Compare commits

..
19 Commits
Author SHA1 Message Date
xavierk 3c07f37c78 fix(release): refresh published XBPS assets 2026-09-29 04:38:58 +05:30
xavierk b098132595 fix(ci): authenticate XBPS repository publish
Set the existing Fenris commit identity and pass the Gitea publish token to Git without storing credentials on the runner.
2026-09-29 04:32:54 +05:30
xavierk 1dbe372714 fix(ci): honor XBPS dispatch input
Handle Gitea boolean inputs in the publish condition. Mark Void available only after the repository publish step succeeds.
2026-09-29 04:27:57 +05:30
xavierk 1063fa4dae test(signing): align key storage contract with Gitea
Release / release (push) Successful in 2m3s
2026-09-29 04:19:23 +05:30
xavierk 3f2dd6a5a1 fix(release): retain XBPS key through publication
The optional XBPS publisher signs repository metadata after package signing. Remove the runner key only after publication and release asset upload.
2026-09-29 04:06:45 +05:30
xavierk a5b84f7566 docs: align signing ceremony with Gitea workflow 2026-09-29 03:57:24 +05:30
xavierk 917c94fd65 chore: prepare v0.6.0 release 2026-09-29 03:51:35 +05:30
xavierk a49808a247 docs: preserve session style instructions 2026-09-29 03:51:24 +05:30
xavierk 87c5dcdac6 test: fix runit missing-script fixture 2026-09-29 03:51:20 +05:30
xavierk 511905f519 docs: record observation history decisions 2026-09-29 03:51:17 +05:30
xavierk e22b99442e fix: validate complete-day monitoring coverage 2026-09-29 03:34:17 +05:30
xavierk d045e88043 Clarify incomplete history readouts (#103) 2026-09-28 17:54:56 +05:30
xavierk c4524cf654 Preserve historical activity selection (#103) 2026-09-28 17:46:18 +05:30
xavierk f780212a02 fix: follow live activity until inspection (#102) 2026-09-28 16:32:13 +05:30
xavierk 3e1ecc90e4 refactor: consolidate pending recovery checks (#101) 2026-09-28 15:05:47 +05:30
xavierk 59d2dd2634 fix: clarify pending admission failure outcome (#101) 2026-09-28 15:05:02 +05:30
xavierk fd4db45ae1 fix: bound pending publication admission (#101) 2026-09-28 15:02:58 +05:30
xavierk 5aeee7b6f1 fix: guard local-day pruning across monitoring gaps (#100) 2026-09-28 14:37:22 +05:30
xavierk d69753690a fix: prune old detail after publication (#100) 2026-09-28 14:03:37 +05:30
33 changed files with 2814 additions and 586 deletions
+39 -12
View File
@@ -159,7 +159,6 @@ jobs:
gpg --batch --yes --delete-secret-keys "${FINGERPRINT}"
gpg --batch --yes --delete-keys "${FINGERPRINT}"
fi
rm -f ~/.ssh/id_xbps
- name: Determine version
id: version
@@ -204,9 +203,21 @@ jobs:
esac
- name: Publish XBPS to distribution repository
if: github.event.inputs.publish_xbps == 'true'
id: publish-xbps
if: ${{ github.event.inputs.publish_xbps == true || github.event.inputs.publish_xbps == 'true' }}
env:
GITEA_PUBLISH_TOKEN: ${{ secrets.GITEAPACKAGETOKEN }}
run: |
set -euo pipefail
if [ -z "${GITEA_PUBLISH_TOKEN}" ]; then
echo "::error::GITEAPACKAGETOKEN repository secret is not configured"
exit 1
fi
git config user.name "xavierk"
git config user.email "xavierk@bongbetic.com"
export GIT_CONFIG_COUNT=1
export GIT_CONFIG_KEY_0='http.https://git.bongbetic.com/.extraheader'
export GIT_CONFIG_VALUE_0="Authorization: token ${GITEA_PUBLISH_TOKEN}"
VERSION=${{ steps.version.outputs.version }}
XBPS_FILE="fenris-${VERSION}_1.x86_64.xbps"
if [ ! -f "${XBPS_FILE}" ]; then
@@ -218,6 +229,7 @@ jobs:
exit 1
fi
bash scripts/xbps-publish.sh --publish
echo "xbps_published=true" >> "$GITHUB_OUTPUT"
- name: Track format availability
id: formats
@@ -227,7 +239,7 @@ jobs:
DEB_EXISTS=$([ -f "dist/fenris_${VERSION}_amd64.deb" ] && echo "true" || echo "false")
RPM_EXISTS=$([ -f "dist/fenris-${VERSION}-1.x86_64.rpm" ] && echo "true" || echo "false")
XBPS_EXISTS=$([ -f "fenris-${VERSION}_1.x86_64.xbps" ] && echo "true" || echo "false")
XBPS_PUBLISHED=$([ "${{ github.event.inputs.publish_xbps }}" = "true" ] && echo "true" || echo "false")
XBPS_PUBLISHED=$([ "${{ steps.publish-xbps.outputs.xbps_published }}" = "true" ] && echo "true" || echo "false")
echo "deb_available=${DEB_EXISTS}" >> "$GITHUB_OUTPUT"
echo "rpm_available=${RPM_EXISTS}" >> "$GITHUB_OUTPUT"
echo "xbps_available=${XBPS_EXISTS}" >> "$GITHUB_OUTPUT"
@@ -318,7 +330,8 @@ jobs:
"https://git.bongbetic.com/api/v1/repos/xavierk/Fenris/releases/tags/v${VERSION}")
RELEASE_ID=$(printf '%s' "${RELEASE_JSON}" \
| python3 -c "import sys,json; print(json.load(sys.stdin)['id'])")
# Attach deb, rpm, clearsigned checksums, and XBPS artifacts once.
PUBLISH_XBPS="${{ steps.formats.outputs.xbps_published }}"
# Attach package artifacts; refresh XBPS assets after publication.
ARTIFACTS=(
"dist/fenris_${VERSION}_amd64.deb"
"dist/fenris-${VERSION}-1.x86_64.rpm"
@@ -327,19 +340,33 @@ jobs:
# Add XBPS artifacts if they exist
XBPS_FILE="fenris-${VERSION}_1.x86_64.xbps"
if [ -f "${XBPS_FILE}" ]; then
ARTIFACTS+=("${XBPS_FILE}")
if [ -f "${XBPS_FILE}.sig2" ]; then
ARTIFACTS+=("${XBPS_FILE}")
ARTIFACTS+=("${XBPS_FILE}.sig2")
else
echo "::warning::Unsigned XBPS artifact omitted from release assets"
fi
fi
for FILE in "${ARTIFACTS[@]}"; do
ASSET_NAME="${FILE##*/}"
if python3 -c 'import json,sys; name=sys.argv[1]; sys.exit(0 if any(a.get("name") == name for a in json.load(sys.stdin).get("assets", [])) else 1)' "${ASSET_NAME}" <<<"${RELEASE_JSON}"; then
echo "${ASSET_NAME}: already attached"
else
curl --fail --silent --show-error -X POST \
-H "Authorization: token ${GITEA_PUBLISH_TOKEN}" \
-F "attachment=@${FILE}" \
"https://git.bongbetic.com/api/v1/repos/xavierk/Fenris/releases/${RELEASE_ID}/assets"
EXISTING_ASSET_ID=$(python3 -c 'import json,sys; name=sys.argv[1]; print(next((a["id"] for a in json.load(sys.stdin).get("assets", []) if a.get("name") == name), ""))' \
"${ASSET_NAME}" <<<"${RELEASE_JSON}")
if [ -n "${EXISTING_ASSET_ID}" ]; then
if [ "${PUBLISH_XBPS}" = "true" ] && [[ "${ASSET_NAME}" = "${XBPS_FILE}" || "${ASSET_NAME}" = "${XBPS_FILE}.sig2" ]]; then
curl --fail --silent --show-error -X DELETE \
-H "Authorization: token ${GITEA_PUBLISH_TOKEN}" \
"https://git.bongbetic.com/api/v1/repos/xavierk/Fenris/releases/${RELEASE_ID}/assets/${EXISTING_ASSET_ID}"
else
echo "${ASSET_NAME}: already attached"
continue
fi
fi
curl --fail --silent --show-error -X POST \
-H "Authorization: token ${GITEA_PUBLISH_TOKEN}" \
-F "attachment=@${FILE}" \
"https://git.bongbetic.com/api/v1/repos/xavierk/Fenris/releases/${RELEASE_ID}/assets"
done
- name: Remove XBPS signing key
if: always()
run: rm -f ~/.ssh/id_xbps
+4
View File
@@ -1,3 +1,7 @@
## Session style
After the first user message in each session, load the global `caveman` skill and activate `ultra` mode. Keep `ultra` mode active until the user changes or stops it.
## Agent skills
## Commit messages
+13
View File
@@ -9,6 +9,19 @@ backfill releases from before this changelog.
## [Unreleased]
## [0.6.0] - 2026-09-29
### Added
- Preserve local-day read and write evidence separately, including ambiguous midnight-spanning volume as shared evidence instead of assigning it to either day.
- Keep valid observations pending when publication fails, and recover them without exposing partial history or counting activity twice.
- Preserve live-point inspection and historical activity selection through graph refresh.
### Changed
- Prune old detail only after durable summaries and required boundary evidence are published.
- Require a complete local day to fit within one monitoring period before it can open the endurance outlook.
## [0.5.0] - 2026-09-19
### Changed
+16
View File
@@ -28,6 +28,10 @@ _Avoid_: Daemon uptime, calibration window
The single SQLite database at `/var/lib/fenris/observations.db` that persists the observation history, monitoring periods, hour observations, day aggregates, and endurance baseline.
_Avoid_: Data directory, history.jsonl, the database (generic)
**Pending publication**:
The condition where valid acquired observations are retained for recovery but their dependent evidence has not yet been published consistently. Those observations are not part of the reader-visible observation history until publication succeeds; readers retain the last consistent evidence with the pending condition made explicit.
_Avoid_: Successful collection, fresh published evidence
**Store fault**:
The condition where the observation store is present but cannot be read or trusted — unreadable, corrupt, or written by a newer Fenris — degrading every view that depends on it rather than crashing or guessing.
_Avoid_: Database error, corruption, broken data
@@ -40,6 +44,18 @@ _Avoid_: Hourly record, hourly.jsonl entry
One row per UTC day derived from hour observations; the grain at which usage-habit evidence is judged.
_Avoid_: Daily summary, daily stats
**Local-day evidence**:
Measured read and write activity attributable to a local calendar day with its recorded timezone and midnight boundaries. Known volumes remain incomplete when gaps or shared local-day evidence prevent an exact total; they are not estimates of the missing activity.
_Avoid_: Estimated daily total, localised UTC day aggregate
**Shared local-day evidence**:
Measured read or write volume spanning a local midnight that cannot honestly be allocated to either adjacent day. It is retained once, separately from either day's known volume, rather than prorated or counted in both days.
_Avoid_: Missing bytes, estimated midnight split
**Activity selection**:
The read/write measurement, date and point being inspected in live or historical drive activity. Live activity follows the newest point until deliberate inspection pins a point; historical selection survives refresh, while an expired live pin is explained before following resumes.
_Avoid_: Projection window, current usage habit
**Usage-history window**:
An exact consecutive span of UTC calendar days ending today, shown from day aggregates; a day without trustworthy evidence remains an explicit gap rather than disappearing or being estimated.
_Avoid_: Available records, dataset range
+1 -1
View File
@@ -22,7 +22,7 @@ The Gitea instance Debian registry signs metadata with its own key. Verify the
instance key fingerprint (TOFU hardening):
```text
Fingerprint: <print after first release — paste beside the curl one-liner>
Fingerprint: F937E81D2FB0736B15BC611884BEBD586DFAC010
```
Add the instance key and repository:
@@ -0,0 +1,20 @@
# 11. Preserve valid unpublished observations without exposing partial history
Status: Accepted; implemented on `main` for the planned v0.6.0 release (commit `e22b994`).
A non-invariant derivation failure must not discard valid acquired observations or report a successful collection: retain the observations as pending publication in the existing observation store and retry through the normal scheduled collection path. Readers continue to see the last consistent published evidence, with an explicit pending-publication explanation, rather than combining newly acquired counters with older derived totals. This trades recovery bookkeeping for preservation of measured evidence and consistent read views; neither discarding every valid acquisition on derivation failure nor exposing partially derived history satisfies both requirements.
## Constraints
- The publication distinction applies to every dependent reader, including activity, freshness, controller-segment interpretation and projection inputs, not just the graph. ADR 0001's freshest-sample signal means the freshest published sample; an unpublished observation must not make published evidence appear fresh.
- Pending-work bookkeeping qualifies ADR 0005's recovery-without-new-state wording: it records unfinished evidence publication, not a new health grade or escalation mechanism. Last-collection outcomes still come from the existing native monitoring and logging paths.
- Invariant violations retain [ADR 0005](0005-failure-detection-and-recovery.md)'s write-nothing rule; retaining recoverable work is not permission to persist invalid observations. Actual store faults retain their existing refusal and degradation behavior.
- Recovery and retention remain collector-owned. Repeated recovery must not duplicate measured volume, and required source evidence cannot be pruned before trustworthy derived evidence is durable.
- This extends [ADR 0001](0001-observation-store-sqlite.md)'s consistent read model and [ADR 0010](0010-local-day-activity-history.md)'s preservation rule without another observation store, sampler, background process or retry cadence. Pending-work metadata is not a persisted projection or a replacement for journalled collection outcomes.
- Pending observations use private staging separate from published samples and derived evidence inside the same observation store. This makes exclusion from ordinary reader and projection queries structural, rather than requiring each query to remember a publication filter; the exact table layout remains an implementation detail.
## Bounded pending work
Pending publication has a fixed initial admission capacity of 6,720 observations, equivalent to 14 days at the default three-minute cadence. This is a count limit, not an expiry rule: older pending observations are never discarded merely to admit newer ones.
The collection module attempts recovery and checks capacity before invoking acquisition. If capacity remains exhausted, the collection is unsuccessful and visibly explains why no new observation was acquired; normal scheduled recovery attempts continue. This deliberately accepts a gap in new observations rather than unlimited pending growth or loss of already retained evidence. It is not a deliberate disable, changes no monitoring intent, and never permits fabricated activity across the gap.
+43 -42
View File
@@ -12,7 +12,7 @@ and destruction.
| UID | `Fenris Packaging <packaging@bongbetic.com>` |
| Expiry | 2 years from creation |
| Hierarchy | Single key — no master/subkey split (single maintainer, manual builds) |
| Private key storage | Password manager only |
| Private key storage | Gitea repository Actions secret `GPG_PRIVATE_KEY` |
| Public key storage | `packaging/keys/fenris-packaging.asc` in-repo, release notes, docs |
| Keyservers | Never — TOFU-over-TLS via raw URL |
@@ -37,71 +37,72 @@ gpg --armor --export packaging@bongbetic.com > packaging/keys/fenris-packaging.a
gpg --fingerprint packaging@bongbetic.com
```
Save the **private key** to the password manager immediately:
Provision the **private key** as the repository Actions secret `GPG_PRIVATE_KEY`.
Run the export on the trusted key-generation machine, then enter its output in
the Gitea repository's Actions secret settings. Do not save it in the checkout,
logs, or a runner directory. The release workflow checks its fingerprint
against the committed public key before signing.
```bash
gpg --armor --export-secret-keys packaging@bongbetic.com
```
Then **delete the private key from the local keyring** — it must never persist
on any build host:
After provisioning the secret, delete the private key from the key-generation
keyring:
```bash
gpg --delete-secret-keys packaging@bongbetic.com
gpg --delete-keys packaging@bongbetic.com
gpg --batch --yes --delete-secret-keys packaging@bongbetic.com
gpg --batch --yes --delete-keys packaging@bongbetic.com
```
The committed `fenris-packaging.asc` must contain the real public key (replace
the placeholder comments).
## XBPS signing key
XBPS uses a separate RSA 3072 key. Its private half is stored as the Gitea
repository Actions secret `XBPS_SIGNING_KEY`. The corresponding public key is
published at
`https://git.bongbetic.com/xavierk/Fenris-xbps/raw/branch/stable/keys/fenris-xbps-signing.pub`,
with fingerprint `SHA256:AvPMRlKMikPg75u0iKr8AUkxlfU/Ad4k/S4o2M9W4/w`.
The secret must match that public key.
The release workflow writes the key to `~/.ssh/id_xbps` to sign the XBPS
package. A requested XBPS publication also uses the key to sign repository
metadata. A final `always()` cleanup removes the runner copy after publication
and release asset upload, including when an earlier step fails.
## Per-release signing flow
Each release performs: **import → sign → delete**. The private key is never
stored on disk longer than the release takes.
Each tagged release performs: **import → verify → sign → delete** on the
repository-scoped Gitea Actions runner. The Gitea secret remains configured;
the runner's keyring copy is removed after the job.
### Step 1: Import the private key
### Step 1: Push the release tag
Retrieve the private key from the password manager and import it:
After updating the version and dated changelog section, push the matching tag:
```bash
gpg --import /tmp/packaging-key-private.asc
rm /f /tmp/packaging-key-private.asc # Shred if possible
git push origin v<version>
```
### Step 2: Build and sign packages
### Step 2: Build, verify, and sign packages
The Makefile target `make release` handles signing automatically when the
key is in the keyring:
The release workflow imports `GPG_PRIVATE_KEY`, checks it against
`packaging/keys/fenris-packaging.asc`, builds packages, signs the RPM and
clearsigned checksum manifest, validates both, and publishes the release. The
workflow imports `XBPS_SIGNING_KEY` separately and signs the XBPS package.
XBPS publication is optional and also signs repository metadata; it requires
host acceptance and explicit selection during workflow dispatch.
```bash
make release # builds, signs RPM, clearsigns SHA256SUMS, prints upload steps
```
### Step 3: Verify runner cleanup
Under the hood:
The workflow's `always()` cleanup removes the GPG key from the runner's keyring
and deletes `~/.ssh/id_xbps`, including after a failed job. Confirm no signing
key remains on the runner after the release job.
1. `rpmsign --addsign` signs the RPM payload with the packaging key
(invoked by `make sign-rpm`).
2. `sha256sum` generates the checksum manifest.
3. `gpg --clearsign` produces `SHA256SUMS.asc` with the packaging key.
### Step 3: Delete the private key
Immediately after signing:
```bash
gpg --delete-secret-keys packaging@bongbetic.com
gpg --delete-keys packaging@bongbetic.com
```
Verify the key is gone:
```bash
gpg --list-keys packaging@bongbetic.com
# Should produce: gpg: keyblock resource ...: No such file or directory
```
The entire import → sign → delete cycle should take minutes. The private key
must never be left in any keyring between releases.
The Gitea Actions secret remains the approved signing source. Do not copy it to
the runner or repository outside the workflow.
## Key rotation (outline)
+117
View File
@@ -0,0 +1,117 @@
## Problem Statement
Fenris users cannot readily understand how much data their monitored drive reads and writes each day, inspect recent activity as it arrives, or select historical dates from the keyboard. Existing graph shortcuts depend on focus, date navigation is limited, collection and screen refresh default to five minutes, and the graph reads hourly/daily summaries rather than three-minute activity. Source inspection also identified derivation and rendering paths that can leave displayed history incomplete or stale despite newly stored readings.
Users want daily data volume as the primary measurement because writes contribute to drive endurance. They also want an understandable end-of-life outlook even with limited history, without presenting an unsupported hardware-failure prediction or a misleading numerical confidence score.
## Solution
Show both local-day read and write totals, updating from background collection targeted every three minutes. Open on a live last-three-hours graph of written data volume, provide a read/write toggle, and support selected-day and longer-history views. Make date navigation discoverable through visible keyboard hints and clickable equivalents.
Retain three-minute detail for 14 days, followed by durable hourly/daily summaries. Preserve the timezone and boundaries of historical local-day summaries, identify incomplete evidence, and never invent missing activity or divide midnight-spanning measurements by assumption.
Show the usage-adjusted theoretical lifespan after one full local calendar day of observations, provided an applicable endurance baseline and usable write rate exist. Explain that this estimates remaining write endurance if observed habits continue, not a physical failure date. Present categorical projection confidence with contributing facts, including limited-history reasons.
## User Stories
1. As a drive owner, I want to see the amount of data written each local day, so that I can understand the activity that contributes to write endurance.
2. As a drive owner, I want to see the amount of data read each local day, so that I can understand the drive's broader activity.
3. As a drive owner, I want read and write totals shown separately, so that reads are not mistaken for writes consuming the endurance allowance.
4. As a TUI user, I want data-volume units rather than transfer speed as the primary graph measurement, so that the graph answers how much data was transferred.
5. As a TUI user, I want today's totals labelled as totals so far, so that a partial day is not presented as a completed day's usage.
6. As a TUI user, I want the opening graph to show written volume over the last three hours, so that recent activity is visible immediately.
7. As a TUI user, I want to switch the graph between writes and reads while both daily totals remain visible, so that I can inspect either measurement without losing context.
8. As a TUI user, I want each new activity point to include transfers between readings, so that brief bursts between collection runs contribute to the graph.
9. As a TUI user, I want new measurements approximately every three minutes, so that the recent-activity view stays useful while I work.
10. As a user who closes the TUI, I want collection to continue in the background, so that reopening it shows activity gathered while I was away.
11. As a user who reboots, I want observation history and the established monitoring lifecycle preserved, so that restarting does not erase activity or confuse monitoring state.
12. As a keyboard user, I want `[` and `]` to select the previous and next day, so that browsing nearby dates takes one action.
13. As a keyboard user, I want `g` to open a date field, so that I can jump directly to a historical date.
14. As a keyboard user, I want `t` to return to today and the live view, so that I can leave historical browsing immediately.
15. As a keyboard user, I want arrows to inspect graph points, so that I can read the volume, time, and evidence state behind a point.
16. As a new TUI user, I want visible shortcut hints that work from normal launch, so that I do not have to discover an invisible graph-focus prerequisite.
17. As a mouse user, I want clickable equivalents for navigation and the read/write toggle, so that I can use the same features without memorizing shortcuts.
18. As a user entering a date, I want typing and cancelling to remain within the date-entry workflow, so that graph or monitoring shortcuts do not fire accidentally.
19. As a user browsing history, I want background refresh to preserve my selected date and measurement, so that new activity does not interrupt inspection.
20. As a user inspecting daily totals, I want to switch between recent detail, a selected day, and longer history, so that I can understand both short bursts and daily habits.
21. As a user outside UTC, I want calendar navigation and timestamps expressed in labelled local time, so that the displayed day matches my calendar.
22. As a user in a timezone with a half-hour offset, I want local-day totals based on the correct midnight boundary, so that relabelled UTC totals do not misrepresent my day.
23. As a user experiencing a daylight-saving transition, I want real local-day boundaries respected, so that a 23-hour or 25-hour day remains understandable.
24. As a user who changes system timezone, I want historical summaries to retain their recorded timezone and boundaries, so that old totals do not silently change meaning.
25. As a user reviewing recent history, I want three-minute detail available for 14 days, so that I can investigate recent usage.
26. As a user reviewing older history, I want durable hourly/daily summaries after detailed readings expire, so that long-term activity remains available.
27. As a user with legacy history, I want dates that cannot be reconstructed at local-day precision labelled incomplete or unavailable, so that old summaries are not presented with invented precision.
28. As a user whose readings straddle midnight, I want the measured volume preserved once and its uncertain day allocation explained, so that it is neither lost nor counted twice.
29. As a new user with only one reading, I want an awaiting-another-reading state, so that a missing interval is not shown as zero activity.
30. As a user with a measured zero-activity interval, I want zero distinguished from missing evidence, so that a quiet drive is not confused with a collection failure.
31. As a user with missed collection runs, I want gaps, actual timestamps, and freshness facts, so that the graph does not imply measurements that never occurred.
32. As a user who deliberately pauses monitoring, I want paused time excluded according to monitoring-period rules, so that the pause does not distort the observed usage habit.
33. As a user whose controller resets or changes, I want counter and identity boundaries respected, so that unrelated readings do not create invalid activity or lifespan estimates.
34. As a user with a store fault, I want a clear explanation instead of guessed history, so that I know which information cannot be trusted.
35. As a drive owner, I want an estimate of the time until remaining write endurance is consumed, so that I can plan around my observed usage habit.
36. As a new user, I want that estimate withheld until one full local calendar day has been observed, so that a few minutes of activity do not immediately produce a lifespan number.
37. As a user who starts monitoring at noon, I want the partial first day excluded from the full-day gate, so that 24 hours since launch is not mistaken for a complete calendar day.
38. As a user with one complete day but limited history, I want an estimate when the other inputs support it, so that I do not have to wait for high confidence before seeing an outlook.
39. As a user with limited evidence, I want a confidence category and concrete reasons, so that I understand why an estimate may change without being shown an unjustified percentage.
40. As a user without an applicable endurance baseline or usable write rate, I want the missing prerequisite explained, so that unavailable evidence is not disguised as merely low confidence.
41. As a drive owner, I want the endurance estimate distinguished from a physical failure date, so that I do not treat a write-endurance projection as a hardware guarantee.
42. As a CLI user, I want status and the TUI to agree on forecast eligibility and confidence, so that choosing a different interface does not change the facts.
43. As a user on a narrow terminal, I want daily totals, dates, evidence labels, and controls to remain accessible, so that terminal size does not hide the information I need.
44. As a user of existing themes and reduced-motion settings, I want those features preserved when navigation shortcuts change, so that better date controls do not remove accessibility or preferences.
45. As a user upgrading Fenris, I want migration, repair, refresh, and pruning to preserve measured evidence without duplication, so that the new graph can be trusted across restarts and upgrades.
## Implementation Decisions
- Extend the existing collector, observation store, derivation, shared status/projection, service scheduling, and TUI responsibilities. The privileged collector remains the single device-acquisition and store-write owner; the TUI remains an unprivileged reader. Do not add another sampler, service, export layer, or generic plotting abstraction.
- Set the default background collection target to three minutes on both systemd and runit. Update shared cadence/freshness semantics, scheduler definitions, user documentation, and relevant acceptance contracts together. Use actual observation timestamps: delayed runs must not be represented as perfectly spaced measurements or synthetic catch-up points.
- Derive interval read/write volumes from compatible cumulative counters. A first reading is an anchor, not an interval. Preserve controller-segment and monitoring-period boundaries, measured zero, unknown time, deliberate disables, and unsupported evidence. Do not compute deltas across a controller identity or counter discontinuity.
- Make successful collection publish consistent data for dependent read views, using the existing transactional and migration patterns. Ensure raw samples, affected summaries, and retained boundary evidence cannot disagree through partial publication. Validate source-level derivation gaps before implementing fixes; this spec does not claim a runtime diagnosis.
- Keep daily read and write totals separate. The write graph is the default; a read/write toggle changes only the displayed measurement. Each live point represents measured interval volume, not instantaneous speed. Daily totals remain visible independently of graph mode, and graph selection must not redefine the projection's evidence window.
- Open on the latest three hours. Offer selected-day and longer-history views at the precision supported by retained evidence. Selected-point readouts expose time, labelled timezone, data-volume units, and incomplete/gap state. Do not imply that a partial current day or an incompletely allocated day is a known full-day total.
- Use `[` / `]` for previous/next day, `g` for date entry, `t` for today/live, and arrows for point inspection, with visible hints and clickable equivalents. Date entry must handle invalid or unavailable dates visibly and must not leak keystrokes into unrelated actions. Cancelling returns to the prior selection; refresh preserves historical selection until the user changes it.
- Resolve the existing `t` theme binding in favor of the accepted today/live action. Preserve theme selection through a discoverable non-conflicting control and update help consistently. Preserve pause, resume, collect-now, disclosures, quit, and reduced-motion behavior; quitting still does not pause monitoring. No exact replacement theme shortcut was selected in the interview.
- Preserve UTC timestamps and the established UTC hour/day evidence used by endurance calculations. Local-day activity is a separate presentation aggregate using the actual local midnight boundaries, including non-whole-hour offsets and daylight-saving transitions. Timestamp relabelling alone is insufficient.
- Extend the existing versioned observation store to retain local-day read/write summaries with their controller identity/segment context, recorded timezone, local date, UTC boundaries, evidence state, and the information needed to explain unallocated boundary volume. These are required semantics, not prescribed table names or a finalized column layout. The collector derives durable summaries before source evidence is eligible for pruning.
- Retain three-minute detail for 14 days and hourly/daily summaries indefinitely. Keep the boundary evidence and anchors necessary for honest derivation. Do not introduce unbounded retention of all fine-grained intervals solely to support arbitrary historical timezone reinterpretation, and do not delete existing durable evidence merely to simplify migration.
- Preserve the timezone and boundaries recorded for historical summaries when the system timezone changes. New observations follow the current local timezone without rewriting old days; labels must make historical timezone context explicit. Do not blend differently bounded summaries into an apparently exact daily total.
- Allocate a measured interval to a local day only when evidence supports that allocation. Preserve an ambiguous midnight-spanning delta once as shared/unallocated boundary evidence; neither prorate it nor copy its full value into both days. If only coarse legacy UTC summaries survive, show them at their actual precision and mark unreconstructable local-day totals incomplete/unavailable. Repeated derivation and repair must be idempotent.
- Use the existing usage-adjusted theoretical lifespan model and categorical projection confidence. Forecast remaining write endurance from an applicable endurance baseline and a usable observed write rate. Lifetime-written counters consume the endurance allowance; observation-window deltas determine the usage rate. Read volume is not included in endurance consumption.
- Add the agreed complete-observation-day gate to the shared projection contract. A completed local midnight-to-midnight day within a monitoring period with usable evidence is required; a partial first day, deliberately disabled span, or day with no usable evidence cannot independently qualify. Monday-noon setup can first qualify at Wednesday 00:00 after observing Tuesday. Real daylight-saving day boundaries apply; do not replace this with a fixed 24-hour timer.
- Before that gate, explain that Fenris is waiting for a full local observation day. Afterward, allow an estimate with Limited confidence when the existing baseline, rate, identity, and evidence-validity rules permit it; preserve the longer warm-up and Supported-confidence requirements. Missing evidence is still evaluated under the existing validity/coverage rules: completing a date does not turn unknown measurements into known data.
- Show confidence as a category plus contributing facts, never a numeric confidence percentage. Explain missing baseline, unusable/zero rate, stale evidence, and controller changes through their applicable states. Do not fabricate a baseline, suppress a missing prerequisite behind a generic confidence warning, or label the output as a predicted hardware-failure date. Preserve the existing baseline setup path and zero-rate explanation.
- Preserve visible freshness, last-collection outcome, paused state, and store-fault behavior. On constrained terminals retain the existing textual fallback/reflow approach so both daily totals, the selected date, evidence state, confidence, and navigation remain accessible; colour alone must not carry meaning.
- Migrate through the existing ordered, versioned, transactional store mechanism. Preserve observation history, refuse unknown newer schemas, establish new durable evidence before pruning, and keep readers consistent while the collector writes. Migration cannot manufacture local precision from historical data that lacks it.
- This design amends ADR 0003's five-minute default on both native service backends under ADR 0008, adds a first-full-local-day eligibility gate to ADR 0002 while retaining its forecasting and confidence model, and extends ADR 0001 with the local-history decision recorded as ADR 0010. ADR 0005's no-fabrication/store-fault contract remains authoritative. Existing historical design documents are not evidence that the current runtime already implements these amendments.
## Testing Decisions
- Prefer one primary acceptance path: controlled acquisition fixtures and an injected clock enter the existing public collection operation; it writes a real temporary observation store; the normal read path supplies the TUI and CLI. Assert resulting displayed volumes, evidence states, selections, forecast eligibility, and public outcomes. Do not mock the internal derivation or query results under test. A good test would continue passing after an internal refactor and fail when a user's observed result becomes incorrect.
- Reuse existing collector tracer and collector-history tracer fixtures, including SMART JSON, a temporary sysfs tree, the injected clock, and a real SQLite store. Compose these with the existing status reader, projection fixtures, and headless TUI driver rather than creating a new production testing interface. Fix defects that this path reproduces; do not make tests reproduce incorrect implementation arithmetic.
- Exercise normal launch, previous/next date, direct date entry, invalid input, cancel, today/live, arrow inspection, read/write switching, refresh during historical browsing, and hourly/day navigation through user actions. Assert visible graph/readout results, not only private selection indices or row counts. In particular, do not manually force graph focus to make the advertised shortcuts work. Existing headless graph, help, theme, motion, and constrained-terminal tests provide prior art.
- Drive known successive read/write counter changes through collection and verify exact interval volumes and aggregate totals. Cover first reading, first compatible pair, measured zero, repeated collection, repeated read/refresh, restart, same-hour accumulation, and a day-boundary crossing. Preserve each measured byte once: attributable totals plus separately retained unallocated evidence must account for the measured delta without duplication.
- Use the same public path for local midnight, Asia/Kolkata's half-hour boundary, a 23-hour local day, a 25-hour local day, repeated local clock labels, and a system-timezone change. Verify historical labels and boundaries remain stable and ambiguous intervals are not silently divided. Add direct derivation cases only when they prove an invariant that the primary path cannot isolate clearly.
- Advance controlled time beyond the 14-day detail window, invoke normal pruning, restart readers, and verify recent detail expiry with durable summary/boundary preservation. Use existing pruning, repair, legacy-migration, and store-migration tests for focused persistence cases that the main path cannot prove, including interruption, reruns, concurrent readers, unknown newer schema, and inability to reconstruct legacy local-day precision.
- Test the forecast gate at concrete local times: Monday noon setup; Tuesday noon with 24 elapsed hours but no complete observed calendar day; and Wednesday 00:00 after a usable Tuesday. Include partial/disabled/unknown days and daylight-saving dates. After eligibility, verify Limited confidence with reasons and a correct estimate where inputs support one; no baseline, invalid counters, zero rate, stale evidence, and segment changes must preserve the relevant unavailable/degraded behavior. Reuse existing projection and status-parity tests.
- Independently calculate an expected endurance result from known lifetime-written counters, a known applicable baseline, and known observed-window deltas. This guards against confusing lifetime consumption with writes inside the rate window. Rendering, browsing another day, and switching to reads must not change that accounting.
- Add focused checks at the existing native scheduler boundary for the three-minute defaults, serialization, failures, and freshness calculations. Reuse systemd/runit control and packaging tests; use only necessary controlled platform probes to verify actual scheduled/background behavior. No live-system scheduling, pause, install, or reboot action is authorized merely by writing this spec. Report any unavailable platform verification during implementation.
- Verify at 80×24 and below that visible totals, date entry, navigation, confidence, and evidence states remain accessible and that theme/reduced-motion controls still work. Avoid pixel-perfect snapshots and assertions about private widget layout when visible text and action results express the contract.
- The affected modules are collection/derivation, observation-store migration and retention, shared projection/status, native service scheduling, and the TUI. Most acceptance coverage should remain at the single collection-to-visible-result path; targeted persistence and scheduler checks are supporting boundaries, not a new layer of test-only architecture.
## Out of Scope
- Adding SATA/HDD acquisition, another monitored device, or multi-drive management; the selected existing NVMe drive remains the source.
- Reporting occupied/free filesystem capacity, making transfer speed the primary measurement, or treating read bytes as write-endurance consumption.
- Forecasting today's final data volume or future daily activity; the selected forecast is the usage-adjusted theoretical lifespan.
- Predicting a physical hardware-failure date, adding numeric confidence percentages, fabricating endurance baselines, or introducing a new forecasting algorithm when the existing model fits.
- Interpolating missing samples, zero-filling unknown activity, prorating ambiguous midnight intervals, or retroactively rewriting historical timezone boundaries.
- Requiring all historical dates to retain three-minute detail forever or to support arbitrary later timezone regrouping.
- A new web UI, plotting service, generic graph framework, notifications, telemetry, theme redesign, or unrelated dashboard rework.
- Package publication, release creation, installation, host-monitoring changes, or runtime diagnosis of the user's deployed observation store as part of this specification task.
## Further Notes
- This spec synthesizes the completed design conversation. The user accepted daily volume, a live three-hour opening view, the specified keyboard date workflow, local dates, 14-day detail retention, a writes-first graph with read toggle and both totals, an endurance forecast, categorical confidence, one full local observation day before forecasting, and historically labelled timezone boundaries with honest incomplete evidence.
- Related prior issue: [Implement Fenris TUI polish and hourly history](https://git.bongbetic.com/xavierk/Fenris/issues/72). This newer accepted design supersedes its conflicting requirements for the opening graph, writes-only presentation, the `t` theme shortcut, unrestricted historical timezone regrouping, indefinite fine-grained interval retention, and the earlier projection-eligibility presentation. Preserve unrelated requirements for identity, status, themes, motion, accessibility, collector publication, evidence conservation, and CLI parity. Do not treat the older issue as a reason to undo these newly agreed choices.
- ADR 0010 records the deliberate trade-off: retain labelled local summaries and necessary boundary evidence before detailed readings expire, rather than retaining every fine-grained interval for arbitrary future timezone reinterpretation. The design adds durable local activity evidence while retaining UTC projection evidence and the existing collector/reader ownership boundary.
- Source inspection identified normal collection paths that do not consistently rebuild daily totals, existing-hour updates that omit read deltas, cross-day deltas added to both dates, and drill-down rendering overwritten by a loading placeholder. These are concrete investigation/regression targets, not claims of executed reproductions or deployed-store corruption.
- The user confirmed the existing collector-to-store-to-TUI/CLI test boundary, with focused migration/pruning and scheduler checks where needed. Publication is an implementation handoff; no application behavior or test result is claimed solely by creating this issue.
+4 -3
View File
@@ -44,7 +44,8 @@
- **RPM payload: signed.** rpmsign with the dedicated packaging key, invoked by `make sign-rpm` after the package is built. This is required, not optional: it is the only working dnf-native verification path.
- **deb: unsigned.** apt never verifies payload signatures; trust = instance-signed `InRelease` (signed-by keyring) + TLS + Acquire-By-Hash. Manual-download integrity is covered by SHA256SUMS.
- **SHA256SUMS: clearsigned** with the packaging key — the trust anchor for manually downloaded release assets, independent of TLS.
- **Packaging key:** single dedicated key, RSA 3072, UID `Fenris Packaging <packaging@bongbetic.com>`, 2-year expiry, no master/subkey hierarchy (single maintainer, manual builds). Private key lives in the password manager only; each release does import → sign → delete — nothing permanent on any build host. The full ceremony is documented in `docs/install/signing-key-ceremony.md`.
- **Packaging key:** single dedicated key, RSA 3072, UID `Fenris Packaging <packaging@bongbetic.com>`, 2-year expiry, no master/subkey hierarchy. The private key is stored as the repository Actions secret `GPG_PRIVATE_KEY`. The release workflow imports it on the self-hosted runner, verifies it against the in-repo public key, signs the RPM and SHA256SUMS, then deletes the runner's keyring copy in an `always()` cleanup step. The full ceremony is documented in `docs/install/signing-key-ceremony.md`.
- **XBPS key:** separate RSA 3072 key stored as the repository Actions secret `XBPS_SIGNING_KEY`; its public key and fingerprint are published in `Fenris-xbps`. The release workflow uses it for the XBPS package and, when publication is explicitly requested, the repository index. A final `always()` cleanup deletes its runner copy after publication and release asset upload.
- **Public key publication:** in-repo `packaging/keys/fenris-packaging.asc` (raw URL doubles as the `.repo` gpgkey target), release notes, docs page. No keyservers — TOFU-over-TLS.
- **Rotation (outline):** new key published alongside old; rpm signed with the new key; `fenris.repo` gpgkey lists both URLs (dnf accepts multiple); old key dropped after one release cycle. Procedure details stay in map fog.
@@ -53,9 +54,9 @@
- **A Release is:** a version tag, its packages in the channel, a Gitea release entry with notes, and a clearsigned SHA256SUMS — all together. **Bare tags are forbidden** (tag without packages + release entry is not a Release).
- **Cadence: on-demand.** Tag when user-visible changes or fixes accumulate; no calendar, no empty releases, no frequency SLA, no RC ceremony — fixes ship as a revision bump of the current version.
- **Versioning: plain semver.** Major = breaking CLI/config/unit change; store schema changes ride the natural bump (the forward-only refusal handles old-reader/new-store).
- **Promotion flow:** tag → `make release` (automated: `make package` → RPM signing via nfpm → SHA256SUMS generation → clearsign → prints registry PUTs + Gitea release steps). The ceremony is documented in `docs/install/signing-key-ceremony.md`.
- **Promotion flow:** bump `pyproject.toml` and the matching dated `CHANGELOG.md` section, then push tag `v<version>`. The repository-scoped Gitea Actions workflow builds and validates the deb, rpm, checksums, and release entry. XBPS publication remains a manual dispatch option after host acceptance. `make release` is for local artifact preparation and does not replace the tag workflow as the supported publication path. The ceremony is documented in `docs/install/signing-key-ceremony.md`.
- **Rollback:** installing an older package over a newer store is **unsupported** — the store's forward-only version refusal fails it by design. Documented rollback = restore the observation-store snapshot, then install the old Release. No automatic downgrade machinery exists or will be built.
- **CI:** no runners are registered on the instance today ([Actions runner research](https://git.bongbetic.com/xavierk/Fenris/issues/37)), so the manual flow above is primary. A dormant `.gitea/workflows/release.yml` (`on: push: tags: ['v*']`, single job, host-mode runner) is committed alongside; if it fires, it replicates `make release`. Cheapest future upgrade: one `act_runner` static binary in host-label mode on the existing Gitea host.
- **CI:** a repository-scoped self-hosted runner is registered and online (checked 2026-09-29). `.gitea/workflows/release.yml` is the tag-triggered release path; maintainers must confirm runner availability and required Gitea secrets before tagging. The workflow publishes deb/rpm packages and release assets; Void publication is withheld unless the signed XBPS host-release step is explicitly requested.
## 6. Package layout and ownership
+1 -1
View File
@@ -1,6 +1,6 @@
[project]
name = "fenris"
version = "0.5.0"
version = "0.6.0"
description = "NVMe wear monitor with persistent TUI"
requires-python = ">=3.10"
license = {file = "LICENSE"}
+132
View File
@@ -0,0 +1,132 @@
"""User choices for live and historical activity views."""
from collections.abc import Sequence
from dataclasses import dataclass
HISTORY_RANGE_DEFAULT = 14
@dataclass(frozen=True)
class IntervalIdentity:
"""Stable identity for one measured interval."""
start_ts: str
end_ts: str
@dataclass(frozen=True)
class HistoryDayIdentity:
"""Stable identity for one recorded local-day summary."""
local_date: str
timezone: str
utc_start: str
utc_end: str
class ActivitySelection:
"""Keep activity navigation state separate from its Textual rendering."""
def __init__(self) -> None:
self.view = "live"
self.browse_date: str | None = None
self.selected_history_day: HistoryDayIdentity | None = None
self.selected_history_hour: str | None = None
self.history_range_days = HISTORY_RANGE_DEFAULT
self.measure = "written"
self.following_live = True
self.selected_live_interval: IntervalIdentity | None = None
self.live_interval_expired = False
def set_view(self, view: str) -> None:
"""Select live, day, or history view."""
if view not in ("live", "day", "history"):
raise ValueError(f"unknown activity view: {view}")
self.view = view
if view == "live":
self.follow_live()
def select_history_date(
self,
date: str | None,
identity: HistoryDayIdentity | None = None,
) -> None:
"""Keep requested date and evidence identity across refresh."""
if date != self.browse_date or (
identity is not None and identity != self.selected_history_day
):
self.selected_history_hour = None
if date != self.browse_date:
self.selected_history_day = None
self.browse_date = date
if identity is not None:
self.selected_history_day = identity
elif date is None:
self.selected_history_day = None
def select_history_hour(self, hour: str | None) -> None:
"""Keep the chosen historical hour across refresh."""
self.selected_history_hour = hour
def toggle_measure(self) -> str:
"""Switch read/write presentation without changing point selection."""
self.measure = "read" if self.measure == "written" else "written"
return self.measure
def follow_live(self) -> None:
"""Resume following the newest interval and clear stale-pin notices."""
self.view = "live"
self.browse_date = None
self.selected_history_day = None
self.selected_history_hour = None
self.following_live = True
self.selected_live_interval = None
self.live_interval_expired = False
def update_live(self, intervals: Sequence[IntervalIdentity]) -> None:
"""Reconcile the selected interval with the rolling live window."""
if self.following_live:
self.selected_live_interval = intervals[-1] if intervals else None
return
if self.selected_live_interval in intervals:
return
self.following_live = True
self.selected_live_interval = intervals[-1] if intervals else None
self.live_interval_expired = True
def move_live(
self, intervals: Sequence[IntervalIdentity], offset: int,
) -> IntervalIdentity | None:
"""Inspect an adjacent interval by stable identity."""
if not intervals:
return None
if self.selected_live_interval in intervals:
current = intervals.index(self.selected_live_interval)
else:
current = len(intervals) - 1
selected = intervals[max(0, min(len(intervals) - 1, current + offset))]
self._pin_live_interval(selected)
return selected
def inspect_live(
self,
intervals: Sequence[IntervalIdentity],
interval: IntervalIdentity,
) -> None:
"""Pin the interval chosen by mouse inspection."""
if interval in intervals:
self._pin_live_interval(interval)
def selected_live_index(self, intervals: Sequence[IntervalIdentity]) -> int:
"""Return the selected point's current render index, or -1."""
if self.selected_live_interval is None:
return -1
try:
return intervals.index(self.selected_live_interval)
except ValueError:
return -1
def _pin_live_interval(self, interval: IntervalIdentity) -> None:
self.selected_live_interval = interval
self.following_live = False
self.live_interval_expired = False
+5
View File
@@ -122,6 +122,11 @@ def main() -> None:
if result["ok"]:
print(f"Collection successful: {result['sample_count']} sample(s)")
if result.get("retention_error"):
print(
f"Retention deferred: {result['retention_error']}",
file=sys.stderr,
)
sys.exit(0)
else:
print(f"Collection failed: {result['error']}", file=sys.stderr)
+146 -59
View File
@@ -32,6 +32,26 @@ class InvariantViolationError(Exception):
pass
class _PendingDerivationError(Exception):
"""A valid staged observation failed while deriving dependent evidence."""
def __init__(self, cause: Exception):
super().__init__(str(cause))
self.cause = cause
class _PendingRecoveryFailure(Exception):
"""A publication attempt failed after earlier pending rows committed."""
def __init__(self, error: Exception, published_count: int):
derivation_error = isinstance(error, _PendingDerivationError)
cause = error.cause if derivation_error else error
super().__init__(str(cause))
self.cause = cause
self.published_count = published_count
self.derivation_error = derivation_error
PENDING_PUBLICATION_LIMIT = 6720
@@ -300,47 +320,50 @@ def _publish_observation(
current_id = seg_info["sample_id"]
prev = find_previous_sample(conn, seg_info.get("segment_id"), current_id)
if prev is not None:
current = {
"id": current_id,
"ts": sample["ts"],
"bytes_written": sample["bytes_written"],
"bytes_read": sample["bytes_read"],
"power_on_hours": sample["power_on_hours"],
"temperature_c": sample["temperature_c"],
"data_units_written": sample["data_units_written"],
"data_units_read": sample["data_units_read"],
"local_tz": tz_name,
}
derive_hours_from_interval(conn, prev, current)
from .local_day import record_local_activity_interval
record_local_activity_interval(
conn,
prev,
current,
start_sample_id=prev["id"],
end_sample_id=current_id,
segment_id=seg_info.get("segment_id"),
)
else:
previous_any_segment = find_previous_sample(conn, None, current_id)
if previous_any_segment is not None:
from .local_day import mark_local_activity_gap
mark_local_activity_gap(
try:
if prev is not None:
current = {
"id": current_id,
"ts": sample["ts"],
"bytes_written": sample["bytes_written"],
"bytes_read": sample["bytes_read"],
"power_on_hours": sample["power_on_hours"],
"temperature_c": sample["temperature_c"],
"data_units_written": sample["data_units_written"],
"data_units_read": sample["data_units_read"],
"local_tz": tz_name,
}
derive_hours_from_interval(conn, prev, current)
from .local_day import record_local_activity_interval
record_local_activity_interval(
conn,
sample["ts"],
tz_name,
previous_any_segment,
prev,
current,
start_sample_id=prev["id"],
end_sample_id=current_id,
segment_id=seg_info.get("segment_id"),
)
else:
previous_any_segment = find_previous_sample(conn, None, current_id)
if previous_any_segment is not None:
from .local_day import mark_local_activity_gap
mark_local_activity_gap(
conn,
sample["ts"],
tz_name,
previous_any_segment,
)
from .day_aggregate import derive_all_days, persist_day_aggregate
for aggregate in derive_all_days(conn):
persist_day_aggregate(conn, aggregate)
from .day_aggregate import derive_all_days, persist_day_aggregate
for aggregate in derive_all_days(conn):
persist_day_aggregate(conn, aggregate)
from .local_day import derive_local_day_summary, persist_local_day
local_summary = derive_local_day_summary(conn, tz_name, observed_at)
if local_summary is not None:
persist_local_day(conn, local_summary)
from .local_day import derive_local_day_summary, persist_local_day
local_summary = derive_local_day_summary(conn, tz_name, observed_at)
if local_summary is not None:
persist_local_day(conn, local_summary)
except Exception as exc: # noqa: BLE001 - preserve valid staged evidence for retry.
raise _PendingDerivationError(exc) from exc
def _pending_count(conn: sqlite3.Connection) -> int:
@@ -381,12 +404,15 @@ def _recover_pending(conn: sqlite3.Connection) -> int:
pending_id, payload = row
observation = json.loads(payload)
_publish_observation(
conn,
observation["sample"],
observation["identity"],
observation["tz_name"],
)
try:
_publish_observation(
conn,
observation["sample"],
observation["identity"],
observation["tz_name"],
)
except Exception as exc:
raise _PendingRecoveryFailure(exc, recovered) from exc
conn.execute("DELETE FROM pending_publications WHERE id = ?", (pending_id,))
conn.commit()
recovered += 1
@@ -395,26 +421,70 @@ def _recover_pending(conn: sqlite3.Connection) -> int:
raise
def _is_store_or_invariant_failure(exc: Exception) -> bool:
return (
isinstance(exc, sqlite3.Error)
and not isinstance(exc, sqlite3.IntegrityError)
) or isinstance(exc, InvariantViolationError)
def _pending_capacity_error(
reason: str,
*,
acquisition_skipped: bool = False,
) -> RuntimeError:
outcome = "; no new observation acquired" if acquisition_skipped else ""
return RuntimeError(
"pending publication capacity full "
f"({PENDING_PUBLICATION_LIMIT} observations){outcome}; "
f"{reason}"
)
def _recover_pending_for_collection(conn: sqlite3.Connection) -> int:
"""Retry queued work and report exhausted capacity without masking store faults."""
try:
return _recover_pending(conn)
except Exception as exc:
if isinstance(exc, sqlite3.Error) and not isinstance(exc, sqlite3.IntegrityError):
raise
if isinstance(exc, InvariantViolationError):
raise
except Exception as exc: # noqa: BLE001 - preserve pending evidence on recovery errors.
cause = exc.cause if isinstance(exc, _PendingRecoveryFailure) else exc
if _is_store_or_invariant_failure(cause):
raise cause
try:
capacity_full = _pending_count(conn) >= PENDING_PUBLICATION_LIMIT
except sqlite3.Error:
raise exc
raise cause
if capacity_full:
raise RuntimeError(
"pending publication capacity full "
f"({PENDING_PUBLICATION_LIMIT} observations); no new observation acquired; "
f"recovery failed: {exc}"
) from exc
raise
raise _pending_capacity_error(f"recovery failed: {cause}") from cause
raise cause
def _recover_pending_for_admission(conn: sqlite3.Connection) -> int:
"""Recover in order, then admit one observation if bounded space remains."""
try:
recovered = _recover_pending(conn)
except _PendingRecoveryFailure as failure:
if (
not failure.derivation_error
or _is_store_or_invariant_failure(failure.cause)
):
raise failure.cause
try:
pending_count = _pending_count(conn)
except sqlite3.Error:
raise failure.cause
if pending_count >= PENDING_PUBLICATION_LIMIT:
raise _pending_capacity_error(
f"recovery failed: {failure.cause}",
acquisition_skipped=True,
) from failure.cause
return failure.published_count
if _pending_count(conn) >= PENDING_PUBLICATION_LIMIT:
raise _pending_capacity_error(
"recovery left the queue full",
acquisition_skipped=True,
)
return recovered
def run_collection(
@@ -444,16 +514,21 @@ def run_collection(
if history_path.exists():
import_legacy_history(conn, history_path, clock=clock)
published_count = _recover_pending_for_collection(conn)
published_count = _recover_pending_for_admission(conn)
checked_pending_count = _pending_count(conn)
# Hold the writer reservation across the capacity check and acquisition.
# A concurrent collector will recheck pending work before it acquires.
# Recheck recovery if another collector appended work during preflight.
while True:
conn.execute("BEGIN IMMEDIATE")
waiting = _pending_count(conn)
if waiting:
if (
waiting >= PENDING_PUBLICATION_LIMIT
or waiting > checked_pending_count
):
conn.rollback()
published_count += _recover_pending_for_collection(conn)
published_count += _recover_pending_for_admission(conn)
checked_pending_count = _pending_count(conn)
continue
from .tz_util import detect_system_tz
@@ -477,9 +552,21 @@ def run_collection(
break
published_count += _recover_pending_for_collection(conn)
retention_error = None
try:
if _pending_count(conn) == 0:
from .pruning import prune_old_samples
prune_old_samples(conn, clock.utcnow())
except Exception as exc: # noqa: BLE001 - publication already committed.
# Publication already committed. Keep collection successful and
# retry atomic retention on the next normal collection.
conn.rollback()
retention_error = str(exc)
return {
"ok": True,
"sample_count": published_count,
"retention_error": retention_error,
"store_path": str(store_path),
}
+22 -31
View File
@@ -33,6 +33,11 @@ from itertools import pairwise
from typing import Any
from zoneinfo import ZoneInfo
from .monitoring_periods import (
as_utc_datetime,
interval_within_one_monitoring_period,
)
@dataclass(frozen=True)
class LocalDaySummary:
@@ -308,27 +313,13 @@ def persist_local_day(
return created
def _utc_datetime(value: str | datetime) -> datetime:
parsed = value if isinstance(value, datetime) else datetime.fromisoformat(value)
if parsed.tzinfo is None:
parsed = parsed.replace(tzinfo=timezone.utc)
return parsed.astimezone(timezone.utc)
def _same_monitoring_period(
conn: sqlite3.Connection,
start: datetime,
end: datetime,
) -> bool:
"""Return whether one recorded monitoring period contains the interval."""
for started_at, ended_at in conn.execute(
"SELECT started_at, ended_at FROM monitoring_periods"
):
period_start = _utc_datetime(started_at)
period_end = _utc_datetime(ended_at) if ended_at else None
if period_start <= start and (period_end is None or end <= period_end):
return True
return False
return interval_within_one_monitoring_period(conn, start, end)
def repair_legacy_local_day_evidence(conn: sqlite3.Connection) -> int:
@@ -365,13 +356,13 @@ def repair_legacy_local_day_evidence(conn: sqlite3.Connection) -> int:
segment_windows = {}
if {"monitoring_periods", "controller_segments"}.issubset(tables):
periods = [
(_utc_datetime(start), _utc_datetime(end) if end else None)
(as_utc_datetime(start), as_utc_datetime(end) if end else None)
for start, end in conn.execute(
"SELECT started_at, ended_at FROM monitoring_periods"
)
]
segment_boundaries = [
(segment_id, _utc_datetime(opened_at))
(segment_id, as_utc_datetime(opened_at))
for segment_id, opened_at in conn.execute(
"SELECT id, opened_at FROM controller_segments ORDER BY opened_at, id"
)
@@ -399,7 +390,7 @@ def repair_legacy_local_day_evidence(conn: sqlite3.Connection) -> int:
)
samples = [
{
"id": row[0], "ts": _utc_datetime(row[1]),
"id": row[0], "ts": as_utc_datetime(row[1]),
"bytes_written": row[2], "bytes_read": row[3],
"segment_id": row[4], "local_tz": row[5],
}
@@ -453,8 +444,8 @@ def repair_legacy_local_day_evidence(conn: sqlite3.Connection) -> int:
"id": row[0],
"local_date": row[1],
"tz_name": row[2],
"start": _utc_datetime(row[3]),
"end": _utc_datetime(row[4]),
"start": as_utc_datetime(row[3]),
"end": as_utc_datetime(row[4]),
"precision": row[6],
}
for row in recorded_days
@@ -482,8 +473,8 @@ def repair_legacy_local_day_evidence(conn: sqlite3.Connection) -> int:
local_day_id, local_date, tz_name, utc_start, utc_end,
prior_incomplete,
) = row
start = _utc_datetime(utc_start)
end = _utc_datetime(utc_end)
start = as_utc_datetime(utc_start)
end = as_utc_datetime(utc_end)
matching = matching_by_day[local_day_id]
if not matching:
@@ -595,8 +586,8 @@ def _record_reconstructed_shared_intervals(
day_rows = [
{
"id": row[0], "local_date": row[1], "tz_name": row[2],
"utc_start": _utc_datetime(row[3]),
"utc_end": _utc_datetime(row[4]),
"utc_start": as_utc_datetime(row[3]),
"utc_end": as_utc_datetime(row[4]),
"precision": row[6],
}
for row in recorded_days
@@ -685,7 +676,7 @@ def _coarse_local_day_activity(
full_hours: dict[datetime, tuple] = {}
boundary_hour_found = False
other_day_bounds = [
(_utc_datetime(row[0]), _utc_datetime(row[1]))
(as_utc_datetime(row[0]), as_utc_datetime(row[1]))
for row in conn.execute(
"SELECT utc_start, utc_end FROM local_days WHERE id != ?",
(local_day_id,),
@@ -698,7 +689,7 @@ def _coarse_local_day_activity(
"FROM hour_observations WHERE hour >= ? AND hour < ? ORDER BY hour",
(first_hour.isoformat(), day_end.isoformat()),
):
hour_start = _utc_datetime(row[0])
hour_start = as_utc_datetime(row[0])
hour_end = hour_start + hour
if hour_start >= day_end or hour_end <= day_start:
continue
@@ -856,8 +847,8 @@ def record_local_activity_interval(
both samples share one monitoring period, and the interval stays inside one
recorded local date. Every other valid difference remains one unallocated row.
"""
start = _utc_datetime(previous["ts"])
end = _utc_datetime(current["ts"])
start = as_utc_datetime(previous["ts"])
end = as_utc_datetime(current["ts"])
if end <= start:
return None
@@ -1004,12 +995,12 @@ def mark_local_activity_gap(
previous: dict | None = None,
) -> None:
"""Mark dates around a controller boundary as missing local activity."""
current = _utc_datetime(current_ts)
current = as_utc_datetime(current_ts)
current_date = current.astimezone(ZoneInfo(current_tz)).date().isoformat()
_upsert_activity_day(conn, current_date, current_tz, incomplete=True)
if previous is None or not previous.get("local_tz"):
return
previous_at = _utc_datetime(previous["ts"])
previous_at = as_utc_datetime(previous["ts"])
previous_tz = previous["local_tz"]
previous_date = previous_at.astimezone(ZoneInfo(previous_tz)).date().isoformat()
_upsert_activity_day(conn, previous_date, previous_tz, incomplete=True)
@@ -1127,7 +1118,7 @@ def query_local_day_summary(
== recorded_date
)
day_seconds = int(
(_utc_datetime(utc_end) - _utc_datetime(utc_start)).total_seconds()
(as_utc_datetime(utc_end) - as_utc_datetime(utc_start)).total_seconds()
)
interval_seconds = activity_seconds or 0
full_activity_day = interval_seconds >= day_seconds or (
+41 -1
View File
@@ -9,7 +9,7 @@ Key contracts:
- End causes: user_disabled, migrated, unknown_gap
"""
import sqlite3
from datetime import datetime
from datetime import datetime, timezone
def ensure_period_open(conn: sqlite3.Connection, run_time: datetime) -> None:
@@ -82,6 +82,46 @@ def is_inside_period(conn: sqlite3.Connection, ts: datetime) -> bool:
return cursor.fetchone() is not None
def interval_within_one_monitoring_period(
conn: sqlite3.Connection,
start: str | datetime,
end: str | datetime,
) -> bool:
"""Return whether one monitoring period contains the full interval.
Compare timestamps as UTC instants. Malformed timestamps fail closed.
"""
try:
interval_start = as_utc_datetime(start)
interval_end = as_utc_datetime(end)
except (TypeError, ValueError, OverflowError):
return False
if interval_end <= interval_start:
return False
for started_at, ended_at in conn.execute(
"SELECT started_at, ended_at FROM monitoring_periods"
):
try:
period_start = as_utc_datetime(started_at)
period_end = as_utc_datetime(ended_at) if ended_at is not None else None
except (TypeError, ValueError, OverflowError):
continue
if period_start <= interval_start and (
period_end is None or interval_end <= period_end
):
return True
return False
def as_utc_datetime(value: str | datetime) -> datetime:
"""Parse a timestamp and normalize it to an aware UTC datetime."""
parsed = value if isinstance(value, datetime) else datetime.fromisoformat(value)
if parsed.tzinfo is None:
parsed = parsed.replace(tzinfo=timezone.utc)
return parsed.astimezone(timezone.utc)
def wall_clock_in_periods(
conn: sqlite3.Connection,
start: datetime,
+9 -10
View File
@@ -28,6 +28,7 @@ from datetime import datetime, timedelta, timezone
from enum import Enum
from typing import Any, Dict, List, Optional, Tuple
from .monitoring_periods import interval_within_one_monitoring_period
# ---------------------------------------------------------------------------
# Constants (spec §6)
@@ -494,18 +495,16 @@ def _has_complete_local_day(conn):
within a monitoring period that has usable observation evidence.
This is the prerequisite for showing an endurance outlook.
"""
return _count_complete_local_days(conn) > 0
def _count_complete_local_days(conn):
"""Count the number of complete local observation days."""
row = conn.execute(
"SELECT COUNT(*) FROM local_days "
local_days = conn.execute(
"SELECT utc_start, utc_end FROM local_days "
"WHERE complete = 1 "
"AND activity_precision IN ('measured', 'coarse') "
"AND activity_intervals > 0"
).fetchone()
return row[0] if row else 0
"AND activity_intervals > 0 ORDER BY id"
).fetchall()
return any(
interval_within_one_monitoring_period(conn, utc_start, utc_end)
for utc_start, utc_end in local_days
)
# ---------------------------------------------------------------------------
+349 -145
View File
@@ -1,131 +1,321 @@
"""Raw sample pruning per spec §3.4, ST-5.
"""Expire detail only after UTC and local_days replacement evidence is durable.
Raw samples are pruned opportunistically to 14 days.
Hour observations and day aggregates are retained indefinitely.
Boundary anchors required for successor evidence are retained.
Local-day summaries (local_days table) are never touched by pruning.
They are persisted at collection time from hour observations and survive
raw-sample pruning because they depend on hour observations, not on raw
samples. This is the mechanism that keeps local-day history trustworthy
after detail expires (issue #93).
The 14-day cutoff never overrides local-day preservation, shared boundary
evidence, publication recovery, or the newest sample's successor-anchor role.
"""
import sqlite3
from datetime import datetime, timedelta, timezone
from datetime import date, datetime, timedelta
from .monitoring_periods import (
as_utc_datetime,
interval_within_one_monitoring_period,
)
# Spec §3.4: Raw-sample retention
RAW_SAMPLE_RETENTION_DAYS = 14
def _parse_sample_time(value: str) -> datetime | None:
"""Parse old timestamps conservatively; malformed legacy values stay."""
try:
parsed = datetime.fromisoformat(value)
except (TypeError, ValueError):
return None
return as_utc_datetime(parsed)
def _sample_rows(conn: sqlite3.Connection) -> list[tuple]:
return conn.execute(
"SELECT id, ts, segment_id, bytes_written, bytes_read, local_tz "
"FROM samples ORDER BY id"
).fetchall()
def _required_hour_window(
start: datetime,
end: datetime,
) -> tuple[datetime, datetime, int] | None:
"""Return first hour, exclusive end and count for [start, end)."""
if end <= start:
return None
first_hour = start.replace(minute=0, second=0, microsecond=0)
last_hour = end.replace(minute=0, second=0, microsecond=0)
try:
exclusive_end = last_hour if end == last_hour else last_hour + timedelta(hours=1)
except OverflowError:
return None
count = int((exclusive_end - first_hour).total_seconds() // 3600)
return first_hour, exclusive_end, count
def _has_utc_replacement(
conn: sqlite3.Connection,
previous: tuple,
current: tuple,
start: datetime,
end: datetime,
) -> bool:
window = _required_hour_window(start, end)
if window is None:
return False
first_hour, exclusive_end, expected_hour_count = window
actual_hour_count = conn.execute(
"SELECT COUNT(*) FROM hour_observations WHERE hour >= ? AND hour < ?",
(
first_hour.strftime("%Y-%m-%dT%H:00:00+00:00"),
exclusive_end.strftime("%Y-%m-%dT%H:00:00+00:00"),
),
).fetchone()[0]
if actual_hour_count < expected_hour_count:
return False
start_written, end_written = previous[3], current[3]
start_read, end_read = previous[4], current[4]
if None in (start_written, end_written, start_read, end_read):
return False
if end_written < start_written or end_read < start_read:
return False
bytes_written = end_written - start_written
bytes_read = end_read - start_read
if expected_hour_count == 1:
totals = conn.execute(
"SELECT bytes_written_delta, bytes_read_delta FROM hour_observations "
"WHERE hour = ?",
(first_hour.strftime("%Y-%m-%dT%H:00:00+00:00"),),
).fetchone()
else:
totals = conn.execute(
"SELECT unattributed_bytes_written, unattributed_bytes_read "
"FROM day_aggregates WHERE day = ?",
(start.date().isoformat(),),
).fetchone()
if totals is None or totals[0] < bytes_written or totals[1] < bytes_read:
return False
last_day = (exclusive_end - timedelta(microseconds=1)).date()
expected_day_count = (last_day - first_hour.date()).days + 1
actual_day_count = conn.execute(
"SELECT COUNT(*) FROM day_aggregates WHERE day >= ? AND day <= ?",
(first_hour.date().isoformat(), last_day.isoformat()),
).fetchone()[0]
return actual_day_count >= expected_day_count
def _local_day_row_exists(
conn: sqlite3.Connection,
local_date: str,
tz_name: str,
) -> bool:
return conn.execute(
"SELECT 1 FROM local_days WHERE local_date = ? AND tz_name = ?",
(local_date, tz_name),
).fetchone() is not None
def _local_day_bounds(
conn: sqlite3.Connection,
local_date: str,
tz_name: str,
) -> tuple[datetime, datetime] | None:
row = conn.execute(
"SELECT utc_start, utc_end FROM local_days "
"WHERE local_date = ? AND tz_name = ?",
(local_date, tz_name),
).fetchone()
if row is None:
return None
start = _parse_sample_time(row[0])
end = _parse_sample_time(row[1])
if start is None or end is None or end <= start:
return None
return start, end
def _contains_instant(
conn: sqlite3.Connection,
local_date: str,
tz_name: str,
instant: datetime,
) -> bool:
bounds = _local_day_bounds(conn, local_date, tz_name)
return bounds is not None and bounds[0] <= instant < bounds[1]
def _monitoring_period_covers(
conn: sqlite3.Connection,
start: datetime,
end: datetime,
) -> bool:
"""Require continuous monitoring before a local-day total replaces detail."""
return interval_within_one_monitoring_period(conn, start, end)
def _has_local_replacement(
conn: sqlite3.Connection,
previous: tuple,
current: tuple,
start: datetime,
end: datetime,
) -> bool:
start_tz = previous[5]
end_tz = current[5]
if not end_tz:
return False
start_id, end_id = previous[0], current[0]
segment_id = current[2]
if (
start_tz
and start_tz == end_tz
and segment_id is not None
and _monitoring_period_covers(conn, start, end)
):
known_days = conn.execute(
"SELECT local_days.utc_start, local_days.utc_end "
"FROM local_days JOIN local_day_segment_totals "
" ON local_day_segment_totals.local_day_id = local_days.id "
"WHERE local_days.tz_name = ? "
" AND local_days.activity_precision IN ('measured', 'coarse') "
" AND local_days.last_sample_id >= ? "
" AND local_days.activity_intervals > 0 "
" AND local_day_segment_totals.segment_id = ? "
" AND local_day_segment_totals.activity_intervals > 0",
(end_tz, end_id, segment_id),
).fetchall()
for row in known_days:
bounds_start = _parse_sample_time(row[0])
bounds_end = _parse_sample_time(row[1])
if (
bounds_start is not None
and bounds_end is not None
and bounds_start <= start
and end < bounds_end
):
return True
evidence = conn.execute(
"SELECT bytes_written, bytes_read, reason, start_local_date, end_local_date, "
" start_tz_name, end_tz_name, started_at, ended_at "
"FROM local_day_unallocated_evidence "
"WHERE start_sample_id = ? AND end_sample_id = ?",
(start_id, end_id),
).fetchone()
if evidence is None:
return False
start_written, end_written = previous[3], current[3]
start_read, end_read = previous[4], current[4]
if None in (start_written, end_written, start_read, end_read):
return False
if end_written < start_written or end_read < start_read:
return False
# The source IDs uniquely identify this interval. Confirm its durable
# deltas and retain every recorded local-day boundary it touches.
if (end_written - start_written, end_read - start_read) != evidence[:2]:
return False
if evidence[5] != start_tz or evidence[6] != end_tz:
return False
if evidence[2] not in {
"counter_discontinuity",
"legacy_timezone_unknown",
"timezone_change",
"monitoring_period",
"local_midnight",
"segment_unknown",
}:
return False
if _parse_sample_time(evidence[7]) != start or _parse_sample_time(evidence[8]) != end:
return False
affected_days: set[tuple[str, str]] = set()
if start_tz:
affected_days.add((evidence[3], start_tz))
affected_days.add((evidence[4], end_tz))
if start_tz and start_tz == end_tz and evidence[3] < evidence[4]:
day = date.fromisoformat(evidence[3]) + timedelta(days=1)
last_day = date.fromisoformat(evidence[4])
while day < last_day:
affected_days.add((day.isoformat(), end_tz))
day += timedelta(days=1)
if not all(
_local_day_row_exists(conn, local_date, zone)
for local_date, zone in affected_days
):
return False
if start_tz and not _contains_instant(conn, evidence[3], start_tz, start):
return False
return _contains_instant(conn, evidence[4], end_tz, end)
def _interval_has_replacement(
conn: sqlite3.Connection,
previous: tuple,
current: tuple,
) -> bool:
start = _parse_sample_time(previous[1])
end = _parse_sample_time(current[1])
if start is None or end is None or end <= start:
return False
try:
return (
_has_utc_replacement(conn, previous, current, start, end)
and _has_local_replacement(conn, previous, current, start, end)
)
except (sqlite3.Error, ValueError, OverflowError, TypeError):
# Missing or unreadable derived evidence never authorizes deletion.
return False
def _sample_needs_preservation(
conn: sqlite3.Connection,
rows: list[tuple],
index: int,
newest_id: int,
) -> bool:
sample = rows[index]
segment_id = sample[2]
if segment_id is None or conn.execute(
"SELECT 1 FROM controller_segments WHERE id = ?",
(segment_id,),
).fetchone() is None:
return True
if sample[0] == newest_id:
# Keep newest sample as the source anchor for the next collection.
return True
pairs = []
if index > 0 and rows[index - 1][2] == segment_id:
pairs.append((rows[index - 1], sample))
if index + 1 < len(rows) and rows[index + 1][2] == segment_id:
pairs.append((sample, rows[index + 1]))
return any(
not _interval_has_replacement(conn, previous, current)
for previous, current in pairs
)
def needs_boundary_anchor(
conn: sqlite3.Connection,
sample_ts: str,
now: datetime,
) -> bool:
"""Check if a sample is needed as a boundary anchor for derivation.
A sample is a boundary anchor if:
1. It's older than retention_days (strictly before cutoff)
2. It has a next sample that forms an interval spanning the retention boundary
3. The interval hasn't been derived yet
The interval spans the boundary if:
- The sample is before the cutoff, AND
- The next sample is strictly after the cutoff (or within retention)
"""
from .derive import _parse_ts
sample_dt = _parse_ts(sample_ts)
retention_cutoff = now - timedelta(days=RAW_SAMPLE_RETENTION_DAYS)
# If sample is within retention (strictly after cutoff), not an anchor
if sample_dt > retention_cutoff:
"""Return whether an old sample still carries unreplaced evidence."""
sample_time = _parse_sample_time(sample_ts)
cutoff = as_utc_datetime(now) - timedelta(days=RAW_SAMPLE_RETENTION_DAYS)
if sample_time is None:
return True
if sample_time >= cutoff:
return False
# Check if this sample has a next sample
cursor = conn.execute(
"""SELECT ts, segment_id FROM samples WHERE ts > ? ORDER BY ts LIMIT 1""",
(sample_ts,),
rows = _sample_rows(conn)
matching = [index for index, row in enumerate(rows) if row[1] == sample_ts]
if not matching:
return False
newest_id = rows[-1][0] if rows else -1
return any(
_sample_needs_preservation(conn, rows, index, newest_id)
for index in matching
)
next_row = cursor.fetchone()
if next_row is None:
# No next sample - this is the last sample
# It's not needed for derivation (no interval to derive)
return False
next_ts_str = next_row[0]
next_segment_id = next_row[1]
next_dt = _parse_ts(next_ts_str)
# Check if the next sample is strictly after the cutoff (i.e., interval spans boundary)
if next_dt > retention_cutoff:
# The interval spans the retention boundary
# Check if the interval needs derivation
# Get current sample's segment_id
cursor = conn.execute(
"SELECT segment_id FROM samples WHERE ts = ?",
(sample_ts,),
)
current_segment_row = cursor.fetchone()
current_segment_id = current_segment_row[0] if current_segment_row else None
# If different segments, no interval to derive
if current_segment_id != next_segment_id:
return False
# Check if the interval [sample_ts, next_ts] needs derivation
# It needs derivation if any hour in the span lacks an observation
current_hour = sample_dt.replace(minute=0, second=0, microsecond=0)
end_hour = next_dt.replace(minute=0, second=0, microsecond=0)
while current_hour <= end_hour:
cursor = conn.execute(
"SELECT id FROM hour_observations WHERE hour = ?",
(current_hour.isoformat(),),
)
if cursor.fetchone() is None:
# This hour lacks an observation - interval needs derivation
return True
current_hour += timedelta(hours=1)
# All hours in the span have observations - interval is derived
return False
else:
# The interval doesn't span the boundary (both samples are old)
# Check if the interval needs derivation
# Get current sample's segment_id
cursor = conn.execute(
"SELECT segment_id FROM samples WHERE ts = ?",
(sample_ts,),
)
current_segment_row = cursor.fetchone()
current_segment_id = current_segment_row[0] if current_segment_row else None
# If different segments, no interval to derive
if current_segment_id != next_segment_id:
return False
# Check if the interval [sample_ts, next_ts] needs derivation
current_hour = sample_dt.replace(minute=0, second=0, microsecond=0)
end_hour = next_dt.replace(minute=0, second=0, microsecond=0)
while current_hour <= end_hour:
cursor = conn.execute(
"SELECT id FROM hour_observations WHERE hour = ?",
(current_hour.isoformat(),),
)
if cursor.fetchone() is None:
# This hour lacks an observation - interval needs derivation
# But only keep if the interval is significant (spans multiple hours)
# or if the next sample is the last sample before a gap
gap = (next_dt - sample_dt).total_seconds()
if gap > 24 * 3600: # Significant gap (> 24 hours)
return True
current_hour += timedelta(hours=1)
# All hours in the span have observations or gap is not significant
return False
def prune_old_samples(
@@ -133,36 +323,50 @@ def prune_old_samples(
now: datetime,
retention_days: int = RAW_SAMPLE_RETENTION_DAYS,
) -> int:
"""Remove raw samples older than retention_days.
"""Delete expired samples only after every dependent fact is durable.
Retains boundary anchors required for successor evidence.
Args:
conn: Connection to the observation store.
now: Current UTC time.
retention_days: Number of days to retain (default 14).
Returns:
Number of samples removed.
Pending publications are stored separately and never selected here. A
sample stays when it is the newest successor anchor, has an unpublishable
neighbour interval, lacks UTC or local-day replacement evidence, or has a
timestamp that cannot be safely interpreted.
"""
cutoff = (now - timedelta(days=retention_days)).isoformat()
# Get all samples older than cutoff
cursor = conn.execute(
"SELECT id, ts FROM samples WHERE ts < ? ORDER BY ts",
(cutoff,),
)
old_samples = cursor.fetchall()
removed = 0
for sample_id, sample_ts in old_samples:
# Check if this sample is a boundary anchor
if needs_boundary_anchor(conn, sample_ts, now):
continue # Skip - it's a boundary anchor
# Remove the sample
conn.execute("DELETE FROM samples WHERE id = ?", (sample_id,))
removed += 1
conn.commit()
return removed
cutoff = as_utc_datetime(now) - timedelta(days=retention_days)
owns_transaction = not conn.in_transaction
savepoint = "fenris_sample_retention"
if owns_transaction:
conn.execute("BEGIN IMMEDIATE")
else:
conn.execute(f"SAVEPOINT {savepoint}")
try:
rows = _sample_rows(conn)
if not rows:
if owns_transaction:
conn.commit()
else:
conn.execute(f"RELEASE SAVEPOINT {savepoint}")
return 0
newest_id = rows[-1][0]
expired = []
for index, row in enumerate(rows):
sample_time = _parse_sample_time(row[1])
if sample_time is None or sample_time >= cutoff:
continue
if _sample_needs_preservation(conn, rows, index, newest_id):
continue
expired.append(row[0])
conn.executemany("DELETE FROM samples WHERE id = ?", ((sample_id,) for sample_id in expired))
if owns_transaction:
conn.commit()
else:
conn.execute(f"RELEASE SAVEPOINT {savepoint}")
return len(expired)
except Exception:
if owns_transaction:
conn.rollback()
else:
conn.execute(f"ROLLBACK TO SAVEPOINT {savepoint}")
conn.execute(f"RELEASE SAVEPOINT {savepoint}")
raise
+637 -157
View File
File diff suppressed because it is too large Load Diff
+94
View File
@@ -0,0 +1,94 @@
"""Activity-selection transitions independent from Textual widgets."""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
from fenris.activity_selection import ActivitySelection, IntervalIdentity
def _interval(n):
return IntervalIdentity(f"start-{n}", f"end-{n}")
def test_live_selection_starts_and_stays_on_newest_interval():
selection = ActivitySelection()
first = [_interval(1), _interval(2)]
selection.update_live(first)
assert selection.following_live
assert selection.selected_live_interval == first[-1]
second = first + [_interval(3)]
selection.update_live(second)
assert selection.selected_live_interval == second[-1]
def test_keyboard_and_mouse_inspection_pin_interval_identity():
intervals = [_interval(1), _interval(2), _interval(3)]
keyboard = ActivitySelection()
mouse = ActivitySelection()
keyboard.update_live(intervals)
keyboard.move_live(intervals, -1)
mouse.update_live(intervals)
mouse.inspect_live(intervals, intervals[1])
assert keyboard.selected_live_interval == mouse.selected_live_interval == intervals[1]
assert not keyboard.following_live
assert not mouse.following_live
refreshed = [_interval(2), _interval(3), _interval(4)]
keyboard.update_live(refreshed)
mouse.update_live(refreshed)
assert keyboard.selected_live_interval == mouse.selected_live_interval == intervals[1]
def test_expired_pin_explains_expiration_and_resumes_following():
selection = ActivitySelection()
intervals = [_interval(1), _interval(2), _interval(3)]
selection.update_live(intervals)
selection.inspect_live(intervals, intervals[0])
newest = _interval(4)
selection.update_live([_interval(2), _interval(3), newest])
assert selection.following_live
assert selection.selected_live_interval == newest
assert selection.live_interval_expired
def test_today_resets_expired_pin_and_read_write_toggle_keeps_identity():
selection = ActivitySelection()
intervals = [_interval(1), _interval(2)]
selection.update_live(intervals)
selection.inspect_live(intervals, intervals[0])
selection.update_live([intervals[1]])
newest = _interval(3)
selection.update_live([intervals[1], newest])
assert selection.live_interval_expired
selection.toggle_measure()
assert selection.measure == "read"
assert selection.selected_live_interval == newest
selection.select_history_date("2026-09-27")
selection.set_view("history")
selection.set_view("live")
assert selection.browse_date is None
assert selection.following_live
assert not selection.live_interval_expired
def test_live_identity_uses_both_interval_boundaries():
first = IntervalIdentity("2026-09-28T10:00:00Z", "2026-09-28T10:03:00Z")
changed_start = IntervalIdentity("2026-09-28T09:57:00Z", "2026-09-28T10:03:00Z")
selection = ActivitySelection()
selection.update_live([first])
selection.inspect_live([first], first)
selection.update_live([changed_start])
assert selection.following_live
assert selection.selected_live_interval == changed_start
assert selection.live_interval_expired
@@ -187,6 +187,39 @@ class TestGateNoCompleteDay:
assert any("full local observation day" in fact
for fact in result.contributing_facts)
def test_day_split_by_deliberate_pause_does_not_open_gate(self, store):
"""A complete-looking summary cannot span separate monitoring periods."""
_insert_baseline(store)
_insert_segment(store)
for i in range(20):
day = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
_insert_day(store, day, bw=1024 * 1024 * 100)
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
store.executemany(
"INSERT INTO monitoring_periods (started_at, ended_at, end_cause) "
"VALUES (?, ?, ?)",
[
("2026-09-29T00:00:00+00:00", "2026-09-29T12:00:00+00:00", "user_disabled"),
("2026-09-29T13:00:00+00:00", "2026-09-30T00:00:00+00:00", "user_disabled"),
("2026-09-30T00:00:00+00:00", None, None),
],
)
store.commit()
_insert_local_day(
store,
"2026-09-29",
utc_start="2026-09-29T00:00:00+00:00",
utc_end="2026-09-30T00:00:00+00:00",
complete=True,
)
result = compute_projection(store, _clock())
assert result.confidence_state == ConfidenceState.UNSUPPORTED
assert any("full local observation day" in fact
for fact in result.contributing_facts)
def test_partial_but_trusted_local_activity_opens_gate(self, store):
"""Known local intervals can coexist with an incomplete day total."""
_insert_baseline(store)
+1
View File
@@ -102,6 +102,7 @@ async def test_tabs_date_entry_and_hourly_inspection_keep_context(dashboard):
app.on_refresh_tick()
await pilot.pause()
assert str(app.query_one("#bar-readout").render()) == readout
assert "Reads" in str(app.query_one("#bar-legend").render())
await pilot.press("g")
await pilot.press(*list("2026-09-17"))
await pilot.press("escape")
+10 -2
View File
@@ -685,11 +685,19 @@ class TestEdgeCases:
def test_runit_collect_missing_script(self):
"""Runit collect exits when fenris-collect script not found."""
with patch("fenris.init_system.Path") as mock_path:
mock_path.return_value.exists.return_value = False
with patch("fenris.init_system.subprocess") as mock_sub, \
patch("fenris.init_system.Path") as mock_path:
missing = MagicMock()
missing.exists.return_value = False
# Fallback path is built via Path(...).parent... / "src" / ...
missing.parent = missing
missing.__truediv__.return_value = missing
mock_path.return_value = missing
mock_sub.run.return_value = MagicMock(returncode=1)
with pytest.raises(SystemExit) as exc_info:
_runit_collect()
assert exc_info.value.code == 1
mock_sub.run.assert_not_called()
def test_systemd_query_state_timeout(self):
"""Systemd query state handles subprocess timeout."""
+341
View File
@@ -0,0 +1,341 @@
"""Normal-collection retention acceptance tests for issue #100."""
import sqlite3
import sys
from datetime import datetime, timedelta, timezone
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
from fenris.collector import run_collection
from fenris.local_day import query_local_day_summary
from fenris.pruning import prune_old_samples
from fenris.status import read_status
from fenris.store import init_store
@pytest.fixture
def sysfs_controller(tmp_path):
controller = tmp_path / "sys" / "class" / "nvme" / "nvme0"
controller.mkdir(parents=True)
(controller / "subsysnqn").write_text("nqn.example:drive-1\n")
(controller / "model").write_text("Fenris Test Drive\n")
(controller / "serial").write_text("drive-1\n")
(controller / "firmware_rev").write_text("1.0\n")
transport = controller / "transport"
transport.mkdir()
(transport / "address").write_text("0000:00:01.0\n")
(transport / "trstring").write_text("pcie\n")
return controller
class FakeClock:
def __init__(self, now):
self.now = now
def utcnow(self):
return self.now
def _smartctl(written, read):
return {
"nvme_smart_health_information_log": {
"critical_warning": 0,
"temperature": 35,
"available_spare": 100,
"percentage_used": 1,
"data_units_written": written,
"data_units_read": read,
"power_on_hours": 100,
"power_cycles": 1,
"unsafe_shutdowns": 0,
"media_errors": 0,
},
"user_capacity": {"bytes": 1_000_000_000_000},
"model_name": "Fenris Test Drive",
"serial_number": "drive-1",
"firmware_version": "1.0",
}
def _collect(config, controller, now, written, read):
return run_collection(
smartctl_data=_smartctl(written, read),
sysfs_path=controller,
config=config,
clock=FakeClock(now),
)
def test_normal_collection_expires_detail_and_keeps_local_day_evidence(
tmp_path, sysfs_controller, monkeypatch,
):
monkeypatch.setenv("TZ", "UTC")
store_path = tmp_path / "observations.db"
config = {"device": "/dev/nvme0", "store_path": str(store_path)}
first_time = datetime(2026, 9, 1, 0, 0, tzinfo=timezone.utc)
second_time = first_time + timedelta(days=15)
first = _collect(config, sysfs_controller, first_time, 1000, 2000)
assert first["ok"] is True, first
second = _collect(config, sysfs_controller, second_time, 1001, 2003)
assert second["ok"] is True, second
with sqlite3.connect(store_path) as conn:
assert conn.execute("SELECT ts FROM samples ORDER BY id").fetchall() == [
(second_time.isoformat(),)
]
assert conn.execute(
"SELECT bytes_written, bytes_read, reason "
"FROM local_day_unallocated_evidence"
).fetchall() == [(512_000, 1_536_000, "local_midnight")]
assert conn.execute("SELECT COUNT(*) FROM pending_publications").fetchone()[0] == 0
assert conn.execute("SELECT COUNT(*) FROM controller_segments").fetchone()[0] == 1
assert conn.execute("SELECT COUNT(*) FROM day_aggregates").fetchone()[0] == 15
monkeypatch.setenv("TZ", "Asia/Kolkata")
with read_status(store_path, second_time, query_services=False) as (reader, state):
assert reader is not None
assert state.store_fault is None
old_day = query_local_day_summary(
reader, "2026-09-01", "UTC", second_time,
)
assert old_day["utc_start"] == "2026-09-01T00:00:00+00:00"
assert old_day["utc_end"] == "2026-09-02T00:00:00+00:00"
assert old_day["bytes_written"] is None
assert old_day["shared_bytes_written"] == 512_000
assert old_day["shared_bytes_read"] == 1_536_000
assert old_day["shared_evidence_count"] == 1
def test_pruning_keeps_samples_when_replacement_evidence_is_missing(tmp_path):
store_path = tmp_path / "incomplete.db"
conn = init_store(store_path)
now = datetime(2026, 9, 30, tzinfo=timezone.utc)
old_time = now - timedelta(days=20)
next_time = now - timedelta(days=1)
conn.execute(
"INSERT INTO controller_segments (id, opened_at, identity_key) "
"VALUES (1, ?, 'test')",
(old_time.isoformat(),),
)
conn.execute(
"INSERT INTO samples (ts, device, bytes_written, bytes_read, segment_id, local_tz) "
"VALUES (?, '/dev/nvme0', 100, 100, 1, 'UTC')",
(old_time.isoformat(),),
)
conn.execute(
"INSERT INTO samples (ts, device, bytes_written, bytes_read, segment_id, local_tz) "
"VALUES (?, '/dev/nvme0', 200, 200, 1, 'UTC')",
(next_time.isoformat(),),
)
conn.commit()
assert prune_old_samples(conn, now) == 0
assert conn.execute("SELECT COUNT(*) FROM samples").fetchone()[0] == 2
conn.close()
def test_later_same_day_activity_does_not_replace_gap_interval(tmp_path):
store_path = tmp_path / "unrelated-local-activity.db"
conn = init_store(store_path)
gap_start = datetime(2026, 9, 1, 0, 0, tzinfo=timezone.utc)
gap_end = gap_start + timedelta(minutes=10)
later_end = gap_start + timedelta(minutes=20)
now = datetime(2026, 9, 30, tzinfo=timezone.utc)
conn.execute(
"INSERT INTO controller_segments (id, opened_at, identity_key) "
"VALUES (1, ?, 'test')",
(gap_start.isoformat(),),
)
for sample_time, written, read in (
(gap_start, 1000, 100),
(gap_end, 2000, 200),
(later_end, 2100, 210),
):
conn.execute(
"INSERT INTO samples (ts, device, bytes_written, bytes_read, segment_id, local_tz) "
"VALUES (?, '/dev/nvme0', ?, ?, 1, 'UTC')",
(sample_time.isoformat(), written, read),
)
# UTC aggregates can replace the old counter pair, but the local-day
# summary contains only the later B->C interval, after monitoring resumed.
conn.execute(
"INSERT INTO hour_observations (hour, bytes_written_delta, bytes_read_delta) "
"VALUES ('2026-09-01T00:00:00+00:00', 1000, 100)"
)
conn.execute(
"INSERT INTO day_aggregates (day) VALUES ('2026-09-01')"
)
conn.execute(
"INSERT INTO monitoring_periods (started_at, ended_at) VALUES (?, ?)",
(gap_end.isoformat(), later_end.isoformat()),
)
last_sample_id = conn.execute("SELECT MAX(id) FROM samples").fetchone()[0]
conn.execute(
"INSERT INTO local_days "
"(local_date, tz_name, tz_offset, utc_start, utc_end, bytes_written, "
" bytes_read, activity_intervals, activity_precision, last_sample_id) "
"VALUES ('2026-09-01', 'UTC', '+00:00', ?, ?, 100, 10, 1, 'measured', ?)",
(
gap_start.isoformat(),
(gap_start + timedelta(days=1)).isoformat(),
last_sample_id,
),
)
local_day_id = conn.execute("SELECT MAX(id) FROM local_days").fetchone()[0]
conn.execute(
"INSERT INTO local_day_segment_totals "
"(local_day_id, segment_id, bytes_written, bytes_read, activity_intervals) "
"VALUES (?, 1, 100, 10, 1)",
(local_day_id,),
)
conn.commit()
assert prune_old_samples(conn, now) == 0
assert conn.execute("SELECT COUNT(*) FROM samples").fetchone()[0] == 3
conn.close()
def test_pruning_keeps_oldest_unpaired_sample_as_successor_anchor(tmp_path):
store_path = tmp_path / "anchor.db"
conn = init_store(store_path)
now = datetime(2026, 9, 30, tzinfo=timezone.utc)
old_time = now - timedelta(days=20)
conn.execute(
"INSERT INTO controller_segments (id, opened_at, identity_key) "
"VALUES (1, ?, 'test')",
(old_time.isoformat(),),
)
conn.execute(
"INSERT INTO samples (ts, device, bytes_written, bytes_read, segment_id, local_tz) "
"VALUES (?, '/dev/nvme0', 100, 100, 1, 'UTC')",
(old_time.isoformat(),),
)
conn.commit()
assert prune_old_samples(conn, now) == 0
assert conn.execute("SELECT ts FROM samples").fetchone()[0] == old_time.isoformat()
conn.close()
def test_pruning_never_expires_pending_publications(tmp_path):
store_path = tmp_path / "pending.db"
conn = init_store(store_path)
old_time = datetime(2026, 9, 1, tzinfo=timezone.utc)
conn.execute(
"INSERT INTO pending_publications (sample_ts, payload) VALUES (?, '{}')",
(old_time.isoformat(),),
)
conn.commit()
assert prune_old_samples(
conn, datetime(2026, 9, 30, tzinfo=timezone.utc),
) == 0
assert conn.execute(
"SELECT sample_ts FROM pending_publications"
).fetchone()[0] == old_time.isoformat()
conn.close()
def test_pruning_preserves_unparseable_legacy_timestamp(tmp_path):
store_path = tmp_path / "legacy.db"
conn = init_store(store_path)
conn.execute(
"INSERT INTO samples (ts, device, bytes_written, bytes_read) "
"VALUES ('legacy timestamp', '/dev/nvme0', 100, 100)"
)
conn.commit()
assert prune_old_samples(conn, datetime(2026, 9, 30, tzinfo=timezone.utc)) == 0
assert conn.execute("SELECT ts FROM samples").fetchone()[0] == "legacy timestamp"
conn.close()
def test_collection_defers_interrupted_pruning_and_retries_next_run(
tmp_path, sysfs_controller, monkeypatch,
):
monkeypatch.setenv("TZ", "UTC")
store_path = tmp_path / "interrupted.db"
config = {"device": "/dev/nvme0", "store_path": str(store_path)}
first_time = datetime(2026, 9, 1, 0, 0, tzinfo=timezone.utc)
second_time = first_time + timedelta(days=15)
assert _collect(config, sysfs_controller, first_time, 1000, 2000)["ok"] is True
writer = sqlite3.connect(store_path)
writer.execute(
"CREATE TRIGGER stop_retention BEFORE DELETE ON samples "
"BEGIN SELECT RAISE(ABORT, 'simulated retention interruption'); END"
)
writer.commit()
writer.close()
second = _collect(config, sysfs_controller, second_time, 1001, 2001)
assert second["ok"] is True, second
assert "simulated retention interruption" in second["retention_error"]
with sqlite3.connect(store_path) as conn:
assert conn.execute("SELECT COUNT(*) FROM samples").fetchone()[0] == 2
assert conn.execute("SELECT COUNT(*) FROM pending_publications").fetchone()[0] == 0
assert conn.execute("SELECT COUNT(*) FROM local_day_unallocated_evidence").fetchone()[0] == 1
writer = sqlite3.connect(store_path)
writer.execute("DROP TRIGGER stop_retention")
writer.commit()
writer.close()
third_time = second_time + timedelta(days=15)
third = _collect(config, sysfs_controller, third_time, 1002, 2002)
assert third["ok"] is True, third
assert third["retention_error"] is None
with sqlite3.connect(store_path) as conn:
assert conn.execute("SELECT ts FROM samples ORDER BY id").fetchall() == [
(third_time.isoformat(),)
]
def test_collection_recovers_old_pending_work_before_pruning(
tmp_path, sysfs_controller, monkeypatch,
):
monkeypatch.setenv("TZ", "UTC")
store_path = tmp_path / "pending-recovery.db"
config = {"device": "/dev/nvme0", "store_path": str(store_path)}
first_time = datetime(2026, 9, 1, 0, 0, tzinfo=timezone.utc)
second_time = first_time + timedelta(days=15)
assert _collect(config, sysfs_controller, first_time, 1000, 2000)["ok"] is True
writer = sqlite3.connect(store_path)
writer.execute(
"CREATE TRIGGER interrupt_publication "
"BEFORE INSERT ON local_day_unallocated_evidence "
"BEGIN SELECT RAISE(ABORT, 'simulated publication interruption'); END"
)
writer.commit()
writer.close()
failed = _collect(config, sysfs_controller, second_time, 1001, 2001)
assert failed["ok"] is False
assert "simulated publication interruption" in failed["error"]
with sqlite3.connect(store_path) as conn:
assert conn.execute("SELECT COUNT(*) FROM samples").fetchone()[0] == 1
assert conn.execute("SELECT COUNT(*) FROM pending_publications").fetchone()[0] == 1
writer = sqlite3.connect(store_path)
writer.execute("DROP TRIGGER interrupt_publication")
writer.commit()
writer.close()
third_time = second_time + timedelta(minutes=5)
recovered = _collect(config, sysfs_controller, third_time, 1002, 2002)
assert recovered["ok"] is True, recovered
assert recovered["retention_error"] is None
with sqlite3.connect(store_path) as conn:
assert conn.execute("SELECT COUNT(*) FROM pending_publications").fetchone()[0] == 0
assert conn.execute("SELECT ts FROM samples ORDER BY id").fetchall() == [
(second_time.isoformat(),),
(third_time.isoformat(),),
]
assert conn.execute(
"SELECT COUNT(*) FROM local_day_unallocated_evidence"
).fetchone()[0] == 1
+257
View File
@@ -0,0 +1,257 @@
"""Bound pending-publication admission before device acquisition."""
import json
import sqlite3
import sys
from datetime import datetime, timedelta, timezone
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
from fenris import collector
class FakeClock:
def __init__(self, now):
self.now = now
def utcnow(self):
return self.now
@pytest.fixture
def sysfs_controller(tmp_path):
controller_path = tmp_path / "sys" / "class" / "nvme" / "nvme0"
controller_path.mkdir(parents=True)
(controller_path / "subsysnqn").write_text("nqn.test:drive\n")
(controller_path / "model").write_text("Test NVMe\n")
(controller_path / "serial").write_text("test-serial\n")
(controller_path / "firmware_rev").write_text("1.0\n")
transport = controller_path / "transport"
transport.mkdir()
(transport / "trstring").write_text("pcie\n")
return controller_path
def _smartctl(units_written):
return {
"nvme_smart_health_information_log": {
"critical_warning": 0,
"temperature": 35,
"available_spare": 100,
"percentage_used": 5,
"data_units_written": units_written,
"data_units_read": 500,
"power_on_hours": 100,
},
"user_capacity": {"bytes": 1_024_000_000_000},
"model_name": "Test NVMe",
"serial_number": "test-serial",
"firmware_version": "1.0",
}
def _collect(config, controller, now, units_written):
return collector.run_collection(
config=config,
clock=FakeClock(now),
acquire=lambda: (_smartctl(units_written), controller),
)
def test_partial_recovery_frees_admission_and_keeps_pending_order(
tmp_path, sysfs_controller, monkeypatch,
):
"""A freed slot admits one observation behind older pending evidence."""
assert collector.PENDING_PUBLICATION_LIMIT == 6_720
monkeypatch.setenv("TZ", "UTC")
monkeypatch.setattr(collector, "PENDING_PUBLICATION_LIMIT", 2)
store_path = tmp_path / "observations.db"
config = {"device": "/dev/nvme0", "store_path": str(store_path)}
old_time = datetime(2026, 8, 1, tzinfo=timezone.utc)
controller = sysfs_controller
assert _collect(config, controller, old_time, 1000)["ok"] is True
original_derive = collector.derive_hours_from_interval
def unavailable_derivation(*_args):
raise RuntimeError("derivation unavailable")
monkeypatch.setattr(collector, "derive_hours_from_interval", unavailable_derivation)
first_pending = _collect(
config, controller, old_time + timedelta(minutes=5), 1010,
)
assert first_pending["ok"] is False
second_pending = _collect(
config, controller, old_time + timedelta(minutes=10), 1020,
)
assert second_pending["ok"] is False
with sqlite3.connect(store_path) as conn:
assert conn.execute(
"SELECT COUNT(*) FROM pending_publications"
).fetchone()[0] == 2
called = False
def must_not_acquire():
nonlocal called
called = True
return _smartctl(1030), controller
full = collector.run_collection(
config=config,
clock=FakeClock(old_time + timedelta(days=40)),
acquire=must_not_acquire,
)
assert full["ok"] is False
assert "capacity full (2 observations)" in full["error"]
assert called is False
with sqlite3.connect(store_path) as conn:
assert conn.execute(
"SELECT COUNT(*), MIN(sample_ts) FROM pending_publications"
).fetchone() == (
2,
(old_time + timedelta(minutes=5)).isoformat(),
)
assert conn.execute("SELECT COUNT(*) FROM samples").fetchone()[0] == 1
def fail_second_pending(conn, previous, current):
if current["data_units_written"] == 1020:
raise RuntimeError("second observation still blocked")
return original_derive(conn, previous, current)
monkeypatch.setattr(collector, "derive_hours_from_interval", fail_second_pending)
acquired = False
def acquire_after_partial_recovery():
nonlocal acquired
acquired = True
return _smartctl(1030), controller
partial = collector.run_collection(
config=config,
clock=FakeClock(old_time + timedelta(days=40, minutes=5)),
acquire=acquire_after_partial_recovery,
)
assert partial["ok"] is False
assert acquired is True, partial
assert "no new observation acquired" not in partial["error"]
with sqlite3.connect(store_path) as conn:
assert conn.execute(
"SELECT data_units_written FROM samples ORDER BY id"
).fetchall() == [(1000,), (1010,)]
queued_units = [
json.loads(row[0])["sample"]["data_units_written"]
for row in conn.execute(
"SELECT payload FROM pending_publications ORDER BY id"
)
]
assert queued_units == [1020, 1030]
assert conn.execute("SELECT COUNT(*) FROM monitoring_periods").fetchone()[0] == 1
recovered_in_order = []
def record_recovery_order(conn, previous, current):
recovered_in_order.append(current["data_units_written"])
return original_derive(conn, previous, current)
monkeypatch.setattr(collector, "derive_hours_from_interval", record_recovery_order)
recovered = collector.run_collection(
config=config,
clock=FakeClock(old_time + timedelta(days=40, minutes=10)),
acquire=lambda: (_smartctl(1040), controller),
)
assert recovered["ok"] is True, recovered
assert recovered_in_order == [1020, 1030, 1040]
with sqlite3.connect(store_path) as conn:
assert conn.execute(
"SELECT data_units_written FROM samples ORDER BY id DESC LIMIT 1"
).fetchone() == (1040,)
assert conn.execute("SELECT COUNT(*) FROM pending_publications").fetchone()[0] == 0
def test_actual_pending_limit_fails_before_acquisition_and_keeps_old_evidence(
tmp_path, sysfs_controller, monkeypatch,
):
"""The production limit retains every old row and refuses device acquisition."""
monkeypatch.setenv("TZ", "UTC")
store_path = tmp_path / "full-observations.db"
config = {"device": "/dev/nvme0", "store_path": str(store_path)}
controller = sysfs_controller
old_time = datetime(2026, 7, 1, tzinfo=timezone.utc)
assert _collect(config, controller, old_time, 1000)["ok"] is True
def unavailable_derivation(*_args):
raise RuntimeError("derivation unavailable")
monkeypatch.setattr(collector, "derive_hours_from_interval", unavailable_derivation)
failed = _collect(
config, controller, old_time + timedelta(minutes=5), 1010,
)
assert failed["ok"] is False
with sqlite3.connect(store_path) as conn:
payload_row = conn.execute(
"SELECT payload FROM pending_publications ORDER BY id LIMIT 1"
).fetchone()
assert payload_row is not None
base = json.loads(payload_row[0])
base_observed_at = datetime.fromisoformat(base["sample"]["ts"])
base_units = base["sample"]["data_units_written"]
rows = []
for offset in range(1, collector.PENDING_PUBLICATION_LIMIT):
observation = json.loads(payload_row[0])
units_written = base_units + offset
observed_at = base_observed_at + timedelta(minutes=5 * offset)
observation["sample"]["ts"] = observed_at.isoformat()
observation["sample"]["data_units_written"] = units_written
observation["sample"]["bytes_written"] = units_written * 512_000
rows.append((observed_at.isoformat(), json.dumps(observation)))
conn.executemany(
"INSERT INTO pending_publications (sample_ts, payload) VALUES (?, ?)",
rows,
)
conn.commit()
called = False
def must_not_acquire():
nonlocal called
called = True
return _smartctl(9000), controller
result = collector.run_collection(
config=config,
clock=FakeClock(old_time + timedelta(days=40)),
acquire=must_not_acquire,
)
assert result["ok"] is False
assert f"capacity full ({collector.PENDING_PUBLICATION_LIMIT} observations)" in result[
"error"
]
assert called is False
with sqlite3.connect(store_path) as conn:
pending_rows = conn.execute(
"SELECT sample_ts, payload FROM pending_publications ORDER BY id"
).fetchall()
assert len(pending_rows) == collector.PENDING_PUBLICATION_LIMIT
assert (pending_rows[0][0], pending_rows[-1][0]) == (
(old_time + timedelta(minutes=5)).isoformat(),
(
old_time
+ timedelta(minutes=5 * collector.PENDING_PUBLICATION_LIMIT)
).isoformat(),
)
assert [
json.loads(payload)["sample"]["data_units_written"]
for _, payload in pending_rows
] == list(range(1010, 1010 + collector.PENDING_PUBLICATION_LIMIT))
assert conn.execute("SELECT COUNT(*) FROM samples").fetchone()[0] == 1
assert conn.execute("SELECT COUNT(*) FROM monitoring_periods").fetchone()[0] == 1
+4
View File
@@ -211,6 +211,8 @@ class TestStalenessFact:
_insert_baseline(conn)
base = datetime(2026, 9, 15, 12, 0, 0, tzinfo=timezone.utc)
ensure_period_open(conn, base)
conn.commit()
for i in range(14):
day = (base + timedelta(days=i)).strftime("%Y-%m-%d")
_insert_day(conn, day, bw=10*1024*1024*1024)
@@ -234,6 +236,8 @@ class TestStalenessFact:
_insert_baseline(conn)
base = datetime(2026, 9, 18, 12, 0, 0, tzinfo=timezone.utc)
ensure_period_open(conn, base)
conn.commit()
for i in range(14):
day = (base + timedelta(days=i)).strftime("%Y-%m-%d")
_insert_day(conn, day, bw=10*1024*1024*1024)
+306 -65
View File
@@ -24,13 +24,12 @@ from fenris.store import init_store
from fenris.monitoring_periods import ensure_period_open
from fenris.local_day import (
LocalDaySummary,
local_day_boundaries,
persist_local_day,
query_local_day_summary,
)
from fenris.tz_util import detect_system_tz
from fenris.tui import (
FenrisTuiApp,
HistoryGraph,
_RANGE_OPTIONS,
_RANGE_DEFAULT,
)
@@ -70,13 +69,18 @@ def _open_period(conn, start="2026-09-01T00:00:00+00:00"):
def _insert_local_day(conn, local_date, tz_name="UTC", tz_offset="+00:00",
bw=300, br=130, complete=True):
bw=300, br=130, complete=True,
utc_start=None, utc_end=None):
if utc_start is None:
utc_start = f"{local_date}T00:00:00+00:00"
if utc_end is None:
utc_end = f"{(datetime.fromisoformat(local_date) + timedelta(days=1)).strftime('%Y-%m-%d')}T00:00:00+00:00"
summary = LocalDaySummary(
local_date=local_date,
tz_name=tz_name,
tz_offset=tz_offset,
utc_start=f"{local_date}T00:00:00+00:00",
utc_end=f"{(datetime.fromisoformat(local_date) + timedelta(days=1)).strftime('%Y-%m-%d')}T00:00:00+00:00",
utc_start=utc_start,
utc_end=utc_end,
bytes_written=bw,
bytes_read=br,
coverage=0.95,
@@ -124,6 +128,10 @@ def _make_app(tmp_path, clock=None):
return app, patcher
def _visible_selected_date(app):
return str(app.query_one("#local-day").render())[:10]
# ---------------------------------------------------------------------------
# AC92-1: Keyboard bindings exist
# ---------------------------------------------------------------------------
@@ -178,14 +186,13 @@ class TestBracketNavigation:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
# Move to an older day first so ] can advance
for _ in range(3):
await pilot.press("left_square_bracket")
day_before = graph._day_data[graph.selected_index].get("day")
day_before = _visible_selected_date(app)
await pilot.press("right_square_bracket")
day_after = graph._day_data[graph.selected_index].get("day")
day_after = _visible_selected_date(app)
# ] should move to a newer day (closer to today)
assert day_after > day_before
@@ -198,11 +205,10 @@ class TestBracketNavigation:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
day_before = graph._day_data[graph.selected_index].get("day")
day_before = _visible_selected_date(app)
await pilot.press("left_square_bracket")
day_after = graph._day_data[graph.selected_index].get("day")
day_after = _visible_selected_date(app)
# [ should move to an older day (further from today)
assert day_after < day_before
@@ -224,9 +230,8 @@ class TestBracketNavigation:
for _ in range(29):
await pilot.press("left_square_bracket")
# Graph should now show a range that includes the oldest day
assert graph.selected_index >= 0
selected_day = graph._day_data[graph.selected_index].get("day")
# Visible readout identifies oldest selected date.
selected_day = _visible_selected_date(app)
oldest_day = (app_clock - timedelta(days=29)).strftime("%Y-%m-%d")
assert selected_day == oldest_day
finally:
@@ -241,11 +246,10 @@ class TestBracketNavigation:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
initial_idx = graph.selected_index
initial_date = _visible_selected_date(app)
await pilot.press("right_square_bracket")
assert graph.selected_index == initial_idx
assert _visible_selected_date(app) == initial_date
# ---------------------------------------------------------------------------
@@ -262,36 +266,54 @@ class TestArrowInspection:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
await pilot.press("v", "v")
graph.focus()
app.query_one("#usage-history").focus()
await pilot.pause()
day_before = graph._day_data[graph.selected_index].get("day")
day_before = _visible_selected_date(app)
await pilot.press("left")
day_after = graph._day_data[graph.selected_index].get("day")
day_after = _visible_selected_date(app)
assert day_after != day_before
@pytest.mark.asyncio
async def test_arrow_keys_inspect_hours_in_drill(self, tmp_path):
"""Arrow keys navigate hours when in hourly drill-down."""
conn, clock = _setup_store(tmp_path, n_days=14)
local_date = datetime.now().astimezone().date().isoformat()
timezone_name = detect_system_tz()
utc_start, utc_end, tz_offset = local_day_boundaries(local_date, timezone_name)
_insert_local_day(
conn, local_date, tz_name=timezone_name, tz_offset=tz_offset,
utc_start=utc_start.isoformat(), utc_end=utc_end.isoformat(),
)
hour = utc_start.replace(minute=0, second=0, microsecond=0)
if hour < utc_start:
hour += timedelta(hours=1)
while hour + timedelta(hours=1) <= utc_end:
conn.execute(
"INSERT INTO hour_observations "
"(hour, bytes_written_delta, bytes_read_delta, coverage, sample_count) "
"VALUES (?, 1000000, 2000000, 1, 1)",
(hour.isoformat(),),
)
hour += timedelta(hours=1)
conn.commit()
conn.close()
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
await pilot.press("v", "v")
await pilot.click("#usage-history")
# Enter drill-down
await pilot.press("enter")
assert graph.view_mode == "hourly"
assert app.query_one("#activity-tabs").active == "view-day"
# Arrow keys navigate hours
readout_before = str(app.query_one("#bar-readout").render())
await pilot.press("left")
assert graph._hourly_selected >= 0
assert str(app.query_one("#bar-readout").render()) != readout_before
# ---------------------------------------------------------------------------
@@ -308,18 +330,17 @@ class TestTodayBinding:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
# Browse to an older day using [
for _ in range(5):
await pilot.press("left_square_bracket")
browsed_day = graph._day_data[graph.selected_index].get("day")
today = clock.strftime("%Y-%m-%d")
browsed_day = _visible_selected_date(app)
today = datetime.now().astimezone().date().isoformat()
assert browsed_day != today
# Press t to return to today
await pilot.press("t")
assert graph.selected_index == len(graph._day_data) - 1
assert app.query_one("#activity-tabs").active == "view-live"
@pytest.mark.asyncio
async def test_t_exits_hourly_drill(self, tmp_path):
@@ -330,17 +351,16 @@ class TestTodayBinding:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
await pilot.press("v", "v")
await pilot.click("#usage-history")
# Enter drill
await pilot.press("enter")
assert graph.view_mode == "hourly"
assert app.query_one("#activity-tabs").active == "view-day"
# t returns to daily
await pilot.press("t")
assert graph.view_mode == "daily"
assert app.query_one("#activity-tabs").active == "view-live"
# ---------------------------------------------------------------------------
@@ -360,7 +380,7 @@ class TestDateEntry:
await pilot.press("g")
await pilot.pause()
# Date picker should be pushed as a screen
assert len(app.screen_stack) > 1
assert app.screen.query_one("#date-input") is not None
@pytest.mark.asyncio
async def test_valid_date_navigates(self, tmp_path):
@@ -375,13 +395,12 @@ class TestDateEntry:
await pilot.pause()
target_date = (app_clock - timedelta(days=5)).strftime("%Y-%m-%d")
# Simulate the date picker callback directly
app._go_to_date_callback(target_date)
await pilot.press("g")
await pilot.press(*target_date)
await pilot.press("enter")
await pilot.pause()
graph = app.query_one("#usage-history")
selected_day = graph._day_data[graph.selected_index].get("day")
assert selected_day == target_date
assert _visible_selected_date(app) == target_date
finally:
patcher.stop()
@@ -400,7 +419,8 @@ class TestDateEntry:
await pilot.press("enter")
await pilot.pause()
# Should remain on date picker (error shown, not dismissed)
assert len(app.screen_stack) > 1
error = str(app.screen.query_one("#date-error").render())
assert "Invalid date format" in error
@pytest.mark.asyncio
async def test_cancel_returns_to_prior(self, tmp_path):
@@ -411,8 +431,7 @@ class TestDateEntry:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
prior_idx = graph.selected_index
prior_readout = str(app.query_one("#bar-readout").render())
await pilot.press("g")
await pilot.pause()
@@ -420,7 +439,7 @@ class TestDateEntry:
await pilot.pause()
# Should be back to the graph with the same selection
assert graph.selected_index == prior_idx
assert str(app.query_one("#bar-readout").render()) == prior_readout
# ---------------------------------------------------------------------------
@@ -442,8 +461,6 @@ class TestLocalDayEvidence:
try:
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
# Navigate to the target day
for _ in range(4):
await pilot.press("left_square_bracket")
@@ -489,12 +506,10 @@ class TestBrowsingStability:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=0.1)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
# Browse to a specific day
for _ in range(3):
await pilot.press("right_square_bracket")
browsed_day = graph._day_data[graph.selected_index].get("day")
await pilot.press("left_square_bracket")
browsed_day = _visible_selected_date(app)
# Wait for at least one refresh tick
import asyncio
@@ -502,8 +517,7 @@ class TestBrowsingStability:
await pilot.pause()
# Selection should be preserved
assert graph.selected_index >= 0
assert graph._day_data[graph.selected_index].get("day") == browsed_day
assert _visible_selected_date(app) == browsed_day
@pytest.mark.asyncio
async def test_navigate_away_does_not_jump_on_refresh(self, tmp_path):
@@ -514,14 +528,12 @@ class TestBrowsingStability:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=0.1)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
# Browse to older day using [
for _ in range(5):
await pilot.press("left_square_bracket")
browsed_day = graph._day_data[graph.selected_index].get("day")
today = clock.strftime("%Y-%m-%d")
browsed_day = _visible_selected_date(app)
today = datetime.now().astimezone().date().isoformat()
assert browsed_day != today
# Wait for refresh
@@ -530,8 +542,8 @@ class TestBrowsingStability:
await pilot.pause()
# Should NOT jump back to today
selected_day = graph._day_data[graph.selected_index].get("day")
assert selected_day != today
assert _visible_selected_date(app) == browsed_day
assert _visible_selected_date(app) != today
# ---------------------------------------------------------------------------
@@ -548,17 +560,16 @@ class TestDrillDownData:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
await pilot.press("v", "v")
await pilot.click("#usage-history")
# Enter drill-down
await pilot.press("enter")
assert graph.view_mode == "hourly"
assert app.query_one("#activity-tabs").active == "view-day"
# After drill callback, data should be populated
# (empty since we have no hour_observations)
assert graph.view_mode == "hourly"
assert "hourly evidence unavailable" in str(
app.query_one("#bar-readout").render()
)
# ---------------------------------------------------------------------------
@@ -575,11 +586,10 @@ class TestConstrainedWidth:
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(80, 24)) as pilot:
await pilot.pause()
graph = app.query_one("#usage-history")
day_before = graph._day_data[graph.selected_index].get("day")
day_before = _visible_selected_date(app)
await pilot.press("left_square_bracket")
day_after = graph._day_data[graph.selected_index].get("day")
day_after = _visible_selected_date(app)
assert day_after < day_before
@pytest.mark.asyncio
@@ -593,4 +603,235 @@ class TestConstrainedWidth:
await pilot.pause()
await pilot.press("g")
await pilot.pause()
assert len(app.screen_stack) > 1
assert app.screen.query_one("#date-input") is not None
class TestClickableNavigation:
@pytest.mark.asyncio
@pytest.mark.parametrize("size", [(80, 24), (70, 20)])
async def test_click_and_keyboard_select_same_local_date(self, tmp_path, size):
"""Clickable and keyboard navigation share date-entry transitions."""
app_clock = _clock()
conn, _ = _setup_store(tmp_path, n_days=14, today=app_clock)
conn.close()
app, patcher = _make_app(tmp_path, clock=app_clock)
expected = (app_clock - timedelta(days=1)).strftime("%Y-%m-%d")
direct = (app_clock - timedelta(days=5)).strftime("%Y-%m-%d")
try:
async with app.run_test(size=size) as pilot:
await pilot.press("left_square_bracket")
keyboard_date = _visible_selected_date(app)
assert keyboard_date == expected
await pilot.press("t")
await pilot.click("#activity-tools", offset=(2, 0))
assert _visible_selected_date(app) == keyboard_date
await pilot.press("g")
await pilot.press(*direct)
await pilot.press("enter")
await pilot.pause()
assert _visible_selected_date(app) == direct
finally:
patcher.stop()
class TestHistoricalSelectionIdentity:
@pytest.mark.asyncio
async def test_unavailable_entered_date_stays_visible_after_refresh(self, tmp_path):
"""An unavailable requested date remains selected and explained."""
conn, clock = _setup_store(tmp_path, n_days=30)
conn.close()
app, patcher = _make_app(tmp_path, clock=clock)
target = (clock - timedelta(days=120)).strftime("%Y-%m-%d")
try:
async with app.run_test(size=(100, 40)) as pilot:
await pilot.press("g")
await pilot.press(*target)
await pilot.press("enter")
await pilot.pause()
local_day = str(app.query_one("#local-day").render())
assert target in local_day
assert "unavailable" in local_day
app.on_refresh_tick()
await pilot.pause()
local_day = str(app.query_one("#local-day").render())
assert target in local_day
assert "unavailable" in local_day
finally:
patcher.stop()
@pytest.mark.asyncio
async def test_retained_summary_outside_recent_window_remains_available(self, tmp_path):
"""Direct date entry can inspect durable local summaries beyond 90 days."""
clock = _clock()
conn, _ = _setup_store(tmp_path, n_days=30, today=clock)
target = (clock - timedelta(days=120)).strftime("%Y-%m-%d")
utc_start, utc_end, tz_offset = local_day_boundaries(
target, "Asia/Kolkata",
)
_insert_local_day(
conn,
target,
tz_name="Asia/Kolkata",
tz_offset=tz_offset,
bw=2_000_000_000,
br=1_000_000_000,
utc_start=utc_start.isoformat(),
utc_end=utc_end.isoformat(),
)
conn.close()
app, patcher = _make_app(tmp_path, clock=clock)
try:
async with app.run_test(size=(100, 40)) as pilot:
await pilot.press("g")
await pilot.press(*target)
await pilot.press("enter")
await pilot.pause()
await pilot.press("v")
await pilot.pause()
readout = str(app.query_one("#bar-readout").render())
assert target in readout
assert "Asia/Kolkata +05:30" in readout
assert "W 2.000 GB known · R 1.000 GB known" in readout
finally:
patcher.stop()
@pytest.mark.asyncio
async def test_constrained_history_keeps_known_totals_and_evidence_state(self, tmp_path):
"""Small textual view retains selected day, totals, and shared evidence."""
clock = _clock()
conn, _ = _setup_store(tmp_path, n_days=14, today=clock)
target = (clock - timedelta(days=2)).strftime("%Y-%m-%d")
utc_start, utc_end, tz_offset = local_day_boundaries(
target, "Asia/Kolkata",
)
_insert_local_day(
conn,
target,
tz_name="Asia/Kolkata",
tz_offset=tz_offset,
bw=1_500_000_000,
br=500_000_000,
utc_start=utc_start.isoformat(),
utc_end=utc_end.isoformat(),
)
conn.execute(
"INSERT INTO local_day_unallocated_evidence "
"(start_sample_id, end_sample_id, start_local_date, end_local_date, "
" start_tz_name, end_tz_name, started_at, ended_at, bytes_written, "
" bytes_read, reason) VALUES (1, 2, ?, ?, ?, ?, ?, ?, ?, ?, 'local_midnight')",
(target, "2026-09-29", "Asia/Kolkata", "Asia/Kolkata",
utc_start.isoformat(), utc_start.isoformat(), 200_000_000, 300_000_000),
)
conn.commit()
conn.close()
app, patcher = _make_app(tmp_path, clock=clock)
try:
async with app.run_test(size=(70, 20)) as pilot:
await pilot.press("g")
await pilot.press(*target)
await pilot.press("enter")
await pilot.press("v")
await pilot.pause()
readout = str(app.query_one("#bar-readout").render())
assert target in readout
assert "Asia/Kolkata +05:30" in readout
assert "incomplete" in readout
assert "W 1.500 GB known · R 0.500 GB known" in readout
assert "shared at midnight W 0.200 GB · R 0.300 GB" in readout
finally:
patcher.stop()
@pytest.mark.asyncio
async def test_date_picker_keystrokes_do_not_toggle_measurement(self, tmp_path):
"""Typing in date entry does not trigger dashboard shortcuts."""
conn, _ = _setup_store(tmp_path, n_days=14)
conn.close()
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
await pilot.press("g")
await pilot.press("w")
await pilot.press("escape")
await pilot.pause()
header = str(app.query_one("#live-header").render())
assert "Writes" in header
@pytest.mark.asyncio
async def test_historical_hour_uses_recorded_timezone_and_utc_identity(self, tmp_path):
"""Repeated local clock labels retain distinct UTC-hour identities."""
clock = _clock(2026, 11, 2, 12)
conn, _ = _setup_store(tmp_path, n_days=30, today=clock)
target = "2026-11-01"
_insert_local_day(
conn,
target,
tz_name="America/New_York",
tz_offset="-04:00",
utc_start="2026-11-01T04:00:00+00:00",
utc_end="2026-11-02T05:00:00+00:00",
)
conn.execute(
"INSERT INTO local_day_unallocated_evidence "
"(start_sample_id, end_sample_id, start_local_date, end_local_date, "
" start_tz_name, end_tz_name, started_at, ended_at, bytes_written, "
" bytes_read, reason) VALUES (1, 2, ?, ?, ?, ?, ?, ?, ?, ?, 'local_midnight')",
(target, "2026-11-02", "America/New_York", "America/New_York",
"2026-11-02T03:59:00+00:00", "2026-11-02T04:01:00+00:00",
2_000_000_000, 3_000_000_000),
)
for index in range(25):
hour = datetime(2026, 11, 1, 4, tzinfo=timezone.utc) + timedelta(hours=index)
conn.execute(
"INSERT INTO hour_observations "
"(hour, bytes_written_delta, bytes_read_delta, coverage, sample_count) "
"VALUES (?, ?, ?, 1, 1)",
(hour.isoformat(), (index + 1) * 1_000_000_000,
(index + 1) * 2_000_000_000),
)
conn.execute(
"UPDATE hour_observations SET coverage = 0.5, unknown_seconds = 1800 "
"WHERE hour = ?",
("2026-11-01T04:00:00+00:00",),
)
conn.commit()
conn.close()
app, patcher = _make_app(tmp_path, clock=clock)
try:
async with app.run_test(size=(100, 40)) as pilot:
await pilot.click("#view-history")
await pilot.press("g")
await pilot.press(*target)
await pilot.press("enter")
await pilot.pause()
local_day = str(app.query_one("#local-day").render())
assert "America/New_York -04:00" in local_day
assert "incomplete" in local_day
assert "shared at midnight W 2.000 GB · R 3.000 GB" in local_day
# Move from final local hour to the second 01:00 occurrence.
await pilot.press(*(["left"] * 22))
second_occurrence = str(app.query_one("#bar-readout").render())
assert "01:00 -0500" in second_occurrence
assert "America/New_York -0500" in second_occurrence
assert "06:00 UTC" in second_occurrence
await pilot.press("left")
first_occurrence = str(app.query_one("#bar-readout").render())
assert "01:00 -0400" in first_occurrence
assert "America/New_York -0400" in first_occurrence
assert "05:00 UTC" in first_occurrence
await pilot.press("left")
incomplete_hour = str(app.query_one("#bar-readout").render())
assert "00:00 -0400" in incomplete_hour
assert "50% coverage · incomplete" in incomplete_hour
finally:
patcher.stop()
+4 -4
View File
@@ -151,8 +151,8 @@ class TestDetailExpiresSummariesRemain:
# Run pruning
pruned = prune_old_samples(conn, now, retention_days=14)
# Old samples should be pruned, recent ones retained
assert pruned >= 1
# The handcrafted totals lack replacement provenance, so retain samples.
assert pruned == 0
# Local-day summaries must still be queryable
old_result = query_local_day_summary(conn, "2026-09-16")
@@ -434,10 +434,10 @@ class TestExistingEntryPoints:
_insert_sample(conn, ts, bw=days_ago * 100)
pruned = prune_old_samples(conn, now, retention_days=14)
assert pruned == 15
assert pruned == 0
cursor = conn.execute("SELECT COUNT(*) FROM samples")
assert cursor.fetchone()[0] == 14
assert cursor.fetchone()[0] == 29
conn.close()
def test_repair_with_real_temp_store(self, tmp_path):
+1
View File
@@ -168,6 +168,7 @@ def test_failed_publication_survives_restart_and_readers_keep_last_consistent_vi
called = True
return _smartctl(1030), sysfs_path
monkeypatch.setattr(collector, "PENDING_PUBLICATION_LIMIT", 1)
blocked = collector.run_collection(
config=config,
clock=FakeClock(now + timedelta(minutes=5)),
+108 -7
View File
@@ -6,26 +6,24 @@ Covers:
- Three-minute cadence constants
- TUI integration with live graph
"""
import sys
from datetime import datetime, timedelta, timezone
from pathlib import Path
import pytest
import sys
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
from fenris.store import init_store
from fenris.activity_selection import IntervalIdentity
from fenris.monitoring_periods import ensure_period_open
from fenris.status import ACCURACY_SEC, CADENCE_DEFAULT_S, FRESH_THRESHOLD_S
from fenris.store import init_store
from fenris.tui import (
LIVE_WINDOW_H,
FenrisTuiApp,
LiveActivityGraph,
_query_live_graph_data,
LIVE_WINDOW_H,
_query_daily_graph_data,
_RANGE_OPTIONS,
)
from fenris.status import CADENCE_DEFAULT_S, FRESH_THRESHOLD_S, ACCURACY_SEC
# ---------------------------------------------------------------------------
# Helpers
@@ -63,6 +61,27 @@ def _open_period(conn, start="2026-09-01T00:00:00+00:00"):
ensure_period_open(conn, datetime.fromisoformat(start))
def _live_intervals(count=4):
first = datetime(2026, 9, 28, 12, 0, tzinfo=timezone.utc)
intervals = []
for index in range(count):
start = first + timedelta(minutes=3 * index)
end = start + timedelta(minutes=3)
intervals.append({
"start_ts": start.isoformat(),
"end_ts": end.isoformat(),
"start_label": start.strftime("%H:%M"),
"end_label": end.strftime("%H:%M"),
"bytes_written": (index + 1) * 1000,
"bytes_read": (index + 1) * 500,
"elapsed_s": 180,
"is_gap": False,
"is_zero": False,
"is_segment_boundary": False,
})
return intervals
# ---------------------------------------------------------------------------
# Cadence constants (issue #91 AC1)
# ---------------------------------------------------------------------------
@@ -322,3 +341,85 @@ class TestTUILiveIntegration:
await pilot.pause()
main_grid = app.query_one("#main-grid")
assert main_grid.has_class("constrained")
@pytest.mark.asyncio
@pytest.mark.parametrize("size", [(100, 40), (70, 20)])
async def test_follow_pin_expiry_and_today_use_visible_readout(self, tmp_path, size):
"""Live selection follows, pins, expires, and resets at both layout sizes."""
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
intervals = _live_intervals()
async with app.run_test(size=size) as pilot:
await pilot.pause()
graph = app.query_one("#live-activity")
graph.set_data(intervals[:3])
readout = str(app.query_one("#live-readout").render())
assert "12:09 UTC" in readout
assert graph.selection.following_live
assert graph.selection.selected_live_interval == IntervalIdentity(
intervals[2]["start_ts"], intervals[2]["end_ts"]
)
# Refresh while following selects the newly measured newest point.
graph.set_data(intervals)
assert "12:12 UTC" in str(app.query_one("#live-readout").render())
# Live graph receives focus at launch; first arrow pins immediately.
await pilot.press("left")
pinned = graph.selection.selected_live_interval
readout = str(app.query_one("#live-readout").render())
assert "12:09 UTC" in readout
assert not graph.selection.following_live
graph.set_data(intervals)
readout = str(app.query_one("#live-readout").render())
assert "12:09 UTC" in readout
assert graph.selection.selected_live_interval == pinned
# Read/write changes presentation only; pinned point stays stable.
await pilot.press("w")
graph.set_data(intervals)
assert graph.measure == "read"
assert graph.selection.selected_live_interval == pinned
assert "12:09 UTC" in str(app.query_one("#live-readout").render())
# Remove pinned interval from rolling window. Explain expiration.
graph.set_data(intervals[3:])
readout = str(app.query_one("#live-readout").render())
assert "Inspected interval expired; following live" in readout
assert "12:12 UTC" in readout
assert graph.selection.following_live
# Today action clears expiration and resumes newest-point following.
await pilot.press("t")
assert graph.selection.following_live
assert not graph.selection.live_interval_expired
graph.set_data(intervals)
readout = str(app.query_one("#live-readout").render())
assert "12:12 UTC" in readout
assert "Inspected interval expired" not in readout
assert graph.selection.following_live
@pytest.mark.asyncio
async def test_mouse_inspection_pins_same_interval_identity(self, tmp_path):
"""Mouse inspection pins the interval at its visible graph location."""
app = FenrisTuiApp(store_path=tmp_path / "test.db", refresh_interval_s=999)
intervals = _live_intervals()
async with app.run_test(size=(100, 40)) as pilot:
await pilot.pause()
graph = app.query_one("#live-activity")
graph.set_data(intervals[:3])
await pilot.pause()
render = app.query_one("#live-render")
await pilot.click(
"#live-render",
offset=(8, max(0, render.content_size.height // 2)),
)
await pilot.pause()
expected = IntervalIdentity(
intervals[0]["start_ts"], intervals[0]["end_ts"]
)
assert graph.selection.selected_live_interval == expected
assert not graph.selection.following_live
assert "12:03 UTC" in str(app.query_one("#live-readout").render())
+17 -15
View File
@@ -27,9 +27,14 @@ def store_conn(tmp_path: Path):
def _insert_sample(conn, ts_iso, device="/dev/nvme0"):
conn.execute(
"INSERT OR IGNORE INTO controller_segments "
"(id, opened_at, identity_key) VALUES (1, '2026-01-01T00:00:00+00:00', 'test')"
)
conn.execute(
"INSERT INTO samples (ts, device, data_units_written, data_units_read, "
" bytes_written, bytes_read, percentage_used) VALUES (?, ?, 0, 0, 0, 0, 0)",
" bytes_written, bytes_read, percentage_used, segment_id, local_tz) "
"VALUES (?, ?, 0, 0, 0, 0, 0, 1, 'UTC')",
(ts_iso, device),
)
conn.commit()
@@ -50,33 +55,30 @@ class TestPruneOldSamples:
cursor = store_conn.execute("SELECT COUNT(*) FROM samples")
assert cursor.fetchone()[0] == 1
def test_removes_old_samples(self, store_conn):
def test_keeps_old_samples_without_replacement_evidence(self, store_conn):
now = datetime(2026, 9, 15, 12, 0, 0, tzinfo=timezone.utc)
# Insert samples at 10, 14, and 15 days ago
# All three are before the cutoff (2026-09-01T12:00:00)
# The 15-day-old sample is not a boundary anchor because
# the next sample (14 days ago) is also before the cutoff
# Old samples stay until UTC and local-day evidence replace them.
for days_ago in [10, 14, 15]:
ts = (now - timedelta(days=days_ago)).isoformat()
_insert_sample(store_conn, ts)
pruned = prune_old_samples(store_conn, now, retention_days=14)
assert pruned == 1 # Only the 15-day-old sample removed
assert pruned == 0
cursor = store_conn.execute("SELECT COUNT(*) FROM samples")
assert cursor.fetchone()[0] == 2
assert cursor.fetchone()[0] == 3
def test_removes_many_old_samples(self, store_conn):
def test_keeps_many_old_samples_without_replacement_evidence(self, store_conn):
now = datetime(2026, 9, 15, 12, 0, 0, tzinfo=timezone.utc)
for days_ago in range(1, 30):
ts = (now - timedelta(days=days_ago)).isoformat()
_insert_sample(store_conn, ts)
pruned = prune_old_samples(store_conn, now, retention_days=14)
assert pruned == 15 # Days 15-29 removed
assert pruned == 0
cursor = store_conn.execute("SELECT COUNT(*) FROM samples")
assert cursor.fetchone()[0] == 14 # Days 1-14 kept
assert cursor.fetchone()[0] == 29
def test_empty_store_no_error(self, store_conn):
now = datetime(2026, 9, 15, 12, 0, 0, tzinfo=timezone.utc)
@@ -130,8 +132,8 @@ class TestPruneOldSamples:
)
assert cursor.fetchone()[0] == 1
def test_old_sample_with_derived_interval_removed(self, store_conn):
"""Old samples with fully derived intervals are removed."""
def test_hour_rows_alone_do_not_replace_local_day_evidence(self, store_conn):
"""An hour row alone cannot authorize source-sample deletion."""
now = datetime(2026, 9, 30, 12, 0, 0, tzinfo=timezone.utc)
# Old sample with derived interval
@@ -148,8 +150,8 @@ class TestPruneOldSamples:
# Run pruning
pruned = prune_old_samples(store_conn, now, retention_days=14)
# Old sample should be removed (interval is derived)
# Keep the source samples until local-day replacement evidence exists.
cursor = store_conn.execute(
"SELECT COUNT(*) FROM samples WHERE ts = '2026-09-10T10:00:00+00:00'"
)
assert cursor.fetchone()[0] == 0
assert cursor.fetchone()[0] == 1
+7 -8
View File
@@ -235,8 +235,8 @@ class TestBoundaryAnchorRetention:
# the interval spans the retention boundary
assert needs_boundary_anchor(store_conn, "2026-09-15T23:55:00+00:00", now)
def test_not_boundary_anchor_if_fully_derived(self, store_conn):
"""Sample that's fully derived is not a boundary anchor."""
def test_hour_only_derivation_does_not_replace_local_day_evidence(self, store_conn):
"""UTC hour evidence alone does not make a sample safe to prune."""
now = datetime(2026, 9, 30, 12, 0, 0, tzinfo=timezone.utc)
# Insert sample and fully derive its interval
@@ -247,8 +247,7 @@ class TestBoundaryAnchorRetention:
_insert_hour(store_conn, "2026-09-14T10:00:00+00:00",
bytes_written_delta=1000000)
# Not a boundary anchor
assert not needs_boundary_anchor(store_conn, "2026-09-14T10:00:00+00:00", now)
assert needs_boundary_anchor(store_conn, "2026-09-14T10:00:00+00:00", now)
def test_pruning_retains_boundary_anchors(self, store_conn):
"""Pruning keeps samples needed as boundary anchors."""
@@ -274,8 +273,8 @@ class TestBoundaryAnchorRetention:
)
assert cursor.fetchone()[0] == 1
def test_pruning_removes_old_sample_with_derived_interval(self, store_conn):
"""Pruning removes old samples when interval is fully derived."""
def test_pruning_keeps_samples_without_local_day_replacement(self, store_conn):
"""Pruning keeps samples when only the UTC-hour row exists."""
now = datetime(2026, 9, 30, 12, 0, 0, tzinfo=timezone.utc)
# Old sample with derived interval
@@ -289,11 +288,11 @@ class TestBoundaryAnchorRetention:
# Run pruning
pruned = prune_old_samples(store_conn, now, retention_days=14)
# Old sample should be removed (interval is derived)
# The source remains until local-day evidence is also durable.
cursor = store_conn.execute(
"SELECT COUNT(*) FROM samples WHERE ts = '2026-09-10T10:00:00+00:00'"
)
assert cursor.fetchone()[0] == 0
assert cursor.fetchone()[0] == 1
# ---------------------------------------------------------------------------
+2 -2
View File
@@ -259,8 +259,8 @@ class TestKeyCeremonyDoc:
def test_documents_private_key_storage(self):
content = self._doc_content()
assert "password manager" in content.lower(), \
"Must document that private key lives in password manager"
assert "gitea repository actions secret `gpg_private_key`" in content.lower(), \
"Must document that Gitea Actions stores the private signing key"
# ---------------------------------------------------------------------------
+30 -21
View File
@@ -60,6 +60,13 @@ def _clock(year=2026, month=9, day=30, hour=12):
return datetime(year, month, day, hour, 0, 0, tzinfo=timezone.utc)
class _FrozenTuiDateTime(datetime):
@classmethod
def now(cls, tz=None):
instant = datetime(2026, 9, 28, 12, 0, 0, tzinfo=timezone.utc)
return instant.replace(tzinfo=None) if tz is None else instant.astimezone(tz)
def _insert_baseline(conn, tbw_tb=1.0, verified=True,
model="Samsung SSD 970 EVO Plus 1TB"):
conn.execute(
@@ -927,6 +934,7 @@ class TestBarGraphTUI:
for i in range(14):
d = (datetime(2026, 9, 15) + timedelta(days=i)).strftime("%Y-%m-%d")
_insert_day(conn, d, bw=1024*1024*100)
_insert_local_day(conn, d, bw=1024*1024*100)
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
conn.close()
@@ -941,11 +949,9 @@ class TestBarGraphTUI:
assert "Unalloc" in legend
assert "Zero" in legend
range_label = str(app.query_one("#bar-range").render())
assert "14 days" in range_label
assert "UTC" in range_label
assert "14 local days" in range_label
plotted = str(app.query_one("#bar-render").render())
assert "Writes (" in legend
assert "UTC" in range_label
assert "─" in plotted
@pytest.mark.asyncio
@@ -957,6 +963,7 @@ class TestBarGraphTUI:
for i in range(14):
d = (datetime(2026, 9, 15) + timedelta(days=i)).strftime("%Y-%m-%d")
_insert_day(conn, d, bw=1024*1024*100)
_insert_local_day(conn, d, bw=1024*1024*100)
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
conn.close()
@@ -1022,7 +1029,7 @@ class TestBarGraphTUI:
assert len(graph._day_data) == 90
@pytest.mark.asyncio
async def test_readout_updates_on_selection(self, tmp_path):
async def test_readout_updates_on_selection(self, tmp_path, monkeypatch):
"""Readout shows selected day info."""
conn = init_store(tmp_path / "test.db")
_insert_segment(conn)
@@ -1030,9 +1037,11 @@ class TestBarGraphTUI:
for i in range(14):
d = (datetime(2026, 9, 15) + timedelta(days=i)).strftime("%Y-%m-%d")
_insert_day(conn, d, bw=1024*1024*100)
_insert_local_day(conn, d, bw=1024*1024*100)
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
conn.close()
monkeypatch.setattr("fenris.tui.datetime", _FrozenTuiDateTime)
app = FenrisTuiApp(store_path=tmp_path / "test.db")
async with app.run_test(size=(80, 24)) as pilot:
await pilot.press("v", "v")
@@ -1043,18 +1052,19 @@ class TestBarGraphTUI:
# Newest day is selected initially.
readout = str(app.query_one("#bar-readout").render())
assert "UTC" in readout
assert "GB" in readout
assert "totals so far" in readout
assert "W 0.105 GB known" in readout
# Moving left updates the selected-day readout.
await pilot.press("left")
await pilot.pause()
readout = str(app.query_one("#bar-readout").render())
assert "2026-09" in readout
assert "GB" in readout
assert "incomplete" in readout
assert "W 0.105 GB known" in readout
@pytest.mark.asyncio
async def test_hourly_drill_down_and_back(self, tmp_path):
async def test_hourly_drill_down_and_back(self, tmp_path, monkeypatch):
"""Enter drills into hourly view, Esc returns to daily."""
conn = init_store(tmp_path / "test.db")
_insert_segment(conn)
@@ -1062,6 +1072,7 @@ class TestBarGraphTUI:
for i in range(14):
d = (datetime(2026, 9, 15) + timedelta(days=i)).strftime("%Y-%m-%d")
_insert_day(conn, d, bw=1024*1024*100)
_insert_local_day(conn, d, bw=1024*1024*100)
# Insert hours for each day
for h in range(24):
hour = "%sT%02d:00:00+00:00" % (d, h)
@@ -1070,6 +1081,7 @@ class TestBarGraphTUI:
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
conn.close()
monkeypatch.setattr("fenris.tui.datetime", _FrozenTuiDateTime)
app = FenrisTuiApp(store_path=tmp_path / "test.db")
async with app.run_test(size=(80, 24)) as pilot:
await pilot.press("v", "v")
@@ -1078,30 +1090,27 @@ class TestBarGraphTUI:
await pilot.pause()
await pilot.pause()
# The newest day is selected automatically.
assert graph.selected_index == len(graph._day_data) - 1
assert graph.view_mode == "daily"
assert app.query_one("#activity-tabs").active == "view-history"
daily_readout = str(app.query_one("#bar-readout").render())
# Enter drill-down
await pilot.press("enter")
await pilot.pause()
assert graph.view_mode == "hourly"
assert graph.drill_day is not None
assert len(graph._hour_data) == 24
assert app.query_one("#activity-tabs").active == "view-day"
assert "UTC" in str(app.query_one("#bar-readout").render())
# Five-minute data refreshes retain the selected hour.
graph._hourly_selected = 12
selected_hour = graph._hour_data[12]["hour"]
# Inspection identity survives refresh through visible readout.
await pilot.press("left")
selected_hour = str(app.query_one("#bar-readout").render())
app._refresh()
await pilot.pause()
assert graph.view_mode == "hourly"
assert graph._hour_data[graph._hourly_selected]["hour"] == selected_hour
assert str(app.query_one("#bar-readout").render()) == selected_hour
# Esc returns to daily
await pilot.press("escape")
await pilot.pause()
assert graph.view_mode == "daily"
assert graph.drill_day is None
assert app.query_one("#activity-tabs").active == "view-history"
assert str(app.query_one("#bar-readout").render()) == daily_readout
@pytest.mark.asyncio
async def test_empty_store_graph(self, tmp_path):