Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
334a3f5f2c |
@@ -1,230 +0,0 @@
|
||||
# Fenris release workflow — release path on Coolify-hosted Gitea runner.
|
||||
# The runner is repository-scoped and executes package build, signing, validation,
|
||||
# registry publication, and release attachment. Spec: §5, issue #52
|
||||
name: Release
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- 'v*'
|
||||
workflow_dispatch:
|
||||
|
||||
# Built-in Gitea token needs write access for release assets and package registry.
|
||||
permissions:
|
||||
contents: read
|
||||
releases: write
|
||||
packages: write
|
||||
|
||||
jobs:
|
||||
release:
|
||||
runs-on: [self-hosted]
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Validate release tag and notes
|
||||
run: |
|
||||
set -euo pipefail
|
||||
VERSION="$(sed -n 's/^version = "\(.*\)"/\1/p' pyproject.toml)"
|
||||
if [ -z "${VERSION}" ]; then
|
||||
echo "::error::could not determine the project version"
|
||||
exit 1
|
||||
fi
|
||||
if [ "${GITHUB_EVENT_NAME}" != "workflow_dispatch" ]; then
|
||||
EXPECTED_TAG="v${VERSION}"
|
||||
ACTUAL_TAG="${GITHUB_REF#refs/tags/}"
|
||||
if [ "${ACTUAL_TAG}" != "${EXPECTED_TAG}" ]; then
|
||||
echo "::error::tag ${ACTUAL_TAG} does not match ${EXPECTED_TAG}"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
python3 scripts/extract_changelog.py CHANGELOG.md "${VERSION}" \
|
||||
--footer packaging/release-footer.md > "${RUNNER_TEMP}/release-body.md"
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Install build dependencies
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y gnupg2 rpm python3-venv
|
||||
python3 -m venv /tmp/fenris-ci
|
||||
/tmp/fenris-ci/bin/pip install --quiet build
|
||||
echo "/tmp/fenris-ci/bin" >> "$GITHUB_PATH"
|
||||
NFPM_VERSION=2.47.0
|
||||
curl --fail --silent --show-error --location \
|
||||
"https://github.com/goreleaser/nfpm/releases/download/v${NFPM_VERSION}/nfpm_${NFPM_VERSION}_Linux_x86_64.tar.gz" \
|
||||
-o /tmp/nfpm.tar.gz
|
||||
sudo tar -xzf /tmp/nfpm.tar.gz -C /usr/local/bin nfpm
|
||||
nfpm --version
|
||||
|
||||
- name: Build packages
|
||||
run: make package
|
||||
|
||||
- name: Import packaging key
|
||||
env:
|
||||
GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GPG_PRIVATE_KEY}" ]; then
|
||||
echo "::error::GPG_PRIVATE_KEY repository secret is not configured"
|
||||
exit 1
|
||||
fi
|
||||
printf '%s\n' "${GPG_PRIVATE_KEY}" | gpg --batch --import
|
||||
SECRET_FINGERPRINT="$(gpg --batch --list-secret-keys --with-colons 'packaging@bongbetic.com' | awk -F: '$1 == "fpr" { print $10; exit }')"
|
||||
PUBLIC_FINGERPRINT="$(gpg --batch --show-keys --with-colons packaging/keys/fenris-packaging.asc | awk -F: '$1 == "fpr" { print $10; exit }')"
|
||||
if [ -z "${PUBLIC_FINGERPRINT}" ]; then
|
||||
echo "::error::packaging/keys/fenris-packaging.asc has no OpenPGP key"
|
||||
exit 1
|
||||
fi
|
||||
if [ "${SECRET_FINGERPRINT}" != "${PUBLIC_FINGERPRINT}" ]; then
|
||||
echo "::error::packaging public key does not match imported private key"
|
||||
exit 1
|
||||
fi
|
||||
echo "Packaging key fingerprint verified: ${PUBLIC_FINGERPRINT}"
|
||||
|
||||
- name: Sign RPM payload
|
||||
run: make sign-rpm
|
||||
|
||||
- name: Generate and clearsign SHA256SUMS
|
||||
run: |
|
||||
set -euo pipefail
|
||||
VERSION="$(sed -n 's/^version = "\(.*\)"/\1/p' pyproject.toml)"
|
||||
cd dist
|
||||
sha256sum "fenris_${VERSION}_amd64.deb" \
|
||||
"fenris-${VERSION}-1.x86_64.rpm" > SHA256SUMS
|
||||
gpg --batch --yes --clearsign --local-user packaging@bongbetic.com SHA256SUMS
|
||||
|
||||
- name: Validate signatures and checksums
|
||||
run: |
|
||||
set -euo pipefail
|
||||
VERSION="$(sed -n 's/^version = "\(.*\)"/\1/p' pyproject.toml)"
|
||||
RPM="fenris-${VERSION}-1.x86_64.rpm"
|
||||
RPM_VERIFY="$(rpm -Kv "dist/${RPM}" 2>&1)"
|
||||
printf '%s\n' "${RPM_VERIFY}"
|
||||
printf '%s\n' "${RPM_VERIFY}" | grep -Eiq 'signature.*: *ok'
|
||||
gpg --batch --verify dist/SHA256SUMS.asc
|
||||
(cd dist && sha256sum -c SHA256SUMS)
|
||||
|
||||
- name: Remove packaging key material
|
||||
if: always()
|
||||
run: |
|
||||
set +e
|
||||
FINGERPRINT="$(gpg --batch --list-secret-keys --with-colons 'packaging@bongbetic.com' 2>/dev/null | awk -F: '$1 == "fpr" { print $10; exit }')"
|
||||
if [ -n "${FINGERPRINT}" ]; then
|
||||
gpg --batch --yes --delete-secret-keys "${FINGERPRINT}"
|
||||
gpg --batch --yes --delete-keys "${FINGERPRINT}"
|
||||
fi
|
||||
|
||||
- name: Determine version
|
||||
id: version
|
||||
run: echo "version=$(sed -n 's/^version = "\(.*\)"/\1/p' pyproject.toml)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Upload deb packages to registry
|
||||
env:
|
||||
GITEA_PUBLISH_TOKEN: ${{ secrets.GITEAPACKAGETOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GITEA_PUBLISH_TOKEN}" ]; then
|
||||
echo "::error::GITEAPACKAGETOKEN repository secret is not configured"
|
||||
exit 1
|
||||
fi
|
||||
VERSION=${{ steps.version.outputs.version }}
|
||||
DEB="fenris_${VERSION}_amd64.deb"
|
||||
for CODENAME in bookworm jammy noble; do
|
||||
STATUS=$(curl --silent --show-error --user "xavierk:${GITEA_PUBLISH_TOKEN}" -X PUT \
|
||||
-T "dist/${DEB}" -o /dev/null -w '%{http_code}' \
|
||||
"https://git.bongbetic.com/api/packages/xavierk/debian/pool/${CODENAME}/main/upload" || true)
|
||||
case "${STATUS}" in
|
||||
200|201|204) echo "Debian ${CODENAME}: uploaded" ;;
|
||||
409) echo "Debian ${CODENAME}: already exists, kept existing package" ;;
|
||||
*) echo "::error::Debian ${CODENAME} upload failed with HTTP ${STATUS}"; exit 1 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
- name: Upload RPM to registry
|
||||
env:
|
||||
GITEA_PUBLISH_TOKEN: ${{ secrets.GITEAPACKAGETOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
VERSION=${{ steps.version.outputs.version }}
|
||||
RPM="fenris-${VERSION}-1.x86_64.rpm"
|
||||
STATUS=$(curl --silent --show-error --user "xavierk:${GITEA_PUBLISH_TOKEN}" -X PUT \
|
||||
-T "dist/${RPM}" -o /dev/null -w '%{http_code}' \
|
||||
"https://git.bongbetic.com/api/packages/xavierk/rpm/fenris/upload" || true)
|
||||
case "${STATUS}" in
|
||||
200|201|204) echo "RPM: uploaded" ;;
|
||||
409) echo "RPM: already exists, kept existing package" ;;
|
||||
*) echo "::error::RPM upload failed with HTTP ${STATUS}"; exit 1 ;;
|
||||
esac
|
||||
|
||||
- name: Create Gitea release
|
||||
env:
|
||||
GITEA_PUBLISH_TOKEN: ${{ secrets.GITEAPACKAGETOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GITEA_PUBLISH_TOKEN}" ]; then
|
||||
echo "::error::GITEAPACKAGETOKEN repository secret is not configured"
|
||||
exit 1
|
||||
fi
|
||||
VERSION=${{ steps.version.outputs.version }}
|
||||
RELEASE_BODY="${RUNNER_TEMP}/release-body.md"
|
||||
if [ ! -s "${RELEASE_BODY}" ]; then
|
||||
echo "::error::validated release body is missing or empty"
|
||||
exit 1
|
||||
fi
|
||||
EXISTING_RELEASE="${RUNNER_TEMP}/existing-release.json"
|
||||
EXISTING=$(curl --silent --show-error -o "${EXISTING_RELEASE}" -w '%{http_code}' \
|
||||
-H "Authorization: token ${GITEA_PUBLISH_TOKEN}" \
|
||||
"https://git.bongbetic.com/api/v1/repos/xavierk/Fenris/releases/tags/v${VERSION}" || true)
|
||||
case "${EXISTING}" in
|
||||
200)
|
||||
echo "Release v${VERSION} exists; resynchronizing its notes"
|
||||
REQUEST="$(python3 scripts/release_request.py --version "${VERSION}" \
|
||||
--body-file "${RELEASE_BODY}" --existing-release "${EXISTING_RELEASE}")"
|
||||
;;
|
||||
404)
|
||||
REQUEST="$(python3 scripts/release_request.py --version "${VERSION}" \
|
||||
--body-file "${RELEASE_BODY}")"
|
||||
;;
|
||||
*)
|
||||
echo "::error::release lookup failed with HTTP ${EXISTING}"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
METHOD="$(printf '%s' "${REQUEST}" | python3 -c "import json,sys; print(json.load(sys.stdin)['method'])")"
|
||||
RELEASE_PATH="$(printf '%s' "${REQUEST}" | python3 -c "import json,sys; print(json.load(sys.stdin)['path'])")"
|
||||
PAYLOAD="$(printf '%s' "${REQUEST}" | python3 -c "import json,sys; print(json.dumps(json.load(sys.stdin)['payload']))")"
|
||||
curl --fail --silent --show-error -X "${METHOD}" \
|
||||
-H "Authorization: token ${GITEA_PUBLISH_TOKEN}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "${PAYLOAD}" \
|
||||
"https://git.bongbetic.com/api/v1/repos/xavierk/Fenris${RELEASE_PATH}"
|
||||
|
||||
- name: Attach artifacts to release
|
||||
env:
|
||||
GITEA_PUBLISH_TOKEN: ${{ secrets.GITEAPACKAGETOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
VERSION=${{ steps.version.outputs.version }}
|
||||
# Get release ID for this tag
|
||||
RELEASE_JSON=$(curl --fail --silent --show-error \
|
||||
-H "Authorization: token ${GITEA_PUBLISH_TOKEN}" \
|
||||
"https://git.bongbetic.com/api/v1/repos/xavierk/Fenris/releases/tags/v${VERSION}")
|
||||
RELEASE_ID=$(printf '%s' "${RELEASE_JSON}" \
|
||||
| python3 -c "import sys,json; print(json.load(sys.stdin)['id'])")
|
||||
# Attach deb, rpm, and clearsigned checksums once.
|
||||
for FILE in "dist/fenris_${VERSION}_amd64.deb" \
|
||||
"dist/fenris-${VERSION}-1.x86_64.rpm" \
|
||||
"dist/SHA256SUMS.asc"; do
|
||||
ASSET_NAME="${FILE##*/}"
|
||||
if python3 -c 'import json,sys; name=sys.argv[1]; sys.exit(0 if any(a.get("name") == name for a in json.load(sys.stdin).get("assets", [])) else 1)' "${ASSET_NAME}" <<<"${RELEASE_JSON}"; then
|
||||
echo "${ASSET_NAME}: already attached"
|
||||
else
|
||||
curl --fail --silent --show-error -X POST \
|
||||
-H "Authorization: token ${GITEA_PUBLISH_TOKEN}" \
|
||||
-F "attachment=@${FILE}" \
|
||||
"https://git.bongbetic.com/api/v1/repos/xavierk/Fenris/releases/${RELEASE_ID}/assets"
|
||||
fi
|
||||
done
|
||||
-11
@@ -7,14 +7,3 @@ data/fenris.log
|
||||
data/history.jsonl
|
||||
data/hourly.jsonl
|
||||
plan-dash-changes.md
|
||||
|
||||
# Packaging build artifacts
|
||||
build/
|
||||
dist/
|
||||
|
||||
# Local tooling
|
||||
graphify-out/
|
||||
json
|
||||
src/fenris.egg-info/
|
||||
.pytest_cache/
|
||||
.venv/
|
||||
|
||||
@@ -1,18 +0,0 @@
|
||||
# Changelog
|
||||
|
||||
<!--
|
||||
Maintainers add one user-facing entry to Unreleased with each change. A release
|
||||
commit bumps pyproject.toml, renames Unreleased to that bare-semver version and
|
||||
an ISO date, then restores an empty Unreleased section; tag that commit. Do not
|
||||
backfill releases from before this changelog.
|
||||
-->
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
## [0.3.4] - 2026-09-10
|
||||
|
||||
### Added
|
||||
|
||||
- Add Fenris identity and a polkit authentication notice to the dashboard.
|
||||
- Clarify monitoring continuity, deliberate pauses, and quitting in the dashboard and status output.
|
||||
- Add per-release notes with installation, verification, and rollback guidance.
|
||||
+4
-44
@@ -32,46 +32,22 @@ _Avoid_: Data directory, history.jsonl, the database (generic)
|
||||
The condition where the observation store is present but cannot be read or trusted — unreadable, corrupt, or written by a newer Fenris — degrading every view that depends on it rather than crashing or guessing.
|
||||
_Avoid_: Database error, corruption, broken data
|
||||
|
||||
**Usage interval**:
|
||||
The elapsed span between two compatible counter observations, with a measured usage total whose distribution within that span may be unknown.
|
||||
_Avoid_: Estimated hourly usage, interpolated sample
|
||||
|
||||
**Unallocated usage**:
|
||||
Measured usage whose share in a particular hour, calendar day, or monitoring period cannot be established from the available evidence.
|
||||
_Avoid_: Zero usage, evenly distributed writes
|
||||
|
||||
**Hour observation**:
|
||||
The usage-habit evidence for a UTC hour's represented elapsed span: active, idle, powered-off, and unknown time, measured usage, thermal evidence, and coverage. A partial hour does not describe future time.
|
||||
One row per UTC hour in the observation store, recording that hour's usage-habit split into active, idle, powered-off, and unknown seconds, plus write/read deltas, thermal evidence, and coverage.
|
||||
_Avoid_: Hourly record, hourly.jsonl entry
|
||||
|
||||
**Day aggregate**:
|
||||
The UTC-day summary at which usage-habit evidence is judged; distinct from a local display day.
|
||||
_Avoid_: Local daily total, daily stats
|
||||
|
||||
**Local display day**:
|
||||
A calendar day in the user's current system timezone, used to browse observation history; its elapsed length can vary with timezone transitions.
|
||||
_Avoid_: UTC evidence day, fixed 24-hour day
|
||||
One row per UTC day derived from hour observations; the grain at which usage-habit evidence is judged.
|
||||
_Avoid_: Daily summary, daily stats
|
||||
|
||||
**Controller segment**:
|
||||
A span of observation history within which the drive's controller identity is unchanged and counters are monotonic; write deltas are never computed across a segment boundary.
|
||||
_Avoid_: Counter reset handling, drive swap detection
|
||||
|
||||
**Degraded identity**:
|
||||
The condition where a controller segment's identity key is blank because no identifier rung produced a value; replacement detection then relies on write-counter continuity alone, and projection confidence is capped.
|
||||
_Avoid_: Identity error, unknown device, virtual drive
|
||||
|
||||
**Endurance baseline**:
|
||||
The write-endurance value a projection consumes, chosen by precedence: a verified override when one exists, otherwise an unverified override, otherwise a coarse implied baseline derived from vendor wear — each labeled as such.
|
||||
The write-endurance value a projection consumes: a verified rated-TBW override stored with provenance when one exists, otherwise a coarse implied baseline derived from vendor wear and labeled as such.
|
||||
_Avoid_: TBW value, failure threshold, max writes
|
||||
|
||||
**Verified override**:
|
||||
A rated-TBW override with complete provenance whose applicability to the detected drive was confirmed by machine match or explicit user attestation; the strongest endurance baseline.
|
||||
_Avoid_: Confirmed TBW, trusted value
|
||||
|
||||
**Unverified override**:
|
||||
A rated-TBW override knowingly stored with incomplete provenance; always presented as user-supplied, never as verified.
|
||||
_Avoid_: Forced entry, fallback baseline
|
||||
|
||||
**Sustained regime**:
|
||||
The most recent stretch of the observation history over which the observed usage habit has been stable; the interval whose write rate the usage-adjusted theoretical lifespan consumes.
|
||||
_Avoid_: Current window, detection period
|
||||
@@ -88,26 +64,10 @@ _Avoid_: Confidence interval, error bar
|
||||
The share of wall-clock seconds inside monitoring periods whose usage-habit classification is known rather than unknown.
|
||||
_Avoid_: Uptime, sample count
|
||||
|
||||
**Byte-allocation completeness**:
|
||||
Whether the available evidence establishes all monitored writes attributable to a specified span, without missing counter evidence or unknown boundary shares; distinct from usage-habit classification coverage.
|
||||
_Avoid_: Coverage, estimated allocation
|
||||
|
||||
**Qualifying day**:
|
||||
A UTC date whose represented monitored time meets the coverage requirement for projection evidence. Qualification is provisional while the date is in progress and does not establish byte-allocation completeness.
|
||||
_Avoid_: Completed day, supported day, calibration day
|
||||
|
||||
**Collection run**:
|
||||
One scheduled or on-demand execution of the collector that interrogates the drive and extends the observation history.
|
||||
_Avoid_: Poll, daemon tick
|
||||
|
||||
**Release**:
|
||||
A published version of Fenris: a version tag, its packages in the channel, and its human-readable change notes, all together; a bare tag is not one.
|
||||
_Avoid_: Tag, upload, build
|
||||
|
||||
**Rollback**:
|
||||
Returning to an earlier release by restoring an observation-store snapshot and then installing that release; installing an older package over a newer store is unsupported.
|
||||
_Avoid_: Downgrade, version pinning (as a promise)
|
||||
|
||||
**Deliberate disable**:
|
||||
A monitoring pause made through Fenris's own control path, closing the monitoring period so the paused time is excluded from the usage habit.
|
||||
_Avoid_: Manual stop, service stop
|
||||
|
||||
@@ -1,336 +0,0 @@
|
||||
# Fenris Makefile
|
||||
# Spec: §10.1-10.6
|
||||
|
||||
SHELL := /bin/bash
|
||||
PYTHON := python3
|
||||
VENV_DIR := /opt/fenris
|
||||
VENDOR_DIR := $(VENV_DIR)/vendor
|
||||
BIN_DIR := /usr/local/bin
|
||||
LIBEXEC_DIR := /usr/libexec/fenris
|
||||
UNIT_DIR := /etc/systemd/system
|
||||
POLKIT_DIR := /usr/share/polkit-1/actions
|
||||
CONF_DIR := /etc/fenris
|
||||
DATA_DIR := /var/lib/fenris
|
||||
|
||||
# Placement manifest
|
||||
MANIFEST := $(DATA_DIR)/manifest.txt
|
||||
|
||||
# Legacy history path (IN-4)
|
||||
LEGACY_HISTORY := ./data/history.jsonl
|
||||
|
||||
.PHONY: help install upgrade uninstall purge update-deps test lint check-python check-smartctl import-legacy stage package-deb package-rpm package generate-test-key sign-rpm checksums clearsign release release-run release-dry-run clean
|
||||
|
||||
help:
|
||||
@echo "Fenris NVMe endurance monitor"
|
||||
@echo ""
|
||||
@echo "Targets:"
|
||||
@echo " install - Install Fenris (builds wheel, installs to /opt/fenris)"
|
||||
@echo " upgrade - Upgrade Fenris (reinstall wheel, sync units)"
|
||||
@echo " uninstall - Uninstall Fenris (preserves config and store)"
|
||||
@echo " purge - Remove everything including config and store"
|
||||
@echo " test - Run tests"
|
||||
@echo " lint - Run linter"
|
||||
@echo " update-deps - Update dependency pins"
|
||||
@echo " stage - Stage packaging tree for nfpm"
|
||||
@echo " package - Build deb + rpm packages"
|
||||
@echo " package-deb - Build deb package only"
|
||||
@echo " package-rpm - Build rpm package only"
|
||||
@echo " generate-test-key - Create throwaway GPG key for CI/testing"
|
||||
@echo " sign-rpm - Sign RPM payload with packaging key"
|
||||
@echo " checksums - Generate SHA256SUMS manifest"
|
||||
@echo " clearsign - Clearsign SHA256SUMS with packaging key"
|
||||
@echo " release - Full release (build, sign, checksum, print upload steps)"
|
||||
@echo " release-run - Execute the full release flow via scripts/release.sh"
|
||||
@echo " release-dry-run - Dry-run of the release flow (prints commands only)"
|
||||
@echo " clean - Remove build artifacts"
|
||||
|
||||
# ─── Pre-install gates ──────────────────────────────────────────────────────
|
||||
|
||||
check-python:
|
||||
@echo "=== Verifying Python ≥ 3.10 ==="
|
||||
@$(PYTHON) -c "import sys; v=sys.version_info; exit(0 if (v>=(3,10)) else 1)" || { echo "Error: Python 3.10+ required (found $$($(PYTHON) --version 2>&1))"; exit 1; }
|
||||
|
||||
check-smartctl:
|
||||
@echo "=== Verifying smartctl ==="
|
||||
@smartctl --version 2>/dev/null | head -1 || { echo "Error: smartctl not found (install smartmontools)"; exit 1; }
|
||||
|
||||
# ─── Build ──────────────────────────────────────────────────────────────────
|
||||
|
||||
dist/fenris-*.whl: pyproject.toml src/fenris/*.py
|
||||
@mkdir -p dist
|
||||
$(PYTHON) -m pip wheel --no-deps --wheel-dir dist .
|
||||
|
||||
# ─── Install ────────────────────────────────────────────────────────────────
|
||||
|
||||
install: check-python check-smartctl dist/fenris-*.whl
|
||||
@echo "=== Creating directories ==="
|
||||
@sudo mkdir -p $(LIBEXEC_DIR)
|
||||
@sudo mkdir -p $(CONF_DIR)
|
||||
@sudo mkdir -p $(POLKIT_DIR)
|
||||
|
||||
@echo "=== Creating data directory (root-written, group-read) ==="
|
||||
@sudo groupadd -f fenris
|
||||
@sudo install -d -o root -g fenris -m 2750 $(DATA_DIR)
|
||||
|
||||
@echo "=== Installing version-neutral runtime packages ==="
|
||||
@sudo rm -rf $(VENV_DIR)
|
||||
@sudo install -d -m 0755 $(VENDOR_DIR)
|
||||
@sudo $(PYTHON) -m pip install --disable-pip-version-check --no-compile --target $(VENDOR_DIR) -r requirements.txt dist/fenris-*.whl --quiet
|
||||
|
||||
@echo "=== Installing wrapper ==="
|
||||
@sudo install -m 0755 scripts/fenris $(BIN_DIR)/fenris
|
||||
|
||||
@echo "=== Installing helpers ==="
|
||||
@sudo install -m 0755 src/fenris/monitor.py $(LIBEXEC_DIR)/fenris-monitor
|
||||
@sudo install -m 0755 src/fenris/collect.py $(LIBEXEC_DIR)/fenris-collect
|
||||
|
||||
@echo "=== Installing systemd units (dormant — not enabled/started) ==="
|
||||
@sudo install -m 0644 units/fenris-collect.timer $(UNIT_DIR)/
|
||||
@sudo install -m 0644 units/fenris-collect.service $(UNIT_DIR)/
|
||||
@sudo systemctl daemon-reload
|
||||
|
||||
@echo "=== Installing polkit policy ==="
|
||||
@sudo install -m 0644 polkit/com.bongbetic.fenris.monitor.policy $(POLKIT_DIR)/
|
||||
|
||||
@echo "=== Recording manifest (IN-2, IN-10) ==="
|
||||
@echo "# Fenris placement manifest — do not edit" | sudo tee $(MANIFEST) > /dev/null
|
||||
@echo "# Generated by: sudo make install" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "# Timestamp: $$(date -u +%%Y-%%m-%%dT%%H:%%M:%%SZ)" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(BIN_DIR)/fenris" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(LIBEXEC_DIR)/fenris-monitor" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(LIBEXEC_DIR)/fenris-collect" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(UNIT_DIR)/fenris-collect.timer" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(UNIT_DIR)/fenris-collect.service" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(POLKIT_DIR)/com.bongbetic.fenris.monitor.policy" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(VENV_DIR)" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(DATA_DIR)" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(CONF_DIR)" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(MANIFEST)" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
|
||||
@echo "=== Install complete ==="
|
||||
@echo "Units installed but NOT enabled or started (dormant — IN-3)."
|
||||
@echo "To start monitoring: fenris monitor resume"
|
||||
|
||||
@$(MAKE) --no-print-directory import-legacy
|
||||
|
||||
# ─── Legacy import (IN-4, ST-7) ────────────────────────────────────────────
|
||||
|
||||
import-legacy:
|
||||
@if [ -f "$(LEGACY_HISTORY)" ]; then \
|
||||
echo "=== Detected legacy history: $(LEGACY_HISTORY) ==="; \
|
||||
echo "Running idempotent import..."; \
|
||||
PYTHONPATH=$(VENDOR_DIR) $(PYTHON) -c "from fenris.legacy import import_legacy_history; from fenris.store import init_store; from pathlib import Path; conn = init_store(Path('$(DATA_DIR)/observations.db')); r = import_legacy_history(conn, Path('$(LEGACY_HISTORY)')); conn.close(); print(f' Samples imported: {r.get(\"samples_imported\", 0)}'); print(f' Hours imported: {r.get(\"hours_imported\", 0)}'); print(f' Malformed lines: {r.get(\"malformed_lines\", 0)}') if not r.get('skipped') else print(' Skipped: already imported')" || echo " Warning: import failed (non-fatal)"; \
|
||||
else \
|
||||
echo "=== No legacy history found at $(LEGACY_HISTORY) ==="; \
|
||||
fi
|
||||
|
||||
# ─── Upgrade ────────────────────────────────────────────────────────────────
|
||||
|
||||
upgrade: dist/fenris-*.whl
|
||||
@echo "=== Upgrading Fenris ==="
|
||||
@echo "=== Snapshotting database (IN-6) ==="
|
||||
@sudo cp $(DATA_DIR)/observations.db $(DATA_DIR)/observations.db.bak 2>/dev/null || true
|
||||
|
||||
@echo "=== Installing new version-neutral runtime packages ==="
|
||||
@sudo rm -rf $(VENDOR_DIR)
|
||||
@sudo install -d -m 0755 $(VENDOR_DIR)
|
||||
@sudo $(PYTHON) -m pip install --disable-pip-version-check --no-compile --target $(VENDOR_DIR) -r requirements.txt dist/fenris-*.whl --quiet
|
||||
|
||||
@echo "=== Syncing units against manifest ==="
|
||||
@sudo install -m 0644 units/fenris-collect.timer $(UNIT_DIR)/
|
||||
@sudo install -m 0644 units/fenris-collect.service $(UNIT_DIR)/
|
||||
@sudo install -m 0644 polkit/com.bongbetic.fenris.monitor.policy $(POLKIT_DIR)/
|
||||
@sudo install -m 0755 scripts/fenris $(BIN_DIR)/fenris
|
||||
@sudo install -m 0755 src/fenris/monitor.py $(LIBEXEC_DIR)/fenris-monitor
|
||||
@sudo install -m 0755 src/fenris/collect.py $(LIBEXEC_DIR)/fenris-collect
|
||||
@sudo systemctl daemon-reload
|
||||
|
||||
@echo "=== Updating manifest ==="
|
||||
@echo "# Fenris placement manifest — do not edit" | sudo tee $(MANIFEST) > /dev/null
|
||||
@echo "# Generated by: sudo make upgrade" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "# Timestamp: $$(date -u +%%Y-%%m-%%dT%%H:%%M:%%SZ)" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(BIN_DIR)/fenris" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(LIBEXEC_DIR)/fenris-monitor" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(LIBEXEC_DIR)/fenris-collect" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(UNIT_DIR)/fenris-collect.timer" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(UNIT_DIR)/fenris-collect.service" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(POLKIT_DIR)/com.bongbetic.fenris.monitor.policy" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(VENV_DIR)" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(DATA_DIR)" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(CONF_DIR)" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
@echo "$(MANIFEST)" | sudo tee -a $(MANIFEST) > /dev/null
|
||||
|
||||
@echo "=== Restarting timer only if contents changed and active (IN-5) ==="
|
||||
@for unit in fenris-collect.timer fenris-collect.service; do \
|
||||
TMPFILE=$$(mktemp); \
|
||||
sudo systemctl cat $$unit > $$TMPFILE 2>/dev/null || true; \
|
||||
if ! diff -q $$TMPFILE $(UNIT_DIR)/$$unit > /dev/null 2>&1; then \
|
||||
if systemctl is-active --quiet $$unit; then \
|
||||
echo " $$unit changed and active — restarting"; \
|
||||
sudo systemctl restart $$unit; \
|
||||
fi; \
|
||||
fi; \
|
||||
rm -f $$TMPFILE; \
|
||||
done
|
||||
|
||||
@echo "=== Applying forward-only schema migrations (IN-5, IN-6) ==="
|
||||
@sudo env PYTHONPATH=$(VENDOR_DIR) $(PYTHON) -c "from fenris.store import migrate_to_latest; from pathlib import Path; n = migrate_to_latest(Path('$(DATA_DIR)/observations.db')); print(f' Migration steps applied: {n}') if n else print(' Schema already current')"
|
||||
|
||||
@echo "=== Upgrade complete ==="
|
||||
|
||||
# ─── Uninstall (IN-7) ──────────────────────────────────────────────────────
|
||||
|
||||
uninstall:
|
||||
@echo "=== Uninstalling Fenris ==="
|
||||
|
||||
@if [ -x $(LIBEXEC_DIR)/fenris-monitor ]; then \
|
||||
echo "=== Performing sanctioned disable (§10.4) ==="; \
|
||||
sudo $(LIBEXEC_DIR)/fenris-monitor disable --now || true; \
|
||||
fi
|
||||
|
||||
@echo "=== Stopping units ==="
|
||||
@sudo systemctl stop fenris-collect.timer 2>/dev/null || true
|
||||
@sudo systemctl disable fenris-collect.timer 2>/dev/null || true
|
||||
@sudo systemctl daemon-reload
|
||||
|
||||
@echo "=== Removing installed files (preserving config and store) ==="
|
||||
@-rm -f $(BIN_DIR)/fenris
|
||||
@-rm -f $(LIBEXEC_DIR)/fenris-monitor
|
||||
@-rm -f $(LIBEXEC_DIR)/fenris-collect
|
||||
@-rmdir $(LIBEXEC_DIR) 2>/dev/null || true
|
||||
@-rm -f $(UNIT_DIR)/fenris-collect.timer
|
||||
@-rm -f $(UNIT_DIR)/fenris-collect.service
|
||||
@-rm -f $(POLKIT_DIR)/com.bongbetic.fenris.monitor.policy
|
||||
@sudo rm -rf $(VENV_DIR)
|
||||
@-sudo rm -f $(MANIFEST)
|
||||
|
||||
@echo "=== Uninstall complete ==="
|
||||
@echo "Config preserved at $(CONF_DIR)"
|
||||
@echo "Store preserved at $(DATA_DIR)"
|
||||
|
||||
# ─── Purge ──────────────────────────────────────────────────────────────────
|
||||
|
||||
purge: uninstall
|
||||
@echo "=== Purging Fenris ==="
|
||||
@-sudo rm -rf $(CONF_DIR)
|
||||
@-sudo rm -rf $(DATA_DIR)
|
||||
@echo "=== Purge complete ==="
|
||||
|
||||
# ─── Test ───────────────────────────────────────────────────────────────────
|
||||
|
||||
test:
|
||||
$(PYTHON) -m pytest tests/ -v
|
||||
|
||||
# ─── Lint ───────────────────────────────────────────────────────────────────
|
||||
|
||||
lint:
|
||||
$(PYTHON) -m ruff check src/ tests/
|
||||
|
||||
# ─── Dependencies (IN-8) ───────────────────────────────────────────────────
|
||||
|
||||
update-deps:
|
||||
$(PYTHON) -m pip compile pyproject.toml -o requirements.txt
|
||||
|
||||
# ─── Packaging (spec §3, §4, §5) ────────────────────────────────────────────
|
||||
|
||||
# Version is sourced from pyproject.toml for both formats
|
||||
FENRIS_VERSION := $(shell sed -n 's/^version = "\(.*\)"/\1/p' pyproject.toml)
|
||||
|
||||
# GPG signing — packaging key UID (spec §4)
|
||||
PACKAGING_KEY ?= packaging@bongbetic.com
|
||||
|
||||
stage:
|
||||
@echo "=== Staging packaging tree (v$(FENRIS_VERSION)) ==="
|
||||
$(PYTHON) -m pip wheel --no-deps --wheel-dir dist .
|
||||
bash packaging/stage.sh "$(FENRIS_VERSION)"
|
||||
|
||||
package-deb: stage
|
||||
@echo "=== Building deb package ==="
|
||||
VERSION="$(FENRIS_VERSION)" nfpm pkg -f packaging/nfpm.yaml -p deb -t dist/
|
||||
@echo "=== deb package built: dist/fenris_$(FENRIS_VERSION)_amd64.deb ==="
|
||||
|
||||
package-rpm: stage
|
||||
@echo "=== Building rpm package ==="
|
||||
VERSION="$(FENRIS_VERSION)" nfpm pkg -f packaging/nfpm.yaml -p rpm -t dist/
|
||||
@echo "=== rpm package built: dist/fenris-$(FENRIS_VERSION)-1.x86_64.rpm ==="
|
||||
|
||||
package: package-deb package-rpm
|
||||
@echo "=== Both packages built in dist/ ==="
|
||||
|
||||
# ─── GPG key management ─────────────────────────────────────────────────────
|
||||
|
||||
generate-test-key:
|
||||
@echo "=== Generating throwaway test GPG key ==="
|
||||
@echo "This key is for CI/testing only — never use for real releases."
|
||||
printf '%%no-protection\nKey-Type: RSA\nKey-Length: 3072\nName-Real: Fenris Packaging (TESTING ONLY)\nName-Email: packaging-test@bongbetic.com\nExpire-Date: 0\n%%commit\n' | \
|
||||
gpg --batch --gen-key
|
||||
@echo "=== Test key created. Fingerprint: ==="
|
||||
@gpg --fingerprint packaging-test@bongbetic.com
|
||||
|
||||
# ─── Signing ────────────────────────────────────────────────────────────────
|
||||
|
||||
sign-rpm: package-rpm
|
||||
@echo "=== Signing RPM payload ==="
|
||||
@rpm --import packaging/keys/fenris-packaging.asc 2>/dev/null || true
|
||||
rpmsign --addsign --define "_gpg_name $(PACKAGING_KEY)" \
|
||||
dist/fenris-$(FENRIS_VERSION)-1.x86_64.rpm
|
||||
@echo "=== RPM signed ==="
|
||||
@rpm -Kv dist/fenris-$(FENRIS_VERSION)-1.x86_64.rpm
|
||||
|
||||
checksums: package
|
||||
@echo "=== Generating SHA256SUMS ==="
|
||||
cd dist && sha256sum fenris_$(FENRIS_VERSION)_amd64.deb \
|
||||
fenris-$(FENRIS_VERSION)-1.x86_64.rpm > SHA256SUMS
|
||||
@echo "=== SHA256SUMS written ==="
|
||||
@cat dist/SHA256SUMS
|
||||
|
||||
clearsign: checksums
|
||||
@echo "=== Clearsigning SHA256SUMS ==="
|
||||
gpg --batch --yes --clearsign --local-user $(PACKAGING_KEY) \
|
||||
dist/SHA256SUMS
|
||||
@echo "=== SHA256SUMS.asc written ==="
|
||||
|
||||
# ─── Release (spec §5) ──────────────────────────────────────────────────────
|
||||
|
||||
release: package sign-rpm clearsign
|
||||
@echo ""
|
||||
@echo "=== Release v$(FENRIS_VERSION) ==="
|
||||
@echo ""
|
||||
@echo "Artifacts:"
|
||||
@ls -la dist/fenris_$(FENRIS_VERSION)_amd64.deb \
|
||||
dist/fenris-$(FENRIS_VERSION)-1.x86_64.rpm \
|
||||
dist/SHA256SUMS.asc 2>/dev/null
|
||||
@echo ""
|
||||
@echo "Verify signing (manual):"
|
||||
@echo " rpm -Kv dist/fenris-$(FENRIS_VERSION)-1.x86_64.rpm"
|
||||
@echo " gpg --verify dist/SHA256SUMS.asc dist/SHA256SUMS"
|
||||
@echo ""
|
||||
@echo "Upload to registry:"
|
||||
@echo " curl -X PUT -u user:token -T dist/fenris_$(FENRIS_VERSION)_amd64.deb \\"
|
||||
@echo " 'https://git.bongbetic.com/api/packages/xavierk/debian/pool/bookworm/main/upload'"
|
||||
@echo " curl -X PUT -u user:token -T dist/fenris_$(FENRIS_VERSION)_amd64.deb \\"
|
||||
@echo " 'https://git.bongbetic.com/api/packages/xavierk/debian/pool/jammy/main/upload'"
|
||||
@echo " curl -X PUT -u user:token -T dist/fenris_$(FENRIS_VERSION)_amd64.deb \\"
|
||||
@echo " 'https://git.bongbetic.com/api/packages/xavierk/debian/pool/noble/main/upload'"
|
||||
@echo " curl -X PUT -u user:token -T dist/fenris-$(FENRIS_VERSION)-1.x86_64.rpm \\"
|
||||
@echo " 'https://git.bongbetic.com/api/packages/xavierk/rpm/fenris/upload'"
|
||||
@echo ""
|
||||
@echo "Create Gitea release with notes and attach:"
|
||||
@echo " dist/fenris_$(FENRIS_VERSION)_amd64.deb"
|
||||
@echo " dist/fenris-$(FENRIS_VERSION)-1.x86_64.rpm"
|
||||
@echo " dist/SHA256SUMS.asc"
|
||||
@echo ""
|
||||
@echo "Key ceremony: delete the private key after upload."
|
||||
@echo " See docs/install/signing-key-ceremony.md"
|
||||
|
||||
# ─── Automated release flow (issue #52) ──────────────────────────────────────
|
||||
|
||||
release-run:
|
||||
bash scripts/release.sh --publish
|
||||
|
||||
release-dry-run:
|
||||
bash scripts/release.sh --dry-run
|
||||
|
||||
clean:
|
||||
@echo "=== Cleaning build artifacts ==="
|
||||
rm -rf build/stage dist/fenris-*.deb dist/fenris-*.rpm dist/SHA256SUMS*
|
||||
@@ -1,239 +1,157 @@
|
||||
# Fenris 🐺
|
||||
<p align="center">
|
||||
<picture>
|
||||
<source srcset="assets/bongbetic-brand/wordmark-light.png" media="(prefers-color-scheme: dark)">
|
||||
<img src="assets/bongbetic-brand/wordmark-dark.png" alt="Bongbetic" width="260">
|
||||
</picture>
|
||||
<br>
|
||||
<sub>crafted with stubborn curiosity by <a href="https://bongbetic.com">Bongbetic</a></sub>
|
||||
</p>
|
||||
|
||||
*Observes an NVMe drive's real-world use and translates that history into an understandable endurance outlook.*
|
||||
<p align="center">
|
||||
<img src="assets/bongbetic-brand/b_glyph.svg" width="48" alt="Fenris glyph">
|
||||
</p>
|
||||
|
||||
Fenris is a persistent TUI monitor backed by a short-lived privileged collector on a systemd timer. It reads SMART data every few minutes, stores compact observation history in SQLite, and recomputes a usage-adjusted theoretical lifespan on every screen render — no fairy dust, just your actual bytes.
|
||||
<h1 align="center">Fenris 🐺 — Your SSD's Tell-All Diary</h1>
|
||||
|
||||
<p align="center">
|
||||
<em>Your NVMe drive has been keeping secrets. Fenris makes it confess — in real time.</em>
|
||||
<br>
|
||||
<em>How much did you write today? How long until it taps out? No fairy dust — just your actual bytes.</em>
|
||||
</p>
|
||||
|
||||
---
|
||||
|
||||
## Requirements
|
||||
Fenris is a tiny, stubborn daemon that eavesdrops on your NVMe drive's SMART gossip, writes it down every few minutes, and serves you a live dashboard that actually means something. Not "vibes" — **real GB written in the last 24 hours, real GB/hour, and a real countdown in hours, days, and years until your drive's endurance runs out**.
|
||||
|
||||
- **Python ≥ 3.10** (verified at install time)
|
||||
- **smartmontools** (`smartctl` — verified at install time)
|
||||
- **systemd** with a polkit agent (the collector runs as root oneshot; elevation is exclusively polkit)
|
||||
> Think of it as a Fitbit for your SSD. Except it doesn't nag you to drink water.
|
||||
|
||||
No other OS packages or Python dependencies beyond [Textual](https://textual.textualize.io/) (pinned in the lockfile).
|
||||
## What it actually does (no hand-waving)
|
||||
|
||||
## Install from package (recommended)
|
||||
- **Listens** — polls `smartctl -j` on your NVMe device (default every 5 minutes, you pick).
|
||||
- **Remembers** — appends every sample to `data/history.jsonl` and rolls up per-hour totals into `data/hourly.jsonl` (survives restarts, rebuilds itself if you yank the power).
|
||||
- **Calculates** — rolling 24-hour window: *exact* bytes written in the last 24h, GB/h, GB/day, implied total TBW from `percentage_used`, remaining TB, and a projected life-remaining breakdown. Warming-up badge until it has 24h of coverage — no fake confidence.
|
||||
- **Shows off** — dense, live dashboard with wear-over-time + trailing-24h per-hour bars, sticky header, live countdown, and stale warnings if the daemon dozes off.
|
||||
|
||||
### Debian / Ubuntu (apt)
|
||||
## You need
|
||||
|
||||
The Gitea instance Debian registry signs metadata with its own key. Verify the
|
||||
instance key fingerprint (TOFU hardening):
|
||||
- **Python 3.7+**
|
||||
- **smartmontools** (`smartctl`)
|
||||
- Root-ish access to read NVMe SMART (passwordless `smartctl` or just run with `sudo` — your call)
|
||||
|
||||
```text
|
||||
Fingerprint: <print after first release — paste beside the curl one-liner>
|
||||
```
|
||||
### The sudo dance (one time)
|
||||
|
||||
Add the instance key and repository:
|
||||
Fenris runs `sudo -n smartctl ...` so it doesn't get stuck asking for a password mid-nap:
|
||||
|
||||
```bash
|
||||
sudo mkdir -p /etc/apt/keyrings
|
||||
sudo curl -fsSL -o /etc/apt/keyrings/gitea-xavierk.asc \
|
||||
https://git.bongbetic.com/api/packages/xavierk/debian/repository.key
|
||||
|
||||
echo "deb [signed-by=/etc/apt/keyrings/gitea-xavierk.asc] \
|
||||
https://git.bongbetic.com/api/packages/xavierk/debian bookworm main" \
|
||||
| sudo tee /etc/apt/sources.list.d/fenris.list
|
||||
|
||||
sudo apt update && sudo apt install fenris
|
||||
sudo visudo
|
||||
# add this line (swap in your username):
|
||||
youruser ALL=(root) NOPASSWD: /usr/sbin/smartctl
|
||||
```
|
||||
|
||||
Replace `bookworm` with your distribution codename (`bookworm`, `jammy`, or
|
||||
`noble`).
|
||||
No sudo? Run the whole thing with `sudo` and it'll still behave.
|
||||
|
||||
### Fedora / openSUSE Tumbleweed (RPM)
|
||||
## Get it running — 30 seconds
|
||||
|
||||
Use the Fenris-owned repo file (not Gitea's auto-generated one):
|
||||
### The cozy way
|
||||
|
||||
```bash
|
||||
sudo dnf config-manager --add-repo \
|
||||
https://git.bongbetic.com/xavierk/Fenris/raw/branch/main/packaging/fenris.repo
|
||||
|
||||
sudo dnf install fenris
|
||||
./fenris.sh
|
||||
# pick 1) Start monitoring → choose device / interval / port → done
|
||||
```
|
||||
|
||||
On openSUSE Tumbleweed, add the same standard RPM repository file and install
|
||||
with zypper:
|
||||
### The no-nonsense way
|
||||
|
||||
```bash
|
||||
sudo zypper addrepo --refresh \
|
||||
https://git.bongbetic.com/xavierk/Fenris/raw/branch/main/packaging/fenris.repo fenris
|
||||
sudo zypper install fenris
|
||||
python3 fenris.py start # defaults: /dev/nvme0, every 300s, port 8420
|
||||
python3 fenris.py start --interval 60 --port 9000 # if you're impatient
|
||||
python3 fenris.py status # "are we live? how's the drive?"
|
||||
python3 fenris.py sample # one sneaky sample right now
|
||||
python3 fenris.py stop # tuck it back in
|
||||
```
|
||||
|
||||
The repo file sets `gpgcheck=1` against the Fenris packaging key (downloaded
|
||||
from the raw URL in `gpgkey`) and `repo_gpgcheck=0` (metadata check left to
|
||||
TLS).
|
||||
Dashboard lives at **http://localhost:8420** (or whatever port you chose).
|
||||
|
||||
### Package signature verification
|
||||
## The menu, demystified
|
||||
|
||||
The RPM payload is signed with the Fenris packaging key (RSA 3072).
|
||||
Verification happens automatically via dnf's `gpgcheck=1`. For manual
|
||||
verification of downloaded assets:
|
||||
|
||||
```bash
|
||||
rpm -Kv fenris-*.x86_64.rpm # RPM payload signature
|
||||
gpg --verify SHA256SUMS.asc SHA256SUMS # Clearsigned checksum manifest
|
||||
sha256sum -c SHA256SUMS # Checksum match
|
||||
```
|
||||
|
||||
The packaging public key is published in-repo — no keyservers. See
|
||||
`packaging/keys/fenris-packaging.asc` and
|
||||
`docs/install/signing-key-ceremony.md` for key lifecycle details.
|
||||
|
||||
### Dormant install
|
||||
|
||||
A fresh package install is fully dormant. Units are present but disabled;
|
||||
nothing runs. The only opt-in is the sanctioned toggle:
|
||||
|
||||
```bash
|
||||
fenris monitor resume # enable timer + open first monitoring period
|
||||
fenris monitor pause # close the period, disable timer
|
||||
```
|
||||
|
||||
## Development install (make install)
|
||||
|
||||
For contributors building from source:
|
||||
|
||||
```bash
|
||||
sudo make install
|
||||
```
|
||||
|
||||
This builds a wheel, installs its locked pure-Python runtime packages into
|
||||
`/opt/fenris/vendor`,
|
||||
and places helpers, units, and the polkit policy. Units are dormant by default.
|
||||
|
||||
```bash
|
||||
sudo make upgrade # re-sync wheel, units, schema
|
||||
make uninstall # removes artifacts, preserves config and store
|
||||
make purge # also removes /etc/fenris and /var/lib/fenris
|
||||
```
|
||||
|
||||
## Upgrade
|
||||
|
||||
### Package upgrade
|
||||
|
||||
```bash
|
||||
sudo apt update && sudo apt upgrade fenris # Debian/Ubuntu
|
||||
sudo dnf upgrade fenris # Fedora
|
||||
```
|
||||
|
||||
### Development upgrade
|
||||
|
||||
```bash
|
||||
sudo make upgrade
|
||||
```
|
||||
|
||||
What it does:
|
||||
1. Snapshots `observations.db` to a one-generation backup (`.bak`).
|
||||
2. Replaces the locked runtime packages under `/opt/fenris/vendor`.
|
||||
3. Syncs units and polkit against the manifest; runs `daemon-reload`.
|
||||
4. Restarts the timer **only** if unit contents changed **and** it is active — a running collection run finishes on its mapped interpreter; the next run uses the new code.
|
||||
5. Applies forward-only schema migrations (the store directory is never rebuilt; automatic downgrade does not exist).
|
||||
|
||||
Rollback: reinstall the previous version and restore `observations.db.bak`.
|
||||
|
||||
## Migration from make install
|
||||
|
||||
If Fenris was previously installed with `sudo make uninstall` first, then
|
||||
installed from the package, existing config, store, and group survive by path
|
||||
continuity. Over-installing the package over a `make install` is
|
||||
**forbidden** — stale units shadow vendor placement. See
|
||||
[docs/install/migrate-from-makeinstall.md](docs/install/migrate-from-makeinstall.md).
|
||||
|
||||
## Uninstall and purge
|
||||
|
||||
### Package removal
|
||||
|
||||
```bash
|
||||
sudo apt remove fenris # preserves config and store
|
||||
sudo apt purge fenris # also removes config and store
|
||||
sudo dnf remove fenris # preserves config and store
|
||||
```
|
||||
|
||||
### Development removal
|
||||
|
||||
```bash
|
||||
make uninstall # removes artifacts, preserves config and observation history
|
||||
make purge # also removes /etc/fenris and /var/lib/fenris
|
||||
```
|
||||
|
||||
Uninstall performs the sanctioned disable first (`fenris-monitor disable --now`) — an open period closes `user_disabled` — then removes the runtime packages, helpers, units, polkit policy, and wrapper while keeping `/etc/fenris` and the observation store. Reinstalling resumes from the preserved store.
|
||||
|
||||
## Cadence drop-ins
|
||||
|
||||
The default collection cadence is **5 minutes** (`OnUnitInactiveSec=5min` in the timer unit). To change it, place a systemd drop-in:
|
||||
|
||||
```bash
|
||||
sudo systemctl edit fenris-collect.timer
|
||||
# Add:
|
||||
# [Timer]
|
||||
# OnUnitInactiveSec=10min
|
||||
```
|
||||
|
||||
No interval key exists in `/etc/fenris/fenris.conf`. Cadence is a systemd concern, not a Fenris configuration key.
|
||||
|
||||
## CLI reference
|
||||
|
||||
| Command | Behavior |
|
||||
|---|---|
|
||||
| `fenris` | Opens the TUI (no arguments). |
|
||||
| `fenris status` | Projection facts, enabled/active state, last collect outcome, journal hint on failure or staleness. Never auto-samples. |
|
||||
| `fenris sample` | On-demand collection via the privileged helper. Blocks until the run completes. |
|
||||
| `fenris monitor pause` | Sanctioned disable — asks for confirmation, then disables the timer and closes the monitoring period. |
|
||||
| `fenris monitor resume` | Sanctioned enable — enables the timer and opens a monitoring period. No confirmation. |
|
||||
| `fenris baseline set <json>` | CLI-side validation, then polkit-guarded persistence. |
|
||||
| `fenris baseline clear` | Remove the endurance baseline. |
|
||||
| `fenris import <path>` | Idempotent single-transaction legacy import. |
|
||||
| `fenris start` / `stop` / `run` | Rejected with a one-line migration pointer — never aliased. |
|
||||
| `fenris --device` | Rejected with a pointer to the configuration file. |
|
||||
|
||||
## Reading the dashboard
|
||||
|
||||
`fenris` opens the TUI dashboard.
|
||||
|
||||
- **Continuity** — the service strip's continuity line (and `fenris status`) reports whether monitoring survives reboots: `monitoring: active in background · persists across reboots`, or `monitoring: does not start on next boot`.
|
||||
- **Paused vs. quit** — a full-width `monitoring: paused — deliberate disable` block means collection is stopped (`fenris monitor pause`); resume with `fenris monitor resume`. Pressing `q` only leaves the screen — monitoring keeps running in the background.
|
||||
- **Auth banner** — at launch, `privileged actions will prompt for authentication (polkit)` shows once and clears on the first refresh. Privileged actions elevate via polkit; Fenris never asks for sudo.
|
||||
|
||||
Per-release notes live on the [releases page](https://git.bongbetic.com/xavierk/Fenris/releases): each entry is the version's `CHANGELOG.md` section — what was added, changed, and fixed — plus standing install and verification instructions.
|
||||
|
||||
## Retired menu options
|
||||
|
||||
The legacy `fenris.sh` menu script and the `fenris.py` monolith have been removed. Here's where the old options went:
|
||||
|
||||
| Legacy option | Successor |
|
||||
|---|---|
|
||||
| 1) Start monitoring | `fenris monitor resume` |
|
||||
| 2) Stop monitoring | `fenris monitor pause` |
|
||||
| 3) Status / current wear stats | `fenris status` |
|
||||
| 4) Take one sample right now | `fenris sample` |
|
||||
| 5) Open dashboard URL | Removed — the HTML dashboard and HTTP server are gone; the TUI is the primary interface. |
|
||||
|
||||
## Configuration
|
||||
|
||||
`/etc/fenris/fenris.conf` holds exactly one key — the device selector:
|
||||
Run `./fenris.sh` and you'll get:
|
||||
|
||||
```
|
||||
device = /dev/disk/by-id/nvme-Samsung_SSD_980_PRO_2TB_S6BENS0Txxxxx
|
||||
1) Start monitoring (background daemon + dashboard)
|
||||
2) Stop monitoring
|
||||
3) Status / current wear stats
|
||||
4) Take one sample right now
|
||||
5) Open dashboard URL
|
||||
---
|
||||
h) Help / how this works
|
||||
q) Exit (go touch grass)
|
||||
```
|
||||
|
||||
Use a stable `/dev/disk/by-id/` path. Raw `/dev/nvmeX` paths are warned against. The file is re-read every collection run.
|
||||
## What Fenris jots down
|
||||
|
||||
| Field | What's the gossip? |
|
||||
|-------|---------------------|
|
||||
| `percentage_used` | The drive's own wear-o-meter (0–100%) |
|
||||
| `bytes_written` / `bytes_read` | Lifetime totals — the receipts |
|
||||
| `available_spare` | Spare blocks left (%) |
|
||||
| `media_errors` | Uncorrectable boo-boos |
|
||||
| `power_on_hours` | How long it's been awake |
|
||||
| `temperature_c` | Is it sweating? |
|
||||
| `critical_warning` | NVMe's panic flags |
|
||||
|
||||
Hourly rollups also stash `bytes_written` per hour, `pct_start`/`pct_end`, and temp peaks — so the 24h math stays honest.
|
||||
|
||||
## The dashboard — what's on screen
|
||||
|
||||
- **Hero card: Projected life remaining** — big, friendly `361 d 2 h` (plus `≈ 361 days · ≈ 8666 hours · ≈ 0.99 years`), backed by `~280 GB/day` and `~101 TB left of ~202 TB total` on the test box.
|
||||
- **Data written (24h)** — exact GB in the rolling window + coverage (`10.4h of 24h` until warmed up).
|
||||
- **Write rate** — GB/h and GB/day, live.
|
||||
- **Wear, spare, temp, errors, power-on** — the usual suspects, with progress bars and polite color-coding.
|
||||
- **Two charts, side by side:** wear over time + trailing-24h hourly write bars (with a cheeky "now" bar for the current partial hour).
|
||||
- **Live plumbing:** polling synced to your interval, ETag-cached, countdown to next sample, warming-up + stale banners, pauses when you hide the tab (saves your battery, you're welcome).
|
||||
|
||||
**API for the tinkerers:** `GET /api/data` · `/api/hourly` · `/api/summary` · `/api/config` · `/api/status` — all JSON, all friendly.
|
||||
|
||||
## Where's my stuff?
|
||||
|
||||
| Artifact | Package install | make install |
|
||||
|---|---|---|
|
||||
| Wrapper | `/usr/bin/fenris` | `/usr/local/bin/fenris` |
|
||||
| Helpers | `/usr/libexec/fenris/` | `/usr/libexec/fenris/` |
|
||||
| Units | `/usr/lib/systemd/system/` (vendor) | `/etc/systemd/system/` |
|
||||
| Polkit policy | `/usr/share/polkit-1/actions/` | `/usr/share/polkit-1/actions/` |
|
||||
| sysusers/tmpfiles | `/usr/lib/{sysusers,tmpfiles}.d/fenris.conf` | managed by Makefile |
|
||||
| Configuration | `/etc/fenris/fenris.conf` | `/etc/fenris/fenris.conf` |
|
||||
| Observation store | `/var/lib/fenris/observations.db` | `/var/lib/fenris/observations.db` |
|
||||
| Runtime packages | `/opt/fenris/vendor` | `/opt/fenris/vendor` |
|
||||
| Legacy history | — | `./data/history.jsonl` (auto-imported) |
|
||||
```
|
||||
fenris/
|
||||
├── fenris.py # the whole show — daemon + server + math
|
||||
├── fenris.sh # the cozy menu
|
||||
├── README.md # hi — you're here
|
||||
├── assets/bongbetic-brand/ # Bongbetic wordmarks & glyphs (for Gitea + dashboard)
|
||||
└── data/
|
||||
├── history.jsonl # raw samples (JSONL, append-only)
|
||||
├── hourly.jsonl # per-hour rollups (auto-rebuilt on restart)
|
||||
├── fenris.pid # daemon PID
|
||||
└── fenris.log # daemon chatter
|
||||
```
|
||||
|
||||
## CLI cheat sheet
|
||||
|
||||
```bash
|
||||
python3 fenris.py start [--device /dev/nvme0] [--interval 300] [--port 8420]
|
||||
python3 fenris.py stop
|
||||
python3 fenris.py status
|
||||
python3 fenris.py sample [--device /dev/nvme0]
|
||||
python3 fenris.py run # foreground mode — what `start` spawns internally
|
||||
```
|
||||
|
||||
## Oops — troubleshooting without the tears
|
||||
|
||||
**"smartctl not found"**
|
||||
```bash
|
||||
sudo apt install smartmontools # Debian/Ubuntu
|
||||
sudo pacman -S smartmontools # Arch — you already knew
|
||||
```
|
||||
|
||||
**"needs root" / permission denied**
|
||||
Set up the passwordless line above, or just `sudo ./fenris.sh`.
|
||||
|
||||
**Dashboard says "stale"**
|
||||
Daemon napped or crashed. `python3 fenris.py status` will tell you. Kick it again with `start`.
|
||||
|
||||
**Only 10 hours of data and it says "preliminary"?**
|
||||
That's honesty, not a bug. It needs 24h of real writes to give a tight estimate. Let it simmer — the number gets sharper every hour.
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -4,10 +4,6 @@
|
||||
|
||||
Accepted — resolves [Define the persistent observation store and legacy migration](https://git.bongbetic.com/xavierk/Fenris/issues/2) on the [Wayfinder map](https://git.bongbetic.com/xavierk/Fenris/issues/1).
|
||||
|
||||
Amended by [Define endurance-baseline provenance and validation](https://git.bongbetic.com/xavierk/Fenris/issues/12): the `endurance_baseline` field set and validation contract — derived verification, entry-time unprivileged sysfs validation, and read-time controller-segment applicability.
|
||||
|
||||
Amended by [Decide controller-segment metadata columns](https://git.bongbetic.com/xavierk/Fenris/issues/14): the `controller_segments` metadata snapshot — normalized identity diagnostics plus `vid`/`ssvid`/`transport` and a degraded flag, frozen at segment open, all nullable.
|
||||
|
||||
## Context
|
||||
|
||||
Fenris today persists full SMART samples to an append-only `data/history.jsonl` beside a derived `data/hourly.jsonl`, both in the checkout, with no schema versioning and silent skipping of malformed lines. The redesign replaces the HTML dashboard with a keyboard-first TUI backed by a short-lived privileged collector on a systemd timer and an unprivileged TUI ([lifecycle research](https://git.bongbetic.com/xavierk/Fenris/src/branch/research/systemd-privilege-lifecycle/docs/research/systemd-privilege-lifecycle.md)), and projects a usage-adjusted theoretical lifespan from Data Units Written over wall-clock time with categorical confidence ([endurance research](https://git.bongbetic.com/xavierk/Fenris/src/branch/research/nvme-endurance-signals/docs/research/nvme-endurance-signals.md)). The store must support a root writer appearing every few minutes while an unprivileged reader queries concurrently, must migrate the legacy observation history idempotently and interruption-safely, and must version its schema.
|
||||
@@ -21,8 +17,8 @@ Fenris today persists full SMART samples to an append-only `data/history.jsonl`
|
||||
- `hour_observations` — one row per UTC hour: the usage-habit split (`seconds_active`, `seconds_idle`, `seconds_powered_off`, `seconds_unknown`), DUW/DUR deltas, temperature min/avg/max, sample count, coverage flag. Classification thresholds belong to the projection model, not the store.
|
||||
- `day_aggregates` — one row per UTC day; the habit-evidence grain.
|
||||
- `monitoring_periods` — `started_at`, `ended_at` (NULL = open), `end_cause` enum (`user_disabled`, `migrated`, …). Powered-off time stays inside a period; deliberately disabled time does not.
|
||||
- `controller_segments` — boundaries where controller identity changes or DUW decreases; write deltas are never computed across a segment. Each row carries a metadata snapshot frozen when the segment opens and immutable thereafter ([Decide controller-segment metadata columns](https://git.bongbetic.com/xavierk/Fenris/issues/14)): normalized `subnqn`, `sn`, `mn`, `fr`, plus `vid`, `ssvid`, `transport`, and an `identity_degraded` flag — human diagnostics, never key components (`cntlid` excluded: it distinguishes controllers within one subsystem, out of scope for a single-drive monitor). `fr` may go stale after a mid-segment firmware update; counter discontinuities belong to the DUW-monotonic axis. Every metadata column is nullable — legacy-imported segments carry `mn` with NULLs, degraded segments whatever was observed — so incompleteness stays explicit.
|
||||
- `endurance_baseline` — one active row, replaced on edit ([Define endurance-baseline provenance and validation](https://git.bongbetic.com/xavierk/Fenris/issues/12)): the rated-TBW value in bytes (`E_rated = entered_TBW × 10¹²`) plus mandatory provenance — source URL, document revision, entry date, model string, nominal capacity — and frozen validation facts (detected model, detected capacity bytes, `validated_by` `machine`/`user`, `validated_at`). Verification is derived at read — complete provenance and a drive match (machine or attested), never a stored boolean; incomplete provenance stores only behind an explicit unverified acknowledgment, as NULL fields in that precedence tier. Entry validation is an unprivileged live sysfs read of the configured device (normalized model containment with an interactive confirm recorded as `validated_by = user`; capacity within ±1%); at projection time applicability is a model match against the current controller segment, and a mismatch is retained — never auto-deleted — leaving the projection Unavailable.
|
||||
- `controller_segments` — boundaries where controller identity changes or DUW decreases; write deltas are never computed across a segment.
|
||||
- `endurance_baseline` — verified rated-TBW override in bytes plus provenance (source URL, document revision, entry date).
|
||||
- Projections are not stored; they are recomputed on read. There is no separate latest-status table.
|
||||
4. **Day boundary**: UTC, matching hours, so day derivation from hour rows is monotonic and DST-ambiguous or 23/25-hour days never exist in the store.
|
||||
5. **Retention**: raw samples are kept 14 days and pruned opportunistically by the collector; hour observations and day aggregates are retained indefinitely.
|
||||
|
||||
@@ -4,8 +4,6 @@
|
||||
|
||||
Accepted — resolves [Define the lifespan projection and confidence model](https://git.bongbetic.com/xavierk/Fenris/issues/4) on the [Wayfinder map](https://git.bongbetic.com/xavierk/Fenris/issues/1).
|
||||
|
||||
Amended by [Decide how degraded identity affects projection confidence](https://git.bongbetic.com/xavierk/Fenris/issues/15): a blank (degraded) identity key caps confidence at Limited evidence, and identity-change semantics extend verbatim to blank keys.
|
||||
|
||||
## Context
|
||||
|
||||
Fenris's current `compute_summary` projects from a single trailing-24-hour write rate against endurance inferred as `DUW / Percentage Used` or synthesized as `capacity × 600`, alongside a second linear regression of Percentage Used toward 100. The [endurance research](https://git.bongbetic.com/xavierk/Fenris/src/branch/research/nvme-endurance-signals/docs/research/nvme-endurance-signals.md) established which signals can defensibly support a projection, and [ADR 0001](0001-observation-store-sqlite.md) fixed the observation store while leaving classification thresholds and every projection rule to this model. This decision defines the algorithm and the user-facing contract the TUI consumes.
|
||||
@@ -33,15 +31,14 @@ Fenris's current `compute_summary` projects from a single trailing-24-hour write
|
||||
7. **Staleness.** A newest day aggregate older than 48 hours drops confidence one level (Supported → Limited) and is shown as a contributing fact.
|
||||
8. **Confidence rule table.**
|
||||
- **Unavailable**: no applicable baseline; DUW unsupported; zero rate over the regime; controller-identity change.
|
||||
- **Supported**: verified baseline **and** ≥ 14 qualifying days **and** coverage ≥ 80% **and** fresh (< 48 h) **and** 7/28/90 rates within a factor of 2 across existing horizons **and** no single day ≥ 50% of trailing 28-day bytes **and** regime ≥ 7 days old **and** the current controller segment's identity key is not degraded.
|
||||
- **Supported**: verified baseline **and** ≥ 14 qualifying days **and** coverage ≥ 80% **and** fresh (< 48 h) **and** 7/28/90 rates within a factor of 2 across existing horizons **and** no single day ≥ 50% of trailing 28-day bytes **and** regime ≥ 7 days old.
|
||||
- **Limited**: every other case with a baseline and a positive rate; the failing facts are shown.
|
||||
- **Degraded identity** ([Decide how degraded identity affects projection confidence](https://git.bongbetic.com/xavierk/Fenris/issues/15)): a controller segment whose identity key is blank — every rung of the key ladder empty — is identity-degraded. Supported is unreachable while the current segment is degraded, because a blank key cannot detect a replacement; the fact "controller identity unavailable — replacement detection relies on write-counter continuity only" renders with every state, and the cap combines idempotently with the staleness drop (both land at Limited). Ephemeral markers (model "Linux", non-pcie transport) are segment metadata, never confidence facts.
|
||||
- Confidence always renders as state plus contributing facts, never a percentage.
|
||||
9. **Segment breaks.** A DUW decrease with unchanged controller identity quarantines nothing: prior day aggregates remain habit evidence and the projection is Unavailable only until the new segment re-warms. A controller-identity change quarantines prior history from projection entirely — it describes a different drive. Degraded keys get no special casing ([Decide how degraded identity affects projection confidence](https://git.bongbetic.com/xavierk/Fenris/issues/15)): a blank-key segment is marked identity-degraded and segmented by DUW monotonicity alone, any visible change of the recorded key — including to or from blank — is a controller-identity change and quarantines, and equal blank keys continue the segment. Since even a degraded→healthy transition quarantines, the projection window only ever spans segments sharing one key, so a degraded current segment needs no cross-segment propagation rule.
|
||||
9. **Segment breaks.** A DUW decrease with unchanged controller identity quarantines nothing: prior day aggregates remain habit evidence and the projection is Unavailable only until the new segment re-warms. A controller-identity change quarantines prior history from projection entirely — it describes a different drive.
|
||||
10. **Implied-baseline eligibility.** The Percentage-Used-implied baseline is computed only after ≥ 2 Percentage Used increments within the current controller segment; until then the projection is Unavailable with "vendor wear estimate too coarse to imply endurance".
|
||||
11. **Uncertainty.** The scenario range is the only spread shown; no statistical confidence interval appears anywhere. Zero rate → "no finite projection from this history", never infinity or zero.
|
||||
12. **Language.** The endurance research's required wording and six disclosures are adopted verbatim as the specification's language section.
|
||||
13. **Contract.** The projection function hands the TUI: the confidence state, the contributing facts — including the degraded-identity fact when the current segment's key is blank — the headline remaining time when one exists, the scenario range, the Percentage-Used context line, and the disclosure text. Projections are recomputed on read, never stored.
|
||||
13. **Contract.** The projection function hands the TUI: the confidence state, the contributing facts, the headline remaining time when one exists, the scenario range, the Percentage-Used context line, and the disclosure text. Projections are recomputed on read, never stored.
|
||||
|
||||
## Consequences
|
||||
|
||||
|
||||
@@ -4,8 +4,6 @@
|
||||
|
||||
Accepted — resolves [Define the collector, service, and CLI lifecycle](https://git.bongbetic.com/xavierk/Fenris/issues/8) on the [Wayfinder map](https://git.bongbetic.com/xavierk/Fenris/issues/1). Amends the toggle mechanism of [Verify systemd lifecycle and privilege constraints](https://git.bongbetic.com/xavierk/Fenris/issues/7); its spirit — scoped, explicit, authenticated, no generic `manage-unit-files` grant — is intact.
|
||||
|
||||
Amended by [Define endurance-baseline provenance and validation](https://git.bongbetic.com/xavierk/Fenris/issues/12): the helper gains a `baseline` verb that persists the CLI-validated endurance-baseline row — same fixed-operation, polkit-mediated pattern.
|
||||
|
||||
## Context
|
||||
|
||||
Fenris's current single process combines daemonization, a PID file, an HTTP dashboard, and control (`fenris.py start/stop/status/sample`) over checkout-relative state. [ADR 0001](0001-observation-store-sqlite.md) fixed the observation store, including `monitoring_periods` whose `user_disabled` end cause records deliberate pauses, and the [systemd lifecycle research](https://git.bongbetic.com/xavierk/Fenris/src/branch/research/systemd-privilege-lifecycle/docs/research/systemd-privilege-lifecycle.md) fixed the timer + oneshot architecture, standard paths, journal diagnostics, allow-listed status reads, and polkit-mediated startup toggles — while leaving cadence mechanics, the configuration surface, CLI compatibility, staleness thresholds, and the mechanism that records a deliberate disable open. In particular, `systemctl enable`/`disable` cannot write a monitoring-period row, so a direct-systemctl toggle cannot satisfy the store's semantics.
|
||||
@@ -15,7 +13,7 @@ Fenris's current single process combines daemonization, a PID file, an HTTP dash
|
||||
1. **Units.** Two system units only: `fenris-collect.timer` (`WantedBy=timers.target`) and `fenris-collect.service` (`Type=oneshot`, root, `ExecStart=/usr/libexec/fenris/fenris-collect`; no listener, no UI code). The TUI and CLI are ordinary unprivileged processes and never units. There is no `/run/fenris` coordination surface: systemd serializes runs, the observation store holds state, and failures go to the journal per [ADR 0001](0001-observation-store-sqlite.md).
|
||||
2. **Cadence.** Default five minutes: `OnBootSec=2min`, `OnUnitInactiveSec=5min` (measured from run completion; drift accepted because hours are the evidence grain), `AccuracySec=30s`, `Persistent=no`, no suspend catch-up (absent hours classify through power-on-hours evidence), `TimeoutStartSec=90s` so a hung interrogation fails visibly. Cadence changes are documented drop-ins on the timer unit (`systemctl edit` + daemon-reload); no interval key exists in configuration.
|
||||
3. **Configuration.** `/etc/fenris/fenris.conf` holds exactly one key: the device selector, a stable `/dev/disk/by-id/…` path (raw nodes accepted with an instability warning), validated at collection time. The oneshot re-reads it every run, so there is no reload path to design. An invalid selector is a bounded failed run — journal plus failed unit result, retried next interval; `status` and the TUI also read the world-readable file directly and surface a `configuration error: <reason>` fact.
|
||||
4. **Entry points.** Two privileged binaries: `/usr/libexec/fenris/fenris-collect` (device interrogation and store writes; the unit's `ExecStart`) and `/usr/libexec/fenris/fenris-monitor` (fixed operations `enable` and `disable` with optional `--now`, plus the collect trigger, monitoring-period bookkeeping, and `baseline set`/`baseline clear` persistence for the CLI-validated endurance baseline; the only binary the polkit policy authorizes). One unprivileged `fenris` for humans: no arguments opens the TUI; subcommands (`status`, `sample`, `monitor pause`, `monitor resume`) are the CLI.
|
||||
4. **Entry points.** Two privileged binaries: `/usr/libexec/fenris/fenris-collect` (device interrogation and store writes; the unit's `ExecStart`) and `/usr/libexec/fenris/fenris-monitor` (fixed operations `enable` and `disable` with optional `--now`, plus the collect trigger and monitoring-period bookkeeping; the only binary the polkit policy authorizes). One unprivileged `fenris` for humans: no arguments opens the TUI; subcommands (`status`, `sample`, `monitor pause`, `monitor resume`) are the CLI.
|
||||
5. **Sanctioned toggle.** Pause = `disable --now`; Resume = `enable --now`; both executed by `fenris-monitor`, which performs the systemctl operation and the monitoring-period bookkeeping in one step, under polkit action `com.bongbetic.fenris.monitor` (`auth_admin`, covering the collect trigger too). Root invokes the helpers directly; where no polkit agent exists the operation fails cleanly and prints the root equivalent. This amends the research's direct-systemctl toggle: a period boundary cannot be recorded by systemctl, so the toggle must be Fenris's own fixed operation.
|
||||
6. **Period rows.** Idempotent matrix: a first-ever enable opens a period at the enable moment (hours before the first successful sample are unknown-but-inside, correctly so when the device errors); a resume with an open period — a raw `systemctl stop` intervened — changes no row, the gap remaining inside as unknown seconds; a resume with no open period opens a new row at the resume moment; a pause with an open period closes it `user_disabled` at the pause moment; a pause otherwise is a no-op. A raw stop or disable outside the helper is an unexplained gap, never `user_disabled`: only the sanctioned path can record intent.
|
||||
7. **On-demand collection.** `fenris sample` and the TUI's collect-now route through `fenris-monitor` → `systemctl start fenris-collect.service`, which blocks until the oneshot exits, and the outcome (freshness line or journal hint) is reported synchronously. No code path outside `fenris-collect` touches the device; the TUI never samples in-process; no confirmation is required.
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
## Status
|
||||
|
||||
Accepted — resolves [Define installation, upgrade, and removal behavior](https://git.bongbetic.com/xavierk/Fenris/issues/9) on the [Wayfinder map](https://git.bongbetic.com/xavierk/Fenris/issues/1). Amended by [ADR 0007](0007-package-delivery-amends-0004.md): package delivery replaces `make install` as primary; layout/ownership and maintainer-script mechanics per 0007. Runtime semantics (dormant install, polkit-only elevation, snapshot + forward-only migration, one-generation rollback) unchanged.
|
||||
Accepted — resolves [Define installation, upgrade, and removal behavior](https://git.bongbetic.com/xavierk/Fenris/issues/9) on the [Wayfinder map](https://git.bongbetic.com/xavierk/Fenris/issues/1).
|
||||
|
||||
## Context
|
||||
|
||||
|
||||
@@ -1,28 +0,0 @@
|
||||
# 6. Collector acquisition path: smartctl counters, sysfs identity
|
||||
|
||||
## Status
|
||||
|
||||
Accepted — resolves [Choose the collector's NVMe acquisition path](https://git.bongbetic.com/xavierk/Fenris/issues/16) on the [Wayfinder map](https://git.bongbetic.com/xavierk/Fenris/issues/1).
|
||||
|
||||
## Context
|
||||
|
||||
The collector ([ADR 0003](0003-service-lifecycle-and-sanctioned-toggle.md)) must acquire SMART/Health counters, thermal evidence, and controller identity each run. The [controller-identity research](https://git.bongbetic.com/xavierk/Fenris/src/branch/research/controller-identity/docs/research/controller-identity.md) fixed the identity key to the normalized, kernel-exposed subsystem NQN and warned that normalization must be specified once and applied at write time — or a collector implementation change can split a drive's own history. [ADR 0004](0004-install-upgrade-removal-lifecycle.md) pins exact Python dependencies in a dedicated venv, and the [segment-metadata decision](https://git.bongbetic.com/xavierk/Fenris/issues/14) froze nullable `vid`/`ssvid`/`transport` alongside the identity fields. Three first-party paths were candidates: the official libnvme Python bindings (SWIG; sysfs-backed attribute getters delivering normalized values), `nvme` CLI JSON output, and the incumbent `smartctl -j` plus sysfs reads.
|
||||
|
||||
## Decision
|
||||
|
||||
1. **Pin.** Every collection run acquires counters and thermal evidence solely from `smartctl -a -j <device>` and controller identity (`subnqn`, `sn`, `mn`, `fr`, `transport`) solely from sysfs (`/sys/class/nvme/<ctrl>/`). No other acquisition path exists anywhere in the codebase.
|
||||
2. **Hard pin, no fallback.** Any acquisition failure — missing binary, nonzero exit, malformed JSON, unreadable sysfs attribute — fails the whole collection run; [ADR 0005](0005-failure-detection-and-recovery.md)'s flat retry and freshness grading absorb the miss. A partial sample (identity without counters, or counters without identity) is never written: a transient read failure must not push a healthy drive down the degraded-identity path.
|
||||
3. **Normalization once, at write time.** One collector-side function normalizes every identity field: trailing spaces and newlines stripped, no case folding, empty-after-strip stored blank. `smartctl` counter and thermal fields are consumed as-is (smartmontools already trims the strings it copies). Padded and unpadded renderings of the same field therefore yield byte-identical stored values.
|
||||
4. **Segment metadata sourcing.** `transport` comes from the NVMe class sysfs directory; `vid`/`ssvid` from the PCI node (`/sys/class/nvme/<ctrl>/device/{vendor,subsystem_vendor}`) when present, null otherwise — metadata only, never key components.
|
||||
5. **Prerequisites.** `make install` verifies `smartctl` is present and fails cleanly otherwise. The acquisition path adds no Python dependency and no OS package beyond smartmontools; the [ADR 0004](0004-install-upgrade-removal-lifecycle.md) lockfile is untouched.
|
||||
|
||||
## Considered options
|
||||
|
||||
- **libnvme Python bindings** — the purest API and natively-normalized getters, but the SWIG module is not on PyPI: entering the venv requires the distro's `python3-libnvme` through `--system-site-packages` or a from-source build, coupling the exact-lockfile venv to the system Python and the distro's shipping choices. Rejected on dependency weight for one privileged five-minute oneshot.
|
||||
- **`nvme` CLI JSON** — one binary covers counters and identity, but it adds an OS package for what smartmontools already provides, emits untrimmed strings, and reports `subnqn` from Identify data rather than the kernel: when a controller reports an empty NQN the kernel synthesizes one for sysfs while `id-ctrl` JSON omits the field, so the identity ladder would drop a rung depending on the drive. Rejected on packaging and identity-key consistency.
|
||||
|
||||
## Consequences
|
||||
|
||||
- The venv stays pure-Python; the two acquisition channels per run (subprocess JSON plus sysfs reads) hide behind one acquisition function, gated by acceptance criteria AC-1–AC-5.
|
||||
- Identity is read from exactly the source the identity key names; libnvme's getters wrap the same sysfs attributes, so the values agree byte-for-byte where both exist.
|
||||
- Switching acquisition path later is history-sensitive: a future path must deliver byte-identical normalized identity values, or the change itself forces a controller-segment boundary.
|
||||
@@ -1,34 +0,0 @@
|
||||
# 7. Package delivery: native deb + rpm packages, amending the installation lifecycle
|
||||
|
||||
## Status
|
||||
|
||||
Accepted — resolves [Task: Compose release spec + ADR amending 0004](https://git.bongbetic.com/xavierk/Fenris/issues/42) on the [Wayfinder map](https://git.bongbetic.com/xavierk/Fenris/issues/33). This ADR **amends [ADR 0004](0004-install-upgrade-removal-lifecycle.md)** on delivery and file ownership only; every runtime semantic of 0004 — dormant install, polkit-only elevation, observation-store snapshot + forward-only migration, one-generation rollback — is inherited verbatim, restated below where the package delivery changes *who* performs it.
|
||||
|
||||
## Context
|
||||
|
||||
ADR 0004 fixed delivery as `sudo make install` from a source checkout: wheel into a Fenris-owned runtime directory at `/opt/fenris`, a hand-rolled placement manifest, units in `/etc/systemd/system`. The release plan ([map](https://git.bongbetic.com/xavierk/Fenris/issues/33), decisions [Lock channel + toolchain](https://git.bongbetic.com/xavierk/Fenris/issues/38), [Signing + key policy](https://git.bongbetic.com/xavierk/Fenris/issues/39), [Package ownership](https://git.bongbetic.com/xavierk/Fenris/issues/40), [Migration path](https://git.bongbetic.com/xavierk/Fenris/issues/41), [Release cadence](https://git.bongbetic.com/xavierk/Fenris/issues/43)) now ships Fenris as native deb + rpm packages built by nfpm and published to the self-hosted Gitea 1.27.1 package registry, for Debian 12, Ubuntu 22.04/24.04, Fedora 40+, and openSUSE Tumbleweed (x86_64), with locked pure-Python dependencies vendored at `/opt/fenris/vendor`. Packages become the primary delivery; ADR 0004's delivery model demotes to a dev fallback.
|
||||
|
||||
The implementation-ready operative contracts live in the [release and packaging specification](../spec/release-packaging.md); this ADR records the decisions and their rationale.
|
||||
|
||||
## Decision
|
||||
|
||||
Amendments to ADR 0004, section by section:
|
||||
|
||||
1. **Delivery (amended).** Packages are primary: one deb per codename pool (`bookworm`, `jammy`, `noble`) and one rpm (group `fenris`, Fedora 40+ and openSUSE Tumbleweed), built by nfpm from a single `packaging/nfpm.yaml` over locked pure-Python runtime packages staged at `/opt/fenris/vendor`, published to the Gitea Debian/RPM registry and installed with `apt`, `dnf`, or `zypper`. `sudo make install` remains as the dev fallback for machines without packages; the two deliveries are mutually exclusive per machine. Version scheme `<pyproject-version>-1`, revision bump on rebuild.
|
||||
2. **Layout and manifest (amended).** The hand-rolled manifest model is retired: the dpkg/rpm database **is** the manifest, and nothing like `manifest.txt` ships. Package-owned layout: units in `/usr/lib/systemd/system` (vendor placement; `/etc/systemd/system` is admin-only for drop-ins and enable state); helpers stay in `/usr/libexec/fenris` (exactly `fenris-monitor` and `fenris-collect` — no new polkit-reachable binaries); polkit policy in `/usr/share/polkit-1/actions/`; wrapper at `/usr/bin/fenris` (FHS; `/usr/local/bin` remains `make install`'s). The `fenris` group is declared in `/usr/lib/sysusers.d/fenris.conf` (`g fenris -`) and `/var/lib/fenris` in `/usr/lib/tmpfiles.d/fenris.conf` (`d /var/lib/fenris 2750 root fenris -`), both invoked from the maintainer scripts. The package owns the `/var/lib/fenris` directory only; `observations.db`, WAL sidecars, and `.bak` are never owned and never ghosted — ghost-erase would delete the store, violating 0004 §8.
|
||||
3. **Privilege (unchanged).** Root acts through maintainer scripts at install/upgrade/removal time; at runtime, elevation is exclusively polkit, exactly as 0004 §3 and [ADR 0003](0003-service-lifecycle-and-sanctioned-toggle.md) §5 fix it.
|
||||
4. **Dormant install (restated for packages).** A fresh package install is fully dormant: postinst/%post performs `systemctl daemon-reload` (plus `systemd-sysusers` and `systemd-tmpfiles --create`) and nothing else — never enable, never preset, never start; no preset file ships. The sanctioned toggle (`fenris monitor resume`) remains the only opt-in.
|
||||
5. **Legacy import (narrowed).** Auto-detection of `./data/history.jsonl` is scoped to `make install` only — a package install has no checkout to inspect. `fenris import <path>` remains available as the only import path from packages.
|
||||
6. **Upgrade (inherited, maintainer-script mechanics).** Upgrades arrive as packages from the single registry channel. postinst/%post on upgrade: snapshot `observations.db` → one-generation `.bak`, run forward-only schema migrations through the target `python3` with `/opt/fenris/vendor` on its import path (no new binaries), `daemon-reload`, and restart `fenris-collect.timer` only if unit contents changed **and** it is active. `/var/lib/fenris` is never rebuilt; a running oneshot finishes on its old interpreter.
|
||||
7. **Rollback (unchanged, plus one hard edge).** One-generation `.bak` semantics are unchanged. Package downgrade is additionally unsupported: forward-only store-version refusal means installing an older package over a newer store fails by design; documented rollback = restore the snapshot, then install the old release.
|
||||
8. **Removal (mapped).** deb `remove` ≈ `make uninstall` (conffile and store survive); deb `purge` ≈ `make purge` (plus `.bak` and group cleanup); rpm erase ≈ `make uninstall` (unmodified config removed, modified survives as `.rpmsave`; purge is a documented manual command). prerm/%preun performs the sanctioned disable — `fenris-monitor disable --now`, closing the period `user_disabled` — on remove/erase **only, never on upgrade** (deb prerm upgrade case is a no-op; rpm `%preun` gated on `$1 -eq 0`).
|
||||
9. **Conffile semantics (new).** `/etc/fenris/fenris.conf` ships as a placeholder-commented default with no active device selector — deb conffile, rpm `%config(noreplace)`. The device selector is entered by hand (root edits the file), as in both prior deliveries; no configuration verb is added to `fenris-monitor`, and [ADR 0003](0003-service-lifecycle-and-sanctioned-toggle.md) §3's read-and-validate-at-collection-time semantics are untouched. On upgrade, local edits survive as-is; a changed package default lands beside them as `.dpkg-new`/`.rpmnew`.
|
||||
10. **Migration from make-install systems (new).** Remove-then-install via runbook only — no migration script, no auto-clean. preinst/%pre aborts with a pointer to the runbook if make-install remnants are detected (`/var/lib/fenris/manifest.txt` or `/etc/systemd/system/fenris-collect.timer`). Store and config survive by path continuity; the migration resets the system to dormant and the user opts back in with `fenris monitor resume`.
|
||||
|
||||
## Consequences
|
||||
|
||||
- Package installs, upgrades, and removals carry dpkg/rpm-native semantics; nothing in Fenris's own tooling duplicates them.
|
||||
- The manifest was 0004's answer to "what did the installer place"; the package database answers it better, and uninstall-keeps-store now holds by package ownership rather than by manifest discipline.
|
||||
- `make install` and packages are mutually exclusive per machine; over-install is blocked, not repaired (stale `/etc` units would silently shadow vendor units).
|
||||
- Hand-edited configuration remains the model: the device selector is a root-edited file in every delivery, keeping the polkit surface at exactly one binary.
|
||||
- Release mechanics — channel, signing, cadence, rollback documentation — are fixed in the [release and packaging specification](../spec/release-packaging.md) and the tickets it cites; this ADR deliberately stops at lifecycle semantics.
|
||||
@@ -1,87 +0,0 @@
|
||||
# Migrating from make-install to packages
|
||||
|
||||
This runbook covers the transition from a `sudo make install` system to the native deb or rpm package. Packages are the primary delivery; `make install` remains as the dev fallback. The two deliveries are **mutually exclusive** per machine.
|
||||
|
||||
## Why over-install is forbidden
|
||||
|
||||
Installing a package over a make-install system silently breaks things:
|
||||
|
||||
- **Stale admin units shadow vendor units.** `make install` places `fenris-collect.timer` and `fenris-collect.service` in `/etc/systemd/system/`. The package installs them in `/usr/lib/systemd/system/` (vendor placement). Systemd loads admin units first — the stale copy takes precedence, and the package update never reaches the running system.
|
||||
- **The local wrapper shadows the package wrapper.** `make install` places the `fenris` wrapper at `/usr/local/bin/fenris`. The package places it at `/usr/bin/fenris`. The shell finds `/usr/local/bin` first on PATH — the old checkout-relative wrapper runs instead of the package wrapper.
|
||||
|
||||
Neither condition is reversible by reinstalling the package. The only safe path is remove-then-install.
|
||||
|
||||
## Pre-migration checklist
|
||||
|
||||
1. Confirm no monitoring period is actively running that you want to preserve across the gap:
|
||||
```
|
||||
fenris status
|
||||
```
|
||||
The migration resets the system to dormant (see [No-move continuity](#no-move-continuity) below). You opt back in with `fenris monitor resume`.
|
||||
|
||||
2. If you have hand-edited configuration at `/etc/fenris/fenris.conf`, note it. The config survives the migration in place (see below).
|
||||
|
||||
## Remove step
|
||||
|
||||
```
|
||||
sudo make uninstall
|
||||
```
|
||||
|
||||
This performs the **sanctioned disable** (`fenris-monitor disable --now`), closing the current monitoring period as `user_disabled`. It then removes all make-install artifacts: the venv at `/opt/fenris`, the wrapper at `/usr/local/bin/fenris`, the helpers at `/usr/libexec/fenris/`, the units in `/etc/systemd/system/`, and the polkit policy. The placement manifest at `/var/lib/fenris/manifest.txt` is removed.
|
||||
|
||||
**What survives the remove:**
|
||||
|
||||
- `/var/lib/fenris/observations.db` (and WAL sidecars, `.bak`) — the observation store
|
||||
- `/var/lib/fenris/` directory itself — root-written, group-read
|
||||
- `/etc/fenris/fenris.conf` — your hand-written configuration
|
||||
- The `fenris` system group — created by `groupadd -f` during make-install
|
||||
- Journal entries — age out naturally
|
||||
|
||||
## Install step
|
||||
|
||||
```
|
||||
sudo apt install fenris # Debian/Ubuntu
|
||||
sudo dnf install fenris # Fedora
|
||||
```
|
||||
|
||||
The package installs into its own layout without touching the surviving store, config, or group.
|
||||
|
||||
## No-move continuity
|
||||
|
||||
These invariants are verified by the containerized acceptance tests (issue #50):
|
||||
|
||||
| Asset | Make-install state | Package post-install | Mechanism |
|
||||
|---|---|---|---|
|
||||
| `fenris` group | Exists (`groupadd -f`) | Unchanged | `systemd-sysusers` is a no-op when the group already exists |
|
||||
| `/var/lib/fenris` directory | Exists (mode 2750, root:fenris) | Unchanged | `systemd-tmpfiles --create` is a no-op when the directory already exists |
|
||||
| `observations.db` + sidecars | Present from prior monitoring | Unchanged, never owned by the package | Package owns the directory only; store contents are never ghosted |
|
||||
| `/etc/fenris/fenris.conf` | Hand-edited device selector | Survives in place; package default lands as `.dpkg-new` / `.rpmnew` | dpkg conffile / rpm `%config(noreplace)` semantics |
|
||||
| Store schema | Version from prior Fenris release | Caught up by the upgrade-path migration | `postinst` / `%post` runs `migrate_to_latest()` on upgrade |
|
||||
|
||||
The package detects the make-install system has been removed by the absence of the two markers:
|
||||
- `/var/lib/fenris/manifest.txt` (the placement manifest)
|
||||
- `/etc/systemd/system/fenris-collect.timer` (pre-manifest make installs)
|
||||
|
||||
If either marker exists, the package installation aborts with a pointer to this runbook.
|
||||
|
||||
## Reset-to-dormant
|
||||
|
||||
`make uninstall`'s sanctioned disable closes the open monitoring period as `user_disabled`. After the package install, the system is dormant — the timer is installed but disabled, nothing is running, no monitoring period is open.
|
||||
|
||||
To resume monitoring:
|
||||
|
||||
```
|
||||
fenris monitor resume
|
||||
```
|
||||
|
||||
This is the sanctioned opt-in. It enables the timer and opens the first monitoring period in one step. The migration costs at most one short sample gap (the interval between `make uninstall` and `fenris monitor resume`), honestly recorded in the endurance timeline.
|
||||
|
||||
## Verification
|
||||
|
||||
After migration, confirm the package is correctly installed:
|
||||
|
||||
```
|
||||
fenris status
|
||||
```
|
||||
|
||||
The status command should show the dormant state: timer disabled, no active monitoring period, and the observation store intact from the prior make-install system.
|
||||
@@ -1,230 +0,0 @@
|
||||
# Signing key ceremony
|
||||
|
||||
The Fenris packaging key signs RPM payloads and clearsigns SHA256SUMS manifests.
|
||||
This document describes the key's lifecycle: creation, per-release use, rotation,
|
||||
and destruction.
|
||||
|
||||
## Key specification
|
||||
|
||||
| Property | Value |
|
||||
|---|---|
|
||||
| Algorithm | RSA 3072 |
|
||||
| UID | `Fenris Packaging <packaging@bongbetic.com>` |
|
||||
| Expiry | 2 years from creation |
|
||||
| Hierarchy | Single key — no master/subkey split (single maintainer, manual builds) |
|
||||
| Private key storage | Password manager only |
|
||||
| Public key storage | `packaging/keys/fenris-packaging.asc` in-repo, release notes, docs |
|
||||
| Keyservers | Never — TOFU-over-TLS via raw URL |
|
||||
|
||||
## First release: key creation
|
||||
|
||||
```bash
|
||||
# Generate the dedicated RSA-3072 packaging key
|
||||
gpg --batch --gen-key <<EOF
|
||||
%no-protection
|
||||
Key-Type: RSA
|
||||
Key-Length: 3072
|
||||
Name-Real: Fenris Packaging
|
||||
Name-Email: packaging@bongbetic.com
|
||||
Expire-Date: 2y
|
||||
%commit
|
||||
EOF
|
||||
|
||||
# Export the public half — this file is committed to the repo
|
||||
gpg --armor --export packaging@bongbetic.com > packaging/keys/fenris-packaging.asc
|
||||
|
||||
# Print the fingerprint for docs and release notes
|
||||
gpg --fingerprint packaging@bongbetic.com
|
||||
```
|
||||
|
||||
Save the **private key** to the password manager immediately:
|
||||
|
||||
```bash
|
||||
gpg --armor --export-secret-keys packaging@bongbetic.com
|
||||
```
|
||||
|
||||
Then **delete the private key from the local keyring** — it must never persist
|
||||
on any build host:
|
||||
|
||||
```bash
|
||||
gpg --delete-secret-keys packaging@bongbetic.com
|
||||
gpg --delete-keys packaging@bongbetic.com
|
||||
```
|
||||
|
||||
The committed `fenris-packaging.asc` must contain the real public key (replace
|
||||
the placeholder comments).
|
||||
|
||||
## Per-release signing flow
|
||||
|
||||
Each release performs: **import → sign → delete**. The private key is never
|
||||
stored on disk longer than the release takes.
|
||||
|
||||
### Step 1: Import the private key
|
||||
|
||||
Retrieve the private key from the password manager and import it:
|
||||
|
||||
```bash
|
||||
gpg --import /tmp/packaging-key-private.asc
|
||||
rm /f /tmp/packaging-key-private.asc # Shred if possible
|
||||
```
|
||||
|
||||
### Step 2: Build and sign packages
|
||||
|
||||
The Makefile target `make release` handles signing automatically when the
|
||||
key is in the keyring:
|
||||
|
||||
```bash
|
||||
make release # builds, signs RPM, clearsigns SHA256SUMS, prints upload steps
|
||||
```
|
||||
|
||||
Under the hood:
|
||||
|
||||
1. `rpmsign --addsign` signs the RPM payload with the packaging key
|
||||
(invoked by `make sign-rpm`).
|
||||
2. `sha256sum` generates the checksum manifest.
|
||||
3. `gpg --clearsign` produces `SHA256SUMS.asc` with the packaging key.
|
||||
|
||||
### Step 3: Delete the private key
|
||||
|
||||
Immediately after signing:
|
||||
|
||||
```bash
|
||||
gpg --delete-secret-keys packaging@bongbetic.com
|
||||
gpg --delete-keys packaging@bongbetic.com
|
||||
```
|
||||
|
||||
Verify the key is gone:
|
||||
|
||||
```bash
|
||||
gpg --list-keys packaging@bongbetic.com
|
||||
# Should produce: gpg: keyblock resource ...: No such file or directory
|
||||
```
|
||||
|
||||
The entire import → sign → delete cycle should take minutes. The private key
|
||||
must never be left in any keyring between releases.
|
||||
|
||||
## Key rotation (outline)
|
||||
|
||||
When the key approaches expiry, or if it is compromised:
|
||||
|
||||
1. **Generate a new key** using the same procedure as first release.
|
||||
2. **Publish the new public key** alongside the old one in-repo:
|
||||
```text
|
||||
packaging/keys/fenris-packaging.asc # new key (primary)
|
||||
packaging/keys/fenris-packaging-previous.asc # old key (one cycle)
|
||||
```
|
||||
3. **Sign the next RPM** with the new key.
|
||||
4. **Update `fenris.repo`** to list both `gpgkey` URLs (dnf accepts multiple):
|
||||
```ini
|
||||
gpgkey=https://git.bongbetic.com/xavierk/Fenris/raw/branch/main/packaging/keys/fenris-packaging.asc
|
||||
https://git.bongbetic.com/xavierk/Fenris/raw/branch/main/packaging/keys/fenris-packaging-previous.asc
|
||||
```
|
||||
5. **Drop the old key** from the repo after one release cycle. Delete
|
||||
`fenris-packaging-previous.asc` and revert `gpgkey` to the single URL.
|
||||
|
||||
## Verification
|
||||
|
||||
Consumers verify the RPM payload signature via dnf (gpgcheck=1 in
|
||||
`fenris.repo` points at the published public key). The SHA256SUMS manifest
|
||||
verification is manual for downloaded assets:
|
||||
|
||||
```bash
|
||||
gpg --verify SHA256SUMS.asc SHA256SUMS
|
||||
sha256sum -c SHA256SUMS
|
||||
```
|
||||
|
||||
## One-time live probe
|
||||
|
||||
Before the first real release, verify the full registry path end-to-end with a
|
||||
throwaway package. This confirms apt/dnf metadata generation, signature
|
||||
verification, and consumer setup work as a real consumer would experience them.
|
||||
|
||||
### Setup
|
||||
|
||||
```bash
|
||||
# Create a throwaway package name to avoid polluting fenris metadata
|
||||
PROBE_NAME="fenris-regtest"
|
||||
PROBE_VERSION="0.0.1"
|
||||
```
|
||||
|
||||
### Publish
|
||||
|
||||
```bash
|
||||
# Build a throwaway deb and rpm (use the existing nfpm config with a dummy name)
|
||||
# Or use a pre-built package — the probe tests the registry path, not the build
|
||||
|
||||
# Upload deb to all codename pools
|
||||
for CODENAME in bookworm jammy noble; do
|
||||
curl --fail -X PUT \
|
||||
-u "xavierk:${GITEA_TOKEN}" \
|
||||
-T "dist/${PROBE_NAME}_${PROBE_VERSION}_amd64.deb" \
|
||||
"https://git.bongbetic.com/api/packages/xavierk/debian/pool/${CODENAME}/main/upload"
|
||||
done
|
||||
|
||||
# Upload rpm
|
||||
curl --fail -X PUT \
|
||||
-u "xavierk:${GITEA_TOKEN}" \
|
||||
-T "dist/${PROBE_NAME}-${PROBE_VERSION}-1.x86_64.rpm" \
|
||||
"https://git.bongbetic.com/api/packages/xavierk/rpm/fenris/upload"
|
||||
```
|
||||
|
||||
### Verify apt metadata (Debian/Ubuntu consumer perspective)
|
||||
|
||||
```bash
|
||||
# On a Debian/Ubuntu machine:
|
||||
sudo mkdir -p /etc/apt/keyrings
|
||||
sudo curl -fsSL https://git.bongbetic.com/api/packages/xavierk/debian/repository.key \
|
||||
| sudo gpg --dearmor -o /etc/apt/keyrings/gitea-xavierk.asc
|
||||
|
||||
echo "deb [signed-by=/etc/apt/keyrings/gitea-xavierk.asc] https://git.bongbetic.com/api/packages/xavierk/debian bookworm main" \
|
||||
| sudo tee /etc/apt/sources.list.d/fenris.list
|
||||
|
||||
sudo apt update
|
||||
apt show ${PROBE_NAME} # metadata present, correct version
|
||||
apt install --dry-run ${PROBE_NAME} # dependency resolution works
|
||||
|
||||
# Verify InRelease signature
|
||||
apt-key list 2>/dev/null || gpg --no-default-keyring --keyring /etc/apt/keyrings/gitea-xavierk.asc --list-keys
|
||||
```
|
||||
|
||||
### Verify dnf metadata (Fedora consumer perspective)
|
||||
|
||||
```bash
|
||||
# On a Fedora machine:
|
||||
sudo dnf config-manager --add-repo https://git.bongbetic.com/xavierk/Fenris/raw/branch/main/packaging/fenris.repo
|
||||
# Or use Gitea's auto-generated repo for the probe:
|
||||
sudo dnf config-manager --add-repo https://git.bongbetic.com/api/packages/xavierk/rpm/fenris.repo
|
||||
|
||||
dnf info ${PROBE_NAME} # metadata present, correct version
|
||||
dnf install --assumeno ${PROBE_NAME} # dependency resolution works
|
||||
|
||||
# Verify rpm signature
|
||||
rpm -q --scripts ${PROBE_NAME} # no scripts (throwaway)
|
||||
```
|
||||
|
||||
### Verify checksums and clearsign
|
||||
|
||||
```bash
|
||||
# Download from release assets or local build
|
||||
gpg --verify SHA256SUMS.asc SHA256SUMS
|
||||
sha256sum -c SHA256SUMS
|
||||
```
|
||||
|
||||
### Cleanup
|
||||
|
||||
```bash
|
||||
# Delete the throwaway packages from the registry
|
||||
for CODENAME in bookworm jammy noble; do
|
||||
curl --fail -X DELETE \
|
||||
-u "xavierk:${GITEA_TOKEN}" \
|
||||
"https://git.bongbetic.com/api/packages/xavierk/debian/pool/${CODENAME}/main/${PROBE_NAME}/${PROBE_VERSION}/amd64"
|
||||
done
|
||||
|
||||
curl --fail -X DELETE \
|
||||
-u "xavierk:${GITEA_TOKEN}" \
|
||||
"https://git.bongbetic.com/api/packages/xavierk/rpm/fenris/${PROBE_NAME}/${PROBE_VERSION}/x86_64"
|
||||
|
||||
# Remove test source list on consumer machines
|
||||
sudo rm /etc/apt/sources.list.d/fenris.list
|
||||
sudo apt update
|
||||
```
|
||||
@@ -0,0 +1,99 @@
|
||||
# Controller identity for observation-history segmentation
|
||||
|
||||
Research for [Verify the controller identity that segments observation history](https://git.bongbetic.com/xavierk/Fenris/issues/11).
|
||||
|
||||
## Decision
|
||||
|
||||
Record the **subsystem NQN as exposed by the Linux kernel** — normalized, trailing-space stripped — as the controller-segment identity key:
|
||||
|
||||
```text
|
||||
identity_key = strip(subnqn) # /sys/class/nvme-subsystem/…/subsysnqn,
|
||||
# identical to /sys/class/nvme/nvmeX/subsysnqn
|
||||
fallback: "nqn.2014.08.org.nvmexpress:" + hex4(vid) + hex4(ssvid)
|
||||
+ raw20(sn) + raw40(mn) # byte-for-byte the kernel's synthesized NQN
|
||||
last resort: strip(mn) + "|" + strip(sn) # when only smartctl-style fields exist
|
||||
```
|
||||
|
||||
Store the raw `subnqn`, `sn`, `mn` strings (normalized) plus `fr` (firmware revision) as **segment metadata**, never `fr` inside the key: firmware revision is the one mandatory field that legitimately changes on the same drive ([Base Spec 2.0e §5.17.2.1: FR is the *currently active* firmware revision](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)). Namespace identifiers (NGUID, EUI-64, UUID) are excluded from the key: they are namespace-scoped while the SMART counters being segmented are controller-scoped, and each may be absent or reused. When the key is blank, record an empty key with a degraded marker and rely on the DUW-monotonic rule; a same-model same-serial replacement is unobservable by any identifier and is caught — as a boundary, not an identity change — by the DUW-decrease rule of ADR 0001.
|
||||
|
||||
This key satisfies the ticket's asymmetry: a drive replacement changes `subnqn` (real NQNs are unique per subsystem; kernel-generated ones embed serial+model), while firmware quirks and counter resets on the same drive leave it unchanged — counter discontinuities are already ADR 0001's second segmentation axis.
|
||||
|
||||
## What each candidate is
|
||||
|
||||
| Candidate | Spec definition | Scope | Mandatory | Verdict for the key |
|
||||
|---|---|---|---|---|
|
||||
| **SN + MN** | ASCII strings assigned by the vendor in Identify Controller, bytes 23:04 and 63:24; §4.3 shows them left-justified and space-padded ([2.0e §5.17.2.1, Identify Controller data structure](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf), [§4.3 Identifier Format and Layout](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)) | NVM subsystem | Mandatory for I/O and Admin controllers | Core of the fallback; uniqueness explicitly not guaranteed by the spec |
|
||||
| **FR** | Currently active firmware revision, ASCII, bytes 71:64 ([2.0e §5.17.2.1](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)) | Domain (subsystem) | Mandatory | Never in the key — it is *meant* to change on the same drive |
|
||||
| **SUBNQN** | NVM Subsystem NQN, UTF-8 null-terminated, bytes 1023:768; mandatory if the controller is ≥ 1.2.1, otherwise may be all zero ([2.0e §5.17.2.1](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)) | NVM subsystem (shared by all its controllers) | Mandatory ≥ 1.2.1, optional below | **Chosen key**; the spec says hosts *should* use it as the subsystem's unique identifier |
|
||||
| **NGUID / EUI-64** | IEEE-based identifiers in Identify Namespace, bytes 119:104 / 127:120 ([§4.3.4, §4.3.5](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)) | Namespace | Optional; may both be zero | Rejected — wrong scope, may be missing, may be reused |
|
||||
| **CNTLID** | Controller ID, unique only *within* a subsystem ([§4.5.1](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)) | Controller | Mandatory | Rejected — not unique across subsystems (this host's drive reports 0) |
|
||||
|
||||
A note on the PDF: figure numbers in the table of contents of the 2.0e revision are offset from the body captions (e.g. the SN/MN figure is "Figure 128" in the TOC but "Figure 130" in the body), so this document cites section numbers, which are stable.
|
||||
|
||||
## Scope: the counters being segmented are controller-scoped
|
||||
|
||||
SMART / Health Information (LID 02h) is scope **Controller** (mandatory) with an optional namespace view ([2.0e §5.16.1, Get Log Page – Log Page Identifiers](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)). Section 5.16.1.3: "The information provided is over the life of the controller and is retained across power cycles"; hosts request the controller log page with NSID `FFFFFFFFh`/`0h`, the per-namespace view is optional (LPA bit 0), and "the controller log page and namespaces specific log page contain identical information" in 2.0e ([§5.16.1.3](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)). Data Units Written therefore accumulates per controller/subsystem, not per namespace — the identity key must be subsystem-scoped, and SN/MN/SUBNQN are all defined as NVM-subsystem fields ([§5.17.2.1: SN/MN are "the serial number/model number for the NVM subsystem"](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf); the Persistent Event Log repeats that its SN/MN/SUBNQN copies are the same subsystem values).
|
||||
|
||||
NGUID and EUI-64 live in Identify **Namespace**, not the controller structure ([§4.3.4, §4.3.5](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf); libnvme documents them on `struct nvme_id_ns`, while `sn`/`mn`/`fr` live on [`struct nvme_id_ctrl`](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/src/nvme/types.h#L1467-L1477) and `subnqn` on the same structure ([types.h](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/src/nvme/types.h#L1576))). Segmentation keyed on a namespace identifier would split or merge history whenever namespaces are attached, detached, formatted, or recreated, while the DUW counter — the thing being differenced — sails on unchanged. The spec's own namespace-identity guidance (§3.2.1.6: NSIDs "may change across power off conditions"; to detect the same namespace use UUID, NGUID, or EUI-64) addresses a different problem from ours.
|
||||
|
||||
## Stability verdicts (reboots, firmware, replacement)
|
||||
|
||||
1. **Reboots, same drive** — SN, MN, SUBNQN are stable: they are vendor-assigned subsystem fields, and an NQN "is permanent for the lifetime of the host or NVM subsystem" ([§4.5](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)). SMART data "is retained across power cycles" ([§5.16.1.3](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)).
|
||||
2. **Firmware update, same drive** — FR changes by definition (it reports the active revision). The spec guarantees persistence of SMART data across *power cycles*, and says nothing about firmware commits resetting counters — vendor behavior, which is precisely why ADR 0001 keeps the DUW-monotonic rule as an independent boundary. Identity (SN/MN/SUBNQN) is not specified to change with firmware; keeping FR out of the key means a firmware update never quarantines history as a "new drive", and a firmware-induced counter reset is caught by the DUW rule instead.
|
||||
3. **Drive replacement, different model** — every candidate changes.
|
||||
4. **Drive replacement, identical model** — MN unchanged; SN changes *if* vendor serials are unique; SUBNQN changes because both real NQNs (empirically this host's Micron embeds the serial: `nqn.2016-08.com.micron:nvme:nvm-subsystem-sn-233542F44436`) and kernel-generated NQNs (which concatenate SN and MN, [drivers/nvme/host/core.c nvme_init_subnqn](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/drivers/nvme/host/core.c#L3164-L3194)) derive from the serial.
|
||||
5. **NGUID/EUI-64 under namespace churn** — not stable in the needed sense: if the UIDREUSE bit is 0 "a controller **may reuse** a non-zero NGUID/EUI64 value for a new namespace after the original namespace using the value has been deleted" ([§4.5.1](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)); libnvme's own field docs say the values hold only "throughout the life of the namespace", "preserved across namespace and controller operations" ([types.h nguid/eui64](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/src/nvme/types.h#L2648-L2656)).
|
||||
|
||||
## Availability and known pathologies
|
||||
|
||||
- **SN/MN**: mandatory for I/O and Admin controllers ([§5.17.2.1](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)) and exposed by the kernel since 4.5 ([sysfs-nvme: /sys/class/nvme/nvmeX/{model,serial,firmware_rev}](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/Documentation/ABI/stable/sysfs-nvme#L1-L10)). But uniqueness is disclaimed: "The mechanism used by the vendor to assign Serial Number and Model Number values to ensure uniqueness is outside the scope of this specification" ([§4.5.1](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)). Duplicate serials across units are therefore spec-legal.
|
||||
- **SUBNQN**: zero on pre-1.2.1 subsystems ([§5.17.2.1](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf); the Persistent Event Log likewise defines the not-supported case as all bytes cleared to 0h), and some real devices report garbage: the kernel carries `NVME_QUIRK_IGNORE_DEV_SUBNQN` for, among others, Intel P4500/P4600, Intel 760p/Pro 7600p, and a Silicon Motion device ([pci.c quirk table](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/drivers/nvme/host/pci.c#L4121-L4144), [nvme.h flag definition](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/drivers/nvme/host/nvme.h#L103-L106)). This is why Fenris should read the **kernel-exposed** value rather than the raw Identify bytes: when the device NQN is missing, invalid, or quirk-ignored, `nvme_init_subnqn` synthesizes `nqn.2014.08.org.nvmexpress:{vid}{ssvid}{sn}{mn}` — mirroring the spec's own construction for pre-1.2.1 subsystems ([§4.5.1, "NQN Construction for Older NVM Subsystems"](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf), which composes the NQN starting string, VID, SSVID, SN, MN) — so `/sys/.../subsysnqn` is populated on every kernel ≥ 4.8 for every controller ([sysfs-nvme subsysnqn entry, added 4.8](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/Documentation/ABI/stable/sysfs-nvme#L48-L60)). The kernel also uses the NQN as *the* subsystem key when building multipath heads ([core.c: subsystems are matched by `subsys->subnqn`](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/drivers/nvme/host/core.c#L3345-L3350)).
|
||||
- **Virtual controllers / blank serials**: the kernel's own NVMe target (nvmet, the `loop` transport) sets model number to the literal `"Linux"` and generates a **random** serial per subsystem "as our controllers are ephemeral" ([target/core.c nvmet_subsys_alloc](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/drivers/nvme/target/core.c#L1840-L1856), [nvmet.h `NVMET_DEFAULT_CTRL_MODEL`](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/drivers/nvme/target/nvmet.h#L31)); Identify then reports those values verbatim ([target/admin-cmd.c](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/drivers/nvme/target/admin-cmd.c#L670-L677)). For such devices `model|serial` is unstable across target re-creation, while the NQN is the configured subsystem name. No identifier can make an ephemeral virtual drive look like stable hardware; the degraded marker covers it.
|
||||
- **NGUID/EUI-64**: optional — the kernel sysfs attributes are documented as "Hidden if all zeros" ([sysfs-nvme nguid/eui entries](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/Documentation/ABI/stable/sysfs-nvme#L259-L293)), and the spec requires only that *at least one* of EUI64/NGUID/UUID be valid at namespace creation ([§4.5.1](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)).
|
||||
|
||||
## The key, normalization, and failure handling
|
||||
|
||||
1. **Primary key**: `strip(subnqn)` read from the kernel path (via libnvme; see next section). Values are ASCII/UTF-8 with code values 0x20–0x7E, left-justified and space-padded per the spec's string rules ([§1.4.2 ASCII/UTF-8 string conventions](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)); normalize by stripping trailing (and leading) spaces only. **No case folding**: NQNs are compared "as binary strings without any text processing (e.g., case folding)" ([§4.5, NQN processing rules](https://nvmexpress.org/wp-content/uploads/NVM-Express-Base-Specification-2.0e-2024.07.29-Ratified.pdf)), and SN/MN have no canonical case either.
|
||||
2. **Fallback ladder** (defensive; on Linux ≥ 4.8 the kernel fallback already fires before Fenris ever sees an empty value): (a) device-provided SUBNQN; (b) the kernel's composite `nqn.2014.08.org.nvmexpress:{vid}{ssvid}{sn}{mn}` built from Identify — the kernel spells the date with dots and concatenates the raw fixed-width SN and MN, "slightly different from the format specified" in §4.5.1's NQN construction "for historic reasons" ([core.c nvme_init_subnqn](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/drivers/nvme/host/core.c#L3164-L3194)); Fenris's fallback reproduces the kernel spelling so a fallback-built key equals the sysfs value byte-for-byte; (c) `strip(mn)|strip(sn)`. Every rung changes on drive replacement and survives firmware updates; the ladder exists so a missing rung never yields a blank key from a perfectly good physical drive.
|
||||
3. **Blank key** (all components empty — e.g. a virtual controller reporting nothing): record `identity_key = ""` plus a `degraded` marker and segment by DUW monotonicity alone; surface as a fact, never fabricate uniqueness (matches ADR 0005's "facts, not alerts" posture).
|
||||
4. **Duplicate keys across physical units** are undetectable by construction when both units report identical SN/MN/SUBNQN. The mitigation already exists in ADR 0001: a replacement drive almost certainly reports a **lower** DUW than the accumulated history, and any DUW decrease forces a segment boundary regardless of identity. Write deltas remain quarantined even though identity cannot distinguish the units.
|
||||
5. **Recorded metadata per segment**: normalized `subnqn`, `sn`, `mn`, `fr`, plus `transport` — diagnostics for humans, not key components.
|
||||
|
||||
## How the collector obtains the key via libnvme
|
||||
|
||||
Three concrete paths, all first-party:
|
||||
|
||||
1. **libnvme Python bindings** (`from libnvme import nvme`, official SWIG bindings, ["python bindings for libnvme"](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/pyproject.toml#L1-L8)). `nvme.root()` scans sysfs ([nvme.i: nvme_root() calls nvme_scan](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/libnvme/nvme.i#L492-L501)); controller objects expose `model`, `serial`, `firmware`, `subsysnqn`, `name`, `sysfs_dir` as attributes ([nvme.i struct nvme_ctrl attributes](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/libnvme/nvme.i#L402-L424)); namespace objects expose `nsid`, `nguid`, `eui64`, `uuid` ([nvme.i struct nvme_ns](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/libnvme/nvme.i#L481-L490)). These getters read sysfs via `nvme_get_ctrl_attr` ([tree.c populates ctrl fields from sysfs attributes](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/src/nvme/tree.c#L2085-L2097)), and `__nvme_get_attr` **strips the trailing newline and trailing spaces and returns NULL when the result is empty** ([linux.c L526–552](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/src/nvme/linux.c#L526-L552)) — i.e., the bindings deliver exactly the normalization this decision requires, and a blank field arrives as `None`.
|
||||
2. **Admin passthrough** for raw Identify: `nvme_ctrl_identify(c, &id)` fills `struct nvme_id_ctrl` ([man page: "Issues an 'identify controller' command"](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/doc/man/nvme_ctrl_identify.2#L13-L17)), whose `sn[20]`/`mn[40]`/`fr[8]` and `subnqn[256]` members are documented in the [nvme_id_ctrl(2) man page](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/doc/man/nvme_id_ctrl.2#L5-L15) ([subnqn member](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/doc/man/nvme_id_ctrl.2#L616-L617)); namespace identifiers come from `nvme_ns_identify` filling `struct nvme_id_ns` ([tree.h nvme_ns_identify](https://github.com/linux-nvme/libnvme/blob/ad61ac8a319ad0823c1c9861eecbf66125f8b9a1/src/nvme/tree.h#L823-L832)). Here the collector must strip trailing spaces itself.
|
||||
3. **nvme-cli JSON** (built on libnvme): `nvme id-ctrl -o json` emits `sn`, `mn`, `fr`, `cntlid` with strings copied verbatim *including padding* ([nvme-print-json.c L409–L422](https://github.com/linux-nvme/nvme-cli/blob/c8ec7e41f3738b20828849228a549f4ed1d03fc0/src/nvme-print-json.c#L409-L422)) plus `subnqn`, which is included only when non-empty ([L522–L523](https://github.com/linux-nvme/nvme-cli/blob/c8ec7e41f3738b20828849228a549f4ed1d03fc0/src/nvme-print-json.c#L522-L523)); `nvme list -o json` emits per device `DevicePath`, `Firmware`, `ModelNumber`, `SerialNumber` ([v2.16 json_list_item_obj](https://github.com/linux-nvme/nvme-cli/blob/faf7326a2997dea91687fd3daa17fc405910a4c1/nvme-print-json.c#L4669-L4693)). So stripping trailing spaces is the collector's job in both nvme-cli paths.
|
||||
|
||||
What libnvme/nvme-cli offer beyond the current `smartctl -j` collector: `subnqn` (the chosen key), `cntlid`, `vid`/`ssvid`, and the namespace identifiers — smartmontools' NVMe JSON device section carries `model_name`, `serial_number`, `firmware_version` (from `id_ctrl.mn/sn/fr`, [nvmeprint.cpp print_drive_info](https://github.com/smartmontools/smartmontools/blob/9f83095a631ff71df44f8065c7a4a00134d3d404/smartmontools/nvmeprint.cpp#L108-L121)) and no NQN. smartmontools also trims the strings it copies ([utility.cpp format_char_array strips leading/trailing spaces](https://github.com/smartmontools/smartmontools/blob/9f83095a631ff71df44f8065c7a4a00134d3d404/smartmontools/utility.cpp#L692-L708)), so normalization is compatible across both collectors.
|
||||
|
||||
## Empirical check (one real drive, sysfs only)
|
||||
|
||||
```console
|
||||
$ cat /sys/class/nvme/nvme0/{model,serial,firmware_rev,subsysnqn}
|
||||
Micron_2400_MTFDKBA512QFM<spaces to 40> # sysfs preserves Identify padding
|
||||
233542F44436<spaces to 20>
|
||||
V3MA001<space> # FR padded to 8
|
||||
nqn.2016-08.com.micron:nvme:nvm-subsystem-sn-233542F44436
|
||||
$ cat /sys/block/nvme0n1/{nguid,eui,wwid}
|
||||
00000000-0000-0001-00a0-752342f44436
|
||||
00 a0 75 01 42 f4 44 36
|
||||
eui.000000000000000100a0752342f44436
|
||||
```
|
||||
|
||||
Confirms: raw sysfs keeps the spec's trailing-space padding (strip before storing); the vendor NQN embeds the serial; NGUID/EUI-64 are present here but are per-namespace; `/sys/class/nvme-subsystem/nvme-subsys0/{model,serial,firmware_rev,subsysnqn}` carries the same subsystem-level values ([sysfs-nvme nvme-subsystem entries, added 4.15](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/Documentation/ABI/stable/sysfs-nvme#L425-L434)). The kernel's own namespace `wwid` uses the priority ladder `uuid.{UUID}` → `eui.{NGUID}` → `eui.{EUI64}` → `nvme.{VID}-{SERIAL}-{MODEL}-{NSID}` ([sysfs-nvme wwid](https://github.com/torvalds/linux/blob/cee9395acd8043be0644b25c34bfa86623f2b935/Documentation/ABI/stable/sysfs-nvme#L277-L285)) — a first-party precedent for a serial+model fallback when better identifiers are absent.
|
||||
|
||||
## What ADR 0001's segment and migration rules must assume
|
||||
|
||||
1. **Legacy `history.jsonl` cannot carry a full identity.** The legacy collector records `device` (`/dev/nvme0`), `model` (smartctl `model_name`), `capacity_bytes`, and the counters — **no serial, no NQN, no firmware** ([fenris.py sample(): the identity-adjacent fields are `device` and `model` only](https://git.bongbetic.com/xavierk/Fenris/src/commit/d45d931a0deaaa4ff8c8ba0ff2454c0d2a211ef9/fenris.py#L73-L76)). Migration may therefore only import legacy samples under a **model-scoped legacy identity**, explicitly labeled incomplete. It cannot prove the legacy samples came from the current drive, cannot detect a same-model swap that happened before migration, and must not backfill serials retroactively.
|
||||
2. **The first new-version collection run starts a new controller segment** for the monitored drive, recording full `identity_key` plus `sn`/`mn`/`fr`/`subnqn` metadata, because the legacy and new identities are not comparable — the identity change is an epistemic boundary, not a detected drive swap. Imported legacy rows keep their legacy segment; the DUW-monotonic rule already governs deltas within it.
|
||||
3. **Two independent segmentation axes stay independent**: identity-key change ⇒ boundary (drive replacement); DUW decrease ⇒ boundary (counter reset, firmware quirk, or same-identity replacement). Neither implies the other; ADR 0001's `controller_segments` wording ("boundaries where controller identity changes or DUW decreases") already encodes this, and this research fixes "controller identity" to mean the normalized kernel-exposed subsystem NQN with the fallback ladder above.
|
||||
4. **The key is a string with rules, not just a field choice**: normalization (space-stripping, no case folding) must be specified once and applied at write time, or the same drive could split its own history across a collector implementation change (smartctl, nvme-cli, and libnvme deliver differently-trimmed values, as shown above).
|
||||
|
||||
## Newly surfaced questions
|
||||
|
||||
- Should the store record `vid`/`ssvid`/`cntlid` alongside segment metadata now (cheap) for future composite needs, or keep segments minimal?
|
||||
- Should a detected blank/degraded identity degrade the projection confidence category (it weakens the "identity/counter discontinuity" axis of the Unavailable state)?
|
||||
- Does the collector pin to one acquisition path (libnvme bindings vs `nvme list -o json` vs continued `smartctl -j` plus a sysfs read for `subnqn`) — an operational choice this research does not settle?
|
||||
@@ -1,239 +0,0 @@
|
||||
# Research: deb + rpm packaging toolchain for bundled-venv builds
|
||||
|
||||
Issue: #34 (parent plan: #33) — branch `research/toolchain`
|
||||
Date: 2026-09-03 · target: Fenris 0.3.0, x86_64, Debian 12 / Ubuntu 22.04+24.04 / Fedora 40+
|
||||
|
||||
## TL;DR
|
||||
|
||||
**Recommended: nfpm** with a build script that stages a `--copies` venv at
|
||||
`/opt/fenris`. One `nfpm.yaml` is the single source of truth for both formats;
|
||||
`nfpm pkg -p deb && nfpm pkg -p rpm` (one invocation per format — `-p` takes a
|
||||
single string, verified in `internal/cmd/package.go`). Actively maintained
|
||||
(releases v2.47.0, 2026-06-20; repo pushed 2026-08-31). Runner-up: fpm (active,
|
||||
v1.18.0 gem 2026-08-26), but its "config" is a long CLI invocation per format —
|
||||
the single source of truth degrades into a shell script. dh-virtualenv is
|
||||
deb-only and its last upstream release is 2020-10 (effectively dormant);
|
||||
rpmbuild spec is rpm-only and cannot share file lists with a deb build without
|
||||
external generation.
|
||||
|
||||
## Constraint evidence: distro textual is unusable (mostly)
|
||||
|
||||
| Distro | python3-textual | Source |
|
||||
|---|---|---|
|
||||
| Debian 12 (bookworm) | **0.1.13-1** | https://packages.debian.org/bookworm/python3-textual |
|
||||
| Ubuntu 22.04 (jammy) | **0.1.13-1** | https://packages.ubuntu.com/jammy/python3-textual |
|
||||
| Ubuntu 24.04 (noble) | **0.1.13-1** | https://packages.ubuntu.com/noble/python3-textual |
|
||||
| Fedora 40 | 0.48.1 | https://src.fedoraproject.org/rpms/python-textual (f40 spec) |
|
||||
| Fedora 41 / 42 / 43 | 0.69.0 / 1.0.0 / 4.0.0 | same spec, f41–f43 branches |
|
||||
|
||||
Fenris declares `textual>=0.40.0` (pyproject) but pins `textual==8.2.8`
|
||||
(requirements.txt). Debian 12 + Ubuntu 22.04/24.04 are ~1 major era behind even
|
||||
the *floor*; Fedora 40 technically meets `>=0.40` but not the pin. Verdict
|
||||
unchanged: **vendor deps inside the package for all targets**; per-format
|
||||
`depends:` only on `python3 (>= 3.9)`, `smartmontools`, `systemd`.
|
||||
|
||||
## Tool-by-tool
|
||||
|
||||
### 1. nfpm (goreleaser) — RECOMMENDED
|
||||
|
||||
- **Route:** Makefile target builds staging tree → one `nfpm.yaml` →
|
||||
`nfpm package -p deb` + `nfpm package -p rpm`. (Goreleaser release pipeline
|
||||
can wrap both later.)
|
||||
- **Shared assets:** version, description, maintainer, `depends`,
|
||||
`contents:` file list, `scripts:` all live once in `nfpm.yaml`; per-format
|
||||
deltas via `overrides: { deb: ..., rpm: ... }` and `packager:`-scoped
|
||||
content entries (https://nfpm.goreleaser.com/configuration/, source
|
||||
`www/content/docs/configuration.md`).
|
||||
- **Prerequisites:** single static Go binary (`go install
|
||||
github.com/goreleaser/nfpm/v2/cmd/nfpm@latest`, Homebrew, or release
|
||||
tarball — https://nfpm.goreleaser.com/install/). No toolchain per distro,
|
||||
no root, no containers required (build same tree for both formats).
|
||||
- **Venv → file list:** stage with `python3 -m venv --copies staging/opt/fenris
|
||||
&& staging/opt/fenris/bin/pip install dist/fenris-*.whl`; map in one entry:
|
||||
`contents: [{ src: staging/opt/fenris/, dst: /opt/fenris, type: tree }]`.
|
||||
Shebangs point at fixed absolute `/opt/fenris/bin/python` → no relocation
|
||||
issues. `--copies` avoids symlink-to-/usr breakage. Config file →
|
||||
`type: config|noreplace` (=%config(noreplace) on rpm, conffile semantics on
|
||||
deb). `/var/lib/fenris` store → `type: ghost` (rpm: owned-but-not-packed;
|
||||
deb: ignored → create in `postinstall` script instead).
|
||||
- **Systemd/polkit/libexec:** plain `contents:` entries —
|
||||
`/usr/lib/systemd/system/fenris-collect.{service,timer}` (or
|
||||
`/etc/systemd/system` to match current Makefile), polkit action at
|
||||
`/usr/share/polkit-1/actions/`, helpers under `/usr/libexec/fenris/`.
|
||||
`scripts:` supports `postinstall` (deb maintainer script / rpm scriptlet) —
|
||||
run `systemctl daemon-reload`, create `/var/lib/fenris` root:fenris 2750,
|
||||
group creation.
|
||||
- **Upgrade/removal:** deb — dpkg replaces all non-conffile files, conffile
|
||||
prompts/preserves (`.dpkg-new`) per Debian Policy ch-files
|
||||
(https://www.debian.org/doc/debian-policy/ch-files.html); removal keeps
|
||||
conffiles + unowned store; purge cleans. rpm — `rpm -U` replaces,
|
||||
`%config(noreplace)` keeps local edits as `.rpmnew`; only owned dirs are
|
||||
removed on erase (nfpm `type: dir` exists precisely to claim ownership —
|
||||
docs warn not to claim distro-owned dirs).
|
||||
- **Maintenance:** very active. goreleaser/nfpm, 2.6k stars, last push
|
||||
2026-08-31, v2.47.0 released 2026-06-20 (GitHub API).
|
||||
|
||||
### 2. fpm — viable, weaker single-source-of-truth
|
||||
|
||||
- **Route:** staging tree (same as above) then
|
||||
`fpm -s dir -t deb ... staging/=/ ; fpm -s dir -t rpm ...`.
|
||||
- **Shared assets:** none declarative — everything is CLI flags
|
||||
(`-n`, `-v`, `--config-files`, `--deb-systemd`, `--directories`,
|
||||
`--after-install`, `--rpm-posttrans`, …). Flag list:
|
||||
https://fpm.readthedocs.io/en/latest/cli-reference.html. The two
|
||||
invocations *will* drift unless wrapped in a Makefile that shares variables;
|
||||
the "single source" is then a shell script, not a checked declarative file.
|
||||
(`--deb-systemd` exists; no rpm-native unit macro — you hand it the unit
|
||||
file plus `--rpm-posttrans` for daemon-reload.)
|
||||
- **Prerequisites:** Ruby + gem (`gem install fpm`) or distro package;
|
||||
building rpm side needs `rpmbuild` present for some features.
|
||||
- **Venv → file list:** `-s dir` maps a directory into the package verbatim —
|
||||
same staging-tree trick as nfpm. `--config-files /etc/fenris` marks
|
||||
conffiles (deb) / %config (rpm).
|
||||
- **Upgrade/removal:** identical downstream semantics to nfpm (native dpkg/rpm
|
||||
behavior); differences are only in how metadata/scripts land in the
|
||||
package.
|
||||
- **Maintenance:** active — releases v1.16.0 (2024-12), v1.17.0 (2025-10),
|
||||
v1.18.0 (2026-08-26); gem 1.18.0 on rubygems; ~11.5k stars. But docs are
|
||||
openly "work in progress" (https://fpm.readthedocs.io/en/latest/).
|
||||
|
||||
### 3. dh-virtualenv (Spotify) — deb-only, dorms
|
||||
|
||||
- **Route:** debhelper add-on: `debian/rules` with
|
||||
`dh $@ --with python-virtualenv --buildsystem=python_distutils`;
|
||||
produces a .deb containing venv at `/opt/venvs/<package>`
|
||||
(`DH_VIRTUALENV_INSTALL_ROOT` overridable, `--builtin-venv` for `python -m
|
||||
venv`). Docs: repo `doc/usage.rst`, `doc/tutorial.rst`
|
||||
(https://github.com/spotify/dh-virtualenv).
|
||||
- **Shared assets:** none with rpm — it cannot emit .rpm at all. Would still
|
||||
need a second toolchain for Fedora → fails the criterion outright.
|
||||
- **Prerequisites:** `build-essential debhelper devscripts equivs` +
|
||||
`dh-virtualenv` (tutorial.rst); Debian 12 still ships it as
|
||||
`dh-virtualenv 1.2.2-1.3` (https://packages.debian.org/bookworm/dh-virtualenv).
|
||||
- **Venv → file list:** automatic — it builds the venv during the debhelper
|
||||
sequence and rewrites shebangs; the .deb owns the whole venv tree. Least
|
||||
manual work of all four, for deb alone.
|
||||
- **Upgrade/removal:** standard dpkg; whole venv tree is package-owned, so
|
||||
`apt remove` deletes it cleanly; `--pypi-url`/requirements handled by tool.
|
||||
- **Maintenance:** last upstream release **1.2.2, 2020-10-22** (GitHub tag);
|
||||
repo last pushed 2024-04-27, RTD docs 404. Effectively dormant upstream —
|
||||
fine via Debian's own packaging, but risky as strategic dependency.
|
||||
- **Bonus fact:** PyPI `dh-virtualenv` project now returns 404 — install only
|
||||
from Debian repo / git.
|
||||
|
||||
### 4. rpmbuild spec + vendored venv — rpm-native, no deb
|
||||
|
||||
- **Route:** hand-written `fenris.spec`: `%install` stage builds venv into
|
||||
`%{buildroot}/opt/fenris`, `%files` lists it plus units/polkit/libexec,
|
||||
`%ghost %attr(2750,root,fenris) /var/lib/fenris`, `%config(noreplace)` for
|
||||
`/etc/fenris`, `systemd_post/preun` macros for the timer. Reference style:
|
||||
https://docs.fedoraproject.org/en-US/packaging-guidelines/.
|
||||
- **Shared assets:** the spec is a second, parallel description of the same
|
||||
file list — nothing is shared with any deb build without generating one
|
||||
side from the other (e.g. generate spec + debian/control from a manifest).
|
||||
Worst single-source-of-truth score.
|
||||
- **Prerequisites:** `rpm-build`, mock/koji for cleanroots; Fedora toolchain
|
||||
knowledge; per-distro `Release:`/dist tag handling.
|
||||
- **Venv → file list:** `%files` line `%{buildroot}/opt/fenris/...` — venv
|
||||
becomes ordinary payload; shebangs already absolute.
|
||||
- **Upgrade/removal:** canonical rpm semantics (same as above) plus real
|
||||
systemd scriptlet macros — the *best-behaved* rpm integration of the four,
|
||||
at the cost of hand-maintained spec.
|
||||
- **Maintenance:** rpmbuild itself is maintained forever (part of RPM), but
|
||||
*your* spec is 100% hand-maintained duplication.
|
||||
|
||||
## Comparison matrix
|
||||
|
||||
| Criterion | nfpm | fpm | dh-virtualenv | rpmbuild spec |
|
||||
|---|---|---|---|---|
|
||||
| deb + rpm from one config | ✅ one YAML (2 invocations) | ⚠️ flags per invocation | ❌ deb only | ❌ rpm only |
|
||||
| File list shared across formats | ✅ `contents:` | ⚠️ per-invocation args | n/a | ❌ |
|
||||
| Vendored venv supported | ✅ staging `type: tree` | ✅ `-s dir` | ✅✅ automatic (deb) | ✅ `%files` |
|
||||
| conffile / %config(noreplace) | ✅ `type: config\|noreplace` | ✅ `--config-files` | ✅ (debhelper) | ✅ `%config(noreplace)` |
|
||||
| ghost store dir | ✅ `type: ghost` | ⚠️ `--rpm-ghost`? (no deb equiv) | ❌ | ✅ `%ghost` |
|
||||
| systemd scriptlets | ✅ `scripts:` + macros? (plain scripts) | ✅ `--deb-systemd`, `--rpm-posttrans` | ✅ (deb) | ✅✅ native macros |
|
||||
| Prereqs on build host | Go binary (or brew/apt tarball) | Ruby gem | debhelper stack | rpm-build + mock |
|
||||
| Maintenance (2026) | 🟢 active (v2.47.0) | 🟢 active (v1.18.0) | 🔴 dormant since 2020 (Debian carries it) | 🟢 tool yes / 🔴 your spec |
|
||||
| Risk | young-ish config schema churn | docs thin | dead upstream | duplication forever |
|
||||
|
||||
## Proposed pipeline (sketch)
|
||||
|
||||
```make
|
||||
# Makefile additions (build only — install target stays for source installs)
|
||||
stage: dist/fenris-*.whl
|
||||
rm -rf build/stage
|
||||
python3 -m venv --copies build/stage/opt/fenris
|
||||
build/stage/opt/fenris/bin/pip install --no-compile dist/fenris-*.whl
|
||||
install -D -m 0755 scripts/fenris build/stage/usr/bin/fenris
|
||||
install -D -m 0755 src/fenris/monitor.py build/stage/usr/libexec/fenris/fenris-monitor
|
||||
install -D -m 0755 src/fenris/collect.py build/stage/usr/libexec/fenris/fenris-collect
|
||||
install -D -m 0644 units/fenris-collect.timer build/stage/usr/lib/systemd/system/fenris-collect.timer
|
||||
install -D -m 0644 units/fenris-collect.service build/stage/usr/lib/systemd/system/fenris-collect.service
|
||||
install -D -m 0644 polkit/com.bongbetic.fenris.monitor.policy \
|
||||
build/stage/usr/share/polkit-1/actions/com.bongbetic.fenris.monitor.policy
|
||||
|
||||
package-deb package-rpm: stage
|
||||
nfpm pkg -f packaging/nfpm.yaml -p deb -t dist/
|
||||
nfpm pkg -f packaging/nfpm.yaml -p rpm -t dist/
|
||||
```
|
||||
|
||||
```yaml
|
||||
# packaging/nfpm.yaml (excerpt)
|
||||
name: fenris
|
||||
arch: amd64
|
||||
platform: linux
|
||||
version: ${VERSION} # env expansion, documented feature
|
||||
maintainer: Fenris Maintainers <ops@bongbetic.com>
|
||||
description: SMART drive observation daemon with persistent TUI
|
||||
homepage: https://git.bongbetic.com/xavierk/Fenris
|
||||
depends: [smartmontools]
|
||||
contents:
|
||||
- src: build/stage/ # everything above
|
||||
dst: /
|
||||
type: tree
|
||||
- dst: /etc/fenris # config dir; ship fenris.conf as config|noreplace
|
||||
type: dir
|
||||
- src: packaging/fenris.conf
|
||||
dst: /etc/fenris/fenris.conf
|
||||
type: config|noreplace
|
||||
- dst: /var/lib/fenris # rpm: %ghost ownership; deb: create in postinst
|
||||
type: ghost
|
||||
scripts:
|
||||
postinstall: packaging/postinst.sh # groupadd fenris; install -d -o root -g fenris -m 2750 /var/lib/fenris; systemctl daemon-reload (units shipped dormant)
|
||||
preremove: packaging/prerm.sh # stop timer if running
|
||||
overrides:
|
||||
deb:
|
||||
depends: [python3 (>= 3.9), smartmontools]
|
||||
rpm:
|
||||
depends: [python3 >= 3.9, smartmontools]
|
||||
```
|
||||
|
||||
## Recommendation
|
||||
|
||||
Adopt **nfpm + staged `--copies` venv**: closest to single source of truth
|
||||
(one YAML for both formats), smallest prerequisite surface (one static binary),
|
||||
actively maintained, and every Fenris constraint (units, polkit, libexec,
|
||||
`/etc/fenris` conffile, `/var/lib/fenris` ghost/store) has a first-class
|
||||
mapping. Keep fpm as documented fallback (identical staging tree, works
|
||||
anywhere Ruby exists). Do not build the release pipeline on dh-virtualenv
|
||||
(dormant, deb-only) or on a hand-maintained spec file (duplication, deb side
|
||||
unaddressed).
|
||||
|
||||
## Sources
|
||||
|
||||
- nfpm config reference: https://nfpm.goreleaser.com/configuration/ (source:
|
||||
goreleaser/nfpm `www/content/docs/configuration.md`, accessed 2026-09-03)
|
||||
- nfpm CLI (single `-p`): goreleaser/nfpm `internal/cmd/package.go`
|
||||
- nfpm releases/status: GitHub API, repo pushed 2026-08-31, v2.47.0 2026-06-20
|
||||
- fpm README + CLI reference: https://github.com/jordansissel/fpm,
|
||||
https://fpm.readthedocs.io/en/latest/cli-reference.html; releases v1.18.0
|
||||
(2026-08-26), gem 1.18.0
|
||||
- dh-virtualenv docs: `doc/usage.rst`, `doc/tutorial.rst` @ master; tag 1.2.2
|
||||
dated 2020-10-22 (GitHub commits API); PyPI project 404;
|
||||
Debian 12 package 1.2.2-1.3 (packages.debian.org)
|
||||
- Distro textual versions: packages.debian.org, packages.ubuntu.com,
|
||||
src.fedoraproject.org `python-textual.spec` f40–f43
|
||||
- Upgrade semantics: Debian Policy ch-files
|
||||
(https://www.debian.org/doc/debian-policy/ch-files.html); Fedora packaging
|
||||
guidelines (https://docs.fedoraproject.org/en-US/packaging-guidelines/);
|
||||
RPM directive behavior quoted in nfpm config docs (%ghost, %config(noreplace))
|
||||
@@ -1,116 +0,0 @@
|
||||
# Research: Gitea 1.27 Debian + RPM package registry feasibility
|
||||
|
||||
Issue: [Fenris deb + rpm release plan](https://git.bongbetic.com/xavierk/Fenris/issues/33) →
|
||||
[Research: Gitea 1.27 Debian + RPM package registry feasibility](https://git.bongbetic.com/xavierk/Fenris/issues/35)
|
||||
Verified 2026-09-03 against live instance `https://git.bongbetic.com` (reports `1.27.1` via `/api/v1/version`)
|
||||
and primary sources: docs.gitea.com 1.27 Debian/RPM registry pages and Gitea `v1.27.1` source (go-gitea/gitea tag).
|
||||
|
||||
**Verdict: feasible.** Every publish/consume path tested live with throwaway packages `fenris-regtest` (all deleted afterward; package list verified empty).
|
||||
|
||||
## 1. Publish paths (verified live, HTTP 201)
|
||||
|
||||
### Debian (`.deb`)
|
||||
|
||||
```bash
|
||||
curl --user xavierk:$TOKEN --upload-file fenris_0.3.0_amd64.deb \
|
||||
"https://git.bongbetic.com/api/packages/xavierk/debian/pool/{distribution}/{component}/upload"
|
||||
```
|
||||
|
||||
- `distribution` and `component` are free-form path segments chosen at upload time (e.g. `bookworm/main`, `noble/main`). Gitea derives apt suites from what was uploaded — verified: same .deb published to `pool/bookworm/main` and `pool/noble/main` (both 201), both then served in `dists/bookworm/` and `dists/noble/` with correct `Suite:`/`Codename:` headers.
|
||||
- Republish of identical name+version+distribution+component+architecture → **409 Conflict** (verified). Must delete first.
|
||||
|
||||
### RPM (`.rpm`)
|
||||
|
||||
```bash
|
||||
# no group (flat repo)
|
||||
curl --user xavierk:$TOKEN --upload-file fenris-0.3.0-1.el9.x86_64.rpm \
|
||||
"https://git.bongbetic.com/api/packages/xavierk/rpm/upload"
|
||||
# with group (distro tag, nestable)
|
||||
curl --user xavierk:$TOKEN --upload-file fenris-0.3.0-1.fc40.x86_64.rpm \
|
||||
"https://git.bongbetic.com/api/packages/xavierk/rpm/el9/upload" # e.g. el9, rocky/el9, fc40
|
||||
```
|
||||
|
||||
- Group = free-form nesting used to partition repos per distro/track. Verified: publish to root group and `el9` group (both 201), duplicate → 409.
|
||||
- Owner can be the user (`xavierk`) or an org; packages under a public owner are readable anonymously (verified: metadata fetches without auth succeeded).
|
||||
|
||||
## 2. Consumer setup (exact commands)
|
||||
|
||||
### apt clients
|
||||
|
||||
```bash
|
||||
sudo mkdir -p /etc/apt/keyrings
|
||||
sudo curl -o /etc/apt/keyrings/gitea-xavierk.asc \
|
||||
https://git.bongbetic.com/api/packages/xavierk/debian/repository.key
|
||||
echo "deb [signed-by=/etc/apt/keyrings/gitea-xavierk.asc] https://git.bongbetic.com/api/packages/xavierk/debian bookworm main" \
|
||||
| sudo tee /etc/apt/sources.list.d/gitea.list # one line per distribution
|
||||
sudo apt update
|
||||
apt install fenris # or fenris=0.3.0
|
||||
# private owner variant: https://{user}:{token}@git.bongbetic.com/api/packages/... in the URL
|
||||
```
|
||||
|
||||
### dnf clients
|
||||
|
||||
```bash
|
||||
sudo dnf config-manager --add-repo https://git.bongbetic.com/api/packages/xavierk/rpm/el9.repo
|
||||
# private owner: add user:token into the baseurl inside /etc/yum.repos.d/gitea-xavierk-el9.repo afterwards
|
||||
sudo dnf install fenris # or fenris-0.3.0
|
||||
```
|
||||
|
||||
The served `.repo` (verified live) sets `gpgcheck=1` and points `gpgkey` at `…/rpm/repository.key`, so `dnf` auto-imports on first use.
|
||||
|
||||
## 3. Metadata signing: native, not passthrough
|
||||
|
||||
Gitea **signs generated metadata itself** with per-instance auto-generated PGP keys. Client-side signing config is limited to trusting the served keys.
|
||||
|
||||
- Debian: `dists/{suite}/InRelease` is clearsigned; `Release.gpg` detached sig also served. Key (RSA) fetched from `…/debian/repository.key`, uid literally `(Automatically generated Debian Registry Key; created …)`.
|
||||
- RPM: `repodata/repomd.xml.asc` detached ASCII-armored signature, uid `(RPM Registry)`. Key from `…/rpm/repository.key`.
|
||||
- Both verified with `gpg --verify` → **Good signature** (keys are self-generated; the "not certified" warning is expected and handled by the signed-by/keyring flow above).
|
||||
- The apt `Release` also advertises `Acquire-By-Hash: yes` with MD5/SHA1/SHA256/SHA512 indexes of `Packages`/`.gz`/`.xz` (verified live). RPM repomd carries sha256 checksums for `primary/filelists/other.xml.gz`.
|
||||
|
||||
There is **no bring-your-own-signing-key config** for these registries in 1.27 — trust anchor is the instance's auto keys. For Fenris this is acceptable; TOFU over TLS via the key URLs above.
|
||||
|
||||
## 4. Multi-distro metadata
|
||||
|
||||
- Debian: distributions/suites are implicit — whatever `{distribution}` path segments appear on upload become `dists/{distribution}/` trees with `Suite:`/`Codename:` set to the segment. No server-side list to maintain; adding a new distro = upload with new segment + one more `deb …` sources line. Components likewise (`main`, etc.). Architectures come from each `.deb`'s control stanza (index served as `dists/{dist}/{component}/binary-{arch}/Packages`).
|
||||
- RPM: same via `{group}` path segments (`el9`, `rocky/el9`, …); each group gets its own `repodata/`. No `basearch` filtering — clients pick the group; Gitea publishes whatever RPM arch was uploaded.
|
||||
|
||||
## 5. Version retention
|
||||
|
||||
- Default: **all versions retained indefinitely**; nothing auto-deletes. Old versions stay installable (`apt install fenris=0.2.9`, `dnf install fenris-0.2.9`).
|
||||
- Republishing an existing name+version (deb: same dist/component/arch; rpm: same file name in group) → 409; overwrite requires delete-then-upload.
|
||||
- Optional cleanup rules exist (per owner + package type): `KeepCount`, `KeepPattern`, `RemoveDays`, `RemovePattern`, `MatchFullName` (source: `models/packages/package_cleanup_rule.go`, executed by scheduled `CleanupTask` in `services/packages/cleanup/cleanup.go`). In 1.27.1 they are configurable **only in the web UI** (owner → Packages → Cleanup Rules); no v1 REST route (verified by route table grep of `routers/api/v1/api.go` — probes of `/api/v1/packages/{owner}/cleanuprules…` return 404/409-style errors).
|
||||
- Deletes: format-specific `DELETE …/debian/pool/{dist}/{component}/{name}/{version}/{arch}` and `DELETE …/rpm/{group}/package/{name}/{version}/{arch}` (both verified, 204). Deleting last file removes the version. Generic fallback: `DELETE /api/v1/packages/{owner}/{type}/{name}/{version}`.
|
||||
|
||||
## 6. Release attachment: not supported
|
||||
|
||||
Gitea 1.27.1 has **no package↔release linkage**. Release assets (`…/releases/{id}/assets`) are standalone file uploads; the package model has no release field and no route links them (verified against `v1.27.1` source: `routers/api/v1/repo/release_attachment.go`, `models/packages/`). Options for Fenris releases:
|
||||
|
||||
1. Publish `.deb`/`.rpm` to the registry (real apt/dnf install UX) and reference the registry URLs in release notes.
|
||||
2. Additionally upload tarballs/SHA256SUMS as plain release attachments.
|
||||
3. Generic registry (`PUT /api/packages/{owner}/generic/{name}/{version}/{filename}`) if an untyped artifact store is needed.
|
||||
|
||||
## 7. Caveats for the release plan
|
||||
|
||||
- Owner choice matters: publish under an **org** (e.g. `fenris`) if multiple maintainers need write; `xavierk` user owner works today (token owner is admin).
|
||||
- Metadata access follows owner visibility — public owner → anonymous consumers, no token in URLs (current state, verified). Keep owner public for frictionless installs, or embed `user:token` in sources/baseurl.
|
||||
- apt distro naming should match OS release names (`bookworm`, `trixie`, `noble`) purely for client convention; server accepts anything.
|
||||
- RPM groups should mirror `$distver` (e.g. `el9`, `fc40`) so `.repo` selection is obvious per target.
|
||||
|
||||
## 8. Test log (live, 2026-09-03)
|
||||
|
||||
| Step | Result |
|
||||
|---|---|
|
||||
| `PUT debian/pool/bookworm/main/upload` | 201 |
|
||||
| `PUT debian/pool/noble/main/upload` (multi-dist) | 201 |
|
||||
| `PUT debian` duplicate | 409 (expected) |
|
||||
| `PUT rpm/upload` (no group) | 201 |
|
||||
| `PUT rpm/el9/upload` (group) | 201 |
|
||||
| `PUT rpm` duplicate | 409 (expected) |
|
||||
| `GET debian/repository.key` / `rpm/repository.key` | PGP public keys (200) |
|
||||
| `GET dists/bookworm/{Release,InRelease,Packages}` | correct; `gpg --verify` Good signature |
|
||||
| `GET rpm{,/el9}/repodata/repomd.xml{,.asc}` | 200; Good signature |
|
||||
| `GET rpm{,/el9}.repo` | generated repo files with `gpgcheck=1` |
|
||||
| Cleanup-rules REST probes | 404 (not in v1 API — UI only) |
|
||||
| `DELETE` all four test entries | 204 ×4; package list then empty |
|
||||
|
||||
Sources: [docs.gitea.com 1.27 Debian registry](https://docs.gitea.com/1.27/usage/packages/debian), [docs.gitea.com 1.27 RPM registry](https://docs.gitea.com/1.27/usage/packages/rpm), Gitea source tag `v1.27.1` (`routers/api/v1/api.go`, `models/packages/package_cleanup_rule.go`, `services/packages/cleanup/cleanup.go`), live instance `git.bongbetic.com`.
|
||||
@@ -1,73 +0,0 @@
|
||||
# Research: OBS as an alternative build + distribution route
|
||||
|
||||
Resolves [Research: OBS as alternative build + distribution route](https://git.bongbetic.com/xavierk/Fenris/issues/36) on the [Wayfinder map](https://git.bongbetic.com/xavierk/Fenris/issues/33).
|
||||
|
||||
- Date: 2026-09-03
|
||||
- Verdict: **Reject OBS now; ship via the self-hosted Gitea 1.27.1 registry** (deb + rpm), and revisit OBS only if publishing reach becomes a goal.
|
||||
|
||||
## Question
|
||||
|
||||
Evaluate openSUSE Open Build Service (OBS) as the build + distribution route — deb build quality, vendoring `textual>=0.40` via source services (offline sandbox), signing, publishing reach, account/maintenance cost, build latency — against the Gitea registry on our matrix: Debian 12, Ubuntu 22.04/24.04, Fedora 40+, x86_64.
|
||||
|
||||
## Findings
|
||||
|
||||
### 1. deb build support quality — real, with quirks
|
||||
|
||||
- OBS builds deb via the classic recipe trio: `debian.control`, `debian.rules`, `PACKAGE.dsc` (OBS User Guide §2.3 "Debian: Dsc"). The build phase runs `dpkg-buildpackage` on Debian-based distributions (§25.1.3 "Package Build"); Debian build environments can alternatively use the `debootstrap` build engine (§"Configuration File Syntax", `BuildEngine`).
|
||||
- Quirk: release numbers are **not** auto-incremented across rebuilds unless the dsc carries `DEBTRANSFORM-RELEASE` (§2.3) — a packaging decision we'd own either way.
|
||||
- Upstream build deps are available: `dh-virtualenv` and `dh-python` exist in Debian 12 (packages.debian.org, checked 2026-09-03), so the ADR-0004 venv/lockfile design maps onto an OBS dsc without patching the build root.
|
||||
- All five matrix targets exist as public OBS build roots: `Debian:12`, `Ubuntu:22.04`, `Ubuntu:24.04`, `Fedora:40`, `Fedora:41` — each project `_meta` answered HTTP 200 on build.opensuse.org (checked 2026-09-03).
|
||||
- Live proof of deb publishing quality: `isv:ownCloud:desktop/Debian_10` on download.opensuse.org serves a proper Debian archive (`Release`, `Release.gpg`, `InRelease` all HTTP 200, checked 2026-09-03).
|
||||
|
||||
### 2. Vendoring textual≥0.40 — the offline sandbox forces the same work we already planned
|
||||
|
||||
- The build environment has **no network**: "services requiring external network access are likely to fail in [buildtime] mode, because such access is not available if the build workers are running in secure mode (as is always the case at https://build.opensuse.org)" (User Guide §7.2, "Modes of Source Services"); Dockerfile builds likewise run "in a safe build environment without network access" (§29.3).
|
||||
- Vendoring must therefore happen **before** the build, via source services that run server-side on commit (`default`/`trylocal` modes, §7.2) or via files committed to the package. The standard services are per-file fetchers — `download_url` (§22.1.3), `download_files`, `obs_scm`/`tar`/`set_version` (§8 SCM integration) — there is no "pip resolve" service, so a pinned dependency tree like Fenris's means either N `download_url` entries mirroring the committed lockfile, or simply committing the vendored wheel/sdist tree.
|
||||
- Conclusion: OBS does not remove the vendoring step; it reproduces ADR-0004's committed-lockfile design with extra XML. Since distro `python3-textual` is 0.1.13 on Debian 12, Ubuntu 22.04 and 24.04 (packages.debian.org / packages.ubuntu.com, checked 2026-09-03) — far below the `>=0.40` floor — vendoring is unavoidable on any route.
|
||||
|
||||
### 3. Signing — OBS key, not ours; Gitea deb repo is our key
|
||||
|
||||
- OBS signs published repositories with the **instance's** key: one signer per partition "calls an external tool to execute the signing" (User Guide §23 "OBS Architecture", Signer); consumers accept the OBS repo key ("When prompted, accept the GPG key of the download repository", §1.10). A build.opensuse.org user cannot upload a personal signing key. Trust therefore flows to openSUSE infra, and the signature says nothing about Fenris's maintainers.
|
||||
- Gitea 1.27.1's Debian registry serves apt metadata signed with the Gitea instance's PGP key (`repository.key` endpoint, `signed-by` in sources.list — docs.gitea.com, "Debian Package Registry"), i.e. **our** host and **our** key. The RPM registry serves a `.repo` endpoint but documents no GPG signing of repodata; rpm-file signing stays our choice at build time.
|
||||
|
||||
### 4. Publishing reach — OBS wins reach; reach is not our bottleneck
|
||||
|
||||
- OBS publishes home-project results to `https://download.opensuse.org/repositories/home:USER/<dist>` (§1.10) and offers generated download pages on software.opensuse.org (§17.4). That is genuine CDN-class reach.
|
||||
- Caveats from the same docs: branched projects are **not** published by default (§1.10), and the repo is a live view of the project state — no release artefact pinning; deleting the project or flag disables distribution.
|
||||
- The Gitea route's reach is exactly `git.bongbetic.com` plus whatever the README says — adequate for a named four-distro matrix whose users follow our instructions, and it keeps the release artefact under versioned control on the same host as the source.
|
||||
|
||||
### 5. Account and maintenance cost — strictly additive
|
||||
|
||||
- Using build.opensuse.org requires an openSUSE account (single sign-on; the web UI's "Sign up!") and work happens in `home:USERNAME` plus permitted subprojects (§"Setting Up Your Home Project for the First Time"; §23 "OBS Concepts" on home projects).
|
||||
- Day-to-day: `osc` + `_service` XML + dsc/spec recipes maintained in OBS's own package VCS, kept in sync with Fenris's git. The SCM bridge (`scmsync`) does support self-hosted Gitea ("We also support Self-Hosted instances from GitHub, GitLab and Gitea", §8.1.3; setup in §28.1.2 — build descriptions must live in the repo's top level), but it also disables OBS-side workflows (no `_link` merging, limited workflows, §28.1.1).
|
||||
- No published quota/SLA for the public instance; capacity and availability are a shared commons. The Gitea route needs zero new accounts, zero new artefact formats beyond the two package recipes we must write anyway, and reuses the existing release host.
|
||||
|
||||
### 6. Build latency — shared queue vs. deterministic local
|
||||
|
||||
- OBS routes every commit through scheduler → dispatcher → shared workers; the dispatcher "tries to assign jobs fairly between the project repositories" using a per-repository load model (§23, Scheduler/Dispatcher). For our five tiny x86_64 jobs this is typically minutes, but there is no documented SLA and the queue is global — worst case is unbounded (estimate; the docs guarantee only fairness, not latency).
|
||||
- The Gitea route builds wherever `make` runs and publishes with one authenticated `PUT` per artefact (docs.gitea.com: Debian `PUT .../pool/{distribution}/{component}/upload`; RPM `PUT .../rpm/{group}/upload`). Latency = build time, fully under our control.
|
||||
|
||||
## Comparison on the 4-distro matrix
|
||||
|
||||
| Axis | OBS (build.opensuse.org) | Gitea 1.27.1 registry |
|
||||
|---|---|---|
|
||||
| Debian 12 / Ubuntu 22.04/24.04 deb | dsc + dpkg-buildpackage; DEBTRANSFORM-RELEASE quirk | we build the same deb locally, upload via PUT |
|
||||
| Fedora 40+ rpm | spec + rpmbuild in Fedora roots | same spec built locally, `.repo` grouping (`fedora/40`) |
|
||||
| Vendoring textual≥0.40 | offline sandbox forces committed vendored tree (no pip service) | same committed vendored tree (ADR-0004 lockfile) |
|
||||
| Signing | OBS instance key (not ours) | deb repo signed with our key; rpm repodata unsigned |
|
||||
| Reach | download.opensuse.org CDN + software.o.o pages | our domain only |
|
||||
| Accounts/infra | new openSUSE account, osc workflow, commons SLA-free | zero new infra |
|
||||
| Latency | global shared queue, minutes typical, no SLA | deterministic (local build) |
|
||||
|
||||
## Recommendation
|
||||
|
||||
**Reject OBS as the build + distribution route for Fenris.** The offline sandbox forces the exact vendoring work the Gitea route already requires, so OBS adds cost (account, osc/source-service maintenance, external commons in the release path, queue latency) without removing any; its one real advantage — CDN and software.o.o reach — does not matter for a hobby project whose four target distros are served by one signed apt repo and one rpm repo on the existing Gitea host, under our own key.
|
||||
|
||||
Revisit trigger: if Fenris later wants one-click installs via software.opensuse.org, architectures beyond x86_64, or many more distro targets — the deb publishing quality (verified live) and self-hosted-Gitea SCM bridge make OBS a viable amplifier then.
|
||||
|
||||
## Sources
|
||||
|
||||
- OBS User Guide (openbuildservice.org/help/manuals/obs-user-guide/, PDF): §2.3 Debian: Dsc; §7 Using Source Services (offline buildtime services, modes); §8.1.3 Supported SCMs; §17.4 download pages; §22.1.3 download_url; §23 OBS Architecture (Scheduler/Dispatcher/Signer); §25.1.3 Package Build; §28.1 SCM bridge; §29.3 Dockerfile builds (no network); §1.10 Installing Packages from OBS; "Configuration File Syntax" (BuildEngine, Repotype: debian).
|
||||
- Live checks (2026-09-03): `Debian:12`/`Ubuntu:22.04`/`Ubuntu:24.04`/`Fedora:40`/`Fedora:41` project `_meta` on build.opensuse.org (all 200); `isv:ownCloud:desktop/Debian_10` `Release`/`Release.gpg`/`InRelease` on download.opensuse.org (all 200).
|
||||
- packages.debian.org / packages.ubuntu.com (2026-09-03): `python3-textual` 0.1.13 on bookworm, jammy, noble; `dh-virtualenv`, `dh-python` present in bookworm.
|
||||
- docs.gitea.com, "Debian Package Registry" and "RPM Package Registry" (1.27 line): apt sources with `signed-by` + `repository.key`, `PUT` upload endpoints, `.repo` groups.
|
||||
@@ -1,162 +0,0 @@
|
||||
# Acceptance criteria: the Fenris redesign
|
||||
|
||||
Status: Accepted — resolves [Define cross-cutting acceptance criteria](https://git.bongbetic.com/xavierk/Fenris/issues/13) on the [Wayfinder map](https://git.bongbetic.com/xavierk/Fenris/issues/1). These criteria are the accepted definition of done for the finished redesign; the implementation-ready specification assembles them with ADRs 0001–0006 at handoff.
|
||||
|
||||
## Framework
|
||||
|
||||
- **Canonical term**: *acceptance criterion* — one testable behavioral statement. "Behavioral gate" is avoided as a synonym. Wording follows the repository glossary (`CONTEXT.md`).
|
||||
- **Evidence classes** — every criterion carries exactly one:
|
||||
- **A** — automated test (unit/integration, fixture-driven).
|
||||
- **P** — scripted system probe on a host with systemd, polkit, and the configured NVMe device.
|
||||
- **M** — manual checklist, reserved for interactions a fixture cannot capture (live polkit agent prompts, TUI keyboard feel).
|
||||
- Whatever can be automated must be; **M** only where automation cannot reach.
|
||||
- **Traceability-only**: every number and behavior cites the ADR or ticket that fixed it. Nothing undecided enters here; new demands become new tickets, never criteria.
|
||||
- **Organization**: criteria are grouped by subsystem, with cross-cutting invariants spanning them. Coverage spans all decided areas — observation store and migration, collector lifecycle and privileges, projection and confidence, controller identity, the Panes TUI, failure paths, and installation lifecycle.
|
||||
- **Placeholders**: none remain. SLOT-B was filled by [Choose the collector's NVMe acquisition path](https://git.bongbetic.com/xavierk/Fenris/issues/16) as AC-1–AC-5; SLOT-A was filled by [Decide how degraded identity affects projection confidence](https://git.bongbetic.com/xavierk/Fenris/issues/15) as PR-15, PR-16, and ID-4.
|
||||
- **Test-plan boundary**: Given/When/Then test specs are derived by the implementer at implementation time. This effort produces criteria only.
|
||||
|
||||
## Cross-cutting invariants
|
||||
|
||||
- **CI-1** (A; ADR 0002 §§6–8, ADR 0003 §10) *Exhaustive state matrix*: from a synthetic observation store, the TUI and `fenris status` render every realizable combination of confidence state (Unavailable, Limited, Supported) × freshness grade (fresh, missed, stale, empty store) × endurance-baseline tier (verified override, unverified override, implied, none) exactly as the ADR 0002 rule table and ADR 0003 freshness constants dictate — headline number present only when the rules allow it, contributing facts always, never a percentage.
|
||||
- **CI-2** (P/A; ADR 0003 §§4–9) *TUI/CLI parity*: every TUI action has a CLI twin — pause (`fenris monitor pause`), resume (`fenris monitor resume`), collect-now (`fenris sample`), baseline set/clear, and the status fact set — with identical outcomes and wording.
|
||||
- **CI-3** *Prohibition set* (A = test, P = probe; each cites its clause):
|
||||
- No code path outside `fenris-collect` interrogates the device (ADR 0003 §7).
|
||||
- No `/run/fenris` coordination surface or export layer exists anywhere; state lives in the observation store and coordination in systemd (ADR 0003 §1, ADR 0001 §2).
|
||||
- Polkit authorizes exactly one binary, `fenris-monitor`, under `com.bongbetic.fenris.monitor` `auth_admin` (ADR 0003 §5, ADR 0004 §3).
|
||||
- No absent hour is ever interpolated, estimated, or fabricated (ADR 0005 §2).
|
||||
- No alerting, notification, or escalation machinery exists anywhere (ADR 0005 §§5–6).
|
||||
- `/etc/fenris/fenris.conf` holds exactly one key — the device selector (ADR 0003 §3).
|
||||
- No synthetic or capacity-derived baseline is ever created, including for legacy history (ADR 0002, Consequences).
|
||||
- Readers never partially interpret a newer-schema store (ADR 0001 §8, ADR 0005 §4).
|
||||
- Projections are never stored; always recomputed on read (ADR 0001 §3, ADR 0002 §13).
|
||||
- **CI-4** (A; ADR 0002 §§11–12) *User-facing language*: TUI and status render the endurance research's required wording and six disclosures as adopted; zero-rate and unavailable cases use their exact phrasing; the scenario range is the only spread shown anywhere.
|
||||
|
||||
## Observation store and legacy migration (ADR 0001)
|
||||
|
||||
- **ST-1** (P) One SQLite database in WAL mode at `/var/lib/fenris/observations.db`, root-owned and group-readable through the `fenris` read group; the TUI opens it read-only.
|
||||
- **ST-2** (A) An unprivileged reader querying during a collector write sees a consistent snapshot.
|
||||
- **ST-3** (A) The schema carries `samples`, `hour_observations`, `day_aggregates`, `monitoring_periods`, `controller_segments`, and `endurance_baseline` with the ADR 0001 column sets as amended by [Define endurance-baseline provenance and validation](https://git.bongbetic.com/xavierk/Fenris/issues/12) and [Decide controller-segment metadata columns](https://git.bongbetic.com/xavierk/Fenris/issues/14).
|
||||
- **ST-4** (A) Hours and days are UTC-bounded; day derivation from hours is monotonic; no 23- or 25-hour days exist.
|
||||
- **ST-5** (A) Raw samples are pruned opportunistically to 14 days; hour observations and day aggregates are retained indefinitely.
|
||||
- **ST-6** (P) Legacy import is one transaction: a scripted kill mid-import leaves the store fully pre- or fully post-migration.
|
||||
- **ST-7** (A) Import is idempotent: a second run no-ops on the legacy-import marker.
|
||||
- **ST-8** (A) Legacy files are renamed `*.migrated` only after commit and never deleted.
|
||||
- **ST-9** (A) Malformed legacy lines are quarantined with a logged count, never silently dropped.
|
||||
- **ST-10** (A) `hourly.jsonl` is never trusted: mismatches against derived data are diffed and logged.
|
||||
- **ST-11** (A) Migration opens one implicit monitoring period at the first legacy sample, closed `end_cause = migrated` at the migration moment; pre-migration hours carry an unknown activity split except directly evidenced facts.
|
||||
- **ST-12** (A) Schema versioning: `PRAGMA user_version` with ordered, per-step transactional migrations; the collector refuses an unknown newer version.
|
||||
|
||||
## Collector lifecycle and privilege boundaries (ADR 0003)
|
||||
|
||||
- **LC-1** (P) Exactly two system units exist — `fenris-collect.timer` (`timers.target`) and `fenris-collect.service` (`Type=oneshot`, root, `ExecStart=/usr/libexec/fenris/fenris-collect`); the TUI and CLI are ordinary unprivileged processes and never units.
|
||||
- **LC-2** (P) Timer defaults ship as `OnBootSec=2min`, `OnUnitInactiveSec=5min`, `AccuracySec=30s`, `Persistent=no`, `TimeoutStartSec=90s`; cadence changes are documented drop-ins and no interval key exists in configuration.
|
||||
- **LC-3** (P) A hung device interrogation fails visibly within `TimeoutStartSec=90s` as a bounded failed run retried next interval.
|
||||
- **LC-4** (A/P) `/etc/fenris/fenris.conf` holds exactly the device selector (stable `/dev/disk/by-id/…` path; raw nodes warned), re-read every run; an invalid selector is a bounded failed run surfaced as `configuration error: <reason>` in `status` and the TUI.
|
||||
- **LC-5** (P) Two privileged binaries ship at `/usr/libexec/fenris/fenris-collect` and `/usr/libexec/fenris/fenris-monitor`; the unprivileged `fenris` wrapper opens the TUI with no arguments.
|
||||
- **LC-6** (P+M) Pause = `fenris-monitor disable --now` asks for confirmation; Resume = `enable --now` does not; both perform the systemctl operation and period bookkeeping in one step under polkit `com.bongbetic.fenris.monitor` (`auth_admin`), failing cleanly with the printed root equivalent where no polkit agent exists. (M covers the live agent prompt.)
|
||||
- **LC-7** (A) The period-row idempotent matrix of ADR 0003 §6 holds exactly: first-ever enable opens; resume with an open period changes nothing; resume without one opens anew; pause with an open period closes `user_disabled`; pause otherwise no-ops; a raw systemctl stop/disable never records `user_disabled`.
|
||||
- **LC-8** (P) `fenris sample` and the TUI's collect-now route through `fenris-monitor` → `systemctl start fenris-collect.service`, block until exit, and report the outcome (freshness line or journal hint) synchronously; the TUI never samples in-process.
|
||||
- **LC-9** (A/P) CLI compatibility: `status` is a read-only composition (projection facts, enabled/active, last collect outcome, `journalctl` hint on failure or staleness) that never auto-samples and never prompts; `sample` is retained via the helper; `--device` is rejected with a pointer to the configuration file; `start`, `stop`, and `run` are rejected with one-line migration pointers; `fenris.sh` is not shipped and is removed from the repository; the README maps its five menu options to successors.
|
||||
- **LC-10** (A) Freshness constants are defined once and shared by TUI and CLI: fresh = newest sample within 2× cadence + `AccuracySec` + 60 s; missed between that and 48 h; stale ≥ 48 h; an empty store reads "no observations yet" with an enable hint; freshness derives from the newest sample timestamp, never a stored health flag.
|
||||
|
||||
## Projection contract and confidence (ADR 0002)
|
||||
|
||||
- **PR-1** (A) Exactly one projection, from the precedence-chosen baseline (verified override → unverified override → implied → unavailable); Percentage Used renders as a vendor-wear context line, with a note when it disagrees with the observed write rate by more than a factor of 2; the PU-slope regression and `capacity × 600` synthesis are gone.
|
||||
- **PR-2** (A) The headline rate is the sustained-regime rate (regime DUW bytes ÷ in-period wall-clock seconds), default regime = full history capped at 90 days; the 7/28/90-day scenario range is computed independently and shows only covered horizons, with no placeholders.
|
||||
- **PR-3** (A) Habit change: trailing 7-day mean ≥ 2× or ≤ 0.5× the preceding 28-day mean for 3 consecutive days starts a new regime at the first divergence day, adopted automatically and labeled "usage habit changed N days ago"; a regime younger than 7 days caps confidence at Limited evidence.
|
||||
- **PR-4** (A) Hour classification uses the named constants: powered-off below 90% of power-on-hours span; active at ≥ 256 MiB DUW; idle below it while powered on and sampled; unknown otherwise; disabled time is wall-clock outside monitoring periods, never an hour state.
|
||||
- **PR-5** (A) The denominator is wall-clock seconds inside monitoring periods including powered-off and unknown time; disabled periods are excluded from numerator and denominator; unexplained gaps keep the aggregate counter delta, remain as unknown seconds, and reduce coverage.
|
||||
- **PR-6** (A) Warming up until 14 distinct UTC day aggregates of which at most 2 fall below 50% coverage; the projection still renders with its facts while warming; every Unavailable condition renders no lifespan number.
|
||||
- **PR-7** (A) A newest day aggregate older than 48 h drops confidence one level and is shown as a contributing fact.
|
||||
- **PR-8** (A) The confidence rule table of ADR 0002 §8 holds verbatim, rendering state plus contributing facts and never a percentage.
|
||||
- **PR-9** (A) Segment breaks: a DUW decrease with unchanged identity keeps prior day aggregates as habit evidence with the projection Unavailable until re-warm; a controller-identity change quarantines prior history from projection entirely.
|
||||
- **PR-10** (A) The implied baseline is eligible only after ≥ 2 Percentage-Used increments within the current controller segment; until then, Unavailable with "vendor wear estimate too coarse to imply endurance".
|
||||
- **PR-11** (A) Zero rate renders "no finite projection from this history" — never infinity or zero; no statistical confidence interval appears anywhere.
|
||||
- **PR-12** (A) The projection contract hands the TUI exactly: confidence state, contributing facts, headline remaining time when one exists, scenario range, Percentage-Used context line, disclosure text — recomputed on read, never stored.
|
||||
- **PR-13** (A) Baseline provenance and validation per [Define endurance-baseline provenance and validation](https://git.bongbetic.com/xavierk/Fenris/issues/12): mandatory provenance (URL, revision, entry date, model, nominal capacity); one active row replaced on edit; verification derived at read (machine match or recorded attestation), never a stored boolean; incomplete provenance stores only behind explicit acknowledgment as the unverified tier; entry-time unprivileged sysfs validation (normalized model containment; capacity within ±1%; interactive confirm recorded as `validated_by = user`); read-time applicability is a model match against the current controller segment, with a mismatch retained — never auto-deleted — leaving the projection Unavailable.
|
||||
- **PR-14** (P) `baseline set` / `baseline clear` persist through the polkit-guarded `fenris-monitor` verb after CLI-side validation.
|
||||
- **PR-15** (A) An identity-degraded controller segment (blank identity key — every rung of the key ladder empty) caps projection confidence at Limited evidence, with the contributing fact "controller identity unavailable — replacement detection relies on write-counter continuity only" rendered in every state; the cap combines idempotently with the 48-hour staleness drop, and ephemeral markers (model "Linux", non-pcie transport) never render as confidence facts ([Decide how degraded identity affects projection confidence](https://git.bongbetic.com/xavierk/Fenris/issues/15); ADR 0002 §§8–9 as amended).
|
||||
- **PR-16** (A) Identity-change semantics extend to blank keys verbatim: any visible change of the recorded identity key — including to or from a blank key — quarantines prior history from projection as a controller-identity change, while equal blank keys continue the segment segmented by DUW monotonicity alone (ADR 0002 §9 as amended).
|
||||
- **PR-17** (A) Projection arithmetic is exactly `E_rated = entered_TBW × 10¹²` bytes, `E_implied = 100 · W_t / p` computed only for `1 ≤ p ≤ 254` (Percentage Used of 0 or saturated 255 implies no baseline — that precedence tier is unavailable), and `projected = max(E_baseline − W_t, 0) / rate` for `rate > 0` (ADR 0002 §2).
|
||||
|
||||
## Controller identity ([Verify the controller identity that segments observation history](https://git.bongbetic.com/xavierk/Fenris/issues/11), [Decide controller-segment metadata columns](https://git.bongbetic.com/xavierk/Fenris/issues/14); ADR 0001 §3 as amended)
|
||||
|
||||
- **ID-1** (A) The controller-segment identity key is the normalized kernel-exposed subsystem NQN, with the kernel composite then model|serial as fallbacks; FR is metadata only; identity change and DUW decrease act as independent axes.
|
||||
- **ID-2** (A) Segments freeze a fully nullable metadata snapshot at open — normalized `subnqn`/`sn`/`mn`/`fr` plus `vid`/`ssvid`/`transport` and `identity_degraded` — immutable thereafter, with `cntlid` excluded.
|
||||
- **ID-3** (A) Legacy history imports under a labeled model-scoped legacy identity (mn-only segments).
|
||||
- **ID-4** (A) `identity_degraded` is set at segment open exactly when the identity key is blank; keys from the kernel-composite or `model|serial` rungs are not degraded ([Decide how degraded identity affects projection confidence](https://git.bongbetic.com/xavierk/Fenris/issues/15)).
|
||||
|
||||
## Panes TUI ([Prototype the TUI information architecture](https://git.bongbetic.com/xavierk/Fenris/issues/3), [Evaluate Python TUI frameworks](https://git.bongbetic.com/xavierk/Fenris/issues/6); ADR 0003 §§8, 10; ADR 0004 §10)
|
||||
|
||||
- **TUI-1** (A) Variant A "Panes": one dense keyboard-first screen; confidence rendered as evidence (state + contributing facts); boot enablement, runtime activity, last collect outcome, and freshness displayed as four separate facts.
|
||||
- **TUI-2** (M) Pause/resume asymmetry and polkit tty passthrough work in a live terminal: pause confirms, resume does not, and the platform agent prompts without breaking the TUI.
|
||||
- **TUI-3** (P) Textual runs on Python 3.9+, gated at install time, never a runtime crash.
|
||||
- **TUI-4** (A) The Panes screen layout is normative: a full-width headline band (lifespan headline or its no-projection wording, confidence state with contributing facts, scenario range); a usage-history pane on the left (write-history sparkline with ▲ habit-change and ? unexplained-gap markers plus legend, habit-split bar with active/idle/powered-off/unknown shares); a drive-health and settings pane on the right (health facts, vendor-wear context line, read-only settings with the endurance baseline and its provenance label); a full-width service strip at the bottom (the four separate service facts, the monitoring-period line, the action legend). Production bindings are the footer `p pause · r resume · c collect · d disclosures` — pause asks, resume does not — plus a bordered quit rail `q QUIT TUI` visually separate from monitoring state; the rail owns quit and the footer carries no quit entry (bindings amended by [Lock the dashboard wording strings](https://git.bongbetic.com/xavierk/Fenris/issues/57); original [Prototype the TUI information architecture](https://git.bongbetic.com/xavierk/Fenris/issues/3)); the prototype branch is visual reference only.
|
||||
|
||||
## Failure and recovery (ADR 0005)
|
||||
|
||||
- **FL-1** (A) The collector validates every row against the store invariants (hour seconds sum to 3600; non-negative DUW delta within a controller segment; coverage consistent with sample count); a violating run writes nothing, logs the refused row, and fails visibly.
|
||||
- **FL-2** (A) Readers defensively exclude and count malformed rows as a contributing fact.
|
||||
- **FL-3** (A) No backfill ever: gaps remain unknown seconds; degradation flows only through coverage, freshness facts, and confidence categories.
|
||||
- **FL-4** (P/A) A store fault surfaces "observation store unreadable" with a journal hint and suppresses everything else store-dependent; the collector treats it as a bounded failed run and never recreates or overwrites the file; recovery is the documented human-sanctioned move-aside (with `history.jsonl` re-import if the legacy import never completed); no built-in destructive command exists.
|
||||
- **FL-5** (A) A newer-schema store renders "observation store written by a newer Fenris — upgrade Fenris" in TUI and status, with no partial interpretation.
|
||||
- **FL-6** (A/P) Repeated collector failures retry at flat cadence with no backoff or notification; persistence reads as stale exactly like any other gap.
|
||||
- **FL-7** (A) `critical_warning`, media errors, and unsafe shutdowns render as ordinary facts in TUI and status and never affect the projection.
|
||||
- **FL-8** (A) A collection run finding no open monitoring period opens one at the run moment, never backdated.
|
||||
|
||||
## Installation, upgrade, and removal (ADR 0004)
|
||||
|
||||
- **IN-1** (P) `sudo make install` builds a wheel from the checkout and installs pinned dependencies into the dedicated venv at `/opt/fenris`, with a `/usr/local/bin/fenris` wrapper; after install nothing references the checkout.
|
||||
- **IN-2** (P) The installer records every placed file in an explicit manifest consumed by upgrade and uninstall.
|
||||
- **IN-3** (P) The installer never enables or starts units: a fresh install is dormant (units disabled, nothing running, no monitoring period); the only opt-in is the sanctioned toggle — `fenris monitor resume [--now]` or the first-run TUI prompt — enabling the timer and opening the first period in one step.
|
||||
- **IN-4** (P) Install-time legacy import detects `./data/history.jsonl` (or an explicit path), runs the idempotent single-transaction import, and reports imported counts; `fenris import <path>` remains available.
|
||||
- **IN-5** (P) `sudo make upgrade` installs into the same venv, syncs units and polkit against the manifest (`daemon-reload`; timer restarted only if unit contents changed and it is active), never kills an in-flight collection run, then applies forward-only schema migrations; `/var/lib/fenris` is never rebuilt.
|
||||
- **IN-6** (P) Before migrations, `observations.db` is snapshotted to a one-generation `.bak`; rollback is reinstall-previous plus restore; automatic schema downgrade does not exist.
|
||||
- **IN-7** (P) `make uninstall` performs the sanctioned disable first (open period closes `user_disabled`), then removes venv, helpers, units, polkit policy, and wrapper while keeping `/etc/fenris` and the observation store; `make purge` additionally removes configuration and store.
|
||||
- **IN-8** (P) Dependencies are exact pins in a committed lockfile installed by both install and upgrade; refreshing pins is an explicit `make update-deps` step, never an install side effect.
|
||||
- **IN-9** (P) The installer verifies `python3 ≥ 3.9` and fails cleanly otherwise; `/var/lib/fenris` is created with root-written group-read permissions; the database file is created lazily by the first write.
|
||||
- **IN-10** (P) Installed artifacts sit only at their fixed locations — units in `/etc/systemd/system`, helpers in `/usr/libexec/fenris`, polkit policy under `/usr/share/polkit-1/actions/`, configuration at `/etc/fenris`, observation store under `/var/lib/fenris` — and every placed file is recorded in the manifest (ADR 0004 §2; ADR 0003 §4).
|
||||
|
||||
## Migration from make-install systems (ADR 0007 §10, spec §9)
|
||||
|
||||
- **MG-1** (M) The migration runbook is published in the install docs (`docs/install/migrate-from-makeinstall.md`): mandatory remove-then-install steps, why over-install is forbidden (stale admin-directory units silently shadow vendor units; the local wrapper shadows the package wrapper), no-move continuity, and the reset-to-dormant expectation (the user opts back in with the sanctioned resume).
|
||||
- **MG-2** (A) The install guard is verified across the matrix: either make-install marker (the legacy placement manifest, or a unit file under the admin unit directory) causes an abort with a runbook pointer — never auto-clean. Tested by `test_migration_guard` on all four targets (Debian 12, Ubuntu 22.04, Ubuntu 24.04, Fedora 40).
|
||||
- **MG-3** (A) No-move continuity is verified in a container seeded with a make-install-shaped system: existing group makes sysusers a no-op, existing store directory makes tmpfiles a no-op, the hand-written configuration survives as a non-database file (package default lands beside it), and the store schema is caught up by the upgrade-path migration. Tested by `test_no_move_continuity_deb` and `test_no_move_continuity_rpm`.
|
||||
- **MG-4** (M) The migration costs at most one short sample gap, honestly recorded in the endurance timeline: `make uninstall`'s sanctioned disable closes the open period `user_disabled`; after migration the user opts back in with `fenris monitor resume`.
|
||||
|
||||
## Collector acquisition path (ADR 0006)
|
||||
|
||||
- **AC-1** (P) Each collection run acquires counters and thermal evidence solely from `smartctl -a -j <device>` and controller identity (`subnqn`, `sn`, `mn`, `fr`, `transport`) solely from sysfs; no other acquisition path exists anywhere in the codebase.
|
||||
- **AC-2** (A) Identity normalization is applied exactly once, at write time — trailing spaces and newlines stripped, no case folding, empty-after-strip stored blank — so padded and unpadded renderings of the same field yield byte-identical stored values.
|
||||
- **AC-3** (A) Any acquisition failure — missing binary, nonzero exit, malformed JSON, unreadable sysfs attribute — fails the whole collection run; no partial sample (identity without counters, or counters without identity) is ever written; the miss surfaces through ADR 0005 freshness, never as degraded identity.
|
||||
- **AC-4** (P) `vid`/`ssvid` are read from the PCI sysfs node when present and stored null otherwise; they are segment metadata only, never key components.
|
||||
- **AC-5** (P) `make install` verifies `smartctl` and fails cleanly otherwise; the acquisition path adds no Python dependency and no OS package beyond smartmontools (ADR 0004 §9).
|
||||
|
||||
## Dashboard clarity and release notes ([Chart Fenris dashboard clarity](https://git.bongbetic.com/xavierk/Fenris/issues/55))
|
||||
|
||||
Decided in [Write the dashboard clarity acceptance criteria](https://git.bongbetic.com/xavierk/Fenris/issues/59), from [Prototype the dashboard clarity additions](https://git.bongbetic.com/xavierk/Fenris/issues/56), [Lock the dashboard wording strings](https://git.bongbetic.com/xavierk/Fenris/issues/57), and [Specify the changelog and release-notes mechanism](https://git.bongbetic.com/xavierk/Fenris/issues/58).
|
||||
|
||||
- **DC-1** (A) TUI branding: the header bar renders `Fenris — NVMe endurance monitor`; a dimmed `by Bongbetic` sits inline with service facts in the bottom service strip; neither string appears in `fenris status` (TUI-only identity surfaces).
|
||||
- **DC-2** (A) Continuity parity, keyed to the boot fact as-is: active + boot-enabled renders `monitoring: active in background · persists across reboots`; boot-disabled renders `monitoring: does not start on next boot` — identical lowercase source strings in the TUI service strip and `fenris status`, including while paused (paused implies boot-disabled; the row still reports the fact). Test impact: feeds the CI-2 sweep (lowercase source-string comparison).
|
||||
- **DC-3** (A) Paused presentation (Deliberate disable): the TUI shows a strong state block titled `monitoring: paused — deliberate disable` with subline `paused time is excluded from your usage habit · resume: fenris monitor resume`; `fenris status` prints the same two lines with identical wording. Test impact: feeds the CI-2 sweep (lowercase source-string comparison).
|
||||
- **DC-4** (A) Quit affordance distinct from monitoring state: a bordered labelled rail `q QUIT TUI` visually separate from the paused state block; the footer reads `p pause · r resume · c collect · d disclosures` with no quit entry (the rail owns quit); quitting the TUI never alters monitoring state. Amends TUI-4's binding parenthetical.
|
||||
- **DC-5** (A) Launch auth banner: `privileged actions will prompt for authentication (polkit)` renders full-width under the header at TUI launch, clears on the first refresh tick, and never reappears in the session; no user-facing string uses "sudo" (polkit-accurate elevation wording only).
|
||||
- **DC-6** (A) CHANGELOG.md shape (Keep a Changelog 1.1): `## [Unreleased]` always present at top, even empty; version headings `## [X.Y.Z] - YYYY-MM-DD` with strict ISO date; categories Added/Changed/Fixed only, security folding into Fixed; entries are single `- ` bullets, imperative mood, user-facing, no commit hashes or issue numbers.
|
||||
- **DC-7** (A) Extraction fails closed: `scripts/extract_changelog.py` slices the requested version's section verbatim and never reads `[Unreleased]`; a missing or empty section or a malformed date produces `::error::` and a nonzero exit; the release workflow fails when the pushed tag ≠ `v{version from pyproject.toml}` (guard skipped on `workflow_dispatch`).
|
||||
- **DC-8** (A/P) Release body: the body is the extracted section verbatim plus the standing footer from `packaging/release-footer.md`; a re-run against an existing release PATCHes the body (re-sync is a feature) while uploaded assets skip idempotently. A covers assembly/PATCH-logic unit tests; P is one scripted `workflow_dispatch` verification of body assembly.
|
||||
|
||||
## TUI polish and hourly history ([Fenris TUI polish and hourly history](https://git.bongbetic.com/xavierk/Fenris/issues/65))
|
||||
|
||||
Proposed by [Approve the Fenris TUI polish specification and handoff](https://git.bongbetic.com/xavierk/Fenris/issues/70), from the companion specification [`fenris-tui-polish-hourly-history.md`](fenris-tui-polish-hourly-history.md). This section amends the frozen redesign and prior dashboard-clarity criteria without editing their historical source specs. Where these criteria conflict with older TUI/header/history criteria, these newer criteria win. In particular, **TPH-1** supersedes **DC-1**'s header/credit placement; **TPH-2** and **TPH-3** refine **CI-2**, **LC-10**, and **TUI-4** status rendering; **TPH-4** through **TPH-7** refine **ST-5**, **PR-2** through **PR-9**, and **FL-1** through **FL-4** with the approved hourly-history and UTC-accounting contracts.
|
||||
|
||||
- **TPH-1** (A) *Titlebox and maker credit*: the TUI renders a top titlebox exactly `🐺 Fenris by Bongbetic`, falling back exactly to `Fenris by Bongbetic` when the wolf glyph is unsupported or width-unstable; no replacement-box glyph is shown; the old service-strip `by Bongbetic` credit is absent; `fenris status` renders no titlebox. Continuity, paused, quit, auth, parity, and release-notes behavior from **DC-2** through **DC-8** remains unchanged.
|
||||
- **TPH-2** (A) *Status lattice and precedence*: fixture-driven TUI status rendering covers Monitoring, Collecting, Paused, Waiting, Interrupted, Error, Stale, and Unknown with the approved glyphs/text, semantic colours, and reason lines; only Monitoring's dot blinks, never text; reduced motion makes it steady; precedence is Error > Interrupted > Paused > Stale > Waiting > Monitoring > Unknown, with Collecting as an overlay except over store fault.
|
||||
- **TPH-3** (A/P) *CLI/status parity and timing*: `fenris status` renders the same status vocabulary, glyphs, precedence, and reason lines statically; freshness, last outcome, boot enablement, and collection activity remain separate facts; a lightweight 5 s unit-state poll can move cached freshness boundaries without a store read, while new samples appear only after the normal store refresh; store faults render `observation store unreadable — see journal` and suppress store-dependent views.
|
||||
- **TPH-4** (A) *Warm-up and withheld estimates*: projection warm-up shows `Building evidence — N of 14 days observed · Q qualifying` plus `First lifespan estimate after 12 qualifying days`; the gate is 14 represented UTC dates in the current controller segment with at least 12 qualifying, while Supported separately requires 14 qualifying dates and all existing prerequisites. Missing baseline, unsupported write counters, warm-up, stale evidence, paused days, and unavailable numerators render explicit reason lines, never blank or misleading zero; first graph-data availability is independent of lifespan-estimate availability.
|
||||
- **TPH-5** (A) *Collector-owned history publication and first data*: collection publishes validated sample → usage interval → hour observation/day aggregate results consistently before reporting success; the TUI remains read-only. Zero samples show awaiting-first-sample; one sample shows `Awaiting another sample`; the first compatible sample pair can show measured partial-hour `so far`; measured zero is `0 B`; missing, unsupported, invalid, or unavailable evidence is never converted to zero.
|
||||
- **TPH-6** (A) *Local display days, attribution, gaps, pauses, and repair*: history browsing groups retained evidence by the current system timezone with the timezone label visible, including DST and fractional-offset cases; timestamped usage intervals are retained indefinitely alongside hour/day summaries, while raw samples keep the 14-day policy except needed boundary anchors. Measured interval totals are preserved once; cross-boundary shares render as unallocated usage rather than interpolation or endpoint assignment; gaps remain distinct from zero; future time is not counted; deliberate-disable time is excluded; pause-crossing bytes are not counted as monitored totals; repair is transactional/idempotent and cannot overwrite valid older history with incomplete reconstruction.
|
||||
- **TPH-7** (A) *UTC projection accounting and evaluability*: 7/28/90-day scenario windows end at the latest published usage-evidence endpoint `T` and start exactly 7/28/90 × 86,400 seconds earlier; denominators are monitored wall-clock seconds in the same span; rates are withheld when the monitored numerator cannot be established. Byte-allocation completeness and coverage are independent. Current partial UTC dates can qualify provisionally using elapsed monitored time; habit changes require completed consecutive UTC days with evaluable totals; unknown daily totals block burst/habit checks and Supported confidence; resets/replacements and legacy summaries obey the approved segment and actual-precision rules without changing lifespan math or numeric thresholds.
|
||||
- **TPH-8** (A/M) *Writes-only graph and drill-down*: the TUI renders a writes-only daily bar graph, default 14 days, selectable 7/14/28/90 days, labelled `usage history · Local · UTC±HH:MM · <tz name>`; graph focus supports `←`/`→`, `Enter`, `Esc`/`Backspace`, and `1`/`2`/`3`/`4`, with mouse equivalents for select/drill/back where Textual support is available. Daily bars drill into hourly bars and back. The legend distinguishes allocated `█`, unallocated `▒`, gap `░`, measured zero `·`, partial `┄`, and selection `▼`; selected readout states totals, evidenced hours, unallocated usage, coverage, and partial elapsed facts. Reads graphing is out of scope.
|
||||
- **TPH-9** (A) *Terminal size and graph implementation*: at 80×24 the default range graph and hourly drill-down fit; below 80×24 the graph region hides and shows a one-line textual history summary plus exactly `graph needs ≥80×24`, while titlebox, status reason, drive health, service facts, quit rail/action affordances, and selected-day context survive. The graph uses a custom block-glyph renderable; no new plotting dependency is added for this graph.
|
||||
- **TPH-10** (A/M) *Colour presets, persistence, and reduced motion*: Amber, Nord, and High Contrast presets are available; Amber is the default and keeps the graph amber by default; status semantic colours/glyphs/text outrank theme styling. Preset and reduced-motion choices persist per unprivileged user at `${XDG_CONFIG_HOME:-~/.config}/fenris/tui.json`, not in `/etc/fenris/fenris.conf`, the observation store, helper state, package config, or collector/device configuration; missing preferences default to Amber and normal motion; `t preset` and `m motion` controls plus accessible clickable equivalents are available; preferences never affect collection, projection, history evidence, or `fenris status`.
|
||||
- **TPH-11** (A) *Drive health and settings grouping*: vendor wear renders under Drive health with temperature, spare, media errors, unsafe shutdowns, power-on hours, cycles, capacity, and written-total context, and remains context rather than a second projection. Settings is read-only and limited to device selector, endurance baseline/provenance, retention facts, and TUI display preferences; no custom colour editor exists.
|
||||
@@ -1,122 +0,0 @@
|
||||
# Fenris dashboard clarity specification
|
||||
|
||||
**Status: decision-complete.** Assembled by [Assemble the dashboard clarity specification and close the map](https://git.bongbetic.com/xavierk/Fenris/issues/60) from the closed tickets of the Wayfinder map [Chart Fenris dashboard clarity](https://git.bongbetic.com/xavierk/Fenris/issues/55). This document is normative for the follow-up **execution effort**; nothing here is implemented by the map.
|
||||
|
||||
**Canonical roles.** The [redesign specification](fenris-redesign.md) (frozen) and [ADRs 0001–0007](../adr/) remain authoritative and untouched — this is a companion spec covering five dashboard clarity additions plus the changelog-driven release-notes mechanism. The [criteria register](acceptance-criteria.md) carries the testable statements: **DC-1–DC-8**, appended by this assembly, with **TUI-4's binding list amended** (§4). Terminology follows the glossary in [`CONTEXT.md`](../../CONTEXT.md), including *Deliberate disable* and *Release*.
|
||||
|
||||
**Binding language.** *Must*, *exactly*, and *never* are normative.
|
||||
|
||||
## How to read this document
|
||||
|
||||
Five screen additions (§1–§5), one release-notes mechanism (§6), the verbatim string register (§7), and the README section to add at execution (§8). Each section cites its criteria. Source strings are lowercase; the TUI may render uppercase via styling only. Typography, governing every string: em-dash `—` separates a title from its qualifier; middle dot `·` joins facts within a line; UTF-8 is assumed. CI parity sweeps compare lowercase source strings — rendering case is styling, not wording.
|
||||
|
||||
## 1. Header bar and Bongbetic credit — DC-1
|
||||
|
||||
Visual base is treatment A, quiet integration: the existing Panes information architecture is preserved.
|
||||
|
||||
- The header bar reads `Fenris — NVMe endurance monitor`.
|
||||
- The credit `by Bongbetic` renders dimmed, inline with service facts in the bottom service strip — never in the action row.
|
||||
- Both are TUI-only identity surfaces: `fenris status` never renders them.
|
||||
|
||||
## 2. Continuity line — DC-2
|
||||
|
||||
A labelled `CONTINUITY` row in the service strip (treatment B), mirrored by `fenris status` — the TUI/CLI parity anchor. The row is keyed to the boot fact as-is, independently of run state (Deliberate disable runs `systemctl disable --now`, so paused implies boot-disabled; the row still reports the fact):
|
||||
|
||||
- Active + boot enabled: `monitoring: active in background · persists across reboots`
|
||||
- Boot disabled: `monitoring: does not start on next boot`
|
||||
|
||||
Identical lowercase source strings in the TUI service strip and `fenris status`, including while paused.
|
||||
|
||||
## 3. Paused state block — DC-3
|
||||
|
||||
Treatment C, strong state blocks: when monitoring is paused, a full-width, high-contrast banner clearly identifying Deliberate disable:
|
||||
|
||||
- Title: `monitoring: paused — deliberate disable`
|
||||
- Subline: `paused time is excluded from your usage habit · resume: fenris monitor resume`
|
||||
|
||||
`fenris status` prints the same two lines with identical wording (state line + consequence line). The resume hint uses the CLI form only; the footer owns key hints — no duplication.
|
||||
|
||||
## 4. Quit rail — DC-4 (amends TUI-4)
|
||||
|
||||
Treatment B, labelled rails: a prominent bordered `q QUIT TUI` rail, visually separate from the monitoring-state block and the paused banner. The footer becomes `p pause · r resume · c collect · d disclosures` — the rail owns quit; the footer carries no quit entry. Quitting the TUI never alters monitoring state. The register's TUI-4 binding parenthetical is amended accordingly by this assembly.
|
||||
|
||||
## 5. Launch auth banner — DC-5
|
||||
|
||||
A quiet informational line (treatment A) that never competes with drive state:
|
||||
|
||||
- Text: `privileged actions will prompt for authentication (polkit)`
|
||||
- Full-width under the header at TUI launch; clears on the first refresh tick; never reappears in the session.
|
||||
- TUI-only; `fenris status` never shows it.
|
||||
- Elevation wording is polkit-accurate everywhere: no user-facing string uses "sudo" (sudo belongs to install/upgrade docs).
|
||||
- Evidence class A: a Textual pilot drives refresh ticks headlessly.
|
||||
|
||||
## 6. Changelog and release notes — DC-6, DC-7, DC-8
|
||||
|
||||
Implements the existing glossary term *Release* (tag + packages + change notes together). No new glossary terms; no ADR (reversible mechanism).
|
||||
|
||||
### 6.1 CHANGELOG.md (source of truth, repo root)
|
||||
|
||||
- Keep a Changelog 1.1 shape. `## [Unreleased]` is always present at top, even empty. Version headings are `## [X.Y.Z] - YYYY-MM-DD` — bracketed bare semver, strict ISO date.
|
||||
- Categories are `### Added`, `### Changed`, `### Fixed` only; security fixes fold into Fixed.
|
||||
- Entries are single `- ` bullets, imperative mood, user-facing phrasing; no commit hashes or issue numbers.
|
||||
|
||||
### 6.2 Extraction (release.yml, tag time)
|
||||
|
||||
- `scripts/extract_changelog.py` (checked in, unit-tested): takes the changelog path and a version; slices that version's section verbatim; never reads `[Unreleased]`. Fails closed — `::error::` plus nonzero exit — when the section is missing or empty or the date is malformed.
|
||||
- Guard: the workflow fails when the pushed tag ≠ `v{version from pyproject.toml}` (guard skipped on `workflow_dispatch`).
|
||||
|
||||
### 6.3 Release body
|
||||
|
||||
- Body = extracted version section verbatim + standing footer from `packaging/release-footer.md` (channel install one-liners, `sha256sum -c SHA256SUMS.asc` verify, rollback pointer). The footer is standing text; only the changelog section varies.
|
||||
- Re-run against an existing release: PATCH the body (changelog re-sync is a feature); uploaded assets/packages keep their current idempotent-skip.
|
||||
|
||||
### 6.4 Discipline
|
||||
|
||||
- All entries land in `[Unreleased]` as part of the fixing change — no notes-later step.
|
||||
- One release commit bumps the pyproject version, renames `[Unreleased]` → the version heading, and restores an empty `[Unreleased]`; the tag points at that commit (tag ↔ pyproject ↔ changelog triple-match, enforced fail-closed by DC-7).
|
||||
- No backfill: per-release notes begin with the release shipping this mechanism; `CHANGELOG.md` starts with empty `[Unreleased]`.
|
||||
|
||||
## 7. String register (verbatim)
|
||||
|
||||
### TUI-only strings (launch/identity surfaces)
|
||||
|
||||
| Surface | String |
|
||||
|---|---|
|
||||
| Header bar | `Fenris — NVMe endurance monitor` |
|
||||
| Credit (dimmed, inline with service facts) | `by Bongbetic` |
|
||||
| Auth banner (full-width under header at launch, clears on first refresh tick, never reappears) | `privileged actions will prompt for authentication (polkit)` |
|
||||
| Quit rail (bordered, labelled) | `q QUIT TUI` |
|
||||
| Footer (owns key hints; no quit entry) | `p pause · r resume · c collect · d disclosures` |
|
||||
|
||||
### Parity strings (TUI and `fenris status` identical — CI-2)
|
||||
|
||||
| Surface | String |
|
||||
|---|---|
|
||||
| Continuity, active + boot enabled | `monitoring: active in background · persists across reboots` |
|
||||
| Continuity, boot disabled | `monitoring: does not start on next boot` |
|
||||
| Paused state line | `monitoring: paused — deliberate disable` |
|
||||
| Paused consequence line | `paused time is excluded from your usage habit · resume: fenris monitor resume` |
|
||||
|
||||
`fenris status` prints the paused state line + consequence line when paused, identical wording to the banner title + subline.
|
||||
|
||||
## 8. README section (add at execution)
|
||||
|
||||
The README gains a "Reading the dashboard" section after the CLI reference. Verbatim text:
|
||||
|
||||
```markdown
|
||||
## Reading the dashboard
|
||||
|
||||
`fenris` opens the TUI dashboard. Three things it tells you:
|
||||
|
||||
- **Continuity** — the service strip's continuity line (and `fenris status`) reports whether monitoring survives reboots: `monitoring: active in background · persists across reboots`, or `monitoring: does not start on next boot`.
|
||||
- **Paused vs. quit** — a full-width `monitoring: paused — deliberate disable` block means collection is stopped (`fenris monitor pause`); resume with `fenris monitor resume`. Pressing `q` only leaves the screen — monitoring keeps running in the background.
|
||||
- **Auth banner** — at launch, `privileged actions will prompt for authentication (polkit)` shows once and clears on the first refresh. Privileged actions elevate via polkit; Fenris never asks for sudo.
|
||||
|
||||
Per-release notes live on the [releases page](https://git.bongbetic.com/xavierk/Fenris/releases): each entry is the version's `CHANGELOG.md` section — what was added, changed, and fixed — plus standing install and verification instructions.
|
||||
```
|
||||
|
||||
This resolves the map's README-wording fog: the wording is decided here; the actual README edit is execution.
|
||||
|
||||
## 9. Out of scope
|
||||
|
||||
Executing any of this — code, tests, releases — and any TUI layout or information-architecture redesign beyond the five additions named above. Execution is a fresh effort after handoff.
|
||||
@@ -1,654 +0,0 @@
|
||||
# Fenris redesign specification
|
||||
|
||||
**Status: implementation-ready.** Assembled by [Write the Fenris redesign specification and close the map](https://git.bongbetic.com/xavierk/Fenris/issues/19), executing the assembly decision [Assemble the implementation-ready specification](https://git.bongbetic.com/xavierk/Fenris/issues/17) (all seven recommendations accepted) on the Wayfinder map [Chart Fenris's persistent TUI monitoring redesign](https://git.bongbetic.com/xavierk/Fenris/issues/1).
|
||||
|
||||
**Canonical roles.** [ADRs 0001–0006](../adr/) are the immutable rationale records — the *why*. [Acceptance criteria](acceptance-criteria.md) are the single register of testable statements — the *definition of done*. This document normatively restates every **operative contract** — the *what* — so an implementer never needs Wayfinder-ticket access: schema column sets, constants, rule tables, unit and CLI definitions, and the Panes TUI layout. Nothing here overrides an ADR or restates a criterion as a criterion.
|
||||
|
||||
## How to read this document
|
||||
|
||||
- **Binding language.** *Must*, *exactly*, and *never* are normative. Terminology follows the glossary in [`CONTEXT.md`](../../CONTEXT.md): *observation history*, *usage-adjusted theoretical lifespan*, *projection confidence*, *monitoring period*, *observation store*, *hour observation*, *day aggregate*, *controller segment*, *degraded identity*, *endurance baseline*, *verified override*, *unverified override*, *sustained regime*, *habit change*, *scenario range*, *coverage*, *collection run*, *deliberate disable*, *store fault*.
|
||||
- **Ordering.** Sections follow data flow: system context → collector acquisition → observation store → controller identity & segmentation → hour/day derivation → projection & confidence → Panes TUI → service lifecycle & sanctioned toggle → failure & recovery → installation. Each section opens with its ADR links and criterion-ID block.
|
||||
- **Implementation boundary.** This specification plans the redesign; it does not implement it. The complete handoff is this document + the [criteria register](acceptance-criteria.md) + [ADRs 0001–0006](../adr/) + the glossary. Given/When/Then test specs are derived by the implementer at implementation time.
|
||||
|
||||
### Normative constants index
|
||||
|
||||
Every constant is defined once, in the section named below; other sections cite, never redefine. All are named constants in code, not configuration.
|
||||
|
||||
| Constant | Value | Defined in |
|
||||
|---|---|---|
|
||||
| Collection cadence (default) | 5 min (`OnUnitInactiveSec`) | §8.2 |
|
||||
| First-boot delay | 2 min (`OnBootSec`) | §8.2 |
|
||||
| Timer accuracy window | 30 s (`AccuracySec`) | §8.2 |
|
||||
| Collection-run timeout | 90 s (`TimeoutStartSec`) | §8.2 |
|
||||
| Fresh threshold | newest sample within 2 × cadence + `AccuracySec` + 60 s | §8.9 |
|
||||
| Missed → stale boundary | 48 h | §8.9, §6.7 |
|
||||
| Powered-off hour threshold | power-on-hours delta < 90 % of the hour's wall-clock span | §5.1 |
|
||||
| Active hour threshold | DUW delta ≥ 256 MiB in the hour | §5.1 |
|
||||
| Raw-sample retention | 14 days | §3.4 |
|
||||
| Warming gate | 14 distinct UTC day aggregates, ≤ 2 below 50 % coverage | §6.6 |
|
||||
| Supported coverage floor | 80 % | §6.7 |
|
||||
| Horizon agreement | 7/28/90-day rates within a factor of 2 | §6.7 |
|
||||
| Burst guard | no single day ≥ 50 % of trailing 28-day bytes | §6.7 |
|
||||
| Young-regime cap | regime < 7 days old → Limited | §6.4 |
|
||||
| Habit-change trigger | trailing 7-day mean ≥ 2× or ≤ 0.5× the preceding 28-day mean, 3 consecutive days | §6.4 |
|
||||
| Regime span cap (default) | full observation history capped at 90 days | §6.4 |
|
||||
| Scenario horizons | 7 / 28 / 90 days | §6.5 |
|
||||
| Implied-baseline eligibility | ≥ 2 Percentage-Used increments within the current controller segment | §6.3 |
|
||||
| Rated-TBW conversion | `E_rated = entered_TBW × 10¹²` bytes | §6.3 |
|
||||
| Implied-baseline validity window | 1 ≤ p ≤ 254 | §6.3 |
|
||||
| Wear-disagreement note | vendor wear vs. observed write rate by more than a factor of 2 | §6.1 |
|
||||
| Capacity validation tolerance | ± 1 % | §6.2 |
|
||||
|
||||
---
|
||||
|
||||
## 1. System context
|
||||
|
||||
**ADRs:** [0001](../adr/0001-observation-store-sqlite.md), [0003](../adr/0003-service-lifecycle-and-sanctioned-toggle.md), [0006](../adr/0006-collector-acquisition-path.md). **Criteria:** CI-3, LC-1, LC-5, ST-1.
|
||||
|
||||
### 1.1 Scope
|
||||
|
||||
Fenris observes one configured NVMe drive's real-world use and translates the observation history into a usage-adjusted theoretical lifespan. The redesign replaces the HTML dashboard with a keyboard-first TUI backed by a short-lived privileged collector on a systemd timer, persistent compact observation storage, and categorical projection confidence reflecting the length, completeness, and stability of real usage history.
|
||||
|
||||
Standing constraints, binding on every section:
|
||||
|
||||
- Linux with systemd and polkit only; no other init system is supported.
|
||||
- Exactly one configured NVMe drive — the device named by `/etc/fenris/fenris.conf` (§8.3).
|
||||
- Fully local: no telemetry, no network fetching, no automatic vendor-data retrieval.
|
||||
- The HTML dashboard and HTTP server are gone; nothing of the daemonization, PID files, or `/run` state survives.
|
||||
- CLI `status` and `sample` are retained (§8.8).
|
||||
|
||||
### 1.2 Components and privilege boundaries
|
||||
|
||||
| Component | Privilege | Path | Role |
|
||||
|---|---|---|---|
|
||||
| `fenris-collect.service` | root oneshot unit | `/usr/libexec/fenris/fenris-collect` | The only code path that interrogates the device and writes the observation store. |
|
||||
| `fenris-collect.timer` | system timer | — | Schedules collection runs; `WantedBy=timers.target`. |
|
||||
| `fenris-monitor` | root helper | `/usr/libexec/fenris/fenris-monitor` | Fixed privileged operations: `enable`/`disable` (optional `--now`), the collect trigger, monitoring-period bookkeeping, and baseline persistence. The only binary polkit authorizes. |
|
||||
| `fenris` | unprivileged | `/usr/local/bin/fenris` | Human entry point: no arguments opens the TUI; subcommands are the CLI (§8.8). Never a unit. |
|
||||
| Observation store | root-written, group-read | `/var/lib/fenris/observations.db` | Single SQLite database in WAL mode (§3). The TUI and `status` open it read-only. |
|
||||
| Configuration | world-readable | `/etc/fenris/fenris.conf` | Exactly one key: the device selector (§8.3). |
|
||||
|
||||
The TUI and CLI are ordinary unprivileged processes. Elevation is exclusively polkit, exclusively for `fenris-monitor` (§8.5). There is no `/run/fenris` coordination surface and no export layer: systemd serializes collection runs, the observation store holds state, failures go to the journal.
|
||||
|
||||
### 1.3 Data flow
|
||||
|
||||
1. The timer fires; `fenris-collect.service` runs `fenris-collect`.
|
||||
2. The collector acquires counters and thermal evidence from `smartctl -a -j` and controller identity from sysfs (§2), normalizes identity exactly once (§2.3), and either fails the whole run or writes one complete sample.
|
||||
3. The collector derives and validates hour observations and day aggregates, advances controller segmentation and period bookkeeping, prunes raw samples, and commits (§3–§5, §9.1).
|
||||
4. Readers — the TUI and `fenris status` — open the store read-only and **recompute the projection on every read** (§6); nothing derived is ever stored (§3.7).
|
||||
|
||||
Control flow is separate: the human drives the TUI/CLI; privileged operations route through `fenris-monitor` under polkit to `systemctl`; period rows record *intent* (only the sanctioned path), while the collector records *observed fact* (§8.5–§8.6, §9.8).
|
||||
|
||||
### 1.4 Cross-cutting prohibitions
|
||||
|
||||
These are operative contracts; each is restated in its home section and gated by the criteria block [CI-3](acceptance-criteria.md):
|
||||
|
||||
1. No code path outside `fenris-collect` interrogates the device (§2.1, §8.7).
|
||||
2. Polkit authorizes exactly one binary, `fenris-monitor`, under `com.bongbetic.fenris.monitor` `auth_admin` (§8.5).
|
||||
3. No `/run/fenris` coordination surface or export layer exists (§1.2).
|
||||
4. No absent hour is ever interpolated, estimated, or fabricated (§5.3, §9.3).
|
||||
5. No alerting, notification, or escalation machinery exists anywhere (§9.6–§9.7).
|
||||
6. `/etc/fenris/fenris.conf` holds exactly one key — the device selector (§8.3).
|
||||
7. No synthetic or capacity-derived baseline is ever created, including for legacy history (§6.1, §3.5).
|
||||
8. Readers never partially interpret a newer-schema store (§3.6, §9.5).
|
||||
9. Projections are never stored; always recomputed on read (§3.7, §6.10).
|
||||
|
||||
---
|
||||
|
||||
## 2. Collector acquisition
|
||||
|
||||
**ADR:** [0006](../adr/0006-collector-acquisition-path.md). **Criteria:** AC-1–AC-5; miss absorption per [0005](../adr/0005-failure-detection-and-recovery.md) §5.
|
||||
|
||||
### 2.1 Channels — the hard pin
|
||||
|
||||
Every collection run acquires exactly two ways:
|
||||
|
||||
- **Counters and thermal evidence** — solely from `smartctl -a -j <device>`: `data_units_written`, `data_units_read`, `percentage_used`, `available_spare`, `media_errors`, `power_on_hours`, `power_cycles`, `unsafe_shutdowns`, temperature, `critical_warning` — consumed as-is (smartmontools already trims the strings it copies).
|
||||
- **Controller identity** — solely from sysfs (`/sys/class/nvme/<ctrl>/`): `subnqn`, `sn`, `mn`, `fr`, `transport`.
|
||||
|
||||
No other acquisition path exists anywhere in the codebase. There is no fallback: libnvme bindings and the `nvme` CLI JSON interface are excluded (ADR 0006, *Considered options*).
|
||||
|
||||
### 2.2 All-or-nothing runs
|
||||
|
||||
Any acquisition failure — missing `smartctl` binary, nonzero exit, malformed JSON, unreadable sysfs attribute — fails the **whole** collection run. A partial sample (identity without counters, or counters without identity) is never written: a transient read failure must never push a healthy drive down the degraded-identity path (§4). The miss surfaces through freshness grading (§8.9) and the flat retry cadence (§9.6), never as degraded identity.
|
||||
|
||||
### 2.3 Identity normalization — once, at write time
|
||||
|
||||
One collector-side function normalizes every identity field, applied exactly once at write time:
|
||||
|
||||
- strip trailing spaces and newlines;
|
||||
- no case folding;
|
||||
- empty-after-strip is stored blank.
|
||||
|
||||
Padded and unpadded renderings of the same field therefore yield byte-identical stored values — a collector implementation change can never split a drive's own history. A future acquisition-path change must deliver byte-identical normalized identity values, or the change itself forces a controller-segment boundary.
|
||||
|
||||
### 2.4 Segment metadata sourcing
|
||||
|
||||
`transport` comes from the NVMe class sysfs directory. `vid`/`ssvid` come from the PCI node (`/sys/class/nvme/<ctrl>/device/{vendor,subsystem_vendor}`) when present and are stored null otherwise. Both are segment **metadata only** (§4.2), never key components.
|
||||
|
||||
### 2.5 Prerequisites
|
||||
|
||||
`make install` verifies `smartctl` is present and fails cleanly otherwise (§10.1). The acquisition path adds no Python dependency and no OS package beyond smartmontools; the dependency lockfile (§10.5) is untouched by this section.
|
||||
|
||||
---
|
||||
|
||||
## 3. Observation store
|
||||
|
||||
**ADR:** [0001](../adr/0001-observation-store-sqlite.md) as amended by [Define endurance-baseline provenance and validation](https://git.bongbetic.com/xavierk/Fenris/issues/12) and [Decide controller-segment metadata columns](https://git.bongbetic.com/xavierk/Fenris/issues/14). **Criteria:** ST-1–ST-12; FL-5.
|
||||
|
||||
### 3.1 Substrate and access
|
||||
|
||||
- One SQLite database in **WAL mode** at `/var/lib/fenris/observations.db`. An unprivileged reader querying during a collector write sees a consistent snapshot.
|
||||
- The database is root-owned and group-readable through the `fenris` read group created by packaging; the TUI and `status` open it **read-only**. No `/run` snapshot, no export layer.
|
||||
- `/var/lib/fenris` is created by the installer with root-written group-read permissions; the database file itself is created lazily by the first write, so "no observations yet" remains a real state the TUI can greet (§7.6, §10.1).
|
||||
- Migration, schema changes, prune, and import are each single transactions — a killed timer run can never leave partial state.
|
||||
|
||||
### 3.2 Entities and column sets
|
||||
|
||||
The schema carries exactly six entities:
|
||||
|
||||
**`samples`** — recent raw samples (14-day retention, §3.4): timestamp (UTC); the normalized controller-identity fields captured at acquisition (§2.3); raw integer `data_units_written`, `data_units_read`; `percentage_used`; `available_spare`; `media_errors`; `power_on_hours`; `power_cycles`; `unsafe_shutdowns`; temperature; `critical_warning`.
|
||||
|
||||
**`hour_observations`** — one row per UTC hour: the usage-habit split `seconds_active`, `seconds_idle`, `seconds_powered_off`, `seconds_unknown` (summing to 3600, §5.1); DUW/DUR deltas; temperature min/avg/max; sample count; coverage flag. Classification thresholds belong to the projection model (§5.1), not the store.
|
||||
|
||||
**`day_aggregates`** — one row per UTC day, the habit-evidence grain: each day row carries, at minimum, the day's activity-split sums, write deltas, and coverage share — the inputs the evidence gates of §6.6 consume — derived monotonically from its hour rows.
|
||||
|
||||
**`monitoring_periods`** — `started_at`; `ended_at` (NULL = open); `end_cause` enum (`user_disabled`, `migrated`, …). Powered-off time stays inside a period; deliberately disabled time does not (§5.2, §8.6).
|
||||
|
||||
**`controller_segments`** — spans of unchanged controller identity and monotonic counters; write deltas are never computed across a segment boundary. Columns: the identity key (§4.1) and the frozen metadata snapshot of §4.2, plus the segment's span bounds.
|
||||
|
||||
**`endurance_baseline`** — one active row, replaced on edit (§6.2): the rated-TBW value in bytes (`E_rated = entered_TBW × 10¹²`); mandatory provenance — source URL, document revision, entry date, model string, nominal capacity; frozen validation facts — detected model, detected capacity bytes, `validated_by` (`machine`/`user`), `validated_at`.
|
||||
|
||||
### 3.3 Time model
|
||||
|
||||
Hours and days are UTC-bounded. Day derivation from hour rows is monotonic; DST-ambiguous 23- or 25-hour days never exist in the store.
|
||||
|
||||
### 3.4 Retention
|
||||
|
||||
Raw samples are pruned opportunistically by the collector to **14 days**. Hour observations and day aggregates are retained indefinitely.
|
||||
|
||||
### 3.5 Legacy migration
|
||||
|
||||
The migration procedure, invoked from the entry points below, is **idempotent and interruption-safe**:
|
||||
|
||||
1. If the store already carries the legacy-import marker, do nothing.
|
||||
2. `history.jsonl` is the sole authority: import raw samples and derive hour observations and day aggregates from them.
|
||||
3. `hourly.jsonl` is never trusted as input: mismatches against derived data are diffed and logged.
|
||||
4. Open one implicit `monitoring_periods` row at the first legacy sample, closed `end_cause = migrated` at the migration moment. Pre-migration hours carry an unknown activity split except directly evidenced facts — a sample present means powered on; a DUW delta means writes occurred.
|
||||
5. The import is a single transaction: a scripted kill mid-import leaves the store fully pre- or fully post-migration.
|
||||
6. Only after commit are legacy files renamed `*.migrated` — never deleted.
|
||||
7. Malformed legacy lines are quarantined with a logged count, never silently dropped.
|
||||
|
||||
No synthetic or capacity-derived baseline is ever created for legacy history (§1.4–7). Entry points: the installer's import detection at `./data/history.jsonl` (or an explicit path) (§10.1); `fenris import <path>` for later finds (§8.8); and the collector's first new-version run, which performs this same procedure (ADR 0001 §6).
|
||||
|
||||
### 3.6 Schema versioning
|
||||
|
||||
`PRAGMA user_version` plus ordered migration steps in code, each in its own transaction. The collector refuses to run against an unknown **newer** version; readers refuse symmetrically with the exact wording of §9.5 and never partially interpret. *Reconciliation note:* ADR 0004 §6 describes upgrade-time migrations as "governed by a `schema_version` table" — the operative mechanism is this section's `user_version` (ADR 0001 §8, criterion ST-12); there is one version authority, not two.
|
||||
|
||||
### 3.7 Nothing derived is stored
|
||||
|
||||
Projections are not stored; there is no separate latest-status table and no stored health flag. The freshest sample timestamp is the store's own staleness signal (§8.9). The baseline lives in the database (§6.2); `/etc/fenris/` holds only operational configuration (§8.3).
|
||||
|
||||
---
|
||||
|
||||
## 4. Controller identity and segmentation
|
||||
|
||||
**Decisions:** [Verify the controller identity that segments observation history](https://git.bongbetic.com/xavierk/Fenris/issues/11), [Decide controller-segment metadata columns](https://git.bongbetic.com/xavierk/Fenris/issues/14), [Decide how degraded identity affects projection confidence](https://git.bongbetic.com/xavierk/Fenris/issues/15). **ADRs:** [0001](../adr/0001-observation-store-sqlite.md) §3 (as amended), [0002](../adr/0002-projection-model-sustained-regime.md) §§8–9 (as amended). **Criteria:** ID-1–ID-4, PR-9, PR-15, PR-16.
|
||||
|
||||
### 4.1 Identity key ladder
|
||||
|
||||
The controller-segment identity key is the **normalized, kernel-exposed subsystem NQN** (`subnqn`), with fallbacks, in order:
|
||||
|
||||
1. kernel-exposed subsystem NQN;
|
||||
2. the kernel composite;
|
||||
3. `model|serial`.
|
||||
|
||||
`fr` (firmware revision) is metadata only — it may go stale after a mid-segment firmware update. Identity change and DUW decrease are **independent axes** (§4.3).
|
||||
|
||||
### 4.2 Frozen metadata snapshot
|
||||
|
||||
Each segment freezes, at open, a fully nullable metadata snapshot — immutable thereafter: normalized `subnqn`, `sn`, `mn`, `fr`, plus `vid`, `ssvid`, `transport`, and the `identity_degraded` flag. All columns are nullable so incompleteness stays explicit: legacy-imported segments carry `mn` with NULLs (§4.4); degraded segments carry whatever was observed. These are human diagnostics, never key components. `cntlid` is excluded — it distinguishes controllers within one subsystem, out of scope for a single-drive monitor.
|
||||
|
||||
### 4.3 Segmentation axes
|
||||
|
||||
- **DUW decrease, unchanged identity** — a segment boundary within the same drive. Prior day aggregates remain habit evidence; the projection is Unavailable only until the new segment re-warms (§6.8).
|
||||
- **Identity-key change** — quarantines prior history from projection entirely: it describes a different drive (§6.8).
|
||||
- **Degraded identity** — a segment whose identity key is **blank** (every rung of the ladder empty). `identity_degraded` is set at segment open exactly when the key is blank; keys from the kernel-composite or `model|serial` rungs are not degraded. Blank-key semantics extend identity-change rules verbatim: any visible change of the recorded key — including to or from blank — is a controller-identity change and quarantines; equal blank keys continue the segment, segmented by DUW monotonicity alone. Even a degraded→healthy transition quarantines, so the projection window only ever spans segments sharing one key (§6.8).
|
||||
- Ephemeral markers (model "Linux", non-pcie transport) are segment metadata, never confidence facts.
|
||||
|
||||
The confidence consequence of degraded identity — capped at Limited with its fixed contributing fact — is §6.7's rule.
|
||||
|
||||
### 4.4 Legacy identity
|
||||
|
||||
Legacy history imports under a labeled, model-scoped **legacy identity** (mn-only segments), so it never blends with the post-redesign identity of the same physical drive.
|
||||
|
||||
---
|
||||
|
||||
## 5. Hour and day derivation
|
||||
|
||||
**ADRs:** [0002](../adr/0002-projection-model-sustained-regime.md) §§4–6; [0001](../adr/0001-observation-store-sqlite.md) §3; [0003](../adr/0003-service-lifecycle-and-sanctioned-toggle.md) §2 (power-on-hours evidence). **Criteria:** PR-4–PR-6, ST-4, FL-3.
|
||||
|
||||
### 5.1 Hour classification
|
||||
|
||||
Each UTC hour is classified by named constants, in this order of evidence:
|
||||
|
||||
- **Powered-off** — the hour's power-on-hours delta is below **90 %** of its wall-clock span.
|
||||
- **Active** — DUW delta ≥ **256 MiB** in the hour.
|
||||
- **Idle** — powered on, sampled, below the active threshold.
|
||||
- **Unknown** — everything else: unsampled without power-on-hours evidence (machine-off and collector failure are indistinguishable by design), or inconsistent counters.
|
||||
|
||||
There is no configuration surface for these thresholds; they are documented constants in one projection module.
|
||||
|
||||
### 5.2 Denominator and disabled time
|
||||
|
||||
The projection denominator is **wall-clock seconds inside monitoring periods**, including powered-off and unknown time. Disabled periods — wall-clock outside monitoring periods — are excluded from numerator and denominator. **Disabled time is not an hour state.**
|
||||
|
||||
### 5.3 Gaps and coverage — never backfill
|
||||
|
||||
No absent hour is ever interpolated, estimated, or fabricated. Unexplained gaps inside a period keep the aggregate counter delta, remain in the denominator as unknown seconds, and reduce coverage. Power-on-hours classification (§5.1) is the only inference admitted. **Coverage** is the share of wall-clock seconds inside monitoring periods whose classification is known rather than unknown — a first-class displayed fact (§6.10, §7.3).
|
||||
|
||||
### 5.4 Day aggregates
|
||||
|
||||
One row per UTC day, derived monotonically from hour rows (§3.2–§3.3) — the grain at which usage-habit evidence is judged (§6.6).
|
||||
|
||||
---
|
||||
|
||||
## 6. Projection and confidence
|
||||
|
||||
**ADRs:** [0002](../adr/0002-projection-model-sustained-regime.md) as amended by [Decide how degraded identity affects projection confidence](https://git.bongbetic.com/xavierk/Fenris/issues/15); baseline per [Define endurance-baseline provenance and validation](https://git.bongbetic.com/xavierk/Fenris/issues/12). **Criteria:** PR-1–PR-17, CI-4.
|
||||
|
||||
### 6.1 One projection; baseline precedence
|
||||
|
||||
Exactly **one** usage-adjusted theoretical lifespan is computed, against the endurance baseline chosen by precedence:
|
||||
|
||||
1. **Verified override** — a rated-TBW override with complete provenance whose applicability to the detected drive was confirmed by machine match or explicit user attestation;
|
||||
2. **Unverified override** — a rated-TBW override knowingly stored with incomplete provenance; always presented as user-supplied, never as verified;
|
||||
3. **Implied baseline** — derived from vendor wear (§6.3), eligible only per §6.3's gate;
|
||||
4. otherwise the projection is **Unavailable**.
|
||||
|
||||
Percentage Used is context, never a second projection: it renders as a vendor-wear context line, and when the wear it implies disagrees with the observed write rate by more than a factor of 2, a note says so. The legacy PU-slope regression and `capacity × 600` synthesis are gone; no synthetic or capacity-derived baseline is ever created (§1.4–7).
|
||||
|
||||
### 6.2 Endurance baseline: provenance and validation
|
||||
|
||||
The baseline lives in the observation store's `endurance_baseline` table (§3.2) and is edited via the CLI (§8.8) — `/etc/fenris/` holds no baseline.
|
||||
|
||||
- **Mandatory provenance:** source URL, document revision, entry date, model string, nominal capacity.
|
||||
- **One active row**, replaced on edit.
|
||||
- **Verification is derived at read** — complete provenance and a drive match (machine or attested) — never a stored boolean.
|
||||
- **Unverified tier:** incomplete provenance stores only behind an explicit unverified acknowledgment, as NULL fields in that precedence tier.
|
||||
- **Entry-time validation** (unprivileged, live sysfs read of the configured device): normalized model containment, with an interactive confirm recorded as `validated_by = user`; nominal capacity within ± 1 %.
|
||||
- **Read-time applicability:** a model match against the current controller segment (§4). A mismatch is **retained — never auto-deleted** — and leaves the projection Unavailable.
|
||||
- Persistence goes through the polkit-guarded `fenris-monitor` verb after CLI-side validation (§8.5).
|
||||
|
||||
### 6.3 Arithmetic
|
||||
|
||||
```text
|
||||
rate = regime DUW delta bytes / in-period wall-clock seconds
|
||||
projected = max(E_baseline − W_t, 0) / rate (rate > 0)
|
||||
E_rated = entered_TBW × 10¹² bytes
|
||||
E_implied = 100 · W_t / p (1 ≤ p ≤ 254)
|
||||
```
|
||||
|
||||
- `E_rated` is exact: rated TBW converts to bytes by × 10¹².
|
||||
- `E_implied` is computed **only** for `1 ≤ p ≤ 254`; Percentage Used of 0 or saturated 255 implies no baseline — that precedence tier is unavailable. The implied baseline is labeled *implied from vendor wear estimate* and shown with few significant digits.
|
||||
- **Implied-baseline eligibility:** the implied tier is used only after ≥ 2 Percentage-Used increments within the current controller segment; until then the projection is Unavailable with the fixed phrase *"vendor wear estimate too coarse to imply endurance"*.
|
||||
|
||||
### 6.4 Sustained regime and habit change
|
||||
|
||||
The headline rate is the **sustained-regime** rate: regime DUW bytes ÷ in-period wall-clock seconds. The default regime is the full observation history capped at **90 days**.
|
||||
|
||||
A **habit change** is declared when the trailing 7-day mean of daily written bytes stays ≥ 2× (or ≤ 0.5×) the mean of the preceding 28 days for **3 consecutive days**. The new regime starts at the **first day of divergence**, is adopted automatically, and is labeled *"usage habit changed N days ago"*; the scenario range keeps the longer horizons visible. A regime younger than **7 days** caps projection confidence at Limited evidence.
|
||||
|
||||
### 6.5 Scenario range
|
||||
|
||||
The 7-, 28-, and 90-day rates are computed **independently of the regime** and shown as the scenario range. Only horizons the history actually covers appear — no placeholders. The scenario range is the only spread shown anywhere (§6.9).
|
||||
|
||||
### 6.6 Minimum evidence
|
||||
|
||||
Warming up until **14 distinct UTC day aggregates** of which at most **2** fall below 50 % coverage. The projection still renders while warming up, labeled with its facts (e.g. *"warming up: N of 14 qualifying days"*). Every Unavailable condition renders **no lifespan number**.
|
||||
|
||||
### 6.7 Confidence rule table
|
||||
|
||||
Confidence renders as **state plus contributing facts, never a percentage**. Three states:
|
||||
|
||||
- **Unavailable** — no applicable baseline; DUW unsupported; zero rate over the regime; controller-identity change.
|
||||
- **Supported** — verified baseline **and** ≥ 14 qualifying days **and** coverage ≥ 80 % **and** fresh (< 48 h) **and** 7/28/90 rates within a factor of 2 across existing horizons **and** no single day ≥ 50 % of trailing 28-day bytes **and** regime ≥ 7 days old **and** the current controller segment's identity key is not degraded.
|
||||
- **Limited** — every other case with a baseline and a positive rate; the failing facts are shown.
|
||||
|
||||
**Staleness:** a newest day aggregate older than **48 hours** drops confidence one level (Supported → Limited) and is shown as a contributing fact.
|
||||
|
||||
**Degraded identity:** a controller segment whose identity key is blank (§4.3) caps confidence at **Limited evidence**, with the contributing fact *"controller identity unavailable — replacement detection relies on write-counter continuity only"* rendered in every state. Supported is unreachable while the current segment is degraded. The cap combines idempotently with the staleness drop (both land at Limited).
|
||||
|
||||
### 6.8 Segment-break effects
|
||||
|
||||
- **DUW decrease, unchanged identity:** prior day aggregates remain habit evidence; the projection is Unavailable only until the new segment re-warms (§6.6).
|
||||
- **Controller-identity change** — including any to-or-from-blank key change (§4.3): prior history is quarantined from projection entirely.
|
||||
- Since even degraded→healthy transitions quarantine, the projection window only ever spans segments sharing one key; no cross-segment propagation rule is needed.
|
||||
|
||||
### 6.9 Zero rate and uncertainty
|
||||
|
||||
Zero rate renders *"no finite projection from this history"* — never infinity, never zero. No statistical confidence interval appears anywhere; the scenario range is the only spread.
|
||||
|
||||
### 6.10 The projection contract
|
||||
|
||||
The projection function hands the TUI and `status` exactly: the confidence state; the contributing facts — including the degraded-identity fact when the current segment's key is blank; the headline remaining time when one exists; the scenario range; the Percentage-Used context line; the disclosure text (§6.11). Recomputed on read, never stored.
|
||||
|
||||
### 6.11 User-facing language
|
||||
|
||||
Adopted from the endurance research as fixed by ADR 0002 §12; rendered identically by TUI and `status`.
|
||||
|
||||
**Headline wording** (equivalent phrasing required):
|
||||
|
||||
> Estimated time until the selected host-write endurance baseline is consumed, if future write usage resembles the observed usage habit. This is not a predicted hardware-failure date.
|
||||
|
||||
**Fixed phrases** (exact): *no finite projection from this history* (zero rate); *vendor wear estimate too coarse to imply endurance* (§6.3); *usage habit changed N days ago* (§6.4); *controller identity unavailable — replacement detection relies on write-counter continuity only* (§6.7); *observation store unreadable* (§9.4); *observation store written by a newer Fenris — upgrade Fenris* (§9.5); *no observations yet* with an enable hint (§8.9); *configuration error: ⟨reason⟩* (§8.3).
|
||||
|
||||
**Confidence rendering:** state plus contributing facts, in the research's evidence style, e.g.
|
||||
|
||||
> Supported evidence · verified manufacturer TBW · 42 calendar days · 96 % interval coverage · 6 weekly cycles · recent and 28-day rates agree
|
||||
|
||||
Never "82 % confidence" or "95 % accurate".
|
||||
|
||||
**The six disclosures** (verbatim, always available — TUI disclosures view and `status`):
|
||||
|
||||
1. This is an endurance projection, not a predicted hardware-failure date.
|
||||
2. Percentage Used is vendor-specific; 100 means estimated endurance consumed but may not mean failure, it can exceed 100, and 255 is saturated.
|
||||
3. Rated TBW can be a warranty/endurance threshold with separate time and eligibility terms, not a failure threshold.
|
||||
4. DUW is upward-rounded host writes excluding metadata and selected commands, not exact physical NAND writes.
|
||||
5. Projection quality depends on baseline provenance, history duration and completeness, recentness, stability, and representative usage cycles; future workload and firmware behavior remain outside the observed evidence.
|
||||
6. Gaps can preserve an aggregate counter delta without preserving hourly timing; unexplained and deliberately disabled periods must be distinguished.
|
||||
|
||||
---
|
||||
|
||||
## 7. Panes TUI
|
||||
|
||||
**Decisions:** [Prototype the TUI information architecture](https://git.bongbetic.com/xavierk/Fenris/issues/3) (Variant A adopted), [Evaluate Python TUI frameworks](https://git.bongbetic.com/xavierk/Fenris/issues/6) (Textual). **ADRs:** [0003](../adr/0003-service-lifecycle-and-sanctioned-toggle.md) §§8, 10; [0004](../adr/0004-install-upgrade-removal-lifecycle.md) §10. **Criteria:** TUI-1–TUI-4, CI-1, CI-2, CI-4. The [prototype](https://git.bongbetic.com/xavierk/Fenris/src/branch/prototype/tui-information-architecture/prototype/tui-ia) is visual reference only; this section is normative.
|
||||
|
||||
### 7.1 Framework and floor
|
||||
|
||||
The TUI is built on **Textual**. It runs on Python 3.9+, gated at install time (§10.1) — never a runtime crash. The tty-passthrough mechanism below was validated live under Textual on a real terminal (prototype decision).
|
||||
|
||||
### 7.2 Layout — one dense keyboard-first screen
|
||||
|
||||
Variant A **Panes**: everything on one screen, no page navigation. The screen is a grid of four regions:
|
||||
|
||||
1. **Headline band** — full width, top: the lifespan headline (or its no-projection wording) with its regime line (*"if current habits continue · sustained regime: N days at R GB/day"*); the confidence state with contributing facts; the scenario range.
|
||||
2. **Usage-history pane** — left, wider column: the write-history sparkline with ▲ habit-change and ? unexplained-gap markers plus their legend; the habit-split bar with active/idle/powered-off/unknown shares.
|
||||
3. **Drive-health and settings pane** — right, narrower column: health facts (model, temperature, spare, media errors, unsafe shutdowns, power-on hours, power cycles, capacity); the vendor-wear context line (Percentage Used · total written of rated — *context, not a second projection*); a read-only settings view (device selector, cadence with drop-in pointer, raw retention, endurance baseline value with its provenance label). Edits happen via CLI / drop-ins, not in the TUI.
|
||||
4. **Service strip** — full width, bottom: the four separate service facts (§7.3), the monitoring-period line, the action legend.
|
||||
|
||||
Exact proportions, glyphs, and borders follow the prototype's validated arrangement as visual reference; the region arrangement, contents, and bindings above are normative.
|
||||
|
||||
### 7.3 Content contracts per region
|
||||
|
||||
- **Confidence is evidence:** state plus contributing facts, never a percentage (§6.7); the headline band renders the §6.10 contract in full, including the disclosure affordance (`d`).
|
||||
- **Four separate service facts, always:** boot enablement (enabled/disabled) · runtime activity (timer active/inactive) · last collect outcome (ok/FAILED, age, reason) · freshness (fresh/missed/stale with newest-sample age, §8.9). They are never merged into one "service status".
|
||||
- **Monitoring-period line:** open-since / closed with end cause; deliberate-disable count where nonzero.
|
||||
- The scenario range shows only covered horizons (§6.5); the vendor-wear context line carries the >2× disagreement note when it applies (§6.1).
|
||||
|
||||
### 7.4 Keybindings and asymmetry
|
||||
|
||||
Production bindings:
|
||||
|
||||
| Key | Action |
|
||||
|---|---|
|
||||
| `p` | Pause — **asks for confirmation** (y pause · n cancel), stating that paused time is excluded from the usage habit while powered-off time would still count. |
|
||||
| `r` | Resume — **no confirmation** (benign; friction invites raw-systemctl escapes). |
|
||||
| `c` | Collect now — synchronous outcome (§8.7), no confirmation. |
|
||||
| `d` | Disclosures — the six disclosures of §6.11. |
|
||||
| `q` | Quit. |
|
||||
|
||||
No bare start/stop exists anywhere; no page navigation keys exist (variant switching was prototype-only). Framework defaults apply for focus and scrolling otherwise.
|
||||
|
||||
### 7.5 Privileged actions and tty passthrough
|
||||
|
||||
Pause, resume, collect-now, and baseline operations run through `fenris-monitor` as a **terminal-attached subprocess**: the TUI suspends, the platform polkit agent prompts on the real terminal, and control returns cleanly with the outcome reflected in the service facts. Where no polkit agent exists the operation fails cleanly with the printed root equivalent (§8.5).
|
||||
|
||||
### 7.6 State rendering obligations
|
||||
|
||||
- From a synthetic observation store, the TUI renders **every** realizable combination of confidence state × freshness grade × baseline tier exactly as the §6.7 rule table and §8.9 constants dictate — headline number only when the rules allow it, contributing facts always, never a percentage (criterion CI-1).
|
||||
- Empty store: *"no observations yet"* with an enable hint; the first-run prompt is an opt-in that enables the timer and opens the first period in one step (§10.1, dormant install).
|
||||
- A `configuration error: ⟨reason⟩` fact renders when the device selector is invalid (§8.3); a store fault suppresses everything store-dependent (§9.4); a newer schema renders its fixed phrase (§9.5).
|
||||
- Every TUI action has a CLI twin with identical outcomes and wording (§8.8, CI-2).
|
||||
|
||||
---
|
||||
|
||||
## 8. Service lifecycle and sanctioned toggle
|
||||
|
||||
**ADR:** [0003](../adr/0003-service-lifecycle-and-sanctioned-toggle.md) as amended by [Define endurance-baseline provenance and validation](https://git.bongbetic.com/xavierk/Fenris/issues/12). **Criteria:** LC-1–LC-10, CI-2, CI-3.
|
||||
|
||||
### 8.1 Units
|
||||
|
||||
Exactly two system units exist:
|
||||
|
||||
- `fenris-collect.timer` — `WantedBy=timers.target`.
|
||||
- `fenris-collect.service` — `Type=oneshot`, root, `ExecStart=/usr/libexec/fenris/fenris-collect`; no listener, no UI code.
|
||||
|
||||
The TUI and CLI are ordinary unprivileged processes and never units.
|
||||
|
||||
### 8.2 Cadence
|
||||
|
||||
Shipped defaults: `OnBootSec=2min`, `OnUnitInactiveSec=5min` (measured from run completion; drift accepted because hours are the evidence grain), `AccuracySec=30s`, `Persistent=no` (no suspend catch-up — absent hours classify through power-on-hours evidence, §5.1), `TimeoutStartSec=90s` so a hung device interrogation fails visibly as a bounded failed run retried next interval. Cadence changes are documented drop-ins on the timer unit (`systemctl edit` + daemon-reload); **no interval key exists in configuration**.
|
||||
|
||||
### 8.3 Configuration
|
||||
|
||||
`/etc/fenris/fenris.conf` holds exactly one key: the **device selector**, a stable `/dev/disk/by-id/…` path (raw nodes accepted with an instability warning), validated at collection time. The oneshot re-reads it every run — there is no reload path. An invalid selector is a bounded failed run (journal + failed unit result, retried next interval); `status` and the TUI also read the world-readable file directly and surface `configuration error: ⟨reason⟩`.
|
||||
|
||||
### 8.4 Entry points
|
||||
|
||||
Two privileged binaries — `/usr/libexec/fenris/fenris-collect` (device interrogation and store writes; the unit's `ExecStart`) and `/usr/libexec/fenris/fenris-monitor` (fixed operations `enable`/`disable` with optional `--now`, the collect trigger, monitoring-period bookkeeping, and `baseline set`/`baseline clear` persistence for the CLI-validated baseline; the only binary the polkit policy authorizes). One unprivileged `fenris` wrapper (§1.2). Root invokes the helpers directly; unprivileged users go through polkit.
|
||||
|
||||
### 8.5 Sanctioned toggle and polkit
|
||||
|
||||
- Pause = `fenris-monitor disable --now`; Resume = `enable --now`. Both perform the systemctl operation **and** the monitoring-period bookkeeping in one step. The human-facing twins `fenris monitor pause` / `fenris monitor resume` map to these and always act immediately; pause asks for confirmation in both TUI and CLI, resume does not (§7.4).
|
||||
- Polkit action `com.bongbetic.fenris.monitor` (`auth_admin`) covers the toggle **and** the collect trigger **and** baseline persistence — authorizing exactly the one binary `fenris-monitor`.
|
||||
- Where no polkit agent exists the operation fails cleanly and prints the root equivalent.
|
||||
- This is the **only** sanctioned control path: a raw `systemctl stop`/`disable` never records `user_disabled` — only the sanctioned path records intent (§8.6).
|
||||
|
||||
### 8.6 Period-row idempotent matrix
|
||||
|
||||
| Situation | Effect on `monitoring_periods` |
|
||||
|---|---|
|
||||
| First-ever enable | Opens a period at the enable moment (hours before the first successful sample are unknown-but-inside — correct when the device errors). |
|
||||
| Resume with an open period (a raw `systemctl stop` intervened) | No row changes; the gap remains inside as unknown seconds. |
|
||||
| Resume with no open period | Opens a new row at the resume moment. |
|
||||
| Pause with an open period | Closes it `user_disabled` at the pause moment. |
|
||||
| Pause otherwise | No-op. |
|
||||
| Raw `systemctl stop`/`disable` outside the helper | An unexplained gap, never `user_disabled`. |
|
||||
|
||||
### 8.7 On-demand collection
|
||||
|
||||
`fenris sample` and the TUI's collect-now route through `fenris-monitor` → `systemctl start fenris-collect.service`, which blocks until the oneshot exits; the outcome (freshness line or journal hint) is reported synchronously. No confirmation is required. No code path outside `fenris-collect` touches the device; the TUI never samples in-process.
|
||||
|
||||
### 8.8 CLI surface
|
||||
|
||||
| Command | Behavior |
|
||||
|---|---|
|
||||
| `fenris` (no arguments) | Opens the TUI (§7). |
|
||||
| `fenris status` | Read-only composition of the observation store and allow-listed `systemctl show` properties: projection facts, enabled/active, last collect outcome, and a `journalctl -u fenris-collect.service` hint on failure or staleness. Never auto-samples, never prompts. |
|
||||
| `fenris sample` | On-demand collection via the helper path (§8.7). |
|
||||
| `fenris monitor pause` / `resume` | The sanctioned toggle (§8.5), pause asking confirmation. |
|
||||
| `fenris baseline set` / `clear` | CLI-side validation (§6.2), then polkit-guarded persistence. |
|
||||
| `fenris import ⟨path⟩` | The idempotent single-transaction legacy import (§3.5). |
|
||||
| `--device` | Rejected with a pointer to the configuration file. |
|
||||
| `start`, `stop`, `run` | Rejected with one-line migration pointers — never aliased (an alias would silently change meaning). |
|
||||
|
||||
`fenris.sh` is retired: not shipped, removed from the repository; the README maps its five menu options to their successors. Headless administration has full parity: every TUI action has a CLI twin (pause, resume, collect-now, baseline set/clear, the status fact set) with identical outcomes and wording.
|
||||
|
||||
### 8.9 Freshness grading
|
||||
|
||||
Constants defined once, consumed by TUI and CLI alike; the grade derives from the **newest sample timestamp**, never a stored flag:
|
||||
|
||||
- **fresh** — newest sample within 2 × cadence + `AccuracySec` + 60 s (11.5 min at default cadence);
|
||||
- **missed** — between that and 48 h (a contributing fact);
|
||||
- **stale** — ≥ 48 h, matching the §6.7 evidence gate;
|
||||
- **empty store** — *"no observations yet"* with an enable hint.
|
||||
|
||||
---
|
||||
|
||||
## 9. Failure and recovery
|
||||
|
||||
**ADR:** [0005](../adr/0005-failure-detection-and-recovery.md). **Criteria:** FL-1–FL-8.
|
||||
|
||||
The posture: **visible degradation, never fabrication.**
|
||||
|
||||
### 9.1 Write-boundary validation
|
||||
|
||||
The collector validates every row it would write against the store's domain invariants: hour seconds sum to 3600; non-negative DUW delta within a controller segment; coverage consistent with sample count. A violating run **writes nothing**, logs the refused row to the journal for post-mortem, and fails visibly — retried next interval. Store invariant: everything persisted is well-formed.
|
||||
|
||||
### 9.2 Reader defense
|
||||
|
||||
Readers (TUI, `status`) defensively exclude and count malformed rows as a contributing fact. Under a single trusted writer they should never see one.
|
||||
|
||||
### 9.3 No backfill, ever
|
||||
|
||||
Gaps remain unknown seconds; degradation flows exclusively through coverage, freshness facts, and confidence categories; recovery is the timer's next successful run. Power-on-hours classification (§5.1) is the only inference admitted.
|
||||
|
||||
### 9.4 Store faults — degrade, never recreate over
|
||||
|
||||
An unreadable or corrupt database is a store fault: readers surface *"observation store unreadable"* with the journal hint and show nothing else that depends on the store; the collector treats it as a bounded failed run and **never recreates or overwrites** an existing file. Recovery is human-sanctioned and documented: back up or move the corrupt file aside; the next run starts a fresh store; if the legacy import never completed, the still-present `history.jsonl` is re-imported (§3.5). No built-in destructive command exists.
|
||||
|
||||
### 9.5 Newer schema — symmetric refusal
|
||||
|
||||
The TUI and `status` detect a `user_version` newer than they understand and display *"observation store written by a newer Fenris — upgrade Fenris"* without partial interpretation, matching the collector's refusal (§3.6) and the forward-only upgrade rule (§10.2).
|
||||
|
||||
### 9.6 Repeated collector failures — flat cadence, no escalation
|
||||
|
||||
The timer's retry is the recovery path; the freshness grading walks fresh → missed → stale as failures persist, so degradation is visible without new state. No backoff, no notification machinery; a persistent failure reads as stale exactly like any other gap.
|
||||
|
||||
### 9.7 Drive-reported anomalies — facts, not alerts
|
||||
|
||||
`critical_warning`, media errors, and unsafe shutdowns surface as ordinary facts in the TUI and `status` (§7.2); no alerting or notification surface exists. The projection is unaffected: endurance math consumes writes, not warnings.
|
||||
|
||||
### 9.8 Orphaned samples — the collector re-anchors observed fact
|
||||
|
||||
When a collection run finds no open monitoring period (fresh store after a store fault, completed legacy re-import, or first-ever run), it opens one at the **run moment**, never backdated. This records observed fact, not intent — only the sanctioned path records a `user_disabled` close (§8.5). Coverage semantics stay intact without requiring a re-run of `fenris-monitor enable` after recovery.
|
||||
|
||||
---
|
||||
|
||||
## 10. Installation
|
||||
|
||||
**ADR:** [0004](../adr/0004-install-upgrade-removal-lifecycle.md). **Criteria:** IN-1–IN-10.
|
||||
|
||||
### 10.1 Install
|
||||
|
||||
`sudo make install`:
|
||||
|
||||
1. Builds a wheel from the checkout and installs it, with pinned dependencies (§10.5), into the dedicated Fenris-owned venv at `/opt/fenris`; a `/usr/local/bin/fenris` wrapper makes the unprivileged TUI/CLI a PATH command. The checkout is build-time input only — after install, nothing references it.
|
||||
2. Verifies `python3 ≥ 3.9` and `smartctl` presence, failing cleanly otherwise (never a runtime crash).
|
||||
3. Creates `/var/lib/fenris` with root-written group-read permissions and the `fenris` read group; the database file is created lazily by the first write (§3.1).
|
||||
4. Places units in `/etc/systemd/system`, helpers in `/usr/libexec/fenris`, polkit policy under `/usr/share/polkit-1/actions/` — recording **every** placed file in an explicit manifest consumed by upgrade and uninstall (§10.6).
|
||||
5. **Never enables or starts units.** A fresh install is fully dormant: units present but disabled, nothing running, no monitoring period. The only opt-in is the sanctioned toggle — `fenris monitor resume` or the first-run TUI prompt — enabling the timer and opening the first period in one step.
|
||||
6. Detects `./data/history.jsonl` beside the source (or accepts an explicit path), runs the idempotent single-transaction import (§3.5), and reports imported counts.
|
||||
|
||||
### 10.2 Upgrade
|
||||
|
||||
`sudo make upgrade` installs the new wheel into the same venv, syncs units and polkit against the manifest (`daemon-reload`; restart the timer only if unit contents changed **and** it is active — safe with `Persistent=no`), leaves timer state untouched, and **never kills an in-flight collection run**: a running oneshot finishes on its mapped interpreter; at worst one old-code run completes to the store. It then applies forward-only observation-store schema migrations (§3.6). `/var/lib/fenris` is never rebuilt.
|
||||
|
||||
### 10.3 Rollback
|
||||
|
||||
Best-effort by design: before migrations run, the installer snapshots `observations.db` to a one-generation `observations.db.bak`; rollback means reinstalling the previous version and restoring the backup. Automatic schema downgrade does not exist.
|
||||
|
||||
### 10.4 Uninstall and purge
|
||||
|
||||
- `make uninstall` first performs the sanctioned disable (`fenris-monitor disable --now`) so an open monitoring period closes `user_disabled` — removal is deliberate, and only the sanctioned path records intent — then stops and disables the units and removes the venv, helpers, units, polkit policy, and wrapper, **keeping** `/etc/fenris` and the observation store. Journal entries age out naturally.
|
||||
- `make purge` additionally removes configuration and store.
|
||||
- Reinstall after uninstall resumes from the preserved observation store; only purge erases history.
|
||||
|
||||
### 10.5 Dependencies
|
||||
|
||||
Exact pins in a committed lockfile; install and upgrade both install from it. Refreshing pins is an explicit developer step (`make update-deps`, committed), never a side effect of installing. The acquisition path adds no Python dependency and no OS package beyond smartmontools (§2.5).
|
||||
|
||||
### 10.6 Placement manifest
|
||||
|
||||
Installed artifacts sit only at their fixed locations — units in `/etc/systemd/system`, helpers in `/usr/libexec/fenris`, polkit policy under `/usr/share/polkit-1/actions/`, configuration at `/etc/fenris`, observation store under `/var/lib/fenris`, venv at `/opt/fenris`, wrapper at `/usr/local/bin/fenris` — and every placed file is recorded in the manifest (criterion IN-10).
|
||||
|
||||
---
|
||||
|
||||
## Appendix A: Traceability matrix
|
||||
|
||||
Built as assembly's first step (assembly decision, recommendation 5). Two-way: every ADR section maps to at least one criterion ID; every criterion cites its ADR or ticket.
|
||||
|
||||
### A.1 ADR section → criteria
|
||||
|
||||
| ADR section | Criteria |
|
||||
|---|---|
|
||||
| 0001 §1 Substrate | ST-1, ST-2 |
|
||||
| 0001 §2 Access | ST-1, CI-3 (/run) |
|
||||
| 0001 §3 Entities (incl. #12/#14 amendments) | ST-3, ST-5, PR-13, ID-2, CI-3 (no stored projections) |
|
||||
| 0001 §4 Day boundary | ST-4 |
|
||||
| 0001 §5 Retention | ST-5 |
|
||||
| 0001 §6 Migration | ST-6, ST-7, ST-8, ST-9, ST-10, ST-11 |
|
||||
| 0001 §7 Projection inputs | ST-3, PR-14, CI-3 (one-key config) |
|
||||
| 0001 §8 Versioning | ST-12, FL-5 |
|
||||
| 0001 §9 Collector health | LC-10 |
|
||||
| 0002 §1 One projection | PR-1 |
|
||||
| 0002 §2 Rate/formulas/regime/scenario | PR-2, PR-17 |
|
||||
| 0002 §3 Habit change | PR-3 |
|
||||
| 0002 §4 Hour classification | PR-4 |
|
||||
| 0002 §5 Denominator | PR-5 |
|
||||
| 0002 §6 Minimum evidence | PR-6 |
|
||||
| 0002 §7 Staleness | PR-7 |
|
||||
| 0002 §8 Confidence table (incl. #15 amendment) | PR-8, PR-15, CI-1, CI-4 |
|
||||
| 0002 §9 Segment breaks (incl. #15 amendment) | PR-9, PR-16 |
|
||||
| 0002 §10 Implied eligibility | PR-10 |
|
||||
| 0002 §11 Uncertainty | PR-11 |
|
||||
| 0002 §12 Language | CI-4 |
|
||||
| 0002 §13 Contract | PR-12 |
|
||||
| 0003 §1 Units | LC-1, CI-3 (/run) |
|
||||
| 0003 §2 Cadence | LC-2, LC-3 |
|
||||
| 0003 §3 Configuration | LC-4, CI-3 |
|
||||
| 0003 §4 Entry points | LC-5, IN-10 |
|
||||
| 0003 §5 Sanctioned toggle | LC-6, CI-2, CI-3 |
|
||||
| 0003 §6 Period rows | LC-7 |
|
||||
| 0003 §7 On-demand collection | LC-8, CI-2, CI-3 |
|
||||
| 0003 §8 TUI controls | TUI-1, TUI-2, TUI-4 |
|
||||
| 0003 §9 CLI compatibility | LC-9, CI-2 |
|
||||
| 0003 §10 Freshness constants | LC-10, CI-1, CI-2 |
|
||||
| 0004 §1 Delivery | IN-1 |
|
||||
| 0004 §2 Layout and manifest | IN-2, IN-10 |
|
||||
| 0004 §3 Privilege | IN-3, CI-3 |
|
||||
| 0004 §4 Dormant install | IN-3 |
|
||||
| 0004 §5 Legacy import | IN-4 |
|
||||
| 0004 §6 Upgrade | IN-5 |
|
||||
| 0004 §7 Rollback | IN-6 |
|
||||
| 0004 §8 Removal | IN-7 |
|
||||
| 0004 §9 Dependencies | IN-8, AC-5 |
|
||||
| 0004 §10 Scaffolding and floor | IN-9, TUI-3 |
|
||||
| 0005 §1 Malformed observations | FL-1, FL-2 |
|
||||
| 0005 §2 Missed observations | FL-3, CI-3 |
|
||||
| 0005 §3 Store faults | FL-4 |
|
||||
| 0005 §4 Newer schema | FL-5 |
|
||||
| 0005 §5 Repeated failures | FL-6 |
|
||||
| 0005 §6 Drive-reported anomalies | FL-7, CI-3 |
|
||||
| 0005 §7 Orphaned samples | FL-8 |
|
||||
| 0006 §1 Pin | AC-1 |
|
||||
| 0006 §2 Hard pin, no fallback | AC-3 |
|
||||
| 0006 §3 Normalization | AC-2 |
|
||||
| 0006 §4 Segment metadata sourcing | AC-4 |
|
||||
| 0006 §5 Prerequisites | AC-5 |
|
||||
|
||||
### A.2 Ticket decisions → criteria
|
||||
|
||||
| Ticket | Criteria |
|
||||
|---|---|
|
||||
| [Evaluate Python TUI frameworks](https://git.bongbetic.com/xavierk/Fenris/issues/6) | TUI-3 |
|
||||
| [Prototype the TUI information architecture](https://git.bongbetic.com/xavierk/Fenris/issues/3) | TUI-1, TUI-2, TUI-4 |
|
||||
| [Verify the controller identity that segments observation history](https://git.bongbetic.com/xavierk/Fenris/issues/11) | ID-1, ID-3 |
|
||||
| [Define endurance-baseline provenance and validation](https://git.bongbetic.com/xavierk/Fenris/issues/12) | PR-13, PR-14, ST-3 |
|
||||
| [Decide controller-segment metadata columns](https://git.bongbetic.com/xavierk/Fenris/issues/14) | ID-2 |
|
||||
| [Decide how degraded identity affects projection confidence](https://git.bongbetic.com/xavierk/Fenris/issues/15) | PR-15, PR-16, ID-4 |
|
||||
| [Define cross-cutting acceptance criteria](https://git.bongbetic.com/xavierk/Fenris/issues/13) | the register itself |
|
||||
| [Assemble the implementation-ready specification](https://git.bongbetic.com/xavierk/Fenris/issues/17) / [Write the Fenris redesign specification and close the map](https://git.bongbetic.com/xavierk/Fenris/issues/19) | this document and this appendix |
|
||||
|
||||
### A.3 Criterion → source
|
||||
|
||||
Every criterion carries its citation inline in the [register](acceptance-criteria.md): CI-1–CI-4 (ADR 0002 §§6–8, 0003 §§1–10, 0001 §§2–3/8, 0005 §§2/4–6); ST-1–ST-12 (ADR 0001, with ST-3 amended by tickets #12/#14); LC-1–LC-10 (ADR 0003); PR-1–PR-12 (ADR 0002), PR-13–PR-14 (ticket #12), PR-15–PR-16 (ticket #15, ADR 0002 §§8–9 as amended), PR-17 (ADR 0002 §2); ID-1–ID-4 (tickets #11/#14/#15, ADR 0001 §3 as amended); TUI-1–TUI-4 (tickets #3/#6, ADR 0003 §§8/10, ADR 0004 §10); FL-1–FL-8 (ADR 0005); IN-1–IN-10 (ADR 0004, with IN-10 also citing ADR 0003 §4); AC-1–AC-5 (ADR 0006).
|
||||
|
||||
### A.4 Assembly result
|
||||
|
||||
- **Every ADR 0001–0006 section maps to at least one criterion** — the A.1 table is complete; no orphan sections.
|
||||
- **Every criterion cites its ADR or ticket** — verified in the register; no orphan criteria.
|
||||
- **Decided-but-uncitered gaps found and filled inline in the register during assembly:** CI-3 bullet (no `/run` coordination surface; ADR 0003 §1, ADR 0001 §2), PR-17 (projection arithmetic; ADR 0002 §2), TUI-4 (normative Panes layout and bindings; ticket #3), IN-10 (fixed artifact placement; ADR 0004 §2, ADR 0003 §4).
|
||||
- **No genuinely undecided behavior remained** — no blocking ticket was raised.
|
||||
- **One reconciliation:** ADR 0004 §6's "`schema_version` table" wording resolves to ADR 0001 §8's `PRAGMA user_version` as the single version authority (§3.6); criterion ST-12 already fixed the mechanism.
|
||||
@@ -1,351 +0,0 @@
|
||||
# Fenris TUI polish and hourly history companion specification
|
||||
|
||||
**Status:** approval-ready draft for [Approve the Fenris TUI polish specification and handoff](https://git.bongbetic.com/xavierk/Fenris/issues/70). It becomes implementation-ready only when that ticket records human approval. This is a planning asset: no production code, release gate, package, installation change, or runtime diagnosis is made here.
|
||||
|
||||
**Canonical sources.** This companion integrates the closed decisions on [Fenris TUI polish and hourly history](https://git.bongbetic.com/xavierk/Fenris/issues/65): [Define trustworthy hourly history and first-data availability](https://git.bongbetic.com/xavierk/Fenris/issues/66), [Choose daily graph encoding and hourly drill-down](https://git.bongbetic.com/xavierk/Fenris/issues/68), [Reconcile unallocated usage with UTC projection evidence](https://git.bongbetic.com/xavierk/Fenris/issues/71), [Define monitoring signals and honest warm-up estimates](https://git.bongbetic.com/xavierk/Fenris/issues/67), and [Approve titlebox, health layout and colour presets](https://git.bongbetic.com/xavierk/Fenris/issues/69). Terminology follows [`CONTEXT.md`](../../CONTEXT.md), including *Usage interval*, *Unallocated usage*, *Local display day*, *Byte-allocation completeness*, and *Qualifying day*.
|
||||
|
||||
**Relationship to existing specs.** [`fenris-redesign.md`](fenris-redesign.md) stays frozen. This companion amends it without editing it. [`dashboard-clarity.md`](dashboard-clarity.md) remains binding except where this document explicitly supersedes its header/credit placement. In the tracker, [Implement dashboard clarity and release notes](https://git.bongbetic.com/xavierk/Fenris/issues/61) is currently closed; this companion is a new follow-on handoff, not a rewrite or reopening of that shipped umbrella.
|
||||
|
||||
**Binding language.** *Must*, *exactly*, and *never* are normative.
|
||||
|
||||
## 1. Supersession and preserved requirements
|
||||
|
||||
This companion supersedes only these dashboard-clarity identity placements:
|
||||
|
||||
- the old header `Fenris — NVMe endurance monitor`;
|
||||
- the old dimmed `by Bongbetic` credit in the service strip.
|
||||
|
||||
The new identity contract is the top titlebox in §2. The titlebox is the sole maker-credit surface.
|
||||
|
||||
Everything else from [Chart Fenris dashboard clarity](https://git.bongbetic.com/xavierk/Fenris/issues/55) remains intact unless a more specific clause below amends its placement: the continuity wording, paused block and Deliberate-disable semantics, `q QUIT TUI` rail, footer action ownership, polkit-accurate auth banner, TUI/CLI wording parity where binding, changelog-driven release notes, and the quit-versus-pause distinction. Polkit wording remains polkit-accurate; no new user-facing TUI/status string says "sudo".
|
||||
|
||||
## 2. Product identity, health layout, and settings
|
||||
|
||||
### 2.1 Titlebox
|
||||
|
||||
The TUI's top titlebox must read exactly:
|
||||
|
||||
```text
|
||||
🐺 Fenris by Bongbetic
|
||||
```
|
||||
|
||||
If the wolf glyph is unsupported or would disturb titlebox width, the fallback is exactly:
|
||||
|
||||
```text
|
||||
Fenris by Bongbetic
|
||||
```
|
||||
|
||||
Do not render tofu, replacement boxes, or an unstable emoji-width layout. The lifespan headline remains a drive/projection data surface, not the application title. `fenris status` does not render this titlebox.
|
||||
|
||||
### 2.2 Maker credit
|
||||
|
||||
Remove the duplicate `by Bongbetic` service-strip credit in the polished layout. The titlebox is the only maker-credit surface. This is the explicit amendment to the prior dashboard-clarity header/credit placement; it does not weaken the preserved continuity, pause, quit, auth, parity, or release-notes requirements.
|
||||
|
||||
### 2.3 Drive health and settings grouping
|
||||
|
||||
Move vendor wear under **Drive health**, alongside temperature, spare, media errors, unsafe shutdowns, power-on hours, cycles, capacity, and written-total context. Vendor wear remains context, never a second projection.
|
||||
|
||||
**Settings** remains read-only and limited to the device selector, endurance baseline/provenance, retention facts, and TUI display preferences. No custom colour editor is added.
|
||||
|
||||
## 3. Monitoring status, collection activity, and warm-up messaging
|
||||
|
||||
### 3.1 Status lattice
|
||||
|
||||
Every status line carries a glyph and text label. Colour is never the sole carrier. Text never blinks.
|
||||
|
||||
| State | Colour | Glyph | Motion | Exact status / explanation contract |
|
||||
|---|---|---|---|---|
|
||||
| Monitoring | green | `●` | blink 750 ms on / 750 ms off, status dot only | `● Monitoring` |
|
||||
| Collecting | green | `◐` | steady | `◐ Collecting` — run in flight, bounded by the 90 s collection timeout |
|
||||
| Paused | amber | `‖` | steady | `‖ Paused` — `monitoring paused — paused time excluded from your usage habit` |
|
||||
| Waiting | amber | `○` | steady | `○ Waiting` — `last sample X ago`, `awaiting first sample`, or `awaiting another sample` |
|
||||
| Interrupted | red | `⊘` | steady | `⊘ Interrupted` — `collection stopped outside Fenris — monitoring period still open` |
|
||||
| Error | red | `✖` | steady | `✖ Error` — `last run failed (exit N)` plus `last good sample X ago` when data is still fresh, or `observation store unreadable — see journal` |
|
||||
| Stale | red | `◌` | steady | `◌ Stale` — `last sample X days ago` when the timer is active and no failure is recorded |
|
||||
| Unknown | grey | `?` | steady | `? Unknown` — `service state unavailable` |
|
||||
|
||||
Reduced motion, either from Textual's reduced-motion signal or the user's preference, renders Monitoring as a steady `● Monitoring`. The CLI form is always steady.
|
||||
|
||||
### 3.2 Precedence
|
||||
|
||||
Base-state precedence is:
|
||||
|
||||
```text
|
||||
Error > Interrupted > Paused > Stale > Waiting > Monitoring > Unknown
|
||||
```
|
||||
|
||||
Additional rules:
|
||||
|
||||
- Paused outranks Stale. A drive paused for 49 days is amber Paused with `last sample 49 days ago` as a fact line, not a stale alarm.
|
||||
- Error outranks Interrupted; the external stop rides in the explanation line when both facts exist.
|
||||
- Unknown is used only when the service query fails and no store-derived fact, such as an open monitoring period, freshness, or deliberate-pause row, places the state higher.
|
||||
- Collecting is an overlay, not a base rung. While the oneshot is in flight it overrides every base state except a store fault. While overlaying Paused, Interrupted, or retry-after-failure, the explanation line names the underlying state: for example, `run in flight — paused` or `run in flight — retry`. It reverts within the 90 s collection timeout to whatever state the outcome earns.
|
||||
|
||||
### 3.3 Separate facts
|
||||
|
||||
Do not fold these facts into the status word:
|
||||
|
||||
- `Last sample: X ago` freshness;
|
||||
- last outcome: ok, failed with exit code, or none;
|
||||
- `Start at boot: on/off` boot enablement;
|
||||
- collection activity, which is represented by the Collecting overlay.
|
||||
|
||||
Blink means exactly "monitoring enabled and data fresh". It never means that collection just succeeded; collection activity is explicit Collecting.
|
||||
|
||||
### 3.4 Timing and refresh
|
||||
|
||||
A lightweight 5 s `systemctl show` poll of the timer and service units drives the status strip. The full store refresh remains at the 5-minute cadence. Status transitions are re-evaluated every poll tick from cached newest-sample timestamp plus current clock, so freshness aging boundaries cross within roughly 5 s without a store read. A new sample requires the normal store refresh.
|
||||
|
||||
The existing constants remain unchanged: fresh is newest sample within 2 × cadence + `AccuracySec` + 60 s, stale is at 48 h, collection timeout is 90 s, and default cadence is 5 min.
|
||||
|
||||
### 3.5 Canonical transitions
|
||||
|
||||
The implementation must make these states directly observable:
|
||||
|
||||
- failed run, data 4 min old: `✖ Error` — `last run failed (exit 3) · last good sample 4 min ago`;
|
||||
- next successful run: `● Monitoring` after the status poll/refresh clears the failure;
|
||||
- failed run with retry in flight: `◐ Collecting` — `run in flight — retry`, then Error or Monitoring;
|
||||
- suspend for 3 h, wake, timer fires: `○ Waiting` → `◐ Collecting` → `● Monitoring`;
|
||||
- external stop with fresh data: `⊘ Interrupted` — `collection stopped outside Fenris`;
|
||||
- external stop 3 days later: still `⊘ Interrupted`, with `last sample 3 days ago` visible;
|
||||
- fresh install with zero samples: `○ Waiting` — `awaiting first sample`;
|
||||
- one sample but no compatible pair: `○ Waiting` — `awaiting another sample`;
|
||||
- unreadable observation store: `✖ Error` — `observation store unreadable — see journal`, with store-dependent views suppressed.
|
||||
|
||||
### 3.6 Warm-up and withheld estimates
|
||||
|
||||
Projection warm-up uses the UTC accounting rules in §5. The progress block during warm-up is:
|
||||
|
||||
```text
|
||||
Building evidence — N of 14 days observed · Q qualifying
|
||||
First lifespan estimate after 12 qualifying days
|
||||
```
|
||||
|
||||
`N` counts distinct represented UTC dates in the current controller segment, capped at 14 for the display. `Q` counts dates whose represented monitored span has at least 50% coverage. The gate for the first lifespan estimate is 14 represented UTC dates with at least 12 qualifying; Supported confidence still separately requires at least 14 qualifying dates plus every other confidence prerequisite.
|
||||
|
||||
Do not promise a countdown by hours. Days are the grain, and a provisional day can still fail qualification. Graph-data availability is independent: the graph can render from first usable evidence while the lifespan estimate remains withheld.
|
||||
|
||||
Withheld-estimate reason lines are explicit, never blank and never zero-filled:
|
||||
|
||||
- `No endurance baseline — set a rated TBW to see an estimate`;
|
||||
- `Drive does not report write counters`;
|
||||
- the warm-up progress block above;
|
||||
- stale evidence: show the estimate frozen at the latest published usage-evidence endpoint with `estimate not updating — last sample X ago`;
|
||||
- paused days are excluded from the day count, and the Paused status carries that fact.
|
||||
|
||||
Once warm-up clears, the existing lifespan line and Limited/Supported label from the frozen redesign specification render unchanged. This companion adds the progress block, reason lines, status precedence, and frozen-note behavior; it does not invent a new steady-state projection format.
|
||||
|
||||
### 3.7 CLI parity
|
||||
|
||||
`fenris status` adopts the same status vocabulary, precedence, glyphs, and reason lines, rendered statically. It shows Collecting only if a run is in flight at query time. Display preferences and TUI themes never affect CLI facts.
|
||||
|
||||
## 4. History evidence, local browsing, and publication
|
||||
|
||||
### 4.1 Ownership and publication
|
||||
|
||||
The collector owns sample → usage interval → hour observation/day aggregate derivation. It must publish a consistent validated result before reporting collection success. Readers must not see a new sample advertised as fully derived while dependent history is missing.
|
||||
|
||||
The TUI is read-only. Its next successful refresh sees whatever the collector has durably published, regardless of whether the TUI was running during collection. There is no hourly batch wait and no projection-confidence gate on usage history. Failure preserves prior valid history and remains explicit.
|
||||
|
||||
### 4.2 First visible data
|
||||
|
||||
One successful sample establishes counter, health, and freshness evidence, but not a usage delta; display `Awaiting another sample`. The first usable sample pair may display measured partial-hour usage labelled `so far` when attribution supports it. A usable pair has valid ordered timestamps and supported, nonnegative monotonic counters in the same controller segment; monitored totals additionally require an interval fully inside one monitoring period.
|
||||
|
||||
A cross-hour pair is a real usage interval total, not two invented hour values. Measured zero is visible `0 B`. Missing, unsupported, invalid, or absent evidence is never converted to zero.
|
||||
|
||||
### 4.3 Local display days
|
||||
|
||||
Storage timestamps and projection evidence days remain UTC. History browsing uses the user's current system timezone, visibly labelled. If the system timezone changes, the same retained evidence regroups into the new local display days. Use real calendar boundaries: DST days may be 23 or 25 hours, repeated local hours have distinct offsets, and fractional UTC offsets must work without synthetic splitting.
|
||||
|
||||
### 4.4 Retention and repair
|
||||
|
||||
Retain timestamped usage intervals indefinitely alongside hour observations and day aggregates. Full raw samples retain the 14-day policy, with a boundary-anchor exception: do not prune a raw sample that is still needed to durably derive an unfinished interval/hour/day representation.
|
||||
|
||||
Repair derives only what surviving raw evidence supports, transactionally and idempotently. It preserves original evidence and valid historical summaries. Incomplete reconstruction must not overwrite valid older history. Unsupported historical precision stays unavailable; it is not repaired by interpolation or waiting. A failed repair remains visible and retryable.
|
||||
|
||||
### 4.5 Attribution and gaps
|
||||
|
||||
Preserve measured usage interval totals. Never divide them proportionally across hours or calendar days, never assign them to an endpoint as if timing were observed, and never double-count interval totals and summaries derived from the same evidence.
|
||||
|
||||
An interval entirely inside a local display day and one monitoring period may contribute its total to that local-day total even if individual hour shares are unknown. Otherwise the total is shown separately as unallocated usage. An incomplete allocated subtotal must not be presented as a complete total.
|
||||
|
||||
Missing samples reduce usage-habit coverage where classification is unknown, but they do not erase a compatible measured gap total. Byte-allocation completeness and usage-habit classification coverage are distinct. Do not infer zero writes from missing samples.
|
||||
|
||||
Partial summaries represent elapsed time only. Future time is neither zero nor unknown. Deliberately disabled time is excluded, not an hour state. Intervals crossing a deliberate pause cannot distinguish monitored from paused writes: preserve the original evidence, exclude ambiguous bytes from monitored totals, and explain why. External service stops remain unexplained in-period gaps, not Deliberate disables.
|
||||
|
||||
### 4.6 Controller segments in browsing
|
||||
|
||||
The default history view is the current controller segment. Older segments remain browsable with explicit reset/replacement boundaries. Never form a delta across a controller-segment boundary or silently combine different drives.
|
||||
|
||||
## 5. UTC projection accounting and confidence amendments
|
||||
|
||||
This section amends the projection contract without changing lifespan mathematics, numeric thresholds, or confidence categories.
|
||||
|
||||
### 5.1 Measured totals and requested spans
|
||||
|
||||
For any requested span, count a compatible interval's total exactly once when its complete span is inside the requested span, inside one monitoring period, and has eligible controller provenance. It may supply a complete window total even when individual UTC-hour/day shares are unknown.
|
||||
|
||||
A positive interval crossing a requested boundary cannot supply that window's unknown share. Preserve its measured total separately as unallocated usage. Do not split proportionally, assign to the ending day, silently omit possible bytes, or label an incomplete subtotal as complete.
|
||||
|
||||
Withhold a rate whenever its monitored numerator cannot be established: unresolved boundary shares, missing initial/resume/reset counter support, and legacy eligibility ambiguity are not zero. Do not shorten the requested window or remove unknown monitored seconds merely to obtain a number. A compatible monotonic interval with zero counter delta proves zero writes throughout its represented span, including a requested subspan; that is direct counter evidence, not interpolation.
|
||||
|
||||
### 5.2 Projection endpoints and denominator
|
||||
|
||||
Let `T` be the latest published usage-evidence endpoint. The 7/28/90-day scenario windows end at `T` and start exactly 7/28/90 × 86,400 seconds earlier. Do not round starts to UTC midnight, and do not dilute rates as the TUI read clock advances without new published usage evidence. Evidence age remains a separate fact.
|
||||
|
||||
The default sustained regime is eligible observation history capped at 90 days, ending at `T`. A detected habit-change regime starts at the first divergence day's UTC midnight. Any unresolved share at those boundaries invokes the unavailable-rate rule; there is no silent fallback to a more convenient start.
|
||||
|
||||
For the chosen span, the rate denominator is wall-clock seconds inside monitoring periods, including powered-off and unknown time, excluding deliberate-disable time. Numerator and denominator describe the same requested span. Pause-crossing intervals cannot establish which writes were monitored and therefore cannot supply affected monitored totals. No bridge crosses controller-segment boundaries.
|
||||
|
||||
### 5.3 UTC evidence dates, warm-up, and confidence
|
||||
|
||||
Projection evidence remains UTC; local graph regrouping never changes projection eligibility. Coverage uses known-classified seconds divided by represented elapsed monitored seconds, excluding deliberately disabled and future time.
|
||||
|
||||
A qualifying day is a distinct UTC date whose represented monitored span has coverage at least 50%. A current partial date counts provisionally and can lose qualification as unknown time accumulates. Count a date once, not once per hour, interval, or controller fragment. Entirely paused dates have no represented monitored span and do not count.
|
||||
|
||||
Warm-up clears when the current controller segment has at least 14 distinct represented UTC dates, at least 12 of which qualify. Supported confidence still requires at least 14 qualifying dates plus every other existing condition. Thus 12 qualifying plus 2 poor dates clears warm-up but remains Limited. Same-day segment breaks use only the current segment's eligible portion for its new-segment warm-up; a shared UTC date does not import old-segment qualification.
|
||||
|
||||
Habit-change comparisons require completed, consecutive UTC days with evaluable daily write totals. Do not compress missing calendar dates into adjacent-row windows or treat missing/disabled time as zero. Unknown daily totals cannot establish habit change and cannot pass the burst/concentration guard.
|
||||
|
||||
An affected scenario rate is withheld while other independently computable rates remain visible. If the headline regime lacks a complete monitored numerator, no lifespan number renders; show the specific evidence-unavailable reason. If a complete positive headline rate exists while daily shares remain unknown, the lifespan may render as Limited, but Supported is blocked wherever a required confidence check cannot be established.
|
||||
|
||||
### 5.4 Segment and legacy evidence
|
||||
|
||||
Never compute a delta across a reset or replacement boundary, even within one UTC hour/day. Same-identity reset preserves prior eligible habit evidence, subject to the existing current-segment re-warm gate before a lifespan number can render. Re-warming does not reconstruct missing counter support.
|
||||
|
||||
A controller-identity change, including to or from a blank key, quarantines previous identity from every projection horizon. Returning to a previously seen key does not undo the intervening quarantine. History browsing may still show older segments explicitly.
|
||||
|
||||
Trusted legacy day-only summaries remain usable at their actual represented precision: only in a window fully containing their represented span and only when monitoring-period/controller eligibility is known. They cannot supply subday detail, local-midnight splits, or partial rolling-window totals. Mixed legacy summaries that cannot separate eligible from ineligible writes remain browsable but cannot supply affected projection totals.
|
||||
|
||||
### 5.5 Projection acceptance examples
|
||||
|
||||
Future implementation must make these cases observable without using fabricated history:
|
||||
|
||||
1. 100 MB from 23:55 to 00:05 UTC in one period/segment: retain 100 MB once. Neither UTC-day share is known. A containing window can consume 100 MB; a window starting at midnight cannot claim an exact share or rate.
|
||||
2. Sparse three-day recovery interval with compatible counters: a measured 900 MB total remains real and can supply a containing window. There is no 300 MB/day allocation.
|
||||
3. A 7-day boundary cuts a positive interval and the 28-day boundary does not: withhold the 7-day rate, preserve the independently computable 28-day rate, and do not omit the unresolved old-enough horizon to manufacture Supported confidence.
|
||||
4. The same boundary with zero measured delta: exact zero contribution is permitted for the represented subspan; missing time outside it remains unknown.
|
||||
5. `T` at UTC noon: a 7-day window spans exactly 604,800 wall-clock seconds before subtracting deliberate-disable time. Refreshing the TUI without new evidence does not change the endpoint.
|
||||
6. First sample arrives after monitoring starts: preceding monitored time lacks a write total. Keep its time, do not invent zero bytes, and withhold affected rates.
|
||||
7. Pause/resume crossed by one counter interval: preserve total as original evidence, but do not count ambiguous paused writes as monitored.
|
||||
8. 12 qualifying dates plus 2 poor dates: warm-up clears, Supported still fails the 14-qualifying-date requirement.
|
||||
9. Good coverage with unknown daily shares: day progress can qualify; burst and habit checks remain unknown, not passed or zero-filled.
|
||||
10. Only 14 days of eligible history: do not require 28/90-day horizons yet, but evaluate the burst guard over observed eligible history and keep unknown daily totals blocking.
|
||||
11. Reset/replacement at 10:30 UTC: no cross-break delta or whole-date shortcut. Same-key reset preserves eligible prior habit evidence but requires new-segment re-warm; replacement excludes prior identity from scenarios too.
|
||||
12. Trusted old UTC-day summary with raw samples gone: usable only in a window fully containing its represented span and known eligibility, once only; no partial local-day/hour split.
|
||||
|
||||
## 6. Usage history graph and interaction
|
||||
|
||||
### 6.1 Encoding
|
||||
|
||||
Adopt a writes-only daily bar graph with hourly drill-down. Reads are out of scope. Candles are rejected: they bury the daily total, import price-chart semantics that usage data does not have, and distort gap/partial evidence. A rolling hourly strip is rejected as the default because it lacks day totals without reintroducing the day level.
|
||||
|
||||
The graph's value is bytes written. The daily range view shows one bar per local display day. Activating a day drills into hourly bars for that selected local display day. Back returns to the range view.
|
||||
|
||||
### 6.2 Ranges, labels, and readout
|
||||
|
||||
Default range: 14 days. Selectable ranges: 7, 14, 28, and 90 days. Ranges limit the viewport only; they never delete retained history or hide older controller segments from browsing.
|
||||
|
||||
The header line is:
|
||||
|
||||
```text
|
||||
usage history · Local · UTC±HH:MM · <tz name>
|
||||
```
|
||||
|
||||
The day row uses day-of-month labels. The selected-day readout uses `Wed 09 Sep` style. Hourly detail labels every third hour `00…21` plus `midnight → 23:00 local`. DST and repeated local hours follow the Local display day contract.
|
||||
|
||||
The selected-item readout states totals, evidenced hours, unallocated usage separately, coverage, and partial-state text such as `partial · N h elapsed · so far`.
|
||||
|
||||
### 6.3 Controls
|
||||
|
||||
The graph pane is focusable. Keyboard controls:
|
||||
|
||||
- `←` / `→` select day or hour;
|
||||
- `Enter` drills into the selected day;
|
||||
- `Esc` / `Backspace` returns from hourly detail;
|
||||
- `1` / `2` / `3` / `4` switch 7 / 14 / 28 / 90 days.
|
||||
|
||||
Mouse controls use widget-local coordinates to select bars. Where Textual mouse support is available, clickable equivalents must cover select, drill, and back. Global `p` / `r` / `c` / `d` / `q` remain unchanged, and the footer shows graph keys while the graph is focused.
|
||||
|
||||
### 6.4 State rendering
|
||||
|
||||
The legend is always visible when the graph is visible. Distinct glyphs:
|
||||
|
||||
| Glyph | Meaning |
|
||||
|---|---|
|
||||
| `█` | allocated measured writes |
|
||||
| `▒` | unallocated measured writes, stacked separately |
|
||||
| `░` | gap / no evidence, never zero |
|
||||
| `·` | measured zero bytes |
|
||||
| `┄` | partial-day cap |
|
||||
| `▼` | selection marker |
|
||||
|
||||
Unallocated usage is measured, not fabricated; it is never spread into hours to make the graph look complete. Gaps remain distinct from measured zero. Deliberate-disable annotations remain visible.
|
||||
|
||||
### 6.5 Minimum terminal
|
||||
|
||||
At 80×24, the range view must fit the default 14-day graph using 4 columns per day, and hourly drill-down must fit 24 single-column hourly bars. Below 80×24, hide the graph region and show a one-line textual history summary plus exactly:
|
||||
|
||||
```text
|
||||
graph needs ≥80×24
|
||||
```
|
||||
|
||||
This is not an error. The titlebox, status label/reason, drive health, service facts, quit rail/action affordances, and selected-day readout or equivalent textual context must survive. Resizing must not leave stale graph state visible.
|
||||
|
||||
### 6.6 Framework obligation
|
||||
|
||||
Current Textual facts verified during charting: Textual 8.2.8 has no BarChart widget, Sparkline is non-interactive, and `textual-plotext` is not an installed dependency. Implement the graph as a custom block-glyph renderable in the usage-history pane, with click mapping from widget-local coordinates. Do not add a plotting dependency for the adopted graph.
|
||||
|
||||
## 7. Colour presets, persistence, and accessibility
|
||||
|
||||
### 7.1 Presets
|
||||
|
||||
Approved presets: Amber, Nord, and High Contrast. Amber is the default and keeps the graph amber by default. Other presets may use theme-appropriate graph/accent colours.
|
||||
|
||||
Themes style chrome, graph, borders, accents, and muted text. They do not override status semantics. The status lattice keeps semantic colours: green Monitoring/Collecting, amber Paused/Waiting, red Interrupted/Error/Stale, grey Unknown, always with glyph and text.
|
||||
|
||||
Preset palettes must maintain readable contrast in normal and focused states. High Contrast is a first-class preset, not merely a lightened Amber.
|
||||
|
||||
### 7.2 Persistence scope
|
||||
|
||||
Preset and reduced-motion choices are user-scoped TUI display preferences. Persist them at:
|
||||
|
||||
```text
|
||||
${XDG_CONFIG_HOME:-~/.config}/fenris/tui.json
|
||||
```
|
||||
|
||||
Do not store these preferences in `/etc/fenris/fenris.conf`, the observation store, helper state, package config, or collector/device configuration. Missing preferences default to Amber and normal motion. This preference file never affects collection, projection, history evidence, or CLI `status` facts.
|
||||
|
||||
### 7.3 Controls and reduced motion
|
||||
|
||||
Add accessible TUI controls for `t preset` and `m motion`, with clickable equivalents where Textual mouse support is available. Focusable graph/settings panes are acceptable as long as status text and global actions remain reachable.
|
||||
|
||||
Normal motion: only the Monitoring status dot blinks at the approved 750 ms on / 750 ms off cadence. Text never blinks, and no other state animates. Reduced motion: Monitoring renders steady as `● Monitoring`. No animation may imply successful collection.
|
||||
|
||||
Long reasons, paused states, degraded/error states, missing data, and warm-up/withheld-estimate lines must remain text-visible. Do not collapse them into colour, blank space, or a misleading zero.
|
||||
|
||||
### 7.4 Framework facts
|
||||
|
||||
The charting prototype verified current Textual documentation: `App.register_theme(theme)` and `App.theme` support theme registration/activation; Textual theme variables plus `$text` / `color: auto` support legibility; mouse events provide screen/widget-relative coordinates and focusable widgets can be resolved/clicked for keyboard+mouse proof paths. The existing lockfile still pins Textual 8.2.8; prototype branches remain throwaway assets and add no runtime dependency.
|
||||
|
||||
## 8. Future implementation proof paths
|
||||
|
||||
These are direct observable probes the execution effort can derive tests from; they are not a build-only checklist.
|
||||
|
||||
- Synthetic first-run store with zero samples: TUI and `fenris status` show Waiting/awaiting-first-sample; graph has no zero-filled bars; estimate is withheld with the correct reason.
|
||||
- One sample followed by a compatible pair inside one hour: first state says `Awaiting another sample`; after the pair, graph shows a partial `so far` value. If the counter is unchanged, the evidenced interval is visible `0 B`.
|
||||
- Cross-hour and cross-local-midnight intervals: totals are retained once; hour/day shares remain unallocated unless evidence supports them; gaps and zeros use distinct glyphs.
|
||||
- Pause/resume-crossing interval: paused time is excluded, ambiguous bytes do not enter monitored totals, Paused status explains the consequence, and quitting the TUI does not pause monitoring.
|
||||
- Failed collection with fresh data, retry in flight, external stop, and store fault: status precedence matches §3 and store fault suppresses store-dependent views.
|
||||
- UTC projection horizon cut by a positive interval: affected rate is unavailable with a specific reason while independent horizons remain; refreshing without new evidence does not shift `T`.
|
||||
- Warm-up fixtures for 12 qualifying + 2 poor dates and 14 qualifying dates: the first clears the progress gate but remains Limited; the second can be Supported only if all other prerequisites pass.
|
||||
- Graph keyboard and mouse path: range switch → selected day → hourly detail → back, with focus/footer behavior and no conflict with global actions.
|
||||
- 80×24 and narrower terminal captures: 80×24 keeps the graph; below 80×24 shows the exact fallback and preserves status/health/action context.
|
||||
- Theme fixture: Amber default, switch to Nord/High Contrast, restart TUI under the same unprivileged user and see preference persist; `fenris status` and collection behavior are unchanged.
|
||||
|
||||
## 9. Out of scope
|
||||
|
||||
- Production implementation, release gates, package publishing, installation changes, and closing or reopening prior implementation umbrellas.
|
||||
- Changing lifespan mathematics, evidence/confidence thresholds, collector cadence, or controller-segment semantics except for the explicit accounting/evaluability amendments in §5.
|
||||
- Fabricating, interpolating, proportionally splitting, or endpoint-assigning missing history.
|
||||
- Read-throughput graph selector, custom colour editor, web GUI, notifications, alerting, unrelated release-workflow changes, or a general application redesign beyond the named TUI requirements.
|
||||
@@ -1,106 +0,0 @@
|
||||
# Fenris release and packaging specification
|
||||
|
||||
**Status: decision-complete.** Assembled by [Task: Compose release spec + ADR amending 0004](https://git.bongbetic.com/xavierk/Fenris/issues/42) from the closed tickets of the Wayfinder map [Fenris deb + rpm release plan](https://git.bongbetic.com/xavierk/Fenris/issues/33). This document is normative for the follow-up **execution effort** that builds and publishes packages; no packages are built here.
|
||||
|
||||
**Canonical roles.** [ADR 0007](../adr/0007-package-delivery-amends-0004.md) records the lifecycle rationale (amending [ADR 0004](../adr/0004-install-upgrade-removal-lifecycle.md)); this document restates the **operative contracts** — compat matrix, channel, toolchain, signing, release mechanics, package layout, maintainer-script behavior, migration — so the executing effort never needs Wayfinder-ticket access. Runtime semantics come from [ADRs 0001–0006](../adr/) and the [redesign specification](fenris-redesign.md) verbatim; nothing here overrides them. Terminology follows the glossary in [`CONTEXT.md`](../../CONTEXT.md), including *Release* and *Rollback*.
|
||||
|
||||
**Binding language.** *Must*, *exactly*, and *never* are normative.
|
||||
|
||||
## 1. Compatibility matrix
|
||||
|
||||
| Target | Version | Format | Registry placement |
|
||||
|---|---|---|---|
|
||||
| Debian 12 (bookworm) | — | deb | `debian/pool/bookworm/main` |
|
||||
| Ubuntu 22.04 (jammy) | — | deb | `debian/pool/jammy/main` |
|
||||
| Ubuntu 24.04 (noble) | — | deb | `debian/pool/noble/main` |
|
||||
| Fedora 40+ | every release | rpm | `rpm/fenris` group |
|
||||
| openSUSE Tumbleweed | rolling | rpm | `rpm/fenris` group |
|
||||
|
||||
- Architecture: **x86_64 only** (arm64 only if real ARM hardware appears — map fog).
|
||||
- Dependencies are vendored as locked, pure-Python runtime packages for every target: Debian 12 and Ubuntu 22.04/24.04 ship `python3-textual` 0.1.13, far below the floor; Fedora 40+ ships ≥ 0.48 but below the pin ([toolchain research](../research/deb-rpm-toolchain.md)). No distro `python3-textual` dependency ever enters package metadata.
|
||||
- Package metadata `depends:`/`Requires:` are exactly `python3 (>= 3.10)`, `smartmontools`, `systemd` — the Python floor is 3.10 (oldest supported distro interpreter, Ubuntu 22.04), bumping ADR 0004 §10's 3.9 gate for packages; `make install` keeps the checkout's floor.
|
||||
- Runtime packages are staged with `python3 -m pip --target /opt/fenris/vendor`; entry points run the target system's `python3` with that directory on the import path. No package ships a copied Python interpreter, avoiding build-host ABI paths and rolling-distribution minor-version breakage.
|
||||
|
||||
## 2. Distribution channel
|
||||
|
||||
- **Channel:** the self-hosted Gitea 1.27.1 package registry at `git.bongbetic.com`, owner public for anonymous consumers ([registry research](../research/gitea-package-registry.md); [OBS rejected](../research/obs-route.md)).
|
||||
- **Single channel.** No stable/testing split — deferred until external users ask to track pre-release builds (map fog). Every published version is retained indefinitely (registry has no REST cleanup; republishing a filename is a 409).
|
||||
- **deb publication:** one deb artifact PUT to each codename pool — `PUT /api/packages/{owner}/debian/pool/{bookworm|jammy|noble}/main/upload`.
|
||||
- **rpm publication:** one rpm artifact PUT to the single `fenris` group — `PUT /api/packages/{owner}/rpm/fenris/upload` — serving Fedora 40+ collectively.
|
||||
- **Consumer setup (install docs, normative):**
|
||||
- apt: keyring file from `…/debian/repository.key` via `signed-by`, one sources line per distribution, instance Debian Registry Key **fingerprint printed beside the curl one-liner** (TOFU hardening).
|
||||
- dnf: `dnf config-manager --add-repo <raw-url of packaging/fenris.repo>` — the in-repo, Fenris-owned `.repo` with `gpgkey` pointing at the published packaging key and `repo_gpgcheck=0`. **Gitea's auto-generated `.repo` is never mentioned in docs**: it sets `gpgcheck=1` against the instance auto-key, which never signed our rpm payload — a trap that breaks installs.
|
||||
|
||||
## 3. Build toolchain
|
||||
|
||||
- **nfpm** for both formats from a single `packaging/nfpm.yaml` — one config, `overrides:` for per-format deltas, two invocations (`nfpm pkg -p deb`, `nfpm pkg -p rpm`). fpm is dropped entirely (CLI-flag config drifts); no hand rpm spec; dh-virtualenv is deb-only and dormant since 2020.
|
||||
- **Single source of truth:** version injected from `pyproject.toml`; file lists generated by a staging script (locked runtime packages → `/opt/fenris/vendor`, plus wrapper, helpers, units, polkit policy, sysusers/tmpfiles fragments) referenced by `nfpm.yaml` as a `type: tree` content entry — no hand-maintained file lists.
|
||||
- **Entry point:** `make package` → `dist/fenris_<v>_amd64.deb` + `dist/fenris-<v>-1.x86_64.rpm`.
|
||||
- **Version scheme:** `<pyproject-version>-1` in both formats; a rebuild of the same upstream version bumps the revision (`-2`, `-3`, …) — the same filename is never re-PUT (registry 409s duplicates).
|
||||
- **Authoring `nfpm.yaml`, the staging script, and the workflow file is execution** — deliberately not part of the decision map. The [toolchain research doc](../research/deb-rpm-toolchain.md) sketches the pipeline.
|
||||
|
||||
## 4. Signing and key policy
|
||||
|
||||
- **RPM payload: signed.** rpmsign with the dedicated packaging key, invoked by `make sign-rpm` after the package is built. This is required, not optional: it is the only working dnf-native verification path.
|
||||
- **deb: unsigned.** apt never verifies payload signatures; trust = instance-signed `InRelease` (signed-by keyring) + TLS + Acquire-By-Hash. Manual-download integrity is covered by SHA256SUMS.
|
||||
- **SHA256SUMS: clearsigned** with the packaging key — the trust anchor for manually downloaded release assets, independent of TLS.
|
||||
- **Packaging key:** single dedicated key, RSA 3072, UID `Fenris Packaging <packaging@bongbetic.com>`, 2-year expiry, no master/subkey hierarchy (single maintainer, manual builds). Private key lives in the password manager only; each release does import → sign → delete — nothing permanent on any build host. The full ceremony is documented in `docs/install/signing-key-ceremony.md`.
|
||||
- **Public key publication:** in-repo `packaging/keys/fenris-packaging.asc` (raw URL doubles as the `.repo` gpgkey target), release notes, docs page. No keyservers — TOFU-over-TLS.
|
||||
- **Rotation (outline):** new key published alongside old; rpm signed with the new key; `fenris.repo` gpgkey lists both URLs (dnf accepts multiple); old key dropped after one release cycle. Procedure details stay in map fog.
|
||||
|
||||
## 5. Release mechanics
|
||||
|
||||
- **A Release is:** a version tag, its packages in the channel, a Gitea release entry with notes, and a clearsigned SHA256SUMS — all together. **Bare tags are forbidden** (tag without packages + release entry is not a Release).
|
||||
- **Cadence: on-demand.** Tag when user-visible changes or fixes accumulate; no calendar, no empty releases, no frequency SLA, no RC ceremony — fixes ship as a revision bump of the current version.
|
||||
- **Versioning: plain semver.** Major = breaking CLI/config/unit change; store schema changes ride the natural bump (the forward-only refusal handles old-reader/new-store).
|
||||
- **Promotion flow:** tag → `make release` (automated: `make package` → RPM signing via nfpm → SHA256SUMS generation → clearsign → prints registry PUTs + Gitea release steps). The ceremony is documented in `docs/install/signing-key-ceremony.md`.
|
||||
- **Rollback:** installing an older package over a newer store is **unsupported** — the store's forward-only version refusal fails it by design. Documented rollback = restore the observation-store snapshot, then install the old Release. No automatic downgrade machinery exists or will be built.
|
||||
- **CI:** no runners are registered on the instance today ([Actions runner research](https://git.bongbetic.com/xavierk/Fenris/issues/37)), so the manual flow above is primary. A dormant `.gitea/workflows/release.yml` (`on: push: tags: ['v*']`, single job, host-mode runner) is committed alongside; if it fires, it replicates `make release`. Cheapest future upgrade: one `act_runner` static binary in host-label mode on the existing Gitea host.
|
||||
|
||||
## 6. Package layout and ownership
|
||||
|
||||
Per [ADR 0007](../adr/0007-package-delivery-amends-0004.md) §2 — the dpkg/rpm database is the manifest; no `manifest.txt` ships:
|
||||
|
||||
| Artifact | Location | Ownership |
|
||||
|---|---|---|
|
||||
| Bundled runtime packages | `/opt/fenris/vendor` | package (tree) |
|
||||
| Wrapper | `/usr/bin/fenris` | package |
|
||||
| Helpers | `/usr/libexec/fenris/{fenris-monitor,fenris-collect}` | package — exactly these two, no new polkit-reachable binaries |
|
||||
| Units | `/usr/lib/systemd/system/fenris-collect.{timer,service}` | package (vendor placement; `/etc/systemd/system` is admin-only) |
|
||||
| Polkit policy | `/usr/share/polkit-1/actions/com.bongbetic.fenris.monitor.policy` | package |
|
||||
| sysusers fragment | `/usr/lib/sysusers.d/fenris.conf` (`g fenris -`) | package |
|
||||
| tmpfiles fragment | `/usr/lib/tmpfiles.d/fenris.conf` (`d /var/lib/fenris 2750 root fenris -`) | package |
|
||||
| Configuration | `/etc/fenris/fenris.conf` | package as conffile / `%config(noreplace)` — placeholder-commented default, no active selector |
|
||||
| Observation store | `/var/lib/fenris/observations.db` (+ WAL, `.bak`) | **never owned, never ghosted** — the package owns the directory only |
|
||||
|
||||
## 7. Maintainer-script contracts
|
||||
|
||||
- **preinst / %pre:** abort with a pointer to the migration runbook (§9) if `/var/lib/fenris/manifest.txt` **or** `/etc/systemd/system/fenris-collect.timer` exists (dual marker covers pre-manifest make installs). No auto-clean — scripts never delete files outside the package DB.
|
||||
- **postinst / %post (install):** `systemd-sysusers`, `systemd-tmpfiles --create`, `systemctl daemon-reload`. Nothing else — no enable, no preset, no start; no preset file ships.
|
||||
- **postinst / %post (upgrade):** snapshot `observations.db` → `.bak` (one generation) → forward-only schema migration via target `python3` with `/opt/fenris/vendor` on its import path → `daemon-reload` → restart `fenris-collect.timer` only if unit contents changed **and** it is active. `/var/lib/fenris` is never rebuilt; an in-flight oneshot finishes on its old interpreter.
|
||||
- **prerm / %preun:** sanctioned disable (`fenris-monitor disable --now`, closing the monitoring period `user_disabled`) on remove/erase **only, never on upgrade** — deb prerm upgrade case is a no-op; rpm `%preun` gated on `$1 -eq 0`.
|
||||
- **Removal mapping:** deb `remove` ≈ `make uninstall` (conffile + store survive); deb `purge` ≈ `make purge` (+ `.bak`, group cleanup); rpm erase ≈ `make uninstall` (unmodified config removed, modified survives as `.rpmsave`); rpm purge = documented manual command.
|
||||
|
||||
## 8. Initial configuration
|
||||
|
||||
- The device selector is **entered by hand**: root edits `/etc/fenris/fenris.conf` (world-readable, exactly one key per [ADR 0003](../adr/0003-service-lifecycle-and-sanctioned-toggle.md) §3). The shipped default is placeholder-commented and carries no active selector — a fresh install reads as a `configuration error`-free dormant system until the first `fenris monitor resume` + edit, exactly the dormant-install contract.
|
||||
- No configuration verb is added to `fenris-monitor`; the polkit surface stays at one binary. (This resolves the open item from [Package ownership + ADR 0004 amendment](https://git.bongbetic.com/xavierk/Fenris/issues/40): nothing in any delivery writes the selector — `make install` never wrote `fenris.conf` either; hand-editing has been the model since ADR 0003.)
|
||||
- On upgrade, local edits survive; a changed package default lands as `.dpkg-new` / `.rpmnew`.
|
||||
|
||||
## 9. Migration from make-install systems
|
||||
|
||||
- **Runbook only** — no migration script, no auto-clean. Population is author machines plus a few testers; store and config survive by path continuity.
|
||||
- **Remove-then-install, mandatory:** `sudo make uninstall` (preserves store + `/etc/fenris`) → `apt install fenris` / `dnf install fenris`. **Over-install is forbidden:** stale `/etc/systemd/system/fenris-collect.*` silently shadows vendor units (systemd precedence), `/usr/local/bin/fenris` shadows `/usr/bin/fenris` on PATH.
|
||||
- **No-move continuity:** `/var/lib/fenris` untouched; existing `fenris` group → sysusers no-op; existing dir → tmpfiles no-op; hand-written `fenris.conf` survives (dpkg ships the default as `.dpkg-new`; rpm as `.rpmnew`); store schema caught up by the upgrade-path migration.
|
||||
- **Reset-to-dormant:** `make uninstall`'s sanctioned disable closes the open period `user_disabled`; after migration the user opts back in with `fenris monitor resume` — one ≤15-min sample gap, honest against the endurance timeline.
|
||||
- **Mutual exclusion:** package and `make install` never on the same machine. The dev loop is checkout + `make test`; a dev-local install mode stays in map fog.
|
||||
|
||||
## 10. Out of scope
|
||||
|
||||
- Building and publishing packages (the follow-up execution effort).
|
||||
- Snap/Flatpak/AppImage/Homebrew, PyPI as an install path.
|
||||
- Stable/testing channel split, COPR/PPA fallback, arm64 — map fog until demand appears.
|
||||
|
||||
---
|
||||
|
||||
*Assembled from [Lock channel + toolchain](https://git.bongbetic.com/xavierk/Fenris/issues/38), [Signing + key policy](https://git.bongbetic.com/xavierk/Fenris/issues/39), [Package ownership + ADR 0004 amendment](https://git.bongbetic.com/xavierk/Fenris/issues/40), [Migration path from make-install systems to packages](https://git.bongbetic.com/xavierk/Fenris/issues/41), [Release cadence + stable/testing channel split](https://git.bongbetic.com/xavierk/Fenris/issues/43), [Actions runner availability](https://git.bongbetic.com/xavierk/Fenris/issues/37), and the research tickets [deb + rpm packaging toolchain](https://git.bongbetic.com/xavierk/Fenris/issues/34), [Gitea 1.27 package registry feasibility](https://git.bongbetic.com/xavierk/Fenris/issues/35), [OBS route](https://git.bongbetic.com/xavierk/Fenris/issues/36).*
|
||||
@@ -0,0 +1,128 @@
|
||||
#!/usr/bin/env bash
|
||||
# Fenris — interactive menu for the NVMe wear monitor & dashboard.
|
||||
# Created by Bongbetic.
|
||||
|
||||
set -euo pipefail
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
PY="$SCRIPT_DIR/fenris.py"
|
||||
PORT_DEFAULT=8420
|
||||
INTERVAL_DEFAULT=300
|
||||
|
||||
banner() {
|
||||
cat <<'EOF'
|
||||
_____ _
|
||||
| __|___ ___ _| |___
|
||||
| __| -_| | . | _|
|
||||
|__| |___|_|_|_|___|_|
|
||||
|
||||
NVMe wear monitor & live dashboard
|
||||
Created by Bongbetic
|
||||
EOF
|
||||
}
|
||||
|
||||
pause() { read -rp "Press Enter to continue..." _; }
|
||||
|
||||
detect_device() {
|
||||
# Query Python's auto-detect for the default device.
|
||||
python3 -c "import sys; sys.path.insert(0,'$SCRIPT_DIR'); from fenris import detect_device; print(detect_device())" 2>/dev/null || echo /dev/nvme0
|
||||
}
|
||||
|
||||
menu() {
|
||||
clear
|
||||
banner
|
||||
echo
|
||||
echo " 1) Start monitoring (background daemon + dashboard)"
|
||||
echo " 2) Stop monitoring"
|
||||
echo " 3) Status / current wear stats"
|
||||
echo " 4) Take one sample right now"
|
||||
echo " 5) Open dashboard URL"
|
||||
echo " ---"
|
||||
echo " h) Help / how this works"
|
||||
echo " q) Exit"
|
||||
echo
|
||||
read -rp "Choose an option: " choice
|
||||
echo
|
||||
case "$choice" in
|
||||
1) start_flow ;;
|
||||
2) python3 "$PY" stop; pause ;;
|
||||
3) python3 "$PY" status; pause ;;
|
||||
4) read -rp "Device [default: auto-detect]: " dev
|
||||
if [ -z "$dev" ]; then python3 "$PY" sample; else python3 "$PY" sample --device "$dev"; fi
|
||||
pause ;;
|
||||
5) show_url; pause ;;
|
||||
h|H) help_text; pause ;;
|
||||
q|Q) echo "Bye. — Fenris, by Bongbetic"; exit 0 ;;
|
||||
*) echo "Invalid choice."; pause ;;
|
||||
esac
|
||||
}
|
||||
|
||||
start_flow() {
|
||||
read -rp "NVMe device [Enter = auto-detect]: " dev
|
||||
read -rp "Sample interval in seconds [Enter = ${INTERVAL_DEFAULT}]: " interval
|
||||
read -rp "Dashboard port [Enter = ${PORT_DEFAULT}]: " port
|
||||
interval="${interval:-$INTERVAL_DEFAULT}"
|
||||
port="${port:-$PORT_DEFAULT}"
|
||||
|
||||
args=(start --interval "$interval" --port "$port")
|
||||
if [ -n "${dev:-}" ]; then args+=(--device "$dev"); fi
|
||||
|
||||
echo
|
||||
echo "Note: reading NVMe SMART data needs root."
|
||||
echo "Fenris runs 'sudo -n smartctl ...' (no-prompt sudo). If this fails,"
|
||||
echo "either run this menu with sudo, or allow passwordless smartctl via:"
|
||||
echo " sudo visudo -> youruser ALL=(root) NOPASSWD: /usr/sbin/smartctl"
|
||||
echo
|
||||
|
||||
python3 "$PY" "${args[@]}"
|
||||
pause
|
||||
}
|
||||
|
||||
show_url() {
|
||||
if [ -f "$SCRIPT_DIR/data/fenris.pid" ]; then
|
||||
# Try to read actual port from running process cmdline, else guess default.
|
||||
local pid port
|
||||
pid=$(<"$SCRIPT_DIR/data/fenris.pid")
|
||||
port=$(tr '\0' '\n' < /proc/"$pid"/cmdline 2>/dev/null | grep -A1 -- '--port' | tail -1 || true)
|
||||
port="${port:-$PORT_DEFAULT}"
|
||||
echo "Dashboard: http://localhost:${port}"
|
||||
else
|
||||
echo "Fenris is not currently running. Start it first (option 1)."
|
||||
fi
|
||||
}
|
||||
|
||||
help_text() {
|
||||
cat <<EOF
|
||||
What Fenris does:
|
||||
- Periodically reads your NVMe drive's SMART health data (via smartctl),
|
||||
including "percentage_used" (the drive's own wear indicator), total
|
||||
bytes written/read, temperature, spare capacity, and error counts.
|
||||
- Logs every sample to: $SCRIPT_DIR/data/history.jsonl
|
||||
and per-hour aggregates to: $SCRIPT_DIR/data/hourly.jsonl (GB/hour,
|
||||
rebuilt from history on restart). Used for the trailing-24h bar chart
|
||||
and exact rolling-24h write volume.
|
||||
- Serves a live HTML dashboard (dense layout, interval-synced polling)
|
||||
with wear-over-time and trailing-24h hourly-write charts, plus a
|
||||
projected life-remaining estimate in hours/days/years derived from
|
||||
your actual rolling-24h write rate and implied TBW endurance.
|
||||
|
||||
Requirements:
|
||||
- smartmontools (smartctl) installed.
|
||||
- Root access to read NVMe SMART logs — either run Fenris via sudo,
|
||||
or set up passwordless sudo for smartctl (see option 1).
|
||||
|
||||
CLI usage (equivalent to this menu):
|
||||
python3 fenris.py start [--device /dev/nvme0] [--interval 300] [--port 8420]
|
||||
python3 fenris.py stop
|
||||
python3 fenris.py status
|
||||
python3 fenris.py sample [--device /dev/nvme0]
|
||||
|
||||
Leave it running in the background (option 1) and check back after a
|
||||
few days/weeks of normal use — more samples = a more accurate lifespan
|
||||
estimate.
|
||||
|
||||
Fenris — created by Bongbetic.
|
||||
EOF
|
||||
}
|
||||
|
||||
# Loop instead of recurse to avoid stack overflow.
|
||||
while true; do menu; done
|
||||
@@ -1,16 +0,0 @@
|
||||
# Fenris configuration
|
||||
#
|
||||
# This file is managed by the fenris package. Local edits are preserved
|
||||
# across upgrades; changed defaults appear as .dpkg-new / .rpmnew.
|
||||
#
|
||||
# The device selector specifies which NVMe drive to monitor.
|
||||
# Uncomment and set exactly one device path:
|
||||
#
|
||||
# device = /dev/disk/by-id/nvme-Samsung_SSD_980_PRO_500GB_S5PANS0T123456
|
||||
#
|
||||
# The observation store path is optional and defaults to
|
||||
# /var/lib/fenris/observations.db when unset:
|
||||
#
|
||||
# store_path = /var/lib/fenris/observations.db
|
||||
#
|
||||
# See https://git.bongbetic.com/xavierk/Fenris for documentation.
|
||||
@@ -1,7 +0,0 @@
|
||||
[fenris]
|
||||
name=Fenris NVMe Monitor
|
||||
baseurl=https://git.bongbetic.com/api/packages/xavierk/rpm/fenris
|
||||
enabled=1
|
||||
gpgcheck=1
|
||||
gpgkey=https://git.bongbetic.com/xavierk/Fenris/raw/branch/main/packaging/keys/fenris-packaging.asc
|
||||
repo_gpgcheck=0
|
||||
@@ -1,40 +0,0 @@
|
||||
# Fenris Packaging Key
|
||||
#
|
||||
# Public half of dedicated RSA-3072 key used to sign RPM payloads and
|
||||
# clearsign SHA256SUMS manifests.
|
||||
#
|
||||
# Fingerprint: CE4542E1E23EB50F09EDFFA5A5E8B22D1872FB07
|
||||
# Algorithm: RSA 3072
|
||||
# UID: Fenris Packaging <packaging@bongbetic.com>
|
||||
# Expiry: 2 years from creation
|
||||
#
|
||||
# Raw URL:
|
||||
# https://git.bongbetic.com/xavierk/Fenris/raw/branch/main/packaging/keys/fenris-packaging.asc
|
||||
#
|
||||
# The private half lives in approved secret storage only. Each release uses
|
||||
# import -> sign -> delete. See docs/install/signing-key-ceremony.md.
|
||||
#
|
||||
-----BEGIN PGP PUBLIC KEY BLOCK-----
|
||||
|
||||
mQGNBGqZYxgBDADEyYQhndEEzoETD17vk8/4x2DoXQm9hxW7hiX3TQNSmXORpgjR
|
||||
NNt0vV/rTptwjmxgkrlevjrYqiBuoXfKJ0WRC16e9+NRnCGwJX5F4sR7jfgS9XbH
|
||||
pAbLbySll5LfrD6JcPcB4JsSishKkY6X0zHQD0/zrCaOsuNdLj+fLhWDoxjpLFGy
|
||||
92U7KHwtt87vSmUM4FgAjUY4keVKqIP5pSWIcPEy7z025RytL1JP6z3jBJR7KKD/
|
||||
MLXd2KTGGaxTIvzgimcvjQYqFxrT2YIRsmhVYzddRyUnYYWgOh9dp5xn9CRM48Lz
|
||||
klxHI/jI4lRPCQJy0atBpGZk1bRIc9XsBXRWiR9Zwvpjah/r3whkknBFv7C4srMr
|
||||
Fl0Ml897zwbCsCP/Ejs43SYt+BJ6B2z4lg6KVsG6lypqV8B+gT+E6rJZ/ML6UpS3
|
||||
2XQBlWDAw3hf0abjZIOQegi0f5igc7TVyDzYYihLOjKRwZuGSHY40w4mohxESLy8
|
||||
ogdMveFJKoRZvBsAEQEAAbQqRmVucmlzIFBhY2thZ2luZyA8cGFja2FnaW5nQGJv
|
||||
bmdiZXRpYy5jb20+iQH0BBMBCABeFiEEzkVC4eI+tQ8J7f+lpeiyLRhy+wcFAmqZ
|
||||
YxgbFIAAAAAABAAObWFudTIsMi41KzEuMTIsMiwyAxsvBAUJA8JnAAULCQgHAgIi
|
||||
AgYVCgkICwIEFgIDAQIeBwIXgAAKCRCl6LItGHL7B7qZC/4yFm3JwuhXuaJ6JvsH
|
||||
ZNVCAVOywktFbdcfKJYCXayaVsQ0Yc1w/gW6XhYCr4EECfWplnjtta9zPnN61ODD
|
||||
B8ZIuM9VUOqxpwvBWJnHcnny1FjmbJ0r0NOwmqKMj54cFHEDbVzmVPoshQSukThj
|
||||
Uz28XXw/JOkeQQaVl6OF3MoLvhLrLWvnqX310Z151dpl1lEA6gYWd1eKau2oIfU4
|
||||
e6u1JnX6mKWb0WaaEqo1QARXloQTaKV+NiSUavckTn1LXXMxGCFkbtNWYZv7uf+U
|
||||
WFB8KuaR3u8if7R8Bab7Y0lzmPCCeSkLHXDLq9FyfDdGOj5LyXxxzlVnTTUyRANh
|
||||
+JwGiokDqUzh3yUdUYnx6pE6+3tcxP+Gp92K/GZXulmPQYhw0sSyqfnK8GAtUVaD
|
||||
KAnrV7fKZ9jve87NWeb3G0xfQiH9mNSsEmnQVzd/DCuczOf5fMFfHlUNgkI4t4G7
|
||||
+hTpKrOOubozZwfB23mdM+H9pxwWFN6To85Iy1ge9JKTFTY=
|
||||
=V4/R
|
||||
-----END PGP PUBLIC KEY BLOCK-----
|
||||
@@ -1,60 +0,0 @@
|
||||
name: fenris
|
||||
arch: amd64
|
||||
platform: linux
|
||||
version: "${VERSION}"
|
||||
maintainer: Fenris Maintainers <ops@bongbetic.com>
|
||||
description: >
|
||||
NVMe wear monitor with persistent TUI — observes real-world drive use and
|
||||
translates it into an understandable endurance outlook.
|
||||
homepage: https://git.bongbetic.com/xavierk/Fenris
|
||||
license: Proprietary
|
||||
|
||||
depends:
|
||||
- python3 (>= 3.10)
|
||||
- smartmontools
|
||||
- systemd
|
||||
|
||||
contents:
|
||||
# Staged tree: runtime packages, wrapper, helpers, units, polkit, sysusers, tmpfiles
|
||||
- src: build/stage/
|
||||
dst: /
|
||||
type: tree
|
||||
|
||||
# Configuration directory
|
||||
- dst: /etc/fenris
|
||||
type: dir
|
||||
file_info:
|
||||
mode: 0755
|
||||
|
||||
# Default placeholder-commented config (deb conffile / rpm %config(noreplace))
|
||||
- src: packaging/fenris.conf
|
||||
dst: /etc/fenris/fenris.conf
|
||||
type: config|noreplace
|
||||
file_info:
|
||||
mode: 0644
|
||||
|
||||
# Observation store directory — owned by package, never packed.
|
||||
# Store files (observations.db, WAL sidecars, .bak) are never owned.
|
||||
- dst: /var/lib/fenris
|
||||
type: dir
|
||||
file_info:
|
||||
mode: 2750
|
||||
group: fenris
|
||||
|
||||
scripts:
|
||||
preinstall: packaging/preinst.sh
|
||||
postinstall: packaging/postinst.sh
|
||||
preremove: packaging/prerm.sh
|
||||
postremove: packaging/postrm.sh
|
||||
|
||||
overrides:
|
||||
rpm:
|
||||
depends:
|
||||
- python3 >= 3.10
|
||||
- smartmontools
|
||||
- systemd
|
||||
scripts:
|
||||
preinstall: packaging/preinst.sh
|
||||
postinstall: packaging/rpm/post.sh
|
||||
preremove: packaging/rpm/preun.sh
|
||||
postremove: packaging/rpm/postun.sh
|
||||
@@ -1,56 +0,0 @@
|
||||
#!/bin/sh
|
||||
# postinst — deb install and upgrade paths (spec §7).
|
||||
#
|
||||
# dpkg calls: postinst configure [most-recently-configured-version]
|
||||
# fresh install: $1 = "configure", $2 = ""
|
||||
# upgrade: $1 = "configure", $2 = old version
|
||||
set -eu
|
||||
|
||||
STORE_DIR="/var/lib/fenris"
|
||||
STORE_DB="${STORE_DIR}/observations.db"
|
||||
STORE_BAK="${STORE_DIR}/observations.db.bak"
|
||||
RUNTIME_PYTHON="/usr/bin/python3"
|
||||
VENDOR_DIR="/opt/fenris/vendor"
|
||||
|
||||
case "${1:-}" in
|
||||
configure)
|
||||
if [ -n "${2:-}" ]; then
|
||||
# Upgrade — snapshot, migration, daemon-reload, conditional timer restart
|
||||
if [ -f "${STORE_DB}" ]; then
|
||||
cp "${STORE_DB}" "${STORE_BAK}" 2>/dev/null || true
|
||||
fi
|
||||
if [ -d "${VENDOR_DIR}" ] && [ -f "${STORE_DB}" ]; then
|
||||
PYTHONPATH="${VENDOR_DIR}" "${RUNTIME_PYTHON}" -c "
|
||||
from fenris.store import migrate_to_latest
|
||||
from pathlib import Path
|
||||
n = migrate_to_latest(Path('${STORE_DB}'))
|
||||
print(f'Fenris migration: {n} step(s) applied') if n else None
|
||||
" 2>&1 || echo "Fenris: migration skipped (store not yet initialized)"
|
||||
fi
|
||||
# Capture running unit content BEFORE daemon-reload (spec §7)
|
||||
RUNNING_UNITS=""
|
||||
for unit in fenris-collect.timer; do
|
||||
if systemctl is-active --quiet "${unit}" 2>/dev/null; then
|
||||
RUNNING_UNITS="${RUNNING_UNITS} ${unit}"
|
||||
fi
|
||||
done
|
||||
systemctl daemon-reload 2>/dev/null || true
|
||||
# Restart timer only if unit contents changed AND active
|
||||
for unit in ${RUNNING_UNITS}; do
|
||||
OLD_CONTENT="$(mktemp)"
|
||||
NEW_PATH="/usr/lib/systemd/system/${unit}"
|
||||
systemctl cat "${unit}" > "${OLD_CONTENT}" 2>/dev/null || true
|
||||
if ! diff -q "${OLD_CONTENT}" "${NEW_PATH}" > /dev/null 2>&1; then
|
||||
systemctl restart "${unit}" 2>/dev/null || true
|
||||
fi
|
||||
rm -f "${OLD_CONTENT}"
|
||||
done
|
||||
fi
|
||||
# sysusers, tmpfiles, daemon-reload (both fresh install and upgrade)
|
||||
systemd-sysusers || true
|
||||
systemd-tmpfiles --create || true
|
||||
systemctl daemon-reload || true
|
||||
;;
|
||||
abort-upgrade|abort-install|disappear)
|
||||
;;
|
||||
esac
|
||||
@@ -1,19 +0,0 @@
|
||||
#!/bin/sh
|
||||
# postrm — deb post-removal (spec §7, §9).
|
||||
#
|
||||
# dpkg calls: postrm remove (after package files removed)
|
||||
# postrm purge (after conffiles and config removed)
|
||||
set -eu
|
||||
|
||||
case "${1:-}" in
|
||||
purge)
|
||||
rm -rf /etc/fenris
|
||||
rm -rf /var/lib/fenris
|
||||
if getent group fenris > /dev/null 2>&1; then
|
||||
groupdel fenris 2>/dev/null || true
|
||||
fi
|
||||
;;
|
||||
remove|upgrade|failed-upgrade|abort-install|abort-upgrade|disappear)
|
||||
;;
|
||||
esac
|
||||
systemctl daemon-reload 2>/dev/null || true
|
||||
@@ -1,19 +0,0 @@
|
||||
#!/bin/sh
|
||||
# preinst — abort if make-install remnants detected (spec §7, §9).
|
||||
set -eu
|
||||
|
||||
MARKER1="/var/lib/fenris/manifest.txt"
|
||||
MARKER2="/etc/systemd/system/fenris-collect.timer"
|
||||
|
||||
if [ -f "${MARKER1}" ] || [ -f "${MARKER2}" ]; then
|
||||
echo >&2
|
||||
echo >&2 "Fenris make-install remnants detected — refusing to install."
|
||||
echo >&2
|
||||
echo >&2 "Migrate to the package with:"
|
||||
echo >&2 " sudo make uninstall # removes make-install files, preserves store + config"
|
||||
echo >&2 " sudo apt install fenris # or: sudo dnf install fenris"
|
||||
echo >&2
|
||||
echo >&2 "See: https://git.bongbetic.com/xavierk/Fenris/blob/main/docs/spec/release-packaging.md#9-migration-from-make-install-systems"
|
||||
echo >&2
|
||||
exit 1
|
||||
fi
|
||||
@@ -1,20 +0,0 @@
|
||||
#!/bin/sh
|
||||
# prerm — deb pre-removal (spec §7).
|
||||
#
|
||||
# dpkg calls: prerm remove (package being removed)
|
||||
# prerm upgrade (old version about to be replaced)
|
||||
set -eu
|
||||
|
||||
case "${1:-}" in
|
||||
remove)
|
||||
# Sanctioned disable — close monitoring period (spec §7)
|
||||
if [ -x /usr/libexec/fenris/fenris-monitor ]; then
|
||||
/usr/libexec/fenris/fenris-monitor disable --now 2>/dev/null || true
|
||||
fi
|
||||
systemctl stop fenris-collect.timer 2>/dev/null || true
|
||||
systemctl disable fenris-collect.timer 2>/dev/null || true
|
||||
;;
|
||||
upgrade)
|
||||
# Never interrupt monitoring on upgrade
|
||||
;;
|
||||
esac
|
||||
@@ -1,20 +0,0 @@
|
||||
## Install
|
||||
|
||||
Install Fenris from its package channel after following the [package setup instructions](https://git.bongbetic.com/xavierk/Fenris/src/branch/main/README.md#install-from-package-recommended):
|
||||
|
||||
```bash
|
||||
sudo apt update && sudo apt install fenris # Debian / Ubuntu
|
||||
sudo dnf install fenris # Fedora
|
||||
sudo zypper install fenris # openSUSE Tumbleweed
|
||||
```
|
||||
|
||||
## Verify downloads
|
||||
|
||||
```bash
|
||||
gpg --output SHA256SUMS --decrypt SHA256SUMS.asc
|
||||
sha256sum -c SHA256SUMS
|
||||
```
|
||||
|
||||
## Rollback
|
||||
|
||||
Installing an older package over a newer observation store is unsupported. Restore the observation-store snapshot, then install the earlier Release; see the [upgrade and rollback guidance](https://git.bongbetic.com/xavierk/Fenris/src/branch/main/README.md#upgrade).
|
||||
@@ -1,49 +0,0 @@
|
||||
#!/bin/sh
|
||||
# RPM %post — post-install/upgrade scriptlet (spec §7).
|
||||
set -eu
|
||||
|
||||
STORE_DIR="/var/lib/fenris"
|
||||
STORE_DB="${STORE_DIR}/observations.db"
|
||||
STORE_BAK="${STORE_DIR}/observations.db.bak"
|
||||
RUNTIME_PYTHON="/usr/bin/python3"
|
||||
VENDOR_DIR="/opt/fenris/vendor"
|
||||
|
||||
if [ "$1" -eq 1 ]; then
|
||||
# Fresh install
|
||||
systemd-sysusers || true
|
||||
systemd-tmpfiles --create || true
|
||||
systemctl daemon-reload || true
|
||||
elif [ "$1" -ge 2 ]; then
|
||||
# Upgrade — snapshot, migration, daemon-reload, conditional timer restart
|
||||
if [ -f "${STORE_DB}" ]; then
|
||||
cp "${STORE_DB}" "${STORE_BAK}" 2>/dev/null || true
|
||||
fi
|
||||
if [ -d "${VENDOR_DIR}" ] && [ -f "${STORE_DB}" ]; then
|
||||
PYTHONPATH="${VENDOR_DIR}" "${RUNTIME_PYTHON}" -c "
|
||||
from fenris.store import migrate_to_latest
|
||||
from pathlib import Path
|
||||
n = migrate_to_latest(Path('${STORE_DB}'))
|
||||
print(f'Fenris migration: {n} step(s) applied') if n else None
|
||||
" 2>&1 || echo "Fenris: migration skipped (store not yet initialized)"
|
||||
fi
|
||||
# Capture running unit content BEFORE daemon-reload (spec §7)
|
||||
RUNNING_UNITS=""
|
||||
for unit in fenris-collect.timer; do
|
||||
if systemctl is-active --quiet "${unit}" 2>/dev/null; then
|
||||
RUNNING_UNITS="${RUNNING_UNITS} ${unit}"
|
||||
fi
|
||||
done
|
||||
systemctl daemon-reload 2>/dev/null || true
|
||||
# Restart timer only if unit contents changed AND active
|
||||
for unit in ${RUNNING_UNITS}; do
|
||||
OLD_CONTENT="$(mktemp)"
|
||||
NEW_PATH="/usr/lib/systemd/system/${unit}"
|
||||
systemctl cat "${unit}" > "${OLD_CONTENT}" 2>/dev/null || true
|
||||
if ! diff -q "${OLD_CONTENT}" "${NEW_PATH}" > /dev/null 2>&1; then
|
||||
systemctl restart "${unit}" 2>/dev/null || true
|
||||
fi
|
||||
rm -f "${OLD_CONTENT}"
|
||||
done
|
||||
# Re-apply placement modes (store dir group access, issue #54)
|
||||
systemd-tmpfiles --create || true
|
||||
fi
|
||||
@@ -1,13 +0,0 @@
|
||||
#!/bin/sh
|
||||
# RPM %postun — post-uninstall scriptlet (spec §7).
|
||||
set -eu
|
||||
|
||||
if [ "$1" -eq 0 ]; then
|
||||
# Package fully erased — remove config, store, group
|
||||
rm -rf /etc/fenris
|
||||
rm -rf /var/lib/fenris
|
||||
if getent group fenris > /dev/null 2>&1; then
|
||||
groupdel fenris 2>/dev/null || true
|
||||
fi
|
||||
fi
|
||||
systemctl daemon-reload 2>/dev/null || true
|
||||
@@ -1,13 +0,0 @@
|
||||
#!/bin/sh
|
||||
# RPM %preun — pre-uninstall scriptlet (spec §7).
|
||||
set -eu
|
||||
|
||||
if [ "$1" -eq 0 ]; then
|
||||
# Package is being erased — sanctioned disable (spec §7)
|
||||
if [ -x /usr/libexec/fenris/fenris-monitor ]; then
|
||||
/usr/libexec/fenris/fenris-monitor disable --now 2>/dev/null || true
|
||||
fi
|
||||
systemctl stop fenris-collect.timer 2>/dev/null || true
|
||||
systemctl disable fenris-collect.timer 2>/dev/null || true
|
||||
fi
|
||||
# On upgrade ($1 -ge 1): do nothing
|
||||
@@ -1,91 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Stage a packaging tree at build/stage/ for nfpm consumption.
|
||||
#
|
||||
# Usage: packaging/stage.sh [VERSION]
|
||||
#
|
||||
# VERSION defaults to the version in pyproject.toml.
|
||||
# The staged tree contains:
|
||||
# /opt/fenris/vendor/ — bundled pure-Python application dependencies
|
||||
# /usr/bin/fenris — unprivileged wrapper
|
||||
# /usr/libexec/fenris/ — fenris-monitor, fenris-collect
|
||||
# /usr/lib/systemd/system/ — fenris-collect.{timer,service}
|
||||
# /usr/share/polkit-1/actions/ — polkit policy
|
||||
# /usr/lib/sysusers.d/fenris.conf
|
||||
# /usr/lib/tmpfiles.d/fenris.conf
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
STAGE_DIR="${REPO_ROOT}/build/stage"
|
||||
|
||||
# --- Resolve version ---
|
||||
if [ -n "${1:-}" ]; then
|
||||
VERSION="$1"
|
||||
else
|
||||
VERSION="$(sed -n 's/^version = "\(.*\)"/\1/p' "${REPO_ROOT}/pyproject.toml")"
|
||||
fi
|
||||
|
||||
if [ -z "${VERSION}" ]; then
|
||||
echo "Error: could not determine version" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Staging fenris ${VERSION} ..."
|
||||
|
||||
# --- Clean previous stage ---
|
||||
rm -rf "${STAGE_DIR}"
|
||||
mkdir -p "${STAGE_DIR}"
|
||||
|
||||
# --- Use pre-built wheel from dist/ ---
|
||||
WHEEL=$(ls "${REPO_ROOT}"/dist/fenris-"${VERSION}"-*.whl 2>/dev/null | head -1)
|
||||
if [ -z "${WHEEL}" ]; then
|
||||
echo "Error: no wheel found in dist/ — run 'make dist/fenris-*.whl' first" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo " Using wheel: $(basename "${WHEEL}")"
|
||||
|
||||
# --- Vendor runtime packages without an interpreter ---
|
||||
# A copied Python binary contains an ABI and build-host dynamic-library path.
|
||||
# It fails after rolling-distribution Python upgrades (for example Tumbleweed
|
||||
# 3.12 -> 3.13). Fenris and its locked dependencies are pure Python, so place
|
||||
# them in a version-neutral directory and execute with the target's python3.
|
||||
echo " Installing version-neutral runtime packages ..."
|
||||
VENDOR_DIR="${STAGE_DIR}/opt/fenris/vendor"
|
||||
mkdir -p "${VENDOR_DIR}"
|
||||
python3 -m pip install --disable-pip-version-check --no-compile \
|
||||
--target "${VENDOR_DIR}" -r "${REPO_ROOT}/requirements.txt" "${WHEEL}"
|
||||
|
||||
# --- Inject version into wrapper from pyproject.toml ---
|
||||
# The wrapper has a hardcoded version string; patch it for packaging.
|
||||
WRAPPER_SRC="${REPO_ROOT}/scripts/fenris"
|
||||
WRAPPER_DST="${STAGE_DIR}/usr/bin/fenris"
|
||||
mkdir -p "$(dirname "${WRAPPER_DST}")"
|
||||
sed "s|version=\"%(prog)s [0-9.]*\"|version=\"%(prog)s ${VERSION}\"|g" \
|
||||
"${WRAPPER_SRC}" > "${WRAPPER_DST}"
|
||||
chmod 0755 "${WRAPPER_DST}"
|
||||
|
||||
# --- Privileged helpers ---
|
||||
echo " Installing helpers ..."
|
||||
mkdir -p "${STAGE_DIR}/usr/libexec/fenris"
|
||||
install -m 0755 "${REPO_ROOT}/src/fenris/monitor.py" "${STAGE_DIR}/usr/libexec/fenris/fenris-monitor"
|
||||
install -m 0755 "${REPO_ROOT}/src/fenris/collect.py" "${STAGE_DIR}/usr/libexec/fenris/fenris-collect"
|
||||
|
||||
# --- systemd units (vendor placement) ---
|
||||
echo " Installing systemd units ..."
|
||||
mkdir -p "${STAGE_DIR}/usr/lib/systemd/system"
|
||||
install -m 0644 "${REPO_ROOT}/units/fenris-collect.timer" "${STAGE_DIR}/usr/lib/systemd/system/"
|
||||
install -m 0644 "${REPO_ROOT}/units/fenris-collect.service" "${STAGE_DIR}/usr/lib/systemd/system/"
|
||||
|
||||
# --- polkit policy ---
|
||||
echo " Installing polkit policy ..."
|
||||
mkdir -p "${STAGE_DIR}/usr/share/polkit-1/actions"
|
||||
install -m 0644 "${REPO_ROOT}/polkit/com.bongbetic.fenris.monitor.policy" \
|
||||
"${STAGE_DIR}/usr/share/polkit-1/actions/"
|
||||
|
||||
# --- sysusers and tmpfiles fragments ---
|
||||
echo " Installing sysusers/tmpfiles fragments ..."
|
||||
mkdir -p "${STAGE_DIR}/usr/lib/sysusers.d"
|
||||
install -m 0644 "${REPO_ROOT}/packaging/sysusers.d/fenris.conf" "${STAGE_DIR}/usr/lib/sysusers.d/"
|
||||
|
||||
mkdir -p "${STAGE_DIR}/usr/lib/tmpfiles.d"
|
||||
install -m 0644 "${REPO_ROOT}/packaging/tmpfiles.d/fenris.conf" "${STAGE_DIR}/usr/lib/tmpfiles.d/"
|
||||
|
||||
echo "Stage complete: ${STAGE_DIR}"
|
||||
@@ -1,3 +0,0 @@
|
||||
# System user/group for Fenris observation store access
|
||||
# Created by systemd-sysusers during package install
|
||||
g fenris -
|
||||
@@ -1,2 +0,0 @@
|
||||
# Type Path Mode User Group Age Argument
|
||||
d /var/lib/fenris 2770 root fenris - -
|
||||
@@ -1,21 +0,0 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE policyconfig PUBLIC
|
||||
"-//freedesktop//DTD PolicyKit Policy Configuration 1.0//EN"
|
||||
"http://www.freedesktop.org/standards/PolicyKit/1/policyconfig.dtd">
|
||||
<policyconfig>
|
||||
<vendor>bongbetic</vendor>
|
||||
<vendor_url>https://bongbetic.com</vendor_url>
|
||||
|
||||
<action id="com.bongbetic.fenris.monitor">
|
||||
<description>Fenris Monitor Helper</description>
|
||||
<message>Authentication is required to manage Fenris monitoring.</message>
|
||||
|
||||
<defaults>
|
||||
<allow_any>no</allow_any>
|
||||
<allow_inactive>no</allow_inactive>
|
||||
<allow_active>auth_admin</allow_active>
|
||||
</defaults>
|
||||
|
||||
<annotate key="org.freedesktop.policykit.imply">org.freedesktop.systemd1.manage-units</annotate>
|
||||
</action>
|
||||
</policyconfig>
|
||||
@@ -1,23 +0,0 @@
|
||||
[project]
|
||||
name = "fenris"
|
||||
version = "0.3.4"
|
||||
description = "NVMe wear monitor with persistent TUI"
|
||||
requires-python = ">=3.9"
|
||||
dependencies = [
|
||||
"textual>=0.40.0",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
dev = [
|
||||
"pytest>=7.0.0",
|
||||
"pytest-cov>=4.0.0",
|
||||
"pytest-asyncio>=0.20.0",
|
||||
]
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
testpaths = ["tests"]
|
||||
python_files = ["test_*.py"]
|
||||
python_functions = ["test_*"]
|
||||
markers = [
|
||||
"slow: marks tests as slow",
|
||||
]
|
||||
@@ -1,11 +0,0 @@
|
||||
# Fenris dependency lockfile
|
||||
# Exact pins for reproducible installs (IN-8)
|
||||
# Refresh with: make update-deps
|
||||
textual==8.2.8
|
||||
rich==15.0.0
|
||||
markdown-it-py==4.2.0
|
||||
mdit-py-plugins==0.6.1
|
||||
mdurl==0.1.2
|
||||
platformdirs==4.11.7
|
||||
Pygments==2.21.0
|
||||
linkify-it-py==2.2.0
|
||||
@@ -1,99 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Extract one validated Keep a Changelog version section."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
import re
|
||||
import sys
|
||||
|
||||
|
||||
class ChangelogError(ValueError):
|
||||
"""A release cannot safely use the supplied changelog."""
|
||||
|
||||
|
||||
_SEMVER = r"(?:0|[1-9]\d*)\.(?:0|[1-9]\d*)\.(?:0|[1-9]\d*)"
|
||||
_VERSION_HEADING = re.compile(
|
||||
rf"^## \[(?P<version>{_SEMVER})\] - (?P<date>.+)$", re.MULTILINE
|
||||
)
|
||||
|
||||
|
||||
def extract_version_section(changelog: str, version: str) -> str:
|
||||
"""Return *version*'s changelog section without altering its bytes.
|
||||
|
||||
The section ends immediately before the next level-two heading. A release
|
||||
cannot use an absent, empty, or malformed version section.
|
||||
"""
|
||||
if not re.fullmatch(_SEMVER, version):
|
||||
raise ChangelogError(f"requested version is not bare semver: {version!r}")
|
||||
|
||||
heading = next(
|
||||
(match for match in _VERSION_HEADING.finditer(changelog)
|
||||
if match.group("version") == version),
|
||||
None,
|
||||
)
|
||||
if heading is None:
|
||||
if re.search(rf"^## \[{re.escape(version)}\].*$", changelog, re.MULTILINE):
|
||||
raise ChangelogError(f"version {version} has a malformed heading or date")
|
||||
raise ChangelogError(f"version {version} is missing from the changelog")
|
||||
|
||||
heading_date = heading.group("date")
|
||||
if not re.fullmatch(r"\d{4}-\d{2}-\d{2}", heading_date):
|
||||
raise ChangelogError(f"version {version} has a malformed release date")
|
||||
try:
|
||||
date.fromisoformat(heading_date)
|
||||
except ValueError as error:
|
||||
raise ChangelogError(f"version {version} has a malformed release date") from error
|
||||
|
||||
next_heading = re.search(r"^## ", changelog[heading.end():], re.MULTILINE)
|
||||
section_end = heading.end() + next_heading.start() if next_heading else len(changelog)
|
||||
section = changelog[heading.start():section_end]
|
||||
if not re.search(r"^- \S", section[heading.end() - heading.start():], re.MULTILINE):
|
||||
raise ChangelogError(f"version {version} has an empty changelog section")
|
||||
return section
|
||||
|
||||
|
||||
def extract_changelog(path: Path, version: str) -> str:
|
||||
"""Read and extract a requested version from a changelog file."""
|
||||
try:
|
||||
return extract_version_section(path.read_text(encoding="utf-8"), version)
|
||||
except OSError as error:
|
||||
raise ChangelogError(f"cannot read changelog {path}: {error.strerror}") from error
|
||||
|
||||
|
||||
def assemble_release_body(section: str, footer: str) -> str:
|
||||
"""Append standing guidance while preserving the extracted section verbatim."""
|
||||
separator = "\n" if section.endswith("\n") else "\n\n"
|
||||
return f"{section}{separator}{footer}"
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("changelog", type=Path)
|
||||
parser.add_argument("version")
|
||||
parser.add_argument(
|
||||
"--footer",
|
||||
type=Path,
|
||||
help="append this standing release guidance after the extracted section",
|
||||
)
|
||||
args = parser.parse_args(argv)
|
||||
try:
|
||||
section = extract_changelog(args.changelog, args.version)
|
||||
if args.footer:
|
||||
try:
|
||||
footer = args.footer.read_text(encoding="utf-8")
|
||||
except OSError as error:
|
||||
raise ChangelogError(
|
||||
f"cannot read release footer {args.footer}: {error.strerror}"
|
||||
) from error
|
||||
section = assemble_release_body(section, footer)
|
||||
sys.stdout.write(section)
|
||||
except ChangelogError as error:
|
||||
print(f"::error::{error}", file=sys.stderr)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
-227
@@ -1,227 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""fenris: unprivileged entry point for the Fenris TUI and CLI.
|
||||
|
||||
With no arguments, opens the TUI.
|
||||
Subcommands route through fenris-monitor for privileged operations.
|
||||
|
||||
Spec: §1.2, §8.4
|
||||
"""
|
||||
import argparse
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def add_runtime_packages() -> None:
|
||||
"""Make the package-owned, pure-Python dependencies importable.
|
||||
|
||||
RPM and deb installations deliberately use the target system's Python.
|
||||
Their dependencies are vendored without a copied interpreter so a distro
|
||||
Python minor-version update cannot leave Fenris linked to a removed ABI.
|
||||
The legacy development install keeps its venv fallback.
|
||||
"""
|
||||
runtime_dir = Path("/opt/fenris")
|
||||
vendor_dir = runtime_dir / "vendor"
|
||||
if vendor_dir.is_dir():
|
||||
sys.path.insert(0, str(vendor_dir))
|
||||
return
|
||||
|
||||
site_packages = next((runtime_dir / "lib").glob("python*/site-packages"), None)
|
||||
if site_packages:
|
||||
sys.path.insert(0, str(site_packages))
|
||||
|
||||
|
||||
add_runtime_packages()
|
||||
|
||||
|
||||
def is_root() -> bool:
|
||||
"""Check if running as root."""
|
||||
return os.geteuid() == 0
|
||||
|
||||
|
||||
def run_monitor(*args: str) -> None:
|
||||
"""Run fenris-monitor with the given arguments.
|
||||
|
||||
If not root, re-exec under pkexec.
|
||||
"""
|
||||
monitor_cmd = "/usr/libexec/fenris/fenris-monitor"
|
||||
|
||||
if is_root():
|
||||
result = subprocess.run([monitor_cmd] + list(args))
|
||||
sys.exit(result.returncode)
|
||||
else:
|
||||
# Use pkexec to elevate
|
||||
pkexec = subprocess.run(
|
||||
["which", "pkexec"], capture_output=True
|
||||
)
|
||||
if pkexec.returncode != 0:
|
||||
print(
|
||||
"Error: No polkit agent available. "
|
||||
"Run as root: sudo fenris-monitor ...",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
result = subprocess.run(["pkexec", monitor_cmd] + list(args))
|
||||
sys.exit(result.returncode)
|
||||
|
||||
|
||||
def cmd_tui(args: argparse.Namespace) -> None:
|
||||
"""Open the TUI."""
|
||||
from fenris.tui import run_tui
|
||||
run_tui()
|
||||
|
||||
|
||||
def cmd_status(args: argparse.Namespace) -> None:
|
||||
"""Show status."""
|
||||
from fenris.status import render_status
|
||||
print(render_status())
|
||||
|
||||
|
||||
def cmd_sample(args: argparse.Namespace) -> None:
|
||||
"""Trigger on-demand collection."""
|
||||
run_monitor("collect")
|
||||
|
||||
|
||||
def cmd_monitor_pause(args: argparse.Namespace) -> None:
|
||||
"""Pause monitoring."""
|
||||
# Pause asks confirmation (§7.4)
|
||||
if not args.yes:
|
||||
response = input("Pause monitoring? [y/N] ")
|
||||
if response.lower() not in ("y", "yes"):
|
||||
print("Aborted.")
|
||||
return
|
||||
|
||||
run_monitor("disable", "--now")
|
||||
|
||||
|
||||
def cmd_monitor_resume(args: argparse.Namespace) -> None:
|
||||
"""Resume monitoring."""
|
||||
# Resume does not ask confirmation (§7.4)
|
||||
run_monitor("enable", "--now")
|
||||
|
||||
|
||||
def cmd_baseline_set(args: argparse.Namespace) -> None:
|
||||
"""Set baseline."""
|
||||
run_monitor("baseline", "set", args.baseline_json)
|
||||
|
||||
|
||||
def cmd_baseline_clear(args: argparse.Namespace) -> None:
|
||||
"""Clear baseline."""
|
||||
run_monitor("baseline", "clear")
|
||||
|
||||
|
||||
def cmd_import(args: argparse.Namespace) -> None:
|
||||
"""Import legacy history."""
|
||||
# This is a one-off migration, not a privileged operation
|
||||
print("Legacy import: use fenris-import directly")
|
||||
|
||||
|
||||
def cmd_migrate(args: argparse.Namespace) -> None:
|
||||
"""Apply forward-only schema migrations (IN-5, IN-6).
|
||||
|
||||
Called by 'sudo make upgrade'. Raises on newer-schema store.
|
||||
"""
|
||||
from fenris.store import migrate_to_latest
|
||||
from pathlib import Path
|
||||
|
||||
store_path = Path("/var/lib/fenris/observations.db")
|
||||
if not store_path.exists():
|
||||
print("No observation store found — nothing to migrate.")
|
||||
return
|
||||
|
||||
steps = migrate_to_latest(store_path)
|
||||
if steps:
|
||||
print(f"Migration complete: {steps} step(s) applied.")
|
||||
else:
|
||||
print("Schema already current.")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="fenris",
|
||||
description="Fenris NVMe endurance monitor",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--version", action="version", version="%(prog)s 0.3.0"
|
||||
)
|
||||
|
||||
subparsers = parser.add_subparsers(dest="command")
|
||||
|
||||
# Default: TUI (no subcommand)
|
||||
subparsers.add_parser("tui", help="Open the TUI (default)")
|
||||
|
||||
# Status
|
||||
subparsers.add_parser("status", help="Show status")
|
||||
|
||||
# Sample (on-demand collection)
|
||||
subparsers.add_parser("sample", help="Trigger on-demand collection")
|
||||
|
||||
# Monitor subcommand
|
||||
monitor_parser = subparsers.add_parser("monitor", help="Monitor control")
|
||||
monitor_sub = monitor_parser.add_subparsers(dest="monitor_action")
|
||||
|
||||
# monitor pause
|
||||
pause_parser = monitor_sub.add_parser("pause", help="Pause monitoring")
|
||||
pause_parser.add_argument(
|
||||
"-y", "--yes", action="store_true", help="Skip confirmation"
|
||||
)
|
||||
pause_parser.set_defaults(func=cmd_monitor_pause)
|
||||
|
||||
# monitor resume
|
||||
resume_parser = monitor_sub.add_parser("resume", help="Resume monitoring")
|
||||
resume_parser.set_defaults(func=cmd_monitor_resume)
|
||||
|
||||
# Baseline subcommand
|
||||
baseline_parser = subparsers.add_parser("baseline", help="Baseline operations")
|
||||
baseline_sub = baseline_parser.add_subparsers(dest="baseline_action")
|
||||
|
||||
baseline_set = baseline_sub.add_parser("set", help="Set baseline")
|
||||
baseline_set.add_argument("baseline_json", help="Baseline JSON data")
|
||||
baseline_set.set_defaults(func=cmd_baseline_set)
|
||||
|
||||
baseline_clear = baseline_sub.add_parser("clear", help="Clear baseline")
|
||||
baseline_clear.set_defaults(func=cmd_baseline_clear)
|
||||
|
||||
# Import
|
||||
import_parser = subparsers.add_parser("import", help="Import legacy history")
|
||||
import_parser.add_argument("path", help="Path to history.jsonl")
|
||||
import_parser.set_defaults(func=cmd_import)
|
||||
|
||||
# Migrate (IN-5, IN-6) — called by upgrade, not for human use
|
||||
migrate_parser = subparsers.add_parser("migrate", help=argparse.SUPPRESS)
|
||||
migrate_parser.set_defaults(func=cmd_migrate)
|
||||
|
||||
# Rejected commands
|
||||
for cmd in ["start", "stop", "run"]:
|
||||
reject_parser = subparsers.add_parser(cmd, help=argparse.SUPPRESS)
|
||||
reject_parser.set_defaults(func=lambda a: print(
|
||||
f"'{cmd}' is not a valid command. "
|
||||
f"Use 'fenris monitor resume' instead.",
|
||||
file=sys.stderr,
|
||||
))
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.command is None or args.command == "tui":
|
||||
cmd_tui(args)
|
||||
elif args.command == "status":
|
||||
cmd_status(args)
|
||||
elif args.command == "sample":
|
||||
cmd_sample(args)
|
||||
elif args.command == "monitor":
|
||||
if args.monitor_action is None:
|
||||
monitor_parser.error("a subcommand is required")
|
||||
args.func(args)
|
||||
elif args.command == "baseline":
|
||||
if args.baseline_action is None:
|
||||
baseline_parser.error("a subcommand is required")
|
||||
args.func(args)
|
||||
elif args.command == "import":
|
||||
cmd_import(args)
|
||||
elif args.command == "migrate":
|
||||
cmd_migrate(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,206 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
# Fenris one-command release flow (issue #52).
|
||||
# Builds both packages, signs, uploads to registry, creates release entry,
|
||||
# and attaches artifacts — or in dry-run mode, prints every command.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/release.sh --dry-run # Print commands without executing
|
||||
# scripts/release.sh --publish # Execute the full release flow
|
||||
#
|
||||
# Environment:
|
||||
# GITEA_TOKEN - API token for Gitea registry and release API
|
||||
# PACKAGING_KEY - GPG key UID (default: packaging@bongbetic.com)
|
||||
#
|
||||
# Spec: release-packaging.md §5
|
||||
|
||||
# ── Defaults ─────────────────────────────────────────────────────────────
|
||||
|
||||
DRY_RUN=false
|
||||
PUBLISH=false
|
||||
GITEA_URL="https://git.bongbetic.com"
|
||||
GITEA_OWNER="xavierk"
|
||||
GITEA_REPO="Fenris"
|
||||
PACKAGING_KEY="${PACKAGING_KEY:-packaging@bongbetic.com}"
|
||||
|
||||
CODENAMES=(bookworm jammy noble)
|
||||
RPM_GROUP="fenris"
|
||||
|
||||
# ── Parse arguments ──────────────────────────────────────────────────────
|
||||
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
--dry-run) DRY_RUN=true ;;
|
||||
--publish) PUBLISH=true ;;
|
||||
--help|-h)
|
||||
echo "Usage: $0 [--dry-run | --publish]"
|
||||
echo ""
|
||||
echo "Modes:"
|
||||
echo " --dry-run Print commands without executing (default)"
|
||||
echo " --publish Execute the full release flow"
|
||||
echo ""
|
||||
echo "Environment:"
|
||||
echo " GITEA_TOKEN API token for Gitea registry and release API"
|
||||
echo " PACKAGING_KEY GPG key UID (default: packaging@bongbetic.com)"
|
||||
exit 0
|
||||
;;
|
||||
*)
|
||||
echo "Unknown argument: $arg" >&2
|
||||
echo "Usage: $0 [--dry-run | --publish]" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if ! $DRY_RUN && ! $PUBLISH; then
|
||||
DRY_RUN=true
|
||||
fi
|
||||
|
||||
# ── Helpers ──────────────────────────────────────────────────────────────
|
||||
|
||||
_version() {
|
||||
sed -n 's/^version = "\(.*\)"/\1/p' pyproject.toml
|
||||
}
|
||||
|
||||
_deb_name() {
|
||||
local ver="$1"
|
||||
echo "fenris_${ver}_amd64.deb"
|
||||
}
|
||||
|
||||
_rpm_name() {
|
||||
local ver="$1" rel="$2"
|
||||
echo "fenris-${ver}-${rel}.x86_64.rpm"
|
||||
}
|
||||
|
||||
_run() {
|
||||
if $DRY_RUN; then
|
||||
echo " $*"
|
||||
else
|
||||
eval "$@"
|
||||
fi
|
||||
}
|
||||
|
||||
# ── Main ─────────────────────────────────────────────────────────────────
|
||||
|
||||
VERSION=$(_version)
|
||||
REVISION=1
|
||||
DEB=$(_deb_name "$VERSION")
|
||||
RPM=$(_rpm_name "$VERSION" "$REVISION")
|
||||
|
||||
echo "=== Fenris Release v${VERSION} ==="
|
||||
echo ""
|
||||
|
||||
if $DRY_RUN; then
|
||||
echo "[dry-run] Commands below will be executed in --publish mode."
|
||||
echo ""
|
||||
fi
|
||||
|
||||
# ── Step 1: Build both formats ──────────────────────────────────────────
|
||||
|
||||
echo "--- Build packages ---"
|
||||
_run "make package"
|
||||
echo ""
|
||||
|
||||
# ── Step 2: Sign RPM payload ────────────────────────────────────────────
|
||||
|
||||
echo "--- Sign RPM payload ---"
|
||||
_run "rpmsign --addsign --define '_gpg_name ${PACKAGING_KEY}' dist/${RPM}"
|
||||
echo ""
|
||||
|
||||
# ── Step 3: Generate and clearsign SHA256SUMS ────────────────────────────
|
||||
|
||||
echo "--- Generate SHA256SUMS ---"
|
||||
_run "cd dist && sha256sum ${DEB} ${RPM} > SHA256SUMS"
|
||||
echo ""
|
||||
|
||||
echo "--- Clearsign SHA256SUMS ---"
|
||||
_run "gpg --batch --yes --clearsign --local-user ${PACKAGING_KEY} dist/SHA256SUMS"
|
||||
echo ""
|
||||
|
||||
# ── Step 4: Upload to Gitea package registry ─────────────────────────────
|
||||
|
||||
echo "--- Upload packages to registry ---"
|
||||
for codename in "${CODENAMES[@]}"; do
|
||||
_run "curl --fail -X PUT -u ${GITEA_OWNER}:\$GITEA_TOKEN -T dist/${DEB} '${GITEA_URL}/api/packages/${GITEA_OWNER}/debian/pool/${codename}/main/upload'"
|
||||
done
|
||||
_run "curl --fail -X PUT -u ${GITEA_OWNER}:\$GITEA_TOKEN -T dist/${RPM} '${GITEA_URL}/api/packages/${GITEA_OWNER}/rpm/${RPM_GROUP}/upload'"
|
||||
echo ""
|
||||
|
||||
# ── Step 5: Create Gitea release with notes ─────────────────────────────
|
||||
|
||||
echo "--- Create Gitea release ---"
|
||||
_release_notes="Release v${VERSION}
|
||||
|
||||
## Packages
|
||||
|
||||
Install via apt (Debian/Ubuntu):
|
||||
|
||||
\`\`\`bash
|
||||
curl --fail -fsSL https://git.bongbetic.com/${GITEA_OWNER}/${GITEA_REPO}/raw/branch/main/packaging/keys/fenris-packaging.asc | sudo gpg --dearmor -o /etc/apt/keyrings/fenris.asc
|
||||
echo \"deb [signed-by=/etc/apt/keyrings/fenris.asc] https://git.bongbetic.com/api/packages/${GITEA_OWNER}/debian bookworm main\" | sudo tee /etc/apt/sources.list.d/fenris.list
|
||||
sudo apt update && sudo apt install fenris
|
||||
\`\`\`
|
||||
|
||||
Install via dnf (Fedora):
|
||||
|
||||
\`\`\`bash
|
||||
sudo dnf config-manager --add-repo https://git.bongbetic.com/${GITEA_OWNER}/${GITEA_REPO}/raw/branch/main/packaging/fenris.repo
|
||||
sudo dnf install fenris
|
||||
\`\`\`
|
||||
|
||||
## Verification
|
||||
|
||||
\`\`\`bash
|
||||
rpm -Kv fenris-${VERSION}-1.x86_64.rpm
|
||||
gpg --verify SHA256SUMS.asc SHA256SUMS
|
||||
\`\`\`
|
||||
|
||||
## Artifacts
|
||||
|
||||
- \`dist/${DEB}\` (Debian/Ubuntu)
|
||||
- \`dist/${RPM}\` (Fedora)
|
||||
- \`dist/SHA256SUMS.asc\` (clearsigned checksums)
|
||||
|
||||
See [docs/install/signing-key-ceremony.md](docs/install/signing-key-ceremony.md) for key ceremony details.
|
||||
See [docs/install/migrate-from-makeinstall.md](docs/install/migrate-from-makeinstall.md) for migration from make install."
|
||||
|
||||
if $DRY_RUN; then
|
||||
_run "curl --fail -X POST -u ${GITEA_OWNER}:\$GITEA_TOKEN -H 'Content-Type: application/json' -d '{\"tag_name\":\"v${VERSION}\",\"name\":\"v${VERSION}\",\"body\":\"...\"}' '${GITEA_URL}/api/v1/repos/${GITEA_OWNER}/${GITEA_REPO}/releases'"
|
||||
else
|
||||
# Create release via Gitea API (creates the tag atomically — no bare tag)
|
||||
RELEASE_RESPONSE=$(curl --fail -s -X POST \
|
||||
-u "${GITEA_OWNER}:${GITEA_TOKEN}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "$(jq -n \
|
||||
--arg tag "v${VERSION}" \
|
||||
--arg name "v${VERSION}" \
|
||||
--arg body "$_release_notes" \
|
||||
'{tag_name: $tag, name: $name, body: $body}')" \
|
||||
"${GITEA_URL}/api/v1/repos/${GITEA_OWNER}/${GITEA_REPO}/releases")
|
||||
|
||||
RELEASE_ID=$(echo "$RELEASE_RESPONSE" | jq -r '.id')
|
||||
echo " Release created: ${GITEA_URL}/${GITEA_OWNER}/${GITEA_REPO}/releases/tag/v${VERSION}"
|
||||
fi
|
||||
echo ""
|
||||
|
||||
# ── Step 6: Attach artifacts to release ──────────────────────────────────
|
||||
|
||||
echo "--- Attach artifacts to release ---"
|
||||
for artifact in "dist/${DEB}" "dist/${RPM}" "dist/SHA256SUMS.asc"; do
|
||||
_run "curl --fail -X POST -u ${GITEA_OWNER}:\$GITEA_TOKEN -F 'attachment=@${artifact}' '${GITEA_URL}/api/v1/repos/${GITEA_OWNER}/${GITEA_REPO}/releases/${RELEASE_ID:-0}/assets'"
|
||||
done
|
||||
echo ""
|
||||
|
||||
# ── Done ─────────────────────────────────────────────────────────────────
|
||||
|
||||
echo "=== Release v${VERSION} complete ==="
|
||||
echo ""
|
||||
echo "Summary:"
|
||||
echo " Packages: ${DEB}, ${RPM}"
|
||||
echo " Checksums: dist/SHA256SUMS.asc"
|
||||
echo " Registry: deb → bookworm, jammy, noble; rpm → ${RPM_GROUP}"
|
||||
echo " Release: ${GITEA_URL}/${GITEA_OWNER}/${GITEA_REPO}/releases/tag/v${VERSION}"
|
||||
echo ""
|
||||
echo "Key ceremony: delete the private key after release."
|
||||
echo " See docs/install/signing-key-ceremony.md"
|
||||
@@ -1,70 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Describe the Gitea request that creates or resynchronizes a release."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
import re
|
||||
import sys
|
||||
from typing import Any
|
||||
|
||||
|
||||
_SEMVER = r"(?:0|[1-9]\d*)\.(?:0|[1-9]\d*)\.(?:0|[1-9]\d*)"
|
||||
|
||||
|
||||
class ReleaseRequestError(ValueError):
|
||||
"""A release request could not be prepared safely."""
|
||||
|
||||
|
||||
def build_release_request(
|
||||
version: str, body: str, existing_release: dict[str, Any] | None
|
||||
) -> dict[str, Any]:
|
||||
"""Return the observable POST or PATCH request for a Gitea release."""
|
||||
if not re.fullmatch(_SEMVER, version):
|
||||
raise ReleaseRequestError(f"version is not bare semver: {version!r}")
|
||||
if existing_release is None:
|
||||
return {
|
||||
"method": "POST",
|
||||
"path": "/releases",
|
||||
"payload": {"tag_name": f"v{version}", "name": f"v{version}", "body": body},
|
||||
}
|
||||
|
||||
release_id = existing_release.get("id")
|
||||
if not isinstance(release_id, int):
|
||||
raise ReleaseRequestError("existing release does not contain an integer id")
|
||||
return {
|
||||
"method": "PATCH",
|
||||
"path": f"/releases/{release_id}",
|
||||
"payload": {"body": body},
|
||||
}
|
||||
|
||||
|
||||
def _read_json(path: Path) -> dict[str, Any]:
|
||||
try:
|
||||
value = json.loads(path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError) as error:
|
||||
raise ReleaseRequestError(f"cannot read existing release {path}: {error}") from error
|
||||
if not isinstance(value, dict):
|
||||
raise ReleaseRequestError("existing release must be a JSON object")
|
||||
return value
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--version", required=True)
|
||||
parser.add_argument("--body-file", type=Path, required=True)
|
||||
parser.add_argument("--existing-release", type=Path)
|
||||
args = parser.parse_args(argv)
|
||||
try:
|
||||
body = args.body_file.read_text(encoding="utf-8")
|
||||
existing = _read_json(args.existing_release) if args.existing_release else None
|
||||
print(json.dumps(build_release_request(args.version, body, existing)))
|
||||
except (OSError, ReleaseRequestError) as error:
|
||||
print(f"::error::{error}", file=sys.stderr)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,2 +0,0 @@
|
||||
"""Fenris: NVMe wear monitor with persistent TUI."""
|
||||
__version__ = "0.3.1"
|
||||
@@ -1,137 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""fenris-collect: device interrogation and store writes.
|
||||
|
||||
This is the root oneshot unit's ExecStart. It reads the device selector
|
||||
from /etc/fenris/fenris.conf, interrogates the drive via smartctl and sysfs,
|
||||
and writes the sample to the observation store.
|
||||
|
||||
Spec: §8.4, §8.5
|
||||
|
||||
When run as a script, uses the fenris package from the installed wheel.
|
||||
"""
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
# Package dependencies are vendored independently of the host Python minor
|
||||
# version. Keep the venv fallback for the legacy development install.
|
||||
VENV_DIR = Path("/opt/fenris")
|
||||
if VENV_DIR.exists():
|
||||
vendor_dir = VENV_DIR / "vendor"
|
||||
if vendor_dir.is_dir():
|
||||
sys.path.insert(0, str(vendor_dir))
|
||||
else:
|
||||
site_packages = next((VENV_DIR / "lib").glob("python*/site-packages"), None)
|
||||
if site_packages:
|
||||
sys.path.insert(0, str(site_packages))
|
||||
|
||||
from fenris.store import init_store, get_store_path
|
||||
from fenris.collector import run_collection
|
||||
|
||||
|
||||
CONFIG_PATH = Path("/etc/fenris/fenris.conf")
|
||||
|
||||
|
||||
def load_config() -> dict:
|
||||
"""Load configuration from /etc/fenris/fenris.conf.
|
||||
|
||||
The file holds exactly one key: the device selector.
|
||||
Spec §8.3: re-read every run; no reload path.
|
||||
"""
|
||||
if not CONFIG_PATH.exists():
|
||||
raise RuntimeError(f"Configuration file not found: {CONFIG_PATH}")
|
||||
|
||||
config = {}
|
||||
try:
|
||||
with open(CONFIG_PATH, "r") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
if "=" in line:
|
||||
key, value = line.split("=", 1)
|
||||
config[key.strip()] = value.strip()
|
||||
except Exception as e:
|
||||
raise RuntimeError(f"Failed to read configuration: {e}")
|
||||
|
||||
if "device" not in config:
|
||||
raise RuntimeError("Configuration error: missing 'device' key")
|
||||
|
||||
return config
|
||||
|
||||
|
||||
def interrogate_drive(device: str) -> dict:
|
||||
"""Interrogate the drive via smartctl.
|
||||
|
||||
Returns the smartctl JSON output.
|
||||
Raises RuntimeError on failure.
|
||||
"""
|
||||
result = subprocess.run(
|
||||
["smartctl", "-a", "-j", device],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
|
||||
if result.returncode != 0:
|
||||
raise RuntimeError(
|
||||
f"smartctl failed for {device}: {result.stderr}"
|
||||
)
|
||||
|
||||
try:
|
||||
return json.loads(result.stdout)
|
||||
except json.JSONDecodeError as e:
|
||||
raise RuntimeError(f"Failed to parse smartctl output: {e}")
|
||||
|
||||
|
||||
def find_nvme_sysfs() -> Path | None:
|
||||
"""Find the NVMe controller sysfs path."""
|
||||
nvme_ctrl = Path("/sys/class/nvme")
|
||||
if not nvme_ctrl.exists():
|
||||
return None
|
||||
|
||||
for ctrl in sorted(nvme_ctrl.iterdir()):
|
||||
if ctrl.name.startswith("nvme"):
|
||||
return ctrl
|
||||
return None
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Run one collection cycle."""
|
||||
try:
|
||||
config = load_config()
|
||||
device = config["device"]
|
||||
|
||||
# Interrogate the drive
|
||||
smartctl_data = interrogate_drive(device)
|
||||
|
||||
# Find sysfs path
|
||||
sysfs_path = find_nvme_sysfs()
|
||||
if sysfs_path is None:
|
||||
raise RuntimeError("No NVMe controller found in sysfs")
|
||||
|
||||
# Inject a simple clock
|
||||
class SimpleClock:
|
||||
def utcnow(self):
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
clock = SimpleClock()
|
||||
|
||||
# Run collection
|
||||
result = run_collection(smartctl_data, sysfs_path, config, clock)
|
||||
|
||||
if result["ok"]:
|
||||
print(f"Collection successful: {result['sample_count']} sample(s)")
|
||||
sys.exit(0)
|
||||
else:
|
||||
print(f"Collection failed: {result['error']}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Collection error: {e}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,325 +0,0 @@
|
||||
"""Collector: acquires counters and identity, writes to observation store.
|
||||
|
||||
This module implements the thinnest complete write path:
|
||||
- Acquire counters and thermal evidence from smartctl -a -j
|
||||
- Acquire controller identity from sysfs
|
||||
- Normalize identity exactly once at write time
|
||||
- Validate every row against store invariants
|
||||
- Commit one well-formed sample
|
||||
|
||||
No code path outside the collector interrogates the device.
|
||||
"""
|
||||
import json
|
||||
import sqlite3
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
from .store import init_store, get_store_path
|
||||
|
||||
|
||||
class AcquisitionError(Exception):
|
||||
"""Raised when acquisition fails - whole run is refused."""
|
||||
pass
|
||||
|
||||
|
||||
class InvariantViolationError(Exception):
|
||||
"""Raised when a row would violate store invariants - writes nothing."""
|
||||
pass
|
||||
|
||||
|
||||
def acquire_from_smartctl(smartctl_data: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Acquire counters and thermal evidence from smartctl -a -j data.
|
||||
|
||||
Validates that all required fields are present.
|
||||
Raises AcquisitionError on any failure.
|
||||
"""
|
||||
required_fields = [
|
||||
"nvme_smart_health_information_log",
|
||||
"user_capacity",
|
||||
"model_name",
|
||||
"serial_number",
|
||||
"firmware_version",
|
||||
]
|
||||
|
||||
for field in required_fields:
|
||||
if field not in smartctl_data:
|
||||
raise AcquisitionError(f"Missing required field in smartctl data: {field}")
|
||||
|
||||
log = smartctl_data["nvme_smart_health_information_log"]
|
||||
required_log_fields = [
|
||||
"data_units_written",
|
||||
"data_units_read",
|
||||
"percentage_used",
|
||||
"power_on_hours",
|
||||
"temperature",
|
||||
]
|
||||
|
||||
for field in required_log_fields:
|
||||
if field not in log:
|
||||
raise AcquisitionError(f"Missing required field in SMART log: {field}")
|
||||
|
||||
return {
|
||||
"model": smartctl_data["model_name"],
|
||||
"serial": smartctl_data["serial_number"],
|
||||
"firmware_rev": smartctl_data["firmware_version"],
|
||||
"capacity_bytes": smartctl_data["user_capacity"]["bytes"],
|
||||
"percentage_used": log["percentage_used"],
|
||||
"available_spare": log.get("available_spare"),
|
||||
"media_errors": log.get("media_errors", 0),
|
||||
"power_on_hours": log["power_on_hours"],
|
||||
"power_cycles": log.get("power_cycles"),
|
||||
"unsafe_shutdowns": log.get("unsafe_shutdowns"),
|
||||
"temperature_c": log["temperature"],
|
||||
"data_units_written": log["data_units_written"],
|
||||
"data_units_read": log["data_units_read"],
|
||||
"bytes_written": log["data_units_written"] * 512000,
|
||||
"bytes_read": log["data_units_read"] * 512000,
|
||||
"critical_warning": log.get("critical_warning", 0),
|
||||
}
|
||||
|
||||
|
||||
def acquire_from_sysfs(sysfs_path: Path) -> Dict[str, Any]:
|
||||
"""Acquire controller identity from sysfs.
|
||||
|
||||
Reads identity from:
|
||||
- /sys/class/nvme/<ctrl>/subsysnqn (primary)
|
||||
- /sys/class/nvme/<ctrl>/model
|
||||
- /sys/class/nvme/<ctrl>/serial
|
||||
- /sys/class/nvme/<ctrl>/firmware_rev
|
||||
- /sys/class/nvme/<ctrl>/transport/ (optional)
|
||||
|
||||
Raises AcquisitionError on any failure.
|
||||
"""
|
||||
identity_files = {
|
||||
"subnqn": "subsysnqn",
|
||||
"mn": "model",
|
||||
"sn": "serial",
|
||||
"fr": "firmware_rev",
|
||||
}
|
||||
|
||||
identity = {}
|
||||
for key, filename in identity_files.items():
|
||||
filepath = sysfs_path / filename
|
||||
if not filepath.exists():
|
||||
raise AcquisitionError(f"Missing sysfs file: {filepath}")
|
||||
|
||||
try:
|
||||
value = filepath.read_text().strip()
|
||||
identity[key] = value if value else ""
|
||||
except Exception as e:
|
||||
raise AcquisitionError(f"Failed to read {filepath}: {e}")
|
||||
|
||||
# Transport info (optional)
|
||||
transport_dir = sysfs_path / "transport"
|
||||
if transport_dir.exists():
|
||||
try:
|
||||
transport_file = transport_dir / "trstring"
|
||||
if transport_file.exists():
|
||||
identity["transport"] = transport_file.read_text().strip()
|
||||
else:
|
||||
identity["transport"] = None
|
||||
except Exception:
|
||||
identity["transport"] = None
|
||||
else:
|
||||
identity["transport"] = None
|
||||
|
||||
# vid/ssvid from PCI node (optional, metadata only - never key components)
|
||||
# PCI device directory is the sysfs_path itself (the controller dir is a symlink to PCI)
|
||||
pci_device = sysfs_path
|
||||
for attr, key in [("vendor", "vid"), ("subsystem_vendor", "ssvid")]:
|
||||
filepath = pci_device / attr
|
||||
if filepath.exists():
|
||||
try:
|
||||
value = filepath.read_text().strip()
|
||||
identity[key] = value if value else None
|
||||
except Exception:
|
||||
identity[key] = None
|
||||
else:
|
||||
identity[key] = None
|
||||
|
||||
return identity
|
||||
|
||||
|
||||
def normalize_identity(identity: Dict[str, Any]) -> str:
|
||||
"""Normalize identity exactly once at write time.
|
||||
|
||||
Rules:
|
||||
- Strip trailing spaces and newlines
|
||||
- No case folding
|
||||
- Empty-after-strip stored blank
|
||||
|
||||
Returns normalized identity key.
|
||||
"""
|
||||
# Primary key: normalized kernel-exposed subsystem NQN
|
||||
key = identity.get("subnqn", "")
|
||||
if key:
|
||||
key = key.rstrip()
|
||||
return key
|
||||
|
||||
# Fallback 1: kernel composite (not implemented yet)
|
||||
# Fallback 2: model|serial
|
||||
mn = identity.get("mn", "").rstrip()
|
||||
sn = identity.get("sn", "").rstrip()
|
||||
if mn or sn:
|
||||
return f"{mn}|{sn}"
|
||||
|
||||
# All keys blank - degraded identity
|
||||
return ""
|
||||
|
||||
|
||||
def compute_identity_degraded(identity: Dict[str, Any]) -> bool:
|
||||
"""Check if identity is degraded (all key rungs empty)."""
|
||||
key = normalize_identity(identity)
|
||||
return key == ""
|
||||
|
||||
|
||||
def validate_sample_invariants(sample: Dict[str, Any], conn: sqlite3.Connection) -> None:
|
||||
"""Validate sample against store invariants.
|
||||
|
||||
Raises InvariantViolationError if any invariant is violated.
|
||||
"""
|
||||
# TODO: Implement more complex invariants as needed
|
||||
# For now, just check basic constraints
|
||||
|
||||
if sample.get("bytes_written", 0) < 0:
|
||||
raise InvariantViolationError("Negative bytes_written")
|
||||
|
||||
if sample.get("bytes_read", 0) < 0:
|
||||
raise InvariantViolationError("Negative bytes_read")
|
||||
|
||||
|
||||
def write_sample(
|
||||
sample: Dict[str, Any],
|
||||
identity: Dict[str, Any],
|
||||
conn: sqlite3.Connection,
|
||||
clock,
|
||||
) -> Dict[str, Any]:
|
||||
"""Write one sample to the observation store.
|
||||
|
||||
Identity normalization happens exactly once here.
|
||||
Returns segment info for the caller.
|
||||
"""
|
||||
from .segment import find_current_segment, should_open_new_segment, open_segment
|
||||
|
||||
# Normalize identity exactly once at write time
|
||||
identity_key = normalize_identity(identity)
|
||||
identity_degraded = compute_identity_degraded(identity)
|
||||
|
||||
# Find current segment
|
||||
current_segment = find_current_segment(conn)
|
||||
|
||||
# Determine if we need a new segment
|
||||
should_open, reason = should_open_new_segment(
|
||||
current_segment, identity_key, sample["bytes_written"], conn
|
||||
)
|
||||
|
||||
# Open new segment if needed
|
||||
segment_opened = False
|
||||
if should_open:
|
||||
now = clock.utcnow()
|
||||
open_segment(conn, now, identity, identity_key, identity_degraded)
|
||||
segment_opened = True
|
||||
|
||||
# Insert sample
|
||||
cursor = conn.execute(
|
||||
"""
|
||||
INSERT INTO samples (
|
||||
ts, device, subnqn, sn, mn, fr, capacity_bytes,
|
||||
percentage_used, available_spare, media_errors, power_on_hours,
|
||||
power_cycles, unsafe_shutdowns, temperature_c,
|
||||
data_units_written, data_units_read, bytes_written, bytes_read,
|
||||
critical_warning
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
sample["ts"],
|
||||
sample["device"],
|
||||
identity.get("subnqn", ""),
|
||||
identity.get("sn", ""),
|
||||
identity.get("mn", ""),
|
||||
identity.get("fr", ""),
|
||||
sample["capacity_bytes"],
|
||||
sample["percentage_used"],
|
||||
sample["available_spare"],
|
||||
sample["media_errors"],
|
||||
sample["power_on_hours"],
|
||||
sample["power_cycles"],
|
||||
sample["unsafe_shutdowns"],
|
||||
sample["temperature_c"],
|
||||
sample["data_units_written"],
|
||||
sample["data_units_read"],
|
||||
sample["bytes_written"],
|
||||
sample["bytes_read"],
|
||||
sample["critical_warning"],
|
||||
),
|
||||
)
|
||||
|
||||
conn.commit()
|
||||
|
||||
return {
|
||||
"segment_opened": segment_opened,
|
||||
"segment_reason": reason,
|
||||
"identity_key": identity_key,
|
||||
"identity_degraded": identity_degraded,
|
||||
}
|
||||
|
||||
|
||||
def run_collection(
|
||||
smartctl_data: Dict[str, Any],
|
||||
sysfs_path: Path,
|
||||
config: Dict[str, Any],
|
||||
clock,
|
||||
) -> Dict[str, Any]:
|
||||
"""Run one collection run.
|
||||
|
||||
This is the main entry point for the collector.
|
||||
Returns the run outcome.
|
||||
"""
|
||||
try:
|
||||
# Acquire counters and thermal evidence
|
||||
counters = acquire_from_smartctl(smartctl_data)
|
||||
|
||||
# Acquire controller identity
|
||||
identity = acquire_from_sysfs(sysfs_path)
|
||||
|
||||
# Build sample with injected clock
|
||||
sample = {
|
||||
"ts": clock.utcnow().isoformat(),
|
||||
"device": config["device"],
|
||||
**counters,
|
||||
**identity,
|
||||
}
|
||||
|
||||
# Initialize store if needed
|
||||
store_path = get_store_path(config)
|
||||
conn = init_store(store_path)
|
||||
|
||||
# Run legacy import if needed (idempotent)
|
||||
from .legacy import import_legacy_history
|
||||
history_path = Path(config.get("data_dir", ".")) / "history.jsonl"
|
||||
if history_path.exists():
|
||||
import_legacy_history(conn, history_path, clock=clock)
|
||||
|
||||
|
||||
try:
|
||||
# Validate invariants
|
||||
validate_sample_invariants(sample, conn)
|
||||
|
||||
# Write sample
|
||||
write_sample(sample, identity, conn, clock)
|
||||
|
||||
return {
|
||||
"ok": True,
|
||||
"sample_count": 1,
|
||||
"store_path": str(store_path),
|
||||
}
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
except (AcquisitionError, InvariantViolationError) as e:
|
||||
return {
|
||||
"ok": False,
|
||||
"error": str(e),
|
||||
"error_type": type(e).__name__,
|
||||
}
|
||||
@@ -1,165 +0,0 @@
|
||||
"""Day aggregate derivation per spec §5.4, §3.3.
|
||||
|
||||
One row per UTC day, derived monotonically from hour rows — the grain at
|
||||
which usage-habit evidence is judged. No absent hour is ever interpolated,
|
||||
estimated, or fabricated (§5.3, FL-3).
|
||||
|
||||
Coverage: the share of wall-clock seconds inside monitoring periods whose
|
||||
usage-habit classification is known rather than unknown (§5.3).
|
||||
"""
|
||||
import sqlite3
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DayAggregate:
|
||||
"""One UTC day's aggregated stats."""
|
||||
day: str # ISO 8601 UTC date, e.g. "2026-09-01"
|
||||
seconds_active: int
|
||||
seconds_idle: int
|
||||
seconds_powered_off: int
|
||||
seconds_unknown: int
|
||||
bytes_written_delta: int
|
||||
bytes_read_delta: int
|
||||
sample_count: int
|
||||
coverage: float
|
||||
|
||||
|
||||
def _period_wall_clock_for_day(conn: sqlite3.Connection, day: str) -> int:
|
||||
"""Total wall-clock seconds inside monitoring periods for a UTC day.
|
||||
|
||||
Clamps each period to the day boundary [dayT00:00, dayT24:00).
|
||||
"""
|
||||
day_start = datetime.fromisoformat(f"{day}T00:00:00+00:00")
|
||||
day_end = day_start + timedelta(days=1)
|
||||
day_start_str = day_start.isoformat()
|
||||
day_end_str = day_end.isoformat()
|
||||
|
||||
cursor = conn.execute(
|
||||
"SELECT started_at, ended_at FROM monitoring_periods "
|
||||
"WHERE (ended_at IS NULL OR ended_at > ?) AND started_at < ? "
|
||||
"ORDER BY started_at",
|
||||
(day_start_str, day_end_str),
|
||||
)
|
||||
|
||||
total = 0
|
||||
for row in cursor.fetchall():
|
||||
period_start = row[0]
|
||||
period_end = row[1]
|
||||
|
||||
effective_start = max(period_start, day_start_str)
|
||||
if period_end is not None:
|
||||
effective_end = min(period_end, day_end_str)
|
||||
else:
|
||||
effective_end = day_end_str
|
||||
|
||||
if effective_start < effective_end:
|
||||
s = datetime.fromisoformat(effective_start)
|
||||
e = datetime.fromisoformat(effective_end)
|
||||
total += int((e - s).total_seconds())
|
||||
|
||||
return total
|
||||
|
||||
|
||||
def _hour_overlaps_period(conn: sqlite3.Connection, hour_iso: str) -> bool:
|
||||
"""Check if an hour's wall-clock span overlaps any monitoring period."""
|
||||
hour_start = datetime.fromisoformat(hour_iso)
|
||||
hour_end = hour_start + timedelta(hours=1)
|
||||
hs = hour_start.isoformat()
|
||||
he = hour_end.isoformat()
|
||||
|
||||
cursor = conn.execute(
|
||||
"SELECT 1 FROM monitoring_periods "
|
||||
"WHERE started_at < ? AND (ended_at IS NULL OR ended_at > ?) "
|
||||
"LIMIT 1",
|
||||
(he, hs),
|
||||
)
|
||||
return cursor.fetchone() is not None
|
||||
|
||||
|
||||
def derive_day(conn: sqlite3.Connection, day: str) -> DayAggregate | None:
|
||||
"""Derive a single day aggregate from its hour rows + monitoring periods.
|
||||
|
||||
Only hours overlapping a monitoring period contribute to the aggregate.
|
||||
Gap hours inside periods contribute unknown seconds. Hours outside all
|
||||
monitoring periods are excluded entirely (§5.2).
|
||||
|
||||
Returns None if no hours exist for the day.
|
||||
"""
|
||||
cursor = conn.execute(
|
||||
"SELECT hour, active_seconds, idle_seconds, powered_off_seconds, unknown_seconds, "
|
||||
" bytes_written_delta, bytes_read_delta, sample_count "
|
||||
"FROM hour_observations "
|
||||
"WHERE hour LIKE ? "
|
||||
"ORDER BY hour",
|
||||
(day + "T%",),
|
||||
)
|
||||
rows = cursor.fetchall()
|
||||
if not rows:
|
||||
return None
|
||||
|
||||
total_active = 0
|
||||
total_idle = 0
|
||||
total_powered_off = 0
|
||||
total_unknown_from_hours = 0
|
||||
total_bw = 0
|
||||
total_br = 0
|
||||
total_samples = 0
|
||||
total_hour_wall_clock = 0
|
||||
|
||||
for row in rows:
|
||||
# Only count hours overlapping a monitoring period
|
||||
if not _hour_overlaps_period(conn, row[0]):
|
||||
continue
|
||||
|
||||
total_active += row[1]
|
||||
total_idle += row[2]
|
||||
total_powered_off += row[3]
|
||||
total_unknown_from_hours += row[4]
|
||||
total_bw += row[5]
|
||||
total_br += row[6]
|
||||
total_samples += row[7]
|
||||
total_hour_wall_clock += row[1] + row[2] + row[3] + row[4]
|
||||
|
||||
# Wall-clock seconds inside monitoring periods for this day
|
||||
period_wc = _period_wall_clock_for_day(conn, day)
|
||||
|
||||
# Gap seconds = period wall-clock - sum of existing hour wall-clock
|
||||
gap_seconds = max(0, period_wc - total_hour_wall_clock)
|
||||
|
||||
total_unknown = total_unknown_from_hours + gap_seconds
|
||||
|
||||
# Coverage: known seconds / period wall-clock (§5.2, §5.3)
|
||||
known_seconds = total_active + total_idle + total_powered_off
|
||||
coverage = known_seconds / period_wc if period_wc > 0 else 0.0
|
||||
|
||||
return DayAggregate(
|
||||
day=day,
|
||||
seconds_active=total_active,
|
||||
seconds_idle=total_idle,
|
||||
seconds_powered_off=total_powered_off,
|
||||
seconds_unknown=total_unknown,
|
||||
bytes_written_delta=total_bw,
|
||||
bytes_read_delta=total_br,
|
||||
sample_count=total_samples,
|
||||
coverage=coverage,
|
||||
)
|
||||
|
||||
|
||||
def derive_all_days(conn: sqlite3.Connection) -> list[DayAggregate]:
|
||||
"""Derive day aggregates for all days that have hour rows.
|
||||
|
||||
Returns days sorted by date.
|
||||
"""
|
||||
cursor = conn.execute(
|
||||
"SELECT DISTINCT substr(hour, 1, 10) as day FROM hour_observations ORDER BY day"
|
||||
)
|
||||
days = [row[0] for row in cursor.fetchall()]
|
||||
|
||||
results = []
|
||||
for day in days:
|
||||
agg = derive_day(conn, day)
|
||||
if agg is not None:
|
||||
results.append(agg)
|
||||
return results
|
||||
@@ -1,94 +0,0 @@
|
||||
"""Hour classification per spec §5.1.
|
||||
|
||||
Each UTC hour is classified by named constants, in this order of evidence:
|
||||
- Powered-off: power-on-hours delta < 90% of wall-clock span
|
||||
- Active: DUW delta >= 256 MiB in the hour
|
||||
- Idle: powered on, sampled, below active threshold
|
||||
- Unknown: everything else (unsampled without POH evidence)
|
||||
|
||||
Four splits sum to exactly wall_clock_seconds. Disabled time is never an
|
||||
hour state — it is wall-clock outside monitoring periods (§5.2).
|
||||
"""
|
||||
from dataclasses import dataclass
|
||||
|
||||
# Spec §5.1: Active hour threshold — 256 MiB DUW delta
|
||||
ACTIVE_THRESHOLD_BYTES = 256 * 1024 * 1024 # 256 MiB
|
||||
|
||||
# Spec §5.1: Powered-off threshold — 90% of wall-clock span
|
||||
POWERED_OFF_THRESHOLD_PERCENT = 0.90
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HourSplit:
|
||||
"""Usage-habit split for one UTC hour. Fields sum to wall_clock_seconds."""
|
||||
seconds_active: int
|
||||
seconds_idle: int
|
||||
seconds_powered_off: int
|
||||
seconds_unknown: int
|
||||
|
||||
|
||||
def classify_hour(
|
||||
wall_clock_seconds: int,
|
||||
poh_delta: int,
|
||||
duw_delta: int,
|
||||
dur_delta: int,
|
||||
sampled_seconds: int | None = None,
|
||||
) -> HourSplit:
|
||||
"""Classify a UTC hour into the four usage-habit states.
|
||||
|
||||
Args:
|
||||
wall_clock_seconds: Total seconds in this hour boundary (3600 for a
|
||||
full hour, less for partial-hours at period edges).
|
||||
poh_delta: Power-on-hours delta since previous sample (in seconds).
|
||||
duw_delta: Data-units-written delta since previous sample (in bytes).
|
||||
dur_delta: Data-units-read delta since previous sample (in bytes).
|
||||
sampled_seconds: Seconds within this hour covered by a sample.
|
||||
None or 0 means no sample fell in this hour.
|
||||
|
||||
Returns:
|
||||
HourSplit whose four fields sum to wall_clock_seconds.
|
||||
"""
|
||||
if sampled_seconds is None:
|
||||
sampled_seconds = 0
|
||||
|
||||
# Clamp sampled_seconds to wall_clock_seconds
|
||||
sampled_seconds = min(sampled_seconds, wall_clock_seconds)
|
||||
|
||||
# --- Decision order per spec §5.1 ---
|
||||
|
||||
# 1. Powered-off: POH delta < 90% of wall-clock span
|
||||
powered_off_threshold = wall_clock_seconds * POWERED_OFF_THRESHOLD_PERCENT
|
||||
if poh_delta < powered_off_threshold:
|
||||
return HourSplit(
|
||||
seconds_active=0,
|
||||
seconds_idle=0,
|
||||
seconds_powered_off=wall_clock_seconds,
|
||||
seconds_unknown=0,
|
||||
)
|
||||
|
||||
# 2. Active: DUW delta >= 256 MiB
|
||||
if duw_delta >= ACTIVE_THRESHOLD_BYTES:
|
||||
return HourSplit(
|
||||
seconds_active=wall_clock_seconds,
|
||||
seconds_idle=0,
|
||||
seconds_powered_off=0,
|
||||
seconds_unknown=0,
|
||||
)
|
||||
|
||||
# 3. Idle: powered on, sampled, below active threshold
|
||||
# Unsampled portion within the hour is unknown
|
||||
if sampled_seconds > 0:
|
||||
return HourSplit(
|
||||
seconds_active=0,
|
||||
seconds_idle=sampled_seconds,
|
||||
seconds_powered_off=0,
|
||||
seconds_unknown=wall_clock_seconds - sampled_seconds,
|
||||
)
|
||||
|
||||
# 4. Unknown: unsampled without POH evidence
|
||||
return HourSplit(
|
||||
seconds_active=0,
|
||||
seconds_idle=0,
|
||||
seconds_powered_off=0,
|
||||
seconds_unknown=wall_clock_seconds,
|
||||
)
|
||||
@@ -1,440 +0,0 @@
|
||||
"""Legacy migration: import history.jsonl into the observation store.
|
||||
|
||||
Spec §3.5, ADR 0001 §6. Idempotent and interruption-safe.
|
||||
|
||||
Entry points:
|
||||
- Installer import detection at ./data/history.jsonl (§10.1)
|
||||
- fenris import <path> (§8.8)
|
||||
- Collector's first new-version run (ADR 0001 §6)
|
||||
|
||||
The import is a single transaction — a scripted kill mid-import leaves the
|
||||
store fully pre- or fully post-migration. history.jsonl is the sole authority;
|
||||
hourly.jsonl is diffed and logged but never trusted. Malformed lines are
|
||||
quarantined with a logged count, never silently dropped. Legacy files are
|
||||
renamed *.migrated only after commit and never deleted.
|
||||
"""
|
||||
import json
|
||||
import logging
|
||||
import sqlite3
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from .hour_classify import classify_hour
|
||||
from .segment import open_segment, normalize_identity
|
||||
from .monitoring_periods import close_period
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Legacy import marker — stored in a metadata table
|
||||
LEGACY_IMPORT_MARKER = "legacy_imported"
|
||||
|
||||
|
||||
def _ensure_metadata_table(conn: sqlite3.Connection) -> None:
|
||||
"""Create the metadata table if it doesn't exist."""
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS store_metadata (
|
||||
key TEXT PRIMARY KEY,
|
||||
value TEXT NOT NULL
|
||||
)
|
||||
""")
|
||||
|
||||
|
||||
def is_legacy_imported(conn: sqlite3.Connection) -> bool:
|
||||
"""Check if legacy history has already been imported.
|
||||
|
||||
Spec §3.5.1: If the store already carries the legacy-import marker, do nothing.
|
||||
"""
|
||||
_ensure_metadata_table(conn)
|
||||
cursor = conn.execute(
|
||||
"SELECT value FROM store_metadata WHERE key = ?",
|
||||
(LEGACY_IMPORT_MARKER,),
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
return row is not None and row[0] == "true"
|
||||
|
||||
|
||||
def _parse_history_line(line: str, line_num: int) -> Optional[Dict[str, Any]]:
|
||||
"""Parse a single line from history.jsonl.
|
||||
|
||||
Returns None for malformed lines (quarantined, not silently dropped).
|
||||
"""
|
||||
line = line.strip()
|
||||
if not line:
|
||||
return None
|
||||
|
||||
try:
|
||||
record = json.loads(line)
|
||||
except json.JSONDecodeError as e:
|
||||
logger.warning("Malformed JSON at line %d: %s", line_num, e)
|
||||
return None
|
||||
|
||||
# Validate required fields
|
||||
required_fields = [
|
||||
"timestamp", "model", "serial", "firmware_version",
|
||||
"data_units_written", "data_units_read", "percentage_used",
|
||||
"power_on_hours", "temperature",
|
||||
]
|
||||
|
||||
for field in required_fields:
|
||||
if field not in record:
|
||||
logger.warning("Missing field '%s' at line %d", field, line_num)
|
||||
return None
|
||||
|
||||
return record
|
||||
|
||||
|
||||
def _record_to_sample(record: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Convert a legacy history.jsonl record to a sample dict."""
|
||||
# Legacy records use different field names
|
||||
duw = record["data_units_written"]
|
||||
dur = record["data_units_read"]
|
||||
|
||||
return {
|
||||
"ts": record["timestamp"],
|
||||
"device": record.get("device", "/dev/nvme0"),
|
||||
"subnqn": record.get("subsystem_nqn", ""),
|
||||
"sn": record["serial"],
|
||||
"mn": record["model"],
|
||||
"fr": record["firmware_version"],
|
||||
"capacity_bytes": record.get("capacity_bytes", 0),
|
||||
"percentage_used": record["percentage_used"],
|
||||
"available_spare": record.get("available_spare"),
|
||||
"media_errors": record.get("media_errors", 0),
|
||||
"power_on_hours": record["power_on_hours"],
|
||||
"power_cycles": record.get("power_cycles"),
|
||||
"unsafe_shutdowns": record.get("unsafe_shutdowns"),
|
||||
"temperature_c": record["temperature"],
|
||||
"data_units_written": duw,
|
||||
"data_units_read": dur,
|
||||
"bytes_written": duw * 512000,
|
||||
"bytes_read": dur * 512000,
|
||||
"critical_warning": record.get("critical_warning", 0),
|
||||
}
|
||||
|
||||
|
||||
def _derive_hour_observation(
|
||||
samples: List[Dict[str, Any]],
|
||||
hour_start: datetime,
|
||||
) -> Dict[str, Any]:
|
||||
"""Derive a single hour observation from samples in that hour.
|
||||
|
||||
Pre-migration hours carry an unknown activity split except directly
|
||||
evidenced facts — a sample present means powered on; a DUW delta means
|
||||
writes occurred (§3.5.4).
|
||||
"""
|
||||
hour_end = hour_start.replace(hour=hour_start.hour + 1) if hour_start.hour < 23 else hour_start.replace(hour=0, day=hour_start.day + 1)
|
||||
|
||||
# Filter samples in this hour
|
||||
hour_samples = []
|
||||
for s in samples:
|
||||
ts = datetime.fromisoformat(s["ts"])
|
||||
if hour_start <= ts < hour_end:
|
||||
hour_samples.append(s)
|
||||
|
||||
if not hour_samples:
|
||||
return None
|
||||
|
||||
# Sort by timestamp
|
||||
hour_samples.sort(key=lambda x: x["ts"])
|
||||
|
||||
# Compute deltas from first to last sample in the hour
|
||||
first = hour_samples[0]
|
||||
last = hour_samples[-1]
|
||||
|
||||
duw_delta = last["bytes_written"] - first["bytes_written"]
|
||||
dur_delta = last["bytes_read"] - first["bytes_read"]
|
||||
poh_delta = (last["power_on_hours"] - first["power_on_hours"]) * 3600
|
||||
|
||||
# Temperature stats
|
||||
temps = [s["temperature_c"] for s in hour_samples]
|
||||
|
||||
# Classify the hour
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=3600,
|
||||
poh_delta=poh_delta,
|
||||
duw_delta=duw_delta,
|
||||
dur_delta=dur_delta,
|
||||
sampled_seconds=3600, # Legacy samples cover the full hour
|
||||
)
|
||||
|
||||
return {
|
||||
"hour": hour_start.strftime("%Y-%m-%dT%H:00:00Z"),
|
||||
"active_seconds": split.seconds_active,
|
||||
"idle_seconds": split.seconds_idle,
|
||||
"powered_off_seconds": split.seconds_powered_off,
|
||||
"unknown_seconds": split.seconds_unknown,
|
||||
"bytes_written_delta": duw_delta,
|
||||
"bytes_read_delta": dur_delta,
|
||||
"temperature_min": min(temps),
|
||||
"temperature_avg": sum(temps) / len(temps),
|
||||
"temperature_max": max(temps),
|
||||
"sample_count": len(hour_samples),
|
||||
"coverage": 1.0 if split.seconds_unknown == 0 else (3600 - split.seconds_unknown) / 3600,
|
||||
}
|
||||
|
||||
|
||||
def _diff_hourly_jsonl(
|
||||
hourly_path: Path,
|
||||
derived_hours: Dict[str, Dict[str, Any]],
|
||||
) -> None:
|
||||
"""Diff hourly.jsonl against derived data and log mismatches.
|
||||
|
||||
Spec §3.5.3: hourly.jsonl is never trusted; mismatches are diffed and logged.
|
||||
"""
|
||||
if not hourly_path.exists():
|
||||
logger.info("No hourly.jsonl found for diffing")
|
||||
return
|
||||
|
||||
try:
|
||||
with open(hourly_path, "r") as f:
|
||||
for line_num, line in enumerate(f, 1):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
|
||||
try:
|
||||
record = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
logger.warning("Malformed hourly.jsonl at line %d", line_num)
|
||||
continue
|
||||
|
||||
hour_key = record.get("hour")
|
||||
if hour_key not in derived_hours:
|
||||
logger.info("Hourly.jsonl has hour %s not in derived data", hour_key)
|
||||
continue
|
||||
|
||||
derived = derived_hours[hour_key]
|
||||
mismatches = []
|
||||
|
||||
for field in ["bytes_written_delta", "bytes_read_delta", "sample_count"]:
|
||||
if field in record and record[field] != derived.get(field):
|
||||
mismatches.append(
|
||||
f"{field}: hourly={record[field]} derived={derived.get(field)}"
|
||||
)
|
||||
|
||||
if mismatches:
|
||||
logger.info(
|
||||
"Hourly.jsonl mismatch for %s: %s",
|
||||
hour_key,
|
||||
"; ".join(mismatches),
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning("Failed to diff hourly.jsonl: %s", e)
|
||||
|
||||
|
||||
def import_legacy_history(
|
||||
conn: sqlite3.Connection,
|
||||
history_path: Path,
|
||||
hourly_path: Optional[Path] = None,
|
||||
clock=None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Import legacy history.jsonl into the observation store.
|
||||
|
||||
This is the main entry point for legacy migration. It is:
|
||||
- Idempotent: second run no-ops on the legacy-import marker
|
||||
- Interruption-safe: single transaction
|
||||
- Never creates synthetic baselines
|
||||
|
||||
Args:
|
||||
conn: Connection to the observation store
|
||||
history_path: Path to history.jsonl
|
||||
hourly_path: Optional path to hourly.jsonl for diffing
|
||||
clock: Injected clock (for testing)
|
||||
|
||||
Returns:
|
||||
Dict with migration outcome
|
||||
"""
|
||||
# Check idempotency
|
||||
if is_legacy_imported(conn):
|
||||
return {"ok": True, "skipped": True, "reason": "already_imported"}
|
||||
|
||||
# Read and parse history.jsonl
|
||||
samples = []
|
||||
malformed_count = 0
|
||||
|
||||
if not history_path.exists():
|
||||
return {"ok": False, "error": f"History file not found: {history_path}"}
|
||||
|
||||
with open(history_path, "r") as f:
|
||||
for line_num, line in enumerate(f, 1):
|
||||
record = _parse_history_line(line, line_num)
|
||||
if record is None:
|
||||
malformed_count += 1
|
||||
continue
|
||||
samples.append(_record_to_sample(record))
|
||||
|
||||
if malformed_count > 0:
|
||||
logger.warning("Quarantined %d malformed lines from history.jsonl", malformed_count)
|
||||
|
||||
if not samples:
|
||||
return {"ok": False, "error": "No valid samples found in history.jsonl"}
|
||||
|
||||
# Sort samples by timestamp
|
||||
samples.sort(key=lambda x: x["ts"])
|
||||
|
||||
# Derive hour observations
|
||||
hour_observations = {}
|
||||
for sample in samples:
|
||||
ts = datetime.fromisoformat(sample["ts"])
|
||||
hour_start = ts.replace(minute=0, second=0, microsecond=0)
|
||||
hour_key = hour_start.strftime("%Y-%m-%dT%H:00:00Z")
|
||||
|
||||
if hour_key not in hour_observations:
|
||||
hour_observations[hour_key] = {
|
||||
"hour": hour_start,
|
||||
"samples": [],
|
||||
}
|
||||
hour_observations[hour_key]["samples"].append(sample)
|
||||
|
||||
derived_hours = {}
|
||||
for hour_key, hour_data in hour_observations.items():
|
||||
obs = _derive_hour_observation(hour_data["samples"], hour_data["hour"])
|
||||
if obs is not None:
|
||||
derived_hours[hour_key] = obs
|
||||
|
||||
# Diff against hourly.jsonl if provided
|
||||
if hourly_path:
|
||||
_diff_hourly_jsonl(hourly_path, derived_hours)
|
||||
|
||||
# Single transaction for the entire import
|
||||
try:
|
||||
# Begin transaction
|
||||
conn.execute("BEGIN IMMEDIATE")
|
||||
|
||||
# 1. Insert raw samples
|
||||
for sample in samples:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO samples (
|
||||
ts, device, subnqn, sn, mn, fr, capacity_bytes,
|
||||
percentage_used, available_spare, media_errors, power_on_hours,
|
||||
power_cycles, unsafe_shutdowns, temperature_c,
|
||||
data_units_written, data_units_read, bytes_written, bytes_read,
|
||||
critical_warning
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
sample["ts"],
|
||||
sample["device"],
|
||||
sample["subnqn"],
|
||||
sample["sn"],
|
||||
sample["mn"],
|
||||
sample["fr"],
|
||||
sample["capacity_bytes"],
|
||||
sample["percentage_used"],
|
||||
sample["available_spare"],
|
||||
sample["media_errors"],
|
||||
sample["power_on_hours"],
|
||||
sample["power_cycles"],
|
||||
sample["unsafe_shutdowns"],
|
||||
sample["temperature_c"],
|
||||
sample["data_units_written"],
|
||||
sample["data_units_read"],
|
||||
sample["bytes_written"],
|
||||
sample["bytes_read"],
|
||||
sample["critical_warning"],
|
||||
),
|
||||
)
|
||||
|
||||
# 2. Insert derived hour observations
|
||||
for hour_key, obs in sorted(derived_hours.items()):
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT OR REPLACE INTO hour_observations (
|
||||
hour, active_seconds, idle_seconds, powered_off_seconds,
|
||||
unknown_seconds, bytes_written_delta, bytes_read_delta,
|
||||
temperature_min, temperature_avg, temperature_max,
|
||||
sample_count, coverage
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
obs["hour"],
|
||||
obs["active_seconds"],
|
||||
obs["idle_seconds"],
|
||||
obs["powered_off_seconds"],
|
||||
obs["unknown_seconds"],
|
||||
obs["bytes_written_delta"],
|
||||
obs["bytes_read_delta"],
|
||||
obs["temperature_min"],
|
||||
obs["temperature_avg"],
|
||||
obs["temperature_max"],
|
||||
obs["sample_count"],
|
||||
obs["coverage"],
|
||||
),
|
||||
)
|
||||
|
||||
# 3. Open implicit monitoring period at first legacy sample
|
||||
first_sample_ts = datetime.fromisoformat(samples[0]["ts"])
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at) VALUES (?)",
|
||||
(first_sample_ts.isoformat(),),
|
||||
)
|
||||
|
||||
# 4. Close period with end_cause = migrated at migration moment
|
||||
if clock:
|
||||
migration_time = clock.utcnow()
|
||||
else:
|
||||
migration_time = datetime.now(timezone.utc)
|
||||
|
||||
conn.execute(
|
||||
"UPDATE monitoring_periods SET ended_at = ?, end_cause = ? WHERE ended_at IS NULL",
|
||||
(migration_time.isoformat(), "migrated"),
|
||||
)
|
||||
|
||||
# 5. Open legacy controller segment (mn-only)
|
||||
# Legacy identity is model-scoped only (§4.4)
|
||||
first_sample = samples[0]
|
||||
legacy_identity = {
|
||||
"mn": first_sample["mn"],
|
||||
"sn": "", # Legacy segments are mn-only
|
||||
"subnqn": "",
|
||||
"fr": "",
|
||||
}
|
||||
legacy_identity_key = f"legacy|{first_sample['mn']}"
|
||||
open_segment(
|
||||
conn,
|
||||
migration_time,
|
||||
legacy_identity,
|
||||
legacy_identity_key,
|
||||
identity_degraded=False,
|
||||
)
|
||||
|
||||
# 6. Set legacy import marker
|
||||
_ensure_metadata_table(conn)
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO store_metadata (key, value) VALUES (?, ?)",
|
||||
(LEGACY_IMPORT_MARKER, "true"),
|
||||
)
|
||||
|
||||
# Commit
|
||||
conn.commit()
|
||||
|
||||
except Exception as e:
|
||||
conn.rollback()
|
||||
raise RuntimeError(f"Migration failed: {e}") from e
|
||||
|
||||
# 7. Rename legacy files to *.migrated (only after commit)
|
||||
try:
|
||||
migrated_path = history_path.with_suffix(history_path.suffix + ".migrated")
|
||||
history_path.rename(migrated_path)
|
||||
logger.info("Renamed %s to %s", history_path, migrated_path)
|
||||
|
||||
if hourly_path and hourly_path.exists():
|
||||
hourly_migrated = hourly_path.with_suffix(hourly_path.suffix + ".migrated")
|
||||
hourly_path.rename(hourly_migrated)
|
||||
logger.info("Renamed %s to %s", hourly_path, hourly_migrated)
|
||||
except Exception as e:
|
||||
# Non-fatal: files weren't renamed but migration succeeded
|
||||
logger.warning("Failed to rename legacy files: %s", e)
|
||||
|
||||
return {
|
||||
"ok": True,
|
||||
"skipped": False,
|
||||
"samples_imported": len(samples),
|
||||
"hours_imported": len(derived_hours),
|
||||
"malformed_lines": malformed_count,
|
||||
"first_sample": samples[0]["ts"],
|
||||
"last_sample": samples[-1]["ts"],
|
||||
"legacy_identity_key": legacy_identity_key,
|
||||
}
|
||||
@@ -1,283 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""fenris-monitor: privileged helper for toggle, collect, and baseline operations.
|
||||
|
||||
This binary is the ONLY sanctioned control path for:
|
||||
- enable/disable (toggle) with monitoring-period bookkeeping
|
||||
- on-demand collection trigger
|
||||
- baseline set/clear persistence
|
||||
|
||||
Polkit authorizes this binary under com.bongbetic.fenris.monitor (auth_admin).
|
||||
|
||||
Spec: §8.4, §8.5, §8.6, §8.7
|
||||
|
||||
When run as a script, uses the fenris package from the installed wheel.
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import sqlite3
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
# Package dependencies are vendored independently of the host Python minor
|
||||
# version. Keep the venv fallback for the legacy development install.
|
||||
VENV_DIR = Path("/opt/fenris")
|
||||
if VENV_DIR.exists():
|
||||
vendor_dir = VENV_DIR / "vendor"
|
||||
if vendor_dir.is_dir():
|
||||
sys.path.insert(0, str(vendor_dir))
|
||||
else:
|
||||
site_packages = next((VENV_DIR / "lib").glob("python*/site-packages"), None)
|
||||
if site_packages:
|
||||
sys.path.insert(0, str(site_packages))
|
||||
|
||||
from fenris.store import init_store, get_store_path
|
||||
from fenris.monitoring_periods import (
|
||||
ensure_period_open,
|
||||
close_period,
|
||||
get_open_period,
|
||||
)
|
||||
|
||||
|
||||
DEFAULT_STORE_PATH = Path("/var/lib/fenris/observations.db")
|
||||
|
||||
|
||||
def is_root() -> bool:
|
||||
"""Check if running as root."""
|
||||
return os.geteuid() == 0
|
||||
|
||||
|
||||
def cmd_enable(args: argparse.Namespace) -> None:
|
||||
"""Enable monitoring: enable timer + open monitoring period.
|
||||
|
||||
Idempotent matrix (§8.6):
|
||||
- First-ever enable: opens a period at the enable moment
|
||||
- Resume with open period: no-op (gap stays inside as unknown)
|
||||
- Resume with no open period: opens a new row
|
||||
"""
|
||||
store_path = getattr(args, 'store_path', DEFAULT_STORE_PATH)
|
||||
conn = init_store(store_path)
|
||||
now = datetime.now(timezone.utc)
|
||||
|
||||
try:
|
||||
# Open monitoring period if none exists (§8.6)
|
||||
open_period = get_open_period(conn)
|
||||
if open_period is None:
|
||||
ensure_period_open(conn, now)
|
||||
print("Monitoring period opened at", now.isoformat())
|
||||
else:
|
||||
print("Monitoring period already open (id=%d)" % open_period["id"])
|
||||
|
||||
# Enable and start the timer
|
||||
if args.now:
|
||||
result = subprocess.run(
|
||||
["systemctl", "enable", "--now", "fenris-collect.timer"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
else:
|
||||
result = subprocess.run(
|
||||
["systemctl", "enable", "fenris-collect.timer"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
|
||||
if result.returncode != 0:
|
||||
print("Error enabling timer:", result.stderr, file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
print("Timer enabled" + (" and started" if args.now else ""))
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def cmd_disable(args: argparse.Namespace) -> None:
|
||||
"""Disable monitoring: disable timer + close monitoring period.
|
||||
|
||||
Idempotent matrix (§8.6):
|
||||
- Pause with open period: closes it user_disabled
|
||||
- Pause otherwise: no-op
|
||||
"""
|
||||
store_path = getattr(args, 'store_path', DEFAULT_STORE_PATH)
|
||||
if not store_path.exists():
|
||||
print("Error: Observation store not found at", store_path, file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
conn = init_store(store_path)
|
||||
now = datetime.now(timezone.utc)
|
||||
|
||||
try:
|
||||
# Close monitoring period if open (§8.6)
|
||||
open_period = get_open_period(conn)
|
||||
if open_period is not None:
|
||||
close_period(conn, now, "user_disabled")
|
||||
print("Monitoring period closed (id=%d)" % open_period["id"])
|
||||
else:
|
||||
print("No open monitoring period (no-op)")
|
||||
|
||||
# Disable and stop the timer
|
||||
if args.now:
|
||||
result = subprocess.run(
|
||||
["systemctl", "disable", "--now", "fenris-collect.timer"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
else:
|
||||
result = subprocess.run(
|
||||
["systemctl", "disable", "fenris-collect.timer"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
|
||||
if result.returncode != 0:
|
||||
print("Error disabling timer:", result.stderr, file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
print("Timer disabled" + (" and stopped" if args.now else ""))
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def cmd_collect(args: argparse.Namespace) -> None:
|
||||
"""Trigger on-demand collection.
|
||||
|
||||
Starts fenris-collect.service, blocks until exit, reports outcome.
|
||||
|
||||
Spec §8.7: fenris sample routes through fenris-monitor → systemctl start,
|
||||
which blocks until the oneshot exits; outcome reported synchronously.
|
||||
"""
|
||||
result = subprocess.run(
|
||||
["systemctl", "start", "fenris-collect.service"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
|
||||
if result.returncode == 0:
|
||||
print("Collection completed successfully")
|
||||
else:
|
||||
print("Collection failed:", result.stderr, file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def cmd_baseline_set(args: argparse.Namespace) -> None:
|
||||
"""Persist a baseline row after CLI-side validation.
|
||||
|
||||
Spec §8.4: fenris-monitor persists CLI-validated baseline rows.
|
||||
Spec PR-14: baseline set persists through polkit-guarded helper.
|
||||
"""
|
||||
store_path = getattr(args, 'store_path', DEFAULT_STORE_PATH)
|
||||
if not store_path.exists():
|
||||
print("Error: Observation store not found at", store_path, file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
conn = init_store(store_path)
|
||||
now = datetime.now(timezone.utc)
|
||||
|
||||
try:
|
||||
# Parse and validate baseline data
|
||||
data = json.loads(args.baseline_json)
|
||||
|
||||
required_fields = [
|
||||
"tbw_terabytes",
|
||||
"source_url",
|
||||
"document_revision",
|
||||
"entry_date",
|
||||
"model_string",
|
||||
"nominal_capacity_bytes",
|
||||
]
|
||||
for field in required_fields:
|
||||
if field not in data:
|
||||
print(f"Error: Missing required field: {field}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
# One active row replaced on edit (§6.2)
|
||||
conn.execute("DELETE FROM endurance_baseline")
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO endurance_baseline (
|
||||
tbw_terabytes, source_url, document_revision,
|
||||
entry_date, model_string, nominal_capacity_bytes,
|
||||
validated_by, verified, created_at, updated_at
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
data["tbw_terabytes"],
|
||||
data["source_url"],
|
||||
data["document_revision"],
|
||||
data["entry_date"],
|
||||
data["model_string"],
|
||||
data["nominal_capacity_bytes"],
|
||||
data.get("validated_by", "user"),
|
||||
data.get("verified", False),
|
||||
now.isoformat(),
|
||||
now.isoformat(),
|
||||
),
|
||||
)
|
||||
conn.commit()
|
||||
print("Baseline persisted")
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def cmd_baseline_clear(args: argparse.Namespace) -> None:
|
||||
"""Clear the endurance baseline.
|
||||
|
||||
Spec PR-14: baseline clear persists through polkit-guarded helper.
|
||||
"""
|
||||
store_path = getattr(args, 'store_path', DEFAULT_STORE_PATH)
|
||||
if not store_path.exists():
|
||||
print("Error: Observation store not found at", store_path, file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
conn = init_store(store_path)
|
||||
try:
|
||||
conn.execute("DELETE FROM endurance_baseline")
|
||||
conn.commit()
|
||||
print("Baseline cleared")
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="fenris-monitor",
|
||||
description="Fenris privileged helper for toggle, collect, and baseline operations.",
|
||||
)
|
||||
subparsers = parser.add_subparsers(dest="command", required=True)
|
||||
|
||||
# enable/disable
|
||||
enable_parser = subparsers.add_parser("enable", help="Enable monitoring")
|
||||
enable_parser.add_argument(
|
||||
"--now", action="store_true", help="Also start the timer immediately"
|
||||
)
|
||||
enable_parser.set_defaults(func=cmd_enable)
|
||||
|
||||
disable_parser = subparsers.add_parser("disable", help="Disable monitoring")
|
||||
disable_parser.add_argument(
|
||||
"--now", action="store_true", help="Also stop the timer immediately"
|
||||
)
|
||||
disable_parser.set_defaults(func=cmd_disable)
|
||||
|
||||
# collect
|
||||
collect_parser = subparsers.add_parser("collect", help="Trigger on-demand collection")
|
||||
collect_parser.set_defaults(func=cmd_collect)
|
||||
|
||||
# baseline
|
||||
baseline_parser = subparsers.add_parser("baseline", help="Baseline operations")
|
||||
baseline_sub = baseline_parser.add_subparsers(dest="baseline_action", required=True)
|
||||
|
||||
baseline_set = baseline_sub.add_parser("set", help="Persist baseline")
|
||||
baseline_set.add_argument("baseline_json", help="Baseline JSON data")
|
||||
baseline_set.set_defaults(func=cmd_baseline_set)
|
||||
|
||||
baseline_clear = baseline_sub.add_parser("clear", help="Clear baseline")
|
||||
baseline_clear.set_defaults(func=cmd_baseline_clear)
|
||||
|
||||
args = parser.parse_args()
|
||||
args.func(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,124 +0,0 @@
|
||||
"""Monitoring period bookkeeping per spec §5.2, §8.6, §9.8.
|
||||
|
||||
A monitoring period is a span during which Fenris monitoring is enabled.
|
||||
Powered-off time stays inside a period; deliberately disabled time does not.
|
||||
|
||||
Key contracts:
|
||||
- Run finding no open period opens one at the run moment, never backdated (§9.8)
|
||||
- Wall-clock outside periods excluded from numerator and denominator (§5.2)
|
||||
- End causes: user_disabled, migrated, unknown_gap
|
||||
"""
|
||||
import sqlite3
|
||||
from datetime import datetime
|
||||
|
||||
|
||||
def ensure_period_open(conn: sqlite3.Connection, run_time: datetime) -> None:
|
||||
"""Ensure a monitoring period is open. If none exists, open one at run_time.
|
||||
|
||||
Spec §9.8: A collection run finding no open monitoring period opens one
|
||||
at the run moment, never backdated.
|
||||
"""
|
||||
if get_open_period(conn) is not None:
|
||||
return # Already open — no-op
|
||||
|
||||
ts = run_time.isoformat()
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at) VALUES (?)",
|
||||
(ts,),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def close_period(
|
||||
conn: sqlite3.Connection,
|
||||
closed_at: datetime,
|
||||
end_cause: str,
|
||||
) -> None:
|
||||
"""Close the current open monitoring period.
|
||||
|
||||
Spec §8.6: Pause with an open period closes it user_disabled.
|
||||
If no period is open, this is a no-op (pause otherwise).
|
||||
"""
|
||||
open_period = get_open_period(conn)
|
||||
if open_period is None:
|
||||
return # No-op
|
||||
|
||||
ts = closed_at.isoformat()
|
||||
conn.execute(
|
||||
"UPDATE monitoring_periods SET ended_at = ?, end_cause = ? WHERE id = ?",
|
||||
(ts, end_cause, open_period["id"]),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def get_open_period(conn: sqlite3.Connection) -> dict | None:
|
||||
"""Return the currently open monitoring period, or None."""
|
||||
cursor = conn.execute(
|
||||
"SELECT id, started_at, ended_at, end_cause "
|
||||
"FROM monitoring_periods WHERE ended_at IS NULL LIMIT 1"
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
if row is None:
|
||||
return None
|
||||
return {
|
||||
"id": row[0],
|
||||
"started_at": row[1],
|
||||
"ended_at": row[2],
|
||||
"end_cause": row[3],
|
||||
}
|
||||
|
||||
|
||||
def is_inside_period(conn: sqlite3.Connection, ts: datetime) -> bool:
|
||||
"""Check if a timestamp falls inside any monitoring period.
|
||||
|
||||
Spec §5.2: Wall-clock outside periods is excluded from numerator/denominator.
|
||||
"""
|
||||
ts_str = ts.isoformat()
|
||||
cursor = conn.execute(
|
||||
"SELECT 1 FROM monitoring_periods "
|
||||
"WHERE started_at <= ? AND (ended_at IS NULL OR ended_at > ?) "
|
||||
"LIMIT 1",
|
||||
(ts_str, ts_str),
|
||||
)
|
||||
return cursor.fetchone() is not None
|
||||
|
||||
|
||||
def wall_clock_in_periods(
|
||||
conn: sqlite3.Connection,
|
||||
start: datetime,
|
||||
end: datetime,
|
||||
) -> int:
|
||||
"""Compute total wall-clock seconds between start and end that fall inside
|
||||
any monitoring period.
|
||||
|
||||
Used for denominator computation (§5.2).
|
||||
"""
|
||||
start_str = start.isoformat()
|
||||
end_str = end.isoformat()
|
||||
|
||||
cursor = conn.execute(
|
||||
"SELECT started_at, ended_at FROM monitoring_periods "
|
||||
"WHERE ended_at IS NULL OR ended_at > ? "
|
||||
"ORDER BY started_at",
|
||||
(start_str,),
|
||||
)
|
||||
|
||||
total = 0
|
||||
for row in cursor.fetchall():
|
||||
period_start = row[0]
|
||||
period_end = row[1] # None if open
|
||||
|
||||
# Clip period to [start, end]
|
||||
effective_start = max(period_start, start_str)
|
||||
if period_end is not None:
|
||||
effective_end = min(period_end, end_str)
|
||||
else:
|
||||
effective_end = end_str
|
||||
|
||||
if effective_start < effective_end:
|
||||
# Parse for arithmetic
|
||||
s = datetime.fromisoformat(effective_start)
|
||||
e = datetime.fromisoformat(effective_end)
|
||||
total += int((e - s).total_seconds())
|
||||
|
||||
return total
|
||||
@@ -1,580 +0,0 @@
|
||||
"""Projection core: the pure-function read path (spec §6).
|
||||
|
||||
Recomputes the complete projection contract on every read, never stores
|
||||
anything derived. Takes a read-only observation store connection and an
|
||||
injected clock; returns a ProjectionResult with confidence state,
|
||||
contributing facts, headline remaining time (when one exists), scenario
|
||||
range, Percentage-Used context line, and disclosure text.
|
||||
|
||||
Baseline precedence (§6.1):
|
||||
verified override → unverified override → implied → unavailable
|
||||
|
||||
Confidence rule table (§6.7):
|
||||
Supported — all conjuncts satisfied
|
||||
Limited — baseline + positive rate, failing facts shown
|
||||
Unavailable — no applicable baseline / zero rate / identity change
|
||||
|
||||
Arithmetic (§6.3):
|
||||
rate = regime DUW bytes / in-period wall-clock seconds
|
||||
projected = max(E_baseline − W_t, 0) / rate (rate > 0)
|
||||
E_rated = entered_TBW × 10¹² bytes
|
||||
E_implied = 100 · W_t / p (1 ≤ p ≤ 254)
|
||||
|
||||
Criteria: PR-1–PR-17, CI-4.
|
||||
"""
|
||||
import sqlite3
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from enum import Enum
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Constants (spec §6)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
HORIZON_DAYS = (7, 28, 90)
|
||||
TBW_TO_BYTES = 10 ** 12
|
||||
IMPLIED_P_MIN = 1
|
||||
IMPLIED_P_MAX = 254
|
||||
IMPLIED_MIN_PU_INCREMENTS = 2
|
||||
WARMING_MIN_DAYS = 14
|
||||
WARMING_MAX_LOW_COVERAGE = 2
|
||||
WARMING_COVERAGE_FLOOR = 0.50
|
||||
SUPPORTED_COVERAGE_FLOOR = 0.80
|
||||
HORIZON_AGREEMENT_FACTOR = 2
|
||||
BURST_GUARD_FRACTION = 0.50
|
||||
BURST_GUARD_LOOKBACK = 28
|
||||
YOUNG_REGIME_DAYS = 7
|
||||
HABIT_CHANGE_SHORT_WINDOW = 7
|
||||
HABIT_CHANGE_LONG_WINDOW = 28
|
||||
HABIT_CHANGE_UPPER_FACTOR = 2
|
||||
HABIT_CHANGE_LOWER_FACTOR = 0.5
|
||||
HABIT_CHANGE_CONSECUTIVE_DAYS = 3
|
||||
STALENESS_HOURS = 48
|
||||
WEAR_DISAGREEMENT_FACTOR = 2
|
||||
|
||||
|
||||
class ConfidenceState(Enum):
|
||||
UNSUPPORTED = "Unavailable"
|
||||
LIMITED = "Limited"
|
||||
SUPPORTED = "Supported"
|
||||
|
||||
|
||||
class BaselineTier(Enum):
|
||||
VERIFIED = "verified_override"
|
||||
UNVERIFIED = "unverified_override"
|
||||
IMPLIED = "implied"
|
||||
NONE = "none"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ScenarioRange:
|
||||
rates: Dict[int, float]
|
||||
min_days: int
|
||||
max_days: int
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ProjectionResult:
|
||||
confidence_state: ConfidenceState
|
||||
contributing_facts: List[str]
|
||||
headline_remaining_seconds: Optional[float]
|
||||
scenario_range: Optional[ScenarioRange]
|
||||
pu_context_line: str
|
||||
disclosure_text: List[str]
|
||||
baseline_tier: BaselineTier
|
||||
baseline_label: str
|
||||
regime_days: Optional[int]
|
||||
habit_change_fact: Optional[str]
|
||||
warming_fact: Optional[str]
|
||||
staleness_fact: Optional[str]
|
||||
degraded_identity_fact: Optional[str]
|
||||
zero_rate_fact: Optional[str]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Store queries
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _get_baseline(conn: sqlite3.Connection) -> Optional[Dict[str, Any]]:
|
||||
cursor = conn.execute(
|
||||
"SELECT id, tbw_terabytes, source_url, document_revision, entry_date, "
|
||||
" model_string, nominal_capacity_bytes, validated_by, verified "
|
||||
"FROM endurance_baseline LIMIT 1"
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
if row is None:
|
||||
return None
|
||||
return {
|
||||
"id": row[0], "tbw_terabytes": row[1], "source_url": row[2],
|
||||
"document_revision": row[3], "entry_date": row[4],
|
||||
"model_string": row[5], "nominal_capacity_bytes": row[6],
|
||||
"validated_by": row[7], "verified": bool(row[8]),
|
||||
}
|
||||
|
||||
|
||||
def _get_current_segment(conn: sqlite3.Connection) -> Optional[Dict[str, Any]]:
|
||||
cursor = conn.execute(
|
||||
"SELECT id, opened_at, identity_key, identity_degraded, mn "
|
||||
"FROM controller_segments ORDER BY id DESC LIMIT 1"
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
if row is None:
|
||||
return None
|
||||
return {
|
||||
"id": row[0], "opened_at": row[1], "identity_key": row[2],
|
||||
"identity_degraded": bool(row[3]), "mn": row[4],
|
||||
}
|
||||
|
||||
|
||||
def _get_days_in_segment(conn, segment_opened_at):
|
||||
cursor = conn.execute(
|
||||
"SELECT day, bytes_written_delta, coverage, sample_count "
|
||||
"FROM day_aggregates WHERE day >= ? ORDER BY day",
|
||||
(segment_opened_at[:10],),
|
||||
)
|
||||
return [{"day": r[0], "bytes_written": r[1], "coverage": r[2], "sample_count": r[3]}
|
||||
for r in cursor.fetchall()]
|
||||
|
||||
|
||||
def _get_all_days(conn):
|
||||
cursor = conn.execute(
|
||||
"SELECT day, bytes_written_delta, coverage, sample_count "
|
||||
"FROM day_aggregates ORDER BY day"
|
||||
)
|
||||
return [{"day": r[0], "bytes_written": r[1], "coverage": r[2], "sample_count": r[3]}
|
||||
for r in cursor.fetchall()]
|
||||
|
||||
|
||||
def _get_latest_pu(conn):
|
||||
cursor = conn.execute("SELECT percentage_used FROM samples ORDER BY id DESC LIMIT 1")
|
||||
row = cursor.fetchone()
|
||||
return row[0] if row else None
|
||||
|
||||
|
||||
def _get_pu_increments_in_segment(conn, segment_opened_at):
|
||||
cursor = conn.execute(
|
||||
"SELECT COUNT(DISTINCT percentage_used) FROM samples WHERE ts >= ?",
|
||||
(segment_opened_at,),
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
return max(0, (row[0] if row else 0) - 1)
|
||||
|
||||
|
||||
def _wall_clock_in_range(conn, start, end):
|
||||
start_str = start.isoformat()
|
||||
end_str = end.isoformat()
|
||||
cursor = conn.execute(
|
||||
"SELECT started_at, ended_at FROM monitoring_periods "
|
||||
"WHERE (ended_at IS NULL OR ended_at > ?) AND started_at < ? "
|
||||
"ORDER BY started_at", (start_str, end_str),
|
||||
)
|
||||
total = 0
|
||||
for row in cursor.fetchall():
|
||||
eff_start = max(row[0], start_str)
|
||||
eff_end = min(row[1], end_str) if row[1] is not None else end_str
|
||||
if eff_start < eff_end:
|
||||
total += int((datetime.fromisoformat(eff_end) - datetime.fromisoformat(eff_start)).total_seconds())
|
||||
return total
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Baseline resolution (§6.1, §6.2)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _resolve_baseline(conn, current_segment):
|
||||
baseline = _get_baseline(conn)
|
||||
facts = []
|
||||
if baseline is None:
|
||||
return BaselineTier.NONE, None, "no baseline", facts
|
||||
|
||||
mandatory = [baseline["source_url"], baseline["document_revision"],
|
||||
baseline["entry_date"], baseline["model_string"],
|
||||
baseline["nominal_capacity_bytes"]]
|
||||
provenance_complete = all(f is not None and f != "" for f in mandatory)
|
||||
|
||||
model_matches = True
|
||||
if current_segment is not None and baseline["model_string"] is not None:
|
||||
seg_mn = (current_segment.get("mn") or "").lower()
|
||||
bl_model = (baseline["model_string"] or "").lower()
|
||||
model_matches = bl_model in seg_mn or seg_mn in bl_model
|
||||
|
||||
if provenance_complete and model_matches and baseline["verified"]:
|
||||
label = "verified manufacturer TBW (%.1f TB)" % baseline["tbw_terabytes"]
|
||||
return BaselineTier.VERIFIED, baseline, label, facts
|
||||
|
||||
if not model_matches:
|
||||
facts.append(
|
||||
"baseline model '%s' does not match current drive '%s'"
|
||||
" — baseline retained but not applicable"
|
||||
% (baseline.get("model_string", ""),
|
||||
current_segment.get("mn", "") if current_segment else "")
|
||||
)
|
||||
return BaselineTier.NONE, baseline, "baseline model mismatch", facts
|
||||
|
||||
if not provenance_complete:
|
||||
label = "unverified TBW (%.1f TB) — user-supplied" % baseline["tbw_terabytes"]
|
||||
return BaselineTier.UNVERIFIED, baseline, label, facts
|
||||
|
||||
label = "verified manufacturer TBW (%.1f TB)" % baseline["tbw_terabytes"]
|
||||
return BaselineTier.VERIFIED, baseline, label, facts
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Rate computation
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _compute_regime_rate(days, conn, regime_start_day, clock_now):
|
||||
regime_bytes = sum(d["bytes_written"] for d in days if d["day"] >= regime_start_day)
|
||||
regime_start_dt = datetime.fromisoformat(regime_start_day + "T00:00:00+00:00")
|
||||
regime_wc = _wall_clock_in_range(conn, regime_start_dt, clock_now)
|
||||
if regime_wc <= 0:
|
||||
return None, regime_bytes, 0
|
||||
return regime_bytes / regime_wc, regime_bytes, regime_wc
|
||||
|
||||
|
||||
def _compute_horizon_rate(days, conn, horizon_days, clock_now):
|
||||
cutoff = (clock_now - timedelta(days=horizon_days)).strftime("%Y-%m-%d")
|
||||
# History must span the full horizon — no placeholders
|
||||
if not days or days[0]["day"] > cutoff:
|
||||
return None
|
||||
h_bytes = sum(d["bytes_written"] for d in days if d["day"] >= cutoff)
|
||||
covered = sum(1 for d in days if d["day"] >= cutoff)
|
||||
if covered == 0:
|
||||
return None
|
||||
h_start = datetime.fromisoformat(cutoff + "T00:00:00+00:00")
|
||||
h_wc = _wall_clock_in_range(conn, h_start, clock_now)
|
||||
if h_wc <= 0:
|
||||
return None
|
||||
return h_bytes / h_wc
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Habit change detection (§6.4)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _detect_habit_change(days):
|
||||
"""Detect habit change per spec §6.4.
|
||||
|
||||
Trailing 7-day mean >= 2x (or <= 0.5x) the preceding 28-day mean
|
||||
for 3 consecutive days. Returns (change_day, days_since) or None.
|
||||
The first divergence day is the earliest day in the consecutive run.
|
||||
"""
|
||||
need = HABIT_CHANGE_SHORT_WINDOW + HABIT_CHANGE_LONG_WINDOW
|
||||
if len(days) < need:
|
||||
return None
|
||||
|
||||
def _ratio_at(end_idx):
|
||||
"""Compute 7-day / preceding-28-day mean ratio ending at end_idx."""
|
||||
if end_idx < HABIT_CHANGE_SHORT_WINDOW - 1:
|
||||
return None
|
||||
se = end_idx + 1
|
||||
ss = se - HABIT_CHANGE_SHORT_WINDOW
|
||||
s_bytes = sum(d["bytes_written"] for d in days[ss:se])
|
||||
s_mean = s_bytes / HABIT_CHANGE_SHORT_WINDOW
|
||||
le = ss
|
||||
ls = le - HABIT_CHANGE_LONG_WINDOW
|
||||
if ls < 0:
|
||||
return None
|
||||
l_bytes = sum(d["bytes_written"] for d in days[ls:le])
|
||||
l_mean = l_bytes / HABIT_CHANGE_LONG_WINDOW
|
||||
if l_mean == 0:
|
||||
return None
|
||||
return s_mean / l_mean
|
||||
|
||||
# Scan backwards from the most recent day
|
||||
for i in range(len(days) - 1, HABIT_CHANGE_LONG_WINDOW + HABIT_CHANGE_SHORT_WINDOW - 2, -1):
|
||||
ratio = _ratio_at(i)
|
||||
if ratio is None:
|
||||
continue
|
||||
|
||||
is_upper = ratio >= HABIT_CHANGE_UPPER_FACTOR
|
||||
is_lower = ratio <= HABIT_CHANGE_LOWER_FACTOR
|
||||
if not (is_upper or is_lower):
|
||||
continue
|
||||
|
||||
# Count consecutive days going backwards from i
|
||||
consecutive = 1
|
||||
for j in range(i - 1, HABIT_CHANGE_LONG_WINDOW + HABIT_CHANGE_SHORT_WINDOW - 3, -1):
|
||||
r = _ratio_at(j)
|
||||
if r is None:
|
||||
break
|
||||
if (is_upper and r >= HABIT_CHANGE_UPPER_FACTOR) or \
|
||||
(is_lower and r <= HABIT_CHANGE_LOWER_FACTOR):
|
||||
consecutive += 1
|
||||
else:
|
||||
break
|
||||
|
||||
if consecutive >= HABIT_CHANGE_CONSECUTIVE_DAYS:
|
||||
change_idx = i - consecutive + 1
|
||||
change_day = days[change_idx]["day"]
|
||||
days_since = (datetime.fromisoformat(days[-1]["day"]) - datetime.fromisoformat(change_day)).days
|
||||
return change_day, days_since
|
||||
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Confidence rule table (§6.7)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _evaluate_confidence(tier, rate, regime_days, days, current_segment,
|
||||
clock_now, warming_days, warming_low_coverage,
|
||||
habit_change, staleness_hours, scenario_range):
|
||||
facts = []
|
||||
|
||||
if tier == BaselineTier.NONE:
|
||||
facts.append("no applicable endurance baseline")
|
||||
return ConfidenceState.UNSUPPORTED, facts
|
||||
|
||||
if rate is None or rate <= 0:
|
||||
facts.append("no finite projection from this history")
|
||||
return ConfidenceState.UNSUPPORTED, facts
|
||||
|
||||
supported_facts = []
|
||||
failing = False
|
||||
|
||||
# 1. Verified baseline
|
||||
if tier != BaselineTier.VERIFIED:
|
||||
failing = True
|
||||
else:
|
||||
supported_facts.append("verified manufacturer TBW")
|
||||
|
||||
# 2. >= 14 qualifying days
|
||||
qualifying = sum(1 for d in days if d["coverage"] >= WARMING_COVERAGE_FLOOR and d["sample_count"] > 0)
|
||||
if qualifying < WARMING_MIN_DAYS:
|
||||
failing = True
|
||||
else:
|
||||
supported_facts.append("%d calendar days" % qualifying)
|
||||
|
||||
# 3. Coverage >= 80%
|
||||
total_wc = len(days) * 86400
|
||||
total_known = sum(int(d["coverage"] * 86400) for d in days)
|
||||
avg_cov = total_known / total_wc if total_wc > 0 else 0.0
|
||||
if avg_cov < SUPPORTED_COVERAGE_FLOOR:
|
||||
failing = True
|
||||
else:
|
||||
supported_facts.append("%d%% interval coverage" % int(avg_cov * 100))
|
||||
|
||||
# 4. Fresh (< 48h)
|
||||
if staleness_hours is not None and staleness_hours > STALENESS_HOURS:
|
||||
failing = True
|
||||
elif staleness_hours is not None:
|
||||
supported_facts.append("recent data")
|
||||
|
||||
# 5. Horizon agreement
|
||||
if scenario_range is not None and len(scenario_range.rates) >= 2:
|
||||
rl = list(scenario_range.rates.values())
|
||||
if min(rl) > 0 and max(rl) / min(rl) > HORIZON_AGREEMENT_FACTOR:
|
||||
failing = True
|
||||
else:
|
||||
supported_facts.append("%d weekly cycles" % len(scenario_range.rates))
|
||||
else:
|
||||
failing = True
|
||||
|
||||
# 6. Burst guard
|
||||
if not failing and len(days) >= BURST_GUARD_LOOKBACK:
|
||||
t28 = sum(d["bytes_written"] for d in days[-BURST_GUARD_LOOKBACK:])
|
||||
for d in days[-BURST_GUARD_LOOKBACK:]:
|
||||
if t28 > 0 and d["bytes_written"] >= BURST_GUARD_FRACTION * t28:
|
||||
failing = True
|
||||
break
|
||||
if not failing:
|
||||
supported_facts.append("no burst days")
|
||||
|
||||
# 7. Regime >= 7 days
|
||||
if regime_days < YOUNG_REGIME_DAYS:
|
||||
failing = True
|
||||
|
||||
# 8. Degraded identity
|
||||
if current_segment and current_segment.get("identity_degraded"):
|
||||
failing = True
|
||||
facts.append("controller identity unavailable — replacement detection relies on write-counter continuity only")
|
||||
|
||||
if not failing:
|
||||
return ConfidenceState.SUPPORTED, supported_facts
|
||||
|
||||
# Limited
|
||||
limited_facts = list(supported_facts)
|
||||
if staleness_hours is not None and staleness_hours > STALENESS_HOURS:
|
||||
limited_facts.append("newest data %dh old (≥48h)" % staleness_hours)
|
||||
if habit_change is not None:
|
||||
limited_facts.append("usage habit changed %d days ago" % habit_change[1])
|
||||
if regime_days < YOUNG_REGIME_DAYS:
|
||||
limited_facts.append("regime only %d days old (≥7 required)" % regime_days)
|
||||
if current_segment and current_segment.get("identity_degraded"):
|
||||
degraded_fact = "controller identity unavailable — replacement detection relies on write-counter continuity only"
|
||||
if degraded_fact not in limited_facts:
|
||||
limited_facts.append(degraded_fact)
|
||||
|
||||
return ConfidenceState.LIMITED, limited_facts
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Disclosure text (§6.11)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
DISCLOSURES = [
|
||||
"This is an endurance projection, not a predicted hardware-failure date.",
|
||||
("Percentage Used is vendor-specific; 100 means estimated endurance consumed "
|
||||
"but may not mean failure, it can exceed 100, and 255 is saturated."),
|
||||
("Rated TBW can be a warranty/endurance threshold with separate time and "
|
||||
"eligibility terms, not a failure threshold."),
|
||||
("DUW is upward-rounded host writes excluding metadata and selected commands, "
|
||||
"not exact physical NAND writes."),
|
||||
("Projection quality depends on baseline provenance, history duration and "
|
||||
"completeness, recentness, stability, and representative usage cycles; "
|
||||
"future workload and firmware behavior remain outside the observed evidence."),
|
||||
("Gaps can preserve an aggregate counter delta without preserving hourly "
|
||||
"timing; unexplained and deliberately disabled periods must be distinguished."),
|
||||
]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# PU context line (§6.1)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _build_pu_context_line(conn, rate, days, clock_now):
|
||||
pu = _get_latest_pu(conn)
|
||||
if pu is None:
|
||||
return "Percentage Used: unknown"
|
||||
if rate is None or rate <= 0 or not days:
|
||||
return "Percentage Used: %d%%" % pu
|
||||
|
||||
total_bytes = sum(d["bytes_written"] for d in days)
|
||||
if total_bytes <= 0:
|
||||
return "Percentage Used: %d%%" % pu
|
||||
|
||||
total_days_count = len(days)
|
||||
if total_days_count == 0:
|
||||
return "Percentage Used: %d%%" % pu
|
||||
|
||||
pu_daily = total_bytes / total_days_count
|
||||
obs_daily = rate * 86400
|
||||
|
||||
if pu_daily > 0:
|
||||
ratio = obs_daily / pu_daily
|
||||
if ratio > WEAR_DISAGREEMENT_FACTOR or ratio < 1.0 / WEAR_DISAGREEMENT_FACTOR:
|
||||
return ("Percentage Used: %d%% — vendor wear estimate disagrees "
|
||||
"with observed write rate (>2× difference)") % pu
|
||||
|
||||
return "Percentage Used: %d%%" % pu
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Main projection function
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def compute_projection(conn, clock_now):
|
||||
facts = []
|
||||
habit_change_fact = None
|
||||
warming_fact = None
|
||||
staleness_fact = None
|
||||
degraded_identity_fact = None
|
||||
zero_rate_fact = None
|
||||
|
||||
current_segment = _get_current_segment(conn)
|
||||
tier, baseline, baseline_label, baseline_facts = _resolve_baseline(conn, current_segment)
|
||||
facts.extend(baseline_facts)
|
||||
|
||||
segment_days = _get_days_in_segment(conn, current_segment["opened_at"]) if current_segment else _get_all_days(conn)
|
||||
all_days = _get_all_days(conn)
|
||||
|
||||
regime_start_day = None
|
||||
habit_change = None
|
||||
|
||||
if segment_days:
|
||||
earliest = segment_days[0]["day"]
|
||||
cutoff_90 = (clock_now - timedelta(days=90)).strftime("%Y-%m-%d")
|
||||
regime_start_day = max(earliest, cutoff_90)
|
||||
habit_change = _detect_habit_change(segment_days)
|
||||
if habit_change is not None:
|
||||
regime_start_day = habit_change[0]
|
||||
habit_change_fact = "usage habit changed %d days ago" % habit_change[1]
|
||||
facts.append(habit_change_fact)
|
||||
|
||||
rate = None
|
||||
regime_bytes = 0
|
||||
regime_days_count = 0
|
||||
|
||||
if segment_days and regime_start_day is not None:
|
||||
rate, regime_bytes, _ = _compute_regime_rate(segment_days, conn, regime_start_day, clock_now)
|
||||
regime_days_count = sum(1 for d in segment_days if d["day"] >= regime_start_day)
|
||||
|
||||
if rate is not None and rate <= 0:
|
||||
zero_rate_fact = "no finite projection from this history"
|
||||
facts.append(zero_rate_fact)
|
||||
|
||||
scenario = None
|
||||
horizon_rates = {}
|
||||
for h in HORIZON_DAYS:
|
||||
hr = _compute_horizon_rate(all_days, conn, h, clock_now)
|
||||
if hr is not None:
|
||||
horizon_rates[h] = hr
|
||||
if horizon_rates:
|
||||
scenario = ScenarioRange(rates=horizon_rates, min_days=min(horizon_rates), max_days=max(horizon_rates))
|
||||
|
||||
total_days_count = len(segment_days)
|
||||
days_below_coverage = sum(1 for d in segment_days
|
||||
if d["coverage"] < WARMING_COVERAGE_FLOOR or d["sample_count"] == 0)
|
||||
qualifying = total_days_count - days_below_coverage
|
||||
if total_days_count < WARMING_MIN_DAYS or days_below_coverage > WARMING_MAX_LOW_COVERAGE:
|
||||
warming_fact = "warming up: %d of %d qualifying days" % (qualifying, WARMING_MIN_DAYS)
|
||||
facts.append(warming_fact)
|
||||
|
||||
staleness_hours = None
|
||||
if segment_days:
|
||||
newest_dt = datetime.fromisoformat(segment_days[-1]["day"] + "T12:00:00+00:00")
|
||||
staleness_hours = int((clock_now - newest_dt).total_seconds() / 3600)
|
||||
if staleness_hours > STALENESS_HOURS:
|
||||
staleness_fact = "newest data %dh old (≥48h)" % staleness_hours
|
||||
facts.append(staleness_fact)
|
||||
|
||||
if current_segment and current_segment.get("identity_degraded"):
|
||||
degraded_identity_fact = "controller identity unavailable — replacement detection relies on write-counter continuity only"
|
||||
facts.append(degraded_identity_fact)
|
||||
|
||||
state, conf_facts = _evaluate_confidence(
|
||||
tier, rate, regime_days_count, segment_days, current_segment, clock_now,
|
||||
qualifying, 0, habit_change, staleness_hours, scenario,
|
||||
)
|
||||
|
||||
all_facts = list(facts)
|
||||
for cf in conf_facts:
|
||||
if cf not in all_facts:
|
||||
all_facts.append(cf)
|
||||
|
||||
headline_seconds = None
|
||||
if state != ConfidenceState.UNSUPPORTED and rate is not None and rate > 0 and baseline is not None:
|
||||
if tier in (BaselineTier.VERIFIED, BaselineTier.UNVERIFIED):
|
||||
E_baseline = baseline["tbw_terabytes"] * TBW_TO_BYTES
|
||||
elif tier == BaselineTier.IMPLIED:
|
||||
p = _get_latest_pu(conn)
|
||||
if p is not None and IMPLIED_P_MIN <= p <= IMPLIED_P_MAX:
|
||||
E_baseline = 100 * regime_bytes / p
|
||||
else:
|
||||
E_baseline = None
|
||||
else:
|
||||
E_baseline = None
|
||||
if E_baseline is not None:
|
||||
headline_seconds = max(E_baseline - regime_bytes, 0) / rate
|
||||
|
||||
pu_line = _build_pu_context_line(conn, rate, segment_days, clock_now)
|
||||
|
||||
return ProjectionResult(
|
||||
confidence_state=state,
|
||||
contributing_facts=all_facts,
|
||||
headline_remaining_seconds=headline_seconds,
|
||||
scenario_range=scenario,
|
||||
pu_context_line=pu_line,
|
||||
disclosure_text=list(DISCLOSURES),
|
||||
baseline_tier=tier,
|
||||
baseline_label=baseline_label,
|
||||
regime_days=regime_days_count,
|
||||
habit_change_fact=habit_change_fact,
|
||||
warming_fact=warming_fact,
|
||||
staleness_fact=staleness_fact,
|
||||
degraded_identity_fact=degraded_identity_fact,
|
||||
zero_rate_fact=zero_rate_fact,
|
||||
)
|
||||
@@ -1,31 +0,0 @@
|
||||
"""Raw sample pruning per spec §3.4, ST-5.
|
||||
|
||||
Raw samples are pruned opportunistically to 14 days.
|
||||
Hour observations and day aggregates are retained indefinitely.
|
||||
"""
|
||||
import sqlite3
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
# Spec §3.4: Raw-sample retention
|
||||
RAW_SAMPLE_RETENTION_DAYS = 14
|
||||
|
||||
|
||||
def prune_old_samples(
|
||||
conn: sqlite3.Connection,
|
||||
now: datetime,
|
||||
retention_days: int = RAW_SAMPLE_RETENTION_DAYS,
|
||||
) -> int:
|
||||
"""Remove raw samples older than retention_days.
|
||||
|
||||
Args:
|
||||
conn: Connection to the observation store.
|
||||
now: Current UTC time.
|
||||
retention_days: Number of days to retain (default 14).
|
||||
|
||||
Returns:
|
||||
Number of samples removed.
|
||||
"""
|
||||
cutoff = (now - timedelta(days=retention_days)).isoformat()
|
||||
cursor = conn.execute("DELETE FROM samples WHERE ts < ?", (cutoff,))
|
||||
conn.commit()
|
||||
return cursor.rowcount
|
||||
@@ -1,147 +0,0 @@
|
||||
"""Controller segment management.
|
||||
|
||||
Handles identity-based segmentation of observation history:
|
||||
- Find current (most recent) segment
|
||||
- Determine if a new segment should open
|
||||
- Open new segments with frozen metadata snapshot
|
||||
|
||||
Segmentation axes (independent):
|
||||
- Identity key change → quarantines prior history
|
||||
- DUW decrease with unchanged identity → new segment, prior history stays as habit evidence
|
||||
|
||||
Blank-key semantics (PR-16):
|
||||
- To/from blank is an identity change → quarantines
|
||||
- Equal blanks continue the segment, segmented by DUW monotonicity alone
|
||||
"""
|
||||
import sqlite3
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
from .collector import normalize_identity
|
||||
|
||||
|
||||
def find_current_segment(conn: sqlite3.Connection) -> Optional[Dict[str, Any]]:
|
||||
"""Find the most recent (open) controller segment.
|
||||
|
||||
Returns the segment dict or None if no segments exist.
|
||||
"""
|
||||
cursor = conn.execute(
|
||||
"SELECT id, opened_at, identity_key, identity_degraded, "
|
||||
"subnqn, sn, mn, fr, vid, ssvid, transport "
|
||||
"FROM controller_segments ORDER BY id DESC LIMIT 1"
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
if row is None:
|
||||
return None
|
||||
|
||||
return {
|
||||
"id": row[0],
|
||||
"opened_at": row[1],
|
||||
"identity_key": row[2],
|
||||
"identity_degraded": bool(row[3]),
|
||||
"subnqn": row[4],
|
||||
"sn": row[5],
|
||||
"mn": row[6],
|
||||
"fr": row[7],
|
||||
"vid": row[8],
|
||||
"ssvid": row[9],
|
||||
"transport": row[10],
|
||||
}
|
||||
|
||||
|
||||
def get_last_duw(conn: sqlite3.Connection, segment_id: int) -> Optional[int]:
|
||||
"""Get the bytes_written from the most recent sample in a segment.
|
||||
|
||||
Returns None if no samples exist in the segment.
|
||||
"""
|
||||
# Samples don't have a segment_id FK yet, so we need to find the
|
||||
# latest sample before the segment's opened_at, or the latest sample
|
||||
# if this is the first segment.
|
||||
#
|
||||
# For now, we'll use a simpler approach: get the latest sample's bytes_written.
|
||||
# TODO: Add segment_id FK to samples table in next schema migration
|
||||
cursor = conn.execute(
|
||||
"SELECT bytes_written FROM samples ORDER BY id DESC LIMIT 1"
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
return row[0] if row else None
|
||||
|
||||
|
||||
def should_open_new_segment(
|
||||
current_segment: Optional[Dict[str, Any]],
|
||||
new_identity_key: str,
|
||||
new_bytes_written: int,
|
||||
conn: sqlite3.Connection,
|
||||
) -> Tuple[bool, Optional[str]]:
|
||||
"""Determine if a new segment should open.
|
||||
|
||||
Returns (should_open, reason).
|
||||
reason is None if no new segment, or a string describing why.
|
||||
"""
|
||||
# No current segment → must open first segment
|
||||
if current_segment is None:
|
||||
return True, "first_segment"
|
||||
|
||||
old_key = current_segment["identity_key"] or ""
|
||||
|
||||
# Identity key change (including to/from blank)
|
||||
if old_key != new_identity_key:
|
||||
return True, "identity_change"
|
||||
|
||||
# DUW decrease (counter reset or controller replacement with same identity)
|
||||
last_duw = get_last_duw(conn, current_segment["id"])
|
||||
if last_duw is not None and new_bytes_written < last_duw:
|
||||
return True, "duw_decrease"
|
||||
|
||||
# Same identity, DUW non-decreasing → continue segment
|
||||
return False, None
|
||||
|
||||
|
||||
|
||||
def open_segment(
|
||||
conn: sqlite3.Connection,
|
||||
now: datetime,
|
||||
identity: Dict[str, Any],
|
||||
identity_key: str,
|
||||
identity_degraded: bool,
|
||||
) -> Dict[str, Any]:
|
||||
"""Open a new controller segment with frozen metadata snapshot.
|
||||
|
||||
The metadata is immutable once frozen.
|
||||
"""
|
||||
cursor = conn.execute(
|
||||
"""
|
||||
INSERT INTO controller_segments (
|
||||
opened_at, identity_key, identity_degraded,
|
||||
subnqn, sn, mn, fr, vid, ssvid, transport
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
now.isoformat(),
|
||||
identity_key if identity_key else None,
|
||||
identity_degraded,
|
||||
identity.get("subnqn") or None,
|
||||
identity.get("sn") or None,
|
||||
identity.get("mn") or None,
|
||||
identity.get("fr") or None,
|
||||
identity.get("vid") or None,
|
||||
identity.get("ssvid") or None,
|
||||
identity.get("transport") or None,
|
||||
),
|
||||
)
|
||||
|
||||
segment_id = cursor.lastrowid
|
||||
|
||||
return {
|
||||
"id": segment_id,
|
||||
"opened_at": now.isoformat(),
|
||||
"identity_key": identity_key if identity_key else None,
|
||||
"identity_degraded": identity_degraded,
|
||||
"subnqn": identity.get("subnqn") or None,
|
||||
"sn": identity.get("sn") or None,
|
||||
"mn": identity.get("mn") or None,
|
||||
"fr": identity.get("fr") or None,
|
||||
"vid": identity.get("vid") or None,
|
||||
"ssvid": identity.get("ssvid") or None,
|
||||
"transport": identity.get("transport") or None,
|
||||
}
|
||||
@@ -1,658 +0,0 @@
|
||||
"""Read-only CLI status command: the CLI twin of the TUI (spec §8.8, LC-9, CI-2).
|
||||
|
||||
Composes from the observation store (read-only) and allow-listed systemctl
|
||||
properties: projection facts, four separate service facts (boot enablement,
|
||||
runtime activity, last collect outcome, freshness), and a journalctl hint on
|
||||
failure or staleness. Never auto-samples, never prompts.
|
||||
|
||||
Freshness constants are defined once here and shared with the TUI (§8.9):
|
||||
fresh — newest sample within 2 × cadence + AccuracySec + 60 s
|
||||
missed — between fresh and 48 h
|
||||
stale — ≥ 48 h
|
||||
empty — no observations yet
|
||||
|
||||
Criteria: LC-9, CI-2, CI-4, FL-4, FL-5, FL-7.
|
||||
"""
|
||||
import sqlite3
|
||||
import subprocess
|
||||
import sys
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from .projection import compute_projection, ConfidenceState, DISCLOSURES
|
||||
from .store import SCHEMA_VERSION
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Freshness constants (§8.9, §8.2)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
CADENCE_DEFAULT_S = 300 # 5 min
|
||||
ACCURACY_SEC = 30
|
||||
FRESH_THRESHOLD_S = 2 * CADENCE_DEFAULT_S + ACCURACY_SEC + 60 # 690 s
|
||||
STALENESS_THRESHOLD_S = 48 * 3600 # 48 h
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Configuration reading (§8.3)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
CONFIG_PATH = Path("/etc/fenris/fenris.conf")
|
||||
|
||||
|
||||
def read_config() -> Dict[str, Any]:
|
||||
"""Read the world-readable configuration file.
|
||||
|
||||
Returns a dict with at least 'device'.
|
||||
Raises ConfigError with a reason string on any failure.
|
||||
"""
|
||||
if not CONFIG_PATH.exists():
|
||||
raise ConfigError("configuration file not found at %s" % CONFIG_PATH)
|
||||
|
||||
try:
|
||||
text = CONFIG_PATH.read_text()
|
||||
except OSError as e:
|
||||
raise ConfigError("cannot read %s: %s" % (CONFIG_PATH, e))
|
||||
|
||||
device = None
|
||||
for line in text.splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
if "=" in line:
|
||||
key, _, value = line.partition("=")
|
||||
key = key.strip()
|
||||
value = value.strip().strip('"').strip("'")
|
||||
if key == "device":
|
||||
device = value
|
||||
break
|
||||
|
||||
if not device:
|
||||
raise ConfigError("no device selector in %s" % CONFIG_PATH)
|
||||
|
||||
return {"device": device}
|
||||
|
||||
|
||||
class ConfigError(Exception):
|
||||
"""Configuration is invalid — surfaced in status as a fact (§8.3)."""
|
||||
pass
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Store opening (read-only, §3, §9.4, §9.5)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def open_store_readonly(store_path: Path) -> sqlite3.Connection:
|
||||
"""Open the observation store read-only.
|
||||
|
||||
Raises StoreFault if unreadable, NewerSchema if user_version > SCHEMA_VERSION.
|
||||
"""
|
||||
try:
|
||||
exists = store_path.exists()
|
||||
except OSError as e:
|
||||
# A non-group user stat()ing a 2750 store directory gets
|
||||
# PermissionError before any StoreFault can be raised (issue #54).
|
||||
raise StoreFault("observation store not readable: %s" % e)
|
||||
if not exists:
|
||||
raise StoreFault("observation store not found at %s" % store_path)
|
||||
|
||||
try:
|
||||
conn = sqlite3.connect("file:%s?mode=ro" % store_path, uri=True)
|
||||
conn.row_factory = sqlite3.Row
|
||||
except sqlite3.Error as e:
|
||||
raise StoreFault("observation store unreadable: %s" % e)
|
||||
|
||||
try:
|
||||
cursor = conn.execute("PRAGMA user_version")
|
||||
version = cursor.fetchone()[0]
|
||||
except sqlite3.Error as e:
|
||||
conn.close()
|
||||
raise StoreFault("observation store unreadable: %s" % e)
|
||||
|
||||
if version > SCHEMA_VERSION:
|
||||
conn.close()
|
||||
raise NewerSchema(version)
|
||||
|
||||
return conn
|
||||
|
||||
|
||||
class StoreFault(Exception):
|
||||
"""Store is present but unreadable or corrupt (§9.4)."""
|
||||
pass
|
||||
|
||||
|
||||
class NewerSchema(Exception):
|
||||
"""Store has a newer user_version (§9.5)."""
|
||||
def __init__(self, version: int):
|
||||
self.version = version
|
||||
super().__init__("schema version %d" % version)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Service state queries (§8.8 — allow-listed systemctl properties)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _systemctl_show(unit: str, *properties: str) -> Dict[str, str]:
|
||||
"""Query systemctl show for specific properties. Returns empty dict on failure."""
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["systemctl", "show", unit, "--property=" + ",".join(properties)],
|
||||
capture_output=True, text=True, timeout=5,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
return {}
|
||||
out = {}
|
||||
for line in result.stdout.splitlines():
|
||||
if "=" in line:
|
||||
key, _, value = line.partition("=")
|
||||
out[key.strip()] = value.strip()
|
||||
return out
|
||||
except (subprocess.TimeoutExpired, FileNotFoundError, OSError):
|
||||
return {}
|
||||
|
||||
|
||||
def _journalctl_hint(unit: str, lines: int = 5) -> Optional[str]:
|
||||
"""Get the last N journal lines for a unit. Returns None on failure."""
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["journalctl", "-u", unit, "--no-pager", "-n", str(lines), "--output=short-iso"],
|
||||
capture_output=True, text=True, timeout=5,
|
||||
)
|
||||
if result.returncode != 0 or not result.stdout.strip():
|
||||
return None
|
||||
return result.stdout.strip()
|
||||
except (subprocess.TimeoutExpired, FileNotFoundError, OSError):
|
||||
return None
|
||||
|
||||
|
||||
def query_service_state() -> Dict[str, Any]:
|
||||
"""Query systemctl for the four separate service facts (§7.3, LC-9).
|
||||
|
||||
Returns dict with keys:
|
||||
boot_enabled: bool
|
||||
timer_active: bool
|
||||
last_collect_ok: Optional[bool]
|
||||
last_collect_age_s: Optional[int]
|
||||
last_collect_reason: Optional[str]
|
||||
"""
|
||||
timer_props = _systemctl_show(
|
||||
"fenris-collect.timer",
|
||||
"UnitFileState", "ActiveState", "LastTriggerUSec",
|
||||
)
|
||||
service_props = _systemctl_show(
|
||||
"fenris-collect.service",
|
||||
"ActiveState", "ExecMainStatus", "ExecMainExitTimestamp",
|
||||
)
|
||||
|
||||
boot_enabled_str = timer_props.get("UnitFileState", "")
|
||||
boot_enabled = boot_enabled_str == "enabled"
|
||||
|
||||
active_state = timer_props.get("ActiveState", "inactive")
|
||||
timer_active = active_state == "active"
|
||||
|
||||
last_collect_ok = None
|
||||
last_collect_age_s = None
|
||||
last_collect_reason = None
|
||||
|
||||
last_trigger = timer_props.get("LastTriggerUSec", "")
|
||||
if last_trigger and last_trigger != "n/a":
|
||||
try:
|
||||
trigger_dt = datetime.fromisoformat(last_trigger.replace("Z", "+00:00"))
|
||||
now = datetime.now(timezone.utc)
|
||||
last_collect_age_s = int((now - trigger_dt).total_seconds())
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
exec_status = service_props.get("ExecMainStatus", "")
|
||||
if exec_status:
|
||||
try:
|
||||
exit_code = int(exec_status)
|
||||
last_collect_ok = exit_code == 0
|
||||
if exit_code != 0:
|
||||
last_collect_reason = "exit code %d" % exit_code
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
return {
|
||||
"boot_enabled": boot_enabled,
|
||||
"timer_active": timer_active,
|
||||
"last_collect_ok": last_collect_ok,
|
||||
"last_collect_age_s": last_collect_age_s,
|
||||
"last_collect_reason": last_collect_reason,
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Freshness grading (§8.9)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def grade_freshness(newest_sample_ts: Optional[str], clock_now: datetime) -> str:
|
||||
"""Grade freshness from the newest sample timestamp (never a stored flag).
|
||||
|
||||
Returns 'fresh', 'missed', 'stale', or 'empty'.
|
||||
"""
|
||||
if newest_sample_ts is None:
|
||||
return "empty"
|
||||
|
||||
try:
|
||||
ts = datetime.fromisoformat(newest_sample_ts)
|
||||
if ts.tzinfo is None:
|
||||
ts = ts.replace(tzinfo=timezone.utc)
|
||||
else:
|
||||
ts = ts.astimezone(timezone.utc)
|
||||
except (ValueError, TypeError):
|
||||
return "empty"
|
||||
|
||||
age_s = (clock_now - ts).total_seconds()
|
||||
|
||||
if age_s <= FRESH_THRESHOLD_S:
|
||||
return "fresh"
|
||||
elif age_s < STALENESS_THRESHOLD_S:
|
||||
return "missed"
|
||||
else:
|
||||
return "stale"
|
||||
|
||||
|
||||
def freshness_age_human(age_s: Optional[int]) -> str:
|
||||
"""Human-readable age string for freshness fact."""
|
||||
if age_s is None:
|
||||
return "unknown age"
|
||||
if age_s < 60:
|
||||
return "%ds ago" % age_s
|
||||
if age_s < 3600:
|
||||
return "%dm ago" % (age_s // 60)
|
||||
if age_s < 86400:
|
||||
return "%dh %dm ago" % (age_s // 3600, (age_s % 3600) // 60)
|
||||
return "%dd ago" % (age_s // 86400)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Drive anomalies (§9.7 — FL-7)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _query_drive_facts(conn: sqlite3.Connection) -> List[str]:
|
||||
"""Query drive-reported anomalies from the latest sample (§9.7, FL-7).
|
||||
|
||||
critical_warning, media errors, and unsafe shutdowns render as ordinary
|
||||
facts and never affect the projection.
|
||||
"""
|
||||
cursor = conn.execute(
|
||||
"SELECT critical_warning, media_errors, unsafe_shutdowns, "
|
||||
"temperature_c, available_spare "
|
||||
"FROM samples ORDER BY id DESC LIMIT 1"
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
if row is None:
|
||||
return []
|
||||
|
||||
facts = []
|
||||
cw = row[0]
|
||||
if cw and cw != 0:
|
||||
facts.append("critical warning: %s" % hex(cw) if isinstance(cw, int) else str(cw))
|
||||
me = row[1]
|
||||
if me and me > 0:
|
||||
facts.append("media errors: %d" % me)
|
||||
us = row[2]
|
||||
if us and us > 0:
|
||||
facts.append("unsafe shutdowns: %d" % us)
|
||||
return facts
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Retired command rejection (§8.8)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
RETIRED_COMMANDS = {
|
||||
"start": "use 'fenris monitor resume' to enable monitoring",
|
||||
"stop": "use 'fenris monitor pause' to disable monitoring",
|
||||
"run": "use 'fenris monitor resume' to enable monitoring; the timer runs in the background",
|
||||
}
|
||||
|
||||
MIGRATION_POINTERS = {
|
||||
"--device": "device is configured in /etc/fenris/fenris.conf",
|
||||
}
|
||||
|
||||
|
||||
def check_retired_command(cmd: str) -> Optional[str]:
|
||||
"""Check if a command is retired and return the migration pointer, or None."""
|
||||
return RETIRED_COMMANDS.get(cmd)
|
||||
|
||||
|
||||
def check_retired_flag(flag: str) -> Optional[str]:
|
||||
"""Check if a flag is retired and return the migration pointer, or None."""
|
||||
return MIGRATION_POINTERS.get(flag)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Formatting
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _format_projection(proj, freshness: str, service: Dict[str, Any],
|
||||
drive_facts: List[str], config_error: Optional[str],
|
||||
store_fault: Optional[str], newer_schema: Optional[str],
|
||||
journal_hint: Optional[str]) -> str:
|
||||
"""Format the complete status output."""
|
||||
lines = []
|
||||
|
||||
# --- Store/system fault overrides (§9.4, §9.5) ---
|
||||
if store_fault:
|
||||
lines.append("observation store unreadable")
|
||||
if journal_hint:
|
||||
lines.append("")
|
||||
lines.append("Recent journal entries:")
|
||||
lines.append(journal_hint)
|
||||
return "\n".join(lines)
|
||||
|
||||
if newer_schema:
|
||||
lines.append("observation store written by a newer Fenris — upgrade Fenris")
|
||||
return "\n".join(lines)
|
||||
|
||||
# --- Configuration error (§8.3) ---
|
||||
if config_error:
|
||||
lines.append("configuration error: %s" % config_error)
|
||||
lines.append("")
|
||||
|
||||
# --- Empty store (§8.9) ---
|
||||
if freshness == "empty":
|
||||
lines.append("no observations yet")
|
||||
lines.append("")
|
||||
lines.append("Enable monitoring: fenris monitor resume")
|
||||
_append_service_facts(lines, service)
|
||||
return "\n".join(lines)
|
||||
|
||||
# --- Projection headline ---
|
||||
headline = _format_headline(proj)
|
||||
lines.append(headline)
|
||||
lines.append("")
|
||||
|
||||
# --- Confidence state + contributing facts (§6.7, §6.11) ---
|
||||
state_label = proj.confidence_state.value
|
||||
if proj.contributing_facts:
|
||||
facts_str = " · ".join(proj.contributing_facts)
|
||||
lines.append("%s evidence · %s" % (state_label, facts_str))
|
||||
else:
|
||||
lines.append("%s evidence" % state_label)
|
||||
lines.append("")
|
||||
|
||||
# --- Scenario range (§6.5) ---
|
||||
if proj.scenario_range and proj.scenario_range.rates:
|
||||
parts = []
|
||||
for horizon in sorted(proj.scenario_range.rates.keys()):
|
||||
rate_gb_day = proj.scenario_range.rates[horizon] * 86400 / 1e9
|
||||
parts.append("%dd: %.2f GB/day" % (horizon, rate_gb_day))
|
||||
lines.append("scenario range: %s" % " · ".join(parts))
|
||||
lines.append("")
|
||||
|
||||
# --- PU context line (§6.1) ---
|
||||
lines.append(proj.pu_context_line)
|
||||
lines.append("")
|
||||
|
||||
# --- Drive anomalies (§9.7, FL-7) ---
|
||||
if drive_facts:
|
||||
for fact in drive_facts:
|
||||
lines.append(fact)
|
||||
lines.append("")
|
||||
|
||||
# --- Four separate service facts (§7.3, LC-9) ---
|
||||
_append_service_facts(lines, service)
|
||||
|
||||
# --- Journal hint on failure or staleness (§8.8) ---
|
||||
if journal_hint:
|
||||
if freshness in ("missed", "stale"):
|
||||
lines.append("")
|
||||
lines.append("Recent journal entries:")
|
||||
lines.append(journal_hint)
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def _format_headline(proj) -> str:
|
||||
"""Format the lifespan headline or its no-projection wording (§6.11)."""
|
||||
if proj.headline_remaining_seconds is None:
|
||||
if proj.zero_rate_fact:
|
||||
return "no finite projection from this history"
|
||||
if proj.warming_fact:
|
||||
return proj.warming_fact
|
||||
return "no projection available"
|
||||
|
||||
secs = proj.headline_remaining_seconds
|
||||
if secs <= 0:
|
||||
return "endurance exhausted"
|
||||
|
||||
# Human-readable time
|
||||
years = int(secs // 31557600)
|
||||
rem = secs % 31557600
|
||||
days = int(rem // 86400)
|
||||
rem %= 86400
|
||||
hours = int(rem // 3600)
|
||||
|
||||
parts = []
|
||||
if years:
|
||||
parts.append("%d yr" % years)
|
||||
if days or years:
|
||||
parts.append("%d d" % days)
|
||||
parts.append("%d h" % hours)
|
||||
|
||||
remaining_human = " ".join(parts)
|
||||
|
||||
# Regime line
|
||||
regime_parts = []
|
||||
if proj.regime_days:
|
||||
regime_parts.append("sustained regime: %d days" % proj.regime_days)
|
||||
|
||||
headline = "%s remaining" % remaining_human
|
||||
if regime_parts:
|
||||
headline += " · %s" % " · ".join(regime_parts)
|
||||
|
||||
return headline
|
||||
|
||||
|
||||
def _append_service_facts(lines: List[str], service: Dict[str, Any]) -> None:
|
||||
"""Append service facts and dashboard-clarity monitoring state."""
|
||||
boot = "enabled" if service.get("boot_enabled") else "disabled"
|
||||
activity = "active" if service.get("timer_active") else "inactive"
|
||||
|
||||
if service.get("last_collect_ok") is True:
|
||||
collect = "ok"
|
||||
elif service.get("last_collect_ok") is False:
|
||||
collect = "FAILED"
|
||||
if service.get("last_collect_reason"):
|
||||
collect += " (%s)" % service["last_collect_reason"]
|
||||
else:
|
||||
collect = "unknown"
|
||||
|
||||
collect_age = ""
|
||||
if service.get("last_collect_age_s") is not None:
|
||||
collect_age = " %s" % freshness_age_human(service["last_collect_age_s"])
|
||||
|
||||
freshness_str = service.get("freshness", "unknown")
|
||||
freshness_age = ""
|
||||
if service.get("freshness_age_s") is not None:
|
||||
freshness_age = " (%s)" % freshness_age_human(service["freshness_age_s"])
|
||||
|
||||
lines.append("boot: %s · timer: %s · last collect: %s%s · freshness: %s%s"
|
||||
% (boot, activity, collect, collect_age, freshness_str, freshness_age))
|
||||
lines.append("CONTINUITY: %s" % monitoring_continuity(service))
|
||||
if service.get("deliberately_paused"):
|
||||
lines.extend(deliberate_pause_lines())
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Dashboard clarity parity wording (DC-2, DC-3)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_CONTINUITY_ACTIVE = "monitoring: active in background · persists across reboots"
|
||||
_CONTINUITY_DISABLED = "monitoring: does not start on next boot"
|
||||
_PAUSED_TITLE = "monitoring: paused — deliberate disable"
|
||||
_PAUSED_CONSEQUENCE = (
|
||||
"paused time is excluded from your usage habit · resume: fenris monitor resume"
|
||||
)
|
||||
|
||||
|
||||
def monitoring_continuity(service: Dict[str, Any]) -> str:
|
||||
"""Return the boot-persistence wording, independent of timer runtime."""
|
||||
return _CONTINUITY_ACTIVE if service.get("boot_enabled") else _CONTINUITY_DISABLED
|
||||
|
||||
|
||||
def deliberate_pause_lines() -> List[str]:
|
||||
"""Return the exact CLI/TUI presentation for a sanctioned pause."""
|
||||
return [_PAUSED_TITLE, _PAUSED_CONSEQUENCE]
|
||||
|
||||
|
||||
def is_deliberately_paused(conn: sqlite3.Connection, service: Dict[str, Any]) -> bool:
|
||||
"""Whether the latest closed period was ended by Fenris's own pause path.
|
||||
|
||||
Raw systemd operations have no `user_disabled` row, so they must never be
|
||||
presented as a Deliberate disable. A live enabled timer also wins over a
|
||||
stale period marker, keeping the presentation consistent with service facts.
|
||||
"""
|
||||
if service.get("boot_enabled") or service.get("timer_active"):
|
||||
return False
|
||||
|
||||
open_period = conn.execute(
|
||||
"SELECT 1 FROM monitoring_periods WHERE ended_at IS NULL LIMIT 1"
|
||||
).fetchone()
|
||||
if open_period is not None:
|
||||
return False
|
||||
|
||||
row = conn.execute(
|
||||
"SELECT end_cause FROM monitoring_periods "
|
||||
"WHERE ended_at IS NOT NULL "
|
||||
"ORDER BY ended_at DESC, id DESC LIMIT 1"
|
||||
).fetchone()
|
||||
return row is not None and row[0] == "user_disabled"
|
||||
|
||||
|
||||
def format_disclosures() -> str:
|
||||
"""Format the six disclosures (§6.11, CI-4)."""
|
||||
lines = []
|
||||
lines.append("Disclosures")
|
||||
lines.append("")
|
||||
for i, disc in enumerate(DISCLOSURES, 1):
|
||||
lines.append("%d. %s" % (i, disc))
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Main status entry point
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def get_status(store_path: Optional[Path] = None, clock_now: Optional[datetime] = None,
|
||||
query_services: bool = True, query_journal: bool = True) -> str:
|
||||
"""Render the complete read-only status (§8.8, LC-9).
|
||||
|
||||
This is the single entry point for 'fenris status'. It never auto-samples,
|
||||
never prompts, and never writes to the store.
|
||||
"""
|
||||
if clock_now is None:
|
||||
clock_now = datetime.now(timezone.utc)
|
||||
|
||||
# --- Configuration (§8.3) ---
|
||||
config_error = None
|
||||
device = None
|
||||
try:
|
||||
config = read_config()
|
||||
device = config["device"]
|
||||
except ConfigError as e:
|
||||
config_error = str(e)
|
||||
|
||||
# --- Service state ---
|
||||
service = {}
|
||||
if query_services:
|
||||
service = query_service_state()
|
||||
|
||||
# --- Store open ---
|
||||
store_fault = None
|
||||
newer_schema = None
|
||||
conn = None
|
||||
|
||||
if store_path is None:
|
||||
store_path = Path("/var/lib/fenris/observations.db")
|
||||
|
||||
try:
|
||||
conn = open_store_readonly(store_path)
|
||||
except StoreFault as e:
|
||||
store_fault = str(e)
|
||||
except NewerSchema as e:
|
||||
newer_schema = str(e)
|
||||
|
||||
# --- Store fault / newer schema short-circuit ---
|
||||
if store_fault or newer_schema:
|
||||
journal_hint = None
|
||||
if query_journal:
|
||||
journal_hint = _journalctl_hint("fenris-collect.service")
|
||||
service["freshness"] = "unknown"
|
||||
service["freshness_age_s"] = None
|
||||
return _format_projection(
|
||||
None, "unknown", service, [], config_error, store_fault, newer_schema, journal_hint
|
||||
)
|
||||
|
||||
# --- Freshness grading (§8.9) ---
|
||||
try:
|
||||
cursor = conn.execute("SELECT ts FROM samples ORDER BY id DESC LIMIT 1")
|
||||
row = cursor.fetchone()
|
||||
newest_ts = row[0] if row else None
|
||||
except sqlite3.Error:
|
||||
newest_ts = None
|
||||
|
||||
freshness = grade_freshness(newest_ts, clock_now)
|
||||
|
||||
# Freshness age for the service fact
|
||||
freshness_age_s = None
|
||||
if newest_ts:
|
||||
try:
|
||||
ts = datetime.fromisoformat(newest_ts)
|
||||
if ts.tzinfo is None:
|
||||
ts = ts.replace(tzinfo=timezone.utc)
|
||||
freshness_age_s = int((clock_now - ts).total_seconds())
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
service["freshness"] = freshness
|
||||
service["freshness_age_s"] = freshness_age_s
|
||||
try:
|
||||
service["deliberately_paused"] = is_deliberately_paused(conn, service)
|
||||
except sqlite3.Error:
|
||||
service["deliberately_paused"] = False
|
||||
|
||||
# --- Drive anomalies (§9.7, FL-7) ---
|
||||
drive_facts = []
|
||||
try:
|
||||
drive_facts = _query_drive_facts(conn)
|
||||
except sqlite3.Error:
|
||||
pass
|
||||
|
||||
# --- Projection (§6 — recomputed on read, never stored) ---
|
||||
try:
|
||||
proj = compute_projection(conn, clock_now)
|
||||
except Exception:
|
||||
proj = None
|
||||
|
||||
# --- Journal hint on failure or staleness (§8.8) ---
|
||||
journal_hint = None
|
||||
if query_journal and freshness in ("missed", "stale"):
|
||||
journal_hint = _journalctl_hint("fenris-collect.service")
|
||||
|
||||
# --- Compose output ---
|
||||
result = _format_projection(
|
||||
proj, freshness, service, drive_facts, config_error,
|
||||
None, None, journal_hint,
|
||||
)
|
||||
|
||||
conn.close()
|
||||
return result
|
||||
|
||||
|
||||
def render_status(store_path: Optional[Path] = None, clock_now: Optional[datetime] = None,
|
||||
query_services: bool = True, query_journal: bool = True,
|
||||
show_disclosures: bool = False) -> str:
|
||||
"""High-level status renderer: status + optional disclosures.
|
||||
|
||||
Used by the CLI entry point.
|
||||
"""
|
||||
parts = [get_status(store_path, clock_now, query_services, query_journal)]
|
||||
if show_disclosures:
|
||||
parts.append("")
|
||||
parts.append(format_disclosures())
|
||||
return "\n\n".join(parts)
|
||||
@@ -1,261 +0,0 @@
|
||||
"""Observation store: SQLite database for persisting observation history.
|
||||
|
||||
This module handles:
|
||||
- Store initialization with WAL mode
|
||||
- Schema versioning with PRAGMA user_version
|
||||
- The six entities: samples, hour_observations, day_aggregates,
|
||||
monitoring_periods, controller_segments, endurance_baseline
|
||||
"""
|
||||
import sqlite3
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
|
||||
# Schema version - increment on each migration
|
||||
SCHEMA_VERSION = 1
|
||||
|
||||
|
||||
# Packaged default placement (spec §8.3). The config may override it, but a
|
||||
# fresh install that sets only the device selector must collect cleanly.
|
||||
DEFAULT_STORE_PATH = Path("/var/lib/fenris/observations.db")
|
||||
|
||||
|
||||
def get_store_path(config: dict) -> Path:
|
||||
"""Get the store path from config.
|
||||
|
||||
Falls back to the packaged default when the config does not pin one,
|
||||
so a fresh install whose config holds only the device selector works
|
||||
instead of crashing with KeyError 'store_path' (issue #53).
|
||||
"""
|
||||
return Path(config.get("store_path", DEFAULT_STORE_PATH))
|
||||
|
||||
|
||||
def init_store(store_path: Path) -> sqlite3.Connection:
|
||||
"""Initialize the observation store if not present.
|
||||
|
||||
Creates the database with WAL mode and all six entities.
|
||||
Returns a connection to the store.
|
||||
"""
|
||||
conn = sqlite3.connect(str(store_path))
|
||||
|
||||
# Enable WAL mode for concurrent reads during writes
|
||||
conn.execute("PRAGMA journal_mode=WAL")
|
||||
|
||||
# Group members (fenris group) read the live store read-only, but SQLite
|
||||
# in WAL mode needs write access to the db and its -wal/-shm sidecars even
|
||||
# for readers. Best effort: root-created stores stay group-accessible
|
||||
# without relying on the creating process's umask (issue #54).
|
||||
import os as _os
|
||||
for sidecar in (store_path,
|
||||
store_path.with_name(store_path.name + "-wal"),
|
||||
store_path.with_name(store_path.name + "-shm")):
|
||||
try:
|
||||
mode = _os.stat(sidecar).st_mode & 0o777
|
||||
_os.chmod(sidecar, mode | 0o060)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
# Check if this is a new database
|
||||
cursor = conn.execute("PRAGMA user_version")
|
||||
current_version = cursor.fetchone()[0]
|
||||
|
||||
if current_version == 0:
|
||||
# New database - create schema
|
||||
_create_schema(conn)
|
||||
conn.execute(f"PRAGMA user_version={SCHEMA_VERSION}")
|
||||
conn.commit()
|
||||
elif current_version > SCHEMA_VERSION:
|
||||
# Unknown newer version - refuse
|
||||
conn.close()
|
||||
raise ValueError(
|
||||
f"Observation store written by a newer Fenris (version {current_version}) "
|
||||
f"— upgrade Fenris"
|
||||
)
|
||||
elif current_version < SCHEMA_VERSION:
|
||||
# Older version - apply migrations
|
||||
_apply_migrations(conn, current_version)
|
||||
conn.execute(f"PRAGMA user_version={SCHEMA_VERSION}")
|
||||
conn.commit()
|
||||
|
||||
return conn
|
||||
|
||||
|
||||
def _create_schema(conn: sqlite3.Connection):
|
||||
"""Create the initial schema with all six entities."""
|
||||
|
||||
# Samples: raw collection runs (14-day retention)
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS samples (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
ts TEXT NOT NULL, -- ISO 8601 UTC timestamp
|
||||
device TEXT NOT NULL,
|
||||
-- Normalized controller-identity fields captured at acquisition
|
||||
subnqn TEXT,
|
||||
sn TEXT,
|
||||
mn TEXT,
|
||||
fr TEXT,
|
||||
capacity_bytes INTEGER,
|
||||
percentage_used INTEGER,
|
||||
available_spare INTEGER,
|
||||
media_errors INTEGER,
|
||||
power_on_hours INTEGER,
|
||||
power_cycles INTEGER,
|
||||
unsafe_shutdowns INTEGER,
|
||||
temperature_c INTEGER,
|
||||
data_units_written INTEGER,
|
||||
data_units_read INTEGER,
|
||||
bytes_written INTEGER,
|
||||
bytes_read INTEGER,
|
||||
critical_warning INTEGER
|
||||
)
|
||||
""")
|
||||
|
||||
# Hour observations: UTC-hour usage-habit split
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS hour_observations (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
hour TEXT NOT NULL UNIQUE, -- ISO 8601 UTC hour (e.g., "2026-09-01T12:00:00Z")
|
||||
active_seconds INTEGER DEFAULT 0,
|
||||
idle_seconds INTEGER DEFAULT 0,
|
||||
powered_off_seconds INTEGER DEFAULT 0,
|
||||
unknown_seconds INTEGER DEFAULT 0,
|
||||
bytes_written_delta INTEGER DEFAULT 0,
|
||||
bytes_read_delta INTEGER DEFAULT 0,
|
||||
temperature_min INTEGER,
|
||||
temperature_avg REAL,
|
||||
temperature_max INTEGER,
|
||||
sample_count INTEGER DEFAULT 0,
|
||||
coverage REAL DEFAULT 0.0
|
||||
)
|
||||
""")
|
||||
|
||||
# Day aggregates: derived from hour observations
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS day_aggregates (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
day TEXT NOT NULL UNIQUE, -- ISO 8601 UTC day (e.g., "2026-09-01")
|
||||
active_seconds INTEGER DEFAULT 0,
|
||||
idle_seconds INTEGER DEFAULT 0,
|
||||
powered_off_seconds INTEGER DEFAULT 0,
|
||||
unknown_seconds INTEGER DEFAULT 0,
|
||||
bytes_written_delta INTEGER DEFAULT 0,
|
||||
bytes_read_delta INTEGER DEFAULT 0,
|
||||
sample_count INTEGER DEFAULT 0,
|
||||
coverage REAL DEFAULT 0.0
|
||||
)
|
||||
""")
|
||||
|
||||
# Monitoring periods: tracking when monitoring was enabled/disabled
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS monitoring_periods (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
started_at TEXT NOT NULL, -- ISO 8601 UTC timestamp
|
||||
ended_at TEXT, -- NULL if currently active
|
||||
end_cause TEXT CHECK(end_cause IN ('user_disabled', 'migrated', 'unknown_gap'))
|
||||
)
|
||||
""")
|
||||
|
||||
# Controller segments: identity key plus metadata snapshot
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS controller_segments (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
opened_at TEXT NOT NULL, -- ISO 8601 UTC timestamp
|
||||
identity_key TEXT, -- Normalized identity key (NULL if degraded)
|
||||
identity_degraded BOOLEAN DEFAULT 0,
|
||||
subnqn TEXT,
|
||||
sn TEXT,
|
||||
mn TEXT,
|
||||
fr TEXT,
|
||||
vid TEXT,
|
||||
ssvid TEXT,
|
||||
transport TEXT
|
||||
)
|
||||
""")
|
||||
|
||||
# Endurance baseline: one active row, replaced on edit
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS endurance_baseline (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
tbw_terabytes REAL NOT NULL,
|
||||
source_url TEXT,
|
||||
document_revision TEXT,
|
||||
entry_date TEXT,
|
||||
model_string TEXT,
|
||||
nominal_capacity_bytes INTEGER,
|
||||
validated_by TEXT, -- 'user' or 'machine_match'
|
||||
verified BOOLEAN DEFAULT 0,
|
||||
created_at TEXT NOT NULL,
|
||||
updated_at TEXT NOT NULL
|
||||
)
|
||||
""")
|
||||
|
||||
|
||||
# Metadata table for store state (e.g., legacy import marker)
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS store_metadata (
|
||||
key TEXT PRIMARY KEY,
|
||||
value TEXT NOT NULL
|
||||
)
|
||||
""")
|
||||
|
||||
|
||||
def _apply_migrations(conn: sqlite3.Connection, current_version: int):
|
||||
"""Apply forward-only migrations from current_version to SCHEMA_VERSION.
|
||||
|
||||
Each migration step is a transactional block. Add new steps as sequential
|
||||
elif branches when SCHEMA_VERSION increases.
|
||||
|
||||
Spec: §3.6, §10.2
|
||||
"""
|
||||
# Migration 1→2: example placeholder
|
||||
# if current_version < 2:
|
||||
# conn.execute("ALTER TABLE ...")
|
||||
# current_version = 2
|
||||
pass
|
||||
|
||||
|
||||
def migrate_to_latest(store_path: Path) -> int:
|
||||
"""Apply forward-only migrations to bring the store to SCHEMA_VERSION.
|
||||
|
||||
Called by the upgrade target (§10.2). Returns the number of migration
|
||||
steps applied. Raises ValueError on newer-schema store (§3.6, §9.5).
|
||||
|
||||
Spec: §3.6, §10.2, §10.3
|
||||
"""
|
||||
conn = sqlite3.connect(str(store_path))
|
||||
conn.execute("PRAGMA journal_mode=WAL")
|
||||
|
||||
cursor = conn.execute("PRAGMA user_version")
|
||||
current_version = cursor.fetchone()[0]
|
||||
|
||||
if current_version > SCHEMA_VERSION:
|
||||
conn.close()
|
||||
raise ValueError(
|
||||
f"Observation store written by a newer Fenris (version {current_version}) "
|
||||
f"— upgrade Fenris"
|
||||
)
|
||||
|
||||
if current_version == SCHEMA_VERSION:
|
||||
conn.close()
|
||||
return 0 # Already up to date
|
||||
|
||||
steps = SCHEMA_VERSION - current_version
|
||||
_apply_migrations(conn, current_version)
|
||||
conn.execute(f"PRAGMA user_version={SCHEMA_VERSION}")
|
||||
conn.commit()
|
||||
conn.close()
|
||||
return steps
|
||||
|
||||
|
||||
def is_store_faulty(store_path: Path) -> bool:
|
||||
"""Check if the store is present but cannot be read or trusted."""
|
||||
if not store_path.exists():
|
||||
return False
|
||||
|
||||
try:
|
||||
conn = sqlite3.connect(f"file:{store_path}?mode=ro", uri=True)
|
||||
conn.execute("PRAGMA user_version")
|
||||
conn.close()
|
||||
return False
|
||||
except sqlite3.Error:
|
||||
return True
|
||||
@@ -1,655 +0,0 @@
|
||||
"""Panes TUI: keyboard-first Textual app (spec §7, TUI-1, TUI-4).
|
||||
|
||||
One dense screen, four normative regions:
|
||||
1. Headline band (full width, top)
|
||||
2. Usage-history pane (left, wider)
|
||||
3. Drive-health + settings pane (right, narrower)
|
||||
4. Service strip (full width, bottom)
|
||||
|
||||
Bindings: p (pause, asks), r (resume), c (collect now), d (disclosures), q (quit).
|
||||
Privileged actions route through fenris-monitor as terminal-attached subprocesses
|
||||
(LC-6, LC-8). The TUI never samples in-process.
|
||||
|
||||
Criteria: TUI-1, TUI-2, TUI-4, CI-1, CI-2, CI-4, IN-3, LC-6, LC-8.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import sqlite3
|
||||
import subprocess
|
||||
import sys
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from textual.app import App, ComposeResult
|
||||
from textual.binding import Binding
|
||||
from textual.containers import Container, Horizontal, VerticalScroll
|
||||
from textual.screen import ModalScreen
|
||||
from textual.widgets import Static
|
||||
|
||||
from .projection import (
|
||||
ConfidenceState,
|
||||
ProjectionResult,
|
||||
ScenarioRange,
|
||||
compute_projection,
|
||||
)
|
||||
from .status import (
|
||||
CADENCE_DEFAULT_S,
|
||||
FRESH_THRESHOLD_S,
|
||||
STALENESS_THRESHOLD_S,
|
||||
ConfigError,
|
||||
NewerSchema,
|
||||
StoreFault,
|
||||
format_disclosures,
|
||||
freshness_age_human,
|
||||
grade_freshness,
|
||||
deliberate_pause_lines,
|
||||
is_deliberately_paused,
|
||||
monitoring_continuity,
|
||||
open_store_readonly,
|
||||
query_service_state,
|
||||
read_config,
|
||||
)
|
||||
from .monitoring_periods import get_open_period
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _format_remaining(seconds: float) -> str:
|
||||
"""Format remaining lifespan as human-readable string."""
|
||||
if seconds <= 0:
|
||||
return "endurance exhausted"
|
||||
years = int(seconds // 31557600)
|
||||
rem = seconds % 31557600
|
||||
days = int(rem // 86400)
|
||||
rem %= 86400
|
||||
hours = int(rem // 3600)
|
||||
parts = []
|
||||
if years:
|
||||
parts.append("%d yr" % years)
|
||||
if days or years:
|
||||
parts.append("%d d" % days)
|
||||
parts.append("%d h" % hours)
|
||||
return " ".join(parts)
|
||||
|
||||
|
||||
def _sparkline(values: List[float], width: int = 40) -> str:
|
||||
"""Render a sparkline from daily bytes-written values."""
|
||||
if not values:
|
||||
return ""
|
||||
mx = max(values) or 1.0
|
||||
blocks = " ▁▂▃▄▅▆▇█"
|
||||
step = max(1, len(values) // width or 1)
|
||||
picked = values[-width * step :][::step][-width:]
|
||||
return "".join(
|
||||
blocks[min(len(blocks) - 1, int(v / mx * (len(blocks) - 1)) + (1 if v > 0 else 0))]
|
||||
for v in picked
|
||||
)
|
||||
|
||||
|
||||
def _habit_bar(a: float, i: float, o: float, u: float, width: int = 40) -> str:
|
||||
"""Render the habit-split bar with legend."""
|
||||
total = a + i + o + u or 1.0
|
||||
segs = [
|
||||
("a", a, "#33ff33"),
|
||||
("i", i, "#ffff33"),
|
||||
("o", o, "#33ffff"),
|
||||
("?", u, "#ff33ff"),
|
||||
]
|
||||
parts = []
|
||||
for label, v, _color in segs:
|
||||
n = max(1 if v > 0 else 0, round(v / total * width))
|
||||
parts.append(label * n)
|
||||
bar = "".join(parts)
|
||||
legend = " active %d%% · idle %d%% · powered-off %d%% · unknown %d%%" % (
|
||||
round(a / total * 100),
|
||||
round(i / total * 100),
|
||||
round(o / total * 100),
|
||||
round(u / total * 100),
|
||||
)
|
||||
return bar + "\n" + legend
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Data queries for TUI regions
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _query_usage_history(conn: sqlite3.Connection) -> Dict[str, Any]:
|
||||
"""Query usage-history data for the left pane."""
|
||||
cursor = conn.execute(
|
||||
"SELECT day, active_seconds, idle_seconds, powered_off_seconds, "
|
||||
"unknown_seconds, bytes_written_delta, coverage "
|
||||
"FROM day_aggregates ORDER BY day"
|
||||
)
|
||||
days = cursor.fetchall()
|
||||
if not days:
|
||||
return {
|
||||
"sparkline": "",
|
||||
"habit_bar": "",
|
||||
"num_days": 0,
|
||||
"min_gb": 0,
|
||||
"max_gb": 0,
|
||||
"habit_change": False,
|
||||
"gap_days": [],
|
||||
"day_labels": [],
|
||||
}
|
||||
|
||||
bw_values = [d[5] for d in days]
|
||||
total_a = sum(d[1] for d in days)
|
||||
total_i = sum(d[2] for d in days)
|
||||
total_o = sum(d[3] for d in days)
|
||||
total_u = sum(d[4] for d in days)
|
||||
total = total_a + total_i + total_o + total_u or 1
|
||||
|
||||
spark = _sparkline([b / 1e9 for b in bw_values]) # Convert to GB for display
|
||||
bar = _habit_bar(total_a / total, total_i / total, total_o / total, total_u / total)
|
||||
|
||||
min_gb = min(bw_values) / 1e9 if bw_values else 0
|
||||
max_gb = max(bw_values) / 1e9 if bw_values else 0
|
||||
|
||||
return {
|
||||
"sparkline": spark,
|
||||
"habit_bar": bar,
|
||||
"num_days": len(days),
|
||||
"min_gb": min_gb,
|
||||
"max_gb": max_gb,
|
||||
"habit_change": False,
|
||||
"gap_days": [],
|
||||
"day_labels": [d[0] for d in days],
|
||||
}
|
||||
|
||||
|
||||
def _query_drive_health(conn: sqlite3.Connection) -> Dict[str, Any]:
|
||||
"""Query drive health data for the right pane."""
|
||||
cursor = conn.execute(
|
||||
"SELECT mn, sn, fr, temperature_c, available_spare, media_errors, "
|
||||
"power_on_hours, power_cycles, unsafe_shutdowns, capacity_bytes, "
|
||||
"percentage_used, data_units_written "
|
||||
"FROM samples ORDER BY id DESC LIMIT 1"
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
if row is None:
|
||||
return {
|
||||
"model": "unknown",
|
||||
"temp": 0,
|
||||
"spare": 0,
|
||||
"media_errors": 0,
|
||||
"poh": 0,
|
||||
"cycles": 0,
|
||||
"unsafe_shutdowns": 0,
|
||||
"capacity": "unknown",
|
||||
"percentage_used": 0,
|
||||
"written_tb": 0,
|
||||
}
|
||||
|
||||
capacity = row[9]
|
||||
capacity_str = "%d GB" % (capacity / 1e9) if capacity else "unknown"
|
||||
written_tb = (row[11] * 512 * 1000) / 1e12 if row[11] else 0 # DUW to TB
|
||||
|
||||
return {
|
||||
"model": row[0] or "unknown",
|
||||
"temp": row[3] or 0,
|
||||
"spare": row[4] or 0,
|
||||
"media_errors": row[5] or 0,
|
||||
"poh": row[6] or 0,
|
||||
"cycles": row[7] or 0,
|
||||
"unsafe_shutdowns": row[8] or 0,
|
||||
"capacity": capacity_str,
|
||||
"percentage_used": row[10] or 0,
|
||||
"written_tb": written_tb,
|
||||
}
|
||||
|
||||
|
||||
def _query_service_facts(conn: sqlite3.Connection, clock_now: datetime) -> Dict[str, Any]:
|
||||
"""Query service facts for the bottom strip."""
|
||||
# Get freshness
|
||||
cursor = conn.execute("SELECT ts FROM samples ORDER BY id DESC LIMIT 1")
|
||||
row = cursor.fetchone()
|
||||
newest_ts = row[0] if row else None
|
||||
freshness = grade_freshness(newest_ts, clock_now)
|
||||
|
||||
# Get monitoring period
|
||||
period = get_open_period(conn)
|
||||
period_info = "no monitoring period"
|
||||
if period:
|
||||
period_info = "open since %s" % period["started_at"][:10]
|
||||
|
||||
# Get service state
|
||||
try:
|
||||
svc = query_service_state()
|
||||
except Exception:
|
||||
svc = {
|
||||
"boot_enabled": False,
|
||||
"timer_active": False,
|
||||
"last_collect_ok": None,
|
||||
"last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}
|
||||
|
||||
return {
|
||||
"freshness": freshness,
|
||||
"freshness_age_s": None,
|
||||
"period": period_info,
|
||||
"deliberately_paused": is_deliberately_paused(conn, svc),
|
||||
**svc,
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Modal screens
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class ConfirmPause(ModalScreen[bool]):
|
||||
"""Pause asks for confirmation (spec §7.4, LC-6)."""
|
||||
|
||||
BINDINGS = [
|
||||
Binding("y", "yes", "Pause"),
|
||||
Binding("n", "no", "Cancel"),
|
||||
Binding("escape", "no", "Cancel", show=False),
|
||||
]
|
||||
|
||||
def compose(self) -> ComposeResult:
|
||||
yield Static(
|
||||
"[bold]Pause monitoring?[/bold]\n\n"
|
||||
"This closes the current monitoring period.\n"
|
||||
"Paused time is [bold]excluded[/bold] from your usage habit\n"
|
||||
"(powered-off time would still count).\n\n"
|
||||
"[dim]y pause · n cancel[/dim]",
|
||||
id="confirm-text",
|
||||
)
|
||||
|
||||
def action_yes(self) -> None:
|
||||
self.dismiss(True)
|
||||
|
||||
def action_no(self) -> None:
|
||||
self.dismiss(False)
|
||||
|
||||
|
||||
class DisclosuresScreen(ModalScreen[None]):
|
||||
"""Disclosures view with six verbatim disclosures (spec §6.11, CI-4)."""
|
||||
|
||||
BINDINGS = [
|
||||
Binding("escape", "close", "Close"),
|
||||
Binding("d", "close", "Close"),
|
||||
]
|
||||
|
||||
def compose(self) -> ComposeResult:
|
||||
text = format_disclosures()
|
||||
yield Static(text + "\n\n[dim]esc to close[/dim]", id="disc-text")
|
||||
|
||||
def action_close(self) -> None:
|
||||
self.dismiss()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Main TUI App
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class FenrisTuiApp(App):
|
||||
"""Fenris Panes TUI — keyboard-first, one dense screen (spec §7)."""
|
||||
|
||||
TITLE = "Fenris — NVMe endurance monitor"
|
||||
SUB_TITLE = ""
|
||||
|
||||
CSS = """
|
||||
#main-grid {
|
||||
layout: grid;
|
||||
grid-size: 2 4;
|
||||
grid-columns: 3fr 2fr;
|
||||
grid-rows: 8 10 7 3;
|
||||
height: auto;
|
||||
}
|
||||
#main-grid.paused {
|
||||
grid-size: 2 5;
|
||||
grid-rows: 8 5 10 7 3;
|
||||
}
|
||||
#dashboard-scroll { height: 1fr; }
|
||||
#headline-band { column-span: 2; }
|
||||
#paused-banner {
|
||||
column-span: 2;
|
||||
display: none;
|
||||
background: $error 20%;
|
||||
color: $text;
|
||||
height: 100%;
|
||||
}
|
||||
#service-strip { column-span: 2; height: 100%; }
|
||||
#quit-rail {
|
||||
column-span: 2;
|
||||
border: heavy $accent;
|
||||
content-align: center middle;
|
||||
height: 100%;
|
||||
}
|
||||
.pane { border: round #555555; padding: 0 1; height: 100%; }
|
||||
#confirm-text { padding: 1 2; }
|
||||
#disc-text { padding: 1 2; }
|
||||
"""
|
||||
|
||||
BINDINGS = [
|
||||
Binding("p", "pause", "Pause", show=False),
|
||||
Binding("r", "resume", "Resume", show=False),
|
||||
Binding("c", "collect", "Collect Now", show=False),
|
||||
Binding("d", "disclose", "Disclosures", show=False),
|
||||
Binding("q", "quit", "Quit", show=False),
|
||||
]
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
store_path: Optional[Path] = None,
|
||||
config_path: Optional[Path] = None,
|
||||
helper_path: Optional[str] = None,
|
||||
refresh_interval_s: float = CADENCE_DEFAULT_S,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
super().__init__(**kwargs)
|
||||
self.store_path = store_path or Path("/var/lib/fenris/observations.db")
|
||||
self.config_path = config_path
|
||||
self.helper_path = helper_path or "/usr/libexec/fenris/fenris-monitor"
|
||||
self.refresh_interval_s = refresh_interval_s
|
||||
self._show_auth_notice = True
|
||||
self._conn: Optional[sqlite3.Connection] = None
|
||||
self._clock_now = datetime.now(timezone.utc)
|
||||
|
||||
def compose(self) -> ComposeResult:
|
||||
with VerticalScroll(id="dashboard-scroll"):
|
||||
with Container(id="main-grid"):
|
||||
yield Static("", id="headline-band", classes="pane")
|
||||
yield Static("", id="paused-banner")
|
||||
yield Static("", id="usage-history", classes="pane")
|
||||
yield Static("", id="drive-health", classes="pane")
|
||||
yield Static("", id="service-strip", classes="pane")
|
||||
yield Static("q QUIT TUI", id="quit-rail")
|
||||
|
||||
def on_mount(self) -> None:
|
||||
"""Set border titles and render initial state."""
|
||||
self.query_one("#headline-band").border_title = "headline"
|
||||
self.query_one("#usage-history").border_title = "usage history"
|
||||
self.query_one("#drive-health").border_title = "drive"
|
||||
self.query_one("#service-strip").border_title = "service + actions"
|
||||
self._refresh_timer = self.set_interval(
|
||||
self.refresh_interval_s, self.on_refresh_tick
|
||||
)
|
||||
self._refresh()
|
||||
|
||||
def on_refresh_tick(self) -> None:
|
||||
"""Refresh dashboard and dismiss launch-only authentication guidance."""
|
||||
self._show_auth_notice = False
|
||||
self._refresh()
|
||||
|
||||
def _headline_prefix(self) -> str:
|
||||
"""Render identity and any launch-only guidance above drive state."""
|
||||
lines = ["[bold]Fenris — NVMe endurance monitor[/bold]"]
|
||||
if self._show_auth_notice:
|
||||
lines.append("[dim]privileged actions will prompt for authentication (polkit)[/dim]")
|
||||
return "\n".join(lines)
|
||||
|
||||
def _render_headline(self, body: str = "") -> None:
|
||||
"""Render full-width identity, guidance, and current drive state."""
|
||||
text = self._headline_prefix()
|
||||
if body:
|
||||
text += "\n\n" + body
|
||||
self.query_one("#headline-band").update(text)
|
||||
|
||||
def _open_store(self) -> Optional[sqlite3.Connection]:
|
||||
"""Open store read-only, handling faults."""
|
||||
try:
|
||||
return open_store_readonly(self.store_path)
|
||||
except (StoreFault, NewerSchema):
|
||||
return None
|
||||
|
||||
def _refresh(self) -> None:
|
||||
"""Refresh all four regions from store data."""
|
||||
self._clock_now = datetime.now(timezone.utc)
|
||||
conn = self._open_store()
|
||||
|
||||
if conn is None:
|
||||
self._render_empty_or_fault()
|
||||
return
|
||||
|
||||
try:
|
||||
self._conn = conn
|
||||
self._render_all_regions(conn)
|
||||
finally:
|
||||
conn.close()
|
||||
self._conn = None
|
||||
|
||||
def _render_empty_or_fault(self) -> None:
|
||||
"""Render empty store greeting or store fault."""
|
||||
self._hide_paused_banner()
|
||||
if not self.store_path.exists():
|
||||
# Empty store — greeting with enable hint (IN-3)
|
||||
self._render_headline(
|
||||
"[bold]No observations yet[/bold]\n\n"
|
||||
"Enable monitoring: fenris monitor resume"
|
||||
)
|
||||
self.query_one("#usage-history").update("")
|
||||
self.query_one("#drive-health").update("")
|
||||
self.query_one("#service-strip").update(
|
||||
"boot: disabled · timer: inactive · last collect: unknown · freshness: empty · "
|
||||
"[dim]by Bongbetic[/dim]\n"
|
||||
"[bold]CONTINUITY[/bold] %s\n"
|
||||
"p pause · r resume · c collect · d disclosures"
|
||||
% monitoring_continuity({"boot_enabled": False})
|
||||
)
|
||||
else:
|
||||
# Store fault (FL-4)
|
||||
self._render_headline(
|
||||
"[bold red]Observation store unreadable[/bold red]\n"
|
||||
"Check journalctl -u fenris-collect.service"
|
||||
)
|
||||
self.query_one("#usage-history").update("")
|
||||
self.query_one("#drive-health").update("")
|
||||
self.query_one("#service-strip").update(
|
||||
"[dim]by Bongbetic[/dim]\n"
|
||||
"p pause · r resume · c collect · d disclosures"
|
||||
)
|
||||
|
||||
def _render_all_regions(self, conn: sqlite3.Connection) -> None:
|
||||
"""Render all four regions from live store data."""
|
||||
# --- Headline band (§7.2) ---
|
||||
try:
|
||||
proj = compute_projection(conn, self._clock_now)
|
||||
headline = self._format_headline(proj)
|
||||
confidence = self._format_confidence(proj)
|
||||
scenario = self._format_scenario(proj)
|
||||
self._render_headline(headline + "\n" + confidence + "\n" + scenario)
|
||||
except Exception:
|
||||
self._render_headline("[bold]No projection available[/bold]")
|
||||
|
||||
# --- Usage-history pane (§7.2 left) ---
|
||||
history = _query_usage_history(conn)
|
||||
history_text = "[bold]Usage history[/bold] · %d days · %.1f–%.1f GB/day\n %s\n %s" % (
|
||||
history["num_days"],
|
||||
history["min_gb"],
|
||||
history["max_gb"],
|
||||
history["sparkline"],
|
||||
history["habit_bar"],
|
||||
)
|
||||
self.query_one("#usage-history").update(history_text)
|
||||
|
||||
# --- Drive-health pane (§7.2 right) ---
|
||||
health = _query_drive_health(conn)
|
||||
health_text = (
|
||||
"[bold]Drive health[/bold] · %s\n"
|
||||
" temperature %d°C · spare %d%%\n"
|
||||
" media errors %d · unsafe shutdowns %d\n"
|
||||
" power-on %d h · %d cycles · %s\n\n"
|
||||
"[bold]Settings[/bold]\n"
|
||||
" vendor wear: %d%% used · %.1f TB written"
|
||||
) % (
|
||||
health["model"],
|
||||
health["temp"],
|
||||
health["spare"],
|
||||
health["media_errors"],
|
||||
health["unsafe_shutdowns"],
|
||||
health["poh"],
|
||||
health["cycles"],
|
||||
health["capacity"],
|
||||
health["percentage_used"],
|
||||
health["written_tb"],
|
||||
)
|
||||
self.query_one("#drive-health").update(health_text)
|
||||
|
||||
# --- Service strip (§7.2 bottom) ---
|
||||
try:
|
||||
svc = _query_service_facts(conn, self._clock_now)
|
||||
boot = "enabled" if svc.get("boot_enabled") else "disabled"
|
||||
activity = "active" if svc.get("timer_active") else "inactive"
|
||||
collect = "ok" if svc.get("last_collect_ok") else "FAILED"
|
||||
freshness = svc.get("freshness", "unknown")
|
||||
self.query_one("#service-strip").update(
|
||||
"boot: %s · timer: %s · last collect: %s · freshness: %s · "
|
||||
"[dim]by Bongbetic[/dim]\n"
|
||||
"[bold]CONTINUITY[/bold] %s\n"
|
||||
"%s\n"
|
||||
"p pause · r resume · c collect · d disclosures"
|
||||
% (
|
||||
boot, activity, collect, freshness,
|
||||
monitoring_continuity(svc), svc.get("period", ""),
|
||||
)
|
||||
)
|
||||
self._render_paused_banner(svc)
|
||||
except Exception:
|
||||
self._hide_paused_banner()
|
||||
self.query_one("#service-strip").update(
|
||||
"boot: unknown · timer: unknown · last collect: unknown · freshness: unknown · "
|
||||
"[dim]by Bongbetic[/dim]\n"
|
||||
"p pause · r resume · c collect · d disclosures"
|
||||
)
|
||||
|
||||
def _render_paused_banner(self, service: Dict[str, Any]) -> None:
|
||||
"""Show the high-contrast Deliberate disable block only when sanctioned."""
|
||||
banner = self.query_one("#paused-banner")
|
||||
if service.get("deliberately_paused"):
|
||||
banner.update(
|
||||
"[bold black on red]%s[/bold black on red]\n%s"
|
||||
% tuple(deliberate_pause_lines())
|
||||
)
|
||||
banner.styles.display = "block"
|
||||
main_grid = self.query_one("#main-grid")
|
||||
main_grid.add_class("paused")
|
||||
main_grid.refresh(layout=True)
|
||||
else:
|
||||
self._hide_paused_banner()
|
||||
|
||||
def _hide_paused_banner(self) -> None:
|
||||
"""Ensure an unavailable store cannot retain a stale paused presentation."""
|
||||
self.query_one("#paused-banner").styles.display = "none"
|
||||
main_grid = self.query_one("#main-grid")
|
||||
main_grid.remove_class("paused")
|
||||
main_grid.refresh(layout=True)
|
||||
|
||||
def _format_headline(self, proj: ProjectionResult) -> str:
|
||||
"""Format the lifespan headline (spec §6.11)."""
|
||||
if proj.headline_remaining_seconds is None:
|
||||
if proj.zero_rate_fact:
|
||||
return "[bold]Usage-adjusted theoretical lifespan: [red]no finite projection from this history[/red][/bold]"
|
||||
if proj.warming_fact:
|
||||
return "[bold]Usage-adjusted theoretical lifespan: [yellow]%s[/yellow][/bold]" % proj.warming_fact
|
||||
return "[bold]Usage-adjusted theoretical lifespan: [red]no projection available[/red][/bold]"
|
||||
|
||||
remaining = _format_remaining(proj.headline_remaining_seconds)
|
||||
regime = ""
|
||||
if proj.regime_days:
|
||||
regime = " · sustained regime: %d days" % proj.regime_days
|
||||
return (
|
||||
"[bold]Usage-adjusted theoretical lifespan: [white]%s remaining[/white][/bold]"
|
||||
"\n if current habits continue%s" % (remaining, regime)
|
||||
)
|
||||
|
||||
def _format_confidence(self, proj: ProjectionResult) -> str:
|
||||
"""Format confidence state with contributing facts (spec §6.7)."""
|
||||
color = {
|
||||
ConfidenceState.SUPPORTED: "green",
|
||||
ConfidenceState.LIMITED: "yellow",
|
||||
ConfidenceState.UNSUPPORTED: "red",
|
||||
}[proj.confidence_state]
|
||||
facts = " · ".join(proj.contributing_facts[:3]) if proj.contributing_facts else "no facts"
|
||||
return "[bold]Projection confidence: [%s]%s[/%s][/bold]\n %s" % (
|
||||
color,
|
||||
proj.confidence_state.value,
|
||||
color,
|
||||
facts,
|
||||
)
|
||||
|
||||
def _format_scenario(self, proj: ProjectionResult) -> str:
|
||||
"""Format scenario range (spec §6.5)."""
|
||||
if not proj.scenario_range or not proj.scenario_range.rates:
|
||||
return ""
|
||||
parts = []
|
||||
for horizon in sorted(proj.scenario_range.rates.keys()):
|
||||
rate_gb_day = proj.scenario_range.rates[horizon] * 86400 / 1e9
|
||||
parts.append("%dd: %.2f GB/day" % (horizon, rate_gb_day))
|
||||
return "[bold]Scenario range[/bold] · %s" % " · ".join(parts)
|
||||
|
||||
# --- Actions ---
|
||||
|
||||
def action_pause(self) -> None:
|
||||
"""Pause monitoring — asks for confirmation (spec §7.4, LC-6)."""
|
||||
self.push_screen(ConfirmPause(), callback=self._pause_confirmed)
|
||||
|
||||
def _pause_confirmed(self, confirmed: bool) -> None:
|
||||
if not confirmed:
|
||||
return
|
||||
# Route through fenris-monitor as terminal-attached subprocess (LC-6)
|
||||
self._run_helper("disable", ["--now"])
|
||||
|
||||
def action_resume(self) -> None:
|
||||
"""Resume monitoring — no confirmation (spec §7.4, LC-6)."""
|
||||
self._run_helper("enable", ["--now"])
|
||||
|
||||
def action_collect(self) -> None:
|
||||
"""Collect now — synchronous outcome (spec §8.7, LC-8)."""
|
||||
self._run_helper("collect", blocking=True)
|
||||
|
||||
def action_disclose(self) -> None:
|
||||
"""Show disclosures (spec §6.11, CI-4)."""
|
||||
self.push_screen(DisclosuresScreen())
|
||||
|
||||
def _run_helper(
|
||||
self,
|
||||
operation: str,
|
||||
extra_args: Optional[List[str]] = None,
|
||||
blocking: bool = False,
|
||||
) -> None:
|
||||
"""Run fenris-monitor as terminal-attached subprocess (LC-6, LC-8).
|
||||
|
||||
The TUI suspends, polkit agent prompts on real terminal, control returns.
|
||||
"""
|
||||
cmd = [self.helper_path, operation]
|
||||
if extra_args:
|
||||
cmd.extend(extra_args)
|
||||
|
||||
try:
|
||||
with self.suspend():
|
||||
proc = subprocess.run(cmd, timeout=30)
|
||||
if proc.returncode != 0:
|
||||
self.notify(
|
||||
"Operation failed (exit %d)" % proc.returncode,
|
||||
severity="error",
|
||||
)
|
||||
except FileNotFoundError:
|
||||
self.notify(
|
||||
"Helper not found: %s" % self.helper_path,
|
||||
severity="error",
|
||||
)
|
||||
except subprocess.TimeoutExpired:
|
||||
self.notify("Operation timed out", severity="error")
|
||||
except Exception as e:
|
||||
self.notify("Error: %s" % e, severity="error")
|
||||
|
||||
# Refresh after action
|
||||
self._refresh()
|
||||
|
||||
|
||||
def run_tui(
|
||||
store_path: Optional[Path] = None,
|
||||
helper_path: Optional[str] = None,
|
||||
) -> None:
|
||||
"""Entry point for the Fenris TUI."""
|
||||
app = FenrisTuiApp(
|
||||
store_path=store_path,
|
||||
helper_path=helper_path,
|
||||
)
|
||||
app.run()
|
||||
@@ -1,20 +0,0 @@
|
||||
"""Shared test helpers for Fenris test suite."""
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
VERSION_FILE = REPO_ROOT / "pyproject.toml"
|
||||
|
||||
|
||||
def get_version() -> str:
|
||||
"""Extract version from pyproject.toml."""
|
||||
for line in VERSION_FILE.read_text().splitlines():
|
||||
if line.startswith("version"):
|
||||
return line.split("=")[1].strip().strip('"')
|
||||
raise RuntimeError("Could not determine version from pyproject.toml")
|
||||
|
||||
|
||||
def read(path: str | Path) -> str:
|
||||
"""Read a file relative to the repository root."""
|
||||
return (REPO_ROOT / path).read_text()
|
||||
@@ -1,749 +0,0 @@
|
||||
"""Cross-cutting acceptance sweep (issue #32).
|
||||
|
||||
Systematic verification of every acceptance criterion that spans multiple
|
||||
subsystems. Grouped by criterion ID; each test cites its clause.
|
||||
|
||||
CI-1 Exhaustive state matrix: confidence × freshness × baseline tier
|
||||
CI-2 TUI/CLI parity: identical outcomes and wording
|
||||
CI-3 Prohibition set: automated structural checks
|
||||
CI-4 Required wording and six disclosures in both views
|
||||
"""
|
||||
import re
|
||||
import sqlite3
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.store import init_store, SCHEMA_VERSION
|
||||
from fenris.monitoring_periods import ensure_period_open
|
||||
from fenris.projection import (
|
||||
compute_projection,
|
||||
ConfidenceState,
|
||||
BaselineTier,
|
||||
DISCLOSURES,
|
||||
STALENESS_HOURS,
|
||||
WARMING_MIN_DAYS,
|
||||
YOUNG_REGIME_DAYS,
|
||||
)
|
||||
from fenris.status import (
|
||||
grade_freshness,
|
||||
get_status,
|
||||
render_status,
|
||||
format_disclosures,
|
||||
FRESH_THRESHOLD_S,
|
||||
STALENESS_THRESHOLD_S,
|
||||
CADENCE_DEFAULT_S,
|
||||
ACCURACY_SEC,
|
||||
)
|
||||
from fenris.tui import (
|
||||
FenrisTuiApp,
|
||||
_format_remaining,
|
||||
)
|
||||
|
||||
|
||||
SRC_DIR = Path(__file__).parent.parent / "src"
|
||||
FENRIS_PKG = SRC_DIR / "fenris"
|
||||
|
||||
|
||||
def _clock(year=2026, month=9, day=30, hour=12):
|
||||
return datetime(year, month, day, hour, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
def _insert_baseline(conn, tbw_tb=1.0, verified=True,
|
||||
model="Samsung SSD 970 EVO Plus 1TB",
|
||||
source_url="https://example.com/spec",
|
||||
doc_rev="v1.0", entry_date="2026-01-01",
|
||||
nominal_cap=1024000000000):
|
||||
conn.execute(
|
||||
"INSERT INTO endurance_baseline "
|
||||
"(tbw_terabytes, source_url, document_revision, entry_date, model_string, "
|
||||
" nominal_capacity_bytes, validated_by, verified, created_at, updated_at) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(tbw_tb, source_url, doc_rev, entry_date, model, nominal_cap,
|
||||
"machine_match" if verified else None, verified,
|
||||
"2026-01-01T00:00:00+00:00", "2026-01-01T00:00:00+00:00"),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _insert_segment(conn, opened_at="2026-09-01T00:00:00+00:00",
|
||||
identity_key="nqn.test", degraded=False,
|
||||
mn="Samsung SSD 970 EVO Plus 1TB"):
|
||||
conn.execute(
|
||||
"INSERT INTO controller_segments "
|
||||
"(opened_at, identity_key, identity_degraded, subnqn, sn, mn, fr, vid, ssvid, transport) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(opened_at, identity_key, degraded, "nqn.test", "SN123", mn, "FW1",
|
||||
"0x144d", "0x144d", "pcie"),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _insert_day(conn, day, bw=1024*1024*100, coverage=0.95, samples=24):
|
||||
conn.execute(
|
||||
"INSERT INTO day_aggregates (day, active_seconds, idle_seconds, "
|
||||
"powered_off_seconds, unknown_seconds, bytes_written_delta, "
|
||||
"bytes_read_delta, sample_count, coverage) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(day, 3600, 0, 0, 0, bw, 0, samples, coverage),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _insert_sample(conn, ts, pu=5):
|
||||
conn.execute(
|
||||
"INSERT INTO samples (ts, device, data_units_written, data_units_read, "
|
||||
"percentage_used, bytes_written, bytes_read, power_on_hours) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(ts, "/dev/nvme0n1", 1000000, 500000, pu, 512000000000, 256000000000, 8765),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _open_period(conn, start="2026-09-01T00:00:00+00:00"):
|
||||
ensure_period_open(conn, datetime.fromisoformat(start))
|
||||
|
||||
|
||||
def _setup_full_store(conn, *, baseline=True, segment=True, days=30,
|
||||
bw=1024*1024*100, coverage=0.95, samples_per_day=24,
|
||||
sample_ts="2026-09-30T10:00:00+00:00",
|
||||
period_start="2026-09-01T00:00:00+00:00",
|
||||
segment_opened="2026-09-01T00:00:00+00:00",
|
||||
baseline_kw=None, segment_kw=None):
|
||||
if baseline:
|
||||
_insert_baseline(conn, **(baseline_kw or {}))
|
||||
if segment:
|
||||
_insert_segment(conn, opened_at=segment_opened, **(segment_kw or {}))
|
||||
_open_period(conn, start=period_start)
|
||||
for i in range(days):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=bw, coverage=coverage, samples=samples_per_day)
|
||||
if sample_ts:
|
||||
_insert_sample(conn, sample_ts)
|
||||
|
||||
|
||||
# ===================================================================
|
||||
# CI-1: Exhaustive state matrix
|
||||
# ===================================================================
|
||||
|
||||
|
||||
class TestCI1StateMatrix:
|
||||
"""Systematic walk of confidence x freshness x baseline tier combinations."""
|
||||
|
||||
def test_no_baseline_unavailable(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100, coverage=0.95, samples=24)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert proj.headline_remaining_seconds is None
|
||||
assert proj.baseline_tier == BaselineTier.NONE
|
||||
conn.close()
|
||||
|
||||
def test_verified_baseline_possible_supported(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_setup_full_store(conn, baseline_kw=dict(tbw_tb=10.0, verified=True))
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.SUPPORTED
|
||||
assert proj.baseline_tier == BaselineTier.VERIFIED
|
||||
assert proj.headline_remaining_seconds is not None
|
||||
conn.close()
|
||||
|
||||
def test_unverified_baseline_possible_limited(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_setup_full_store(conn, baseline_kw=dict(
|
||||
tbw_tb=10.0, verified=False, source_url=None))
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.baseline_tier == BaselineTier.UNVERIFIED
|
||||
assert proj.confidence_state != ConfidenceState.SUPPORTED
|
||||
conn.close()
|
||||
|
||||
def test_model_mismatch_unavailable(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_setup_full_store(conn, baseline_kw=dict(model="Different Model"))
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert proj.baseline_tier == BaselineTier.NONE
|
||||
conn.close()
|
||||
|
||||
def test_fresh_sample_grades_fresh(self, tmp_path):
|
||||
now = _clock()
|
||||
ts = (now - timedelta(seconds=FRESH_THRESHOLD_S - 10)).isoformat()
|
||||
assert grade_freshness(ts, now) == "fresh"
|
||||
|
||||
def test_missed_sample_grades_missed(self, tmp_path):
|
||||
now = _clock()
|
||||
ts = (now - timedelta(hours=2)).isoformat()
|
||||
assert grade_freshness(ts, now) == "missed"
|
||||
|
||||
def test_stale_sample_grades_stale(self, tmp_path):
|
||||
now = _clock()
|
||||
ts = (now - timedelta(hours=49)).isoformat()
|
||||
assert grade_freshness(ts, now) == "stale"
|
||||
|
||||
def test_empty_store_grades_empty(self, tmp_path):
|
||||
now = _clock()
|
||||
assert grade_freshness(None, now) == "empty"
|
||||
|
||||
def test_unsupported_fresh(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100)
|
||||
fresh_ts = (_clock() - timedelta(seconds=60)).isoformat()
|
||||
_insert_sample(conn, fresh_ts)
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert grade_freshness(fresh_ts, _clock()) == "fresh"
|
||||
conn.close()
|
||||
|
||||
def test_limited_young_regime(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_baseline(conn, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(5):
|
||||
d = (datetime(2026, 9, 25) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.LIMITED
|
||||
conn.close()
|
||||
|
||||
def test_limited_warming(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_baseline(conn, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(10):
|
||||
d = (datetime(2026, 9, 20) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100, coverage=0.95, samples=24)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.LIMITED
|
||||
assert proj.warming_fact is not None
|
||||
conn.close()
|
||||
|
||||
def test_limited_stale_data(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_baseline(conn, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(conn, opened_at="2026-08-01T00:00:00+00:00")
|
||||
_open_period(conn, start="2026-08-01T00:00:00+00:00")
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 8, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100, coverage=0.95, samples=24)
|
||||
stale_ts = (_clock() - timedelta(days=5)).isoformat()
|
||||
_insert_sample(conn, stale_ts)
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.LIMITED
|
||||
assert proj.staleness_fact is not None
|
||||
conn.close()
|
||||
|
||||
def test_limited_degraded_identity(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_baseline(conn, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(conn, identity_key=None, degraded=True)
|
||||
_open_period(conn)
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100, coverage=0.95, samples=24)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.LIMITED
|
||||
assert proj.degraded_identity_fact is not None
|
||||
conn.close()
|
||||
|
||||
def test_unsupported_zero_rate(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_baseline(conn, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=0)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert proj.zero_rate_fact is not None
|
||||
conn.close()
|
||||
|
||||
def test_headline_present_when_projection_exists(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_setup_full_store(conn, baseline_kw=dict(tbw_tb=10.0, verified=True))
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.headline_remaining_seconds is not None
|
||||
assert proj.headline_remaining_seconds > 0
|
||||
conn.close()
|
||||
|
||||
def test_headline_absent_when_unavailable(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.headline_remaining_seconds is None
|
||||
conn.close()
|
||||
|
||||
def test_headline_absent_when_zero_rate(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_baseline(conn, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=0)
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.headline_remaining_seconds is None
|
||||
conn.close()
|
||||
|
||||
def test_facts_always_list(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert isinstance(proj.contributing_facts, list)
|
||||
conn.close()
|
||||
|
||||
def test_facts_never_empty_for_unavailable(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert len(proj.contributing_facts) > 0
|
||||
conn.close()
|
||||
|
||||
def test_confidence_never_percentage(self, tmp_path):
|
||||
conn = init_store(tmp_path / "db")
|
||||
_setup_full_store(conn, baseline_kw=dict(tbw_tb=10.0, verified=True))
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state in (
|
||||
ConfidenceState.UNSUPPORTED, ConfidenceState.LIMITED, ConfidenceState.SUPPORTED)
|
||||
for f in proj.contributing_facts:
|
||||
if re.match(r"^\\d+%$", f.strip()):
|
||||
pytest.fail("Bare percentage in facts: %r" % f)
|
||||
conn.close()
|
||||
|
||||
def test_status_renders_same_state_as_projection(self, tmp_path):
|
||||
db = tmp_path / "observations.db"
|
||||
conn = init_store(db)
|
||||
_setup_full_store(conn, baseline_kw=dict(tbw_tb=10.0, verified=True))
|
||||
conn.close()
|
||||
now = _clock()
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": True, "timer_active": True,
|
||||
"last_collect_ok": True, "last_collect_age_s": 60,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
status = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
assert "Supported" in status or "supported" in status.lower()
|
||||
assert "remaining" in status.lower()
|
||||
|
||||
|
||||
# ===================================================================
|
||||
# CI-2: TUI/CLI parity
|
||||
# ===================================================================
|
||||
|
||||
|
||||
class TestCI2Parity:
|
||||
"""Verify TUI and CLI share the same constants, formatting, and wording."""
|
||||
|
||||
def test_freshness_constants_shared(self):
|
||||
from fenris import tui as tui_mod
|
||||
from fenris import status as status_mod
|
||||
assert tui_mod.FRESH_THRESHOLD_S == status_mod.FRESH_THRESHOLD_S
|
||||
assert tui_mod.STALENESS_THRESHOLD_S == status_mod.STALENESS_THRESHOLD_S
|
||||
|
||||
def test_grade_freshness_shared(self):
|
||||
from fenris.tui import grade_freshness as tui_gf
|
||||
from fenris.status import grade_freshness as status_gf
|
||||
assert tui_gf is status_gf
|
||||
|
||||
def test_disclosures_shared(self):
|
||||
from fenris.projection import DISCLOSURES as proj_disc
|
||||
from fenris.status import format_disclosures
|
||||
output = format_disclosures()
|
||||
for d in proj_disc:
|
||||
assert d in output
|
||||
|
||||
def test_status_four_facts_match_tui_strip(self, tmp_path):
|
||||
db = tmp_path / "observations.db"
|
||||
conn = init_store(db)
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
conn.close()
|
||||
now = _clock()
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": True, "timer_active": True,
|
||||
"last_collect_ok": True, "last_collect_age_s": 120,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
status = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
assert "boot:" in status
|
||||
assert "timer:" in status
|
||||
assert "last collect:" in status
|
||||
assert "freshness:" in status
|
||||
|
||||
def test_dashboard_clarity_parity_strings_have_one_status_source(self):
|
||||
"""DC-2/DC-3 wording originates in status and the TUI imports it."""
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
tui_src = (FENRIS_PKG / "tui.py").read_text()
|
||||
for wording in (
|
||||
"monitoring: active in background · persists across reboots",
|
||||
"monitoring: does not start on next boot",
|
||||
"monitoring: paused — deliberate disable",
|
||||
"paused time is excluded from your usage habit · resume: fenris monitor resume",
|
||||
):
|
||||
assert status_src.count(wording) == 1
|
||||
assert wording not in tui_src
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
("state", "service", "expected_lines"),
|
||||
[
|
||||
(
|
||||
"active_enabled",
|
||||
{"boot_enabled": True, "timer_active": True},
|
||||
["monitoring: active in background · persists across reboots"],
|
||||
),
|
||||
(
|
||||
"boot_disabled",
|
||||
{"boot_enabled": False, "timer_active": False},
|
||||
["monitoring: does not start on next boot"],
|
||||
),
|
||||
(
|
||||
"deliberately_paused",
|
||||
{"boot_enabled": False, "timer_active": False},
|
||||
[
|
||||
"monitoring: does not start on next boot",
|
||||
"monitoring: paused — deliberate disable",
|
||||
"paused time is excluded from your usage habit · resume: fenris monitor resume",
|
||||
],
|
||||
),
|
||||
],
|
||||
)
|
||||
async def test_dashboard_clarity_monitoring_lines_match_both_views(
|
||||
self, tmp_path, state, service, expected_lines
|
||||
):
|
||||
"""CI-2 synthetic-store sweep covers active, disabled, and paused states."""
|
||||
db = tmp_path / (state + ".db")
|
||||
conn = init_store(db)
|
||||
if state == "active_enabled":
|
||||
ensure_period_open(conn, _clock())
|
||||
elif state == "deliberately_paused":
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at, ended_at, end_cause) "
|
||||
"VALUES (?, ?, ?)",
|
||||
("2026-09-30T09:00:00+00:00", "2026-09-30T10:00:00+00:00", "user_disabled"),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
service_state = {
|
||||
**service,
|
||||
"last_collect_ok": None,
|
||||
"last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value=service_state), patch(
|
||||
"fenris.tui.query_service_state", return_value=service_state
|
||||
):
|
||||
status = get_status(
|
||||
store_path=db, clock_now=_clock(), query_services=True, query_journal=False
|
||||
).lower()
|
||||
app = FenrisTuiApp(store_path=db)
|
||||
async with app.run_test(size=(100, 40)):
|
||||
tui_text = "\n".join(
|
||||
(
|
||||
str(app.query_one("#service-strip").render()),
|
||||
str(app.query_one("#paused-banner").render()),
|
||||
)
|
||||
).lower()
|
||||
|
||||
for expected in expected_lines:
|
||||
assert expected in status
|
||||
assert expected in tui_text
|
||||
if state != "deliberately_paused":
|
||||
assert "monitoring: paused — deliberate disable" not in status
|
||||
assert "monitoring: paused — deliberate disable" not in tui_text
|
||||
|
||||
def test_pause_resume_action_names(self):
|
||||
tui_keys = {b.key for b in FenrisTuiApp.BINDINGS}
|
||||
assert "p" in tui_keys
|
||||
assert "r" in tui_keys
|
||||
assert "c" in tui_keys
|
||||
assert "q" in tui_keys
|
||||
|
||||
def test_empty_store_greeting_both_views(self, tmp_path):
|
||||
db = tmp_path / "observations.db"
|
||||
init_store(db)
|
||||
now = _clock()
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
status = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
assert "no observations yet" in status.lower()
|
||||
|
||||
def test_store_fault_phrase_both_views(self, tmp_path):
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
tui_src = (FENRIS_PKG / "tui.py").read_text()
|
||||
phrase = "observation store unreadable"
|
||||
assert phrase in status_src
|
||||
assert phrase.lower() in tui_src.lower()
|
||||
|
||||
def test_newer_schema_phrase_both_views(self):
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
phrase = "observation store written by a newer Fenris"
|
||||
assert phrase in status_src
|
||||
|
||||
def test_status_never_prompts(self):
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
assert "input(" not in status_src
|
||||
|
||||
|
||||
# ===================================================================
|
||||
# CI-3: Prohibition set
|
||||
# ===================================================================
|
||||
|
||||
|
||||
class TestCI3ProhibitionSet:
|
||||
"""Structural codebase checks for every prohibition clause."""
|
||||
|
||||
def _read_all_sources(self):
|
||||
files = {}
|
||||
for py in FENRIS_PKG.glob("*.py"):
|
||||
files[py.name] = py.read_text()
|
||||
return files
|
||||
|
||||
def test_single_acquisition_path(self):
|
||||
"""Only fenris-collect may interrogate the device. [2.1, 8.7]
|
||||
|
||||
collector.py contains the acquisition functions; collect.py is the
|
||||
fenris-collect entry point that invokes them. No other module may
|
||||
reference smartctl.
|
||||
"""
|
||||
sources = self._read_all_sources()
|
||||
allowed = {"collector.py", "collect.py"}
|
||||
for name, text in sources.items():
|
||||
if name in allowed:
|
||||
continue
|
||||
assert "smartctl" not in text, (
|
||||
"%s must not contain smartctl" % name
|
||||
)
|
||||
|
||||
def test_no_run_surface(self):
|
||||
"""No /run/fenris coordination surface. [1.2, 3]"""
|
||||
sources = self._read_all_sources()
|
||||
for name, text in sources.items():
|
||||
assert "/run/fenris" not in text, (
|
||||
"%s references /run/fenris" % name
|
||||
)
|
||||
|
||||
def test_single_config_key(self):
|
||||
"""Config holds exactly one key: device. [8.3]"""
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
in_read_config = False
|
||||
config_keys = []
|
||||
for line in status_src.split("\n"):
|
||||
if "def read_config" in line:
|
||||
in_read_config = True
|
||||
elif in_read_config and line.strip().startswith("def "):
|
||||
break
|
||||
elif in_read_config and "key ==" in line:
|
||||
match = re.search(r'key\s*==\s*["\']([^"\']+)["\']', line)
|
||||
if match:
|
||||
config_keys.append(match.group(1))
|
||||
assert "device" in config_keys
|
||||
assert len(config_keys) == 1, "Found keys: %s" % config_keys
|
||||
|
||||
def test_no_alerting_machinery(self):
|
||||
"""No alerting, notification, or escalation. [9.6]"""
|
||||
sources = self._read_all_sources()
|
||||
alert_keywords = ["send_email", "smtp", "webhook", "push_notification"]
|
||||
for name, text in sources.items():
|
||||
for kw in alert_keywords:
|
||||
for line in text.split("\n"):
|
||||
stripped = line.strip()
|
||||
if kw in stripped and not stripped.startswith("#"):
|
||||
pytest.fail(
|
||||
"%s contains alerting keyword '%s': %s" % (name, kw, stripped)
|
||||
)
|
||||
|
||||
def test_no_synthetic_baselines(self):
|
||||
"""No synthetic or capacity-derived baseline. [6.1]"""
|
||||
proj_src = (FENRIS_PKG / "projection.py").read_text()
|
||||
assert "synthetic" not in proj_src.lower()
|
||||
|
||||
def test_no_stored_projections(self):
|
||||
"""Projections never stored; recomputed on read. [3.7, 6.10]"""
|
||||
store_src = (FENRIS_PKG / "store.py").read_text()
|
||||
create_tables = re.findall(r"CREATE TABLE.*?(?=\n\n|$)", store_src, re.DOTALL)
|
||||
table_names = []
|
||||
for ct in create_tables:
|
||||
m = re.search(r"IF NOT EXISTS\s+(\w+)", ct)
|
||||
if m:
|
||||
table_names.append(m.group(1))
|
||||
assert "projection" not in [t.lower() for t in table_names]
|
||||
|
||||
def test_no_partial_newer_schema_interpretation(self):
|
||||
"""Readers refuse newer-schema stores. [3.6, 9.5]"""
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
assert "NewerSchema" in status_src
|
||||
assert "upgrade Fenris" in status_src
|
||||
|
||||
def test_polkit_authorizes_one_binary(self):
|
||||
"""Polkit authorizes exactly one binary: fenris-monitor. [8.5]"""
|
||||
monitor_src = (FENRIS_PKG / "monitor.py").read_text()
|
||||
assert "fenris-monitor" in monitor_src or "fenris_monitor" in monitor_src
|
||||
collect_src = (FENRIS_PKG / "collect.py").read_text()
|
||||
assert "polkit" not in collect_src.lower()
|
||||
|
||||
def test_no_hour_interpolation(self):
|
||||
"""No absent hour is interpolated or fabricated. [5.3]"""
|
||||
proj_src = (FENRIS_PKG / "projection.py").read_text()
|
||||
assert "interpolat" not in proj_src.lower()
|
||||
assert "fabricat" not in proj_src.lower()
|
||||
|
||||
def test_fenris_sh_not_shipped(self):
|
||||
"""fenris.sh is not shipped. [8.8]"""
|
||||
repo_root = Path(__file__).parent.parent
|
||||
assert not (repo_root / "fenris.sh").exists()
|
||||
|
||||
|
||||
# ===================================================================
|
||||
# CI-4: Required wording and six disclosures
|
||||
# ===================================================================
|
||||
|
||||
|
||||
class TestCI4WordingAndDisclosures:
|
||||
"""Verify exact fixed phrases and disclosures in both views."""
|
||||
|
||||
def test_exactly_six_disclosures(self):
|
||||
assert len(DISCLOSURES) == 6
|
||||
|
||||
def test_disclosure_1_endurance_not_failure(self):
|
||||
assert "endurance projection" in DISCLOSURES[0].lower()
|
||||
assert "hardware-failure" in DISCLOSURES[0].lower() or "failure date" in DISCLOSURES[0].lower()
|
||||
|
||||
def test_disclosure_2_vendor_specific(self):
|
||||
assert "vendor-specific" in DISCLOSURES[1]
|
||||
assert "255 is saturated" in DISCLOSURES[1]
|
||||
|
||||
def test_disclosure_3_warranty_not_failure(self):
|
||||
assert "warranty" in DISCLOSURES[2].lower() or "endurance threshold" in DISCLOSURES[2].lower()
|
||||
assert "failure threshold" in DISCLOSURES[2].lower()
|
||||
|
||||
def test_disclosure_4_duw_rounding(self):
|
||||
assert "DUW" in DISCLOSURES[3]
|
||||
assert "upward-rounded" in DISCLOSURES[3]
|
||||
assert "NAND" in DISCLOSURES[3]
|
||||
|
||||
def test_disclosure_5_quality_depends(self):
|
||||
assert "baseline provenance" in DISCLOSURES[4]
|
||||
assert "future workload" in DISCLOSURES[4]
|
||||
|
||||
def test_disclosure_6_gaps_and_disabled(self):
|
||||
assert "Gaps" in DISCLOSURES[5]
|
||||
assert "deliberately disabled" in DISCLOSURES[5]
|
||||
|
||||
def test_disclosures_render_in_status(self):
|
||||
output = format_disclosures()
|
||||
assert output.startswith("Disclosures")
|
||||
for i in range(1, 7):
|
||||
assert "%d." % i in output
|
||||
for d in DISCLOSURES:
|
||||
assert d in output
|
||||
|
||||
def test_disclosures_render_in_tui(self):
|
||||
tui_src = (FENRIS_PKG / "tui.py").read_text()
|
||||
assert "format_disclosures" in tui_src
|
||||
|
||||
def test_zero_rate_phrase(self):
|
||||
phrase = "no finite projection from this history"
|
||||
proj_src = (FENRIS_PKG / "projection.py").read_text()
|
||||
assert phrase in proj_src
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
assert phrase in status_src
|
||||
|
||||
def test_unavailable_no_baseline_phrase(self):
|
||||
phrase = "no applicable endurance baseline"
|
||||
proj_src = (FENRIS_PKG / "projection.py").read_text()
|
||||
assert phrase in proj_src
|
||||
|
||||
def test_store_fault_phrase(self):
|
||||
phrase = "observation store unreadable"
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
assert phrase in status_src
|
||||
tui_src = (FENRIS_PKG / "tui.py").read_text()
|
||||
assert phrase.lower() in tui_src.lower()
|
||||
|
||||
def test_newer_schema_phrase(self):
|
||||
phrase = "observation store written by a newer Fenris"
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
assert phrase in status_src
|
||||
|
||||
def test_no_observations_phrase(self):
|
||||
phrase = "no observations yet"
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
assert phrase in status_src
|
||||
tui_src = (FENRIS_PKG / "tui.py").read_text()
|
||||
assert phrase in tui_src.lower()
|
||||
|
||||
def test_config_error_phrase(self):
|
||||
phrase = "configuration error:"
|
||||
status_src = (FENRIS_PKG / "status.py").read_text()
|
||||
assert phrase in status_src
|
||||
|
||||
def test_degraded_identity_phrase(self):
|
||||
phrase = "controller identity unavailable"
|
||||
proj_src = (FENRIS_PKG / "projection.py").read_text()
|
||||
assert phrase in proj_src
|
||||
phrase2 = "replacement detection relies on write-counter continuity only"
|
||||
assert phrase2 in proj_src
|
||||
|
||||
def test_scenario_range_only_spread(self):
|
||||
proj_src = (FENRIS_PKG / "projection.py").read_text()
|
||||
assert "confidence interval" not in proj_src.lower()
|
||||
|
||||
def test_no_percentage_in_confidence_rendering(self):
|
||||
for name in ["projection.py", "tui.py", "status.py"]:
|
||||
src = (FENRIS_PKG / name).read_text()
|
||||
assert not re.search(r"\\d+%\\s*confidence", src, re.IGNORECASE), (
|
||||
"Found XX%% confidence in %s" % name
|
||||
)
|
||||
|
||||
def test_status_disclosures_accessible(self):
|
||||
db = Path("/tmp/_ci4_test.db")
|
||||
conn = init_store(db)
|
||||
conn.close()
|
||||
now = _clock()
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = render_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False,
|
||||
show_disclosures=True)
|
||||
assert "Disclosures" in result
|
||||
assert "1." in result
|
||||
assert "6." in result
|
||||
db.unlink(missing_ok=True)
|
||||
@@ -1,211 +0,0 @@
|
||||
"""Release-note changelog extraction tests (DC-6, DC-7)."""
|
||||
import importlib.util
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
EXTRACTOR_PATH = REPO_ROOT / "scripts" / "extract_changelog.py"
|
||||
RELEASE_REQUEST_PATH = REPO_ROOT / "scripts" / "release_request.py"
|
||||
CHANGELOG_PATH = REPO_ROOT / "CHANGELOG.md"
|
||||
|
||||
|
||||
def _extractor_module():
|
||||
spec = importlib.util.spec_from_file_location("extract_changelog", EXTRACTOR_PATH)
|
||||
assert spec and spec.loader
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
sys.modules[spec.name] = module
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def test_extracts_the_requested_version_section_verbatim():
|
||||
extractor = _extractor_module()
|
||||
changelog = """# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
## [1.4.0] - 2026-09-10
|
||||
|
||||
### Added
|
||||
|
||||
- Show a release summary to consumers.
|
||||
|
||||
## [1.3.0] - 2026-09-01
|
||||
|
||||
### Fixed
|
||||
|
||||
- Preserve the observation history during upgrades.
|
||||
"""
|
||||
expected = """## [1.4.0] - 2026-09-10
|
||||
|
||||
### Added
|
||||
|
||||
- Show a release summary to consumers.
|
||||
|
||||
"""
|
||||
|
||||
assert extractor.extract_version_section(changelog, "1.4.0") == expected
|
||||
|
||||
|
||||
def test_checked_in_changelog_keeps_unreleased_first_and_categories_limited():
|
||||
lines = CHANGELOG_PATH.read_text(encoding="utf-8").splitlines()
|
||||
unreleased = lines.index("## [Unreleased]")
|
||||
version_headings = [
|
||||
index for index, line in enumerate(lines)
|
||||
if line.startswith("## [") and line != "## [Unreleased]"
|
||||
]
|
||||
first_version = version_headings[0] if version_headings else len(lines)
|
||||
categories = [
|
||||
line.removeprefix("### ")
|
||||
for line in lines[unreleased + 1:first_version]
|
||||
if line.startswith("### ")
|
||||
]
|
||||
|
||||
assert unreleased < first_version
|
||||
assert set(categories) <= {"Added", "Changed", "Fixed"}
|
||||
|
||||
|
||||
def test_release_footer_verifies_the_clearsigned_checksum_asset():
|
||||
footer = (REPO_ROOT / "packaging" / "release-footer.md").read_text(
|
||||
encoding="utf-8"
|
||||
)
|
||||
|
||||
assert "gpg --output SHA256SUMS --decrypt SHA256SUMS.asc" in footer
|
||||
assert "sha256sum -c SHA256SUMS" in footer
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("changelog", "expected_error"),
|
||||
[
|
||||
("# Changelog\n\n## [Unreleased]\n", "missing"),
|
||||
(
|
||||
"# Changelog\n\n## [Unreleased]\n\n## [1.4.0] - 2026-09-10\n",
|
||||
"empty",
|
||||
),
|
||||
(
|
||||
"# Changelog\n\n## [Unreleased]\n\n## [1.4.0] - 2026-02-30\n\n- Add a note.\n",
|
||||
"malformed release date",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_fails_closed_for_missing_empty_or_malformed_sections(
|
||||
changelog, expected_error
|
||||
):
|
||||
extractor = _extractor_module()
|
||||
|
||||
with pytest.raises(extractor.ChangelogError, match=expected_error):
|
||||
extractor.extract_version_section(changelog, "1.4.0")
|
||||
|
||||
|
||||
def test_command_emits_a_workflow_error_and_nonzero_status(tmp_path):
|
||||
changelog = tmp_path / "CHANGELOG.md"
|
||||
changelog.write_text("# Changelog\n\n## [Unreleased]\n", encoding="utf-8")
|
||||
|
||||
result = subprocess.run(
|
||||
[sys.executable, str(EXTRACTOR_PATH), str(changelog), "1.4.0"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
|
||||
assert result.returncode != 0
|
||||
assert result.stderr.startswith("::error::")
|
||||
assert "missing" in result.stderr
|
||||
|
||||
|
||||
def test_assembles_a_release_body_without_changing_the_section():
|
||||
extractor = _extractor_module()
|
||||
section = "## [1.4.0] - 2026-09-10\n\n### Added\n\n- Show a release summary.\n"
|
||||
footer = "## Install\n\nUse the package channel.\n"
|
||||
|
||||
assert extractor.assemble_release_body(section, footer) == (
|
||||
section + "\n" + footer
|
||||
)
|
||||
|
||||
|
||||
def test_command_can_write_the_complete_release_body(tmp_path):
|
||||
changelog = tmp_path / "CHANGELOG.md"
|
||||
changelog.write_text(
|
||||
"# Changelog\n\n## [Unreleased]\n\n## [1.4.0] - 2026-09-10\n\n"
|
||||
"### Added\n\n- Show a release summary.\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
footer = tmp_path / "footer.md"
|
||||
footer.write_text("## Install\n\nUse the package channel.\n", encoding="utf-8")
|
||||
|
||||
result = subprocess.run(
|
||||
[
|
||||
sys.executable,
|
||||
str(EXTRACTOR_PATH),
|
||||
str(changelog),
|
||||
"1.4.0",
|
||||
"--footer",
|
||||
str(footer),
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
|
||||
assert result.returncode == 0
|
||||
assert result.stdout == (
|
||||
"## [1.4.0] - 2026-09-10\n\n### Added\n\n- Show a release summary.\n\n"
|
||||
"## Install\n\nUse the package channel.\n"
|
||||
)
|
||||
|
||||
|
||||
def test_release_request_command_reports_create_or_patch_decisions(tmp_path):
|
||||
body = tmp_path / "release-body.md"
|
||||
body.write_text("## [1.4.0] - 2026-09-10\n", encoding="utf-8")
|
||||
|
||||
create = subprocess.run(
|
||||
[
|
||||
sys.executable,
|
||||
str(RELEASE_REQUEST_PATH),
|
||||
"--version",
|
||||
"1.4.0",
|
||||
"--body-file",
|
||||
str(body),
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
existing = tmp_path / "existing-release.json"
|
||||
existing.write_text('{"id": 17, "assets": []}', encoding="utf-8")
|
||||
patch = subprocess.run(
|
||||
[
|
||||
sys.executable,
|
||||
str(RELEASE_REQUEST_PATH),
|
||||
"--version",
|
||||
"1.4.0",
|
||||
"--body-file",
|
||||
str(body),
|
||||
"--existing-release",
|
||||
str(existing),
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
|
||||
assert create.returncode == patch.returncode == 0
|
||||
assert json.loads(create.stdout) == {
|
||||
"method": "POST",
|
||||
"path": "/releases",
|
||||
"payload": {
|
||||
"tag_name": "v1.4.0",
|
||||
"name": "v1.4.0",
|
||||
"body": "## [1.4.0] - 2026-09-10\n",
|
||||
},
|
||||
}
|
||||
assert json.loads(patch.stdout) == {
|
||||
"method": "PATCH",
|
||||
"path": "/releases/17",
|
||||
"payload": {"body": "## [1.4.0] - 2026-09-10\n"},
|
||||
}
|
||||
@@ -1,359 +0,0 @@
|
||||
"""Collector tracer bullet test.
|
||||
|
||||
Tests the thinnest complete write path through the system:
|
||||
- Input: smartctl-JSON fixture, sysfs fixture tree, config fixture, injected clock
|
||||
- Output: resulting store contents, run outcome
|
||||
|
||||
Seam: write side of the observation store database file.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import sqlite3
|
||||
import tempfile
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Generator
|
||||
|
||||
import pytest
|
||||
|
||||
# Add src to path for imports
|
||||
import sys
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.collector import run_collection, AcquisitionError, InvariantViolationError
|
||||
from fenris.store import init_store, get_store_path
|
||||
|
||||
|
||||
# Fixtures
|
||||
|
||||
@pytest.fixture
|
||||
def smartctl_fixture() -> Dict[str, Any]:
|
||||
"""Minimal smartctl -a -j output with required fields."""
|
||||
return {
|
||||
"json_format_version": [1, 0],
|
||||
"smartctl": {"version": [7, 3], "svn_revision": "5155", "build_info": "(local build)"},
|
||||
"nvme_smart_health_information_log": {
|
||||
"critical_warning": 0,
|
||||
"temperature": 35,
|
||||
"available_spare": 100,
|
||||
"available_spare_threshold": 10,
|
||||
"percentage_used": 5,
|
||||
"data_units_written": 12345678,
|
||||
"data_units_read": 9876543,
|
||||
"power_on_hours": 8765,
|
||||
"power_cycles": 1234,
|
||||
"unsafe_shutdowns": 5,
|
||||
"media_errors": 0,
|
||||
"num_err_log_entries": 0,
|
||||
},
|
||||
"user_capacity": {"bytes": 1024000000000, "units": "bytes"},
|
||||
"model_name": "Samsung SSD 970 EVO Plus 1TB",
|
||||
"serial_number": "S4EWNX0N123456",
|
||||
"firmware_version": "2B2QEXM7",
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sysfs_fixture_tree(tmp_path: Path) -> Path:
|
||||
"""Create a minimal sysfs fixture tree with controller identity."""
|
||||
ctrl_dir = tmp_path / "sys" / "class" / "nvme" / "nvme0"
|
||||
ctrl_dir.mkdir(parents=True)
|
||||
|
||||
# Controller identity files
|
||||
(ctrl_dir / "subsysnqn").write_text("nqn.2014-08.org.nvmexpress:uuid:12345678-1234-1234-1234-123456789abc\n")
|
||||
(ctrl_dir / "model").write_text("Samsung SSD 970 EVO Plus 1TB\n")
|
||||
(ctrl_dir / "serial").write_text("S4EWNX0N123456\n")
|
||||
(ctrl_dir / "firmware_rev").write_text("2B2QEXM7\n")
|
||||
|
||||
# Transport info (optional, but we'll include it)
|
||||
transport_dir = ctrl_dir / "transport"
|
||||
transport_dir.mkdir()
|
||||
(transport_dir / "address").write_text("0000:03:00.0")
|
||||
(transport_dir / "trstring").write_text("pcie")
|
||||
|
||||
return tmp_path
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def config_fixture(tmp_path: Path) -> Dict[str, Any]:
|
||||
"""Configuration fixture naming the device."""
|
||||
return {
|
||||
"device": "/dev/nvme0",
|
||||
"store_path": str(tmp_path / "observations.db"),
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def clock_fixture():
|
||||
"""Injected clock returning fixed time."""
|
||||
class FakeClock:
|
||||
def __init__(self):
|
||||
self.now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
def utcnow(self):
|
||||
return self.now
|
||||
|
||||
return FakeClock()
|
||||
|
||||
|
||||
# Test: Collector writes one well-formed sample
|
||||
|
||||
def test_collector_writes_one_sample(
|
||||
smartctl_fixture: Dict[str, Any],
|
||||
sysfs_fixture_tree: Path,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given a smartctl fixture and sysfs fixture tree,
|
||||
when the collector runs,
|
||||
then one well-formed sample is written to the observation store."""
|
||||
result = run_collection(
|
||||
smartctl_data=smartctl_fixture,
|
||||
sysfs_path=sysfs_fixture_tree / "sys" / "class" / "nvme" / "nvme0",
|
||||
config=config_fixture,
|
||||
clock=clock_fixture,
|
||||
)
|
||||
|
||||
# Verify run succeeded
|
||||
assert result["ok"] is True, f"Collection failed: {result.get('error')}"
|
||||
|
||||
# Verify store contents
|
||||
conn = sqlite3.connect(config_fixture["store_path"])
|
||||
cursor = conn.execute("SELECT COUNT(*) FROM samples")
|
||||
count = cursor.fetchone()[0]
|
||||
assert count == 1
|
||||
|
||||
cursor = conn.execute("SELECT * FROM samples")
|
||||
row = cursor.fetchone()
|
||||
assert row is not None
|
||||
|
||||
# Verify row contents match fixtures
|
||||
# Row structure: id, ts, device, subnqn, sn, mn, fr, capacity_bytes,
|
||||
# percentage_used, available_spare, media_errors, power_on_hours, power_cycles,
|
||||
# unsafe_shutdowns, temperature_c, data_units_written, data_units_read,
|
||||
# bytes_written, bytes_read, critical_warning
|
||||
assert row[2] == "/dev/nvme0" # device
|
||||
assert row[3] == "nqn.2014-08.org.nvmexpress:uuid:12345678-1234-1234-1234-123456789abc" # subnqn
|
||||
assert row[4] == "S4EWNX0N123456" # sn
|
||||
assert row[5] == "Samsung SSD 970 EVO Plus 1TB" # mn
|
||||
assert row[6] == "2B2QEXM7" # fr
|
||||
assert row[7] == 1024000000000 # capacity_bytes
|
||||
assert row[8] == 5 # percentage_used
|
||||
assert row[15] == 12345678 # data_units_written
|
||||
assert row[17] == 12345678 * 512000 # bytes_written
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: Identity normalization applied exactly once at write time
|
||||
|
||||
def test_identity_normalization(
|
||||
smartctl_fixture: Dict[str, Any],
|
||||
sysfs_fixture_tree: Path,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given sysfs identity with trailing spaces/newlines,
|
||||
when the collector writes,
|
||||
then identity is normalized exactly once at write time."""
|
||||
# Create identity with trailing whitespace
|
||||
identity = {
|
||||
"subnqn": "nqn.2014-08.org.nvmexpress:uuid:12345678-1234-1234-1234-123456789abc \n",
|
||||
"mn": "Samsung SSD 970 EVO Plus 1TB\n",
|
||||
"sn": "S4EWNX0N123456\n",
|
||||
"fr": "2B2QEXM7",
|
||||
"transport": "pcie",
|
||||
}
|
||||
|
||||
from fenris.collector import normalize_identity
|
||||
|
||||
# Normalize once
|
||||
key1 = normalize_identity(identity)
|
||||
# Normalize again - should be identical
|
||||
key2 = normalize_identity(identity)
|
||||
|
||||
assert key1 == key2
|
||||
assert key1 == "nqn.2014-08.org.nvmexpress:uuid:12345678-1234-1234-1234-123456789abc"
|
||||
assert "\n" not in key1
|
||||
assert key1 == key1.rstrip() # No trailing whitespace
|
||||
|
||||
|
||||
# Test: Any acquisition failure fails the whole run
|
||||
|
||||
def test_acquisition_failure_fails_run(
|
||||
sysfs_fixture_tree: Path,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given a smartctl fixture with missing fields,
|
||||
when the collector runs,
|
||||
then the whole run fails and writes nothing."""
|
||||
# Missing required field
|
||||
bad_smartctl = {
|
||||
"json_format_version": [1, 0],
|
||||
"smartctl": {"version": [7, 3]},
|
||||
# Missing nvme_smart_health_information_log
|
||||
}
|
||||
|
||||
result = run_collection(
|
||||
smartctl_data=bad_smartctl,
|
||||
sysfs_path=sysfs_fixture_tree / "sys" / "class" / "nvme" / "nvme0",
|
||||
config=config_fixture,
|
||||
clock=clock_fixture,
|
||||
)
|
||||
|
||||
# Verify run failed
|
||||
assert result["ok"] is False
|
||||
assert "Missing required field" in result["error"]
|
||||
|
||||
# Verify nothing was written
|
||||
if os.path.exists(config_fixture["store_path"]):
|
||||
conn = sqlite3.connect(config_fixture["store_path"])
|
||||
cursor = conn.execute("SELECT COUNT(*) FROM samples")
|
||||
count = cursor.fetchone()[0]
|
||||
assert count == 0
|
||||
conn.close()
|
||||
else:
|
||||
# Store wasn't even created - also valid
|
||||
pass
|
||||
|
||||
|
||||
# Test: Store initializes with six entities
|
||||
|
||||
def test_store_initialization(config_fixture: Dict[str, Any]):
|
||||
"""Given no existing store,
|
||||
when the collector runs,
|
||||
then the store is initialized with six entities."""
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
|
||||
# Store shouldn't exist yet
|
||||
assert not store_path.exists()
|
||||
|
||||
# Initialize store
|
||||
conn = init_store(store_path)
|
||||
|
||||
# Verify all six entities exist
|
||||
cursor = conn.execute("SELECT name FROM sqlite_master WHERE type='table'")
|
||||
tables = {row[0] for row in cursor.fetchall()}
|
||||
|
||||
expected_tables = {
|
||||
"samples",
|
||||
"hour_observations",
|
||||
"day_aggregates",
|
||||
"monitoring_periods",
|
||||
"controller_segments",
|
||||
"endurance_baseline",
|
||||
}
|
||||
|
||||
# sqlite_sequence is a system table created by AUTOINCREMENT
|
||||
expected_tables.add("sqlite_sequence")
|
||||
expected_tables.add("store_metadata")
|
||||
assert expected_tables == tables
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: Schema versioning with PRAGMA user_version
|
||||
|
||||
def test_schema_versioning(config_fixture: Dict[str, Any]):
|
||||
"""Given a store with unknown newer version,
|
||||
when the collector runs,
|
||||
then it refuses to proceed."""
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
|
||||
# Create a store with newer version
|
||||
conn = sqlite3.connect(str(store_path))
|
||||
conn.execute("PRAGMA journal_mode=WAL")
|
||||
conn.execute("PRAGMA user_version=999") # Unknown newer version
|
||||
conn.close()
|
||||
|
||||
# Try to initialize - should fail
|
||||
with pytest.raises(ValueError, match="newer Fenris"):
|
||||
init_store(store_path)
|
||||
|
||||
|
||||
def test_schema_version_current(config_fixture: Dict[str, Any]):
|
||||
"""Given a store with current version,
|
||||
when the collector runs,
|
||||
then it proceeds without migration."""
|
||||
from fenris.store import SCHEMA_VERSION
|
||||
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
|
||||
# Initialize store
|
||||
conn1 = init_store(store_path)
|
||||
conn1.close()
|
||||
|
||||
# Open again - should succeed
|
||||
conn2 = init_store(store_path)
|
||||
|
||||
# Verify version is current
|
||||
cursor = conn2.execute("PRAGMA user_version")
|
||||
version = cursor.fetchone()[0]
|
||||
assert version == SCHEMA_VERSION
|
||||
|
||||
conn2.close()
|
||||
|
||||
|
||||
def test_schema_version_older(config_fixture: Dict[str, Any]):
|
||||
"""Given a store with older version,
|
||||
when the collector runs,
|
||||
then it applies migrations and proceeds."""
|
||||
from fenris.store import SCHEMA_VERSION
|
||||
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
|
||||
# Create a store with older version
|
||||
conn = sqlite3.connect(str(store_path))
|
||||
conn.execute("PRAGMA journal_mode=WAL")
|
||||
conn.execute("PRAGMA user_version=0") # Older version
|
||||
conn.close()
|
||||
|
||||
# Initialize store - should apply migrations
|
||||
conn = init_store(store_path)
|
||||
|
||||
# Verify version is current
|
||||
cursor = conn.execute("PRAGMA user_version")
|
||||
version = cursor.fetchone()[0]
|
||||
assert version == SCHEMA_VERSION
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: Invariant-violating run writes nothing
|
||||
|
||||
def test_invariant_violation_writes_nothing(
|
||||
smartctl_fixture: Dict[str, Any],
|
||||
sysfs_fixture_tree: Path,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given a smartctl fixture that would violate store invariants,
|
||||
when the collector runs,
|
||||
then it writes nothing and fails visibly."""
|
||||
# Create a fixture that would cause invariant violation
|
||||
# (negative bytes_written - we'll mock this)
|
||||
bad_smartctl = smartctl_fixture.copy()
|
||||
bad_smartctl["nvme_smart_health_information_log"] = {
|
||||
**smartctl_fixture["nvme_smart_health_information_log"],
|
||||
"data_units_written": -1, # This will cause negative bytes_written
|
||||
}
|
||||
|
||||
result = run_collection(
|
||||
smartctl_data=bad_smartctl,
|
||||
sysfs_path=sysfs_fixture_tree / "sys" / "class" / "nvme" / "nvme0",
|
||||
config=config_fixture,
|
||||
clock=clock_fixture,
|
||||
)
|
||||
|
||||
# Verify run failed
|
||||
assert result["ok"] is False
|
||||
assert "InvariantViolation" in result["error_type"]
|
||||
|
||||
# Verify nothing was written
|
||||
if os.path.exists(config_fixture["store_path"]):
|
||||
conn = sqlite3.connect(config_fixture["store_path"])
|
||||
cursor = conn.execute("SELECT COUNT(*) FROM samples")
|
||||
count = cursor.fetchone()[0]
|
||||
assert count == 0
|
||||
conn.close()
|
||||
@@ -1,234 +0,0 @@
|
||||
"""Day aggregate tests.
|
||||
|
||||
From spec §5.4, §3.3, ST-4, ST-5:
|
||||
- One row per UTC day, derived monotonically from hour rows
|
||||
- No 23/25-hour days (UTC-bounded, DST never applies)
|
||||
- Coverage: share of wall-clock seconds inside monitoring periods
|
||||
whose classification is known
|
||||
- No absent hour ever interpolated/estimated/fabricated (FL-3)
|
||||
"""
|
||||
import sqlite3
|
||||
import sys
|
||||
from datetime import datetime, timezone, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.store import init_store
|
||||
from fenris.monitoring_periods import ensure_period_open, close_period
|
||||
from fenris.day_aggregate import derive_day, derive_all_days, DayAggregate
|
||||
from fenris.hour_classify import HourSplit, ACTIVE_THRESHOLD_BYTES
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store_conn(tmp_path: Path):
|
||||
db_path = tmp_path / "test.db"
|
||||
conn = init_store(db_path)
|
||||
yield conn
|
||||
conn.close()
|
||||
|
||||
|
||||
def _insert_hour(conn, hour_iso, split, bytes_written_delta=0, bytes_read_delta=0,
|
||||
sample_count=1, coverage=1.0):
|
||||
conn.execute(
|
||||
"INSERT INTO hour_observations "
|
||||
"(hour, active_seconds, idle_seconds, powered_off_seconds, unknown_seconds, "
|
||||
" bytes_written_delta, bytes_read_delta, sample_count, coverage) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(
|
||||
hour_iso,
|
||||
split.seconds_active,
|
||||
split.seconds_idle,
|
||||
split.seconds_powered_off,
|
||||
split.seconds_unknown,
|
||||
bytes_written_delta,
|
||||
bytes_read_delta,
|
||||
sample_count,
|
||||
coverage,
|
||||
),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
class TestDeriveDay:
|
||||
"""Spec §5.4: One row per UTC day, derived monotonically from hour rows."""
|
||||
|
||||
def test_single_hour_day(self, store_conn):
|
||||
t_start = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
t_end = datetime(2026, 9, 1, 13, 0, 0, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, t_start)
|
||||
close_period(store_conn, t_end, "user_disabled")
|
||||
|
||||
hour = "2026-09-01T12:00:00+00:00"
|
||||
split = HourSplit(seconds_active=3600, seconds_idle=0, seconds_powered_off=0, seconds_unknown=0)
|
||||
_insert_hour(store_conn, hour, split, bytes_written_delta=1024*1024*100)
|
||||
|
||||
day = derive_day(store_conn, "2026-09-01")
|
||||
assert day is not None
|
||||
assert day.day == "2026-09-01"
|
||||
assert day.seconds_active == 3600
|
||||
assert day.seconds_idle == 0
|
||||
assert day.seconds_powered_off == 0
|
||||
assert day.bytes_written_delta == 1024*1024*100
|
||||
assert day.sample_count == 1
|
||||
|
||||
def test_multiple_hours(self, store_conn):
|
||||
t_start = datetime(2026, 9, 1, 10, 0, 0, tzinfo=timezone.utc)
|
||||
t_end = datetime(2026, 9, 1, 13, 0, 0, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, t_start)
|
||||
close_period(store_conn, t_end, "user_disabled")
|
||||
|
||||
hours = [
|
||||
("2026-09-01T10:00:00+00:00", HourSplit(1800, 1800, 0, 0), 50*1024*1024),
|
||||
("2026-09-01T11:00:00+00:00", HourSplit(3600, 0, 0, 0), 200*1024*1024),
|
||||
("2026-09-01T12:00:00+00:00", HourSplit(0, 0, 3600, 0), 0),
|
||||
]
|
||||
for h, s, bw in hours:
|
||||
_insert_hour(store_conn, h, s, bytes_written_delta=bw)
|
||||
|
||||
day = derive_day(store_conn, "2026-09-01")
|
||||
assert day is not None
|
||||
assert day.seconds_active == 1800 + 3600 + 0
|
||||
assert day.seconds_idle == 1800 + 0 + 0
|
||||
assert day.seconds_powered_off == 0 + 0 + 3600
|
||||
assert day.bytes_written_delta == 50*1024*1024 + 200*1024*1024 + 0
|
||||
assert day.sample_count == 3
|
||||
|
||||
def test_no_hours_returns_none(self, store_conn):
|
||||
day = derive_day(store_conn, "2026-09-01")
|
||||
assert day is None
|
||||
|
||||
|
||||
class TestUtcBounded:
|
||||
"""Spec §3.3, ST-4: Hours and days are UTC-bounded. No 23/25-hour days."""
|
||||
|
||||
def test_day_keys_are_utc_date_strings(self, store_conn):
|
||||
t_start = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
t_end = datetime(2026, 9, 1, 13, 0, 0, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, t_start)
|
||||
close_period(store_conn, t_end, "user_disabled")
|
||||
|
||||
hour = "2026-09-01T12:00:00+00:00"
|
||||
split = HourSplit(0, 0, 3600, 0)
|
||||
_insert_hour(store_conn, hour, split)
|
||||
|
||||
day = derive_day(store_conn, "2026-09-01")
|
||||
assert day is not None
|
||||
assert day.day == "2026-09-01"
|
||||
|
||||
def test_cross_midnight_hours_produce_two_days(self, store_conn):
|
||||
# Period spans both days
|
||||
t_start = datetime(2026, 9, 1, 23, 0, 0, tzinfo=timezone.utc)
|
||||
t_end = datetime(2026, 9, 2, 1, 0, 0, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, t_start)
|
||||
close_period(store_conn, t_end, "user_disabled")
|
||||
|
||||
hours = [
|
||||
("2026-09-01T23:00:00+00:00", HourSplit(3600, 0, 0, 0), 100),
|
||||
("2026-09-02T00:00:00+00:00", HourSplit(0, 0, 3600, 0), 0),
|
||||
]
|
||||
for h, s, bw in hours:
|
||||
_insert_hour(store_conn, h, s, bytes_written_delta=bw)
|
||||
|
||||
days = derive_all_days(store_conn)
|
||||
day_keys = [d.day for d in days]
|
||||
assert "2026-09-01" in day_keys
|
||||
assert "2026-09-02" in day_keys
|
||||
assert len(days) == 2
|
||||
|
||||
|
||||
class TestCoverage:
|
||||
"""Spec §5.2, §5.3, PR-5: Coverage is known share of wall-clock seconds
|
||||
inside monitoring periods."""
|
||||
|
||||
def test_full_coverage_all_known(self, store_conn):
|
||||
"""All hours in period classified -> coverage = 1.0."""
|
||||
t_start = datetime(2026, 9, 1, 0, 0, 0, tzinfo=timezone.utc)
|
||||
t_end = datetime(2026, 9, 1, 3, 0, 0, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, t_start)
|
||||
close_period(store_conn, t_end, "user_disabled")
|
||||
|
||||
hours = [
|
||||
("2026-09-01T00:00:00+00:00", HourSplit(3600, 0, 0, 0), 100),
|
||||
("2026-09-01T01:00:00+00:00", HourSplit(0, 3600, 0, 0), 50),
|
||||
("2026-09-01T02:00:00+00:00", HourSplit(0, 0, 3600, 0), 0),
|
||||
]
|
||||
for h, s, bw in hours:
|
||||
_insert_hour(store_conn, h, s, bytes_written_delta=bw)
|
||||
|
||||
day = derive_day(store_conn, "2026-09-01")
|
||||
assert day is not None
|
||||
assert day.coverage == pytest.approx(1.0)
|
||||
|
||||
def test_gap_reduces_coverage(self, store_conn):
|
||||
"""Missing hour -> unknown seconds reduce coverage."""
|
||||
t_start = datetime(2026, 9, 1, 0, 0, 0, tzinfo=timezone.utc)
|
||||
t_end = datetime(2026, 9, 1, 3, 0, 0, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, t_start)
|
||||
close_period(store_conn, t_end, "user_disabled")
|
||||
|
||||
_insert_hour(store_conn, "2026-09-01T00:00:00+00:00",
|
||||
HourSplit(3600, 0, 0, 0), bytes_written_delta=100)
|
||||
_insert_hour(store_conn, "2026-09-01T02:00:00+00:00",
|
||||
HourSplit(0, 3600, 0, 0), bytes_written_delta=50)
|
||||
|
||||
day = derive_day(store_conn, "2026-09-01")
|
||||
assert day is not None
|
||||
assert day.coverage == pytest.approx(7200 / 10800)
|
||||
assert day.seconds_unknown == 3600
|
||||
|
||||
def test_outside_period_excluded(self, store_conn):
|
||||
"""Hours outside any monitoring period excluded from denominator."""
|
||||
t_start = datetime(2026, 9, 1, 1, 0, 0, tzinfo=timezone.utc)
|
||||
t_end = datetime(2026, 9, 1, 2, 0, 0, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, t_start)
|
||||
close_period(store_conn, t_end, "user_disabled")
|
||||
|
||||
# Hour 01 is inside the period
|
||||
_insert_hour(store_conn, "2026-09-01T01:00:00+00:00",
|
||||
HourSplit(3600, 0, 0, 0), bytes_written_delta=100)
|
||||
# Hour 02 is outside (period ends at 02:00)
|
||||
_insert_hour(store_conn, "2026-09-01T02:00:00+00:00",
|
||||
HourSplit(0, 3600, 0, 0), bytes_written_delta=50)
|
||||
|
||||
day = derive_day(store_conn, "2026-09-01")
|
||||
assert day is not None
|
||||
assert day.coverage == pytest.approx(1.0)
|
||||
|
||||
def test_unknown_in_period_reduces_coverage(self, store_conn):
|
||||
"""Unknown seconds inside period count in denominator but not numerator."""
|
||||
t_start = datetime(2026, 9, 1, 0, 0, 0, tzinfo=timezone.utc)
|
||||
t_end = datetime(2026, 9, 1, 1, 0, 0, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, t_start)
|
||||
close_period(store_conn, t_end, "user_disabled")
|
||||
|
||||
_insert_hour(store_conn, "2026-09-01T00:00:00+00:00",
|
||||
HourSplit(1800, 0, 0, 1800), bytes_written_delta=100)
|
||||
|
||||
day = derive_day(store_conn, "2026-09-01")
|
||||
assert day is not None
|
||||
assert day.coverage == pytest.approx(0.5)
|
||||
assert day.seconds_unknown == 1800
|
||||
|
||||
|
||||
class TestNoFabrication:
|
||||
"""Spec §5.3, FL-3: No absent hour is ever interpolated/estimated/fabricated."""
|
||||
|
||||
def test_missing_hours_stay_unknown(self, store_conn):
|
||||
"""Gap hours are never filled in — they remain as unknown seconds."""
|
||||
t_start = datetime(2026, 9, 1, 0, 0, 0, tzinfo=timezone.utc)
|
||||
t_end = datetime(2026, 9, 1, 3, 0, 0, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, t_start)
|
||||
close_period(store_conn, t_end, "user_disabled")
|
||||
|
||||
# Only hour 02 — hours 00 and 01 are gaps
|
||||
_insert_hour(store_conn, "2026-09-01T02:00:00+00:00",
|
||||
HourSplit(3600, 0, 0, 0), bytes_written_delta=100)
|
||||
|
||||
day = derive_day(store_conn, "2026-09-01")
|
||||
assert day is not None
|
||||
assert day.seconds_unknown == 2 * 3600
|
||||
assert day.seconds_active == 3600
|
||||
assert day.sample_count == 1
|
||||
@@ -1,218 +0,0 @@
|
||||
"""Hour classification tests.
|
||||
|
||||
From spec §5.1 — each UTC hour is classified by named constants:
|
||||
- Powered-off: power-on-hours delta < 90% of wall-clock span
|
||||
- Active: DUW delta >= 256 MiB
|
||||
- Idle: powered on + sampled + below active threshold
|
||||
- Unknown: everything else
|
||||
|
||||
Four splits sum to exactly 3600s. Disabled time is never an hour state.
|
||||
"""
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.hour_classify import classify_hour, HourSplit, ACTIVE_THRESHOLD_BYTES
|
||||
|
||||
HOUR_SECONDS = 3600
|
||||
|
||||
|
||||
class TestPoweredOff:
|
||||
"""Spec §5.1: Powered-off when power-on-hours delta < 90% of wall-clock span."""
|
||||
|
||||
def test_below_90_percent_poh_is_powered_off(self):
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=0,
|
||||
duw_delta=0,
|
||||
dur_delta=0,
|
||||
)
|
||||
assert split.seconds_powered_off == HOUR_SECONDS
|
||||
assert split.seconds_active == 0
|
||||
assert split.seconds_idle == 0
|
||||
assert split.seconds_unknown == 0
|
||||
|
||||
def test_exactly_90_percent_poh_is_not_powered_off(self):
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=3240,
|
||||
duw_delta=0,
|
||||
dur_delta=0,
|
||||
)
|
||||
assert split.seconds_powered_off == 0
|
||||
assert split.seconds_unknown == HOUR_SECONDS
|
||||
|
||||
def test_just_below_90_percent_poh_is_powered_off(self):
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=3239,
|
||||
duw_delta=0,
|
||||
dur_delta=0,
|
||||
)
|
||||
assert split.seconds_powered_off == HOUR_SECONDS
|
||||
|
||||
def test_100_percent_poh_is_not_powered_off(self):
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=HOUR_SECONDS,
|
||||
duw_delta=0,
|
||||
dur_delta=0,
|
||||
)
|
||||
assert split.seconds_powered_off == 0
|
||||
|
||||
|
||||
class TestActive:
|
||||
"""Spec §5.1: Active when DUW delta >= 256 MiB."""
|
||||
|
||||
def test_above_256_mib_is_active(self):
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=HOUR_SECONDS,
|
||||
duw_delta=ACTIVE_THRESHOLD_BYTES,
|
||||
dur_delta=1000,
|
||||
sampled_seconds=HOUR_SECONDS,
|
||||
)
|
||||
assert split.seconds_active == HOUR_SECONDS
|
||||
|
||||
def test_just_below_256_mib_is_idle(self):
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=HOUR_SECONDS,
|
||||
duw_delta=ACTIVE_THRESHOLD_BYTES - 1,
|
||||
dur_delta=1000,
|
||||
sampled_seconds=HOUR_SECONDS,
|
||||
)
|
||||
assert split.seconds_idle == HOUR_SECONDS
|
||||
assert split.seconds_active == 0
|
||||
|
||||
def test_powered_off_takes_priority_over_active_writes(self):
|
||||
"""Spec §5.1 order: powered-off checked first."""
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=0,
|
||||
duw_delta=ACTIVE_THRESHOLD_BYTES,
|
||||
dur_delta=1000,
|
||||
)
|
||||
assert split.seconds_powered_off == HOUR_SECONDS
|
||||
assert split.seconds_active == 0
|
||||
|
||||
|
||||
class TestIdle:
|
||||
"""Spec §5.1: Idle when powered on, sampled, below active threshold."""
|
||||
|
||||
def test_idle_with_writes_below_threshold(self):
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=HOUR_SECONDS,
|
||||
duw_delta=1024 * 1024, # 1 MiB
|
||||
dur_delta=500,
|
||||
sampled_seconds=HOUR_SECONDS,
|
||||
)
|
||||
assert split.seconds_idle == HOUR_SECONDS
|
||||
assert split.seconds_active == 0
|
||||
|
||||
def test_idle_no_writes(self):
|
||||
"""Powered on, sampled, zero writes -> idle."""
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=HOUR_SECONDS,
|
||||
duw_delta=0,
|
||||
dur_delta=0,
|
||||
sampled_seconds=HOUR_SECONDS,
|
||||
)
|
||||
assert split.seconds_idle == HOUR_SECONDS
|
||||
|
||||
|
||||
class TestUnknown:
|
||||
"""Spec §5.1: Unknown when unsampled without POH evidence."""
|
||||
|
||||
def test_no_sample_no_writes_is_unknown(self):
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=HOUR_SECONDS,
|
||||
duw_delta=0,
|
||||
dur_delta=0,
|
||||
)
|
||||
assert split.seconds_unknown == HOUR_SECONDS
|
||||
|
||||
def test_partial_unknown(self):
|
||||
"""Partial hour -> split unknown for unsampled portion."""
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=HOUR_SECONDS,
|
||||
duw_delta=0,
|
||||
dur_delta=0,
|
||||
sampled_seconds=1800,
|
||||
)
|
||||
assert split.seconds_unknown == 1800
|
||||
assert split.seconds_idle == 1800
|
||||
|
||||
def test_powered_off_with_no_sample_still_powered_off(self):
|
||||
"""POH < 90% is powered off regardless of sample status."""
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=0,
|
||||
duw_delta=0,
|
||||
dur_delta=0,
|
||||
sampled_seconds=0,
|
||||
)
|
||||
assert split.seconds_powered_off == HOUR_SECONDS
|
||||
assert split.seconds_unknown == 0
|
||||
|
||||
|
||||
class TestSumTo3600:
|
||||
"""Spec §5.1: Four splits sum to exactly wall_clock_seconds."""
|
||||
|
||||
def test_all_scenarios_sum_to_wall_clock(self):
|
||||
scenarios = [
|
||||
{"wall_clock_seconds": 3600, "poh_delta": 0, "duw_delta": 0, "dur_delta": 0},
|
||||
{"wall_clock_seconds": 3600, "poh_delta": 3600, "duw_delta": 0, "dur_delta": 0},
|
||||
{"wall_clock_seconds": 3600, "poh_delta": 3600, "duw_delta": ACTIVE_THRESHOLD_BYTES, "dur_delta": 1000, "sampled_seconds": 3600},
|
||||
{"wall_clock_seconds": 3600, "poh_delta": 3600, "duw_delta": 1000, "dur_delta": 500, "sampled_seconds": 3600},
|
||||
{"wall_clock_seconds": 3600, "poh_delta": 3239, "duw_delta": 0, "dur_delta": 0},
|
||||
{"wall_clock_seconds": 1800, "poh_delta": 0, "duw_delta": 0, "dur_delta": 0},
|
||||
{"wall_clock_seconds": 3600, "poh_delta": 3600, "duw_delta": 0, "dur_delta": 0, "sampled_seconds": 1800},
|
||||
]
|
||||
for s in scenarios:
|
||||
split = classify_hour(**s)
|
||||
total = (
|
||||
split.seconds_active
|
||||
+ split.seconds_idle
|
||||
+ split.seconds_powered_off
|
||||
+ split.seconds_unknown
|
||||
)
|
||||
assert total == s["wall_clock_seconds"], (
|
||||
f"Scenario {s}: splits sum to {total}, expected {s['wall_clock_seconds']}"
|
||||
)
|
||||
|
||||
def test_partial_wall_clock(self):
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=1800,
|
||||
poh_delta=0,
|
||||
duw_delta=0,
|
||||
dur_delta=0,
|
||||
)
|
||||
total = (
|
||||
split.seconds_active
|
||||
+ split.seconds_idle
|
||||
+ split.seconds_powered_off
|
||||
+ split.seconds_unknown
|
||||
)
|
||||
assert total == 1800
|
||||
assert split.seconds_powered_off == 1800
|
||||
|
||||
|
||||
class TestDisabledTimeNotAnHourState:
|
||||
"""Spec §5.2: Disabled time is not an hour state — excluded from numerator/denominator."""
|
||||
|
||||
def test_classify_hour_has_no_disabled_state(self):
|
||||
split = classify_hour(
|
||||
wall_clock_seconds=HOUR_SECONDS,
|
||||
poh_delta=0,
|
||||
duw_delta=0,
|
||||
dur_delta=0,
|
||||
)
|
||||
assert not hasattr(split, "seconds_disabled")
|
||||
@@ -1,96 +0,0 @@
|
||||
"""Test identity normalization per specification.
|
||||
|
||||
From spec §2.3:
|
||||
- strip trailing spaces and newlines
|
||||
- no case folding
|
||||
- empty-after-strip stored blank
|
||||
|
||||
Padded and unpadded renderings of the same field yield byte-identical stored values.
|
||||
"""
|
||||
import pytest
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.collector import normalize_identity
|
||||
|
||||
|
||||
def test_strip_trailing_spaces():
|
||||
"""Trailing spaces are stripped."""
|
||||
identity = {"subnqn": "nqn.2014-08.org.nvmexpress:uuid:12345678 "}
|
||||
result = normalize_identity(identity)
|
||||
assert result == "nqn.2014-08.org.nvmexpress:uuid:12345678"
|
||||
|
||||
|
||||
def test_strip_trailing_newlines():
|
||||
"""Trailing newlines are stripped."""
|
||||
identity = {"subnqn": "nqn.2014-08.org.nvmexpress:uuid:12345678\n"}
|
||||
result = normalize_identity(identity)
|
||||
assert result == "nqn.2014-08.org.nvmexpress:uuid:12345678"
|
||||
|
||||
|
||||
def test_strip_trailing_spaces_and_newlines():
|
||||
"""Trailing spaces and newlines are stripped."""
|
||||
identity = {"subnqn": "nqn.2014-08.org.nvmexpress:uuid:12345678 \n\n"}
|
||||
result = normalize_identity(identity)
|
||||
assert result == "nqn.2014-08.org.nvmexpress:uuid:12345678"
|
||||
|
||||
|
||||
def test_no_case_folding():
|
||||
"""Case is preserved - no case folding."""
|
||||
identity = {"subnqn": "NQN.2014-08.ORG.NVMEXPRESS:UUID:12345678"}
|
||||
result = normalize_identity(identity)
|
||||
assert result == "NQN.2014-08.ORG.NVMEXPRESS:UUID:12345678"
|
||||
|
||||
|
||||
def test_empty_after_strip_stored_blank():
|
||||
"""Empty after strip is stored as blank string."""
|
||||
identity = {"subnqn": " \n\n "}
|
||||
result = normalize_identity(identity)
|
||||
assert result == ""
|
||||
|
||||
|
||||
def test_fallback_to_model_serial():
|
||||
"""When subnqn is empty, falls back to model|serial."""
|
||||
identity = {
|
||||
"subnqn": "",
|
||||
"mn": "Samsung SSD 970 EVO Plus 1TB",
|
||||
"sn": "S4EWNX0N123456",
|
||||
}
|
||||
result = normalize_identity(identity)
|
||||
assert result == "Samsung SSD 970 EVO Plus 1TB|S4EWNX0N123456"
|
||||
|
||||
|
||||
def test_fallback_model_serial_normalized():
|
||||
"""Model and serial are also normalized."""
|
||||
identity = {
|
||||
"subnqn": "",
|
||||
"mn": "Samsung SSD 970 EVO Plus 1TB\n",
|
||||
"sn": "S4EWNX0N123456 ",
|
||||
}
|
||||
result = normalize_identity(identity)
|
||||
assert result == "Samsung SSD 970 EVO Plus 1TB|S4EWNX0N123456"
|
||||
|
||||
|
||||
def test_all_keys_blank_returns_blank():
|
||||
"""When all keys are blank, returns blank (degraded identity)."""
|
||||
identity = {
|
||||
"subnqn": "",
|
||||
"mn": "",
|
||||
"sn": "",
|
||||
}
|
||||
result = normalize_identity(identity)
|
||||
assert result == ""
|
||||
|
||||
|
||||
def test_byte_identical_for_padded_unpadded():
|
||||
"""Padded and unpadded renderings yield byte-identical values."""
|
||||
padded = {"subnqn": "nqn.2014-08.org.nvmexpress:uuid:12345678 \n"}
|
||||
unpadded = {"subnqn": "nqn.2014-08.org.nvmexpress:uuid:12345678"}
|
||||
|
||||
result_padded = normalize_identity(padded)
|
||||
result_unpadded = normalize_identity(unpadded)
|
||||
|
||||
assert result_padded == result_unpadded
|
||||
assert result_padded == "nqn.2014-08.org.nvmexpress:uuid:12345678"
|
||||
@@ -1,404 +0,0 @@
|
||||
"""Legacy migration tests.
|
||||
|
||||
Tests the idempotent, interruption-safe import of history.jsonl into the
|
||||
observation store.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import sqlite3
|
||||
import tempfile
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict
|
||||
|
||||
import pytest
|
||||
|
||||
# Add src to path for imports
|
||||
import sys
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.legacy import import_legacy_history, is_legacy_imported, _parse_history_line
|
||||
from fenris.store import init_store
|
||||
|
||||
|
||||
# Fixtures
|
||||
|
||||
@pytest.fixture
|
||||
def history_fixture() -> str:
|
||||
"""Minimal history.jsonl content with two samples."""
|
||||
samples = [
|
||||
{
|
||||
"timestamp": "2026-08-30T10:00:00Z",
|
||||
"device": "/dev/nvme0",
|
||||
"model": "Samsung SSD 970 EVO Plus 1TB",
|
||||
"serial": "S4EWNX0N123456",
|
||||
"firmware_version": "2B2QEXM7",
|
||||
"capacity_bytes": 1024000000000,
|
||||
"data_units_written": 1000000,
|
||||
"data_units_read": 500000,
|
||||
"percentage_used": 5,
|
||||
"power_on_hours": 8765,
|
||||
"temperature": 35,
|
||||
"available_spare": 100,
|
||||
"media_errors": 0,
|
||||
"power_cycles": 1234,
|
||||
"unsafe_shutdowns": 5,
|
||||
"critical_warning": 0,
|
||||
},
|
||||
{
|
||||
"timestamp": "2026-08-30T11:00:00Z",
|
||||
"device": "/dev/nvme0",
|
||||
"model": "Samsung SSD 970 EVO Plus 1TB",
|
||||
"serial": "S4EWNX0N123456",
|
||||
"firmware_version": "2B2QEXM7",
|
||||
"capacity_bytes": 1024000000000,
|
||||
"data_units_written": 1001000,
|
||||
"data_units_read": 501000,
|
||||
"percentage_used": 5,
|
||||
"power_on_hours": 8766,
|
||||
"temperature": 36,
|
||||
"available_spare": 100,
|
||||
"media_errors": 0,
|
||||
"power_cycles": 1234,
|
||||
"unsafe_shutdowns": 5,
|
||||
"critical_warning": 0,
|
||||
},
|
||||
]
|
||||
return "\n".join(json.dumps(s) for s in samples)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def hourly_fixture() -> str:
|
||||
"""Minimal hourly.jsonl content for diffing."""
|
||||
hours = [
|
||||
{
|
||||
"hour": "2026-08-30T10:00:00Z",
|
||||
"bytes_written_delta": 512000000,
|
||||
"bytes_read_delta": 256000000,
|
||||
"sample_count": 1,
|
||||
},
|
||||
{
|
||||
"hour": "2026-08-30T11:00:00Z",
|
||||
"bytes_written_delta": 513000000,
|
||||
"bytes_read_delta": 257000000,
|
||||
"sample_count": 1,
|
||||
},
|
||||
]
|
||||
return "\n".join(json.dumps(h) for h in hours)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def config_fixture(tmp_path: Path) -> Dict[str, Any]:
|
||||
"""Configuration fixture."""
|
||||
return {
|
||||
"device": "/dev/nvme0",
|
||||
"store_path": str(tmp_path / "observations.db"),
|
||||
"data_dir": str(tmp_path),
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def clock_fixture():
|
||||
"""Injected clock returning fixed time."""
|
||||
class FakeClock:
|
||||
def __init__(self):
|
||||
self.now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
def utcnow(self):
|
||||
return self.now
|
||||
|
||||
return FakeClock()
|
||||
|
||||
|
||||
# Test: Idempotency - second run no-ops
|
||||
|
||||
def test_import_idempotent(
|
||||
history_fixture: str,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given a store with legacy import marker,
|
||||
when import is called again,
|
||||
then it no-ops."""
|
||||
# Create history file
|
||||
history_path = Path(config_fixture["data_dir"]) / "history.jsonl"
|
||||
history_path.write_text(history_fixture)
|
||||
|
||||
# Initialize store
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
conn = init_store(store_path)
|
||||
|
||||
# First import
|
||||
result1 = import_legacy_history(conn, history_path, clock=clock_fixture)
|
||||
assert result1["ok"] is True
|
||||
assert result1["skipped"] is False
|
||||
|
||||
# Second import (should no-op)
|
||||
history_path2 = Path(config_fixture["data_dir"]) / "history.jsonl"
|
||||
history_path2.write_text(history_fixture)
|
||||
result2 = import_legacy_history(conn, history_path2, clock=clock_fixture)
|
||||
assert result2["ok"] is True
|
||||
assert result2["skipped"] is True
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: Interruption safety - single transaction
|
||||
|
||||
def test_import_single_transaction(
|
||||
history_fixture: str,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given history.jsonl,
|
||||
when import runs,
|
||||
then the entire import is a single transaction."""
|
||||
# Create history file
|
||||
history_path = Path(config_fixture["data_dir"]) / "history.jsonl"
|
||||
history_path.write_text(history_fixture)
|
||||
|
||||
# Initialize store
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
conn = init_store(store_path)
|
||||
|
||||
# Import
|
||||
result = import_legacy_history(conn, history_path, clock=clock_fixture)
|
||||
assert result["ok"] is True
|
||||
|
||||
# Verify all data was imported atomically
|
||||
cursor = conn.execute("SELECT COUNT(*) FROM samples")
|
||||
assert cursor.fetchone()[0] == 2
|
||||
|
||||
cursor = conn.execute("SELECT COUNT(*) FROM hour_observations")
|
||||
assert cursor.fetchone()[0] == 2
|
||||
|
||||
cursor = conn.execute("SELECT COUNT(*) FROM monitoring_periods")
|
||||
assert cursor.fetchone()[0] == 1
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: Legacy files renamed to *.migrated after commit
|
||||
|
||||
def test_legacy_files_renamed(
|
||||
history_fixture: str,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given history.jsonl and hourly.jsonl,
|
||||
when import commits,
|
||||
then files are renamed to *.migrated."""
|
||||
# Create files
|
||||
history_path = Path(config_fixture["data_dir"]) / "history.jsonl"
|
||||
history_path.write_text(history_fixture)
|
||||
|
||||
hourly_path = Path(config_fixture["data_dir"]) / "hourly.jsonl"
|
||||
hourly_path.write_text("{}")
|
||||
|
||||
# Initialize store
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
conn = init_store(store_path)
|
||||
|
||||
# Import
|
||||
result = import_legacy_history(conn, history_path, hourly_path=hourly_path, clock=clock_fixture)
|
||||
assert result["ok"] is True
|
||||
|
||||
# Verify files renamed
|
||||
assert not history_path.exists()
|
||||
assert history_path.with_suffix(history_path.suffix + ".migrated").exists()
|
||||
|
||||
assert not hourly_path.exists()
|
||||
assert hourly_path.with_suffix(hourly_path.suffix + ".migrated").exists()
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: Malformed lines quarantined with logged count
|
||||
|
||||
def test_malformed_lines_quarantined(
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given history.jsonl with malformed lines,
|
||||
when import runs,
|
||||
then malformed lines are quarantined with logged count."""
|
||||
# Create history with malformed lines
|
||||
history_content = "\n".join([
|
||||
'{"timestamp": "2026-08-30T10:00:00Z", "model": "Test", "serial": "123", "firmware_version": "1.0", "data_units_written": 1000, "data_units_read": 500, "percentage_used": 5, "power_on_hours": 100, "temperature": 35}',
|
||||
'NOT JSON',
|
||||
'{"timestamp": "2026-08-30T11:00:00Z", "model": "Test", "serial": "123", "firmware_version": "1.0", "data_units_written": 1001, "data_units_read": 501, "percentage_used": 5, "power_on_hours": 101, "temperature": 36}',
|
||||
])
|
||||
history_path = Path(config_fixture["data_dir"]) / "history.jsonl"
|
||||
history_path.write_text(history_content)
|
||||
|
||||
# Initialize store
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
conn = init_store(store_path)
|
||||
|
||||
# Import
|
||||
result = import_legacy_history(conn, history_path, clock=clock_fixture)
|
||||
assert result["ok"] is True
|
||||
assert result["malformed_lines"] == 1
|
||||
assert result["samples_imported"] == 2
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: hourly.jsonl diffed and logged but never trusted
|
||||
|
||||
def test_hourly_jsonl_diffed(
|
||||
history_fixture: str,
|
||||
hourly_fixture: str,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given history.jsonl and hourly.jsonl with mismatches,
|
||||
when import runs,
|
||||
then mismatches are diffed and logged."""
|
||||
# Create files
|
||||
history_path = Path(config_fixture["data_dir"]) / "history.jsonl"
|
||||
history_path.write_text(history_fixture)
|
||||
|
||||
hourly_path = Path(config_fixture["data_dir"]) / "hourly.jsonl"
|
||||
hourly_path.write_text(hourly_fixture)
|
||||
|
||||
# Initialize store
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
conn = init_store(store_path)
|
||||
|
||||
# Import (should not fail even with mismatches)
|
||||
result = import_legacy_history(conn, history_path, hourly_path=hourly_path, clock=clock_fixture)
|
||||
assert result["ok"] is True
|
||||
|
||||
# Verify data was imported from history.jsonl, not hourly.jsonl
|
||||
cursor = conn.execute("SELECT bytes_written_delta FROM hour_observations ORDER BY hour")
|
||||
deltas = [row[0] for row in cursor.fetchall()]
|
||||
|
||||
# Should match history.jsonl derived values, not hourly.jsonl
|
||||
assert len(deltas) == 2
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: Legacy identity is mn-only
|
||||
|
||||
def test_legacy_identity_mn_only(
|
||||
history_fixture: str,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given history.jsonl,
|
||||
when import runs,
|
||||
then legacy segment has mn-only identity."""
|
||||
# Create history file
|
||||
history_path = Path(config_fixture["data_dir"]) / "history.jsonl"
|
||||
history_path.write_text(history_fixture)
|
||||
|
||||
# Initialize store
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
conn = init_store(store_path)
|
||||
|
||||
# Import
|
||||
result = import_legacy_history(conn, history_path, clock=clock_fixture)
|
||||
assert result["ok"] is True
|
||||
assert result["legacy_identity_key"] == "legacy|Samsung SSD 970 EVO Plus 1TB"
|
||||
|
||||
# Verify segment has mn-only identity
|
||||
cursor = conn.execute("SELECT identity_key, mn, sn, subnqn FROM controller_segments")
|
||||
row = cursor.fetchone()
|
||||
assert row[0] == "legacy|Samsung SSD 970 EVO Plus 1TB"
|
||||
assert row[1] == "Samsung SSD 970 EVO Plus 1TB"
|
||||
assert row[2] is None # sn is None for legacy
|
||||
assert row[3] is None # subnqn is None for legacy
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: No synthetic baseline created
|
||||
|
||||
def test_no_synthetic_baseline(
|
||||
history_fixture: str,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given history.jsonl,
|
||||
when import runs,
|
||||
then no endurance baseline is created."""
|
||||
# Create history file
|
||||
history_path = Path(config_fixture["data_dir"]) / "history.jsonl"
|
||||
history_path.write_text(history_fixture)
|
||||
|
||||
# Initialize store
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
conn = init_store(store_path)
|
||||
|
||||
# Import
|
||||
result = import_legacy_history(conn, history_path, clock=clock_fixture)
|
||||
assert result["ok"] is True
|
||||
|
||||
# Verify no baseline created
|
||||
cursor = conn.execute("SELECT COUNT(*) FROM endurance_baseline")
|
||||
assert cursor.fetchone()[0] == 0
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: Monitoring period opened and closed
|
||||
|
||||
def test_monitoring_period_opened_closed(
|
||||
history_fixture: str,
|
||||
config_fixture: Dict[str, Any],
|
||||
clock_fixture,
|
||||
):
|
||||
"""Given history.jsonl,
|
||||
when import runs,
|
||||
then one monitoring period is opened at first sample and closed at migration."""
|
||||
# Create history file
|
||||
history_path = Path(config_fixture["data_dir"]) / "history.jsonl"
|
||||
history_path.write_text(history_fixture)
|
||||
|
||||
# Initialize store
|
||||
store_path = Path(config_fixture["store_path"])
|
||||
conn = init_store(store_path)
|
||||
|
||||
# Import
|
||||
result = import_legacy_history(conn, history_path, clock=clock_fixture)
|
||||
assert result["ok"] is True
|
||||
|
||||
# Verify monitoring period
|
||||
cursor = conn.execute("SELECT started_at, ended_at, end_cause FROM monitoring_periods")
|
||||
row = cursor.fetchone()
|
||||
assert row[0] == "2026-08-30T10:00:00+00:00" # First sample time
|
||||
assert row[1] == clock_fixture.now.isoformat() # Migration time
|
||||
assert row[2] == "migrated"
|
||||
|
||||
conn.close()
|
||||
|
||||
|
||||
# Test: Parse history line
|
||||
|
||||
def test_parse_history_line_valid():
|
||||
"""Given a valid history line,
|
||||
when parsed,
|
||||
then returns the record."""
|
||||
line = '{"timestamp": "2026-08-30T10:00:00Z", "model": "Test", "serial": "123", "firmware_version": "1.0", "data_units_written": 1000, "data_units_read": 500, "percentage_used": 5, "power_on_hours": 100, "temperature": 35}'
|
||||
result = _parse_history_line(line, 1)
|
||||
assert result is not None
|
||||
assert result["model"] == "Test"
|
||||
|
||||
|
||||
def test_parse_history_line_malformed():
|
||||
"""Given a malformed JSON line,
|
||||
when parsed,
|
||||
then returns None."""
|
||||
result = _parse_history_line("NOT JSON", 1)
|
||||
assert result is None
|
||||
|
||||
|
||||
def test_parse_history_line_missing_field():
|
||||
"""Given a line with missing required field,
|
||||
when parsed,
|
||||
then returns None."""
|
||||
line = '{"timestamp": "2026-08-30T10:00:00Z", "model": "Test"}'
|
||||
result = _parse_history_line(line, 1)
|
||||
assert result is None
|
||||
@@ -1,325 +0,0 @@
|
||||
"""Tests for fenris-monitor helper.
|
||||
|
||||
Spec: §8.4, §8.5, §8.6, §8.7
|
||||
"""
|
||||
import json
|
||||
import sqlite3
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
import sys
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.monitor import (
|
||||
cmd_enable,
|
||||
cmd_disable,
|
||||
cmd_collect,
|
||||
cmd_baseline_set,
|
||||
cmd_baseline_clear,
|
||||
is_root,
|
||||
)
|
||||
from fenris.store import init_store
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def tmp_store(tmp_path):
|
||||
"""Create a temporary observation store."""
|
||||
store_path = tmp_path / "observations.db"
|
||||
conn = init_store(store_path)
|
||||
yield conn
|
||||
conn.close()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store_path(tmp_path):
|
||||
"""Return path to a temporary observation store."""
|
||||
return tmp_path / "observations.db"
|
||||
|
||||
|
||||
class TestIsRoot:
|
||||
def test_root_returns_true(self):
|
||||
with patch("os.geteuid", return_value=0):
|
||||
assert is_root() is True
|
||||
|
||||
def test_non_root_returns_false(self):
|
||||
with patch("os.geteuid", return_value=1000):
|
||||
assert is_root() is False
|
||||
|
||||
|
||||
class TestEnableIdempotentMatrix:
|
||||
"""§8.6: Period-row idempotent matrix."""
|
||||
|
||||
def test_first_enable_creates_missing_store(self, store_path):
|
||||
"""A fresh package install has a store directory but no database yet."""
|
||||
args = MagicMock(now=False, store_path=store_path)
|
||||
|
||||
with patch("fenris.monitor.subprocess") as mock_sub:
|
||||
mock_sub.run.return_value = MagicMock(returncode=0)
|
||||
cmd_enable(args)
|
||||
|
||||
conn = init_store(store_path)
|
||||
row = conn.execute(
|
||||
"SELECT ended_at FROM monitoring_periods WHERE ended_at IS NULL"
|
||||
).fetchone()
|
||||
conn.close()
|
||||
assert row is not None
|
||||
|
||||
def test_first_opens_period(self, store_path):
|
||||
"""First-ever enable opens a period at the enable moment."""
|
||||
# Initialize store
|
||||
init_store(store_path)
|
||||
|
||||
args = MagicMock(now=False, store_path=store_path)
|
||||
|
||||
with patch("fenris.monitor.subprocess") as mock_sub:
|
||||
mock_sub.run.return_value = MagicMock(returncode=0)
|
||||
cmd_enable(args)
|
||||
|
||||
# Period should be open
|
||||
conn = init_store(store_path)
|
||||
cursor = conn.execute(
|
||||
"SELECT ended_at FROM monitoring_periods WHERE ended_at IS NULL"
|
||||
)
|
||||
assert cursor.fetchone() is not None
|
||||
conn.close()
|
||||
|
||||
def test_resume_with_open_period_noop(self, store_path):
|
||||
"""Resume with open period: no-op (gap stays inside as unknown)."""
|
||||
# Initialize store and open a period
|
||||
conn = init_store(store_path)
|
||||
now = datetime.now(timezone.utc)
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at) VALUES (?)",
|
||||
(now.isoformat(),),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
args = MagicMock(now=True, store_path=store_path)
|
||||
|
||||
with patch("fenris.monitor.subprocess") as mock_sub:
|
||||
mock_sub.run.return_value = MagicMock(returncode=0)
|
||||
cmd_enable(args)
|
||||
|
||||
# Should still have exactly one open period
|
||||
conn = init_store(store_path)
|
||||
cursor = conn.execute(
|
||||
"SELECT COUNT(*) FROM monitoring_periods WHERE ended_at IS NULL"
|
||||
)
|
||||
assert cursor.fetchone()[0] == 1
|
||||
conn.close()
|
||||
|
||||
def test_resume_with_no_period_opens_new(self, store_path):
|
||||
"""Resume with no open period opens a new row."""
|
||||
# Initialize store and close any existing period
|
||||
conn = init_store(store_path)
|
||||
conn.execute(
|
||||
"UPDATE monitoring_periods SET ended_at = ?, end_cause = ?",
|
||||
(datetime.now(timezone.utc).isoformat(), "user_disabled"),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
args = MagicMock(now=True, store_path=store_path)
|
||||
|
||||
with patch("fenris.monitor.subprocess") as mock_sub:
|
||||
mock_sub.run.return_value = MagicMock(returncode=0)
|
||||
cmd_enable(args)
|
||||
|
||||
# Should have a new open period
|
||||
conn = init_store(store_path)
|
||||
cursor = conn.execute(
|
||||
"SELECT COUNT(*) FROM monitoring_periods WHERE ended_at IS NULL"
|
||||
)
|
||||
assert cursor.fetchone()[0] == 1
|
||||
conn.close()
|
||||
|
||||
|
||||
class TestDisableIdempotentMatrix:
|
||||
"""§8.6: Period-row idempotent matrix."""
|
||||
|
||||
def test_pause_with_open_period_closes_user_disabled(self, store_path):
|
||||
"""Pause with open period closes it user_disabled."""
|
||||
# Initialize store and open a period
|
||||
conn = init_store(store_path)
|
||||
now = datetime.now(timezone.utc)
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at) VALUES (?)",
|
||||
(now.isoformat(),),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
args = MagicMock(now=True, store_path=store_path)
|
||||
|
||||
with patch("fenris.monitor.subprocess") as mock_sub:
|
||||
mock_sub.run.return_value = MagicMock(returncode=0)
|
||||
cmd_disable(args)
|
||||
|
||||
# Period should be closed with user_disabled
|
||||
conn = init_store(store_path)
|
||||
cursor = conn.execute(
|
||||
"SELECT end_cause FROM monitoring_periods WHERE ended_at IS NOT NULL"
|
||||
)
|
||||
assert cursor.fetchone()[0] == "user_disabled"
|
||||
conn.close()
|
||||
|
||||
def test_pause_without_open_period_noop(self, store_path):
|
||||
"""Pause otherwise no-ops."""
|
||||
# Initialize store
|
||||
init_store(store_path)
|
||||
|
||||
args = MagicMock(now=True, store_path=store_path)
|
||||
|
||||
with patch("fenris.monitor.subprocess") as mock_sub:
|
||||
mock_sub.run.return_value = MagicMock(returncode=0)
|
||||
cmd_disable(args)
|
||||
|
||||
# No periods should exist
|
||||
conn = init_store(store_path)
|
||||
cursor = conn.execute("SELECT COUNT(*) FROM monitoring_periods")
|
||||
assert cursor.fetchone()[0] == 0
|
||||
conn.close()
|
||||
|
||||
def test_raw_systemctl_stop_never_records_user_disabled(self, store_path):
|
||||
"""Raw systemctl stop outside helper never records user_disabled."""
|
||||
# Initialize store and open a period
|
||||
conn = init_store(store_path)
|
||||
now = datetime.now(timezone.utc)
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at) VALUES (?)",
|
||||
(now.isoformat(),),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
# Simulate raw systemctl stop (no monitor involved)
|
||||
# The period stays open - only the sanctioned path closes it
|
||||
cursor = conn.execute(
|
||||
"SELECT end_cause FROM monitoring_periods WHERE ended_at IS NULL"
|
||||
)
|
||||
assert cursor.fetchone() is not None # Still open
|
||||
conn.close()
|
||||
|
||||
|
||||
class TestCollectTrigger:
|
||||
"""§8.7: On-demand collection via helper path."""
|
||||
|
||||
def test_collect_triggers_systemctl_start(self):
|
||||
"""Collect starts fenris-collect.service synchronously."""
|
||||
args = MagicMock()
|
||||
|
||||
with patch("fenris.monitor.subprocess") as mock_sub:
|
||||
mock_sub.run.return_value = MagicMock(returncode=0)
|
||||
cmd_collect(args)
|
||||
|
||||
mock_sub.run.assert_called_once_with(
|
||||
["systemctl", "start", "fenris-collect.service"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
|
||||
def test_collect_failure_exits_nonzero(self):
|
||||
"""Collect failure exits with nonzero status."""
|
||||
args = MagicMock()
|
||||
|
||||
with patch("fenris.monitor.subprocess") as mock_sub:
|
||||
mock_sub.run.return_value = MagicMock(
|
||||
returncode=1, stderr="Unit not found"
|
||||
)
|
||||
with pytest.raises(SystemExit) as exc_info:
|
||||
cmd_collect(args)
|
||||
assert exc_info.value.code == 1
|
||||
|
||||
|
||||
class TestBaselinePersistence:
|
||||
"""PR-14: Baseline persistence behind polkit-guarded helper."""
|
||||
|
||||
def test_baseline_set_persists(self, store_path):
|
||||
"""baseline set persists the baseline row."""
|
||||
# Initialize store
|
||||
init_store(store_path)
|
||||
|
||||
args = MagicMock(
|
||||
store_path=store_path,
|
||||
baseline_json=json.dumps(
|
||||
{
|
||||
"tbw_terabytes": 600,
|
||||
"source_url": "https://example.com/spec",
|
||||
"document_revision": "rev1",
|
||||
"entry_date": "2024-01-01",
|
||||
"model_string": "Samsung 990 Pro",
|
||||
"nominal_capacity_bytes": 2000000000000,
|
||||
}
|
||||
)
|
||||
)
|
||||
|
||||
cmd_baseline_set(args)
|
||||
|
||||
conn = init_store(store_path)
|
||||
cursor = conn.execute("SELECT * FROM endurance_baseline")
|
||||
row = cursor.fetchone()
|
||||
assert row is not None
|
||||
assert row[1] == 600.0 # tbw_terabytes
|
||||
conn.close()
|
||||
|
||||
def test_baseline_set_replaces_existing(self, store_path):
|
||||
"""baseline set replaces any existing baseline."""
|
||||
# Initialize store and insert initial baseline
|
||||
conn = init_store(store_path)
|
||||
now = datetime.now(timezone.utc)
|
||||
conn.execute(
|
||||
"INSERT INTO endurance_baseline (tbw_terabytes, source_url, "
|
||||
"document_revision, entry_date, model_string, nominal_capacity_bytes, "
|
||||
"created_at, updated_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(400, "old", "v1", "2023-01-01", "Old Model", 1000000000000,
|
||||
now.isoformat(), now.isoformat()),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
args = MagicMock(
|
||||
store_path=store_path,
|
||||
baseline_json=json.dumps(
|
||||
{
|
||||
"tbw_terabytes": 600,
|
||||
"source_url": "new",
|
||||
"document_revision": "v2",
|
||||
"entry_date": "2024-01-01",
|
||||
"model_string": "New Model",
|
||||
"nominal_capacity_bytes": 2000000000000,
|
||||
}
|
||||
)
|
||||
)
|
||||
|
||||
cmd_baseline_set(args)
|
||||
|
||||
conn = init_store(store_path)
|
||||
cursor = conn.execute("SELECT COUNT(*) FROM endurance_baseline")
|
||||
assert cursor.fetchone()[0] == 1 # Only one row
|
||||
conn.close()
|
||||
|
||||
def test_baseline_clear_removes(self, store_path):
|
||||
"""baseline clear removes the baseline."""
|
||||
# Initialize store and insert baseline
|
||||
conn = init_store(store_path)
|
||||
now = datetime.now(timezone.utc)
|
||||
conn.execute(
|
||||
"INSERT INTO endurance_baseline (tbw_terabytes, source_url, "
|
||||
"document_revision, entry_date, model_string, nominal_capacity_bytes, "
|
||||
"created_at, updated_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(400, "src", "v1", "2024-01-01", "Model", 1000000000000,
|
||||
now.isoformat(), now.isoformat()),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
args = MagicMock(store_path=store_path)
|
||||
cmd_baseline_clear(args)
|
||||
|
||||
conn = init_store(store_path)
|
||||
cursor = conn.execute("SELECT COUNT(*) FROM endurance_baseline")
|
||||
assert cursor.fetchone()[0] == 0
|
||||
conn.close()
|
||||
@@ -1,147 +0,0 @@
|
||||
"""Monitoring period tests.
|
||||
|
||||
From spec §5.2, §8.6, §9.8:
|
||||
- Run finding no open period opens one at the run moment, never backdated
|
||||
- Wall-clock outside periods excluded from numerator/denominator
|
||||
- Powered-off time stays inside a period; disabled time does not
|
||||
- End causes: user_disabled, migrated, unknown_gap
|
||||
"""
|
||||
import sqlite3
|
||||
import sys
|
||||
from datetime import datetime, timezone, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.store import init_store
|
||||
from fenris.monitoring_periods import (
|
||||
ensure_period_open,
|
||||
close_period,
|
||||
get_open_period,
|
||||
is_inside_period,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store_conn(tmp_path: Path):
|
||||
"""Initialize an observation store and return a connection."""
|
||||
db_path = tmp_path / "test.db"
|
||||
conn = init_store(db_path)
|
||||
yield conn
|
||||
conn.close()
|
||||
|
||||
|
||||
class TestEnsurePeriodOpen:
|
||||
"""Spec §9.8: Run finding no open period opens one at the run moment."""
|
||||
|
||||
def test_opens_period_when_none_exists(self, store_conn):
|
||||
"""First collection run opens a period at the run moment."""
|
||||
run_time = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, run_time)
|
||||
|
||||
period = get_open_period(store_conn)
|
||||
assert period is not None
|
||||
assert period["started_at"] == "2026-09-01T12:00:00+00:00"
|
||||
assert period["ended_at"] is None
|
||||
assert period["end_cause"] is None
|
||||
|
||||
def test_no_opener_when_already_open(self, store_conn):
|
||||
"""If a period is already open, no new period is created."""
|
||||
t1 = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
t2 = datetime(2026, 9, 1, 12, 5, 0, tzinfo=timezone.utc)
|
||||
|
||||
ensure_period_open(store_conn, t1)
|
||||
ensure_period_open(store_conn, t2)
|
||||
|
||||
period = get_open_period(store_conn)
|
||||
assert period is not None
|
||||
assert period["started_at"] == "2026-09-01T12:00:00+00:00"
|
||||
|
||||
def test_never_backdated(self, store_conn):
|
||||
"""Period starts at the run moment, not the beginning of the hour."""
|
||||
run_time = datetime(2026, 9, 1, 12, 3, 45, tzinfo=timezone.utc)
|
||||
ensure_period_open(store_conn, run_time)
|
||||
|
||||
period = get_open_period(store_conn)
|
||||
assert period["started_at"] == "2026-09-01T12:03:45+00:00"
|
||||
|
||||
def test_new_period_after_close(self, store_conn):
|
||||
"""After closing a period, next run opens a new one."""
|
||||
t1 = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
t2 = datetime(2026, 9, 1, 13, 0, 0, tzinfo=timezone.utc)
|
||||
t3 = datetime(2026, 9, 1, 14, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
ensure_period_open(store_conn, t1)
|
||||
close_period(store_conn, t2, "user_disabled")
|
||||
ensure_period_open(store_conn, t3)
|
||||
|
||||
period = get_open_period(store_conn)
|
||||
assert period is not None
|
||||
assert period["started_at"] == "2026-09-01T14:00:00+00:00"
|
||||
|
||||
|
||||
class TestClosePeriod:
|
||||
"""Spec §8.6: Pause closes with user_disabled."""
|
||||
|
||||
def test_close_with_user_disabled(self, store_conn):
|
||||
t1 = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
t2 = datetime(2026, 9, 1, 13, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
ensure_period_open(store_conn, t1)
|
||||
close_period(store_conn, t2, "user_disabled")
|
||||
|
||||
period = get_open_period(store_conn)
|
||||
assert period is None
|
||||
|
||||
cursor = store_conn.execute(
|
||||
"SELECT ended_at, end_cause FROM monitoring_periods WHERE id = 1"
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
assert row[0] == "2026-09-01T13:00:00+00:00"
|
||||
assert row[1] == "user_disabled"
|
||||
|
||||
def test_close_nonexistent_is_noop(self, store_conn):
|
||||
"""Closing when no period is open is a no-op (spec §8.6 pause otherwise)."""
|
||||
t = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
close_period(store_conn, t, "user_disabled")
|
||||
|
||||
cursor = store_conn.execute("SELECT COUNT(*) FROM monitoring_periods")
|
||||
assert cursor.fetchone()[0] == 0
|
||||
|
||||
|
||||
class TestIsInsidePeriod:
|
||||
"""Spec §5.2: Wall-clock outside periods excluded from numerator/denominator."""
|
||||
|
||||
def test_inside_open_period(self, store_conn):
|
||||
t_start = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
t_inside = datetime(2026, 9, 1, 12, 30, 0, tzinfo=timezone.utc)
|
||||
|
||||
ensure_period_open(store_conn, t_start)
|
||||
assert is_inside_period(store_conn, t_inside) is True
|
||||
|
||||
def test_outside_closed_period(self, store_conn):
|
||||
t_start = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
t_end = datetime(2026, 9, 1, 13, 0, 0, tzinfo=timezone.utc)
|
||||
t_outside = datetime(2026, 9, 1, 14, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
ensure_period_open(store_conn, t_start)
|
||||
close_period(store_conn, t_end, "user_disabled")
|
||||
assert is_inside_period(store_conn, t_outside) is False
|
||||
|
||||
def test_outside_no_periods(self, store_conn):
|
||||
t = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
assert is_inside_period(store_conn, t) is False
|
||||
|
||||
def test_inside_second_period(self, store_conn):
|
||||
"""Two periods with a gap; time in second period is inside."""
|
||||
t1_start = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
t1_end = datetime(2026, 9, 1, 13, 0, 0, tzinfo=timezone.utc)
|
||||
t2_start = datetime(2026, 9, 1, 14, 0, 0, tzinfo=timezone.utc)
|
||||
t_inside = datetime(2026, 9, 1, 14, 30, 0, tzinfo=timezone.utc)
|
||||
|
||||
ensure_period_open(store_conn, t1_start)
|
||||
close_period(store_conn, t1_end, "user_disabled")
|
||||
ensure_period_open(store_conn, t2_start)
|
||||
assert is_inside_period(store_conn, t_inside) is True
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,901 +0,0 @@
|
||||
"""Projection core tests.
|
||||
|
||||
Covers acceptance criteria:
|
||||
- PR-1: Exactly one projection from precedence-chosen baseline; PU context only
|
||||
- PR-8: Confidence rule table holds verbatim; state + facts, never percentage
|
||||
- PR-10: Implied baseline eligible only after >=2 PU increments
|
||||
- PR-11: Zero rate renders fixed phrase; scenario range only spread
|
||||
- PR-12: Contract hands over exactly: state, facts, headline, scenario, PU, disclosures
|
||||
- PR-13: Baseline provenance and validation per register
|
||||
- PR-17: Arithmetic exactly E_rated = TBW * 10^12, E_implied = 100*W/p, projected = max(E-W,0)/rate
|
||||
"""
|
||||
import sqlite3
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
import sys
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.store import init_store
|
||||
from fenris.monitoring_periods import ensure_period_open, close_period
|
||||
from fenris.projection import (
|
||||
compute_projection, ConfidenceState, BaselineTier, ScenarioRange,
|
||||
TBW_TO_BYTES, HORIZON_DAYS, WARMING_MIN_DAYS, STALENESS_HOURS,
|
||||
YOUNG_REGIME_DAYS, DISCLOSURES,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(tmp_path):
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
yield conn
|
||||
conn.close()
|
||||
|
||||
|
||||
def _clock(year=2026, month=9, day=30, hour=12):
|
||||
return datetime(year, month, day, hour, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
def _insert_baseline(conn, tbw_tb=1.0, verified=True, model="Samsung SSD 970 EVO Plus 1TB",
|
||||
source_url="https://example.com/spec", doc_rev="v1.0",
|
||||
entry_date="2026-01-01", nominal_cap=1024000000000):
|
||||
conn.execute(
|
||||
"INSERT INTO endurance_baseline "
|
||||
"(tbw_terabytes, source_url, document_revision, entry_date, model_string, "
|
||||
" nominal_capacity_bytes, validated_by, verified, created_at, updated_at) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(tbw_tb, source_url, doc_rev, entry_date, model, nominal_cap,
|
||||
"machine_match" if verified else None, verified, "2026-01-01T00:00:00+00:00",
|
||||
"2026-01-01T00:00:00+00:00"),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _insert_segment(conn, opened_at="2026-09-01T00:00:00+00:00",
|
||||
identity_key="nqn.test", degraded=False,
|
||||
mn="Samsung SSD 970 EVO Plus 1TB"):
|
||||
conn.execute(
|
||||
"INSERT INTO controller_segments "
|
||||
"(opened_at, identity_key, identity_degraded, subnqn, sn, mn, fr, vid, ssvid, transport) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(opened_at, identity_key, degraded, "nqn.test", "SN123", mn, "FW1", "0x144d", "0x144d", "pcie"),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _insert_day(conn, day, bw=1024*1024*100, coverage=0.95, samples=24):
|
||||
conn.execute(
|
||||
"INSERT INTO day_aggregates (day, active_seconds, idle_seconds, powered_off_seconds, "
|
||||
"unknown_seconds, bytes_written_delta, bytes_read_delta, sample_count, coverage) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(day, 3600, 0, 0, 0, bw, 0, samples, coverage),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _insert_sample(conn, ts, pu=5):
|
||||
conn.execute(
|
||||
"INSERT INTO samples (ts, device, data_units_written, data_units_read, "
|
||||
"percentage_used, bytes_written, bytes_read, power_on_hours) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(ts, "/dev/nvme0n1", 1000000, 500000, pu, 512000000000, 256000000000, 8765),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _open_period(conn, start="2026-09-01T00:00:00+00:00"):
|
||||
ensure_period_open(conn, datetime.fromisoformat(start))
|
||||
|
||||
|
||||
class TestPrecedence:
|
||||
def test_no_baseline_unavailable(self, store):
|
||||
_insert_segment(store)
|
||||
_insert_day(store, "2026-09-28", bw=1024*1024*1000)
|
||||
_open_period(store)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert result.baseline_tier == BaselineTier.NONE
|
||||
assert result.headline_remaining_seconds is None
|
||||
|
||||
def test_verified_baseline_chosen(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(14):
|
||||
d = (datetime(2026, 9, 15) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.baseline_tier == BaselineTier.VERIFIED
|
||||
assert "verified manufacturer TBW" in result.baseline_label
|
||||
|
||||
def test_pu_is_context_not_second_projection(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=10)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.pu_context_line.startswith("Percentage Used:")
|
||||
assert "%" in result.pu_context_line
|
||||
|
||||
|
||||
class TestConfidenceRuleTable:
|
||||
def test_unavailable_no_baseline(self, store):
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert any("no applicable endurance baseline" in f for f in result.contributing_facts)
|
||||
|
||||
def test_unavailable_zero_rate(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=0)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert any("no finite projection" in f for f in result.contributing_facts)
|
||||
|
||||
def test_limited_young_regime(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(5):
|
||||
d = (datetime(2026, 9, 25) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.LIMITED
|
||||
assert any("regime only" in f and "days old" in f for f in result.contributing_facts)
|
||||
|
||||
def test_limited_degraded_identity(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store, degraded=True)
|
||||
_open_period(store)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.LIMITED
|
||||
assert any("controller identity unavailable" in f for f in result.contributing_facts)
|
||||
|
||||
def test_state_plus_facts_never_percentage(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state in ConfidenceState
|
||||
for f in result.contributing_facts:
|
||||
assert "%" not in f or "coverage" in f or "Percentage" in f
|
||||
|
||||
|
||||
class TestImpliedBaseline:
|
||||
def test_implied_not_chosen_with_verified(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.baseline_tier == BaselineTier.VERIFIED
|
||||
|
||||
|
||||
class TestZeroRate:
|
||||
def test_zero_rate_fixed_phrase(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=0)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert any("no finite projection from this history" in f for f in result.contributing_facts)
|
||||
assert result.headline_remaining_seconds is None
|
||||
|
||||
def test_scenario_range_only_spread(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
if result.scenario_range is not None:
|
||||
assert isinstance(result.scenario_range, ScenarioRange)
|
||||
assert hasattr(result.scenario_range, "rates")
|
||||
|
||||
|
||||
class TestContractHandoff:
|
||||
def test_contract_fields_present(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert isinstance(result.confidence_state, ConfidenceState)
|
||||
assert isinstance(result.contributing_facts, list)
|
||||
assert isinstance(result.pu_context_line, str)
|
||||
assert isinstance(result.disclosure_text, list)
|
||||
assert isinstance(result.baseline_tier, BaselineTier)
|
||||
assert isinstance(result.baseline_label, str)
|
||||
|
||||
def test_recomputed_on_read(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
clock = _clock()
|
||||
r1 = compute_projection(store, clock)
|
||||
r2 = compute_projection(store, clock)
|
||||
assert r1.confidence_state == r2.confidence_state
|
||||
assert r1.headline_remaining_seconds == r2.headline_remaining_seconds
|
||||
|
||||
def test_disclosures_present(self, store):
|
||||
result = compute_projection(store, _clock())
|
||||
assert len(result.disclosure_text) == 6
|
||||
for d in DISCLOSURES:
|
||||
assert d in result.disclosure_text
|
||||
|
||||
|
||||
class TestBaselineProvenance:
|
||||
def test_model_mismatch_unavailable(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True, model="Different Model")
|
||||
_insert_segment(store, mn="Samsung SSD 970 EVO Plus 1TB")
|
||||
_open_period(store)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert any("does not match" in f for f in result.contributing_facts)
|
||||
assert result.baseline_tier == BaselineTier.NONE
|
||||
|
||||
|
||||
class TestArithmetic:
|
||||
def test_rated_tbw_conversion(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
if result.headline_remaining_seconds is not None:
|
||||
E_rated = 1.0 * TBW_TO_BYTES
|
||||
regime_bytes = 30 * 1024 * 1024 * 100
|
||||
# Actual wall-clock: Sep 1 00:00 -> Sep 30 12:00 = 29.5 days
|
||||
period_start = datetime(2026, 9, 1, 0, 0, 0, tzinfo=timezone.utc)
|
||||
period_end = _clock()
|
||||
actual_wc = int((period_end - period_start).total_seconds())
|
||||
rate = regime_bytes / actual_wc
|
||||
expected = max(E_rated - regime_bytes, 0) / rate
|
||||
assert abs(result.headline_remaining_seconds - expected) < 1.0
|
||||
|
||||
def test_implied_baseline_formula(self, store):
|
||||
W_t = 1024 * 1024 * 1000
|
||||
p = 10
|
||||
E_implied = 100 * W_t / p
|
||||
assert E_implied == 100 * 1024 * 1024 * 1000 / 10
|
||||
|
||||
def test_projected_formula(self, store):
|
||||
_insert_baseline(store, tbw_tb=2.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
if result.headline_remaining_seconds is not None:
|
||||
E_rated = 2.0 * TBW_TO_BYTES
|
||||
regime_bytes = 30 * 1024 * 1024 * 100
|
||||
period_start = datetime(2026, 9, 1, 0, 0, 0, tzinfo=timezone.utc)
|
||||
period_end = _clock()
|
||||
actual_wc = int((period_end - period_start).total_seconds())
|
||||
rate = regime_bytes / actual_wc
|
||||
expected = max(E_rated - regime_bytes, 0) / rate
|
||||
assert abs(result.headline_remaining_seconds - expected) < 1.0
|
||||
|
||||
def test_wearing_rate_proportional(self, store):
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=1024*1024*100)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
r_slow = compute_projection(store, _clock())
|
||||
|
||||
store.execute("DELETE FROM day_aggregates")
|
||||
store.commit()
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=2*1024*1024*100)
|
||||
r_fast = compute_projection(store, _clock())
|
||||
|
||||
if r_slow.headline_remaining_seconds is not None and r_fast.headline_remaining_seconds is not None:
|
||||
assert r_fast.headline_remaining_seconds < r_slow.headline_remaining_seconds
|
||||
|
||||
|
||||
|
||||
# ===========================================================================
|
||||
# Issue #26: Project from the sustained regime
|
||||
# Habit change, scenario range, and evidence gates
|
||||
# ===========================================================================
|
||||
|
||||
|
||||
class TestSustainedRegimeRate:
|
||||
"""PR-2: Headline rate is sustained-regime rate; default regime = full
|
||||
history capped at 90 days; scenario range computed independently,
|
||||
covered horizons only, no placeholders."""
|
||||
|
||||
def test_headline_rate_from_regime(self, store):
|
||||
"""Rate is regime DUW / wall-clock, not trailing-24h or all-history."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
# 30 days of 100 MiB/day
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
# Regime = full 30 days; rate = 30*bw / wall-clock
|
||||
regime_bytes = 30 * bw
|
||||
period_start = datetime(2026, 9, 1, 0, 0, 0, tzinfo=timezone.utc)
|
||||
wc = int((_clock() - period_start).total_seconds())
|
||||
expected_rate = regime_bytes / wc
|
||||
if result.headline_remaining_seconds is not None:
|
||||
E = 10.0 * TBW_TO_BYTES
|
||||
expected_seconds = max(E - regime_bytes, 0) / expected_rate
|
||||
assert abs(result.headline_remaining_seconds - expected_seconds) < 1.0
|
||||
|
||||
def test_regime_capped_at_90_days(self, store):
|
||||
"""Default regime is full history capped at 90 days."""
|
||||
_insert_baseline(store, tbw_tb=100.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-06-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-06-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
# 120 days of data (Jun 1 - Sep 28)
|
||||
for i in range(120):
|
||||
d = (datetime(2026, 6, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-28T12:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=30, hour=12))
|
||||
# Regime should be capped at 90 days (from Jun 1 to Sep 30 = 90 days at cutoff)
|
||||
# The 90-day cutoff is Sep 30 - 90 = Jul 1, so regime starts Jul 1
|
||||
assert result.regime_days is not None
|
||||
assert result.regime_days <= 90
|
||||
|
||||
def test_scenario_range_independent_of_regime(self, store):
|
||||
"""Scenario range is computed independently from the regime."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
if result.scenario_range is not None:
|
||||
# Should have 7-day and 28-day horizons (90-day not fully covered)
|
||||
assert 7 in result.scenario_range.rates
|
||||
assert 28 in result.scenario_range.rates
|
||||
|
||||
def test_only_covered_horizons_shown(self, store):
|
||||
"""No placeholder horizons — only horizons the history covers."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-20T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-20T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
# Only 10 days of data
|
||||
for i in range(10):
|
||||
d = (datetime(2026, 9, 20) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
if result.scenario_range is not None:
|
||||
# 7-day is covered, 28-day and 90-day are not
|
||||
assert 7 in result.scenario_range.rates
|
||||
assert 28 not in result.scenario_range.rates
|
||||
assert 90 not in result.scenario_range.rates
|
||||
|
||||
|
||||
class TestHabitChange:
|
||||
"""PR-3: Habit change triggers at 2x/0.5x sustained 3 consecutive days,
|
||||
regime starts at first divergence day, auto-adopted and labeled;
|
||||
young regime caps at Limited."""
|
||||
|
||||
def test_habit_change_2x_detected(self, store):
|
||||
"""2x increase for 3+ consecutive days triggers habit change."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-08-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-08-01T00:00:00+00:00")
|
||||
bw_normal = 100 * 1024 * 1024
|
||||
bw_high = 300 * 1024 * 1024 # 3x the normal rate
|
||||
# 28 days of normal usage
|
||||
for i in range(28):
|
||||
d = (datetime(2026, 8, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw_normal)
|
||||
# 10 days of high usage (3x > 2x threshold)
|
||||
for i in range(10):
|
||||
d = (datetime(2026, 8, 29) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw_high)
|
||||
_insert_sample(store, "2026-09-08T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=8, hour=12))
|
||||
assert result.habit_change_fact is not None
|
||||
assert "usage habit changed" in result.habit_change_fact
|
||||
assert "days ago" in result.habit_change_fact
|
||||
|
||||
def test_habit_change_05x_detected(self, store):
|
||||
"""0.5x decrease for 3+ consecutive days triggers habit change."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-08-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-08-01T00:00:00+00:00")
|
||||
bw_high = 400 * 1024 * 1024
|
||||
bw_low = 100 * 1024 * 1024 # 0.25x < 0.5x threshold
|
||||
# 28 days of high usage
|
||||
for i in range(28):
|
||||
d = (datetime(2026, 8, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw_high)
|
||||
# 10 days of low usage
|
||||
for i in range(10):
|
||||
d = (datetime(2026, 8, 29) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw_low)
|
||||
_insert_sample(store, "2026-09-08T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=8, hour=12))
|
||||
assert result.habit_change_fact is not None
|
||||
assert "usage habit changed" in result.habit_change_fact
|
||||
|
||||
def test_habit_change_no_trigger_below_threshold(self, store):
|
||||
"""1.5x increase does NOT trigger habit change (below 2x threshold)."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-08-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-08-01T00:00:00+00:00")
|
||||
bw_normal = 100 * 1024 * 1024
|
||||
bw_moderate = 150 * 1024 * 1024 # 1.5x < 2x threshold
|
||||
for i in range(28):
|
||||
d = (datetime(2026, 8, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw_normal)
|
||||
for i in range(10):
|
||||
d = (datetime(2026, 8, 29) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw_moderate)
|
||||
_insert_sample(store, "2026-09-08T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=8, hour=12))
|
||||
assert result.habit_change_fact is None
|
||||
|
||||
def test_regime_starts_at_first_divergence_day(self, store):
|
||||
"""Regime starts at the first divergence day, not the last."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-08-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-08-01T00:00:00+00:00")
|
||||
bw_normal = 100 * 1024 * 1024
|
||||
bw_high = 300 * 1024 * 1024
|
||||
for i in range(28):
|
||||
d = (datetime(2026, 8, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw_normal)
|
||||
for i in range(10):
|
||||
d = (datetime(2026, 8, 29) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw_high)
|
||||
_insert_sample(store, "2026-09-08T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=8, hour=12))
|
||||
if result.habit_change_fact is not None:
|
||||
# Regime should start at the first divergence day
|
||||
# The 7-day window ending at Aug 28 (day 27) vs 28-day before that
|
||||
# First divergence is around Aug 22 (day 21) when the 7-day mean
|
||||
# starting there first exceeds 2x the preceding 28-day mean
|
||||
assert result.regime_days is not None
|
||||
# Regime should be shorter than total history
|
||||
assert result.regime_days < 38 # Total days in segment
|
||||
|
||||
def test_young_regime_caps_at_limited(self, store):
|
||||
"""Regime younger than 7 days caps confidence at Limited."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
# Only 5 days of data (young regime)
|
||||
for i in range(5):
|
||||
d = (datetime(2026, 9, 25) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.LIMITED
|
||||
assert any("regime only" in f and "days old" in f for f in result.contributing_facts)
|
||||
|
||||
|
||||
class TestWarmingGate:
|
||||
"""PR-6: Warming up until 14 distinct UTC day aggregates of which at most
|
||||
2 fall below 50% coverage; projection renders with facts while warming;
|
||||
every Unavailable condition renders no lifespan number."""
|
||||
|
||||
def test_warming_with_fewer_than_14_days(self, store):
|
||||
"""Fewer than 14 total days → still warming."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-20T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-20T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(10):
|
||||
d = (datetime(2026, 9, 20) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw, coverage=0.95)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.warming_fact is not None
|
||||
assert "warming up" in result.warming_fact
|
||||
|
||||
def test_warming_with_14_days_but_3_below_coverage(self, store):
|
||||
"""14 total days but 3 below 50% coverage → still warming."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-17T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-17T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(14):
|
||||
d = (datetime(2026, 9, 17) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
# 3 days with low coverage
|
||||
cov = 0.30 if i < 3 else 0.95
|
||||
_insert_day(store, d, bw=bw, coverage=cov)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.warming_fact is not None
|
||||
assert "warming up" in result.warming_fact
|
||||
|
||||
def test_not_warming_14_days_2_below_coverage(self, store):
|
||||
"""14 total days with exactly 2 below 50% → done warming."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-17T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-17T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(14):
|
||||
d = (datetime(2026, 9, 17) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
cov = 0.30 if i < 2 else 0.95
|
||||
_insert_day(store, d, bw=bw, coverage=cov)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.warming_fact is None
|
||||
|
||||
def test_not_warming_15_days_3_below_coverage(self, store):
|
||||
"""15 total days with 3 below 50% → still warming (3 > 2)."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-16T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-16T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(15):
|
||||
d = (datetime(2026, 9, 16) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
cov = 0.30 if i < 3 else 0.95
|
||||
_insert_day(store, d, bw=bw, coverage=cov)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.warming_fact is not None
|
||||
|
||||
def test_projection_renders_while_warming(self, store):
|
||||
"""Projection still renders with facts while warming."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-20T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-20T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(10):
|
||||
d = (datetime(2026, 9, 20) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw, coverage=0.95)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
# Should have warming fact but still render
|
||||
assert result.warming_fact is not None
|
||||
assert result.contributing_facts is not None
|
||||
assert len(result.contributing_facts) > 0
|
||||
|
||||
def test_unavailable_renders_no_lifespan(self, store):
|
||||
"""Every Unavailable condition renders no lifespan number."""
|
||||
# No baseline → Unavailable
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
_insert_day(store, "2026-09-28", bw=100*1024*1024)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert result.headline_remaining_seconds is None
|
||||
|
||||
def test_unavailable_zero_rate_no_lifespan(self, store):
|
||||
"""Zero rate → Unavailable with no lifespan number."""
|
||||
_insert_baseline(store, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(store)
|
||||
_open_period(store)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=0)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert result.headline_remaining_seconds is None
|
||||
assert any("no finite projection" in f for f in result.contributing_facts)
|
||||
|
||||
|
||||
class TestStalenessDrop:
|
||||
"""PR-7: Newest day aggregate older than 48 h drops confidence one level,
|
||||
shown as a contributing fact."""
|
||||
|
||||
def test_staleness_drops_to_limited(self, store):
|
||||
"""Stale data (>48h) drops Supported → Limited."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw, coverage=0.95)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
# Clock is 3 days after last data → staleness > 48h
|
||||
clock = datetime(2026, 10, 3, 12, 0, 0, tzinfo=timezone.utc)
|
||||
result = compute_projection(store, clock)
|
||||
assert any("48h" in f or "stale" in f.lower() or "old" in f for f in result.contributing_facts)
|
||||
|
||||
def test_staleness_fact_shown(self, store):
|
||||
"""Staleness is shown as a contributing fact."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw, coverage=0.95)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
clock = datetime(2026, 10, 3, 12, 0, 0, tzinfo=timezone.utc)
|
||||
result = compute_projection(store, clock)
|
||||
assert result.staleness_fact is not None
|
||||
assert "old" in result.staleness_fact or "48h" in result.staleness_fact
|
||||
|
||||
def test_fresh_data_no_staleness_fact(self, store):
|
||||
"""Fresh data (<48h) produces no staleness fact."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw, coverage=0.95)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.staleness_fact is None
|
||||
|
||||
|
||||
class TestSegmentBreakProjection:
|
||||
"""PR-9: Segment breaks — DUW decrease keeps prior day aggregates as
|
||||
habit evidence with Unavailable until re-warm; identity change
|
||||
quarantines prior history entirely."""
|
||||
|
||||
def test_duw_decrease_keeps_prior_as_habit_evidence(self, store):
|
||||
"""DUW decrease: prior days remain in store, projection based on
|
||||
current segment days only."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
# First segment: Sep 1-15
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(15):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
# DUW decrease → new segment Sep 16
|
||||
_insert_segment(store, opened_at="2026-09-16T00:00:00+00:00")
|
||||
# 5 days in new segment
|
||||
for i in range(5):
|
||||
d = (datetime(2026, 9, 16) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-20T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=20, hour=12))
|
||||
# Prior days exist in store but projection uses current segment
|
||||
# 5 days in segment → regime_days = 5
|
||||
assert result.regime_days is not None
|
||||
assert result.regime_days <= 5
|
||||
|
||||
def test_duw_decrease_unavailable_until_rewarm(self, store):
|
||||
"""DUW decrease: projection Unavailable until new segment re-warms."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
# DUW decrease → new segment Sep 25; clear old days to avoid duplicates
|
||||
_insert_segment(store, opened_at="2026-09-25T00:00:00+00:00")
|
||||
store.execute("DELETE FROM day_aggregates WHERE day >= '2026-09-01'")
|
||||
store.commit()
|
||||
# Only 3 days in new segment (not enough for warming)
|
||||
for i in range(3):
|
||||
d = (datetime(2026, 9, 25) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-28T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=28, hour=12))
|
||||
# Young regime (3 days) → Limited, not enough data for full confidence
|
||||
assert result.confidence_state == ConfidenceState.LIMITED
|
||||
assert result.warming_fact is not None
|
||||
|
||||
def test_identity_change_quarantines_prior_history(self, store):
|
||||
"""Identity change: prior history quarantined entirely."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
# First segment with lots of data
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00",
|
||||
identity_key="nqn.drive-a")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
# Identity change → new segment Sep 25; clear old days
|
||||
_insert_segment(store, opened_at="2026-09-25T00:00:00+00:00",
|
||||
identity_key="nqn.drive-b")
|
||||
store.execute("DELETE FROM day_aggregates WHERE day >= '2026-09-01'")
|
||||
store.commit()
|
||||
# Only 3 days in new segment
|
||||
for i in range(3):
|
||||
d = (datetime(2026, 9, 25) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-28T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=28, hour=12))
|
||||
# Prior history quarantined; only 3 days in new segment
|
||||
assert result.regime_days is not None
|
||||
assert result.regime_days <= 3
|
||||
# Should be Limited due to young regime
|
||||
assert result.confidence_state == ConfidenceState.LIMITED
|
||||
|
||||
|
||||
class TestDegradedIdentity:
|
||||
"""PR-15: Degraded identity caps at Limited with fixed fact in every state;
|
||||
cap combines idempotently with staleness; ephemeral markers never render
|
||||
as confidence facts."""
|
||||
|
||||
def test_degraded_identity_fact_in_every_state(self, store):
|
||||
"""Degraded identity fact renders even when Unavailable."""
|
||||
_insert_segment(store, identity_key=None, degraded=True)
|
||||
_open_period(store)
|
||||
_insert_day(store, "2026-09-28", bw=100*1024*1024)
|
||||
# No baseline → Unavailable
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert any("controller identity unavailable" in f for f in result.contributing_facts)
|
||||
|
||||
def test_degraded_identity_caps_at_limited(self, store):
|
||||
"""Degraded identity makes Supported unreachable → Limited."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, identity_key=None, degraded=True,
|
||||
opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw, coverage=0.95)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
assert result.confidence_state == ConfidenceState.LIMITED
|
||||
assert any("controller identity unavailable" in f for f in result.contributing_facts)
|
||||
|
||||
def test_degraded_idempotent_with_staleness(self, store):
|
||||
"""Degraded + staleness both land at Limited (idempotent)."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, identity_key=None, degraded=True,
|
||||
opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw, coverage=0.95)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
# Stale clock (>48h)
|
||||
clock = datetime(2026, 10, 5, 12, 0, 0, tzinfo=timezone.utc)
|
||||
result = compute_projection(store, clock)
|
||||
# Both degraded and stale → still Limited (not worse)
|
||||
assert result.confidence_state == ConfidenceState.LIMITED
|
||||
assert any("controller identity unavailable" in f for f in result.contributing_facts)
|
||||
assert any("old" in f or "48h" in f for f in result.contributing_facts)
|
||||
|
||||
def test_ephemeral_markers_never_render_as_facts(self, store):
|
||||
"""Model 'Linux' and non-pcie transport never appear as confidence facts."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True, model="Linux")
|
||||
_insert_segment(store, identity_key="nqn.test", degraded=False,
|
||||
opened_at="2026-09-01T00:00:00+00:00", mn="Linux")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw, coverage=0.95)
|
||||
_insert_sample(store, "2026-09-30T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock())
|
||||
for fact in result.contributing_facts:
|
||||
# Ephemeral markers (model, transport) never appear as confidence facts
|
||||
assert "transport" not in fact.lower() or "transport" in fact.lower()
|
||||
# The key check: model name should not appear as a confidence-quality fact
|
||||
# (it may appear in baseline label, but not in confidence contributing facts)
|
||||
confidence_facts = [f for f in result.contributing_facts
|
||||
if f not in ["verified manufacturer TBW", "no applicable endurance baseline"]]
|
||||
# No fact should mention transport as a quality indicator
|
||||
for cf in confidence_facts:
|
||||
assert "non-pcie" not in cf.lower()
|
||||
assert "usb transport" not in cf.lower()
|
||||
|
||||
|
||||
class TestIdentityChangeBlankKeys:
|
||||
"""PR-16: Identity-change semantics extend to blank keys verbatim —
|
||||
to/from blank quarantines, equal blanks continue."""
|
||||
|
||||
def test_to_blank_quarantines_in_projection(self, store):
|
||||
"""Transition to blank key quarantines prior history."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
# First segment: healthy key
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00",
|
||||
identity_key="nqn.healthy")
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
# Blank key → new segment Sep 21
|
||||
_insert_segment(store, opened_at="2026-09-21T00:00:00+00:00",
|
||||
identity_key=None, degraded=True)
|
||||
for i in range(5):
|
||||
d = (datetime(2026, 9, 21) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-26T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=26, hour=12))
|
||||
# Prior history quarantined; only 5 days in new segment
|
||||
assert result.regime_days is not None
|
||||
assert result.regime_days <= 5
|
||||
|
||||
def test_from_blank_quarantines_in_projection(self, store):
|
||||
"""Transition from blank to healthy key quarantines prior history."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
# First segment: blank key
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00",
|
||||
identity_key=None, degraded=True)
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
# Healthy key → new segment Sep 21
|
||||
_insert_segment(store, opened_at="2026-09-21T00:00:00+00:00",
|
||||
identity_key="nqn.restored")
|
||||
for i in range(5):
|
||||
d = (datetime(2026, 9, 21) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-26T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=26, hour=12))
|
||||
assert result.regime_days is not None
|
||||
assert result.regime_days <= 5
|
||||
|
||||
def test_equal_blanks_continue_segment(self, store):
|
||||
"""Equal blank keys continue the segment (no quarantine)."""
|
||||
_insert_baseline(store, tbw_tb=10.0, verified=True)
|
||||
_insert_segment(store, opened_at="2026-09-01T00:00:00+00:00",
|
||||
identity_key=None, degraded=True)
|
||||
_open_period(store, start="2026-09-01T00:00:00+00:00")
|
||||
bw = 100 * 1024 * 1024
|
||||
for i in range(25):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(store, d, bw=bw)
|
||||
_insert_sample(store, "2026-09-26T10:00:00+00:00", pu=5)
|
||||
result = compute_projection(store, _clock(year=2026, month=9, day=26, hour=12))
|
||||
# All 25 days in same segment (equal blanks continue)
|
||||
assert result.regime_days is not None
|
||||
assert result.regime_days >= 20 # Most of the history
|
||||
@@ -1,104 +0,0 @@
|
||||
"""Raw sample pruning tests.
|
||||
|
||||
Spec §3.4, ST-5: Raw samples pruned to 14 days; hour observations and
|
||||
day aggregates retained indefinitely.
|
||||
"""
|
||||
import sqlite3
|
||||
import sys
|
||||
from datetime import datetime, timezone, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.store import init_store
|
||||
from fenris.pruning import prune_old_samples
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store_conn(tmp_path: Path):
|
||||
db_path = tmp_path / "test.db"
|
||||
conn = init_store(db_path)
|
||||
yield conn
|
||||
conn.close()
|
||||
|
||||
|
||||
def _insert_sample(conn, ts_iso, device="/dev/nvme0"):
|
||||
conn.execute(
|
||||
"INSERT INTO samples (ts, device, data_units_written, data_units_read, "
|
||||
" bytes_written, bytes_read, percentage_used) VALUES (?, ?, 0, 0, 0, 0, 0)",
|
||||
(ts_iso, device),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
class TestPruneOldSamples:
|
||||
"""Spec §3.4: Raw samples pruned to 14 days."""
|
||||
|
||||
def test_keeps_recent_samples(self, store_conn):
|
||||
now = datetime(2026, 9, 15, 12, 0, 0, tzinfo=timezone.utc)
|
||||
# Insert a sample 1 day ago
|
||||
ts = (now - timedelta(days=1)).isoformat()
|
||||
_insert_sample(store_conn, ts)
|
||||
|
||||
pruned = prune_old_samples(store_conn, now, retention_days=14)
|
||||
assert pruned == 0
|
||||
|
||||
cursor = store_conn.execute("SELECT COUNT(*) FROM samples")
|
||||
assert cursor.fetchone()[0] == 1
|
||||
|
||||
def test_removes_old_samples(self, store_conn):
|
||||
now = datetime(2026, 9, 15, 12, 0, 0, tzinfo=timezone.utc)
|
||||
# Insert samples at 10, 14, and 15 days ago
|
||||
for days_ago in [10, 14, 15]:
|
||||
ts = (now - timedelta(days=days_ago)).isoformat()
|
||||
_insert_sample(store_conn, ts)
|
||||
|
||||
pruned = prune_old_samples(store_conn, now, retention_days=14)
|
||||
assert pruned == 1 # Only the 15-day-old sample removed
|
||||
|
||||
cursor = store_conn.execute("SELECT COUNT(*) FROM samples")
|
||||
assert cursor.fetchone()[0] == 2
|
||||
|
||||
def test_removes_many_old_samples(self, store_conn):
|
||||
now = datetime(2026, 9, 15, 12, 0, 0, tzinfo=timezone.utc)
|
||||
for days_ago in range(1, 30):
|
||||
ts = (now - timedelta(days=days_ago)).isoformat()
|
||||
_insert_sample(store_conn, ts)
|
||||
|
||||
pruned = prune_old_samples(store_conn, now, retention_days=14)
|
||||
assert pruned == 15 # Days 15-29 removed
|
||||
|
||||
cursor = store_conn.execute("SELECT COUNT(*) FROM samples")
|
||||
assert cursor.fetchone()[0] == 14 # Days 1-14 kept
|
||||
|
||||
def test_empty_store_no_error(self, store_conn):
|
||||
now = datetime(2026, 9, 15, 12, 0, 0, tzinfo=timezone.utc)
|
||||
pruned = prune_old_samples(store_conn, now, retention_days=14)
|
||||
assert pruned == 0
|
||||
|
||||
def test_hour_observations_not_pruned(self, store_conn):
|
||||
"""Hour observations are retained indefinitely."""
|
||||
now = datetime(2026, 9, 15, 12, 0, 0, tzinfo=timezone.utc)
|
||||
# Insert an old sample and a recent sample
|
||||
_insert_sample(store_conn, (now - timedelta(days=20)).isoformat())
|
||||
_insert_sample(store_conn, (now - timedelta(days=1)).isoformat())
|
||||
|
||||
# Insert an old hour observation
|
||||
store_conn.execute(
|
||||
"INSERT INTO hour_observations (hour, active_seconds, sample_count) "
|
||||
"VALUES (?, 3600, 1)",
|
||||
((now - timedelta(days=20)).replace(hour=0, minute=0, second=0).isoformat(),),
|
||||
)
|
||||
store_conn.commit()
|
||||
|
||||
prune_old_samples(store_conn, now, retention_days=14)
|
||||
|
||||
# Sample removed
|
||||
cursor = store_conn.execute("SELECT COUNT(*) FROM samples")
|
||||
assert cursor.fetchone()[0] == 1
|
||||
|
||||
# Hour observation retained
|
||||
cursor = store_conn.execute("SELECT COUNT(*) FROM hour_observations")
|
||||
assert cursor.fetchone()[0] == 1
|
||||
@@ -1,390 +0,0 @@
|
||||
"""Release flow tests (issue #52).
|
||||
|
||||
Tests the one-command release flow: build, sign, publish, and attach — with
|
||||
dry-run mode that is what the tests assert. All assertions are structural:
|
||||
dry-run output contains the expected commands without any network or registry
|
||||
access.
|
||||
|
||||
Requirements:
|
||||
- scripts/release.sh exists and is executable
|
||||
- No network access required for dry-run tests
|
||||
- No GPG key or registry token required for dry-run tests
|
||||
|
||||
Spec: release-packaging.md §5, issue #52 acceptance criteria
|
||||
"""
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
RELEASE_SCRIPT = REPO_ROOT / "scripts" / "release.sh"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _read(path: str | Path) -> str:
|
||||
from tests.conftest import read
|
||||
return read(path)
|
||||
|
||||
|
||||
def _get_version() -> str:
|
||||
"""Extract version from pyproject.toml."""
|
||||
from tests.conftest import get_version
|
||||
return get_version()
|
||||
|
||||
|
||||
def _run_dry_run(*args: str) -> tuple[int, str]:
|
||||
"""Run the release script in dry-run mode and return (exit_code, stdout)."""
|
||||
cmd = ["bash", str(RELEASE_SCRIPT), "--dry-run"] + list(args)
|
||||
r = subprocess.run(
|
||||
cmd, capture_output=True, text=True, timeout=30,
|
||||
cwd=REPO_ROOT,
|
||||
)
|
||||
return r.returncode, r.stdout + r.stderr
|
||||
|
||||
|
||||
def _deb_filename(version: str, release: int = 1) -> str:
|
||||
"""Expected deb filename for a given version and release."""
|
||||
return f"fenris_{version}_amd64.deb"
|
||||
|
||||
|
||||
def _rpm_filename(version: str, release: int = 1) -> str:
|
||||
"""Expected RPM filename for a given version and release."""
|
||||
return f"fenris-{version}-{release}.x86_64.rpm"
|
||||
|
||||
|
||||
def _registry_upload_deb_url(version: str) -> str:
|
||||
"""Expected registry upload URL for a deb package."""
|
||||
return f"debian/pool/bookworm/main/upload"
|
||||
|
||||
|
||||
def _registry_upload_rpm_url() -> str:
|
||||
"""Expected registry upload URL for an RPM package."""
|
||||
return "rpm/fenris/upload"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — release script existence and permissions
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestReleaseScriptExists:
|
||||
"""Verify the release script is present and executable."""
|
||||
|
||||
def test_script_exists(self):
|
||||
assert RELEASE_SCRIPT.exists(), \
|
||||
"scripts/release.sh must exist"
|
||||
|
||||
def test_script_is_executable(self):
|
||||
assert RELEASE_SCRIPT.stat().st_mode & 0o111, \
|
||||
"scripts/release.sh must be executable"
|
||||
|
||||
def test_script_has_shebang(self):
|
||||
first_line = RELEASE_SCRIPT.read_text().splitlines()[0]
|
||||
assert first_line.startswith("#!/"), \
|
||||
"scripts/release.sh must have a shebang"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — dry-run prints all expected commands
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestDryRunCommandPrintout:
|
||||
"""Verify dry-run prints every command that would execute."""
|
||||
|
||||
def test_dry_run_exits_zero(self):
|
||||
rc, _ = _run_dry_run()
|
||||
assert rc == 0, "Dry-run must exit zero"
|
||||
|
||||
def test_dry_run_prints_make_package(self):
|
||||
_, output = _run_dry_run()
|
||||
assert "make" in output.lower() and "package" in output.lower(), \
|
||||
"Dry-run must print the make package command"
|
||||
|
||||
def test_dry_run_prints_rpm_signing(self):
|
||||
_, output = _run_dry_run()
|
||||
assert "rpmsign" in output or "sign" in output.lower(), \
|
||||
"Dry-run must print RPM signing step"
|
||||
|
||||
def test_dry_run_prints_sha256sums(self):
|
||||
_, output = _run_dry_run()
|
||||
assert "sha256sum" in output, \
|
||||
"Dry-run must print SHA256SUMS generation"
|
||||
|
||||
def test_dry_run_prints_clearsign(self):
|
||||
_, output = _run_dry_run()
|
||||
assert "clearsign" in output or "SHA256SUMS.asc" in output, \
|
||||
"Dry-run must print clearsign step"
|
||||
|
||||
def test_dry_run_prints_deb_upload(self):
|
||||
version = _get_version()
|
||||
_, output = _run_dry_run()
|
||||
assert _registry_upload_deb_url(version) in output, \
|
||||
f"Dry-run must print deb upload URL ({_registry_upload_deb_url(version)})"
|
||||
|
||||
def test_dry_run_prints_deb_upload_for_all_codenames(self):
|
||||
_, output = _run_dry_run()
|
||||
for codename in ("bookworm", "jammy", "noble"):
|
||||
assert codename in output, \
|
||||
f"Dry-run must include upload for {codename}"
|
||||
|
||||
def test_dry_run_prints_rpm_upload(self):
|
||||
_, output = _run_dry_run()
|
||||
assert _registry_upload_rpm_url() in output, \
|
||||
f"Dry-run must print RPM upload URL ({_registry_upload_rpm_url()})"
|
||||
|
||||
def test_dry_run_prints_release_creation(self):
|
||||
_, output = _run_dry_run()
|
||||
assert "release" in output.lower(), \
|
||||
"Dry-run must print release creation step"
|
||||
|
||||
def test_dry_run_prints_attachment_upload(self):
|
||||
_, output = _run_dry_run()
|
||||
assert "SHA256SUMS.asc" in output, \
|
||||
"Dry-run must print SHA256SUMS.asc attachment upload"
|
||||
|
||||
def test_dry_run_prints_tag_push(self):
|
||||
_, output = _run_dry_run()
|
||||
# The tag is created atomically by the Gitea release API (step 5),
|
||||
# not by a separate git push. Verify the release creation step is present.
|
||||
assert "tag_name" in output or "release" in output.lower(), \
|
||||
"Dry-run must print release creation (which creates the tag)"
|
||||
|
||||
def test_dry_run_no_network_calls(self):
|
||||
"""Dry-run must not execute curl, rpmsign, or any network tools."""
|
||||
_, output = _run_dry_run()
|
||||
# The dry-run mode prints a marker at the top; all commands are
|
||||
# echoed (prefixed by spaces) but never executed. Verify the
|
||||
# marker is present, confirming we're in dry-run mode.
|
||||
assert "[dry-run]" in output, \
|
||||
"Output must contain [dry-run] marker"
|
||||
# Verify dangerous tools only appear as printed commands (not executed).
|
||||
# Printed commands are indented; the dry-run section header confirms
|
||||
# no commands were actually run.
|
||||
assert "Commands below will be executed" in output, \
|
||||
"Dry-run must indicate commands are for display only"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — dry-run prints correct package filenames
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestDryRunFilenames:
|
||||
"""Verify dry-run uses the correct artifact filenames."""
|
||||
|
||||
def test_deb_filename_in_output(self):
|
||||
version = _get_version()
|
||||
_, output = _run_dry_run()
|
||||
expected = _deb_filename(version)
|
||||
assert expected in output, \
|
||||
f"Dry-run must reference deb filename {expected}"
|
||||
|
||||
def test_rpm_filename_in_output(self):
|
||||
version = _get_version()
|
||||
_, output = _run_dry_run()
|
||||
expected = _rpm_filename(version)
|
||||
assert expected in output, \
|
||||
f"Dry-run must reference RPM filename {expected}"
|
||||
|
||||
def test_checksums_filename_in_output(self):
|
||||
_, output = _run_dry_run()
|
||||
assert "SHA256SUMS" in output, \
|
||||
"Dry-run must reference SHA256SUMS filename"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — dry-run does not create artifacts or tags
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestDryRunNoSideEffects:
|
||||
"""Verify dry-run creates no filesystem or git side effects."""
|
||||
|
||||
def test_dry_run_no_git_tag_created(self):
|
||||
version = _get_version()
|
||||
tag = f"v{version}"
|
||||
# Ensure tag doesn't exist before
|
||||
r = subprocess.run(
|
||||
["git", "tag", "-l", tag], capture_output=True, text=True,
|
||||
cwd=REPO_ROOT,
|
||||
)
|
||||
pre_tags = r.stdout.strip()
|
||||
|
||||
_run_dry_run()
|
||||
|
||||
# Verify tag was not created
|
||||
r = subprocess.run(
|
||||
["git", "tag", "-l", tag], capture_output=True, text=True,
|
||||
cwd=REPO_ROOT,
|
||||
)
|
||||
post_tags = r.stdout.strip()
|
||||
assert pre_tags == post_tags, \
|
||||
f"Dry-run must not create git tag {tag}"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — revision bumping (structural: output contains incremented release)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestRevisionBumping:
|
||||
"""Verify the release script handles revision bumping.
|
||||
|
||||
When a version already exists in the registry (HTTP 409), the script
|
||||
bumps the revision and retries. These tests verify the dry-run output
|
||||
reflects the correct revision logic — without any network access.
|
||||
"""
|
||||
|
||||
def test_dry_run_starts_at_revision_one(self):
|
||||
version = _get_version()
|
||||
_, output = _run_dry_run()
|
||||
rpm_expected = _rpm_filename(version, 1)
|
||||
assert rpm_expected in output, \
|
||||
f"Dry-run must start at release 1: expected {rpm_expected} in output"
|
||||
|
||||
def test_revision_bump_changes_rpm_filename(self):
|
||||
"""When revision is bumped, the RPM filename changes accordingly."""
|
||||
version = _get_version()
|
||||
rpm_r1 = _rpm_filename(version, 1)
|
||||
rpm_r2 = _rpm_filename(version, 2)
|
||||
# R2 filename must differ from R1
|
||||
assert rpm_r1 != rpm_r2, \
|
||||
"R2 filename must differ from R1"
|
||||
# Both must contain the version
|
||||
assert version in rpm_r1
|
||||
assert version in rpm_r2
|
||||
|
||||
def test_revision_bump_changes_deb_filename(self):
|
||||
"""When revision is bumped, the deb filename also changes."""
|
||||
version = _get_version()
|
||||
# Deb filename includes release in nfpm naming
|
||||
deb_r1 = f"fenris_{version}_amd64.deb"
|
||||
deb_r2 = f"fenris_{version}_amd64.deb"
|
||||
# For deb, the filename doesn't change with revision (deb uses epoch)
|
||||
# But the RPM does — this verifies we test RPM revision correctly
|
||||
rpm_r1 = _rpm_filename(version, 1)
|
||||
rpm_r2 = _rpm_filename(version, 2)
|
||||
assert "-1." in rpm_r1, "R1 RPM must contain -1."
|
||||
assert "-2." in rpm_r2, "R2 RPM must contain -2."
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — bare tag prevention (structural)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestBareTagPrevention:
|
||||
"""Verify the flow prevents bare tags.
|
||||
|
||||
A bare tag (tag without packages, release entry, notes, and checksums)
|
||||
must not result from the flow. The script checks for existing bare
|
||||
tags before proceeding. These tests verify the dry-run doesn't create
|
||||
any tags.
|
||||
"""
|
||||
|
||||
def test_dry_run_does_not_push_tag(self):
|
||||
_, output = _run_dry_run()
|
||||
# The dry-run marker confirms no commands are executed.
|
||||
# git push appears only as a printed command, never executed.
|
||||
assert "[dry-run]" in output, \
|
||||
"Must be in dry-run mode"
|
||||
# Tag push is printed but the [dry-run] marker confirms nothing ran
|
||||
assert "Commands below will be executed" in output, \
|
||||
"Dry-run must indicate commands are for display only"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — CI workflow file
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestCIWorkflow:
|
||||
"""Verify the dormant CI workflow is present and correctly structured."""
|
||||
|
||||
def test_workflow_file_exists(self):
|
||||
path = REPO_ROOT / ".gitea" / "workflows" / "release.yml"
|
||||
assert path.exists(), \
|
||||
".gitea/workflows/release.yml must exist"
|
||||
|
||||
def test_workflow_triggers_on_tags(self):
|
||||
content = _read(".gitea/workflows/release.yml")
|
||||
assert "v*" in content, \
|
||||
"Workflow must trigger on version tags (v*)"
|
||||
|
||||
def test_workflow_has_release_step(self):
|
||||
content = _read(".gitea/workflows/release.yml")
|
||||
assert "release" in content.lower(), \
|
||||
"Workflow must have a release step"
|
||||
|
||||
def test_workflow_mentions_signing(self):
|
||||
content = _read(".gitea/workflows/release.yml")
|
||||
assert "sign" in content.lower(), \
|
||||
"Workflow must include signing step"
|
||||
|
||||
def test_workflow_mentions_upload(self):
|
||||
content = _read(".gitea/workflows/release.yml")
|
||||
assert "upload" in content.lower() or "publish" in content.lower(), \
|
||||
"Workflow must include upload/publish step"
|
||||
|
||||
def test_workflow_validates_notes_before_publication(self):
|
||||
content = _read(".gitea/workflows/release.yml")
|
||||
assert "scripts/extract_changelog.py" in content, \
|
||||
"Workflow must fail before publication if release notes cannot be extracted"
|
||||
assert "--footer packaging/release-footer.md" in content, \
|
||||
"Workflow must assemble the body from the standing release footer"
|
||||
|
||||
def test_workflow_resynchronizes_existing_release_bodies(self):
|
||||
content = _read(".gitea/workflows/release.yml")
|
||||
assert "scripts/release_request.py" in content, \
|
||||
"Workflow must make the create-versus-update decision through the request seam"
|
||||
assert '"${METHOD}"' in content, \
|
||||
"Workflow must execute the helper-selected create-or-update request"
|
||||
assert 'RELEASE_PATH="$(printf' in content and '\n PATH="$(printf' not in content, \
|
||||
"Workflow must not overwrite the shell PATH while preparing the request URL"
|
||||
|
||||
|
||||
def test_readme_points_consumers_to_release_notes():
|
||||
readme = _read("README.md")
|
||||
|
||||
assert "Per-release notes live on the [releases page]" in readme
|
||||
assert "standing install and verification instructions" in readme
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — Makefile release targets
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestMakefileReleaseTargets:
|
||||
"""Verify the Makefile exposes release-related targets."""
|
||||
|
||||
def _makefile_content(self) -> str:
|
||||
return _read("Makefile")
|
||||
|
||||
def test_release_run_target_exists(self):
|
||||
content = self._makefile_content()
|
||||
assert "release-run:" in content, \
|
||||
"Makefile must have a release-run target"
|
||||
|
||||
def test_release_dry_run_target_exists(self):
|
||||
content = self._makefile_content()
|
||||
assert "release-dry-run:" in content, \
|
||||
"Makefile must have a release-dry-run target"
|
||||
|
||||
def test_release_run_calls_script(self):
|
||||
content = self._makefile_content()
|
||||
assert "release.sh" in content, \
|
||||
"release-run target must call scripts/release.sh"
|
||||
|
||||
def test_release_dry_run_uses_dry_run_flag(self):
|
||||
content = self._makefile_content()
|
||||
# Find the release-dry-run target and verify it passes --dry-run
|
||||
in_target = False
|
||||
for line in content.splitlines():
|
||||
if line.startswith("release-dry-run:"):
|
||||
in_target = True
|
||||
continue
|
||||
if in_target and line.strip():
|
||||
if "--dry-run" in line:
|
||||
break
|
||||
if not line.startswith("\t"):
|
||||
break
|
||||
else:
|
||||
pytest.fail("release-dry-run target must pass --dry-run to release.sh")
|
||||
@@ -1,465 +0,0 @@
|
||||
"""Controller segmentation tests.
|
||||
|
||||
Covers acceptance criteria:
|
||||
- ID-1: Identity key ladder, FR metadata only, independent axes
|
||||
- ID-2: Frozen metadata snapshot at segment open
|
||||
- ID-4: identity_degraded set exactly when key is blank
|
||||
- AC-4: vid/ssvid from PCI node, stored null otherwise
|
||||
- PR-16: Blank-key semantics (to/from blank quarantines, equal blanks continue)
|
||||
"""
|
||||
import sqlite3
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict
|
||||
|
||||
import pytest
|
||||
|
||||
import sys
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.store import init_store
|
||||
from fenris.collector import normalize_identity, compute_identity_degraded, acquire_from_sysfs
|
||||
from fenris.segment import find_current_segment, should_open_new_segment, open_segment
|
||||
|
||||
|
||||
# Fixtures
|
||||
|
||||
@pytest.fixture
|
||||
def store(tmp_path: Path) -> sqlite3.Connection:
|
||||
conn = init_store(tmp_path / "observations.db")
|
||||
yield conn
|
||||
conn.close()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def clock():
|
||||
class FakeClock:
|
||||
def __init__(self):
|
||||
self.now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
def utcnow(self):
|
||||
return self.now
|
||||
return FakeClock()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def identity_nqn() -> Dict[str, Any]:
|
||||
return {
|
||||
"subnqn": "nqn.2014-08.org.nvmexpress:uuid:12345678-1234-1234-1234-123456789abc",
|
||||
"mn": "Samsung SSD 970 EVO Plus 1TB",
|
||||
"sn": "S4EWNX0N123456",
|
||||
"fr": "2B2QEXM7",
|
||||
"vid": "0x144d",
|
||||
"ssvid": "0x144d",
|
||||
"transport": "pcie",
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def identity_model_serial() -> Dict[str, Any]:
|
||||
return {
|
||||
"subnqn": "",
|
||||
"mn": "Samsung SSD 970 EVO Plus 1TB",
|
||||
"sn": "S4EWNX0N123456",
|
||||
"fr": "2B2QEXM7",
|
||||
"vid": "0x144d",
|
||||
"ssvid": "0x144d",
|
||||
"transport": "pcie",
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def identity_degraded() -> Dict[str, Any]:
|
||||
return {
|
||||
"subnqn": "",
|
||||
"mn": "",
|
||||
"sn": "",
|
||||
"fr": "2B2QEXM7",
|
||||
"vid": "0x144d",
|
||||
"ssvid": "0x144d",
|
||||
"transport": "pcie",
|
||||
}
|
||||
|
||||
|
||||
# ID-1: Identity key ladder
|
||||
|
||||
class TestIdentityKeyLadder:
|
||||
def test_subnqn_primary(self, identity_nqn):
|
||||
key = normalize_identity(identity_nqn)
|
||||
assert key == "nqn.2014-08.org.nvmexpress:uuid:12345678-1234-1234-1234-123456789abc"
|
||||
|
||||
def test_fallback_to_model_serial(self, identity_model_serial):
|
||||
key = normalize_identity(identity_model_serial)
|
||||
assert key == "Samsung SSD 970 EVO Plus 1TB|S4EWNX0N123456"
|
||||
|
||||
def test_all_blank_degraded(self, identity_degraded):
|
||||
key = normalize_identity(identity_degraded)
|
||||
assert key == ""
|
||||
|
||||
def test_firmware_is_metadata_only(self, identity_nqn):
|
||||
key = normalize_identity(identity_nqn)
|
||||
assert "2B2QEXM7" not in key
|
||||
|
||||
|
||||
# ID-2: Frozen metadata snapshot
|
||||
|
||||
class TestFrozenMetadataSnapshot:
|
||||
def test_segment_stores_all_metadata_fields(self, store, clock, identity_nqn):
|
||||
key = normalize_identity(identity_nqn)
|
||||
degraded = compute_identity_degraded(identity_nqn)
|
||||
|
||||
segment = open_segment(store, clock.now, identity_nqn, key, degraded)
|
||||
|
||||
assert segment["identity_key"] == key
|
||||
assert segment["identity_degraded"] is False
|
||||
assert segment["subnqn"] == identity_nqn["subnqn"]
|
||||
assert segment["sn"] == identity_nqn["sn"]
|
||||
assert segment["mn"] == identity_nqn["mn"]
|
||||
assert segment["fr"] == identity_nqn["fr"]
|
||||
assert segment["vid"] == identity_nqn["vid"]
|
||||
assert segment["ssvid"] == identity_nqn["ssvid"]
|
||||
assert segment["transport"] == identity_nqn["transport"]
|
||||
|
||||
def test_metadata_nullable(self, store, clock):
|
||||
sparse_identity = {
|
||||
"subnqn": "",
|
||||
"mn": "Legacy Model",
|
||||
"sn": "LEGACY123",
|
||||
"fr": "1.0",
|
||||
"vid": None,
|
||||
"ssvid": None,
|
||||
"transport": None,
|
||||
}
|
||||
|
||||
key = normalize_identity(sparse_identity)
|
||||
degraded = compute_identity_degraded(sparse_identity)
|
||||
segment = open_segment(store, clock.now, sparse_identity, key, degraded)
|
||||
|
||||
assert segment["vid"] is None
|
||||
assert segment["ssvid"] is None
|
||||
assert segment["transport"] is None
|
||||
|
||||
def test_cntlid_excluded(self, store, clock, identity_nqn):
|
||||
key = normalize_identity(identity_nqn)
|
||||
degraded = compute_identity_degraded(identity_nqn)
|
||||
|
||||
segment = open_segment(store, clock.now, identity_nqn, key, degraded)
|
||||
|
||||
assert "cntlid" not in segment
|
||||
|
||||
def test_metadata_immutable_after_open(self, store, clock, identity_nqn):
|
||||
key = normalize_identity(identity_nqn)
|
||||
degraded = compute_identity_degraded(identity_nqn)
|
||||
|
||||
open_segment(store, clock.now, identity_nqn, key, degraded)
|
||||
|
||||
found = find_current_segment(store)
|
||||
assert found["subnqn"] == identity_nqn["subnqn"]
|
||||
assert found["vid"] == identity_nqn["vid"]
|
||||
assert found["ssvid"] == identity_nqn["ssvid"]
|
||||
|
||||
|
||||
# ID-4: identity_degraded
|
||||
|
||||
class TestIdentityDegraded:
|
||||
def test_degraded_when_blank_key(self, store, clock, identity_degraded):
|
||||
key = normalize_identity(identity_degraded)
|
||||
degraded = compute_identity_degraded(identity_degraded)
|
||||
|
||||
assert key == ""
|
||||
assert degraded is True
|
||||
|
||||
segment = open_segment(store, clock.now, identity_degraded, key, degraded)
|
||||
assert segment["identity_degraded"] is True
|
||||
|
||||
def test_not_degraded_with_subnqn(self, store, clock, identity_nqn):
|
||||
key = normalize_identity(identity_nqn)
|
||||
degraded = compute_identity_degraded(identity_nqn)
|
||||
|
||||
assert key != ""
|
||||
assert degraded is False
|
||||
|
||||
segment = open_segment(store, clock.now, identity_nqn, key, degraded)
|
||||
assert segment["identity_degraded"] is False
|
||||
|
||||
def test_not_degraded_with_model_serial(self, store, clock, identity_model_serial):
|
||||
key = normalize_identity(identity_model_serial)
|
||||
degraded = compute_identity_degraded(identity_model_serial)
|
||||
|
||||
assert key != ""
|
||||
assert degraded is False
|
||||
|
||||
segment = open_segment(store, clock.now, identity_model_serial, key, degraded)
|
||||
assert segment["identity_degraded"] is False
|
||||
|
||||
|
||||
# AC-4: vid/ssvid from PCI node
|
||||
|
||||
class TestVidSsvidAcquisition:
|
||||
def test_vid_ssvid_from_pci_node(self, tmp_path: Path):
|
||||
ctrl_dir = tmp_path / "nvme0"
|
||||
ctrl_dir.mkdir()
|
||||
|
||||
(ctrl_dir / "subsysnqn").write_text("nqn.test\n")
|
||||
(ctrl_dir / "model").write_text("Test Model\n")
|
||||
(ctrl_dir / "serial").write_text("TEST123\n")
|
||||
(ctrl_dir / "firmware_rev").write_text("1.0\n")
|
||||
(ctrl_dir / "vendor").write_text("0x144d\n")
|
||||
(ctrl_dir / "subsystem_vendor").write_text("0x144d\n")
|
||||
|
||||
identity = acquire_from_sysfs(ctrl_dir)
|
||||
|
||||
assert identity["vid"] == "0x144d"
|
||||
assert identity["ssvid"] == "0x144d"
|
||||
|
||||
def test_vid_ssvid_null_when_absent(self, tmp_path: Path):
|
||||
ctrl_dir = tmp_path / "nvme0"
|
||||
ctrl_dir.mkdir()
|
||||
|
||||
(ctrl_dir / "subsysnqn").write_text("nqn.test\n")
|
||||
(ctrl_dir / "model").write_text("Test Model\n")
|
||||
(ctrl_dir / "serial").write_text("TEST123\n")
|
||||
(ctrl_dir / "firmware_rev").write_text("1.0\n")
|
||||
|
||||
identity = acquire_from_sysfs(ctrl_dir)
|
||||
|
||||
assert identity["vid"] is None
|
||||
assert identity["ssvid"] is None
|
||||
|
||||
def test_vid_ssvid_metadata_only(self, store, clock, identity_nqn):
|
||||
identity_a = {**identity_nqn, "vid": "0x144d", "ssvid": "0x144d"}
|
||||
identity_b = {**identity_nqn, "vid": "0xFFFF", "ssvid": "0xFFFF"}
|
||||
|
||||
key_a = normalize_identity(identity_a)
|
||||
key_b = normalize_identity(identity_b)
|
||||
|
||||
assert key_a == key_b
|
||||
|
||||
|
||||
# PR-16: Blank-key semantics
|
||||
|
||||
class TestBlankKeySemantics:
|
||||
def test_to_blank_quarantines(self, store, clock):
|
||||
healthy = {
|
||||
"subnqn": "nqn.healthy",
|
||||
"mn": "Model A", "sn": "SN1", "fr": "1.0",
|
||||
"vid": "0x1", "ssvid": "0x1", "transport": "pcie",
|
||||
}
|
||||
key_healthy = normalize_identity(healthy)
|
||||
degraded_healthy = compute_identity_degraded(healthy)
|
||||
|
||||
open_segment(store, clock.now, healthy, key_healthy, degraded_healthy)
|
||||
current = find_current_segment(store)
|
||||
|
||||
blank = {
|
||||
"subnqn": "", "mn": "", "sn": "",
|
||||
"fr": "1.0", "vid": "0x1", "ssvid": "0x1", "transport": "pcie",
|
||||
}
|
||||
key_blank = normalize_identity(blank)
|
||||
|
||||
should_open, reason = should_open_new_segment(
|
||||
current, key_blank, 1000, store
|
||||
)
|
||||
|
||||
assert should_open is True
|
||||
assert reason == "identity_change"
|
||||
|
||||
def test_from_blank_quarantines(self, store, clock):
|
||||
blank = {
|
||||
"subnqn": "", "mn": "", "sn": "",
|
||||
"fr": "1.0", "vid": "0x1", "ssvid": "0x1", "transport": "pcie",
|
||||
}
|
||||
key_blank = normalize_identity(blank)
|
||||
degraded_blank = compute_identity_degraded(blank)
|
||||
|
||||
open_segment(store, clock.now, blank, key_blank, degraded_blank)
|
||||
current = find_current_segment(store)
|
||||
|
||||
healthy = {
|
||||
"subnqn": "nqn.healthy",
|
||||
"mn": "Model A", "sn": "SN1", "fr": "1.0",
|
||||
"vid": "0x1", "ssvid": "0x1", "transport": "pcie",
|
||||
}
|
||||
key_healthy = normalize_identity(healthy)
|
||||
|
||||
should_open, reason = should_open_new_segment(
|
||||
current, key_healthy, 1000, store
|
||||
)
|
||||
|
||||
assert should_open is True
|
||||
assert reason == "identity_change"
|
||||
|
||||
def test_equal_blanks_continue_by_duw(self, store, clock):
|
||||
blank = {
|
||||
"subnqn": "", "mn": "", "sn": "",
|
||||
"fr": "1.0", "vid": "0x1", "ssvid": "0x1", "transport": "pcie",
|
||||
}
|
||||
key_blank = normalize_identity(blank)
|
||||
degraded_blank = compute_identity_degraded(blank)
|
||||
|
||||
open_segment(store, clock.now, blank, key_blank, degraded_blank)
|
||||
current = find_current_segment(store)
|
||||
|
||||
store.execute(
|
||||
"INSERT INTO samples (ts, device, bytes_written) VALUES (?, ?, ?)",
|
||||
("2026-09-01T12:00:00Z", "/dev/nvme0", 1000)
|
||||
)
|
||||
store.commit()
|
||||
|
||||
should_open, reason = should_open_new_segment(
|
||||
current, key_blank, 2000, store
|
||||
)
|
||||
|
||||
assert should_open is False
|
||||
assert reason is None
|
||||
|
||||
def test_equal_blanks_duw_decrease_opens(self, store, clock):
|
||||
blank = {
|
||||
"subnqn": "", "mn": "", "sn": "",
|
||||
"fr": "1.0", "vid": "0x1", "ssvid": "0x1", "transport": "pcie",
|
||||
}
|
||||
key_blank = normalize_identity(blank)
|
||||
degraded_blank = compute_identity_degraded(blank)
|
||||
|
||||
open_segment(store, clock.now, blank, key_blank, degraded_blank)
|
||||
current = find_current_segment(store)
|
||||
|
||||
store.execute(
|
||||
"INSERT INTO samples (ts, device, bytes_written) VALUES (?, ?, ?)",
|
||||
("2026-09-01T12:00:00Z", "/dev/nvme0", 2000)
|
||||
)
|
||||
store.commit()
|
||||
|
||||
should_open, reason = should_open_new_segment(
|
||||
current, key_blank, 1000, store
|
||||
)
|
||||
|
||||
assert should_open is True
|
||||
assert reason == "duw_decrease"
|
||||
|
||||
|
||||
# Independent segmentation axes
|
||||
|
||||
class TestIndependentAxes:
|
||||
def test_identity_change_with_duw_increase(self, store, clock):
|
||||
identity_a = {
|
||||
"subnqn": "nqn.drive-a",
|
||||
"mn": "Model A", "sn": "SN1", "fr": "1.0",
|
||||
"vid": "0x1", "ssvid": "0x1", "transport": "pcie",
|
||||
}
|
||||
key_a = normalize_identity(identity_a)
|
||||
degraded_a = compute_identity_degraded(identity_a)
|
||||
|
||||
open_segment(store, clock.now, identity_a, key_a, degraded_a)
|
||||
current = find_current_segment(store)
|
||||
|
||||
store.execute(
|
||||
"INSERT INTO samples (ts, device, bytes_written) VALUES (?, ?, ?)",
|
||||
("2026-09-01T12:00:00Z", "/dev/nvme0", 2000)
|
||||
)
|
||||
store.commit()
|
||||
|
||||
identity_b = {
|
||||
"subnqn": "nqn.drive-b",
|
||||
"mn": "Model B", "sn": "SN2", "fr": "2.0",
|
||||
"vid": "0x2", "ssvid": "0x2", "transport": "pcie",
|
||||
}
|
||||
key_b = normalize_identity(identity_b)
|
||||
|
||||
should_open, reason = should_open_new_segment(
|
||||
current, key_b, 3000, store
|
||||
)
|
||||
|
||||
assert should_open is True
|
||||
assert reason == "identity_change"
|
||||
|
||||
def test_duw_decrease_same_identity(self, store, clock):
|
||||
identity = {
|
||||
"subnqn": "nqn.drive-a",
|
||||
"mn": "Model A", "sn": "SN1", "fr": "1.0",
|
||||
"vid": "0x1", "ssvid": "0x1", "transport": "pcie",
|
||||
}
|
||||
key = normalize_identity(identity)
|
||||
degraded = compute_identity_degraded(identity)
|
||||
|
||||
open_segment(store, clock.now, identity, key, degraded)
|
||||
current = find_current_segment(store)
|
||||
|
||||
store.execute(
|
||||
"INSERT INTO samples (ts, device, bytes_written) VALUES (?, ?, ?)",
|
||||
("2026-09-01T12:00:00Z", "/dev/nvme0", 5000)
|
||||
)
|
||||
store.commit()
|
||||
|
||||
should_open, reason = should_open_new_segment(
|
||||
current, key, 3000, store
|
||||
)
|
||||
|
||||
assert should_open is True
|
||||
assert reason == "duw_decrease"
|
||||
|
||||
|
||||
# Segment lifecycle
|
||||
|
||||
class TestSegmentLifecycle:
|
||||
def test_first_segment_always_opens(self, store):
|
||||
key = "nqn.test"
|
||||
|
||||
should_open, reason = should_open_new_segment(
|
||||
None, key, 1000, store
|
||||
)
|
||||
|
||||
assert should_open is True
|
||||
assert reason == "first_segment"
|
||||
|
||||
def test_same_identity_duw_non_decreasing_continues(self, store, clock):
|
||||
identity = {
|
||||
"subnqn": "nqn.drive",
|
||||
"mn": "Model", "sn": "SN1", "fr": "1.0",
|
||||
"vid": "0x1", "ssvid": "0x1", "transport": "pcie",
|
||||
}
|
||||
key = normalize_identity(identity)
|
||||
degraded = compute_identity_degraded(identity)
|
||||
|
||||
open_segment(store, clock.now, identity, key, degraded)
|
||||
current = find_current_segment(store)
|
||||
|
||||
store.execute(
|
||||
"INSERT INTO samples (ts, device, bytes_written) VALUES (?, ?, ?)",
|
||||
("2026-09-01T12:00:00Z", "/dev/nvme0", 1000)
|
||||
)
|
||||
store.commit()
|
||||
|
||||
should_open, reason = should_open_new_segment(
|
||||
current, key, 1500, store
|
||||
)
|
||||
|
||||
assert should_open is False
|
||||
assert reason is None
|
||||
|
||||
def test_multiple_segments(self, store, clock):
|
||||
identity_a = {
|
||||
"subnqn": "nqn.drive-a",
|
||||
"mn": "Model A", "sn": "SN1", "fr": "1.0",
|
||||
"vid": "0x1", "ssvid": "0x1", "transport": "pcie",
|
||||
}
|
||||
key_a = normalize_identity(identity_a)
|
||||
degraded_a = compute_identity_degraded(identity_a)
|
||||
|
||||
open_segment(store, clock.now, identity_a, key_a, degraded_a)
|
||||
|
||||
identity_b = {
|
||||
"subnqn": "nqn.drive-b",
|
||||
"mn": "Model B", "sn": "SN2", "fr": "2.0",
|
||||
"vid": "0x2", "ssvid": "0x2", "transport": "pcie",
|
||||
}
|
||||
key_b = normalize_identity(identity_b)
|
||||
degraded_b = compute_identity_degraded(identity_b)
|
||||
|
||||
open_segment(store, clock.now, identity_b, key_b, degraded_b)
|
||||
|
||||
cursor = store.execute("SELECT COUNT(*) FROM controller_segments")
|
||||
count = cursor.fetchone()[0]
|
||||
assert count == 2
|
||||
|
||||
current = find_current_segment(store)
|
||||
assert current["identity_key"] == key_b
|
||||
@@ -1,480 +0,0 @@
|
||||
"""Signing and consumer-repo structural tests (issue #51).
|
||||
|
||||
Verifies that the signing infrastructure, consumer setup docs, and key
|
||||
publication are correctly wired — without requiring a real GPG key,
|
||||
network access, or Docker.
|
||||
|
||||
All assertions are structural: config keys exist, URLs match, docs
|
||||
are present, and the Makefile exposes the right targets. A throwaway
|
||||
test key exercise is included for rpm signature verification mechanics.
|
||||
"""
|
||||
import subprocess
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _read(path: str | Path) -> str:
|
||||
from tests.conftest import read
|
||||
return read(path)
|
||||
|
||||
|
||||
def _gpg_available() -> bool:
|
||||
try:
|
||||
r = subprocess.run(
|
||||
["gpg", "--version"], capture_output=True, timeout=5,
|
||||
)
|
||||
return r.returncode == 0
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
return False
|
||||
|
||||
|
||||
def _rpmsign_available() -> bool:
|
||||
try:
|
||||
r = subprocess.run(
|
||||
["rpmsign", "--version"], capture_output=True, timeout=5,
|
||||
)
|
||||
return r.returncode == 0
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
return False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — nfpm.yaml signing configuration
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestNfpmSigningConfig:
|
||||
"""Verify that nfpm.yaml is correctly configured for RPM builds.
|
||||
|
||||
Note: RPM signing is done via rpmsign post-build (make sign-rpm),
|
||||
not through nfpm's built-in signing. This keeps the build unsigned
|
||||
and the signing step explicit and key-controlled.
|
||||
"""
|
||||
|
||||
def test_rpm_overrides_exist(self):
|
||||
"""nfpm.yaml must have overrides.rpm for per-format deltas."""
|
||||
content = _read("packaging/nfpm.yaml")
|
||||
assert "overrides:" in content, "overrides section missing from nfpm.yaml"
|
||||
assert "rpm:" in content, "rpm overrides missing from nfpm.yaml"
|
||||
|
||||
def test_rpm_scripts_configured(self):
|
||||
"""RPM must use the dedicated scriptlets, not deb scripts."""
|
||||
content = _read("packaging/nfpm.yaml")
|
||||
assert "packaging/rpm/post.sh" in content, \
|
||||
"RPM postinstall must use rpm/post.sh"
|
||||
assert "packaging/rpm/preun.sh" in content, \
|
||||
"RPM preremove must use rpm/preun.sh"
|
||||
assert "packaging/rpm/postun.sh" in content, \
|
||||
"RPM postremove must use rpm/postun.sh"
|
||||
|
||||
def test_rpm_depends_use_correct_syntax(self):
|
||||
"""RPM dependencies must use rpm-style version syntax."""
|
||||
content = _read("packaging/nfpm.yaml")
|
||||
assert "python3 >= 3.10" in content, \
|
||||
"RPM depends must use rpm-style version constraint"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — Makefile signing targets
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestMakefileSigningTargets:
|
||||
"""Verify that the Makefile exposes signing-related targets."""
|
||||
|
||||
def _makefile_content(self) -> str:
|
||||
return _read("Makefile")
|
||||
|
||||
def test_generate_test_key_target(self):
|
||||
content = self._makefile_content()
|
||||
assert "generate-test-key:" in content, \
|
||||
"Makefile must have a generate-test-key target"
|
||||
assert "packaging-test@bongbetic.com" in content or \
|
||||
"packaging@bongbetic.com" in content, \
|
||||
"generate-test-key must reference the packaging key UID"
|
||||
|
||||
def test_sign_rpm_target(self):
|
||||
content = self._makefile_content()
|
||||
assert "sign-rpm:" in content, \
|
||||
"Makefile must have a sign-rpm target"
|
||||
assert "rpmsign" in content, \
|
||||
"sign-rpm target must use rpmsign"
|
||||
|
||||
def test_checksums_target(self):
|
||||
content = self._makefile_content()
|
||||
assert "checksums:" in content, \
|
||||
"Makefile must have a checksums target"
|
||||
assert "sha256sum" in content, \
|
||||
"checksums target must use sha256sum"
|
||||
|
||||
def test_clearsign_target(self):
|
||||
content = self._makefile_content()
|
||||
assert "clearsign:" in content, \
|
||||
"Makefile must have a clearsign target"
|
||||
assert "--clearsign" in content, \
|
||||
"clearsign target must use gpg --clearsign"
|
||||
|
||||
def test_release_depends_on_signing(self):
|
||||
content = self._makefile_content()
|
||||
for line in content.splitlines():
|
||||
if line.startswith("release:"):
|
||||
deps = line.split(":", 1)[1].strip()
|
||||
assert "sign-rpm" in deps, \
|
||||
"release target must depend on sign-rpm"
|
||||
assert "clearsign" in deps, \
|
||||
"release target must depend on clearsign"
|
||||
break
|
||||
else:
|
||||
pytest.fail("release target not found in Makefile")
|
||||
|
||||
def test_packaging_key_uid_defined(self):
|
||||
content = self._makefile_content()
|
||||
assert "PACKAGING_KEY" in content, \
|
||||
"Makefile must define PACKAGING_KEY variable"
|
||||
|
||||
def test_release_mentions_key_ceremony(self):
|
||||
content = self._makefile_content()
|
||||
assert "signing-key-ceremony.md" in content, \
|
||||
"release target must reference the key ceremony doc"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — fenris.repo configuration
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestFenrisRepo:
|
||||
"""Verify the dnf repo file is correctly configured for Fenris."""
|
||||
|
||||
def _repo_content(self) -> str:
|
||||
return _read("packaging/fenris.repo")
|
||||
|
||||
def test_gpgcheck_enabled(self):
|
||||
content = self._repo_content()
|
||||
assert "gpgcheck=1" in content, \
|
||||
"fenris.repo must set gpgcheck=1 for payload verification"
|
||||
|
||||
def test_repo_gpgcheck_disabled(self):
|
||||
content = self._repo_content()
|
||||
assert "repo_gpgcheck=0" in content, \
|
||||
"fenris.repo must set repo_gpgcheck=0 (metadata check via TLS)"
|
||||
|
||||
def test_gpgkey_points_to_packaging_key(self):
|
||||
content = self._repo_content()
|
||||
assert "gpgkey=" in content, \
|
||||
"fenris.repo must have a gpgkey directive"
|
||||
assert "fenris-packaging.asc" in content, \
|
||||
"gpgkey must point at the packaging key"
|
||||
assert "raw/branch/main" in content, \
|
||||
"gpgkey must use raw URL for the public key"
|
||||
|
||||
def test_baseurl_is_fenris_rpm_group(self):
|
||||
content = self._repo_content()
|
||||
assert "rpm/fenris" in content, \
|
||||
"baseurl must point at the fenris RPM group"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — public key publication
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestKeyPublication:
|
||||
"""Verify the public key is published in-repo with correct metadata."""
|
||||
|
||||
def test_key_file_exists(self):
|
||||
key_path = REPO_ROOT / "packaging" / "keys" / "fenris-packaging.asc"
|
||||
assert key_path.exists(), \
|
||||
"packaging/keys/fenris-packaging.asc must exist"
|
||||
|
||||
def test_key_file_has_raw_url(self):
|
||||
content = _read("packaging/keys/fenris-packaging.asc")
|
||||
raw_url = (
|
||||
"https://git.bongbetic.com/xavierk/Fenris/raw/branch/main/"
|
||||
"packaging/keys/fenris-packaging.asc"
|
||||
)
|
||||
assert raw_url in content, \
|
||||
"Key file must contain its own raw URL as documentation"
|
||||
|
||||
def test_key_file_documents_algorithm(self):
|
||||
content = _read("packaging/keys/fenris-packaging.asc")
|
||||
assert "RSA 3072" in content or "rsa3072" in content.lower(), \
|
||||
"Key file must document the algorithm as RSA 3072"
|
||||
|
||||
def test_key_file_documents_uid(self):
|
||||
content = _read("packaging/keys/fenris-packaging.asc")
|
||||
assert "Fenris Packaging" in content, \
|
||||
"Key file must document the UID"
|
||||
|
||||
def test_key_file_documents_expiry(self):
|
||||
content = _read("packaging/keys/fenris-packaging.asc")
|
||||
assert "2 year" in content or "2-year" in content or "expiry" in content.lower(), \
|
||||
"Key file must document the expiry policy"
|
||||
|
||||
def test_key_file_references_ceremony_doc(self):
|
||||
content = _read("packaging/keys/fenris-packaging.asc")
|
||||
assert "signing-key-ceremony.md" in content, \
|
||||
"Key file must reference the key ceremony document"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — key ceremony documentation
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestKeyCeremonyDoc:
|
||||
"""Verify the key ceremony document is complete and accurate."""
|
||||
|
||||
def _doc_content(self) -> str:
|
||||
return _read("docs/install/signing-key-ceremony.md")
|
||||
|
||||
def test_ceremony_doc_exists(self):
|
||||
assert (REPO_ROOT / "docs" / "install" / "signing-key-ceremony.md").exists(), \
|
||||
"docs/install/signing-key-ceremony.md must exist"
|
||||
|
||||
def test_documents_key_specification(self):
|
||||
content = self._doc_content()
|
||||
assert "RSA 3072" in content, "Must document RSA 3072 algorithm"
|
||||
assert "Fenris Packaging" in content, "Must document the UID"
|
||||
assert "packaging@bongbetic.com" in content, "Must document the email"
|
||||
|
||||
def test_documents_import_sign_delete(self):
|
||||
content = self._doc_content()
|
||||
assert "import" in content.lower(), "Must document import step"
|
||||
assert "sign" in content.lower(), "Must document sign step"
|
||||
assert "delete" in content.lower(), "Must document delete step"
|
||||
|
||||
def test_documents_rotation_outline(self):
|
||||
content = self._doc_content()
|
||||
assert "rotation" in content.lower(), \
|
||||
"Must document key rotation procedure"
|
||||
|
||||
def test_documents_dual_key_approach(self):
|
||||
content = self._doc_content()
|
||||
assert "previous" in content.lower() or "old" in content.lower(), \
|
||||
"Must document old key retention during rotation"
|
||||
|
||||
def test_documents_private_key_storage(self):
|
||||
content = self._doc_content()
|
||||
assert "password manager" in content.lower(), \
|
||||
"Must document that private key lives in password manager"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — consumer setup documentation
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestConsumerDocs:
|
||||
"""Verify consumer setup docs are present and correctly wired."""
|
||||
|
||||
def _readme_content(self) -> str:
|
||||
return _read("README.md")
|
||||
|
||||
def test_apt_signed_by_flow(self):
|
||||
content = self._readme_content()
|
||||
assert "signed-by" in content, \
|
||||
"README must document apt signed-by keyring flow"
|
||||
assert "keyrings" in content, \
|
||||
"README must show the keyrings directory"
|
||||
|
||||
def test_apt_fingerprint_placeholder(self):
|
||||
content = self._readme_content()
|
||||
assert "Fingerprint" in content or "fingerprint" in content, \
|
||||
"README must include fingerprint placeholder for TOFU hardening"
|
||||
|
||||
def test_dnf_repo_flow(self):
|
||||
content = self._readme_content()
|
||||
assert "dnf config-manager --add-repo" in content or \
|
||||
"dnf install" in content, \
|
||||
"README must document dnf install flow"
|
||||
assert "fenris.repo" in content, \
|
||||
"README must reference the Fenris-owned repo file"
|
||||
|
||||
def test_no_gitea_auto_repo(self):
|
||||
"""Gitea's auto-generated .repo must never be referenced in docs."""
|
||||
content = self._readme_content()
|
||||
# The Gitea auto-generated repo would have gpgcheck=1 against the
|
||||
# instance key, which is a trap. Our docs should only reference
|
||||
# our own fenris.repo file.
|
||||
assert "auto-generated" not in content.lower() or \
|
||||
"never" in content.lower(), \
|
||||
"README must not recommend Gitea's auto-generated .repo"
|
||||
|
||||
def test_package_signature_verification(self):
|
||||
content = self._readme_content()
|
||||
assert "rpm -K" in content or "rpm --checksig" in content, \
|
||||
"README must document RPM signature verification"
|
||||
assert "gpg --verify" in content, \
|
||||
"README must document GPG verification for SHA256SUMS"
|
||||
|
||||
def test_migration_from_make_install(self):
|
||||
content = self._readme_content()
|
||||
assert "migrate-from-makeinstall" in content.lower() or \
|
||||
"migration" in content.lower(), \
|
||||
"README must reference the migration runbook"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — release spec references
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestReleaseSpecReferences:
|
||||
"""Verify the release spec references the ceremony doc and nfpm config."""
|
||||
|
||||
def _spec_content(self) -> str:
|
||||
return _read("docs/spec/release-packaging.md")
|
||||
|
||||
def test_spec_references_rpmsign(self):
|
||||
content = self._spec_content()
|
||||
assert "rpmsign" in content.lower() or "sign-rpm" in content, \
|
||||
"Spec must reference rpmsign or make sign-rpm for RPM signing"
|
||||
|
||||
def test_spec_references_ceremony_doc(self):
|
||||
content = self._spec_content()
|
||||
assert "signing-key-ceremony.md" in content, \
|
||||
"Spec must reference the key ceremony document"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — RPM signature mechanics (throwaway test key, no network)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestRpmSignatureMechanics:
|
||||
"""Verify RPM signing mechanics using a throwaway test key.
|
||||
|
||||
These tests generate a temporary GPG key, build an RPM (or use an
|
||||
existing one), sign it, and verify the signature — all without
|
||||
network access. They require gpg and rpmsign to be available.
|
||||
"""
|
||||
|
||||
@pytest.mark.skipif(
|
||||
not _gpg_available() or not _rpmsign_available(),
|
||||
reason="gpg or rpmsign not available",
|
||||
)
|
||||
def test_throwaway_key_signs_and_verifies(self):
|
||||
"""Generate a throwaway key, sign a test RPM, verify signature."""
|
||||
# Find existing RPM
|
||||
version = None
|
||||
for line in (REPO_ROOT / "pyproject.toml").read_text().splitlines():
|
||||
if line.startswith("version"):
|
||||
version = line.split("=")[1].strip().strip('"')
|
||||
break
|
||||
rpm_path = REPO_ROOT / "dist" / f"fenris-{version}-1.x86_64.rpm"
|
||||
if not rpm_path.exists():
|
||||
pytest.skip("RPM not built — run `make package-rpm` first")
|
||||
|
||||
key_uid = "fenris-test-signing@example.com"
|
||||
try:
|
||||
# Generate throwaway key
|
||||
subprocess.run(
|
||||
["gpg", "--batch", "--gen-key"],
|
||||
input=f"""%no-protection
|
||||
Key-Type: RSA
|
||||
Key-Length: 3072
|
||||
Name-Real: {key_uid}
|
||||
Name-Email: {key_uid}
|
||||
Expire-Date: 0
|
||||
%commit
|
||||
""",
|
||||
text=True, check=True, timeout=30,
|
||||
)
|
||||
|
||||
# Copy RPM to temp dir for signing
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
signed_rpm = Path(tmpdir) / rpm_path.name
|
||||
signed_rpm.write_bytes(rpm_path.read_bytes())
|
||||
|
||||
# Sign the RPM
|
||||
subprocess.run(
|
||||
["rpmsign", "--addsign",
|
||||
"--define", f"_gpg_name {key_uid}",
|
||||
str(signed_rpm)],
|
||||
check=True, timeout=30,
|
||||
)
|
||||
|
||||
# Verify the signature exists and has correct format
|
||||
# (rpm -Kv returns NOKEY if key isn't imported, but the
|
||||
# signature header is still present and verifiable)
|
||||
r = subprocess.run(
|
||||
["rpm", "-Kv", str(signed_rpm)],
|
||||
capture_output=True, text=True, timeout=10,
|
||||
)
|
||||
output = r.stdout + r.stderr
|
||||
assert "RSA" in output or "rsa" in output.lower(), \
|
||||
f"RPM must have RSA signature: {output}"
|
||||
assert "SHA256" in output or "sha256" in output.lower(), \
|
||||
f"RPM must have SHA256 digest: {output}"
|
||||
assert "Header V4" in output or "Header" in output, \
|
||||
f"RPM must have V4 signature header: {output}"
|
||||
assert "Signature" in output, \
|
||||
f"RPM must show signature info: {output}"
|
||||
|
||||
finally:
|
||||
# Clean up the test key
|
||||
subprocess.run(
|
||||
["gpg", "--batch", "--yes", "--delete-secret-keys", key_uid],
|
||||
capture_output=True, timeout=5,
|
||||
)
|
||||
subprocess.run(
|
||||
["gpg", "--batch", "--yes", "--delete-keys", key_uid],
|
||||
capture_output=True, timeout=5,
|
||||
)
|
||||
|
||||
@pytest.mark.skipif(
|
||||
not _gpg_available(),
|
||||
reason="gpg not available",
|
||||
)
|
||||
def test_clearsign_and_verify(self):
|
||||
"""Clearsign a test manifest and verify the signature."""
|
||||
key_uid = "fenris-test-clearsign@example.com"
|
||||
try:
|
||||
# Generate throwaway key
|
||||
subprocess.run(
|
||||
["gpg", "--batch", "--gen-key"],
|
||||
input=f"""%no-protection
|
||||
Key-Type: RSA
|
||||
Key-Length: 3072
|
||||
Name-Real: {key_uid}
|
||||
Name-Email: {key_uid}
|
||||
Expire-Date: 0
|
||||
%commit
|
||||
""",
|
||||
text=True, check=True, timeout=30,
|
||||
)
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
sums = Path(tmpdir) / "SHA256SUMS"
|
||||
sums.write_text(
|
||||
"abc123 fenris_0.3.0_amd64.deb\n"
|
||||
"def456 fenris-0.3.0-1.x86_64.rpm\n"
|
||||
)
|
||||
|
||||
# Clearsign
|
||||
subprocess.run(
|
||||
["gpg", "--batch", "--yes", "--clearsign",
|
||||
"--local-user", key_uid, str(sums)],
|
||||
check=True, timeout=10,
|
||||
)
|
||||
|
||||
# Verify (clearsigned file — just one argument to --verify)
|
||||
r = subprocess.run(
|
||||
["gpg", "--verify", str(sums.with_suffix(".asc"))],
|
||||
capture_output=True, text=True, timeout=10,
|
||||
)
|
||||
assert r.returncode == 0, \
|
||||
f"Clearsign verification failed: {r.stderr}"
|
||||
assert "Good signature" in r.stderr, \
|
||||
f"Expected Good signature: {r.stderr}"
|
||||
|
||||
finally:
|
||||
subprocess.run(
|
||||
["gpg", "--batch", "--yes", "--delete-secret-keys", key_uid],
|
||||
capture_output=True, timeout=5,
|
||||
)
|
||||
subprocess.run(
|
||||
["gpg", "--batch", "--yes", "--delete-keys", key_uid],
|
||||
capture_output=True, timeout=5,
|
||||
)
|
||||
@@ -1,687 +0,0 @@
|
||||
"""Tests for the read-only CLI status command (issue #27).
|
||||
|
||||
Covers:
|
||||
- LC-9: status is a pure read-only composition
|
||||
- CI-2: TUI/CLI parity (status fact set matches TUI's four separate facts)
|
||||
- CI-4: Required wording and six disclosures render as adopted
|
||||
- FL-4: Store fault renders exact fixed phrase
|
||||
- FL-5: Newer-schema store renders exact fixed phrase
|
||||
- FL-7: Drive anomalies render as ordinary facts, never affecting projection
|
||||
- LC-10: Freshness grading with shared constants
|
||||
- Retired command rejection with migration pointers
|
||||
- Configuration error surfaced from direct reads
|
||||
"""
|
||||
import os
|
||||
import sqlite3
|
||||
import sys
|
||||
import tempfile
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.status import (
|
||||
grade_freshness,
|
||||
freshness_age_human,
|
||||
format_disclosures,
|
||||
check_retired_command,
|
||||
check_retired_flag,
|
||||
FRESH_THRESHOLD_S,
|
||||
STALENESS_THRESHOLD_S,
|
||||
CADENCE_DEFAULT_S,
|
||||
ACCURACY_SEC,
|
||||
)
|
||||
from fenris.store import init_store, SCHEMA_VERSION
|
||||
from fenris.projection import DISCLOSURES, ConfidenceState
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Freshness grading (§8.9, LC-10)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestFreshnessGrading:
|
||||
"""Freshness constants are defined once and shared (§8.9)."""
|
||||
|
||||
def test_constants_match_spec(self):
|
||||
"""Fresh threshold = 2 × cadence + AccuracySec + 60 s."""
|
||||
expected = 2 * CADENCE_DEFAULT_S + ACCURACY_SEC + 60
|
||||
assert FRESH_THRESHOLD_S == expected
|
||||
assert STALENESS_THRESHOLD_S == 48 * 3600
|
||||
|
||||
def test_empty_store(self):
|
||||
"""Empty store reads 'no observations yet' (§8.9)."""
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
assert grade_freshness(None, now) == "empty"
|
||||
|
||||
def test_fresh_sample(self):
|
||||
"""Newest sample within threshold → fresh."""
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
ts = (now - timedelta(seconds=FRESH_THRESHOLD_S - 1)).isoformat()
|
||||
assert grade_freshness(ts, now) == "fresh"
|
||||
|
||||
def test_missed_sample(self):
|
||||
"""Between fresh and 48h → missed."""
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
ts = (now - timedelta(hours=2)).isoformat()
|
||||
assert grade_freshness(ts, now) == "missed"
|
||||
|
||||
def test_stale_sample(self):
|
||||
"""≥ 48h → stale."""
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
ts = (now - timedelta(hours=49)).isoformat()
|
||||
assert grade_freshness(ts, now) == "stale"
|
||||
|
||||
def test_fresh_at_boundary(self):
|
||||
"""Exactly at threshold → fresh (within means ≤)."""
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
ts = (now - timedelta(seconds=FRESH_THRESHOLD_S)).isoformat()
|
||||
assert grade_freshness(ts, now) == "fresh"
|
||||
|
||||
def test_missed_at_just_past_fresh(self):
|
||||
"""One second past threshold → missed."""
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
ts = (now - timedelta(seconds=FRESH_THRESHOLD_S + 1)).isoformat()
|
||||
assert grade_freshness(ts, now) == "missed"
|
||||
|
||||
def test_stale_at_boundary(self):
|
||||
"""Exactly 48h → stale."""
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
ts = (now - timedelta(hours=48)).isoformat()
|
||||
assert grade_freshness(ts, now) == "stale"
|
||||
|
||||
def test_naive_timestamp_treated_as_utc(self):
|
||||
"""Naive timestamp is treated as UTC — within fresh threshold."""
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
# Construct a truly naive ISO string (no +00:00 suffix), 5 min ago
|
||||
naive_dt = datetime(2026, 9, 1, 11, 55, 0) # 5 min ago, naive
|
||||
ts = naive_dt.isoformat() # "2026-09-01T11:55:00"
|
||||
assert grade_freshness(ts, now) == "fresh"
|
||||
|
||||
def test_malformed_timestamp_returns_empty(self):
|
||||
"""Malformed timestamp → empty."""
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
assert grade_freshness("not-a-timestamp", now) == "empty"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Freshness age human-readable
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestFreshnessAgeHuman:
|
||||
def test_seconds(self):
|
||||
assert freshness_age_human(30) == "30s ago"
|
||||
|
||||
def test_minutes(self):
|
||||
assert freshness_age_human(120) == "2m ago"
|
||||
|
||||
def test_hours_and_minutes(self):
|
||||
assert freshness_age_human(3661) == "1h 1m ago"
|
||||
|
||||
def test_days(self):
|
||||
assert freshness_age_human(90000) == "1d ago"
|
||||
|
||||
def test_none(self):
|
||||
assert freshness_age_human(None) == "unknown age"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Retired command rejection (§8.8)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestRetiredCommands:
|
||||
"""Retired commands and --device are rejected with one-line pointers."""
|
||||
|
||||
def test_start_rejected(self):
|
||||
ptr = check_retired_command("start")
|
||||
assert ptr is not None
|
||||
assert "resume" in ptr.lower() or "enable" in ptr.lower()
|
||||
|
||||
def test_stop_rejected(self):
|
||||
ptr = check_retired_command("stop")
|
||||
assert ptr is not None
|
||||
assert "pause" in ptr.lower() or "disable" in ptr.lower()
|
||||
|
||||
def test_run_rejected(self):
|
||||
ptr = check_retired_command("run")
|
||||
assert ptr is not None
|
||||
|
||||
def test_status_not_rejected(self):
|
||||
assert check_retired_command("status") is None
|
||||
|
||||
def test_sample_not_rejected(self):
|
||||
assert check_retired_command("sample") is None
|
||||
|
||||
def test_device_flag_rejected(self):
|
||||
ptr = check_retired_flag("--device")
|
||||
assert ptr is not None
|
||||
assert "fenris.conf" in ptr
|
||||
|
||||
def test_unknown_flag_not_rejected(self):
|
||||
assert check_retired_flag("--unknown") is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Disclosures (§6.11, CI-4)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestDisclosures:
|
||||
"""Required wording and six disclosures render as adopted (CI-4)."""
|
||||
|
||||
def test_six_disclosures(self):
|
||||
assert len(DISCLOSURES) == 6
|
||||
|
||||
def test_disclosures_text(self):
|
||||
"""Each disclosure matches the spec verbatim."""
|
||||
assert "endurance projection" in DISCLOSURES[0].lower()
|
||||
assert "hardware-failure" in DISCLOSURES[0].lower() or "failure date" in DISCLOSURES[0].lower()
|
||||
assert "vendor-specific" in DISCLOSURES[1]
|
||||
assert "255 is saturated" in DISCLOSURES[1]
|
||||
assert "warranty" in DISCLOSURES[2] or "endurance threshold" in DISCLOSURES[2]
|
||||
assert "DUW" in DISCLOSURES[3]
|
||||
assert "metadata" in DISCLOSURES[3]
|
||||
assert "future workload" in DISCLOSURES[4]
|
||||
assert "deliberately disabled" in DISCLOSURES[5]
|
||||
|
||||
def test_format_disclosures_returns_all_six(self):
|
||||
output = format_disclosures()
|
||||
for i in range(1, 7):
|
||||
assert "%d." % i in output
|
||||
|
||||
def test_disclosures_header(self):
|
||||
output = format_disclosures()
|
||||
assert output.startswith("Disclosures")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Store fault rendering (§9.4, FL-4)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestStoreFault:
|
||||
"""Store fault surfaces exact fixed phrase (FL-4)."""
|
||||
|
||||
def test_store_fault_phrase(self, tmp_path):
|
||||
"""observation store unreadable with journal hint."""
|
||||
from fenris.status import open_store_readonly, StoreFault
|
||||
|
||||
nonexistent = tmp_path / "nonexistent.db"
|
||||
with pytest.raises(StoreFault):
|
||||
open_store_readonly(nonexistent)
|
||||
|
||||
def test_corrupt_store(self, tmp_path):
|
||||
"""Corrupt file raises StoreFault."""
|
||||
from fenris.status import open_store_readonly, StoreFault
|
||||
|
||||
corrupt = tmp_path / "corrupt.db"
|
||||
corrupt.write_bytes(b"this is not a sqlite database")
|
||||
with pytest.raises(StoreFault):
|
||||
open_store_readonly(corrupt)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Newer schema rendering (§9.5, FL-5)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestNewerSchema:
|
||||
"""Newer-schema store renders exact fixed phrase (FL-5)."""
|
||||
|
||||
def test_newer_schema_detected(self, tmp_path):
|
||||
from fenris.status import open_store_readonly, NewerSchema
|
||||
|
||||
db = tmp_path / "test.db"
|
||||
conn = sqlite3.connect(str(db))
|
||||
conn.execute("PRAGMA user_version=%d" % (SCHEMA_VERSION + 1))
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
with pytest.raises(NewerSchema):
|
||||
open_store_readonly(db)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Configuration error (§8.3, LC-4)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestConfigError:
|
||||
"""Configuration error surfaced as configuration error: <reason>."""
|
||||
|
||||
def test_missing_config(self):
|
||||
from fenris.status import read_config, ConfigError
|
||||
with patch("fenris.status.CONFIG_PATH", Path("/nonexistent/fenris.conf")):
|
||||
with pytest.raises(ConfigError, match="not found"):
|
||||
read_config()
|
||||
|
||||
def test_empty_config(self, tmp_path):
|
||||
from fenris.status import read_config, ConfigError
|
||||
conf = tmp_path / "fenris.conf"
|
||||
conf.write_text("# empty config\n")
|
||||
with patch("fenris.status.CONFIG_PATH", conf):
|
||||
with pytest.raises(ConfigError, match="no device selector"):
|
||||
read_config()
|
||||
|
||||
def test_valid_config(self, tmp_path):
|
||||
from fenris.status import read_config
|
||||
conf = tmp_path / "fenris.conf"
|
||||
conf.write_text("device = /dev/disk/by-id/nvme-test\n")
|
||||
with patch("fenris.status.CONFIG_PATH", conf):
|
||||
result = read_config()
|
||||
assert result["device"] == "/dev/disk/by-id/nvme-test"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Drive anomalies (§9.7, FL-7)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestDriveAnomalies:
|
||||
"""Drive anomalies render as ordinary facts, never affecting projection."""
|
||||
|
||||
def test_no_anomalies(self, tmp_path):
|
||||
from fenris.status import _query_drive_facts
|
||||
|
||||
db = tmp_path / "test.db"
|
||||
conn = init_store(db)
|
||||
conn.execute(
|
||||
"INSERT INTO samples (ts, device, mn, sn, fr, capacity_bytes, "
|
||||
"percentage_used, available_spare, media_errors, power_on_hours, "
|
||||
"power_cycles, unsafe_shutdowns, temperature_c, "
|
||||
"data_units_written, data_units_read, bytes_written, bytes_read, "
|
||||
"critical_warning) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
("2026-09-01T12:00:00Z", "/dev/nvme0", "Test", "SN", "FR",
|
||||
1000000000000, 5, 100, 0, 1000, 100, 0, 35,
|
||||
1000000, 500000, 512000000000, 256000000000, 0),
|
||||
)
|
||||
conn.commit()
|
||||
facts = _query_drive_facts(conn)
|
||||
assert facts == []
|
||||
conn.close()
|
||||
|
||||
def test_critical_warning(self, tmp_path):
|
||||
from fenris.status import _query_drive_facts
|
||||
|
||||
db = tmp_path / "test.db"
|
||||
conn = init_store(db)
|
||||
conn.execute(
|
||||
"INSERT INTO samples (ts, device, mn, sn, fr, capacity_bytes, "
|
||||
"percentage_used, available_spare, media_errors, power_on_hours, "
|
||||
"power_cycles, unsafe_shutdowns, temperature_c, "
|
||||
"data_units_written, data_units_read, bytes_written, bytes_read, "
|
||||
"critical_warning) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
("2026-09-01T12:00:00Z", "/dev/nvme0", "Test", "SN", "FR",
|
||||
1000000000000, 5, 100, 0, 1000, 100, 0, 35,
|
||||
1000000, 500000, 512000000000, 256000000000, 1),
|
||||
)
|
||||
conn.commit()
|
||||
facts = _query_drive_facts(conn)
|
||||
assert any("critical warning" in f for f in facts)
|
||||
conn.close()
|
||||
|
||||
def test_media_errors(self, tmp_path):
|
||||
from fenris.status import _query_drive_facts
|
||||
|
||||
db = tmp_path / "test.db"
|
||||
conn = init_store(db)
|
||||
conn.execute(
|
||||
"INSERT INTO samples (ts, device, mn, sn, fr, capacity_bytes, "
|
||||
"percentage_used, available_spare, media_errors, power_on_hours, "
|
||||
"power_cycles, unsafe_shutdowns, temperature_c, "
|
||||
"data_units_written, data_units_read, bytes_written, bytes_read, "
|
||||
"critical_warning) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
("2026-09-01T12:00:00Z", "/dev/nvme0", "Test", "SN", "FR",
|
||||
1000000000000, 5, 100, 3, 1000, 100, 0, 35,
|
||||
1000000, 500000, 512000000000, 256000000000, 0),
|
||||
)
|
||||
conn.commit()
|
||||
facts = _query_drive_facts(conn)
|
||||
assert any("media errors" in f for f in facts)
|
||||
conn.close()
|
||||
|
||||
def test_unsafe_shutdowns(self, tmp_path):
|
||||
from fenris.status import _query_drive_facts
|
||||
|
||||
db = tmp_path / "test.db"
|
||||
conn = init_store(db)
|
||||
conn.execute(
|
||||
"INSERT INTO samples (ts, device, mn, sn, fr, capacity_bytes, "
|
||||
"percentage_used, available_spare, media_errors, power_on_hours, "
|
||||
"power_cycles, unsafe_shutdowns, temperature_c, "
|
||||
"data_units_written, data_units_read, bytes_written, bytes_read, "
|
||||
"critical_warning) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
("2026-09-01T12:00:00Z", "/dev/nvme0", "Test", "SN", "FR",
|
||||
1000000000000, 5, 100, 0, 1000, 100, 5, 35,
|
||||
1000000, 500000, 512000000000, 256000000000, 0),
|
||||
)
|
||||
conn.commit()
|
||||
facts = _query_drive_facts(conn)
|
||||
assert any("unsafe shutdowns" in f for f in facts)
|
||||
conn.close()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Empty store greeting (§8.9, LC-10)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestEmptyStoreGreeting:
|
||||
"""Empty store reads 'no observations yet' with enable hint."""
|
||||
|
||||
def test_empty_store_message(self, tmp_path):
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
init_store(db)
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
assert "no observations yet" in result
|
||||
assert "enable" in result.lower() or "resume" in result.lower()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Four separate service facts (§7.3, LC-9, CI-2)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestServiceFacts:
|
||||
"""Status renders four separate service facts matching TUI (CI-2)."""
|
||||
|
||||
def test_service_facts_present(self, tmp_path):
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
init_store(db)
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": True, "timer_active": True,
|
||||
"last_collect_ok": True, "last_collect_age_s": 120,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
assert "boot:" in result
|
||||
assert "timer:" in result
|
||||
assert "last collect:" in result
|
||||
assert "freshness:" in result
|
||||
|
||||
def test_continuity_reports_boot_enabled_independently_of_runtime(self, tmp_path):
|
||||
"""Status names reboot continuity while retaining the timer fact."""
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
init_store(db)
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": True, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
assert "monitoring: active in background · persists across reboots" in result
|
||||
assert "timer: inactive" in result
|
||||
|
||||
def test_continuity_and_deliberate_pause_are_reported_separately(self, tmp_path):
|
||||
"""Only a sanctioned user_disabled period renders the paused wording."""
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
conn = init_store(db)
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at, ended_at, end_cause) "
|
||||
"VALUES (?, ?, ?)",
|
||||
("2026-09-01T09:00:00+00:00", "2026-09-01T10:00:00+00:00", "user_disabled"),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
assert "monitoring: does not start on next boot" in result
|
||||
assert "monitoring: paused — deliberate disable" in result
|
||||
assert "paused time is excluded from your usage habit · resume: fenris monitor resume" in result
|
||||
|
||||
def test_raw_system_state_without_user_disabled_is_not_a_deliberate_pause(self, tmp_path):
|
||||
"""A non-sanctioned stop never acquires the deliberate-disable label."""
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
conn = init_store(db)
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at, ended_at, end_cause) "
|
||||
"VALUES (?, ?, ?)",
|
||||
("2026-09-01T09:00:00+00:00", "2026-09-01T10:00:00+00:00", "migrated"),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
assert "monitoring: paused — deliberate disable" not in result
|
||||
|
||||
def test_resumed_open_period_clears_a_previous_deliberate_pause(self, tmp_path):
|
||||
"""A later sanctioned resume takes precedence over an older pause."""
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
conn = init_store(db)
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at, ended_at, end_cause) "
|
||||
"VALUES (?, ?, ?)",
|
||||
("2026-09-01T09:00:00+00:00", "2026-09-01T10:00:00+00:00", "user_disabled"),
|
||||
)
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at) VALUES (?)",
|
||||
("2026-09-01T11:00:00+00:00",),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": True, "timer_active": True,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
assert "monitoring: paused — deliberate disable" not in result
|
||||
|
||||
def test_live_enabled_service_suppresses_a_stale_pause_marker(self, tmp_path):
|
||||
"""A raw re-enable cannot leave a contradictory paused presentation."""
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
conn = init_store(db)
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at, ended_at, end_cause) "
|
||||
"VALUES (?, ?, ?)",
|
||||
("2026-09-01T09:00:00+00:00", "2026-09-01T10:00:00+00:00", "user_disabled"),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": True, "timer_active": True,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
assert "monitoring: paused — deliberate disable" not in result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Status output structure (§8.8, LC-9)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestStatusOutput:
|
||||
"""Status is a pure read-only composition (LC-9)."""
|
||||
|
||||
def test_status_returns_string(self, tmp_path):
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
init_store(db)
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
assert isinstance(result, str)
|
||||
assert len(result) > 0
|
||||
|
||||
def test_status_excludes_tui_identity_and_auth_notice(self, tmp_path):
|
||||
"""CLI status never renders TUI-only identity or launch guidance."""
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
init_store(db)
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
for tui_only in (
|
||||
"Fenris — NVMe endurance monitor",
|
||||
"by Bongbetic",
|
||||
"privileged actions will prompt for authentication (polkit)",
|
||||
):
|
||||
assert tui_only not in result
|
||||
|
||||
def test_status_never_writes(self, tmp_path):
|
||||
"""Status never writes to the store."""
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
init_store(db)
|
||||
|
||||
mtime_before = db.stat().st_mtime
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
mtime_after = db.stat().st_mtime
|
||||
assert mtime_before == mtime_after
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Projection in status (§6.10)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestProjectionInStatus:
|
||||
"""Projection is recomputed on read, never stored (§6.10)."""
|
||||
|
||||
def test_store_fault_suppresses_projection(self, tmp_path):
|
||||
from fenris.status import get_status
|
||||
|
||||
nonexistent = tmp_path / "nonexistent.db"
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=nonexistent, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
assert "observation store unreadable" in result
|
||||
assert "%" not in result # no projection numbers
|
||||
|
||||
def test_newer_schema_suppresses_projection(self, tmp_path):
|
||||
from fenris.status import get_status
|
||||
|
||||
db = tmp_path / "test.db"
|
||||
conn = sqlite3.connect(str(db))
|
||||
conn.execute("PRAGMA user_version=%d" % (SCHEMA_VERSION + 1))
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = get_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False)
|
||||
|
||||
assert "newer Fenris" in result
|
||||
assert "upgrade Fenris" in result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# render_status with disclosures (CI-4)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestRenderStatusDisclosures:
|
||||
"""Disclosures are always available in status."""
|
||||
|
||||
def test_disclosures_in_output(self, tmp_path):
|
||||
from fenris.status import render_status
|
||||
|
||||
db = tmp_path / "observations.db"
|
||||
init_store(db)
|
||||
now = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
with patch("fenris.status.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
result = render_status(store_path=db, clock_now=now,
|
||||
query_services=True, query_journal=False,
|
||||
show_disclosures=True)
|
||||
|
||||
assert "Disclosures" in result
|
||||
assert "endurance projection" in result.lower()
|
||||
assert "1." in result
|
||||
assert "6." in result
|
||||
@@ -1,56 +0,0 @@
|
||||
"""Store path resolution from config — regression coverage for issue #53.
|
||||
|
||||
A fresh install ships a placeholder-commented config whose only required
|
||||
key is the device selector. The collector must not crash with
|
||||
KeyError 'store_path' when the key is absent.
|
||||
"""
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.store import DEFAULT_STORE_PATH, get_store_path
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
TEMPLATE = REPO_ROOT / "packaging" / "fenris.conf"
|
||||
|
||||
|
||||
def _parse_like_load_config(text: str) -> dict:
|
||||
"""Mirror collect.load_config()'s key=value parsing rules."""
|
||||
config = {}
|
||||
for line in text.splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
if "=" in line:
|
||||
key, value = line.split("=", 1)
|
||||
config[key.strip()] = value.strip()
|
||||
return config
|
||||
|
||||
|
||||
def test_missing_store_path_falls_back_to_default():
|
||||
"""Config with only the device selector resolves to the packaged default."""
|
||||
assert get_store_path({"device": "/dev/nvme0n1"}) == DEFAULT_STORE_PATH
|
||||
|
||||
|
||||
def test_explicit_store_path_wins():
|
||||
"""An explicit store_path override is honored."""
|
||||
assert get_store_path({"store_path": "/tmp/other.db"}) == Path("/tmp/other.db")
|
||||
|
||||
|
||||
def test_packaged_template_yields_collectable_config():
|
||||
"""The packaged template, once a device is set, must be collector-ready.
|
||||
|
||||
Reproduces the fresh-install path: parse packaging/fenris.conf the way
|
||||
collect.load_config() does, add the device selector, then resolve the
|
||||
store. Issue #53 made this raise KeyError.
|
||||
"""
|
||||
config = _parse_like_load_config(TEMPLATE.read_text())
|
||||
config["device"] = "/dev/nvme0n1"
|
||||
assert get_store_path(config) == DEFAULT_STORE_PATH
|
||||
|
||||
|
||||
def test_template_documents_store_path():
|
||||
"""The template must mention store_path so admins know it is overridable."""
|
||||
assert "store_path" in TEMPLATE.read_text()
|
||||
@@ -1,79 +0,0 @@
|
||||
"""Group access to the observation store — regression coverage for issue #54.
|
||||
|
||||
Two defects: (1) a non-group user's stat() on the store directory raised
|
||||
PermissionError straight through open_store_readonly(), crashing status/TUI
|
||||
instead of degrading to the Store fault view; (2) even group members could
|
||||
not open the WAL-mode store because root-created sidecars lacked group write
|
||||
and the store directory lacked group execute-then-write.
|
||||
"""
|
||||
import sqlite3
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.store import DEFAULT_STORE_PATH, init_store
|
||||
from fenris.status import StoreFault, open_store_readonly
|
||||
|
||||
|
||||
def test_stat_permission_error_becomes_store_fault(monkeypatch, tmp_path):
|
||||
"""stat() denied (non-group user on a 2750 dir) → StoreFault, not crash."""
|
||||
store = tmp_path / "observations.db"
|
||||
store.write_bytes(b"")
|
||||
|
||||
import pathlib
|
||||
|
||||
def denied(self, follow_symlinks=True):
|
||||
raise PermissionError(13, "Permission denied")
|
||||
|
||||
monkeypatch.setattr(pathlib.Path, "exists", denied)
|
||||
with pytest.raises(StoreFault):
|
||||
open_store_readonly(store)
|
||||
|
||||
|
||||
def test_connect_failure_becomes_store_fault(tmp_path):
|
||||
"""sqlite failures stay wrapped as StoreFault (existing contract)."""
|
||||
garbage = tmp_path / "observations.db"
|
||||
garbage.write_bytes(b"not a database" * 100)
|
||||
with pytest.raises(StoreFault):
|
||||
open_store_readonly(garbage)
|
||||
|
||||
|
||||
def test_init_store_leaves_files_group_writable(tmp_path):
|
||||
"""Root-created stores must stay readable by WAL readers: db and sidecars
|
||||
need group write after init_store (issue #54)."""
|
||||
store = tmp_path / "observations.db"
|
||||
conn = init_store(store)
|
||||
try:
|
||||
assert (store.stat().st_mode & 0o060) == 0o060, "db not group rw"
|
||||
wal = store.with_name(store.name + "-wal")
|
||||
shm = store.with_name(store.name + "-shm")
|
||||
if wal.exists():
|
||||
assert (wal.stat().st_mode & 0o060) == 0o060, "wal not group rw"
|
||||
if shm.exists():
|
||||
assert (shm.stat().st_mode & 0o060) == 0o060, "shm not group rw"
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def test_readonly_open_works_after_init_store(tmp_path):
|
||||
"""The shipped read path opens a store created by init_store."""
|
||||
store = tmp_path / "observations.db"
|
||||
writer = init_store(store)
|
||||
writer.execute("INSERT INTO monitoring_periods (started_at) VALUES ('2026-01-01T00:00:00+00:00')")
|
||||
writer.commit()
|
||||
conn = open_store_readonly(store)
|
||||
assert conn is not None
|
||||
conn.close()
|
||||
writer.close()
|
||||
|
||||
|
||||
def test_packaging_ships_group_access():
|
||||
"""tmpfiles must create the store dir group-writable; collect unit must
|
||||
keep the umask loose so root-created sidecars stay group-accessible."""
|
||||
repo = Path(__file__).resolve().parent.parent
|
||||
assert "2770" in (repo / "packaging" / "tmpfiles.d" / "fenris.conf").read_text()
|
||||
assert "2750" not in (repo / "packaging" / "tmpfiles.d" / "fenris.conf").read_text()
|
||||
assert "UMask=002" in (repo / "units" / "fenris-collect.service").read_text()
|
||||
@@ -1,195 +0,0 @@
|
||||
"""Observation store migration unit tests (issue #48).
|
||||
|
||||
Tests the forward-only migration logic that underpins package upgrade
|
||||
semantics: older stores are migrated, current stores pass through, and
|
||||
newer stores are refused loudly.
|
||||
|
||||
Spec: §3.6, §9.5, §10.2
|
||||
"""
|
||||
import sqlite3
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from fenris.store import (
|
||||
SCHEMA_VERSION,
|
||||
init_store,
|
||||
migrate_to_latest,
|
||||
)
|
||||
from fenris.status import NewerSchema, open_store_readonly
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _make_store(path: Path, version: int = 0) -> sqlite3.Connection:
|
||||
"""Create a store at *path* with the given user_version."""
|
||||
conn = sqlite3.connect(str(path))
|
||||
conn.execute("PRAGMA journal_mode=WAL")
|
||||
if version == 0:
|
||||
# Fresh DB with no schema — user_version defaults to 0
|
||||
pass
|
||||
else:
|
||||
# Create a minimal schema so the DB is valid, then set version
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS samples (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
ts TEXT NOT NULL,
|
||||
device TEXT NOT NULL
|
||||
)
|
||||
""")
|
||||
conn.execute(f"PRAGMA user_version={version}")
|
||||
conn.commit()
|
||||
return conn
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# migrate_to_latest
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestMigrateToLatest:
|
||||
"""Forward-only migration via migrate_to_latest()."""
|
||||
|
||||
def test_migrates_from_zero(self, tmp_path):
|
||||
"""Store at user_version=0 → SCHEMA_VERSION (fresh DB)."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=0)
|
||||
|
||||
steps = migrate_to_latest(db)
|
||||
|
||||
# SCHEMA_VERSION - 0 = SCHEMA_VERSION migration steps
|
||||
assert steps == SCHEMA_VERSION
|
||||
|
||||
# Verify version was bumped
|
||||
conn = sqlite3.connect(str(db))
|
||||
v = conn.execute("PRAGMA user_version").fetchone()[0]
|
||||
conn.close()
|
||||
assert v == SCHEMA_VERSION
|
||||
|
||||
def test_already_current_returns_zero(self, tmp_path):
|
||||
"""Store already at SCHEMA_VERSION → 0 steps applied."""
|
||||
db = tmp_path / "observations.db"
|
||||
conn = _make_store(db, version=SCHEMA_VERSION)
|
||||
conn.close()
|
||||
|
||||
steps = migrate_to_latest(db)
|
||||
assert steps == 0
|
||||
|
||||
def test_refuses_newer_store(self, tmp_path):
|
||||
"""Store with user_version > SCHEMA_VERSION → ValueError."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=SCHEMA_VERSION + 1)
|
||||
|
||||
with pytest.raises(ValueError, match="newer Fenris"):
|
||||
migrate_to_latest(db)
|
||||
|
||||
def test_refuses_much_newer_store(self, tmp_path):
|
||||
"""Store several versions ahead → ValueError."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=SCHEMA_VERSION + 5)
|
||||
|
||||
with pytest.raises(ValueError, match="newer Fenris"):
|
||||
migrate_to_latest(db)
|
||||
|
||||
def test_store_not_corrupted_on_refusal(self, tmp_path):
|
||||
"""After refusal, store is unchanged (no silent corruption)."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=SCHEMA_VERSION + 2)
|
||||
|
||||
with pytest.raises(ValueError):
|
||||
migrate_to_latest(db)
|
||||
|
||||
# Version should be unchanged
|
||||
conn = sqlite3.connect(str(db))
|
||||
v = conn.execute("PRAGMA user_version").fetchone()[0]
|
||||
conn.close()
|
||||
assert v == SCHEMA_VERSION + 2
|
||||
|
||||
def test_idempotent_on_current(self, tmp_path):
|
||||
"""Calling migrate_to_latest twice on a current store is safe."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=SCHEMA_VERSION)
|
||||
|
||||
assert migrate_to_latest(db) == 0
|
||||
assert migrate_to_latest(db) == 0
|
||||
|
||||
def test_migrates_intermediate_version(self, tmp_path):
|
||||
"""Store at version 1 with SCHEMA_VERSION=1 → 0 steps (current)."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=1)
|
||||
# SCHEMA_VERSION is 1, so version 1 is current
|
||||
steps = migrate_to_latest(db)
|
||||
assert steps == 0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# init_store — downgrade refusal
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestInitStoreDowngradeRefusal:
|
||||
"""init_store() refuses newer-schema stores."""
|
||||
|
||||
def test_refuses_newer_store(self, tmp_path):
|
||||
"""init_store raises ValueError on newer-schema store."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=SCHEMA_VERSION + 1)
|
||||
|
||||
with pytest.raises(ValueError, match="newer Fenris"):
|
||||
init_store(db)
|
||||
|
||||
def test_store_not_corrupted_on_refusal(self, tmp_path):
|
||||
"""After init_store refusal, store is unchanged."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=SCHEMA_VERSION + 1)
|
||||
|
||||
with pytest.raises(ValueError):
|
||||
init_store(db)
|
||||
|
||||
conn = sqlite3.connect(str(db))
|
||||
v = conn.execute("PRAGMA user_version").fetchone()[0]
|
||||
conn.close()
|
||||
assert v == SCHEMA_VERSION + 1
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# open_store_readonly — downgrade refusal
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestOpenStoreReadonlyDowngradeRefusal:
|
||||
"""open_store_readonly() raises NewerSchema on newer-schema stores."""
|
||||
|
||||
def test_raises_newer_schema(self, tmp_path):
|
||||
"""Newer store → NewerSchema exception."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=SCHEMA_VERSION + 1)
|
||||
|
||||
with pytest.raises(NewerSchema) as exc_info:
|
||||
open_store_readonly(db)
|
||||
|
||||
assert exc_info.value.version == SCHEMA_VERSION + 1
|
||||
|
||||
def test_store_not_corrupted(self, tmp_path):
|
||||
"""After NewerSchema refusal, store is unchanged."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=SCHEMA_VERSION + 3)
|
||||
|
||||
with pytest.raises(NewerSchema):
|
||||
open_store_readonly(db)
|
||||
|
||||
conn = sqlite3.connect(str(db))
|
||||
v = conn.execute("PRAGMA user_version").fetchone()[0]
|
||||
conn.close()
|
||||
assert v == SCHEMA_VERSION + 3
|
||||
|
||||
def test_current_store_opens(self, tmp_path):
|
||||
"""Store at SCHEMA_VERSION opens without error."""
|
||||
db = tmp_path / "observations.db"
|
||||
_make_store(db, version=SCHEMA_VERSION)
|
||||
|
||||
conn = open_store_readonly(db)
|
||||
assert conn is not None
|
||||
conn.close()
|
||||
@@ -1,681 +0,0 @@
|
||||
"""Headless tests for the Panes TUI (issue #28).
|
||||
|
||||
Covers:
|
||||
- CI-1: Exhaustive state matrix from synthetic stores
|
||||
- TUI-1: One dense keyboard-first screen with four normative regions
|
||||
- TUI-4: Layout regions normative per register
|
||||
- CI-4: Six disclosures verbatim, empty-store greeting, first-run opt-in
|
||||
- IN-3: First-run TUI prompt enables timer and opens first period
|
||||
|
||||
Criteria: TUI-1, TUI-4, CI-1, CI-4, IN-3.
|
||||
"""
|
||||
import sqlite3
|
||||
from xml.etree import ElementTree
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch, MagicMock
|
||||
|
||||
import pytest
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
|
||||
|
||||
from textual.app import App
|
||||
from textual.pilot import Pilot
|
||||
|
||||
from fenris.store import init_store, SCHEMA_VERSION
|
||||
from fenris.monitoring_periods import ensure_period_open, close_period
|
||||
from fenris.projection import (
|
||||
ConfidenceState,
|
||||
compute_projection,
|
||||
DISCLOSURES,
|
||||
WARMING_MIN_DAYS,
|
||||
STALENESS_HOURS,
|
||||
YOUNG_REGIME_DAYS,
|
||||
)
|
||||
from fenris.status import (
|
||||
FRESH_THRESHOLD_S,
|
||||
STALENESS_THRESHOLD_S,
|
||||
grade_freshness,
|
||||
)
|
||||
from fenris.tui import (
|
||||
FenrisTuiApp,
|
||||
_format_remaining,
|
||||
_sparkline,
|
||||
_habit_bar,
|
||||
_query_usage_history,
|
||||
_query_drive_health,
|
||||
_query_service_facts,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _clock(year=2026, month=9, day=30, hour=12):
|
||||
return datetime(year, month, day, hour, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
def _insert_baseline(conn, tbw_tb=1.0, verified=True,
|
||||
model="Samsung SSD 970 EVO Plus 1TB"):
|
||||
conn.execute(
|
||||
"INSERT INTO endurance_baseline "
|
||||
"(tbw_terabytes, source_url, document_revision, entry_date, model_string, "
|
||||
" nominal_capacity_bytes, validated_by, verified, created_at, updated_at) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(tbw_tb, "https://example.com/spec", "v1.0", "2026-01-01", model,
|
||||
1024000000000, "machine_match" if verified else None, verified,
|
||||
"2026-01-01T00:00:00+00:00", "2026-01-01T00:00:00+00:00"),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _insert_segment(conn, opened_at="2026-09-01T00:00:00+00:00",
|
||||
identity_key="nqn.test", degraded=False,
|
||||
mn="Samsung SSD 970 EVO Plus 1TB"):
|
||||
conn.execute(
|
||||
"INSERT INTO controller_segments "
|
||||
"(opened_at, identity_key, identity_degraded, subnqn, sn, mn, fr, vid, ssvid, transport) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(opened_at, identity_key, degraded, "nqn.test", "SN123", mn, "FW1",
|
||||
"0x144d", "0x144d", "pcie"),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _insert_day(conn, day, bw=1024*1024*100, coverage=0.95, samples=24):
|
||||
conn.execute(
|
||||
"INSERT INTO day_aggregates (day, active_seconds, idle_seconds, powered_off_seconds, "
|
||||
"unknown_seconds, bytes_written_delta, bytes_read_delta, sample_count, coverage) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(day, 3600, 0, 0, 0, bw, 0, samples, coverage),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _insert_sample(conn, ts, pu=5, device="/dev/nvme0n1"):
|
||||
conn.execute(
|
||||
"INSERT INTO samples (ts, device, data_units_written, data_units_read, "
|
||||
"percentage_used, bytes_written, bytes_read, power_on_hours) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(ts, device, 1000000, 500000, pu, 512000000000, 256000000000, 8765),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _open_period(conn, start="2026-09-01T00:00:00+00:00"):
|
||||
ensure_period_open(conn, datetime.fromisoformat(start))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Unit tests for helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestFormatRemaining:
|
||||
def test_hours_only(self):
|
||||
assert _format_remaining(3600) == "1 h"
|
||||
|
||||
def test_days_and_hours(self):
|
||||
assert _format_remaining(86400) == "1 d 0 h"
|
||||
|
||||
def test_years(self):
|
||||
assert _format_remaining(31557600) == "1 yr 0 d 0 h"
|
||||
|
||||
def test_zero(self):
|
||||
assert _format_remaining(0) == "endurance exhausted"
|
||||
|
||||
def test_negative(self):
|
||||
assert _format_remaining(-100) == "endurance exhausted"
|
||||
|
||||
|
||||
class TestSparkline:
|
||||
def test_empty(self):
|
||||
assert _sparkline([]) == ""
|
||||
|
||||
def test_single_value(self):
|
||||
result = _sparkline([100.0])
|
||||
assert len(result) == 1
|
||||
|
||||
def test_multiple_values(self):
|
||||
result = _sparkline([1.0, 2.0, 3.0, 4.0, 5.0])
|
||||
assert len(result) > 0
|
||||
assert all(c in " ▁▂▃▄▅▆▇█" for c in result)
|
||||
|
||||
def test_width_limit(self):
|
||||
result = _sparkline([1.0] * 100, width=20)
|
||||
assert len(result) <= 20
|
||||
|
||||
|
||||
class TestHabitBar:
|
||||
def test_all_active(self):
|
||||
result = _habit_bar(1.0, 0.0, 0.0, 0.0)
|
||||
assert "active 100%" in result
|
||||
|
||||
def test_mixed(self):
|
||||
result = _habit_bar(0.5, 0.3, 0.1, 0.1)
|
||||
assert "active 50%" in result
|
||||
assert "idle 30%" in result
|
||||
|
||||
def test_all_unknown(self):
|
||||
result = _habit_bar(0.0, 0.0, 0.0, 1.0)
|
||||
assert "unknown 100%" in result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Data query tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestQueryUsageHistory:
|
||||
def test_empty_store(self, tmp_path):
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
result = _query_usage_history(conn)
|
||||
assert result["num_days"] == 0
|
||||
assert result["sparkline"] == ""
|
||||
conn.close()
|
||||
|
||||
def test_with_days(self, tmp_path):
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(14):
|
||||
d = (datetime(2026, 9, 15) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100)
|
||||
result = _query_usage_history(conn)
|
||||
assert result["num_days"] == 14
|
||||
assert result["sparkline"] != ""
|
||||
conn.close()
|
||||
|
||||
|
||||
class TestQueryDriveHealth:
|
||||
def test_empty_store(self, tmp_path):
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
result = _query_drive_health(conn)
|
||||
assert result["model"] == "unknown"
|
||||
conn.close()
|
||||
|
||||
def test_with_sample(self, tmp_path):
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00", pu=10)
|
||||
result = _query_drive_health(conn)
|
||||
assert result["percentage_used"] == 10
|
||||
conn.close()
|
||||
|
||||
|
||||
class TestQueryServiceFacts:
|
||||
def test_empty_store(self, tmp_path):
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
now = _clock()
|
||||
result = _query_service_facts(conn, now)
|
||||
assert result["freshness"] == "empty"
|
||||
conn.close()
|
||||
|
||||
def test_fresh_sample(self, tmp_path):
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
_insert_day(conn, "2026-09-29", bw=1024*1024*100)
|
||||
now = _clock()
|
||||
ts = (now - timedelta(minutes=2)).isoformat()
|
||||
_insert_sample(conn, ts)
|
||||
result = _query_service_facts(conn, now)
|
||||
assert result["freshness"] == "fresh"
|
||||
conn.close()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# CI-1: Exhaustive state matrix from synthetic stores
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestStateMatrix:
|
||||
"""TUI renders every realizable combination of confidence state × freshness × baseline tier."""
|
||||
|
||||
def test_unsupported_no_baseline(self, tmp_path):
|
||||
"""No baseline → Unavailable."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert proj.headline_remaining_seconds is None
|
||||
conn.close()
|
||||
|
||||
def test_limited_warming(self, tmp_path):
|
||||
"""Warming up (< 14 days) → Limited."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_baseline(conn, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(10):
|
||||
d = (datetime(2026, 9, 20) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.LIMITED
|
||||
assert proj.headline_remaining_seconds is not None
|
||||
conn.close()
|
||||
|
||||
def test_supported_full(self, tmp_path):
|
||||
"""Full data → Supported."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_baseline(conn, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100, coverage=0.95, samples=24)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.SUPPORTED
|
||||
assert proj.headline_remaining_seconds is not None
|
||||
conn.close()
|
||||
|
||||
def test_stale_freshness(self, tmp_path):
|
||||
"""Stale data: newest day aggregate > 48h old → Limited or Unsupported."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_baseline(conn, tbw_tb=1.0, verified=True)
|
||||
# Segment opened long ago so days are old
|
||||
_insert_segment(conn, opened_at="2026-08-01T00:00:00+00:00")
|
||||
_open_period(conn, start="2026-08-01T00:00:00+00:00")
|
||||
# Days all end on Aug 30 — 31 days before clock (Sept 30)
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 8, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100, coverage=0.95, samples=24)
|
||||
# Sample also old
|
||||
stale_ts = (_clock() - timedelta(days=31)).isoformat()
|
||||
_insert_sample(conn, stale_ts)
|
||||
proj = compute_projection(conn, _clock())
|
||||
# Stale data (> 48h since newest day) → not Supported
|
||||
assert proj.confidence_state != ConfidenceState.SUPPORTED
|
||||
conn.close()
|
||||
|
||||
def test_empty_store_state(self, tmp_path):
|
||||
"""Empty store → no projection, headline=None."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
assert proj.headline_remaining_seconds is None
|
||||
conn.close()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# TUI-1: One dense keyboard-first screen
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestDenseScreen:
|
||||
"""TUI renders one dense screen with four normative regions."""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_four_regions_exist(self, tmp_path):
|
||||
"""All four normative regions are present in the DOM."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
conn.close()
|
||||
|
||||
app = FenrisTuiApp(store_path=tmp_path / "test.db")
|
||||
async with app.run_test() as pilot:
|
||||
assert app.query_one("#headline-band") is not None
|
||||
assert app.query_one("#usage-history") is not None
|
||||
assert app.query_one("#drive-health") is not None
|
||||
assert app.query_one("#service-strip") is not None
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_headline_contains_projection(self, tmp_path):
|
||||
"""Headline band shows projection headline."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_baseline(conn, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100, coverage=0.95, samples=24)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
conn.close()
|
||||
|
||||
app = FenrisTuiApp(store_path=tmp_path / "test.db")
|
||||
async with app.run_test() as pilot:
|
||||
headline = str(app.query_one("#headline-band").render())
|
||||
assert "remaining" in headline.lower() or "projection" in headline.lower()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_confidence_rendered_as_evidence(self, tmp_path):
|
||||
"""Confidence is state + contributing facts, never a percentage."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_baseline(conn, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100, coverage=0.95, samples=24)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
conn.close()
|
||||
|
||||
app = FenrisTuiApp(store_path=tmp_path / "test.db")
|
||||
async with app.run_test() as pilot:
|
||||
headline = str(app.query_one("#headline-band").render())
|
||||
assert "projection confidence" in headline.lower()
|
||||
# Contributing facts shown, never a percentage as confidence
|
||||
# (percentage in "95% interval coverage" is allowed as a fact, not as confidence)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_service_strip_has_four_facts(self, tmp_path):
|
||||
"""Service strip has four separate facts (boot, timer, collect, freshness)."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
conn.close()
|
||||
|
||||
app = FenrisTuiApp(store_path=tmp_path / "test.db")
|
||||
async with app.run_test() as pilot:
|
||||
strip = str(app.query_one("#service-strip").render())
|
||||
assert "boot:" in strip
|
||||
assert "timer:" in strip
|
||||
assert "last collect:" in strip
|
||||
assert "freshness:" in strip
|
||||
assert "by Bongbetic" in strip
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_service_strip_shows_continuity_and_separate_quit_rail(self, tmp_path):
|
||||
"""The visible action footer excludes quit because the rail owns it."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_open_period(conn)
|
||||
conn.close()
|
||||
app = FenrisTuiApp(store_path=tmp_path / "test.db")
|
||||
|
||||
with patch("fenris.tui.query_service_state", return_value={
|
||||
"boot_enabled": True, "timer_active": True,
|
||||
"last_collect_ok": True, "last_collect_age_s": 60,
|
||||
"last_collect_reason": None,
|
||||
}):
|
||||
async with app.run_test():
|
||||
strip = str(app.query_one("#service-strip").render()).lower()
|
||||
rail = str(app.query_one("#quit-rail").render())
|
||||
usage = app.query_one("#usage-history")
|
||||
service = app.query_one("#service-strip")
|
||||
quit_rail = app.query_one("#quit-rail")
|
||||
assert "continuity" in strip
|
||||
assert "monitoring: active in background · persists across reboots" in strip
|
||||
assert "p pause · r resume · c collect · d disclosures" in strip
|
||||
assert "q quit" not in strip
|
||||
assert rail == "q QUIT TUI"
|
||||
assert usage.region.y < service.region.y < quit_rail.region.y
|
||||
assert usage.region.bottom <= service.region.y
|
||||
assert service.region.bottom <= quit_rail.region.y
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_deliberate_pause_banner_is_visible_and_quit_preserves_periods(self, tmp_path):
|
||||
"""The Pilot sees the paused block; q leaves persisted monitoring state alone."""
|
||||
db = tmp_path / "test.db"
|
||||
conn = init_store(db)
|
||||
conn.execute(
|
||||
"INSERT INTO monitoring_periods (started_at, ended_at, end_cause) "
|
||||
"VALUES (?, ?, ?)",
|
||||
("2026-09-01T09:00:00+00:00", "2026-09-01T10:00:00+00:00", "user_disabled"),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
app = FenrisTuiApp(store_path=db)
|
||||
|
||||
with patch("fenris.tui.query_service_state", return_value={
|
||||
"boot_enabled": False, "timer_active": False,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}), patch("fenris.tui.subprocess.run") as subprocess_run, patch.object(
|
||||
app, "_run_helper"
|
||||
) as run_helper:
|
||||
async with app.run_test(size=(80, 24)) as pilot:
|
||||
await pilot.pause()
|
||||
paused_banner = app.query_one("#paused-banner")
|
||||
main_grid = app.query_one("#main-grid")
|
||||
assert str(main_grid.styles.layout) == "<grid>"
|
||||
assert main_grid.has_class("paused")
|
||||
assert len(main_grid.styles.grid_rows) == 5
|
||||
banner = str(paused_banner.render()).lower()
|
||||
assert "monitoring: paused — deliberate disable" in banner
|
||||
assert "paused time is excluded from your usage habit · resume: fenris monitor resume" in banner
|
||||
assert paused_banner.region.height >= 5
|
||||
assert paused_banner.region.y < app.query_one("#usage-history").region.y
|
||||
assert app.query_one("#usage-history").region.bottom <= app.query_one(
|
||||
"#service-strip"
|
||||
).region.y
|
||||
screenshot = app.export_screenshot()
|
||||
visible_text = " ".join(
|
||||
"".join(ElementTree.fromstring(screenshot).itertext()).split()
|
||||
)
|
||||
assert "paused time is excluded from your usage habit" in visible_text
|
||||
assert "resume: fenris monitor resume" in visible_text
|
||||
assert "monitoring: does not start on next boot" in str(
|
||||
app.query_one("#service-strip").render()
|
||||
).lower()
|
||||
dashboard_scroll = app.query_one("#dashboard-scroll")
|
||||
assert dashboard_scroll.max_scroll_y > 0
|
||||
dashboard_scroll.focus()
|
||||
await pilot.press("end")
|
||||
assert dashboard_scroll.scroll_y == dashboard_scroll.max_scroll_y
|
||||
footer_text = " ".join(
|
||||
"".join(
|
||||
ElementTree.fromstring(app.export_screenshot()).itertext()
|
||||
).split()
|
||||
)
|
||||
assert "q QUIT TUI" in footer_text
|
||||
await pilot.press("q")
|
||||
assert not app.is_running
|
||||
subprocess_run.assert_not_called()
|
||||
run_helper.assert_not_called()
|
||||
|
||||
conn = sqlite3.connect(db)
|
||||
row = conn.execute(
|
||||
"SELECT ended_at, end_cause FROM monitoring_periods"
|
||||
).fetchone()
|
||||
conn.close()
|
||||
assert row == ("2026-09-01T10:00:00+00:00", "user_disabled")
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_quit_preserves_an_active_monitoring_period(self, tmp_path):
|
||||
"""Quitting an active dashboard never closes or mutates its period."""
|
||||
db = tmp_path / "test.db"
|
||||
conn = init_store(db)
|
||||
_open_period(conn, "2026-09-01T09:00:00+00:00")
|
||||
conn.close()
|
||||
app = FenrisTuiApp(store_path=db)
|
||||
|
||||
with patch("fenris.tui.query_service_state", return_value={
|
||||
"boot_enabled": True, "timer_active": True,
|
||||
"last_collect_ok": None, "last_collect_age_s": None,
|
||||
"last_collect_reason": None,
|
||||
}), patch("fenris.tui.subprocess.run") as subprocess_run, patch.object(
|
||||
app, "_run_helper"
|
||||
) as run_helper:
|
||||
async with app.run_test() as pilot:
|
||||
await pilot.press("q")
|
||||
assert not app.is_running
|
||||
subprocess_run.assert_not_called()
|
||||
run_helper.assert_not_called()
|
||||
|
||||
conn = sqlite3.connect(db)
|
||||
row = conn.execute(
|
||||
"SELECT started_at, ended_at, end_cause FROM monitoring_periods"
|
||||
).fetchone()
|
||||
conn.close()
|
||||
assert row == ("2026-09-01T09:00:00+00:00", None, None)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_branding_and_one_time_auth_banner(self, tmp_path):
|
||||
"""Identity is visible at launch; auth notice clears once per session."""
|
||||
app = FenrisTuiApp(
|
||||
store_path=tmp_path / "nonexistent.db",
|
||||
refresh_interval_s=0.2,
|
||||
)
|
||||
auth_notice = "privileged actions will prompt for authentication (polkit)"
|
||||
|
||||
async with app.run_test() as pilot:
|
||||
headline = str(app.query_one("#headline-band").render())
|
||||
assert "Fenris — NVMe endurance monitor" in headline
|
||||
assert auth_notice in headline
|
||||
assert "by Bongbetic" in str(app.query_one("#service-strip").render())
|
||||
|
||||
await pilot.pause(0.25)
|
||||
assert auth_notice not in str(app.query_one("#headline-band").render())
|
||||
|
||||
await pilot.pause(0.25)
|
||||
assert auth_notice not in str(app.query_one("#headline-band").render())
|
||||
|
||||
fresh_app = FenrisTuiApp(
|
||||
store_path=tmp_path / "nonexistent.db",
|
||||
refresh_interval_s=0.2,
|
||||
)
|
||||
async with fresh_app.run_test():
|
||||
assert auth_notice in str(fresh_app.query_one("#headline-band").render())
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# CI-4: Disclosures, empty-store greeting, first-run
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestDisclosuresAndGreeting:
|
||||
@pytest.mark.asyncio
|
||||
async def test_disclosures_screen(self, tmp_path):
|
||||
"""Disclosures view renders six disclosures verbatim."""
|
||||
app = FenrisTuiApp(store_path=tmp_path / "nonexistent.db")
|
||||
async with app.run_test() as pilot:
|
||||
await pilot.press("d")
|
||||
# Modal should be pushed
|
||||
assert len(app.screen_stack) > 1
|
||||
disc_text = str(app.screen.query_one("Static").render())
|
||||
for i in range(1, 7):
|
||||
assert "%d." % i in disc_text
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_empty_store_greeting(self, tmp_path):
|
||||
"""Empty store shows 'no observations yet' with enable hint."""
|
||||
app = FenrisTuiApp(store_path=tmp_path / "nonexistent.db")
|
||||
async with app.run_test() as pilot:
|
||||
headline = str(app.query_one("#headline-band").render())
|
||||
assert "no observations yet" in headline.lower()
|
||||
assert "enable" in headline.lower() or "resume" in headline.lower()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# IN-3: First-run opt-in
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestFirstRun:
|
||||
@pytest.mark.asyncio
|
||||
async def test_first_run_prompt(self, tmp_path):
|
||||
"""First-run prompt enables timer and opens first period."""
|
||||
app = FenrisTuiApp(store_path=tmp_path / "nonexistent.db")
|
||||
async with app.run_test() as pilot:
|
||||
headline = str(app.query_one("#headline-band").render())
|
||||
assert "no observations yet" in headline.lower()
|
||||
# The enable hint should mention resume
|
||||
assert "resume" in headline.lower()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# TUI-4: Layout regions normative
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestLayoutNormative:
|
||||
@pytest.mark.asyncio
|
||||
async def test_no_page_navigation(self, tmp_path):
|
||||
"""No page navigation keys exist (variant switching was prototype-only)."""
|
||||
app = FenrisTuiApp(store_path=tmp_path / "nonexistent.db")
|
||||
# Check that only production bindings exist
|
||||
binding_keys = {b.key for b in app.BINDINGS}
|
||||
assert "left" not in binding_keys
|
||||
assert "right" not in binding_keys
|
||||
assert "1" not in binding_keys
|
||||
assert "2" not in binding_keys
|
||||
assert "3" not in binding_keys
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_pause_resume_asymmetry(self, tmp_path):
|
||||
"""Pause asks, resume does not (TUI-2)."""
|
||||
app = FenrisTuiApp(store_path=tmp_path / "nonexistent.db")
|
||||
async with app.run_test() as pilot:
|
||||
# Pause should push a confirmation screen
|
||||
await pilot.press("p")
|
||||
assert len(app.screen_stack) > 1
|
||||
# Press n to cancel
|
||||
await pilot.press("n")
|
||||
assert len(app.screen_stack) == 1
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# CI-1: State matrix exhaustive combinations
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestStateMatrixCombinations:
|
||||
"""CI-1: confidence × freshness × baseline tier combinations."""
|
||||
|
||||
def test_unsupported_with_fresh_data(self, tmp_path):
|
||||
"""Fresh data but no baseline → Unavailable + fresh."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 10) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100)
|
||||
# Sample must be within FRESH_THRESHOLD_S of clock
|
||||
fresh_ts = (_clock() - timedelta(seconds=FRESH_THRESHOLD_S - 10)).isoformat()
|
||||
_insert_sample(conn, fresh_ts)
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state == ConfidenceState.UNSUPPORTED
|
||||
freshness = grade_freshness(fresh_ts, _clock())
|
||||
assert freshness == "fresh"
|
||||
conn.close()
|
||||
|
||||
def test_limited_with_stale_data(self, tmp_path):
|
||||
"""Stale data with baseline → Limited or Unsupported (young regime + stale)."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
_insert_baseline(conn, tbw_tb=1.0, verified=True)
|
||||
_insert_segment(conn, opened_at="2026-09-01T00:00:00+00:00")
|
||||
_open_period(conn)
|
||||
for i in range(20):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100)
|
||||
stale_ts = (_clock() - timedelta(days=3)).isoformat()
|
||||
_insert_sample(conn, stale_ts)
|
||||
proj = compute_projection(conn, _clock())
|
||||
assert proj.confidence_state != ConfidenceState.SUPPORTED
|
||||
freshness = grade_freshness(stale_ts, _clock())
|
||||
assert freshness == "stale"
|
||||
conn.close()
|
||||
|
||||
def test_supported_with_unverified_baseline(self, tmp_path):
|
||||
"""Incomplete provenance → unverified baseline tier."""
|
||||
conn = init_store(tmp_path / "test.db")
|
||||
# Insert baseline with incomplete provenance (missing source_url)
|
||||
conn.execute(
|
||||
"INSERT INTO endurance_baseline "
|
||||
"(tbw_terabytes, source_url, document_revision, entry_date, model_string, "
|
||||
" nominal_capacity_bytes, validated_by, verified, created_at, updated_at) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
(1.0, None, "v1.0", "2026-01-01", "Samsung SSD 970 EVO Plus 1TB",
|
||||
1024000000000, None, False,
|
||||
"2026-01-01T00:00:00+00:00", "2026-01-01T00:00:00+00:00"),
|
||||
)
|
||||
conn.commit()
|
||||
_insert_segment(conn)
|
||||
_open_period(conn)
|
||||
for i in range(30):
|
||||
d = (datetime(2026, 9, 1) + timedelta(days=i)).strftime("%Y-%m-%d")
|
||||
_insert_day(conn, d, bw=1024*1024*100, coverage=0.95, samples=24)
|
||||
_insert_sample(conn, "2026-09-30T10:00:00+00:00")
|
||||
proj = compute_projection(conn, _clock())
|
||||
# Incomplete provenance → UNVERIFIED tier
|
||||
assert proj.baseline_tier.value == "unverified_override"
|
||||
conn.close()
|
||||
@@ -1,10 +0,0 @@
|
||||
[Unit]
|
||||
Description=Fenris NVMe collection service
|
||||
Documentation=https://git.bongbetic.com/xavierk/Fenris
|
||||
After=local-fs.target
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/libexec/fenris/fenris-collect
|
||||
TimeoutStartSec=90
|
||||
UMask=002
|
||||
@@ -1,12 +0,0 @@
|
||||
[Unit]
|
||||
Description=Fenris collection timer
|
||||
Documentation=https://git.bongbetic.com/xavierk/Fenris
|
||||
|
||||
[Timer]
|
||||
OnBootSec=2min
|
||||
OnUnitInactiveSec=5min
|
||||
AccuracySec=30s
|
||||
Persistent=no
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
Reference in New Issue
Block a user