feat(hour/day derivation): hour classification, monitoring periods, day aggregates, pruning

Hour classification (PR-4):
- Powered-off: POH delta < 90% of wall-clock span
- Active: DUW delta >= 256 MiB
- Idle: powered on + sampled + below active threshold
- Unknown: unsampled without POH evidence
- Four splits sum to exactly wall_clock_seconds
- Disabled time is never an hour state

Monitoring periods (FL-8):
- ensure_period_open: opens period at run moment if none exists
- close_period: closes with end cause
- is_inside_period: checks timestamp against period bounds
- Never backdated; wall-clock outside periods excluded from denominator

Day aggregates (ST-4, PR-5):
- Derived monotonically from hour rows
- UTC-bounded; no 23/25-hour days
- Coverage: known seconds / period wall-clock
- Gap hours inside periods contribute unknown seconds
- Hours outside periods excluded entirely
- No absent hour interpolated/estimated/fabricated (FL-3)

Raw sample pruning (ST-5):
- Prunes samples older than 14 days
- Hour observations and day aggregates retained indefinitely

Closes #22
This commit is contained in:
xavierk
2026-09-01 22:58:03 +05:30
parent 2217b00ff6
commit 7d219c4697
8 changed files with 1117 additions and 0 deletions
+165
View File
@@ -0,0 +1,165 @@
"""Day aggregate derivation per spec §5.4, §3.3.
One row per UTC day, derived monotonically from hour rows — the grain at
which usage-habit evidence is judged. No absent hour is ever interpolated,
estimated, or fabricated (§5.3, FL-3).
Coverage: the share of wall-clock seconds inside monitoring periods whose
usage-habit classification is known rather than unknown (§5.3).
"""
import sqlite3
from dataclasses import dataclass
from datetime import datetime, timedelta, timezone
@dataclass(frozen=True)
class DayAggregate:
"""One UTC day's aggregated stats."""
day: str # ISO 8601 UTC date, e.g. "2026-09-01"
seconds_active: int
seconds_idle: int
seconds_powered_off: int
seconds_unknown: int
bytes_written_delta: int
bytes_read_delta: int
sample_count: int
coverage: float
def _period_wall_clock_for_day(conn: sqlite3.Connection, day: str) -> int:
"""Total wall-clock seconds inside monitoring periods for a UTC day.
Clamps each period to the day boundary [dayT00:00, dayT24:00).
"""
day_start = datetime.fromisoformat(f"{day}T00:00:00+00:00")
day_end = day_start + timedelta(days=1)
day_start_str = day_start.isoformat()
day_end_str = day_end.isoformat()
cursor = conn.execute(
"SELECT started_at, ended_at FROM monitoring_periods "
"WHERE (ended_at IS NULL OR ended_at > ?) AND started_at < ? "
"ORDER BY started_at",
(day_start_str, day_end_str),
)
total = 0
for row in cursor.fetchall():
period_start = row[0]
period_end = row[1]
effective_start = max(period_start, day_start_str)
if period_end is not None:
effective_end = min(period_end, day_end_str)
else:
effective_end = day_end_str
if effective_start < effective_end:
s = datetime.fromisoformat(effective_start)
e = datetime.fromisoformat(effective_end)
total += int((e - s).total_seconds())
return total
def _hour_overlaps_period(conn: sqlite3.Connection, hour_iso: str) -> bool:
"""Check if an hour's wall-clock span overlaps any monitoring period."""
hour_start = datetime.fromisoformat(hour_iso)
hour_end = hour_start + timedelta(hours=1)
hs = hour_start.isoformat()
he = hour_end.isoformat()
cursor = conn.execute(
"SELECT 1 FROM monitoring_periods "
"WHERE started_at < ? AND (ended_at IS NULL OR ended_at > ?) "
"LIMIT 1",
(he, hs),
)
return cursor.fetchone() is not None
def derive_day(conn: sqlite3.Connection, day: str) -> DayAggregate | None:
"""Derive a single day aggregate from its hour rows + monitoring periods.
Only hours overlapping a monitoring period contribute to the aggregate.
Gap hours inside periods contribute unknown seconds. Hours outside all
monitoring periods are excluded entirely (§5.2).
Returns None if no hours exist for the day.
"""
cursor = conn.execute(
"SELECT hour, active_seconds, idle_seconds, powered_off_seconds, unknown_seconds, "
" bytes_written_delta, bytes_read_delta, sample_count "
"FROM hour_observations "
"WHERE hour LIKE ? "
"ORDER BY hour",
(day + "T%",),
)
rows = cursor.fetchall()
if not rows:
return None
total_active = 0
total_idle = 0
total_powered_off = 0
total_unknown_from_hours = 0
total_bw = 0
total_br = 0
total_samples = 0
total_hour_wall_clock = 0
for row in rows:
# Only count hours overlapping a monitoring period
if not _hour_overlaps_period(conn, row[0]):
continue
total_active += row[1]
total_idle += row[2]
total_powered_off += row[3]
total_unknown_from_hours += row[4]
total_bw += row[5]
total_br += row[6]
total_samples += row[7]
total_hour_wall_clock += row[1] + row[2] + row[3] + row[4]
# Wall-clock seconds inside monitoring periods for this day
period_wc = _period_wall_clock_for_day(conn, day)
# Gap seconds = period wall-clock - sum of existing hour wall-clock
gap_seconds = max(0, period_wc - total_hour_wall_clock)
total_unknown = total_unknown_from_hours + gap_seconds
# Coverage: known seconds / period wall-clock (§5.2, §5.3)
known_seconds = total_active + total_idle + total_powered_off
coverage = known_seconds / period_wc if period_wc > 0 else 0.0
return DayAggregate(
day=day,
seconds_active=total_active,
seconds_idle=total_idle,
seconds_powered_off=total_powered_off,
seconds_unknown=total_unknown,
bytes_written_delta=total_bw,
bytes_read_delta=total_br,
sample_count=total_samples,
coverage=coverage,
)
def derive_all_days(conn: sqlite3.Connection) -> list[DayAggregate]:
"""Derive day aggregates for all days that have hour rows.
Returns days sorted by date.
"""
cursor = conn.execute(
"SELECT DISTINCT substr(hour, 1, 10) as day FROM hour_observations ORDER BY day"
)
days = [row[0] for row in cursor.fetchall()]
results = []
for day in days:
agg = derive_day(conn, day)
if agg is not None:
results.append(agg)
return results
+94
View File
@@ -0,0 +1,94 @@
"""Hour classification per spec §5.1.
Each UTC hour is classified by named constants, in this order of evidence:
- Powered-off: power-on-hours delta < 90% of wall-clock span
- Active: DUW delta >= 256 MiB in the hour
- Idle: powered on, sampled, below active threshold
- Unknown: everything else (unsampled without POH evidence)
Four splits sum to exactly wall_clock_seconds. Disabled time is never an
hour state — it is wall-clock outside monitoring periods (§5.2).
"""
from dataclasses import dataclass
# Spec §5.1: Active hour threshold — 256 MiB DUW delta
ACTIVE_THRESHOLD_BYTES = 256 * 1024 * 1024 # 256 MiB
# Spec §5.1: Powered-off threshold — 90% of wall-clock span
POWERED_OFF_THRESHOLD_PERCENT = 0.90
@dataclass(frozen=True)
class HourSplit:
"""Usage-habit split for one UTC hour. Fields sum to wall_clock_seconds."""
seconds_active: int
seconds_idle: int
seconds_powered_off: int
seconds_unknown: int
def classify_hour(
wall_clock_seconds: int,
poh_delta: int,
duw_delta: int,
dur_delta: int,
sampled_seconds: int | None = None,
) -> HourSplit:
"""Classify a UTC hour into the four usage-habit states.
Args:
wall_clock_seconds: Total seconds in this hour boundary (3600 for a
full hour, less for partial-hours at period edges).
poh_delta: Power-on-hours delta since previous sample (in seconds).
duw_delta: Data-units-written delta since previous sample (in bytes).
dur_delta: Data-units-read delta since previous sample (in bytes).
sampled_seconds: Seconds within this hour covered by a sample.
None or 0 means no sample fell in this hour.
Returns:
HourSplit whose four fields sum to wall_clock_seconds.
"""
if sampled_seconds is None:
sampled_seconds = 0
# Clamp sampled_seconds to wall_clock_seconds
sampled_seconds = min(sampled_seconds, wall_clock_seconds)
# --- Decision order per spec §5.1 ---
# 1. Powered-off: POH delta < 90% of wall-clock span
powered_off_threshold = wall_clock_seconds * POWERED_OFF_THRESHOLD_PERCENT
if poh_delta < powered_off_threshold:
return HourSplit(
seconds_active=0,
seconds_idle=0,
seconds_powered_off=wall_clock_seconds,
seconds_unknown=0,
)
# 2. Active: DUW delta >= 256 MiB
if duw_delta >= ACTIVE_THRESHOLD_BYTES:
return HourSplit(
seconds_active=wall_clock_seconds,
seconds_idle=0,
seconds_powered_off=0,
seconds_unknown=0,
)
# 3. Idle: powered on, sampled, below active threshold
# Unsampled portion within the hour is unknown
if sampled_seconds > 0:
return HourSplit(
seconds_active=0,
seconds_idle=sampled_seconds,
seconds_powered_off=0,
seconds_unknown=wall_clock_seconds - sampled_seconds,
)
# 4. Unknown: unsampled without POH evidence
return HourSplit(
seconds_active=0,
seconds_idle=0,
seconds_powered_off=0,
seconds_unknown=wall_clock_seconds,
)
+124
View File
@@ -0,0 +1,124 @@
"""Monitoring period bookkeeping per spec §5.2, §8.6, §9.8.
A monitoring period is a span during which Fenris monitoring is enabled.
Powered-off time stays inside a period; deliberately disabled time does not.
Key contracts:
- Run finding no open period opens one at the run moment, never backdated (§9.8)
- Wall-clock outside periods excluded from numerator and denominator (§5.2)
- End causes: user_disabled, migrated, unknown_gap
"""
import sqlite3
from datetime import datetime
def ensure_period_open(conn: sqlite3.Connection, run_time: datetime) -> None:
"""Ensure a monitoring period is open. If none exists, open one at run_time.
Spec §9.8: A collection run finding no open monitoring period opens one
at the run moment, never backdated.
"""
if get_open_period(conn) is not None:
return # Already open — no-op
ts = run_time.isoformat()
conn.execute(
"INSERT INTO monitoring_periods (started_at) VALUES (?)",
(ts,),
)
conn.commit()
def close_period(
conn: sqlite3.Connection,
closed_at: datetime,
end_cause: str,
) -> None:
"""Close the current open monitoring period.
Spec §8.6: Pause with an open period closes it user_disabled.
If no period is open, this is a no-op (pause otherwise).
"""
open_period = get_open_period(conn)
if open_period is None:
return # No-op
ts = closed_at.isoformat()
conn.execute(
"UPDATE monitoring_periods SET ended_at = ?, end_cause = ? WHERE id = ?",
(ts, end_cause, open_period["id"]),
)
conn.commit()
def get_open_period(conn: sqlite3.Connection) -> dict | None:
"""Return the currently open monitoring period, or None."""
cursor = conn.execute(
"SELECT id, started_at, ended_at, end_cause "
"FROM monitoring_periods WHERE ended_at IS NULL LIMIT 1"
)
row = cursor.fetchone()
if row is None:
return None
return {
"id": row[0],
"started_at": row[1],
"ended_at": row[2],
"end_cause": row[3],
}
def is_inside_period(conn: sqlite3.Connection, ts: datetime) -> bool:
"""Check if a timestamp falls inside any monitoring period.
Spec §5.2: Wall-clock outside periods is excluded from numerator/denominator.
"""
ts_str = ts.isoformat()
cursor = conn.execute(
"SELECT 1 FROM monitoring_periods "
"WHERE started_at <= ? AND (ended_at IS NULL OR ended_at > ?) "
"LIMIT 1",
(ts_str, ts_str),
)
return cursor.fetchone() is not None
def wall_clock_in_periods(
conn: sqlite3.Connection,
start: datetime,
end: datetime,
) -> int:
"""Compute total wall-clock seconds between start and end that fall inside
any monitoring period.
Used for denominator computation (§5.2).
"""
start_str = start.isoformat()
end_str = end.isoformat()
cursor = conn.execute(
"SELECT started_at, ended_at FROM monitoring_periods "
"WHERE ended_at IS NULL OR ended_at > ? "
"ORDER BY started_at",
(start_str,),
)
total = 0
for row in cursor.fetchall():
period_start = row[0]
period_end = row[1] # None if open
# Clip period to [start, end]
effective_start = max(period_start, start_str)
if period_end is not None:
effective_end = min(period_end, end_str)
else:
effective_end = end_str
if effective_start < effective_end:
# Parse for arithmetic
s = datetime.fromisoformat(effective_start)
e = datetime.fromisoformat(effective_end)
total += int((e - s).total_seconds())
return total
+31
View File
@@ -0,0 +1,31 @@
"""Raw sample pruning per spec §3.4, ST-5.
Raw samples are pruned opportunistically to 14 days.
Hour observations and day aggregates are retained indefinitely.
"""
import sqlite3
from datetime import datetime, timedelta, timezone
# Spec §3.4: Raw-sample retention
RAW_SAMPLE_RETENTION_DAYS = 14
def prune_old_samples(
conn: sqlite3.Connection,
now: datetime,
retention_days: int = RAW_SAMPLE_RETENTION_DAYS,
) -> int:
"""Remove raw samples older than retention_days.
Args:
conn: Connection to the observation store.
now: Current UTC time.
retention_days: Number of days to retain (default 14).
Returns:
Number of samples removed.
"""
cutoff = (now - timedelta(days=retention_days)).isoformat()
cursor = conn.execute("DELETE FROM samples WHERE ts < ?", (cutoff,))
conn.commit()
return cursor.rowcount