Files
Fenris/src/fenris/projection.py
T

669 lines
25 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Projection core: the pure-function read path (spec §6).
Recomputes the complete projection contract on every read, never stores
anything derived. Takes a read-only observation store connection and an
injected clock; returns a ProjectionResult with confidence state,
contributing facts, headline remaining time (when one exists), scenario
range, Percentage-Used context line, and disclosure text.
Baseline precedence (§6.1):
verified override → unverified override → implied → unavailable
Confidence rule table (§6.7):
Supported — all conjuncts satisfied
Limited — baseline + positive rate, failing facts shown
Unavailable — no applicable baseline / zero rate / identity change
Arithmetic (§6.3):
rate = regime DUW bytes / in-period wall-clock seconds
projected = max(E_baseline − W_t, 0) / rate (rate > 0)
E_rated = entered_TBW × 10¹² bytes
E_implied = 100 · W_t / p (1 ≤ p ≤ 254)
Criteria: PR-1–PR-17, CI-4.
"""
import sqlite3
from dataclasses import dataclass, field
from datetime import datetime, timedelta, timezone
from enum import Enum
from typing import Any, Dict, List, Optional, Tuple
from .monitoring_periods import interval_within_one_monitoring_period
# ---------------------------------------------------------------------------
# Constants (spec §6)
# ---------------------------------------------------------------------------
HORIZON_DAYS = (7, 28, 90)
TBW_TO_BYTES = 10 ** 12
IMPLIED_P_MIN = 1
IMPLIED_P_MAX = 254
IMPLIED_MIN_PU_INCREMENTS = 2
WARMING_MIN_DAYS = 14
WARMING_MAX_LOW_COVERAGE = 2
WARMING_COVERAGE_FLOOR = 0.50
SUPPORTED_COVERAGE_FLOOR = 0.80
HORIZON_AGREEMENT_FACTOR = 2
BURST_GUARD_FRACTION = 0.50
BURST_GUARD_LOOKBACK = 28
YOUNG_REGIME_DAYS = 7
HABIT_CHANGE_SHORT_WINDOW = 7
HABIT_CHANGE_LONG_WINDOW = 28
HABIT_CHANGE_UPPER_FACTOR = 2
HABIT_CHANGE_LOWER_FACTOR = 0.5
HABIT_CHANGE_CONSECUTIVE_DAYS = 3
STALENESS_HOURS = 48
WEAR_DISAGREEMENT_FACTOR = 2
class ConfidenceState(Enum):
UNSUPPORTED = "Unavailable"
LIMITED = "Limited"
SUPPORTED = "Supported"
class BaselineTier(Enum):
VERIFIED = "verified_override"
UNVERIFIED = "unverified_override"
IMPLIED = "implied"
NONE = "none"
@dataclass(frozen=True)
class ScenarioRange:
rates: Dict[int, float]
min_days: int
max_days: int
horizon_reasons: Dict[int, str] = field(default_factory=dict)
@dataclass(frozen=True)
class ProjectionResult:
confidence_state: ConfidenceState
contributing_facts: List[str]
headline_remaining_seconds: Optional[float]
scenario_range: Optional[ScenarioRange]
pu_context_line: str
disclosure_text: List[str]
baseline_tier: BaselineTier
baseline_label: str
regime_days: Optional[int]
habit_change_fact: Optional[str]
warming_fact: Optional[str]
staleness_fact: Optional[str]
degraded_identity_fact: Optional[str]
zero_rate_fact: Optional[str]
qualifying_days_progress: Optional[str] = None # Issue #77: honest qualifying-day progress
# ---------------------------------------------------------------------------
# Store queries
# ---------------------------------------------------------------------------
def _get_baseline(conn: sqlite3.Connection) -> Optional[Dict[str, Any]]:
cursor = conn.execute(
"SELECT id, tbw_terabytes, source_url, document_revision, entry_date, "
" model_string, nominal_capacity_bytes, validated_by, verified "
"FROM endurance_baseline LIMIT 1"
)
row = cursor.fetchone()
if row is None:
return None
return {
"id": row[0], "tbw_terabytes": row[1], "source_url": row[2],
"document_revision": row[3], "entry_date": row[4],
"model_string": row[5], "nominal_capacity_bytes": row[6],
"validated_by": row[7], "verified": bool(row[8]),
}
def _get_current_segment(conn: sqlite3.Connection) -> Optional[Dict[str, Any]]:
cursor = conn.execute(
"SELECT id, opened_at, identity_key, identity_degraded, mn "
"FROM controller_segments ORDER BY id DESC LIMIT 1"
)
row = cursor.fetchone()
if row is None:
return None
return {
"id": row[0], "opened_at": row[1], "identity_key": row[2],
"identity_degraded": bool(row[3]), "mn": row[4],
}
def _get_days_in_segment(conn, segment_opened_at):
cursor = conn.execute(
"SELECT day, bytes_written_delta, coverage, sample_count "
"FROM day_aggregates WHERE day >= ? ORDER BY day",
(segment_opened_at[:10],),
)
return [{"day": r[0], "bytes_written": r[1], "coverage": r[2], "sample_count": r[3]}
for r in cursor.fetchall()]
def _get_all_days(conn):
cursor = conn.execute(
"SELECT day, bytes_written_delta, coverage, sample_count "
"FROM day_aggregates ORDER BY day"
)
return [{"day": r[0], "bytes_written": r[1], "coverage": r[2], "sample_count": r[3]}
for r in cursor.fetchall()]
def _get_latest_pu(conn):
cursor = conn.execute("SELECT percentage_used FROM samples ORDER BY id DESC LIMIT 1")
row = cursor.fetchone()
return row[0] if row else None
def _get_pu_increments_in_segment(conn, segment_opened_at):
cursor = conn.execute(
"SELECT COUNT(DISTINCT percentage_used) FROM samples WHERE ts >= ?",
(segment_opened_at,),
)
row = cursor.fetchone()
return max(0, (row[0] if row else 0) - 1)
def _wall_clock_in_range(conn, start, end):
start_str = start.isoformat()
end_str = end.isoformat()
cursor = conn.execute(
"SELECT started_at, ended_at FROM monitoring_periods "
"WHERE (ended_at IS NULL OR ended_at > ?) AND started_at < ? "
"ORDER BY started_at", (start_str, end_str),
)
total = 0
for row in cursor.fetchall():
eff_start = max(row[0], start_str)
eff_end = min(row[1], end_str) if row[1] is not None else end_str
if eff_start < eff_end:
total += int((datetime.fromisoformat(eff_end) - datetime.fromisoformat(eff_start)).total_seconds())
return total
# ---------------------------------------------------------------------------
# Baseline resolution (§6.1, §6.2)
# ---------------------------------------------------------------------------
def _resolve_baseline(conn, current_segment):
baseline = _get_baseline(conn)
facts: list[str] = []
if baseline is None:
return BaselineTier.NONE, None, "no baseline", facts
mandatory = [baseline["source_url"], baseline["document_revision"],
baseline["entry_date"], baseline["model_string"],
baseline["nominal_capacity_bytes"]]
provenance_complete = all(f is not None and f != "" for f in mandatory)
model_matches = True
if current_segment is not None and baseline["model_string"] is not None:
seg_mn = (current_segment.get("mn") or "").lower()
bl_model = (baseline["model_string"] or "").lower()
model_matches = bl_model in seg_mn or seg_mn in bl_model
if provenance_complete and model_matches and baseline["verified"]:
label = "verified manufacturer TBW (%.1f TB)" % baseline["tbw_terabytes"]
return BaselineTier.VERIFIED, baseline, label, facts
if not model_matches:
facts.append(
"baseline model '%s' does not match current drive '%s'"
" — baseline retained but not applicable"
% (baseline.get("model_string", ""),
current_segment.get("mn", "") if current_segment else "")
)
return BaselineTier.NONE, baseline, "baseline model mismatch", facts
if not provenance_complete:
label = "unverified TBW (%.1f TB) — user-supplied" % baseline["tbw_terabytes"]
return BaselineTier.UNVERIFIED, baseline, label, facts
label = "verified manufacturer TBW (%.1f TB)" % baseline["tbw_terabytes"]
return BaselineTier.VERIFIED, baseline, label, facts
# ---------------------------------------------------------------------------
# Rate computation
# ---------------------------------------------------------------------------
def _compute_regime_rate(days, conn, regime_start_day, clock_now):
regime_bytes = sum(d["bytes_written"] for d in days if d["day"] >= regime_start_day)
regime_start_dt = datetime.fromisoformat(regime_start_day + "T00:00:00+00:00")
regime_wc = _wall_clock_in_range(conn, regime_start_dt, clock_now)
if regime_wc <= 0:
return None, regime_bytes, 0
return regime_bytes / regime_wc, regime_bytes, regime_wc
def _compute_horizon_rate(days, conn, horizon_days, evidence_endpoint):
"""Compute horizon rate anchored at the latest evidence endpoint T.
Uses exact trailing horizon_days × 86400 seconds from T, not clock_now.
Reader refresh alone never moves T or dilutes rates.
Returns (rate, reason) where reason is None on success or a string
describing why the rate is unavailable.
"""
if not days:
return None, "no observation history"
# T is the latest evidence endpoint — the end of the last day aggregate
T = datetime.fromisoformat(days[-1]["day"] + "T23:59:59+00:00")
# Exact trailing start: T minus horizon_days × 86400 seconds
h_start = T - timedelta(days=horizon_days)
cutoff = h_start.strftime("%Y-%m-%d")
# History must span the full horizon — no placeholders
if days[0]["day"] > cutoff:
return None, "%d-day window starts before earliest data" % horizon_days
h_bytes = sum(d["bytes_written"] for d in days if d["day"] >= cutoff)
covered = sum(1 for d in days if d["day"] >= cutoff)
if covered == 0:
return None, "%d-day window has no data" % horizon_days
h_wc = _wall_clock_in_range(conn, h_start, T)
if h_wc <= 0:
return None, "%d-day window has no monitored wall-clock time" % horizon_days
return h_bytes / h_wc, None
# ---------------------------------------------------------------------------
# Habit change detection (§6.4)
# ---------------------------------------------------------------------------
def _detect_habit_change(days):
"""Detect habit change per spec §6.4.
Trailing 7-day mean >= 2x (or <= 0.5x) the preceding 28-day mean
for 3 consecutive days. Returns (change_day, days_since) or None.
The first divergence day is the earliest day in the consecutive run.
"""
need = HABIT_CHANGE_SHORT_WINDOW + HABIT_CHANGE_LONG_WINDOW
if len(days) < need:
return None
def _ratio_at(end_idx):
"""Compute 7-day / preceding-28-day mean ratio ending at end_idx."""
if end_idx < HABIT_CHANGE_SHORT_WINDOW - 1:
return None
se = end_idx + 1
ss = se - HABIT_CHANGE_SHORT_WINDOW
s_bytes = sum(d["bytes_written"] for d in days[ss:se])
s_mean = s_bytes / HABIT_CHANGE_SHORT_WINDOW
le = ss
ls = le - HABIT_CHANGE_LONG_WINDOW
if ls < 0:
return None
l_bytes = sum(d["bytes_written"] for d in days[ls:le])
l_mean = l_bytes / HABIT_CHANGE_LONG_WINDOW
if l_mean == 0:
return None
return s_mean / l_mean
# Scan backwards from the most recent day
for i in range(len(days) - 1, HABIT_CHANGE_LONG_WINDOW + HABIT_CHANGE_SHORT_WINDOW - 2, -1):
ratio = _ratio_at(i)
if ratio is None:
continue
is_upper = ratio >= HABIT_CHANGE_UPPER_FACTOR
is_lower = ratio <= HABIT_CHANGE_LOWER_FACTOR
if not (is_upper or is_lower):
continue
# Count consecutive days going backwards from i
consecutive = 1
for j in range(i - 1, HABIT_CHANGE_LONG_WINDOW + HABIT_CHANGE_SHORT_WINDOW - 3, -1):
r = _ratio_at(j)
if r is None:
break
if (is_upper and r >= HABIT_CHANGE_UPPER_FACTOR) or \
(is_lower and r <= HABIT_CHANGE_LOWER_FACTOR):
consecutive += 1
else:
break
if consecutive >= HABIT_CHANGE_CONSECUTIVE_DAYS:
change_idx = i - consecutive + 1
change_day = days[change_idx]["day"]
days_since = (datetime.fromisoformat(days[-1]["day"]) - datetime.fromisoformat(change_day)).days
return change_day, days_since
return None
# ---------------------------------------------------------------------------
# Confidence rule table (§6.7)
# ---------------------------------------------------------------------------
def _evaluate_confidence(tier, rate, regime_days, days, current_segment,
clock_now, warming_days, warming_low_coverage,
habit_change, staleness_hours, scenario_range):
facts = []
if tier == BaselineTier.NONE:
facts.append("no applicable endurance baseline")
return ConfidenceState.UNSUPPORTED, facts
if rate is None or rate <= 0:
facts.append("no finite projection from this history")
return ConfidenceState.UNSUPPORTED, facts
supported_facts = []
failing = False
# 1. Verified baseline
if tier != BaselineTier.VERIFIED:
failing = True
else:
supported_facts.append("verified manufacturer TBW")
# 2. >= 14 qualifying days
qualifying = sum(1 for d in days if d["coverage"] >= WARMING_COVERAGE_FLOOR and d["sample_count"] > 0)
if qualifying < WARMING_MIN_DAYS:
failing = True
else:
supported_facts.append("%d calendar days" % qualifying)
# 3. Coverage >= 80%
total_wc = len(days) * 86400
total_known = sum(int(d["coverage"] * 86400) for d in days)
avg_cov = total_known / total_wc if total_wc > 0 else 0.0
if avg_cov < SUPPORTED_COVERAGE_FLOOR:
failing = True
else:
supported_facts.append("%d%% interval coverage" % int(avg_cov * 100))
# 4. Fresh (< 48h)
if staleness_hours is not None and staleness_hours > STALENESS_HOURS:
failing = True
elif staleness_hours is not None:
supported_facts.append("recent data")
# 5. Horizon agreement
if scenario_range is not None and len(scenario_range.rates) >= 2:
rl = list(scenario_range.rates.values())
if min(rl) > 0 and max(rl) / min(rl) > HORIZON_AGREEMENT_FACTOR:
failing = True
else:
supported_facts.append("%d weekly cycles" % len(scenario_range.rates))
else:
failing = True
# 6. Burst guard
if not failing and len(days) >= BURST_GUARD_LOOKBACK:
t28 = sum(d["bytes_written"] for d in days[-BURST_GUARD_LOOKBACK:])
for d in days[-BURST_GUARD_LOOKBACK:]:
if t28 > 0 and d["bytes_written"] >= BURST_GUARD_FRACTION * t28:
failing = True
break
if not failing:
supported_facts.append("no burst days")
# 7. Regime >= 7 days
if regime_days < YOUNG_REGIME_DAYS:
failing = True
# 8. Degraded identity
if current_segment and current_segment.get("identity_degraded"):
failing = True
facts.append("controller identity unavailable — replacement detection relies on write-counter continuity only")
if not failing:
return ConfidenceState.SUPPORTED, supported_facts
# Limited
limited_facts = list(supported_facts)
if staleness_hours is not None and staleness_hours > STALENESS_HOURS:
limited_facts.append("newest data %dh old (≥48h)" % staleness_hours)
if habit_change is not None:
limited_facts.append("usage habit changed %d days ago" % habit_change[1])
if regime_days < YOUNG_REGIME_DAYS:
limited_facts.append("regime only %d days old (≥7 required)" % regime_days)
if current_segment and current_segment.get("identity_degraded"):
degraded_fact = "controller identity unavailable — replacement detection relies on write-counter continuity only"
if degraded_fact not in limited_facts:
limited_facts.append(degraded_fact)
return ConfidenceState.LIMITED, limited_facts
# ---------------------------------------------------------------------------
# Disclosure text (§6.11)
# ---------------------------------------------------------------------------
DISCLOSURES = [
"This is an endurance projection, not a predicted hardware-failure date.",
("Percentage Used is vendor-specific; 100 means estimated endurance consumed "
"but may not mean failure, it can exceed 100, and 255 is saturated."),
("Rated TBW can be a warranty/endurance threshold with separate time and "
"eligibility terms, not a failure threshold."),
("DUW is upward-rounded host writes excluding metadata and selected commands, "
"not exact physical NAND writes."),
("Projection quality depends on baseline provenance, history duration and "
"completeness, recentness, stability, and representative usage cycles; "
"future workload and firmware behavior remain outside the observed evidence."),
("Gaps can preserve an aggregate counter delta without preserving hourly "
"timing; unexplained and deliberately disabled periods must be distinguished."),
]
# ---------------------------------------------------------------------------
# PU context line (§6.1)
# ---------------------------------------------------------------------------
def _build_pu_context_line(conn, rate, days, clock_now):
pu = _get_latest_pu(conn)
if pu is None:
return "Percentage Used: unknown"
if rate is None or rate <= 0 or not days:
return "Percentage Used: %d%%" % pu
total_bytes = sum(d["bytes_written"] for d in days)
if total_bytes <= 0:
return "Percentage Used: %d%%" % pu
total_days_count = len(days)
if total_days_count == 0:
return "Percentage Used: %d%%" % pu
pu_daily = total_bytes / total_days_count
obs_daily = rate * 86400
if pu_daily > 0:
ratio = obs_daily / pu_daily
if ratio > WEAR_DISAGREEMENT_FACTOR or ratio < 1.0 / WEAR_DISAGREEMENT_FACTOR:
return ("Percentage Used: %d%% — vendor wear estimate disagrees "
"with observed write rate (>2× difference)") % pu
return "Percentage Used: %d%%" % pu
# ---------------------------------------------------------------------------
# Complete observation day gate (issue #94)
# ---------------------------------------------------------------------------
def _has_complete_local_day(conn):
"""Check if at least one complete local observation day exists.
A complete local day is a full midnight-to-midnight calendar day
within a monitoring period that has usable observation evidence.
This is the prerequisite for showing an endurance outlook.
"""
local_days = conn.execute(
"SELECT utc_start, utc_end FROM local_days "
"WHERE complete = 1 "
"AND activity_precision IN ('measured', 'coarse') "
"AND activity_intervals > 0 ORDER BY id"
).fetchall()
return any(
interval_within_one_monitoring_period(conn, utc_start, utc_end)
for utc_start, utc_end in local_days
)
# ---------------------------------------------------------------------------
# Main projection function
# ---------------------------------------------------------------------------
def compute_projection(conn, clock_now):
facts = []
habit_change_fact = None
warming_fact = None
staleness_fact = None
degraded_identity_fact = None
zero_rate_fact = None
current_segment = _get_current_segment(conn)
tier, baseline, baseline_label, baseline_facts = _resolve_baseline(conn, current_segment)
facts.extend(baseline_facts)
# --- Complete observation day gate (issue #94) ---
# An endurance outlook requires at least one complete local
# midnight-to-midnight calendar day with usable observation evidence.
has_complete_day = _has_complete_local_day(conn)
if not has_complete_day:
facts.append("waiting for a full local observation day")
return ProjectionResult(
confidence_state=ConfidenceState.UNSUPPORTED,
contributing_facts=facts,
headline_remaining_seconds=None,
scenario_range=None,
pu_context_line="Percentage Used: unknown" if _get_latest_pu(conn) is None else "Percentage Used: %d%%" % (_get_latest_pu(conn) or 0),
disclosure_text=list(DISCLOSURES),
baseline_tier=tier,
baseline_label=baseline_label,
regime_days=None,
habit_change_fact=None,
warming_fact=None,
staleness_fact=None,
degraded_identity_fact=None,
zero_rate_fact=None,
qualifying_days_progress=None,
)
segment_days = _get_days_in_segment(conn, current_segment["opened_at"]) if current_segment else _get_all_days(conn)
all_days = _get_all_days(conn)
regime_start_day = None
habit_change = None
if segment_days:
earliest = segment_days[0]["day"]
cutoff_90 = (clock_now - timedelta(days=90)).strftime("%Y-%m-%d")
regime_start_day = max(earliest, cutoff_90)
habit_change = _detect_habit_change(segment_days)
if habit_change is not None:
regime_start_day = habit_change[0]
habit_change_fact = "usage habit changed %d days ago" % habit_change[1]
facts.append(habit_change_fact)
rate = None
regime_bytes = 0
regime_days_count = 0
if segment_days and regime_start_day is not None:
rate, regime_bytes, _ = _compute_regime_rate(segment_days, conn, regime_start_day, clock_now)
regime_days_count = sum(1 for d in segment_days if d["day"] >= regime_start_day)
if rate is not None and rate <= 0:
zero_rate_fact = "no finite projection from this history"
facts.append(zero_rate_fact)
scenario = None
horizon_rates = {}
horizon_reasons = {}
for h in HORIZON_DAYS:
hr, reason = _compute_horizon_rate(all_days, conn, h, clock_now)
if hr is not None:
horizon_rates[h] = hr
else:
horizon_reasons[h] = reason
if horizon_rates:
scenario = ScenarioRange(
rates=horizon_rates,
min_days=min(horizon_rates),
max_days=max(horizon_rates),
horizon_reasons=horizon_reasons,
)
total_days_count = len(segment_days)
days_below_coverage = sum(1 for d in segment_days
if d["coverage"] < WARMING_COVERAGE_FLOOR or d["sample_count"] == 0)
qualifying = total_days_count - days_below_coverage
# Issue #77: Show honest qualifying-day progress
if total_days_count >= WARMING_MIN_DAYS:
# After warm-up, show qualifying day details for transparency
qualifying_days_progress = "%d of %d qualifying days" % (qualifying, total_days_count)
if days_below_coverage > 0:
qualifying_days_progress += " (%d below coverage)" % days_below_coverage
else:
qualifying_days_progress = None
if total_days_count < WARMING_MIN_DAYS or days_below_coverage > WARMING_MAX_LOW_COVERAGE:
warming_fact = "warming up: %d of %d qualifying days" % (qualifying, WARMING_MIN_DAYS)
facts.append(warming_fact)
staleness_hours = None
if segment_days:
newest_dt = datetime.fromisoformat(segment_days[-1]["day"] + "T12:00:00+00:00")
staleness_hours = int((clock_now - newest_dt).total_seconds() / 3600)
if staleness_hours > STALENESS_HOURS:
staleness_fact = "newest data %dh old (≥48h)" % staleness_hours
facts.append(staleness_fact)
if current_segment and current_segment.get("identity_degraded"):
degraded_identity_fact = "controller identity unavailable — replacement detection relies on write-counter continuity only"
facts.append(degraded_identity_fact)
state, conf_facts = _evaluate_confidence(
tier, rate, regime_days_count, segment_days, current_segment, clock_now,
qualifying, 0, habit_change, staleness_hours, scenario,
)
all_facts = list(facts)
for cf in conf_facts:
if cf not in all_facts:
all_facts.append(cf)
headline_seconds = None
if state != ConfidenceState.UNSUPPORTED and rate is not None and rate > 0 and baseline is not None:
if tier in (BaselineTier.VERIFIED, BaselineTier.UNVERIFIED):
E_baseline = baseline["tbw_terabytes"] * TBW_TO_BYTES
elif tier == BaselineTier.IMPLIED:
p = _get_latest_pu(conn)
if p is not None and IMPLIED_P_MIN <= p <= IMPLIED_P_MAX:
E_baseline = 100 * regime_bytes / p
else:
E_baseline = None
else:
E_baseline = None
if E_baseline is not None:
headline_seconds = max(E_baseline - regime_bytes, 0) / rate
pu_line = _build_pu_context_line(conn, rate, segment_days, clock_now)
return ProjectionResult(
confidence_state=state,
contributing_facts=all_facts,
headline_remaining_seconds=headline_seconds,
scenario_range=scenario,
pu_context_line=pu_line,
disclosure_text=list(DISCLOSURES),
baseline_tier=tier,
baseline_label=baseline_label,
regime_days=regime_days_count,
habit_change_fact=habit_change_fact,
warming_fact=warming_fact,
staleness_fact=staleness_fact,
degraded_identity_fact=degraded_identity_fact,
zero_rate_fact=zero_rate_fact,
qualifying_days_progress=qualifying_days_progress,
)