"""Projection core: the pure-function read path (spec §6). Recomputes the complete projection contract on every read, never stores anything derived. Takes a read-only observation store connection and an injected clock; returns a ProjectionResult with confidence state, contributing facts, headline remaining time (when one exists), scenario range, Percentage-Used context line, and disclosure text. Baseline precedence (§6.1): verified override → unverified override → implied → unavailable Confidence rule table (§6.7): Supported — all conjuncts satisfied Limited — baseline + positive rate, failing facts shown Unavailable — no applicable baseline / zero rate / identity change Arithmetic (§6.3): rate = regime DUW bytes / in-period wall-clock seconds projected = max(E_baseline − W_t, 0) / rate (rate > 0) E_rated = entered_TBW × 10¹² bytes E_implied = 100 · W_t / p (1 ≤ p ≤ 254) Criteria: PR-1–PR-17, CI-4. """ import sqlite3 from dataclasses import dataclass, field from datetime import datetime, timedelta, timezone from enum import Enum from typing import Any, Dict, List, Optional, Tuple # --------------------------------------------------------------------------- # Constants (spec §6) # --------------------------------------------------------------------------- HORIZON_DAYS = (7, 28, 90) TBW_TO_BYTES = 10 ** 12 IMPLIED_P_MIN = 1 IMPLIED_P_MAX = 254 IMPLIED_MIN_PU_INCREMENTS = 2 WARMING_MIN_DAYS = 14 WARMING_MAX_LOW_COVERAGE = 2 WARMING_COVERAGE_FLOOR = 0.50 SUPPORTED_COVERAGE_FLOOR = 0.80 HORIZON_AGREEMENT_FACTOR = 2 BURST_GUARD_FRACTION = 0.50 BURST_GUARD_LOOKBACK = 28 YOUNG_REGIME_DAYS = 7 HABIT_CHANGE_SHORT_WINDOW = 7 HABIT_CHANGE_LONG_WINDOW = 28 HABIT_CHANGE_UPPER_FACTOR = 2 HABIT_CHANGE_LOWER_FACTOR = 0.5 HABIT_CHANGE_CONSECUTIVE_DAYS = 3 STALENESS_HOURS = 48 WEAR_DISAGREEMENT_FACTOR = 2 class ConfidenceState(Enum): UNSUPPORTED = "Unavailable" LIMITED = "Limited" SUPPORTED = "Supported" class BaselineTier(Enum): VERIFIED = "verified_override" UNVERIFIED = "unverified_override" IMPLIED = "implied" NONE = "none" @dataclass(frozen=True) class ScenarioRange: rates: Dict[int, float] min_days: int max_days: int horizon_reasons: Dict[int, str] = field(default_factory=dict) @dataclass(frozen=True) class ProjectionResult: confidence_state: ConfidenceState contributing_facts: List[str] headline_remaining_seconds: Optional[float] scenario_range: Optional[ScenarioRange] pu_context_line: str disclosure_text: List[str] baseline_tier: BaselineTier baseline_label: str regime_days: Optional[int] habit_change_fact: Optional[str] warming_fact: Optional[str] staleness_fact: Optional[str] degraded_identity_fact: Optional[str] zero_rate_fact: Optional[str] # --------------------------------------------------------------------------- # Store queries # --------------------------------------------------------------------------- def _get_baseline(conn: sqlite3.Connection) -> Optional[Dict[str, Any]]: cursor = conn.execute( "SELECT id, tbw_terabytes, source_url, document_revision, entry_date, " " model_string, nominal_capacity_bytes, validated_by, verified " "FROM endurance_baseline LIMIT 1" ) row = cursor.fetchone() if row is None: return None return { "id": row[0], "tbw_terabytes": row[1], "source_url": row[2], "document_revision": row[3], "entry_date": row[4], "model_string": row[5], "nominal_capacity_bytes": row[6], "validated_by": row[7], "verified": bool(row[8]), } def _get_current_segment(conn: sqlite3.Connection) -> Optional[Dict[str, Any]]: cursor = conn.execute( "SELECT id, opened_at, identity_key, identity_degraded, mn " "FROM controller_segments ORDER BY id DESC LIMIT 1" ) row = cursor.fetchone() if row is None: return None return { "id": row[0], "opened_at": row[1], "identity_key": row[2], "identity_degraded": bool(row[3]), "mn": row[4], } def _get_days_in_segment(conn, segment_opened_at): cursor = conn.execute( "SELECT day, bytes_written_delta, coverage, sample_count " "FROM day_aggregates WHERE day >= ? ORDER BY day", (segment_opened_at[:10],), ) return [{"day": r[0], "bytes_written": r[1], "coverage": r[2], "sample_count": r[3]} for r in cursor.fetchall()] def _get_all_days(conn): cursor = conn.execute( "SELECT day, bytes_written_delta, coverage, sample_count " "FROM day_aggregates ORDER BY day" ) return [{"day": r[0], "bytes_written": r[1], "coverage": r[2], "sample_count": r[3]} for r in cursor.fetchall()] def _get_latest_pu(conn): cursor = conn.execute("SELECT percentage_used FROM samples ORDER BY id DESC LIMIT 1") row = cursor.fetchone() return row[0] if row else None def _get_pu_increments_in_segment(conn, segment_opened_at): cursor = conn.execute( "SELECT COUNT(DISTINCT percentage_used) FROM samples WHERE ts >= ?", (segment_opened_at,), ) row = cursor.fetchone() return max(0, (row[0] if row else 0) - 1) def _wall_clock_in_range(conn, start, end): start_str = start.isoformat() end_str = end.isoformat() cursor = conn.execute( "SELECT started_at, ended_at FROM monitoring_periods " "WHERE (ended_at IS NULL OR ended_at > ?) AND started_at < ? " "ORDER BY started_at", (start_str, end_str), ) total = 0 for row in cursor.fetchall(): eff_start = max(row[0], start_str) eff_end = min(row[1], end_str) if row[1] is not None else end_str if eff_start < eff_end: total += int((datetime.fromisoformat(eff_end) - datetime.fromisoformat(eff_start)).total_seconds()) return total # --------------------------------------------------------------------------- # Baseline resolution (§6.1, §6.2) # --------------------------------------------------------------------------- def _resolve_baseline(conn, current_segment): baseline = _get_baseline(conn) facts = [] if baseline is None: return BaselineTier.NONE, None, "no baseline", facts mandatory = [baseline["source_url"], baseline["document_revision"], baseline["entry_date"], baseline["model_string"], baseline["nominal_capacity_bytes"]] provenance_complete = all(f is not None and f != "" for f in mandatory) model_matches = True if current_segment is not None and baseline["model_string"] is not None: seg_mn = (current_segment.get("mn") or "").lower() bl_model = (baseline["model_string"] or "").lower() model_matches = bl_model in seg_mn or seg_mn in bl_model if provenance_complete and model_matches and baseline["verified"]: label = "verified manufacturer TBW (%.1f TB)" % baseline["tbw_terabytes"] return BaselineTier.VERIFIED, baseline, label, facts if not model_matches: facts.append( "baseline model '%s' does not match current drive '%s'" " — baseline retained but not applicable" % (baseline.get("model_string", ""), current_segment.get("mn", "") if current_segment else "") ) return BaselineTier.NONE, baseline, "baseline model mismatch", facts if not provenance_complete: label = "unverified TBW (%.1f TB) — user-supplied" % baseline["tbw_terabytes"] return BaselineTier.UNVERIFIED, baseline, label, facts label = "verified manufacturer TBW (%.1f TB)" % baseline["tbw_terabytes"] return BaselineTier.VERIFIED, baseline, label, facts # --------------------------------------------------------------------------- # Rate computation # --------------------------------------------------------------------------- def _compute_regime_rate(days, conn, regime_start_day, clock_now): regime_bytes = sum(d["bytes_written"] for d in days if d["day"] >= regime_start_day) regime_start_dt = datetime.fromisoformat(regime_start_day + "T00:00:00+00:00") regime_wc = _wall_clock_in_range(conn, regime_start_dt, clock_now) if regime_wc <= 0: return None, regime_bytes, 0 return regime_bytes / regime_wc, regime_bytes, regime_wc def _compute_horizon_rate(days, conn, horizon_days, evidence_endpoint): """Compute horizon rate anchored at the latest evidence endpoint T. Uses exact trailing horizon_days × 86400 seconds from T, not clock_now. Reader refresh alone never moves T or dilutes rates. Returns (rate, reason) where reason is None on success or a string describing why the rate is unavailable. """ if not days: return None, "no observation history" # T is the latest evidence endpoint — the end of the last day aggregate T = datetime.fromisoformat(days[-1]["day"] + "T23:59:59+00:00") # Exact trailing start: T minus horizon_days × 86400 seconds h_start = T - timedelta(days=horizon_days) cutoff = h_start.strftime("%Y-%m-%d") # History must span the full horizon — no placeholders if days[0]["day"] > cutoff: return None, "%d-day window starts before earliest data" % horizon_days h_bytes = sum(d["bytes_written"] for d in days if d["day"] >= cutoff) covered = sum(1 for d in days if d["day"] >= cutoff) if covered == 0: return None, "%d-day window has no data" % horizon_days h_wc = _wall_clock_in_range(conn, h_start, T) if h_wc <= 0: return None, "%d-day window has no monitored wall-clock time" % horizon_days return h_bytes / h_wc, None # --------------------------------------------------------------------------- # Habit change detection (§6.4) # --------------------------------------------------------------------------- def _detect_habit_change(days): """Detect habit change per spec §6.4. Trailing 7-day mean >= 2x (or <= 0.5x) the preceding 28-day mean for 3 consecutive days. Returns (change_day, days_since) or None. The first divergence day is the earliest day in the consecutive run. """ need = HABIT_CHANGE_SHORT_WINDOW + HABIT_CHANGE_LONG_WINDOW if len(days) < need: return None def _ratio_at(end_idx): """Compute 7-day / preceding-28-day mean ratio ending at end_idx.""" if end_idx < HABIT_CHANGE_SHORT_WINDOW - 1: return None se = end_idx + 1 ss = se - HABIT_CHANGE_SHORT_WINDOW s_bytes = sum(d["bytes_written"] for d in days[ss:se]) s_mean = s_bytes / HABIT_CHANGE_SHORT_WINDOW le = ss ls = le - HABIT_CHANGE_LONG_WINDOW if ls < 0: return None l_bytes = sum(d["bytes_written"] for d in days[ls:le]) l_mean = l_bytes / HABIT_CHANGE_LONG_WINDOW if l_mean == 0: return None return s_mean / l_mean # Scan backwards from the most recent day for i in range(len(days) - 1, HABIT_CHANGE_LONG_WINDOW + HABIT_CHANGE_SHORT_WINDOW - 2, -1): ratio = _ratio_at(i) if ratio is None: continue is_upper = ratio >= HABIT_CHANGE_UPPER_FACTOR is_lower = ratio <= HABIT_CHANGE_LOWER_FACTOR if not (is_upper or is_lower): continue # Count consecutive days going backwards from i consecutive = 1 for j in range(i - 1, HABIT_CHANGE_LONG_WINDOW + HABIT_CHANGE_SHORT_WINDOW - 3, -1): r = _ratio_at(j) if r is None: break if (is_upper and r >= HABIT_CHANGE_UPPER_FACTOR) or \ (is_lower and r <= HABIT_CHANGE_LOWER_FACTOR): consecutive += 1 else: break if consecutive >= HABIT_CHANGE_CONSECUTIVE_DAYS: change_idx = i - consecutive + 1 change_day = days[change_idx]["day"] days_since = (datetime.fromisoformat(days[-1]["day"]) - datetime.fromisoformat(change_day)).days return change_day, days_since return None # --------------------------------------------------------------------------- # Confidence rule table (§6.7) # --------------------------------------------------------------------------- def _evaluate_confidence(tier, rate, regime_days, days, current_segment, clock_now, warming_days, warming_low_coverage, habit_change, staleness_hours, scenario_range): facts = [] if tier == BaselineTier.NONE: facts.append("no applicable endurance baseline") return ConfidenceState.UNSUPPORTED, facts if rate is None or rate <= 0: facts.append("no finite projection from this history") return ConfidenceState.UNSUPPORTED, facts supported_facts = [] failing = False # 1. Verified baseline if tier != BaselineTier.VERIFIED: failing = True else: supported_facts.append("verified manufacturer TBW") # 2. >= 14 qualifying days qualifying = sum(1 for d in days if d["coverage"] >= WARMING_COVERAGE_FLOOR and d["sample_count"] > 0) if qualifying < WARMING_MIN_DAYS: failing = True else: supported_facts.append("%d calendar days" % qualifying) # 3. Coverage >= 80% total_wc = len(days) * 86400 total_known = sum(int(d["coverage"] * 86400) for d in days) avg_cov = total_known / total_wc if total_wc > 0 else 0.0 if avg_cov < SUPPORTED_COVERAGE_FLOOR: failing = True else: supported_facts.append("%d%% interval coverage" % int(avg_cov * 100)) # 4. Fresh (< 48h) if staleness_hours is not None and staleness_hours > STALENESS_HOURS: failing = True elif staleness_hours is not None: supported_facts.append("recent data") # 5. Horizon agreement if scenario_range is not None and len(scenario_range.rates) >= 2: rl = list(scenario_range.rates.values()) if min(rl) > 0 and max(rl) / min(rl) > HORIZON_AGREEMENT_FACTOR: failing = True else: supported_facts.append("%d weekly cycles" % len(scenario_range.rates)) else: failing = True # 6. Burst guard if not failing and len(days) >= BURST_GUARD_LOOKBACK: t28 = sum(d["bytes_written"] for d in days[-BURST_GUARD_LOOKBACK:]) for d in days[-BURST_GUARD_LOOKBACK:]: if t28 > 0 and d["bytes_written"] >= BURST_GUARD_FRACTION * t28: failing = True break if not failing: supported_facts.append("no burst days") # 7. Regime >= 7 days if regime_days < YOUNG_REGIME_DAYS: failing = True # 8. Degraded identity if current_segment and current_segment.get("identity_degraded"): failing = True facts.append("controller identity unavailable — replacement detection relies on write-counter continuity only") if not failing: return ConfidenceState.SUPPORTED, supported_facts # Limited limited_facts = list(supported_facts) if staleness_hours is not None and staleness_hours > STALENESS_HOURS: limited_facts.append("newest data %dh old (≥48h)" % staleness_hours) if habit_change is not None: limited_facts.append("usage habit changed %d days ago" % habit_change[1]) if regime_days < YOUNG_REGIME_DAYS: limited_facts.append("regime only %d days old (≥7 required)" % regime_days) if current_segment and current_segment.get("identity_degraded"): degraded_fact = "controller identity unavailable — replacement detection relies on write-counter continuity only" if degraded_fact not in limited_facts: limited_facts.append(degraded_fact) return ConfidenceState.LIMITED, limited_facts # --------------------------------------------------------------------------- # Disclosure text (§6.11) # --------------------------------------------------------------------------- DISCLOSURES = [ "This is an endurance projection, not a predicted hardware-failure date.", ("Percentage Used is vendor-specific; 100 means estimated endurance consumed " "but may not mean failure, it can exceed 100, and 255 is saturated."), ("Rated TBW can be a warranty/endurance threshold with separate time and " "eligibility terms, not a failure threshold."), ("DUW is upward-rounded host writes excluding metadata and selected commands, " "not exact physical NAND writes."), ("Projection quality depends on baseline provenance, history duration and " "completeness, recentness, stability, and representative usage cycles; " "future workload and firmware behavior remain outside the observed evidence."), ("Gaps can preserve an aggregate counter delta without preserving hourly " "timing; unexplained and deliberately disabled periods must be distinguished."), ] # --------------------------------------------------------------------------- # PU context line (§6.1) # --------------------------------------------------------------------------- def _build_pu_context_line(conn, rate, days, clock_now): pu = _get_latest_pu(conn) if pu is None: return "Percentage Used: unknown" if rate is None or rate <= 0 or not days: return "Percentage Used: %d%%" % pu total_bytes = sum(d["bytes_written"] for d in days) if total_bytes <= 0: return "Percentage Used: %d%%" % pu total_days_count = len(days) if total_days_count == 0: return "Percentage Used: %d%%" % pu pu_daily = total_bytes / total_days_count obs_daily = rate * 86400 if pu_daily > 0: ratio = obs_daily / pu_daily if ratio > WEAR_DISAGREEMENT_FACTOR or ratio < 1.0 / WEAR_DISAGREEMENT_FACTOR: return ("Percentage Used: %d%% — vendor wear estimate disagrees " "with observed write rate (>2× difference)") % pu return "Percentage Used: %d%%" % pu # --------------------------------------------------------------------------- # Main projection function # --------------------------------------------------------------------------- def compute_projection(conn, clock_now): facts = [] habit_change_fact = None warming_fact = None staleness_fact = None degraded_identity_fact = None zero_rate_fact = None current_segment = _get_current_segment(conn) tier, baseline, baseline_label, baseline_facts = _resolve_baseline(conn, current_segment) facts.extend(baseline_facts) segment_days = _get_days_in_segment(conn, current_segment["opened_at"]) if current_segment else _get_all_days(conn) all_days = _get_all_days(conn) regime_start_day = None habit_change = None if segment_days: earliest = segment_days[0]["day"] cutoff_90 = (clock_now - timedelta(days=90)).strftime("%Y-%m-%d") regime_start_day = max(earliest, cutoff_90) habit_change = _detect_habit_change(segment_days) if habit_change is not None: regime_start_day = habit_change[0] habit_change_fact = "usage habit changed %d days ago" % habit_change[1] facts.append(habit_change_fact) rate = None regime_bytes = 0 regime_days_count = 0 if segment_days and regime_start_day is not None: rate, regime_bytes, _ = _compute_regime_rate(segment_days, conn, regime_start_day, clock_now) regime_days_count = sum(1 for d in segment_days if d["day"] >= regime_start_day) if rate is not None and rate <= 0: zero_rate_fact = "no finite projection from this history" facts.append(zero_rate_fact) scenario = None horizon_rates = {} horizon_reasons = {} for h in HORIZON_DAYS: hr, reason = _compute_horizon_rate(all_days, conn, h, clock_now) if hr is not None: horizon_rates[h] = hr else: horizon_reasons[h] = reason if horizon_rates: scenario = ScenarioRange( rates=horizon_rates, min_days=min(horizon_rates), max_days=max(horizon_rates), horizon_reasons=horizon_reasons, ) total_days_count = len(segment_days) days_below_coverage = sum(1 for d in segment_days if d["coverage"] < WARMING_COVERAGE_FLOOR or d["sample_count"] == 0) qualifying = total_days_count - days_below_coverage if total_days_count < WARMING_MIN_DAYS or days_below_coverage > WARMING_MAX_LOW_COVERAGE: warming_fact = "warming up: %d of %d qualifying days" % (qualifying, WARMING_MIN_DAYS) facts.append(warming_fact) staleness_hours = None if segment_days: newest_dt = datetime.fromisoformat(segment_days[-1]["day"] + "T12:00:00+00:00") staleness_hours = int((clock_now - newest_dt).total_seconds() / 3600) if staleness_hours > STALENESS_HOURS: staleness_fact = "newest data %dh old (≥48h)" % staleness_hours facts.append(staleness_fact) if current_segment and current_segment.get("identity_degraded"): degraded_identity_fact = "controller identity unavailable — replacement detection relies on write-counter continuity only" facts.append(degraded_identity_fact) state, conf_facts = _evaluate_confidence( tier, rate, regime_days_count, segment_days, current_segment, clock_now, qualifying, 0, habit_change, staleness_hours, scenario, ) all_facts = list(facts) for cf in conf_facts: if cf not in all_facts: all_facts.append(cf) headline_seconds = None if state != ConfidenceState.UNSUPPORTED and rate is not None and rate > 0 and baseline is not None: if tier in (BaselineTier.VERIFIED, BaselineTier.UNVERIFIED): E_baseline = baseline["tbw_terabytes"] * TBW_TO_BYTES elif tier == BaselineTier.IMPLIED: p = _get_latest_pu(conn) if p is not None and IMPLIED_P_MIN <= p <= IMPLIED_P_MAX: E_baseline = 100 * regime_bytes / p else: E_baseline = None else: E_baseline = None if E_baseline is not None: headline_seconds = max(E_baseline - regime_bytes, 0) / rate pu_line = _build_pu_context_line(conn, rate, segment_days, clock_now) return ProjectionResult( confidence_state=state, contributing_facts=all_facts, headline_remaining_seconds=headline_seconds, scenario_range=scenario, pu_context_line=pu_line, disclosure_text=list(DISCLOSURES), baseline_tier=tier, baseline_label=baseline_label, regime_days=regime_days_count, habit_change_fact=habit_change_fact, warming_fact=warming_fact, staleness_fact=staleness_fact, degraded_identity_fact=degraded_identity_fact, zero_rate_fact=zero_rate_fact, )