669 lines
25 KiB
Python
669 lines
25 KiB
Python
"""Projection core: the pure-function read path (spec §6).
|
||
|
||
Recomputes the complete projection contract on every read, never stores
|
||
anything derived. Takes a read-only observation store connection and an
|
||
injected clock; returns a ProjectionResult with confidence state,
|
||
contributing facts, headline remaining time (when one exists), scenario
|
||
range, Percentage-Used context line, and disclosure text.
|
||
|
||
Baseline precedence (§6.1):
|
||
verified override → unverified override → implied → unavailable
|
||
|
||
Confidence rule table (§6.7):
|
||
Supported — all conjuncts satisfied
|
||
Limited — baseline + positive rate, failing facts shown
|
||
Unavailable — no applicable baseline / zero rate / identity change
|
||
|
||
Arithmetic (§6.3):
|
||
rate = regime DUW bytes / in-period wall-clock seconds
|
||
projected = max(E_baseline − W_t, 0) / rate (rate > 0)
|
||
E_rated = entered_TBW × 10¹² bytes
|
||
E_implied = 100 · W_t / p (1 ≤ p ≤ 254)
|
||
|
||
Criteria: PR-1–PR-17, CI-4.
|
||
"""
|
||
import sqlite3
|
||
from dataclasses import dataclass, field
|
||
from datetime import datetime, timedelta, timezone
|
||
from enum import Enum
|
||
from typing import Any, Dict, List, Optional, Tuple
|
||
|
||
from .monitoring_periods import interval_within_one_monitoring_period
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Constants (spec §6)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
HORIZON_DAYS = (7, 28, 90)
|
||
TBW_TO_BYTES = 10 ** 12
|
||
IMPLIED_P_MIN = 1
|
||
IMPLIED_P_MAX = 254
|
||
IMPLIED_MIN_PU_INCREMENTS = 2
|
||
WARMING_MIN_DAYS = 14
|
||
WARMING_MAX_LOW_COVERAGE = 2
|
||
WARMING_COVERAGE_FLOOR = 0.50
|
||
SUPPORTED_COVERAGE_FLOOR = 0.80
|
||
HORIZON_AGREEMENT_FACTOR = 2
|
||
BURST_GUARD_FRACTION = 0.50
|
||
BURST_GUARD_LOOKBACK = 28
|
||
YOUNG_REGIME_DAYS = 7
|
||
HABIT_CHANGE_SHORT_WINDOW = 7
|
||
HABIT_CHANGE_LONG_WINDOW = 28
|
||
HABIT_CHANGE_UPPER_FACTOR = 2
|
||
HABIT_CHANGE_LOWER_FACTOR = 0.5
|
||
HABIT_CHANGE_CONSECUTIVE_DAYS = 3
|
||
STALENESS_HOURS = 48
|
||
WEAR_DISAGREEMENT_FACTOR = 2
|
||
|
||
|
||
class ConfidenceState(Enum):
|
||
UNSUPPORTED = "Unavailable"
|
||
LIMITED = "Limited"
|
||
SUPPORTED = "Supported"
|
||
|
||
|
||
class BaselineTier(Enum):
|
||
VERIFIED = "verified_override"
|
||
UNVERIFIED = "unverified_override"
|
||
IMPLIED = "implied"
|
||
NONE = "none"
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class ScenarioRange:
|
||
rates: Dict[int, float]
|
||
min_days: int
|
||
max_days: int
|
||
horizon_reasons: Dict[int, str] = field(default_factory=dict)
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class ProjectionResult:
|
||
confidence_state: ConfidenceState
|
||
contributing_facts: List[str]
|
||
headline_remaining_seconds: Optional[float]
|
||
scenario_range: Optional[ScenarioRange]
|
||
pu_context_line: str
|
||
disclosure_text: List[str]
|
||
baseline_tier: BaselineTier
|
||
baseline_label: str
|
||
regime_days: Optional[int]
|
||
habit_change_fact: Optional[str]
|
||
warming_fact: Optional[str]
|
||
staleness_fact: Optional[str]
|
||
degraded_identity_fact: Optional[str]
|
||
zero_rate_fact: Optional[str]
|
||
qualifying_days_progress: Optional[str] = None # Issue #77: honest qualifying-day progress
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Store queries
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _get_baseline(conn: sqlite3.Connection) -> Optional[Dict[str, Any]]:
|
||
cursor = conn.execute(
|
||
"SELECT id, tbw_terabytes, source_url, document_revision, entry_date, "
|
||
" model_string, nominal_capacity_bytes, validated_by, verified "
|
||
"FROM endurance_baseline LIMIT 1"
|
||
)
|
||
row = cursor.fetchone()
|
||
if row is None:
|
||
return None
|
||
return {
|
||
"id": row[0], "tbw_terabytes": row[1], "source_url": row[2],
|
||
"document_revision": row[3], "entry_date": row[4],
|
||
"model_string": row[5], "nominal_capacity_bytes": row[6],
|
||
"validated_by": row[7], "verified": bool(row[8]),
|
||
}
|
||
|
||
|
||
def _get_current_segment(conn: sqlite3.Connection) -> Optional[Dict[str, Any]]:
|
||
cursor = conn.execute(
|
||
"SELECT id, opened_at, identity_key, identity_degraded, mn "
|
||
"FROM controller_segments ORDER BY id DESC LIMIT 1"
|
||
)
|
||
row = cursor.fetchone()
|
||
if row is None:
|
||
return None
|
||
return {
|
||
"id": row[0], "opened_at": row[1], "identity_key": row[2],
|
||
"identity_degraded": bool(row[3]), "mn": row[4],
|
||
}
|
||
|
||
|
||
def _get_days_in_segment(conn, segment_opened_at):
|
||
cursor = conn.execute(
|
||
"SELECT day, bytes_written_delta, coverage, sample_count "
|
||
"FROM day_aggregates WHERE day >= ? ORDER BY day",
|
||
(segment_opened_at[:10],),
|
||
)
|
||
return [{"day": r[0], "bytes_written": r[1], "coverage": r[2], "sample_count": r[3]}
|
||
for r in cursor.fetchall()]
|
||
|
||
|
||
def _get_all_days(conn):
|
||
cursor = conn.execute(
|
||
"SELECT day, bytes_written_delta, coverage, sample_count "
|
||
"FROM day_aggregates ORDER BY day"
|
||
)
|
||
return [{"day": r[0], "bytes_written": r[1], "coverage": r[2], "sample_count": r[3]}
|
||
for r in cursor.fetchall()]
|
||
|
||
|
||
def _get_latest_pu(conn):
|
||
cursor = conn.execute("SELECT percentage_used FROM samples ORDER BY id DESC LIMIT 1")
|
||
row = cursor.fetchone()
|
||
return row[0] if row else None
|
||
|
||
|
||
def _get_pu_increments_in_segment(conn, segment_opened_at):
|
||
cursor = conn.execute(
|
||
"SELECT COUNT(DISTINCT percentage_used) FROM samples WHERE ts >= ?",
|
||
(segment_opened_at,),
|
||
)
|
||
row = cursor.fetchone()
|
||
return max(0, (row[0] if row else 0) - 1)
|
||
|
||
|
||
def _wall_clock_in_range(conn, start, end):
|
||
start_str = start.isoformat()
|
||
end_str = end.isoformat()
|
||
cursor = conn.execute(
|
||
"SELECT started_at, ended_at FROM monitoring_periods "
|
||
"WHERE (ended_at IS NULL OR ended_at > ?) AND started_at < ? "
|
||
"ORDER BY started_at", (start_str, end_str),
|
||
)
|
||
total = 0
|
||
for row in cursor.fetchall():
|
||
eff_start = max(row[0], start_str)
|
||
eff_end = min(row[1], end_str) if row[1] is not None else end_str
|
||
if eff_start < eff_end:
|
||
total += int((datetime.fromisoformat(eff_end) - datetime.fromisoformat(eff_start)).total_seconds())
|
||
return total
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Baseline resolution (§6.1, §6.2)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _resolve_baseline(conn, current_segment):
|
||
baseline = _get_baseline(conn)
|
||
facts: list[str] = []
|
||
if baseline is None:
|
||
return BaselineTier.NONE, None, "no baseline", facts
|
||
|
||
mandatory = [baseline["source_url"], baseline["document_revision"],
|
||
baseline["entry_date"], baseline["model_string"],
|
||
baseline["nominal_capacity_bytes"]]
|
||
provenance_complete = all(f is not None and f != "" for f in mandatory)
|
||
|
||
model_matches = True
|
||
if current_segment is not None and baseline["model_string"] is not None:
|
||
seg_mn = (current_segment.get("mn") or "").lower()
|
||
bl_model = (baseline["model_string"] or "").lower()
|
||
model_matches = bl_model in seg_mn or seg_mn in bl_model
|
||
|
||
if provenance_complete and model_matches and baseline["verified"]:
|
||
label = "verified manufacturer TBW (%.1f TB)" % baseline["tbw_terabytes"]
|
||
return BaselineTier.VERIFIED, baseline, label, facts
|
||
|
||
if not model_matches:
|
||
facts.append(
|
||
"baseline model '%s' does not match current drive '%s'"
|
||
" — baseline retained but not applicable"
|
||
% (baseline.get("model_string", ""),
|
||
current_segment.get("mn", "") if current_segment else "")
|
||
)
|
||
return BaselineTier.NONE, baseline, "baseline model mismatch", facts
|
||
|
||
if not provenance_complete:
|
||
label = "unverified TBW (%.1f TB) — user-supplied" % baseline["tbw_terabytes"]
|
||
return BaselineTier.UNVERIFIED, baseline, label, facts
|
||
|
||
label = "verified manufacturer TBW (%.1f TB)" % baseline["tbw_terabytes"]
|
||
return BaselineTier.VERIFIED, baseline, label, facts
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Rate computation
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _compute_regime_rate(days, conn, regime_start_day, clock_now):
|
||
regime_bytes = sum(d["bytes_written"] for d in days if d["day"] >= regime_start_day)
|
||
regime_start_dt = datetime.fromisoformat(regime_start_day + "T00:00:00+00:00")
|
||
regime_wc = _wall_clock_in_range(conn, regime_start_dt, clock_now)
|
||
if regime_wc <= 0:
|
||
return None, regime_bytes, 0
|
||
return regime_bytes / regime_wc, regime_bytes, regime_wc
|
||
|
||
|
||
def _compute_horizon_rate(days, conn, horizon_days, evidence_endpoint):
|
||
"""Compute horizon rate anchored at the latest evidence endpoint T.
|
||
|
||
Uses exact trailing horizon_days × 86400 seconds from T, not clock_now.
|
||
Reader refresh alone never moves T or dilutes rates.
|
||
|
||
Returns (rate, reason) where reason is None on success or a string
|
||
describing why the rate is unavailable.
|
||
"""
|
||
if not days:
|
||
return None, "no observation history"
|
||
|
||
# T is the latest evidence endpoint — the end of the last day aggregate
|
||
T = datetime.fromisoformat(days[-1]["day"] + "T23:59:59+00:00")
|
||
|
||
# Exact trailing start: T minus horizon_days × 86400 seconds
|
||
h_start = T - timedelta(days=horizon_days)
|
||
cutoff = h_start.strftime("%Y-%m-%d")
|
||
|
||
# History must span the full horizon — no placeholders
|
||
if days[0]["day"] > cutoff:
|
||
return None, "%d-day window starts before earliest data" % horizon_days
|
||
|
||
h_bytes = sum(d["bytes_written"] for d in days if d["day"] >= cutoff)
|
||
covered = sum(1 for d in days if d["day"] >= cutoff)
|
||
if covered == 0:
|
||
return None, "%d-day window has no data" % horizon_days
|
||
|
||
h_wc = _wall_clock_in_range(conn, h_start, T)
|
||
if h_wc <= 0:
|
||
return None, "%d-day window has no monitored wall-clock time" % horizon_days
|
||
|
||
return h_bytes / h_wc, None
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Habit change detection (§6.4)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _detect_habit_change(days):
|
||
"""Detect habit change per spec §6.4.
|
||
|
||
Trailing 7-day mean >= 2x (or <= 0.5x) the preceding 28-day mean
|
||
for 3 consecutive days. Returns (change_day, days_since) or None.
|
||
The first divergence day is the earliest day in the consecutive run.
|
||
"""
|
||
need = HABIT_CHANGE_SHORT_WINDOW + HABIT_CHANGE_LONG_WINDOW
|
||
if len(days) < need:
|
||
return None
|
||
|
||
def _ratio_at(end_idx):
|
||
"""Compute 7-day / preceding-28-day mean ratio ending at end_idx."""
|
||
if end_idx < HABIT_CHANGE_SHORT_WINDOW - 1:
|
||
return None
|
||
se = end_idx + 1
|
||
ss = se - HABIT_CHANGE_SHORT_WINDOW
|
||
s_bytes = sum(d["bytes_written"] for d in days[ss:se])
|
||
s_mean = s_bytes / HABIT_CHANGE_SHORT_WINDOW
|
||
le = ss
|
||
ls = le - HABIT_CHANGE_LONG_WINDOW
|
||
if ls < 0:
|
||
return None
|
||
l_bytes = sum(d["bytes_written"] for d in days[ls:le])
|
||
l_mean = l_bytes / HABIT_CHANGE_LONG_WINDOW
|
||
if l_mean == 0:
|
||
return None
|
||
return s_mean / l_mean
|
||
|
||
# Scan backwards from the most recent day
|
||
for i in range(len(days) - 1, HABIT_CHANGE_LONG_WINDOW + HABIT_CHANGE_SHORT_WINDOW - 2, -1):
|
||
ratio = _ratio_at(i)
|
||
if ratio is None:
|
||
continue
|
||
|
||
is_upper = ratio >= HABIT_CHANGE_UPPER_FACTOR
|
||
is_lower = ratio <= HABIT_CHANGE_LOWER_FACTOR
|
||
if not (is_upper or is_lower):
|
||
continue
|
||
|
||
# Count consecutive days going backwards from i
|
||
consecutive = 1
|
||
for j in range(i - 1, HABIT_CHANGE_LONG_WINDOW + HABIT_CHANGE_SHORT_WINDOW - 3, -1):
|
||
r = _ratio_at(j)
|
||
if r is None:
|
||
break
|
||
if (is_upper and r >= HABIT_CHANGE_UPPER_FACTOR) or \
|
||
(is_lower and r <= HABIT_CHANGE_LOWER_FACTOR):
|
||
consecutive += 1
|
||
else:
|
||
break
|
||
|
||
if consecutive >= HABIT_CHANGE_CONSECUTIVE_DAYS:
|
||
change_idx = i - consecutive + 1
|
||
change_day = days[change_idx]["day"]
|
||
days_since = (datetime.fromisoformat(days[-1]["day"]) - datetime.fromisoformat(change_day)).days
|
||
return change_day, days_since
|
||
|
||
return None
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Confidence rule table (§6.7)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _evaluate_confidence(tier, rate, regime_days, days, current_segment,
|
||
clock_now, warming_days, warming_low_coverage,
|
||
habit_change, staleness_hours, scenario_range):
|
||
facts = []
|
||
|
||
if tier == BaselineTier.NONE:
|
||
facts.append("no applicable endurance baseline")
|
||
return ConfidenceState.UNSUPPORTED, facts
|
||
|
||
if rate is None or rate <= 0:
|
||
facts.append("no finite projection from this history")
|
||
return ConfidenceState.UNSUPPORTED, facts
|
||
|
||
supported_facts = []
|
||
failing = False
|
||
|
||
# 1. Verified baseline
|
||
if tier != BaselineTier.VERIFIED:
|
||
failing = True
|
||
else:
|
||
supported_facts.append("verified manufacturer TBW")
|
||
|
||
# 2. >= 14 qualifying days
|
||
qualifying = sum(1 for d in days if d["coverage"] >= WARMING_COVERAGE_FLOOR and d["sample_count"] > 0)
|
||
if qualifying < WARMING_MIN_DAYS:
|
||
failing = True
|
||
else:
|
||
supported_facts.append("%d calendar days" % qualifying)
|
||
|
||
# 3. Coverage >= 80%
|
||
total_wc = len(days) * 86400
|
||
total_known = sum(int(d["coverage"] * 86400) for d in days)
|
||
avg_cov = total_known / total_wc if total_wc > 0 else 0.0
|
||
if avg_cov < SUPPORTED_COVERAGE_FLOOR:
|
||
failing = True
|
||
else:
|
||
supported_facts.append("%d%% interval coverage" % int(avg_cov * 100))
|
||
|
||
# 4. Fresh (< 48h)
|
||
if staleness_hours is not None and staleness_hours > STALENESS_HOURS:
|
||
failing = True
|
||
elif staleness_hours is not None:
|
||
supported_facts.append("recent data")
|
||
|
||
# 5. Horizon agreement
|
||
if scenario_range is not None and len(scenario_range.rates) >= 2:
|
||
rl = list(scenario_range.rates.values())
|
||
if min(rl) > 0 and max(rl) / min(rl) > HORIZON_AGREEMENT_FACTOR:
|
||
failing = True
|
||
else:
|
||
supported_facts.append("%d weekly cycles" % len(scenario_range.rates))
|
||
else:
|
||
failing = True
|
||
|
||
# 6. Burst guard
|
||
if not failing and len(days) >= BURST_GUARD_LOOKBACK:
|
||
t28 = sum(d["bytes_written"] for d in days[-BURST_GUARD_LOOKBACK:])
|
||
for d in days[-BURST_GUARD_LOOKBACK:]:
|
||
if t28 > 0 and d["bytes_written"] >= BURST_GUARD_FRACTION * t28:
|
||
failing = True
|
||
break
|
||
if not failing:
|
||
supported_facts.append("no burst days")
|
||
|
||
# 7. Regime >= 7 days
|
||
if regime_days < YOUNG_REGIME_DAYS:
|
||
failing = True
|
||
|
||
# 8. Degraded identity
|
||
if current_segment and current_segment.get("identity_degraded"):
|
||
failing = True
|
||
facts.append("controller identity unavailable — replacement detection relies on write-counter continuity only")
|
||
|
||
if not failing:
|
||
return ConfidenceState.SUPPORTED, supported_facts
|
||
|
||
# Limited
|
||
limited_facts = list(supported_facts)
|
||
if staleness_hours is not None and staleness_hours > STALENESS_HOURS:
|
||
limited_facts.append("newest data %dh old (≥48h)" % staleness_hours)
|
||
if habit_change is not None:
|
||
limited_facts.append("usage habit changed %d days ago" % habit_change[1])
|
||
if regime_days < YOUNG_REGIME_DAYS:
|
||
limited_facts.append("regime only %d days old (≥7 required)" % regime_days)
|
||
if current_segment and current_segment.get("identity_degraded"):
|
||
degraded_fact = "controller identity unavailable — replacement detection relies on write-counter continuity only"
|
||
if degraded_fact not in limited_facts:
|
||
limited_facts.append(degraded_fact)
|
||
|
||
return ConfidenceState.LIMITED, limited_facts
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Disclosure text (§6.11)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
DISCLOSURES = [
|
||
"This is an endurance projection, not a predicted hardware-failure date.",
|
||
("Percentage Used is vendor-specific; 100 means estimated endurance consumed "
|
||
"but may not mean failure, it can exceed 100, and 255 is saturated."),
|
||
("Rated TBW can be a warranty/endurance threshold with separate time and "
|
||
"eligibility terms, not a failure threshold."),
|
||
("DUW is upward-rounded host writes excluding metadata and selected commands, "
|
||
"not exact physical NAND writes."),
|
||
("Projection quality depends on baseline provenance, history duration and "
|
||
"completeness, recentness, stability, and representative usage cycles; "
|
||
"future workload and firmware behavior remain outside the observed evidence."),
|
||
("Gaps can preserve an aggregate counter delta without preserving hourly "
|
||
"timing; unexplained and deliberately disabled periods must be distinguished."),
|
||
]
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# PU context line (§6.1)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _build_pu_context_line(conn, rate, days, clock_now):
|
||
pu = _get_latest_pu(conn)
|
||
if pu is None:
|
||
return "Percentage Used: unknown"
|
||
if rate is None or rate <= 0 or not days:
|
||
return "Percentage Used: %d%%" % pu
|
||
|
||
total_bytes = sum(d["bytes_written"] for d in days)
|
||
if total_bytes <= 0:
|
||
return "Percentage Used: %d%%" % pu
|
||
|
||
total_days_count = len(days)
|
||
if total_days_count == 0:
|
||
return "Percentage Used: %d%%" % pu
|
||
|
||
pu_daily = total_bytes / total_days_count
|
||
obs_daily = rate * 86400
|
||
|
||
if pu_daily > 0:
|
||
ratio = obs_daily / pu_daily
|
||
if ratio > WEAR_DISAGREEMENT_FACTOR or ratio < 1.0 / WEAR_DISAGREEMENT_FACTOR:
|
||
return ("Percentage Used: %d%% — vendor wear estimate disagrees "
|
||
"with observed write rate (>2× difference)") % pu
|
||
|
||
return "Percentage Used: %d%%" % pu
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Complete observation day gate (issue #94)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _has_complete_local_day(conn):
|
||
"""Check if at least one complete local observation day exists.
|
||
|
||
A complete local day is a full midnight-to-midnight calendar day
|
||
within a monitoring period that has usable observation evidence.
|
||
This is the prerequisite for showing an endurance outlook.
|
||
"""
|
||
local_days = conn.execute(
|
||
"SELECT utc_start, utc_end FROM local_days "
|
||
"WHERE complete = 1 "
|
||
"AND activity_precision IN ('measured', 'coarse') "
|
||
"AND activity_intervals > 0 ORDER BY id"
|
||
).fetchall()
|
||
return any(
|
||
interval_within_one_monitoring_period(conn, utc_start, utc_end)
|
||
for utc_start, utc_end in local_days
|
||
)
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Main projection function
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def compute_projection(conn, clock_now):
|
||
facts = []
|
||
habit_change_fact = None
|
||
warming_fact = None
|
||
staleness_fact = None
|
||
degraded_identity_fact = None
|
||
zero_rate_fact = None
|
||
|
||
current_segment = _get_current_segment(conn)
|
||
tier, baseline, baseline_label, baseline_facts = _resolve_baseline(conn, current_segment)
|
||
facts.extend(baseline_facts)
|
||
|
||
# --- Complete observation day gate (issue #94) ---
|
||
# An endurance outlook requires at least one complete local
|
||
# midnight-to-midnight calendar day with usable observation evidence.
|
||
has_complete_day = _has_complete_local_day(conn)
|
||
if not has_complete_day:
|
||
facts.append("waiting for a full local observation day")
|
||
return ProjectionResult(
|
||
confidence_state=ConfidenceState.UNSUPPORTED,
|
||
contributing_facts=facts,
|
||
headline_remaining_seconds=None,
|
||
scenario_range=None,
|
||
pu_context_line="Percentage Used: unknown" if _get_latest_pu(conn) is None else "Percentage Used: %d%%" % (_get_latest_pu(conn) or 0),
|
||
disclosure_text=list(DISCLOSURES),
|
||
baseline_tier=tier,
|
||
baseline_label=baseline_label,
|
||
regime_days=None,
|
||
habit_change_fact=None,
|
||
warming_fact=None,
|
||
staleness_fact=None,
|
||
degraded_identity_fact=None,
|
||
zero_rate_fact=None,
|
||
qualifying_days_progress=None,
|
||
)
|
||
|
||
segment_days = _get_days_in_segment(conn, current_segment["opened_at"]) if current_segment else _get_all_days(conn)
|
||
all_days = _get_all_days(conn)
|
||
|
||
regime_start_day = None
|
||
habit_change = None
|
||
|
||
if segment_days:
|
||
earliest = segment_days[0]["day"]
|
||
cutoff_90 = (clock_now - timedelta(days=90)).strftime("%Y-%m-%d")
|
||
regime_start_day = max(earliest, cutoff_90)
|
||
habit_change = _detect_habit_change(segment_days)
|
||
if habit_change is not None:
|
||
regime_start_day = habit_change[0]
|
||
habit_change_fact = "usage habit changed %d days ago" % habit_change[1]
|
||
facts.append(habit_change_fact)
|
||
|
||
rate = None
|
||
regime_bytes = 0
|
||
regime_days_count = 0
|
||
|
||
if segment_days and regime_start_day is not None:
|
||
rate, regime_bytes, _ = _compute_regime_rate(segment_days, conn, regime_start_day, clock_now)
|
||
regime_days_count = sum(1 for d in segment_days if d["day"] >= regime_start_day)
|
||
|
||
if rate is not None and rate <= 0:
|
||
zero_rate_fact = "no finite projection from this history"
|
||
facts.append(zero_rate_fact)
|
||
|
||
scenario = None
|
||
horizon_rates = {}
|
||
horizon_reasons = {}
|
||
for h in HORIZON_DAYS:
|
||
hr, reason = _compute_horizon_rate(all_days, conn, h, clock_now)
|
||
if hr is not None:
|
||
horizon_rates[h] = hr
|
||
else:
|
||
horizon_reasons[h] = reason
|
||
if horizon_rates:
|
||
scenario = ScenarioRange(
|
||
rates=horizon_rates,
|
||
min_days=min(horizon_rates),
|
||
max_days=max(horizon_rates),
|
||
horizon_reasons=horizon_reasons,
|
||
)
|
||
|
||
total_days_count = len(segment_days)
|
||
days_below_coverage = sum(1 for d in segment_days
|
||
if d["coverage"] < WARMING_COVERAGE_FLOOR or d["sample_count"] == 0)
|
||
qualifying = total_days_count - days_below_coverage
|
||
|
||
# Issue #77: Show honest qualifying-day progress
|
||
if total_days_count >= WARMING_MIN_DAYS:
|
||
# After warm-up, show qualifying day details for transparency
|
||
qualifying_days_progress = "%d of %d qualifying days" % (qualifying, total_days_count)
|
||
if days_below_coverage > 0:
|
||
qualifying_days_progress += " (%d below coverage)" % days_below_coverage
|
||
else:
|
||
qualifying_days_progress = None
|
||
|
||
if total_days_count < WARMING_MIN_DAYS or days_below_coverage > WARMING_MAX_LOW_COVERAGE:
|
||
warming_fact = "warming up: %d of %d qualifying days" % (qualifying, WARMING_MIN_DAYS)
|
||
facts.append(warming_fact)
|
||
|
||
staleness_hours = None
|
||
if segment_days:
|
||
newest_dt = datetime.fromisoformat(segment_days[-1]["day"] + "T12:00:00+00:00")
|
||
staleness_hours = int((clock_now - newest_dt).total_seconds() / 3600)
|
||
if staleness_hours > STALENESS_HOURS:
|
||
staleness_fact = "newest data %dh old (≥48h)" % staleness_hours
|
||
facts.append(staleness_fact)
|
||
|
||
if current_segment and current_segment.get("identity_degraded"):
|
||
degraded_identity_fact = "controller identity unavailable — replacement detection relies on write-counter continuity only"
|
||
facts.append(degraded_identity_fact)
|
||
|
||
state, conf_facts = _evaluate_confidence(
|
||
tier, rate, regime_days_count, segment_days, current_segment, clock_now,
|
||
qualifying, 0, habit_change, staleness_hours, scenario,
|
||
)
|
||
|
||
all_facts = list(facts)
|
||
for cf in conf_facts:
|
||
if cf not in all_facts:
|
||
all_facts.append(cf)
|
||
|
||
headline_seconds = None
|
||
if state != ConfidenceState.UNSUPPORTED and rate is not None and rate > 0 and baseline is not None:
|
||
if tier in (BaselineTier.VERIFIED, BaselineTier.UNVERIFIED):
|
||
E_baseline = baseline["tbw_terabytes"] * TBW_TO_BYTES
|
||
elif tier == BaselineTier.IMPLIED:
|
||
p = _get_latest_pu(conn)
|
||
if p is not None and IMPLIED_P_MIN <= p <= IMPLIED_P_MAX:
|
||
E_baseline = 100 * regime_bytes / p
|
||
else:
|
||
E_baseline = None
|
||
else:
|
||
E_baseline = None
|
||
if E_baseline is not None:
|
||
headline_seconds = max(E_baseline - regime_bytes, 0) / rate
|
||
|
||
pu_line = _build_pu_context_line(conn, rate, segment_days, clock_now)
|
||
|
||
return ProjectionResult(
|
||
confidence_state=state,
|
||
contributing_facts=all_facts,
|
||
headline_remaining_seconds=headline_seconds,
|
||
scenario_range=scenario,
|
||
pu_context_line=pu_line,
|
||
disclosure_text=list(DISCLOSURES),
|
||
baseline_tier=tier,
|
||
baseline_label=baseline_label,
|
||
regime_days=regime_days_count,
|
||
habit_change_fact=habit_change_fact,
|
||
warming_fact=warming_fact,
|
||
staleness_fact=staleness_fact,
|
||
degraded_identity_fact=degraded_identity_fact,
|
||
zero_rate_fact=zero_rate_fact,
|
||
qualifying_days_progress=qualifying_days_progress,
|
||
)
|