#!/usr/bin/env python3 """ Fenris — NVMe wear monitor + live dashboard. Created by Bongbetic. Samples NVMe SMART health data (via smartctl -j), logs it over time, and serves a self-contained HTML dashboard estimating SSD lifespan from your actual daily usage trend. """ import json import os import subprocess import sys import time import signal import argparse import threading import hashlib import mimetypes from datetime import datetime, timezone, timedelta from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) DATA_DIR = os.path.join(SCRIPT_DIR, "data") DATA_FILE = os.path.join(DATA_DIR, "history.jsonl") HOURLY_FILE = os.path.join(DATA_DIR, "hourly.jsonl") PID_FILE = os.path.join(DATA_DIR, "fenris.pid") LOG_FILE = os.path.join(DATA_DIR, "fenris.log") ASSETS_DIR = os.path.join(SCRIPT_DIR, "assets") VERSION = "0.2.0" os.makedirs(DATA_DIR, exist_ok=True) _CONFIG = {"interval": 300, "device": "/dev/nvme0", "port": 8420, "version": VERSION} _hour_lock = threading.Lock() _current_hour = None def detect_device(): for cand in ("/dev/nvme0", "/dev/nvme1"): if os.path.exists(cand): return cand return "/dev/nvme0" def sample(device): try: out = subprocess.run( ["sudo", "-n", "smartctl", "-a", "-j", device], capture_output=True, text=True, timeout=15 ) except FileNotFoundError: print("ERROR: smartctl not found. Install smartmontools.", file=sys.stderr) return None if out.returncode not in (0, 4): print(f"smartctl failed (exit {out.returncode}): {out.stderr.strip()}", file=sys.stderr) print("Hint: needs root. Run 'sudo visudo' and allow passwordless " "'smartctl' for your user, or run Fenris with sudo.", file=sys.stderr) return None try: d = json.loads(out.stdout) except json.JSONDecodeError: return None log = d.get("nvme_smart_health_information_log") if not log: print("No NVMe SMART data in smartctl output (not an NVMe device?).", file=sys.stderr) return None capacity_bytes = (d.get("user_capacity") or {}).get("bytes", 0) units_written = log.get("data_units_written", 0) units_read = log.get("data_units_read", 0) return { "ts": datetime.now(timezone.utc).isoformat(), "device": device, "model": d.get("model_name", "unknown"), "capacity_bytes": capacity_bytes, "percentage_used": log.get("percentage_used"), "available_spare": log.get("available_spare"), "media_errors": log.get("media_errors"), "power_on_hours": log.get("power_on_hours"), "power_cycles": log.get("power_cycles"), "unsafe_shutdowns": log.get("unsafe_shutdowns"), "temperature_c": log.get("temperature"), "data_units_written": units_written, "data_units_read": units_read, "bytes_written": units_written * 512000, "bytes_read": units_read * 512000, "critical_warning": log.get("critical_warning"), } def append_sample(rec): with open(DATA_FILE, "a") as f: f.write(json.dumps(rec) + "\n") def load_history(): if not os.path.exists(DATA_FILE): return [] out = [] with open(DATA_FILE) as f: for line in f: line = line.strip() if line: try: out.append(json.loads(line)) except json.JSONDecodeError: continue return out def hour_key(ts_str): try: dt = datetime.fromisoformat(ts_str) if dt.tzinfo is None: dt = dt.replace(tzinfo=timezone.utc) else: dt = dt.astimezone(timezone.utc) dt = dt.replace(minute=0, second=0, microsecond=0) return dt.isoformat().replace("+00:00", "Z") except Exception: return ts_str[:13] + ":00:00Z" def load_hourly(): if not os.path.exists(HOURLY_FILE): return [] out = [] with open(HOURLY_FILE) as f: for line in f: line = line.strip() if line: try: out.append(json.loads(line)) except json.JSONDecodeError: continue out.sort(key=lambda r: r.get("hour", "")) return out def append_hourly(rec): with _hour_lock: with open(HOURLY_FILE, "a") as f: f.write(json.dumps(rec) + "\n") def _hourly_record_for_bucket(hour_str, recs): if not recs: return None first = recs[0] last = recs[-1] bw = (last.get("bytes_written", 0) - first.get("bytes_written", 0)) if len(recs) > 1 else 0 br = (last.get("bytes_read", 0) - first.get("bytes_read", 0)) if len(recs) > 1 else 0 temps = [r.get("temperature_c") for r in recs if r.get("temperature_c") is not None] return { "hour": hour_str, "samples": len(recs), "bytes_written": max(bw, 0), "bytes_read": max(br, 0), "pct_start": first.get("percentage_used"), "pct_end": last.get("percentage_used"), "temp_avg": round(sum(temps) / len(temps), 1) if temps else None, "temp_max": max(temps) if temps else None, "media_errors": last.get("media_errors", 0), "available_spare": last.get("available_spare"), } def rebuild_hourly_from_history(): history = load_history() if not history: return buckets = {} for r in history: hk = hour_key(r["ts"]) buckets.setdefault(hk, []).append(r) existing = {r["hour"]: r for r in load_hourly()} for hk in sorted(buckets.keys()): if hk in existing: continue rec = _hourly_record_for_bucket(hk, buckets[hk]) if rec: append_hourly(rec) def update_hour_bucket(rec): global _current_hour hk = hour_key(rec["ts"]) if _current_hour is None or _current_hour["hour"] != hk: if _current_hour is not None: flushed = _hourly_record_for_bucket(_current_hour["hour"], _current_hour["recs"]) if flushed: existing_hours = {r["hour"] for r in load_hourly()} if flushed["hour"] not in existing_hours: append_hourly(flushed) _current_hour = {"hour": hk, "recs": [rec]} else: _current_hour["recs"].append(rec) def _parse_ts(ts_str): try: dt = datetime.fromisoformat(ts_str) if dt.tzinfo is None: dt = dt.replace(tzinfo=timezone.utc) return dt.astimezone(timezone.utc) except Exception: return None def compute_summary(history=None, hourly=None): if history is None: history = load_history() if not history: return { "window24h": {"bytes": 0, "gb": 0, "coverage_hours": 0}, "gb_per_hour": 0, "gb_per_day": 0, "endurance_tb": None, "endurance_estimated": True, "remaining_tb": None, "remaining_bytes": 0, "seconds_remaining": None, "breakdown": {"years": 0, "days": 0, "hours": 0, "human": "—"}, "wear_model_days": None, "preliminary": True, "notes": ["No data yet"], "latest": None, } latest = history[-1] latest_ts = _parse_ts(latest["ts"]) earliest_ts = _parse_ts(history[0]["ts"]) total_span_sec = (latest_ts - earliest_ts).total_seconds() if latest_ts and earliest_ts else 0 window_sec = 24 * 3600 window_start_ts = latest_ts - timedelta(seconds=window_sec) if latest_ts else None idx = 0 if window_start_ts: for i, r in enumerate(history): dt = _parse_ts(r["ts"]) if dt and dt >= window_start_ts: idx = i break else: idx = 0 window_start_rec = history[idx] window_start_ts_actual = _parse_ts(window_start_rec["ts"]) coverage_sec = (latest_ts - window_start_ts_actual).total_seconds() if latest_ts and window_start_ts_actual else 0 if coverage_sec < 1: coverage_sec = total_span_sec if total_span_sec > 0 else 1 bytes_in_window = latest.get("bytes_written", 0) - window_start_rec.get("bytes_written", 0) if bytes_in_window < 0: bytes_in_window = 0 rate = bytes_in_window / coverage_sec if coverage_sec > 0 else 0 gb_per_hour = rate * 3600 / 1e9 gb_per_day = gb_per_hour * 24 coverage_hours = coverage_sec / 3600 pct = latest.get("percentage_used") cap = latest.get("capacity_bytes") or 0 bw_total = latest.get("bytes_written", 0) if pct is not None and pct > 0: endurance_bytes = bw_total / (pct / 100) endurance_estimated = False else: endurance_bytes = cap * 600 if cap else 0 endurance_estimated = True remaining_bytes = max(endurance_bytes - bw_total, 0) if endurance_bytes else 0 endurance_tb = endurance_bytes / 1e12 if endurance_bytes else None remaining_tb = remaining_bytes / 1e12 if remaining_bytes else 0 seconds_remaining = None if rate > 0 and remaining_bytes > 0: seconds_remaining = remaining_bytes / rate def _humanize(secs): if secs is None or secs <= 0: return "—" years = int(secs // 31557600) rem = secs % 31557600 days = int(rem // 86400) rem %= 86400 hours = int(rem // 3600) parts = [] if years: parts.append(f"{years} yr") if days or years: parts.append(f"{days} d") parts.append(f"{hours} h") return " ".join(parts) human = _humanize(seconds_remaining) breakdown = { "years": (seconds_remaining / 31557600) if seconds_remaining else 0, "days": (seconds_remaining / 86400) if seconds_remaining else 0, "hours": (seconds_remaining / 3600) if seconds_remaining else 0, "human": human, } wear_days = None pts = [( _parse_ts(r["ts"]).timestamp(), r["percentage_used"]) for r in history if r.get("percentage_used") is not None and _parse_ts(r["ts"]) is not None] if len(pts) >= 2: win_pts = [p for p in pts if p[0] >= (latest_ts.timestamp() - window_sec)] if latest_ts else pts if len(win_pts) < 2: win_pts = pts n = len(win_pts) mean_x = sum(p[0] for p in win_pts) / n mean_y = sum(p[1] for p in win_pts) / n num = sum((p[0]-mean_x)*(p[1]-mean_y) for p in win_pts) den = sum((p[0]-mean_x)**2 for p in win_pts) if den != 0: slope = num/den if slope > 0: last_pct = win_pts[-1][1] secs_to_100 = (100 - last_pct) / slope wear_days = secs_to_100 / 86400 preliminary = coverage_hours < 24 notes = [] if preliminary: notes.append(f"Warming up — {coverage_hours:.1f}h of 24h") if endurance_estimated: notes.append("TBW estimated from capacity (pct=0)") if rate <= 0: notes.append("No writes in window") if wear_days is None: notes.append("Wear flat — write model only") return { "window24h": {"bytes": bytes_in_window, "gb": bytes_in_window / 1e9, "coverage_hours": coverage_hours}, "gb_per_hour": gb_per_hour, "gb_per_day": gb_per_day, "endurance_tb": endurance_tb, "endurance_estimated": endurance_estimated, "remaining_tb": remaining_tb, "remaining_bytes": remaining_bytes, "seconds_remaining": seconds_remaining, "breakdown": breakdown, "wear_model_days": wear_days, "preliminary": preliminary, "notes": notes, "latest": latest, } def collector_loop(device, interval, stop_event): while not stop_event.is_set(): rec = sample(device) if rec: append_sample(rec) update_hour_bucket(rec) stop_event.wait(interval) DASHBOARD_HTML = r"""