#!/usr/bin/env python3 """ Fenris — NVMe wear monitor + live dashboard. Created by Bongbetic. Samples NVMe SMART health data (via smartctl -j), logs it over time, and serves a self-contained HTML dashboard estimating SSD lifespan from your actual daily usage trend. """ import json import os import subprocess import sys import time import signal import argparse import threading import hashlib import mimetypes from datetime import datetime, timezone, timedelta from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) DATA_DIR = os.path.join(SCRIPT_DIR, "data") DATA_FILE = os.path.join(DATA_DIR, "history.jsonl") HOURLY_FILE = os.path.join(DATA_DIR, "hourly.jsonl") PID_FILE = os.path.join(DATA_DIR, "fenris.pid") LOG_FILE = os.path.join(DATA_DIR, "fenris.log") ASSETS_DIR = os.path.join(SCRIPT_DIR, "assets") VERSION = "0.2.0" os.makedirs(DATA_DIR, exist_ok=True) _CONFIG = {"interval": 300, "device": "/dev/nvme0", "port": 8420, "version": VERSION} _hour_lock = threading.Lock() _current_hour = None def detect_device(): for cand in ("/dev/nvme0", "/dev/nvme1"): if os.path.exists(cand): return cand return "/dev/nvme0" def sample(device): try: out = subprocess.run( ["sudo", "-n", "smartctl", "-a", "-j", device], capture_output=True, text=True, timeout=15 ) except FileNotFoundError: print("ERROR: smartctl not found. Install smartmontools.", file=sys.stderr) return None if out.returncode not in (0, 4): print(f"smartctl failed (exit {out.returncode}): {out.stderr.strip()}", file=sys.stderr) print("Hint: needs root. Run 'sudo visudo' and allow passwordless " "'smartctl' for your user, or run Fenris with sudo.", file=sys.stderr) return None try: d = json.loads(out.stdout) except json.JSONDecodeError: return None log = d.get("nvme_smart_health_information_log") if not log: print("No NVMe SMART data in smartctl output (not an NVMe device?).", file=sys.stderr) return None capacity_bytes = (d.get("user_capacity") or {}).get("bytes", 0) units_written = log.get("data_units_written", 0) units_read = log.get("data_units_read", 0) return { "ts": datetime.now(timezone.utc).isoformat(), "device": device, "model": d.get("model_name", "unknown"), "capacity_bytes": capacity_bytes, "percentage_used": log.get("percentage_used"), "available_spare": log.get("available_spare"), "media_errors": log.get("media_errors"), "power_on_hours": log.get("power_on_hours"), "power_cycles": log.get("power_cycles"), "unsafe_shutdowns": log.get("unsafe_shutdowns"), "temperature_c": log.get("temperature"), "data_units_written": units_written, "data_units_read": units_read, "bytes_written": units_written * 512000, "bytes_read": units_read * 512000, "critical_warning": log.get("critical_warning"), } def append_sample(rec): with open(DATA_FILE, "a") as f: f.write(json.dumps(rec) + "\n") def load_history(): if not os.path.exists(DATA_FILE): return [] out = [] with open(DATA_FILE) as f: for line in f: line = line.strip() if line: try: out.append(json.loads(line)) except json.JSONDecodeError: continue return out def hour_key(ts_str): try: dt = datetime.fromisoformat(ts_str) if dt.tzinfo is None: dt = dt.replace(tzinfo=timezone.utc) else: dt = dt.astimezone(timezone.utc) dt = dt.replace(minute=0, second=0, microsecond=0) return dt.isoformat().replace("+00:00", "Z") except Exception: return ts_str[:13] + ":00:00Z" def load_hourly(): if not os.path.exists(HOURLY_FILE): return [] out = [] with open(HOURLY_FILE) as f: for line in f: line = line.strip() if line: try: out.append(json.loads(line)) except json.JSONDecodeError: continue out.sort(key=lambda r: r.get("hour", "")) return out def append_hourly(rec): with _hour_lock: with open(HOURLY_FILE, "a") as f: f.write(json.dumps(rec) + "\n") def _hourly_record_for_bucket(hour_str, recs): if not recs: return None first = recs[0] last = recs[-1] bw = (last.get("bytes_written", 0) - first.get("bytes_written", 0)) if len(recs) > 1 else 0 br = (last.get("bytes_read", 0) - first.get("bytes_read", 0)) if len(recs) > 1 else 0 temps = [r.get("temperature_c") for r in recs if r.get("temperature_c") is not None] return { "hour": hour_str, "samples": len(recs), "bytes_written": max(bw, 0), "bytes_read": max(br, 0), "pct_start": first.get("percentage_used"), "pct_end": last.get("percentage_used"), "temp_avg": round(sum(temps) / len(temps), 1) if temps else None, "temp_max": max(temps) if temps else None, "media_errors": last.get("media_errors", 0), "available_spare": last.get("available_spare"), } def rebuild_hourly_from_history(): history = load_history() if not history: return buckets = {} for r in history: hk = hour_key(r["ts"]) buckets.setdefault(hk, []).append(r) existing = {r["hour"]: r for r in load_hourly()} for hk in sorted(buckets.keys()): if hk in existing: continue rec = _hourly_record_for_bucket(hk, buckets[hk]) if rec: append_hourly(rec) def update_hour_bucket(rec): global _current_hour hk = hour_key(rec["ts"]) if _current_hour is None or _current_hour["hour"] != hk: if _current_hour is not None: flushed = _hourly_record_for_bucket(_current_hour["hour"], _current_hour["recs"]) if flushed: existing_hours = {r["hour"] for r in load_hourly()} if flushed["hour"] not in existing_hours: append_hourly(flushed) _current_hour = {"hour": hk, "recs": [rec]} else: _current_hour["recs"].append(rec) def _parse_ts(ts_str): try: dt = datetime.fromisoformat(ts_str) if dt.tzinfo is None: dt = dt.replace(tzinfo=timezone.utc) return dt.astimezone(timezone.utc) except Exception: return None def compute_summary(history=None, hourly=None): if history is None: history = load_history() if not history: return { "window24h": {"bytes": 0, "gb": 0, "coverage_hours": 0}, "gb_per_hour": 0, "gb_per_day": 0, "endurance_tb": None, "endurance_estimated": True, "remaining_tb": None, "remaining_bytes": 0, "seconds_remaining": None, "breakdown": {"years": 0, "days": 0, "hours": 0, "human": "—"}, "wear_model_days": None, "preliminary": True, "notes": ["No data yet"], "latest": None, } latest = history[-1] latest_ts = _parse_ts(latest["ts"]) earliest_ts = _parse_ts(history[0]["ts"]) total_span_sec = (latest_ts - earliest_ts).total_seconds() if latest_ts and earliest_ts else 0 window_sec = 24 * 3600 window_start_ts = latest_ts - timedelta(seconds=window_sec) if latest_ts else None idx = 0 if window_start_ts: for i, r in enumerate(history): dt = _parse_ts(r["ts"]) if dt and dt >= window_start_ts: idx = i break else: idx = 0 window_start_rec = history[idx] window_start_ts_actual = _parse_ts(window_start_rec["ts"]) coverage_sec = (latest_ts - window_start_ts_actual).total_seconds() if latest_ts and window_start_ts_actual else 0 if coverage_sec < 1: coverage_sec = total_span_sec if total_span_sec > 0 else 1 bytes_in_window = latest.get("bytes_written", 0) - window_start_rec.get("bytes_written", 0) if bytes_in_window < 0: bytes_in_window = 0 rate = bytes_in_window / coverage_sec if coverage_sec > 0 else 0 gb_per_hour = rate * 3600 / 1e9 gb_per_day = gb_per_hour * 24 coverage_hours = coverage_sec / 3600 pct = latest.get("percentage_used") cap = latest.get("capacity_bytes") or 0 bw_total = latest.get("bytes_written", 0) if pct is not None and pct > 0: endurance_bytes = bw_total / (pct / 100) endurance_estimated = False else: endurance_bytes = cap * 600 if cap else 0 endurance_estimated = True remaining_bytes = max(endurance_bytes - bw_total, 0) if endurance_bytes else 0 endurance_tb = endurance_bytes / 1e12 if endurance_bytes else None remaining_tb = remaining_bytes / 1e12 if remaining_bytes else 0 seconds_remaining = None if rate > 0 and remaining_bytes > 0: seconds_remaining = remaining_bytes / rate def _humanize(secs): if secs is None or secs <= 0: return "—" years = int(secs // 31557600) rem = secs % 31557600 days = int(rem // 86400) rem %= 86400 hours = int(rem // 3600) parts = [] if years: parts.append(f"{years} yr") if days or years: parts.append(f"{days} d") parts.append(f"{hours} h") return " ".join(parts) human = _humanize(seconds_remaining) breakdown = { "years": (seconds_remaining / 31557600) if seconds_remaining else 0, "days": (seconds_remaining / 86400) if seconds_remaining else 0, "hours": (seconds_remaining / 3600) if seconds_remaining else 0, "human": human, } wear_days = None pts = [( _parse_ts(r["ts"]).timestamp(), r["percentage_used"]) for r in history if r.get("percentage_used") is not None and _parse_ts(r["ts"]) is not None] if len(pts) >= 2: win_pts = [p for p in pts if p[0] >= (latest_ts.timestamp() - window_sec)] if latest_ts else pts if len(win_pts) < 2: win_pts = pts n = len(win_pts) mean_x = sum(p[0] for p in win_pts) / n mean_y = sum(p[1] for p in win_pts) / n num = sum((p[0]-mean_x)*(p[1]-mean_y) for p in win_pts) den = sum((p[0]-mean_x)**2 for p in win_pts) if den != 0: slope = num/den if slope > 0: last_pct = win_pts[-1][1] secs_to_100 = (100 - last_pct) / slope wear_days = secs_to_100 / 86400 preliminary = coverage_hours < 24 notes = [] if preliminary: notes.append(f"Warming up — {coverage_hours:.1f}h of 24h") if endurance_estimated: notes.append("TBW estimated from capacity (pct=0)") if rate <= 0: notes.append("No writes in window") if wear_days is None: notes.append("Wear flat — write model only") return { "window24h": {"bytes": bytes_in_window, "gb": bytes_in_window / 1e9, "coverage_hours": coverage_hours}, "gb_per_hour": gb_per_hour, "gb_per_day": gb_per_day, "endurance_tb": endurance_tb, "endurance_estimated": endurance_estimated, "remaining_tb": remaining_tb, "remaining_bytes": remaining_bytes, "seconds_remaining": seconds_remaining, "breakdown": breakdown, "wear_model_days": wear_days, "preliminary": preliminary, "notes": notes, "latest": latest, } def collector_loop(device, interval, stop_event): while not stop_event.is_set(): rec = sample(device) if rec: append_sample(rec) update_hour_bucket(rec) stop_event.wait(interval) DASHBOARD_HTML = r""" Fenris — NVMe Wear Dashboard
Bongbetic Bongbetic ·
FENRIS
loading…
● Live every — next —
Wear (% used) over time
Trailing 24h — GB written per hour
""" class Handler(BaseHTTPRequestHandler): def log_message(self, fmt, *args): pass def do_GET(self): path = self.path.split("?")[0] if path == "/" or path.startswith("/index"): body = DASHBOARD_HTML.encode() self.send_response(200) self.send_header("Content-Type", "text/html; charset=utf-8") self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) return if path.startswith("/assets/"): rel = path[len("/assets/"):] if ".." in rel or rel.startswith("/"): self.send_response(404); self.end_headers(); return fp = os.path.join(ASSETS_DIR, rel) if not os.path.isfile(fp): self.send_response(404); self.end_headers(); return ctype = mimetypes.guess_type(fp)[0] or "application/octet-stream" with open(fp, "rb") as f: body = f.read() self.send_response(200) self.send_header("Content-Type", ctype) self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) return if path == "/api/config": body = json.dumps(_CONFIG).encode() self.send_response(200) self.send_header("Content-Type", "application/json") self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) return if path == "/api/status": hist = load_history() latest = hist[-1] if hist else None summ = compute_summary(hist) body = json.dumps({ "alive": True, "samples": len(hist), "last_ts": latest["ts"] if latest else None, "version": VERSION, "config": _CONFIG, "summary": summ, }).encode() self.send_response(200) self.send_header("Content-Type", "application/json") self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) return if path.startswith("/api/data"): hist = load_history() body = json.dumps(hist).encode() etag = f'"{hashlib.md5(body).hexdigest()}"' inm = self.headers.get("If-None-Match") if inm and inm == etag: self.send_response(304); self.end_headers(); return self.send_response(200) self.send_header("Content-Type", "application/json") self.send_header("ETag", etag) self.send_header("Cache-Control", "no-cache") self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) return if path.startswith("/api/hourly"): hourly = load_hourly() body = json.dumps(hourly).encode() etag = f'"{hashlib.md5(body).hexdigest()}"' inm = self.headers.get("If-None-Match") if inm and inm == etag: self.send_response(304); self.end_headers(); return self.send_response(200) self.send_header("Content-Type", "application/json") self.send_header("ETag", etag) self.send_header("Cache-Control", "no-cache") self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) return if path.startswith("/api/summary"): summ = compute_summary() body = json.dumps(summ).encode() self.send_response(200) self.send_header("Content-Type", "application/json") self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) return self.send_response(404) self.end_headers() def run_foreground(device, interval, port): global _CONFIG, _current_hour _CONFIG = {"interval": interval, "device": device, "port": port, "version": VERSION} stop_event = threading.Event() def handle_sig(signum, frame): stop_event.set() signal.signal(signal.SIGTERM, handle_sig) signal.signal(signal.SIGINT, handle_sig) def cleanup_pid(): try: os.remove(PID_FILE) except OSError: pass import atexit atexit.register(cleanup_pid) print(f"Fenris starting — device={device} interval={interval}s port={port}") print(f"Data log: {DATA_FILE}") rebuild_hourly_from_history() rec = sample(device) if rec: append_sample(rec) update_hour_bucket(rec) else: print("WARNING: initial sample failed — check smartctl/sudo setup. " "The daemon will keep retrying.", file=sys.stderr) t = threading.Thread(target=collector_loop, args=(device, interval, stop_event), daemon=True) t.start() server = ThreadingHTTPServer(("0.0.0.0", port), Handler) server.timeout = 1 print(f"Dashboard: http://localhost:{port}") def server_loop(): while not stop_event.is_set(): server.handle_request() st = threading.Thread(target=server_loop, daemon=True) st.start() while not stop_event.is_set(): time.sleep(0.5) print("Fenris stopping.") def cmd_start(args): if os.path.exists(PID_FILE): with open(PID_FILE) as f: pid = int(f.read().strip()) if pid_alive(pid): print(f"Fenris already running (pid {pid}). Use 'status' or 'stop'.") return os.remove(PID_FILE) log_f = open(LOG_FILE, "a") proc = subprocess.Popen( [sys.executable, os.path.abspath(__file__), "run", "--device", args.device, "--interval", str(args.interval), "--port", str(args.port)], stdout=log_f, stderr=log_f, stdin=subprocess.DEVNULL, start_new_session=True, ) with open(PID_FILE, "w") as f: f.write(str(proc.pid)) time.sleep(0.5) print(f"Fenris started in background (pid {proc.pid}).") print(f"Dashboard: http://localhost:{args.port}") print(f"Logs: {LOG_FILE}") def pid_alive(pid): try: os.kill(pid, 0) return True except OSError: return False def cmd_stop(args): if not os.path.exists(PID_FILE): print("Fenris is not running (no pid file).") return with open(PID_FILE) as f: pid = int(f.read().strip()) if pid_alive(pid): os.kill(pid, signal.SIGTERM) for _ in range(50): if not pid_alive(pid): break time.sleep(0.1) if pid_alive(pid): print(f"SIGTERM did not stop pid {pid}, sending SIGKILL...") os.kill(pid, signal.SIGKILL) for _ in range(20): if not pid_alive(pid): break time.sleep(0.1) if pid_alive(pid): print(f"ERROR: could not stop pid {pid}. Manual intervention needed.") return print(f"Stopped Fenris (pid {pid}).") else: print("Stale pid file — process was not running.") try: os.remove(PID_FILE) except OSError: pass def cmd_status(args): running = False if os.path.exists(PID_FILE): with open(PID_FILE) as f: pid = int(f.read().strip()) running = pid_alive(pid) print(f"Fenris daemon: {'RUNNING (pid ' + str(pid) + ')' if running else 'not running (stale pid file)'}") else: print("Fenris daemon: not running") rows = load_history() if rows: summ = compute_summary(rows) latest = rows[-1] print(f"Samples collected: {len(rows)}") print(f"Last sample: {latest['ts']}") print(f"Wear (percentage_used): {latest.get('percentage_used')}%") print(f"Total written: {latest.get('bytes_written', 0) / 1e9:.1f} GB") print(f"Written (24h rolling): {summ['window24h']['gb']:.2f} GB over {summ['window24h']['coverage_hours']:.1f}h") print(f"Write rate: {summ['gb_per_hour']:.2f} GB/h ({summ['gb_per_day']:.1f} GB/day)") if summ["seconds_remaining"]: print(f"Projected life remaining: {summ['breakdown']['human']} (≈{summ['breakdown']['days']:.0f} days / {summ['breakdown']['hours']:.0f} hours / {summ['breakdown']['years']:.2f} years)") print(f"Endurance: {summ['endurance_tb']:.1f} TB total, {summ['remaining_tb']:.1f} TB remaining" + (" (estimated)" if summ["endurance_estimated"] else "")) if summ["wear_model_days"]: print(f"Wear-model cross-check: ~{summ['wear_model_days']:.0f} days at current wear rate") if summ["preliminary"]: print("Note: preliminary — less than 24h coverage") else: print("Projected life remaining: — (no writes in window or no endurance data)") hourly = load_hourly() if hourly: print(f"Hourly buckets: {len(hourly)} (last {hourly[-1]['hour']}: {hourly[-1]['bytes_written']/1e9:.2f} GB)") else: print("No samples collected yet.") def cmd_run(args): run_foreground(args.device, args.interval, args.port) def cmd_sample_once(args): rec = sample(args.device) if rec: append_sample(rec) update_hour_bucket(rec) print(json.dumps(rec, indent=2)) summ = compute_summary() print(f"\n24h: {summ['window24h']['gb']:.2f} GB rate {summ['gb_per_hour']:.2f} GB/h remaining {summ['breakdown']['human']}") else: sys.exit(1) def main(): p = argparse.ArgumentParser(description="Fenris — NVMe wear monitor & dashboard (by Bongbetic)") sub = p.add_subparsers(dest="cmd", required=True) def add_common(sp): sp.add_argument("--device", default=detect_device(), help="NVMe device, e.g. /dev/nvme0") sp.add_argument("--interval", type=int, default=300, help="seconds between samples (default 300)") sp.add_argument("--port", type=int, default=8420, help="dashboard HTTP port (default 8420)") sp = sub.add_parser("start", help="start monitoring in background") add_common(sp); sp.set_defaults(func=cmd_start) sp = sub.add_parser("stop", help="stop background monitoring") sp.set_defaults(func=cmd_stop) sp = sub.add_parser("status", help="show daemon + latest wear stats") sp.set_defaults(func=cmd_status) sp = sub.add_parser("run", help="run in foreground (used internally by 'start')") add_common(sp); sp.set_defaults(func=cmd_run) sp = sub.add_parser("sample", help="take one sample immediately and print it") sp.add_argument("--device", default=detect_device()) sp.set_defaults(func=cmd_sample_once) args = p.parse_args() args.func(args) if __name__ == "__main__": main()