commit 53cc6c81b9abac5356e42236332f2316bec1bb5c Author: xavierk Date: Sun Aug 23 18:47:12 2026 +0530 Initial commit: Fenris NVMe wear monitor with rolling-24h dashboard diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..69ac2c8 --- /dev/null +++ b/.gitignore @@ -0,0 +1,7 @@ +__pycache__/ +*.pyc +.commandcode/ +data/fenris.pid +data/fenris.log +data/history.jsonl +data/hourly.jsonl diff --git a/README.md b/README.md new file mode 100644 index 0000000..b60ee71 --- /dev/null +++ b/README.md @@ -0,0 +1,145 @@ +# Fenris — NVMe Wear Monitor & Live Dashboard + +Created by Bongbetic. + +Periodically reads your NVMe drive's SMART health data, logs it over time, and serves a self-contained HTML dashboard estimating SSD lifespan from your actual daily usage trend. + +## Requirements + +- **Python 3.7+** +- **smartmontools** (`smartctl`) installed +- Root access to read NVMe SMART logs + +### Setting up passwordless smartctl + +Fenris runs `sudo -n smartctl ...` (no-prompt sudo). Either run with sudo or allow passwordless access: + +```bash +sudo visudo +# Add this line (replace youruser with your username): +youruser ALL=(root) NOPASSWD: /usr/sbin/smartctl +``` + +## Quick Start + +### Interactive Menu + +```bash +./fenris.sh +``` + +### CLI + +```bash +# Start daemon + dashboard in background +python3 fenris.py start + +# Check status and latest wear stats +python3 fenris.py status + +# Take one sample now +python3 fenris.py sample + +# Stop the daemon +python3 fenris.py stop +``` + +Dashboard available at: `http://localhost:8420` + +## CLI Reference + +### fenris.py start + +Start monitoring in background (daemon + dashboard). + +```bash +python3 fenris.py start [OPTIONS] + +Options: + --device PATH NVMe device (default: auto-detect, e.g. /dev/nvme0) + --interval SEC Seconds between samples (default: 300) + --port PORT Dashboard HTTP port (default: 8420) +``` + +### fenris.py stop + +Stop background monitoring and clean up PID file. + +### fenris.py status + +Show daemon status and latest wear statistics. + +### fenris.py sample + +Take one sample immediately and print it. + +```bash +python3 fenris.py sample [--device /dev/nvme0] +``` + +### fenris.py run + +Run in foreground (used internally by `start`). Not intended for direct use. + +## Interactive Menu + +Run `./fenris.sh` for a guided interface: + +``` + 1) Start monitoring (background daemon + dashboard) + 2) Stop monitoring + 3) Status / current wear stats + 4) Take one sample right now + 5) Open dashboard URL + --- + h) Help / how this works + q) Exit +``` + +## Data Collected + +| Field | Description | +|-------|-------------| +| `percentage_used` | SSD's own wear indicator (0-100%) | +| `bytes_written` / `bytes_read` | Total data written/read | +| `available_spare` | Remaining spare capacity (%) | +| `media_errors` | Number of uncorrectable errors | +| `power_on_hours` | Total power-on hours | +| `temperature_c` | Current temperature | +| `critical_warning` | NVMe critical warning flags | + +## Dashboard Features + +- Real-time wear level + projected life remaining in **hours / days / years** from the actual **rolling-24h write rate** and implied TBW endurance +- Exact **GB written in the last 24 hours** + GB/h and GB/day rate (updates every poll interval) +- Wear-over-time chart + **trailing-24h per-hour write bars** +- Dense layout with sticky header, ETag-cached polling synced to the daemon interval, countdown and live badge, stale/preliminary banners +- API: `GET /api/data`, `/api/hourly`, `/api/summary`, `/api/config`, `/api/status` + +## File Structure + +``` +fenris/ +├── fenris.py # Main Python script +├── fenris.sh # Interactive menu wrapper +├── README.md +└── data/ + ├── history.jsonl # Raw sample log (JSONL) + ├── hourly.jsonl # Per-hour aggregates (rebuilt from history on restart) + ├── fenris.pid # Daemon PID file + └── fenris.log # Daemon log output +``` + +## Troubleshooting + +**"smartctl not found"** +```bash +sudo apt install smartmontools # Debian/Ubuntu +sudo pacman -S smartmontools # Arch +``` + +**"needs root" / permission denied** +Set up passwordless sudo (see Requirements) or run with sudo. + +**Dashboard shows "stale"** +Daemon not running. Check with `python3 fenris.py status` and restart if needed. diff --git a/fenris.py b/fenris.py new file mode 100755 index 0000000..d13150a --- /dev/null +++ b/fenris.py @@ -0,0 +1,1035 @@ +#!/usr/bin/env python3 +""" +Fenris — NVMe wear monitor + live dashboard. +Created by Bongbetic. + +Samples NVMe SMART health data (via smartctl -j), logs it over time, +and serves a self-contained HTML dashboard estimating SSD lifespan +from your actual daily usage trend. +""" + +import json +import os +import subprocess +import sys +import time +import signal +import argparse +import threading +import hashlib +import mimetypes +from datetime import datetime, timezone, timedelta +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +DATA_DIR = os.path.join(SCRIPT_DIR, "data") +DATA_FILE = os.path.join(DATA_DIR, "history.jsonl") +HOURLY_FILE = os.path.join(DATA_DIR, "hourly.jsonl") +PID_FILE = os.path.join(DATA_DIR, "fenris.pid") +LOG_FILE = os.path.join(DATA_DIR, "fenris.log") +ASSETS_DIR = os.path.join(SCRIPT_DIR, "assets") +VERSION = "0.2.0" + +os.makedirs(DATA_DIR, exist_ok=True) + +_CONFIG = {"interval": 300, "device": "/dev/nvme0", "port": 8420, "version": VERSION} +_hour_lock = threading.Lock() +_current_hour = None + + +def detect_device(): + for cand in ("/dev/nvme0", "/dev/nvme1"): + if os.path.exists(cand): + return cand + return "/dev/nvme0" + + +def sample(device): + try: + out = subprocess.run( + ["sudo", "-n", "smartctl", "-a", "-j", device], + capture_output=True, text=True, timeout=15 + ) + except FileNotFoundError: + print("ERROR: smartctl not found. Install smartmontools.", file=sys.stderr) + return None + if out.returncode not in (0, 4): + print(f"smartctl failed (exit {out.returncode}): {out.stderr.strip()}", file=sys.stderr) + print("Hint: needs root. Run 'sudo visudo' and allow passwordless " + "'smartctl' for your user, or run Fenris with sudo.", file=sys.stderr) + return None + try: + d = json.loads(out.stdout) + except json.JSONDecodeError: + return None + log = d.get("nvme_smart_health_information_log") + if not log: + print("No NVMe SMART data in smartctl output (not an NVMe device?).", file=sys.stderr) + return None + capacity_bytes = (d.get("user_capacity") or {}).get("bytes", 0) + units_written = log.get("data_units_written", 0) + units_read = log.get("data_units_read", 0) + return { + "ts": datetime.now(timezone.utc).isoformat(), + "device": device, + "model": d.get("model_name", "unknown"), + "capacity_bytes": capacity_bytes, + "percentage_used": log.get("percentage_used"), + "available_spare": log.get("available_spare"), + "media_errors": log.get("media_errors"), + "power_on_hours": log.get("power_on_hours"), + "power_cycles": log.get("power_cycles"), + "unsafe_shutdowns": log.get("unsafe_shutdowns"), + "temperature_c": log.get("temperature"), + "data_units_written": units_written, + "data_units_read": units_read, + "bytes_written": units_written * 512000, + "bytes_read": units_read * 512000, + "critical_warning": log.get("critical_warning"), + } + + +def append_sample(rec): + with open(DATA_FILE, "a") as f: + f.write(json.dumps(rec) + "\n") + + +def load_history(): + if not os.path.exists(DATA_FILE): + return [] + out = [] + with open(DATA_FILE) as f: + for line in f: + line = line.strip() + if line: + try: + out.append(json.loads(line)) + except json.JSONDecodeError: + continue + return out + + +def hour_key(ts_str): + try: + dt = datetime.fromisoformat(ts_str) + if dt.tzinfo is None: + dt = dt.replace(tzinfo=timezone.utc) + else: + dt = dt.astimezone(timezone.utc) + dt = dt.replace(minute=0, second=0, microsecond=0) + return dt.isoformat().replace("+00:00", "Z") + except Exception: + return ts_str[:13] + ":00:00Z" + + +def load_hourly(): + if not os.path.exists(HOURLY_FILE): + return [] + out = [] + with open(HOURLY_FILE) as f: + for line in f: + line = line.strip() + if line: + try: + out.append(json.loads(line)) + except json.JSONDecodeError: + continue + out.sort(key=lambda r: r.get("hour", "")) + return out + + +def append_hourly(rec): + with _hour_lock: + with open(HOURLY_FILE, "a") as f: + f.write(json.dumps(rec) + "\n") + + +def _hourly_record_for_bucket(hour_str, recs): + if not recs: + return None + first = recs[0] + last = recs[-1] + bw = (last.get("bytes_written", 0) - first.get("bytes_written", 0)) if len(recs) > 1 else 0 + br = (last.get("bytes_read", 0) - first.get("bytes_read", 0)) if len(recs) > 1 else 0 + temps = [r.get("temperature_c") for r in recs if r.get("temperature_c") is not None] + return { + "hour": hour_str, + "samples": len(recs), + "bytes_written": max(bw, 0), + "bytes_read": max(br, 0), + "pct_start": first.get("percentage_used"), + "pct_end": last.get("percentage_used"), + "temp_avg": round(sum(temps) / len(temps), 1) if temps else None, + "temp_max": max(temps) if temps else None, + "media_errors": last.get("media_errors", 0), + "available_spare": last.get("available_spare"), + } + + +def rebuild_hourly_from_history(): + history = load_history() + if not history: + return + buckets = {} + for r in history: + hk = hour_key(r["ts"]) + buckets.setdefault(hk, []).append(r) + existing = {r["hour"]: r for r in load_hourly()} + for hk in sorted(buckets.keys()): + if hk in existing: + continue + rec = _hourly_record_for_bucket(hk, buckets[hk]) + if rec: + append_hourly(rec) + + +def update_hour_bucket(rec): + global _current_hour + hk = hour_key(rec["ts"]) + if _current_hour is None or _current_hour["hour"] != hk: + if _current_hour is not None: + flushed = _hourly_record_for_bucket(_current_hour["hour"], _current_hour["recs"]) + if flushed: + existing_hours = {r["hour"] for r in load_hourly()} + if flushed["hour"] not in existing_hours: + append_hourly(flushed) + _current_hour = {"hour": hk, "recs": [rec]} + else: + _current_hour["recs"].append(rec) + + +def _parse_ts(ts_str): + try: + dt = datetime.fromisoformat(ts_str) + if dt.tzinfo is None: + dt = dt.replace(tzinfo=timezone.utc) + return dt.astimezone(timezone.utc) + except Exception: + return None + + +def compute_summary(history=None, hourly=None): + if history is None: + history = load_history() + if not history: + return { + "window24h": {"bytes": 0, "gb": 0, "coverage_hours": 0}, + "gb_per_hour": 0, + "gb_per_day": 0, + "endurance_tb": None, + "endurance_estimated": True, + "remaining_tb": None, + "remaining_bytes": 0, + "seconds_remaining": None, + "breakdown": {"years": 0, "days": 0, "hours": 0, "human": "—"}, + "wear_model_days": None, + "preliminary": True, + "notes": ["No data yet"], + "latest": None, + } + latest = history[-1] + latest_ts = _parse_ts(latest["ts"]) + earliest_ts = _parse_ts(history[0]["ts"]) + total_span_sec = (latest_ts - earliest_ts).total_seconds() if latest_ts and earliest_ts else 0 + window_sec = 24 * 3600 + window_start_ts = latest_ts - timedelta(seconds=window_sec) if latest_ts else None + + idx = 0 + if window_start_ts: + for i, r in enumerate(history): + dt = _parse_ts(r["ts"]) + if dt and dt >= window_start_ts: + idx = i + break + else: + idx = 0 + window_start_rec = history[idx] + window_start_ts_actual = _parse_ts(window_start_rec["ts"]) + coverage_sec = (latest_ts - window_start_ts_actual).total_seconds() if latest_ts and window_start_ts_actual else 0 + if coverage_sec < 1: + coverage_sec = total_span_sec if total_span_sec > 0 else 1 + + bytes_in_window = latest.get("bytes_written", 0) - window_start_rec.get("bytes_written", 0) + if bytes_in_window < 0: + bytes_in_window = 0 + + rate = bytes_in_window / coverage_sec if coverage_sec > 0 else 0 + gb_per_hour = rate * 3600 / 1e9 + gb_per_day = gb_per_hour * 24 + coverage_hours = coverage_sec / 3600 + + pct = latest.get("percentage_used") + cap = latest.get("capacity_bytes") or 0 + bw_total = latest.get("bytes_written", 0) + if pct is not None and pct > 0: + endurance_bytes = bw_total / (pct / 100) + endurance_estimated = False + else: + endurance_bytes = cap * 600 if cap else 0 + endurance_estimated = True + remaining_bytes = max(endurance_bytes - bw_total, 0) if endurance_bytes else 0 + endurance_tb = endurance_bytes / 1e12 if endurance_bytes else None + remaining_tb = remaining_bytes / 1e12 if remaining_bytes else 0 + + seconds_remaining = None + if rate > 0 and remaining_bytes > 0: + seconds_remaining = remaining_bytes / rate + + def _humanize(secs): + if secs is None or secs <= 0: + return "—" + years = int(secs // 31557600) + rem = secs % 31557600 + days = int(rem // 86400) + rem %= 86400 + hours = int(rem // 3600) + parts = [] + if years: + parts.append(f"{years} yr") + if days or years: + parts.append(f"{days} d") + parts.append(f"{hours} h") + return " ".join(parts) + + human = _humanize(seconds_remaining) + breakdown = { + "years": (seconds_remaining / 31557600) if seconds_remaining else 0, + "days": (seconds_remaining / 86400) if seconds_remaining else 0, + "hours": (seconds_remaining / 3600) if seconds_remaining else 0, + "human": human, + } + + wear_days = None + pts = [( _parse_ts(r["ts"]).timestamp(), r["percentage_used"]) for r in history if r.get("percentage_used") is not None and _parse_ts(r["ts"]) is not None] + if len(pts) >= 2: + win_pts = [p for p in pts if p[0] >= (latest_ts.timestamp() - window_sec)] if latest_ts else pts + if len(win_pts) < 2: + win_pts = pts + n = len(win_pts) + mean_x = sum(p[0] for p in win_pts) / n + mean_y = sum(p[1] for p in win_pts) / n + num = sum((p[0]-mean_x)*(p[1]-mean_y) for p in win_pts) + den = sum((p[0]-mean_x)**2 for p in win_pts) + if den != 0: + slope = num/den + if slope > 0: + last_pct = win_pts[-1][1] + secs_to_100 = (100 - last_pct) / slope + wear_days = secs_to_100 / 86400 + + preliminary = coverage_hours < 24 + notes = [] + if preliminary: + notes.append(f"Warming up — {coverage_hours:.1f}h of 24h") + if endurance_estimated: + notes.append("TBW estimated from capacity (pct=0)") + if rate <= 0: + notes.append("No writes in window") + if wear_days is None: + notes.append("Wear flat — write model only") + + return { + "window24h": {"bytes": bytes_in_window, "gb": bytes_in_window / 1e9, "coverage_hours": coverage_hours}, + "gb_per_hour": gb_per_hour, + "gb_per_day": gb_per_day, + "endurance_tb": endurance_tb, + "endurance_estimated": endurance_estimated, + "remaining_tb": remaining_tb, + "remaining_bytes": remaining_bytes, + "seconds_remaining": seconds_remaining, + "breakdown": breakdown, + "wear_model_days": wear_days, + "preliminary": preliminary, + "notes": notes, + "latest": latest, + } + + +def collector_loop(device, interval, stop_event): + while not stop_event.is_set(): + rec = sample(device) + if rec: + append_sample(rec) + update_hour_bucket(rec) + stop_event.wait(interval) + + +DASHBOARD_HTML = r""" + + + + +Fenris — NVMe Wear Dashboard + + + + + +
+
+
+
F
+
+
FENRIS by Bongbetic
+
loading…
+
+
+
+ ● Live + every — + next — +
+
+
+ +
+ + + +
+ +
+
+
Wear (% used) over time
+ +
+
+
Trailing 24h — GB written per hour
+ +
+
+
+ + + + + + +""" + + +class Handler(BaseHTTPRequestHandler): + def log_message(self, fmt, *args): + pass + + def do_GET(self): + path = self.path.split("?")[0] + if path == "/" or path.startswith("/index"): + body = DASHBOARD_HTML.encode() + self.send_response(200) + self.send_header("Content-Type", "text/html; charset=utf-8") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return + if path.startswith("/assets/"): + rel = path[len("/assets/"):] + if ".." in rel or rel.startswith("/"): + self.send_response(404); self.end_headers(); return + fp = os.path.join(ASSETS_DIR, rel) + if not os.path.isfile(fp): + self.send_response(404); self.end_headers(); return + ctype = mimetypes.guess_type(fp)[0] or "application/octet-stream" + with open(fp, "rb") as f: + body = f.read() + self.send_response(200) + self.send_header("Content-Type", ctype) + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return + if path == "/api/config": + body = json.dumps(_CONFIG).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return + if path == "/api/status": + hist = load_history() + latest = hist[-1] if hist else None + summ = compute_summary(hist) + body = json.dumps({ + "alive": True, + "samples": len(hist), + "last_ts": latest["ts"] if latest else None, + "version": VERSION, + "config": _CONFIG, + "summary": summ, + }).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return + if path.startswith("/api/data"): + hist = load_history() + body = json.dumps(hist).encode() + etag = f'"{hashlib.md5(body).hexdigest()}"' + inm = self.headers.get("If-None-Match") + if inm and inm == etag: + self.send_response(304); self.end_headers(); return + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("ETag", etag) + self.send_header("Cache-Control", "no-cache") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return + if path.startswith("/api/hourly"): + hourly = load_hourly() + body = json.dumps(hourly).encode() + etag = f'"{hashlib.md5(body).hexdigest()}"' + inm = self.headers.get("If-None-Match") + if inm and inm == etag: + self.send_response(304); self.end_headers(); return + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("ETag", etag) + self.send_header("Cache-Control", "no-cache") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return + if path.startswith("/api/summary"): + summ = compute_summary() + body = json.dumps(summ).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return + self.send_response(404) + self.end_headers() + + +def run_foreground(device, interval, port): + global _CONFIG, _current_hour + _CONFIG = {"interval": interval, "device": device, "port": port, "version": VERSION} + stop_event = threading.Event() + + def handle_sig(signum, frame): + stop_event.set() + + signal.signal(signal.SIGTERM, handle_sig) + signal.signal(signal.SIGINT, handle_sig) + + def cleanup_pid(): + try: + os.remove(PID_FILE) + except OSError: + pass + import atexit + atexit.register(cleanup_pid) + + print(f"Fenris starting — device={device} interval={interval}s port={port}") + print(f"Data log: {DATA_FILE}") + + rebuild_hourly_from_history() + + rec = sample(device) + if rec: + append_sample(rec) + update_hour_bucket(rec) + else: + print("WARNING: initial sample failed — check smartctl/sudo setup. " + "The daemon will keep retrying.", file=sys.stderr) + + t = threading.Thread(target=collector_loop, args=(device, interval, stop_event), daemon=True) + t.start() + + server = ThreadingHTTPServer(("0.0.0.0", port), Handler) + server.timeout = 1 + print(f"Dashboard: http://localhost:{port}") + + def server_loop(): + while not stop_event.is_set(): + server.handle_request() + + st = threading.Thread(target=server_loop, daemon=True) + st.start() + + while not stop_event.is_set(): + time.sleep(0.5) + print("Fenris stopping.") + + +def cmd_start(args): + if os.path.exists(PID_FILE): + with open(PID_FILE) as f: + pid = int(f.read().strip()) + if pid_alive(pid): + print(f"Fenris already running (pid {pid}). Use 'status' or 'stop'.") + return + os.remove(PID_FILE) + log_f = open(LOG_FILE, "a") + proc = subprocess.Popen( + [sys.executable, os.path.abspath(__file__), "run", + "--device", args.device, "--interval", str(args.interval), "--port", str(args.port)], + stdout=log_f, stderr=log_f, stdin=subprocess.DEVNULL, + start_new_session=True, + ) + with open(PID_FILE, "w") as f: + f.write(str(proc.pid)) + time.sleep(0.5) + print(f"Fenris started in background (pid {proc.pid}).") + print(f"Dashboard: http://localhost:{args.port}") + print(f"Logs: {LOG_FILE}") + + +def pid_alive(pid): + try: + os.kill(pid, 0) + return True + except OSError: + return False + + +def cmd_stop(args): + if not os.path.exists(PID_FILE): + print("Fenris is not running (no pid file).") + return + with open(PID_FILE) as f: + pid = int(f.read().strip()) + if pid_alive(pid): + os.kill(pid, signal.SIGTERM) + for _ in range(50): + if not pid_alive(pid): + break + time.sleep(0.1) + if pid_alive(pid): + print(f"SIGTERM did not stop pid {pid}, sending SIGKILL...") + os.kill(pid, signal.SIGKILL) + for _ in range(20): + if not pid_alive(pid): + break + time.sleep(0.1) + if pid_alive(pid): + print(f"ERROR: could not stop pid {pid}. Manual intervention needed.") + return + print(f"Stopped Fenris (pid {pid}).") + else: + print("Stale pid file — process was not running.") + try: + os.remove(PID_FILE) + except OSError: + pass + + +def cmd_status(args): + running = False + if os.path.exists(PID_FILE): + with open(PID_FILE) as f: + pid = int(f.read().strip()) + running = pid_alive(pid) + print(f"Fenris daemon: {'RUNNING (pid ' + str(pid) + ')' if running else 'not running (stale pid file)'}") + else: + print("Fenris daemon: not running") + rows = load_history() + if rows: + summ = compute_summary(rows) + latest = rows[-1] + print(f"Samples collected: {len(rows)}") + print(f"Last sample: {latest['ts']}") + print(f"Wear (percentage_used): {latest.get('percentage_used')}%") + print(f"Total written: {latest.get('bytes_written', 0) / 1e9:.1f} GB") + print(f"Written (24h rolling): {summ['window24h']['gb']:.2f} GB over {summ['window24h']['coverage_hours']:.1f}h") + print(f"Write rate: {summ['gb_per_hour']:.2f} GB/h ({summ['gb_per_day']:.1f} GB/day)") + if summ["seconds_remaining"]: + print(f"Projected life remaining: {summ['breakdown']['human']} (≈{summ['breakdown']['days']:.0f} days / {summ['breakdown']['hours']:.0f} hours / {summ['breakdown']['years']:.2f} years)") + print(f"Endurance: {summ['endurance_tb']:.1f} TB total, {summ['remaining_tb']:.1f} TB remaining" + (" (estimated)" if summ["endurance_estimated"] else "")) + if summ["wear_model_days"]: + print(f"Wear-model cross-check: ~{summ['wear_model_days']:.0f} days at current wear rate") + if summ["preliminary"]: + print("Note: preliminary — less than 24h coverage") + else: + print("Projected life remaining: — (no writes in window or no endurance data)") + hourly = load_hourly() + if hourly: + print(f"Hourly buckets: {len(hourly)} (last {hourly[-1]['hour']}: {hourly[-1]['bytes_written']/1e9:.2f} GB)") + else: + print("No samples collected yet.") + + +def cmd_run(args): + run_foreground(args.device, args.interval, args.port) + + +def cmd_sample_once(args): + rec = sample(args.device) + if rec: + append_sample(rec) + update_hour_bucket(rec) + print(json.dumps(rec, indent=2)) + summ = compute_summary() + print(f"\n24h: {summ['window24h']['gb']:.2f} GB rate {summ['gb_per_hour']:.2f} GB/h remaining {summ['breakdown']['human']}") + else: + sys.exit(1) + + +def main(): + p = argparse.ArgumentParser(description="Fenris — NVMe wear monitor & dashboard (by Bongbetic)") + sub = p.add_subparsers(dest="cmd", required=True) + + def add_common(sp): + sp.add_argument("--device", default=detect_device(), help="NVMe device, e.g. /dev/nvme0") + sp.add_argument("--interval", type=int, default=300, help="seconds between samples (default 300)") + sp.add_argument("--port", type=int, default=8420, help="dashboard HTTP port (default 8420)") + + sp = sub.add_parser("start", help="start monitoring in background") + add_common(sp); sp.set_defaults(func=cmd_start) + sp = sub.add_parser("stop", help="stop background monitoring") + sp.set_defaults(func=cmd_stop) + sp = sub.add_parser("status", help="show daemon + latest wear stats") + sp.set_defaults(func=cmd_status) + sp = sub.add_parser("run", help="run in foreground (used internally by 'start')") + add_common(sp); sp.set_defaults(func=cmd_run) + sp = sub.add_parser("sample", help="take one sample immediately and print it") + sp.add_argument("--device", default=detect_device()) + sp.set_defaults(func=cmd_sample_once) + + args = p.parse_args() + args.func(args) + + +if __name__ == "__main__": + main() diff --git a/fenris.sh b/fenris.sh new file mode 100755 index 0000000..82c9bd0 --- /dev/null +++ b/fenris.sh @@ -0,0 +1,128 @@ +#!/usr/bin/env bash +# Fenris — interactive menu for the NVMe wear monitor & dashboard. +# Created by Bongbetic. + +set -euo pipefail +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +PY="$SCRIPT_DIR/fenris.py" +PORT_DEFAULT=8420 +INTERVAL_DEFAULT=300 + +banner() { + cat <<'EOF' + _____ _ +| __|___ ___ _| |___ +| __| -_| | . | _| +|__| |___|_|_|_|___|_| + + NVMe wear monitor & live dashboard + Created by Bongbetic +EOF +} + +pause() { read -rp "Press Enter to continue..." _; } + +detect_device() { + # Query Python's auto-detect for the default device. + python3 -c "import sys; sys.path.insert(0,'$SCRIPT_DIR'); from fenris import detect_device; print(detect_device())" 2>/dev/null || echo /dev/nvme0 +} + +menu() { + clear + banner + echo + echo " 1) Start monitoring (background daemon + dashboard)" + echo " 2) Stop monitoring" + echo " 3) Status / current wear stats" + echo " 4) Take one sample right now" + echo " 5) Open dashboard URL" + echo " ---" + echo " h) Help / how this works" + echo " q) Exit" + echo + read -rp "Choose an option: " choice + echo + case "$choice" in + 1) start_flow ;; + 2) python3 "$PY" stop; pause ;; + 3) python3 "$PY" status; pause ;; + 4) read -rp "Device [default: auto-detect]: " dev + if [ -z "$dev" ]; then python3 "$PY" sample; else python3 "$PY" sample --device "$dev"; fi + pause ;; + 5) show_url; pause ;; + h|H) help_text; pause ;; + q|Q) echo "Bye. — Fenris, by Bongbetic"; exit 0 ;; + *) echo "Invalid choice."; pause ;; + esac +} + +start_flow() { + read -rp "NVMe device [Enter = auto-detect]: " dev + read -rp "Sample interval in seconds [Enter = ${INTERVAL_DEFAULT}]: " interval + read -rp "Dashboard port [Enter = ${PORT_DEFAULT}]: " port + interval="${interval:-$INTERVAL_DEFAULT}" + port="${port:-$PORT_DEFAULT}" + + args=(start --interval "$interval" --port "$port") + if [ -n "${dev:-}" ]; then args+=(--device "$dev"); fi + + echo + echo "Note: reading NVMe SMART data needs root." + echo "Fenris runs 'sudo -n smartctl ...' (no-prompt sudo). If this fails," + echo "either run this menu with sudo, or allow passwordless smartctl via:" + echo " sudo visudo -> youruser ALL=(root) NOPASSWD: /usr/sbin/smartctl" + echo + + python3 "$PY" "${args[@]}" + pause +} + +show_url() { + if [ -f "$SCRIPT_DIR/data/fenris.pid" ]; then + # Try to read actual port from running process cmdline, else guess default. + local pid port + pid=$(<"$SCRIPT_DIR/data/fenris.pid") + port=$(tr '\0' '\n' < /proc/"$pid"/cmdline 2>/dev/null | grep -A1 -- '--port' | tail -1 || true) + port="${port:-$PORT_DEFAULT}" + echo "Dashboard: http://localhost:${port}" + else + echo "Fenris is not currently running. Start it first (option 1)." + fi +} + +help_text() { + cat < Status: PLAN ONLY — no code touched. **Option A approved.** Data-management removed; cumulative write chart X = hours-in-day (0–24); dense layout; hourly diagnostics + real-time forecast model. + +## 1. Context & Goals + +**User request (consolidated):** +- Dashboard live/dynamic, updating at each poll interval without refresh — **Option A (zero-build).** +- Bongbetic branding + logo everywhere (`~/Documents/bongbetic/Logo`). +- shadcn/ui styling (CSS parity, no React build). +- **Remove data-management feature** (backup/restore/backups/repair/purge). +- **Cumulative data-written graph X-axis = hours in a day (0–24).** +- **Graphs smaller, side-by-side, flexible.** **Cards tighter/dense.** +- **Estimated life remaining = real-time, based on live hourly GB usage.** After 24h of run, app must keep recording hourly, store diagnostics, log GB/hour, and forecast remaining days from current hourly rate. + +**Current state (verified 2026-08-19):** +- `fenris.py` single `DASHBOARD_HTML` via `ThreadingHTTPServer`. Hand-rolled dark CSS, 2× `` stacked vertically (`height=220`, full-width), `setInterval(render,30000)` hardcoded, full `innerHTML` replace. +- `GET /api/data` → full `history.jsonl`; no `/api/config`; interval not exposed. +- Forecast: `linearForecast()` on `percentage_used` trend (simple linear regression, whole history). No hourly bucket, no GB/hour model, no diagnostics persistence beyond raw history. +- Cards: `minmax(220px,1fr)` gap 1rem, padding `1rem 1.2rem`, value `1.7rem` — loose. +- No asset vendoring, no Tailwind/shadcn. + +**Goals:** +1. Interval-synced live refresh (no reload), ETag/diff, pulse. +2. Bongbetic header/logo/footer + shadcn tokens. +3. Dense cards + two graphs side-by-side, responsive, reduced height. +4. After 24h: hourly diagnostics log (`GB/hour`), real-time days-remaining forecast from live hourly write rate (not just wear %). + +## 2. Non-Goals & Explicit Removals + +- No auth/multi-tenant, no SSE/WebSocket v1, no Vite/React (Option B rejected). +- **Data-management removed:** delete `BACKUP_DIR`, 5 commands, parsers, `data/backups/`, `fenris.sh` items 6–10. Data = append-only `data/history.jsonl` (+ new `data/hourly.jsonl` per §5.2); orphan `backups/` logs “safe to delete”. + +## 3. Logo / Brand Audit + +``` +bongbetic-logo-dark.svg (751×220, #F7F1E7 currentColor) — dark bg +bongbetic-symbol-dark.svg (192×192) — favicon/header mark +bongbetic-brand/favicon-32.png, icon-dark-512.png, site.webmanifest, b_glyph.svg +``` + +Vendor into `assets/` and serve via `Handler` static branch; inline SVG so `currentColor` follows `text-foreground`. + +## 4. Target Architecture — Option A (Approved) + +Zero-build: Tailwind CDN + shadcn CSS variables (Slate/Zinc dark). Semantic HTML + Tailwind mimics `Card/Badge/Progress/Alert/Skeleton`. Canvas charts wrapped in `Card`; wear + write side-by-side flex grid (see §6.1). Python exposes `/api/config`, `/api/status`, `/api/hourly`, static `/assets/*`. + +## 5. Live / Dynamic Behavior + +### 5.0 Polling (unchanged from prior plan) +- `GET /api/config → {interval, device, port, version}` → `POLL_MS = interval*1000` (clamp 5s–3600s, 30s fallback). +- Diff: hash `last_ts+length`; skip render if unchanged; else patch cards (textContent morph + `ring-2` pulse), append chart point via `requestAnimationFrame`. +- `visibilitychange` pause/resume, `navigator.onLine` backoff 1/2/4…60s, header badge “Live • every 5m • next in 03:42”. +- `GET /api/data` gains `ETag` (mtime+length) + 304. + +### 5.1 Cumulative Write Chart — Hours-in-Day (Requirement) + +- **Intraday 0–24h view:** `hours = h + m/60 + s/3600` (0–23.99). `todayRows = rows.filter(same DateString)` → `[hours, (bytes_written - midnightBaseline)/1e9]`. +- Axis: `minX=0, maxX=24`, ticks `00:00/06:00/12:00/18:00/24:00` via `opts.xIsHoursInDay`. Title “Cumulative written today (GB) — hours of day”. Empty → “No samples today yet”. Midnight → reset to 0 GB, caption “Resets at midnight — today only”. +- Wear chart stays absolute-time (weeks trend). Both canvases slimmed (see §6.1). + +### 5.2 Real-Time Hourly Diagnostics & Forecast Model (New — Core Requirement) + +**Problem with current forecast:** `linearForecast(points)` regresses `percentage_used` over full history; single slope, insensitive to bursty hourly writes, no GB/hour visibility. + +**Required model:** +- Real-time on every poll, not daily batch. +- After 24h wall time since first sample, switch to **hourly GB/hour regime**; before 24h, show warming-up estimate. +- Persist per-hour diagnostics and GB/hour log. + +**Design:** + +1. **Raw source stays `data/history.jsonl`** (poll interval samples, e.g., every 300s). + +2. **New derived log `data/hourly.jsonl`** (append-only, one record per wall hour): + ```json + {"hour":"2026-08-19T14:00:00Z","samples":12,"gb_written":4.21,"gb_read":1.03, + "pct_start":1.2,"pct_end":1.21,"pct_delta":0.01, + "temp_avg":42.1,"temp_max":48,"spare":98,"media_errors":0} + ``` + Fields: hour bucket start (UTC), count, deltas from first/last sample in hour, averages. Written by `collector_loop` helper `flush_hourly()`. + +3. **Hourly rollup logic (Python):** + - In-memory `current_hour_bucket`; on each `sample()`, accumulate `bytes_written` delta vs bucket start. + - On hour boundary (or every 60min since daemon start if clock not trusted), `append_hourly(rec);` also `append_sample(raw)`. + - On daemon (re)start, rebuild missing hours by scanning `history.jsonl` and aggregating by `hour = ts truncated to hour` (idempotent — dedupe by hour string). + - `/api/hourly` returns parsed `hourly.jsonl` (array, sorted). Add ETag similarly. + +4. **Forecast — two complementary signals, UI shows primary “days remaining (hourly write model)” + secondary wear model:** + - **GB/hour → TBW model (primary, per user ask “present hourly usage → days”):** + ``` + hourly_avg = mean(last 24 hourly gb_written) // rolling 24, or EWMA α=0.3 if <24h + // derive endurance from vendor wear if available: + if pct_used>0: endurance_TB = bytes_written / (pct_used/100) // total TBW implied + else: endurance_TB = capacity_bytes * 600 // fallback: ~600× capacity (conservative), or mark unknown + remaining_TB = max(endurance_TB - bytes_written/1e12, 0) + days_remaining = remaining_TB / (hourly_avg *24) // hourly_avg in TB + ``` + If endurance derivable, show; else show wear-based only and badge “TBW unknown — using wear rate”. + - **Wear/hour model (secondary, cross-check):** + ``` + wear_per_hour = mean(last 24 pct_delta per hour) + hours_to_100 = (100 - pct_now) / wear_per_hour + days_wear = hours_to_100/24 + ``` + Shown as tooltip / small “also ~X days at current wear rate”. + + - **<24h warming up:** `hourly_avg` over available hours (n<24); badge “Warming up — Xh to confident forecast (now ~Y days, n=Nh)”. `days_remaining` still computed but flagged `preliminary`. + - **Real-time update:** every poll, frontend refetches `/api/data` + `/api/hourly`, recomputes hourly_avg client-side too (so UI reflects instantly even before next hourly flush); backend hourly file ensures persistence across restarts. + +5. **Storage & retention:** `hourly.jsonl` append 24 records/day → ~9k/year, trivial. Keep forever; same manual-truncate philosophy (no purge command). Document `jq` one-liner to trim. + +6. **UI integration:** new card “Est. days remaining (live hourly)” with large `N days` + sub `3.2 GB/hour avg (24h) • 1.1 TB remaining • updates each 5m`; secondary line wear model. New mini sparkline/bar inside card showing last 24h `gb_written` per hour (or when `<24h`, show available). Wear chart tooltip cross-links. + +**API additions:** +```python +GET /api/config → {interval, device, port, version} +GET /api/status → {alive, samples, last_ts, pid, uptime_hours, hourly_samples} +GET /api/hourly → [hourlyRec, ...] # ETag + no-store +GET /api/data → unchanged + ETag +# no backup/restore +``` + +## 6. Layout & shadcn Mapping (Dense + Side-by-Side) + +### 6.1 Dense Cards + Flexible Graphs + +- **Cards:** tighter — Tailwind: `grid gap-3` (was 1rem), `grid-cols-2 md:grid-cols-3 xl:grid-cols-5`, card `p-3` (was 1rem 1.2rem), `rounded-lg` (was 10px), label `text-[0.70rem] tracking-wide`, value `text-xl font-semibold` (was 1.7rem), sub `text-xs`. Row height uniform via `min-h-[96px]`. +- **Graphs:** `section.grid.grid-cols-1.lg:grid-cols-2.gap-4` — two `Card`s side-by-side on ≥1024px, stacked below. Each Card: `p-4`, header `CardTitle` 0.9rem, canvas `h-[200px] lg:h-[220px] w-full` (down from 220 full-width), `flex-1 min-w-0` so canvases shrink. `drawLine` canvas `width=clientWidth, height=200` → responsive. No fixed page width; container `max-w-[1400px] mx-auto px-4`. +- **Flex guarantees:** `canvas { width:100%; height:100%; display:block }`, `chart-wrap { flex:1 min-w-0 }`, `canvas` DPR scaled but CSS size flexes. ResizeObserver re-draw on container resize. +- **Overall page:** less vertical scroll — header sticky + dense cards + two graphs in one row + footer. + +| Current | shadcn | Tight classes | +|---|---|---| +| `.grid .card` | `Card` | `rounded-lg border bg-card p-3 shadow-sm gap-3` | +| value | `CardTitle` numeric | `text-xl font-semibold tabular-nums` | +| wear % | `Progress` | `h-1.5 bg-primary` | +| Spare/errors | `Badge` + `Alert` | `text-xs px-1.5 py-0` | +| charts | `Card` flex | `h-[200px] lg:h-[220px] p-4` | +| footer | muted | `text-xs text-muted-foreground` | + +## 7. Branding Spec + +Header sticky `[symbol 28px | wordmark] Fenris — NVMe Wear Monitor` left, `[Live ● next 03:42] [every 5m]` right, `bg-background/80 backdrop-blur`. Favicon `favicon-32.png`, `icon-dark-512`. Footer “© Bongbetic — Fenris · interval 300s · v0.1” + `b_glyph.svg` 16px + link to diagnostics count. No `data/backups`. + +## 8. File / Code Changes (Option A) + +**`fenris.py`:** +- Delete data-mgmt: `BACKUP_DIR`, 5 cmds, parsers; keep `start/stop/status/run/sample`. +- Add globals `_interval`, `_device`, `_port`; vendored `assets/` static branch. +- New `data/hourly.jsonl` + helpers `append_hourly()`, `load_hourly()`, `flush_hourly()`, `rebuild_hourly_from_history()`. +- `collector_loop`: on each sample also `update_hour_bucket`; hourly flush. +- Endpoints: `/api/config`, `/api/status`, `/api/hourly`, ETag on `/api/data`+`/api/hourly`, static `/assets/*`. +- Replace `DASHBOARD_HTML`: Tailwind CDN + shadcn vars, sticky header logos, dense grid, side-by-side Cards (200px canvases, flex), hours-in-day write chart (§5.1), live hourly forecast card (§5.2) with `hourly_avg` + `days_remaining` + 24h bar sparkline, midnight caption, footer glyph. + +**`fenris.sh`:** remove items 6–10 branches, renumber Help/Exit, update `help_text`. + +**`README.md`:** drop Data Management; add “Hourly diagnostics `data/hourly.jsonl` (GB/hour) + live days-remaining forecast after 24h” + `jq` truncate note. + +**Assets:** +```bash +mkdir -p assets +cp ~/Documents/bongbetic/Logo/bongbetic-logo-dark.svg assets/ +cp ~/Documents/bongbetic/Logo/bongbetic-symbol-dark.svg assets/ +cp ~/Documents/bongbetic/Logo/bongbetic-brand/favicon-32.png assets/favicon.ico +cp ~/Documents/bongbetic/Logo/bongbetic-brand/icon-dark-512.png assets/ +``` + +Keep `collector_loop` otherwise unchanged; no Vite. + +## 9. Implementation Phases (Option A) + +**Phase 0 — Approved** (this plan): Option A + removals + hours-in-day + dense layout + hourly model. + +**Phase 1 — Deletion + API + hourly store:** delete data-mgmt, add config/status/hourly endpoints, hourly rollup helpers + rebuild, verify `curl /api/hourly | jq`. + +**Phase 2 — Shell/layout:** dense cards, side-by-side flex graphs (200px), Tailwind CDN + shadcn vars, header/footer logos. + +**Phase 3 — Live JS + charts:** interval-synced poll, diff patch, writeChart 0–24h (§5.1), wear chart, visibility/backoff, ETag 304. + +**Phase 4 — Hourly forecast live:** `hourly_avg` rolling 24h, TBW-derived `days_remaining` + wear cross-check, warming-up badge <24h, per-hour GB bar in forecast card, midnight reset; wire `GET /api/hourly`. + +**Phase 5 — Polish:** `Progress/Badge` variants, `Skeleton`, temp alert, responsive, `prefers-color-scheme`, ResizeObserver. + +## 10. Risks + +| Risk | Mitigation | +|---|---| +| Tailwind CDN offline | Vendor `assets/tailwind.css` offline fallback. | +| Logo currentColor | Inline SVG, test Chrome/Firefox. | +| Interval desync | Python global single source. | +| Flicker | `requestAnimationFrame` + keep rows. | +| History large | `hourly.jsonl` is compact; keep `/api/data` full v1, add `?since=` v2. | +| Data-mgmt removal confusion | `--help` no backup/restore; log orphan `backups/`. | +| Midnight reset | Caption “Resets at midnight — today only”. | +| <24h forecast noisy | Flag “preliminary (n=Xh)”, EWMA; show both GB and wear models. | +| TBW derivation unstable when pct=0 | Fallback to wear model only; badge “TBW unknown”. | +| Side-by-side overflow | `min-w-0 flex-1` + `grid-cols-1 lg:grid-cols-2` + 200px height; test 1024/1440. | + +## 11. Verification + +- [ ] `python3 -m py_compile fenris.py` / `bash -n fenris.sh` pass +- [ ] `python3 fenris.py backup` → unknown command; menu 1–5 only +- [ ] `python3 fenris.py start --interval 10` → badge `every 10s` +- [ ] `/api/config`, `/api/hourly`, `/api/data` 200; 304 when unchanged; `curl /assets/bongbetic-logo-dark.svg` svg+xml +- [ ] Cards tight: 5/col xl, p-3, no overflow; graphs side-by-side lg, stacked sm, each ~200px, flex resize no clipping +- [ ] writeChart X `00:00 06:00 12:00 18:00 24:00` today-only; midnight resets to 0 +- [ ] Kill/restart daemon → `/api/hourly` rebuilt from history, no dupe hours +- [ ] `<24h`: forecast badge “Warming up (n=Xh) ~Y days (preliminary)” +- [ ] `≥24h`: let run 24h (or fake hourly file with 24 records) → card shows `hourly_avg` over 24, `days_remaining` updates each poll (change write load → forecast moves within one interval); bar of 24h gb/hour visible +- [ ] Tab hidden → pause, visible → immediate fetch; stale banner > interval*2 +- [ ] `prefers-color-scheme` legible + +## 12. Out of Scope + +- Vite/React build, SSE, auth, purge — removed intentionally. + +--- +*Option A — dense side-by-side + hourly live forecast — ready to implement.*