fix(dashboard): harden generation against common failures

This commit is contained in:
2026-08-24 11:10:39 +02:00
parent 12a0eca44a
commit f5b1ade48c
3 changed files with 263 additions and 418 deletions
+71 -13
View File
@@ -28,11 +28,25 @@ import os
import re
import sys
import tempfile
import traceback
from datetime import datetime
from pathlib import Path
import xml.etree.ElementTree as ET
ROOT = Path(__file__).resolve().parent.parent
LOG_FILE = ROOT / "tools" / "plot_progress.log"
def log(msg):
"""Timestamped line to stderr AND tools/plot_progress.log (survives reboots;
the watcher has no terminal to read errors from)."""
line = f"{datetime.now():%F %T} {msg}"
print(line)
try:
with open(LOG_FILE, "a") as fh:
fh.write(line + "\n")
except OSError:
pass
EVAL_RE = re.compile(r">>> \[eval\] win rate: (\d+)/(\d+) \(([\d.]+)%\) vs (\S+)")
EVAL_BLOCK_RE = re.compile(r">>> \[eval\] \d+ deterministic rounds vs (\S+)")
ROUND_RE = re.compile(r"Round \d+/\d+\D+ticks:\d+ score:(\d+) win:")
@@ -235,8 +249,8 @@ def write_svg(path, text):
except ET.ParseError as e:
print(f"[error] {path.name}: generated SVG invalid, keeping old file ({e})")
return False
tmp = path.with_name(path.name + ".tmp")
tmp.write_text(text)
tmp = path.with_name(f"{path.name}.{os.getpid()}.tmp") # unique: a manual
tmp.write_text(text) # run + watcher can race
os.replace(tmp, path)
return True
@@ -302,7 +316,8 @@ def legend(items, x, y):
def map_fn(p0, p1, vmin, vmax, log=False):
def f(v):
t = (math.log10(v) - vmin) / (vmax - vmin) if log else (v - vmin) / (vmax - vmin)
span = (vmax - vmin) or 1 # single-point series -> degenerate range
t = (math.log10(v) - vmin) / span if log else (v - vmin) / span
return p0 + max(0.0, min(1.0, t)) * (p1 - p0)
return f
@@ -354,7 +369,7 @@ def panel_test_matches(series, geo):
if not series:
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
"no eval lines found</text>\n"
return
return s
xmax = max(max(len(v) for v in series.values()), 2)
xm, ym = map_fn(x0, x1, 1, xmax), map_fn(pb, pt, 0, 100)
s += hgrid(x0, x1, [ym(v) for v in range(0, 101, 20)])
@@ -381,7 +396,7 @@ def panel_max_score(maxes, geo):
if not maxes:
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
"no eval lines found</text>\n"
return
return s
ncyc = max(len(v) for v in maxes.values())
xmax = max(ncyc, 2)
hi = max(1, max(v for vals in maxes.values() for v in vals)) * 1.1
@@ -421,11 +436,18 @@ def panel_losses(rows, geo):
if not (critic or actor or alpha):
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
"no valid loss points</text>\n"
return
return s
x1 -= 46 # room on the right for the twin alpha axis labels
n = len(rows)
xm = lambda i: x0 + (x1 - x0) * i / max(n - 1, 1)
base = critic + actor or alpha # log-domain source (losses in practice)
# log-domain source (losses in practice); drop non-positive values so a
# run of alpha==0 rows can't raise math domain error and kill the build
base = [(i, v) for i, v in critic + actor if v > 0] or \
[(i, v) for i, v in alpha if v > 0]
if not base:
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
"no valid loss points</text>\n"
return s
lo = math.floor(math.log10(min(v for _, v in base)))
hi = math.ceil(math.log10(max(v for _, v in base)))
if lo == hi:
@@ -473,7 +495,7 @@ def panel_throughput(rows, geo):
if not rates:
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
"no usable epoch timestamps</text>\n"
return
return s
bm = bucket_means(rates, RATE_BUCKET)
xm = map_fn(x0, x1, 1, len(rates))
ymax = max(max(rates), max(bm)) * 1.1
@@ -528,7 +550,13 @@ def build_dashboard(campaign, metrics_f, out):
for geo, title, guide, sub, drawer in drawers:
s += header(geo[3], geo[0], title, sub)
s += panel_guide(geo[3], geo[0] + 33, guide)
s += drawer() or "" # panels bare-return None on their no-data path
try:
body = drawer()
except Exception as e: # one bad panel must not kill the whole page
log(f"[warn] panel '{title}' failed, rendering placeholder: {e}")
body = (f'<text x="{geo[3] + 10}" y="{geo[1] + 40}" font-size="12" '
f'fill="#a00">panel error: {esc(e)}</text>')
s += body or "" # panels bare-return None on their no-data path
s += signs_block()
s += RELOAD_JS + "\n"
@@ -625,6 +653,32 @@ def selftest():
(xa, ya), (xb, yb) = pts
assert yb < ya, f"y-axis inverted: win % rose 10->90 but ink moved down ({ya} -> {yb})"
assert xb > xa, f"x-axis reversed: newer eval plotted left ({xa} -> {xb})"
# resilience: missing/empty inputs render placeholders, never crash
empty_dash = td / "empty.svg"
assert build_dashboard(td / "nope.log", td / "nope.jsonl", empty_dash)
etext = empty_dash.read_text()
assert etext.count("no eval lines found") == 2, etext.count("no eval lines found")
assert "no valid loss points" in etext
assert "no usable epoch timestamps" in etext
# all-zero alpha rows (post-crash trainer state) must not raise in the
# losses panel's log-domain math -> placeholder instead of dead build
(td / "zero.jsonl").write_text(
'{"epoch": 1000.0, "alpha": 0}\n{"epoch": 1060.0, "alpha": 0}\n')
zero_dash = td / "zero.svg"
assert build_dashboard(td / "log", td / "zero.jsonl", zero_dash)
assert "no valid loss points" in zero_dash.read_text()
# single throughput interval (fresh 2-row metrics file right after a
# restart) used to divide by zero in map_fn and kill the whole build
(td / "one.jsonl").write_text(
'{"epoch": 1000.0, "critic_loss": 10, "alpha": 0.5}\n'
'{"epoch": 1060.0, "critic_loss": 5, "alpha": 0.4}\n')
one_dash = td / "one.svg"
assert build_dashboard(td / "log", td / "one.jsonl", one_dash)
assert "panel error" not in one_dash.read_text()
assert "<polyline" in one_dash.read_text()
print("selftest OK")
@@ -636,17 +690,21 @@ def main():
campaign = Path(args[0]) if len(args) > 0 else ROOT / "campaign_v4_stdout.log"
metrics = Path(args[1]) if len(args) > 1 else ROOT / "training_metrics.jsonl"
outdir = Path(args[2]) if len(args) > 2 else ROOT / "docs"
outdir.mkdir(parents=True, exist_ok=True)
try:
outdir.mkdir(parents=True, exist_ok=True)
for p in (campaign, metrics):
if not p.is_file():
# loudest symptom of a watcher launched from a stale checkout
log(f"[warn] input missing: {p} - is this the live checkout?")
ok = build_dashboard(campaign, metrics, outdir / "campaign_dashboard.svg")
except Exception as e:
print(f"[error] dashboard build failed: {e}")
except Exception:
log(f"[error] dashboard build failed:\n{traceback.format_exc().rstrip()}")
ok = False
if ok:
print(f"[done] dashboard written to {outdir / 'campaign_dashboard.svg'}")
else:
print("[error] dashboard NOT updated - check paths above")
print("[error] dashboard NOT updated - see " + str(LOG_FILE))
sys.exit(1)
+16 -3
View File
@@ -1,10 +1,23 @@
#!/bin/sh
# Keep docs/campaign_dashboard.svg fresh: regenerate every 60 s.
# Errors go to stderr and never exit the loop silently.
dir=$(dirname "$0")
# Failures are appended to tools/plot_progress.log (by this wrapper and by
# plot_progress.py itself) - check that file when the dashboard looks stale.
# The startup banner records WHICH tree this instance watches, so a copy
# launched from a stale checkout (the post-reboot failure mode) is visible.
set -u
dir=$(cd "$(dirname "$0")" && pwd)
log="$dir/plot_progress.log"
if command -v flock >/dev/null 2>&1; then
exec 9>"$dir/.watch_dashboard.lock"
flock -n 9 || { echo "[watch_dashboard] another instance already running, exiting" >&2; exit 0; }
fi
echo "[watch_dashboard] $(date '+%F %T') started, watching root=$(cd "$dir/.." && pwd)" >> "$log" 2>/dev/null || true
while :; do
if ! python3 "$dir/plot_progress.py"; then
echo "[watch_dashboard] $(date '+%F %T') regeneration failed (see error above)" >&2
echo "[watch_dashboard] $(date '+%F %T') regeneration failed (see $log)" >&2
echo "[watch_dashboard] $(date '+%F %T') regeneration failed" >> "$log" 2>/dev/null || true
fi
sleep 60
done