diff --git a/SAC_LSTM_Bot/docs/campaign_dashboard.svg b/SAC_LSTM_Bot/docs/campaign_dashboard.svg index 6b7a84d..66993f4 100644 --- a/SAC_LSTM_Bot/docs/campaign_dashboard.svg +++ b/SAC_LSTM_Bot/docs/campaign_dashboard.svg @@ -1,7 +1,7 @@ SAC-LSTM campaign dashboard - live run (current only) -generated 2026-08-24 10:18:35 - auto-reloads every 60 s (open this file in Chrome) +generated 2026-08-24 11:10:31 - auto-reloads every 60 s (open this file in Chrome) Test matches - win % vs opponents raw dots = single test matches, thick = rolling-mean-10 What: How often the bot wins against each opponent in test battles. @@ -17,74 +17,14 @@ 1 - -2 - -3 - -4 - -5 - -6 - -7 - -8 - -9 - -10 - -11 - -12 - -13 - -14 - -15 - -16 - -17 + +2 -18 - -19 - -20 - -21 - -22 - -23 - -24 - -25 - -26 - -27 - -28 - -29 - -30 - -31 - -32 - -33 - -34 +3 + +4 -35 +5 0 @@ -100,119 +40,29 @@ test match number (each opponent) win rate (%) - - - - - - - - - - - - - - - - + - - - - - - - - - - - - - - - - + - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + - + - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + - + -Corners - 35 evals +Corners - 5 evals -Crazy - 35 evals +Crazy - 5 evals -Target - 35 evals +Target - 5 evals Training losses (log) & alpha (linear) training_metrics.jsonl - big early spikes are normal What: How well the brain is learning: losses should fall; alpha sets explore/exploit. @@ -228,13 +78,13 @@ 1 -19 +24 -36 +46 -54 +69 -72 +92 0.01 @@ -247,8 +97,8 @@ 100 metric line number loss (log) - - + + @@ -266,7 +116,7 @@ 1 alpha - + critic_loss @@ -285,26 +135,26 @@ - -7 - -14 - -21 - -28 - -35 - -42 - -49 - -56 - -63 - -70 + +9 + +18 + +27 + +36 + +45 + +54 + +63 + +72 + +81 + +90 0 @@ -318,77 +168,97 @@ chunk interval (10-game chunks) games / hour - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Max score per eval cycle campaign_v4_stdout.log - max of the 10 deterministic round scores per eval What: Best single-round score the bot managed in each test cycle. @@ -401,149 +271,53 @@ - -4 - -8 - -12 - -16 - -20 - -24 - -28 - -32 + +1 + +2 + +3 + +4 + +5 0 -52 +46 -103 +92 -155 +138 -207 +184 eval cycle number best single-round score - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + -Corners - 35 evals +Corners - 5 evals -Crazy - 35 evals +Crazy - 5 evals -Target - 35 evals +Target - 5 evals combined max diff --git a/SAC_LSTM_Bot/tools/plot_progress.py b/SAC_LSTM_Bot/tools/plot_progress.py index 32710e6..46fcda9 100644 --- a/SAC_LSTM_Bot/tools/plot_progress.py +++ b/SAC_LSTM_Bot/tools/plot_progress.py @@ -28,11 +28,25 @@ import os import re import sys import tempfile +import traceback from datetime import datetime from pathlib import Path import xml.etree.ElementTree as ET ROOT = Path(__file__).resolve().parent.parent +LOG_FILE = ROOT / "tools" / "plot_progress.log" + + +def log(msg): + """Timestamped line to stderr AND tools/plot_progress.log (survives reboots; + the watcher has no terminal to read errors from).""" + line = f"{datetime.now():%F %T} {msg}" + print(line) + try: + with open(LOG_FILE, "a") as fh: + fh.write(line + "\n") + except OSError: + pass EVAL_RE = re.compile(r">>> \[eval\] win rate: (\d+)/(\d+) \(([\d.]+)%\) vs (\S+)") EVAL_BLOCK_RE = re.compile(r">>> \[eval\] \d+ deterministic rounds vs (\S+)") ROUND_RE = re.compile(r"Round \d+/\d+\D+ticks:\d+ score:(\d+) win:") @@ -235,8 +249,8 @@ def write_svg(path, text): except ET.ParseError as e: print(f"[error] {path.name}: generated SVG invalid, keeping old file ({e})") return False - tmp = path.with_name(path.name + ".tmp") - tmp.write_text(text) + tmp = path.with_name(f"{path.name}.{os.getpid()}.tmp") # unique: a manual + tmp.write_text(text) # run + watcher can race os.replace(tmp, path) return True @@ -302,7 +316,8 @@ def legend(items, x, y): def map_fn(p0, p1, vmin, vmax, log=False): def f(v): - t = (math.log10(v) - vmin) / (vmax - vmin) if log else (v - vmin) / (vmax - vmin) + span = (vmax - vmin) or 1 # single-point series -> degenerate range + t = (math.log10(v) - vmin) / span if log else (v - vmin) / span return p0 + max(0.0, min(1.0, t)) * (p1 - p0) return f @@ -354,7 +369,7 @@ def panel_test_matches(series, geo): if not series: s += f'' \ "no eval lines found\n" - return + return s xmax = max(max(len(v) for v in series.values()), 2) xm, ym = map_fn(x0, x1, 1, xmax), map_fn(pb, pt, 0, 100) s += hgrid(x0, x1, [ym(v) for v in range(0, 101, 20)]) @@ -381,7 +396,7 @@ def panel_max_score(maxes, geo): if not maxes: s += f'' \ "no eval lines found\n" - return + return s ncyc = max(len(v) for v in maxes.values()) xmax = max(ncyc, 2) hi = max(1, max(v for vals in maxes.values() for v in vals)) * 1.1 @@ -421,11 +436,18 @@ def panel_losses(rows, geo): if not (critic or actor or alpha): s += f'' \ "no valid loss points\n" - return + return s x1 -= 46 # room on the right for the twin alpha axis labels n = len(rows) xm = lambda i: x0 + (x1 - x0) * i / max(n - 1, 1) - base = critic + actor or alpha # log-domain source (losses in practice) + # log-domain source (losses in practice); drop non-positive values so a + # run of alpha==0 rows can't raise math domain error and kill the build + base = [(i, v) for i, v in critic + actor if v > 0] or \ + [(i, v) for i, v in alpha if v > 0] + if not base: + s += f'' \ + "no valid loss points\n" + return s lo = math.floor(math.log10(min(v for _, v in base))) hi = math.ceil(math.log10(max(v for _, v in base))) if lo == hi: @@ -473,7 +495,7 @@ def panel_throughput(rows, geo): if not rates: s += f'' \ "no usable epoch timestamps\n" - return + return s bm = bucket_means(rates, RATE_BUCKET) xm = map_fn(x0, x1, 1, len(rates)) ymax = max(max(rates), max(bm)) * 1.1 @@ -528,7 +550,13 @@ def build_dashboard(campaign, metrics_f, out): for geo, title, guide, sub, drawer in drawers: s += header(geo[3], geo[0], title, sub) s += panel_guide(geo[3], geo[0] + 33, guide) - s += drawer() or "" # panels bare-return None on their no-data path + try: + body = drawer() + except Exception as e: # one bad panel must not kill the whole page + log(f"[warn] panel '{title}' failed, rendering placeholder: {e}") + body = (f'panel error: {esc(e)}') + s += body or "" # panels bare-return None on their no-data path s += signs_block() s += RELOAD_JS + "\n" @@ -625,6 +653,32 @@ def selftest(): (xa, ya), (xb, yb) = pts assert yb < ya, f"y-axis inverted: win % rose 10->90 but ink moved down ({ya} -> {yb})" assert xb > xa, f"x-axis reversed: newer eval plotted left ({xa} -> {xb})" + + # resilience: missing/empty inputs render placeholders, never crash + empty_dash = td / "empty.svg" + assert build_dashboard(td / "nope.log", td / "nope.jsonl", empty_dash) + etext = empty_dash.read_text() + assert etext.count("no eval lines found") == 2, etext.count("no eval lines found") + assert "no valid loss points" in etext + assert "no usable epoch timestamps" in etext + + # all-zero alpha rows (post-crash trainer state) must not raise in the + # losses panel's log-domain math -> placeholder instead of dead build + (td / "zero.jsonl").write_text( + '{"epoch": 1000.0, "alpha": 0}\n{"epoch": 1060.0, "alpha": 0}\n') + zero_dash = td / "zero.svg" + assert build_dashboard(td / "log", td / "zero.jsonl", zero_dash) + assert "no valid loss points" in zero_dash.read_text() + + # single throughput interval (fresh 2-row metrics file right after a + # restart) used to divide by zero in map_fn and kill the whole build + (td / "one.jsonl").write_text( + '{"epoch": 1000.0, "critic_loss": 10, "alpha": 0.5}\n' + '{"epoch": 1060.0, "critic_loss": 5, "alpha": 0.4}\n') + one_dash = td / "one.svg" + assert build_dashboard(td / "log", td / "one.jsonl", one_dash) + assert "panel error" not in one_dash.read_text() + assert " 0 else ROOT / "campaign_v4_stdout.log" metrics = Path(args[1]) if len(args) > 1 else ROOT / "training_metrics.jsonl" outdir = Path(args[2]) if len(args) > 2 else ROOT / "docs" - outdir.mkdir(parents=True, exist_ok=True) try: + outdir.mkdir(parents=True, exist_ok=True) + for p in (campaign, metrics): + if not p.is_file(): + # loudest symptom of a watcher launched from a stale checkout + log(f"[warn] input missing: {p} - is this the live checkout?") ok = build_dashboard(campaign, metrics, outdir / "campaign_dashboard.svg") - except Exception as e: - print(f"[error] dashboard build failed: {e}") + except Exception: + log(f"[error] dashboard build failed:\n{traceback.format_exc().rstrip()}") ok = False if ok: print(f"[done] dashboard written to {outdir / 'campaign_dashboard.svg'}") else: - print("[error] dashboard NOT updated - check paths above") + print("[error] dashboard NOT updated - see " + str(LOG_FILE)) sys.exit(1) diff --git a/SAC_LSTM_Bot/tools/watch_dashboard.sh b/SAC_LSTM_Bot/tools/watch_dashboard.sh index e16fd41..537afe8 100755 --- a/SAC_LSTM_Bot/tools/watch_dashboard.sh +++ b/SAC_LSTM_Bot/tools/watch_dashboard.sh @@ -1,10 +1,23 @@ #!/bin/sh # Keep docs/campaign_dashboard.svg fresh: regenerate every 60 s. -# Errors go to stderr and never exit the loop silently. -dir=$(dirname "$0") +# Failures are appended to tools/plot_progress.log (by this wrapper and by +# plot_progress.py itself) - check that file when the dashboard looks stale. +# The startup banner records WHICH tree this instance watches, so a copy +# launched from a stale checkout (the post-reboot failure mode) is visible. +set -u +dir=$(cd "$(dirname "$0")" && pwd) +log="$dir/plot_progress.log" + +if command -v flock >/dev/null 2>&1; then + exec 9>"$dir/.watch_dashboard.lock" + flock -n 9 || { echo "[watch_dashboard] another instance already running, exiting" >&2; exit 0; } +fi + +echo "[watch_dashboard] $(date '+%F %T') started, watching root=$(cd "$dir/.." && pwd)" >> "$log" 2>/dev/null || true while :; do if ! python3 "$dir/plot_progress.py"; then - echo "[watch_dashboard] $(date '+%F %T') regeneration failed (see error above)" >&2 + echo "[watch_dashboard] $(date '+%F %T') regeneration failed (see $log)" >&2 + echo "[watch_dashboard] $(date '+%F %T') regeneration failed" >> "$log" 2>/dev/null || true fi sleep 60 done