#!/usr/bin/env python3 """Build the single self-updating SAC training dashboard. Pure-stdlib SVG output (matplotlib not available on this box). Generates ONE file: docs/campaign_dashboard.svg - five panels, current (v2) run only: 1. test-match win % vs opponents (campaign_v2_stdout.log eval lines) 2. real-fight win % per opponent (training_log.jsonl, ~100-game buckets) 3. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y) 4. alpha temperature (training_metrics.jsonl, linear) 5. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas; training_log.jsonl has NO timestamps - verified field names - and one metrics row == one 10-game chunk, counts match the stdout "=== Chunk N/N ===" markers) plus an embedded JS snippet that reloads the page every 60 s when the SVG is opened as a top-level document in Chrome. Usage: python3 tools/plot_progress.py [campaign_log] [metrics_jsonl] [games_jsonl] [outdir] All args optional; defaults relative to the SAC_LSTM_Bot/ root (parent of tools/). python3 tools/plot_progress.py --selftest # tiny built-in sanity check """ import json import math import os import re import sys import tempfile from datetime import datetime from pathlib import Path import xml.etree.ElementTree as ET ROOT = Path(__file__).resolve().parent.parent EVAL_RE = re.compile(r">>> \[eval\] win rate: (\d+)/(\d+) \(([\d.]+)%\) vs (\S+)") TREND_WINDOW = 10 # rolling mean shown as the thick trend line (panel 1) GAME_BUCKET = 100 # games per bucket, real-fight panel RATE_BUCKET = 20 # metric intervals per throughput bucket (~200 games) GAMES_PER_ROW = 10 # one training_metrics.jsonl row per 10-round chunk COLORS = {"Corners": "#d62728", "Crazy": "#1f77b4", "Target": "#2ca02c", "RamFire": "#ff7f0e", "SacTwin": "#9467bd"} CRITIC_C, ACTOR_C = "#1f77b4", "#ff7f0e" RELOAD_JS = ('') W, H = 1400, 2000 TITLE_H, GUIDE_H = 80, 180 # rows: (header_y, panel_top_y, panel_bottom_y, x_left, x_right) C1_L, C1_R = 70, 697 C2_L, C2_R = 747, 1375 ROWS = { 1: (100, 120, 720, C1_L, C1_R), 2: (100, 120, 720, C2_L, C2_R), 3: (940, 958, 1428, C1_L, C1_R), 4: (940, 958, 1428, C2_L, C2_R), 5: (1600, 1618, 1780, C1_L, C2_R), } PANEL_TITLES = [ "Test matches - win % vs opponents", "Real fights - win % per opponent", "Training losses (log scale)", "Alpha temperature", "Throughput - games per hour", ] GUIDE_LINES = [ "Test matches: dots are single fights, thick line shows trend.", "Real battles only. Rising lines mean the bot improves.", "Loss spikes are normal early; endless growth is bad.", "Alpha high means experimenting; falling too fast freezes habits.", "Throughput flat is healthy; dips mean something slowed.", "This file reloads itself in Chrome every sixty seconds.", "Regenerate anytime with tools/watch_dashboard.sh or the python command.", ] # ---------- parsing ---------- def parse_eval_series(path): """Return {opponent: [win% per eval, in file order]}.""" series = {} if not path.is_file(): print(f"[skip] eval log not found: {path}") return series for line in path.read_text(errors="replace").splitlines(): m = EVAL_RE.search(line) if not m: continue try: pct = float(m.group(3)) except ValueError: continue series.setdefault(m.group(4), []).append(pct) return series def parse_metrics(path): """Return list of metric dicts, skipping malformed lines.""" rows = [] if not path.is_file(): print(f"[skip] metrics file not found: {path}") return rows for line in path.read_text(errors="replace").splitlines(): line = line.strip() if not line: continue try: rows.append(json.loads(line)) except json.JSONDecodeError: continue return rows def parse_games(path): """Return [(opponent, won_bool)] for type=='game' rows, skipping junk.""" out = [] if not path.is_file(): print(f"[skip] games log not found: {path}") return out for line in path.read_text(errors="replace").splitlines(): try: r = json.loads(line) except json.JSONDecodeError: continue if r.get("type") != "game": continue opp, win = r.get("opponent"), r.get("win") if isinstance(opp, str) and isinstance(win, bool): out.append((opp, win)) return out def metric_col(rows, key, positive=False): """[(index, value)] for float-parseable rows; abs() applied; optional >0 filter.""" out = [] for i, r in enumerate(rows): try: v = abs(float(r[key])) except (KeyError, TypeError, ValueError): continue if positive and v <= 0: continue out.append((i, v)) return out def rolling(vals, w=TREND_WINDOW): out, s = [], 0.0 for i, v in enumerate(vals): s += v if i >= w: s -= vals[i - w] out.append(s / min(i + 1, w)) return out def bucket_means(vals, n): """Split vals into <=n contiguous buckets of near-equal size; per-bucket mean.""" if not vals: return [] n = min(n, len(vals)) k, rem = divmod(len(vals), n) out, i = [], 0 for b in range(n): size = k + (1 if b < rem else 0) out.append(sum(vals[i:i + size]) / size) i += size return out # ---------- tiny SVG helpers ---------- def esc(s): return str(s).replace("&", "&").replace("<", "<").replace(">", ">") def write_svg(path, text): """Validate the finished SVG, then atomically swap it into place. Readers never see partial output; an invalid render aborts without touching the previous good file.""" try: ET.fromstring(text) except ET.ParseError as e: print(f"[error] {path.name}: generated SVG invalid, keeping old file ({e})") return False tmp = path.with_name(path.name + ".tmp") tmp.write_text(text) os.replace(tmp, path) return True def polyline(pts, color, width=1.5, dash=None, opacity=1.0): if len(pts) < 2: return "" d = f' stroke-dasharray="{dash}"' if dash else "" p = " ".join(f"{x:.1f},{y:.1f}" for x, y in pts) return (f'\n') def dots(pts, color, r=2, opacity=0.25): return "".join(f'\n' for x, y in pts) def hgrid(x0, x1, ys): return "".join(f'\n' for y in ys) def axis(x0, y0, x1, y1, xt, yt, xlabel, ylabel, ylog=False): """Draw axes + ticks + labels. xt/yt are (value, px) tick lists.""" s = (f'\n' f'\n') for v, px in xt: s += (f'\n' f'' f"{esc(v)}\n") for v, py in yt: s += (f'\n' f'' f"{esc(v)}\n") s += (f'' f"{esc(xlabel)}\n" f'{esc(ylabel)}\n') return s def ticks_linear(vmin, vmax, p0, p1, n=5, fmt="{:.0f}"): return [(fmt.format(vmin + (vmax - vmin) * i / (n - 1)), p0 + (p1 - p0) * i / (n - 1)) for i in range(n)] def ticks_log(vmin, vmax, p0, p1, n=5): vals = [10 ** (vmin + (vmax - vmin) * i / (n - 1)) for i in range(n)] return [("{:.3g}".format(v), p0 + (p1 - p0) * i / (n - 1)) for i, v in enumerate(vals)] def legend(items, x, y): """items: [(color, label)]""" s = "" for i, (color, label) in enumerate(items): yy = y + i * 18 s += (f'\n' f'{esc(label)}\n') return s def map_fn(p0, p1, vmin, vmax, log=False): def f(v): t = (math.log10(v) - vmin) / (vmax - vmin) if log else (v - vmin) / (vmax - vmin) return p0 + max(0.0, min(1.0, t)) * (p1 - p0) return f def guide_block(w, h, lines): """Plain-English "How to read" band at the bottom of the canvas.""" y = H - GUIDE_H s = f'\n' s += (f'' f"How to read\n" f'Regenerate anytime: python3 tools/plot_progress.py\n') for i, ln in enumerate(lines): s += f'{esc(ln)}\n' return s def header(x, y, title, sub=None): s = (f'' f"{esc(title)}\n") if sub: s += (f'' f"{esc(sub)}\n") return s # ---------- panels ---------- def panel_test_matches(series, geo): _, pt, pb, x0, x1 = geo s = "" if not series: s += f'' \ "no eval lines found\n" return xmax = max(max(len(v) for v in series.values()), 2) xm, ym = map_fn(x0, x1, 1, xmax), map_fn(pb, pt, 0, 100) s += hgrid(x0, x1, [ym(v) for v in range(0, 101, 20)]) step = max(1, (xmax // 8 // 10) * 10) xt = [(str(v), xm(v)) for v in range(step, xmax + 1, step)] or [("1", xm(1))] s += axis(x0, pb, x1, pt, xt, ticks_linear(0, 100, pb, pt, n=6), "test match number (each opponent)", "win rate (%)") items = [] for name in ("Corners", "Crazy", "Target"): vals = series.get(name, []) if not vals: continue c = COLORS[name] s += dots([(xm(i + 1), ym(v)) for i, v in enumerate(vals)], c) s += polyline([(xm(i + 1), ym(v)) for i, v in enumerate(rolling(vals))], c, 3.5) items.append((c, f"{name} - {len(vals)} evals")) s += legend(items, x0 + 12, pb + 52) return s def panel_real_fights(games, geo): _, pt, pb, x0, x1 = geo s = "" if not games: s += f'' \ "no game rows found\n" return edges = list(range(0, len(games) + 1, GAME_BUCKET)) if edges[-1] != len(games): edges.append(len(games)) buckets = list(zip(edges[:-1], edges[1:])) xm, ym = map_fn(x0, x1, 1, max(len(games), 2)), map_fn(pb, pt, 0, 100) s += hgrid(x0, x1, [ym(v) for v in range(0, 101, 20)]) step = max(GAME_BUCKET, GAME_BUCKET * (len(games) // GAME_BUCKET // 8 + 1)) xt = [(str(v), xm(v)) for v in range(step, len(games) + 1, step)] s += axis(x0, pb, x1, pt, xt, ticks_linear(0, 100, pb, pt, n=6), f"game number ({GAME_BUCKET}-game buckets)", "win %") opponents = [] for opp, _ in games: if opp not in opponents: opponents.append(opp) items = [] for name in opponents: c = COLORS.get(name, "#7f7f7f") by_b = {} for bi, (lo, hi) in enumerate(buckets): sub = [w for o, w in games[lo:hi] if o == name] if sub: by_b[bi] = 100.0 * sum(sub) / len(sub) pts = [] segs, prev = [], None for bi in sorted(by_b): if prev is not None and bi != prev + 1: segs.append(pts) pts = [] center = (buckets[bi][0] + buckets[bi][1]) / 2 pts.append((xm(center), ym(by_b[bi]))) prev = bi if len(pts) >= 2: segs.append(pts) for seg in segs: s += polyline(seg, c, 3.5) n_played = sum(1 for o, _ in games if o == name) items.append((c, f"{name} ({n_played} games)")) s += legend(items, x0 + 12, pb + 52) return s def panel_losses(rows, geo): _, pt, pb, x0, x1 = geo s = "" critic = metric_col(rows, "critic_loss", positive=True) actor = metric_col(rows, "actor_loss") # abs() applied; sign dropped actor = [(i, v) for i, v in actor if v > 0] if not (critic or actor): s += f'' \ "no valid loss points\n" return allv = [v for _, v in critic + actor] lo, hi = math.floor(math.log10(min(allv))), math.ceil(math.log10(max(allv))) if lo == hi: hi = lo + 1 n = len(rows) xm = lambda i: x0 + (x1 - x0) * i / max(n - 1, 1) ym = map_fn(pb, pt, lo, hi, log=True) s += hgrid(x0, x1, [ym(10 ** e) for e in range(lo, hi + 1)]) s += axis(x0, pb, x1, pt, ticks_linear(1, n, x0, x1, n=5), ticks_log(lo, hi, pb, pt), "metric line number", "loss (log)") s += polyline([(xm(i), ym(v)) for i, v in critic], CRITIC_C, 1.8) s += polyline([(xm(i), ym(v)) for i, v in actor], ACTOR_C, 1.8) s += legend([(CRITIC_C, "critic_loss"), (ACTOR_C, "|actor_loss|")], x0 + 12, pb + 52) return s def panel_alpha(rows, geo): _, pt, pb, x0, x1 = geo s = "" alpha = metric_col(rows, "alpha") if not alpha: s += f'' \ "no alpha points\n" return n = len(rows) hi = max(1.0, max(v for _, v in alpha)) xm = lambda i: x0 + (x1 - x0) * i / max(n - 1, 1) ym = map_fn(pb, pt, 0, hi) s += hgrid(x0, x1, [ym(v) for v in [hi * k / 4 for k in range(5)]]) s += axis(x0, pb, x1, pt, ticks_linear(1, n, x0, x1, n=5), ticks_linear(0, hi, pb, pt, n=5, fmt="{:.3g}"), "metric line number", "alpha") s += polyline([(xm(i), ym(v)) for i, v in alpha], "#9467bd", 1.8) return s def panel_throughput(rows, geo): _, pt, pb, x0, x1 = geo s = "" eps = [] for r in rows: try: eps.append(float(r["epoch"])) except (KeyError, TypeError, ValueError): continue rates = [] # games/hour per inter-row interval for a, b in zip(eps, eps[1:]): dt = b - a if dt > 0: rates.append(3600.0 * GAMES_PER_ROW / dt) if not rates: s += f'' \ "no usable epoch timestamps\n" return bm = bucket_means(rates, RATE_BUCKET) xm = map_fn(x0, x1, 1, len(rates)) ymax = max(max(rates), max(bm)) * 1.1 ym = map_fn(pb, pt, 0, ymax) s += hgrid(x0, x1, [ym(ymax * k / 4) for k in range(5)]) step = max(1, len(rates) // 10) xt = [(str(v), xm(v)) for v in range(step, len(rates) + 1, step)] s += axis(x0, pb, x1, pt, xt, ticks_linear(0, ymax, pb, pt, n=5, fmt="{:.0f}"), f"chunk interval ({GAMES_PER_ROW}-game chunks)", "games / hour") s += dots([(xm(i + 1), ym(v)) for i, v in enumerate(rates)], "#7f7f7f", r=1.6) if len(bm) >= 2: ctr = [xm(round((i + 0.5) * len(rates) / len(bm))) for i in range(len(bm))] s += polyline(list(zip(ctr, map(ym, bm))), "#2ca02c", 3.5) return s # ---------- assembly ---------- def build_dashboard(campaign, metrics_f, games_f, out): series = parse_eval_series(campaign) print("[info] evals parsed: " + (", ".join(f"{k}={len(v)}" for k, v in sorted(series.items())) or "(none)")) rows = parse_metrics(metrics_f) print(f"[info] metric rows parsed: {len(rows)}") games = parse_games(games_f) print(f"[info] game rows parsed: {len(games)}") s = (f'\n' f'\n' f'SAC-LSTM campaign dashboard - live run (current only)\n' f'' f'generated {datetime.now():%Y-%m-%d %H:%M:%S} - auto-reloads every 60 s ' f'(open this file in Chrome)\n') drawers = [ (ROWS[1], PANEL_TITLES[0], "raw dots = single test matches, thick = rolling-mean-%d" % TREND_WINDOW, lambda: panel_test_matches(series, ROWS[1])), (ROWS[2], PANEL_TITLES[1], "training_log.jsonl only - learning in REAL battles, not tests", lambda: panel_real_fights(games, ROWS[2])), (ROWS[3], PANEL_TITLES[2], "training_metrics.jsonl - big early spikes are normal", lambda: panel_losses(rows, ROWS[3])), (ROWS[4], PANEL_TITLES[3], "training_metrics.jsonl - high = exploring, low = exploiting", lambda: panel_alpha(rows, ROWS[4])), (ROWS[5], PANEL_TITLES[4], "method: training_metrics.jsonl 'epoch' deltas (training_log.jsonl has " "no timestamps); 1 row = one 10-game chunk", lambda: panel_throughput(rows, ROWS[5])), ] for geo, title, sub, drawer in drawers: s += header(geo[3], geo[0], title, sub) s += drawer() s += guide_block(W, H, GUIDE_LINES) s += RELOAD_JS + "\n" s += "\n" return write_svg(out, s) # ---------- selftest ---------- def selftest(): with tempfile.TemporaryDirectory() as td: td = Path(td) (td / "log").write_text( ">>> [eval] win rate: 3/10 (30%) vs Corners\n" "garbage line\n" ">>> [eval] win rate: 7/10 (70%) vs Crazy\n" ">>> [eval] win rate: broken\n" ">>> [eval] win rate: 5/10 (50%) vs Corners\n" ">>> [eval] win rate: 4/10 (40%) vs Crazy\n") ser = parse_eval_series(td / "log") assert ser == {"Corners": [30.0, 50.0], "Crazy": [70.0, 40.0]}, ser assert rolling([10] * 25, 20)[-1] == 10.0 assert rolling([1, 2, 3], 20) == [1.0, 1.5, 2.0] assert len(bucket_means(list(range(1287)), RATE_BUCKET)) == RATE_BUCKET bm = bucket_means([0, 10], RATE_BUCKET) assert bm == [0.0, 10.0], bm # fewer points than buckets -> no empty buckets (td / "m.jsonl").write_text( '{"epoch": 1000.0, "critic_loss": 10, "actor_loss": -2, "alpha": 0.5}\n' "not json\n" '{"epoch": 1060.0, "critic_loss": 100, "actor_loss": -4, "alpha": 0.25}\n' '{"epoch": 1090.0, "critic_loss": 50, "actor_loss": 3, "alpha": 0.2}\n') rows = parse_metrics(td / "m.jsonl") assert len(rows) == 3 and rows[1]["critic_loss"] == 100 glines = [] for i in range(150): # 2 full GAME_BUCKETs, both opponents in both glines.append(json.dumps( {"type": "game", "round": i % 10 + 1, "ticks": 100, "score": i % 3, "total_score": i, "win": i % 3 == 0, "opponent": ("Corners", "Crazy")[i % 2]})) (td / "g.jsonl").write_text("\n".join(glines) + "\n") games = parse_games(td / "g.jsonl") assert len(games) == 150 and games[0] == ("Corners", True) assert games[-1] == ("Crazy", False) # i=149: odd -> Crazy; 149%3!=0 -> loss assert metric_col(rows, "actor_loss") == [(0, 2.0), (1, 4.0), (2, 3.0)] dash = td / "dash.svg" assert build_dashboard(td / "log", td / "m.jsonl", td / "g.jsonl", dash) text = dash.read_text() ET.fromstring(text) # whole doc must parse -> closing tag present assert RELOAD_JS in text, "auto-reload script missing" for t in PANEL_TITLES: assert t in text, f"panel title missing: {t}" assert text.count(PANEL_TITLES[0]) == 1 assert "How to read" in text, "reading guide missing" assert 'width="1400"' in text and 'height="2000"' in text # circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 5) assert text.count(" 90%) rendered through # the FULL build path must plot upward (smaller SVG y) and forward in # time (larger x). Fails loudly if axis mapping is ever inverted again. ori_log = td / "ori.log" ori_log.write_text( ">>> [eval] win rate: 1/10 (10%) vs Corners\n" ">>> [eval] win rate: 9/10 (90%) vs Corners\n") ori_dash = td / "ori.svg" assert build_dashboard(ori_log, td / "m.jsonl", td / "g.jsonl", ori_dash) m = re.search(r'90 but ink moved down ({ya} -> {yb})" assert xb > xa, f"x-axis reversed: newer eval plotted left ({xa} -> {xb})" print("selftest OK") def main(): if "--selftest" in sys.argv: selftest() return args = [a for a in sys.argv[1:] if not a.startswith("-")] campaign = Path(args[0]) if len(args) > 0 else ROOT / "campaign_v2_stdout.log" metrics = Path(args[1]) if len(args) > 1 else ROOT / "training_metrics.jsonl" games = Path(args[2]) if len(args) > 2 else ROOT / "training_log.jsonl" outdir = Path(args[3]) if len(args) > 3 else ROOT / "docs" outdir.mkdir(parents=True, exist_ok=True) try: ok = build_dashboard(campaign, metrics, games, outdir / "campaign_dashboard.svg") except Exception as e: print(f"[error] dashboard build failed: {e}") ok = False if ok: print(f"[done] dashboard written to {outdir / 'campaign_dashboard.svg'}") else: print("[error] dashboard NOT updated - check paths above") sys.exit(1) if __name__ == "__main__": main()