From 8eb7c53dd0d11dc65cde79e9e96f183e80d572c9 Mon Sep 17 00:00:00 2001 From: Davide Cappellini Date: Mon, 24 Aug 2026 09:05:41 +0200 Subject: [PATCH] fix(dashboard): remove broken unused win%-per-opponent panel --- SAC_LSTM_Bot/docs/campaign_dashboard.svg | 295 +++++++++++++---------- SAC_LSTM_Bot/tools/plot_progress.py | 136 ++--------- 2 files changed, 185 insertions(+), 246 deletions(-) diff --git a/SAC_LSTM_Bot/docs/campaign_dashboard.svg b/SAC_LSTM_Bot/docs/campaign_dashboard.svg index 9c5967d..60025cb 100644 --- a/SAC_LSTM_Bot/docs/campaign_dashboard.svg +++ b/SAC_LSTM_Bot/docs/campaign_dashboard.svg @@ -1,7 +1,7 @@ - - + + SAC-LSTM campaign dashboard - live run (current only) -generated 2026-08-24 08:29:26 - auto-reloads every 60 s (open this file in Chrome) +generated 2026-08-24 09:05:32 - auto-reloads every 60 s (open this file in Chrome) Test matches - win % vs opponents raw dots = single test matches, thick = rolling-mean-10 @@ -14,16 +14,22 @@ 1 - -2 - -3 - -4 - -5 + +2 + +3 + +4 + +5 + +6 + +7 + +8 -6 +9 0 @@ -39,98 +45,112 @@ test match number (each opponent) win rate (%) - - - - + + + + + + + - + - - - - + + + + + + + - + - - - - - + + + + + + + + + -Corners - 6 evals +Corners - 9 evals -Crazy - 6 evals +Crazy - 9 evals -Target - 5 evals -Real fights - win % per opponent -training_log.jsonl only - learning in REAL battles, not tests +Target - 9 evals +Training losses (log scale) +training_metrics.jsonl - big early spikes are normal - - - - + + + + +1 + +6 + +10 + +15 + +20 -0 - -20 - -40 - -60 - -80 +0.01 + +0.1 + +1 + +10 100 -game number (100-game buckets) -win % - -RamFire (10 games) - -Target (10 games) - -Corners (40 games) -Training losses (log scale) -training_metrics.jsonl - big early spikes are normal +metric line number +loss (log) + + + +critic_loss + +|actor_loss| +Alpha temperature +training_metrics.jsonl - high = exploring, low = exploiting + + 1 -2 +6 -4 +10 -5 +15 -6 +20 -1 +0 -3.16 +0.25 -10 +0.5 -31.6 +0.75 -100 +1 metric line number -loss (log) - - - -critic_loss - -|actor_loss| -Alpha temperature -training_metrics.jsonl - high = exploring, low = exploiting +alpha + +Throughput - games per hour +method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk @@ -140,73 +160,82 @@ 1 - -2 + +2 + +3 + +4 + +5 + +6 + +7 + +8 + +9 -4 - -5 +10 + +11 + +12 + +13 + +14 + +15 + +16 + +17 + +18 -6 +19 0 -0.25 +1653 -0.5 +3305 -0.75 +4958 -1 -metric line number -alpha - -Throughput - games per hour -method: training_metrics.jsonl 'epoch' deltas (training_log.jsonl has no timestamps); 1 row = one 10-game chunk - - - - - - - - -1 - -2 - -3 - -4 - -5 - -0 - -70 - -141 - -211 - -282 -chunk interval (10-game chunks) -games / hour - - - - - - - -How to read -Regenerate anytime: python3 tools/plot_progress.py -Test matches: dots are single fights, thick line shows trend. -Real battles only. Rising lines mean the bot improves. -Loss spikes are normal early; endless growth is bad. -Alpha high means experimenting; falling too fast freezes habits. -Throughput flat is healthy; dips mean something slowed. -This file reloads itself in Chrome every sixty seconds. -Regenerate anytime with tools/watch_dashboard.sh or the python command. +6611 +chunk interval (10-game chunks) +games / hour + + + + + + + + + + + + + + + + + + + + + +How to read +Regenerate anytime: python3 tools/plot_progress.py +Test matches: dots are single fights, thick line shows trend. +Loss spikes are normal early; endless growth is bad. +Alpha high means experimenting; falling too fast freezes habits. +Throughput flat is healthy; dips mean something slowed. +This file reloads itself in Chrome every sixty seconds. +Regenerate anytime with tools/watch_dashboard.sh or the python command. diff --git a/SAC_LSTM_Bot/tools/plot_progress.py b/SAC_LSTM_Bot/tools/plot_progress.py index da040e7..4054cf2 100644 --- a/SAC_LSTM_Bot/tools/plot_progress.py +++ b/SAC_LSTM_Bot/tools/plot_progress.py @@ -3,21 +3,18 @@ Pure-stdlib SVG output (matplotlib not available on this box). Generates ONE file: - docs/campaign_dashboard.svg - five panels, current (v2) run only: + docs/campaign_dashboard.svg - four panels, current (v2) run only: 1. test-match win % vs opponents (campaign_v4_stdout.log eval lines) - 2. real-fight win % per opponent (training_log.jsonl, ~100-game buckets) - 3. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y) - 4. alpha temperature (training_metrics.jsonl, linear) - 5. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas; - training_log.jsonl has NO timestamps - - verified field names - and one metrics - row == one 10-game chunk, counts match - the stdout "=== Chunk N/N ===" markers) + 2. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y) + 3. alpha temperature (training_metrics.jsonl, linear) + 4. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas; + 1 metrics row == one 10-game chunk, + counts match stdout chunk markers) plus an embedded JS snippet that reloads the page every 60 s when the SVG is opened as a top-level document in Chrome. Usage: - python3 tools/plot_progress.py [campaign_log] [metrics_jsonl] [games_jsonl] [outdir] + python3 tools/plot_progress.py [campaign_log] [metrics_jsonl] [outdir] All args optional; defaults relative to the SAC_LSTM_Bot/ root (parent of tools/). python3 tools/plot_progress.py --selftest # tiny built-in sanity check """ @@ -34,7 +31,6 @@ import xml.etree.ElementTree as ET ROOT = Path(__file__).resolve().parent.parent EVAL_RE = re.compile(r">>> \[eval\] win rate: (\d+)/(\d+) \(([\d.]+)%\) vs (\S+)") TREND_WINDOW = 10 # rolling mean shown as the thick trend line (panel 1) -GAME_BUCKET = 100 # games per bucket, real-fight panel RATE_BUCKET = 20 # metric intervals per throughput bucket (~200 games) GAMES_PER_ROW = 10 # one training_metrics.jsonl row per 10-round chunk COLORS = {"Corners": "#d62728", "Crazy": "#1f77b4", "Target": "#2ca02c", @@ -42,7 +38,7 @@ COLORS = {"Corners": "#d62728", "Crazy": "#1f77b4", "Target": "#2ca02c", CRITIC_C, ACTOR_C = "#1f77b4", "#ff7f0e" RELOAD_JS = ('') -W, H = 1400, 2000 +W, H = 1400, 1720 TITLE_H, GUIDE_H = 80, 180 # rows: (header_y, panel_top_y, panel_bottom_y, x_left, x_right) C1_L, C1_R = 70, 697 @@ -52,18 +48,15 @@ ROWS = { 2: (100, 120, 720, C2_L, C2_R), 3: (940, 958, 1428, C1_L, C1_R), 4: (940, 958, 1428, C2_L, C2_R), - 5: (1600, 1618, 1780, C1_L, C2_R), } PANEL_TITLES = [ "Test matches - win % vs opponents", - "Real fights - win % per opponent", "Training losses (log scale)", "Alpha temperature", "Throughput - games per hour", ] GUIDE_LINES = [ "Test matches: dots are single fights, thick line shows trend.", - "Real battles only. Rising lines mean the bot improves.", "Loss spikes are normal early; endless growth is bad.", "Alpha high means experimenting; falling too fast freezes habits.", "Throughput flat is healthy; dips mean something slowed.", @@ -109,25 +102,6 @@ def parse_metrics(path): return rows -def parse_games(path): - """Return [(opponent, won_bool)] for type=='game' rows, skipping junk.""" - out = [] - if not path.is_file(): - print(f"[skip] games log not found: {path}") - return out - for line in path.read_text(errors="replace").splitlines(): - try: - r = json.loads(line) - except json.JSONDecodeError: - continue - if r.get("type") != "game": - continue - opp, win = r.get("opponent"), r.get("win") - if isinstance(opp, str) and isinstance(win, bool): - out.append((opp, win)) - return out - - def metric_col(rows, key, positive=False): """[(index, value)] for float-parseable rows; abs() applied; optional >0 filter.""" out = [] @@ -305,53 +279,6 @@ def panel_test_matches(series, geo): s += legend(items, x0 + 12, pb + 52) return s -def panel_real_fights(games, geo): - _, pt, pb, x0, x1 = geo - s = "" - if not games: - s += f'' \ - "no game rows found\n" - return - edges = list(range(0, len(games) + 1, GAME_BUCKET)) - if edges[-1] != len(games): - edges.append(len(games)) - buckets = list(zip(edges[:-1], edges[1:])) - xm, ym = map_fn(x0, x1, 1, max(len(games), 2)), map_fn(pb, pt, 0, 100) - s += hgrid(x0, x1, [ym(v) for v in range(0, 101, 20)]) - step = max(GAME_BUCKET, GAME_BUCKET * (len(games) // GAME_BUCKET // 8 + 1)) - xt = [(str(v), xm(v)) for v in range(step, len(games) + 1, step)] - s += axis(x0, pb, x1, pt, xt, ticks_linear(0, 100, pb, pt, n=6), - f"game number ({GAME_BUCKET}-game buckets)", "win %") - opponents = [] - for opp, _ in games: - if opp not in opponents: - opponents.append(opp) - items = [] - for name in opponents: - c = COLORS.get(name, "#7f7f7f") - by_b = {} - for bi, (lo, hi) in enumerate(buckets): - sub = [w for o, w in games[lo:hi] if o == name] - if sub: - by_b[bi] = 100.0 * sum(sub) / len(sub) - pts = [] - segs, prev = [], None - for bi in sorted(by_b): - if prev is not None and bi != prev + 1: - segs.append(pts) - pts = [] - center = (buckets[bi][0] + buckets[bi][1]) / 2 - pts.append((xm(center), ym(by_b[bi]))) - prev = bi - if len(pts) >= 2: - segs.append(pts) - for seg in segs: - s += polyline(seg, c, 3.5) - n_played = sum(1 for o, _ in games if o == name) - items.append((c, f"{name} ({n_played} games)")) - s += legend(items, x0 + 12, pb + 52) - return s - def panel_losses(rows, geo): _, pt, pb, x0, x1 = geo s = "" @@ -434,14 +361,12 @@ def panel_throughput(rows, geo): # ---------- assembly ---------- -def build_dashboard(campaign, metrics_f, games_f, out): +def build_dashboard(campaign, metrics_f, out): series = parse_eval_series(campaign) print("[info] evals parsed: " + (", ".join(f"{k}={len(v)}" for k, v in sorted(series.items())) or "(none)")) rows = parse_metrics(metrics_f) print(f"[info] metric rows parsed: {len(rows)}") - games = parse_games(games_f) - print(f"[info] game rows parsed: {len(games)}") s = (f'\n' @@ -457,18 +382,14 @@ def build_dashboard(campaign, metrics_f, games_f, out): "raw dots = single test matches, thick = rolling-mean-%d" % TREND_WINDOW, lambda: panel_test_matches(series, ROWS[1])), (ROWS[2], PANEL_TITLES[1], - "training_log.jsonl only - learning in REAL battles, not tests", - lambda: panel_real_fights(games, ROWS[2])), - (ROWS[3], PANEL_TITLES[2], "training_metrics.jsonl - big early spikes are normal", - lambda: panel_losses(rows, ROWS[3])), - (ROWS[4], PANEL_TITLES[3], + lambda: panel_losses(rows, ROWS[2])), + (ROWS[3], PANEL_TITLES[2], "training_metrics.jsonl - high = exploring, low = exploiting", - lambda: panel_alpha(rows, ROWS[4])), - (ROWS[5], PANEL_TITLES[4], - "method: training_metrics.jsonl 'epoch' deltas (training_log.jsonl has " - "no timestamps); 1 row = one 10-game chunk", - lambda: panel_throughput(rows, ROWS[5])), + lambda: panel_alpha(rows, ROWS[3])), + (ROWS[4], PANEL_TITLES[3], + "method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk", + lambda: panel_throughput(rows, ROWS[4])), ] for geo, title, sub, drawer in drawers: s += header(geo[3], geo[0], title, sub) @@ -506,20 +427,10 @@ def selftest(): '{"epoch": 1090.0, "critic_loss": 50, "actor_loss": 3, "alpha": 0.2}\n') rows = parse_metrics(td / "m.jsonl") assert len(rows) == 3 and rows[1]["critic_loss"] == 100 - glines = [] - for i in range(150): # 2 full GAME_BUCKETs, both opponents in both - glines.append(json.dumps( - {"type": "game", "round": i % 10 + 1, "ticks": 100, - "score": i % 3, "total_score": i, "win": i % 3 == 0, - "opponent": ("Corners", "Crazy")[i % 2]})) - (td / "g.jsonl").write_text("\n".join(glines) + "\n") - games = parse_games(td / "g.jsonl") - assert len(games) == 150 and games[0] == ("Corners", True) - assert games[-1] == ("Crazy", False) # i=149: odd -> Crazy; 149%3!=0 -> loss assert metric_col(rows, "actor_loss") == [(0, 2.0), (1, 4.0), (2, 3.0)] dash = td / "dash.svg" - assert build_dashboard(td / "log", td / "m.jsonl", td / "g.jsonl", dash) + assert build_dashboard(td / "log", td / "m.jsonl", dash) text = dash.read_text() ET.fromstring(text) # whole doc must parse -> closing tag present assert RELOAD_JS in text, "auto-reload script missing" @@ -527,11 +438,11 @@ def selftest(): assert t in text, f"panel title missing: {t}" assert text.count(PANEL_TITLES[0]) == 1 assert "How to read" in text, "reading guide missing" - assert 'width="1400"' in text and 'height="2000"' in text - # circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 5) + assert 'width="1400"' in text and 'height="1720"' in text + # circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 4) assert text.count(" 90%) rendered through # the FULL build path must plot upward (smaller SVG y) and forward in @@ -541,7 +452,7 @@ def selftest(): ">>> [eval] win rate: 1/10 (10%) vs Corners\n" ">>> [eval] win rate: 9/10 (90%) vs Corners\n") ori_dash = td / "ori.svg" - assert build_dashboard(ori_log, td / "m.jsonl", td / "g.jsonl", ori_dash) + assert build_dashboard(ori_log, td / "m.jsonl", ori_dash) m = re.search(r' 0 else ROOT / "campaign_v4_stdout.log" metrics = Path(args[1]) if len(args) > 1 else ROOT / "training_metrics.jsonl" - games = Path(args[2]) if len(args) > 2 else ROOT / "training_log.jsonl" - outdir = Path(args[3]) if len(args) > 3 else ROOT / "docs" + outdir = Path(args[2]) if len(args) > 2 else ROOT / "docs" outdir.mkdir(parents=True, exist_ok=True) try: - ok = build_dashboard(campaign, metrics, games, outdir / "campaign_dashboard.svg") + ok = build_dashboard(campaign, metrics, outdir / "campaign_dashboard.svg") except Exception as e: print(f"[error] dashboard build failed: {e}") ok = False