From 08d1c44ca81558884063f764db49e4866eac50cf Mon Sep 17 00:00:00 2001 From: Davide Cappellini Date: Mon, 24 Aug 2026 09:25:14 +0200 Subject: [PATCH] docs(dashboard): add how-to-read guide per panel --- SAC_LSTM_Bot/docs/campaign_dashboard.svg | 706 ++++++++++++----------- SAC_LSTM_Bot/tools/plot_progress.py | 56 +- 2 files changed, 398 insertions(+), 364 deletions(-) diff --git a/SAC_LSTM_Bot/docs/campaign_dashboard.svg b/SAC_LSTM_Bot/docs/campaign_dashboard.svg index fa0ab5a..cf0a607 100644 --- a/SAC_LSTM_Bot/docs/campaign_dashboard.svg +++ b/SAC_LSTM_Bot/docs/campaign_dashboard.svg @@ -1,363 +1,369 @@ SAC-LSTM campaign dashboard - live run (current only) -generated 2026-08-24 09:22:32 - auto-reloads every 60 s (open this file in Chrome) +generated 2026-08-24 09:24:45 - auto-reloads every 60 s (open this file in Chrome) Test matches - win % vs opponents raw dots = single test matches, thick = rolling-mean-10 - - - - - - - - - -1 - -2 - -3 - -4 - -5 - -6 - -7 - -8 - -9 - -10 - -11 - -12 - -13 - -14 - -15 - -0 - -20 - -40 - -60 - -80 - -100 -test match number (each opponent) -win rate (%) - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -Corners - 15 evals - -Crazy - 14 evals - -Target - 14 evals +What: How often the bot wins against each opponent in test battles. +Axes: X = training progress (match number); Y = win rate, 0-100%. +Better: higher is better. + + + + + + + + + +1 + +2 + +3 + +4 + +5 + +6 + +7 + +8 + +9 + +10 + +11 + +12 + +13 + +14 + +15 + +0 + +20 + +40 + +60 + +80 + +100 +test match number (each opponent) +win rate (%) + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Corners - 15 evals + +Crazy - 15 evals + +Target - 15 evals Training losses (log) & alpha (linear) training_metrics.jsonl - big early spikes are normal - - - - - - - - -1 - -8 - -16 - -23 - -30 - -0.01 - -0.1 - -1 - -10 - -100 -metric line number -loss (log) - - - - - - - - - -0 - -0.25 - -0.5 - -0.75 - -1 -alpha - - -critic_loss - -|actor_loss| - -alpha +What: How well the brain is learning: losses should fall; alpha sets explore/exploit. +Axes: X = training progress; left Y (log) = losses; right Y (linear) = alpha. +Better: lower losses = better; alpha falls over time as the bot gets confident. + + + + + + + + +1 + +9 + +16 + +24 + +32 + +0.01 + +0.1 + +1 + +10 + +100 +metric line number +loss (log) + + + + + + + + + +0 + +0.25 + +0.5 + +0.75 + +1 +alpha + + +critic_loss + +|actor_loss| + +alpha Throughput - games per hour method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk - - - - - - - - -2 - -4 - -6 - -8 - -10 - -12 - -14 - -16 - -18 - -20 - -22 - -24 - -26 - -28 - -0 - -1653 - -3305 - -4958 - -6611 -chunk interval (10-game chunks) -games / hour - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - +What: How many games per hour the bot trains (its learning speed). +Axes: X = training progress (10-game chunks); Y = games per hour. +Better: higher = faster learning; a steady line beats a spiky one. + + + + + + + + +3 + +6 + +9 + +12 + +15 + +18 + +21 + +24 + +27 + +30 + +0 + +1653 + +3305 + +4958 + +6611 +chunk interval (10-game chunks) +games / hour + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Max score per eval cycle campaign_v4_stdout.log - max of the 10 deterministic round scores per eval - - - - - - - - -1 - -2 - -3 - -4 - -5 - -6 - -7 - -8 - -9 - -10 - -11 - -12 - -13 - -14 - -15 - -0 - -52 - -103 - -155 - -207 -eval cycle number -best single-round score - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -Corners - 15 evals - -Crazy - 14 evals - -Target - 14 evals - -combined max - -How to read -Regenerate anytime: python3 tools/plot_progress.py -Test matches: dots are single fights, thick line shows trend. -Loss spikes are normal early; endless growth is bad. -Throughput flat is healthy; dips mean something slowed. -Max score: best single-round score the bot managed in that eval cycle. -This file reloads itself in Chrome every sixty seconds. -Regenerate anytime with tools/watch_dashboard.sh or the python command. +What: Best single-round score the bot managed in each test cycle. +Axes: X = eval cycle number; Y = best score achieved. +Better: higher is better; a rising trend means the bot is improving. + + + + + + + + +1 + +2 + +3 + +4 + +5 + +6 + +7 + +8 + +9 + +10 + +11 + +12 + +13 + +14 + +15 + +0 + +52 + +103 + +155 + +207 +eval cycle number +best single-round score + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Corners - 15 evals + +Crazy - 15 evals + +Target - 15 evals + +combined max + +How to read +Regenerate anytime: python3 tools/plot_progress.py +This file reloads itself in Chrome every sixty seconds. +Regenerate anytime with tools/watch_dashboard.sh or the python command. diff --git a/SAC_LSTM_Bot/tools/plot_progress.py b/SAC_LSTM_Bot/tools/plot_progress.py index aca71ce..6272568 100644 --- a/SAC_LSTM_Bot/tools/plot_progress.py +++ b/SAC_LSTM_Bot/tools/plot_progress.py @@ -46,15 +46,17 @@ ALPHA_C = "#9467bd" # alpha line on the losses panel RELOAD_JS = ('') W, H = 1400, 1720 -TITLE_H, GUIDE_H = 80, 180 +TITLE_H, GUIDE_H = 80, 80 # rows: (header_y, panel_top_y, panel_bottom_y, x_left, x_right) +# panel_top sits low enough to leave room under header+sub for the +# per-panel how-to-read guide (3 italic lines, see panel_guide) C1_L, C1_R = 70, 697 C2_L, C2_R = 747, 1375 ROWS = { - 1: (100, 120, 720, C1_L, C1_R), - 2: (100, 120, 720, C2_L, C2_R), - 3: (940, 958, 1428, C1_L, C1_R), - 4: (940, 958, 1428, C2_L, C2_R), + 1: (100, 184, 784, C1_L, C1_R), + 2: (100, 184, 784, C2_L, C2_R), + 3: (940, 1024, 1494, C1_L, C1_R), + 4: (940, 1024, 1494, C2_L, C2_R), } PANEL_TITLES = [ "Test matches - win % vs opponents", @@ -62,11 +64,22 @@ PANEL_TITLES = [ "Throughput - games per hour", "Max score per eval cycle", ] +# per-panel how-to-read notes: (what it shows, axes, which direction is better) +GUIDES = { + 1: ("How often the bot wins against each opponent in test battles.", + "X = training progress (match number); Y = win rate, 0-100%.", + "higher is better."), + 2: ("How well the brain is learning: losses should fall; alpha sets explore/exploit.", + "X = training progress; left Y (log) = losses; right Y (linear) = alpha.", + "lower losses = better; alpha falls over time as the bot gets confident."), + 3: ("How many games per hour the bot trains (its learning speed).", + "X = training progress (10-game chunks); Y = games per hour.", + "higher = faster learning; a steady line beats a spiky one."), + 4: ("Best single-round score the bot managed in each test cycle.", + "X = eval cycle number; Y = best score achieved.", + "higher is better; a rising trend means the bot is improving."), +} GUIDE_LINES = [ - "Test matches: dots are single fights, thick line shows trend.", - "Loss spikes are normal early; endless growth is bad.", - "Throughput flat is healthy; dips mean something slowed.", - "Max score: best single-round score the bot managed in that eval cycle.", "This file reloads itself in Chrome every sixty seconds.", "Regenerate anytime with tools/watch_dashboard.sh or the python command.", ] @@ -289,6 +302,16 @@ def header(x, y, title, sub=None): return s +def panel_guide(x0, y, guide): + """Italic how-to-read note between a panel's header and its plot area.""" + what, axes, better = guide + s = "" + for i, txt in enumerate((f"What: {what}", f"Axes: {axes}", f"Better: {better}")): + s += (f'{esc(txt)}\n') + return s + + # ---------- panels ---------- def panel_test_matches(series, geo): @@ -455,21 +478,22 @@ def build_dashboard(campaign, metrics_f, out): f'(open this file in Chrome)\n') drawers = [ - (ROWS[1], PANEL_TITLES[0], + (ROWS[1], PANEL_TITLES[0], GUIDES[1], "raw dots = single test matches, thick = rolling-mean-%d" % TREND_WINDOW, lambda: panel_test_matches(series, ROWS[1])), - (ROWS[2], PANEL_TITLES[1], + (ROWS[2], PANEL_TITLES[1], GUIDES[2], "training_metrics.jsonl - big early spikes are normal", lambda: panel_losses(rows, ROWS[2])), - (ROWS[3], PANEL_TITLES[2], + (ROWS[3], PANEL_TITLES[2], GUIDES[3], "method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk", lambda: panel_throughput(rows, ROWS[3])), - (ROWS[4], PANEL_TITLES[3], + (ROWS[4], PANEL_TITLES[3], GUIDES[4], "campaign_v4_stdout.log - max of the 10 deterministic round scores per eval", lambda: panel_max_score(maxes, ROWS[4])), ] - for geo, title, sub, drawer in drawers: + for geo, title, guide, sub, drawer in drawers: s += header(geo[3], geo[0], title, sub) + s += panel_guide(geo[3], geo[0] + 33, guide) s += drawer() or "" # panels bare-return None on their no-data path s += guide_block(W, H, GUIDE_LINES) @@ -530,6 +554,10 @@ def selftest(): assert text.count(PANEL_TITLES[0]) == 1 assert "How to read" in text, "reading guide missing" assert 'width="1400"' in text and 'height="1720"' in text + # every panel carries its own What/Axes/Better how-to-read note + assert text.count("What:") == len(PANEL_TITLES), text.count("What:") + for g in GUIDES.values(): + assert esc(g[0]) in text, f"panel guide missing: {g[0]}" # circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 3) # + 2 cycles x 2 opponents max-score dots (panel 4) assert text.count("