diff --git a/SAC_LSTM_Bot/docs/campaign_dashboard.svg b/SAC_LSTM_Bot/docs/campaign_dashboard.svg index cf0a607..6b7a84d 100644 --- a/SAC_LSTM_Bot/docs/campaign_dashboard.svg +++ b/SAC_LSTM_Bot/docs/campaign_dashboard.svg @@ -1,7 +1,7 @@ - - + + SAC-LSTM campaign dashboard - live run (current only) -generated 2026-08-24 09:24:45 - auto-reloads every 60 s (open this file in Chrome) +generated 2026-08-24 10:18:35 - auto-reloads every 60 s (open this file in Chrome) Test matches - win % vs opponents raw dots = single test matches, thick = rolling-mean-10 What: How often the bot wins against each opponent in test battles. @@ -17,34 +17,74 @@ 1 - -2 - -3 - -4 - -5 - -6 - -7 + +2 + +3 + +4 + +5 + +6 + +7 + +8 + +9 + +10 + +11 + +12 + +13 + +14 + +15 + +16 + +17 -8 - -9 - -10 - -11 - -12 - -13 - -14 +18 + +19 + +20 + +21 + +22 + +23 + +24 + +25 + +26 + +27 + +28 + +29 + +30 + +31 + +32 + +33 + +34 -15 +35 0 @@ -60,59 +100,119 @@ test match number (each opponent) win rate (%) - - - - - - + + + + + + + + + + + + + + + + - - - - - - + + + + + + + + + + + + + + + + - + - - - - - - + + + + + + + + + + + + + + + + - - - - - - - - + + + + + + + + + + + + + + + + + + - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + -Corners - 15 evals +Corners - 35 evals -Crazy - 15 evals +Crazy - 35 evals -Target - 15 evals +Target - 35 evals Training losses (log) & alpha (linear) training_metrics.jsonl - big early spikes are normal What: How well the brain is learning: losses should fall; alpha sets explore/exploit. @@ -128,13 +228,13 @@ 1 -9 +19 -16 +36 -24 +54 -32 +72 0.01 @@ -147,8 +247,8 @@ 100 metric line number loss (log) - - + + @@ -166,7 +266,7 @@ 1 alpha - + critic_loss @@ -185,26 +285,26 @@ - -3 - -6 - -9 - -12 - -15 - -18 - -21 - -24 - -27 - -30 + +7 + +14 + +21 + +28 + +35 + +42 + +49 + +56 + +63 + +70 0 @@ -218,37 +318,77 @@ chunk interval (10-game chunks) games / hour - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Max score per eval cycle campaign_v4_stdout.log - max of the 10 deterministic round scores per eval What: Best single-round score the bot managed in each test cycle. @@ -261,36 +401,22 @@ - -1 - -2 - -3 - -4 - -5 - -6 - -7 - -8 - -9 - -10 - -11 - -12 - -13 - -14 - -15 + +4 + +8 + +12 + +16 + +20 + +24 + +28 + +32 0 @@ -304,66 +430,152 @@ eval cycle number best single-round score - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + - + - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + -Corners - 15 evals +Corners - 35 evals -Crazy - 15 evals +Crazy - 35 evals -Target - 15 evals +Target - 35 evals combined max - -How to read -Regenerate anytime: python3 tools/plot_progress.py -This file reloads itself in Chrome every sixty seconds. -Regenerate anytime with tools/watch_dashboard.sh or the python command. + +Reading the signs +Regenerate anytime: python3 tools/plot_progress.py +Healthy patterns ✅ +• Both critic and actor losses trending down over time +• Alpha decaying slowly from ~1.0, + then plateauing — this is normal +• Win rates appearing and increasing in the eval panel +• Max scores rising in the max-score panel +• Alpha plateauing is NOT a problem — it means + exploration level is stable +Warning signs ⚠️ +• Losses exploding (suddenly jumping to + millions or billions) +• Win rates staying at 0% for a long time + after the first ~20 eval cycles +• Alpha reaching 0 — bot stops exploring entirely + (gets stuck) +• Max scores flatlining (no improvement + over many eval cycles) +• Any single loss value above 1e6 +What each metric means (brief) +• Critic loss: how wrong the bot's value + estimates are — should go down +• Actor loss: how well the bot's action policy + is doing — should go down overall + (some bumps are normal) +• Alpha: exploration-exploitation tradeoff — starts + high, settles at a positive value (NOT zero) +• Max score: best score achieved per eval cycle — + rising trend = learning diff --git a/SAC_LSTM_Bot/tools/plot_progress.py b/SAC_LSTM_Bot/tools/plot_progress.py index 6272568..32710e6 100644 --- a/SAC_LSTM_Bot/tools/plot_progress.py +++ b/SAC_LSTM_Bot/tools/plot_progress.py @@ -45,8 +45,8 @@ CRITIC_C, ACTOR_C = "#1f77b4", "#ff7f0e" ALPHA_C = "#9467bd" # alpha line on the losses panel RELOAD_JS = ('') -W, H = 1400, 1720 -TITLE_H, GUIDE_H = 80, 80 +W, H = 1400, 1850 +HEALTH_H = 205 # bottom "Reading the signs" cheat-sheet band # rows: (header_y, panel_top_y, panel_bottom_y, x_left, x_right) # panel_top sits low enough to leave room under header+sub for the # per-panel how-to-read guide (3 italic lines, see panel_guide) @@ -79,9 +79,36 @@ GUIDES = { "X = eval cycle number; Y = best score achieved.", "higher is better; a rising trend means the bot is improving."), } -GUIDE_LINES = [ - "This file reloads itself in Chrome every sixty seconds.", - "Regenerate anytime with tools/watch_dashboard.sh or the python command.", +# bottom health-check cheat-sheet: one column per subsection; each bullet is +# a tuple of pre-wrapped text lines (first line gets the bullet marker) +SIGNS_COLUMNS = [ + ("Healthy patterns ✅", [ + ("Both critic and actor losses trending down over time",), + ("Alpha decaying slowly from ~1.0,", "then plateauing — this is normal"), + ("Win rates appearing and increasing in the eval panel",), + ("Max scores rising in the max-score panel",), + ("Alpha plateauing is NOT a problem — it means", + "exploration level is stable"), + ]), + ("Warning signs ⚠️", [ + ("Losses exploding (suddenly jumping to", "millions or billions)"), + ("Win rates staying at 0% for a long time", + "after the first ~20 eval cycles"), + ("Alpha reaching 0 — bot stops exploring entirely", "(gets stuck)"), + ("Max scores flatlining (no improvement", + "over many eval cycles)"), + ("Any single loss value above 1e6",), + ]), + ("What each metric means (brief)", [ + ("Critic loss: how wrong the bot's value", + "estimates are — should go down"), + ("Actor loss: how well the bot's action policy", + "is doing — should go down overall", "(some bumps are normal)"), + ("Alpha: exploration-exploitation tradeoff — starts", + "high, settles at a positive value (NOT zero)"), + ("Max score: best score achieved per eval cycle —", + "rising trend = learning"), + ]), ] @@ -280,16 +307,23 @@ def map_fn(p0, p1, vmin, vmax, log=False): return f -def guide_block(w, h, lines): - """Plain-English "How to read" band at the bottom of the canvas.""" - y = H - GUIDE_H - s = f'\n' - s += (f'' - f"How to read\n" - f'Regenerate anytime: python3 tools/plot_progress.py\n') - for i, ln in enumerate(lines): - s += f'{esc(ln)}\n' +def signs_block(): + """Health-check cheat-sheet band along the bottom of the canvas.""" + y0 = H - HEALTH_H + s = f'\n' + s += (f'' + "Reading the signs\n" + f'Regenerate anytime: python3 tools/plot_progress.py\n') + for (head, bullets), x in zip(SIGNS_COLUMNS, (16, 500, 985)): + s += (f'' + f"{esc(head)}\n") + y = y0 + 61 + for lines in bullets: + for j, ln in enumerate(lines): + s += (f'' + f"{esc(('• ' if j == 0 else ' ') + ln)}\n") + y += 14 return s @@ -496,7 +530,7 @@ def build_dashboard(campaign, metrics_f, out): s += panel_guide(geo[3], geo[0] + 33, guide) s += drawer() or "" # panels bare-return None on their no-data path - s += guide_block(W, H, GUIDE_LINES) + s += signs_block() s += RELOAD_JS + "\n" s += "\n" return write_svg(out, s) @@ -552,8 +586,13 @@ def selftest(): for t in PANEL_TITLES: assert esc(t) in text, f"panel title missing: {t}" assert text.count(PANEL_TITLES[0]) == 1 - assert "How to read" in text, "reading guide missing" - assert 'width="1400"' in text and 'height="1720"' in text + assert "Reading the signs" in text, "health-check section missing" + for head, bullets in SIGNS_COLUMNS: + assert esc(head) in text, f"health-check column missing: {head}" + for lines in bullets: + assert esc("• " + lines[0]) in text, f"bullet missing: {lines[0]}" + assert text.count("• ") == sum(len(b) for _, b in SIGNS_COLUMNS) + assert 'width="1400"' in text and 'height="1850"' in text # every panel carries its own What/Axes/Better how-to-read note assert text.count("What:") == len(PANEL_TITLES), text.count("What:") for g in GUIDES.values():