diff --git a/SAC_LSTM_Bot/docs/campaign_dashboard.svg b/SAC_LSTM_Bot/docs/campaign_dashboard.svg index 60025cb..26023a4 100644 --- a/SAC_LSTM_Bot/docs/campaign_dashboard.svg +++ b/SAC_LSTM_Bot/docs/campaign_dashboard.svg @@ -1,7 +1,7 @@ SAC-LSTM campaign dashboard - live run (current only) -generated 2026-08-24 09:05:32 - auto-reloads every 60 s (open this file in Chrome) +generated 2026-08-24 09:10:32 - auto-reloads every 60 s (open this file in Chrome) Test matches - win % vs opponents raw dots = single test matches, thick = rolling-mean-10 @@ -14,22 +14,26 @@ 1 - -2 - -3 - -4 + +2 + +3 + +4 + +5 -5 - -6 - -7 - -8 +6 + +7 + +8 + +9 + +10 -9 +11 0 @@ -45,42 +49,48 @@ test match number (each opponent) win rate (%) - - - + + + + - - - + + + + - + - - - + + + + - - - + + + + - + - - - + + + + - - - + + + + - + -Corners - 9 evals +Corners - 11 evals -Crazy - 9 evals +Crazy - 11 evals -Target - 9 evals -Training losses (log scale) +Target - 11 evals +Training losses & alpha (log scale) training_metrics.jsonl - big early spikes are normal @@ -92,13 +102,13 @@ 1 -6 +7 -10 +12 -15 +18 -20 +24 0.01 @@ -111,12 +121,15 @@ 100 metric line number loss (log) - - + + + critic_loss |actor_loss| + +alpha Alpha temperature training_metrics.jsonl - high = exploring, low = exploiting @@ -129,13 +142,13 @@ 1 -6 +7 -10 +12 -15 +18 -20 +24 0 @@ -148,7 +161,7 @@ 1 metric line number alpha - + Throughput - games per hour method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk @@ -158,44 +171,28 @@ - -1 - -2 - -3 - -4 - -5 - -6 - -7 - -8 - -9 + +2 + +4 + +6 + +8 + +10 -10 - -11 - -12 - -13 - -14 - -15 - -16 - -17 - -18 - -19 +12 + +14 + +16 + +18 + +20 + +22 0 @@ -209,25 +206,29 @@ chunk interval (10-game chunks) games / hour - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + How to read Regenerate anytime: python3 tools/plot_progress.py diff --git a/SAC_LSTM_Bot/tools/plot_progress.py b/SAC_LSTM_Bot/tools/plot_progress.py index 4054cf2..dfab006 100644 --- a/SAC_LSTM_Bot/tools/plot_progress.py +++ b/SAC_LSTM_Bot/tools/plot_progress.py @@ -5,7 +5,7 @@ Pure-stdlib SVG output (matplotlib not available on this box). Generates ONE file: docs/campaign_dashboard.svg - four panels, current (v2) run only: 1. test-match win % vs opponents (campaign_v4_stdout.log eval lines) - 2. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y) + 2. critic_loss / |actor_loss| / alpha (training_metrics.jsonl, shared log-y) 3. alpha temperature (training_metrics.jsonl, linear) 4. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas; 1 metrics row == one 10-game chunk, @@ -36,6 +36,7 @@ GAMES_PER_ROW = 10 # one training_metrics.jsonl row per 10-round chunk COLORS = {"Corners": "#d62728", "Crazy": "#1f77b4", "Target": "#2ca02c", "RamFire": "#ff7f0e", "SacTwin": "#9467bd"} CRITIC_C, ACTOR_C = "#1f77b4", "#ff7f0e" +ALPHA_C = "#9467bd" # same purple as the dedicated alpha panel RELOAD_JS = ('') W, H = 1400, 1720 @@ -51,7 +52,7 @@ ROWS = { } PANEL_TITLES = [ "Test matches - win % vs opponents", - "Training losses (log scale)", + "Training losses & alpha (log scale)", "Alpha temperature", "Throughput - games per hour", ] @@ -285,11 +286,12 @@ def panel_losses(rows, geo): critic = metric_col(rows, "critic_loss", positive=True) actor = metric_col(rows, "actor_loss") # abs() applied; sign dropped actor = [(i, v) for i, v in actor if v > 0] - if not (critic or actor): + alpha = metric_col(rows, "alpha") + if not (critic or actor or alpha): s += f'' \ "no valid loss points\n" return - allv = [v for _, v in critic + actor] + allv = [v for _, v in critic + actor + alpha] lo, hi = math.floor(math.log10(min(allv))), math.ceil(math.log10(max(allv))) if lo == hi: hi = lo + 1 @@ -301,8 +303,11 @@ def panel_losses(rows, geo): ticks_log(lo, hi, pb, pt), "metric line number", "loss (log)") s += polyline([(xm(i), ym(v)) for i, v in critic], CRITIC_C, 1.8) s += polyline([(xm(i), ym(v)) for i, v in actor], ACTOR_C, 1.8) - s += legend([(CRITIC_C, "critic_loss"), (ACTOR_C, "|actor_loss|")], - x0 + 12, pb + 52) + s += polyline([(xm(i), ym(v)) for i, v in alpha], ALPHA_C, 1.8) + items = [(CRITIC_C, "critic_loss"), (ACTOR_C, "|actor_loss|")] + if alpha: + items.append((ALPHA_C, "alpha")) + s += legend(items, x0 + 12, pb + 52) return s def panel_alpha(rows, geo): @@ -435,14 +440,15 @@ def selftest(): ET.fromstring(text) # whole doc must parse -> closing tag present assert RELOAD_JS in text, "auto-reload script missing" for t in PANEL_TITLES: - assert t in text, f"panel title missing: {t}" + assert esc(t) in text, f"panel title missing: {t}" assert text.count(PANEL_TITLES[0]) == 1 assert "How to read" in text, "reading guide missing" assert 'width="1400"' in text and 'height="1720"' in text # circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 4) assert text.count(" 90%) rendered through # the FULL build path must plot upward (smaller SVG y) and forward in