diff --git a/SAC_LSTM_Bot/docs/campaign_dashboard.svg b/SAC_LSTM_Bot/docs/campaign_dashboard.svg index 0ef8c26..acc6b24 100644 --- a/SAC_LSTM_Bot/docs/campaign_dashboard.svg +++ b/SAC_LSTM_Bot/docs/campaign_dashboard.svg @@ -1,35 +1,35 @@ SAC-LSTM campaign dashboard - live run (current only) -generated 2026-08-23 10:29:53 - auto-reloads every 60 s (open this file in Chrome) +generated 2026-08-23 10:46:52 - auto-reloads every 60 s (open this file in Chrome) Test matches - win % vs opponents raw dots = single test matches, thick = rolling-mean-10 - - - - - + + + + + - -10 - -20 - -30 - -40 - -50 - -60 - -70 - -80 - -90 + +10 + +20 + +30 + +40 + +50 + +60 + +70 + +80 + +90 0 @@ -44,306 +44,331 @@ 100 test match number (each opponent) win rate (%) - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + -Corners - 90 evals +Corners - 98 evals -Crazy - 90 evals +Crazy - 98 evals -Target - 89 evals +Target - 98 evals Real fights - win % per opponent training_log.jsonl only - learning in REAL battles, not tests - - - - - + + + + + - -300 - -600 - -900 - -1200 - -1500 - -1800 + +300 + +600 + +900 + +1200 + +1500 + +1800 0 @@ -358,60 +383,60 @@ 100 game number (100-game buckets) win % - - - - - - - - - - - + + + + + + + + + + + -Crazy (420 games) +Crazy (470 games) -RamFire (420 games) +RamFire (450 games) -Target (190 games) +Target (220 games) -SacTwin (180 games) +SacTwin (190 games) -Corners (610 games) +Corners (670 games) Training losses (log scale) training_metrics.jsonl - big early spikes are normal - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + 1 -46 +51 -92 +101 -138 +151 -183 +201 0.1 @@ -424,31 +449,31 @@ 1e+17 metric line number loss (log) - - + + critic_loss |actor_loss| Alpha temperature training_metrics.jsonl - high = exploring, low = exploiting - - - - + + + + 1 -46 +51 -92 +101 -138 +151 -183 +201 0 @@ -461,36 +486,36 @@ 1.01 metric line number alpha - + Throughput - games per hour method: training_metrics.jsonl 'epoch' deltas (training_log.jsonl has no timestamps); 1 row = one 10-game chunk - - - - + + + + - -18 - -36 - -54 - -72 - -90 - -108 - -126 - -144 - -162 - -180 + +20 + +40 + +60 + +80 + +100 + +120 + +140 + +160 + +180 + +200 0 @@ -503,189 +528,207 @@ 6750 chunk interval (10-game chunks) games / hour - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + How to read Regenerate anytime: python3 tools/plot_progress.py diff --git a/SAC_LSTM_Bot/docs/campaign_notebook.md b/SAC_LSTM_Bot/docs/campaign_notebook.md index a482140..47a429b 100644 --- a/SAC_LSTM_Bot/docs/campaign_notebook.md +++ b/SAC_LSTM_Bot/docs/campaign_notebook.md @@ -322,3 +322,11 @@ SACLSTM_LR_CRITIC=1e-4 \ ### Progress graphs One live dashboard: `docs/campaign_dashboard.svg` (current run only — test wins, real-fight wins, losses, alpha, throughput; auto-reloads every 60 s when open in Chrome). Keep it fresh with `tools/watch_dashboard.sh` (regenerates every 60 s), or one-shot `python3 tools/plot_progress.py` (pure stdlib; paths overridable via argv, `--selftest` for sanity check). + +## ~10:47 — the dashboard's axes were upside-down since creation + +The progress graphs have been lying since they were made: 0% was drawn at the TOP of every panel and the newest games appeared on the LEFT. The cause is a one-line formula bug in `tools/plot_progress.py`: `map_fn` interpolated as `p1 - t*(p1-p0)` instead of `p0 + t*(p1-p0)`, so all five panels plotted `100 - value` on y and reversed time on x. Tick labels were computed by separate (correct) code, which is why the numbers on the axes never matched the ink. + +Fix + guard: formula corrected; every call site audited (panels 1/2/5 use `map_fn` for both axes and are fixed by the same line; panels 3/4 already used a correct local x-lambda; no other consumer of `map_fn` exists in the repo). `--selftest` now renders a known rising series through the full build path and fails loudly unless higher value = smaller SVG y and newer data = further right — proven to catch this exact bug when the old formula is re-injected. Dashboard regenerated from live logs. + +Honest status while reading the now-correct charts: run 3 is only hours old and winning ~0% — recent evals are 0/10 vs Corners, Crazy and Target alike, and real-fight buckets sit at 0–1 wins per 100 games. Expected for a fresh brain. Night-1's gains were real but were intentionally reset by the stability restart that began attempt-3; the curve starts from zero again here. diff --git a/SAC_LSTM_Bot/tools/plot_progress.py b/SAC_LSTM_Bot/tools/plot_progress.py index 38d1500..b3d5df3 100644 --- a/SAC_LSTM_Bot/tools/plot_progress.py +++ b/SAC_LSTM_Bot/tools/plot_progress.py @@ -250,7 +250,7 @@ def legend(items, x, y): def map_fn(p0, p1, vmin, vmax, log=False): def f(v): t = (math.log10(v) - vmin) / (vmax - vmin) if log else (v - vmin) / (vmax - vmin) - return p1 - max(0.0, min(1.0, t)) * (p1 - p0) + return p0 + max(0.0, min(1.0, t)) * (p1 - p0) return f @@ -532,6 +532,22 @@ def selftest(): assert text.count(" 90%) rendered through + # the FULL build path must plot upward (smaller SVG y) and forward in + # time (larger x). Fails loudly if axis mapping is ever inverted again. + ori_log = td / "ori.log" + ori_log.write_text( + ">>> [eval] win rate: 1/10 (10%) vs Corners\n" + ">>> [eval] win rate: 9/10 (90%) vs Corners\n") + ori_dash = td / "ori.svg" + assert build_dashboard(ori_log, td / "m.jsonl", td / "g.jsonl", ori_dash) + m = re.search(r'90 but ink moved down ({ya} -> {yb})" + assert xb > xa, f"x-axis reversed: newer eval plotted left ({xa} -> {xb})" print("selftest OK")