From 55d69a9352b592281222d41aae4f62d0381bbe10 Mon Sep 17 00:00:00 2001 From: Davide Cappellini Date: Mon, 24 Aug 2026 09:21:43 +0200 Subject: [PATCH] fix(dashboard): dual y-axis for losses+alpha panel --- SAC_LSTM_Bot/docs/campaign_dashboard.svg | 532 ++++++++++++----------- SAC_LSTM_Bot/tools/plot_progress.py | 101 ++--- 2 files changed, 319 insertions(+), 314 deletions(-) diff --git a/SAC_LSTM_Bot/docs/campaign_dashboard.svg b/SAC_LSTM_Bot/docs/campaign_dashboard.svg index 33601d1..4022579 100644 --- a/SAC_LSTM_Bot/docs/campaign_dashboard.svg +++ b/SAC_LSTM_Bot/docs/campaign_dashboard.svg @@ -1,7 +1,7 @@ - - + + SAC-LSTM campaign dashboard - live run (current only) -generated 2026-08-24 09:14:55 - auto-reloads every 60 s (open this file in Chrome) +generated 2026-08-24 09:21:32 - auto-reloads every 60 s (open this file in Chrome) Test matches - win % vs opponents raw dots = single test matches, thick = rolling-mean-10 @@ -14,28 +14,32 @@ 1 - -2 - -3 - -4 - -5 - -6 - -7 - -8 - -9 - -10 - -11 + +2 + +3 + +4 + +5 + +6 + +7 + +8 + +9 + +10 + +11 + +12 + +13 -12 +14 0 @@ -51,69 +55,75 @@ test match number (each opponent) win rate (%) - - - - - - - - - - - - + + + + + + + + + + + + + + - - - - - - - - - - - - + + + + + + + + + + + + + + - - - - - - - - - - - - + + + + + + + + + + + + + + -Corners - 12 evals +Corners - 14 evals -Crazy - 12 evals +Crazy - 14 evals -Target - 12 evals -Training losses & alpha (log scale) +Target - 14 evals +Training losses (log) & alpha (linear) training_metrics.jsonl - big early spikes are normal - - - - - - + + + + + + 1 - -7 - -14 - -20 - -26 + +8 + +16 + +23 + +30 0.01 @@ -124,19 +134,36 @@ 10 100 -metric line number +metric line number loss (log) - - - + + + + + + + + + +0 + +0.25 + +0.5 + +0.75 + +1 +alpha + critic_loss |actor_loss| alpha -Alpha temperature -training_metrics.jsonl - high = exploring, low = exploiting +Throughput - games per hour +method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk @@ -144,31 +171,78 @@ - -1 + +2 + +4 + +6 -7 - -14 +8 + +10 + +12 + +14 + +16 + +18 + +20 -20 - -26 +22 + +24 + +26 + +28 0 -0.25 +1653 -0.5 +3305 -0.75 +4958 -1 -metric line number -alpha - -Throughput - games per hour -method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk +6611 +chunk interval (10-game chunks) +games / hour + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Max score per eval cycle +campaign_v4_stdout.log - max of the 10 deterministic round scores per eval @@ -176,170 +250,108 @@ - -2 - -4 - -6 - -8 - -10 - -12 - -14 - -16 - -18 - -20 - -22 - -24 + +1 + +2 + +3 + +4 + +5 + +6 + +7 + +8 + +9 + +10 + +11 + +12 + +13 + +14 0 -1653 +52 -3305 +103 -4958 +155 -6611 -chunk interval (10-game chunks) -games / hour - - - - - - - - - - - - - - - - - - - - - - - - - - -Max score per eval cycle -campaign_v4_stdout.log - max of the 10 deterministic round scores per eval - - - - - - - - -1 - -2 - -3 - -4 - -5 - -6 - -7 - -8 - -9 - -10 - -11 - -12 - -0 - -52 - -103 - -155 - -207 -eval cycle number -best single-round score - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -Corners - 12 evals - -Crazy - 12 evals - -Target - 12 evals - -combined max - -How to read -Regenerate anytime: python3 tools/plot_progress.py -Test matches: dots are single fights, thick line shows trend. -Loss spikes are normal early; endless growth is bad. -Alpha high means experimenting; falling too fast freezes habits. -Throughput flat is healthy; dips mean something slowed. -Max score: best single-round score the bot managed in that eval cycle. -This file reloads itself in Chrome every sixty seconds. -Regenerate anytime with tools/watch_dashboard.sh or the python command. +207 +eval cycle number +best single-round score + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Corners - 14 evals + +Crazy - 14 evals + +Target - 14 evals + +combined max + +How to read +Regenerate anytime: python3 tools/plot_progress.py +Test matches: dots are single fights, thick line shows trend. +Loss spikes are normal early; endless growth is bad. +Throughput flat is healthy; dips mean something slowed. +Max score: best single-round score the bot managed in that eval cycle. +This file reloads itself in Chrome every sixty seconds. +Regenerate anytime with tools/watch_dashboard.sh or the python command. diff --git a/SAC_LSTM_Bot/tools/plot_progress.py b/SAC_LSTM_Bot/tools/plot_progress.py index 6f6121a..aca71ce 100644 --- a/SAC_LSTM_Bot/tools/plot_progress.py +++ b/SAC_LSTM_Bot/tools/plot_progress.py @@ -3,17 +3,17 @@ Pure-stdlib SVG output (matplotlib not available on this box). Generates ONE file: - docs/campaign_dashboard.svg - five panels, current (v2) run only: + docs/campaign_dashboard.svg - four panels, current (v2) run only: 1. test-match win % vs opponents (campaign_v4_stdout.log eval lines) - 2. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y) - 3. alpha temperature (training_metrics.jsonl, linear) - 4. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas; - 1 metrics row == one 10-game chunk, - counts match stdout chunk markers) - 5. max score per eval cycle (campaign_v4_stdout.log eval blocks; - eval_log.jsonl only ever holds the - LATEST cycle, so history comes from - the stdout log) + 2. critic_loss / |actor_loss| / alpha (training_metrics.jsonl; losses log + left axis, alpha linear right axis) + 3. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas; + 1 metrics row == one 10-game chunk, + counts match stdout chunk markers) + 4. max score per eval cycle (campaign_v4_stdout.log eval blocks; + eval_log.jsonl only ever holds the + LATEST cycle, so history comes from + the stdout log) plus an embedded JS snippet that reloads the page every 60 s when the SVG is opened as a top-level document in Chrome. @@ -42,10 +42,10 @@ GAMES_PER_ROW = 10 # one training_metrics.jsonl row per 10-round chunk COLORS = {"Corners": "#d62728", "Crazy": "#1f77b4", "Target": "#2ca02c", "RamFire": "#ff7f0e", "SacTwin": "#9467bd"} CRITIC_C, ACTOR_C = "#1f77b4", "#ff7f0e" -ALPHA_C = "#9467bd" # same purple as the dedicated alpha panel +ALPHA_C = "#9467bd" # alpha line on the losses panel RELOAD_JS = ('') -W, H = 1400, 2520 +W, H = 1400, 1720 TITLE_H, GUIDE_H = 80, 180 # rows: (header_y, panel_top_y, panel_bottom_y, x_left, x_right) C1_L, C1_R = 70, 697 @@ -55,19 +55,16 @@ ROWS = { 2: (100, 120, 720, C2_L, C2_R), 3: (940, 958, 1428, C1_L, C1_R), 4: (940, 958, 1428, C2_L, C2_R), - 5: (1648, 1666, 2270, C1_L, C1_R), # third row, left column; right slot empty } PANEL_TITLES = [ "Test matches - win % vs opponents", - "Training losses & alpha (log scale)", - "Alpha temperature", + "Training losses (log) & alpha (linear)", "Throughput - games per hour", "Max score per eval cycle", ] GUIDE_LINES = [ "Test matches: dots are single fights, thick line shows trend.", "Loss spikes are normal early; endless growth is bad.", - "Alpha high means experimenting; falling too fast freezes habits.", "Throughput flat is healthy; dips mean something slowed.", "Max score: best single-round score the bot managed in that eval cycle.", "This file reloads itself in Chrome every sixty seconds.", @@ -218,9 +215,9 @@ def dots(pts, color, r=2, opacity=0.25): f'fill="{color}" opacity="{opacity}"/>\n' for x, y in pts) -def hgrid(x0, x1, ys): +def hgrid(x0, x1, ys, color="#dddddd"): return "".join(f'\n' for y in ys) + f'stroke="{color}"/>\n' for y in ys) def axis(x0, y0, x1, y1, xt, yt, xlabel, ylabel, ylog=False): @@ -368,45 +365,40 @@ def panel_losses(rows, geo): s += f'' \ "no valid loss points\n" return - allv = [v for _, v in critic + actor + alpha] - lo, hi = math.floor(math.log10(min(allv))), math.ceil(math.log10(max(allv))) - if lo == hi: - hi = lo + 1 + x1 -= 46 # room on the right for the twin alpha axis labels n = len(rows) xm = lambda i: x0 + (x1 - x0) * i / max(n - 1, 1) + base = critic + actor or alpha # log-domain source (losses in practice) + lo = math.floor(math.log10(min(v for _, v in base))) + hi = math.ceil(math.log10(max(v for _, v in base))) + if lo == hi: + hi = lo + 1 ym = map_fn(pb, pt, lo, hi, log=True) s += hgrid(x0, x1, [ym(10 ** e) for e in range(lo, hi + 1)]) s += axis(x0, pb, x1, pt, ticks_linear(1, n, x0, x1, n=5), ticks_log(lo, hi, pb, pt), "metric line number", "loss (log)") s += polyline([(xm(i), ym(v)) for i, v in critic], CRITIC_C, 1.8) s += polyline([(xm(i), ym(v)) for i, v in actor], ACTOR_C, 1.8) - s += polyline([(xm(i), ym(v)) for i, v in alpha], ALPHA_C, 1.8) + if alpha: # twin axis: alpha on its own linear scale, purple like the line + ahi = max(1.0, max(v for _, v in alpha)) + yma = map_fn(pb, pt, 0, ahi) + s += hgrid(x0, x1, [yma(ahi * k / 4) for k in range(5)], "#e9dcf5") + mid = (pt + pb) // 2 + s += f'\n' + for v, py in ticks_linear(0, ahi, pb, pt, n=5, fmt="{:.3g}"): + s += (f'\n' + f'{esc(v)}\n') + s += (f'alpha\n') + s += polyline([(xm(i), yma(v)) for i, v in alpha], ALPHA_C, 1.8) items = [(CRITIC_C, "critic_loss"), (ACTOR_C, "|actor_loss|")] if alpha: items.append((ALPHA_C, "alpha")) s += legend(items, x0 + 12, pb + 52) return s -def panel_alpha(rows, geo): - _, pt, pb, x0, x1 = geo - s = "" - alpha = metric_col(rows, "alpha") - if not alpha: - s += f'' \ - "no alpha points\n" - return - n = len(rows) - hi = max(1.0, max(v for _, v in alpha)) - xm = lambda i: x0 + (x1 - x0) * i / max(n - 1, 1) - ym = map_fn(pb, pt, 0, hi) - s += hgrid(x0, x1, [ym(v) for v in - [hi * k / 4 for k in range(5)]]) - s += axis(x0, pb, x1, pt, ticks_linear(1, n, x0, x1, n=5), - ticks_linear(0, hi, pb, pt, n=5, fmt="{:.3g}"), - "metric line number", "alpha") - s += polyline([(xm(i), ym(v)) for i, v in alpha], "#9467bd", 1.8) - return s - def panel_throughput(rows, geo): _, pt, pb, x0, x1 = geo s = "" @@ -470,14 +462,11 @@ def build_dashboard(campaign, metrics_f, out): "training_metrics.jsonl - big early spikes are normal", lambda: panel_losses(rows, ROWS[2])), (ROWS[3], PANEL_TITLES[2], - "training_metrics.jsonl - high = exploring, low = exploiting", - lambda: panel_alpha(rows, ROWS[3])), - (ROWS[4], PANEL_TITLES[3], "method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk", - lambda: panel_throughput(rows, ROWS[4])), - (ROWS[5], PANEL_TITLES[4], + lambda: panel_throughput(rows, ROWS[3])), + (ROWS[4], PANEL_TITLES[3], "campaign_v4_stdout.log - max of the 10 deterministic round scores per eval", - lambda: panel_max_score(maxes, ROWS[5])), + lambda: panel_max_score(maxes, ROWS[4])), ] for geo, title, sub, drawer in drawers: s += header(geo[3], geo[0], title, sub) @@ -540,14 +529,18 @@ def selftest(): assert esc(t) in text, f"panel title missing: {t}" assert text.count(PANEL_TITLES[0]) == 1 assert "How to read" in text, "reading guide missing" - assert 'width="1400"' in text and 'height="2520"' in text - # circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 4) - # + 2 cycles x 2 opponents max-score dots (panel 5) + assert 'width="1400"' in text and 'height="1720"' in text + # circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 3) + # + 2 cycles x 2 opponents max-score dots (panel 4) assert text.count("alpha<") == 2, text.count(">alpha<") + assert text.count('fill="#9467bd"') == 6, text.count('fill="#9467bd"') assert "combined max" in text, "max-score combined line missing" # orientation guard: a known rising series (10% -> 90%) rendered through