fix(dashboard): dual y-axis for losses+alpha panel

This commit is contained in:
2026-08-24 09:21:43 +02:00
parent 4a6e3347c6
commit 55d69a9352
2 changed files with 319 additions and 314 deletions
+47 -54
View File
@@ -3,17 +3,17 @@
Pure-stdlib SVG output (matplotlib not available on this box).
Generates ONE file:
docs/campaign_dashboard.svg - five panels, current (v2) run only:
docs/campaign_dashboard.svg - four panels, current (v2) run only:
1. test-match win % vs opponents (campaign_v4_stdout.log eval lines)
2. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y)
3. alpha temperature (training_metrics.jsonl, linear)
4. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas;
1 metrics row == one 10-game chunk,
counts match stdout chunk markers)
5. max score per eval cycle (campaign_v4_stdout.log eval blocks;
eval_log.jsonl only ever holds the
LATEST cycle, so history comes from
the stdout log)
2. critic_loss / |actor_loss| / alpha (training_metrics.jsonl; losses log
left axis, alpha linear right axis)
3. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas;
1 metrics row == one 10-game chunk,
counts match stdout chunk markers)
4. max score per eval cycle (campaign_v4_stdout.log eval blocks;
eval_log.jsonl only ever holds the
LATEST cycle, so history comes from
the stdout log)
plus an embedded JS snippet that reloads the page every 60 s when the SVG is
opened as a top-level document in Chrome.
@@ -42,10 +42,10 @@ GAMES_PER_ROW = 10 # one training_metrics.jsonl row per 10-round chunk
COLORS = {"Corners": "#d62728", "Crazy": "#1f77b4", "Target": "#2ca02c",
"RamFire": "#ff7f0e", "SacTwin": "#9467bd"}
CRITIC_C, ACTOR_C = "#1f77b4", "#ff7f0e"
ALPHA_C = "#9467bd" # same purple as the dedicated alpha panel
ALPHA_C = "#9467bd" # alpha line on the losses panel
RELOAD_JS = ('<script type="text/javascript"><![CDATA[ '
'setTimeout(function(){ location.reload(); }, 60000); ]]></script>')
W, H = 1400, 2520
W, H = 1400, 1720
TITLE_H, GUIDE_H = 80, 180
# rows: (header_y, panel_top_y, panel_bottom_y, x_left, x_right)
C1_L, C1_R = 70, 697
@@ -55,19 +55,16 @@ ROWS = {
2: (100, 120, 720, C2_L, C2_R),
3: (940, 958, 1428, C1_L, C1_R),
4: (940, 958, 1428, C2_L, C2_R),
5: (1648, 1666, 2270, C1_L, C1_R), # third row, left column; right slot empty
}
PANEL_TITLES = [
"Test matches - win % vs opponents",
"Training losses & alpha (log scale)",
"Alpha temperature",
"Training losses (log) & alpha (linear)",
"Throughput - games per hour",
"Max score per eval cycle",
]
GUIDE_LINES = [
"Test matches: dots are single fights, thick line shows trend.",
"Loss spikes are normal early; endless growth is bad.",
"Alpha high means experimenting; falling too fast freezes habits.",
"Throughput flat is healthy; dips mean something slowed.",
"Max score: best single-round score the bot managed in that eval cycle.",
"This file reloads itself in Chrome every sixty seconds.",
@@ -218,9 +215,9 @@ def dots(pts, color, r=2, opacity=0.25):
f'fill="{color}" opacity="{opacity}"/>\n' for x, y in pts)
def hgrid(x0, x1, ys):
def hgrid(x0, x1, ys, color="#dddddd"):
return "".join(f'<line x1="{x0}" y1="{y:.1f}" x2="{x1}" y2="{y:.1f}" '
f'stroke="#dddddd"/>\n' for y in ys)
f'stroke="{color}"/>\n' for y in ys)
def axis(x0, y0, x1, y1, xt, yt, xlabel, ylabel, ylog=False):
@@ -368,45 +365,40 @@ def panel_losses(rows, geo):
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
"no valid loss points</text>\n"
return
allv = [v for _, v in critic + actor + alpha]
lo, hi = math.floor(math.log10(min(allv))), math.ceil(math.log10(max(allv)))
if lo == hi:
hi = lo + 1
x1 -= 46 # room on the right for the twin alpha axis labels
n = len(rows)
xm = lambda i: x0 + (x1 - x0) * i / max(n - 1, 1)
base = critic + actor or alpha # log-domain source (losses in practice)
lo = math.floor(math.log10(min(v for _, v in base)))
hi = math.ceil(math.log10(max(v for _, v in base)))
if lo == hi:
hi = lo + 1
ym = map_fn(pb, pt, lo, hi, log=True)
s += hgrid(x0, x1, [ym(10 ** e) for e in range(lo, hi + 1)])
s += axis(x0, pb, x1, pt, ticks_linear(1, n, x0, x1, n=5),
ticks_log(lo, hi, pb, pt), "metric line number", "loss (log)")
s += polyline([(xm(i), ym(v)) for i, v in critic], CRITIC_C, 1.8)
s += polyline([(xm(i), ym(v)) for i, v in actor], ACTOR_C, 1.8)
s += polyline([(xm(i), ym(v)) for i, v in alpha], ALPHA_C, 1.8)
if alpha: # twin axis: alpha on its own linear scale, purple like the line
ahi = max(1.0, max(v for _, v in alpha))
yma = map_fn(pb, pt, 0, ahi)
s += hgrid(x0, x1, [yma(ahi * k / 4) for k in range(5)], "#e9dcf5")
mid = (pt + pb) // 2
s += f'<line x1="{x1}" y1="{pb}" x2="{x1}" y2="{pt}" stroke="{ALPHA_C}"/>\n'
for v, py in ticks_linear(0, ahi, pb, pt, n=5, fmt="{:.3g}"):
s += (f'<line x1="{x1}" y1="{py:.1f}" x2="{x1 + 4}" y2="{py:.1f}" '
f'stroke="{ALPHA_C}"/>\n'
f'<text x="{x1 + 7}" y="{py + 4:.1f}" font-size="11" '
f'fill="{ALPHA_C}">{esc(v)}</text>\n')
s += (f'<text x="{x1 + 17}" y="{mid}" text-anchor="middle" font-size="12" '
f'fill="{ALPHA_C}" transform="rotate(90 {x1 + 17} {mid})">alpha</text>\n')
s += polyline([(xm(i), yma(v)) for i, v in alpha], ALPHA_C, 1.8)
items = [(CRITIC_C, "critic_loss"), (ACTOR_C, "|actor_loss|")]
if alpha:
items.append((ALPHA_C, "alpha"))
s += legend(items, x0 + 12, pb + 52)
return s
def panel_alpha(rows, geo):
_, pt, pb, x0, x1 = geo
s = ""
alpha = metric_col(rows, "alpha")
if not alpha:
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
"no alpha points</text>\n"
return
n = len(rows)
hi = max(1.0, max(v for _, v in alpha))
xm = lambda i: x0 + (x1 - x0) * i / max(n - 1, 1)
ym = map_fn(pb, pt, 0, hi)
s += hgrid(x0, x1, [ym(v) for v in
[hi * k / 4 for k in range(5)]])
s += axis(x0, pb, x1, pt, ticks_linear(1, n, x0, x1, n=5),
ticks_linear(0, hi, pb, pt, n=5, fmt="{:.3g}"),
"metric line number", "alpha")
s += polyline([(xm(i), ym(v)) for i, v in alpha], "#9467bd", 1.8)
return s
def panel_throughput(rows, geo):
_, pt, pb, x0, x1 = geo
s = ""
@@ -470,14 +462,11 @@ def build_dashboard(campaign, metrics_f, out):
"training_metrics.jsonl - big early spikes are normal",
lambda: panel_losses(rows, ROWS[2])),
(ROWS[3], PANEL_TITLES[2],
"training_metrics.jsonl - high = exploring, low = exploiting",
lambda: panel_alpha(rows, ROWS[3])),
(ROWS[4], PANEL_TITLES[3],
"method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk",
lambda: panel_throughput(rows, ROWS[4])),
(ROWS[5], PANEL_TITLES[4],
lambda: panel_throughput(rows, ROWS[3])),
(ROWS[4], PANEL_TITLES[3],
"campaign_v4_stdout.log - max of the 10 deterministic round scores per eval",
lambda: panel_max_score(maxes, ROWS[5])),
lambda: panel_max_score(maxes, ROWS[4])),
]
for geo, title, sub, drawer in drawers:
s += header(geo[3], geo[0], title, sub)
@@ -540,14 +529,18 @@ def selftest():
assert esc(t) in text, f"panel title missing: {t}"
assert text.count(PANEL_TITLES[0]) == 1
assert "How to read" in text, "reading guide missing"
assert 'width="1400"' in text and 'height="2520"' in text
# circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 4)
# + 2 cycles x 2 opponents max-score dots (panel 5)
assert 'width="1400"' in text and 'height="1720"' in text
# circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 3)
# + 2 cycles x 2 opponents max-score dots (panel 4)
assert text.count("<circle") == 10, text.count("<circle")
# 2 trends + 3 losses-panel polylines (critic, actor, alpha)
# + 1 dedicated alpha panel + 1 throughput
# + 1 throughput
# + 3 max-score panel (Corners, Crazy, combined; Target absent in fixture)
assert text.count("<polyline") == 10, text.count("<polyline")
assert text.count("<polyline") == 9, text.count("<polyline")
# losses panel twin axis: "alpha" text = legend + right ylabel; purple
# fills = 5 right-axis tick labels + rotated ylabel
assert text.count(">alpha<") == 2, text.count(">alpha<")
assert text.count('fill="#9467bd"') == 6, text.count('fill="#9467bd"')
assert "combined max" in text, "max-score combined line missing"
# orientation guard: a known rising series (10% -> 90%) rendered through