feat(dashboard): max-score-per-eval-cycle panel

This commit is contained in:
2026-08-24 09:15:24 +02:00
parent 8872be3ff6
commit 4a6e3347c6
2 changed files with 333 additions and 130 deletions
+109 -9
View File
@@ -3,13 +3,17 @@
Pure-stdlib SVG output (matplotlib not available on this box).
Generates ONE file:
docs/campaign_dashboard.svg - four panels, current (v2) run only:
docs/campaign_dashboard.svg - five panels, current (v2) run only:
1. test-match win % vs opponents (campaign_v4_stdout.log eval lines)
2. critic_loss / |actor_loss| / alpha (training_metrics.jsonl, shared log-y)
2. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y)
3. alpha temperature (training_metrics.jsonl, linear)
4. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas;
1 metrics row == one 10-game chunk,
counts match stdout chunk markers)
5. max score per eval cycle (campaign_v4_stdout.log eval blocks;
eval_log.jsonl only ever holds the
LATEST cycle, so history comes from
the stdout log)
plus an embedded JS snippet that reloads the page every 60 s when the SVG is
opened as a top-level document in Chrome.
@@ -30,6 +34,8 @@ import xml.etree.ElementTree as ET
ROOT = Path(__file__).resolve().parent.parent
EVAL_RE = re.compile(r">>> \[eval\] win rate: (\d+)/(\d+) \(([\d.]+)%\) vs (\S+)")
EVAL_BLOCK_RE = re.compile(r">>> \[eval\] \d+ deterministic rounds vs (\S+)")
ROUND_RE = re.compile(r"Round \d+/\d+\D+ticks:\d+ score:(\d+) win:")
TREND_WINDOW = 10 # rolling mean shown as the thick trend line (panel 1)
RATE_BUCKET = 20 # metric intervals per throughput bucket (~200 games)
GAMES_PER_ROW = 10 # one training_metrics.jsonl row per 10-round chunk
@@ -39,7 +45,7 @@ CRITIC_C, ACTOR_C = "#1f77b4", "#ff7f0e"
ALPHA_C = "#9467bd" # same purple as the dedicated alpha panel
RELOAD_JS = ('<script type="text/javascript"><![CDATA[ '
'setTimeout(function(){ location.reload(); }, 60000); ]]></script>')
W, H = 1400, 1720
W, H = 1400, 2520
TITLE_H, GUIDE_H = 80, 180
# rows: (header_y, panel_top_y, panel_bottom_y, x_left, x_right)
C1_L, C1_R = 70, 697
@@ -49,18 +55,21 @@ ROWS = {
2: (100, 120, 720, C2_L, C2_R),
3: (940, 958, 1428, C1_L, C1_R),
4: (940, 958, 1428, C2_L, C2_R),
5: (1648, 1666, 2270, C1_L, C1_R), # third row, left column; right slot empty
}
PANEL_TITLES = [
"Test matches - win % vs opponents",
"Training losses & alpha (log scale)",
"Alpha temperature",
"Throughput - games per hour",
"Max score per eval cycle",
]
GUIDE_LINES = [
"Test matches: dots are single fights, thick line shows trend.",
"Loss spikes are normal early; endless growth is bad.",
"Alpha high means experimenting; falling too fast freezes habits.",
"Throughput flat is healthy; dips mean something slowed.",
"Max score: best single-round score the bot managed in that eval cycle.",
"This file reloads itself in Chrome every sixty seconds.",
"Regenerate anytime with tools/watch_dashboard.sh or the python command.",
]
@@ -86,6 +95,38 @@ def parse_eval_series(path):
return series
def parse_max_scores(path):
"""Return {opponent: [best single-round score per eval cycle, in file order]}.
eval_log.jsonl is atomically overwritten every cycle (sac_train.sh mv), so
per-cycle history only exists in the stdout log: each eval prints a
'>>> [eval] N deterministic rounds vs X' header, then Round/score lines,
closed by the '[eval] win rate' (or crashed / no results) line. Training
rounds share the Round-line format, so they are ignored unless inside a block.
"""
out, cur, best = {}, None, None
if not path.is_file():
print(f"[skip] campaign log not found: {path}")
return out
for line in path.read_text(errors="replace").splitlines():
m = EVAL_BLOCK_RE.search(line)
if m:
cur, best = m.group(1), None
continue
if cur is None:
continue
if "[eval]" in line: # win-rate / crashed / no-results closes the block
if best is not None:
out.setdefault(cur, []).append(best)
cur, best = None, None
continue
m = ROUND_RE.search(line)
if m:
v = int(m.group(1))
best = v if best is None else max(best, v)
return out
def parse_metrics(path):
"""Return list of metric dicts, skipping malformed lines."""
rows = []
@@ -280,6 +321,42 @@ def panel_test_matches(series, geo):
s += legend(items, x0 + 12, pb + 52)
return s
def panel_max_score(maxes, geo):
_, pt, pb, x0, x1 = geo
s = ""
if not maxes:
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
"no eval lines found</text>\n"
return
ncyc = max(len(v) for v in maxes.values())
xmax = max(ncyc, 2)
hi = max(1, max(v for vals in maxes.values() for v in vals)) * 1.1
xm, ym = map_fn(x0, x1, 1, xmax), map_fn(pb, pt, 0, hi)
s += hgrid(x0, x1, [ym(hi * k / 4) for k in range(5)])
step = max(1, xmax // 8)
xt = [(str(v), xm(v)) for v in range(step, xmax + 1, step)] or [("1", xm(1))]
s += axis(x0, pb, x1, pt, xt, ticks_linear(0, hi, pb, pt, n=5),
"eval cycle number", "best single-round score")
items, per_opp = [], {}
for name in ("Corners", "Crazy", "Target"):
vals = maxes.get(name, [])
if not vals:
continue
c = COLORS[name]
pts = [(xm(i + 1), ym(v)) for i, v in enumerate(vals)]
per_opp[name] = vals
s += dots(pts, c)
s += polyline(pts, c, 2.5)
items.append((c, f"{name} - {len(vals)} evals"))
if len(items) >= 2: # combined best across opponents, cycle-aligned
comb = [max(vals[i] for vals in per_opp.values() if i < len(vals))
for i in range(ncyc)]
s += polyline([(xm(i + 1), ym(v)) for i, v in enumerate(comb)],
"#555555", 2.5, dash="6 4")
items.append(("#555555", "combined max"))
s += legend(items, x0 + 12, pb + 52)
return s
def panel_losses(rows, geo):
_, pt, pb, x0, x1 = geo
s = ""
@@ -370,6 +447,9 @@ def build_dashboard(campaign, metrics_f, out):
series = parse_eval_series(campaign)
print("[info] evals parsed: " +
(", ".join(f"{k}={len(v)}" for k, v in sorted(series.items())) or "(none)"))
maxes = parse_max_scores(campaign)
print("[info] eval max-score cycles parsed: " +
(", ".join(f"{k}={len(v)}" for k, v in sorted(maxes.items())) or "(none)"))
rows = parse_metrics(metrics_f)
print(f"[info] metric rows parsed: {len(rows)}")
@@ -395,10 +475,13 @@ def build_dashboard(campaign, metrics_f, out):
(ROWS[4], PANEL_TITLES[3],
"method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk",
lambda: panel_throughput(rows, ROWS[4])),
(ROWS[5], PANEL_TITLES[4],
"campaign_v4_stdout.log - max of the 10 deterministic round scores per eval",
lambda: panel_max_score(maxes, ROWS[5])),
]
for geo, title, sub, drawer in drawers:
s += header(geo[3], geo[0], title, sub)
s += drawer()
s += drawer() or "" # panels bare-return None on their no-data path
s += guide_block(W, H, GUIDE_LINES)
s += RELOAD_JS + "\n"
@@ -412,14 +495,28 @@ def selftest():
with tempfile.TemporaryDirectory() as td:
td = Path(td)
(td / "log").write_text(
">>> [eval] win rate: 3/10 (30%) vs Corners\n"
">>> [eval] 10 deterministic rounds vs Corners\n"
"Round 1/10 - ticks:100 score:3 win:true\n"
"garbage line\n"
"Round 2/10 - ticks:100 score:7 win:false\n"
">>> [eval] win rate: 3/10 (30%) vs Corners\n"
">>> [eval] 10 deterministic rounds vs Crazy\n"
"Round 1/10 - ticks:100 score:70 win:false\n"
">>> [eval] win rate: 7/10 (70%) vs Crazy\n"
">>> [eval] win rate: broken\n"
">>> [eval] broken\n"
"Round 9/9 - ticks:1 score:999 win:false\n"
">>> [eval] 10 deterministic rounds vs Corners\n"
"Round 1/10 - ticks:100 score:50 win:true\n"
">>> [eval] win rate: 5/10 (50%) vs Corners\n"
">>> [eval] 10 deterministic rounds vs Crazy\n"
"Round 1/10 - ticks:100 score:40 win:true\n"
">>> [eval] win rate: 4/10 (40%) vs Crazy\n")
ser = parse_eval_series(td / "log")
assert ser == {"Corners": [30.0, 50.0], "Crazy": [70.0, 40.0]}, ser
mxs = parse_max_scores(td / "log")
# stray Round 999 after the unclosed '[eval] broken' line is ignored;
# per-cycle max of the Round scores above
assert mxs == {"Corners": [7, 50], "Crazy": [70, 40]}, mxs
assert rolling([10] * 25, 20)[-1] == 10.0
assert rolling([1, 2, 3], 20) == [1.0, 1.5, 2.0]
assert len(bucket_means(list(range(1287)), RATE_BUCKET)) == RATE_BUCKET
@@ -443,12 +540,15 @@ def selftest():
assert esc(t) in text, f"panel title missing: {t}"
assert text.count(PANEL_TITLES[0]) == 1
assert "How to read" in text, "reading guide missing"
assert 'width="1400"' in text and 'height="1720"' in text
assert 'width="1400"' in text and 'height="2520"' in text
# circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 4)
assert text.count("<circle") == 6, text.count("<circle")
# + 2 cycles x 2 opponents max-score dots (panel 5)
assert text.count("<circle") == 10, text.count("<circle")
# 2 trends + 3 losses-panel polylines (critic, actor, alpha)
# + 1 dedicated alpha panel + 1 throughput
assert text.count("<polyline") == 7, text.count("<polyline")
# + 3 max-score panel (Corners, Crazy, combined; Target absent in fixture)
assert text.count("<polyline") == 10, text.count("<polyline")
assert "combined max" in text, "max-score combined line missing"
# orientation guard: a known rising series (10% -> 90%) rendered through
# the FULL build path must plot upward (smaller SVG y) and forward in