feat(dashboard): max-score-per-eval-cycle panel
This commit is contained in:
@@ -3,13 +3,17 @@
|
||||
|
||||
Pure-stdlib SVG output (matplotlib not available on this box).
|
||||
Generates ONE file:
|
||||
docs/campaign_dashboard.svg - four panels, current (v2) run only:
|
||||
docs/campaign_dashboard.svg - five panels, current (v2) run only:
|
||||
1. test-match win % vs opponents (campaign_v4_stdout.log eval lines)
|
||||
2. critic_loss / |actor_loss| / alpha (training_metrics.jsonl, shared log-y)
|
||||
2. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y)
|
||||
3. alpha temperature (training_metrics.jsonl, linear)
|
||||
4. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas;
|
||||
1 metrics row == one 10-game chunk,
|
||||
counts match stdout chunk markers)
|
||||
5. max score per eval cycle (campaign_v4_stdout.log eval blocks;
|
||||
eval_log.jsonl only ever holds the
|
||||
LATEST cycle, so history comes from
|
||||
the stdout log)
|
||||
plus an embedded JS snippet that reloads the page every 60 s when the SVG is
|
||||
opened as a top-level document in Chrome.
|
||||
|
||||
@@ -30,6 +34,8 @@ import xml.etree.ElementTree as ET
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
EVAL_RE = re.compile(r">>> \[eval\] win rate: (\d+)/(\d+) \(([\d.]+)%\) vs (\S+)")
|
||||
EVAL_BLOCK_RE = re.compile(r">>> \[eval\] \d+ deterministic rounds vs (\S+)")
|
||||
ROUND_RE = re.compile(r"Round \d+/\d+\D+ticks:\d+ score:(\d+) win:")
|
||||
TREND_WINDOW = 10 # rolling mean shown as the thick trend line (panel 1)
|
||||
RATE_BUCKET = 20 # metric intervals per throughput bucket (~200 games)
|
||||
GAMES_PER_ROW = 10 # one training_metrics.jsonl row per 10-round chunk
|
||||
@@ -39,7 +45,7 @@ CRITIC_C, ACTOR_C = "#1f77b4", "#ff7f0e"
|
||||
ALPHA_C = "#9467bd" # same purple as the dedicated alpha panel
|
||||
RELOAD_JS = ('<script type="text/javascript"><![CDATA[ '
|
||||
'setTimeout(function(){ location.reload(); }, 60000); ]]></script>')
|
||||
W, H = 1400, 1720
|
||||
W, H = 1400, 2520
|
||||
TITLE_H, GUIDE_H = 80, 180
|
||||
# rows: (header_y, panel_top_y, panel_bottom_y, x_left, x_right)
|
||||
C1_L, C1_R = 70, 697
|
||||
@@ -49,18 +55,21 @@ ROWS = {
|
||||
2: (100, 120, 720, C2_L, C2_R),
|
||||
3: (940, 958, 1428, C1_L, C1_R),
|
||||
4: (940, 958, 1428, C2_L, C2_R),
|
||||
5: (1648, 1666, 2270, C1_L, C1_R), # third row, left column; right slot empty
|
||||
}
|
||||
PANEL_TITLES = [
|
||||
"Test matches - win % vs opponents",
|
||||
"Training losses & alpha (log scale)",
|
||||
"Alpha temperature",
|
||||
"Throughput - games per hour",
|
||||
"Max score per eval cycle",
|
||||
]
|
||||
GUIDE_LINES = [
|
||||
"Test matches: dots are single fights, thick line shows trend.",
|
||||
"Loss spikes are normal early; endless growth is bad.",
|
||||
"Alpha high means experimenting; falling too fast freezes habits.",
|
||||
"Throughput flat is healthy; dips mean something slowed.",
|
||||
"Max score: best single-round score the bot managed in that eval cycle.",
|
||||
"This file reloads itself in Chrome every sixty seconds.",
|
||||
"Regenerate anytime with tools/watch_dashboard.sh or the python command.",
|
||||
]
|
||||
@@ -86,6 +95,38 @@ def parse_eval_series(path):
|
||||
return series
|
||||
|
||||
|
||||
def parse_max_scores(path):
|
||||
"""Return {opponent: [best single-round score per eval cycle, in file order]}.
|
||||
|
||||
eval_log.jsonl is atomically overwritten every cycle (sac_train.sh mv), so
|
||||
per-cycle history only exists in the stdout log: each eval prints a
|
||||
'>>> [eval] N deterministic rounds vs X' header, then Round/score lines,
|
||||
closed by the '[eval] win rate' (or crashed / no results) line. Training
|
||||
rounds share the Round-line format, so they are ignored unless inside a block.
|
||||
"""
|
||||
out, cur, best = {}, None, None
|
||||
if not path.is_file():
|
||||
print(f"[skip] campaign log not found: {path}")
|
||||
return out
|
||||
for line in path.read_text(errors="replace").splitlines():
|
||||
m = EVAL_BLOCK_RE.search(line)
|
||||
if m:
|
||||
cur, best = m.group(1), None
|
||||
continue
|
||||
if cur is None:
|
||||
continue
|
||||
if "[eval]" in line: # win-rate / crashed / no-results closes the block
|
||||
if best is not None:
|
||||
out.setdefault(cur, []).append(best)
|
||||
cur, best = None, None
|
||||
continue
|
||||
m = ROUND_RE.search(line)
|
||||
if m:
|
||||
v = int(m.group(1))
|
||||
best = v if best is None else max(best, v)
|
||||
return out
|
||||
|
||||
|
||||
def parse_metrics(path):
|
||||
"""Return list of metric dicts, skipping malformed lines."""
|
||||
rows = []
|
||||
@@ -280,6 +321,42 @@ def panel_test_matches(series, geo):
|
||||
s += legend(items, x0 + 12, pb + 52)
|
||||
return s
|
||||
|
||||
def panel_max_score(maxes, geo):
|
||||
_, pt, pb, x0, x1 = geo
|
||||
s = ""
|
||||
if not maxes:
|
||||
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
|
||||
"no eval lines found</text>\n"
|
||||
return
|
||||
ncyc = max(len(v) for v in maxes.values())
|
||||
xmax = max(ncyc, 2)
|
||||
hi = max(1, max(v for vals in maxes.values() for v in vals)) * 1.1
|
||||
xm, ym = map_fn(x0, x1, 1, xmax), map_fn(pb, pt, 0, hi)
|
||||
s += hgrid(x0, x1, [ym(hi * k / 4) for k in range(5)])
|
||||
step = max(1, xmax // 8)
|
||||
xt = [(str(v), xm(v)) for v in range(step, xmax + 1, step)] or [("1", xm(1))]
|
||||
s += axis(x0, pb, x1, pt, xt, ticks_linear(0, hi, pb, pt, n=5),
|
||||
"eval cycle number", "best single-round score")
|
||||
items, per_opp = [], {}
|
||||
for name in ("Corners", "Crazy", "Target"):
|
||||
vals = maxes.get(name, [])
|
||||
if not vals:
|
||||
continue
|
||||
c = COLORS[name]
|
||||
pts = [(xm(i + 1), ym(v)) for i, v in enumerate(vals)]
|
||||
per_opp[name] = vals
|
||||
s += dots(pts, c)
|
||||
s += polyline(pts, c, 2.5)
|
||||
items.append((c, f"{name} - {len(vals)} evals"))
|
||||
if len(items) >= 2: # combined best across opponents, cycle-aligned
|
||||
comb = [max(vals[i] for vals in per_opp.values() if i < len(vals))
|
||||
for i in range(ncyc)]
|
||||
s += polyline([(xm(i + 1), ym(v)) for i, v in enumerate(comb)],
|
||||
"#555555", 2.5, dash="6 4")
|
||||
items.append(("#555555", "combined max"))
|
||||
s += legend(items, x0 + 12, pb + 52)
|
||||
return s
|
||||
|
||||
def panel_losses(rows, geo):
|
||||
_, pt, pb, x0, x1 = geo
|
||||
s = ""
|
||||
@@ -370,6 +447,9 @@ def build_dashboard(campaign, metrics_f, out):
|
||||
series = parse_eval_series(campaign)
|
||||
print("[info] evals parsed: " +
|
||||
(", ".join(f"{k}={len(v)}" for k, v in sorted(series.items())) or "(none)"))
|
||||
maxes = parse_max_scores(campaign)
|
||||
print("[info] eval max-score cycles parsed: " +
|
||||
(", ".join(f"{k}={len(v)}" for k, v in sorted(maxes.items())) or "(none)"))
|
||||
rows = parse_metrics(metrics_f)
|
||||
print(f"[info] metric rows parsed: {len(rows)}")
|
||||
|
||||
@@ -395,10 +475,13 @@ def build_dashboard(campaign, metrics_f, out):
|
||||
(ROWS[4], PANEL_TITLES[3],
|
||||
"method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk",
|
||||
lambda: panel_throughput(rows, ROWS[4])),
|
||||
(ROWS[5], PANEL_TITLES[4],
|
||||
"campaign_v4_stdout.log - max of the 10 deterministic round scores per eval",
|
||||
lambda: panel_max_score(maxes, ROWS[5])),
|
||||
]
|
||||
for geo, title, sub, drawer in drawers:
|
||||
s += header(geo[3], geo[0], title, sub)
|
||||
s += drawer()
|
||||
s += drawer() or "" # panels bare-return None on their no-data path
|
||||
|
||||
s += guide_block(W, H, GUIDE_LINES)
|
||||
s += RELOAD_JS + "\n"
|
||||
@@ -412,14 +495,28 @@ def selftest():
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
td = Path(td)
|
||||
(td / "log").write_text(
|
||||
">>> [eval] win rate: 3/10 (30%) vs Corners\n"
|
||||
">>> [eval] 10 deterministic rounds vs Corners\n"
|
||||
"Round 1/10 - ticks:100 score:3 win:true\n"
|
||||
"garbage line\n"
|
||||
"Round 2/10 - ticks:100 score:7 win:false\n"
|
||||
">>> [eval] win rate: 3/10 (30%) vs Corners\n"
|
||||
">>> [eval] 10 deterministic rounds vs Crazy\n"
|
||||
"Round 1/10 - ticks:100 score:70 win:false\n"
|
||||
">>> [eval] win rate: 7/10 (70%) vs Crazy\n"
|
||||
">>> [eval] win rate: broken\n"
|
||||
">>> [eval] broken\n"
|
||||
"Round 9/9 - ticks:1 score:999 win:false\n"
|
||||
">>> [eval] 10 deterministic rounds vs Corners\n"
|
||||
"Round 1/10 - ticks:100 score:50 win:true\n"
|
||||
">>> [eval] win rate: 5/10 (50%) vs Corners\n"
|
||||
">>> [eval] 10 deterministic rounds vs Crazy\n"
|
||||
"Round 1/10 - ticks:100 score:40 win:true\n"
|
||||
">>> [eval] win rate: 4/10 (40%) vs Crazy\n")
|
||||
ser = parse_eval_series(td / "log")
|
||||
assert ser == {"Corners": [30.0, 50.0], "Crazy": [70.0, 40.0]}, ser
|
||||
mxs = parse_max_scores(td / "log")
|
||||
# stray Round 999 after the unclosed '[eval] broken' line is ignored;
|
||||
# per-cycle max of the Round scores above
|
||||
assert mxs == {"Corners": [7, 50], "Crazy": [70, 40]}, mxs
|
||||
assert rolling([10] * 25, 20)[-1] == 10.0
|
||||
assert rolling([1, 2, 3], 20) == [1.0, 1.5, 2.0]
|
||||
assert len(bucket_means(list(range(1287)), RATE_BUCKET)) == RATE_BUCKET
|
||||
@@ -443,12 +540,15 @@ def selftest():
|
||||
assert esc(t) in text, f"panel title missing: {t}"
|
||||
assert text.count(PANEL_TITLES[0]) == 1
|
||||
assert "How to read" in text, "reading guide missing"
|
||||
assert 'width="1400"' in text and 'height="1720"' in text
|
||||
assert 'width="1400"' in text and 'height="2520"' in text
|
||||
# circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 4)
|
||||
assert text.count("<circle") == 6, text.count("<circle")
|
||||
# + 2 cycles x 2 opponents max-score dots (panel 5)
|
||||
assert text.count("<circle") == 10, text.count("<circle")
|
||||
# 2 trends + 3 losses-panel polylines (critic, actor, alpha)
|
||||
# + 1 dedicated alpha panel + 1 throughput
|
||||
assert text.count("<polyline") == 7, text.count("<polyline")
|
||||
# + 3 max-score panel (Corners, Crazy, combined; Target absent in fixture)
|
||||
assert text.count("<polyline") == 10, text.count("<polyline")
|
||||
assert "combined max" in text, "max-score combined line missing"
|
||||
|
||||
# orientation guard: a known rising series (10% -> 90%) rendered through
|
||||
# the FULL build path must plot upward (smaller SVG y) and forward in
|
||||
|
||||
Reference in New Issue
Block a user