fix(dashboard): remove broken unused win%-per-opponent panel
This commit is contained in:
@@ -3,21 +3,18 @@
|
||||
|
||||
Pure-stdlib SVG output (matplotlib not available on this box).
|
||||
Generates ONE file:
|
||||
docs/campaign_dashboard.svg - five panels, current (v2) run only:
|
||||
docs/campaign_dashboard.svg - four panels, current (v2) run only:
|
||||
1. test-match win % vs opponents (campaign_v4_stdout.log eval lines)
|
||||
2. real-fight win % per opponent (training_log.jsonl, ~100-game buckets)
|
||||
3. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y)
|
||||
4. alpha temperature (training_metrics.jsonl, linear)
|
||||
5. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas;
|
||||
training_log.jsonl has NO timestamps -
|
||||
verified field names - and one metrics
|
||||
row == one 10-game chunk, counts match
|
||||
the stdout "=== Chunk N/N ===" markers)
|
||||
2. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y)
|
||||
3. alpha temperature (training_metrics.jsonl, linear)
|
||||
4. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas;
|
||||
1 metrics row == one 10-game chunk,
|
||||
counts match stdout chunk markers)
|
||||
plus an embedded JS snippet that reloads the page every 60 s when the SVG is
|
||||
opened as a top-level document in Chrome.
|
||||
|
||||
Usage:
|
||||
python3 tools/plot_progress.py [campaign_log] [metrics_jsonl] [games_jsonl] [outdir]
|
||||
python3 tools/plot_progress.py [campaign_log] [metrics_jsonl] [outdir]
|
||||
All args optional; defaults relative to the SAC_LSTM_Bot/ root (parent of tools/).
|
||||
python3 tools/plot_progress.py --selftest # tiny built-in sanity check
|
||||
"""
|
||||
@@ -34,7 +31,6 @@ import xml.etree.ElementTree as ET
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
EVAL_RE = re.compile(r">>> \[eval\] win rate: (\d+)/(\d+) \(([\d.]+)%\) vs (\S+)")
|
||||
TREND_WINDOW = 10 # rolling mean shown as the thick trend line (panel 1)
|
||||
GAME_BUCKET = 100 # games per bucket, real-fight panel
|
||||
RATE_BUCKET = 20 # metric intervals per throughput bucket (~200 games)
|
||||
GAMES_PER_ROW = 10 # one training_metrics.jsonl row per 10-round chunk
|
||||
COLORS = {"Corners": "#d62728", "Crazy": "#1f77b4", "Target": "#2ca02c",
|
||||
@@ -42,7 +38,7 @@ COLORS = {"Corners": "#d62728", "Crazy": "#1f77b4", "Target": "#2ca02c",
|
||||
CRITIC_C, ACTOR_C = "#1f77b4", "#ff7f0e"
|
||||
RELOAD_JS = ('<script type="text/javascript"><![CDATA[ '
|
||||
'setTimeout(function(){ location.reload(); }, 60000); ]]></script>')
|
||||
W, H = 1400, 2000
|
||||
W, H = 1400, 1720
|
||||
TITLE_H, GUIDE_H = 80, 180
|
||||
# rows: (header_y, panel_top_y, panel_bottom_y, x_left, x_right)
|
||||
C1_L, C1_R = 70, 697
|
||||
@@ -52,18 +48,15 @@ ROWS = {
|
||||
2: (100, 120, 720, C2_L, C2_R),
|
||||
3: (940, 958, 1428, C1_L, C1_R),
|
||||
4: (940, 958, 1428, C2_L, C2_R),
|
||||
5: (1600, 1618, 1780, C1_L, C2_R),
|
||||
}
|
||||
PANEL_TITLES = [
|
||||
"Test matches - win % vs opponents",
|
||||
"Real fights - win % per opponent",
|
||||
"Training losses (log scale)",
|
||||
"Alpha temperature",
|
||||
"Throughput - games per hour",
|
||||
]
|
||||
GUIDE_LINES = [
|
||||
"Test matches: dots are single fights, thick line shows trend.",
|
||||
"Real battles only. Rising lines mean the bot improves.",
|
||||
"Loss spikes are normal early; endless growth is bad.",
|
||||
"Alpha high means experimenting; falling too fast freezes habits.",
|
||||
"Throughput flat is healthy; dips mean something slowed.",
|
||||
@@ -109,25 +102,6 @@ def parse_metrics(path):
|
||||
return rows
|
||||
|
||||
|
||||
def parse_games(path):
|
||||
"""Return [(opponent, won_bool)] for type=='game' rows, skipping junk."""
|
||||
out = []
|
||||
if not path.is_file():
|
||||
print(f"[skip] games log not found: {path}")
|
||||
return out
|
||||
for line in path.read_text(errors="replace").splitlines():
|
||||
try:
|
||||
r = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if r.get("type") != "game":
|
||||
continue
|
||||
opp, win = r.get("opponent"), r.get("win")
|
||||
if isinstance(opp, str) and isinstance(win, bool):
|
||||
out.append((opp, win))
|
||||
return out
|
||||
|
||||
|
||||
def metric_col(rows, key, positive=False):
|
||||
"""[(index, value)] for float-parseable rows; abs() applied; optional >0 filter."""
|
||||
out = []
|
||||
@@ -305,53 +279,6 @@ def panel_test_matches(series, geo):
|
||||
s += legend(items, x0 + 12, pb + 52)
|
||||
return s
|
||||
|
||||
def panel_real_fights(games, geo):
|
||||
_, pt, pb, x0, x1 = geo
|
||||
s = ""
|
||||
if not games:
|
||||
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
|
||||
"no game rows found</text>\n"
|
||||
return
|
||||
edges = list(range(0, len(games) + 1, GAME_BUCKET))
|
||||
if edges[-1] != len(games):
|
||||
edges.append(len(games))
|
||||
buckets = list(zip(edges[:-1], edges[1:]))
|
||||
xm, ym = map_fn(x0, x1, 1, max(len(games), 2)), map_fn(pb, pt, 0, 100)
|
||||
s += hgrid(x0, x1, [ym(v) for v in range(0, 101, 20)])
|
||||
step = max(GAME_BUCKET, GAME_BUCKET * (len(games) // GAME_BUCKET // 8 + 1))
|
||||
xt = [(str(v), xm(v)) for v in range(step, len(games) + 1, step)]
|
||||
s += axis(x0, pb, x1, pt, xt, ticks_linear(0, 100, pb, pt, n=6),
|
||||
f"game number ({GAME_BUCKET}-game buckets)", "win %")
|
||||
opponents = []
|
||||
for opp, _ in games:
|
||||
if opp not in opponents:
|
||||
opponents.append(opp)
|
||||
items = []
|
||||
for name in opponents:
|
||||
c = COLORS.get(name, "#7f7f7f")
|
||||
by_b = {}
|
||||
for bi, (lo, hi) in enumerate(buckets):
|
||||
sub = [w for o, w in games[lo:hi] if o == name]
|
||||
if sub:
|
||||
by_b[bi] = 100.0 * sum(sub) / len(sub)
|
||||
pts = []
|
||||
segs, prev = [], None
|
||||
for bi in sorted(by_b):
|
||||
if prev is not None and bi != prev + 1:
|
||||
segs.append(pts)
|
||||
pts = []
|
||||
center = (buckets[bi][0] + buckets[bi][1]) / 2
|
||||
pts.append((xm(center), ym(by_b[bi])))
|
||||
prev = bi
|
||||
if len(pts) >= 2:
|
||||
segs.append(pts)
|
||||
for seg in segs:
|
||||
s += polyline(seg, c, 3.5)
|
||||
n_played = sum(1 for o, _ in games if o == name)
|
||||
items.append((c, f"{name} ({n_played} games)"))
|
||||
s += legend(items, x0 + 12, pb + 52)
|
||||
return s
|
||||
|
||||
def panel_losses(rows, geo):
|
||||
_, pt, pb, x0, x1 = geo
|
||||
s = ""
|
||||
@@ -434,14 +361,12 @@ def panel_throughput(rows, geo):
|
||||
|
||||
# ---------- assembly ----------
|
||||
|
||||
def build_dashboard(campaign, metrics_f, games_f, out):
|
||||
def build_dashboard(campaign, metrics_f, out):
|
||||
series = parse_eval_series(campaign)
|
||||
print("[info] evals parsed: " +
|
||||
(", ".join(f"{k}={len(v)}" for k, v in sorted(series.items())) or "(none)"))
|
||||
rows = parse_metrics(metrics_f)
|
||||
print(f"[info] metric rows parsed: {len(rows)}")
|
||||
games = parse_games(games_f)
|
||||
print(f"[info] game rows parsed: {len(games)}")
|
||||
|
||||
s = (f'<svg xmlns="http://www.w3.org/2000/svg" width="{W}" height="{H}" '
|
||||
f'viewBox="0 0 {W} {H}" font-family="sans-serif">\n'
|
||||
@@ -457,18 +382,14 @@ def build_dashboard(campaign, metrics_f, games_f, out):
|
||||
"raw dots = single test matches, thick = rolling-mean-%d" % TREND_WINDOW,
|
||||
lambda: panel_test_matches(series, ROWS[1])),
|
||||
(ROWS[2], PANEL_TITLES[1],
|
||||
"training_log.jsonl only - learning in REAL battles, not tests",
|
||||
lambda: panel_real_fights(games, ROWS[2])),
|
||||
(ROWS[3], PANEL_TITLES[2],
|
||||
"training_metrics.jsonl - big early spikes are normal",
|
||||
lambda: panel_losses(rows, ROWS[3])),
|
||||
(ROWS[4], PANEL_TITLES[3],
|
||||
lambda: panel_losses(rows, ROWS[2])),
|
||||
(ROWS[3], PANEL_TITLES[2],
|
||||
"training_metrics.jsonl - high = exploring, low = exploiting",
|
||||
lambda: panel_alpha(rows, ROWS[4])),
|
||||
(ROWS[5], PANEL_TITLES[4],
|
||||
"method: training_metrics.jsonl 'epoch' deltas (training_log.jsonl has "
|
||||
"no timestamps); 1 row = one 10-game chunk",
|
||||
lambda: panel_throughput(rows, ROWS[5])),
|
||||
lambda: panel_alpha(rows, ROWS[3])),
|
||||
(ROWS[4], PANEL_TITLES[3],
|
||||
"method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk",
|
||||
lambda: panel_throughput(rows, ROWS[4])),
|
||||
]
|
||||
for geo, title, sub, drawer in drawers:
|
||||
s += header(geo[3], geo[0], title, sub)
|
||||
@@ -506,20 +427,10 @@ def selftest():
|
||||
'{"epoch": 1090.0, "critic_loss": 50, "actor_loss": 3, "alpha": 0.2}\n')
|
||||
rows = parse_metrics(td / "m.jsonl")
|
||||
assert len(rows) == 3 and rows[1]["critic_loss"] == 100
|
||||
glines = []
|
||||
for i in range(150): # 2 full GAME_BUCKETs, both opponents in both
|
||||
glines.append(json.dumps(
|
||||
{"type": "game", "round": i % 10 + 1, "ticks": 100,
|
||||
"score": i % 3, "total_score": i, "win": i % 3 == 0,
|
||||
"opponent": ("Corners", "Crazy")[i % 2]}))
|
||||
(td / "g.jsonl").write_text("\n".join(glines) + "\n")
|
||||
games = parse_games(td / "g.jsonl")
|
||||
assert len(games) == 150 and games[0] == ("Corners", True)
|
||||
assert games[-1] == ("Crazy", False) # i=149: odd -> Crazy; 149%3!=0 -> loss
|
||||
assert metric_col(rows, "actor_loss") == [(0, 2.0), (1, 4.0), (2, 3.0)]
|
||||
|
||||
dash = td / "dash.svg"
|
||||
assert build_dashboard(td / "log", td / "m.jsonl", td / "g.jsonl", dash)
|
||||
assert build_dashboard(td / "log", td / "m.jsonl", dash)
|
||||
text = dash.read_text()
|
||||
ET.fromstring(text) # whole doc must parse -> closing tag present
|
||||
assert RELOAD_JS in text, "auto-reload script missing"
|
||||
@@ -527,11 +438,11 @@ def selftest():
|
||||
assert t in text, f"panel title missing: {t}"
|
||||
assert text.count(PANEL_TITLES[0]) == 1
|
||||
assert "How to read" in text, "reading guide missing"
|
||||
assert 'width="1400"' in text and 'height="2000"' in text
|
||||
# circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 5)
|
||||
assert 'width="1400"' in text and 'height="1720"' in text
|
||||
# circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 4)
|
||||
assert text.count("<circle") == 6, text.count("<circle")
|
||||
# 2 trends + 2 real-fight opp lines + 2 losses + 1 alpha + 1 throughput
|
||||
assert text.count("<polyline") == 8, text.count("<polyline")
|
||||
# 2 trends + 2 losses + 1 alpha + 1 throughput
|
||||
assert text.count("<polyline") == 6, text.count("<polyline")
|
||||
|
||||
# orientation guard: a known rising series (10% -> 90%) rendered through
|
||||
# the FULL build path must plot upward (smaller SVG y) and forward in
|
||||
@@ -541,7 +452,7 @@ def selftest():
|
||||
">>> [eval] win rate: 1/10 (10%) vs Corners\n"
|
||||
">>> [eval] win rate: 9/10 (90%) vs Corners\n")
|
||||
ori_dash = td / "ori.svg"
|
||||
assert build_dashboard(ori_log, td / "m.jsonl", td / "g.jsonl", ori_dash)
|
||||
assert build_dashboard(ori_log, td / "m.jsonl", ori_dash)
|
||||
m = re.search(r'<polyline points="([^"]+)"', ori_dash.read_text())
|
||||
pts = [tuple(map(float, p.split(","))) for p in m.group(1).split()]
|
||||
assert len(pts) == 2, pts
|
||||
@@ -558,12 +469,11 @@ def main():
|
||||
args = [a for a in sys.argv[1:] if not a.startswith("-")]
|
||||
campaign = Path(args[0]) if len(args) > 0 else ROOT / "campaign_v4_stdout.log"
|
||||
metrics = Path(args[1]) if len(args) > 1 else ROOT / "training_metrics.jsonl"
|
||||
games = Path(args[2]) if len(args) > 2 else ROOT / "training_log.jsonl"
|
||||
outdir = Path(args[3]) if len(args) > 3 else ROOT / "docs"
|
||||
outdir = Path(args[2]) if len(args) > 2 else ROOT / "docs"
|
||||
outdir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
try:
|
||||
ok = build_dashboard(campaign, metrics, games, outdir / "campaign_dashboard.svg")
|
||||
ok = build_dashboard(campaign, metrics, outdir / "campaign_dashboard.svg")
|
||||
except Exception as e:
|
||||
print(f"[error] dashboard build failed: {e}")
|
||||
ok = False
|
||||
|
||||
Reference in New Issue
Block a user