diff --git a/SAC_LSTM_Bot/docs/graph_eval_winrates.svg b/SAC_LSTM_Bot/docs/graph_eval_winrates.svg index e5f2fad..7c9f190 100644 --- a/SAC_LSTM_Bot/docs/graph_eval_winrates.svg +++ b/SAC_LSTM_Bot/docs/graph_eval_winrates.svg @@ -1,5 +1,5 @@ - - + + SAC-LSTM campaign: eval win rate vs opponents (Run 3 live, Run 1 archived) @@ -11,20 +11,22 @@ win % per test match — thick line = trend - -10 - -20 - -30 - -40 - -50 - -60 - -70 + +10 + +20 + +30 + +40 + +50 + +60 + +70 + +80 0 @@ -40,250 +42,258 @@ eval number win rate (%) - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + - + - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + -Corners — 79 evals +Corners — 82 evals -Crazy — 79 evals +Crazy — 81 evals -Target — 78 evals +Target — 81 evals @@ -320,9 +330,17 @@ fraction of run (start → end) win rate (%) - + Run 1 (old) — 1287 evals in 40 buckets Run 3 trend (rolling-10) + +How to read +Regenerate anytime: python3 tools/plot_progress.py +One small dot = one test match. 0 means lost all ten fights. +A thick line shows the recent trend. Up = learning. +Watch the three colors. Red = Corners. Blue = Crazy. Green = Target. +Gray line = old run. Red line = new run. Same enemy. +If red ends higher than gray, the new training worked. diff --git a/SAC_LSTM_Bot/docs/graph_losses.svg b/SAC_LSTM_Bot/docs/graph_losses.svg index 2297aa5..f7212a4 100644 --- a/SAC_LSTM_Bot/docs/graph_losses.svg +++ b/SAC_LSTM_Bot/docs/graph_losses.svg @@ -1,5 +1,5 @@ - - + + SAC-LSTM training losses (x = metric line number) critic_loss (log scale) @@ -7,13 +7,13 @@ 1 -41 +42 -81 +84 -121 +126 -161 +167 0.1 @@ -26,20 +26,20 @@ 1e+17 critic_loss - + actor_loss |abs| (log scale) 1 -41 +42 -81 +84 -121 +126 -161 +167 1 @@ -52,20 +52,20 @@ 1e+08 |actor_loss| - + alpha (temperature) 1 -41 +42 -81 +84 -121 +126 -161 +167 0 @@ -78,6 +78,14 @@ 1.01 alpha - -metric line number (161 rows) + +metric line number (167 rows) + +How to read +Regenerate anytime: python3 tools/plot_progress.py +These are internal error numbers. They are not game scores. +Critic and actor panels: big spikes are normal early. +Steady growth forever is bad. +Alpha near 1 = bot still experimenting. +If alpha falls too fast, the bot freezes its habits. diff --git a/SAC_LSTM_Bot/tools/plot_progress.py b/SAC_LSTM_Bot/tools/plot_progress.py index 2eae4dc..ed871b9 100644 --- a/SAC_LSTM_Bot/tools/plot_progress.py +++ b/SAC_LSTM_Bot/tools/plot_progress.py @@ -29,6 +29,21 @@ BUCKETS = 40 # v1 downsampling for the Run1-vs-Run3 panel COLORS = {"Corners": "#d62728", "Crazy": "#1f77b4", "Target": "#2ca02c"} V1_COLOR = "#999999" W, M_L, M_R, M_T, M_B = 1000, 70, 20, 40, 45 # graph 2 geometry (unchanged) +GUIDE_H = 124 # bottom band holding the plain-English "How to read" panel +WINRATE_GUIDE = [ + "One small dot = one test match. 0 means lost all ten fights.", + "A thick line shows the recent trend. Up = learning.", + "Watch the three colors. Red = Corners. Blue = Crazy. Green = Target.", + "Gray line = old run. Red line = new run. Same enemy.", + "If red ends higher than gray, the new training worked.", +] +LOSSES_GUIDE = [ + "These are internal error numbers. They are not game scores.", + "Critic and actor panels: big spikes are normal early.", + "Steady growth forever is bad.", + "Alpha near 1 = bot still experimenting.", + "If alpha falls too fast, the bot freezes its habits.", +] def parse_eval_series(path): @@ -186,6 +201,19 @@ def map_fn(p0, p1, vmin, vmax, log=False): return f +def guide_block(w, h, lines): + """Plain-English "How to read" band at the bottom of the canvas.""" + y = h - GUIDE_H + s = f'\n' + s += (f'' + f"How to read\n" + f'Regenerate anytime: python3 tools/plot_progress.py\n') + for i, ln in enumerate(lines): + s += f'{esc(ln)}\n' + return s + + # ---------- graph 1: eval win rates ---------- def graph_eval(series_v2, series_v1, out): @@ -195,7 +223,7 @@ def graph_eval(series_v2, series_v1, out): # panel geometry (local: graph 2 keeps the module-level constants) CW, ML, MR = 1200, 64, 24 TOP, PH, GAP, MB = 78, 310, 84, 56 - H = TOP + PH + GAP + PH + MB # 838 >= 800 + H = TOP + PH + GAP + PH + MB + GUIDE_H x0, x1 = ML, CW - MR s = svg_open(CW, H, "SAC-LSTM campaign: eval win rate vs opponents (Run 3 live, Run 1 archived)") @@ -256,6 +284,7 @@ def graph_eval(series_v2, series_v1, out): items_b.append((COLORS["Corners"], f"Run 3 trend (rolling-{TREND_WINDOW})")) s += legend(items_b, x0 + 12, y1b + 14) + s += guide_block(CW, H, WINRATE_GUIDE) s += "\n" return 1 if write_svg(out, s) else 0 @@ -287,7 +316,7 @@ def graph_losses(rows, out): return 0 PH, GAP = 210, 55 - H = M_T + 3 * PH + 2 * GAP + M_B + H = M_T + 3 * PH + 2 * GAP + M_B + GUIDE_H x0, x1 = M_L, W - M_R n = len(rows) s = svg_open(W, H, "SAC-LSTM training losses (x = metric line number)") @@ -335,8 +364,10 @@ def graph_losses(rows, out): panel(M_T + PH + GAP, actor, "actor_loss |abs| (log scale)", True, "|actor_loss|") panel(M_T + 2 * (PH + GAP), alpha, "alpha (temperature)", False, "alpha", fixed_range=(0, max(1.0, max(alpha)))) - s += (f'metric line number ({n} rows)\n\n') + s += (f'metric line number ({n} rows)\n') + s += guide_block(W, H, LOSSES_GUIDE) + s += "\n" return 1 if write_svg(out, s) else 0 @@ -368,8 +399,10 @@ def selftest(): ok = graph_losses(rows, td / "g2.svg") and graph_eval(ser, {"Corners": [0, 10]}, td / "g1.svg") assert ok and (td / "g1.svg").stat().st_size > 500 g1 = (td / "g1.svg").read_text() + g2 = (td / "g2.svg").read_text() ET.fromstring(g1) # whole doc must parse -> closing tag present - ET.fromstring((td / "g2.svg").read_text()) + ET.fromstring(g2) + assert "How to read" in g1 and "How to read" in g2, "reading guide missing" assert 'width="1200"' in g1, "canvas must be >=1200 wide" assert g1.count("= 2) + 2 # trends + v1 buckets + v3 trend