Compare commits
18 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| f5b1ade48c | |||
| 12a0eca44a | |||
| 08d1c44ca8 | |||
| f47925eeb0 | |||
| 55d69a9352 | |||
| 4a6e3347c6 | |||
| 8872be3ff6 | |||
| 8eb7c53dd0 | |||
| 51e7f33ecb | |||
| d2d205e6f6 | |||
| 3851f28cc5 | |||
| 6241c41e52 | |||
| 156b4ae8db | |||
| eee48dec38 | |||
| eb8faa7c99 | |||
| cb1bbc35dc | |||
| b7b10af811 | |||
| e455566c8e |
+59
@@ -0,0 +1,59 @@
|
||||
# Evo_Bot
|
||||
|
||||
Robocode Tank Royale bot with a modular gun system where evolved neural networks learn to predict enemy dodge behavior.
|
||||
|
||||
## Language
|
||||
|
||||
**Evo_Bot**:
|
||||
The bot itself — handles movement and firing discipline. Guns are pluggable modules.
|
||||
_Avoid_: robot, tank
|
||||
|
||||
**Guess Factor (GF)**:
|
||||
A value from -1 to +1 representing where on the maximum escape angle arc the enemy is. 0 = directly ahead, -1 = full left dodge, +1 = full right dodge. The gun's prediction target.
|
||||
_Avoid_: aim offset, dodge index
|
||||
|
||||
**Max Escape Angle (MEA)**:
|
||||
The widest angle the enemy can reach before a bullet arrives, computed from distance and bullet speed. Guess factor is multiplied by MEA to get the aim offset.
|
||||
|
||||
**Lateral Velocity**:
|
||||
Enemy speed projected perpendicular to the line between you and them. The primary signal for guess factor prediction.
|
||||
_Avoid_: tangential speed, sideways velocity
|
||||
|
||||
**Sliding Window**:
|
||||
The last N ticks (default 30) of enemy state fed as input to the network. Each tick contains lateral velocity, heading delta, and wall distance ahead.
|
||||
_Avoid_: observation buffer, input history
|
||||
|
||||
**Replay Tape**:
|
||||
Rolling buffer of recorded enemy states (~2000 ticks). The evolution thread evaluates gun fitness against this tape.
|
||||
_Avoid_: experience buffer, replay buffer
|
||||
|
||||
**Virtual Gun**:
|
||||
A gun that runs in parallel without actually firing. It tracks where it would have aimed and whether a simulated bullet would have hit. Used to compare gun variants.
|
||||
|
||||
**Virtual Bullet**:
|
||||
A simulated bullet fired by a virtual gun. Never actually sent to the game engine.
|
||||
|
||||
**TOPO_Gun**:
|
||||
Fixed-topology ANN gun evolved by GA. Network shape is predetermined (e.g., 91-8-1), only weights are evolved.
|
||||
_Avoid_: static gun, fixed gun
|
||||
|
||||
**NEAT_Gun**:
|
||||
Variable-topology ANN gun where evolution can add/remove neurons and connections (NEAT algorithm). Deferred — only built if TOPO_Gun hits a ceiling.
|
||||
|
||||
**Population**:
|
||||
The set of candidate networks (default 64-200) being evolved. Each member is a complete set of ANN weights.
|
||||
|
||||
**Champion**:
|
||||
The best-performing network in the current GA population. The champion's weights are pushed to the inference side when it beats the current best.
|
||||
_Avoid_: best, winner, elite
|
||||
|
||||
**Fitness**:
|
||||
Hit count when a network's aim predictions are evaluated as virtual bullets against sampled ticks from the replay tape.
|
||||
_Avoid_: score, reward
|
||||
|
||||
**Cold Start**:
|
||||
The first-ever battle with no saved weights. The gun does not fire until the evolution thread produces its first champion. Subsequent battles load persisted weights.
|
||||
|
||||
**Weight Persistence**:
|
||||
Saving evolved weights to disk. Load order: per-opponent file, then global fallback, then random initialization.
|
||||
_Avoid_: model saving, checkpointing
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"name": "OscillatorBot",
|
||||
"version": "0.1.0",
|
||||
"authors": ["Davide Cappellini"],
|
||||
"description": "Predictable zigzag sparring partner for gun testing",
|
||||
"homepage": "",
|
||||
"countryCodes": ["IT"],
|
||||
"gameTypes": ["classic", "1v1"],
|
||||
"platform": "Nim",
|
||||
"programmingLang": "Nim"
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
# OscillatorBot — predictable zigzag sparring partner for GA gun testing.
|
||||
# Reverses direction + turn every PERIOD ticks. Head-on targeting only.
|
||||
|
||||
import std/[math, os]
|
||||
import tankroyale_botapi
|
||||
|
||||
const botJsonPath = currentSourcePath().parentDir / "OscillatorBot.json"
|
||||
|
||||
const
|
||||
SPEED = 7.0 # forward/backward speed
|
||||
PERIOD = 25 # ticks between direction reversals
|
||||
|
||||
type OscillatorBot = ref object of Bot
|
||||
tickCount: int
|
||||
moveSign: float # +1 forward, -1 backward
|
||||
turnSign: float # +1 right, -1 left
|
||||
|
||||
method onRoundStarted*(bot: OscillatorBot, e: RoundStartedEvent) =
|
||||
setAdjustGunForBodyTurn(true)
|
||||
setAdjustRadarForBodyTurn(true)
|
||||
setAdjustRadarForGunTurn(true)
|
||||
bot.tickCount = 0
|
||||
bot.moveSign = 1.0
|
||||
bot.turnSign = 1.0
|
||||
|
||||
method onScannedBot*(bot: OscillatorBot, e: ScannedBotEvent) =
|
||||
# Head-on targeting: aim gun directly at enemy, fire medium power
|
||||
let bearing = directionTo(getX(), getY(), e.x, e.y)
|
||||
let gunDelta = normalizeRelativeAngle(bearing - getGunDirection())
|
||||
setGunTurnRate(gunDelta.clamp(-MAX_GUN_TURN_RATE, MAX_GUN_TURN_RATE))
|
||||
if abs(gunDelta) < 10.0 and getGunHeat() <= 0.0:
|
||||
discard setFire(2.0)
|
||||
|
||||
method run*(bot: OscillatorBot) =
|
||||
while isRunning():
|
||||
inc bot.tickCount
|
||||
if bot.tickCount mod PERIOD == 0:
|
||||
bot.moveSign *= -1.0
|
||||
bot.turnSign *= -1.0
|
||||
|
||||
setTargetSpeed(bot.moveSign * SPEED)
|
||||
setTurnRate(bot.turnSign * 4.0)
|
||||
setRadarTurnRate(45.0) # spin radar to keep scanning
|
||||
go()
|
||||
|
||||
when isMainModule:
|
||||
var bot = OscillatorBot(moveSign: 1.0, turnSign: 1.0)
|
||||
start(bot, botJsonPath)
|
||||
Executable
+3
@@ -0,0 +1,3 @@
|
||||
#!/bin/sh
|
||||
cd "$(dirname "$0")"
|
||||
exec ./OscillatorBot 2>> /tmp/oscillatorbot_stderr.log
|
||||
@@ -0,0 +1 @@
|
||||
--path:"../libs"
|
||||
File diff suppressed because it is too large
Load Diff
|
Before Width: | Height: | Size: 63 KiB After Width: | Height: | Size: 30 KiB |
+317
-183
@@ -3,21 +3,22 @@
|
||||
|
||||
Pure-stdlib SVG output (matplotlib not available on this box).
|
||||
Generates ONE file:
|
||||
docs/campaign_dashboard.svg - five panels, current (v2) run only:
|
||||
1. test-match win % vs opponents (campaign_v2_stdout.log eval lines)
|
||||
2. real-fight win % per opponent (training_log.jsonl, ~100-game buckets)
|
||||
3. critic_loss / |actor_loss| (training_metrics.jsonl, shared log-y)
|
||||
4. alpha temperature (training_metrics.jsonl, linear)
|
||||
5. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas;
|
||||
training_log.jsonl has NO timestamps -
|
||||
verified field names - and one metrics
|
||||
row == one 10-game chunk, counts match
|
||||
the stdout "=== Chunk N/N ===" markers)
|
||||
docs/campaign_dashboard.svg - four panels, current (v2) run only:
|
||||
1. test-match win % vs opponents (campaign_v4_stdout.log eval lines)
|
||||
2. critic_loss / |actor_loss| / alpha (training_metrics.jsonl; losses log
|
||||
left axis, alpha linear right axis)
|
||||
3. throughput, games/hour buckets (training_metrics.jsonl 'epoch' deltas;
|
||||
1 metrics row == one 10-game chunk,
|
||||
counts match stdout chunk markers)
|
||||
4. max score per eval cycle (campaign_v4_stdout.log eval blocks;
|
||||
eval_log.jsonl only ever holds the
|
||||
LATEST cycle, so history comes from
|
||||
the stdout log)
|
||||
plus an embedded JS snippet that reloads the page every 60 s when the SVG is
|
||||
opened as a top-level document in Chrome.
|
||||
|
||||
Usage:
|
||||
python3 tools/plot_progress.py [campaign_log] [metrics_jsonl] [games_jsonl] [outdir]
|
||||
python3 tools/plot_progress.py [campaign_log] [metrics_jsonl] [outdir]
|
||||
All args optional; defaults relative to the SAC_LSTM_Bot/ root (parent of tools/).
|
||||
python3 tools/plot_progress.py --selftest # tiny built-in sanity check
|
||||
"""
|
||||
@@ -27,48 +28,101 @@ import os
|
||||
import re
|
||||
import sys
|
||||
import tempfile
|
||||
import traceback
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
import xml.etree.ElementTree as ET
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
LOG_FILE = ROOT / "tools" / "plot_progress.log"
|
||||
|
||||
|
||||
def log(msg):
|
||||
"""Timestamped line to stderr AND tools/plot_progress.log (survives reboots;
|
||||
the watcher has no terminal to read errors from)."""
|
||||
line = f"{datetime.now():%F %T} {msg}"
|
||||
print(line)
|
||||
try:
|
||||
with open(LOG_FILE, "a") as fh:
|
||||
fh.write(line + "\n")
|
||||
except OSError:
|
||||
pass
|
||||
EVAL_RE = re.compile(r">>> \[eval\] win rate: (\d+)/(\d+) \(([\d.]+)%\) vs (\S+)")
|
||||
EVAL_BLOCK_RE = re.compile(r">>> \[eval\] \d+ deterministic rounds vs (\S+)")
|
||||
ROUND_RE = re.compile(r"Round \d+/\d+\D+ticks:\d+ score:(\d+) win:")
|
||||
TREND_WINDOW = 10 # rolling mean shown as the thick trend line (panel 1)
|
||||
GAME_BUCKET = 100 # games per bucket, real-fight panel
|
||||
RATE_BUCKET = 20 # metric intervals per throughput bucket (~200 games)
|
||||
GAMES_PER_ROW = 10 # one training_metrics.jsonl row per 10-round chunk
|
||||
COLORS = {"Corners": "#d62728", "Crazy": "#1f77b4", "Target": "#2ca02c",
|
||||
"RamFire": "#ff7f0e", "SacTwin": "#9467bd"}
|
||||
CRITIC_C, ACTOR_C = "#1f77b4", "#ff7f0e"
|
||||
ALPHA_C = "#9467bd" # alpha line on the losses panel
|
||||
RELOAD_JS = ('<script type="text/javascript"><![CDATA[ '
|
||||
'setTimeout(function(){ location.reload(); }, 60000); ]]></script>')
|
||||
W, H = 1400, 2000
|
||||
TITLE_H, GUIDE_H = 80, 180
|
||||
W, H = 1400, 1850
|
||||
HEALTH_H = 205 # bottom "Reading the signs" cheat-sheet band
|
||||
# rows: (header_y, panel_top_y, panel_bottom_y, x_left, x_right)
|
||||
# panel_top sits low enough to leave room under header+sub for the
|
||||
# per-panel how-to-read guide (3 italic lines, see panel_guide)
|
||||
C1_L, C1_R = 70, 697
|
||||
C2_L, C2_R = 747, 1375
|
||||
ROWS = {
|
||||
1: (100, 120, 720, C1_L, C1_R),
|
||||
2: (100, 120, 720, C2_L, C2_R),
|
||||
3: (940, 958, 1428, C1_L, C1_R),
|
||||
4: (940, 958, 1428, C2_L, C2_R),
|
||||
5: (1600, 1618, 1780, C1_L, C2_R),
|
||||
1: (100, 184, 784, C1_L, C1_R),
|
||||
2: (100, 184, 784, C2_L, C2_R),
|
||||
3: (940, 1024, 1494, C1_L, C1_R),
|
||||
4: (940, 1024, 1494, C2_L, C2_R),
|
||||
}
|
||||
PANEL_TITLES = [
|
||||
"Test matches - win % vs opponents",
|
||||
"Real fights - win % per opponent",
|
||||
"Training losses (log scale)",
|
||||
"Alpha temperature",
|
||||
"Training losses (log) & alpha (linear)",
|
||||
"Throughput - games per hour",
|
||||
"Max score per eval cycle",
|
||||
]
|
||||
GUIDE_LINES = [
|
||||
"Test matches: dots are single fights, thick line shows trend.",
|
||||
"Real battles only. Rising lines mean the bot improves.",
|
||||
"Loss spikes are normal early; endless growth is bad.",
|
||||
"Alpha high means experimenting; falling too fast freezes habits.",
|
||||
"Throughput flat is healthy; dips mean something slowed.",
|
||||
"This file reloads itself in Chrome every sixty seconds.",
|
||||
"Regenerate anytime with tools/watch_dashboard.sh or the python command.",
|
||||
# per-panel how-to-read notes: (what it shows, axes, which direction is better)
|
||||
GUIDES = {
|
||||
1: ("How often the bot wins against each opponent in test battles.",
|
||||
"X = training progress (match number); Y = win rate, 0-100%.",
|
||||
"higher is better."),
|
||||
2: ("How well the brain is learning: losses should fall; alpha sets explore/exploit.",
|
||||
"X = training progress; left Y (log) = losses; right Y (linear) = alpha.",
|
||||
"lower losses = better; alpha falls over time as the bot gets confident."),
|
||||
3: ("How many games per hour the bot trains (its learning speed).",
|
||||
"X = training progress (10-game chunks); Y = games per hour.",
|
||||
"higher = faster learning; a steady line beats a spiky one."),
|
||||
4: ("Best single-round score the bot managed in each test cycle.",
|
||||
"X = eval cycle number; Y = best score achieved.",
|
||||
"higher is better; a rising trend means the bot is improving."),
|
||||
}
|
||||
# bottom health-check cheat-sheet: one column per subsection; each bullet is
|
||||
# a tuple of pre-wrapped text lines (first line gets the bullet marker)
|
||||
SIGNS_COLUMNS = [
|
||||
("Healthy patterns ✅", [
|
||||
("Both critic and actor losses trending down over time",),
|
||||
("Alpha decaying slowly from ~1.0,", "then plateauing — this is normal"),
|
||||
("Win rates appearing and increasing in the eval panel",),
|
||||
("Max scores rising in the max-score panel",),
|
||||
("Alpha plateauing is NOT a problem — it means",
|
||||
"exploration level is stable"),
|
||||
]),
|
||||
("Warning signs ⚠️", [
|
||||
("Losses exploding (suddenly jumping to", "millions or billions)"),
|
||||
("Win rates staying at 0% for a long time",
|
||||
"after the first ~20 eval cycles"),
|
||||
("Alpha reaching 0 — bot stops exploring entirely", "(gets stuck)"),
|
||||
("Max scores flatlining (no improvement",
|
||||
"over many eval cycles)"),
|
||||
("Any single loss value above 1e6",),
|
||||
]),
|
||||
("What each metric means (brief)", [
|
||||
("Critic loss: how wrong the bot's value",
|
||||
"estimates are — should go down"),
|
||||
("Actor loss: how well the bot's action policy",
|
||||
"is doing — should go down overall", "(some bumps are normal)"),
|
||||
("Alpha: exploration-exploitation tradeoff — starts",
|
||||
"high, settles at a positive value (NOT zero)"),
|
||||
("Max score: best score achieved per eval cycle —",
|
||||
"rising trend = learning"),
|
||||
]),
|
||||
]
|
||||
|
||||
|
||||
@@ -92,6 +146,38 @@ def parse_eval_series(path):
|
||||
return series
|
||||
|
||||
|
||||
def parse_max_scores(path):
|
||||
"""Return {opponent: [best single-round score per eval cycle, in file order]}.
|
||||
|
||||
eval_log.jsonl is atomically overwritten every cycle (sac_train.sh mv), so
|
||||
per-cycle history only exists in the stdout log: each eval prints a
|
||||
'>>> [eval] N deterministic rounds vs X' header, then Round/score lines,
|
||||
closed by the '[eval] win rate' (or crashed / no results) line. Training
|
||||
rounds share the Round-line format, so they are ignored unless inside a block.
|
||||
"""
|
||||
out, cur, best = {}, None, None
|
||||
if not path.is_file():
|
||||
print(f"[skip] campaign log not found: {path}")
|
||||
return out
|
||||
for line in path.read_text(errors="replace").splitlines():
|
||||
m = EVAL_BLOCK_RE.search(line)
|
||||
if m:
|
||||
cur, best = m.group(1), None
|
||||
continue
|
||||
if cur is None:
|
||||
continue
|
||||
if "[eval]" in line: # win-rate / crashed / no-results closes the block
|
||||
if best is not None:
|
||||
out.setdefault(cur, []).append(best)
|
||||
cur, best = None, None
|
||||
continue
|
||||
m = ROUND_RE.search(line)
|
||||
if m:
|
||||
v = int(m.group(1))
|
||||
best = v if best is None else max(best, v)
|
||||
return out
|
||||
|
||||
|
||||
def parse_metrics(path):
|
||||
"""Return list of metric dicts, skipping malformed lines."""
|
||||
rows = []
|
||||
@@ -109,25 +195,6 @@ def parse_metrics(path):
|
||||
return rows
|
||||
|
||||
|
||||
def parse_games(path):
|
||||
"""Return [(opponent, won_bool)] for type=='game' rows, skipping junk."""
|
||||
out = []
|
||||
if not path.is_file():
|
||||
print(f"[skip] games log not found: {path}")
|
||||
return out
|
||||
for line in path.read_text(errors="replace").splitlines():
|
||||
try:
|
||||
r = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if r.get("type") != "game":
|
||||
continue
|
||||
opp, win = r.get("opponent"), r.get("win")
|
||||
if isinstance(opp, str) and isinstance(win, bool):
|
||||
out.append((opp, win))
|
||||
return out
|
||||
|
||||
|
||||
def metric_col(rows, key, positive=False):
|
||||
"""[(index, value)] for float-parseable rows; abs() applied; optional >0 filter."""
|
||||
out = []
|
||||
@@ -182,8 +249,8 @@ def write_svg(path, text):
|
||||
except ET.ParseError as e:
|
||||
print(f"[error] {path.name}: generated SVG invalid, keeping old file ({e})")
|
||||
return False
|
||||
tmp = path.with_name(path.name + ".tmp")
|
||||
tmp.write_text(text)
|
||||
tmp = path.with_name(f"{path.name}.{os.getpid()}.tmp") # unique: a manual
|
||||
tmp.write_text(text) # run + watcher can race
|
||||
os.replace(tmp, path)
|
||||
return True
|
||||
|
||||
@@ -202,9 +269,9 @@ def dots(pts, color, r=2, opacity=0.25):
|
||||
f'fill="{color}" opacity="{opacity}"/>\n' for x, y in pts)
|
||||
|
||||
|
||||
def hgrid(x0, x1, ys):
|
||||
def hgrid(x0, x1, ys, color="#dddddd"):
|
||||
return "".join(f'<line x1="{x0}" y1="{y:.1f}" x2="{x1}" y2="{y:.1f}" '
|
||||
f'stroke="#dddddd"/>\n' for y in ys)
|
||||
f'stroke="{color}"/>\n' for y in ys)
|
||||
|
||||
|
||||
def axis(x0, y0, x1, y1, xt, yt, xlabel, ylabel, ylog=False):
|
||||
@@ -249,21 +316,29 @@ def legend(items, x, y):
|
||||
|
||||
def map_fn(p0, p1, vmin, vmax, log=False):
|
||||
def f(v):
|
||||
t = (math.log10(v) - vmin) / (vmax - vmin) if log else (v - vmin) / (vmax - vmin)
|
||||
span = (vmax - vmin) or 1 # single-point series -> degenerate range
|
||||
t = (math.log10(v) - vmin) / span if log else (v - vmin) / span
|
||||
return p0 + max(0.0, min(1.0, t)) * (p1 - p0)
|
||||
return f
|
||||
|
||||
|
||||
def guide_block(w, h, lines):
|
||||
"""Plain-English "How to read" band at the bottom of the canvas."""
|
||||
y = H - GUIDE_H
|
||||
s = f'<rect x="0" y="{y}" width="{w}" height="{GUIDE_H}" fill="#f2f2f2"/>\n'
|
||||
s += (f'<text x="16" y="{y + 21}" font-size="15" font-weight="bold">'
|
||||
f"How to read</text>\n"
|
||||
f'<text x="{w - 16}" y="{y + 20}" text-anchor="end" font-size="11" '
|
||||
f'fill="#666">Regenerate anytime: python3 tools/plot_progress.py</text>\n')
|
||||
for i, ln in enumerate(lines):
|
||||
s += f'<text x="16" y="{y + 43 + i * 17}" font-size="13">{esc(ln)}</text>\n'
|
||||
def signs_block():
|
||||
"""Health-check cheat-sheet band along the bottom of the canvas."""
|
||||
y0 = H - HEALTH_H
|
||||
s = f'<rect x="0" y="{y0}" width="{W}" height="{HEALTH_H}" fill="#f2f2f2"/>\n'
|
||||
s += (f'<text x="16" y="{y0 + 21}" font-size="15" font-weight="bold">'
|
||||
"Reading the signs</text>\n"
|
||||
f'<text x="{W - 16}" y="{y0 + 20}" text-anchor="end" font-size="11" '
|
||||
'fill="#666">Regenerate anytime: python3 tools/plot_progress.py</text>\n')
|
||||
for (head, bullets), x in zip(SIGNS_COLUMNS, (16, 500, 985)):
|
||||
s += (f'<text x="{x}" y="{y0 + 43}" font-size="13" font-weight="bold">'
|
||||
f"{esc(head)}</text>\n")
|
||||
y = y0 + 61
|
||||
for lines in bullets:
|
||||
for j, ln in enumerate(lines):
|
||||
s += (f'<text x="{x}" y="{y}" font-size="12">'
|
||||
f"{esc(('• ' if j == 0 else ' ') + ln)}</text>\n")
|
||||
y += 14
|
||||
return s
|
||||
|
||||
|
||||
@@ -276,6 +351,16 @@ def header(x, y, title, sub=None):
|
||||
return s
|
||||
|
||||
|
||||
def panel_guide(x0, y, guide):
|
||||
"""Italic how-to-read note between a panel's header and its plot area."""
|
||||
what, axes, better = guide
|
||||
s = ""
|
||||
for i, txt in enumerate((f"What: {what}", f"Axes: {axes}", f"Better: {better}")):
|
||||
s += (f'<text x="{x0}" y="{y + i * 16}" font-size="12" font-style="italic" '
|
||||
f'fill="#555">{esc(txt)}</text>\n')
|
||||
return s
|
||||
|
||||
|
||||
# ---------- panels ----------
|
||||
|
||||
def panel_test_matches(series, geo):
|
||||
@@ -284,7 +369,7 @@ def panel_test_matches(series, geo):
|
||||
if not series:
|
||||
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
|
||||
"no eval lines found</text>\n"
|
||||
return
|
||||
return s
|
||||
xmax = max(max(len(v) for v in series.values()), 2)
|
||||
xm, ym = map_fn(x0, x1, 1, xmax), map_fn(pb, pt, 0, 100)
|
||||
s += hgrid(x0, x1, [ym(v) for v in range(0, 101, 20)])
|
||||
@@ -305,50 +390,39 @@ def panel_test_matches(series, geo):
|
||||
s += legend(items, x0 + 12, pb + 52)
|
||||
return s
|
||||
|
||||
def panel_real_fights(games, geo):
|
||||
def panel_max_score(maxes, geo):
|
||||
_, pt, pb, x0, x1 = geo
|
||||
s = ""
|
||||
if not games:
|
||||
if not maxes:
|
||||
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
|
||||
"no game rows found</text>\n"
|
||||
return
|
||||
edges = list(range(0, len(games) + 1, GAME_BUCKET))
|
||||
if edges[-1] != len(games):
|
||||
edges.append(len(games))
|
||||
buckets = list(zip(edges[:-1], edges[1:]))
|
||||
xm, ym = map_fn(x0, x1, 1, max(len(games), 2)), map_fn(pb, pt, 0, 100)
|
||||
s += hgrid(x0, x1, [ym(v) for v in range(0, 101, 20)])
|
||||
step = max(GAME_BUCKET, GAME_BUCKET * (len(games) // GAME_BUCKET // 8 + 1))
|
||||
xt = [(str(v), xm(v)) for v in range(step, len(games) + 1, step)]
|
||||
s += axis(x0, pb, x1, pt, xt, ticks_linear(0, 100, pb, pt, n=6),
|
||||
f"game number ({GAME_BUCKET}-game buckets)", "win %")
|
||||
opponents = []
|
||||
for opp, _ in games:
|
||||
if opp not in opponents:
|
||||
opponents.append(opp)
|
||||
items = []
|
||||
for name in opponents:
|
||||
c = COLORS.get(name, "#7f7f7f")
|
||||
by_b = {}
|
||||
for bi, (lo, hi) in enumerate(buckets):
|
||||
sub = [w for o, w in games[lo:hi] if o == name]
|
||||
if sub:
|
||||
by_b[bi] = 100.0 * sum(sub) / len(sub)
|
||||
pts = []
|
||||
segs, prev = [], None
|
||||
for bi in sorted(by_b):
|
||||
if prev is not None and bi != prev + 1:
|
||||
segs.append(pts)
|
||||
pts = []
|
||||
center = (buckets[bi][0] + buckets[bi][1]) / 2
|
||||
pts.append((xm(center), ym(by_b[bi])))
|
||||
prev = bi
|
||||
if len(pts) >= 2:
|
||||
segs.append(pts)
|
||||
for seg in segs:
|
||||
s += polyline(seg, c, 3.5)
|
||||
n_played = sum(1 for o, _ in games if o == name)
|
||||
items.append((c, f"{name} ({n_played} games)"))
|
||||
"no eval lines found</text>\n"
|
||||
return s
|
||||
ncyc = max(len(v) for v in maxes.values())
|
||||
xmax = max(ncyc, 2)
|
||||
hi = max(1, max(v for vals in maxes.values() for v in vals)) * 1.1
|
||||
xm, ym = map_fn(x0, x1, 1, xmax), map_fn(pb, pt, 0, hi)
|
||||
s += hgrid(x0, x1, [ym(hi * k / 4) for k in range(5)])
|
||||
step = max(1, xmax // 8)
|
||||
xt = [(str(v), xm(v)) for v in range(step, xmax + 1, step)] or [("1", xm(1))]
|
||||
s += axis(x0, pb, x1, pt, xt, ticks_linear(0, hi, pb, pt, n=5),
|
||||
"eval cycle number", "best single-round score")
|
||||
items, per_opp = [], {}
|
||||
for name in ("Corners", "Crazy", "Target"):
|
||||
vals = maxes.get(name, [])
|
||||
if not vals:
|
||||
continue
|
||||
c = COLORS[name]
|
||||
pts = [(xm(i + 1), ym(v)) for i, v in enumerate(vals)]
|
||||
per_opp[name] = vals
|
||||
s += dots(pts, c)
|
||||
s += polyline(pts, c, 2.5)
|
||||
items.append((c, f"{name} - {len(vals)} evals"))
|
||||
if len(items) >= 2: # combined best across opponents, cycle-aligned
|
||||
comb = [max(vals[i] for vals in per_opp.values() if i < len(vals))
|
||||
for i in range(ncyc)]
|
||||
s += polyline([(xm(i + 1), ym(v)) for i, v in enumerate(comb)],
|
||||
"#555555", 2.5, dash="6 4")
|
||||
items.append(("#555555", "combined max"))
|
||||
s += legend(items, x0 + 12, pb + 52)
|
||||
return s
|
||||
|
||||
@@ -358,44 +432,50 @@ def panel_losses(rows, geo):
|
||||
critic = metric_col(rows, "critic_loss", positive=True)
|
||||
actor = metric_col(rows, "actor_loss") # abs() applied; sign dropped
|
||||
actor = [(i, v) for i, v in actor if v > 0]
|
||||
if not (critic or actor):
|
||||
alpha = metric_col(rows, "alpha")
|
||||
if not (critic or actor or alpha):
|
||||
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
|
||||
"no valid loss points</text>\n"
|
||||
return
|
||||
allv = [v for _, v in critic + actor]
|
||||
lo, hi = math.floor(math.log10(min(allv))), math.ceil(math.log10(max(allv)))
|
||||
if lo == hi:
|
||||
hi = lo + 1
|
||||
return s
|
||||
x1 -= 46 # room on the right for the twin alpha axis labels
|
||||
n = len(rows)
|
||||
xm = lambda i: x0 + (x1 - x0) * i / max(n - 1, 1)
|
||||
# log-domain source (losses in practice); drop non-positive values so a
|
||||
# run of alpha==0 rows can't raise math domain error and kill the build
|
||||
base = [(i, v) for i, v in critic + actor if v > 0] or \
|
||||
[(i, v) for i, v in alpha if v > 0]
|
||||
if not base:
|
||||
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
|
||||
"no valid loss points</text>\n"
|
||||
return s
|
||||
lo = math.floor(math.log10(min(v for _, v in base)))
|
||||
hi = math.ceil(math.log10(max(v for _, v in base)))
|
||||
if lo == hi:
|
||||
hi = lo + 1
|
||||
ym = map_fn(pb, pt, lo, hi, log=True)
|
||||
s += hgrid(x0, x1, [ym(10 ** e) for e in range(lo, hi + 1)])
|
||||
s += axis(x0, pb, x1, pt, ticks_linear(1, n, x0, x1, n=5),
|
||||
ticks_log(lo, hi, pb, pt), "metric line number", "loss (log)")
|
||||
s += polyline([(xm(i), ym(v)) for i, v in critic], CRITIC_C, 1.8)
|
||||
s += polyline([(xm(i), ym(v)) for i, v in actor], ACTOR_C, 1.8)
|
||||
s += legend([(CRITIC_C, "critic_loss"), (ACTOR_C, "|actor_loss|")],
|
||||
x0 + 12, pb + 52)
|
||||
return s
|
||||
|
||||
def panel_alpha(rows, geo):
|
||||
_, pt, pb, x0, x1 = geo
|
||||
s = ""
|
||||
alpha = metric_col(rows, "alpha")
|
||||
if not alpha:
|
||||
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
|
||||
"no alpha points</text>\n"
|
||||
return
|
||||
n = len(rows)
|
||||
hi = max(1.0, max(v for _, v in alpha))
|
||||
xm = lambda i: x0 + (x1 - x0) * i / max(n - 1, 1)
|
||||
ym = map_fn(pb, pt, 0, hi)
|
||||
s += hgrid(x0, x1, [ym(v) for v in
|
||||
[hi * k / 4 for k in range(5)]])
|
||||
s += axis(x0, pb, x1, pt, ticks_linear(1, n, x0, x1, n=5),
|
||||
ticks_linear(0, hi, pb, pt, n=5, fmt="{:.3g}"),
|
||||
"metric line number", "alpha")
|
||||
s += polyline([(xm(i), ym(v)) for i, v in alpha], "#9467bd", 1.8)
|
||||
if alpha: # twin axis: alpha on its own linear scale, purple like the line
|
||||
ahi = max(1.0, max(v for _, v in alpha))
|
||||
yma = map_fn(pb, pt, 0, ahi)
|
||||
s += hgrid(x0, x1, [yma(ahi * k / 4) for k in range(5)], "#e9dcf5")
|
||||
mid = (pt + pb) // 2
|
||||
s += f'<line x1="{x1}" y1="{pb}" x2="{x1}" y2="{pt}" stroke="{ALPHA_C}"/>\n'
|
||||
for v, py in ticks_linear(0, ahi, pb, pt, n=5, fmt="{:.3g}"):
|
||||
s += (f'<line x1="{x1}" y1="{py:.1f}" x2="{x1 + 4}" y2="{py:.1f}" '
|
||||
f'stroke="{ALPHA_C}"/>\n'
|
||||
f'<text x="{x1 + 7}" y="{py + 4:.1f}" font-size="11" '
|
||||
f'fill="{ALPHA_C}">{esc(v)}</text>\n')
|
||||
s += (f'<text x="{x1 + 17}" y="{mid}" text-anchor="middle" font-size="12" '
|
||||
f'fill="{ALPHA_C}" transform="rotate(90 {x1 + 17} {mid})">alpha</text>\n')
|
||||
s += polyline([(xm(i), yma(v)) for i, v in alpha], ALPHA_C, 1.8)
|
||||
items = [(CRITIC_C, "critic_loss"), (ACTOR_C, "|actor_loss|")]
|
||||
if alpha:
|
||||
items.append((ALPHA_C, "alpha"))
|
||||
s += legend(items, x0 + 12, pb + 52)
|
||||
return s
|
||||
|
||||
def panel_throughput(rows, geo):
|
||||
@@ -415,7 +495,7 @@ def panel_throughput(rows, geo):
|
||||
if not rates:
|
||||
s += f'<text x="{x0 + 10}" y="{pt + 40}" font-size="12" fill="#a00">' \
|
||||
"no usable epoch timestamps</text>\n"
|
||||
return
|
||||
return s
|
||||
bm = bucket_means(rates, RATE_BUCKET)
|
||||
xm = map_fn(x0, x1, 1, len(rates))
|
||||
ymax = max(max(rates), max(bm)) * 1.1
|
||||
@@ -434,14 +514,15 @@ def panel_throughput(rows, geo):
|
||||
|
||||
# ---------- assembly ----------
|
||||
|
||||
def build_dashboard(campaign, metrics_f, games_f, out):
|
||||
def build_dashboard(campaign, metrics_f, out):
|
||||
series = parse_eval_series(campaign)
|
||||
print("[info] evals parsed: " +
|
||||
(", ".join(f"{k}={len(v)}" for k, v in sorted(series.items())) or "(none)"))
|
||||
maxes = parse_max_scores(campaign)
|
||||
print("[info] eval max-score cycles parsed: " +
|
||||
(", ".join(f"{k}={len(v)}" for k, v in sorted(maxes.items())) or "(none)"))
|
||||
rows = parse_metrics(metrics_f)
|
||||
print(f"[info] metric rows parsed: {len(rows)}")
|
||||
games = parse_games(games_f)
|
||||
print(f"[info] game rows parsed: {len(games)}")
|
||||
|
||||
s = (f'<svg xmlns="http://www.w3.org/2000/svg" width="{W}" height="{H}" '
|
||||
f'viewBox="0 0 {W} {H}" font-family="sans-serif">\n'
|
||||
@@ -453,28 +534,31 @@ def build_dashboard(campaign, metrics_f, games_f, out):
|
||||
f'(open this file in Chrome)</text>\n')
|
||||
|
||||
drawers = [
|
||||
(ROWS[1], PANEL_TITLES[0],
|
||||
(ROWS[1], PANEL_TITLES[0], GUIDES[1],
|
||||
"raw dots = single test matches, thick = rolling-mean-%d" % TREND_WINDOW,
|
||||
lambda: panel_test_matches(series, ROWS[1])),
|
||||
(ROWS[2], PANEL_TITLES[1],
|
||||
"training_log.jsonl only - learning in REAL battles, not tests",
|
||||
lambda: panel_real_fights(games, ROWS[2])),
|
||||
(ROWS[3], PANEL_TITLES[2],
|
||||
(ROWS[2], PANEL_TITLES[1], GUIDES[2],
|
||||
"training_metrics.jsonl - big early spikes are normal",
|
||||
lambda: panel_losses(rows, ROWS[3])),
|
||||
(ROWS[4], PANEL_TITLES[3],
|
||||
"training_metrics.jsonl - high = exploring, low = exploiting",
|
||||
lambda: panel_alpha(rows, ROWS[4])),
|
||||
(ROWS[5], PANEL_TITLES[4],
|
||||
"method: training_metrics.jsonl 'epoch' deltas (training_log.jsonl has "
|
||||
"no timestamps); 1 row = one 10-game chunk",
|
||||
lambda: panel_throughput(rows, ROWS[5])),
|
||||
lambda: panel_losses(rows, ROWS[2])),
|
||||
(ROWS[3], PANEL_TITLES[2], GUIDES[3],
|
||||
"method: training_metrics.jsonl 'epoch' deltas; 1 row = one 10-game chunk",
|
||||
lambda: panel_throughput(rows, ROWS[3])),
|
||||
(ROWS[4], PANEL_TITLES[3], GUIDES[4],
|
||||
"campaign_v4_stdout.log - max of the 10 deterministic round scores per eval",
|
||||
lambda: panel_max_score(maxes, ROWS[4])),
|
||||
]
|
||||
for geo, title, sub, drawer in drawers:
|
||||
for geo, title, guide, sub, drawer in drawers:
|
||||
s += header(geo[3], geo[0], title, sub)
|
||||
s += drawer()
|
||||
s += panel_guide(geo[3], geo[0] + 33, guide)
|
||||
try:
|
||||
body = drawer()
|
||||
except Exception as e: # one bad panel must not kill the whole page
|
||||
log(f"[warn] panel '{title}' failed, rendering placeholder: {e}")
|
||||
body = (f'<text x="{geo[3] + 10}" y="{geo[1] + 40}" font-size="12" '
|
||||
f'fill="#a00">panel error: {esc(e)}</text>')
|
||||
s += body or "" # panels bare-return None on their no-data path
|
||||
|
||||
s += guide_block(W, H, GUIDE_LINES)
|
||||
s += signs_block()
|
||||
s += RELOAD_JS + "\n"
|
||||
s += "</svg>\n"
|
||||
return write_svg(out, s)
|
||||
@@ -486,14 +570,28 @@ def selftest():
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
td = Path(td)
|
||||
(td / "log").write_text(
|
||||
">>> [eval] win rate: 3/10 (30%) vs Corners\n"
|
||||
">>> [eval] 10 deterministic rounds vs Corners\n"
|
||||
"Round 1/10 - ticks:100 score:3 win:true\n"
|
||||
"garbage line\n"
|
||||
"Round 2/10 - ticks:100 score:7 win:false\n"
|
||||
">>> [eval] win rate: 3/10 (30%) vs Corners\n"
|
||||
">>> [eval] 10 deterministic rounds vs Crazy\n"
|
||||
"Round 1/10 - ticks:100 score:70 win:false\n"
|
||||
">>> [eval] win rate: 7/10 (70%) vs Crazy\n"
|
||||
">>> [eval] win rate: broken\n"
|
||||
">>> [eval] broken\n"
|
||||
"Round 9/9 - ticks:1 score:999 win:false\n"
|
||||
">>> [eval] 10 deterministic rounds vs Corners\n"
|
||||
"Round 1/10 - ticks:100 score:50 win:true\n"
|
||||
">>> [eval] win rate: 5/10 (50%) vs Corners\n"
|
||||
">>> [eval] 10 deterministic rounds vs Crazy\n"
|
||||
"Round 1/10 - ticks:100 score:40 win:true\n"
|
||||
">>> [eval] win rate: 4/10 (40%) vs Crazy\n")
|
||||
ser = parse_eval_series(td / "log")
|
||||
assert ser == {"Corners": [30.0, 50.0], "Crazy": [70.0, 40.0]}, ser
|
||||
mxs = parse_max_scores(td / "log")
|
||||
# stray Round 999 after the unclosed '[eval] broken' line is ignored;
|
||||
# per-cycle max of the Round scores above
|
||||
assert mxs == {"Corners": [7, 50], "Crazy": [70, 40]}, mxs
|
||||
assert rolling([10] * 25, 20)[-1] == 10.0
|
||||
assert rolling([1, 2, 3], 20) == [1.0, 1.5, 2.0]
|
||||
assert len(bucket_means(list(range(1287)), RATE_BUCKET)) == RATE_BUCKET
|
||||
@@ -506,32 +604,39 @@ def selftest():
|
||||
'{"epoch": 1090.0, "critic_loss": 50, "actor_loss": 3, "alpha": 0.2}\n')
|
||||
rows = parse_metrics(td / "m.jsonl")
|
||||
assert len(rows) == 3 and rows[1]["critic_loss"] == 100
|
||||
glines = []
|
||||
for i in range(150): # 2 full GAME_BUCKETs, both opponents in both
|
||||
glines.append(json.dumps(
|
||||
{"type": "game", "round": i % 10 + 1, "ticks": 100,
|
||||
"score": i % 3, "total_score": i, "win": i % 3 == 0,
|
||||
"opponent": ("Corners", "Crazy")[i % 2]}))
|
||||
(td / "g.jsonl").write_text("\n".join(glines) + "\n")
|
||||
games = parse_games(td / "g.jsonl")
|
||||
assert len(games) == 150 and games[0] == ("Corners", True)
|
||||
assert games[-1] == ("Crazy", False) # i=149: odd -> Crazy; 149%3!=0 -> loss
|
||||
assert metric_col(rows, "actor_loss") == [(0, 2.0), (1, 4.0), (2, 3.0)]
|
||||
|
||||
dash = td / "dash.svg"
|
||||
assert build_dashboard(td / "log", td / "m.jsonl", td / "g.jsonl", dash)
|
||||
assert build_dashboard(td / "log", td / "m.jsonl", dash)
|
||||
text = dash.read_text()
|
||||
ET.fromstring(text) # whole doc must parse -> closing tag present
|
||||
assert RELOAD_JS in text, "auto-reload script missing"
|
||||
for t in PANEL_TITLES:
|
||||
assert t in text, f"panel title missing: {t}"
|
||||
assert esc(t) in text, f"panel title missing: {t}"
|
||||
assert text.count(PANEL_TITLES[0]) == 1
|
||||
assert "How to read" in text, "reading guide missing"
|
||||
assert 'width="1400"' in text and 'height="2000"' in text
|
||||
# circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 5)
|
||||
assert text.count("<circle") == 6, text.count("<circle")
|
||||
# 2 trends + 2 real-fight opp lines + 2 losses + 1 alpha + 1 throughput
|
||||
assert text.count("<polyline") == 8, text.count("<polyline")
|
||||
assert "Reading the signs" in text, "health-check section missing"
|
||||
for head, bullets in SIGNS_COLUMNS:
|
||||
assert esc(head) in text, f"health-check column missing: {head}"
|
||||
for lines in bullets:
|
||||
assert esc("• " + lines[0]) in text, f"bullet missing: {lines[0]}"
|
||||
assert text.count("• ") == sum(len(b) for _, b in SIGNS_COLUMNS)
|
||||
assert 'width="1400"' in text and 'height="1850"' in text
|
||||
# every panel carries its own What/Axes/Better how-to-read note
|
||||
assert text.count("What:") == len(PANEL_TITLES), text.count("What:")
|
||||
for g in GUIDES.values():
|
||||
assert esc(g[0]) in text, f"panel guide missing: {g[0]}"
|
||||
# circles: 4 eval dots (panel 1) + 2 throughput rate dots (panel 3)
|
||||
# + 2 cycles x 2 opponents max-score dots (panel 4)
|
||||
assert text.count("<circle") == 10, text.count("<circle")
|
||||
# 2 trends + 3 losses-panel polylines (critic, actor, alpha)
|
||||
# + 1 throughput
|
||||
# + 3 max-score panel (Corners, Crazy, combined; Target absent in fixture)
|
||||
assert text.count("<polyline") == 9, text.count("<polyline")
|
||||
# losses panel twin axis: "alpha" text = legend + right ylabel; purple
|
||||
# fills = 5 right-axis tick labels + rotated ylabel
|
||||
assert text.count(">alpha<") == 2, text.count(">alpha<")
|
||||
assert text.count('fill="#9467bd"') == 6, text.count('fill="#9467bd"')
|
||||
assert "combined max" in text, "max-score combined line missing"
|
||||
|
||||
# orientation guard: a known rising series (10% -> 90%) rendered through
|
||||
# the FULL build path must plot upward (smaller SVG y) and forward in
|
||||
@@ -541,13 +646,39 @@ def selftest():
|
||||
">>> [eval] win rate: 1/10 (10%) vs Corners\n"
|
||||
">>> [eval] win rate: 9/10 (90%) vs Corners\n")
|
||||
ori_dash = td / "ori.svg"
|
||||
assert build_dashboard(ori_log, td / "m.jsonl", td / "g.jsonl", ori_dash)
|
||||
assert build_dashboard(ori_log, td / "m.jsonl", ori_dash)
|
||||
m = re.search(r'<polyline points="([^"]+)"', ori_dash.read_text())
|
||||
pts = [tuple(map(float, p.split(","))) for p in m.group(1).split()]
|
||||
assert len(pts) == 2, pts
|
||||
(xa, ya), (xb, yb) = pts
|
||||
assert yb < ya, f"y-axis inverted: win % rose 10->90 but ink moved down ({ya} -> {yb})"
|
||||
assert xb > xa, f"x-axis reversed: newer eval plotted left ({xa} -> {xb})"
|
||||
|
||||
# resilience: missing/empty inputs render placeholders, never crash
|
||||
empty_dash = td / "empty.svg"
|
||||
assert build_dashboard(td / "nope.log", td / "nope.jsonl", empty_dash)
|
||||
etext = empty_dash.read_text()
|
||||
assert etext.count("no eval lines found") == 2, etext.count("no eval lines found")
|
||||
assert "no valid loss points" in etext
|
||||
assert "no usable epoch timestamps" in etext
|
||||
|
||||
# all-zero alpha rows (post-crash trainer state) must not raise in the
|
||||
# losses panel's log-domain math -> placeholder instead of dead build
|
||||
(td / "zero.jsonl").write_text(
|
||||
'{"epoch": 1000.0, "alpha": 0}\n{"epoch": 1060.0, "alpha": 0}\n')
|
||||
zero_dash = td / "zero.svg"
|
||||
assert build_dashboard(td / "log", td / "zero.jsonl", zero_dash)
|
||||
assert "no valid loss points" in zero_dash.read_text()
|
||||
|
||||
# single throughput interval (fresh 2-row metrics file right after a
|
||||
# restart) used to divide by zero in map_fn and kill the whole build
|
||||
(td / "one.jsonl").write_text(
|
||||
'{"epoch": 1000.0, "critic_loss": 10, "alpha": 0.5}\n'
|
||||
'{"epoch": 1060.0, "critic_loss": 5, "alpha": 0.4}\n')
|
||||
one_dash = td / "one.svg"
|
||||
assert build_dashboard(td / "log", td / "one.jsonl", one_dash)
|
||||
assert "panel error" not in one_dash.read_text()
|
||||
assert "<polyline" in one_dash.read_text()
|
||||
print("selftest OK")
|
||||
|
||||
|
||||
@@ -556,21 +687,24 @@ def main():
|
||||
selftest()
|
||||
return
|
||||
args = [a for a in sys.argv[1:] if not a.startswith("-")]
|
||||
campaign = Path(args[0]) if len(args) > 0 else ROOT / "campaign_v2_stdout.log"
|
||||
campaign = Path(args[0]) if len(args) > 0 else ROOT / "campaign_v4_stdout.log"
|
||||
metrics = Path(args[1]) if len(args) > 1 else ROOT / "training_metrics.jsonl"
|
||||
games = Path(args[2]) if len(args) > 2 else ROOT / "training_log.jsonl"
|
||||
outdir = Path(args[3]) if len(args) > 3 else ROOT / "docs"
|
||||
outdir.mkdir(parents=True, exist_ok=True)
|
||||
outdir = Path(args[2]) if len(args) > 2 else ROOT / "docs"
|
||||
|
||||
try:
|
||||
ok = build_dashboard(campaign, metrics, games, outdir / "campaign_dashboard.svg")
|
||||
except Exception as e:
|
||||
print(f"[error] dashboard build failed: {e}")
|
||||
outdir.mkdir(parents=True, exist_ok=True)
|
||||
for p in (campaign, metrics):
|
||||
if not p.is_file():
|
||||
# loudest symptom of a watcher launched from a stale checkout
|
||||
log(f"[warn] input missing: {p} - is this the live checkout?")
|
||||
ok = build_dashboard(campaign, metrics, outdir / "campaign_dashboard.svg")
|
||||
except Exception:
|
||||
log(f"[error] dashboard build failed:\n{traceback.format_exc().rstrip()}")
|
||||
ok = False
|
||||
if ok:
|
||||
print(f"[done] dashboard written to {outdir / 'campaign_dashboard.svg'}")
|
||||
else:
|
||||
print("[error] dashboard NOT updated - check paths above")
|
||||
print("[error] dashboard NOT updated - see " + str(LOG_FILE))
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
|
||||
@@ -1,10 +1,23 @@
|
||||
#!/bin/sh
|
||||
# Keep docs/campaign_dashboard.svg fresh: regenerate every 60 s.
|
||||
# Errors go to stderr and never exit the loop silently.
|
||||
dir=$(dirname "$0")
|
||||
# Failures are appended to tools/plot_progress.log (by this wrapper and by
|
||||
# plot_progress.py itself) - check that file when the dashboard looks stale.
|
||||
# The startup banner records WHICH tree this instance watches, so a copy
|
||||
# launched from a stale checkout (the post-reboot failure mode) is visible.
|
||||
set -u
|
||||
dir=$(cd "$(dirname "$0")" && pwd)
|
||||
log="$dir/plot_progress.log"
|
||||
|
||||
if command -v flock >/dev/null 2>&1; then
|
||||
exec 9>"$dir/.watch_dashboard.lock"
|
||||
flock -n 9 || { echo "[watch_dashboard] another instance already running, exiting" >&2; exit 0; }
|
||||
fi
|
||||
|
||||
echo "[watch_dashboard] $(date '+%F %T') started, watching root=$(cd "$dir/.." && pwd)" >> "$log" 2>/dev/null || true
|
||||
while :; do
|
||||
if ! python3 "$dir/plot_progress.py"; then
|
||||
echo "[watch_dashboard] $(date '+%F %T') regeneration failed (see error above)" >&2
|
||||
echo "[watch_dashboard] $(date '+%F %T') regeneration failed (see $log)" >&2
|
||||
echo "[watch_dashboard] $(date '+%F %T') regeneration failed" >> "$log" 2>/dev/null || true
|
||||
fi
|
||||
sleep 60
|
||||
done
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
# Neuroevolution gun with fixed-topology ANN evolved by GA
|
||||
|
||||
Evo_Bot needs a gun that adapts to each opponent's dodge patterns during a match, finds nonlinear movement patterns that histograms miss, and is original. We chose a fixed-topology feedforward ANN (91->8->1, 745 weights) whose weights are evolved by a mutation-only GA running on a parallel thread. This beats the alternatives (RL too slow to adapt in-match, Q-learning collapses to histogram for single-shot decisions, guess-factor histograms are unoriginal, transformer/LLM-style prediction is data-starved at ~14k ticks per match) while keeping implementation risk low by deferring topology evolution (NEAT) until the fixed network hits its ceiling.
|
||||
|
||||
## Considered Options
|
||||
|
||||
- **PPO / SAC (end-to-end RL):** Too slow -- thousands of rounds to converge, cannot adapt mid-match. Explored in other bots in this repo.
|
||||
- **Q-learning for aiming:** Collapses to a histogram. Single-shot aiming has no sequential decision structure for Q-learning to exploit.
|
||||
- **Guess-factor histogram:** Proven and fast to converge (~15 ticks), but unoriginal -- 20 years of community tuning.
|
||||
- **Transformer / LLM-style sequence prediction:** Data-starved. ~14k ticks per match vs billions needed for attention-based models.
|
||||
- **GA with crossover:** Literature uniformly shows crossover is harmful for ANN weight evolution -- it breaks co-adapted weight configurations. Every modern neuroevolution paper (Uber Deep GA, NRA, OpenAI ES) drops it.
|
||||
- **CMA-ES:** Ideal at d=750 weights (self-adapts sigma and covariance). More complex to implement; upgrade path from simple GA when needed.
|
||||
|
||||
## Decision
|
||||
|
||||
**Architecture:**
|
||||
- Evo_Bot (1v1) with modular gun interface: `feed(state)` / `aim() -> (angle, power)`
|
||||
- Gun owns its evolution thread (parallel, never blocks inference)
|
||||
|
||||
**TOPO_Gun (first gun implementation):**
|
||||
- Network: 91->8->1 (hidden size configurable), 745 weights
|
||||
- Input: 30 ticks x (lateral_vel, delta_heading, wall_distance_ahead) + current distance = 91
|
||||
- Output: guess factor (-1 to +1)
|
||||
- Bullet power: deterministic distance-based formula (not learned)
|
||||
- Evolution: population 300, clone loaded champion + small Gaussian mutations (ALL weights, sigma=0.005-0.01), NO crossover, single elite preserved unchanged
|
||||
- Fitness: virtual bullet hits on 100 randomly sampled replay tape ticks, using real distance-based power
|
||||
- Replay tape: rolling window ~2000 ticks (configurable)
|
||||
- Weight persistence: per-opponent -> global fallback -> random init (load order)
|
||||
- Cold start: first-ever run, don't fire until champion emerges; subsequent runs load weights, fire from tick 1
|
||||
- Push new champion weights to inference when it hits better than current
|
||||
|
||||
**Deferred:**
|
||||
- NEAT_Gun: deferred until TOPO_Gun hits its ceiling
|
||||
- Virtual Guns: run multiple guns in parallel, fire whichever has best virtual hit rate
|
||||
|
||||
**Boundaries:**
|
||||
- Bot controls firing discipline (when to shoot, energy management); gun always returns best aim
|
||||
|
||||
## Consequences
|
||||
|
||||
- GA+ANN finds nonlinear patterns histograms miss, but needs more data (~50+ ticks vs ~15 for guess-factor histogram) before it outperforms
|
||||
- Fixed topology before NEAT reduces implementation risk
|
||||
- Mutation-only evolution simplifies implementation (no crossover logic)
|
||||
- Parallel evolution thread reuses the pattern from SAC_LSTM_Bot (#48)
|
||||
- Per-opponent weight persistence eliminates cold start after first encounter
|
||||
- CMA-ES is the natural upgrade path if simple GA convergence is too slow (d=750 is CMA-ES sweet spot)
|
||||
@@ -0,0 +1,111 @@
|
||||
## ga_spike.nim — throwaway GA neuroevolution spike: evolve ANN to predict sin(x)
|
||||
## Validates: population init, fitness eval, truncation selection, gaussian
|
||||
## mutation on all weights, single-elite preservation, champion tracking.
|
||||
## Issue #66.
|
||||
|
||||
import std/[math, random, algorithm, strformat]
|
||||
|
||||
const
|
||||
InputDim = 10 # sliding window of 10 past sin values
|
||||
HiddenDim = 4
|
||||
OutputDim = 1
|
||||
NumWeights = (InputDim * HiddenDim + HiddenDim) + # W1 + b1
|
||||
(HiddenDim * OutputDim + OutputDim) # W2 + b2 = 49
|
||||
PopSize = 200
|
||||
Generations = 200
|
||||
TopFrac = 0.2 # keep top 20%
|
||||
Sigma = 0.01 # gaussian mutation sigma on ALL weights
|
||||
EvalPoints = 50 # fitness eval sample size
|
||||
|
||||
type Individual = object
|
||||
weights: seq[float64]
|
||||
fitness: float64
|
||||
|
||||
proc forward(w: seq[float64]; input: array[InputDim, float64]): float64 =
|
||||
## 10->4->1 tanh ANN, flat weight layout: W1[40], b1[4], W2[4], b2[1]
|
||||
var hidden: array[HiddenDim, float64]
|
||||
for h in 0..<HiddenDim:
|
||||
var s = w[InputDim * HiddenDim + h] # b1[h]
|
||||
for i in 0..<InputDim:
|
||||
s += w[h * InputDim + i] * input[i] # W1[h,i]
|
||||
hidden[h] = tanh(s)
|
||||
var s = w[InputDim * HiddenDim + HiddenDim + HiddenDim] # b2[0]
|
||||
for h in 0..<HiddenDim:
|
||||
s += w[InputDim * HiddenDim + HiddenDim + h] * hidden[h] # W2[h]
|
||||
result = tanh(s)
|
||||
|
||||
proc evaluate(ind: var Individual; rng: var Rand) =
|
||||
## Fitness = -MSE on EvalPoints random samples of sin prediction.
|
||||
var mse = 0.0
|
||||
for _ in 0..<EvalPoints:
|
||||
let x = rng.rand(0.0 .. 20.0 * PI)
|
||||
var input: array[InputDim, float64]
|
||||
for i in 0..<InputDim:
|
||||
input[i] = sin(x - float64(InputDim - 1 - i))
|
||||
let target = sin(x)
|
||||
let pred = forward(ind.weights, input)
|
||||
mse += (pred - target) * (pred - target)
|
||||
ind.fitness = -mse / EvalPoints.float64
|
||||
|
||||
proc mutate(ind: var Individual; rng: var Rand) =
|
||||
## Additive gaussian on ALL weights, per GA-parameters research.
|
||||
for i in 0..<ind.weights.len:
|
||||
ind.weights[i] += rng.gauss(0.0, Sigma)
|
||||
|
||||
when isMainModule:
|
||||
var rng = initRand(42)
|
||||
|
||||
# Init population: random weights in [-0.5, 0.5]
|
||||
var pop = newSeq[Individual](PopSize)
|
||||
for i in 0..<PopSize:
|
||||
pop[i].weights = newSeq[float64](NumWeights)
|
||||
for j in 0..<NumWeights:
|
||||
pop[i].weights[j] = rng.rand(-0.5 .. 0.5)
|
||||
|
||||
echo &"GA spike: {PopSize} individuals, {NumWeights} weights, {Generations} gens, sigma={Sigma}"
|
||||
|
||||
for gen in 0..<Generations:
|
||||
# Evaluate
|
||||
for i in 0..<PopSize:
|
||||
pop[i].evaluate(rng)
|
||||
|
||||
# Sort descending by fitness (higher = better)
|
||||
pop.sort(proc(a, b: Individual): int = cmp(b.fitness, a.fitness))
|
||||
|
||||
if gen mod 20 == 0 or gen == Generations - 1:
|
||||
echo &" gen {gen:3d} best_fitness={pop[0].fitness:.6f} (MSE={-pop[0].fitness:.6f})"
|
||||
|
||||
# Selection: keep top 20%
|
||||
let nParents = max(1, int(PopSize.float64 * TopFrac))
|
||||
|
||||
# Next generation: elite(1) + mutated children from parents
|
||||
var next = newSeq[Individual](PopSize)
|
||||
next[0] = pop[0] # single elite, unchanged
|
||||
for i in 1..<PopSize:
|
||||
let parent = rng.rand(0..<nParents)
|
||||
next[i].weights = pop[parent].weights # clone parent
|
||||
next[i].mutate(rng)
|
||||
pop = next
|
||||
|
||||
# Final evaluation of champion
|
||||
pop[0].evaluate(rng)
|
||||
echo &"\nChampion fitness: {pop[0].fitness:.6f} (MSE={-pop[0].fitness:.6f})"
|
||||
|
||||
# Print predictions vs actual
|
||||
echo "\nPredictions (sample):"
|
||||
var totalErr = 0.0
|
||||
let nSamples = 20
|
||||
for i in 0..<nSamples:
|
||||
let x = float64(i) * 1.0
|
||||
var input: array[InputDim, float64]
|
||||
for j in 0..<InputDim:
|
||||
input[j] = sin(x - float64(InputDim - 1 - j))
|
||||
let target = sin(x)
|
||||
let pred = forward(pop[0].weights, input)
|
||||
totalErr += abs(pred - target)
|
||||
echo &" x={x:5.1f} sin(x)={target:+.4f} pred={pred:+.4f} err={abs(pred-target):.4f}"
|
||||
|
||||
let avgErr = totalErr / nSamples.float64
|
||||
echo &"\nAverage absolute error: {avgErr:.4f}"
|
||||
assert avgErr < 0.1, &"Champion avg error {avgErr:.4f} >= 0.1 — evolution did not converge"
|
||||
echo "PASS: assert avgErr < 0.1"
|
||||
Reference in New Issue
Block a user