8efa627c05
The repo's first multi-opponent gun measurement. Adds tools/ab/gauntlet_run.sh (per-opponent A/B over the legacy roster, subject = frozen ModularBot), tools/ab/gauntlet_analyze.py (paired per-opponent deltas, cross-opponent sign test, style split, MDE) and the arm/opponent fixtures. Result: BitBrain does NOT generalize beyond DrussGT. 32 opponents x 2 arms x 3 runs x 5 rounds = 192 battles / 960 rounds, 0 failed, 0 retries: damage/run 214.5 (pattern) vs 210.9 (bb), sign-flip p=0.53; round wins 237/480 vs 239/480, p=0.91. Sign test: bb better on 13/32 opponents (damage). The DrussGT-only penalty does not carry. The owner's 'killer vs regular movers' sub-claim is not supported: regular bucket +1.3 dmg/run vs dodgers -0.2 (MW p=0.85), and the measured movement predictability does not correlate with the delta.
645 lines
26 KiB
Python
Executable File
645 lines
26 KiB
Python
Executable File
#!/usr/bin/env python3
|
||
"""gauntlet_analyze.py — per-adversary BitBrain-vs-Pattern gauntlet analysis.
|
||
|
||
python3 tools/ab/gauntlet_analyze.py <session_dir> [--report FILE]
|
||
|
||
Reads a session dir produced by tools/ab/gauntlet_run.sh and prints:
|
||
|
||
* the PER-OPPONENT table (opponent, inferred movement style, damage/run in
|
||
each arm, paired damage delta, round wins in each arm, paired wins delta,
|
||
rounds sampled, opponent fire count for liveness);
|
||
* the CROSS-OPPONENT SIGN TEST on the paired deltas — the headline: on how
|
||
many opponents does each arm win (bb - pattern > 0);
|
||
* the pooled paired estimate WITH the between-opponent spread (so a single
|
||
outlier bot cannot carry it), a sign-flip permutation test and the MDE;
|
||
* the STYLE SPLIT: is the bb - pattern delta larger on the regular movers
|
||
(wall-followers / campers / rammers / periodic) than on the dodgers?
|
||
|
||
Standard library only, deterministic.
|
||
|
||
Metrics (the standing rule): damage dealt per run and ROUND WINS. Never hit
|
||
rate alone. Subject of every capture is ModularBot itself, so `subject event
|
||
counts` are ModularBot's own counts and `firstPlaces` is ModularBot's own
|
||
name-based round-win count.
|
||
|
||
Liveness / false-negative guard: a run is dropped unless the capture shows real
|
||
rows, ModularBot appears in the runner's RESULTS block (firstPlaces) and the
|
||
event sidecar attributes owners unambiguously. An opponent is dropped unless it
|
||
fired at least once in the retained runs — an opponent that never fires is not
|
||
an opponent (per tools/robocode_shim/LEGACY_BOTS.md the synthetic smoke test has
|
||
false negatives, so only a real battle counts).
|
||
"""
|
||
import glob
|
||
import itertools
|
||
import json
|
||
import math
|
||
import os
|
||
import random
|
||
import re
|
||
import statistics
|
||
import sys
|
||
|
||
BOT_NAME = "ModularBot"
|
||
EXACT_PAIR_CAP = 20 # 2**20 = 1.05M sign-flip enumerations, still fast
|
||
MC_DRAWS = 1_000_000
|
||
MC_SEED = 0x5EED5EED
|
||
Z_ALPHA_POWER = 1.959963984540054 + 0.8416212335729143 # 2-sample MDE const
|
||
|
||
|
||
# ── loading / parsing ────────────────────────────────────────────────────────
|
||
|
||
def parse_envspec(envspec):
|
||
"""`TR_A=1 TR_B=2` -> {'TR_A': '1', 'TR_B': '2'}."""
|
||
out = {}
|
||
for tok in (envspec or "").split():
|
||
if "=" in tok:
|
||
k, v = tok.split("=", 1)
|
||
out[k] = v
|
||
return out
|
||
|
||
|
||
def read_lines(path):
|
||
try:
|
||
with open(path, errors="replace") as fh:
|
||
return fh.readlines()
|
||
except OSError:
|
||
return []
|
||
|
||
|
||
def parse_events(path):
|
||
evs = []
|
||
for line in read_lines(path):
|
||
line = line.strip()
|
||
if not line:
|
||
continue
|
||
try:
|
||
evs.append(json.loads(line))
|
||
except json.JSONDecodeError:
|
||
continue
|
||
return evs
|
||
|
||
|
||
def parse_first_places(log_text):
|
||
"""ModularBot's run-level `firstPlaces` (name-based round wins)."""
|
||
m = re.search(
|
||
r"^\s*#\d+\s+" + re.escape(BOT_NAME) +
|
||
r"\s+totalScore=-?\d+\s+firstPlaces=(\d+)", log_text, re.MULTILINE)
|
||
return int(m.group(1)) if m else None
|
||
|
||
|
||
def parse_counters(log_text):
|
||
m = re.search(
|
||
r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)"
|
||
r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)", log_text)
|
||
if not m:
|
||
return None
|
||
return {"fired": int(m.group(2)), "hits": int(m.group(3)),
|
||
"hits_taken": int(m.group(6))}
|
||
|
||
|
||
def round_count(path):
|
||
try:
|
||
return len(json.load(open(path)).get("rounds", []))
|
||
except (OSError, json.JSONDecodeError, AttributeError):
|
||
return 0
|
||
|
||
|
||
def attribute_subject(evs, counters):
|
||
"""Return (subject_id, other_id) from the fire/hit event counts.
|
||
|
||
The subject is ModularBot, so its counts must equal the capture's subject
|
||
counters line. Returns (None, None) when attribution is ambiguous.
|
||
"""
|
||
fires, hits, victim_hits = {}, {}, {}
|
||
for o in evs:
|
||
t = o.get("type")
|
||
if t == "fire":
|
||
fires[o["owner"]] = fires.get(o["owner"], 0) + 1
|
||
elif t == "hit":
|
||
hits[o["owner"]] = hits.get(o["owner"], 0) + 1
|
||
if "victim" in o:
|
||
victim_hits[o["victim"]] = victim_hits.get(o["victim"], 0) + 1
|
||
if not fires:
|
||
return None, None
|
||
if counters:
|
||
strict = [o for o in fires
|
||
if fires[o] == counters["fired"]
|
||
and hits.get(o, 0) == counters["hits"]
|
||
and victim_hits.get(o, 0) == counters["hits_taken"]]
|
||
if len(strict) == 1:
|
||
subj = strict[0]
|
||
other = [o for o in fires if o != subj]
|
||
# other may legitimately be absent: an inert opponent that never fires
|
||
return subj, (other[0] if len(other) == 1 else None)
|
||
if len(fires) == 2:
|
||
ids = list(fires)
|
||
# fall back to the owner that fired the subject's declared count
|
||
if counters:
|
||
cand = [o for o in ids if fires[o] == counters["fired"]]
|
||
if len(cand) == 1:
|
||
subj = cand[0]
|
||
return subj, [o for o in ids if o != subj][0]
|
||
return ids[0], ids[1]
|
||
if len(fires) == 1 and counters:
|
||
only = next(iter(fires))
|
||
if fires[only] == counters["fired"]:
|
||
return only, None
|
||
return None, None
|
||
|
||
|
||
def parse_run(armdir, run, env_req=None):
|
||
"""Per-run data; `valid` is False for a false-negative / unstarted battle."""
|
||
evs = parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))
|
||
log_text = "".join(read_lines(os.path.join(armdir, f"run{run}.battle.log")))
|
||
counters = parse_counters(log_text)
|
||
wins = parse_first_places(log_text)
|
||
nrounds = round_count(os.path.join(armdir, f"run{run}.jsonl.rounds.json"))
|
||
|
||
r = {"run": run, "valid": False, "rounds": nrounds, "wins": wins,
|
||
"damage": 0.0, "damage_taken": 0.0, "opp_fired": 0, "mb_fired": 0,
|
||
"hits": 0, "attempts": None, "env_ok": True}
|
||
rf = os.path.join(armdir, f"run{run}.attempts")
|
||
if os.path.exists(rf):
|
||
try:
|
||
r["attempts"] = int(open(rf).read().strip())
|
||
except ValueError:
|
||
pass
|
||
|
||
# env liveness: every declared VAR=VALUE must appear verbatim in the bot's
|
||
# own boot report, else the arm setting never reached the process.
|
||
if env_req:
|
||
stdout_text = "".join(read_lines(os.path.join(armdir, f"run{run}.bot.stdout.log")))
|
||
r["env_ok"] = all(
|
||
re.search(re.escape(k) + r"\s*=\s*" + re.escape(v) + r"(\s|$)", stdout_text)
|
||
is not None for k, v in env_req.items())
|
||
|
||
if wins is None or counters is None or nrounds == 0:
|
||
return r # bot never appeared in the results block
|
||
subj, other = attribute_subject(evs, counters)
|
||
if subj is None:
|
||
return r # ambiguous owner attribution
|
||
|
||
for o in evs:
|
||
if o.get("type") == "hit":
|
||
dmg = o.get("damage", 0.0)
|
||
if o.get("owner") == subj:
|
||
r["damage"] += dmg
|
||
r["hits"] += 1
|
||
if o.get("victim") == subj:
|
||
r["damage_taken"] += dmg
|
||
elif o.get("type") == "fire":
|
||
if o.get("owner") == subj:
|
||
r["mb_fired"] += 1
|
||
elif o.get("owner") == other:
|
||
r["opp_fired"] += 1
|
||
r["valid"] = True
|
||
return r
|
||
|
||
|
||
# ── statistics ───────────────────────────────────────────────────────────────
|
||
|
||
def sign_test(deltas):
|
||
"""Two-sided exact sign test. Returns (pos, neg, ties, p)."""
|
||
pos = sum(1 for d in deltas if d > 1e-12)
|
||
neg = sum(1 for d in deltas if d < -1e-12)
|
||
ties = len(deltas) - pos - neg
|
||
n = pos + neg
|
||
if n == 0:
|
||
return pos, neg, ties, 1.0
|
||
k = min(pos, neg)
|
||
tail = sum(math.comb(n, i) for i in range(0, k + 1)) / (2 ** n)
|
||
return pos, neg, ties, min(1.0, 2.0 * tail)
|
||
|
||
|
||
def signflip_perm(deltas):
|
||
"""Paired sign-flip permutation test on mean(delta)==0 (two-sided)."""
|
||
n = len(deltas)
|
||
if n == 0:
|
||
return None
|
||
obs = abs(sum(deltas) / n)
|
||
if n <= EXACT_PAIR_CAP:
|
||
cnt = 0
|
||
for signs in itertools.product((1.0, -1.0), repeat=n):
|
||
m = abs(sum(s * d for s, d in zip(signs, deltas)) / n)
|
||
if m >= obs - 1e-12:
|
||
cnt += 1
|
||
return {"p": cnt / (2 ** n), "method": "exact", "draws": 2 ** n,
|
||
"se": 0.0}
|
||
rng = random.Random(MC_SEED)
|
||
B = MC_DRAWS
|
||
cnt = 0
|
||
for _ in range(B):
|
||
m = abs(sum((1.0 if rng.random() < 0.5 else -1.0) * d for d in deltas)) / n
|
||
if m >= obs - 1e-12:
|
||
cnt += 1
|
||
p = (cnt + 1) / (B + 1)
|
||
return {"p": p, "method": "monte-carlo", "draws": B,
|
||
"se": math.sqrt(p * (1.0 - p) / (B + 1))}
|
||
|
||
|
||
def mannwhitney_p(xa, xb):
|
||
"""Two-sided Mann-Whitney U, normal approx with tie + continuity correction."""
|
||
na, nb = len(xa), len(xb)
|
||
if na == 0 or nb == 0:
|
||
return None
|
||
vals = sorted([(v, 0) for v in xa] + [(v, 1) for v in xb])
|
||
n = na + nb
|
||
rank = [0.0] * n
|
||
tie_term = 0.0
|
||
i = 0
|
||
while i < n:
|
||
j = i
|
||
while j + 1 < n and vals[j + 1][0] == vals[i][0]:
|
||
j += 1
|
||
t = j - i + 1
|
||
tie_term += t ** 3 - t
|
||
avg = (i + j) / 2.0 + 1.0
|
||
for k in range(i, j + 1):
|
||
rank[k] = avg
|
||
i = j + 1
|
||
r1 = sum(rank[k] for k in range(n) if vals[k][1] == 0)
|
||
u1 = r1 - na * (na + 1) / 2.0
|
||
mu = na * nb / 2.0
|
||
sigma2 = (na * nb / 12.0) * ((n + 1) - tie_term / (n * (n - 1)))
|
||
if sigma2 <= 0:
|
||
return 1.0, min(u1, na * nb - u1)
|
||
z = (abs(u1 - mu) - 0.5) / math.sqrt(sigma2)
|
||
z = max(0.0, z)
|
||
return math.erfc(z / math.sqrt(2.0)), min(u1, na * nb - u1)
|
||
|
||
|
||
def describe(deltas):
|
||
n = len(deltas)
|
||
if n == 0:
|
||
return None
|
||
mean = sum(deltas) / n
|
||
sd = statistics.stdev(deltas) if n > 1 else 0.0
|
||
se = sd / math.sqrt(n) if n > 1 else 0.0
|
||
mde = Z_ALPHA_POWER * sd / math.sqrt(n) if n > 1 else 0.0
|
||
return {"n": n, "mean": mean, "sd": sd, "se": se, "mde": mde,
|
||
"min": min(deltas), "max": max(deltas)}
|
||
|
||
|
||
# ── aggregation ──────────────────────────────────────────────────────────────
|
||
|
||
def arm_summary(armdir, runs, env_req=None):
|
||
tot = {"valid_runs": 0, "runs": len(runs), "damage": 0.0,
|
||
"damage_taken": 0.0, "wins": 0, "rounds": 0, "opp_fired": 0,
|
||
"mb_fired": 0, "attempts": 0, "dropped": [], "env_bad": []}
|
||
for run in runs:
|
||
r = parse_run(armdir, run, env_req)
|
||
if not r["valid"]:
|
||
tot["dropped"].append(run)
|
||
continue
|
||
if not r["env_ok"]:
|
||
tot["env_bad"].append(run)
|
||
tot["valid_runs"] += 1
|
||
tot["damage"] += r["damage"]
|
||
tot["damage_taken"] += r["damage_taken"]
|
||
tot["wins"] += r["wins"]
|
||
tot["rounds"] += r["rounds"]
|
||
tot["opp_fired"] += r["opp_fired"]
|
||
tot["mb_fired"] += r["mb_fired"]
|
||
tot["attempts"] += (r["attempts"] or 1)
|
||
vr = tot["valid_runs"]
|
||
tot["damage_per_run"] = tot["damage"] / vr if vr else None
|
||
tot["wins_per_run"] = tot["wins"] / vr if vr else None
|
||
tot["wins_per_round"] = tot["wins"] / tot["rounds"] if tot["rounds"] else None
|
||
return tot
|
||
|
||
|
||
def discover_runs(armdir):
|
||
return sorted(int(m.group(1)) for m in
|
||
(re.fullmatch(r"run(\d+)\.jsonl", f) for f in os.listdir(armdir))
|
||
if m)
|
||
|
||
|
||
def movement_metrics(outdir, opponent, arms):
|
||
"""MEASURED character of the OPPONENT's movement, pooled over both arms.
|
||
|
||
The capture stores the opponent as the `s*` bot (sx,sy,sh,ss). We report
|
||
wall hugging, straight-line and PERIODICITY signatures. `periodicity` is
|
||
the strongest match of the signed per-tick turn series against any lag in
|
||
2..60 (a spinner/oscillator repeats its turn rate; a random dodger does
|
||
not) and `repeat` is the lag-1 version. This lets the inferred style
|
||
labels be checked against data and gives a continuous 'regular mover' axis.
|
||
"""
|
||
wall = full = n = 0
|
||
straight = 0.0
|
||
turns = []
|
||
prev = None
|
||
for arm in arms:
|
||
adir = os.path.join(outdir, opponent, arm)
|
||
if not os.path.isdir(adir):
|
||
continue
|
||
for run in discover_runs(adir):
|
||
p = os.path.join(adir, f"run{run}.jsonl")
|
||
try:
|
||
fh = open(p, errors="replace")
|
||
except OSError:
|
||
continue
|
||
with fh:
|
||
for line in fh:
|
||
if '"sx"' not in line:
|
||
continue
|
||
try:
|
||
o = json.loads(line)
|
||
except json.JSONDecodeError:
|
||
continue
|
||
if "sx" not in o:
|
||
continue
|
||
x, y, sp, hd = o["sx"], o["sy"], o.get("ss", 0.0), o.get("sh", 0.0)
|
||
n += 1
|
||
if min(x, 800 - x, y, 600 - y) < 50:
|
||
wall += 1
|
||
if sp >= 7.5:
|
||
full += 1
|
||
if prev is not None:
|
||
px, py, ph = prev
|
||
dist = math.hypot(x - px, y - py)
|
||
if dist <= 30.0: # ignore round-reset teleports
|
||
dh = (hd - ph + 180.0) % 360.0 - 180.0
|
||
turns.append(dh)
|
||
if abs(dh) < 1.0:
|
||
straight += 1
|
||
prev = (x, y, hd)
|
||
if n == 0:
|
||
return None
|
||
t = len(turns)
|
||
repeat = 0
|
||
if t > 1:
|
||
repeat = sum(1 for i in range(1, t) if abs(turns[i] - turns[i - 1]) < 0.5) / (t - 1)
|
||
# turn-mode fraction: the single most common 1-degree |turn| value. A
|
||
# wall-follower going straight (|turn|~0), a spinner (constant |turn|) and
|
||
# a fixed oscillator all concentrate here; an adaptive surfer does not.
|
||
mode_frac = 0.0
|
||
mode_val = 0.0
|
||
if t:
|
||
hist = {}
|
||
for v in turns:
|
||
b = int(round(abs(v)))
|
||
hist[b] = hist.get(b, 0) + 1
|
||
mode_val, cnt = max(hist.items(), key=lambda kv: kv[1])
|
||
mode_frac = cnt / t
|
||
return {"n": n, "wall_frac": wall / n, "full_frac": full / n,
|
||
"straight_frac": straight / t if t else 0.0,
|
||
"repeat": repeat, "mode_frac": mode_frac, "mode_val": mode_val,
|
||
"predictability": mode_frac}
|
||
|
||
|
||
def spearman(xs, ys):
|
||
"""Spearman rank correlation, returns (rho, two-sided p) or None."""
|
||
n = len(xs)
|
||
if n < 4:
|
||
return None
|
||
def ranks(v):
|
||
order = sorted(range(n), key=lambda i: v[i])
|
||
r = [0.0] * n
|
||
i = 0
|
||
while i < n:
|
||
j = i
|
||
while j + 1 < n and v[order[j + 1]] == v[order[i]]:
|
||
j += 1
|
||
avg = (i + j) / 2.0 + 1.0
|
||
for k in range(i, j + 1):
|
||
r[order[k]] = avg
|
||
i = j + 1
|
||
return r
|
||
rx, ry = ranks(xs), ranks(ys)
|
||
mx, my = sum(rx) / n, sum(ry) / n
|
||
sxy = sum((rx[i] - mx) * (ry[i] - my) for i in range(n))
|
||
sxx = sum((rx[i] - mx) ** 2 for i in range(n))
|
||
syy = sum((ry[i] - my) ** 2 for i in range(n))
|
||
if sxx == 0 or syy == 0:
|
||
return 0.0, 1.0
|
||
rho = sxy / math.sqrt(sxx * syy)
|
||
# t-approximation for the p-value
|
||
t = rho * math.sqrt((n - 2) / max(1e-12, 1 - rho * rho))
|
||
# two-sided p from the Student-t survival via the normal fallback (n>=10
|
||
# in practice here); use a small series-accurate normal approx on t.
|
||
p = math.erfc(abs(t) / math.sqrt(2.0))
|
||
return rho, p
|
||
|
||
|
||
def main():
|
||
if len(sys.argv) < 2:
|
||
print(__doc__)
|
||
sys.exit(2)
|
||
outdir = sys.argv[1].rstrip("/")
|
||
report_path = None
|
||
if "--report" in sys.argv:
|
||
report_path = sys.argv[sys.argv.index("--report") + 1]
|
||
|
||
session = {}
|
||
sp = os.path.join(outdir, "session.json")
|
||
if os.path.exists(sp):
|
||
session = json.load(open(sp))
|
||
arms = [a["name"] for a in session.get("arms", [])]
|
||
if len(arms) != 2:
|
||
print("ERROR: gauntlet analysis needs exactly 2 arms, got:", arms)
|
||
sys.exit(1)
|
||
ref, test = arms[0], arms[1]
|
||
arm_env = {a["name"]: parse_envspec(a.get("env", ""))
|
||
for a in session.get("arms", [])}
|
||
opp_meta = {o["name"]: o for o in session.get("opponents", [])}
|
||
|
||
lines = []
|
||
def out(s=""):
|
||
lines.append(s)
|
||
print(s)
|
||
|
||
out(f"# gauntlet analysis — {ref} (reference) vs {test} (treatment)")
|
||
out()
|
||
out(f"session: {outdir}")
|
||
out(f"commit: {session.get('commit')} binary sha256 {session.get('binary_sha256')}")
|
||
out(f"design: {session.get('runs')} runs x {session.get('rounds')} rounds per "
|
||
f"opponent per arm, conc={session.get('conc')}")
|
||
out(f"arms: {ref} = {next((a['env'] for a in session.get('arms',[]) if a['name']==ref), '') or '(none)'}")
|
||
out(f" {test} = {next((a['env'] for a in session.get('arms',[]) if a['name']==test), '')}")
|
||
out()
|
||
|
||
opponents = []
|
||
for d in sorted(os.listdir(outdir)):
|
||
full = os.path.join(outdir, d)
|
||
if not os.path.isdir(full) or d in ("frozen", ".work", "nimcache"):
|
||
continue
|
||
ra, rb = os.path.join(full, ref), os.path.join(full, test)
|
||
if os.path.isdir(ra) and os.path.isdir(rb):
|
||
opponents.append(d)
|
||
|
||
rows = []
|
||
dropped_opp = []
|
||
env_bad_total = 0
|
||
for opp in opponents:
|
||
sa = arm_summary(os.path.join(outdir, opp, ref), discover_runs(os.path.join(outdir, opp, ref)), arm_env.get(ref))
|
||
sb = arm_summary(os.path.join(outdir, opp, test), discover_runs(os.path.join(outdir, opp, test)), arm_env.get(test))
|
||
env_bad_total += len(sa["env_bad"]) + len(sb["env_bad"])
|
||
style = (opp_meta.get(opp, {}).get("style") or "other").strip() or "other"
|
||
ok = (sa["valid_runs"] > 0 and sb["valid_runs"] > 0
|
||
and sa["opp_fired"] + sb["opp_fired"] >= 10)
|
||
row = {"opp": opp, "style": style, "ref": sa, "test": sb, "ok": ok}
|
||
if sa["valid_runs"] == 0 or sb["valid_runs"] == 0:
|
||
dropped_opp.append((opp, f"unstarted runs ref={sa['dropped']} test={sb['dropped']}"))
|
||
elif sa["opp_fired"] + sb["opp_fired"] < 10:
|
||
dropped_opp.append(
|
||
(opp, f"inert (only {sa['opp_fired'] + sb['opp_fired']} fires "
|
||
f"in both arms — never meaningfully fought)"))
|
||
if ok:
|
||
row["d_dmg"] = sb["damage_per_run"] - sa["damage_per_run"]
|
||
row["d_win"] = sb["wins_per_run"] - sa["wins_per_run"]
|
||
rows.append(row)
|
||
# drop the raw summary dicts to keep printing clean
|
||
|
||
valid = [r for r in rows if r["ok"]]
|
||
out(f"opponents: {len(valid)} retained of {len(rows)} attempted; "
|
||
f"{len(dropped_opp)} dropped")
|
||
for opp, why in dropped_opp:
|
||
out(f" dropped {opp}: {why}")
|
||
out()
|
||
|
||
# ── per-opponent table ───────────────────────────────────────────────────
|
||
out("## Per-opponent table (MEASURED; damage is damage DEALT by ModularBot)")
|
||
out()
|
||
out(f"| opponent | style | {ref} dmg/run | {test} dmg/run | Δ dmg | "
|
||
f"{ref} wins/run | {test} wins/run | Δ wins | rounds/arm | opp fired | runs |")
|
||
out("|---|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|")
|
||
for r in sorted(valid, key=lambda x: x["style"] + x["opp"]):
|
||
sa, sb = r["ref"], r["test"]
|
||
out(f"| {r['opp']} | {r['style']} | {sa['damage_per_run']:.1f} | "
|
||
f"{sb['damage_per_run']:.1f} | {r['d_dmg']:+.1f} | "
|
||
f"{sa['wins_per_run']:.2f} | {sb['wins_per_run']:.2f} | "
|
||
f"{r['d_win']:+.2f} | {sa['rounds']}/{sb['rounds']} | "
|
||
f"{sa['opp_fired'] + sb['opp_fired']} | {sa['valid_runs']}/{sb['valid_runs']} |")
|
||
out()
|
||
|
||
d_dmg = [r["d_dmg"] for r in valid]
|
||
d_win = [r["d_win"] for r in valid]
|
||
|
||
# ── pooled arm totals ────────────────────────────────────────────────
|
||
out("## Pooled arm totals (all retained opponents)")
|
||
out()
|
||
for arm, key in ((ref, "ref"), (test, "test")):
|
||
dm = sum(r[key]["damage"] for r in valid)
|
||
wn = sum(r[key]["wins"] for r in valid)
|
||
rd = sum(r[key]["rounds"] for r in valid)
|
||
out(f"* `{arm}`: damage/run = {dm / sum(r[key]['valid_runs'] for r in valid):.1f}, "
|
||
f"round wins = {wn}/{rd} = {100.0 * wn / rd:.1f}%, "
|
||
f"damage taken/run = "
|
||
f"{sum(r[key]['damage_taken'] for r in valid) / sum(r[key]['valid_runs'] for r in valid):.1f}")
|
||
out()
|
||
# ── sign test across opponents (the headline) ────────────────────────────
|
||
out("## Cross-opponent sign test (HEADLINE)")
|
||
out()
|
||
for label, ds in (("damage/run", d_dmg), ("wins/run", d_win)):
|
||
pos, neg, ties, p = sign_test(ds)
|
||
out(f"* **{label}**: `{test}` better on **{pos}** opponents, worse on "
|
||
f"**{neg}**, tie {ties} (of {len(ds)}) — two-sided sign-test p={p:.4g}")
|
||
out()
|
||
|
||
# ── pooled paired estimate with between-opponent spread ─────────────────
|
||
out("## Pooled paired estimate (per-opponent deltas; outlier-resistant view)")
|
||
out()
|
||
for label, ds in (("damage/run", d_dmg), ("wins/run", d_win)):
|
||
d = describe(ds)
|
||
sf = signflip_perm(ds)
|
||
out(f"* **{label}**: mean Δ = {d['mean']:+.3f}, between-opponent SD = "
|
||
f"{d['sd']:.3f}, SE = {d['se']:.3f}, 95% CI ≈ "
|
||
f"[{d['mean'] - 1.96 * d['se']:+.3f}, {d['mean'] + 1.96 * d['se']:+.3f}], "
|
||
f"range [{d['min']:+.3f}, {d['max']:+.3f}]")
|
||
out(f" sign-flip permutation ({sf['method']}, {sf['draws']} draws): "
|
||
f"p={sf['p']:.4g}" + (f" (MC se={sf['se']:.1e})" if sf["se"] else ""))
|
||
out(f" MDE at n={d['n']} opponents (α=0.05, 80% power, paired): "
|
||
f"{d['mde']:.3f} {label.split('/')[0]} per run")
|
||
out()
|
||
|
||
# ── style split: the user's specific sub-claim ───────────────────────────
|
||
out("## Style split — is Δ larger on REGULAR movers than on dodgers?")
|
||
out()
|
||
out("(style labels are INFERRED from robots.json name/docs — see "
|
||
"tools/ab/opponents_gauntlet.txt)")
|
||
out()
|
||
groups = {}
|
||
for r in valid:
|
||
groups.setdefault(r["style"], []).append(r)
|
||
for style in sorted(groups):
|
||
rs = groups[style]
|
||
dd = describe([r["d_dmg"] for r in rs])
|
||
dw = describe([r["d_win"] for r in rs])
|
||
out(f"* **{style}** (n={len(rs)}): mean Δdmg/run = {dd['mean']:+.1f} "
|
||
f"(SD {dd['sd']:.1f}, MDE {dd['mde']:.1f}); mean Δwins/run = "
|
||
f"{dw['mean']:+.2f} (SD {dw['sd']:.2f}, MDE {dw['mde']:.2f})")
|
||
out()
|
||
if "regular" in groups and "dodger" in groups:
|
||
reg = [r["d_dmg"] for r in groups["regular"]]
|
||
dod = [r["d_dmg"] for r in groups["dodger"]]
|
||
p, u = mannwhitney_p(reg, dod)
|
||
out(f"* regular vs dodger Δdmg/run: Mann-Whitney p={p:.4g} (U={u:.0f}); "
|
||
f"regular n={len(reg)}, dodger n={len(dod)}")
|
||
reg = [r["d_win"] for r in groups["regular"]]
|
||
dod = [r["d_win"] for r in groups["dodger"]]
|
||
p, u = mannwhitney_p(reg, dod)
|
||
out(f"* regular vs dodger Δwins/run: Mann-Whitney p={p:.4g} (U={u:.0f})")
|
||
out()
|
||
|
||
# ── movement characterisation (MEASURED) + continuous regularity test ────
|
||
out("## Movement characterisation (MEASURED from the opponent's own trace)")
|
||
out()
|
||
out("straight% = ticks with |turn| < 1 deg; wall% = within 50 px of a wall; "
|
||
"mode% = single most common 1-deg |turn| value (a straight-liner, a "
|
||
"spinner and a fixed oscillator all concentrate here — the observable "
|
||
"signature of a 'regular' mover)")
|
||
out()
|
||
out("| opponent | inferred | straight% | wall% | full-spd% | mode% (|turn|) | Δ dmg |")
|
||
out("|---|---|---:|---:|---:|---:|---:|")
|
||
movers = {}
|
||
for r in valid:
|
||
m = movement_metrics(outdir, r["opp"], [ref, test])
|
||
if m is None:
|
||
continue
|
||
movers[r["opp"]] = m
|
||
out(f"| {r['opp']} | {r['style']} | {100 * m['straight_frac']:.1f} | "
|
||
f"{100 * m['wall_frac']:.1f} | {100 * m['full_frac']:.1f} | "
|
||
f"{100 * m['mode_frac']:.1f} ({m['mode_val']:.0f}) | {r['d_dmg']:+.1f} |")
|
||
out()
|
||
both = [r for r in valid if r["opp"] in movers]
|
||
if len(both) >= 4:
|
||
xs = [movers[r["opp"]]["predictability"] for r in both]
|
||
sp = spearman(xs, [r["d_dmg"] for r in both])
|
||
out(f"* Spearman(predictability, Δdmg/run) = {sp[0]:+.3f} (p≈{sp[1]:.3g}, "
|
||
f"n={len(both)})")
|
||
sp = spearman(xs, [r["d_win"] for r in both])
|
||
out(f"* Spearman(predictability, Δwins/run) = {sp[0]:+.3f} (p≈{sp[1]:.3g})")
|
||
for style in ("regular", "dodger"):
|
||
ps = [movers[r["opp"]]["predictability"] for r in both if r["style"] == style]
|
||
if ps:
|
||
out(f"* mean measured predictability of inferred '{style}' "
|
||
f"movers: {statistics.mean(ps):.2f} (n={len(ps)})")
|
||
out()
|
||
|
||
# ── liveness / validity ──────────────────────────────────────────────────
|
||
tot_attempts = sum(r["ref"]["attempts"] + r["test"]["attempts"] for r in valid)
|
||
retried = sum(1 for r in valid
|
||
for a, d in (("ref", r["ref"]), ("test", r["test"]))
|
||
if d["attempts"] > d["valid_runs"])
|
||
out("## Validity / liveness")
|
||
out()
|
||
out(f"* every retained opponent fired: min opponent fires (summed over both arms) "
|
||
f"= {min(r['ref']['opp_fired'] + r['test']['opp_fired'] for r in valid)}")
|
||
out(f"* arm env reached the bot in every run that was kept: "
|
||
f"{env_bad_total} run(s) had a missing declared env var"
|
||
+ (" — LOUD FAIL" if env_bad_total else " (verified from each bot's own boot report)"))
|
||
out(f"* total battle attempts {tot_attempts} for "
|
||
f"{sum(r['ref']['valid_runs'] + r['test']['valid_runs'] for r in valid)} "
|
||
f"retained runs ({retried} arm/opponent cells needed a retry)")
|
||
out(f"* retained rounds: ref={sum(r['ref']['rounds'] for r in valid)}, "
|
||
f"test={sum(r['test']['rounds'] for r in valid)}")
|
||
out()
|
||
|
||
if report_path:
|
||
with open(report_path, "w") as fh:
|
||
fh.write("\n".join(lines) + "\n")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|