#!/usr/bin/env python3 """ab_analyze.py — read a session dir produced by ab_run.sh and print the report. python3 tools/ab/ab_analyze.py [--reference ARM] Standard library only, deterministic. For every arm it prints runs, damage/run, damage taken/run, ROUND WINS, shots/run, hits taken/run and the PER-RUN values (wins cluster at 0/7 and single-run damage swings ~200, so the mean alone lies). Tests (all two-sided, on PER-RUN values unless stated): * permutation test on the difference of means. Full enumeration when C(n, na) <= EXACT_CAP (7v7 -> C(14,7)=3432, always exact); otherwise a Monte-Carlo permutation test with a fixed MC_DRAWS draws and the fixed seed MC_SEED, reported with its Monte-Carlo standard error. Every p-value says which method produced it (ncomb is C(60,30) ~ 1.2e17 at 30v30). * Mann-Whitney U (rank-sum) cross-check, normal approximation with the standard tie correction and a continuity correction. * minimum detectable effect for n-per-arm (alpha=0.05 two-sided, 80% power) derived from the observed pooled per-run SD, so a null result can be told apart from an under-powered one. * a round-level Fisher exact test on pooled rounds, clearly labelled anti-conservative (rounds cluster within runs). Round-win attribution comes from the events sidecar (the bot that does NOT die wins the round) and is cross-checked against the runner's `firstPlaces`, which is name-based. `firstPlaces` is the run-level truth: the per-round lines in *.results.json are CUMULATIVE standings, not round winners — do not use them. Liveness: each arm's declared env vars must appear verbatim in the bot's own boot environment report (/run.bot.stdout.log, `[env] VAR=VALUE`), so an arm whose setting never reached the process is a loud FAIL rather than a plausible-looking number. Owner attribution never relies on fired-power values (the power policy fires a continuous 0.15–1.15 range). The subject (DrussGT) is identified by matching the events sidecar's per-owner fire/hit counts against the capture's own `subject event counts:` line; ModularBot is the other bot. """ import glob import itertools import json import math import os import random import re import sys # beyond this many combinations we sample (deterministic seed) and say so; # 7v7 = C(14,7) = 3432 is always exact. EXACT_CAP = 20_000_000 # Monte-Carlo permutation test: fixed draw count and fixed seed, so the p-value # is reproducible byte-for-byte across runs of this analyzer. MC_DRAWS = 1_000_000 MC_SEED = 0x5EED5EED # z_{0.975} + z_{0.80}, the constant in the two-sample MDE at alpha=0.05 (two # sided) and 80% power: MDE = 2.8016 * sd * sqrt(2/n). Z_ALPHA_POWER = 1.959963984540054 + 0.8416212335729143 BOT_NAME = "ModularBot" SUBJECT_NAME = "DrussGT" # the capture subject (our bot is the adversary) # ── loading ────────────────────────────────────────────────────────────────── def load_session(outdir): p = os.path.join(outdir, "session.json") if os.path.exists(p): try: return json.load(open(p)) except (json.JSONDecodeError, OSError): return None return None def parse_envspec(envspec): """`TR_A=1 TR_B=2` -> {'TR_A': '1', 'TR_B': '2'}.""" out = {} for tok in (envspec or "").split(): if "=" in tok: k, v = tok.split("=", 1) out[k] = v return out def discover_arms(outdir, session): arms = [] if session and isinstance(session.get("arms"), list): for a in session["arms"]: name = a.get("name") if not name: continue arms.append({"name": name, "env": parse_envspec(a.get("env", "")), "label": a.get("label", "")}) if arms: return arms # fallback: any subdir holding run*.jsonl for d in sorted(os.listdir(outdir)): full = os.path.join(outdir, d) if os.path.isdir(full) and discover_runs(full): arms.append({"name": d, "env": {}, "label": ""}) return arms def discover_runs(armdir): runs = [] for f in os.listdir(armdir): m = re.fullmatch(r"run(\d+)\.jsonl", f) if m: runs.append(int(m.group(1))) return sorted(runs) def read_lines(path): try: with open(path, errors="replace") as fh: return fh.readlines() except OSError: return [] def parse_events(path): evs = [] for line in read_lines(path): line = line.strip() if not line: continue try: evs.append(json.loads(line)) except json.JSONDecodeError: continue # partial line from an interrupted battle return evs def subject_counters(log_text): m = re.search( r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)" r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)", log_text) if not m: return None return {"fired": int(m.group(2)), "hits": int(m.group(3)), "hits_taken": int(m.group(6))} def parse_round_wins_by_name(log_text): """ModularBot's `firstPlaces` from the runner's final standings block.""" for m in re.finditer( r"^\s*#\d+\s+(\S+)\s+totalScore=-?\d+\s+firstPlaces=(\d+)", log_text, re.MULTILINE): if m.group(1) == BOT_NAME: return int(m.group(2)) return None def parse_round_scores(path): """Per-round score delta per bot NAME from the cumulative *.results.json lines. Used only to break ties (mutual kill / no death), never as the primary attribution.""" try: lines = json.load(open(path)).get("roundResults", []) except (OSError, json.JSONDecodeError, AttributeError): return {} prev, out = {}, {} for i, line in enumerate(lines, 1): cur = {m.group(1): int(m.group(2)) for m in re.finditer(r"(\S+) rank=\d+ score=(-?\d+)", line)} if not cur: continue out[i] = {name: sc - prev.get(name, 0) for name, sc in cur.items()} prev = cur return out def attribute_owners(evs, counters): """Return (subject_id, other_id) using fire/hit counts, never powers.""" fires, hits, victim_hits = {}, {}, {} for o in evs: t = o.get("type") if t == "fire": fires[o["owner"]] = fires.get(o["owner"], 0) + 1 elif t == "hit": hits[o["owner"]] = hits.get(o["owner"], 0) + 1 if "victim" in o: victim_hits[o["victim"]] = victim_hits.get(o["victim"], 0) + 1 if not fires: return None, None if counters: strict = [o for o in fires if fires[o] == counters["fired"] and hits.get(o, 0) == counters["hits"] and victim_hits.get(o, 0) == counters["hits_taken"]] if len(strict) == 1: subj = strict[0] return subj, _other(fires, subj) # relaxed fallbacks (logged by caller via `counters is None` etc.) cand = [o for o in fires if counters and fires[o] == counters["fired"]] if len(cand) != 1: cand = [o for o in fires if counters and victim_hits.get(o, 0) == counters["hits_taken"]] if len(cand) != 1: cand = list(fires) if len(cand) == 2: return cand[0], cand[1] if len(cand) == 1: return cand[0], _other(fires, cand[0]) return None, None def _other(owners, subj): others = [o for o in owners if o != subj] return others[0] if len(others) == 1 else None def parse_run(armdir, run): """Everything the report needs for one run; None-free on partial data.""" evs = parse_events(os.path.join(armdir, f"run{run}.events.jsonl")) log_text = "".join(read_lines(os.path.join(armdir, f"run{run}.battle.log"))) counters = subject_counters(log_text) subj, other = attribute_owners(evs, counters) r = {"run": run, "rounds": 0, "wins": None, "first_places": parse_round_wins_by_name(log_text), "subject_id": subj, "other_id": other, "attribution_exact": counters is not None, "shots": 0, "damage": 0.0, "damage_taken": 0.0, "hits_taken": 0, "hits_dealt": 0, "death_solved": 0, "death_score_agree": 0, "ambiguous": 0} if subj is None or other is None: # still count rounds so the arm shows up, but flags will be set r["rounds"] = _round_count(armdir, run) return r deaths = {} for o in evs: t = o.get("type") if t == "fire" and o.get("owner") == other: r["shots"] += 1 elif t == "hit": if o.get("owner") == other: r["damage"] += o.get("damage", 0.0) r["hits_dealt"] += 1 if o.get("victim") == other: r["damage_taken"] += o.get("damage", 0.0) r["hits_taken"] += 1 elif t == "death": deaths.setdefault(o.get("round"), []).append(o.get("victim")) nrounds = _round_count(armdir, run) r["rounds"] = nrounds scores = parse_round_scores( os.path.join(armdir, f"run{run}.jsonl.results.json")) # Per-round winner: primary = death events (a lone death means the OTHER bot # won). A round with two or zero deaths (a mutual kill or a timeout) cannot # be resolved from deaths alone, so the name-based per-round score delta is # the tie-break. The two methods are cross-checked on the unambiguous rounds. wins = death_solved = agree = ambiguous = 0 for rd in range(1, nrounds + 1): vics = deaths.get(rd, []) winner = None if len(vics) == 1: winner = other if vics[0] == subj else subj death_solved += 1 delta = scores.get(rd) if delta and SUBJECT_NAME in delta and BOT_NAME in delta: score_winner = (subj if delta[SUBJECT_NAME] > delta[BOT_NAME] else other) if winner is None: winner = score_winner ambiguous += 1 elif winner == score_winner: agree += 1 if winner == other: wins += 1 r["wins"] = wins r["death_solved"] = death_solved r["death_score_agree"] = agree r["ambiguous"] = ambiguous return r def _round_count(armdir, run): p = os.path.join(armdir, f"run{run}.jsonl.rounds.json") try: return len(json.load(open(p)).get("rounds", [])) except (OSError, json.JSONDecodeError, AttributeError): pass # fall back to distinct round numbers in the events sidecar rds = {o.get("round") for o in parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))} return len(rds) # ── statistics ─────────────────────────────────────────────────────────────── def perm_test(xa, xb): """Two-sided permutation test on the difference of means. Exact full enumeration when C(n, na) <= EXACT_CAP (7v7 is 3432), otherwise a Monte-Carlo permutation test with MC_DRAWS fixed draws and the fixed seed MC_SEED. Returns None if either arm is empty, else a dict: obs observed |mean(xa) - mean(xb)| (what the p-value tests) signed mean(xa) - mean(xb) (for display; negative means xb is larger) p two-sided p-value draws number of permutations (enumerated or sampled) method 'exact' | 'monte-carlo' se Monte-Carlo standard error of p (0.0 for the exact test) seed MC seed (None for the exact test) The MC p-value uses the (cnt + 1) / (B + 1) estimator, so it is never 0. """ na, nb = len(xa), len(xb) if na == 0 or nb == 0: return None obs = abs(sum(xa) / na - sum(xb) / nb) pooled = list(xa) + list(xb) n = na + nb total = sum(pooled) ncomb = math.comb(n, na) if ncomb <= EXACT_CAP: cnt = 0 for combo in itertools.combinations(range(n), na): sa = sum(pooled[i] for i in combo) if abs(sa / na - (total - sa) / nb) >= obs - 1e-9: cnt += 1 return {"obs": obs, "signed": sum(xa) / na - sum(xb) / nb, "p": cnt / ncomb, "draws": ncomb, "method": "exact", "se": 0.0, "seed": None} rng = random.Random(MC_SEED) sample = rng.sample idx_range = range(n) B = MC_DRAWS cnt = 0 for _ in range(B): sa = 0 for i in sample(idx_range, na): sa += pooled[i] if abs(sa / na - (total - sa) / nb) >= obs - 1e-9: cnt += 1 p = (cnt + 1) / (B + 1) se = math.sqrt(p * (1.0 - p) / (B + 1)) return {"obs": obs, "signed": sum(xa) / na - sum(xb) / nb, "p": p, "draws": B, "method": "monte-carlo", "se": se, "seed": MC_SEED} def mannwhitney_p(xa, xb): """Two-sided Mann-Whitney U (rank-sum) test, normal approximation with the standard tie correction and a continuity correction. Returns (p, U) or None when either arm is empty. U is the smaller of U1, U2.""" na, nb = len(xa), len(xb) n = na + nb if na == 0 or nb == 0: return None vals = sorted([(v, 0) for v in xa] + [(v, 1) for v in xb]) rank = [0.0] * n tie_term = 0.0 i = 0 while i < n: j = i while j + 1 < n and vals[j + 1][0] == vals[i][0]: j += 1 t = j - i + 1 avg = (i + j) / 2.0 + 1.0 tie_term += t ** 3 - t for k in range(i, j + 1): rank[k] = avg i = j + 1 r1 = sum(rank[k] for k in range(n) if vals[k][1] == 0) u1 = r1 - na * (na + 1) / 2.0 mu = na * nb / 2.0 sigma2 = (na * nb / 12.0) * ((n + 1) - tie_term / (n * (n - 1))) if sigma2 <= 0: return (1.0, min(u1, na * nb - u1)) z = (abs(u1 - mu) - 0.5) / math.sqrt(sigma2) if z < 0.0: z = 0.0 p = math.erfc(z / math.sqrt(2.0)) return (p, min(u1, na * nb - u1)) def _var(x): m = sum(x) / len(x) return sum((v - m) ** 2 for v in x) / (len(x) - 1) def min_detectable_effect(sd, n_per_arm): """Two-sample MDE at alpha=0.05 two-sided, 80% power, equal n.""" return Z_ALPHA_POWER * sd * math.sqrt(2.0 / n_per_arm) def fisher_two_sided(a, b, c, d): """Two-sided Fisher exact on [[a,b],[c,d]].""" n = a + b + c + d if n == 0: return None row1, col1, col2 = a + b, a + c, b + d denom = math.comb(n, row1) if denom == 0: return None def hyper(x): return math.comb(col1, x) * math.comb(col2, row1 - x) / denom lo = max(0, col1 - (c + d)) hi = min(row1, col1) obs = hyper(a) return sum(hyper(x) for x in range(lo, hi + 1) if hyper(x) <= obs + 1e-12) # ── liveness ───────────────────────────────────────────────────────────────── def liveness(armdir, runs, env): """OK iff every declared env var appears verbatim in the bot's boot report of every run, the report itself ran, and the bot did not warn that the name is unrecognised (an unknown TR_*/GUN_* var is silently ignored).""" if not runs: return "FAIL", "no runs" report_seen = {r: False for r in runs} matched = {r: True for r in runs} warned = {r: [] for r in runs} for r in runs: text = "".join(read_lines(os.path.join( armdir, f"run{r}.bot.stdout.log"))) report_seen[r] = "=== ENVIRONMENT (boot report) ===" in text for k, v in env.items(): if f"[env] {k}={v}" not in text: matched[r] = False if f"[env] WARNING: {k}" in text: warned[r].append(k) dead = [r for r in runs if not report_seen[r]] miss = [r for r in runs if not matched[r]] if dead: return "FAIL", f"no bot env report in run(s) {dead}" warn_runs = [r for r in runs if warned[r]] if warn_runs: names = sorted({k for r in warn_runs for k in warned[r]}) return "FAIL", (f"bot ignored arm env {names} (unrecognised) in " f"run(s) {warn_runs}") if env and miss: vars_txt = " ".join(f"{k}={v}" for k, v in env.items()) return "FAIL", f"arm env '{vars_txt}' not in boot report of run(s) {miss}" if not env: return "OK", f"{len(runs)}/{len(runs)} runs: no arm env; report present" vars_txt = " ".join(f"{k}={v}" for k, v in env.items()) return "OK", f"{len(runs)}/{len(runs)} runs: {vars_txt} applied" # ── [bb] shift liveness ────────────────────────────────────────────────────── _BB_SHIFT_RE = re.compile(r"\[bb\].*?shift=([+-]?\d+(?:\.\d+)?)deg") def bb_shifts(armdir, runs): """All `[bb] ... shift=Xdeg` values in an arm's bot stdout (TR_BITBRAIN_LOG=1). Returns (values, runs_with_lines, runs). A placebo that never applies a correction emits ZERO [bb] lines; `bbLog` is only reached inside the same `trained >= minObs` branch that computes the applied shift.""" vals = [] runs_with = 0 for r in runs: text = "".join(read_lines(os.path.join( armdir, f"run{r}.bot.stdout.log"))) found = _BB_SHIFT_RE.findall(text) if found: runs_with += 1 vals.extend(float(x) for x in found) return vals, runs_with, len(runs) # ── report ─────────────────────────────────────────────────────────────────── def main(): args = sys.argv[1:] if not args or args[0] in ("-h", "--help"): print(__doc__) return 0 if args else 2 outdir = args[0] reference = None if "--reference" in args: reference = args[args.index("--reference") + 1] if not os.path.isdir(outdir): print(f"ERROR: not a directory: {outdir}", file=sys.stderr) return 2 session = load_session(outdir) arms = discover_arms(outdir, session) if not arms: print(f"ERROR: no arms found in {outdir}", file=sys.stderr) return 2 if reference is None: reference = arms[0]["name"] data = {} for arm in arms: armdir = os.path.join(outdir, arm["name"]) runs = discover_runs(armdir) per = [parse_run(armdir, r) for r in runs] data[arm["name"]] = {"dir": armdir, "runs": runs, "per": per, "env": arm["env"], "label": arm["label"]} if session: print(f"# session {outdir}") print(f"# commit={session.get('commit','?')} " f"binary_sha256={session.get('binary_sha256','?')} " f"rounds={session.get('rounds','?')} runs={session.get('runs','?')} " f"conc={session.get('conc','?')} ts={session.get('timestamp','?')}") print() # ── summary table ──────────────────────────────────────────────────────── print("ARM SUMMARY") hdr = (f"{'arm':<14} {'runs':>4} {'dmg/run':>8} {'dmgtk/run':>9} " f"{'wins':>7} {'win%':>6} {'shots/run':>9} {'hitstk/run':>10}") print(hdr) print("-" * len(hdr)) for arm in arms: name = arm["name"] per = data[name]["per"] n = len(per) if n == 0: print(f"{name:<14} {0:>4} (no runs)") continue dmg = sum(p["damage"] for p in per) / n dmgv = sum(p["damage_taken"] for p in per) / n shots = sum(p["shots"] for p in per) / n htk = sum(p["hits_taken"] for p in per) / n wins = sum(p["wins"] for p in per if p["wins"] is not None) rounds = sum(p["rounds"] for p in per) pct = f"{100.0 * wins / rounds:.1f}" if rounds else "n/a" print(f"{name:<14} {n:>4} {dmg:>8.0f} {dmgv:>9.0f} " f"{str(wins) + '/' + str(rounds):>7} {pct:>6} {shots:>9.0f} {htk:>10.1f}") # ── per-run values ─────────────────────────────────────────────────────── print("\nPER-RUN (never just the mean)") for arm in arms: name = arm["name"] per = data[name]["per"] dmgs = " ".join( f"r{p['run']}={p['damage']:.0f}" for p in per) wins = " ".join( f"r{p['run']}={p['wins']}/{p['rounds']}" if p["wins"] is not None else f"r{p['run']}=?/{p['rounds']}" for p in per) print(f" {name:<14} dmg: {dmgs}") print(f" {'':<14} wins: {wins}") # ── permutation tests (exact for 7v7, Monte-Carlo for 30v30) ───────────── if reference not in data: print(f"\nWARNING: reference arm '{reference}' not found; skipping tests") return 1 ref = data[reference]["per"] print("\nPAIRWISE PERMUTATION TEST (per-run values) + MANN-WHITNEY CROSS-CHECK") print(f"permutation: exact when C(n,na) <= {EXACT_CAP:,}; otherwise " f"Monte-Carlo {MC_DRAWS:,} draws, seed={MC_SEED:#x}, " f"p = (cnt+1)/(B+1), se = sqrt(p(1-p)/(B+1))") hdr = (f"{'metric':<11} {'A':<11} {'B':<11} {'diff(A-B)':>11} " f"{'perm p':>9} {'method':<12} {'MC se':>7} {'MW p':>9} {'MW U':>8}") print(hdr) print("-" * len(hdr)) for i in range(len(arms)): for j in range(i + 1, len(arms)): ana, bnb = arms[i]["name"], arms[j]["name"] for label, key in (("dmg/run", "damage"), ("round wins", "wins")): xa = [p[key] for p in data[ana]["per"] if p[key] is not None] xb = [p[key] for p in data[bnb]["per"] if p[key] is not None] res = perm_test(xa, xb) mw = mannwhitney_p(xa, xb) if res is None: print(f"{label:<11} {ana:<11} {bnb:<11} {'n/a':>11}") continue mstr = ("exact" if res["method"] == "exact" else f"MC/B={res['draws']:,}") sestr = "-" if res["method"] == "exact" else f"{res['se']:.4f}" mwp = f"{mw[0]:.4f}" if mw else "n/a" mwu = f"{mw[1]:.1f}" if mw else "n/a" print(f"{label:<11} {ana:<11} {bnb:<11} {res['signed']:>+11.3f} " f"{res['p']:>9.4f} {mstr:<12} {sestr:>7} {mwp:>9} {mwu:>8}") # ── minimum detectable effect ────────────────────────────────────────── print("\nMINIMUM DETECTABLE EFFECT (two-sample, alpha=0.05 two-sided, " "80% power; MDE = 2.8016*sd*sqrt(2/n))") print(f"{'metric':<11} {'n/arm':>6} {'sd(control)':>12} {'MDE(abs)':>10} " f"{'MDE vs control mean':>22}") print("-" * 64) ctrl = data[reference]["per"] for label, key in (("dmg/run", "damage"), ("round wins", "wins")): vals = [p[key] for p in ctrl if p[key] is not None] if len(vals) < 2: continue sd = math.sqrt(_var(vals)) mde = min_detectable_effect(sd, len(vals)) cmean = sum(vals) / len(vals) rel = (f"{100.0*mde/abs(cmean):.1f}% of {cmean:.1f}" if cmean else "n/a") print(f"{label:<11} {len(vals):>6} {sd:>12.3f} {mde:>10.3f} {rel:>22}") # ── round-level test (anti-conservative) ───────────────────────────────── print("\nROUND-LEVEL TEST (pooled rounds, Fisher exact) vs " f"`{reference}` — ANTI-CONSERVATIVE: rounds cluster within runs") print(f"{'arm':<14} {'ref wins':>10} {'arm wins':>10} {'p':>8}") print("-" * 46) refw = sum(p["wins"] for p in ref if p["wins"] is not None) refr = sum(p["rounds"] for p in ref) for arm in arms: name = arm["name"] if name == reference: continue per = data[name]["per"] w = sum(p["wins"] for p in per if p["wins"] is not None) rr = sum(p["rounds"] for p in per) p = fisher_two_sided(w, rr - w, refw, refr - refw) pstr = f"{p:.4f}" if p is not None else "n/a" print(f"{name:<14} {str(refw) + '/' + str(refr):>10} " f"{str(w) + '/' + str(rr):>10} {pstr:>8}") # ── liveness ───────────────────────────────────────────────────────────── print("\nLIVENESS (arm env applied in the bot's own boot report)") lv_fail = 0 for arm in arms: name = arm["name"] status, why = liveness(data[name]["dir"], data[name]["runs"], data[name]["env"]) if status == "FAIL": lv_fail += 1 print(f" {name:<14} {status:<4} ({why})") # ── [bb] applied-shift check (a placebo must apply exactly 0) ──────────── print("\n[bb] APPLIED-SHIFT CHECK (from bot stdout; needs TR_BITBRAIN_LOG=1). " "A provably-zero placebo emits ZERO [bb] lines.") print(f" {'arm':<14} {'runs w/log':>11} {'lines':>7} {'min':>9} " f"{'max':>9} {'zeros':>6}") for arm in arms: name = arm["name"] vals, runs_with, nruns = bb_shifts(data[name]["dir"], data[name]["runs"]) if not vals: print(f" {name:<14} {runs_with:>4}/{nruns:<4} {0:>7} " f"{'-':>9} {'-':>9} {'-':>6}") else: zeros = sum(1 for v in vals if v == 0.0) print(f" {name:<14} {runs_with:>4}/{nruns:<4} {len(vals):>7} " f"{min(vals):>+9.2f} {max(vals):>+9.2f} {zeros:>6}") # ── round-win attribution cross-check ──────────────────────────────────── print("\nROUND-WIN ATTRIBUTION (events primary; score tie-break for " "mutual-kill / timeout rounds)") for arm in arms: name = arm["name"] runs_ok = runs_tot = 0 dsolve = dagree = damb = 0 for p in data[name]["per"]: if p["wins"] is not None and p["first_places"] is not None: runs_tot += 1 if p["wins"] == p["first_places"]: runs_ok += 1 dsolve += p["death_solved"] dagree += p["death_score_agree"] damb += p["ambiguous"] status = ("OK" if runs_tot and runs_ok == runs_tot else ("n/a" if not runs_tot else "MISMATCH")) print(f" {name:<14} wins==firstPlaces {runs_ok}/{runs_tot} runs {status}; " f"single-death rounds agree with score {dagree}/{dsolve} " f"({damb} tie-broken)") return 1 if lv_fail else 0 if __name__ == "__main__": sys.exit(main())