#!/usr/bin/env python3 """ab_analyze.py — read a session dir produced by ab_run.sh and print the report. python3 tools/ab/ab_analyze.py [--reference ARM] Standard library only, deterministic. For every arm it prints runs, damage/run, damage taken/run, ROUND WINS, shots/run, hits taken/run and the PER-RUN values (wins cluster at 0/7 and single-run damage swings ~200, so the mean alone lies). Tests: * exact two-sided permutation test on PER-RUN values (damage/run and round wins) vs a reference arm (default: the first arm). C(14,7)=3432 for 7v7 — enumerated exactly, never sampled, whenever the combination count is small. * a round-level Fisher exact test on pooled rounds, clearly labelled anti-conservative (rounds cluster within runs). Round-win attribution comes from the events sidecar (the bot that does NOT die wins the round) and is cross-checked against the runner's `firstPlaces`, which is name-based. `firstPlaces` is the run-level truth: the per-round lines in *.results.json are CUMULATIVE standings, not round winners — do not use them. Liveness: each arm's declared env vars must appear verbatim in the bot's own boot environment report (/run.bot.stdout.log, `[env] VAR=VALUE`), so an arm whose setting never reached the process is a loud FAIL rather than a plausible-looking number. Owner attribution never relies on fired-power values (the power policy fires a continuous 0.15–1.15 range). The subject (DrussGT) is identified by matching the events sidecar's per-owner fire/hit counts against the capture's own `subject event counts:` line; ModularBot is the other bot. """ import glob import itertools import json import math import os import re import sys # beyond this many combinations we sample (deterministic seed) and say so; # 7v7 = C(14,7) = 3432 is always exact. EXACT_CAP = 20_000_000 BOT_NAME = "ModularBot" SUBJECT_NAME = "DrussGT" # the capture subject (our bot is the adversary) # ── loading ────────────────────────────────────────────────────────────────── def load_session(outdir): p = os.path.join(outdir, "session.json") if os.path.exists(p): try: return json.load(open(p)) except (json.JSONDecodeError, OSError): return None return None def parse_envspec(envspec): """`TR_A=1 TR_B=2` -> {'TR_A': '1', 'TR_B': '2'}.""" out = {} for tok in (envspec or "").split(): if "=" in tok: k, v = tok.split("=", 1) out[k] = v return out def discover_arms(outdir, session): arms = [] if session and isinstance(session.get("arms"), list): for a in session["arms"]: name = a.get("name") if not name: continue arms.append({"name": name, "env": parse_envspec(a.get("env", "")), "label": a.get("label", "")}) if arms: return arms # fallback: any subdir holding run*.jsonl for d in sorted(os.listdir(outdir)): full = os.path.join(outdir, d) if os.path.isdir(full) and discover_runs(full): arms.append({"name": d, "env": {}, "label": ""}) return arms def discover_runs(armdir): runs = [] for f in os.listdir(armdir): m = re.fullmatch(r"run(\d+)\.jsonl", f) if m: runs.append(int(m.group(1))) return sorted(runs) def read_lines(path): try: with open(path, errors="replace") as fh: return fh.readlines() except OSError: return [] def parse_events(path): evs = [] for line in read_lines(path): line = line.strip() if not line: continue try: evs.append(json.loads(line)) except json.JSONDecodeError: continue # partial line from an interrupted battle return evs def subject_counters(log_text): m = re.search( r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)" r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)", log_text) if not m: return None return {"fired": int(m.group(2)), "hits": int(m.group(3)), "hits_taken": int(m.group(6))} def parse_round_wins_by_name(log_text): """ModularBot's `firstPlaces` from the runner's final standings block.""" for m in re.finditer( r"^\s*#\d+\s+(\S+)\s+totalScore=-?\d+\s+firstPlaces=(\d+)", log_text, re.MULTILINE): if m.group(1) == BOT_NAME: return int(m.group(2)) return None def parse_round_scores(path): """Per-round score delta per bot NAME from the cumulative *.results.json lines. Used only to break ties (mutual kill / no death), never as the primary attribution.""" try: lines = json.load(open(path)).get("roundResults", []) except (OSError, json.JSONDecodeError, AttributeError): return {} prev, out = {}, {} for i, line in enumerate(lines, 1): cur = {m.group(1): int(m.group(2)) for m in re.finditer(r"(\S+) rank=\d+ score=(-?\d+)", line)} if not cur: continue out[i] = {name: sc - prev.get(name, 0) for name, sc in cur.items()} prev = cur return out def attribute_owners(evs, counters): """Return (subject_id, other_id) using fire/hit counts, never powers.""" fires, hits, victim_hits = {}, {}, {} for o in evs: t = o.get("type") if t == "fire": fires[o["owner"]] = fires.get(o["owner"], 0) + 1 elif t == "hit": hits[o["owner"]] = hits.get(o["owner"], 0) + 1 if "victim" in o: victim_hits[o["victim"]] = victim_hits.get(o["victim"], 0) + 1 if not fires: return None, None if counters: strict = [o for o in fires if fires[o] == counters["fired"] and hits.get(o, 0) == counters["hits"] and victim_hits.get(o, 0) == counters["hits_taken"]] if len(strict) == 1: subj = strict[0] return subj, _other(fires, subj) # relaxed fallbacks (logged by caller via `counters is None` etc.) cand = [o for o in fires if counters and fires[o] == counters["fired"]] if len(cand) != 1: cand = [o for o in fires if counters and victim_hits.get(o, 0) == counters["hits_taken"]] if len(cand) != 1: cand = list(fires) if len(cand) == 2: return cand[0], cand[1] if len(cand) == 1: return cand[0], _other(fires, cand[0]) return None, None def _other(owners, subj): others = [o for o in owners if o != subj] return others[0] if len(others) == 1 else None def parse_run(armdir, run): """Everything the report needs for one run; None-free on partial data.""" evs = parse_events(os.path.join(armdir, f"run{run}.events.jsonl")) log_text = "".join(read_lines(os.path.join(armdir, f"run{run}.battle.log"))) counters = subject_counters(log_text) subj, other = attribute_owners(evs, counters) r = {"run": run, "rounds": 0, "wins": None, "first_places": parse_round_wins_by_name(log_text), "subject_id": subj, "other_id": other, "attribution_exact": counters is not None, "shots": 0, "damage": 0.0, "damage_taken": 0.0, "hits_taken": 0, "hits_dealt": 0, "death_solved": 0, "death_score_agree": 0, "ambiguous": 0} if subj is None or other is None: # still count rounds so the arm shows up, but flags will be set r["rounds"] = _round_count(armdir, run) return r deaths = {} for o in evs: t = o.get("type") if t == "fire" and o.get("owner") == other: r["shots"] += 1 elif t == "hit": if o.get("owner") == other: r["damage"] += o.get("damage", 0.0) r["hits_dealt"] += 1 if o.get("victim") == other: r["damage_taken"] += o.get("damage", 0.0) r["hits_taken"] += 1 elif t == "death": deaths.setdefault(o.get("round"), []).append(o.get("victim")) nrounds = _round_count(armdir, run) r["rounds"] = nrounds scores = parse_round_scores( os.path.join(armdir, f"run{run}.jsonl.results.json")) # Per-round winner: primary = death events (a lone death means the OTHER bot # won). A round with two or zero deaths (a mutual kill or a timeout) cannot # be resolved from deaths alone, so the name-based per-round score delta is # the tie-break. The two methods are cross-checked on the unambiguous rounds. wins = death_solved = agree = ambiguous = 0 for rd in range(1, nrounds + 1): vics = deaths.get(rd, []) winner = None if len(vics) == 1: winner = other if vics[0] == subj else subj death_solved += 1 delta = scores.get(rd) if delta and SUBJECT_NAME in delta and BOT_NAME in delta: score_winner = (subj if delta[SUBJECT_NAME] > delta[BOT_NAME] else other) if winner is None: winner = score_winner ambiguous += 1 elif winner == score_winner: agree += 1 if winner == other: wins += 1 r["wins"] = wins r["death_solved"] = death_solved r["death_score_agree"] = agree r["ambiguous"] = ambiguous return r def _round_count(armdir, run): p = os.path.join(armdir, f"run{run}.jsonl.rounds.json") try: return len(json.load(open(p)).get("rounds", [])) except (OSError, json.JSONDecodeError, AttributeError): pass # fall back to distinct round numbers in the events sidecar rds = {o.get("round") for o in parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))} return len(rds) # ── statistics ─────────────────────────────────────────────────────────────── def perm_test(xa, xb): """Exact two-sided permutation test on the difference of means. Returns (obs, p, n_perm, exact_bool).""" na, nb = len(xa), len(xb) if na == 0 or nb == 0: return None obs = abs(sum(xa) / na - sum(xb) / nb) pooled = list(xa) + list(xb) n = na + nb total_sum = sum(pooled) ncomb = math.comb(n, na) if ncomb <= EXACT_CAP: cnt = 0 for combo in itertools.combinations(range(n), na): sa = sum(pooled[i] for i in combo) if abs(sa / na - (total_sum - sa) / nb) >= obs - 1e-9: cnt += 1 return obs, cnt / ncomb, ncomb, True # deterministic fallback for very large n (never hit at the default 7 runs) import random rng = random.Random(0xA1B2C3) B = 200_000 cnt = 0 for _ in range(B): idx = rng.sample(range(n), na) sa = sum(pooled[i] for i in idx) if abs(sa / na - (total_sum - sa) / nb) >= obs - 1e-9: cnt += 1 return obs, (cnt + 1) / (B + 1), ncomb, False def fisher_two_sided(a, b, c, d): """Two-sided Fisher exact on [[a,b],[c,d]].""" n = a + b + c + d if n == 0: return None row1, col1, col2 = a + b, a + c, b + d denom = math.comb(n, row1) if denom == 0: return None def hyper(x): return math.comb(col1, x) * math.comb(col2, row1 - x) / denom lo = max(0, col1 - (c + d)) hi = min(row1, col1) obs = hyper(a) return sum(hyper(x) for x in range(lo, hi + 1) if hyper(x) <= obs + 1e-12) # ── liveness ───────────────────────────────────────────────────────────────── def liveness(armdir, runs, env): """OK iff every declared env var appears verbatim in the bot's boot report of every run, the report itself ran, and the bot did not warn that the name is unrecognised (an unknown TR_*/GUN_* var is silently ignored).""" if not runs: return "FAIL", "no runs" report_seen = {r: False for r in runs} matched = {r: True for r in runs} warned = {r: [] for r in runs} for r in runs: text = "".join(read_lines(os.path.join( armdir, f"run{r}.bot.stdout.log"))) report_seen[r] = "=== ENVIRONMENT (boot report) ===" in text for k, v in env.items(): if f"[env] {k}={v}" not in text: matched[r] = False if f"[env] WARNING: {k}" in text: warned[r].append(k) dead = [r for r in runs if not report_seen[r]] miss = [r for r in runs if not matched[r]] if dead: return "FAIL", f"no bot env report in run(s) {dead}" warn_runs = [r for r in runs if warned[r]] if warn_runs: names = sorted({k for r in warn_runs for k in warned[r]}) return "FAIL", (f"bot ignored arm env {names} (unrecognised) in " f"run(s) {warn_runs}") if env and miss: vars_txt = " ".join(f"{k}={v}" for k, v in env.items()) return "FAIL", f"arm env '{vars_txt}' not in boot report of run(s) {miss}" if not env: return "OK", f"{len(runs)}/{len(runs)} runs: no arm env; report present" vars_txt = " ".join(f"{k}={v}" for k, v in env.items()) return "OK", f"{len(runs)}/{len(runs)} runs: {vars_txt} applied" # ── report ─────────────────────────────────────────────────────────────────── def main(): args = sys.argv[1:] if not args or args[0] in ("-h", "--help"): print(__doc__) return 0 if args else 2 outdir = args[0] reference = None if "--reference" in args: reference = args[args.index("--reference") + 1] if not os.path.isdir(outdir): print(f"ERROR: not a directory: {outdir}", file=sys.stderr) return 2 session = load_session(outdir) arms = discover_arms(outdir, session) if not arms: print(f"ERROR: no arms found in {outdir}", file=sys.stderr) return 2 if reference is None: reference = arms[0]["name"] data = {} for arm in arms: armdir = os.path.join(outdir, arm["name"]) runs = discover_runs(armdir) per = [parse_run(armdir, r) for r in runs] data[arm["name"]] = {"dir": armdir, "runs": runs, "per": per, "env": arm["env"], "label": arm["label"]} if session: print(f"# session {outdir}") print(f"# commit={session.get('commit','?')} " f"binary_sha256={session.get('binary_sha256','?')} " f"rounds={session.get('rounds','?')} runs={session.get('runs','?')} " f"conc={session.get('conc','?')} ts={session.get('timestamp','?')}") print() # ── summary table ──────────────────────────────────────────────────────── print("ARM SUMMARY") hdr = (f"{'arm':<14} {'runs':>4} {'dmg/run':>8} {'dmgtk/run':>9} " f"{'wins':>7} {'win%':>6} {'shots/run':>9} {'hitstk/run':>10}") print(hdr) print("-" * len(hdr)) for arm in arms: name = arm["name"] per = data[name]["per"] n = len(per) if n == 0: print(f"{name:<14} {0:>4} (no runs)") continue dmg = sum(p["damage"] for p in per) / n dmgv = sum(p["damage_taken"] for p in per) / n shots = sum(p["shots"] for p in per) / n htk = sum(p["hits_taken"] for p in per) / n wins = sum(p["wins"] for p in per if p["wins"] is not None) rounds = sum(p["rounds"] for p in per) pct = f"{100.0 * wins / rounds:.1f}" if rounds else "n/a" print(f"{name:<14} {n:>4} {dmg:>8.0f} {dmgv:>9.0f} " f"{str(wins) + '/' + str(rounds):>7} {pct:>6} {shots:>9.0f} {htk:>10.1f}") # ── per-run values ─────────────────────────────────────────────────────── print("\nPER-RUN (never just the mean)") for arm in arms: name = arm["name"] per = data[name]["per"] dmgs = " ".join( f"r{p['run']}={p['damage']:.0f}" for p in per) wins = " ".join( f"r{p['run']}={p['wins']}/{p['rounds']}" if p["wins"] is not None else f"r{p['run']}=?/{p['rounds']}" for p in per) print(f" {name:<14} dmg: {dmgs}") print(f" {'':<14} wins: {wins}") # ── exact permutation tests ────────────────────────────────────────────── if reference not in data: print(f"\nWARNING: reference arm '{reference}' not found; skipping tests") return 1 print(f"\nEXACT TWO-SIDED PERMUTATION TEST (per-run values) vs `{reference}`") print(f"{'metric':<12} {'arm':<14} {'obs(diff)':>12} {'p':>8} permutations") print("-" * 64) ref = data[reference]["per"] for arm in arms: name = arm["name"] if name == reference: continue for label, key in (("dmg/run", "damage"), ("round wins", "wins")): xa = [p[key] for p in ref if p[key] is not None] xb = [p[key] for p in data[name]["per"] if p[key] is not None] res = perm_test(xa, xb) if res is None: print(f"{label:<12} {name:<14} {'n/a':>12} {'n/a':>8}") continue obs, p, ncomb, exact = res note = f"C({len(xa)+len(xb)},{len(xa)})={ncomb}" + ( "" if exact else " SAMPLED") print(f"{label:<12} {name:<14} {obs:>+12.3f} {p:>8.4f} {note}") # ── round-level test (anti-conservative) ───────────────────────────────── print("\nROUND-LEVEL TEST (pooled rounds, Fisher exact) vs " f"`{reference}` — ANTI-CONSERVATIVE: rounds cluster within runs") print(f"{'arm':<14} {'ref wins':>10} {'arm wins':>10} {'p':>8}") print("-" * 46) refw = sum(p["wins"] for p in ref if p["wins"] is not None) refr = sum(p["rounds"] for p in ref) for arm in arms: name = arm["name"] if name == reference: continue per = data[name]["per"] w = sum(p["wins"] for p in per if p["wins"] is not None) rr = sum(p["rounds"] for p in per) p = fisher_two_sided(w, rr - w, refw, refr - refw) pstr = f"{p:.4f}" if p is not None else "n/a" print(f"{name:<14} {str(refw) + '/' + str(refr):>10} " f"{str(w) + '/' + str(rr):>10} {pstr:>8}") # ── liveness ───────────────────────────────────────────────────────────── print("\nLIVENESS (arm env applied in the bot's own boot report)") lv_fail = 0 for arm in arms: name = arm["name"] status, why = liveness(data[name]["dir"], data[name]["runs"], data[name]["env"]) if status == "FAIL": lv_fail += 1 print(f" {name:<14} {status:<4} ({why})") # ── round-win attribution cross-check ──────────────────────────────────── print("\nROUND-WIN ATTRIBUTION (events primary; score tie-break for " "mutual-kill / timeout rounds)") for arm in arms: name = arm["name"] runs_ok = runs_tot = 0 dsolve = dagree = damb = 0 for p in data[name]["per"]: if p["wins"] is not None and p["first_places"] is not None: runs_tot += 1 if p["wins"] == p["first_places"]: runs_ok += 1 dsolve += p["death_solved"] dagree += p["death_score_agree"] damb += p["ambiguous"] status = ("OK" if runs_tot and runs_ok == runs_tot else ("n/a" if not runs_tot else "MISMATCH")) print(f" {name:<14} wins==firstPlaces {runs_ok}/{runs_tot} runs {status}; " f"single-death rounds agree with score {dagree}/{dsolve} " f"({damb} tie-broken)") return 1 if lv_fail else 0 if __name__ == "__main__": sys.exit(main())