521 lines
20 KiB
Python
Executable File
521 lines
20 KiB
Python
Executable File
#!/usr/bin/env python3
|
||
"""ab_analyze.py — read a session dir produced by ab_run.sh and print the report.
|
||
|
||
python3 tools/ab/ab_analyze.py <session_dir> [--reference ARM]
|
||
|
||
Standard library only, deterministic. For every arm it prints runs, damage/run,
|
||
damage taken/run, ROUND WINS, shots/run, hits taken/run and the PER-RUN values
|
||
(wins cluster at 0/7 and single-run damage swings ~200, so the mean alone lies).
|
||
|
||
Tests:
|
||
* exact two-sided permutation test on PER-RUN values (damage/run and round
|
||
wins) vs a reference arm (default: the first arm). C(14,7)=3432 for 7v7 —
|
||
enumerated exactly, never sampled, whenever the combination count is small.
|
||
* a round-level Fisher exact test on pooled rounds, clearly labelled
|
||
anti-conservative (rounds cluster within runs).
|
||
|
||
Round-win attribution comes from the events sidecar (the bot that does NOT die
|
||
wins the round) and is cross-checked against the runner's `firstPlaces`, which
|
||
is name-based. `firstPlaces` is the run-level truth: the per-round lines in
|
||
*.results.json are CUMULATIVE standings, not round winners — do not use them.
|
||
|
||
Liveness: each arm's declared env vars must appear verbatim in the bot's own
|
||
boot environment report (<arm>/run<N>.bot.stdout.log, `[env] VAR=VALUE`), so an
|
||
arm whose setting never reached the process is a loud FAIL rather than a
|
||
plausible-looking number.
|
||
|
||
Owner attribution never relies on fired-power values (the power policy fires a
|
||
continuous 0.15–1.15 range). The subject (DrussGT) is identified by matching
|
||
the events sidecar's per-owner fire/hit counts against the capture's own
|
||
`subject event counts:` line; ModularBot is the other bot.
|
||
"""
|
||
import glob
|
||
import itertools
|
||
import json
|
||
import math
|
||
import os
|
||
import re
|
||
import sys
|
||
|
||
# beyond this many combinations we sample (deterministic seed) and say so;
|
||
# 7v7 = C(14,7) = 3432 is always exact.
|
||
EXACT_CAP = 20_000_000
|
||
BOT_NAME = "ModularBot"
|
||
SUBJECT_NAME = "DrussGT" # the capture subject (our bot is the adversary)
|
||
|
||
|
||
# ── loading ──────────────────────────────────────────────────────────────────
|
||
|
||
def load_session(outdir):
|
||
p = os.path.join(outdir, "session.json")
|
||
if os.path.exists(p):
|
||
try:
|
||
return json.load(open(p))
|
||
except (json.JSONDecodeError, OSError):
|
||
return None
|
||
return None
|
||
|
||
|
||
def parse_envspec(envspec):
|
||
"""`TR_A=1 TR_B=2` -> {'TR_A': '1', 'TR_B': '2'}."""
|
||
out = {}
|
||
for tok in (envspec or "").split():
|
||
if "=" in tok:
|
||
k, v = tok.split("=", 1)
|
||
out[k] = v
|
||
return out
|
||
|
||
|
||
def discover_arms(outdir, session):
|
||
arms = []
|
||
if session and isinstance(session.get("arms"), list):
|
||
for a in session["arms"]:
|
||
name = a.get("name")
|
||
if not name:
|
||
continue
|
||
arms.append({"name": name,
|
||
"env": parse_envspec(a.get("env", "")),
|
||
"label": a.get("label", "")})
|
||
if arms:
|
||
return arms
|
||
# fallback: any subdir holding run*.jsonl
|
||
for d in sorted(os.listdir(outdir)):
|
||
full = os.path.join(outdir, d)
|
||
if os.path.isdir(full) and discover_runs(full):
|
||
arms.append({"name": d, "env": {}, "label": ""})
|
||
return arms
|
||
|
||
|
||
def discover_runs(armdir):
|
||
runs = []
|
||
for f in os.listdir(armdir):
|
||
m = re.fullmatch(r"run(\d+)\.jsonl", f)
|
||
if m:
|
||
runs.append(int(m.group(1)))
|
||
return sorted(runs)
|
||
|
||
|
||
def read_lines(path):
|
||
try:
|
||
with open(path, errors="replace") as fh:
|
||
return fh.readlines()
|
||
except OSError:
|
||
return []
|
||
|
||
|
||
def parse_events(path):
|
||
evs = []
|
||
for line in read_lines(path):
|
||
line = line.strip()
|
||
if not line:
|
||
continue
|
||
try:
|
||
evs.append(json.loads(line))
|
||
except json.JSONDecodeError:
|
||
continue # partial line from an interrupted battle
|
||
return evs
|
||
|
||
|
||
def subject_counters(log_text):
|
||
m = re.search(
|
||
r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)"
|
||
r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)", log_text)
|
||
if not m:
|
||
return None
|
||
return {"fired": int(m.group(2)), "hits": int(m.group(3)),
|
||
"hits_taken": int(m.group(6))}
|
||
|
||
|
||
def parse_round_wins_by_name(log_text):
|
||
"""ModularBot's `firstPlaces` from the runner's final standings block."""
|
||
for m in re.finditer(
|
||
r"^\s*#\d+\s+(\S+)\s+totalScore=-?\d+\s+firstPlaces=(\d+)", log_text,
|
||
re.MULTILINE):
|
||
if m.group(1) == BOT_NAME:
|
||
return int(m.group(2))
|
||
return None
|
||
|
||
|
||
def parse_round_scores(path):
|
||
"""Per-round score delta per bot NAME from the cumulative *.results.json
|
||
lines. Used only to break ties (mutual kill / no death), never as the
|
||
primary attribution."""
|
||
try:
|
||
lines = json.load(open(path)).get("roundResults", [])
|
||
except (OSError, json.JSONDecodeError, AttributeError):
|
||
return {}
|
||
prev, out = {}, {}
|
||
for i, line in enumerate(lines, 1):
|
||
cur = {m.group(1): int(m.group(2))
|
||
for m in re.finditer(r"(\S+) rank=\d+ score=(-?\d+)", line)}
|
||
if not cur:
|
||
continue
|
||
out[i] = {name: sc - prev.get(name, 0) for name, sc in cur.items()}
|
||
prev = cur
|
||
return out
|
||
|
||
|
||
def attribute_owners(evs, counters):
|
||
"""Return (subject_id, other_id) using fire/hit counts, never powers."""
|
||
fires, hits, victim_hits = {}, {}, {}
|
||
for o in evs:
|
||
t = o.get("type")
|
||
if t == "fire":
|
||
fires[o["owner"]] = fires.get(o["owner"], 0) + 1
|
||
elif t == "hit":
|
||
hits[o["owner"]] = hits.get(o["owner"], 0) + 1
|
||
if "victim" in o:
|
||
victim_hits[o["victim"]] = victim_hits.get(o["victim"], 0) + 1
|
||
if not fires:
|
||
return None, None
|
||
if counters:
|
||
strict = [o for o in fires
|
||
if fires[o] == counters["fired"]
|
||
and hits.get(o, 0) == counters["hits"]
|
||
and victim_hits.get(o, 0) == counters["hits_taken"]]
|
||
if len(strict) == 1:
|
||
subj = strict[0]
|
||
return subj, _other(fires, subj)
|
||
# relaxed fallbacks (logged by caller via `counters is None` etc.)
|
||
cand = [o for o in fires if counters and fires[o] == counters["fired"]]
|
||
if len(cand) != 1:
|
||
cand = [o for o in fires
|
||
if counters and victim_hits.get(o, 0) == counters["hits_taken"]]
|
||
if len(cand) != 1:
|
||
cand = list(fires)
|
||
if len(cand) == 2:
|
||
return cand[0], cand[1]
|
||
if len(cand) == 1:
|
||
return cand[0], _other(fires, cand[0])
|
||
return None, None
|
||
|
||
|
||
def _other(owners, subj):
|
||
others = [o for o in owners if o != subj]
|
||
return others[0] if len(others) == 1 else None
|
||
|
||
|
||
def parse_run(armdir, run):
|
||
"""Everything the report needs for one run; None-free on partial data."""
|
||
evs = parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))
|
||
log_text = "".join(read_lines(os.path.join(armdir, f"run{run}.battle.log")))
|
||
counters = subject_counters(log_text)
|
||
subj, other = attribute_owners(evs, counters)
|
||
|
||
r = {"run": run, "rounds": 0, "wins": None, "first_places":
|
||
parse_round_wins_by_name(log_text), "subject_id": subj,
|
||
"other_id": other,
|
||
"attribution_exact": counters is not None, "shots": 0, "damage": 0.0,
|
||
"damage_taken": 0.0, "hits_taken": 0, "hits_dealt": 0,
|
||
"death_solved": 0, "death_score_agree": 0, "ambiguous": 0}
|
||
if subj is None or other is None:
|
||
# still count rounds so the arm shows up, but flags will be set
|
||
r["rounds"] = _round_count(armdir, run)
|
||
return r
|
||
|
||
deaths = {}
|
||
for o in evs:
|
||
t = o.get("type")
|
||
if t == "fire" and o.get("owner") == other:
|
||
r["shots"] += 1
|
||
elif t == "hit":
|
||
if o.get("owner") == other:
|
||
r["damage"] += o.get("damage", 0.0)
|
||
r["hits_dealt"] += 1
|
||
if o.get("victim") == other:
|
||
r["damage_taken"] += o.get("damage", 0.0)
|
||
r["hits_taken"] += 1
|
||
elif t == "death":
|
||
deaths.setdefault(o.get("round"), []).append(o.get("victim"))
|
||
|
||
nrounds = _round_count(armdir, run)
|
||
r["rounds"] = nrounds
|
||
scores = parse_round_scores(
|
||
os.path.join(armdir, f"run{run}.jsonl.results.json"))
|
||
# Per-round winner: primary = death events (a lone death means the OTHER bot
|
||
# won). A round with two or zero deaths (a mutual kill or a timeout) cannot
|
||
# be resolved from deaths alone, so the name-based per-round score delta is
|
||
# the tie-break. The two methods are cross-checked on the unambiguous rounds.
|
||
wins = death_solved = agree = ambiguous = 0
|
||
for rd in range(1, nrounds + 1):
|
||
vics = deaths.get(rd, [])
|
||
winner = None
|
||
if len(vics) == 1:
|
||
winner = other if vics[0] == subj else subj
|
||
death_solved += 1
|
||
delta = scores.get(rd)
|
||
if delta and SUBJECT_NAME in delta and BOT_NAME in delta:
|
||
score_winner = (subj if delta[SUBJECT_NAME] > delta[BOT_NAME]
|
||
else other)
|
||
if winner is None:
|
||
winner = score_winner
|
||
ambiguous += 1
|
||
elif winner == score_winner:
|
||
agree += 1
|
||
if winner == other:
|
||
wins += 1
|
||
r["wins"] = wins
|
||
r["death_solved"] = death_solved
|
||
r["death_score_agree"] = agree
|
||
r["ambiguous"] = ambiguous
|
||
return r
|
||
|
||
|
||
def _round_count(armdir, run):
|
||
p = os.path.join(armdir, f"run{run}.jsonl.rounds.json")
|
||
try:
|
||
return len(json.load(open(p)).get("rounds", []))
|
||
except (OSError, json.JSONDecodeError, AttributeError):
|
||
pass
|
||
# fall back to distinct round numbers in the events sidecar
|
||
rds = {o.get("round") for o in
|
||
parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))}
|
||
return len(rds)
|
||
|
||
|
||
# ── statistics ───────────────────────────────────────────────────────────────
|
||
|
||
def perm_test(xa, xb):
|
||
"""Exact two-sided permutation test on the difference of means. Returns
|
||
(obs, p, n_perm, exact_bool)."""
|
||
na, nb = len(xa), len(xb)
|
||
if na == 0 or nb == 0:
|
||
return None
|
||
obs = abs(sum(xa) / na - sum(xb) / nb)
|
||
pooled = list(xa) + list(xb)
|
||
n = na + nb
|
||
total_sum = sum(pooled)
|
||
ncomb = math.comb(n, na)
|
||
if ncomb <= EXACT_CAP:
|
||
cnt = 0
|
||
for combo in itertools.combinations(range(n), na):
|
||
sa = sum(pooled[i] for i in combo)
|
||
if abs(sa / na - (total_sum - sa) / nb) >= obs - 1e-9:
|
||
cnt += 1
|
||
return obs, cnt / ncomb, ncomb, True
|
||
# deterministic fallback for very large n (never hit at the default 7 runs)
|
||
import random
|
||
rng = random.Random(0xA1B2C3)
|
||
B = 200_000
|
||
cnt = 0
|
||
for _ in range(B):
|
||
idx = rng.sample(range(n), na)
|
||
sa = sum(pooled[i] for i in idx)
|
||
if abs(sa / na - (total_sum - sa) / nb) >= obs - 1e-9:
|
||
cnt += 1
|
||
return obs, (cnt + 1) / (B + 1), ncomb, False
|
||
|
||
|
||
def fisher_two_sided(a, b, c, d):
|
||
"""Two-sided Fisher exact on [[a,b],[c,d]]."""
|
||
n = a + b + c + d
|
||
if n == 0:
|
||
return None
|
||
row1, col1, col2 = a + b, a + c, b + d
|
||
denom = math.comb(n, row1)
|
||
if denom == 0:
|
||
return None
|
||
|
||
def hyper(x):
|
||
return math.comb(col1, x) * math.comb(col2, row1 - x) / denom
|
||
|
||
lo = max(0, col1 - (c + d))
|
||
hi = min(row1, col1)
|
||
obs = hyper(a)
|
||
return sum(hyper(x) for x in range(lo, hi + 1) if hyper(x) <= obs + 1e-12)
|
||
|
||
|
||
# ── liveness ─────────────────────────────────────────────────────────────────
|
||
|
||
def liveness(armdir, runs, env):
|
||
"""OK iff every declared env var appears verbatim in the bot's boot report
|
||
of every run, the report itself ran, and the bot did not warn that the name
|
||
is unrecognised (an unknown TR_*/GUN_* var is silently ignored)."""
|
||
if not runs:
|
||
return "FAIL", "no runs"
|
||
report_seen = {r: False for r in runs}
|
||
matched = {r: True for r in runs}
|
||
warned = {r: [] for r in runs}
|
||
for r in runs:
|
||
text = "".join(read_lines(os.path.join(
|
||
armdir, f"run{r}.bot.stdout.log")))
|
||
report_seen[r] = "=== ENVIRONMENT (boot report) ===" in text
|
||
for k, v in env.items():
|
||
if f"[env] {k}={v}" not in text:
|
||
matched[r] = False
|
||
if f"[env] WARNING: {k}" in text:
|
||
warned[r].append(k)
|
||
dead = [r for r in runs if not report_seen[r]]
|
||
miss = [r for r in runs if not matched[r]]
|
||
if dead:
|
||
return "FAIL", f"no bot env report in run(s) {dead}"
|
||
warn_runs = [r for r in runs if warned[r]]
|
||
if warn_runs:
|
||
names = sorted({k for r in warn_runs for k in warned[r]})
|
||
return "FAIL", (f"bot ignored arm env {names} (unrecognised) in "
|
||
f"run(s) {warn_runs}")
|
||
if env and miss:
|
||
vars_txt = " ".join(f"{k}={v}" for k, v in env.items())
|
||
return "FAIL", f"arm env '{vars_txt}' not in boot report of run(s) {miss}"
|
||
if not env:
|
||
return "OK", f"{len(runs)}/{len(runs)} runs: no arm env; report present"
|
||
vars_txt = " ".join(f"{k}={v}" for k, v in env.items())
|
||
return "OK", f"{len(runs)}/{len(runs)} runs: {vars_txt} applied"
|
||
|
||
|
||
# ── report ───────────────────────────────────────────────────────────────────
|
||
|
||
def main():
|
||
args = sys.argv[1:]
|
||
if not args or args[0] in ("-h", "--help"):
|
||
print(__doc__)
|
||
return 0 if args else 2
|
||
outdir = args[0]
|
||
reference = None
|
||
if "--reference" in args:
|
||
reference = args[args.index("--reference") + 1]
|
||
if not os.path.isdir(outdir):
|
||
print(f"ERROR: not a directory: {outdir}", file=sys.stderr)
|
||
return 2
|
||
|
||
session = load_session(outdir)
|
||
arms = discover_arms(outdir, session)
|
||
if not arms:
|
||
print(f"ERROR: no arms found in {outdir}", file=sys.stderr)
|
||
return 2
|
||
if reference is None:
|
||
reference = arms[0]["name"]
|
||
|
||
data = {}
|
||
for arm in arms:
|
||
armdir = os.path.join(outdir, arm["name"])
|
||
runs = discover_runs(armdir)
|
||
per = [parse_run(armdir, r) for r in runs]
|
||
data[arm["name"]] = {"dir": armdir, "runs": runs, "per": per,
|
||
"env": arm["env"], "label": arm["label"]}
|
||
|
||
if session:
|
||
print(f"# session {outdir}")
|
||
print(f"# commit={session.get('commit','?')} "
|
||
f"binary_sha256={session.get('binary_sha256','?')} "
|
||
f"rounds={session.get('rounds','?')} runs={session.get('runs','?')} "
|
||
f"conc={session.get('conc','?')} ts={session.get('timestamp','?')}")
|
||
print()
|
||
|
||
# ── summary table ────────────────────────────────────────────────────────
|
||
print("ARM SUMMARY")
|
||
hdr = (f"{'arm':<14} {'runs':>4} {'dmg/run':>8} {'dmgtk/run':>9} "
|
||
f"{'wins':>7} {'win%':>6} {'shots/run':>9} {'hitstk/run':>10}")
|
||
print(hdr)
|
||
print("-" * len(hdr))
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
per = data[name]["per"]
|
||
n = len(per)
|
||
if n == 0:
|
||
print(f"{name:<14} {0:>4} (no runs)")
|
||
continue
|
||
dmg = sum(p["damage"] for p in per) / n
|
||
dmgv = sum(p["damage_taken"] for p in per) / n
|
||
shots = sum(p["shots"] for p in per) / n
|
||
htk = sum(p["hits_taken"] for p in per) / n
|
||
wins = sum(p["wins"] for p in per if p["wins"] is not None)
|
||
rounds = sum(p["rounds"] for p in per)
|
||
pct = f"{100.0 * wins / rounds:.1f}" if rounds else "n/a"
|
||
print(f"{name:<14} {n:>4} {dmg:>8.0f} {dmgv:>9.0f} "
|
||
f"{str(wins) + '/' + str(rounds):>7} {pct:>6} {shots:>9.0f} {htk:>10.1f}")
|
||
|
||
# ── per-run values ───────────────────────────────────────────────────────
|
||
print("\nPER-RUN (never just the mean)")
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
per = data[name]["per"]
|
||
dmgs = " ".join(
|
||
f"r{p['run']}={p['damage']:.0f}" for p in per)
|
||
wins = " ".join(
|
||
f"r{p['run']}={p['wins']}/{p['rounds']}" if p["wins"] is not None
|
||
else f"r{p['run']}=?/{p['rounds']}" for p in per)
|
||
print(f" {name:<14} dmg: {dmgs}")
|
||
print(f" {'':<14} wins: {wins}")
|
||
|
||
# ── exact permutation tests ──────────────────────────────────────────────
|
||
if reference not in data:
|
||
print(f"\nWARNING: reference arm '{reference}' not found; skipping tests")
|
||
return 1
|
||
print(f"\nEXACT TWO-SIDED PERMUTATION TEST (per-run values) vs `{reference}`")
|
||
print(f"{'metric':<12} {'arm':<14} {'obs(diff)':>12} {'p':>8} permutations")
|
||
print("-" * 64)
|
||
ref = data[reference]["per"]
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
if name == reference:
|
||
continue
|
||
for label, key in (("dmg/run", "damage"), ("round wins", "wins")):
|
||
xa = [p[key] for p in ref if p[key] is not None]
|
||
xb = [p[key] for p in data[name]["per"] if p[key] is not None]
|
||
res = perm_test(xa, xb)
|
||
if res is None:
|
||
print(f"{label:<12} {name:<14} {'n/a':>12} {'n/a':>8}")
|
||
continue
|
||
obs, p, ncomb, exact = res
|
||
note = f"C({len(xa)+len(xb)},{len(xa)})={ncomb}" + (
|
||
"" if exact else " SAMPLED")
|
||
print(f"{label:<12} {name:<14} {obs:>+12.3f} {p:>8.4f} {note}")
|
||
|
||
# ── round-level test (anti-conservative) ─────────────────────────────────
|
||
print("\nROUND-LEVEL TEST (pooled rounds, Fisher exact) vs "
|
||
f"`{reference}` — ANTI-CONSERVATIVE: rounds cluster within runs")
|
||
print(f"{'arm':<14} {'ref wins':>10} {'arm wins':>10} {'p':>8}")
|
||
print("-" * 46)
|
||
refw = sum(p["wins"] for p in ref if p["wins"] is not None)
|
||
refr = sum(p["rounds"] for p in ref)
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
if name == reference:
|
||
continue
|
||
per = data[name]["per"]
|
||
w = sum(p["wins"] for p in per if p["wins"] is not None)
|
||
rr = sum(p["rounds"] for p in per)
|
||
p = fisher_two_sided(w, rr - w, refw, refr - refw)
|
||
pstr = f"{p:.4f}" if p is not None else "n/a"
|
||
print(f"{name:<14} {str(refw) + '/' + str(refr):>10} "
|
||
f"{str(w) + '/' + str(rr):>10} {pstr:>8}")
|
||
|
||
# ── liveness ─────────────────────────────────────────────────────────────
|
||
print("\nLIVENESS (arm env applied in the bot's own boot report)")
|
||
lv_fail = 0
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
status, why = liveness(data[name]["dir"], data[name]["runs"],
|
||
data[name]["env"])
|
||
if status == "FAIL":
|
||
lv_fail += 1
|
||
print(f" {name:<14} {status:<4} ({why})")
|
||
|
||
# ── round-win attribution cross-check ────────────────────────────────────
|
||
print("\nROUND-WIN ATTRIBUTION (events primary; score tie-break for "
|
||
"mutual-kill / timeout rounds)")
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
runs_ok = runs_tot = 0
|
||
dsolve = dagree = damb = 0
|
||
for p in data[name]["per"]:
|
||
if p["wins"] is not None and p["first_places"] is not None:
|
||
runs_tot += 1
|
||
if p["wins"] == p["first_places"]:
|
||
runs_ok += 1
|
||
dsolve += p["death_solved"]
|
||
dagree += p["death_score_agree"]
|
||
damb += p["ambiguous"]
|
||
status = ("OK" if runs_tot and runs_ok == runs_tot
|
||
else ("n/a" if not runs_tot else "MISMATCH"))
|
||
print(f" {name:<14} wins==firstPlaces {runs_ok}/{runs_tot} runs {status}; "
|
||
f"single-death rounds agree with score {dagree}/{dsolve} "
|
||
f"({damb} tie-broken)")
|
||
|
||
return 1 if lv_fail else 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|