Files
SirRoboGarage/tools/ab/ab_analyze.py
T

521 lines
20 KiB
Python
Executable File
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""ab_analyze.py — read a session dir produced by ab_run.sh and print the report.
python3 tools/ab/ab_analyze.py <session_dir> [--reference ARM]
Standard library only, deterministic. For every arm it prints runs, damage/run,
damage taken/run, ROUND WINS, shots/run, hits taken/run and the PER-RUN values
(wins cluster at 0/7 and single-run damage swings ~200, so the mean alone lies).
Tests:
* exact two-sided permutation test on PER-RUN values (damage/run and round
wins) vs a reference arm (default: the first arm). C(14,7)=3432 for 7v7 —
enumerated exactly, never sampled, whenever the combination count is small.
* a round-level Fisher exact test on pooled rounds, clearly labelled
anti-conservative (rounds cluster within runs).
Round-win attribution comes from the events sidecar (the bot that does NOT die
wins the round) and is cross-checked against the runner's `firstPlaces`, which
is name-based. `firstPlaces` is the run-level truth: the per-round lines in
*.results.json are CUMULATIVE standings, not round winners — do not use them.
Liveness: each arm's declared env vars must appear verbatim in the bot's own
boot environment report (<arm>/run<N>.bot.stdout.log, `[env] VAR=VALUE`), so an
arm whose setting never reached the process is a loud FAIL rather than a
plausible-looking number.
Owner attribution never relies on fired-power values (the power policy fires a
continuous 0.15–1.15 range). The subject (DrussGT) is identified by matching
the events sidecar's per-owner fire/hit counts against the capture's own
`subject event counts:` line; ModularBot is the other bot.
"""
import glob
import itertools
import json
import math
import os
import re
import sys
# beyond this many combinations we sample (deterministic seed) and say so;
# 7v7 = C(14,7) = 3432 is always exact.
EXACT_CAP = 20_000_000
BOT_NAME = "ModularBot"
SUBJECT_NAME = "DrussGT" # the capture subject (our bot is the adversary)
# ── loading ──────────────────────────────────────────────────────────────────
def load_session(outdir):
p = os.path.join(outdir, "session.json")
if os.path.exists(p):
try:
return json.load(open(p))
except (json.JSONDecodeError, OSError):
return None
return None
def parse_envspec(envspec):
"""`TR_A=1 TR_B=2` -> {'TR_A': '1', 'TR_B': '2'}."""
out = {}
for tok in (envspec or "").split():
if "=" in tok:
k, v = tok.split("=", 1)
out[k] = v
return out
def discover_arms(outdir, session):
arms = []
if session and isinstance(session.get("arms"), list):
for a in session["arms"]:
name = a.get("name")
if not name:
continue
arms.append({"name": name,
"env": parse_envspec(a.get("env", "")),
"label": a.get("label", "")})
if arms:
return arms
# fallback: any subdir holding run*.jsonl
for d in sorted(os.listdir(outdir)):
full = os.path.join(outdir, d)
if os.path.isdir(full) and discover_runs(full):
arms.append({"name": d, "env": {}, "label": ""})
return arms
def discover_runs(armdir):
runs = []
for f in os.listdir(armdir):
m = re.fullmatch(r"run(\d+)\.jsonl", f)
if m:
runs.append(int(m.group(1)))
return sorted(runs)
def read_lines(path):
try:
with open(path, errors="replace") as fh:
return fh.readlines()
except OSError:
return []
def parse_events(path):
evs = []
for line in read_lines(path):
line = line.strip()
if not line:
continue
try:
evs.append(json.loads(line))
except json.JSONDecodeError:
continue # partial line from an interrupted battle
return evs
def subject_counters(log_text):
m = re.search(
r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)"
r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)", log_text)
if not m:
return None
return {"fired": int(m.group(2)), "hits": int(m.group(3)),
"hits_taken": int(m.group(6))}
def parse_round_wins_by_name(log_text):
"""ModularBot's `firstPlaces` from the runner's final standings block."""
for m in re.finditer(
r"^\s*#\d+\s+(\S+)\s+totalScore=-?\d+\s+firstPlaces=(\d+)", log_text,
re.MULTILINE):
if m.group(1) == BOT_NAME:
return int(m.group(2))
return None
def parse_round_scores(path):
"""Per-round score delta per bot NAME from the cumulative *.results.json
lines. Used only to break ties (mutual kill / no death), never as the
primary attribution."""
try:
lines = json.load(open(path)).get("roundResults", [])
except (OSError, json.JSONDecodeError, AttributeError):
return {}
prev, out = {}, {}
for i, line in enumerate(lines, 1):
cur = {m.group(1): int(m.group(2))
for m in re.finditer(r"(\S+) rank=\d+ score=(-?\d+)", line)}
if not cur:
continue
out[i] = {name: sc - prev.get(name, 0) for name, sc in cur.items()}
prev = cur
return out
def attribute_owners(evs, counters):
"""Return (subject_id, other_id) using fire/hit counts, never powers."""
fires, hits, victim_hits = {}, {}, {}
for o in evs:
t = o.get("type")
if t == "fire":
fires[o["owner"]] = fires.get(o["owner"], 0) + 1
elif t == "hit":
hits[o["owner"]] = hits.get(o["owner"], 0) + 1
if "victim" in o:
victim_hits[o["victim"]] = victim_hits.get(o["victim"], 0) + 1
if not fires:
return None, None
if counters:
strict = [o for o in fires
if fires[o] == counters["fired"]
and hits.get(o, 0) == counters["hits"]
and victim_hits.get(o, 0) == counters["hits_taken"]]
if len(strict) == 1:
subj = strict[0]
return subj, _other(fires, subj)
# relaxed fallbacks (logged by caller via `counters is None` etc.)
cand = [o for o in fires if counters and fires[o] == counters["fired"]]
if len(cand) != 1:
cand = [o for o in fires
if counters and victim_hits.get(o, 0) == counters["hits_taken"]]
if len(cand) != 1:
cand = list(fires)
if len(cand) == 2:
return cand[0], cand[1]
if len(cand) == 1:
return cand[0], _other(fires, cand[0])
return None, None
def _other(owners, subj):
others = [o for o in owners if o != subj]
return others[0] if len(others) == 1 else None
def parse_run(armdir, run):
"""Everything the report needs for one run; None-free on partial data."""
evs = parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))
log_text = "".join(read_lines(os.path.join(armdir, f"run{run}.battle.log")))
counters = subject_counters(log_text)
subj, other = attribute_owners(evs, counters)
r = {"run": run, "rounds": 0, "wins": None, "first_places":
parse_round_wins_by_name(log_text), "subject_id": subj,
"other_id": other,
"attribution_exact": counters is not None, "shots": 0, "damage": 0.0,
"damage_taken": 0.0, "hits_taken": 0, "hits_dealt": 0,
"death_solved": 0, "death_score_agree": 0, "ambiguous": 0}
if subj is None or other is None:
# still count rounds so the arm shows up, but flags will be set
r["rounds"] = _round_count(armdir, run)
return r
deaths = {}
for o in evs:
t = o.get("type")
if t == "fire" and o.get("owner") == other:
r["shots"] += 1
elif t == "hit":
if o.get("owner") == other:
r["damage"] += o.get("damage", 0.0)
r["hits_dealt"] += 1
if o.get("victim") == other:
r["damage_taken"] += o.get("damage", 0.0)
r["hits_taken"] += 1
elif t == "death":
deaths.setdefault(o.get("round"), []).append(o.get("victim"))
nrounds = _round_count(armdir, run)
r["rounds"] = nrounds
scores = parse_round_scores(
os.path.join(armdir, f"run{run}.jsonl.results.json"))
# Per-round winner: primary = death events (a lone death means the OTHER bot
# won). A round with two or zero deaths (a mutual kill or a timeout) cannot
# be resolved from deaths alone, so the name-based per-round score delta is
# the tie-break. The two methods are cross-checked on the unambiguous rounds.
wins = death_solved = agree = ambiguous = 0
for rd in range(1, nrounds + 1):
vics = deaths.get(rd, [])
winner = None
if len(vics) == 1:
winner = other if vics[0] == subj else subj
death_solved += 1
delta = scores.get(rd)
if delta and SUBJECT_NAME in delta and BOT_NAME in delta:
score_winner = (subj if delta[SUBJECT_NAME] > delta[BOT_NAME]
else other)
if winner is None:
winner = score_winner
ambiguous += 1
elif winner == score_winner:
agree += 1
if winner == other:
wins += 1
r["wins"] = wins
r["death_solved"] = death_solved
r["death_score_agree"] = agree
r["ambiguous"] = ambiguous
return r
def _round_count(armdir, run):
p = os.path.join(armdir, f"run{run}.jsonl.rounds.json")
try:
return len(json.load(open(p)).get("rounds", []))
except (OSError, json.JSONDecodeError, AttributeError):
pass
# fall back to distinct round numbers in the events sidecar
rds = {o.get("round") for o in
parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))}
return len(rds)
# ── statistics ───────────────────────────────────────────────────────────────
def perm_test(xa, xb):
"""Exact two-sided permutation test on the difference of means. Returns
(obs, p, n_perm, exact_bool)."""
na, nb = len(xa), len(xb)
if na == 0 or nb == 0:
return None
obs = abs(sum(xa) / na - sum(xb) / nb)
pooled = list(xa) + list(xb)
n = na + nb
total_sum = sum(pooled)
ncomb = math.comb(n, na)
if ncomb <= EXACT_CAP:
cnt = 0
for combo in itertools.combinations(range(n), na):
sa = sum(pooled[i] for i in combo)
if abs(sa / na - (total_sum - sa) / nb) >= obs - 1e-9:
cnt += 1
return obs, cnt / ncomb, ncomb, True
# deterministic fallback for very large n (never hit at the default 7 runs)
import random
rng = random.Random(0xA1B2C3)
B = 200_000
cnt = 0
for _ in range(B):
idx = rng.sample(range(n), na)
sa = sum(pooled[i] for i in idx)
if abs(sa / na - (total_sum - sa) / nb) >= obs - 1e-9:
cnt += 1
return obs, (cnt + 1) / (B + 1), ncomb, False
def fisher_two_sided(a, b, c, d):
"""Two-sided Fisher exact on [[a,b],[c,d]]."""
n = a + b + c + d
if n == 0:
return None
row1, col1, col2 = a + b, a + c, b + d
denom = math.comb(n, row1)
if denom == 0:
return None
def hyper(x):
return math.comb(col1, x) * math.comb(col2, row1 - x) / denom
lo = max(0, col1 - (c + d))
hi = min(row1, col1)
obs = hyper(a)
return sum(hyper(x) for x in range(lo, hi + 1) if hyper(x) <= obs + 1e-12)
# ── liveness ─────────────────────────────────────────────────────────────────
def liveness(armdir, runs, env):
"""OK iff every declared env var appears verbatim in the bot's boot report
of every run, the report itself ran, and the bot did not warn that the name
is unrecognised (an unknown TR_*/GUN_* var is silently ignored)."""
if not runs:
return "FAIL", "no runs"
report_seen = {r: False for r in runs}
matched = {r: True for r in runs}
warned = {r: [] for r in runs}
for r in runs:
text = "".join(read_lines(os.path.join(
armdir, f"run{r}.bot.stdout.log")))
report_seen[r] = "=== ENVIRONMENT (boot report) ===" in text
for k, v in env.items():
if f"[env] {k}={v}" not in text:
matched[r] = False
if f"[env] WARNING: {k}" in text:
warned[r].append(k)
dead = [r for r in runs if not report_seen[r]]
miss = [r for r in runs if not matched[r]]
if dead:
return "FAIL", f"no bot env report in run(s) {dead}"
warn_runs = [r for r in runs if warned[r]]
if warn_runs:
names = sorted({k for r in warn_runs for k in warned[r]})
return "FAIL", (f"bot ignored arm env {names} (unrecognised) in "
f"run(s) {warn_runs}")
if env and miss:
vars_txt = " ".join(f"{k}={v}" for k, v in env.items())
return "FAIL", f"arm env '{vars_txt}' not in boot report of run(s) {miss}"
if not env:
return "OK", f"{len(runs)}/{len(runs)} runs: no arm env; report present"
vars_txt = " ".join(f"{k}={v}" for k, v in env.items())
return "OK", f"{len(runs)}/{len(runs)} runs: {vars_txt} applied"
# ── report ───────────────────────────────────────────────────────────────────
def main():
args = sys.argv[1:]
if not args or args[0] in ("-h", "--help"):
print(__doc__)
return 0 if args else 2
outdir = args[0]
reference = None
if "--reference" in args:
reference = args[args.index("--reference") + 1]
if not os.path.isdir(outdir):
print(f"ERROR: not a directory: {outdir}", file=sys.stderr)
return 2
session = load_session(outdir)
arms = discover_arms(outdir, session)
if not arms:
print(f"ERROR: no arms found in {outdir}", file=sys.stderr)
return 2
if reference is None:
reference = arms[0]["name"]
data = {}
for arm in arms:
armdir = os.path.join(outdir, arm["name"])
runs = discover_runs(armdir)
per = [parse_run(armdir, r) for r in runs]
data[arm["name"]] = {"dir": armdir, "runs": runs, "per": per,
"env": arm["env"], "label": arm["label"]}
if session:
print(f"# session {outdir}")
print(f"# commit={session.get('commit','?')} "
f"binary_sha256={session.get('binary_sha256','?')} "
f"rounds={session.get('rounds','?')} runs={session.get('runs','?')} "
f"conc={session.get('conc','?')} ts={session.get('timestamp','?')}")
print()
# ── summary table ────────────────────────────────────────────────────────
print("ARM SUMMARY")
hdr = (f"{'arm':<14} {'runs':>4} {'dmg/run':>8} {'dmgtk/run':>9} "
f"{'wins':>7} {'win%':>6} {'shots/run':>9} {'hitstk/run':>10}")
print(hdr)
print("-" * len(hdr))
for arm in arms:
name = arm["name"]
per = data[name]["per"]
n = len(per)
if n == 0:
print(f"{name:<14} {0:>4} (no runs)")
continue
dmg = sum(p["damage"] for p in per) / n
dmgv = sum(p["damage_taken"] for p in per) / n
shots = sum(p["shots"] for p in per) / n
htk = sum(p["hits_taken"] for p in per) / n
wins = sum(p["wins"] for p in per if p["wins"] is not None)
rounds = sum(p["rounds"] for p in per)
pct = f"{100.0 * wins / rounds:.1f}" if rounds else "n/a"
print(f"{name:<14} {n:>4} {dmg:>8.0f} {dmgv:>9.0f} "
f"{str(wins) + '/' + str(rounds):>7} {pct:>6} {shots:>9.0f} {htk:>10.1f}")
# ── per-run values ───────────────────────────────────────────────────────
print("\nPER-RUN (never just the mean)")
for arm in arms:
name = arm["name"]
per = data[name]["per"]
dmgs = " ".join(
f"r{p['run']}={p['damage']:.0f}" for p in per)
wins = " ".join(
f"r{p['run']}={p['wins']}/{p['rounds']}" if p["wins"] is not None
else f"r{p['run']}=?/{p['rounds']}" for p in per)
print(f" {name:<14} dmg: {dmgs}")
print(f" {'':<14} wins: {wins}")
# ── exact permutation tests ──────────────────────────────────────────────
if reference not in data:
print(f"\nWARNING: reference arm '{reference}' not found; skipping tests")
return 1
print(f"\nEXACT TWO-SIDED PERMUTATION TEST (per-run values) vs `{reference}`")
print(f"{'metric':<12} {'arm':<14} {'obs(diff)':>12} {'p':>8} permutations")
print("-" * 64)
ref = data[reference]["per"]
for arm in arms:
name = arm["name"]
if name == reference:
continue
for label, key in (("dmg/run", "damage"), ("round wins", "wins")):
xa = [p[key] for p in ref if p[key] is not None]
xb = [p[key] for p in data[name]["per"] if p[key] is not None]
res = perm_test(xa, xb)
if res is None:
print(f"{label:<12} {name:<14} {'n/a':>12} {'n/a':>8}")
continue
obs, p, ncomb, exact = res
note = f"C({len(xa)+len(xb)},{len(xa)})={ncomb}" + (
"" if exact else " SAMPLED")
print(f"{label:<12} {name:<14} {obs:>+12.3f} {p:>8.4f} {note}")
# ── round-level test (anti-conservative) ─────────────────────────────────
print("\nROUND-LEVEL TEST (pooled rounds, Fisher exact) vs "
f"`{reference}` — ANTI-CONSERVATIVE: rounds cluster within runs")
print(f"{'arm':<14} {'ref wins':>10} {'arm wins':>10} {'p':>8}")
print("-" * 46)
refw = sum(p["wins"] for p in ref if p["wins"] is not None)
refr = sum(p["rounds"] for p in ref)
for arm in arms:
name = arm["name"]
if name == reference:
continue
per = data[name]["per"]
w = sum(p["wins"] for p in per if p["wins"] is not None)
rr = sum(p["rounds"] for p in per)
p = fisher_two_sided(w, rr - w, refw, refr - refw)
pstr = f"{p:.4f}" if p is not None else "n/a"
print(f"{name:<14} {str(refw) + '/' + str(refr):>10} "
f"{str(w) + '/' + str(rr):>10} {pstr:>8}")
# ── liveness ─────────────────────────────────────────────────────────────
print("\nLIVENESS (arm env applied in the bot's own boot report)")
lv_fail = 0
for arm in arms:
name = arm["name"]
status, why = liveness(data[name]["dir"], data[name]["runs"],
data[name]["env"])
if status == "FAIL":
lv_fail += 1
print(f" {name:<14} {status:<4} ({why})")
# ── round-win attribution cross-check ────────────────────────────────────
print("\nROUND-WIN ATTRIBUTION (events primary; score tie-break for "
"mutual-kill / timeout rounds)")
for arm in arms:
name = arm["name"]
runs_ok = runs_tot = 0
dsolve = dagree = damb = 0
for p in data[name]["per"]:
if p["wins"] is not None and p["first_places"] is not None:
runs_tot += 1
if p["wins"] == p["first_places"]:
runs_ok += 1
dsolve += p["death_solved"]
dagree += p["death_score_agree"]
damb += p["ambiguous"]
status = ("OK" if runs_tot and runs_ok == runs_tot
else ("n/a" if not runs_tot else "MISMATCH"))
print(f" {name:<14} wins==firstPlaces {runs_ok}/{runs_tot} runs {status}; "
f"single-death rounds agree with score {dagree}/{dsolve} "
f"({damb} tie-broken)")
return 1 if lv_fail else 0
if __name__ == "__main__":
sys.exit(main())