#!/usr/bin/env python3 """Per-arm DrussGT dodge quality from an `ab_run.sh` session. The gun-mixing hypothesis is: admitting TMHorizon AND BitBrain makes DrussGT's surfgun dodge *worse*, because two guns that disagree on the same tick fire inconsistent bullets and poison its wave matching. The random 7-round win counts an A/B session produces cannot see that (a few rounds, huge variance); the shots can. This reads the SAME live captures `ab_analyze.py` uses, applies the SAME per-shot instrument as `common_libs/tests/analyze_drussgt_dodge_vs_power.py` (miss at arrival, miss per flight tick, fixed-window lateral displacement, and the bullet-free CONTROL window), and splits it PER ARM. It is a thin driver: it imports the validated instrument, only adds (a) the per-arm split, (b) an aim-offset / power dispersion reading of OUR OWN shooting (the liveness check for the mechanism), and (c) between-arm tests whose uncertainty is clustered by RUN (7 runs per arm -> C(14,7)=3432 exact). The power-neutral metrics (miss/tick, fixed/control-window lateral displacement) are the ones to trust; raw miss distance is dominated by our own aim error and by the flight window (see docs/drussgt_dodge_vs_power.md). Usage: python3 tools/ab/ab_dodge_analyze.py /tmp/ab/mix [--json out.json] [--reps 2000] """ from __future__ import annotations import argparse import collections import glob import itertools import json import math import os import random import re import statistics import sys HERE = os.path.dirname(os.path.abspath(__file__)) REPO = os.path.dirname(os.path.dirname(HERE)) sys.path.insert(0, os.path.join(REPO, "common_libs", "tests")) import analyze_drussgt_dodge_vs_power as dodge # noqa: E402 BAND_LABELS = dodge.BAND_LABELS # 0-100 .. 450+ STRAT_BANDS = ["200-300", "300-450", "450+"] # bands with enough shots for a contrast DODGE_METRICS = ["miss", "miss_per_tick", "lat_fixed_abs", "lat_ctrl_abs", "lat_disp_per_tick", "hit"] TEST_METRICS = ["miss", "miss_per_tick", "lat_fixed_abs", "lat_ctrl_abs", "hit"] class ArmRun(dodge.Run): """The validated instrument, plus the aim offset we fired at. `aim_off` = (fired direction - bearing to DrussGT at the fire tick), in degrees, wrapped to (-180, 180]. It is the gun's answer relative to the straight-at-target line: its dispersion across shots is how *inconsistent* our bullets are, which is exactly the property that would poison a surfgun. """ def _shot(self, ev, t0): s = super()._shot(ev, t0) if s is None: return None r0 = self.by_tick[t0] bearing = math.degrees(math.atan2(r0["ey"] - s["_y"], r0["ex"] - s["_x"])) s["aim_off"] = ((ev["dir"] - bearing + 180.0) % 360.0) - 180.0 return s # ── discovery ──────────────────────────────────────────────────────────────── def discover_arm(armdir): runs = [] for cap in sorted(glob.glob(os.path.join(armdir, "run*.jsonl"))): if cap.endswith(".events.jsonl"): continue ev = cap[:-len(".jsonl")] + ".events.jsonl" rj = cap + ".rounds.json" if os.path.exists(ev) and os.path.exists(rj): runs.append(ArmRun(cap, ev, rj)) return runs # ── shooting-variation (liveness for the mechanism) ────────────────────────── def iqr(xs): if len(xs) < 4: return float("nan") xs = sorted(xs) return dodge.pct(xs, 0.75) - dodge.pct(xs, 0.25) def shooting_variation(shots): pows = [s["power"] for s in shots] offs = [s["aim_off"] for s in shots] ranges = [s["range"] for s in shots] pwr_hist = collections.Counter(round(p, 2) for p in pows) off_hist = collections.Counter(round(o / 5.0) * 5 for o in offs) return { "power_mean": dodge.mean(pows), "power_sd": statistics.pstdev(pows) if len(pows) > 1 else 0.0, "power_iqr": iqr(pows), "power_distinct": len(set(pows)), "aimoff_mean": dodge.mean(offs), "aimoff_sd": statistics.pstdev(offs) if len(offs) > 1 else 0.0, "aimoff_iqr": iqr(offs), "range_mean": dodge.mean(ranges), "power_hist": dict(pwr_hist.most_common(10)), "aimoff_hist": dict(sorted(off_hist.items())), } # ── gun-selection liveness (does the mix actually alternate guns?) ──────────── CONFIG_GUN_RE = re.compile(r"gun=([A-Za-z]+)") ANSI_RE = re.compile(r"\x1b\[[0-9;]*m") def gun_liveness(runs): """From the `[config] gun=` lines the bot emits when its selection changes: how often each gun was selected and how many times the selection switched. A one-gun arm never switches; a real two-gun mix switches often. """ total = collections.Counter() switches = runs_with = 0 for r in runs: armdir = os.path.dirname(r.cap_path) num = os.path.basename(r.cap_path)[len("run"):-len(".jsonl")] path = os.path.join(armdir, "run%s.bot.stdout.log" % num) if not os.path.exists(path): continue seq = [] for line in open(path, errors="replace"): if "[config]" not in line: continue m = CONFIG_GUN_RE.search(ANSI_RE.sub("", line)) if m: seq.append(m.group(1)) if seq: runs_with += 1 total.update(seq) switches += sum(1 for i in range(1, len(seq)) if seq[i] != seq[i - 1]) return total, switches, runs_with, len(runs) # ── per-run summaries + clustered tests ────────────────────────────────────── def run_summaries(runs_shots, key): """Per run: (sum, count) for `key`, and per range band.""" out = [] for ss in runs_shots: vals = [s[key] for s in ss if s.get(key) is not None] bands = {b: [0.0, 0] for b in BAND_LABELS} for s in ss: v = s.get(key) if v is None: continue bands[s["band"]][0] += v bands[s["band"]][1] += 1 out.append({"sum": sum(vals), "n": len(vals), "bands": {b: tuple(v) for b, v in bands.items()}}) return out def cluster_ci(summary, reps=2000, seed=1): """95% CI of the pooled mean, resampling RUNS with replacement.""" rng = random.Random(seed) R = len(summary) if R == 0: return float("nan"), float("nan") means = [] for _ in range(reps): s = c = 0 for _ in range(R): it = summary[rng.randrange(R)] s += it["sum"] c += it["n"] if c: means.append(s / c) means.sort() return dodge.pct(means, 0.025), dodge.pct(means, 0.975) def _pooled(summary, idx): s = c = 0 for i in idx: s += summary[i]["sum"] c += summary[i]["n"] return s, c def _strat(summary, idx, rest, bands): num = den = 0.0 for b in bands: sa = ca = sb = cb = 0.0 for i in idx: sa += summary[i]["bands"][b][0] ca += summary[i]["bands"][b][1] for i in rest: sb += summary[i]["bands"][b][0] cb += summary[i]["bands"][b][1] w = ca + cb if ca > 0 and cb > 0: num += w * (sa / ca - sb / cb) den += w return num / den if den else float("nan") EXACT_CAP = 20_000_000 # 7v7 -> C(14,7)=3432 exact; 30v30 -> Monte-Carlo MC_DRAWS = 1_000_000 MC_SEED = 0x5EED5EED def cluster_perm(summary_a, summary_b, stat): """Two-sided run-cluster permutation test. `stat(summary, idxA, idxB)` is evaluated on the observed split and on relabellings of the pooled RUNS; the null is the difference itself, centred at 0, so p = P(|stat_perm| >= |stat_obs|). Clustering by run keeps the within-run shot correlation, which a shot-level test would ignore. Full enumeration when C(2R,R) <= EXACT_CAP (7v7 -> 3432, exact); otherwise a fixed-seed Monte-Carlo run (reported as such). """ summary = list(summary_a) + list(summary_b) n, na = len(summary), len(summary_a) obs = stat(summary, range(na), range(na, n)) ncomb = math.comb(n, na) if ncomb <= EXACT_CAP: cnt = 0 for combo in itertools.combinations(range(n), na): cset = set(combo) rest = [i for i in range(n) if i not in cset] v = stat(summary, combo, rest) if v == v and abs(v) >= abs(obs) - 1e-12: cnt += 1 return obs, (cnt + 1) / (ncomb + 1), "exact" rng = random.Random(MC_SEED) cnt = 0 for _ in range(MC_DRAWS): idx = rng.sample(range(n), na) cset = set(idx) rest = [i for i in range(n) if i not in cset] v = stat(summary, idx, rest) if v == v and abs(v) >= abs(obs) - 1e-12: cnt += 1 return obs, (cnt + 1) / (MC_DRAWS + 1), "monte-carlo/%d" % MC_DRAWS def pooled_stat(summary, idx, rest): sa, ca = _pooled(summary, idx) sb, cb = _pooled(summary, rest) if ca == 0 or cb == 0: return float("nan") return sa / ca - sb / cb def strat_stat(summary, idx, rest): return _strat(summary, idx, rest, STRAT_BANDS) # ── report ─────────────────────────────────────────────────────────────────── def main(): ap = argparse.ArgumentParser() ap.add_argument("outdir") ap.add_argument("--json", default=None) ap.add_argument("--reps", type=int, default=2000) ap.add_argument("--reference", default=None) a = ap.parse_args() session = json.load(open(os.path.join(a.outdir, "session.json"))) arm_names = [x["name"] for x in session["arms"]] ref = a.reference or arm_names[0] arms = {} for name in arm_names: runs = discover_arm(os.path.join(a.outdir, name)) runs_shots = [[s for s in r.shots()] for r in runs] shots = [s for ss in runs_shots for s in ss] rounds = sum(len(r.rounds) for r in runs) arms[name] = {"runs": runs, "runs_shots": runs_shots, "shots": shots, "rounds": rounds, "variation": shooting_variation(shots), "guns": gun_liveness(runs)} out = {"commit": session["commit"], "runs": session["runs"], "rounds": session["rounds"], "arms": {}} # ── corpus sanity: attribution is per battle, so restate the death check ── bad = tot = 0 for name in arm_names: for r in arms[name]["runs"]: for ev in r.events: if ev.get("type") != "death": continue end = r.start[ev["round"]] + r.count[ev["round"]] - 1 row = r.by_tick.get(end) if row is None: continue tot += 1 vs = r.owner_side.get(ev["victim"]) if (vs == "e" and row["ee"] > 1.0) or (vs == "s" and row["se"] > 1.0): bad += 1 print("=" * 100) print("PER-ARM DRUSSTGT DODGE QUALITY (session commit %s, %d runs x %d rounds)" % (session["commit"][:9], session["runs"], session["rounds"])) print("attribution cross-check: %d/%d deaths have the mapped victim at ~0 energy" % (tot - bad, tot)) print("shipped baseline vs DrussGT is ~49%% round wins; this instrument is per " "SHOT (~thousands/arm), not per round.") for name in arm_names: A = arms[name] v = A["variation"] print("\n--- ARM `%s` --- %d runs, %d rounds, %d shots" % (name, len(A["runs"]), A["rounds"], len(A["shots"]))) print(" OUR shooting: power mean=%.3f sd=%.3f iqr=%.3f distinct=%d | " "aim-offset sd=%.2f deg iqr=%.2f | fire range mean=%.0f px" % (v["power_mean"], v["power_sd"], v["power_iqr"], v["power_distinct"], v["aimoff_sd"], v["aimoff_iqr"], v["range_mean"])) print(" power histogram (top): " + " ".join("%.2f:%d" % (p, n) for p, n in v["power_hist"].items())) print(" aim-offset histogram (deg bucket): " + " ".join("%+d:%d" % (b, n) for b, n in v["aimoff_hist"].items())) gt, gsw, grw, gnr = A["guns"] print(" GUN SELECTION ([config] switch lines): %s | switches=%d over %d/%d runs" % (" ".join("%s:%d" % kv for kv in gt.most_common()), gsw, grw, gnr)) print("\n" + "=" * 100) print("DRUSSGT DODGE QUALITY PER ARM (mean, 95%% CI = run-cluster bootstrap)") hdr = (" %-6s %7s | %8s %16s | %10s %16s | %10s %16s | %8s" % ("arm", "shots", "miss", "miss/tick [95%CI]", "lat12", "lat12 [95%CI]", "lat_ctrl", "lat_ctrl [95%CI]", "hit%")) print(hdr) print(" " + "-" * (len(hdr) - 2)) for name in arm_names: A = arms[name] shots = A["shots"] d = A.setdefault("dodge", {}) for key in DODGE_METRICS: vals = [s[key] for s in shots if s.get(key) is not None] d[key] = dodge.mean(vals) ci_mt = cluster_ci(run_summaries(A["runs_shots"], "miss_per_tick"), a.reps) ci_lf = cluster_ci(run_summaries(A["runs_shots"], "lat_fixed_abs"), a.reps) ci_lc = cluster_ci(run_summaries(A["runs_shots"], "lat_ctrl_abs"), a.reps) print(" %-6s %7d | %8.1f %7.2f[%5.2f,%6.2f] | %10.1f [%5.1f,%6.1f] | " "%10.1f [%5.1f,%6.1f] | %8.3f" % ( name, len(shots), d["miss"], d["miss_per_tick"], ci_mt[0], ci_mt[1], d["lat_fixed_abs"], ci_lf[0], ci_lf[1], d["lat_ctrl_abs"], ci_lc[0], ci_lc[1], d["hit"])) gt, gsw, grw, gnr = A["guns"] out["arms"][name] = { "runs": len(A["runs"]), "rounds": A["rounds"], "shots": len(shots), "variation": A["variation"], "dodge": {k: d[k] for k in DODGE_METRICS}, "guns": {"counts": dict(gt), "switches": gsw, "runs_with": grw}, "ci": {"miss_per_tick": ci_mt, "lat_fixed_abs": ci_lf, "lat_ctrl_abs": ci_lc}, } # ── minimum detectable effect for the shot-level mechanism metrics ─────── print("\n" + "=" * 100) print("MINIMUM DETECTABLE EFFECT (run-cluster level, n=%d/arm, alpha=0.05 two-sided," " 80%% power; MDE = 2.8016*sd_perrun*sqrt(2/n))" % session["runs"]) print(" the shot counts are large but the RUNS are what set the between-arm " "uncertainty, so this is the honest floor") print(" %-16s %14s %14s %-22s" % ("metric", "sd(per-run)", "MDE(abs)", "MDE vs ref mean")) Z_ALPHA_POWER = 1.959963984540054 + 0.8416212335729143 # 2.8016 for label, key in (("miss", "miss"), ("miss/tick", "miss_per_tick"), ("lat12", "lat_fixed_abs"), ("lat_ctrl", "lat_ctrl_abs"), ("hit%", "hit")): summ = run_summaries(arms[ref]["runs_shots"], key) means = [it["sum"] / it["n"] for it in summ if it["n"]] if len(means) < 2: continue sd = statistics.stdev(means) mde = Z_ALPHA_POWER * sd * math.sqrt(2.0 / len(means)) rm = arms[ref]["dodge"] refmean = rm[key] * (100.0 if key == "hit" else 1.0) print(" %-16s %14.3f %14.3f %-22s" % ( label, sd, mde, "%.1f%% of %.3f" % (100 * mde / refmean, refmean) if refmean else "-")) # ── between-arm tests, clustered by run ─────────────────────────────────── print("\n" + "=" * 100) print("BETWEEN-ARM CONTRASTS (exact run-cluster permutation, C(14,7)=3432)") print(" pooled = pooled-shot difference A-B; strat = range-stratified " "(bands %s, n-weighted)" % ",".join(STRAT_BANDS)) print(" a mix that poisons DrussGT should show miss/tick, lat12 and lat_ctrl " "LOWER (worse dodging) than the best single gun") hdr2 = (" %-24s %-24s %9s %8s | %9s %8s" % ("metric", "A vs B", "pooled", "perm p", "strat", "perm p")) print(hdr2) print(" " + "-" * (len(hdr2) - 2)) compare_pairs = [("mix", n) for n in arm_names if n != "mix" and n != ref] compare_pairs += [(n, ref) for n in arm_names if n not in (ref,)] seen = set() out["contrasts"] = {} for A, B in compare_pairs: if (A, B) in seen: continue seen.add((A, B)) for key in TEST_METRICS: sa = run_summaries(arms[A]["runs_shots"], key) sb = run_summaries(arms[B]["runs_shots"], key) obs_p, p_p, meth = cluster_perm(sa, sb, pooled_stat) obs_s, p_s, _ = cluster_perm(sa, sb, strat_stat) print(" %-24s %-24s %+9.3f %8.4f | %+9.3f %8.4f [%s]" % (key, "%s - %s" % (A, B), obs_p, p_p, obs_s, p_s, meth)) out["contrasts"].setdefault("%s-%s" % (A, B), {})[key] = { "pooled": obs_p, "perm_p": p_p, "method": meth, "strat": obs_s, "strat_perm_p": p_s} print() if a.json: with open(a.json, "w") as f: json.dump(out, f, indent=1, sort_keys=True) print("JSON written to %s" % a.json) return out if __name__ == "__main__": main()