#!/usr/bin/env python3 """spinner_analyze.py — the SPINNER-CLAIM analyzer (j124, owner claim 1). python3 tools/ab/spinner_analyze.py [--reference pattern] [--report FILE] Reads a session produced by `tools/ab/tournament_run.sh` (layout: ///run.{battle.log,events.jsonl,jsonl,jsonl.rounds.json,bot.stdout.log}) and answers the owner's claim: "I never seen a gun that learns wall movement or circular movement like spinning bot so fast - like a gun made on purpose for those movements." Unlike the gun campaign's gauntlet_analyze.py (exactly 2 arms), this supports N arms and adds the CONVERGENCE layer the claim is really about: our gun's per-round hit rate on the first rounds vs the later rounds, per arm, and within-round (first-half vs second-half ticks) for the spinner opponents. Primary metrics stay damage/run and ROUND WINS. Hit rate (ours, per-round) is the SECONDARY mechanism metric, reported but never the verdict. MDE is printed for every paired test so a null is distinguishable from an under-powered null. Standard library only. Deterministic (fixed permutation seed). """ import itertools import json import math import os import random import re import statistics import sys sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import gauntlet_analyze as ga # noqa: E402 (reuse the proven parsers/stats) BOT_NAME = "ModularBot" EXACT_PAIR_CAP = 20 MC_DRAWS = 200_000 MC_SEED = 0x5EED5EED # ── parsing one run ────────────────────────────────────────────────────────── def read_round_starts(path): try: data = json.load(open(path)) except (OSError, json.JSONDecodeError): return {} return {r["round"]: r["startTick"] for r in data.get("rounds", [])} def parse_run(armdir, run, env_req=None): evs = ga.parse_events(os.path.join(armdir, f"run{run}.events.jsonl")) log_text = "".join(ga.read_lines(os.path.join(armdir, f"run{run}.battle.log"))) counters = ga.parse_counters(log_text) wins = ga.parse_first_places(log_text) nrounds = ga.round_count(os.path.join(armdir, f"run{run}.jsonl.rounds.json")) starts = read_round_starts(os.path.join(armdir, f"run{run}.jsonl.rounds.json")) r = {"run": run, "valid": False, "rounds": nrounds, "wins": wins, "damage": 0.0, "damage_taken": 0.0, "mb_fired": 0, "mb_hits": 0, "opp_fired": 0, "opp_hits": 0, "env_ok": True, "per_round": {}} if env_req: text = "".join(ga.read_lines(os.path.join(armdir, f"run{run}.bot.stdout.log"))) r["env_ok"] = all( re.search(re.escape(k) + r"\s*=\s*" + re.escape(v) + r"(\s|$)", text) is not None for k, v in env_req.items()) if wins is None or counters is None or nrounds == 0: return r subj, other = ga.attribute_subject(evs, counters) if subj is None: return r def slot(rnd): return r["per_round"].setdefault(rnd, { "mb_fired": 0, "mb_hits": 0, "mb_damage": 0.0, "opp_fired": 0, "opp_hits": 0, "opp_damage": 0.0}) for o in evs: rnd = o.get("round", 0) t = o.get("type") if t == "fire": if o.get("owner") == subj: r["mb_fired"] += 1; slot(rnd)["mb_fired"] += 1 elif o.get("owner") == other: r["opp_fired"] += 1; slot(rnd)["opp_fired"] += 1 elif t == "hit": d = float(o.get("damage", 0.0)) if o.get("owner") == subj: r["damage"] += d; r["mb_hits"] += 1 slot(rnd)["mb_damage"] += d; slot(rnd)["mb_hits"] += 1 if o.get("owner") == other: r["opp_hits"] += 1; slot(rnd)["opp_damage"] += d slot(rnd)["opp_hits"] += 1 if o.get("victim") == subj: r["damage_taken"] += d r["valid"] = True r["round_starts"] = starts return r def discover_runs(armdir): if not os.path.isdir(armdir): return [] return sorted(int(m.group(1)) for m in (re.fullmatch(r"run(\d+)\.jsonl", f) for f in os.listdir(armdir)) if m) def load_arm(opp_dir, arm, runs, env_req): valid, invalid = [], [] for run in range(1, runs + 1): r = parse_run(os.path.join(opp_dir, arm), run, env_req) (valid if (r["valid"] and r["env_ok"]) else invalid).append(r) return valid, invalid # ── aggregation ────────────────────────────────────────────────────────────── def sumruns(runs, key): return sum(r[key] for r in runs) def pooled(runs): n = len(runs) if n == 0: return None rounds = sumruns(runs, "rounds") mb_fired = sumruns(runs, "mb_fired") opp_fired = sumruns(runs, "opp_fired") return { "runs": n, "damage": sumruns(runs, "damage") / n, "damage_taken": sumruns(runs, "damage_taken") / n, "wins": sumruns(runs, "wins") / n, "win_rate": sumruns(runs, "wins") / rounds if rounds else float("nan"), "rounds": rounds, "hit_rate": 100.0 * sumruns(runs, "mb_hits") / mb_fired if mb_fired else float("nan"), "incoming_hit_rate": 100.0 * sumruns(runs, "opp_hits") / opp_fired if opp_fired else float("nan"), "mb_fired": mb_fired, "opp_fired": opp_fired, } def round_hit_rates(runs): """{round_index: (hits, fires)} pooled over runs.""" out = {} for r in runs: for rnd, d in r["per_round"].items(): h, f = out.get(rnd, (0, 0)) out[rnd] = (h + d["mb_hits"], f + d["mb_fired"]) return out # ── main ───────────────────────────────────────────────────────────────────── def main(): args = sys.argv[1:] if not args: print(__doc__) return 2 session_dir = args[0].rstrip("/") ref = "pattern" if "--reference" in args: ref = args[args.index("--reference") + 1] report_path = None if "--report" in args: report_path = args[args.index("--report") + 1] session = json.load(open(os.path.join(session_dir, "session.json"))) arms = [a["name"] for a in session["arms"]] arm_env = {a["name"]: ga.parse_envspec(a.get("env", "")) for a in session["arms"]} if ref not in arms: ref = arms[0] opps = [o["name"] for o in session["opponents"]] style_of = {o["name"]: (o.get("style") or "other").strip() or "other" for o in session["opponents"]} lines = [] def out(s=""): lines.append(s) print(s) out("### MEASURED: session") out() out(f"* commit `{session['commit']}`, frozen binary sha256 " f"`{session['binary_sha256'][:12]}…`") out(f"* {len(opps)} opponents × {len(arms)} arms × {session['runs']} runs × " f"{session['rounds']} rounds = {len(opps) * len(arms) * session['runs']} battles") out(f"* arms file `{os.path.basename(session.get('arms_file', '?'))}`, " f"panel file `{os.path.basename(session.get('panel_file', '?'))}`") out(f"* reference arm: **`{ref}`** (deltas are arm − reference, opponent by opponent)") out() data = {} invalid_total = 0 for o in opps: data[o] = {} for a in arms: valid, invalid = load_arm(os.path.join(session_dir, o), a, session["runs"], arm_env.get(a)) data[o][a] = valid invalid_total += len(invalid) out(f"* liveness: {invalid_total} run(s) excluded " f"({len(opps) * len(arms) * session['runs']} total)") out() # ── spinner movement validation (MEASURED from the opponent trace) ────── out("### MEASURED: the spinner fixture is a true constant-turn spinner") out() out("Per-tick body turn and speed of each spinner opponent, read from the") out("capture's adversary trace (`s*` fields), pooled over its runs.") out() out("| opponent | ticks | turn mode (deg/tick) | mode share | turn SD | speed mode | speed mode share | wall-hug frac |") out("|---|---:|---:|---:|---:|---:|---:|---:|") spinner_opps = [o for o in opps if style_of[o] == "spinner"] for o in spinner_opps: adir = os.path.join(session_dir, o, ref) turns, speeds, wall, n, prev = [], [], 0, 0, None for run in discover_runs(adir): try: fh = open(os.path.join(adir, f"run{run}.jsonl"), errors="replace") except OSError: continue with fh: for line in fh: if '"sx"' not in line: continue try: e = json.loads(line) except json.JSONDecodeError: continue if "sx" not in e: continue x, y, h, s = e["sx"], e["sy"], e["sh"], e.get("ss", 0.0) n += 1 if min(x, 800 - x, y, 600 - y) < 50: wall += 1 if prev is not None: px, py, ph = prev if math.hypot(x - px, y - py) < 30: turns.append((h - ph + 180.0) % 360.0 - 180.0) speeds.append(s) prev = (x, y, h) if not turns: out(f"| {o} | 0 | n/a | n/a | n/a | n/a | n/a | n/a |") continue tc = {} for t in turns: tc[round(t, 1)] = tc.get(round(t, 1), 0) + 1 tmode, tcnt = max(tc.items(), key=lambda kv: kv[1]) sc = {} for s in speeds: sc[round(s, 1)] = sc.get(round(s, 1), 0) + 1 smode, scnt = max(sc.items(), key=lambda kv: kv[1]) sd = statistics.pstdev(turns) out(f"| {o} | {n} | {tmode} | {100 * tcnt / len(turns):.1f}% | {sd:.3f} " f"| {smode} | {100 * scnt / len(speeds):.1f}% | {100 * wall / n:.1f}% |") out() # ── per-opponent paired table ─────────────────────────────────────────── per_opp = {a: {} for a in arms} for o in opps: ref_runs = data[o][ref] if not ref_runs: for a in arms: per_opp[a][o] = None continue rm = pooled(ref_runs) for a in arms: m = pooled(data[o][a]) if m is None: per_opp[a][o] = None continue per_opp[a][o] = { "m": m, "ref": rm, "d_damage": m["damage"] - rm["damage"], "d_wins": m["wins"] - rm["wins"], "d_hit": m["hit_rate"] - rm["hit_rate"], "d_taken": m["damage_taken"] - rm["damage_taken"], } out("### MEASURED: per-opponent paired table") out() for a in arms: if a == ref: continue out(f"#### `{a}` vs `{ref}`") out() out("| opponent | style | dmg/run ref→arm | Δdmg | wins/run ref→arm | Δwins | our hit% ref→arm | Δhit (pp) | dmg taken Δ |") out("|---|---|---:|---:|---:|---:|---:|---:|---:|") for o in opps: e = per_opp[a][o] if e is None: out(f"| {o} | {style_of[o]} | n/a | n/a | n/a | n/a | n/a | n/a | n/a |") continue m, rm = e["m"], e["ref"] out(f"| {o} | {style_of[o]} | {rm['damage']:.1f}→{m['damage']:.1f} " f"| {e['d_damage']:+.1f} | {rm['wins']:.2f}→{m['wins']:.2f} " f"| {e['d_wins']:+.2f} | {rm['hit_rate']:.2f}→{m['hit_rate']:.2f} " f"| {e['d_hit']:+.2f} | {e['d_taken']:+.1f} |") out() # ── pooled dashboard ──────────────────────────────────────────────────── out("### MEASURED: pooled dashboard (all valid runs; NOT the verdict)") out() out("| arm | runs | dmg/run | dmg taken/run | wins/run | round wins | win rate | our hit rate | incoming hit rate |") out("|---|---:|---:|---:|---:|---:|---:|---:|---:|") pooled_arm = {} for a in arms: runs = [r for o in opps for r in data[o][a]] m = pooled(runs) pooled_arm[a] = m out(f"| `{a}` | {m['runs']} | {m['damage']:.1f} | {m['damage_taken']:.1f} " f"| {m['wins']:.2f} | {int(sumruns(runs, 'wins'))}/{m['rounds']} " f"| {100 * m['win_rate']:.1f}% | {m['hit_rate']:.2f}% " f"| {m['incoming_hit_rate']:.2f}% |") out() # ── cross-opponent aggregation ────────────────────────────────────────── out("### MEASURED: cross-opponent aggregation (the verdict layer)") out() out("| arm | metric | mean Δ | spread (SD) | 95% CI | sign test | p(sign) | p(sign-flip) | p(Wilcoxon) | MDE |") out("|---|---|---:|---:|---|---:|---:|---:|---:|---:|") stats = {} for a in arms: if a == ref: continue st = {} for key, mkey in (("d_damage", "damage"), ("d_wins", "wins"), ("d_hit", "our hit rate (pp)")): ds = [per_opp[a][o][key] for o in opps if per_opp[a][o] and not math.isnan(per_opp[a][o][key])] if len(ds) < 2: continue desc = ga.describe(ds) pos, neg, ties, p_sign = ga.sign_test(ds) sf = ga.signflip_perm(ds) pw = wilcoxon_signed(ds) st[key] = {"desc": desc, "pos": pos, "neg": neg, "ties": ties, "p_sign": p_sign, "p_flip": sf["p"], "p_wilcox": pw} out(f"| `{a}` | {mkey} | {desc['mean']:+.2f} | {desc['sd']:.2f} " f"| [{desc['mean'] - 1.96 * desc['se']:+.2f}, " f"{desc['mean'] + 1.96 * desc['se']:+.2f}] " f"| {pos}/{pos + neg} | {p_sign:.4g} | {sf['p']:.4g} " f"| {pw:.4g} | {desc['mde']:.2f} |") stats[a] = st out() # ── style split ───────────────────────────────────────────────────────── styles = sorted({style_of[o] for o in opps}) out("### MEASURED: by style (explanation only)") out() out("| arm | style | n | mean Δdmg | mean Δwins | mean Δour-hit (pp) |") out("|---|---|---:|---:|---:|---:|") for a in arms: if a == ref: continue for s in styles: es = [per_opp[a][o] for o in opps if per_opp[a][o] and style_of[o] == s] if not es: continue out(f"| `{a}` | {s} | {len(es)} " f"| {statistics.mean([e['d_damage'] for e in es]):+.2f} " f"| {statistics.mean([e['d_wins'] for e in es]):+.2f} " f"| {statistics.mean([e['d_hit'] for e in es]):+.2f} |") out() # ── CONVERGENCE ───────────────────────────────────────────────────────── out("### MEASURED: CONVERGENCE — our per-round hit rate") out() out("Pooled over every valid run (sum of hits / sum of shots, per round index).") out() maxr = session["rounds"] out("| arm | " + " | ".join(f"R{i}" for i in range(1, maxr + 1)) + " | R1→Rlast (pp) |") out("|---|" + "---:|" * (maxr + 1)) conv_all = {} for a in arms: runs = [r for o in opps for r in data[o][a]] rhs = round_hit_rates(runs) vals = [] for i in range(1, maxr + 1): h, f = rhs.get(i, (0, 0)) vals.append(100.0 * h / f if f else float("nan")) conv_all[a] = vals delta = (vals[-1] - vals[0]) if not (math.isnan(vals[0]) or math.isnan(vals[-1])) else float("nan") out(f"| `{a}` | " + " | ".join("n/a" if math.isnan(v) else f"{v:.1f}%" for v in vals) + f" | {delta:+.1f} |") out() out("Same, restricted to the TRUE-SPINNER opponents " f"({', '.join(spinner_opps)}):") out() out("| arm | " + " | ".join(f"R{i}" for i in range(1, maxr + 1)) + " | R1→Rlast (pp) |") out("|---|" + "---:|" * (maxr + 1)) conv_spin = {} for a in arms: runs = [r for o in spinner_opps for r in data[o][a]] rhs = round_hit_rates(runs) vals = [] for i in range(1, maxr + 1): h, f = rhs.get(i, (0, 0)) vals.append(100.0 * h / f if f else float("nan")) conv_spin[a] = vals delta = (vals[-1] - vals[0]) if not (math.isnan(vals[0]) or math.isnan(vals[-1])) else float("nan") out(f"| `{a}` | " + " | ".join("n/a" if math.isnan(v) else f"{v:.1f}%" for v in vals) + f" | {delta:+.1f} |") out() # within-round early vs late (fast adaptation), spinner opponents out("Within-round adaptation on the true spinners: our hit rate over the") out("first vs second half of each round (round-relative tick from the") out("rounds sidecar), pooled over runs.") out() out("| arm | first-half hit% | second-half hit% | late−early (pp) |") out("|---|---:|---:|---:|") for a in arms: early = [0, 0] late = [0, 0] for o in spinner_opps: adir = os.path.join(session_dir, o, a) for run in discover_runs(adir): evs = ga.parse_events(os.path.join(adir, f"run{run}.events.jsonl")) rpath = os.path.join(adir, f"run{run}.jsonl.rounds.json") starts = read_round_starts(rpath) log_text = "".join(ga.read_lines(os.path.join(adir, f"run{run}.battle.log"))) counters = ga.parse_counters(log_text) if counters is None or not starts: continue subj, other = ga.attribute_subject(evs, counters) if subj is None: continue counts = {rr["round"]: rr["count"] for rr in json.load(open(rpath))["rounds"]} for e in evs: rnd = e.get("round", 0) st = starts.get(rnd) if st is None: continue rel = e.get("tick", 0) - st bucket = early if rel < counts.get(rnd, 0) / 2 else late if e.get("type") == "fire" and e.get("owner") == subj: bucket[1] += 1 elif e.get("type") == "hit" and e.get("owner") == subj: bucket[0] += 1 if early[1] and late[1]: eh = 100.0 * early[0] / early[1] lh = 100.0 * late[0] / late[1] out(f"| `{a}` | {eh:.2f}% | {lh:.2f}% | {lh - eh:+.2f} |") else: out(f"| `{a}` | n/a | n/a | n/a |") out() # ── verdict ───────────────────────────────────────────────────────────── out("### The pre-registered reading") out() out("PRIMARY = dmg/run and round wins. Our per-round hit rate is the") out("SECONDARY mechanism metric. An arm is BETTER only if a primary improves") out("with sign-test p<0.05 while the other primary does not go down.") out() out("| arm | Δwins/run | Δdmg/run | Δour-hit (pp) | sign test dmg | verdict |") out("|---|---:|---:|---:|---|---|") for a in arms: if a == ref: continue st = stats.get(a, {}) d = st.get("d_damage") w = st.get("d_wins") h = st.get("d_hit") if not d or not w: out(f"| `{a}` | n/a | n/a | n/a | n/a | n/a |") continue up_d = d["desc"]["mean"] > 0 and d["p_sign"] < 0.05 up_w = w["desc"]["mean"] > 0 and w["p_sign"] < 0.05 if (up_d and w["desc"]["mean"] >= 0) or (up_w and d["desc"]["mean"] >= 0): verdict = "**BETTER**" elif d["desc"]["mean"] < -d["desc"]["mde"] or w["desc"]["mean"] < -w["desc"]["mde"]: verdict = "WORSE" else: verdict = "not distinguishable" out(f"| `{a}` | {w['desc']['mean']:+.2f} | {d['desc']['mean']:+.2f} " f"| {h['desc']['mean']:+.2f} | {d['pos']}/{d['pos'] + d['neg']} " f"p={d['p_sign']:.4g} | {verdict} |") out() if report_path: with open(report_path, "w") as fh: fh.write("\n".join(lines) + "\n") return 0 def wilcoxon_signed(deltas): """Two-sided Wilcoxon signed-rank normal approximation with tie correction.""" nz = [d for d in deltas if d != 0.0] n = len(nz) if n < 3: return 1.0 order = sorted(range(n), key=lambda i: abs(nz[i])) ranks = [0.0] * n i = 0 while i < n: j = i while j + 1 < n and abs(nz[order[j + 1]]) == abs(nz[order[i]]): j += 1 avg = (i + j) / 2.0 + 1.0 for k in range(i, j + 1): ranks[order[k]] = avg i = j + 1 w_plus = sum(ranks[i] for i in range(n) if nz[i] > 0) mu = n * (n + 1) / 4.0 cnt = {} for d in nz: cnt[abs(d)] = cnt.get(abs(d), 0) + 1 tie = sum(c ** 3 - c for c in cnt.values()) sigma2 = n * (n + 1) * (2 * n + 1) / 24.0 - tie / 48.0 if sigma2 <= 0: return 1.0 z = (w_plus - mu - 0.5 * (1 if w_plus > mu else -1)) / math.sqrt(sigma2) return min(1.0, 2.0 * 0.5 * math.erfc(abs(z) / math.sqrt(2.0))) if __name__ == "__main__": sys.exit(main())