#!/usr/bin/env python3 """tournament_analyze.py — paired multi-opponent ranking of movement arms. python3 tools/ab/tournament_analyze.py [--reference ARM] Reads a session produced by `tournament_run.sh` (layout: ///run.{battle.log,events.jsonl,jsonl,bot.stdout.log}) and answers the only question the movement campaign asks: DOES ARM A MOVE BETTER THAN THE REFERENCE ARM, ACROSS OPPONENTS? Method, in one paragraph. Every arm fights the SAME panel with the SAME frozen binary; for each opponent the arm's metric is averaged over its runs and subtracted from the reference arm's average for that same opponent. Those per-opponent deltas are the unit of evidence: the MEAN of the deltas is the effect, the SPREAD of the deltas across opponents is the honest error bar (one weird opponent cannot carry it), and a SIGN TEST over the deltas says how many opponents the arm actually wins. A pooled number over all runs is also printed, but it is reported as the descriptive dashboard, never as the verdict. CLI judgment (pre-registered in docs/movement_campaign.md): the primary metrics are damage/run and ROUND WINS. An arm is BETTER than the reference only if one of the two improves with the sign test at p<0.05 while the other does not degrade; hit rate and distance are explanation, never the verdict. MDE (alpha=0.05 two-sided, 80% power) is printed for every test so a null can be told apart from an under-powered null. Standard library only. Deterministic: the exact sign-flip test is enumerated when n_opponents <= 20, otherwise sampled with a fixed seed. """ import json import math import os import random import re import sys BOT_NAME = "ModularBot" # The metric keys, in report order. METRICS = ["damage", "damage_taken", "wins", "hit_rate", "dist"] # z_{0.975} + z_{0.80}: the constant in MDE = C * sd * sqrt(2/n) is for a # two-SAMPLE design; for the paired per-opponent design used here the analogous # constant with n = number of opponents is C * sd(deltas) / sqrt(n). MDE_C = 1.959963984540054 + 0.8416212335729143 EXACT_SIGNCAP = 20 # 2^20 = 1M sign vectors is still instant MC_DRAWS = 200_000 MC_SEED = 0x5EED5EED T975 = {1: 12.706, 2: 4.303, 3: 3.182, 4: 2.776, 5: 2.571, 6: 2.447, 7: 2.365, 8: 2.306, 9: 2.262, 10: 2.228, 11: 2.201, 12: 2.179, 13: 2.160, 14: 2.145, 15: 2.131, 16: 2.120, 17: 2.110, 18: 2.101, 19: 2.093, 20: 2.086, 21: 2.080, 22: 2.074, 23: 2.069, 24: 2.064, 25: 2.060, 26: 2.056, 27: 2.052, 28: 2.048, 29: 2.045, 30: 2.042} # ── small statistics helpers ───────────────────────────────────────────────── def mean(xs): return sum(xs) / len(xs) if xs else float("nan") def sd(xs): """Sample standard deviation (n-1). 0.0 for n<2.""" n = len(xs) if n < 2: return 0.0 m = mean(xs) return math.sqrt(sum((x - m) ** 2 for x in xs) / (n - 1)) def median(xs): s = sorted(xs) n = len(s) if n == 0: return float("nan") return s[n // 2] if n % 2 else 0.5 * (s[n // 2 - 1] + s[n // 2]) def binom_two_sided(k, n): """Exact two-sided sign-test p-value (p=0.5), ties already removed.""" if n == 0: return 1.0 def c(nn, kk): return math.comb(nn, kk) tail = sum(c(n, i) for i in range(0, min(k, n - k) + 1)) / 2 ** n return min(1.0, 2.0 * tail) def signflip_p(deltas): """Two-sided sign-flip permutation test on the MEAN of the deltas. Exact (all 2^n sign vectors) for n <= EXACT_SIGNCAP; otherwise a deterministic Monte-Carlo draw. Returns (p, method_string).""" n = len(deltas) if n == 0: return 1.0, "n/a" obs = abs(mean(deltas)) if obs == 0.0: return 1.0, "exact (degenerate)" tol = 1e-12 if n <= EXACT_SIGNCAP: total = 1 << n hits = 0 for mask in range(total): s = 0.0 for i, d in enumerate(deltas): s += -d if (mask >> i) & 1 else d if abs(s / n) >= obs - tol: hits += 1 return hits / total, f"exact 2^{n}" rng = random.Random(MC_SEED) hits = 0 for _ in range(MC_DRAWS): s = 0.0 for d in deltas: s += -d if rng.getrandbits(1) else d if abs(s / n) >= obs - tol: hits += 1 return hits / MC_DRAWS, f"MC {MC_DRAWS}" def wilcoxon_p(deltas): """Two-sided Wilcoxon signed-rank, normal approximation with tie correction. Returns (p, W). p=1.0 when there is nothing to test.""" nz = [d for d in deltas if d != 0.0] n = len(nz) if n < 3: return 1.0, float("nan") order = sorted(range(n), key=lambda i: abs(nz[i])) ranks = [0.0] * n i = 0 while i < n: j = i while j + 1 < n and abs(nz[order[j + 1]]) == abs(nz[order[i]]): j += 1 avg = (i + j) / 2.0 + 1.0 for k in range(i, j + 1): ranks[order[k]] = avg i = j + 1 w_plus = sum(ranks[i] for i in range(n) if nz[i] > 0) mu = n * (n + 1) / 4.0 # tie correction for sigma from collections import Counter cnt = Counter(abs(d) for d in nz) tie = sum(c ** 3 - c for c in cnt.values()) sigma2 = n * (n + 1) * (2 * n + 1) / 24.0 - tie / 48.0 if sigma2 <= 0: return 1.0, w_plus z = (w_plus - mu - 0.5 * (1 if w_plus > mu else -1)) / math.sqrt(sigma2) p = 2.0 * 0.5 * math.erfc(abs(z) / math.sqrt(2.0)) return min(1.0, p), w_plus def t_crit(df): return T975.get(df, 1.96) def mde(deltas): """Minimum detectable effect for the paired per-opponent design at alpha=0.05 (two-sided), power 80%, from the observed spread of the deltas.""" n = len(deltas) if n < 2: return float("nan") return MDE_C * sd(deltas) / math.sqrt(n) # ── parsing ────────────────────────────────────────────────────────────────── COUNTERS_RE = re.compile( r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)" r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)") FIRSTPLACES_RE = re.compile( r"^\s*#\d+\s+(\S+)\s+totalScore=(-?\d+)\s+firstPlaces=(\d+)\s+survival=(\d+)", re.MULTILINE) DIST_RE = re.compile(r"DISTANCE: mean=([\d.]+)") ROWS_RE = re.compile(r"rows=(\d+)") def read_text(path): try: with open(path, "r", errors="replace") as fh: return fh.read() except OSError: return "" def parse_events(path): out = [] try: with open(path, "r", errors="replace") as fh: for line in fh: line = line.strip() if not line: continue try: out.append(json.loads(line)) except json.JSONDecodeError: continue # a partial line from a killed battle except OSError: out = [] return out def parse_counters(text): m = COUNTERS_RE.search(text) if not m: return None return {"scans": int(m.group(1)), "fired": int(m.group(2)), "hits": int(m.group(3)), "misses": int(m.group(4)), "hit_bullets": int(m.group(5)), "hits_taken": int(m.group(6))} def parse_first_places(text): for m in FIRSTPLACES_RE.finditer(text): if m.group(1) == BOT_NAME: return int(m.group(3)) return None def parse_rounds(path): try: with open(path) as fh: return len(json.load(fh).get("rounds", [])) except (OSError, json.JSONDecodeError): return None def attribute_subject(evs, counters): """Which event `owner` id is our bot? Matched on fire/hit/hits-taken counts, never on fired power (the power policy is continuous). Returns (subject_id, other_id, how). `subject_id` is None only when nothing could be identified; `other_id` is None when the opponent never fired a single bullet (then the incoming hit rate is simply undefined, and the run is still a valid damage measurement).""" fires, hits, victim_hits = {}, {}, {} owner_ids = set() for ev in evs: t = ev.get("type") o = ev.get("owner") if o is not None: owner_ids.add(o) if t == "fire": fires[o] = fires.get(o, 0) + 1 elif t == "hit": hits[o] = hits.get(o, 0) + 1 v = ev.get("victim") if v is not None: victim_hits[v] = victim_hits.get(v, 0) + 1 if not fires: return None, None, "no fire events" def other_of(subj): others = [o for o in owner_ids if o != subj] return others[0] if len(others) == 1 else None if counters: strict = [o for o in fires if fires[o] == counters["fired"] and hits.get(o, 0) == counters["hits"] and victim_hits.get(o, 0) == counters["hits_taken"]] if len(strict) == 1: return strict[0], other_of(strict[0]), "exact" cand = [o for o in fires if counters and fires[o] == counters["fired"]] if len(cand) != 1: cand = [o for o in fires if counters and victim_hits.get(o, 0) == counters["hits_taken"]] if len(cand) != 1: cand = list(fires) if len(cand) == 1: return cand[0], other_of(cand[0]), "fires-only" return None, None, "ambiguous" def liveness(arm_env_tokens, stdout_path): """Liveness: every declared env token must appear verbatim in OUR bot's own boot environment report, so an arm whose setting never reached the process is a loud FAIL instead of a plausible-looking number. A TR_MOVEMENT the arm did NOT declare is fatal too (the baseline would be contaminated).""" text = read_text(stdout_path) if not text: return False, "no boot env report", False missing = [t for t in arm_env_tokens if t not in text] declared_keys = {t.split("=", 1)[0] for t in arm_env_tokens} leaked_movement = ("TR_MOVEMENT" not in declared_keys and re.search(r"^\[env\] TR_MOVEMENT=", text, re.MULTILINE) is not None) ok = (not missing) and (not leaked_movement) why = [] if missing: why.append("not seen in bot env report: " + ", ".join(missing)) if leaked_movement: why.append("undeclared TR_MOVEMENT leaked into the process") return ok, ("; ".join(why) if why else "ok"), bool(leaked_movement) def parse_run(opp_dir, arm_dir, run): """One run -> dict of MEASURED numbers, or ok=False with a reason.""" base = os.path.join(opp_dir, arm_dir) log_path = os.path.join(base, f"run{run}.battle.log") ev_path = os.path.join(base, f"run{run}.events.jsonl") rounds_path = os.path.join(base, f"run{run}.jsonl.rounds.json") log = read_text(log_path) out = {"run": run, "ok": False, "reason": "no capture", "damage": 0.0, "damage_taken": 0.0, "wins": None, "rounds": None, "opp_fired": 0, "opp_hits": 0, "dist": None, "scans": 0, "liveness": "n/a"} if not log: return out counters = parse_counters(log) if counters is None: out["reason"] = "battle never started (no subject counters)" return out evs = parse_events(ev_path) subj, other, how = attribute_subject(evs, counters) if subj is None: out["reason"] = f"owner attribution failed ({how})" return out dmg = dmg_taken = 0.0 opp_fired = 0 for ev in evs: t = ev.get("type") if t == "fire" and ev.get("owner") == other: opp_fired += 1 elif t == "hit": d = float(ev.get("damage", 0.0)) if ev.get("owner") == subj: dmg += d elif ev.get("owner") == other: dmg_taken += d opp_hits = sum(1 for ev in evs if ev.get("type") == "hit" and ev.get("owner") == other) wins = parse_first_places(log) if wins is None: out["reason"] = "no final standings in the capture" return out m = DIST_RE.search(log) out.update({ "ok": True, "reason": "ok", "counters": counters, "subject_id": subj, "owner_attribution": how, "damage": dmg, "damage_taken": dmg_taken, "wins": wins, "rounds": parse_rounds(rounds_path), "opp_fired": opp_fired, "opp_hits": opp_hits, "dist": float(m.group(1)) if m else None, "scans": counters["scans"], }) return out # ── arm/opponent aggregation ───────────────────────────────────────────────── def arm_metrics(runs): """Aggregate a list of valid run dicts into one metric dict.""" n = len(runs) total_rounds = sum(r["rounds"] or 0 for r in runs) opp_fired = sum(r["opp_fired"] for r in runs) opp_hits = sum(r["opp_hits"] for r in runs) dists = [r["dist"] for r in runs if r["dist"] is not None] return { "runs": n, "damage": mean([r["damage"] for r in runs]), "damage_taken": mean([r["damage_taken"] for r in runs]), "wins": mean([r["wins"] for r in runs]), "win_rate": (sum(r["wins"] for r in runs) / total_rounds if total_rounds else float("nan")), "rounds": total_rounds, "hit_rate": (100.0 * opp_hits / opp_fired) if opp_fired else float("nan"), "dist": mean(dists) if dists else float("nan"), "scans": mean([r["scans"] for r in runs]), } def collect(session): """-> (data, notes); data[opp][arm] = {"valid": [...], "invalid": [...]}""" data = {} notes = [] for opp in session["opponents"]: oname = opp["name"] data[oname] = {} opp_dir = os.path.join(session["outdir"], oname) for arm in session["arms"]: aname = arm["name"] tokens = arm["env"].split() if arm["env"] else [] node = {"valid": [], "invalid": [], "tokens": tokens, "style": opp.get("style", ""), "label": arm.get("label", "")} for r in range(1, session["runs"] + 1): rr = parse_run(opp_dir, aname, r) live_ok, live_why, fatal_leak = liveness( tokens, os.path.join(opp_dir, aname, f"run{r}.bot.stdout.log")) rr["liveness"] = live_why if not rr["ok"]: node["invalid"].append(rr) elif not live_ok: rr["reason"] = "liveness FAIL: " + live_why node["invalid"].append(rr) else: node["valid"].append(rr) data[oname][aname] = node return data, notes # ── the report ─────────────────────────────────────────────────────────────── def fmt(x, nd=1): return "n/a" if x is None or (isinstance(x, float) and math.isnan(x)) \ else f"{x:.{nd}f}" def fmt_s(x, nd=1): return "n/a" if x is None or (isinstance(x, float) and math.isnan(x)) \ else f"{x:+.{nd}f}" def main(): args = sys.argv[1:] if not args: print(__doc__) return 2 session_dir = args[0] ref = None if "--reference" in args: ref = args[args.index("--reference") + 1] try: with open(os.path.join(session_dir, "session.json")) as fh: session = json.load(fh) except (OSError, json.JSONDecodeError) as exc: print(f"ERROR: cannot read {session_dir}/session.json: {exc}", file=sys.stderr) return 2 session["outdir"] = session_dir arms = [a["name"] for a in session["arms"]] if ref is None: ref = session.get("reference") or arms[0] if ref not in arms: print(f"ERROR: reference arm '{ref}' not in session arms {arms}", file=sys.stderr) return 2 opps = [o["name"] for o in session["opponents"]] style_of = {o["name"]: o.get("style", "") for o in session["opponents"]} data, _ = collect(session) out = [] def p(s=""): out.append(s) print(s) p("### MEASURED: session") p() p(f"* commit `{session['commit']}`, frozen binary sha256 `{session['binary_sha256'][:12]}…`") p(f"* {len(opps)} opponents × {len(arms)} arms × {session['runs']} runs × " f"{session['rounds']} rounds = {len(opps) * len(arms) * session['runs']} battles, " f"conc={session.get('conc', '?')}") p(f"* arms file `{os.path.basename(session.get('arms_file', '?'))}`, " f"panel file `{os.path.basename(session.get('panel_file', '?'))}`") p(f"* reference arm: **`{ref}`** — every delta below is (arm − {ref}), " f"opponent by opponent") p() invalid = [(o, a, r) for o in opps for a in arms for r in data[o][a]["invalid"]] p(f"* liveness: {len(invalid)} run(s) excluded " f"({len(opps) * len(arms) * session['runs']} total)") for o, a, r in invalid[:20]: p(f" * `{o}/{a}` run{r['run']}: {r['reason']}") if len(invalid) > 20: p(f" * … and {len(invalid) - 20} more") p() # ── per-opponent × per-arm deltas ─────────────────────────────────────── per_opp = {a: {} for a in arms} for o in opps: ref_runs = data[o][ref]["valid"] ref_m = arm_metrics(ref_runs) if ref_runs else None for a in arms: m = arm_metrics(data[o][a]["valid"]) if data[o][a]["valid"] else None if m is None or ref_m is None: per_opp[a][o] = None continue per_opp[a][o] = { "m": m, "ref": ref_m, "d_damage": m["damage"] - ref_m["damage"], "d_wins": m["wins"] - ref_m["wins"], "d_damage_taken": m["damage_taken"] - ref_m["damage_taken"], "d_hit_rate": m["hit_rate"] - ref_m["hit_rate"], "d_dist": m["dist"] - ref_m["dist"], } # ── per-arm tables ────────────────────────────────────────────────────── p("### MEASURED: per-opponent paired table (per arm)") p() for a in arms: lbl = next((x.get("label") for x in session["arms"] if x["name"] == a), "") cnt = sum(1 for o in opps if per_opp[a][o]) p(f"#### `{a}`" + (f" — {lbl}" if lbl else "") + f" (paired on {cnt} opponents)") p() p("| opponent | style | dmg/run ref→arm | Δdmg | wins/run ref→arm | Δwins | Δdmg taken | Δhit rate (pp) | dist ref→arm |") p("|---|---|---:|---:|---:|---:|---:|---:|---:|") for o in opps: e = per_opp[a][o] if e is None: p(f"| {o} | {style_of[o]} | n/a | n/a | n/a | n/a | n/a | n/a | n/a |") continue m, rm = e["m"], e["ref"] p(f"| {o} | {style_of[o]} | {fmt(rm['damage'])}→{fmt(m['damage'])} " f"| {fmt_s(e['d_damage'])} " f"| {fmt(rm['wins'], 2)}→{fmt(m['wins'], 2)} | {fmt_s(e['d_wins'], 2)} " f"| {fmt_s(e['d_damage_taken'])} | {fmt_s(e['d_hit_rate'], 2)} " f"| {fmt(rm['dist'], 0)}→{fmt(m['dist'], 0)} |") p() # ── aggregate dashboard, pooled over every valid run ──────────────────── p("### MEASURED: pooled dashboard (all valid runs, NOT the verdict)") p() p("| arm | runs | dmg/run | dmg taken/run | wins/run | round wins | win rate | incoming hit rate | mean distance |") p("|---|---:|---:|---:|---:|---:|---:|---:|---:|") pooled = {} for a in arms: runs = [r for o in opps for r in data[o][a]["valid"]] if not runs: p(f"| `{a}` | 0 | n/a | n/a | n/a | n/a | n/a | n/a | n/a |") continue m = arm_metrics(runs) pooled[a] = m p(f"| `{a}` | {m['runs']} | {fmt(m['damage'])} | {fmt(m['damage_taken'])} " f"| {fmt(m['wins'], 2)} | {int(sum(r['wins'] for r in runs))}/{m['rounds']} " f"| {fmt(100 * m['win_rate'])}% | {fmt(m['hit_rate'], 2)}% " f"| {fmt(m['dist'], 0)} |") p() # ── cross-opponent aggregation: mean delta, spread, sign tests, MDE ───── p("### MEASURED: cross-opponent aggregation (the verdict layer)") p() p("Deltas are per-opponent (arm − reference). `spread` is the SD of those " "deltas ACROSS opponents; `SE` = spread/√n; `95% CI` = mean ± t·SE. " "Sign test = how many opponents the arm wins (ties dropped), exact " "binomial; sign-flip = permutation test on the mean of the deltas.") p() p("| arm | metric | mean Δ | spread (SD) | SE | 95% CI | sign test (wins/n) | p(sign) | p(sign-flip) | Wilcoxon p | MDE |") p("|---|---|---:|---:|---:|---|---:|---:|---:|---:|---:|") stats = {} for a in arms: if a == ref: continue ds = [per_opp[a][o] for o in opps if per_opp[a][o]] st = {"n": len(ds)} for key, mkey in (("d_damage", "damage"), ("d_wins", "wins"), ("d_damage_taken", "damage_taken"), ("d_hit_rate", "hit_rate"), ("d_dist", "dist")): vals = [e[key] for e in ds if not math.isnan(e[key])] if len(vals) < 2: continue mu = mean(vals) s = sd(vals) se = s / math.sqrt(len(vals)) crit = t_crit(len(vals) - 1) nz = [v for v in vals if v != 0.0] wins_sign = sum(1 for v in nz if v > 0) ps = binom_two_sided(wins_sign, len(nz)) pf, method = signflip_p(vals) pw, _ = wilcoxon_p(vals) st[key] = {"mean": mu, "sd": s, "se": se, "ci": (mu - crit * se, mu + crit * se), "k": wins_sign, "nz": len(nz), "p_sign": ps, "p_flip": pf, "p_flip_method": method, "p_wilcox": pw, "mde": mde(vals)} p(f"| `{a}` | {mkey} | {fmt_s(mu, 2)} | {fmt(s, 2)} | {fmt(se, 2)} " f"| [{fmt_s(mu - crit * se, 2)}, {fmt_s(mu + crit * se, 2)}] " f"| {wins_sign}/{len(nz)} | {ps:.4g} | {pf:.4g} ({method}) " f"| {pw:.4g} | {fmt(st[key]['mde'], 2)} |") stats[a] = st p() # ── style breakdown (leg/governance only) ─────────────────────────────── styles = sorted({style_of[o] for o in opps if style_of[o]}) if len(styles) > 1: p("#### By inferred style (explanation only, never the verdict)") p() p("| arm | style | n | mean Δdmg | mean Δwins | mean Δhit rate (pp) |") p("|---|---|---:|---:|---:|---:|") for a in arms: if a == ref: continue for s in styles: ds = [per_opp[a][o] for o in opps if per_opp[a][o] and style_of[o] == s] if not ds: continue p(f"| `{a}` | {s} | {len(ds)} " f"| {fmt_s(mean([e['d_damage'] for e in ds]))} " f"| {fmt_s(mean([e['d_wins'] for e in ds]), 2)} " f"| {fmt_s(mean([e['d_hit_rate'] for e in ds]), 2)} |") p() # ── the pre-registered verdict rules ──────────────────────────────────── p("### The pre-registered verdict (rules fixed in `docs/movement_campaign.md`)") p() p("PRIMARY metrics are dmg/run and wins/run; hit rate is never the verdict. " "The pre-registered rule says an arm is BETTER when one primary metric is " "UP at sign-test p<0.05 `while the other does not go down`. That phrase has " "two readings and BOTH are printed:") p() p("* **strict** — the other metric's mean delta is not negative at all " "(`Δ >= 0`). Nothing can be BETTER while it costs *any* mean damage.") p("* **substantive** — the other metric's delta is not *detectably* down: " "the sign test is not significant **and** the delta is smaller than that " "metric's MDE (the pre-registered rule 3 says an effect under the MDE is " "not detectable, so it cannot count as a loss).") p() ranked = [] for a in arms: if a == ref: continue st = stats.get(a, {}) d, w = st.get("d_damage"), st.get("d_wins") if d is None or w is None: ranked.append((a, "n/a", "n/a", float("nan"), float("nan"))) continue up_d = d["mean"] > 0 and d["p_sign"] < 0.05 dn_d = d["mean"] < 0 and d["p_sign"] < 0.05 up_w = w["mean"] > 0 and w["p_sign"] < 0.05 dn_w = w["mean"] < 0 and w["p_sign"] < 0.05 def not_down(o): return o["mean"] >= 0 def not_down_subst(o): return o["mean"] > -o["mde"] and o["p_sign"] >= 0.05 def not_up_subst(o): return o["mean"] < o["mde"] and o["p_sign"] >= 0.05 if (up_d and not_down(w)) or (up_w and not_down(d)): strict = "BETTER" elif (dn_d and w["mean"] <= 0) or (dn_w and d["mean"] <= 0): strict = "WORSE" else: strict = "not distinguishable" if (up_d and not_down_subst(w)) or (up_w and not_down_subst(d)): subst = "BETTER" elif (dn_d and not_up_subst(w)) or (dn_w and not_up_subst(d)): subst = "WORSE" else: subst = "not distinguishable" ranked.append((a, strict, subst, w["mean"], d["mean"])) ranked.sort(key=lambda t: (-(t[3] if t[3] == t[3] else -1e9), -(t[4] if t[4] == t[4] else -1e9))) p("| rank | arm | Δwins/run | Δdmg/run | sign test wins | sign test dmg | verdict (strict) | verdict (substantive) |") p("|---:|---|---:|---:|---|---|---|---|") for i, (a, strict, subst, w, d) in enumerate(ranked, 1): st = stats.get(a, {}) dd = st.get("d_damage", {}) ww = st.get("d_wins", {}) p(f"| {i} | `{a}` | {fmt_s(w, 2)} | {fmt_s(d)} " f"| {ww.get('k', 'n/a')}/{ww.get('nz', 'n/a')} p={ww.get('p_sign', float('nan')):.4g} " f"| {dd.get('k', 'n/a')}/{dd.get('nz', 'n/a')} p={dd.get('p_sign', float('nan')):.4g} " f"| **{strict}** | **{subst}** |") p() p(f"Reference `{ref}`: {fmt(pooled.get(ref, {}).get('damage'))} dmg/run, " f"{fmt(pooled.get(ref, {}).get('wins'), 2)} wins/run, " f"{fmt(pooled.get(ref, {}).get('hit_rate'), 2)}% incoming, " f"{fmt(pooled.get(ref, {}).get('dist'), 0)} px.") best = ranked[0] if ranked else None if best: p() p(f"Highest wins delta: `{best[0]}` ({fmt_s(best[3], 2)} wins/run, " f"{fmt_s(best[4])} dmg/run) — strict: **{best[1]}**, " f"substantive: **{best[2]}**.") return 0 if __name__ == "__main__": sys.exit(main())