melee A/B doc: correct the per-arm [bb-reset] counts (perRound 68.75, retained 68.19, decay 66.75)

This commit is contained in:
2026-09-26 00:43:04 +02:00
parent da4a971ca9
commit 1984a780f4
6 changed files with 1342 additions and 3 deletions
+639
View File
@@ -0,0 +1,639 @@
#!/usr/bin/env python3
"""tournament_analyze.py — paired multi-opponent ranking of movement arms.
python3 tools/ab/tournament_analyze.py <session_dir> [--reference ARM]
Reads a session produced by `tournament_run.sh` (layout:
<outdir>/<opponent>/<arm>/run<N>.{battle.log,events.jsonl,jsonl,bot.stdout.log})
and answers the only question the movement campaign asks:
DOES ARM A MOVE BETTER THAN THE REFERENCE ARM, ACROSS OPPONENTS?
Method, in one paragraph. Every arm fights the SAME panel with the SAME frozen
binary; for each opponent the arm's metric is averaged over its runs and
subtracted from the reference arm's average for that same opponent. Those
per-opponent deltas are the unit of evidence: the MEAN of the deltas is the
effect, the SPREAD of the deltas across opponents is the honest error bar (one
weird opponent cannot carry it), and a SIGN TEST over the deltas says how many
opponents the arm actually wins. A pooled number over all runs is also printed,
but it is reported as the descriptive dashboard, never as the verdict.
CLI judgment (pre-registered in docs/movement_campaign.md): the primary metrics
are damage/run and ROUND WINS. An arm is BETTER than the reference only if one
of the two improves with the sign test at p<0.05 while the other does not
degrade; hit rate and distance are explanation, never the verdict. MDE
(alpha=0.05 two-sided, 80% power) is printed for every test so a null can be
told apart from an under-powered null.
Standard library only. Deterministic: the exact sign-flip test is enumerated
when n_opponents <= 20, otherwise sampled with a fixed seed.
"""
import json
import math
import os
import random
import re
import sys
BOT_NAME = "ModularBot"
# The metric keys, in report order.
METRICS = ["damage", "damage_taken", "wins", "hit_rate", "dist"]
# z_{0.975} + z_{0.80}: the constant in MDE = C * sd * sqrt(2/n) is for a
# two-SAMPLE design; for the paired per-opponent design used here the analogous
# constant with n = number of opponents is C * sd(deltas) / sqrt(n).
MDE_C = 1.959963984540054 + 0.8416212335729143
EXACT_SIGNCAP = 20 # 2^20 = 1M sign vectors is still instant
MC_DRAWS = 200_000
MC_SEED = 0x5EED5EED
T975 = {1: 12.706, 2: 4.303, 3: 3.182, 4: 2.776, 5: 2.571, 6: 2.447,
7: 2.365, 8: 2.306, 9: 2.262, 10: 2.228, 11: 2.201, 12: 2.179,
13: 2.160, 14: 2.145, 15: 2.131, 16: 2.120, 17: 2.110, 18: 2.101,
19: 2.093, 20: 2.086, 21: 2.080, 22: 2.074, 23: 2.069, 24: 2.064,
25: 2.060, 26: 2.056, 27: 2.052, 28: 2.048, 29: 2.045, 30: 2.042}
# ── small statistics helpers ─────────────────────────────────────────────────
def mean(xs):
return sum(xs) / len(xs) if xs else float("nan")
def sd(xs):
"""Sample standard deviation (n-1). 0.0 for n<2."""
n = len(xs)
if n < 2:
return 0.0
m = mean(xs)
return math.sqrt(sum((x - m) ** 2 for x in xs) / (n - 1))
def median(xs):
s = sorted(xs)
n = len(s)
if n == 0:
return float("nan")
return s[n // 2] if n % 2 else 0.5 * (s[n // 2 - 1] + s[n // 2])
def binom_two_sided(k, n):
"""Exact two-sided sign-test p-value (p=0.5), ties already removed."""
if n == 0:
return 1.0
def c(nn, kk):
return math.comb(nn, kk)
tail = sum(c(n, i) for i in range(0, min(k, n - k) + 1)) / 2 ** n
return min(1.0, 2.0 * tail)
def signflip_p(deltas):
"""Two-sided sign-flip permutation test on the MEAN of the deltas.
Exact (all 2^n sign vectors) for n <= EXACT_SIGNCAP; otherwise a
deterministic Monte-Carlo draw. Returns (p, method_string)."""
n = len(deltas)
if n == 0:
return 1.0, "n/a"
obs = abs(mean(deltas))
if obs == 0.0:
return 1.0, "exact (degenerate)"
tol = 1e-12
if n <= EXACT_SIGNCAP:
total = 1 << n
hits = 0
for mask in range(total):
s = 0.0
for i, d in enumerate(deltas):
s += -d if (mask >> i) & 1 else d
if abs(s / n) >= obs - tol:
hits += 1
return hits / total, f"exact 2^{n}"
rng = random.Random(MC_SEED)
hits = 0
for _ in range(MC_DRAWS):
s = 0.0
for d in deltas:
s += -d if rng.getrandbits(1) else d
if abs(s / n) >= obs - tol:
hits += 1
return hits / MC_DRAWS, f"MC {MC_DRAWS}"
def wilcoxon_p(deltas):
"""Two-sided Wilcoxon signed-rank, normal approximation with tie
correction. Returns (p, W). p=1.0 when there is nothing to test."""
nz = [d for d in deltas if d != 0.0]
n = len(nz)
if n < 3:
return 1.0, float("nan")
order = sorted(range(n), key=lambda i: abs(nz[i]))
ranks = [0.0] * n
i = 0
while i < n:
j = i
while j + 1 < n and abs(nz[order[j + 1]]) == abs(nz[order[i]]):
j += 1
avg = (i + j) / 2.0 + 1.0
for k in range(i, j + 1):
ranks[order[k]] = avg
i = j + 1
w_plus = sum(ranks[i] for i in range(n) if nz[i] > 0)
mu = n * (n + 1) / 4.0
# tie correction for sigma
from collections import Counter
cnt = Counter(abs(d) for d in nz)
tie = sum(c ** 3 - c for c in cnt.values())
sigma2 = n * (n + 1) * (2 * n + 1) / 24.0 - tie / 48.0
if sigma2 <= 0:
return 1.0, w_plus
z = (w_plus - mu - 0.5 * (1 if w_plus > mu else -1)) / math.sqrt(sigma2)
p = 2.0 * 0.5 * math.erfc(abs(z) / math.sqrt(2.0))
return min(1.0, p), w_plus
def t_crit(df):
return T975.get(df, 1.96)
def mde(deltas):
"""Minimum detectable effect for the paired per-opponent design at
alpha=0.05 (two-sided), power 80%, from the observed spread of the deltas."""
n = len(deltas)
if n < 2:
return float("nan")
return MDE_C * sd(deltas) / math.sqrt(n)
# ── parsing ──────────────────────────────────────────────────────────────────
COUNTERS_RE = re.compile(
r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)"
r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)")
FIRSTPLACES_RE = re.compile(
r"^\s*#\d+\s+(\S+)\s+totalScore=(-?\d+)\s+firstPlaces=(\d+)\s+survival=(\d+)",
re.MULTILINE)
DIST_RE = re.compile(r"DISTANCE: mean=([\d.]+)")
ROWS_RE = re.compile(r"rows=(\d+)")
def read_text(path):
try:
with open(path, "r", errors="replace") as fh:
return fh.read()
except OSError:
return ""
def parse_events(path):
out = []
try:
with open(path, "r", errors="replace") as fh:
for line in fh:
line = line.strip()
if not line:
continue
try:
out.append(json.loads(line))
except json.JSONDecodeError:
continue # a partial line from a killed battle
except OSError:
out = []
return out
def parse_counters(text):
m = COUNTERS_RE.search(text)
if not m:
return None
return {"scans": int(m.group(1)), "fired": int(m.group(2)),
"hits": int(m.group(3)), "misses": int(m.group(4)),
"hit_bullets": int(m.group(5)), "hits_taken": int(m.group(6))}
def parse_first_places(text):
for m in FIRSTPLACES_RE.finditer(text):
if m.group(1) == BOT_NAME:
return int(m.group(3))
return None
def parse_rounds(path):
try:
with open(path) as fh:
return len(json.load(fh).get("rounds", []))
except (OSError, json.JSONDecodeError):
return None
def attribute_subject(evs, counters):
"""Which event `owner` id is our bot? Matched on fire/hit/hits-taken counts,
never on fired power (the power policy is continuous).
Returns (subject_id, other_id, how). `subject_id` is None only when nothing
could be identified; `other_id` is None when the opponent never fired a
single bullet (then the incoming hit rate is simply undefined, and the run
is still a valid damage measurement)."""
fires, hits, victim_hits = {}, {}, {}
owner_ids = set()
for ev in evs:
t = ev.get("type")
o = ev.get("owner")
if o is not None:
owner_ids.add(o)
if t == "fire":
fires[o] = fires.get(o, 0) + 1
elif t == "hit":
hits[o] = hits.get(o, 0) + 1
v = ev.get("victim")
if v is not None:
victim_hits[v] = victim_hits.get(v, 0) + 1
if not fires:
return None, None, "no fire events"
def other_of(subj):
others = [o for o in owner_ids if o != subj]
return others[0] if len(others) == 1 else None
if counters:
strict = [o for o in fires
if fires[o] == counters["fired"]
and hits.get(o, 0) == counters["hits"]
and victim_hits.get(o, 0) == counters["hits_taken"]]
if len(strict) == 1:
return strict[0], other_of(strict[0]), "exact"
cand = [o for o in fires if counters and fires[o] == counters["fired"]]
if len(cand) != 1:
cand = [o for o in fires
if counters and victim_hits.get(o, 0) == counters["hits_taken"]]
if len(cand) != 1:
cand = list(fires)
if len(cand) == 1:
return cand[0], other_of(cand[0]), "fires-only"
return None, None, "ambiguous"
def liveness(arm_env_tokens, stdout_path):
"""Liveness: every declared env token must appear verbatim in OUR bot's own
boot environment report, so an arm whose setting never reached the process
is a loud FAIL instead of a plausible-looking number. A TR_MOVEMENT the arm
did NOT declare is fatal too (the baseline would be contaminated)."""
text = read_text(stdout_path)
if not text:
return False, "no boot env report", False
missing = [t for t in arm_env_tokens if t not in text]
declared_keys = {t.split("=", 1)[0] for t in arm_env_tokens}
leaked_movement = ("TR_MOVEMENT" not in declared_keys
and re.search(r"^\[env\] TR_MOVEMENT=", text, re.MULTILINE)
is not None)
ok = (not missing) and (not leaked_movement)
why = []
if missing:
why.append("not seen in bot env report: " + ", ".join(missing))
if leaked_movement:
why.append("undeclared TR_MOVEMENT leaked into the process")
return ok, ("; ".join(why) if why else "ok"), bool(leaked_movement)
def parse_run(opp_dir, arm_dir, run):
"""One run -> dict of MEASURED numbers, or ok=False with a reason."""
base = os.path.join(opp_dir, arm_dir)
log_path = os.path.join(base, f"run{run}.battle.log")
ev_path = os.path.join(base, f"run{run}.events.jsonl")
rounds_path = os.path.join(base, f"run{run}.jsonl.rounds.json")
log = read_text(log_path)
out = {"run": run, "ok": False, "reason": "no capture",
"damage": 0.0, "damage_taken": 0.0, "wins": None, "rounds": None,
"opp_fired": 0, "opp_hits": 0, "dist": None, "scans": 0,
"liveness": "n/a"}
if not log:
return out
counters = parse_counters(log)
if counters is None:
out["reason"] = "battle never started (no subject counters)"
return out
evs = parse_events(ev_path)
subj, other, how = attribute_subject(evs, counters)
if subj is None:
out["reason"] = f"owner attribution failed ({how})"
return out
dmg = dmg_taken = 0.0
opp_fired = 0
for ev in evs:
t = ev.get("type")
if t == "fire" and ev.get("owner") == other:
opp_fired += 1
elif t == "hit":
d = float(ev.get("damage", 0.0))
if ev.get("owner") == subj:
dmg += d
elif ev.get("owner") == other:
dmg_taken += d
opp_hits = sum(1 for ev in evs
if ev.get("type") == "hit" and ev.get("owner") == other)
wins = parse_first_places(log)
if wins is None:
out["reason"] = "no final standings in the capture"
return out
m = DIST_RE.search(log)
out.update({
"ok": True, "reason": "ok",
"counters": counters, "subject_id": subj, "owner_attribution": how,
"damage": dmg, "damage_taken": dmg_taken,
"wins": wins, "rounds": parse_rounds(rounds_path),
"opp_fired": opp_fired, "opp_hits": opp_hits,
"dist": float(m.group(1)) if m else None,
"scans": counters["scans"],
})
return out
# ── arm/opponent aggregation ─────────────────────────────────────────────────
def arm_metrics(runs):
"""Aggregate a list of valid run dicts into one metric dict."""
n = len(runs)
total_rounds = sum(r["rounds"] or 0 for r in runs)
opp_fired = sum(r["opp_fired"] for r in runs)
opp_hits = sum(r["opp_hits"] for r in runs)
dists = [r["dist"] for r in runs if r["dist"] is not None]
return {
"runs": n,
"damage": mean([r["damage"] for r in runs]),
"damage_taken": mean([r["damage_taken"] for r in runs]),
"wins": mean([r["wins"] for r in runs]),
"win_rate": (sum(r["wins"] for r in runs) / total_rounds
if total_rounds else float("nan")),
"rounds": total_rounds,
"hit_rate": (100.0 * opp_hits / opp_fired) if opp_fired else float("nan"),
"dist": mean(dists) if dists else float("nan"),
"scans": mean([r["scans"] for r in runs]),
}
def collect(session):
"""-> (data, notes); data[opp][arm] = {"valid": [...], "invalid": [...]}"""
data = {}
notes = []
for opp in session["opponents"]:
oname = opp["name"]
data[oname] = {}
opp_dir = os.path.join(session["outdir"], oname)
for arm in session["arms"]:
aname = arm["name"]
tokens = arm["env"].split() if arm["env"] else []
node = {"valid": [], "invalid": [], "tokens": tokens,
"style": opp.get("style", ""), "label": arm.get("label", "")}
for r in range(1, session["runs"] + 1):
rr = parse_run(opp_dir, aname, r)
live_ok, live_why, fatal_leak = liveness(
tokens, os.path.join(opp_dir, aname, f"run{r}.bot.stdout.log"))
rr["liveness"] = live_why
if not rr["ok"]:
node["invalid"].append(rr)
elif not live_ok:
rr["reason"] = "liveness FAIL: " + live_why
node["invalid"].append(rr)
else:
node["valid"].append(rr)
data[oname][aname] = node
return data, notes
# ── the report ───────────────────────────────────────────────────────────────
def fmt(x, nd=1):
return "n/a" if x is None or (isinstance(x, float) and math.isnan(x)) \
else f"{x:.{nd}f}"
def fmt_s(x, nd=1):
return "n/a" if x is None or (isinstance(x, float) and math.isnan(x)) \
else f"{x:+.{nd}f}"
def main():
args = sys.argv[1:]
if not args:
print(__doc__)
return 2
session_dir = args[0]
ref = None
if "--reference" in args:
ref = args[args.index("--reference") + 1]
try:
with open(os.path.join(session_dir, "session.json")) as fh:
session = json.load(fh)
except (OSError, json.JSONDecodeError) as exc:
print(f"ERROR: cannot read {session_dir}/session.json: {exc}", file=sys.stderr)
return 2
session["outdir"] = session_dir
arms = [a["name"] for a in session["arms"]]
if ref is None:
ref = session.get("reference") or arms[0]
if ref not in arms:
print(f"ERROR: reference arm '{ref}' not in session arms {arms}",
file=sys.stderr)
return 2
opps = [o["name"] for o in session["opponents"]]
style_of = {o["name"]: o.get("style", "") for o in session["opponents"]}
data, _ = collect(session)
out = []
def p(s=""):
out.append(s)
print(s)
p("### MEASURED: session")
p()
p(f"* commit `{session['commit']}`, frozen binary sha256 `{session['binary_sha256'][:12]}…`")
p(f"* {len(opps)} opponents × {len(arms)} arms × {session['runs']} runs × "
f"{session['rounds']} rounds = {len(opps) * len(arms) * session['runs']} battles, "
f"conc={session.get('conc', '?')}")
p(f"* arms file `{os.path.basename(session.get('arms_file', '?'))}`, "
f"panel file `{os.path.basename(session.get('panel_file', '?'))}`")
p(f"* reference arm: **`{ref}`** — every delta below is (arm − {ref}), "
f"opponent by opponent")
p()
invalid = [(o, a, r) for o in opps for a in arms for r in data[o][a]["invalid"]]
p(f"* liveness: {len(invalid)} run(s) excluded "
f"({len(opps) * len(arms) * session['runs']} total)")
for o, a, r in invalid[:20]:
p(f" * `{o}/{a}` run{r['run']}: {r['reason']}")
if len(invalid) > 20:
p(f" * … and {len(invalid) - 20} more")
p()
# ── per-opponent × per-arm deltas ───────────────────────────────────────
per_opp = {a: {} for a in arms}
for o in opps:
ref_runs = data[o][ref]["valid"]
ref_m = arm_metrics(ref_runs) if ref_runs else None
for a in arms:
m = arm_metrics(data[o][a]["valid"]) if data[o][a]["valid"] else None
if m is None or ref_m is None:
per_opp[a][o] = None
continue
per_opp[a][o] = {
"m": m, "ref": ref_m,
"d_damage": m["damage"] - ref_m["damage"],
"d_wins": m["wins"] - ref_m["wins"],
"d_damage_taken": m["damage_taken"] - ref_m["damage_taken"],
"d_hit_rate": m["hit_rate"] - ref_m["hit_rate"],
"d_dist": m["dist"] - ref_m["dist"],
}
# ── per-arm tables ──────────────────────────────────────────────────────
p("### MEASURED: per-opponent paired table (per arm)")
p()
for a in arms:
lbl = next((x.get("label") for x in session["arms"] if x["name"] == a), "")
cnt = sum(1 for o in opps if per_opp[a][o])
p(f"#### `{a}`" + (f" — {lbl}" if lbl else "") + f" (paired on {cnt} opponents)")
p()
p("| opponent | style | dmg/run ref→arm | Δdmg | wins/run ref→arm | Δwins | Δdmg taken | Δhit rate (pp) | dist ref→arm |")
p("|---|---|---:|---:|---:|---:|---:|---:|---:|")
for o in opps:
e = per_opp[a][o]
if e is None:
p(f"| {o} | {style_of[o]} | n/a | n/a | n/a | n/a | n/a | n/a | n/a |")
continue
m, rm = e["m"], e["ref"]
p(f"| {o} | {style_of[o]} | {fmt(rm['damage'])}→{fmt(m['damage'])} "
f"| {fmt_s(e['d_damage'])} "
f"| {fmt(rm['wins'], 2)}→{fmt(m['wins'], 2)} | {fmt_s(e['d_wins'], 2)} "
f"| {fmt_s(e['d_damage_taken'])} | {fmt_s(e['d_hit_rate'], 2)} "
f"| {fmt(rm['dist'], 0)}→{fmt(m['dist'], 0)} |")
p()
# ── aggregate dashboard, pooled over every valid run ────────────────────
p("### MEASURED: pooled dashboard (all valid runs, NOT the verdict)")
p()
p("| arm | runs | dmg/run | dmg taken/run | wins/run | round wins | win rate | incoming hit rate | mean distance |")
p("|---|---:|---:|---:|---:|---:|---:|---:|---:|")
pooled = {}
for a in arms:
runs = [r for o in opps for r in data[o][a]["valid"]]
if not runs:
p(f"| `{a}` | 0 | n/a | n/a | n/a | n/a | n/a | n/a | n/a |")
continue
m = arm_metrics(runs)
pooled[a] = m
p(f"| `{a}` | {m['runs']} | {fmt(m['damage'])} | {fmt(m['damage_taken'])} "
f"| {fmt(m['wins'], 2)} | {int(sum(r['wins'] for r in runs))}/{m['rounds']} "
f"| {fmt(100 * m['win_rate'])}% | {fmt(m['hit_rate'], 2)}% "
f"| {fmt(m['dist'], 0)} |")
p()
# ── cross-opponent aggregation: mean delta, spread, sign tests, MDE ─────
p("### MEASURED: cross-opponent aggregation (the verdict layer)")
p()
p("Deltas are per-opponent (arm − reference). `spread` is the SD of those "
"deltas ACROSS opponents; `SE` = spread/√n; `95% CI` = mean ± t·SE. "
"Sign test = how many opponents the arm wins (ties dropped), exact "
"binomial; sign-flip = permutation test on the mean of the deltas.")
p()
p("| arm | metric | mean Δ | spread (SD) | SE | 95% CI | sign test (wins/n) | p(sign) | p(sign-flip) | Wilcoxon p | MDE |")
p("|---|---|---:|---:|---:|---|---:|---:|---:|---:|---:|")
stats = {}
for a in arms:
if a == ref:
continue
ds = [per_opp[a][o] for o in opps if per_opp[a][o]]
st = {"n": len(ds)}
for key, mkey in (("d_damage", "damage"), ("d_wins", "wins"),
("d_damage_taken", "damage_taken"),
("d_hit_rate", "hit_rate"), ("d_dist", "dist")):
vals = [e[key] for e in ds if not math.isnan(e[key])]
if len(vals) < 2:
continue
mu = mean(vals)
s = sd(vals)
se = s / math.sqrt(len(vals))
crit = t_crit(len(vals) - 1)
nz = [v for v in vals if v != 0.0]
wins_sign = sum(1 for v in nz if v > 0)
ps = binom_two_sided(wins_sign, len(nz))
pf, method = signflip_p(vals)
pw, _ = wilcoxon_p(vals)
st[key] = {"mean": mu, "sd": s, "se": se,
"ci": (mu - crit * se, mu + crit * se),
"k": wins_sign, "nz": len(nz), "p_sign": ps,
"p_flip": pf, "p_flip_method": method, "p_wilcox": pw,
"mde": mde(vals)}
p(f"| `{a}` | {mkey} | {fmt_s(mu, 2)} | {fmt(s, 2)} | {fmt(se, 2)} "
f"| [{fmt_s(mu - crit * se, 2)}, {fmt_s(mu + crit * se, 2)}] "
f"| {wins_sign}/{len(nz)} | {ps:.4g} | {pf:.4g} ({method}) "
f"| {pw:.4g} | {fmt(st[key]['mde'], 2)} |")
stats[a] = st
p()
# ── style breakdown (leg/governance only) ───────────────────────────────
styles = sorted({style_of[o] for o in opps if style_of[o]})
if len(styles) > 1:
p("#### By inferred style (explanation only, never the verdict)")
p()
p("| arm | style | n | mean Δdmg | mean Δwins | mean Δhit rate (pp) |")
p("|---|---|---:|---:|---:|---:|")
for a in arms:
if a == ref:
continue
for s in styles:
ds = [per_opp[a][o] for o in opps
if per_opp[a][o] and style_of[o] == s]
if not ds:
continue
p(f"| `{a}` | {s} | {len(ds)} "
f"| {fmt_s(mean([e['d_damage'] for e in ds]))} "
f"| {fmt_s(mean([e['d_wins'] for e in ds]), 2)} "
f"| {fmt_s(mean([e['d_hit_rate'] for e in ds]), 2)} |")
p()
# ── the pre-registered verdict rules ────────────────────────────────────
p("### The pre-registered verdict (rules fixed in `docs/movement_campaign.md`)")
p()
p("BETTER = one primary metric (dmg/run, wins/run) up at sign-test p<0.05 "
"with the other not down; WORSE = the mirror image; otherwise NOT "
"DISTINGUISHABLE. Hit rate is never the verdict.")
p()
ranked = []
for a in arms:
if a == ref:
continue
st = stats.get(a, {})
d, w = st.get("d_damage"), st.get("d_wins")
if d is None or w is None:
ranked.append((a, "n/a", float("nan"), float("nan")))
continue
up_d = d["mean"] > 0 and d["p_sign"] < 0.05
dn_d = d["mean"] < 0 and d["p_sign"] < 0.05
up_w = w["mean"] > 0 and w["p_sign"] < 0.05
dn_w = w["mean"] < 0 and w["p_sign"] < 0.05
if (up_d and w["mean"] >= 0) or (up_w and d["mean"] >= 0):
verdict = "BETTER than reference"
elif (dn_d and w["mean"] <= 0) or (dn_w and d["mean"] <= 0):
verdict = "WORSE than reference"
else:
verdict = "NOT DISTINGUISHABLE from reference"
ranked.append((a, verdict, w["mean"], d["mean"]))
ranked.sort(key=lambda t: (-(t[2] if t[2] == t[2] else -1e9),
-(t[3] if t[3] == t[3] else -1e9)))
p("| rank | arm | Δwins/run | Δdmg/run | sign test dmg | sign test wins | verdict |")
p("|---:|---|---:|---:|---|---|---|")
for i, (a, verdict, w, d) in enumerate(ranked, 1):
st = stats.get(a, {})
dd = st.get("d_damage", {})
ww = st.get("d_wins", {})
p(f"| {i} | `{a}` | {fmt_s(w, 2)} | {fmt_s(d)} "
f"| {dd.get('k', 'n/a')}/{dd.get('nz', 'n/a')} p={dd.get('p_sign', float('nan')):.4g} "
f"| {ww.get('k', 'n/a')}/{ww.get('nz', 'n/a')} p={ww.get('p_sign', float('nan')):.4g} "
f"| **{verdict}** |")
p()
p(f"Reference `{ref}`: {fmt(pooled.get(ref, {}).get('damage'))} dmg/run, "
f"{fmt(pooled.get(ref, {}).get('wins'), 2)} wins/run, "
f"{fmt(pooled.get(ref, {}).get('hit_rate'), 2)}% incoming, "
f"{fmt(pooled.get(ref, {}).get('dist'), 0)} px.")
best = ranked[0] if ranked else None
if best:
p()
p(f"Highest wins delta: `{best[0]}` ({fmt_s(best[2], 2)} wins/run, "
f"{fmt_s(best[3])} dmg/run) — **{best[1]}**.")
return 0
if __name__ == "__main__":
sys.exit(main())