Files
SirRoboGarage/tools/ab/tournament_analyze.py
T
SirStone 07766303f5 movement Batch 1: pure strafe (range tilt OFF) beats the shipped tfil on round wins across a 15-opponent panel
225 battles, one frozen binary, five env-only arms, the frozen panel, 0 invalid
runs. Paired per opponent vs the shipped tfil:

  strafe_notilt  wins/run +0.38  [CI +0.16,+0.60]  9/9 opponents p=0.0039
                 dmg/run  -10.2  [CI -25.8,+5.5]   p=0.61, MDE 20.4 (not detectable)
                 incoming hit rate 12.24% vs 18.17%, dmg taken 150 vs 200
  strafe_325     wins/run +0.33  [CI +0.04,+0.63]  10/12 p=0.0386
  ring           dmg/run  +31.2  [CI +11.5,+50.9]  13/15 p=0.0074, wins/run -0.04 (ns)
                 but hit rate 29.4% at 236 px: a damage/survival trade, not a win
  ring_notemp    indistinguishable from tfil on both primaries

Round wins in this harness are survival wins (in 216/219 attributable runs the
win count equals the rounds the opponent died in), and the winner takes ~1/3
fewer hits while fighting ~74 px farther out. The shipped tfil is last of five
on wins: the DrussGT-only picture did not generalize.

Also: tournament_analyze.py now prints BOTH readings of the pre-registered
'while the other does not go down' clause (strict: nothing is better;
substantive: the two strafe arms and ring are better on one metric each).
2026-09-26 01:13:29 +02:00

665 lines
27 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""tournament_analyze.py — paired multi-opponent ranking of movement arms.
python3 tools/ab/tournament_analyze.py <session_dir> [--reference ARM]
Reads a session produced by `tournament_run.sh` (layout:
<outdir>/<opponent>/<arm>/run<N>.{battle.log,events.jsonl,jsonl,bot.stdout.log})
and answers the only question the movement campaign asks:
DOES ARM A MOVE BETTER THAN THE REFERENCE ARM, ACROSS OPPONENTS?
Method, in one paragraph. Every arm fights the SAME panel with the SAME frozen
binary; for each opponent the arm's metric is averaged over its runs and
subtracted from the reference arm's average for that same opponent. Those
per-opponent deltas are the unit of evidence: the MEAN of the deltas is the
effect, the SPREAD of the deltas across opponents is the honest error bar (one
weird opponent cannot carry it), and a SIGN TEST over the deltas says how many
opponents the arm actually wins. A pooled number over all runs is also printed,
but it is reported as the descriptive dashboard, never as the verdict.
CLI judgment (pre-registered in docs/movement_campaign.md): the primary metrics
are damage/run and ROUND WINS. An arm is BETTER than the reference only if one
of the two improves with the sign test at p<0.05 while the other does not
degrade; hit rate and distance are explanation, never the verdict. MDE
(alpha=0.05 two-sided, 80% power) is printed for every test so a null can be
told apart from an under-powered null.
Standard library only. Deterministic: the exact sign-flip test is enumerated
when n_opponents <= 20, otherwise sampled with a fixed seed.
"""
import json
import math
import os
import random
import re
import sys
BOT_NAME = "ModularBot"
# The metric keys, in report order.
METRICS = ["damage", "damage_taken", "wins", "hit_rate", "dist"]
# z_{0.975} + z_{0.80}: the constant in MDE = C * sd * sqrt(2/n) is for a
# two-SAMPLE design; for the paired per-opponent design used here the analogous
# constant with n = number of opponents is C * sd(deltas) / sqrt(n).
MDE_C = 1.959963984540054 + 0.8416212335729143
EXACT_SIGNCAP = 20 # 2^20 = 1M sign vectors is still instant
MC_DRAWS = 200_000
MC_SEED = 0x5EED5EED
T975 = {1: 12.706, 2: 4.303, 3: 3.182, 4: 2.776, 5: 2.571, 6: 2.447,
7: 2.365, 8: 2.306, 9: 2.262, 10: 2.228, 11: 2.201, 12: 2.179,
13: 2.160, 14: 2.145, 15: 2.131, 16: 2.120, 17: 2.110, 18: 2.101,
19: 2.093, 20: 2.086, 21: 2.080, 22: 2.074, 23: 2.069, 24: 2.064,
25: 2.060, 26: 2.056, 27: 2.052, 28: 2.048, 29: 2.045, 30: 2.042}
# ── small statistics helpers ─────────────────────────────────────────────────
def mean(xs):
return sum(xs) / len(xs) if xs else float("nan")
def sd(xs):
"""Sample standard deviation (n-1). 0.0 for n<2."""
n = len(xs)
if n < 2:
return 0.0
m = mean(xs)
return math.sqrt(sum((x - m) ** 2 for x in xs) / (n - 1))
def median(xs):
s = sorted(xs)
n = len(s)
if n == 0:
return float("nan")
return s[n // 2] if n % 2 else 0.5 * (s[n // 2 - 1] + s[n // 2])
def binom_two_sided(k, n):
"""Exact two-sided sign-test p-value (p=0.5), ties already removed."""
if n == 0:
return 1.0
def c(nn, kk):
return math.comb(nn, kk)
tail = sum(c(n, i) for i in range(0, min(k, n - k) + 1)) / 2 ** n
return min(1.0, 2.0 * tail)
def signflip_p(deltas):
"""Two-sided sign-flip permutation test on the MEAN of the deltas.
Exact (all 2^n sign vectors) for n <= EXACT_SIGNCAP; otherwise a
deterministic Monte-Carlo draw. Returns (p, method_string)."""
n = len(deltas)
if n == 0:
return 1.0, "n/a"
obs = abs(mean(deltas))
if obs == 0.0:
return 1.0, "exact (degenerate)"
tol = 1e-12
if n <= EXACT_SIGNCAP:
total = 1 << n
hits = 0
for mask in range(total):
s = 0.0
for i, d in enumerate(deltas):
s += -d if (mask >> i) & 1 else d
if abs(s / n) >= obs - tol:
hits += 1
return hits / total, f"exact 2^{n}"
rng = random.Random(MC_SEED)
hits = 0
for _ in range(MC_DRAWS):
s = 0.0
for d in deltas:
s += -d if rng.getrandbits(1) else d
if abs(s / n) >= obs - tol:
hits += 1
return hits / MC_DRAWS, f"MC {MC_DRAWS}"
def wilcoxon_p(deltas):
"""Two-sided Wilcoxon signed-rank, normal approximation with tie
correction. Returns (p, W). p=1.0 when there is nothing to test."""
nz = [d for d in deltas if d != 0.0]
n = len(nz)
if n < 3:
return 1.0, float("nan")
order = sorted(range(n), key=lambda i: abs(nz[i]))
ranks = [0.0] * n
i = 0
while i < n:
j = i
while j + 1 < n and abs(nz[order[j + 1]]) == abs(nz[order[i]]):
j += 1
avg = (i + j) / 2.0 + 1.0
for k in range(i, j + 1):
ranks[order[k]] = avg
i = j + 1
w_plus = sum(ranks[i] for i in range(n) if nz[i] > 0)
mu = n * (n + 1) / 4.0
# tie correction for sigma
from collections import Counter
cnt = Counter(abs(d) for d in nz)
tie = sum(c ** 3 - c for c in cnt.values())
sigma2 = n * (n + 1) * (2 * n + 1) / 24.0 - tie / 48.0
if sigma2 <= 0:
return 1.0, w_plus
z = (w_plus - mu - 0.5 * (1 if w_plus > mu else -1)) / math.sqrt(sigma2)
p = 2.0 * 0.5 * math.erfc(abs(z) / math.sqrt(2.0))
return min(1.0, p), w_plus
def t_crit(df):
return T975.get(df, 1.96)
def mde(deltas):
"""Minimum detectable effect for the paired per-opponent design at
alpha=0.05 (two-sided), power 80%, from the observed spread of the deltas."""
n = len(deltas)
if n < 2:
return float("nan")
return MDE_C * sd(deltas) / math.sqrt(n)
# ── parsing ──────────────────────────────────────────────────────────────────
COUNTERS_RE = re.compile(
r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)"
r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)")
FIRSTPLACES_RE = re.compile(
r"^\s*#\d+\s+(\S+)\s+totalScore=(-?\d+)\s+firstPlaces=(\d+)\s+survival=(\d+)",
re.MULTILINE)
DIST_RE = re.compile(r"DISTANCE: mean=([\d.]+)")
ROWS_RE = re.compile(r"rows=(\d+)")
def read_text(path):
try:
with open(path, "r", errors="replace") as fh:
return fh.read()
except OSError:
return ""
def parse_events(path):
out = []
try:
with open(path, "r", errors="replace") as fh:
for line in fh:
line = line.strip()
if not line:
continue
try:
out.append(json.loads(line))
except json.JSONDecodeError:
continue # a partial line from a killed battle
except OSError:
out = []
return out
def parse_counters(text):
m = COUNTERS_RE.search(text)
if not m:
return None
return {"scans": int(m.group(1)), "fired": int(m.group(2)),
"hits": int(m.group(3)), "misses": int(m.group(4)),
"hit_bullets": int(m.group(5)), "hits_taken": int(m.group(6))}
def parse_first_places(text):
for m in FIRSTPLACES_RE.finditer(text):
if m.group(1) == BOT_NAME:
return int(m.group(3))
return None
def parse_rounds(path):
try:
with open(path) as fh:
return len(json.load(fh).get("rounds", []))
except (OSError, json.JSONDecodeError):
return None
def attribute_subject(evs, counters):
"""Which event `owner` id is our bot? Matched on fire/hit/hits-taken counts,
never on fired power (the power policy is continuous).
Returns (subject_id, other_id, how). `subject_id` is None only when nothing
could be identified; `other_id` is None when the opponent never fired a
single bullet (then the incoming hit rate is simply undefined, and the run
is still a valid damage measurement)."""
fires, hits, victim_hits = {}, {}, {}
owner_ids = set()
for ev in evs:
t = ev.get("type")
o = ev.get("owner")
if o is not None:
owner_ids.add(o)
if t == "fire":
fires[o] = fires.get(o, 0) + 1
elif t == "hit":
hits[o] = hits.get(o, 0) + 1
v = ev.get("victim")
if v is not None:
victim_hits[v] = victim_hits.get(v, 0) + 1
if not fires:
return None, None, "no fire events"
def other_of(subj):
others = [o for o in owner_ids if o != subj]
return others[0] if len(others) == 1 else None
if counters:
strict = [o for o in fires
if fires[o] == counters["fired"]
and hits.get(o, 0) == counters["hits"]
and victim_hits.get(o, 0) == counters["hits_taken"]]
if len(strict) == 1:
return strict[0], other_of(strict[0]), "exact"
cand = [o for o in fires if counters and fires[o] == counters["fired"]]
if len(cand) != 1:
cand = [o for o in fires
if counters and victim_hits.get(o, 0) == counters["hits_taken"]]
if len(cand) != 1:
cand = list(fires)
if len(cand) == 1:
return cand[0], other_of(cand[0]), "fires-only"
return None, None, "ambiguous"
def liveness(arm_env_tokens, stdout_path):
"""Liveness: every declared env token must appear verbatim in OUR bot's own
boot environment report, so an arm whose setting never reached the process
is a loud FAIL instead of a plausible-looking number. A TR_MOVEMENT the arm
did NOT declare is fatal too (the baseline would be contaminated)."""
text = read_text(stdout_path)
if not text:
return False, "no boot env report", False
missing = [t for t in arm_env_tokens if t not in text]
declared_keys = {t.split("=", 1)[0] for t in arm_env_tokens}
leaked_movement = ("TR_MOVEMENT" not in declared_keys
and re.search(r"^\[env\] TR_MOVEMENT=", text, re.MULTILINE)
is not None)
ok = (not missing) and (not leaked_movement)
why = []
if missing:
why.append("not seen in bot env report: " + ", ".join(missing))
if leaked_movement:
why.append("undeclared TR_MOVEMENT leaked into the process")
return ok, ("; ".join(why) if why else "ok"), bool(leaked_movement)
def parse_run(opp_dir, arm_dir, run):
"""One run -> dict of MEASURED numbers, or ok=False with a reason."""
base = os.path.join(opp_dir, arm_dir)
log_path = os.path.join(base, f"run{run}.battle.log")
ev_path = os.path.join(base, f"run{run}.events.jsonl")
rounds_path = os.path.join(base, f"run{run}.jsonl.rounds.json")
log = read_text(log_path)
out = {"run": run, "ok": False, "reason": "no capture",
"damage": 0.0, "damage_taken": 0.0, "wins": None, "rounds": None,
"opp_fired": 0, "opp_hits": 0, "dist": None, "scans": 0,
"liveness": "n/a"}
if not log:
return out
counters = parse_counters(log)
if counters is None:
out["reason"] = "battle never started (no subject counters)"
return out
evs = parse_events(ev_path)
subj, other, how = attribute_subject(evs, counters)
if subj is None:
out["reason"] = f"owner attribution failed ({how})"
return out
dmg = dmg_taken = 0.0
opp_fired = 0
for ev in evs:
t = ev.get("type")
if t == "fire" and ev.get("owner") == other:
opp_fired += 1
elif t == "hit":
d = float(ev.get("damage", 0.0))
if ev.get("owner") == subj:
dmg += d
elif ev.get("owner") == other:
dmg_taken += d
opp_hits = sum(1 for ev in evs
if ev.get("type") == "hit" and ev.get("owner") == other)
wins = parse_first_places(log)
if wins is None:
out["reason"] = "no final standings in the capture"
return out
m = DIST_RE.search(log)
out.update({
"ok": True, "reason": "ok",
"counters": counters, "subject_id": subj, "owner_attribution": how,
"damage": dmg, "damage_taken": dmg_taken,
"wins": wins, "rounds": parse_rounds(rounds_path),
"opp_fired": opp_fired, "opp_hits": opp_hits,
"dist": float(m.group(1)) if m else None,
"scans": counters["scans"],
})
return out
# ── arm/opponent aggregation ─────────────────────────────────────────────────
def arm_metrics(runs):
"""Aggregate a list of valid run dicts into one metric dict."""
n = len(runs)
total_rounds = sum(r["rounds"] or 0 for r in runs)
opp_fired = sum(r["opp_fired"] for r in runs)
opp_hits = sum(r["opp_hits"] for r in runs)
dists = [r["dist"] for r in runs if r["dist"] is not None]
return {
"runs": n,
"damage": mean([r["damage"] for r in runs]),
"damage_taken": mean([r["damage_taken"] for r in runs]),
"wins": mean([r["wins"] for r in runs]),
"win_rate": (sum(r["wins"] for r in runs) / total_rounds
if total_rounds else float("nan")),
"rounds": total_rounds,
"hit_rate": (100.0 * opp_hits / opp_fired) if opp_fired else float("nan"),
"dist": mean(dists) if dists else float("nan"),
"scans": mean([r["scans"] for r in runs]),
}
def collect(session):
"""-> (data, notes); data[opp][arm] = {"valid": [...], "invalid": [...]}"""
data = {}
notes = []
for opp in session["opponents"]:
oname = opp["name"]
data[oname] = {}
opp_dir = os.path.join(session["outdir"], oname)
for arm in session["arms"]:
aname = arm["name"]
tokens = arm["env"].split() if arm["env"] else []
node = {"valid": [], "invalid": [], "tokens": tokens,
"style": opp.get("style", ""), "label": arm.get("label", "")}
for r in range(1, session["runs"] + 1):
rr = parse_run(opp_dir, aname, r)
live_ok, live_why, fatal_leak = liveness(
tokens, os.path.join(opp_dir, aname, f"run{r}.bot.stdout.log"))
rr["liveness"] = live_why
if not rr["ok"]:
node["invalid"].append(rr)
elif not live_ok:
rr["reason"] = "liveness FAIL: " + live_why
node["invalid"].append(rr)
else:
node["valid"].append(rr)
data[oname][aname] = node
return data, notes
# ── the report ───────────────────────────────────────────────────────────────
def fmt(x, nd=1):
return "n/a" if x is None or (isinstance(x, float) and math.isnan(x)) \
else f"{x:.{nd}f}"
def fmt_s(x, nd=1):
return "n/a" if x is None or (isinstance(x, float) and math.isnan(x)) \
else f"{x:+.{nd}f}"
def main():
args = sys.argv[1:]
if not args:
print(__doc__)
return 2
session_dir = args[0]
ref = None
if "--reference" in args:
ref = args[args.index("--reference") + 1]
try:
with open(os.path.join(session_dir, "session.json")) as fh:
session = json.load(fh)
except (OSError, json.JSONDecodeError) as exc:
print(f"ERROR: cannot read {session_dir}/session.json: {exc}", file=sys.stderr)
return 2
session["outdir"] = session_dir
arms = [a["name"] for a in session["arms"]]
if ref is None:
ref = session.get("reference") or arms[0]
if ref not in arms:
print(f"ERROR: reference arm '{ref}' not in session arms {arms}",
file=sys.stderr)
return 2
opps = [o["name"] for o in session["opponents"]]
style_of = {o["name"]: o.get("style", "") for o in session["opponents"]}
data, _ = collect(session)
out = []
def p(s=""):
out.append(s)
print(s)
p("### MEASURED: session")
p()
p(f"* commit `{session['commit']}`, frozen binary sha256 `{session['binary_sha256'][:12]}…`")
p(f"* {len(opps)} opponents × {len(arms)} arms × {session['runs']} runs × "
f"{session['rounds']} rounds = {len(opps) * len(arms) * session['runs']} battles, "
f"conc={session.get('conc', '?')}")
p(f"* arms file `{os.path.basename(session.get('arms_file', '?'))}`, "
f"panel file `{os.path.basename(session.get('panel_file', '?'))}`")
p(f"* reference arm: **`{ref}`** — every delta below is (arm − {ref}), "
f"opponent by opponent")
p()
invalid = [(o, a, r) for o in opps for a in arms for r in data[o][a]["invalid"]]
p(f"* liveness: {len(invalid)} run(s) excluded "
f"({len(opps) * len(arms) * session['runs']} total)")
for o, a, r in invalid[:20]:
p(f" * `{o}/{a}` run{r['run']}: {r['reason']}")
if len(invalid) > 20:
p(f" * … and {len(invalid) - 20} more")
p()
# ── per-opponent × per-arm deltas ───────────────────────────────────────
per_opp = {a: {} for a in arms}
for o in opps:
ref_runs = data[o][ref]["valid"]
ref_m = arm_metrics(ref_runs) if ref_runs else None
for a in arms:
m = arm_metrics(data[o][a]["valid"]) if data[o][a]["valid"] else None
if m is None or ref_m is None:
per_opp[a][o] = None
continue
per_opp[a][o] = {
"m": m, "ref": ref_m,
"d_damage": m["damage"] - ref_m["damage"],
"d_wins": m["wins"] - ref_m["wins"],
"d_damage_taken": m["damage_taken"] - ref_m["damage_taken"],
"d_hit_rate": m["hit_rate"] - ref_m["hit_rate"],
"d_dist": m["dist"] - ref_m["dist"],
}
# ── per-arm tables ──────────────────────────────────────────────────────
p("### MEASURED: per-opponent paired table (per arm)")
p()
for a in arms:
lbl = next((x.get("label") for x in session["arms"] if x["name"] == a), "")
cnt = sum(1 for o in opps if per_opp[a][o])
p(f"#### `{a}`" + (f" — {lbl}" if lbl else "") + f" (paired on {cnt} opponents)")
p()
p("| opponent | style | dmg/run ref→arm | Δdmg | wins/run ref→arm | Δwins | Δdmg taken | Δhit rate (pp) | dist ref→arm |")
p("|---|---|---:|---:|---:|---:|---:|---:|---:|")
for o in opps:
e = per_opp[a][o]
if e is None:
p(f"| {o} | {style_of[o]} | n/a | n/a | n/a | n/a | n/a | n/a | n/a |")
continue
m, rm = e["m"], e["ref"]
p(f"| {o} | {style_of[o]} | {fmt(rm['damage'])}→{fmt(m['damage'])} "
f"| {fmt_s(e['d_damage'])} "
f"| {fmt(rm['wins'], 2)}→{fmt(m['wins'], 2)} | {fmt_s(e['d_wins'], 2)} "
f"| {fmt_s(e['d_damage_taken'])} | {fmt_s(e['d_hit_rate'], 2)} "
f"| {fmt(rm['dist'], 0)}→{fmt(m['dist'], 0)} |")
p()
# ── aggregate dashboard, pooled over every valid run ────────────────────
p("### MEASURED: pooled dashboard (all valid runs, NOT the verdict)")
p()
p("| arm | runs | dmg/run | dmg taken/run | wins/run | round wins | win rate | incoming hit rate | mean distance |")
p("|---|---:|---:|---:|---:|---:|---:|---:|---:|")
pooled = {}
for a in arms:
runs = [r for o in opps for r in data[o][a]["valid"]]
if not runs:
p(f"| `{a}` | 0 | n/a | n/a | n/a | n/a | n/a | n/a | n/a |")
continue
m = arm_metrics(runs)
pooled[a] = m
p(f"| `{a}` | {m['runs']} | {fmt(m['damage'])} | {fmt(m['damage_taken'])} "
f"| {fmt(m['wins'], 2)} | {int(sum(r['wins'] for r in runs))}/{m['rounds']} "
f"| {fmt(100 * m['win_rate'])}% | {fmt(m['hit_rate'], 2)}% "
f"| {fmt(m['dist'], 0)} |")
p()
# ── cross-opponent aggregation: mean delta, spread, sign tests, MDE ─────
p("### MEASURED: cross-opponent aggregation (the verdict layer)")
p()
p("Deltas are per-opponent (arm − reference). `spread` is the SD of those "
"deltas ACROSS opponents; `SE` = spread/√n; `95% CI` = mean ± t·SE. "
"Sign test = how many opponents the arm wins (ties dropped), exact "
"binomial; sign-flip = permutation test on the mean of the deltas.")
p()
p("| arm | metric | mean Δ | spread (SD) | SE | 95% CI | sign test (wins/n) | p(sign) | p(sign-flip) | Wilcoxon p | MDE |")
p("|---|---|---:|---:|---:|---|---:|---:|---:|---:|---:|")
stats = {}
for a in arms:
if a == ref:
continue
ds = [per_opp[a][o] for o in opps if per_opp[a][o]]
st = {"n": len(ds)}
for key, mkey in (("d_damage", "damage"), ("d_wins", "wins"),
("d_damage_taken", "damage_taken"),
("d_hit_rate", "hit_rate"), ("d_dist", "dist")):
vals = [e[key] for e in ds if not math.isnan(e[key])]
if len(vals) < 2:
continue
mu = mean(vals)
s = sd(vals)
se = s / math.sqrt(len(vals))
crit = t_crit(len(vals) - 1)
nz = [v for v in vals if v != 0.0]
wins_sign = sum(1 for v in nz if v > 0)
ps = binom_two_sided(wins_sign, len(nz))
pf, method = signflip_p(vals)
pw, _ = wilcoxon_p(vals)
st[key] = {"mean": mu, "sd": s, "se": se,
"ci": (mu - crit * se, mu + crit * se),
"k": wins_sign, "nz": len(nz), "p_sign": ps,
"p_flip": pf, "p_flip_method": method, "p_wilcox": pw,
"mde": mde(vals)}
p(f"| `{a}` | {mkey} | {fmt_s(mu, 2)} | {fmt(s, 2)} | {fmt(se, 2)} "
f"| [{fmt_s(mu - crit * se, 2)}, {fmt_s(mu + crit * se, 2)}] "
f"| {wins_sign}/{len(nz)} | {ps:.4g} | {pf:.4g} ({method}) "
f"| {pw:.4g} | {fmt(st[key]['mde'], 2)} |")
stats[a] = st
p()
# ── style breakdown (leg/governance only) ───────────────────────────────
styles = sorted({style_of[o] for o in opps if style_of[o]})
if len(styles) > 1:
p("#### By inferred style (explanation only, never the verdict)")
p()
p("| arm | style | n | mean Δdmg | mean Δwins | mean Δhit rate (pp) |")
p("|---|---|---:|---:|---:|---:|")
for a in arms:
if a == ref:
continue
for s in styles:
ds = [per_opp[a][o] for o in opps
if per_opp[a][o] and style_of[o] == s]
if not ds:
continue
p(f"| `{a}` | {s} | {len(ds)} "
f"| {fmt_s(mean([e['d_damage'] for e in ds]))} "
f"| {fmt_s(mean([e['d_wins'] for e in ds]), 2)} "
f"| {fmt_s(mean([e['d_hit_rate'] for e in ds]), 2)} |")
p()
# ── the pre-registered verdict rules ────────────────────────────────────
p("### The pre-registered verdict (rules fixed in `docs/movement_campaign.md`)")
p()
p("PRIMARY metrics are dmg/run and wins/run; hit rate is never the verdict. "
"The pre-registered rule says an arm is BETTER when one primary metric is "
"UP at sign-test p<0.05 `while the other does not go down`. That phrase has "
"two readings and BOTH are printed:")
p()
p("* **strict** — the other metric's mean delta is not negative at all "
"(`Δ >= 0`). Nothing can be BETTER while it costs *any* mean damage.")
p("* **substantive** — the other metric's delta is not *detectably* down: "
"the sign test is not significant **and** the delta is smaller than that "
"metric's MDE (the pre-registered rule 3 says an effect under the MDE is "
"not detectable, so it cannot count as a loss).")
p()
ranked = []
for a in arms:
if a == ref:
continue
st = stats.get(a, {})
d, w = st.get("d_damage"), st.get("d_wins")
if d is None or w is None:
ranked.append((a, "n/a", "n/a", float("nan"), float("nan")))
continue
up_d = d["mean"] > 0 and d["p_sign"] < 0.05
dn_d = d["mean"] < 0 and d["p_sign"] < 0.05
up_w = w["mean"] > 0 and w["p_sign"] < 0.05
dn_w = w["mean"] < 0 and w["p_sign"] < 0.05
def not_down(o):
return o["mean"] >= 0
def not_down_subst(o):
return o["mean"] > -o["mde"] and o["p_sign"] >= 0.05
def not_up_subst(o):
return o["mean"] < o["mde"] and o["p_sign"] >= 0.05
if (up_d and not_down(w)) or (up_w and not_down(d)):
strict = "BETTER"
elif (dn_d and w["mean"] <= 0) or (dn_w and d["mean"] <= 0):
strict = "WORSE"
else:
strict = "not distinguishable"
if (up_d and not_down_subst(w)) or (up_w and not_down_subst(d)):
subst = "BETTER"
elif (dn_d and not_up_subst(w)) or (dn_w and not_up_subst(d)):
subst = "WORSE"
else:
subst = "not distinguishable"
ranked.append((a, strict, subst, w["mean"], d["mean"]))
ranked.sort(key=lambda t: (-(t[3] if t[3] == t[3] else -1e9),
-(t[4] if t[4] == t[4] else -1e9)))
p("| rank | arm | Δwins/run | Δdmg/run | sign test wins | sign test dmg | verdict (strict) | verdict (substantive) |")
p("|---:|---|---:|---:|---|---|---|---|")
for i, (a, strict, subst, w, d) in enumerate(ranked, 1):
st = stats.get(a, {})
dd = st.get("d_damage", {})
ww = st.get("d_wins", {})
p(f"| {i} | `{a}` | {fmt_s(w, 2)} | {fmt_s(d)} "
f"| {ww.get('k', 'n/a')}/{ww.get('nz', 'n/a')} p={ww.get('p_sign', float('nan')):.4g} "
f"| {dd.get('k', 'n/a')}/{dd.get('nz', 'n/a')} p={dd.get('p_sign', float('nan')):.4g} "
f"| **{strict}** | **{subst}** |")
p()
p(f"Reference `{ref}`: {fmt(pooled.get(ref, {}).get('damage'))} dmg/run, "
f"{fmt(pooled.get(ref, {}).get('wins'), 2)} wins/run, "
f"{fmt(pooled.get(ref, {}).get('hit_rate'), 2)}% incoming, "
f"{fmt(pooled.get(ref, {}).get('dist'), 0)} px.")
best = ranked[0] if ranked else None
if best:
p()
p(f"Highest wins delta: `{best[0]}` ({fmt_s(best[3], 2)} wins/run, "
f"{fmt_s(best[4])} dmg/run) — strict: **{best[1]}**, "
f"substantive: **{best[2]}**.")
return 0
if __name__ == "__main__":
sys.exit(main())