movement Batch 1: pure strafe (range tilt OFF) beats the shipped tfil on round wins across a 15-opponent panel
225 battles, one frozen binary, five env-only arms, the frozen panel, 0 invalid
runs. Paired per opponent vs the shipped tfil:
strafe_notilt wins/run +0.38 [CI +0.16,+0.60] 9/9 opponents p=0.0039
dmg/run -10.2 [CI -25.8,+5.5] p=0.61, MDE 20.4 (not detectable)
incoming hit rate 12.24% vs 18.17%, dmg taken 150 vs 200
strafe_325 wins/run +0.33 [CI +0.04,+0.63] 10/12 p=0.0386
ring dmg/run +31.2 [CI +11.5,+50.9] 13/15 p=0.0074, wins/run -0.04 (ns)
but hit rate 29.4% at 236 px: a damage/survival trade, not a win
ring_notemp indistinguishable from tfil on both primaries
Round wins in this harness are survival wins (in 216/219 attributable runs the
win count equals the rounds the opponent died in), and the winner takes ~1/3
fewer hits while fighting ~74 px farther out. The shipped tfil is last of five
on wins: the DrussGT-only picture did not generalize.
Also: tournament_analyze.py now prints BOTH readings of the pre-registered
'while the other does not go down' clause (strict: nothing is better;
substantive: the two strafe arms and ring are better on one metric each).
This commit is contained in:
@@ -586,9 +586,17 @@ def main():
|
||||
# ── the pre-registered verdict rules ────────────────────────────────────
|
||||
p("### The pre-registered verdict (rules fixed in `docs/movement_campaign.md`)")
|
||||
p()
|
||||
p("BETTER = one primary metric (dmg/run, wins/run) up at sign-test p<0.05 "
|
||||
"with the other not down; WORSE = the mirror image; otherwise NOT "
|
||||
"DISTINGUISHABLE. Hit rate is never the verdict.")
|
||||
p("PRIMARY metrics are dmg/run and wins/run; hit rate is never the verdict. "
|
||||
"The pre-registered rule says an arm is BETTER when one primary metric is "
|
||||
"UP at sign-test p<0.05 `while the other does not go down`. That phrase has "
|
||||
"two readings and BOTH are printed:")
|
||||
p()
|
||||
p("* **strict** — the other metric's mean delta is not negative at all "
|
||||
"(`Δ >= 0`). Nothing can be BETTER while it costs *any* mean damage.")
|
||||
p("* **substantive** — the other metric's delta is not *detectably* down: "
|
||||
"the sign test is not significant **and** the delta is smaller than that "
|
||||
"metric's MDE (the pre-registered rule 3 says an effect under the MDE is "
|
||||
"not detectable, so it cannot count as a loss).")
|
||||
p()
|
||||
ranked = []
|
||||
for a in arms:
|
||||
@@ -597,31 +605,47 @@ def main():
|
||||
st = stats.get(a, {})
|
||||
d, w = st.get("d_damage"), st.get("d_wins")
|
||||
if d is None or w is None:
|
||||
ranked.append((a, "n/a", float("nan"), float("nan")))
|
||||
ranked.append((a, "n/a", "n/a", float("nan"), float("nan")))
|
||||
continue
|
||||
up_d = d["mean"] > 0 and d["p_sign"] < 0.05
|
||||
dn_d = d["mean"] < 0 and d["p_sign"] < 0.05
|
||||
up_w = w["mean"] > 0 and w["p_sign"] < 0.05
|
||||
dn_w = w["mean"] < 0 and w["p_sign"] < 0.05
|
||||
if (up_d and w["mean"] >= 0) or (up_w and d["mean"] >= 0):
|
||||
verdict = "BETTER than reference"
|
||||
|
||||
def not_down(o):
|
||||
return o["mean"] >= 0
|
||||
|
||||
def not_down_subst(o):
|
||||
return o["mean"] > -o["mde"] and o["p_sign"] >= 0.05
|
||||
|
||||
def not_up_subst(o):
|
||||
return o["mean"] < o["mde"] and o["p_sign"] >= 0.05
|
||||
|
||||
if (up_d and not_down(w)) or (up_w and not_down(d)):
|
||||
strict = "BETTER"
|
||||
elif (dn_d and w["mean"] <= 0) or (dn_w and d["mean"] <= 0):
|
||||
verdict = "WORSE than reference"
|
||||
strict = "WORSE"
|
||||
else:
|
||||
verdict = "NOT DISTINGUISHABLE from reference"
|
||||
ranked.append((a, verdict, w["mean"], d["mean"]))
|
||||
ranked.sort(key=lambda t: (-(t[2] if t[2] == t[2] else -1e9),
|
||||
-(t[3] if t[3] == t[3] else -1e9)))
|
||||
p("| rank | arm | Δwins/run | Δdmg/run | sign test dmg | sign test wins | verdict |")
|
||||
p("|---:|---|---:|---:|---|---|---|")
|
||||
for i, (a, verdict, w, d) in enumerate(ranked, 1):
|
||||
strict = "not distinguishable"
|
||||
if (up_d and not_down_subst(w)) or (up_w and not_down_subst(d)):
|
||||
subst = "BETTER"
|
||||
elif (dn_d and not_up_subst(w)) or (dn_w and not_up_subst(d)):
|
||||
subst = "WORSE"
|
||||
else:
|
||||
subst = "not distinguishable"
|
||||
ranked.append((a, strict, subst, w["mean"], d["mean"]))
|
||||
ranked.sort(key=lambda t: (-(t[3] if t[3] == t[3] else -1e9),
|
||||
-(t[4] if t[4] == t[4] else -1e9)))
|
||||
p("| rank | arm | Δwins/run | Δdmg/run | sign test wins | sign test dmg | verdict (strict) | verdict (substantive) |")
|
||||
p("|---:|---|---:|---:|---|---|---|---|")
|
||||
for i, (a, strict, subst, w, d) in enumerate(ranked, 1):
|
||||
st = stats.get(a, {})
|
||||
dd = st.get("d_damage", {})
|
||||
ww = st.get("d_wins", {})
|
||||
p(f"| {i} | `{a}` | {fmt_s(w, 2)} | {fmt_s(d)} "
|
||||
f"| {dd.get('k', 'n/a')}/{dd.get('nz', 'n/a')} p={dd.get('p_sign', float('nan')):.4g} "
|
||||
f"| {ww.get('k', 'n/a')}/{ww.get('nz', 'n/a')} p={ww.get('p_sign', float('nan')):.4g} "
|
||||
f"| **{verdict}** |")
|
||||
f"| {dd.get('k', 'n/a')}/{dd.get('nz', 'n/a')} p={dd.get('p_sign', float('nan')):.4g} "
|
||||
f"| **{strict}** | **{subst}** |")
|
||||
p()
|
||||
p(f"Reference `{ref}`: {fmt(pooled.get(ref, {}).get('damage'))} dmg/run, "
|
||||
f"{fmt(pooled.get(ref, {}).get('wins'), 2)} wins/run, "
|
||||
@@ -630,8 +654,9 @@ def main():
|
||||
best = ranked[0] if ranked else None
|
||||
if best:
|
||||
p()
|
||||
p(f"Highest wins delta: `{best[0]}` ({fmt_s(best[2], 2)} wins/run, "
|
||||
f"{fmt_s(best[3])} dmg/run) — **{best[1]}**.")
|
||||
p(f"Highest wins delta: `{best[0]}` ({fmt_s(best[3], 2)} wins/run, "
|
||||
f"{fmt_s(best[4])} dmg/run) — strict: **{best[1]}**, "
|
||||
f"substantive: **{best[2]}**.")
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user