BitBrain verdict: clean negative at 30 runs/arm; analyzer gets MC + Mann-Whitney + MDE

- docs/bitbrain_gun_verdict.md: control vs bb_decay (decay SBC memory) vs a
  provably-zero placebo, 30 runs/arm vs real DrussGT. Nothing separates
  (bb_decay +3.3 dmg/run, p=0.71; round wins 97/210 vs 97/210, p=1.00); the
  7-run shape does not replicate. TR_BITBRAIN_RANGE=0 is clamped to 1.0 deg
  (bitbrain_gun.nim:207) so it is NOT a zero-shift placebo; TR_BITBRAIN_MIN_OBS
  unreachable is used instead.
- tools/ab/ab_analyze.py: keep exact enumeration for C(n,na)<=20e6 (7v7), add
  a seeded Monte-Carlo permutation test (1e6 draws, 0x5eed5eed) with its
  standard error, a tie-corrected Mann-Whitney U cross-check, a minimum
  detectable effect line, all-pairs comparisons, and a [bb] shift check.
- tools/ab/README.md: document the new analyzer output.
This commit is contained in:
2026-09-24 23:00:20 +02:00
parent f91e121965
commit d93ce444c0
3 changed files with 377 additions and 38 deletions
+178 -35
View File
@@ -7,10 +7,17 @@ Standard library only, deterministic. For every arm it prints runs, damage/run,
damage taken/run, ROUND WINS, shots/run, hits taken/run and the PER-RUN values
(wins cluster at 0/7 and single-run damage swings ~200, so the mean alone lies).
Tests:
* exact two-sided permutation test on PER-RUN values (damage/run and round
wins) vs a reference arm (default: the first arm). C(14,7)=3432 for 7v7 —
enumerated exactly, never sampled, whenever the combination count is small.
Tests (all two-sided, on PER-RUN values unless stated):
* permutation test on the difference of means. Full enumeration when
C(n, na) <= EXACT_CAP (7v7 -> C(14,7)=3432, always exact); otherwise a
Monte-Carlo permutation test with a fixed MC_DRAWS draws and the fixed
seed MC_SEED, reported with its Monte-Carlo standard error. Every p-value
says which method produced it (ncomb is C(60,30) ~ 1.2e17 at 30v30).
* Mann-Whitney U (rank-sum) cross-check, normal approximation with the
standard tie correction and a continuity correction.
* minimum detectable effect for n-per-arm (alpha=0.05 two-sided, 80% power)
derived from the observed pooled per-run SD, so a null result can be told
apart from an under-powered one.
* a round-level Fisher exact test on pooled rounds, clearly labelled
anti-conservative (rounds cluster within runs).
@@ -34,12 +41,20 @@ import itertools
import json
import math
import os
import random
import re
import sys
# beyond this many combinations we sample (deterministic seed) and say so;
# 7v7 = C(14,7) = 3432 is always exact.
EXACT_CAP = 20_000_000
# Monte-Carlo permutation test: fixed draw count and fixed seed, so the p-value
# is reproducible byte-for-byte across runs of this analyzer.
MC_DRAWS = 1_000_000
MC_SEED = 0x5EED5EED
# z_{0.975} + z_{0.80}, the constant in the two-sample MDE at alpha=0.05 (two
# sided) and 80% power: MDE = 2.8016 * sd * sqrt(2/n).
Z_ALPHA_POWER = 1.959963984540054 + 0.8416212335729143
BOT_NAME = "ModularBot"
SUBJECT_NAME = "DrussGT" # the capture subject (our bot is the adversary)
@@ -276,34 +291,98 @@ def _round_count(armdir, run):
# ── statistics ───────────────────────────────────────────────────────────────
def perm_test(xa, xb):
"""Exact two-sided permutation test on the difference of means. Returns
(obs, p, n_perm, exact_bool)."""
"""Two-sided permutation test on the difference of means.
Exact full enumeration when C(n, na) <= EXACT_CAP (7v7 is 3432), otherwise
a Monte-Carlo permutation test with MC_DRAWS fixed draws and the fixed seed
MC_SEED. Returns None if either arm is empty, else a dict:
obs observed |mean(xa) - mean(xb)| (what the p-value tests)
signed mean(xa) - mean(xb) (for display; negative means xb is larger)
p two-sided p-value
draws number of permutations (enumerated or sampled)
method 'exact' | 'monte-carlo'
se Monte-Carlo standard error of p (0.0 for the exact test)
seed MC seed (None for the exact test)
The MC p-value uses the (cnt + 1) / (B + 1) estimator, so it is never 0.
"""
na, nb = len(xa), len(xb)
if na == 0 or nb == 0:
return None
obs = abs(sum(xa) / na - sum(xb) / nb)
pooled = list(xa) + list(xb)
n = na + nb
total_sum = sum(pooled)
total = sum(pooled)
ncomb = math.comb(n, na)
if ncomb <= EXACT_CAP:
cnt = 0
for combo in itertools.combinations(range(n), na):
sa = sum(pooled[i] for i in combo)
if abs(sa / na - (total_sum - sa) / nb) >= obs - 1e-9:
if abs(sa / na - (total - sa) / nb) >= obs - 1e-9:
cnt += 1
return obs, cnt / ncomb, ncomb, True
# deterministic fallback for very large n (never hit at the default 7 runs)
import random
rng = random.Random(0xA1B2C3)
B = 200_000
return {"obs": obs, "signed": sum(xa) / na - sum(xb) / nb,
"p": cnt / ncomb, "draws": ncomb,
"method": "exact", "se": 0.0, "seed": None}
rng = random.Random(MC_SEED)
sample = rng.sample
idx_range = range(n)
B = MC_DRAWS
cnt = 0
for _ in range(B):
idx = rng.sample(range(n), na)
sa = sum(pooled[i] for i in idx)
if abs(sa / na - (total_sum - sa) / nb) >= obs - 1e-9:
sa = 0
for i in sample(idx_range, na):
sa += pooled[i]
if abs(sa / na - (total - sa) / nb) >= obs - 1e-9:
cnt += 1
return obs, (cnt + 1) / (B + 1), ncomb, False
p = (cnt + 1) / (B + 1)
se = math.sqrt(p * (1.0 - p) / (B + 1))
return {"obs": obs, "signed": sum(xa) / na - sum(xb) / nb,
"p": p, "draws": B, "method": "monte-carlo",
"se": se, "seed": MC_SEED}
def mannwhitney_p(xa, xb):
"""Two-sided Mann-Whitney U (rank-sum) test, normal approximation with the
standard tie correction and a continuity correction. Returns (p, U) or
None when either arm is empty. U is the smaller of U1, U2."""
na, nb = len(xa), len(xb)
n = na + nb
if na == 0 or nb == 0:
return None
vals = sorted([(v, 0) for v in xa] + [(v, 1) for v in xb])
rank = [0.0] * n
tie_term = 0.0
i = 0
while i < n:
j = i
while j + 1 < n and vals[j + 1][0] == vals[i][0]:
j += 1
t = j - i + 1
avg = (i + j) / 2.0 + 1.0
tie_term += t ** 3 - t
for k in range(i, j + 1):
rank[k] = avg
i = j + 1
r1 = sum(rank[k] for k in range(n) if vals[k][1] == 0)
u1 = r1 - na * (na + 1) / 2.0
mu = na * nb / 2.0
sigma2 = (na * nb / 12.0) * ((n + 1) - tie_term / (n * (n - 1)))
if sigma2 <= 0:
return (1.0, min(u1, na * nb - u1))
z = (abs(u1 - mu) - 0.5) / math.sqrt(sigma2)
if z < 0.0:
z = 0.0
p = math.erfc(z / math.sqrt(2.0))
return (p, min(u1, na * nb - u1))
def _var(x):
m = sum(x) / len(x)
return sum((v - m) ** 2 for v in x) / (len(x) - 1)
def min_detectable_effect(sd, n_per_arm):
"""Two-sample MDE at alpha=0.05 two-sided, 80% power, equal n."""
return Z_ALPHA_POWER * sd * math.sqrt(2.0 / n_per_arm)
def fisher_two_sided(a, b, c, d):
@@ -363,6 +442,28 @@ def liveness(armdir, runs, env):
return "OK", f"{len(runs)}/{len(runs)} runs: {vars_txt} applied"
# ── [bb] shift liveness ──────────────────────────────────────────────────────
_BB_SHIFT_RE = re.compile(r"\[bb\].*?shift=([+-]?\d+(?:\.\d+)?)deg")
def bb_shifts(armdir, runs):
"""All `[bb] ... shift=Xdeg` values in an arm's bot stdout (TR_BITBRAIN_LOG=1).
Returns (values, runs_with_lines, runs). A placebo that never applies a
correction emits ZERO [bb] lines; `bbLog` is only reached inside the same
`trained >= minObs` branch that computes the applied shift."""
vals = []
runs_with = 0
for r in runs:
text = "".join(read_lines(os.path.join(
armdir, f"run{r}.bot.stdout.log")))
found = _BB_SHIFT_RE.findall(text)
if found:
runs_with += 1
vals.extend(float(x) for x in found)
return vals, runs_with, len(runs)
# ── report ───────────────────────────────────────────────────────────────────
def main():
@@ -438,29 +539,55 @@ def main():
print(f" {name:<14} dmg: {dmgs}")
print(f" {'':<14} wins: {wins}")
# ── exact permutation tests ──────────────────────────────────────────────
# ── permutation tests (exact for 7v7, Monte-Carlo for 30v30) ─────────────
if reference not in data:
print(f"\nWARNING: reference arm '{reference}' not found; skipping tests")
return 1
print(f"\nEXACT TWO-SIDED PERMUTATION TEST (per-run values) vs `{reference}`")
print(f"{'metric':<12} {'arm':<14} {'obs(diff)':>12} {'p':>8} permutations")
print("-" * 64)
ref = data[reference]["per"]
for arm in arms:
name = arm["name"]
if name == reference:
print("\nPAIRWISE PERMUTATION TEST (per-run values) + MANN-WHITNEY CROSS-CHECK")
print(f"permutation: exact when C(n,na) <= {EXACT_CAP:,}; otherwise "
f"Monte-Carlo {MC_DRAWS:,} draws, seed={MC_SEED:#x}, "
f"p = (cnt+1)/(B+1), se = sqrt(p(1-p)/(B+1))")
hdr = (f"{'metric':<11} {'A':<11} {'B':<11} {'diff(A-B)':>11} "
f"{'perm p':>9} {'method':<12} {'MC se':>7} {'MW p':>9} {'MW U':>8}")
print(hdr)
print("-" * len(hdr))
for i in range(len(arms)):
for j in range(i + 1, len(arms)):
ana, bnb = arms[i]["name"], arms[j]["name"]
for label, key in (("dmg/run", "damage"), ("round wins", "wins")):
xa = [p[key] for p in data[ana]["per"] if p[key] is not None]
xb = [p[key] for p in data[bnb]["per"] if p[key] is not None]
res = perm_test(xa, xb)
mw = mannwhitney_p(xa, xb)
if res is None:
print(f"{label:<11} {ana:<11} {bnb:<11} {'n/a':>11}")
continue
mstr = ("exact" if res["method"] == "exact"
else f"MC/B={res['draws']:,}")
sestr = "-" if res["method"] == "exact" else f"{res['se']:.4f}"
mwp = f"{mw[0]:.4f}" if mw else "n/a"
mwu = f"{mw[1]:.1f}" if mw else "n/a"
print(f"{label:<11} {ana:<11} {bnb:<11} {res['signed']:>+11.3f} "
f"{res['p']:>9.4f} {mstr:<12} {sestr:>7} {mwp:>9} {mwu:>8}")
# ── minimum detectable effect ──────────────────────────────────────────
print("\nMINIMUM DETECTABLE EFFECT (two-sample, alpha=0.05 two-sided, "
"80% power; MDE = 2.8016*sd*sqrt(2/n))")
print(f"{'metric':<11} {'n/arm':>6} {'sd(control)':>12} {'MDE(abs)':>10} "
f"{'MDE vs control mean':>22}")
print("-" * 64)
ctrl = data[reference]["per"]
for label, key in (("dmg/run", "damage"), ("round wins", "wins")):
vals = [p[key] for p in ctrl if p[key] is not None]
if len(vals) < 2:
continue
for label, key in (("dmg/run", "damage"), ("round wins", "wins")):
xa = [p[key] for p in ref if p[key] is not None]
xb = [p[key] for p in data[name]["per"] if p[key] is not None]
res = perm_test(xa, xb)
if res is None:
print(f"{label:<12} {name:<14} {'n/a':>12} {'n/a':>8}")
continue
obs, p, ncomb, exact = res
note = f"C({len(xa)+len(xb)},{len(xa)})={ncomb}" + (
"" if exact else " SAMPLED")
print(f"{label:<12} {name:<14} {obs:>+12.3f} {p:>8.4f} {note}")
sd = math.sqrt(_var(vals))
mde = min_detectable_effect(sd, len(vals))
cmean = sum(vals) / len(vals)
rel = (f"{100.0*mde/abs(cmean):.1f}% of {cmean:.1f}"
if cmean else "n/a")
print(f"{label:<11} {len(vals):>6} {sd:>12.3f} {mde:>10.3f} {rel:>22}")
# ── round-level test (anti-conservative) ─────────────────────────────────
print("\nROUND-LEVEL TEST (pooled rounds, Fisher exact) vs "
@@ -492,6 +619,22 @@ def main():
lv_fail += 1
print(f" {name:<14} {status:<4} ({why})")
# ── [bb] applied-shift check (a placebo must apply exactly 0) ────────────
print("\n[bb] APPLIED-SHIFT CHECK (from bot stdout; needs TR_BITBRAIN_LOG=1). "
"A provably-zero placebo emits ZERO [bb] lines.")
print(f" {'arm':<14} {'runs w/log':>11} {'lines':>7} {'min':>9} "
f"{'max':>9} {'zeros':>6}")
for arm in arms:
name = arm["name"]
vals, runs_with, nruns = bb_shifts(data[name]["dir"], data[name]["runs"])
if not vals:
print(f" {name:<14} {runs_with:>4}/{nruns:<4} {0:>7} "
f"{'-':>9} {'-':>9} {'-':>6}")
else:
zeros = sum(1 for v in vals if v == 0.0)
print(f" {name:<14} {runs_with:>4}/{nruns:<4} {len(vals):>7} "
f"{min(vals):>+9.2f} {max(vals):>+9.2f} {zeros:>6}")
# ── round-win attribution cross-check ────────────────────────────────────
print("\nROUND-WIN ATTRIBUTION (events primary; score tie-break for "
"mutual-kill / timeout rounds)")