Files
SirStone d93ce444c0 BitBrain verdict: clean negative at 30 runs/arm; analyzer gets MC + Mann-Whitney + MDE
- docs/bitbrain_gun_verdict.md: control vs bb_decay (decay SBC memory) vs a
  provably-zero placebo, 30 runs/arm vs real DrussGT. Nothing separates
  (bb_decay +3.3 dmg/run, p=0.71; round wins 97/210 vs 97/210, p=1.00); the
  7-run shape does not replicate. TR_BITBRAIN_RANGE=0 is clamped to 1.0 deg
  (bitbrain_gun.nim:207) so it is NOT a zero-shift placebo; TR_BITBRAIN_MIN_OBS
  unreachable is used instead.
- tools/ab/ab_analyze.py: keep exact enumeration for C(n,na)<=20e6 (7v7), add
  a seeded Monte-Carlo permutation test (1e6 draws, 0x5eed5eed) with its
  standard error, a tie-corrected Mann-Whitney U cross-check, a minimum
  detectable effect line, all-pairs comparisons, and a [bb] shift check.
- tools/ab/README.md: document the new analyzer output.
2026-09-24 23:02:59 +02:00

664 lines
27 KiB
Python
Executable File
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""ab_analyze.py — read a session dir produced by ab_run.sh and print the report.
python3 tools/ab/ab_analyze.py <session_dir> [--reference ARM]
Standard library only, deterministic. For every arm it prints runs, damage/run,
damage taken/run, ROUND WINS, shots/run, hits taken/run and the PER-RUN values
(wins cluster at 0/7 and single-run damage swings ~200, so the mean alone lies).
Tests (all two-sided, on PER-RUN values unless stated):
* permutation test on the difference of means. Full enumeration when
C(n, na) <= EXACT_CAP (7v7 -> C(14,7)=3432, always exact); otherwise a
Monte-Carlo permutation test with a fixed MC_DRAWS draws and the fixed
seed MC_SEED, reported with its Monte-Carlo standard error. Every p-value
says which method produced it (ncomb is C(60,30) ~ 1.2e17 at 30v30).
* Mann-Whitney U (rank-sum) cross-check, normal approximation with the
standard tie correction and a continuity correction.
* minimum detectable effect for n-per-arm (alpha=0.05 two-sided, 80% power)
derived from the observed pooled per-run SD, so a null result can be told
apart from an under-powered one.
* a round-level Fisher exact test on pooled rounds, clearly labelled
anti-conservative (rounds cluster within runs).
Round-win attribution comes from the events sidecar (the bot that does NOT die
wins the round) and is cross-checked against the runner's `firstPlaces`, which
is name-based. `firstPlaces` is the run-level truth: the per-round lines in
*.results.json are CUMULATIVE standings, not round winners — do not use them.
Liveness: each arm's declared env vars must appear verbatim in the bot's own
boot environment report (<arm>/run<N>.bot.stdout.log, `[env] VAR=VALUE`), so an
arm whose setting never reached the process is a loud FAIL rather than a
plausible-looking number.
Owner attribution never relies on fired-power values (the power policy fires a
continuous 0.15–1.15 range). The subject (DrussGT) is identified by matching
the events sidecar's per-owner fire/hit counts against the capture's own
`subject event counts:` line; ModularBot is the other bot.
"""
import glob
import itertools
import json
import math
import os
import random
import re
import sys
# beyond this many combinations we sample (deterministic seed) and say so;
# 7v7 = C(14,7) = 3432 is always exact.
EXACT_CAP = 20_000_000
# Monte-Carlo permutation test: fixed draw count and fixed seed, so the p-value
# is reproducible byte-for-byte across runs of this analyzer.
MC_DRAWS = 1_000_000
MC_SEED = 0x5EED5EED
# z_{0.975} + z_{0.80}, the constant in the two-sample MDE at alpha=0.05 (two
# sided) and 80% power: MDE = 2.8016 * sd * sqrt(2/n).
Z_ALPHA_POWER = 1.959963984540054 + 0.8416212335729143
BOT_NAME = "ModularBot"
SUBJECT_NAME = "DrussGT" # the capture subject (our bot is the adversary)
# ── loading ──────────────────────────────────────────────────────────────────
def load_session(outdir):
p = os.path.join(outdir, "session.json")
if os.path.exists(p):
try:
return json.load(open(p))
except (json.JSONDecodeError, OSError):
return None
return None
def parse_envspec(envspec):
"""`TR_A=1 TR_B=2` -> {'TR_A': '1', 'TR_B': '2'}."""
out = {}
for tok in (envspec or "").split():
if "=" in tok:
k, v = tok.split("=", 1)
out[k] = v
return out
def discover_arms(outdir, session):
arms = []
if session and isinstance(session.get("arms"), list):
for a in session["arms"]:
name = a.get("name")
if not name:
continue
arms.append({"name": name,
"env": parse_envspec(a.get("env", "")),
"label": a.get("label", "")})
if arms:
return arms
# fallback: any subdir holding run*.jsonl
for d in sorted(os.listdir(outdir)):
full = os.path.join(outdir, d)
if os.path.isdir(full) and discover_runs(full):
arms.append({"name": d, "env": {}, "label": ""})
return arms
def discover_runs(armdir):
runs = []
for f in os.listdir(armdir):
m = re.fullmatch(r"run(\d+)\.jsonl", f)
if m:
runs.append(int(m.group(1)))
return sorted(runs)
def read_lines(path):
try:
with open(path, errors="replace") as fh:
return fh.readlines()
except OSError:
return []
def parse_events(path):
evs = []
for line in read_lines(path):
line = line.strip()
if not line:
continue
try:
evs.append(json.loads(line))
except json.JSONDecodeError:
continue # partial line from an interrupted battle
return evs
def subject_counters(log_text):
m = re.search(
r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)"
r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)", log_text)
if not m:
return None
return {"fired": int(m.group(2)), "hits": int(m.group(3)),
"hits_taken": int(m.group(6))}
def parse_round_wins_by_name(log_text):
"""ModularBot's `firstPlaces` from the runner's final standings block."""
for m in re.finditer(
r"^\s*#\d+\s+(\S+)\s+totalScore=-?\d+\s+firstPlaces=(\d+)", log_text,
re.MULTILINE):
if m.group(1) == BOT_NAME:
return int(m.group(2))
return None
def parse_round_scores(path):
"""Per-round score delta per bot NAME from the cumulative *.results.json
lines. Used only to break ties (mutual kill / no death), never as the
primary attribution."""
try:
lines = json.load(open(path)).get("roundResults", [])
except (OSError, json.JSONDecodeError, AttributeError):
return {}
prev, out = {}, {}
for i, line in enumerate(lines, 1):
cur = {m.group(1): int(m.group(2))
for m in re.finditer(r"(\S+) rank=\d+ score=(-?\d+)", line)}
if not cur:
continue
out[i] = {name: sc - prev.get(name, 0) for name, sc in cur.items()}
prev = cur
return out
def attribute_owners(evs, counters):
"""Return (subject_id, other_id) using fire/hit counts, never powers."""
fires, hits, victim_hits = {}, {}, {}
for o in evs:
t = o.get("type")
if t == "fire":
fires[o["owner"]] = fires.get(o["owner"], 0) + 1
elif t == "hit":
hits[o["owner"]] = hits.get(o["owner"], 0) + 1
if "victim" in o:
victim_hits[o["victim"]] = victim_hits.get(o["victim"], 0) + 1
if not fires:
return None, None
if counters:
strict = [o for o in fires
if fires[o] == counters["fired"]
and hits.get(o, 0) == counters["hits"]
and victim_hits.get(o, 0) == counters["hits_taken"]]
if len(strict) == 1:
subj = strict[0]
return subj, _other(fires, subj)
# relaxed fallbacks (logged by caller via `counters is None` etc.)
cand = [o for o in fires if counters and fires[o] == counters["fired"]]
if len(cand) != 1:
cand = [o for o in fires
if counters and victim_hits.get(o, 0) == counters["hits_taken"]]
if len(cand) != 1:
cand = list(fires)
if len(cand) == 2:
return cand[0], cand[1]
if len(cand) == 1:
return cand[0], _other(fires, cand[0])
return None, None
def _other(owners, subj):
others = [o for o in owners if o != subj]
return others[0] if len(others) == 1 else None
def parse_run(armdir, run):
"""Everything the report needs for one run; None-free on partial data."""
evs = parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))
log_text = "".join(read_lines(os.path.join(armdir, f"run{run}.battle.log")))
counters = subject_counters(log_text)
subj, other = attribute_owners(evs, counters)
r = {"run": run, "rounds": 0, "wins": None, "first_places":
parse_round_wins_by_name(log_text), "subject_id": subj,
"other_id": other,
"attribution_exact": counters is not None, "shots": 0, "damage": 0.0,
"damage_taken": 0.0, "hits_taken": 0, "hits_dealt": 0,
"death_solved": 0, "death_score_agree": 0, "ambiguous": 0}
if subj is None or other is None:
# still count rounds so the arm shows up, but flags will be set
r["rounds"] = _round_count(armdir, run)
return r
deaths = {}
for o in evs:
t = o.get("type")
if t == "fire" and o.get("owner") == other:
r["shots"] += 1
elif t == "hit":
if o.get("owner") == other:
r["damage"] += o.get("damage", 0.0)
r["hits_dealt"] += 1
if o.get("victim") == other:
r["damage_taken"] += o.get("damage", 0.0)
r["hits_taken"] += 1
elif t == "death":
deaths.setdefault(o.get("round"), []).append(o.get("victim"))
nrounds = _round_count(armdir, run)
r["rounds"] = nrounds
scores = parse_round_scores(
os.path.join(armdir, f"run{run}.jsonl.results.json"))
# Per-round winner: primary = death events (a lone death means the OTHER bot
# won). A round with two or zero deaths (a mutual kill or a timeout) cannot
# be resolved from deaths alone, so the name-based per-round score delta is
# the tie-break. The two methods are cross-checked on the unambiguous rounds.
wins = death_solved = agree = ambiguous = 0
for rd in range(1, nrounds + 1):
vics = deaths.get(rd, [])
winner = None
if len(vics) == 1:
winner = other if vics[0] == subj else subj
death_solved += 1
delta = scores.get(rd)
if delta and SUBJECT_NAME in delta and BOT_NAME in delta:
score_winner = (subj if delta[SUBJECT_NAME] > delta[BOT_NAME]
else other)
if winner is None:
winner = score_winner
ambiguous += 1
elif winner == score_winner:
agree += 1
if winner == other:
wins += 1
r["wins"] = wins
r["death_solved"] = death_solved
r["death_score_agree"] = agree
r["ambiguous"] = ambiguous
return r
def _round_count(armdir, run):
p = os.path.join(armdir, f"run{run}.jsonl.rounds.json")
try:
return len(json.load(open(p)).get("rounds", []))
except (OSError, json.JSONDecodeError, AttributeError):
pass
# fall back to distinct round numbers in the events sidecar
rds = {o.get("round") for o in
parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))}
return len(rds)
# ── statistics ───────────────────────────────────────────────────────────────
def perm_test(xa, xb):
"""Two-sided permutation test on the difference of means.
Exact full enumeration when C(n, na) <= EXACT_CAP (7v7 is 3432), otherwise
a Monte-Carlo permutation test with MC_DRAWS fixed draws and the fixed seed
MC_SEED. Returns None if either arm is empty, else a dict:
obs observed |mean(xa) - mean(xb)| (what the p-value tests)
signed mean(xa) - mean(xb) (for display; negative means xb is larger)
p two-sided p-value
draws number of permutations (enumerated or sampled)
method 'exact' | 'monte-carlo'
se Monte-Carlo standard error of p (0.0 for the exact test)
seed MC seed (None for the exact test)
The MC p-value uses the (cnt + 1) / (B + 1) estimator, so it is never 0.
"""
na, nb = len(xa), len(xb)
if na == 0 or nb == 0:
return None
obs = abs(sum(xa) / na - sum(xb) / nb)
pooled = list(xa) + list(xb)
n = na + nb
total = sum(pooled)
ncomb = math.comb(n, na)
if ncomb <= EXACT_CAP:
cnt = 0
for combo in itertools.combinations(range(n), na):
sa = sum(pooled[i] for i in combo)
if abs(sa / na - (total - sa) / nb) >= obs - 1e-9:
cnt += 1
return {"obs": obs, "signed": sum(xa) / na - sum(xb) / nb,
"p": cnt / ncomb, "draws": ncomb,
"method": "exact", "se": 0.0, "seed": None}
rng = random.Random(MC_SEED)
sample = rng.sample
idx_range = range(n)
B = MC_DRAWS
cnt = 0
for _ in range(B):
sa = 0
for i in sample(idx_range, na):
sa += pooled[i]
if abs(sa / na - (total - sa) / nb) >= obs - 1e-9:
cnt += 1
p = (cnt + 1) / (B + 1)
se = math.sqrt(p * (1.0 - p) / (B + 1))
return {"obs": obs, "signed": sum(xa) / na - sum(xb) / nb,
"p": p, "draws": B, "method": "monte-carlo",
"se": se, "seed": MC_SEED}
def mannwhitney_p(xa, xb):
"""Two-sided Mann-Whitney U (rank-sum) test, normal approximation with the
standard tie correction and a continuity correction. Returns (p, U) or
None when either arm is empty. U is the smaller of U1, U2."""
na, nb = len(xa), len(xb)
n = na + nb
if na == 0 or nb == 0:
return None
vals = sorted([(v, 0) for v in xa] + [(v, 1) for v in xb])
rank = [0.0] * n
tie_term = 0.0
i = 0
while i < n:
j = i
while j + 1 < n and vals[j + 1][0] == vals[i][0]:
j += 1
t = j - i + 1
avg = (i + j) / 2.0 + 1.0
tie_term += t ** 3 - t
for k in range(i, j + 1):
rank[k] = avg
i = j + 1
r1 = sum(rank[k] for k in range(n) if vals[k][1] == 0)
u1 = r1 - na * (na + 1) / 2.0
mu = na * nb / 2.0
sigma2 = (na * nb / 12.0) * ((n + 1) - tie_term / (n * (n - 1)))
if sigma2 <= 0:
return (1.0, min(u1, na * nb - u1))
z = (abs(u1 - mu) - 0.5) / math.sqrt(sigma2)
if z < 0.0:
z = 0.0
p = math.erfc(z / math.sqrt(2.0))
return (p, min(u1, na * nb - u1))
def _var(x):
m = sum(x) / len(x)
return sum((v - m) ** 2 for v in x) / (len(x) - 1)
def min_detectable_effect(sd, n_per_arm):
"""Two-sample MDE at alpha=0.05 two-sided, 80% power, equal n."""
return Z_ALPHA_POWER * sd * math.sqrt(2.0 / n_per_arm)
def fisher_two_sided(a, b, c, d):
"""Two-sided Fisher exact on [[a,b],[c,d]]."""
n = a + b + c + d
if n == 0:
return None
row1, col1, col2 = a + b, a + c, b + d
denom = math.comb(n, row1)
if denom == 0:
return None
def hyper(x):
return math.comb(col1, x) * math.comb(col2, row1 - x) / denom
lo = max(0, col1 - (c + d))
hi = min(row1, col1)
obs = hyper(a)
return sum(hyper(x) for x in range(lo, hi + 1) if hyper(x) <= obs + 1e-12)
# ── liveness ─────────────────────────────────────────────────────────────────
def liveness(armdir, runs, env):
"""OK iff every declared env var appears verbatim in the bot's boot report
of every run, the report itself ran, and the bot did not warn that the name
is unrecognised (an unknown TR_*/GUN_* var is silently ignored)."""
if not runs:
return "FAIL", "no runs"
report_seen = {r: False for r in runs}
matched = {r: True for r in runs}
warned = {r: [] for r in runs}
for r in runs:
text = "".join(read_lines(os.path.join(
armdir, f"run{r}.bot.stdout.log")))
report_seen[r] = "=== ENVIRONMENT (boot report) ===" in text
for k, v in env.items():
if f"[env] {k}={v}" not in text:
matched[r] = False
if f"[env] WARNING: {k}" in text:
warned[r].append(k)
dead = [r for r in runs if not report_seen[r]]
miss = [r for r in runs if not matched[r]]
if dead:
return "FAIL", f"no bot env report in run(s) {dead}"
warn_runs = [r for r in runs if warned[r]]
if warn_runs:
names = sorted({k for r in warn_runs for k in warned[r]})
return "FAIL", (f"bot ignored arm env {names} (unrecognised) in "
f"run(s) {warn_runs}")
if env and miss:
vars_txt = " ".join(f"{k}={v}" for k, v in env.items())
return "FAIL", f"arm env '{vars_txt}' not in boot report of run(s) {miss}"
if not env:
return "OK", f"{len(runs)}/{len(runs)} runs: no arm env; report present"
vars_txt = " ".join(f"{k}={v}" for k, v in env.items())
return "OK", f"{len(runs)}/{len(runs)} runs: {vars_txt} applied"
# ── [bb] shift liveness ──────────────────────────────────────────────────────
_BB_SHIFT_RE = re.compile(r"\[bb\].*?shift=([+-]?\d+(?:\.\d+)?)deg")
def bb_shifts(armdir, runs):
"""All `[bb] ... shift=Xdeg` values in an arm's bot stdout (TR_BITBRAIN_LOG=1).
Returns (values, runs_with_lines, runs). A placebo that never applies a
correction emits ZERO [bb] lines; `bbLog` is only reached inside the same
`trained >= minObs` branch that computes the applied shift."""
vals = []
runs_with = 0
for r in runs:
text = "".join(read_lines(os.path.join(
armdir, f"run{r}.bot.stdout.log")))
found = _BB_SHIFT_RE.findall(text)
if found:
runs_with += 1
vals.extend(float(x) for x in found)
return vals, runs_with, len(runs)
# ── report ───────────────────────────────────────────────────────────────────
def main():
args = sys.argv[1:]
if not args or args[0] in ("-h", "--help"):
print(__doc__)
return 0 if args else 2
outdir = args[0]
reference = None
if "--reference" in args:
reference = args[args.index("--reference") + 1]
if not os.path.isdir(outdir):
print(f"ERROR: not a directory: {outdir}", file=sys.stderr)
return 2
session = load_session(outdir)
arms = discover_arms(outdir, session)
if not arms:
print(f"ERROR: no arms found in {outdir}", file=sys.stderr)
return 2
if reference is None:
reference = arms[0]["name"]
data = {}
for arm in arms:
armdir = os.path.join(outdir, arm["name"])
runs = discover_runs(armdir)
per = [parse_run(armdir, r) for r in runs]
data[arm["name"]] = {"dir": armdir, "runs": runs, "per": per,
"env": arm["env"], "label": arm["label"]}
if session:
print(f"# session {outdir}")
print(f"# commit={session.get('commit','?')} "
f"binary_sha256={session.get('binary_sha256','?')} "
f"rounds={session.get('rounds','?')} runs={session.get('runs','?')} "
f"conc={session.get('conc','?')} ts={session.get('timestamp','?')}")
print()
# ── summary table ────────────────────────────────────────────────────────
print("ARM SUMMARY")
hdr = (f"{'arm':<14} {'runs':>4} {'dmg/run':>8} {'dmgtk/run':>9} "
f"{'wins':>7} {'win%':>6} {'shots/run':>9} {'hitstk/run':>10}")
print(hdr)
print("-" * len(hdr))
for arm in arms:
name = arm["name"]
per = data[name]["per"]
n = len(per)
if n == 0:
print(f"{name:<14} {0:>4} (no runs)")
continue
dmg = sum(p["damage"] for p in per) / n
dmgv = sum(p["damage_taken"] for p in per) / n
shots = sum(p["shots"] for p in per) / n
htk = sum(p["hits_taken"] for p in per) / n
wins = sum(p["wins"] for p in per if p["wins"] is not None)
rounds = sum(p["rounds"] for p in per)
pct = f"{100.0 * wins / rounds:.1f}" if rounds else "n/a"
print(f"{name:<14} {n:>4} {dmg:>8.0f} {dmgv:>9.0f} "
f"{str(wins) + '/' + str(rounds):>7} {pct:>6} {shots:>9.0f} {htk:>10.1f}")
# ── per-run values ───────────────────────────────────────────────────────
print("\nPER-RUN (never just the mean)")
for arm in arms:
name = arm["name"]
per = data[name]["per"]
dmgs = " ".join(
f"r{p['run']}={p['damage']:.0f}" for p in per)
wins = " ".join(
f"r{p['run']}={p['wins']}/{p['rounds']}" if p["wins"] is not None
else f"r{p['run']}=?/{p['rounds']}" for p in per)
print(f" {name:<14} dmg: {dmgs}")
print(f" {'':<14} wins: {wins}")
# ── permutation tests (exact for 7v7, Monte-Carlo for 30v30) ─────────────
if reference not in data:
print(f"\nWARNING: reference arm '{reference}' not found; skipping tests")
return 1
ref = data[reference]["per"]
print("\nPAIRWISE PERMUTATION TEST (per-run values) + MANN-WHITNEY CROSS-CHECK")
print(f"permutation: exact when C(n,na) <= {EXACT_CAP:,}; otherwise "
f"Monte-Carlo {MC_DRAWS:,} draws, seed={MC_SEED:#x}, "
f"p = (cnt+1)/(B+1), se = sqrt(p(1-p)/(B+1))")
hdr = (f"{'metric':<11} {'A':<11} {'B':<11} {'diff(A-B)':>11} "
f"{'perm p':>9} {'method':<12} {'MC se':>7} {'MW p':>9} {'MW U':>8}")
print(hdr)
print("-" * len(hdr))
for i in range(len(arms)):
for j in range(i + 1, len(arms)):
ana, bnb = arms[i]["name"], arms[j]["name"]
for label, key in (("dmg/run", "damage"), ("round wins", "wins")):
xa = [p[key] for p in data[ana]["per"] if p[key] is not None]
xb = [p[key] for p in data[bnb]["per"] if p[key] is not None]
res = perm_test(xa, xb)
mw = mannwhitney_p(xa, xb)
if res is None:
print(f"{label:<11} {ana:<11} {bnb:<11} {'n/a':>11}")
continue
mstr = ("exact" if res["method"] == "exact"
else f"MC/B={res['draws']:,}")
sestr = "-" if res["method"] == "exact" else f"{res['se']:.4f}"
mwp = f"{mw[0]:.4f}" if mw else "n/a"
mwu = f"{mw[1]:.1f}" if mw else "n/a"
print(f"{label:<11} {ana:<11} {bnb:<11} {res['signed']:>+11.3f} "
f"{res['p']:>9.4f} {mstr:<12} {sestr:>7} {mwp:>9} {mwu:>8}")
# ── minimum detectable effect ──────────────────────────────────────────
print("\nMINIMUM DETECTABLE EFFECT (two-sample, alpha=0.05 two-sided, "
"80% power; MDE = 2.8016*sd*sqrt(2/n))")
print(f"{'metric':<11} {'n/arm':>6} {'sd(control)':>12} {'MDE(abs)':>10} "
f"{'MDE vs control mean':>22}")
print("-" * 64)
ctrl = data[reference]["per"]
for label, key in (("dmg/run", "damage"), ("round wins", "wins")):
vals = [p[key] for p in ctrl if p[key] is not None]
if len(vals) < 2:
continue
sd = math.sqrt(_var(vals))
mde = min_detectable_effect(sd, len(vals))
cmean = sum(vals) / len(vals)
rel = (f"{100.0*mde/abs(cmean):.1f}% of {cmean:.1f}"
if cmean else "n/a")
print(f"{label:<11} {len(vals):>6} {sd:>12.3f} {mde:>10.3f} {rel:>22}")
# ── round-level test (anti-conservative) ─────────────────────────────────
print("\nROUND-LEVEL TEST (pooled rounds, Fisher exact) vs "
f"`{reference}` — ANTI-CONSERVATIVE: rounds cluster within runs")
print(f"{'arm':<14} {'ref wins':>10} {'arm wins':>10} {'p':>8}")
print("-" * 46)
refw = sum(p["wins"] for p in ref if p["wins"] is not None)
refr = sum(p["rounds"] for p in ref)
for arm in arms:
name = arm["name"]
if name == reference:
continue
per = data[name]["per"]
w = sum(p["wins"] for p in per if p["wins"] is not None)
rr = sum(p["rounds"] for p in per)
p = fisher_two_sided(w, rr - w, refw, refr - refw)
pstr = f"{p:.4f}" if p is not None else "n/a"
print(f"{name:<14} {str(refw) + '/' + str(refr):>10} "
f"{str(w) + '/' + str(rr):>10} {pstr:>8}")
# ── liveness ─────────────────────────────────────────────────────────────
print("\nLIVENESS (arm env applied in the bot's own boot report)")
lv_fail = 0
for arm in arms:
name = arm["name"]
status, why = liveness(data[name]["dir"], data[name]["runs"],
data[name]["env"])
if status == "FAIL":
lv_fail += 1
print(f" {name:<14} {status:<4} ({why})")
# ── [bb] applied-shift check (a placebo must apply exactly 0) ────────────
print("\n[bb] APPLIED-SHIFT CHECK (from bot stdout; needs TR_BITBRAIN_LOG=1). "
"A provably-zero placebo emits ZERO [bb] lines.")
print(f" {'arm':<14} {'runs w/log':>11} {'lines':>7} {'min':>9} "
f"{'max':>9} {'zeros':>6}")
for arm in arms:
name = arm["name"]
vals, runs_with, nruns = bb_shifts(data[name]["dir"], data[name]["runs"])
if not vals:
print(f" {name:<14} {runs_with:>4}/{nruns:<4} {0:>7} "
f"{'-':>9} {'-':>9} {'-':>6}")
else:
zeros = sum(1 for v in vals if v == 0.0)
print(f" {name:<14} {runs_with:>4}/{nruns:<4} {len(vals):>7} "
f"{min(vals):>+9.2f} {max(vals):>+9.2f} {zeros:>6}")
# ── round-win attribution cross-check ────────────────────────────────────
print("\nROUND-WIN ATTRIBUTION (events primary; score tie-break for "
"mutual-kill / timeout rounds)")
for arm in arms:
name = arm["name"]
runs_ok = runs_tot = 0
dsolve = dagree = damb = 0
for p in data[name]["per"]:
if p["wins"] is not None and p["first_places"] is not None:
runs_tot += 1
if p["wins"] == p["first_places"]:
runs_ok += 1
dsolve += p["death_solved"]
dagree += p["death_score_agree"]
damb += p["ambiguous"]
status = ("OK" if runs_tot and runs_ok == runs_tot
else ("n/a" if not runs_tot else "MISMATCH"))
print(f" {name:<14} wins==firstPlaces {runs_ok}/{runs_tot} runs {status}; "
f"single-death rounds agree with score {dagree}/{dsolve} "
f"({damb} tie-broken)")
return 1 if lv_fail else 0
if __name__ == "__main__":
sys.exit(main())