d93ce444c0
- docs/bitbrain_gun_verdict.md: control vs bb_decay (decay SBC memory) vs a provably-zero placebo, 30 runs/arm vs real DrussGT. Nothing separates (bb_decay +3.3 dmg/run, p=0.71; round wins 97/210 vs 97/210, p=1.00); the 7-run shape does not replicate. TR_BITBRAIN_RANGE=0 is clamped to 1.0 deg (bitbrain_gun.nim:207) so it is NOT a zero-shift placebo; TR_BITBRAIN_MIN_OBS unreachable is used instead. - tools/ab/ab_analyze.py: keep exact enumeration for C(n,na)<=20e6 (7v7), add a seeded Monte-Carlo permutation test (1e6 draws, 0x5eed5eed) with its standard error, a tie-corrected Mann-Whitney U cross-check, a minimum detectable effect line, all-pairs comparisons, and a [bb] shift check. - tools/ab/README.md: document the new analyzer output.
664 lines
27 KiB
Python
Executable File
664 lines
27 KiB
Python
Executable File
#!/usr/bin/env python3
|
||
"""ab_analyze.py — read a session dir produced by ab_run.sh and print the report.
|
||
|
||
python3 tools/ab/ab_analyze.py <session_dir> [--reference ARM]
|
||
|
||
Standard library only, deterministic. For every arm it prints runs, damage/run,
|
||
damage taken/run, ROUND WINS, shots/run, hits taken/run and the PER-RUN values
|
||
(wins cluster at 0/7 and single-run damage swings ~200, so the mean alone lies).
|
||
|
||
Tests (all two-sided, on PER-RUN values unless stated):
|
||
* permutation test on the difference of means. Full enumeration when
|
||
C(n, na) <= EXACT_CAP (7v7 -> C(14,7)=3432, always exact); otherwise a
|
||
Monte-Carlo permutation test with a fixed MC_DRAWS draws and the fixed
|
||
seed MC_SEED, reported with its Monte-Carlo standard error. Every p-value
|
||
says which method produced it (ncomb is C(60,30) ~ 1.2e17 at 30v30).
|
||
* Mann-Whitney U (rank-sum) cross-check, normal approximation with the
|
||
standard tie correction and a continuity correction.
|
||
* minimum detectable effect for n-per-arm (alpha=0.05 two-sided, 80% power)
|
||
derived from the observed pooled per-run SD, so a null result can be told
|
||
apart from an under-powered one.
|
||
* a round-level Fisher exact test on pooled rounds, clearly labelled
|
||
anti-conservative (rounds cluster within runs).
|
||
|
||
Round-win attribution comes from the events sidecar (the bot that does NOT die
|
||
wins the round) and is cross-checked against the runner's `firstPlaces`, which
|
||
is name-based. `firstPlaces` is the run-level truth: the per-round lines in
|
||
*.results.json are CUMULATIVE standings, not round winners — do not use them.
|
||
|
||
Liveness: each arm's declared env vars must appear verbatim in the bot's own
|
||
boot environment report (<arm>/run<N>.bot.stdout.log, `[env] VAR=VALUE`), so an
|
||
arm whose setting never reached the process is a loud FAIL rather than a
|
||
plausible-looking number.
|
||
|
||
Owner attribution never relies on fired-power values (the power policy fires a
|
||
continuous 0.15–1.15 range). The subject (DrussGT) is identified by matching
|
||
the events sidecar's per-owner fire/hit counts against the capture's own
|
||
`subject event counts:` line; ModularBot is the other bot.
|
||
"""
|
||
import glob
|
||
import itertools
|
||
import json
|
||
import math
|
||
import os
|
||
import random
|
||
import re
|
||
import sys
|
||
|
||
# beyond this many combinations we sample (deterministic seed) and say so;
|
||
# 7v7 = C(14,7) = 3432 is always exact.
|
||
EXACT_CAP = 20_000_000
|
||
# Monte-Carlo permutation test: fixed draw count and fixed seed, so the p-value
|
||
# is reproducible byte-for-byte across runs of this analyzer.
|
||
MC_DRAWS = 1_000_000
|
||
MC_SEED = 0x5EED5EED
|
||
# z_{0.975} + z_{0.80}, the constant in the two-sample MDE at alpha=0.05 (two
|
||
# sided) and 80% power: MDE = 2.8016 * sd * sqrt(2/n).
|
||
Z_ALPHA_POWER = 1.959963984540054 + 0.8416212335729143
|
||
BOT_NAME = "ModularBot"
|
||
SUBJECT_NAME = "DrussGT" # the capture subject (our bot is the adversary)
|
||
|
||
|
||
# ── loading ──────────────────────────────────────────────────────────────────
|
||
|
||
def load_session(outdir):
|
||
p = os.path.join(outdir, "session.json")
|
||
if os.path.exists(p):
|
||
try:
|
||
return json.load(open(p))
|
||
except (json.JSONDecodeError, OSError):
|
||
return None
|
||
return None
|
||
|
||
|
||
def parse_envspec(envspec):
|
||
"""`TR_A=1 TR_B=2` -> {'TR_A': '1', 'TR_B': '2'}."""
|
||
out = {}
|
||
for tok in (envspec or "").split():
|
||
if "=" in tok:
|
||
k, v = tok.split("=", 1)
|
||
out[k] = v
|
||
return out
|
||
|
||
|
||
def discover_arms(outdir, session):
|
||
arms = []
|
||
if session and isinstance(session.get("arms"), list):
|
||
for a in session["arms"]:
|
||
name = a.get("name")
|
||
if not name:
|
||
continue
|
||
arms.append({"name": name,
|
||
"env": parse_envspec(a.get("env", "")),
|
||
"label": a.get("label", "")})
|
||
if arms:
|
||
return arms
|
||
# fallback: any subdir holding run*.jsonl
|
||
for d in sorted(os.listdir(outdir)):
|
||
full = os.path.join(outdir, d)
|
||
if os.path.isdir(full) and discover_runs(full):
|
||
arms.append({"name": d, "env": {}, "label": ""})
|
||
return arms
|
||
|
||
|
||
def discover_runs(armdir):
|
||
runs = []
|
||
for f in os.listdir(armdir):
|
||
m = re.fullmatch(r"run(\d+)\.jsonl", f)
|
||
if m:
|
||
runs.append(int(m.group(1)))
|
||
return sorted(runs)
|
||
|
||
|
||
def read_lines(path):
|
||
try:
|
||
with open(path, errors="replace") as fh:
|
||
return fh.readlines()
|
||
except OSError:
|
||
return []
|
||
|
||
|
||
def parse_events(path):
|
||
evs = []
|
||
for line in read_lines(path):
|
||
line = line.strip()
|
||
if not line:
|
||
continue
|
||
try:
|
||
evs.append(json.loads(line))
|
||
except json.JSONDecodeError:
|
||
continue # partial line from an interrupted battle
|
||
return evs
|
||
|
||
|
||
def subject_counters(log_text):
|
||
m = re.search(
|
||
r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)"
|
||
r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)", log_text)
|
||
if not m:
|
||
return None
|
||
return {"fired": int(m.group(2)), "hits": int(m.group(3)),
|
||
"hits_taken": int(m.group(6))}
|
||
|
||
|
||
def parse_round_wins_by_name(log_text):
|
||
"""ModularBot's `firstPlaces` from the runner's final standings block."""
|
||
for m in re.finditer(
|
||
r"^\s*#\d+\s+(\S+)\s+totalScore=-?\d+\s+firstPlaces=(\d+)", log_text,
|
||
re.MULTILINE):
|
||
if m.group(1) == BOT_NAME:
|
||
return int(m.group(2))
|
||
return None
|
||
|
||
|
||
def parse_round_scores(path):
|
||
"""Per-round score delta per bot NAME from the cumulative *.results.json
|
||
lines. Used only to break ties (mutual kill / no death), never as the
|
||
primary attribution."""
|
||
try:
|
||
lines = json.load(open(path)).get("roundResults", [])
|
||
except (OSError, json.JSONDecodeError, AttributeError):
|
||
return {}
|
||
prev, out = {}, {}
|
||
for i, line in enumerate(lines, 1):
|
||
cur = {m.group(1): int(m.group(2))
|
||
for m in re.finditer(r"(\S+) rank=\d+ score=(-?\d+)", line)}
|
||
if not cur:
|
||
continue
|
||
out[i] = {name: sc - prev.get(name, 0) for name, sc in cur.items()}
|
||
prev = cur
|
||
return out
|
||
|
||
|
||
def attribute_owners(evs, counters):
|
||
"""Return (subject_id, other_id) using fire/hit counts, never powers."""
|
||
fires, hits, victim_hits = {}, {}, {}
|
||
for o in evs:
|
||
t = o.get("type")
|
||
if t == "fire":
|
||
fires[o["owner"]] = fires.get(o["owner"], 0) + 1
|
||
elif t == "hit":
|
||
hits[o["owner"]] = hits.get(o["owner"], 0) + 1
|
||
if "victim" in o:
|
||
victim_hits[o["victim"]] = victim_hits.get(o["victim"], 0) + 1
|
||
if not fires:
|
||
return None, None
|
||
if counters:
|
||
strict = [o for o in fires
|
||
if fires[o] == counters["fired"]
|
||
and hits.get(o, 0) == counters["hits"]
|
||
and victim_hits.get(o, 0) == counters["hits_taken"]]
|
||
if len(strict) == 1:
|
||
subj = strict[0]
|
||
return subj, _other(fires, subj)
|
||
# relaxed fallbacks (logged by caller via `counters is None` etc.)
|
||
cand = [o for o in fires if counters and fires[o] == counters["fired"]]
|
||
if len(cand) != 1:
|
||
cand = [o for o in fires
|
||
if counters and victim_hits.get(o, 0) == counters["hits_taken"]]
|
||
if len(cand) != 1:
|
||
cand = list(fires)
|
||
if len(cand) == 2:
|
||
return cand[0], cand[1]
|
||
if len(cand) == 1:
|
||
return cand[0], _other(fires, cand[0])
|
||
return None, None
|
||
|
||
|
||
def _other(owners, subj):
|
||
others = [o for o in owners if o != subj]
|
||
return others[0] if len(others) == 1 else None
|
||
|
||
|
||
def parse_run(armdir, run):
|
||
"""Everything the report needs for one run; None-free on partial data."""
|
||
evs = parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))
|
||
log_text = "".join(read_lines(os.path.join(armdir, f"run{run}.battle.log")))
|
||
counters = subject_counters(log_text)
|
||
subj, other = attribute_owners(evs, counters)
|
||
|
||
r = {"run": run, "rounds": 0, "wins": None, "first_places":
|
||
parse_round_wins_by_name(log_text), "subject_id": subj,
|
||
"other_id": other,
|
||
"attribution_exact": counters is not None, "shots": 0, "damage": 0.0,
|
||
"damage_taken": 0.0, "hits_taken": 0, "hits_dealt": 0,
|
||
"death_solved": 0, "death_score_agree": 0, "ambiguous": 0}
|
||
if subj is None or other is None:
|
||
# still count rounds so the arm shows up, but flags will be set
|
||
r["rounds"] = _round_count(armdir, run)
|
||
return r
|
||
|
||
deaths = {}
|
||
for o in evs:
|
||
t = o.get("type")
|
||
if t == "fire" and o.get("owner") == other:
|
||
r["shots"] += 1
|
||
elif t == "hit":
|
||
if o.get("owner") == other:
|
||
r["damage"] += o.get("damage", 0.0)
|
||
r["hits_dealt"] += 1
|
||
if o.get("victim") == other:
|
||
r["damage_taken"] += o.get("damage", 0.0)
|
||
r["hits_taken"] += 1
|
||
elif t == "death":
|
||
deaths.setdefault(o.get("round"), []).append(o.get("victim"))
|
||
|
||
nrounds = _round_count(armdir, run)
|
||
r["rounds"] = nrounds
|
||
scores = parse_round_scores(
|
||
os.path.join(armdir, f"run{run}.jsonl.results.json"))
|
||
# Per-round winner: primary = death events (a lone death means the OTHER bot
|
||
# won). A round with two or zero deaths (a mutual kill or a timeout) cannot
|
||
# be resolved from deaths alone, so the name-based per-round score delta is
|
||
# the tie-break. The two methods are cross-checked on the unambiguous rounds.
|
||
wins = death_solved = agree = ambiguous = 0
|
||
for rd in range(1, nrounds + 1):
|
||
vics = deaths.get(rd, [])
|
||
winner = None
|
||
if len(vics) == 1:
|
||
winner = other if vics[0] == subj else subj
|
||
death_solved += 1
|
||
delta = scores.get(rd)
|
||
if delta and SUBJECT_NAME in delta and BOT_NAME in delta:
|
||
score_winner = (subj if delta[SUBJECT_NAME] > delta[BOT_NAME]
|
||
else other)
|
||
if winner is None:
|
||
winner = score_winner
|
||
ambiguous += 1
|
||
elif winner == score_winner:
|
||
agree += 1
|
||
if winner == other:
|
||
wins += 1
|
||
r["wins"] = wins
|
||
r["death_solved"] = death_solved
|
||
r["death_score_agree"] = agree
|
||
r["ambiguous"] = ambiguous
|
||
return r
|
||
|
||
|
||
def _round_count(armdir, run):
|
||
p = os.path.join(armdir, f"run{run}.jsonl.rounds.json")
|
||
try:
|
||
return len(json.load(open(p)).get("rounds", []))
|
||
except (OSError, json.JSONDecodeError, AttributeError):
|
||
pass
|
||
# fall back to distinct round numbers in the events sidecar
|
||
rds = {o.get("round") for o in
|
||
parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))}
|
||
return len(rds)
|
||
|
||
|
||
# ── statistics ───────────────────────────────────────────────────────────────
|
||
|
||
def perm_test(xa, xb):
|
||
"""Two-sided permutation test on the difference of means.
|
||
|
||
Exact full enumeration when C(n, na) <= EXACT_CAP (7v7 is 3432), otherwise
|
||
a Monte-Carlo permutation test with MC_DRAWS fixed draws and the fixed seed
|
||
MC_SEED. Returns None if either arm is empty, else a dict:
|
||
obs observed |mean(xa) - mean(xb)| (what the p-value tests)
|
||
signed mean(xa) - mean(xb) (for display; negative means xb is larger)
|
||
p two-sided p-value
|
||
draws number of permutations (enumerated or sampled)
|
||
method 'exact' | 'monte-carlo'
|
||
se Monte-Carlo standard error of p (0.0 for the exact test)
|
||
seed MC seed (None for the exact test)
|
||
The MC p-value uses the (cnt + 1) / (B + 1) estimator, so it is never 0.
|
||
"""
|
||
na, nb = len(xa), len(xb)
|
||
if na == 0 or nb == 0:
|
||
return None
|
||
obs = abs(sum(xa) / na - sum(xb) / nb)
|
||
pooled = list(xa) + list(xb)
|
||
n = na + nb
|
||
total = sum(pooled)
|
||
ncomb = math.comb(n, na)
|
||
if ncomb <= EXACT_CAP:
|
||
cnt = 0
|
||
for combo in itertools.combinations(range(n), na):
|
||
sa = sum(pooled[i] for i in combo)
|
||
if abs(sa / na - (total - sa) / nb) >= obs - 1e-9:
|
||
cnt += 1
|
||
return {"obs": obs, "signed": sum(xa) / na - sum(xb) / nb,
|
||
"p": cnt / ncomb, "draws": ncomb,
|
||
"method": "exact", "se": 0.0, "seed": None}
|
||
rng = random.Random(MC_SEED)
|
||
sample = rng.sample
|
||
idx_range = range(n)
|
||
B = MC_DRAWS
|
||
cnt = 0
|
||
for _ in range(B):
|
||
sa = 0
|
||
for i in sample(idx_range, na):
|
||
sa += pooled[i]
|
||
if abs(sa / na - (total - sa) / nb) >= obs - 1e-9:
|
||
cnt += 1
|
||
p = (cnt + 1) / (B + 1)
|
||
se = math.sqrt(p * (1.0 - p) / (B + 1))
|
||
return {"obs": obs, "signed": sum(xa) / na - sum(xb) / nb,
|
||
"p": p, "draws": B, "method": "monte-carlo",
|
||
"se": se, "seed": MC_SEED}
|
||
|
||
|
||
def mannwhitney_p(xa, xb):
|
||
"""Two-sided Mann-Whitney U (rank-sum) test, normal approximation with the
|
||
standard tie correction and a continuity correction. Returns (p, U) or
|
||
None when either arm is empty. U is the smaller of U1, U2."""
|
||
na, nb = len(xa), len(xb)
|
||
n = na + nb
|
||
if na == 0 or nb == 0:
|
||
return None
|
||
vals = sorted([(v, 0) for v in xa] + [(v, 1) for v in xb])
|
||
rank = [0.0] * n
|
||
tie_term = 0.0
|
||
i = 0
|
||
while i < n:
|
||
j = i
|
||
while j + 1 < n and vals[j + 1][0] == vals[i][0]:
|
||
j += 1
|
||
t = j - i + 1
|
||
avg = (i + j) / 2.0 + 1.0
|
||
tie_term += t ** 3 - t
|
||
for k in range(i, j + 1):
|
||
rank[k] = avg
|
||
i = j + 1
|
||
r1 = sum(rank[k] for k in range(n) if vals[k][1] == 0)
|
||
u1 = r1 - na * (na + 1) / 2.0
|
||
mu = na * nb / 2.0
|
||
sigma2 = (na * nb / 12.0) * ((n + 1) - tie_term / (n * (n - 1)))
|
||
if sigma2 <= 0:
|
||
return (1.0, min(u1, na * nb - u1))
|
||
z = (abs(u1 - mu) - 0.5) / math.sqrt(sigma2)
|
||
if z < 0.0:
|
||
z = 0.0
|
||
p = math.erfc(z / math.sqrt(2.0))
|
||
return (p, min(u1, na * nb - u1))
|
||
|
||
|
||
def _var(x):
|
||
m = sum(x) / len(x)
|
||
return sum((v - m) ** 2 for v in x) / (len(x) - 1)
|
||
|
||
|
||
def min_detectable_effect(sd, n_per_arm):
|
||
"""Two-sample MDE at alpha=0.05 two-sided, 80% power, equal n."""
|
||
return Z_ALPHA_POWER * sd * math.sqrt(2.0 / n_per_arm)
|
||
|
||
|
||
def fisher_two_sided(a, b, c, d):
|
||
"""Two-sided Fisher exact on [[a,b],[c,d]]."""
|
||
n = a + b + c + d
|
||
if n == 0:
|
||
return None
|
||
row1, col1, col2 = a + b, a + c, b + d
|
||
denom = math.comb(n, row1)
|
||
if denom == 0:
|
||
return None
|
||
|
||
def hyper(x):
|
||
return math.comb(col1, x) * math.comb(col2, row1 - x) / denom
|
||
|
||
lo = max(0, col1 - (c + d))
|
||
hi = min(row1, col1)
|
||
obs = hyper(a)
|
||
return sum(hyper(x) for x in range(lo, hi + 1) if hyper(x) <= obs + 1e-12)
|
||
|
||
|
||
# ── liveness ─────────────────────────────────────────────────────────────────
|
||
|
||
def liveness(armdir, runs, env):
|
||
"""OK iff every declared env var appears verbatim in the bot's boot report
|
||
of every run, the report itself ran, and the bot did not warn that the name
|
||
is unrecognised (an unknown TR_*/GUN_* var is silently ignored)."""
|
||
if not runs:
|
||
return "FAIL", "no runs"
|
||
report_seen = {r: False for r in runs}
|
||
matched = {r: True for r in runs}
|
||
warned = {r: [] for r in runs}
|
||
for r in runs:
|
||
text = "".join(read_lines(os.path.join(
|
||
armdir, f"run{r}.bot.stdout.log")))
|
||
report_seen[r] = "=== ENVIRONMENT (boot report) ===" in text
|
||
for k, v in env.items():
|
||
if f"[env] {k}={v}" not in text:
|
||
matched[r] = False
|
||
if f"[env] WARNING: {k}" in text:
|
||
warned[r].append(k)
|
||
dead = [r for r in runs if not report_seen[r]]
|
||
miss = [r for r in runs if not matched[r]]
|
||
if dead:
|
||
return "FAIL", f"no bot env report in run(s) {dead}"
|
||
warn_runs = [r for r in runs if warned[r]]
|
||
if warn_runs:
|
||
names = sorted({k for r in warn_runs for k in warned[r]})
|
||
return "FAIL", (f"bot ignored arm env {names} (unrecognised) in "
|
||
f"run(s) {warn_runs}")
|
||
if env and miss:
|
||
vars_txt = " ".join(f"{k}={v}" for k, v in env.items())
|
||
return "FAIL", f"arm env '{vars_txt}' not in boot report of run(s) {miss}"
|
||
if not env:
|
||
return "OK", f"{len(runs)}/{len(runs)} runs: no arm env; report present"
|
||
vars_txt = " ".join(f"{k}={v}" for k, v in env.items())
|
||
return "OK", f"{len(runs)}/{len(runs)} runs: {vars_txt} applied"
|
||
|
||
|
||
# ── [bb] shift liveness ──────────────────────────────────────────────────────
|
||
|
||
_BB_SHIFT_RE = re.compile(r"\[bb\].*?shift=([+-]?\d+(?:\.\d+)?)deg")
|
||
|
||
|
||
def bb_shifts(armdir, runs):
|
||
"""All `[bb] ... shift=Xdeg` values in an arm's bot stdout (TR_BITBRAIN_LOG=1).
|
||
Returns (values, runs_with_lines, runs). A placebo that never applies a
|
||
correction emits ZERO [bb] lines; `bbLog` is only reached inside the same
|
||
`trained >= minObs` branch that computes the applied shift."""
|
||
vals = []
|
||
runs_with = 0
|
||
for r in runs:
|
||
text = "".join(read_lines(os.path.join(
|
||
armdir, f"run{r}.bot.stdout.log")))
|
||
found = _BB_SHIFT_RE.findall(text)
|
||
if found:
|
||
runs_with += 1
|
||
vals.extend(float(x) for x in found)
|
||
return vals, runs_with, len(runs)
|
||
|
||
|
||
# ── report ───────────────────────────────────────────────────────────────────
|
||
|
||
def main():
|
||
args = sys.argv[1:]
|
||
if not args or args[0] in ("-h", "--help"):
|
||
print(__doc__)
|
||
return 0 if args else 2
|
||
outdir = args[0]
|
||
reference = None
|
||
if "--reference" in args:
|
||
reference = args[args.index("--reference") + 1]
|
||
if not os.path.isdir(outdir):
|
||
print(f"ERROR: not a directory: {outdir}", file=sys.stderr)
|
||
return 2
|
||
|
||
session = load_session(outdir)
|
||
arms = discover_arms(outdir, session)
|
||
if not arms:
|
||
print(f"ERROR: no arms found in {outdir}", file=sys.stderr)
|
||
return 2
|
||
if reference is None:
|
||
reference = arms[0]["name"]
|
||
|
||
data = {}
|
||
for arm in arms:
|
||
armdir = os.path.join(outdir, arm["name"])
|
||
runs = discover_runs(armdir)
|
||
per = [parse_run(armdir, r) for r in runs]
|
||
data[arm["name"]] = {"dir": armdir, "runs": runs, "per": per,
|
||
"env": arm["env"], "label": arm["label"]}
|
||
|
||
if session:
|
||
print(f"# session {outdir}")
|
||
print(f"# commit={session.get('commit','?')} "
|
||
f"binary_sha256={session.get('binary_sha256','?')} "
|
||
f"rounds={session.get('rounds','?')} runs={session.get('runs','?')} "
|
||
f"conc={session.get('conc','?')} ts={session.get('timestamp','?')}")
|
||
print()
|
||
|
||
# ── summary table ────────────────────────────────────────────────────────
|
||
print("ARM SUMMARY")
|
||
hdr = (f"{'arm':<14} {'runs':>4} {'dmg/run':>8} {'dmgtk/run':>9} "
|
||
f"{'wins':>7} {'win%':>6} {'shots/run':>9} {'hitstk/run':>10}")
|
||
print(hdr)
|
||
print("-" * len(hdr))
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
per = data[name]["per"]
|
||
n = len(per)
|
||
if n == 0:
|
||
print(f"{name:<14} {0:>4} (no runs)")
|
||
continue
|
||
dmg = sum(p["damage"] for p in per) / n
|
||
dmgv = sum(p["damage_taken"] for p in per) / n
|
||
shots = sum(p["shots"] for p in per) / n
|
||
htk = sum(p["hits_taken"] for p in per) / n
|
||
wins = sum(p["wins"] for p in per if p["wins"] is not None)
|
||
rounds = sum(p["rounds"] for p in per)
|
||
pct = f"{100.0 * wins / rounds:.1f}" if rounds else "n/a"
|
||
print(f"{name:<14} {n:>4} {dmg:>8.0f} {dmgv:>9.0f} "
|
||
f"{str(wins) + '/' + str(rounds):>7} {pct:>6} {shots:>9.0f} {htk:>10.1f}")
|
||
|
||
# ── per-run values ───────────────────────────────────────────────────────
|
||
print("\nPER-RUN (never just the mean)")
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
per = data[name]["per"]
|
||
dmgs = " ".join(
|
||
f"r{p['run']}={p['damage']:.0f}" for p in per)
|
||
wins = " ".join(
|
||
f"r{p['run']}={p['wins']}/{p['rounds']}" if p["wins"] is not None
|
||
else f"r{p['run']}=?/{p['rounds']}" for p in per)
|
||
print(f" {name:<14} dmg: {dmgs}")
|
||
print(f" {'':<14} wins: {wins}")
|
||
|
||
# ── permutation tests (exact for 7v7, Monte-Carlo for 30v30) ─────────────
|
||
if reference not in data:
|
||
print(f"\nWARNING: reference arm '{reference}' not found; skipping tests")
|
||
return 1
|
||
ref = data[reference]["per"]
|
||
print("\nPAIRWISE PERMUTATION TEST (per-run values) + MANN-WHITNEY CROSS-CHECK")
|
||
print(f"permutation: exact when C(n,na) <= {EXACT_CAP:,}; otherwise "
|
||
f"Monte-Carlo {MC_DRAWS:,} draws, seed={MC_SEED:#x}, "
|
||
f"p = (cnt+1)/(B+1), se = sqrt(p(1-p)/(B+1))")
|
||
hdr = (f"{'metric':<11} {'A':<11} {'B':<11} {'diff(A-B)':>11} "
|
||
f"{'perm p':>9} {'method':<12} {'MC se':>7} {'MW p':>9} {'MW U':>8}")
|
||
print(hdr)
|
||
print("-" * len(hdr))
|
||
for i in range(len(arms)):
|
||
for j in range(i + 1, len(arms)):
|
||
ana, bnb = arms[i]["name"], arms[j]["name"]
|
||
for label, key in (("dmg/run", "damage"), ("round wins", "wins")):
|
||
xa = [p[key] for p in data[ana]["per"] if p[key] is not None]
|
||
xb = [p[key] for p in data[bnb]["per"] if p[key] is not None]
|
||
res = perm_test(xa, xb)
|
||
mw = mannwhitney_p(xa, xb)
|
||
if res is None:
|
||
print(f"{label:<11} {ana:<11} {bnb:<11} {'n/a':>11}")
|
||
continue
|
||
mstr = ("exact" if res["method"] == "exact"
|
||
else f"MC/B={res['draws']:,}")
|
||
sestr = "-" if res["method"] == "exact" else f"{res['se']:.4f}"
|
||
mwp = f"{mw[0]:.4f}" if mw else "n/a"
|
||
mwu = f"{mw[1]:.1f}" if mw else "n/a"
|
||
print(f"{label:<11} {ana:<11} {bnb:<11} {res['signed']:>+11.3f} "
|
||
f"{res['p']:>9.4f} {mstr:<12} {sestr:>7} {mwp:>9} {mwu:>8}")
|
||
|
||
# ── minimum detectable effect ──────────────────────────────────────────
|
||
print("\nMINIMUM DETECTABLE EFFECT (two-sample, alpha=0.05 two-sided, "
|
||
"80% power; MDE = 2.8016*sd*sqrt(2/n))")
|
||
print(f"{'metric':<11} {'n/arm':>6} {'sd(control)':>12} {'MDE(abs)':>10} "
|
||
f"{'MDE vs control mean':>22}")
|
||
print("-" * 64)
|
||
ctrl = data[reference]["per"]
|
||
for label, key in (("dmg/run", "damage"), ("round wins", "wins")):
|
||
vals = [p[key] for p in ctrl if p[key] is not None]
|
||
if len(vals) < 2:
|
||
continue
|
||
sd = math.sqrt(_var(vals))
|
||
mde = min_detectable_effect(sd, len(vals))
|
||
cmean = sum(vals) / len(vals)
|
||
rel = (f"{100.0*mde/abs(cmean):.1f}% of {cmean:.1f}"
|
||
if cmean else "n/a")
|
||
print(f"{label:<11} {len(vals):>6} {sd:>12.3f} {mde:>10.3f} {rel:>22}")
|
||
|
||
# ── round-level test (anti-conservative) ─────────────────────────────────
|
||
print("\nROUND-LEVEL TEST (pooled rounds, Fisher exact) vs "
|
||
f"`{reference}` — ANTI-CONSERVATIVE: rounds cluster within runs")
|
||
print(f"{'arm':<14} {'ref wins':>10} {'arm wins':>10} {'p':>8}")
|
||
print("-" * 46)
|
||
refw = sum(p["wins"] for p in ref if p["wins"] is not None)
|
||
refr = sum(p["rounds"] for p in ref)
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
if name == reference:
|
||
continue
|
||
per = data[name]["per"]
|
||
w = sum(p["wins"] for p in per if p["wins"] is not None)
|
||
rr = sum(p["rounds"] for p in per)
|
||
p = fisher_two_sided(w, rr - w, refw, refr - refw)
|
||
pstr = f"{p:.4f}" if p is not None else "n/a"
|
||
print(f"{name:<14} {str(refw) + '/' + str(refr):>10} "
|
||
f"{str(w) + '/' + str(rr):>10} {pstr:>8}")
|
||
|
||
# ── liveness ─────────────────────────────────────────────────────────────
|
||
print("\nLIVENESS (arm env applied in the bot's own boot report)")
|
||
lv_fail = 0
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
status, why = liveness(data[name]["dir"], data[name]["runs"],
|
||
data[name]["env"])
|
||
if status == "FAIL":
|
||
lv_fail += 1
|
||
print(f" {name:<14} {status:<4} ({why})")
|
||
|
||
# ── [bb] applied-shift check (a placebo must apply exactly 0) ────────────
|
||
print("\n[bb] APPLIED-SHIFT CHECK (from bot stdout; needs TR_BITBRAIN_LOG=1). "
|
||
"A provably-zero placebo emits ZERO [bb] lines.")
|
||
print(f" {'arm':<14} {'runs w/log':>11} {'lines':>7} {'min':>9} "
|
||
f"{'max':>9} {'zeros':>6}")
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
vals, runs_with, nruns = bb_shifts(data[name]["dir"], data[name]["runs"])
|
||
if not vals:
|
||
print(f" {name:<14} {runs_with:>4}/{nruns:<4} {0:>7} "
|
||
f"{'-':>9} {'-':>9} {'-':>6}")
|
||
else:
|
||
zeros = sum(1 for v in vals if v == 0.0)
|
||
print(f" {name:<14} {runs_with:>4}/{nruns:<4} {len(vals):>7} "
|
||
f"{min(vals):>+9.2f} {max(vals):>+9.2f} {zeros:>6}")
|
||
|
||
# ── round-win attribution cross-check ────────────────────────────────────
|
||
print("\nROUND-WIN ATTRIBUTION (events primary; score tie-break for "
|
||
"mutual-kill / timeout rounds)")
|
||
for arm in arms:
|
||
name = arm["name"]
|
||
runs_ok = runs_tot = 0
|
||
dsolve = dagree = damb = 0
|
||
for p in data[name]["per"]:
|
||
if p["wins"] is not None and p["first_places"] is not None:
|
||
runs_tot += 1
|
||
if p["wins"] == p["first_places"]:
|
||
runs_ok += 1
|
||
dsolve += p["death_solved"]
|
||
dagree += p["death_score_agree"]
|
||
damb += p["ambiguous"]
|
||
status = ("OK" if runs_tot and runs_ok == runs_tot
|
||
else ("n/a" if not runs_tot else "MISMATCH"))
|
||
print(f" {name:<14} wins==firstPlaces {runs_ok}/{runs_tot} runs {status}; "
|
||
f"single-death rounds agree with score {dagree}/{dsolve} "
|
||
f"({damb} tie-broken)")
|
||
|
||
return 1 if lv_fail else 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|