Files
SirStone da4a971ca9 MELEE A/B: BitBrain vs shipped Pattern rack — repo's first melee measurement (null)
Adds common_libs/tests/measure_melee_bitbrain_ab.nim (+ .sh driver, .py analyzer,
committed per-run fixtures) and docs/melee_bitbrain_ab.md.

Experiment: 4-bot Free-For-All (ModularBot + WaveSurfer + PatternMover +
RandomMover), 4 arms x 16 runs x 7 rounds, frozen ModularBot from git archive
HEAD (commit 0f5cfe3, binary 11bba27), shipped tfil movement in every run.
Arms differ only in the gun rack: pattern (shipped), bb_round, bb_ret, bb_learn.

Result: NOT DETECTABLE. Score (server round score = damage + survival bonus)
differs by -63..+33 pts (perm p=0.16-0.71) against an MDE of 151 (~5.1%).
Every arm finishes rank 1. Round wins hint BitBrain's way (112/112 and 111/112
vs 109/112) but p=0.225 (MW 0.080), half the 0.40-win MDE.

Liveness proven: rack boot lines flip (rack active melee = PATTERN / BITBRAIN),
every run faced 3 distinct targets and ~66-69 target changes, and the bb arms
logged one [bb-reset] reason=target_change per switch. The melee premise was
exercised; the fast adaptation bought no measurable score edge at this sample.
2026-09-26 00:35:44 +02:00

216 lines
8.9 KiB
Python

#!/usr/bin/env python3
"""analyze_melee_ab.py — analyze a melee A/B session produced by
common_libs/tests/measure_melee_bitbrain_ab.nim (+ run_melee_bitbrain_ab.sh).
python3 common_libs/tests/analyze_melee_ab.py <outdir> [--reference ARM]
Melee scoring caveat: a round score (and the run `totalScore`) is the SERVER's
round score = damage dealt + the survival bonus; it is not raw damage. Per-round
placement is the 1..N rank in that round. `firstPlaces` is the count of rounds
won (rank 1) and `survivalCount` the total ticks survived.
Statistics are reused verbatim from tools/ab/ab_analyze.py (MC permutation test
+ Mann-Whitney U + MDE), so the melee numbers use the same machinery as the 1v1
A/Bs. All tests are two-sided on PER-RUN values.
"""
import glob
import importlib.util
import json
import math
import os
import statistics
import sys
REPO = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
_AB = os.path.join(REPO, "tools", "ab", "ab_analyze.py")
_spec = importlib.util.spec_from_file_location("ab_analyze", _AB)
ab = importlib.util.module_from_spec(_spec)
_spec.loader.exec_module(ab)
def load_runs(outdir, arm):
runs = []
for p in sorted(glob.glob(os.path.join(outdir, arm, "run*.json"))):
try:
d = json.load(open(p))
except (OSError, json.JSONDecodeError):
continue
if not d.get("ok"):
continue
mb = d.get("modularbot", {})
pr = d.get("perRound", [])
ranks = [r["rank"] for r in pr]
runs.append({
"run": d["run"],
"wins": mb.get("firstPlaces", 0),
"rounds": d.get("rounds", len(pr)),
"score": mb.get("totalScore", 0),
"survival": mb.get("survivalCount", 0),
"rank": mb.get("rank", 0),
"mean_rank": (sum(ranks) / len(ranks)) if ranks else float("nan"),
"field_score": sum(f.get("totalScore", 0) for f in d.get("field", [])),
"live": d.get("live", {}),
})
return runs
def score_share(r):
tot = r["score"] + r["field_score"]
return r["score"] / tot if tot else float("nan")
def fmt(xs, width=5):
return " ".join(f"{x:>{width}g}" for x in xs)
def main():
args = sys.argv[1:]
if not args:
print(__doc__)
return 2
outdir = args[0]
reference = "pattern"
if "--reference" in args:
reference = args[args.index("--reference") + 1]
arms = [d for d in sorted(os.listdir(outdir))
if os.path.isdir(os.path.join(outdir, d)) and load_runs(outdir, d)]
if reference not in arms:
print(f"ERROR: reference '{reference}' not among {arms}", file=sys.stderr)
return 2
data = {a: load_runs(outdir, a) for a in arms}
print(f"# melee A/B session: {outdir}")
print(f"# arms: {', '.join(arms)} reference: {reference}")
# commit / binary sha
anyj = glob.glob(os.path.join(outdir, reference, "run*.json"))
if anyj:
d0 = json.load(open(sorted(anyj)[0]))
print(f"# commit={d0.get('commit')} binary_sha256={d0.get('binary_sha256')}")
print()
print("ARM SUMMARY (melee: score=server round score dmg+survival bonus)")
hdr = (f"{'arm':<10} {'runs':>4} {'wins/rd':>9} {'score':>8} "
f"{'surv':>7} {'rank':>5} {'meanrank':>9} {'score share':>11} "
f"{'targets':>8} {'tchanges':>9}")
print(hdr)
print("-" * len(hdr))
for a in arms:
rs = data[a]
n = len(rs)
wins = sum(r["wins"] for r in rs)
rounds = sum(r["rounds"] for r in rs)
sc = statistics.mean(r["score"] for r in rs)
sv = statistics.mean(r["survival"] for r in rs)
rk = statistics.mean(r["rank"] for r in rs)
mr = statistics.mean(r["mean_rank"] for r in rs)
ss = statistics.mean(score_share(r) for r in rs)
tg = statistics.mean(r["live"].get("distinct_targets", 0) for r in rs)
tc = statistics.mean(r["live"].get("target_changes", 0) for r in rs)
print(f"{a:<10} {n:>4} {str(wins)+'/'+str(rounds):>9} {sc:>8.0f} "
f"{sv:>7.0f} {rk:>5.2f} {mr:>9.3f} {ss*100:>10.1f}% "
f"{tg:>8.2f} {tc:>9.2f}")
print("\nPER-RUN VALUES (never just the mean)")
for a in arms:
rs = data[a]
print(f" {a:<10} wins: " + " ".join(f"r{r['run']}={r['wins']}" for r in rs))
print(f" {'':<10} score: " + " ".join(f"r{r['run']}={r['score']}" for r in rs))
print(f" {'':<10} surv: " + " ".join(f"r{r['run']}={r['survival']}" for r in rs))
print(f" {'':<10} mrank: " + " ".join(f"r{r['run']}={r['mean_rank']:.2f}" for r in rs))
metrics = [("score", "score"), ("survival", "survival"),
("wins", "wins"), ("mean_rank", "mean_rank"),
("rank", "rank"), ("score_share", "__share__")]
def vals(a, key, f):
if key == "__share__":
return [score_share(r) for r in data[a]]
return [r[key] for r in data[a]]
print("\nPAIRWISE PERMUTATION TEST (per-run) + MANN-WHITNEY CROSS-CHECK")
print(f" perm: exact when C(n,na) <= {ab.EXACT_CAP:,}; else Monte-Carlo "
f"{ab.MC_DRAWS:,} draws seed={ab.MC_SEED:#x}")
hdr = (f"{'metric':<12} {'A':<10} {'B':<10} {'diff(A-B)':>11} "
f"{'perm p':>9} {'method':<12} {'MW p':>9} {'MW U':>8}")
print(hdr)
print("-" * len(hdr))
for label, key in metrics:
for a in arms:
if a == reference:
continue
xa = vals(reference, key, None)
xb = vals(a, key, None)
res = ab.perm_test(xa, xb)
mw = ab.mannwhitney_p(xa, xb)
if res is None:
continue
mstr = ("exact" if res["method"] == "exact"
else f"MC/B={res['draws']:,}")
mwp = f"{mw[0]:.4f}" if mw else "n/a"
mwu = f"{mw[1]:.1f}" if mw else "n/a"
print(f"{label:<12} {reference:<10} {a:<10} {res['signed']:>+11.4f} "
f"{res['p']:>9.4f} {mstr:<12} {mwp:>9} {mwu:>8}")
print("\nMINIMUM DETECTABLE EFFECT (two-sample, alpha=0.05 two-sided, "
"80% power; MDE = 2.8016*sd*sqrt(2/n))")
print(f"{'metric':<12} {'n/arm':>6} {'sd(ref)':>9} {'MDE':>10} {'ref mean':>10}")
print("-" * 52)
for label, key in [("score", "score"), ("survival", "survival"),
("wins", "wins"), ("mean_rank", "mean_rank")]:
xa = vals(reference, key, None)
if len(xa) < 2:
continue
sd = statistics.stdev(xa)
mde = ab.min_detectable_effect(sd, len(xa))
print(f"{label:<12} {len(xa):>6} {sd:>9.2f} {mde:>10.2f} {statistics.mean(xa):>10.3f}")
print("\nLIVENESS (per arm, from the bot's own boot report + log)")
print(f"{'arm':<10} {'rack_melee':>11} {'mem':>9} {'targets':>8} "
f"{'tchanges':>9} {'bb-reset':>9} {'OK':>4}")
for a in arms:
rs = data[a]
racks = {r["live"].get("rack_active_melee", "?") for r in rs}
mems = {r["live"].get("env_TR_BITBRAIN_MEM", "?") for r in rs}
tgt = statistics.mean(r["live"].get("distinct_targets", 0) for r in rs)
tch = statistics.mean(r["live"].get("target_changes", 0) for r in rs)
rst = statistics.mean(r["live"].get("bb_reset_target_change", 0) for r in rs)
if a == reference:
ok = all(r["live"].get("rack_active_melee") == "PATTERN"
and r["live"].get("env_TR_RACK_BITBRAIN") == "off"
and r["live"].get("bb_reset_target_change", -1) == 0
for r in rs)
else:
ok = all(r["live"].get("rack_active_melee") == "BITBRAIN"
and r["live"].get("env_TR_RACK_PATTERN") == "off"
and r["live"].get("env_TR_RACK_BITBRAIN") == "melee"
and r["live"].get("bb_reset_target_change", 0) > 0
for r in rs)
print(f"{a:<10} {','.join(sorted(racks)):>11} {','.join(sorted(mems)):>9} "
f"{tgt:>8.2f} {tch:>9.2f} {rst:>9.2f} {'yes' if ok else 'NO':>4}")
# machine-readable dump for the doc
dump = {"reference": reference, "arms": {}}
for a in arms:
rs = data[a]
dump["arms"][a] = {
"n": len(rs),
"wins": [r["wins"] for r in rs],
"rounds": [r["rounds"] for r in rs],
"score": [r["score"] for r in rs],
"survival": [r["survival"] for r in rs],
"mean_rank": [round(r["mean_rank"], 4) for r in rs],
"rack_melee": [r["live"].get("rack_active_melee") for r in rs],
"target_changes": [r["live"].get("target_changes") for r in rs],
"distinct_targets": [r["live"].get("distinct_targets") for r in rs],
"bb_reset": [r["live"].get("bb_reset_target_change") for r in rs],
}
with open(os.path.join(outdir, "analysis.json"), "w") as fh:
json.dump(dump, fh, indent=1)
print(f"\n# machine-readable: {os.path.join(outdir, 'analysis.json')}")
return 0
if __name__ == "__main__":
sys.exit(main())