da4a971ca9
Adds common_libs/tests/measure_melee_bitbrain_ab.nim (+ .sh driver, .py analyzer,
committed per-run fixtures) and docs/melee_bitbrain_ab.md.
Experiment: 4-bot Free-For-All (ModularBot + WaveSurfer + PatternMover +
RandomMover), 4 arms x 16 runs x 7 rounds, frozen ModularBot from git archive
HEAD (commit 0f5cfe3, binary 11bba27), shipped tfil movement in every run.
Arms differ only in the gun rack: pattern (shipped), bb_round, bb_ret, bb_learn.
Result: NOT DETECTABLE. Score (server round score = damage + survival bonus)
differs by -63..+33 pts (perm p=0.16-0.71) against an MDE of 151 (~5.1%).
Every arm finishes rank 1. Round wins hint BitBrain's way (112/112 and 111/112
vs 109/112) but p=0.225 (MW 0.080), half the 0.40-win MDE.
Liveness proven: rack boot lines flip (rack active melee = PATTERN / BITBRAIN),
every run faced 3 distinct targets and ~66-69 target changes, and the bb arms
logged one [bb-reset] reason=target_change per switch. The melee premise was
exercised; the fast adaptation bought no measurable score edge at this sample.
216 lines
8.9 KiB
Python
216 lines
8.9 KiB
Python
#!/usr/bin/env python3
|
|
"""analyze_melee_ab.py — analyze a melee A/B session produced by
|
|
common_libs/tests/measure_melee_bitbrain_ab.nim (+ run_melee_bitbrain_ab.sh).
|
|
|
|
python3 common_libs/tests/analyze_melee_ab.py <outdir> [--reference ARM]
|
|
|
|
Melee scoring caveat: a round score (and the run `totalScore`) is the SERVER's
|
|
round score = damage dealt + the survival bonus; it is not raw damage. Per-round
|
|
placement is the 1..N rank in that round. `firstPlaces` is the count of rounds
|
|
won (rank 1) and `survivalCount` the total ticks survived.
|
|
|
|
Statistics are reused verbatim from tools/ab/ab_analyze.py (MC permutation test
|
|
+ Mann-Whitney U + MDE), so the melee numbers use the same machinery as the 1v1
|
|
A/Bs. All tests are two-sided on PER-RUN values.
|
|
"""
|
|
import glob
|
|
import importlib.util
|
|
import json
|
|
import math
|
|
import os
|
|
import statistics
|
|
import sys
|
|
|
|
REPO = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
|
_AB = os.path.join(REPO, "tools", "ab", "ab_analyze.py")
|
|
_spec = importlib.util.spec_from_file_location("ab_analyze", _AB)
|
|
ab = importlib.util.module_from_spec(_spec)
|
|
_spec.loader.exec_module(ab)
|
|
|
|
|
|
def load_runs(outdir, arm):
|
|
runs = []
|
|
for p in sorted(glob.glob(os.path.join(outdir, arm, "run*.json"))):
|
|
try:
|
|
d = json.load(open(p))
|
|
except (OSError, json.JSONDecodeError):
|
|
continue
|
|
if not d.get("ok"):
|
|
continue
|
|
mb = d.get("modularbot", {})
|
|
pr = d.get("perRound", [])
|
|
ranks = [r["rank"] for r in pr]
|
|
runs.append({
|
|
"run": d["run"],
|
|
"wins": mb.get("firstPlaces", 0),
|
|
"rounds": d.get("rounds", len(pr)),
|
|
"score": mb.get("totalScore", 0),
|
|
"survival": mb.get("survivalCount", 0),
|
|
"rank": mb.get("rank", 0),
|
|
"mean_rank": (sum(ranks) / len(ranks)) if ranks else float("nan"),
|
|
"field_score": sum(f.get("totalScore", 0) for f in d.get("field", [])),
|
|
"live": d.get("live", {}),
|
|
})
|
|
return runs
|
|
|
|
|
|
def score_share(r):
|
|
tot = r["score"] + r["field_score"]
|
|
return r["score"] / tot if tot else float("nan")
|
|
|
|
|
|
def fmt(xs, width=5):
|
|
return " ".join(f"{x:>{width}g}" for x in xs)
|
|
|
|
|
|
def main():
|
|
args = sys.argv[1:]
|
|
if not args:
|
|
print(__doc__)
|
|
return 2
|
|
outdir = args[0]
|
|
reference = "pattern"
|
|
if "--reference" in args:
|
|
reference = args[args.index("--reference") + 1]
|
|
arms = [d for d in sorted(os.listdir(outdir))
|
|
if os.path.isdir(os.path.join(outdir, d)) and load_runs(outdir, d)]
|
|
if reference not in arms:
|
|
print(f"ERROR: reference '{reference}' not among {arms}", file=sys.stderr)
|
|
return 2
|
|
|
|
data = {a: load_runs(outdir, a) for a in arms}
|
|
|
|
print(f"# melee A/B session: {outdir}")
|
|
print(f"# arms: {', '.join(arms)} reference: {reference}")
|
|
# commit / binary sha
|
|
anyj = glob.glob(os.path.join(outdir, reference, "run*.json"))
|
|
if anyj:
|
|
d0 = json.load(open(sorted(anyj)[0]))
|
|
print(f"# commit={d0.get('commit')} binary_sha256={d0.get('binary_sha256')}")
|
|
print()
|
|
|
|
print("ARM SUMMARY (melee: score=server round score dmg+survival bonus)")
|
|
hdr = (f"{'arm':<10} {'runs':>4} {'wins/rd':>9} {'score':>8} "
|
|
f"{'surv':>7} {'rank':>5} {'meanrank':>9} {'score share':>11} "
|
|
f"{'targets':>8} {'tchanges':>9}")
|
|
print(hdr)
|
|
print("-" * len(hdr))
|
|
for a in arms:
|
|
rs = data[a]
|
|
n = len(rs)
|
|
wins = sum(r["wins"] for r in rs)
|
|
rounds = sum(r["rounds"] for r in rs)
|
|
sc = statistics.mean(r["score"] for r in rs)
|
|
sv = statistics.mean(r["survival"] for r in rs)
|
|
rk = statistics.mean(r["rank"] for r in rs)
|
|
mr = statistics.mean(r["mean_rank"] for r in rs)
|
|
ss = statistics.mean(score_share(r) for r in rs)
|
|
tg = statistics.mean(r["live"].get("distinct_targets", 0) for r in rs)
|
|
tc = statistics.mean(r["live"].get("target_changes", 0) for r in rs)
|
|
print(f"{a:<10} {n:>4} {str(wins)+'/'+str(rounds):>9} {sc:>8.0f} "
|
|
f"{sv:>7.0f} {rk:>5.2f} {mr:>9.3f} {ss*100:>10.1f}% "
|
|
f"{tg:>8.2f} {tc:>9.2f}")
|
|
|
|
print("\nPER-RUN VALUES (never just the mean)")
|
|
for a in arms:
|
|
rs = data[a]
|
|
print(f" {a:<10} wins: " + " ".join(f"r{r['run']}={r['wins']}" for r in rs))
|
|
print(f" {'':<10} score: " + " ".join(f"r{r['run']}={r['score']}" for r in rs))
|
|
print(f" {'':<10} surv: " + " ".join(f"r{r['run']}={r['survival']}" for r in rs))
|
|
print(f" {'':<10} mrank: " + " ".join(f"r{r['run']}={r['mean_rank']:.2f}" for r in rs))
|
|
|
|
metrics = [("score", "score"), ("survival", "survival"),
|
|
("wins", "wins"), ("mean_rank", "mean_rank"),
|
|
("rank", "rank"), ("score_share", "__share__")]
|
|
|
|
def vals(a, key, f):
|
|
if key == "__share__":
|
|
return [score_share(r) for r in data[a]]
|
|
return [r[key] for r in data[a]]
|
|
|
|
print("\nPAIRWISE PERMUTATION TEST (per-run) + MANN-WHITNEY CROSS-CHECK")
|
|
print(f" perm: exact when C(n,na) <= {ab.EXACT_CAP:,}; else Monte-Carlo "
|
|
f"{ab.MC_DRAWS:,} draws seed={ab.MC_SEED:#x}")
|
|
hdr = (f"{'metric':<12} {'A':<10} {'B':<10} {'diff(A-B)':>11} "
|
|
f"{'perm p':>9} {'method':<12} {'MW p':>9} {'MW U':>8}")
|
|
print(hdr)
|
|
print("-" * len(hdr))
|
|
for label, key in metrics:
|
|
for a in arms:
|
|
if a == reference:
|
|
continue
|
|
xa = vals(reference, key, None)
|
|
xb = vals(a, key, None)
|
|
res = ab.perm_test(xa, xb)
|
|
mw = ab.mannwhitney_p(xa, xb)
|
|
if res is None:
|
|
continue
|
|
mstr = ("exact" if res["method"] == "exact"
|
|
else f"MC/B={res['draws']:,}")
|
|
mwp = f"{mw[0]:.4f}" if mw else "n/a"
|
|
mwu = f"{mw[1]:.1f}" if mw else "n/a"
|
|
print(f"{label:<12} {reference:<10} {a:<10} {res['signed']:>+11.4f} "
|
|
f"{res['p']:>9.4f} {mstr:<12} {mwp:>9} {mwu:>8}")
|
|
|
|
print("\nMINIMUM DETECTABLE EFFECT (two-sample, alpha=0.05 two-sided, "
|
|
"80% power; MDE = 2.8016*sd*sqrt(2/n))")
|
|
print(f"{'metric':<12} {'n/arm':>6} {'sd(ref)':>9} {'MDE':>10} {'ref mean':>10}")
|
|
print("-" * 52)
|
|
for label, key in [("score", "score"), ("survival", "survival"),
|
|
("wins", "wins"), ("mean_rank", "mean_rank")]:
|
|
xa = vals(reference, key, None)
|
|
if len(xa) < 2:
|
|
continue
|
|
sd = statistics.stdev(xa)
|
|
mde = ab.min_detectable_effect(sd, len(xa))
|
|
print(f"{label:<12} {len(xa):>6} {sd:>9.2f} {mde:>10.2f} {statistics.mean(xa):>10.3f}")
|
|
|
|
print("\nLIVENESS (per arm, from the bot's own boot report + log)")
|
|
print(f"{'arm':<10} {'rack_melee':>11} {'mem':>9} {'targets':>8} "
|
|
f"{'tchanges':>9} {'bb-reset':>9} {'OK':>4}")
|
|
for a in arms:
|
|
rs = data[a]
|
|
racks = {r["live"].get("rack_active_melee", "?") for r in rs}
|
|
mems = {r["live"].get("env_TR_BITBRAIN_MEM", "?") for r in rs}
|
|
tgt = statistics.mean(r["live"].get("distinct_targets", 0) for r in rs)
|
|
tch = statistics.mean(r["live"].get("target_changes", 0) for r in rs)
|
|
rst = statistics.mean(r["live"].get("bb_reset_target_change", 0) for r in rs)
|
|
if a == reference:
|
|
ok = all(r["live"].get("rack_active_melee") == "PATTERN"
|
|
and r["live"].get("env_TR_RACK_BITBRAIN") == "off"
|
|
and r["live"].get("bb_reset_target_change", -1) == 0
|
|
for r in rs)
|
|
else:
|
|
ok = all(r["live"].get("rack_active_melee") == "BITBRAIN"
|
|
and r["live"].get("env_TR_RACK_PATTERN") == "off"
|
|
and r["live"].get("env_TR_RACK_BITBRAIN") == "melee"
|
|
and r["live"].get("bb_reset_target_change", 0) > 0
|
|
for r in rs)
|
|
print(f"{a:<10} {','.join(sorted(racks)):>11} {','.join(sorted(mems)):>9} "
|
|
f"{tgt:>8.2f} {tch:>9.2f} {rst:>9.2f} {'yes' if ok else 'NO':>4}")
|
|
|
|
# machine-readable dump for the doc
|
|
dump = {"reference": reference, "arms": {}}
|
|
for a in arms:
|
|
rs = data[a]
|
|
dump["arms"][a] = {
|
|
"n": len(rs),
|
|
"wins": [r["wins"] for r in rs],
|
|
"rounds": [r["rounds"] for r in rs],
|
|
"score": [r["score"] for r in rs],
|
|
"survival": [r["survival"] for r in rs],
|
|
"mean_rank": [round(r["mean_rank"], 4) for r in rs],
|
|
"rack_melee": [r["live"].get("rack_active_melee") for r in rs],
|
|
"target_changes": [r["live"].get("target_changes") for r in rs],
|
|
"distinct_targets": [r["live"].get("distinct_targets") for r in rs],
|
|
"bb_reset": [r["live"].get("bb_reset_target_change") for r in rs],
|
|
}
|
|
with open(os.path.join(outdir, "analysis.json"), "w") as fh:
|
|
json.dump(dump, fh, indent=1)
|
|
print(f"\n# machine-readable: {os.path.join(outdir, 'analysis.json')}")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|