MELEE A/B: BitBrain vs shipped Pattern rack — repo's first melee measurement (null)
Adds common_libs/tests/measure_melee_bitbrain_ab.nim (+ .sh driver, .py analyzer,
committed per-run fixtures) and docs/melee_bitbrain_ab.md.
Experiment: 4-bot Free-For-All (ModularBot + WaveSurfer + PatternMover +
RandomMover), 4 arms x 16 runs x 7 rounds, frozen ModularBot from git archive
HEAD (commit 0f5cfe3, binary 11bba27), shipped tfil movement in every run.
Arms differ only in the gun rack: pattern (shipped), bb_round, bb_ret, bb_learn.
Result: NOT DETECTABLE. Score (server round score = damage + survival bonus)
differs by -63..+33 pts (perm p=0.16-0.71) against an MDE of 151 (~5.1%).
Every arm finishes rank 1. Round wins hint BitBrain's way (112/112 and 111/112
vs 109/112) but p=0.225 (MW 0.080), half the 0.40-win MDE.
Liveness proven: rack boot lines flip (rack active melee = PATTERN / BITBRAIN),
every run faced 3 distinct targets and ~66-69 target changes, and the bb arms
logged one [bb-reset] reason=target_change per switch. The melee premise was
exercised; the fast adaptation bought no measurable score edge at this sample.
This commit is contained in:
@@ -0,0 +1,215 @@
|
||||
#!/usr/bin/env python3
|
||||
"""analyze_melee_ab.py — analyze a melee A/B session produced by
|
||||
common_libs/tests/measure_melee_bitbrain_ab.nim (+ run_melee_bitbrain_ab.sh).
|
||||
|
||||
python3 common_libs/tests/analyze_melee_ab.py <outdir> [--reference ARM]
|
||||
|
||||
Melee scoring caveat: a round score (and the run `totalScore`) is the SERVER's
|
||||
round score = damage dealt + the survival bonus; it is not raw damage. Per-round
|
||||
placement is the 1..N rank in that round. `firstPlaces` is the count of rounds
|
||||
won (rank 1) and `survivalCount` the total ticks survived.
|
||||
|
||||
Statistics are reused verbatim from tools/ab/ab_analyze.py (MC permutation test
|
||||
+ Mann-Whitney U + MDE), so the melee numbers use the same machinery as the 1v1
|
||||
A/Bs. All tests are two-sided on PER-RUN values.
|
||||
"""
|
||||
import glob
|
||||
import importlib.util
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import statistics
|
||||
import sys
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
_AB = os.path.join(REPO, "tools", "ab", "ab_analyze.py")
|
||||
_spec = importlib.util.spec_from_file_location("ab_analyze", _AB)
|
||||
ab = importlib.util.module_from_spec(_spec)
|
||||
_spec.loader.exec_module(ab)
|
||||
|
||||
|
||||
def load_runs(outdir, arm):
|
||||
runs = []
|
||||
for p in sorted(glob.glob(os.path.join(outdir, arm, "run*.json"))):
|
||||
try:
|
||||
d = json.load(open(p))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
continue
|
||||
if not d.get("ok"):
|
||||
continue
|
||||
mb = d.get("modularbot", {})
|
||||
pr = d.get("perRound", [])
|
||||
ranks = [r["rank"] for r in pr]
|
||||
runs.append({
|
||||
"run": d["run"],
|
||||
"wins": mb.get("firstPlaces", 0),
|
||||
"rounds": d.get("rounds", len(pr)),
|
||||
"score": mb.get("totalScore", 0),
|
||||
"survival": mb.get("survivalCount", 0),
|
||||
"rank": mb.get("rank", 0),
|
||||
"mean_rank": (sum(ranks) / len(ranks)) if ranks else float("nan"),
|
||||
"field_score": sum(f.get("totalScore", 0) for f in d.get("field", [])),
|
||||
"live": d.get("live", {}),
|
||||
})
|
||||
return runs
|
||||
|
||||
|
||||
def score_share(r):
|
||||
tot = r["score"] + r["field_score"]
|
||||
return r["score"] / tot if tot else float("nan")
|
||||
|
||||
|
||||
def fmt(xs, width=5):
|
||||
return " ".join(f"{x:>{width}g}" for x in xs)
|
||||
|
||||
|
||||
def main():
|
||||
args = sys.argv[1:]
|
||||
if not args:
|
||||
print(__doc__)
|
||||
return 2
|
||||
outdir = args[0]
|
||||
reference = "pattern"
|
||||
if "--reference" in args:
|
||||
reference = args[args.index("--reference") + 1]
|
||||
arms = [d for d in sorted(os.listdir(outdir))
|
||||
if os.path.isdir(os.path.join(outdir, d)) and load_runs(outdir, d)]
|
||||
if reference not in arms:
|
||||
print(f"ERROR: reference '{reference}' not among {arms}", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
data = {a: load_runs(outdir, a) for a in arms}
|
||||
|
||||
print(f"# melee A/B session: {outdir}")
|
||||
print(f"# arms: {', '.join(arms)} reference: {reference}")
|
||||
# commit / binary sha
|
||||
anyj = glob.glob(os.path.join(outdir, reference, "run*.json"))
|
||||
if anyj:
|
||||
d0 = json.load(open(sorted(anyj)[0]))
|
||||
print(f"# commit={d0.get('commit')} binary_sha256={d0.get('binary_sha256')}")
|
||||
print()
|
||||
|
||||
print("ARM SUMMARY (melee: score=server round score dmg+survival bonus)")
|
||||
hdr = (f"{'arm':<10} {'runs':>4} {'wins/rd':>9} {'score':>8} "
|
||||
f"{'surv':>7} {'rank':>5} {'meanrank':>9} {'score share':>11} "
|
||||
f"{'targets':>8} {'tchanges':>9}")
|
||||
print(hdr)
|
||||
print("-" * len(hdr))
|
||||
for a in arms:
|
||||
rs = data[a]
|
||||
n = len(rs)
|
||||
wins = sum(r["wins"] for r in rs)
|
||||
rounds = sum(r["rounds"] for r in rs)
|
||||
sc = statistics.mean(r["score"] for r in rs)
|
||||
sv = statistics.mean(r["survival"] for r in rs)
|
||||
rk = statistics.mean(r["rank"] for r in rs)
|
||||
mr = statistics.mean(r["mean_rank"] for r in rs)
|
||||
ss = statistics.mean(score_share(r) for r in rs)
|
||||
tg = statistics.mean(r["live"].get("distinct_targets", 0) for r in rs)
|
||||
tc = statistics.mean(r["live"].get("target_changes", 0) for r in rs)
|
||||
print(f"{a:<10} {n:>4} {str(wins)+'/'+str(rounds):>9} {sc:>8.0f} "
|
||||
f"{sv:>7.0f} {rk:>5.2f} {mr:>9.3f} {ss*100:>10.1f}% "
|
||||
f"{tg:>8.2f} {tc:>9.2f}")
|
||||
|
||||
print("\nPER-RUN VALUES (never just the mean)")
|
||||
for a in arms:
|
||||
rs = data[a]
|
||||
print(f" {a:<10} wins: " + " ".join(f"r{r['run']}={r['wins']}" for r in rs))
|
||||
print(f" {'':<10} score: " + " ".join(f"r{r['run']}={r['score']}" for r in rs))
|
||||
print(f" {'':<10} surv: " + " ".join(f"r{r['run']}={r['survival']}" for r in rs))
|
||||
print(f" {'':<10} mrank: " + " ".join(f"r{r['run']}={r['mean_rank']:.2f}" for r in rs))
|
||||
|
||||
metrics = [("score", "score"), ("survival", "survival"),
|
||||
("wins", "wins"), ("mean_rank", "mean_rank"),
|
||||
("rank", "rank"), ("score_share", "__share__")]
|
||||
|
||||
def vals(a, key, f):
|
||||
if key == "__share__":
|
||||
return [score_share(r) for r in data[a]]
|
||||
return [r[key] for r in data[a]]
|
||||
|
||||
print("\nPAIRWISE PERMUTATION TEST (per-run) + MANN-WHITNEY CROSS-CHECK")
|
||||
print(f" perm: exact when C(n,na) <= {ab.EXACT_CAP:,}; else Monte-Carlo "
|
||||
f"{ab.MC_DRAWS:,} draws seed={ab.MC_SEED:#x}")
|
||||
hdr = (f"{'metric':<12} {'A':<10} {'B':<10} {'diff(A-B)':>11} "
|
||||
f"{'perm p':>9} {'method':<12} {'MW p':>9} {'MW U':>8}")
|
||||
print(hdr)
|
||||
print("-" * len(hdr))
|
||||
for label, key in metrics:
|
||||
for a in arms:
|
||||
if a == reference:
|
||||
continue
|
||||
xa = vals(reference, key, None)
|
||||
xb = vals(a, key, None)
|
||||
res = ab.perm_test(xa, xb)
|
||||
mw = ab.mannwhitney_p(xa, xb)
|
||||
if res is None:
|
||||
continue
|
||||
mstr = ("exact" if res["method"] == "exact"
|
||||
else f"MC/B={res['draws']:,}")
|
||||
mwp = f"{mw[0]:.4f}" if mw else "n/a"
|
||||
mwu = f"{mw[1]:.1f}" if mw else "n/a"
|
||||
print(f"{label:<12} {reference:<10} {a:<10} {res['signed']:>+11.4f} "
|
||||
f"{res['p']:>9.4f} {mstr:<12} {mwp:>9} {mwu:>8}")
|
||||
|
||||
print("\nMINIMUM DETECTABLE EFFECT (two-sample, alpha=0.05 two-sided, "
|
||||
"80% power; MDE = 2.8016*sd*sqrt(2/n))")
|
||||
print(f"{'metric':<12} {'n/arm':>6} {'sd(ref)':>9} {'MDE':>10} {'ref mean':>10}")
|
||||
print("-" * 52)
|
||||
for label, key in [("score", "score"), ("survival", "survival"),
|
||||
("wins", "wins"), ("mean_rank", "mean_rank")]:
|
||||
xa = vals(reference, key, None)
|
||||
if len(xa) < 2:
|
||||
continue
|
||||
sd = statistics.stdev(xa)
|
||||
mde = ab.min_detectable_effect(sd, len(xa))
|
||||
print(f"{label:<12} {len(xa):>6} {sd:>9.2f} {mde:>10.2f} {statistics.mean(xa):>10.3f}")
|
||||
|
||||
print("\nLIVENESS (per arm, from the bot's own boot report + log)")
|
||||
print(f"{'arm':<10} {'rack_melee':>11} {'mem':>9} {'targets':>8} "
|
||||
f"{'tchanges':>9} {'bb-reset':>9} {'OK':>4}")
|
||||
for a in arms:
|
||||
rs = data[a]
|
||||
racks = {r["live"].get("rack_active_melee", "?") for r in rs}
|
||||
mems = {r["live"].get("env_TR_BITBRAIN_MEM", "?") for r in rs}
|
||||
tgt = statistics.mean(r["live"].get("distinct_targets", 0) for r in rs)
|
||||
tch = statistics.mean(r["live"].get("target_changes", 0) for r in rs)
|
||||
rst = statistics.mean(r["live"].get("bb_reset_target_change", 0) for r in rs)
|
||||
if a == reference:
|
||||
ok = all(r["live"].get("rack_active_melee") == "PATTERN"
|
||||
and r["live"].get("env_TR_RACK_BITBRAIN") == "off"
|
||||
and r["live"].get("bb_reset_target_change", -1) == 0
|
||||
for r in rs)
|
||||
else:
|
||||
ok = all(r["live"].get("rack_active_melee") == "BITBRAIN"
|
||||
and r["live"].get("env_TR_RACK_PATTERN") == "off"
|
||||
and r["live"].get("env_TR_RACK_BITBRAIN") == "melee"
|
||||
and r["live"].get("bb_reset_target_change", 0) > 0
|
||||
for r in rs)
|
||||
print(f"{a:<10} {','.join(sorted(racks)):>11} {','.join(sorted(mems)):>9} "
|
||||
f"{tgt:>8.2f} {tch:>9.2f} {rst:>9.2f} {'yes' if ok else 'NO':>4}")
|
||||
|
||||
# machine-readable dump for the doc
|
||||
dump = {"reference": reference, "arms": {}}
|
||||
for a in arms:
|
||||
rs = data[a]
|
||||
dump["arms"][a] = {
|
||||
"n": len(rs),
|
||||
"wins": [r["wins"] for r in rs],
|
||||
"rounds": [r["rounds"] for r in rs],
|
||||
"score": [r["score"] for r in rs],
|
||||
"survival": [r["survival"] for r in rs],
|
||||
"mean_rank": [round(r["mean_rank"], 4) for r in rs],
|
||||
"rack_melee": [r["live"].get("rack_active_melee") for r in rs],
|
||||
"target_changes": [r["live"].get("target_changes") for r in rs],
|
||||
"distinct_targets": [r["live"].get("distinct_targets") for r in rs],
|
||||
"bb_reset": [r["live"].get("bb_reset_target_change") for r in rs],
|
||||
}
|
||||
with open(os.path.join(outdir, "analysis.json"), "w") as fh:
|
||||
json.dump(dump, fh, indent=1)
|
||||
print(f"\n# machine-readable: {os.path.join(outdir, 'analysis.json')}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user