#!/usr/bin/env python3 """analyze_melee_ab.py — analyze a melee A/B session produced by common_libs/tests/measure_melee_bitbrain_ab.nim (+ run_melee_bitbrain_ab.sh). python3 common_libs/tests/analyze_melee_ab.py [--reference ARM] Melee scoring caveat: a round score (and the run `totalScore`) is the SERVER's round score = damage dealt + the survival bonus; it is not raw damage. Per-round placement is the 1..N rank in that round. `firstPlaces` is the count of rounds won (rank 1) and `survivalCount` the total ticks survived. Statistics are reused verbatim from tools/ab/ab_analyze.py (MC permutation test + Mann-Whitney U + MDE), so the melee numbers use the same machinery as the 1v1 A/Bs. All tests are two-sided on PER-RUN values. """ import glob import importlib.util import json import math import os import statistics import sys REPO = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) _AB = os.path.join(REPO, "tools", "ab", "ab_analyze.py") _spec = importlib.util.spec_from_file_location("ab_analyze", _AB) ab = importlib.util.module_from_spec(_spec) _spec.loader.exec_module(ab) def load_runs(outdir, arm): runs = [] for p in sorted(glob.glob(os.path.join(outdir, arm, "run*.json"))): try: d = json.load(open(p)) except (OSError, json.JSONDecodeError): continue if not d.get("ok"): continue mb = d.get("modularbot", {}) pr = d.get("perRound", []) ranks = [r["rank"] for r in pr] runs.append({ "run": d["run"], "wins": mb.get("firstPlaces", 0), "rounds": d.get("rounds", len(pr)), "score": mb.get("totalScore", 0), "survival": mb.get("survivalCount", 0), "rank": mb.get("rank", 0), "mean_rank": (sum(ranks) / len(ranks)) if ranks else float("nan"), "field_score": sum(f.get("totalScore", 0) for f in d.get("field", [])), "live": d.get("live", {}), }) return runs def score_share(r): tot = r["score"] + r["field_score"] return r["score"] / tot if tot else float("nan") def fmt(xs, width=5): return " ".join(f"{x:>{width}g}" for x in xs) def main(): args = sys.argv[1:] if not args: print(__doc__) return 2 outdir = args[0] reference = "pattern" if "--reference" in args: reference = args[args.index("--reference") + 1] arms = [d for d in sorted(os.listdir(outdir)) if os.path.isdir(os.path.join(outdir, d)) and load_runs(outdir, d)] if reference not in arms: print(f"ERROR: reference '{reference}' not among {arms}", file=sys.stderr) return 2 data = {a: load_runs(outdir, a) for a in arms} print(f"# melee A/B session: {outdir}") print(f"# arms: {', '.join(arms)} reference: {reference}") # commit / binary sha anyj = glob.glob(os.path.join(outdir, reference, "run*.json")) if anyj: d0 = json.load(open(sorted(anyj)[0])) print(f"# commit={d0.get('commit')} binary_sha256={d0.get('binary_sha256')}") print() print("ARM SUMMARY (melee: score=server round score dmg+survival bonus)") hdr = (f"{'arm':<10} {'runs':>4} {'wins/rd':>9} {'score':>8} " f"{'surv':>7} {'rank':>5} {'meanrank':>9} {'score share':>11} " f"{'targets':>8} {'tchanges':>9}") print(hdr) print("-" * len(hdr)) for a in arms: rs = data[a] n = len(rs) wins = sum(r["wins"] for r in rs) rounds = sum(r["rounds"] for r in rs) sc = statistics.mean(r["score"] for r in rs) sv = statistics.mean(r["survival"] for r in rs) rk = statistics.mean(r["rank"] for r in rs) mr = statistics.mean(r["mean_rank"] for r in rs) ss = statistics.mean(score_share(r) for r in rs) tg = statistics.mean(r["live"].get("distinct_targets", 0) for r in rs) tc = statistics.mean(r["live"].get("target_changes", 0) for r in rs) print(f"{a:<10} {n:>4} {str(wins)+'/'+str(rounds):>9} {sc:>8.0f} " f"{sv:>7.0f} {rk:>5.2f} {mr:>9.3f} {ss*100:>10.1f}% " f"{tg:>8.2f} {tc:>9.2f}") print("\nPER-RUN VALUES (never just the mean)") for a in arms: rs = data[a] print(f" {a:<10} wins: " + " ".join(f"r{r['run']}={r['wins']}" for r in rs)) print(f" {'':<10} score: " + " ".join(f"r{r['run']}={r['score']}" for r in rs)) print(f" {'':<10} surv: " + " ".join(f"r{r['run']}={r['survival']}" for r in rs)) print(f" {'':<10} mrank: " + " ".join(f"r{r['run']}={r['mean_rank']:.2f}" for r in rs)) metrics = [("score", "score"), ("survival", "survival"), ("wins", "wins"), ("mean_rank", "mean_rank"), ("rank", "rank"), ("score_share", "__share__")] def vals(a, key, f): if key == "__share__": return [score_share(r) for r in data[a]] return [r[key] for r in data[a]] print("\nPAIRWISE PERMUTATION TEST (per-run) + MANN-WHITNEY CROSS-CHECK") print(f" perm: exact when C(n,na) <= {ab.EXACT_CAP:,}; else Monte-Carlo " f"{ab.MC_DRAWS:,} draws seed={ab.MC_SEED:#x}") hdr = (f"{'metric':<12} {'A':<10} {'B':<10} {'diff(A-B)':>11} " f"{'perm p':>9} {'method':<12} {'MW p':>9} {'MW U':>8}") print(hdr) print("-" * len(hdr)) for label, key in metrics: for a in arms: if a == reference: continue xa = vals(reference, key, None) xb = vals(a, key, None) res = ab.perm_test(xa, xb) mw = ab.mannwhitney_p(xa, xb) if res is None: continue mstr = ("exact" if res["method"] == "exact" else f"MC/B={res['draws']:,}") mwp = f"{mw[0]:.4f}" if mw else "n/a" mwu = f"{mw[1]:.1f}" if mw else "n/a" print(f"{label:<12} {reference:<10} {a:<10} {res['signed']:>+11.4f} " f"{res['p']:>9.4f} {mstr:<12} {mwp:>9} {mwu:>8}") print("\nMINIMUM DETECTABLE EFFECT (two-sample, alpha=0.05 two-sided, " "80% power; MDE = 2.8016*sd*sqrt(2/n))") print(f"{'metric':<12} {'n/arm':>6} {'sd(ref)':>9} {'MDE':>10} {'ref mean':>10}") print("-" * 52) for label, key in [("score", "score"), ("survival", "survival"), ("wins", "wins"), ("mean_rank", "mean_rank")]: xa = vals(reference, key, None) if len(xa) < 2: continue sd = statistics.stdev(xa) mde = ab.min_detectable_effect(sd, len(xa)) print(f"{label:<12} {len(xa):>6} {sd:>9.2f} {mde:>10.2f} {statistics.mean(xa):>10.3f}") print("\nLIVENESS (per arm, from the bot's own boot report + log)") print(f"{'arm':<10} {'rack_melee':>11} {'mem':>9} {'targets':>8} " f"{'tchanges':>9} {'bb-reset':>9} {'OK':>4}") for a in arms: rs = data[a] racks = {r["live"].get("rack_active_melee", "?") for r in rs} mems = {r["live"].get("env_TR_BITBRAIN_MEM", "?") for r in rs} tgt = statistics.mean(r["live"].get("distinct_targets", 0) for r in rs) tch = statistics.mean(r["live"].get("target_changes", 0) for r in rs) rst = statistics.mean(r["live"].get("bb_reset_target_change", 0) for r in rs) if a == reference: ok = all(r["live"].get("rack_active_melee") == "PATTERN" and r["live"].get("env_TR_RACK_BITBRAIN") == "off" and r["live"].get("bb_reset_target_change", -1) == 0 for r in rs) else: ok = all(r["live"].get("rack_active_melee") == "BITBRAIN" and r["live"].get("env_TR_RACK_PATTERN") == "off" and r["live"].get("env_TR_RACK_BITBRAIN") == "melee" and r["live"].get("bb_reset_target_change", 0) > 0 for r in rs) print(f"{a:<10} {','.join(sorted(racks)):>11} {','.join(sorted(mems)):>9} " f"{tgt:>8.2f} {tch:>9.2f} {rst:>9.2f} {'yes' if ok else 'NO':>4}") # machine-readable dump for the doc dump = {"reference": reference, "arms": {}} for a in arms: rs = data[a] dump["arms"][a] = { "n": len(rs), "wins": [r["wins"] for r in rs], "rounds": [r["rounds"] for r in rs], "score": [r["score"] for r in rs], "survival": [r["survival"] for r in rs], "mean_rank": [round(r["mean_rank"], 4) for r in rs], "rack_melee": [r["live"].get("rack_active_melee") for r in rs], "target_changes": [r["live"].get("target_changes") for r in rs], "distinct_targets": [r["live"].get("distinct_targets") for r in rs], "bb_reset": [r["live"].get("bb_reset_target_change") for r in rs], } with open(os.path.join(outdir, "analysis.json"), "w") as fh: json.dump(dump, fh, indent=1) print(f"\n# machine-readable: {os.path.join(outdir, 'analysis.json')}") return 0 if __name__ == "__main__": sys.exit(main())