diff --git a/tools/ab/README.md b/tools/ab/README.md new file mode 100644 index 0000000..4ef0056 --- /dev/null +++ b/tools/ab/README.md @@ -0,0 +1,65 @@ +# tools/ab — reusable A/B harness + +Two tools, built once and reused for every variant test. Adding an arm costs +nothing: the frozen bot is built once per session and every arm reuses it. + +## 1. Run a session + +```sh +tools/ab/ab_run.sh --arms tools/ab/arms.example.txt --runs 7 --outdir /tmp/ab/power --conc 7 +``` + +* builds ONE frozen ModularBot from **current HEAD** (`git archive HEAD` + + `nim c -d:release`) and reuses that binary for every arm — a dirty tree cannot + leak into the measurement; +* runs arm × run battles vs real DrussGT in parallel (ephemeral ports, one + DrussGT botdir/data per run, one ModularBot botdir per run); +* writes `session.json` (commit, binary sha256, arms, runs/rounds, timestamp) + and `//run.{jsonl,jsonl.rounds.json,jsonl.results.json,events.jsonl,battle.log,bot.stdout.log}`. + +Options: `--arms FILE` (required) `--runs N` (default 7) `--outdir DIR` +(required) `--conc K` (default 7) `--rounds R` (default 7). + +It kills its own children (own process group + outdir-tagged backstop) on +EXIT/INT/TERM, so a Ctrl-C does not leave orphan battles. + +Prerequisites (fails loudly if any is missing): +`/tmp/robocode/install/libs/robocode.jar`, `/tmp/drussgt/DrussGT.jar`, +the Tank Royale runner jar, the bot-API jar, `nim`, and the shim `out/` classes. +`/tmp/tr_bots/DrussGT` is recreated via `make_botdir.sh` if absent (the actual +battles still use per-run copies). + +## 2. Analyze a session + +```sh +python3 tools/ab/ab_analyze.py /tmp/ab/power [--reference control] +``` + +Prints per-arm damage/run, damage taken/run, round wins, shots/run, hits +taken/run, the **per-run** values, an exact two-sided permutation test on +per-run damage and wins vs the reference arm (default: first arm), a +round-level Fisher test (labelled anti-conservative), a liveness OK/FAIL line, +and a round-win attribution cross-check. + +Round wins come from the events sidecar (the bot that does not die wins) and +are cross-checked against the runner's `firstPlaces`. The per-round lines in +`*.results.json` are **cumulative** standings — not round winners. + +## Arm file + +See `arms.example.txt`: + +``` +name | ENV_VAR=value ENV_VAR2=value2 | optional label +``` + +## Known gotchas + +* Ports: the runner picks ephemeral ports itself; nothing to configure. +* Races: never share a DrussGT botdir/data or a ModularBot stdout log across + parallel runs — `ab_run.sh` already gives every run its own. +* `pkill -f run_bridge_battle` matches the pkill command itself; use the + `[r]un_bridge_battle` trick (as `ab_run.sh` does). +* Liveness reads the bot's `[env]` boot report from + `/run.bot.stdout.log`; if an arm's variable is missing there it is a + FAIL, not a measurement. diff --git a/tools/ab/ab_analyze.py b/tools/ab/ab_analyze.py new file mode 100755 index 0000000..b794b4e --- /dev/null +++ b/tools/ab/ab_analyze.py @@ -0,0 +1,520 @@ +#!/usr/bin/env python3 +"""ab_analyze.py — read a session dir produced by ab_run.sh and print the report. + + python3 tools/ab/ab_analyze.py [--reference ARM] + +Standard library only, deterministic. For every arm it prints runs, damage/run, +damage taken/run, ROUND WINS, shots/run, hits taken/run and the PER-RUN values +(wins cluster at 0/7 and single-run damage swings ~200, so the mean alone lies). + +Tests: + * exact two-sided permutation test on PER-RUN values (damage/run and round + wins) vs a reference arm (default: the first arm). C(14,7)=3432 for 7v7 — + enumerated exactly, never sampled, whenever the combination count is small. + * a round-level Fisher exact test on pooled rounds, clearly labelled + anti-conservative (rounds cluster within runs). + +Round-win attribution comes from the events sidecar (the bot that does NOT die +wins the round) and is cross-checked against the runner's `firstPlaces`, which +is name-based. `firstPlaces` is the run-level truth: the per-round lines in +*.results.json are CUMULATIVE standings, not round winners — do not use them. + +Liveness: each arm's declared env vars must appear verbatim in the bot's own +boot environment report (/run.bot.stdout.log, `[env] VAR=VALUE`), so an +arm whose setting never reached the process is a loud FAIL rather than a +plausible-looking number. + +Owner attribution never relies on fired-power values (the power policy fires a +continuous 0.15–1.15 range). The subject (DrussGT) is identified by matching +the events sidecar's per-owner fire/hit counts against the capture's own +`subject event counts:` line; ModularBot is the other bot. +""" +import glob +import itertools +import json +import math +import os +import re +import sys + +# beyond this many combinations we sample (deterministic seed) and say so; +# 7v7 = C(14,7) = 3432 is always exact. +EXACT_CAP = 20_000_000 +BOT_NAME = "ModularBot" +SUBJECT_NAME = "DrussGT" # the capture subject (our bot is the adversary) + + +# ── loading ────────────────────────────────────────────────────────────────── + +def load_session(outdir): + p = os.path.join(outdir, "session.json") + if os.path.exists(p): + try: + return json.load(open(p)) + except (json.JSONDecodeError, OSError): + return None + return None + + +def parse_envspec(envspec): + """`TR_A=1 TR_B=2` -> {'TR_A': '1', 'TR_B': '2'}.""" + out = {} + for tok in (envspec or "").split(): + if "=" in tok: + k, v = tok.split("=", 1) + out[k] = v + return out + + +def discover_arms(outdir, session): + arms = [] + if session and isinstance(session.get("arms"), list): + for a in session["arms"]: + name = a.get("name") + if not name: + continue + arms.append({"name": name, + "env": parse_envspec(a.get("env", "")), + "label": a.get("label", "")}) + if arms: + return arms + # fallback: any subdir holding run*.jsonl + for d in sorted(os.listdir(outdir)): + full = os.path.join(outdir, d) + if os.path.isdir(full) and discover_runs(full): + arms.append({"name": d, "env": {}, "label": ""}) + return arms + + +def discover_runs(armdir): + runs = [] + for f in os.listdir(armdir): + m = re.fullmatch(r"run(\d+)\.jsonl", f) + if m: + runs.append(int(m.group(1))) + return sorted(runs) + + +def read_lines(path): + try: + with open(path, errors="replace") as fh: + return fh.readlines() + except OSError: + return [] + + +def parse_events(path): + evs = [] + for line in read_lines(path): + line = line.strip() + if not line: + continue + try: + evs.append(json.loads(line)) + except json.JSONDecodeError: + continue # partial line from an interrupted battle + return evs + + +def subject_counters(log_text): + m = re.search( + r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)" + r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)", log_text) + if not m: + return None + return {"fired": int(m.group(2)), "hits": int(m.group(3)), + "hits_taken": int(m.group(6))} + + +def parse_round_wins_by_name(log_text): + """ModularBot's `firstPlaces` from the runner's final standings block.""" + for m in re.finditer( + r"^\s*#\d+\s+(\S+)\s+totalScore=-?\d+\s+firstPlaces=(\d+)", log_text, + re.MULTILINE): + if m.group(1) == BOT_NAME: + return int(m.group(2)) + return None + + +def parse_round_scores(path): + """Per-round score delta per bot NAME from the cumulative *.results.json + lines. Used only to break ties (mutual kill / no death), never as the + primary attribution.""" + try: + lines = json.load(open(path)).get("roundResults", []) + except (OSError, json.JSONDecodeError, AttributeError): + return {} + prev, out = {}, {} + for i, line in enumerate(lines, 1): + cur = {m.group(1): int(m.group(2)) + for m in re.finditer(r"(\S+) rank=\d+ score=(-?\d+)", line)} + if not cur: + continue + out[i] = {name: sc - prev.get(name, 0) for name, sc in cur.items()} + prev = cur + return out + + +def attribute_owners(evs, counters): + """Return (subject_id, other_id) using fire/hit counts, never powers.""" + fires, hits, victim_hits = {}, {}, {} + for o in evs: + t = o.get("type") + if t == "fire": + fires[o["owner"]] = fires.get(o["owner"], 0) + 1 + elif t == "hit": + hits[o["owner"]] = hits.get(o["owner"], 0) + 1 + if "victim" in o: + victim_hits[o["victim"]] = victim_hits.get(o["victim"], 0) + 1 + if not fires: + return None, None + if counters: + strict = [o for o in fires + if fires[o] == counters["fired"] + and hits.get(o, 0) == counters["hits"] + and victim_hits.get(o, 0) == counters["hits_taken"]] + if len(strict) == 1: + subj = strict[0] + return subj, _other(fires, subj) + # relaxed fallbacks (logged by caller via `counters is None` etc.) + cand = [o for o in fires if counters and fires[o] == counters["fired"]] + if len(cand) != 1: + cand = [o for o in fires + if counters and victim_hits.get(o, 0) == counters["hits_taken"]] + if len(cand) != 1: + cand = list(fires) + if len(cand) == 2: + return cand[0], cand[1] + if len(cand) == 1: + return cand[0], _other(fires, cand[0]) + return None, None + + +def _other(owners, subj): + others = [o for o in owners if o != subj] + return others[0] if len(others) == 1 else None + + +def parse_run(armdir, run): + """Everything the report needs for one run; None-free on partial data.""" + evs = parse_events(os.path.join(armdir, f"run{run}.events.jsonl")) + log_text = "".join(read_lines(os.path.join(armdir, f"run{run}.battle.log"))) + counters = subject_counters(log_text) + subj, other = attribute_owners(evs, counters) + + r = {"run": run, "rounds": 0, "wins": None, "first_places": + parse_round_wins_by_name(log_text), "subject_id": subj, + "other_id": other, + "attribution_exact": counters is not None, "shots": 0, "damage": 0.0, + "damage_taken": 0.0, "hits_taken": 0, "hits_dealt": 0, + "death_solved": 0, "death_score_agree": 0, "ambiguous": 0} + if subj is None or other is None: + # still count rounds so the arm shows up, but flags will be set + r["rounds"] = _round_count(armdir, run) + return r + + deaths = {} + for o in evs: + t = o.get("type") + if t == "fire" and o.get("owner") == other: + r["shots"] += 1 + elif t == "hit": + if o.get("owner") == other: + r["damage"] += o.get("damage", 0.0) + r["hits_dealt"] += 1 + if o.get("victim") == other: + r["damage_taken"] += o.get("damage", 0.0) + r["hits_taken"] += 1 + elif t == "death": + deaths.setdefault(o.get("round"), []).append(o.get("victim")) + + nrounds = _round_count(armdir, run) + r["rounds"] = nrounds + scores = parse_round_scores( + os.path.join(armdir, f"run{run}.jsonl.results.json")) + # Per-round winner: primary = death events (a lone death means the OTHER bot + # won). A round with two or zero deaths (a mutual kill or a timeout) cannot + # be resolved from deaths alone, so the name-based per-round score delta is + # the tie-break. The two methods are cross-checked on the unambiguous rounds. + wins = death_solved = agree = ambiguous = 0 + for rd in range(1, nrounds + 1): + vics = deaths.get(rd, []) + winner = None + if len(vics) == 1: + winner = other if vics[0] == subj else subj + death_solved += 1 + delta = scores.get(rd) + if delta and SUBJECT_NAME in delta and BOT_NAME in delta: + score_winner = (subj if delta[SUBJECT_NAME] > delta[BOT_NAME] + else other) + if winner is None: + winner = score_winner + ambiguous += 1 + elif winner == score_winner: + agree += 1 + if winner == other: + wins += 1 + r["wins"] = wins + r["death_solved"] = death_solved + r["death_score_agree"] = agree + r["ambiguous"] = ambiguous + return r + + +def _round_count(armdir, run): + p = os.path.join(armdir, f"run{run}.jsonl.rounds.json") + try: + return len(json.load(open(p)).get("rounds", [])) + except (OSError, json.JSONDecodeError, AttributeError): + pass + # fall back to distinct round numbers in the events sidecar + rds = {o.get("round") for o in + parse_events(os.path.join(armdir, f"run{run}.events.jsonl"))} + return len(rds) + + +# ── statistics ─────────────────────────────────────────────────────────────── + +def perm_test(xa, xb): + """Exact two-sided permutation test on the difference of means. Returns + (obs, p, n_perm, exact_bool).""" + na, nb = len(xa), len(xb) + if na == 0 or nb == 0: + return None + obs = abs(sum(xa) / na - sum(xb) / nb) + pooled = list(xa) + list(xb) + n = na + nb + total_sum = sum(pooled) + ncomb = math.comb(n, na) + if ncomb <= EXACT_CAP: + cnt = 0 + for combo in itertools.combinations(range(n), na): + sa = sum(pooled[i] for i in combo) + if abs(sa / na - (total_sum - sa) / nb) >= obs - 1e-9: + cnt += 1 + return obs, cnt / ncomb, ncomb, True + # deterministic fallback for very large n (never hit at the default 7 runs) + import random + rng = random.Random(0xA1B2C3) + B = 200_000 + cnt = 0 + for _ in range(B): + idx = rng.sample(range(n), na) + sa = sum(pooled[i] for i in idx) + if abs(sa / na - (total_sum - sa) / nb) >= obs - 1e-9: + cnt += 1 + return obs, (cnt + 1) / (B + 1), ncomb, False + + +def fisher_two_sided(a, b, c, d): + """Two-sided Fisher exact on [[a,b],[c,d]].""" + n = a + b + c + d + if n == 0: + return None + row1, col1, col2 = a + b, a + c, b + d + denom = math.comb(n, row1) + if denom == 0: + return None + + def hyper(x): + return math.comb(col1, x) * math.comb(col2, row1 - x) / denom + + lo = max(0, col1 - (c + d)) + hi = min(row1, col1) + obs = hyper(a) + return sum(hyper(x) for x in range(lo, hi + 1) if hyper(x) <= obs + 1e-12) + + +# ── liveness ───────────────────────────────────────────────────────────────── + +def liveness(armdir, runs, env): + """OK iff every declared env var appears verbatim in the bot's boot report + of every run, the report itself ran, and the bot did not warn that the name + is unrecognised (an unknown TR_*/GUN_* var is silently ignored).""" + if not runs: + return "FAIL", "no runs" + report_seen = {r: False for r in runs} + matched = {r: True for r in runs} + warned = {r: [] for r in runs} + for r in runs: + text = "".join(read_lines(os.path.join( + armdir, f"run{r}.bot.stdout.log"))) + report_seen[r] = "=== ENVIRONMENT (boot report) ===" in text + for k, v in env.items(): + if f"[env] {k}={v}" not in text: + matched[r] = False + if f"[env] WARNING: {k}" in text: + warned[r].append(k) + dead = [r for r in runs if not report_seen[r]] + miss = [r for r in runs if not matched[r]] + if dead: + return "FAIL", f"no bot env report in run(s) {dead}" + warn_runs = [r for r in runs if warned[r]] + if warn_runs: + names = sorted({k for r in warn_runs for k in warned[r]}) + return "FAIL", (f"bot ignored arm env {names} (unrecognised) in " + f"run(s) {warn_runs}") + if env and miss: + vars_txt = " ".join(f"{k}={v}" for k, v in env.items()) + return "FAIL", f"arm env '{vars_txt}' not in boot report of run(s) {miss}" + if not env: + return "OK", f"{len(runs)}/{len(runs)} runs: no arm env; report present" + vars_txt = " ".join(f"{k}={v}" for k, v in env.items()) + return "OK", f"{len(runs)}/{len(runs)} runs: {vars_txt} applied" + + +# ── report ─────────────────────────────────────────────────────────────────── + +def main(): + args = sys.argv[1:] + if not args or args[0] in ("-h", "--help"): + print(__doc__) + return 0 if args else 2 + outdir = args[0] + reference = None + if "--reference" in args: + reference = args[args.index("--reference") + 1] + if not os.path.isdir(outdir): + print(f"ERROR: not a directory: {outdir}", file=sys.stderr) + return 2 + + session = load_session(outdir) + arms = discover_arms(outdir, session) + if not arms: + print(f"ERROR: no arms found in {outdir}", file=sys.stderr) + return 2 + if reference is None: + reference = arms[0]["name"] + + data = {} + for arm in arms: + armdir = os.path.join(outdir, arm["name"]) + runs = discover_runs(armdir) + per = [parse_run(armdir, r) for r in runs] + data[arm["name"]] = {"dir": armdir, "runs": runs, "per": per, + "env": arm["env"], "label": arm["label"]} + + if session: + print(f"# session {outdir}") + print(f"# commit={session.get('commit','?')} " + f"binary_sha256={session.get('binary_sha256','?')} " + f"rounds={session.get('rounds','?')} runs={session.get('runs','?')} " + f"conc={session.get('conc','?')} ts={session.get('timestamp','?')}") + print() + + # ── summary table ──────────────────────────────────────────────────────── + print("ARM SUMMARY") + hdr = (f"{'arm':<14} {'runs':>4} {'dmg/run':>8} {'dmgtk/run':>9} " + f"{'wins':>7} {'win%':>6} {'shots/run':>9} {'hitstk/run':>10}") + print(hdr) + print("-" * len(hdr)) + for arm in arms: + name = arm["name"] + per = data[name]["per"] + n = len(per) + if n == 0: + print(f"{name:<14} {0:>4} (no runs)") + continue + dmg = sum(p["damage"] for p in per) / n + dmgv = sum(p["damage_taken"] for p in per) / n + shots = sum(p["shots"] for p in per) / n + htk = sum(p["hits_taken"] for p in per) / n + wins = sum(p["wins"] for p in per if p["wins"] is not None) + rounds = sum(p["rounds"] for p in per) + pct = f"{100.0 * wins / rounds:.1f}" if rounds else "n/a" + print(f"{name:<14} {n:>4} {dmg:>8.0f} {dmgv:>9.0f} " + f"{str(wins) + '/' + str(rounds):>7} {pct:>6} {shots:>9.0f} {htk:>10.1f}") + + # ── per-run values ─────────────────────────────────────────────────────── + print("\nPER-RUN (never just the mean)") + for arm in arms: + name = arm["name"] + per = data[name]["per"] + dmgs = " ".join( + f"r{p['run']}={p['damage']:.0f}" for p in per) + wins = " ".join( + f"r{p['run']}={p['wins']}/{p['rounds']}" if p["wins"] is not None + else f"r{p['run']}=?/{p['rounds']}" for p in per) + print(f" {name:<14} dmg: {dmgs}") + print(f" {'':<14} wins: {wins}") + + # ── exact permutation tests ────────────────────────────────────────────── + if reference not in data: + print(f"\nWARNING: reference arm '{reference}' not found; skipping tests") + return 1 + print(f"\nEXACT TWO-SIDED PERMUTATION TEST (per-run values) vs `{reference}`") + print(f"{'metric':<12} {'arm':<14} {'obs(diff)':>12} {'p':>8} permutations") + print("-" * 64) + ref = data[reference]["per"] + for arm in arms: + name = arm["name"] + if name == reference: + continue + for label, key in (("dmg/run", "damage"), ("round wins", "wins")): + xa = [p[key] for p in ref if p[key] is not None] + xb = [p[key] for p in data[name]["per"] if p[key] is not None] + res = perm_test(xa, xb) + if res is None: + print(f"{label:<12} {name:<14} {'n/a':>12} {'n/a':>8}") + continue + obs, p, ncomb, exact = res + note = f"C({len(xa)+len(xb)},{len(xa)})={ncomb}" + ( + "" if exact else " SAMPLED") + print(f"{label:<12} {name:<14} {obs:>+12.3f} {p:>8.4f} {note}") + + # ── round-level test (anti-conservative) ───────────────────────────────── + print("\nROUND-LEVEL TEST (pooled rounds, Fisher exact) vs " + f"`{reference}` — ANTI-CONSERVATIVE: rounds cluster within runs") + print(f"{'arm':<14} {'ref wins':>10} {'arm wins':>10} {'p':>8}") + print("-" * 46) + refw = sum(p["wins"] for p in ref if p["wins"] is not None) + refr = sum(p["rounds"] for p in ref) + for arm in arms: + name = arm["name"] + if name == reference: + continue + per = data[name]["per"] + w = sum(p["wins"] for p in per if p["wins"] is not None) + rr = sum(p["rounds"] for p in per) + p = fisher_two_sided(w, rr - w, refw, refr - refw) + pstr = f"{p:.4f}" if p is not None else "n/a" + print(f"{name:<14} {str(refw) + '/' + str(refr):>10} " + f"{str(w) + '/' + str(rr):>10} {pstr:>8}") + + # ── liveness ───────────────────────────────────────────────────────────── + print("\nLIVENESS (arm env applied in the bot's own boot report)") + lv_fail = 0 + for arm in arms: + name = arm["name"] + status, why = liveness(data[name]["dir"], data[name]["runs"], + data[name]["env"]) + if status == "FAIL": + lv_fail += 1 + print(f" {name:<14} {status:<4} ({why})") + + # ── round-win attribution cross-check ──────────────────────────────────── + print("\nROUND-WIN ATTRIBUTION (events primary; score tie-break for " + "mutual-kill / timeout rounds)") + for arm in arms: + name = arm["name"] + runs_ok = runs_tot = 0 + dsolve = dagree = damb = 0 + for p in data[name]["per"]: + if p["wins"] is not None and p["first_places"] is not None: + runs_tot += 1 + if p["wins"] == p["first_places"]: + runs_ok += 1 + dsolve += p["death_solved"] + dagree += p["death_score_agree"] + damb += p["ambiguous"] + status = ("OK" if runs_tot and runs_ok == runs_tot + else ("n/a" if not runs_tot else "MISMATCH")) + print(f" {name:<14} wins==firstPlaces {runs_ok}/{runs_tot} runs {status}; " + f"single-death rounds agree with score {dagree}/{dsolve} " + f"({damb} tie-broken)") + + return 1 if lv_fail else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tools/ab/ab_run.sh b/tools/ab/ab_run.sh new file mode 100755 index 0000000..b161cf3 --- /dev/null +++ b/tools/ab/ab_run.sh @@ -0,0 +1,270 @@ +#!/usr/bin/env bash +# ───────────────────────────────────────────────────────────────────────────── +# ab_run.sh — run an A/B session of N arms vs real DrussGT. +# +# tools/ab/ab_run.sh --arms arms.txt --runs 7 --outdir /tmp/ab/foo --conc 7 +# +# What it does, once per SESSION (not once per arm): +# 1. builds ONE frozen ModularBot from the CURRENT HEAD via `git archive HEAD` +# (so a dirty tree cannot leak into the measurement) and records the commit +# sha + binary sha256; +# 2. for every arm × run, launches a real battle against DrussGT in parallel, +# each with its own ports (the runner picks ephemeral ones), its own +# DrussGT botdir/data, its own ModularBot botdir and its own output files. +# +# Arm file format (one arm per line): +# name | ENV_VAR=value ENV_VAR2=value2 | optional label +# `name` alone or `name |` are also accepted (no arm env). Blank lines and +# lines starting with `#` are ignored. See arms.example.txt. +# +# Output layout (what ab_analyze.py reads): +# /session.json commit, binary sha, arms, runs… +# //run.jsonl tick capture +# //run.jsonl.rounds.json round tick boundaries +# //run.jsonl.results.json +# //run.events.jsonl fire/hit/death sidecar +# //run.battle.log capture stdout (firstPlaces, …) +# //run.bot.stdout.log ModularBot boot env report (liveness) +# +# Cleanup: every job runs in its own process group (setsid); on EXIT/INT/TERM +# this script SIGTERMs/SIGKILLs those groups and, as a backstop, pkills anything +# whose command line references this session's outdir. So a Ctrl-C does not +# leave orphan battles running. +# ───────────────────────────────────────────────────────────────────────────── +set -euo pipefail + +REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +RUN_BRIDGE="$REPO/tools/robocode_shim/run_bridge_battle.sh" +MAKE_BOTDIR="$REPO/tools/robocode_shim/make_botdir.sh" + +ARMS_FILE="" +RUNS=7 +CONC=7 +OUTDIR="" +ROUNDS=7 + +usage() { + sed -n '2,40p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//' + exit "${1:-0}" +} + +while [[ $# -gt 0 ]]; do + case "$1" in + --arms) ARMS_FILE="$2"; shift 2;; + --runs) RUNS="$2"; shift 2;; + --outdir) OUTDIR="$2"; shift 2;; + --conc) CONC="$2"; shift 2;; + --rounds) ROUNDS="$2"; shift 2;; + -h|--help) usage 0;; + *) echo "unknown argument: $1" >&2; usage 1;; + esac +done + +[[ -n "$ARMS_FILE" ]] || { echo "ERROR: --arms is required" >&2; usage 1; } +[[ -f "$ARMS_FILE" ]] || { echo "ERROR: arm file not found: $ARMS_FILE" >&2; exit 1; } +[[ -n "$OUTDIR" ]] || { echo "ERROR: --outdir is required" >&2; usage 1; } +[[ "$OUTDIR" != "/" && "$OUTDIR" != "" ]] || { echo "ERROR: refusing outdir '$OUTDIR'" >&2; exit 1; } +OUTDIR="$(mkdir -p "$OUTDIR" && cd "$OUTDIR" && pwd)" + +# ── prerequisites ──────────────────────────────────────────────────────────── +ROBOCODE_JAR="${ROBOCODE_JAR:-/tmp/robocode/install/libs/robocode.jar}" +DRUSSGT_JAR="${DRUSSGT_JAR:-/tmp/drussgt/DrussGT.jar}" +RUNNER_JAR="${TR_RUNNER_JAR:-/home/davide/Projects/tank-royale/runner/examples/lib/robocode-tankroyale-runner.jar}" +BOT_API_JAR="${TR_BOT_API_JAR:-$HOME/Downloads/sample-bots-java-1.0.2/lib/robocode-tankroyale-bot-api-1.0.2.jar}" +DRUSSGT_SHARED="${TR_DRUSSGT_BOTDIR:-/tmp/tr_bots/DrussGT}" + +missing=0 +for f in "$ROBOCODE_JAR" "$DRUSSGT_JAR" "$RUNNER_JAR" "$BOT_API_JAR"; do + if [[ ! -f "$f" ]]; then echo "ERROR: missing prerequisite: $f" >&2; missing=1; fi +done +command -v nim >/dev/null 2>&1 || { echo "ERROR: nim not on PATH" >&2; missing=1; } +command -v git >/dev/null 2>&1 || { echo "ERROR: git not on PATH" >&2; missing=1; } +[[ -f "$RUN_BRIDGE" ]] || { echo "ERROR: missing $RUN_BRIDGE" >&2; missing=1; } +[[ -f "$MAKE_BOTDIR" ]] || { echo "ERROR: missing $MAKE_BOTDIR" >&2; missing=1; } +(( missing == 0 )) || { echo "ERROR: prerequisites missing — not starting a broken session" >&2; exit 1; } + +if [[ ! -d "$DRUSSGT_SHARED" ]]; then + echo "[ab] recreating shared DrussGT botdir: $DRUSSGT_SHARED" + "$MAKE_BOTDIR" "$DRUSSGT_SHARED" >/dev/null +fi + +# ── parse the arm file into parallel arrays ────────────────────────────────── +trim() { local s="$1"; s="${s#"${s%%[![:space:]]*}"}"; s="${s%"${s##*[![:space:]]}"}"; printf '%s' "$s"; } + +ARM_NAMES=(); ARM_ENVS=(); ARM_LABELS=() +while IFS= read -r line || [[ -n "$line" ]]; do + [[ -z "${line//[[:space:]]/}" ]] && continue + [[ "$line" =~ ^[[:space:]]*# ]] && continue + name="$(trim "${line%%|*}")" + rest="${line#*|}" + if [[ "$line" == *"|"* ]]; then + envspec="$(trim "${rest%%|*}")" + label="$(trim "${rest#*|}")" + else + envspec=""; label="" + fi + [[ -n "$name" ]] || { echo "ERROR: arm with empty name in $ARMS_FILE" >&2; exit 1; } + ARM_NAMES+=("$name"); ARM_ENVS+=("$envspec"); ARM_LABELS+=("$label") +done < "$ARMS_FILE" +(( ${#ARM_NAMES[@]} > 0 )) || { echo "ERROR: no arms parsed from $ARMS_FILE" >&2; exit 1; } +# duplicate names would silently overwrite each other's run files +dupes="$(printf '%s\n' "${ARM_NAMES[@]}" | sort | uniq -d)" +[[ -z "$dupes" ]] || { echo "ERROR: duplicate arm name(s): $dupes" >&2; exit 1; } + +# ── fresh session dir ──────────────────────────────────────────────────────── +rm -rf "$OUTDIR" +mkdir -p "$OUTDIR" +WORK="$OUTDIR/.work"; mkdir -p "$WORK" + +# ── build ONE frozen bot from HEAD ─────────────────────────────────────────── +COMMIT="$(git -C "$REPO" rev-parse HEAD)" +FROZEN_DIR="$OUTDIR/frozen/ModularBot" +FROZEN_BIN="$FROZEN_DIR/ModularBot_bin" +FROZEN_JSON="$FROZEN_DIR/ModularBot.json" +mkdir -p "$FROZEN_DIR" +BUILDDIR="$WORK/head" +mkdir -p "$BUILDDIR" +echo "[ab] exporting HEAD ($COMMIT) -> $BUILDDIR" +git -C "$REPO" archive HEAD | tar -x -C "$BUILDDIR" + +echo "[ab] building frozen ModularBot (nim c -d:release)…" +( cd "$BUILDDIR/ModularBot_garage" && \ + nim c -d:release --nimcache:"$OUTDIR/nimcache" --out:"$FROZEN_BIN" src/ModularBot.nim ) +[[ -x "$FROZEN_BIN" ]] || { echo "ERROR: frozen build failed" >&2; exit 1; } +BINSHA="$(sha256sum "$FROZEN_BIN" | awk '{print $1}')" +echo "[ab] frozen binary sha256=$BINSHA" + +# canonical bot identity for every arm (the dir is /named/ ModularBot) +cat > "$FROZEN_JSON" <<'JSON' +{ + "name": "ModularBot", + "version": "0.1.0", + "authors": ["Davide Cappellini"], + "description": "frozen A/B build of ModularBot (ab_run.sh)", + "homepage": "", + "countryCodes": ["IT"], + "gameTypes": ["classic", "1v1"], + "platform": "Nim", + "programmingLang": "Nim" +} +JSON + +# ── session.json ───────────────────────────────────────────────────────────── +json_esc() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; } +{ + printf '{\n' + printf ' "commit": "%s",\n' "$COMMIT" + printf ' "binary_sha256": "%s",\n' "$BINSHA" + printf ' "binary": "frozen/ModularBot/ModularBot_bin",\n' + printf ' "rounds": %d,\n' "$ROUNDS" + printf ' "runs": %d,\n' "$RUNS" + printf ' "conc": %d,\n' "$CONC" + printf ' "timestamp": "%s",\n' "$(date -Is)" + printf ' "outdir": "%s",\n' "$(json_esc "$OUTDIR")" + printf ' "arms": [\n' + for i in "${!ARM_NAMES[@]}"; do + printf ' {"name": "%s", "env": "%s", "label": "%s"}' \ + "$(json_esc "${ARM_NAMES[$i]}")" "$(json_esc "${ARM_ENVS[$i]}")" "$(json_esc "${ARM_LABELS[$i]}")" + (( i + 1 < ${#ARM_NAMES[@]} )) && printf ',' || true + printf '\n' + done + printf ' ]\n}\n' +} > "$OUTDIR/session.json" + +# ── one job (run in its own process group) ─────────────────────────────────── +# exported so the detached `bash -c` subshell can call it; config via env +export REPO RUN_BRIDGE ROUNDS OUTDIR WORK FROZEN_BIN FROZEN_JSON + +ab_run_one() { + local arm="$1" run="$2" envspec="$3" + local dir="$OUTDIR/$arm" + local work="$WORK/$arm/run$run" + local botdir="$work/bots/ModularBot" + mkdir -p "$dir" "$botdir" "$work/drussgt" "$work/data" + + cp "$FROZEN_JSON" "$botdir/ModularBot.json" + cat > "$botdir/ModularBot.sh" < "$dir/run$run.bot.stdout.log" 2> "$dir/run$run.bot.stderr.log" +SH + chmod +x "$botdir/ModularBot.sh" + ln -sf "$FROZEN_BIN" "$botdir/ModularBot_bin" + + rm -f "$dir/run$run.jsonl" "$dir/run$run.jsonl.rounds.json" \ + "$dir/run$run.jsonl.results.json" "$dir/run$run.events.jsonl" \ + "$dir/run$run.status" + + local rc=0 + # shellcheck disable=SC2086 # envspec is intentionally word-split + env $envspec \ + DRUSSGT_BOTDIR="$work/drussgt/DrussGT" \ + DRUSSGT_DATA="$work/data" \ + AB_SESSION="$OUTDIR" \ + TR_EVENTS_OUT="$dir/run$run.events.jsonl" \ + timeout 900 "$RUN_BRIDGE" "$botdir" "$ROUNDS" "$dir/run$run.jsonl" \ + > "$dir/run$run.battle.log" 2>&1 || rc=$? + echo "$rc" > "$dir/run$run.status" + return 0 +} +export -f ab_run_one + +# ── cleanup: kill this session's children ──────────────────────────────────── +JOB_PIDS=() +cleanup() { + local rc=$? + trap - EXIT INT TERM + for p in "${JOB_PIDS[@]:-}"; do + [[ -n "$p" ]] || continue + kill -TERM -- "-$p" 2>/dev/null || kill -TERM "$p" 2>/dev/null || true + done + sleep 0.5 + for p in "${JOB_PIDS[@]:-}"; do + [[ -n "$p" ]] || continue + kill -KILL -- "-$p" 2>/dev/null || true + done + # backstop: any survivor whose cmdline references this session's outdir. + # `[r]un_…` cannot match this script's own pkill command line. + pkill -f "[r]un_bridge_battle.sh.*$OUTDIR" 2>/dev/null || true + pkill -f "[r]obocode_shim.TrBattleCapture.*$OUTDIR" 2>/dev/null || true + exit "$rc" +} +trap cleanup EXIT INT TERM + +# ── launch the jobs, K at a time ───────────────────────────────────────────── +echo "[ab] session: ${#ARM_NAMES[@]} arms × $RUNS runs, rounds=$ROUNDS, conc=$CONC" +echo "[ab] outdir: $OUTDIR" +RUNNING=0 +for ai in "${!ARM_NAMES[@]}"; do + arm="${ARM_NAMES[$ai]}"; envspec="${ARM_ENVS[$ai]}" + for (( r=1; r<=RUNS; r++ )); do + setsid bash -c 'ab_run_one "$1" "$2" "$3"' _ "$arm" "$r" "$envspec" & + JOB_PIDS+=("$!") + RUNNING=$((RUNNING + 1)) + if (( RUNNING >= CONC )); then + wait -n || true + RUNNING=$((RUNNING - 1)) + fi + done +done +wait || true + +# jobs are done — do not let the EXIT trap signal reused PIDs +JOB_PIDS=() + +# ── aggregate status ───────────────────────────────────────────────────────── +FAILED=0 +for ai in "${!ARM_NAMES[@]}"; do + arm="${ARM_NAMES[$ai]}" + for (( r=1; r<=RUNS; r++ )); do + st="$(cat "$OUTDIR/$arm/run$r.status" 2>/dev/null || echo 999)" + if [[ "$st" != "0" ]]; then + echo "[ab] FAIL $arm run$r (rc=$st) — see $OUTDIR/$arm/run$r.battle.log" >&2 + FAILED=$((FAILED + 1)) + fi + done +done + +echo "[ab] done: $(( ${#ARM_NAMES[@]} * RUNS - FAILED )) ok, $FAILED failed" +echo "[ab] analyze with: python3 $REPO/tools/ab/ab_analyze.py $OUTDIR" +(( FAILED == 0 )) || exit 1 diff --git a/tools/ab/arms.example.txt b/tools/ab/arms.example.txt new file mode 100644 index 0000000..371753d --- /dev/null +++ b/tools/ab/arms.example.txt @@ -0,0 +1,15 @@ +# A/B arm file — one arm per line. Passed to ab_run.sh via --arms. +# +# Format: +# name | ENV_VAR=value ENV_VAR2=value2 | optional label +# +# * `name` must be unique; it becomes //. +# * the env part is exported into the battle process (so the bot inherits it); +# ab_analyze.py verifies each pair shows up in the bot's boot env report. +# * the first arm is the default reference for the permutation test. +# * `name` alone, or `name |`, is a control arm with no env override. +# * blank lines and lines starting with `#` are ignored. + +control | | unmodified power policy (TR_POWER_ENERGY_MIN default 0.5) +pmin100 | TR_POWER_ENERGY_MIN=1.0 | never fire below power 1.0 +pmin075 | TR_POWER_ENERGY_MIN=0.75 | never fire below power 0.75