Gun mixing (TMHorizon+BitBrain) vs DrussGT: clean negative, no dodge disruption
4 arms x 7 runs vs real DrussGT. mix alternates the two guns 476 times/7 runs (liveness OK) but our bullets are no more varied (power sd / aim-offset sd flat) and DrussGT's dodge quality is unchanged (miss/tick mix-pat +0.03, p=0.66; MDE 3.8%). mix wins 24/49 = the 49% baseline; the user's 6/10 has P=0.353 at 49%. New tools/ab/ab_dodge_analyze.py splits the validated per-shot dodge instrument by arm and adds gun-switch/power/bearing liveness; fixtures committed.
This commit is contained in:
@@ -0,0 +1,417 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Per-arm DrussGT dodge quality from an `ab_run.sh` session.
|
||||
|
||||
The gun-mixing hypothesis is: admitting TMHorizon AND BitBrain makes DrussGT's
|
||||
surfgun dodge *worse*, because two guns that disagree on the same tick fire
|
||||
inconsistent bullets and poison its wave matching. The random 7-round win
|
||||
counts an A/B session produces cannot see that (a few rounds, huge variance);
|
||||
the shots can. This reads the SAME live captures `ab_analyze.py` uses, applies
|
||||
the SAME per-shot instrument as
|
||||
`common_libs/tests/analyze_drussgt_dodge_vs_power.py` (miss at arrival, miss per
|
||||
flight tick, fixed-window lateral displacement, and the bullet-free CONTROL
|
||||
window), and splits it PER ARM.
|
||||
|
||||
It is a thin driver: it imports the validated instrument, only adds (a) the
|
||||
per-arm split, (b) an aim-offset / power dispersion reading of OUR OWN shooting
|
||||
(the liveness check for the mechanism), and (c) between-arm tests whose
|
||||
uncertainty is clustered by RUN (7 runs per arm -> C(14,7)=3432 exact).
|
||||
The power-neutral metrics (miss/tick, fixed/control-window lateral displacement)
|
||||
are the ones to trust; raw miss distance is dominated by our own aim error and
|
||||
by the flight window (see docs/drussgt_dodge_vs_power.md).
|
||||
|
||||
Usage:
|
||||
python3 tools/ab/ab_dodge_analyze.py /tmp/ab/mix [--json out.json] [--reps 2000]
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import collections
|
||||
import glob
|
||||
import itertools
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import random
|
||||
import re
|
||||
import statistics
|
||||
import sys
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
REPO = os.path.dirname(os.path.dirname(HERE))
|
||||
sys.path.insert(0, os.path.join(REPO, "common_libs", "tests"))
|
||||
import analyze_drussgt_dodge_vs_power as dodge # noqa: E402
|
||||
|
||||
BAND_LABELS = dodge.BAND_LABELS # 0-100 .. 450+
|
||||
STRAT_BANDS = ["200-300", "300-450", "450+"] # bands with enough shots for a contrast
|
||||
DODGE_METRICS = ["miss", "miss_per_tick", "lat_fixed_abs", "lat_ctrl_abs",
|
||||
"lat_disp_per_tick", "hit"]
|
||||
TEST_METRICS = ["miss", "miss_per_tick", "lat_fixed_abs", "lat_ctrl_abs", "hit"]
|
||||
|
||||
|
||||
class ArmRun(dodge.Run):
|
||||
"""The validated instrument, plus the aim offset we fired at.
|
||||
|
||||
`aim_off` = (fired direction - bearing to DrussGT at the fire tick), in
|
||||
degrees, wrapped to (-180, 180]. It is the gun's answer relative to the
|
||||
straight-at-target line: its dispersion across shots is how *inconsistent*
|
||||
our bullets are, which is exactly the property that would poison a surfgun.
|
||||
"""
|
||||
|
||||
def _shot(self, ev, t0):
|
||||
s = super()._shot(ev, t0)
|
||||
if s is None:
|
||||
return None
|
||||
r0 = self.by_tick[t0]
|
||||
bearing = math.degrees(math.atan2(r0["ey"] - s["_y"], r0["ex"] - s["_x"]))
|
||||
s["aim_off"] = ((ev["dir"] - bearing + 180.0) % 360.0) - 180.0
|
||||
return s
|
||||
|
||||
|
||||
# ── discovery ────────────────────────────────────────────────────────────────
|
||||
def discover_arm(armdir):
|
||||
runs = []
|
||||
for cap in sorted(glob.glob(os.path.join(armdir, "run*.jsonl"))):
|
||||
if cap.endswith(".events.jsonl"):
|
||||
continue
|
||||
ev = cap[:-len(".jsonl")] + ".events.jsonl"
|
||||
rj = cap + ".rounds.json"
|
||||
if os.path.exists(ev) and os.path.exists(rj):
|
||||
runs.append(ArmRun(cap, ev, rj))
|
||||
return runs
|
||||
|
||||
|
||||
# ── shooting-variation (liveness for the mechanism) ──────────────────────────
|
||||
def iqr(xs):
|
||||
if len(xs) < 4:
|
||||
return float("nan")
|
||||
xs = sorted(xs)
|
||||
return dodge.pct(xs, 0.75) - dodge.pct(xs, 0.25)
|
||||
|
||||
|
||||
def shooting_variation(shots):
|
||||
pows = [s["power"] for s in shots]
|
||||
offs = [s["aim_off"] for s in shots]
|
||||
ranges = [s["range"] for s in shots]
|
||||
pwr_hist = collections.Counter(round(p, 2) for p in pows)
|
||||
off_hist = collections.Counter(round(o / 5.0) * 5 for o in offs)
|
||||
return {
|
||||
"power_mean": dodge.mean(pows),
|
||||
"power_sd": statistics.pstdev(pows) if len(pows) > 1 else 0.0,
|
||||
"power_iqr": iqr(pows),
|
||||
"power_distinct": len(set(pows)),
|
||||
"aimoff_mean": dodge.mean(offs),
|
||||
"aimoff_sd": statistics.pstdev(offs) if len(offs) > 1 else 0.0,
|
||||
"aimoff_iqr": iqr(offs),
|
||||
"range_mean": dodge.mean(ranges),
|
||||
"power_hist": dict(pwr_hist.most_common(10)),
|
||||
"aimoff_hist": dict(sorted(off_hist.items())),
|
||||
}
|
||||
|
||||
|
||||
# ── gun-selection liveness (does the mix actually alternate guns?) ────────────
|
||||
CONFIG_GUN_RE = re.compile(r"gun=([A-Za-z]+)")
|
||||
ANSI_RE = re.compile(r"\x1b\[[0-9;]*m")
|
||||
|
||||
|
||||
def gun_liveness(runs):
|
||||
"""From the `[config] gun=<name>` lines the bot emits when its selection
|
||||
changes: how often each gun was selected and how many times the selection
|
||||
switched. A one-gun arm never switches; a real two-gun mix switches often.
|
||||
"""
|
||||
total = collections.Counter()
|
||||
switches = runs_with = 0
|
||||
for r in runs:
|
||||
armdir = os.path.dirname(r.cap_path)
|
||||
num = os.path.basename(r.cap_path)[len("run"):-len(".jsonl")]
|
||||
path = os.path.join(armdir, "run%s.bot.stdout.log" % num)
|
||||
if not os.path.exists(path):
|
||||
continue
|
||||
seq = []
|
||||
for line in open(path, errors="replace"):
|
||||
if "[config]" not in line:
|
||||
continue
|
||||
m = CONFIG_GUN_RE.search(ANSI_RE.sub("", line))
|
||||
if m:
|
||||
seq.append(m.group(1))
|
||||
if seq:
|
||||
runs_with += 1
|
||||
total.update(seq)
|
||||
switches += sum(1 for i in range(1, len(seq)) if seq[i] != seq[i - 1])
|
||||
return total, switches, runs_with, len(runs)
|
||||
|
||||
|
||||
# ── per-run summaries + clustered tests ──────────────────────────────────────
|
||||
def run_summaries(runs_shots, key):
|
||||
"""Per run: (sum, count) for `key`, and per range band."""
|
||||
out = []
|
||||
for ss in runs_shots:
|
||||
vals = [s[key] for s in ss if s.get(key) is not None]
|
||||
bands = {b: [0.0, 0] for b in BAND_LABELS}
|
||||
for s in ss:
|
||||
v = s.get(key)
|
||||
if v is None:
|
||||
continue
|
||||
bands[s["band"]][0] += v
|
||||
bands[s["band"]][1] += 1
|
||||
out.append({"sum": sum(vals), "n": len(vals),
|
||||
"bands": {b: tuple(v) for b, v in bands.items()}})
|
||||
return out
|
||||
|
||||
|
||||
def cluster_ci(summary, reps=2000, seed=1):
|
||||
"""95% CI of the pooled mean, resampling RUNS with replacement."""
|
||||
rng = random.Random(seed)
|
||||
R = len(summary)
|
||||
if R == 0:
|
||||
return float("nan"), float("nan")
|
||||
means = []
|
||||
for _ in range(reps):
|
||||
s = c = 0
|
||||
for _ in range(R):
|
||||
it = summary[rng.randrange(R)]
|
||||
s += it["sum"]
|
||||
c += it["n"]
|
||||
if c:
|
||||
means.append(s / c)
|
||||
means.sort()
|
||||
return dodge.pct(means, 0.025), dodge.pct(means, 0.975)
|
||||
|
||||
|
||||
def _pooled(summary, idx):
|
||||
s = c = 0
|
||||
for i in idx:
|
||||
s += summary[i]["sum"]
|
||||
c += summary[i]["n"]
|
||||
return s, c
|
||||
|
||||
|
||||
def _strat(summary, idx, rest, bands):
|
||||
num = den = 0.0
|
||||
for b in bands:
|
||||
sa = ca = sb = cb = 0.0
|
||||
for i in idx:
|
||||
sa += summary[i]["bands"][b][0]
|
||||
ca += summary[i]["bands"][b][1]
|
||||
for i in rest:
|
||||
sb += summary[i]["bands"][b][0]
|
||||
cb += summary[i]["bands"][b][1]
|
||||
w = ca + cb
|
||||
if ca > 0 and cb > 0:
|
||||
num += w * (sa / ca - sb / cb)
|
||||
den += w
|
||||
return num / den if den else float("nan")
|
||||
|
||||
|
||||
EXACT_CAP = 20_000_000 # 7v7 -> C(14,7)=3432 exact; 30v30 -> Monte-Carlo
|
||||
MC_DRAWS = 1_000_000
|
||||
MC_SEED = 0x5EED5EED
|
||||
|
||||
|
||||
def cluster_perm(summary_a, summary_b, stat):
|
||||
"""Two-sided run-cluster permutation test.
|
||||
|
||||
`stat(summary, idxA, idxB)` is evaluated on the observed split and on
|
||||
relabellings of the pooled RUNS; the null is the difference itself, centred
|
||||
at 0, so p = P(|stat_perm| >= |stat_obs|). Clustering by run keeps the
|
||||
within-run shot correlation, which a shot-level test would ignore. Full
|
||||
enumeration when C(2R,R) <= EXACT_CAP (7v7 -> 3432, exact); otherwise a
|
||||
fixed-seed Monte-Carlo run (reported as such).
|
||||
"""
|
||||
summary = list(summary_a) + list(summary_b)
|
||||
n, na = len(summary), len(summary_a)
|
||||
obs = stat(summary, range(na), range(na, n))
|
||||
ncomb = math.comb(n, na)
|
||||
if ncomb <= EXACT_CAP:
|
||||
cnt = 0
|
||||
for combo in itertools.combinations(range(n), na):
|
||||
cset = set(combo)
|
||||
rest = [i for i in range(n) if i not in cset]
|
||||
v = stat(summary, combo, rest)
|
||||
if v == v and abs(v) >= abs(obs) - 1e-12:
|
||||
cnt += 1
|
||||
return obs, (cnt + 1) / (ncomb + 1), "exact"
|
||||
rng = random.Random(MC_SEED)
|
||||
cnt = 0
|
||||
for _ in range(MC_DRAWS):
|
||||
idx = rng.sample(range(n), na)
|
||||
cset = set(idx)
|
||||
rest = [i for i in range(n) if i not in cset]
|
||||
v = stat(summary, idx, rest)
|
||||
if v == v and abs(v) >= abs(obs) - 1e-12:
|
||||
cnt += 1
|
||||
return obs, (cnt + 1) / (MC_DRAWS + 1), "monte-carlo/%d" % MC_DRAWS
|
||||
|
||||
|
||||
def pooled_stat(summary, idx, rest):
|
||||
sa, ca = _pooled(summary, idx)
|
||||
sb, cb = _pooled(summary, rest)
|
||||
if ca == 0 or cb == 0:
|
||||
return float("nan")
|
||||
return sa / ca - sb / cb
|
||||
|
||||
|
||||
def strat_stat(summary, idx, rest):
|
||||
return _strat(summary, idx, rest, STRAT_BANDS)
|
||||
|
||||
|
||||
# ── report ───────────────────────────────────────────────────────────────────
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("outdir")
|
||||
ap.add_argument("--json", default=None)
|
||||
ap.add_argument("--reps", type=int, default=2000)
|
||||
ap.add_argument("--reference", default=None)
|
||||
a = ap.parse_args()
|
||||
|
||||
session = json.load(open(os.path.join(a.outdir, "session.json")))
|
||||
arm_names = [x["name"] for x in session["arms"]]
|
||||
ref = a.reference or arm_names[0]
|
||||
|
||||
arms = {}
|
||||
for name in arm_names:
|
||||
runs = discover_arm(os.path.join(a.outdir, name))
|
||||
runs_shots = [[s for s in r.shots()] for r in runs]
|
||||
shots = [s for ss in runs_shots for s in ss]
|
||||
rounds = sum(len(r.rounds) for r in runs)
|
||||
arms[name] = {"runs": runs, "runs_shots": runs_shots, "shots": shots,
|
||||
"rounds": rounds,
|
||||
"variation": shooting_variation(shots),
|
||||
"guns": gun_liveness(runs)}
|
||||
|
||||
out = {"commit": session["commit"], "runs": session["runs"],
|
||||
"rounds": session["rounds"], "arms": {}}
|
||||
|
||||
# ── corpus sanity: attribution is per battle, so restate the death check ──
|
||||
bad = tot = 0
|
||||
for name in arm_names:
|
||||
for r in arms[name]["runs"]:
|
||||
for ev in r.events:
|
||||
if ev.get("type") != "death":
|
||||
continue
|
||||
end = r.start[ev["round"]] + r.count[ev["round"]] - 1
|
||||
row = r.by_tick.get(end)
|
||||
if row is None:
|
||||
continue
|
||||
tot += 1
|
||||
vs = r.owner_side.get(ev["victim"])
|
||||
if (vs == "e" and row["ee"] > 1.0) or (vs == "s" and row["se"] > 1.0):
|
||||
bad += 1
|
||||
print("=" * 100)
|
||||
print("PER-ARM DRUSSTGT DODGE QUALITY (session commit %s, %d runs x %d rounds)"
|
||||
% (session["commit"][:9], session["runs"], session["rounds"]))
|
||||
print("attribution cross-check: %d/%d deaths have the mapped victim at ~0 energy"
|
||||
% (tot - bad, tot))
|
||||
print("shipped baseline vs DrussGT is ~49%% round wins; this instrument is per "
|
||||
"SHOT (~thousands/arm), not per round.")
|
||||
|
||||
for name in arm_names:
|
||||
A = arms[name]
|
||||
v = A["variation"]
|
||||
print("\n--- ARM `%s` --- %d runs, %d rounds, %d shots" %
|
||||
(name, len(A["runs"]), A["rounds"], len(A["shots"])))
|
||||
print(" OUR shooting: power mean=%.3f sd=%.3f iqr=%.3f distinct=%d | "
|
||||
"aim-offset sd=%.2f deg iqr=%.2f | fire range mean=%.0f px" %
|
||||
(v["power_mean"], v["power_sd"], v["power_iqr"], v["power_distinct"],
|
||||
v["aimoff_sd"], v["aimoff_iqr"], v["range_mean"]))
|
||||
print(" power histogram (top): " +
|
||||
" ".join("%.2f:%d" % (p, n) for p, n in v["power_hist"].items()))
|
||||
print(" aim-offset histogram (deg bucket): " +
|
||||
" ".join("%+d:%d" % (b, n) for b, n in v["aimoff_hist"].items()))
|
||||
gt, gsw, grw, gnr = A["guns"]
|
||||
print(" GUN SELECTION ([config] switch lines): %s | switches=%d over %d/%d runs"
|
||||
% (" ".join("%s:%d" % kv for kv in gt.most_common()), gsw, grw, gnr))
|
||||
|
||||
print("\n" + "=" * 100)
|
||||
print("DRUSSGT DODGE QUALITY PER ARM (mean, 95%% CI = run-cluster bootstrap)")
|
||||
hdr = (" %-6s %7s | %8s %16s | %10s %16s | %10s %16s | %8s" %
|
||||
("arm", "shots", "miss", "miss/tick [95%CI]", "lat12",
|
||||
"lat12 [95%CI]", "lat_ctrl", "lat_ctrl [95%CI]", "hit%"))
|
||||
print(hdr)
|
||||
print(" " + "-" * (len(hdr) - 2))
|
||||
for name in arm_names:
|
||||
A = arms[name]
|
||||
shots = A["shots"]
|
||||
d = A.setdefault("dodge", {})
|
||||
for key in DODGE_METRICS:
|
||||
vals = [s[key] for s in shots if s.get(key) is not None]
|
||||
d[key] = dodge.mean(vals)
|
||||
ci_mt = cluster_ci(run_summaries(A["runs_shots"], "miss_per_tick"), a.reps)
|
||||
ci_lf = cluster_ci(run_summaries(A["runs_shots"], "lat_fixed_abs"), a.reps)
|
||||
ci_lc = cluster_ci(run_summaries(A["runs_shots"], "lat_ctrl_abs"), a.reps)
|
||||
print(" %-6s %7d | %8.1f %7.2f[%5.2f,%6.2f] | %10.1f [%5.1f,%6.1f] | "
|
||||
"%10.1f [%5.1f,%6.1f] | %8.3f" % (
|
||||
name, len(shots), d["miss"], d["miss_per_tick"],
|
||||
ci_mt[0], ci_mt[1], d["lat_fixed_abs"], ci_lf[0], ci_lf[1],
|
||||
d["lat_ctrl_abs"], ci_lc[0], ci_lc[1], d["hit"]))
|
||||
gt, gsw, grw, gnr = A["guns"]
|
||||
out["arms"][name] = {
|
||||
"runs": len(A["runs"]), "rounds": A["rounds"], "shots": len(shots),
|
||||
"variation": A["variation"], "dodge": {k: d[k] for k in DODGE_METRICS},
|
||||
"guns": {"counts": dict(gt), "switches": gsw, "runs_with": grw},
|
||||
"ci": {"miss_per_tick": ci_mt, "lat_fixed_abs": ci_lf,
|
||||
"lat_ctrl_abs": ci_lc},
|
||||
}
|
||||
|
||||
# ── minimum detectable effect for the shot-level mechanism metrics ───────
|
||||
print("\n" + "=" * 100)
|
||||
print("MINIMUM DETECTABLE EFFECT (run-cluster level, n=%d/arm, alpha=0.05 two-sided,"
|
||||
" 80%% power; MDE = 2.8016*sd_perrun*sqrt(2/n))" % session["runs"])
|
||||
print(" the shot counts are large but the RUNS are what set the between-arm "
|
||||
"uncertainty, so this is the honest floor")
|
||||
print(" %-16s %14s %14s %-22s" % ("metric", "sd(per-run)", "MDE(abs)", "MDE vs ref mean"))
|
||||
Z_ALPHA_POWER = 1.959963984540054 + 0.8416212335729143 # 2.8016
|
||||
for label, key in (("miss", "miss"), ("miss/tick", "miss_per_tick"),
|
||||
("lat12", "lat_fixed_abs"), ("lat_ctrl", "lat_ctrl_abs"),
|
||||
("hit%", "hit")):
|
||||
summ = run_summaries(arms[ref]["runs_shots"], key)
|
||||
means = [it["sum"] / it["n"] for it in summ if it["n"]]
|
||||
if len(means) < 2:
|
||||
continue
|
||||
sd = statistics.stdev(means)
|
||||
mde = Z_ALPHA_POWER * sd * math.sqrt(2.0 / len(means))
|
||||
rm = arms[ref]["dodge"]
|
||||
refmean = rm[key] * (100.0 if key == "hit" else 1.0)
|
||||
print(" %-16s %14.3f %14.3f %-22s" % (
|
||||
label, sd, mde,
|
||||
"%.1f%% of %.3f" % (100 * mde / refmean, refmean) if refmean else "-"))
|
||||
|
||||
# ── between-arm tests, clustered by run ───────────────────────────────────
|
||||
print("\n" + "=" * 100)
|
||||
print("BETWEEN-ARM CONTRASTS (exact run-cluster permutation, C(14,7)=3432)")
|
||||
print(" pooled = pooled-shot difference A-B; strat = range-stratified "
|
||||
"(bands %s, n-weighted)" % ",".join(STRAT_BANDS))
|
||||
print(" a mix that poisons DrussGT should show miss/tick, lat12 and lat_ctrl "
|
||||
"LOWER (worse dodging) than the best single gun")
|
||||
hdr2 = (" %-24s %-24s %9s %8s | %9s %8s" %
|
||||
("metric", "A vs B", "pooled", "perm p", "strat", "perm p"))
|
||||
print(hdr2)
|
||||
print(" " + "-" * (len(hdr2) - 2))
|
||||
compare_pairs = [("mix", n) for n in arm_names if n != "mix" and n != ref]
|
||||
compare_pairs += [(n, ref) for n in arm_names if n not in (ref,)]
|
||||
seen = set()
|
||||
out["contrasts"] = {}
|
||||
for A, B in compare_pairs:
|
||||
if (A, B) in seen:
|
||||
continue
|
||||
seen.add((A, B))
|
||||
for key in TEST_METRICS:
|
||||
sa = run_summaries(arms[A]["runs_shots"], key)
|
||||
sb = run_summaries(arms[B]["runs_shots"], key)
|
||||
obs_p, p_p, meth = cluster_perm(sa, sb, pooled_stat)
|
||||
obs_s, p_s, _ = cluster_perm(sa, sb, strat_stat)
|
||||
print(" %-24s %-24s %+9.3f %8.4f | %+9.3f %8.4f [%s]" %
|
||||
(key, "%s - %s" % (A, B), obs_p, p_p, obs_s, p_s, meth))
|
||||
out["contrasts"].setdefault("%s-%s" % (A, B), {})[key] = {
|
||||
"pooled": obs_p, "perm_p": p_p, "method": meth,
|
||||
"strat": obs_s, "strat_perm_p": p_s}
|
||||
print()
|
||||
|
||||
if a.json:
|
||||
with open(a.json, "w") as f:
|
||||
json.dump(out, f, indent=1, sort_keys=True)
|
||||
print("JSON written to %s" % a.json)
|
||||
return out
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user