Files
SirRoboGarage/tools/ab/which_gun_analyze.py
SirStone a54ae6a162 SETTLED: no small rack beats Pattern alone; the selector is negative value on a GOOD rack
Follow-up to e0666a5, which showed Pattern alone (10.78%) beats the full rack
(6.93%). That left two open questions: is a SMALL rack of good guns better than
Pattern alone, and does the selector add value on a good rack (rather than only
on the bloated one)? Both are now answered: NO and NO.

6 arms x 7 runs x 7 rounds, one frozen binary from CLEAN HEAD e0666a5 (built via
`git archive`, source verified byte-identical to the clean tree), rack knobs
only, 8 concurrent battles, real server-side hit rate vs the real DrussGT, exact
two-sided permutation test on per-run rates.

  arm          guns (selector active?)                    real %  dmg/run  p vs onlyPattern
  onlyPattern  Pattern, NO selection                       10.36    264     --
  lean8        HeadOn,Linear,Circular,Accel,Pattern,GF,KNN,WallBounce  6.31  146  0.0169
  lean6        lean8 - HeadOn                               8.83    212     0.0262
  pairPC       Pattern + Circular                           8.23    185     0.0460
  pairPK       Pattern + KNN                                9.80    264     0.3998
  pairPL       Pattern + Linear                             8.23    200     0.0035

The control replicates the prior run (10.36% vs 10.78% before; same binary tree,
different build path).

THE MECHANISM, from the per-arm selected-gun mix - the virtual signal keeps
ranking the WRONG guns first, even on a two-gun rack:
  lean8: HeadOn 46.2% of ticks at 2.0% REAL; Pattern only 14.4% (12.6% real)
  lean6: Pattern 29.2% (10.4% real) vs KNN 25.3% (8.1%) and Linear 15.2% (8.0%)
  pairPC: Circular 66.8% (6.9% real) vs Pattern 33.2% (11.2% real) - over-picks Circular
  pairPL: Linear 57.7% (6.0% real) vs Pattern 42.3% (11.1% real) - over-picks Linear
  pairPK: Pattern 86.4% - ties ONLY because the selector happens to pick Pattern
          most of the time; it is numerically lower with identical dmg/run
So the failure is NOT rack size. Pruning does not fix it; the ranking is wrong.

VERDICT: ship `onlyPattern` - Pattern alone with selection bypassed - at 10.36%
real and 264 dmg/run, vs lean8 6.31%/146 and the prior full rack 6.93%/159.
This DIRECTLY CONTRADICTS the standing user directive to keep virtual-fitness
selection, so it is recorded here plainly rather than quietly acted on: disable
the selector (`TR_RACK_<every gun but PATTERN>=off`) pending a better fitness
signal. The mechanism itself is left intact and functional so it can be re-enabled
with one env var, and so it can be fixed rather than discarded.

REMAINING CAVEAT: ONE ADVERSARY. All of this is vs DrussGT. Pattern as the default
must be re-checked against other bots first - that is the next job.

Extends the reusable harness (tools/ab/which_gun_arm_env.sh now has lean8/lean6/
pairPC/pairPK/pairPL; which_gun_analyze.py is parameterised by WHICHGUN_OUT and
compares against both `full` and `onlyPattern`).
2026-09-22 01:35:51 +02:00

215 lines
7.1 KiB
Python

#!/usr/bin/env python3
"""Analyze the which-gun A/B from server-side events sidecars.
Ground truth: real hit rate from the events sidecar (fire/hit, server-side).
Per-run rates -> exact two-sided permutation test (enumerated when small).
Also reads gun_stats.jsonl to prove each arm was LIVE (selected-gun mix).
"""
import json, glob, os, re, sys, collections, itertools, math
OUT = os.environ.get("WHICHGUN_OUT", "/tmp/whichgun")
ARMS = ["full", "onlyLinear", "onlyPattern", "onlyGF", "onlyKNN",
"lean8", "lean6", "pairPC", "pairPK", "pairPL"]
BASELINES = ["full", "onlyPattern"]
MB_POWERS = {1.0, 1.5, 2.0, 3.0}
def runs_for(arm):
rs = set()
for f in glob.glob(f"{OUT}/events_{arm}_r*.json"):
m = re.search(rf"_{re.escape(arm)}_r(\d+)\.json$", f)
if m:
rs.add(int(m.group(1)))
return sorted(rs)
def parse_subject_fired(logpath):
if not os.path.exists(logpath):
return None
for line in open(logpath, errors="replace"):
m = re.search(r"subject event counts:.*bulletsFired=(\d+)", line)
if m:
return int(m.group(1))
return None
def ev_stats(path, subject_fired):
fires = collections.Counter()
hits = collections.Counter()
dmg = collections.Counter()
powers = collections.defaultdict(set)
for line in open(path):
line = line.strip()
if not line:
continue
try:
o = json.loads(line)
except json.JSONDecodeError:
continue # partial line from an in-progress battle
t = o.get("type")
if t == "fire":
fires[o["owner"]] += 1
powers[o["owner"]].add(round(o.get("power", -1), 3))
elif t == "hit":
hits[o["owner"]] += 1
dmg[o["owner"]] += o.get("damage", 0.0)
# ModularBot = owner whose power set is a subset of the discrete bins
mb = None
for own, pw in powers.items():
if fires[own] > 0 and pw <= MB_POWERS:
mb = own
if mb is None:
# fallback: owner that is NOT the subject (identified by bulletsFired)
subj = None
if subject_fired is not None:
for own, n in fires.items():
if n == subject_fired:
subj = own
cand = [o for o in fires if o != subj]
if len(cand) == 1:
mb = cand[0]
if mb is None:
return None
return fires[mb], hits[mb], dmg[mb]
def gun_mix(arm):
"""Sum selected/realShots/realHits per gun across all rounds of the arm."""
mix = collections.defaultdict(lambda: collections.Counter())
rounds = 0
for r in runs_for(arm):
p = f"{OUT}/gun_stats_{arm}_r{r}.jsonl"
if not os.path.exists(p):
continue
for line in open(p):
line = line.strip()
if not line:
continue
o = json.loads(line)
rounds += 1
for g in o["guns"]:
a = mix[g["name"]]
a["selected"] += g["selected"]
a["realShots"] += g["realShots"]
a["realHits"] += g["realHits"]
return mix, rounds
def analyze(arm, verbose=False):
per = []
totF = totH = 0
totD = 0.0
for r in runs_for(arm):
ev = ev_stats(f"{OUT}/events_{arm}_r{r}.json", parse_subject_fired(f"{OUT}/battle_{arm}_r{r}.log"))
if not ev:
if verbose:
print(f" r{r}: no events")
continue
f, h, d = ev
rate = 100.0 * h / max(1, f)
per.append({"run": r, "fires": f, "hits": h, "rate": rate, "dmg": d})
totF += f
totH += h
totD += d
if verbose:
print(f" r{r}: {h}/{f} = {rate:.2f}% dmg={d:.0f}")
return dict(arm=arm, per=per, shots=totF, hits=totH, dmg=totD,
rate=100.0 * totH / max(1, totF), n=len(per))
def perm_test_exact(a, b):
"""Exact two-sided permutation test on per-run rates (difference of means)."""
xa = [p["rate"] for p in a["per"]]
xb = [p["rate"] for p in b["per"]]
na, nb = len(xa), len(xb)
if na == 0 or nb == 0:
return float("nan"), float("nan")
obs = abs(sum(xa) / na - sum(xb) / nb)
pooled = xa + xb
n = na + nb
cnt = 0
total = 0
# exact enumeration of C(n, na); fall back to sampling for large n
if math.comb(n, na) <= 200000:
for combo in itertools.combinations(range(n), na):
s = set(combo)
ma = sum(pooled[i] for i in combo) / na
mb = sum(pooled[i] for i in range(n) if i not in s) / nb
if abs(ma - mb) >= obs - 1e-12:
cnt += 1
total += 1
return obs, cnt / total
# sampled
import random
rng = random.Random(12345)
B = 200000
for _ in range(B):
idx = rng.sample(range(n), na)
s = set(idx)
ma = sum(pooled[i] for i in idx) / na
mb = sum(pooled[i] for i in range(n) if i not in s) / nb
if abs(ma - mb) >= obs - 1e-12:
cnt += 1
return obs, (cnt + 1) / (B + 1)
def fmt_range(per):
if not per:
return "n/a"
rs = [p["rate"] for p in per]
return f"{min(rs):.2f}-{max(rs):.2f}"
def main():
arms = [a for a in ARMS if runs_for(a)]
extra = sys.argv[1:]
if extra:
arms = extra
res = {a: analyze(a, verbose=("-v" in sys.argv)) for a in arms}
print("=" * 96)
print("REAL HIT RATE (server-side events sidecar) — ModularBot vs real DrussGT")
print("=" * 96)
hdr = f"{'arm':<12} {'runs':>4} {'shots':>6} {'hits':>5} {'real %':>7} {'dmg/run':>8} {'per-run range':>14}"
print(hdr)
print("-" * len(hdr))
for a in arms:
r = res[a]
print(f"{a:<12} {r['n']:>4} {r['shots']:>6} {r['hits']:>5} {r['rate']:>7.2f} "
f"{r['dmg']/max(1,r['n']):>8.0f} {fmt_range(r['per']):>14}")
print("\nper-run rates:")
for a in arms:
pr = " ".join(f"r{p['run']}={p['rate']:.2f}" for p in res[a]["per"])
print(f" {a:<12} {pr}")
for base in BASELINES:
if base not in res or len(arms) < 2:
continue
print("\n" + "=" * 96)
print(f"EXACT TWO-SIDED PERMUTATION TEST vs `{base}` (per-run rates)")
print("=" * 96)
for a in arms:
if a == base:
continue
obs, p = perm_test_exact(res[base], res[a])
print(f" {base} vs {a:<12}: mean-rate diff = {obs:+.2f} pp, exact two-sided p = {p:.4f}")
print("\n" + "=" * 96)
print("LIVENESS — selected-gun mix (sum of per-tick `selected` over all rounds)")
print("=" * 96)
for a in arms:
mix, rounds = gun_mix(a)
tot_sel = sum(v["selected"] for v in mix.values())
parts = []
for name, v in sorted(mix.items(), key=lambda kv: -kv[1]["selected"]):
if v["selected"] == 0:
continue
share = 100.0 * v["selected"] / max(1, tot_sel)
parts.append(f"{name} {share:.1f}% (sel={v['selected']}, real={v['realHits']}/{v['realShots']})")
print(f" {a:<12} rounds={rounds:>3} " + ("; ".join(parts) if parts else "NO SELECTION RECORDED"))
if __name__ == "__main__":
main()