a54ae6a162
Follow-up toe0666a5, which showed Pattern alone (10.78%) beats the full rack (6.93%). That left two open questions: is a SMALL rack of good guns better than Pattern alone, and does the selector add value on a good rack (rather than only on the bloated one)? Both are now answered: NO and NO. 6 arms x 7 runs x 7 rounds, one frozen binary from CLEAN HEADe0666a5(built via `git archive`, source verified byte-identical to the clean tree), rack knobs only, 8 concurrent battles, real server-side hit rate vs the real DrussGT, exact two-sided permutation test on per-run rates. arm guns (selector active?) real % dmg/run p vs onlyPattern onlyPattern Pattern, NO selection 10.36 264 -- lean8 HeadOn,Linear,Circular,Accel,Pattern,GF,KNN,WallBounce 6.31 146 0.0169 lean6 lean8 - HeadOn 8.83 212 0.0262 pairPC Pattern + Circular 8.23 185 0.0460 pairPK Pattern + KNN 9.80 264 0.3998 pairPL Pattern + Linear 8.23 200 0.0035 The control replicates the prior run (10.36% vs 10.78% before; same binary tree, different build path). THE MECHANISM, from the per-arm selected-gun mix - the virtual signal keeps ranking the WRONG guns first, even on a two-gun rack: lean8: HeadOn 46.2% of ticks at 2.0% REAL; Pattern only 14.4% (12.6% real) lean6: Pattern 29.2% (10.4% real) vs KNN 25.3% (8.1%) and Linear 15.2% (8.0%) pairPC: Circular 66.8% (6.9% real) vs Pattern 33.2% (11.2% real) - over-picks Circular pairPL: Linear 57.7% (6.0% real) vs Pattern 42.3% (11.1% real) - over-picks Linear pairPK: Pattern 86.4% - ties ONLY because the selector happens to pick Pattern most of the time; it is numerically lower with identical dmg/run So the failure is NOT rack size. Pruning does not fix it; the ranking is wrong. VERDICT: ship `onlyPattern` - Pattern alone with selection bypassed - at 10.36% real and 264 dmg/run, vs lean8 6.31%/146 and the prior full rack 6.93%/159. This DIRECTLY CONTRADICTS the standing user directive to keep virtual-fitness selection, so it is recorded here plainly rather than quietly acted on: disable the selector (`TR_RACK_<every gun but PATTERN>=off`) pending a better fitness signal. The mechanism itself is left intact and functional so it can be re-enabled with one env var, and so it can be fixed rather than discarded. REMAINING CAVEAT: ONE ADVERSARY. All of this is vs DrussGT. Pattern as the default must be re-checked against other bots first - that is the next job. Extends the reusable harness (tools/ab/which_gun_arm_env.sh now has lean8/lean6/ pairPC/pairPK/pairPL; which_gun_analyze.py is parameterised by WHICHGUN_OUT and compares against both `full` and `onlyPattern`).
215 lines
7.1 KiB
Python
215 lines
7.1 KiB
Python
#!/usr/bin/env python3
|
|
"""Analyze the which-gun A/B from server-side events sidecars.
|
|
|
|
Ground truth: real hit rate from the events sidecar (fire/hit, server-side).
|
|
Per-run rates -> exact two-sided permutation test (enumerated when small).
|
|
Also reads gun_stats.jsonl to prove each arm was LIVE (selected-gun mix).
|
|
"""
|
|
import json, glob, os, re, sys, collections, itertools, math
|
|
|
|
OUT = os.environ.get("WHICHGUN_OUT", "/tmp/whichgun")
|
|
ARMS = ["full", "onlyLinear", "onlyPattern", "onlyGF", "onlyKNN",
|
|
"lean8", "lean6", "pairPC", "pairPK", "pairPL"]
|
|
BASELINES = ["full", "onlyPattern"]
|
|
MB_POWERS = {1.0, 1.5, 2.0, 3.0}
|
|
|
|
|
|
def runs_for(arm):
|
|
rs = set()
|
|
for f in glob.glob(f"{OUT}/events_{arm}_r*.json"):
|
|
m = re.search(rf"_{re.escape(arm)}_r(\d+)\.json$", f)
|
|
if m:
|
|
rs.add(int(m.group(1)))
|
|
return sorted(rs)
|
|
|
|
|
|
def parse_subject_fired(logpath):
|
|
if not os.path.exists(logpath):
|
|
return None
|
|
for line in open(logpath, errors="replace"):
|
|
m = re.search(r"subject event counts:.*bulletsFired=(\d+)", line)
|
|
if m:
|
|
return int(m.group(1))
|
|
return None
|
|
|
|
|
|
def ev_stats(path, subject_fired):
|
|
fires = collections.Counter()
|
|
hits = collections.Counter()
|
|
dmg = collections.Counter()
|
|
powers = collections.defaultdict(set)
|
|
for line in open(path):
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
try:
|
|
o = json.loads(line)
|
|
except json.JSONDecodeError:
|
|
continue # partial line from an in-progress battle
|
|
t = o.get("type")
|
|
if t == "fire":
|
|
fires[o["owner"]] += 1
|
|
powers[o["owner"]].add(round(o.get("power", -1), 3))
|
|
elif t == "hit":
|
|
hits[o["owner"]] += 1
|
|
dmg[o["owner"]] += o.get("damage", 0.0)
|
|
# ModularBot = owner whose power set is a subset of the discrete bins
|
|
mb = None
|
|
for own, pw in powers.items():
|
|
if fires[own] > 0 and pw <= MB_POWERS:
|
|
mb = own
|
|
if mb is None:
|
|
# fallback: owner that is NOT the subject (identified by bulletsFired)
|
|
subj = None
|
|
if subject_fired is not None:
|
|
for own, n in fires.items():
|
|
if n == subject_fired:
|
|
subj = own
|
|
cand = [o for o in fires if o != subj]
|
|
if len(cand) == 1:
|
|
mb = cand[0]
|
|
if mb is None:
|
|
return None
|
|
return fires[mb], hits[mb], dmg[mb]
|
|
|
|
|
|
def gun_mix(arm):
|
|
"""Sum selected/realShots/realHits per gun across all rounds of the arm."""
|
|
mix = collections.defaultdict(lambda: collections.Counter())
|
|
rounds = 0
|
|
for r in runs_for(arm):
|
|
p = f"{OUT}/gun_stats_{arm}_r{r}.jsonl"
|
|
if not os.path.exists(p):
|
|
continue
|
|
for line in open(p):
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
o = json.loads(line)
|
|
rounds += 1
|
|
for g in o["guns"]:
|
|
a = mix[g["name"]]
|
|
a["selected"] += g["selected"]
|
|
a["realShots"] += g["realShots"]
|
|
a["realHits"] += g["realHits"]
|
|
return mix, rounds
|
|
|
|
|
|
def analyze(arm, verbose=False):
|
|
per = []
|
|
totF = totH = 0
|
|
totD = 0.0
|
|
for r in runs_for(arm):
|
|
ev = ev_stats(f"{OUT}/events_{arm}_r{r}.json", parse_subject_fired(f"{OUT}/battle_{arm}_r{r}.log"))
|
|
if not ev:
|
|
if verbose:
|
|
print(f" r{r}: no events")
|
|
continue
|
|
f, h, d = ev
|
|
rate = 100.0 * h / max(1, f)
|
|
per.append({"run": r, "fires": f, "hits": h, "rate": rate, "dmg": d})
|
|
totF += f
|
|
totH += h
|
|
totD += d
|
|
if verbose:
|
|
print(f" r{r}: {h}/{f} = {rate:.2f}% dmg={d:.0f}")
|
|
return dict(arm=arm, per=per, shots=totF, hits=totH, dmg=totD,
|
|
rate=100.0 * totH / max(1, totF), n=len(per))
|
|
|
|
|
|
def perm_test_exact(a, b):
|
|
"""Exact two-sided permutation test on per-run rates (difference of means)."""
|
|
xa = [p["rate"] for p in a["per"]]
|
|
xb = [p["rate"] for p in b["per"]]
|
|
na, nb = len(xa), len(xb)
|
|
if na == 0 or nb == 0:
|
|
return float("nan"), float("nan")
|
|
obs = abs(sum(xa) / na - sum(xb) / nb)
|
|
pooled = xa + xb
|
|
n = na + nb
|
|
cnt = 0
|
|
total = 0
|
|
# exact enumeration of C(n, na); fall back to sampling for large n
|
|
if math.comb(n, na) <= 200000:
|
|
for combo in itertools.combinations(range(n), na):
|
|
s = set(combo)
|
|
ma = sum(pooled[i] for i in combo) / na
|
|
mb = sum(pooled[i] for i in range(n) if i not in s) / nb
|
|
if abs(ma - mb) >= obs - 1e-12:
|
|
cnt += 1
|
|
total += 1
|
|
return obs, cnt / total
|
|
# sampled
|
|
import random
|
|
rng = random.Random(12345)
|
|
B = 200000
|
|
for _ in range(B):
|
|
idx = rng.sample(range(n), na)
|
|
s = set(idx)
|
|
ma = sum(pooled[i] for i in idx) / na
|
|
mb = sum(pooled[i] for i in range(n) if i not in s) / nb
|
|
if abs(ma - mb) >= obs - 1e-12:
|
|
cnt += 1
|
|
return obs, (cnt + 1) / (B + 1)
|
|
|
|
|
|
def fmt_range(per):
|
|
if not per:
|
|
return "n/a"
|
|
rs = [p["rate"] for p in per]
|
|
return f"{min(rs):.2f}-{max(rs):.2f}"
|
|
|
|
|
|
def main():
|
|
arms = [a for a in ARMS if runs_for(a)]
|
|
extra = sys.argv[1:]
|
|
if extra:
|
|
arms = extra
|
|
res = {a: analyze(a, verbose=("-v" in sys.argv)) for a in arms}
|
|
|
|
print("=" * 96)
|
|
print("REAL HIT RATE (server-side events sidecar) — ModularBot vs real DrussGT")
|
|
print("=" * 96)
|
|
hdr = f"{'arm':<12} {'runs':>4} {'shots':>6} {'hits':>5} {'real %':>7} {'dmg/run':>8} {'per-run range':>14}"
|
|
print(hdr)
|
|
print("-" * len(hdr))
|
|
for a in arms:
|
|
r = res[a]
|
|
print(f"{a:<12} {r['n']:>4} {r['shots']:>6} {r['hits']:>5} {r['rate']:>7.2f} "
|
|
f"{r['dmg']/max(1,r['n']):>8.0f} {fmt_range(r['per']):>14}")
|
|
|
|
print("\nper-run rates:")
|
|
for a in arms:
|
|
pr = " ".join(f"r{p['run']}={p['rate']:.2f}" for p in res[a]["per"])
|
|
print(f" {a:<12} {pr}")
|
|
|
|
for base in BASELINES:
|
|
if base not in res or len(arms) < 2:
|
|
continue
|
|
print("\n" + "=" * 96)
|
|
print(f"EXACT TWO-SIDED PERMUTATION TEST vs `{base}` (per-run rates)")
|
|
print("=" * 96)
|
|
for a in arms:
|
|
if a == base:
|
|
continue
|
|
obs, p = perm_test_exact(res[base], res[a])
|
|
print(f" {base} vs {a:<12}: mean-rate diff = {obs:+.2f} pp, exact two-sided p = {p:.4f}")
|
|
|
|
print("\n" + "=" * 96)
|
|
print("LIVENESS — selected-gun mix (sum of per-tick `selected` over all rounds)")
|
|
print("=" * 96)
|
|
for a in arms:
|
|
mix, rounds = gun_mix(a)
|
|
tot_sel = sum(v["selected"] for v in mix.values())
|
|
parts = []
|
|
for name, v in sorted(mix.items(), key=lambda kv: -kv[1]["selected"]):
|
|
if v["selected"] == 0:
|
|
continue
|
|
share = 100.0 * v["selected"] / max(1, tot_sel)
|
|
parts.append(f"{name} {share:.1f}% (sel={v['selected']}, real={v['realHits']}/{v['realShots']})")
|
|
print(f" {a:<12} rounds={rounds:>3} " + ("; ".join(parts) if parts else "NO SELECTION RECORDED"))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|