e0666a562d
MEASURED against the real DrussGT, one frozen binary built from clean HEAD, rack knobs only (no source edits), 5 arms x 7 runs x 7 rounds, 8 concurrent battles, judged ONLY on server-side real hit rate from the events sidecar, exact two-sided permutation test on per-run rates. arm runs shots hits real % dmg/run p vs full full (shipped) 7 3898 270 6.93 159 -- onlyPattern 7 4582 494 10.78 287 0.0012 <- BETTER onlyKNN 7 4033 207 5.13 119 0.1340 onlyLinear 7 3215 105 3.27 65 0.0082 onlyGF 7 3193 72 2.25 45 0.0012 Firing Pattern ALONE gives +3.85pp pooled hit rate and +80% damage per run, and it fires MORE shots (4582 vs 3898) - it dominates on rate and volume. This is not "any single gun wins" (full beats Linear, GF and KNN); it is specifically "Pattern alone beats the rack". WHY - the virtual fitness signal mis-ranks guns against real outcomes: - HeadOn is massively over-selected: 31.4% of ticks, the most real shots (1070), but only 4.5% REAL. It alone drags the rack down. - Pattern has the best virtual rank and near-best real rate (11.9%, rank 2), yet is selected only 22.6% of the time. - Linear's apparent strength was SELECTION BIAS: conditional on being selected it looked like 15.2% (n=33), but its UNCONDITIONAL rate (onlyLinear) is 3.27%. Every earlier per-gun "real rate" in this repo is conditional on selection and is therefore confounded. This experiment is the clean measurement. NOT YET SETTLED (do not overclaim): - ONE ADVERSARY. All of this is vs DrussGT. Pattern must be re-checked against other bots before it becomes the default on this evidence alone. - Whether a SMALL rack of good guns beats Pattern alone. The selector is negative value on the CURRENT bloated rack; that does not prove it is negative value on a rack of only good guns. That is the next experiment and it decides whether the selection apparatus is fixed or disabled. - The user's standing directive is to KEEP virtual-fitness selection. This measurement conflicts with it, so the next step tests the selector on a small good rack rather than assuming either answer. Context - three prior selection-side attempts all failed: hysteresis (7.02% -> 5.10%, p=0.002), commitment (7.17% -> 4.44%, p=0.0012), arrival-accuracy tie-break (7.08%, p=0.88 null). The per-tick random draw is load-bearing on three independent measurements. This experiment locates the real problem one level up: which guns are in the rack, and that the virtual signal ranks them wrongly. Preserves the reusable harness (tools/ab/which_gun_run_one.sh, which_gun_arm_env.sh, which_gun_analyze.py) and the full writeup (docs/selector_negative_value.md).
208 lines
6.8 KiB
Python
208 lines
6.8 KiB
Python
#!/usr/bin/env python3
|
|
"""Analyze the which-gun A/B from server-side events sidecars.
|
|
|
|
Ground truth: real hit rate from the events sidecar (fire/hit, server-side).
|
|
Per-run rates -> exact two-sided permutation test (enumerated when small).
|
|
Also reads gun_stats.jsonl to prove each arm was LIVE (selected-gun mix).
|
|
"""
|
|
import json, glob, os, re, sys, collections, itertools, math
|
|
|
|
OUT = "/tmp/whichgun"
|
|
ARMS = ["full", "onlyLinear", "onlyPattern", "onlyGF", "onlyKNN"]
|
|
MB_POWERS = {1.0, 1.5, 2.0, 3.0}
|
|
|
|
|
|
def runs_for(arm):
|
|
rs = set()
|
|
for f in glob.glob(f"{OUT}/events_{arm}_r*.json"):
|
|
m = re.search(rf"_{re.escape(arm)}_r(\d+)\.json$", f)
|
|
if m:
|
|
rs.add(int(m.group(1)))
|
|
return sorted(rs)
|
|
|
|
|
|
def parse_subject_fired(logpath):
|
|
if not os.path.exists(logpath):
|
|
return None
|
|
for line in open(logpath, errors="replace"):
|
|
m = re.search(r"subject event counts:.*bulletsFired=(\d+)", line)
|
|
if m:
|
|
return int(m.group(1))
|
|
return None
|
|
|
|
|
|
def ev_stats(path, subject_fired):
|
|
fires = collections.Counter()
|
|
hits = collections.Counter()
|
|
dmg = collections.Counter()
|
|
powers = collections.defaultdict(set)
|
|
for line in open(path):
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
o = json.loads(line)
|
|
t = o.get("type")
|
|
if t == "fire":
|
|
fires[o["owner"]] += 1
|
|
powers[o["owner"]].add(round(o.get("power", -1), 3))
|
|
elif t == "hit":
|
|
hits[o["owner"]] += 1
|
|
dmg[o["owner"]] += o.get("damage", 0.0)
|
|
# ModularBot = owner whose power set is a subset of the discrete bins
|
|
mb = None
|
|
for own, pw in powers.items():
|
|
if fires[own] > 0 and pw <= MB_POWERS:
|
|
mb = own
|
|
if mb is None:
|
|
# fallback: owner that is NOT the subject (identified by bulletsFired)
|
|
subj = None
|
|
if subject_fired is not None:
|
|
for own, n in fires.items():
|
|
if n == subject_fired:
|
|
subj = own
|
|
cand = [o for o in fires if o != subj]
|
|
if len(cand) == 1:
|
|
mb = cand[0]
|
|
if mb is None:
|
|
return None
|
|
return fires[mb], hits[mb], dmg[mb]
|
|
|
|
|
|
def gun_mix(arm):
|
|
"""Sum selected/realShots/realHits per gun across all rounds of the arm."""
|
|
mix = collections.defaultdict(lambda: collections.Counter())
|
|
rounds = 0
|
|
for r in runs_for(arm):
|
|
p = f"{OUT}/gun_stats_{arm}_r{r}.jsonl"
|
|
if not os.path.exists(p):
|
|
continue
|
|
for line in open(p):
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
o = json.loads(line)
|
|
rounds += 1
|
|
for g in o["guns"]:
|
|
a = mix[g["name"]]
|
|
a["selected"] += g["selected"]
|
|
a["realShots"] += g["realShots"]
|
|
a["realHits"] += g["realHits"]
|
|
return mix, rounds
|
|
|
|
|
|
def analyze(arm, verbose=False):
|
|
per = []
|
|
totF = totH = 0
|
|
totD = 0.0
|
|
for r in runs_for(arm):
|
|
ev = ev_stats(f"{OUT}/events_{arm}_r{r}.json", parse_subject_fired(f"{OUT}/battle_{arm}_r{r}.log"))
|
|
if not ev:
|
|
if verbose:
|
|
print(f" r{r}: no events")
|
|
continue
|
|
f, h, d = ev
|
|
rate = 100.0 * h / max(1, f)
|
|
per.append({"run": r, "fires": f, "hits": h, "rate": rate, "dmg": d})
|
|
totF += f
|
|
totH += h
|
|
totD += d
|
|
if verbose:
|
|
print(f" r{r}: {h}/{f} = {rate:.2f}% dmg={d:.0f}")
|
|
return dict(arm=arm, per=per, shots=totF, hits=totH, dmg=totD,
|
|
rate=100.0 * totH / max(1, totF), n=len(per))
|
|
|
|
|
|
def perm_test_exact(a, b):
|
|
"""Exact two-sided permutation test on per-run rates (difference of means)."""
|
|
xa = [p["rate"] for p in a["per"]]
|
|
xb = [p["rate"] for p in b["per"]]
|
|
na, nb = len(xa), len(xb)
|
|
if na == 0 or nb == 0:
|
|
return float("nan"), float("nan")
|
|
obs = abs(sum(xa) / na - sum(xb) / nb)
|
|
pooled = xa + xb
|
|
n = na + nb
|
|
cnt = 0
|
|
total = 0
|
|
# exact enumeration of C(n, na); fall back to sampling for large n
|
|
if math.comb(n, na) <= 200000:
|
|
for combo in itertools.combinations(range(n), na):
|
|
s = set(combo)
|
|
ma = sum(pooled[i] for i in combo) / na
|
|
mb = sum(pooled[i] for i in range(n) if i not in s) / nb
|
|
if abs(ma - mb) >= obs - 1e-12:
|
|
cnt += 1
|
|
total += 1
|
|
return obs, cnt / total
|
|
# sampled
|
|
import random
|
|
rng = random.Random(12345)
|
|
B = 200000
|
|
for _ in range(B):
|
|
idx = rng.sample(range(n), na)
|
|
s = set(idx)
|
|
ma = sum(pooled[i] for i in idx) / na
|
|
mb = sum(pooled[i] for i in range(n) if i not in s) / nb
|
|
if abs(ma - mb) >= obs - 1e-12:
|
|
cnt += 1
|
|
return obs, (cnt + 1) / (B + 1)
|
|
|
|
|
|
def fmt_range(per):
|
|
if not per:
|
|
return "n/a"
|
|
rs = [p["rate"] for p in per]
|
|
return f"{min(rs):.2f}-{max(rs):.2f}"
|
|
|
|
|
|
def main():
|
|
arms = [a for a in ARMS if runs_for(a)]
|
|
extra = sys.argv[1:]
|
|
if extra:
|
|
arms = extra
|
|
res = {a: analyze(a, verbose=("-v" in sys.argv)) for a in arms}
|
|
|
|
print("=" * 96)
|
|
print("REAL HIT RATE (server-side events sidecar) — ModularBot vs real DrussGT")
|
|
print("=" * 96)
|
|
hdr = f"{'arm':<12} {'runs':>4} {'shots':>6} {'hits':>5} {'real %':>7} {'dmg/run':>8} {'per-run range':>14}"
|
|
print(hdr)
|
|
print("-" * len(hdr))
|
|
for a in arms:
|
|
r = res[a]
|
|
print(f"{a:<12} {r['n']:>4} {r['shots']:>6} {r['hits']:>5} {r['rate']:>7.2f} "
|
|
f"{r['dmg']/max(1,r['n']):>8.0f} {fmt_range(r['per']):>14}")
|
|
|
|
print("\nper-run rates:")
|
|
for a in arms:
|
|
pr = " ".join(f"r{p['run']}={p['rate']:.2f}" for p in res[a]["per"])
|
|
print(f" {a:<12} {pr}")
|
|
|
|
if "full" in res and len(arms) > 1:
|
|
print("\n" + "=" * 96)
|
|
print("EXACT TWO-SIDED PERMUTATION TEST vs `full` (per-run rates)")
|
|
print("=" * 96)
|
|
for a in arms:
|
|
if a == "full":
|
|
continue
|
|
obs, p = perm_test_exact(res["full"], res[a])
|
|
print(f" full vs {a:<12}: mean-rate diff = {obs:+.2f} pp, exact two-sided p = {p:.4f}")
|
|
|
|
print("\n" + "=" * 96)
|
|
print("LIVENESS — selected-gun mix (sum of per-tick `selected` over all rounds)")
|
|
print("=" * 96)
|
|
for a in arms:
|
|
mix, rounds = gun_mix(a)
|
|
tot_sel = sum(v["selected"] for v in mix.values())
|
|
parts = []
|
|
for name, v in sorted(mix.items(), key=lambda kv: -kv[1]["selected"]):
|
|
if v["selected"] == 0:
|
|
continue
|
|
share = 100.0 * v["selected"] / max(1, tot_sel)
|
|
parts.append(f"{name} {share:.1f}% (sel={v['selected']}, real={v['realHits']}/{v['realShots']})")
|
|
print(f" {a:<12} rounds={rounds:>3} " + ("; ".join(parts) if parts else "NO SELECTION RECORDED"))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|