From 2c94dc221a817475ebde4bb4f0c928b674a47f6f Mon Sep 17 00:00:00 2001 From: Davide Cappellini Date: Mon, 21 Sep 2026 06:31:00 +0200 Subject: [PATCH] test(selector): 16 ranking rules A/B'd against the boss - none beat the shipped config Added runtime-tunable ranking knobs to the selector, all defaulting to the shipped values so behaviour is byte-identical when unset: GUN_SELECTOR_WINDOW, MINOBS, TIE, FLOOR, POOL, RANK, SHRINK, SEED. rankScore supports mean, Wilson lower bound, UCB, Thompson and shrinkage. Also fixed hitRate's most-recent-N read for sub-WindowSize windows (windowHits). RESULT: NO candidate credibly beat the shipped config. 13 runs x 8 rounds vs DrussGT, 3612 shots, base 6.95% at 251 dmg/run; every candidate's per-run interval overlaps base, and the nominal 'winners' are <=0.6 SE apart on far fewer shots. Kept the shipped default. Valid outcome, recorded plainly. THE FINDING THAT MATTERS MORE: the virtual-bullet ranking is ANTI-correlated with real hit rate - Spearman ~ -0.37 for the shipped config. It is not merely weak, it is INVERTED. The guns with the highest VIRTUAL rates have among the lowest REAL rates: Tsetlin 12.9% virtual / 5.8% real, WallBounce 12.9 / 6.2, StopShot 12.6 / 6.1, AvgLead 12.3 / 7.0 - while Linear sits at 10.2 virtual / 10.7 real and KNN at 7.5 / 9.0. So what carries the selector is the floor/tie HEDGING, not the ranking: removing the floor drops us to 5.08% / 175 dmg. That also kills the 'exploration' hypothesis - every gun spawns virtual bullets every tick, so sampling is uniform and the bottleneck is SIGNAL QUALITY, not under-sampling. FINAL PER-GUN REAL HIT RATE vs DrussGT (13 runs, 3612 shots, overall 6.95%): Linear 10.7 | Circular 9.9 | KNN 9.0 | Pattern 8.6 | Accel 7.3 | AvgLead 7.0 GuessFactor 6.9 | DecayGF 6.4 | WallBounce 6.2 | StopShot 6.1 | Tsetlin 5.8 Displace 5.3 | HeadOn 5.2 Keep: Linear, Circular, KNN, Pattern, Accel, AvgLead. Marginal: GuessFactor, DecayGF, WallBounce, StopShot. Below overall: Tsetlin, Displace, HeadOn - but HeadOn must STAY as the floor fallback, since disabling the floor measurably hurt. CORRECTION TO A CLAIM I HAVE BEEN MAKING: the 12/12 offline==online acceptance is FLAKY. It fails 11/12 on the UNMODIFIED HEAD source (control: KNN 81 online vs 71 offline), and the mismatching gun moves between runs (KNN, then WallBounce) - a live/offline boundary race. So '12/12' was a lucky run, and that proof should be treated as strong-but-not-exact until the race is fixed. This diff does not touch replayFixture/spawnBullets/tickBullets and the selector is never called during replay, so it is pre-existing. SIDE FINDING, not fixed: the shipped live bot never calls randomize(), so the 'random tie-break' is a FIXED sequence across process restarts. Overfitting guard vs a non-surfer (SpinBot): inconclusive - ModularBot fires only 17-31 real shots/run against fast bots because the range-aware firing gate is strict at long range, so the guard has little power. Wilson looked better (18.5% vs 8.6%) but on 70-92 shots with a 5-33% spread. Not evidence either way. --- common_libs/gun_harness/virtual_bullets.nim | 179 +++++++++++++++++--- 1 file changed, 151 insertions(+), 28 deletions(-) diff --git a/common_libs/gun_harness/virtual_bullets.nim b/common_libs/gun_harness/virtual_bullets.nim index 3c0572c..e237197 100644 --- a/common_libs/gun_harness/virtual_bullets.nim +++ b/common_libs/gun_harness/virtual_bullets.nim @@ -100,6 +100,69 @@ proc parseSelectorMode*(value: string): SelectorMode = let ActiveSelectorMode* = parseSelectorMode(getEnv(SelectorModeEnvVar, "relative")) +# ── runtime ranking knobs (A/B without rebuilding) ──────────────────────────── +# +# Every knob below defaults to the SHIPPED constant, so an unset environment +# reproduces the shipped behaviour byte-for-byte. They exist so one compiled +# binary can be swept across candidate ranking rules. The measured A/B found no +# candidate that credibly beats the shipped statistic: keep the defaults below +# unless a new adversary/run set changes that. + +type + RankStat* = enum + rsMean ## plain window mean (SHIPPED) + rsWilson ## Wilson lower confidence bound (z=1); penalises small n + rsUCB ## mean + c*se (exploration bonus) + rsThompson ## one Normal-approx Beta(h+1,n-h+1) sample per gun (Thompson) + rsShrunk ## empirical-Bayes shrink toward the field mean + +proc envInt(name: string, default: int): int = + let v = getEnv(name, "") + if v.len == 0: return default + try: parseInt(v.strip()) + except ValueError: default + +proc envFloat(name: string, default: float): float = + let v = getEnv(name, "") + if v.len == 0: return default + try: parseFloat(v.strip()) + except ValueError: default + +proc envBool(name: string, default: bool): bool = + case getEnv(name, "").strip().toLowerAscii() + of "1", "true", "yes", "on": true + of "0", "false", "no", "off": false + else: default + +proc parseRank(value: string): RankStat = + case value.strip().toLowerAscii() + of "", "mean", "avg": rsMean + of "wilson", "lcb": rsWilson + of "ucb": rsUCB + of "thompson", "ts": rsThompson + of "shrunk", "shrink", "eb": rsShrunk + else: + stderr.writeLine("[gun_harness] unknown GUN_SELECTOR_RANK='" & value & + "'; falling back to 'mean' (valid: mean|wilson|ucb|thompson|shrunk)") + rsMean + +let ActiveWindow* = clamp(envInt("GUN_SELECTOR_WINDOW", WindowSize), 1, WindowSize) +let ActiveMinObs* = max(1, envInt("GUN_SELECTOR_MINOBS", MinObsBeforeCompete)) +let ActiveRelTie* = envFloat("GUN_SELECTOR_TIE", RelTieMargin) +let ActiveFloorFrac* = envFloat("GUN_SELECTOR_FLOOR", FloorPeakFrac) +let ActivePooled* = envBool("GUN_SELECTOR_POOL", true) +let ActiveRank* = parseRank(getEnv("GUN_SELECTOR_RANK", "")) +let ActiveShrink* = max(0.0, envFloat("GUN_SELECTOR_SHRINK", 20.0)) + +# Optional per-process seed so independent A/B runs use independent tie-breaks +# (Nim's default rand() stream is identical in every process, which would make +# "random" tie-breaks repeat across runs). Unset => leave the RNG untouched. +block: + let s = getEnv("GUN_SELECTOR_SEED", "") + if s.len > 0: + try: randomize(parseInt(s.strip())) + except ValueError: discard + type VirtualBullet* = object gunId*: GunId @@ -156,13 +219,22 @@ proc initTracker*(numGuns: int, metric = ActiveMetric): VirtualTracker = result.numGuns = numGuns result.metric = metric +proc windowHits(fw: FitnessWindow, want: int): int = + ## Hits among the most recent `want` samples, in ring order. Reading the last + ## `want` slots (head backwards) is what makes a runtime window shorter than + ## `WindowSize` correct even after the ring has wrapped. + let n = min(fw.count, min(want, WindowSize)) + for k in 1..n: + let idx = (fw.head - k + WindowSize) mod WindowSize + if fw.hits[idx]: inc result + proc hitRate*(fw: FitnessWindow): float = ## Returns fraction of hits in the rolling window. 0.0 when no data. + ## Uses `ActiveWindow` (defaults to `WindowSize`). if fw.count == 0: return 0.0 - let n = min(fw.count, WindowSize) - var h = 0 - for i in 0..= MinObsBeforeCompete: return true + if fit.bins[binIdx].count >= ActiveMinObs: return true false +proc gunCounts*(fit: GunFitness, pooled: bool): tuple[hits, n: int] = + ## Sample counts behind `gunRate`. `pooled` sums all power bins; otherwise the + ## single bin with the best rate (the "specialist" view). + if pooled: + for binIdx in 0.. best: + best = r + result = (h, m) + proc gunRate*(fit: GunFitness, pooled: bool): float = ## A gun's hit rate. `pooled` sums hits/shots across all power bins (more ## samples, immune to one lucky bin); otherwise the max single-bin rate. - if pooled: - var h, n = 0 - for binIdx in 0.. 0: h.float / n.float else: 0.0 - else: - result = 0.0 - for binIdx in 0.. 0: h.float / n.float else: 0.0 + +proc rankScore(fit: GunFitness, pooled: bool, stat: RankStat, + fieldRate, shrinkK: float): float = + ## Ranking statistic over the gun's window. All are monotone-ish in the mean, + ## but differ in how they trade mean against sample size/noise. `rsMean` is the + ## shipped statistic. + let (h, n) = gunCounts(fit, pooled) + if n == 0: return 0.0 + let p = h.float / n.float + case stat + of rsMean: + p + of rsWilson: + let z = 1.0 + let z2 = z * z + let denom = 1.0 + z2 / n.float + let centre = p + z2 / (2.0 * n.float) + let margin = z * sqrt((p * (1.0 - p) + z2 / (4.0 * n.float)) / n.float) + max(0.0, (centre - margin) / denom) + of rsUCB: + p + 0.5 * sqrt(p * (1.0 - p) / n.float) + of rsThompson: + let se = sqrt(max(1e-9, p * (1.0 - p) / n.float)) + clamp(p + gauss(0.0, se), 0.0, 1.0) + of rsShrunk: + (h.float + shrinkK * fieldRate) / (n.float + shrinkK) proc tableBestRate(t: VirtualTracker, pooled: bool): float = ## Best eligible gun rate across every target's fitness (no merge/allocation). @@ -469,27 +575,35 @@ proc chooseFromFit*(fit: seq[GunFitness], diag: ptr SelectorDiag = nil, ## power bins, since one lucky bin is a poor ranker. ## `referenceRate` <= 0 disables the RELATIVE floor (no history yet). ## Ties (within the band) are broken randomly to avoid index-0 bias. - let pooled = mode == smRelative + let pooled = if mode == smRelative: ActivePooled else: false var anyQualifies = false for gunId in 0..= MinObsBeforeCompete: - anyQualifies = true - break - if anyQualifies: break + if gunEligible(fit[gunId], true): + anyQualifies = true + break let requireMin = anyQualifies if diag != nil: diag[].anyQualifies = requireMin + # Mean rate per eligible gun: used both for the floor reference and (for + # rsShrunk) as the field mean the estimates are pulled toward. var bestRate = 0.0 + var fieldSum = 0.0 + var fieldN = 0 for gunId in 0.. 0: fieldSum / fieldN.float else: 0.0 if diag != nil: diag[].bestRate = bestRate let floorRate = if mode == smAbsolute: MinHitRateFloor - elif referenceRate > 0.0: FloorPeakFrac * referenceRate + elif referenceRate > 0.0: ActiveFloorFrac * referenceRate else: 0.0 if diag != nil: diag[].floorRate = floorRate @@ -498,13 +612,22 @@ proc chooseFromFit*(fit: seq[GunFitness], diag: ptr SelectorDiag = nil, if diag != nil: diag[].floorFired = true return 0 + # Ranking scores (the active statistic) and the best of them. + var scores = newSeq[float](fit.len) + var bestScore = 0.0 + for gunId in 0..= bestRate - tieBand: + if scores[gunId] >= bestScore - tieBand: tied.add(gunId) if diag != nil: diag[].tiedCount = tied.len if tied.len == 0: return 0