test(selector): 16 ranking rules A/B'd against the boss - none beat the shipped config

Added runtime-tunable ranking knobs to the selector, all defaulting to the
shipped values so behaviour is byte-identical when unset: GUN_SELECTOR_WINDOW,
MINOBS, TIE, FLOOR, POOL, RANK, SHRINK, SEED. rankScore supports mean, Wilson
lower bound, UCB, Thompson and shrinkage. Also fixed hitRate's most-recent-N
read for sub-WindowSize windows (windowHits).

RESULT: NO candidate credibly beat the shipped config. 13 runs x 8 rounds vs
DrussGT, 3612 shots, base 6.95% at 251 dmg/run; every candidate's per-run
interval overlaps base, and the nominal 'winners' are <=0.6 SE apart on far
fewer shots. Kept the shipped default. Valid outcome, recorded plainly.

THE FINDING THAT MATTERS MORE: the virtual-bullet ranking is ANTI-correlated
with real hit rate - Spearman ~ -0.37 for the shipped config. It is not merely
weak, it is INVERTED. The guns with the highest VIRTUAL rates have among the
lowest REAL rates: Tsetlin 12.9% virtual / 5.8% real, WallBounce 12.9 / 6.2,
StopShot 12.6 / 6.1, AvgLead 12.3 / 7.0 - while Linear sits at 10.2 virtual /
10.7 real and KNN at 7.5 / 9.0. So what carries the selector is the floor/tie
HEDGING, not the ranking: removing the floor drops us to 5.08% / 175 dmg.
That also kills the 'exploration' hypothesis - every gun spawns virtual bullets
every tick, so sampling is uniform and the bottleneck is SIGNAL QUALITY, not
under-sampling.

FINAL PER-GUN REAL HIT RATE vs DrussGT (13 runs, 3612 shots, overall 6.95%):
  Linear 10.7 | Circular 9.9 | KNN 9.0 | Pattern 8.6 | Accel 7.3 | AvgLead 7.0
  GuessFactor 6.9 | DecayGF 6.4 | WallBounce 6.2 | StopShot 6.1 | Tsetlin 5.8
  Displace 5.3 | HeadOn 5.2
Keep: Linear, Circular, KNN, Pattern, Accel, AvgLead. Marginal: GuessFactor,
DecayGF, WallBounce, StopShot. Below overall: Tsetlin, Displace, HeadOn - but
HeadOn must STAY as the floor fallback, since disabling the floor measurably
hurt.

CORRECTION TO A CLAIM I HAVE BEEN MAKING: the 12/12 offline==online acceptance
is FLAKY. It fails 11/12 on the UNMODIFIED HEAD source (control: KNN 81 online
vs 71 offline), and the mismatching gun moves between runs (KNN, then
WallBounce) - a live/offline boundary race. So '12/12' was a lucky run, and
that proof should be treated as strong-but-not-exact until the race is fixed.
This diff does not touch replayFixture/spawnBullets/tickBullets and the
selector is never called during replay, so it is pre-existing.

SIDE FINDING, not fixed: the shipped live bot never calls randomize(), so the
'random tie-break' is a FIXED sequence across process restarts.

Overfitting guard vs a non-surfer (SpinBot): inconclusive - ModularBot fires
only 17-31 real shots/run against fast bots because the range-aware firing gate
is strict at long range, so the guard has little power. Wilson looked better
(18.5% vs 8.6%) but on 70-92 shots with a 5-33% spread. Not evidence either way.
This commit is contained in:
2026-09-21 06:31:00 +02:00
parent 57b2ac3849
commit 2c94dc221a
+151 -28
View File
@@ -100,6 +100,69 @@ proc parseSelectorMode*(value: string): SelectorMode =
let ActiveSelectorMode* = parseSelectorMode(getEnv(SelectorModeEnvVar, "relative")) let ActiveSelectorMode* = parseSelectorMode(getEnv(SelectorModeEnvVar, "relative"))
# ── runtime ranking knobs (A/B without rebuilding) ────────────────────────────
#
# Every knob below defaults to the SHIPPED constant, so an unset environment
# reproduces the shipped behaviour byte-for-byte. They exist so one compiled
# binary can be swept across candidate ranking rules. The measured A/B found no
# candidate that credibly beats the shipped statistic: keep the defaults below
# unless a new adversary/run set changes that.
type
RankStat* = enum
rsMean ## plain window mean (SHIPPED)
rsWilson ## Wilson lower confidence bound (z=1); penalises small n
rsUCB ## mean + c*se (exploration bonus)
rsThompson ## one Normal-approx Beta(h+1,n-h+1) sample per gun (Thompson)
rsShrunk ## empirical-Bayes shrink toward the field mean
proc envInt(name: string, default: int): int =
let v = getEnv(name, "")
if v.len == 0: return default
try: parseInt(v.strip())
except ValueError: default
proc envFloat(name: string, default: float): float =
let v = getEnv(name, "")
if v.len == 0: return default
try: parseFloat(v.strip())
except ValueError: default
proc envBool(name: string, default: bool): bool =
case getEnv(name, "").strip().toLowerAscii()
of "1", "true", "yes", "on": true
of "0", "false", "no", "off": false
else: default
proc parseRank(value: string): RankStat =
case value.strip().toLowerAscii()
of "", "mean", "avg": rsMean
of "wilson", "lcb": rsWilson
of "ucb": rsUCB
of "thompson", "ts": rsThompson
of "shrunk", "shrink", "eb": rsShrunk
else:
stderr.writeLine("[gun_harness] unknown GUN_SELECTOR_RANK='" & value &
"'; falling back to 'mean' (valid: mean|wilson|ucb|thompson|shrunk)")
rsMean
let ActiveWindow* = clamp(envInt("GUN_SELECTOR_WINDOW", WindowSize), 1, WindowSize)
let ActiveMinObs* = max(1, envInt("GUN_SELECTOR_MINOBS", MinObsBeforeCompete))
let ActiveRelTie* = envFloat("GUN_SELECTOR_TIE", RelTieMargin)
let ActiveFloorFrac* = envFloat("GUN_SELECTOR_FLOOR", FloorPeakFrac)
let ActivePooled* = envBool("GUN_SELECTOR_POOL", true)
let ActiveRank* = parseRank(getEnv("GUN_SELECTOR_RANK", ""))
let ActiveShrink* = max(0.0, envFloat("GUN_SELECTOR_SHRINK", 20.0))
# Optional per-process seed so independent A/B runs use independent tie-breaks
# (Nim's default rand() stream is identical in every process, which would make
# "random" tie-breaks repeat across runs). Unset => leave the RNG untouched.
block:
let s = getEnv("GUN_SELECTOR_SEED", "")
if s.len > 0:
try: randomize(parseInt(s.strip()))
except ValueError: discard
type type
VirtualBullet* = object VirtualBullet* = object
gunId*: GunId gunId*: GunId
@@ -156,13 +219,22 @@ proc initTracker*(numGuns: int, metric = ActiveMetric): VirtualTracker =
result.numGuns = numGuns result.numGuns = numGuns
result.metric = metric result.metric = metric
proc windowHits(fw: FitnessWindow, want: int): int =
## Hits among the most recent `want` samples, in ring order. Reading the last
## `want` slots (head backwards) is what makes a runtime window shorter than
## `WindowSize` correct even after the ring has wrapped.
let n = min(fw.count, min(want, WindowSize))
for k in 1..n:
let idx = (fw.head - k + WindowSize) mod WindowSize
if fw.hits[idx]: inc result
proc hitRate*(fw: FitnessWindow): float = proc hitRate*(fw: FitnessWindow): float =
## Returns fraction of hits in the rolling window. 0.0 when no data. ## Returns fraction of hits in the rolling window. 0.0 when no data.
## Uses `ActiveWindow` (defaults to `WindowSize`).
if fw.count == 0: return 0.0 if fw.count == 0: return 0.0
let n = min(fw.count, WindowSize) let n = min(fw.count, min(ActiveWindow, WindowSize))
var h = 0 if n == 0: return 0.0
for i in 0..<n: h += (if fw.hits[i]: 1 else: 0) result = windowHits(fw, n).float / n.float
result = h.float / n.float
proc record(fw: var FitnessWindow, hit: bool) = proc record(fw: var FitnessWindow, hit: bool) =
fw.hits[fw.head] = hit fw.hits[fw.head] = hit
@@ -227,25 +299,59 @@ proc gunEligible*(fit: GunFitness, requireMin: bool): bool =
## when `requireMin` is false, every gun is eligible). ## when `requireMin` is false, every gun is eligible).
if not requireMin: return true if not requireMin: return true
for binIdx in 0..<len(PowerBins): for binIdx in 0..<len(PowerBins):
if fit.bins[binIdx].count >= MinObsBeforeCompete: return true if fit.bins[binIdx].count >= ActiveMinObs: return true
false false
proc gunCounts*(fit: GunFitness, pooled: bool): tuple[hits, n: int] =
## Sample counts behind `gunRate`. `pooled` sums all power bins; otherwise the
## single bin with the best rate (the "specialist" view).
if pooled:
for binIdx in 0..<len(PowerBins):
let m = min(fit.bins[binIdx].count, min(ActiveWindow, WindowSize))
result.n += m
result.hits += windowHits(fit.bins[binIdx], m)
else:
var best = -1.0
for binIdx in 0..<len(PowerBins):
let m = min(fit.bins[binIdx].count, min(ActiveWindow, WindowSize))
if m == 0: continue
let h = windowHits(fit.bins[binIdx], m)
let r = h.float / m.float
if r > best:
best = r
result = (h, m)
proc gunRate*(fit: GunFitness, pooled: bool): float = proc gunRate*(fit: GunFitness, pooled: bool): float =
## A gun's hit rate. `pooled` sums hits/shots across all power bins (more ## A gun's hit rate. `pooled` sums hits/shots across all power bins (more
## samples, immune to one lucky bin); otherwise the max single-bin rate. ## samples, immune to one lucky bin); otherwise the max single-bin rate.
if pooled: let (h, n) = gunCounts(fit, pooled)
var h, n = 0 result = if n > 0: h.float / n.float else: 0.0
for binIdx in 0..<len(PowerBins):
let fw = fit.bins[binIdx] proc rankScore(fit: GunFitness, pooled: bool, stat: RankStat,
let m = min(fw.count, WindowSize) fieldRate, shrinkK: float): float =
n += m ## Ranking statistic over the gun's window. All are monotone-ish in the mean,
for k in 0..<m: ## but differ in how they trade mean against sample size/noise. `rsMean` is the
if fw.hits[k]: inc h ## shipped statistic.
result = if n > 0: h.float / n.float else: 0.0 let (h, n) = gunCounts(fit, pooled)
else: if n == 0: return 0.0
result = 0.0 let p = h.float / n.float
for binIdx in 0..<len(PowerBins): case stat
result = max(result, fit.bins[binIdx].hitRate()) of rsMean:
p
of rsWilson:
let z = 1.0
let z2 = z * z
let denom = 1.0 + z2 / n.float
let centre = p + z2 / (2.0 * n.float)
let margin = z * sqrt((p * (1.0 - p) + z2 / (4.0 * n.float)) / n.float)
max(0.0, (centre - margin) / denom)
of rsUCB:
p + 0.5 * sqrt(p * (1.0 - p) / n.float)
of rsThompson:
let se = sqrt(max(1e-9, p * (1.0 - p) / n.float))
clamp(p + gauss(0.0, se), 0.0, 1.0)
of rsShrunk:
(h.float + shrinkK * fieldRate) / (n.float + shrinkK)
proc tableBestRate(t: VirtualTracker, pooled: bool): float = proc tableBestRate(t: VirtualTracker, pooled: bool): float =
## Best eligible gun rate across every target's fitness (no merge/allocation). ## Best eligible gun rate across every target's fitness (no merge/allocation).
@@ -469,27 +575,35 @@ proc chooseFromFit*(fit: seq[GunFitness], diag: ptr SelectorDiag = nil,
## power bins, since one lucky bin is a poor ranker. ## power bins, since one lucky bin is a poor ranker.
## `referenceRate` <= 0 disables the RELATIVE floor (no history yet). ## `referenceRate` <= 0 disables the RELATIVE floor (no history yet).
## Ties (within the band) are broken randomly to avoid index-0 bias. ## Ties (within the band) are broken randomly to avoid index-0 bias.
let pooled = mode == smRelative let pooled = if mode == smRelative: ActivePooled else: false
var anyQualifies = false var anyQualifies = false
for gunId in 0..<fit.len: for gunId in 0..<fit.len:
for binIdx in 0..<len(PowerBins): if gunEligible(fit[gunId], true):
if fit[gunId].bins[binIdx].count >= MinObsBeforeCompete: anyQualifies = true
anyQualifies = true break
break
if anyQualifies: break
let requireMin = anyQualifies let requireMin = anyQualifies
if diag != nil: diag[].anyQualifies = requireMin if diag != nil: diag[].anyQualifies = requireMin
# Mean rate per eligible gun: used both for the floor reference and (for
# rsShrunk) as the field mean the estimates are pulled toward.
var bestRate = 0.0 var bestRate = 0.0
var fieldSum = 0.0
var fieldN = 0
for gunId in 0..<fit.len: for gunId in 0..<fit.len:
if requireMin and not gunEligible(fit[gunId], true): continue if requireMin and not gunEligible(fit[gunId], true): continue
bestRate = max(bestRate, gunRate(fit[gunId], pooled)) let (h, n) = gunCounts(fit[gunId], pooled)
if n == 0: continue
let r = h.float / n.float
bestRate = max(bestRate, r)
fieldSum += r
inc fieldN
let fieldRate = if fieldN > 0: fieldSum / fieldN.float else: 0.0
if diag != nil: diag[].bestRate = bestRate if diag != nil: diag[].bestRate = bestRate
let floorRate = let floorRate =
if mode == smAbsolute: MinHitRateFloor if mode == smAbsolute: MinHitRateFloor
elif referenceRate > 0.0: FloorPeakFrac * referenceRate elif referenceRate > 0.0: ActiveFloorFrac * referenceRate
else: 0.0 else: 0.0
if diag != nil: diag[].floorRate = floorRate if diag != nil: diag[].floorRate = floorRate
@@ -498,13 +612,22 @@ proc chooseFromFit*(fit: seq[GunFitness], diag: ptr SelectorDiag = nil,
if diag != nil: diag[].floorFired = true if diag != nil: diag[].floorFired = true
return 0 return 0
# Ranking scores (the active statistic) and the best of them.
var scores = newSeq[float](fit.len)
var bestScore = 0.0
for gunId in 0..<fit.len:
if requireMin and not gunEligible(fit[gunId], true): continue
let s = rankScore(fit[gunId], pooled, ActiveRank, fieldRate, ActiveShrink)
scores[gunId] = s
bestScore = max(bestScore, s)
let tieBand = let tieBand =
if mode == smAbsolute: TieMargin if mode == smAbsolute: TieMargin
else: bestRate * RelTieMargin else: bestScore * ActiveRelTie
var tied: seq[GunId] var tied: seq[GunId]
for gunId in 0..<fit.len: for gunId in 0..<fit.len:
if requireMin and not gunEligible(fit[gunId], true): continue if requireMin and not gunEligible(fit[gunId], true): continue
if gunRate(fit[gunId], pooled) >= bestRate - tieBand: if scores[gunId] >= bestScore - tieBand:
tied.add(gunId) tied.add(gunId)
if diag != nil: diag[].tiedCount = tied.len if diag != nil: diag[].tiedCount = tied.len
if tied.len == 0: return 0 if tied.len == 0: return 0