test(selector): 16 ranking rules A/B'd against the boss - none beat the shipped config
Added runtime-tunable ranking knobs to the selector, all defaulting to the shipped values so behaviour is byte-identical when unset: GUN_SELECTOR_WINDOW, MINOBS, TIE, FLOOR, POOL, RANK, SHRINK, SEED. rankScore supports mean, Wilson lower bound, UCB, Thompson and shrinkage. Also fixed hitRate's most-recent-N read for sub-WindowSize windows (windowHits). RESULT: NO candidate credibly beat the shipped config. 13 runs x 8 rounds vs DrussGT, 3612 shots, base 6.95% at 251 dmg/run; every candidate's per-run interval overlaps base, and the nominal 'winners' are <=0.6 SE apart on far fewer shots. Kept the shipped default. Valid outcome, recorded plainly. THE FINDING THAT MATTERS MORE: the virtual-bullet ranking is ANTI-correlated with real hit rate - Spearman ~ -0.37 for the shipped config. It is not merely weak, it is INVERTED. The guns with the highest VIRTUAL rates have among the lowest REAL rates: Tsetlin 12.9% virtual / 5.8% real, WallBounce 12.9 / 6.2, StopShot 12.6 / 6.1, AvgLead 12.3 / 7.0 - while Linear sits at 10.2 virtual / 10.7 real and KNN at 7.5 / 9.0. So what carries the selector is the floor/tie HEDGING, not the ranking: removing the floor drops us to 5.08% / 175 dmg. That also kills the 'exploration' hypothesis - every gun spawns virtual bullets every tick, so sampling is uniform and the bottleneck is SIGNAL QUALITY, not under-sampling. FINAL PER-GUN REAL HIT RATE vs DrussGT (13 runs, 3612 shots, overall 6.95%): Linear 10.7 | Circular 9.9 | KNN 9.0 | Pattern 8.6 | Accel 7.3 | AvgLead 7.0 GuessFactor 6.9 | DecayGF 6.4 | WallBounce 6.2 | StopShot 6.1 | Tsetlin 5.8 Displace 5.3 | HeadOn 5.2 Keep: Linear, Circular, KNN, Pattern, Accel, AvgLead. Marginal: GuessFactor, DecayGF, WallBounce, StopShot. Below overall: Tsetlin, Displace, HeadOn - but HeadOn must STAY as the floor fallback, since disabling the floor measurably hurt. CORRECTION TO A CLAIM I HAVE BEEN MAKING: the 12/12 offline==online acceptance is FLAKY. It fails 11/12 on the UNMODIFIED HEAD source (control: KNN 81 online vs 71 offline), and the mismatching gun moves between runs (KNN, then WallBounce) - a live/offline boundary race. So '12/12' was a lucky run, and that proof should be treated as strong-but-not-exact until the race is fixed. This diff does not touch replayFixture/spawnBullets/tickBullets and the selector is never called during replay, so it is pre-existing. SIDE FINDING, not fixed: the shipped live bot never calls randomize(), so the 'random tie-break' is a FIXED sequence across process restarts. Overfitting guard vs a non-surfer (SpinBot): inconclusive - ModularBot fires only 17-31 real shots/run against fast bots because the range-aware firing gate is strict at long range, so the guard has little power. Wilson looked better (18.5% vs 8.6%) but on 70-92 shots with a 5-33% spread. Not evidence either way.
This commit is contained in:
@@ -100,6 +100,69 @@ proc parseSelectorMode*(value: string): SelectorMode =
|
||||
|
||||
let ActiveSelectorMode* = parseSelectorMode(getEnv(SelectorModeEnvVar, "relative"))
|
||||
|
||||
# ── runtime ranking knobs (A/B without rebuilding) ────────────────────────────
|
||||
#
|
||||
# Every knob below defaults to the SHIPPED constant, so an unset environment
|
||||
# reproduces the shipped behaviour byte-for-byte. They exist so one compiled
|
||||
# binary can be swept across candidate ranking rules. The measured A/B found no
|
||||
# candidate that credibly beats the shipped statistic: keep the defaults below
|
||||
# unless a new adversary/run set changes that.
|
||||
|
||||
type
|
||||
RankStat* = enum
|
||||
rsMean ## plain window mean (SHIPPED)
|
||||
rsWilson ## Wilson lower confidence bound (z=1); penalises small n
|
||||
rsUCB ## mean + c*se (exploration bonus)
|
||||
rsThompson ## one Normal-approx Beta(h+1,n-h+1) sample per gun (Thompson)
|
||||
rsShrunk ## empirical-Bayes shrink toward the field mean
|
||||
|
||||
proc envInt(name: string, default: int): int =
|
||||
let v = getEnv(name, "")
|
||||
if v.len == 0: return default
|
||||
try: parseInt(v.strip())
|
||||
except ValueError: default
|
||||
|
||||
proc envFloat(name: string, default: float): float =
|
||||
let v = getEnv(name, "")
|
||||
if v.len == 0: return default
|
||||
try: parseFloat(v.strip())
|
||||
except ValueError: default
|
||||
|
||||
proc envBool(name: string, default: bool): bool =
|
||||
case getEnv(name, "").strip().toLowerAscii()
|
||||
of "1", "true", "yes", "on": true
|
||||
of "0", "false", "no", "off": false
|
||||
else: default
|
||||
|
||||
proc parseRank(value: string): RankStat =
|
||||
case value.strip().toLowerAscii()
|
||||
of "", "mean", "avg": rsMean
|
||||
of "wilson", "lcb": rsWilson
|
||||
of "ucb": rsUCB
|
||||
of "thompson", "ts": rsThompson
|
||||
of "shrunk", "shrink", "eb": rsShrunk
|
||||
else:
|
||||
stderr.writeLine("[gun_harness] unknown GUN_SELECTOR_RANK='" & value &
|
||||
"'; falling back to 'mean' (valid: mean|wilson|ucb|thompson|shrunk)")
|
||||
rsMean
|
||||
|
||||
let ActiveWindow* = clamp(envInt("GUN_SELECTOR_WINDOW", WindowSize), 1, WindowSize)
|
||||
let ActiveMinObs* = max(1, envInt("GUN_SELECTOR_MINOBS", MinObsBeforeCompete))
|
||||
let ActiveRelTie* = envFloat("GUN_SELECTOR_TIE", RelTieMargin)
|
||||
let ActiveFloorFrac* = envFloat("GUN_SELECTOR_FLOOR", FloorPeakFrac)
|
||||
let ActivePooled* = envBool("GUN_SELECTOR_POOL", true)
|
||||
let ActiveRank* = parseRank(getEnv("GUN_SELECTOR_RANK", ""))
|
||||
let ActiveShrink* = max(0.0, envFloat("GUN_SELECTOR_SHRINK", 20.0))
|
||||
|
||||
# Optional per-process seed so independent A/B runs use independent tie-breaks
|
||||
# (Nim's default rand() stream is identical in every process, which would make
|
||||
# "random" tie-breaks repeat across runs). Unset => leave the RNG untouched.
|
||||
block:
|
||||
let s = getEnv("GUN_SELECTOR_SEED", "")
|
||||
if s.len > 0:
|
||||
try: randomize(parseInt(s.strip()))
|
||||
except ValueError: discard
|
||||
|
||||
type
|
||||
VirtualBullet* = object
|
||||
gunId*: GunId
|
||||
@@ -156,13 +219,22 @@ proc initTracker*(numGuns: int, metric = ActiveMetric): VirtualTracker =
|
||||
result.numGuns = numGuns
|
||||
result.metric = metric
|
||||
|
||||
proc windowHits(fw: FitnessWindow, want: int): int =
|
||||
## Hits among the most recent `want` samples, in ring order. Reading the last
|
||||
## `want` slots (head backwards) is what makes a runtime window shorter than
|
||||
## `WindowSize` correct even after the ring has wrapped.
|
||||
let n = min(fw.count, min(want, WindowSize))
|
||||
for k in 1..n:
|
||||
let idx = (fw.head - k + WindowSize) mod WindowSize
|
||||
if fw.hits[idx]: inc result
|
||||
|
||||
proc hitRate*(fw: FitnessWindow): float =
|
||||
## Returns fraction of hits in the rolling window. 0.0 when no data.
|
||||
## Uses `ActiveWindow` (defaults to `WindowSize`).
|
||||
if fw.count == 0: return 0.0
|
||||
let n = min(fw.count, WindowSize)
|
||||
var h = 0
|
||||
for i in 0..<n: h += (if fw.hits[i]: 1 else: 0)
|
||||
result = h.float / n.float
|
||||
let n = min(fw.count, min(ActiveWindow, WindowSize))
|
||||
if n == 0: return 0.0
|
||||
result = windowHits(fw, n).float / n.float
|
||||
|
||||
proc record(fw: var FitnessWindow, hit: bool) =
|
||||
fw.hits[fw.head] = hit
|
||||
@@ -227,25 +299,59 @@ proc gunEligible*(fit: GunFitness, requireMin: bool): bool =
|
||||
## when `requireMin` is false, every gun is eligible).
|
||||
if not requireMin: return true
|
||||
for binIdx in 0..<len(PowerBins):
|
||||
if fit.bins[binIdx].count >= MinObsBeforeCompete: return true
|
||||
if fit.bins[binIdx].count >= ActiveMinObs: return true
|
||||
false
|
||||
|
||||
proc gunCounts*(fit: GunFitness, pooled: bool): tuple[hits, n: int] =
|
||||
## Sample counts behind `gunRate`. `pooled` sums all power bins; otherwise the
|
||||
## single bin with the best rate (the "specialist" view).
|
||||
if pooled:
|
||||
for binIdx in 0..<len(PowerBins):
|
||||
let m = min(fit.bins[binIdx].count, min(ActiveWindow, WindowSize))
|
||||
result.n += m
|
||||
result.hits += windowHits(fit.bins[binIdx], m)
|
||||
else:
|
||||
var best = -1.0
|
||||
for binIdx in 0..<len(PowerBins):
|
||||
let m = min(fit.bins[binIdx].count, min(ActiveWindow, WindowSize))
|
||||
if m == 0: continue
|
||||
let h = windowHits(fit.bins[binIdx], m)
|
||||
let r = h.float / m.float
|
||||
if r > best:
|
||||
best = r
|
||||
result = (h, m)
|
||||
|
||||
proc gunRate*(fit: GunFitness, pooled: bool): float =
|
||||
## A gun's hit rate. `pooled` sums hits/shots across all power bins (more
|
||||
## samples, immune to one lucky bin); otherwise the max single-bin rate.
|
||||
if pooled:
|
||||
var h, n = 0
|
||||
for binIdx in 0..<len(PowerBins):
|
||||
let fw = fit.bins[binIdx]
|
||||
let m = min(fw.count, WindowSize)
|
||||
n += m
|
||||
for k in 0..<m:
|
||||
if fw.hits[k]: inc h
|
||||
result = if n > 0: h.float / n.float else: 0.0
|
||||
else:
|
||||
result = 0.0
|
||||
for binIdx in 0..<len(PowerBins):
|
||||
result = max(result, fit.bins[binIdx].hitRate())
|
||||
let (h, n) = gunCounts(fit, pooled)
|
||||
result = if n > 0: h.float / n.float else: 0.0
|
||||
|
||||
proc rankScore(fit: GunFitness, pooled: bool, stat: RankStat,
|
||||
fieldRate, shrinkK: float): float =
|
||||
## Ranking statistic over the gun's window. All are monotone-ish in the mean,
|
||||
## but differ in how they trade mean against sample size/noise. `rsMean` is the
|
||||
## shipped statistic.
|
||||
let (h, n) = gunCounts(fit, pooled)
|
||||
if n == 0: return 0.0
|
||||
let p = h.float / n.float
|
||||
case stat
|
||||
of rsMean:
|
||||
p
|
||||
of rsWilson:
|
||||
let z = 1.0
|
||||
let z2 = z * z
|
||||
let denom = 1.0 + z2 / n.float
|
||||
let centre = p + z2 / (2.0 * n.float)
|
||||
let margin = z * sqrt((p * (1.0 - p) + z2 / (4.0 * n.float)) / n.float)
|
||||
max(0.0, (centre - margin) / denom)
|
||||
of rsUCB:
|
||||
p + 0.5 * sqrt(p * (1.0 - p) / n.float)
|
||||
of rsThompson:
|
||||
let se = sqrt(max(1e-9, p * (1.0 - p) / n.float))
|
||||
clamp(p + gauss(0.0, se), 0.0, 1.0)
|
||||
of rsShrunk:
|
||||
(h.float + shrinkK * fieldRate) / (n.float + shrinkK)
|
||||
|
||||
proc tableBestRate(t: VirtualTracker, pooled: bool): float =
|
||||
## Best eligible gun rate across every target's fitness (no merge/allocation).
|
||||
@@ -469,27 +575,35 @@ proc chooseFromFit*(fit: seq[GunFitness], diag: ptr SelectorDiag = nil,
|
||||
## power bins, since one lucky bin is a poor ranker.
|
||||
## `referenceRate` <= 0 disables the RELATIVE floor (no history yet).
|
||||
## Ties (within the band) are broken randomly to avoid index-0 bias.
|
||||
let pooled = mode == smRelative
|
||||
let pooled = if mode == smRelative: ActivePooled else: false
|
||||
|
||||
var anyQualifies = false
|
||||
for gunId in 0..<fit.len:
|
||||
for binIdx in 0..<len(PowerBins):
|
||||
if fit[gunId].bins[binIdx].count >= MinObsBeforeCompete:
|
||||
anyQualifies = true
|
||||
break
|
||||
if anyQualifies: break
|
||||
if gunEligible(fit[gunId], true):
|
||||
anyQualifies = true
|
||||
break
|
||||
let requireMin = anyQualifies
|
||||
if diag != nil: diag[].anyQualifies = requireMin
|
||||
|
||||
# Mean rate per eligible gun: used both for the floor reference and (for
|
||||
# rsShrunk) as the field mean the estimates are pulled toward.
|
||||
var bestRate = 0.0
|
||||
var fieldSum = 0.0
|
||||
var fieldN = 0
|
||||
for gunId in 0..<fit.len:
|
||||
if requireMin and not gunEligible(fit[gunId], true): continue
|
||||
bestRate = max(bestRate, gunRate(fit[gunId], pooled))
|
||||
let (h, n) = gunCounts(fit[gunId], pooled)
|
||||
if n == 0: continue
|
||||
let r = h.float / n.float
|
||||
bestRate = max(bestRate, r)
|
||||
fieldSum += r
|
||||
inc fieldN
|
||||
let fieldRate = if fieldN > 0: fieldSum / fieldN.float else: 0.0
|
||||
if diag != nil: diag[].bestRate = bestRate
|
||||
|
||||
let floorRate =
|
||||
if mode == smAbsolute: MinHitRateFloor
|
||||
elif referenceRate > 0.0: FloorPeakFrac * referenceRate
|
||||
elif referenceRate > 0.0: ActiveFloorFrac * referenceRate
|
||||
else: 0.0
|
||||
if diag != nil: diag[].floorRate = floorRate
|
||||
|
||||
@@ -498,13 +612,22 @@ proc chooseFromFit*(fit: seq[GunFitness], diag: ptr SelectorDiag = nil,
|
||||
if diag != nil: diag[].floorFired = true
|
||||
return 0
|
||||
|
||||
# Ranking scores (the active statistic) and the best of them.
|
||||
var scores = newSeq[float](fit.len)
|
||||
var bestScore = 0.0
|
||||
for gunId in 0..<fit.len:
|
||||
if requireMin and not gunEligible(fit[gunId], true): continue
|
||||
let s = rankScore(fit[gunId], pooled, ActiveRank, fieldRate, ActiveShrink)
|
||||
scores[gunId] = s
|
||||
bestScore = max(bestScore, s)
|
||||
|
||||
let tieBand =
|
||||
if mode == smAbsolute: TieMargin
|
||||
else: bestRate * RelTieMargin
|
||||
else: bestScore * ActiveRelTie
|
||||
var tied: seq[GunId]
|
||||
for gunId in 0..<fit.len:
|
||||
if requireMin and not gunEligible(fit[gunId], true): continue
|
||||
if gunRate(fit[gunId], pooled) >= bestRate - tieBand:
|
||||
if scores[gunId] >= bestScore - tieBand:
|
||||
tied.add(gunId)
|
||||
if diag != nil: diag[].tiedCount = tied.len
|
||||
if tied.len == 0: return 0
|
||||
|
||||
Reference in New Issue
Block a user