diff --git a/common_libs/gun_harness/virtual_bullets.nim b/common_libs/gun_harness/virtual_bullets.nim index 3c0572c..e237197 100644 --- a/common_libs/gun_harness/virtual_bullets.nim +++ b/common_libs/gun_harness/virtual_bullets.nim @@ -100,6 +100,69 @@ proc parseSelectorMode*(value: string): SelectorMode = let ActiveSelectorMode* = parseSelectorMode(getEnv(SelectorModeEnvVar, "relative")) +# ── runtime ranking knobs (A/B without rebuilding) ──────────────────────────── +# +# Every knob below defaults to the SHIPPED constant, so an unset environment +# reproduces the shipped behaviour byte-for-byte. They exist so one compiled +# binary can be swept across candidate ranking rules. The measured A/B found no +# candidate that credibly beats the shipped statistic: keep the defaults below +# unless a new adversary/run set changes that. + +type + RankStat* = enum + rsMean ## plain window mean (SHIPPED) + rsWilson ## Wilson lower confidence bound (z=1); penalises small n + rsUCB ## mean + c*se (exploration bonus) + rsThompson ## one Normal-approx Beta(h+1,n-h+1) sample per gun (Thompson) + rsShrunk ## empirical-Bayes shrink toward the field mean + +proc envInt(name: string, default: int): int = + let v = getEnv(name, "") + if v.len == 0: return default + try: parseInt(v.strip()) + except ValueError: default + +proc envFloat(name: string, default: float): float = + let v = getEnv(name, "") + if v.len == 0: return default + try: parseFloat(v.strip()) + except ValueError: default + +proc envBool(name: string, default: bool): bool = + case getEnv(name, "").strip().toLowerAscii() + of "1", "true", "yes", "on": true + of "0", "false", "no", "off": false + else: default + +proc parseRank(value: string): RankStat = + case value.strip().toLowerAscii() + of "", "mean", "avg": rsMean + of "wilson", "lcb": rsWilson + of "ucb": rsUCB + of "thompson", "ts": rsThompson + of "shrunk", "shrink", "eb": rsShrunk + else: + stderr.writeLine("[gun_harness] unknown GUN_SELECTOR_RANK='" & value & + "'; falling back to 'mean' (valid: mean|wilson|ucb|thompson|shrunk)") + rsMean + +let ActiveWindow* = clamp(envInt("GUN_SELECTOR_WINDOW", WindowSize), 1, WindowSize) +let ActiveMinObs* = max(1, envInt("GUN_SELECTOR_MINOBS", MinObsBeforeCompete)) +let ActiveRelTie* = envFloat("GUN_SELECTOR_TIE", RelTieMargin) +let ActiveFloorFrac* = envFloat("GUN_SELECTOR_FLOOR", FloorPeakFrac) +let ActivePooled* = envBool("GUN_SELECTOR_POOL", true) +let ActiveRank* = parseRank(getEnv("GUN_SELECTOR_RANK", "")) +let ActiveShrink* = max(0.0, envFloat("GUN_SELECTOR_SHRINK", 20.0)) + +# Optional per-process seed so independent A/B runs use independent tie-breaks +# (Nim's default rand() stream is identical in every process, which would make +# "random" tie-breaks repeat across runs). Unset => leave the RNG untouched. +block: + let s = getEnv("GUN_SELECTOR_SEED", "") + if s.len > 0: + try: randomize(parseInt(s.strip())) + except ValueError: discard + type VirtualBullet* = object gunId*: GunId @@ -156,13 +219,22 @@ proc initTracker*(numGuns: int, metric = ActiveMetric): VirtualTracker = result.numGuns = numGuns result.metric = metric +proc windowHits(fw: FitnessWindow, want: int): int = + ## Hits among the most recent `want` samples, in ring order. Reading the last + ## `want` slots (head backwards) is what makes a runtime window shorter than + ## `WindowSize` correct even after the ring has wrapped. + let n = min(fw.count, min(want, WindowSize)) + for k in 1..n: + let idx = (fw.head - k + WindowSize) mod WindowSize + if fw.hits[idx]: inc result + proc hitRate*(fw: FitnessWindow): float = ## Returns fraction of hits in the rolling window. 0.0 when no data. + ## Uses `ActiveWindow` (defaults to `WindowSize`). if fw.count == 0: return 0.0 - let n = min(fw.count, WindowSize) - var h = 0 - for i in 0..= MinObsBeforeCompete: return true + if fit.bins[binIdx].count >= ActiveMinObs: return true false +proc gunCounts*(fit: GunFitness, pooled: bool): tuple[hits, n: int] = + ## Sample counts behind `gunRate`. `pooled` sums all power bins; otherwise the + ## single bin with the best rate (the "specialist" view). + if pooled: + for binIdx in 0.. best: + best = r + result = (h, m) + proc gunRate*(fit: GunFitness, pooled: bool): float = ## A gun's hit rate. `pooled` sums hits/shots across all power bins (more ## samples, immune to one lucky bin); otherwise the max single-bin rate. - if pooled: - var h, n = 0 - for binIdx in 0.. 0: h.float / n.float else: 0.0 - else: - result = 0.0 - for binIdx in 0.. 0: h.float / n.float else: 0.0 + +proc rankScore(fit: GunFitness, pooled: bool, stat: RankStat, + fieldRate, shrinkK: float): float = + ## Ranking statistic over the gun's window. All are monotone-ish in the mean, + ## but differ in how they trade mean against sample size/noise. `rsMean` is the + ## shipped statistic. + let (h, n) = gunCounts(fit, pooled) + if n == 0: return 0.0 + let p = h.float / n.float + case stat + of rsMean: + p + of rsWilson: + let z = 1.0 + let z2 = z * z + let denom = 1.0 + z2 / n.float + let centre = p + z2 / (2.0 * n.float) + let margin = z * sqrt((p * (1.0 - p) + z2 / (4.0 * n.float)) / n.float) + max(0.0, (centre - margin) / denom) + of rsUCB: + p + 0.5 * sqrt(p * (1.0 - p) / n.float) + of rsThompson: + let se = sqrt(max(1e-9, p * (1.0 - p) / n.float)) + clamp(p + gauss(0.0, se), 0.0, 1.0) + of rsShrunk: + (h.float + shrinkK * fieldRate) / (n.float + shrinkK) proc tableBestRate(t: VirtualTracker, pooled: bool): float = ## Best eligible gun rate across every target's fitness (no merge/allocation). @@ -469,27 +575,35 @@ proc chooseFromFit*(fit: seq[GunFitness], diag: ptr SelectorDiag = nil, ## power bins, since one lucky bin is a poor ranker. ## `referenceRate` <= 0 disables the RELATIVE floor (no history yet). ## Ties (within the band) are broken randomly to avoid index-0 bias. - let pooled = mode == smRelative + let pooled = if mode == smRelative: ActivePooled else: false var anyQualifies = false for gunId in 0..= MinObsBeforeCompete: - anyQualifies = true - break - if anyQualifies: break + if gunEligible(fit[gunId], true): + anyQualifies = true + break let requireMin = anyQualifies if diag != nil: diag[].anyQualifies = requireMin + # Mean rate per eligible gun: used both for the floor reference and (for + # rsShrunk) as the field mean the estimates are pulled toward. var bestRate = 0.0 + var fieldSum = 0.0 + var fieldN = 0 for gunId in 0.. 0: fieldSum / fieldN.float else: 0.0 if diag != nil: diag[].bestRate = bestRate let floorRate = if mode == smAbsolute: MinHitRateFloor - elif referenceRate > 0.0: FloorPeakFrac * referenceRate + elif referenceRate > 0.0: ActiveFloorFrac * referenceRate else: 0.0 if diag != nil: diag[].floorRate = floorRate @@ -498,13 +612,22 @@ proc chooseFromFit*(fit: seq[GunFitness], diag: ptr SelectorDiag = nil, if diag != nil: diag[].floorFired = true return 0 + # Ranking scores (the active statistic) and the best of them. + var scores = newSeq[float](fit.len) + var bestScore = 0.0 + for gunId in 0..= bestRate - tieBand: + if scores[gunId] >= bestScore - tieBand: tied.add(gunId) if diag != nil: diag[].tiedCount = tied.len if tied.len == 0: return 0