Files

1297 lines
60 KiB
Nim
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
## Virtual bullet tracker.
## Spawns virtual bullets per gun×power bin every tick (no real firing).
## Resolves by travel distance. Rolling window fitness per gun×power.
## Calls onResult() on the owning gun when a bullet resolves.
import std/math
import std/tables
import std/random
import std/algorithm
import std/os
import std/strutils
import gun_interface
const
PowerBins* = [1.0, 1.5, 2.0, 3.0] ## 4 bins; ponytail: fixed array, add runtime config if needed
WindowSize* = 100 ## rolling window ticks for fitness
MaxBullets* = 8192 ## hard cap; ring buffer. 52 spawns/tick and a
## full-map long shot (~90 ticks) need ~4700 slots;
## 8192 wraps only after ~157 ticks. Each VirtualBullet
## is ~120 bytes, so this array costs ~960 KiB.
MinHitRate* = 0.40 ## LEGACY absolute bar; no longer used by bestPower
## (no bin on the live path-metric scale cleared it, so
## once every bin had data bestPower fell to power 1.0).
PowerBarFrac* = 0.50 ## RELATIVE power bar (dimensionless): a bin is
## acceptable when its virtual hit rate is at least this
## FRACTION of the same gun's best bin rate. Scales with
## the metric instead of assuming a ~40% hit rate.
MinObsBeforeCompete* = 50 ## min observations before a gun×bin enters competition
TieMargin* = 0.02 ## ABSOLUTE mode: guns within this hit-rate margin of best are tied
MinHitRateFloor* = 0.10 ## ABSOLUTE mode: if best gun < this, fall back to gun 0 (HeadOn)
RelTieMargin* = 0.20 ## RELATIVE mode: tied if rate >= bestRate*(1-this). Dimensionless
## fraction of the best rate, so it scales with the metric.
FloorPeakFrac* = 0.25 ## RELATIVE mode: floor fires if bestRate < this*peakRateRef.
## Dimensionless: only if the field collapsed vs its own recent best.
SelectorWindow* = 256 ## ticks of per-tick bestRate kept for the RELATIVE floor reference
# ── selector hysteresis (anti-chatter) ─────────────────────────────────────
#
# Without these, `chooseFromFit` performs a fresh uniform draw among the tied
# set EVERY tick. With the 20% relative tie band that is most of the rack, so
# the turret re-aims on ~37% of ticks (live config log; offline replay of the
# same selection path: 54.3 switches / 100 ticks). The two constants below
# make the selector commit.
#
# Shipped values are the CONSERVATIVE end of a measured sweep against the real
# DrussGT (8-16 runs x 7 rounds per setting, server-side events sidecar).
# Offline replay of the same selection path gives the chatter column:
# dwell 0 / margin 0.00 : chatter 54.3/100t real 7.02% 217 dmg/run
# dwell 10 / margin 0.05 : chatter 1.95/100t real 6.22% 191 dmg/run
# dwell 30 / margin 0.15 : chatter 1.33/100t real 5.10% 156 dmg/run
# dwell 60 / margin 0.30 : real 5.72% 175 dmg/run (8 runs)
# The light setting cuts chatter ~28x and is the only one not distinguishable
# from the no-hysteresis baseline (overlapping per-run ranges, permutation
# p=0.18); the heavier settings cost real hit rate (moderate p=0.002). So the
# conservative setting ships. Set BOTH env knobs to 0 to disable hysteresis.
GunDwellTicks* = 10 ## TICKS: once selected, a gun is held for at least
## this many ticks unless it becomes disqualified
## (below the eligibility sample gate) or the field
## collapses (existing floor path -> HeadOn).
GunSwitchMargin* = 0.05 ## DIMENSIONLESS fraction of the incumbent's score: a
## challenger must beat the incumbent by more than
## this (score > incumbent * (1 + margin)) before it is
## allowed to displace it. A merely-tied gun cannot.
MetricEnvVar* = "GUN_VBULLET_METRIC"
## RUNTIME switch selecting how a virtual bullet is scored. Read once per
## process at module init, so the SAME compiled binary can be A/B'd by
## exporting it — no rebuild needed. Both the live ModularBot tracker and
## the offline range replay call `initTracker`, so they always agree.
type
GunId* = int ## index into the guns seq
BulletMetric* = enum
bmPoint ## A bullet is scored at the single point it reaches at the
## fire-time aim distance. HIT iff that point is within BotRadius of
## the target on that tick. Measures prediction accuracy (does the
## bullet arrive at the predicted point at the right time).
bmPath ## DEFAULT. The bullet flies along its straight ray until it leaves
## the arena. Each tick the swept segment (previous -> new position)
## is tested against the target's radius; HIT iff ANY segment came
## within BotRadius. Measures hypothetical hit chance against the
## target's real path. Chosen by the DrussGT A/B: 7.2-7.6% real hit
## rate vs 3.8% for point (p<0.0001).
const DefaultMetric* = bmPath
## Shipped virtual-bullet scoring model. `GUN_VBULLET_METRIC` overrides it at
## runtime; an unset OR empty value means this default.
proc parseMetric*(value: string): BulletMetric =
## Parse a `GUN_VBULLET_METRIC` value. Empty / unknown values fall back to
## the shipped `DefaultMetric` and emit a one-line warning on stderr, so a
## typo can never silently change the metric and a bad value can never take
## the bot down.
case value.strip().toLowerAscii()
of "", "default": DefaultMetric
of "point", "points", "bmpoint": bmPoint
of "path", "paths", "bmpath": bmPath
else:
stderr.writeLine("[gun_harness] unknown " & MetricEnvVar & "='" & value &
"'; falling back to '" & $DefaultMetric & "' (valid: point|path)")
DefaultMetric
let ActiveMetric* = parseMetric(getEnv(MetricEnvVar, ""))
## The metric every tracker uses unless a caller overrides it explicitly in
## `initTracker`. Frozen at process start from the environment.
const SelectorModeEnvVar* = "GUN_SELECTOR_MODE"
## RUNTIME switch selecting the selection-threshold model. Read once per
## process, so one binary can A/B both (§ virtual_bullets).
type
SelectorMode* = enum
smAbsolute ## legacy: fixed 2pp tie band + 10% absolute floor. Correct only
## if the virtual hit-rate scale happens to land near 10%.
smRelative ## scale-aware: tie band is a fraction of the best rate; the floor
## fires only when the field has collapsed vs its own recent peak.
proc parseSelectorMode*(value: string): SelectorMode =
## Empty / unknown values fall back to the shipped `relative` model and warn.
case value.strip().toLowerAscii()
of "", "relative", "rel": smRelative
of "absolute", "abs", "legacy": smAbsolute
else:
stderr.writeLine("[gun_harness] unknown " & SelectorModeEnvVar & "='" & value &
"'; falling back to 'relative' (valid: absolute|relative)")
smRelative
let ActiveSelectorMode* = parseSelectorMode(getEnv(SelectorModeEnvVar, "relative"))
# ── arrival-accuracy tie-break (GUN_SELECTOR_TIEBREAK) ───────────────────────
#
# The shipped selector ranks guns by the `path` metric (the swept-ray score,
# measurably the better coarse signal: 7.43% vs 4.70% real hit rate) and then
# draws the shot UNIFORMLY at random inside a tie band built on that rate. The
# randomness is load-bearing (replacing it with commitment to the virtual best
# cost real hit rate, 7.02% -> 5.10%), but `path` is deliberately generous: it
# asks "does the ray ever sweep the target's path", which a gun can satisfy
# while its bullets ARRIVE poorly. Such a gun then sits inside the band and
# gets picked.
#
# The tie-break below keeps the band construction on `path` (and therefore the
# band's measured advantage) and makes the DRAW narrower by arrival accuracy:
# each virtual bullet is additionally scored with the `point` model (the
# prediction-accuracy score at the exact tick the bullet reaches its aim
# distance) into a parallel `pointBins` window, and the tied set is restricted
# to the guns whose point rate is within `GUN_SELECTOR_POINT_TIE` of the best
# point rate in the band. It is still a random draw inside a band — just a
# better-informed band — so the load-bearing randomness is preserved.
#
# `off` is the SHIPPED default, so an unset environment behaves exactly as
# before, and the parallel point scoring is not even performed.
const TieBreakEnvVar* = "GUN_SELECTOR_TIEBREAK"
type
TieBreakMode* = enum
tbOff ## DEFAULT: pure `path` band, uniform random draw.
tbPoint ## narrow the `path` tie band to the point-accurate guns.
tbPointCommit ## CONTROL (removes the randomness): take the best point rate
## inside the band, deterministically. Measured to be worse
## when done on the path rate; included so the point variant
## can be compared against its own commitment control.
proc parseTieBreak*(value: string): TieBreakMode =
## Empty / unknown values fall back to the shipped `off` and warn.
case value.strip().toLowerAscii()
of "", "off", "none", "0", "false": tbOff
of "point", "arrival", "p": tbPoint
of "commit", "pointcommit", "best": tbPointCommit
else:
stderr.writeLine("[gun_harness] unknown " & TieBreakEnvVar & "='" & value &
"'; falling back to 'off' (valid: off|point|commit)")
tbOff
# ── rack membership (melee vs 1v1) ────────────────────────────────────────────
#
# The selector can run two racks and switch between them on SERVER truth —
# `getEnemyCount() == 1` -> 1v1, `> 1` -> melee (`rackMode`) — the same value
# the live radar uses to pick RadarLock vs AdaptiveMelee. Each gun declares
# which rack(s) admit it. DEFAULT IS `rmBoth` FOR EVERY GUN, so an unset
# environment reproduces the pre-change (single-rack) selection byte-for-byte.
#
# The membership test lives here, next to `chooseFromFit`, because it filters
# that function's candidate set. The env parsing / gun-name table live in
# selector.nim, which owns the rack's gun names (this low-level module is
# generic over `fit.len` and does not).
type
RackMode* = enum
rm1v1 ## exactly one enemy alive (server truth)
rmMelee ## two or more enemies alive (server truth)
RackMembership* = enum
rmBoth ## DEFAULT: eligible in 1v1 AND melee
rmOnly1v1 ## eligible only while exactly one enemy is alive
rmOnlyMelee ## eligible only in melee (2+ enemies)
rmOff ## never eligible (removed from the selector's rack)
proc admits*(m: RackMembership, mode: RackMode): bool =
## True when membership `m` includes rack `mode`.
case m
of rmBoth: true
of rmOnly1v1: mode == rm1v1
of rmOnlyMelee: mode == rmMelee
of rmOff: false
proc rackMode*(enemyCount: int): RackMode =
## THE single definition of the mode, derived from server truth. The live
## radar derives its target mode from the same count in ModularBot.run.
if enemyCount == 1: rm1v1 else: rmMelee
proc rackAdmitted*(gunId: int, mode: RackMode,
membership: openArray[RackMembership]): bool =
## True when `gunId` is a candidate in `mode`. An id past the membership table
## is admitted, so an empty table means "no filtering".
if gunId < 0: return false
if gunId < membership.len: return membership[gunId].admits(mode)
true
proc admittedGuns*(fitLen: int, mode: RackMode,
membership: openArray[RackMembership]): seq[GunId] =
## Gun ids whose membership admits `mode`. GRACEFUL DEGRADATION: if the
## filtered set is EMPTY (e.g. every gun is `off`, or no membership table is
## supplied), return the full rack so the selector can never end up with no
## gun.
for gunId in 0..<fitLen:
if gunId < membership.len and not membership[gunId].admits(mode):
continue
result.add gunId
if result.len == 0:
for gunId in 0..<fitLen: result.add gunId
proc vBulletAdmitted*(gunId: int, mode: RackMode,
membership: openArray[RackMembership],
admitOnly: bool = true): bool =
## `TR_VBULLET_ADMIT_ONLY` gate for the LIVE predict+spawn block.
##
## Rack membership historically filtered only SELECTION, so every unselected
## gun still ran `predict`/`spawnBullets` each tick to feed a fitness table the
## selector would never read. This predicate lets the live loop skip BOTH for a
## gun the current rack does not admit.
##
## admitOnly = true (SHIPPED default) — a gun spawns only while admitted.
## admitOnly = false — every gun spawns (the pre-change behaviour).
##
## `onResult` feedback is deliberately NOT covered here: a bullet spawned
## while admitted must still resolve and be attributed to its gun even if the
## rack changes mid-round.
(not admitOnly) or rackAdmitted(gunId, mode, membership)
# ── runtime ranking knobs (A/B without rebuilding) ────────────────────────────
#
# Every knob below defaults to the SHIPPED constant, so an unset environment
# reproduces the shipped behaviour byte-for-byte. They exist so one compiled
# binary can be swept across candidate ranking rules. The measured A/B found no
# candidate that credibly beats the shipped statistic: keep the defaults below
# unless a new adversary/run set changes that.
type
RankStat* = enum
rsMean ## plain window mean (SHIPPED)
rsWilson ## Wilson lower confidence bound (z=1); penalises small n
rsUCB ## mean + c*se (exploration bonus)
rsThompson ## one Normal-approx Beta(h+1,n-h+1) sample per gun (Thompson)
rsShrunk ## empirical-Bayes shrink toward the field mean
proc envInt(name: string, default: int): int =
let v = getEnv(name, "")
if v.len == 0: return default
try: parseInt(v.strip())
except ValueError: default
proc envFloat(name: string, default: float): float =
let v = getEnv(name, "")
if v.len == 0: return default
try: parseFloat(v.strip())
except ValueError: default
proc envBool(name: string, default: bool): bool =
case getEnv(name, "").strip().toLowerAscii()
of "1", "true", "yes", "on": true
of "0", "false", "no", "off": false
else: default
proc parseRank(value: string): RankStat =
case value.strip().toLowerAscii()
of "", "mean", "avg": rsMean
of "wilson", "lcb": rsWilson
of "ucb": rsUCB
of "thompson", "ts": rsThompson
of "shrunk", "shrink", "eb": rsShrunk
else:
stderr.writeLine("[gun_harness] unknown GUN_SELECTOR_RANK='" & value &
"'; falling back to 'mean' (valid: mean|wilson|ucb|thompson|shrunk)")
rsMean
let ActiveWindow* = clamp(envInt("GUN_SELECTOR_WINDOW", WindowSize), 1, WindowSize)
let ActiveMinObs* = max(1, envInt("GUN_SELECTOR_MINOBS", MinObsBeforeCompete))
let ActiveRelTie* = envFloat("GUN_SELECTOR_TIE", RelTieMargin)
let ActiveFloorFrac* = envFloat("GUN_SELECTOR_FLOOR", FloorPeakFrac)
let ActivePooled* = envBool("GUN_SELECTOR_POOL", true)
let ActiveRank* = parseRank(getEnv("GUN_SELECTOR_RANK", ""))
let ActiveShrink* = max(0.0, envFloat("GUN_SELECTOR_SHRINK", 20.0))
# Runtime hysteresis knobs, read once per process like the rest, so one binary
# can be swept. Defaults equal the shipped constants; setting BOTH to 0
# reproduces the pre-hysteresis (pure per-tick) selection for A/B.
let ActiveDwellTicks* = max(0, envInt("GUN_SELECTOR_DWELL", GunDwellTicks))
let ActiveSwitchMargin* = max(0.0, envFloat("GUN_SELECTOR_MARGIN", GunSwitchMargin))
# Arrival-accuracy tie-break: mode plus the RELATIVE width of the point band
# (fraction of the best in-band point rate), matching the `RelTieMargin` style.
let ActiveTieBreak* = parseTieBreak(getEnv(TieBreakEnvVar, ""))
let ActivePointTie* = clamp(envFloat("GUN_SELECTOR_POINT_TIE", 0.5), 0.0, 1.0)
# A `point` tie-break needs the parallel point windows, which only the `path`
# metric records (under `point` the primary ranking already IS arrival
# accuracy, so the tie-break would be a no-op). Warn loudly rather than
# silently measuring nothing.
block:
if ActiveTieBreak != tbOff and ActiveMetric == bmPoint:
stderr.writeLine("[gun_harness] " & TieBreakEnvVar & "=" & $ActiveTieBreak &
" has no effect under " & MetricEnvVar & "=point (the primary " &
"ranking already uses arrival accuracy)")
elif ActiveTieBreak != tbOff:
# One audit line per process so a live run's log records which arm it ran.
stderr.writeLine("[gun_harness] " & TieBreakEnvVar & "=" & $ActiveTieBreak &
" pointTie=" & $ActivePointTie)
# Optional per-process seed so independent A/B runs use independent tie-breaks
# (Nim's default rand() stream is identical in every process, which would make
# "random" tie-breaks repeat across runs). Unset => leave the RNG untouched.
block:
let s = getEnv("GUN_SELECTOR_SEED", "")
if s.len > 0:
try: randomize(parseInt(s.strip()))
except ValueError: discard
type
VirtualBullet* = object
gunId*: GunId
powerBin*: int ## index into PowerBins
targetId*: int ## enemy bot ID this bullet was aimed at
fireTick*: int ## tick this bullet was spawned; lets a gun pair its
## predict() trace with the exact resolution event
fireX*, fireY*: float
aimX*, aimY*: float ## predicted target (absolute)
bulletSpeed*: float
travelDist*: float ## accumulated px so far
fireDist*: float ## distance to target at fire time
confidence*: float ## spawn-time GunPrediction.confidence (TMComposites gate)
active*: bool
pointScored*: bool ## the parallel point-model score for this flight has
## been recorded (tie-break mode only)
# --- path-metric bookkeeping (unused by the point metric) ---
hitSeen*: bool ## a swept segment already touched the target
bestMissDist*: float ## closest segment->target distance seen so far
bestMissX*: float ## target position at that closest approach
bestMissY*: float
FitnessWindow* = object
## Ring buffer of hit booleans.
hits*: array[WindowSize, bool]
count*: int ## total samples so far (capped at WindowSize for rate)
head*: int
GunFitness* = object
bins*: array[len(PowerBins), FitnessWindow]
pointBins*: array[len(PowerBins), FitnessWindow]
## Parallel `point`-model window, populated ONLY while the arrival-accuracy
## tie-break is active. The `path` ranking continues to read `bins`; the
## tie-break reads `pointBins` for the SAME bullets, so both scores come
## from one flight with no extra virtual bullets.
SelectorDiag* = object
## Optional observability for `chooseFromFit`/`bestGun`. Never needed by the
## bot; lets the offline range report WHY a gun was selected (floor vs tie).
bestRate*: float ## max hit rate over eligible guns (the floor comparison value)
floorFired*: bool ## bestRate below the active floor -> returned gun 0
tiedCount*: int ## eligible guns within the tie band of bestRate (0 if floor fired)
anyQualifies*: bool ## at least one gun reached MinObsBeforeCompete
floorRate*: float ## the floor actually applied this tick
incumbentKept*: bool ## hysteresis retained the incumbent this tick
pointTiedCount*: int ## guns kept after the arrival-accuracy tie-break
bestPointRate*: float ## best in-band point rate (the tie-break reference)
pointTieFired*: bool ## the tie-break actually narrowed the band
VirtualTracker* = object
bullets*: array[MaxBullets, VirtualBullet]
head*: int ## ring buffer head
numGuns*: int
metric*: BulletMetric ## scoring model (defaults to ActiveMetric)
fitness*: Table[int, seq[GunFitness]] ## keyed by enemy bot ID, indexed by GunId
droppedBullets*: int ## unresolved bullets clobbered by the ring buffer (should stay 0)
# RELATIVE-mode floor reference: per-tick bestRate history + its running max.
rateHist*: array[SelectorWindow, float]
rateHistHead*: int
rateHistCount*: int
peakRateRef*: float ## max bestRate in rateHist; 0.0 = not enough history yet
# Hysteresis state (see `selectGun`): the incumbent gun and the tick it was
# chosen on. `-1` means no gun selected yet. Only the live `selectGun` path
# reads/writes these; `bestGun`/`chooseFromFit` stay memoryless.
currentGun*: GunId
currentSince*: int
# Forced-share allocator state (`TR_RACK_SHARE`, see `selectSharedGun`).
# `shareDeficit` accumulates one round of weights per allocation EPOCH; the
# chosen gun has 1.0 subtracted, so the long-run allocation follows the
# weights exactly. Empty/unused when the share is inactive.
shareDeficit*: seq[float]
proc initTracker*(numGuns: int, metric = ActiveMetric): VirtualTracker =
## `metric` defaults to the process-wide `GUN_VBULLET_METRIC` switch; pass it
## explicitly only from tests that need both models in one process.
result.numGuns = numGuns
result.metric = metric
result.currentGun = -1
proc windowHits(fw: FitnessWindow, want: int): int =
## Hits among the most recent `want` samples, in ring order. Reading the last
## `want` slots (head backwards) is what makes a runtime window shorter than
## `WindowSize` correct even after the ring has wrapped.
let n = min(fw.count, min(want, WindowSize))
for k in 1..n:
let idx = (fw.head - k + WindowSize) mod WindowSize
if fw.hits[idx]: inc result
proc hitRate*(fw: FitnessWindow): float =
## Returns fraction of hits in the rolling window. 0.0 when no data.
## Uses `ActiveWindow` (defaults to `WindowSize`).
if fw.count == 0: return 0.0
let n = min(fw.count, min(ActiveWindow, WindowSize))
if n == 0: return 0.0
result = windowHits(fw, n).float / n.float
proc record(fw: var FitnessWindow, hit: bool) =
fw.hits[fw.head] = hit
fw.head = (fw.head + 1) mod WindowSize
inc fw.count
proc spawnBullets*(t: var VirtualTracker, gunId: GunId,
predictions: array[len(PowerBins), GunPrediction],
state: WorldState, targetId: int) =
## Call once per gun per tick with predictions for all power bins.
## Lazily creates fitness entry for targetId on first spawn.
if targetId notin t.fitness:
t.fitness[targetId] = newSeq[GunFitness](t.numGuns)
for binIdx in 0..<len(PowerBins):
let power = PowerBins[binIdx]
let speed = bulletSpeed(power)
let pred = predictions[binIdx]
let fireDist = hypot(pred.x - state.selfX, pred.y - state.selfY)
let slot = t.head mod MaxBullets
# Measurement integrity: if the slot we are about to overwrite still holds an
# unresolved bullet, that bullet will never be scored. Count it instead of
# silently dropping it (non-zero after a battle means MaxBullets is too small).
if t.bullets[slot].active:
inc t.droppedBullets
t.bullets[slot] = VirtualBullet(
gunId: gunId,
powerBin: binIdx,
targetId: targetId,
fireTick: state.tick,
fireX: state.selfX,
fireY: state.selfY,
aimX: pred.x,
aimY: pred.y,
bulletSpeed: speed,
travelDist: 0.0,
fireDist: fireDist,
confidence: pred.confidence,
active: true,
pointScored: false,
hitSeen: false,
bestMissDist: Inf,
bestMissX: 0.0,
bestMissY: 0.0,
)
t.head = (t.head + 1) mod MaxBullets
const StaleTicks* = 20 ## discard bullet if target not seen within this many ticks
proc distPointToSegment*(px, py, ax, ay, bx, by: float): float =
## Shortest distance from point P to the segment A-B (A/B are bullet
## positions on consecutive ticks).
let abx = bx - ax
let aby = by - ay
let abLen2 = abx*abx + aby*aby
var s = 0.0
if abLen2 > 1e-12:
s = clamp(((px - ax)*abx + (py - ay)*aby) / abLen2, 0.0, 1.0)
hypot(px - (ax + s*abx), py - (ay + s*aby))
# ── rate helpers (shared by the selector and the reference tracker) ──────────
proc gunEligible*(fit: GunFitness, requireMin: bool): bool =
## True if a gun has >= MinObsBeforeCompete samples in at least one bin (or
## when `requireMin` is false, every gun is eligible).
if not requireMin: return true
for binIdx in 0..<len(PowerBins):
if fit.bins[binIdx].count >= ActiveMinObs: return true
false
proc gunCounts*(fit: GunFitness, pooled: bool): tuple[hits, n: int] =
## Sample counts behind `gunRate`. `pooled` sums all power bins; otherwise the
## single bin with the best rate (the "specialist" view).
if pooled:
for binIdx in 0..<len(PowerBins):
let m = min(fit.bins[binIdx].count, min(ActiveWindow, WindowSize))
result.n += m
result.hits += windowHits(fit.bins[binIdx], m)
else:
var best = -1.0
for binIdx in 0..<len(PowerBins):
let m = min(fit.bins[binIdx].count, min(ActiveWindow, WindowSize))
if m == 0: continue
let h = windowHits(fit.bins[binIdx], m)
let r = h.float / m.float
if r > best:
best = r
result = (h, m)
proc gunRate*(fit: GunFitness, pooled: bool): float =
## A gun's hit rate. `pooled` sums hits/shots across all power bins (more
## samples, immune to one lucky bin); otherwise the max single-bin rate.
let (h, n) = gunCounts(fit, pooled)
result = if n > 0: h.float / n.float else: 0.0
proc pointCounts*(fit: GunFitness, pooled: bool): tuple[hits, n: int] =
## Sample counts behind `pointRate`: the PARALLEL point-model window. All zero
## unless the arrival-accuracy tie-break is active, which is what makes the
## tie-break a graceful no-op on every existing caller.
if pooled:
for binIdx in 0..<len(PowerBins):
let m = min(fit.pointBins[binIdx].count, min(ActiveWindow, WindowSize))
result.n += m
result.hits += windowHits(fit.pointBins[binIdx], m)
else:
var best = -1.0
for binIdx in 0..<len(PowerBins):
let m = min(fit.pointBins[binIdx].count, min(ActiveWindow, WindowSize))
if m == 0: continue
let h = windowHits(fit.pointBins[binIdx], m)
let r = h.float / m.float
if r > best:
best = r
result = (h, m)
proc pointRate*(fit: GunFitness, pooled: bool): float =
## Arrival-accuracy rate over the parallel point window (0.0 when no data).
let (h, n) = pointCounts(fit, pooled)
result = if n > 0: h.float / n.float else: 0.0
proc rankScore(fit: GunFitness, pooled: bool, stat: RankStat,
fieldRate, shrinkK: float): float =
## Ranking statistic over the gun's window. All are monotone-ish in the mean,
## but differ in how they trade mean against sample size/noise. `rsMean` is the
## shipped statistic.
let (h, n) = gunCounts(fit, pooled)
if n == 0: return 0.0
let p = h.float / n.float
case stat
of rsMean:
p
of rsWilson:
let z = 1.0
let z2 = z * z
let denom = 1.0 + z2 / n.float
let centre = p + z2 / (2.0 * n.float)
let margin = z * sqrt((p * (1.0 - p) + z2 / (4.0 * n.float)) / n.float)
max(0.0, (centre - margin) / denom)
of rsUCB:
p + 0.5 * sqrt(p * (1.0 - p) / n.float)
of rsThompson:
let se = sqrt(max(1e-9, p * (1.0 - p) / n.float))
clamp(p + gauss(0.0, se), 0.0, 1.0)
of rsShrunk:
(h.float + shrinkK * fieldRate) / (n.float + shrinkK)
proc tableBestRate(t: VirtualTracker, pooled: bool): float =
## Best eligible gun rate across every target's fitness (no merge/allocation).
## Only guns with >= MinObsBeforeCompete samples count: under-sampled bins
## produce 100%/-looking spikes that would inflate the floor reference and
## force HeadOn for the whole window. Zero when nothing is warmed up yet.
result = 0.0
for _, perEnemy in t.fitness:
for gunId in 0..<perEnemy.len:
if not gunEligible(perEnemy[gunId], true): continue
result = max(result, gunRate(perEnemy[gunId], pooled))
proc noteBestRate*(t: var VirtualTracker) =
## Record this tick's field-best rate and refresh the RELATIVE-mode floor
## reference: the max bestRate over the last SelectorWindow ticks. Call once
## per selection tick; the live loop does this from `tickBullets`.
let best = tableBestRate(t, pooled = true)
t.rateHist[t.rateHistHead] = best
t.rateHistHead = (t.rateHistHead + 1) mod SelectorWindow
if t.rateHistCount < SelectorWindow: inc t.rateHistCount
var peak = 0.0
for i in 0..<t.rateHistCount:
if t.rateHist[i] > peak: peak = t.rateHist[i]
t.peakRateRef = peak
proc tickBullets*(t: var VirtualTracker, state: WorldState,
enemies: Table[int, tuple[x, y: float, lastSeenTick: int, alive: bool]],
onResolved: proc(gunId: GunId, binIdx: int, e: FeedbackEvent),
tieBreak: TieBreakMode = ActiveTieBreak) =
## Advance all active bullets one tick.
##
## The `bmPoint` branch (default) is unchanged: resolve when the bullet
## reaches the fire-time aim distance and score the single point it lands on.
##
## The `bmPath` branch flies the bullet along its ray until it leaves the
## arena and tests each tick's swept segment against the target's radius. It
## records exactly one outcome per bullet (at the wall), so every resolved
## bullet contributes exactly one fitness sample. A bullet that goes dead or
## stale is discarded without scoring, mirroring the point metric at
## resolution time.
for i in 0..<MaxBullets:
var b = addr t.bullets[i]
if not b.active: continue
b.travelDist += b.bulletSpeed
case t.metric
of bmPoint:
if b.travelDist < b.fireDist: continue
# Resolved: look up the correct enemy position
var ex, ey: float
if b.targetId in enemies:
let e = enemies[b.targetId]
if not e.alive or (state.tick - e.lastSeenTick) > StaleTicks:
b.active = false
continue
ex = e.x; ey = e.y
else:
# No data for this target — fall back to selected enemy in state
ex = state.enemyX; ey = state.enemyY
let dx = b.aimX - b.fireX
let dy = b.aimY - b.fireY
let dist = hypot(dx, dy)
let (bx, by) =
if dist < 1e-6: (b.aimX, b.aimY)
else: (b.fireX + dx / dist * b.travelDist,
b.fireY + dy / dist * b.travelDist)
let missDist = hypot(bx - ex, by - ey)
let hit = missDist < BotRadius
if b.targetId in t.fitness:
t.fitness[b.targetId][b.gunId].bins[b.powerBin].record(hit)
let fe = FeedbackEvent(
prediction: GunPrediction(x: b.aimX, y: b.aimY, confidence: b.confidence),
actualX: ex,
actualY: ey,
bulletPower: PowerBins[b.powerBin],
fireTick: b.fireTick,
powerBin: b.powerBin,
missDistance: missDist,
hit: hit,
confidence: b.confidence,
)
onResolved(b.gunId, b.powerBin, fe)
b.active = false
of bmPath:
# Enemy pose for this tick. A dead/stale target abandons the bullet
# without scoring, exactly as the point metric does at resolution time.
var ex, ey: float
if b.targetId in enemies:
let e = enemies[b.targetId]
if not e.alive or (state.tick - e.lastSeenTick) > StaleTicks:
b.active = false
continue
ex = e.x; ey = e.y
else:
ex = state.enemyX; ey = state.enemyY
let dx = b.aimX - b.fireX
let dy = b.aimY - b.fireY
let dist = hypot(dx, dy)
var ux, uy: float
if dist < 1e-6: ux = 0.0; uy = 0.0
else: ux = dx / dist; uy = dy / dist
let prevD = max(0.0, b.travelDist - b.bulletSpeed)
let ax = b.fireX + ux * prevD
let ay = b.fireY + uy * prevD
let bx = b.fireX + ux * b.travelDist
let by = b.fireY + uy * b.travelDist
let segMiss = distPointToSegment(ex, ey, ax, ay, bx, by)
# Arrival-accuracy probe (tie-break only; skipped entirely when off). The
# tick the bullet first reaches its fire-time aim distance is exactly the
# tick `bmPoint` would resolve on, so this records the SAME outcome the
# point model would have — just into a parallel window, leaving the path
# score (and therefore the shipped ranking) untouched.
if tieBreak != tbOff and not b.pointScored and b.travelDist >= b.fireDist:
b.pointScored = true
let (px, py) =
if dist < 1e-6: (b.aimX, b.aimY)
else: (b.fireX + ux * b.fireDist, b.fireY + uy * b.fireDist)
let pMiss = hypot(px - ex, py - ey)
if b.targetId in t.fitness:
t.fitness[b.targetId][b.gunId].pointBins[b.powerBin].record(pMiss < BotRadius)
if not b.hitSeen:
if segMiss < BotRadius:
# First physical contact — freeze it so a later closer approach
# cannot overwrite the contact position the guns learn from.
b.hitSeen = true
b.bestMissDist = segMiss
b.bestMissX = ex
b.bestMissY = ey
elif segMiss < b.bestMissDist:
b.bestMissDist = segMiss
b.bestMissX = ex
b.bestMissY = ey
# Despawn only at a wall (a degenerate zero-length ray also ends here).
let outside =
dist < 1e-6 or
bx < 0.0 or bx > state.arenaWidth or
by < 0.0 or by > state.arenaHeight
if outside:
let missDist = if b.bestMissDist == Inf: segMiss else: b.bestMissDist
let rx = if b.bestMissDist == Inf: ex else: b.bestMissX
let ry = if b.bestMissDist == Inf: ey else: b.bestMissY
if b.targetId in t.fitness:
t.fitness[b.targetId][b.gunId].bins[b.powerBin].record(b.hitSeen)
let fe = FeedbackEvent(
prediction: GunPrediction(x: b.aimX, y: b.aimY, confidence: b.confidence),
actualX: rx,
actualY: ry,
bulletPower: PowerBins[b.powerBin],
fireTick: b.fireTick,
powerBin: b.powerBin,
missDistance: missDist,
hit: b.hitSeen,
confidence: b.confidence,
)
onResolved(b.gunId, b.powerBin, fe)
b.active = false
if ActiveSelectorMode == smRelative:
noteBestRate(t)
proc fitnessFor*(t: VirtualTracker, targetId: int): seq[GunFitness] =
## Returns fitness seq for targetId, or merges all enemies as fallback.
##
## The fallback is a RECENCY-WEIGHTED AGGREGATE over the last WindowSize
## samples, NOT a pooled rate: each per-enemy window is replayed into one fresh
## window, so once the total exceeds WindowSize the earliest samples are
## overwritten by later ones. Enemies are visited in ascending target-id order
## so the result is identical on every run (std/tables iteration order is hash
## order and therefore nondeterministic).
## ponytail: merge is O(enemies*guns*bins*WindowSize), fine for small counts
if targetId >= 0 and targetId in t.fitness:
return t.fitness[targetId]
# Aggregate across all enemies, deterministically ordered.
result = newSeq[GunFitness](t.numGuns)
var enemyIds: seq[int]
for id in t.fitness.keys: enemyIds.add id
enemyIds.sort()
for id in enemyIds:
let perEnemy = t.fitness[id]
for gunId in 0..<t.numGuns:
for binIdx in 0..<len(PowerBins):
let src = perEnemy[gunId].bins[binIdx]
for k in 0..<min(src.count, WindowSize):
result[gunId].bins[binIdx].record(src.hits[k])
# The parallel point window (tie-break) must be merged by the same rule,
# or the aggregate fallback would silently lose the arrival score.
let psrc = perEnemy[gunId].pointBins[binIdx]
for k in 0..<min(psrc.count, WindowSize):
result[gunId].pointBins[binIdx].record(psrc.hits[k])
proc bestPower*(t: VirtualTracker, gunId: GunId, targetId: int = -1): (int, float) =
## Returns (binIdx, power). Prefers the HIGHEST power bin whose virtual hit
## rate is acceptable, where "acceptable" is measured RELATIVE to the same
## gun's best bin (`rate >= PowerBarFrac * bestBinRate`, dimensionless) — not
## against the legacy absolute `MinHitRate`. On the live path-metric scale a
## gun's rates sit around 3-40%, so the absolute 40% bar never fired once every
## bin had data and bestPower silently collapsed to power 1.0; the relative bar
## discriminates between bins at any scale.
##
## An EMPTY bin is still handed out (highest power first) so every bin keeps
## getting sampled, and a fully cold gun (no data anywhere) returns the lowest
## power bin. Uses per-enemy fitness when targetId >= 0 and data exists; else
## the deterministic aggregate.
let fit = t.fitnessFor(targetId)
result = (0, PowerBins[0])
var anyObs = false
var bestRate = 0.0
for binIdx in 0..<len(PowerBins):
if fit[gunId].bins[binIdx].count > 0:
anyObs = true
bestRate = max(bestRate, fit[gunId].bins[binIdx].hitRate())
if not anyObs:
return (0, PowerBins[0])
let bar = PowerBarFrac * bestRate
for binIdx in countdown(len(PowerBins) - 1, 0):
let fw = fit[gunId].bins[binIdx]
if fw.count == 0 or fw.hitRate() >= bar:
return (binIdx, PowerBins[binIdx])
# ── energy-aware power policy (TR_POWER_*) ───────────────────────────────────
#
# `bestPower` answers "which power bin does this gun's own virtual data prefer?"
# and is deliberately left untouched. The policy below CAPS that preference using
# only cheap, always-available state — range, our own energy, the ENEMY's energy,
# and the gun's own rate. It never RAISES power, so the shipped behaviour is
# exactly the `cap = 3.0` case, which is also the control arm
# (`TR_POWER_POLICY=0`). Every rule is an ADDITIONAL cap: the applied cap is the
# minimum of all of them.
#
# Two rules are the energy-economy improvement:
#
# 1. ENERGY SLOPE (replaces the old `TR_POWER_LOW_ENERGY` CLIFF at 50): cap our
# power as a LINEAR function of OUR energy — `TR_POWER_ENERGY_MAX` at/above
# `TR_POWER_ENERGY_HI`, `TR_POWER_ENERGY_MIN` at/below `TR_POWER_ENERGY_LO`,
# linear in between. A lower power is a FASTER bullet (speed = 20-3p, so less
# lead error -> higher hit chance), fires more often (interval 10+2p) and
# drains energy more slowly (cost p/shot). Energy math: E[dE] = p(3P-1), so
# the break-even hit probability is 1/3 INDEPENDENT of power; our measured
# hit rates are 5-27%, far below 1/3, so every point of power above the
# minimum costs more than it returns. Below HI the old code was a hard step.
#
# 2. FINISHING SLOPE (`TR_POWER_FINISH_KILL`): when the ENEMY is low, cap power
# at the SMALLEST bullet that still removes its remaining energy. Server
# 1.3.1 caps the `bulletDamage` score at the energy ACTUALLY REMOVED, so
# overkill is WASTED damage AND ~6x the energy for ZERO extra score. Damage
# is `4p` for p<=1 and `6p-2` for p>1, so the minimum power that covers `E`
# is `E/4` (clamped to >=0.1) for E<=4, `(E+2)/6` for 4<E<=16, and no cap
# above 16 (even p=3.0 removes only 16).
#
# Measured basis (real shots vs DrussGT, 8-16 runs): hit rate 21.6% at 0-100px,
# 27.1% at 100-200, then 19.3% at 200-300, 10.9% at 300-400, 6.8% at 400-600 and
# 5.4% at 600-800. Damage is 4p (p<=1) / 6p-2 (p>1).
const
PowerPolicyEnvVar* = "TR_POWER_POLICY" ## 0 = control arm (uncapped)
PowerFarDistEnvVar* = "TR_POWER_FAR_DIST" ## px; beyond this = bad-chances zone
PowerFarCapEnvVar* = "TR_POWER_FAR_CAP" ## cap for far
PowerMidCapEnvVar* = "TR_POWER_MID_CAP" ## cap when close+healthy but not above avg
PowerRefEnvVar* = "TR_POWER_REF" ## 0 = gun's own mean; >0 = fixed P_ref
PowerEnergyHiEnvVar* = "TR_POWER_ENERGY_HI" ## self energy at/above which = no slope cap
PowerEnergyLoEnvVar* = "TR_POWER_ENERGY_LO" ## self energy at/below which = ENERGY_MIN
PowerEnergyMinEnvVar* = "TR_POWER_ENERGY_MIN" ## the cap at/below ENERGY_LO
PowerEnergyMaxEnvVar* = "TR_POWER_ENERGY_MAX" ## the cap at/above ENERGY_HI (3.0 = uncapped)
PowerFinishKillEnvVar* = "TR_POWER_FINISH_KILL" ## 1 = cap to the smallest killing bullet
let PowerPolicyEnabled* = envBool(PowerPolicyEnvVar, true)
let PowerFarDist* = envFloat(PowerFarDistEnvVar, 200.0)
let PowerFarCap* = envFloat(PowerFarCapEnvVar, 1.0)
let PowerMidCap* = envFloat(PowerMidCapEnvVar, 2.0)
let PowerRefFixed* = envFloat(PowerRefEnvVar, 0.0)
let PowerEnergyHi* = envFloat(PowerEnergyHiEnvVar, 80.0)
let PowerEnergyLo* = envFloat(PowerEnergyLoEnvVar, 20.0)
let PowerEnergyMin* = envFloat(PowerEnergyMinEnvVar, 0.5)
let PowerEnergyMax* = envFloat(PowerEnergyMaxEnvVar, 3.0)
let PowerFinishKill* = envBool(PowerFinishKillEnvVar, true)
type
PowerReason* = enum
prFull ## no cap binds -> the gun's own preference
prFar ## beyond TR_POWER_FAR_DIST -> bad-chances zone
prEnergySlope ## OUR energy below TR_POWER_ENERGY_HI -> linear conserve cap
prFinishKill ## ENEMY energy low -> smallest bullet that still finishes it
prBelowAvg ## chances not above the gun's own average -> no power 3.0
prRam ## ramming: exempt (at contact P->1, so 3.0 is correct)
PowerCap* = object
power*: float ## the power to fire (<= the gun's preference)
cap*: float ## the cap applied (3.0 = uncapped)
reason*: PowerReason ## why
proc powerReasonName*(r: PowerReason): string =
case r
of prFull: "full"
of prFar: "far"
of prEnergySlope: "energySlope"
of prFinishKill: "finishKill"
of prBelowAvg: "belowAvg"
of prRam: "ram"
proc bulletDamageAtPower*(power: float): float =
## Energy the server removes for a bullet of `power`: `4p` for `p <= 1` and
## `6p - 2` for `p > 1` (and 0 for `p <= 0`), mirroring the server's
## `rules/math.kt calcBulletDamage`. Kept local so `gun_harness` has no
## dependency on the movement modules (which define the same `bulletDamage`).
if power <= 0.0: return 0.0
var p = power
if p < 0.1: p = 0.1
elif p > 3.0: p = 3.0
result = 4.0 * p
if p > 1.0: result += 2.0 * (p - 1.0)
proc powerToKill*(enemyEnergy: float): float =
## The SMALLEST power whose bullet damage covers `enemyEnergy`, i.e. the
## inverse of `bulletDamageAtPower`:
## E <= 4 -> p = E/4 (clamped to >= 0.1)
## E > 4 -> p = (E+2)/6
## E > 16 -> 3.0 (no cap: even p=3.0 removes only 16, so nothing smaller
## helps and the normal rules decide)
## Exactly covers E in the first two branches (`4p = E` and `6p-2 = E`); the
## clamp makes it cover E <= 0.4 too. PURE, exposed for testing.
if enemyEnergy > 16.0: return 3.0
if enemyEnergy <= 4.0: return max(0.1, enemyEnergy / 4.0)
(enemyEnergy + 2.0) / 6.0
proc energySlopeCap*(selfEnergy: float,
hi = PowerEnergyHi,
lo = PowerEnergyLo,
capMin = PowerEnergyMin,
capMax = PowerEnergyMax): float =
## Linear power cap vs OUR energy: `capMax` at/above `hi`, `capMin` at/below
## `lo`, linear in POWER in between, clamped to `[capMin, capMax]`. The
## shipped defaults (3.0 at 80, 0.5 at 20) mean "no cap above 80, half-power at
## or below 20". `capMax` is the value at/above HI; at its default 3.0 that is
## exactly "no cap from this rule". PURE, exposed for testing.
if selfEnergy >= hi: return capMax
if selfEnergy <= lo: return capMin
let t = (selfEnergy - lo) / (hi - lo)
clamp(capMin + t * (capMax - capMin), capMin, capMax)
proc binIndexForPower*(power: float): int =
## Index of `power` in `PowerBins`; if it is not an exact bin value, the
## highest bin whose power does not exceed it (0 if none). Keeps the returned
## bin index consistent with a capped power.
result = 0
for i in 0..<len(PowerBins):
if abs(PowerBins[i] - power) < 1e-9: return i
if PowerBins[i] <= power: result = i
proc applyPowerPolicy*(preferredPower, dist, selfEnergy, pEst, pRef: float,
ramming: bool,
enemyEnergy = 100.0,
enabled = PowerPolicyEnabled,
farDist = PowerFarDist,
energyHi = PowerEnergyHi,
energyLo = PowerEnergyLo,
energyMin = PowerEnergyMin,
energyMax = PowerEnergyMax,
finishKill = PowerFinishKill,
farCap = PowerFarCap,
midCap = PowerMidCap): PowerCap =
## PURE cap core — no tracker, no battle. `power = min(preferredPower, cap)`,
## so the result can only ever LOWER the gun's own preference. Each rule is an
## additional cap; the applied cap is the MINIMUM of all of them, and `reason`
## names the rule that set it (ties resolved by the order below, which is also
## the precedence order): ram (exempt) > far > energy slope > below average >
## finishing. `prFull` means nothing capped.
##
## `pEst` is the gun's rate for the bin it chose (or its aggregate when that
## bin is empty); `pRef` is the gun's aggregate mean (or the fixed
## `TR_POWER_REF`). A cold gun has no data, so `pEst <= pRef` is vacuously
## true and it gets the mid cap — deliberately conservative until it has
## evidence its chances are above average.
##
## `enemyEnergy` drives the finishing rule; its default (100) means "no
## finishing cap", so every pre-existing caller is unchanged. The finishing
## rule is skipped for `enemyEnergy <= 0` (a dead/unknown target), so a zero
## energy reading cannot collapse power to 0.1.
if ramming:
return PowerCap(power: preferredPower, cap: 3.0, reason: prRam)
if not enabled:
return PowerCap(power: preferredPower, cap: 3.0, reason: prFull)
var cap = 3.0
var reason = prFull
template noteCap(c: float, r: PowerReason) =
## Adopt `c` as the cap only when it is STRICTLY smaller, so ties keep the
## higher-precedence rule's reason (the order of the calls below).
if c < cap:
cap = c
reason = r
if dist > farDist: noteCap(farCap, prFar)
noteCap(energySlopeCap(selfEnergy, energyHi, energyLo, energyMin, energyMax),
prEnergySlope)
if pEst <= pRef: noteCap(midCap, prBelowAvg)
if finishKill and enemyEnergy > 0.0:
noteCap(powerToKill(enemyEnergy), prFinishKill)
PowerCap(power: min(preferredPower, cap), cap: cap, reason: reason)
proc chooseFromFit*(fit: seq[GunFitness], diag: ptr SelectorDiag = nil,
mode: SelectorMode = smAbsolute,
referenceRate = -1.0,
incumbent: GunId = -1,
switchMargin = 0.0,
rackMode: RackMode = rm1v1,
membership: openArray[RackMembership] = [],
tieBreak: TieBreakMode = ActiveTieBreak,
pointTieMargin: float = ActivePointTie): GunId =
## Core gun ranking over an already-resolved fitness seq. Split out from
## `bestGun` so the offline range can rank without copying a VirtualTracker,
## and so callers can request `diag` for the selection internals.
##
## `rackMode`/`membership` restrict the candidate set to the guns whose rack
## membership admits the mode. An empty `membership` (the default) admits every
## gun, so existing callers and the offline path are unchanged; an empty
## FILTERED set also falls back to the full rack.
##
## Guns with fewer than MinObsBeforeCompete observations are skipped unless
## every gun is below threshold (then fall back to best of all).
##
## `mode` chooses the threshold model:
## smAbsolute — legacy fixed TieMargin / MinHitRateFloor.
## smRelative — tie band = bestRate*RelTieMargin; floor = FloorPeakFrac
## * `referenceRate` (the recent field-best rate). Pooled over
## power bins, since one lucky bin is a poor ranker.
## `referenceRate` <= 0 disables the RELATIVE floor (no history yet).
##
## `incumbent` (>= 0) enables switch hysteresis: the incumbent is retained
## unless the best eligible gun beats it by the RELATIVE `switchMargin`
## (score > incumbentScore * (1 + margin)). Defaults keep the pure-ranking
## behaviour the offline analyzer and the `bestGun` tests rely on.
##
## Ties (within the band) are broken randomly to avoid index-0 bias. The draw
## runs ONLY when a switch is actually permitted — a tick that retains the
## incumbent returns before `rand`, so the tie-break no longer re-decides
## every tick.
##
## `tieBreak` (see `TieBreakMode`) optionally narrows the tied set by the
## PARALLEL arrival-accuracy window (`GunFitness.pointBins`, filled by
## `tickBullets` only while the tie-break is active) before that random draw.
## The default is the shipped `off`, and a fitness seq with no point data
## leaves the band untouched, so every existing caller is byte-identical.
let pooled = if mode == smRelative: ActivePooled else: false
let admitted = admittedGuns(fit.len, rackMode, membership)
if admitted.len == 0: return 0
var anyQualifies = false
for gunId in admitted:
if gunEligible(fit[gunId], true):
anyQualifies = true
break
let requireMin = anyQualifies
if diag != nil: diag[].anyQualifies = requireMin
# Mean rate per eligible gun: used both for the floor reference and (for
# rsShrunk) as the field mean the estimates are pulled toward.
var bestRate = 0.0
var fieldSum = 0.0
var fieldN = 0
for gunId in admitted:
if requireMin and not gunEligible(fit[gunId], true): continue
let (h, n) = gunCounts(fit[gunId], pooled)
if n == 0: continue
let r = h.float / n.float
bestRate = max(bestRate, r)
fieldSum += r
inc fieldN
let fieldRate = if fieldN > 0: fieldSum / fieldN.float else: 0.0
if diag != nil: diag[].bestRate = bestRate
let floorRate =
if mode == smAbsolute: MinHitRateFloor
elif referenceRate > 0.0: ActiveFloorFrac * referenceRate
else: 0.0
if diag != nil: diag[].floorRate = floorRate
# No hit at all, or the field collapsed below its own recent peak: the first
# admitted gun (HeadOn by default).
if bestRate <= 0.0 or bestRate < floorRate:
if diag != nil: diag[].floorFired = true
return admitted[0]
# Ranking scores (the active statistic) and the best of them.
var scores = newSeq[float](fit.len)
var bestScore = 0.0
for gunId in admitted:
if requireMin and not gunEligible(fit[gunId], true): continue
let s = rankScore(fit[gunId], pooled, ActiveRank, fieldRate, ActiveShrink)
scores[gunId] = s
bestScore = max(bestScore, s)
let tieBand =
if mode == smAbsolute: TieMargin
else: bestScore * ActiveRelTie
var tied: seq[GunId]
for gunId in admitted:
if requireMin and not gunEligible(fit[gunId], true): continue
if scores[gunId] >= bestScore - tieBand:
tied.add(gunId)
if diag != nil: diag[].tiedCount = tied.len
# ── switch hysteresis ──────────────────────────────────────────────────────
# A settled incumbent is displaced only by a challenger that clears the
# relative margin. This is what stops a merely-tied gun from churning the
# turret. `incumbent < 0` disables the rule (pure ranking).
if incumbent >= 0 and incumbent < fit.len and
rackAdmitted(incumbent, rackMode, membership):
var incumbentEligible = true
if requireMin and not gunEligible(fit[incumbent], true):
incumbentEligible = false
if incumbentEligible:
let incumbentScore = scores[incumbent]
let bar = incumbentScore * (1.0 + switchMargin)
if bestScore <= bar:
if diag != nil: diag[].incumbentKept = true
return incumbent
# Keep only genuine challengers (guns that clear the margin); if none do,
# the incumbent survives even when `bestScore` technically exceeds the bar
# (the top gun may sit inside the tie band but not be switched to).
var challengers: seq[GunId]
for g in tied:
if scores[g] > bar: challengers.add g
if challengers.len > 0:
tied = challengers
else:
if diag != nil: diag[].incumbentKept = true
return incumbent
if tied.len == 0: return 0
# ── arrival-accuracy tie-break (GUN_SELECTOR_TIEBREAK) ──────────────────────
# `tied` is the `path` band (optionally already narrowed to the challengers
# that cleared the switch margin). Re-order it by the parallel point window so
# the uniform draw below lands on guns whose bullets actually ARRIVE, not just
# on guns whose ray sweeps the target generously.
#
# tbPoint — narrow the band to guns within `pointTieMargin` of the best
# in-band point rate, then keep the uniform random draw.
# tbPointCommit — CONTROL: deterministically take the best point rate,
# removing the random draw. Included so the narrowed band can
# be compared against its own commitment control.
#
# A gun with NO point samples is KEPT: it cannot be judged, and dropping it
# would turn "cold" into "bad". If no tied gun has any point data the band is
# left untouched, so the tie-break is a strict no-op until the point window
# warms up.
if tieBreak != tbOff and tied.len > 1:
var bestPoint = 0.0
var anyPoint = false
for g in tied:
let (h, n) = pointCounts(fit[g], pooled = true)
if n == 0: continue
anyPoint = true
bestPoint = max(bestPoint, h.float / n.float)
if diag != nil: diag[].bestPointRate = bestPoint
if anyPoint and bestPoint > 0.0:
if tieBreak == tbPointCommit:
var bestGun = tied[0]
var bestP = -1.0
for g in tied:
let p = pointRate(fit[g], pooled = true)
if p > bestP:
bestP = p
bestGun = g
if diag != nil:
diag[].pointTieFired = true
diag[].pointTiedCount = 1
return bestGun
var kept: seq[GunId]
for g in tied:
let (h, n) = pointCounts(fit[g], pooled = true)
if n == 0 or h.float / n.float >= bestPoint * (1.0 - pointTieMargin):
kept.add g
if kept.len > 0 and kept.len < tied.len:
if diag != nil: diag[].pointTieFired = true
tied = kept
if diag != nil: diag[].pointTiedCount = kept.len
result = tied[rand(tied.len - 1)]
proc bestGun*(t: VirtualTracker, targetId: int = -1,
diag: ptr SelectorDiag = nil,
rackMode: RackMode = rm1v1,
membership: openArray[RackMembership] = []): GunId =
## Pick gun with highest hit rate across all power bins.
## Uses per-enemy fitness when targetId >= 0 and data exists; else aggregate.
## `diag`, when non-nil, receives the selection internals (bestRate, floor,
## tie count) exactly as used by the decision.
##
## This is the PURE, memoryless ranking primitive: it starts a fresh tie-break
## every call. The live bot uses `selectGun` (below), which wraps it with the
## dwell/switch-margin hysteresis. Keeping this pure is what lets the offline
## analyzer and the harness tests stay deterministic and side-effect free.
result = chooseFromFit(t.fitnessFor(targetId), diag,
mode = ActiveSelectorMode,
referenceRate = t.peakRateRef,
rackMode = rackMode, membership = membership)
proc selectSharedGun*(t: var VirtualTracker, share: openArray[float],
tick: int, rackMode: RackMode,
membership: openArray[RackMembership]): GunId =
## Forced-share allocation (`TR_RACK_SHARE`). Deterministic deficit
## round-robin over the named guns that the CURRENT rack admits; returns -1
## when no named gun is admitted (caller falls back to the normal ranking).
##
## The schedule advances once per allocation EPOCH, not once per tick: the
## incumbent is held for `ActiveDwellTicks`, so the turret converges on the
## allocated gun before the next allocation — the same anti-chatter reason the
## shipped hysteresis exists. Selected-tick counts therefore follow the weight
## ratio, which is what `gun_stats.jsonl` reports as `selected`.
var guns: seq[GunId]
var total = 0.0
for g in 0..<share.len:
if share[g] > 0.0 and rackAdmitted(g, rackMode, membership):
guns.add g
total += share[g]
if guns.len == 0 or total <= 0.0: return -1
# Hold the incumbent through its dwell window (it is already one of the
# admitted share guns). This is what keeps aim convergence; the deficit below
# is NOT advanced on a held tick. The `tick >= t.currentSince` guard is
# load-bearing: `bot.tick` RESETS to 0 each round while the tracker (and its
# `currentSince`) persists, so without it a negative elapsed would lock the
# round-1-end gun for every later round. The shipped ranking path keeps its
# own (single-gun-in-practice) dwell untouched.
if t.currentGun in guns and tick >= t.currentSince and
(tick - t.currentSince) < ActiveDwellTicks:
return t.currentGun
if t.shareDeficit.len != share.len:
t.shareDeficit.setLen(share.len)
for g in guns:
t.shareDeficit[g] += share[g] / total
var pick = guns[0]
var best = t.shareDeficit[pick]
for g in guns:
if t.shareDeficit[g] > best:
best = t.shareDeficit[g]
pick = g
t.shareDeficit[pick] -= 1.0
t.currentGun = pick
t.currentSince = tick
result = pick
proc selectGun*(t: var VirtualTracker, targetId: int = -1, tick = 0,
diag: ptr SelectorDiag = nil,
rackMode: RackMode = rm1v1,
membership: openArray[RackMembership] = [],
share: seq[float] = @[]): GunId =
## Stateful, sticky gun selection — the LIVE path (`selector.selectShot` calls
## this). Wraps the pure `chooseFromFit` ranking with two commitments:
##
## * minimum dwell — once selected, a gun is held for at least
## `GUN_SELECTOR_DWELL` ticks unless it becomes disqualified (below the
## eligibility sample gate) or the field collapses (existing floor path ->
## HeadOn);
## * switch margin — a challenger must beat the incumbent by
## `GUN_SELECTOR_MARGIN` (a fraction of the incumbent's score) before it can
## displace it, so a merely-tied gun does not.
##
## Setting BOTH knobs to 0 disables hysteresis entirely and reproduces the old
## per-tick behaviour, so the same binary can A/B against the baseline.
##
## Seam: the incumbent lives in the `VirtualTracker` because the tracker
## already owns all other selection state (`fitness`, the floor's
## `peakRateRef` history) and is threaded through the whole live loop; the bot
## does not need to know about it. `bestGun`/`chooseFromFit` stay pure for the
## offline tools.
##
## `share` (non-empty) is the `TR_RACK_SHARE` forced allocation: it bypasses
## `chooseFromFit` entirely (see `selectSharedGun`). Empty is the shipped
## ranking path, byte-for-byte.
if share.len > 0:
let shared = t.selectSharedGun(share, tick, rackMode, membership)
if shared >= 0: return shared
let fit = t.fitnessFor(targetId)
var local: SelectorDiag
let d = if diag != nil: diag else: addr local
# Both zero == no hysteresis: pass no incumbent, matching the pre-change
# (pure per-tick) selector for a clean A/B.
let useIncumbent = ActiveDwellTicks > 0 or ActiveSwitchMargin > 0.0
# A rack change (1v1 <-> melee) can retire the incumbent; do not let a gun the
# current rack excludes survive on hysteresis.
let incumbentAdmitted =
t.currentGun < 0 or rackAdmitted(t.currentGun, rackMode, membership)
let incumbent = if useIncumbent and incumbentAdmitted: t.currentGun else: -1
let challenger = chooseFromFit(fit, d, ActiveSelectorMode, t.peakRateRef,
incumbent = incumbent,
switchMargin = ActiveSwitchMargin,
rackMode = rackMode, membership = membership)
# The floor collapses the field to the first admitted gun regardless of dwell.
if d[].floorFired:
let floorGun = challenger
if t.currentGun != floorGun:
t.currentGun = floorGun
t.currentSince = tick
return floorGun
result = challenger
# Minimum dwell: hold the incumbent while it is still eligible AND admitted.
if useIncumbent and incumbentAdmitted and
t.currentGun >= 0 and t.currentGun < fit.len:
let eligible = (not d[].anyQualifies) or gunEligible(fit[t.currentGun], true)
if eligible and (tick - t.currentSince) < ActiveDwellTicks:
result = t.currentGun
if result != t.currentGun:
t.currentGun = result
t.currentSince = tick