Files
SirRoboGarage/common_libs/tests/range_guns.nim
T
SirStone 4cd5618435 fix(test): repair the flaky offline==online acceptance; make the tie-break truly random
TASK 2 - THE FLAKY ACCEPTANCE TEST, root-caused. It was NOT a live/offline
boundary race as suspected. The replay spawned gun 13 (TMSelect) while live has
EnableTmSelector = false and never does. The shared VirtualTracker ring is
ORDER-SENSITIVE, so gun 13's extra 4 bullets/tick shift the ring head and
permute the per-tick RESOLUTION ORDER of every other gun. The learning guns
append observations in resolution order, so their predictions shifted and
produced small hit deltas that moved between runs.
Evidence: the first KNN divergence was at rtick=174 with the SAME resolution
set merely reordered (live ft133,138,139,142,148,150,151 vs offline
ft150,151,133,138,139,142,148); after closing gun 13's ready gate offline the
live and offline KNN traces became BYTE-IDENTICAL (diff empty, 904/904 lines).
Fix: mirror the live rack in the replay. No tick exclusion, no tolerance
loosening. Stability: 5/5 consecutive runs now report 12/12 exact, each with
enemyDied=true - the death boundary is included, not excluded. The proof is
now real rather than a lucky run.

TASK 1 - the tie-break was not random. randomize() was only reached
incidentally through initTsetlinGun(), so a rack without Tsetlin had a fixed
rand() stream and ties always resolved the same way across process restarts.
Added seedSelectorRng() after gun construction, honouring GUN_SELECTOR_SEED.
Evidence: unseeded, 6 separate processes gave different pick sequences; with
GUN_SELECTOR_SEED=42, 3 processes gave identical sequences.

TASK 3 - PRUNING DOES NOT HELP; keep the full rack. 15 PAIRED runs per variant
vs DrussGT, 8 rounds, identical seeds:
  baseline             3238 shots  6.18%  (events 6.16%)  200 dmg/run
  Tsetlin disabled     3522 shots  5.76%  (events 5.71%)  197 dmg/run
  Tsetlin+Displace     3478 shots  5.46%  (events 5.37%)  183 dmg/run
Paired permutation tests: -0.34pp p=0.57 and -0.70pp p=0.21. Per-run
distributions completely overlap (baseline range [2.68, 10.00]; 15/15 and 14/15
runs inside it). A Crazy control showed no separation either. So removing the
measured-worst real performers is neutral-to-slightly-negative, and with
sd ~1.8pp a definitive claim either way would need far more runs.

CORRECTION TO A CLAIM I MADE: the 'virtual metric is INVERTED' finding does NOT
reproduce. Job-24 measured Spearman -0.374; this job measures +0.335 over the
same 13 guns with a different but equally defensible aggregation. Two opposite
signs means the correlation is NOT robustly negative - it is WEAK AND
SIGN-UNSTABLE. The honest statement is that virtual hit rate is a poor ranker,
not an inverted one. The docs assert the inversion and need correcting.

Also adds per-process GUN_STATS_PATH/GUN_SHOTLOG_PATH so concurrent A/B runs do
not clobber each other, and an env-gated GUN_RACK_DISABLE for rack A/Bs. All
default behaviour is unchanged when the env vars are unset.
2026-09-21 06:56:44 +02:00

94 lines
3.8 KiB
Nim

## Shared helper: build the 13 ModularBot guns as type-erased offline range
## drivers, in ModularBot's gun-id order, with the same readiness gate the live
## loop uses (Tsetlin only spawns once its 10-frame window is full).
import std/random
import gun_harness/offline_range
import guns/head_on
import guns/linear
import guns/circular
import guns/tsetlin
import guns/guess_factor
import guns/pattern_matcher
import guns/wall_bounce
import guns/accel_predictor
import guns/stop_shot
import guns/displacement
import guns/averaged_lead
import guns/decay_gf
import guns/knn_gun
import guns/tm_selector
proc buildAllGunDrivers*(seed = -1, enableTmSelector = true): seq[GunDriver] =
## seed >= 0 re-seeds the global RNG after constructing the stochastic guns
## (Tsetlin and the TM selector both call randomize() in their constructors),
## so their learning is reproducible for offline runs.
##
## Order matches ModularBot's gun ids exactly (TMSelect appended at 13).
##
## `enableTmSelector` must MIRROR the live rack. The shipped ModularBot has
## `EnableTmSelector = false`, so the live loop never spawns gun-13 virtual
## bullets. The replay's shared VirtualTracker ring is order-sensitive: extra
## gun-13 spawns shift the ring head and permute the per-tick resolution ORDER
## of every other gun, which scrambles the `obs` insertion order of the
## learning guns (KNN, DecayGF) and makes the offline metric diverge from the
## live one. Callers that mirror the shipped bot must pass false (the
## acceptance test does); tests that specifically exercise TMSelect pass true.
var tsetlin = initTsetlinGun()
var tmSelector = initTmSelectorGun()
if seed >= 0:
randomize(seed)
result = @[
makeDriver("HeadOn", HeadOnGun()),
makeDriver("Linear", LinearGun()),
makeDriver("Tsetlin", tsetlin),
makeDriver("Circular", CircularGun()),
makeDriver("GuessFactor", initGFGun()),
makeDriver("Pattern", PatternMatcherGun()),
makeDriver("WallBounce", initWallBounceGun()),
makeDriver("Accel", initAccelGun()),
makeDriver("StopShot", initStopShotGun()),
makeDriver("Displace", initDisplacementGun()),
makeDriver("AvgLead", initAveragedLeadGun()),
makeDriver("DecayGF", initDecayGFGun()),
makeDriver("KNN", initKNNGun()),
makeDriver("TMSelect", tmSelector),
]
if not enableTmSelector:
## Same effect as the live `if EnableTmSelector` gate: never spawn gun 13.
## Keep the slot so gun ids / report indices are unchanged.
result[13].readyCb = proc(): bool = false
proc makeTsetlinDriver*(seed = -1): tuple[driver: GunDriver, gun: ref TsetlinGun] =
## Same as makeDriver("Tsetlin", ...) but keeps a handle to the concrete gun,
## so a test can read its clause-sparsity / correction instrumentation after a
## replay. `makeDriver` heap-boxes a copy internally and drops the handle.
let g = new(TsetlinGun)
g[] = initTsetlinGun()
if seed >= 0:
randomize(seed)
result.gun = g
result.driver = GunDriver(
name: "Tsetlin",
predictCb: proc(state: WorldState, bulletSpeed: float): GunPrediction =
g[].predict(state, bulletSpeed),
resultCb: proc(e: FeedbackEvent) = g[].onResult(e),
readyCb: proc(): bool = g[].isWarmedUp(),
)
proc makeTmSelectorDriver*(seed = -1): tuple[driver: GunDriver, gun: ref TmSelectorGun] =
## Same as makeDriver("TMSelect", ...) but keeps a handle to the concrete gun
## so a test can inspect its votes / clause interpretability after a replay.
let g = new(TmSelectorGun)
g[] = initTmSelectorGun()
if seed >= 0:
randomize(seed)
result.gun = g
result.driver = GunDriver(
name: "TMSelect",
predictCb: proc(state: WorldState, bulletSpeed: float): GunPrediction =
g[].predict(state, bulletSpeed),
resultCb: proc(e: FeedbackEvent) = g[].onResult(e),
readyCb: proc(): bool = g[].isWarmedUp(),
)