e40c8493a6
AUDIT (docs/offline_harness_trust.md, new): - Re-ran acceptance_offline_vs_online myself TWICE: 12/12 deterministic guns exact both times (264 ticks/enemyId=1, 244 ticks/enemyId=2), death boundary included. The offline range reproduces the live bot's own per-gun virtual telemetry exactly. - Re-verified the (fireTick, powerBin) wave-pairing fix: exact-key lookup, collisions counted not silently mislabelled; test_wave_pairing 17/17 PASS. - The offline score is the live TELEMETRY (last-100 virtual hit rate) but NOT the live BATTLE score (damage/round wins). Two-level answer, documented. - bmPoint scores up to one tick-step (~17px) PAST its documented aim distance, while the tie-break probe scores exactly the aim point. Real, low-impact, deliberately NOT fixed (point metric is non-default, measured negative, and the committed point baselines would silently change). - bmPoint/bmPath, perfect-info captures, conditional-on-selection live rates, and hit-rate-as-objective-for-movement all catalogued as non-apples comparisons. FIX (unambiguous, fail-before/pass-after): - common_libs/tests/range_guns.nim: buildAllGunDrivers defaulted to enableTmSelector=true, so run_range / analyze_selector / test_power_selection / measure_power_policy spawned gun 13 (TMSelect) - a gun the shipped bot NEVER spawns. The shared VirtualTracker ring is order-sensitive, so those 4 spawns/tick permuted the learning guns' resolution order (the exact confound4cd5618fixed for the acceptance test, left broken for every default caller). Default is now false (mirror the shipped rack). Impact on tr_drussgt_vs_modularbot: Tsetlin 18.8->18.5%, KNN 7.5->7.2%, TMSelect 15.2->0. - New guard common_libs/tests/test_range_rack_parity.nim (3 checks); proven to FAIL before and PASS after by stash-reverting the fix. CALIBRATION (offline prediction vs live outcome, 9 usable arms): - Direction agreement 3/9 = 33%. Split by domain: open-loop (single-tick prediction / metric / threshold) 3/3; closed-loop (adaptation / range / movement / selection) 0/6. Small, non-random, hand-assembled set - no correlation coefficient is claimed. - The four motivating "offline wins" re-attributed: ring mover was NEVER offline (it is a live server-side hit rate, mislabelled "offline" in env_reference.md:342 and commit7f6ccfb); TMHorizon window/NSTATES and the TM gun are the H3 classifier-accuracy harness (not hit rate); TFIL is the H2 open-loop movement replay, whose mechanism prediction was right and whose outcome prediction was wrong. - Open-loop hypothesis tested: TR_RACK_* knobs leave the offline range output BYTE-IDENTICAL (the replay never calls the selector), and the range has no driver for guns 14/15 (TMPATTERN/TMHORIZON). BUG vs LIMIT separated. VERDICT: trust the harness for single-tick prediction quality only; never for anything running through the closed loop. MEASURED vs INFERRED labelled. Green counts unchanged: test_gun_harness 39, test_vbullet_metric 11, test_power_selection 3, test_power_policy 58, test_adaptive_radar 41, test_tfil_ring_weights 24, test_ram_decision 40, test_rack_membership 48, test_selector_tiebreak 19, test_tm_pattern_registration 20, test_vbullet_admit_gate 12, test_tm_horizon 104, test_tm_diag 48, test_tm_automata_diag 55, test_tm_clause_shape 66, test_env_report 25, test_tfil_commit_env 30 (as-is). New: test_range_rack_parity 3.
100 lines
4.2 KiB
Nim
100 lines
4.2 KiB
Nim
## Shared helper: build the 13 ModularBot guns as type-erased offline range
|
|
## drivers, in ModularBot's gun-id order, with the same readiness gate the live
|
|
## loop uses (Tsetlin only spawns once its 10-frame window is full).
|
|
|
|
import std/random
|
|
import gun_harness/offline_range
|
|
import guns/head_on
|
|
import guns/linear
|
|
import guns/circular
|
|
import guns/tsetlin
|
|
import guns/guess_factor
|
|
import guns/pattern_matcher
|
|
import guns/wall_bounce
|
|
import guns/accel_predictor
|
|
import guns/stop_shot
|
|
import guns/displacement
|
|
import guns/averaged_lead
|
|
import guns/decay_gf
|
|
import guns/knn_gun
|
|
import guns/tm_selector
|
|
|
|
proc buildAllGunDrivers*(seed = -1, enableTmSelector = false): seq[GunDriver] =
|
|
## seed >= 0 re-seeds the global RNG after constructing the stochastic guns
|
|
## (Tsetlin and the TM selector both call randomize() in their constructors),
|
|
## so their learning is reproducible for offline runs.
|
|
##
|
|
## Order matches ModularBot's gun ids exactly (TMSelect appended at 13).
|
|
##
|
|
## `enableTmSelector` must MIRROR the live rack. The shipped ModularBot has
|
|
## `EnableTmSelector = false`, so the live loop never spawns gun-13 virtual
|
|
## bullets. The replay's shared VirtualTracker ring is ORDER-SENSITIVE: extra
|
|
## gun-13 spawns shift the ring head and permute the per-tick resolution ORDER
|
|
## of every other gun, which scrambles the `obs` insertion order of the
|
|
## learning guns (KNN, DecayGF) and shifts their predictions. That is exactly
|
|
## the defect the acceptance test was fixed for (commit 4cd5618): with gun 13
|
|
## spawning offline, the live and offline KNN traces diverged; with its ready
|
|
## gate closed they became byte-identical.
|
|
##
|
|
## The DEFAULT is therefore `false` (mirror the shipped rack). Only a caller
|
|
## that is deliberately measuring TMSelect as a gun should pass `true`; the
|
|
## default must never silently inject a gun the shipped bot does not spawn,
|
|
## because the corruption lands on the OTHER guns' numbers.
|
|
var tsetlin = initTsetlinGun()
|
|
var tmSelector = initTmSelectorGun()
|
|
if seed >= 0:
|
|
randomize(seed)
|
|
result = @[
|
|
makeDriver("HeadOn", HeadOnGun()),
|
|
makeDriver("Linear", LinearGun()),
|
|
makeDriver("Tsetlin", tsetlin),
|
|
makeDriver("Circular", CircularGun()),
|
|
makeDriver("GuessFactor", initGFGun()),
|
|
makeDriver("Pattern", PatternMatcherGun()),
|
|
makeDriver("WallBounce", initWallBounceGun()),
|
|
makeDriver("Accel", initAccelGun()),
|
|
makeDriver("StopShot", initStopShotGun()),
|
|
makeDriver("Displace", initDisplacementGun()),
|
|
makeDriver("AvgLead", initAveragedLeadGun()),
|
|
makeDriver("DecayGF", initDecayGFGun()),
|
|
makeDriver("KNN", initKNNGun()),
|
|
makeDriver("TMSelect", tmSelector),
|
|
]
|
|
if not enableTmSelector:
|
|
## Same effect as the live `if EnableTmSelector` gate: never spawn gun 13.
|
|
## Keep the slot so gun ids / report indices are unchanged.
|
|
result[13].readyCb = proc(): bool = false
|
|
|
|
proc makeTsetlinDriver*(seed = -1): tuple[driver: GunDriver, gun: ref TsetlinGun] =
|
|
## Same as makeDriver("Tsetlin", ...) but keeps a handle to the concrete gun,
|
|
## so a test can read its clause-sparsity / correction instrumentation after a
|
|
## replay. `makeDriver` heap-boxes a copy internally and drops the handle.
|
|
let g = new(TsetlinGun)
|
|
g[] = initTsetlinGun()
|
|
if seed >= 0:
|
|
randomize(seed)
|
|
result.gun = g
|
|
result.driver = GunDriver(
|
|
name: "Tsetlin",
|
|
predictCb: proc(state: WorldState, bulletSpeed: float): GunPrediction =
|
|
g[].predict(state, bulletSpeed),
|
|
resultCb: proc(e: FeedbackEvent) = g[].onResult(e),
|
|
readyCb: proc(): bool = g[].isWarmedUp(),
|
|
)
|
|
|
|
proc makeTmSelectorDriver*(seed = -1): tuple[driver: GunDriver, gun: ref TmSelectorGun] =
|
|
## Same as makeDriver("TMSelect", ...) but keeps a handle to the concrete gun
|
|
## so a test can inspect its votes / clause interpretability after a replay.
|
|
let g = new(TmSelectorGun)
|
|
g[] = initTmSelectorGun()
|
|
if seed >= 0:
|
|
randomize(seed)
|
|
result.gun = g
|
|
result.driver = GunDriver(
|
|
name: "TMSelect",
|
|
predictCb: proc(state: WorldState, bulletSpeed: float): GunPrediction =
|
|
g[].predict(state, bulletSpeed),
|
|
resultCb: proc(e: FeedbackEvent) = g[].onResult(e),
|
|
readyCb: proc(): bool = g[].isWarmedUp(),
|
|
)
|