Files
SirRoboGarage/common_libs/tests/test_vbullet_metric.nim
T
SirStone 3b5d70b7c3 feat(gun_harness): runtime metric switch + A/B proving the point metric mis-selects
Adds GUN_VBULLET_METRIC (point|path, default point = unchanged behaviour) so
the virtual-bullet hit model can be selected at runtime with no rebuild. Both
the live tracker and the offline replay read the same value, so the 12/12
offline==online acceptance holds under EITHER setting (verified for both).

A/B AGAINST THE LIVE BOSS, real server-side hit rate as ground truth, 5
battles x 12 rounds per metric on one frozen binary:
  point  4660 shots / 219 hits = 4.70%   (per-run 3.16-5.53)
  path   4834 shots / 359 hits = 7.43%   (per-run 6.55-8.24)
The distributions DO NOT OVERLAP: path's worst run beats point's best run.
+2.73pp, +58% relative, z = 5.56, p < 0.0001. Range distributions were
identical (~460-478 px), so this is not a range confound.

MECHANISM - and this is the important part. The gain is SELECTION, not better
gun learning. Under the point model every gun's virtual rate is compressed
into 0.6-4.4%, so HeadOn sits inside the 2pp tie margin and takes 72.6% of
selection ticks / 76.9% of shots - while HeadOn is 11th of 13 by REAL hit rate
(2.3%). The path model widens the band to 4.7-13.7% and ranks HeadOn 10th, so
its shot share falls to 35.9% and Pattern/Accel/WallBounce get picked instead.
Counterfactual: applying the point model's per-gun real rates to the path
model's shot mix yields 7.65%, i.e. essentially the whole observed gain.
So the selector, not the guns, is where the win lives.

PER-GUN REAL HIT RATE vs DrussGT (path mix, the answer to 'which guns are
worth keeping'): WallBounce 10.8, Pattern 10.5, Accel 10.0, Displace 9.3,
Circular 9.2, AvgLead 8.5, KNN 5.7, StopShot 5.2, GuessFactor 3.7,
Tsetlin 2.9. Per-gun N is small (hundreds of shots) so single-gun ordering is
indicative, not definitive.

TWO CAVEATS, recorded because they undercut a naive reading:
1. One adversary. DrussGT is a wave surfer and HeadOn is genuinely bad against
   surfers, so part of this may be matchup-specific.
2. The path model is NOT a better general ranker. Spearman(virtual rank, real
   rank) is 0.52 under point vs -0.04 under path. It wins by accidentally
   fixing HeadOn's mis-rank, not by ranking guns better. A more durable fix is
   to address the selection logic directly - which is the next job.

Also adds a focused guard test (test_vbullet_metric) covering parsing/default,
a receding-target point-miss/path-hit, a perpendicular-target path-miss, and
replay determinism.

Verified: 33 guard checks, 12/12 acceptance under both metrics, tsetlin tests
green, range 34.3% (point, unchanged) / 50.8% (path).
2026-09-21 03:58:27 +02:00

128 lines
5.8 KiB
Nim

## Focused guard for the virtual-bullet METRIC switch
## (`GUN_VBULLET_METRIC`, see common_libs/gun_harness/virtual_bullets.nim).
##
## The switch selects how a virtual bullet is scored:
## point (default) — single point at the fire-time aim distance;
## path — swept-segment collision along the whole flight.
##
## This test pins the geometry that actually distinguishes the two models, and
## proves the switch is a real runtime override (not a compile-time constant) by
## driving BOTH models from one process via the explicit `initTracker` /
## `replayFixture` metric parameter.
##
## Run: nim c -r common_libs/tests/test_vbullet_metric.nim
import std/[math, tables, strformat]
import gun_harness/gun_interface
import gun_harness/virtual_bullets
import gun_harness/offline_range
import guns/head_on
import guns/linear
var failures = 0
proc check(name: string, ok: bool) =
if ok: echo "PASS: ", name
else: echo "FAIL: ", name; inc failures
# ── parsing / default ─────────────────────────────────────────────────────────
proc testParsing() =
check "metric parse: empty -> point (default)",
parseMetric("") == bmPoint
check "metric parse: 'point' -> bmPoint",
parseMetric("point") == bmPoint
check "metric parse: 'path' -> bmPath",
parseMetric("path") == bmPath
check "metric parse: case/space insensitive",
parseMetric(" PaTh ") == bmPath
check "metric parse: unknown -> point (safe fallback, warns)",
parseMetric("definitely-not-a-metric") == bmPoint
# ── geometry that distinguishes the models ────────────────────────────────────
proc mkState(tick: int, ex, ey: float): WorldState =
WorldState(selfX: 100.0, selfY: 100.0,
enemyX: ex, enemyY: ey, enemySpeed: 0.0, enemyHeading: 0.0,
arenaWidth: 1000.0, arenaHeight: 1000.0, tick: tick)
proc runOneBullet(metric: BulletMetric, states: seq[WorldState],
targetId = 7): tuple[hits, shots: int] =
## Spawn one bullet per power bin at tick 0, aimed at the enemy's tick-0
## position, then tick the tracker over the rest of the stream.
var t = initTracker(1, metric)
let s0 = states[0]
let preds = [GunPrediction(x: s0.enemyX, y: s0.enemyY),
GunPrediction(x: s0.enemyX, y: s0.enemyY),
GunPrediction(x: s0.enemyX, y: s0.enemyY),
GunPrediction(x: s0.enemyX, y: s0.enemyY)]
t.spawnBullets(0, preds, s0, targetId)
for i in 1..<states.len:
let st = states[i]
var enemies: Table[int, tuple[x, y: float, lastSeenTick: int, alive: bool]]
enemies[targetId] = (x: st.enemyX, y: st.enemyY,
lastSeenTick: st.tick, alive: true)
t.tickBullets(st, enemies,
proc(gid: GunId, bin: int, e: FeedbackEvent) = discard)
let fit = t.fitnessFor(targetId)
for bin in 0..<len(PowerBins):
let n = min(fit[0].bins[bin].count, WindowSize)
result.shots += n
for k in 0..<n:
if fit[0].bins[bin].hits[k]: inc result.hits
proc testRecedingTarget() =
## Enemy recedes radially along the bullet's own ray, faster than the slowest
## bullet. The point model fires at the tick-0 aim distance (200 px) and the
## enemy has already moved past it -> miss. The path model keeps the bullet
## flying until the wall and it physically catches up -> hit.
var states: seq[WorldState]
for t in 0..<140:
states.add mkState(t, 300.0 + 4.0*t.float, 100.0)
let p = runOneBullet(bmPoint, states)
let q = runOneBullet(bmPath, states)
check "receding target: point model scores 0 hits (enemy left the aim point)",
p.hits == 0
check "receding target: path model catches the receding enemy", q.hits > 0
check "receding target: both models emit the same number of samples",
p.shots == q.shots and q.shots == len(PowerBins)
echo fmt" receding: point {p.hits}/{p.shots}, path {q.hits}/{q.shots}"
proc testPerpendicularTarget() =
## Enemy crosses the ray perpendicularly. The ray and the enemy's trajectory
## intersect at the enemy's tick-0 position, but the bullet is 200 px away at
## that instant. A physical swept-collision model must MISS; a naive
## path-intersection model would wrongly hit.
var states: seq[WorldState]
for t in 0..<140:
states.add mkState(t, 300.0, 100.0 + 8.0*t.float)
let q = runOneBullet(bmPath, states)
check "perpendicular target: path model misses (timing matters, not just path)",
q.hits == 0
echo fmt" perpendicular: path {q.hits}/{q.shots}"
# ── switch drives the offline replay ─────────────────────────────────────────
proc testReplayMetricOverride() =
let fx = synthesizeConstantVelocity(ticks = 140)
let p = replayFixture(fx, @[makeDriver("Linear", LinearGun())], metric = bmPoint)
let q = replayFixture(fx, @[makeDriver("Linear", LinearGun())], metric = bmPath)
check "replay: both metrics record samples on a fixture",
p[0].shots > 0 and q[0].shots > 0
check "replay: path model is deterministic run-to-run",
block:
let q2 = replayFixture(fx, @[makeDriver("Linear", LinearGun())], metric = bmPath)
q2[0].hits == q[0].hits and q2[0].shots == q[0].shots
echo fmt" constant-velocity Linear: point {p[0].hits}/{p[0].shots}, path {q[0].hits}/{q[0].shots}"
# ── driver ────────────────────────────────────────────────────────────────────
testParsing()
testRecedingTarget()
testPerpendicularTarget()
testReplayMetricOverride()
if failures > 0:
echo "\n", failures, " check(s) FAILED"
quit(1)
echo "\nAll virtual-bullet metric checks passed."