dea4dcb574
The selection thresholds were calibrated for a rate scale that does not exist.
MEASURED on an exact offline replay of a fogged live WorldState vs DrussGT
(1397 selection ticks), the 0.10 absolute floor fires on 53.0% of point-metric
ticks and forces HeadOn, which has a REAL hit rate of 2.0-4.4% - worst or
near-worst of 13 guns. HeadOn's selection share: 69.1% (abs+point) -> 43.5%
(rel+point). My earlier claim that the floor fires ALWAYS is REFUTED - it is
53%, because bestRate is a max over gun x bin and a >=50-sample bin
occasionally clears 10%. The mechanism is confirmed; the literal statement was
not.
Scale-aware mode (GUN_SELECTOR_MODE, absolute|relative, default relative):
RelTieMargin = 0.20 dimensionless FRACTION of bestRate, replacing the
fixed 2pp band so the band scales with the metric
FloorPeakFrac = 0.25 the floor fires iff bestRate < 0.25 * peakRateRef,
SelectorWindow = 256 where peakRateRef is the field-best rate over the last
256 selection ticks - keeping the original 'don't trust
a collapsed field' purpose but only when the field is
bad RELATIVE TO ITS OWN RECENT BEST, and counting only
guns with >= MinObsBeforeCompete samples so cold-start
100% spikes cannot pin HeadOn
also pools the rate over power bins instead of taking the max over bins, so
one lucky bin no longer wins
absolute mode is preserved byte-for-byte for rollback.
A/B vs DrussGT, real server hit rate, 3 runs x 10 rounds per config, one frozen
binary:
absolute+point 3.66 / 2.45 / 5.01 pooled 3.76%
absolute+path 7.55 / 8.21 / 6.83 pooled 7.57%
relative+point 7.66 / 6.18 / 5.79 pooled 6.59%
relative+path 7.15 / 7.55 / 6.90 pooled 7.21%
absolute+point is SEPARATED from all three (p < 0.0001); the other three
OVERLAP each other (p = 0.18-0.64). So the METRIC is the dominant lever and
under path the two threshold models are statistically tied.
DEFAULT SET: metric = path, thresholds = relative. absolute+path was nominally
0.35pp higher but indistinguishable (p = 0.64); relative is the principled
scale-aware fix, is the only model that works under BOTH metrics, and prevents
the point-metric catastrophe if anyone switches back. Shipping absolute would
ship the accidental side-effect this work exists to remove.
STILL NOT SOLVED: the selector remains only a moderate ranker.
Spearman(virtual rank, real rank) is 0.52 for the winning config, 0.36 pooled
for path and 0.04 for point - and it is INCONSISTENT across run sets. The
metric switch won by de-selecting HeadOn, not by ranking guns better. That is
the next problem.
TASK B, report only: do NOT drive selection from raw real hit rates yet.
Only the selected gun fires, so unselected guns get near-zero real shots
(GuessFactor 20, Linear 24 vs HeadOn 733); noise is fatal (n=470 at p=10% gives
+/-2.8pp, most guns n<200 gives +/-5pp+ across a 3-15% spread); and real rate is
conditional on when the gun was selected. A blended signal with forced
exploration and shrinkage is defensible in principle but needs thousands of
shots per gun across many battles. Real rate is best used OFFLINE as the
evaluation metric - which is exactly what this A/B did.
RELATED BUG FLAGGED, not fixed: MinHitRate = 0.40 in bestPower is on the same
wrong scale - no bin ever clears 40%, so once every bin has data, power
selection falls back to bin 0 (power 1.0) late in a round.
Verified: 33/33 guard checks, 11/11 metric checks, tsetlin green, 12/12
offline==online acceptance under the shipped default, run_range rc=0 over 20
fixtures. Adds analyze_selector.nim to measure floor/tie/bestRate/HeadOn-share
per config on any fixture.
128 lines
5.9 KiB
Nim
128 lines
5.9 KiB
Nim
## Focused guard for the virtual-bullet METRIC switch
|
|
## (`GUN_VBULLET_METRIC`, see common_libs/gun_harness/virtual_bullets.nim).
|
|
##
|
|
## The switch selects how a virtual bullet is scored:
|
|
## point (default) — single point at the fire-time aim distance;
|
|
## path — swept-segment collision along the whole flight.
|
|
##
|
|
## This test pins the geometry that actually distinguishes the two models, and
|
|
## proves the switch is a real runtime override (not a compile-time constant) by
|
|
## driving BOTH models from one process via the explicit `initTracker` /
|
|
## `replayFixture` metric parameter.
|
|
##
|
|
## Run: nim c -r common_libs/tests/test_vbullet_metric.nim
|
|
|
|
import std/[math, tables, strformat]
|
|
import gun_harness/gun_interface
|
|
import gun_harness/virtual_bullets
|
|
import gun_harness/offline_range
|
|
import guns/head_on
|
|
import guns/linear
|
|
|
|
var failures = 0
|
|
proc check(name: string, ok: bool) =
|
|
if ok: echo "PASS: ", name
|
|
else: echo "FAIL: ", name; inc failures
|
|
|
|
# ── parsing / default ─────────────────────────────────────────────────────────
|
|
|
|
proc testParsing() =
|
|
check "metric parse: empty -> shipped default",
|
|
parseMetric("") == DefaultMetric
|
|
check "metric parse: 'point' -> bmPoint",
|
|
parseMetric("point") == bmPoint
|
|
check "metric parse: 'path' -> bmPath",
|
|
parseMetric("path") == bmPath
|
|
check "metric parse: case/space insensitive",
|
|
parseMetric(" PaTh ") == bmPath
|
|
check "metric parse: unknown -> shipped default (safe fallback, warns)",
|
|
parseMetric("definitely-not-a-metric") == DefaultMetric
|
|
|
|
# ── geometry that distinguishes the models ────────────────────────────────────
|
|
|
|
proc mkState(tick: int, ex, ey: float): WorldState =
|
|
WorldState(selfX: 100.0, selfY: 100.0,
|
|
enemyX: ex, enemyY: ey, enemySpeed: 0.0, enemyHeading: 0.0,
|
|
arenaWidth: 1000.0, arenaHeight: 1000.0, tick: tick)
|
|
|
|
proc runOneBullet(metric: BulletMetric, states: seq[WorldState],
|
|
targetId = 7): tuple[hits, shots: int] =
|
|
## Spawn one bullet per power bin at tick 0, aimed at the enemy's tick-0
|
|
## position, then tick the tracker over the rest of the stream.
|
|
var t = initTracker(1, metric)
|
|
let s0 = states[0]
|
|
let preds = [GunPrediction(x: s0.enemyX, y: s0.enemyY),
|
|
GunPrediction(x: s0.enemyX, y: s0.enemyY),
|
|
GunPrediction(x: s0.enemyX, y: s0.enemyY),
|
|
GunPrediction(x: s0.enemyX, y: s0.enemyY)]
|
|
t.spawnBullets(0, preds, s0, targetId)
|
|
for i in 1..<states.len:
|
|
let st = states[i]
|
|
var enemies: Table[int, tuple[x, y: float, lastSeenTick: int, alive: bool]]
|
|
enemies[targetId] = (x: st.enemyX, y: st.enemyY,
|
|
lastSeenTick: st.tick, alive: true)
|
|
t.tickBullets(st, enemies,
|
|
proc(gid: GunId, bin: int, e: FeedbackEvent) = discard)
|
|
let fit = t.fitnessFor(targetId)
|
|
for bin in 0..<len(PowerBins):
|
|
let n = min(fit[0].bins[bin].count, WindowSize)
|
|
result.shots += n
|
|
for k in 0..<n:
|
|
if fit[0].bins[bin].hits[k]: inc result.hits
|
|
|
|
proc testRecedingTarget() =
|
|
## Enemy recedes radially along the bullet's own ray, faster than the slowest
|
|
## bullet. The point model fires at the tick-0 aim distance (200 px) and the
|
|
## enemy has already moved past it -> miss. The path model keeps the bullet
|
|
## flying until the wall and it physically catches up -> hit.
|
|
var states: seq[WorldState]
|
|
for t in 0..<140:
|
|
states.add mkState(t, 300.0 + 4.0*t.float, 100.0)
|
|
let p = runOneBullet(bmPoint, states)
|
|
let q = runOneBullet(bmPath, states)
|
|
check "receding target: point model scores 0 hits (enemy left the aim point)",
|
|
p.hits == 0
|
|
check "receding target: path model catches the receding enemy", q.hits > 0
|
|
check "receding target: both models emit the same number of samples",
|
|
p.shots == q.shots and q.shots == len(PowerBins)
|
|
echo fmt" receding: point {p.hits}/{p.shots}, path {q.hits}/{q.shots}"
|
|
|
|
proc testPerpendicularTarget() =
|
|
## Enemy crosses the ray perpendicularly. The ray and the enemy's trajectory
|
|
## intersect at the enemy's tick-0 position, but the bullet is 200 px away at
|
|
## that instant. A physical swept-collision model must MISS; a naive
|
|
## path-intersection model would wrongly hit.
|
|
var states: seq[WorldState]
|
|
for t in 0..<140:
|
|
states.add mkState(t, 300.0, 100.0 + 8.0*t.float)
|
|
let q = runOneBullet(bmPath, states)
|
|
check "perpendicular target: path model misses (timing matters, not just path)",
|
|
q.hits == 0
|
|
echo fmt" perpendicular: path {q.hits}/{q.shots}"
|
|
|
|
# ── switch drives the offline replay ─────────────────────────────────────────
|
|
|
|
proc testReplayMetricOverride() =
|
|
let fx = synthesizeConstantVelocity(ticks = 140)
|
|
let p = replayFixture(fx, @[makeDriver("Linear", LinearGun())], metric = bmPoint)
|
|
let q = replayFixture(fx, @[makeDriver("Linear", LinearGun())], metric = bmPath)
|
|
check "replay: both metrics record samples on a fixture",
|
|
p[0].shots > 0 and q[0].shots > 0
|
|
check "replay: path model is deterministic run-to-run",
|
|
block:
|
|
let q2 = replayFixture(fx, @[makeDriver("Linear", LinearGun())], metric = bmPath)
|
|
q2[0].hits == q[0].hits and q2[0].shots == q[0].shots
|
|
echo fmt" constant-velocity Linear: point {p[0].hits}/{p[0].shots}, path {q[0].hits}/{q[0].shots}"
|
|
|
|
# ── driver ────────────────────────────────────────────────────────────────────
|
|
|
|
testParsing()
|
|
testRecedingTarget()
|
|
testPerpendicularTarget()
|
|
testReplayMetricOverride()
|
|
|
|
if failures > 0:
|
|
echo "\n", failures, " check(s) FAILED"
|
|
quit(1)
|
|
echo "\nAll virtual-bullet metric checks passed."
|