wave pairing: 36-58% of GF/DecayGF/KNN learning samples were MISLABELLED

The audit inferred (from code) that GF/DecayGF/KNN pop the OLDEST wave on
resolution, while under bmPath bullets leave the arena in NON-FIFO order - so an
outcome could be attached to the wrong wave. It also noted that `starved=0` does
NOT rule this out. Both halves are now MEASURED.

MISPAIRING RATE (10 DrussGT fixtures, real VirtualTracker, 344k resolutions/gun):
  gun         bmPath mispair   label err      bmPoint mispair   label err
  GuessFactor     36.48%         19.39%           18.24%          7.62%
  DecayGF         36.85%         19.52%           20.57%          8.64%
  KNN             57.91%         27.63%           29.75%         11.58%
  (starved = 0 everywhere, exactly as the audit predicted)
So ~1 in 5 GF/DecayGF learning samples and ~1 in 4 KNN samples carried a WRONG
guess-factor bin. This is a material corruption of the learning signal.

FIX: the same fireTick-keyed ring scheme `tsetlin.nim`/`tm_selector.nim` already
use - `slot = (fireTick*4 + bin) mod 1024` (period 256 ticks, longer than the
~91-tick max flight), looked up by exact key. Public interfaces unchanged; added
`waveResolved`/`waveMispaired` integrity counters. AFTER: mispaired = 0 and
starved = 0, both metrics, all three guns.

EFFECT ON HIT RATE: SMALL AND NOT SIGNIFICANT. bmPath 4000 samples/gun:
  GuessFactor 23.20% -> 23.02% (-0.18pp, per-run sign-flip p=0.750)
  DecayGF     23.80% -> 24.25% (+0.45pp, p=0.625)
  KNN         18.27% -> 18.80% (+0.53pp, p=0.547)
bmPoint: +0.05 / +0.33 / -0.15pp, p = 1.00 / 0.50 / 0.50. Per-run ranges overlap
almost completely. A bullet-level z-test is anti-conservative (bullets within a
fixture share a trajectory) and its KNN p=1.9e-16 cannot be trusted given ~10
effective independent runs.
PLAIN READING: this is a CORRECTNESS fix, not a measurable hit-rate win. It
removes a 36-58% mislabelling of the learning signal; the point estimates move by
at most ~0.5pp, within run-to-run noise. Stated plainly rather than oversold.

A REGRESSION IT CAUGHT IN ITSELF (and this explains the SIGSEGV another job saw
and correctly attributed to a concurrent knn_gun.nim rewrite): the first
implementation put an inline `array[1024, KNNWave]` (~100KB) inside each gun,
which overflowed the default 8MB stack and made `test_power_selection` SIGSEGV.
Causation was proven by stashing only the three gun files (test passed), then
fixed by making the rings heap-backed `seq`. Verified: `test_power_selection`
3 PASS on the default stack, and zero inline `array[1024]` remain.

Guards: test_wave_pairing 17 (new, pure), test_gun_harness 39,
test_vbullet_metric 11, test_power_selection 3, test_adaptive_radar 41,
test_tfil_ring_weights 24, test_power_policy 26, test_ram_decision 28.
ModularBot compiles. Adds audit_wave_pairing.nim and compare_pairing.nim.
This commit is contained in:
2026-09-22 01:33:31 +02:00
parent 1ea72c7f14
commit 4657fe715e
6 changed files with 586 additions and 65 deletions
+146
View File
@@ -0,0 +1,146 @@
## Task 3 analysis: compare the FIFO (before) and fireTick-keyed (after) pairing
## for the three learned GF guns.
##
## Two analyses:
## 1. Per-fixture (per-run) paired comparison — the repo's convention. The 10
## DrussGT fixtures are the independent runs; the paired delta is
## after - before. An exact sign-flip permutation test (2^10 = 1024 sign
## patterns) gives the p-value, and the per-run ranges give the overlap.
## 2. Bullet-level two-sample permutation test on the raw hit booleans dumped
## by audit_wave_pairing.nim. ANTI-CONSERVATIVE: bullets within a fixture
## share a trajectory and are correlated, so treat this as an upper bound on
## significance, not the headline.
##
## Run: nim c -r common_libs/tests/compare_pairing.nim
import std/[math, strformat, random, os, strutils]
const
FixtureNames = ["drussgt_vs_corners", "drussgt_vs_crazy", "drussgt_vs_drussgt",
"drussgt_vs_ramfire", "drussgt_vs_spinbot",
"tr_drussgt_vs_corners", "tr_drussgt_vs_crazy",
"tr_drussgt_vs_modularbot", "tr_drussgt_vs_modularbot_shield",
"tr_drussgt_vs_spinbot"]
ShotsPerFixture = 400 # WindowSize(100) x 4 power bins
GunNames = ["GuessFactor", "DecayGF", "KNN"]
# Captured from `audit_wave_pairing.nim <tag> both` (fresh guns per fixture;
# hits out of 400). Deterministic guns -> reproducible.
BeforePath: array[3, array[10, int]] = [
[36, 139, 23, 202, 233, 155, 9, 47, 42, 42], # GuessFactor
[61, 108, 8, 204, 232, 118, 9, 44, 108, 60], # DecayGF
[16, 94, 33, 188, 176, 123, 12, 33, 19, 37], # KNN
]
AfterPath: array[3, array[10, int]] = [
[36, 139, 23, 202, 226, 155, 12, 44, 42, 42],
[61, 105, 19, 213, 226, 140, 9, 47, 108, 42],
[38, 94, 38, 193, 158, 129, 11, 29, 25, 37],
]
BeforePoint: array[3, array[10, int]] = [
[5, 33, 15, 59, 59, 36, 0, 26, 3, 21],
[1, 8, 2, 37, 52, 48, 0, 0, 33, 6],
[0, 19, 14, 53, 34, 39, 13, 15, 5, 25],
]
AfterPoint: array[3, array[10, int]] = [
[5, 33, 15, 59, 59, 36, 0, 28, 3, 21],
[1, 9, 2, 37, 52, 48, 0, 0, 45, 6],
[0, 20, 15, 48, 33, 38, 13, 14, 5, 25],
]
proc sum(a: array[10, int]): int =
for x in a: result += x
proc meanPct(a: array[10, int]): float = sum(a).float / 10.0 / ShotsPerFixture.float * 100.0
proc minPct(a: array[10, int]): float =
result = 1e9
for x in a: result = min(result, x.float / ShotsPerFixture.float * 100.0)
proc maxPct(a: array[10, int]): float =
result = -1e9
for x in a: result = max(result, x.float / ShotsPerFixture.float * 100.0)
proc signFlipP(before, after: array[10, int]): tuple[p, obsMeanPp: float, nPos, nNeg, nZero: int] =
## Exact sign-flip permutation test on the paired per-fixture deltas.
var deltas: array[10, float]
for i in 0..<10:
deltas[i] = (after[i] - before[i]).float / ShotsPerFixture.float * 100.0
if deltas[i] > 1e-9: inc result.nPos
elif deltas[i] < -1e-9: inc result.nNeg
else: inc result.nZero
result.obsMeanPp += deltas[i] / 10.0
let obs = abs(result.obsMeanPp)
var ge = 0
for mask in 0..<(1 shl 10):
var m = 0.0
for i in 0..<10:
let s = if ((mask shr i) and 1) == 1: -1.0 else: 1.0
m += s * deltas[i] / 10.0
if abs(m) >= obs - 1e-12: inc ge
result.p = ge.float / 1024.0
proc loadDump(path: string): seq[bool] =
if not fileExists(path):
return @[]
for line in lines(path):
let s = line.strip()
if s.len == 0: continue
result.add (s == "1")
proc zTest(a, b: seq[bool]): tuple[p, diffPp, z: float] =
## Two-proportion z-test (analytic; the permutation equivalent is exact but
## 344k-element shuffles are needlessly slow). ANTI-CONSERVATIVE because the
## bullets are correlated within a fixture.
if a.len == 0 or b.len == 0: return (1.0, 0.0, 0.0)
var ha, hb: int
for x in a: (if x: inc ha)
for x in b: (if x: inc hb)
let p1 = ha.float / a.len.float
let p2 = hb.float / b.len.float
result.diffPp = (p2 - p1) * 100.0
let p = (ha + hb).float / (a.len + b.len).float
let se = sqrt(max(1e-30, p * (1.0 - p) * (1.0/a.len.float + 1.0/b.len.float)))
result.z = (p2 - p1) / se
result.p = erfc(abs(result.z) / sqrt(2.0))
proc reportMetric(mname: string,
before, after: array[3, array[10, int]]) =
echo "══════════════════════════════════════════════════════════════════"
echo " METRIC = ", mname
echo "══════════════════════════════════════════════════════════════════"
for gi in 0..<2:
let b = before[gi]
let a = after[gi]
echo fmt"{GunNames[gi]}:"
echo fmt" before {sum(b):>4}/{ShotsPerFixture*10} = {meanPct(b):5.2f}% per-run {minPct(b):5.2f}..{maxPct(b):5.2f}%"
echo fmt" after {sum(a):>4}/{ShotsPerFixture*10} = {meanPct(a):5.2f}% per-run {minPct(a):5.2f}..{maxPct(a):5.2f}%"
let sf = signFlipP(b, a)
echo fmt" delta {meanPct(a)-meanPct(b):+5.2f}pp paired sign-flip permutation p={sf.p:.3f} (+{sf.nPos}/-{sf.nNeg}/0:{sf.nZero} of 10)"
let b = before[2]
let a = after[2]
echo fmt"{GunNames[2]}:"
echo fmt" before {sum(b):>4}/{ShotsPerFixture*10} = {meanPct(b):5.2f}% per-run {minPct(b):5.2f}..{maxPct(b):5.2f}%"
echo fmt" after {sum(a):>4}/{ShotsPerFixture*10} = {meanPct(a):5.2f}% per-run {minPct(a):5.2f}..{maxPct(a):5.2f}%"
let sf = signFlipP(b, a)
echo fmt" delta {meanPct(a)-meanPct(b):+5.2f}pp paired sign-flip permutation p={sf.p:.3f} (+{sf.nPos}/-{sf.nNeg}/0:{sf.nZero} of 10)"
echo ""
proc main() =
randomize(12345)
reportMetric("bmPath (shipped)", BeforePath, AfterPath)
reportMetric("bmPoint", BeforePoint, AfterPoint)
echo "══════════════════════════════════════════════════════════════════"
echo " bullet-level two-proportion z-test (bmPath dumps) — ANTI-CONSERVATIVE"
echo "══════════════════════════════════════════════════════════════════"
for gi in 0..<3:
let bf = loadDump(fmt"/tmp/wavepair_before_path_{GunNames[gi]}.txt")
let af = loadDump(fmt"/tmp/wavepair_after_path_{GunNames[gi]}.txt")
let r = zTest(bf, af)
var hb, ha: int
for x in bf: (if x: inc hb)
for x in af: (if x: inc ha)
echo fmt"{GunNames[gi]:<12} before {hb:>6}/{bf.len:<6} {hb.float/bf.len.float*100:5.2f}% after {ha:>6}/{af.len:<6} {ha.float/af.len.float*100:5.2f}% diff {r.diffPp:+5.2f}pp z={r.z:+5.2f} p={r.p:.2g}"
when isMainModule:
main()