4657fe715e
The audit inferred (from code) that GF/DecayGF/KNN pop the OLDEST wave on resolution, while under bmPath bullets leave the arena in NON-FIFO order - so an outcome could be attached to the wrong wave. It also noted that `starved=0` does NOT rule this out. Both halves are now MEASURED. MISPAIRING RATE (10 DrussGT fixtures, real VirtualTracker, 344k resolutions/gun): gun bmPath mispair label err bmPoint mispair label err GuessFactor 36.48% 19.39% 18.24% 7.62% DecayGF 36.85% 19.52% 20.57% 8.64% KNN 57.91% 27.63% 29.75% 11.58% (starved = 0 everywhere, exactly as the audit predicted) So ~1 in 5 GF/DecayGF learning samples and ~1 in 4 KNN samples carried a WRONG guess-factor bin. This is a material corruption of the learning signal. FIX: the same fireTick-keyed ring scheme `tsetlin.nim`/`tm_selector.nim` already use - `slot = (fireTick*4 + bin) mod 1024` (period 256 ticks, longer than the ~91-tick max flight), looked up by exact key. Public interfaces unchanged; added `waveResolved`/`waveMispaired` integrity counters. AFTER: mispaired = 0 and starved = 0, both metrics, all three guns. EFFECT ON HIT RATE: SMALL AND NOT SIGNIFICANT. bmPath 4000 samples/gun: GuessFactor 23.20% -> 23.02% (-0.18pp, per-run sign-flip p=0.750) DecayGF 23.80% -> 24.25% (+0.45pp, p=0.625) KNN 18.27% -> 18.80% (+0.53pp, p=0.547) bmPoint: +0.05 / +0.33 / -0.15pp, p = 1.00 / 0.50 / 0.50. Per-run ranges overlap almost completely. A bullet-level z-test is anti-conservative (bullets within a fixture share a trajectory) and its KNN p=1.9e-16 cannot be trusted given ~10 effective independent runs. PLAIN READING: this is a CORRECTNESS fix, not a measurable hit-rate win. It removes a 36-58% mislabelling of the learning signal; the point estimates move by at most ~0.5pp, within run-to-run noise. Stated plainly rather than oversold. A REGRESSION IT CAUGHT IN ITSELF (and this explains the SIGSEGV another job saw and correctly attributed to a concurrent knn_gun.nim rewrite): the first implementation put an inline `array[1024, KNNWave]` (~100KB) inside each gun, which overflowed the default 8MB stack and made `test_power_selection` SIGSEGV. Causation was proven by stashing only the three gun files (test passed), then fixed by making the rings heap-backed `seq`. Verified: `test_power_selection` 3 PASS on the default stack, and zero inline `array[1024]` remain. Guards: test_wave_pairing 17 (new, pure), test_gun_harness 39, test_vbullet_metric 11, test_power_selection 3, test_adaptive_radar 41, test_tfil_ring_weights 24, test_power_policy 26, test_ram_decision 28. ModularBot compiles. Adds audit_wave_pairing.nim and compare_pairing.nim.
147 lines
6.8 KiB
Nim
147 lines
6.8 KiB
Nim
## Task 3 analysis: compare the FIFO (before) and fireTick-keyed (after) pairing
|
|
## for the three learned GF guns.
|
|
##
|
|
## Two analyses:
|
|
## 1. Per-fixture (per-run) paired comparison — the repo's convention. The 10
|
|
## DrussGT fixtures are the independent runs; the paired delta is
|
|
## after - before. An exact sign-flip permutation test (2^10 = 1024 sign
|
|
## patterns) gives the p-value, and the per-run ranges give the overlap.
|
|
## 2. Bullet-level two-sample permutation test on the raw hit booleans dumped
|
|
## by audit_wave_pairing.nim. ANTI-CONSERVATIVE: bullets within a fixture
|
|
## share a trajectory and are correlated, so treat this as an upper bound on
|
|
## significance, not the headline.
|
|
##
|
|
## Run: nim c -r common_libs/tests/compare_pairing.nim
|
|
|
|
import std/[math, strformat, random, os, strutils]
|
|
|
|
const
|
|
FixtureNames = ["drussgt_vs_corners", "drussgt_vs_crazy", "drussgt_vs_drussgt",
|
|
"drussgt_vs_ramfire", "drussgt_vs_spinbot",
|
|
"tr_drussgt_vs_corners", "tr_drussgt_vs_crazy",
|
|
"tr_drussgt_vs_modularbot", "tr_drussgt_vs_modularbot_shield",
|
|
"tr_drussgt_vs_spinbot"]
|
|
ShotsPerFixture = 400 # WindowSize(100) x 4 power bins
|
|
GunNames = ["GuessFactor", "DecayGF", "KNN"]
|
|
|
|
# Captured from `audit_wave_pairing.nim <tag> both` (fresh guns per fixture;
|
|
# hits out of 400). Deterministic guns -> reproducible.
|
|
BeforePath: array[3, array[10, int]] = [
|
|
[36, 139, 23, 202, 233, 155, 9, 47, 42, 42], # GuessFactor
|
|
[61, 108, 8, 204, 232, 118, 9, 44, 108, 60], # DecayGF
|
|
[16, 94, 33, 188, 176, 123, 12, 33, 19, 37], # KNN
|
|
]
|
|
AfterPath: array[3, array[10, int]] = [
|
|
[36, 139, 23, 202, 226, 155, 12, 44, 42, 42],
|
|
[61, 105, 19, 213, 226, 140, 9, 47, 108, 42],
|
|
[38, 94, 38, 193, 158, 129, 11, 29, 25, 37],
|
|
]
|
|
BeforePoint: array[3, array[10, int]] = [
|
|
[5, 33, 15, 59, 59, 36, 0, 26, 3, 21],
|
|
[1, 8, 2, 37, 52, 48, 0, 0, 33, 6],
|
|
[0, 19, 14, 53, 34, 39, 13, 15, 5, 25],
|
|
]
|
|
AfterPoint: array[3, array[10, int]] = [
|
|
[5, 33, 15, 59, 59, 36, 0, 28, 3, 21],
|
|
[1, 9, 2, 37, 52, 48, 0, 0, 45, 6],
|
|
[0, 20, 15, 48, 33, 38, 13, 14, 5, 25],
|
|
]
|
|
|
|
proc sum(a: array[10, int]): int =
|
|
for x in a: result += x
|
|
|
|
proc meanPct(a: array[10, int]): float = sum(a).float / 10.0 / ShotsPerFixture.float * 100.0
|
|
|
|
proc minPct(a: array[10, int]): float =
|
|
result = 1e9
|
|
for x in a: result = min(result, x.float / ShotsPerFixture.float * 100.0)
|
|
|
|
proc maxPct(a: array[10, int]): float =
|
|
result = -1e9
|
|
for x in a: result = max(result, x.float / ShotsPerFixture.float * 100.0)
|
|
|
|
proc signFlipP(before, after: array[10, int]): tuple[p, obsMeanPp: float, nPos, nNeg, nZero: int] =
|
|
## Exact sign-flip permutation test on the paired per-fixture deltas.
|
|
var deltas: array[10, float]
|
|
for i in 0..<10:
|
|
deltas[i] = (after[i] - before[i]).float / ShotsPerFixture.float * 100.0
|
|
if deltas[i] > 1e-9: inc result.nPos
|
|
elif deltas[i] < -1e-9: inc result.nNeg
|
|
else: inc result.nZero
|
|
result.obsMeanPp += deltas[i] / 10.0
|
|
let obs = abs(result.obsMeanPp)
|
|
var ge = 0
|
|
for mask in 0..<(1 shl 10):
|
|
var m = 0.0
|
|
for i in 0..<10:
|
|
let s = if ((mask shr i) and 1) == 1: -1.0 else: 1.0
|
|
m += s * deltas[i] / 10.0
|
|
if abs(m) >= obs - 1e-12: inc ge
|
|
result.p = ge.float / 1024.0
|
|
|
|
proc loadDump(path: string): seq[bool] =
|
|
if not fileExists(path):
|
|
return @[]
|
|
for line in lines(path):
|
|
let s = line.strip()
|
|
if s.len == 0: continue
|
|
result.add (s == "1")
|
|
|
|
proc zTest(a, b: seq[bool]): tuple[p, diffPp, z: float] =
|
|
## Two-proportion z-test (analytic; the permutation equivalent is exact but
|
|
## 344k-element shuffles are needlessly slow). ANTI-CONSERVATIVE because the
|
|
## bullets are correlated within a fixture.
|
|
if a.len == 0 or b.len == 0: return (1.0, 0.0, 0.0)
|
|
var ha, hb: int
|
|
for x in a: (if x: inc ha)
|
|
for x in b: (if x: inc hb)
|
|
let p1 = ha.float / a.len.float
|
|
let p2 = hb.float / b.len.float
|
|
result.diffPp = (p2 - p1) * 100.0
|
|
let p = (ha + hb).float / (a.len + b.len).float
|
|
let se = sqrt(max(1e-30, p * (1.0 - p) * (1.0/a.len.float + 1.0/b.len.float)))
|
|
result.z = (p2 - p1) / se
|
|
result.p = erfc(abs(result.z) / sqrt(2.0))
|
|
|
|
proc reportMetric(mname: string,
|
|
before, after: array[3, array[10, int]]) =
|
|
echo "══════════════════════════════════════════════════════════════════"
|
|
echo " METRIC = ", mname
|
|
echo "══════════════════════════════════════════════════════════════════"
|
|
for gi in 0..<2:
|
|
let b = before[gi]
|
|
let a = after[gi]
|
|
echo fmt"{GunNames[gi]}:"
|
|
echo fmt" before {sum(b):>4}/{ShotsPerFixture*10} = {meanPct(b):5.2f}% per-run {minPct(b):5.2f}..{maxPct(b):5.2f}%"
|
|
echo fmt" after {sum(a):>4}/{ShotsPerFixture*10} = {meanPct(a):5.2f}% per-run {minPct(a):5.2f}..{maxPct(a):5.2f}%"
|
|
let sf = signFlipP(b, a)
|
|
echo fmt" delta {meanPct(a)-meanPct(b):+5.2f}pp paired sign-flip permutation p={sf.p:.3f} (+{sf.nPos}/-{sf.nNeg}/0:{sf.nZero} of 10)"
|
|
let b = before[2]
|
|
let a = after[2]
|
|
echo fmt"{GunNames[2]}:"
|
|
echo fmt" before {sum(b):>4}/{ShotsPerFixture*10} = {meanPct(b):5.2f}% per-run {minPct(b):5.2f}..{maxPct(b):5.2f}%"
|
|
echo fmt" after {sum(a):>4}/{ShotsPerFixture*10} = {meanPct(a):5.2f}% per-run {minPct(a):5.2f}..{maxPct(a):5.2f}%"
|
|
let sf = signFlipP(b, a)
|
|
echo fmt" delta {meanPct(a)-meanPct(b):+5.2f}pp paired sign-flip permutation p={sf.p:.3f} (+{sf.nPos}/-{sf.nNeg}/0:{sf.nZero} of 10)"
|
|
echo ""
|
|
|
|
proc main() =
|
|
randomize(12345)
|
|
reportMetric("bmPath (shipped)", BeforePath, AfterPath)
|
|
reportMetric("bmPoint", BeforePoint, AfterPoint)
|
|
|
|
echo "══════════════════════════════════════════════════════════════════"
|
|
echo " bullet-level two-proportion z-test (bmPath dumps) — ANTI-CONSERVATIVE"
|
|
echo "══════════════════════════════════════════════════════════════════"
|
|
for gi in 0..<3:
|
|
let bf = loadDump(fmt"/tmp/wavepair_before_path_{GunNames[gi]}.txt")
|
|
let af = loadDump(fmt"/tmp/wavepair_after_path_{GunNames[gi]}.txt")
|
|
let r = zTest(bf, af)
|
|
var hb, ha: int
|
|
for x in bf: (if x: inc hb)
|
|
for x in af: (if x: inc ha)
|
|
echo fmt"{GunNames[gi]:<12} before {hb:>6}/{bf.len:<6} {hb.float/bf.len.float*100:5.2f}% after {ha:>6}/{af.len:<6} {ha.float/af.len.float*100:5.2f}% diff {r.diffPp:+5.2f}pp z={r.z:+5.2f} p={r.p:.2g}"
|
|
|
|
when isMainModule:
|
|
main()
|