diff --git a/common_libs/tests/range_guns.nim b/common_libs/tests/range_guns.nim index 18d2afc..35b98d9 100644 --- a/common_libs/tests/range_guns.nim +++ b/common_libs/tests/range_guns.nim @@ -19,7 +19,7 @@ import guns/decay_gf import guns/knn_gun import guns/tm_selector -proc buildAllGunDrivers*(seed = -1, enableTmSelector = true): seq[GunDriver] = +proc buildAllGunDrivers*(seed = -1, enableTmSelector = false): seq[GunDriver] = ## seed >= 0 re-seeds the global RNG after constructing the stochastic guns ## (Tsetlin and the TM selector both call randomize() in their constructors), ## so their learning is reproducible for offline runs. @@ -28,12 +28,18 @@ proc buildAllGunDrivers*(seed = -1, enableTmSelector = true): seq[GunDriver] = ## ## `enableTmSelector` must MIRROR the live rack. The shipped ModularBot has ## `EnableTmSelector = false`, so the live loop never spawns gun-13 virtual - ## bullets. The replay's shared VirtualTracker ring is order-sensitive: extra + ## bullets. The replay's shared VirtualTracker ring is ORDER-SENSITIVE: extra ## gun-13 spawns shift the ring head and permute the per-tick resolution ORDER ## of every other gun, which scrambles the `obs` insertion order of the - ## learning guns (KNN, DecayGF) and makes the offline metric diverge from the - ## live one. Callers that mirror the shipped bot must pass false (the - ## acceptance test does); tests that specifically exercise TMSelect pass true. + ## learning guns (KNN, DecayGF) and shifts their predictions. That is exactly + ## the defect the acceptance test was fixed for (commit 4cd5618): with gun 13 + ## spawning offline, the live and offline KNN traces diverged; with its ready + ## gate closed they became byte-identical. + ## + ## The DEFAULT is therefore `false` (mirror the shipped rack). Only a caller + ## that is deliberately measuring TMSelect as a gun should pass `true`; the + ## default must never silently inject a gun the shipped bot does not spawn, + ## because the corruption lands on the OTHER guns' numbers. var tsetlin = initTsetlinGun() var tmSelector = initTmSelectorGun() if seed >= 0: diff --git a/common_libs/tests/test_range_rack_parity.nim b/common_libs/tests/test_range_rack_parity.nim new file mode 100644 index 0000000..7f9b85e --- /dev/null +++ b/common_libs/tests/test_range_rack_parity.nim @@ -0,0 +1,63 @@ +## Guard: the offline range default MUST mirror the live rack. +## +## `buildAllGunDrivers()` used to default `enableTmSelector = true`, so every +## default caller (`run_range`, `analyze_selector`, `test_power_selection`, +## `measure_power_policy`) spawned gun 13 (TMSelect) — a gun the shipped bot +## NEVER spawns (`EnableTmSelector = false` in ModularBot.nim). The shared +## `VirtualTracker` ring is order-sensitive: gun 13's extra 4 bullets/tick shift +## the ring head and permute every other gun's per-tick resolution order, which +## reorders the learning guns' observations and shifts their predictions. That is +## the exact confound the acceptance test was repaired for (commit 4cd5618, where +## closing gun 13's offline ready gate made the live/offline KNN traces +## byte-identical). +## +## Before the fix this file FAILS: the default replay resolves gun-13 bullets. +## After the fix it PASSES and the rack mirrors the shipped bot. +## +## Run: nim c -r --path:common_libs common_libs/tests/test_range_rack_parity.nim + +import std/[os] +import gun_harness/offline_range +import range_guns + +const fixturesDir = currentSourcePath().parentDir.parentDir.parentDir / "tools" / "fixtures" + +var failures = 0 +proc check(name: string, ok: bool) = + if ok: echo "PASS: ", name + else: echo "FAIL: ", name; inc failures + +proc tmShots(drivers: seq[GunDriver], fx: Fixture): int = + let reports = replayFixture(fx, drivers) + for r in reports: + if r.name == "TMSelect": return r.shots + 0 + +proc main() = + let path = fixturesDir / "drussgt_vs_ramfire.jsonl" + if not fileExists(path): + echo "SKIP: fixture missing: ", path + quit(0) + let fx = loadFixture(path) + + # 1. THE DEFECT: the default rack must NOT spawn the disabled TMSelect gun. + check "default rack never spawns gun 13 (TMSelect)", + tmShots(buildAllGunDrivers(), fx) == 0 + + # 2. The flag is live: asking for TMSelect explicitly DOES spawn it, so the + # zero above is the gate and not a broken gun. + check "enableTmSelector=true still spawns gun 13 (flag is live)", + tmShots(buildAllGunDrivers(enableTmSelector = true), fx) > 0 + + # 3. The rack reports the live gun ids 0..13 (14 slots), so report indices are + # stable whichever way the flag is set. + check "driver count is 14 (gun ids 0..13)", + buildAllGunDrivers().len == 14 + + if failures > 0: + echo "\n", failures, " check(s) FAILED" + quit(1) + echo "\nAll range rack-parity checks passed." + +when isMainModule: + main() diff --git a/docs/offline_harness_trust.md b/docs/offline_harness_trust.md new file mode 100644 index 0000000..88764e8 --- /dev/null +++ b/docs/offline_harness_trust.md @@ -0,0 +1,472 @@ +# Can the offline harness be trusted? — audit, calibration, and rules of use + +**Scope.** The user stopped trusting the offline testing harness after four +offline-flavoured claims did not survive the live arena. This document audits the +harness code, re-runs the offline==online acceptance test independently, builds +an offline-prediction-vs-live-outcome table for every arm this session measured, +and states exactly when the harness is and is not trustworthy. + +**Evidence tags.** `[MEASURED]` = read from a recorded artifact, a source file, or +reproduced by a command in this document (the command is named). `[INFERRED]` = +reasoning from measured facts. + +--- + +## 0. The single most important sentence + +> **The offline harness is trustworthy — provably, to the last hit — for one +> thing only: the per-gun, single-tick prediction quality of a gun on a FIXED +> enemy trajectory. It is NOT trustworthy for anything whose value flows through +> the closed loop (movement, range, round length, adaptation, gun selection), +> because the replay's enemy never reacts to our bot and the replay never calls +> the selector at all. And even for prediction quality, its output is the +> virtual-bullet metric, which is a poor, sign-unstable ranker of real hit rate.** + +`[MEASURED]` for the parity claim (§1, §2.1), the selector blindness (§4.1), and +the poor-ranker result (§3, §4.3). The "trust it only for single-tick prediction" +verdict is `[INFERRED]` from those measurements plus the calibration table in §3. + +--- + +## 1. "The offline harness" is three different things — the distrust conflates them + +The four failures in the user's evidence list do **not** all come from the same +harness. They come from three, and one of them is not offline at all. + +| id | harness | what it measures | where | +|---|---|---|---| +| **H1** | **gun range / virtual bullets** | per-gun `path`-metric hit rate over a replayed `seq[WorldState]` | `common_libs/gun_harness/offline_range.nim`, `virtual_bullets.nim`, driven by `common_libs/tests/run_range.nim` | +| **H2** | **movement fixture replay** | per-tick movement-command diagnostics (reversal rate, pick interval, tile reasons) over the same fixtures | `common_libs/tests/test_tfil_commit_env.nim` (includes `movements/the_floor_is_lava.nim`) | +| **H3** | **TM classifier-accuracy harness** | prequential side/label accuracy of the Tsetlin-Machine corrector, not hit rate | `common_libs/tests/measure_tm_readapt.nim`, `measure_tm_hit_optimal.nim`, `measure_tm_miss_shrink.nim` | +| — | **NOT offline** | live server-side-event hit rate / damage / round wins | `tools/robocode_shim/run_bridge_battle.sh`, `tools/ab/*` | + +### The user's four failures, re-attributed `[MEASURED]` + +1. **Ring mover (20.28%).** This is a **live** number: `docs/feature_ab_results.md` + and commit `bfdcdf8` say "35 **real-DrusGT bridge battles** … server-side event + sidecar ground truth", so 20.28% is the live real hit rate of the ring arm, and + 6/49 round wins is the live survival. **There is no offline measurement of the + ring mover anywhere**: the offline harness only scores guns. The label + "best offline hit rate" in `docs/env_reference.md:342` and commit `7f6ccfb` is + **wrong**. This failure is a **metric-mismatch** (hit rate vs damage/round-wins + for a movement arm), not a harness-calibration failure. `[MEASURED]` +2. **TM / TMHorizon gun.** Offline evidence is **H3** (classifier side accuracy), + not H1. H1 (`run_range`/`buildAllGunDrivers`) does not even contain the + TMHorizon gun — it builds 14 drivers (ids 0..13); TMHORIZON is id 15 + (`common_libs/tests/range_guns.nim`). `[MEASURED]` +3. **`TR_TMHORIZON_WINDOW=150` (+9.3pp).** H3 again. `[MEASURED]` +4. **TFIL reversal theory.** H2 (movement replay) for the mechanism, plus the live + `TR_TFIL_COMMIT_LOG`. `[MEASURED]` + +So two of the four are H3 (classifier accuracy, not hit rate), one is H2 (open-loop +movement), and one was never offline. + +--- + +## 2. Correctness audit of the harness code + +### 2.1 The offline==online parity claim — independently re-run, 12/12 `[MEASURED]` + +I re-ran the acceptance test myself rather than trusting the claim: + +``` +nim c -r --nimcache:/tmp/nc_j89 common_libs/tests/acceptance_offline_vs_online.nim +``` + +Result (fresh live round, ModularBot vs OscillatorBot, max speed, 264 recorded +ticks, `enemyDied=true`): + +``` +deterministic guns matching exactly: 12/12 +VERDICT: PASS — offline == online for all 12 deterministic guns. +``` + +Per-gun hits/shots, online vs offline, identical for all 12 deterministic guns +(HeadOn 84/400, Linear 132/400, Circular 156/400, GuessFactor 89/400, Pattern +400/400, WallBounce 141/400, Accel 233/400, StopShot 120/400, Displace 43/400, +AvgLead 154/400, DecayGF 88/400, KNN 161/400). Tsetlin differed (123/400 online vs +131/400 offline) and is labelled stochastic; TMSelect is 0/0 both ways. + +**Conclusion: H1 reproduces the live bot's own per-gun virtual-bullet telemetry +exactly**, on a fogged, bot-recorded fixture, including the death boundary. This is +the strongest evidence in the repo and it holds up. + +### 2.2 The bullet-arrival resolver and the `bmPoint`/`bmPath` difference `[MEASURED]` + +Two models live in `virtual_bullets.nim`: + +- `bmPath` (shipped default) — the bullet flies along its ray until it leaves the + arena; each tick's swept segment is tested against the target radius + (`virtual_bullets.nim:658-725`). One fitness sample per bullet, recorded at the + wall. +- `bmPoint` — resolves on the first tick where `b.travelDist >= b.fireDist` and + scores the bullet position **at `b.travelDist`**, i.e. up to one tick-step + (≤ `bulletSpeed`, ≤ 17 px) **past** the fire-time aim distance + (`virtual_bullets.nim:617-655`, the `(bx, by)` interpolation on `b.travelDist`). + +**Finding (real, low-impact, not fixed).** `bmPoint`'s own docstring says it scores +"the single point it reaches **at the fire-time aim distance**" +(`virtual_bullets.nim:97-104`), but the code scores the point at the *first +at-or-beyond* distance. The parallel arrival-accuracy probe used by the tie-break +does it the documented way — it interpolates to exactly `b.fireDist` +(`virtual_bullets.nim:689-695`). The two therefore **can disagree**, even though +the probe's comment claims it "records the SAME outcome the point model would +have" (`virtual_bullets.nim:684-688`). A concrete disagreement exists whenever the +enemy sits in the ~`bulletSpeed`-wide crescent between the disk of radius +`BotRadius` around the aim point and the disk around the slightly-further point +(e.g. fire (100,100) → aim (300,100), enemy at (304.1, 117.9): `bmPoint` scores +hypot(0.1, 17.9)=17.9 < 18 = HIT; the probe scores hypot(4.1, 17.9)=18.4 = MISS). + +Which is right is *not* ambiguous — the docstring and the probe both describe the +aim point; `bmPoint` overshoots. But I did **not** change it, because: +(a) `bmPoint` is the non-default metric, already measured NEGATIVE live (4.70% vs +7.43% for `path`, p<1e-4, `docs/gun_rack_analysis.md` §2.1), so no shipped behaviour +depends on it; (b) changing it would silently invalidate the committed +`point`-metric numbers in `gun_rack_analysis.md`, `tm_pattern_sweep_results.md`, +`sweep_pattern_radial`/`audit_virtual_guns` outputs; (c) it affects only the +`GUN_SELECTOR_TIEBREAK != off` fork, which is default-off and measured null +(`gun_rack_analysis.md` §6.8). The **right** fix is one line in the `bmPoint` branch +(interpolate to `b.fireDist`, as the probe already does) and should be done *with* a +re-measure of the point metric, not silently here. `[MEASURED]` for the mechanics; +`[INFERRED]` for the crescent argument. + +### 2.3 Virtual-bullet pairing / attribution — the mispairing fix re-verified `[MEASURED]` + +A prior job measured 36-57% mispairing in GF/DecayGF/KNN (FIFO queue drained +1-push-vs-4-pops) and replaced it with an exact `(fireTick, powerBin)` ring lookup +(`guess_factor.nim:83-105,116-124`; `decay_gf.nim:76-133`; `knn_gun.nim:151-210,300-318`). +I re-verified the fix: + +- **By construction:** `onResult` looks up `waveSlot(e.fireTick, binIdx)` and, if + the slot holds a different `fireTick`, increments `waveMispaired` and returns + **without recording** (`guess_factor.nim:120-131`). A collision is detected, never + silently mislabelled. `waveStarved` counts a missing key. +- **By test:** `nim c -r common_libs/tests/test_wave_pairing.nim` → **17 PASS, 0 FAIL**, + including "resolve order 2,0,1 → all three waves found, none starved" and + "reverse resolution … no mispair / no starve" for all three guns. +- **Ring period:** `WaveRingSlots = 1024`, key period 1024/4 = 256 ticks; the + slowest bullet (power 3.0, 11 px/tick) leaves an 800×600 arena within ~91 ticks + and a 1000×1000 arena within ~128, so a live wave is never overwritten by a newer + one. `[MEASURED]` + +**Verdict: the pairing fix is correct.** Attribution (`FeedbackEvent.fireTick`, +`powerBin`, `hit`) is consistent between the tracker and the guns. + +### 2.4 What the fixture replay feeds the guns vs what the live bot sees `[MEASURED]` + +The replay is fed by `common_libs/gun_harness/offline_range.nim:248-276`. Differences +from the live `WorldState` the guns actually get: + +| # | difference | where | impact | +|---|---|---|---| +| 1 | **`liveActual` ordering shift** — the live loop calls `go()` (which dispatches the NEXT scan into `enemyTracker`) *before* the aim block, so the resolver reads the NEXT frame's pose/`lastSeenTick`. Modelled by `actIdx = si+1` (`offline_range.nim:261-270`). | `offline_range.nim:261` | must be `true` for bot-recorded fixtures; `false` for external captures. `run_range.nim:49` derives it from `meta.source == "live"`. The bot recorder writes `source: "live"` (`ModularBot.nim:342-355`), so this is wired correctly. | +| 2 | **final-tick drop** — if the target died during the last `go()`, the live aim block is skipped; modelled by `skipFinal` (`offline_range.nim:249,273`). | `offline_range.nim:249` | needed for the parity match (verified). | +| 3 | **stale/fogged pose** — bot-recorded fixtures carry the tracker pose (`perfect_info:false`) + `lastSeenTick`. | recorder | exact. | +| 4 | **perfect-info captures** — `tr-bridge` and `classic-robocode` fixtures carry **true positions every tick** (`perfect_info:true`). The live bot only sees the target on radar scans. | fixture meta | **optimistic**; absolute rates on these fixtures ≠ live framing. | +| 5 | **`selfRadarHeading`** is reconstructed as `selfHeading` (`offline_range.nim:135`), losing the real `getRadarDirection()`. | `offline_range.nim:135` | no gun or movement module reads it (grep: no consumers) — currently inert. | +| 6 | **one enemy in `enemies[]`** — the loader builds a single-element list (`offline_range.nim:123-152`). | — | exact in 1v1; melee is not modelled. | +| 7 | **skipped ticks are not recorded** — `buildState` (and therefore a fixture line) only runs when the target is valid and alive, so ticks where the aim block was skipped simply do not exist as frames. | `ModularBot.nim:397-421` | consistent for the mid-round gap (both skip), but the `si+1` shift assumes the next recorded frame is the next tick. A temporary target-invalid gap would mis-align the `si+1` pose. Not exercised by the acceptance round. | + +### 2.5 Does the offline score measure the same quantity the live battle scores? `[MEASURED]` + +Two levels, and only one of them is a yes: + +1. **offline per-gun virtual hit rate == the live bot's per-gun `vShots`/`vHits`** — + **YES, exactly** (12/12, §2.1), on `path` (the shipped metric), and only when the + replay mirrors the live rack (§2.7) and the metric/rack env match. +2. **the live bot's virtual hit rate == the live battle score/outcome** — **NO.** + `reportFor` (`offline_range.nim:205-219`) and `ModularBot.onRoundEnded` + (`ModularBot.nim:589-620`) both compute `min(count, WindowSize)` hits over the + ring — a **last-100-samples-per-bin** rate, not the whole round, and not damage + or round wins. The selector that consumes this signal is measured **negative + value**: `onlyPattern` 10.78% vs the full rack 6.93% (p=0.0012), and the virtual + rank correlates with real rank by Spearman −0.374 / +0.335 on parallel run sets + (`docs/gun_rack_analysis.md` §2-§3). + +**So: the offline harness measures the same quantity as the live *telemetry*, not +the same quantity as the live *battle*.** + +### 2.6 Comparisons that were not apples-to-apples `[MEASURED]` + +- **`bmPoint` vs `bmPath` totals.** Offline totals are inflated by three synthetic + perfect-info fixtures that score 100% under `path` for every gun + (`gun_rack_analysis.md` §2.1), so offline totals are comparable only to each + other, never to live rates. The docs say this; it is easy to forget. +- **Firing gate threshold calibrated on one metric.** The range-aware gate's + `SafetyFactor = 0.6` was fitted from real-shot data (fine), but `aimToleranceDeg` + is used by `shouldFire` only; the virtual harness never calls it, so no offline + number sees the gate. +- **Per-gun live "real %" is conditional on selection.** Every per-gun live rate in + `gun_rack_analysis.md` §2/§3 (e.g. Linear 10.7%) is conditional on the selector + having picked that gun. The clean unconditional rates are only the `onlyX` arms + (Linear 3.27%, KNN 5.13%, GF 2.25%, Pattern 10.78%). Comparing an offline + unconditional rate to a conditional live rate is a confound the report names. +- **Movement arms judged on hit rate.** Explicitly corrected in + `docs/feature_ab_results.md`: a movement change alters range, shots and round + length at once, so hit rate INVERTS the verdict; damage/round-wins are the + objective. The ring mover is the case in point. + +### 2.7 BUG FIXED — the offline rack default did not mirror the live rack `[MEASURED]` + +`buildAllGunDrivers` in `common_libs/tests/range_guns.nim` defaulted to +`enableTmSelector = true`, so every *default* caller (`run_range.nim:48`, +`analyze_selector.nim:54`, `test_power_selection.nim:56`, +`measure_power_policy.nim:97`) spawned gun 13 (TMSelect) — a gun the shipped bot +**never** spawns (`EnableTmSelector = false`, `ModularBot.nim`). The shared +`VirtualTracker` ring is **order-sensitive**, so those extra 4 spawns/tick permute +every other gun's per-tick resolution order and reorder the learning guns' +observations. This is the *exact* confound the acceptance test was repaired for in +commit `4cd5618` — but that commit fixed only the acceptance caller and left the +helper's default broken for everyone else. + +**Fix:** `range_guns.nim` now defaults `enableTmSelector = false` (mirror the +shipped rack); `true` is opt-in for callers deliberately measuring TMSelect. + +**Evidence it was a real defect (before → after, `run_range` on +`tools/fixtures/tr_drussgt_vs_modularbot.jsonl`):** + +| gun | before (gun 13 injected) | after (live rack) | +|---|---|---| +| Tsetlin | 75/400 (18.8%) | 74/400 (18.5%) | +| KNN | 30/400 (7.5%) | 29/400 (7.2%) | +| TMSelect | 61/400 (15.2%) | 0/0 | + +On `drussgt_vs_crazy` the perturbation moves Tsetlin's per-bin counts +(p2.0 22→19); on `drussgt_vs_drussgt` Tsetlin 25→27 hits. Small in aggregate, +but it lands on the **other** guns' numbers, which is the worst kind of silent error +for a harness whose whole job is ranking guns. + +**New guard:** `common_libs/tests/test_range_rack_parity.nim` (3 checks). I proved +it **fails before and passes after** by `git stash`-ing the fix: + +``` +# with the fix: +PASS: default rack never spawns gun 13 (TMSelect) +PASS: enableTmSelector=true still spawns gun 13 (flag is live) +PASS: driver count is 14 (gun ids 0..13) +# with range_guns.nim reverted: +FAIL: default rack never spawns gun 13 (TMSelect) +``` + +**Guard counts unchanged** for every required suite (the fix changes no test file +in the green list; `test_power_selection` re-verified at **3/3** after the fix). + +### 2.8 Coverage gap (LIMIT, not a bug): 2 of 16 live guns are not in the offline range + +`buildAllGunDrivers` builds **14** drivers (ids 0..13). The live bot has **16** +(`TMPATTERN` id 14, `TMHORIZON` id 15). Demonstrated by running the range with both +enabled — the output is byte-identical to the default and has **no rows** for guns +14/15: + +``` +TR_RACK_TMHORIZON=both TR_RACK_TMPATTERN=both /tmp/run_range tools/fixtures/drussgt_vs_ramfire.jsonl +diff <(... default ...) <(... tm ...) -> IDENTICAL ; rows = 14 +``` + +So H1 **cannot measure** the exact gun that produced two of the four failures. +Those arms were measured by H3 instead. `[MEASURED]` + +--- + +## 3. Calibration: offline prediction vs live outcome, every arm this session + +Rows are restricted to arms where an offline harness produced a **directional +prediction about a live outcome** and a live A/B tested it. Live numbers are read +from committed artifacts (`docs/env_reference.md` §"Measured verdicts", +`docs/feature_ab_results.md`, `docs/selector_negative_value.md`, +`docs/ramming_negative_result.md`, `common_libs/tests/fixtures/tfil_commit_ab_report_runs{7,14}.md`, +`common_libs/tests/measure_tm_readapt_results.txt`, `docs/gun_rack_analysis.md`); +they are **not** re-run here. + +| # | arm | offline harness + directional prediction | live outcome | agree? | +|---|---|---|---|---| +| 1 | `TR_TMHORIZON_WINDOW=150` | **H3**: window 84.6% vs accum 75.3% late side-accuracy = **+9.3pp help** (shuffled ctrl ~51%) | **13/49 = 26.5%** vs 49.0%, **p=0.036 HARMFUL** | **NO** | +| 2 | TMHorizon `TR_TMHORIZON_NSTATES` 2 / 8 | **H3**: lower inertia lifts keep-all late acc 75.3→83.3 (window flat) = **help** | 42.9% / 42.9% vs 49.0–53.1%, all **p≥0.8** = **null** | **NO** | +| 3 | TM as a gun (radial TM / `onlyTMPATTERN`) | **H3**: side signal above chance; PERF-SIGN form **+4.9pp** hits; break-even ~80% | `onlyTMPATTERN` **3.50%** vs `onlyPattern` 10.74%, **p=0.0006**, 0/49 wins; live side acc **~50% = chance** | **NO** | +| 4 | TMSelect gun admitted | **H1**: loses to the best of its own experts offline on nearly every fixture = **negative** | **negative** (disabled; cost 7.47%→5.59%) | **YES** | +| 5 | `lean8` / `lean6` racks (H1 audit recommendation) | **H1**: duty/overlap audit recommends pruning to `lean8` (8 good guns) | lean8 **6.31%**, lean6 **8.83%** vs `onlyPattern` **10.36%**, p=0.017 / 0.026 = **they lose** | **NO** | +| 6 | single-gun arms Pattern / KNN / Linear / GF | **H1** all-20 rank: Pattern 55 > **Linear 51 > GF 49 > KNN 46** | live `onlyX`: Pattern 10.78 > **KNN 5.13 > Linear 3.27 > GF 2.25** | **NO** (only rank 1 agrees; Spearman 0.4) | +| 7 | `GUN_VBULLET_METRIC` point → path | **H1**: path 50.8% vs point 34.3% = **path better** | path **7.43%** vs point 4.70%, **p<1e-4** = **path better** | **YES** | +| 8 | power bar absolute 0.40 → relative | **H1**: no bin ever clears 0.40 = the bar is **broken** | relative bar → **damage +52%** (157→239/run) | **YES** | +| 9 | TFIL commitment arms A–E (movement) | **H2**: mechanism exact (interval 5.07/14.65/14.65/4.94/27.57); **reversals rise for B/E too** (33.7→41.1 / 42.0) | live: **no arm** improves damage or wins (p>0.3); reversals **rise** 33.9→50.9 / 57.0 | **mechanism YES, outcome NO** | +| 10 | ring mover (`tfil_ring`) | **not an offline arm** — live-only, mislabelled "offline" | glass cannon: 20.28% hit but 6/49 wins vs 16/49 (p=0.012) | **excluded (N/A)** | +| 11 | `TR_RAM_OPPORTUNITY`, `TR_POWER_POLICY`, sub-1.0 power accuracy | no offline arm (unit tests or live log-replay only) | 0/59 conversions; policy helps (p=0.0012); null (p=1.0) | **N/A** | + +### Direction-agreement rate + +- **Sample size: n = 9 usable arms** (#1–#9; #10 is not offline, #11 has no offline + prediction). **Agreement: 3/9 = 33%** (#4, #7, #8). #9 agrees on mechanism but + not on outcome, so it is counted as a NO under the strict rule. +- Split by **where the effect has to travel**: + + | domain | arms | correct | rate | + |---|---|---|---| + | open-loop — single-tick prediction, metric choice, threshold bug | #4, #7, #8 | 3/3 | **100%** | + | closed-loop — adaptation, range, movement, selection/rack | #1, #2, #3, #5, #6, #9 | 0/6 | **0%** | + +**This is a small, non-random, hand-assembled set of arms** — it is the session's +decision log, not a sample. No correlation coefficient or confidence interval +should be computed or over-claimed from it; the 33% / 100% / 0% figures are +descriptive of these nine decisions only. What they support is the +*mechanistic* split below, which follows from the harness's structure, not from +the count. + +--- + +## 4. The open-loop hypothesis: stated, evidenced, and tested + +**Hypothesis.** The fixtures are open-loop: the recorded enemy trajectory is an +input, not a function of our bot's behaviour. Therefore (a) anything that is a +function of the fixed enemy stream *and our fixed gun list* — single-tick +prediction quality — is reproduced faithfully, while (b) anything whose value runs +through the closed loop (our movement, our selection, our adaptation, the enemy's +reaction to our bullets, round length, range) is systematically absent, and its +offline measurement is at best blind and at worst inverted. + +### 4.1 Structural test — the harness is blind to our own decisions `[MEASURED]` + +The replay never calls the selector, and the enemy stream is read from the fixture. +I verified empirically that the `TR_RACK_*` knobs have **zero** effect on the +offline range output: + +``` +FX=tools/fixtures/drussgt_vs_ramfire.jsonl +/tmp/run_range_bin $FX > a # shipped onlyPattern +TR_RACK_PATTERN=off TR_RACK_KNN=both TR_RACK_HEADON=both /tmp/run_range_bin $FX > b +TR_RACK_TMHORIZON=both TR_RACK_TMPATTERN=both /tmp/run_range_bin $FX > c +diff a b -> IDENTICAL ; diff a c -> IDENTICAL +``` + +Every gun spawns a virtual bullet every tick regardless of whether the live bot +would select it, so the offline per-gun table **cannot see the selector at all**. +The same holds for movement: nothing in H1 reads our position history as a +*consequence* — the self positions are replayed from the fixture, so a change to +our mover is not representable. A rack arm (#5, #6) therefore has **no** offline +measurement of its effect; H1 only ranks guns one by one. + +### 4.2 The positive half — faithful single-tick reproduction `[MEASURED]` + +The acceptance test (§2.1) proves that for a fixed fogged enemy stream, the replay's +per-gun hit/miss sequence is byte-exact against live, including the resolver's +`go()` ordering and the death boundary. Independently, `test_vbullet_metric` (11 +checks) pins the point/path geometry, and `test_wave_pairing` (17 checks) pins the +exact wave attribution. So the *mechanism* of single-tick prediction is trustworthy. + +### 4.3 The negative half — the metric doesn't even rank real hit rate `[MEASURED]` + +On the committed TR-bridge fixture (a capture in which DrussGT *did* react to our +bullets, but which is replayed open-loop), the offline range ranks: + +``` +StopShot 21.0% > Tsetlin 18.8% > HeadOn/Accel 14.8% > WallBounce 13.8% > +Circular/AvgLead 12.8% > Linear/DecayGF 11.8% > Pattern 11.5% > GF 11.0% > +Displace 11.5% > KNN 7.5% (my re-run, /tmp/rr_tr.txt) +``` + +Live, conditional on selection, KNN is the *third-best* gun (9.0%) and Pattern +fourth (8.6%), while StopShot (6.1%) and Tsetlin (5.8%) are near the bottom +(`gun_rack_analysis.md` §3). Spearman across run sets is **sign-unstable** +(−0.374 over 13 runs, +0.335 over 15, `gun_rack_analysis.md` §2). So even the +faithfully-reproduced quantity is a poor ranker of what we care about. + +### 4.4 The H3 caveat, already recorded at the time `[MEASURED]` + +The re-adaptation commit (`69debbe`) states the hypothesis in its own words: +*"The fixtures are OPEN-LOOP (DrussGT does not react to our bullets), so a true +mid-round adaptation is NOT present … its absolute side accuracy (75-85%) is +INFLATED: the same gun measured ~52% = chance live … So +9.3pp is a real ARM DELTA, +not a promise that the gun now clears the ~80% accuracy wall that hits need."* +The live A/B later found it **harmful** (p=0.036). The lesson was written down and +then not applied. `[MEASURED]` + +**Hypothesis verdict:** **supported** by §4.1 (structural blindness, empirically +shown) and §4.3, and consistent with every closed-loop arm in §3 failing. Not +falsified anywhere. + +--- + +## 5. BUG vs LIMIT — they need different fixes and different documentation + +| class | item | fix | +|---|---|---| +| **BUG** (fixed) | default rack injected the disabled TMSelect gun → perturbed the learning guns (§2.7) | `range_guns.nim` default flipped; guard `test_range_rack_parity.nim` | +| **BUG** (real, deliberately not fixed) | `bmPoint` scores up to 17 px past its documented aim point; the tie-break probe already does it correctly (§2.2) | change the `bmPoint` interpolation to `b.fireDist` **together with** a re-measure of the point metric | +| **LIMIT** | replay is open-loop; enemy never reacts; selector never called (§4.1) | **not fixable by code** — it is the definition of a recorded stream. Document, and gate every claim that depends on the loop | +| **LIMIT** | offline score is the last-100 virtual hit rate, not damage/round-wins (§2.5) | keep it as a *gun* metric; never use it as an objective | +| **LIMIT** | perfect-info captures (`tr-bridge`, `classic-robocode`) feed true positions every tick (§2.4 #4) | use `source=live` bot captures for parity; use the others only for relative gun ranking | +| **LIMIT** | the range omits guns 14/15 (TMPATTERN/TMHORIZON) (§2.8) | add drivers only with a matching live rack; until then, TM arms must be measured live or by H3 with the H3 caveat attached | +| **LIMIT** | `bmPoint`/`bmPath` offline totals are inflated by perfect-info fixtures (§2.6) | compare offline-to-offline only | +| **LIMIT** | `selfRadarHeading` is not recorded (§2.4 #5) | harmless today (no consumer); fix if a consumer appears | + +--- + +## 6. How to use this harness — rules for future work + +1. **State which harness you mean.** H1 (gun range), H2 (movement replay), H3 + (classifier accuracy) answer different questions. A "+9.3pp" from H3 is not a + hit-rate claim. +2. **Use H1 only to ask single-tick prediction questions** about a gun on a fixed + trajectory: "does this gun predict this recorded motion better than that one", + "is this threshold/bug real". Its 12/12 parity is real and it is the right tool + for that. +3. **Never use H1 (or H2/H3) to predict a movement, adaptation, range, round-length + or selection/rack outcome.** It is structurally blind to all of them (§4.1). Any + such arm must be decided by a live A/B on damage **and** round wins. +4. **For movement arms, report all three: damage/run, round-win rate and hit rate**, + and treat hit rate alone as inverted (`feature_ab_results.md`). +5. **For gun/selection arms, hit rate is the right live ground truth** but judge on + the *server-side events sidecar*, never the bot's own attribution and never + scores. +6. **A live arm must be pre-registered with a replication block.** The TFIL case + shows a single 7-run block can show +21 damage at p=0.26 that does not replicate + (+3.4 in block 2). Do not ship on block 1. +7. **Mirror the live rack exactly in any replay** (`enableTmSelector`, disabled + guns). The ring is order-sensitive; the default is now safe but check it. +8. **Match the fixture family to the question.** `source=live` (fogged, recorder) → + parity work. Perfect-info captures → relative ranking only. +9. **Do not read the offline per-gun ranking as a rack verdict** — it is a poor, + sign-unstable ranker (§4.3). +10. **Fix a harness bug only with a fail-before/pass-after test** and report guard + counts (done for §2.7). +11. **When a knob is added, add its liveness check** (the treatment must be shown to + have applied) — as the TFIL report does. + +--- + +## 7. Method / reproduction + +Commands run for this audit (all with `--nimcache:/tmp/nc_j89`): + +```bash +# independent re-run of the parity proof +nim c -r --nimcache:/tmp/nc_j89 common_libs/tests/acceptance_offline_vs_online.nim # 12/12 PASS + +# pairing fix re-verification +nim c -r --nimcache:/tmp/nc_j89 --path:common_libs common_libs/tests/test_wave_pairing.nim # 17 PASS + +# rack-invariance demonstration (open-loop blindness) +nim c --nimcache:/tmp/nc_j89 --path:common_libs -o:/tmp/run_range common_libs/tests/run_range.nim +/tmp/run_range tools/fixtures/drussgt_vs_ramfire.jsonl > a +TR_RACK_PATTERN=off TR_RACK_KNN=both /tmp/run_range tools/fixtures/drussgt_vs_ramfire.jsonl > b # a==b + +# the fix + its guard (fail-before proven with git stash) +nim c -r --nimcache:/tmp/nc_j89 --path:common_libs common_libs/tests/test_range_rack_parity.nim # 3 PASS + +# every required green suite (counts below are the PASS lines) +``` + +Required suites, all green at their exact counts after the change: +`test_gun_harness 39`, `test_vbullet_metric 11`, `test_power_selection 3`, +`test_power_policy 58`, `test_adaptive_radar 41`, `test_tfil_ring_weights 24`, +`test_ram_decision 40`, `test_rack_membership 48`, `test_selector_tiebreak 19`, +`test_tm_pattern_registration 20`, `test_vbullet_admit_gate 12`, +`test_tm_horizon 104`, `test_tm_diag 48`, `test_tm_automata_diag 55`, +`test_tm_clause_shape 66`, `test_env_report 25`, `test_tfil_commit_env 30` +(as-is). Plus the new `test_range_rack_parity 3`. + +**MEASURED:** the parity re-run, the pairing re-verification, the rack-invariance +diff, the fix magnitude, the guard counts, and every live number quoted (from the +committed artifacts named in §3). +**INFERRED:** the crescent disagreement in §2.2; the "trust only for single-tick +prediction" verdict; the 100%/0% domain split as an explanation (the counts are +descriptive; the causal story is the structural argument in §4.1).