From b0654d18eb80d5da48b657a45c4498d1d0f8bf48 Mon Sep 17 00:00:00 2001 From: Davide Cappellini Date: Tue, 22 Sep 2026 08:21:49 +0200 Subject: [PATCH] TM verdict, settled: it loses LIVE and sits at/below its majority class - (c) The user pushed back on "the TM can't be your best 1v1 gun", correctly, because two decisive tests had never been run. Both are now run and they agree. TASK 1 - THE GF HEAD vs ITS MAJORITY-CLASS BASELINE (offline, n=1,751,067): label histogram [254286, 284578, 678879, 297055, 236269] majority class = 2 (the CENTRE bucket) = 38.77% RAW head accuracy = 36.69% -> margin **-2.08 pp, BELOW majority** GATED head accuracy = 40.37% vs 38.75% majority -> +1.62 pp, BUT it predicts the majority class on 62.4% of ticks and its minority recall is 13.6% / 12.9% - a base-rate predictor wearing a classifier's clothes. Shuffled control sits at its own majority (20.04% vs 20.12%), confirming chance. **THE OLD "46% vs 20% CHANCE" FIGURE I QUOTED WAS WRONG ON TWO COUNTS:** the baseline is 38.8%, not 20%, and the 46% predated the deferred-label fix. Against the correct baseline the head is BELOW it. TASK 2 - THE FIRST-EVER LIVE A/B OF THE TM GUN (7 runs x 7 rounds per arm, one frozen binary from git archive HEAD = eb74f9b2, sha256 cb66d66b..., real DrussGT, every arm forced alone with TR_RACK_=both and all 14 others off, liveness confirmed per run): arm shots real % dmg/run round wins onlyPattern 4610 10.74% 285 25/49 onlyTMPATTERN (radial) 3374 3.50% 71 0/49 onlyLinear 3218 3.23% 61 0/49 Pattern vs TM: +7.22 pp / +213.7 dmg, exact p=0.0006 TM vs Linear: +0.30 pp, p=0.659 (dmg p=0.438) **The TM is statistically INDISTINGUISHABLE from its own Linear base live.** So it is not "the TM works and we are aiming it wrong". DIRECT ANSWER: **(c) It loses live AND sits at/below majority - the target carries no learnable signal beyond the base rate, and that is the reason.** The reason is not the machine, not the knobs, and not the application alone: the thing it was asked to predict is dominated by the modal answer. This closes the TM-as-gun thread. If a TM is wanted in the bot, a firing gate or a movement decision is a better fit for a boolean-rule classifier than an aim point - that is untested and is a different project. A LIVE GF-MODE ARM WAS NOT RUN (stated as unmeasured): the task pinned one frozen HEAD binary and HEAD registers the TM gun as radial only; Task 1 already makes GF the unpromising candidate. HARNESS FIX WORTH KEEPING: `tools/ab/which_gun_arm_env.sh` left the TARGET gun unset, so with the now-Pattern-only default it silently fell back to the FULL rack - an arm could appear to test a single gun while actually running the whole rack. It now emits `TR_RACK_=both` for the target and `=off` for all 14 others. (Earlier which-gun results are unaffected: they ran before the Pattern-only default, or - as in the melee/1v1 campaign - set the explicit `=both` themselves.) tm_pattern.nim gains a per-class confusion matrix (warm samples only) to support the majority baseline; no behaviour change. Adds Round 4 to tm_pattern_sweep_results.md with both tasks and the interpretation rule. --- common_libs/guns/tm_pattern.nim | 10 + common_libs/tests/sweep_tm_pattern.nim | 47 ++++ common_libs/tests/tm_pattern_sweep_results.md | 205 ++++++++++++++++++ tools/ab/which_gun_arm_env.sh | 28 ++- tools/ab/which_gun_run_one.sh | 2 +- 5 files changed, 281 insertions(+), 11 deletions(-) diff --git a/common_libs/guns/tm_pattern.nim b/common_libs/guns/tm_pattern.nim index e3ba0b4..2581d93 100644 --- a/common_libs/guns/tm_pattern.nim +++ b/common_libs/guns/tm_pattern.nim @@ -188,6 +188,14 @@ type radOffsetN*: int classCorrect*: int ## warm predictions whose class matched the eventual label classTotal*: int ## warm predictions with a resolvable label + ## Per-class confusion matrix for the GF head, indexed [true label][predicted + ## class], counted over WARM predictions only (the same samples `classTotal` + ## scores). This is what makes the majority-class baseline and per-class + ## precision/recall measurable. Row sums = the warm label histogram; the + ## diagonal sum = classCorrect. + confusion*: array[TM_CLASSES, array[TM_CLASSES, int]] + ## Same for the radial head. + radConfusion*: array[TM_CLASSES, array[TM_CLASSES, int]] lastChosen*: int shuffleLabels*: bool ## control: replace the computed GF label with a random class forceBase*: bool ## measurement: ignore the TM, emit the pure LinearGun base @@ -504,6 +512,7 @@ proc tmResolveTrace(g: var TmPatternGun, t: TmPatternTrace, power: float) = if t.warm: inc g.classTotal if winner == t.chosen: inc g.classCorrect + inc g.confusion[winner][t.chosen] # Radial label: enemy radius at the base arrival tick minus the base fire # distance. Independent of our own aim, so it is a clean target. @@ -517,6 +526,7 @@ proc tmResolveTrace(g: var TmPatternGun, t: TmPatternTrace, power: float) = if t.warm: inc g.radTotal if radWinner == t.radChosen: inc g.radCorrect + inc g.radConfusion[radWinner][t.radChosen] # Reversal label: net heading turn over the flight, opposite to the direction # the enemy was turning at fire time. diff --git a/common_libs/tests/sweep_tm_pattern.nim b/common_libs/tests/sweep_tm_pattern.nim index 4cbacb6..f01608a 100644 --- a/common_libs/tests/sweep_tm_pattern.nim +++ b/common_libs/tests/sweep_tm_pattern.nim @@ -45,6 +45,8 @@ type labHist, choHist: array[TM_CLASSES, int] classCorrect, classTotal: int radCorrect, radTotal, revCorrect, revTotal: int + confusion: array[TM_CLASSES, array[TM_CLASSES, int]] + radConfusion: array[TM_CLASSES, array[TM_CLASSES, int]] Adapt = object h100, n100, h300, n300, hall, nall, f100, m100: int @@ -103,6 +105,9 @@ proc addAdapt(dst: var Adapt, src: Adapt) = for c in 0.. rowSum[maj]: maj = c + let majShare = if total > 0: rowSum[maj].float / total.float else: 0.0 + let acc = if total > 0: diag.float / total.float else: 0.0 + var hs = "" + for c in 0.. 0: cm[c][c].float / rowSum[c].float else: 0.0 + let prec = if colSum[c] > 0: cm[c][c].float / colSum[c].float else: 0.0 + echo &"# class{c}: trueN={rowSum[c]} predN={colSum[c]} TP={cm[c][c]} " & + &"recall={rec*100:.1f}% precision={prec*100:.1f}%" + proc main() = var nSeeds = 3 var set = "real" @@ -358,6 +397,14 @@ proc main() = for c in 0..=both` + and **all 14 other guns `off`**, including TMPATTERN in the non-TM arms (the + repo `arm_env.sh` was fixed to emit `=both` explicitly so an arm can never fall + back to the FULL rack). +* **7 runs × 7 rounds per arm, 7 concurrent**, distinct ports/output files. +* **Liveness (MEASURED)**: the `[rack] mode=1v1 active=` line is exactly the + target gun in every run (`PATTERN` / `TMPATTERN` / `LINEAR`) and the + selected-gun mix is 100% that gun (Pattern 79401 sel, TMPattern 83478 sel, + Linear 80233 sel). TMPattern fired real bullets every run. + +### Live results — damage/run is PRIMARY + +| arm | runs | shots | hits | real % | **dmg/run** | **round wins** | mean round len | real % range | dmg range | +|---|---|---|---|---|---|---|---|---|---| +| **onlyPattern** | 7 | 4610 | 495 | **10.74%** | **285** | **25/49** | 1625 | 9.31–12.01 | 236–344 | +| onlyTMPATTERN | 7 | 3374 | 118 | 3.50% | 71 | 0/49 | 1743 | 2.22–5.41 | 44–104 | +| onlyLinear | 7 | 3218 | 104 | 3.23% | 61 | 0/49 | 1642 | 1.43–5.14 | 28–106 | + +Per-run values (`run: hits/shots rate, dmg, roundWins/7, meanTurn`): + +* `onlyPattern` — r1 `86/716 12.01% 344 4/7 1772`; r2 `67/620 10.81% 274 4/7 1481`; + r3 `60/534 11.24% 246 0/7 1314`; r4 `73/703 10.38% 292 7/7 1719`; + r5 `59/634 9.31% 236 0/7 1646`; r6 `74/695 10.65% 299 7/7 1692`; + r7 `76/708 10.73% 304 3/7 1753`. +* `onlyTMPATTERN` — r1 `19/437 4.35% 76 0/7 1691`; r2 `14/489 2.86% 68 0/7 1837`; + r3 `21/490 4.29% 84 0/7 1805`; r4 `16/505 3.17% 73 0/7 1681`; + r5 `11/477 2.31% 50 0/7 1713`; r6 `11/495 2.22% 44 0/7 1723`; + r7 `26/481 5.41% 104 0/7 1747`. +* `onlyLinear` — r1 `19/465 4.09% 76 0/7 1781`; r2 `7/491 1.43% 28 0/7 1674`; + r3 `25/486 5.14% 106 0/7 1837`; r4 `7/425 1.65% 34 0/7 1447`; + r5 `14/417 3.36% 56 0/7 1456`; r6 `17/454 3.74% 68 0/7 1638`; + r7 `15/480 3.12% 60 0/7 1664`. + +Exact two-sided permutation test on per-run values (7 vs 7, C(14,7)=3432 exact): + +| A vs B | metric | mean diff (A−B) | exact p | +|---|---|---|---| +| onlyPattern vs onlyTMPATTERN | hit-rate pp | **+7.22** | **0.0006** | +| onlyPattern vs onlyTMPATTERN | dmg | **+213.7** | **0.0006** | +| onlyPattern vs onlyLinear | hit-rate pp | +7.51 | 0.0006 | +| onlyPattern vs onlyLinear | dmg | +223.9 | 0.0006 | +| **onlyTMPATTERN vs onlyLinear** | hit-rate pp | **+0.30** | **0.6591** | +| **onlyTMPATTERN vs onlyLinear** | dmg | **+10.1** | **0.4376** | + +**Task 2 verdict (MEASURED):** Pattern **decisively** beats the TM gun live — +10.74% vs 3.50% hit rate, 285 vs 71 dmg/run, 25 vs **0** round wins of 49, +p=0.0006 on both hit rate and damage, with **non-overlapping** per-run ranges +(Pattern 9.31–12.01 vs TM 2.22–5.41). The TM gun is **statistically +indistinguishable from its own Linear base** (p=0.66 hit rate, p=0.44 damage): +the whole radial TM correction is a live no-op, exactly as Round 3 predicted for +the structural-radial dead end. The TM gun cannot be the best 1v1 gun. + +## DIRECT ANSWER — can the TM gun be the best 1v1 gun? + +**(c) It loses live AND it sits at/below its majority class — the target carries +no learnable conditional signal, and that is the reason.** + +* LIVE: the registered TM (radial) loses decisively to Pattern (3.50% vs 10.74%, + p=0.0006; 0 vs 25 round wins) and is no better than its own Linear base + (p=0.66). It was never the best gun live. +* OFFLINE: the GF head at its honest (raw) readout is 2.1 pp BELOW the 38.8% + majority class; the gated config is only 1.4–1.6 pp above, by predicting the + majority 62% of the time and with 13–14% recall on the two minority classes the + correction would need. The radial head was already measured at/below majority + in Round 3 (56.2% vs 57.2%). +* Therefore the TM machine is not the limiting factor in the sense "it learns and + we aim it wrong" (that would be (b)); the discrete GF/radial targets simply do + not contain conditional information beyond the base rate for these surfers, and + the live result agrees with the offline majority-class analysis. + +## MEASURED vs INFERRED (round 4) + +* **MEASURED**: the GF head confusion matrices, warm label histograms, majority + shares, accuracies, margins, per-class precision/recall, and the shuffled + controls (seeds=1 RAW+GATED, seeds=3 GATED); the fact that RAW is below + majority and GATED is +1.4–1.6 pp above by majority-class over-prediction. +* **MEASURED**: the live table (shots, hits, real %, dmg/run, round wins, mean + round length, per-run ranges, exact permutation p vs onlyPattern and TM vs + Linear), the frozen-binary commit + sha256, and the per-arm `[rack] active=` + liveness lines. +* **MEASURED**: `onlyTMPATTERN` and `onlyLinear` are indistinguishable live + (p=0.66/0.44) — the radial correction changes nothing in real physics. +* **INFERRED**: that the drop from the old 46.0% to the current 36.7%/40.4% is + caused by the deferred-label fix enlarging the sample set (1.26 M → 1.75 M) + with the short-aim samples the head gets wrong; the current numbers are + measured, the causal attribution is not separately ablated. +* **INFERRED (not measured)**: a live GF-mode arm — not run (one frozen HEAD + binary; HEAD registers radial only). Task 1 makes it the unpromising candidate. diff --git a/tools/ab/which_gun_arm_env.sh b/tools/ab/which_gun_arm_env.sh index af552a8..0a45299 100755 --- a/tools/ab/which_gun_arm_env.sh +++ b/tools/ab/which_gun_arm_env.sh @@ -1,13 +1,20 @@ -# emits `TR_RACK_=off ...` for every gun NOT in the arm. -# usage: arm_env.sh +# emits `TR_RACK_=both` for the kept gun and `TR_RACK_=off` for every +# other gun, so the rack can never silently fall back to the FULL-rack +# degradation when the target gun's `=both` was forgotten. +# usage: arm_env.sh arm="$1" -GUNS="HEADON LINEAR TSETLIN CIRCULAR GUESSFACTOR PATTERN WALLBOUNCE ACCEL STOPSHOT DISPLACE AVGLEAD DECAYGF KNN TMSELECT" +GUNS="HEADON LINEAR TSETLIN CIRCULAR GUESSFACTOR PATTERN WALLBOUNCE ACCEL STOPSHOT DISPLACE AVGLEAD DECAYGF KNN TMSELECT TMPATTERN" +if [ "$arm" = "full" ]; then + # Every gun `both` (the process default is onlyPattern, so this must be + # explicit). + keep="$GUNS" +else case "$arm" in - full) keep="";; - onlyLinear) keep="LINEAR";; - onlyPattern) keep="PATTERN";; - onlyGF) keep="GUESSFACTOR";; - onlyKNN) keep="KNN";; + onlyLinear) keep="LINEAR";; + onlyPattern) keep="PATTERN";; + onlyTMPATTERN) keep="TMPATTERN";; + onlyGF) keep="GUESSFACTOR";; + onlyKNN) keep="KNN";; # ── small-good-rack follow-up ────────────────────────────────────────────── lean8) keep="HEADON LINEAR CIRCULAR ACCEL PATTERN GUESSFACTOR KNN WALLBOUNCE";; lean6) keep="LINEAR CIRCULAR ACCEL PATTERN GUESSFACTOR KNN";; @@ -16,11 +23,12 @@ case "$arm" in pairPL) keep="PATTERN LINEAR";; *) echo "unknown arm $arm" >&2; exit 1;; esac +fi out="" for g in $GUNS; do case " $keep " in - *" $g "*) continue;; + *" $g "*) out="$out TR_RACK_${g}=both";; + *) out="$out TR_RACK_${g}=off";; esac - out="$out TR_RACK_${g}=off" done echo "$out" diff --git a/tools/ab/which_gun_run_one.sh b/tools/ab/which_gun_run_one.sh index 275292d..f5a3c27 100755 --- a/tools/ab/which_gun_run_one.sh +++ b/tools/ab/which_gun_run_one.sh @@ -4,7 +4,7 @@ set -u ARM="$1"; RUN="$2"; ROUNDS="${3:-7}" ROOT=/home/davide/Projects/SirRoboGarage -OUT=/tmp/whichgun +OUT="${WHICHGUN_OUT:-/tmp/whichgun}" BOTDIR="$OUT/drussgt_bots/${ARM}_${RUN}/DrussGT" mkdir -p "$BOTDIR" "$OUT/botlog"