TM gun round 2: base was never behind; RADIAL target beats Linear on bmPoint

=== TASK 1: MY PREMISE WAS REFUTED ===
I instructed the job to "fix the baseline" because an earlier measurement said the
TM gun's base did not iterate flight time like `LinearGun`. MEASURED: the new gun's
base is BYTE-FOR-BYTE `LinearGun` - 18/18 runs tie exactly, p=1.000, every per-run
row byte-identical. The "non-iterating baseline" belonged to the OLD `tsetlin.nim`,
not this gun. So no fix was needed, and the earlier inference should not have been
generalised to the new gun. (It did still align the zero-correction clamp to
LinearGun's exact [0, arena] range, and reports the old BotRadius-inset base was a
wash/marginally better at 34.2%/24.7%.)

=== TASK 2: THE RADIAL TARGET - A CONTROL-VALIDATED WIN, BUT ONLY ON bmPoint ===
Instead of the lateral (GF-bucket) component - which the linear lead already
captures - the TM now predicts the RADIAL component: will the enemy be nearer or
farther than the base prediction when our bullet arrives? A 5-class radial head
sharing the same 40-bit context and TM core; the readout advances/retards the aim
distance along the base bearing.

  under bmPath (the SHIPPED metric): STRUCTURAL NO-OP
    synthetic 8/8 exact ties, p=1.0; real 33.9%/24.1% vs Linear 34.0%/24.3%
  under bmPoint: A WIN, control-validated
    TMRadial 9.4% (6013/63785) / 5.8% (42079/726652)
    Linear   7.2% / 4.7%          overall 17/1, p=0.0001
    Tsetlin  7.0% / 4.8%          overall 15/3, p=0.0075
    shuffled 7.0% / 3.6%          early 17/1 p=0.0001; overall 18/0, p<0.0001
  radial head online accuracy 48.8% vs 19.9% shuffled chance and 36.7% majority
  -> it is CONDITIONAL learning, not a constant short-range bias.
Best config: TM_RADIAL_RANGE=60, TM_RAD_MARGIN=0.25, 5 classes.

CAVEAT THAT MATTERS: a win on `bmPoint` is NOT yet evidence of a real win. `bmPath`
is the shipped SELECTION metric precisely because it beat `bmPoint` on real hit
rate (7.43% vs 4.70%). But that A/B was about which gun to PICK, not about gun
QUALITY - a gun can be better in reality while scoring worse on the selection
metric. So this needs a LIVE test, and it is the decisive one.

=== TASK 3: REVERSAL TARGET - CLEAN NEGATIVE ===
The label positive rate is only 9.7% (rev=[24772,2673]) and the head's 86.8%
accuracy is BELOW the 90.3% majority baseline: it does not learn the positive
class at all. Hit-rate effect neutral (bmPath 19.5%/18.4% vs shuffled 19.1%/17.8%,
p=0.24/0.82). Dropped.

=== OVERALL ===
Not competitive on the shipped bmPath metric (gated GF 28.3%/22.2% vs Linear
34.0%/24.3%, p=0.0075). Better than Linear on bmPoint via TMRadial (+2.2pp early,
+1.1pp overall). Per-enemy reset exists; a fresh gun per round; NO cross-battle
persistence (the user's non-negotiable).

MEASURED LIMITATION: radial mode has a high labelMiss because aiming short
resolves BEFORE the base arrival tick, biasing training toward resolvable samples.
The metric win is label-independent. A deferred-label fix is the next refinement.
INFERRED: the mechanism is surfers being NEARER than the base prediction
(range-holding); a constant-short-offset ablation would separate a learned
short-range bias from genuine per-tick conditional prediction.
This commit is contained in:
2026-09-22 01:27:59 +02:00
parent 78975a35c4
commit 1ea72c7f14
3 changed files with 435 additions and 38 deletions
+86 -9
View File
@@ -24,6 +24,17 @@ import gun_harness/offline_range
import range_guns
import guns/tm_pattern
import guns/linear
import guns/lead_forecast
type
LinearInsetGun = object ## pre-fix TMPattern base: same forecast, BotRadius clamp
proc predict*(g: var LinearInsetGun, state: WorldState, bulletSpeed: float): GunPrediction =
let f = forecastLinear(state, bulletSpeed)
GunPrediction(x: clamp(f.x, BotRadius, state.arenaWidth - BotRadius),
y: clamp(f.y, BotRadius, state.arenaHeight - BotRadius))
proc onResult*(g: var LinearInsetGun, e: FeedbackEvent) = discard
const repoRoot = currentSourcePath().parentDir.parentDir.parentDir
const fixturesDir = repoRoot / "tools" / "fixtures"
@@ -33,6 +44,7 @@ type
obs, labelMiss, traceMiss: int
labHist, choHist: array[TM_CLASSES, int]
classCorrect, classTotal: int
radCorrect, radTotal, revCorrect, revTotal: int
Adapt = object
h100, n100, h300, n300, hall, nall, f100, m100: int
@@ -47,14 +59,21 @@ type
r: Adapt
VariantKind = enum
vLinear, vTsetlin, vTmpat, vTmpatShuf
vLinear, vLinearOld, vTsetlin, vTmpat, vTmpatShuf, vTmpatBase,
vTmpatRad, vTmpatRadShuf, vTmpatRev, vTmpatRevShuf
proc variantName(v: VariantKind): string =
case v
of vLinear: "Linear"
of vLinearOld: "LinearOldClamp"
of vTsetlin: "Tsetlin"
of vTmpat: "TMPattern"
of vTmpatShuf: "TMPatternShuf"
of vTmpatBase: "TMPatternBase"
of vTmpatRad: "TMRadial"
of vTmpatRadShuf: "TMRadialShuf"
of vTmpatRev: "TMReversal"
of vTmpatRevShuf: "TMReversalShuf"
proc loadRounds(path: string): seq[RoundSpan] =
let dir = path.parentDir
@@ -77,6 +96,10 @@ proc addAdapt(dst: var Adapt, src: Adapt) =
dst.st.traceMiss += src.st.traceMiss
dst.st.classCorrect += src.st.classCorrect
dst.st.classTotal += src.st.classTotal
dst.st.radCorrect += src.st.radCorrect
dst.st.radTotal += src.st.radTotal
dst.st.revCorrect += src.st.revCorrect
dst.st.revTotal += src.st.revTotal
for c in 0..<TM_CLASSES:
dst.st.labHist[c] += src.st.labHist[c]
dst.st.choHist[c] += src.st.choHist[c]
@@ -128,6 +151,10 @@ proc replayRound(states: seq[WorldState], lastSeen: seq[int], enemyId, baseTick:
res.st.traceMiss = after.traceMiss - obsBefore.traceMiss
res.st.classCorrect = after.classCorrect - obsBefore.classCorrect
res.st.classTotal = after.classTotal - obsBefore.classTotal
res.st.radCorrect = after.radCorrect - obsBefore.radCorrect
res.st.radTotal = after.radTotal - obsBefore.radTotal
res.st.revCorrect = after.revCorrect - obsBefore.revCorrect
res.st.revTotal = after.revTotal - obsBefore.revTotal
for c in 0..<TM_CLASSES:
res.st.labHist[c] = after.labHist[c] - obsBefore.labHist[c]
res.st.choHist[c] = after.choHist[c] - obsBefore.choHist[c]
@@ -157,15 +184,18 @@ proc replayFixture(fx: Fixture, path: string, driver: GunDriver, metric: BulletM
proc emptyStats(): GunStats = GunStats()
proc makeTmpatDriver(seed: int, shuffle: bool):
proc makeTmpatDriver(seed: int, shuffle: bool, forceBase = false,
mode = tmGF):
tuple[driver: GunDriver, gun: ref TmPatternGun] =
let g = new(TmPatternGun)
g[] = initTmPatternGun()
g[].shuffleLabels = shuffle
g[].forceBase = forceBase
g[].targetMode = mode
if seed >= 0: randomize(seed)
result.gun = g
result.driver = GunDriver(
name: (if shuffle: "TMPatternShuf" else: "TMPattern"),
name: (if forceBase: "TMPatternBase" elif shuffle: "TMPatternShuf" else: "TMPattern"),
predictCb: proc(state: WorldState, bulletSpeed: float): GunPrediction =
g[].predict(state, bulletSpeed),
resultCb: proc(e: FeedbackEvent) = g[].onResult(e),
@@ -179,6 +209,10 @@ proc tmpatStats(g: ref TmPatternGun): GunStats =
result.choHist = g[].chosenHist
result.classCorrect = g[].classCorrect
result.classTotal = g[].classTotal
result.radCorrect = g[].radCorrect
result.radTotal = g[].radTotal
result.revCorrect = g[].revCorrect
result.revTotal = g[].revTotal
proc fixtureSet(name: string): seq[string] =
case name
@@ -235,9 +269,15 @@ proc main() =
for tok in a[11..^1].split(','):
case tok.strip()
of "linear": variants.add vLinear
of "linear_old": variants.add vLinearOld
of "tsetlin": variants.add vTsetlin
of "tmpat": variants.add vTmpat
of "tmpat_shuf": variants.add vTmpatShuf
of "tmbase": variants.add vTmpatBase
of "tmrad": variants.add vTmpatRad
of "tmrad_shuf": variants.add vTmpatRadShuf
of "tmrev": variants.add vTmpatRev
of "tmrev_shuf": variants.add vTmpatRevShuf
else: discard
let metric = if metricName == "point": bmPoint else: bmPath
let names = fixtureSet(set)
@@ -245,11 +285,13 @@ proc main() =
echo "variant,fixture,seed,h100,n100,h300,n300,hall,nall,f100,m100,rounds,obs,labelMiss,traceMiss"
var rows: seq[Row]
var revLabTab = initTable[string, array[2, int]]()
var radLabTab = initTable[string, array[TM_CLASSES, int]]()
for name in names:
let (fx, path) = resolve(name)
let fxName = path.extractFilename.replace(".jsonl", "")
for v in variants:
let nIter = if v == vLinear: 1 else: nSeeds
let nIter = if v in [vLinear, vLinearOld, vTmpatBase]: 1 else: nSeeds
for seed in 1..nIter:
var drv: GunDriver
var gun: ref TmPatternGun
@@ -257,16 +299,32 @@ proc main() =
case v
of vLinear:
drv = makeDriver("Linear", LinearGun())
of vLinearOld:
drv = makeDriver("LinearOldClamp", LinearInsetGun())
of vTsetlin:
let pair = makeTsetlinDriver(seed = seed)
drv = pair.driver
of vTmpat, vTmpatShuf:
let pair = makeTmpatDriver(seed = seed, shuffle = (v == vTmpatShuf))
of vTmpat, vTmpatShuf, vTmpatBase, vTmpatRad, vTmpatRadShuf,
vTmpatRev, vTmpatRevShuf:
let mode =
case v
of vTmpatRad, vTmpatRadShuf: tmRadial
of vTmpatRev, vTmpatRevShuf: tmReversal
else: tmGF
let shuf = v in [vTmpatShuf, vTmpatRadShuf, vTmpatRevShuf]
let pair = makeTmpatDriver(seed = seed, shuffle = shuf,
forceBase = (v == vTmpatBase), mode = mode)
drv = pair.driver
gun = pair.gun
obsCount = proc(): GunStats = tmpatStats(gun)
let r = replayFixture(fx, path, drv, metric, emptyStats(), obsCount, maxRounds)
rows.add Row(variant: variantName(v), fixture: fxName, seed: seed, r: r)
if gun != nil:
if variantName(v) notin revLabTab: revLabTab[variantName(v)] = [0, 0]
if variantName(v) notin radLabTab:
radLabTab[variantName(v)] = default(array[TM_CLASSES, int])
for c in 0..<2: revLabTab[variantName(v)][c] += gun[].revLabelHist[c]
for c in 0..<TM_CLASSES: radLabTab[variantName(v)][c] += gun[].radLabelHist[c]
echo &"{variantName(v)},{fxName},{seed},{r.h100},{r.n100},{r.h300},{r.n300}," &
&"{r.hall},{r.nall},{r.f100},{r.m100},{r.rounds},{r.st.obs},{r.st.labelMiss},{r.st.traceMiss}"
@@ -282,8 +340,8 @@ proc main() =
&"{a.h300},{a.n300},{rateStr(a.h300, a.n300)},{a.hall},{a.nall}," &
&"{rateStr(a.hall, a.nall)},{a.st.obs},{a.st.labelMiss},{a.st.traceMiss}"
# ── label vs chosen class histogram (TMPattern only) ──
for v in [vTmpat, vTmpatShuf]:
# ── label vs chosen class histogram (all TM variants) ──
for v in [vTmpat, vTmpatShuf, vTmpatRad, vTmpatRadShuf, vTmpatRev, vTmpatRevShuf]:
if v in variants:
let a = pooled[variantName(v)]
var ls, cs: string
@@ -291,7 +349,14 @@ proc main() =
ls.add &"{a.st.labHist[c]},"
cs.add &"{a.st.choHist[c]},"
echo &"\n# class histogram {variantName(v)}: labels=[{ls}] chosen=[{cs}] " &
&"onlineAcc={a.st.classCorrect}/{a.st.classTotal}"
&"onlineAcc={a.st.classCorrect}/{a.st.classTotal} " &
&"radAcc={a.st.radCorrect}/{a.st.radTotal} " &
&"revAcc={a.st.revCorrect}/{a.st.revTotal}"
if variantName(v) in revLabTab:
var rl, rdl: string
for c in 0..<2: rl.add &"{revLabTab[variantName(v)][c]},"
for c in 0..<TM_CLASSES: rdl.add &"{radLabTab[variantName(v)][c]},"
echo &"# label hist {variantName(v)}: radial=[{rdl}] rev=[{rl}]"
# ── per-run distributions (a "run" = one fixture × one seed) ──
# Linear is deterministic: replicate its one row per fixture across seeds so a
@@ -313,6 +378,18 @@ proc main() =
for s in 2..nSeeds:
byVariant["Linear"][(row.fixture, s)] = byVariant["Linear"][(row.fixture, 1)]
byVariantAll["Linear"][(row.fixture, s)] = byVariantAll["Linear"][(row.fixture, 1)]
if vLinearOld in variants:
for row in rows:
if row.variant == "LinearOldClamp":
for s in 2..nSeeds:
byVariant["LinearOldClamp"][(row.fixture, s)] = byVariant["LinearOldClamp"][(row.fixture, 1)]
byVariantAll["LinearOldClamp"][(row.fixture, s)] = byVariantAll["LinearOldClamp"][(row.fixture, 1)]
if vTmpatBase in variants:
for row in rows:
if row.variant == "TMPatternBase":
for s in 2..nSeeds:
byVariant["TMPatternBase"][(row.fixture, s)] = byVariant["TMPatternBase"][(row.fixture, 1)]
byVariantAll["TMPatternBase"][(row.fixture, s)] = byVariantAll["TMPatternBase"][(row.fixture, 1)]
echo "\n# ── per-run early-rate distribution (mean / min / max, n runs) ──"
echo "variant,earlyMean%,earlyMin%,earlyMax%,overallMean%,overallMin%,overallMax%,n"