TM radial gun: registered (default OFF) + label-bias fix that removes the bias but

retracts its own earlier learning claim

=== TASK 1: REGISTERED AS GUN 14, DEFAULT `off` ===
The radial TM gun is now a first-class rack member (`TMPATTERN`, id 14), forceable
alone with `TR_RACK_TMPATTERN=both` plus every other `TR_RACK_*=off`.
DEFAULT IS `off`, and the justification matters: `both` would let it compete for
selection AND (because the shared VirtualTracker ring is order-sensitive) shift
every other gun's learning order, so it CANNOT leave the default path unchanged.
With `off` its predict and spawnBullets are additionally GATED on rack admission
(the only gun wired that way), so the shipped default never spawns it at all:
zero cost, zero ring perturbation.
Live proof: 1-round battle with only TMPATTERN racked ->
  `gun 14 (TMPattern): vShots=400 selected=104 other-gun selections=0`.
Default-path-unchanged proof: parity checks that the 15-gun default bestGun/
selectGun equals the old 14-gun rack RNG-draw-for-RNG-draw, that gun 14 is never
selected by default, and acceptance 12/12.
Cost: 0.36 ms/tick (predict 0.30 + onResult 0.05) ~= 3% of the 13.16 ms budget.
Tsetlin in the same harness is 1.62 ms/tick, so the new gun is ~4.5x cheaper.

=== TASK 2: THE LABEL-BIAS FIX - AND A RETRACTION ===
Root cause confirmed: under bmPoint a SHORT radial correction resolves the virtual
bullet BEFORE the base arrival tick, so the label was dropped (labelMisses).
Fix: defer the label in a pending queue and flush it once the arrival tick is
recorded; labels still come from the BASE arrival tick.
  labelMisses        4,281,695  ->  0
  training samples   1,071,824  ->  5,345,847  (x5)
  radial head acc         48.8% ->  57.0%   (shuffled control 20.0%)
  bmPoint hit rate     9.4/5.8% ->  9.1/5.7%  (unchanged, within noise)
So the fix IMPROVES LEARNING but NOT the metric.

**RETRACTION OF THE PREVIOUS JOB'S CLAIM.** It reported the radial head's 48.8%
against a 36.7% majority baseline and concluded "conditional learning, not a
constant bias". With the bias removed, the correctly-measured majority baseline is
**58.2%** - so the head at 57.0% is AT/BELOW majority. The earlier apparent
conditional learning was PARTLY AN ARTEFACT OF THE BIASED SAMPLE. The bmPoint
metric win is real (TMRadial > Linear early 16/2 p=0.0013, overall 18/0 p<0.0001;
> shuffled 18/0 p<0.0001) but it comes from a NET-POSITIVE AVERAGE RADIAL SHIFT,
not from beating a majority classifier. Recorded plainly rather than left standing.

Guards: test_tm_pattern_registration 20 (new), test_tm_pattern_rack_live 4 (new),
test_gun_harness 39, test_vbullet_metric 11, test_power_selection 3 (the SIGSEGV is
gone - the knn_gun rewrite is now committed), test_adaptive_radar 41,
test_tfil_ring_weights 24, test_power_policy 26, test_ram_decision 28,
test_rack_membership 38, test_selector_tiebreak 19, test_tm_pattern_learning 3,
acceptance_offline_vs_online 12/12. ModularBot compiles (release).

Note: `common_libs/tests/range_guns.nim` still builds 14 offline drivers (the
offline sweep constructs TmPatternGun directly and acceptance only inspects ids
0..13), so nothing breaks - but a future job wanting it in the offline rack must
add a 15th driver and mirror the live admission gating. gun_stats.jsonl now emits
15 rows; downstream tooling should ignore id 14.
This commit is contained in:
2026-09-22 01:58:33 +02:00
parent 394b3deeed
commit 589a230106
7 changed files with 675 additions and 89 deletions
+133 -65
View File
@@ -70,6 +70,13 @@ const
TM_SOFT_BETA* = parseFloat(TM_SOFT_BETA_DEF)
TM_TRACE_SLOTS = 1024
POS_RING = 512
## Deferred-label queue (Task 2): a virtual bullet whose radial correction
## aimed SHORT resolves BEFORE its BASE arrival tick, when the arrival-tick
## position is not yet in `posRing`. Instead of dropping the sample
## (`labelMisses`), the trace is copied here and resolved on the first later
## `predict` tick at which the base arrival tick's position exists, so every
## fired virtual bullet contributes an unbiased training sample.
TM_PENDING_SLOTS = 1024
DebugTMPattern* = false
## ── radial head (Task 2) ────────────────────────────────────────────────
## Radial label = (enemy radius at the BASE arrival tick) - (base fire
@@ -125,12 +132,26 @@ type
heading: float
valid: bool
PendingResolve = object
## A fired virtual bullet whose label was not yet resolvable at resolution
## time. `trace` is a COPY of the fire-time trace (features + clause
## caches), so the deferred training update is identical to an immediate
## one, just later.
arrivalTick: int
powerBin: int
power: float
trace: TmPatternTrace
TmPatternGun* = object
teams: array[TM_CLASSES, seq[int16]]
radTeams: array[TM_CLASSES, seq[int16]]
revTeams: array[2, seq[int16]]
targetMode*: TmTargetMode
traces: array[TM_TRACE_SLOTS, TmPatternTrace]
# ── deferred labels (Task 2) ──
pending: array[TM_PENDING_SLOTS, PendingResolve]
pendingCount: int
pendingDropped*: int
# ── history ──
posRing: array[POS_RING, PosSample]
lastTick: int
@@ -268,6 +289,14 @@ proc initTmPatternGun*(): TmPatternGun =
randomize()
result.debugGraphics = false
proc initTmRadialGun*(): TmPatternGun =
## The RACK-REGISTERED instance: the RADIAL target mode, which is the
## control-validated winner under `bmPoint` (see tm_pattern_sweep_results.md,
## Round 2 Task 2). The gun type carries all three heads; the live rack only
## ever selects this radial-mode instance.
result = initTmPatternGun()
result.targetMode = tmRadial
proc isWarmedUp*(g: TmPatternGun): bool {.inline.} = true
proc resetLearning*(g: var TmPatternGun) =
@@ -284,6 +313,7 @@ proc resetLearning*(g: var TmPatternGun) =
g.sinceReversal = 0
g.radialFracSm = 0.0
g.latPersist = 0
g.pendingCount = 0
proc tmUpdateHistory(g: var TmPatternGun, state: WorldState) =
if state.tick == g.lastTick: return
@@ -442,6 +472,90 @@ proc tmChooseAt(votes: openArray[float], centre: int, margin: float,
proc tmChooseClass(g: var TmPatternGun, votes: array[TM_CLASSES, float]): int =
tmChooseAt(votes, (TM_CLASSES - 1) div 2, TM_CONF_MARGIN, g.totalObs)
proc tmResolveTrace(g: var TmPatternGun, t: TmPatternTrace, power: float) =
## One label + one TM update for a fired virtual bullet, using the enemy
## position recorded at the BASE arrival tick. `t` is a value copy of the
## fire-time trace, so this is safe to call either from `onResult` (the label
## is already resolvable) or from `tmFlushPending` (the label was deferred
## because the bullet resolved before its base arrival tick).
let s = ((t.arrivalTick mod POS_RING) + POS_RING) mod POS_RING
let speed = bulletSpeed(power)
let mea = arcsin(clamp(8.0 / speed, -1.0, 1.0))
let actualBearing = arctan2(g.posRing[s].y - t.fireY, g.posRing[s].x - t.fireX)
var delta = actualBearing - t.baseBearing
while delta > PI: delta -= 2.0 * PI
while delta < -PI: delta += 2.0 * PI
let gf = if mea > 1e-10: clamp(delta / mea, -1.0, 1.0) else: 0.0
# The shuffled control randomises ONLY the head the current mode is claiming.
let shuffleGF = g.shuffleLabels and g.targetMode == tmGF
let shuffleRad = g.shuffleLabels and g.targetMode == tmRadial
let shuffleRev = g.shuffleLabels and g.targetMode == tmReversal
let winner = if shuffleGF: rand(TM_CLASSES - 1) else: gfToBucket(gf)
inc g.labelHist[winner]
if t.warm:
inc g.classTotal
if winner == t.chosen: inc g.classCorrect
# Radial label: enemy radius at the base arrival tick minus the base fire
# distance. Independent of our own aim, so it is a clean target.
let actualRadius = hypot(g.posRing[s].x - t.fireX, g.posRing[s].y - t.fireY)
let radDelta = actualRadius - t.fireDist
let radWinner = if shuffleRad: rand(TM_CLASSES - 1) else: radToBucket(radDelta)
inc g.radLabelHist[radWinner]
if t.warm:
inc g.radTotal
if radWinner == t.radChosen: inc g.radCorrect
# Reversal label: net heading turn over the flight, opposite to the direction
# the enemy was turning at fire time.
let dh = normDeg(g.posRing[s].heading - t.fireHeading)
let netTurn = if dh > TM_REV_TURN_DEG: 1 elif dh < -TM_REV_TURN_DEG: -1 else: 0
let revWinner =
if shuffleRev: rand(1)
elif t.fireTurn != 0 and netTurn != 0 and netTurn != t.fireTurn: 1
else: 0
inc g.revLabelHist[revWinner]
if t.warm:
inc g.revTotal
if revWinner == t.revChosen: inc g.revCorrect
case g.targetMode
of tmGF:
for c in 0..<TM_CLASSES:
let d = if c == winner: 1.0 else: -1.0
g.teams[c].tmLearnDir(t.lits, t.cache[c], t.votes[c], d)
of tmRadial:
for c in 0..<TM_CLASSES:
let d = if c == radWinner: 1.0 else: -1.0
g.radTeams[c].tmLearnDir(t.lits, t.radCache[c], t.radVotes[c], d)
of tmReversal:
for c in 0..<TM_CLASSES:
let d = if c == winner: 1.0 else: -1.0
g.teams[c].tmLearnDir(t.lits, t.cache[c], t.votes[c], d)
for c in 0..<2:
let d = if c == revWinner: 1.0 else: -1.0
g.revTeams[c].tmLearnDir(t.lits, t.revCache[c], t.revVotes[c], d)
inc g.totalObs
inc g.trainCalls
proc tmFlushPending(g: var TmPatternGun) =
## Resolve every deferred trace whose BASE arrival tick is now recorded in
## `posRing`. Called once per `predict` right after `tmUpdateHistory`, so the
## just-written current tick is visible. Entries are compacted in place.
if g.pendingCount == 0: return
var w = 0
for i in 0..<g.pendingCount:
let p = addr g.pending[i]
let s = ((p.arrivalTick mod POS_RING) + POS_RING) mod POS_RING
if g.posRing[s].valid and g.posRing[s].tick == p.arrivalTick:
g.tmResolveTrace(p.trace, p.power)
else:
if w != i: g.pending[w] = g.pending[i]
inc w
g.pendingCount = w
proc predict*(g: var TmPatternGun, state: WorldState, bulletSpeed: float):
GunPrediction =
inc g.predictCalls
@@ -454,6 +568,9 @@ proc predict*(g: var TmPatternGun, state: WorldState, bulletSpeed: float):
g.currentTarget = tid
g.tmUpdateHistory(state)
# Deferred-label flush (Task 2): resolve any fired bullet whose BASE arrival
# tick is now in the ring, before the cold-start gate reads totalObs.
g.tmFlushPending()
if bulletSpeed <= 0.0:
return GunPrediction(x: state.enemyX, y: state.enemyY)
@@ -560,69 +677,20 @@ proc onResult*(g: var TmPatternGun, e: FeedbackEvent) =
# Clean label: enemy position at the BASE arrival tick from our own history.
let s = ((t.arrivalTick mod POS_RING) + POS_RING) mod POS_RING
if not g.posRing[s].valid or g.posRing[s].tick != t.arrivalTick:
inc g.labelMisses
t.alive = false
return
let speed = bulletSpeed(e.bulletPower)
let mea = arcsin(clamp(8.0 / speed, -1.0, 1.0))
let actualBearing = arctan2(g.posRing[s].y - t.fireY, g.posRing[s].x - t.fireX)
var delta = actualBearing - t.baseBearing
while delta > PI: delta -= 2.0 * PI
while delta < -PI: delta += 2.0 * PI
let gf = if mea > 1e-10: clamp(delta / mea, -1.0, 1.0) else: 0.0
# The shuffled control randomises ONLY the head the current mode is claiming.
let shuffleGF = g.shuffleLabels and g.targetMode == tmGF
let shuffleRad = g.shuffleLabels and g.targetMode == tmRadial
let shuffleRev = g.shuffleLabels and g.targetMode == tmReversal
let winner = if shuffleGF: rand(TM_CLASSES - 1) else: gfToBucket(gf)
inc g.labelHist[winner]
if t.warm:
inc g.classTotal
if winner == t.chosen: inc g.classCorrect
# Radial label: enemy radius at the base arrival tick minus the base fire
# distance. Independent of our own aim, so it is a clean target.
let actualRadius = hypot(g.posRing[s].x - t.fireX, g.posRing[s].y - t.fireY)
let radDelta = actualRadius - t.fireDist
let radWinner = if shuffleRad: rand(TM_CLASSES - 1) else: radToBucket(radDelta)
inc g.radLabelHist[radWinner]
if t.warm:
inc g.radTotal
if radWinner == t.radChosen: inc g.radCorrect
# Reversal label: net heading turn over the flight, opposite to the direction
# the enemy was turning at fire time.
let dh = normDeg(g.posRing[s].heading - t.fireHeading)
let netTurn = if dh > TM_REV_TURN_DEG: 1 elif dh < -TM_REV_TURN_DEG: -1 else: 0
let revWinner =
if shuffleRev: rand(1)
elif t.fireTurn != 0 and netTurn != 0 and netTurn != t.fireTurn: 1
else: 0
inc g.revLabelHist[revWinner]
if t.warm:
inc g.revTotal
if revWinner == t.revChosen: inc g.revCorrect
case g.targetMode
of tmGF:
for c in 0..<TM_CLASSES:
let d = if c == winner: 1.0 else: -1.0
g.teams[c].tmLearnDir(t.lits, t.cache[c], t.votes[c], d)
of tmRadial:
for c in 0..<TM_CLASSES:
let d = if c == radWinner: 1.0 else: -1.0
g.radTeams[c].tmLearnDir(t.lits, t.radCache[c], t.radVotes[c], d)
of tmReversal:
for c in 0..<TM_CLASSES:
let d = if c == winner: 1.0 else: -1.0
g.teams[c].tmLearnDir(t.lits, t.cache[c], t.votes[c], d)
for c in 0..<2:
let d = if c == revWinner: 1.0 else: -1.0
g.revTeams[c].tmLearnDir(t.lits, t.revCache[c], t.revVotes[c], d)
inc g.totalObs
inc g.trainCalls
if g.posRing[s].valid and g.posRing[s].tick == t.arrivalTick:
g.tmResolveTrace(t[], e.bulletPower)
else:
# DEFER (Task 2): the bullet resolved BEFORE its BASE arrival tick, which
# happens whenever the radial correction aimed SHORT. The arrival-tick
# position is not recorded yet, so keep a COPY of the trace and train on it
# once that tick is in the ring (`tmFlushPending`). Dropping it here is what
# biased the training set toward only the resolvable (long/centre) aims.
if g.pendingCount < TM_PENDING_SLOTS:
g.pending[g.pendingCount] = PendingResolve(
arrivalTick: t.arrivalTick, powerBin: binIdx,
power: e.bulletPower, trace: t[])
inc g.pendingCount
else:
inc g.pendingDropped
inc g.labelMisses
t.alive = false