TM radial gun: registered (default OFF) + label-bias fix that removes the bias but
retracts its own earlier learning claim === TASK 1: REGISTERED AS GUN 14, DEFAULT `off` === The radial TM gun is now a first-class rack member (`TMPATTERN`, id 14), forceable alone with `TR_RACK_TMPATTERN=both` plus every other `TR_RACK_*=off`. DEFAULT IS `off`, and the justification matters: `both` would let it compete for selection AND (because the shared VirtualTracker ring is order-sensitive) shift every other gun's learning order, so it CANNOT leave the default path unchanged. With `off` its predict and spawnBullets are additionally GATED on rack admission (the only gun wired that way), so the shipped default never spawns it at all: zero cost, zero ring perturbation. Live proof: 1-round battle with only TMPATTERN racked -> `gun 14 (TMPattern): vShots=400 selected=104 other-gun selections=0`. Default-path-unchanged proof: parity checks that the 15-gun default bestGun/ selectGun equals the old 14-gun rack RNG-draw-for-RNG-draw, that gun 14 is never selected by default, and acceptance 12/12. Cost: 0.36 ms/tick (predict 0.30 + onResult 0.05) ~= 3% of the 13.16 ms budget. Tsetlin in the same harness is 1.62 ms/tick, so the new gun is ~4.5x cheaper. === TASK 2: THE LABEL-BIAS FIX - AND A RETRACTION === Root cause confirmed: under bmPoint a SHORT radial correction resolves the virtual bullet BEFORE the base arrival tick, so the label was dropped (labelMisses). Fix: defer the label in a pending queue and flush it once the arrival tick is recorded; labels still come from the BASE arrival tick. labelMisses 4,281,695 -> 0 training samples 1,071,824 -> 5,345,847 (x5) radial head acc 48.8% -> 57.0% (shuffled control 20.0%) bmPoint hit rate 9.4/5.8% -> 9.1/5.7% (unchanged, within noise) So the fix IMPROVES LEARNING but NOT the metric. **RETRACTION OF THE PREVIOUS JOB'S CLAIM.** It reported the radial head's 48.8% against a 36.7% majority baseline and concluded "conditional learning, not a constant bias". With the bias removed, the correctly-measured majority baseline is **58.2%** - so the head at 57.0% is AT/BELOW majority. The earlier apparent conditional learning was PARTLY AN ARTEFACT OF THE BIASED SAMPLE. The bmPoint metric win is real (TMRadial > Linear early 16/2 p=0.0013, overall 18/0 p<0.0001; > shuffled 18/0 p<0.0001) but it comes from a NET-POSITIVE AVERAGE RADIAL SHIFT, not from beating a majority classifier. Recorded plainly rather than left standing. Guards: test_tm_pattern_registration 20 (new), test_tm_pattern_rack_live 4 (new), test_gun_harness 39, test_vbullet_metric 11, test_power_selection 3 (the SIGSEGV is gone - the knn_gun rewrite is now committed), test_adaptive_radar 41, test_tfil_ring_weights 24, test_power_policy 26, test_ram_decision 28, test_rack_membership 38, test_selector_tiebreak 19, test_tm_pattern_learning 3, acceptance_offline_vs_online 12/12. ModularBot compiles (release). Note: `common_libs/tests/range_guns.nim` still builds 14 offline drivers (the offline sweep constructs TmPatternGun directly and acceptance only inspects ids 0..13), so nothing breaks - but a future job wanting it in the offline rack must add a 15th driver and mirror the live admission gating. gun_stats.jsonl now emits 15 rows; downstream tooling should ignore id 14.
This commit is contained in:
+133
-65
@@ -70,6 +70,13 @@ const
|
||||
TM_SOFT_BETA* = parseFloat(TM_SOFT_BETA_DEF)
|
||||
TM_TRACE_SLOTS = 1024
|
||||
POS_RING = 512
|
||||
## Deferred-label queue (Task 2): a virtual bullet whose radial correction
|
||||
## aimed SHORT resolves BEFORE its BASE arrival tick, when the arrival-tick
|
||||
## position is not yet in `posRing`. Instead of dropping the sample
|
||||
## (`labelMisses`), the trace is copied here and resolved on the first later
|
||||
## `predict` tick at which the base arrival tick's position exists, so every
|
||||
## fired virtual bullet contributes an unbiased training sample.
|
||||
TM_PENDING_SLOTS = 1024
|
||||
DebugTMPattern* = false
|
||||
## ── radial head (Task 2) ────────────────────────────────────────────────
|
||||
## Radial label = (enemy radius at the BASE arrival tick) - (base fire
|
||||
@@ -125,12 +132,26 @@ type
|
||||
heading: float
|
||||
valid: bool
|
||||
|
||||
PendingResolve = object
|
||||
## A fired virtual bullet whose label was not yet resolvable at resolution
|
||||
## time. `trace` is a COPY of the fire-time trace (features + clause
|
||||
## caches), so the deferred training update is identical to an immediate
|
||||
## one, just later.
|
||||
arrivalTick: int
|
||||
powerBin: int
|
||||
power: float
|
||||
trace: TmPatternTrace
|
||||
|
||||
TmPatternGun* = object
|
||||
teams: array[TM_CLASSES, seq[int16]]
|
||||
radTeams: array[TM_CLASSES, seq[int16]]
|
||||
revTeams: array[2, seq[int16]]
|
||||
targetMode*: TmTargetMode
|
||||
traces: array[TM_TRACE_SLOTS, TmPatternTrace]
|
||||
# ── deferred labels (Task 2) ──
|
||||
pending: array[TM_PENDING_SLOTS, PendingResolve]
|
||||
pendingCount: int
|
||||
pendingDropped*: int
|
||||
# ── history ──
|
||||
posRing: array[POS_RING, PosSample]
|
||||
lastTick: int
|
||||
@@ -268,6 +289,14 @@ proc initTmPatternGun*(): TmPatternGun =
|
||||
randomize()
|
||||
result.debugGraphics = false
|
||||
|
||||
proc initTmRadialGun*(): TmPatternGun =
|
||||
## The RACK-REGISTERED instance: the RADIAL target mode, which is the
|
||||
## control-validated winner under `bmPoint` (see tm_pattern_sweep_results.md,
|
||||
## Round 2 Task 2). The gun type carries all three heads; the live rack only
|
||||
## ever selects this radial-mode instance.
|
||||
result = initTmPatternGun()
|
||||
result.targetMode = tmRadial
|
||||
|
||||
proc isWarmedUp*(g: TmPatternGun): bool {.inline.} = true
|
||||
|
||||
proc resetLearning*(g: var TmPatternGun) =
|
||||
@@ -284,6 +313,7 @@ proc resetLearning*(g: var TmPatternGun) =
|
||||
g.sinceReversal = 0
|
||||
g.radialFracSm = 0.0
|
||||
g.latPersist = 0
|
||||
g.pendingCount = 0
|
||||
|
||||
proc tmUpdateHistory(g: var TmPatternGun, state: WorldState) =
|
||||
if state.tick == g.lastTick: return
|
||||
@@ -442,6 +472,90 @@ proc tmChooseAt(votes: openArray[float], centre: int, margin: float,
|
||||
proc tmChooseClass(g: var TmPatternGun, votes: array[TM_CLASSES, float]): int =
|
||||
tmChooseAt(votes, (TM_CLASSES - 1) div 2, TM_CONF_MARGIN, g.totalObs)
|
||||
|
||||
proc tmResolveTrace(g: var TmPatternGun, t: TmPatternTrace, power: float) =
|
||||
## One label + one TM update for a fired virtual bullet, using the enemy
|
||||
## position recorded at the BASE arrival tick. `t` is a value copy of the
|
||||
## fire-time trace, so this is safe to call either from `onResult` (the label
|
||||
## is already resolvable) or from `tmFlushPending` (the label was deferred
|
||||
## because the bullet resolved before its base arrival tick).
|
||||
let s = ((t.arrivalTick mod POS_RING) + POS_RING) mod POS_RING
|
||||
let speed = bulletSpeed(power)
|
||||
let mea = arcsin(clamp(8.0 / speed, -1.0, 1.0))
|
||||
let actualBearing = arctan2(g.posRing[s].y - t.fireY, g.posRing[s].x - t.fireX)
|
||||
var delta = actualBearing - t.baseBearing
|
||||
while delta > PI: delta -= 2.0 * PI
|
||||
while delta < -PI: delta += 2.0 * PI
|
||||
let gf = if mea > 1e-10: clamp(delta / mea, -1.0, 1.0) else: 0.0
|
||||
|
||||
# The shuffled control randomises ONLY the head the current mode is claiming.
|
||||
let shuffleGF = g.shuffleLabels and g.targetMode == tmGF
|
||||
let shuffleRad = g.shuffleLabels and g.targetMode == tmRadial
|
||||
let shuffleRev = g.shuffleLabels and g.targetMode == tmReversal
|
||||
|
||||
let winner = if shuffleGF: rand(TM_CLASSES - 1) else: gfToBucket(gf)
|
||||
inc g.labelHist[winner]
|
||||
if t.warm:
|
||||
inc g.classTotal
|
||||
if winner == t.chosen: inc g.classCorrect
|
||||
|
||||
# Radial label: enemy radius at the base arrival tick minus the base fire
|
||||
# distance. Independent of our own aim, so it is a clean target.
|
||||
let actualRadius = hypot(g.posRing[s].x - t.fireX, g.posRing[s].y - t.fireY)
|
||||
let radDelta = actualRadius - t.fireDist
|
||||
let radWinner = if shuffleRad: rand(TM_CLASSES - 1) else: radToBucket(radDelta)
|
||||
inc g.radLabelHist[radWinner]
|
||||
if t.warm:
|
||||
inc g.radTotal
|
||||
if radWinner == t.radChosen: inc g.radCorrect
|
||||
|
||||
# Reversal label: net heading turn over the flight, opposite to the direction
|
||||
# the enemy was turning at fire time.
|
||||
let dh = normDeg(g.posRing[s].heading - t.fireHeading)
|
||||
let netTurn = if dh > TM_REV_TURN_DEG: 1 elif dh < -TM_REV_TURN_DEG: -1 else: 0
|
||||
let revWinner =
|
||||
if shuffleRev: rand(1)
|
||||
elif t.fireTurn != 0 and netTurn != 0 and netTurn != t.fireTurn: 1
|
||||
else: 0
|
||||
inc g.revLabelHist[revWinner]
|
||||
if t.warm:
|
||||
inc g.revTotal
|
||||
if revWinner == t.revChosen: inc g.revCorrect
|
||||
|
||||
case g.targetMode
|
||||
of tmGF:
|
||||
for c in 0..<TM_CLASSES:
|
||||
let d = if c == winner: 1.0 else: -1.0
|
||||
g.teams[c].tmLearnDir(t.lits, t.cache[c], t.votes[c], d)
|
||||
of tmRadial:
|
||||
for c in 0..<TM_CLASSES:
|
||||
let d = if c == radWinner: 1.0 else: -1.0
|
||||
g.radTeams[c].tmLearnDir(t.lits, t.radCache[c], t.radVotes[c], d)
|
||||
of tmReversal:
|
||||
for c in 0..<TM_CLASSES:
|
||||
let d = if c == winner: 1.0 else: -1.0
|
||||
g.teams[c].tmLearnDir(t.lits, t.cache[c], t.votes[c], d)
|
||||
for c in 0..<2:
|
||||
let d = if c == revWinner: 1.0 else: -1.0
|
||||
g.revTeams[c].tmLearnDir(t.lits, t.revCache[c], t.revVotes[c], d)
|
||||
inc g.totalObs
|
||||
inc g.trainCalls
|
||||
|
||||
proc tmFlushPending(g: var TmPatternGun) =
|
||||
## Resolve every deferred trace whose BASE arrival tick is now recorded in
|
||||
## `posRing`. Called once per `predict` right after `tmUpdateHistory`, so the
|
||||
## just-written current tick is visible. Entries are compacted in place.
|
||||
if g.pendingCount == 0: return
|
||||
var w = 0
|
||||
for i in 0..<g.pendingCount:
|
||||
let p = addr g.pending[i]
|
||||
let s = ((p.arrivalTick mod POS_RING) + POS_RING) mod POS_RING
|
||||
if g.posRing[s].valid and g.posRing[s].tick == p.arrivalTick:
|
||||
g.tmResolveTrace(p.trace, p.power)
|
||||
else:
|
||||
if w != i: g.pending[w] = g.pending[i]
|
||||
inc w
|
||||
g.pendingCount = w
|
||||
|
||||
proc predict*(g: var TmPatternGun, state: WorldState, bulletSpeed: float):
|
||||
GunPrediction =
|
||||
inc g.predictCalls
|
||||
@@ -454,6 +568,9 @@ proc predict*(g: var TmPatternGun, state: WorldState, bulletSpeed: float):
|
||||
g.currentTarget = tid
|
||||
|
||||
g.tmUpdateHistory(state)
|
||||
# Deferred-label flush (Task 2): resolve any fired bullet whose BASE arrival
|
||||
# tick is now in the ring, before the cold-start gate reads totalObs.
|
||||
g.tmFlushPending()
|
||||
|
||||
if bulletSpeed <= 0.0:
|
||||
return GunPrediction(x: state.enemyX, y: state.enemyY)
|
||||
@@ -560,69 +677,20 @@ proc onResult*(g: var TmPatternGun, e: FeedbackEvent) =
|
||||
|
||||
# Clean label: enemy position at the BASE arrival tick from our own history.
|
||||
let s = ((t.arrivalTick mod POS_RING) + POS_RING) mod POS_RING
|
||||
if not g.posRing[s].valid or g.posRing[s].tick != t.arrivalTick:
|
||||
inc g.labelMisses
|
||||
t.alive = false
|
||||
return
|
||||
|
||||
let speed = bulletSpeed(e.bulletPower)
|
||||
let mea = arcsin(clamp(8.0 / speed, -1.0, 1.0))
|
||||
let actualBearing = arctan2(g.posRing[s].y - t.fireY, g.posRing[s].x - t.fireX)
|
||||
var delta = actualBearing - t.baseBearing
|
||||
while delta > PI: delta -= 2.0 * PI
|
||||
while delta < -PI: delta += 2.0 * PI
|
||||
let gf = if mea > 1e-10: clamp(delta / mea, -1.0, 1.0) else: 0.0
|
||||
|
||||
# The shuffled control randomises ONLY the head the current mode is claiming.
|
||||
let shuffleGF = g.shuffleLabels and g.targetMode == tmGF
|
||||
let shuffleRad = g.shuffleLabels and g.targetMode == tmRadial
|
||||
let shuffleRev = g.shuffleLabels and g.targetMode == tmReversal
|
||||
|
||||
let winner = if shuffleGF: rand(TM_CLASSES - 1) else: gfToBucket(gf)
|
||||
inc g.labelHist[winner]
|
||||
if t.warm:
|
||||
inc g.classTotal
|
||||
if winner == t.chosen: inc g.classCorrect
|
||||
|
||||
# Radial label: enemy radius at the base arrival tick minus the base fire
|
||||
# distance. Independent of our own aim, so it is a clean target.
|
||||
let actualRadius = hypot(g.posRing[s].x - t.fireX, g.posRing[s].y - t.fireY)
|
||||
let radDelta = actualRadius - t.fireDist
|
||||
let radWinner = if shuffleRad: rand(TM_CLASSES - 1) else: radToBucket(radDelta)
|
||||
inc g.radLabelHist[radWinner]
|
||||
if t.warm:
|
||||
inc g.radTotal
|
||||
if radWinner == t.radChosen: inc g.radCorrect
|
||||
|
||||
# Reversal label: net heading turn over the flight, opposite to the direction
|
||||
# the enemy was turning at fire time.
|
||||
let dh = normDeg(g.posRing[s].heading - t.fireHeading)
|
||||
let netTurn = if dh > TM_REV_TURN_DEG: 1 elif dh < -TM_REV_TURN_DEG: -1 else: 0
|
||||
let revWinner =
|
||||
if shuffleRev: rand(1)
|
||||
elif t.fireTurn != 0 and netTurn != 0 and netTurn != t.fireTurn: 1
|
||||
else: 0
|
||||
inc g.revLabelHist[revWinner]
|
||||
if t.warm:
|
||||
inc g.revTotal
|
||||
if revWinner == t.revChosen: inc g.revCorrect
|
||||
|
||||
case g.targetMode
|
||||
of tmGF:
|
||||
for c in 0..<TM_CLASSES:
|
||||
let d = if c == winner: 1.0 else: -1.0
|
||||
g.teams[c].tmLearnDir(t.lits, t.cache[c], t.votes[c], d)
|
||||
of tmRadial:
|
||||
for c in 0..<TM_CLASSES:
|
||||
let d = if c == radWinner: 1.0 else: -1.0
|
||||
g.radTeams[c].tmLearnDir(t.lits, t.radCache[c], t.radVotes[c], d)
|
||||
of tmReversal:
|
||||
for c in 0..<TM_CLASSES:
|
||||
let d = if c == winner: 1.0 else: -1.0
|
||||
g.teams[c].tmLearnDir(t.lits, t.cache[c], t.votes[c], d)
|
||||
for c in 0..<2:
|
||||
let d = if c == revWinner: 1.0 else: -1.0
|
||||
g.revTeams[c].tmLearnDir(t.lits, t.revCache[c], t.revVotes[c], d)
|
||||
inc g.totalObs
|
||||
inc g.trainCalls
|
||||
if g.posRing[s].valid and g.posRing[s].tick == t.arrivalTick:
|
||||
g.tmResolveTrace(t[], e.bulletPower)
|
||||
else:
|
||||
# DEFER (Task 2): the bullet resolved BEFORE its BASE arrival tick, which
|
||||
# happens whenever the radial correction aimed SHORT. The arrival-tick
|
||||
# position is not recorded yet, so keep a COPY of the trace and train on it
|
||||
# once that tick is in the ring (`tmFlushPending`). Dropping it here is what
|
||||
# biased the training set toward only the resolvable (long/centre) aims.
|
||||
if g.pendingCount < TM_PENDING_SLOTS:
|
||||
g.pending[g.pendingCount] = PendingResolve(
|
||||
arrivalTick: t.arrivalTick, powerBin: binIdx,
|
||||
power: e.bulletPower, trace: t[])
|
||||
inc g.pendingCount
|
||||
else:
|
||||
inc g.pendingDropped
|
||||
inc g.labelMisses
|
||||
t.alive = false
|
||||
|
||||
Reference in New Issue
Block a user