fix(guns): speed-sensitive caches, dead stop-shot branch, exact TM trace pairing

Four guns cached a whole prediction per tick while predict() is called once
per power bin, so every bin after the first (and the real fired shot, which
shares lastState) reused the power-1.0 lead. Fixed by caching only the
speed-INDEPENDENT derived state and recomputing the lead per requested speed:
- stop_shot: also fixes prevSpeed being written before it was read, which
  made abs(speed) < abs(prev) permanently false and the entire
  stop-prediction branch unreachable (it was just Linear).
- displacement: the cache key included bulletSpeed, so the guard missed on
  all four bins and the 15-tick window advanced ~4x/tick, making the
  inferred velocity ~4x too small.
- averaged_lead: tick cache removed outright. pattern_matcher: split into
  speed-independent match+path and per-call lead.

FeedbackEvent gains fireTick/powerBin (additive; only virtual_bullets
constructs one) so guns can pair feedback to the exact shot instead of
guessing by coordinates. tsetlin uses it: traces are now keyed exactly by
(fireTick, powerBin) with a 1024-slot ring, and the 10-frame window shifts
at most once per tick (it was shifting ~4-5x/tick, so isWarmedUp tripped
after ~2 ticks).

KNOWN INCOMPLETE: tsetlin still does not diverge from Linear in battle. The
two named bugs are fixed (a 600-tick sim shows trainedShots=2141,
traceMisses=0, and a fixed-input probe converges to a 9.6px correction), but
the TM's clause feedback itself is broken: ~131 of 1740 literals end up
included per clause, so its conjunction never fires. Sweeping TM_S,
TM_N_CLAUSES and a two-branch Type-I update did not change the correction
from 0. Needs a real TM fix or removal, not another bug fix.

First-ever guard tests for the gun selector: common_libs/tests/
test_gun_harness.nim (14 checks, headless, no Java). There were none before,
which is how six broken guns survived a full analysis cycle. Against the
previous HEAD, 5 of these checks FAIL - that is the regression guard.
This commit is contained in:
2026-09-20 22:47:26 +02:00
parent 0cc682152d
commit e53690036b
9 changed files with 480 additions and 171 deletions
+76 -30
View File
@@ -4,6 +4,7 @@
import std/[math, random, strformat]
import gun_harness/gun_interface
import gun_harness/virtual_bullets as vb # PowerBins (power-bin count for trace keys)
# ── Binary encoding (adapted from BNNBot_garage/src/binary_encoding.nim) ─────
@@ -188,13 +189,18 @@ proc tmLearnOne(net: var TmNet, outIdx: int, lits: array[TM_N_LITERALS, uint8],
# ── TsetlinGun public type ────────────────────────────────────────────────────
const
TM_TRACE_SLOTS = 64 # ring buffer of pending traces
# ponytail: 64 slots >> TRACE_MAX_AGE=40 ticks, safe margin; grow if many guns/bins
# Ring of pending traces keyed EXACTLY by (fireTick, powerBin). A power-3 shot
# can take ~fireDist/speed ~ 128 ticks to resolve, and the rack stores 4 traces
# per tick, so 1024 slots (> 128*4) guarantee a live trace is never overwritten
# by a newer one. The old 64-slot ring held only ~13 ticks of traces.
TM_TRACE_SLOTS = 1024
DebugTM* = false # set true to print [tm-dbg] lines per onResult call
type
TmTrace = object
predX, predY: float # key: matches FeedbackEvent.prediction
fireTick: int # key part: tick the bullet was fired
powerBin: int # key part: power bin the bullet belonged to
predX, predY: float # stored prediction, for the directional residual
input: TmBinaryVector
cache: TmClauseCache
alive: bool
@@ -203,14 +209,30 @@ type
net: TmNet
frameBuffer: array[TM_WINDOW_SIZE, TmFrameEncoded]
bufferCount: int
frameTick: int # last tick the window was shifted (once per tick)
traces: array[TM_TRACE_SLOTS, TmTrace]
traceHead: int
shotCount: int ## total onResult calls received
trainedShots*: int ## onResult calls that found and trained their exact trace
traceMisses*: int ## onResult calls whose trace was gone (integrity counter)
debugGraphics*: bool
proc tmBinForSpeed(spd: float): int {.inline.} =
## Map a virtual-bullet speed back to its power-bin index.
for i in 0..<len(vb.PowerBins):
if abs(spd - bulletSpeed(vb.PowerBins[i])) < 1e-6:
return i
-1
proc tmTraceSlot(fireTick, binIdx: int): int {.inline.} =
## Exact (fireTick, powerBin) key -> ring slot. TM_TRACE_SLOTS is a multiple of
## the bin count and larger than maxResolveTicks*bins, so live traces never
## collide with newer ones; unresolved traces are evicted after ~256 ticks.
((fireTick * len(vb.PowerBins)) + binIdx) mod TM_TRACE_SLOTS
proc initTsetlinGun*(): TsetlinGun =
# states init at 0 (boundary); one Type I step crosses into Include
for s in result.net.states.mitems: s = 0'i16
result.frameTick = -1
randomize()
result.debugGraphics = false
@@ -228,11 +250,16 @@ proc predict*(g: var TsetlinGun, state: WorldState, bulletSpeed: float): GunPred
state.arenaWidth - state.enemyX, state.enemyX,
state.selfEnergy, # use self energy as proxy (enemy energy not in WorldState)
)
# Shift window: index 0 = newest
for i in countdown(TM_WINDOW_SIZE - 1, 1):
g.frameBuffer[i] = g.frameBuffer[i - 1]
g.frameBuffer[0] = frame
if g.bufferCount < TM_WINDOW_SIZE: inc g.bufferCount
# Shift window: index 0 = newest. Do this at most once per tick — the harness
# calls predict() 4-5x/tick (once per power bin), which used to shift the
# 10-frame window ~4-5x/tick (representing ~2 real ticks and tripping
# isWarmedUp after 2-3 ticks instead of 10).
if state.tick != g.frameTick:
g.frameTick = state.tick
for i in countdown(TM_WINDOW_SIZE - 1, 1):
g.frameBuffer[i] = g.frameBuffer[i - 1]
g.frameBuffer[0] = frame
if g.bufferCount < TM_WINDOW_SIZE: inc g.bufferCount
# Warm-up: until window is full, fall back to linear extrapolation
let ticksToArrive = if bulletSpeed > 0.0: dist / bulletSpeed else: 1.0
@@ -258,29 +285,48 @@ proc predict*(g: var TsetlinGun, state: WorldState, bulletSpeed: float): GunPred
let predX = clamp(linearX + cx, 0.0, state.arenaWidth)
let predY = clamp(linearY + cy, 0.0, state.arenaHeight)
# Store trace keyed by prediction coords
let slot = g.traceHead mod TM_TRACE_SLOTS
g.traces[slot] = TmTrace(predX: predX, predY: predY, input: vec, cache: cache, alive: true)
g.traceHead = (slot + 1) mod TM_TRACE_SLOTS
# Store trace keyed exactly by (fireTick, powerBin) so the resolution event
# can find it no matter how many other guns/bins fired in between.
let binIdx = tmBinForSpeed(bulletSpeed)
if binIdx >= 0:
let slot = tmTraceSlot(state.tick, binIdx)
g.traces[slot] = TmTrace(
fireTick: state.tick,
powerBin: binIdx,
predX: predX,
predY: predY,
input: vec,
cache: cache,
alive: true,
)
GunPrediction(x: predX, y: predY)
proc onResult*(g: var TsetlinGun, e: FeedbackEvent) =
inc g.shotCount
# Find matching trace by prediction coords
for i in 0..<TM_TRACE_SLOTS:
var t = addr g.traces[i]
if not t.alive: continue
if abs(t.predX - e.prediction.x) > 0.01 or abs(t.predY - e.prediction.y) > 0.01:
continue
# Directional residual: actual enemy pos minus our prediction
# On hit residual is 0 (we were right); on miss we push toward actual position.
let rx = if e.hit: 0.0 else: clamp(e.actualX - t.predX, -TM_RESID_MAX, TM_RESID_MAX)
let ry = if e.hit: 0.0 else: clamp(e.actualY - t.predY, -TM_RESID_MAX, TM_RESID_MAX)
let lits = tmMakeLiterals(t.input)
g.net.tmLearnOne(0, lits, t.cache, rx)
g.net.tmLearnOne(1, lits, t.cache, ry)
when DebugTM:
echo fmt"[tm-dbg] shot={g.shotCount} miss={e.missDistance:.1f}px predicted=({t.predX:.0f},{t.predY:.0f}) actual=({e.actualX:.0f},{e.actualY:.0f}) rx={rx:.1f} ry={ry:.1f} hit={e.hit}"
t.alive = false
break
# Exact pairing: index the trace by the tick the bullet was fired and the power
# bin it belonged to. The old coordinate-matched 64-slot ring lost the trace
# long before a long shot resolved, so the TM never trained and its output was
# pure linear extrapolation.
let binIdx = if e.powerBin >= 0 and e.powerBin < len(vb.PowerBins): e.powerBin
else: tmBinForSpeed(bulletSpeed(e.bulletPower))
if binIdx < 0:
inc g.traceMisses
return
let slot = tmTraceSlot(e.fireTick, binIdx)
var t = addr g.traces[slot]
if not t.alive or t.fireTick != e.fireTick or t.powerBin != binIdx:
inc g.traceMisses
return
# Directional residual: actual enemy pos minus our prediction
# On hit residual is 0 (we were right); on miss we push toward actual position.
let rx = if e.hit: 0.0 else: clamp(e.actualX - t.predX, -TM_RESID_MAX, TM_RESID_MAX)
let ry = if e.hit: 0.0 else: clamp(e.actualY - t.predY, -TM_RESID_MAX, TM_RESID_MAX)
let lits = tmMakeLiterals(t.input)
g.net.tmLearnOne(0, lits, t.cache, rx)
g.net.tmLearnOne(1, lits, t.cache, ry)
when DebugTM:
echo fmt"[tm-dbg] shot={g.shotCount} tick={e.fireTick} bin={binIdx} miss={e.missDistance:.1f}px predicted=({t.predX:.0f},{t.predY:.0f}) actual=({e.actualX:.0f},{e.actualY:.0f}) rx={rx:.1f} ry={ry:.1f} hit={e.hit}"
t.alive = false
inc g.trainedShots