fix(guns): speed-sensitive caches, dead stop-shot branch, exact TM trace pairing
Four guns cached a whole prediction per tick while predict() is called once per power bin, so every bin after the first (and the real fired shot, which shares lastState) reused the power-1.0 lead. Fixed by caching only the speed-INDEPENDENT derived state and recomputing the lead per requested speed: - stop_shot: also fixes prevSpeed being written before it was read, which made abs(speed) < abs(prev) permanently false and the entire stop-prediction branch unreachable (it was just Linear). - displacement: the cache key included bulletSpeed, so the guard missed on all four bins and the 15-tick window advanced ~4x/tick, making the inferred velocity ~4x too small. - averaged_lead: tick cache removed outright. pattern_matcher: split into speed-independent match+path and per-call lead. FeedbackEvent gains fireTick/powerBin (additive; only virtual_bullets constructs one) so guns can pair feedback to the exact shot instead of guessing by coordinates. tsetlin uses it: traces are now keyed exactly by (fireTick, powerBin) with a 1024-slot ring, and the 10-frame window shifts at most once per tick (it was shifting ~4-5x/tick, so isWarmedUp tripped after ~2 ticks). KNOWN INCOMPLETE: tsetlin still does not diverge from Linear in battle. The two named bugs are fixed (a 600-tick sim shows trainedShots=2141, traceMisses=0, and a fixed-input probe converges to a 9.6px correction), but the TM's clause feedback itself is broken: ~131 of 1740 literals end up included per clause, so its conjunction never fires. Sweeping TM_S, TM_N_CLAUSES and a two-branch Type-I update did not change the correction from 0. Needs a real TM fix or removal, not another bug fix. First-ever guard tests for the gun selector: common_libs/tests/ test_gun_harness.nim (14 checks, headless, no Java). There were none before, which is how six broken guns survived a full analysis cycle. Against the previous HEAD, 5 of these checks FAIL - that is the regression guard.
This commit is contained in:
@@ -4,6 +4,7 @@
|
||||
|
||||
import std/[math, random, strformat]
|
||||
import gun_harness/gun_interface
|
||||
import gun_harness/virtual_bullets as vb # PowerBins (power-bin count for trace keys)
|
||||
|
||||
# ── Binary encoding (adapted from BNNBot_garage/src/binary_encoding.nim) ─────
|
||||
|
||||
@@ -188,13 +189,18 @@ proc tmLearnOne(net: var TmNet, outIdx: int, lits: array[TM_N_LITERALS, uint8],
|
||||
# ── TsetlinGun public type ────────────────────────────────────────────────────
|
||||
|
||||
const
|
||||
TM_TRACE_SLOTS = 64 # ring buffer of pending traces
|
||||
# ponytail: 64 slots >> TRACE_MAX_AGE=40 ticks, safe margin; grow if many guns/bins
|
||||
# Ring of pending traces keyed EXACTLY by (fireTick, powerBin). A power-3 shot
|
||||
# can take ~fireDist/speed ~ 128 ticks to resolve, and the rack stores 4 traces
|
||||
# per tick, so 1024 slots (> 128*4) guarantee a live trace is never overwritten
|
||||
# by a newer one. The old 64-slot ring held only ~13 ticks of traces.
|
||||
TM_TRACE_SLOTS = 1024
|
||||
DebugTM* = false # set true to print [tm-dbg] lines per onResult call
|
||||
|
||||
type
|
||||
TmTrace = object
|
||||
predX, predY: float # key: matches FeedbackEvent.prediction
|
||||
fireTick: int # key part: tick the bullet was fired
|
||||
powerBin: int # key part: power bin the bullet belonged to
|
||||
predX, predY: float # stored prediction, for the directional residual
|
||||
input: TmBinaryVector
|
||||
cache: TmClauseCache
|
||||
alive: bool
|
||||
@@ -203,14 +209,30 @@ type
|
||||
net: TmNet
|
||||
frameBuffer: array[TM_WINDOW_SIZE, TmFrameEncoded]
|
||||
bufferCount: int
|
||||
frameTick: int # last tick the window was shifted (once per tick)
|
||||
traces: array[TM_TRACE_SLOTS, TmTrace]
|
||||
traceHead: int
|
||||
shotCount: int ## total onResult calls received
|
||||
trainedShots*: int ## onResult calls that found and trained their exact trace
|
||||
traceMisses*: int ## onResult calls whose trace was gone (integrity counter)
|
||||
debugGraphics*: bool
|
||||
|
||||
proc tmBinForSpeed(spd: float): int {.inline.} =
|
||||
## Map a virtual-bullet speed back to its power-bin index.
|
||||
for i in 0..<len(vb.PowerBins):
|
||||
if abs(spd - bulletSpeed(vb.PowerBins[i])) < 1e-6:
|
||||
return i
|
||||
-1
|
||||
|
||||
proc tmTraceSlot(fireTick, binIdx: int): int {.inline.} =
|
||||
## Exact (fireTick, powerBin) key -> ring slot. TM_TRACE_SLOTS is a multiple of
|
||||
## the bin count and larger than maxResolveTicks*bins, so live traces never
|
||||
## collide with newer ones; unresolved traces are evicted after ~256 ticks.
|
||||
((fireTick * len(vb.PowerBins)) + binIdx) mod TM_TRACE_SLOTS
|
||||
|
||||
proc initTsetlinGun*(): TsetlinGun =
|
||||
# states init at 0 (boundary); one Type I step crosses into Include
|
||||
for s in result.net.states.mitems: s = 0'i16
|
||||
result.frameTick = -1
|
||||
randomize()
|
||||
result.debugGraphics = false
|
||||
|
||||
@@ -228,11 +250,16 @@ proc predict*(g: var TsetlinGun, state: WorldState, bulletSpeed: float): GunPred
|
||||
state.arenaWidth - state.enemyX, state.enemyX,
|
||||
state.selfEnergy, # use self energy as proxy (enemy energy not in WorldState)
|
||||
)
|
||||
# Shift window: index 0 = newest
|
||||
for i in countdown(TM_WINDOW_SIZE - 1, 1):
|
||||
g.frameBuffer[i] = g.frameBuffer[i - 1]
|
||||
g.frameBuffer[0] = frame
|
||||
if g.bufferCount < TM_WINDOW_SIZE: inc g.bufferCount
|
||||
# Shift window: index 0 = newest. Do this at most once per tick — the harness
|
||||
# calls predict() 4-5x/tick (once per power bin), which used to shift the
|
||||
# 10-frame window ~4-5x/tick (representing ~2 real ticks and tripping
|
||||
# isWarmedUp after 2-3 ticks instead of 10).
|
||||
if state.tick != g.frameTick:
|
||||
g.frameTick = state.tick
|
||||
for i in countdown(TM_WINDOW_SIZE - 1, 1):
|
||||
g.frameBuffer[i] = g.frameBuffer[i - 1]
|
||||
g.frameBuffer[0] = frame
|
||||
if g.bufferCount < TM_WINDOW_SIZE: inc g.bufferCount
|
||||
|
||||
# Warm-up: until window is full, fall back to linear extrapolation
|
||||
let ticksToArrive = if bulletSpeed > 0.0: dist / bulletSpeed else: 1.0
|
||||
@@ -258,29 +285,48 @@ proc predict*(g: var TsetlinGun, state: WorldState, bulletSpeed: float): GunPred
|
||||
let predX = clamp(linearX + cx, 0.0, state.arenaWidth)
|
||||
let predY = clamp(linearY + cy, 0.0, state.arenaHeight)
|
||||
|
||||
# Store trace keyed by prediction coords
|
||||
let slot = g.traceHead mod TM_TRACE_SLOTS
|
||||
g.traces[slot] = TmTrace(predX: predX, predY: predY, input: vec, cache: cache, alive: true)
|
||||
g.traceHead = (slot + 1) mod TM_TRACE_SLOTS
|
||||
# Store trace keyed exactly by (fireTick, powerBin) so the resolution event
|
||||
# can find it no matter how many other guns/bins fired in between.
|
||||
let binIdx = tmBinForSpeed(bulletSpeed)
|
||||
if binIdx >= 0:
|
||||
let slot = tmTraceSlot(state.tick, binIdx)
|
||||
g.traces[slot] = TmTrace(
|
||||
fireTick: state.tick,
|
||||
powerBin: binIdx,
|
||||
predX: predX,
|
||||
predY: predY,
|
||||
input: vec,
|
||||
cache: cache,
|
||||
alive: true,
|
||||
)
|
||||
|
||||
GunPrediction(x: predX, y: predY)
|
||||
|
||||
proc onResult*(g: var TsetlinGun, e: FeedbackEvent) =
|
||||
inc g.shotCount
|
||||
# Find matching trace by prediction coords
|
||||
for i in 0..<TM_TRACE_SLOTS:
|
||||
var t = addr g.traces[i]
|
||||
if not t.alive: continue
|
||||
if abs(t.predX - e.prediction.x) > 0.01 or abs(t.predY - e.prediction.y) > 0.01:
|
||||
continue
|
||||
# Directional residual: actual enemy pos minus our prediction
|
||||
# On hit residual is 0 (we were right); on miss we push toward actual position.
|
||||
let rx = if e.hit: 0.0 else: clamp(e.actualX - t.predX, -TM_RESID_MAX, TM_RESID_MAX)
|
||||
let ry = if e.hit: 0.0 else: clamp(e.actualY - t.predY, -TM_RESID_MAX, TM_RESID_MAX)
|
||||
let lits = tmMakeLiterals(t.input)
|
||||
g.net.tmLearnOne(0, lits, t.cache, rx)
|
||||
g.net.tmLearnOne(1, lits, t.cache, ry)
|
||||
when DebugTM:
|
||||
echo fmt"[tm-dbg] shot={g.shotCount} miss={e.missDistance:.1f}px predicted=({t.predX:.0f},{t.predY:.0f}) actual=({e.actualX:.0f},{e.actualY:.0f}) rx={rx:.1f} ry={ry:.1f} hit={e.hit}"
|
||||
t.alive = false
|
||||
break
|
||||
# Exact pairing: index the trace by the tick the bullet was fired and the power
|
||||
# bin it belonged to. The old coordinate-matched 64-slot ring lost the trace
|
||||
# long before a long shot resolved, so the TM never trained and its output was
|
||||
# pure linear extrapolation.
|
||||
let binIdx = if e.powerBin >= 0 and e.powerBin < len(vb.PowerBins): e.powerBin
|
||||
else: tmBinForSpeed(bulletSpeed(e.bulletPower))
|
||||
if binIdx < 0:
|
||||
inc g.traceMisses
|
||||
return
|
||||
let slot = tmTraceSlot(e.fireTick, binIdx)
|
||||
var t = addr g.traces[slot]
|
||||
if not t.alive or t.fireTick != e.fireTick or t.powerBin != binIdx:
|
||||
inc g.traceMisses
|
||||
return
|
||||
|
||||
# Directional residual: actual enemy pos minus our prediction
|
||||
# On hit residual is 0 (we were right); on miss we push toward actual position.
|
||||
let rx = if e.hit: 0.0 else: clamp(e.actualX - t.predX, -TM_RESID_MAX, TM_RESID_MAX)
|
||||
let ry = if e.hit: 0.0 else: clamp(e.actualY - t.predY, -TM_RESID_MAX, TM_RESID_MAX)
|
||||
let lits = tmMakeLiterals(t.input)
|
||||
g.net.tmLearnOne(0, lits, t.cache, rx)
|
||||
g.net.tmLearnOne(1, lits, t.cache, ry)
|
||||
when DebugTM:
|
||||
echo fmt"[tm-dbg] shot={g.shotCount} tick={e.fireTick} bin={binIdx} miss={e.missDistance:.1f}px predicted=({t.predX:.0f},{t.predY:.0f}) actual=({e.actualX:.0f},{e.actualY:.0f}) rx={rx:.1f} ry={ry:.1f} hit={e.hit}"
|
||||
t.alive = false
|
||||
inc g.trainedShots
|
||||
|
||||
Reference in New Issue
Block a user