TM horizon: retain learning ACROSS ROUNDS, reset only when the ENEMY changes
The user's requirement: "every battle i means from round 1 to round end-battle, so retain all learning until the enemy change." What was built wiped the Tsetlin machines EVERY ROUND, in two places (`onRoundStarted` and the gun's own tick-regression self-reset), so in a 7-round battle each round started cold, trained ~360 samples and threw them away - discarding most of its one chance to do what was asked: overfit the current enemy over the whole battle. THE FIX - two kinds of state, two triggers: - **`resetRoundState` (per ROUND)**: the observation ring, pending/deferred labels, the bullet proxy, motion history, per-tick caches, `roundStartTrained`. These MUST clear every round, because bots teleport back to the starting corners between rounds - an old position would build a garbage label. (That exact class of bug shipped 36-58% wrong labels in the old gun.) - **`resetLearning` (per BATTLE / per ENEMY)**: both Tsetlin machines, `trained`, `sideCorrect/sideTotal`, all histograms, the magnitude median, `pendingDropped`, `observedTargetId`. These now SURVIVE round boundaries. Triggers for the machine wipe: `onGameStarted` (primary) plus a redundant `roundNumber <= 1` fallback in `onRoundStarted`; and a TARGET CHANGE (`targetChanged`, knob `TR_TMHORIZON_RESET_ON_TARGET` default on - a no-op in 1v1, fires on melee target switches; first acquisition never wipes). The tick-regression self-reset now clears ONLY per-round state. Still NO cross-battle persistence: grep for file I/O in the gun finds none. PROOF IT WORKS (live 2-round battle, `TR_RACK_PATTERN=off TR_RACK_TMHORIZON=both`): [tmh-reset] reason=game_start trained_was=0 [tmh-reset] reason=round1 trained_was=0 ...exactly TWO reset lines in the whole battle, both at battle start, and NONE at the round-2 boundary. And the per-round summaries: [tmh-round] trained=1249 thisRound=1249 ... sideAcc=711/1177 (60.4%) [tmh-round] trained=2226 thisRound=977 ... sideAcc=1361/2154 (63.2%) `trained` CLIMBED 1249 -> 2226 across the boundary, and side accuracy rose 60.4% -> 63.2% in round 2 (one battle - suggestive, not proof). Unit tests: `test_tm_horizon` 79 (was 54), including "trained SURVIVES the boundary", "clause states SURVIVE", "trained climbs round1->round2", "game-start wipes and all clauses end Exclude", "different enemy wipes / same enemy does not / knob-off does not", and crucially "a label CANNOT be built across a round boundary" (ringValidCount==0, ringHas(oldTick)==false, pendingCount==0) - the single most dangerous interaction of this change. Guards: test_tm_horizon 79, test_rack_membership 48, test_gun_harness 39, test_vbullet_metric 11, test_power_selection 3, test_adaptive_radar 41, test_tfil_ring_weights 24, test_power_policy 26, test_ram_decision 40, test_selector_tiebreak 19, test_tm_pattern_registration 20, test_vbullet_admit_gate 12, test_tm_diag 48, test_tm_automata_diag 55, test_tm_clause_shape 66. acceptance_offline_vs_online 12/12 VERDICT PASS. Shipped rack unchanged: DefaultRackMembership is still Pattern-only, TMHORIZON off. Residual (pre-existing, out of scope, stated): the internal base `PatternMatcherGun` has a rolling move-history buffer that is NOT cleared at round boundaries - it never was, and the SHIPPED Pattern gun carries history across rounds too. It cannot affect label correctness (labels come from `g.ring`), only base-prediction quality in a round's first ticks.
This commit is contained in:
@@ -1,5 +1,5 @@
|
||||
## ModularBot — plugin gun architecture tracer bullet.
|
||||
## Guns: HeadOnGun (0), LinearGun (1), TsetlinGun (2), CircularGun (3), GFGun (4), PatternMatcherGun (5), WallBounceGun (6), AccelGun (7), StopShotGun (8), DisplacementGun (9), AveragedLeadGun (10), DecayGFGun (11), KNNGun (12), TmSelectorGun (13), TmPatternGun (14) via GunHarness.
|
||||
## Guns: HeadOnGun (0), LinearGun (1), TsetlinGun (2), CircularGun (3), GFGun (4), PatternMatcherGun (5), WallBounceGun (6), AccelGun (7), StopShotGun (8), DisplacementGun (9), AveragedLeadGun (10), DecayGFGun (11), KNNGun (12), TmSelectorGun (13), TmPatternGun (14), TmHorizonGun (15) via GunHarness.
|
||||
## Radar: RadarLockModule (1v1) / AdaptiveMeleeRadarModule (2+ enemies), auto-switched per tick.
|
||||
## Movement: OscillatorModule (perpendicular strafing).
|
||||
|
||||
@@ -26,6 +26,7 @@ import guns/decay_gf
|
||||
import guns/knn_gun
|
||||
import guns/tm_selector
|
||||
import guns/tm_pattern
|
||||
import guns/tm_horizon
|
||||
import movements/phantom_meteor
|
||||
import movements/rammer
|
||||
import movements/ram_decision
|
||||
@@ -119,7 +120,7 @@ let ShotLogPath = getEnv("GUN_SHOTLOG_PATH", "/tmp/shot_log.jsonl")
|
||||
## shows why the cap moved. The policy itself lives in the shared gun harness
|
||||
## (`applyPowerPolicy`), so both live and offline paths see the same rule.
|
||||
let PowerLog = existsEnv("TR_POWER_LOG")
|
||||
const GunNames = ["HeadOn", "Linear", "Tsetlin", "Circular", "GuessFactor", "Pattern", "WallBounce", "Accel", "StopShot", "Displace", "AvgLead", "DecayGF", "KNN", "TMSelect", "TMPattern"]
|
||||
const GunNames = ["HeadOn", "Linear", "Tsetlin", "Circular", "GuessFactor", "Pattern", "WallBounce", "Accel", "StopShot", "Displace", "AvgLead", "DecayGF", "KNN", "TMSelect", "TMPattern", "TMHorizon"]
|
||||
|
||||
## Rack id of the new TM pattern gun. It defaults to `TR_RACK_TMPATTERN=off`;
|
||||
## unlike the other guns, its virtual-bullet spawn is gated on rack admission
|
||||
@@ -128,6 +129,12 @@ const GunNames = ["HeadOn", "Linear", "Tsetlin", "Circular", "GuessFactor", "Pat
|
||||
## the default selection sequence — is byte-for-byte unchanged.
|
||||
const TmPatternId = 14
|
||||
|
||||
## Rack id of the horizon-based TM corrector. Like TMPATTERN it defaults to
|
||||
## `TR_RACK_TMHORIZON=off` and its virtual-bullet spawn is gated on rack
|
||||
## admission, so the shipped default never spawns it and the shared tracker ring
|
||||
## head — hence every other gun's learning order — is byte-for-byte unchanged.
|
||||
const TmHorizonId = 15
|
||||
|
||||
const
|
||||
CLR_GUN = "\e[33m" # yellow
|
||||
CLR_MOVE = "\e[36m" # cyan
|
||||
@@ -174,6 +181,7 @@ type
|
||||
knnGun: KNNGun
|
||||
tmSelector: TmSelectorGun
|
||||
tmPattern: TmPatternGun
|
||||
tmHorizon: TmHorizonGun
|
||||
mover: TFILModule
|
||||
ringMover: TFILRingModule
|
||||
rammer: RammerModule
|
||||
@@ -199,18 +207,18 @@ type
|
||||
roundNumber: int
|
||||
realShotsFired: int
|
||||
realHits: int
|
||||
gunRealShots: array[15, int]
|
||||
gunRealHits: array[15, int]
|
||||
gunRealShots: array[16, int]
|
||||
gunRealHits: array[16, int]
|
||||
# Same real-shot accounting split by the rack in force at fire time, so a
|
||||
# later data-driven pass can rank guns per mode (1v1 vs melee).
|
||||
gunRealShotsByMode: array[vb.RackMode, array[15, int]]
|
||||
gunRealHitsByMode: array[vb.RackMode, array[15, int]]
|
||||
gunRealShotsByMode: array[vb.RackMode, array[16, int]]
|
||||
gunRealHitsByMode: array[vb.RackMode, array[16, int]]
|
||||
pendingFires: seq[PendingShot] ## FIFO of fired shots awaiting onBulletFired bulletId stamp
|
||||
bulletGun: Table[int, int] ## bulletId -> gun id, filled on onBulletFired, drained on resolution
|
||||
bulletMode: Table[int, vb.RackMode] ## bulletId -> rack at fire time
|
||||
bulletShot: Table[int, PendingShot] ## bulletId -> shot metadata (Task A shot log)
|
||||
pendingHitBullets: HashSet[int] ## hit bulletIds seen before their onBulletFired stamp
|
||||
gunSelectionCount: array[15, int]
|
||||
gunSelectionCount: array[16, int]
|
||||
lastPowerLogKey: string ## change detector for the TR_POWER_LOG line
|
||||
lastKnownTargetId: int ## persists through death, used for round-end stats
|
||||
# Radar measurement instrumentation (only touched when RadarScanLog is set).
|
||||
@@ -569,6 +577,9 @@ method onRoundEnded*(bot: ModularBot, e: RoundEndedEventForBot) =
|
||||
## Dump per-gun virtual bullet stats to /tmp/gun_stats.jsonl (one line per round).
|
||||
if RecordWorldState:
|
||||
bot.finishWorldStateRecord()
|
||||
# Per-round TM-horizon summary (change-gated behind TR_TMHORIZON_LOG=1) so the
|
||||
# user can watch it learn across the round.
|
||||
bot.tmHorizon.roundSummary()
|
||||
# Use lastKnownTargetId: currentTargetId is -1 if enemy died before round end
|
||||
let targetId = if bot.currentTargetId >= 0: bot.currentTargetId else: bot.lastKnownTargetId
|
||||
# Per-target fitness when known, else a deterministic recency-weighted aggregate
|
||||
@@ -577,7 +588,7 @@ method onRoundEnded*(bot: ModularBot, e: RoundEndedEventForBot) =
|
||||
let fit = bot.tracker.fitnessFor(targetId)
|
||||
|
||||
var gunsArr = newJArray()
|
||||
for gid in 0..<15:
|
||||
for gid in 0..<16:
|
||||
var totalShots = 0
|
||||
var totalHits = 0
|
||||
for binIdx in 0..<len(vb.PowerBins):
|
||||
@@ -690,12 +701,12 @@ method onRoundStarted*(bot: ModularBot, e: RoundStartedEvent) =
|
||||
bot.radarAcquireTicks = 0
|
||||
bot.radarTrackTicks = 0
|
||||
bot.radarMeleeActive = false
|
||||
for i in 0..<15:
|
||||
for i in 0..<16:
|
||||
bot.gunSelectionCount[i] = 0
|
||||
bot.gunRealShots[i] = 0
|
||||
bot.gunRealHits[i] = 0
|
||||
for m in vb.RackMode:
|
||||
for i in 0..<15:
|
||||
for i in 0..<16:
|
||||
bot.gunRealShotsByMode[m][i] = 0
|
||||
bot.gunRealHitsByMode[m][i] = 0
|
||||
# Reset per-round integrity counters so each /tmp/gun_stats.jsonl line reports
|
||||
@@ -720,6 +731,14 @@ method onRoundStarted*(bot: ModularBot, e: RoundStartedEvent) =
|
||||
# its clause teams and motion history explicitly, so a same-id opponent in the
|
||||
# next round cannot inherit the previous round's overfit net.
|
||||
bot.tmPattern.resetLearning()
|
||||
# The horizon TM corrector KEEPS its machines across rounds: only the
|
||||
# per-round observation ring and deferred labels are cleared (bots teleport
|
||||
# between rounds, so old positions are meaningless). The machines are wiped on
|
||||
# a NEW BATTLE in onGameStarted, with a round-1 fallback here in case that
|
||||
# callback does not fire in this harness.
|
||||
bot.tmHorizon.resetRoundState()
|
||||
if e.roundNumber <= 1:
|
||||
bot.tmHorizon.resetLearning("round1")
|
||||
bot.isRamming = false
|
||||
bot.ramDurationTicks = 0
|
||||
bot.ramCooldownTicks = 0
|
||||
@@ -773,6 +792,11 @@ method onBotDeath*(bot: ModularBot, e: BotDeathEvent) =
|
||||
bot.currentTargetId = -1
|
||||
|
||||
method onGameStarted*(bot: ModularBot, e: GameStartedEventForBot) =
|
||||
# A NEW BATTLE begins: wipe the horizon TM's machines and learned statistics
|
||||
# (no persistence across battles). `onRoundStarted` only clears per-round
|
||||
# state, so this is the only place the learning is dropped at a battle
|
||||
# boundary. The round-1 fallback in `onRoundStarted` covers a missed callback.
|
||||
bot.tmHorizon.resetLearning("game_start")
|
||||
# minNumberOfParticipants == maxNumberOfParticipants for fixed battles; self is -1
|
||||
bot.initialEnemyCount = e.gameSetup.minNumberOfParticipants - 1
|
||||
# The custom event must be registered AFTER `start()` ran `initGlobals()`,
|
||||
@@ -863,6 +887,10 @@ method run*(bot: ModularBot) =
|
||||
bot.currentTargetId = candidateId
|
||||
bot.targetSwitchTick = bot.tick
|
||||
bot.cfgDirty = true # emit at tick end, after gun selection
|
||||
# The horizon TM learns ONE enemy at a time: wipe its machines when the
|
||||
# target changes to a different bot id (gated by
|
||||
# TR_TMHORIZON_RESET_ON_TARGET, default on; no-op in 1v1).
|
||||
discard bot.tmHorizon.targetChanged(candidateId)
|
||||
if bot.currentTargetId >= 0:
|
||||
bot.lastKnownTargetId = bot.currentTargetId
|
||||
|
||||
@@ -998,6 +1026,7 @@ method run*(bot: ModularBot) =
|
||||
var knnPreds: array[len(PowerBins), GunPrediction]
|
||||
var tmselPreds: array[len(PowerBins), GunPrediction]
|
||||
var tmpPreds: array[len(PowerBins), GunPrediction]
|
||||
var tmhPreds: array[len(PowerBins), GunPrediction]
|
||||
# ── TR_VBULLET_ADMIT_ONLY gate ──────────────────────────────────────
|
||||
# Rack membership used to filter only SELECTION, so every unselected gun
|
||||
# still ran predict()+spawnBullets() each tick to feed a fitness table the
|
||||
@@ -1007,12 +1036,13 @@ method run*(bot: ModularBot) =
|
||||
# thinning to 1v1). Bullets already in flight are NOT cancelled: they are
|
||||
# resolved by `tickBullets` and the feedback `case` below still calls the
|
||||
# owning gun's onResult, so attribution survives a mid-round rack change.
|
||||
# TMPATTERN keeps its own admission gate even when the knob is 0, so
|
||||
# TR_VBULLET_ADMIT_ONLY=0 reproduces the exact pre-change rack.
|
||||
var admit: array[15, bool]
|
||||
for gi in 0..<15:
|
||||
# TMPATTERN and TMHORIZON keep their own admission gate even when the knob
|
||||
# is 0, so TR_VBULLET_ADMIT_ONLY=0 reproduces the exact pre-change rack.
|
||||
var admit: array[16, bool]
|
||||
for gi in 0..<16:
|
||||
admit[gi] = vBulletAdmitted(gi, bot.rackMode, ActiveRackMembership,
|
||||
VBulletAdmitOnly or gi == TmPatternId)
|
||||
VBulletAdmitOnly or gi == TmPatternId or
|
||||
gi == TmHorizonId)
|
||||
for i in 0..<len(PowerBins):
|
||||
if admit[0]: headsUp[i] = bot.headOn.predict(bot.lastState, bulletSpeed(PowerBins[i]))
|
||||
if admit[1]: linPreds[i] = bot.linear.predict(bot.lastState, bulletSpeed(PowerBins[i]))
|
||||
@@ -1032,6 +1062,9 @@ method run*(bot: ModularBot) =
|
||||
# TMPATTERN keeps its own rack-admission gate (see `admit` above).
|
||||
if admit[TmPatternId]:
|
||||
tmpPreds[i] = bot.tmPattern.predict(bot.lastState, bulletSpeed(PowerBins[i]))
|
||||
# TMHORIZON likewise: Pattern base + horizon TM correction.
|
||||
if admit[TmHorizonId]:
|
||||
tmhPreds[i] = bot.tmHorizon.predict(bot.lastState, bulletSpeed(PowerBins[i]))
|
||||
|
||||
if admit[0]: bot.tracker.spawnBullets(0, headsUp, bot.lastState, tid)
|
||||
if admit[1] and not gunDisabled(1): bot.tracker.spawnBullets(1, linPreds, bot.lastState, tid)
|
||||
@@ -1051,6 +1084,8 @@ method run*(bot: ModularBot) =
|
||||
bot.tracker.spawnBullets(13, tmselPreds, bot.lastState, tid)
|
||||
if admit[TmPatternId] and not gunDisabled(TmPatternId):
|
||||
bot.tracker.spawnBullets(TmPatternId, tmpPreds, bot.lastState, tid)
|
||||
if admit[TmHorizonId] and not gunDisabled(TmHorizonId):
|
||||
bot.tracker.spawnBullets(TmHorizonId, tmhPreds, bot.lastState, tid)
|
||||
|
||||
# Build slim enemy table for tickBullets
|
||||
var enemyPositions: Table[int, tuple[x, y: float, lastSeenTick: int, alive: bool]]
|
||||
@@ -1080,6 +1115,7 @@ method run*(bot: ModularBot) =
|
||||
of 12: bot.knnGun.onResult(fe)
|
||||
of 13: bot.tmSelector.onResult(fe)
|
||||
of TmPatternId: bot.tmPattern.onResult(fe)
|
||||
of TmHorizonId: bot.tmHorizon.onResult(fe)
|
||||
else: discard
|
||||
if fe.hit: inc bot.virtualHits else: inc bot.virtualMiss
|
||||
when DebugVBullets:
|
||||
@@ -1122,6 +1158,7 @@ method run*(bot: ModularBot) =
|
||||
of 12: setTurretColor("#CC00CC"); setBulletColor("#FF44FF")
|
||||
of 13: setTurretColor("#00FFCC"); setBulletColor("#66FFDD")
|
||||
of TmPatternId: setTurretColor("#AAFF00"); setBulletColor("#CCFF66")
|
||||
of TmHorizonId: setTurretColor("#00AAFF"); setBulletColor("#66CCFF")
|
||||
else: discard
|
||||
|
||||
let pred = case selectedGun
|
||||
@@ -1139,6 +1176,7 @@ method run*(bot: ModularBot) =
|
||||
of 12: bot.knnGun.predict(bot.lastState, bulletSpeed(power))
|
||||
of 13: bot.tmSelector.predict(bot.lastState, bulletSpeed(power))
|
||||
of TmPatternId: bot.tmPattern.predict(bot.lastState, bulletSpeed(power))
|
||||
of TmHorizonId: bot.tmHorizon.predict(bot.lastState, bulletSpeed(power))
|
||||
else: bot.headOn.predict(bot.lastState, bulletSpeed(power))
|
||||
let aimTarget = aimAngle(getX(), getY(), pred.x, pred.y)
|
||||
|
||||
@@ -1193,7 +1231,7 @@ proc seedSelectorRng() =
|
||||
|
||||
when isMainModule:
|
||||
var bot = ModularBot(
|
||||
tracker: vb.initTracker(15), # 0: HeadOn, 1: Linear, 2: Tsetlin, 3: Circular, 4: GuessFactor, 5: Pattern, 6: WallBounce, 7: Accel, 8: StopShot, 9: Displace, 10: AvgLead, 11: DecayGF, 12: KNN, 13: TMSelect, 14: TMPattern
|
||||
tracker: vb.initTracker(16), # 0: HeadOn, 1: Linear, 2: Tsetlin, 3: Circular, 4: GuessFactor, 5: Pattern, 6: WallBounce, 7: Accel, 8: StopShot, 9: Displace, 10: AvgLead, 11: DecayGF, 12: KNN, 13: TMSelect, 14: TMPattern, 15: TMHorizon
|
||||
headOn: HeadOnGun(),
|
||||
linear: LinearGun(),
|
||||
circular: CircularGun(),
|
||||
@@ -1209,6 +1247,7 @@ when isMainModule:
|
||||
knnGun: initKNNGun(),
|
||||
tmSelector: initTmSelectorGun(),
|
||||
tmPattern: initTmRadialGun(),
|
||||
tmHorizon: initTmHorizonGun(),
|
||||
radar: RadarLockModule(),
|
||||
meleeRadar: initAdaptiveMeleeRadar(),
|
||||
mover: TFILModule(debugGraphics: true),
|
||||
|
||||
Reference in New Issue
Block a user