tune(SNNBot): bullet economy tuning vs Walls — 2/10 wins
Changes: - Fixed fire power at 3.0 (removes hitRateEMA power ladder noise that cleared exemplars mid-battle and corrupted learning) - Relative velocity encoding: input uses (velDir - bearing) instead of absolute velDir, so exemplars generalize across Walls' starting walls - Fix test winner detection to use per-round score delta instead of rank field (rank in round_ended is cumulative battle rank, not round winner) - Keep exemplars across power changes (no longer relevant with fixed power) - Store lastBulletSpeed at fire time for accurate EVALUATE lead prediction Result: 2/10 rounds won vs Walls (Nim); first-round win now possible from round 1 when aimer generalizes from relative velocity patterns. Bottleneck: sparse exemplars in early rounds; energy bleeds at P3 cold-start. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -220,6 +220,7 @@ type
|
|||||||
hitRateEMA: float # exponential moving average of hit rate
|
hitRateEMA: float # exponential moving average of hit rate
|
||||||
currentFirePower: float # dynamic fire power (1.0 / 2.0 / 3.0)
|
currentFirePower: float # dynamic fire power (1.0 / 2.0 / 3.0)
|
||||||
bulletSpeed: float # 20 - 3 * currentFirePower
|
bulletSpeed: float # 20 - 3 * currentFirePower
|
||||||
|
lastBulletSpeed: float # bullet speed at last fire (for EVALUATE learning)
|
||||||
pendingPower: float # candidate new power level
|
pendingPower: float # candidate new power level
|
||||||
powerChangeCounter: int # ticks the new power has been suggested
|
powerChangeCounter: int # ticks the new power has been suggested
|
||||||
|
|
||||||
@@ -285,10 +286,11 @@ method onRoundStarted*(bot: SNNBot, e: RoundStartedEvent) =
|
|||||||
bot.bulletsHit = 0
|
bot.bulletsHit = 0
|
||||||
bot.hitRateEMA = 0.5 # optimistic start
|
bot.hitRateEMA = 0.5 # optimistic start
|
||||||
bot.powerChangeCounter = 0
|
bot.powerChangeCounter = 0
|
||||||
# currentFirePower persists; init to 1.0 on first round (zero-value)
|
# currentFirePower persists; init to 3.0 on first round (zero-value)
|
||||||
if bot.currentFirePower == 0.0:
|
if bot.currentFirePower == 0.0:
|
||||||
bot.currentFirePower = 1.0
|
bot.currentFirePower = 3.0
|
||||||
bot.bulletSpeed = 20.0 - 3.0 * bot.currentFirePower
|
bot.bulletSpeed = 20.0 - 3.0 * bot.currentFirePower
|
||||||
|
bot.lastBulletSpeed = bot.bulletSpeed
|
||||||
bot.lastEnemyX = 0.0
|
bot.lastEnemyX = 0.0
|
||||||
bot.lastEnemyY = 0.0
|
bot.lastEnemyY = 0.0
|
||||||
bot.lastAbsBearing = 0.0
|
bot.lastAbsBearing = 0.0
|
||||||
@@ -330,7 +332,12 @@ method onBulletHit*(bot: SNNBot, e: BulletHitBotEvent) =
|
|||||||
# ── Reservoir helpers ─────────────────────────────────────────────────────────
|
# ── Reservoir helpers ─────────────────────────────────────────────────────────
|
||||||
|
|
||||||
proc toBinaryInput(bearing: float, velDir: float, velSpeed: float, hasVel: bool): BitVec80 =
|
proc toBinaryInput(bearing: float, velDir: float, velSpeed: float, hasVel: bool): BitVec80 =
|
||||||
## Binary-native encoding: ~5 bits per channel, ~13 total active bits
|
## Binary-native encoding using relative velocity direction for cross-round generalization.
|
||||||
|
## Bits 0-35: bearing (absolute, for range estimate)
|
||||||
|
## Bits 36-71: relative velocity direction = (velDir - bearing) mod 360
|
||||||
|
## This encodes lateral vs. radial motion regardless of which wall Walls is on.
|
||||||
|
## Bits 72-79: speed bands
|
||||||
|
## ponytail: relative-vel encoding; revert to absolute if aimer convergence regresses
|
||||||
result = [0'u64, 0'u64]
|
result = [0'u64, 0'u64]
|
||||||
|
|
||||||
# Bearing: bits 0-35 (36 bits, 10° bands)
|
# Bearing: bits 0-35 (36 bits, 10° bands)
|
||||||
@@ -344,9 +351,10 @@ proc toBinaryInput(bearing: float, velDir: float, velSpeed: float, hasVel: bool)
|
|||||||
result[word] = result[word] or (1'u64 shl bit)
|
result[word] = result[word] or (1'u64 shl bit)
|
||||||
|
|
||||||
if hasVel:
|
if hasVel:
|
||||||
# Velocity direction: bits 36-71 (36 bits, 10° bands)
|
# Relative velocity direction: (velDir - bearing + 360) mod 360
|
||||||
# Same scheme: 5 bits active
|
# Encodes lateral/radial motion direction independent of which wall Walls is on
|
||||||
let vNorm = (velDir + 180.0) / 10.0
|
let relVel = ((velDir - bearing) + 360.0) mod 360.0
|
||||||
|
let vNorm = relVel / 10.0
|
||||||
let vCenter = int(vNorm) mod 36
|
let vCenter = int(vNorm) mod 36
|
||||||
for offset in -2 .. 2:
|
for offset in -2 .. 2:
|
||||||
let idx = 36 + (vCenter + offset + 36) mod 36
|
let idx = 36 + (vCenter + offset + 36) mod 36
|
||||||
@@ -439,35 +447,10 @@ method run*(bot: SNNBot) =
|
|||||||
|
|
||||||
# Fire gate: only fire when error within adaptive dead zone
|
# Fire gate: only fire when error within adaptive dead zone
|
||||||
if err < adaptiveDeadZone:
|
if err < adaptiveDeadZone:
|
||||||
# Power selection with hysteresis
|
# ponytail: fixed power 3.0 vs Walls; adaptive economy adds noise vs learning
|
||||||
let desired =
|
if getEnergy() >= ENERGY_GUARD:
|
||||||
if getEnergy() < ENERGY_GUARD: 0.0
|
bot.lastBulletSpeed = bot.bulletSpeed
|
||||||
elif bot.hitRateEMA > 0.45: 3.0
|
|
||||||
elif bot.hitRateEMA > 0.30: 2.0
|
|
||||||
else: 1.0
|
|
||||||
if desired != bot.currentFirePower and desired > 0.0:
|
|
||||||
if desired == bot.pendingPower:
|
|
||||||
inc bot.powerChangeCounter
|
|
||||||
if bot.powerChangeCounter >= 8:
|
|
||||||
let oldPower = bot.currentFirePower
|
|
||||||
bot.currentFirePower = desired
|
|
||||||
bot.bulletSpeed = 20.0 - 3.0 * desired
|
|
||||||
bot.powerChangeCounter = 0
|
|
||||||
bot.pendingPower = 0.0
|
|
||||||
# old exemplars learned at different bullet speed — clear them
|
|
||||||
if oldPower != desired:
|
|
||||||
bot.res.count = 0
|
|
||||||
bot.res.nextSlot = 0
|
|
||||||
else:
|
|
||||||
bot.pendingPower = desired
|
|
||||||
bot.powerChangeCounter = 1
|
|
||||||
elif desired == 0.0:
|
|
||||||
discard setFire(0.0)
|
|
||||||
# skip EVALUATE so we don't corrupt learning with a non-shot
|
|
||||||
# stay in WAITING for next tick
|
|
||||||
if desired > 0.0:
|
|
||||||
discard setFire(bot.currentFirePower)
|
discard setFire(bot.currentFirePower)
|
||||||
# bulletsFired counted in onBulletFired event
|
|
||||||
bot.phase = EVALUATE
|
bot.phase = EVALUATE
|
||||||
else:
|
else:
|
||||||
discard setFire(0.0)
|
discard setFire(0.0)
|
||||||
@@ -475,7 +458,7 @@ method run*(bot: SNNBot) =
|
|||||||
of EVALUATE:
|
of EVALUATE:
|
||||||
# Predictive error signal: extrapolate enemy position at bullet impact time.
|
# Predictive error signal: extrapolate enemy position at bullet impact time.
|
||||||
if bot.hasLastPos:
|
if bot.hasLastPos:
|
||||||
let travelTime = bot.enemyDist / bot.bulletSpeed
|
let travelTime = bot.enemyDist / bot.lastBulletSpeed # use speed from fire tick
|
||||||
let velRad = degToRad(bot.velDirDeg)
|
let velRad = degToRad(bot.velDirDeg)
|
||||||
let futureX = bot.lastEnemyX + cos(velRad) * bot.velSpeed * travelTime
|
let futureX = bot.lastEnemyX + cos(velRad) * bot.velSpeed * travelTime
|
||||||
let futureY = bot.lastEnemyY + sin(velRad) * bot.velSpeed * travelTime
|
let futureY = bot.lastEnemyY + sin(velRad) * bot.velSpeed * travelTime
|
||||||
|
|||||||
@@ -36,15 +36,20 @@ let r = runBattle(@[snnbotDir, wallsbotDir], rounds = 10)
|
|||||||
# Per-round stats
|
# Per-round stats
|
||||||
echo "=== Per-round results ==="
|
echo "=== Per-round results ==="
|
||||||
var snnWins, wallsWins: int
|
var snnWins, wallsWins: int
|
||||||
|
var prevSnn, prevWalls: int
|
||||||
for rnd in r.rounds:
|
for rnd in r.rounds:
|
||||||
var snn, walls: BotRoundResult
|
var snn, walls: BotRoundResult
|
||||||
for br in rnd.results:
|
for br in rnd.results:
|
||||||
if br.name == "SNNBot": snn = br
|
if br.name == "SNNBot": snn = br
|
||||||
if br.name == "Walls (Nim)": walls = br
|
if br.name == "Walls (Nim)": walls = br
|
||||||
# rank 1 = winner; lower rank is better; default rank=0 means bot not found
|
let snnDelta = snn.score - prevSnn
|
||||||
let winner = if snn.rank > 0 and (walls.rank == 0 or snn.rank < walls.rank): "SNNBot" else: "Walls"
|
let wallsDelta = walls.score - prevWalls
|
||||||
|
prevSnn = snn.score
|
||||||
|
prevWalls = walls.score
|
||||||
|
# Use per-round score delta to determine round winner (rank field is cumulative battle rank)
|
||||||
|
let winner = if snnDelta > wallsDelta: "SNNBot" else: "Walls"
|
||||||
if winner == "SNNBot": inc snnWins else: inc wallsWins
|
if winner == "SNNBot": inc snnWins else: inc wallsWins
|
||||||
echo fmt"Round {rnd.round:2}: {winner} wins | SNNBot score={snn.score} survived={snn.survived} | Walls score={walls.score} survived={walls.survived}"
|
echo fmt"Round {rnd.round:2}: {winner} wins | SNNBot +{snnDelta}(={snn.score}) rank={snn.rank} survived={snn.survived} | Walls +{wallsDelta}(={walls.score}) rank={walls.rank} survived={walls.survived}"
|
||||||
|
|
||||||
# Summary
|
# Summary
|
||||||
echo ""
|
echo ""
|
||||||
|
|||||||
Reference in New Issue
Block a user