Files
SirRoboGarage/PPO_Bot/PPO_Bot.nim
T
SirStone 64697f917e fix(botapi): static event queue storage + end-of-battle train wait
The event queue's heap seq was the last GC'd block surviving across
rounds: each round runs on a freshly spawned bot thread, so the N+1
thread realloc'd a block grown by dead thread N's allocator mid-round
(at the next capacity doubling, ~turn 104) -> rawDealloc SIGSEGV in
addEvent (7 gdb-confirmed coredumps). Replace with a static
array[MAX_QUEUE_SIZE, BotEvent] + eventsLen: no heap block crosses
threads, realloc can never happen.

Also fix the harness aborting the final round mid-train: PPO_Bot's
onRoundEnded trains synchronously after the runner's RoundEndedEvent,
so the counter read right after awaitResults() is the stale pre-train
value and System.exit killed the bot inside ppoUpdate. Poll up to 60s
for the counter to catch up before declaring the battle incomplete.

Verified: 72 consecutive rounds vs Fire, 100% wins, all rounds trained
(counter advanced 1:1), zero coredumps since the fix.
2026-08-19 03:12:22 +02:00

294 lines
13 KiB
Nim

## PPO_Bot — enemy tracker + state vector wired into the game loop.
## Training: trajectory collected per tick, PPO update in background thread.
import std/[os, strformat, strutils, math, times]
import arraymancer
import tankroyale_botapi
import network
import actions
import training
import weights
import ./enemy_tracker
import ./state_vector
# ── Hyperparameters from env vars (PPOB_ prefix) ─────────────────────────────
# All optional; defaults match ppoUpdate signature in training.nim.
proc getEnvFloat(name: string, default: float32): float32 =
let v = getEnv(name)
if v.len == 0: default else: parseFloat(v).float32
proc getEnvInt(name: string, default: int): int =
let v = getEnv(name)
if v.len == 0: default else: parseInt(v)
var
hpLr: float32 = getEnvFloat("PPOB_LR", 3e-4'f32)
hpClipEpsilon: float32 = getEnvFloat("PPOB_CLIP_EPSILON", 0.2'f32)
hpEntropyCoeff: float32 = getEnvFloat("PPOB_ENTROPY_COEFF", 0.01'f32)
hpValueLossCoeff: float32 = getEnvFloat("PPOB_VALUE_LOSS_COEFF", 0.5'f32)
hpMaxGradNorm: float32 = getEnvFloat("PPOB_MAX_GRAD_NORM", 0.5'f32)
hpGamma: float32 = getEnvFloat("PPOB_GAMMA", 0.99'f32)
hpLam: float32 = getEnvFloat("PPOB_LAM", 0.95'f32)
hpEpochs: int = getEnvInt("PPOB_EPOCHS", 4)
hpMiniBatchSize: int = getEnvInt("PPOB_MINI_BATCH_SIZE", 64)
# Wire logStd tunable params into network module vars (read before initActorCritic)
logStdFloor = getEnvFloat("PPOB_LOG_STD_FLOOR", -3.0'f32)
initialLogStd = getEnvFloat("PPOB_INITIAL_LOG_STD", 0.0'f32)
# ── Structured log output ─────────────────────────────────────────────────────
let logFile = getEnv("PPOB_LOG_FILE") # empty → no JSON logging
let evalOnly = getEnv("PPOB_EVAL_ONLY") == "1" # freeze training (pure evaluation)
proc appendJsonLine(path, line: string) =
## Append a JSON line to path; no-op if path is empty.
if path.len == 0: return
let f = open(path, fmAppend)
f.writeLine(line)
f.close()
proc jsonFloat(v: float32): string =
## Serialize a float for JSON; non-finite → null (keeps JSONL parseable).
if v == v and v > -1e30'f32 and v < 1e30'f32: $v else: "null"
proc hyperparmSnapshot(): string =
## Compact JSON object of current hyperparams (no outer braces).
&"\"lr\":{hpLr},\"clipEpsilon\":{hpClipEpsilon}," &
&"\"entropyCoeff\":{hpEntropyCoeff},\"valueLossCoeff\":{hpValueLossCoeff}," &
&"\"maxGradNorm\":{hpMaxGradNorm},\"gamma\":{hpGamma},\"lam\":{hpLam}," &
&"\"epochs\":{hpEpochs},\"miniBatchSize\":{hpMiniBatchSize}," &
&"\"logStdFloor\":{logStdFloor},\"initialLogStd\":{initialLogStd}"
const botJsonPath = currentSourcePath().parentDir / "PPO_Bot.json"
const weightsRoot = currentSourcePath().parentDir / "weights"
type PPOBot = ref object of Bot
tracker: EnemyTracker
buffer: TrajectoryBuffer
prevEnergy: float32 # own energy last tick
prevEnemyE: float32 # enemy energy last tick (from tracker)
lastState: array[STATE_DIM, float32] # plain arrays — tensors NEVER cross threads
lastAction: array[ACTION_DIM, float32]
lastLogP: float32
lastValue: float32
hasLastTrans: bool
lastActions: BotActions # previous tick's decoded actions (for state vector)
roundRewardSum: float32 # cumulative reward this round (for live display)
roundTicks: int # ticks this round
var ac = initActorCritic()
var gAdamStates: ACAdamStates # persists across rounds
var roundCounter = 0
# ── Bot methods ───────────────────────────────────────────────────────────────
method onScannedBot*(bot: PPOBot, e: ScannedBotEvent) =
bot.tracker.update(e.x, e.y, e.direction, e.speed, e.energy)
method onRoundStarted*(bot: PPOBot, e: RoundStartedEvent) =
setAdjustGunForBodyTurn(true)
setAdjustRadarForBodyTurn(true)
setAdjustRadarForGunTurn(true)
bot.tracker = initEnemyTracker()
bot.buffer = initTrajectoryBuffer()
bot.prevEnergy = 0.0'f32
bot.prevEnemyE = 0.0'f32
bot.hasLastTrans = false
bot.roundRewardSum = 0.0'f32
bot.roundTicks = 0
method onRoundEnded*(bot: PPOBot, e: RoundEndedEventForBot) =
inc roundCounter
debugLog("[PO-ENTER] round=" & $roundCounter & " tid=" & $getThreadId())
# Add round-end score bonus to last transition (if any)
let roundReward = computeRoundReward(e.results.totalScore.float32)
if bot.hasLastTrans and bot.buffer.len > 0:
bot.buffer.transitions[bot.buffer.len - 1].reward += roundReward
# Training progress display — one line per round in the UI console
let ticks = bot.buffer.len
var avgR = 0.0'f32
if ticks > 0:
var rewardSum = 0.0'f32
for i in 0 ..< bot.buffer.len: rewardSum += bot.buffer.transitions[i].reward
avgR = rewardSum / ticks.float32
let avgRStr = formatFloat(avgR.float, ffDecimal, 3)
printToStdOut(&"R:{roundCounter} ticks:{ticks} avgR:{avgRStr} score:{e.results.totalScore}\n")
echo &"R:{roundCounter} ticks:{ticks} avgR:{avgRStr} score:{e.results.totalScore}"
if bot.buffer.len == 0:
bot.hasLastTrans = false
return
# Emit per-round game-stats JSON line
let ts = int(epochTime())
let jline = &"""{{"type":"round","round":{roundCounter},"ticks":{ticks},"avgReward":{avgR},"score":{e.results.totalScore},"ts":{ts}}}"""
appendJsonLine(logFile, jline)
# PPOB_EVAL_ONLY=1 → freeze training (pure evaluation): skip ppoUpdate and
# checkpoint save, but keep advancing/writing round_counter.txt so run.sh's
# remaining-rounds bookkeeping still works, and keep the game line above.
if evalOnly:
writeFile(weightsRoot / "round_counter.txt", $roundCounter)
bot.buffer.clear()
bot.hasLastTrans = false
return
# ponytail: synchronous update. Arraymancer tensors can't cross threads under
# ORC — training-thread ppoUpdate frees/rebinds tensors owned by the bot
# thread's heap (SIGSEGV, reproduced with a lone trainer thread on a fixed
# buffer; save/channel/forward exonerated). The bot API runs events on one
# bot thread, so inline is single-threaded; ~0.5s per round, and every round
# trains (the old drop-loop trained ~1 in 60). Revert to a background thread
# only if tensors are rebuilt from plain data on that thread.
printToStdOut(&" train→ R:{roundCounter} ticks:{ticks}\n")
echo &" train→ R:{roundCounter} ticks:{ticks}"
let m = ppoUpdate(ac, bot.buffer,
lastValue = 0.0'f32,
adamStates = gAdamStates,
epochs = hpEpochs,
miniBatchSize = hpMiniBatchSize,
clipEpsilon = hpClipEpsilon,
entropyCoeff = hpEntropyCoeff,
valueLossCoeff = hpValueLossCoeff,
lr = hpLr,
maxGradNorm = hpMaxGradNorm,
gamma = hpGamma,
lam = hpLam)
saveCheckpoint(ac, gAdamStates, weightsRoot, roundCounter)
printToStdOut(&" trained R:{roundCounter} aLoss:{formatFloat(m.actorLoss.float, ffDecimal, 4)} vLoss:{formatFloat(m.valueLoss.float, ffDecimal, 4)} gNorm:{formatFloat(m.gradNorm.float, ffDecimal, 3)}\n")
echo &" trained R:{roundCounter} aLoss:{formatFloat(m.actorLoss.float, ffDecimal, 4)} vLoss:{formatFloat(m.valueLoss.float, ffDecimal, 4)} gNorm:{formatFloat(m.gradNorm.float, ffDecimal, 3)}"
# Emit training-health JSON line
let hp = hyperparmSnapshot()
let ts2 = int(epochTime())
let jline2 = &"""{{"type":"train","round":{roundCounter},"actorLoss":{jsonFloat(m.actorLoss)},"valueLoss":{jsonFloat(m.valueLoss)},"gradNorm":{jsonFloat(m.gradNorm)},"ts":{ts2},{hp}}}"""
appendJsonLine(logFile, jline2)
bot.buffer.clear()
bot.hasLastTrans = false
debugLog("[PO-EXIT] round=" & $roundCounter & " tid=" & $getThreadId())
method run(bot: PPOBot) =
debugLog("[RUN-ENTER] tid=" & $getThreadId())
# Seed energy and goto/aimTo targets on first tick (remainingDistance = 0 initially)
bot.prevEnergy = getEnergy().float32
bot.prevEnemyE = if bot.tracker.hasContact: bot.tracker.current.energy.float32 else: 0.0'f32
bot.lastActions.gotoX = getX()
bot.lastActions.gotoY = getY()
bot.lastActions.aimToX = getX()
bot.lastActions.aimToY = getY()
while isRunning():
bot.tracker.deadReckon()
setRadarTurnRate(bot.tracker.getRadarTurnRate(getX(), getY(), getDirection(), getRadarDirection()))
let botData = BotStateData(
x: getX(),
y: getY(),
direction: getDirection(),
speed: getSpeed(),
energy: getEnergy(),
gunDirection: getGunDirection(),
gunHeat: getGunHeat(),
arenaWidth: float64(getArenaWidth()),
arenaHeight: float64(getArenaHeight()),
)
let remainingGotoDistance = hypot(bot.lastActions.gotoX - botData.x,
bot.lastActions.gotoY - botData.y)
let remainingGunAngle = abs(normalizeRelativeAngle(
directionTo(botData.x, botData.y, bot.lastActions.aimToX, bot.lastActions.aimToY) -
botData.gunDirection))
let state = buildStateVector(botData, bot.tracker, remainingGotoDistance, remainingGunAngle)
let (rawActs, logP) = ac.actorForward(state)
let value = ac.criticForward(state)
let ex = if bot.tracker.hasContact: bot.tracker.current.x else: botData.arenaWidth / 2.0
let ey = if bot.tracker.hasContact: bot.tracker.current.y else: botData.arenaHeight / 2.0
let acts = mapActions(rawActs,
getGunHeat().float,
botData.arenaWidth, botData.arenaHeight,
botData.x, botData.y,
botData.direction, botData.speed, botData.gunDirection,
ex, ey)
# Compute tick reward from energy deltas + dense shaping
let curEnergy = getEnergy().float32
let curEnemyE = if bot.tracker.hasContact: bot.tracker.current.energy.float32 else: bot.prevEnemyE
let myDelta = curEnergy - bot.prevEnergy
let enemyDelta = curEnemyE - bot.prevEnemyE
let arenaDiag = float32(sqrt(botData.arenaWidth * botData.arenaWidth +
botData.arenaHeight * botData.arenaHeight))
let distEnemy = if bot.tracker.hasContact:
float32(hypot(bot.tracker.current.x - botData.x,
bot.tracker.current.y - botData.y))
else: arenaDiag
let gunToEnemy = if bot.tracker.hasContact:
abs(normalizeRelativeAngle(
arctan2(bot.tracker.current.y - botData.y,
bot.tracker.current.x - botData.x) * 180.0 / PI -
botData.gunDirection)).float32
else: 180.0'f32
let tickReward = computeTickReward(myDelta, enemyDelta,
distToEnemy = distEnemy,
maxDist = arenaDiag,
gunBearingAbs = gunToEnemy)
# Track running reward for in-game display
bot.roundRewardSum += tickReward
inc bot.roundTicks
# Finalise previous transition with the reward from this tick's state change
if bot.hasLastTrans:
let tr = Transition(
state: bot.lastState,
action: bot.lastAction,
logProb: bot.lastLogP,
reward: tickReward,
value: bot.lastValue,
)
bot.buffer.add(tr)
# Store current for next tick — plain arrays only. `state`/`rawActs` tensors
# live and die on this thread; a NEW bot thread runs each round, so storing
# tensors in the shared bot object would free round-N's heap memory from
# round N+1's thread (SIGSEGV; confirmed empirically).
bot.lastState = stateToArr(state)
bot.lastAction = actionToArr(rawActs)
bot.lastLogP = logP
bot.lastValue = value
bot.prevEnergy = curEnergy
bot.prevEnemyE = curEnemyE
bot.hasLastTrans = true
bot.lastActions = acts
setTargetSpeed(acts.targetSpeed.float)
setTurnRate(acts.turnRate.float)
setGunTurnRate(acts.gunTurnRate.float)
if acts.shouldFire:
discard setFire(acts.firePower.float)
# In-game training progress overlay
let avgR = if bot.roundTicks > 0: bot.roundRewardSum / bot.roundTicks.float32
else: 0.0'f32
let avgRStr = formatFloat(avgR.float, ffDecimal, 2)
drawText(&"R:{roundCounter} avg:{avgRStr}", getX(), getY() - 40.0)
go()
when isMainModule:
createDir(weightsRoot)
cleanStaleTempDirs(weightsRoot)
let loadResult = loadBestAvailable(ac, gAdamStates, weightsRoot)
if loadResult.loaded:
roundCounter = loadResult.roundNum
var bot = PPOBot(
tracker: initEnemyTracker(),
buffer: initTrajectoryBuffer(),
)
start(bot, botJsonPath)