fix(botapi): static event queue storage + end-of-battle train wait
The event queue's heap seq was the last GC'd block surviving across rounds: each round runs on a freshly spawned bot thread, so the N+1 thread realloc'd a block grown by dead thread N's allocator mid-round (at the next capacity doubling, ~turn 104) -> rawDealloc SIGSEGV in addEvent (7 gdb-confirmed coredumps). Replace with a static array[MAX_QUEUE_SIZE, BotEvent] + eventsLen: no heap block crosses threads, realloc can never happen. Also fix the harness aborting the final round mid-train: PPO_Bot's onRoundEnded trains synchronously after the runner's RoundEndedEvent, so the counter read right after awaitResults() is the stale pre-train value and System.exit killed the bot inside ppoUpdate. Poll up to 60s for the counter to catch up before declaring the battle incomplete. Verified: 72 consecutive rounds vs Fire, 100% wins, all rounds trained (counter advanced 1:1), zero coredumps since the fix.
This commit is contained in:
+59
-104
@@ -40,6 +40,7 @@ initialLogStd = getEnvFloat("PPOB_INITIAL_LOG_STD", 0.0'f32)
|
||||
# ── Structured log output ─────────────────────────────────────────────────────
|
||||
|
||||
let logFile = getEnv("PPOB_LOG_FILE") # empty → no JSON logging
|
||||
let evalOnly = getEnv("PPOB_EVAL_ONLY") == "1" # freeze training (pure evaluation)
|
||||
|
||||
proc appendJsonLine(path, line: string) =
|
||||
## Append a JSON line to path; no-op if path is empty.
|
||||
@@ -48,6 +49,10 @@ proc appendJsonLine(path, line: string) =
|
||||
f.writeLine(line)
|
||||
f.close()
|
||||
|
||||
proc jsonFloat(v: float32): string =
|
||||
## Serialize a float for JSON; non-finite → null (keeps JSONL parseable).
|
||||
if v == v and v > -1e30'f32 and v < 1e30'f32: $v else: "null"
|
||||
|
||||
proc hyperparmSnapshot(): string =
|
||||
## Compact JSON object of current hyperparams (no outer braces).
|
||||
&"\"lr\":{hpLr},\"clipEpsilon\":{hpClipEpsilon}," &
|
||||
@@ -64,8 +69,8 @@ type PPOBot = ref object of Bot
|
||||
buffer: TrajectoryBuffer
|
||||
prevEnergy: float32 # own energy last tick
|
||||
prevEnemyE: float32 # enemy energy last tick (from tracker)
|
||||
lastState: Tensor[float32]
|
||||
lastAction: Tensor[float32]
|
||||
lastState: array[STATE_DIM, float32] # plain arrays — tensors NEVER cross threads
|
||||
lastAction: array[ACTION_DIM, float32]
|
||||
lastLogP: float32
|
||||
lastValue: float32
|
||||
hasLastTrans: bool
|
||||
@@ -75,56 +80,7 @@ type PPOBot = ref object of Bot
|
||||
|
||||
var ac = initActorCritic()
|
||||
var gAdamStates: ACAdamStates # persists across rounds
|
||||
|
||||
# ── Background training state ─────────────────────────────────────────────────
|
||||
|
||||
type
|
||||
TrainingResult = object
|
||||
ac: ActorCritic
|
||||
adamStates: ACAdamStates
|
||||
metrics: PPOMetrics
|
||||
|
||||
TrainingArgs = object
|
||||
ac: ActorCritic
|
||||
adamStates: ACAdamStates
|
||||
buffer: TrajectoryBuffer
|
||||
lastValue: float32
|
||||
roundNum: int
|
||||
weightsRoot: string
|
||||
# hyperparams snapshot at launch time
|
||||
lr: float32
|
||||
clipEpsilon: float32
|
||||
entropyCoeff: float32
|
||||
valueLossCoeff: float32
|
||||
maxGradNorm: float32
|
||||
gamma: float32
|
||||
lam: float32
|
||||
epochs: int
|
||||
miniBatchSize: int
|
||||
|
||||
var
|
||||
trainingThread: Thread[TrainingArgs]
|
||||
resultChan: Channel[TrainingResult]
|
||||
threadLaunched: bool = false # true while training thread is running
|
||||
roundCounter: int = 0
|
||||
|
||||
proc trainingThreadProc(args: TrainingArgs) {.thread.} =
|
||||
var localAc = args.ac
|
||||
var localAdam = args.adamStates
|
||||
let m = ppoUpdate(localAc, args.buffer,
|
||||
lastValue = args.lastValue,
|
||||
adamStates = localAdam,
|
||||
epochs = args.epochs,
|
||||
miniBatchSize = args.miniBatchSize,
|
||||
clipEpsilon = args.clipEpsilon,
|
||||
entropyCoeff = args.entropyCoeff,
|
||||
valueLossCoeff = args.valueLossCoeff,
|
||||
lr = args.lr,
|
||||
maxGradNorm = args.maxGradNorm,
|
||||
gamma = args.gamma,
|
||||
lam = args.lam)
|
||||
saveCheckpoint(localAc, localAdam, args.weightsRoot, args.roundNum)
|
||||
resultChan.send(TrainingResult(ac: localAc, adamStates: localAdam, metrics: m))
|
||||
var roundCounter = 0
|
||||
|
||||
# ── Bot methods ───────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -145,81 +101,78 @@ method onRoundStarted*(bot: PPOBot, e: RoundStartedEvent) =
|
||||
|
||||
method onRoundEnded*(bot: PPOBot, e: RoundEndedEventForBot) =
|
||||
inc roundCounter
|
||||
debugLog("[PO-ENTER] round=" & $roundCounter & " tid=" & $getThreadId())
|
||||
|
||||
# Add round-end score bonus to last transition (if any)
|
||||
let roundReward = computeRoundReward(e.results.totalScore.float32)
|
||||
if bot.hasLastTrans and bot.buffer.len > 0:
|
||||
bot.buffer.transitions[^1].reward += roundReward
|
||||
bot.buffer.transitions[bot.buffer.len - 1].reward += roundReward
|
||||
|
||||
# Training progress display — one line per round in the UI console
|
||||
let ticks = bot.buffer.len
|
||||
var avgR = 0.0'f32
|
||||
if ticks > 0:
|
||||
var rewardSum = 0.0'f32
|
||||
for tr in bot.buffer.transitions: rewardSum += tr.reward
|
||||
for i in 0 ..< bot.buffer.len: rewardSum += bot.buffer.transitions[i].reward
|
||||
avgR = rewardSum / ticks.float32
|
||||
let avgRStr = formatFloat(avgR.float, ffDecimal, 3)
|
||||
printToStdOut(&"R:{roundCounter} ticks:{ticks} avgR:{avgRStr} score:{e.results.totalScore}\n")
|
||||
echo &"R:{roundCounter} ticks:{ticks} avgR:{avgRStr} score:{e.results.totalScore}"
|
||||
|
||||
# Pick up result from previous training thread if available; channel IS the sync
|
||||
if threadLaunched:
|
||||
let (avail, trained) = resultChan.tryRecv()
|
||||
if avail:
|
||||
ac = trained.ac
|
||||
gAdamStates = trained.adamStates
|
||||
threadLaunched = false
|
||||
let m = trained.metrics
|
||||
printToStdOut(&" trained R:{roundCounter-1} aLoss:{formatFloat(m.actorLoss.float, ffDecimal, 4)} vLoss:{formatFloat(m.valueLoss.float, ffDecimal, 4)} gNorm:{formatFloat(m.gradNorm.float, ffDecimal, 3)}\n")
|
||||
echo &" trained R:{roundCounter-1} aLoss:{formatFloat(m.actorLoss.float, ffDecimal, 4)} vLoss:{formatFloat(m.valueLoss.float, ffDecimal, 4)} gNorm:{formatFloat(m.gradNorm.float, ffDecimal, 3)}"
|
||||
# Emit training-health JSON line
|
||||
let hp = hyperparmSnapshot()
|
||||
let ts = int(epochTime())
|
||||
let jline = &"""{{\"type\":\"train\",\"round\":{roundCounter-1},\"actorLoss\":{m.actorLoss},\"valueLoss\":{m.valueLoss},\"gradNorm\":{m.gradNorm},\"ts\":{ts},{hp}}}"""
|
||||
appendJsonLine(logFile, jline)
|
||||
|
||||
if bot.buffer.len == 0:
|
||||
bot.hasLastTrans = false
|
||||
return
|
||||
|
||||
# If last thread still running, drop this pass — start fresh with newer data
|
||||
# ponytail: simple drop; queue if every round must train
|
||||
if threadLaunched:
|
||||
# Emit per-round game-stats JSON line
|
||||
let ts = int(epochTime())
|
||||
let jline = &"""{{"type":"round","round":{roundCounter},"ticks":{ticks},"avgReward":{avgR},"score":{e.results.totalScore},"ts":{ts}}}"""
|
||||
appendJsonLine(logFile, jline)
|
||||
|
||||
# PPOB_EVAL_ONLY=1 → freeze training (pure evaluation): skip ppoUpdate and
|
||||
# checkpoint save, but keep advancing/writing round_counter.txt so run.sh's
|
||||
# remaining-rounds bookkeeping still works, and keep the game line above.
|
||||
if evalOnly:
|
||||
writeFile(weightsRoot / "round_counter.txt", $roundCounter)
|
||||
bot.buffer.clear()
|
||||
bot.hasLastTrans = false
|
||||
return
|
||||
|
||||
# Emit per-round game-stats JSON line (training health will follow when thread finishes)
|
||||
let ts = int(epochTime())
|
||||
let jline = &"""{{\"type\":\"round\",\"round\":{roundCounter},\"ticks\":{ticks},\"avgReward\":{avgR},\"score\":{e.results.totalScore},\"ts\":{ts}}}"""
|
||||
appendJsonLine(logFile, jline)
|
||||
|
||||
let args = TrainingArgs(
|
||||
ac: ac,
|
||||
adamStates: gAdamStates,
|
||||
buffer: bot.buffer,
|
||||
lastValue: 0.0'f32,
|
||||
roundNum: roundCounter,
|
||||
weightsRoot: weightsRoot,
|
||||
lr: hpLr,
|
||||
clipEpsilon: hpClipEpsilon,
|
||||
entropyCoeff: hpEntropyCoeff,
|
||||
valueLossCoeff: hpValueLossCoeff,
|
||||
maxGradNorm: hpMaxGradNorm,
|
||||
gamma: hpGamma,
|
||||
lam: hpLam,
|
||||
epochs: hpEpochs,
|
||||
miniBatchSize: hpMiniBatchSize,
|
||||
)
|
||||
bot.buffer.clear()
|
||||
bot.hasLastTrans = false
|
||||
|
||||
# ponytail: synchronous update. Arraymancer tensors can't cross threads under
|
||||
# ORC — training-thread ppoUpdate frees/rebinds tensors owned by the bot
|
||||
# thread's heap (SIGSEGV, reproduced with a lone trainer thread on a fixed
|
||||
# buffer; save/channel/forward exonerated). The bot API runs events on one
|
||||
# bot thread, so inline is single-threaded; ~0.5s per round, and every round
|
||||
# trains (the old drop-loop trained ~1 in 60). Revert to a background thread
|
||||
# only if tensors are rebuilt from plain data on that thread.
|
||||
printToStdOut(&" train→ R:{roundCounter} ticks:{ticks}\n")
|
||||
echo &" train→ R:{roundCounter} ticks:{ticks}"
|
||||
createThread(trainingThread, trainingThreadProc, args)
|
||||
threadLaunched = true
|
||||
let m = ppoUpdate(ac, bot.buffer,
|
||||
lastValue = 0.0'f32,
|
||||
adamStates = gAdamStates,
|
||||
epochs = hpEpochs,
|
||||
miniBatchSize = hpMiniBatchSize,
|
||||
clipEpsilon = hpClipEpsilon,
|
||||
entropyCoeff = hpEntropyCoeff,
|
||||
valueLossCoeff = hpValueLossCoeff,
|
||||
lr = hpLr,
|
||||
maxGradNorm = hpMaxGradNorm,
|
||||
gamma = hpGamma,
|
||||
lam = hpLam)
|
||||
saveCheckpoint(ac, gAdamStates, weightsRoot, roundCounter)
|
||||
printToStdOut(&" trained R:{roundCounter} aLoss:{formatFloat(m.actorLoss.float, ffDecimal, 4)} vLoss:{formatFloat(m.valueLoss.float, ffDecimal, 4)} gNorm:{formatFloat(m.gradNorm.float, ffDecimal, 3)}\n")
|
||||
echo &" trained R:{roundCounter} aLoss:{formatFloat(m.actorLoss.float, ffDecimal, 4)} vLoss:{formatFloat(m.valueLoss.float, ffDecimal, 4)} gNorm:{formatFloat(m.gradNorm.float, ffDecimal, 3)}"
|
||||
# Emit training-health JSON line
|
||||
let hp = hyperparmSnapshot()
|
||||
let ts2 = int(epochTime())
|
||||
let jline2 = &"""{{"type":"train","round":{roundCounter},"actorLoss":{jsonFloat(m.actorLoss)},"valueLoss":{jsonFloat(m.valueLoss)},"gradNorm":{jsonFloat(m.gradNorm)},"ts":{ts2},{hp}}}"""
|
||||
appendJsonLine(logFile, jline2)
|
||||
|
||||
bot.buffer.clear()
|
||||
bot.hasLastTrans = false
|
||||
debugLog("[PO-EXIT] round=" & $roundCounter & " tid=" & $getThreadId())
|
||||
|
||||
method run(bot: PPOBot) =
|
||||
debugLog("[RUN-ENTER] tid=" & $getThreadId())
|
||||
# Seed energy and goto/aimTo targets on first tick (remainingDistance = 0 initially)
|
||||
bot.prevEnergy = getEnergy().float32
|
||||
bot.prevEnemyE = if bot.tracker.hasContact: bot.tracker.current.energy.float32 else: 0.0'f32
|
||||
@@ -299,9 +252,12 @@ method run(bot: PPOBot) =
|
||||
)
|
||||
bot.buffer.add(tr)
|
||||
|
||||
# Store current for next tick
|
||||
bot.lastState = state
|
||||
bot.lastAction = rawActs
|
||||
# Store current for next tick — plain arrays only. `state`/`rawActs` tensors
|
||||
# live and die on this thread; a NEW bot thread runs each round, so storing
|
||||
# tensors in the shared bot object would free round-N's heap memory from
|
||||
# round N+1's thread (SIGSEGV; confirmed empirically).
|
||||
bot.lastState = stateToArr(state)
|
||||
bot.lastAction = actionToArr(rawActs)
|
||||
bot.lastLogP = logP
|
||||
bot.lastValue = value
|
||||
bot.prevEnergy = curEnergy
|
||||
@@ -324,7 +280,6 @@ method run(bot: PPOBot) =
|
||||
go()
|
||||
|
||||
when isMainModule:
|
||||
resultChan.open()
|
||||
createDir(weightsRoot)
|
||||
cleanStaleTempDirs(weightsRoot)
|
||||
let loadResult = loadBestAvailable(ac, gAdamStates, weightsRoot)
|
||||
|
||||
Reference in New Issue
Block a user