fix(botapi): static event queue storage + end-of-battle train wait

The event queue's heap seq was the last GC'd block surviving across
rounds: each round runs on a freshly spawned bot thread, so the N+1
thread realloc'd a block grown by dead thread N's allocator mid-round
(at the next capacity doubling, ~turn 104) -> rawDealloc SIGSEGV in
addEvent (7 gdb-confirmed coredumps). Replace with a static
array[MAX_QUEUE_SIZE, BotEvent] + eventsLen: no heap block crosses
threads, realloc can never happen.

Also fix the harness aborting the final round mid-train: PPO_Bot's
onRoundEnded trains synchronously after the runner's RoundEndedEvent,
so the counter read right after awaitResults() is the stale pre-train
value and System.exit killed the bot inside ppoUpdate. Poll up to 60s
for the counter to catch up before declaring the battle incomplete.

Verified: 72 consecutive rounds vs Fire, 100% wins, all rounds trained
(counter advanced 1:1), zero coredumps since the fix.
This commit is contained in:
2026-08-19 03:12:22 +02:00
parent 766b9e03ee
commit 64697f917e
66 changed files with 2927 additions and 127 deletions
+50 -4
View File
@@ -14,14 +14,15 @@ template check(cond: bool, msg: string) =
block testTickReward:
# I lost 2, enemy lost 10 → reward = -2 - (-10) = 8
# + default closeness shaping 0.01*(1-0/maxDist) = 0.01 (gunBearingAbs=180 → 0)
let r = computeTickReward(-2.0'f32, -10.0'f32)
check abs(r - 8.0'f32) < 1e-6'f32, "computeTickReward(-2, -10) == 8, got " & $r
check abs(r - 8.01'f32) < 1e-6'f32, "computeTickReward(-2, -10) == 8.01, got " & $r
# ── computeRoundReward ────────────────────────────────────────────────────────
block testRoundReward:
let r = computeRoundReward(350.0'f32)
check abs(r - 3.5'f32) < 1e-6'f32, "computeRoundReward(350) == 3.5, got " & $r
check abs(r - 7.0'f32) < 1e-6'f32, "computeRoundReward(350) == 7.0, got " & $r
# ── TrajectoryBuffer ──────────────────────────────────────────────────────────
@@ -29,7 +30,8 @@ block testBuffer:
var buf = initTrajectoryBuffer()
check buf.len == 0, "empty buffer len == 0"
let t1 = Transition(state: zeros[float32](STATE_DIM), action: zeros[float32](ACTION_DIM),
let t1 = Transition(state: zeros[float32](STATE_DIM).stateToArr,
action: zeros[float32](ACTION_DIM).actionToArr,
logProb: -1.0'f32, reward: 0.5'f32, value: 0.3'f32)
buf.add(t1)
buf.add(t1)
@@ -84,7 +86,7 @@ block testPpoUpdate:
let a = randomNormalTensor[float32](ACTION_DIM)
let lp = ac.computeLogProb(s, a)
let v = ac.criticForward(s)
buf.add(Transition(state: s, action: a, logProb: lp, reward: 0.1'f32, value: v))
buf.add(Transition(state: s.stateToArr, action: a.actionToArr, logProb: lp, reward: 0.1'f32, value: v))
var adam: ACAdamStates
discard ppoUpdate(ac, buf, lastValue = 0.0'f32, adamStates = adam, epochs = 2, miniBatchSize = 5)
@@ -100,4 +102,48 @@ block testPpoUpdate:
break
check changed, "actor w1 should change after ppoUpdate"
# ── ppoUpdate on constant-reward trajectory: zero-variance guard ─────────────
# A passive round has near-constant per-tick rewards; with constant values the
# GAE advantages are identical → zero variance. The normalization must not
# amplify/NaN on this — update must complete with finite losses.
block testPpoUpdateConstantReward:
randomize(43)
var ac = initActorCritic()
var buf = initTrajectoryBuffer()
for _ in 0..<64:
let s = randomNormalTensor[float32](STATE_DIM)
let a = randomNormalTensor[float32](ACTION_DIM)
let lp = ac.computeLogProb(s, a)
buf.add(Transition(state: s.stateToArr, action: a.actionToArr, logProb: lp,
reward: 0.05'f32, value: 0.5'f32)) # constant reward+value
var adam: ACAdamStates
let m = ppoUpdate(ac, buf, lastValue = 0.5'f32, adamStates = adam,
epochs = 2, miniBatchSize = 16)
check m.actorLoss == m.actorLoss, "actorLoss NaN on constant-reward round"
check m.valueLoss == m.valueLoss, "valueLoss NaN on constant-reward round"
check m.gradNorm == m.gradNorm, "gradNorm NaN on constant-reward round"
# ── ppoUpdate on normal-reward trajectory: finite losses ─────────────────────
block testPpoUpdateNormalReward:
randomize(44)
var ac = initActorCritic()
var buf = initTrajectoryBuffer()
for i in 0..<64:
let s = randomNormalTensor[float32](STATE_DIM)
let a = randomNormalTensor[float32](ACTION_DIM)
let lp = ac.computeLogProb(s, a)
let v = ac.criticForward(s)
let r = 0.05'f32 + 0.5'f32 * sin(float32(i))
buf.add(Transition(state: s.stateToArr, action: a.actionToArr, logProb: lp, reward: r, value: v))
var adam: ACAdamStates
let m = ppoUpdate(ac, buf, lastValue = 0.0'f32, adamStates = adam,
epochs = 2, miniBatchSize = 16)
check m.actorLoss == m.actorLoss, "actorLoss NaN on normal-reward round"
check m.valueLoss == m.valueLoss, "valueLoss NaN on normal-reward round"
check m.gradNorm == m.gradNorm, "gradNorm NaN on normal-reward round"
echo "All tests passed"