fix(PPO_Bot): SIGSEGV crash fixes + static buffers for thread safety

- bullets: seq[InFlightBullet] → array[4, InFlightBullet] + bulletCount
  (eliminates cross-thread heap realloc under ORC)
- hasFired: edge-triggered (cleared after state build, not level-triggered)
- round_counter parseInt: wrapped for empty/torn file → 0
- Static SVG + intent buffers to kill cross-thread heap realloc
- Tick-local alive/bulletData also fixed arrays

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-08-20 14:23:08 +02:00
parent 75e32e3315
commit 6ad51148f4
11 changed files with 239 additions and 52 deletions
+23
View File
@@ -149,4 +149,27 @@ block testPpoUpdateNormalReward:
check m.valueLoss == m.valueLoss, "valueLoss NaN on normal-reward round"
check m.gradNorm == m.gradNorm, "gradNorm NaN on normal-reward round"
# ── logStd ceiling: raw param must never drift above the collection clamp ─────
# Regression for the train/collection std mismatch: logStd starting above the
# ceiling (as in the warm-start snapshot) must be clamped back to [floor, ceiling]
# by the first Adam step, so recomputed logP matches the acting policy's std.
block testLogStdCeilingClamp:
randomize(45)
var ac = initActorCritic()
ac.logStd = newTensor[float32](ACTION_DIM).map(proc(v: float32): float32 = 3.0'f32)
var buf = initTrajectoryBuffer()
for _ in 0..<16:
let s = randomNormalTensor[float32](STATE_DIM)
let a = randomNormalTensor[float32](ACTION_DIM)
let lp = ac.computeLogProb(s, a)
buf.add(Transition(state: s.stateToArr, action: a.actionToArr, logProb: lp,
reward: 0.1'f32, value: 0.5'f32))
var adam: ACAdamStates
discard ppoUpdate(ac, buf, lastValue = 0.0'f32, adamStates = adam,
epochs = 1, miniBatchSize = 16)
for v in ac.logStd:
check v <= logStdCeiling + 1e-6'f32, "logStd above ceiling after ppoUpdate"
check v >= logStdFloor - 1e-6'f32, "logStd below floor after ppoUpdate"
echo "All tests passed"