diff --git a/SAC_LSTM_Bot/src/SAC_LSTM_Bot/integration.nim b/SAC_LSTM_Bot/src/SAC_LSTM_Bot/integration.nim index 8d3c3d6..aa9102c 100644 --- a/SAC_LSTM_Bot/src/SAC_LSTM_Bot/integration.nim +++ b/SAC_LSTM_Bot/src/SAC_LSTM_Bot/integration.nim @@ -290,6 +290,15 @@ proc trainPass*(st: var TrainState; drained: int) = break discard sacUpdate(st.trainer, seqs) inc st.stepCount + # Save check INSIDE the step loop (#56 launch finding): at production sizes + # (hidden 256 ⇒ ~1 s/step) a drain burst queues minutes of steps; checking + # only between passes meant the process died mid-loop before stepCount ever + # reached nextSave — zero checkpoints persisted for the whole campaign. + # Mid-loop checks + SAVE_INTERVAL≤20 (#54) keep saves ~20 s apart. + if st.stepCount >= st.nextSave: + st.nextSave += getSaveInterval() + var full = packFull(st.trainer) + discard gSaveChan.trySend(move(full)) # cap-1: drop if I/O thread is busy (Q5) # Publish latest actor (Q7): in-place write under the lock, bump version. withLock(gWeightLock): assert gSharedSnap.hiddenDim == st.trainer.actor.hiddenDim, @@ -297,10 +306,6 @@ proc trainPass*(st: var TrainState; drained: int) = var c = 0 packActor(st.trainer.actor, gSharedSnap.data, c) inc gSharedSnap.version - if st.stepCount >= st.nextSave: - st.nextSave += getSaveInterval() - var full = packFull(st.trainer) - discard gSaveChan.trySend(move(full)) # cap-1: drop if I/O thread is busy (Q5) # ── Threads ───────────────────────────────────────────────────────────────────