feat(PPO_Bot): weight persistence + background training (#17)

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-08-16 15:35:42 +02:00
parent eadd177d3b
commit 473d67f644
4 changed files with 269 additions and 9 deletions
+6 -4
View File
@@ -149,9 +149,11 @@ proc initACAdamStates(ac: ActorCritic): ACAdamStates =
result.cb3 = initAdamState(ac.critic.b3)
result.logStd = initAdamState(ac.logStd)
# Persistent Adam state — survives across ppoUpdate calls (lives in training module)
var gAdamStates: ACAdamStates
var gAdamInit = false
# Persistent Adam state — per-thread so training thread can call ppoUpdate safely.
# ponytail: threadvar resets Adam each new training thread; persist across calls
# within one thread. If Adam across rounds matters, embed state in TrainingArgs.
var gAdamStates {.threadvar.}: ACAdamStates
var gAdamInit {.threadvar.}: bool
# ── Gradient clipping ─────────────────────────────────────────────────────────
@@ -172,7 +174,7 @@ proc ppoUpdate*(ac: var ActorCritic;
entropyCoeff: float32 = 0.01'f32;
valueLossCoeff: float32 = 0.5'f32;
lr: float32 = 3e-4'f32;
maxGradNorm: float32 = 0.5'f32) =
maxGradNorm: float32 = 0.5'f32) {.gcsafe.} =
if buffer.len == 0: return
# Initialise Adam states once (persists across rounds)