feat(PPO_Bot): weight persistence + background training (#17)
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -149,9 +149,11 @@ proc initACAdamStates(ac: ActorCritic): ACAdamStates =
|
||||
result.cb3 = initAdamState(ac.critic.b3)
|
||||
result.logStd = initAdamState(ac.logStd)
|
||||
|
||||
# Persistent Adam state — survives across ppoUpdate calls (lives in training module)
|
||||
var gAdamStates: ACAdamStates
|
||||
var gAdamInit = false
|
||||
# Persistent Adam state — per-thread so training thread can call ppoUpdate safely.
|
||||
# ponytail: threadvar resets Adam each new training thread; persist across calls
|
||||
# within one thread. If Adam across rounds matters, embed state in TrainingArgs.
|
||||
var gAdamStates {.threadvar.}: ACAdamStates
|
||||
var gAdamInit {.threadvar.}: bool
|
||||
|
||||
# ── Gradient clipping ─────────────────────────────────────────────────────────
|
||||
|
||||
@@ -172,7 +174,7 @@ proc ppoUpdate*(ac: var ActorCritic;
|
||||
entropyCoeff: float32 = 0.01'f32;
|
||||
valueLossCoeff: float32 = 0.5'f32;
|
||||
lr: float32 = 3e-4'f32;
|
||||
maxGradNorm: float32 = 0.5'f32) =
|
||||
maxGradNorm: float32 = 0.5'f32) {.gcsafe.} =
|
||||
if buffer.len == 0: return
|
||||
|
||||
# Initialise Adam states once (persists across rounds)
|
||||
|
||||
Reference in New Issue
Block a user