feat(PPO_Bot): persist Adam optimizer state and round counter across restarts (#35)

Save ACAdamStates (m/v tensors + t counters) as .npy files alongside
network weights in latest/ and checkpoint dirs; save round counter to
round_counter.txt. loadBestAvailable restores both on startup; fresh
start works unchanged when files are absent.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-08-18 15:40:54 +02:00
parent cdde60d79f
commit 0b17430735
4 changed files with 118 additions and 19 deletions
+6 -6
View File
@@ -71,9 +71,9 @@ proc computeGAE*(rewards, values: seq[float32];
# ── Manual Adam state ─────────────────────────────────────────────────────────
type
AdamState = object
m, v: Tensor[float32]
t: int
AdamState* = object
m*, v*: Tensor[float32]
t*: int
proc initAdamState(like: Tensor[float32]): AdamState =
result.m = zeros_like(like)
@@ -130,9 +130,9 @@ proc mlpBackward(mlp: MLP; fwd: MLPFwd; x: Tensor[float32];
type ACAdamStates* = object
## One AdamState per learnable tensor in ActorCritic.
aw1, ab1, aw2, ab2, aw3, ab3: AdamState # actor MLP
cw1, cb1, cw2, cb2, cw3, cb3: AdamState # critic MLP
logStd: AdamState
aw1*, ab1*, aw2*, ab2*, aw3*, ab3*: AdamState # actor MLP
cw1*, cb1*, cw2*, cb2*, cw3*, cb3*: AdamState # critic MLP
logStd*: AdamState
initialized*: bool
proc initACAdamStates*(ac: ActorCritic): ACAdamStates =