feat(PPO_Bot): multi-round transition accumulation (UPDATE_INTERVAL=10)

- Accumulate transitions across 10 rounds (~3000) before PPO update
  (was per-round ~300 — gradient estimates were far too noisy)
- training.nim: MAX_TRANSITIONS 4096→8192, done flag on transitions,
  GAE handles episode boundaries correctly
- PPO_Bot.nim: buffer persists across rounds, update every N rounds
- training.env: lr 5e-5→1e-4, entropy 0.001, UPDATE_INTERVAL=10
This commit is contained in:
2026-08-20 15:06:00 +02:00
parent 82eeb53e5c
commit fedab54bc0
3 changed files with 71 additions and 50 deletions
+19 -12
View File
@@ -14,7 +14,7 @@ import ./network
# threads. Tensors are rebuilt from the arrays on the consuming (training) thread.
const
MAX_TRANSITIONS* = 4096 # server rounds are 2000 ticks; headroom for config drift
MAX_TRANSITIONS* = 8192 # 10 rounds × ~300 ticks + headroom
type
Transition* = object
@@ -23,6 +23,7 @@ type
logProb*: float32
reward*: float32
value*: float32 # critic estimate at collection time
done*: bool # true at episode (round) boundary
TrajectoryBuffer* = object
transitions*: array[MAX_TRANSITIONS, Transition]
@@ -34,9 +35,8 @@ proc initTrajectoryBuffer*(): TrajectoryBuffer =
result = TrajectoryBuffer()
proc add*(buf: var TrajectoryBuffer, t: Transition) =
## ponytail: fixed 4096 cap — server rounds run 2000 ticks; if a round ever
## exceeds the cap new transitions are dropped (oldest kept). Raise the cap
## if arena rounds get longer.
## ponytail: fixed 8192 cap — 10 rounds × ~300 ticks with headroom. Drops
## new transitions when full. Raise cap if accumulation window grows.
if buf.len < MAX_TRANSITIONS:
buf.transitions[buf.len] = t
inc buf.len
@@ -76,21 +76,27 @@ proc computeRoundReward*(roundScore: float32): float32 =
# ── GAE ───────────────────────────────────────────────────────────────────────
proc computeGAE*(rewards, values: seq[float32];
dones: seq[bool];
lastValue: float32;
gamma: float32 = 0.99'f32;
lam: float32 = 0.95'f32):
tuple[advantages: seq[float32], returns: seq[float32]] =
## Generalised Advantage Estimation — reverse sweep.
## lastValue = 0 for natural episode end (death/win).
## Generalised Advantage Estimation — reverse sweep with episode boundaries.
## When done=true on transition t, bootstrap value and accumulated GAE are
## reset to 0 at that boundary (terminal state has no future value).
let n = rewards.len
var advantages = newSeq[float32](n)
var gaeAcc = 0.0'f32
var lastGae = 0.0'f32
for t in countdown(n - 1, 0):
let nextVal = if t == n - 1: lastValue else: values[t + 1]
let delta = rewards[t] + gamma * nextVal - values[t]
gaeAcc = delta + gamma * lam * gaeAcc
advantages[t] = gaeAcc
let nextVal: float32 =
if t == n - 1 or dones[t]: 0.0'f32
else: values[t + 1]
if t == n - 1 or dones[t]:
lastGae = 0.0'f32
let delta = rewards[t] + gamma * nextVal - values[t]
lastGae = delta + gamma * lam * lastGae
advantages[t] = lastGae
var returns = newSeq[float32](n)
for t in 0..<n:
@@ -229,7 +235,8 @@ proc ppoUpdate*(ac: var ActorCritic;
# 1. GAE
let rewards = buffer.transitions[0 ..< buffer.len].mapIt(it.reward)
let values = buffer.transitions[0 ..< buffer.len].mapIt(it.value)
let (advantages, returns) = computeGAE(rewards, values, lastValue, gamma = gamma, lam = lam)
let dones = buffer.transitions[0 ..< buffer.len].mapIt(it.done)
let (advantages, returns) = computeGAE(rewards, values, dones, lastValue, gamma = gamma, lam = lam)
# 2. Normalise advantages
let n = advantages.len.float32