feat(PPO_Bot): multi-round transition accumulation (UPDATE_INTERVAL=10)
- Accumulate transitions across 10 rounds (~3000) before PPO update (was per-round ~300 — gradient estimates were far too noisy) - training.nim: MAX_TRANSITIONS 4096→8192, done flag on transitions, GAE handles episode boundaries correctly - PPO_Bot.nim: buffer persists across rounds, update every N rounds - training.env: lr 5e-5→1e-4, entropy 0.001, UPDATE_INTERVAL=10
This commit is contained in:
+19
-12
@@ -14,7 +14,7 @@ import ./network
|
||||
# threads. Tensors are rebuilt from the arrays on the consuming (training) thread.
|
||||
|
||||
const
|
||||
MAX_TRANSITIONS* = 4096 # server rounds are 2000 ticks; headroom for config drift
|
||||
MAX_TRANSITIONS* = 8192 # 10 rounds × ~300 ticks + headroom
|
||||
|
||||
type
|
||||
Transition* = object
|
||||
@@ -23,6 +23,7 @@ type
|
||||
logProb*: float32
|
||||
reward*: float32
|
||||
value*: float32 # critic estimate at collection time
|
||||
done*: bool # true at episode (round) boundary
|
||||
|
||||
TrajectoryBuffer* = object
|
||||
transitions*: array[MAX_TRANSITIONS, Transition]
|
||||
@@ -34,9 +35,8 @@ proc initTrajectoryBuffer*(): TrajectoryBuffer =
|
||||
result = TrajectoryBuffer()
|
||||
|
||||
proc add*(buf: var TrajectoryBuffer, t: Transition) =
|
||||
## ponytail: fixed 4096 cap — server rounds run 2000 ticks; if a round ever
|
||||
## exceeds the cap new transitions are dropped (oldest kept). Raise the cap
|
||||
## if arena rounds get longer.
|
||||
## ponytail: fixed 8192 cap — 10 rounds × ~300 ticks with headroom. Drops
|
||||
## new transitions when full. Raise cap if accumulation window grows.
|
||||
if buf.len < MAX_TRANSITIONS:
|
||||
buf.transitions[buf.len] = t
|
||||
inc buf.len
|
||||
@@ -76,21 +76,27 @@ proc computeRoundReward*(roundScore: float32): float32 =
|
||||
# ── GAE ───────────────────────────────────────────────────────────────────────
|
||||
|
||||
proc computeGAE*(rewards, values: seq[float32];
|
||||
dones: seq[bool];
|
||||
lastValue: float32;
|
||||
gamma: float32 = 0.99'f32;
|
||||
lam: float32 = 0.95'f32):
|
||||
tuple[advantages: seq[float32], returns: seq[float32]] =
|
||||
## Generalised Advantage Estimation — reverse sweep.
|
||||
## lastValue = 0 for natural episode end (death/win).
|
||||
## Generalised Advantage Estimation — reverse sweep with episode boundaries.
|
||||
## When done=true on transition t, bootstrap value and accumulated GAE are
|
||||
## reset to 0 at that boundary (terminal state has no future value).
|
||||
let n = rewards.len
|
||||
var advantages = newSeq[float32](n)
|
||||
var gaeAcc = 0.0'f32
|
||||
var lastGae = 0.0'f32
|
||||
|
||||
for t in countdown(n - 1, 0):
|
||||
let nextVal = if t == n - 1: lastValue else: values[t + 1]
|
||||
let delta = rewards[t] + gamma * nextVal - values[t]
|
||||
gaeAcc = delta + gamma * lam * gaeAcc
|
||||
advantages[t] = gaeAcc
|
||||
let nextVal: float32 =
|
||||
if t == n - 1 or dones[t]: 0.0'f32
|
||||
else: values[t + 1]
|
||||
if t == n - 1 or dones[t]:
|
||||
lastGae = 0.0'f32
|
||||
let delta = rewards[t] + gamma * nextVal - values[t]
|
||||
lastGae = delta + gamma * lam * lastGae
|
||||
advantages[t] = lastGae
|
||||
|
||||
var returns = newSeq[float32](n)
|
||||
for t in 0..<n:
|
||||
@@ -229,7 +235,8 @@ proc ppoUpdate*(ac: var ActorCritic;
|
||||
# 1. GAE
|
||||
let rewards = buffer.transitions[0 ..< buffer.len].mapIt(it.reward)
|
||||
let values = buffer.transitions[0 ..< buffer.len].mapIt(it.value)
|
||||
let (advantages, returns) = computeGAE(rewards, values, lastValue, gamma = gamma, lam = lam)
|
||||
let dones = buffer.transitions[0 ..< buffer.len].mapIt(it.done)
|
||||
let (advantages, returns) = computeGAE(rewards, values, dones, lastValue, gamma = gamma, lam = lam)
|
||||
|
||||
# 2. Normalise advantages
|
||||
let n = advantages.len.float32
|
||||
|
||||
Reference in New Issue
Block a user