feat(PPO_Bot): enemy-centered action space + reward shaping for 100% vs Target

- actions.nim: goto/aimTo coordinates now offset from enemy position
  (enemyX + tanh(raw) * scale) instead of absolute arena coords
  (sigmoid(raw) * arenaSize). Initial random policy defaults to
  approaching and aiming at enemy.
- training.nim: added dense reward shaping (distance closeness +
  gun bearing) to computeTickReward, doubled round reward scaling.
- PPO_Bot.nim: passes enemy position to mapActions, computes
  gun-to-enemy bearing for reward shaping.

Result: 100/100 win rate vs Target with frozen weights.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-08-18 17:40:59 +02:00
parent 0b17430735
commit 766b9e03ee
4 changed files with 164 additions and 37 deletions
+19 -8
View File
@@ -35,13 +35,22 @@ proc len*(buf: TrajectoryBuffer): int =
# ── Reward helpers ─────────────────────────────────────────────────────────────
proc computeTickReward*(myEnergyDelta, enemyEnergyDelta: float32): float32 =
proc computeTickReward*(myEnergyDelta, enemyEnergyDelta: float32;
distToEnemy: float32 = 0.0'f32;
maxDist: float32 = 1.0'f32;
gunBearingAbs: float32 = 180.0'f32): float32 =
## Positive when we deal more damage than we receive.
result = myEnergyDelta - enemyEnergyDelta
## Dense shaping: closeness (0-0.01/tick) + aim quality (0-0.02/tick).
## ponytail: magnitudes 10x smaller than original to keep shaping as a nudge,
## not the dominant signal. Increase if bot ignores positioning entirely.
let sparseReward = myEnergyDelta - enemyEnergyDelta
let distReward = 0.01'f32 * (1.0'f32 - distToEnemy / maxDist)
let aimReward = 0.02'f32 * (1.0'f32 - gunBearingAbs / 180.0'f32)
result = sparseReward + distReward + aimReward
proc computeRoundReward*(roundScore: float32): float32 =
## Normalise round-end score to a rough ±3 range.
result = roundScore / 100.0'f32
## Normalise round-end score to a rough ±6 range (doubled win signal).
result = roundScore / 50.0'f32
# ── GAE ───────────────────────────────────────────────────────────────────────
@@ -178,7 +187,9 @@ proc ppoUpdate*(ac: var ActorCritic;
entropyCoeff: float32 = 0.01'f32;
valueLossCoeff: float32 = 0.5'f32;
lr: float32 = 3e-4'f32;
maxGradNorm: float32 = 0.5'f32): PPOMetrics {.gcsafe.} =
maxGradNorm: float32 = 0.5'f32;
gamma: float32 = 0.99'f32;
lam: float32 = 0.95'f32): PPOMetrics {.gcsafe.} =
if buffer.len == 0: return
var totalActorLoss = 0.0'f32
@@ -193,7 +204,7 @@ proc ppoUpdate*(ac: var ActorCritic;
# 1. GAE
let rewards = buffer.transitions.mapIt(it.reward)
let values = buffer.transitions.mapIt(it.value)
let (advantages, returns) = computeGAE(rewards, values, lastValue)
let (advantages, returns) = computeGAE(rewards, values, lastValue, gamma = gamma, lam = lam)
# 2. Normalise advantages
let n = advantages.len.float32
@@ -244,7 +255,7 @@ proc ppoUpdate*(ac: var ActorCritic;
let actorFwd = mlpForwardCached(ac.actor, tr.state)
let newMean = actorFwd.y # [ACTION_DIM]
let logStdClamped = ac.logStd.map(proc(v: float32): float32 = max(v, -3.0'f32))
let logStdClamped = ac.logStd.map(proc(v: float32): float32 = max(v, logStdFloor))
let std = logStdClamped.map(proc(v: float32): float32 = exp(v))
# New log prob
@@ -285,7 +296,7 @@ proc ppoUpdate*(ac: var ActorCritic;
# Also: d(newLogP)/d(logStd_i) when logStd_i > -3:
# = (action_i - mean_i)^2/std_i^2 - 1
for i in 0..<ACTION_DIM:
let isClamped = (ac.logStd[i] <= -3.0'f32)
let isClamped = (ac.logStd[i] <= logStdFloor)
if not isClamped:
let s = std[i]
let diff = (tr.action[i] - newMean[i]) / s