feat(PPO_Bot): network forward pass + action mapping (#14)
Two-hidden-layer MLP actor-critic (42→64→64→5/1) with stochastic actorForward, logStd floor at -3, and BotAction mapper wired into the run() loop. Assert-based test suite covers shapes, finiteness, logStd collapse, and all action range bounds. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,33 @@
|
||||
## actions.nim — map raw network output to Tank Royale bot commands.
|
||||
|
||||
import arraymancer
|
||||
import std/math
|
||||
|
||||
type
|
||||
BotActions* = object
|
||||
targetSpeed*: float32
|
||||
turnRate*: float32
|
||||
gunTurnRate*: float32
|
||||
shouldFire*: bool
|
||||
firePower*: float32
|
||||
|
||||
proc mapActions*(rawActions: Tensor[float32], currentSpeed: float32, gunHeat: float32): BotActions =
|
||||
## rawActions: [5] tensor from actorForward.
|
||||
## Ranges:
|
||||
## targetSpeed ∈ [-8, 8]
|
||||
## turnRate ∈ [-(10 - 0.75|v|), (10 - 0.75|v|)]
|
||||
## gunTurnRate ∈ [-20, 20]
|
||||
## firePower ∈ [0.1, 3.0]
|
||||
let r0 = tanh(rawActions[0].float64).float32
|
||||
let r1 = tanh(rawActions[1].float64).float32
|
||||
let r2 = tanh(rawActions[2].float64).float32
|
||||
let r3 = tanh(rawActions[3].float64).float32
|
||||
let r4 = rawActions[4]
|
||||
|
||||
result.targetSpeed = r0 * 8.0'f32
|
||||
result.turnRate = r1 * (10.0'f32 - 0.75'f32 * abs(currentSpeed))
|
||||
result.gunTurnRate = r2 * 20.0'f32
|
||||
let fireDecision = r3
|
||||
result.shouldFire = fireDecision >= 0.0'f32 and gunHeat <= 0.0'f32
|
||||
# sigmoid(r4) * 2.9 + 0.1 → [0.1, 3.0]
|
||||
result.firePower = (1.0'f32 / (1.0'f32 + exp(-r4.float64).float32)) * 2.9'f32 + 0.1'f32
|
||||
Reference in New Issue
Block a user