feat(PPO_Bot): command abstraction layer — goto/aimTo controllers (#24)
- Add gotoTick/aimToTick controller functions (#25) - Update network dims: actor 5→6, state 42→44 (#26) - Rewrite mapActions for 6-dim command space (#27) - Delete stale weight files (shape mismatch) - Fix existing tests for new signatures Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
+13
-9
@@ -3,6 +3,10 @@
|
||||
import arraymancer
|
||||
import std/[math, random]
|
||||
|
||||
const
|
||||
STATE_DIM* = 44
|
||||
ACTION_DIM* = 6
|
||||
|
||||
type
|
||||
MLP* = object
|
||||
w1*, b1*: Tensor[float32] # [hidden, input], [hidden]
|
||||
@@ -12,7 +16,7 @@ type
|
||||
ActorCritic* = object
|
||||
actor*: MLP
|
||||
critic*: MLP
|
||||
logStd*: Tensor[float32] # [5] — one per action dim
|
||||
logStd*: Tensor[float32] # [ACTION_DIM] — one per action dim
|
||||
|
||||
proc initMLP*(inputDim, hiddenDim, outputDim: int): MLP =
|
||||
# Xavier/He-style init: scale weights by sqrt(2/fan_in)
|
||||
@@ -24,9 +28,9 @@ proc initMLP*(inputDim, hiddenDim, outputDim: int): MLP =
|
||||
result.b3 = zeros[float32](outputDim)
|
||||
|
||||
proc initActorCritic*(): ActorCritic =
|
||||
result.actor = initMLP(42, 64, 5)
|
||||
result.critic = initMLP(42, 64, 1)
|
||||
result.logStd = zeros[float32](5) # init to 0 → std=1
|
||||
result.actor = initMLP(STATE_DIM, 64, ACTION_DIM)
|
||||
result.critic = initMLP(STATE_DIM, 64, 1)
|
||||
result.logStd = zeros[float32](ACTION_DIM) # init to 0 → std=1
|
||||
|
||||
proc forward*(mlp: MLP, x: Tensor[float32]): Tensor[float32] =
|
||||
## x shape: [inputDim] (1D vector)
|
||||
@@ -35,15 +39,15 @@ proc forward*(mlp: MLP, x: Tensor[float32]): Tensor[float32] =
|
||||
result = mlp.w3 * h2 + mlp.b3
|
||||
|
||||
proc actorForward*(ac: ActorCritic, state: Tensor[float32]): tuple[actions: Tensor[float32], logProb: float32] =
|
||||
## state: [42]. Returns sampled actions [5] and sum log-prob.
|
||||
## state: [STATE_DIM]. Returns sampled actions [ACTION_DIM] and sum log-prob.
|
||||
let mean = ac.actor.forward(state)
|
||||
# Floor logStd at -3 before exp → min std ≈ 0.05
|
||||
var clampedLogStd = ac.logStd.map(proc(v: float32): float32 = max(v, -3.0'f32))
|
||||
let std = clampedLogStd.map(proc(v: float32): float32 = exp(v))
|
||||
|
||||
var actions = newTensor[float32](5)
|
||||
var actions = newTensor[float32](ACTION_DIM)
|
||||
var logP = 0.0'f32
|
||||
for i in 0..<5:
|
||||
for i in 0..<ACTION_DIM:
|
||||
let mu = mean[i]
|
||||
let s = std[i]
|
||||
let z = gauss(0.0'f64, 1.0'f64).float32
|
||||
@@ -55,7 +59,7 @@ proc actorForward*(ac: ActorCritic, state: Tensor[float32]): tuple[actions: Tens
|
||||
result = (actions: actions, logProb: logP)
|
||||
|
||||
proc criticForward*(ac: ActorCritic, state: Tensor[float32]): float32 =
|
||||
## state: [42]. Returns scalar value estimate.
|
||||
## state: [STATE_DIM]. Returns scalar value estimate.
|
||||
let val = ac.critic.forward(state)
|
||||
result = val[0]
|
||||
|
||||
@@ -65,7 +69,7 @@ proc computeLogProb*(ac: ActorCritic, state, action: Tensor[float32]): float32 =
|
||||
let logStdClamped = ac.logStd.map(proc(v: float32): float32 = max(v, -3.0'f32))
|
||||
let std = logStdClamped.map(proc(v: float32): float32 = exp(v))
|
||||
var logP = 0.0'f32
|
||||
for i in 0..<5:
|
||||
for i in 0..<ACTION_DIM:
|
||||
let mu = mean[i]
|
||||
let s = std[i]
|
||||
let diff = (action[i] - mu) / s
|
||||
|
||||
Reference in New Issue
Block a user