feat(PPO_Bot): command abstraction layer — goto/aimTo controllers (#24)

- Add gotoTick/aimToTick controller functions (#25)
- Update network dims: actor 5→6, state 42→44 (#26)
- Rewrite mapActions for 6-dim command space (#27)
- Delete stale weight files (shape mismatch)
- Fix existing tests for new signatures

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-08-17 19:08:31 +02:00
parent 56e0b306c9
commit cdde60d79f
39 changed files with 490 additions and 78 deletions
+10 -4
View File
@@ -1,4 +1,4 @@
## State vector builder — produces 42-float normalized tensor for PPO policy.
## State vector builder — produces 44-float normalized tensor for PPO policy.
## No bot API imports; takes plain BotState + EnemyTracker structs.
import std/math
@@ -16,10 +16,12 @@ type
gunHeat*: float64
arenaWidth*, arenaHeight*: float64
proc buildStateVector*(bot: BotStateData; enemy: EnemyTracker): Tensor[float32] =
## Build the 42-float normalized state tensor.
proc buildStateVector*(bot: BotStateData; enemy: EnemyTracker;
remainingGotoDistance: float64 = 0.0;
remainingGunAngle: float64 = 0.0): Tensor[float32] =
## Build the 44-float normalized state tensor.
## All values clipped to roughly [-1, 1] via division by physical maxima.
result = zeros[float32](42)
result = zeros[float32](44)
let aW = bot.arenaWidth
let aH = bot.arenaHeight
@@ -85,3 +87,7 @@ proc buildStateVector*(bot: BotStateData; enemy: EnemyTracker): Tensor[float32]
result[base + 2] = float32(enemy.history[i].direction / 360.0)
result[base + 3] = float32(enemy.history[i].speed / 8.0)
# else: remain 0.0 (pad)
# --- Goto controller inputs (indices 42-43) ---
result[42] = float32(remainingGotoDistance / diag)
result[43] = float32(remainingGunAngle / 180.0)