feat(PPO_Bot): enemy-centered action space + reward shaping for 100% vs Target

- actions.nim: goto/aimTo coordinates now offset from enemy position
  (enemyX + tanh(raw) * scale) instead of absolute arena coords
  (sigmoid(raw) * arenaSize). Initial random policy defaults to
  approaching and aiming at enemy.
- training.nim: added dense reward shaping (distance closeness +
  gun bearing) to computeTickReward, doubled round reward scaling.
- PPO_Bot.nim: passes enemy position to mapActions, computes
  gun-to-enemy bearing for reward shaping.

Result: 100/100 win rate vs Target with frozen weights.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-08-18 17:40:59 +02:00
parent 0b17430735
commit 766b9e03ee
4 changed files with 164 additions and 37 deletions
+10 -6
View File
@@ -21,13 +21,17 @@ type
proc mapActions*(rawActions: Tensor[float32],
gunHeat: float,
arenaWidth, arenaHeight: float,
botX, botY, direction, speed, gunDirection: float): BotActions =
botX, botY, direction, speed, gunDirection: float,
enemyX, enemyY: float): BotActions =
## rawActions: [6] tensor from actorForward.
## Dims 0–1: goto x/y, 2–3: aimTo x/y, 4: fire decision, 5: fire power.
let gotoX = sigmoid(rawActions[0].float) * arenaWidth
let gotoY = sigmoid(rawActions[1].float) * arenaHeight
let aimToX = sigmoid(rawActions[2].float) * arenaWidth
let aimToY = sigmoid(rawActions[3].float) * arenaHeight
## Dims 0–1: goto x/y offset from enemy, 2–3: aimTo x/y offset from enemy,
## 4: fire decision, 5: fire power.
## Enemy-centred mapping: tanh gives [-1,1]; scale by arena/4 (goto) and
## arena/8 (aimTo) so zero-init defaults the bot toward the enemy.
let gotoX = clamp(enemyX + tanh(rawActions[0].float) * arenaWidth * 0.25, 0.0, arenaWidth)
let gotoY = clamp(enemyY + tanh(rawActions[1].float) * arenaHeight * 0.25, 0.0, arenaHeight)
let aimToX = clamp(enemyX + tanh(rawActions[2].float) * arenaWidth * 0.125, 0.0, arenaWidth)
let aimToY = clamp(enemyY + tanh(rawActions[3].float) * arenaHeight * 0.125, 0.0, arenaHeight)
let fireDec = tanh(rawActions[4].float)
let fp = sigmoid(rawActions[5].float) * 2.9 + 0.1