766b9e03ee
- actions.nim: goto/aimTo coordinates now offset from enemy position (enemyX + tanh(raw) * scale) instead of absolute arena coords (sigmoid(raw) * arenaSize). Initial random policy defaults to approaching and aiming at enemy. - training.nim: added dense reward shaping (distance closeness + gun bearing) to computeTickReward, doubled round reward scaling. - PPO_Bot.nim: passes enemy position to mapActions, computes gun-to-enemy bearing for reward shaping. Result: 100/100 win rate vs Target with frozen weights. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
50 lines
1.9 KiB
Nim
50 lines
1.9 KiB
Nim
## actions.nim — map raw network output to Tank Royale bot commands.
|
||
|
||
import arraymancer
|
||
import std/math
|
||
import ./controllers
|
||
|
||
func sigmoid(x: float): float = 1.0 / (1.0 + exp(-x))
|
||
|
||
type
|
||
BotActions* = object
|
||
targetSpeed*: float
|
||
turnRate*: float
|
||
gunTurnRate*: float
|
||
shouldFire*: bool
|
||
firePower*: float
|
||
gotoX*: float
|
||
gotoY*: float
|
||
aimToX*: float
|
||
aimToY*: float
|
||
|
||
proc mapActions*(rawActions: Tensor[float32],
|
||
gunHeat: float,
|
||
arenaWidth, arenaHeight: float,
|
||
botX, botY, direction, speed, gunDirection: float,
|
||
enemyX, enemyY: float): BotActions =
|
||
## rawActions: [6] tensor from actorForward.
|
||
## Dims 0–1: goto x/y offset from enemy, 2–3: aimTo x/y offset from enemy,
|
||
## 4: fire decision, 5: fire power.
|
||
## Enemy-centred mapping: tanh gives [-1,1]; scale by arena/4 (goto) and
|
||
## arena/8 (aimTo) so zero-init defaults the bot toward the enemy.
|
||
let gotoX = clamp(enemyX + tanh(rawActions[0].float) * arenaWidth * 0.25, 0.0, arenaWidth)
|
||
let gotoY = clamp(enemyY + tanh(rawActions[1].float) * arenaHeight * 0.25, 0.0, arenaHeight)
|
||
let aimToX = clamp(enemyX + tanh(rawActions[2].float) * arenaWidth * 0.125, 0.0, arenaWidth)
|
||
let aimToY = clamp(enemyY + tanh(rawActions[3].float) * arenaHeight * 0.125, 0.0, arenaHeight)
|
||
let fireDec = tanh(rawActions[4].float)
|
||
let fp = sigmoid(rawActions[5].float) * 2.9 + 0.1
|
||
|
||
let (ts, tr) = gotoTick(gotoX, gotoY, botX, botY, direction, speed)
|
||
let gtr = aimToTick(aimToX, aimToY, botX, botY, gunDirection)
|
||
|
||
result.gotoX = gotoX
|
||
result.gotoY = gotoY
|
||
result.aimToX = aimToX
|
||
result.aimToY = aimToY
|
||
result.targetSpeed = ts
|
||
result.turnRate = tr
|
||
result.gunTurnRate = gtr
|
||
result.shouldFire = fireDec >= 0.0 and gunHeat <= 0.0
|
||
result.firePower = fp
|