Merge branch 'worktree-agent-a393a8b9' (ticket #44 reward module)

This commit is contained in:
2026-08-20 23:41:23 +02:00
2 changed files with 128 additions and 0 deletions
+49
View File
@@ -0,0 +1,49 @@
## rewards.nim — Raw reward computation + running mean/variance normalizer.
## Welford online algorithm; safe cold-start (0 or 1 samples).
import std/math
# ── Raw reward ────────────────────────────────────────────────────────────────
proc computeReward*(
damageInflicted: float64 = 0.0, # fire power p of own shot that hit
damageReceived: float64 = 0.0, # fire power p_e of enemy shot that hit
wallHitTicks: int = 0, # ticks in wall contact this step
wastedShotPower: float64 = 0.0, # fire power of shot that missed/hit wall
win: bool = false,
loss: bool = false
): float64 =
## Returns the raw (un-normalized) reward for one decision step.
## Damage formula: 4p + 2(p-1) = 6p - 2 (matches Tank Royale bullet rules).
let p = damageInflicted
let pe = damageReceived
if p > 0.0: result += 6.0 * p - 2.0
if pe > 0.0: result -= 6.0 * pe - 2.0
result -= 5.0 * wallHitTicks.float64
if wastedShotPower > 0.0: result -= 0.1 * wastedShotPower
if win: result += 20.0
if loss: result -= 10.0
# ── Running normalizer (Welford) ──────────────────────────────────────────────
const NormEps = 1e-8
type
RewardNormalizer* = object
n*: int # samples seen
mean*: float64
m2*: float64 # sum of squared deviations (Welford M2)
proc update*(rn: var RewardNormalizer; r: float64) =
rn.n += 1
let delta = r - rn.mean
rn.mean += delta / rn.n.float64
let delta2 = r - rn.mean
rn.m2 += delta * delta2
proc normalize*(rn: RewardNormalizer; r: float64): float64 =
## Returns (r - mean) / (std + eps).
## Cold start (n < 2): returns 0.0 to avoid NaN/inf.
if rn.n < 2: return 0.0
let variance = rn.m2 / rn.n.float64 # ponytail: population var; switch to n-1 if bias matters
result = (r - rn.mean) / (sqrt(variance) + NormEps)