Merge branch 'worktree-agent-a393a8b9' (ticket #44 reward module)
This commit is contained in:
@@ -0,0 +1,49 @@
|
||||
## rewards.nim — Raw reward computation + running mean/variance normalizer.
|
||||
## Welford online algorithm; safe cold-start (0 or 1 samples).
|
||||
|
||||
import std/math
|
||||
|
||||
# ── Raw reward ────────────────────────────────────────────────────────────────
|
||||
|
||||
proc computeReward*(
|
||||
damageInflicted: float64 = 0.0, # fire power p of own shot that hit
|
||||
damageReceived: float64 = 0.0, # fire power p_e of enemy shot that hit
|
||||
wallHitTicks: int = 0, # ticks in wall contact this step
|
||||
wastedShotPower: float64 = 0.0, # fire power of shot that missed/hit wall
|
||||
win: bool = false,
|
||||
loss: bool = false
|
||||
): float64 =
|
||||
## Returns the raw (un-normalized) reward for one decision step.
|
||||
## Damage formula: 4p + 2(p-1) = 6p - 2 (matches Tank Royale bullet rules).
|
||||
let p = damageInflicted
|
||||
let pe = damageReceived
|
||||
if p > 0.0: result += 6.0 * p - 2.0
|
||||
if pe > 0.0: result -= 6.0 * pe - 2.0
|
||||
result -= 5.0 * wallHitTicks.float64
|
||||
if wastedShotPower > 0.0: result -= 0.1 * wastedShotPower
|
||||
if win: result += 20.0
|
||||
if loss: result -= 10.0
|
||||
|
||||
# ── Running normalizer (Welford) ──────────────────────────────────────────────
|
||||
|
||||
const NormEps = 1e-8
|
||||
|
||||
type
|
||||
RewardNormalizer* = object
|
||||
n*: int # samples seen
|
||||
mean*: float64
|
||||
m2*: float64 # sum of squared deviations (Welford M2)
|
||||
|
||||
proc update*(rn: var RewardNormalizer; r: float64) =
|
||||
rn.n += 1
|
||||
let delta = r - rn.mean
|
||||
rn.mean += delta / rn.n.float64
|
||||
let delta2 = r - rn.mean
|
||||
rn.m2 += delta * delta2
|
||||
|
||||
proc normalize*(rn: RewardNormalizer; r: float64): float64 =
|
||||
## Returns (r - mean) / (std + eps).
|
||||
## Cold start (n < 2): returns 0.0 to avoid NaN/inf.
|
||||
if rn.n < 2: return 0.0
|
||||
let variance = rn.m2 / rn.n.float64 # ponytail: population var; switch to n-1 if bias matters
|
||||
result = (r - rn.mean) / (sqrt(variance) + NormEps)
|
||||
Reference in New Issue
Block a user