feat(SAC_LSTM_Bot): campaign v2 levers — aggression/anti-ram reward shaping + stability knob overrides (part 2)
Lever 2 (#59): x1.25 aggression mult on damage dealt, flat +0.5 hit bonus, -3.0 per bot-bot collision (server deals RAM_DAMAGE=0.6 to both parties but only notifies the hitter), escalating proximity deterrent below 12% arena diagonal suppressed while dealing damage. Win/loss terminals unchanged and dominant. All weights TUNABLE consts marked ponytail. SACLSTM_REWARD_DEBUG=1 env-gated reward_debug.log for calibration greps. Lever 5 (#59): no code needed — SACLSTM_LR_ACTOR/LR_CRITIC/LR_ALPHA (3e-4) and SACLSTM_TARGET_ENTROPY (-4.0) were already env-overridable in training.nim. Smoke vs RamFire+Crazy (hidden=32, random init, isolated weights): 75 ram penalties, 381 charge events, hit bonuses firing, 0 crashes, metrics JSONL flowing. Tests: 8/8 suites green incl. new assert-level term math.
This commit is contained in:
@@ -11,17 +11,46 @@ template check(cond: bool, msg: string) =
|
||||
# ── computeReward ─────────────────────────────────────────────────────────────
|
||||
|
||||
block damageInflicted:
|
||||
# p=1: 6*1 - 2 = 4
|
||||
check abs(computeReward(damageInflicted = 1.0) - 4.0) < 1e-9, "p=1 damage = +4"
|
||||
# p=3: 6*3 - 2 = 16
|
||||
check abs(computeReward(damageInflicted = 3.0) - 16.0) < 1e-9, "p=3 damage = +16"
|
||||
# p=1: 1.25 * (6*1 - 2) = 5 (lever 2 aggression mult, #59)
|
||||
check abs(computeReward(damageInflicted = 1.0) - 5.0) < 1e-9, "p=1 damage = +5"
|
||||
# p=3: 1.25 * (6*3 - 2) = 20
|
||||
check abs(computeReward(damageInflicted = 3.0) - 20.0) < 1e-9, "p=3 damage = +20"
|
||||
# low-power spam stays unprofitable: 1.25*(6*0.1-2) < 0
|
||||
check computeReward(damageInflicted = 0.1) < 0.0, "p=0.1 spam still negative"
|
||||
|
||||
block damageReceived:
|
||||
# p_e=1: -(6*1 - 2) = -4
|
||||
# p_e=1: -(6*1 - 2) = -4 (unchanged by lever 2)
|
||||
check abs(computeReward(damageReceived = 1.0) - (-4.0)) < 1e-9, "p_e=1 received = -4"
|
||||
# p_e=3: -(6*3 - 2) = -16
|
||||
check abs(computeReward(damageReceived = 3.0) - (-16.0)) < 1e-9, "p_e=3 received = -16"
|
||||
|
||||
block hitBonus:
|
||||
# flat +0.5 per landed shot: p=1 hit -> 5.0 + 0.5
|
||||
check abs(computeReward(damageInflicted = 1.0, hitCount = 1) - 5.5) < 1e-9,
|
||||
"p=1 hit = +5.5"
|
||||
# two hits in one step: 1.25*(6*2-2) + 2*0.5 = 12.5 + 1.0 = 13.5
|
||||
check abs(computeReward(damageInflicted = 2.0, hitCount = 2) - 13.5) < 1e-9,
|
||||
"two hits = +13.5"
|
||||
|
||||
block ramTaken:
|
||||
# flat per victim collision (#59)
|
||||
check abs(computeReward(ramTakenCount = 1) - (-3.0)) < 1e-9, "ram taken x1 = -3"
|
||||
check abs(computeReward(ramTakenCount = 2) - (-6.0)) < 1e-9, "ram taken x2 = -6"
|
||||
|
||||
block chargeDeterrent:
|
||||
# zero-damage case at half threshold depth: -2 * (1 - 0.06/0.12) = -1
|
||||
let rHalf = computeReward(enemyDistFrac = 0.06)
|
||||
check abs(rHalf - (-1.0)) < 1e-9, "charge at frac 0.06 = -1"
|
||||
# at zero distance: full ceiling
|
||||
check abs(computeReward(enemyDistFrac = 0.0) - (-2.0)) < 1e-9, "charge at frac 0 = -2"
|
||||
# at/beyond threshold and no-contact sentinel: no penalty
|
||||
check abs(computeReward(enemyDistFrac = 0.12)) < 1e-9, "at threshold = 0"
|
||||
check abs(computeReward(enemyDistFrac = 0.5)) < 1e-9, "beyond threshold = 0"
|
||||
check abs(computeReward(enemyDistFrac = 2.0)) < 1e-9, "no-contact sentinel = 0"
|
||||
# suppressed while dealing damage that step (fighting back at close range is fine)
|
||||
let rFight = computeReward(damageInflicted = 1.0, enemyDistFrac = 0.06)
|
||||
check abs(rFight - 5.0) < 1e-9, "dealing damage cancels charge penalty"
|
||||
|
||||
block wallHit:
|
||||
check abs(computeReward(wallHitTicks = 1) - (-5.0)) < 1e-9, "1 wall tick = -5"
|
||||
|
||||
@@ -30,6 +59,7 @@ block wastedShot:
|
||||
check abs(computeReward(wastedShotPower = 2.0) - (-0.2)) < 1e-9, "missed p=2 = -0.2"
|
||||
|
||||
block winLoss:
|
||||
# terminal terms stay dominant over shaping (#59 scale discipline)
|
||||
check abs(computeReward(win = true) - 20.0) < 1e-9, "win = +20"
|
||||
check abs(computeReward(loss = true) - (-10.0)) < 1e-9, "loss = -10"
|
||||
|
||||
|
||||
Reference in New Issue
Block a user