feat(PPO_Bot): deterministic eval + fix logStd warm-start
- actorForward: deterministic param, uses mean-only when PPOB_EVAL_ONLY=1 (eval was adding unit Gaussian noise to every action — unreliable scores) - warm_start.py: log_std initialized to -1.0 (std≈0.37) instead of copying snapshot values (were 2.27-4.68 → std 9-108, completely drowning signal) - training.env: LOG_STD_CEILING 0.0→-0.5 (cap exploration at std≈0.6)
This commit is contained in:
+1
-1
@@ -281,7 +281,7 @@ method run(bot: PPOBot) =
|
||||
# detection-tick pulse, and the next iteration's spawn check sees false —
|
||||
# one shot → exactly one bullet, even across deadReckon gaps.
|
||||
bot.tracker.current.hasFired = false
|
||||
let (rawActs, logP) = ac.actorForward(state)
|
||||
let (rawActs, logP) = ac.actorForward(state, deterministic = evalOnly)
|
||||
let value = ac.criticForward(state)
|
||||
let ex = if bot.tracker.hasContact: bot.tracker.current.x else: botData.arenaWidth / 2.0
|
||||
let ey = if bot.tracker.hasContact: bot.tracker.current.y else: botData.arenaHeight / 2.0
|
||||
|
||||
Reference in New Issue
Block a user