From 0d35646dc99fc40cc240f9be331b07f17b3ca8c3 Mon Sep 17 00:00:00 2001 From: Davide Cappellini Date: Thu, 20 Aug 2026 15:18:07 +0200 Subject: [PATCH] feat(PPO_Bot): deterministic eval + fix logStd warm-start MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - actorForward: deterministic param, uses mean-only when PPOB_EVAL_ONLY=1 (eval was adding unit Gaussian noise to every action — unreliable scores) - warm_start.py: log_std initialized to -1.0 (std≈0.37) instead of copying snapshot values (were 2.27-4.68 → std 9-108, completely drowning signal) - training.env: LOG_STD_CEILING 0.0→-0.5 (cap exploration at std≈0.6) --- PPO_Bot/PPO_Bot.nim | 2 +- PPO_Bot/network.nim | 8 ++++++-- tools/training_runner/training.env | 2 +- tools/warm_start.py | 6 +++++- 4 files changed, 13 insertions(+), 5 deletions(-) diff --git a/PPO_Bot/PPO_Bot.nim b/PPO_Bot/PPO_Bot.nim index 79dcf15..efb0a54 100644 --- a/PPO_Bot/PPO_Bot.nim +++ b/PPO_Bot/PPO_Bot.nim @@ -281,7 +281,7 @@ method run(bot: PPOBot) = # detection-tick pulse, and the next iteration's spawn check sees false — # one shot → exactly one bullet, even across deadReckon gaps. bot.tracker.current.hasFired = false - let (rawActs, logP) = ac.actorForward(state) + let (rawActs, logP) = ac.actorForward(state, deterministic = evalOnly) let value = ac.criticForward(state) let ex = if bot.tracker.hasContact: bot.tracker.current.x else: botData.arenaWidth / 2.0 let ey = if bot.tracker.hasContact: bot.tracker.current.y else: botData.arenaHeight / 2.0 diff --git a/PPO_Bot/network.nim b/PPO_Bot/network.nim index 4b21fd4..9089fd3 100644 --- a/PPO_Bot/network.nim +++ b/PPO_Bot/network.nim @@ -43,9 +43,13 @@ proc forward*(mlp: MLP, x: Tensor[float32]): Tensor[float32] = let h2 = tanh(mlp.w2 * h1 + mlp.b2) result = mlp.w3 * h2 + mlp.b3 -proc actorForward*(ac: ActorCritic, state: Tensor[float32]): tuple[actions: Tensor[float32], logProb: float32] = - ## state: [STATE_DIM]. Returns sampled actions [ACTION_DIM] and sum log-prob. +proc actorForward*(ac: ActorCritic, state: Tensor[float32], deterministic = false): tuple[actions: Tensor[float32], logProb: float32] = + ## state: [STATE_DIM]. Returns actions [ACTION_DIM] and sum log-prob. + ## deterministic=true: return mean only (no noise), logProb=0. let mean = ac.actor.forward(state) + if deterministic: + return (actions: mean, logProb: 0.0'f32) + # Floor logStd at logStdFloor before exp → min std ≈ exp(logStdFloor) var clampedLogStd = ac.logStd.map(proc(v: float32): float32 = clamp(v, logStdFloor, logStdCeiling)) let std = clampedLogStd.map(proc(v: float32): float32 = exp(v)) diff --git a/tools/training_runner/training.env b/tools/training_runner/training.env index a22278e..40e1c84 100644 --- a/tools/training_runner/training.env +++ b/tools/training_runner/training.env @@ -12,7 +12,7 @@ PPOB_LAM=0.95 PPOB_EPOCHS=4 PPOB_MINI_BATCH_SIZE=64 PPOB_LOG_STD_FLOOR=-3.0 -PPOB_LOG_STD_CEILING=0.0 +PPOB_LOG_STD_CEILING=-0.5 PPOB_INITIAL_LOG_STD=-0.5 # PPOB_EVAL_ONLY=1 → freeze training (pure evaluation) diff --git a/tools/warm_start.py b/tools/warm_start.py index 7cd4172..7c06f11 100644 --- a/tools/warm_start.py +++ b/tools/warm_start.py @@ -31,13 +31,17 @@ for prefix in ("actor", "critic"): unchanged = [ "actor_w2", "actor_w3", "actor_b1", "actor_b2", "actor_b3", "critic_w2", "critic_w3", "critic_b1", "critic_b2", "critic_b3", - "log_std", ] for name in unchanged: data = np.load(SRC / f"{name}.npy") np.save(DST / f"{name}.npy", data) print(f" {name}: {data.shape} copied") +# Initialize log_std to -1.0 (std ≈ 0.37) — snapshot values (2.27–4.68) are too high for fine-tuning +log_std = np.full(6, -1.0, dtype=np.float32) +np.save(DST / "log_std.npy", log_std) +print(f" log_std: initialized to -1.0 (std≈0.37), shape={log_std.shape}") + # Copy unchanged Adam moments (all except w1, which were handled above) unchanged_adam = [ "adam_aw2", "adam_cw2",