feat(PPO_Bot): deterministic eval + fix logStd warm-start

- actorForward: deterministic param, uses mean-only when PPOB_EVAL_ONLY=1
  (eval was adding unit Gaussian noise to every action — unreliable scores)
- warm_start.py: log_std initialized to -1.0 (std≈0.37) instead of copying
  snapshot values (were 2.27-4.68 → std 9-108, completely drowning signal)
- training.env: LOG_STD_CEILING 0.0→-0.5 (cap exploration at std≈0.6)
This commit is contained in:
2026-08-20 15:18:07 +02:00
parent c834d2cbee
commit 0d35646dc9
4 changed files with 13 additions and 5 deletions
+5 -1
View File
@@ -31,13 +31,17 @@ for prefix in ("actor", "critic"):
unchanged = [
"actor_w2", "actor_w3", "actor_b1", "actor_b2", "actor_b3",
"critic_w2", "critic_w3", "critic_b1", "critic_b2", "critic_b3",
"log_std",
]
for name in unchanged:
data = np.load(SRC / f"{name}.npy")
np.save(DST / f"{name}.npy", data)
print(f" {name}: {data.shape} copied")
# Initialize log_std to -1.0 (std ≈ 0.37) — snapshot values (2.27–4.68) are too high for fine-tuning
log_std = np.full(6, -1.0, dtype=np.float32)
np.save(DST / "log_std.npy", log_std)
print(f" log_std: initialized to -1.0 (std≈0.37), shape={log_std.shape}")
# Copy unchanged Adam moments (all except w1, which were handled above)
unchanged_adam = [
"adam_aw2", "adam_cw2",