tune(PPO_Bot): logStd=-2.0 (std≈0.135), entropy=0, ceiling=-1.0
Stochastic eval at std≈0.37 was 0/10 vs Corners (deterministic: 10/10). Warm-start policy is correct but brittle — any noise breaks it. - log_std initialized to -2.0 (std≈0.135) for moderate exploration - entropy_coeff=0.0 (no push toward exploration during fine-tuning) - logStd ceiling=-1.0 (cap at std≈0.37)
This commit is contained in:
@@ -4,7 +4,7 @@ PPOB_LOG_FILE=/home/davide/Projects/SirRoboGarage/tools/training_runner/logs/fir
|
||||
PPOB_LR=1e-4
|
||||
PPOB_UPDATE_INTERVAL=10
|
||||
PPOB_CLIP_EPSILON=0.2
|
||||
PPOB_ENTROPY_COEFF=0.001
|
||||
PPOB_ENTROPY_COEFF=0.0
|
||||
PPOB_VALUE_LOSS_COEFF=0.5
|
||||
PPOB_MAX_GRAD_NORM=0.5
|
||||
PPOB_GAMMA=0.99
|
||||
@@ -12,7 +12,7 @@ PPOB_LAM=0.95
|
||||
PPOB_EPOCHS=4
|
||||
PPOB_MINI_BATCH_SIZE=64
|
||||
PPOB_LOG_STD_FLOOR=-3.0
|
||||
PPOB_LOG_STD_CEILING=-0.5
|
||||
PPOB_LOG_STD_CEILING=-1.0
|
||||
PPOB_INITIAL_LOG_STD=-0.5
|
||||
|
||||
# PPOB_EVAL_ONLY=1 → freeze training (pure evaluation)
|
||||
|
||||
+3
-3
@@ -37,10 +37,10 @@ for name in unchanged:
|
||||
np.save(DST / f"{name}.npy", data)
|
||||
print(f" {name}: {data.shape} copied")
|
||||
|
||||
# Initialize log_std to -1.0 (std ≈ 0.37) — snapshot values (2.27–4.68) are too high for fine-tuning
|
||||
log_std = np.full(6, -1.0, dtype=np.float32)
|
||||
# Initialize log_std to -2.0 (std ≈ 0.135) — tighter than -1.0, proven workable
|
||||
log_std = np.full(6, -2.0, dtype=np.float32)
|
||||
np.save(DST / "log_std.npy", log_std)
|
||||
print(f" log_std: initialized to -1.0 (std≈0.37), shape={log_std.shape}")
|
||||
print(f" log_std: initialized to -2.0 (std≈0.135), shape={log_std.shape}")
|
||||
|
||||
# Copy unchanged Adam moments (all except w1, which were handled above)
|
||||
unchanged_adam = [
|
||||
|
||||
Reference in New Issue
Block a user