feat(SAC_LSTM_Bot): campaign v2 levers — loss metrics, eval-mode gate, eval rotation + MA gating (part 1)

Levers 3, 4, 1 of the #57 sign-off (execution order 3->4->1), tracked in #59.

- Lever 3 (#59): one JSONL line per trainPass in training_metrics.jsonl with
  exactly the scalars sacUpdate already exposes (SACMetrics: critic/actor/alpha
  losses + alpha, averaged per pass) plus epoch, buffer size (replay_buffer.len),
  cumulative steps and drained count. No trainer change needed.
- Lever 4 (#59): sendTrainingMsg drops all training input while SACLSTM_EVAL_MODE=1
  (existing #49 harness mechanism) — eval battles can neither pollute the replay
  buffer nor trigger gradient updates; one-time stderr notice at bot init.
- Lever 1 (#59): sac_train.sh evaluates every SAC_EVAL_OPPONENTS entry per cycle
  (results carry opponent name in eval_log.jsonl); best-gating now uses a
  composite = mean over opponents of the last-5-evals moving average per
  opponent. best_score.txt format change: float composite replaces the
  single-opponent integer win rate semantics (retired).
- Tests: metricsLine JSONL scalars + eval-mode suppression asserts.

Refs: #59, #57
This commit is contained in:
2026-08-22 18:58:47 +02:00
parent 2619ba06fc
commit a07e5305f5
3 changed files with 146 additions and 24 deletions
+26
View File
@@ -122,3 +122,29 @@ block:
assert alpha == t0.alpha()
discard tc1
echo "PASS flat snapshot pack/unpack round-trip"
# ── 5. Lever 3 (#59): metricsLine emits exactly the exposed trainer scalars ───
block:
let line = metricsLine(1787394115.123, 42, 500, 20, 20,
SACMetrics(criticLoss: 0.5'f32, actorLoss: -1.5'f32,
alphaLoss: 0.25'f32, alpha: 2.0'f32))
let j = parseJson(line) # throws on malformed JSONL
assert j["steps"].getInt() == 42 and j["buffer_size"].getInt() == 500
assert j["drained"].getInt() == 20 and j["grad_steps"].getInt() == 20
assert abs(j["critic_loss"].getFloat() - 0.5) < 1e-3
assert abs(j["actor_loss"].getFloat() + 1.5) < 1e-3
assert abs(j["alpha_loss"].getFloat() - 0.25) < 1e-3
assert abs(j["alpha"].getFloat() - 2.0) < 1e-3
assert j["epoch"].getFloat() > 1e9
echo "PASS metricsLine JSONL scalars"
# ── 6. Lever 4 (#59): SACLSTM_EVAL_MODE=1 suppresses training input ───────────
block:
putEnv("SACLSTM_EVAL_MODE", "1")
# Gate fires before any channel traffic: false = dropped, nothing enqueued.
assert not sendTrainingMsg(TrainingMsg(kind: tmkTransition)),
"eval mode must drop transitions"
assert not sendTrainingMsg(TrainingMsg(kind: tmkNewBattle, enemyId: 7)),
"eval mode must drop NewBattle (no buffer clears from eval)"
delEnv("SACLSTM_EVAL_MODE")
echo "PASS eval-mode training-input suppression"