feat(SAC_LSTM_Bot): campaign v2 levers — loss metrics, eval-mode gate, eval rotation + MA gating (part 1)
Levers 3, 4, 1 of the #57 sign-off (execution order 3->4->1), tracked in #59. - Lever 3 (#59): one JSONL line per trainPass in training_metrics.jsonl with exactly the scalars sacUpdate already exposes (SACMetrics: critic/actor/alpha losses + alpha, averaged per pass) plus epoch, buffer size (replay_buffer.len), cumulative steps and drained count. No trainer change needed. - Lever 4 (#59): sendTrainingMsg drops all training input while SACLSTM_EVAL_MODE=1 (existing #49 harness mechanism) — eval battles can neither pollute the replay buffer nor trigger gradient updates; one-time stderr notice at bot init. - Lever 1 (#59): sac_train.sh evaluates every SAC_EVAL_OPPONENTS entry per cycle (results carry opponent name in eval_log.jsonl); best-gating now uses a composite = mean over opponents of the last-5-evals moving average per opponent. best_score.txt format change: float composite replaces the single-opponent integer win rate semantics (retired). - Tests: metricsLine JSONL scalars + eval-mode suppression asserts. Refs: #59, #57
This commit is contained in:
@@ -12,7 +12,7 @@
|
||||
## Decisions Q1–Q14: Gitea #48.
|
||||
|
||||
import arraymancer except Linear
|
||||
import std/[locks, os, math, random, strutils]
|
||||
import std/[locks, os, math, random, strutils, times]
|
||||
import tankroyale_botapi # getBotName (#49 name-based opponent identity)
|
||||
import SAC_LSTM_Bot/network
|
||||
import SAC_LSTM_Bot/state # STATE_DIM
|
||||
@@ -216,8 +216,18 @@ proc pullWeights*(myVersion: var int; hidden: var int;
|
||||
hidden = gSharedSnap.hiddenDim
|
||||
true
|
||||
|
||||
proc evalModeActive*(): bool {.inline.} =
|
||||
## Lever 4 (#59): the harness's deterministic eval battles already run the bot
|
||||
## with SACLSTM_EVAL_MODE=1 (sac_train.sh eval_checkpoint, mechanism from #49).
|
||||
## While set, eval ticks must NOT feed the trainer — transitions would pollute
|
||||
## the replay buffer with eval-only data and trigger gradient updates.
|
||||
getEnv("SACLSTM_EVAL_MODE") == "1"
|
||||
|
||||
proc sendTrainingMsg*(msg: TrainingMsg): bool {.inline.} =
|
||||
## Bot-side enqueue (cap-256, drops on overflow per Q10). Thread-safe.
|
||||
## Lever 4 (#59): fully suppressed in eval mode — NewBattle drops too, so an
|
||||
## eval battle can neither add transitions nor clear/retarget the buffer.
|
||||
if evalModeActive(): return false
|
||||
gTrainChan.trySend(msg)
|
||||
|
||||
# ── Training state (testable without threads) ─────────────────────────────────
|
||||
@@ -256,6 +266,41 @@ proc initTrainState*(initial: FullSnap): TrainState =
|
||||
result.lastEnemyKey = ""
|
||||
result.nextSave = getSaveInterval()
|
||||
|
||||
# ── Training-loss metrics (campaign v2 lever 3, #59) ──────────────────────────
|
||||
|
||||
proc metricsFilePath*(): string =
|
||||
## Sits next to the weights dir's parent: SAC_LSTM_Bot/training_metrics.jsonl
|
||||
## under the #49 harness (weights live in SAC_LSTM_Bot/weights/).
|
||||
getWeightsPath().parentDir.parentDir / "training_metrics.jsonl"
|
||||
|
||||
proc metricsLine*(epoch: float64; stepCount, bufferLen, drained, gradSteps: int;
|
||||
m: SACMetrics): string =
|
||||
## One JSONL line with exactly the scalars SACTrainer.sacUpdate exposes
|
||||
## (#59 lever 3 — SACMetrics was already returned, no trainer change needed):
|
||||
## losses/alpha averaged over this pass's gradient steps, buffer size from
|
||||
## replay_buffer.len, cumulative step count and drained transition count.
|
||||
"{\"epoch\":" & $epoch &
|
||||
",\"steps\":" & $stepCount &
|
||||
",\"buffer_size\":" & $bufferLen &
|
||||
",\"drained\":" & $drained &
|
||||
",\"grad_steps\":" & $gradSteps &
|
||||
",\"critic_loss\":" & $m.criticLoss &
|
||||
",\"actor_loss\":" & $m.actorLoss &
|
||||
",\"alpha_loss\":" & $m.alphaLoss &
|
||||
",\"alpha\":" & $m.alpha & "}"
|
||||
|
||||
proc appendMetricsLine(st: TrainState; drained, gradSteps: int; m: SACMetrics) =
|
||||
## Lever 3 (#59): one append per trainPass (never per gradient step). Open,
|
||||
## write, close — cheap and crash-tolerant; a metrics failure never kills
|
||||
## training.
|
||||
try:
|
||||
let f = open(metricsFilePath(), fmAppend)
|
||||
f.writeLine(metricsLine(epochTime(), st.stepCount, st.buf.len,
|
||||
drained, gradSteps, m))
|
||||
f.close()
|
||||
except CatchableError:
|
||||
discard
|
||||
|
||||
proc handleTrainingMsg*(st: var TrainState; msg: TrainingMsg): bool =
|
||||
## Process one message. Returns false for Shutdown (caller stops).
|
||||
## Tensors are born HERE from the message's plain arrays — training thread only.
|
||||
@@ -284,11 +329,16 @@ proc trainPass*(st: var TrainState; drained: int) =
|
||||
if drained <= 0 or not st.buf.canSample:
|
||||
return
|
||||
let steps = drained * getUtdRatio() # Q2
|
||||
var gradSteps = 0
|
||||
var sumCritic, sumActor, sumAlphaLoss, sumAlpha = 0.0'f32
|
||||
for i in 1 .. steps:
|
||||
let seqs = st.buf.sampleSequences(getBatchSize())
|
||||
if seqs.len == 0:
|
||||
break
|
||||
discard sacUpdate(st.trainer, seqs)
|
||||
let m = sacUpdate(st.trainer, seqs)
|
||||
sumCritic += m.criticLoss; sumActor += m.actorLoss
|
||||
sumAlphaLoss += m.alphaLoss; sumAlpha += m.alpha
|
||||
inc gradSteps
|
||||
inc st.stepCount
|
||||
# Save check INSIDE the step loop (#56 launch finding): at production sizes
|
||||
# (hidden 256 ⇒ ~1 s/step) a drain burst queues minutes of steps; checking
|
||||
@@ -299,6 +349,13 @@ proc trainPass*(st: var TrainState; drained: int) =
|
||||
st.nextSave += getSaveInterval()
|
||||
var full = packFull(st.trainer)
|
||||
discard gSaveChan.trySend(move(full)) # cap-1: drop if I/O thread is busy (Q5)
|
||||
if gradSteps > 0:
|
||||
# Lever 3 (#59): one metrics line per pass, losses averaged over its steps.
|
||||
appendMetricsLine(st, drained, gradSteps, SACMetrics(
|
||||
criticLoss: sumCritic / gradSteps.float32,
|
||||
actorLoss: sumActor / gradSteps.float32,
|
||||
alphaLoss: sumAlphaLoss / gradSteps.float32,
|
||||
alpha: sumAlpha / gradSteps.float32))
|
||||
# Publish latest actor (Q7): in-place write under the lock, bump version.
|
||||
withLock(gWeightLock):
|
||||
assert gSharedSnap.hiddenDim == st.trainer.actor.hiddenDim,
|
||||
@@ -382,6 +439,10 @@ proc initIntegration*() =
|
||||
initLock(gWeightLock)
|
||||
gTrainChan.open(256) # Q10 cap-256
|
||||
gSaveChan.open(1) # Q5 cap-1
|
||||
# Lever 4 (#59): one-time visibility for the suppression gate (see
|
||||
# sendTrainingMsg) — the eval bot trains nothing by design.
|
||||
if evalModeActive():
|
||||
stderr.writeLine "[sac] SACLSTM_EVAL_MODE=1 — training input suppressed (lever 4, #59)"
|
||||
createThread(gTrainingThread, trainingThreadEntry)
|
||||
createThread(gIoThread, ioThreadEntry)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user