feat(SAC_LSTM_Bot): campaign v2 levers — loss metrics, eval-mode gate, eval rotation + MA gating (part 1)

Levers 3, 4, 1 of the #57 sign-off (execution order 3->4->1), tracked in #59.

- Lever 3 (#59): one JSONL line per trainPass in training_metrics.jsonl with
  exactly the scalars sacUpdate already exposes (SACMetrics: critic/actor/alpha
  losses + alpha, averaged per pass) plus epoch, buffer size (replay_buffer.len),
  cumulative steps and drained count. No trainer change needed.
- Lever 4 (#59): sendTrainingMsg drops all training input while SACLSTM_EVAL_MODE=1
  (existing #49 harness mechanism) — eval battles can neither pollute the replay
  buffer nor trigger gradient updates; one-time stderr notice at bot init.
- Lever 1 (#59): sac_train.sh evaluates every SAC_EVAL_OPPONENTS entry per cycle
  (results carry opponent name in eval_log.jsonl); best-gating now uses a
  composite = mean over opponents of the last-5-evals moving average per
  opponent. best_score.txt format change: float composite replaces the
  single-opponent integer win rate semantics (retired).
- Tests: metricsLine JSONL scalars + eval-mode suppression asserts.

Refs: #59, #57
This commit is contained in:
2026-08-22 18:58:47 +02:00
parent 2619ba06fc
commit a07e5305f5
3 changed files with 146 additions and 24 deletions
+57 -22
View File
@@ -5,7 +5,8 @@
# owns server lifecycle, opponent connection and dead-bot liveness detection
# through weights/round_counter.txt), samples opponents by weight per chunk,
# runs deterministic evaluation (SACLSTM_EVAL_MODE=1) every N chunks, and keeps
# the best checkpoint (weights/sac_best.zip) by eval win rate.
# the best checkpoint (weights/sac_best.zip) by a moving-average composite over
# the eval opponent set (campaign v2 lever 1, #59).
#
# Config (env vars):
# SAC_OPPONENTS "Name:weight,Name:weight,..." (default below)
@@ -13,7 +14,9 @@
# SAC_CHUNK_SIZE rounds per RunTraining battle (default 10)
# SAC_EVAL_INTERVAL eval every N chunks (default 2)
# SAC_EVAL_ROUNDS rounds per evaluation battle (default 10)
# SAC_EVAL_OPPONENT fixed eval opponent (default first opponent)
# SAC_EVAL_OPPONENTS comma-separated eval set (default Corners,Crazy,Target)
# — each cycle evaluates EVERY one; results all land in
# eval_log.jsonl (lines carry "opponent":"Name")
# SAC_MAX_CRASHES consecutive crashes before abort (default 5)
# SAC_LOG_FILE / SAC_EVAL_LOG_FILE (JSON-lines logs)
# SACLSTM_* passed through to the bot (UTD_RATIO, BATCH_SIZE, ...)
@@ -38,7 +41,8 @@ TOTAL_ROUNDS="${SAC_TOTAL_ROUNDS:-100}"
CHUNK_SIZE="${SAC_CHUNK_SIZE:-10}"
EVAL_INTERVAL="${SAC_EVAL_INTERVAL:-2}"
EVAL_ROUNDS="${SAC_EVAL_ROUNDS:-10}"
EVAL_OPPONENT="${SAC_EVAL_OPPONENT:-${OPPONENTS%%:*}}"
EVAL_OPPONENTS="${SAC_EVAL_OPPONENTS:-Corners,Crazy,Target}"
MA_WINDOW=5 # lever 1 (#59): per-opponent moving average over last N evals
MAX_CRASHES="${SAC_MAX_CRASHES:-5}"
LOG_FILE="${SAC_LOG_FILE:-$SCRIPT_DIR/training_log.jsonl}"
EVAL_LOG_FILE="${SAC_EVAL_LOG_FILE:-$SCRIPT_DIR/eval_log.jsonl}"
@@ -46,7 +50,7 @@ CLASSES_DIR="/tmp/opencode/sac_train_classes"
echo "=== SAC_LSTM_Bot training harness ==="
echo "Opponents: $OPPONENTS | budget: $TOTAL_ROUNDS rounds in chunks of $CHUNK_SIZE"
echo "Eval: every $EVAL_INTERVAL chunks, $EVAL_ROUNDS rounds vs $EVAL_OPPONENT"
echo "Eval: every $EVAL_INTERVAL chunks, $EVAL_ROUNDS rounds vs [$EVAL_OPPONENTS], MA-$MA_WINDOW composite best-gating"
echo "Weights: $SACLSTM_WEIGHTS_PATH"
# ── compile bot + java runner ─────────────────────────────────────────────────
@@ -76,29 +80,60 @@ run_battle() { # $1=opponent $2=rounds $3=log file
PPOB_LOG_FILE="$3" java -cp "$CLASSES_DIR:$JAR" RunTraining "$1" "$2"
}
# ── Lever 1 (#59): eval rotation + MA best-gating ─────────────────────────────
# best_score.txt FORMAT CHANGE: it used to store the single-opponent integer
# win rate (%); that semantics is retired. It now stores the COMPOSITE score —
# the mean over SAC_EVAL_OPPONENTS of each opponent's moving average (last
# MA_WINDOW eval win rates, %). sac_best.zip is rewritten only when the
# composite strictly improves.
ma_hist_file() { echo "$WEIGHTS_DIR/ma_history_$1.txt"; }
composite_of() { # reads one "w w w ..." history line per opponent on stdin
awk -v W="$MA_WINDOW" '
NF > 0 { n=NF; k=(n>W)?W:n; s=0; for(j=n-k+1;j<=n;j++) s+=$j; tot+=s/k; c++ }
END { if (c>0) printf "%.4f", tot/c; else print "-1" }'
}
eval_checkpoint() {
local tmp="$EVAL_LOG_FILE.tmp" wins rounds wr best
# ponytail: opponent names are split by whitespace — fine for Tank Royale bot
# names (no spaces); switch to a mapfile IFS=',\n' read if that ever changes.
local opps=(${EVAL_OPPONENTS//,/ })
local tmp="$EVAL_LOG_FILE.tmp" otmp opp wins rounds wr composite best
: > "$tmp"
echo ">>> [eval] $EVAL_ROUNDS deterministic rounds vs $EVAL_OPPONENT"
if ! SACLSTM_EVAL_MODE=1 run_battle "$EVAL_OPPONENT" "$EVAL_ROUNDS" "$tmp"; then
rm -f "$tmp"
echo ">>> [eval] crashed — keeping previous best"
return 0
fi
for opp in "${opps[@]}"; do
otmp="$EVAL_LOG_FILE.$opp.tmp"
: > "$otmp"
echo ">>> [eval] $EVAL_ROUNDS deterministic rounds vs $opp"
if ! SACLSTM_EVAL_MODE=1 run_battle "$opp" "$EVAL_ROUNDS" "$otmp"; then
rm -f "$otmp" "$tmp"
echo ">>> [eval] crashed vs $opp — keeping previous best"
return 0
fi
wins=$(grep -c '"win":true' "$otmp" || true)
rounds=$(grep -c '"type":"game"' "$otmp" || true)
if (( rounds == 0 )); then
rm -f "$otmp" "$tmp"
echo ">>> [eval] no results vs $opp — keeping previous best"
return 0
fi
wr=$(( 100 * wins / rounds ))
echo ">>> [eval] win rate: $wins/$rounds ($wr%) vs $opp"
cat "$otmp" >> "$tmp"; rm -f "$otmp"
# Per-opponent history: append this cycle's win rate, keep last MA_WINDOW.
printf '%s\n' "$(cat "$(ma_hist_file "$opp")" 2>/dev/null)" "$wr" \
| tail -n "$MA_WINDOW" | tr '\n' ' ' > "$(ma_hist_file "$opp")"
done
mv "$tmp" "$EVAL_LOG_FILE"
wins=$(grep -c '"win":true' "$EVAL_LOG_FILE" || true)
rounds=$(grep -c '"type":"game"' "$EVAL_LOG_FILE" || true)
(( rounds == 0 )) && { echo ">>> [eval] no results"; return 0; }
wr=$(( 100 * wins / rounds ))
echo ">>> [eval] win rate: $wins/$rounds ($wr%) vs $EVAL_OPPONENT"
composite=$(for opp in "${opps[@]}"; do cat "$(ma_hist_file "$opp")"; echo; done | composite_of)
# ponytail: best-score state is a plain file next to the checkpoint; survives
# harness restarts, no lock needed (single harness instance assumed).
best=-1
[ -f "$WEIGHTS_DIR/best_score.txt" ] && best=$(cat "$WEIGHTS_DIR/best_score.txt")
if (( wr > best )) && [ -f "$SACLSTM_WEIGHTS_PATH" ]; then
echo "$wr" > "$WEIGHTS_DIR/best_score.txt"
best=$(cat "$WEIGHTS_DIR/best_score.txt" 2>/dev/null)
[ -z "$best" ] && best=-1
if awk -v a="$composite" -v b="$best" 'BEGIN{exit !(a+0 > b+0)}' \
&& [ -f "$SACLSTM_WEIGHTS_PATH" ]; then
echo "$composite" > "$WEIGHTS_DIR/best_score.txt"
cp "$SACLSTM_WEIGHTS_PATH" "$WEIGHTS_DIR/sac_best.zip"
echo ">>> [eval] new best ($wr%) -> sac_best.zip"
echo ">>> [eval] new best composite ($composite) -> sac_best.zip"
fi
}
@@ -134,5 +169,5 @@ echo ">>> training complete: $NUM_CHUNKS chunks. Logs:"
echo " training: $LOG_FILE"
echo " eval: $EVAL_LOG_FILE"
[ -f "$WEIGHTS_DIR/sac_best.zip" ] && \
echo " best: $WEIGHTS_DIR/sac_best.zip ($(cat "$WEIGHTS_DIR/best_score.txt")%)"
echo " best: $WEIGHTS_DIR/sac_best.zip (composite $(cat "$WEIGHTS_DIR/best_score.txt"))"
exit 0