#!/usr/bin/env bash # sac_train.sh — #49 training orchestration for SAC_LSTM_Bot. # # Drives chunked self-play via tools/training_runner/RunTraining.java (which # owns server lifecycle, opponent connection and dead-bot liveness detection # through weights/round_counter.txt), samples opponents by weight per chunk, # runs deterministic evaluation (SACLSTM_EVAL_MODE=1) every N chunks, and keeps # the best checkpoint (weights/sac_best.zip) by a moving-average composite over # the eval opponent set (campaign v2 lever 1, #59). # # Config (env vars): # SAC_OPPONENTS "Name:weight,Name:weight,..." (default below) # SAC_TOTAL_ROUNDS total training-round budget (default 100) # SAC_CHUNK_SIZE rounds per RunTraining battle (default 10) # SAC_EVAL_INTERVAL eval every N chunks (default 2) # SAC_EVAL_ROUNDS rounds per evaluation battle (default 10) # SAC_EVAL_OPPONENTS comma-separated eval set (default Corners,Crazy,Target) # — each cycle evaluates EVERY one; results all land in # eval_log.jsonl (lines carry "opponent":"Name") # SAC_MAX_CRASHES consecutive crashes before abort (default 5) # SAC_LOG_FILE / SAC_EVAL_LOG_FILE (JSON-lines logs) # SACLSTM_* passed through to the bot (UTD_RATIO, BATCH_SIZE, ...) set -uo pipefail SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" RUNNER_DIR="$REPO_ROOT/tools/training_runner" JAR="${TANK_ROYALE_JAR:-/home/davide/Projects/tank-royale/runner/examples/lib/robocode-tankroyale-runner.jar}" export PPO_BOT_DIR="$SCRIPT_DIR" # runner launches THIS bot dir export BOT_NAME="${BOT_NAME:-SAC_LSTM_Bot}" # RunTraining result matching export SAMPLE_BOTS_DIR="${SAMPLE_BOTS_DIR:-/home/davide/Projects/tank-royale/sample-bots/java/build/archive}" # Liveness contract (#49): RunTraining.java watches $BOT_DIR/weights/round_counter.txt # and integration.bumpRoundCounter() writes it next to the weights — so the bot's # weights path is pinned here, NOT env-overridable. export SACLSTM_WEIGHTS_PATH="$SCRIPT_DIR/weights/sac_latest.zip" WEIGHTS_DIR="$(dirname "$SACLSTM_WEIGHTS_PATH")" OPPONENTS="${SAC_OPPONENTS:-Corners:3,Crazy:2,RamFire:1,Target:1}" TOTAL_ROUNDS="${SAC_TOTAL_ROUNDS:-100}" CHUNK_SIZE="${SAC_CHUNK_SIZE:-10}" EVAL_INTERVAL="${SAC_EVAL_INTERVAL:-2}" EVAL_ROUNDS="${SAC_EVAL_ROUNDS:-10}" EVAL_OPPONENTS="${SAC_EVAL_OPPONENTS:-Corners,Crazy,Target}" MA_WINDOW=5 # lever 1 (#59): per-opponent moving average over last N evals MAX_CRASHES="${SAC_MAX_CRASHES:-5}" LOG_FILE="${SAC_LOG_FILE:-$SCRIPT_DIR/training_log.jsonl}" EVAL_LOG_FILE="${SAC_EVAL_LOG_FILE:-$SCRIPT_DIR/eval_log.jsonl}" CLASSES_DIR="/tmp/opencode/sac_train_classes" echo "=== SAC_LSTM_Bot training harness ===" echo "Opponents: $OPPONENTS | budget: $TOTAL_ROUNDS rounds in chunks of $CHUNK_SIZE" echo "Eval: every $EVAL_INTERVAL chunks, $EVAL_ROUNDS rounds vs [$EVAL_OPPONENTS], MA-$MA_WINDOW composite best-gating" echo "Weights: $SACLSTM_WEIGHTS_PATH" # ── compile bot + java runner ───────────────────────────────────────────────── (cd "$SCRIPT_DIR" && nimble build -d:release) || { echo ">>> bot build failed"; exit 1; } mkdir -p "$WEIGHTS_DIR" "$CLASSES_DIR" # Startup sweep: atomic saves leave sac_latest.zip.tmp.*.part corpses behind if # a run is SIGKILLed (libzip modify-path, see notebook forensics) — clear them. rm -f "$WEIGHTS_DIR"/sac_latest.zip.tmp.*.part "$WEIGHTS_DIR"/sac_latest.zip.tmp javac -cp "$JAR" -d "$CLASSES_DIR" "$RUNNER_DIR/RunTraining.java" || { echo ">>> javac failed"; exit 1; } # ── weighted opponent pick over "Name:w,Name:w" ─────────────────────────────── pick_opponent() { local total=0 p name w r local pairs IFS=',' read -ra pairs <<< "$OPPONENTS" for p in "${pairs[@]}"; do total=$(( total + ${p##*:} )); done r=$(( RANDOM % total )) for p in "${pairs[@]}"; do name="${p%%:*}"; w="${p##*:}" if (( r < w )); then echo "$name"; return; fi r=$(( r - w )) done echo "${pairs[0]%%:*}" } run_battle() { # $1=opponent $2=rounds $3=log file PPOB_LOG_FILE="$3" java -cp "$CLASSES_DIR:$JAR" RunTraining "$1" "$2" } # ── Lever 1 (#59): eval rotation + MA best-gating ───────────────────────────── # best_score.txt FORMAT CHANGE: it used to store the single-opponent integer # win rate (%); that semantics is retired. It now stores the COMPOSITE score — # the mean over SAC_EVAL_OPPONENTS of each opponent's moving average (last # MA_WINDOW eval win rates, %). sac_best.zip is rewritten only when the # composite strictly improves. ma_hist_file() { echo "$WEIGHTS_DIR/ma_history_$1.txt"; } composite_of() { # reads one "w w w ..." history line per opponent on stdin awk -v W="$MA_WINDOW" ' NF > 0 { n=NF; k=(n>W)?W:n; s=0; for(j=n-k+1;j<=n;j++) s+=$j; tot+=s/k; c++ } END { if (c>0) printf "%.4f", tot/c; else print "-1" }' } eval_checkpoint() { # ponytail: opponent names are split by whitespace — fine for Tank Royale bot # names (no spaces); switch to a mapfile IFS=',\n' read if that ever changes. local opps=(${EVAL_OPPONENTS//,/ }) local tmp="$EVAL_LOG_FILE.tmp" otmp opp wins rounds wr composite best : > "$tmp" for opp in "${opps[@]}"; do otmp="$EVAL_LOG_FILE.$opp.tmp" : > "$otmp" echo ">>> [eval] $EVAL_ROUNDS deterministic rounds vs $opp" if ! SACLSTM_EVAL_MODE=1 run_battle "$opp" "$EVAL_ROUNDS" "$otmp"; then rm -f "$otmp" "$tmp" echo ">>> [eval] crashed vs $opp — keeping previous best" return 0 fi wins=$(grep -c '"win":true' "$otmp" || true) rounds=$(grep -c '"type":"game"' "$otmp" || true) if (( rounds == 0 )); then rm -f "$otmp" "$tmp" echo ">>> [eval] no results vs $opp — keeping previous best" return 0 fi wr=$(( 100 * wins / rounds )) echo ">>> [eval] win rate: $wins/$rounds ($wr%) vs $opp" cat "$otmp" >> "$tmp"; rm -f "$otmp" # Per-opponent history: append this cycle's win rate, keep last MA_WINDOW. printf '%s\n' "$(cat "$(ma_hist_file "$opp")" 2>/dev/null)" "$wr" \ | tail -n "$MA_WINDOW" | tr '\n' ' ' > "$(ma_hist_file "$opp")" done mv "$tmp" "$EVAL_LOG_FILE" composite=$(for opp in "${opps[@]}"; do cat "$(ma_hist_file "$opp")"; echo; done | composite_of) # ponytail: best-score state is a plain file next to the checkpoint; survives # harness restarts, no lock needed (single harness instance assumed). best=$(cat "$WEIGHTS_DIR/best_score.txt" 2>/dev/null) [ -z "$best" ] && best=-1 if awk -v a="$composite" -v b="$best" 'BEGIN{exit !(a+0 > b+0)}' \ && [ -f "$SACLSTM_WEIGHTS_PATH" ]; then echo "$composite" > "$WEIGHTS_DIR/best_score.txt" cp "$SACLSTM_WEIGHTS_PATH" "$WEIGHTS_DIR/sac_best.zip" echo ">>> [eval] new best composite ($composite) -> sac_best.zip" fi } NUM_CHUNKS=$(( (TOTAL_ROUNDS + CHUNK_SIZE - 1) / CHUNK_SIZE )) fails=0 chunk=1 # while, not `for chunk in $(seq ...)`: a crash on the FINAL chunk must rerun # it (#54 — seq list is exhausted by then, so ((chunk--));continue fell through # and the harness exited 0 with the budget incomplete). while (( chunk <= NUM_CHUNKS )); do ROUNDS=$CHUNK_SIZE (( TOTAL_ROUNDS - (chunk - 1) * CHUNK_SIZE < CHUNK_SIZE )) && \ ROUNDS=$(( TOTAL_ROUNDS - (chunk - 1) * CHUNK_SIZE )) OPP=$(pick_opponent) echo "=== Chunk $chunk/$NUM_CHUNKS: $ROUNDS rounds vs $OPP ===" if ! run_battle "$OPP" "$ROUNDS" "$LOG_FILE"; then fails=$(( fails + 1 )) if (( fails >= MAX_CRASHES )); then echo ">>> aborted: $fails consecutive crashes (bot process dying?)" exit 1 fi # Crash recovery: RunTraining's liveness detection exited; the bot reloads # its latest checkpoint on restart, so just rerun this chunk. echo ">>> crash #$fails — restarting chunk from latest checkpoint" (( chunk-- )); continue fi fails=0 (( chunk % EVAL_INTERVAL == 0 )) && eval_checkpoint ((chunk += 1)) done echo ">>> training complete: $NUM_CHUNKS chunks. Logs:" echo " training: $LOG_FILE" echo " eval: $EVAL_LOG_FILE" [ -f "$WEIGHTS_DIR/sac_best.zip" ] && \ echo " best: $WEIGHTS_DIR/sac_best.zip (composite $(cat "$WEIGHTS_DIR/best_score.txt"))" exit 0