Files
SirRoboGarage/SAC_LSTM_Bot/sac_train.sh
T
SirStone 6a294ad7ad feat(SAC_LSTM_Bot): mirror-twin sparring partner + readiness check (#54)
- make_twin.sh: generates self-contained SacTwin dir in the sample-bots
  archive (own json/sh identity, own weights dir seeded from a frozen
  sac_best.zip copy, own round_counter) so RunTraining.java resolves it
  like any sample bot; re-running resets the twin to the frozen baseline.
- src/SAC_LSTM_Bot.nim: SACLSTM_BOT_JSON env overrides the baked-in bot
  json (loadBotInfo gives json total precedence, #49) so the same binary
  boots under the twin's name.
- sac_train.sh: chunk loop is a while, not for-over-seq — a crash on the
  FINAL chunk previously fell through ((chunk--);continue on an exhausted
  seq list) and exited 0 with budget incomplete; observed live vs SacTwin.

Readiness dry-run (#54): weighted pool Corners:1,SacTwin:3 picked the twin
in 3/4 chunks; all battles counter-checked; deterministic eval parsed;
main sac_best.zip/counter untouched by twin (twin counter advanced
independently); crash-restart proven end-to-end incl. final-chunk retry.
2026-08-21 23:19:27 +02:00

136 lines
6.1 KiB
Bash
Executable File

#!/usr/bin/env bash
# sac_train.sh — #49 training orchestration for SAC_LSTM_Bot.
#
# Drives chunked self-play via tools/training_runner/RunTraining.java (which
# owns server lifecycle, opponent connection and dead-bot liveness detection
# through weights/round_counter.txt), samples opponents by weight per chunk,
# runs deterministic evaluation (SACLSTM_EVAL_MODE=1) every N chunks, and keeps
# the best checkpoint (weights/sac_best.zip) by eval win rate.
#
# Config (env vars):
# SAC_OPPONENTS "Name:weight,Name:weight,..." (default below)
# SAC_TOTAL_ROUNDS total training-round budget (default 100)
# SAC_CHUNK_SIZE rounds per RunTraining battle (default 10)
# SAC_EVAL_INTERVAL eval every N chunks (default 2)
# SAC_EVAL_ROUNDS rounds per evaluation battle (default 10)
# SAC_EVAL_OPPONENT fixed eval opponent (default first opponent)
# SAC_MAX_CRASHES consecutive crashes before abort (default 5)
# SAC_LOG_FILE / SAC_EVAL_LOG_FILE (JSON-lines logs)
# SACLSTM_* passed through to the bot (UTD_RATIO, BATCH_SIZE, ...)
set -uo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
RUNNER_DIR="$REPO_ROOT/tools/training_runner"
JAR="${TANK_ROYALE_JAR:-/home/davide/Projects/tank-royale/runner/examples/lib/robocode-tankroyale-runner.jar}"
export PPO_BOT_DIR="$SCRIPT_DIR" # runner launches THIS bot dir
export BOT_NAME="${BOT_NAME:-SAC_LSTM_Bot}" # RunTraining result matching
export SAMPLE_BOTS_DIR="${SAMPLE_BOTS_DIR:-/home/davide/Projects/tank-royale/sample-bots/java/build/archive}"
# Liveness contract (#49): RunTraining.java watches $BOT_DIR/weights/round_counter.txt
# and integration.bumpRoundCounter() writes it next to the weights — so the bot's
# weights path is pinned here, NOT env-overridable.
export SACLSTM_WEIGHTS_PATH="$SCRIPT_DIR/weights/sac_latest.zip"
WEIGHTS_DIR="$(dirname "$SACLSTM_WEIGHTS_PATH")"
OPPONENTS="${SAC_OPPONENTS:-Corners:3,Crazy:2,RamFire:1,Target:1}"
TOTAL_ROUNDS="${SAC_TOTAL_ROUNDS:-100}"
CHUNK_SIZE="${SAC_CHUNK_SIZE:-10}"
EVAL_INTERVAL="${SAC_EVAL_INTERVAL:-2}"
EVAL_ROUNDS="${SAC_EVAL_ROUNDS:-10}"
EVAL_OPPONENT="${SAC_EVAL_OPPONENT:-${OPPONENTS%%:*}}"
MAX_CRASHES="${SAC_MAX_CRASHES:-5}"
LOG_FILE="${SAC_LOG_FILE:-$SCRIPT_DIR/training_log.jsonl}"
EVAL_LOG_FILE="${SAC_EVAL_LOG_FILE:-$SCRIPT_DIR/eval_log.jsonl}"
CLASSES_DIR="/tmp/opencode/sac_train_classes"
echo "=== SAC_LSTM_Bot training harness ==="
echo "Opponents: $OPPONENTS | budget: $TOTAL_ROUNDS rounds in chunks of $CHUNK_SIZE"
echo "Eval: every $EVAL_INTERVAL chunks, $EVAL_ROUNDS rounds vs $EVAL_OPPONENT"
echo "Weights: $SACLSTM_WEIGHTS_PATH"
# ── compile bot + java runner ─────────────────────────────────────────────────
(cd "$SCRIPT_DIR" && nimble build -d:release) || { echo ">>> bot build failed"; exit 1; }
mkdir -p "$WEIGHTS_DIR" "$CLASSES_DIR"
javac -cp "$JAR" -d "$CLASSES_DIR" "$RUNNER_DIR/RunTraining.java" || { echo ">>> javac failed"; exit 1; }
# ── weighted opponent pick over "Name:w,Name:w" ───────────────────────────────
pick_opponent() {
local total=0 p name w r
local pairs
IFS=',' read -ra pairs <<< "$OPPONENTS"
for p in "${pairs[@]}"; do total=$(( total + ${p##*:} )); done
r=$(( RANDOM % total ))
for p in "${pairs[@]}"; do
name="${p%%:*}"; w="${p##*:}"
if (( r < w )); then echo "$name"; return; fi
r=$(( r - w ))
done
echo "${pairs[0]%%:*}"
}
run_battle() { # $1=opponent $2=rounds $3=log file
PPOB_LOG_FILE="$3" java -cp "$CLASSES_DIR:$JAR" RunTraining "$1" "$2"
}
eval_checkpoint() {
local tmp="$EVAL_LOG_FILE.tmp" wins rounds wr best
: > "$tmp"
echo ">>> [eval] $EVAL_ROUNDS deterministic rounds vs $EVAL_OPPONENT"
if ! SACLSTM_EVAL_MODE=1 run_battle "$EVAL_OPPONENT" "$EVAL_ROUNDS" "$tmp"; then
rm -f "$tmp"
echo ">>> [eval] crashed — keeping previous best"
return 0
fi
mv "$tmp" "$EVAL_LOG_FILE"
wins=$(grep -c '"win":true' "$EVAL_LOG_FILE" || true)
rounds=$(grep -c '"type":"game"' "$EVAL_LOG_FILE" || true)
(( rounds == 0 )) && { echo ">>> [eval] no results"; return 0; }
wr=$(( 100 * wins / rounds ))
echo ">>> [eval] win rate: $wins/$rounds ($wr%) vs $EVAL_OPPONENT"
# ponytail: best-score state is a plain file next to the checkpoint; survives
# harness restarts, no lock needed (single harness instance assumed).
best=-1
[ -f "$WEIGHTS_DIR/best_score.txt" ] && best=$(cat "$WEIGHTS_DIR/best_score.txt")
if (( wr > best )) && [ -f "$SACLSTM_WEIGHTS_PATH" ]; then
echo "$wr" > "$WEIGHTS_DIR/best_score.txt"
cp "$SACLSTM_WEIGHTS_PATH" "$WEIGHTS_DIR/sac_best.zip"
echo ">>> [eval] new best ($wr%) -> sac_best.zip"
fi
}
NUM_CHUNKS=$(( (TOTAL_ROUNDS + CHUNK_SIZE - 1) / CHUNK_SIZE ))
fails=0
chunk=1
# while, not `for chunk in $(seq ...)`: a crash on the FINAL chunk must rerun
# it (#54 — seq list is exhausted by then, so ((chunk--));continue fell through
# and the harness exited 0 with the budget incomplete).
while (( chunk <= NUM_CHUNKS )); do
ROUNDS=$CHUNK_SIZE
(( TOTAL_ROUNDS - (chunk - 1) * CHUNK_SIZE < CHUNK_SIZE )) && \
ROUNDS=$(( TOTAL_ROUNDS - (chunk - 1) * CHUNK_SIZE ))
OPP=$(pick_opponent)
echo "=== Chunk $chunk/$NUM_CHUNKS: $ROUNDS rounds vs $OPP ==="
if ! run_battle "$OPP" "$ROUNDS" "$LOG_FILE"; then
fails=$(( fails + 1 ))
if (( fails >= MAX_CRASHES )); then
echo ">>> aborted: $fails consecutive crashes (bot process dying?)"
exit 1
fi
# Crash recovery: RunTraining's liveness detection exited; the bot reloads
# its latest checkpoint on restart, so just rerun this chunk.
echo ">>> crash #$fails — restarting chunk from latest checkpoint"
(( chunk-- )); continue
fi
fails=0
(( chunk % EVAL_INTERVAL == 0 )) && eval_checkpoint
((chunk += 1))
done
echo ">>> training complete: $NUM_CHUNKS chunks. Logs:"
echo " training: $LOG_FILE"
echo " eval: $EVAL_LOG_FILE"
[ -f "$WEIGHTS_DIR/sac_best.zip" ] && \
echo " best: $WEIGHTS_DIR/sac_best.zip ($(cat "$WEIGHTS_DIR/best_score.txt")%)"
exit 0