139 lines
6.4 KiB
Bash
Executable File
139 lines
6.4 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# sac_train.sh — #49 training orchestration for SAC_LSTM_Bot.
|
|
#
|
|
# Drives chunked self-play via tools/training_runner/RunTraining.java (which
|
|
# owns server lifecycle, opponent connection and dead-bot liveness detection
|
|
# through weights/round_counter.txt), samples opponents by weight per chunk,
|
|
# runs deterministic evaluation (SACLSTM_EVAL_MODE=1) every N chunks, and keeps
|
|
# the best checkpoint (weights/sac_best.zip) by eval win rate.
|
|
#
|
|
# Config (env vars):
|
|
# SAC_OPPONENTS "Name:weight,Name:weight,..." (default below)
|
|
# SAC_TOTAL_ROUNDS total training-round budget (default 100)
|
|
# SAC_CHUNK_SIZE rounds per RunTraining battle (default 10)
|
|
# SAC_EVAL_INTERVAL eval every N chunks (default 2)
|
|
# SAC_EVAL_ROUNDS rounds per evaluation battle (default 10)
|
|
# SAC_EVAL_OPPONENT fixed eval opponent (default first opponent)
|
|
# SAC_MAX_CRASHES consecutive crashes before abort (default 5)
|
|
# SAC_LOG_FILE / SAC_EVAL_LOG_FILE (JSON-lines logs)
|
|
# SACLSTM_* passed through to the bot (UTD_RATIO, BATCH_SIZE, ...)
|
|
set -uo pipefail
|
|
|
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
|
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
|
RUNNER_DIR="$REPO_ROOT/tools/training_runner"
|
|
JAR="${TANK_ROYALE_JAR:-/home/davide/Projects/tank-royale/runner/examples/lib/robocode-tankroyale-runner.jar}"
|
|
|
|
export PPO_BOT_DIR="$SCRIPT_DIR" # runner launches THIS bot dir
|
|
export BOT_NAME="${BOT_NAME:-SAC_LSTM_Bot}" # RunTraining result matching
|
|
export SAMPLE_BOTS_DIR="${SAMPLE_BOTS_DIR:-/home/davide/Projects/tank-royale/sample-bots/java/build/archive}"
|
|
# Liveness contract (#49): RunTraining.java watches $BOT_DIR/weights/round_counter.txt
|
|
# and integration.bumpRoundCounter() writes it next to the weights — so the bot's
|
|
# weights path is pinned here, NOT env-overridable.
|
|
export SACLSTM_WEIGHTS_PATH="$SCRIPT_DIR/weights/sac_latest.zip"
|
|
WEIGHTS_DIR="$(dirname "$SACLSTM_WEIGHTS_PATH")"
|
|
|
|
OPPONENTS="${SAC_OPPONENTS:-Corners:3,Crazy:2,RamFire:1,Target:1}"
|
|
TOTAL_ROUNDS="${SAC_TOTAL_ROUNDS:-100}"
|
|
CHUNK_SIZE="${SAC_CHUNK_SIZE:-10}"
|
|
EVAL_INTERVAL="${SAC_EVAL_INTERVAL:-2}"
|
|
EVAL_ROUNDS="${SAC_EVAL_ROUNDS:-10}"
|
|
EVAL_OPPONENT="${SAC_EVAL_OPPONENT:-${OPPONENTS%%:*}}"
|
|
MAX_CRASHES="${SAC_MAX_CRASHES:-5}"
|
|
LOG_FILE="${SAC_LOG_FILE:-$SCRIPT_DIR/training_log.jsonl}"
|
|
EVAL_LOG_FILE="${SAC_EVAL_LOG_FILE:-$SCRIPT_DIR/eval_log.jsonl}"
|
|
CLASSES_DIR="/tmp/opencode/sac_train_classes"
|
|
|
|
echo "=== SAC_LSTM_Bot training harness ==="
|
|
echo "Opponents: $OPPONENTS | budget: $TOTAL_ROUNDS rounds in chunks of $CHUNK_SIZE"
|
|
echo "Eval: every $EVAL_INTERVAL chunks, $EVAL_ROUNDS rounds vs $EVAL_OPPONENT"
|
|
echo "Weights: $SACLSTM_WEIGHTS_PATH"
|
|
|
|
# ── compile bot + java runner ─────────────────────────────────────────────────
|
|
(cd "$SCRIPT_DIR" && nimble build -d:release) || { echo ">>> bot build failed"; exit 1; }
|
|
mkdir -p "$WEIGHTS_DIR" "$CLASSES_DIR"
|
|
# Startup sweep: atomic saves leave sac_latest.zip.tmp.*.part corpses behind if
|
|
# a run is SIGKILLed (libzip modify-path, see notebook forensics) — clear them.
|
|
rm -f "$WEIGHTS_DIR"/sac_latest.zip.tmp.*.part "$WEIGHTS_DIR"/sac_latest.zip.tmp
|
|
javac -cp "$JAR" -d "$CLASSES_DIR" "$RUNNER_DIR/RunTraining.java" || { echo ">>> javac failed"; exit 1; }
|
|
|
|
# ── weighted opponent pick over "Name:w,Name:w" ───────────────────────────────
|
|
pick_opponent() {
|
|
local total=0 p name w r
|
|
local pairs
|
|
IFS=',' read -ra pairs <<< "$OPPONENTS"
|
|
for p in "${pairs[@]}"; do total=$(( total + ${p##*:} )); done
|
|
r=$(( RANDOM % total ))
|
|
for p in "${pairs[@]}"; do
|
|
name="${p%%:*}"; w="${p##*:}"
|
|
if (( r < w )); then echo "$name"; return; fi
|
|
r=$(( r - w ))
|
|
done
|
|
echo "${pairs[0]%%:*}"
|
|
}
|
|
|
|
run_battle() { # $1=opponent $2=rounds $3=log file
|
|
PPOB_LOG_FILE="$3" java -cp "$CLASSES_DIR:$JAR" RunTraining "$1" "$2"
|
|
}
|
|
|
|
eval_checkpoint() {
|
|
local tmp="$EVAL_LOG_FILE.tmp" wins rounds wr best
|
|
: > "$tmp"
|
|
echo ">>> [eval] $EVAL_ROUNDS deterministic rounds vs $EVAL_OPPONENT"
|
|
if ! SACLSTM_EVAL_MODE=1 run_battle "$EVAL_OPPONENT" "$EVAL_ROUNDS" "$tmp"; then
|
|
rm -f "$tmp"
|
|
echo ">>> [eval] crashed — keeping previous best"
|
|
return 0
|
|
fi
|
|
mv "$tmp" "$EVAL_LOG_FILE"
|
|
wins=$(grep -c '"win":true' "$EVAL_LOG_FILE" || true)
|
|
rounds=$(grep -c '"type":"game"' "$EVAL_LOG_FILE" || true)
|
|
(( rounds == 0 )) && { echo ">>> [eval] no results"; return 0; }
|
|
wr=$(( 100 * wins / rounds ))
|
|
echo ">>> [eval] win rate: $wins/$rounds ($wr%) vs $EVAL_OPPONENT"
|
|
# ponytail: best-score state is a plain file next to the checkpoint; survives
|
|
# harness restarts, no lock needed (single harness instance assumed).
|
|
best=-1
|
|
[ -f "$WEIGHTS_DIR/best_score.txt" ] && best=$(cat "$WEIGHTS_DIR/best_score.txt")
|
|
if (( wr > best )) && [ -f "$SACLSTM_WEIGHTS_PATH" ]; then
|
|
echo "$wr" > "$WEIGHTS_DIR/best_score.txt"
|
|
cp "$SACLSTM_WEIGHTS_PATH" "$WEIGHTS_DIR/sac_best.zip"
|
|
echo ">>> [eval] new best ($wr%) -> sac_best.zip"
|
|
fi
|
|
}
|
|
|
|
NUM_CHUNKS=$(( (TOTAL_ROUNDS + CHUNK_SIZE - 1) / CHUNK_SIZE ))
|
|
fails=0
|
|
chunk=1
|
|
# while, not `for chunk in $(seq ...)`: a crash on the FINAL chunk must rerun
|
|
# it (#54 — seq list is exhausted by then, so ((chunk--));continue fell through
|
|
# and the harness exited 0 with the budget incomplete).
|
|
while (( chunk <= NUM_CHUNKS )); do
|
|
ROUNDS=$CHUNK_SIZE
|
|
(( TOTAL_ROUNDS - (chunk - 1) * CHUNK_SIZE < CHUNK_SIZE )) && \
|
|
ROUNDS=$(( TOTAL_ROUNDS - (chunk - 1) * CHUNK_SIZE ))
|
|
OPP=$(pick_opponent)
|
|
echo "=== Chunk $chunk/$NUM_CHUNKS: $ROUNDS rounds vs $OPP ==="
|
|
if ! run_battle "$OPP" "$ROUNDS" "$LOG_FILE"; then
|
|
fails=$(( fails + 1 ))
|
|
if (( fails >= MAX_CRASHES )); then
|
|
echo ">>> aborted: $fails consecutive crashes (bot process dying?)"
|
|
exit 1
|
|
fi
|
|
# Crash recovery: RunTraining's liveness detection exited; the bot reloads
|
|
# its latest checkpoint on restart, so just rerun this chunk.
|
|
echo ">>> crash #$fails — restarting chunk from latest checkpoint"
|
|
(( chunk-- )); continue
|
|
fi
|
|
fails=0
|
|
(( chunk % EVAL_INTERVAL == 0 )) && eval_checkpoint
|
|
((chunk += 1))
|
|
done
|
|
|
|
echo ">>> training complete: $NUM_CHUNKS chunks. Logs:"
|
|
echo " training: $LOG_FILE"
|
|
echo " eval: $EVAL_LOG_FILE"
|
|
[ -f "$WEIGHTS_DIR/sac_best.zip" ] && \
|
|
echo " best: $WEIGHTS_DIR/sac_best.zip ($(cat "$WEIGHTS_DIR/best_score.txt")%)"
|
|
exit 0
|