4cd5618435
TASK 2 - THE FLAKY ACCEPTANCE TEST, root-caused. It was NOT a live/offline boundary race as suspected. The replay spawned gun 13 (TMSelect) while live has EnableTmSelector = false and never does. The shared VirtualTracker ring is ORDER-SENSITIVE, so gun 13's extra 4 bullets/tick shift the ring head and permute the per-tick RESOLUTION ORDER of every other gun. The learning guns append observations in resolution order, so their predictions shifted and produced small hit deltas that moved between runs. Evidence: the first KNN divergence was at rtick=174 with the SAME resolution set merely reordered (live ft133,138,139,142,148,150,151 vs offline ft150,151,133,138,139,142,148); after closing gun 13's ready gate offline the live and offline KNN traces became BYTE-IDENTICAL (diff empty, 904/904 lines). Fix: mirror the live rack in the replay. No tick exclusion, no tolerance loosening. Stability: 5/5 consecutive runs now report 12/12 exact, each with enemyDied=true - the death boundary is included, not excluded. The proof is now real rather than a lucky run. TASK 1 - the tie-break was not random. randomize() was only reached incidentally through initTsetlinGun(), so a rack without Tsetlin had a fixed rand() stream and ties always resolved the same way across process restarts. Added seedSelectorRng() after gun construction, honouring GUN_SELECTOR_SEED. Evidence: unseeded, 6 separate processes gave different pick sequences; with GUN_SELECTOR_SEED=42, 3 processes gave identical sequences. TASK 3 - PRUNING DOES NOT HELP; keep the full rack. 15 PAIRED runs per variant vs DrussGT, 8 rounds, identical seeds: baseline 3238 shots 6.18% (events 6.16%) 200 dmg/run Tsetlin disabled 3522 shots 5.76% (events 5.71%) 197 dmg/run Tsetlin+Displace 3478 shots 5.46% (events 5.37%) 183 dmg/run Paired permutation tests: -0.34pp p=0.57 and -0.70pp p=0.21. Per-run distributions completely overlap (baseline range [2.68, 10.00]; 15/15 and 14/15 runs inside it). A Crazy control showed no separation either. So removing the measured-worst real performers is neutral-to-slightly-negative, and with sd ~1.8pp a definitive claim either way would need far more runs. CORRECTION TO A CLAIM I MADE: the 'virtual metric is INVERTED' finding does NOT reproduce. Job-24 measured Spearman -0.374; this job measures +0.335 over the same 13 guns with a different but equally defensible aggregation. Two opposite signs means the correlation is NOT robustly negative - it is WEAK AND SIGN-UNSTABLE. The honest statement is that virtual hit rate is a poor ranker, not an inverted one. The docs assert the inversion and need correcting. Also adds per-process GUN_STATS_PATH/GUN_SHOTLOG_PATH so concurrent A/B runs do not clobber each other, and an env-gated GUN_RACK_DISABLE for rack A/Bs. All default behaviour is unchanged when the env vars are unset.
127 lines
4.9 KiB
Nim
127 lines
4.9 KiB
Nim
## Task 3: acceptance test — prove the offline range reproduces the live
|
|
## virtual-bullet metric, or the range is worthless.
|
|
##
|
|
## Steps:
|
|
## 1. run ONE live ModularBot vs OscillatorBot round with the ModularBot
|
|
## recorder enabled for this battle only (the test exports
|
|
## TR_RECORD_WORLDSTATE=1, which the bot reads at RUNTIME; ordinary
|
|
## builds leave it unset and write no fixture),
|
|
## 2. read the online per-gun virtual fitness from /tmp/gun_stats.jsonl,
|
|
## 3. replay the recorded WorldState fixture offline through the same guns,
|
|
## 4. compare.
|
|
##
|
|
## Tsetlin is stochastic (tmLearnOne calls rand(); its constructor calls
|
|
## randomize()), so byte-identical replay is impossible for it. The 12
|
|
## deterministic guns must match EXACTLY; Tsetlin is reported separately.
|
|
##
|
|
## Run with:
|
|
## nim c -r common_libs/tests/acceptance_offline_vs_online.nim
|
|
##
|
|
## Requires TR_SERVER_JAR / TR_BATTLE_RUNNER (or the default dev paths below).
|
|
|
|
import std/[json, os, strformat, strutils, math]
|
|
import test_framework/test_framework
|
|
import gun_harness/offline_range
|
|
import range_guns
|
|
|
|
const
|
|
repoRoot = currentSourcePath().parentDir.parentDir.parentDir
|
|
modularBotDir = repoRoot / "ModularBot_garage"
|
|
adversaryDir = repoRoot / "common_libs" / "test_framework" / "adversaries" / "OscillatorBot"
|
|
statsPath = "/tmp/gun_stats.jsonl"
|
|
recordPath = "/tmp/worldstate_record.jsonl"
|
|
serverJar = "/home/davide/Projects/tank-royale/server/build/libs/robocode-tankroyale-server-0.35.5-all.jar"
|
|
runnerJar = "/home/davide/Projects/tank-royale/runner/examples/lib/robocode-tankroyale-runner.jar"
|
|
|
|
const TsetlinId = 2
|
|
const TmSelectorId = 13 ## also stochastic (rand() in Gate choose + TM feedback)
|
|
|
|
proc isStochastic(id: int): bool = id == TsetlinId or id == TmSelectorId
|
|
|
|
proc lastOnlineRound(path: string): JsonNode =
|
|
result = nil
|
|
for line in lines(path):
|
|
let s = line.strip()
|
|
if s.len == 0: continue
|
|
let node = parseJson(s)
|
|
if node.hasKey("guns"): result = node
|
|
|
|
proc main() =
|
|
if not fileExists(serverJar) or not fileExists(runnerJar):
|
|
echo "Skipping: TR JARs not found (server=", serverJar, ", runner=", runnerJar, ")"
|
|
quit(0)
|
|
if not fileExists(modularBotDir / "src" / "ModularBot.nim"):
|
|
echo "Skipping: ModularBot source not found"
|
|
quit(0)
|
|
|
|
for p in [statsPath, recordPath]:
|
|
if fileExists(p): removeFile(p)
|
|
|
|
# Enable the ModularBot's runtime world-state recorder for THIS battle only.
|
|
# The env var is inherited by the battle-runner process and then by the bot
|
|
# processes it spawns, so a single invocation of this test is self-contained.
|
|
putEnv("TR_RECORD_WORLDSTATE", "1")
|
|
defer: delEnv("TR_RECORD_WORLDSTATE")
|
|
|
|
echo "=== live battle: ModularBot vs OscillatorBot, 1 round, max speed ==="
|
|
let battle = runBattle(@[modularBotDir, adversaryDir], rounds = 1,
|
|
timeout = 240000, maxSpeed = true)
|
|
for res in battle.results:
|
|
echo fmt" {res.name:<14} rank={res.rank} score={res.totalScore}"
|
|
|
|
if not fileExists(recordPath):
|
|
echo "FAIL: recorder produced no fixture (TR_RECORD_WORLDSTATE not inherited?)"
|
|
quit(1)
|
|
if not fileExists(statsPath):
|
|
echo "FAIL: no /tmp/gun_stats.jsonl"
|
|
quit(1)
|
|
|
|
let online = lastOnlineRound(statsPath)
|
|
let fx = loadFixture(recordPath)
|
|
let reports = replayFixture(fx, buildAllGunDrivers(enableTmSelector = false), liveActual = true)
|
|
|
|
# Map online stats by gun id.
|
|
var onShots: array[14, int]
|
|
var onHits: array[14, int]
|
|
var onNames: array[14, string]
|
|
for g in online["guns"]:
|
|
let id = g["id"].getInt()
|
|
if id >= 0 and id < 14:
|
|
onShots[id] = g["vShots"].getInt()
|
|
onHits[id] = g["vHits"].getInt()
|
|
onNames[id] = g["name"].getStr()
|
|
|
|
echo ""
|
|
echo fmt"fixture: {recordPath} ticks={fx.states.len} enemyId={fx.enemyId} enemyDied={fx.enemyDied}"
|
|
echo "online stats: /tmp/gun_stats.jsonl round ", online["round"].getInt()
|
|
echo ""
|
|
echo "gun online(vHits/vShots) offline(hits/shots) verdict"
|
|
echo "-----------------------------------------------------------------------"
|
|
var matches = 0
|
|
var deterministic = 0
|
|
for id in 0..<14:
|
|
let r = reports[id]
|
|
let match = r.hits == onHits[id] and r.shots == onShots[id]
|
|
var verdict: string
|
|
if isStochastic(id):
|
|
verdict = if match: "MATCH (stochastic)" else: "differs (stochastic, expected)"
|
|
else:
|
|
inc deterministic
|
|
if match:
|
|
inc matches
|
|
verdict = "OK"
|
|
else:
|
|
verdict = "MISMATCH"
|
|
echo fmt"{r.name:<12} {onHits[id]:>5}/{onShots[id]:<5} {r.hits:>5}/{r.shots:<5} {verdict}"
|
|
|
|
echo ""
|
|
echo fmt"deterministic guns matching exactly: {matches}/{deterministic}"
|
|
if matches != deterministic:
|
|
echo "VERDICT: FAIL — offline range does NOT reproduce the live metric."
|
|
quit(1)
|
|
echo "VERDICT: PASS — offline == online for all 12 deterministic guns."
|
|
echo "(Tsetlin and TMSelect are stochastic and are allowed to differ.)"
|
|
|
|
when isMainModule:
|
|
main()
|