diff --git a/common_libs/tests/fixtures/gauntlet_bitbrain_vs_pattern_report.txt b/common_libs/tests/fixtures/gauntlet_bitbrain_vs_pattern_report.txt new file mode 100644 index 0000000..1e5c788 --- /dev/null +++ b/common_libs/tests/fixtures/gauntlet_bitbrain_vs_pattern_report.txt @@ -0,0 +1,129 @@ +# gauntlet analysis — pattern (reference) vs bb (treatment) + +session: /tmp/ab/j117_gauntlet +commit: da4a971ca99e81894e86a35a521bbef9ba081d3f binary sha256 f5876e7e8c14389de8f335b1529868da80d987cea5ca298abc058ecaff1925cf +design: 3 runs x 5 rounds per opponent per arm, conc=4 +arms: pattern = (none) + bb = TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.0,1.25,1.5,2.0 TR_BITBRAIN_MEM=decay + +opponents: 32 retained of 33 attempted; 1 dropped + dropped Aurora: inert (only 1 fires in both arms — never meaningfully fought) + +## Per-opponent table (MEASURED; damage is damage DEALT by ModularBot) + +| opponent | style | pattern dmg/run | bb dmg/run | Δ dmg | pattern wins/run | bb wins/run | Δ wins | rounds/arm | opp fired | runs | +|---|---|---:|---:|---:|---:|---:|---:|---:|---:|---:| +| Aristocles | dodger | 409.3 | 380.0 | -29.3 | 5.00 | 5.00 | +0.00 | 15/15 | 416 | 3/3 | +| CassiusClay | dodger | 133.1 | 127.0 | -6.1 | 0.67 | 1.00 | +0.33 | 15/15 | 1259 | 3/3 | +| Cigaret | dodger | 124.7 | 192.8 | +68.1 | 1.33 | 2.33 | +1.00 | 15/15 | 792 | 3/3 | +| CigaretBH | dodger | 145.2 | 186.6 | +41.4 | 2.33 | 3.33 | +1.00 | 15/15 | 863 | 3/3 | +| Diamond | dodger | 99.4 | 91.4 | -8.0 | 0.00 | 0.33 | +0.33 | 15/15 | 1717 | 3/3 | +| Dookious | dodger | 153.3 | 158.8 | +5.5 | 1.67 | 1.67 | +0.00 | 15/15 | 1206 | 3/3 | +| GresSuffurd | dodger | 192.8 | 163.7 | -29.1 | 1.67 | 2.67 | +1.00 | 15/15 | 1357 | 3/3 | +| Jen | dodger | 354.5 | 362.5 | +8.0 | 5.00 | 5.00 | +0.00 | 15/15 | 730 | 3/3 | +| Komarious | dodger | 223.9 | 265.8 | +41.9 | 2.67 | 3.33 | +0.67 | 15/15 | 1341 | 3/3 | +| KurtWaveSurfer | dodger | 357.4 | 354.3 | -3.1 | 3.33 | 2.33 | -1.00 | 15/15 | 1466 | 3/3 | +| LionWWSVMvoid | dodger | 469.3 | 473.9 | +4.5 | 5.00 | 5.00 | +0.00 | 15/15 | 849 | 3/3 | +| Lukious | dodger | 238.8 | 201.7 | -37.1 | 2.67 | 3.00 | +0.33 | 15/15 | 1307 | 3/3 | +| RougeDC | dodger | 327.9 | 301.1 | -26.8 | 5.00 | 5.00 | +0.00 | 15/15 | 682 | 3/3 | +| WaveSurferGF | dodger | 171.1 | 115.2 | -55.9 | 0.33 | 0.00 | -0.33 | 15/15 | 851 | 3/3 | +| WaveSurferPG | dodger | 123.1 | 146.2 | +23.1 | 0.67 | 0.33 | -0.33 | 15/15 | 877 | 3/3 | +| Ascendant | other | 96.8 | 85.3 | -11.5 | 0.00 | 0.00 | +0.00 | 15/15 | 1078 | 3/3 | +| BrokenSword | other | 140.4 | 124.9 | -15.5 | 3.33 | 3.33 | +0.00 | 15/15 | 631 | 3/3 | +| Coriantumr | other | 108.6 | 153.1 | +44.5 | 1.67 | 2.00 | +0.33 | 15/15 | 1055 | 3/3 | +| DiamondHawk | other | 234.0 | 252.0 | +18.0 | 3.00 | 3.67 | +0.67 | 15/15 | 1444 | 3/3 | +| DrussGT | other | 203.2 | 185.1 | -18.1 | 2.67 | 2.00 | -0.67 | 15/15 | 3602 | 3/3 | +| LightningBug | other | 408.7 | 402.7 | -6.0 | 5.00 | 5.00 | +0.00 | 15/15 | 403 | 3/3 | +| Phoenix | other | 112.8 | 83.4 | -29.4 | 0.33 | 0.00 | -0.33 | 15/15 | 1135 | 3/3 | +| RetroGirl | other | 321.0 | 227.9 | -93.1 | 3.00 | 2.00 | -1.00 | 15/15 | 1317 | 3/3 | +| Shadow | other | 113.4 | 103.5 | -9.9 | 0.33 | 0.33 | +0.00 | 15/15 | 1251 | 3/3 | +| TripHammer | other | 91.4 | 88.0 | -3.4 | 0.33 | 0.33 | +0.00 | 15/15 | 1434 | 3/3 | +| YersiniaPestis | other | 84.8 | 88.5 | +3.8 | 0.33 | 0.00 | -0.33 | 15/15 | 1192 | 3/3 | +| BlitzBat | regular | 129.0 | 115.4 | -13.6 | 3.67 | 3.33 | -0.33 | 15/15 | 646 | 3/3 | +| DiamondStealer | regular | 273.6 | 264.8 | -8.8 | 1.33 | 1.33 | +0.00 | 15/15 | 797 | 3/3 | +| FloodMini | regular | 392.7 | 399.3 | +6.7 | 5.00 | 5.00 | +0.00 | 15/15 | 301 | 3/3 | +| HawkOnFire | regular | 236.6 | 210.7 | -25.9 | 3.33 | 2.33 | -1.00 | 15/15 | 978 | 3/3 | +| PatternRobot | regular | 111.7 | 116.3 | +4.6 | 5.00 | 5.00 | +0.00 | 15/15 | 136 | 3/3 | +| WallAvoider | regular | 281.1 | 325.8 | +44.7 | 3.33 | 3.67 | +0.33 | 15/15 | 886 | 3/3 | + +## Pooled arm totals (all retained opponents) + +* `pattern`: damage/run = 214.5, round wins = 237/480 = 49.4%, damage taken/run = 291.5 +* `bb`: damage/run = 210.9, round wins = 239/480 = 49.8%, damage taken/run = 296.4 + +## Cross-opponent sign test (HEADLINE) + +* **damage/run**: `bb` better on **13** opponents, worse on **19**, tie 0 (of 32) — two-sided sign-test p=0.3771 +* **wins/run**: `bb` better on **10** opponents, worse on **9**, tie 13 (of 32) — two-sided sign-test p=1 + +## Pooled paired estimate (per-opponent deltas; outlier-resistant view) + +* **damage/run**: mean Δ = -3.621, between-opponent SD = 31.654, SE = 5.596, 95% CI ≈ [-14.588, +7.347], range [-93.098, +68.117] + sign-flip permutation (monte-carlo, 1000000 draws): p=0.5266 (MC se=5.0e-04) + MDE at n=32 opponents (α=0.05, 80% power, paired): 15.677 damage per run +* **wins/run**: mean Δ = +0.021, between-opponent SD = 0.521, SE = 0.092, 95% CI ≈ [-0.160, +0.202], range [-1.000, +1.000] + sign-flip permutation (monte-carlo, 1000000 draws): p=0.9106 (MC se=2.9e-04) + MDE at n=32 opponents (α=0.05, 80% power, paired): 0.258 wins per run + +## Style split — is Δ larger on REGULAR movers than on dodgers? + +(style labels are INFERRED from robots.json name/docs — see tools/ab/opponents_gauntlet.txt) + +* **dodger** (n=15): mean Δdmg/run = -0.2 (SD 33.6, MDE 24.3); mean Δwins/run = +0.20 (SD 0.56, MDE 0.41) +* **other** (n=11): mean Δdmg/run = -11.0 (SD 33.7, MDE 28.5); mean Δwins/run = -0.12 (SD 0.45, MDE 0.38) +* **regular** (n=6): mean Δdmg/run = +1.3 (SD 24.4, MDE 27.9); mean Δwins/run = -0.17 (SD 0.46, MDE 0.53) + +* regular vs dodger Δdmg/run: Mann-Whitney p=0.8457 (U=42); regular n=6, dodger n=15 +* regular vs dodger Δwins/run: Mann-Whitney p=0.1862 (U=28) + +## Movement characterisation (MEASURED from the opponent's own trace) + +straight% = ticks with |turn| < 1 deg; wall% = within 50 px of a wall; mode% = single most common 1-deg |turn| value (a straight-liner, a spinner and a fixed oscillator all concentrate here — the observable signature of a 'regular' mover) + +| opponent | inferred | straight% | wall% | full-spd% | mode% (|turn|) | Δ dmg | +|---|---|---:|---:|---:|---:|---:| +| Aristocles | dodger | 94.7 | 37.4 | 0.6 | 94.7 (0) | -29.3 | +| Ascendant | other | 47.0 | 14.4 | 16.7 | 40.1 (1) | -11.5 | +| BlitzBat | regular | 78.5 | 12.6 | 38.5 | 78.0 (0) | -13.6 | +| BrokenSword | other | 79.4 | 11.4 | 36.1 | 79.2 (0) | -15.5 | +| CassiusClay | dodger | 43.0 | 25.8 | 24.3 | 41.5 (1) | -6.1 | +| Cigaret | dodger | 83.9 | 16.6 | 17.4 | 83.3 (0) | +68.1 | +| CigaretBH | dodger | 84.1 | 44.2 | 15.5 | 83.3 (0) | +41.4 | +| Coriantumr | other | 80.7 | 33.4 | 31.4 | 80.0 (0) | +44.5 | +| Diamond | dodger | 62.5 | 38.2 | 20.3 | 54.9 (0) | -8.0 | +| DiamondHawk | other | 81.8 | 48.7 | 35.5 | 81.4 (0) | +18.0 | +| DiamondStealer | regular | 10.8 | 2.3 | 13.7 | 29.0 (4) | -8.8 | +| Dookious | dodger | 73.7 | 47.4 | 21.7 | 72.0 (0) | +5.5 | +| DrussGT | other | 67.8 | 31.2 | 21.7 | 63.4 (0) | -18.1 | +| FloodMini | regular | 93.8 | 39.9 | 0.0 | 93.7 (0) | +6.7 | +| GresSuffurd | dodger | 65.2 | 56.3 | 22.1 | 60.2 (0) | -29.1 | +| HawkOnFire | regular | 80.7 | 46.8 | 37.0 | 80.3 (0) | -25.9 | +| Jen | dodger | 71.0 | 77.8 | 48.2 | 69.5 (0) | +8.0 | +| Komarious | dodger | 74.9 | 61.3 | 14.0 | 72.9 (0) | +41.9 | +| KurtWaveSurfer | dodger | 76.9 | 29.0 | 20.7 | 75.3 (0) | -3.1 | +| LightningBug | other | 96.1 | 25.1 | 0.0 | 96.0 (0) | -6.0 | +| LionWWSVMvoid | dodger | 77.9 | 49.5 | 9.4 | 75.4 (0) | +4.5 | +| Lukious | dodger | 71.9 | 58.9 | 32.5 | 70.4 (0) | -37.1 | +| PatternRobot | regular | 49.8 | 32.7 | 15.7 | 39.5 (0) | +4.6 | +| Phoenix | other | 47.5 | 21.7 | 22.8 | 31.2 (0) | -29.4 | +| RetroGirl | other | 64.8 | 43.9 | 30.9 | 63.4 (0) | -93.1 | +| RougeDC | dodger | 63.7 | 65.2 | 46.7 | 52.6 (0) | -26.8 | +| Shadow | other | 50.3 | 17.2 | 19.6 | 39.9 (1) | -9.9 | +| TripHammer | other | 56.6 | 47.5 | 15.9 | 42.7 (0) | -3.4 | +| WallAvoider | regular | 38.5 | 11.4 | 29.2 | 27.8 (0) | +44.7 | +| WaveSurferGF | dodger | 48.5 | 19.2 | 20.3 | 32.9 (0) | -55.9 | +| WaveSurferPG | dodger | 52.5 | 20.3 | 22.7 | 37.1 (1) | +23.1 | +| YersiniaPestis | other | 45.1 | 14.1 | 15.3 | 32.1 (1) | +3.8 | + +* Spearman(predictability, Δdmg/run) = +0.178 (p≈0.321, n=32) +* Spearman(predictability, Δwins/run) = +0.238 (p≈0.179) +* mean measured predictability of inferred 'regular' movers: 0.58 (n=6) +* mean measured predictability of inferred 'dodger' movers: 0.65 (n=15) + +## Validity / liveness + +* every retained opponent fired: min opponent fires (summed over both arms) = 136 +* arm env reached the bot in every run that was kept: 0 run(s) had a missing declared env var (verified from each bot's own boot report) +* total battle attempts 192 for 192 retained runs (0 arm/opponent cells needed a retry) +* retained rounds: ref=480, test=480 + diff --git a/common_libs/tests/fixtures/gauntlet_bitbrain_vs_pattern_session.json b/common_libs/tests/fixtures/gauntlet_bitbrain_vs_pattern_session.json new file mode 100644 index 0000000..864756a --- /dev/null +++ b/common_libs/tests/fixtures/gauntlet_bitbrain_vs_pattern_session.json @@ -0,0 +1,49 @@ +{ + "commit": "da4a971ca99e81894e86a35a521bbef9ba081d3f", + "binary_sha256": "f5876e7e8c14389de8f335b1529868da80d987cea5ca298abc058ecaff1925cf", + "binary": "frozen/ModularBot/ModularBot_bin", + "rounds": 5, + "runs": 3, + "conc": 4, + "timestamp": "2026-09-26T00:36:25+02:00", + "outdir": "/tmp/ab/j117_gauntlet", + "arms": [ + {"name": "pattern", "env": "", "label": ""}, + {"name": "bb", "env": "TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.0,1.25,1.5,2.0 TR_BITBRAIN_MEM=decay", "label": "BitBrain learned gain"} + ], + "opponents": [ + {"name": "WallAvoider", "dir": "/tmp/tr_bots/WallAvoider", "style": "regular"}, + {"name": "HawkOnFire", "dir": "/tmp/tr_bots/HawkOnFire", "style": "regular"}, + {"name": "DiamondStealer", "dir": "/tmp/tr_bots/DiamondStealer", "style": "regular"}, + {"name": "BlitzBat", "dir": "/tmp/tr_bots/BlitzBat", "style": "regular"}, + {"name": "PatternRobot", "dir": "/tmp/tr_bots/PatternRobot", "style": "regular"}, + {"name": "FloodMini", "dir": "/tmp/tr_bots/FloodMini", "style": "regular"}, + {"name": "Aurora", "dir": "/tmp/tr_bots/Aurora", "style": "regular"}, + {"name": "Diamond", "dir": "/tmp/tr_bots/Diamond", "style": "dodger"}, + {"name": "Dookious", "dir": "/tmp/tr_bots/Dookious", "style": "dodger"}, + {"name": "Lukious", "dir": "/tmp/tr_bots/Lukious", "style": "dodger"}, + {"name": "GresSuffurd", "dir": "/tmp/tr_bots/GresSuffurd", "style": "dodger"}, + {"name": "KurtWaveSurfer", "dir": "/tmp/tr_bots/KurtWaveSurfer", "style": "dodger"}, + {"name": "CassiusClay", "dir": "/tmp/tr_bots/CassiusClay", "style": "dodger"}, + {"name": "WaveSurferGF", "dir": "/tmp/tr_bots/WaveSurferGF", "style": "dodger"}, + {"name": "WaveSurferPG", "dir": "/tmp/tr_bots/WaveSurferPG", "style": "dodger"}, + {"name": "Jen", "dir": "/tmp/tr_bots/Jen", "style": "dodger"}, + {"name": "Cigaret", "dir": "/tmp/tr_bots/Cigaret", "style": "dodger"}, + {"name": "CigaretBH", "dir": "/tmp/tr_bots/CigaretBH", "style": "dodger"}, + {"name": "Komarious", "dir": "/tmp/tr_bots/Komarious", "style": "dodger"}, + {"name": "LionWWSVMvoid", "dir": "/tmp/tr_bots/LionWWSVMvoid", "style": "dodger"}, + {"name": "RougeDC", "dir": "/tmp/tr_bots/RougeDC", "style": "dodger"}, + {"name": "Aristocles", "dir": "/tmp/tr_bots/Aristocles", "style": "dodger"}, + {"name": "DrussGT", "dir": "/tmp/tr_bots/DrussGT", "style": "other"}, + {"name": "Shadow", "dir": "/tmp/tr_bots/Shadow", "style": "other"}, + {"name": "Phoenix", "dir": "/tmp/tr_bots/Phoenix", "style": "other"}, + {"name": "Coriantumr", "dir": "/tmp/tr_bots/Coriantumr", "style": "other"}, + {"name": "BrokenSword", "dir": "/tmp/tr_bots/BrokenSword", "style": "other"}, + {"name": "TripHammer", "dir": "/tmp/tr_bots/TripHammer", "style": "other"}, + {"name": "RetroGirl", "dir": "/tmp/tr_bots/RetroGirl", "style": "other"}, + {"name": "DiamondHawk", "dir": "/tmp/tr_bots/DiamondHawk", "style": "other"}, + {"name": "Ascendant", "dir": "/tmp/tr_bots/Ascendant", "style": "other"}, + {"name": "YersiniaPestis", "dir": "/tmp/tr_bots/YersiniaPestis", "style": "other"}, + {"name": "LightningBug", "dir": "/tmp/tr_bots/LightningBug", "style": "other"} + ] +} diff --git a/docs/gauntlet_bitbrain_vs_pattern.md b/docs/gauntlet_bitbrain_vs_pattern.md new file mode 100644 index 0000000..36af331 --- /dev/null +++ b/docs/gauntlet_bitbrain_vs_pattern.md @@ -0,0 +1,220 @@ +# BitBrain vs Pattern across 32 legacy opponents — the generalization gauntlet + +**This is the repo's first multi-opponent gun measurement.** Every previous +gun/movement claim in this project was measured against the single opponent +DrussGT; that caveat was flagged repeatedly and never closed until now. j114's +28 validated legacy champions (`tools/robocode_shim/robots.json`) made a +per-adversary gauntlet possible, so this test asks whether the DrussGT-only +conclusions *generalize*. + +The claim under test is the bot owner's, from watching the GUI (2026-09-25): +*"BitBrain is a killer gun, especially against regular movements (spinners, +wall-followers), and its fast adaptation is the reason."* The counter-evidence +was 30 runs/arm vs DrussGT where BitBrain and Pattern were **identical** +(97/210 vs 97/210 round wins) and the learned-gain config was the worst arm +(`docs/bitbrain_vs_tmhorizon_ab.md`, `docs/bitbrain_gun_verdict.md`). + +> ## DIRECT ANSWERS (MEASURED, 32 opponents × 2 arms × 3 runs × 5 rounds) +> +> * **Does BitBrain generalize beyond DrussGT?** **No.** Pooled over 32 +> opponents the two arms are indistinguishable: damage/run **214.5 vs 210.9** +> (`bb - pattern` = **-3.6** ± 5.6, sign-flip p=0.53) and round wins +> **237/480 = 49.4% vs 239/480 = 49.8%** (delta **+0.02** wins/run, p=0.91). +> The cross-opponent sign test favours neither arm: BitBrain wins on **13/32** +> opponents on damage (Pattern on 19) and on **10** opponents on round wins +> (Pattern 9, 13 ties). **The DrussGT-specific penalty does not carry** — on +> DrussGT alone BitBrain was -18.1 damage/run, but across the field the +> estimate collapses to ~0. BitBrain is a wash, not a killer. +> * **Is it specifically stronger against regular/periodic movers?** **No.** +> On the 6 inferred regular movers (wall-followers, campers, rammers, +> flood-fill) the paired delta is **+1.3 damage/run** (SD 24.4); on the 15 +> inferred dodgers it is **-0.2** (SD 33.6). Regular-vs-dodger Mann-Whitney +> **p=0.85** (damage) / **p=0.19** (wins). The measured movement predictability +> does not correlate with the delta either (Spearman +0.18, p=0.32). The +> owner's observation is **not** supported: the tournament's MDE at 32 +> opponents is **15.7 damage/run** and **0.26 wins/run**, and the point +> estimate sits at zero, so an effect of the claimed (visible) size would have +> shown up. +> * **Caveat that limits the sub-claim:** the legacy roster has **no true +> constant-turn spinner** — the "regular" bucket is wall-followers / corner +> campers / a rammer / a flood-filler. The spinner-specific half of the claim +> is therefore *untested*, not refuted. What is refuted is the general +> "killer vs regular movers" reading. + +## Setup `[MEASURED]` + +One frozen `ModularBot` built from `git archive HEAD` at run-time commit +`da4a971ca99e81894e86a35a521bbef9ba081d3f` (binary sha256 +`f5876e7e8c14389de8f335b1529868da80d987cea5ca298abc058ecaff1925cf`). **33 +opponents attempted, 32 retained, 2 arms, 3 runs × 5 rounds each = +192 battles / 960 rounds**, `--conc 4`, **0 failed, 0 retries needed**. Raw +per-tick captures live at `/tmp/ab/j117_gauntlet/` and are **not** committed; +the full machine report is +`common_libs/tests/fixtures/gauntlet_bitbrain_vs_pattern_report.txt` and the +session header (arms, opponents, styles) is +`..._session.json`. + +| arm | env | role | +|---|---|---| +| `pattern` | *(none)* | shipped default (`onlyPattern` rack) — reference | +| `bb` | `TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.0,1.25,1.5,2.0 TR_BITBRAIN_MEM=decay` | BitBrain-only, the learned-gain config the owner most likely ran | + +Runner `tools/ab/gauntlet_run.sh` (per-opponent A/B; subject = the frozen +ModularBot itself, adversary iterates over the roster), arm list +`tools/ab/arms_gauntlet.txt`, opponents + inferred movement styles +`tools/ab/opponents_gauntlet.txt`, analyzer `tools/ab/gauntlet_analyze.py`. + +## Validity / false-negative guard `[MEASURED]` + +Per j114's warning, `run_smoke.sh` has false negatives and only a real battle is +authoritative, so each battle is validated and (if the bot failed to connect +inside the booter's 30 s window) retried up to 4 times. Results: + +* **0 of 192 battles needed a retry**; every *retained* opponent fired and was + scored (the roster also includes the 5 `WORKS_WEAK` bots — valid movement + targets — of which only Aurora turned out inert). +* The least-active retained opponent fired **136** times across both arms; the + one inert bot (Aurora — a single fire in 6 battles) was **dropped**. +* Every declared arm env var reached the process, verified from each bot's own + `[env]` boot report: `TR_RACK_BITBRAIN = both`, `TR_RACK_PATTERN = off`, + `TR_BITBRAIN_GAINS = 1.0,1.25,1.5,2.0`, `TR_BITBRAIN_MEM = decay`, and the + active 1v1 rack line reads `BITBRAIN` for `bb` and `PATTERN` for `pattern`. + +## 1. Per-opponent result `[MEASURED]` + +The unit of evidence is the **number of opponents**, so the design spends its +budget on breadth (3 runs/opponent) rather than depth. Full table in the fixture; +the deltas that matter: + +| opponent | style (INFERRED) | Δ dmg/run | Δ wins/run | +|---|---|---:|---:| +| Cigaret | dodger | +68.1 | +1.00 | +| WallAvoider | regular | +44.7 | +0.33 | +| Coriantumr | other | +44.5 | +0.33 | +| Komarious | dodger | +41.9 | +0.67 | +| CigaretBH | dodger | +41.4 | +1.00 | +| WaveSurferPG | dodger | +23.1 | -0.33 | +| DiamondHawk | other | +18.0 | +0.67 | +| Jen | dodger | +8.0 | 0.00 | +| FloodMini | regular | +6.7 | 0.00 | +| Dookious | dodger | +5.5 | 0.00 | +| PatternRobot | regular | +4.6 | 0.00 | +| LionWWSVMvoid | dodger | +4.5 | 0.00 | +| YersiniaPestis | other | +3.8 | -0.33 | +| KurtWaveSurfer | dodger | -3.1 | -1.00 | +| TripHammer | other | -3.4 | 0.00 | +| LightningBug | other | -6.0 | 0.00 | +| CassiusClay | dodger | -6.1 | +0.33 | +| Diamond | dodger | -8.0 | +0.33 | +| DiamondStealer | regular | -8.8 | 0.00 | +| Shadow | other | -9.9 | 0.00 | +| Ascendant | other | -11.5 | 0.00 | +| BlitzBat | regular | -13.6 | -0.33 | +| BrokenSword | other | -15.5 | 0.00 | +| DrussGT | other | -18.1 | -0.67 | +| HawkOnFire | regular | -25.9 | -1.00 | +| RougeDC | dodger | -26.8 | 0.00 | +| GresSuffurd | dodger | -29.1 | +1.00 | +| Aristocles | dodger | -29.3 | 0.00 | +| Phoenix | other | -29.4 | -0.33 | +| Lukious | dodger | -37.1 | +0.33 | +| WaveSurferGF | dodger | -55.9 | -0.33 | +| RetroGirl | other | -93.1 | -1.00 | + +The individual deltas swing from -93 to +68 damage/run — with only 3 runs per +opponent, a single battle dominates each cell, so **no individual row is +evidence on its own.** The evidence is the aggregate below. + +## 2. Cross-opponent sign test — the headline `[MEASURED]` + +On how many opponents does each arm win (paired `bb - pattern`)? + +``` +metric bb better pattern better tie two-sided sign-test p +damage/run 13 19 0 0.377 +wins/run 10 9 13 1.000 +``` + +Neither arm wins on the majority of opponents. On round wins the two arms are +dead even (10 vs 9, 13 exact ties), which is the round-level echo of the +30-run-vs-DrussGT finding (97/210 vs 97/210). + +## 3. Pooled paired estimate with between-opponent spread `[MEASURED]` + +Per-opponent paired deltas, so one outlier bot cannot carry the result: + +``` +metric mean Δ between-opp SD SE 95% CI sign-flip p MDE(n=32) +damage/run -3.62 31.65 5.60 [-14.59, +7.35] 0.527 15.68 +wins/run +0.02 0.52 0.09 [-0.16, +0.20] 0.911 0.258 +``` + +Pooled arm totals over the 32 opponents: + +``` +arm damage/run damage taken/run round wins +pattern 214.5 291.5 237/480 = 49.4% +bb 210.9 296.4 239/480 = 49.8% +``` + +Both verdict metrics (damage/run and round wins, per the standing rule) say the +same thing: **no detectable difference.** The MDE at 32 opponents is 15.7 +damage/run (≈7% of the ~214 baseline) and 0.258 wins/run (≈5.2 pp); the point +estimates are ~0. So a *large* BitBrain advantage is excluded, and the observed +effect is at the null. The usual caveat stands: a genuinely small positive +effect (<~8%) is **not** excluded by this design — but there is no evidence of +one. + +## 4. Style split and measured movement character `[MEASURED]` + +Style labels are **INFERRED** from the bot names/docs in `robots.json`. To +check them against data, the opponent's own captured trajectory (`sx,sy,sh,ss`) +was characterised by its single most-common turn magnitude (`mode%`): a +straight-liner, a spinner and a fixed oscillator all concentrate there, an +adaptive surfer does not. + +``` +style n mean Δdmg/run (SD) mean Δwins/run (SD) measured mode% +regular 6 +1.3 (24.4) -0.17 (0.46) 0.58 +dodger 15 -0.2 (33.6) +0.20 (0.56) 0.65 +other 11 -11.0 (33.7) -0.12 (0.45) 0.59 + +regular vs dodger: Δdmg/run Mann-Whitney p=0.85 + Δwins/run Mann-Whitney p=0.19 +Spearman(measured mode%, Δdmg/run) = +0.178 (p≈0.32, n=32) +``` + +**The directional sub-claim is not supported and not even close**: if BitBrain +were a killer against regular movers, the `regular` bucket should sit clearly +above the `dodger` bucket; instead both are ~0 and the difference is noise. +The measured movement character also fails to separate the inferred groups (the +"regular" bucket is not measurably more periodic than the "dodger" bucket), so +the sub-claim is doubly weak: not only is the delta not larger on regular +movers, the label itself is not confirmed by the traces. The several big +per-opponent deltas are scattered across both buckets in both directions +(e.g. +68 Cigaret/dodger, -56 WaveSurferGF/dodger, +45 WallAvoider/regular, +-26 HawkOnFire/regular), which is what pure noise looks like. + +## 5. What this changes about the earlier DrussGT-only conclusions `[MEASURED/INFERRED]` + +* **MEASURED:** on DrussGT alone in this gauntlet BitBrain is -18.1 damage/run + (1 run-mean of 3); in the larger j113 test the learned arm was -25.9 + damage/run vs Pattern (p=0.048). Both are *single-opponent* results. +* **MEASURED:** across 32 opponents the pooled estimate is -3.6 (p=0.53). +* **INFERRED (the lesson):** the DrussGT-only penalty is an opponent-specific + interaction, not a general property of the gun. Likewise the owner's + GUI impression of a "killer gun" is an opponent-specific (or + small-sample) impression that does not survive a broad field. If anything, + the broad field slightly favours Pattern on damage (19/32 opponents). + +## MEASURED vs INFERRED + +* **MEASURED:** every per-opponent delta and round-win count, the pooled + estimates, the sign test, the sign-flip permutation p-values, the MDE, the + movement traces (`straight%`, `wall%`, `mode%`), the liveness checks (fire + counts, 0 retries, env reached the bot), and the raw captures. +* **INFERRED:** the movement *style* labels (from names/docs; the measured + `mode%` does not confirm them), and the reading of the DrussGT-vs-field + discrepancy as an opponent-specific interaction rather than a code + difference. The spinner-specific half of the owner's claim is untested + because no constant-turn spinner is in the roster. diff --git a/tools/ab/arms_gauntlet.txt b/tools/ab/arms_gauntlet.txt new file mode 100644 index 0000000..4c1ff20 --- /dev/null +++ b/tools/ab/arms_gauntlet.txt @@ -0,0 +1,8 @@ +# BitBrain vs Pattern gauntlet — two arms only, so the per-opponent sample is +# as large as the battle budget allows. +# +# pattern = shipped default (onlyPattern rack), NO env reference +# bb = Pattern off, BitBrain on, the learned-gain config the owner +# most likely ran (gains 1.0..2.0, decay memory) +pattern | +bb | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.0,1.25,1.5,2.0 TR_BITBRAIN_MEM=decay | BitBrain learned gain diff --git a/tools/ab/gauntlet_analyze.py b/tools/ab/gauntlet_analyze.py new file mode 100755 index 0000000..0afc016 --- /dev/null +++ b/tools/ab/gauntlet_analyze.py @@ -0,0 +1,644 @@ +#!/usr/bin/env python3 +"""gauntlet_analyze.py — per-adversary BitBrain-vs-Pattern gauntlet analysis. + + python3 tools/ab/gauntlet_analyze.py [--report FILE] + +Reads a session dir produced by tools/ab/gauntlet_run.sh and prints: + + * the PER-OPPONENT table (opponent, inferred movement style, damage/run in + each arm, paired damage delta, round wins in each arm, paired wins delta, + rounds sampled, opponent fire count for liveness); + * the CROSS-OPPONENT SIGN TEST on the paired deltas — the headline: on how + many opponents does each arm win (bb - pattern > 0); + * the pooled paired estimate WITH the between-opponent spread (so a single + outlier bot cannot carry it), a sign-flip permutation test and the MDE; + * the STYLE SPLIT: is the bb - pattern delta larger on the regular movers + (wall-followers / campers / rammers / periodic) than on the dodgers? + +Standard library only, deterministic. + +Metrics (the standing rule): damage dealt per run and ROUND WINS. Never hit +rate alone. Subject of every capture is ModularBot itself, so `subject event +counts` are ModularBot's own counts and `firstPlaces` is ModularBot's own +name-based round-win count. + +Liveness / false-negative guard: a run is dropped unless the capture shows real +rows, ModularBot appears in the runner's RESULTS block (firstPlaces) and the +event sidecar attributes owners unambiguously. An opponent is dropped unless it +fired at least once in the retained runs — an opponent that never fires is not +an opponent (per tools/robocode_shim/LEGACY_BOTS.md the synthetic smoke test has +false negatives, so only a real battle counts). +""" +import glob +import itertools +import json +import math +import os +import random +import re +import statistics +import sys + +BOT_NAME = "ModularBot" +EXACT_PAIR_CAP = 20 # 2**20 = 1.05M sign-flip enumerations, still fast +MC_DRAWS = 1_000_000 +MC_SEED = 0x5EED5EED +Z_ALPHA_POWER = 1.959963984540054 + 0.8416212335729143 # 2-sample MDE const + + +# ── loading / parsing ──────────────────────────────────────────────────────── + +def parse_envspec(envspec): + """`TR_A=1 TR_B=2` -> {'TR_A': '1', 'TR_B': '2'}.""" + out = {} + for tok in (envspec or "").split(): + if "=" in tok: + k, v = tok.split("=", 1) + out[k] = v + return out + + +def read_lines(path): + try: + with open(path, errors="replace") as fh: + return fh.readlines() + except OSError: + return [] + + +def parse_events(path): + evs = [] + for line in read_lines(path): + line = line.strip() + if not line: + continue + try: + evs.append(json.loads(line)) + except json.JSONDecodeError: + continue + return evs + + +def parse_first_places(log_text): + """ModularBot's run-level `firstPlaces` (name-based round wins).""" + m = re.search( + r"^\s*#\d+\s+" + re.escape(BOT_NAME) + + r"\s+totalScore=-?\d+\s+firstPlaces=(\d+)", log_text, re.MULTILINE) + return int(m.group(1)) if m else None + + +def parse_counters(log_text): + m = re.search( + r"subject event counts: scans=(\d+) bulletsFired=(\d+) bulletHits=(\d+)" + r" bulletMisses=(\d+) bulletHitBullets=(\d+) hitsTaken=(\d+)", log_text) + if not m: + return None + return {"fired": int(m.group(2)), "hits": int(m.group(3)), + "hits_taken": int(m.group(6))} + + +def round_count(path): + try: + return len(json.load(open(path)).get("rounds", [])) + except (OSError, json.JSONDecodeError, AttributeError): + return 0 + + +def attribute_subject(evs, counters): + """Return (subject_id, other_id) from the fire/hit event counts. + + The subject is ModularBot, so its counts must equal the capture's subject + counters line. Returns (None, None) when attribution is ambiguous. + """ + fires, hits, victim_hits = {}, {}, {} + for o in evs: + t = o.get("type") + if t == "fire": + fires[o["owner"]] = fires.get(o["owner"], 0) + 1 + elif t == "hit": + hits[o["owner"]] = hits.get(o["owner"], 0) + 1 + if "victim" in o: + victim_hits[o["victim"]] = victim_hits.get(o["victim"], 0) + 1 + if not fires: + return None, None + if counters: + strict = [o for o in fires + if fires[o] == counters["fired"] + and hits.get(o, 0) == counters["hits"] + and victim_hits.get(o, 0) == counters["hits_taken"]] + if len(strict) == 1: + subj = strict[0] + other = [o for o in fires if o != subj] + # other may legitimately be absent: an inert opponent that never fires + return subj, (other[0] if len(other) == 1 else None) + if len(fires) == 2: + ids = list(fires) + # fall back to the owner that fired the subject's declared count + if counters: + cand = [o for o in ids if fires[o] == counters["fired"]] + if len(cand) == 1: + subj = cand[0] + return subj, [o for o in ids if o != subj][0] + return ids[0], ids[1] + if len(fires) == 1 and counters: + only = next(iter(fires)) + if fires[only] == counters["fired"]: + return only, None + return None, None + + +def parse_run(armdir, run, env_req=None): + """Per-run data; `valid` is False for a false-negative / unstarted battle.""" + evs = parse_events(os.path.join(armdir, f"run{run}.events.jsonl")) + log_text = "".join(read_lines(os.path.join(armdir, f"run{run}.battle.log"))) + counters = parse_counters(log_text) + wins = parse_first_places(log_text) + nrounds = round_count(os.path.join(armdir, f"run{run}.jsonl.rounds.json")) + + r = {"run": run, "valid": False, "rounds": nrounds, "wins": wins, + "damage": 0.0, "damage_taken": 0.0, "opp_fired": 0, "mb_fired": 0, + "hits": 0, "attempts": None, "env_ok": True} + rf = os.path.join(armdir, f"run{run}.attempts") + if os.path.exists(rf): + try: + r["attempts"] = int(open(rf).read().strip()) + except ValueError: + pass + + # env liveness: every declared VAR=VALUE must appear verbatim in the bot's + # own boot report, else the arm setting never reached the process. + if env_req: + stdout_text = "".join(read_lines(os.path.join(armdir, f"run{run}.bot.stdout.log"))) + r["env_ok"] = all( + re.search(re.escape(k) + r"\s*=\s*" + re.escape(v) + r"(\s|$)", stdout_text) + is not None for k, v in env_req.items()) + + if wins is None or counters is None or nrounds == 0: + return r # bot never appeared in the results block + subj, other = attribute_subject(evs, counters) + if subj is None: + return r # ambiguous owner attribution + + for o in evs: + if o.get("type") == "hit": + dmg = o.get("damage", 0.0) + if o.get("owner") == subj: + r["damage"] += dmg + r["hits"] += 1 + if o.get("victim") == subj: + r["damage_taken"] += dmg + elif o.get("type") == "fire": + if o.get("owner") == subj: + r["mb_fired"] += 1 + elif o.get("owner") == other: + r["opp_fired"] += 1 + r["valid"] = True + return r + + +# ── statistics ─────────────────────────────────────────────────────────────── + +def sign_test(deltas): + """Two-sided exact sign test. Returns (pos, neg, ties, p).""" + pos = sum(1 for d in deltas if d > 1e-12) + neg = sum(1 for d in deltas if d < -1e-12) + ties = len(deltas) - pos - neg + n = pos + neg + if n == 0: + return pos, neg, ties, 1.0 + k = min(pos, neg) + tail = sum(math.comb(n, i) for i in range(0, k + 1)) / (2 ** n) + return pos, neg, ties, min(1.0, 2.0 * tail) + + +def signflip_perm(deltas): + """Paired sign-flip permutation test on mean(delta)==0 (two-sided).""" + n = len(deltas) + if n == 0: + return None + obs = abs(sum(deltas) / n) + if n <= EXACT_PAIR_CAP: + cnt = 0 + for signs in itertools.product((1.0, -1.0), repeat=n): + m = abs(sum(s * d for s, d in zip(signs, deltas)) / n) + if m >= obs - 1e-12: + cnt += 1 + return {"p": cnt / (2 ** n), "method": "exact", "draws": 2 ** n, + "se": 0.0} + rng = random.Random(MC_SEED) + B = MC_DRAWS + cnt = 0 + for _ in range(B): + m = abs(sum((1.0 if rng.random() < 0.5 else -1.0) * d for d in deltas)) / n + if m >= obs - 1e-12: + cnt += 1 + p = (cnt + 1) / (B + 1) + return {"p": p, "method": "monte-carlo", "draws": B, + "se": math.sqrt(p * (1.0 - p) / (B + 1))} + + +def mannwhitney_p(xa, xb): + """Two-sided Mann-Whitney U, normal approx with tie + continuity correction.""" + na, nb = len(xa), len(xb) + if na == 0 or nb == 0: + return None + vals = sorted([(v, 0) for v in xa] + [(v, 1) for v in xb]) + n = na + nb + rank = [0.0] * n + tie_term = 0.0 + i = 0 + while i < n: + j = i + while j + 1 < n and vals[j + 1][0] == vals[i][0]: + j += 1 + t = j - i + 1 + tie_term += t ** 3 - t + avg = (i + j) / 2.0 + 1.0 + for k in range(i, j + 1): + rank[k] = avg + i = j + 1 + r1 = sum(rank[k] for k in range(n) if vals[k][1] == 0) + u1 = r1 - na * (na + 1) / 2.0 + mu = na * nb / 2.0 + sigma2 = (na * nb / 12.0) * ((n + 1) - tie_term / (n * (n - 1))) + if sigma2 <= 0: + return 1.0, min(u1, na * nb - u1) + z = (abs(u1 - mu) - 0.5) / math.sqrt(sigma2) + z = max(0.0, z) + return math.erfc(z / math.sqrt(2.0)), min(u1, na * nb - u1) + + +def describe(deltas): + n = len(deltas) + if n == 0: + return None + mean = sum(deltas) / n + sd = statistics.stdev(deltas) if n > 1 else 0.0 + se = sd / math.sqrt(n) if n > 1 else 0.0 + mde = Z_ALPHA_POWER * sd / math.sqrt(n) if n > 1 else 0.0 + return {"n": n, "mean": mean, "sd": sd, "se": se, "mde": mde, + "min": min(deltas), "max": max(deltas)} + + +# ── aggregation ────────────────────────────────────────────────────────────── + +def arm_summary(armdir, runs, env_req=None): + tot = {"valid_runs": 0, "runs": len(runs), "damage": 0.0, + "damage_taken": 0.0, "wins": 0, "rounds": 0, "opp_fired": 0, + "mb_fired": 0, "attempts": 0, "dropped": [], "env_bad": []} + for run in runs: + r = parse_run(armdir, run, env_req) + if not r["valid"]: + tot["dropped"].append(run) + continue + if not r["env_ok"]: + tot["env_bad"].append(run) + tot["valid_runs"] += 1 + tot["damage"] += r["damage"] + tot["damage_taken"] += r["damage_taken"] + tot["wins"] += r["wins"] + tot["rounds"] += r["rounds"] + tot["opp_fired"] += r["opp_fired"] + tot["mb_fired"] += r["mb_fired"] + tot["attempts"] += (r["attempts"] or 1) + vr = tot["valid_runs"] + tot["damage_per_run"] = tot["damage"] / vr if vr else None + tot["wins_per_run"] = tot["wins"] / vr if vr else None + tot["wins_per_round"] = tot["wins"] / tot["rounds"] if tot["rounds"] else None + return tot + + +def discover_runs(armdir): + return sorted(int(m.group(1)) for m in + (re.fullmatch(r"run(\d+)\.jsonl", f) for f in os.listdir(armdir)) + if m) + + +def movement_metrics(outdir, opponent, arms): + """MEASURED character of the OPPONENT's movement, pooled over both arms. + + The capture stores the opponent as the `s*` bot (sx,sy,sh,ss). We report + wall hugging, straight-line and PERIODICITY signatures. `periodicity` is + the strongest match of the signed per-tick turn series against any lag in + 2..60 (a spinner/oscillator repeats its turn rate; a random dodger does + not) and `repeat` is the lag-1 version. This lets the inferred style + labels be checked against data and gives a continuous 'regular mover' axis. + """ + wall = full = n = 0 + straight = 0.0 + turns = [] + prev = None + for arm in arms: + adir = os.path.join(outdir, opponent, arm) + if not os.path.isdir(adir): + continue + for run in discover_runs(adir): + p = os.path.join(adir, f"run{run}.jsonl") + try: + fh = open(p, errors="replace") + except OSError: + continue + with fh: + for line in fh: + if '"sx"' not in line: + continue + try: + o = json.loads(line) + except json.JSONDecodeError: + continue + if "sx" not in o: + continue + x, y, sp, hd = o["sx"], o["sy"], o.get("ss", 0.0), o.get("sh", 0.0) + n += 1 + if min(x, 800 - x, y, 600 - y) < 50: + wall += 1 + if sp >= 7.5: + full += 1 + if prev is not None: + px, py, ph = prev + dist = math.hypot(x - px, y - py) + if dist <= 30.0: # ignore round-reset teleports + dh = (hd - ph + 180.0) % 360.0 - 180.0 + turns.append(dh) + if abs(dh) < 1.0: + straight += 1 + prev = (x, y, hd) + if n == 0: + return None + t = len(turns) + repeat = 0 + if t > 1: + repeat = sum(1 for i in range(1, t) if abs(turns[i] - turns[i - 1]) < 0.5) / (t - 1) + # turn-mode fraction: the single most common 1-degree |turn| value. A + # wall-follower going straight (|turn|~0), a spinner (constant |turn|) and + # a fixed oscillator all concentrate here; an adaptive surfer does not. + mode_frac = 0.0 + mode_val = 0.0 + if t: + hist = {} + for v in turns: + b = int(round(abs(v))) + hist[b] = hist.get(b, 0) + 1 + mode_val, cnt = max(hist.items(), key=lambda kv: kv[1]) + mode_frac = cnt / t + return {"n": n, "wall_frac": wall / n, "full_frac": full / n, + "straight_frac": straight / t if t else 0.0, + "repeat": repeat, "mode_frac": mode_frac, "mode_val": mode_val, + "predictability": mode_frac} + + +def spearman(xs, ys): + """Spearman rank correlation, returns (rho, two-sided p) or None.""" + n = len(xs) + if n < 4: + return None + def ranks(v): + order = sorted(range(n), key=lambda i: v[i]) + r = [0.0] * n + i = 0 + while i < n: + j = i + while j + 1 < n and v[order[j + 1]] == v[order[i]]: + j += 1 + avg = (i + j) / 2.0 + 1.0 + for k in range(i, j + 1): + r[order[k]] = avg + i = j + 1 + return r + rx, ry = ranks(xs), ranks(ys) + mx, my = sum(rx) / n, sum(ry) / n + sxy = sum((rx[i] - mx) * (ry[i] - my) for i in range(n)) + sxx = sum((rx[i] - mx) ** 2 for i in range(n)) + syy = sum((ry[i] - my) ** 2 for i in range(n)) + if sxx == 0 or syy == 0: + return 0.0, 1.0 + rho = sxy / math.sqrt(sxx * syy) + # t-approximation for the p-value + t = rho * math.sqrt((n - 2) / max(1e-12, 1 - rho * rho)) + # two-sided p from the Student-t survival via the normal fallback (n>=10 + # in practice here); use a small series-accurate normal approx on t. + p = math.erfc(abs(t) / math.sqrt(2.0)) + return rho, p + + +def main(): + if len(sys.argv) < 2: + print(__doc__) + sys.exit(2) + outdir = sys.argv[1].rstrip("/") + report_path = None + if "--report" in sys.argv: + report_path = sys.argv[sys.argv.index("--report") + 1] + + session = {} + sp = os.path.join(outdir, "session.json") + if os.path.exists(sp): + session = json.load(open(sp)) + arms = [a["name"] for a in session.get("arms", [])] + if len(arms) != 2: + print("ERROR: gauntlet analysis needs exactly 2 arms, got:", arms) + sys.exit(1) + ref, test = arms[0], arms[1] + arm_env = {a["name"]: parse_envspec(a.get("env", "")) + for a in session.get("arms", [])} + opp_meta = {o["name"]: o for o in session.get("opponents", [])} + + lines = [] + def out(s=""): + lines.append(s) + print(s) + + out(f"# gauntlet analysis — {ref} (reference) vs {test} (treatment)") + out() + out(f"session: {outdir}") + out(f"commit: {session.get('commit')} binary sha256 {session.get('binary_sha256')}") + out(f"design: {session.get('runs')} runs x {session.get('rounds')} rounds per " + f"opponent per arm, conc={session.get('conc')}") + out(f"arms: {ref} = {next((a['env'] for a in session.get('arms',[]) if a['name']==ref), '') or '(none)'}") + out(f" {test} = {next((a['env'] for a in session.get('arms',[]) if a['name']==test), '')}") + out() + + opponents = [] + for d in sorted(os.listdir(outdir)): + full = os.path.join(outdir, d) + if not os.path.isdir(full) or d in ("frozen", ".work", "nimcache"): + continue + ra, rb = os.path.join(full, ref), os.path.join(full, test) + if os.path.isdir(ra) and os.path.isdir(rb): + opponents.append(d) + + rows = [] + dropped_opp = [] + env_bad_total = 0 + for opp in opponents: + sa = arm_summary(os.path.join(outdir, opp, ref), discover_runs(os.path.join(outdir, opp, ref)), arm_env.get(ref)) + sb = arm_summary(os.path.join(outdir, opp, test), discover_runs(os.path.join(outdir, opp, test)), arm_env.get(test)) + env_bad_total += len(sa["env_bad"]) + len(sb["env_bad"]) + style = (opp_meta.get(opp, {}).get("style") or "other").strip() or "other" + ok = (sa["valid_runs"] > 0 and sb["valid_runs"] > 0 + and sa["opp_fired"] + sb["opp_fired"] >= 10) + row = {"opp": opp, "style": style, "ref": sa, "test": sb, "ok": ok} + if sa["valid_runs"] == 0 or sb["valid_runs"] == 0: + dropped_opp.append((opp, f"unstarted runs ref={sa['dropped']} test={sb['dropped']}")) + elif sa["opp_fired"] + sb["opp_fired"] < 10: + dropped_opp.append( + (opp, f"inert (only {sa['opp_fired'] + sb['opp_fired']} fires " + f"in both arms — never meaningfully fought)")) + if ok: + row["d_dmg"] = sb["damage_per_run"] - sa["damage_per_run"] + row["d_win"] = sb["wins_per_run"] - sa["wins_per_run"] + rows.append(row) + # drop the raw summary dicts to keep printing clean + + valid = [r for r in rows if r["ok"]] + out(f"opponents: {len(valid)} retained of {len(rows)} attempted; " + f"{len(dropped_opp)} dropped") + for opp, why in dropped_opp: + out(f" dropped {opp}: {why}") + out() + + # ── per-opponent table ─────────────────────────────────────────────────── + out("## Per-opponent table (MEASURED; damage is damage DEALT by ModularBot)") + out() + out(f"| opponent | style | {ref} dmg/run | {test} dmg/run | Δ dmg | " + f"{ref} wins/run | {test} wins/run | Δ wins | rounds/arm | opp fired | runs |") + out("|---|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|") + for r in sorted(valid, key=lambda x: x["style"] + x["opp"]): + sa, sb = r["ref"], r["test"] + out(f"| {r['opp']} | {r['style']} | {sa['damage_per_run']:.1f} | " + f"{sb['damage_per_run']:.1f} | {r['d_dmg']:+.1f} | " + f"{sa['wins_per_run']:.2f} | {sb['wins_per_run']:.2f} | " + f"{r['d_win']:+.2f} | {sa['rounds']}/{sb['rounds']} | " + f"{sa['opp_fired'] + sb['opp_fired']} | {sa['valid_runs']}/{sb['valid_runs']} |") + out() + + d_dmg = [r["d_dmg"] for r in valid] + d_win = [r["d_win"] for r in valid] + + # ── pooled arm totals ──────────────────────────────────────────────── + out("## Pooled arm totals (all retained opponents)") + out() + for arm, key in ((ref, "ref"), (test, "test")): + dm = sum(r[key]["damage"] for r in valid) + wn = sum(r[key]["wins"] for r in valid) + rd = sum(r[key]["rounds"] for r in valid) + out(f"* `{arm}`: damage/run = {dm / sum(r[key]['valid_runs'] for r in valid):.1f}, " + f"round wins = {wn}/{rd} = {100.0 * wn / rd:.1f}%, " + f"damage taken/run = " + f"{sum(r[key]['damage_taken'] for r in valid) / sum(r[key]['valid_runs'] for r in valid):.1f}") + out() + # ── sign test across opponents (the headline) ──────────────────────────── + out("## Cross-opponent sign test (HEADLINE)") + out() + for label, ds in (("damage/run", d_dmg), ("wins/run", d_win)): + pos, neg, ties, p = sign_test(ds) + out(f"* **{label}**: `{test}` better on **{pos}** opponents, worse on " + f"**{neg}**, tie {ties} (of {len(ds)}) — two-sided sign-test p={p:.4g}") + out() + + # ── pooled paired estimate with between-opponent spread ───────────────── + out("## Pooled paired estimate (per-opponent deltas; outlier-resistant view)") + out() + for label, ds in (("damage/run", d_dmg), ("wins/run", d_win)): + d = describe(ds) + sf = signflip_perm(ds) + out(f"* **{label}**: mean Δ = {d['mean']:+.3f}, between-opponent SD = " + f"{d['sd']:.3f}, SE = {d['se']:.3f}, 95% CI ≈ " + f"[{d['mean'] - 1.96 * d['se']:+.3f}, {d['mean'] + 1.96 * d['se']:+.3f}], " + f"range [{d['min']:+.3f}, {d['max']:+.3f}]") + out(f" sign-flip permutation ({sf['method']}, {sf['draws']} draws): " + f"p={sf['p']:.4g}" + (f" (MC se={sf['se']:.1e})" if sf["se"] else "")) + out(f" MDE at n={d['n']} opponents (α=0.05, 80% power, paired): " + f"{d['mde']:.3f} {label.split('/')[0]} per run") + out() + + # ── style split: the user's specific sub-claim ─────────────────────────── + out("## Style split — is Δ larger on REGULAR movers than on dodgers?") + out() + out("(style labels are INFERRED from robots.json name/docs — see " + "tools/ab/opponents_gauntlet.txt)") + out() + groups = {} + for r in valid: + groups.setdefault(r["style"], []).append(r) + for style in sorted(groups): + rs = groups[style] + dd = describe([r["d_dmg"] for r in rs]) + dw = describe([r["d_win"] for r in rs]) + out(f"* **{style}** (n={len(rs)}): mean Δdmg/run = {dd['mean']:+.1f} " + f"(SD {dd['sd']:.1f}, MDE {dd['mde']:.1f}); mean Δwins/run = " + f"{dw['mean']:+.2f} (SD {dw['sd']:.2f}, MDE {dw['mde']:.2f})") + out() + if "regular" in groups and "dodger" in groups: + reg = [r["d_dmg"] for r in groups["regular"]] + dod = [r["d_dmg"] for r in groups["dodger"]] + p, u = mannwhitney_p(reg, dod) + out(f"* regular vs dodger Δdmg/run: Mann-Whitney p={p:.4g} (U={u:.0f}); " + f"regular n={len(reg)}, dodger n={len(dod)}") + reg = [r["d_win"] for r in groups["regular"]] + dod = [r["d_win"] for r in groups["dodger"]] + p, u = mannwhitney_p(reg, dod) + out(f"* regular vs dodger Δwins/run: Mann-Whitney p={p:.4g} (U={u:.0f})") + out() + + # ── movement characterisation (MEASURED) + continuous regularity test ──── + out("## Movement characterisation (MEASURED from the opponent's own trace)") + out() + out("straight% = ticks with |turn| < 1 deg; wall% = within 50 px of a wall; " + "mode% = single most common 1-deg |turn| value (a straight-liner, a " + "spinner and a fixed oscillator all concentrate here — the observable " + "signature of a 'regular' mover)") + out() + out("| opponent | inferred | straight% | wall% | full-spd% | mode% (|turn|) | Δ dmg |") + out("|---|---|---:|---:|---:|---:|---:|") + movers = {} + for r in valid: + m = movement_metrics(outdir, r["opp"], [ref, test]) + if m is None: + continue + movers[r["opp"]] = m + out(f"| {r['opp']} | {r['style']} | {100 * m['straight_frac']:.1f} | " + f"{100 * m['wall_frac']:.1f} | {100 * m['full_frac']:.1f} | " + f"{100 * m['mode_frac']:.1f} ({m['mode_val']:.0f}) | {r['d_dmg']:+.1f} |") + out() + both = [r for r in valid if r["opp"] in movers] + if len(both) >= 4: + xs = [movers[r["opp"]]["predictability"] for r in both] + sp = spearman(xs, [r["d_dmg"] for r in both]) + out(f"* Spearman(predictability, Δdmg/run) = {sp[0]:+.3f} (p≈{sp[1]:.3g}, " + f"n={len(both)})") + sp = spearman(xs, [r["d_win"] for r in both]) + out(f"* Spearman(predictability, Δwins/run) = {sp[0]:+.3f} (p≈{sp[1]:.3g})") + for style in ("regular", "dodger"): + ps = [movers[r["opp"]]["predictability"] for r in both if r["style"] == style] + if ps: + out(f"* mean measured predictability of inferred '{style}' " + f"movers: {statistics.mean(ps):.2f} (n={len(ps)})") + out() + + # ── liveness / validity ────────────────────────────────────────────────── + tot_attempts = sum(r["ref"]["attempts"] + r["test"]["attempts"] for r in valid) + retried = sum(1 for r in valid + for a, d in (("ref", r["ref"]), ("test", r["test"])) + if d["attempts"] > d["valid_runs"]) + out("## Validity / liveness") + out() + out(f"* every retained opponent fired: min opponent fires (summed over both arms) " + f"= {min(r['ref']['opp_fired'] + r['test']['opp_fired'] for r in valid)}") + out(f"* arm env reached the bot in every run that was kept: " + f"{env_bad_total} run(s) had a missing declared env var" + + (" — LOUD FAIL" if env_bad_total else " (verified from each bot's own boot report)")) + out(f"* total battle attempts {tot_attempts} for " + f"{sum(r['ref']['valid_runs'] + r['test']['valid_runs'] for r in valid)} " + f"retained runs ({retried} arm/opponent cells needed a retry)") + out(f"* retained rounds: ref={sum(r['ref']['rounds'] for r in valid)}, " + f"test={sum(r['test']['rounds'] for r in valid)}") + out() + + if report_path: + with open(report_path, "w") as fh: + fh.write("\n".join(lines) + "\n") + + +if __name__ == "__main__": + main() diff --git a/tools/ab/gauntlet_run.sh b/tools/ab/gauntlet_run.sh new file mode 100755 index 0000000..6707cd9 --- /dev/null +++ b/tools/ab/gauntlet_run.sh @@ -0,0 +1,317 @@ +#!/usr/bin/env bash +# ───────────────────────────────────────────────────────────────────────────── +# gauntlet_run.sh — run a PER-ADVERSARY A/B gauntlet of N arms against MANY +# legacy opponents (the generalization test for the gun claims that have, until +# now, always been measured against the single opponent DrussGT). +# +# tools/ab/gauntlet_run.sh \ +# --arms tools/ab/arms_gauntlet.txt \ +# --opponents tools/ab/opponents_gauntlet.txt \ +# --runs 3 --rounds 3 --conc 5 --outdir /tmp/ab/j117_gauntlet +# +# Difference from ab_run.sh: ab_run.sh hard-wires exactly ONE adversary +# (DrussGT, via run_bridge_battle's legacy default subject). Here the SUBJECT +# is the frozen ModularBot itself (so the capture's `subject event counts` are +# ModularBot's own fire/hit/taken counts and `firstPlaces` is ModularBot's +# name-based round wins) and the ADVERSARY iterates over a list of legacy +# champ bot dirs from tools/robocode_shim/robots.json. +# +# For every (opponent, arm, run) it launches a real Tank Royale battle through +# tools/robocode_shim/run_bridge_battle.sh, each in its own process group with +# ephemeral ports and an isolated classic data dir. +# +# Arm file format (same as ab_run.sh): +# name | ENV_VAR=value ENV_VAR2=value2 | optional label +# +# Opponent file format (one per line): +# /tmp/tr_bots/ | style-label # or a bare Name (=> /tmp/tr_bots/) +# Blank lines and #-comments ignored. +# +# Output layout (what gauntlet_analyze.py reads): +# /session.json +# ///run.jsonl tick capture +# ///run.jsonl.rounds.json round boundaries +# ///run.jsonl.results.json cumulative scores +# ///run.events.jsonl fire/hit/death +# ///run.battle.log capture stdout +# ///run.bot.stdout.log ModularBot env +# +# Cleanup: every job runs in its own process group (setsid); on EXIT/INT/TERM +# this script SIGTERMs/SIGKILLs those groups and pkills anything referencing +# this session's outdir, so a Ctrl-C leaves no orphan battles. +# ───────────────────────────────────────────────────────────────────────────── +set -euo pipefail + +REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +RUN_BRIDGE="$REPO/tools/robocode_shim/run_bridge_battle.sh" + +ARMS_FILE="" +OPP_FILE="" +RUNS=3 +ROUNDS=3 +CONC=4 +OUTDIR="" +# A bot can fail to connect within the booter's 30 s window under load +# ("Only 1 of 2 bots started"), which is a FALSE NEGATIVE, not a real battle. +# Each job is retried until the capture shows real rows AND a subject counter +# line, so the analyzer never sees a phantom 0-0 run. +MAX_ATTEMPTS="${GAUNTLET_MAX_ATTEMPTS:-4}" + +usage() { + sed -n '2,45p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//' + exit "${1:-0}" +} + +while [[ $# -gt 0 ]]; do + case "$1" in + --arms) ARMS_FILE="$2"; shift 2;; + --opponents) OPP_FILE="$2"; shift 2;; + --runs) RUNS="$2"; shift 2;; + --rounds) ROUNDS="$2"; shift 2;; + --conc) CONC="$2"; shift 2;; + --outdir) OUTDIR="$2"; shift 2;; + -h|--help) usage 0;; + *) echo "unknown argument: $1" >&2; usage 1;; + esac +done + +[[ -n "$ARMS_FILE" ]] || { echo "ERROR: --arms is required" >&2; usage 1; } +[[ -n "$OPP_FILE" ]] || { echo "ERROR: --opponents is required" >&2; usage 1; } +[[ -f "$ARMS_FILE" ]] || { echo "ERROR: arm file not found: $ARMS_FILE" >&2; exit 1; } +[[ -f "$OPP_FILE" ]] || { echo "ERROR: opponent file not found: $OPP_FILE" >&2; exit 1; } +[[ -n "$OUTDIR" ]] || { echo "ERROR: --outdir is required" >&2; exit 1; } +[[ "$OUTDIR" != "/" && "$OUTDIR" != "" ]] || { echo "ERROR: refusing outdir '$OUTDIR'" >&2; exit 1; } +OUTDIR="$(mkdir -p "$OUTDIR" && cd "$OUTDIR" && pwd)" + +# ── prerequisites ──────────────────────────────────────────────────────────── +ROBOCODE_JAR="${ROBOCODE_JAR:-/tmp/robocode/install/libs/robocode.jar}" +RUNNER_JAR="${TR_RUNNER_JAR:-/home/davide/Projects/tank-royale/runner/examples/lib/robocode-tankroyale-runner.jar}" +BOT_API_JAR="${TR_BOT_API_JAR:-$HOME/Downloads/sample-bots-java-1.0.2/lib/robocode-tankroyale-bot-api-1.0.2.jar}" + +missing=0 +for f in "$ROBOCODE_JAR" "$RUNNER_JAR" "$BOT_API_JAR"; do + if [[ ! -f "$f" ]]; then echo "ERROR: missing prerequisite: $f" >&2; missing=1; fi +done +command -v nim >/dev/null 2>&1 || { echo "ERROR: nim not on PATH" >&2; missing=1; } +command -v git >/dev/null 2>&1 || { echo "ERROR: git not on PATH" >&2; missing=1; } +[[ -f "$RUN_BRIDGE" ]] || { echo "ERROR: missing $RUN_BRIDGE" >&2; missing=1; } +(( missing == 0 )) || { echo "ERROR: prerequisites missing — not starting a broken session" >&2; exit 1; } + +# ── parse the arm file into parallel arrays ────────────────────────────────── +trim() { local s="$1"; s="${s#"${s%%[![:space:]]*}"}"; s="${s%"${s##*[![:space:]]}"}"; printf '%s' "$s"; } + +ARM_NAMES=(); ARM_ENVS=(); ARM_LABELS=() +while IFS= read -r line || [[ -n "$line" ]]; do + [[ -z "${line//[[:space:]]/}" ]] && continue + [[ "$line" =~ ^[[:space:]]*# ]] && continue + name="$(trim "${line%%|*}")" + rest="${line#*|}" + if [[ "$line" == *"|"* ]]; then + envspec="$(trim "${rest%%|*}")" + label="$(trim "${rest#*|}")" + else + envspec=""; label="" + fi + [[ -n "$name" ]] || { echo "ERROR: arm with empty name in $ARMS_FILE" >&2; exit 1; } + ARM_NAMES+=("$name"); ARM_ENVS+=("$envspec"); ARM_LABELS+=("$label") +done < "$ARMS_FILE" +(( ${#ARM_NAMES[@]} > 0 )) || { echo "ERROR: no arms parsed from $ARMS_FILE" >&2; exit 1; } +dupes="$(printf '%s\n' "${ARM_NAMES[@]}" | sort | uniq -d)" +[[ -z "$dupes" ]] || { echo "ERROR: duplicate arm name(s): $dupes" >&2; exit 1; } + +# ── parse the opponent file ────────────────────────────────────────────────── +OPP_DIRS=(); OPP_NAMES=(); OPP_STYLES=() +while IFS= read -r line || [[ -n "$line" ]]; do + [[ -z "${line//[[:space:]]/}" ]] && continue + [[ "$line" =~ ^[[:space:]]*# ]] && continue + dir="$(trim "${line%%|*}")" + style="" + if [[ "$line" == *"|"* ]]; then style="$(trim "${line#*|}")"; fi + [[ "$dir" == */* ]] || dir="/tmp/tr_bots/$dir" + [[ -d "$dir" ]] || { echo "ERROR: opponent bot dir not found: $dir" >&2; exit 1; } + nm="$(basename "$dir")" + OPP_DIRS+=("$dir"); OPP_NAMES+=("$nm"); OPP_STYLES+=("$style") +done < "$OPP_FILE" +(( ${#OPP_NAMES[@]} > 0 )) || { echo "ERROR: no opponents parsed from $OPP_FILE" >&2; exit 1; } + +# ── fresh session dir ──────────────────────────────────────────────────────── +rm -rf "$OUTDIR" +mkdir -p "$OUTDIR" +WORK="$OUTDIR/.work"; mkdir -p "$WORK" + +# ── build ONE frozen bot from HEAD ─────────────────────────────────────────── +COMMIT="$(git -C "$REPO" rev-parse HEAD)" +FROZEN_DIR="$OUTDIR/frozen/ModularBot" +FROZEN_BIN="$FROZEN_DIR/ModularBot_bin" +FROZEN_JSON="$FROZEN_DIR/ModularBot.json" +mkdir -p "$FROZEN_DIR" +BUILDDIR="$WORK/head" +mkdir -p "$BUILDDIR" +echo "[gauntlet] exporting HEAD ($COMMIT) -> $BUILDDIR" +git -C "$REPO" archive HEAD | tar -x -C "$BUILDDIR" + +echo "[gauntlet] building frozen ModularBot (nim c -d:release)…" +( cd "$BUILDDIR/ModularBot_garage" && \ + nim c -d:release --nimcache:"$OUTDIR/nimcache" --out:"$FROZEN_BIN" src/ModularBot.nim ) +[[ -x "$FROZEN_BIN" ]] || { echo "ERROR: frozen build failed" >&2; exit 1; } +BINSHA="$(sha256sum "$FROZEN_BIN" | awk '{print $1}')" +echo "[gauntlet] frozen binary sha256=$BINSHA" + +cat > "$FROZEN_JSON" <<'JSON' +{ + "name": "ModularBot", + "version": "0.1.0", + "authors": ["Davide Cappellini"], + "description": "frozen A/B build of ModularBot (gauntlet_run.sh)", + "homepage": "", + "countryCodes": ["IT"], + "gameTypes": ["classic", "1v1"], + "platform": "Nim", + "programmingLang": "Nim" +} +JSON + +# ── session.json ───────────────────────────────────────────────────────────── +json_esc() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; } +{ + printf '{\n' + printf ' "commit": "%s",\n' "$COMMIT" + printf ' "binary_sha256": "%s",\n' "$BINSHA" + printf ' "binary": "frozen/ModularBot/ModularBot_bin",\n' + printf ' "rounds": %d,\n' "$ROUNDS" + printf ' "runs": %d,\n' "$RUNS" + printf ' "conc": %d,\n' "$CONC" + printf ' "timestamp": "%s",\n' "$(date -Is)" + printf ' "outdir": "%s",\n' "$(json_esc "$OUTDIR")" + printf ' "arms": [\n' + for i in "${!ARM_NAMES[@]}"; do + printf ' {"name": "%s", "env": "%s", "label": "%s"}' \ + "$(json_esc "${ARM_NAMES[$i]}")" "$(json_esc "${ARM_ENVS[$i]}")" "$(json_esc "${ARM_LABELS[$i]}")" + (( i + 1 < ${#ARM_NAMES[@]} )) && printf ',' || true + printf '\n' + done + printf ' ],\n' + printf ' "opponents": [\n' + for i in "${!OPP_NAMES[@]}"; do + printf ' {"name": "%s", "dir": "%s", "style": "%s"}' \ + "$(json_esc "${OPP_NAMES[$i]}")" "$(json_esc "${OPP_DIRS[$i]}")" "$(json_esc "${OPP_STYLES[$i]}")" + (( i + 1 < ${#OPP_NAMES[@]} )) && printf ',' || true + printf '\n' + done + printf ' ]\n}\n' +} > "$OUTDIR/session.json" + +# ── one job (run in its own process group) ─────────────────────────────────── +# exported so the detached `bash -c` subshell can call it; config via env +export REPO RUN_BRIDGE ROUNDS OUTDIR WORK FROZEN_BIN FROZEN_JSON + +gauntlet_run_one() { + local opponent_dir="$1" opponent="$2" arm="$3" run="$4" envspec="$5" + local dir="$OUTDIR/$opponent/$arm" + local work="$WORK/$opponent/$arm/run$run" + local botdir="$work/bots/ModularBot" + mkdir -p "$dir" "$botdir" "$work/data" + + cp "$FROZEN_JSON" "$botdir/ModularBot.json" + cat > "$botdir/ModularBot.sh" < "$dir/run$run.bot.stdout.log" 2> "$dir/run$run.bot.stderr.log" +SH + chmod +x "$botdir/ModularBot.sh" + ln -sf "$FROZEN_BIN" "$botdir/ModularBot_bin" + + local attempt=0 + while :; do + attempt=$((attempt + 1)) + rm -f "$dir/run$run.jsonl" "$dir/run$run.jsonl.rounds.json" \ + "$dir/run$run.jsonl.results.json" "$dir/run$run.events.jsonl" \ + "$dir/run$run.status" "$dir/run$run.bot.stdout.log" "$dir/run$run.bot.stderr.log" + + local rc=0 + # shellcheck disable=SC2086 # envspec is intentionally word-split + env $envspec \ + SHIM_BOTDIR="$botdir" \ + SHIM_BOT_NAME="ModularBot" \ + SHIM_DATA="$work/data" \ + TR_EVENTS_OUT="$dir/run$run.events.jsonl" \ + timeout 900 "$RUN_BRIDGE" "$opponent_dir" "$ROUNDS" "$dir/run$run.jsonl" \ + > "$dir/run$run.battle.log" 2>&1 || rc=$? + echo "$rc" > "$dir/run$run.status" + + # liveness: a real battle has captured rows and a subject counter line + if grep -q "subject event counts" "$dir/run$run.battle.log" 2>/dev/null \ + && grep -Eq "rows=[1-9]" "$dir/run$run.battle.log" 2>/dev/null; then + break + fi + if (( attempt >= MAX_ATTEMPTS )); then + echo "[gauntlet] WARNING: ${opponent}/${arm} run${run} never started after ${attempt} attempts" >&2 + break + fi + sleep $((attempt * 2)) + done + echo "$attempt" > "$dir/run$run.attempts" + return 0 +} +export -f gauntlet_run_one + +# ── cleanup: kill this session's children ──────────────────────────────────── +JOB_PIDS=() +cleanup() { + local rc=$? + trap - EXIT INT TERM + for p in "${JOB_PIDS[@]:-}"; do + [[ -n "$p" ]] || continue + kill -TERM -- "-$p" 2>/dev/null || kill -TERM "$p" 2>/dev/null || true + done + sleep 0.5 + for p in "${JOB_PIDS[@]:-}"; do + [[ -n "$p" ]] || continue + kill -KILL -- "-$p" 2>/dev/null || true + done + pkill -f "[r]un_bridge_battle.sh.*$OUTDIR" 2>/dev/null || true + pkill -f "[r]obocode_shim.TrBattleCapture.*$OUTDIR" 2>/dev/null || true + exit "$rc" +} +trap cleanup EXIT INT TERM + +# ── launch the jobs, K at a time ───────────────────────────────────────────── +NJOBS=$(( ${#OPP_NAMES[@]} * ${#ARM_NAMES[@]} * RUNS )) +echo "[gauntlet] session: ${#OPP_NAMES[@]} opponents × ${#ARM_NAMES[@]} arms × $RUNS runs = $NJOBS battles, rounds=$ROUNDS, conc=$CONC" +echo "[gauntlet] outdir: $OUTDIR" +RUNNING=0 +for oi in "${!OPP_NAMES[@]}"; do + for ai in "${!ARM_NAMES[@]}"; do + for (( r=1; r<=RUNS; r++ )); do + setsid bash -c 'gauntlet_run_one "$1" "$2" "$3" "$4" "$5"' _ \ + "${OPP_DIRS[$oi]}" "${OPP_NAMES[$oi]}" "${ARM_NAMES[$ai]}" "$r" "${ARM_ENVS[$ai]}" & + JOB_PIDS+=("$!") + RUNNING=$((RUNNING + 1)) + if (( RUNNING >= CONC )); then + wait -n || true + RUNNING=$((RUNNING - 1)) + fi + done + done +done +wait || true + +JOB_PIDS=() + +# ── aggregate status ───────────────────────────────────────────────────────── +FAILED=0 +for oi in "${!OPP_NAMES[@]}"; do + for ai in "${!ARM_NAMES[@]}"; do + for (( r=1; r<=RUNS; r++ )); do + st="$(cat "$OUTDIR/${OPP_NAMES[$oi]}/${ARM_NAMES[$ai]}/run$r.status" 2>/dev/null || echo 999)" + if [[ "$st" != "0" ]]; then + echo "[gauntlet] FAIL ${OPP_NAMES[$oi]}/${ARM_NAMES[$ai]} run$r (rc=$st)" >&2 + FAILED=$((FAILED + 1)) + fi + done + done +done + +echo "[gauntlet] done: $(( NJOBS - FAILED )) ok, $FAILED failed" +echo "[gauntlet] analyze with: python3 $REPO/tools/ab/gauntlet_analyze.py $OUTDIR" +(( FAILED == 0 )) || exit 1 diff --git a/tools/ab/opponents_gauntlet.txt b/tools/ab/opponents_gauntlet.txt new file mode 100644 index 0000000..7fe64b2 --- /dev/null +++ b/tools/ab/opponents_gauntlet.txt @@ -0,0 +1,42 @@ +# Opponents for the BitBrain-vs-Pattern generalization gauntlet. +# Every entry is a validated legacy-champ bot dir from tools/robocode_shim/robots.json. +# The trailing label is the MOVEMENT style (INFERRED from the bot's name/docs in +# robots.json, not decompiled): regular = periodic/predictable/wall-following/ +# ramming, dodger = wave-surfing/adaptive evasion, other = mixed/unknown. +# +# regular: the owner's sub-claim ("killer vs spinners, wall-followers") lives here +/tmp/tr_bots/WallAvoider | regular +/tmp/tr_bots/HawkOnFire | regular +/tmp/tr_bots/DiamondStealer| regular +/tmp/tr_bots/BlitzBat | regular +/tmp/tr_bots/PatternRobot | regular +/tmp/tr_bots/FloodMini | regular +/tmp/tr_bots/Aurora | regular +# dodger: wave surfers / adaptive evasion +/tmp/tr_bots/Diamond | dodger +/tmp/tr_bots/Dookious | dodger +/tmp/tr_bots/Lukious | dodger +/tmp/tr_bots/GresSuffurd | dodger +/tmp/tr_bots/KurtWaveSurfer| dodger +/tmp/tr_bots/CassiusClay | dodger +/tmp/tr_bots/WaveSurferGF | dodger +/tmp/tr_bots/WaveSurferPG | dodger +/tmp/tr_bots/Jen | dodger +/tmp/tr_bots/Cigaret | dodger +/tmp/tr_bots/CigaretBH | dodger +/tmp/tr_bots/Komarious | dodger +/tmp/tr_bots/LionWWSVMvoid | dodger +/tmp/tr_bots/RougeDC | dodger +/tmp/tr_bots/Aristocles | dodger +# other: mixed / unknown movement +/tmp/tr_bots/DrussGT | other +/tmp/tr_bots/Shadow | other +/tmp/tr_bots/Phoenix | other +/tmp/tr_bots/Coriantumr | other +/tmp/tr_bots/BrokenSword | other +/tmp/tr_bots/TripHammer | other +/tmp/tr_bots/RetroGirl | other +/tmp/tr_bots/DiamondHawk | other +/tmp/tr_bots/Ascendant | other +/tmp/tr_bots/YersiniaPestis| other +/tmp/tr_bots/LightningBug | other