From 2a98aba91b0ec023f580cd2485cafba2404d275d Mon Sep 17 00:00:00 2001 From: Davide Cappellini Date: Sat, 26 Sep 2026 04:01:51 +0200 Subject: [PATCH] gun j123 Task A+B prereg: expose Pattern's TR_PATTERN_LEN/TR_PATTERN_DEPTH (default parity) and pre-register the 6-arm match-parameter sweep --- ModularBot_garage/src/env_report.nim | 11 ++- common_libs/guns/pattern_matcher.nim | 71 +++++++++++++++--- docs/gun_campaign.md | 106 +++++++++++++++++++++++++++ tools/ab/arms_gun_b3.txt | 53 ++++++++++++++ 4 files changed, 231 insertions(+), 10 deletions(-) create mode 100644 tools/ab/arms_gun_b3.txt diff --git a/ModularBot_garage/src/env_report.nim b/ModularBot_garage/src/env_report.nim index 4cd1873..9c4eaa2 100644 --- a/ModularBot_garage/src/env_report.nim +++ b/ModularBot_garage/src/env_report.nim @@ -365,9 +365,17 @@ proc printEffectiveValues(ctx: EnvReportContext) = emit("TR_BITBRAIN_RESET_ON_TARGET", onOff(ctx.bitbrain.resetOnTarget), sourceOf(BB_RESET_ON_TARGET_ENV)) - # ── Pattern radial knobs ────────────────────────────────────────────────── + # ── Pattern match-shape + radial knobs ──────────────────────────────────── # These are resolved lazily inside `predict` (which has not run at boot), so # unless something already forced them we report the raw env value and say so. + if ctx.patternMatcher.paramConfigured: + emit("TR_PATTERN_LEN", $ctx.patternMatcher.patternLen, sourceOf("TR_PATTERN_LEN")) + emit("TR_PATTERN_DEPTH", $ctx.patternMatcher.histDepth, sourceOf("TR_PATTERN_DEPTH")) + else: + echo "[env] TR_PATTERN_LEN = ", + getEnv("TR_PATTERN_LEN", $PatternLen), " (raw - not resolved here)" + echo "[env] TR_PATTERN_DEPTH = ", + getEnv("TR_PATTERN_DEPTH", $PatternDepth), " (raw - not resolved here)" if ctx.patternMatcher.radConfigured: emit("TR_PATTERN_RAD_OFFSET", $ctx.patternMatcher.radOffset, sourceOf("TR_PATTERN_RAD_OFFSET")) emit("TR_PATTERN_RAD_SCALE", $ctx.patternMatcher.radScale, sourceOf("TR_PATTERN_RAD_SCALE")) @@ -428,6 +436,7 @@ proc knownEnvNames*(): seq[string] = PowerRefEnvVar, PowerEnergyHiEnvVar, PowerEnergyLoEnvVar, PowerEnergyMinEnvVar, PowerEnergyMaxEnvVar, PowerFinishKillEnvVar, PatternRadOffsetEnvVar, PatternRadScaleEnvVar, + PatternLenEnvVar, PatternDepthEnvVar, TMH_SHIFT_ENV, TMH_BIG_MULT_ENV, TMH_LOG_ENV, TMH_RESET_ON_TARGET_ENV, TMH_WINDOW_ENV, TMH_RESET_DROP_ENV, TMH_NSTATES_ENV, TMH_ACCURVE_ENV, TMH_RETRAIN_EVERY_ENV, TMH_EPOCHS_ENV, diff --git a/common_libs/guns/pattern_matcher.nim b/common_libs/guns/pattern_matcher.nim index c1dd56c..7119833 100644 --- a/common_libs/guns/pattern_matcher.nim +++ b/common_libs/guns/pattern_matcher.nim @@ -18,9 +18,20 @@ import gun_harness/gun_interface const HistorySize* = 500 - PatternLen* = 10 # ticks used as search key; ponytail: fixed, expose if tuning needed + PatternLen* = 10 # ticks used as search key; shipped default of TR_PATTERN_LEN + ## Shipped defaults of the two MATCH-SHAPE runtime knobs (Task A, gun j123). + ## `TR_PATTERN_LEN` is the length of the movement segment compared (the + ## search key); `TR_PATTERN_DEPTH` is how many recent history entries the + ## matcher is allowed to scan. Both default to the exact pre-knob constants, + ## so an unset environment reproduces today's gun byte-for-byte. + PatternDepth* = HistorySize # max history entries scanned; default = full buffer PatternRadOffsetEnvVar* = "TR_PATTERN_RAD_OFFSET" ## px; negative = aim short PatternRadScaleEnvVar* = "TR_PATTERN_RAD_SCALE" ## multiplier on aim distance + PatternLenEnvVar* = "TR_PATTERN_LEN" ## search-key length (ticks) + ## History/replay depth: the search never looks further back than this many + ## entries. The buffer itself is a fixed `array[HistorySize]`, so the useful + ## range is 1..HistorySize; the ceiling cannot be raised at runtime. + PatternDepthEnvVar* = "TR_PATTERN_DEPTH" proc patternEnvFloat(name: string, default: float): float = ## Read an env knob like every other runtime switch in the harness: unset or @@ -30,6 +41,18 @@ proc patternEnvFloat(name: string, default: float): float = try: parseFloat(v.strip()) except ValueError: default +proc patternEnvInt(name: string, default: int): int = + ## Integer twin of `patternEnvFloat`: unset or unparsable -> shipped default. + let v = getEnv(name, "") + if v.len == 0: return default + try: parseInt(v.strip()) + except ValueError: default + +proc clampInt(v, lo, hi: int): int {.inline.} = + result = v + if result < lo: result = lo + if result > hi: result = hi + type MoveTick = object velocity: float ## signed speed (px/tick) @@ -60,6 +83,10 @@ type pathY: array[HistorySize + 1, float] pathHeading: array[HistorySize + 1, float] ## radians after s steps pathSpeed: array[HistorySize + 1, float] ## speed after s steps + # ── match-shape knobs (defaults leave the prediction byte-identical) ── + patternLen*: int ## search-key length actually used (default PatternLen) + histDepth*: int ## history entries actually scanned (default full) + paramConfigured*: bool ## true once the env/setter has populated the two above # ── radial offset knob (defaults leave the prediction byte-identical) ── radScale*: float ## multiplier on the predicted aim distance radOffset*: float ## px added to the predicted aim distance @@ -95,20 +122,30 @@ proc findBestMatch(g: var PatternMatcherGun): int = ## Speed-independent history search. Returns the start index of the best ## matching pattern, or -1 when there is not enough history. Stores the best ## match cost in `g.lastMatchScore` for the confidence readout. + ## + ## `g.patternLen` (TR_PATTERN_LEN) is the key length; `g.histDepth` + ## (TR_PATTERN_DEPTH) bounds how far back the scan may reach. With the shipped + ## defaults (10, HistorySize) this is the exact pre-knob loop. + let plen = g.patternLen g.lastMatchScore = Inf - if g.count < PatternLen * 2: + if g.count < plen * 2: return -1 - # key = last PatternLen entries - let keyStart = g.count - PatternLen + # key = last plen entries + let keyStart = g.count - plen - # scan backwards for best match (exclude the key itself) + # scan backwards for best match (exclude the key itself), but never further + # back than `histDepth` entries; when count <= histDepth this floor is 0 and + # the loop is identical to the original. var bestScore = Inf var bestMatch = -1 - let scanEnd = g.count - PatternLen - 1 # last valid match start - for i in countdown(scanEnd, 0): + let scanEnd = g.count - plen - 1 # last valid match start + let scanFloor = max(0, g.count - g.histDepth) + if scanEnd < scanFloor: + return -1 + for i in countdown(scanEnd, scanFloor): var score = 0.0 - for k in 0 ..< PatternLen: + for k in 0 ..< plen: let a = g.readAt(keyStart + k) let b = g.readAt(i + k) let dv = a.velocity - b.velocity @@ -123,7 +160,7 @@ proc findBestMatch(g: var PatternMatcherGun): int = proc buildPath(g: var PatternMatcherGun, state: WorldState, bestMatch: int) = ## Precompute the matched pattern replayed forward from the current state. ## Only depends on observed movement, so it is valid for every power bin. - g.playStart = bestMatch + PatternLen + g.playStart = bestMatch + g.patternLen g.playAvail = g.count - 1 - g.playStart # ticks we can replay g.pathX[0] = state.enemyX g.pathY[0] = state.enemyY @@ -173,6 +210,21 @@ proc ensureRadialConfig(g: var PatternMatcherGun) {.inline.} = g.radOffset = patternEnvFloat(PatternRadOffsetEnvVar, 0.0) g.radConfigured = true +proc ensureParamConfig(g: var PatternMatcherGun) {.inline.} = + ## Lazily read the match-shape env knobs on first use, mirroring the radial + ## knobs: the shipped binary reads the env once per gun instance. + if g.paramConfigured: return + g.patternLen = clampInt(patternEnvInt(PatternLenEnvVar, PatternLen), 1, HistorySize) + g.histDepth = clampInt(patternEnvInt(PatternDepthEnvVar, PatternDepth), 1, HistorySize) + g.paramConfigured = true + +proc setMatchParams*(g: var PatternMatcherGun, patternLen, histDepth: int) = + ## Explicit per-gun override used by offline sweeps/tests. Writes the same + ## fields the env path writes, so the measured code path is identical. + g.patternLen = clampInt(patternLen, 1, HistorySize) + g.histDepth = clampInt(histDepth, 1, HistorySize) + g.paramConfigured = true + proc setRadialCorrection*(g: var PatternMatcherGun, scale, offsetPx: float) = ## Explicit per-gun override used by the offline sweep. Writes the same fields ## the env path writes, so the measured code path is identical. @@ -220,6 +272,7 @@ proc predict*(g: var PatternMatcherGun, state: WorldState, g.cacheValid = true g.cacheTick = state.tick + g.ensureParamConfig() g.bestMatch = g.findBestMatch() if g.bestMatch >= 0: g.buildPath(state, g.bestMatch) diff --git a/docs/gun_campaign.md b/docs/gun_campaign.md index 43428cc..fcc62fa 100644 --- a/docs/gun_campaign.md +++ b/docs/gun_campaign.md @@ -464,3 +464,109 @@ different reference is a free pairwise comparison with no battles. |---|---|---:|---|---| | `/tmp/ab/j121_g1` | `1d8143a` | 270 (0 failed, 0 never started, 0 excluded) | pattern, bitbrain, tmhorizon, knn, rack_pk, rack_pt | **nothing beats the shipped `pattern`**; all arms not distinguishable except `knn` = WORSE; `onlyPattern` CONFIRMED | | `/tmp/ab/j121_g2` | `c343c00` | 297 (0 failed, 0 never started, 0 excluded) | pattern, bitbrain, tmhorizon | **the Batch-1 damage hint does not survive**: `bitbrain` +1.8 dmg/run p=0.49 (wash); `tmhorizon` −10.4 dmg/run p=0.035 (WORSE); wins flat | + +--- + +# Phase 2: tuning the incumbent + +**Owner's mandate (phase 2):** the phase-1 campaign closed the "other gun / other +rack" design space (nothing beats the shipped `Pattern`; the correctors are a +wash or a loss). What remains OPEN is the **incumbent's own tuning**: Pattern's +match-length / history parameters had NEVER been swept. Phase 2 asks the direct +question: + +> **Can the shipped `Pattern` be improved by tuning its own match parameters — +> and if so, by how much on damage/run and round wins?** + +The lead-amplitude axis is already known dead (`docs/bitbrain_campaign.md`), and +the radial knobs are known non-winners (job j99: `bmPath` structural no-op; live ++0.28 pp p=0.62 for scale 0.98, −0.42 pp p=0.46 for offset −20). Those are +therefore **controls** here, not candidates. + +## Task A — exposing Pattern's match-shape parameters (what and why) + +**MEASURED (code read).** `common_libs/guns/pattern_matcher.nim` had exactly two +tunable knobs, both RADIAL (`TR_PATTERN_RAD_SCALE`, `TR_PATTERN_RAD_OFFSET`), and +neither can change the lead bearing (job j99 proved the bearing is untouched), so +neither was ever the "lead information" axis. The parameters that actually +control **how the pattern is matched** were compile-time constants: + +| parameter | code | what it controls | exposed as | +|---|---|---|---| +| match-key length | `PatternLen = 10` | the length of the movement segment compared (the search key); also how far after the match the replay starts | `TR_PATTERN_LEN` (int, default 10) | +| search depth | implicit `HistorySize = 500` | how far back the best-match scan may reach (`scanEnd` was always `count−PatternLen−1`) | `TR_PATTERN_DEPTH` (int, default 500 = full buffer) | +| similarity/search radius | **does not exist** | the search always takes the single lowest-cost match; there is no acceptance threshold or radius to expose | **not exposed — nothing to expose** | +| history buffer capacity | `HistorySize = 500` | the fixed `array[HistorySize]` backing store | runtime depth limit only; the buffer ceiling cannot be raised at runtime (see below) | + +**Why env and not `-d:`.** The phase-1 instrument (`tools/ab/tournament_run.sh`) +builds ONE frozen binary from `git archive HEAD` and every arm differs only by its +env dict. A `{.intdefine.}` knob would need one binary per arm, which the +instrument forbids. Both new knobs are therefore resolved lazily from the +environment on the first `predict`, exactly like the radial knobs, and default to +the pre-knob constants. `setMatchParams(patternLen, histDepth)` is the explicit +offline/unit-test twin that writes the same fields. + +**What could NOT be exposed cheaply (MEASURED).** `HistorySize` sizes four fixed +`array[HistorySize(+1)]` fields in the gun object. Raising it above 500 at +runtime is impossible without a heap buffer; the runtime `TR_PATTERN_DEPTH` knob +therefore *lowers* the effective search depth within the existing 500-entry +buffer. A depth above 500 is clamped to 500, and a non-positive or unparsable +value falls back to the shipped default. `PatternLen` is clamped to +`1..HistorySize`. + +## Task A — default parity (MEASURED, byte-for-byte) + +* **Baseline-vs-new parity dump.** A throwaway harness replayed 800 ticks of + `tools/fixtures/drussgt_vs_crazy.jsonl` through one `PatternMatcherGun` at all + four power bins (3200 predictions) and printed every point at full precision. + Compiled once against a `HEAD` worktree (`/tmp/j123_base`, before the change) + and once against the modified tree: **the two dumps are byte-identical** + (`diff -q` clean). The shipped default path is unchanged. +* **The knobs are live and self-falling-back.** `TR_PATTERN_LEN=6/16` and + `TR_PATTERN_DEPTH=100/20` each change the dump; `TR_PATTERN_LEN=banana` + reproduces the default dump exactly. +* **Existing guards pass** (no new guard tests, per the "cut ceremony" rule): + `common_libs/tests/test_pattern_radial_offset.nim` (6 checks) and + `common_libs/tests/test_gun_harness.nim` (all checks) pass; the env-report + guard `test_env_report.nim` passes with `TR_PATTERN_LEN` / `TR_PATTERN_DEPTH` + registered in `knownEnvNames()` and the effective-values report. +* **Clean-archive compile** is exercised by `tournament_run.sh` itself, which + builds the frozen binary from `git archive HEAD`. + +## Task B — Batch 1 (pre-registered BEFORE any battle) + +**Design.** One frozen binary, six env-only arms, the FROZEN 15-opponent panel +`tools/ab/panel_movement.txt`, **3 runs × 3 rounds per (opponent, arm) = 270 +battles**, `--conc 6`, `--wait-arena`. Arm file: `tools/ab/arms_gun_b3.txt`. +Reference: `pattern`. Movement pinned `TR_MOVEMENT=strafe` in every arm. + +| # | arm | env over the pin | what it isolates | +|---|---|---|---| +| 1 | `pattern` | (none) | the arm to beat (shipped `onlyPattern` rack) | +| 2 | `len6` | `TR_PATTERN_LEN=6` | shorter movement segment compared | +| 3 | `len16` | `TR_PATTERN_LEN=16` | longer movement segment compared | +| 4 | `depth100` | `TR_PATTERN_DEPTH=100` | shallower history / match search | +| 5 | `rad_offset` | `TR_PATTERN_RAD_OFFSET=-20` | control: aim 20 px short | +| 6 | `rad_scale` | `TR_PATTERN_RAD_SCALE=0.95` | control: scale the aim distance | + +**Pre-registered prediction (written BEFORE the battles finished):** +1. **No arm beats `pattern` on both primaries.** The incumbent is already tuned — + the phase-2 null. The point estimates sit inside the MDE. +2. `len16` is **WORSE or flat** on damage: a 16-tick key matches rarely in a + ~500-tick buffer, so the gun falls back to the linear forecast (the weaker + base) more often; wins flat. +3. `len6` is **flat** on damage (more matches but a noisier replay) and flat on + wins; possibly a sub-MDE wobble in either direction. +4. `depth100` is **not distinguishable** from `pattern` — the full buffer already + contains the useful candidates; a sub-MDE damage wobble is possible. +5. The two radial controls are **not distinguishable** and, per job j99, cannot + change the real aim bearing; any damage delta is the fire-gate channel only. +6. If any surprise exists, it is a **damage-only wobble without wins** — the + `ring`-mover trap — and will not be read as a win. + +**Pre-registered Batch-2 trigger (task rule).** A Batch-1 arm is promoted to a +higher-power Batch 2 **only if** it beats the reference `pattern` by the +campaign rule 2 (one primary up at cross-opponent sign-test p<0.05 while the +other does not go down) **AND** its damage CI excludes 0 **AND** the sign-flip +permutation test gives p<0.05. Otherwise Batch 1 is the answer. + diff --git a/tools/ab/arms_gun_b3.txt b/tools/ab/arms_gun_b3.txt new file mode 100644 index 0000000..6274748 --- /dev/null +++ b/tools/ab/arms_gun_b3.txt @@ -0,0 +1,53 @@ +# ───────────────────────────────────────────────────────────────────────────── +# arms_gun_b3.txt — PHASE 2 BATCH 1 of the GUN campaign: does tuning the +# incumbent's OWN match-shape parameters beat the shipped `Pattern`? +# +# Format: name | ENV=value ENV=value | label +# +# ONE frozen binary (tournament_run.sh builds it from `git archive HEAD`); every +# arm below differs ONLY by its env dict. No per-arm rebuild. +# +# MOVEMENT IS PINNED IN EVERY ARM (`TR_MOVEMENT=strafe`). Reason (hard rule): +# a parallel job (j124) may be fighting, and the shipped movement default may +# move; pinning the engine in EVERY arm removes movement as a confound, makes all +# arms share one movement, and keeps the comparison a pure GUN comparison. It +# also satisfies the analyzer liveness rule (every declared token must appear in +# the bot's own raw-env report, and an undeclared TR_MOVEMENT is fatal). +# +# Task A (this job) exposed Pattern's own match-shape parameters as runtime env +# knobs, defaults byte-identical to the pre-knob gun (proven by the parity dump +# in the ledger): +# TR_PATTERN_LEN — length of the movement segment compared (search key). +# Shipped default 10. +# TR_PATTERN_DEPTH — how many recent history entries the matcher may scan. +# Shipped default 500 (= the whole HistorySize buffer). +# The two pre-existing radial knobs are kept as cheap CONTROLS. +# +# Prior data this batch is built on (taken as given; NOT re-derived): +# * `docs/gun_campaign.md` Batches 1-2: nothing beats the shipped `pattern` +# across 15 and 33 opponents; `onlyPattern` is a measured optimum. +# * `docs/bitbrain_campaign.md`: lead AMPLITUDE is a dead axis; lead +# INFORMATION is the open one. +# * `common_libs/tests/pattern_radial_results.md` (job j99): the radial knobs +# move the aim point along an UNCHANGED bearing, so under the shipped +# `bmPath` metric they are a structural no-op; live they were +0.28 pp +# (p=0.62) / −0.42 pp (p=0.46). +# ───────────────────────────────────────────────────────────────────────────── + +# 1. REFERENCE. The shipped `onlyPattern` rack, movement pinned, no shape env. +pattern | TR_MOVEMENT=strafe | shipped onlyPattern rack (REFERENCE) + +# 2. Shorter match key: more candidate matches, noisier replay. +len6 | TR_MOVEMENT=strafe TR_PATTERN_LEN=6 | match key = 6 ticks + +# 3. Longer match key: fewer, more specific matches; more linear fallback. +len16 | TR_MOVEMENT=strafe TR_PATTERN_LEN=16 | match key = 16 ticks + +# 4. Shallow history: the search never looks further back than 100 entries. +depth100 | TR_MOVEMENT=strafe TR_PATTERN_DEPTH=100 | history search = 100 ticks + +# 5. CONTROL: aim 20 px short (the radial-knob header's intended use). +rad_offset | TR_MOVEMENT=strafe TR_PATTERN_RAD_OFFSET=-20 | radial control: aim short 20 px + +# 6. CONTROL: scale the aim distance by 0.95. +rad_scale | TR_MOVEMENT=strafe TR_PATTERN_RAD_SCALE=0.95 | radial control: scale 0.95