gun j123 Task A+B prereg: expose Pattern's TR_PATTERN_LEN/TR_PATTERN_DEPTH (default parity) and pre-register the 6-arm match-parameter sweep

This commit is contained in:
2026-09-26 04:01:51 +02:00
parent 467e07a6a4
commit 2a98aba91b
4 changed files with 231 additions and 10 deletions
+10 -1
View File
@@ -365,9 +365,17 @@ proc printEffectiveValues(ctx: EnvReportContext) =
emit("TR_BITBRAIN_RESET_ON_TARGET", onOff(ctx.bitbrain.resetOnTarget),
sourceOf(BB_RESET_ON_TARGET_ENV))
# ── Pattern radial knobs ──────────────────────────────────────────────────
# ── Pattern match-shape + radial knobs ────────────────────────────────────
# These are resolved lazily inside `predict` (which has not run at boot), so
# unless something already forced them we report the raw env value and say so.
if ctx.patternMatcher.paramConfigured:
emit("TR_PATTERN_LEN", $ctx.patternMatcher.patternLen, sourceOf("TR_PATTERN_LEN"))
emit("TR_PATTERN_DEPTH", $ctx.patternMatcher.histDepth, sourceOf("TR_PATTERN_DEPTH"))
else:
echo "[env] TR_PATTERN_LEN = ",
getEnv("TR_PATTERN_LEN", $PatternLen), " (raw - not resolved here)"
echo "[env] TR_PATTERN_DEPTH = ",
getEnv("TR_PATTERN_DEPTH", $PatternDepth), " (raw - not resolved here)"
if ctx.patternMatcher.radConfigured:
emit("TR_PATTERN_RAD_OFFSET", $ctx.patternMatcher.radOffset, sourceOf("TR_PATTERN_RAD_OFFSET"))
emit("TR_PATTERN_RAD_SCALE", $ctx.patternMatcher.radScale, sourceOf("TR_PATTERN_RAD_SCALE"))
@@ -428,6 +436,7 @@ proc knownEnvNames*(): seq[string] =
PowerRefEnvVar, PowerEnergyHiEnvVar, PowerEnergyLoEnvVar,
PowerEnergyMinEnvVar, PowerEnergyMaxEnvVar, PowerFinishKillEnvVar,
PatternRadOffsetEnvVar, PatternRadScaleEnvVar,
PatternLenEnvVar, PatternDepthEnvVar,
TMH_SHIFT_ENV, TMH_BIG_MULT_ENV, TMH_LOG_ENV, TMH_RESET_ON_TARGET_ENV,
TMH_WINDOW_ENV, TMH_RESET_DROP_ENV, TMH_NSTATES_ENV, TMH_ACCURVE_ENV,
TMH_RETRAIN_EVERY_ENV, TMH_EPOCHS_ENV,
+62 -9
View File
@@ -18,9 +18,20 @@ import gun_harness/gun_interface
const
HistorySize* = 500
PatternLen* = 10 # ticks used as search key; ponytail: fixed, expose if tuning needed
PatternLen* = 10 # ticks used as search key; shipped default of TR_PATTERN_LEN
## Shipped defaults of the two MATCH-SHAPE runtime knobs (Task A, gun j123).
## `TR_PATTERN_LEN` is the length of the movement segment compared (the
## search key); `TR_PATTERN_DEPTH` is how many recent history entries the
## matcher is allowed to scan. Both default to the exact pre-knob constants,
## so an unset environment reproduces today's gun byte-for-byte.
PatternDepth* = HistorySize # max history entries scanned; default = full buffer
PatternRadOffsetEnvVar* = "TR_PATTERN_RAD_OFFSET" ## px; negative = aim short
PatternRadScaleEnvVar* = "TR_PATTERN_RAD_SCALE" ## multiplier on aim distance
PatternLenEnvVar* = "TR_PATTERN_LEN" ## search-key length (ticks)
## History/replay depth: the search never looks further back than this many
## entries. The buffer itself is a fixed `array[HistorySize]`, so the useful
## range is 1..HistorySize; the ceiling cannot be raised at runtime.
PatternDepthEnvVar* = "TR_PATTERN_DEPTH"
proc patternEnvFloat(name: string, default: float): float =
## Read an env knob like every other runtime switch in the harness: unset or
@@ -30,6 +41,18 @@ proc patternEnvFloat(name: string, default: float): float =
try: parseFloat(v.strip())
except ValueError: default
proc patternEnvInt(name: string, default: int): int =
## Integer twin of `patternEnvFloat`: unset or unparsable -> shipped default.
let v = getEnv(name, "")
if v.len == 0: return default
try: parseInt(v.strip())
except ValueError: default
proc clampInt(v, lo, hi: int): int {.inline.} =
result = v
if result < lo: result = lo
if result > hi: result = hi
type
MoveTick = object
velocity: float ## signed speed (px/tick)
@@ -60,6 +83,10 @@ type
pathY: array[HistorySize + 1, float]
pathHeading: array[HistorySize + 1, float] ## radians after s steps
pathSpeed: array[HistorySize + 1, float] ## speed after s steps
# ── match-shape knobs (defaults leave the prediction byte-identical) ──
patternLen*: int ## search-key length actually used (default PatternLen)
histDepth*: int ## history entries actually scanned (default full)
paramConfigured*: bool ## true once the env/setter has populated the two above
# ── radial offset knob (defaults leave the prediction byte-identical) ──
radScale*: float ## multiplier on the predicted aim distance
radOffset*: float ## px added to the predicted aim distance
@@ -95,20 +122,30 @@ proc findBestMatch(g: var PatternMatcherGun): int =
## Speed-independent history search. Returns the start index of the best
## matching pattern, or -1 when there is not enough history. Stores the best
## match cost in `g.lastMatchScore` for the confidence readout.
##
## `g.patternLen` (TR_PATTERN_LEN) is the key length; `g.histDepth`
## (TR_PATTERN_DEPTH) bounds how far back the scan may reach. With the shipped
## defaults (10, HistorySize) this is the exact pre-knob loop.
let plen = g.patternLen
g.lastMatchScore = Inf
if g.count < PatternLen * 2:
if g.count < plen * 2:
return -1
# key = last PatternLen entries
let keyStart = g.count - PatternLen
# key = last plen entries
let keyStart = g.count - plen
# scan backwards for best match (exclude the key itself)
# scan backwards for best match (exclude the key itself), but never further
# back than `histDepth` entries; when count <= histDepth this floor is 0 and
# the loop is identical to the original.
var bestScore = Inf
var bestMatch = -1
let scanEnd = g.count - PatternLen - 1 # last valid match start
for i in countdown(scanEnd, 0):
let scanEnd = g.count - plen - 1 # last valid match start
let scanFloor = max(0, g.count - g.histDepth)
if scanEnd < scanFloor:
return -1
for i in countdown(scanEnd, scanFloor):
var score = 0.0
for k in 0 ..< PatternLen:
for k in 0 ..< plen:
let a = g.readAt(keyStart + k)
let b = g.readAt(i + k)
let dv = a.velocity - b.velocity
@@ -123,7 +160,7 @@ proc findBestMatch(g: var PatternMatcherGun): int =
proc buildPath(g: var PatternMatcherGun, state: WorldState, bestMatch: int) =
## Precompute the matched pattern replayed forward from the current state.
## Only depends on observed movement, so it is valid for every power bin.
g.playStart = bestMatch + PatternLen
g.playStart = bestMatch + g.patternLen
g.playAvail = g.count - 1 - g.playStart # ticks we can replay
g.pathX[0] = state.enemyX
g.pathY[0] = state.enemyY
@@ -173,6 +210,21 @@ proc ensureRadialConfig(g: var PatternMatcherGun) {.inline.} =
g.radOffset = patternEnvFloat(PatternRadOffsetEnvVar, 0.0)
g.radConfigured = true
proc ensureParamConfig(g: var PatternMatcherGun) {.inline.} =
## Lazily read the match-shape env knobs on first use, mirroring the radial
## knobs: the shipped binary reads the env once per gun instance.
if g.paramConfigured: return
g.patternLen = clampInt(patternEnvInt(PatternLenEnvVar, PatternLen), 1, HistorySize)
g.histDepth = clampInt(patternEnvInt(PatternDepthEnvVar, PatternDepth), 1, HistorySize)
g.paramConfigured = true
proc setMatchParams*(g: var PatternMatcherGun, patternLen, histDepth: int) =
## Explicit per-gun override used by offline sweeps/tests. Writes the same
## fields the env path writes, so the measured code path is identical.
g.patternLen = clampInt(patternLen, 1, HistorySize)
g.histDepth = clampInt(histDepth, 1, HistorySize)
g.paramConfigured = true
proc setRadialCorrection*(g: var PatternMatcherGun, scale, offsetPx: float) =
## Explicit per-gun override used by the offline sweep. Writes the same fields
## the env path writes, so the measured code path is identical.
@@ -220,6 +272,7 @@ proc predict*(g: var PatternMatcherGun, state: WorldState,
g.cacheValid = true
g.cacheTick = state.tick
g.ensureParamConfig()
g.bestMatch = g.findBestMatch()
if g.bestMatch >= 0:
g.buildPath(state, g.bestMatch)
+106
View File
@@ -464,3 +464,109 @@ different reference is a free pairwise comparison with no battles.
|---|---|---:|---|---|
| `/tmp/ab/j121_g1` | `1d8143a` | 270 (0 failed, 0 never started, 0 excluded) | pattern, bitbrain, tmhorizon, knn, rack_pk, rack_pt | **nothing beats the shipped `pattern`**; all arms not distinguishable except `knn` = WORSE; `onlyPattern` CONFIRMED |
| `/tmp/ab/j121_g2` | `c343c00` | 297 (0 failed, 0 never started, 0 excluded) | pattern, bitbrain, tmhorizon | **the Batch-1 damage hint does not survive**: `bitbrain` +1.8 dmg/run p=0.49 (wash); `tmhorizon` −10.4 dmg/run p=0.035 (WORSE); wins flat |
---
# Phase 2: tuning the incumbent
**Owner's mandate (phase 2):** the phase-1 campaign closed the "other gun / other
rack" design space (nothing beats the shipped `Pattern`; the correctors are a
wash or a loss). What remains OPEN is the **incumbent's own tuning**: Pattern's
match-length / history parameters had NEVER been swept. Phase 2 asks the direct
question:
> **Can the shipped `Pattern` be improved by tuning its own match parameters —
> and if so, by how much on damage/run and round wins?**
The lead-amplitude axis is already known dead (`docs/bitbrain_campaign.md`), and
the radial knobs are known non-winners (job j99: `bmPath` structural no-op; live
+0.28 pp p=0.62 for scale 0.98, −0.42 pp p=0.46 for offset −20). Those are
therefore **controls** here, not candidates.
## Task A — exposing Pattern's match-shape parameters (what and why)
**MEASURED (code read).** `common_libs/guns/pattern_matcher.nim` had exactly two
tunable knobs, both RADIAL (`TR_PATTERN_RAD_SCALE`, `TR_PATTERN_RAD_OFFSET`), and
neither can change the lead bearing (job j99 proved the bearing is untouched), so
neither was ever the "lead information" axis. The parameters that actually
control **how the pattern is matched** were compile-time constants:
| parameter | code | what it controls | exposed as |
|---|---|---|---|
| match-key length | `PatternLen = 10` | the length of the movement segment compared (the search key); also how far after the match the replay starts | `TR_PATTERN_LEN` (int, default 10) |
| search depth | implicit `HistorySize = 500` | how far back the best-match scan may reach (`scanEnd` was always `count−PatternLen−1`) | `TR_PATTERN_DEPTH` (int, default 500 = full buffer) |
| similarity/search radius | **does not exist** | the search always takes the single lowest-cost match; there is no acceptance threshold or radius to expose | **not exposed — nothing to expose** |
| history buffer capacity | `HistorySize = 500` | the fixed `array[HistorySize]` backing store | runtime depth limit only; the buffer ceiling cannot be raised at runtime (see below) |
**Why env and not `-d:`.** The phase-1 instrument (`tools/ab/tournament_run.sh`)
builds ONE frozen binary from `git archive HEAD` and every arm differs only by its
env dict. A `{.intdefine.}` knob would need one binary per arm, which the
instrument forbids. Both new knobs are therefore resolved lazily from the
environment on the first `predict`, exactly like the radial knobs, and default to
the pre-knob constants. `setMatchParams(patternLen, histDepth)` is the explicit
offline/unit-test twin that writes the same fields.
**What could NOT be exposed cheaply (MEASURED).** `HistorySize` sizes four fixed
`array[HistorySize(+1)]` fields in the gun object. Raising it above 500 at
runtime is impossible without a heap buffer; the runtime `TR_PATTERN_DEPTH` knob
therefore *lowers* the effective search depth within the existing 500-entry
buffer. A depth above 500 is clamped to 500, and a non-positive or unparsable
value falls back to the shipped default. `PatternLen` is clamped to
`1..HistorySize`.
## Task A — default parity (MEASURED, byte-for-byte)
* **Baseline-vs-new parity dump.** A throwaway harness replayed 800 ticks of
`tools/fixtures/drussgt_vs_crazy.jsonl` through one `PatternMatcherGun` at all
four power bins (3200 predictions) and printed every point at full precision.
Compiled once against a `HEAD` worktree (`/tmp/j123_base`, before the change)
and once against the modified tree: **the two dumps are byte-identical**
(`diff -q` clean). The shipped default path is unchanged.
* **The knobs are live and self-falling-back.** `TR_PATTERN_LEN=6/16` and
`TR_PATTERN_DEPTH=100/20` each change the dump; `TR_PATTERN_LEN=banana`
reproduces the default dump exactly.
* **Existing guards pass** (no new guard tests, per the "cut ceremony" rule):
`common_libs/tests/test_pattern_radial_offset.nim` (6 checks) and
`common_libs/tests/test_gun_harness.nim` (all checks) pass; the env-report
guard `test_env_report.nim` passes with `TR_PATTERN_LEN` / `TR_PATTERN_DEPTH`
registered in `knownEnvNames()` and the effective-values report.
* **Clean-archive compile** is exercised by `tournament_run.sh` itself, which
builds the frozen binary from `git archive HEAD`.
## Task B — Batch 1 (pre-registered BEFORE any battle)
**Design.** One frozen binary, six env-only arms, the FROZEN 15-opponent panel
`tools/ab/panel_movement.txt`, **3 runs × 3 rounds per (opponent, arm) = 270
battles**, `--conc 6`, `--wait-arena`. Arm file: `tools/ab/arms_gun_b3.txt`.
Reference: `pattern`. Movement pinned `TR_MOVEMENT=strafe` in every arm.
| # | arm | env over the pin | what it isolates |
|---|---|---|---|
| 1 | `pattern` | (none) | the arm to beat (shipped `onlyPattern` rack) |
| 2 | `len6` | `TR_PATTERN_LEN=6` | shorter movement segment compared |
| 3 | `len16` | `TR_PATTERN_LEN=16` | longer movement segment compared |
| 4 | `depth100` | `TR_PATTERN_DEPTH=100` | shallower history / match search |
| 5 | `rad_offset` | `TR_PATTERN_RAD_OFFSET=-20` | control: aim 20 px short |
| 6 | `rad_scale` | `TR_PATTERN_RAD_SCALE=0.95` | control: scale the aim distance |
**Pre-registered prediction (written BEFORE the battles finished):**
1. **No arm beats `pattern` on both primaries.** The incumbent is already tuned —
the phase-2 null. The point estimates sit inside the MDE.
2. `len16` is **WORSE or flat** on damage: a 16-tick key matches rarely in a
~500-tick buffer, so the gun falls back to the linear forecast (the weaker
base) more often; wins flat.
3. `len6` is **flat** on damage (more matches but a noisier replay) and flat on
wins; possibly a sub-MDE wobble in either direction.
4. `depth100` is **not distinguishable** from `pattern` — the full buffer already
contains the useful candidates; a sub-MDE damage wobble is possible.
5. The two radial controls are **not distinguishable** and, per job j99, cannot
change the real aim bearing; any damage delta is the fire-gate channel only.
6. If any surprise exists, it is a **damage-only wobble without wins** — the
`ring`-mover trap — and will not be read as a win.
**Pre-registered Batch-2 trigger (task rule).** A Batch-1 arm is promoted to a
higher-power Batch 2 **only if** it beats the reference `pattern` by the
campaign rule 2 (one primary up at cross-opponent sign-test p<0.05 while the
other does not go down) **AND** its damage CI excludes 0 **AND** the sign-flip
permutation test gives p<0.05. Otherwise Batch 1 is the answer.
+53
View File
@@ -0,0 +1,53 @@
# ─────────────────────────────────────────────────────────────────────────────
# arms_gun_b3.txt — PHASE 2 BATCH 1 of the GUN campaign: does tuning the
# incumbent's OWN match-shape parameters beat the shipped `Pattern`?
#
# Format: name | ENV=value ENV=value | label
#
# ONE frozen binary (tournament_run.sh builds it from `git archive HEAD`); every
# arm below differs ONLY by its env dict. No per-arm rebuild.
#
# MOVEMENT IS PINNED IN EVERY ARM (`TR_MOVEMENT=strafe`). Reason (hard rule):
# a parallel job (j124) may be fighting, and the shipped movement default may
# move; pinning the engine in EVERY arm removes movement as a confound, makes all
# arms share one movement, and keeps the comparison a pure GUN comparison. It
# also satisfies the analyzer liveness rule (every declared token must appear in
# the bot's own raw-env report, and an undeclared TR_MOVEMENT is fatal).
#
# Task A (this job) exposed Pattern's own match-shape parameters as runtime env
# knobs, defaults byte-identical to the pre-knob gun (proven by the parity dump
# in the ledger):
# TR_PATTERN_LEN — length of the movement segment compared (search key).
# Shipped default 10.
# TR_PATTERN_DEPTH — how many recent history entries the matcher may scan.
# Shipped default 500 (= the whole HistorySize buffer).
# The two pre-existing radial knobs are kept as cheap CONTROLS.
#
# Prior data this batch is built on (taken as given; NOT re-derived):
# * `docs/gun_campaign.md` Batches 1-2: nothing beats the shipped `pattern`
# across 15 and 33 opponents; `onlyPattern` is a measured optimum.
# * `docs/bitbrain_campaign.md`: lead AMPLITUDE is a dead axis; lead
# INFORMATION is the open one.
# * `common_libs/tests/pattern_radial_results.md` (job j99): the radial knobs
# move the aim point along an UNCHANGED bearing, so under the shipped
# `bmPath` metric they are a structural no-op; live they were +0.28 pp
# (p=0.62) / −0.42 pp (p=0.46).
# ─────────────────────────────────────────────────────────────────────────────
# 1. REFERENCE. The shipped `onlyPattern` rack, movement pinned, no shape env.
pattern | TR_MOVEMENT=strafe | shipped onlyPattern rack (REFERENCE)
# 2. Shorter match key: more candidate matches, noisier replay.
len6 | TR_MOVEMENT=strafe TR_PATTERN_LEN=6 | match key = 6 ticks
# 3. Longer match key: fewer, more specific matches; more linear fallback.
len16 | TR_MOVEMENT=strafe TR_PATTERN_LEN=16 | match key = 16 ticks
# 4. Shallow history: the search never looks further back than 100 entries.
depth100 | TR_MOVEMENT=strafe TR_PATTERN_DEPTH=100 | history search = 100 ticks
# 5. CONTROL: aim 20 px short (the radial-knob header's intended use).
rad_offset | TR_MOVEMENT=strafe TR_PATTERN_RAD_OFFSET=-20 | radial control: aim short 20 px
# 6. CONTROL: scale the aim distance by 0.95.
rad_scale | TR_MOVEMENT=strafe TR_PATTERN_RAD_SCALE=0.95 | radial control: scale 0.95