gun j123 Task A+B prereg: expose Pattern's TR_PATTERN_LEN/TR_PATTERN_DEPTH (default parity) and pre-register the 6-arm match-parameter sweep
This commit is contained in:
@@ -365,9 +365,17 @@ proc printEffectiveValues(ctx: EnvReportContext) =
|
||||
emit("TR_BITBRAIN_RESET_ON_TARGET", onOff(ctx.bitbrain.resetOnTarget),
|
||||
sourceOf(BB_RESET_ON_TARGET_ENV))
|
||||
|
||||
# ── Pattern radial knobs ──────────────────────────────────────────────────
|
||||
# ── Pattern match-shape + radial knobs ────────────────────────────────────
|
||||
# These are resolved lazily inside `predict` (which has not run at boot), so
|
||||
# unless something already forced them we report the raw env value and say so.
|
||||
if ctx.patternMatcher.paramConfigured:
|
||||
emit("TR_PATTERN_LEN", $ctx.patternMatcher.patternLen, sourceOf("TR_PATTERN_LEN"))
|
||||
emit("TR_PATTERN_DEPTH", $ctx.patternMatcher.histDepth, sourceOf("TR_PATTERN_DEPTH"))
|
||||
else:
|
||||
echo "[env] TR_PATTERN_LEN = ",
|
||||
getEnv("TR_PATTERN_LEN", $PatternLen), " (raw - not resolved here)"
|
||||
echo "[env] TR_PATTERN_DEPTH = ",
|
||||
getEnv("TR_PATTERN_DEPTH", $PatternDepth), " (raw - not resolved here)"
|
||||
if ctx.patternMatcher.radConfigured:
|
||||
emit("TR_PATTERN_RAD_OFFSET", $ctx.patternMatcher.radOffset, sourceOf("TR_PATTERN_RAD_OFFSET"))
|
||||
emit("TR_PATTERN_RAD_SCALE", $ctx.patternMatcher.radScale, sourceOf("TR_PATTERN_RAD_SCALE"))
|
||||
@@ -428,6 +436,7 @@ proc knownEnvNames*(): seq[string] =
|
||||
PowerRefEnvVar, PowerEnergyHiEnvVar, PowerEnergyLoEnvVar,
|
||||
PowerEnergyMinEnvVar, PowerEnergyMaxEnvVar, PowerFinishKillEnvVar,
|
||||
PatternRadOffsetEnvVar, PatternRadScaleEnvVar,
|
||||
PatternLenEnvVar, PatternDepthEnvVar,
|
||||
TMH_SHIFT_ENV, TMH_BIG_MULT_ENV, TMH_LOG_ENV, TMH_RESET_ON_TARGET_ENV,
|
||||
TMH_WINDOW_ENV, TMH_RESET_DROP_ENV, TMH_NSTATES_ENV, TMH_ACCURVE_ENV,
|
||||
TMH_RETRAIN_EVERY_ENV, TMH_EPOCHS_ENV,
|
||||
|
||||
@@ -18,9 +18,20 @@ import gun_harness/gun_interface
|
||||
|
||||
const
|
||||
HistorySize* = 500
|
||||
PatternLen* = 10 # ticks used as search key; ponytail: fixed, expose if tuning needed
|
||||
PatternLen* = 10 # ticks used as search key; shipped default of TR_PATTERN_LEN
|
||||
## Shipped defaults of the two MATCH-SHAPE runtime knobs (Task A, gun j123).
|
||||
## `TR_PATTERN_LEN` is the length of the movement segment compared (the
|
||||
## search key); `TR_PATTERN_DEPTH` is how many recent history entries the
|
||||
## matcher is allowed to scan. Both default to the exact pre-knob constants,
|
||||
## so an unset environment reproduces today's gun byte-for-byte.
|
||||
PatternDepth* = HistorySize # max history entries scanned; default = full buffer
|
||||
PatternRadOffsetEnvVar* = "TR_PATTERN_RAD_OFFSET" ## px; negative = aim short
|
||||
PatternRadScaleEnvVar* = "TR_PATTERN_RAD_SCALE" ## multiplier on aim distance
|
||||
PatternLenEnvVar* = "TR_PATTERN_LEN" ## search-key length (ticks)
|
||||
## History/replay depth: the search never looks further back than this many
|
||||
## entries. The buffer itself is a fixed `array[HistorySize]`, so the useful
|
||||
## range is 1..HistorySize; the ceiling cannot be raised at runtime.
|
||||
PatternDepthEnvVar* = "TR_PATTERN_DEPTH"
|
||||
|
||||
proc patternEnvFloat(name: string, default: float): float =
|
||||
## Read an env knob like every other runtime switch in the harness: unset or
|
||||
@@ -30,6 +41,18 @@ proc patternEnvFloat(name: string, default: float): float =
|
||||
try: parseFloat(v.strip())
|
||||
except ValueError: default
|
||||
|
||||
proc patternEnvInt(name: string, default: int): int =
|
||||
## Integer twin of `patternEnvFloat`: unset or unparsable -> shipped default.
|
||||
let v = getEnv(name, "")
|
||||
if v.len == 0: return default
|
||||
try: parseInt(v.strip())
|
||||
except ValueError: default
|
||||
|
||||
proc clampInt(v, lo, hi: int): int {.inline.} =
|
||||
result = v
|
||||
if result < lo: result = lo
|
||||
if result > hi: result = hi
|
||||
|
||||
type
|
||||
MoveTick = object
|
||||
velocity: float ## signed speed (px/tick)
|
||||
@@ -60,6 +83,10 @@ type
|
||||
pathY: array[HistorySize + 1, float]
|
||||
pathHeading: array[HistorySize + 1, float] ## radians after s steps
|
||||
pathSpeed: array[HistorySize + 1, float] ## speed after s steps
|
||||
# ── match-shape knobs (defaults leave the prediction byte-identical) ──
|
||||
patternLen*: int ## search-key length actually used (default PatternLen)
|
||||
histDepth*: int ## history entries actually scanned (default full)
|
||||
paramConfigured*: bool ## true once the env/setter has populated the two above
|
||||
# ── radial offset knob (defaults leave the prediction byte-identical) ──
|
||||
radScale*: float ## multiplier on the predicted aim distance
|
||||
radOffset*: float ## px added to the predicted aim distance
|
||||
@@ -95,20 +122,30 @@ proc findBestMatch(g: var PatternMatcherGun): int =
|
||||
## Speed-independent history search. Returns the start index of the best
|
||||
## matching pattern, or -1 when there is not enough history. Stores the best
|
||||
## match cost in `g.lastMatchScore` for the confidence readout.
|
||||
##
|
||||
## `g.patternLen` (TR_PATTERN_LEN) is the key length; `g.histDepth`
|
||||
## (TR_PATTERN_DEPTH) bounds how far back the scan may reach. With the shipped
|
||||
## defaults (10, HistorySize) this is the exact pre-knob loop.
|
||||
let plen = g.patternLen
|
||||
g.lastMatchScore = Inf
|
||||
if g.count < PatternLen * 2:
|
||||
if g.count < plen * 2:
|
||||
return -1
|
||||
|
||||
# key = last PatternLen entries
|
||||
let keyStart = g.count - PatternLen
|
||||
# key = last plen entries
|
||||
let keyStart = g.count - plen
|
||||
|
||||
# scan backwards for best match (exclude the key itself)
|
||||
# scan backwards for best match (exclude the key itself), but never further
|
||||
# back than `histDepth` entries; when count <= histDepth this floor is 0 and
|
||||
# the loop is identical to the original.
|
||||
var bestScore = Inf
|
||||
var bestMatch = -1
|
||||
let scanEnd = g.count - PatternLen - 1 # last valid match start
|
||||
for i in countdown(scanEnd, 0):
|
||||
let scanEnd = g.count - plen - 1 # last valid match start
|
||||
let scanFloor = max(0, g.count - g.histDepth)
|
||||
if scanEnd < scanFloor:
|
||||
return -1
|
||||
for i in countdown(scanEnd, scanFloor):
|
||||
var score = 0.0
|
||||
for k in 0 ..< PatternLen:
|
||||
for k in 0 ..< plen:
|
||||
let a = g.readAt(keyStart + k)
|
||||
let b = g.readAt(i + k)
|
||||
let dv = a.velocity - b.velocity
|
||||
@@ -123,7 +160,7 @@ proc findBestMatch(g: var PatternMatcherGun): int =
|
||||
proc buildPath(g: var PatternMatcherGun, state: WorldState, bestMatch: int) =
|
||||
## Precompute the matched pattern replayed forward from the current state.
|
||||
## Only depends on observed movement, so it is valid for every power bin.
|
||||
g.playStart = bestMatch + PatternLen
|
||||
g.playStart = bestMatch + g.patternLen
|
||||
g.playAvail = g.count - 1 - g.playStart # ticks we can replay
|
||||
g.pathX[0] = state.enemyX
|
||||
g.pathY[0] = state.enemyY
|
||||
@@ -173,6 +210,21 @@ proc ensureRadialConfig(g: var PatternMatcherGun) {.inline.} =
|
||||
g.radOffset = patternEnvFloat(PatternRadOffsetEnvVar, 0.0)
|
||||
g.radConfigured = true
|
||||
|
||||
proc ensureParamConfig(g: var PatternMatcherGun) {.inline.} =
|
||||
## Lazily read the match-shape env knobs on first use, mirroring the radial
|
||||
## knobs: the shipped binary reads the env once per gun instance.
|
||||
if g.paramConfigured: return
|
||||
g.patternLen = clampInt(patternEnvInt(PatternLenEnvVar, PatternLen), 1, HistorySize)
|
||||
g.histDepth = clampInt(patternEnvInt(PatternDepthEnvVar, PatternDepth), 1, HistorySize)
|
||||
g.paramConfigured = true
|
||||
|
||||
proc setMatchParams*(g: var PatternMatcherGun, patternLen, histDepth: int) =
|
||||
## Explicit per-gun override used by offline sweeps/tests. Writes the same
|
||||
## fields the env path writes, so the measured code path is identical.
|
||||
g.patternLen = clampInt(patternLen, 1, HistorySize)
|
||||
g.histDepth = clampInt(histDepth, 1, HistorySize)
|
||||
g.paramConfigured = true
|
||||
|
||||
proc setRadialCorrection*(g: var PatternMatcherGun, scale, offsetPx: float) =
|
||||
## Explicit per-gun override used by the offline sweep. Writes the same fields
|
||||
## the env path writes, so the measured code path is identical.
|
||||
@@ -220,6 +272,7 @@ proc predict*(g: var PatternMatcherGun, state: WorldState,
|
||||
g.cacheValid = true
|
||||
g.cacheTick = state.tick
|
||||
|
||||
g.ensureParamConfig()
|
||||
g.bestMatch = g.findBestMatch()
|
||||
if g.bestMatch >= 0:
|
||||
g.buildPath(state, g.bestMatch)
|
||||
|
||||
@@ -464,3 +464,109 @@ different reference is a free pairwise comparison with no battles.
|
||||
|---|---|---:|---|---|
|
||||
| `/tmp/ab/j121_g1` | `1d8143a` | 270 (0 failed, 0 never started, 0 excluded) | pattern, bitbrain, tmhorizon, knn, rack_pk, rack_pt | **nothing beats the shipped `pattern`**; all arms not distinguishable except `knn` = WORSE; `onlyPattern` CONFIRMED |
|
||||
| `/tmp/ab/j121_g2` | `c343c00` | 297 (0 failed, 0 never started, 0 excluded) | pattern, bitbrain, tmhorizon | **the Batch-1 damage hint does not survive**: `bitbrain` +1.8 dmg/run p=0.49 (wash); `tmhorizon` −10.4 dmg/run p=0.035 (WORSE); wins flat |
|
||||
|
||||
---
|
||||
|
||||
# Phase 2: tuning the incumbent
|
||||
|
||||
**Owner's mandate (phase 2):** the phase-1 campaign closed the "other gun / other
|
||||
rack" design space (nothing beats the shipped `Pattern`; the correctors are a
|
||||
wash or a loss). What remains OPEN is the **incumbent's own tuning**: Pattern's
|
||||
match-length / history parameters had NEVER been swept. Phase 2 asks the direct
|
||||
question:
|
||||
|
||||
> **Can the shipped `Pattern` be improved by tuning its own match parameters —
|
||||
> and if so, by how much on damage/run and round wins?**
|
||||
|
||||
The lead-amplitude axis is already known dead (`docs/bitbrain_campaign.md`), and
|
||||
the radial knobs are known non-winners (job j99: `bmPath` structural no-op; live
|
||||
+0.28 pp p=0.62 for scale 0.98, −0.42 pp p=0.46 for offset −20). Those are
|
||||
therefore **controls** here, not candidates.
|
||||
|
||||
## Task A — exposing Pattern's match-shape parameters (what and why)
|
||||
|
||||
**MEASURED (code read).** `common_libs/guns/pattern_matcher.nim` had exactly two
|
||||
tunable knobs, both RADIAL (`TR_PATTERN_RAD_SCALE`, `TR_PATTERN_RAD_OFFSET`), and
|
||||
neither can change the lead bearing (job j99 proved the bearing is untouched), so
|
||||
neither was ever the "lead information" axis. The parameters that actually
|
||||
control **how the pattern is matched** were compile-time constants:
|
||||
|
||||
| parameter | code | what it controls | exposed as |
|
||||
|---|---|---|---|
|
||||
| match-key length | `PatternLen = 10` | the length of the movement segment compared (the search key); also how far after the match the replay starts | `TR_PATTERN_LEN` (int, default 10) |
|
||||
| search depth | implicit `HistorySize = 500` | how far back the best-match scan may reach (`scanEnd` was always `count−PatternLen−1`) | `TR_PATTERN_DEPTH` (int, default 500 = full buffer) |
|
||||
| similarity/search radius | **does not exist** | the search always takes the single lowest-cost match; there is no acceptance threshold or radius to expose | **not exposed — nothing to expose** |
|
||||
| history buffer capacity | `HistorySize = 500` | the fixed `array[HistorySize]` backing store | runtime depth limit only; the buffer ceiling cannot be raised at runtime (see below) |
|
||||
|
||||
**Why env and not `-d:`.** The phase-1 instrument (`tools/ab/tournament_run.sh`)
|
||||
builds ONE frozen binary from `git archive HEAD` and every arm differs only by its
|
||||
env dict. A `{.intdefine.}` knob would need one binary per arm, which the
|
||||
instrument forbids. Both new knobs are therefore resolved lazily from the
|
||||
environment on the first `predict`, exactly like the radial knobs, and default to
|
||||
the pre-knob constants. `setMatchParams(patternLen, histDepth)` is the explicit
|
||||
offline/unit-test twin that writes the same fields.
|
||||
|
||||
**What could NOT be exposed cheaply (MEASURED).** `HistorySize` sizes four fixed
|
||||
`array[HistorySize(+1)]` fields in the gun object. Raising it above 500 at
|
||||
runtime is impossible without a heap buffer; the runtime `TR_PATTERN_DEPTH` knob
|
||||
therefore *lowers* the effective search depth within the existing 500-entry
|
||||
buffer. A depth above 500 is clamped to 500, and a non-positive or unparsable
|
||||
value falls back to the shipped default. `PatternLen` is clamped to
|
||||
`1..HistorySize`.
|
||||
|
||||
## Task A — default parity (MEASURED, byte-for-byte)
|
||||
|
||||
* **Baseline-vs-new parity dump.** A throwaway harness replayed 800 ticks of
|
||||
`tools/fixtures/drussgt_vs_crazy.jsonl` through one `PatternMatcherGun` at all
|
||||
four power bins (3200 predictions) and printed every point at full precision.
|
||||
Compiled once against a `HEAD` worktree (`/tmp/j123_base`, before the change)
|
||||
and once against the modified tree: **the two dumps are byte-identical**
|
||||
(`diff -q` clean). The shipped default path is unchanged.
|
||||
* **The knobs are live and self-falling-back.** `TR_PATTERN_LEN=6/16` and
|
||||
`TR_PATTERN_DEPTH=100/20` each change the dump; `TR_PATTERN_LEN=banana`
|
||||
reproduces the default dump exactly.
|
||||
* **Existing guards pass** (no new guard tests, per the "cut ceremony" rule):
|
||||
`common_libs/tests/test_pattern_radial_offset.nim` (6 checks) and
|
||||
`common_libs/tests/test_gun_harness.nim` (all checks) pass; the env-report
|
||||
guard `test_env_report.nim` passes with `TR_PATTERN_LEN` / `TR_PATTERN_DEPTH`
|
||||
registered in `knownEnvNames()` and the effective-values report.
|
||||
* **Clean-archive compile** is exercised by `tournament_run.sh` itself, which
|
||||
builds the frozen binary from `git archive HEAD`.
|
||||
|
||||
## Task B — Batch 1 (pre-registered BEFORE any battle)
|
||||
|
||||
**Design.** One frozen binary, six env-only arms, the FROZEN 15-opponent panel
|
||||
`tools/ab/panel_movement.txt`, **3 runs × 3 rounds per (opponent, arm) = 270
|
||||
battles**, `--conc 6`, `--wait-arena`. Arm file: `tools/ab/arms_gun_b3.txt`.
|
||||
Reference: `pattern`. Movement pinned `TR_MOVEMENT=strafe` in every arm.
|
||||
|
||||
| # | arm | env over the pin | what it isolates |
|
||||
|---|---|---|---|
|
||||
| 1 | `pattern` | (none) | the arm to beat (shipped `onlyPattern` rack) |
|
||||
| 2 | `len6` | `TR_PATTERN_LEN=6` | shorter movement segment compared |
|
||||
| 3 | `len16` | `TR_PATTERN_LEN=16` | longer movement segment compared |
|
||||
| 4 | `depth100` | `TR_PATTERN_DEPTH=100` | shallower history / match search |
|
||||
| 5 | `rad_offset` | `TR_PATTERN_RAD_OFFSET=-20` | control: aim 20 px short |
|
||||
| 6 | `rad_scale` | `TR_PATTERN_RAD_SCALE=0.95` | control: scale the aim distance |
|
||||
|
||||
**Pre-registered prediction (written BEFORE the battles finished):**
|
||||
1. **No arm beats `pattern` on both primaries.** The incumbent is already tuned —
|
||||
the phase-2 null. The point estimates sit inside the MDE.
|
||||
2. `len16` is **WORSE or flat** on damage: a 16-tick key matches rarely in a
|
||||
~500-tick buffer, so the gun falls back to the linear forecast (the weaker
|
||||
base) more often; wins flat.
|
||||
3. `len6` is **flat** on damage (more matches but a noisier replay) and flat on
|
||||
wins; possibly a sub-MDE wobble in either direction.
|
||||
4. `depth100` is **not distinguishable** from `pattern` — the full buffer already
|
||||
contains the useful candidates; a sub-MDE damage wobble is possible.
|
||||
5. The two radial controls are **not distinguishable** and, per job j99, cannot
|
||||
change the real aim bearing; any damage delta is the fire-gate channel only.
|
||||
6. If any surprise exists, it is a **damage-only wobble without wins** — the
|
||||
`ring`-mover trap — and will not be read as a win.
|
||||
|
||||
**Pre-registered Batch-2 trigger (task rule).** A Batch-1 arm is promoted to a
|
||||
higher-power Batch 2 **only if** it beats the reference `pattern` by the
|
||||
campaign rule 2 (one primary up at cross-opponent sign-test p<0.05 while the
|
||||
other does not go down) **AND** its damage CI excludes 0 **AND** the sign-flip
|
||||
permutation test gives p<0.05. Otherwise Batch 1 is the answer.
|
||||
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# arms_gun_b3.txt — PHASE 2 BATCH 1 of the GUN campaign: does tuning the
|
||||
# incumbent's OWN match-shape parameters beat the shipped `Pattern`?
|
||||
#
|
||||
# Format: name | ENV=value ENV=value | label
|
||||
#
|
||||
# ONE frozen binary (tournament_run.sh builds it from `git archive HEAD`); every
|
||||
# arm below differs ONLY by its env dict. No per-arm rebuild.
|
||||
#
|
||||
# MOVEMENT IS PINNED IN EVERY ARM (`TR_MOVEMENT=strafe`). Reason (hard rule):
|
||||
# a parallel job (j124) may be fighting, and the shipped movement default may
|
||||
# move; pinning the engine in EVERY arm removes movement as a confound, makes all
|
||||
# arms share one movement, and keeps the comparison a pure GUN comparison. It
|
||||
# also satisfies the analyzer liveness rule (every declared token must appear in
|
||||
# the bot's own raw-env report, and an undeclared TR_MOVEMENT is fatal).
|
||||
#
|
||||
# Task A (this job) exposed Pattern's own match-shape parameters as runtime env
|
||||
# knobs, defaults byte-identical to the pre-knob gun (proven by the parity dump
|
||||
# in the ledger):
|
||||
# TR_PATTERN_LEN — length of the movement segment compared (search key).
|
||||
# Shipped default 10.
|
||||
# TR_PATTERN_DEPTH — how many recent history entries the matcher may scan.
|
||||
# Shipped default 500 (= the whole HistorySize buffer).
|
||||
# The two pre-existing radial knobs are kept as cheap CONTROLS.
|
||||
#
|
||||
# Prior data this batch is built on (taken as given; NOT re-derived):
|
||||
# * `docs/gun_campaign.md` Batches 1-2: nothing beats the shipped `pattern`
|
||||
# across 15 and 33 opponents; `onlyPattern` is a measured optimum.
|
||||
# * `docs/bitbrain_campaign.md`: lead AMPLITUDE is a dead axis; lead
|
||||
# INFORMATION is the open one.
|
||||
# * `common_libs/tests/pattern_radial_results.md` (job j99): the radial knobs
|
||||
# move the aim point along an UNCHANGED bearing, so under the shipped
|
||||
# `bmPath` metric they are a structural no-op; live they were +0.28 pp
|
||||
# (p=0.62) / −0.42 pp (p=0.46).
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
# 1. REFERENCE. The shipped `onlyPattern` rack, movement pinned, no shape env.
|
||||
pattern | TR_MOVEMENT=strafe | shipped onlyPattern rack (REFERENCE)
|
||||
|
||||
# 2. Shorter match key: more candidate matches, noisier replay.
|
||||
len6 | TR_MOVEMENT=strafe TR_PATTERN_LEN=6 | match key = 6 ticks
|
||||
|
||||
# 3. Longer match key: fewer, more specific matches; more linear fallback.
|
||||
len16 | TR_MOVEMENT=strafe TR_PATTERN_LEN=16 | match key = 16 ticks
|
||||
|
||||
# 4. Shallow history: the search never looks further back than 100 entries.
|
||||
depth100 | TR_MOVEMENT=strafe TR_PATTERN_DEPTH=100 | history search = 100 ticks
|
||||
|
||||
# 5. CONTROL: aim 20 px short (the radial-knob header's intended use).
|
||||
rad_offset | TR_MOVEMENT=strafe TR_PATTERN_RAD_OFFSET=-20 | radial control: aim short 20 px
|
||||
|
||||
# 6. CONTROL: scale the aim distance by 0.95.
|
||||
rad_scale | TR_MOVEMENT=strafe TR_PATTERN_RAD_SCALE=0.95 | radial control: scale 0.95
|
||||
Reference in New Issue
Block a user