From 2747ebd32364eb7128dde39aff81ce493b035122 Mon Sep 17 00:00:00 2001 From: Davide Cappellini Date: Fri, 25 Sep 2026 00:15:27 +0200 Subject: [PATCH] BitBrain: TR_BITBRAIN_GAINS env knob (candidate set + fixed-gain degenerate) Task A of campaign phase 2: the lead-gain candidate set is now pure env, so the live arms need no recompile. - common_libs/guns/bitbrain_gun.nim: BB_GAINS_ENV (TR_BITBRAIN_GAINS); the candidate list is parsed once at gun construction into a dynamic seq, so the hit counts/hit rates are sized to it. Unset/unparsable -> the shipped BB_CAND set [0,0.25,0.5,0.75,1.0] (byte-identical behaviour). Exactly ONE candidate degenerates to a FIXED gain applied from the first shot (learning bypassed), still gated to the long bands. parseGains clamps to [0,8], de-dupes and sorts so the argmax tie rule is unchanged. The [bb] line now prints the APPLIED gain AND the resulting angular shift, so a run's correction is auditable from stdout. - ModularBot_garage/src/env_report.nim: emit TR_BITBRAIN_GAINS (resolved candidate set) and add BB_GAINS_ENV to the known-name list. - tools/ab/arms_leadgain.txt: the 6-arm phase-2 sweep definition. --- ModularBot_garage/src/env_report.nim | 4 +- common_libs/guns/bitbrain_gun.nim | 99 +++++++++++++++++++++++----- tools/ab/arms_leadgain.txt | 22 +++++++ 3 files changed, 106 insertions(+), 19 deletions(-) create mode 100644 tools/ab/arms_leadgain.txt diff --git a/ModularBot_garage/src/env_report.nim b/ModularBot_garage/src/env_report.nim index f3bf419..58c221f 100644 --- a/ModularBot_garage/src/env_report.nim +++ b/ModularBot_garage/src/env_report.nim @@ -289,6 +289,8 @@ proc printEffectiveValues(ctx: EnvReportContext) = # `predict`, which is why `[rack]` membership is the actual enable switch. emit("TR_BITBRAIN_MEM", memModeName(ctx.bitbrain.memMode), sourceOf(BB_MEM_ENV)) + emit("TR_BITBRAIN_GAINS", bbGainsString(ctx.bitbrain.cands), + sourceOf(BB_GAINS_ENV)) emit("TR_BITBRAIN_N", $ctx.bitbrain.nClasses, sourceOf(BB_N_ENV)) emit("TR_BITBRAIN_NADE", $ctx.bitbrain.nAde, sourceOf(BB_NADE_ENV)) emit("TR_BITBRAIN_RANGE", $ctx.bitbrain.maxDeg, sourceOf(BB_RANGE_ENV)) @@ -371,7 +373,7 @@ proc knownEnvNames*(): seq[string] = TMH_SHIFT_ENV, TMH_BIG_MULT_ENV, TMH_LOG_ENV, TMH_RESET_ON_TARGET_ENV, TMH_WINDOW_ENV, TMH_RESET_DROP_ENV, TMH_NSTATES_ENV, TMH_ACCURVE_ENV, TMH_RETRAIN_EVERY_ENV, TMH_EPOCHS_ENV, - BB_MEM_ENV, BB_N_ENV, BB_NADE_ENV, BB_RANGE_ENV, BB_LOG_ENV, + BB_MEM_ENV, BB_GAINS_ENV, BB_N_ENV, BB_NADE_ENV, BB_RANGE_ENV, BB_LOG_ENV, BB_MIN_OBS_ENV, BB_WARMUP_ENV, BB_ADAPT_ENV, BB_CALIB_ENV, BB_DECAY_ENV, BB_DECAY_FRAC_ENV, BB_SEED_ENV, BB_RESET_ON_TARGET_ENV, ] diff --git a/common_libs/guns/bitbrain_gun.nim b/common_libs/guns/bitbrain_gun.nim index 8e9378d..655bfde 100644 --- a/common_libs/guns/bitbrain_gun.nim +++ b/common_libs/guns/bitbrain_gun.nim @@ -60,11 +60,23 @@ ## `common_libs/bitbrain/` library is untouched and still tested by ## `test_bitbrain.nim`. ## +## ── TR_BITBRAIN_GAINS (the candidate set as an env knob) ───────────────────── +## `TR_BITBRAIN_GAINS` is a comma-separated candidate list, e.g. +## `TR_BITBRAIN_GAINS=1.0,1.25,1.5,2.0`. It replaces the fixed shipped candidate +## set `{0, 0.25, 0.5, 0.75, 1.0}` for this gun instance, so every live arm is +## pure-env (no recompile). Two degenerate cases are deliberate: +## * unset / unparsable -> the shipped `BB_CAND` set, byte-identical behaviour; +## * exactly ONE value -> a FIXED gain, applied from the first shot with NO +## learning at all (the learner is bypassed), still gated to the long bands. +## The applied `gain` (and the resulting angular `shift`) is printed on the +## existing change-gated `[bb]` line, so a run's liveness AND the correction it +## actually applied are both auditable from the bot's stdout. +## ## DEFAULT OFF / PARITY: this gun is admitted ONLY when `TR_RACK_BITBRAIN` says so ## (default `off`). The shipped rack never calls `predict`, so `ensureInit` never ## runs and the shipped bot is byte-for-byte unchanged. -import std/[math, os, strutils, strformat] +import std/[math, os, strutils, strformat, algorithm] import gun_harness/gun_interface import guns/tm_horizon import guns/pattern_matcher @@ -72,6 +84,7 @@ import guns/pattern_matcher const ## ── env knobs (all resolved once at gun construction) ───────────────────── BB_MEM_ENV* = "TR_BITBRAIN_MEM" ## perRound|retained|decay + BB_GAINS_ENV* = "TR_BITBRAIN_GAINS" ## comma-separated candidate gains BB_N_ENV* = "TR_BITBRAIN_N" ## (legacy geometry; inert) BB_NADE_ENV* = "TR_BITBRAIN_NADE" ## (legacy ADE count; inert) BB_RANGE_ENV* = "TR_BITBRAIN_RANGE" ## (legacy class half-range; inert) @@ -91,8 +104,9 @@ const BB_NHB* = 4 ## horizon buckets (for the per-tick label dedupe) BB_BAND_LO* = [0.0, 100.0, 200.0, 300.0, 450.0] BB_BAND_HI* = [100.0, 200.0, 300.0, 450.0, 1.0e18] - ## The candidate lead gains the band selector picks from. 0.0 == HeadOn (aim - ## at the current position) and 1.0 == Pattern (use the full lead). + ## The DEFAULT candidate lead gains the band selector picks from. 0.0 == HeadOn + ## (aim at the current position) and 1.0 == Pattern (use the full lead). + ## `TR_BITBRAIN_GAINS` replaces this set per gun; unset -> this exact set. BB_CAND* = [0.0, 0.25, 0.50, 0.75, 1.0] BB_NCAND* = 5 BB_BB_RADIUS* = 18.0 ## hit-detection radius in px (ruler tolerance) @@ -146,9 +160,13 @@ type decayFrac*: float seed*: int64 resetOnTarget*: bool + # ── the candidate gain set (resolved once at construction) ──────────────── + ## Ascending; one entry == a FIXED gain with no learning. `bandHits` and + ## `bandN` are sized to it, so the loops below never touch a stale column. + cands*: seq[float] # ── gain learner: hit counts per (range band x candidate gain) ──────────── - bandHits*: array[BB_NBANDS, array[BB_NCAND, float64]] - bandN*: array[BB_NBANDS, float64] + bandHits*: seq[seq[float64]] + bandN*: seq[float64] trained*: int sinceDecay: int decays*: int @@ -179,6 +197,37 @@ proc memModeName*(m: BitMemMode): string = of bmRetained: "retained" of bmDecay: "decay" +proc bbGainsString*(cands: seq[float]): string = + ## The resolved candidate set as the env's comma-separated form (boot report). + for i, c in cands: + if i > 0: result.add "," + result.add $c + +proc parseGains*(value: string): seq[float] = + ## Parse `TR_BITBRAIN_GAINS`. Empty / unparsable / out-of-range / duplicate + ## input cannot silently select a different regime: it falls back to the + ## shipped `BB_CAND` set, exactly like the other env knobs fall back to their + ## defaults. Values are clamped to [0, 8] (0 == HeadOn, 1 == Pattern) and + ## de-duplicated, then sorted so the argmax tie rule (keep the smaller + ## candidate) is unchanged. + var seen: seq[float] + for tok in value.split(','): + let t = tok.strip() + if t.len == 0: continue + var v: float + try: v = parseFloat(t) + except ValueError: continue + if v < 0.0 or v > 8.0: continue + var dup = false + for u in seen: + if abs(u - v) < 1e-9: dup = true + if not dup: seen.add v + if seen.len == 0: + for c in BB_CAND: seen.add c + return seen + seen.sort() + seen + proc parseMemMode*(value: string): BitMemMode = ## Empty / unknown values fall back to the shipped `perRound`, so a typo ## cannot silently select another regime. @@ -245,6 +294,11 @@ proc initBitBrainGun*(): BitBrainGun = result.decayFrac = clamp(envFloatBB(BB_DECAY_FRAC_ENV, BB_DECAY_FRAC_DEF), 0.0, 1.0) result.seed = int64(envIntBB(BB_SEED_ENV, BB_SEED_DEF)) result.resetOnTarget = envBoolBB(BB_RESET_ON_TARGET_ENV, BB_RESET_ON_TARGET_DEF) + result.cands = parseGains(getEnv(BB_GAINS_ENV, "")) + result.bandN = newSeq[float64](BB_NBANDS) + result.bandHits = newSeq[seq[float64]](BB_NBANDS) + for b in 0 ..< BB_NBANDS: + result.bandHits[b] = newSeq[float64](result.cands.len) result.lastTick = -1 result.lastEnqTick = -1 result.lastEnqBucket = -1 @@ -263,8 +317,8 @@ proc ensureInit*(g: var BitBrainGun) = proc bbAccumulate(g: var BitBrainGun, leadDeg, reqDeg, tolDeg: float, band: int) = ## Score every candidate gain on this resolved sample: a candidate "hits" when ## it would have put the aim within the target's angular half-width. - for ci in 0 ..< BB_NCAND: - if abs(BB_CAND[ci] * leadDeg - reqDeg) <= tolDeg: + for ci in 0 ..< g.cands.len: + if abs(g.cands[ci] * leadDeg - reqDeg) <= tolDeg: g.bandHits[band][ci] += 1.0 g.bandN[band] += 1.0 inc g.trained @@ -275,7 +329,7 @@ proc bbApplyDecay(g: var BitBrainGun) = let f = 1.0 - g.decayFrac if f >= 1.0: return for b in 0 ..< BB_NBANDS: - for ci in 0 ..< BB_NCAND: + for ci in 0 ..< g.cands.len: g.bandHits[b][ci] *= f g.bandN[b] *= f inc g.decays @@ -286,15 +340,20 @@ proc bbGain(g: BitBrainGun, band: int): float = ## conservative choice for the long-range regime this corrector targets. ## Returns 1.0 (Pattern) below the range gate or when the band is cold. if band < BB_GAIN_BAND_MIN: return 1.0 + if g.cands.len == 0: return 1.0 + # A single candidate is a FIXED gain: apply it from the first shot, never + # consult the counts. This is the no-learning arm of the live sweep. + if g.cands.len == 1: return g.cands[0] if g.bandN[band] < float(g.minObs): return 1.0 - var best = 4 # gain 1.0 + var best = -1 var bestRate = -1.0 - for ci in 0 ..< BB_NCAND: + for ci in 0 ..< g.cands.len: let rate = g.bandHits[band][ci] / g.bandN[band] if rate > bestRate: bestRate = rate best = ci - BB_CAND[best] + if best < 0: return 1.0 + g.cands[best] # ── deferred-label resolution (prequential learning) ───────────────────────── @@ -324,18 +383,22 @@ proc resolvePending(g: var BitBrainGun, state: WorldState) = # ── logging ────────────────────────────────────────────────────────────────── -proc bbLog(g: var BitBrainGun, state: WorldState, band: int, gain: float) = +proc bbLog(g: var BitBrainGun, state: WorldState, band: int, gain, leadDeg: float) = ## ONE change-gated `[bb]` line (behind TR_BITBRAIN_LOG=1) so a user tailing - ## the GUI log sees the gain the corrector is applying. + ## the GUI log sees the gain the corrector is applying. The APPLIED gain and + ## the resulting angular `shift` are both on the line: the boot report proves + ## the knob reached the process, this proves the gun actually used it. if not g.logEnabled: return + let shiftDeg = (gain - 1.0) * leadDeg let key = fmt"{gain:.2f}|{band}" if key == g.lastLogKey: return g.lastLogKey = key var rate = 0.0 - for ci in 0 ..< BB_NCAND: - if abs(BB_CAND[ci] - gain) < 1e-9: rate = g.bandHits[band][ci] / max(1.0, g.bandN[band]) + for ci in 0 ..< g.cands.len: + if abs(g.cands[ci] - gain) < 1e-9: rate = g.bandHits[band][ci] / max(1.0, g.bandN[band]) echo fmt"[bb] t={state.tick} band={BB_BAND_LO[band]:.0f}+ gain={gain:.2f} " & - fmt"rate={rate:.3f} n={g.bandN[band]:.0f} trained={g.trained} " & + fmt"shift={shiftDeg:+.2f}deg rate={rate:.3f} n={g.bandN[band]:.0f} " & + fmt"ncand={g.cands.len} trained={g.trained} " & fmt"pend={g.pendingCount} dropped={g.pendingDropped} mode={memModeName(g.memMode)}" # ── reset hooks (mirroring TmHorizonGun) ───────────────────────────────────── @@ -359,7 +422,7 @@ proc resetLearning*(g: var BitBrainGun, reason = "") = ## PER-BATTLE / PER-ENEMY wipe: gain counts, counters and the round state. if not g.initialized: return for b in 0 ..< BB_NBANDS: - for ci in 0 ..< BB_NCAND: g.bandHits[b][ci] = 0.0 + for ci in 0 ..< g.cands.len: g.bandHits[b][ci] = 0.0 g.bandN[b] = 0.0 g.lastGain[b] = 1.0 g.trained = 0 @@ -437,7 +500,7 @@ proc predict*(g: var BitBrainGun, state: WorldState, g.lastGain[band] = gain if abs(gain - 1.0) < 1e-9: return base inc g.corrections - g.bbLog(state, band, gain) + g.bbLog(state, band, gain, radToDeg(lead)) tmhApplyShift(state.selfX, state.selfY, base.x, base.y, radToDeg((gain - 1.0) * lead)) diff --git a/tools/ab/arms_leadgain.txt b/tools/ab/arms_leadgain.txt new file mode 100644 index 0000000..9dfb883 --- /dev/null +++ b/tools/ab/arms_leadgain.txt @@ -0,0 +1,22 @@ +# Phase 2 live A/B: does a lead gain ABOVE 1.0 beat shipped Pattern? +# +# One frozen binary (git archive HEAD), real DrussGT, 6 arms x 7 runs x 7 rounds. +# Every BitBrain arm swaps the admitted rack gun (Pattern off, BitBrain on) so the +# ONLY thing that changes is the lead gain: BitBrain's base prediction IS Pattern +# (`tmh.pattern.predict`), scaled over the line of sight by `gain`. The correction +# is applied only in the long bands (range >= 300 px, BB_GAIN_BAND_MIN), i.e. the +# region never tested live. TR_BITBRAIN_LOG=1 puts the APPLIED gain on the [bb] +# line; liveness is read from the boot env report (raw `[env] VAR=VALUE`). +# +# control = shipped Pattern-only rack, no env (reference) +# g100 = single candidate 1.0 -> FIXED gain, identity: MUST match control +# glo = learner restricted to <= 1 (expected HARMFUL per the live HeadOn kill) +# ghi = learner allowed above 1 (THE HYPOTHESIS) +# gfix150 = fixed 1.5, no learning +# gfix125 = fixed 1.25, no learning +control | +g100 | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.0 TR_BITBRAIN_LOG=1 | fixed gain 1.0 (identity/validity) +glo | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=0.25,0.5,0.75,1.0 TR_BITBRAIN_LOG=1 | learner restricted to <= 1 +ghi | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.0,1.25,1.5,2.0 TR_BITBRAIN_LOG=1 | learner allowed above 1 (HYPOTHESIS) +gfix150 | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.5 TR_BITBRAIN_LOG=1 | fixed gain 1.5, no learning +gfix125 | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.25 TR_BITBRAIN_LOG=1 | fixed gain 1.25, no learning