BitBrain: TR_BITBRAIN_GAINS env knob (candidate set + fixed-gain degenerate)

Task A of campaign phase 2: the lead-gain candidate set is now pure env, so the
live arms need no recompile.

- common_libs/guns/bitbrain_gun.nim: BB_GAINS_ENV (TR_BITBRAIN_GAINS); the
  candidate list is parsed once at gun construction into a dynamic seq, so the
  hit counts/hit rates are sized to it. Unset/unparsable -> the shipped
  BB_CAND set [0,0.25,0.5,0.75,1.0] (byte-identical behaviour). Exactly ONE
  candidate degenerates to a FIXED gain applied from the first shot (learning
  bypassed), still gated to the long bands. parseGains clamps to [0,8],
  de-dupes and sorts so the argmax tie rule is unchanged. The [bb] line now
  prints the APPLIED gain AND the resulting angular shift, so a run's
  correction is auditable from stdout.
- ModularBot_garage/src/env_report.nim: emit TR_BITBRAIN_GAINS (resolved
  candidate set) and add BB_GAINS_ENV to the known-name list.
- tools/ab/arms_leadgain.txt: the 6-arm phase-2 sweep definition.
This commit is contained in:
2026-09-25 00:15:27 +02:00
parent c305ef4212
commit 2747ebd323
3 changed files with 106 additions and 19 deletions
+3 -1
View File
@@ -289,6 +289,8 @@ proc printEffectiveValues(ctx: EnvReportContext) =
# `predict`, which is why `[rack]` membership is the actual enable switch.
emit("TR_BITBRAIN_MEM", memModeName(ctx.bitbrain.memMode),
sourceOf(BB_MEM_ENV))
emit("TR_BITBRAIN_GAINS", bbGainsString(ctx.bitbrain.cands),
sourceOf(BB_GAINS_ENV))
emit("TR_BITBRAIN_N", $ctx.bitbrain.nClasses, sourceOf(BB_N_ENV))
emit("TR_BITBRAIN_NADE", $ctx.bitbrain.nAde, sourceOf(BB_NADE_ENV))
emit("TR_BITBRAIN_RANGE", $ctx.bitbrain.maxDeg, sourceOf(BB_RANGE_ENV))
@@ -371,7 +373,7 @@ proc knownEnvNames*(): seq[string] =
TMH_SHIFT_ENV, TMH_BIG_MULT_ENV, TMH_LOG_ENV, TMH_RESET_ON_TARGET_ENV,
TMH_WINDOW_ENV, TMH_RESET_DROP_ENV, TMH_NSTATES_ENV, TMH_ACCURVE_ENV,
TMH_RETRAIN_EVERY_ENV, TMH_EPOCHS_ENV,
BB_MEM_ENV, BB_N_ENV, BB_NADE_ENV, BB_RANGE_ENV, BB_LOG_ENV,
BB_MEM_ENV, BB_GAINS_ENV, BB_N_ENV, BB_NADE_ENV, BB_RANGE_ENV, BB_LOG_ENV,
BB_MIN_OBS_ENV, BB_WARMUP_ENV, BB_ADAPT_ENV, BB_CALIB_ENV, BB_DECAY_ENV,
BB_DECAY_FRAC_ENV, BB_SEED_ENV, BB_RESET_ON_TARGET_ENV,
]
+81 -18
View File
@@ -60,11 +60,23 @@
## `common_libs/bitbrain/` library is untouched and still tested by
## `test_bitbrain.nim`.
##
## ── TR_BITBRAIN_GAINS (the candidate set as an env knob) ─────────────────────
## `TR_BITBRAIN_GAINS` is a comma-separated candidate list, e.g.
## `TR_BITBRAIN_GAINS=1.0,1.25,1.5,2.0`. It replaces the fixed shipped candidate
## set `{0, 0.25, 0.5, 0.75, 1.0}` for this gun instance, so every live arm is
## pure-env (no recompile). Two degenerate cases are deliberate:
## * unset / unparsable -> the shipped `BB_CAND` set, byte-identical behaviour;
## * exactly ONE value -> a FIXED gain, applied from the first shot with NO
## learning at all (the learner is bypassed), still gated to the long bands.
## The applied `gain` (and the resulting angular `shift`) is printed on the
## existing change-gated `[bb]` line, so a run's liveness AND the correction it
## actually applied are both auditable from the bot's stdout.
##
## DEFAULT OFF / PARITY: this gun is admitted ONLY when `TR_RACK_BITBRAIN` says so
## (default `off`). The shipped rack never calls `predict`, so `ensureInit` never
## runs and the shipped bot is byte-for-byte unchanged.
import std/[math, os, strutils, strformat]
import std/[math, os, strutils, strformat, algorithm]
import gun_harness/gun_interface
import guns/tm_horizon
import guns/pattern_matcher
@@ -72,6 +84,7 @@ import guns/pattern_matcher
const
## ── env knobs (all resolved once at gun construction) ─────────────────────
BB_MEM_ENV* = "TR_BITBRAIN_MEM" ## perRound|retained|decay
BB_GAINS_ENV* = "TR_BITBRAIN_GAINS" ## comma-separated candidate gains
BB_N_ENV* = "TR_BITBRAIN_N" ## (legacy geometry; inert)
BB_NADE_ENV* = "TR_BITBRAIN_NADE" ## (legacy ADE count; inert)
BB_RANGE_ENV* = "TR_BITBRAIN_RANGE" ## (legacy class half-range; inert)
@@ -91,8 +104,9 @@ const
BB_NHB* = 4 ## horizon buckets (for the per-tick label dedupe)
BB_BAND_LO* = [0.0, 100.0, 200.0, 300.0, 450.0]
BB_BAND_HI* = [100.0, 200.0, 300.0, 450.0, 1.0e18]
## The candidate lead gains the band selector picks from. 0.0 == HeadOn (aim
## at the current position) and 1.0 == Pattern (use the full lead).
## The DEFAULT candidate lead gains the band selector picks from. 0.0 == HeadOn
## (aim at the current position) and 1.0 == Pattern (use the full lead).
## `TR_BITBRAIN_GAINS` replaces this set per gun; unset -> this exact set.
BB_CAND* = [0.0, 0.25, 0.50, 0.75, 1.0]
BB_NCAND* = 5
BB_BB_RADIUS* = 18.0 ## hit-detection radius in px (ruler tolerance)
@@ -146,9 +160,13 @@ type
decayFrac*: float
seed*: int64
resetOnTarget*: bool
# ── the candidate gain set (resolved once at construction) ────────────────
## Ascending; one entry == a FIXED gain with no learning. `bandHits` and
## `bandN` are sized to it, so the loops below never touch a stale column.
cands*: seq[float]
# ── gain learner: hit counts per (range band x candidate gain) ────────────
bandHits*: array[BB_NBANDS, array[BB_NCAND, float64]]
bandN*: array[BB_NBANDS, float64]
bandHits*: seq[seq[float64]]
bandN*: seq[float64]
trained*: int
sinceDecay: int
decays*: int
@@ -179,6 +197,37 @@ proc memModeName*(m: BitMemMode): string =
of bmRetained: "retained"
of bmDecay: "decay"
proc bbGainsString*(cands: seq[float]): string =
## The resolved candidate set as the env's comma-separated form (boot report).
for i, c in cands:
if i > 0: result.add ","
result.add $c
proc parseGains*(value: string): seq[float] =
## Parse `TR_BITBRAIN_GAINS`. Empty / unparsable / out-of-range / duplicate
## input cannot silently select a different regime: it falls back to the
## shipped `BB_CAND` set, exactly like the other env knobs fall back to their
## defaults. Values are clamped to [0, 8] (0 == HeadOn, 1 == Pattern) and
## de-duplicated, then sorted so the argmax tie rule (keep the smaller
## candidate) is unchanged.
var seen: seq[float]
for tok in value.split(','):
let t = tok.strip()
if t.len == 0: continue
var v: float
try: v = parseFloat(t)
except ValueError: continue
if v < 0.0 or v > 8.0: continue
var dup = false
for u in seen:
if abs(u - v) < 1e-9: dup = true
if not dup: seen.add v
if seen.len == 0:
for c in BB_CAND: seen.add c
return seen
seen.sort()
seen
proc parseMemMode*(value: string): BitMemMode =
## Empty / unknown values fall back to the shipped `perRound`, so a typo
## cannot silently select another regime.
@@ -245,6 +294,11 @@ proc initBitBrainGun*(): BitBrainGun =
result.decayFrac = clamp(envFloatBB(BB_DECAY_FRAC_ENV, BB_DECAY_FRAC_DEF), 0.0, 1.0)
result.seed = int64(envIntBB(BB_SEED_ENV, BB_SEED_DEF))
result.resetOnTarget = envBoolBB(BB_RESET_ON_TARGET_ENV, BB_RESET_ON_TARGET_DEF)
result.cands = parseGains(getEnv(BB_GAINS_ENV, ""))
result.bandN = newSeq[float64](BB_NBANDS)
result.bandHits = newSeq[seq[float64]](BB_NBANDS)
for b in 0 ..< BB_NBANDS:
result.bandHits[b] = newSeq[float64](result.cands.len)
result.lastTick = -1
result.lastEnqTick = -1
result.lastEnqBucket = -1
@@ -263,8 +317,8 @@ proc ensureInit*(g: var BitBrainGun) =
proc bbAccumulate(g: var BitBrainGun, leadDeg, reqDeg, tolDeg: float, band: int) =
## Score every candidate gain on this resolved sample: a candidate "hits" when
## it would have put the aim within the target's angular half-width.
for ci in 0 ..< BB_NCAND:
if abs(BB_CAND[ci] * leadDeg - reqDeg) <= tolDeg:
for ci in 0 ..< g.cands.len:
if abs(g.cands[ci] * leadDeg - reqDeg) <= tolDeg:
g.bandHits[band][ci] += 1.0
g.bandN[band] += 1.0
inc g.trained
@@ -275,7 +329,7 @@ proc bbApplyDecay(g: var BitBrainGun) =
let f = 1.0 - g.decayFrac
if f >= 1.0: return
for b in 0 ..< BB_NBANDS:
for ci in 0 ..< BB_NCAND:
for ci in 0 ..< g.cands.len:
g.bandHits[b][ci] *= f
g.bandN[b] *= f
inc g.decays
@@ -286,15 +340,20 @@ proc bbGain(g: BitBrainGun, band: int): float =
## conservative choice for the long-range regime this corrector targets.
## Returns 1.0 (Pattern) below the range gate or when the band is cold.
if band < BB_GAIN_BAND_MIN: return 1.0
if g.cands.len == 0: return 1.0
# A single candidate is a FIXED gain: apply it from the first shot, never
# consult the counts. This is the no-learning arm of the live sweep.
if g.cands.len == 1: return g.cands[0]
if g.bandN[band] < float(g.minObs): return 1.0
var best = 4 # gain 1.0
var best = -1
var bestRate = -1.0
for ci in 0 ..< BB_NCAND:
for ci in 0 ..< g.cands.len:
let rate = g.bandHits[band][ci] / g.bandN[band]
if rate > bestRate:
bestRate = rate
best = ci
BB_CAND[best]
if best < 0: return 1.0
g.cands[best]
# ── deferred-label resolution (prequential learning) ─────────────────────────
@@ -324,18 +383,22 @@ proc resolvePending(g: var BitBrainGun, state: WorldState) =
# ── logging ──────────────────────────────────────────────────────────────────
proc bbLog(g: var BitBrainGun, state: WorldState, band: int, gain: float) =
proc bbLog(g: var BitBrainGun, state: WorldState, band: int, gain, leadDeg: float) =
## ONE change-gated `[bb]` line (behind TR_BITBRAIN_LOG=1) so a user tailing
## the GUI log sees the gain the corrector is applying.
## the GUI log sees the gain the corrector is applying. The APPLIED gain and
## the resulting angular `shift` are both on the line: the boot report proves
## the knob reached the process, this proves the gun actually used it.
if not g.logEnabled: return
let shiftDeg = (gain - 1.0) * leadDeg
let key = fmt"{gain:.2f}|{band}"
if key == g.lastLogKey: return
g.lastLogKey = key
var rate = 0.0
for ci in 0 ..< BB_NCAND:
if abs(BB_CAND[ci] - gain) < 1e-9: rate = g.bandHits[band][ci] / max(1.0, g.bandN[band])
for ci in 0 ..< g.cands.len:
if abs(g.cands[ci] - gain) < 1e-9: rate = g.bandHits[band][ci] / max(1.0, g.bandN[band])
echo fmt"[bb] t={state.tick} band={BB_BAND_LO[band]:.0f}+ gain={gain:.2f} " &
fmt"rate={rate:.3f} n={g.bandN[band]:.0f} trained={g.trained} " &
fmt"shift={shiftDeg:+.2f}deg rate={rate:.3f} n={g.bandN[band]:.0f} " &
fmt"ncand={g.cands.len} trained={g.trained} " &
fmt"pend={g.pendingCount} dropped={g.pendingDropped} mode={memModeName(g.memMode)}"
# ── reset hooks (mirroring TmHorizonGun) ─────────────────────────────────────
@@ -359,7 +422,7 @@ proc resetLearning*(g: var BitBrainGun, reason = "") =
## PER-BATTLE / PER-ENEMY wipe: gain counts, counters and the round state.
if not g.initialized: return
for b in 0 ..< BB_NBANDS:
for ci in 0 ..< BB_NCAND: g.bandHits[b][ci] = 0.0
for ci in 0 ..< g.cands.len: g.bandHits[b][ci] = 0.0
g.bandN[b] = 0.0
g.lastGain[b] = 1.0
g.trained = 0
@@ -437,7 +500,7 @@ proc predict*(g: var BitBrainGun, state: WorldState,
g.lastGain[band] = gain
if abs(gain - 1.0) < 1e-9: return base
inc g.corrections
g.bbLog(state, band, gain)
g.bbLog(state, band, gain, radToDeg(lead))
tmhApplyShift(state.selfX, state.selfY, base.x, base.y,
radToDeg((gain - 1.0) * lead))
+22
View File
@@ -0,0 +1,22 @@
# Phase 2 live A/B: does a lead gain ABOVE 1.0 beat shipped Pattern?
#
# One frozen binary (git archive HEAD), real DrussGT, 6 arms x 7 runs x 7 rounds.
# Every BitBrain arm swaps the admitted rack gun (Pattern off, BitBrain on) so the
# ONLY thing that changes is the lead gain: BitBrain's base prediction IS Pattern
# (`tmh.pattern.predict`), scaled over the line of sight by `gain`. The correction
# is applied only in the long bands (range >= 300 px, BB_GAIN_BAND_MIN), i.e. the
# region never tested live. TR_BITBRAIN_LOG=1 puts the APPLIED gain on the [bb]
# line; liveness is read from the boot env report (raw `[env] VAR=VALUE`).
#
# control = shipped Pattern-only rack, no env (reference)
# g100 = single candidate 1.0 -> FIXED gain, identity: MUST match control
# glo = learner restricted to <= 1 (expected HARMFUL per the live HeadOn kill)
# ghi = learner allowed above 1 (THE HYPOTHESIS)
# gfix150 = fixed 1.5, no learning
# gfix125 = fixed 1.25, no learning
control |
g100 | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.0 TR_BITBRAIN_LOG=1 | fixed gain 1.0 (identity/validity)
glo | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=0.25,0.5,0.75,1.0 TR_BITBRAIN_LOG=1 | learner restricted to <= 1
ghi | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.0,1.25,1.5,2.0 TR_BITBRAIN_LOG=1 | learner allowed above 1 (HYPOTHESIS)
gfix150 | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.5 TR_BITBRAIN_LOG=1 | fixed gain 1.5, no learning
gfix125 | TR_RACK_PATTERN=off TR_RACK_BITBRAIN=both TR_BITBRAIN_GAINS=1.25 TR_BITBRAIN_LOG=1 | fixed gain 1.25, no learning