## bitbrain_gun.nim — BitBrain (id 16), REBUILT as a LEAD-GAIN CORRECTOR. ## ## ── WHY THIS FILE WAS REWRITTEN (Phase 0/1 evidence) ────────────────────────── ## The previous design was an ADDITIVE angular shift: an ADE+SBC network ## classified the +h-tick angular error over ±`TR_BITBRAIN_RANGE` degrees and ## added the argmax class centre to Pattern's bearing. Phase 0 measured it as ## statistically identical to Pattern (`docs/bitbrain_gun_verdict.md`, ## commit d93ce44) and as carrying no measurable aim information ## (450+: 16.200 deg vs Pattern's 16.193; `docs/bitbrain_campaign.md` §0.3.5). ## ## Phase 1 measured the actual lever. The gain sweep found that gains >= 1 are ## strictly worse at every band and that the optimal gain is BELOW 1.0 at long ## range (450+: ~0.25). A fractional gain leaves the Pearson lead *correlation* ## unchanged (correlation is invariant under positive scaling), so a smaller ## gain does not add information — it shrinks the magnitude of an uninformative ## Pattern lead toward the low-variance static (HeadOn) aim. The right output is ## therefore a multiplicative GAIN on Pattern's lead, not a class-based additive ## shift. See `docs/bitbrain_campaign.md` §Phase 1 for the measured curve. ## ## ── THE DESIGN ──────────────────────────────────────────────────────────────── ## * BASE — the shipped Pattern gun's prediction (`guns/pattern_matcher`), ## reached through the TmHorizonGun observation ring. ## * OUTPUT — `aim = LOS + gain * (patternAim - LOS)`, i.e. Pattern's lead over ## the line of sight is multiplied by a learned `gain` (one of ## `BB_CAND`, so it may be BELOW 1.0 — the point). ## * LABEL — the same deferred-label path the old corrector used: at fire ## time we remember the base lead and the aim tolerance; `h = ## round(dist/speed)` ticks later `tmhObservedAt` returns the ## enemy's OBSERVED bearing from the firing position. `requiredLead ## = observedBearing - LOS` and `baseLead = baseBearing - LOS`, so a ## candidate gain scores a hit on this sample when ## `|gain*baseLead - requiredLead| <= tolerance`. ## * TRAIN — ONLINE / PREQUENTIAL per range band: for each candidate gain we ## count the fraction of resolved samples that would have been ## within the target's angular half-width (`atan(18/range)`, the ## SAME tolerance the offline ruler uses). The band's gain is the ## argmax hit rate. THIS is the key lesson of Phase 1: the ## least-squares gain and the hit-probability-optimal gain DIVERGE ## (Pattern's lead errors are bimodal), so the learner optimises the ## hit-probability proxy directly instead of mean squared error. ## * STATE — the range band (the ruler's 5 bands). Range is known causally at ## fire time, so a per-band gain table is shippable with no learning ## at all; BitBrain learns that table online. The correction is ## additionally gated to bands with range >= 300 px ## (`BB_GAIN_BAND_MIN`), where Phase 1 measured Pattern's lead to be ## uninformative. That gate is causal (range is known). ## ## The gain statistics are battle-scale: a round boundary wipes the observation ## ring and deferred labels but NOT the gain counts (a new round is not a new ## enemy). `resetLearning` wipes them on a new battle / target change; with ## `TR_BITBRAIN_MEM=decay` every `TR_BITBRAIN_DECAY` resolved samples decays the ## counts by `TR_BITBRAIN_DECAY_FRAC` toward the gain-1.0 column. ## ## ── WHAT IS STILL HERE ONLY FOR THE BOOT REPORT / GUARD TESTS ───────────────── ## The ADE+SBC network is GONE from the gun. The 53-bit TMH input, the class ## geometry (`bbCenterDeg`/`bbClassOf`), `TR_BITBRAIN_N`/`NADE`/`WARMUP`/`ADAPT`/ ## `CALIB`/`SEED` and `TR_BITBRAIN_RANGE` are retained as resolved configuration ## so the boot report (`env_report.nim`) and the registration guard tests keep ## working unchanged; they no longer affect the gain learner. The generic ## `common_libs/bitbrain/` library is untouched and still tested by ## `test_bitbrain.nim`. ## ## ── TR_BITBRAIN_GAINS (the candidate set as an env knob) ───────────────────── ## `TR_BITBRAIN_GAINS` is a comma-separated candidate list, e.g. ## `TR_BITBRAIN_GAINS=1.0,1.25,1.5,2.0`. It replaces the fixed shipped candidate ## set `{0, 0.25, 0.5, 0.75, 1.0}` for this gun instance, so every live arm is ## pure-env (no recompile). Two degenerate cases are deliberate: ## * unset / unparsable -> the shipped `BB_CAND` set, byte-identical behaviour; ## * exactly ONE value -> a FIXED gain, applied from the first shot with NO ## learning at all (the learner is bypassed), still gated to the long bands. ## The applied `gain` (and the resulting angular `shift`) is printed on the ## existing change-gated `[bb]` line, so a run's liveness AND the correction it ## actually applied are both auditable from the bot's stdout. ## ## DEFAULT OFF / PARITY: this gun is admitted ONLY when `TR_RACK_BITBRAIN` says so ## (default `off`). The shipped rack never calls `predict`, so `ensureInit` never ## runs and the shipped bot is byte-for-byte unchanged. import std/[math, os, strutils, strformat, algorithm] import gun_harness/gun_interface import guns/tm_horizon import guns/pattern_matcher const ## ── env knobs (all resolved once at gun construction) ───────────────────── BB_MEM_ENV* = "TR_BITBRAIN_MEM" ## perRound|retained|decay BB_GAINS_ENV* = "TR_BITBRAIN_GAINS" ## comma-separated candidate gains BB_N_ENV* = "TR_BITBRAIN_N" ## (legacy geometry; inert) BB_NADE_ENV* = "TR_BITBRAIN_NADE" ## (legacy ADE count; inert) BB_RANGE_ENV* = "TR_BITBRAIN_RANGE" ## (legacy class half-range; inert) BB_LOG_ENV* = "TR_BITBRAIN_LOG" ## 1 = per-change [bb] log BB_MIN_OBS_ENV* = "TR_BITBRAIN_MIN_OBS" ## samples before a band is trusted BB_WARMUP_ENV* = "TR_BITBRAIN_WARMUP" ## (legacy; inert) BB_ADAPT_ENV* = "TR_BITBRAIN_ADAPT" ## (legacy; inert) BB_CALIB_ENV* = "TR_BITBRAIN_CALIB" ## (legacy; inert) BB_DECAY_ENV* = "TR_BITBRAIN_DECAY" ## decay interval (samples) BB_DECAY_FRAC_ENV* = "TR_BITBRAIN_DECAY_FRAC" ## per-decay count shrink BB_SEED_ENV* = "TR_BITBRAIN_SEED" ## (legacy; inert) BB_RESET_ON_TARGET_ENV* = "TR_BITBRAIN_RESET_ON_TARGET" ## ── fixed geometry ──────────────────────────────────────────────────────── BB_PENDING_CAP* = 512 ## deferred-label queue (>= 4 buckets x 50 ticks) ## ── the gain learner ────────────────────────────────────────────────────── BB_NBANDS* = 5 ## the ruler's range bands BB_NHB* = 4 ## horizon buckets (for the per-tick label dedupe) BB_BAND_LO* = [0.0, 100.0, 200.0, 300.0, 450.0] BB_BAND_HI* = [100.0, 200.0, 300.0, 450.0, 1.0e18] ## The DEFAULT candidate lead gains the band selector picks from. 0.0 == HeadOn ## (aim at the current position) and 1.0 == Pattern (use the full lead). ## `TR_BITBRAIN_GAINS` replaces this set per gun; unset -> this exact set. BB_CAND* = [0.0, 0.25, 0.50, 0.75, 1.0] BB_NCAND* = 5 BB_BB_RADIUS* = 18.0 ## hit-detection radius in px (ruler tolerance) ## Apply the correction only from this band up (range >= BB_BAND_LO[3] = 300). ## [MEASURED] below 300 Pattern's lead is informative and shrinking it loses ## hits; see the header note. BB_GAIN_BAND_MIN* = 3 ## ── shipped defaults ────────────────────────────────────────────────────── BB_N_DEF = 32 BB_NADE_DEF = 256 BB_RANGE_DEF = 40.0 BB_MIN_OBS_DEF = 8 BB_WARMUP_DEF = 400 BB_ADAPT_DEF = 32 BB_CALIB_DEF = 512 BB_DECAY_DEF = 250 BB_DECAY_FRAC_DEF = 0.02 BB_SEED_DEF = 20240921 BB_RESET_ON_TARGET_DEF = true type BitMemMode* = enum bmPerRound, bmRetained, bmDecay BbPending = object ## One deferred training sample. `lead` is Pattern's lead over LOS at fire ## time (radians) and `tol` the target's angular half-width then; the label ## is resolved `horizon` ticks later. fireTick: int horizon: int band: int selfX*, selfY: float baseBearing: float lead: float tolDeg: float BitBrainGun* = object tmh: TmHorizonGun initialized: bool # ── resolved config (kept in the boot report) ───────────────────────────── nClasses*: int maxDeg*: float nAde*: int memMode*: BitMemMode logEnabled*: bool minObs*: int warmupN*: int adaptEvery*: int calibEvery*: int decayEvery*: int decayFrac*: float seed*: int64 resetOnTarget*: bool # ── the candidate gain set (resolved once at construction) ──────────────── ## Ascending; one entry == a FIXED gain with no learning. `bandHits` and ## `bandN` are sized to it, so the loops below never touch a stale column. cands*: seq[float] # ── gain learner: hit counts per (range band x candidate gain) ──────────── bandHits*: seq[seq[float64]] bandN*: seq[float64] trained*: int sinceDecay: int decays*: int # ── readout / accounting ────────────────────────────────────────────────── lastGain*: array[BB_NBANDS, float] corrections*: int lastLogKey: string # ── deferred labels ─────────────────────────────────────────────────────── pending: array[BB_PENDING_CAP, BbPending] pendingCount*: int pendingDropped*: int # ── per-tick caches ─────────────────────────────────────────────────────── lastTick: int lastEnqTick: int lastEnqBucket: int observedTargetId*: int # ── small pure helpers ─────────────────────────────────────────────────────── proc wrapRadBB(r: float): float {.inline.} = result = r while result > PI: result -= 2.0 * PI while result < -PI: result += 2.0 * PI proc memModeName*(m: BitMemMode): string = case m of bmPerRound: "perRound" of bmRetained: "retained" of bmDecay: "decay" proc bbGainsString*(cands: seq[float]): string = ## The resolved candidate set as the env's comma-separated form (boot report). for i, c in cands: if i > 0: result.add "," result.add $c proc parseGains*(value: string): seq[float] = ## Parse `TR_BITBRAIN_GAINS`. Empty / unparsable / out-of-range / duplicate ## input cannot silently select a different regime: it falls back to the ## shipped `BB_CAND` set, exactly like the other env knobs fall back to their ## defaults. Values are clamped to [0, 8] (0 == HeadOn, 1 == Pattern) and ## de-duplicated, then sorted so the argmax tie rule (keep the smaller ## candidate) is unchanged. var seen: seq[float] for tok in value.split(','): let t = tok.strip() if t.len == 0: continue var v: float try: v = parseFloat(t) except ValueError: continue if v < 0.0 or v > 8.0: continue var dup = false for u in seen: if abs(u - v) < 1e-9: dup = true if not dup: seen.add v if seen.len == 0: for c in BB_CAND: seen.add c return seen seen.sort() seen proc parseMemMode*(value: string): BitMemMode = ## Empty / unknown values fall back to the shipped `perRound`, so a typo ## cannot silently select another regime. case value.strip().toLowerAscii() of "retained", "retain", "accum", "accumulate": bmRetained of "decay", "forget", "age": bmDecay else: bmPerRound proc envFloatBB(name: string, default: float): float = let v = getEnv(name, "") if v.len == 0: return default try: parseFloat(v.strip()) except ValueError: default proc envIntBB(name: string, default: int): int = let v = getEnv(name, "") if v.len == 0: return default try: parseInt(v.strip()) except ValueError: default proc envBoolBB(name: string, default: bool): bool = case getEnv(name, "").strip().toLowerAscii() of "1", "true", "yes", "on": true of "0", "false", "no", "off": false else: default proc bbCenterDeg*(k, nClasses: int, maxDeg: float): float = ## Centre (degrees) of correction class `k` over ±maxDeg. Retained for the ## registration guard test and the boot report; inert for the gain learner. let w = 2.0 * maxDeg / float(nClasses) -maxDeg + (float(k) + 0.5) * w proc bbClassOf*(errRad: float, nClasses: int, maxDeg: float): int = ## Bin a signed angular error (radians) into one of `nClasses` bins over ## [−maxDeg, +maxDeg]. Retained for the registration guard test; inert. let x = radToDeg(errRad) var k = int((x + maxDeg) / (2.0 * maxDeg) * float(nClasses)) if k < 0: k = 0 if k >= nClasses: k = nClasses - 1 k proc bbBandOf*(range: float): int {.inline.} = ## Range band (the ruler's bands), known causally at fire time. for b in 0 ..< BB_NBANDS: if range >= BB_BAND_LO[b] and range < BB_BAND_HI[b]: return b BB_NBANDS - 1 proc bbTolDeg*(range: float): float {.inline.} = ## The target's angular half-width at `range` — atan(18/range) — i.e. the exact ## tolerance the offline ruler uses for its hit-probability proxy. radToDeg(arctan2(BB_BB_RADIUS, max(range, 1e-9))) # ── construction / lazy init ───────────────────────────────────────────────── proc initBitBrainGun*(): BitBrainGun = result.nClasses = clamp(envIntBB(BB_N_ENV, BB_N_DEF), 2, 512) result.nAde = clamp(envIntBB(BB_NADE_ENV, BB_NADE_DEF), 8, 4096) result.maxDeg = clamp(envFloatBB(BB_RANGE_ENV, BB_RANGE_DEF), 1.0, 180.0) result.memMode = parseMemMode(getEnv(BB_MEM_ENV, "")) result.logEnabled = envBoolBB(BB_LOG_ENV, false) result.minObs = max(1, envIntBB(BB_MIN_OBS_ENV, BB_MIN_OBS_DEF)) result.warmupN = max(0, envIntBB(BB_WARMUP_ENV, BB_WARMUP_DEF)) result.adaptEvery = max(1, envIntBB(BB_ADAPT_ENV, BB_ADAPT_DEF)) result.calibEvery = max(1, envIntBB(BB_CALIB_ENV, BB_CALIB_DEF)) result.decayEvery = max(1, envIntBB(BB_DECAY_ENV, BB_DECAY_DEF)) result.decayFrac = clamp(envFloatBB(BB_DECAY_FRAC_ENV, BB_DECAY_FRAC_DEF), 0.0, 1.0) result.seed = int64(envIntBB(BB_SEED_ENV, BB_SEED_DEF)) result.resetOnTarget = envBoolBB(BB_RESET_ON_TARGET_ENV, BB_RESET_ON_TARGET_DEF) result.cands = parseGains(getEnv(BB_GAINS_ENV, "")) result.bandN = newSeq[float64](BB_NBANDS) result.bandHits = newSeq[seq[float64]](BB_NBANDS) for b in 0 ..< BB_NBANDS: result.bandHits[b] = newSeq[float64](result.cands.len) result.lastTick = -1 result.lastEnqTick = -1 result.lastEnqBucket = -1 result.observedTargetId = -1 for b in 0 ..< BB_NBANDS: result.lastGain[b] = 1.0 proc ensureInit*(g: var BitBrainGun) = ## Build the observation ring on first use. No network, no global-RNG use, so ## the shipped default path is untouched and construction stays cheap. if g.initialized: return g.initialized = true g.tmh = initTmHorizonGun() # ── the gain learner ───────────────────────────────────────────────────────── proc bbAccumulate(g: var BitBrainGun, leadDeg, reqDeg, tolDeg: float, band: int) = ## Score every candidate gain on this resolved sample: a candidate "hits" when ## it would have put the aim within the target's angular half-width. for ci in 0 ..< g.cands.len: if abs(g.cands[ci] * leadDeg - reqDeg) <= tolDeg: g.bandHits[band][ci] += 1.0 g.bandN[band] += 1.0 inc g.trained proc bbApplyDecay(g: var BitBrainGun) = ## Forgetting for `TR_BITBRAIN_MEM=decay`: shrink the hit counts and, more ## strongly, pull them toward the gain-1.0 column so stale evidence ages out. let f = 1.0 - g.decayFrac if f >= 1.0: return for b in 0 ..< BB_NBANDS: for ci in 0 ..< g.cands.len: g.bandHits[b][ci] *= f g.bandN[b] *= f inc g.decays proc bbGain(g: BitBrainGun, band: int): float = ## The band's gain is the candidate with the highest observed hit rate. ## Ties keep the SMALLER candidate (the scan is ascending), which is the ## conservative choice for the long-range regime this corrector targets. ## Returns 1.0 (Pattern) below the range gate or when the band is cold. if band < BB_GAIN_BAND_MIN: return 1.0 if g.cands.len == 0: return 1.0 # A single candidate is a FIXED gain: apply it from the first shot, never # consult the counts. This is the no-learning arm of the live sweep. if g.cands.len == 1: return g.cands[0] if g.bandN[band] < float(g.minObs): return 1.0 var best = -1 var bestRate = -1.0 for ci in 0 ..< g.cands.len: let rate = g.bandHits[band][ci] / g.bandN[band] if rate > bestRate: bestRate = rate best = ci if best < 0: return 1.0 g.cands[best] # ── deferred-label resolution (prequential learning) ───────────────────────── proc resolvePending(g: var BitBrainGun, state: WorldState) = var w = 0 for i in 0 ..< g.pendingCount: let p = g.pending[i] let due = p.fireTick + p.horizon if due > state.tick: g.pending[w] = p inc w elif due == state.tick: let obs = tmhObservedAt(g.tmh, state.tick, p.selfX, p.selfY) if obs.ok and (state.tick - obs.lastSeenTick) <= TMH_STALE_MAX: let err = wrapRadBB(obs.bearing - p.baseBearing) let reqLead = wrapRadBB(err + p.lead) g.bbAccumulate(radToDeg(p.lead), radToDeg(reqLead), p.tolDeg, p.band) inc g.sinceDecay if g.memMode == bmDecay and g.sinceDecay >= g.decayEvery: g.bbApplyDecay() g.sinceDecay = 0 else: inc g.pendingDropped else: inc g.pendingDropped g.pendingCount = w # ── logging ────────────────────────────────────────────────────────────────── proc bbLog(g: var BitBrainGun, state: WorldState, band: int, gain, leadDeg: float) = ## ONE change-gated `[bb]` line (behind TR_BITBRAIN_LOG=1) so a user tailing ## the GUI log sees the gain the corrector is applying. The APPLIED gain and ## the resulting angular `shift` are both on the line: the boot report proves ## the knob reached the process, this proves the gun actually used it. if not g.logEnabled: return let shiftDeg = (gain - 1.0) * leadDeg let key = fmt"{gain:.2f}|{band}" if key == g.lastLogKey: return g.lastLogKey = key var rate = 0.0 for ci in 0 ..< g.cands.len: if abs(g.cands[ci] - gain) < 1e-9: rate = g.bandHits[band][ci] / max(1.0, g.bandN[band]) echo fmt"[bb] t={state.tick} band={BB_BAND_LO[band]:.0f}+ gain={gain:.2f} " & fmt"shift={shiftDeg:+.2f}deg rate={rate:.3f} n={g.bandN[band]:.0f} " & fmt"ncand={g.cands.len} trained={g.trained} " & fmt"pend={g.pendingCount} dropped={g.pendingDropped} mode={memModeName(g.memMode)}" # ── reset hooks (mirroring TmHorizonGun) ───────────────────────────────────── proc resetRound(g: var BitBrainGun) = ## PER-ROUND reset: observation ring, deferred labels and per-tick caches (the ## bots teleport between rounds). The gain counts are deliberately KEPT — they ## are battle-scale and a new round is not a new enemy. g.tmh.resetRoundState() g.pendingCount = 0 g.lastTick = -1 g.lastEnqTick = -1 g.lastEnqBucket = -1 g.lastLogKey = "" proc resetRoundState*(g: var BitBrainGun) = if not g.initialized: return g.resetRound() proc resetLearning*(g: var BitBrainGun, reason = "") = ## PER-BATTLE / PER-ENEMY wipe: gain counts, counters and the round state. if not g.initialized: return for b in 0 ..< BB_NBANDS: for ci in 0 ..< g.cands.len: g.bandHits[b][ci] = 0.0 g.bandN[b] = 0.0 g.lastGain[b] = 1.0 g.trained = 0 g.sinceDecay = 0 g.decays = 0 g.corrections = 0 g.observedTargetId = -1 g.resetRound() if reason.len > 0 and g.logEnabled: echo fmt"[bb-reset] reason={reason}" proc targetChanged*(g: var BitBrainGun, enemyId: int): bool = ## Per-ENEMY reset: wipe when the target changes to a different bot id. First ## acquisition never wipes, so the round-start pick does not cold-start us. if not g.resetOnTarget: return false if enemyId < 0: return false if g.observedTargetId >= 0 and enemyId != g.observedTargetId: g.resetLearning("target_change") g.observedTargetId = enemyId return true g.observedTargetId = enemyId false # ── Gun interface ──────────────────────────────────────────────────────────── proc isWarmedUp*(g: BitBrainGun): bool {.inline.} = true proc networkBytes*(g: BitBrainGun): int = ## No neural network is held any more; kept for the boot report / guard test. 0 proc predict*(g: var BitBrainGun, state: WorldState, bulletSpeed: float): GunPrediction = g.ensureInit() # Round boundary: a tick regression means a new round. if state.tick < g.lastTick: g.resetRound() # Once per tick: observe the world, then resolve any labels now due. if state.tick != g.lastTick: tmhUpdateHistory(g.tmh, state) g.resolvePending(state) g.lastTick = state.tick # The base prediction is Pattern; BitBrain only scales its lead over LOS. let base = g.tmh.pattern.predict(state, bulletSpeed) if bulletSpeed <= 0.0: return base let dist = hypot(state.enemyX - state.selfX, state.enemyY - state.selfY) let h = tmhHorizonFor(dist, bulletSpeed) let hb = tmhHorizonBucket(h) let band = bbBandOf(dist) let los = arctan2(state.enemyY - state.selfY, state.enemyX - state.selfX) let baseBearing = arctan2(base.y - state.selfY, base.x - state.selfX) let lead = wrapRadBB(baseBearing - los) # Enqueue one deferred sample per (tick, horizon bucket): `predict` runs once # per power bin, so all four horizons contribute evidence. if g.lastEnqTick != state.tick or g.lastEnqBucket != hb: if g.pendingCount < BB_PENDING_CAP: g.pending[g.pendingCount] = BbPending( fireTick: state.tick, horizon: h, band: band, selfX: state.selfX, selfY: state.selfY, baseBearing: baseBearing, lead: lead, tolDeg: bbTolDeg(dist)) inc g.pendingCount else: inc g.pendingDropped g.lastEnqTick = state.tick g.lastEnqBucket = hb # Readout: a fractional gain may be BELOW 1.0. When cold / gated out the # learner returns 1.0 and the base prediction is returned unchanged. let gain = g.bbGain(band) g.lastGain[band] = gain if abs(gain - 1.0) < 1e-9: return base inc g.corrections g.bbLog(state, band, gain, radToDeg(lead)) tmhApplyShift(state.selfX, state.selfY, base.x, base.y, radToDeg((gain - 1.0) * lead)) proc onResult*(g: var BitBrainGun, e: FeedbackEvent) = ## Labels come from our own observation ring, not from virtual-bullet ## feedback, so there is nothing to do here. The hook exists for the rack. discard