## lead_gain.nim — LEADGAIN (rack id 16): a per-range-band LEAD-GAIN corrector. ## ## ── THE NAME ────────────────────────────────────────────────────────────────── ## This gun used to be called `BITBRAIN` and to live in `guns/bitbrain_gun.nim`, ## but its ADE+SBC network was removed when it was rebuilt into what it actually ## is: **it learns a multiplier for Pattern's lead, separately per range band.** ## `LEADGAIN` says that; `BITBRAIN` (a neural network) did not. The rack id 16 ## is UNCHANGED (many tests assert the id literals) and the `BITBRAIN` name is ## the gun's to keep: the ADE+SBC gun that briefly held it (rack id 17, ## `guns/bitbrain_net.nim`) was RETIRED and removed — see ## `docs/bitbrain_campaign.md` §RETIRED. ## ## ── WHY THE FILE WAS REBUILT (Phase 0/1 evidence) ──────────────────────────── ## The previous design was an ADDITIVE angular shift: an ADE+SBC network ## classified the +h-tick angular error over ±`TR_LEADGAIN_RANGE` degrees and ## added the argmax class centre to Pattern's bearing. Phase 0 measured it as ## statistically identical to Pattern (`docs/bitbrain_gun_verdict.md`, ## commit d93ce44) and as carrying no measurable aim information ## (450+: 16.200 deg vs Pattern's 16.193; `docs/bitbrain_campaign.md` §0.3.5). ## ## Phase 1 measured the actual lever. The gain sweep found that gains >= 1 are ## strictly worse at every band and that the optimal gain is BELOW 1.0 at long ## range (450+: ~0.25). A fractional gain leaves the Pearson lead *correlation* ## unchanged (correlation is invariant under positive scaling), so a smaller ## gain does not add information — it shrinks the magnitude of an uninformative ## Pattern lead toward the low-variance static (HeadOn) aim. The right output is ## therefore a multiplicative GAIN on Pattern's lead, not a class-based additive ## shift. See `docs/bitbrain_campaign.md` §Phase 1 for the measured curve. ## ## ── THE DESIGN ──────────────────────────────────────────────────────────────── ## * BASE — the shipped Pattern gun's prediction (`guns/pattern_matcher`), ## reached through the TmHorizonGun observation ring. ## * OUTPUT — `aim = LOS + gain * (patternAim - LOS)`, i.e. Pattern's lead over ## the line of sight is multiplied by a learned `gain` (one of ## `LG_CAND`, so it may be BELOW 1.0 — the point). ## * LABEL — the same deferred-label path the old corrector used: at fire ## time we remember the base lead and the aim tolerance; `h = ## round(dist/speed)` ticks later `tmhObservedAt` returns the ## enemy's OBSERVED bearing from the firing position. `requiredLead ## = observedBearing - LOS` and `baseLead = baseBearing - LOS`, so a ## candidate gain scores a hit on this sample when ## `|gain*baseLead - requiredLead| <= tolerance`. ## * TRAIN — ONLINE / PREQUENTIAL per range band: for each candidate gain we ## count the fraction of resolved samples that would have been ## within the target's angular half-width (`atan(18/range)`, the ## SAME tolerance the offline ruler uses). The band's gain is the ## argmax hit rate. THIS is the key lesson of Phase 1: the ## least-squares gain and the hit-probability-optimal gain DIVERGE ## (Pattern's lead errors are bimodal), so the learner optimises the ## hit-probability proxy directly instead of mean squared error. ## * STATE — the range band (the ruler's 5 bands). Range is known causally at ## fire time, so a per-band gain table is shippable with no learning ## at all; this gun learns that table online. The correction is ## additionally gated to bands with range >= 300 px ## (`LG_GAIN_BAND_MIN`), where Phase 1 measured Pattern's lead to be ## uninformative. That gate is causal (range is known). ## ## The gain statistics are battle-scale: a round boundary wipes the observation ## ring and deferred labels but NOT the gain counts (a new round is not a new ## enemy). `resetLearning` wipes them on a new battle / target change; with ## `TR_LEADGAIN_MEM=decay` every `TR_LEADGAIN_DECAY` resolved samples decays the ## counts by `TR_LEADGAIN_DECAY_FRAC` toward the gain-1.0 column. ## ## ── WHAT IS STILL HERE ONLY FOR THE BOOT REPORT / GUARD TESTS ───────────────── ## The ADE+SBC network is GONE from the gun. The class geometry ## (`lgCenterDeg`/`lgClassOf`) and `TR_LEADGAIN_N`/`NADE`/`WARMUP`/`ADAPT`/ ## `CALIB`/`SEED`/`RANGE` are retained as resolved configuration so the boot ## report (`env_report.nim`) and the registration guard tests keep working ## unchanged; they no longer affect the gain learner. The generic ## `common_libs/bitbrain/` library is untouched and still tested by ## `test_bitbrain.nim` — `movements/learned_surfer.nim` still imports ## `bitbrain/sbc`, so the library is NOT the gun. ## ## ── BACKWARD COMPATIBILITY: the `TR_BITBRAIN_*` legacy aliases ─────────────── ## The owner's live `.env` predates the rename and contains `TR_RACK_BITBRAIN`, ## `TR_BITBRAIN_GAINS`, `TR_BITBRAIN_MEM`, `TR_BITBRAIN_LOG`, … Those names are ## the OLD corrector's knobs and MUST keep working unchanged. ## ## The `TR_BITBRAIN_*` prefix was briefly shared with the ADE+SBC gun, which ## had a two-mode disambiguator, `TR_BITBRAIN_NET`. That gun was RETIRED and ## removed (it did not learn; the owner watched it live and threw it away), so ## the legacy namespace is now the ONLY namespace and the mapping is ## UNCONDITIONAL: every `TR_BITBRAIN_` in the frozen `LegacyKnobEnvNames` ## set is an alias for `TR_LEADGAIN_`, and `TR_RACK_BITBRAIN` always selects ## this gun (rack id 16). A stale `TR_BITBRAIN_NET=1` left in a `.env` is now an ## unrecognised variable: the boot report warns about it and ignores it. ## ## ── TR_LEADGAIN_GAINS (the candidate set as an env knob) ───────────────────── ## `TR_LEADGAIN_GAINS` is a comma-separated candidate list, e.g. ## `TR_LEADGAIN_GAINS=1.0,1.25,1.5,2.0`. It replaces the fixed shipped candidate ## set `{0, 0.25, 0.5, 0.75, 1.0}` for this gun instance, so every live arm is ## pure-env (no recompile). Two degenerate cases are deliberate: ## * unset / unparsable -> the shipped `LG_CAND` set, byte-identical behaviour; ## * exactly ONE value -> a FIXED gain, applied from the first shot with NO ## learning at all (the learner is bypassed), still gated to the long bands. ## The applied `gain` (and the resulting angular `shift`) is printed on the ## existing change-gated `[lg]` line, so a run's liveness AND the correction it ## actually applied are both auditable from the bot's stdout. ## ## DEFAULT OFF / PARITY: this gun is admitted ONLY when `TR_RACK_LEADGAIN` says so ## (default `off`). The shipped rack never calls `predict`, so `ensureInit` never ## runs and the shipped bot is byte-for-byte unchanged. import std/[math, os, strutils, strformat, algorithm] import gun_harness/gun_interface import guns/tm_horizon import guns/pattern_matcher import gun_harness/selector const ## ── env knobs (all resolved once at gun construction) ───────────────────── LG_MEM_ENV* = "TR_LEADGAIN_MEM" ## perRound|retained|decay LG_GAINS_ENV* = "TR_LEADGAIN_GAINS" ## comma-separated candidate gains LG_N_ENV* = "TR_LEADGAIN_N" ## (legacy geometry; inert) LG_NADE_ENV* = "TR_LEADGAIN_NADE" ## (legacy ADE count; inert) LG_RANGE_ENV* = "TR_LEADGAIN_RANGE" ## (legacy class half-range; inert) LG_LOG_ENV* = "TR_LEADGAIN_LOG" ## 1 = per-change [lg] log ## ── the legacy `TR_BITBRAIN_*` namespace ────────────────────────────────── ## This gun OWNS the `TR_BITBRAIN_*` prefix unconditionally (the ADE+SBC gun ## that briefly shared it was retired), so the frozen suffix set below maps ## every legacy name onto this gun's `TR_LEADGAIN_` with no switch. LG_MIN_OBS_ENV* = "TR_LEADGAIN_MIN_OBS" ## samples before a band is trusted LG_WARMUP_ENV* = "TR_LEADGAIN_WARMUP" ## (legacy; inert) LG_ADAPT_ENV* = "TR_LEADGAIN_ADAPT" ## (legacy; inert) LG_CALIB_ENV* = "TR_LEADGAIN_CALIB" ## (legacy; inert) LG_DECAY_ENV* = "TR_LEADGAIN_DECAY" ## decay interval (samples) LG_DECAY_FRAC_ENV* = "TR_LEADGAIN_DECAY_FRAC" ## per-decay count shrink LG_SEED_ENV* = "TR_LEADGAIN_SEED" ## (legacy; inert) LG_RESET_ON_TARGET_ENV* = "TR_LEADGAIN_RESET_ON_TARGET" ## ── fixed geometry ──────────────────────────────────────────────────────── LG_PENDING_CAP* = 512 ## deferred-label queue (>= 4 buckets x 50 ticks) ## ── the gain learner ────────────────────────────────────────────────────── LG_NBANDS* = 5 ## the ruler's range bands LG_NHB* = 4 ## horizon buckets (for the per-tick label dedupe) LG_BAND_LO* = [0.0, 100.0, 200.0, 300.0, 450.0] LG_BAND_HI* = [100.0, 200.0, 300.0, 450.0, 1.0e18] ## The DEFAULT candidate lead gains the band selector picks from. 0.0 == HeadOn ## (aim at the current position) and 1.0 == Pattern (use the full lead). ## `TR_LEADGAIN_GAINS` replaces this set per gun; unset -> this exact set. LG_CAND* = [0.0, 0.25, 0.50, 0.75, 1.0] LG_NCAND* = 5 LG_BOT_RADIUS* = 18.0 ## hit-detection radius in px (ruler tolerance) ## Apply the correction only from this band up (range >= LG_BAND_LO[3] = 300). ## [MEASURED] below 300 Pattern's lead is informative and shrinking it loses ## hits; see the header note. LG_GAIN_BAND_MIN* = 3 ## ── shipped defaults ────────────────────────────────────────────────────── LG_N_DEF = 32 LG_NADE_DEF = 256 LG_RANGE_DEF = 40.0 LG_MIN_OBS_DEF = 8 LG_WARMUP_DEF = 400 LG_ADAPT_DEF = 32 LG_CALIB_DEF = 512 LG_DECAY_DEF = 250 LG_DECAY_FRAC_DEF = 0.02 LG_SEED_DEF = 20240921 LG_RESET_ON_TARGET_DEF = true type LeadMemMode* = enum lgPerRound, lgRetained, lgDecay LgPending = object ## One deferred training sample. `lead` is Pattern's lead over LOS at fire ## time (radians) and `tol` the target's angular half-width then; the label ## is resolved `horizon` ticks later. fireTick: int horizon: int band: int selfX*, selfY: float baseBearing: float lead: float tolDeg: float LeadGainGun* = object tmh: TmHorizonGun initialized: bool # ── resolved config (kept in the boot report) ───────────────────────────── nClasses*: int maxDeg*: float nAde*: int memMode*: LeadMemMode logEnabled*: bool minObs*: int warmupN*: int adaptEvery*: int calibEvery*: int decayEvery*: int decayFrac*: float seed*: int64 resetOnTarget*: bool # ── the candidate gain set (resolved once at construction) ──────────────── ## Ascending; one entry == a FIXED gain with no learning. `bandHits` and ## `bandN` are sized to it, so the loops below never touch a stale column. cands*: seq[float] # ── gain learner: hit counts per (range band x candidate gain) ──────────── bandHits*: seq[seq[float64]] bandN*: seq[float64] trained*: int sinceDecay: int decays*: int # ── readout / accounting ────────────────────────────────────────────────── lastGain*: array[LG_NBANDS, float] corrections*: int lastLogKey: string # ── deferred labels ─────────────────────────────────────────────────────── pending: array[LG_PENDING_CAP, LgPending] pendingCount*: int pendingDropped*: int # ── per-tick caches ─────────────────────────────────────────────────────── lastTick: int lastEnqTick: int lastEnqBucket: int observedTargetId*: int # ── small pure helpers ─────────────────────────────────────────────────────── proc wrapRadLg(r: float): float {.inline.} = result = r while result > PI: result -= 2.0 * PI while result < -PI: result += 2.0 * PI proc memModeName*(m: LeadMemMode): string = case m of lgPerRound: "perRound" of lgRetained: "retained" of lgDecay: "decay" proc lgGainsString*(cands: seq[float]): string = ## The resolved candidate set as the env's comma-separated form (boot report). for i, c in cands: if i > 0: result.add "," result.add $c proc parseLgGains*(value: string): seq[float] = ## Parse `TR_LEADGAIN_GAINS`. Empty / unparsable / out-of-range / duplicate ## input cannot silently select a different regime: it falls back to the ## shipped `LG_CAND` set, exactly like the other env knobs fall back to their ## defaults. Values are clamped to [0, 8] (0 == HeadOn, 1 == Pattern) and ## de-duplicated, then sorted so the argmax tie rule (keep the smaller ## candidate) is unchanged. var seen: seq[float] for tok in value.split(','): let t = tok.strip() if t.len == 0: continue var v: float try: v = parseFloat(t) except ValueError: continue if v < 0.0 or v > 8.0: continue var dup = false for u in seen: if abs(u - v) < 1e-9: dup = true if not dup: seen.add v if seen.len == 0: for c in LG_CAND: seen.add c return seen seen.sort() seen proc parseLgMemMode*(value: string): LeadMemMode = ## Empty / unknown values fall back to the shipped `perRound`, so a typo ## cannot silently select another regime. case value.strip().toLowerAscii() of "retained", "retain", "accum", "accumulate": lgRetained of "decay", "forget", "age": lgDecay else: lgPerRound proc lgEnv(name: string): string ## Forward declaration: the legacy-alias lookup is defined below, after the ## frozen `LegacyKnobEnvNames` table it depends on. proc envFloatLg(name: string, default: float): float = let v = lgEnv(name) if v.len == 0: return default try: parseFloat(v.strip()) except ValueError: default # ── legacy `TR_BITBRAIN_*` aliases (backward compatibility) ─────────────────── const LegacyPrefix* = "TR_BITBRAIN_" NewPrefix* = "TR_LEADGAIN_" ## The COMPLETE, FROZEN set of the old corrector's knob suffixes. A ## `TR_BITBRAIN_` in this set is a legacy alias for `TR_LEADGAIN_`. ## This gun is the sole owner of the prefix (the ADE+SBC gun was retired), so ## the mapping is total and unconditional — no name is claimed twice. LegacyKnobEnvNames* = [ "GAINS", "MEM", "MIN_OBS", "DECAY", "DECAY_FRAC", "LOG", "RESET_ON_TARGET", "N", "NADE", "RANGE", "WARMUP", "ADAPT", "CALIB", "SEED"] ## Knobs that actually change behaviour (the rest are inert configuration kept ## for the boot report). A deprecation line is only worth printing for these ## plus the inert ones, because a stale inert name is still a stale name. ## The legacy rack knob, from `gun_harness/selector`'s alias table so the ## two can never drift apart. It is NOT a current rack name any more (it was ## `RackGunNames[RackNetGunId]` while the retired ADE+SBC gun existed). LegacyRackEnvName* = RackLegacyAlias[0][0] var deprecationShown = false proc lgDeprecationLine*(): string = ## The single clear deprecation line the owner sees. Names every legacy ## `TR_BITBRAIN_*` knob that is actually set in the environment and the new ## name that now owns it. Empty when there is nothing to migrate. var parts: seq[string] for suffix in LegacyKnobEnvNames: let old = LegacyPrefix & suffix if getEnv(old, "").len > 0: parts.add old & " -> " & NewPrefix & suffix if getEnv(LegacyRackEnvName, "").len > 0: parts.add LegacyRackEnvName & " -> TR_RACK_LEADGAIN" if parts.len == 0: return "" result = "[depr] " & LegacyPrefix & "* is the OLD lead-gain corrector's namespace; " & "it was renamed to " & NewPrefix & "* (gun LEADGAIN, rack id 16). " & "Still honoured: " & parts.join("; ") & "." proc lgEnv(name: string): string = ## Read a `TR_LEADGAIN_` knob, falling back to the legacy ## `TR_BITBRAIN_` alias. The NEW name always wins when both are set, so a ## migrated config is authoritative. var v = getEnv(name, "") if v.len > 0: return v let suffix = if name.startsWith(NewPrefix): name[NewPrefix.len .. ^1] else: "" if suffix.len == 0: return "" for s in LegacyKnobEnvNames: if s == suffix: return getEnv(LegacyPrefix & suffix, "") "" proc envIntLg(name: string, default: int): int = let v = lgEnv(name) if v.len == 0: return default try: parseInt(v.strip()) except ValueError: default proc envBoolLg(name: string, default: bool): bool = case lgEnv(name).strip().toLowerAscii() of "1", "true", "yes", "on": true of "0", "false", "no", "off": false else: default proc lgCenterDeg*(k, nClasses: int, maxDeg: float): float = ## Centre (degrees) of correction class `k` over ±maxDeg. Retained for the ## registration guard test and the boot report; inert for the gain learner. let w = 2.0 * maxDeg / float(nClasses) -maxDeg + (float(k) + 0.5) * w proc lgClassOf*(errRad: float, nClasses: int, maxDeg: float): int = ## Bin a signed angular error (radians) into one of `nClasses` bins over ## [−maxDeg, +maxDeg]. Retained for the registration guard test; inert. let x = radToDeg(errRad) var k = int((x + maxDeg) / (2.0 * maxDeg) * float(nClasses)) if k < 0: k = 0 if k >= nClasses: k = nClasses - 1 k proc lgBandOf*(range: float): int {.inline.} = ## Range band (the ruler's bands), known causally at fire time. for b in 0 ..< LG_NBANDS: if range >= LG_BAND_LO[b] and range < LG_BAND_HI[b]: return b LG_NBANDS - 1 proc lgTolDeg*(range: float): float {.inline.} = ## The target's angular half-width at `range` — atan(18/range) — i.e. the exact ## tolerance the offline ruler uses for its hit-probability proxy. radToDeg(arctan2(LG_BOT_RADIUS, max(range, 1e-9))) # ── construction / lazy init ───────────────────────────────────────────────── proc initLeadGainGun*(): LeadGainGun = result.nClasses = clamp(envIntLg(LG_N_ENV, LG_N_DEF), 2, 512) result.nAde = clamp(envIntLg(LG_NADE_ENV, LG_NADE_DEF), 8, 4096) result.maxDeg = clamp(envFloatLg(LG_RANGE_ENV, LG_RANGE_DEF), 1.0, 180.0) result.memMode = parseLgMemMode(lgEnv(LG_MEM_ENV)) result.logEnabled = envBoolLg(LG_LOG_ENV, false) result.minObs = max(1, envIntLg(LG_MIN_OBS_ENV, LG_MIN_OBS_DEF)) result.warmupN = max(0, envIntLg(LG_WARMUP_ENV, LG_WARMUP_DEF)) result.adaptEvery = max(1, envIntLg(LG_ADAPT_ENV, LG_ADAPT_DEF)) result.calibEvery = max(1, envIntLg(LG_CALIB_ENV, LG_CALIB_DEF)) result.decayEvery = max(1, envIntLg(LG_DECAY_ENV, LG_DECAY_DEF)) result.decayFrac = clamp(envFloatLg(LG_DECAY_FRAC_ENV, LG_DECAY_FRAC_DEF), 0.0, 1.0) result.seed = int64(envIntLg(LG_SEED_ENV, LG_SEED_DEF)) result.resetOnTarget = envBoolLg(LG_RESET_ON_TARGET_ENV, LG_RESET_ON_TARGET_DEF) result.cands = parseLgGains(lgEnv(LG_GAINS_ENV)) result.bandN = newSeq[float64](LG_NBANDS) result.bandHits = newSeq[seq[float64]](LG_NBANDS) for b in 0 ..< LG_NBANDS: result.bandHits[b] = newSeq[float64](result.cands.len) result.lastTick = -1 result.lastEnqTick = -1 result.lastEnqBucket = -1 result.observedTargetId = -1 for b in 0 ..< LG_NBANDS: result.lastGain[b] = 1.0 # ONE deprecation line per process, naming the new `TR_LEADGAIN_*` names. let dep = lgDeprecationLine() if dep.len > 0 and not deprecationShown: deprecationShown = true stderr.writeLine(dep) proc ensureInit*(g: var LeadGainGun) = ## Build the observation ring on first use. No network, no global-RNG use, so ## the shipped default path is untouched and construction stays cheap. if g.initialized: return g.initialized = true g.tmh = initTmHorizonGun() # ── the gain learner ───────────────────────────────────────────────────────── proc lgAccumulate(g: var LeadGainGun, leadDeg, reqDeg, tolDeg: float, band: int) = ## Score every candidate gain on this resolved sample: a candidate "hits" when ## it would have put the aim within the target's angular half-width. for ci in 0 ..< g.cands.len: if abs(g.cands[ci] * leadDeg - reqDeg) <= tolDeg: g.bandHits[band][ci] += 1.0 g.bandN[band] += 1.0 inc g.trained proc lgApplyDecay(g: var LeadGainGun) = ## Forgetting for `TR_LEADGAIN_MEM=decay`: shrink the hit counts and, more ## strongly, pull them toward the gain-1.0 column so stale evidence ages out. let f = 1.0 - g.decayFrac if f >= 1.0: return for b in 0 ..< LG_NBANDS: for ci in 0 ..< g.cands.len: g.bandHits[b][ci] *= f g.bandN[b] *= f inc g.decays proc lgGainFor(g: LeadGainGun, band: int): float = ## The band's gain is the candidate with the highest observed hit rate. ## Ties keep the SMALLER candidate (the scan is ascending), which is the ## conservative choice for the long-range regime this corrector targets. ## Returns 1.0 (Pattern) below the range gate or when the band is cold. if band < LG_GAIN_BAND_MIN: return 1.0 if g.cands.len == 0: return 1.0 # A single candidate is a FIXED gain: apply it from the first shot, never # consult the counts. This is the no-learning arm of the live sweep. if g.cands.len == 1: return g.cands[0] if g.bandN[band] < float(g.minObs): return 1.0 var best = -1 var bestRate = -1.0 for ci in 0 ..< g.cands.len: let rate = g.bandHits[band][ci] / g.bandN[band] if rate > bestRate: bestRate = rate best = ci if best < 0: return 1.0 g.cands[best] # ── deferred-label resolution (prequential learning) ───────────────────────── proc resolvePending(g: var LeadGainGun, state: WorldState) = var w = 0 for i in 0 ..< g.pendingCount: let p = g.pending[i] let due = p.fireTick + p.horizon if due > state.tick: g.pending[w] = p inc w elif due == state.tick: let obs = tmhObservedAt(g.tmh, state.tick, p.selfX, p.selfY) if obs.ok and (state.tick - obs.lastSeenTick) <= TMH_STALE_MAX: let err = wrapRadLg(obs.bearing - p.baseBearing) let reqLead = wrapRadLg(err + p.lead) g.lgAccumulate(radToDeg(p.lead), radToDeg(reqLead), p.tolDeg, p.band) inc g.sinceDecay if g.memMode == lgDecay and g.sinceDecay >= g.decayEvery: g.lgApplyDecay() g.sinceDecay = 0 else: inc g.pendingDropped else: inc g.pendingDropped g.pendingCount = w # ── logging ────────────────────────────────────────────────────────────────── proc lgLog(g: var LeadGainGun, state: WorldState, band: int, gain, leadDeg: float) = ## ONE change-gated `[lg]` line (behind TR_LEADGAIN_LOG=1) so a user tailing ## the GUI log sees the gain the corrector is applying. The APPLIED gain and ## the resulting angular `shift` are both on the line: the boot report proves ## the knob reached the process, this proves the gun actually used it. if not g.logEnabled: return let shiftDeg = (gain - 1.0) * leadDeg let key = fmt"{gain:.2f}|{band}" if key == g.lastLogKey: return g.lastLogKey = key var rate = 0.0 for ci in 0 ..< g.cands.len: if abs(g.cands[ci] - gain) < 1e-9: rate = g.bandHits[band][ci] / max(1.0, g.bandN[band]) echo fmt"[lg] t={state.tick} band={LG_BAND_LO[band]:.0f}+ gain={gain:.2f} " & fmt"shift={shiftDeg:+.2f}deg rate={rate:.3f} n={g.bandN[band]:.0f} " & fmt"ncand={g.cands.len} trained={g.trained} " & fmt"pend={g.pendingCount} dropped={g.pendingDropped} mode={memModeName(g.memMode)}" # ── reset hooks (mirroring TmHorizonGun) ───────────────────────────────────── proc resetRound(g: var LeadGainGun) = ## PER-ROUND reset: observation ring, deferred labels and per-tick caches (the ## bots teleport between rounds). The gain counts are deliberately KEPT — they ## are battle-scale and a new round is not a new enemy. g.tmh.resetRoundState() g.pendingCount = 0 g.lastTick = -1 g.lastEnqTick = -1 g.lastEnqBucket = -1 g.lastLogKey = "" proc resetRoundState*(g: var LeadGainGun) = if not g.initialized: return g.resetRound() proc resetLearning*(g: var LeadGainGun, reason = "") = ## PER-BATTLE / PER-ENEMY wipe: gain counts, counters and the round state. if not g.initialized: return for b in 0 ..< LG_NBANDS: for ci in 0 ..< g.cands.len: g.bandHits[b][ci] = 0.0 g.bandN[b] = 0.0 g.lastGain[b] = 1.0 g.trained = 0 g.sinceDecay = 0 g.decays = 0 g.corrections = 0 g.observedTargetId = -1 g.resetRound() if reason.len > 0 and g.logEnabled: echo fmt"[lg-reset] reason={reason}" proc targetChanged*(g: var LeadGainGun, enemyId: int): bool = ## Per-ENEMY reset: wipe when the target changes to a different bot id. First ## acquisition never wipes, so the round-start pick does not cold-start us. if not g.resetOnTarget: return false if enemyId < 0: return false if g.observedTargetId >= 0 and enemyId != g.observedTargetId: g.resetLearning("target_change") g.observedTargetId = enemyId return true g.observedTargetId = enemyId false # ── Gun interface ──────────────────────────────────────────────────────────── proc isWarmedUp*(g: LeadGainGun): bool {.inline.} = true proc networkBytes*(g: LeadGainGun): int = ## No neural network is held any more; kept for the boot report / guard test. 0 proc predict*(g: var LeadGainGun, state: WorldState, bulletSpeed: float): GunPrediction = g.ensureInit() # Round boundary: a tick regression means a new round. if state.tick < g.lastTick: g.resetRound() # Once per tick: observe the world, then resolve any labels now due. if state.tick != g.lastTick: tmhUpdateHistory(g.tmh, state) g.resolvePending(state) g.lastTick = state.tick # The base prediction is Pattern; LEADGAIN only scales its lead over LOS. let base = g.tmh.pattern.predict(state, bulletSpeed) if bulletSpeed <= 0.0: return base let dist = hypot(state.enemyX - state.selfX, state.enemyY - state.selfY) let h = tmhHorizonFor(dist, bulletSpeed) let hb = tmhHorizonBucket(h) let band = lgBandOf(dist) let los = arctan2(state.enemyY - state.selfY, state.enemyX - state.selfX) let baseBearing = arctan2(base.y - state.selfY, base.x - state.selfX) let lead = wrapRadLg(baseBearing - los) # Enqueue one deferred sample per (tick, horizon bucket): `predict` runs once # per power bin, so all four horizons contribute evidence. if g.lastEnqTick != state.tick or g.lastEnqBucket != hb: if g.pendingCount < LG_PENDING_CAP: g.pending[g.pendingCount] = LgPending( fireTick: state.tick, horizon: h, band: band, selfX: state.selfX, selfY: state.selfY, baseBearing: baseBearing, lead: lead, tolDeg: lgTolDeg(dist)) inc g.pendingCount else: inc g.pendingDropped g.lastEnqTick = state.tick g.lastEnqBucket = hb # Readout: a fractional gain may be BELOW 1.0. When cold / gated out the # learner returns 1.0 and the base prediction is returned unchanged. let gain = g.lgGainFor(band) g.lastGain[band] = gain if abs(gain - 1.0) < 1e-9: return base inc g.corrections g.lgLog(state, band, gain, radToDeg(lead)) tmhApplyShift(state.selfX, state.selfY, base.x, base.y, radToDeg((gain - 1.0) * lead)) proc onResult*(g: var LeadGainGun, e: FeedbackEvent) = ## Labels come from our own observation ring, not from virtual-bullet ## feedback, so there is nothing to do here. The hook exists for the rack. discard