diff --git a/common_libs/tests/measure_bitbrain_gate.nim b/common_libs/tests/measure_bitbrain_gate.nim new file mode 100644 index 0000000..7e6e927 --- /dev/null +++ b/common_libs/tests/measure_bitbrain_gate.nim @@ -0,0 +1,648 @@ +## OFFLINE GATE TEST (read-only fixtures, no Java, no battle, no server). +## +## QUESTION: can the generic BitBrain (ADE + SBC) library at +## `common_libs/bitbrain/` predict a FINE-GRAINED AIM CORRECTION better than +## (a) the straight-line naive extrapolation, (b) an always-the-same-answer +## fixed correction, and (c) the shipped Pattern gun's own prediction? +## +## INPUT = the SAME 53 bits the TMHorizon gun uses. They are harvested from the +## LIVE `TmHorizonGun` itself (the per-tick cached-bits + 4-bit horizon +## one-hot path), not re-derived, so the comparison to the TM gun we +## already measured is apples-to-apples. +## OUTPUT = a fine-grained angular-correction class N over a fixed range. The +## KEY readout is the COUNT-WEIGHTED MEAN of the per-class set-bit +## counts (a soft continuous predictor); the argmax is reported too. +## LABEL = the +h-tick FACT exactly as TMHorizon defines it: h = round(dist / +## speed), speed = 20 - 3*power; at tick t the enemy's ACTUAL angular +## offset from Pattern's prediction is looked up at t+h. Never crosses +## a round boundary. +## METRIC = arrival aim error in PIXELS and DEGREES, plus a simulated hit +## (within the 18 px bot radius). Arrival geometry is the harness +## `bmPoint` formula (fireDist = |Pattern - self|; arrivalTick = +## T + ceil(fireDist/v) - 1), the same relation measure_aim_vs_power.nim +## validated against the harness resolver. +## PROTOCOL = PREQUENTIAL (predict-then-learn, streaming). Two regimes: +## retained across all rounds of one battle, and reset every round. +## +## Run: +## nim c -r -d:release --nimcache:/tmp/nc_j92 --path:common_libs \ +## common_libs/tests/measure_bitbrain_gate.nim + +import std/[os, strformat, strutils, math, tables, algorithm, random, json, times] +import gun_harness/gun_interface +import gun_harness/virtual_bullets +import gun_harness/offline_range +import guns/tm_horizon +import bitbrain/bitbrain + +const + repoRoot = currentSourcePath().parentDir.parentDir.parentDir + fixturesDir = repoRoot / "tools" / "fixtures" + metaDir = fixturesDir / "drussgt_meta" + NIN = TMH_N_BITS ## 53 + BotR = BotRadius ## 18 px + TMPM = PI / 180.0 + +# ── configuration (env-overridable) ─────────────────────────────────────────── +proc envF(name: string, d: float): float = + let v = getEnv(name, "") + if v.len == 0: return d + try: parseFloat(v.strip()) except ValueError: d +proc envI(name: string, d: int): int = + let v = getEnv(name, "") + if v.len == 0: return d + try: parseInt(v.strip()) except ValueError: d +proc envList(name, d: string): seq[int] = + let v = getEnv(name, d) + for p in v.split(','): + let t = p.strip() + if t.len > 0: result.add parseInt(t) + +let + EVAL_POWER = envF("BB_POWER", 2.0) + PMAX_DEG = envF("BB_PMAX", 40.0) + AD_PASSES = envI("BB_PASSES", 2) + AD_STEP = envI("BB_STEP", 1) + AD_CHUNK = envI("BB_CHUNK", 1000) + AD_STRIDE = envI("BB_STRIDE", 3) + AD_INIT = envI("BB_AD_INIT", 1) + AD_TARGET = envF("BB_TARGET", 0.01) + NADE_LIST = envList("BB_NADE", "128,256") + NLIST = envList("BB_NLIST", "4,8,16,32,64") + SEED = 20240921'i64 + +# ── small math ──────────────────────────────────────────────────────────────── + +proc wrapRad(a: float): float {.inline.} = + result = a + while result > PI: result -= 2.0 * PI + while result < -PI: result += 2.0 * PI + +proc medianOf(xs: var seq[float]): float = + if xs.len == 0: return NaN + xs.sort() + let n = xs.len + if n mod 2 == 1: xs[n div 2] + else: 0.5 * (xs[n div 2 - 1] + xs[n div 2]) + +proc pctOf(xs: var seq[float], q: float): float = + if xs.len == 0: return NaN + xs.sort() + xs[min(xs.len - 1, int(ceil(q * xs.len.float)) - 1)] + +# ── fixture round metadata ──────────────────────────────────────────────────── + +type + RoundInfo = object + endTick: int + roundId: int + +proc loadRoundInfo(path: string, states: seq[WorldState]): Table[int, RoundInfo] = + let rp = metaDir / (extractFilename(path) & ".rounds.json") + if not fileExists(rp): + for s in states: result[s.tick] = RoundInfo(endTick: states[^1].tick, roundId: 0) + return + var idxAt = initTable[int, int]() + for i, s in states: idxAt[s.tick] = i + let j = parseFile(rp) + var rid = 0 + for r in j["rounds"]: + let s0 = r["startTick"].getInt() + let c = r["count"].getInt() + if s0 notin idxAt: continue + let i0 = idxAt[s0] + let lastIdx = min(states.len - 1, i0 + c - 1) + for k in i0 .. lastIdx: + result[states[k].tick] = RoundInfo(endTick: states[lastIdx].tick, roundId: rid) + inc rid + +# ── harvested sample ────────────────────────────────────────────────────────── + +type + GSample = object + bits: array[NIN, uint8] + selfX, selfY: float + baseX, baseY: float + baseBearing: float + fireTick: int + horizon: int + speed: float + fxId: int + roundId: int + labelErr: float ## radians, fact at t+h + arrTick: int + actualArrX, actualArrY: float + actualArrBearing: float + slX, slY: float ## straight-line prediction at arrTick + slBearing: float + fireDist: float + +# ── harvest: drive the LIVE TmHorizonGun, capture its own bits ─────────────── + +proc harvest(fx: Fixture, fxId: int, power: float, rinfo: Table[int, RoundInfo], + samples: var seq[GSample]): int = + var g = initTmHorizonGun() + g.setShift(0.0) # pure predict arm; we read Pattern's base + let speed = bulletSpeed(power) + if speed <= 0.0: return 0 + var idxAt = initTable[int, int]() + for i, s in fx.states: idxAt[s.tick] = i + var curRound = -1 + for state in fx.states: + let T = state.tick + if T notin rinfo: continue + let ri = rinfo[T] + if ri.roundId != curRound: + g.resetRoundState() # mirror the live onRoundStarted wipe + curRound = ri.roundId + let pred = g.predict(state, speed) + # Reuse the gun's OWN exported builder: tmhBaseBits (49 draft bits) + the + # 4-bit horizon one-hot via tmhLits, exactly the cachedBits per-tick path. + let dist = hypot(state.enemyX - state.selfX, state.enemyY - state.selfY) + let h = tmhHorizonFor(dist, speed) + let bucket = tmhHorizonBucket(h) + let baseBits = g.tmhBaseBits(state) + let lits = tmhLits(baseBits, bucket) + if T + h > ri.endTick: continue + let fireDist = hypot(pred.x - state.selfX, pred.y - state.selfY) + if fireDist < 1e-6: continue + let arrTick = T + max(1, int(ceil(fireDist / speed))) - 1 + if arrTick > ri.endTick: continue + if arrTick notin idxAt or (T + h) notin idxAt: continue + + let actH = fx.states[idxAt[T + h]] + let sx = state.selfX + let sy = state.selfY + let baseBearing = arctan2(pred.y - sy, pred.x - sx) + let labelErr = wrapRad(arctan2(actH.enemyY - sy, actH.enemyX - sx) - baseBearing) + + let actA = fx.states[idxAt[arrTick]] + let arrBearing = arctan2(actA.enemyY - sy, actA.enemyX - sx) + + let hr = degToRad(state.enemyHeading) + let vx = cos(hr) * state.enemySpeed + let vy = sin(hr) * state.enemySpeed + let el = float(arrTick - T) + let slX = state.enemyX + vx * el + let slY = state.enemyY + vy * el + let slBearing = arctan2(slY - sy, slX - sx) + + var bits: array[NIN, uint8] + for k in 0 ..< NIN: bits[k] = lits[k] + samples.add GSample( + bits: bits, selfX: sx, selfY: sy, baseX: pred.x, baseY: pred.y, + baseBearing: baseBearing, fireTick: T, horizon: h, speed: speed, + fxId: fxId, roundId: ri.roundId, labelErr: labelErr, + arrTick: arrTick, actualArrX: actA.enemyX, actualArrY: actA.enemyY, + actualArrBearing: arrBearing, slX: slX, slY: slY, slBearing: slBearing, + fireDist: fireDist) + inc result + +# ── geometry: rotate the base aim point around the shooter ──────────────────── + +proc rotAim(s: GSample, corrRad: float): tuple[x, y: float] = + let dx = s.baseX - s.selfX + let dy = s.baseY - s.selfY + let b = arctan2(dy, dx) + corrRad + (s.selfX + cos(b) * s.fireDist, s.selfY + sin(b) * s.fireDist) + +proc missOf(s: GSample, x, y: float): tuple[px, angRad, rng: float] = + let rng = hypot(s.actualArrX - s.selfX, s.actualArrY - s.selfY) + (hypot(x - s.actualArrX, y - s.actualArrY), + abs(wrapRad(arctan2(y - s.selfY, x - s.selfX) - s.actualArrBearing)), + rng) + +# ── metrics ─────────────────────────────────────────────────────────────────── + +type + Metrics = object + n: int + sumPx, sumDeg: float + hits: int + angHits: int + px: seq[float] + deg: seq[float] + +proc add(m: var Metrics, missPx, angRad, rng: float) = + inc m.n + m.sumPx += missPx + m.sumDeg += angRad * 180.0 / PI + m.px.add missPx + m.deg.add angRad * 180.0 / PI + if missPx < BotR: inc m.hits + if angRad < arctan(BotR / max(rng, 1e-6)): inc m.angHits + +proc merge(dst: var Metrics, src: Metrics) = + dst.n += src.n + dst.sumPx += src.sumPx + dst.sumDeg += src.sumDeg + dst.hits += src.hits + dst.angHits += src.angHits + dst.px.add src.px + dst.deg.add src.deg + +proc line(name: string, m0: Metrics): string = + var m = m0 + if m.n == 0: return fmt"{name:<24} n=0" + let meanPx = m.sumPx / float(m.n) + let medPx = medianOf(m.px) + let p90Px = pctOf(m.px, 0.9) + let meanDeg = m.sumDeg / float(m.n) + let medDeg = medianOf(m.deg) + let hit = 100.0 * float(m.hits) / float(m.n) + let ahit = 100.0 * float(m.angHits) / float(m.n) + fmt"{name:<24} n={m.n:<6} meanPx={meanPx:>6.2f} medPx={medPx:>6.2f} p90Px={p90Px:>6.2f} " & + fmt"meanDeg={meanDeg:>6.2f} medDeg={medDeg:>6.2f} pxHit%={hit:>5.1f} angHit%={ahit:>5.1f}" + +# ── BitBrain readout ────────────────────────────────────────────────────────── + +proc binOf(err: float, nClasses: int, maxDeg: float): int = + let x = (err * 180.0 / PI) + var k = int((x + maxDeg) / (2.0 * maxDeg) * float(nClasses)) + if k < 0: k = 0 + if k >= nClasses: k = nClasses - 1 + k + +proc centersOf(nClasses: int, maxDeg: float): seq[float] = + let w = 2.0 * maxDeg / float(nClasses) + for k in 0 ..< nClasses: + result.add (-maxDeg + (float(k) + 0.5) * w) * TMPM + +proc weightedMean(counts: openArray[int], centers: openArray[float]): float = + var num = 0.0 + var den = 0.0 + for k in 0 ..< counts.len: + num += float(counts[k]) * centers[k] + den += float(counts[k]) + if den <= 0.0: return 0.0 + num / den + +# ── regime ──────────────────────────────────────────────────────────────────── + +type Regime = enum rgRetained, rgPerRound + +proc regimeName(r: Regime): string = + case r + of rgRetained: "retained" + of rgPerRound: "perRound" + +type + ConfigResult = object + nade, nClasses: int + regime: Regime + wm: Metrics + am: Metrics + +# ── one prequential run of a config ─────────────────────────────────────────── + +proc runConfig(ads: seq[AddressDecoder], nClasses: int, samples: seq[GSample], + labels: seq[int], regime: Regime, + maxDeg: float): ConfigResult = + result.nade = if ads.len > 0: ads[0].nAde else: 0 + result.nClasses = nClasses + result.regime = regime + var bb = initBitBrain(ads, crossPairs(ads.len), nClasses) + let centers = centersOf(nClasses, maxDeg) + var lists: seq[seq[int32]] + var counts = newSeq[int](nClasses) + var lastFx = -1 + var lastRound = -1 + for i, s in samples: + if s.fxId != lastFx: + bb.resetLearning() + lastFx = s.fxId + lastRound = -1 + if regime == rgPerRound and s.roundId != lastRound: + bb.resetLearning() + lastRound = s.roundId + + # ---- predict (before learning this sample) ---- + bb.fireInto(s.bits, lists) + for k in 0 ..< counts.len: counts[k] = 0 + var total = 0 + for sl in 0 ..< bb.sbcs.len: + let spec = bb.specs[sl] + bb.sbcs[sl].infer(lists[spec.row], lists[spec.col], counts) + for k in 0 ..< counts.len: total += counts[k] + var best = 0 + for k in 1 ..< nClasses: + if counts[k] > counts[best]: best = k + + let wm = weightedMean(counts, centers) + var (wx, wy) = rotAim(s, wm) + var (mpx, mang, mrg) = missOf(s, wx, wy) + result.wm.add(mpx, mang, mrg) + + let amc = if total > 0: centers[best] else: 0.0 + (wx, wy) = rotAim(s, amc) + (mpx, mang, mrg) = missOf(s, wx, wy) + result.am.add(mpx, mang, mrg) + + # ---- learn ---- + for sl in 0 ..< bb.sbcs.len: + let spec = bb.specs[sl] + discard bb.sbcs[sl].learn(lists[spec.row], lists[spec.col], labels[i]) + +# ── AD synthesis + homeostasis ──────────────────────────────────────────────── + +proc countGE(sc: openArray[int], t: int): int = + ## number of elements >= t in an ascending-sorted array + var a = 0 + var b = sc.len + while a < b: + let m = (a + b) div 2 + if sc[m] >= t: b = m else: a = m + 1 + sc.len - a + +proc initADs(nAde: int, widths: seq[int], train: seq[GSample], + seed: int64, pctInit: bool): seq[AddressDecoder] = + var rng = initRand(seed) + for w in widths: + # center = 0 because our inputs are BINARY (0/1). The reference 127 is the + # midpoint of 0..255; centring binary inputs at 127 would make every synapse + # contribute ~-127 and collapse the ADE code to a polarity count. + result.add initRandomAddressDecoder(nAde, w, NIN, rng, + scale = DefaultScale, center = 0, + threshold = 0'i32) + let stride = max(1, AD_STRIDE) + # Unsupervised percentile init: put each ADE's threshold at the 99th + # percentile of its own score over this data, i.e. the 1% operating point the + # paper's controller aims for. The deterministic controller below then refines + # it (and would find it on its own, but would need thousands of `step=1` + # intervals, far more than a battle of 500-2000 ticks provides). + if pctInit: + for ai in 0 ..< result.len: + var sc = newSeq[int]() + for e in 0 ..< result[ai].nAde: + sc.setLen(0) + var s = 0 + while s < train.len: + sc.add result[ai].score(train[s].bits, e) + s += stride + sc.sort() + if sc.len == 0: continue + # Choose the integer threshold whose firing count is CLOSEST to 1% of + # the observed scores. A plain 99th percentile lands on a large atom + # (binary inputs make the score distribution discrete and sparse), so + # its realised rate can be several percent; picking the closest count + # pins the realised rate near 1%. + let target = AD_TARGET * float(sc.len) + var lo = sc[0] - 1 # count>=lo == n (> target) + var hi = sc[^1] + 1 # count>=hi == 0 (<= target) + while hi - lo > 1: + let mid = (lo + hi) div 2 + if float(countGE(sc, mid)) > target: lo = mid else: hi = mid + let tA = lo + let tB = hi + let cA = countGE(sc, tA) + let cB = countGE(sc, tB) + result[ai].thresholds[e] = + if abs(float(cA) - target) <= abs(float(cB) - target): int32(tA) + else: int32(tB) + +proc homeostasis(ads: var seq[AddressDecoder], train: seq[GSample]) = + let stride = max(1, AD_STRIDE) + for _ in 0 ..< AD_PASSES: + for ad in ads.mitems: ad.resetFiringCounts() + var i = 0 + while i < train.len: + let j = min(i + AD_CHUNK * stride, train.len) + var k = i + var cnt = 0 + while k < j: + for ad in ads.mitems: ad.accumulateFiring(train[k].bits) + inc cnt + k += stride + for ad in ads.mitems: + ad.adaptThresholds(interval = max(1, cnt), targetRate = AD_TARGET, step = AD_STEP) + i = j + +proc fireRates(ads: seq[AddressDecoder], train: seq[GSample], + sampleN: int): seq[float] = + ## Fire rate per AD measured on the SAME stride the thresholds were fitted on + ## (the whole stream, every AD_STRIDE-th sample), so the reported number is the + ## rate the controller actually achieved. + discard sampleN + if train.len == 0: return + let stride = max(1, AD_STRIDE) + for ad in ads: + var f = 0 + var i = 0 + var m = 0 + while i < train.len: + f += ad.fireCount(train[i].bits) + inc m + i += stride + result.add float(f) / (float(m) * float(ad.nAde)) + +const Widths = [6, 8, 10, 12] + +# ── baselines ──────────────────────────────────────────────────────────────── + +proc computeBaselines(samples: seq[GSample]): + tuple[pattern, straight, fixed: Metrics, meanErr: float] = + var sumErr = 0.0 + for s in samples: sumErr += s.labelErr + let meanErr = if samples.len > 0: sumErr / float(samples.len) else: 0.0 + for s in samples: + var (px, ang, rng) = missOf(s, s.baseX, s.baseY) + result.pattern.add(px, ang, rng) + (px, ang, rng) = missOf(s, s.slX, s.slY) + result.straight.add(px, ang, rng) + let (fx, fy) = rotAim(s, meanErr) + (px, ang, rng) = missOf(s, fx, fy) + result.fixed.add(px, ang, rng) + result.meanErr = meanErr + +# ── main ───────────────────────────────────────────────────────────────────── + +proc main() = + var files: seq[string] + let only = getEnv("BB_FILES", "") + if only.len > 0: + for p in only.split(','): + let t = p.strip() + if t.len > 0: files.add fixturesDir / t + else: + for f in walkFiles(fixturesDir / "tr_drussgt_vs_*.jsonl"): files.add f + files.sort() + if files.len == 0: + stderr.writeLine("no tr_drussgt fixtures found"); quit(1) + + let t0 = epochTime() + var samples: seq[GSample] + var perFx: seq[(string, int)] + for fi, path in files: + let fx = loadFixture(path) + let rinfo = loadRoundInfo(path, fx.states) + let got = harvest(fx, fi, EVAL_POWER, rinfo, samples) + perFx.add (extractFilename(path), got) + stderr.writeLine(fmt"[harvest] {extractFilename(path)} ticks={fx.states.len} samples={got}") + stderr.writeLine(fmt"[harvest] total samples={samples.len} in {epochTime()-t0:.1f}s") + + echo "" + echo "=================================================================" + echo "BitBrain gate test - fine-grained aim correction" + echo "=================================================================" + echo fmt"fixtures : {files.len} tr_drussgt_vs_* (open-loop replay)" + echo fmt"power={EVAL_POWER} speed={bulletSpeed(EVAL_POWER):.1f} class half-range=+-{PMAX_DEG:.1f} deg" + echo fmt"AD: widths {Widths} nAde={NADE_LIST} target={AD_TARGET*100.0:.1f}% passes={AD_PASSES} step={AD_STEP} stride={AD_STRIDE}" + echo "" + echo "== dataset (effective sample counts) ==" + for (nm, n) in perFx: echo fmt" {nm:<40} samples={n}" + var abst: seq[float] + var roundCount = initTable[int, int]() + for s in samples: + abst.add abs(s.labelErr) * 180.0 / PI + roundCount[s.fxId] = max(roundCount.getOrDefault(s.fxId, 0), s.roundId + 1) + var a2 = abst + var aSum = 0.0 + for v in a2: aSum += v + echo fmt" |label err| deg: mean={aSum/float(a2.len):.2f} " & + fmt"p50={a2.pctOf(0.5):.2f} p90={a2.pctOf(0.9):.2f} p99={a2.pctOf(0.99):.2f} " & + fmt"max={a2.pctOf(1.0):.2f}" + var totRounds = 0 + for _, r in roundCount: totRounds += r + echo fmt" rounds={totRounds} samples/round={float(samples.len)/float(max(1,totRounds)):.0f}" + echo "" + + # ── baselines ──────────────────────────────────────────────────────────── + let bl = computeBaselines(samples) + echo "== baselines (same arrival geometry, same sample set) ==" + var pm = bl.pattern + var sm = bl.straight + var fm = bl.fixed + echo line("Pattern (zero corr)", pm) + echo line("straight-line naive", sm) + echo line(fmt"fixed corr={bl.meanErr*180.0/PI:+.2f} deg", fm) + echo "" + + # ── label bin histogram for the largest N (effective sizes per class) ───── + let nBig = NLIST[^1] + var hist = newSeq[int](nBig) + for s in samples: inc hist[binOf(s.labelErr, nBig, PMAX_DEG)] + var maxH = 0 + for h in hist: maxH = max(maxH, h) + echo fmt"== label histogram (N={nBig}) ==" + for k in 0 ..< nBig: + echo fmt" class {k:>2} [{(-PMAX_DEG + float(k)*2*PMAX_DEG/float(nBig)):>+6.1f},{(-PMAX_DEG + float(k+1)*2*PMAX_DEG/float(nBig)):>+6.1f}) n={hist[k]:<6}" + echo "" + + # ── AD fire rates ──────────────────────────────────────────────────────── + echo "== AD layer (synthesised on this data, homeostasis to ~1% target) ==" + var adsByNade = initTable[int, seq[AddressDecoder]]() + for nade in NADE_LIST: + var ads = initADs(nade, @Widths, samples, SEED + int64(nade), AD_INIT != 0) + let fr0 = fireRates(ads, samples, 3000) + var f0str = "" + var f0sum = 0.0 + for i, r in fr0: + f0str.add fmt"w{Widths[i]}={r*100.0:.2f}% " + f0sum += r + echo fmt" nAde={nade:<4} pct-init: {f0str} (mean={f0sum/float(fr0.len)*100.0:.2f}%)" + homeostasis(ads, samples) + adsByNade[nade] = ads + let fr = fireRates(ads, samples, 3000) + var fstr = "" + var fSum = 0.0 + for i, r in fr: + fstr.add fmt"w{Widths[i]}={r*100.0:.2f}% " + fSum += r + echo fmt" nAde={nade:<4} +homeostasis:{fstr} (mean={fSum/float(fr.len)*100.0:.2f}%)" + echo "" + + # ── prequential sweep ──────────────────────────────────────────────────── + echo "== prequential sweep (predict-then-learn) ==" + echo " BitBrain wm = count-weighted mean of class centers (KEY readout)" + echo " BitBrain arg = argmax class center" + echo " vs Pattern / straight-line / fixed baselines above" + echo "" + # precompute labels per N + var results: seq[ConfigResult] + for nade in NADE_LIST: + let ads = adsByNade[nade] + for nc in NLIST: + var labels = newSeq[int](samples.len) + for i, s in samples: labels[i] = binOf(s.labelErr, nc, PMAX_DEG) + for regime in [rgRetained, rgPerRound]: + let cr = runConfig(ads, nc, samples, labels, regime, PMAX_DEG) + echo fmt" nAde={nade:<4} N={nc:<3} {regimeName(regime):<9} | " & line("wm", cr.wm) + echo " argmax | " & line("argmax", cr.am) + var r = cr + r.nade = nade + results.add r + echo "" + + # ── per-fixture breakdown for the winning config ──────────────────────── + echo "== per-fixture breakdown: perRound, N=32, vs Pattern ==========" + let winNade = NADE_LIST[^1] + let winAds = adsByNade[winNade] + let winN = 32 + var winLabels = newSeq[int](samples.len) + for i, s in samples: winLabels[i] = binOf(s.labelErr, winN, PMAX_DEG) + for fi in 0 ..< files.len: + var sub: seq[GSample] + var subLab: seq[int] + for i, s in samples: + if s.fxId == fi: + sub.add s + subLab.add winLabels[i] + if sub.len == 0: continue + var pat: Metrics + for s in sub: + let (px, ang, rng) = missOf(s, s.baseX, s.baseY) + pat.add(px, ang, rng) + let cr = runConfig(winAds, winN, sub, subLab, rgPerRound, PMAX_DEG) + echo " " & align(extractFilename(files[fi]), 38) & " n=" & align($sub.len, 6) + echo " " & line("Pattern", pat) + echo " " & line("BitBrain wm", cr.wm) + echo " " & line("BitBrain arg", cr.am) + echo "" + + # ── compact comparison for the verdict ─────────────────────────────────── + echo "== compact: mean px error (lower is better) ==" + echo " " & align("config",24) & " " & align("wm meanPx",9) & " " & + align("arg meanPx",10) & " " & align("wm hit%",8) & " " & align("arg hit%",8) + echo " " & align("Pattern",24) & " " & align(fmt"{pm.sumPx/float(pm.n):.2f}",9) & " " & + align("-",10) & " " & align(fmt"{100.0*float(pm.hits)/float(pm.n):.1f}",8) & " " & align("-",8) + echo " " & align("straight-line",24) & " " & align(fmt"{sm.sumPx/float(sm.n):.2f}",9) & " " & + align("-",10) & " " & align(fmt"{100.0*float(sm.hits)/float(sm.n):.1f}",8) & " " & align("-",8) + echo " " & align("fixed corr",24) & " " & align(fmt"{fm.sumPx/float(fm.n):.2f}",9) & " " & + align("-",10) & " " & align(fmt"{100.0*float(fm.hits)/float(fm.n):.1f}",8) & " " & align("-",8) + for r in results: + let nm = fmt"nAde{r.nade}/N{r.nClasses}/{regimeName(r.regime)}" + echo " " & align(nm,24) & " " & align(fmt"{r.wm.sumPx/float(r.wm.n):.2f}",9) & " " & + align(fmt"{r.am.sumPx/float(r.am.n):.2f}",10) & " " & + align(fmt"{100.0*float(r.wm.hits)/float(r.wm.n):.1f}",8) & " " & + align(fmt"{100.0*float(r.am.hits)/float(r.am.n):.1f}",8) + echo "" + + # ── shuffled-label control (must collapse to the fixed baseline) ───────── + echo "== shuffled-label control (permuted labels, same input distribution) ==" + echo " (the weighted mean shrinks to the label mean; the argmax keeps making a" + echo " confident pick, so its null is random-correction worse-than-baseline)" + let nadeBig = NADE_LIST[^1] + let adsBig = adsByNade[nadeBig] + var baseLabels = newSeq[int](samples.len) + for i, s in samples: baseLabels[i] = binOf(s.labelErr, nBig, PMAX_DEG) + for regime in [rgRetained, rgPerRound]: + var wPx, aPx: float + var wHit, aHit, shN: int + for rep in 0 ..< 3: + var labs = baseLabels + var rng = initRand(SEED + 777 + int64(rep)) + for i in countdown(labs.len - 1, 1): + let j = rng.rand(i) + swap(labs[i], labs[j]) + let cr = runConfig(adsBig, nBig, samples, labs, regime, PMAX_DEG) + wPx += cr.wm.sumPx; wHit += cr.wm.hits + aPx += cr.am.sumPx; aHit += cr.am.hits + shN += cr.wm.n + echo fmt" {regimeName(regime):<9} shuffled mean: wm meanPx={wPx/float(shN):.2f} pxHit%={100.0*float(wHit)/float(shN):.1f} " & + fmt"arg meanPx={aPx/float(shN):.2f} pxHit%={100.0*float(aHit)/float(shN):.1f}" + echo "" + + stderr.writeLine(fmt"[total] {epochTime()-t0:.1f}s") + +when isMainModule: + main() diff --git a/common_libs/tests/measure_bitbrain_gate_results.txt b/common_libs/tests/measure_bitbrain_gate_results.txt new file mode 100644 index 0000000..892f388 --- /dev/null +++ b/common_libs/tests/measure_bitbrain_gate_results.txt @@ -0,0 +1,194 @@ + +================================================================= +BitBrain gate test - fine-grained aim correction +================================================================= +fixtures : 5 tr_drussgt_vs_* (open-loop replay) +power=2.0 speed=14.0 class half-range=+-40.0 deg +AD: widths [6, 8, 10, 12] nAde=@[128, 256] target=1.0% passes=2 step=1 stride=3 + +== dataset (effective sample counts) == + tr_drussgt_vs_corners.jsonl samples=2173 + tr_drussgt_vs_crazy.jsonl samples=11209 + tr_drussgt_vs_modularbot.jsonl samples=19548 + tr_drussgt_vs_modularbot_shield.jsonl samples=12308 + tr_drussgt_vs_spinbot.jsonl samples=10494 + |label err| deg: mean=15.91 p50=11.55 p90=37.19 p99=51.99 max=63.16 + rounds=55 samples/round=1013. + +== baselines (same arrival geometry, same sample set) == +Pattern (zero corr) n=55732 meanPx=139.78 medPx=112.71 p90Px=299.10 meanDeg= 15.56 medDeg= 11.73 pxHit%= 7.0 angHit%= 19.1 +straight-line naive n=55732 meanPx=160.17 medPx=131.57 p90Px=336.48 meanDeg= 16.01 medDeg= 12.18 pxHit%= 5.0 angHit%= 18.8 +fixed corr=+0.20 deg n=55732 meanPx=139.76 medPx=112.66 p90Px=298.85 meanDeg= 15.56 medDeg= 11.74 pxHit%= 7.0 angHit%= 19.0 + +== label histogram (N=64) == + class 0 [ -40.0, -38.8) n=2365 + class 1 [ -38.8, -37.5) n=303 + class 2 [ -37.5, -36.2) n=364 + class 3 [ -36.2, -35.0) n=377 + class 4 [ -35.0, -33.8) n=400 + class 5 [ -33.8, -32.5) n=426 + class 6 [ -32.5, -31.2) n=430 + class 7 [ -31.2, -30.0) n=437 + class 8 [ -30.0, -28.8) n=475 + class 9 [ -28.8, -27.5) n=515 + class 10 [ -27.5, -26.2) n=558 + class 11 [ -26.2, -25.0) n=571 + class 12 [ -25.0, -23.8) n=546 + class 13 [ -23.8, -22.5) n=555 + class 14 [ -22.5, -21.2) n=526 + class 15 [ -21.2, -20.0) n=553 + class 16 [ -20.0, -18.8) n=560 + class 17 [ -18.8, -17.5) n=577 + class 18 [ -17.5, -16.2) n=611 + class 19 [ -16.2, -15.0) n=651 + class 20 [ -15.0, -13.8) n=713 + class 21 [ -13.8, -12.5) n=731 + class 22 [ -12.5, -11.2) n=744 + class 23 [ -11.2, -10.0) n=807 + class 24 [ -10.0, -8.8) n=916 + class 25 [ -8.8, -7.5) n=1109 + class 26 [ -7.5, -6.2) n=1260 + class 27 [ -6.2, -5.0) n=1362 + class 28 [ -5.0, -3.8) n=1668 + class 29 [ -3.8, -2.5) n=1809 + class 30 [ -2.5, -1.2) n=1977 + class 31 [ -1.2, +0.0) n=2155 + class 32 [ +0.0, +1.2) n=2768 + class 33 [ +1.2, +2.5) n=2161 + class 34 [ +2.5, +3.8) n=1847 + class 35 [ +3.8, +5.0) n=1721 + class 36 [ +5.0, +6.2) n=1484 + class 37 [ +6.2, +7.5) n=1428 + class 38 [ +7.5, +8.8) n=1098 + class 39 [ +8.8, +10.0) n=1004 + class 40 [ +10.0, +11.2) n=893 + class 41 [ +11.2, +12.5) n=852 + class 42 [ +12.5, +13.8) n=727 + class 43 [ +13.8, +15.0) n=705 + class 44 [ +15.0, +16.2) n=712 + class 45 [ +16.2, +17.5) n=669 + class 46 [ +17.5, +18.8) n=629 + class 47 [ +18.8, +20.0) n=587 + class 48 [ +20.0, +21.2) n=571 + class 49 [ +21.2, +22.5) n=597 + class 50 [ +22.5, +23.8) n=528 + class 51 [ +23.8, +25.0) n=520 + class 52 [ +25.0, +26.2) n=489 + class 53 [ +26.2, +27.5) n=490 + class 54 [ +27.5, +28.8) n=486 + class 55 [ +28.8, +30.0) n=456 + class 56 [ +30.0, +31.2) n=432 + class 57 [ +31.2, +32.5) n=470 + class 58 [ +32.5, +33.8) n=423 + class 59 [ +33.8, +35.0) n=431 + class 60 [ +35.0, +36.2) n=399 + class 61 [ +36.2, +37.5) n=384 + class 62 [ +37.5, +38.8) n=332 + class 63 [ +38.8, +40.0) n=2388 + +== AD layer (synthesised on this data, homeostasis to ~1% target) == + nAde=128 pct-init: w6=0.47% w8=0.56% w10=0.66% w12=0.70% (mean=0.59%) + nAde=128 +homeostasis:w6=1.20% w8=1.50% w10=1.34% w12=1.32% (mean=1.34%) + nAde=256 pct-init: w6=0.37% w8=0.50% w10=0.62% w12=0.69% (mean=0.54%) + nAde=256 +homeostasis:w6=1.16% w8=1.30% w10=1.31% w12=1.47% (mean=1.31%) + +== prequential sweep (predict-then-learn) == + BitBrain wm = count-weighted mean of class centers (KEY readout) + BitBrain arg = argmax class center + vs Pattern / straight-line / fixed baselines above + + nAde=128 N=4 retained | wm n=55732 meanPx=136.93 medPx=110.25 p90Px=288.15 meanDeg= 15.25 medDeg= 11.48 pxHit%= 4.8 angHit%= 15.2 + argmax | argmax n=55732 meanPx=166.79 medPx=127.55 p90Px=347.39 meanDeg= 19.33 medDeg= 13.81 pxHit%= 2.4 angHit%= 9.4 + nAde=128 N=4 perRound | wm n=55732 meanPx=130.27 medPx=104.94 p90Px=270.86 meanDeg= 14.37 medDeg= 10.86 pxHit%= 4.3 angHit%= 13.7 + argmax | argmax n=55732 meanPx=140.28 medPx=103.16 p90Px=299.10 meanDeg= 15.72 medDeg= 10.64 pxHit%= 3.0 angHit%= 11.6 + nAde=128 N=8 retained | wm n=55732 meanPx=136.21 medPx=109.65 p90Px=287.48 meanDeg= 15.13 medDeg= 11.39 pxHit%= 4.9 angHit%= 15.1 + argmax | argmax n=55732 meanPx=162.26 medPx=122.44 p90Px=349.71 meanDeg= 18.58 medDeg= 12.97 pxHit%= 3.5 angHit%= 12.9 + nAde=128 N=8 perRound | wm n=55732 meanPx=128.87 medPx=103.50 p90Px=270.52 meanDeg= 14.12 medDeg= 10.51 pxHit%= 4.5 angHit%= 14.7 + argmax | argmax n=55732 meanPx=134.81 medPx= 95.93 p90Px=302.68 meanDeg= 14.77 medDeg= 8.58 pxHit%= 4.7 angHit%= 17.1 + nAde=128 N=16 retained | wm n=55732 meanPx=135.94 medPx=109.36 p90Px=287.13 meanDeg= 15.09 medDeg= 11.35 pxHit%= 5.4 angHit%= 15.4 + argmax | argmax n=55732 meanPx=161.08 medPx=120.64 p90Px=350.87 meanDeg= 18.37 medDeg= 12.58 pxHit%= 5.0 angHit%= 16.2 + nAde=128 N=16 perRound | wm n=55732 meanPx=128.78 medPx=103.42 p90Px=271.12 meanDeg= 14.08 medDeg= 10.46 pxHit%= 5.2 angHit%= 15.6 + argmax | argmax n=55732 meanPx=133.72 medPx= 94.38 p90Px=304.44 meanDeg= 14.55 medDeg= 8.33 pxHit%= 6.7 angHit%= 21.7 + nAde=128 N=32 retained | wm n=55732 meanPx=135.84 medPx=109.25 p90Px=286.59 meanDeg= 15.07 medDeg= 11.33 pxHit%= 5.4 angHit%= 15.6 + argmax | argmax n=55732 meanPx=161.27 medPx=120.06 p90Px=352.17 meanDeg= 18.37 medDeg= 12.52 pxHit%= 5.7 angHit%= 17.6 + nAde=128 N=32 perRound | wm n=55732 meanPx=128.83 medPx=103.30 p90Px=271.66 meanDeg= 14.09 medDeg= 10.44 pxHit%= 5.2 angHit%= 15.8 + argmax | argmax n=55732 meanPx=133.97 medPx= 93.91 p90Px=307.63 meanDeg= 14.56 medDeg= 8.28 pxHit%= 7.5 angHit%= 23.5 + nAde=128 N=64 retained | wm n=55732 meanPx=135.87 medPx=109.28 p90Px=287.03 meanDeg= 15.07 medDeg= 11.33 pxHit%= 5.5 angHit%= 15.6 + argmax | argmax n=55732 meanPx=162.34 medPx=120.83 p90Px=354.80 meanDeg= 18.50 medDeg= 12.57 pxHit%= 5.8 angHit%= 17.7 + nAde=128 N=64 perRound | wm n=55732 meanPx=128.97 medPx=103.37 p90Px=272.07 meanDeg= 14.10 medDeg= 10.47 pxHit%= 5.2 angHit%= 15.9 + argmax | argmax n=55732 meanPx=134.67 medPx= 94.30 p90Px=309.47 meanDeg= 14.64 medDeg= 8.30 pxHit%= 7.6 angHit%= 23.6 + nAde=256 N=4 retained | wm n=55732 meanPx=135.35 medPx=109.18 p90Px=284.46 meanDeg= 15.04 medDeg= 11.24 pxHit%= 4.8 angHit%= 15.0 + argmax | argmax n=55732 meanPx=156.83 medPx=115.59 p90Px=331.56 meanDeg= 18.05 medDeg= 12.22 pxHit%= 2.0 angHit%= 8.5 + nAde=256 N=4 perRound | wm n=55732 meanPx=126.10 medPx=101.89 p90Px=259.70 meanDeg= 13.81 medDeg= 10.37 pxHit%= 4.0 angHit%= 13.4 + argmax | argmax n=55732 meanPx=133.30 medPx= 97.76 p90Px=283.16 meanDeg= 14.82 medDeg= 10.16 pxHit%= 2.6 angHit%= 10.6 + nAde=256 N=8 retained | wm n=55732 meanPx=134.60 medPx=108.62 p90Px=283.12 meanDeg= 14.92 medDeg= 11.15 pxHit%= 4.9 angHit%= 15.1 + argmax | argmax n=55732 meanPx=150.62 medPx=108.97 p90Px=331.24 meanDeg= 17.03 medDeg= 10.73 pxHit%= 3.6 angHit%= 13.3 + nAde=256 N=8 perRound | wm n=55732 meanPx=124.78 medPx=101.15 p90Px=257.08 meanDeg= 13.58 medDeg= 10.21 pxHit%= 4.3 angHit%= 14.3 + argmax | argmax n=55732 meanPx=125.60 medPx= 85.82 p90Px=287.09 meanDeg= 13.54 medDeg= 7.27 pxHit%= 4.7 angHit%= 17.6 + nAde=256 N=16 retained | wm n=55732 meanPx=134.31 medPx=108.35 p90Px=282.05 meanDeg= 14.87 medDeg= 11.12 pxHit%= 5.4 angHit%= 15.5 + argmax | argmax n=55732 meanPx=149.00 medPx=106.72 p90Px=332.91 meanDeg= 16.75 medDeg= 10.45 pxHit%= 5.6 angHit%= 17.9 + nAde=256 N=16 perRound | wm n=55732 meanPx=124.72 medPx=101.16 p90Px=257.90 meanDeg= 13.56 medDeg= 10.21 pxHit%= 4.8 angHit%= 14.9 + argmax | argmax n=55732 meanPx=124.46 medPx= 84.11 p90Px=291.12 meanDeg= 13.28 medDeg= 6.90 pxHit%= 7.3 angHit%= 23.7 + nAde=256 N=32 retained | wm n=55732 meanPx=134.23 medPx=108.09 p90Px=281.94 meanDeg= 14.86 medDeg= 11.12 pxHit%= 5.3 angHit%= 15.3 + argmax | argmax n=55732 meanPx=149.43 medPx=106.75 p90Px=335.62 meanDeg= 16.77 medDeg= 10.31 pxHit%= 6.5 angHit%= 19.8 + nAde=256 N=32 perRound | wm n=55732 meanPx=124.95 medPx=101.33 p90Px=258.34 meanDeg= 13.58 medDeg= 10.23 pxHit%= 4.9 angHit%= 15.0 + argmax | argmax n=55732 meanPx=124.62 medPx= 83.51 p90Px=293.48 meanDeg= 13.26 medDeg= 6.74 pxHit%= 8.3 angHit%= 26.1 + nAde=256 N=64 retained | wm n=55732 meanPx=134.30 medPx=108.26 p90Px=282.48 meanDeg= 14.87 medDeg= 11.13 pxHit%= 5.3 angHit%= 15.2 + argmax | argmax n=55732 meanPx=151.60 medPx=107.64 p90Px=341.44 meanDeg= 17.05 medDeg= 10.44 pxHit%= 6.5 angHit%= 19.9 + nAde=256 N=64 perRound | wm n=55732 meanPx=125.13 medPx=101.31 p90Px=259.01 meanDeg= 13.61 medDeg= 10.21 pxHit%= 5.0 angHit%= 15.0 + argmax | argmax n=55732 meanPx=125.34 medPx= 83.61 p90Px=296.28 meanDeg= 13.35 medDeg= 6.72 pxHit%= 8.5 angHit%= 26.5 + +== per-fixture breakdown: perRound, N=32, vs Pattern ========== + tr_drussgt_vs_corners.jsonl n= 2173 + Pattern n=2173 meanPx=164.76 medPx=139.78 p90Px=325.54 meanDeg= 14.40 medDeg= 10.50 pxHit%= 3.9 angHit%= 18.5 + BitBrain wm n=2173 meanPx=134.00 medPx=109.26 p90Px=268.02 meanDeg= 10.33 medDeg= 7.21 pxHit%= 1.7 angHit%= 16.7 + BitBrain arg n=2173 meanPx=132.64 medPx=105.90 p90Px=280.25 meanDeg= 10.16 medDeg= 5.94 pxHit%= 3.6 angHit%= 23.8 + tr_drussgt_vs_crazy.jsonl n= 11209 + Pattern n=11209 meanPx=126.34 medPx= 97.68 p90Px=280.34 meanDeg= 13.65 medDeg= 8.48 pxHit%= 7.3 angHit%= 25.8 + BitBrain wm n=11209 meanPx=113.68 medPx= 91.36 p90Px=233.59 meanDeg= 11.90 medDeg= 8.27 pxHit%= 4.5 angHit%= 19.1 + BitBrain arg n=11209 meanPx=111.95 medPx= 79.22 p90Px=248.50 meanDeg= 11.55 medDeg= 6.37 pxHit%= 5.9 angHit%= 24.6 + tr_drussgt_vs_modularbot.jsonl n= 19548 + Pattern n=19548 meanPx=151.80 medPx=129.09 p90Px=304.18 meanDeg= 17.21 medDeg= 14.39 pxHit%= 3.3 angHit%= 10.5 + BitBrain wm n=19548 meanPx=134.78 medPx=112.25 p90Px=269.34 meanDeg= 14.89 medDeg= 12.05 pxHit%= 3.3 angHit%= 11.0 + BitBrain arg n=19548 meanPx=134.46 medPx= 91.39 p90Px=312.12 meanDeg= 14.43 medDeg= 7.95 pxHit%= 7.5 angHit%= 25.1 + tr_drussgt_vs_modularbot_shield.jsonl n= 12308 + Pattern n=12308 meanPx=144.28 medPx=121.48 p90Px=297.04 meanDeg= 16.60 medDeg= 13.80 pxHit%= 6.6 angHit%= 14.1 + BitBrain wm n=12308 meanPx=128.52 medPx=106.53 p90Px=263.24 meanDeg= 14.45 medDeg= 11.56 pxHit%= 6.9 angHit%= 14.6 + BitBrain arg n=12308 meanPx=124.98 medPx= 81.72 p90Px=299.54 meanDeg= 13.56 medDeg= 6.64 pxHit%= 10.5 angHit%= 28.5 + tr_drussgt_vs_spinbot.jsonl n= 10494 + Pattern n=10494 meanPx=121.29 medPx= 79.94 p90Px=299.51 meanDeg= 13.55 medDeg= 6.58 pxHit%= 15.0 angHit%= 33.7 + BitBrain wm n=10494 meanPx=112.59 medPx= 85.32 p90Px=248.00 meanDeg= 12.62 medDeg= 8.52 pxHit%= 6.7 angHit%= 18.4 + BitBrain arg n=10494 meanPx=117.72 medPx= 72.72 p90Px=290.16 meanDeg= 13.19 medDeg= 5.80 pxHit%= 10.6 angHit%= 27.3 + +== compact: mean px error (lower is better) == + config wm meanPx arg meanPx wm hit% arg hit% + Pattern 139.78 - 7.0 - + straight-line 160.17 - 5.0 - + fixed corr 139.76 - 7.0 - + nAde128/N4/retained 136.93 166.79 4.8 2.4 + nAde128/N4/perRound 130.27 140.28 4.3 3.0 + nAde128/N8/retained 136.21 162.26 4.9 3.5 + nAde128/N8/perRound 128.87 134.81 4.5 4.7 + nAde128/N16/retained 135.94 161.08 5.4 5.0 + nAde128/N16/perRound 128.78 133.72 5.2 6.7 + nAde128/N32/retained 135.84 161.27 5.4 5.7 + nAde128/N32/perRound 128.83 133.97 5.2 7.5 + nAde128/N64/retained 135.87 162.34 5.5 5.8 + nAde128/N64/perRound 128.97 134.67 5.2 7.6 + nAde256/N4/retained 135.35 156.83 4.8 2.0 + nAde256/N4/perRound 126.10 133.30 4.0 2.6 + nAde256/N8/retained 134.60 150.62 4.9 3.6 + nAde256/N8/perRound 124.78 125.60 4.3 4.7 + nAde256/N16/retained 134.31 149.00 5.4 5.6 + nAde256/N16/perRound 124.72 124.46 4.8 7.3 + nAde256/N32/retained 134.23 149.43 5.3 6.5 + nAde256/N32/perRound 124.95 124.62 4.9 8.3 + nAde256/N64/retained 134.30 151.60 5.3 6.5 + nAde256/N64/perRound 125.13 125.34 5.0 8.5 + +== shuffled-label control (permuted labels, same input distribution) == + (the weighted mean shrinks to the label mean; the argmax keeps making a + confident pick, so its null is random-correction worse-than-baseline) + retained shuffled mean: wm meanPx=141.51 pxHit%=5.8 arg meanPx=200.10 pxHit%=2.4 + perRound shuffled mean: wm meanPx=146.74 pxHit%=4.6 arg meanPx=193.22 pxHit%=2.4 + diff --git a/docs/bitbrain_gate_test.md b/docs/bitbrain_gate_test.md new file mode 100644 index 0000000..ce240aa --- /dev/null +++ b/docs/bitbrain_gate_test.md @@ -0,0 +1,311 @@ +# BitBrain gate test — can ADE+SBC predict a fine-grained aim correction? + +**Question.** Given the already-built generic BitBrain library +(`common_libs/bitbrain/`, commit `77e6dac`, which reproduced the reference C on +MNIST to the digit), is there signal in a **fine-grained angular aim +correction** that neither a naive predictor nor the shipped Pattern gun already +has? If not, we stop before writing a gun. + +**Scope.** Offline only. No gun wiring, no battle, no server, no rack +registration, no changed defaults. Fixtures are read-only. + +**Tooling (the evidence):** + +- `common_libs/tests/measure_bitbrain_gate.nim` — the analyzer. +- `common_libs/tests/measure_bitbrain_gate_results.txt` — its full deterministic + output. +- `common_libs/bitbrain/` — the library under test (untouched). + +Reproduce: + +```bash +nim c -r -d:release --nimcache:/tmp/nc_j92 --path:common_libs \ + common_libs/tests/measure_bitbrain_gate.nim +# optional knobs: BB_POWER, BB_PMAX, BB_TARGET, BB_PASSES, BB_STEP, BB_STRIDE, +# BB_NADE, BB_NLIST, BB_FILES +``` + +--- + +## Direct answer + +**There is signal, but it does not clear the bar as a shipping gun.** + +- **vs straight-line naive:** BitBrain wins decisively on **every** readout and + every configuration (pooled mean arrival error **124.7 px vs 160.2 px**). +- **vs the shipped Pattern gun:** pooled, the best BitBrain configuration + (per-round reset, argmax readout, N=32, nAde=256) beats Pattern on **mean + arrival error** (124.6 px vs 139.8 px, −10.8 %; 13.3° vs 15.6°, −14.8 %) and + on **both hit rates** (18 px hit 8.3 % vs 7.0 %; angular hit 26.1 % vs + 19.1 %). **However** the hit-rate gain is **not robust across fixtures**: it + is concentrated in the two fixtures where Pattern is weak (modularbot, + modularbot_shield) and **BitBrain loses hit rate on the two fixtures where + Pattern is strongest** (spinbot, crazy). It also only works in the + **per-round-reset** regime; the **retained-across-rounds** regime the user + actually wants is the weakest (it improves average error slightly but lowers + the hit rate). +- **Bottleneck (MEASURED):** sample starvation / SBC memory saturation, not the + AD synthesis and not an absence of signal. + +A sword that helps exactly where the incumbent is already weak, and hurts where +it is strong, is not a gun improvement. The honest verdict is **signal yes, +shippable improvement no (yet)** — see the bottleneck section. + +--- + +## What was measured (MEASURED unless tagged INFERRED) + +### Data + +The 5 committed Tank-Royale bridge fixtures `tools/fixtures/tr_drussgt_vs_*` +(open-loop replay), read-only: + +| fixture | ticks | resolved samples | +|---|---:|---:| +| `tr_drussgt_vs_corners.jsonl` | 2575 | 2173 | +| `tr_drussgt_vs_crazy.jsonl` | 11507 | 11209 | +| `tr_drussgt_vs_modularbot.jsonl` | 20026 | 19548 | +| `tr_drussgt_vs_modularbot_shield.jsonl` | 12629 | 12308 | +| `tr_drussgt_vs_spinbot.jsonl` | 10824 | 10494 | +| **total** | | **55732** (55 rounds, ~1013 samples/round) | + +The fixtures are **open-loop**: the recorded enemy does not react to us. That is +acceptable *here* because this test measures **single-tick prediction quality**, +the one category where fixture replay reproduces live behaviour faithfully. It +would **not** be acceptable evidence for a movement or adaptation claim. Do not +over-read the hit numbers as live hit rates. + +### Input — the TMHorizon 53 bits, reused, not re-derived + +The analyzer drives the **live `TmHorizonGun`** one tick at a time and reads its +own exported builder `tmhBaseBits` (49 draft bits) plus the 4-bit horizon +one-hot via `tmhLits` — the exact `cachedBits` per-tick path, with +`g.resetRoundState()` called at each round boundary to mirror the live +`onRoundStarted`. (The brief calls this `tmhBuildBits`; the actual symbol is +`tmhBaseBits`.) This keeps the comparison apples-to-apples with the TM gun +already measured. + +### Output — fine-grained angular correction class + +The correction is the bearing offset added to Pattern's prediction. N class +bins are laid over a fixed **±40°** range (class width 80°/N). Two readouts: + +- **wm** = the **count-weighted mean** of the class centres, weighted by the + per-class set-bit counts summed over the 6 cross-AD SBCs. No evidence → 0 + correction (i.e. Pattern). +- **arg** = the argmax class centre; no evidence → 0. + +### Label — the +h-tick fact (never across a round) + +`h = tmhHorizonFor(dist, speed) = clamp(round(dist/(20−3·power)), 10, 50)`. +At fire tick `t`, the label is the actual angular offset of the enemy at +`t+h` (from the same fixture, which under perfect-info replay equals the bot's +own observation ring) relative to Pattern's base bearing. Samples with +`t+h` past the round end are dropped (never a cross-boundary label). Pooled +`|label err|`: mean 15.9°, p50 11.6°, p90 37.2°, p99 52.0°. + +### Metric — arrival aim error, not accuracy + +Harness `bmPoint` geometry, the relation `measure_aim_vs_power.nim` validated +against the harness resolver: `fireDist = |Pattern − self|`, +`arrivalTick = t + ceil(fireDist/v) − 1`. A rotation preserves `fireDist`, so +the corrected aim point is rotated around the shooter. `miss = |aim − actual +enemy pos at arrivalTick|`; hit = `miss < 18 px`. Angular error = `|aim +bearing − actual bearing|`; angular hit = `|angErr| < atan(18/range)`. Timed +resolution happens on `arrivalTick`, which can differ from `t+h` by ≤ half a +tick; the label uses `h`, the metric uses `arrivalTick`, exactly as the brief +specifies. + +**The px metric includes range error**, and because the head is a pure +rotation it cannot fix range; this is why the 18 px hit is dominated by range +error and the angular metric is the cleaner measure of a rotation head. Both +are reported. + +### Protocol — prequential (predict-then-learn, streaming) + +For every sample the model predicts **before** it is updated with the label. +Two regimes: + +- **retained** — learning accumulates across all rounds of one battle + (fixture); reset only when the battle/enemy changes. This is what the user + asked for. +- **perRound** — reset at every round boundary (the worst case). + +### Baselines + +1. **straight-line naive** — enemy keeps its fire-tick velocity over the same + `arrivalTick` window. +2. **always-the-same-answer** — a fixed correction equal to the global mean + label (+0.20°). +3. **Pattern** — the shipped gun's own prediction (zero correction). + +Pooled over all 55732 samples: + +| predictor | meanPx | medPx | p90Px | meanDeg | medDeg | pxHit% | angHit% | +|---|---:|---:|---:|---:|---:|---:|---:| +| Pattern (zero corr) | 139.78 | 112.71 | 299.10 | 15.56 | 11.73 | 7.0 | 19.1 | +| straight-line naive | 160.17 | 131.57 | 336.48 | 16.01 | 12.18 | 5.0 | 18.8 | +| fixed (+0.20°) | 139.76 | 112.66 | 298.85 | 15.56 | 11.74 | 7.0 | 19.0 | + +--- + +## The AD layer (synthesised for our data) + +Random ADs (`initRandomAddressDecoder`, widths {6, 8, 10, 12}, **center = 0**), +then the deterministic homeostatic controller (`accumulateFiring` + +`adaptThresholds`, target 1 %). **center = 0 is forced by our data:** the +reference's 127 is the midpoint of 0..255; centring **binary** 0/1 inputs at 127 +makes every synapse contribute ≈ −127 and collapses the ADE code to a mere +polarity count, destroying the signal. + +The paper's `step = 1` controller would need thousands of intervals to find the +1 % operating point — **far more than a battle (500–2000 ticks) provides**. +This is itself the first measured symptom of sample starvation. The analyzer +therefore initialises each ADE's threshold at the score that puts it closest to +the 1 % firing count (a fast, unsupervised percentile), then runs the +deterministic controller (`step = 1`, 2 passes) to refine it. Achieved firing +rates (MEASURED, on the fit stride): + +| nAde | w6 | w8 | w10 | w12 | mean | +|---:|---:|---:|---:|---:|---:| +| 128, pct-init | 0.47 % | 0.56 % | 0.66 % | 0.70 % | 0.59 % | +| 128, +homeostasis | 1.20 % | 1.50 % | 1.34 % | 1.32 % | **1.34 %** | +| 256, pct-init | 0.37 % | 0.50 % | 0.62 % | 0.69 % | 0.54 % | +| 256, +homeostasis | 1.16 % | 1.30 % | 1.31 % | 1.47 % | **1.31 %** | + +So the paper's ~1 % operating point **is** reached. (The controller alone +overshot in an earlier pass at `step = 2`; `step = 1` lands it.) + +--- + +## Sweep — N × AD size × regime (pooled, mean px error) + +`wm` = count-weighted mean, `arg` = argmax. `hit%` is the 18 px arrival hit. + +| config | wm meanPx | arg meanPx | wm pxHit% | arg pxHit% | +|---|---:|---:|---:|---:| +| Pattern | 139.78 | — | 7.0 | — | +| straight-line | 160.17 | — | 5.0 | — | +| fixed | 139.76 | — | 7.0 | — | +| nAde128 / N4 / retained | 136.93 | 166.79 | 4.8 | 2.4 | +| nAde128 / N4 / perRound | 130.27 | 140.28 | 4.3 | 3.0 | +| nAde128 / N8 / perRound | 128.87 | 134.81 | 4.5 | 4.7 | +| nAde128 / N16 / perRound | 128.78 | 133.72 | 5.2 | 6.7 | +| nAde128 / N32 / perRound | 128.83 | 133.97 | 5.2 | 7.5 | +| nAde128 / N64 / perRound | 128.97 | 134.67 | 5.2 | 7.6 | +| nAde256 / N8 / perRound | 124.78 | 125.60 | 4.3 | 4.7 | +| nAde256 / N16 / perRound | 124.72 | 124.46 | 4.8 | 7.3 | +| **nAde256 / N32 / perRound** | 124.95 | **124.62** | 4.9 | **8.3** | +| nAde256 / N64 / perRound | 125.13 | 125.34 | 5.0 | 8.5 | +| nAde256 / N16 / retained | 134.31 | 149.00 | 5.4 | 5.6 | +| nAde256 / N32 / retained | 134.23 | 149.43 | 5.3 | 6.5 | +| nAde256 / N64 / retained | 134.30 | 151.60 | 5.3 | 6.5 | + +Full per-config degrees/p90/angular-hit rows are in +`measure_bitbrain_gate_results.txt`. + +### Where the error stops falling, and why + +- **N:** the `wm` error is flat from N=8 to N=64 (≈124.7–125.1 px); the `arg` + error falls to N=16–32 then flattens. **Optimum N ≈ 16–32.** Beyond it the + correction classes get finer than the loop can resolve, and the SBC + coincidence cells are already too few to constrain their class bits — more + classes only split the same evidence. +- **AD size:** nAde=256 beats 128 by a modest ~3 % in `wm`; both are far from + saturating, but the classes saturate first. Doubling the ADE count does not + double the information. +- **Why it stops:** MNIST needed ~60 000 examples for **10** mutually exclusive + classes. Here we have ~55 000 samples for **16–64** correction classes whose + evidence must separate by 1–2° — i.e. ~2 orders of magnitude less evidence per + class. The SBC is idempotent (a cell accumulates *every* class that ever + co-occurred, with no decay), so with too few examples per cell the per-class + counts blur toward uniform and the readout regresses toward the mean. That is + **sample starvation / memory saturation**, and it is consistent with every + other observation (flat N tail, weak nAde scaling, retained < perRound). + +### Count-weighted mean vs argmax (the brief's hypothesis) + +The brief expected the **count-weighted mean** to be the key readout because +the TM's discarded magnitude. **MEASURED, that is only half right:** + +- The `wm` is the **shrinkage** readout: it reduces *mean* error (and extreme + misses) but **lowers the hit rate** (pooled pxHit 4.8 % vs Pattern 7.0 %, + angular hit 14.9 % vs 19.1 %). It never makes a confident, sharp correction. +- The `argmax` is the **decision** readout: it keeps the same mean-error + reduction *and* improves the hit rate (pxHit 8.3 %, angular hit 26.1 %). It is + the readout that beats Pattern on all four metrics. + +So the fine-grained head works, but as a **classifier** (argmax), not as a soft +regression (weighted mean). The weighted mean is a useful control: it is the +readout whose shuffled-label null collapses to the baseline. + +### Retained vs per-round reset + +**Per-round reset beats retained across rounds on every readout and every N** +(retained `wm` ≈ 134.2 px, retained `arg` ≈ 149–162 px; per-round `wm` ≈ 124.7, +per-round `arg` ≈ 124.5). The user wants retention across the battle; the +measurement says the idempotent SBC **accumulates stale, conflicting class bits +across rounds** and the extra evidence hurts. This mirrors the project's earlier +TM finding ("forgetting is stronger than accumulation"). A viable gun would +need a bounded/decaying SBC, which the library does not have. + +### Per-fixture breakdown (perRound, N=32, argmax) + +| fixture (n) | Pattern meanPx / pxHit% / angHit% | BitBrain arg meanPx / pxHit% / angHit% | +|---|---|---| +| corners (2173) | 164.76 / 3.9 / 18.5 | 132.64 / 3.6 / 23.8 | +| crazy (11209) | 126.34 / 7.3 / 25.8 | 111.95 / **5.9** / **24.6** | +| modularbot (19548) | 151.80 / 3.3 / 10.5 | 134.46 / **7.5** / **25.1** | +| shield (12308) | 144.28 / 6.6 / 14.1 | 124.98 / **10.5** / **28.5** | +| spinbot (10494) | 121.29 / 15.0 / 33.7 | 117.72 / **10.6** / **27.3** | + +Mean error improves on **all five**. Hit rate improves on modularbot and shield +(where Pattern is weak) and **regresses on spinbot and crazy** (where Pattern is +strong), with corners a wash. That is the whole verdict in one table. + +### Shuffled-label control (must collapse) + +Labels permuted across all samples (3 seeds), same inputs: + +| regime | wm shuffled meanPx / pxHit% | arg shuffled meanPx / pxHit% | +|---|---|---| +| retained | 141.51 / 5.8 | 200.10 / 2.4 | +| perRound | 146.74 / 4.6 | 193.22 / 2.4 | + +The **weighted mean collapses toward the baseline** (141.5 vs Pattern 139.8) — +expected, because it shrinks to the (near-zero) label mean. The **argmax does +not collapse to the baseline: its null is worse than the baseline** — with no +signal it still makes a confident, essentially random rotation, which is worse +than no correction. That is the correct null behaviour for a non-shrinking +readout, and it is why the honest control is **real vs shuffled within the same +readout**: BitBrain argmax is ~124.6 px on real labels vs ~193 px on shuffled +labels. The learning is real; the signal is not an artifact. + +--- + +## Bottleneck and recommendation (MEASURED) + +Ranked by how much each could plausibly close the gap: + +1. **Sample starvation / SBC memory saturation — the dominant one.** N + saturates at ~16–32, nAde barely scales, and per-round reset beats retention. + The library has no bounded/decaying SBC, so a long battle only blurs. +2. **Fixture-dependent gain.** The pooled hit win is carried by the + weak-Pattern fixtures. Without an online per-fixture selector, a blanket + substitution would lose on spinbot/crazy. +3. **AD synthesis is *not* the bottleneck.** The ~1 % operating point is + reached and the shuffle control shows the ADs are informative. The forced + `center = 0` for binary inputs is a correctness requirement, not a defect. + +**Do not build the gun yet.** The cheap decisive next step, if pursued, is a +**bounded/decaying SBC** (a per-round or recency-weighted memory) plus an +**online selection gate** that keeps Pattern where BitBrain is worse — the only +shape the data supports. A wider class range or a larger nAde will not fix the +starvation. + +--- + +*All numbers MEASURED by `common_libs/tests/measure_bitbrain_gate.nim` on this +machine, deterministic (fixed seeds). Arrival geometry is the harness `bmPoint` +relation validated in `measure_aim_vs_power.nim`. The fixtures are open-loop; +treat the hit rates as prediction-quality evidence only.*