## SCALING SWEEP for the ADE+SBC gun (rack id 17) — RAM vs ms/tick vs quality. ## ## THE QUESTION: "how do inference, timing and size of RAM scale with input and ## output size?" This answers it on the REAL gun by replaying a recorded live ## corpus through it, not from theory. ## ## For each setting it reports: ## RAM — `memoryBytes` (ADs + SBC tensors) and the SBC tensor alone, ## with the counted-vs-bitset factor spelled out; ## ms/tick — wall time per RECORDED TICK over the real replay, against the ## project's 13.16 ms/tick budget. One `predict` per power bin ## per tick, which is what the live loop does, so the number ## is directly comparable with the budget; ## quality — the OFFLINE ruler: mean |angular error| against the true ## interception point (`gun_harness/prediction_quality`), plus ## the hit proxy (|err| <= atan(18/range)). ## ## VETO-CAPABLE CHECK ONLY. Per `docs/offline_harness_trust.md` the offline ## harness is trustworthy for per-gun, single-tick prediction quality on a FIXED ## trajectory and for NOTHING that flows through the closed loop. A win here is ## NOT a live win and is never presented as one. ## ## Usage: ## nim c -r -d:release --path:common_libs common_libs/tests/measure_bitbrain_scaling.nim \ ## [--corpus /tmp/tfil_ab2/out] [--limit N] ## ## Env knobs are set per SETTING by this program (putEnv), so the sweep is a ## pure-env experiment: no recompile between arms. import std/[os, strformat, strutils, times, math, sequtils, algorithm] import gun_harness/gun_interface import gun_harness/prediction_quality import gun_harness/virtual_bullets import guns/pattern_matcher import guns/bitbrain_net const BudgetMsPerTick* = 13.16 ## the project's live tick budget type Setting = object label: string inputWidth: int nClasses: int nAde: int widths: string features: string mode: string Row = object s: Setting ramBytes: int inputWidth: int sbcBytes: int adBytes: int msPerTick: float meanAbsDeg: float hitProxy: float patternAbsDeg: float n: int # ── the sweep grid ─────────────────────────────────────────────────────────── const FullFeatures = "epos:4,evel:3,eturn:2,eself:2,dist:5,bear:4,walls:2,bull:2,hzn:4" proc smallFeatures(): string = "epos:2,evel:2,eturn:1,eself:1,dist:2,bear:2,walls:1,bull:1,hzn:2" proc largeFeatures(): string = "epos:6,evel:5,eturn:4,eself:4,dist:8,bear:6,walls:3,bull:3,hzn:6" proc settings(): seq[Setting] = ## Three points along the INPUT axis (classes/geometry held at the default) and ## three along the OUTPUT axis (input held at the default 52), all in BOTH ## storage modes, because the counted/bitset factor is part of the answer. let base = Setting(label: "default", inputWidth: 0, nClasses: 8, nAde: 256, widths: "4,5,6", features: FullFeatures, mode: "counted") var inp: seq[Setting] for (lbl, feat, w) in [("input-small", smallFeatures(), 0), ("input-medium", FullFeatures, 0), ("input-large", largeFeatures(), 0)]: inp.add Setting(label: lbl, inputWidth: w, nClasses: 8, nAde: 256, widths: "4,5,6", features: feat, mode: "counted") var outp: seq[Setting] for (lbl, nc) in [("classes-small", 2), ("classes-medium", 8), ("classes-large", 64)]: outp.add Setting(label: lbl, inputWidth: 0, nClasses: nc, nAde: 256, widths: "4,5,6", features: FullFeatures, mode: "counted") # the nAde axis is the third one, because RAM is quadratic in it var ade: seq[Setting] for (lbl, n) in [("nAde-small", 64), ("nAde-medium", 256), ("nAde-large", 512)]: ade.add Setting(label: lbl, inputWidth: 0, nClasses: 8, nAde: n, widths: "4,5,6", features: FullFeatures, mode: "counted") var both: seq[Setting] for m in ["bitset", "counted"]: both.add Setting(label: "mode-" & m, inputWidth: 0, nClasses: 8, nAde: 256, widths: "4,5,6", features: FullFeatures, mode: m) result = inp & outp & ade & both discard base # ── one arm ────────────────────────────────────────────────────────────────── proc applySetting(s: Setting) = for n in BitbrainNetEnvNames: delEnv(n) putEnv(BBN_NET_ENV, "1") putEnv(BBN_FEATURES_ENV, s.features) if s.inputWidth > 0: putEnv(BBN_INPUT_ENV, $s.inputWidth) putEnv(BBN_CLASSES_ENV, $s.nClasses) putEnv(BBN_NADES_ENV, $s.nAde) putEnv(BBN_WIDTHS_ENV, s.widths) putEnv(BBN_MODE_ENV, s.mode) putEnv(BBN_MINOBS_ENV, "1") putEnv(BBN_DECAY_EVERY_ENV, "64") putEnv(BBN_DECAY_SHIFT_ENV, "3") # ── the corpus, turned once into a fixed sample set ───────────────────────── # # The interception solve (the ruler) is INDEPENDENT of the arm, so it is done # ONCE and cached. That does two things: every arm is scored on byte-identical # labels, and the timed region contains ONLY the gun's `predict` calls — the # ruler's own cost cannot contaminate the ms/tick number. type Sample = object st: WorldState speed: float targetLead: float range: float tol: float proc buildSamples(runs: seq[string]): seq[Sample] = for rp in runs: let c = loadCorpus(rp) if c.n == 0: continue for r in 0 ..< c.rStart.len: let base = int(c.rStart[r]) - c.base let cnt = int(c.rCount[r]) let iEnd = base + cnt for i in base ..< iEnd: let ox = c.sx(i) let oy = c.sy(i) let localTick = int(c.tick[i]) - int(c.rStart[r]) let baseState = WorldState( arenaWidth: c.arenaW, arenaHeight: c.arenaH, tick: localTick, enemyX: c.ex(i), enemyY: c.ey(i), enemyHeading: c.eh(i), enemySpeed: c.es(i), enemyEnergy: c.ee(i), selfX: ox, selfY: oy, selfHeading: c.sh(i), selfSpeed: c.ss(i), selfEnergy: c.se(i), selfRadarHeading: c.sh(i)) let los = bearingDeg(ox, oy, c.ex(i), c.ey(i)) for bin in 0 ..< len(PowerBins): let speed = bulletSpeed(PowerBins[bin]) let ib = interceptBearing(c, i, iEnd, ox, oy, speed, true) if not ib.ok: continue result.add Sample(st: baseState, speed: speed, targetLead: wrap180(ib.bearing - los), range: ib.range, tol: tolDeg(ib.range)) proc runArm(s: Setting, samples: seq[Sample]): Row = applySetting(s) var g = initBitbrainNetGun() var pat = PatternMatcherGun() var sumAbs = 0.0 var sumPat = 0.0 var hits = 0 var n = 0 # `samples` is ordered round-by-round, so the gun sees rounds in order. var t0 = epochTime() for smp in samples: let bp = predict(g, smp.st, smp.speed) let los = bearingDeg(smp.st.selfX, smp.st.selfY, smp.st.enemyX, smp.st.enemyY) let bl = wrap180(bearingDeg(smp.st.selfX, smp.st.selfY, bp.x, bp.y) - los) let err = abs(wrap180(bl - smp.targetLead)) sumAbs += err if err <= smp.tol: inc hits let pp = predict(pat, smp.st, smp.speed) let pl = wrap180(bearingDeg(smp.st.selfX, smp.st.selfY, pp.x, pp.y) - los) sumPat += abs(wrap180(pl - smp.targetLead)) inc n let elapsed = epochTime() - t0 # ms per recorded TICK: `n` samples over `len(PowerBins)` samples per tick. let ticks = max(1, samples.len div len(PowerBins)) result = Row(s: s, inputWidth: g.inputWidth, ramBytes: g.networkBytes(), sbcBytes: g.sbcBytes(), adBytes: g.networkBytes() - g.sbcBytes(), msPerTick: elapsed * 1000.0 / float(ticks), meanAbsDeg: if n > 0: sumAbs / float(n) else: NaN, hitProxy: if n > 0: float(hits) / float(n) else: NaN, patternAbsDeg: if n > 0: sumPat / float(n) else: NaN, n: n) # ── driver ─────────────────────────────────────────────────────────────────── proc main() = var corpusRoot = "/tmp/tfil_ab2/out" var limit = 4 var i = 1 while i <= paramCount(): case paramStr(i) of "--corpus": inc i; corpusRoot = paramStr(i) of "--limit": inc i; limit = parseInt(paramStr(i)) else: stderr.writeLine("unknown arg: " & paramStr(i)); quit(2) inc i var runs = discoverRuns(corpusRoot) if limit > 0 and runs.len > limit: runs.setLen(limit) if runs.len == 0: stderr.writeLine("no runs under " & corpusRoot); quit(1) echo "=".repeat(118) echo "BITBRAIN (ADE+SBC, rack id 17) SCALING -- RAM / ms-per-tick / offline prediction quality" echo "=".repeat(118) echo fmt"corpus : {corpusRoot} ({runs.len} recorded run(s))" echo fmt"nAde : ADEs per address decoder; the SBC tensor is nAde^2 x nClasses" echo " cells, so RAM is QUADRATIC in nAde and LINEAR in nClasses." echo fmt"budget : {BudgetMsPerTick} ms/tick (one predict per power bin per recorded tick)" echo "quality : mean |angular error| vs the true interception point, over every" echo " tick x power-bin. VETO-CAPABLE OFFLINE CHECK ONLY (docs/offline_harness_trust.md):" echo " a win here is NOT a live win." echo "" stderr.writeLine("building the ruler sample set once from " & $runs.len & " run(s)...") let tSamp = epochTime() let samples = buildSamples(runs) echo fmt"sample set : {samples.len} tick x power-bin samples " & fmt"({samples.len div max(1, len(PowerBins))} recorded ticks) in " & fmt"{epochTime()-tSamp:.1f}s — arm-independent, so the timed region below" echo " contains ONLY the gun's predict calls." echo "" var rows: seq[Row] for s in settings(): rows.add runArm(s, samples) # Pattern is the same on every arm (it never reads the net's knobs), so the # Pattern column is taken from one arm and is identical for all of them. let patRef = rows[1] echo "arm in nCl nAde mode RAM B SBC B AD B " & "ms/tick %bud mean|err| hit% Pattern|err|" echo "-".repeat(118) for r in rows: echo fmt"{r.s.label:<15} {r.inputWidth:>4} {r.s.nClasses:>5} " & fmt"{r.s.nAde:>5} {r.s.mode:<8} {r.ramBytes:>8} {r.sbcBytes:>9} " & fmt"{r.adBytes:>7} {r.msPerTick:>8.3f} {100.0*r.msPerTick/BudgetMsPerTick:>6.2f} " & fmt"{r.meanAbsDeg:>10.3f} {100.0*r.hitProxy:>6.2f} {patRef.patternAbsDeg:>13.3f}" echo "" echo "(in = the resolved input width, i.e. the CONFIGURED feature-block total;" echo " Pattern|err| is the shipped Pattern gun on the same ticks and is identical" echo " across arms, since Pattern never reads any of these knobs.)" for n in BitbrainNetEnvNames: delEnv(n) main()