## TASK 4 — the tm_diag kit against the EXISTING guns/tm_pattern.nim GF head. ## ## Replays the committed DrussGT fixtures through the offline gun range with a ## FRESH, diagCapture-enabled TM per fixture (the live semantics: cold every ## battle, overfit within the battle), then reports: ## * pooled label balance + majority baseline vs the gun's own warm accuracy; ## * the DEAD-INPUT LIST for the actual 40-bit tm_pattern feature set; ## * the top firing positive clauses and the recovered necessary literals. ## ## Offline only. Fixtures are READ-ONLY. ## Run: nim c -r -d:release --path:common_libs common_libs/tests/diag_tm_pattern_offline.nim ## [--fixtures=a,b,c] [--maxsamples=N] import std/[os, strformat, strutils, algorithm, random] import gun_harness/offline_range import guns/tm_pattern import tm_diag/diagnostics const repoRoot = currentSourcePath().parentDir.parentDir.parentDir const fixturesDir = repoRoot / "tools" / "fixtures" proc toDiag(s: TmDiagSample): DiagSample = result.lits = newSeq[uint8](TM_NLITS) for i in 0.. rowSum[maj]: maj = c let majShare = if total > 0: rowSum[maj].float / total.float else: 0.0 let acc = if total > 0: correct.float / total.float else: 0.0 echo &"# pooled warm accuracy = {correct}/{total} = {acc*100:.2f}%" echo &"# majority class = {maj} share/baseline = {rowSum[maj]}/{total} = {majShare*100:.2f}%" echo &"# margin = {(acc-majShare)*100:+.2f}pp pred-majority share = " & &"{colSum[maj].float/max(1,total).float*100:.2f}%" for c in 0.. 0: cm[c][c].float / rowSum[c].float else: 0.0 let prec = if colSum[c] > 0: cm[c][c].float / colSum[c].float else: 0.0 echo &"# class{c}: trueN={rowSum[c]:<6} predN={colSum[c]:<6} TP={cm[c][c]:<6} " & &"recall={rec*100:5.1f}% precision={prec*100:5.1f}%" proc main() = var names = @["drussgt_vs_crazy", "drussgt_vs_spinbot", "drussgt_vs_drussgt", "tr_drussgt_vs_crazy", "tr_drussgt_vs_spinbot", "tr_drussgt_vs_modularbot"] var maxSamples = 40000 for i in 1..paramCount(): let a = paramStr(i) if a.startsWith("--fixtures="): names = a[11..^1].split(',') elif a.startsWith("--maxsamples="): maxSamples = parseInt(a[13..^1]) let spec = tmPatternSpec() echo &"# tm_pattern GF head over DrussGT fixtures (spec nBits={spec.nBits}, " & &"TM_CLASSES={TM_CLASSES}, TM_NCLAUSES={TM_NCLAUSES})" echo "# fixture,samples,captured,classTotal,labelHist" var pooledLabels: seq[int] var pooledConfusion: array[TM_CLASSES, array[TM_CLASSES, int]] var pooledClassCorrect, pooledClassTotal = 0 var pooledLabelHist: array[TM_CLASSES, int] var bestGun: TmPatternGun var bestSamples: seq[DiagSample] var bestName = "" var bestClassTotal = -1 for name in names: let path = fixturesDir / (name & ".jsonl") if not fileExists(path): echo &"# SKIP missing fixture {path}" continue let fx = loadFixture(path) let g = new(TmPatternGun) g[] = initTmPatternGun() g[].targetMode = tmGF g[].diagCapture = true randomize(1234) let drv = GunDriver( name: "TMPatGF", predictCb: proc(state: WorldState, bs: float): GunPrediction = g[].predict(state, bs), resultCb: proc(e: FeedbackEvent) = g[].onResult(e), readyCb: proc(): bool = true) discard replayFixture(fx, @[drv], fx.enemyId) var hs: string for c in 0.. bestClassTotal and thisSamples.len > 0: bestClassTotal = g[].classTotal bestGun = g[] bestSamples = thisSamples bestName = name # ── pooled label balance + majority baseline vs real accuracy ── for c in 0..=1={summ.firedAtLeastOnce} " & &"neverFired={summ.neverFired} posFired={summ.posFired} negFired={summ.negFired}" echo &"# clause length: mean={summ.meanLength:.2f} max={summ.maxLength} hist={summ.lengthHist}" for cb in clauseBalanceByClass(infos, TM_CLASSES): echo &"# class{cb.cls}: nonEmpty={cb.nonEmpty} empty={cb.empty} " & &"posFired={cb.posFired} negFired={cb.negFired} neverFired={cb.neverFired}" let contribs = featureContributions(m, bestSamples, spec) let ranked = rankedInputs(contribs) let dead = deadInputs(contribs) let never = neverUsedInputs(contribs) let consts = constantInputs(bestSamples, TM_NBITS) echo "\n## INPUT VALUE RANKING (top 15 by weighted vote-share)" for i in 0..2} {ranked[i].name:<26} appearances={ranked[i].appearances:<8} weighted={ranked[i].weighted:.1f}" echo "\n## DEAD-INPUT LIST (weighted < 5% of top, or never in a voting clause):" echo "# ", dead echo "## never-used (strict appearances==0): ", never echo "## constant inputs (zero variance in this fixture): ", consts echo "\n## TOP FIRING POSITIVE CLAUSES per class" for cls in 0.. necessary literals: {spec.describeClause(necessaryLiterals(m, bestSamples, cls), cls)}" # ── CLAUSE-SHAPE checker (Task 2/4): distribution + healthy-band verdict ── # The default healthy band is 3-8 literals for a ~50-bit problem; the shipped # tm_pattern is 40 bits, so the same band is used and the verdict is explicit. let sd = clauseShapeDiagnostics(m, bestSamples, spec) echo "\n## CLAUSE-SHAPE readout (healthy band 3-8 literals)" echo formatClauseShapeReport(sd) echo "# SHAPE VERDICT: ", sd.verdict # ── AUTOMATA-LEVEL metrics on the shipped GF head (Task 5) ── # settledness / diversity / histogram / per-input confidence / disagreement # are read DIRECTLY off the exported teams; churn is measured on a tm_core # retrain (same algorithm) over the captured samples, because the live gun # does not expose a per-sample state trace. var ad = AutomataDiag() ad.machine = m ad.settledness = settledness(m, 0.5) ad.diversity = clauseDiversity(m) ad.histogram = stateHistogram(m, 9) ad.inputConfidence = perInputConfidence(m, spec, bestSamples, 0.5) ad.disagreement = voteDisagreement(m, bestSamples) ad.shape = clauseShapeDiagnostics(m, bestSamples, spec) let churnN = min(bestSamples.len, 10000) var churnSamples = newSeq[DiagSample](churnN) for i in 0..