## Pure unit + synthetic-validation tests for the CLAUSE-SHAPE checker (group 8). ## ## Covers: ## * percentile / lengthStats / clauseSummary distribution fields; ## * clauseShapeVerdict against the healthy band (boundaries + custom band); ## * per-block length contribution and coverage/concentration; ## * synthetic ground truth: a SHORT planted rule reads healthy and is ## recovered near its true length; RANDOM labels are reported as measured; ## the block containing the planted rule dominates and an irrelevant block ## contributes ~0. ## ## Run: nim c -r -d:release --path:common_libs common_libs/tests/test_tm_clause_shape.nim import std/[random, strformat, strutils] import tm_diag/diagnostics var checks = 0 var failures = 0 proc check(name: string, ok: bool) = inc checks if ok: echo "PASS: ", name else: echo "FAIL: ", name; inc failures # ── percentile / lengthStats ───────────────────────────────────────────────── proc testPercentile() = let s = @[1, 2, 3, 4] check "percentile p0 = min", percentile(s, 0.0) == 1.0 check "percentile p100 = max", percentile(s, 1.0) == 4.0 check "percentile median interpolates", abs(percentile(s, 0.5) - 2.5) < 1e-9 check "percentile p25", abs(percentile(s, 0.25) - 1.75) < 1e-9 check "percentile of empty = 0", percentile(@[], 0.5) == 0.0 check "percentile clamps p", percentile(s, 2.0) == 4.0 proc testLengthStats() = let st = lengthStats(@[1, 2, 2, 3, 3, 3, 4]) check "lengthStats n", st.n == 7 check "lengthStats min/max", st.minL == 1 and st.maxL == 4 check "lengthStats mean = 18/7", abs(st.mean - 18.0 / 7.0) < 1e-9 check "lengthStats median = 3", abs(st.median - 3.0) < 1e-9 check "lengthStats p90", abs(st.p90 - 3.4) < 1e-9 check "lengthStats histogram", st.hist == @[0, 1, 2, 3, 1] let empty = lengthStats(@[]) check "lengthStats empty is well-defined", empty.n == 0 and empty.mean == 0.0 and empty.maxL == 0 # ── clauseSummary distribution / polarity / per-class ──────────────────────── proc tinySpec(): FeatureSpec = var s = FeatureSpec() s.addBlock("b0", 4, @["b0.0", "b0.1", "b0.2", "b0.3"]) s.addBlock("b1", 4, @["b1.0", "b1.1", "b1.2", "b1.3"]) s proc shapeModel(): TmMachine = ## nBits=8, 2 classes, 4 clauses/class (half=2 pos, 2 neg), 16 literals. ## class0 pos clause0 = {+b0.0, +b0.1, +b1.0} (len 3, block0 x2 + block1 x1) ## class0 pos clause1 = {+b0.2} (len 1, block0) ## class0 neg clause2 = {+b1.1, +b1.2} (len 2, block1) ## everything else empty. result = newMachine(8, 2, 4, 64, 3.0, 1) let nl = result.nLiterals result.teams[0][0 * nl + 0] = 1 result.teams[0][0 * nl + 1] = 1 result.teams[0][0 * nl + 4] = 1 result.teams[0][1 * nl + 2] = 1 result.teams[0][2 * nl + 5] = 1 # negative clause (index 2 >= half=2) result.teams[0][2 * nl + 6] = 1 proc testClauseSummaryShape() = let m = shapeModel() let spec = tinySpec() let samples = @[makeSample(8, @[1, 1, 1, 0, 1, 1, 1, 0], 0)] let summ = clauseSummary(clauseInfo(m, samples, spec)) check "summary nonEmpty = 3", summ.nonEmpty == 3 check "summary empty = 5", summ.emptyClauses == 5 check "summary mean length = 2.0", abs(summ.meanLength - 2.0) < 1e-9 check "summary median = 2.0", abs(summ.medianLength - 2.0) < 1e-9 check "summary min = 1", summ.minLength == 1 check "summary max = 3", summ.maxLength == 3 check "summary lengthHist", summ.lengthHist == @[0, 1, 1, 1] check "summary posMean = (3+1)/2", abs(summ.posMeanLength - 2.0) < 1e-9 check "summary negMean = 2", abs(summ.negMeanLength - 2.0) < 1e-9 check "summary per-class mean = 2.0", abs(summ.perClassMeanLength[0] - 2.0) < 1e-9 check "summary emptyFraction = 5/8", abs(summ.emptyFraction - 5.0 / 8.0) < 1e-9 # ── healthy-band verdict ───────────────────────────────────────────────────── proc verdictOf(mean: float, n = 4): string = var s = ClauseSummary() s.nonEmpty = n s.totalClauses = 8 s.meanLength = mean clauseShapeVerdict(s) proc testShapeVerdict() = check "mean 1.5 -> collapsed", verdictOf(1.5) == "collapsed" check "mean 2.5 -> short", verdictOf(2.5) == "short" check "mean 3.0 -> healthy (band low edge)", verdictOf(3.0) == "healthy" check "mean 8.0 -> healthy (band high edge)", verdictOf(8.0) == "healthy" check "mean 12.0 -> too long", verdictOf(12.0) == "too long" check "mean 19.17 -> too long (the shipped gun)", verdictOf(19.17) == "too long" check "no non-empty clauses -> n/a", verdictOf(0.0, 0) == "n/a" # custom band is honoured var s = ClauseSummary() s.nonEmpty = 4 s.totalClauses = 8 s.meanLength = 2.5 check "custom band 2-6 makes mean 2.5 healthy", clauseShapeVerdict(s, 2.0, 6.0, 1.0) == "healthy" check "custom collapsedMax makes mean 2.5 collapsed", clauseShapeVerdict(s, 2.0, 6.0, 3.0) == "collapsed" proc testLengthHistText() = let t = lengthHistText(@[0, 3, 1]) check "lengthHistText skips empty bins", "len 0" notin t check "lengthHistText renders len 1", "len 1: 3" in t check "lengthHistText renders len 2", "len 2: 1" in t # ── per-block contribution ─────────────────────────────────────────────────── proc testBlockContribution() = let m = shapeModel() let spec = tinySpec() let blocks = blockLengthContributions(m, spec) check "one entry per block", blocks.len == 2 # block0 (bits 0-3): 2 in clause0 + 1 in clause1 = 3 over 3 clauses check "block0 total = 3", blocks[0].totalLits == 3 check "block0 mean/clause = 1.0", abs(blocks[0].meanPerClause - 1.0) < 1e-9 check "block0 clausesUsing = 2", blocks[0].clausesUsing == 2 check "block0 max per clause = 2", blocks[0].perClauseMax == 2 # block1 (bits 4-7): 1 in clause0 + 2 in neg clause2 = 3 over 3 clauses check "block1 total = 3", blocks[1].totalLits == 3 check "block1 mean/clause = 1.0", abs(blocks[1].meanPerClause - 1.0) < 1e-9 check "shares sum to 1", abs(blocks[0].share + blocks[1].share - 1.0) < 1e-9 # skipEmpty = false averages over all 8 clauses instead of 3 let blocksAll = blockLengthContributions(m, spec, skipEmpty = false) check "skipEmpty=false dilutes the mean", abs(blocksAll[0].meanPerClause - 3.0 / 8.0) < 1e-9 # ── coverage / concentration ───────────────────────────────────────────────── proc testCoverage() = ## 2 classes x 2 clauses = 4 clauses. class0/clause0 = {+b0.0} fires on the ## all-ones sample; nothing else fires. So 1 of 4 clauses fires. var m = newMachine(8, 2, 2, 64, 3.0, 1) let nl = m.nLiterals m.teams[0][0 * nl + 0] = 1 let samples = @[ makeSample(8, @[1, 0, 0, 0, 0, 0, 0, 0], 0), makeSample(8, @[1, 1, 1, 1, 1, 1, 1, 1], 0), ] let cov = clauseCoverage(m, samples) check "coverage totalClauses = 4", cov.totalClauses == 4 check "coverage mean firing = 1", abs(cov.meanFiringClauses - 1.0) < 1e-9 check "coverage fraction = 1/4", abs(cov.meanFiringFraction - 0.25) < 1e-9 check "one firing clause", cov.firingClauseCount == 1 check "effective clauses = 1 (single clause does all voting)", abs(cov.effectiveClauses - 1.0) < 1e-9 check "top3 share = 1 (all fires from one clause)", abs(cov.top3Share - 1.0) < 1e-9 # ── report helpers ─────────────────────────────────────────────────────────── proc testReportHelpers() = let m = shapeModel() let spec = tinySpec() let samples = @[makeSample(8, @[1, 1, 1, 0, 1, 1, 1, 0], 0)] let d = clauseShapeDiagnostics(m, samples, spec) check "shapeLine mentions shape/coverage", "shape mean=" in shapeLine(d) and "coverage=" in shapeLine(d) let rep = formatClauseShapeReport(d) check "report has length histogram", "length histogram" in rep check "report has per-block contribution", "per-block literal contribution" in rep check "report has per-class mean", "per-class mean/median" in rep # ── Task 3: synthetic validation with KNOWN geometry ───────────────────────── const NBits = 49 BitA = 0 ## dist-to-nearest-wall (WALLS block) BitB = 45 ## bullet-lateral-offset (BULLETS block) NoiseBit = 17 NClasses = 3 proc genRule(n, seed: int): seq[DiagSample] = ## class2 = A AND B ; class1 = A AND NOT B ; class0 = NOT A. var rng = initRand(seed) for i in 0..= 3.0 and d.summary.meanLength <= 8.0 # the recovered necessary literals are the 2-literal rule var c2ok = false let c2 = necessaryLiterals(m, ev, 2) var pos2: set[uint8] for l in c2: if l < NBits: pos2.incl uint8(l) c2ok = pos2 == {uint8(BitA), uint8(BitB)} check "SHORT rule recovered exactly (A AND B)", c2ok check "SHORT rule is NOT called too long", d.verdict != "too long" # ── per-block contribution: planted blocks dominate, irrelevant block dead ── # Use the sparser s=3 model here: with less padding the irrelevant block is # unambiguously negligible (below the uniform 1/12 = 8.3% share). let btmpl = newMachine(NBits, NClasses, nClauses = 40, nStates = 64, sValue = 3.0, seed = 1) let bm = trainModel(btmpl, train, epochs = 25, seed = 777) let bd = clauseShapeDiagnostics(bm, ev, spec) let walls = blockByName(bd.blocks, "dist-to-nearest-wall") let bullets = blockByName(bd.blocks, "bullet-lateral-offset") let usBlock = blockByName(bd.blocks, "dist-from-us") echo &"# per-block (s=3.0): walls(mean={walls.meanPerClause:.2f} share={walls.share*100:.1f}%) " & &"bullets(mean={bullets.meanPerClause:.2f} share={bullets.share*100:.1f}%) " & &"dist-from-us(mean={usBlock.meanPerClause:.2f} share={usBlock.share*100:.1f}%) " & &"[uniform=8.3%]" check "the WALLS block (holds the planted A) dominates", walls.share > 0.25 check "the BULLETS block (holds the planted B) is a top contributor", bullets.share > 0.15 check "the irrelevant US block is negligible (below uniform share)", usBlock.share < 0.083 check "the planted blocks outweigh the irrelevant block by >5x", usBlock.totalLits > 0 and walls.totalLits + bullets.totalLits > 5 * usBlock.totalLits # the noise MOTION bit is not singled out as a top block check "the pure-noise turn-direction block is far smaller than WALLS", blockByName(bd.blocks, "turn-direction").totalLits < walls.totalLits # ── RANDOM labels: report whatever the checker actually says ── let rtrain = genRandom(3000, 5) let rev = genRandom(1500, 6) let rtmpl = newMachine(NBits, NClasses, nClauses = 40, nStates = 64, sValue = 3.0, seed = 1) let rm = trainModel(rtmpl, rtrain, epochs = 25, seed = 777) let rd = clauseShapeDiagnostics(rm, rev, spec) echo &"# RANDOM labels s=3.0: mean={rd.summary.meanLength:.2f} " & &"median={rd.summary.medianLength:.2f} p90={rd.summary.p90Length:.2f} " & &"max={rd.summary.maxLength} empty={rd.summary.emptyClauses}/" & &"{rd.summary.totalClauses} verdict={rd.verdict} " & &"acc={evalAcc(rm, rev)*100:.1f}% eff={rd.coverage.effectiveClauses:.1f} " & &"top3={rd.coverage.top3Share*100:.1f}%" # On THIS encoding noise does not pad clauses long; it fails to commit at all # (short clauses, many firing, low concentration). Report and assert only that # the verdict is a legal word and that it is not "healthy". check "RANDOM labels are not reported healthy", rd.verdict in ["collapsed", "short", "too long"] check "RANDOM-label accuracy is near the majority baseline", abs(evalAcc(rm, rev) - 1.0 / 3.0) < 0.10 echo "" if failures > 0: echo &"{failures} / {checks} check(s) FAILED" quit(1) echo "All ", checks, " clause-shape checks passed."