ab8d383121
Added the four automata-level metrics to the TM diagnostics kit (settledness, clause diversity, churn, vote disagreement) plus a state histogram, a per-input confidence table and a one-line health summary, and validated them on a learnable-vs-noise pair. STATE CONVENTIONS, read off OUR code rather than from memory: range [-nStates, nStates] as int16; nStates = 64 for tm_pattern, 32 for tsetlin initial value 0 = the Exclude boundary INCLUDE iff state > 0; EXCLUDE iff state <= 0 flip boundary sits between state 0 and 1; commitment = abs(st)/nStates in [0,1] === THE GATE, AND A RESULT THAT MATTERS === Case A (learnable planted rule) vs Case B (shuffled labels), 49 bits, N=64: metric A (learnable) B (shuffled) settledness mean 0.970 0.719 churn flip/sample 0.000055 FALLING 0.000788 FLAT clause-change/sample 0.00263 falling 0.0595 flat diversity (Jaccard) 0.176 0.014 disagreement 0.003 0.298 verdict settling mixed (NOT settling) **SETTLEDNESS ALONE DOES NOT WORK.** On noise the automata still COMMIT (0.719) - they just commit to the wrong thing. The decisive separators are **churn TREND (falling vs flat)** and **vote DISAGREEMENT (0.003 vs 0.298)**. Had we built only the settledness metric - the one that seems most obvious - we would have been misled. That is now recorded in the README. INERTIA SWEEP: A vs B separate at N=16/32/64/128. **Raising N raises A's commitment but does NOT reduce B's noise-fitting** - so more inertia does not rescue a noise-fitting TM. === REAL READING ON THE SHIPPED GUN, AND THE INFERENCE IT SUPPORTS === tm_pattern GF head over the DrussGT fixtures: settledness 0.484 (settling), diversity 0.267 (moderate), churn 0.094/100 FALLING, disagreement 0.145 (coherent). **VERDICT: SETTLING** - not fidgeting, not collapsed. Constant inputs flagged: 38/39 (the known never-written bits) plus 19/36/37. Context: pooled warm accuracy 35.72% vs 34.24% majority = +1.48pp. So: **the old gun was NOT failing because of inertia or instability - it settled properly and its settled rules still barely beat a lazy guess.** Its settledness (0.484) is LOWER than both synthetic cases (0.97/0.72), which is the signature of WEAK OR CONFLICTING SIGNAL rather than too much inertia. CONCLUSION: **N and s are not the observed bottleneck. The target/representation is.** That is exactly why the new design changes the target and the label pipeline rather than sweeping knobs - and it means we should NOT spend effort on an N/s sweep expecting it to fix anything. Also adds `diag_automata_validation.nim` (Case A/B/C + inertia sweep) and `test_tm_automata_diag.nim` (55 pure checks); `test_tm_diag` 48 and `diag_synthetic` 17 still pass, plus all other guards. acceptance_offline_vs_online was NOT run (it needs a live battle and there is no tm_diag dependency). Caveat: churn on the real gun is a PROXY (a tm_core retrain over captured samples in live order) because the live gun exposes no per-sample state trace; the other metrics are read directly off the exported teams.
146 lines
6.4 KiB
Nim
146 lines
6.4 KiB
Nim
## Task 5 — VALIDATE THE AUTOMATA-LEVEL METRICS against known ground truth.
|
|
##
|
|
## Case A: the planted-rule set from diag_synthetic.nim (learnable).
|
|
## Case B: the SAME inputs with the labels shuffled (pure noise).
|
|
## The metrics must SEPARATE learning from fidgeting:
|
|
## A: settledness high and rising, churn trend falling, disagreement low.
|
|
## B: settledness low, churn high / not falling, disagreement high.
|
|
## Case C: a CONSTANT input, to confirm the kit surfaces an information-free bit.
|
|
##
|
|
## Every number printed here is MEASURED.
|
|
## Run: nim c -r -d:release --path:common_libs common_libs/tests/diag_automata_validation.nim
|
|
|
|
import std/[random, strformat, strutils, algorithm]
|
|
import tm_diag/diagnostics
|
|
|
|
const
|
|
NBits = 49
|
|
BitA = 0
|
|
BitB = 45
|
|
NoiseBit = 17
|
|
ConstBit = 20
|
|
NClasses = 3
|
|
Epochs = 15
|
|
|
|
var failures = 0
|
|
proc check(name: string, ok: bool) =
|
|
if ok: echo "PASS: ", name
|
|
else: echo "FAIL: ", name; inc failures
|
|
|
|
proc genDataset(n, seed: int, constant = false): seq[DiagSample] =
|
|
var rng = initRand(seed)
|
|
for i in 0..<n:
|
|
var raw = newSeq[int](NBits)
|
|
for b in 0..<NBits:
|
|
raw[b] = (if rng.rand(1.0) < 0.5: 1 else: 0)
|
|
let a = if rng.rand(1.0) < 0.4: 1 else: 0
|
|
let bb = if rng.rand(1.0) < 0.5: 1 else: 0
|
|
raw[BitA] = a
|
|
raw[BitB] = bb
|
|
if constant: raw[ConstBit] = 1
|
|
let label =
|
|
if a == 1 and bb == 1: 2
|
|
elif a == 1: 1
|
|
else: 0
|
|
result.add makeSample(NBits, raw, label, i)
|
|
|
|
proc shuffleLabels(s: seq[DiagSample], seed: int): seq[DiagSample] =
|
|
result = s
|
|
var rng = initRand(seed)
|
|
var labels = newSeq[int](s.len)
|
|
for i in 0..<s.len: labels[i] = s[i].label
|
|
for i in countdown(s.len - 1, 1):
|
|
let j = rng.rand(i)
|
|
swap(labels[i], labels[j])
|
|
for i in 0..<s.len: result[i].label = labels[i]
|
|
|
|
proc report(tag: string, d: AutomataDiag) =
|
|
echo &"\n## {tag}"
|
|
echo "# ", d.summary
|
|
echo "# verdict=", automataVerdict(d)
|
|
echo &"# settledness: mean={d.settledness.overallMean:.4f} " &
|
|
&"settledFrac={d.settledness.overallSettledFraction:.4f} " &
|
|
&"(threshold={d.settledness.threshold:.2f})"
|
|
echo &"# diversity: pos={d.diversity.jaccardPos:.4f} neg={d.diversity.jaccardNeg:.4f} " &
|
|
&"overall={d.diversity.jaccardOverall:.4f} pairs={d.diversity.nPairs}"
|
|
echo &"# churn: flip/sample={d.churn.flipRate:.6f} ({d.churn.flipRatePer100:.4f}/100) " &
|
|
&"trend={d.churn.flipTrend} first={d.churn.flipFirst:.6f} last={d.churn.flipLast:.6f}"
|
|
echo &"# clause-change/sample={d.churn.clauseChangeRate:.6f} " &
|
|
&"({d.churn.clauseChangePer100:.4f}/100) trend={d.churn.clauseTrend} " &
|
|
&"first={d.churn.clauseFirst:.6f} last={d.churn.clauseLast:.6f}"
|
|
echo &"# disagreement: overall={d.disagreement.overall:.4f} perClass={d.disagreement.perClass}"
|
|
|
|
when isMainModule:
|
|
let spec = draftTMSpec()
|
|
let trainA = genDataset(3000, 1)
|
|
let trainB = shuffleLabels(trainA, 99)
|
|
let tmpl = newMachine(NBits, NClasses, nClauses = 40, nStates = 64,
|
|
sValue = 3.0, seed = 1)
|
|
|
|
echo &"# automata validation: nBits={NBits} classes={NClasses} clauses=40 states=64 " &
|
|
&"samples={trainA.len} epochs={Epochs}"
|
|
|
|
let dA = automataDiagnostics(tmpl, trainA, spec, epochs = Epochs, seed = 777,
|
|
settleThreshold = 0.5)
|
|
let dB = automataDiagnostics(tmpl, trainB, spec, epochs = Epochs, seed = 777,
|
|
settleThreshold = 0.5)
|
|
report("CASE A — learnable planted rule", dA)
|
|
report("CASE B — shuffled (noise) labels", dB)
|
|
|
|
# settledness RISING: early prefix vs full training.
|
|
let early = trainModel(tmpl, trainA[0..<500], epochs = 3, seed = 777)
|
|
let earlyS = settledness(early, 0.5)
|
|
let lateS = dA.settledness
|
|
echo &"\n## CASE A settledness trajectory: early(prefix 500 x3)={earlyS.overallMean:.4f} " &
|
|
&"-> late(full)={lateS.overallMean:.4f} ({settlednessTrend(earlyS, lateS)})"
|
|
|
|
echo "\n## CHECKS"
|
|
check "A settledness is high at convergence (>= 0.50)", dA.settledness.overallMean >= 0.50
|
|
check "A churn TRENDS DOWN (falling)",
|
|
dA.churn.flipTrend == "falling" and dA.churn.flipLast < dA.churn.flipFirst
|
|
check "A disagreement is low (< 0.15)", dA.disagreement.overall < 0.15
|
|
check "A settledness RISES from early to late",
|
|
lateS.overallMean > earlyS.overallMean
|
|
check "B churn does NOT fall (flat/rising/frozen)",
|
|
dB.churn.flipTrend in ["flat", "rising", "frozen"]
|
|
check "A is clearly more settled than B (A - B >= 0.10)",
|
|
dA.settledness.overallMean - dB.settledness.overallMean >= 0.10
|
|
check "B disagrees clearly more than A (B - A >= 0.10)",
|
|
dB.disagreement.overall - dA.disagreement.overall >= 0.10
|
|
check "the pair is separated (A verdict settling, B not settling)",
|
|
automataVerdict(dA) == "settling" and automataVerdict(dB) != "settling"
|
|
|
|
# ── Case C: constant input ──
|
|
let trainC = genDataset(2000, 3, constant = true)
|
|
let dC = automataDiagnostics(tmpl, trainC, spec, epochs = Epochs, seed = 777)
|
|
echo "\n## CASE C — constant input (bit 20 forced to 1)"
|
|
var cbit: InputConfidence
|
|
for ic in dC.inputConfidence:
|
|
if ic.bit == ConstBit: cbit = ic
|
|
echo &"# bit{ConstBit} constant={cbit.constant} meanCommit={cbit.meanCommitment:.4f} " &
|
|
&"settled={cbit.settledFraction:.4f}"
|
|
let consts = constantInputs(trainC, NBits)
|
|
echo &"# constantInputs(trainC) = {consts}"
|
|
check "the constant bit is flagged in the confidence table", cbit.constant
|
|
check "constantInputs() surfaces the constant bit", ConstBit in consts
|
|
check "a real planted bit is NOT flagged constant",
|
|
not dC.inputConfidence[BitA].constant
|
|
|
|
# ── Inertia sweep: how the automata metrics move with N ──
|
|
echo "\n## INERTIA SWEEP — settledness / churn / disagreement vs N (10 epochs)"
|
|
echo "# case,N,settledness,churnTrend,flipPer100,disagreement,verdict"
|
|
for n in [16, 32, 64, 128]:
|
|
let tmplN = newMachine(NBits, NClasses, 40, n, 3.0, 1)
|
|
let aN = automataDiagnostics(tmplN, trainA, spec, epochs = 10, seed = 777)
|
|
let bN = automataDiagnostics(tmplN, trainB, spec, epochs = 10, seed = 777)
|
|
echo &"# A,{n},{aN.settledness.overallMean:.3f},{aN.churn.flipTrend}," &
|
|
&"{aN.churn.flipRatePer100:.3f},{aN.disagreement.overall:.3f},{automataVerdict(aN)}"
|
|
echo &"# B,{n},{bN.settledness.overallMean:.3f},{bN.churn.flipTrend}," &
|
|
&"{bN.churn.flipRatePer100:.3f},{bN.disagreement.overall:.3f},{automataVerdict(bN)}"
|
|
|
|
echo ""
|
|
if failures > 0:
|
|
echo &"{failures} check(s) FAILED"
|
|
quit(1)
|
|
echo "All automata-validation checks passed."
|