From f58d65d2e8206cebd5b7d0a9465950dd9e6d2c28 Mon Sep 17 00:00:00 2001 From: Davide Cappellini Date: Fri, 25 Sep 2026 22:02:28 +0200 Subject: [PATCH] TMComposites gate: per-gun confidence faithful for 3 guns; no pair composes Adds a per-sample intrinsic-confidence field (GunPrediction.confidence, threaded through FeedbackEvent/VirtualBullet, populated by Pattern, DecayGF, KNN, GuessFactor, Tsetlin, TMHorizon) and an offline recorder + analyzer that reproduce the paper's Figure 2 per gun and its Eq-8 composite. Measured on 3 held-out tr-bridge DrussGT battles (33k ticks, ~133k samples/gun): - FAITHFUL: DecayGF (rho +0.133), KNN (+0.090), Pattern (+0.064, weak). - GuessFactor is ANTI-faithful (rho -0.067); Tsetlin c_max is useless (0.001). - No pair of guns specialises complementarily: the same gun dominates both high-confidence slices in every pair. - Eq-8 alpha-normalised confidence-weighted composite: 18.41% vs Pattern 20.45% (McNemar p=3.1e-126). Faithful-only variant 18.68%, still loses. Shuffle control passes weakly (composite > shuffle, p=4e-14) so ~0.7pp of competence is real but ~2pp short. Offline veto: design is dead. See docs/tmcomposites_gate.md. --- common_libs/gun_harness/gun_interface.nim | 9 + common_libs/gun_harness/virtual_bullets.nim | 8 +- common_libs/guns/decay_gf.nim | 1 + common_libs/guns/guess_factor.nim | 3 + common_libs/guns/knn_gun.nim | 4 + common_libs/guns/pattern_matcher.nim | 16 +- common_libs/guns/tm_horizon.nim | 13 +- common_libs/guns/tsetlin.nim | 9 +- common_libs/tests/analyze_tmcomposites.py | 438 +++++ .../tests/fixtures/tmcomposites_gate.json | 1573 +++++++++++++++++ common_libs/tests/measure_tmcomposites.nim | 285 +++ docs/tmcomposites_gate.md | 318 ++++ 12 files changed, 2666 insertions(+), 11 deletions(-) create mode 100644 common_libs/tests/analyze_tmcomposites.py create mode 100644 common_libs/tests/fixtures/tmcomposites_gate.json create mode 100644 common_libs/tests/measure_tmcomposites.nim create mode 100644 docs/tmcomposites_gate.md diff --git a/common_libs/gun_harness/gun_interface.nim b/common_libs/gun_harness/gun_interface.nim index c250784..16a1232 100644 --- a/common_libs/gun_harness/gun_interface.nim +++ b/common_libs/gun_harness/gun_interface.nim @@ -35,6 +35,14 @@ type GunPrediction* = object ## Absolute (x, y) where the gun predicts the enemy will be. x*, y*: float + confidence*: float + ## The gun's INTRINSIC, per-sample confidence in THIS prediction, on any + ## positive scale the gun likes (the TMComposites gate normalises it by + ## the sample range, Eq 7). 0.0 means "this gun has no confidence signal" + ## (every deterministic geometric gun): such a gun casts no composite + ## vote. Deliberately NOT a rolling hit-rate: that is the selector + ## mechanism already measured negative (docs/gun_rack_analysis.md). + ## See docs/tmcomposites_gate.md and common_libs/tests/measure_tmcomposites.nim. FeedbackEvent* = object ## Outcome of a resolved virtual bullet. @@ -45,6 +53,7 @@ type powerBin*: int ## index into PowerBins the bullet belongs to missDistance*: float ## px; < BotRadius = hit hit*: bool + confidence*: float ## copied from the spawn-time GunPrediction (above) proc bulletSpeed*(power: float): float {.inline.} = 20.0 - 3.0 * power diff --git a/common_libs/gun_harness/virtual_bullets.nim b/common_libs/gun_harness/virtual_bullets.nim index cf8fe46..efdd691 100644 --- a/common_libs/gun_harness/virtual_bullets.nim +++ b/common_libs/gun_harness/virtual_bullets.nim @@ -348,6 +348,7 @@ type bulletSpeed*: float travelDist*: float ## accumulated px so far fireDist*: float ## distance to target at fire time + confidence*: float ## spawn-time GunPrediction.confidence (TMComposites gate) active*: bool pointScored*: bool ## the parallel point-model score for this flight has ## been recorded (tie-break mode only) @@ -461,6 +462,7 @@ proc spawnBullets*(t: var VirtualTracker, gunId: GunId, bulletSpeed: speed, travelDist: 0.0, fireDist: fireDist, + confidence: pred.confidence, active: true, pointScored: false, hitSeen: false, @@ -643,7 +645,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState, t.fitness[b.targetId][b.gunId].bins[b.powerBin].record(hit) let fe = FeedbackEvent( - prediction: GunPrediction(x: b.aimX, y: b.aimY), + prediction: GunPrediction(x: b.aimX, y: b.aimY, confidence: b.confidence), actualX: ex, actualY: ey, bulletPower: PowerBins[b.powerBin], @@ -651,6 +653,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState, powerBin: b.powerBin, missDistance: missDist, hit: hit, + confidence: b.confidence, ) onResolved(b.gunId, b.powerBin, fe) b.active = false @@ -720,7 +723,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState, if b.targetId in t.fitness: t.fitness[b.targetId][b.gunId].bins[b.powerBin].record(b.hitSeen) let fe = FeedbackEvent( - prediction: GunPrediction(x: b.aimX, y: b.aimY), + prediction: GunPrediction(x: b.aimX, y: b.aimY, confidence: b.confidence), actualX: rx, actualY: ry, bulletPower: PowerBins[b.powerBin], @@ -728,6 +731,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState, powerBin: b.powerBin, missDistance: missDist, hit: b.hitSeen, + confidence: b.confidence, ) onResolved(b.gunId, b.powerBin, fe) b.active = false diff --git a/common_libs/guns/decay_gf.nim b/common_libs/guns/decay_gf.nim index 3c10ef5..5fdbfe7 100644 --- a/common_libs/guns/decay_gf.nim +++ b/common_libs/guns/decay_gf.nim @@ -113,6 +113,7 @@ proc predict*(g: var DecayGFGun, state: WorldState, bulletSpeed: float): GunPred GunPrediction( x: clamp(px, BotRadius, state.arenaWidth - BotRadius), y: clamp(py, BotRadius, state.arenaHeight - BotRadius), + confidence: g.bins[peak], # class-sum max over GF bins (see guess_factor.nim) ) proc onResult*(g: var DecayGFGun, e: FeedbackEvent) = diff --git a/common_libs/guns/guess_factor.nim b/common_libs/guns/guess_factor.nim index 584fc25..c147c31 100644 --- a/common_libs/guns/guess_factor.nim +++ b/common_libs/guns/guess_factor.nim @@ -141,6 +141,9 @@ proc predict*(g: var GFGun, state: WorldState, bulletSpeed: float): GunPredictio GunPrediction( x: clamp(px, BotRadius, state.arenaWidth - BotRadius), y: clamp(py, BotRadius, state.arenaHeight - BotRadius), + # TMComposites Eq 4 analogue: the GF histogram is the class distribution over + # GF bins, so the class-sum max c_max is the peak bin's accumulated weight. + confidence: g.bins[peak], ) proc onResult*(g: var GFGun, e: FeedbackEvent) = diff --git a/common_libs/guns/knn_gun.nim b/common_libs/guns/knn_gun.nim index 0aa4511..8b43ce1 100644 --- a/common_libs/guns/knn_gun.nim +++ b/common_libs/guns/knn_gun.nim @@ -299,6 +299,10 @@ proc predict*(g: var KNNGun, state: WorldState, bulletSpd: float): GunPrediction GunPrediction( x: clamp(px, BotRadius, state.arenaWidth - BotRadius), y: clamp(py, BotRadius, state.arenaHeight - BotRadius), + # TMComposites Eq 4 analogue: the KNN Gaussian density over GF candidates is + # the class distribution; bestScore is the class-sum max. Cold-start returns + # (no data / < 5 neighbours) leave the default 0.0 = no vote. + confidence: bestScore, ) proc onResult*(g: var KNNGun, e: FeedbackEvent) = diff --git a/common_libs/guns/pattern_matcher.nim b/common_libs/guns/pattern_matcher.nim index 4dc1918..c1dd56c 100644 --- a/common_libs/guns/pattern_matcher.nim +++ b/common_libs/guns/pattern_matcher.nim @@ -50,6 +50,10 @@ type cacheValid: bool cacheTick: int bestMatch: int ## -1 = no usable match (linear fallback) + lastMatchScore*: float ## best pattern-match cost (lower = better); + ## set by findBestMatch, exposed as the gun's + ## intrinsic per-sample confidence + ## (TMComposites gate, docs/tmcomposites_gate.md) playStart: int playAvail: int pathX: array[HistorySize + 1, float] @@ -87,9 +91,11 @@ proc linearPredict(state: WorldState, bulletSpeed: float): (float, float) = # --- pattern search + play-forward --- -proc findBestMatch(g: PatternMatcherGun): int = +proc findBestMatch(g: var PatternMatcherGun): int = ## Speed-independent history search. Returns the start index of the best - ## matching pattern, or -1 when there is not enough history. + ## matching pattern, or -1 when there is not enough history. Stores the best + ## match cost in `g.lastMatchScore` for the confidence readout. + g.lastMatchScore = Inf if g.count < PatternLen * 2: return -1 @@ -111,6 +117,7 @@ proc findBestMatch(g: PatternMatcherGun): int = if score < bestScore: bestScore = score bestMatch = i + g.lastMatchScore = bestScore bestMatch proc buildPath(g: var PatternMatcherGun, state: WorldState, bestMatch: int) = @@ -222,7 +229,10 @@ proc predict*(g: var PatternMatcherGun, state: WorldState, return g.applyRadial(state, px, py) let (px, py) = g.projectFromPath(state, bulletSpeed) - g.applyRadial(state, px, py) + result = g.applyRadial(state, px, py) + # Match quality as a confidence: a perfect historical match (cost 0) gives 1.0, + # a worse match decays toward 0. Deterministic and per-sample. + result.confidence = 1.0 / (1.0 + max(0.0, g.lastMatchScore)) proc onResult*(g: var PatternMatcherGun, e: FeedbackEvent) = discard # pattern matcher learns from movement observation, not feedback diff --git a/common_libs/guns/tm_horizon.nim b/common_libs/guns/tm_horizon.nim index 7c6e8be..7db36f1 100644 --- a/common_libs/guns/tm_horizon.nim +++ b/common_libs/guns/tm_horizon.nim @@ -1091,13 +1091,16 @@ proc predict*(g: var TmHorizonGun, state: WorldState, let sgn = if side == 1: 1.0 else: -1.0 shift = sgn * g.shiftDeg * magScale - let pred = - if shift == 0.0: base - else: tmhApplyShift(state.selfX, state.selfY, base.x, base.y, shift) + # TMComposites Eq 4 analogue: the side-machine's class-sum margin (normalised + # by the clause half-count) is the per-sample confidence in WHICH WAY to shift. + var predOut = base + if shift != 0.0: + predOut = tmhApplyShift(state.selfX, state.selfY, base.x, base.y, shift) + if warm: predOut.confidence = ev.sideConf - let aimDeg = radToDeg(arctan2(pred.y - state.selfY, pred.x - state.selfX)) + let aimDeg = radToDeg(arctan2(predOut.y - state.selfY, predOut.x - state.selfX)) g.tmhLog(state, bulletSpeed, h, side, mag, ev.sideConf, shift, aimDeg) - pred + predOut proc onResult*(g: var TmHorizonGun, e: FeedbackEvent) = ## The label comes from our own observation ring, not from virtual-bullet diff --git a/common_libs/guns/tsetlin.nim b/common_libs/guns/tsetlin.nim index 6bdf9f2..6186cfc 100644 --- a/common_libs/guns/tsetlin.nim +++ b/common_libs/guns/tsetlin.nim @@ -413,7 +413,14 @@ proc predict*(g: var TsetlinGun, state: WorldState, bulletSpeed: float): GunPred alive: true, ) - GunPrediction(x: predX, y: predY) + GunPrediction( + x: predX, + y: predY, + # TMComposites Eq 4: the two output teams' clamped clause sums are (vx, vy); + # their magnitude is how hard the machine is voting to move the correction. + # Warm-up fallback (window not full) leaves the default 0.0 = no vote. + confidence: hypot(vx, vy), + ) proc onResult*(g: var TsetlinGun, e: FeedbackEvent) = inc g.shotCount diff --git a/common_libs/tests/analyze_tmcomposites.py b/common_libs/tests/analyze_tmcomposites.py new file mode 100644 index 0000000..fce7dd8 --- /dev/null +++ b/common_libs/tests/analyze_tmcomposites.py @@ -0,0 +1,438 @@ +#!/usr/bin/env python3 +"""TMComposites GATE — analyse the per-sample confidence dump. + +Reads the JSONL produced by ``common_libs/tests/measure_tmcomposites.nim`` +(one row per resolved virtual bullet: fixture, split, gun, tick, bin, conf, +hit, relDeg, range, missPx) and answers the three questions of +``docs/tmcomposites_gate.md``: + + A. Is each gun's intrinsic confidence FAITHFUL? + Rank samples by the gun's own confidence; report the accuracy-vs-confidence + curve, Spearman(confidence, hit), and top-half vs bottom-half accuracy with + a two-proportion p-value. + + B. Are the faithful guns COMPLEMENTARY SPECIALISTS? + For each pair, on the samples where A's alpha-normalised confidence beats + B's, is A the more accurate one? Report the two slices and the win rates. + + C. Does the Eq-8 alpha-normalised confidence-weighted composite beat the best + single gun on held-out battles, and is the gain attributable to competence? + The recorder already ran ``Composite`` and its within-sample + confidence-shuffle control ``CompositeShuf`` through the SAME virtual-bullet + geometry, so the comparison is paired (McNemar). + +Pure stdlib (no numpy/scipy on this machine). MEASURED = every number printed; +INFERRED = the causal reading in the doc. +""" +from __future__ import annotations + +import argparse +import collections +import json +import math +import random +import sys + +DETERMINISTIC = ["HeadOn", "Linear", "Circular", "WallBounce", "Accel", + "StopShot", "Displace", "AvgLead"] + + +# ───────────────────────────── stats (stdlib) ────────────────────────────── +def mean(xs): + return sum(xs) / len(xs) if xs else float("nan") + + +def spearman(xs, ys): + """Spearman rho with average ranks for ties, and a normal-approx p.""" + n = len(xs) + if n < 3: + return float("nan"), float("nan") + + def ranks(v): + order = sorted(range(n), key=lambda i: v[i]) + r = [0.0] * n + i = 0 + while i < n: + j = i + while j + 1 < n and v[order[j + 1]] == v[order[i]]: + j += 1 + avg = (i + j) / 2.0 + 1.0 + for k in range(i, j + 1): + r[order[k]] = avg + i = j + 1 + return r + + rx, ry = ranks(xs), ranks(ys) + mx, my = mean(rx), mean(ry) + num = sum((a - mx) * (b - my) for a, b in zip(rx, ry)) + den = math.sqrt(sum((a - mx) ** 2 for a in rx) * sum((b - my) ** 2 for b in ry)) + if den == 0: + return 0.0, 1.0 + rho = num / den + rho = max(-1.0, min(1.0, rho)) + z = rho * math.sqrt(n - 1) + p = math.erfc(abs(z) / math.sqrt(2.0)) + return rho, p + + +def norm_two_prop(z): + return math.erfc(abs(z) / math.sqrt(2.0)) + + +def two_prop_p(h1, n1, h2, n2): + if n1 == 0 or n2 == 0: + return float("nan"), float("nan") + p1, p2 = h1 / n1, h2 / n2 + p = (h1 + h2) / (n1 + n2) + se = math.sqrt(p * (1 - p) * (1 / n1 + 1 / n2)) + if se == 0: + return float("nan"), float("nan") + z = (p1 - p2) / se + return z, norm_two_prop(z) + + +def binom_two_sided(k, n, p=0.5): + """Exact two-sided binomial p (used for McNemar's discordant pairs).""" + if n == 0: + return 1.0 + def pmf(i): + return math.comb(n, i) * p ** i * (1 - p) ** (n - i) + obs = pmf(k) + tot = 0.0 + for i in range(n + 1): + if pmf(i) <= obs + 1e-12: + tot += pmf(i) + return min(1.0, tot) + + +def mcnemar_p(a_hit_b_miss, a_miss_b_hit): + """McNemar: exact binomial for small discordant counts, normal approx for + large (the exact branch overflows math.comb for n in the hundred-thousands).""" + b, c = a_hit_b_miss, a_miss_b_hit + n = b + c + if n == 0: + return 1.0 + if n < 500: + return binom_two_sided(b, n) + z = (b - c) / math.sqrt(n) + return math.erfc(abs(z) / math.sqrt(2.0)) + + +# ───────────────────────────── data loading ──────────────────────────────── +class Dump: + def __init__(self, path): + self.rows = [] # list of dicts + self.by_gun = collections.defaultdict(list) + # key -> {gun: (conf, hit)} + self.by_sample = collections.defaultdict(dict) + with open(path) as f: + for line in f: + line = line.strip() + if not line: + continue + o = json.loads(line) + self.rows.append(o) + self.by_gun[o["gun"]].append(o) + self.by_sample[(o["fixture"], o["tick"], o["bin"])][o["gun"]] = ( + o["conf"], bool(o["hit"])) + + def guns(self): + return sorted(self.by_gun.keys()) + + def split(self, gun, split): + return [r for r in self.by_gun[gun] if r["split"] == split] + + +def faithful_curve(rows, bins=10): + rows = sorted(rows, key=lambda r: r["conf"]) + n = len(rows) + if n == 0: + return [] + out = [] + for b in range(bins): + lo = b * n // bins + hi = (b + 1) * n // bins + chunk = rows[lo:hi] + if not chunk: + continue + out.append(dict( + lo=chunk[0]["conf"], hi=chunk[-1]["conf"], n=len(chunk), + hit=mean([1.0 if r["hit"] else 0.0 for r in chunk]), + )) + return out + + +def faithfulness_report(dump): + """Question A: per-gun faithfulness on the pooled test samples.""" + out = {} + for gun in dump.guns(): + rows = dump.split(gun, "test") + if not rows: + continue + confs = [r["conf"] for r in rows] + hits = [1.0 if r["hit"] else 0.0 for r in rows] + nz = sum(1 for c in confs if c > 1e-12) + rho, p = spearman(confs, hits) + order = sorted(range(len(rows)), key=lambda i: confs[i]) + half = len(order) // 2 + lo_idx, hi_idx = order[:half], order[half:] + h_lo = sum(hits[i] for i in lo_idx) + h_hi = sum(hits[i] for i in hi_idx) + z, phalf = two_prop_p(h_hi, len(hi_idx), h_lo, len(lo_idx)) + out[gun] = dict( + n=len(rows), nonzero_conf=nz, + base=mean(hits), rho=rho, rho_p=p, + bottom_half_acc=(h_lo / len(lo_idx)) if lo_idx else float("nan"), + top_half_acc=(h_hi / len(hi_idx)) if hi_idx else float("nan"), + half_z=z, half_p=phalf, + curve=faithful_curve(rows), + ) + return out + + +def alphas(dump): + """alpha_t = max-min of the gun's confidence over the TRAIN samples (Eq 7).""" + out = {} + for gun in dump.guns(): + rows = dump.split(gun, "train") + if not rows: + continue + cs = [r["conf"] for r in rows] + out[gun] = max(1e-12, max(cs) - min(min(cs), 0.0)) + return out + + +def normalised(conf, gun, alpha): + return conf / alpha.get(gun, 1.0) + + +def complementarity(dump, alphas_, faithful): + """Question B: pairwise complementary slices over the test samples. + + For each pair we split the samples by which gun has the higher + alpha-normalised confidence and, ON EACH SLICE, measure BOTH guns' accuracy. + A pair is complementary when each gun is the more accurate one on its own + winning slice, and each slice is a substantial (>=10%) share. + """ + tfs = test_fixtures(dump) + pairs = [] + names = [g for g in faithful + if faithful[g]["nonzero_conf"] > 0 + and not g.startswith("Composite")] + for i in range(len(names)): + for j in range(i + 1, len(names)): + A, B = names[i], names[j] + a_win = b_win = 0 + # hits ON the A-winning slice, and ON the B-winning slice + aA = bA = aB = bB = 0 + aA_only = bA_only = aB_only = bB_only = 0 + for key, guns in dump.by_sample.items(): + if key[0] not in tfs: + continue + if A not in guns or B not in guns: + continue + ca = normalised(guns[A][0], A, alphas_) + cb = normalised(guns[B][0], B, alphas_) + if ca <= 0 and cb <= 0: + continue + ha, hb = int(guns[A][1]), int(guns[B][1]) + if ca > cb: + a_win += 1; aA += ha; bA += hb + if ha and not hb: aA_only += 1 + elif hb and not ha: bA_only += 1 + elif cb > ca: + b_win += 1; aB += ha; bB += hb + if ha and not hb: aB_only += 1 + elif hb and not ha: bB_only += 1 + both = a_win + b_win + def frac(h, n): + return h / n if n else float("nan") + accA_on_A, accB_on_A = frac(aA, a_win), frac(bA, a_win) + accA_on_B, accB_on_B = frac(aB, b_win), frac(bB, b_win) + comp = (both > 0 and a_win >= 0.10 * both and b_win >= 0.10 * both + and accA_on_A > accB_on_A and accB_on_B > accA_on_B) + pairs.append(dict( + A=A, B=B, a_win=a_win, b_win=b_win, both=both, + A_acc_on_A_slice=accA_on_A, B_acc_on_A_slice=accB_on_A, + A_acc_on_B_slice=accA_on_B, B_acc_on_B_slice=accB_on_B, + p_on_A_slice=mcnemar_p(aA_only, bA_only), + p_on_B_slice=mcnemar_p(bB_only, aB_only), + complementary=comp)) + return pairs + + +def compl_ok(a_win, b_win, both, aa, ba): + """Retained for backwards compatibility; superseded by the per-slice test + in `complementarity`.""" + if both == 0 or a_win == 0 or b_win == 0: + return False + return a_win >= 0.10 * both and b_win >= 0.10 * both + + +# ───────────────────────── composite comparison ──────────────────────────── +def paired(dump, gun_a, gun_b, tfs): + """Return (a_hit_b_miss, a_miss_b_hit, a_hits, b_hits) over the test fixtures.""" + ab = ba = na = nb = 0 + for key, guns in dump.by_sample.items(): + if key[0] not in tfs: + continue + if gun_a in guns and gun_b in guns: + a = guns[gun_a][1] + b = guns[gun_b][1] + if a and not b: + ab += 1 + elif b and not a: + ba += 1 + if a: + na += 1 + if b: + nb += 1 + return ab, ba, na, nb + + +def test_fixtures(dump): + out = set() + for r in dump.rows: + if r["split"] == "test": + out.add(r["fixture"]) + return out + + +def composite_report(dump): + """Question C: composite vs best single vs shuffle control on test.""" + tfs = test_fixtures(dump) + acc = {} + n = {} + for gun in dump.guns(): + rows = [r for r in dump.by_gun[gun] if r["split"] == "test"] + if not rows: + continue + acc[gun] = mean([1.0 if r["hit"] else 0.0 for r in rows]) + n[gun] = len(rows) + members = [g for g in dump.guns() + if not g.startswith("Composite") and g not in DETERMINISTIC] + best_member = max(members, key=lambda g: acc[g]) if members else None + res = dict(acc=acc, n=n, best_member=best_member) + # Oracle ceiling: if a perfect per-sample selector could pick ANY member, + # how often would it hit? This bounds what a member-selection composite + # could ever reach (the vote can do worse but not better than this). + oracle = 0 + oracle_n = 0 + for key, guns in dump.by_sample.items(): + if key[0] not in tfs: + continue + hits = [guns[m][1] for m in members if m in guns] + if hits: + oracle += int(any(hits)) + oracle_n += 1 + res["member_oracle"] = (oracle / oracle_n) if oracle_n else float("nan") + res["member_oracle_n"] = oracle_n + comps = [g for g in dump.guns() if g.startswith("Composite") and not g.endswith("Shuf")] + res["composites"] = {} + for comp in comps: + entry = {} + if best_member: + ab, ba, na, nb = paired(dump, comp, best_member, tfs) + entry["vs_best"] = dict(best=best_member, comp_hit=na, best_hit=nb, + total=n[comp], mcnemar_ab=ab, mcnemar_ba=ba, + p=mcnemar_p(ab, ba)) + if "Pattern" in acc: + ab, ba, na, nb = paired(dump, comp, "Pattern", tfs) + entry["vs_pattern"] = dict(comp_hit=na, pattern_hit=nb, total=n[comp], + mcnemar_ab=ab, mcnemar_ba=ba, p=mcnemar_p(ab, ba)) + shuf = comp + "Shuf" + if shuf in acc: + ab, ba, na, nb = paired(dump, comp, shuf, tfs) + entry["vs_shuffle"] = dict(comp_hit=na, shuffle_hit=nb, total=n[comp], + mcnemar_ab=ab, mcnemar_ba=ba, p=mcnemar_p(ab, ba)) + res["composites"][comp] = entry + # per-fixture composite vs best single vs shuffle + per = {} + for fx in sorted(tfs): + row = {} + for gun in ([best_member] if best_member else []) + comps: + rows = [r for r in dump.by_gun[gun] if r["split"] == "test" and r["fixture"] == fx] + if rows: + row[gun] = dict(n=len(rows), acc=mean([1.0 if r["hit"] else 0.0 for r in rows])) + per[fx] = row + res["per_fixture"] = per + return res + + +# ───────────────────────────────── main ──────────────────────────────────── +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--input", default="/tmp/tmc_full.jsonl") + ap.add_argument("--json", default=None) + args = ap.parse_args() + + dump = Dump(args.input) + print(f"rows={len(dump.rows)} guns={len(dump.guns())} " + f"test fixtures={sorted(test_fixtures(dump))}") + + a = faithfulness_report(dump) + print("\n=== A. FAITHFULNESS (test samples; rank by own confidence) ===") + print(f"{'gun':<14}{'n':>7}{'nonzero':>8}{'base%':>7}{'rho':>8}{'rho_p':>9}" + f"{'bot%':>7}{'top%':>7}{'z':>7}{'p':>9} verdict") + verdicts = {} + for gun in sorted(a, key=lambda g: -a[g]["base"]): + r = a[gun] + if r["nonzero_conf"] == 0: + v = "NO SIGNAL" + elif r["rho_p"] < 0.01 and r["rho"] > 0.05: + v = "FAITHFUL" + elif r["rho_p"] < 0.01 and r["rho"] < -0.05: + v = "ANTI-FAITHFUL" + else: + v = "USELESS" + verdicts[gun] = v + print(f"{gun:<14}{r['n']:>7}{r['nonzero_conf']:>8}{100*r['base']:>7.2f}" + f"{r['rho']:>8.3f}{r['rho_p']:>9.2g}{100*r['bottom_half_acc']:>7.2f}" + f"{100*r['top_half_acc']:>7.2f}{r['half_z']:>7.2f}{r['half_p']:>9.2g} {v}") + + print("\ncurves (deciles, low->high confidence):") + for gun in sorted(a, key=lambda g: -a[g]["base"]): + if verdicts[gun] == "NO SIGNAL": + continue + cur = " ".join(f"{100*c['hit']:.0f}%" for c in a[gun]["curve"]) + print(f" {gun:<14} {cur}") + + al = alphas(dump) + pairs = complementarity(dump, al, a) + print("\n=== B. PAIRWISE COMPLEMENTARITY (test; alpha-normalised confidence) ===") + print(f"{'A':<14}{'B':<14}{'A_wins':>8}{'A|A':>7}{'B|A':>7}{'p_A':>9}| {'B_wins':>8}{'A|B':>7}{'B|B':>7}{'p_B':>9} comp") + for p in pairs: + print(f"{p['A']:<14}{p['B']:<14}{p['a_win']:>8}" + f"{100*p['A_acc_on_A_slice']:>7.1f}{100*p['B_acc_on_A_slice']:>7.1f}{p['p_on_A_slice']:>9.2g}| " + f"{p['b_win']:>8}{100*p['A_acc_on_B_slice']:>7.1f}" + f"{100*p['B_acc_on_B_slice']:>7.1f}{p['p_on_B_slice']:>9.2g} {p['complementary']}") + print(" (A|A = A's accuracy on the slice A wins; B|A = B's accuracy on that same slice; etc.)") + + c = composite_report(dump) + print("\n=== C. COMPOSITE vs BEST SINGLE vs SHUFFLE CONTROL (test) ===") + for gun in sorted(c["acc"], key=lambda g: -c["acc"][g]): + print(f" {gun:<14} {100*c['acc'][gun]:>6.2f}% n={c['n'][gun]}") + if "member_oracle" in c: + print(f" member_oracle (any member hits, per sample) : {100*c['member_oracle']:.2f}%") + for comp, entry in c.get("composites", {}).items(): + print(f" {comp}:") + for k, r in entry.items(): + print(f" {k}: {r}") + print("\nper-fixture:") + for fx, row in c["per_fixture"].items(): + parts = " ".join(f"{g}={100*v['acc']:.1f}%({v['n']})" for g, v in row.items()) + print(f" {fx:<32} {parts}") + + if args.json: + blob = dict( + faithfulness=a, alphas=al, verdicts=verdicts, + complementarity=pairs, composite=c, + test_fixtures=sorted(test_fixtures(dump)), + ) + with open(args.json, "w") as f: + json.dump(blob, f, indent=2) + print(f"\n[json] wrote {args.json}") + + +if __name__ == "__main__": + main() diff --git a/common_libs/tests/fixtures/tmcomposites_gate.json b/common_libs/tests/fixtures/tmcomposites_gate.json new file mode 100644 index 0000000..8ff81a3 --- /dev/null +++ b/common_libs/tests/fixtures/tmcomposites_gate.json @@ -0,0 +1,1573 @@ +{ + "faithfulness": { + "Accel": { + "n": 133101, + "nonzero_conf": 0, + "base": 0.21021630190607132, + "rho": 0.0, + "rho_p": 1.0, + "bottom_half_acc": 0.14471825694966192, + "top_half_acc": 0.27571336268425717, + "half_z": 58.64465757250358, + "half_p": 0.0, + "curve": [ + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.1549962434259955 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.11111945905334335 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.16574004507888807 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.1334335086401202 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.15830202854996245 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.10811419984973704 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.34913598797896317 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.3019534184823441 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.30022539444027047 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.3191345503718729 + } + ] + }, + "AvgLead": { + "n": 133100, + "nonzero_conf": 0, + "base": 0.2008114199849737, + "rho": 0.0, + "rho_p": 1.0, + "bottom_half_acc": 0.14428249436513899, + "top_half_acc": 0.2573403456048084, + "half_z": 51.48028242957263, + "half_p": 0.0, + "curve": [ + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.15432006010518406 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.12374154770848986 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.16356123215627347 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.12644628099173555 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.15334335086401202 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.10721262208865515 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.3495867768595041 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.2725770097670924 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.2535687453042825 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.3037565740045079 + } + ] + }, + "Circular": { + "n": 133107, + "nonzero_conf": 0, + "base": 0.1726956508673473, + "rho": 0.0, + "rho_p": 1.0, + "bottom_half_acc": 0.12675611918320737, + "top_half_acc": 0.21863449229197343, + "half_z": 44.341501434293924, + "half_p": 0.0, + "curve": [ + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.12975206611570247 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.09991736158064758 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.14679588310419953 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.1236664162283997 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.133648861843588 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.09721283149275035 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.28549962434259957 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.23431748178198483 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.22627901735406805 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.24986852978739388 + } + ] + }, + "Composite": { + "n": 133094, + "nonzero_conf": 133094, + "base": 0.18412550528198116, + "rho": 0.0088488504464142, + "rho_p": 0.0012455880311271538, + "bottom_half_acc": 0.18370475002629721, + "top_half_acc": 0.1845462605376651, + "half_z": 0.39604098688467776, + "half_p": 0.6920747918778563, + "curve": [ + { + "lo": 0.0003648682386073272, + "hi": 0.7293044058500786, + "n": 13309, + "hit": 0.10917424299346307 + }, + { + "lo": 0.7293044058500786, + "hi": 0.8632782544157275, + "n": 13309, + "hit": 0.15553384927492675 + }, + { + "lo": 0.8633076503729116, + "hi": 0.963730921978428, + "n": 13310, + "hit": 0.2081893313298272 + }, + { + "lo": 0.9637341288188656, + "hi": 0.9908785195880181, + "n": 13309, + "hit": 0.22450973025772034 + }, + { + "lo": 0.9908785195880181, + "hi": 0.9976484200640967, + "n": 13310, + "hit": 0.22111194590533434 + }, + { + "lo": 0.9976491859409187, + "hi": 0.9998337129326115, + "n": 13309, + "hit": 0.23660680742354798 + }, + { + "lo": 0.9998337129326115, + "hi": 1.0063463491902216, + "n": 13309, + "hit": 0.24231722894282065 + }, + { + "lo": 1.0063463491902216, + "hi": 1.1189947998872034, + "n": 13310, + "hit": 0.1293764087152517 + }, + { + "lo": 1.1189947998872034, + "hi": 1.2378819792750102, + "n": 13309, + "hit": 0.14005560147268767 + }, + { + "lo": 1.2379491215331564, + "hi": 3.037483163630121, + "n": 13310, + "hit": 0.1743801652892562 + } + ] + }, + "CompositeF": { + "n": 133126, + "nonzero_conf": 133126, + "base": 0.18682300978020822, + "rho": 0.11439273200936113, + "rho_p": 0.0, + "bottom_half_acc": 0.1520514399891832, + "top_half_acc": 0.22159457957123327, + "half_z": 32.54977691376144, + "half_p": 2.1090750867380953e-232, + "curve": [ + { + "lo": 0.0003252316422504398, + "hi": 0.6523030295379743, + "n": 13312, + "hit": 0.08916766826923077 + }, + { + "lo": 0.6523333693309337, + "hi": 0.7320461727923427, + "n": 13313, + "hit": 0.09930143468789905 + }, + { + "lo": 0.7320461727923427, + "hi": 0.8459995543137109, + "n": 13312, + "hit": 0.1455078125 + }, + { + "lo": 0.8460419408811006, + "hi": 0.9540086046234093, + "n": 13313, + "hit": 0.21092165552467512 + }, + { + "lo": 0.9540086046234093, + "hi": 0.9834965465240503, + "n": 13313, + "hit": 0.21535341395628332 + }, + { + "lo": 0.9834965465240503, + "hi": 0.9939996324022881, + "n": 13312, + "hit": 0.21627103365384615 + }, + { + "lo": 0.9939996324022881, + "hi": 0.997989885572357, + "n": 13313, + "hit": 0.2065650116427552 + }, + { + "lo": 0.997989885572357, + "hi": 0.9997245674149878, + "n": 13312, + "hit": 0.22783954326923078 + }, + { + "lo": 0.9997245674149878, + "hi": 0.9999999964938535, + "n": 13313, + "hit": 0.22534364906482385 + }, + { + "lo": 0.9999999964938535, + "hi": 2.3088298828108833, + "n": 13313, + "hit": 0.23195372943739204 + } + ] + }, + "CompositeFShuf": { + "n": 133119, + "nonzero_conf": 133119, + "base": 0.18064288343512197, + "rho": 0.06555127384597137, + "rho_p": 2.0577714602185925e-126, + "bottom_half_acc": 0.15790501660181192, + "top_half_acc": 0.20338040865384616, + "half_z": 21.56350907672538, + "half_p": 3.954561480655304e-103, + "curve": [ + { + "lo": 0.0003252316422504398, + "hi": 0.9927495862142239, + "n": 13311, + "hit": 0.16505146119750583 + }, + { + "lo": 0.9927526173108666, + "hi": 14.197404502916669, + "n": 13312, + "hit": 0.20462740384615385 + }, + { + "lo": 14.198092147703614, + "hi": 17.307603914642247, + "n": 13312, + "hit": 0.12439903846153846 + }, + { + "lo": 17.307671001275732, + "hi": 19.293129066496917, + "n": 13312, + "hit": 0.13529146634615385 + }, + { + "lo": 19.29415671136869, + "hi": 20.979916138678714, + "n": 13312, + "hit": 0.16015625 + }, + { + "lo": 20.980038476057583, + "hi": 22.99105090689128, + "n": 13312, + "hit": 0.19884314903846154 + }, + { + "lo": 22.99125154781924, + "hi": 394.79845403414635, + "n": 13312, + "hit": 0.193359375 + }, + { + "lo": 394.79845403414635, + "hi": 474.1544020257505, + "n": 13312, + "hit": 0.12755408653846154 + }, + { + "lo": 474.165264526892, + "hi": 576.4441539605992, + "n": 13312, + "hit": 0.18689903846153846 + }, + { + "lo": 576.4657205924276, + "hi": 843.1494057302995, + "n": 13312, + "hit": 0.3102463942307692 + } + ] + }, + "CompositeShuf": { + "n": 133103, + "nonzero_conf": 133103, + "base": 0.17614929791214323, + "rho": 0.03129357624957575, + "rho_p": 3.443706234142166e-30, + "bottom_half_acc": 0.1692536550915839, + "top_half_acc": 0.18304483711984615, + "half_z": 6.603903230539594, + "half_p": 4.0047098974433546e-11, + "curve": [ + { + "lo": 0.00036682746011704995, + "hi": 15.46962039728178, + "n": 13310, + "hit": 0.1720510894064613 + }, + { + "lo": 15.47270923059888, + "hi": 21.12887136780713, + "n": 13310, + "hit": 0.12945154019534186 + }, + { + "lo": 21.128940649173003, + "hi": 29.1739879593678, + "n": 13310, + "hit": 0.18655146506386175 + }, + { + "lo": 29.17559773111444, + "hi": 179.79606987569198, + "n": 13311, + "hit": 0.2035158891142664 + }, + { + "lo": 179.80971961983835, + "hi": 385.68049046671655, + "n": 13310, + "hit": 0.15469571750563485 + }, + { + "lo": 385.6860801329345, + "hi": 480.8583906459061, + "n": 13310, + "hit": 0.1290007513148009 + }, + { + "lo": 480.8640659212374, + "hi": 586.8560660045789, + "n": 13311, + "hit": 0.17947562166629102 + }, + { + "lo": 586.8815589462752, + "hi": 785.2768612735734, + "n": 13310, + "hit": 0.21570247933884298 + }, + { + "lo": 785.2768612735734, + "hi": 7737.118985752055, + "n": 13310, + "hit": 0.20510894064613072 + }, + { + "lo": 7737.845176228245, + "hi": 18212.90587819798, + "n": 13311, + "hit": 0.18593644354293443 + } + ] + }, + "DecayGF": { + "n": 133104, + "nonzero_conf": 133104, + "base": 0.1482750330568578, + "rho": 0.13264131571711138, + "rho_p": 0.0, + "bottom_half_acc": 0.10794566654645991, + "top_half_acc": 0.1886043995672557, + "half_z": 41.40313721318057, + "half_p": 0.0, + "curve": [ + { + "lo": 0.556047406291762, + "hi": 396.25822339191745, + "n": 13310, + "hit": 0.10240420736288505 + }, + { + "lo": 396.25822339191745, + "hi": 424.34215837190305, + "n": 13310, + "hit": 0.0889556724267468 + }, + { + "lo": 424.34891458650884, + "hi": 448.7083002629722, + "n": 13311, + "hit": 0.10908271354518818 + }, + { + "lo": 448.7158220532212, + "hi": 474.6532756908656, + "n": 13310, + "hit": 0.14507888805409466 + }, + { + "lo": 474.6532756908656, + "hi": 502.470482806322, + "n": 13311, + "hit": 0.09420779806175343 + }, + { + "lo": 502.470482806322, + "hi": 539.350529281079, + "n": 13310, + "hit": 0.11427498121712998 + }, + { + "lo": 539.364943883612, + "hi": 576.763702492511, + "n": 13310, + "hit": 0.1589030803906837 + }, + { + "lo": 576.7798120938062, + "hi": 610.4661212081966, + "n": 13311, + "hit": 0.20787318758921192 + }, + { + "lo": 610.472355268001, + "hi": 663.2419449545197, + "n": 13310, + "hit": 0.2308039068369647 + }, + { + "lo": 663.2419449545197, + "hi": 842.9566899329488, + "n": 13311, + "hit": 0.23116219667943805 + } + ] + }, + "Displace": { + "n": 133105, + "nonzero_conf": 0, + "base": 0.13946884038916646, + "rho": 0.0, + "rho_p": 1.0, + "bottom_half_acc": 0.11421144368313499, + "top_half_acc": 0.16472585758718616, + "half_z": 26.598712318364225, + "half_p": 7.024916765901703e-156, + "curve": [ + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.12126220886551466 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.10720456765081511 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.10676183320811419 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.1123882503192848 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.12344102178812923 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.10036811659529712 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.2396694214876033 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.19164600706182855 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.14868519909842223 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.14326496882277814 + } + ] + }, + "GuessFactor": { + "n": 133108, + "nonzero_conf": 133108, + "base": 0.1589461189410103, + "rho": -0.06701377950217692, + "rho_p": 5.132616723013723e-132, + "bottom_half_acc": 0.18137151786519218, + "top_half_acc": 0.13652072001682844, + "half_z": -22.377181417817297, + "half_p": 6.56663113481927e-111, + "curve": [ + { + "lo": 0.6, + "hi": 1308.429940892439, + "n": 13310, + "hit": 0.164763335837716 + }, + { + "lo": 1308.429940892439, + "hi": 2631.5920330486097, + "n": 13311, + "hit": 0.2090752009616107 + }, + { + "lo": 2631.5920330486097, + "hi": 4241.519507453279, + "n": 13311, + "hit": 0.20531890917286455 + }, + { + "lo": 4241.606869192732, + "hi": 5988.043113014076, + "n": 13311, + "hit": 0.14747201562617385 + }, + { + "lo": 5988.043113014076, + "hi": 7765.872324815058, + "n": 13311, + "hit": 0.18022688002404028 + }, + { + "lo": 7765.872324815058, + "hi": 9382.10941803352, + "n": 13310, + "hit": 0.13591284748309543 + }, + { + "lo": 9382.10941803352, + "hi": 11083.456536032649, + "n": 13311, + "hit": 0.1777477274434678 + }, + { + "lo": 11083.456536032649, + "hi": 12694.844299170474, + "n": 13311, + "hit": 0.1408609420779806 + }, + { + "lo": 12694.844299170474, + "hi": 15396.958642755233, + "n": 13311, + "hit": 0.13116970926301555 + }, + { + "lo": 15397.427232498827, + "hi": 18211.62611629322, + "n": 13311, + "hit": 0.09691232814965066 + } + ] + }, + "HeadOn": { + "n": 133162, + "nonzero_conf": 0, + "base": 0.06180441867800123, + "rho": 0.0, + "rho_p": 1.0, + "bottom_half_acc": 0.07092113365674892, + "top_half_acc": 0.05268770369925354, + "half_z": -13.815674020284371, + "half_p": 2.0502616163033433e-43, + "curve": [ + { + "lo": 0.0, + "hi": 0.0, + "n": 13316, + "hit": 0.05407029137879243 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13316, + "hit": 0.07126764794232503 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13316, + "hit": 0.07967858215680385 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13316, + "hit": 0.06991589065785521 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13317, + "hit": 0.07967259893369377 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13316, + "hit": 0.08823971162511264 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13316, + "hit": 0.03416942024632021 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13316, + "hit": 0.04881345749474317 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13316, + "hit": 0.0514418744367678 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13317, + "hit": 0.040774949312908315 + } + ] + }, + "KNN": { + "n": 133121, + "nonzero_conf": 132685, + "base": 0.12320370189526822, + "rho": 0.09017502000221088, + "rho_p": 2.1330122689777176e-237, + "bottom_half_acc": 0.096484375, + "top_half_acc": 0.14992262736437253, + "half_z": 29.6608986030642, + "half_p": 2.4539906628991957e-193, + "curve": [ + { + "lo": 0.0, + "hi": 15.638339141930848, + "n": 13312, + "hit": 0.10321514423076923 + }, + { + "lo": 15.638339141930848, + "hi": 17.036923326108976, + "n": 13312, + "hit": 0.09149639423076923 + }, + { + "lo": 17.036999410096538, + "hi": 18.030581282296783, + "n": 13312, + "hit": 0.08849158653846154 + }, + { + "lo": 18.030581282296783, + "hi": 18.827000971765557, + "n": 13312, + "hit": 0.0967548076923077 + }, + { + "lo": 18.827000971765557, + "hi": 19.518491870355092, + "n": 13312, + "hit": 0.1024639423076923 + }, + { + "lo": 19.518491870355092, + "hi": 20.23071031835459, + "n": 13312, + "hit": 0.11576021634615384 + }, + { + "lo": 20.23071031835459, + "hi": 20.952282251438355, + "n": 13312, + "hit": 0.12770432692307693 + }, + { + "lo": 20.952282251438355, + "hi": 21.757155049516914, + "n": 13312, + "hit": 0.15016526442307693 + }, + { + "lo": 21.757155049516914, + "hi": 22.812605936942635, + "n": 13312, + "hit": 0.16376201923076922 + }, + { + "lo": 22.812605936942635, + "hi": 26.877360693741576, + "n": 13313, + "hit": 0.19221813265229476 + } + ] + }, + "Linear": { + "n": 133104, + "nonzero_conf": 0, + "base": 0.17308270224786634, + "rho": 0.0, + "rho_p": 1.0, + "bottom_half_acc": 0.13042432984733743, + "top_half_acc": 0.21574107464839523, + "half_z": 41.137885289820915, + "half_p": 0.0, + "curve": [ + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.14199849737039819 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.1009015777610819 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.15626173841183982 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.12381667918858001 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.12914131169709264 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.09368895567242674 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.3176558978211871 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.23469311096085943 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.2202854996243426 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.21238073773570731 + } + ] + }, + "Pattern": { + "n": 133100, + "nonzero_conf": 132860, + "base": 0.2044778362133734, + "rho": 0.06414480610840442, + "rho_p": 4.0989076672362305e-121, + "bottom_half_acc": 0.1861457550713749, + "top_half_acc": 0.2228099173553719, + "half_z": 16.582558430549508, + "half_p": 9.31761587086433e-62, + "curve": [ + { + "lo": 0.0, + "hi": 0.09254953188445986, + "n": 13310, + "hit": 0.13696468820435762 + }, + { + "lo": 0.09267265094740694, + "hi": 0.25053352547051444, + "n": 13310, + "hit": 0.15619834710743802 + }, + { + "lo": 0.25053352547051444, + "hi": 0.7601042123131311, + "n": 13310, + "hit": 0.18775356874530427 + }, + { + "lo": 0.7601042123131311, + "hi": 0.9485863156445888, + "n": 13310, + "hit": 0.23140495867768596 + }, + { + "lo": 0.9485863156445888, + "hi": 0.9807168098299544, + "n": 13310, + "hit": 0.21840721262208865 + }, + { + "lo": 0.9807257421009, + "hi": 0.9926808037026422, + "n": 13310, + "hit": 0.22058602554470322 + }, + { + "lo": 0.9926808037026422, + "hi": 0.9969332089098565, + "n": 13310, + "hit": 0.2080390683696469 + }, + { + "lo": 0.9969332089098565, + "hi": 0.9991909887963832, + "n": 13310, + "hit": 0.23388429752066114 + }, + { + "lo": 0.9991909887963832, + "hi": 0.9999566687733649, + "n": 13310, + "hit": 0.21908339594290008 + }, + { + "lo": 0.9999566687733649, + "hi": 1.0, + "n": 13310, + "hit": 0.23245679939894817 + } + ] + }, + "StopShot": { + "n": 133078, + "nonzero_conf": 0, + "base": 0.1821112430304032, + "rho": 0.0, + "rho_p": 1.0, + "bottom_half_acc": 0.1281504080313801, + "top_half_acc": 0.23607207802942634, + "half_z": 51.00541618271553, + "half_p": 0.0, + "curve": [ + { + "lo": 0.0, + "hi": 0.0, + "n": 13307, + "hit": 0.11039302622679792 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13308, + "hit": 0.12759242560865644 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13308, + "hit": 0.11421701232341448 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13308, + "hit": 0.14322212203186052 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13308, + "hit": 0.14532611962729186 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13307, + "hit": 0.09634027203727362 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13308, + "hit": 0.328073339344755 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13308, + "hit": 0.2673579801623084 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13308, + "hit": 0.27329425909227534 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13308, + "hit": 0.21528403967538323 + } + ] + }, + "Tsetlin": { + "n": 132964, + "nonzero_conf": 105353, + "base": 0.18178604735116272, + "rho": 0.0011184099384213854, + "rho_p": 0.6834072770419826, + "bottom_half_acc": 0.17937186005234498, + "top_half_acc": 0.18420023464998045, + "half_z": 2.282570933914721, + "half_p": 0.022455654714715015, + "curve": [ + { + "lo": 0.0, + "hi": 0.0, + "n": 13296, + "hit": 0.13372442839951865 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13296, + "hit": 0.2486462093862816 + }, + { + "lo": 0.0, + "hi": 1.0, + "n": 13297, + "hit": 0.11814695043994886 + }, + { + "lo": 1.0, + "hi": 1.0, + "n": 13296, + "hit": 0.20570096269554752 + }, + { + "lo": 1.0, + "hi": 2.0, + "n": 13297, + "hit": 0.19064450627961194 + }, + { + "lo": 2.0, + "hi": 2.0, + "n": 13296, + "hit": 0.1953219013237064 + }, + { + "lo": 2.0, + "hi": 3.0, + "n": 13296, + "hit": 0.15719013237063778 + }, + { + "lo": 3.0, + "hi": 4.0, + "n": 13297, + "hit": 0.1778596675941942 + }, + { + "lo": 4.0, + "hi": 5.385164807134504, + "n": 13296, + "hit": 0.20314380264741275 + }, + { + "lo": 5.385164807134504, + "hi": 23.021728866442675, + "n": 13297, + "hit": 0.1874858990749793 + } + ] + }, + "WallBounce": { + "n": 133103, + "nonzero_conf": 0, + "base": 0.20148306198958701, + "rho": 0.0, + "rho_p": 1.0, + "bottom_half_acc": 0.14839746960977296, + "top_half_acc": 0.2545678567135473, + "half_z": 48.284305564860375, + "half_p": 0.0, + "curve": [ + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.1516153268219384 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.12637114951164538 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.16604057099924868 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.13154533844189017 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.16641622839969947 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.12163786626596544 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.36270753512132825 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.2639368895567243 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13310, + "hit": 0.25649887302779867 + }, + { + "lo": 0.0, + "hi": 0.0, + "n": 13311, + "hit": 0.2680489820449253 + } + ] + } + }, + "alphas": { + "Accel": 1e-12, + "AvgLead": 1e-12, + "Circular": 1e-12, + "Composite": 15871.91599852515, + "CompositeF": 1800.10939824932, + "CompositeFShuf": 1800.10939824932, + "CompositeShuf": 15880.47005887853, + "DecayGF": 1772.43078264789, + "Displace": 1e-12, + "GuessFactor": 15137.52580059114, + "HeadOn": 1e-12, + "KNN": 28.0463854286614, + "Linear": 1e-12, + "Pattern": 1.0, + "StopShot": 1e-12, + "Tsetlin": 21.37755832643195, + "WallBounce": 1e-12 + }, + "verdicts": { + "Accel": "NO SIGNAL", + "Pattern": "FAITHFUL", + "WallBounce": "NO SIGNAL", + "AvgLead": "NO SIGNAL", + "CompositeF": "FAITHFUL", + "Composite": "USELESS", + "StopShot": "NO SIGNAL", + "Tsetlin": "USELESS", + "CompositeFShuf": "FAITHFUL", + "CompositeShuf": "USELESS", + "Linear": "NO SIGNAL", + "Circular": "NO SIGNAL", + "GuessFactor": "ANTI-FAITHFUL", + "DecayGF": "FAITHFUL", + "Displace": "NO SIGNAL", + "KNN": "FAITHFUL", + "HeadOn": "NO SIGNAL" + }, + "complementarity": [ + { + "A": "DecayGF", + "B": "GuessFactor", + "a_win": 43493, + "b_win": 89608, + "both": 133101, + "A_acc_on_A_slice": 0.1909962522704803, + "B_acc_on_A_slice": 0.19998620467661463, + "A_acc_on_B_slice": 0.12753325595928935, + "B_acc_on_B_slice": 0.13903892509597357, + "p_on_A_slice": 4.2133394063733146e-14, + "p_on_B_slice": 8.388980732271656e-46, + "complementary": false + }, + { + "A": "DecayGF", + "B": "KNN", + "a_win": 436, + "b_win": 132656, + "both": 133092, + "A_acc_on_A_slice": 0.30045871559633025, + "B_acc_on_A_slice": 0.3073394495412844, + "A_acc_on_B_slice": 0.1477430346158485, + "B_acc_on_B_slice": 0.12261789892654686, + "p_on_A_slice": 0.25, + "p_on_B_slice": 3.709286550752466e-99, + "complementary": false + }, + { + "A": "DecayGF", + "B": "Pattern", + "a_win": 26788, + "b_win": 106279, + "both": 133067, + "A_acc_on_A_slice": 0.10971330446468568, + "B_acc_on_A_slice": 0.14835000746602955, + "A_acc_on_B_slice": 0.15777340772871404, + "B_acc_on_B_slice": 0.21868854618504127, + "p_on_A_slice": 1.2951175231420152e-51, + "p_on_B_slice": 0.0, + "complementary": false + }, + { + "A": "DecayGF", + "B": "Tsetlin", + "a_win": 121163, + "b_win": 11795, + "both": 132958, + "A_acc_on_A_slice": 0.14675272154040425, + "B_acc_on_A_slice": 0.18446225332816124, + "A_acc_on_B_slice": 0.16184824077999152, + "B_acc_on_B_slice": 0.1540483255616787, + "p_on_A_slice": 3.169590284134502e-211, + "p_on_B_slice": 0.04539822572882434, + "complementary": false + }, + { + "A": "GuessFactor", + "B": "KNN", + "a_win": 44521, + "b_win": 88570, + "both": 133091, + "A_acc_on_A_slice": 0.12344736191909436, + "B_acc_on_A_slice": 0.0848363693537881, + "A_acc_on_B_slice": 0.1768093033758609, + "B_acc_on_B_slice": 0.14252004064581686, + "p_on_A_slice": 1.620185923314916e-84, + "p_on_B_slice": 9.029631750754458e-115, + "complementary": false + }, + { + "A": "GuessFactor", + "B": "Pattern", + "a_win": 40605, + "b_win": 92459, + "both": 133064, + "A_acc_on_A_slice": 0.10592291589705702, + "B_acc_on_A_slice": 0.14271641423470016, + "A_acc_on_B_slice": 0.18229701813776916, + "B_acc_on_B_slice": 0.23168106944699812, + "p_on_A_slice": 7.233373529180136e-69, + "p_on_B_slice": 2.018348664294041e-207, + "complementary": false + }, + { + "A": "GuessFactor", + "B": "Tsetlin", + "a_win": 117799, + "b_win": 15156, + "both": 132955, + "A_acc_on_A_slice": 0.1547636227811781, + "B_acc_on_A_slice": 0.17894888751177854, + "A_acc_on_B_slice": 0.19180522565320665, + "B_acc_on_B_slice": 0.20354974927421482, + "p_on_A_slice": 5.921470342848188e-92, + "p_on_B_slice": 0.0012357591951468305, + "complementary": false + }, + { + "A": "KNN", + "B": "Pattern", + "a_win": 38485, + "b_win": 94344, + "both": 132829, + "A_acc_on_A_slice": 0.0968689099649214, + "B_acc_on_A_slice": 0.15834740808107053, + "A_acc_on_B_slice": 0.13373399474264394, + "B_acc_on_B_slice": 0.22368142118205717, + "p_on_A_slice": 6.383187817616419e-155, + "p_on_B_slice": 0.0, + "complementary": false + }, + { + "A": "KNN", + "B": "Tsetlin", + "a_win": 131820, + "b_win": 804, + "both": 132624, + "A_acc_on_A_slice": 0.12191624943104233, + "B_acc_on_A_slice": 0.18118646639356698, + "A_acc_on_B_slice": 0.21641791044776118, + "B_acc_on_B_slice": 0.2997512437810945, + "p_on_A_slice": 0.0, + "p_on_B_slice": 6.351396455890198e-06, + "complementary": false + }, + { + "A": "Pattern", + "B": "Tsetlin", + "a_win": 121591, + "b_win": 11221, + "both": 132812, + "A_acc_on_A_slice": 0.21102713194233125, + "B_acc_on_A_slice": 0.18796621460469937, + "A_acc_on_B_slice": 0.13608412797433384, + "B_acc_on_B_slice": 0.11647803226093931, + "p_on_A_slice": 1.902586881414242e-58, + "p_on_B_slice": 2.0891951235349046e-07, + "complementary": false + } + ], + "composite": { + "acc": { + "Accel": 0.21021630190607132, + "AvgLead": 0.2008114199849737, + "Circular": 0.1726956508673473, + "Composite": 0.18412550528198116, + "CompositeF": 0.18682300978020822, + "CompositeFShuf": 0.18064288343512197, + "CompositeShuf": 0.17614929791214323, + "DecayGF": 0.1482750330568578, + "Displace": 0.13946884038916646, + "GuessFactor": 0.1589461189410103, + "HeadOn": 0.06180441867800123, + "KNN": 0.12320370189526822, + "Linear": 0.17308270224786634, + "Pattern": 0.2044778362133734, + "StopShot": 0.1821112430304032, + "Tsetlin": 0.18178604735116272, + "WallBounce": 0.20148306198958701 + }, + "n": { + "Accel": 133101, + "AvgLead": 133100, + "Circular": 133107, + "Composite": 133094, + "CompositeF": 133126, + "CompositeFShuf": 133119, + "CompositeShuf": 133103, + "DecayGF": 133104, + "Displace": 133105, + "GuessFactor": 133108, + "HeadOn": 133162, + "KNN": 133121, + "Linear": 133104, + "Pattern": 133100, + "StopShot": 133078, + "Tsetlin": 132964, + "WallBounce": 133103 + }, + "best_member": "Pattern", + "member_oracle": 0.4171554307256738, + "member_oracle_n": 133159, + "composites": { + "Composite": { + "vs_best": { + "best": "Pattern", + "comp_hit": 24488, + "best_hit": 27215, + "total": 133094, + "mcnemar_ab": 5146, + "mcnemar_ba": 7873, + "p": 3.069104908009734e-126 + }, + "vs_pattern": { + "comp_hit": 24488, + "pattern_hit": 27215, + "total": 133094, + "mcnemar_ab": 5146, + "mcnemar_ba": 7873, + "p": 3.069104908009734e-126 + }, + "vs_shuffle": { + "comp_hit": 24496, + "shuffle_hit": 23443, + "total": 133094, + "mcnemar_ab": 10228, + "mcnemar_ba": 9175, + "p": 4.045752414569227e-14 + } + }, + "CompositeF": { + "vs_best": { + "best": "Pattern", + "comp_hit": 24854, + "best_hit": 27215, + "total": 133126, + "mcnemar_ab": 3425, + "mcnemar_ba": 5786, + "p": 1.2500472522844938e-133 + }, + "vs_pattern": { + "comp_hit": 24854, + "pattern_hit": 27215, + "total": 133126, + "mcnemar_ab": 3425, + "mcnemar_ba": 5786, + "p": 1.2500472522844938e-133 + }, + "vs_shuffle": { + "comp_hit": 24864, + "shuffle_hit": 24040, + "total": 133126, + "mcnemar_ab": 5914, + "mcnemar_ba": 5090, + "p": 3.994416394978975e-15 + } + } + }, + "per_fixture": { + "tr_drussgt_vs_corners": { + "Pattern": { + "n": 10062, + "acc": 0.22411051480818922 + }, + "Composite": { + "n": 10082, + "acc": 0.22287244594326522 + }, + "CompositeF": { + "n": 10082, + "acc": 0.2232691926205118 + } + }, + "tr_drussgt_vs_modularbot": { + "Pattern": { + "n": 79966, + "acc": 0.12960508215991798 + }, + "Composite": { + "n": 79943, + "acc": 0.12208698697822198 + }, + "CompositeF": { + "n": 79965, + "acc": 0.12204089289063966 + } + }, + "tr_drussgt_vs_spinbot": { + "Pattern": { + "n": 43072, + "acc": 0.3388976597325409 + }, + "Composite": { + "n": 43069, + "acc": 0.29020873482086884 + }, + "CompositeF": { + "n": 43079, + "acc": 0.29854453445994567 + } + } + } + }, + "test_fixtures": [ + "tr_drussgt_vs_corners", + "tr_drussgt_vs_modularbot", + "tr_drussgt_vs_spinbot" + ] +} \ No newline at end of file diff --git a/common_libs/tests/measure_tmcomposites.nim b/common_libs/tests/measure_tmcomposites.nim new file mode 100644 index 0000000..5e30d3e --- /dev/null +++ b/common_libs/tests/measure_tmcomposites.nim @@ -0,0 +1,285 @@ +## TMComposites GATE — offline per-sample confidence + outcome recorder. +## +## Implements the measurement half of docs/tmcomposites_gate.md. It replays +## recorded fixtures through the SHIPPED rack exactly as `offline_range` does +## (same `VirtualTracker`, same `bmPath` virtual-bullet ground truth), but also +## captures, for every resolved virtual bullet, the gun's INTRINSIC per-sample +## `GunPrediction.confidence` (see gun_interface.nim) and the bullet's aim +## bearing relative to the fire-time line of sight. +## +## It also runs two COMPOSITE arms through the SAME tracker, so their hits are +## scored by the identical geometry as every member: +## * Composite — TMComposites Eq 8: each confident gun casts its +## alpha-normalised confidence into the angular bin of its +## own aim; the argmax bin wins. +## * CompositeShuf — the MANDATORY control: the same confidences are randomly +## permuted among the members within each sample, so the +## weighting distribution is preserved but competence is +## destroyed. +## +## alpha (Eq 7) is calibrated on the TRAIN fixtures (per member: max-min of its +## confidence over train) and then frozen for the TEST fixtures, so the test +## composite never sees test labels while choosing its weights. +## +## Output: one JSONL row per resolved bullet, to --out. Aggregation lives in +## common_libs/tests/analyze_tmcomposites.py. +## +## Usage: +## nim c -r --nimcache:/tmp/nc_j104 common_libs/tests/measure_tmcomposites.nim \ +## --out /tmp/tmc.jsonl \ +## --train fx_a.jsonl fx_b.jsonl --test fx_c.jsonl + +import std/[os, strformat, json, math, strutils, random, algorithm, tables, times] +import gun_harness/offline_range +import gun_harness/gun_interface +import gun_harness/virtual_bullets +import range_guns + +const + BinDeg = 0.5 + HalfSpanDeg = 45.0 + NumBins = int(2.0 * HalfSpanDeg / BinDeg) + ## guns with a genuine intrinsic confidence signal (deterministic geometric + ## guns leave GunPrediction.confidence at 0.0 and cast no composite vote). + ConfidentGuns = ["Tsetlin", "GuessFactor", "Pattern", "DecayGF", "KNN"] + +type + Spawn = object + relDeg: float + range: float + conf: float + +proc wrapDeg(d: float): float = + result = d + while result > 180.0: result -= 360.0 + while result < -180.0: result += 360.0 + +proc binOf(relDeg: float): int = + result = int(floor((relDeg + HalfSpanDeg) / BinDeg)) + if result < 0: result = 0 + elif result >= NumBins: result = NumBins - 1 + +var alphaByName: ref Table[string, float] + +proc alphaFor(name: string): float = + ## Eq 7 alpha_t for a member, keyed by GUN NAME so each composite's member + ## order cannot mis-assign a scale. 1.0 until calibration writes the table. + if alphaByName.isNil: return 1.0 + alphaByName[].getOrDefault(name, 1.0) + +proc makeComposite(name: string, members: seq[GunDriver], + shuffle: bool): GunDriver = + ## One composite arm. `members` are the SAME driver closures the tracker uses + ## for the member guns, so the composite reads their live state (calling + ## predict twice in a tick is idempotent for every gun: GF/KNN guard their + ## wave queue on (tick,bin), Pattern/Tsetlin/TMHorizon cache per tick). + result.name = name + let memberList = members + result.predictCb = proc(state: WorldState, speed: float): GunPrediction = + let n = memberList.len + var rels = newSeq[float](n) + var dists = newSeq[float](n) + var confs = newSeq[float](n) + let los = arctan2(state.enemyY - state.selfY, state.enemyX - state.selfX) + for i in 0.. votes[best]: best = b + + # aim distance = confidence-weighted mean distance of the winning bin's voters + var dsum = 0.0 + var wsum = 0.0 + for i in 0.. 1e-12: dsum / wsum else: hypot(state.enemyX - state.selfX, + state.enemyY - state.selfY) + let ang = los + degToRad(-HalfSpanDeg + (best.float + 0.5) * BinDeg) + GunPrediction(x: state.selfX + cos(ang) * d, + y: state.selfY + sin(ang) * d, + confidence: votes[best]) + result.resultCb = proc(e: FeedbackEvent) = discard + result.readyCb = nil + +proc buildRack(): tuple[drivers: seq[GunDriver], memberLocal: seq[int]] = + ## Fresh members + the composite arms, mirroring the standard per-fixture + ## replay (guns start cold for every fixture, exactly as run_range does). + ## + ## Composite — all 5 confidence guns (Tsetlin, GF, Pattern, DecayGF, KNN) + ## CompositeShuf — its within-sample confidence shuffle control + ## CompositeF — only the guns the faithfulness test finds FAITHFUL + ## (Pattern, DecayGF, KNN); GF is anti-faithful and Tsetlin + ## useless, so this is the strongest reasonable variant + ## CompositeFShuf — its shuffle control + let allDrivers = buildAllGunDrivers(seed = 1) + var members: seq[GunDriver] + var memberLocal: seq[int] + var faithMembers: seq[GunDriver] + for i, d in allDrivers: + if d.name in ConfidentGuns: + members.add d + memberLocal.add i + if d.name in ["Pattern", "DecayGF", "KNN"]: + faithMembers.add d + doAssert members.len == ConfidentGuns.len + result.drivers = allDrivers + result.drivers.add makeComposite("Composite", members, shuffle = false) + result.drivers.add makeComposite("CompositeShuf", members, shuffle = true) + result.drivers.add makeComposite("CompositeF", faithMembers, shuffle = false) + result.drivers.add makeComposite("CompositeFShuf", faithMembers, shuffle = true) + result.memberLocal = memberLocal + +proc runFixture(fx: Fixture, fixtureName, split: string, + calMin, calMax: ref seq[float], outFile: File) = + let (drivers, memberLocal) = buildRack() + var byTick = initTable[int, WorldState]() + for s in fx.states: byTick[s.tick] = s + var spawns = initTable[tuple[gunId, tick, bin: int], Spawn]() + var tracker = initTracker(drivers.len, ActiveMetric) + + for si in 0.. 0: + for e in state.enemies: + enemyPositions[e.id] = (x: e.x, y: e.y, lastSeenTick: e.lastSeenTick, alive: true) + else: + enemyPositions[fx.enemyId] = (x: state.enemyX, y: state.enemyY, + lastSeenTick: state.tick, alive: true) + + tracker.tickBullets(state, enemyPositions, + proc(gunId: int, binIdx: int, e: FeedbackEvent) = + # The gun must LEARN from its own resolved bullet (exactly as + # offline_range.replayFixture does); the recorder is an extra hook. + drivers[gunId].resultCb(e) + let key = (gunId, e.fireTick, binIdx) + if not spawns.hasKey(key): return + let sp = spawns[key] + spawns.del(key) + if split == "train": + for mi, gid in memberLocal: + if gunId == gid: + calMin[][mi] = min(calMin[][mi], sp.conf) + calMax[][mi] = max(calMax[][mi], sp.conf) + outFile.writeLine($(%*{ + "fixture": fixtureName, + "split": split, + "gun": drivers[gunId].name, + "tick": e.fireTick, + "bin": binIdx, + "conf": sp.conf, + "hit": e.hit, + "relDeg": sp.relDeg, + "range": sp.range, + "missPx": e.missDistance, + })) + ) + +proc loadAll(paths: seq[string]): seq[tuple[name: string, fx: Fixture]] = + for p in paths: + let fx = loadFixture(p) + let name = extractFilename(p).replace(".jsonl", "") + result.add (name: name, fx: fx) + +proc main() = + var outPath = "/tmp/tmc.jsonl" + var trainPaths, testPaths: seq[string] + var args: seq[string] + for i in 1..paramCount(): args.add paramStr(i) + var mode = "" + for a in args: + if a == "--out": mode = "out"; continue + if a == "--train": mode = "train"; continue + if a == "--test": mode = "test"; continue + case mode + of "out": outPath = a + of "train": trainPaths.add a + of "test": testPaths.add a + else: discard + + var calMinRef = new(seq[float]) + var calMaxRef = new(seq[float]) + calMinRef[] = newSeq[float](ConfidentGuns.len) + calMaxRef[] = newSeq[float](ConfidentGuns.len) + for i in 0..6} {epochTime()-a:6.1f}s" + + # Eq 7 alpha_t = max-min over the training input set; frozen for test. + alphaByName = new(Table[string, float]) + for i in 0..6} {epochTime()-a:6.1f}s" + + echo fmt"# done in {epochTime()-t0:.1f}s -> {outPath}" + +when isMainModule: + main() diff --git a/docs/tmcomposites_gate.md b/docs/tmcomposites_gate.md new file mode 100644 index 0000000..8d3a2b0 --- /dev/null +++ b/docs/tmcomposites_gate.md @@ -0,0 +1,318 @@ +# TMComposites gate: is each gun's intrinsic confidence faithful, and are the guns complementary specialists? + +**Scope.** Test the mechanism of *TMComposites: Plug-and-Play Collaboration +Between Specialized Tsetlin Machines* (Granmo, arXiv:2309.04801v2, §3) on our +gun rack: (A) is each gun's per-sample confidence faithful to its own accuracy? +(B) are the guns complementary specialists? (C) does an Eq-8 alpha-normalised +confidence-weighted composite beat the best single gun offline, and is any gain +attributable to competence (shuffle control)? This is the one untested idea that +could plausibly beat `onlyPattern`, whose selector was measured **negative value** +(`docs/selector_negative_value.md`, `docs/gun_rack_analysis.md`, commit `e0666a5`) +because it decides with a rolling hit-rate instead of a per-sample signal. + +**Evidence tags.** `[MEASURED]` = read from the committed result +`common_libs/tests/fixtures/tmcomposites_gate.json`, reproduced by the named +command, or a source fact. `[INFERRED]` = reasoning from those measurements. + +--- + +## 0. Direct answers + +1. **Which guns are faithfully confident?** `[MEASURED]` + **Pattern (weak), DecayGF (strong), KNN (moderate)** are faithful: rank samples + by the gun's own confidence and accuracy rises. **GuessFactor is + ANTI-faithful** — it is more accurate when it is *less* confident. **Tsetlin's + class-sum max is useless** (Spearman 0.001, p=0.68). The eight deterministic + geometric guns (HeadOn, Linear, Circular, WallBounce, Accel, StopShot, + Displace, AvgLead) expose **no per-sample confidence at all**. + +2. **Do any pairs specialise complementarily?** `[MEASURED]` **No.** In every one + of the 10 pairs, on the slice where A's normalised confidence beats B's, A is + *not* the more accurate gun — or the same gun dominates both slices (e.g. + GuessFactor is more accurate than DecayGF and than KNN even on their own + high-confidence slices). The paper's premise — one member's weakness is + another's strength, decided by confidence — does not hold on our rack. Our + guns are different *estimators of the same target*, not different *specialists*. + +3. **Does the composite beat the best single gun?** `[MEASURED]` **No, and the + design is dead offline.** The Eq-8 composite scores **18.41%** vs Pattern's + **20.45%** (McNemar p=3.1e-126); the faithful-only variant (Pattern + DecayGF + + KNN) scores **18.68%**, still losing to Pattern by 1.77pp (p=1.25e-133). Both + lose on all three held-out battles. The **confidence-shuffle control passes in + the weak sense** — the composite beats its own shuffle (18.41 vs 17.61, + p=4.0e-14) — so the weighting carries a *real but tiny* competence signal; it + is simply nowhere near enough to beat Pattern. Per the `docs/offline_harness_trust.md` + rule (offline is veto-only), this negative **kills the design**; do not take a + composite to a live A/B. + +--- + +## 1. What was measured `[MEASURED]` + +The offline gun range (`common_libs/gun_harness/offline_range.nim` + +`virtual_bullets.nim`), driven by a new recorder +`common_libs/tests/measure_tmcomposites.nim`. For every resolved virtual bullet +it records `(gun, tick, powerBin, confidence, hit, aim-bearing-relative-to-LOS, +range)`. Ground truth is the shipped `bmPath` virtual-bullet metric (18 px hit +radius — `BotRadius`), reproduced exactly per bullet, not a rolling rate. + +| | fixtures | ticks | note | +|---|---|---:|---| +| **train** (alpha calibration only) | `drussgt_vs_crazy`, `drussgt_vs_spinbot`, `drussgt_vs_ramfire`, `drussgt_vs_corners` | 22 242 | classic-Robocode DrussGT | +| **test** (all reported numbers) | `tr_drussgt_vs_modularbot`, `tr_drussgt_vs_spinbot`, `tr_drussgt_vs_corners` | 33 425 | Tank-Royale bridge captures | + +Each fixture is replayed with a **fresh rack** (as `run_range` does); the test +battles are **held out by battle**, never by tick. The dump is 3 763 298 rows; +per-gun test `n ≈ 133 000` (33 425 ticks × 4 power bins). + +**Per-gun confidence signal defined in source** `[MEASURED]` (the new +`GunPrediction.confidence` field, `common_libs/gun_harness/gun_interface.nim`): + +| gun | intrinsic per-sample signal | source | +|---|---|---| +| GuessFactor | peak GF bin weight `max_i bins[i]` (Eq 4 analogue) | `guess_factor.nim` | +| DecayGF | peak decayed GF bin weight | `decay_gf.nim` | +| KNN | peak Gaussian density over GF candidates (`bestScore`) | `knn_gun.nim` | +| Pattern | match quality `1/(1+bestMatchCost)` | `pattern_matcher.nim` | +| Tsetlin | magnitude of the clamped clause-sum vote `hypot(vx,vy)` | `tsetlin.nim` | +| TMHorizon | side-class margin `|votes1-votes0|/(2*half)` | `tm_horizon.nim` (instrumented; see §7) | +| 8 geometric guns | none — confidence 0.0 | `head_on/linear/circular/wall_bounce/accel_predictor/stop_shot/displacement/averaged_lead` | + +**WARNING — this is exactly the mechanism class the task forbids:** all five are +per-sample statistics of the gun's own internal state at prediction time. None is +a rolling accuracy or hit-rate average. + +--- + +## 2. The mechanism (paper §3, Eqs 4/6/7/8) + +- Member `t` outputs class sums `c^i_{t,d}` (Eq 6); confidence is `c_max` (Eq 4). +- Eq 7 normalises by `alpha_t = max_{d,i}(c) - min_{d,i}(c)`. +- Eq 8: `y_d = argmax_i sum_t (1/alpha_t) c^i_{t,d}`. + +Adapted to an **angle output**: the class is a 0.5° aim bin relative to the +fire-time line of sight (`[-45°, +45°]`, 180 bins); each confident gun casts its +alpha-normalised confidence into the bin of its own aim; the argmax bin wins, and +the aim distance is the confidence-weighted mean distance of that bin's voters. +`alpha_t` is calibrated on the **train** fixtures and frozen for test, so the test +composite never sees test data when choosing its weights (Eq 7's `X` = train). +Two composites are scored: + +- **Composite** = all five confidence guns {Tsetlin, GuessFactor, Pattern, DecayGF, KNN}. +- **CompositeF** = only the guns §3 finds faithful {Pattern, DecayGF, KNN}. + +--- + +## 3. Task A — confidence faithfulness `[MEASURED]` + +Test samples ranked by each gun's own confidence. `rho` = Spearman(confidence, +hit); `bottom/top` = accuracy in the lower/upper confidence half; the two-prop +p is for top vs bottom. + +| gun | n | nonzero | base | rho | rho p | bottom% | top% | verdict | +|---|---:|---:|---:|---:|---:|---:|---:|---| +| Accel | 133 101 | 0 | 21.02 | — | — | 14.47 | 27.57 | NO SIGNAL | +| **Pattern** | 133 100 | 132 860 | 20.45 | **+0.064** | 4.1e-121 | 18.61 | **22.28** | **FAITHFUL (weak)** | +| WallBounce | 133 103 | 0 | 20.15 | — | — | 14.84 | 25.46 | NO SIGNAL | +| AvgLead | 133 100 | 0 | 20.08 | — | — | 14.43 | 25.73 | NO SIGNAL | +| StopShot | 133 078 | 0 | 18.21 | — | — | 12.82 | 23.61 | NO SIGNAL | +| Tsetlin | 132 964 | 105 353 | 18.18 | +0.001 | 0.68 | 17.94 | 18.42 | **USELESS** | +| Linear | 133 104 | 0 | 17.31 | — | — | 13.04 | 21.57 | NO SIGNAL | +| Circular | 133 107 | 0 | 17.27 | — | — | 12.68 | 21.86 | NO SIGNAL | +| **GuessFactor** | 133 108 | 133 108 | 15.89 | **−0.067** | 5.1e-132 | 18.14 | **13.65** | **ANTI-FAITHFUL** | +| **DecayGF** | 133 104 | 133 104 | 14.83 | **+0.133** | <1e-300 | 10.79 | **18.86** | **FAITHFUL (strong)** | +| Displace | 133 105 | 0 | 13.95 | — | — | 11.42 | 16.47 | NO SIGNAL | +| **KNN** | 133 121 | 132 685 | 12.32 | **+0.090** | 2.1e-237 | 9.65 | **14.99** | **FAITHFUL** | +| HeadOn | 133 162 | 0 | 6.18 | — | — | 7.09 | 5.27 | NO SIGNAL | + +Accuracy-vs-confidence curves (deciles of confidence, low→high), the paper's +Figure 2 reproduced per gun: + +``` +Pattern 14% 16% 19% 23% 22% 22% 21% 23% 22% 23% rises, then flat +DecayGF 10% 9% 11% 15% 9% 11% 16% 21% 23% 23% rises (noisy) +KNN 10% 9% 9% 10% 10% 12% 13% 15% 16% 19% monotone rise +GuessFactor 16% 21% 21% 15% 18% 14% 18% 14% 13% 10% FALLS +Tsetlin 13% 25% 12% 21% 19% 20% 16% 18% 20% 19% flat/noise +``` + +**Reading.** DecayGF and KNN are the textbook faithful shapes (accuracy climbs +with confidence). Pattern is faithful but weakly — its one prior is strong +(`18.61% → 22.28%`) and then flattens, i.e. the match cost discriminates +"no/poor match" from "match" but not much *between* matches. GuessFactor is the +surprise and the most important negative: its histogram peak is **anti-faithful**, +consistent with the earlier finding that the GF code path is degenerate on these +`tr-bridge` captures. Tsetlin's `c_max` (clause-sum magnitude) carries **no** +information about whether the shot hits — `[INFERRED]` because the TM's +regression correction is trained on a residual and never learns the surfer +(`tsetlin.nim` header: every variant sat at chance vs a shuffled-control). + +--- + +## 4. Task B — complementary specialists? `[MEASURED]` + +For each pair we split test samples by which gun has the higher **alpha-normalised** +confidence and measure **both** guns on each slice. `A|A` = A's accuracy on the +slice A wins, `B|A` = B's accuracy on that same slice; `p` is a paired McNemar. + +| A | B | A_wins | A\|A | B\|A | p_A | B_wins | A\|B | B\|B | p_B | complementary | +|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---| +| DecayGF | GuessFactor | 43 493 | 19.1 | **20.0** | 4e-14 | 89 608 | 12.8 | **13.9** | 8e-46 | **No** (GF dominates both) | +| DecayGF | KNN | 436 | 30.0 | **30.7** | 0.25 | 132 656 | **14.8** | 12.3 | 4e-99 | **No** (DecayGF wins 0.3% only) | +| DecayGF | Pattern | 26 788 | 11.0 | **14.8** | 1e-51 | 106 279 | 15.8 | **21.9** | 0 | **No** (Pattern dominates) | +| DecayGF | Tsetlin | 121 163 | 14.7 | **18.4** | 3e-211 | 11 795 | **16.2** | 15.4 | 0.045 | **No** | +| GuessFactor | KNN | 44 521 | **12.3** | 8.5 | 2e-84 | 88 570 | **17.7** | 14.3 | 9e-115 | **No** (GF dominates both) | +| GuessFactor | Pattern | 40 605 | 10.6 | **14.3** | 7e-69 | 92 459 | 18.2 | **23.2** | 2e-207 | **No** | +| GuessFactor | Tsetlin | 117 799 | 15.5 | **17.9** | 6e-92 | 15 156 | 19.2 | **20.4** | 0.0012 | **No** | +| KNN | Pattern | 38 485 | 9.7 | **15.8** | 6e-155 | 94 344 | 13.4 | **22.4** | 0 | **No** | +| KNN | Tsetlin | 131 820 | 12.2 | **18.1** | 0 | 804 | 21.6 | **30.0** | 6e-06 | **No** (KNN wins 0.6% only) | +| Pattern | Tsetlin | 121 591 | **21.1** | 18.8 | 2e-58 | 11 221 | **13.6** | 11.6 | 2e-07 | **No** (Pattern dominates) | + +**No pair is complementary.** The pattern in nearly every row is that the more +accurate gun is the same one on *both* slices — the confidence orderings are not +aligned across guns, so "A is confident here" does not mean "A is the expert +here". The two pairs with an unequal split (DecayGF vs KNN, KNN vs Tsetlin) are +ones where the "loser" wins *only* 0.3–0.6% of samples — not a usable slice. A +pair with no complementary slices cannot form a useful composite, and none does. +`[INFERRED]` the mechanism: all five guns estimate the *same* quantity (the +enemy's intercept) from the *same* recorded trajectory; differing booleanisations +create different *noise*, not different *competence regions*, so their confidence +rankings carry no cross-gun information. + +--- + +## 5. Task C — composite vs best single, with the shuffle control `[MEASURED]` + +Held-out test battles, paired per sample. `member_oracle` = accuracy if a +perfect per-sample selector could pick any confidence member (the ceiling for +any member-picking composite; a free-aim oracle could only be higher). + +| arm | accuracy | n | +|---|---:|---:| +| **Accel** (best single overall — no confidence) | **21.02%** | 133 101 | +| **Pattern** (best single *with* confidence / incumbent) | **20.45%** | 133 100 | +| WallBounce | 20.15% | 133 103 | +| AvgLead | 20.08% | 133 100 | +| **CompositeF** (faithful only: Pattern+DecayGF+KNN) | **18.68%** | 133 126 | +| **Composite** (all 5) | **18.41%** | 133 094 | +| Tsetlin | 18.18% | 132 964 | +| **CompositeFShuf** (control) | **18.06%** | 133 119 | +| **CompositeShuf** (control) | **17.61%** | 133 103 | +| GuessFactor | 15.89% | 133 108 | +| DecayGF | 14.83% | 133 104 | +| KNN | 12.32% | 133 121 | +| **member_oracle** (perfect member picker) | **41.72%** | 133 094 | + +Paired comparisons (McNemar on discordant pairs): + +| comparison | hits | opponents | p | +|---|---:|---:|---:| +| Composite vs **Pattern** | 24 488 | 27 215 | **3.1e-126** (composite loses) | +| CompositeF vs **Pattern** | 24 854 | 27 215 | **1.25e-133** (composite loses) | +| Composite vs its **shuffle** | 24 496 | 23 443 | 4.0e-14 (composite wins) | +| CompositeF vs its **shuffle** | 24 864 | 24 040 | 4.0e-15 (composite wins) | +| CompositeF vs Composite | — | — | (faithful-only is marginally better, +0.27pp) | + +Per held-out battle (composite vs Pattern): `tr_drussgt_vs_modularbot` 12.2% vs +13.0%; `tr_drussgt_vs_spinbot` 29.0% vs 33.9%; `tr_drussgt_vs_corners` 22.3% vs +22.4%. The composite loses **all three**. + +**Verdict: the design is dead offline.** The alpha-normalised confidence-weighted +composite does **not** beat the best single gun; it loses to Pattern by ~1.8–2.0pp +with p≈1e-130, and to the best single overall (Accel) by more. The shuffle control +**passes in the weak sense**: the composite is genuinely (p≈1e-14) better than a +version with the same weighting distribution but shuffled competences, so the +confidence signal is not pure noise — but the effect is ~0.6–0.8pp versus a ~2pp +deficit, i.e. real but far too small. The `member_oracle` of **41.72%** shows +enormous headroom exists — it is not reachable from these confidence signals. + +--- + +## 6. Why it fails `[INFERRED]` + +1. **Faithfulness is not competence.** A gun can be perfectly confidence-ordered + and still be worse than another gun everywhere. DecayGF is *more* faithful than + Pattern (rho 0.133 vs 0.064) yet 5.6pp less accurate, so its (correct) ordering + contributes weak votes against a stronger member. +2. **The members are not specialists.** §4 shows the confidence orderings do not + identify competence regions across guns; every gun attacks the whole input + space. The paper's win comes from members that are *good on disjoint subsets*. +3. **The signal is diluted by anti-faithful members.** GuessFactor is anti-faithful + and Tsetlin is useless; CompositeF (faithful only) is better than Composite + (18.68 vs 18.41) and more faithful (rho 0.114 vs 0.009), which confirms the + poison — but removing it still leaves the composite below Pattern. +4. **Alpha normalisation is unstable for online-learning guns.** Eq 7 sets + `alpha_t` to the train range; GuessFactor's histogram accumulates unboundedly + (alpha≈1.5e4 here), so its normalised confidence is tiny on test and it almost + never casts a decisive vote. `[INFERRED]` this mutes the very member whose + confidence was measured anti-faithful. + +--- + +## 7. Limits and caveats + +- **Open-loop corpus.** `[MEASURED]`/`[INFERRED]` The fixtures are recorded + trajectories; the enemy never reacts to the composite's (or any) bullets, and + the `tr-bridge` battles were recorded while **Pattern's selector** was shooting. + A gun that behaves like Pattern is therefore favoured in *framing*. This cannot + rescue the composite: it loses to Pattern **and** to Accel/AvgLead/WallBounce, so + the deficit is not a framing artefact. Per `docs/offline_harness_trust.md`, this + is exactly the closed-loop class of question where the offline harness is + **veto-only** — a negative kills the design, a positive would have proved nothing. +- **Coverage gap (documented LIMIT).** `[MEASURED]` The offline range builds guns + 0..13; TMPATTERN (14) and TMHORIZON (15) are not in it + (`docs/offline_harness_trust.md` §2.8). TMHorizon's `sideConf` is instrumented in + source but not measured here. Since Tsetlin's c_max is already useless and no pair + composes, the omission does not change the verdict. +- **One composite architecture.** `[INFERRED]` The vote uses each gun's scalar + confidence at the bin of its own aim. The paper's members also expose a full + class distribution (GF/KNN do). Feeding those full distributions could refine the + composite, but §4 shows the core premise — cross-gun competence specialisation — + is absent, so a refined vote has no signal to exploit. + +--- + +## 8. Recommendation + +**Do not build or A/B the composite.** The offline gate is a veto and it vetoes. + +Two by-products worth keeping: + +- **GuessFactor / anti-faithful warning.** GuessFactor's own histogram peak is + anti-faithful on this corpus; do **not** use it as a competence signal (e.g. for + a confidence-gated firing decision or a power policy). Pattern, DecayGF and KNN + confidences are faithful and *could* gate a shot ("do not fire when not + confident") — a different, narrower mechanism than the composite, and still + subject to the live veto. +- **The selector conclusion is reinforced.** The rack's problem is not the + *decision statistic* the selector uses (rolling rate vs per-sample confidence): + even a per-sample intrinsic confidence, applied cross-gun, cannot beat Pattern. + The rack does not contain complementary specialists. + +--- + +## 9. Reproduction `[MEASURED]` + +```bash +# 1. record per-sample confidence + hit (≈ 6.5 min; fresh rack per fixture) +nim c -d:release --nimcache:/tmp/nc_j104 --path:common_libs \ + -o:/tmp/measure_tmc common_libs/tests/measure_tmcomposites.nim +/tmp/measure_tmc --out /tmp/tmc_full.jsonl \ + --train tools/fixtures/drussgt_vs_crazy.jsonl tools/fixtures/drussgt_vs_spinbot.jsonl \ + tools/fixtures/drussgt_vs_ramfire.jsonl tools/fixtures/drussgt_vs_corners.jsonl \ + --test tools/fixtures/tr_drussgt_vs_modularbot.jsonl tools/fixtures/tr_drussgt_vs_spinbot.jsonl \ + tools/fixtures/tr_drussgt_vs_corners.jsonl + +# 2. analyse (≈ 40 s); committed result is common_libs/tests/fixtures/tmcomposites_gate.json +python3 common_libs/tests/analyze_tmcomposites.py \ + --input /tmp/tmc_full.jsonl --json common_libs/tests/fixtures/tmcomposites_gate.json +``` + +The recorder threads a new per-sample `confidence` field through +`GunPrediction`/`FeedbackEvent`/`VirtualBullet` (`gun_interface.nim`, +`virtual_bullets.nim`) and populates it in `guess_factor.nim`, `decay_gf.nim`, +`knn_gun.nim`, `pattern_matcher.nim`, `tsetlin.nim`, `tm_horizon.nim`; the field +defaults to 0.0, so every existing caller and test is unchanged. Verified +`test_gun_harness`, `test_vbullet_metric`, `test_wave_pairing`, +`test_tm_horizon`, `test_pattern_radial_offset`, `test_range_rack_parity`, +`test_selector_tiebreak` all pass.