TMComposites gate: per-gun confidence faithful for 3 guns; no pair composes

Adds a per-sample intrinsic-confidence field (GunPrediction.confidence,
threaded through FeedbackEvent/VirtualBullet, populated by Pattern, DecayGF,
KNN, GuessFactor, Tsetlin, TMHorizon) and an offline recorder + analyzer that
reproduce the paper's Figure 2 per gun and its Eq-8 composite.

Measured on 3 held-out tr-bridge DrussGT battles (33k ticks, ~133k samples/gun):
- FAITHFUL: DecayGF (rho +0.133), KNN (+0.090), Pattern (+0.064, weak).
- GuessFactor is ANTI-faithful (rho -0.067); Tsetlin c_max is useless (0.001).
- No pair of guns specialises complementarily: the same gun dominates both
  high-confidence slices in every pair.
- Eq-8 alpha-normalised confidence-weighted composite: 18.41% vs Pattern
  20.45% (McNemar p=3.1e-126). Faithful-only variant 18.68%, still loses.
  Shuffle control passes weakly (composite > shuffle, p=4e-14) so ~0.7pp of
  competence is real but ~2pp short. Offline veto: design is dead.

See docs/tmcomposites_gate.md.
This commit is contained in:
2026-09-25 22:02:28 +02:00
parent d0750ab020
commit f58d65d2e8
12 changed files with 2666 additions and 11 deletions
@@ -35,6 +35,14 @@ type
GunPrediction* = object
## Absolute (x, y) where the gun predicts the enemy will be.
x*, y*: float
confidence*: float
## The gun's INTRINSIC, per-sample confidence in THIS prediction, on any
## positive scale the gun likes (the TMComposites gate normalises it by
## the sample range, Eq 7). 0.0 means "this gun has no confidence signal"
## (every deterministic geometric gun): such a gun casts no composite
## vote. Deliberately NOT a rolling hit-rate: that is the selector
## mechanism already measured negative (docs/gun_rack_analysis.md).
## See docs/tmcomposites_gate.md and common_libs/tests/measure_tmcomposites.nim.
FeedbackEvent* = object
## Outcome of a resolved virtual bullet.
@@ -45,6 +53,7 @@ type
powerBin*: int ## index into PowerBins the bullet belongs to
missDistance*: float ## px; < BotRadius = hit
hit*: bool
confidence*: float ## copied from the spawn-time GunPrediction (above)
proc bulletSpeed*(power: float): float {.inline.} =
20.0 - 3.0 * power
+6 -2
View File
@@ -348,6 +348,7 @@ type
bulletSpeed*: float
travelDist*: float ## accumulated px so far
fireDist*: float ## distance to target at fire time
confidence*: float ## spawn-time GunPrediction.confidence (TMComposites gate)
active*: bool
pointScored*: bool ## the parallel point-model score for this flight has
## been recorded (tie-break mode only)
@@ -461,6 +462,7 @@ proc spawnBullets*(t: var VirtualTracker, gunId: GunId,
bulletSpeed: speed,
travelDist: 0.0,
fireDist: fireDist,
confidence: pred.confidence,
active: true,
pointScored: false,
hitSeen: false,
@@ -643,7 +645,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState,
t.fitness[b.targetId][b.gunId].bins[b.powerBin].record(hit)
let fe = FeedbackEvent(
prediction: GunPrediction(x: b.aimX, y: b.aimY),
prediction: GunPrediction(x: b.aimX, y: b.aimY, confidence: b.confidence),
actualX: ex,
actualY: ey,
bulletPower: PowerBins[b.powerBin],
@@ -651,6 +653,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState,
powerBin: b.powerBin,
missDistance: missDist,
hit: hit,
confidence: b.confidence,
)
onResolved(b.gunId, b.powerBin, fe)
b.active = false
@@ -720,7 +723,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState,
if b.targetId in t.fitness:
t.fitness[b.targetId][b.gunId].bins[b.powerBin].record(b.hitSeen)
let fe = FeedbackEvent(
prediction: GunPrediction(x: b.aimX, y: b.aimY),
prediction: GunPrediction(x: b.aimX, y: b.aimY, confidence: b.confidence),
actualX: rx,
actualY: ry,
bulletPower: PowerBins[b.powerBin],
@@ -728,6 +731,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState,
powerBin: b.powerBin,
missDistance: missDist,
hit: b.hitSeen,
confidence: b.confidence,
)
onResolved(b.gunId, b.powerBin, fe)
b.active = false
+1
View File
@@ -113,6 +113,7 @@ proc predict*(g: var DecayGFGun, state: WorldState, bulletSpeed: float): GunPred
GunPrediction(
x: clamp(px, BotRadius, state.arenaWidth - BotRadius),
y: clamp(py, BotRadius, state.arenaHeight - BotRadius),
confidence: g.bins[peak], # class-sum max over GF bins (see guess_factor.nim)
)
proc onResult*(g: var DecayGFGun, e: FeedbackEvent) =
+3
View File
@@ -141,6 +141,9 @@ proc predict*(g: var GFGun, state: WorldState, bulletSpeed: float): GunPredictio
GunPrediction(
x: clamp(px, BotRadius, state.arenaWidth - BotRadius),
y: clamp(py, BotRadius, state.arenaHeight - BotRadius),
# TMComposites Eq 4 analogue: the GF histogram is the class distribution over
# GF bins, so the class-sum max c_max is the peak bin's accumulated weight.
confidence: g.bins[peak],
)
proc onResult*(g: var GFGun, e: FeedbackEvent) =
+4
View File
@@ -299,6 +299,10 @@ proc predict*(g: var KNNGun, state: WorldState, bulletSpd: float): GunPrediction
GunPrediction(
x: clamp(px, BotRadius, state.arenaWidth - BotRadius),
y: clamp(py, BotRadius, state.arenaHeight - BotRadius),
# TMComposites Eq 4 analogue: the KNN Gaussian density over GF candidates is
# the class distribution; bestScore is the class-sum max. Cold-start returns
# (no data / < 5 neighbours) leave the default 0.0 = no vote.
confidence: bestScore,
)
proc onResult*(g: var KNNGun, e: FeedbackEvent) =
+13 -3
View File
@@ -50,6 +50,10 @@ type
cacheValid: bool
cacheTick: int
bestMatch: int ## -1 = no usable match (linear fallback)
lastMatchScore*: float ## best pattern-match cost (lower = better);
## set by findBestMatch, exposed as the gun's
## intrinsic per-sample confidence
## (TMComposites gate, docs/tmcomposites_gate.md)
playStart: int
playAvail: int
pathX: array[HistorySize + 1, float]
@@ -87,9 +91,11 @@ proc linearPredict(state: WorldState, bulletSpeed: float): (float, float) =
# --- pattern search + play-forward ---
proc findBestMatch(g: PatternMatcherGun): int =
proc findBestMatch(g: var PatternMatcherGun): int =
## Speed-independent history search. Returns the start index of the best
## matching pattern, or -1 when there is not enough history.
## matching pattern, or -1 when there is not enough history. Stores the best
## match cost in `g.lastMatchScore` for the confidence readout.
g.lastMatchScore = Inf
if g.count < PatternLen * 2:
return -1
@@ -111,6 +117,7 @@ proc findBestMatch(g: PatternMatcherGun): int =
if score < bestScore:
bestScore = score
bestMatch = i
g.lastMatchScore = bestScore
bestMatch
proc buildPath(g: var PatternMatcherGun, state: WorldState, bestMatch: int) =
@@ -222,7 +229,10 @@ proc predict*(g: var PatternMatcherGun, state: WorldState,
return g.applyRadial(state, px, py)
let (px, py) = g.projectFromPath(state, bulletSpeed)
g.applyRadial(state, px, py)
result = g.applyRadial(state, px, py)
# Match quality as a confidence: a perfect historical match (cost 0) gives 1.0,
# a worse match decays toward 0. Deterministic and per-sample.
result.confidence = 1.0 / (1.0 + max(0.0, g.lastMatchScore))
proc onResult*(g: var PatternMatcherGun, e: FeedbackEvent) =
discard # pattern matcher learns from movement observation, not feedback
+8 -5
View File
@@ -1091,13 +1091,16 @@ proc predict*(g: var TmHorizonGun, state: WorldState,
let sgn = if side == 1: 1.0 else: -1.0
shift = sgn * g.shiftDeg * magScale
let pred =
if shift == 0.0: base
else: tmhApplyShift(state.selfX, state.selfY, base.x, base.y, shift)
# TMComposites Eq 4 analogue: the side-machine's class-sum margin (normalised
# by the clause half-count) is the per-sample confidence in WHICH WAY to shift.
var predOut = base
if shift != 0.0:
predOut = tmhApplyShift(state.selfX, state.selfY, base.x, base.y, shift)
if warm: predOut.confidence = ev.sideConf
let aimDeg = radToDeg(arctan2(pred.y - state.selfY, pred.x - state.selfX))
let aimDeg = radToDeg(arctan2(predOut.y - state.selfY, predOut.x - state.selfX))
g.tmhLog(state, bulletSpeed, h, side, mag, ev.sideConf, shift, aimDeg)
pred
predOut
proc onResult*(g: var TmHorizonGun, e: FeedbackEvent) =
## The label comes from our own observation ring, not from virtual-bullet
+8 -1
View File
@@ -413,7 +413,14 @@ proc predict*(g: var TsetlinGun, state: WorldState, bulletSpeed: float): GunPred
alive: true,
)
GunPrediction(x: predX, y: predY)
GunPrediction(
x: predX,
y: predY,
# TMComposites Eq 4: the two output teams' clamped clause sums are (vx, vy);
# their magnitude is how hard the machine is voting to move the correction.
# Warm-up fallback (window not full) leaves the default 0.0 = no vote.
confidence: hypot(vx, vy),
)
proc onResult*(g: var TsetlinGun, e: FeedbackEvent) =
inc g.shotCount
+438
View File
@@ -0,0 +1,438 @@
#!/usr/bin/env python3
"""TMComposites GATE — analyse the per-sample confidence dump.
Reads the JSONL produced by ``common_libs/tests/measure_tmcomposites.nim``
(one row per resolved virtual bullet: fixture, split, gun, tick, bin, conf,
hit, relDeg, range, missPx) and answers the three questions of
``docs/tmcomposites_gate.md``:
A. Is each gun's intrinsic confidence FAITHFUL?
Rank samples by the gun's own confidence; report the accuracy-vs-confidence
curve, Spearman(confidence, hit), and top-half vs bottom-half accuracy with
a two-proportion p-value.
B. Are the faithful guns COMPLEMENTARY SPECIALISTS?
For each pair, on the samples where A's alpha-normalised confidence beats
B's, is A the more accurate one? Report the two slices and the win rates.
C. Does the Eq-8 alpha-normalised confidence-weighted composite beat the best
single gun on held-out battles, and is the gain attributable to competence?
The recorder already ran ``Composite`` and its within-sample
confidence-shuffle control ``CompositeShuf`` through the SAME virtual-bullet
geometry, so the comparison is paired (McNemar).
Pure stdlib (no numpy/scipy on this machine). MEASURED = every number printed;
INFERRED = the causal reading in the doc.
"""
from __future__ import annotations
import argparse
import collections
import json
import math
import random
import sys
DETERMINISTIC = ["HeadOn", "Linear", "Circular", "WallBounce", "Accel",
"StopShot", "Displace", "AvgLead"]
# ───────────────────────────── stats (stdlib) ──────────────────────────────
def mean(xs):
return sum(xs) / len(xs) if xs else float("nan")
def spearman(xs, ys):
"""Spearman rho with average ranks for ties, and a normal-approx p."""
n = len(xs)
if n < 3:
return float("nan"), float("nan")
def ranks(v):
order = sorted(range(n), key=lambda i: v[i])
r = [0.0] * n
i = 0
while i < n:
j = i
while j + 1 < n and v[order[j + 1]] == v[order[i]]:
j += 1
avg = (i + j) / 2.0 + 1.0
for k in range(i, j + 1):
r[order[k]] = avg
i = j + 1
return r
rx, ry = ranks(xs), ranks(ys)
mx, my = mean(rx), mean(ry)
num = sum((a - mx) * (b - my) for a, b in zip(rx, ry))
den = math.sqrt(sum((a - mx) ** 2 for a in rx) * sum((b - my) ** 2 for b in ry))
if den == 0:
return 0.0, 1.0
rho = num / den
rho = max(-1.0, min(1.0, rho))
z = rho * math.sqrt(n - 1)
p = math.erfc(abs(z) / math.sqrt(2.0))
return rho, p
def norm_two_prop(z):
return math.erfc(abs(z) / math.sqrt(2.0))
def two_prop_p(h1, n1, h2, n2):
if n1 == 0 or n2 == 0:
return float("nan"), float("nan")
p1, p2 = h1 / n1, h2 / n2
p = (h1 + h2) / (n1 + n2)
se = math.sqrt(p * (1 - p) * (1 / n1 + 1 / n2))
if se == 0:
return float("nan"), float("nan")
z = (p1 - p2) / se
return z, norm_two_prop(z)
def binom_two_sided(k, n, p=0.5):
"""Exact two-sided binomial p (used for McNemar's discordant pairs)."""
if n == 0:
return 1.0
def pmf(i):
return math.comb(n, i) * p ** i * (1 - p) ** (n - i)
obs = pmf(k)
tot = 0.0
for i in range(n + 1):
if pmf(i) <= obs + 1e-12:
tot += pmf(i)
return min(1.0, tot)
def mcnemar_p(a_hit_b_miss, a_miss_b_hit):
"""McNemar: exact binomial for small discordant counts, normal approx for
large (the exact branch overflows math.comb for n in the hundred-thousands)."""
b, c = a_hit_b_miss, a_miss_b_hit
n = b + c
if n == 0:
return 1.0
if n < 500:
return binom_two_sided(b, n)
z = (b - c) / math.sqrt(n)
return math.erfc(abs(z) / math.sqrt(2.0))
# ───────────────────────────── data loading ────────────────────────────────
class Dump:
def __init__(self, path):
self.rows = [] # list of dicts
self.by_gun = collections.defaultdict(list)
# key -> {gun: (conf, hit)}
self.by_sample = collections.defaultdict(dict)
with open(path) as f:
for line in f:
line = line.strip()
if not line:
continue
o = json.loads(line)
self.rows.append(o)
self.by_gun[o["gun"]].append(o)
self.by_sample[(o["fixture"], o["tick"], o["bin"])][o["gun"]] = (
o["conf"], bool(o["hit"]))
def guns(self):
return sorted(self.by_gun.keys())
def split(self, gun, split):
return [r for r in self.by_gun[gun] if r["split"] == split]
def faithful_curve(rows, bins=10):
rows = sorted(rows, key=lambda r: r["conf"])
n = len(rows)
if n == 0:
return []
out = []
for b in range(bins):
lo = b * n // bins
hi = (b + 1) * n // bins
chunk = rows[lo:hi]
if not chunk:
continue
out.append(dict(
lo=chunk[0]["conf"], hi=chunk[-1]["conf"], n=len(chunk),
hit=mean([1.0 if r["hit"] else 0.0 for r in chunk]),
))
return out
def faithfulness_report(dump):
"""Question A: per-gun faithfulness on the pooled test samples."""
out = {}
for gun in dump.guns():
rows = dump.split(gun, "test")
if not rows:
continue
confs = [r["conf"] for r in rows]
hits = [1.0 if r["hit"] else 0.0 for r in rows]
nz = sum(1 for c in confs if c > 1e-12)
rho, p = spearman(confs, hits)
order = sorted(range(len(rows)), key=lambda i: confs[i])
half = len(order) // 2
lo_idx, hi_idx = order[:half], order[half:]
h_lo = sum(hits[i] for i in lo_idx)
h_hi = sum(hits[i] for i in hi_idx)
z, phalf = two_prop_p(h_hi, len(hi_idx), h_lo, len(lo_idx))
out[gun] = dict(
n=len(rows), nonzero_conf=nz,
base=mean(hits), rho=rho, rho_p=p,
bottom_half_acc=(h_lo / len(lo_idx)) if lo_idx else float("nan"),
top_half_acc=(h_hi / len(hi_idx)) if hi_idx else float("nan"),
half_z=z, half_p=phalf,
curve=faithful_curve(rows),
)
return out
def alphas(dump):
"""alpha_t = max-min of the gun's confidence over the TRAIN samples (Eq 7)."""
out = {}
for gun in dump.guns():
rows = dump.split(gun, "train")
if not rows:
continue
cs = [r["conf"] for r in rows]
out[gun] = max(1e-12, max(cs) - min(min(cs), 0.0))
return out
def normalised(conf, gun, alpha):
return conf / alpha.get(gun, 1.0)
def complementarity(dump, alphas_, faithful):
"""Question B: pairwise complementary slices over the test samples.
For each pair we split the samples by which gun has the higher
alpha-normalised confidence and, ON EACH SLICE, measure BOTH guns' accuracy.
A pair is complementary when each gun is the more accurate one on its own
winning slice, and each slice is a substantial (>=10%) share.
"""
tfs = test_fixtures(dump)
pairs = []
names = [g for g in faithful
if faithful[g]["nonzero_conf"] > 0
and not g.startswith("Composite")]
for i in range(len(names)):
for j in range(i + 1, len(names)):
A, B = names[i], names[j]
a_win = b_win = 0
# hits ON the A-winning slice, and ON the B-winning slice
aA = bA = aB = bB = 0
aA_only = bA_only = aB_only = bB_only = 0
for key, guns in dump.by_sample.items():
if key[0] not in tfs:
continue
if A not in guns or B not in guns:
continue
ca = normalised(guns[A][0], A, alphas_)
cb = normalised(guns[B][0], B, alphas_)
if ca <= 0 and cb <= 0:
continue
ha, hb = int(guns[A][1]), int(guns[B][1])
if ca > cb:
a_win += 1; aA += ha; bA += hb
if ha and not hb: aA_only += 1
elif hb and not ha: bA_only += 1
elif cb > ca:
b_win += 1; aB += ha; bB += hb
if ha and not hb: aB_only += 1
elif hb and not ha: bB_only += 1
both = a_win + b_win
def frac(h, n):
return h / n if n else float("nan")
accA_on_A, accB_on_A = frac(aA, a_win), frac(bA, a_win)
accA_on_B, accB_on_B = frac(aB, b_win), frac(bB, b_win)
comp = (both > 0 and a_win >= 0.10 * both and b_win >= 0.10 * both
and accA_on_A > accB_on_A and accB_on_B > accA_on_B)
pairs.append(dict(
A=A, B=B, a_win=a_win, b_win=b_win, both=both,
A_acc_on_A_slice=accA_on_A, B_acc_on_A_slice=accB_on_A,
A_acc_on_B_slice=accA_on_B, B_acc_on_B_slice=accB_on_B,
p_on_A_slice=mcnemar_p(aA_only, bA_only),
p_on_B_slice=mcnemar_p(bB_only, aB_only),
complementary=comp))
return pairs
def compl_ok(a_win, b_win, both, aa, ba):
"""Retained for backwards compatibility; superseded by the per-slice test
in `complementarity`."""
if both == 0 or a_win == 0 or b_win == 0:
return False
return a_win >= 0.10 * both and b_win >= 0.10 * both
# ───────────────────────── composite comparison ────────────────────────────
def paired(dump, gun_a, gun_b, tfs):
"""Return (a_hit_b_miss, a_miss_b_hit, a_hits, b_hits) over the test fixtures."""
ab = ba = na = nb = 0
for key, guns in dump.by_sample.items():
if key[0] not in tfs:
continue
if gun_a in guns and gun_b in guns:
a = guns[gun_a][1]
b = guns[gun_b][1]
if a and not b:
ab += 1
elif b and not a:
ba += 1
if a:
na += 1
if b:
nb += 1
return ab, ba, na, nb
def test_fixtures(dump):
out = set()
for r in dump.rows:
if r["split"] == "test":
out.add(r["fixture"])
return out
def composite_report(dump):
"""Question C: composite vs best single vs shuffle control on test."""
tfs = test_fixtures(dump)
acc = {}
n = {}
for gun in dump.guns():
rows = [r for r in dump.by_gun[gun] if r["split"] == "test"]
if not rows:
continue
acc[gun] = mean([1.0 if r["hit"] else 0.0 for r in rows])
n[gun] = len(rows)
members = [g for g in dump.guns()
if not g.startswith("Composite") and g not in DETERMINISTIC]
best_member = max(members, key=lambda g: acc[g]) if members else None
res = dict(acc=acc, n=n, best_member=best_member)
# Oracle ceiling: if a perfect per-sample selector could pick ANY member,
# how often would it hit? This bounds what a member-selection composite
# could ever reach (the vote can do worse but not better than this).
oracle = 0
oracle_n = 0
for key, guns in dump.by_sample.items():
if key[0] not in tfs:
continue
hits = [guns[m][1] for m in members if m in guns]
if hits:
oracle += int(any(hits))
oracle_n += 1
res["member_oracle"] = (oracle / oracle_n) if oracle_n else float("nan")
res["member_oracle_n"] = oracle_n
comps = [g for g in dump.guns() if g.startswith("Composite") and not g.endswith("Shuf")]
res["composites"] = {}
for comp in comps:
entry = {}
if best_member:
ab, ba, na, nb = paired(dump, comp, best_member, tfs)
entry["vs_best"] = dict(best=best_member, comp_hit=na, best_hit=nb,
total=n[comp], mcnemar_ab=ab, mcnemar_ba=ba,
p=mcnemar_p(ab, ba))
if "Pattern" in acc:
ab, ba, na, nb = paired(dump, comp, "Pattern", tfs)
entry["vs_pattern"] = dict(comp_hit=na, pattern_hit=nb, total=n[comp],
mcnemar_ab=ab, mcnemar_ba=ba, p=mcnemar_p(ab, ba))
shuf = comp + "Shuf"
if shuf in acc:
ab, ba, na, nb = paired(dump, comp, shuf, tfs)
entry["vs_shuffle"] = dict(comp_hit=na, shuffle_hit=nb, total=n[comp],
mcnemar_ab=ab, mcnemar_ba=ba, p=mcnemar_p(ab, ba))
res["composites"][comp] = entry
# per-fixture composite vs best single vs shuffle
per = {}
for fx in sorted(tfs):
row = {}
for gun in ([best_member] if best_member else []) + comps:
rows = [r for r in dump.by_gun[gun] if r["split"] == "test" and r["fixture"] == fx]
if rows:
row[gun] = dict(n=len(rows), acc=mean([1.0 if r["hit"] else 0.0 for r in rows]))
per[fx] = row
res["per_fixture"] = per
return res
# ───────────────────────────────── main ────────────────────────────────────
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--input", default="/tmp/tmc_full.jsonl")
ap.add_argument("--json", default=None)
args = ap.parse_args()
dump = Dump(args.input)
print(f"rows={len(dump.rows)} guns={len(dump.guns())} "
f"test fixtures={sorted(test_fixtures(dump))}")
a = faithfulness_report(dump)
print("\n=== A. FAITHFULNESS (test samples; rank by own confidence) ===")
print(f"{'gun':<14}{'n':>7}{'nonzero':>8}{'base%':>7}{'rho':>8}{'rho_p':>9}"
f"{'bot%':>7}{'top%':>7}{'z':>7}{'p':>9} verdict")
verdicts = {}
for gun in sorted(a, key=lambda g: -a[g]["base"]):
r = a[gun]
if r["nonzero_conf"] == 0:
v = "NO SIGNAL"
elif r["rho_p"] < 0.01 and r["rho"] > 0.05:
v = "FAITHFUL"
elif r["rho_p"] < 0.01 and r["rho"] < -0.05:
v = "ANTI-FAITHFUL"
else:
v = "USELESS"
verdicts[gun] = v
print(f"{gun:<14}{r['n']:>7}{r['nonzero_conf']:>8}{100*r['base']:>7.2f}"
f"{r['rho']:>8.3f}{r['rho_p']:>9.2g}{100*r['bottom_half_acc']:>7.2f}"
f"{100*r['top_half_acc']:>7.2f}{r['half_z']:>7.2f}{r['half_p']:>9.2g} {v}")
print("\ncurves (deciles, low->high confidence):")
for gun in sorted(a, key=lambda g: -a[g]["base"]):
if verdicts[gun] == "NO SIGNAL":
continue
cur = " ".join(f"{100*c['hit']:.0f}%" for c in a[gun]["curve"])
print(f" {gun:<14} {cur}")
al = alphas(dump)
pairs = complementarity(dump, al, a)
print("\n=== B. PAIRWISE COMPLEMENTARITY (test; alpha-normalised confidence) ===")
print(f"{'A':<14}{'B':<14}{'A_wins':>8}{'A|A':>7}{'B|A':>7}{'p_A':>9}| {'B_wins':>8}{'A|B':>7}{'B|B':>7}{'p_B':>9} comp")
for p in pairs:
print(f"{p['A']:<14}{p['B']:<14}{p['a_win']:>8}"
f"{100*p['A_acc_on_A_slice']:>7.1f}{100*p['B_acc_on_A_slice']:>7.1f}{p['p_on_A_slice']:>9.2g}| "
f"{p['b_win']:>8}{100*p['A_acc_on_B_slice']:>7.1f}"
f"{100*p['B_acc_on_B_slice']:>7.1f}{p['p_on_B_slice']:>9.2g} {p['complementary']}")
print(" (A|A = A's accuracy on the slice A wins; B|A = B's accuracy on that same slice; etc.)")
c = composite_report(dump)
print("\n=== C. COMPOSITE vs BEST SINGLE vs SHUFFLE CONTROL (test) ===")
for gun in sorted(c["acc"], key=lambda g: -c["acc"][g]):
print(f" {gun:<14} {100*c['acc'][gun]:>6.2f}% n={c['n'][gun]}")
if "member_oracle" in c:
print(f" member_oracle (any member hits, per sample) : {100*c['member_oracle']:.2f}%")
for comp, entry in c.get("composites", {}).items():
print(f" {comp}:")
for k, r in entry.items():
print(f" {k}: {r}")
print("\nper-fixture:")
for fx, row in c["per_fixture"].items():
parts = " ".join(f"{g}={100*v['acc']:.1f}%({v['n']})" for g, v in row.items())
print(f" {fx:<32} {parts}")
if args.json:
blob = dict(
faithfulness=a, alphas=al, verdicts=verdicts,
complementarity=pairs, composite=c,
test_fixtures=sorted(test_fixtures(dump)),
)
with open(args.json, "w") as f:
json.dump(blob, f, indent=2)
print(f"\n[json] wrote {args.json}")
if __name__ == "__main__":
main()
File diff suppressed because it is too large Load Diff
+285
View File
@@ -0,0 +1,285 @@
## TMComposites GATE — offline per-sample confidence + outcome recorder.
##
## Implements the measurement half of docs/tmcomposites_gate.md. It replays
## recorded fixtures through the SHIPPED rack exactly as `offline_range` does
## (same `VirtualTracker`, same `bmPath` virtual-bullet ground truth), but also
## captures, for every resolved virtual bullet, the gun's INTRINSIC per-sample
## `GunPrediction.confidence` (see gun_interface.nim) and the bullet's aim
## bearing relative to the fire-time line of sight.
##
## It also runs two COMPOSITE arms through the SAME tracker, so their hits are
## scored by the identical geometry as every member:
## * Composite — TMComposites Eq 8: each confident gun casts its
## alpha-normalised confidence into the angular bin of its
## own aim; the argmax bin wins.
## * CompositeShuf — the MANDATORY control: the same confidences are randomly
## permuted among the members within each sample, so the
## weighting distribution is preserved but competence is
## destroyed.
##
## alpha (Eq 7) is calibrated on the TRAIN fixtures (per member: max-min of its
## confidence over train) and then frozen for the TEST fixtures, so the test
## composite never sees test labels while choosing its weights.
##
## Output: one JSONL row per resolved bullet, to --out. Aggregation lives in
## common_libs/tests/analyze_tmcomposites.py.
##
## Usage:
## nim c -r --nimcache:/tmp/nc_j104 common_libs/tests/measure_tmcomposites.nim \
## --out /tmp/tmc.jsonl \
## --train fx_a.jsonl fx_b.jsonl --test fx_c.jsonl
import std/[os, strformat, json, math, strutils, random, algorithm, tables, times]
import gun_harness/offline_range
import gun_harness/gun_interface
import gun_harness/virtual_bullets
import range_guns
const
BinDeg = 0.5
HalfSpanDeg = 45.0
NumBins = int(2.0 * HalfSpanDeg / BinDeg)
## guns with a genuine intrinsic confidence signal (deterministic geometric
## guns leave GunPrediction.confidence at 0.0 and cast no composite vote).
ConfidentGuns = ["Tsetlin", "GuessFactor", "Pattern", "DecayGF", "KNN"]
type
Spawn = object
relDeg: float
range: float
conf: float
proc wrapDeg(d: float): float =
result = d
while result > 180.0: result -= 360.0
while result < -180.0: result += 360.0
proc binOf(relDeg: float): int =
result = int(floor((relDeg + HalfSpanDeg) / BinDeg))
if result < 0: result = 0
elif result >= NumBins: result = NumBins - 1
var alphaByName: ref Table[string, float]
proc alphaFor(name: string): float =
## Eq 7 alpha_t for a member, keyed by GUN NAME so each composite's member
## order cannot mis-assign a scale. 1.0 until calibration writes the table.
if alphaByName.isNil: return 1.0
alphaByName[].getOrDefault(name, 1.0)
proc makeComposite(name: string, members: seq[GunDriver],
shuffle: bool): GunDriver =
## One composite arm. `members` are the SAME driver closures the tracker uses
## for the member guns, so the composite reads their live state (calling
## predict twice in a tick is idempotent for every gun: GF/KNN guard their
## wave queue on (tick,bin), Pattern/Tsetlin/TMHorizon cache per tick).
result.name = name
let memberList = members
result.predictCb = proc(state: WorldState, speed: float): GunPrediction =
let n = memberList.len
var rels = newSeq[float](n)
var dists = newSeq[float](n)
var confs = newSeq[float](n)
let los = arctan2(state.enemyY - state.selfY, state.enemyX - state.selfX)
for i in 0..<n:
let p = memberList[i].predictCb(state, speed)
rels[i] = wrapDeg(radToDeg(arctan2(p.y - state.selfY, p.x - state.selfX) - los))
dists[i] = hypot(p.x - state.selfX, p.y - state.selfY)
confs[i] = p.confidence
if shuffle:
for i in countdown(n - 1, 1):
let j = rand(i)
swap(confs[i], confs[j])
var votes = newSeq[float](NumBins)
var total = 0.0
for i in 0..<n:
let a = max(1e-12, alphaFor(memberList[i].name))
let w = confs[i] / a
if w <= 0.0: continue
votes[binOf(rels[i])] += w
total += w
if total <= 0.0:
# cold: no member has any confidence yet. Make no claim.
return GunPrediction(x: state.enemyX, y: state.enemyY, confidence: 0.0)
var best = 0
for b in 1..<NumBins:
if votes[b] > votes[best]: best = b
# aim distance = confidence-weighted mean distance of the winning bin's voters
var dsum = 0.0
var wsum = 0.0
for i in 0..<n:
if binOf(rels[i]) != best: continue
let a = max(1e-12, alphaFor(memberList[i].name))
let w = confs[i] / a
dsum += dists[i] * w
wsum += w
let d = if wsum > 1e-12: dsum / wsum else: hypot(state.enemyX - state.selfX,
state.enemyY - state.selfY)
let ang = los + degToRad(-HalfSpanDeg + (best.float + 0.5) * BinDeg)
GunPrediction(x: state.selfX + cos(ang) * d,
y: state.selfY + sin(ang) * d,
confidence: votes[best])
result.resultCb = proc(e: FeedbackEvent) = discard
result.readyCb = nil
proc buildRack(): tuple[drivers: seq[GunDriver], memberLocal: seq[int]] =
## Fresh members + the composite arms, mirroring the standard per-fixture
## replay (guns start cold for every fixture, exactly as run_range does).
##
## Composite — all 5 confidence guns (Tsetlin, GF, Pattern, DecayGF, KNN)
## CompositeShuf — its within-sample confidence shuffle control
## CompositeF — only the guns the faithfulness test finds FAITHFUL
## (Pattern, DecayGF, KNN); GF is anti-faithful and Tsetlin
## useless, so this is the strongest reasonable variant
## CompositeFShuf — its shuffle control
let allDrivers = buildAllGunDrivers(seed = 1)
var members: seq[GunDriver]
var memberLocal: seq[int]
var faithMembers: seq[GunDriver]
for i, d in allDrivers:
if d.name in ConfidentGuns:
members.add d
memberLocal.add i
if d.name in ["Pattern", "DecayGF", "KNN"]:
faithMembers.add d
doAssert members.len == ConfidentGuns.len
result.drivers = allDrivers
result.drivers.add makeComposite("Composite", members, shuffle = false)
result.drivers.add makeComposite("CompositeShuf", members, shuffle = true)
result.drivers.add makeComposite("CompositeF", faithMembers, shuffle = false)
result.drivers.add makeComposite("CompositeFShuf", faithMembers, shuffle = true)
result.memberLocal = memberLocal
proc runFixture(fx: Fixture, fixtureName, split: string,
calMin, calMax: ref seq[float], outFile: File) =
let (drivers, memberLocal) = buildRack()
var byTick = initTable[int, WorldState]()
for s in fx.states: byTick[s.tick] = s
var spawns = initTable[tuple[gunId, tick, bin: int], Spawn]()
var tracker = initTracker(drivers.len, ActiveMetric)
for si in 0..<fx.states.len:
let state = fx.states[si]
for gi in 0..<drivers.len:
var preds: array[len(PowerBins), GunPrediction]
var sps: array[len(PowerBins), Spawn]
let los = arctan2(state.enemyY - state.selfY, state.enemyX - state.selfX)
for i in 0..<len(PowerBins):
preds[i] = drivers[gi].predictCb(state, bulletSpeed(PowerBins[i]))
sps[i] = Spawn(
relDeg: wrapDeg(radToDeg(arctan2(preds[i].y - state.selfY,
preds[i].x - state.selfX) - los)),
range: hypot(preds[i].x - state.selfX, preds[i].y - state.selfY),
conf: preds[i].confidence)
let ready = if drivers[gi].readyCb == nil: true else: drivers[gi].readyCb()
if ready:
for i in 0..<len(PowerBins):
spawns[(gi, state.tick, i)] = sps[i]
tracker.spawnBullets(gi, preds, state, fx.enemyId)
var enemyPositions: Table[int, tuple[x, y: float, lastSeenTick: int, alive: bool]]
if state.enemies.len > 0:
for e in state.enemies:
enemyPositions[e.id] = (x: e.x, y: e.y, lastSeenTick: e.lastSeenTick, alive: true)
else:
enemyPositions[fx.enemyId] = (x: state.enemyX, y: state.enemyY,
lastSeenTick: state.tick, alive: true)
tracker.tickBullets(state, enemyPositions,
proc(gunId: int, binIdx: int, e: FeedbackEvent) =
# The gun must LEARN from its own resolved bullet (exactly as
# offline_range.replayFixture does); the recorder is an extra hook.
drivers[gunId].resultCb(e)
let key = (gunId, e.fireTick, binIdx)
if not spawns.hasKey(key): return
let sp = spawns[key]
spawns.del(key)
if split == "train":
for mi, gid in memberLocal:
if gunId == gid:
calMin[][mi] = min(calMin[][mi], sp.conf)
calMax[][mi] = max(calMax[][mi], sp.conf)
outFile.writeLine($(%*{
"fixture": fixtureName,
"split": split,
"gun": drivers[gunId].name,
"tick": e.fireTick,
"bin": binIdx,
"conf": sp.conf,
"hit": e.hit,
"relDeg": sp.relDeg,
"range": sp.range,
"missPx": e.missDistance,
}))
)
proc loadAll(paths: seq[string]): seq[tuple[name: string, fx: Fixture]] =
for p in paths:
let fx = loadFixture(p)
let name = extractFilename(p).replace(".jsonl", "")
result.add (name: name, fx: fx)
proc main() =
var outPath = "/tmp/tmc.jsonl"
var trainPaths, testPaths: seq[string]
var args: seq[string]
for i in 1..paramCount(): args.add paramStr(i)
var mode = ""
for a in args:
if a == "--out": mode = "out"; continue
if a == "--train": mode = "train"; continue
if a == "--test": mode = "test"; continue
case mode
of "out": outPath = a
of "train": trainPaths.add a
of "test": testPaths.add a
else: discard
var calMinRef = new(seq[float])
var calMaxRef = new(seq[float])
calMinRef[] = newSeq[float](ConfidentGuns.len)
calMaxRef[] = newSeq[float](ConfidentGuns.len)
for i in 0..<ConfidentGuns.len:
calMinRef[][i] = 1e18
calMaxRef[][i] = -1e18
let trainFx = loadAll(trainPaths)
let testFx = loadAll(testPaths)
let outFile = open(outPath, fmWrite)
defer: outFile.close()
randomize(20250925)
echo fmt"# TMComposites recorder: {trainFx.len} train fixtures, {testFx.len} test fixtures"
echo fmt"# members: {ConfidentGuns}"
var t0 = epochTime()
for (name, fx) in trainFx:
let a = epochTime()
runFixture(fx, name, "train", calMinRef, calMaxRef, outFile)
echo fmt" [train] {name:<28} ticks={fx.states.len:>6} {epochTime()-a:6.1f}s"
# Eq 7 alpha_t = max-min over the training input set; frozen for test.
alphaByName = new(Table[string, float])
for i in 0..<ConfidentGuns.len:
alphaByName[][ConfidentGuns[i]] =
max(1e-9, calMaxRef[][i] - min(calMinRef[][i], 0.0))
echo "# alpha_t (Eq 7, from train): " &
(block:
var s = ""
for i in 0..<ConfidentGuns.len:
s.add fmt"{ConfidentGuns[i]}={alphaByName[][ConfidentGuns[i]]:.4g} "
s)
for (name, fx) in testFx:
let a = epochTime()
runFixture(fx, name, "test", calMinRef, calMaxRef, outFile)
echo fmt" [test ] {name:<28} ticks={fx.states.len:>6} {epochTime()-a:6.1f}s"
echo fmt"# done in {epochTime()-t0:.1f}s -> {outPath}"
when isMainModule:
main()