TMComposites gate: per-gun confidence faithful for 3 guns; no pair composes
Adds a per-sample intrinsic-confidence field (GunPrediction.confidence, threaded through FeedbackEvent/VirtualBullet, populated by Pattern, DecayGF, KNN, GuessFactor, Tsetlin, TMHorizon) and an offline recorder + analyzer that reproduce the paper's Figure 2 per gun and its Eq-8 composite. Measured on 3 held-out tr-bridge DrussGT battles (33k ticks, ~133k samples/gun): - FAITHFUL: DecayGF (rho +0.133), KNN (+0.090), Pattern (+0.064, weak). - GuessFactor is ANTI-faithful (rho -0.067); Tsetlin c_max is useless (0.001). - No pair of guns specialises complementarily: the same gun dominates both high-confidence slices in every pair. - Eq-8 alpha-normalised confidence-weighted composite: 18.41% vs Pattern 20.45% (McNemar p=3.1e-126). Faithful-only variant 18.68%, still loses. Shuffle control passes weakly (composite > shuffle, p=4e-14) so ~0.7pp of competence is real but ~2pp short. Offline veto: design is dead. See docs/tmcomposites_gate.md.
This commit is contained in:
@@ -35,6 +35,14 @@ type
|
|||||||
GunPrediction* = object
|
GunPrediction* = object
|
||||||
## Absolute (x, y) where the gun predicts the enemy will be.
|
## Absolute (x, y) where the gun predicts the enemy will be.
|
||||||
x*, y*: float
|
x*, y*: float
|
||||||
|
confidence*: float
|
||||||
|
## The gun's INTRINSIC, per-sample confidence in THIS prediction, on any
|
||||||
|
## positive scale the gun likes (the TMComposites gate normalises it by
|
||||||
|
## the sample range, Eq 7). 0.0 means "this gun has no confidence signal"
|
||||||
|
## (every deterministic geometric gun): such a gun casts no composite
|
||||||
|
## vote. Deliberately NOT a rolling hit-rate: that is the selector
|
||||||
|
## mechanism already measured negative (docs/gun_rack_analysis.md).
|
||||||
|
## See docs/tmcomposites_gate.md and common_libs/tests/measure_tmcomposites.nim.
|
||||||
|
|
||||||
FeedbackEvent* = object
|
FeedbackEvent* = object
|
||||||
## Outcome of a resolved virtual bullet.
|
## Outcome of a resolved virtual bullet.
|
||||||
@@ -45,6 +53,7 @@ type
|
|||||||
powerBin*: int ## index into PowerBins the bullet belongs to
|
powerBin*: int ## index into PowerBins the bullet belongs to
|
||||||
missDistance*: float ## px; < BotRadius = hit
|
missDistance*: float ## px; < BotRadius = hit
|
||||||
hit*: bool
|
hit*: bool
|
||||||
|
confidence*: float ## copied from the spawn-time GunPrediction (above)
|
||||||
|
|
||||||
proc bulletSpeed*(power: float): float {.inline.} =
|
proc bulletSpeed*(power: float): float {.inline.} =
|
||||||
20.0 - 3.0 * power
|
20.0 - 3.0 * power
|
||||||
|
|||||||
@@ -348,6 +348,7 @@ type
|
|||||||
bulletSpeed*: float
|
bulletSpeed*: float
|
||||||
travelDist*: float ## accumulated px so far
|
travelDist*: float ## accumulated px so far
|
||||||
fireDist*: float ## distance to target at fire time
|
fireDist*: float ## distance to target at fire time
|
||||||
|
confidence*: float ## spawn-time GunPrediction.confidence (TMComposites gate)
|
||||||
active*: bool
|
active*: bool
|
||||||
pointScored*: bool ## the parallel point-model score for this flight has
|
pointScored*: bool ## the parallel point-model score for this flight has
|
||||||
## been recorded (tie-break mode only)
|
## been recorded (tie-break mode only)
|
||||||
@@ -461,6 +462,7 @@ proc spawnBullets*(t: var VirtualTracker, gunId: GunId,
|
|||||||
bulletSpeed: speed,
|
bulletSpeed: speed,
|
||||||
travelDist: 0.0,
|
travelDist: 0.0,
|
||||||
fireDist: fireDist,
|
fireDist: fireDist,
|
||||||
|
confidence: pred.confidence,
|
||||||
active: true,
|
active: true,
|
||||||
pointScored: false,
|
pointScored: false,
|
||||||
hitSeen: false,
|
hitSeen: false,
|
||||||
@@ -643,7 +645,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState,
|
|||||||
t.fitness[b.targetId][b.gunId].bins[b.powerBin].record(hit)
|
t.fitness[b.targetId][b.gunId].bins[b.powerBin].record(hit)
|
||||||
|
|
||||||
let fe = FeedbackEvent(
|
let fe = FeedbackEvent(
|
||||||
prediction: GunPrediction(x: b.aimX, y: b.aimY),
|
prediction: GunPrediction(x: b.aimX, y: b.aimY, confidence: b.confidence),
|
||||||
actualX: ex,
|
actualX: ex,
|
||||||
actualY: ey,
|
actualY: ey,
|
||||||
bulletPower: PowerBins[b.powerBin],
|
bulletPower: PowerBins[b.powerBin],
|
||||||
@@ -651,6 +653,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState,
|
|||||||
powerBin: b.powerBin,
|
powerBin: b.powerBin,
|
||||||
missDistance: missDist,
|
missDistance: missDist,
|
||||||
hit: hit,
|
hit: hit,
|
||||||
|
confidence: b.confidence,
|
||||||
)
|
)
|
||||||
onResolved(b.gunId, b.powerBin, fe)
|
onResolved(b.gunId, b.powerBin, fe)
|
||||||
b.active = false
|
b.active = false
|
||||||
@@ -720,7 +723,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState,
|
|||||||
if b.targetId in t.fitness:
|
if b.targetId in t.fitness:
|
||||||
t.fitness[b.targetId][b.gunId].bins[b.powerBin].record(b.hitSeen)
|
t.fitness[b.targetId][b.gunId].bins[b.powerBin].record(b.hitSeen)
|
||||||
let fe = FeedbackEvent(
|
let fe = FeedbackEvent(
|
||||||
prediction: GunPrediction(x: b.aimX, y: b.aimY),
|
prediction: GunPrediction(x: b.aimX, y: b.aimY, confidence: b.confidence),
|
||||||
actualX: rx,
|
actualX: rx,
|
||||||
actualY: ry,
|
actualY: ry,
|
||||||
bulletPower: PowerBins[b.powerBin],
|
bulletPower: PowerBins[b.powerBin],
|
||||||
@@ -728,6 +731,7 @@ proc tickBullets*(t: var VirtualTracker, state: WorldState,
|
|||||||
powerBin: b.powerBin,
|
powerBin: b.powerBin,
|
||||||
missDistance: missDist,
|
missDistance: missDist,
|
||||||
hit: b.hitSeen,
|
hit: b.hitSeen,
|
||||||
|
confidence: b.confidence,
|
||||||
)
|
)
|
||||||
onResolved(b.gunId, b.powerBin, fe)
|
onResolved(b.gunId, b.powerBin, fe)
|
||||||
b.active = false
|
b.active = false
|
||||||
|
|||||||
@@ -113,6 +113,7 @@ proc predict*(g: var DecayGFGun, state: WorldState, bulletSpeed: float): GunPred
|
|||||||
GunPrediction(
|
GunPrediction(
|
||||||
x: clamp(px, BotRadius, state.arenaWidth - BotRadius),
|
x: clamp(px, BotRadius, state.arenaWidth - BotRadius),
|
||||||
y: clamp(py, BotRadius, state.arenaHeight - BotRadius),
|
y: clamp(py, BotRadius, state.arenaHeight - BotRadius),
|
||||||
|
confidence: g.bins[peak], # class-sum max over GF bins (see guess_factor.nim)
|
||||||
)
|
)
|
||||||
|
|
||||||
proc onResult*(g: var DecayGFGun, e: FeedbackEvent) =
|
proc onResult*(g: var DecayGFGun, e: FeedbackEvent) =
|
||||||
|
|||||||
@@ -141,6 +141,9 @@ proc predict*(g: var GFGun, state: WorldState, bulletSpeed: float): GunPredictio
|
|||||||
GunPrediction(
|
GunPrediction(
|
||||||
x: clamp(px, BotRadius, state.arenaWidth - BotRadius),
|
x: clamp(px, BotRadius, state.arenaWidth - BotRadius),
|
||||||
y: clamp(py, BotRadius, state.arenaHeight - BotRadius),
|
y: clamp(py, BotRadius, state.arenaHeight - BotRadius),
|
||||||
|
# TMComposites Eq 4 analogue: the GF histogram is the class distribution over
|
||||||
|
# GF bins, so the class-sum max c_max is the peak bin's accumulated weight.
|
||||||
|
confidence: g.bins[peak],
|
||||||
)
|
)
|
||||||
|
|
||||||
proc onResult*(g: var GFGun, e: FeedbackEvent) =
|
proc onResult*(g: var GFGun, e: FeedbackEvent) =
|
||||||
|
|||||||
@@ -299,6 +299,10 @@ proc predict*(g: var KNNGun, state: WorldState, bulletSpd: float): GunPrediction
|
|||||||
GunPrediction(
|
GunPrediction(
|
||||||
x: clamp(px, BotRadius, state.arenaWidth - BotRadius),
|
x: clamp(px, BotRadius, state.arenaWidth - BotRadius),
|
||||||
y: clamp(py, BotRadius, state.arenaHeight - BotRadius),
|
y: clamp(py, BotRadius, state.arenaHeight - BotRadius),
|
||||||
|
# TMComposites Eq 4 analogue: the KNN Gaussian density over GF candidates is
|
||||||
|
# the class distribution; bestScore is the class-sum max. Cold-start returns
|
||||||
|
# (no data / < 5 neighbours) leave the default 0.0 = no vote.
|
||||||
|
confidence: bestScore,
|
||||||
)
|
)
|
||||||
|
|
||||||
proc onResult*(g: var KNNGun, e: FeedbackEvent) =
|
proc onResult*(g: var KNNGun, e: FeedbackEvent) =
|
||||||
|
|||||||
@@ -50,6 +50,10 @@ type
|
|||||||
cacheValid: bool
|
cacheValid: bool
|
||||||
cacheTick: int
|
cacheTick: int
|
||||||
bestMatch: int ## -1 = no usable match (linear fallback)
|
bestMatch: int ## -1 = no usable match (linear fallback)
|
||||||
|
lastMatchScore*: float ## best pattern-match cost (lower = better);
|
||||||
|
## set by findBestMatch, exposed as the gun's
|
||||||
|
## intrinsic per-sample confidence
|
||||||
|
## (TMComposites gate, docs/tmcomposites_gate.md)
|
||||||
playStart: int
|
playStart: int
|
||||||
playAvail: int
|
playAvail: int
|
||||||
pathX: array[HistorySize + 1, float]
|
pathX: array[HistorySize + 1, float]
|
||||||
@@ -87,9 +91,11 @@ proc linearPredict(state: WorldState, bulletSpeed: float): (float, float) =
|
|||||||
|
|
||||||
# --- pattern search + play-forward ---
|
# --- pattern search + play-forward ---
|
||||||
|
|
||||||
proc findBestMatch(g: PatternMatcherGun): int =
|
proc findBestMatch(g: var PatternMatcherGun): int =
|
||||||
## Speed-independent history search. Returns the start index of the best
|
## Speed-independent history search. Returns the start index of the best
|
||||||
## matching pattern, or -1 when there is not enough history.
|
## matching pattern, or -1 when there is not enough history. Stores the best
|
||||||
|
## match cost in `g.lastMatchScore` for the confidence readout.
|
||||||
|
g.lastMatchScore = Inf
|
||||||
if g.count < PatternLen * 2:
|
if g.count < PatternLen * 2:
|
||||||
return -1
|
return -1
|
||||||
|
|
||||||
@@ -111,6 +117,7 @@ proc findBestMatch(g: PatternMatcherGun): int =
|
|||||||
if score < bestScore:
|
if score < bestScore:
|
||||||
bestScore = score
|
bestScore = score
|
||||||
bestMatch = i
|
bestMatch = i
|
||||||
|
g.lastMatchScore = bestScore
|
||||||
bestMatch
|
bestMatch
|
||||||
|
|
||||||
proc buildPath(g: var PatternMatcherGun, state: WorldState, bestMatch: int) =
|
proc buildPath(g: var PatternMatcherGun, state: WorldState, bestMatch: int) =
|
||||||
@@ -222,7 +229,10 @@ proc predict*(g: var PatternMatcherGun, state: WorldState,
|
|||||||
return g.applyRadial(state, px, py)
|
return g.applyRadial(state, px, py)
|
||||||
|
|
||||||
let (px, py) = g.projectFromPath(state, bulletSpeed)
|
let (px, py) = g.projectFromPath(state, bulletSpeed)
|
||||||
g.applyRadial(state, px, py)
|
result = g.applyRadial(state, px, py)
|
||||||
|
# Match quality as a confidence: a perfect historical match (cost 0) gives 1.0,
|
||||||
|
# a worse match decays toward 0. Deterministic and per-sample.
|
||||||
|
result.confidence = 1.0 / (1.0 + max(0.0, g.lastMatchScore))
|
||||||
|
|
||||||
proc onResult*(g: var PatternMatcherGun, e: FeedbackEvent) =
|
proc onResult*(g: var PatternMatcherGun, e: FeedbackEvent) =
|
||||||
discard # pattern matcher learns from movement observation, not feedback
|
discard # pattern matcher learns from movement observation, not feedback
|
||||||
|
|||||||
@@ -1091,13 +1091,16 @@ proc predict*(g: var TmHorizonGun, state: WorldState,
|
|||||||
let sgn = if side == 1: 1.0 else: -1.0
|
let sgn = if side == 1: 1.0 else: -1.0
|
||||||
shift = sgn * g.shiftDeg * magScale
|
shift = sgn * g.shiftDeg * magScale
|
||||||
|
|
||||||
let pred =
|
# TMComposites Eq 4 analogue: the side-machine's class-sum margin (normalised
|
||||||
if shift == 0.0: base
|
# by the clause half-count) is the per-sample confidence in WHICH WAY to shift.
|
||||||
else: tmhApplyShift(state.selfX, state.selfY, base.x, base.y, shift)
|
var predOut = base
|
||||||
|
if shift != 0.0:
|
||||||
|
predOut = tmhApplyShift(state.selfX, state.selfY, base.x, base.y, shift)
|
||||||
|
if warm: predOut.confidence = ev.sideConf
|
||||||
|
|
||||||
let aimDeg = radToDeg(arctan2(pred.y - state.selfY, pred.x - state.selfX))
|
let aimDeg = radToDeg(arctan2(predOut.y - state.selfY, predOut.x - state.selfX))
|
||||||
g.tmhLog(state, bulletSpeed, h, side, mag, ev.sideConf, shift, aimDeg)
|
g.tmhLog(state, bulletSpeed, h, side, mag, ev.sideConf, shift, aimDeg)
|
||||||
pred
|
predOut
|
||||||
|
|
||||||
proc onResult*(g: var TmHorizonGun, e: FeedbackEvent) =
|
proc onResult*(g: var TmHorizonGun, e: FeedbackEvent) =
|
||||||
## The label comes from our own observation ring, not from virtual-bullet
|
## The label comes from our own observation ring, not from virtual-bullet
|
||||||
|
|||||||
@@ -413,7 +413,14 @@ proc predict*(g: var TsetlinGun, state: WorldState, bulletSpeed: float): GunPred
|
|||||||
alive: true,
|
alive: true,
|
||||||
)
|
)
|
||||||
|
|
||||||
GunPrediction(x: predX, y: predY)
|
GunPrediction(
|
||||||
|
x: predX,
|
||||||
|
y: predY,
|
||||||
|
# TMComposites Eq 4: the two output teams' clamped clause sums are (vx, vy);
|
||||||
|
# their magnitude is how hard the machine is voting to move the correction.
|
||||||
|
# Warm-up fallback (window not full) leaves the default 0.0 = no vote.
|
||||||
|
confidence: hypot(vx, vy),
|
||||||
|
)
|
||||||
|
|
||||||
proc onResult*(g: var TsetlinGun, e: FeedbackEvent) =
|
proc onResult*(g: var TsetlinGun, e: FeedbackEvent) =
|
||||||
inc g.shotCount
|
inc g.shotCount
|
||||||
|
|||||||
@@ -0,0 +1,438 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""TMComposites GATE — analyse the per-sample confidence dump.
|
||||||
|
|
||||||
|
Reads the JSONL produced by ``common_libs/tests/measure_tmcomposites.nim``
|
||||||
|
(one row per resolved virtual bullet: fixture, split, gun, tick, bin, conf,
|
||||||
|
hit, relDeg, range, missPx) and answers the three questions of
|
||||||
|
``docs/tmcomposites_gate.md``:
|
||||||
|
|
||||||
|
A. Is each gun's intrinsic confidence FAITHFUL?
|
||||||
|
Rank samples by the gun's own confidence; report the accuracy-vs-confidence
|
||||||
|
curve, Spearman(confidence, hit), and top-half vs bottom-half accuracy with
|
||||||
|
a two-proportion p-value.
|
||||||
|
|
||||||
|
B. Are the faithful guns COMPLEMENTARY SPECIALISTS?
|
||||||
|
For each pair, on the samples where A's alpha-normalised confidence beats
|
||||||
|
B's, is A the more accurate one? Report the two slices and the win rates.
|
||||||
|
|
||||||
|
C. Does the Eq-8 alpha-normalised confidence-weighted composite beat the best
|
||||||
|
single gun on held-out battles, and is the gain attributable to competence?
|
||||||
|
The recorder already ran ``Composite`` and its within-sample
|
||||||
|
confidence-shuffle control ``CompositeShuf`` through the SAME virtual-bullet
|
||||||
|
geometry, so the comparison is paired (McNemar).
|
||||||
|
|
||||||
|
Pure stdlib (no numpy/scipy on this machine). MEASURED = every number printed;
|
||||||
|
INFERRED = the causal reading in the doc.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import collections
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import random
|
||||||
|
import sys
|
||||||
|
|
||||||
|
DETERMINISTIC = ["HeadOn", "Linear", "Circular", "WallBounce", "Accel",
|
||||||
|
"StopShot", "Displace", "AvgLead"]
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────────────────── stats (stdlib) ──────────────────────────────
|
||||||
|
def mean(xs):
|
||||||
|
return sum(xs) / len(xs) if xs else float("nan")
|
||||||
|
|
||||||
|
|
||||||
|
def spearman(xs, ys):
|
||||||
|
"""Spearman rho with average ranks for ties, and a normal-approx p."""
|
||||||
|
n = len(xs)
|
||||||
|
if n < 3:
|
||||||
|
return float("nan"), float("nan")
|
||||||
|
|
||||||
|
def ranks(v):
|
||||||
|
order = sorted(range(n), key=lambda i: v[i])
|
||||||
|
r = [0.0] * n
|
||||||
|
i = 0
|
||||||
|
while i < n:
|
||||||
|
j = i
|
||||||
|
while j + 1 < n and v[order[j + 1]] == v[order[i]]:
|
||||||
|
j += 1
|
||||||
|
avg = (i + j) / 2.0 + 1.0
|
||||||
|
for k in range(i, j + 1):
|
||||||
|
r[order[k]] = avg
|
||||||
|
i = j + 1
|
||||||
|
return r
|
||||||
|
|
||||||
|
rx, ry = ranks(xs), ranks(ys)
|
||||||
|
mx, my = mean(rx), mean(ry)
|
||||||
|
num = sum((a - mx) * (b - my) for a, b in zip(rx, ry))
|
||||||
|
den = math.sqrt(sum((a - mx) ** 2 for a in rx) * sum((b - my) ** 2 for b in ry))
|
||||||
|
if den == 0:
|
||||||
|
return 0.0, 1.0
|
||||||
|
rho = num / den
|
||||||
|
rho = max(-1.0, min(1.0, rho))
|
||||||
|
z = rho * math.sqrt(n - 1)
|
||||||
|
p = math.erfc(abs(z) / math.sqrt(2.0))
|
||||||
|
return rho, p
|
||||||
|
|
||||||
|
|
||||||
|
def norm_two_prop(z):
|
||||||
|
return math.erfc(abs(z) / math.sqrt(2.0))
|
||||||
|
|
||||||
|
|
||||||
|
def two_prop_p(h1, n1, h2, n2):
|
||||||
|
if n1 == 0 or n2 == 0:
|
||||||
|
return float("nan"), float("nan")
|
||||||
|
p1, p2 = h1 / n1, h2 / n2
|
||||||
|
p = (h1 + h2) / (n1 + n2)
|
||||||
|
se = math.sqrt(p * (1 - p) * (1 / n1 + 1 / n2))
|
||||||
|
if se == 0:
|
||||||
|
return float("nan"), float("nan")
|
||||||
|
z = (p1 - p2) / se
|
||||||
|
return z, norm_two_prop(z)
|
||||||
|
|
||||||
|
|
||||||
|
def binom_two_sided(k, n, p=0.5):
|
||||||
|
"""Exact two-sided binomial p (used for McNemar's discordant pairs)."""
|
||||||
|
if n == 0:
|
||||||
|
return 1.0
|
||||||
|
def pmf(i):
|
||||||
|
return math.comb(n, i) * p ** i * (1 - p) ** (n - i)
|
||||||
|
obs = pmf(k)
|
||||||
|
tot = 0.0
|
||||||
|
for i in range(n + 1):
|
||||||
|
if pmf(i) <= obs + 1e-12:
|
||||||
|
tot += pmf(i)
|
||||||
|
return min(1.0, tot)
|
||||||
|
|
||||||
|
|
||||||
|
def mcnemar_p(a_hit_b_miss, a_miss_b_hit):
|
||||||
|
"""McNemar: exact binomial for small discordant counts, normal approx for
|
||||||
|
large (the exact branch overflows math.comb for n in the hundred-thousands)."""
|
||||||
|
b, c = a_hit_b_miss, a_miss_b_hit
|
||||||
|
n = b + c
|
||||||
|
if n == 0:
|
||||||
|
return 1.0
|
||||||
|
if n < 500:
|
||||||
|
return binom_two_sided(b, n)
|
||||||
|
z = (b - c) / math.sqrt(n)
|
||||||
|
return math.erfc(abs(z) / math.sqrt(2.0))
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────────────────── data loading ────────────────────────────────
|
||||||
|
class Dump:
|
||||||
|
def __init__(self, path):
|
||||||
|
self.rows = [] # list of dicts
|
||||||
|
self.by_gun = collections.defaultdict(list)
|
||||||
|
# key -> {gun: (conf, hit)}
|
||||||
|
self.by_sample = collections.defaultdict(dict)
|
||||||
|
with open(path) as f:
|
||||||
|
for line in f:
|
||||||
|
line = line.strip()
|
||||||
|
if not line:
|
||||||
|
continue
|
||||||
|
o = json.loads(line)
|
||||||
|
self.rows.append(o)
|
||||||
|
self.by_gun[o["gun"]].append(o)
|
||||||
|
self.by_sample[(o["fixture"], o["tick"], o["bin"])][o["gun"]] = (
|
||||||
|
o["conf"], bool(o["hit"]))
|
||||||
|
|
||||||
|
def guns(self):
|
||||||
|
return sorted(self.by_gun.keys())
|
||||||
|
|
||||||
|
def split(self, gun, split):
|
||||||
|
return [r for r in self.by_gun[gun] if r["split"] == split]
|
||||||
|
|
||||||
|
|
||||||
|
def faithful_curve(rows, bins=10):
|
||||||
|
rows = sorted(rows, key=lambda r: r["conf"])
|
||||||
|
n = len(rows)
|
||||||
|
if n == 0:
|
||||||
|
return []
|
||||||
|
out = []
|
||||||
|
for b in range(bins):
|
||||||
|
lo = b * n // bins
|
||||||
|
hi = (b + 1) * n // bins
|
||||||
|
chunk = rows[lo:hi]
|
||||||
|
if not chunk:
|
||||||
|
continue
|
||||||
|
out.append(dict(
|
||||||
|
lo=chunk[0]["conf"], hi=chunk[-1]["conf"], n=len(chunk),
|
||||||
|
hit=mean([1.0 if r["hit"] else 0.0 for r in chunk]),
|
||||||
|
))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def faithfulness_report(dump):
|
||||||
|
"""Question A: per-gun faithfulness on the pooled test samples."""
|
||||||
|
out = {}
|
||||||
|
for gun in dump.guns():
|
||||||
|
rows = dump.split(gun, "test")
|
||||||
|
if not rows:
|
||||||
|
continue
|
||||||
|
confs = [r["conf"] for r in rows]
|
||||||
|
hits = [1.0 if r["hit"] else 0.0 for r in rows]
|
||||||
|
nz = sum(1 for c in confs if c > 1e-12)
|
||||||
|
rho, p = spearman(confs, hits)
|
||||||
|
order = sorted(range(len(rows)), key=lambda i: confs[i])
|
||||||
|
half = len(order) // 2
|
||||||
|
lo_idx, hi_idx = order[:half], order[half:]
|
||||||
|
h_lo = sum(hits[i] for i in lo_idx)
|
||||||
|
h_hi = sum(hits[i] for i in hi_idx)
|
||||||
|
z, phalf = two_prop_p(h_hi, len(hi_idx), h_lo, len(lo_idx))
|
||||||
|
out[gun] = dict(
|
||||||
|
n=len(rows), nonzero_conf=nz,
|
||||||
|
base=mean(hits), rho=rho, rho_p=p,
|
||||||
|
bottom_half_acc=(h_lo / len(lo_idx)) if lo_idx else float("nan"),
|
||||||
|
top_half_acc=(h_hi / len(hi_idx)) if hi_idx else float("nan"),
|
||||||
|
half_z=z, half_p=phalf,
|
||||||
|
curve=faithful_curve(rows),
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def alphas(dump):
|
||||||
|
"""alpha_t = max-min of the gun's confidence over the TRAIN samples (Eq 7)."""
|
||||||
|
out = {}
|
||||||
|
for gun in dump.guns():
|
||||||
|
rows = dump.split(gun, "train")
|
||||||
|
if not rows:
|
||||||
|
continue
|
||||||
|
cs = [r["conf"] for r in rows]
|
||||||
|
out[gun] = max(1e-12, max(cs) - min(min(cs), 0.0))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def normalised(conf, gun, alpha):
|
||||||
|
return conf / alpha.get(gun, 1.0)
|
||||||
|
|
||||||
|
|
||||||
|
def complementarity(dump, alphas_, faithful):
|
||||||
|
"""Question B: pairwise complementary slices over the test samples.
|
||||||
|
|
||||||
|
For each pair we split the samples by which gun has the higher
|
||||||
|
alpha-normalised confidence and, ON EACH SLICE, measure BOTH guns' accuracy.
|
||||||
|
A pair is complementary when each gun is the more accurate one on its own
|
||||||
|
winning slice, and each slice is a substantial (>=10%) share.
|
||||||
|
"""
|
||||||
|
tfs = test_fixtures(dump)
|
||||||
|
pairs = []
|
||||||
|
names = [g for g in faithful
|
||||||
|
if faithful[g]["nonzero_conf"] > 0
|
||||||
|
and not g.startswith("Composite")]
|
||||||
|
for i in range(len(names)):
|
||||||
|
for j in range(i + 1, len(names)):
|
||||||
|
A, B = names[i], names[j]
|
||||||
|
a_win = b_win = 0
|
||||||
|
# hits ON the A-winning slice, and ON the B-winning slice
|
||||||
|
aA = bA = aB = bB = 0
|
||||||
|
aA_only = bA_only = aB_only = bB_only = 0
|
||||||
|
for key, guns in dump.by_sample.items():
|
||||||
|
if key[0] not in tfs:
|
||||||
|
continue
|
||||||
|
if A not in guns or B not in guns:
|
||||||
|
continue
|
||||||
|
ca = normalised(guns[A][0], A, alphas_)
|
||||||
|
cb = normalised(guns[B][0], B, alphas_)
|
||||||
|
if ca <= 0 and cb <= 0:
|
||||||
|
continue
|
||||||
|
ha, hb = int(guns[A][1]), int(guns[B][1])
|
||||||
|
if ca > cb:
|
||||||
|
a_win += 1; aA += ha; bA += hb
|
||||||
|
if ha and not hb: aA_only += 1
|
||||||
|
elif hb and not ha: bA_only += 1
|
||||||
|
elif cb > ca:
|
||||||
|
b_win += 1; aB += ha; bB += hb
|
||||||
|
if ha and not hb: aB_only += 1
|
||||||
|
elif hb and not ha: bB_only += 1
|
||||||
|
both = a_win + b_win
|
||||||
|
def frac(h, n):
|
||||||
|
return h / n if n else float("nan")
|
||||||
|
accA_on_A, accB_on_A = frac(aA, a_win), frac(bA, a_win)
|
||||||
|
accA_on_B, accB_on_B = frac(aB, b_win), frac(bB, b_win)
|
||||||
|
comp = (both > 0 and a_win >= 0.10 * both and b_win >= 0.10 * both
|
||||||
|
and accA_on_A > accB_on_A and accB_on_B > accA_on_B)
|
||||||
|
pairs.append(dict(
|
||||||
|
A=A, B=B, a_win=a_win, b_win=b_win, both=both,
|
||||||
|
A_acc_on_A_slice=accA_on_A, B_acc_on_A_slice=accB_on_A,
|
||||||
|
A_acc_on_B_slice=accA_on_B, B_acc_on_B_slice=accB_on_B,
|
||||||
|
p_on_A_slice=mcnemar_p(aA_only, bA_only),
|
||||||
|
p_on_B_slice=mcnemar_p(bB_only, aB_only),
|
||||||
|
complementary=comp))
|
||||||
|
return pairs
|
||||||
|
|
||||||
|
|
||||||
|
def compl_ok(a_win, b_win, both, aa, ba):
|
||||||
|
"""Retained for backwards compatibility; superseded by the per-slice test
|
||||||
|
in `complementarity`."""
|
||||||
|
if both == 0 or a_win == 0 or b_win == 0:
|
||||||
|
return False
|
||||||
|
return a_win >= 0.10 * both and b_win >= 0.10 * both
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────────────── composite comparison ────────────────────────────
|
||||||
|
def paired(dump, gun_a, gun_b, tfs):
|
||||||
|
"""Return (a_hit_b_miss, a_miss_b_hit, a_hits, b_hits) over the test fixtures."""
|
||||||
|
ab = ba = na = nb = 0
|
||||||
|
for key, guns in dump.by_sample.items():
|
||||||
|
if key[0] not in tfs:
|
||||||
|
continue
|
||||||
|
if gun_a in guns and gun_b in guns:
|
||||||
|
a = guns[gun_a][1]
|
||||||
|
b = guns[gun_b][1]
|
||||||
|
if a and not b:
|
||||||
|
ab += 1
|
||||||
|
elif b and not a:
|
||||||
|
ba += 1
|
||||||
|
if a:
|
||||||
|
na += 1
|
||||||
|
if b:
|
||||||
|
nb += 1
|
||||||
|
return ab, ba, na, nb
|
||||||
|
|
||||||
|
|
||||||
|
def test_fixtures(dump):
|
||||||
|
out = set()
|
||||||
|
for r in dump.rows:
|
||||||
|
if r["split"] == "test":
|
||||||
|
out.add(r["fixture"])
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def composite_report(dump):
|
||||||
|
"""Question C: composite vs best single vs shuffle control on test."""
|
||||||
|
tfs = test_fixtures(dump)
|
||||||
|
acc = {}
|
||||||
|
n = {}
|
||||||
|
for gun in dump.guns():
|
||||||
|
rows = [r for r in dump.by_gun[gun] if r["split"] == "test"]
|
||||||
|
if not rows:
|
||||||
|
continue
|
||||||
|
acc[gun] = mean([1.0 if r["hit"] else 0.0 for r in rows])
|
||||||
|
n[gun] = len(rows)
|
||||||
|
members = [g for g in dump.guns()
|
||||||
|
if not g.startswith("Composite") and g not in DETERMINISTIC]
|
||||||
|
best_member = max(members, key=lambda g: acc[g]) if members else None
|
||||||
|
res = dict(acc=acc, n=n, best_member=best_member)
|
||||||
|
# Oracle ceiling: if a perfect per-sample selector could pick ANY member,
|
||||||
|
# how often would it hit? This bounds what a member-selection composite
|
||||||
|
# could ever reach (the vote can do worse but not better than this).
|
||||||
|
oracle = 0
|
||||||
|
oracle_n = 0
|
||||||
|
for key, guns in dump.by_sample.items():
|
||||||
|
if key[0] not in tfs:
|
||||||
|
continue
|
||||||
|
hits = [guns[m][1] for m in members if m in guns]
|
||||||
|
if hits:
|
||||||
|
oracle += int(any(hits))
|
||||||
|
oracle_n += 1
|
||||||
|
res["member_oracle"] = (oracle / oracle_n) if oracle_n else float("nan")
|
||||||
|
res["member_oracle_n"] = oracle_n
|
||||||
|
comps = [g for g in dump.guns() if g.startswith("Composite") and not g.endswith("Shuf")]
|
||||||
|
res["composites"] = {}
|
||||||
|
for comp in comps:
|
||||||
|
entry = {}
|
||||||
|
if best_member:
|
||||||
|
ab, ba, na, nb = paired(dump, comp, best_member, tfs)
|
||||||
|
entry["vs_best"] = dict(best=best_member, comp_hit=na, best_hit=nb,
|
||||||
|
total=n[comp], mcnemar_ab=ab, mcnemar_ba=ba,
|
||||||
|
p=mcnemar_p(ab, ba))
|
||||||
|
if "Pattern" in acc:
|
||||||
|
ab, ba, na, nb = paired(dump, comp, "Pattern", tfs)
|
||||||
|
entry["vs_pattern"] = dict(comp_hit=na, pattern_hit=nb, total=n[comp],
|
||||||
|
mcnemar_ab=ab, mcnemar_ba=ba, p=mcnemar_p(ab, ba))
|
||||||
|
shuf = comp + "Shuf"
|
||||||
|
if shuf in acc:
|
||||||
|
ab, ba, na, nb = paired(dump, comp, shuf, tfs)
|
||||||
|
entry["vs_shuffle"] = dict(comp_hit=na, shuffle_hit=nb, total=n[comp],
|
||||||
|
mcnemar_ab=ab, mcnemar_ba=ba, p=mcnemar_p(ab, ba))
|
||||||
|
res["composites"][comp] = entry
|
||||||
|
# per-fixture composite vs best single vs shuffle
|
||||||
|
per = {}
|
||||||
|
for fx in sorted(tfs):
|
||||||
|
row = {}
|
||||||
|
for gun in ([best_member] if best_member else []) + comps:
|
||||||
|
rows = [r for r in dump.by_gun[gun] if r["split"] == "test" and r["fixture"] == fx]
|
||||||
|
if rows:
|
||||||
|
row[gun] = dict(n=len(rows), acc=mean([1.0 if r["hit"] else 0.0 for r in rows]))
|
||||||
|
per[fx] = row
|
||||||
|
res["per_fixture"] = per
|
||||||
|
return res
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────────────────────── main ────────────────────────────────────
|
||||||
|
def main():
|
||||||
|
ap = argparse.ArgumentParser()
|
||||||
|
ap.add_argument("--input", default="/tmp/tmc_full.jsonl")
|
||||||
|
ap.add_argument("--json", default=None)
|
||||||
|
args = ap.parse_args()
|
||||||
|
|
||||||
|
dump = Dump(args.input)
|
||||||
|
print(f"rows={len(dump.rows)} guns={len(dump.guns())} "
|
||||||
|
f"test fixtures={sorted(test_fixtures(dump))}")
|
||||||
|
|
||||||
|
a = faithfulness_report(dump)
|
||||||
|
print("\n=== A. FAITHFULNESS (test samples; rank by own confidence) ===")
|
||||||
|
print(f"{'gun':<14}{'n':>7}{'nonzero':>8}{'base%':>7}{'rho':>8}{'rho_p':>9}"
|
||||||
|
f"{'bot%':>7}{'top%':>7}{'z':>7}{'p':>9} verdict")
|
||||||
|
verdicts = {}
|
||||||
|
for gun in sorted(a, key=lambda g: -a[g]["base"]):
|
||||||
|
r = a[gun]
|
||||||
|
if r["nonzero_conf"] == 0:
|
||||||
|
v = "NO SIGNAL"
|
||||||
|
elif r["rho_p"] < 0.01 and r["rho"] > 0.05:
|
||||||
|
v = "FAITHFUL"
|
||||||
|
elif r["rho_p"] < 0.01 and r["rho"] < -0.05:
|
||||||
|
v = "ANTI-FAITHFUL"
|
||||||
|
else:
|
||||||
|
v = "USELESS"
|
||||||
|
verdicts[gun] = v
|
||||||
|
print(f"{gun:<14}{r['n']:>7}{r['nonzero_conf']:>8}{100*r['base']:>7.2f}"
|
||||||
|
f"{r['rho']:>8.3f}{r['rho_p']:>9.2g}{100*r['bottom_half_acc']:>7.2f}"
|
||||||
|
f"{100*r['top_half_acc']:>7.2f}{r['half_z']:>7.2f}{r['half_p']:>9.2g} {v}")
|
||||||
|
|
||||||
|
print("\ncurves (deciles, low->high confidence):")
|
||||||
|
for gun in sorted(a, key=lambda g: -a[g]["base"]):
|
||||||
|
if verdicts[gun] == "NO SIGNAL":
|
||||||
|
continue
|
||||||
|
cur = " ".join(f"{100*c['hit']:.0f}%" for c in a[gun]["curve"])
|
||||||
|
print(f" {gun:<14} {cur}")
|
||||||
|
|
||||||
|
al = alphas(dump)
|
||||||
|
pairs = complementarity(dump, al, a)
|
||||||
|
print("\n=== B. PAIRWISE COMPLEMENTARITY (test; alpha-normalised confidence) ===")
|
||||||
|
print(f"{'A':<14}{'B':<14}{'A_wins':>8}{'A|A':>7}{'B|A':>7}{'p_A':>9}| {'B_wins':>8}{'A|B':>7}{'B|B':>7}{'p_B':>9} comp")
|
||||||
|
for p in pairs:
|
||||||
|
print(f"{p['A']:<14}{p['B']:<14}{p['a_win']:>8}"
|
||||||
|
f"{100*p['A_acc_on_A_slice']:>7.1f}{100*p['B_acc_on_A_slice']:>7.1f}{p['p_on_A_slice']:>9.2g}| "
|
||||||
|
f"{p['b_win']:>8}{100*p['A_acc_on_B_slice']:>7.1f}"
|
||||||
|
f"{100*p['B_acc_on_B_slice']:>7.1f}{p['p_on_B_slice']:>9.2g} {p['complementary']}")
|
||||||
|
print(" (A|A = A's accuracy on the slice A wins; B|A = B's accuracy on that same slice; etc.)")
|
||||||
|
|
||||||
|
c = composite_report(dump)
|
||||||
|
print("\n=== C. COMPOSITE vs BEST SINGLE vs SHUFFLE CONTROL (test) ===")
|
||||||
|
for gun in sorted(c["acc"], key=lambda g: -c["acc"][g]):
|
||||||
|
print(f" {gun:<14} {100*c['acc'][gun]:>6.2f}% n={c['n'][gun]}")
|
||||||
|
if "member_oracle" in c:
|
||||||
|
print(f" member_oracle (any member hits, per sample) : {100*c['member_oracle']:.2f}%")
|
||||||
|
for comp, entry in c.get("composites", {}).items():
|
||||||
|
print(f" {comp}:")
|
||||||
|
for k, r in entry.items():
|
||||||
|
print(f" {k}: {r}")
|
||||||
|
print("\nper-fixture:")
|
||||||
|
for fx, row in c["per_fixture"].items():
|
||||||
|
parts = " ".join(f"{g}={100*v['acc']:.1f}%({v['n']})" for g, v in row.items())
|
||||||
|
print(f" {fx:<32} {parts}")
|
||||||
|
|
||||||
|
if args.json:
|
||||||
|
blob = dict(
|
||||||
|
faithfulness=a, alphas=al, verdicts=verdicts,
|
||||||
|
complementarity=pairs, composite=c,
|
||||||
|
test_fixtures=sorted(test_fixtures(dump)),
|
||||||
|
)
|
||||||
|
with open(args.json, "w") as f:
|
||||||
|
json.dump(blob, f, indent=2)
|
||||||
|
print(f"\n[json] wrote {args.json}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
+1573
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,285 @@
|
|||||||
|
## TMComposites GATE — offline per-sample confidence + outcome recorder.
|
||||||
|
##
|
||||||
|
## Implements the measurement half of docs/tmcomposites_gate.md. It replays
|
||||||
|
## recorded fixtures through the SHIPPED rack exactly as `offline_range` does
|
||||||
|
## (same `VirtualTracker`, same `bmPath` virtual-bullet ground truth), but also
|
||||||
|
## captures, for every resolved virtual bullet, the gun's INTRINSIC per-sample
|
||||||
|
## `GunPrediction.confidence` (see gun_interface.nim) and the bullet's aim
|
||||||
|
## bearing relative to the fire-time line of sight.
|
||||||
|
##
|
||||||
|
## It also runs two COMPOSITE arms through the SAME tracker, so their hits are
|
||||||
|
## scored by the identical geometry as every member:
|
||||||
|
## * Composite — TMComposites Eq 8: each confident gun casts its
|
||||||
|
## alpha-normalised confidence into the angular bin of its
|
||||||
|
## own aim; the argmax bin wins.
|
||||||
|
## * CompositeShuf — the MANDATORY control: the same confidences are randomly
|
||||||
|
## permuted among the members within each sample, so the
|
||||||
|
## weighting distribution is preserved but competence is
|
||||||
|
## destroyed.
|
||||||
|
##
|
||||||
|
## alpha (Eq 7) is calibrated on the TRAIN fixtures (per member: max-min of its
|
||||||
|
## confidence over train) and then frozen for the TEST fixtures, so the test
|
||||||
|
## composite never sees test labels while choosing its weights.
|
||||||
|
##
|
||||||
|
## Output: one JSONL row per resolved bullet, to --out. Aggregation lives in
|
||||||
|
## common_libs/tests/analyze_tmcomposites.py.
|
||||||
|
##
|
||||||
|
## Usage:
|
||||||
|
## nim c -r --nimcache:/tmp/nc_j104 common_libs/tests/measure_tmcomposites.nim \
|
||||||
|
## --out /tmp/tmc.jsonl \
|
||||||
|
## --train fx_a.jsonl fx_b.jsonl --test fx_c.jsonl
|
||||||
|
|
||||||
|
import std/[os, strformat, json, math, strutils, random, algorithm, tables, times]
|
||||||
|
import gun_harness/offline_range
|
||||||
|
import gun_harness/gun_interface
|
||||||
|
import gun_harness/virtual_bullets
|
||||||
|
import range_guns
|
||||||
|
|
||||||
|
const
|
||||||
|
BinDeg = 0.5
|
||||||
|
HalfSpanDeg = 45.0
|
||||||
|
NumBins = int(2.0 * HalfSpanDeg / BinDeg)
|
||||||
|
## guns with a genuine intrinsic confidence signal (deterministic geometric
|
||||||
|
## guns leave GunPrediction.confidence at 0.0 and cast no composite vote).
|
||||||
|
ConfidentGuns = ["Tsetlin", "GuessFactor", "Pattern", "DecayGF", "KNN"]
|
||||||
|
|
||||||
|
type
|
||||||
|
Spawn = object
|
||||||
|
relDeg: float
|
||||||
|
range: float
|
||||||
|
conf: float
|
||||||
|
|
||||||
|
proc wrapDeg(d: float): float =
|
||||||
|
result = d
|
||||||
|
while result > 180.0: result -= 360.0
|
||||||
|
while result < -180.0: result += 360.0
|
||||||
|
|
||||||
|
proc binOf(relDeg: float): int =
|
||||||
|
result = int(floor((relDeg + HalfSpanDeg) / BinDeg))
|
||||||
|
if result < 0: result = 0
|
||||||
|
elif result >= NumBins: result = NumBins - 1
|
||||||
|
|
||||||
|
var alphaByName: ref Table[string, float]
|
||||||
|
|
||||||
|
proc alphaFor(name: string): float =
|
||||||
|
## Eq 7 alpha_t for a member, keyed by GUN NAME so each composite's member
|
||||||
|
## order cannot mis-assign a scale. 1.0 until calibration writes the table.
|
||||||
|
if alphaByName.isNil: return 1.0
|
||||||
|
alphaByName[].getOrDefault(name, 1.0)
|
||||||
|
|
||||||
|
proc makeComposite(name: string, members: seq[GunDriver],
|
||||||
|
shuffle: bool): GunDriver =
|
||||||
|
## One composite arm. `members` are the SAME driver closures the tracker uses
|
||||||
|
## for the member guns, so the composite reads their live state (calling
|
||||||
|
## predict twice in a tick is idempotent for every gun: GF/KNN guard their
|
||||||
|
## wave queue on (tick,bin), Pattern/Tsetlin/TMHorizon cache per tick).
|
||||||
|
result.name = name
|
||||||
|
let memberList = members
|
||||||
|
result.predictCb = proc(state: WorldState, speed: float): GunPrediction =
|
||||||
|
let n = memberList.len
|
||||||
|
var rels = newSeq[float](n)
|
||||||
|
var dists = newSeq[float](n)
|
||||||
|
var confs = newSeq[float](n)
|
||||||
|
let los = arctan2(state.enemyY - state.selfY, state.enemyX - state.selfX)
|
||||||
|
for i in 0..<n:
|
||||||
|
let p = memberList[i].predictCb(state, speed)
|
||||||
|
rels[i] = wrapDeg(radToDeg(arctan2(p.y - state.selfY, p.x - state.selfX) - los))
|
||||||
|
dists[i] = hypot(p.x - state.selfX, p.y - state.selfY)
|
||||||
|
confs[i] = p.confidence
|
||||||
|
if shuffle:
|
||||||
|
for i in countdown(n - 1, 1):
|
||||||
|
let j = rand(i)
|
||||||
|
swap(confs[i], confs[j])
|
||||||
|
|
||||||
|
var votes = newSeq[float](NumBins)
|
||||||
|
var total = 0.0
|
||||||
|
for i in 0..<n:
|
||||||
|
let a = max(1e-12, alphaFor(memberList[i].name))
|
||||||
|
let w = confs[i] / a
|
||||||
|
if w <= 0.0: continue
|
||||||
|
votes[binOf(rels[i])] += w
|
||||||
|
total += w
|
||||||
|
if total <= 0.0:
|
||||||
|
# cold: no member has any confidence yet. Make no claim.
|
||||||
|
return GunPrediction(x: state.enemyX, y: state.enemyY, confidence: 0.0)
|
||||||
|
|
||||||
|
var best = 0
|
||||||
|
for b in 1..<NumBins:
|
||||||
|
if votes[b] > votes[best]: best = b
|
||||||
|
|
||||||
|
# aim distance = confidence-weighted mean distance of the winning bin's voters
|
||||||
|
var dsum = 0.0
|
||||||
|
var wsum = 0.0
|
||||||
|
for i in 0..<n:
|
||||||
|
if binOf(rels[i]) != best: continue
|
||||||
|
let a = max(1e-12, alphaFor(memberList[i].name))
|
||||||
|
let w = confs[i] / a
|
||||||
|
dsum += dists[i] * w
|
||||||
|
wsum += w
|
||||||
|
let d = if wsum > 1e-12: dsum / wsum else: hypot(state.enemyX - state.selfX,
|
||||||
|
state.enemyY - state.selfY)
|
||||||
|
let ang = los + degToRad(-HalfSpanDeg + (best.float + 0.5) * BinDeg)
|
||||||
|
GunPrediction(x: state.selfX + cos(ang) * d,
|
||||||
|
y: state.selfY + sin(ang) * d,
|
||||||
|
confidence: votes[best])
|
||||||
|
result.resultCb = proc(e: FeedbackEvent) = discard
|
||||||
|
result.readyCb = nil
|
||||||
|
|
||||||
|
proc buildRack(): tuple[drivers: seq[GunDriver], memberLocal: seq[int]] =
|
||||||
|
## Fresh members + the composite arms, mirroring the standard per-fixture
|
||||||
|
## replay (guns start cold for every fixture, exactly as run_range does).
|
||||||
|
##
|
||||||
|
## Composite — all 5 confidence guns (Tsetlin, GF, Pattern, DecayGF, KNN)
|
||||||
|
## CompositeShuf — its within-sample confidence shuffle control
|
||||||
|
## CompositeF — only the guns the faithfulness test finds FAITHFUL
|
||||||
|
## (Pattern, DecayGF, KNN); GF is anti-faithful and Tsetlin
|
||||||
|
## useless, so this is the strongest reasonable variant
|
||||||
|
## CompositeFShuf — its shuffle control
|
||||||
|
let allDrivers = buildAllGunDrivers(seed = 1)
|
||||||
|
var members: seq[GunDriver]
|
||||||
|
var memberLocal: seq[int]
|
||||||
|
var faithMembers: seq[GunDriver]
|
||||||
|
for i, d in allDrivers:
|
||||||
|
if d.name in ConfidentGuns:
|
||||||
|
members.add d
|
||||||
|
memberLocal.add i
|
||||||
|
if d.name in ["Pattern", "DecayGF", "KNN"]:
|
||||||
|
faithMembers.add d
|
||||||
|
doAssert members.len == ConfidentGuns.len
|
||||||
|
result.drivers = allDrivers
|
||||||
|
result.drivers.add makeComposite("Composite", members, shuffle = false)
|
||||||
|
result.drivers.add makeComposite("CompositeShuf", members, shuffle = true)
|
||||||
|
result.drivers.add makeComposite("CompositeF", faithMembers, shuffle = false)
|
||||||
|
result.drivers.add makeComposite("CompositeFShuf", faithMembers, shuffle = true)
|
||||||
|
result.memberLocal = memberLocal
|
||||||
|
|
||||||
|
proc runFixture(fx: Fixture, fixtureName, split: string,
|
||||||
|
calMin, calMax: ref seq[float], outFile: File) =
|
||||||
|
let (drivers, memberLocal) = buildRack()
|
||||||
|
var byTick = initTable[int, WorldState]()
|
||||||
|
for s in fx.states: byTick[s.tick] = s
|
||||||
|
var spawns = initTable[tuple[gunId, tick, bin: int], Spawn]()
|
||||||
|
var tracker = initTracker(drivers.len, ActiveMetric)
|
||||||
|
|
||||||
|
for si in 0..<fx.states.len:
|
||||||
|
let state = fx.states[si]
|
||||||
|
for gi in 0..<drivers.len:
|
||||||
|
var preds: array[len(PowerBins), GunPrediction]
|
||||||
|
var sps: array[len(PowerBins), Spawn]
|
||||||
|
let los = arctan2(state.enemyY - state.selfY, state.enemyX - state.selfX)
|
||||||
|
for i in 0..<len(PowerBins):
|
||||||
|
preds[i] = drivers[gi].predictCb(state, bulletSpeed(PowerBins[i]))
|
||||||
|
sps[i] = Spawn(
|
||||||
|
relDeg: wrapDeg(radToDeg(arctan2(preds[i].y - state.selfY,
|
||||||
|
preds[i].x - state.selfX) - los)),
|
||||||
|
range: hypot(preds[i].x - state.selfX, preds[i].y - state.selfY),
|
||||||
|
conf: preds[i].confidence)
|
||||||
|
let ready = if drivers[gi].readyCb == nil: true else: drivers[gi].readyCb()
|
||||||
|
if ready:
|
||||||
|
for i in 0..<len(PowerBins):
|
||||||
|
spawns[(gi, state.tick, i)] = sps[i]
|
||||||
|
tracker.spawnBullets(gi, preds, state, fx.enemyId)
|
||||||
|
|
||||||
|
var enemyPositions: Table[int, tuple[x, y: float, lastSeenTick: int, alive: bool]]
|
||||||
|
if state.enemies.len > 0:
|
||||||
|
for e in state.enemies:
|
||||||
|
enemyPositions[e.id] = (x: e.x, y: e.y, lastSeenTick: e.lastSeenTick, alive: true)
|
||||||
|
else:
|
||||||
|
enemyPositions[fx.enemyId] = (x: state.enemyX, y: state.enemyY,
|
||||||
|
lastSeenTick: state.tick, alive: true)
|
||||||
|
|
||||||
|
tracker.tickBullets(state, enemyPositions,
|
||||||
|
proc(gunId: int, binIdx: int, e: FeedbackEvent) =
|
||||||
|
# The gun must LEARN from its own resolved bullet (exactly as
|
||||||
|
# offline_range.replayFixture does); the recorder is an extra hook.
|
||||||
|
drivers[gunId].resultCb(e)
|
||||||
|
let key = (gunId, e.fireTick, binIdx)
|
||||||
|
if not spawns.hasKey(key): return
|
||||||
|
let sp = spawns[key]
|
||||||
|
spawns.del(key)
|
||||||
|
if split == "train":
|
||||||
|
for mi, gid in memberLocal:
|
||||||
|
if gunId == gid:
|
||||||
|
calMin[][mi] = min(calMin[][mi], sp.conf)
|
||||||
|
calMax[][mi] = max(calMax[][mi], sp.conf)
|
||||||
|
outFile.writeLine($(%*{
|
||||||
|
"fixture": fixtureName,
|
||||||
|
"split": split,
|
||||||
|
"gun": drivers[gunId].name,
|
||||||
|
"tick": e.fireTick,
|
||||||
|
"bin": binIdx,
|
||||||
|
"conf": sp.conf,
|
||||||
|
"hit": e.hit,
|
||||||
|
"relDeg": sp.relDeg,
|
||||||
|
"range": sp.range,
|
||||||
|
"missPx": e.missDistance,
|
||||||
|
}))
|
||||||
|
)
|
||||||
|
|
||||||
|
proc loadAll(paths: seq[string]): seq[tuple[name: string, fx: Fixture]] =
|
||||||
|
for p in paths:
|
||||||
|
let fx = loadFixture(p)
|
||||||
|
let name = extractFilename(p).replace(".jsonl", "")
|
||||||
|
result.add (name: name, fx: fx)
|
||||||
|
|
||||||
|
proc main() =
|
||||||
|
var outPath = "/tmp/tmc.jsonl"
|
||||||
|
var trainPaths, testPaths: seq[string]
|
||||||
|
var args: seq[string]
|
||||||
|
for i in 1..paramCount(): args.add paramStr(i)
|
||||||
|
var mode = ""
|
||||||
|
for a in args:
|
||||||
|
if a == "--out": mode = "out"; continue
|
||||||
|
if a == "--train": mode = "train"; continue
|
||||||
|
if a == "--test": mode = "test"; continue
|
||||||
|
case mode
|
||||||
|
of "out": outPath = a
|
||||||
|
of "train": trainPaths.add a
|
||||||
|
of "test": testPaths.add a
|
||||||
|
else: discard
|
||||||
|
|
||||||
|
var calMinRef = new(seq[float])
|
||||||
|
var calMaxRef = new(seq[float])
|
||||||
|
calMinRef[] = newSeq[float](ConfidentGuns.len)
|
||||||
|
calMaxRef[] = newSeq[float](ConfidentGuns.len)
|
||||||
|
for i in 0..<ConfidentGuns.len:
|
||||||
|
calMinRef[][i] = 1e18
|
||||||
|
calMaxRef[][i] = -1e18
|
||||||
|
|
||||||
|
let trainFx = loadAll(trainPaths)
|
||||||
|
let testFx = loadAll(testPaths)
|
||||||
|
let outFile = open(outPath, fmWrite)
|
||||||
|
defer: outFile.close()
|
||||||
|
|
||||||
|
randomize(20250925)
|
||||||
|
|
||||||
|
echo fmt"# TMComposites recorder: {trainFx.len} train fixtures, {testFx.len} test fixtures"
|
||||||
|
echo fmt"# members: {ConfidentGuns}"
|
||||||
|
|
||||||
|
var t0 = epochTime()
|
||||||
|
for (name, fx) in trainFx:
|
||||||
|
let a = epochTime()
|
||||||
|
runFixture(fx, name, "train", calMinRef, calMaxRef, outFile)
|
||||||
|
echo fmt" [train] {name:<28} ticks={fx.states.len:>6} {epochTime()-a:6.1f}s"
|
||||||
|
|
||||||
|
# Eq 7 alpha_t = max-min over the training input set; frozen for test.
|
||||||
|
alphaByName = new(Table[string, float])
|
||||||
|
for i in 0..<ConfidentGuns.len:
|
||||||
|
alphaByName[][ConfidentGuns[i]] =
|
||||||
|
max(1e-9, calMaxRef[][i] - min(calMinRef[][i], 0.0))
|
||||||
|
echo "# alpha_t (Eq 7, from train): " &
|
||||||
|
(block:
|
||||||
|
var s = ""
|
||||||
|
for i in 0..<ConfidentGuns.len:
|
||||||
|
s.add fmt"{ConfidentGuns[i]}={alphaByName[][ConfidentGuns[i]]:.4g} "
|
||||||
|
s)
|
||||||
|
|
||||||
|
for (name, fx) in testFx:
|
||||||
|
let a = epochTime()
|
||||||
|
runFixture(fx, name, "test", calMinRef, calMaxRef, outFile)
|
||||||
|
echo fmt" [test ] {name:<28} ticks={fx.states.len:>6} {epochTime()-a:6.1f}s"
|
||||||
|
|
||||||
|
echo fmt"# done in {epochTime()-t0:.1f}s -> {outPath}"
|
||||||
|
|
||||||
|
when isMainModule:
|
||||||
|
main()
|
||||||
@@ -0,0 +1,318 @@
|
|||||||
|
# TMComposites gate: is each gun's intrinsic confidence faithful, and are the guns complementary specialists?
|
||||||
|
|
||||||
|
**Scope.** Test the mechanism of *TMComposites: Plug-and-Play Collaboration
|
||||||
|
Between Specialized Tsetlin Machines* (Granmo, arXiv:2309.04801v2, §3) on our
|
||||||
|
gun rack: (A) is each gun's per-sample confidence faithful to its own accuracy?
|
||||||
|
(B) are the guns complementary specialists? (C) does an Eq-8 alpha-normalised
|
||||||
|
confidence-weighted composite beat the best single gun offline, and is any gain
|
||||||
|
attributable to competence (shuffle control)? This is the one untested idea that
|
||||||
|
could plausibly beat `onlyPattern`, whose selector was measured **negative value**
|
||||||
|
(`docs/selector_negative_value.md`, `docs/gun_rack_analysis.md`, commit `e0666a5`)
|
||||||
|
because it decides with a rolling hit-rate instead of a per-sample signal.
|
||||||
|
|
||||||
|
**Evidence tags.** `[MEASURED]` = read from the committed result
|
||||||
|
`common_libs/tests/fixtures/tmcomposites_gate.json`, reproduced by the named
|
||||||
|
command, or a source fact. `[INFERRED]` = reasoning from those measurements.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 0. Direct answers
|
||||||
|
|
||||||
|
1. **Which guns are faithfully confident?** `[MEASURED]`
|
||||||
|
**Pattern (weak), DecayGF (strong), KNN (moderate)** are faithful: rank samples
|
||||||
|
by the gun's own confidence and accuracy rises. **GuessFactor is
|
||||||
|
ANTI-faithful** — it is more accurate when it is *less* confident. **Tsetlin's
|
||||||
|
class-sum max is useless** (Spearman 0.001, p=0.68). The eight deterministic
|
||||||
|
geometric guns (HeadOn, Linear, Circular, WallBounce, Accel, StopShot,
|
||||||
|
Displace, AvgLead) expose **no per-sample confidence at all**.
|
||||||
|
|
||||||
|
2. **Do any pairs specialise complementarily?** `[MEASURED]` **No.** In every one
|
||||||
|
of the 10 pairs, on the slice where A's normalised confidence beats B's, A is
|
||||||
|
*not* the more accurate gun — or the same gun dominates both slices (e.g.
|
||||||
|
GuessFactor is more accurate than DecayGF and than KNN even on their own
|
||||||
|
high-confidence slices). The paper's premise — one member's weakness is
|
||||||
|
another's strength, decided by confidence — does not hold on our rack. Our
|
||||||
|
guns are different *estimators of the same target*, not different *specialists*.
|
||||||
|
|
||||||
|
3. **Does the composite beat the best single gun?** `[MEASURED]` **No, and the
|
||||||
|
design is dead offline.** The Eq-8 composite scores **18.41%** vs Pattern's
|
||||||
|
**20.45%** (McNemar p=3.1e-126); the faithful-only variant (Pattern + DecayGF +
|
||||||
|
KNN) scores **18.68%**, still losing to Pattern by 1.77pp (p=1.25e-133). Both
|
||||||
|
lose on all three held-out battles. The **confidence-shuffle control passes in
|
||||||
|
the weak sense** — the composite beats its own shuffle (18.41 vs 17.61,
|
||||||
|
p=4.0e-14) — so the weighting carries a *real but tiny* competence signal; it
|
||||||
|
is simply nowhere near enough to beat Pattern. Per the `docs/offline_harness_trust.md`
|
||||||
|
rule (offline is veto-only), this negative **kills the design**; do not take a
|
||||||
|
composite to a live A/B.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 1. What was measured `[MEASURED]`
|
||||||
|
|
||||||
|
The offline gun range (`common_libs/gun_harness/offline_range.nim` +
|
||||||
|
`virtual_bullets.nim`), driven by a new recorder
|
||||||
|
`common_libs/tests/measure_tmcomposites.nim`. For every resolved virtual bullet
|
||||||
|
it records `(gun, tick, powerBin, confidence, hit, aim-bearing-relative-to-LOS,
|
||||||
|
range)`. Ground truth is the shipped `bmPath` virtual-bullet metric (18 px hit
|
||||||
|
radius — `BotRadius`), reproduced exactly per bullet, not a rolling rate.
|
||||||
|
|
||||||
|
| | fixtures | ticks | note |
|
||||||
|
|---|---|---:|---|
|
||||||
|
| **train** (alpha calibration only) | `drussgt_vs_crazy`, `drussgt_vs_spinbot`, `drussgt_vs_ramfire`, `drussgt_vs_corners` | 22 242 | classic-Robocode DrussGT |
|
||||||
|
| **test** (all reported numbers) | `tr_drussgt_vs_modularbot`, `tr_drussgt_vs_spinbot`, `tr_drussgt_vs_corners` | 33 425 | Tank-Royale bridge captures |
|
||||||
|
|
||||||
|
Each fixture is replayed with a **fresh rack** (as `run_range` does); the test
|
||||||
|
battles are **held out by battle**, never by tick. The dump is 3 763 298 rows;
|
||||||
|
per-gun test `n ≈ 133 000` (33 425 ticks × 4 power bins).
|
||||||
|
|
||||||
|
**Per-gun confidence signal defined in source** `[MEASURED]` (the new
|
||||||
|
`GunPrediction.confidence` field, `common_libs/gun_harness/gun_interface.nim`):
|
||||||
|
|
||||||
|
| gun | intrinsic per-sample signal | source |
|
||||||
|
|---|---|---|
|
||||||
|
| GuessFactor | peak GF bin weight `max_i bins[i]` (Eq 4 analogue) | `guess_factor.nim` |
|
||||||
|
| DecayGF | peak decayed GF bin weight | `decay_gf.nim` |
|
||||||
|
| KNN | peak Gaussian density over GF candidates (`bestScore`) | `knn_gun.nim` |
|
||||||
|
| Pattern | match quality `1/(1+bestMatchCost)` | `pattern_matcher.nim` |
|
||||||
|
| Tsetlin | magnitude of the clamped clause-sum vote `hypot(vx,vy)` | `tsetlin.nim` |
|
||||||
|
| TMHorizon | side-class margin `|votes1-votes0|/(2*half)` | `tm_horizon.nim` (instrumented; see §7) |
|
||||||
|
| 8 geometric guns | none — confidence 0.0 | `head_on/linear/circular/wall_bounce/accel_predictor/stop_shot/displacement/averaged_lead` |
|
||||||
|
|
||||||
|
**WARNING — this is exactly the mechanism class the task forbids:** all five are
|
||||||
|
per-sample statistics of the gun's own internal state at prediction time. None is
|
||||||
|
a rolling accuracy or hit-rate average.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 2. The mechanism (paper §3, Eqs 4/6/7/8)
|
||||||
|
|
||||||
|
- Member `t` outputs class sums `c^i_{t,d}` (Eq 6); confidence is `c_max` (Eq 4).
|
||||||
|
- Eq 7 normalises by `alpha_t = max_{d,i}(c) - min_{d,i}(c)`.
|
||||||
|
- Eq 8: `y_d = argmax_i sum_t (1/alpha_t) c^i_{t,d}`.
|
||||||
|
|
||||||
|
Adapted to an **angle output**: the class is a 0.5° aim bin relative to the
|
||||||
|
fire-time line of sight (`[-45°, +45°]`, 180 bins); each confident gun casts its
|
||||||
|
alpha-normalised confidence into the bin of its own aim; the argmax bin wins, and
|
||||||
|
the aim distance is the confidence-weighted mean distance of that bin's voters.
|
||||||
|
`alpha_t` is calibrated on the **train** fixtures and frozen for test, so the test
|
||||||
|
composite never sees test data when choosing its weights (Eq 7's `X` = train).
|
||||||
|
Two composites are scored:
|
||||||
|
|
||||||
|
- **Composite** = all five confidence guns {Tsetlin, GuessFactor, Pattern, DecayGF, KNN}.
|
||||||
|
- **CompositeF** = only the guns §3 finds faithful {Pattern, DecayGF, KNN}.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 3. Task A — confidence faithfulness `[MEASURED]`
|
||||||
|
|
||||||
|
Test samples ranked by each gun's own confidence. `rho` = Spearman(confidence,
|
||||||
|
hit); `bottom/top` = accuracy in the lower/upper confidence half; the two-prop
|
||||||
|
p is for top vs bottom.
|
||||||
|
|
||||||
|
| gun | n | nonzero | base | rho | rho p | bottom% | top% | verdict |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|---|
|
||||||
|
| Accel | 133 101 | 0 | 21.02 | — | — | 14.47 | 27.57 | NO SIGNAL |
|
||||||
|
| **Pattern** | 133 100 | 132 860 | 20.45 | **+0.064** | 4.1e-121 | 18.61 | **22.28** | **FAITHFUL (weak)** |
|
||||||
|
| WallBounce | 133 103 | 0 | 20.15 | — | — | 14.84 | 25.46 | NO SIGNAL |
|
||||||
|
| AvgLead | 133 100 | 0 | 20.08 | — | — | 14.43 | 25.73 | NO SIGNAL |
|
||||||
|
| StopShot | 133 078 | 0 | 18.21 | — | — | 12.82 | 23.61 | NO SIGNAL |
|
||||||
|
| Tsetlin | 132 964 | 105 353 | 18.18 | +0.001 | 0.68 | 17.94 | 18.42 | **USELESS** |
|
||||||
|
| Linear | 133 104 | 0 | 17.31 | — | — | 13.04 | 21.57 | NO SIGNAL |
|
||||||
|
| Circular | 133 107 | 0 | 17.27 | — | — | 12.68 | 21.86 | NO SIGNAL |
|
||||||
|
| **GuessFactor** | 133 108 | 133 108 | 15.89 | **−0.067** | 5.1e-132 | 18.14 | **13.65** | **ANTI-FAITHFUL** |
|
||||||
|
| **DecayGF** | 133 104 | 133 104 | 14.83 | **+0.133** | <1e-300 | 10.79 | **18.86** | **FAITHFUL (strong)** |
|
||||||
|
| Displace | 133 105 | 0 | 13.95 | — | — | 11.42 | 16.47 | NO SIGNAL |
|
||||||
|
| **KNN** | 133 121 | 132 685 | 12.32 | **+0.090** | 2.1e-237 | 9.65 | **14.99** | **FAITHFUL** |
|
||||||
|
| HeadOn | 133 162 | 0 | 6.18 | — | — | 7.09 | 5.27 | NO SIGNAL |
|
||||||
|
|
||||||
|
Accuracy-vs-confidence curves (deciles of confidence, low→high), the paper's
|
||||||
|
Figure 2 reproduced per gun:
|
||||||
|
|
||||||
|
```
|
||||||
|
Pattern 14% 16% 19% 23% 22% 22% 21% 23% 22% 23% rises, then flat
|
||||||
|
DecayGF 10% 9% 11% 15% 9% 11% 16% 21% 23% 23% rises (noisy)
|
||||||
|
KNN 10% 9% 9% 10% 10% 12% 13% 15% 16% 19% monotone rise
|
||||||
|
GuessFactor 16% 21% 21% 15% 18% 14% 18% 14% 13% 10% FALLS
|
||||||
|
Tsetlin 13% 25% 12% 21% 19% 20% 16% 18% 20% 19% flat/noise
|
||||||
|
```
|
||||||
|
|
||||||
|
**Reading.** DecayGF and KNN are the textbook faithful shapes (accuracy climbs
|
||||||
|
with confidence). Pattern is faithful but weakly — its one prior is strong
|
||||||
|
(`18.61% → 22.28%`) and then flattens, i.e. the match cost discriminates
|
||||||
|
"no/poor match" from "match" but not much *between* matches. GuessFactor is the
|
||||||
|
surprise and the most important negative: its histogram peak is **anti-faithful**,
|
||||||
|
consistent with the earlier finding that the GF code path is degenerate on these
|
||||||
|
`tr-bridge` captures. Tsetlin's `c_max` (clause-sum magnitude) carries **no**
|
||||||
|
information about whether the shot hits — `[INFERRED]` because the TM's
|
||||||
|
regression correction is trained on a residual and never learns the surfer
|
||||||
|
(`tsetlin.nim` header: every variant sat at chance vs a shuffled-control).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 4. Task B — complementary specialists? `[MEASURED]`
|
||||||
|
|
||||||
|
For each pair we split test samples by which gun has the higher **alpha-normalised**
|
||||||
|
confidence and measure **both** guns on each slice. `A|A` = A's accuracy on the
|
||||||
|
slice A wins, `B|A` = B's accuracy on that same slice; `p` is a paired McNemar.
|
||||||
|
|
||||||
|
| A | B | A_wins | A\|A | B\|A | p_A | B_wins | A\|B | B\|B | p_B | complementary |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---|
|
||||||
|
| DecayGF | GuessFactor | 43 493 | 19.1 | **20.0** | 4e-14 | 89 608 | 12.8 | **13.9** | 8e-46 | **No** (GF dominates both) |
|
||||||
|
| DecayGF | KNN | 436 | 30.0 | **30.7** | 0.25 | 132 656 | **14.8** | 12.3 | 4e-99 | **No** (DecayGF wins 0.3% only) |
|
||||||
|
| DecayGF | Pattern | 26 788 | 11.0 | **14.8** | 1e-51 | 106 279 | 15.8 | **21.9** | 0 | **No** (Pattern dominates) |
|
||||||
|
| DecayGF | Tsetlin | 121 163 | 14.7 | **18.4** | 3e-211 | 11 795 | **16.2** | 15.4 | 0.045 | **No** |
|
||||||
|
| GuessFactor | KNN | 44 521 | **12.3** | 8.5 | 2e-84 | 88 570 | **17.7** | 14.3 | 9e-115 | **No** (GF dominates both) |
|
||||||
|
| GuessFactor | Pattern | 40 605 | 10.6 | **14.3** | 7e-69 | 92 459 | 18.2 | **23.2** | 2e-207 | **No** |
|
||||||
|
| GuessFactor | Tsetlin | 117 799 | 15.5 | **17.9** | 6e-92 | 15 156 | 19.2 | **20.4** | 0.0012 | **No** |
|
||||||
|
| KNN | Pattern | 38 485 | 9.7 | **15.8** | 6e-155 | 94 344 | 13.4 | **22.4** | 0 | **No** |
|
||||||
|
| KNN | Tsetlin | 131 820 | 12.2 | **18.1** | 0 | 804 | 21.6 | **30.0** | 6e-06 | **No** (KNN wins 0.6% only) |
|
||||||
|
| Pattern | Tsetlin | 121 591 | **21.1** | 18.8 | 2e-58 | 11 221 | **13.6** | 11.6 | 2e-07 | **No** (Pattern dominates) |
|
||||||
|
|
||||||
|
**No pair is complementary.** The pattern in nearly every row is that the more
|
||||||
|
accurate gun is the same one on *both* slices — the confidence orderings are not
|
||||||
|
aligned across guns, so "A is confident here" does not mean "A is the expert
|
||||||
|
here". The two pairs with an unequal split (DecayGF vs KNN, KNN vs Tsetlin) are
|
||||||
|
ones where the "loser" wins *only* 0.3–0.6% of samples — not a usable slice. A
|
||||||
|
pair with no complementary slices cannot form a useful composite, and none does.
|
||||||
|
`[INFERRED]` the mechanism: all five guns estimate the *same* quantity (the
|
||||||
|
enemy's intercept) from the *same* recorded trajectory; differing booleanisations
|
||||||
|
create different *noise*, not different *competence regions*, so their confidence
|
||||||
|
rankings carry no cross-gun information.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 5. Task C — composite vs best single, with the shuffle control `[MEASURED]`
|
||||||
|
|
||||||
|
Held-out test battles, paired per sample. `member_oracle` = accuracy if a
|
||||||
|
perfect per-sample selector could pick any confidence member (the ceiling for
|
||||||
|
any member-picking composite; a free-aim oracle could only be higher).
|
||||||
|
|
||||||
|
| arm | accuracy | n |
|
||||||
|
|---|---:|---:|
|
||||||
|
| **Accel** (best single overall — no confidence) | **21.02%** | 133 101 |
|
||||||
|
| **Pattern** (best single *with* confidence / incumbent) | **20.45%** | 133 100 |
|
||||||
|
| WallBounce | 20.15% | 133 103 |
|
||||||
|
| AvgLead | 20.08% | 133 100 |
|
||||||
|
| **CompositeF** (faithful only: Pattern+DecayGF+KNN) | **18.68%** | 133 126 |
|
||||||
|
| **Composite** (all 5) | **18.41%** | 133 094 |
|
||||||
|
| Tsetlin | 18.18% | 132 964 |
|
||||||
|
| **CompositeFShuf** (control) | **18.06%** | 133 119 |
|
||||||
|
| **CompositeShuf** (control) | **17.61%** | 133 103 |
|
||||||
|
| GuessFactor | 15.89% | 133 108 |
|
||||||
|
| DecayGF | 14.83% | 133 104 |
|
||||||
|
| KNN | 12.32% | 133 121 |
|
||||||
|
| **member_oracle** (perfect member picker) | **41.72%** | 133 094 |
|
||||||
|
|
||||||
|
Paired comparisons (McNemar on discordant pairs):
|
||||||
|
|
||||||
|
| comparison | hits | opponents | p |
|
||||||
|
|---|---:|---:|---:|
|
||||||
|
| Composite vs **Pattern** | 24 488 | 27 215 | **3.1e-126** (composite loses) |
|
||||||
|
| CompositeF vs **Pattern** | 24 854 | 27 215 | **1.25e-133** (composite loses) |
|
||||||
|
| Composite vs its **shuffle** | 24 496 | 23 443 | 4.0e-14 (composite wins) |
|
||||||
|
| CompositeF vs its **shuffle** | 24 864 | 24 040 | 4.0e-15 (composite wins) |
|
||||||
|
| CompositeF vs Composite | — | — | (faithful-only is marginally better, +0.27pp) |
|
||||||
|
|
||||||
|
Per held-out battle (composite vs Pattern): `tr_drussgt_vs_modularbot` 12.2% vs
|
||||||
|
13.0%; `tr_drussgt_vs_spinbot` 29.0% vs 33.9%; `tr_drussgt_vs_corners` 22.3% vs
|
||||||
|
22.4%. The composite loses **all three**.
|
||||||
|
|
||||||
|
**Verdict: the design is dead offline.** The alpha-normalised confidence-weighted
|
||||||
|
composite does **not** beat the best single gun; it loses to Pattern by ~1.8–2.0pp
|
||||||
|
with p≈1e-130, and to the best single overall (Accel) by more. The shuffle control
|
||||||
|
**passes in the weak sense**: the composite is genuinely (p≈1e-14) better than a
|
||||||
|
version with the same weighting distribution but shuffled competences, so the
|
||||||
|
confidence signal is not pure noise — but the effect is ~0.6–0.8pp versus a ~2pp
|
||||||
|
deficit, i.e. real but far too small. The `member_oracle` of **41.72%** shows
|
||||||
|
enormous headroom exists — it is not reachable from these confidence signals.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 6. Why it fails `[INFERRED]`
|
||||||
|
|
||||||
|
1. **Faithfulness is not competence.** A gun can be perfectly confidence-ordered
|
||||||
|
and still be worse than another gun everywhere. DecayGF is *more* faithful than
|
||||||
|
Pattern (rho 0.133 vs 0.064) yet 5.6pp less accurate, so its (correct) ordering
|
||||||
|
contributes weak votes against a stronger member.
|
||||||
|
2. **The members are not specialists.** §4 shows the confidence orderings do not
|
||||||
|
identify competence regions across guns; every gun attacks the whole input
|
||||||
|
space. The paper's win comes from members that are *good on disjoint subsets*.
|
||||||
|
3. **The signal is diluted by anti-faithful members.** GuessFactor is anti-faithful
|
||||||
|
and Tsetlin is useless; CompositeF (faithful only) is better than Composite
|
||||||
|
(18.68 vs 18.41) and more faithful (rho 0.114 vs 0.009), which confirms the
|
||||||
|
poison — but removing it still leaves the composite below Pattern.
|
||||||
|
4. **Alpha normalisation is unstable for online-learning guns.** Eq 7 sets
|
||||||
|
`alpha_t` to the train range; GuessFactor's histogram accumulates unboundedly
|
||||||
|
(alpha≈1.5e4 here), so its normalised confidence is tiny on test and it almost
|
||||||
|
never casts a decisive vote. `[INFERRED]` this mutes the very member whose
|
||||||
|
confidence was measured anti-faithful.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 7. Limits and caveats
|
||||||
|
|
||||||
|
- **Open-loop corpus.** `[MEASURED]`/`[INFERRED]` The fixtures are recorded
|
||||||
|
trajectories; the enemy never reacts to the composite's (or any) bullets, and
|
||||||
|
the `tr-bridge` battles were recorded while **Pattern's selector** was shooting.
|
||||||
|
A gun that behaves like Pattern is therefore favoured in *framing*. This cannot
|
||||||
|
rescue the composite: it loses to Pattern **and** to Accel/AvgLead/WallBounce, so
|
||||||
|
the deficit is not a framing artefact. Per `docs/offline_harness_trust.md`, this
|
||||||
|
is exactly the closed-loop class of question where the offline harness is
|
||||||
|
**veto-only** — a negative kills the design, a positive would have proved nothing.
|
||||||
|
- **Coverage gap (documented LIMIT).** `[MEASURED]` The offline range builds guns
|
||||||
|
0..13; TMPATTERN (14) and TMHORIZON (15) are not in it
|
||||||
|
(`docs/offline_harness_trust.md` §2.8). TMHorizon's `sideConf` is instrumented in
|
||||||
|
source but not measured here. Since Tsetlin's c_max is already useless and no pair
|
||||||
|
composes, the omission does not change the verdict.
|
||||||
|
- **One composite architecture.** `[INFERRED]` The vote uses each gun's scalar
|
||||||
|
confidence at the bin of its own aim. The paper's members also expose a full
|
||||||
|
class distribution (GF/KNN do). Feeding those full distributions could refine the
|
||||||
|
composite, but §4 shows the core premise — cross-gun competence specialisation —
|
||||||
|
is absent, so a refined vote has no signal to exploit.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 8. Recommendation
|
||||||
|
|
||||||
|
**Do not build or A/B the composite.** The offline gate is a veto and it vetoes.
|
||||||
|
|
||||||
|
Two by-products worth keeping:
|
||||||
|
|
||||||
|
- **GuessFactor / anti-faithful warning.** GuessFactor's own histogram peak is
|
||||||
|
anti-faithful on this corpus; do **not** use it as a competence signal (e.g. for
|
||||||
|
a confidence-gated firing decision or a power policy). Pattern, DecayGF and KNN
|
||||||
|
confidences are faithful and *could* gate a shot ("do not fire when not
|
||||||
|
confident") — a different, narrower mechanism than the composite, and still
|
||||||
|
subject to the live veto.
|
||||||
|
- **The selector conclusion is reinforced.** The rack's problem is not the
|
||||||
|
*decision statistic* the selector uses (rolling rate vs per-sample confidence):
|
||||||
|
even a per-sample intrinsic confidence, applied cross-gun, cannot beat Pattern.
|
||||||
|
The rack does not contain complementary specialists.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 9. Reproduction `[MEASURED]`
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 1. record per-sample confidence + hit (≈ 6.5 min; fresh rack per fixture)
|
||||||
|
nim c -d:release --nimcache:/tmp/nc_j104 --path:common_libs \
|
||||||
|
-o:/tmp/measure_tmc common_libs/tests/measure_tmcomposites.nim
|
||||||
|
/tmp/measure_tmc --out /tmp/tmc_full.jsonl \
|
||||||
|
--train tools/fixtures/drussgt_vs_crazy.jsonl tools/fixtures/drussgt_vs_spinbot.jsonl \
|
||||||
|
tools/fixtures/drussgt_vs_ramfire.jsonl tools/fixtures/drussgt_vs_corners.jsonl \
|
||||||
|
--test tools/fixtures/tr_drussgt_vs_modularbot.jsonl tools/fixtures/tr_drussgt_vs_spinbot.jsonl \
|
||||||
|
tools/fixtures/tr_drussgt_vs_corners.jsonl
|
||||||
|
|
||||||
|
# 2. analyse (≈ 40 s); committed result is common_libs/tests/fixtures/tmcomposites_gate.json
|
||||||
|
python3 common_libs/tests/analyze_tmcomposites.py \
|
||||||
|
--input /tmp/tmc_full.jsonl --json common_libs/tests/fixtures/tmcomposites_gate.json
|
||||||
|
```
|
||||||
|
|
||||||
|
The recorder threads a new per-sample `confidence` field through
|
||||||
|
`GunPrediction`/`FeedbackEvent`/`VirtualBullet` (`gun_interface.nim`,
|
||||||
|
`virtual_bullets.nim`) and populates it in `guess_factor.nim`, `decay_gf.nim`,
|
||||||
|
`knn_gun.nim`, `pattern_matcher.nim`, `tsetlin.nim`, `tm_horizon.nim`; the field
|
||||||
|
defaults to 0.0, so every existing caller and test is unchanged. Verified
|
||||||
|
`test_gun_harness`, `test_vbullet_metric`, `test_wave_pairing`,
|
||||||
|
`test_tm_horizon`, `test_pattern_radial_offset`, `test_range_rack_parity`,
|
||||||
|
`test_selector_tiebreak` all pass.
|
||||||
Reference in New Issue
Block a user