TM verdict, settled: it loses LIVE and sits at/below its majority class - (c)
The user pushed back on "the TM can't be your best 1v1 gun", correctly, because two
decisive tests had never been run. Both are now run and they agree.
TASK 1 - THE GF HEAD vs ITS MAJORITY-CLASS BASELINE (offline, n=1,751,067):
label histogram [254286, 284578, 678879, 297055, 236269]
majority class = 2 (the CENTRE bucket) = 38.77%
RAW head accuracy = 36.69% -> margin **-2.08 pp, BELOW majority**
GATED head accuracy = 40.37% vs 38.75% majority -> +1.62 pp, BUT it predicts the
majority class on 62.4% of ticks and its minority recall is 13.6% / 12.9% - a
base-rate predictor wearing a classifier's clothes.
Shuffled control sits at its own majority (20.04% vs 20.12%), confirming chance.
**THE OLD "46% vs 20% CHANCE" FIGURE I QUOTED WAS WRONG ON TWO COUNTS:** the
baseline is 38.8%, not 20%, and the 46% predated the deferred-label fix. Against
the correct baseline the head is BELOW it.
TASK 2 - THE FIRST-EVER LIVE A/B OF THE TM GUN (7 runs x 7 rounds per arm, one
frozen binary from git archive HEAD = eb74f9b2, sha256 cb66d66b..., real DrussGT,
every arm forced alone with TR_RACK_<GUN>=both and all 14 others off, liveness
confirmed per run):
arm shots real % dmg/run round wins
onlyPattern 4610 10.74% 285 25/49
onlyTMPATTERN (radial) 3374 3.50% 71 0/49
onlyLinear 3218 3.23% 61 0/49
Pattern vs TM: +7.22 pp / +213.7 dmg, exact p=0.0006
TM vs Linear: +0.30 pp, p=0.659 (dmg p=0.438)
**The TM is statistically INDISTINGUISHABLE from its own Linear base live.** So it
is not "the TM works and we are aiming it wrong".
DIRECT ANSWER: **(c) It loses live AND sits at/below majority - the target carries
no learnable signal beyond the base rate, and that is the reason.** The reason is
not the machine, not the knobs, and not the application alone: the thing it was
asked to predict is dominated by the modal answer.
This closes the TM-as-gun thread. If a TM is wanted in the bot, a firing gate or a
movement decision is a better fit for a boolean-rule classifier than an aim point -
that is untested and is a different project.
A LIVE GF-MODE ARM WAS NOT RUN (stated as unmeasured): the task pinned one frozen
HEAD binary and HEAD registers the TM gun as radial only; Task 1 already makes GF
the unpromising candidate.
HARNESS FIX WORTH KEEPING: `tools/ab/which_gun_arm_env.sh` left the TARGET gun
unset, so with the now-Pattern-only default it silently fell back to the FULL rack
- an arm could appear to test a single gun while actually running the whole rack.
It now emits `TR_RACK_<GUN>=both` for the target and `=off` for all 14 others.
(Earlier which-gun results are unaffected: they ran before the Pattern-only default,
or - as in the melee/1v1 campaign - set the explicit `=both` themselves.)
tm_pattern.nim gains a per-class confusion matrix (warm samples only) to support the
majority baseline; no behaviour change. Adds Round 4 to
tm_pattern_sweep_results.md with both tasks and the interpretation rule.
This commit is contained in:
@@ -45,6 +45,8 @@ type
|
||||
labHist, choHist: array[TM_CLASSES, int]
|
||||
classCorrect, classTotal: int
|
||||
radCorrect, radTotal, revCorrect, revTotal: int
|
||||
confusion: array[TM_CLASSES, array[TM_CLASSES, int]]
|
||||
radConfusion: array[TM_CLASSES, array[TM_CLASSES, int]]
|
||||
|
||||
Adapt = object
|
||||
h100, n100, h300, n300, hall, nall, f100, m100: int
|
||||
@@ -103,6 +105,9 @@ proc addAdapt(dst: var Adapt, src: Adapt) =
|
||||
for c in 0..<TM_CLASSES:
|
||||
dst.st.labHist[c] += src.st.labHist[c]
|
||||
dst.st.choHist[c] += src.st.choHist[c]
|
||||
for p in 0..<TM_CLASSES:
|
||||
dst.st.confusion[c][p] += src.st.confusion[c][p]
|
||||
dst.st.radConfusion[c][p] += src.st.radConfusion[c][p]
|
||||
|
||||
proc replayRound(states: seq[WorldState], lastSeen: seq[int], enemyId, baseTick: int,
|
||||
driver: GunDriver, metric: BulletMetric,
|
||||
@@ -158,6 +163,9 @@ proc replayRound(states: seq[WorldState], lastSeen: seq[int], enemyId, baseTick:
|
||||
for c in 0..<TM_CLASSES:
|
||||
res.st.labHist[c] = after.labHist[c] - obsBefore.labHist[c]
|
||||
res.st.choHist[c] = after.choHist[c] - obsBefore.choHist[c]
|
||||
for p in 0..<TM_CLASSES:
|
||||
res.st.confusion[c][p] = after.confusion[c][p] - obsBefore.confusion[c][p]
|
||||
res.st.radConfusion[c][p] = after.radConfusion[c][p] - obsBefore.radConfusion[c][p]
|
||||
result = res
|
||||
|
||||
proc replayFixture(fx: Fixture, path: string, driver: GunDriver, metric: BulletMetric,
|
||||
@@ -209,6 +217,8 @@ proc tmpatStats(g: ref TmPatternGun): GunStats =
|
||||
result.choHist = g[].chosenHist
|
||||
result.classCorrect = g[].classCorrect
|
||||
result.classTotal = g[].classTotal
|
||||
result.confusion = g[].confusion
|
||||
result.radConfusion = g[].radConfusion
|
||||
result.radCorrect = g[].radCorrect
|
||||
result.radTotal = g[].radTotal
|
||||
result.revCorrect = g[].revCorrect
|
||||
@@ -252,6 +262,35 @@ proc signTestP(wins, n: int): float =
|
||||
for k in 0..lo: s += binomPmf(k, n)
|
||||
min(1.0, 2.0 * s)
|
||||
|
||||
proc printConfusion(variant: string, cm: array[TM_CLASSES, array[TM_CLASSES, int]]) =
|
||||
## Task 1 deliverable: majority-class baseline vs online accuracy on the SAME
|
||||
## (warm) samples, plus per-class precision/recall. `cm[true][pred]`.
|
||||
var total = 0
|
||||
var diag = 0
|
||||
var rowSum: array[TM_CLASSES, int]
|
||||
var colSum: array[TM_CLASSES, int]
|
||||
for c in 0..<TM_CLASSES:
|
||||
for p in 0..<TM_CLASSES:
|
||||
total += cm[c][p]
|
||||
if c == p: diag += cm[c][p]
|
||||
rowSum[c] += cm[c][p]
|
||||
colSum[p] += cm[c][p]
|
||||
var maj = 0
|
||||
for c in 1..<TM_CLASSES:
|
||||
if rowSum[c] > rowSum[maj]: maj = c
|
||||
let majShare = if total > 0: rowSum[maj].float / total.float else: 0.0
|
||||
let acc = if total > 0: diag.float / total.float else: 0.0
|
||||
var hs = ""
|
||||
for c in 0..<TM_CLASSES: hs.add &"{rowSum[c]},"
|
||||
echo &"# {variant}: warmTotal={total} warmLabelHist=[{hs}] majority=class{maj} " &
|
||||
&"majShare={majShare*100:.1f}% acc={diag}/{total}={acc*100:.1f}% " &
|
||||
&"margin={((acc-majShare)*100):+.1f}pp"
|
||||
for c in 0..<TM_CLASSES:
|
||||
let rec = if rowSum[c] > 0: cm[c][c].float / rowSum[c].float else: 0.0
|
||||
let prec = if colSum[c] > 0: cm[c][c].float / colSum[c].float else: 0.0
|
||||
echo &"# class{c}: trueN={rowSum[c]} predN={colSum[c]} TP={cm[c][c]} " &
|
||||
&"recall={rec*100:.1f}% precision={prec*100:.1f}%"
|
||||
|
||||
proc main() =
|
||||
var nSeeds = 3
|
||||
var set = "real"
|
||||
@@ -358,6 +397,14 @@ proc main() =
|
||||
for c in 0..<TM_CLASSES: rdl.add &"{radLabTab[variantName(v)][c]},"
|
||||
echo &"# label hist {variantName(v)}: radial=[{rdl}] rev=[{rl}]"
|
||||
|
||||
# ── TASK 1: majority-class baseline vs accuracy + per-class P/R ──
|
||||
echo "\n# ── GF head majority-class test (warm samples only) ──"
|
||||
for v in [vTmpat, vTmpatShuf]:
|
||||
if v in variants: printConfusion(variantName(v), pooled[variantName(v)].st.confusion)
|
||||
echo "\n# ── radial head confusion (context) ──"
|
||||
for v in [vTmpatRad, vTmpatRadShuf]:
|
||||
if v in variants: printConfusion(variantName(v), pooled[variantName(v)].st.radConfusion)
|
||||
|
||||
# ── per-run distributions (a "run" = one fixture × one seed) ──
|
||||
# Linear is deterministic: replicate its one row per fixture across seeds so a
|
||||
# paired comparison against a stochastic variant has a partner per run.
|
||||
|
||||
Reference in New Issue
Block a user