Files
SirRoboGarage/common_libs/tests/label_inversion_three_way.py
T

112 lines
4.3 KiB
Python

#!/usr/bin/env python3
"""Task C — the label-inversion question under ONE consistent computation (j133).
Reproduces, on the recorded corpus and with the SAME extraction/metric, the
correlation between each candidate DANGER MAP and the realised per-bin hit rate
`P(hit | b_our = g)` (the j128 metric, whose histogram value is -0.342):
(i) histogram label danger(g) = P(arrival bin = g) [j128]
(ii) outcome proxy label danger(g) = P(hit and |g - b_our| <= w) [j130]
(iii) EXACT bullet-line label danger(g) = P(|g - b_bullet| <= w) [j131]
(iv) the STATE-CONDITIONAL outcome model's own predicted danger, held out
by battle danger(g) = mean_test P_hat(hit | state, g) [new]
Negative = minimising the danger steers INTO where the observed hits happen.
If the physically-EXACT label (iii) is still negative, exact geometry does NOT
fix the inversion and the observable STATE is the binding constraint.
Run:
python3 common_libs/tests/label_inversion_three_way.py --corpus /tmp/tfil_ab2/out
"""
from __future__ import annotations
import argparse
import os
import statistics
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import analyze_drussgt_dodge_vs_power as adp
import outcome_label_gate as olg
NBINS = olg.NBINS
def realised(recs):
n = [0] * NBINS
h = [0.0] * NBINS
for r in recs:
n[r["b_our"]] += 1
h[r["b_our"]] += r["hit"]
used = [b for b in range(NBINS) if n[b] > 0]
rate = [h[b] / n[b] for b in used]
return used, rate
def label_corrs(recs):
used, rate = realised(recs)
hist = statistics.correlation([sum(1 for r in recs if r["b_our"] == b) / len(recs)
for b in used], rate)
proxy = statistics.correlation(
[statistics.fmean(olg.hitwin(r, b) for r in recs) for b in used], rate)
exact = statistics.correlation(
[statistics.fmean(1 if abs(b - r["b_bullet"]) <= r["w"] else 0
for r in recs) for b in used], rate)
return hist, proxy, exact
def model_corr(recs, seed, state_free):
"""Held-out (split BY BATTLE) state-conditional model danger vs test hit rate."""
tr_b, te_b = olg.split_battles({r["battle"] for r in recs}, seed)
tr = [r for r in recs if r["battle"] in tr_b]
te = [r for r in recs if r["battle"] in te_b]
om = olg.OutcomeModel(decay=128, shift=1, state_free=state_free)
edges = dict(olg.CANON)
for r in tr:
st = olg.code_of(r, edges)
for g in range(NBINS):
om.learn(st, g, olg.hitwin(r, g))
used, rate = realised(te)
danger = [statistics.fmean(om.predict_hit(olg.code_of(r, edges), b) for r in te)
for b in used]
return statistics.correlation(danger, rate)
def main() -> int:
ap = argparse.ArgumentParser()
ap.add_argument("--corpus", default="/tmp/tfil_ab2/out")
ap.add_argument("--seeds", type=int, default=3)
args = ap.parse_args()
recs = olg.extract(adp.discover_tfil(args.corpus))
hist, proxy, exact = label_corrs(recs)
ms = [model_corr(recs, s, False) for s in range(args.seeds)]
mf = [model_corr(recs, s, True) for s in range(args.seeds)]
print(f"corpus: {args.corpus} records: {len(recs)} "
f"base hit: {statistics.fmean(r['hit'] for r in recs) * 100:.2f}%")
print()
print("corr( danger(g) , P(hit | b_our = g) ) [the j128 metric]")
print("------------------------------------------------------------")
print(f"(i) histogram label (j128) : {hist:+.3f}")
print(f"(ii) outcome proxy label (j130) : {proxy:+.3f}")
print(f"(iii) EXACT bullet-line label (j131) : {exact:+.3f}")
print(f"(iv) state-CONDITIONAL outcome model : {statistics.fmean(ms):+.3f} "
f"(seeds {['%+.3f' % v for v in ms]})")
print(f" state-FREE outcome model : {statistics.fmean(mf):+.3f} "
f"(seeds {['%+.3f' % v for v in mf]})")
print()
if exact < 0:
print("VERDICT: the physically-exact label is STILL negative -> the LABEL "
"was never the problem; the observable STATE is the binding "
"constraint (closes the learned-movement family).")
else:
print("VERDICT: the exact label is positive -> geometry, not state, was "
"the binding constraint.")
return 0
if __name__ == "__main__":
sys.exit(main())