diff --git a/common_libs/tests/gate2b_hit_optimal_results.txt b/common_libs/tests/gate2b_hit_optimal_results.txt new file mode 100644 index 0000000..4dfa1f5 --- /dev/null +++ b/common_libs/tests/gate2b_hit_optimal_results.txt @@ -0,0 +1,1204 @@ +========================================================================================== +GATE 2b - is the shift-per-bucket appliance STRUCTURALLY DEAD or MISCALIBRATED? +========================================================================================== +fixtures : tr_drussgt_vs_modularbot.jsonl, tr_drussgt_vs_modularbot_shield.jsonl +bits : 53 (49 draft + 4 horizon one-hot) +TM : 50 clauses, 64 states, s=3.0, 5 epochs, fresh per round +split : within-round, train 50%, calibrate 70.%, eval rest +appliance : shift on {-20..+20 deg, 0.5 step} maximising HIT COUNT + on the calibration slice; applied out-of-sample +estimated hit : |residual| < atan(18px / range) -- ABSOLUTE IS OPTIMISTIC +DECISIVE arm : PERF-SIGN uses the TRUE sign (cheating) + hit-optimal shift +bullet block : INFERRED from self-energy drops (no gun heading recorded) +========================================================================================== + +## FIXTURE tr_drussgt_vs_modularbot.jsonl: 15 rounds, 20026 ticks + round 1 (L= 1252) nEval/h = [361, 356, 351, 346] + h arm med p90 hit% + 15 naive 4.20 11.73 30.2 + 15 TM-med 4.54 12.06 29.9 + 15 TM-hit 4.93 12.23 24.4 + 15 TM-side 4.32 11.92 28.5 + 15 TM-x0.25 4.43 11.86 30.5 + 15 TM-x0.5 4.66 11.81 29.4 + 15 PERF-SIGN 2.60 10.23 46.3 + 15 shuffled 4.56 12.96 25.5 + 20 naive 6.41 18.49 16.0 + 20 TM-med 6.19 18.21 19.1 + 20 TM-hit 6.50 18.52 18.3 + 20 TM-side 6.60 18.53 17.1 + 20 TM-x0.25 6.41 19.15 20.2 + 20 TM-x0.5 6.27 18.62 21.1 + 20 PERF-SIGN 3.69 16.34 36.2 + 20 shuffled 6.50 18.86 17.7 + 25 naive 8.45 26.51 12.3 + 25 TM-med 9.14 22.74 8.3 + 25 TM-hit 9.40 23.40 10.0 + 25 TM-side 9.81 23.53 11.1 + 25 TM-x0.25 8.46 25.22 16.8 + 25 TM-x0.5 8.54 24.12 16.2 + 25 PERF-SIGN 4.40 22.40 31.1 + 25 shuffled 9.17 26.16 11.4 + 30 naive 8.91 33.37 13.3 + 30 TM-med 11.18 29.74 6.6 + 30 TM-hit 11.97 28.64 7.2 + 30 TM-side 12.27 29.66 8.1 + 30 TM-x0.25 9.85 33.52 13.0 + 30 TM-x0.5 9.93 31.89 12.4 + 30 PERF-SIGN 6.03 25.37 20.5 + 30 shuffled 9.36 37.04 13.9 + + round 2 (L= 1416) nEval/h = [410, 405, 400, 395] + h arm med p90 hit% + 15 naive 2.49 13.34 47.8 + 15 TM-med 4.42 11.90 28.8 + 15 TM-hit 3.73 12.97 40.5 + 15 TM-side 2.75 13.24 45.4 + 15 TM-x0.25 2.83 12.91 42.7 + 15 TM-x0.5 3.24 12.53 39.3 + 15 PERF-SIGN 1.50 11.84 54.9 + 15 shuffled 3.39 13.33 42.7 + 20 naive 4.91 21.45 35.6 + 20 TM-med 7.21 17.83 25.2 + 20 TM-hit 5.21 20.49 33.1 + 20 TM-side 4.42 20.84 36.5 + 20 TM-x0.25 5.17 20.31 31.6 + 20 TM-x0.5 5.25 19.62 30.9 + 20 PERF-SIGN 5.58 16.00 33.1 + 20 shuffled 7.91 20.89 25.7 + 25 naive 7.54 29.02 28.8 + 25 TM-med 8.97 27.06 21.8 + 25 TM-hit 8.81 27.65 18.5 + 25 TM-side 7.24 28.57 25.2 + 25 TM-x0.25 7.25 27.67 20.5 + 25 TM-x0.5 8.04 26.35 21.0 + 25 PERF-SIGN 7.12 23.73 25.2 + 25 shuffled 8.53 29.93 26.2 + 30 naive 12.27 36.16 21.8 + 30 TM-med 10.55 31.84 13.7 + 30 TM-hit 14.50 41.65 13.9 + 30 TM-side 11.12 35.98 15.7 + 30 TM-x0.25 12.39 36.59 17.2 + 30 TM-x0.5 12.65 37.68 15.9 + 30 PERF-SIGN 11.02 29.88 14.9 + 30 shuffled 17.75 43.76 7.3 + + round 3 (L= 845) nEval/h = [239, 234, 229, 224] + h arm med p90 hit% + 15 naive 1.09 13.57 59.4 + 15 TM-med 1.08 13.53 59.8 + 15 TM-hit 0.98 13.43 58.2 + 15 TM-side 1.76 13.66 56.5 + 15 TM-x0.25 1.03 13.67 59.8 + 15 TM-x0.5 0.98 13.54 59.4 + 15 PERF-SIGN 1.87 11.57 65.3 + 15 shuffled 0.96 13.79 57.7 + 20 naive 2.94 23.90 46.2 + 20 TM-med 3.27 22.89 43.2 + 20 TM-hit 3.44 22.99 41.9 + 20 TM-side 3.44 22.99 41.9 + 20 TM-x0.25 2.94 23.52 45.7 + 20 TM-x0.5 2.94 23.02 44.0 + 20 PERF-SIGN 1.95 21.90 54.7 + 20 shuffled 2.97 23.52 46.6 + 25 naive 7.12 32.44 29.3 + 25 TM-med 6.66 31.73 21.4 + 25 TM-hit 9.15 31.77 16.2 + 25 TM-side 7.12 32.44 27.5 + 25 TM-x0.25 6.67 32.12 25.8 + 25 TM-x0.5 6.40 32.24 21.8 + 25 PERF-SIGN 5.12 30.44 44.1 + 25 shuffled 6.01 32.06 29.7 + 30 naive 10.18 42.99 17.9 + 30 TM-med 12.32 40.72 20.1 + 30 TM-hit 12.25 41.36 19.2 + 30 TM-side 9.98 42.05 15.6 + 30 TM-x0.25 9.23 41.92 13.4 + 30 TM-x0.5 9.23 41.66 15.6 + 30 PERF-SIGN 7.98 40.55 22.3 + 30 shuffled 11.09 42.12 15.6 + + round 4 (L= 1446) nEval/h = [419, 414, 409, 404] + h arm med p90 hit% + 15 naive 4.97 11.21 23.9 + 15 TM-med 4.77 12.28 24.3 + 15 TM-hit 5.56 12.71 22.0 + 15 TM-side 5.00 11.30 24.3 + 15 TM-x0.25 4.73 10.83 26.5 + 15 TM-x0.5 4.86 11.03 26.0 + 15 PERF-SIGN 3.26 8.09 33.7 + 15 shuffled 5.88 13.48 19.6 + 20 naive 7.53 16.45 13.5 + 20 TM-med 6.59 16.82 12.3 + 20 TM-hit 6.07 16.22 15.2 + 20 TM-side 6.82 16.19 15.9 + 20 TM-x0.25 6.66 15.64 16.2 + 20 TM-x0.5 6.15 15.56 15.5 + 20 PERF-SIGN 4.30 12.85 29.0 + 20 shuffled 7.19 16.72 14.0 + 25 naive 9.99 23.25 10.5 + 25 TM-med 9.03 22.95 11.7 + 25 TM-hit 10.30 24.95 8.6 + 25 TM-side 9.31 21.88 11.0 + 25 TM-x0.25 9.12 22.81 9.3 + 25 TM-x0.5 8.95 23.07 11.5 + 25 PERF-SIGN 6.11 18.11 20.5 + 25 shuffled 10.51 28.54 12.5 + 30 naive 12.60 29.60 6.9 + 30 TM-med 11.19 29.12 8.7 + 30 TM-hit 10.37 29.63 11.1 + 30 TM-side 11.72 29.66 9.4 + 30 TM-x0.25 11.42 28.81 7.2 + 30 TM-x0.5 10.49 29.33 8.4 + 30 PERF-SIGN 6.41 18.64 14.4 + 30 shuffled 13.68 34.90 6.2 + + round 5 (L= 1893) nEval/h = [553, 548, 543, 538] + h arm med p90 hit% + 15 naive 3.33 13.59 44.7 + 15 TM-med 3.13 13.70 45.4 + 15 TM-hit 3.34 13.60 44.1 + 15 TM-side 3.55 13.70 42.1 + 15 TM-x0.25 3.34 13.60 44.7 + 15 TM-x0.5 3.35 13.60 44.8 + 15 PERF-SIGN 1.93 11.60 54.4 + 15 shuffled 3.76 14.26 42.7 + 20 naive 5.75 21.91 34.9 + 20 TM-med 5.23 21.62 31.8 + 20 TM-hit 5.73 22.28 33.9 + 20 TM-side 5.79 21.59 32.8 + 20 TM-x0.25 5.62 20.81 35.2 + 20 TM-x0.5 5.41 21.21 34.5 + 20 PERF-SIGN 4.25 20.41 42.5 + 20 shuffled 6.70 22.56 30.1 + 25 naive 8.09 29.86 27.6 + 25 TM-med 6.89 29.83 17.5 + 25 TM-hit 7.03 28.99 23.9 + 25 TM-side 7.44 29.54 22.1 + 25 TM-x0.25 7.72 29.57 27.1 + 25 TM-x0.5 7.40 29.22 27.3 + 25 PERF-SIGN 6.24 28.02 37.0 + 25 shuffled 7.99 30.33 25.0 + 30 naive 11.17 35.88 21.2 + 30 TM-med 9.42 36.62 15.4 + 30 TM-hit 10.31 36.05 20.4 + 30 TM-side 11.25 36.37 19.0 + 30 TM-x0.25 10.76 35.58 19.7 + 30 TM-x0.5 10.42 35.30 20.4 + 30 PERF-SIGN 8.94 33.45 30.1 + 30 shuffled 11.82 37.70 18.2 + + round 6 (L= 1080) nEval/h = [309, 304, 299, 294] + h arm med p90 hit% + 15 naive 6.66 13.01 20.7 + 15 TM-med 6.26 12.61 17.5 + 15 TM-hit 6.46 12.93 21.0 + 15 TM-side 6.45 13.23 20.7 + 15 TM-x0.25 6.71 12.93 21.0 + 15 TM-x0.5 6.68 12.93 21.0 + 15 PERF-SIGN 5.16 11.51 27.5 + 15 shuffled 6.61 12.81 19.7 + 20 naive 9.00 20.62 12.2 + 20 TM-med 8.75 18.74 13.2 + 20 TM-hit 9.11 20.21 11.8 + 20 TM-side 9.16 20.23 13.2 + 20 TM-x0.25 8.59 19.92 11.2 + 20 TM-x0.5 8.30 18.93 11.2 + 20 PERF-SIGN 7.50 19.12 18.8 + 20 shuffled 9.09 20.62 12.2 + 25 naive 11.36 26.48 8.0 + 25 TM-med 9.05 23.09 11.4 + 25 TM-hit 10.24 26.44 7.4 + 25 TM-side 11.69 27.48 7.4 + 25 TM-x0.25 10.91 26.51 9.7 + 25 TM-x0.5 10.89 25.81 10.0 + 25 PERF-SIGN 10.09 25.32 13.4 + 25 shuffled 11.05 27.32 9.0 + 30 naive 14.01 32.46 6.5 + 30 TM-med 10.03 29.06 13.6 + 30 TM-hit 13.32 36.28 5.4 + 30 TM-side 13.67 34.07 5.4 + 30 TM-x0.25 13.50 32.39 7.1 + 30 TM-x0.5 13.07 33.78 6.5 + 30 PERF-SIGN 12.14 30.74 11.2 + 30 shuffled 12.97 32.16 5.8 + + round 7 (L= 1460) nEval/h = [424, 419, 414, 409] + h arm med p90 hit% + 15 naive 4.85 14.20 33.5 + 15 TM-med 5.08 13.66 27.1 + 15 TM-hit 5.68 14.20 29.0 + 15 TM-side 4.78 14.20 33.7 + 15 TM-x0.25 4.87 14.06 30.4 + 15 TM-x0.5 4.98 14.31 30.4 + 15 PERF-SIGN 3.35 12.70 42.7 + 15 shuffled 6.25 16.14 24.8 + 20 naive 7.32 20.48 21.7 + 20 TM-med 8.11 19.04 11.5 + 20 TM-hit 7.97 20.48 19.3 + 20 TM-side 7.36 20.71 22.0 + 20 TM-x0.25 7.34 20.58 20.8 + 20 TM-x0.5 7.37 20.48 20.5 + 20 PERF-SIGN 5.82 18.98 32.5 + 20 shuffled 8.64 22.33 17.7 + 25 naive 10.81 28.52 16.4 + 25 TM-med 10.16 26.15 8.7 + 25 TM-hit 11.15 28.67 17.1 + 25 TM-side 10.71 28.54 16.7 + 25 TM-x0.25 11.02 28.54 17.1 + 25 TM-x0.5 10.91 28.54 17.4 + 25 PERF-SIGN 9.31 27.02 23.9 + 25 shuffled 10.86 30.63 13.3 + 30 naive 14.36 34.95 9.5 + 30 TM-med 13.26 32.06 5.9 + 30 TM-hit 14.61 33.93 8.3 + 30 TM-side 14.95 34.44 7.8 + 30 TM-x0.25 14.37 34.70 10.0 + 30 TM-x0.5 13.59 33.80 7.8 + 30 PERF-SIGN 12.86 33.45 18.1 + 30 shuffled 15.53 37.65 9.8 + + round 8 (L= 1213) nEval/h = [349, 344, 339, 334] + h arm med p90 hit% + 15 naive 3.89 15.17 37.2 + 15 TM-med 4.06 13.55 38.7 + 15 TM-hit 4.33 14.87 37.0 + 15 TM-side 4.88 15.52 35.5 + 15 TM-x0.25 3.59 14.45 34.1 + 15 TM-x0.5 4.02 13.76 37.5 + 15 PERF-SIGN 2.39 13.67 47.9 + 15 shuffled 4.27 14.97 37.0 + 20 naive 7.39 21.18 30.5 + 20 TM-med 6.32 21.44 32.0 + 20 TM-hit 6.07 23.20 31.7 + 20 TM-side 7.04 25.02 29.9 + 20 TM-x0.25 6.11 20.33 28.2 + 20 TM-x0.5 6.39 21.14 29.9 + 20 PERF-SIGN 5.89 19.68 38.1 + 20 shuffled 7.76 21.29 33.1 + 25 naive 10.75 26.76 28.3 + 25 TM-med 7.12 29.72 23.9 + 25 TM-hit 6.77 30.96 24.2 + 25 TM-side 7.99 31.04 24.5 + 25 TM-x0.25 9.12 26.47 23.6 + 25 TM-x0.5 8.20 27.55 23.3 + 25 PERF-SIGN 9.70 25.26 32.2 + 25 shuffled 10.70 26.45 30.1 + 30 naive 13.98 31.31 24.9 + 30 TM-med 6.90 32.68 20.7 + 30 TM-hit 8.03 33.07 22.2 + 30 TM-side 9.22 33.06 20.4 + 30 TM-x0.25 12.70 31.94 22.8 + 30 TM-x0.5 11.07 32.56 20.7 + 30 PERF-SIGN 12.48 29.81 28.1 + 30 shuffled 13.01 32.78 18.6 + + round 9 (L= 1759) nEval/h = [513, 508, 503, 498] + h arm med p90 hit% + 15 naive 2.58 14.40 49.3 + 15 TM-med 4.59 13.54 19.9 + 15 TM-hit 2.58 14.23 49.3 + 15 TM-side 2.70 14.50 48.1 + 15 TM-x0.25 2.58 14.38 49.5 + 15 TM-x0.5 2.57 14.03 49.7 + 15 PERF-SIGN 1.78 12.40 60.0 + 15 shuffled 2.82 16.40 47.4 + 20 naive 5.52 22.86 37.0 + 20 TM-med 7.28 20.73 11.8 + 20 TM-hit 9.98 20.53 8.9 + 20 TM-side 5.42 22.41 33.9 + 20 TM-x0.25 5.26 21.56 25.8 + 20 TM-x0.5 6.63 20.64 11.2 + 20 PERF-SIGN 5.42 18.99 33.3 + 20 shuffled 6.92 23.11 26.4 + 25 naive 8.13 31.70 33.6 + 25 TM-med 10.37 29.16 8.5 + 25 TM-hit 9.59 29.43 7.4 + 25 TM-side 8.85 32.87 7.2 + 25 TM-x0.25 8.13 30.60 25.2 + 25 TM-x0.5 8.36 30.00 12.7 + 25 PERF-SIGN 5.21 27.31 24.9 + 25 shuffled 9.92 30.06 17.5 + 30 naive 9.83 39.66 28.1 + 30 TM-med 13.29 36.45 8.4 + 30 TM-hit 11.41 37.37 6.0 + 30 TM-side 9.77 40.57 15.1 + 30 TM-x0.25 10.29 38.90 20.3 + 30 TM-x0.5 10.43 38.33 9.6 + 30 PERF-SIGN 10.56 30.54 11.6 + 30 shuffled 13.28 41.47 15.9 + + round 10 (L= 1380) nEval/h = [400, 395, 390, 385] + h arm med p90 hit% + 15 naive 4.01 13.67 39.5 + 15 TM-med 4.47 13.06 27.0 + 15 TM-hit 3.90 13.40 32.0 + 15 TM-side 3.82 13.34 30.5 + 15 TM-x0.25 4.14 13.54 40.2 + 15 TM-x0.5 4.16 13.42 34.2 + 15 PERF-SIGN 3.17 11.23 36.5 + 15 shuffled 4.31 15.73 37.2 + 20 naive 7.63 20.99 22.8 + 20 TM-med 7.74 21.04 16.7 + 20 TM-hit 7.55 21.65 19.7 + 20 TM-side 7.52 20.37 16.5 + 20 TM-x0.25 7.29 20.99 25.1 + 20 TM-x0.5 7.18 20.97 24.6 + 20 PERF-SIGN 5.27 17.22 20.5 + 20 shuffled 7.41 21.15 22.3 + 25 naive 10.69 28.86 15.4 + 25 TM-med 9.74 28.43 8.2 + 25 TM-hit 8.99 29.45 13.8 + 25 TM-side 8.36 28.36 16.4 + 25 TM-x0.25 10.57 28.96 19.2 + 25 TM-x0.5 9.43 28.93 18.5 + 25 PERF-SIGN 6.08 24.46 14.6 + 25 shuffled 10.36 29.31 11.5 + 30 naive 13.79 35.78 11.2 + 30 TM-med 13.76 34.64 7.0 + 30 TM-hit 12.52 36.57 11.9 + 30 TM-side 12.63 34.28 11.2 + 30 TM-x0.25 13.47 35.90 12.7 + 30 TM-x0.5 12.91 36.28 13.0 + 30 PERF-SIGN 9.67 31.90 19.0 + 30 shuffled 13.23 40.25 9.6 + + round 11 (L= 1127) nEval/h = [324, 319, 314, 309] + h arm med p90 hit% + 15 naive 6.80 14.92 20.7 + 15 TM-med 6.86 15.11 19.1 + 15 TM-hit 8.24 17.47 20.4 + 15 TM-side 6.70 14.85 21.0 + 15 TM-x0.25 6.67 14.12 16.0 + 15 TM-x0.5 6.99 14.85 17.9 + 15 PERF-SIGN 3.97 11.82 27.2 + 15 shuffled 7.93 15.17 13.9 + 20 naive 9.68 22.92 15.4 + 20 TM-med 9.87 22.80 11.9 + 20 TM-hit 10.39 23.44 12.5 + 20 TM-side 10.58 27.27 10.7 + 20 TM-x0.25 9.26 22.24 12.5 + 20 TM-x0.5 9.61 22.49 15.7 + 20 PERF-SIGN 7.98 16.55 18.2 + 20 shuffled 10.36 24.28 9.7 + 25 naive 10.88 30.80 13.7 + 25 TM-med 11.11 28.15 10.2 + 25 TM-hit 10.96 28.03 11.1 + 25 TM-side 12.05 33.22 10.2 + 25 TM-x0.25 10.67 29.83 11.1 + 25 TM-x0.5 10.49 29.19 11.1 + 25 PERF-SIGN 9.86 22.35 14.0 + 25 shuffled 12.39 32.57 11.5 + 30 naive 13.77 38.10 10.0 + 30 TM-med 12.41 33.61 9.7 + 30 TM-hit 13.28 35.20 7.8 + 30 TM-side 14.86 40.87 8.4 + 30 TM-x0.25 12.42 36.61 6.5 + 30 TM-x0.5 12.67 35.72 5.5 + 30 PERF-SIGN 11.54 27.84 11.0 + 30 shuffled 13.87 38.88 5.8 + + round 12 (L= 1258) nEval/h = [363, 358, 353, 348] + h arm med p90 hit% + 15 naive 2.76 13.10 47.4 + 15 TM-med 2.72 12.78 46.8 + 15 TM-hit 2.88 13.16 45.7 + 15 TM-side 2.83 13.00 45.7 + 15 TM-x0.25 2.78 13.12 47.4 + 15 TM-x0.5 2.82 13.07 46.8 + 15 PERF-SIGN 1.96 11.10 56.7 + 15 shuffled 3.28 13.54 41.0 + 20 naive 5.47 19.38 39.4 + 20 TM-med 4.74 19.08 39.9 + 20 TM-hit 4.13 18.92 36.9 + 20 TM-side 5.48 19.50 39.4 + 20 TM-x0.25 5.07 19.36 39.9 + 20 TM-x0.5 4.62 18.92 38.8 + 20 PERF-SIGN 2.99 17.36 44.7 + 20 shuffled 5.80 19.50 33.2 + 25 naive 8.00 26.76 32.0 + 25 TM-med 7.08 26.69 31.2 + 25 TM-hit 6.95 27.50 24.9 + 25 TM-side 7.52 27.50 29.7 + 25 TM-x0.25 7.47 27.31 33.7 + 25 TM-x0.5 6.93 27.24 33.4 + 25 PERF-SIGN 5.40 24.13 37.4 + 25 shuffled 8.52 27.50 29.5 + 30 naive 10.11 33.73 24.1 + 30 TM-med 10.04 32.63 14.4 + 30 TM-hit 8.71 33.81 26.7 + 30 TM-side 13.08 36.25 15.2 + 30 TM-x0.25 9.48 33.57 26.1 + 30 TM-x0.5 9.07 33.26 27.0 + 30 PERF-SIGN 7.32 30.69 27.9 + 30 shuffled 11.52 33.81 14.9 + + round 13 (L= 1330) nEval/h = [385, 380, 375, 370] + h arm med p90 hit% + 15 naive 5.81 16.12 32.7 + 15 TM-med 8.07 16.09 16.1 + 15 TM-hit 6.07 15.20 20.0 + 15 TM-side 5.60 15.63 33.0 + 15 TM-x0.25 5.71 15.60 32.5 + 15 TM-x0.5 5.78 15.45 27.0 + 15 PERF-SIGN 3.81 12.02 39.5 + 15 shuffled 5.75 15.63 27.3 + 20 naive 9.24 23.32 21.8 + 20 TM-med 12.84 24.67 13.7 + 20 TM-hit 14.16 25.89 13.4 + 20 TM-side 10.30 25.00 13.7 + 20 TM-x0.25 9.98 22.98 12.4 + 20 TM-x0.5 11.54 23.51 13.7 + 20 PERF-SIGN 6.89 21.12 32.4 + 20 shuffled 10.76 24.56 18.9 + 25 naive 12.70 31.07 13.1 + 25 TM-med 16.49 33.55 9.9 + 25 TM-hit 17.74 33.22 7.7 + 25 TM-side 13.73 32.50 13.6 + 25 TM-x0.25 13.06 30.05 8.5 + 25 TM-x0.5 14.35 30.52 8.3 + 25 PERF-SIGN 10.21 28.53 19.5 + 25 shuffled 13.18 31.22 14.9 + 30 naive 14.84 36.94 10.0 + 30 TM-med 20.28 42.39 8.6 + 30 TM-hit 19.28 40.64 7.0 + 30 TM-side 16.56 40.65 12.7 + 30 TM-x0.25 15.59 37.84 7.0 + 30 TM-x0.5 16.53 37.51 7.8 + 30 PERF-SIGN 11.84 34.44 18.4 + 30 shuffled 15.03 36.94 10.0 + + round 14 (L= 1474) nEval/h = [428, 423, 418, 413] + h arm med p90 hit% + 15 naive 3.33 13.96 42.5 + 15 TM-med 5.17 11.86 23.4 + 15 TM-hit 3.18 12.11 43.0 + 15 TM-side 3.26 12.58 42.8 + 15 TM-x0.25 3.31 12.89 42.5 + 15 TM-x0.5 3.33 12.42 42.5 + 15 PERF-SIGN 2.25 11.79 50.7 + 15 shuffled 3.29 13.96 42.1 + 20 naive 6.05 21.34 29.1 + 20 TM-med 8.08 19.45 15.6 + 20 TM-hit 4.96 19.95 32.6 + 20 TM-side 5.44 20.96 31.7 + 20 TM-x0.25 6.07 19.75 30.0 + 20 TM-x0.5 5.68 19.70 30.0 + 20 PERF-SIGN 4.04 18.96 36.6 + 20 shuffled 7.04 21.43 23.9 + 25 naive 9.48 28.59 24.2 + 25 TM-med 11.16 26.70 11.7 + 25 TM-hit 7.73 27.76 23.0 + 25 TM-side 7.97 30.31 23.2 + 25 TM-x0.25 9.29 27.37 23.9 + 25 TM-x0.5 9.28 27.00 24.6 + 25 PERF-SIGN 7.56 21.35 22.2 + 25 shuffled 13.05 27.93 11.5 + 30 naive 12.80 37.10 20.8 + 30 TM-med 15.48 35.95 10.2 + 30 TM-hit 12.59 35.19 8.0 + 30 TM-side 11.55 38.10 21.8 + 30 TM-x0.25 11.85 36.02 16.9 + 30 TM-x0.5 11.59 34.42 9.7 + 30 PERF-SIGN 9.85 28.54 14.5 + 30 shuffled 13.71 36.23 14.5 + + round 15 (L= 1093) nEval/h = [313, 308, 303, 298] + h arm med p90 hit% + 15 naive 4.30 18.06 40.9 + 15 TM-med 7.14 14.79 14.1 + 15 TM-hit 4.81 17.25 37.4 + 15 TM-side 4.53 16.87 36.4 + 15 TM-x0.25 4.28 17.53 41.5 + 15 TM-x0.5 4.20 17.61 38.0 + 15 PERF-SIGN 2.30 16.06 51.4 + 15 shuffled 4.58 17.48 39.9 + 20 naive 7.49 26.79 34.1 + 20 TM-med 9.24 21.96 11.4 + 20 TM-hit 7.82 25.26 30.2 + 20 TM-side 9.08 25.94 29.2 + 20 TM-x0.25 7.87 26.03 36.0 + 20 TM-x0.5 7.47 25.71 31.8 + 20 PERF-SIGN 5.55 20.63 15.6 + 20 shuffled 9.11 28.75 7.5 + 25 naive 10.87 35.16 23.1 + 25 TM-med 13.81 29.13 7.9 + 25 TM-hit 10.99 31.50 25.4 + 25 TM-side 11.68 29.23 6.3 + 25 TM-x0.25 11.01 34.22 22.4 + 25 TM-x0.5 10.97 33.25 22.4 + 25 PERF-SIGN 7.48 29.73 32.0 + 25 shuffled 13.51 40.29 15.5 + 30 naive 15.47 43.12 18.5 + 30 TM-med 20.37 36.64 5.0 + 30 TM-hit 15.73 39.72 15.4 + 30 TM-side 15.47 39.30 16.4 + 30 TM-x0.25 15.76 41.51 17.1 + 30 TM-x0.5 15.74 41.15 17.4 + 30 PERF-SIGN 11.61 32.14 9.7 + 30 shuffled 16.96 42.39 9.1 + + +## PER-FIXTURE pooled (tr_drussgt_vs_modularbot.jsonl) + h arm N med|err| p90|err| hit% miss% + 15 naive 5790 4.26 14.10 38.3 61.7 + 15 TM-med 5790 4.85 13.39 28.9 71.1 + 15 TM-hit 5790 4.46 14.20 35.2 64.8 + 15 TM-side 5790 4.26 13.78 36.6 63.4 + 15 TM-x0.0 5790 4.26 14.10 38.3 61.7 + 15 TM-x0.25 5790 4.18 13.78 37.6 62.4 + 15 TM-x0.5 5790 4.28 13.63 36.5 63.5 + 15 PERF-SIGN 5790 2.61 11.88 46.6 53.4 + 15 PS-shift0 5790 4.26 14.10 38.3 61.7 + 15 shuffled 5790 4.61 14.63 34.8 65.2 + 20 naive 5715 6.97 21.49 27.4 72.6 + 20 TM-med 5715 7.61 20.46 20.2 79.8 + 20 TM-hit 5715 7.61 21.44 23.6 76.4 + 20 TM-side 5715 7.06 21.73 25.8 74.2 + 20 TM-x0.0 5715 6.97 21.49 27.4 72.6 + 20 TM-x0.25 5715 6.72 20.93 25.9 74.1 + 20 TM-x0.5 5715 6.84 20.68 24.5 75.5 + 20 PERF-SIGN 5715 5.30 18.38 32.6 67.4 + 20 PS-shift0 5715 6.97 21.49 27.4 72.6 + 20 shuffled 5715 7.76 22.06 22.6 77.4 + 25 naive 5640 9.74 29.02 21.5 78.5 + 25 TM-med 5640 9.89 27.86 13.9 86.1 + 25 TM-hit 5640 9.76 28.99 16.0 84.0 + 25 TM-side 5640 9.72 29.02 16.8 83.2 + 25 TM-x0.0 5640 9.74 29.02 21.5 78.5 + 25 TM-x0.25 5640 9.44 28.60 19.9 80.1 + 25 TM-x0.5 5640 9.33 28.43 18.8 81.2 + 25 PERF-SIGN 5640 7.12 25.48 26.0 74.0 + 25 PS-shift0 5640 9.74 29.02 21.5 78.5 + 25 shuffled 5640 10.68 29.83 17.9 82.1 + 30 naive 5565 12.40 36.16 16.7 83.3 + 30 TM-med 5565 12.75 34.51 11.0 89.0 + 30 TM-hit 5565 12.70 36.23 12.6 87.4 + 30 TM-side 5565 12.61 36.37 13.7 86.3 + 30 TM-x0.0 5565 12.40 36.16 16.7 83.3 + 30 TM-x0.25 5565 12.03 35.77 14.8 85.2 + 30 TM-x0.5 5565 11.94 35.62 13.2 86.8 + 30 PERF-SIGN 5565 9.77 30.85 18.3 81.7 + 30 PS-shift0 5565 12.40 36.16 16.7 83.3 + 30 shuffled 5565 13.63 37.96 11.9 88.1 + +## PER-FIXTURE learnability (tr_drussgt_vs_modularbot.jsonl) + h N TMquad% shufquad% TMside% majQuad% pred[Ls Lb Rs Rb] / true[Ls Lb Rs Rb] + 15 5790 35.0 22.2 58.2 25.4 [1237 1608 898 2047 ] / [1448 1427 1471 1444 ] + 20 5715 35.3 22.3 58.9 25.3 [1273 1542 849 2051 ] / [1392 1438 1445 1440 ] + 25 5640 34.6 24.0 59.0 25.6 [1227 1558 802 2053 ] / [1368 1414 1445 1413 ] + 30 5565 35.4 23.7 61.1 26.1 [1254 1469 834 2008 ] / [1340 1361 1452 1412 ] + +## FIXTURE tr_drussgt_vs_modularbot_shield.jsonl: 10 rounds, 12629 ticks + round 1 (L= 593) nEval/h = [163, 158, 153, 148] + h arm med p90 hit% + 15 naive 2.08 16.95 61.3 + 15 TM-med 2.08 16.95 61.3 + 15 TM-hit 2.08 16.95 61.3 + 15 TM-side 2.08 16.95 61.3 + 15 TM-x0.25 2.08 16.95 61.3 + 15 TM-x0.5 2.08 16.95 61.3 + 15 PERF-SIGN 2.08 16.95 61.3 + 15 shuffled 2.08 16.95 61.3 + 20 naive 4.14 25.55 39.9 + 20 TM-med 4.14 25.55 39.9 + 20 TM-hit 4.14 25.55 39.9 + 20 TM-side 4.14 25.55 39.9 + 20 TM-x0.25 4.14 25.55 39.9 + 20 TM-x0.5 4.14 25.55 39.9 + 20 PERF-SIGN 4.14 25.55 39.9 + 20 shuffled 4.14 25.55 39.9 + 25 naive 6.92 34.33 22.9 + 25 TM-med 6.92 34.33 22.9 + 25 TM-hit 7.00 33.98 20.3 + 25 TM-side 7.00 33.98 20.3 + 25 TM-x0.25 6.91 34.36 22.9 + 25 TM-x0.5 7.03 34.23 20.9 + 25 PERF-SIGN 6.78 33.98 22.9 + 25 shuffled 7.00 33.98 20.3 + 30 naive 10.18 41.79 18.2 + 30 TM-med 10.18 41.79 18.2 + 30 TM-hit 10.18 41.79 18.2 + 30 TM-side 10.18 41.79 18.2 + 30 TM-x0.25 10.18 41.79 18.2 + 30 TM-x0.5 10.18 41.79 18.2 + 30 PERF-SIGN 10.08 40.74 19.6 + 30 shuffled 10.18 41.79 18.2 + + round 2 (L= 1081) nEval/h = [310, 305, 300, 295] + h arm med p90 hit% + 15 naive 2.43 9.95 45.5 + 15 TM-med 3.47 8.70 26.8 + 15 TM-hit 2.78 9.58 37.4 + 15 TM-side 2.29 9.62 47.4 + 15 TM-x0.25 2.43 9.78 46.5 + 15 TM-x0.5 2.58 9.76 43.9 + 15 PERF-SIGN 1.49 8.45 55.5 + 15 shuffled 3.78 14.45 39.7 + 20 naive 4.96 15.77 32.5 + 20 TM-med 4.85 12.53 21.3 + 20 TM-hit 4.74 15.32 30.2 + 20 TM-side 4.65 15.77 34.4 + 20 TM-x0.25 4.85 15.65 33.8 + 20 TM-x0.5 4.73 15.83 33.4 + 20 PERF-SIGN 3.22 14.27 35.7 + 20 shuffled 5.03 17.36 27.9 + 25 naive 8.08 23.11 26.3 + 25 TM-med 7.10 18.45 15.7 + 25 TM-hit 6.56 18.83 16.7 + 25 TM-side 7.22 21.36 21.3 + 25 TM-x0.25 6.97 21.62 21.0 + 25 TM-x0.5 6.65 20.40 15.7 + 25 PERF-SIGN 6.93 17.29 24.0 + 25 shuffled 10.10 25.17 14.0 + 30 naive 11.82 30.98 18.6 + 30 TM-med 8.81 24.99 9.8 + 30 TM-hit 11.94 32.49 9.2 + 30 TM-side 10.21 25.96 16.9 + 30 TM-x0.25 10.47 30.05 10.2 + 30 TM-x0.5 9.89 29.49 8.1 + 30 PERF-SIGN 9.68 28.49 19.3 + 30 shuffled 12.16 30.47 15.6 + + round 3 (L= 1237) nEval/h = [357, 352, 347, 342] + h arm med p90 hit% + 15 naive 4.45 17.62 41.5 + 15 TM-med 7.33 14.79 17.1 + 15 TM-hit 8.23 14.27 17.1 + 15 TM-side 6.05 16.78 24.9 + 15 TM-x0.25 3.62 16.38 32.2 + 15 TM-x0.5 4.92 15.47 17.6 + 15 PERF-SIGN 3.55 12.63 43.7 + 15 shuffled 7.07 21.51 24.6 + 20 naive 7.78 25.57 32.1 + 20 TM-med 11.92 22.17 9.7 + 20 TM-hit 12.52 24.04 9.9 + 20 TM-side 9.69 25.29 9.9 + 20 TM-x0.25 6.75 23.57 11.9 + 20 TM-x0.5 8.36 22.02 9.9 + 20 PERF-SIGN 5.80 20.42 37.8 + 20 shuffled 11.00 27.24 17.0 + 25 naive 11.50 31.52 19.9 + 25 TM-med 16.11 30.15 6.9 + 25 TM-hit 15.99 27.52 9.8 + 25 TM-side 13.31 32.73 7.2 + 25 TM-x0.25 10.33 29.39 11.8 + 25 TM-x0.5 10.29 28.48 7.2 + 25 PERF-SIGN 8.18 26.52 31.7 + 25 shuffled 14.94 36.70 4.6 + 30 naive 14.83 38.77 14.3 + 30 TM-med 20.87 34.41 5.8 + 30 TM-hit 16.61 35.13 5.8 + 30 TM-side 15.39 42.28 6.4 + 30 TM-x0.25 13.01 35.75 15.2 + 30 TM-x0.5 12.75 35.15 6.1 + 30 PERF-SIGN 8.95 26.30 7.3 + 30 shuffled 19.25 46.58 3.8 + + round 4 (L= 1296) nEval/h = [374, 369, 364, 359] + h arm med p90 hit% + 15 naive 1.72 12.89 54.8 + 15 TM-med 3.62 14.68 27.0 + 15 TM-hit 1.97 13.89 54.5 + 15 TM-side 1.91 13.33 53.7 + 15 TM-x0.25 1.69 13.14 55.1 + 15 TM-x0.5 1.65 13.50 54.0 + 15 PERF-SIGN 1.84 11.33 61.8 + 15 shuffled 2.34 13.13 49.2 + 20 naive 3.40 21.79 42.8 + 20 TM-med 7.10 24.36 16.5 + 20 TM-hit 3.90 22.29 38.8 + 20 TM-side 3.73 22.02 39.6 + 20 TM-x0.25 3.52 21.92 43.9 + 20 TM-x0.5 3.47 22.22 44.7 + 20 PERF-SIGN 2.67 19.22 45.8 + 20 shuffled 4.98 21.72 22.8 + 25 naive 5.26 30.32 32.1 + 25 TM-med 9.22 34.28 13.5 + 25 TM-hit 5.86 31.25 24.2 + 25 TM-side 6.84 30.71 9.1 + 25 TM-x0.25 5.42 30.56 32.1 + 25 TM-x0.5 5.43 30.55 31.6 + 25 PERF-SIGN 2.91 27.68 36.8 + 25 shuffled 7.26 29.97 17.6 + 30 naive 6.85 38.04 23.7 + 30 TM-med 10.38 42.47 15.3 + 30 TM-hit 8.11 39.17 11.1 + 30 TM-side 9.46 38.46 7.5 + 30 TM-x0.25 6.73 38.70 24.2 + 30 TM-x0.5 7.53 38.37 20.3 + 30 PERF-SIGN 5.01 32.04 22.3 + 30 shuffled 9.72 37.96 9.5 + + round 5 (L= 1550) nEval/h = [450, 445, 440, 435] + h arm med p90 hit% + 15 naive 4.76 14.38 37.1 + 15 TM-med 4.14 13.43 37.3 + 15 TM-hit 4.17 13.46 36.0 + 15 TM-side 4.32 14.18 37.6 + 15 TM-x0.25 4.39 13.75 34.4 + 15 TM-x0.5 4.50 13.31 34.7 + 15 PERF-SIGN 2.76 12.38 46.2 + 15 shuffled 4.31 14.47 38.7 + 20 naive 7.25 22.56 22.5 + 20 TM-med 7.31 19.53 21.8 + 20 TM-hit 6.87 19.92 18.0 + 20 TM-side 7.50 22.31 21.8 + 20 TM-x0.25 7.17 21.56 22.5 + 20 TM-x0.5 6.91 20.92 20.0 + 20 PERF-SIGN 5.25 20.56 37.5 + 20 shuffled 7.50 21.93 21.3 + 25 naive 9.70 30.01 11.4 + 25 TM-med 9.72 26.88 12.3 + 25 TM-hit 9.33 26.70 12.0 + 25 TM-side 9.03 29.72 15.5 + 25 TM-x0.25 9.42 28.77 10.9 + 25 TM-x0.5 9.06 27.75 11.6 + 25 PERF-SIGN 6.31 26.30 21.6 + 25 shuffled 8.87 29.42 18.9 + 30 naive 12.35 36.18 8.0 + 30 TM-med 13.08 34.35 8.3 + 30 TM-hit 13.31 34.33 9.7 + 30 TM-side 12.96 36.80 10.6 + 30 TM-x0.25 11.54 35.43 9.2 + 30 TM-x0.5 12.07 34.68 8.7 + 30 PERF-SIGN 7.33 31.78 17.7 + 30 shuffled 13.00 36.19 9.0 + + round 6 (L= 1284) nEval/h = [371, 366, 361, 356] + h arm med p90 hit% + 15 naive 4.35 13.01 29.9 + 15 TM-med 5.08 12.79 14.3 + 15 TM-hit 4.38 12.73 30.5 + 15 TM-side 4.44 12.78 29.4 + 15 TM-x0.25 4.19 13.05 30.7 + 15 TM-x0.5 4.14 12.93 31.5 + 15 PERF-SIGN 2.57 11.38 41.5 + 15 shuffled 4.44 13.43 30.2 + 20 naive 7.35 18.96 18.9 + 20 TM-med 9.10 20.33 11.7 + 20 TM-hit 7.27 18.22 16.4 + 20 TM-side 7.77 18.48 15.8 + 20 TM-x0.25 7.33 19.22 18.9 + 20 TM-x0.5 7.26 18.58 17.8 + 20 PERF-SIGN 5.29 16.93 27.0 + 20 shuffled 7.61 20.85 15.8 + 25 naive 11.26 25.97 10.5 + 25 TM-med 13.50 30.78 9.4 + 25 TM-hit 10.81 25.06 10.2 + 25 TM-side 11.13 25.48 9.1 + 25 TM-x0.25 11.13 25.29 11.1 + 25 TM-x0.5 10.64 25.14 10.8 + 25 PERF-SIGN 7.49 17.37 13.9 + 25 shuffled 11.54 28.86 7.8 + 30 naive 15.01 33.20 7.9 + 30 TM-med 15.25 36.96 7.9 + 30 TM-hit 14.61 34.12 5.3 + 30 TM-side 14.53 33.14 5.3 + 30 TM-x0.25 15.10 33.00 7.0 + 30 TM-x0.5 14.50 33.25 5.9 + 30 PERF-SIGN 10.37 24.48 9.8 + 30 shuffled 17.36 38.99 4.2 + + round 7 (L= 1364) nEval/h = [395, 390, 385, 380] + h arm med p90 hit% + 15 naive 2.46 13.59 49.1 + 15 TM-med 5.33 12.88 36.2 + 15 TM-hit 5.90 14.89 33.9 + 15 TM-side 2.44 13.55 49.4 + 15 TM-x0.25 3.66 13.50 38.5 + 15 TM-x0.5 4.70 12.88 37.0 + 15 PERF-SIGN 1.93 11.59 59.0 + 15 shuffled 2.68 13.92 47.8 + 20 naive 5.01 21.20 34.4 + 20 TM-med 8.50 20.85 20.5 + 20 TM-hit 8.41 21.77 24.4 + 20 TM-side 4.82 21.36 35.1 + 20 TM-x0.25 5.52 20.92 24.9 + 20 TM-x0.5 7.35 20.36 23.6 + 20 PERF-SIGN 3.01 19.20 49.0 + 20 shuffled 5.38 21.36 38.2 + 25 naive 6.96 29.57 24.2 + 25 TM-med 12.04 29.71 14.0 + 25 TM-hit 11.75 29.61 18.4 + 25 TM-side 8.86 30.31 16.1 + 25 TM-x0.25 6.94 29.32 19.5 + 25 TM-x0.5 9.52 29.07 22.6 + 25 PERF-SIGN 5.06 27.22 41.3 + 25 shuffled 7.81 29.22 23.9 + 30 naive 10.37 35.68 23.4 + 30 TM-med 16.49 38.72 13.7 + 30 TM-hit 14.92 39.07 12.9 + 30 TM-side 11.80 39.61 13.4 + 30 TM-x0.25 9.43 37.27 15.5 + 30 TM-x0.5 10.13 37.74 17.4 + 30 PERF-SIGN 8.32 33.53 35.5 + 30 shuffled 10.82 35.67 27.6 + + round 8 (L= 1546) nEval/h = [449, 444, 439, 434] + h arm med p90 hit% + 15 naive 4.25 14.89 40.1 + 15 TM-med 5.51 16.01 31.4 + 15 TM-hit 5.57 15.51 32.7 + 15 TM-side 8.92 18.26 10.7 + 15 TM-x0.25 4.19 14.38 36.5 + 15 TM-x0.5 4.95 14.63 32.3 + 15 PERF-SIGN 4.98 10.49 32.7 + 15 shuffled 10.99 19.29 18.5 + 20 naive 7.75 23.03 33.8 + 20 TM-med 7.73 25.17 10.8 + 20 TM-hit 9.48 29.07 6.8 + 20 TM-side 11.88 28.52 8.3 + 20 TM-x0.25 6.88 21.78 20.0 + 20 TM-x0.5 7.40 22.06 9.9 + 20 PERF-SIGN 8.10 12.16 16.2 + 20 shuffled 12.43 28.17 5.6 + 25 naive 11.52 31.02 24.4 + 25 TM-med 11.97 33.75 8.4 + 25 TM-hit 13.73 35.77 8.0 + 25 TM-side 13.75 35.51 6.4 + 25 TM-x0.25 10.42 29.90 13.2 + 25 TM-x0.5 9.67 30.11 6.2 + 25 PERF-SIGN 8.21 18.65 10.5 + 25 shuffled 15.01 37.19 6.6 + 30 naive 14.47 39.88 17.1 + 30 TM-med 13.82 41.47 7.1 + 30 TM-hit 14.73 41.51 6.2 + 30 TM-side 18.06 44.33 3.9 + 30 TM-x0.25 12.20 38.44 13.4 + 30 TM-x0.5 12.11 38.29 6.9 + 30 PERF-SIGN 11.55 23.35 10.1 + 30 shuffled 16.57 44.29 6.0 + + round 9 (L= 1286) nEval/h = [371, 366, 361, 356] + h arm med p90 hit% + 15 naive 5.97 14.60 24.0 + 15 TM-med 5.56 13.46 27.0 + 15 TM-hit 5.37 13.80 26.7 + 15 TM-side 5.73 14.42 27.0 + 15 TM-x0.25 5.75 14.42 24.3 + 15 TM-x0.5 5.74 14.05 25.1 + 15 PERF-SIGN 3.97 12.60 35.0 + 15 shuffled 6.08 14.79 23.7 + 20 naive 9.31 21.51 15.8 + 20 TM-med 8.73 20.98 17.5 + 20 TM-hit 9.52 22.48 14.5 + 20 TM-side 9.96 22.77 12.6 + 20 TM-x0.25 9.05 21.15 17.5 + 20 TM-x0.5 9.10 21.12 16.1 + 20 PERF-SIGN 7.25 19.56 24.6 + 20 shuffled 9.25 21.24 19.4 + 25 naive 12.35 27.72 12.2 + 25 TM-med 11.16 27.17 8.3 + 25 TM-hit 13.17 28.72 9.1 + 25 TM-side 12.17 27.07 10.8 + 25 TM-x0.25 12.00 27.22 11.4 + 25 TM-x0.5 12.02 27.20 7.5 + 25 PERF-SIGN 10.42 26.19 19.1 + 25 shuffled 12.62 27.73 10.0 + 30 naive 14.72 31.71 11.0 + 30 TM-med 14.05 30.50 7.3 + 30 TM-hit 16.13 31.61 9.3 + 30 TM-side 16.31 32.14 7.0 + 30 TM-x0.25 14.69 31.28 8.7 + 30 TM-x0.5 15.72 31.27 4.8 + 30 PERF-SIGN 12.69 29.14 16.6 + 30 shuffled 14.78 32.29 7.9 + + round 10 (L= 1392) nEval/h = [403, 398, 393, 388] + h arm med p90 hit% + 15 naive 4.35 15.34 39.0 + 15 TM-med 4.67 14.34 28.8 + 15 TM-hit 4.13 14.57 34.7 + 15 TM-side 4.29 15.20 29.8 + 15 TM-x0.25 4.32 14.89 38.7 + 15 TM-x0.5 4.18 14.73 36.5 + 15 PERF-SIGN 2.35 13.34 50.4 + 15 shuffled 4.44 14.98 39.5 + 20 naive 7.44 22.49 28.9 + 20 TM-med 8.41 21.46 20.1 + 20 TM-hit 7.10 22.18 30.4 + 20 TM-side 7.19 22.61 27.9 + 20 TM-x0.25 7.34 22.29 28.4 + 20 TM-x0.5 7.22 22.18 28.6 + 20 PERF-SIGN 5.58 20.68 37.7 + 20 shuffled 7.55 23.14 17.6 + 25 naive 11.21 30.56 20.1 + 25 TM-med 11.67 30.44 13.5 + 25 TM-hit 10.51 29.88 20.9 + 25 TM-side 10.93 29.52 19.3 + 25 TM-x0.25 10.94 30.19 21.6 + 25 TM-x0.5 10.82 29.49 21.9 + 25 PERF-SIGN 9.71 29.06 29.3 + 25 shuffled 11.80 31.40 11.2 + 30 naive 15.74 37.20 17.8 + 30 TM-med 14.96 37.52 8.8 + 30 TM-hit 16.07 39.13 11.9 + 30 TM-side 15.74 37.35 16.8 + 30 TM-x0.25 13.88 36.78 10.6 + 30 TM-x0.5 13.44 37.54 10.6 + 30 PERF-SIGN 13.74 35.20 24.5 + 30 shuffled 16.69 41.02 3.9 + + +## PER-FIXTURE pooled (tr_drussgt_vs_modularbot_shield.jsonl) + h arm N med|err| p90|err| hit% miss% + 15 naive 3643 3.83 14.38 41.0 59.0 + 15 TM-med 3643 4.71 13.67 29.3 70.7 + 15 TM-hit 3643 4.47 14.37 35.0 65.0 + 15 TM-side 3643 4.66 14.50 35.1 64.9 + 15 TM-x0.0 3643 3.83 14.38 41.0 59.0 + 15 TM-x0.25 3643 3.75 13.92 38.3 61.7 + 15 TM-x0.5 3643 4.32 13.63 35.8 64.2 + 15 PERF-SIGN 3643 2.48 11.87 47.6 52.4 + 15 PS-shift0 3643 3.83 14.38 41.0 59.0 + 15 shuffled 3643 4.82 15.41 35.7 64.3 + 20 naive 3593 6.50 21.96 29.5 70.5 + 20 TM-med 3593 7.63 21.39 17.7 82.3 + 20 TM-hit 3593 8.00 21.89 21.5 78.5 + 20 TM-side 3593 7.96 22.26 23.2 76.8 + 20 TM-x0.0 3593 6.50 21.96 29.5 70.5 + 20 TM-x0.25 3593 6.27 21.22 25.1 74.9 + 20 TM-x0.5 3593 6.86 21.05 23.0 77.0 + 20 PERF-SIGN 3593 5.00 18.74 34.6 65.4 + 20 PS-shift0 3593 6.50 21.96 29.5 70.5 + 20 shuffled 3593 7.92 22.91 21.2 78.8 + 25 naive 3543 9.68 29.41 20.1 79.9 + 25 TM-med 3543 10.94 29.71 11.8 88.2 + 25 TM-hit 3543 11.21 28.74 14.5 85.5 + 25 TM-side 3543 10.88 29.50 13.0 87.0 + 25 TM-x0.0 3543 9.68 29.41 20.1 79.9 + 25 TM-x0.25 3543 9.14 28.62 17.0 83.0 + 25 TM-x0.5 3543 9.30 28.15 15.1 84.9 + 25 PERF-SIGN 3543 7.51 25.11 25.0 75.0 + 25 PS-shift0 3543 9.68 29.41 20.1 79.9 + 25 shuffled 3543 11.39 30.95 13.1 86.9 + 30 naive 3493 12.97 36.08 15.7 84.3 + 30 TM-med 3493 13.78 36.34 9.7 90.3 + 30 TM-hit 3493 14.23 36.47 9.4 90.6 + 30 TM-side 3493 13.68 37.39 10.0 90.0 + 30 TM-x0.0 3493 12.97 36.08 15.7 84.3 + 30 TM-x0.25 3493 12.20 35.70 12.9 87.1 + 30 TM-x0.5 3493 11.91 35.68 10.2 89.8 + 30 PERF-SIGN 3493 9.61 30.76 18.2 81.8 + 30 PS-shift0 3493 12.97 36.08 15.7 84.3 + 30 shuffled 3493 14.36 38.36 10.0 90.0 + +## PER-FIXTURE learnability (tr_drussgt_vs_modularbot_shield.jsonl) + h N TMquad% shufquad% TMside% majQuad% pred[Ls Lb Rs Rb] / true[Ls Lb Rs Rb] + 15 3643 35.5 24.1 61.7 26.3 [1057 1007 549 1030 ] / [959 887 906 891 ] + 20 3593 34.0 25.6 61.1 25.4 [1008 1037 537 1011 ] / [913 893 898 889 ] + 25 3543 33.7 23.4 60.9 25.6 [981 1000 564 998 ] / [860 908 870 905 ] + 30 3493 32.5 26.5 60.8 26.1 [938 1032 541 982 ] / [810 912 869 902 ] + +========================================================================================== + +## POOLED PRIMARY (tr_drussgt_vs_modularbot*): arm table + h arm N med|err| p90|err| hit% miss% + 15 naive 9433 4.08 14.24 39.3 60.7 + 15 TM-med 9433 4.81 13.53 29.0 71.0 + 15 TM-hit 9433 4.46 14.29 35.1 64.9 + 15 TM-side 9433 4.40 14.12 36.0 64.0 + 15 TM-x0.0 9433 4.08 14.24 39.3 60.7 + 15 TM-x0.25 9433 3.99 13.84 37.9 62.1 + 15 TM-x0.5 9433 4.30 13.63 36.3 63.7 + 15 PERF-SIGN 9433 2.57 11.88 47.0 53.0 + 15 PS-shift0 9433 4.08 14.24 39.3 60.7 + 15 shuffled 9433 4.68 14.97 35.1 64.9 + 20 naive 9308 6.81 21.60 28.2 71.8 + 20 TM-med 9308 7.61 20.85 19.2 80.8 + 20 TM-hit 9308 7.75 21.62 22.8 77.2 + 20 TM-side 9308 7.36 21.96 24.8 75.2 + 20 TM-x0.0 9308 6.81 21.60 28.2 71.8 + 20 TM-x0.25 9308 6.58 21.08 25.6 74.4 + 20 TM-x0.5 9308 6.85 20.82 23.9 76.1 + 20 PERF-SIGN 9308 5.22 18.50 33.4 66.6 + 20 PS-shift0 9308 6.81 21.60 28.2 71.8 + 20 shuffled 9308 7.81 22.33 22.0 78.0 + 25 naive 9183 9.71 29.13 20.9 79.1 + 25 TM-med 9183 10.33 28.63 13.1 86.9 + 25 TM-hit 9183 10.22 28.90 15.4 84.6 + 25 TM-side 9183 10.12 29.21 15.3 84.7 + 25 TM-x0.0 9183 9.71 29.13 20.9 79.1 + 25 TM-x0.25 9183 9.31 28.61 18.8 81.2 + 25 TM-x0.5 9183 9.31 28.29 17.4 82.6 + 25 PERF-SIGN 9183 7.30 25.26 25.6 74.4 + 25 PS-shift0 9183 9.71 29.13 20.9 79.1 + 25 shuffled 9183 10.96 30.19 16.0 84.0 + 30 naive 9058 12.62 36.12 16.4 83.6 + 30 TM-med 9058 13.09 35.17 10.5 89.5 + 30 TM-hit 9058 13.27 36.28 11.4 88.6 + 30 TM-side 9058 12.94 36.88 12.3 87.7 + 30 TM-x0.0 9058 12.62 36.12 16.4 83.6 + 30 TM-x0.25 9058 12.12 35.75 14.1 85.9 + 30 TM-x0.5 9058 11.93 35.63 12.1 87.9 + 30 PERF-SIGN 9058 9.72 30.82 18.3 81.7 + 30 PS-shift0 9058 12.62 36.12 16.4 83.6 + 30 shuffled 9058 13.90 38.14 11.2 88.8 + +## DELTAS (percentage points of estimated hit fraction) + h naive% dTMmed dTMhit dTMside dTMs0 dTMs25 dTMs50 dPS dPS0 dShuf (hit pp vs naive) + 15 39.3 -10.3 -4.2 -3.3 +0.0 -1.5 -3.0 +7.7 +0.0 -4.2 + 20 28.2 -9.0 -5.4 -3.4 +0.0 -2.6 -4.3 +5.1 +0.0 -6.2 + 25 20.9 -7.8 -5.5 -5.6 +0.0 -2.2 -3.6 +4.7 +0.0 -4.9 + 30 16.4 -5.9 -5.0 -4.1 +0.0 -2.3 -4.3 +1.9 +0.0 -5.2 + +## POOLED learnability diagnostics (is the learner learning?) + h N TMquad% shufquad% TMside% majQuad% pred[Ls Lb Rs Rb] / true[Ls Lb Rs Rb] + 15 9433 35.2 22.9 59.6 25.5 [2294 2615 1447 3077 ] / [2407 2314 2377 2335 ] + 20 9308 34.8 23.6 59.8 25.2 [2281 2579 1386 3062 ] / [2305 2331 2343 2329 ] + 25 9183 34.3 23.8 59.7 25.3 [2208 2558 1366 3051 ] / [2228 2322 2315 2318 ] + 30 9058 34.3 24.8 61.0 25.6 [2192 2501 1375 2990 ] / [2150 2273 2321 2314 ] + +## FITTED HIT-OPTIMAL SHIFT MAGNITUDES + (median across rounds of the fitted HIT-OPTIMAL shift, deg) + h TM[Ls Lb Rs Rb] PS[left right] shuf[Ls Lb Rs Rb] + 15 [ 0.00 1.50 0.00 -1.00 ] [ 2.00 -2.00 ] [ 0.00 0.00 0.00 -0.50 ] + 20 [ 0.00 2.00 -0.50 -1.00 ] [ 2.00 -2.00 ] [ 0.00 0.00 -0.50 0.00 ] + 25 [ 0.50 5.50 0.00 -2.71 ] [ 2.50 -2.50 ] [ 0.00 0.00 1.00 1.50 ] + 30 [ 1.50 8.00 0.00 -5.00 ] [ 3.50 -3.00 ] [ 0.00 1.00 1.00 1.50 ] + +## HIT FRACTION vs CONSTANT SHIFT MAGNITUDE (calibration) + shift h=15 h=20 h=25 h=30 (hit% on calibration, pooled) + -20.0 1.2 4.0 4.2 4.4 + -18.0 2.1 4.5 4.7 4.8 + -16.0 3.4 5.1 5.7 5.2 + -14.0 5.1 6.3 5.9 5.7 + -12.0 6.6 7.4 6.4 5.6 + -10.0 8.4 8.1 7.2 6.2 + -8.0 10.2 8.6 7.6 6.4 + -6.0 11.3 8.9 7.6 6.3 + -4.0 14.2 11.8 10.2 7.8 + -2.0 26.8 19.2 14.4 10.9 + +0.0 36.2 25.3 18.0 12.8 + +2.0 27.4 19.9 15.4 12.9 + +4.0 14.1 12.5 11.3 11.0 + +6.0 11.7 9.2 9.0 8.5 + +8.0 10.7 8.6 7.7 7.2 + +10.0 8.3 8.0 6.8 5.7 + +12.0 6.4 7.3 6.2 5.4 + +14.0 4.7 6.4 5.8 5.5 + +16.0 3.6 5.6 5.6 5.1 + +18.0 2.1 4.6 5.6 5.2 + +20.0 1.1 3.7 4.8 4.8 + +## SIDE-ACCURACY SWEEP (synthetic q-accurate predictor, out-of-sample) + q fitShift(h=20)[L R] hit% h15 h20 h25 h30 d(hit) pp: h15 h20 h25 h30 mean + 0.50 [ 0.00 -0.50] 37.8 25.1 16.8 11.5 -1.5 -3.1 -4.1 -4.9 -3.4 + 0.55 [ 0.50 0.00] 37.6 25.8 16.6 11.5 -1.7 -2.4 -4.3 -4.8 -3.3 + 0.60 [ 0.50 -0.50] 38.0 25.7 17.0 12.0 -1.3 -2.5 -3.9 -4.3 -3.0 + 0.65 [ 1.00 -1.00] 39.2 25.1 17.7 12.6 -0.1 -3.1 -3.3 -3.8 -2.6 + 0.70 [ 1.50 -1.00] 39.4 26.5 19.0 13.6 +0.1 -1.8 -2.0 -2.8 -1.6 + 0.80 [ 1.50 -1.50] 40.5 28.5 20.7 15.0 +1.2 +0.2 -0.3 -1.4 -0.0 + 1.00 [ 2.00 -2.00] 47.0 33.4 25.6 18.3 +7.7 +5.1 +4.7 +1.9 +4.9 + +========================================================================================== +## PER-ROUND hit-delta integrity (pp vs naive) +fixture round h TMmed TMhit TMside TMs25 TMs50 PS shuf +tr_drussgt_vs_modularbot.jsonl 1 15 -0.3 -5.8 -1.7 +0.3 -0.8 +16.1 -4.7 +tr_drussgt_vs_modularbot.jsonl 1 20 +3.1 +2.2 +1.1 +4.2 +5.1 +20.2 +1.7 +tr_drussgt_vs_modularbot.jsonl 1 25 -4.0 -2.3 -1.1 +4.6 +4.0 +18.8 -0.9 +tr_drussgt_vs_modularbot.jsonl 1 30 -6.6 -6.1 -5.2 -0.3 -0.9 +7.2 +0.6 +tr_drussgt_vs_modularbot.jsonl 2 15 -19.0 -7.3 -2.4 -5.1 -8.5 +7.1 -5.1 +tr_drussgt_vs_modularbot.jsonl 2 20 -10.4 -2.5 +1.0 -4.0 -4.7 -2.5 -9.9 +tr_drussgt_vs_modularbot.jsonl 2 25 -7.0 -10.2 -3.5 -8.2 -7.8 -3.5 -2.5 +tr_drussgt_vs_modularbot.jsonl 2 30 -8.1 -7.8 -6.1 -4.6 -5.8 -6.8 -14.4 +tr_drussgt_vs_modularbot.jsonl 3 15 +0.4 -1.3 -2.9 +0.4 +0.0 +5.9 -1.7 +tr_drussgt_vs_modularbot.jsonl 3 20 -3.0 -4.3 -4.3 -0.4 -2.1 +8.5 +0.4 +tr_drussgt_vs_modularbot.jsonl 3 25 -7.9 -13.1 -1.7 -3.5 -7.4 +14.8 +0.4 +tr_drussgt_vs_modularbot.jsonl 3 30 +2.2 +1.3 -2.2 -4.5 -2.2 +4.5 -2.2 +tr_drussgt_vs_modularbot.jsonl 4 15 +0.5 -1.9 +0.5 +2.6 +2.1 +9.8 -4.3 +tr_drussgt_vs_modularbot.jsonl 4 20 -1.2 +1.7 +2.4 +2.7 +1.9 +15.5 +0.5 +tr_drussgt_vs_modularbot.jsonl 4 25 +1.2 -2.0 +0.5 -1.2 +1.0 +10.0 +2.0 +tr_drussgt_vs_modularbot.jsonl 4 30 +1.7 +4.2 +2.5 +0.2 +1.5 +7.4 -0.7 +tr_drussgt_vs_modularbot.jsonl 5 15 +0.7 -0.5 -2.5 +0.0 +0.2 +9.8 -2.0 +tr_drussgt_vs_modularbot.jsonl 5 20 -3.1 -0.9 -2.0 +0.4 -0.4 +7.7 -4.7 +tr_drussgt_vs_modularbot.jsonl 5 25 -10.1 -3.7 -5.5 -0.6 -0.4 +9.4 -2.6 +tr_drussgt_vs_modularbot.jsonl 5 30 -5.8 -0.7 -2.2 -1.5 -0.7 +8.9 -3.0 +tr_drussgt_vs_modularbot.jsonl 6 15 -3.2 +0.3 +0.0 +0.3 +0.3 +6.8 -1.0 +tr_drussgt_vs_modularbot.jsonl 6 20 +1.0 -0.3 +1.0 -1.0 -1.0 +6.6 +0.0 +tr_drussgt_vs_modularbot.jsonl 6 25 +3.3 -0.7 -0.7 +1.7 +2.0 +5.4 +1.0 +tr_drussgt_vs_modularbot.jsonl 6 30 +7.1 -1.0 -1.0 +0.7 +0.0 +4.8 -0.7 +tr_drussgt_vs_modularbot.jsonl 7 15 -6.4 -4.5 +0.2 -3.1 -3.1 +9.2 -8.7 +tr_drussgt_vs_modularbot.jsonl 7 20 -10.3 -2.4 +0.2 -1.0 -1.2 +10.7 -4.1 +tr_drussgt_vs_modularbot.jsonl 7 25 -7.7 +0.7 +0.2 +0.7 +1.0 +7.5 -3.1 +tr_drussgt_vs_modularbot.jsonl 7 30 -3.7 -1.2 -1.7 +0.5 -1.7 +8.6 +0.2 +tr_drussgt_vs_modularbot.jsonl 8 15 +1.4 -0.3 -1.7 -3.2 +0.3 +10.6 -0.3 +tr_drussgt_vs_modularbot.jsonl 8 20 +1.5 +1.2 -0.6 -2.3 -0.6 +7.6 +2.6 +tr_drussgt_vs_modularbot.jsonl 8 25 -4.4 -4.1 -3.8 -4.7 -5.0 +3.8 +1.8 +tr_drussgt_vs_modularbot.jsonl 8 30 -4.2 -2.7 -4.5 -2.1 -4.2 +3.3 -6.3 +tr_drussgt_vs_modularbot.jsonl 9 15 -29.4 +0.0 -1.2 +0.2 +0.4 +10.7 -1.9 +tr_drussgt_vs_modularbot.jsonl 9 20 -25.2 -28.1 -3.1 -11.2 -25.8 -3.7 -10.6 +tr_drussgt_vs_modularbot.jsonl 9 25 -25.0 -26.2 -26.4 -8.3 -20.9 -8.7 -16.1 +tr_drussgt_vs_modularbot.jsonl 9 30 -19.7 -22.1 -13.1 -7.8 -18.5 -16.5 -12.2 +tr_drussgt_vs_modularbot.jsonl 10 15 -12.5 -7.5 -9.0 +0.8 -5.2 -3.0 -2.2 +tr_drussgt_vs_modularbot.jsonl 10 20 -6.1 -3.0 -6.3 +2.3 +1.8 -2.3 -0.5 +tr_drussgt_vs_modularbot.jsonl 10 25 -7.2 -1.5 +1.0 +3.8 +3.1 -0.8 -3.8 +tr_drussgt_vs_modularbot.jsonl 10 30 -4.2 +0.8 +0.0 +1.6 +1.8 +7.8 -1.6 +tr_drussgt_vs_modularbot.jsonl 11 15 -1.5 -0.3 +0.3 -4.6 -2.8 +6.5 -6.8 +tr_drussgt_vs_modularbot.jsonl 11 20 -3.4 -2.8 -4.7 -2.8 +0.3 +2.8 -5.6 +tr_drussgt_vs_modularbot.jsonl 11 25 -3.5 -2.5 -3.5 -2.5 -2.5 +0.3 -2.2 +tr_drussgt_vs_modularbot.jsonl 11 30 -0.3 -2.3 -1.6 -3.6 -4.5 +1.0 -4.2 +tr_drussgt_vs_modularbot.jsonl 12 15 -0.6 -1.7 -1.7 +0.0 -0.6 +9.4 -6.3 +tr_drussgt_vs_modularbot.jsonl 12 20 +0.6 -2.5 +0.0 +0.6 -0.6 +5.3 -6.1 +tr_drussgt_vs_modularbot.jsonl 12 25 -0.8 -7.1 -2.3 +1.7 +1.4 +5.4 -2.5 +tr_drussgt_vs_modularbot.jsonl 12 30 -9.8 +2.6 -8.9 +2.0 +2.9 +3.7 -9.2 +tr_drussgt_vs_modularbot.jsonl 13 15 -16.6 -12.7 +0.3 -0.3 -5.7 +6.8 -5.5 +tr_drussgt_vs_modularbot.jsonl 13 20 -8.2 -8.4 -8.2 -9.5 -8.2 +10.5 -2.9 +tr_drussgt_vs_modularbot.jsonl 13 25 -3.2 -5.3 +0.5 -4.5 -4.8 +6.4 +1.9 +tr_drussgt_vs_modularbot.jsonl 13 30 -1.4 -3.0 +2.7 -3.0 -2.2 +8.4 +0.0 +tr_drussgt_vs_modularbot.jsonl 14 15 -19.2 +0.5 +0.2 +0.0 +0.0 +8.2 -0.5 +tr_drussgt_vs_modularbot.jsonl 14 20 -13.5 +3.5 +2.6 +0.9 +0.9 +7.6 -5.2 +tr_drussgt_vs_modularbot.jsonl 14 25 -12.4 -1.2 -1.0 -0.2 +0.5 -1.9 -12.7 +tr_drussgt_vs_modularbot.jsonl 14 30 -10.7 -12.8 +1.0 -3.9 -11.1 -6.3 -6.3 +tr_drussgt_vs_modularbot.jsonl 15 15 -26.8 -3.5 -4.5 +0.6 -2.9 +10.5 -1.0 +tr_drussgt_vs_modularbot.jsonl 15 20 -22.7 -3.9 -4.9 +1.9 -2.3 -18.5 -26.6 +tr_drussgt_vs_modularbot.jsonl 15 25 -15.2 +2.3 -16.8 -0.7 -0.7 +8.9 -7.6 +tr_drussgt_vs_modularbot.jsonl 15 30 -13.4 -3.0 -2.0 -1.3 -1.0 -8.7 -9.4 +tr_drussgt_vs_modularbot_shield.jsonl 1 15 +0.0 +0.0 +0.0 +0.0 +0.0 +0.0 +0.0 +tr_drussgt_vs_modularbot_shield.jsonl 1 20 +0.0 +0.0 +0.0 +0.0 +0.0 +0.0 +0.0 +tr_drussgt_vs_modularbot_shield.jsonl 1 25 +0.0 -2.6 -2.6 +0.0 -2.0 +0.0 -2.6 +tr_drussgt_vs_modularbot_shield.jsonl 1 30 +0.0 +0.0 +0.0 +0.0 +0.0 +1.4 +0.0 +tr_drussgt_vs_modularbot_shield.jsonl 2 15 -18.7 -8.1 +1.9 +1.0 -1.6 +10.0 -5.8 +tr_drussgt_vs_modularbot_shield.jsonl 2 20 -11.1 -2.3 +2.0 +1.3 +1.0 +3.3 -4.6 +tr_drussgt_vs_modularbot_shield.jsonl 2 25 -10.7 -9.7 -5.0 -5.3 -10.7 -2.3 -12.3 +tr_drussgt_vs_modularbot_shield.jsonl 2 30 -8.8 -9.5 -1.7 -8.5 -10.5 +0.7 -3.1 +tr_drussgt_vs_modularbot_shield.jsonl 3 15 -24.4 -24.4 -16.5 -9.2 -23.8 +2.2 -16.8 +tr_drussgt_vs_modularbot_shield.jsonl 3 20 -22.4 -22.2 -22.2 -20.2 -22.2 +5.7 -15.1 +tr_drussgt_vs_modularbot_shield.jsonl 3 25 -13.0 -10.1 -12.7 -8.1 -12.7 +11.8 -15.3 +tr_drussgt_vs_modularbot_shield.jsonl 3 30 -8.5 -8.5 -7.9 +0.9 -8.2 -7.0 -10.5 +tr_drussgt_vs_modularbot_shield.jsonl 4 15 -27.8 -0.3 -1.1 +0.3 -0.8 +7.0 -5.6 +tr_drussgt_vs_modularbot_shield.jsonl 4 20 -26.3 -4.1 -3.3 +1.1 +1.9 +3.0 -20.1 +tr_drussgt_vs_modularbot_shield.jsonl 4 25 -18.7 -8.0 -23.1 +0.0 -0.5 +4.7 -14.6 +tr_drussgt_vs_modularbot_shield.jsonl 4 30 -8.4 -12.5 -16.2 +0.6 -3.3 -1.4 -14.2 +tr_drussgt_vs_modularbot_shield.jsonl 5 15 +0.2 -1.1 +0.4 -2.7 -2.4 +9.1 +1.6 +tr_drussgt_vs_modularbot_shield.jsonl 5 20 -0.7 -4.5 -0.7 +0.0 -2.5 +15.1 -1.1 +tr_drussgt_vs_modularbot_shield.jsonl 5 25 +0.9 +0.7 +4.1 -0.5 +0.2 +10.2 +7.5 +tr_drussgt_vs_modularbot_shield.jsonl 5 30 +0.2 +1.6 +2.5 +1.1 +0.7 +9.7 +0.9 +tr_drussgt_vs_modularbot_shield.jsonl 6 15 -15.6 +0.5 -0.5 +0.8 +1.6 +11.6 +0.3 +tr_drussgt_vs_modularbot_shield.jsonl 6 20 -7.1 -2.5 -3.0 +0.0 -1.1 +8.2 -3.0 +tr_drussgt_vs_modularbot_shield.jsonl 6 25 -1.1 -0.3 -1.4 +0.6 +0.3 +3.3 -2.8 +tr_drussgt_vs_modularbot_shield.jsonl 6 30 +0.0 -2.5 -2.5 -0.8 -2.0 +2.0 -3.7 +tr_drussgt_vs_modularbot_shield.jsonl 7 15 -12.9 -15.2 +0.3 -10.6 -12.2 +9.9 -1.3 +tr_drussgt_vs_modularbot_shield.jsonl 7 20 -13.8 -10.0 +0.8 -9.5 -10.8 +14.6 +3.8 +tr_drussgt_vs_modularbot_shield.jsonl 7 25 -10.1 -5.7 -8.1 -4.7 -1.6 +17.1 -0.3 +tr_drussgt_vs_modularbot_shield.jsonl 7 30 -9.7 -10.5 -10.0 -7.9 -6.1 +12.1 +4.2 +tr_drussgt_vs_modularbot_shield.jsonl 8 15 -8.7 -7.3 -29.4 -3.6 -7.8 -7.3 -21.6 +tr_drussgt_vs_modularbot_shield.jsonl 8 20 -23.0 -27.0 -25.5 -13.7 -23.9 -17.6 -28.2 +tr_drussgt_vs_modularbot_shield.jsonl 8 25 -15.9 -16.4 -18.0 -11.2 -18.2 -13.9 -17.8 +tr_drussgt_vs_modularbot_shield.jsonl 8 30 -9.9 -10.8 -13.1 -3.7 -10.1 -6.9 -11.1 +tr_drussgt_vs_modularbot_shield.jsonl 9 15 +3.0 +2.7 +3.0 +0.3 +1.1 +11.1 -0.3 +tr_drussgt_vs_modularbot_shield.jsonl 9 20 +1.6 -1.4 -3.3 +1.6 +0.3 +8.7 +3.6 +tr_drussgt_vs_modularbot_shield.jsonl 9 25 -3.9 -3.0 -1.4 -0.8 -4.7 +6.9 -2.2 +tr_drussgt_vs_modularbot_shield.jsonl 9 30 -3.7 -1.7 -3.9 -2.2 -6.2 +5.6 -3.1 +tr_drussgt_vs_modularbot_shield.jsonl 10 15 -10.2 -4.2 -9.2 -0.2 -2.5 +11.4 +0.5 +tr_drussgt_vs_modularbot_shield.jsonl 10 20 -8.8 +1.5 -1.0 -0.5 -0.3 +8.8 -11.3 +tr_drussgt_vs_modularbot_shield.jsonl 10 25 -6.6 +0.8 -0.8 +1.5 +1.8 +9.2 -8.9 +tr_drussgt_vs_modularbot_shield.jsonl 10 30 -9.0 -5.9 -1.0 -7.2 -7.2 +6.7 -13.9 + +========================================================================================== +## VERDICT INPUTS (pooled hit pp vs naive) +h= 15 PERF-SIGN +7.7 pp TM-side -3.3 pp TM-quad -4.2 pp shuffled -4.2 pp +h= 20 PERF-SIGN +5.1 pp TM-side -3.4 pp TM-quad -5.4 pp shuffled -6.2 pp +h= 25 PERF-SIGN +4.7 pp TM-side -5.6 pp TM-quad -5.5 pp shuffled -4.9 pp +h= 30 PERF-SIGN +1.9 pp TM-side -4.1 pp TM-quad -5.0 pp shuffled -5.2 pp diff --git a/common_libs/tests/measure_tm_hit_optimal.nim b/common_libs/tests/measure_tm_hit_optimal.nim new file mode 100644 index 0000000..d6de2b7 --- /dev/null +++ b/common_libs/tests/measure_tm_hit_optimal.nim @@ -0,0 +1,1054 @@ +## GATE 2b — is the "shift the aim per predicted bucket" application form +## STRUCTURALLY DEAD, or merely MISCALIBRATED? OFFLINE ONLY. +## +## Gate 2 (fd2f7f6) calibrated each quadrant to the conditional MEDIAN error and +## LOST hits everywhere. The gate metric is HITS, not median error, so the +## calibration objective was wrong. This file changes ONLY the calibration / +## appliance step: each bucket's offset is the shift (fine grid, -20..+20 deg in +## 0.5 deg steps) that MAXIMISES THE HIT COUNT on the out-of-sample calibration +## slice. It adds the DECISIVE arm: PERFECT-SIGN (cheating, the true sign) with +## the hit-optimal shift. If even perfect side knowledge cannot beat naive on +## hits, the application form is structurally dead. +## +## Original Gate 2 pipeline (unchanged): extract the draft 49-bit spec + 4-bit +## horizon, derive quadrant labels, train tm_core fresh per round, split each +## round train|calibrate|eval, measure residual angular error + hit fraction. +## +## Pipeline (all offline, fixtures READ-ONLY): +## 1. Extract the DRAFT 49-bit TM feature spec (walls / us / motion / bullets) +## from the committed DrussGT fixtures, plus a 4-bit one-hot horizon block +## (h in {15,20,25,30}) -> 53 raw bits. +## 2. Build FACT labels: for a sample at tick t and horizon h, look up where +## the enemy ACTUALLY was at t+h (never across a round boundary; the last h +## ticks of each round are dropped). Two binaries: +## (a) side: enemy LEFT / RIGHT of the naive straight-line guess, +## (b) magnitude: |angular error| bigger / smaller than the TRAIN median. +## Four quadrants: left-small / left-big / right-small / right-big. +## 3. Train a real Tsetlin Machine with the validated core at +## `common_libs/tm_diag/tm_core.nim` (one fresh model per round = the +## intended "fresh every round, overfit the current enemy" semantics). +## 4. Map each predicted quadrant to a representative signed angular offset +## (median signed error of the TRAINING samples in that quadrant), apply it +## to the naive aim, and measure the residual angular error. +## +## Four arms + floor: +## naive : straight-line guess (baseline) +## TM : trained model +## shuffled : same pipeline, labels randomised (pipeline-integrity control) +## turn-only : uses ONLY the enemy's current turn direction (critical arm) +## majority : constant majority-quadrant offset (floor) +## +## Metric: median / p90 residual |angular error| (deg) and the estimated hit +## fraction (|residual| < atan(18px / range)). The absolute hit fraction is +## OPTIMISTIC (perfect arrival knowledge every tick = bmPoint-style); only the +## DELTA before/after is meaningful. +## +## Protocol: within-round split, train on the EARLY portion, evaluate on the +## LATER portion. Focus horizons h = 15,20,25 (+30). +## +## Run: nim c -r -d:release --path:common_libs \ +## common_libs/tests/measure_tm_miss_shrink.nim +## [--fixtures=a,b] [--epochs=N] [--trainfrac=F] [--clauses=N] + +import std/[json, os, strformat, strutils, math, algorithm, tables] +import tm_diag/tm_core + +# ── configuration ──────────────────────────────────────────────────────────── + +const repoRoot* = currentSourcePath().parentDir.parentDir.parentDir +const fixturesDir* = repoRoot / "tools" / "fixtures" +const metaDir* = fixturesDir / "drussgt_meta" + +const HORIZONS = [15, 20, 25, 30] +const NH = 4 +const N_BASE = 49 # draftTMSpec() bit count +const N_BITS = N_BASE + 4 # + 4-bit horizon one-hot +const N_CLASSES = 4 # left-small, left-big, right-small, right-big +const MIN_I = 12 # need 10 ticks of history for the motion features +const BOT_RADIUS = 18.0 + +## Hit-optimal shift search grid: -20..+20 deg in 0.5 deg steps. +const SHIFT_LO = -20.0 +const SHIFT_HI = 20.0 +const SHIFT_STEP = 0.5 +const NSHIFT = int((SHIFT_HI - SHIFT_LO) / SHIFT_STEP) + 1 # 81 + +## Hits-vs-shift-magnitude diagnostic curve: -20..+20 deg in 2 deg steps. +const NCURVE = 21 +proc curveShift(k: int): float {.inline.} = -20.0 + 2.0 * float(k) + +## Side-accuracy sweep: a synthetic side predictor that is correct with +## probability q (independent draws on calibration and eval). +const NQ = 7 +const ACCS = [0.50, 0.55, 0.60, 0.65, 0.70, 0.80, 1.00] + +proc shiftGrid(): seq[float] = + for k in 0.. 180.0: r -= 360.0 + while r <= -180.0: r += 360.0 + r + +proc signf(x: float): int {.inline.} = + if x > 1e-9: 1 elif x < -1e-9: -1 else: 0 + +proc jf(d: JsonNode, k: string): float = + let n = d[k] + case n.kind + of JFloat: n.getFloat + of JInt: float(n.getInt) + else: parseFloat(n.getStr) + +# ── data model ─────────────────────────────────────────────────────────────── + +type + Tick* = object + tick*: int + ex*, ey*, eh*, es*, ee*: float + sx*, sy*, sh*, ss*, se*: float + + Rnd* = object + roundNo*: int + st*: seq[Tick] + + BulletSeries* = object + ## Per-tick proxy for OUR in-flight bullets (INFERRED from self-energy + ## drops; the fixture records no gun heading/power). + tta*: seq[int] ## ticks until the nearest in-flight bullet arrives, -1 none + lat*: seq[float] ## lateral offset of the enemy from the fired path (px) + + ## Arms: + ## aNaive : straight-line, no shift (baseline) + ## aTMmed : OLD gate-2 appliance (conditional median per predicted quadrant) + ## aTMhit : TM quadrant -> HIT-OPTIMAL shift (shrink x1.0) + ## aTMside : TM predicted SIDE (quadrant collapsed) -> hit-optimal shift + ## aTMs0 : TM bucket -> hit-optimal shift x0.0 (== naive, sanity) + ## aTMs25 : TM bucket -> hit-optimal shift x0.25 + ## aTMs50 : TM bucket -> hit-optimal shift x0.50 + ## aPS : PERFECT-SIGN -> hit-optimal shift (DECISIVE, cheating) + ## aPS0 : PERFECT-SIGN, shift 0 (== naive, sanity) + ## aShuf : shuffled-label model -> hit-optimal shift (integrity control) + Arm = enum aNaive, aTMmed, aTMhit, aTMside, aTMs0, aTMs25, aTMs50, + aPS, aPS0, aShuf + Hist = object + errs: array[Arm, array[NH, seq[float]]] + hits: array[Arm, array[NH, int]] + n: array[NH, int] + # diagnostics + tmQCor: array[NH, int] ## TM predicted the true quadrant + tmQTot: array[NH, int] + tmHitCor: array[NH, int] ## TM predicted the true sign + tmHitTot: array[NH, int] + shufQCor: array[NH, int] ## shuffled-model predicted the true quadrant + shufQTot: array[NH, int] + predCounts: array[NH, array[N_CLASSES, int]] + trueCounts: array[NH, array[N_CLASSES, int]] + # fitted hit-optimal shifts (diagnostic) + fitTM: array[NH, array[N_CLASSES, float]] + fitTMSide: array[NH, array[2, float]] + fitShuf: array[NH, array[N_CLASSES, float]] + fitPS: array[NH, array[2, float]] + # hits-vs-constant-shift curve, pooled calibration samples per horizon + curveHits: array[NH, array[NCURVE, int]] + curveN: array[NH, int] + # side-accuracy sweep: hits on eval for a synthetic q-accurate predictor + accHits: array[NQ, array[NH, int]] + accN: array[NQ, array[NH, int]] + accFit: array[NQ, array[NH, array[2, float]]] + +# ── fixture loading ────────────────────────────────────────────────────────── + +proc loadTicks(path: string): seq[Tick] = + for line in lines(path): + let ln = line.strip() + if ln.len == 0: continue + let d = parseJson(ln) + if not d.hasKey("tick"): continue + result.add Tick(tick: d["tick"].getInt, + ex: jf(d, "ex"), ey: jf(d, "ey"), eh: jf(d, "eh"), + es: jf(d, "es"), ee: jf(d, "ee"), + sx: jf(d, "sx"), sy: jf(d, "sy"), sh: jf(d, "sh"), + ss: jf(d, "ss"), se: jf(d, "se")) + +proc loadRounds(path: string, ticks: seq[Tick]): seq[Rnd] = + let rp = metaDir / (extractFilename(path) & ".rounds.json") + var spans: seq[(int, int)] + if fileExists(rp): + let j = parseFile(rp) + for r in j["rounds"]: + spans.add (r["startTick"].getInt, r["count"].getInt) + elif ticks.len > 0: + spans.add (ticks[0].tick, ticks.len) + var idxByTick = initTable[int, int]() + for i, t in ticks: idxByTick[t.tick] = i + for sp in spans: + let (s0, c) = sp + if not idxByTick.hasKey(s0): continue + let i0 = idxByTick[s0] + var st: seq[Tick] + for k in 0.. 0: + result.add Rnd(roundNo: result.len + 1, st: st) + +# ── bullet proxy (INFERRED) ────────────────────────────────────────────────── + +proc buildBulletSeries(r: Rnd): BulletSeries = + let L = r.st.len + result.tta = newSeq[int](L) + for i in 0.. 3.1: continue # damage, not a fire + var power = drop + if power < 0.1: power = 0.1 + if power > 3.0: power = 3.0 + let speed = 20.0 - 3.0 * power + let rng = hypot(r.st[t0].ex - r.st[t0].sx, r.st[t0].ey - r.st[t0].sy) + let flight = int(ceil(rng / speed)) + let dx = r.st[t0].ex - r.st[t0].sx + let dy = r.st[t0].ey - r.st[t0].sy + let nrm = max(1e-6, hypot(dx, dy)) + let ux = dx / nrm + let uy = dy / nrm + for k in 0..flight: + let t = t0 + k + if t >= L: break + let ta = t0 + flight - t + if result.tta[t] < 0 or ta < result.tta[t]: + result.tta[t] = ta + let vx = r.st[t].ex - r.st[t0].sx + let vy = r.st[t].ey - r.st[t0].sy + result.lat[t] = ux * vy - uy * vx + +# ── feature extraction: the 49 draft bits (causal, at tick i) ──────────────── +# +# Block layout (mirrors draftTMSpec()): +# 0..3 dist-to-nearest-wall (4 one-hot) +# 4..7 which-wall-nearest (4 one-hot) +# 8..13 dist-from-us (6 one-hot) +# 14..16 enemy-heading-vs-line-to-us (3 one-hot) +# 17..19 turn-direction t, t-1, t-2 (3 boolean "was turning left") +# 20..24 ticks-since-reversal (5 one-hot) +# 25..27 turn-consistency-10 (3 one-hot) +# 28..30 distance-moved-10 (3 one-hot) +# 31..33 speed-trend-10 (3 one-hot) +# 34..36 turn-rate-change-5 (3 one-hot) +# 37..41 time-until-bullet (5 one-hot) +# 42..48 bullet-lateral-offset (7 one-hot) + +proc buildBase(r: Rnd, bi: BulletSeries, i: int, sinceRev: seq[int]): array[N_BASE, int] = + let s = r.st + let cur = s[i] + + # walls + let dL = cur.ex + let dR = 800.0 - cur.ex + let dT = 600.0 - cur.ey + let dBottom = cur.ey + let dmin = min(min(dL, dR), min(dT, dBottom)) + var wallBin = 3 + if dmin < 50.0: wallBin = 0 + elif dmin < 100.0: wallBin = 1 + elif dmin < 200.0: wallBin = 2 + result[wallBin] = 1 + var wb = 0 + let walls = [dL, dR, dT, dBottom] + for w in 1..3: + if walls[w] < walls[wb]: wb = w + result[4 + wb] = 1 + + # us + let rng = hypot(cur.ex - cur.sx, cur.ey - cur.sy) + var ub = 5 + if rng < 100.0: ub = 0 + elif rng < 200.0: ub = 1 + elif rng < 300.0: ub = 2 + elif rng < 400.0: ub = 3 + elif rng < 600.0: ub = 4 + result[8 + ub] = 1 + + let lane = arctan2(cur.sy - cur.ey, cur.sx - cur.ex) + let hdg = cur.eh * PI / 180.0 + let perp = abs(sin(hdg - lane)) + var hb = 1 + if perp < 0.5: hb = 2 + elif perp > 0.866: hb = 0 + result[14 + hb] = 1 + + # motion: turn direction + for k in 0..2: + if i - 1 - k >= 0: + let d = wrap180(s[i - k].eh - s[i - 1 - k].eh) + if d > 1e-6: result[17 + k] = 1 + + # ticks since reversal + var rb = 4 + let sr = sinceRev[i] + if sr < 5: rb = 0 + elif sr < 10: rb = 1 + elif sr < 20: rb = 2 + elif sr < 40: rb = 3 + result[20 + rb] = 1 + + # turn consistency over last 10 + var pos = 0 + var neg = 0 + for k in 0..9: + if i - 1 - k < 0: break + let d = wrap180(s[i - k].eh - s[i - 1 - k].eh) + if d > 1e-6: inc pos + elif d < -1e-6: inc neg + let tot = pos + neg + let cons = if tot > 0: max(pos, neg).float / tot.float else: 0.0 + var cb = 0 + if cons > 0.8: cb = 2 + elif cons >= 0.5: cb = 1 + result[25 + cb] = 1 + + # distance moved over 10 + let j0 = max(0, i - 10) + let dm = hypot(cur.ex - s[j0].ex, cur.ey - s[j0].ey) + var mb = 1 + if dm < 20.0: mb = 0 + elif dm > 50.0: mb = 2 + result[28 + mb] = 1 + + # speed trend over 10 + let st10 = abs(s[max(0, i - 10)].es) + let spdDiff = abs(cur.es) - st10 + var sb = 1 + if spdDiff < -0.5: sb = 0 + elif spdDiff > 0.5: sb = 2 + result[31 + sb] = 1 + + # turn-rate change: last 5 deltas vs previous 5 + var r1 = 0.0 + var n1 = 0 + for k in 0..4: + if i - 1 - k >= 0: + r1 += abs(wrap180(s[i - k].eh - s[i - 1 - k].eh)); inc n1 + var r2 = 0.0 + var n2 = 0 + for k in 5..9: + if i - 1 - k >= 0: + r2 += abs(wrap180(s[i - k].eh - s[i - 1 - k].eh)); inc n2 + let m1 = if n1 > 0: r1 / float(n1) else: 0.0 + let m2 = if n2 > 0: r2 / float(n2) else: 0.0 + let dtr = m1 - m2 + var tb = 1 + if dtr < -0.3: tb = 0 + elif dtr > 0.3: tb = 2 + result[34 + tb] = 1 + + # bullets + let tta = bi.tta[i] + var b1 = 0 + if tta >= 0: + if tta < 5: b1 = 1 + elif tta < 10: b1 = 2 + elif tta < 20: b1 = 3 + else: b1 = 4 + result[37 + b1] = 1 + + let lat = bi.lat[i] + var lb = 3 + if lat < -72.0: lb = 0 + elif lat < -36.0: lb = 1 + elif lat < -18.0: lb = 2 + elif lat <= 18.0: lb = 3 + elif lat <= 36.0: lb = 4 + elif lat <= 72.0: lb = 5 + else: lb = 6 + result[42 + lb] = 1 + +proc toLits(base: array[N_BASE, int], h: int): seq[uint8] = + var raw: array[N_BITS, int] + for i in 0.. bestV: + bestV = v + result = c + +# ── statistics ─────────────────────────────────────────────────────────────── + +proc medOf(v: seq[float]): float = + if v.len == 0: return NaN + var s = v + s.sort() + s[s.len div 2] + +proc qOf(v: seq[float], q: float): float = + if v.len == 0: return NaN + var s = v + s.sort() + s[min(s.len - 1, max(0, int(q * float(s.len - 1) + 0.5)))] + +proc medOfInts(v: seq[int]): float = + if v.len == 0: return NaN + var s = v + s.sort() + float(s[s.len div 2]) + +proc fitHitOptimal(errs, halves: seq[float], grid: seq[float]): float = + ## The shift on `grid` that MAXIMISES the number of hits + ## (|err - shift| < half) on this bucket. Ties prefer the + ## smallest-magnitude shift, so an optimum of 0 is reported as 0. + var best = 0.0 + var bestHits = -1 + for sh in grid: + var hits = 0 + for k in 0.. bestHits or (hits == bestHits and abs(sh) < abs(best)): + bestHits = hits + best = sh + best + +proc mergeHist(dst: var Hist, src: Hist) = + for a in Arm: + for hi in 0..offset mapping is + # calibrated OUT-OF-SAMPLE so an overfit in-sample median cannot leak. + let fitEnd = max(MIN_I + 1, int(0.50 * float(L))) + let calEnd = max(fitEnd + 1, int(cfgTrainFrac * float(L))) + + # local collector: samples with i in [lo,stop) and the label j = i+h < stop + # (so NOTHING here reads past `stop` — no cross-region label leakage). + proc collect(lo, stop, hidx: int): + tuple[hs: seq[int], es: seq[float], ls: seq[seq[uint8]], + ts: seq[float], rs: seq[float]] = + let h = HORIZONS[hidx] + for i in lo..= stop: continue + let cur = s[i] + let gx = cur.ex + cur.es * cos(cur.eh * PI / 180.0) * float(h) + let gy = cur.ey + cur.es * sin(cur.eh * PI / 180.0) * float(h) + let ba = arctan2(s[j].ey - cur.sy, s[j].ex - cur.sx) + let bg = arctan2(gy - cur.sy, gx - cur.sx) + let err = radToDeg(arctan2(sin(ba - bg), cos(ba - bg))) + result.hs.add hidx + result.es.add err + result.ls.add toLits(base[i], h) + result.ts.add(if i - 1 >= 0: wrap180(cur.eh - s[i - 1].eh) else: 0.0) + result.rs.add hypot(s[j].ex - cur.sx, s[j].ey - cur.sy) + + var trH: seq[int] + var trErr: seq[float] + var trLits: seq[seq[uint8]] + var trTurn: seq[float] + var calH: seq[int] + var calErr: seq[float] + var calLits: seq[seq[uint8]] + var calTurn: seq[float] + var calRange: seq[float] + var evH: seq[int] + var evIdx: seq[int] + var evErr: seq[float] + var evRange: seq[float] + var evTurn: seq[float] + + for hi in 0..= L: continue + let cur = s[i] + let gx = cur.ex + cur.es * cos(cur.eh * PI / 180.0) * float(h) + let gy = cur.ey + cur.es * sin(cur.eh * PI / 180.0) * float(h) + let ba = arctan2(s[j].ey - cur.sy, s[j].ex - cur.sx) + let bg = arctan2(gy - cur.sy, gx - cur.sx) + let err = radToDeg(arctan2(sin(ba - bg), cos(ba - bg))) + evH.add hi + evIdx.add i + evErr.add err + evRange.add hypot(s[j].ex - cur.sx, s[j].ey - cur.sy) + evTurn.add(if i - 1 >= 0: wrap180(cur.eh - s[i - 1].eh) else: 0.0) + + # ── labels (true quadrants); magnitude threshold = FIT median |err| ── + var medAbs: array[NH, float] + for hi in 0.. 0.0: 0 else: 2) + + (if abs(trErr[k]) > medAbs[hi]: 1 else: 0)) + tlits.add trLits[k] + terr.add trErr[k] + th.add hi + tturn.add trTurn[k] + tlab.add cls + + proc sideMedian(errs: seq[float]): array[2, float] = + var l, r: seq[float] + for e in errs: + if e > 1e-9: l.add e + elif e < -1e-9: r.add e + result[0] = medOf(l) + result[1] = medOf(r) + + # ── train TM (fresh per round) ── + var tm = newMachine(N_BITS, N_CLASSES, cfgClauses, cfgStates, cfgS, + seed = 12345'u64 + uint64(r.roundNo)) + trainMachine(tm, tlits, tlab, cfgEpochs, seed = 999'u64 + uint64(r.roundNo)) + + # ── calibration on the out-of-sample slice ───────────────────────────── + var calHalf = newSeq[float](calErr.len) + for k in 0.. 1e-9: 0 elif e < -1e-9: 1 else: -1 + + # TM predictions on the calibration slice (computed once, reused below). + var calPredTM = newSeq[int](calErr.len) + var calPredShuf = newSeq[int](calErr.len) + block: + var cache = newSeq[uint8](tm.nClauses) + for idx in 0.. 0: medOf(clsE[hi][c]) + else: (if c < 2: sm2[0] else: sm2[1]) + + # HIT-OPTIMAL shift per predicted quadrant (the new appliance). + block: + var clsE: array[NH, array[N_CLASSES, seq[float]]] + var clsH: array[NH, array[N_CLASSES, seq[float]]] + for idx in 0.. 0: fitHitOptimal(clsE[hi][c], clsH[hi][c], grid) + else: (if c < 2: sm2[0] else: sm2[1]) + + # HIT-OPTIMAL shift per predicted SIDE (quadrant collapsed) — apples-to-apples + # with PERF-SIGN: separates "the quadrant split is noisy" from "60% is too low". + block: + var sideE: array[NH, array[2, seq[float]]] + var sideH: array[NH, array[2, seq[float]]] + for idx in 0.. 0: fitHitOptimal(sideE[hi][sd], sideH[hi][sd], grid) + else: 0.0 + + # ── shuffled-label control ── + # Permute the fit labels within each horizon, retrain, and apply the SAME + # hit-optimal appliance to the shuffled model. + var slab = tlab + block: + var rng = seedRng(4242'u64 + uint64(r.roundNo)) + for hi in 0.. 0: fitHitOptimal(clsE[hi][c], clsH[hi][c], grid) + else: (if c < 2: sm2[0] else: sm2[1]) + + # ── PERFECT-SIGN + hit-optimal shift (DECISIVE, cheating arm) ── + # The bucket is the TRUE sign of the residual (not available in reality); + # the shift is the hit-optimal shift on the calibration slice per true side. + block: + var sideE: array[NH, array[2, seq[float]]] + var sideH: array[NH, array[2, seq[float]]] + for idx in 0.. 0: fitHitOptimal(sideE[hi][sd], sideH[hi][sd], grid) + else: 0.0 + + # side-accuracy sweep: what q does the form need? For each target accuracy, + # a synthetic predictor correct with probability q (independent draws on the + # calibration and eval slices); fit per predicted side, apply out-of-sample. + block: + for qi, q in ACCS: + var rng = seedRng(31337'u64 + uint64(r.roundNo) * 131'u64 + uint64(qi)) + var calSide = newSeq[int](calErr.len) + for idx in 0.. 0: fitHitOptimal(sideE[hi][sd], sideH[hi][sd], grid) + else: 0.0 + result.accFit[qi][hi][sd] = off[hi][sd] + for k in 0..= 0: result.fitPS[hi][ps] else: 0.0), + aPS0: 0.0, + aShuf: result.fitShuf[hi][predShuf]] + + # diagnostics on the SAME eval tick + let trueCls = ((if err > 0.0: 0 else: 2) + + (if abs(err) > medAbs[hi]: 1 else: 0)) + inc result.trueCounts[hi][trueCls] + inc result.predCounts[hi][predTM] + inc result.tmQTot[hi] + if predTM == trueCls: inc result.tmQCor[hi] + inc result.shufQTot[hi] + if predShuf == trueCls: inc result.shufQCor[hi] + inc result.tmHitTot[hi] + if (predTM <= 1) == (trueCls <= 1): inc result.tmHitCor[hi] + + for a in Arm: + let res = err - offs[a] + result.errs[a][hi].add abs(res) + if abs(res) < half: inc result.hits[a][hi] + + # ── per-round table ── + if emit: + echo &" round {r.roundNo:>2} (L={L:>5}) " & + "nEval/h = " & $[result.n[0], result.n[1], result.n[2], result.n[3]] + echo " h arm med p90 hit%" + for hi in 0.. 0: 100.0 * result.hits[a][hi].float / result.n[hi].float else: NaN + echo &" {HORIZONS[hi]:>3} {ArmName[a]:<9} {m:>6.2f} {p:>6.2f} {hit:>6.1f}" + +# ── table printing ─────────────────────────────────────────────────────────── + +proc printPooled(h: Hist, title: string) = + echo "\n" & title + echo " h arm N med|err| p90|err| hit% miss%" + for hi in 0.. 0: 100.0 * h.hits[a][hi].float / n.float else: NaN + echo &"{HORIZONS[hi]:>3} {ArmName[a]:<9} {n:>7} {m:>9.2f} {p:>9.2f} {hit:>7.1f} {100.0-hit:>7.1f}" + +proc printDeltas(h: Hist, title: string) = + echo "\n" & title + echo " h naive% dTMmed dTMhit dTMside dTMs0 dTMs25 dTMs50 dPS dPS0 dShuf (hit pp vs naive)" + for hi in 0..3} {base:>6.1f} " + for a in Arm: + if a == aNaive: continue + let v = 100.0 * h.hits[a][hi].float / n - base + row.add &"{v:>+6.1f} " + echo row + +proc printDiagnostics(h: Hist, title: string) = + ## Is the learner actually learning? (quadrant / side accuracy vs chance) + echo "\n" & title + echo " h N TMquad% shufquad% TMside% majQuad% pred[Ls Lb Rs Rb] / true[Ls Lb Rs Rb]" + for hi in 0..3} {n:>6} {tmq:>8.1f} {shufq:>10.1f} {tms:>8.1f} {maj:>9.1f} [{pcs}] / [{tcs}]" + +proc printFittedSummary(rounds: seq[Hist], title: string) = + ## Median across rounds of the fitted hit-optimal shift (deg). + echo "\n" & title + echo " (median across rounds of the fitted HIT-OPTIMAL shift, deg)" + echo " h TM[Ls Lb Rs Rb] PS[left right] shuf[Ls Lb Rs Rb]" + for hi in 0..6.2f} " + for sd in 0..<2: s2.add &"{medOf(psC[sd]):>7.2f} " + for c in 0..6.2f} " + echo &"{HORIZONS[hi]:>3} [{s1}] [{s2}] [{s3}]" + +proc printCurve(h: Hist, title: string) = + ## Hit fraction as a function of a CONSTANT shift applied to every + ## calibration sample (pooled across buckets), per horizon. + echo "\n" & title + echo " shift h=15 h=20 h=25 h=30 (hit% on calibration, pooled)" + for k in 0..+6.1f} " + for hi in 0.. 0: 100.0 * h.curveHits[hi][k].float / n.float else: NaN + row.add &"{v:>7.1f} " + echo row + +proc printAccuracySweep(rounds: seq[Hist], title: string) = + ## Gain (pp vs naive) for a synthetic side predictor of accuracy q. + echo "\n" & title + echo " q fitShift(h=20)[L R] hit% h15 h20 h25 h30 d(hit) pp: h15 h20 h25 h30 mean" + var naiveHits, naiveN: array[NH, int] + for rh in rounds: + for hi in 0.. 0: 100.0 * hits[hi].float / nn[hi].float else: NaN + let np = if naiveN[hi] > 0: 100.0 * naiveHits[hi].float / naiveN[hi].float else: NaN + hRow.add &"{hp:>6.1f}" + dRow.add &"{hp - np:>+6.1f}" + dsum += hp - np + echo &" {q:>4.2f} [{medOf(sh[0]):>6.2f} {medOf(sh[1]):>6.2f}] {hRow} {dRow} {dsum/float(NH):>+6.1f}" + +# ── main ───────────────────────────────────────────────────────────────────── + +proc main() = + var names = @PRIMARY + for i in 1..paramCount(): + let a = paramStr(i) + if a.startsWith("--fixtures="): names = a[11..^1].split(',') + elif a.startsWith("--epochs="): cfgEpochs = parseInt(a[9..^1]) + elif a.startsWith("--trainfrac="): cfgTrainFrac = parseFloat(a[12..^1]) + elif a.startsWith("--clauses="): cfgClauses = parseInt(a[10..^1]) + + echo "=" .repeat(90) + echo "GATE 2b - is the shift-per-bucket appliance STRUCTURALLY DEAD or MISCALIBRATED?" + echo "=" .repeat(90) + echo &"fixtures : {names.join(\", \")}" + echo &"bits : {N_BITS} ({N_BASE} draft + 4 horizon one-hot)" + echo &"TM : {cfgClauses} clauses, {cfgStates} states, s={cfgS}, " & + &"{cfgEpochs} epochs, fresh per round" + echo &"split : within-round, train 50%, calibrate {cfgTrainFrac*100:.0f}%, eval rest" + echo "appliance : shift on {-20..+20 deg, 0.5 step} maximising HIT COUNT" + echo &" on the calibration slice; applied out-of-sample" + echo &"estimated hit : |residual| < atan(18px / range) -- ABSOLUTE IS OPTIMISTIC" + echo &"DECISIVE arm : PERF-SIGN uses the TRUE sign (cheating) + hit-optimal shift" + echo &"bullet block : INFERRED from self-energy drops (no gun heading recorded)" + echo "=" .repeat(90) + + var pooled = Hist() + var pooledByFix = initTable[string, Hist]() + var roundHists = initTable[string, seq[Hist]]() + var allRounds: seq[Hist] + + for name in names: + let path = if name.endsWith(".jsonl"): fixturesDir / name + else: fixturesDir / (name & ".jsonl") + if not fileExists(path): + echo &"# SKIP missing fixture {path}" + continue + let ticks = loadTicks(path) + let rounds = loadRounds(path, ticks) + echo &"\n## FIXTURE {name}: {rounds.len} rounds, {ticks.len} ticks" + var fhist = Hist() + var rlist: seq[Hist] + for r in rounds: + let bi = buildBulletSeries(r) + var cache = newSeq[uint8](cfgClauses) + let rh = runRound(r, bi, cache, emit = true) + mergeHist(fhist, rh) + mergeHist(pooled, rh) + rlist.add rh + allRounds.add rh + echo "" + pooledByFix[name] = fhist + roundHists[name] = rlist + printPooled(fhist, &"## PER-FIXTURE pooled ({name})") + printDiagnostics(fhist, &"## PER-FIXTURE learnability ({name})") + + echo "\n" & "=".repeat(90) + printPooled(pooled, "## POOLED PRIMARY (tr_drussgt_vs_modularbot*): arm table") + printDeltas(pooled, "## DELTAS (percentage points of estimated hit fraction)") + printDiagnostics(pooled, "## POOLED learnability diagnostics (is the learner learning?)") + printFittedSummary(allRounds, "## FITTED HIT-OPTIMAL SHIFT MAGNITUDES") + printCurve(pooled, "## HIT FRACTION vs CONSTANT SHIFT MAGNITUDE (calibration)") + printAccuracySweep(allRounds, "## SIDE-ACCURACY SWEEP (synthetic q-accurate predictor, out-of-sample)") + + echo "\n" & "=".repeat(90) + echo "## PER-ROUND hit-delta integrity (pp vs naive)" + echo "fixture round h TMmed TMhit TMside TMs25 TMs50 PS shuf" + for name in names: + if not roundHists.hasKey(name): continue + for rn, rh in roundHists[name]: + for hi in 0..5} {HORIZONS[hi]:>3} {dTMmed:>+6.1f} " & + &"{dTMhit:>+6.1f} {dTMside:>+6.1f} {dTMs25:>+6.1f} {dTMs50:>+6.1f} " & + &"{dPS:>+6.1f} {dShuf:>+6.1f}" + + # ── unhedged verdict inputs ── + echo "\n" & "=".repeat(90) + echo "## VERDICT INPUTS (pooled hit pp vs naive)" + for hi in 0..3} PERF-SIGN {ps:>+6.1f} pp TM-side {tms:>+6.1f} pp TM-quad {tm:>+6.1f} pp shuffled {sh:>+6.1f} pp" + +when isMainModule: + main()