diff --git a/ModularBot_garage/.env.example b/ModularBot_garage/.env.example index 1cee78b..fc88559 100644 --- a/ModularBot_garage/.env.example +++ b/ModularBot_garage/.env.example @@ -98,6 +98,8 @@ TR_RAM_PLAN=off # the change-of-plan trigger (enemy outguns us while TR_RAM_PLAN_DIST=250.0 # px; max range at which the plan trigger may fire TR_RAM_PLAN_MARGIN=20.0 # energy advantage the plan trigger needs TR_RAM_PLAN_HITRATE=0.05 # pooled virtual hit rate below which the gun duel counts as failing +TR_RAM_FLOOR_ENERGY=0.0 # j160 firing floor: at/below this self energy stop firing (0 = off) +TR_RAM_ENEMY_ENERGY=0.0 # j160 exhaustion: last-scanned enemy energy <= this -> ram (0 = off) # ── movement internals: tfil (the floor-is-lava field) ────────────────────── TR_TFIL_RANGE_LO=100.0 # px; lower edge of the range band the ring mover prefers diff --git a/ModularBot_garage/src/ModularBot.nim b/ModularBot_garage/src/ModularBot.nim index cc06e23..676fcb6 100644 --- a/ModularBot_garage/src/ModularBot.nim +++ b/ModularBot_garage/src/ModularBot.nim @@ -1459,9 +1459,14 @@ method run*(bot: ModularBot) = let distPx = hypot(pred.x - getX(), pred.y - getY()) if shouldFire(gunDir, aimTarget, gunHeat, distPx): + # j160 FIRING FLOOR: at/below TR_RAM_FLOOR_ENERGY self energy we hold + # the reserve for the ram instead of spending it on a shot. Off by + # default (`RamFloorEnergy = 0.0`), and `ramming` (the exhaustion + # trigger) wins the conflict, so an engaged ram never starves itself. + let floorBlocks = fireFloorBlocks(RamFloorEnergy, getEnergy(), shouldRam) # Enqueue the selected gun so onBulletFired can stamp the server's bulletId. # getEnergy() > power mirrors the server's "bot.energy <= firepower" reject. - if setFire(power) and getEnergy() > power: + if not floorBlocks and setFire(power) and getEnergy() > power: bot.pendingFires.add(PendingShot( gunId: selectedGun, angleErr: abs(normDelta), diff --git a/ModularBot_garage/src/env_report.nim b/ModularBot_garage/src/env_report.nim index c3e7640..9bc48dd 100644 --- a/ModularBot_garage/src/env_report.nim +++ b/ModularBot_garage/src/env_report.nim @@ -416,6 +416,8 @@ proc printEffectiveValues(ctx: EnvReportContext) = emit("TR_RAM_PLAN_MARGIN", $RamPlanMargin, sourceOf("TR_RAM_PLAN_MARGIN")) emit("TR_RAM_PLAN_HITRATE", $RamPlanHitRate, sourceOf("TR_RAM_PLAN_HITRATE")) emit("TR_RAM_LOG", onOff(RamLog), sourceOfPresence("TR_RAM_LOG")) + emit("TR_RAM_FLOOR_ENERGY", $RamFloorEnergy, sourceOf("TR_RAM_FLOOR_ENERGY")) + emit("TR_RAM_ENEMY_ENERGY", $RamEnemyEnergy, sourceOf("TR_RAM_ENEMY_ENERGY")) # ── the horizon TM gun ──────────────────────────────────────────────────── # `resetLearning`/`targetChanged` resolve the lazily-read fields at round @@ -667,6 +669,7 @@ proc knownEnvNames*(): seq[string] = "TR_RAM_OPPORTUNITY", "TR_RAM_OPP_DIST", "TR_RAM_OPP_MARGIN", "TR_RAM_ABORT_DMG", "TR_RAM_PLAN", "TR_RAM_PLAN_DIST", "TR_RAM_PLAN_MARGIN", "TR_RAM_PLAN_HITRATE", "TR_RAM_LOG", + "TR_RAM_FLOOR_ENERGY", "TR_RAM_ENEMY_ENERGY", "TR_TFIL_RANGE_LO", "TR_TFIL_RANGE_HI", "TR_TFIL_RANGE_TEMP", "TR_TFIL_RING_COMMIT_ARRIVAL", "TR_TFIL_RING_NOREV_SPEED", "TR_TFIL_RANGE_K", "TR_TFIL_CORRIDOR_HEAT", "TR_TFIL_WALL_HOTNESS", diff --git a/common_libs/movements/ram_decision.nim b/common_libs/movements/ram_decision.nim index 4b6a003..3fbadc0 100644 --- a/common_libs/movements/ram_decision.nim +++ b/common_libs/movements/ram_decision.nim @@ -46,7 +46,29 @@ ## TR_RAM_PLAN_MARGIN default 20.0 change-of-plan energy advantage ## TR_RAM_PLAN_HITRATE default 0.05 selected gun's pooled virtual hit rate ## below which the gun duel counts as failing -## TR_RAM_LOG=1 emit one change-gated `[ram]` line +## ## TR_RAM_LOG=1 emit one change-gated `[ram]` line + ## TR_RAM_FLOOR_ENERGY default 0.0 FIRING FLOOR (j160). At or below this + ## self energy we stop firing to keep a + ## ram reserve. 0 = off = today's behaviour. + ## TR_RAM_ENEMY_ENERGY default 0.0 ENEMY-EXHAUSTION trigger (j160). The + ## last-scanned enemy energy <= this -> + ## ram mode. 0 = off. +## +## ── j160: the energy-reserve + exhaustion policy ─────────────────────────── +## Energy NEVER regenerates and has no cap; the only gain in the whole game is +## `+3 * power` per bullet hit LANDED (server `rules.kt`). So not firing denies +## the enemy its only refill AND keeps our ram reserve intact — the two halves +## of the policy are the same bet. +## +## Floor sizing: one likely return hit (`bulletDamage(1.0)` = 4.0) plus two +## 0.1-power shots (0.1 each) is 4.2. The knob DEFAULT stays 0.0 so the default +## path is byte-identical; the operator sets 5-ish. +## +## The exhaustion trigger is the FINISHER with the energy tolerance promoted to +## an operator knob. It deliberately KEEPS the finisher's own +## `selfEnergy > enemyEnergy` surplus guard: `RAM_DAMAGE 0.6` is applied to BOTH +## bots on every contact tick, so a head-on contact is a symmetric bleed decided +## by who walks in with the surplus. import std/[os, strutils] @@ -95,10 +117,15 @@ let RamPlanDist* = getEnvFloat("TR_RAM_PLAN_DIST", DefaultRamPlanDist) let RamPlanMargin* = getEnvFloat("TR_RAM_PLAN_MARGIN", DefaultRamPlanMargin) let RamPlanHitRate* = getEnvFloat("TR_RAM_PLAN_HITRATE", DefaultRamPlanHitRate) let RamLog* = existsEnv("TR_RAM_LOG") +## j160. 0.0 = off on BOTH knobs, which is the shipped behaviour. +let RamFloorEnergy* = getEnvFloat("TR_RAM_FLOOR_ENERGY", 0.0) +let RamEnemyEnergy* = getEnvFloat("TR_RAM_ENEMY_ENERGY", 0.0) type RamReason* = enum rrNone ## no trigger fires + rrExhausted ## j160: last-scanned enemy energy <= TR_RAM_ENEMY_ENERGY + ## and we hold the surplus (it is out of ammo, we are not) rrFinisher ## enemy < 20 energy, we are healthier, dist < 300 rrOpportunity ## we clearly out-energise and are close enough to close rrDesperation ## both nearly dead, short range @@ -131,7 +158,8 @@ proc ramTrigger*(inp: RamInputs, planEnabled = RamPlanEnabled, planDist = RamPlanDist, planMargin = RamPlanMargin, - planHitRate = RamPlanHitRate): RamReason = + planHitRate = RamPlanHitRate, + enemyEnergyTol = RamEnemyEnergy): RamReason = ## Pure trigger evaluation. Returns the FIRST matching reason in priority ## order, or `rrNone`. Cooldown/duration/abort are deliberately NOT here — the ## caller composes those, so this function has no state and is unit-testable. @@ -143,6 +171,13 @@ proc ramTrigger*(inp: RamInputs, ## `desperation` and `finisher` are kept: they are rare, short-range, and the ## finisher is the only measured conversion. `plan` remains opt-in and off. if inp.enemyEnergy <= 0.0: return rrNone + # j160 exhaustion trigger. Checked FIRST so the operator-set tolerance wins + # the label when it is set; it is the finisher's own shape (same surplus and + # distance guards) with the 20.0 energy tolerance promoted to a knob. With + # `enemyEnergyTol = 0.0` (the default) this arm can never fire. + if enemyEnergyTol > 0.0 and inp.enemyEnergy <= enemyEnergyTol and + inp.dist < RamFinisherDist and inp.selfEnergy > inp.enemyEnergy: + return rrExhausted if inp.dist < RamFinisherDist and inp.enemyEnergy < RamFinisherEnergy and inp.selfEnergy > inp.enemyEnergy: return rrFinisher @@ -158,9 +193,24 @@ proc ramTrigger*(inp: RamInputs, return rrPlan rrNone +proc fireFloorBlocks*(floor, selfEnergy: float, ramming = false): bool = + ## j160 FIRING FLOOR. True when the reserve is thin enough that we must not + ## commit a NEW shot. `floor = 0.0` (the default) disables the floor entirely + ## and returns false for every input, so the default path is unchanged. + ## + ## `ramming` WINS over the floor: once ram mode is engaged the duel is over, + ## so the reserve is being spent on the contact, not held for it. This is the + ## same exemption `ramming` already gets in `applyPowerPolicy`. + ## + ## The floor blocks only NEW shots. A bullet already in the air (gun heat > 0) + ## is untouched — `shouldFire` already gates on `gunHeat <= 0.0`, so there is + ## no committed shot for the floor to suppress or cancel. + not ramming and floor > 0.0 and selfEnergy <= floor + proc reasonName*(r: RamReason): string = case r of rrNone: "none" + of rrExhausted: "exhausted" of rrFinisher: "finisher" of rrOpportunity: "opportunity" of rrDesperation: "desperation" diff --git a/common_libs/tests/measure_ram_exhaustion b/common_libs/tests/measure_ram_exhaustion new file mode 100755 index 0000000..72ad1fd --- /dev/null +++ b/common_libs/tests/measure_ram_exhaustion @@ -0,0 +1,278 @@ +#!/usr/bin/env python3 +"""j162 DECISIVE measurement: does the bot ever actually run out of energy? + +The firing floor (TR_RAM_FLOOR_ENERGY) only pays if the bot regularly creeps +down to a few energy and gets disabled. This answers that from the ALREADY +RECORDED closed-loop corpus, state only: + + A) self energy AT DEATH (the reserve we actually held when the killing blow + landed) -- the floor's entire claim + B) how long we stay at energy <= 0 (isDisabled) before the round ends + C) recovery: how often self energy RISES tick-over-tick, and from what level + (the only refill in the game is +3*power per landed bullet hit, so a rise + is a landed hit -- this is "can we climb back out by shooting") + D) what a floor at {3,5,10,20} would cost: % ticks suppressed, run length, and + the heat-limited ceiling on how much energy it could possibly save + +NO battle, NO server, NO counterfactual replay, NO damage estimate (the offline +harness scored 0/6 on closed-loop questions, docs/offline_harness_trust.md). + +Usage: python3 common_libs/tests/measure_ram_exhaustion [glob-dir] +""" +import glob, json, os, statistics, sys +from array import array +from multiprocessing import Pool + +ROOTS = sys.argv[1:] or ["/tmp"] + +# Tank Royale gun heat: heat += 1 + power/5 and the gun cools 0.1/tick, so a +# power-p shot can be fired at most once per 10 + 2p ticks and costs p energy. +# The cost bracket is therefore p/(10+2p) energy per tick, from 0.0098 at the +# cheapest legal shot (0.1) to 0.1875 at the most expensive (3.0). +def per_tick(power): + return power / (10.0 + 2.0 * power) + + +def num(line, key): + i = line.find('"' + key + '":') + if i < 0: + return None + i += len(key) + 3 + j = line.find(',', i) + if j < 0: + j = line.find('}', i) + try: + return float(line[i:j]) + except ValueError: + return None + + +def load(path): + """[(self, enemy)] per tick, with round boundaries from the round map.""" + rows = [] + with open(path) as fh: + for line in fh: + if '"tick"' not in line: + continue + t, se, ee = num(line, 'tick'), num(line, 'se'), num(line, 'ee') + if t is None or se is None or ee is None: + continue + rows.append((t, se, ee)) + if not rows: + return [] + rf = path.replace(".jsonl", ".jsonl.rounds.json") + bounds = [] + if os.path.exists(rf): + try: + for r in json.load(open(rf))["rounds"]: + bounds.append((r["startTick"], r["startTick"] + r["count"])) + except Exception: + bounds = [] + if not bounds: + # no map: a round is the span between RISES from depleted to full, + # never the first ticks of a round where both bots sit at 100. + starts = [0] + [i for i in range(1, len(rows)) + if rows[i][1] >= 100 > rows[i - 1][1]] + bounds = [(starts[k], starts[k + 1] if k + 1 < len(starts) else len(rows)) + for k in range(len(starts))] + rounds = [] + for s, e in bounds: + r = [(se, ee) for t, se, ee in rows if s <= t < e] + if r: + rounds.append(r) + return rounds + + +def corpus(): + files = [] + for root in ROOTS: + for f in glob.glob(os.path.join(root, "**", "*.jsonl"), recursive=True): + if f.endswith(".events.jsonl"): + continue + try: + with open(f) as fh: + first = fh.readline() + except OSError: + continue + if '"closed_loop":true' not in first.replace(" ", ""): + continue + files.append(f) + out = [] + for r in Pool(8).imap(load, sorted(files), chunksize=32): + out += r + return sorted(files), out + + +def pct(sorted_x, q): + if not sorted_x: + return 0.0 + i = q * (len(sorted_x) - 1) + lo, hi = int(i), min(int(i) + 1, len(sorted_x) - 1) + return sorted_x[lo] + (sorted_x[hi] - sorted_x[lo]) * (i - lo) + + +def main(): + files, rounds = corpus() + N = sum(len(r) for r in rounds) + print(f"recordings={len(files)} rounds={len(rounds)} ticks={N}\n") + + # ---- A) how each round ends, and the reserve held at that moment -------- + self_dead = enemy_dead = both_dead = alive_end = 0 + last_alive = [] # self energy on the last tick we were alive + death_tick = [] # self energy on the tick we crossed 0 (can be < 0) + zero_runs = [] # ticks spent at self energy <= 0 before round end + over = [] # reserve that would have absorbed the killing blow + for r in rounds: + sd = ed = None + for i, (a, b) in enumerate(r): + if sd is None and a <= 0: + sd = i + if ed is None and b <= 0: + ed = i + if sd is not None and ed is not None: + break + if sd is None and ed is None: + alive_end += 1 + continue + if sd is not None and ed is not None: + both_dead += 1 + elif sd is not None: + self_dead += 1 + else: + enemy_dead += 1 + if sd is not None: + last_alive.append(r[sd - 1][0] if sd > 0 else r[0][0]) + death_tick.append(r[sd][0]) + over.append(-r[sd][0]) + j = len(r) + while j > sd and r[j - 1][0] <= 0: + j -= 1 + zero_runs.append(len(r) - j) + m = len(rounds) + print("=== A) how each round ends ===") + print(f" self reached energy<=0 : {self_dead:>6} rounds ({100*self_dead/m:5.1f}%)") + print(f" only the enemy did : {enemy_dead:>6} rounds ({100*enemy_dead/m:5.1f}%)") + print(f" both in the same round : {both_dead:>6} rounds ({100*both_dead/m:5.1f}%)") + print(f" neither (truncated) : {alive_end:>6} rounds ({100*alive_end/m:5.1f}%)") + + print("\n=== B) SELF ENERGY AT DEATH (last value above 0 before the kill) ===") + s = sorted(last_alive) + if s: + print(f" n={len(s)} min {s[0]:.2f} p10 {pct(s,.10):.2f} median {pct(s,.5):.2f}" + f" mean {statistics.fmean(s):.2f} p90 {pct(s,.90):.2f} max {s[-1]:.2f}") + for t in (0, 1, 3, 5, 10, 20): + c = sum(1 for x in s if x <= t) + print(f" <= {t:>2} energy: {c:>6} ({100*c/len(s):5.1f}% of self deaths," + f" {100*c/m:5.2f}% of all rounds)") + d = sorted(death_tick) + if d: + print(f" crossing value: median {pct(d,.5):.2f} p10 {pct(d,.10):.2f}" + f" p90 {pct(d,.90):.2f} (negative = overshoot of the killing hit)") + over = sorted(over) + print(" reserve that WOULD have survived the killing blow (overshoot):") + print(f" median {pct(over,.5):.2f} p75 {pct(over,.75):.2f}" + f" p90 {pct(over,.90):.2f} p99 {pct(over,.99):.2f} max {over[-1]:.2f}") + for F in (3, 5, 10, 20): + c = sum(1 for x in over if x < F) + print(f" a reserve of {F:>2} would have absorbed it in {c:>6} self deaths" + f" ({100*c/len(over):5.1f}%)") + + print("\n=== C) time spent at energy<=0 (isDisabled) before the round ends ===") + z = sorted(zero_runs) + if z: + print(f" ticks disabled: median {pct(z,.5):.0f} p90 {pct(z,.9):.0f}" + f" max {z[-1]} total {sum(z)} of {N} ticks" + f" ({100*sum(z)/N:.4f}%)") + + # ---- D) recovery: energy RISES tick-over-tick = a landed bullet hit ----- + rises, pre = 0, [] + pre_low = {20: 0, 10: 0, 5: 0, 3: 0} + tot_ticks = 0 + for r in rounds: + for i in range(1, len(r)): + tot_ticks += 1 + if r[i][0] - r[i - 1][0] > 0.01: + rises += 1 + pre.append(r[i - 1][0]) + for t in pre_low: + if r[i - 1][0] <= t: + pre_low[t] += 1 + print("\n=== D) RECOVERY: self energy rises tick-over-tick (a landed hit) ===") + print(f" rising transitions: {rises} of {tot_ticks} tick-pairs" + f" ({100*rises/tot_ticks:.3f}%), i.e. ~{rises/len(rounds):.2f} per round") + p = sorted(pre) + if p: + print(f" self energy just BEFORE the rise: median {pct(p,.5):.2f}" + f" p10 {pct(p,.10):.2f} p90 {pct(p,.90):.2f}") + print(" climbs that started from a low reserve:") + for t in sorted(pre_low, reverse=True): + print(f" from <= {t:>2}: {pre_low[t]:>6} rises" + f" ({100*pre_low[t]/rises:5.2f}% of rises)") + + # the decisive conditional: sitting low, do we climb back out or die? + # "death" counts the ONE tick that crosses 0. The long zero tails a few + # recordings hold afterwards are a recorder artefact, not a state lived in. + print("\n P(climb out | low) vs P(die | low), per tick spent at that level:") + death_idx = [] + for r in rounds: + death_idx.append(next((i for i, (a, _) in enumerate(r) if a <= 0), -1)) + for F in (3, 5, 10, 20): + at = rise = died = 0 + for r, di in zip(rounds, death_idx): + for i, (a, _) in enumerate(r): + if a > F: + continue + at += 1 + if i and r[i][0] - r[i - 1][0] > 0.01: + rise += 1 + if i == di: + died += 1 + if at: + print(f" energy <= {F:>2}: {at:>8} ticks | climb next tick" + f" {100*rise/at:6.3f}% | killed on this tick {100*died/at:6.3f}%" + f" -> dying is {died/max(1,rise):.1f}x more likely than recovering") + + # ---- E) what the floor would cost --------------------------------------- + print("\n=== E) COST of TR_RAM_FLOOR_ENERGY: ticks where a new shot is blocked ===") + print(f"{'floor':>5} {'%ticks':>7} {'rounds':>7} {'med run':>8} {'p90 run':>8}" + f" {'max run':>8} {'energy saved, corpus (0.1..3.0 p)':>34}" + f" {'per med run @1.0p':>19}") + for F in (3, 5, 10, 20): + tot, hit, lens = 0, 0, [] + for r in rounds: + cur, got = 0, False + for a, _ in r: + if a <= F: + cur += 1 + tot += 1 + got = True + elif cur: + lens.append(cur) + cur = 0 + if cur: + lens.append(cur) + hit += 1 if got else 0 + lens.sort() + # The gun may not fire more often than 1/(10*heat) ticks, so the floor + # can never save more than the suppressed ticks x power-per-shot x + # shots-per-tick. Report the bracket: 0.1 power (cheapest legal shot) to + # 3.0 power (most expensive legal shot). + med = pct(lens, .5) if lens else 0 + lo, hi = tot * per_tick(0.1), tot * per_tick(3.0) + mid = med * per_tick(1.0) + print(f"{F:>5} {100*tot/N:>6.2f}% {hit:>7} {med:>8.0f} " + f"{pct(lens,.9) if lens else 0:>8.0f} {lens[-1] if lens else 0:>8}" + f" {lo:>7.0f} .. {hi:>7.0f} {mid:>6.2f}") + print(" energy saved over the WHOLE corpus, heat-limited: the 0.1..3.0 power") + print(" bracket, then the p=1.0 column = what one median suppressed RUN is worth") + print(" (1.0 power is the mode of the measured landed-hit histogram).") + print(" Median run lengths 34/53/89/139 ticks; one 1.0-power landed hit = 3.0.") + print() + print(" CAVEAT, measured: the recorded energy ledger closes EXACTLY on") + print(" start + landed-gains - damage = end (residual -0.00 over 35065 rounds),") + print(" i.e. these captures DO NOT charge the firepower cost. The cost column") + print(" is therefore computed from the game rules, not read off the data.") + + +if __name__ == "__main__": + main() diff --git a/common_libs/tests/measure_ram_exhaustion_results.txt b/common_libs/tests/measure_ram_exhaustion_results.txt new file mode 100644 index 0000000..a3a2d7c --- /dev/null +++ b/common_libs/tests/measure_ram_exhaustion_results.txt @@ -0,0 +1,57 @@ +recordings=8149 rounds=35163 ticks=34461805 + +=== A) how each round ends === + self reached energy<=0 : 21518 rounds ( 61.2%) + only the enemy did : 12906 rounds ( 36.7%) + both in the same round : 347 rounds ( 1.0%) + neither (truncated) : 392 rounds ( 1.1%) + +=== B) SELF ENERGY AT DEATH (last value above 0 before the kill) === + n=21865 min 0.00 p10 0.10 median 0.83 mean 2.50 p90 8.90 max 24.83 + <= 0 energy: 0 ( 0.0% of self deaths, 0.00% of all rounds) + <= 1 energy: 12217 ( 55.9% of self deaths, 34.74% of all rounds) + <= 3 energy: 16793 ( 76.8% of self deaths, 47.76% of all rounds) + <= 5 energy: 18420 ( 84.2% of self deaths, 52.38% of all rounds) + <= 10 energy: 20290 ( 92.8% of self deaths, 57.70% of all rounds) + <= 20 energy: 21864 (100.0% of self deaths, 62.18% of all rounds) + crossing value: median -0.40 p10 -6.90 p90 0.00 (negative = overshoot of the killing hit) + reserve that WOULD have survived the killing blow (overshoot): + median 0.40 p75 2.00 p90 6.90 p99 15.00 max 19.50 + a reserve of 3 would have absorbed it in 17301 self deaths ( 79.1%) + a reserve of 5 would have absorbed it in 18610 self deaths ( 85.1%) + a reserve of 10 would have absorbed it in 20690 self deaths ( 94.6%) + a reserve of 20 would have absorbed it in 21865 self deaths (100.0%) + +=== C) time spent at energy<=0 (isDisabled) before the round ends === + ticks disabled: median 1 p90 18 max 452 total 593030 of 34461805 ticks (1.7208%) + +=== D) RECOVERY: self energy rises tick-over-tick (a landed hit) === + rising transitions: 205754 of 34426642 tick-pairs (0.598%), i.e. ~5.85 per round + self energy just BEFORE the rise: median 45.24 p10 8.04 p90 89.00 + climbs that started from a low reserve: + from <= 20: 52256 rises (25.40% of rises) + from <= 10: 25352 rises (12.32% of rises) + from <= 5: 12812 rises ( 6.23% of rises) + from <= 3: 8013 rises ( 3.89% of rises) + + P(climb out | low) vs P(die | low), per tick spent at that level: + energy <= 3: 2633881 ticks | climb next tick 0.130% | killed on this tick 0.830% -> dying is 6.4x more likely than recovering + energy <= 5: 3299807 ticks | climb next tick 0.232% | killed on this tick 0.663% -> dying is 2.9x more likely than recovering + energy <= 10: 5002524 ticks | climb next tick 0.365% | killed on this tick 0.437% -> dying is 1.2x more likely than recovering + energy <= 20: 8526129 ticks | climb next tick 0.500% | killed on this tick 0.256% -> dying is 0.5x more likely than recovering + +=== E) COST of TR_RAM_FLOOR_ENERGY: ticks where a new shot is blocked === +floor %ticks rounds med run p90 run max run energy saved, corpus (0.1..3.0 p) per med run @1.0p + 3 7.64% 23086 34 299 1104 25822 .. 493853 2.83 + 5 9.58% 23739 53 330 1104 32351 .. 618714 4.42 + 10 14.52% 25211 89 433 1141 49044 .. 937973 7.42 + 20 24.74% 27799 139 621 2059 83590 .. 1598649 11.58 + energy saved over the WHOLE corpus, heat-limited: the 0.1..3.0 power + bracket, then the p=1.0 column = what one median suppressed RUN is worth + (1.0 power is the mode of the measured landed-hit histogram). + Median run lengths 34/53/89/139 ticks; one 1.0-power landed hit = 3.0. + + CAVEAT, measured: the recorded energy ledger closes EXACTLY on + start + landed-gains - damage = end (residual -0.00 over 35065 rounds), + i.e. these captures DO NOT charge the firepower cost. The cost column + is therefore computed from the game rules, not read off the data. diff --git a/common_libs/tests/measure_ramfloor_energy b/common_libs/tests/measure_ramfloor_energy new file mode 100644 index 0000000..b0a071e --- /dev/null +++ b/common_libs/tests/measure_ramfloor_energy @@ -0,0 +1,141 @@ +#!/usr/bin/env python3 +"""j160 open-loop energy measurement for the FIRING FLOOR / EXHAUSTION RAM. + +NO battle, NO server, NO counterfactual replay. This reads the ALREADY RECORDED +closed-loop captures under /tmp and reports, per tick: + + * how often SELF energy sits below a floor candidate, + * whether the owner's "both low, nobody firing" situation actually occurs, + * who crosses a low-energy line FIRST (self or the enemy), + * how often the enemy is low while we are healthy -- the opportunity the + exhaustion trigger (TR_RAM_ENEMY_ENERGY) would act on. + +Deliberately produces NO "damage if we had not fired" number: the offline +harness scored 0/6 on closed-loop questions (docs/offline_harness_trust.md), +so that class of number is worthless here. + +Usage: python3 common_libs/tests/measure_ramfloor_energy [glob-dir] +""" +import json, os, glob, statistics, sys, array + +ROOTS = sys.argv[1:] or ["/tmp"] + +def recordings(): + out = [] + for root in ROOTS: + for f in glob.glob(os.path.join(root, "**", "*.jsonl"), recursive=True): + if f.endswith(".events.jsonl"): continue + try: + with open(f) as fh: first = fh.readline() + except OSError: continue + if '"closed_loop":true' not in first.replace(" ", ""): continue + out.append(f) + return sorted(out) + +def split_rounds(path): + """Yield per-round [(self, enemy)] from a recording, using its round map.""" + rf = path.replace(".jsonl", ".jsonl.rounds.json") + bounds = [] + if os.path.exists(rf): + try: + for r in json.load(open(rf))["rounds"]: + bounds.append((r["startTick"], r["startTick"] + r["count"])) + except Exception: bounds = [] + rows = [] + with open(path) as fh: + for line in fh: + if '"tick"' not in line: continue + try: d = json.loads(line) + except ValueError: continue + if "se" in d and "ee" in d: rows.append((d["tick"], d["se"], d["ee"])) + if not rows: return [] + if not bounds: bounds = [(rows[0][0], rows[-1][0] + 1)] + rounds = [] + for s, e in bounds: + r = [(se, ee) for t, se, ee in rows if s <= t < e] + if not r: continue + # trim the trailing both-disabled tail: a dead bot sits at ~0 forever + last = max(i for i, (a, b) in enumerate(r) if a > 0 and b > 0) + rounds.append(r[:last + 1]) + return rounds + +def main(): + files = recordings() + rounds = [] + for f in files: rounds += split_rounds(f) + if not rounds: + print("no closed-loop recordings found"); return + N = sum(len(r) for r in rounds) + se, ee = array.array("d"), array.array("d") + for r in rounds: + for a, b in r: se.append(a); ee.append(b) + print(f"recordings={len(files)} rounds={len(rounds)} ticks={N}\n") + + THR = [5, 10, 15, 20, 25] + print("=== A) SELF energy below a floor candidate (share of ticks) ===") + print(f"{'floor':>5} {'pct':>7} {'rounds hit':>10} {'med run':>8} {'p90 run':>8} {'max run':>8}") + for t in THR: + tot, hit, lens = 0, 0, [] + for r in rounds: + cur, got = 0, False + for a, _ in r: + if a <= t: cur += 1; tot += 1; got = True + elif cur: lens.append(cur); cur = 0 + if cur: lens.append(cur) + hit += 1 if got else 0 + lens.sort() + print(f"{t:>5} {100*tot/N:>6.2f}% {hit:>10} " + f"{statistics.median(lens) if lens else 0:>8.0f} " + f"{lens[int(.9*len(lens))] if lens else 0:>8} " + f"{lens[-1] if lens else 0:>8}") + + print("\n=== B) the owner's \"both low, nobody firing\" situation ===") + for t in THR: + both = sum(1 for a, b in zip(se, ee) if a <= t and b <= t) + sonly = sum(1 for a, b in zip(se, ee) if a <= t < b) + eonly = sum(1 for a, b in zip(se, ee) if b <= t < a) + print(f" both<={t:>2}: {100*both/N:6.3f}% self-only {100*sonly/N:6.2f}%" + f" enemy-only {100*eonly/N:6.2f}%") + + print("\n=== C) who crosses a low-energy line FIRST (per round) ===") + for t in [10, 15, 20, 25]: + s = e = n = 0 + for r in rounds: + fs = next((i for i, x in enumerate(r) if x[0] <= t), None) + fe = next((i for i, x in enumerate(r) if x[1] <= t), None) + if fs is None and fe is None: n += 1 + elif fs is None or (fe is not None and fs < fe): s += 1 + else: e += 1 + m = len(rounds) + print(f" t={t:>2}: self-first {s:>5} ({100*s/m:5.1f}%) " + f"enemy-first {e:>5} ({100*e/m:5.1f}%) neither {n:>4} ({100*n/m:4.1f}%)") + + print("\n=== D) \"the enemy can no longer fire\" (server rejects energy <= power) ===") + for p in (0.4, 1.0, 1.95, 3.0): + c = sum(1 for b in ee if b <= p) + c2 = sum(1 for a, b in zip(se, ee) if b <= p and a > 20) + print(f" enemy <= {p:>4}: {100*c/N:6.3f}% and self>20: {100*c2/N:6.3f}%") + + print("\n=== E) exhaustion-trigger OPPORTUNITY: enemy low while we are healthy ===") + for t in [10, 20, 30]: + row = " ".join(f"self>{fl}: {100*sum(1 for a,b in zip(se,ee) if b<=t and a>fl)/N:6.2f}%" + for fl in (0, 20, 25)) + print(f" enemy<={t:>2} {row}") + + print("\n=== F) the ALREADY-SHIPPED finisher (enemy<20 & self>enemy & dist<300) ===") + c = 0 + # dist needs the raw file; recompute over the whole trimmed corpus + for f in files: + with open(f) as fh: + for line in fh: + if '"se"' not in line: continue + try: d = json.loads(line) + except ValueError: continue + if "se" not in d: continue + if d["ee"] < 20 and d["se"] > d["ee"] and \ + ((d["ex"]-d["sx"])**2 + (d["ey"]-d["sy"])**2) ** .5 < 300: + c += 1 + print(f" {c} ticks ({100*c/N:.4f}% of the trimmed corpus)") + +if __name__ == "__main__": + main() diff --git a/common_libs/tests/test_tfil_commit_env.nim b/common_libs/tests/test_tfil_commit_env.nim index bc9bee8..e3ca46e 100644 --- a/common_libs/tests/test_tfil_commit_env.nim +++ b/common_libs/tests/test_tfil_commit_env.nim @@ -40,6 +40,7 @@ import std/[os, json, random, math, sequtils, sets] import std/strutils except fromHex # `fromHex` would clash with color.fromHex import gun_harness/gun_interface +import movements/ram_decision # Private-field access: include (do NOT import) the shipped mover. include movements/the_floor_is_lava # j165: the TFIL-RING fork's arrival commitment. `tfil_ring_replay` includes @@ -1607,6 +1608,110 @@ when declared(TfilRingCommitArrival): firstArmed >= 0 and armed[firstArmed].picked and abs(states[firstArmed].selfSpeed) < 4.0 +# ── j160: the energy-reserve FIRING FLOOR + ENEMY-EXHAUSTION ram trigger ───── +proc testJ160() = + ## j160 — the energy-reserve FIRING FLOOR (TR_RAM_FLOOR_ENERGY) and the + ## ENEMY-EXHAUSTION ram trigger (TR_RAM_ENEMY_ENERGY). Both default 0.0 = + ## today's behaviour. Pure logic only; no bot, no server. + const F = 5.0 + + # 1. DEFAULT PARITY, floor: with the knob unset the floor blocks NOTHING over + # a grid that includes every measured low-energy region (<=5: 9.4% of + # ticks in the 8612-round closed-loop corpus, p01 self energy = 0.2). + var blocked = 0 + for e in [-1.0, 0.0, 0.1, 1.0, 2.5, 5.0, 7.0, 20.0, 46.0, 100.0, 120.0]: + if fireFloorBlocks(0.0, e): inc blocked + check "j160: TR_RAM_FLOOR_ENERGY unset (=0) suppresses fire on NO input, " & + "so the default path is byte-for-byte today's (" & $blocked & " blocked)", + blocked == 0 + + # 2. DEFAULT PARITY, trigger: with the knob unset the reason over a grid is + # the PRE-j160 result — the new arm is unreachable, and every old arm still + # returns exactly what it returned before. + var newArm, mismatch: int + for dist in [10.0, 100.0, 299.0, 301.0, 500.0]: + for se in [0.5, 5.0, 19.0, 21.0, 60.0, 100.0]: + for ee in [0.0, 1.0, 5.0, 19.9, 20.0, 40.0, 100.0]: + let inp = RamInputs(dist: dist, selfEnergy: se, enemyEnergy: ee) + let got = ramTrigger(inp) + if got == rrExhausted: inc newArm + # the pre-j160 body, verbatim + var want: RamReason = rrNone + if ee > 0.0: + if dist < RamFinisherDist and ee < RamFinisherEnergy and se > ee: want = rrFinisher + elif se < RamDesperationEnergy and ee < RamDesperationEnergy and dist < RamDesperationDist: want = rrDesperation + if got != want: inc mismatch + check "j160: TR_RAM_ENEMY_ENERGY unset (=0) makes the exhaustion arm " & + "unreachable (" & $newArm & " hits) and leaves the old finisher / " & + "desperation verdicts identical (" & $mismatch & " mismatches over 210 " & + "input combinations)", + newArm == 0 and mismatch == 0 + + # 3. the floor suppresses AT the threshold, not above it. + check "j160: the floor blocks AT the threshold (self == 5.0 <= floor 5.0)", + fireFloorBlocks(F, 5.0) + check "j160: the floor blocks just below it and not just above it — one " & + "tick of hysteresis, no dead band", + fireFloorBlocks(F, 4.999) and not fireFloorBlocks(F, 5.001) + + # 4. it never suppresses while we are healthy, at ANY floor setting. + var healthy = 0 + for floor in [0.5, 1.0, 5.0, 20.0, 25.0, 40.0]: + for e in [floor, 46.0, 60.0, 100.0, 120.0]: + if e > floor and fireFloorBlocks(floor, e): inc healthy + check "j160: the floor NEVER blocks above its own threshold — healthy energy " & + "fires for every floor/energy pair (" & $healthy & " violations)", + healthy == 0 + + # 5. the trigger switches to ram EXACTLY at the tolerance, no earlier. + let inp2 = RamInputs(dist: 100.0, selfEnergy: 60.0, enemyEnergy: 10.0) + check "j160: the exhaustion trigger fires EXACTLY at TR_RAM_ENEMY_ENERGY " & + "(enemy 10.0 <= tol 10.0) and not one tick above (10.001)", + ramTrigger(inp2, enemyEnergyTol = 10.0) == rrExhausted and + ramTrigger(RamInputs(dist: 100.0, selfEnergy: 60.0, enemyEnergy: 10.001), + enemyEnergyTol = 10.0) != rrExhausted + + # 6. the surplus guard survives: 0.6/contact is applied to BOTH bots, so we + # only ram an exhausted enemy while WE hold the surplus. + check "j160: the exhaustion trigger keeps the finisher's energy-surplus " & + "guard — an exhausted enemy while WE are lower is a ram we lose", + ramTrigger(RamInputs(dist: 250.0, selfEnergy: 2.0, enemyEnergy: 3.0), + enemyEnergyTol = 10.0) != rrExhausted + + # 7. it is the finisher's own shape: the existing range guard still applies. + check "j160: the exhaustion trigger keeps the finisher's 300px range guard " & + "(enemy exhausted at 301px is not a ram)", + ramTrigger(RamInputs(dist: 301.0, selfEnergy: 60.0, enemyEnergy: 3.0), + enemyEnergyTol = 10.0) != rrExhausted + + # 8. COMPOSITION: ramming WINS. The floor is the reserve FOR the ram, so once + # the ram is engaged the reserve is being spent, not held. No starvation: + # the floor alone can never make us unable to close. + check "j160: RAMMING wins the conflict — at self energy 0.1 (below any " & + "sane floor) an engaged ram is never floor-blocked, so the two " & + "compose instead of deadlocking each other", + fireFloorBlocks(F, 0.1, ramming = true) == false and + fireFloorBlocks(F, 0.1, ramming = false) == true + + # 9. and the floor can never be engaged at all without self energy being + # genuinely low — the guard the owner asked for, stated as a property. + var unsafe = 0 + for floor in [0.5, 5.0, 20.0, 25.0]: + for e in [0.0, 1.0, 10.0, 25.0, 50.0, 100.0]: + if fireFloorBlocks(floor, e, ramming = false) and e > floor: inc unsafe + check "j160: the floor is a LOW-ENERGY guard only — it can never suppress " & + "fire while we are healthy, in any configuration (" & $unsafe & ")", + unsafe == 0 + + # 10. both knobs together: the exhausted trigger still fires while the floor + # is at full strength, and the floor still holds when no ram is engaged. + check "j160: both knobs ON compose — exhaustion (enemy 3, us 60) still " & + "ram-bypasses the floor, and a non-ramming low-energy tick still holds", + fireFloorBlocks(F, 3.0, ramming = false) and + ramTrigger(RamInputs(dist: 100.0, selfEnergy: 60.0, enemyEnergy: 3.0), + enemyEnergyTol = 10.0) == rrExhausted and + not fireFloorBlocks(F, 3.0, ramming = true) + # ── driver ─────────────────────────────────────────────────────────────────── testDefaultParity() @@ -1624,6 +1729,7 @@ when declared(loadTfilCommitEnv): testJ153() when declared(TfilRingCommitArrival): testJ165() + testJ160() if failures > 0: echo "\n", failures, " check(s) FAILED" diff --git a/docs/env_reference.md b/docs/env_reference.md index 95bdeef..b9eec3e 100644 --- a/docs/env_reference.md +++ b/docs/env_reference.md @@ -226,6 +226,8 @@ shooting *look* like missing). | `TR_POWER_POLICY` | `1` | `0` = uncapped control arm (today's behaviour without the energy policy) | | `TR_POWER_LOG` | off | **presence-based**: if the var exists at all (even `=0`) log each power decision | | `TR_RAM_LOG` | off | **presence-based**: log ram on/off with the reason | +| `TR_RAM_FLOOR_ENERGY` | `0.0` | j160 firing floor: at/below this self energy we start no NEW shot, holding a ram reserve. `0` = off. `~5` = one p=1.0 return hit + two 0.1 shots. Bypassed while ramming | +| `TR_RAM_ENEMY_ENERGY` | `0.0` | j160 exhaustion trigger: last-scanned enemy energy `<=` this -> ram mode. `0` = off. Keeps the finisher's energy-surplus and 300px guards | | `TR_MOVEMENT_LOG` | off | **presence-based**: log movement band/class changes | | `TR_TMHORIZON_LOG` | off | value-based: `1` = let the horizon TM gun log its thinking per shot | | `TR_ENV_REPORT` | `1` | print the boot-time `[env]` report to stdout; `0` suppresses it | diff --git a/docs/ram_floor_exhaustion_ab.md b/docs/ram_floor_exhaustion_ab.md new file mode 100644 index 0000000..f7a0c9b --- /dev/null +++ b/docs/ram_floor_exhaustion_ab.md @@ -0,0 +1,366 @@ +# j162 — the FIRING FLOOR: exhaustion measurement + a re-sized A/B proposal + +**No battle, server, GUI or A/B was started for this job.** Both knobs remain +default-`0.0`; `out/ModularBot` was not rebuilt. Everything below the divider is +a proposal awaiting the owner's explicit permission. + +--- + +## MEASURED (`common_libs/tests/measure_ram_exhaustion`, offline, state only) + +Same corpus as j160: **8149 closed-loop recordings / 35163 rounds / 34.46M +ticks**. No counterfactual replay (the offline harness scored 0/6 on +closed-loop questions, `docs/offline_harness_trust.md`). + +### 1. We die BROKE, and it is a death event — not a state we sit disabled in + +* **21865 rounds (61.2%) end with self energy crossing 0**; 99.0% of rounds end + in *some* death. Energy on the last tick we were alive: + **median 0.83, mean 2.50, p90 8.90, max 24.83**. + 55.9% of self-deaths at <=1, 84.2% at <=5, 92.8% at <=10, **100% at <=20**. +* Time at energy <= 0 before the round ends: **median 1 tick** (p90 18). The + round ends on the crossing tick. The 1.72%-of-ticks figure is dominated by a + handful of recordings that hold a dead bot for hundreds of ticks — a recorder + artefact, not a lived state. **There is no recoverable disabled window to + defend.** +* The reserve that *would* have absorbed the killing blow (the overshoot of the + final hit): **median 0.40, p75 2.00, p90 6.90, p99 15.0**. A free 5-energy + reserve would have saved 85.1% of self-deaths. + +So the prompt's hypothesis ("it dies at 40 energy, so the floor protects +against nothing") is **false**: the bot dies with nothing, every time. The +floor's premise is real. + +### 2. We cannot climb back out of a low-energy dip + +Energy rises on 0.598% of tick-pairs (~5.87 landed hits/round; **mean landed +power 1.42**, mode 1.0 — not 0.1). Per tick spent at a given level: + +| energy | climb next tick | killed this tick | ratio | +|---|---|---|---| +| <= 3 | 0.130% | 0.830% | dying **6.4x** more likely | +| <= 5 | 0.232% | 0.663% | dying **2.9x** more likely | +| <= 10 | 0.365% | 0.437% | dying 1.2x more likely | +| <= 20 | 0.500% | 0.256% | recovering **2x** more likely | + +Below ~10 energy a landed hit is not coming; below 20 it usually is. **A floor +at 20 would therefore block the only zone where recovery is actually +plausible.** + +### 3. What the floor costs + +| floor | % ticks blocked | rounds | med run | p90 run | mean run | % runs ending in death | bank @1.0p | +|---|---|---|---|---|---|---|---| +| 3 | 7.64% | 23086 | 34 | 299 | 98 | 80.8% | 8.2 | +| 5 | **9.58%** | 23739 | 53 | 330 | 118 | 78.0% | **9.9** | +| 10 | 14.52% | 25211 | 89 | 433 | 162 | 70.3% | 13.5 | +| 20 | 24.74% | 27799 | 139 | 621 | 236 | 60.0% | 19.6 | + +"Bank" = mean suppressed run x `p/(10+2p)` energy/tick, the gun-heat ceiling +(`heat = 1 + p/5`, cool 0.1/tick). At 0.1 power it is 0.52 energy for floor 5; +at 2.0 power, 16.9. + +**Measured caveat, and it matters:** the recorded energy ledger closes *exactly* +— `start + landed-gains - damage - end = -0.00` over 35065 rounds. **These +captures do not charge the firepower cost**, so the cost column is derived from +the game rules, not read off the data. The landed-hit *power* distribution is +read off the data (mode 1.0, mean 1.42) and is what sets the bracket. + +### 4. Honest read — materially DIFFERENT from the geometry arm + +Firing is **net energy-negative** for this bot on this panel: a landed hit +returns `3p` for `p` spent (break-even hit rate 1/3), and the hit rate cannot +exceed 5.87 hits / 76 shots-per-round heat ceiling = **7.7%**. So not firing +really does bank energy — about 9.9 at floor 5. + +That is the same *kind* of trade the geometry arm made — spend offence, buy +protection — but a different *magnitude*: + +* **Safety claim is stronger.** The geometry arm's safety gain did not convert + into wins. Here the hazard is measured directly: 0.66%/tick death at energy + <= 5, against a p75 overshoot of 2.0 and a bank of 9.9. The bank is above + p75 and near p90 — a genuinely material reserve, not a rounding error. +* **Damage cost is ~an order of magnitude smaller.** Floor 5 suppresses ~4.4 + shots per median run; at 4 damage/hit and a 7.7% hit rate that is ~1.4 + damage per suppressed run, ~2 damage/run. The geometry arm lost **8.83 + damage/run** for its unconverted gain. +* **It is a light touch in time, not in behaviour**: 9.6% of ticks, median run + 53 ticks. Not a blackout. + +**Verdict: the floor is worth an A/B. It is not the clean negative.** But the +honest counterweight is on the record: 78% of suppressed runs still end in +death, and the bank is only reached *because* we stopped shooting. + +**Chosen value: `TR_RAM_FLOOR_ENERGY=5`** — bank 9.9 (above p75 overshoot 2.0, +near p90 6.9) at 7.6%-vs-9.6% less tick cost than 10. Floor 20 is dropped on +the measurement: it costs 24.7% of ticks and sits on top of the <= 20 recovery +window. + +--- + +## PROPOSAL — PRE-REGISTERED, **NOT RUN** + +* Harness: `tools/ab/tournament_run.sh` + `tools/ab/tournament_analyze.py`, + unmodified. Panel: `tools/ab/panel_movement.txt` (frozen 15-opponent movement + panel). Unit of evidence is the opponent, not the battle. One frozen binary + from `git archive` of `j160-ramfloor`. +* **Contamination control — j159's three guarantees, unchanged**: (1) per-arm + `TR_ENV_FILE` in this job's own outdir (`/tmp/j162_floor/env/.env`); + (2) the per-run botdir holds only `.json`, `.sh` and a symlink to the frozen + binary — no `.env`, and the loader does not walk up; (3) **every** run's + `[env]` boot report is checked against its arm, and any disagreement voids + the session. **A session-record check runs BEFORE analysis**, not as a rewrite + afterwards (j159 had a mid-analysis `session.json` rewrite; not repeated here). + +| arm | env | role | +|---|---|---| +| `A_off` | both unset | REFERENCE | +| `B_floor` | `TR_RAM_FLOOR_ENERGY=5` | the floor alone (value from the measurement) | +| `C_exhaust` | `TR_RAM_ENEMY_ENERGY=20` | the exhaustion trigger alone | +| `D_both` | `TR_RAM_FLOOR_ENERGY=5 TR_RAM_ENEMY_ENERGY=20` | the combined policy | + +No arm is inert, so none is dropped: `C_exhaust=20` acts on 7.1% of ticks +(enemy <= 20 while we are > 20) and `D_both` is the only arm that answers the +composition rule. A floor *sweep* arm is deliberately omitted — the measurement +chose the value, and the budget is better spent on n. + +### Size — corrected for the real throughput + +j159 measured **420 battles in 1110 s = 23 battles/min** (not the ~7/min the +previous estimate assumed). MDE scales as `1/sqrt(n)`; the shipped default +movement gate resolved **0.17 wins/run at 210 runs/arm**. For a target MDE of +**0.10 wins/run**: `n = 210 * (0.17/0.10)^2 = 607` runs/arm, rounded up to +**42 runs/opponent = 630 runs/arm**. + +* 15 opponents x 4 arms x 42 runs x 3 rounds = **3780 battles ≈ 2.7 h**. +* Decision-only 2-arm version (`A_off` vs `B_floor`): 15 x 2 x 42 x 3 = + **1890 battles ≈ 1.4 h**, still at MDE 0.10. + +### Metrics (fixed now) + +**Primaries: damage/run, round-win rate.** Mechanism, never a verdict: self +energy at death, ticks spent disabled, shots fired/run, ram-kill count. +Paired per-opponent deltas, mean/SD/SE/95% CI, sign test, sign-flip +permutation, Wilcoxon cross-check, reported MDE. Two-sided. + +### Verdict rule (fixed now) + +Adopt only if BOTH primaries favour the arm with `p(sign-flip) < 0.05` **and** +the effect is at or above the reported MDE. Otherwise do not ship; both knobs +stay `0.0`. A clean null is a fully acceptable result. No subsetting, no +dropping opponents, no re-running to chase a p-value. + +--- + +# j163 — PRE-REGISTRATION: the FIRING FLOOR A/B (2 arms), BEFORE ANY BATTLE + +**Written and committed before a single battle of this design was run.** Nothing +below was chosen after seeing data. Worktree `j160-ramfloor` @ `64e23e2`, one +frozen binary built by `tournament_run.sh` from `git archive HEAD`, both knobs +default-`0.0`, no code changed by this job. + +## Hypothesis + +The bot dies broke: **61.2% of rounds (21865/35753) end with self energy +crossing 0**, and energy on the last alive tick is median 0.83. Below **5** +energy the next tick brings death **2.9x** more often than a landed hit +(0.663%/tick vs 0.232%/tick); the bank of a suppressed run at floor 5 is ~9.9 +energy against a p75 overshoot of 2.0 / p90 6.9 — a free 5-energy reserve would +have saved 85.1% of self-deaths. Firing is net energy-negative here (a landed +hit returns `3p` for `p` spent; the hit rate is capped at 5.87/76 = 7.7%), so +holding a reserve in the sub-5 zone should convert safety into **round wins**. + +## Arms — two, differing in exactly one variable (`TR_MOVEMENT=tfil` pinned) + +| arm | per-arm env file | role | +|---|---|---| +| `A_baseline` | `TR_RAM_FLOOR_ENERGY=0` | REFERENCE (today's shipped behaviour) | +| `B_floor5` | `TR_RAM_FLOOR_ENERGY=5` | treatment, the value chosen by the j162 measurement | + +**The `TR_RAM_ENEMY_ENERGY` (ram-exhaustion) arm is DELIBERATELY EXCLUDED.** +Its own measurement found the trigger is rare at its literal threshold and that +whether it fires is close to a coin flip in direction — it is not a +well-founded mechanism. Excluding it here is a decision, **not an oversight**; +this job tests only the one well-founded mechanism. If the floor is adopted, the +exhaustion trigger needs its own design and its own A/B. + +## Primaries (fixed now) + +1. **round-win rate** (rounds won / rounds fought) — **the deciding primary** +2. **damage/run** + +## THE DAMAGE MDE IS STATED UP FRONT, BECAUSE IT IS BIGGER THAN THE EFFECT + +Expected damage cost of floor 5: **~2 damage/run** (4.4 suppressed shots per +median run x 4 damage/hit x 7.7% hit rate) — versus **8.83 damage/run** for the +already-rejected geometry arm. The design's **damage MDE is 7.65**. The damage +effect is therefore **~3.8x below what this design can resolve**. + +> **Recorded before any data: we EXPECT TO BE UNABLE TO MEASURE THE DAMAGE +> COST DIRECTLY. A null on damage/run is the predicted outcome, not a surprise, +> and must NOT be re-read after the fact as evidence either for or against the +> floor.** The verdict is judged on **ROUND WINS**. The damage MDE is a +> one-sided blind spot of this design, fixed in advance. + +## Counterweight (also on the record before any data) + +**78.0% of suppressed runs still end in death.** The floor protects the tail of +the energy ledger; it is not a shield. A mechanism-positive / outcome-null +result is the fifth such in this campaign (j144, j145, j146, j147, j159). + +## Mechanism metrics (reported, never a verdict) + +* self energy at death (per round); +* share of rounds ending at self energy **<= 0** (baseline **61.2%**); +* shots/run; +* share of ticks with firing suppressed (**floor 5 predicts ~9.6% of ticks, + median suppressed run ~53 ticks**). +* Reported **per opponent as well as pooled**: j159's re-analysis showed a + pooled test hid a real per-opponent effect (safety signal p=0.0008 + per-opponent, null pooled). The unit of evidence is the opponent. + +## Size, MDE and the time floor + +15 frozen opponents x 2 arms x **42 runs** x 3 rounds = **1890 battles**. +MDE ~**0.10 wins/run** (`210 x (0.17/0.10)^2`; 210 was the design that resolved +0.17 wins/run). At the measured 22.7-23.2 runs/min that is **~1.4 h — a floor +on elapsed time, not an estimate**: opponent heterogeneity does not average +down with added runs. If time runs short, the achieved n and the MDE actually +reached are reported exactly; the panel and the arms are **not** silently +shrunk. + +## Contamination controls (all three, in order) + +1. Each arm is launched with `TR_ENV_FILE` pointing at a **per-arm file this + job generated** in its own directory (`/tmp/j163_env/.env`) — never a + shell export, because the dotenv loader gives the FILE priority. The file + dir is deliberately **outside** `--outdir` (`tournament_run.sh` `rm -rf`s + the outdir). This matters: the owner has an 18 KB `.env` at + `ModularBot_garage/out/.env` in the main tree. The per-run botdir holds only + `.json`, `.sh` and a symlink to the frozen binary; the frozen binary's own + directory holds no `.env`; the loader's fallbacks are `./.env` then + exe-adjacent with **no parent walk**, so the owner's file is unreachable. +2. **Every run's `[env]` boot block is verified against its arm as runs + complete** — `TR_RAM_FLOOR_ENERGY` and `TR_MOVEMENT=tfil` — and `mis-set` + is counted and reported. A previous session was invalidated-risk because + this was checked too late. +3. The session record is **read before analysis, never rewritten**. j159 had a + mid-analysis `session.json` rewrite; declaring `TR_MOVEMENT` explicitly + disables the analyzer's leaked-`TR_MOVEMENT` fatal check, so the built-in + leak guard is **not trustworthy here** — the explicit per-run `[env]` + verification above is the primary control. If the guard misbehaves it is + **reported as a finding, not worked around**. + +## Verdict rule (fixed now, two-sided) + +**Adopt** only if round-win rate favours `B_floor5` with a per-opponent +**sign-flip permutation p < 0.05** AND the effect is at or above the reported +MDE. Otherwise **do not ship**; `TR_RAM_FLOOR_ENERGY` stays `0.0`. No +subsetting, no dropping opponents, no re-running to chase a p-value, no +reinterpreting the bar after seeing the data. **A clean null is a fully +acceptable result** — and a null here licenses only "no effect >= MDE is +detectable at this design", never "the knob is harmless". + +--- + +## MEASURED + +*(appended after the battles — everything above was committed first, at +`6cfb169`)* + +### MEASURED — the live A/B, 450 runs/arm, 2700 rounds (j163) + +* **Provenance.** Frozen 15-opponent movement panel + (`tools/ab/panel_movement.txt`), 15 x 2 x **30 runs** x 3 rounds = + **450 runs/arm, 2700 rounds**, `conc=6`, **0 failed, 0 never started**. + Frozen binary `d9a39c3b8472…`, `TR_MOVEMENT=tfil` in both arms; the only + difference is `TR_RAM_FLOOR_ENERGY` `0` (`A_floor0`, reference) vs `5` + (`B_floor5`). +* **Env verification — 0 mis-set.** Every run's `[env]` boot block was checked + against its arm as the session progressed, live at **24 / 193 / 410 / 826 / + 900** runs completed: no disagreement at any checkpoint. Per-arm + `TR_ENV_FILE` in the session's own directory, botdir without a `.env`, loader + does not walk up parents. This is contamination control #2 from the + pre-registration, satisfied. +* **DEVIATION FROM THE PRE-REGISTRATION, DISCLOSED.** The design asked for + **42 runs/opponent** (1890 battles, MDE ~0.10 wins/run). Measured throughput + was **14-22 runs/min**, not the assumed 22.7-23.2, so the wall-clock cost of + the pre-registered n was not affordable. As the pre-registration required, + the **panel and the arms were NOT shrunk**: the full 15 opponents and both + arms were kept and the **runs per opponent were reduced to 30**. The MDE + actually reached is reported below and is the honest resolution limit of this + run. No opponent was dropped, no arm was re-run to chase a p-value. + +**Pooled dashboard (descriptive, NOT the verdict):** + +| arm | runs | dmg/run | wins/run | round wins | round win rate | +|---|---:|---:|---:|---:|---:| +| `A_floor0` | 450 | 113.65 | 0.200 | — | **40.30%** | +| `B_floor5` | 450 | 112.70 | 0.182 | — | **39.70%** | + +**Verdict layer** (per-opponent paired deltas, arm − reference; the sign-flip +is the exact 2^15 permutation the pre-registration names as the decision test): + +| metric | mean Δ | 95% CI | p(sign-flip) | sign test | Wilcoxon p | **MDE reached** | +|---|---:|---|---:|---:|---:|---:| +| **round-win rate** | **-0.59 pp** | [-3.90, +2.72] | **0.7676** | 1.00 | 0.84 | **4.73 pp** | +| wins/run | -0.0178 | [-0.117, +0.082] | 0.7676 | 1.00 | 0.84 | 0.1420 | +| damage/run | -0.95 | [-3.88, +1.98] | 0.5298 | 1.00 | 0.84 | 4.19 | + +### HEADLINE FINDING — the mechanism barely fired; the offline energy corpus did not survive contact with the live game + +1. **Suppression was 0.04% of ticks, not the 9.6% the offline ruler predicted** — + a **~200x** smaller effect. The pre-registration's own mechanism metric + (`share of ticks with firing suppressed`, predicted ~9.6%, median suppressed + run ~53 ticks) is the number that failed, and it failed by two orders of + magnitude. +2. **Only 4.8% of shots are ever taken in the low-energy zone, and the floor + removed 14% of those.** The gate therefore touches a small slice of a small + slice: a bot that almost never wants to fire at low energy. The pre- + registration's expected damage cost (~2 damage/run) was the arithmetic + consequence of the 9.6% figure; with 0.04% it is ~200x smaller still, which + is why the damage MDE (4.19) is unreachable by construction and not by bad + luck. +3. **Rounds ending at self energy <= 0: 40.6% live vs 61.2% implied by the + offline corpus** (the j162 baseline the pre-registration quoted). The + recorded-fixture corpus over-states how often we die broke by ~1.5x. This is + the second, independent way the same corpus mis-called the live game. +4. **Median self energy at death 14.1 -> 15.5** — the floor moved the death + energy by +1.4, real but tiny, and nowhere near the "we sit disabled" state + the floor was built for. +5. **The pre-registered blind spot was called correctly.** The pre-registration + states, in advance, that we expect to be unable to measure the damage cost + directly and that a damage null must not be re-read afterwards as evidence. + That call was right, and it is the reason this run cannot be misread. + +### VERDICT — DO NOT ADOPT + +1. **The pre-registered verdict rule is not met.** Adopt required round-win + rate favouring `B_floor5` with per-opponent sign-flip p < 0.05 AND the + effect at or above the reported MDE. Observed: **-0.59 pp against**, p = + **0.7676**, under the 4.73 pp MDE. Round wins and wins/run are the same + null (p = 0.7676, MDE 0.1420 wins/run). +2. **`TR_RAM_FLOOR_ENERGY` stays `0.0`.** Do not adopt, do not ship, and **do + not re-test this knob.** The design that could resolve a real effect does + not exist at an affordable run count, and the mechanism it was built to + suppress is nearly absent in the live game. Re-running buys resolution on an + effect that is not there. +3. **A null here does NOT prove the knob inert — the opposite.** The mechanism + fired on 0.04% of ticks. This run licenses only: "no effect >= 4.73 pp of + round-win rate at 450 runs/arm". It says nothing about the ~0.04% of ticks + it did suppress, because too few of them existed to measure. +4. **The generalisable finding is the corpus, not the knob.** The offline + energy corpus over-predicted both the size of the low-energy firing window + (9.6% -> 0.04%) and the rate of dying broke (61.2% -> 40.6%). An offline + ruler built on recorded fixtures is only as representative as the fixtures; + the death-energy corpus does not represent the live energy ledger. Future + offline rulers for exhaustion must be calibrated against a live + death-energy distribution before their predictions are pre-registered as + expectations, not just as a rationale. + +**Ship state: unchanged. `TR_RAM_FLOOR_ENERGY=0.0` and `TR_RAM_ENEMY_ENERGY=0.0` +remain the shipped defaults, and the ram path keeps its pre-j160 behaviour.** +This is the sixth mechanism-positive-or-presumed / outcome-not-positive result +in the campaign (j144, j145, j146, j147, j159, j163) — and the first where the +mechanism was not merely ineffective but **~200x smaller than the offline ruler +said it would be**.