diff --git a/ModularBot_garage/src/ModularBot.nim b/ModularBot_garage/src/ModularBot.nim index 674a2e5..92e0b77 100644 --- a/ModularBot_garage/src/ModularBot.nim +++ b/ModularBot_garage/src/ModularBot.nim @@ -620,6 +620,15 @@ method onBulletHitWall*(bot: ModularBot, e: BulletHitWallEvent) = bot.strafeMover.removeBulletNear(e.bullet.x, e.bullet.y) bot.surfMover.removeBulletNear(e.bullet.x, e.bullet.y) bot.learnedMover.removeBulletNear(e.bullet.x, e.bullet.y) + # Real wall endpoint of an ENEMY bullet. On the running server (0.35.5) a + # BulletHitWallEvent is delivered only to the bullet's OWNER + # (`addPrivateBotEvent(bullet.botId, ...)`), so this branch is a no-op live; + # it is wired so a server that exposes enemy wall hit-points feeds the + # exact-geometry wave resolution instead of the arrival-deadline proxy. + if e.bullet.ownerId != getMyId(): + discard bot.learnedMover.resolveEnemyBullet( + e.bullet.x, e.bullet.y, degToRad(e.bullet.direction), + e.bullet.ownerId, bot.tick, false) method onBulletHitBullet*(bot: ModularBot, e: BulletHitBulletEvent) = discard bot.resolveOwnBullet(e.bullet.bulletId) # bullet-vs-bullet: free the slot @@ -630,6 +639,13 @@ method onBulletHitBullet*(bot: ModularBot, e: BulletHitBulletEvent) = bot.strafeMover.removeBulletNear(e.bullet.x, e.bullet.y) bot.surfMover.removeBulletNear(e.bullet.x, e.bullet.y) bot.learnedMover.removeBulletNear(e.bullet.x, e.bullet.y) + # `e.bullet` is OUR bullet; `e.hitBullet` is the bullet it struck. When that + # is an ENEMY bullet, the event exposes its real endpoint + heading - a real + # resolution for a wave that would otherwise wait for the deadline. + if e.hitBullet.ownerId != getMyId() and e.hitBullet.bulletId != 0: + discard bot.learnedMover.resolveEnemyBullet( + e.hitBullet.x, e.hitBullet.y, degToRad(e.hitBullet.direction), + e.hitBullet.ownerId, bot.tick, false) method onHitByBullet*(bot: ModularBot, e: HitByBulletEvent) = # Feeds the ram bullet-rain abort window. Accumulate REAL ENERGY (the server's @@ -637,6 +653,11 @@ method onHitByBullet*(bot: ModularBot, e: HitByBulletEvent) = # energy/turn rate (see movements/ram_decision.bulletDamage). bot.ramDmgAccum += bulletDamage(e.bullet.power) bot.moveTracker.registerHit(e.bullet.power, e.bullet.direction, getX(), getY()) + # A real ENEMY bullet hit us: its endpoint (our impact point) + heading give + # the wave's EXACT straight line, so resolve and drop that live wave now. + discard bot.learnedMover.resolveEnemyBullet( + e.bullet.x, e.bullet.y, degToRad(e.bullet.direction), + e.bullet.ownerId, bot.tick, true) proc enemyEnergyForReport(bot: ModularBot): float = ## Last-scanned energy of the enemy the report is about (current target, diff --git a/ModularBot_garage/src/env_report.nim b/ModularBot_garage/src/env_report.nim index 413f955..9d14ad1 100644 --- a/ModularBot_garage/src/env_report.nim +++ b/ModularBot_garage/src/env_report.nim @@ -511,7 +511,7 @@ proc knownEnvNames*(): seq[string] = LearnedDecayEveryEnv, LearnedDecayShiftEnv, LearnedAlphaEnv, LearnedTravelEnv, LearnedReversalEnv, LearnedPrefDistEnv, LearnedDistBandEnv, LearnedRadialFracEnv, LearnedWallMarginEnv, - LearnedGlobalEnv, LearnedLabelEnv, LearnedLogEnv, + LearnedGlobalEnv, LearnedLabelEnv, LearnedRealEventsEnv, LearnedLogEnv, # harness vars (read by the test framework, inherited by the bot, so they # must NOT be reported as typos) "TR_SERVER_JAR", "TR_BATTLE_RUNNER", "TR_BATTLE_RUNNER_DIR", diff --git a/ModularBot_garage/tests/test_learned_surfer.nim b/ModularBot_garage/tests/test_learned_surfer.nim index 3243ea0..e57ddb0 100644 --- a/ModularBot_garage/tests/test_learned_surfer.nim +++ b/ModularBot_garage/tests/test_learned_surfer.nim @@ -88,6 +88,66 @@ proc counters(m: LearnedSurferModule): int = if c != 0'u8: inc n n +proc testRealEvents() = + ## `TR_LEARNED_REAL_EVENTS`: a wave resolves on the REAL bullet event (exact + ## origin->endpoint line) instead of the arrival deadline, and is dropped the + ## moment it resolves (ghost cleanup). Default-off parity is also pinned. + const Ex = 400.0 + const Ey = 100.0 + const Ux = 400.0 + const Uy = 300.0 + + proc wsAt(tick: int; eEnergy: float): WorldState = + WorldState( + enemyX: Ex, enemyY: Ey, enemyEnergy: eEnergy, + selfX: Ux, selfY: Uy, selfSpeed: 8.0, selfHeading: 0.0, + arenaWidth: 800.0, arenaHeight: 600.0, tick: tick, + enemies: @[EnemyInfo(id: 1, x: Ex, y: Ey, + heading: 180.0, speed: 0.0, energy: eEnergy)]) + + # ── parity: with the knob off a real event changes nothing ────────────── + putEnv(LearnedLabelEnv, "histogram") + putEnv(LearnedRealEventsEnv, "") + loadLearnedEnv() + var mOff = initLearnedSurfer() + discard mOff.computeMove(wsAt(0, 100.0)) # tick 0: baseline energy sample + discard mOff.computeMove(wsAt(1, 99.0)) # tick 1: a 1.0 firepower drop + check "RE default-off: the fire is detected as a live wave", + mOff.liveWaves == 1 + check "RE default-off: resolveEnemyBullet is a no-op", + (not mOff.resolveEnemyBullet(Ux, Uy, degToRad(90.0), 1, 13, true)) and + mOff.liveWaves == 1 + + # ── ON: the exact origin->endpoint line resolves the wave immediately ──── + putEnv(LearnedRealEventsEnv, "1") + loadLearnedEnv() + var mOn = initLearnedSurfer() + discard mOn.computeMove(wsAt(0, 100.0)) + discard mOn.computeMove(wsAt(1, 99.0)) + check "RE on: the fire is detected as a live wave", mOn.liveWaves == 1 + # endpoint = our position on the centre line (GF 0 -> bin 15) at ~nominal + # (200 px at speed 17 -> ~12 ticks after the fire at tick 1) + check "RE on: the real hit endpoint resolves the wave", + mOn.resolveEnemyBullet(Ux, Uy, degToRad(90.0), 1, 13, true) + check "RE on: the wave is dropped at once (ghost cleanup)", mOn.liveWaves == 0 + check "RE on: a real resolution is counted", mOn.resolvedReal == 1 + check "RE on: the exact straight line lands in the centre bin", + mOn.glob[15] == 1 + check "RE on: the real flight time is recorded", mOn.lastFlightErr > -20.0 + + # ── ON: an unmatched wave still resolves (as a WALL MISS) at the deadline ─ + var mMiss = initLearnedSurfer() + discard mMiss.computeMove(wsAt(0, 100.0)) + discard mMiss.computeMove(wsAt(1, 99.0)) + check "RE on: a wave is pending before any event", mMiss.liveWaves == 1 + for t in 2..<40: discard mMiss.computeMove(wsAt(t, 99.0)) + check "RE on: the unmatched wall wave eventually resolves", + mMiss.liveWaves == 0 and mMiss.resolvedDead >= 1 + + putEnv(LearnedRealEventsEnv, "") + putEnv(LearnedLabelEnv, "") + loadLearnedEnv() + proc main() = # 1. concept check "MovementModule concept", isMovementModule(LearnedSurferModule) @@ -170,6 +230,8 @@ proc main() = putEnv(LearnedDecayEveryEnv, "") loadLearnedEnv() + echo "" + testRealEvents() echo "" echo "checks=", checks, " failures=", failures if failures > 0: quit(1) diff --git a/common_libs/movements/learned_surfer.nim b/common_libs/movements/learned_surfer.nim index 0dd7d95..5bfb2f5 100644 --- a/common_libs/movements/learned_surfer.nim +++ b/common_libs/movements/learned_surfer.nim @@ -71,6 +71,13 @@ ## = the bot radius as an angle at the wave's ## distance). See docs/movement_campaign.md, ## "outcome label". +## TR_LEARNED_REAL_EVENTS =1: resolve a wave on the REAL bullet event +## (onHitByBullet / an enemy bullet intercepted by +## ours) using the exact origin->endpoint line and +## the real flight time, dropping the wave at once. +## Default off (arrival-deadline proxy). Enemy WALL +## hits are owner-private on server 0.35.5 and stay +## deadline misses - see the const-block note. ## TR_LEARNED_LOG per-decision log line import std/[math, os] @@ -108,8 +115,30 @@ const LearnedWallMarginEnv* = "TR_LEARNED_WALL_MARGIN" LearnedGlobalEnv* = "TR_LEARNED_GLOBAL" LearnedLabelEnv* = "TR_LEARNED_LABEL" + LearnedRealEventsEnv* = "TR_LEARNED_REAL_EVENTS" LearnedLogEnv* = "TR_LEARNED_LOG" + ## Real-event matching (job j131). + ## A wave is normally resolved on the nominal arrival tick + ## `ceil(startDist/speed)`. With `TR_LEARNED_REAL_EVENTS=1` a wave is instead + ## resolved by the REAL server event that carries the bullet's endpoint: + ## * `onHitByBullet` -> the bullet HIT us; endpoint = our impact point; + ## * a bullet-vs-bullet intercept of an ENEMY bullet (our bullet hit theirs) + ## -> the enemy bullet's endpoint/heading are in the event. + ## The bullet's raw straight line (origin at fire -> endpoint) then gives the + ## EXACT GF bin and the real flight time (a cross-check on the energy-drop + ## speed inference), and the wave is dropped immediately (no ghost build-up). + ## + ## ON THE RUNNING SERVER (0.35.5, verified from the server bytecode + + ## `TurnToTickEventForBotMapper`) an ENEMY bullet that hits a WALL produces a + ## `BulletHitWallEvent` only for the bullet's OWNER (`addPrivateBotEvent( + ## bullet.botId, ...)`), and `bulletStates` is filtered to the bot's own + ## bullets. So a wall HIT is NOT observable by the dodger; those waves fall + ## back to the arrival deadline and are labelled a MISS. `resolveEnemyBullet` + ## accepts a wall endpoint anyway so a future/other server can feed it. + RealEventsMatchTol = 12.0 ## max |real flight - nominal| to accept a match, ticks + RealEventsGrace = 8 ## ticks past nominal before an unmatched wave resolves + ## Number of joint (vlat,dist,room,turn) state codes = 4^4. LS_STATES = LS_Q * LS_Q * LS_Q * LS_Q ## Prior-mix weight for the 2-class outcome readout. @@ -132,6 +161,7 @@ var LearnedWallMargin* = 48.0 LearnedGlobal* = false LearnedLabel* = llHistogram + LearnedRealEvents* = false LearnedLog* = false proc getEnvFloat(name: string, default: float): float = @@ -164,6 +194,7 @@ proc loadLearnedEnv*() = LearnedRadialFrac = clamp(getEnvFloat(LearnedRadialFracEnv, 0.35), 0.0, 1.0) LearnedWallMargin = max(0.0, getEnvFloat(LearnedWallMarginEnv, 48.0)) LearnedGlobal = envOn(LearnedGlobalEnv) + LearnedRealEvents = envOn(LearnedRealEventsEnv) LearnedLog = envOn(LearnedLogEnv) LearnedLabel = case getEnv(LearnedLabelEnv, "").strip().toLowerAscii() @@ -219,6 +250,8 @@ proc roomToWall(px, py, dx, dy, arenaW, arenaH: float64): float64 = type LSWave = object + ownerId: int ## enemy that fired (energies are per-enemy) + fireTick: int ## `ws.tick` at the fire tick (real flight time) originX, originY: float64 bearing: float64 ## enemy -> us at the fire tick (centre line) speed: float64 @@ -246,6 +279,9 @@ type prevHeading: float64 debugGraphics*: bool decisions*: int ## decisions taken (diagnostic) + resolvedReal*: int ## waves resolved by a real bullet event + resolvedDead*: int ## waves resolved on the arrival deadline + lastFlightErr*: float ## real flight - nominal flight, last resolution proc resetRound*(m: var LearnedSurferModule) = ## Per-ROUND reset: the waves and the smoothed global prior are per round, but @@ -261,6 +297,9 @@ proc resetRound*(m: var LearnedSurferModule) = m.prevY = 0.0 m.prevHeading = 0.0 m.decisions = 0 + m.resolvedReal = 0 + m.resolvedDead = 0 + m.lastFlightErr = 0.0 for i in 0.. one training sample + removal. Shared by the arrival + ## deadline (unobserved wall misses) and the real-event path, so a wave is + ## ALWAYS trained and dropped exactly once - no ghost accumulation. + let w = m.waves[idx] + let nom = w.startDist / max(w.speed, 1e-9) + let realFlight = float(currentTick - w.fireTick) + m.lastFlightErr = realFlight - nom + if LearnedLog: + echo "[learned] resolve bin=", bin, " state=", w.stateRow, "/", + w.stateCol, " d=", w.startDist.int, " e=", w.originX.int, ",", + w.originY.int, " flight=", realFlight.int, " nominal=", nom.int, + " hit=", hit + m.learnWave(w, bin, hit) + m.waves.del(idx) + +proc missileLineBin(w: LSWave, x, y, headingRad: float): int = + ## GF bin of the bullet's real straight line through `(x,y)` (the endpoint), + ## falling back to the real heading when the endpoint is degenerate. This is + ## the EXACT geometry: origin at the fire tick + real endpoint, no timing + ## guess. + let maxA = mea(w.speed) + if maxA < 1e-9: return gfToBin(0.0) + let ex = x - w.originX + let ey = y - w.originY + let lineDir = + if hypot(ex, ey) > 1.0: arctan2(ey, ex) + else: headingRad + gfToBin(clamp(wrapPi(lineDir - w.bearing) / maxA, -1.0, 1.0)) + +proc resolveEnemyBullet*(m: var LearnedSurferModule, x, y, headingRad: float, + ownerId, currentTick: int, hit: bool): bool = + ## Resolve (and DROP) the live wave matching a REAL enemy-bullet event. + ## + ## `x,y` the bullet's real endpoint (our impact point for a HIT, the + ## wall point for a wall hit, the intercept point for a + ## bullet-vs-bullet hit), + ## `headingRad` the bullet's real heading (fallback when the endpoint is + ## degenerate), + ## `hit` true only for a HIT on us. + ## The exact straight line origin->endpoint sets the label's GF bin and the + ## real flight time `currentTick - fireTick` is recorded, which cross-checks + ## the energy-drop speed inference. No-op unless `TR_LEARNED_REAL_EVENTS=1`. + result = false + if not LearnedRealEvents or m.waves.len == 0: return + # The wave whose nominal arrival is closest to now is the one this bullet + # belongs to; ownerId disambiguates when several enemies are firing. + var best = -1 + var bestKey = Inf + for i in 0..= 0 and w.ownerId != ownerId: continue + let key = abs(float(currentTick - w.fireTick) - + w.startDist / max(w.speed, 1e-9)) + if key < bestKey: + bestKey = key + best = i + if best < 0: # no wave from that enemy: fall back to time-only matching + for i in 0.. RealEventsMatchTol: return + let w = m.waves[best] + let bin = missileLineBin(w, x, y, headingRad) + m.resolveWaveIdx(best, bin, hit, currentTick) + inc m.resolvedReal + result = true + proc predictHit*(m: LearnedSurferModule, row, col, g: int): float = ## P(hit | state, candidate bin g) — the `outcome` danger (lower = safer), ## from the 2-class counted SBC read out with the per-cell posterior and @@ -416,6 +529,7 @@ proc detectFire(m: var LearnedSurferModule, id: int, ex, ey, eenergy: float, let col = code(room, RoomEdges) * LS_Q + code(turn, TurnEdges) m.waves.add LSWave( + ownerId: id, fireTick: ws.tick, originX: ex, originY: ey, bearing: bearing, speed: bspeed, startDist: d, power: drop, ticksLeft: max(1, int(ceil(d / max(bspeed, 1e-9)))), @@ -445,24 +559,27 @@ proc computeMove*(m: var LearnedSurferModule, ws: WorldState): MoveCommand = m.waves[i].fresh = false # created this tick: not one tick old yet else: dec m.waves[i].ticksLeft - if m.waves[i].ticksLeft <= 0: + # With real events on, wait `RealEventsGrace` ticks past the nominal arrival + # so a late HitByBullet can still claim the wave; a wave no event claims is + # a WALL MISS (the wall event is owner-private - see the const block). + let deadline = if LearnedRealEvents: -RealEventsGrace else: 0 + if m.waves[i].ticksLeft <= deadline: let w = m.waves[i] let maxA = mea(w.speed) if maxA >= 1e-9: let off = wrapPi(arctan2(botY - w.originY, botX - w.originX) - w.bearing) let bin = gfToBin(clamp(off / maxA, -1.0, 1.0)) - # llOutcome label: did THIS wave hit us? Our own energy dropped since - # the fire tick. (One wave is live at a time in 1v1; ramming also drops - # energy, so this is a proxy, not an oracle.) - let hit = ws.selfEnergy < w.selfEnergyAtFire - 0.01 - m.learnWave(w, bin, hit) - if LearnedLog: - echo "[learned] resolve bin=", bin, " state=", w.stateRow, "/", - w.stateCol, " d=", w.startDist.int, " e=", w.originX.int, ",", - w.originY.int - m.waves.del(i) - else: - inc i + # llOutcome label: did THIS wave hit us? The energy drop is the proxy; + # with real events on the HIT is taken from `onHitByBullet` instead, so + # an unmatched wave is a wall MISS. + let hit = (not LearnedRealEvents) and + ws.selfEnergy < w.selfEnergyAtFire - 0.01 + m.resolveWaveIdx(i, bin, hit, ws.tick) + inc m.resolvedDead + else: + m.waves.del(i) + continue + inc i # ── 3. danger of every candidate bin, summed over every live wave ──────── # llHistogram: precompute the predicted arrival-bin distribution per wave. diff --git a/common_libs/tests/exact_geometry_gate.py b/common_libs/tests/exact_geometry_gate.py new file mode 100644 index 0000000..b2c4ccb --- /dev/null +++ b/common_libs/tests/exact_geometry_gate.py @@ -0,0 +1,224 @@ +#!/usr/bin/env python3 +"""Exact-geometry Gate A/B for the learned movement (job j131). + +Question: does labelling/resolving a wave by the REAL bullet endpoint (the +exact origin->endpoint straight line, available live from `onHitByBullet` and +from a bullet-vs-bullet intercept) fix the danger-map inversion that j128 +measured with the histogram label (`corr = -0.342`) and that j130 replaced with +a suspicious live-computable proxy (`corr = +0.566`)? + +This is the SAME corpus, SAME per-shot extraction and SAME metric as +`outcome_label_gate.py` (which job j130 used), so the three danger maps are +computed under ONE consistent computation and are directly comparable: + + (a) histogram label danger(g) = P(arrival bin = g) (j128) + (b) outcome proxy label danger(g) = P(hit and |g - b_our| <= w) (j130 live) + (c) EXACT bullet line danger(g) = P(|g - b_bullet| <= w) (this job) + +`b_our` is the GF of OUR position at the nominal arrival tick; `b_bullet` is +the GF of the bullet's own straight line (from the recorded fire direction - +exactly the line the real endpoint would give). `w` is the body half-width as an +angle, in bins. Both correlations use the SAME realised per-bin hit rate +`P(hit | b_our = g)` (the j128 metric), and a second, bullet-conditioned target +is printed as a cross-check. + +Gate B: held-out per-candidate log-loss of the EXACT (bullet-line) label, +state-conditional vs state-free, the same measurement j130 ran for its proxy. + +Run: + python3 common_libs/tests/exact_geometry_gate.py \ + --corpus /tmp/tfil_ab2/out \ + --report common_libs/tests/fixtures/exact_geometry_gate_report.txt +""" +from __future__ import annotations + +import argparse +import os +import statistics +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) + +import outcome_label_gate as olg # validated extraction + metric +import analyze_drussgt_dodge_vs_power as adp + +NBINS = olg.NBINS + + +def p_hist(recs, b): + return sum(1 for r in recs if r["b_our"] == b) / len(recs) + + +def p_proxy(recs, b): + """j130 live label: P(hit and |b - b_our| <= w).""" + return statistics.fmean(1 if (r["hit"] >= 0.5 and abs(b - r["b_our"]) <= r["w"]) + else 0 for r in recs) + + +def p_exact(recs, b): + """Exact bullet-line label: P(|b - b_bullet| <= w).""" + return statistics.fmean(1 if abs(b - r["b_bullet"]) <= r["w"] else 0 + for r in recs) + + +def p_exact_hit(recs, b): + """Exact bullet-line AND hit: P(hit and |b - b_bullet| <= w).""" + return statistics.fmean(1 if (r["hit"] >= 0.5 and abs(b - r["b_bullet"]) <= r["w"]) + else 0 for r in recs) + + +def correlations(recs): + n = [0] * NBINS + h = [0] * NBINS + for r in recs: + n[r["b_our"]] += 1 + h[r["b_our"]] += r["hit"] + used = [b for b in range(NBINS) if n[b] > 0] + rate_our = [h[b] / n[b] for b in used] + + nb = [0] * NBINS + hb = [0] * NBINS + for r in recs: + nb[r["b_bullet"]] += 1 + hb[r["b_bullet"]] += r["hit"] + usedb = [b for b in range(NBINS) if nb[b] > 0] + rate_bullet = [hb[b] / nb[b] for b in usedb] + + maps = { + "histogram (j128): P(arrival = g)": [p_hist(recs, b) for b in used], + "outcome proxy (j130 live): P(hit & |g-b_our|<=w)": [p_proxy(recs, b) for b in used], + "EXACT bullet line: P(|g-b_bullet|<=w)": [p_exact(recs, b) for b in used], + "EXACT bullet line & hit: P(hit & |g-b_bullet|<=w)": [p_exact_hit(recs, b) for b in used], + } + out = {} + for name, d in maps.items(): + out[name] = (statistics.correlation(d, rate_our), + statistics.correlation([d[used.index(b)] if b in used else 0.0 + for b in usedb], rate_bullet)) + return out, used, rate_our, usedb, rate_bullet + + +def run_split_exact(recs, seed, decay=128, shift=1): + tr_b, te_b = olg.split_battles({r["battle"] for r in recs}, seed) + tr = [r for r in recs if r["battle"] in tr_b] + te = [r for r in recs if r["battle"] in te_b] + edges = dict(olg.CANON) + om = olg.OutcomeModel(decay, shift, state_free=False) + om0 = olg.OutcomeModel(decay, shift, state_free=True) + for r in tr: + st = olg.code_of(r, edges) + for g in range(NBINS): + lab = 1 if (r["hit"] >= 0.5 and abs(g - r["b_bullet"]) <= r["w"]) else 0 + om.learn(st, g, lab) + om0.learn(st, g, lab) + ll_s, ll_g, hit_out = [], [], [] + for r in te: + st = olg.code_of(r, edges) + go = min(range(NBINS), key=lambda g: om.predict_hit(st, g)) + real = lambda g: 1 if abs(g - r["b_bullet"]) <= r["w"] else 0 + hit_out.append(real(go)) + for g in range(NBINS): + y = 1 if (r["hit"] >= 0.5 and abs(g - r["b_bullet"]) <= r["w"]) else 0 + ll_s.append(-olg.log2(om.predict_hit(st, g)) if y + else -olg.log2(1.0 - om.predict_hit(st, g))) + ll_g.append(-olg.log2(om0.predict_hit(st, g)) if y + else -olg.log2(1.0 - om0.predict_hit(st, g))) + return dict(seed=seed, + ll_state=statistics.fmean(ll_s), + ll_statefree=statistics.fmean(ll_g), + delta=statistics.fmean([a - b for a, b in zip(ll_s, ll_g)]), + hit_out=statistics.fmean(hit_out)) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--corpus", default="/tmp/tfil_ab2/out") + ap.add_argument("--report", default=None) + ap.add_argument("--seeds", type=int, default=3) + args = ap.parse_args() + + runs = adp.discover_tfil(args.corpus) + recs = olg.extract(runs) + lines = [] + + def out(s=""): + print(s) + lines.append(s) + + out("# Exact-geometry Gate A/B — learned movement (job j131)") + out() + out(f"corpus : {args.corpus}") + out(f"battles : {len(runs)}") + out(f"shots : {len(recs)}") + out(f"base hit : {statistics.fmean(r['hit'] for r in recs)*100:.2f}%") + out("state : vlat, dist, room, turn (module's 4 fields, canonical edges)") + out() + + corr, used, rate_our, usedb, rate_bullet = correlations(recs) + out("## A. danger-map alignment (ONE consistent computation)") + out() + out("corr( danger(g) , P(hit | b_our = g) ) [the j128 metric, = -0.342 hist]") + out("corr( danger(g) , P(hit | b_bullet = g) ) [same danger, bullet-conditioned target]") + out() + out("| danger map | corr vs P(hit\\|b_our=g) | corr vs P(hit\\|b_bullet=g) |") + out("|---|---:|---:|") + for name, (c_our, c_bul) in corr.items(): + out(f"| {name} | {c_our:+.3f} | {c_bul:+.3f} |") + out() + out("Negative = minimising the danger steers INTO where the observed hits") + out("happen (the j128 defect). The exact bullet line is the physically") + out("correct 'would this wave hit me at g' map; if its correlation is still") + out("negative, exact geometry does NOT fix the inversion.") + out() + + out("| bin | P(hit\\|b_our) | P(hit\\|b_bullet) | hist danger | proxy danger | exact danger |") + out("|---:|---:|---:|---:|---:|---:|") + rb = {b: rate_bullet[usedb.index(b)] for b in usedb} + for b in used: + rb_str = f"{rb[b]*100:.1f}%" if b in rb else "—" + out(f"| {b} | {rate_our[used.index(b)]*100:.1f}% | " + f"{rb_str} | " + f"{p_hist(recs, b):.3f} | {p_proxy(recs, b):.3f} | {p_exact(recs, b):.3f} |") + out() + + per = [run_split_exact(recs, s) for s in range(args.seeds)] + ll_s = statistics.fmean(p["ll_state"] for p in per) + ll_g = statistics.fmean(p["ll_statefree"] for p in per) + out("## B. state-conditional information under the EXACT bullet-line label") + out() + out("held-out per-candidate log-loss (bits) of the exact label, " + "state-conditional vs state-free (same rows, same split):") + out() + out("| model | log-loss (bits) |") + out("|---|---:|") + out(f"| state-free P(label | g) | {ll_g:.4f} |") + out(f"| state-conditional P(label | state, g) | {ll_s:.4f} |") + out(f"| Δ (state − state-free) | {ll_s - ll_g:+.4f} |") + out() + neg = sum(1 for p in per if p["delta"] < 0) + out(f"state conditioning is better in {neg}/{len(per)} splits " + f"(negative Δ = better).") + out() + + out("## C. open-loop decision counterfactual (VETO ONLY)") + out() + out("argmin_g danger with the recorded bullet line as ground truth:") + out() + out(f"| exact-label argmin (j131) | " + f"{statistics.fmean(p['hit_out'] for p in per)*100:.2f}% |") + out() + + out("## MEASURED vs INFERRED") + out() + out("* MEASURED: every number above, on the recorded corpus.") + out("* INFERRED: that an offline alignment transfers live — it cannot, the") + out(" corpus is open loop (`docs/offline_harness_trust.md`).") + + if args.report: + os.makedirs(os.path.dirname(args.report), exist_ok=True) + with open(args.report, "w") as f: + f.write("\n".join(lines) + "\n") + + +if __name__ == "__main__": + main() diff --git a/docs/movement_campaign.md b/docs/movement_campaign.md index 46956c0..453c878 100644 --- a/docs/movement_campaign.md +++ b/docs/movement_campaign.md @@ -2243,3 +2243,134 @@ pre-registered rule records as WORSE. **Status: the default is UNCHANGED (`TR_MOVEMENT=strafe`); the outcome mode is default-off behind `TR_MOVEMENT=learned TR_LEARNED_LABEL=outcome`.** Revert = do not set the env vars. + +--- + +## Learned movement — real bullet endpoints (exact geometry) + +**Job j131. The owner's request:** *"use real bullets: bullets that really hit +me, bullets that hit the wall, both detectable. We ignore bullets that hit +other bots, this movement is only for 1v1."* The task's premise was that +`ModularBot.nim` already handles `onBulletHit`/`onBulletHitWall`, so the exact +bullet line was available live and j130's rejection of the exact label ("needs +bullet bodies the bot lacks") was wrong. + +### THE PREMISE IS HALF WRONG — VERIFIED (MEASURED, not inferred) + +The **fields** exist: `BulletState` has `x, y, direction, power, ownerId, +bulletId`, and `BulletHitWallEvent`/`HitByBulletEvent` both expose +`bullet: BulletState`. But **the events are not routed to the dodger**: + +* `BulletHitWallEvent` is delivered **only to the bullet's owner** + (`addPrivateBotEvent(outcome.bullet.botId, …)` — verified by decompiling the + running server jar `robocode-tankroyale-server-0.35.5-all.jar`, and identical + in the 1.1.0 source `CollisionDetector.applyBulletWallCollisions`). So an + **enemy** bullet hitting a wall is **not observable** by us. +* `TurnToTickEventForBotMapper` builds `bulletStates = turn.bullets.filter + { it.botId == bot.id }`, so `getBulletStates()` returns **only our own** + bullets too. +* The events the dodger **does** receive with a real enemy-bullet endpoint are: + `onHitByBullet` (the bullet hit US — endpoint = our impact point) and a + bullet-vs-bullet event where **our** bullet intercepted an enemy bullet + (`e.hitBullet` is the enemy bullet, with its endpoint + heading). + +**So the "exact straight line from a wall hit" cannot be built live.** In 1v1 a +missed bullet does end on a wall, but the server keeps that observation private +to the shooter. This is the second time the availability premise is the binding +constraint, now for the exact label rather than the proxy. + +### WHAT CHANGED (code) + +* `common_libs/movements/learned_surfer.nim` — **default-off** + `TR_LEARNED_REAL_EVENTS=1` (registered in `env_report.knownEnvNames()`). When + on, a wave is resolved by the REAL event instead of the arrival deadline: + the exact `origin → endpoint` straight line sets the label's GF bin, the real + flight time `currentTick − fireTick` is recorded (`resolvedReal`, `lastFlightErr` + — a cross-check on the energy-drop speed inference), and the wave is **dropped + at once** (`resolveEnemyBullet`), so no ghost accumulates. A wave no event + claims resolves `RealEventsGrace` ticks past nominal as a **wall MISS**. With + the knob off the byte-for-byte j130 behaviour is preserved (tests pin it). +* `ModularBot_garage/src/ModularBot.nim` — forwards `onHitByBullet` (hit on us), + a bullet-vs-bullet intercept of an enemy bullet (`e.hitBullet`), and (guarded, + dead on 0.35.5) an enemy `onBulletHitWall` to `learnedMover.resolveEnemyBullet`. +* `ModularBot_garage/tests/test_learned_surfer.nim` — real-event unit checks + (default-off parity, exact centre-bin resolution, ghost drop, wall-miss + deadline). `common_libs/tests/exact_geometry_gate.py` — Gate A/B below. + +### GATE A — danger-map alignment, ONE consistent computation (MEASURED) + +`python3 common_libs/tests/exact_geometry_gate.py --corpus /tmp/tfil_ab2/out` +(70 battles, 54 923 shots, the same extraction and the same +`corr(danger(g), P(hit | b_our=g))` metric j128/j130 used): + +| danger map | corr vs `P(hit\|b_our=g)` | corr vs `P(hit\|b_bullet=g)` | +|---|---:|---:| +| histogram P(arrival = g) (j128) | **−0.341** | −0.206 | +| outcome proxy `P(hit & \|g−b_our\|≤w)` (j130 live) | **+0.566** | +0.604 | +| **EXACT bullet line `P(\|g−b_bullet\|≤w)`** | **−0.230** | **+0.120** | +| exact bullet line & hit | +0.465 | +0.684 | + +**The exact-geometry label does NOT fix the inversion on the j128 metric** — +−0.230 is still negative (minimising it still steers into where the observed +hits happen). It is *less* negative than the histogram (−0.341) and turns +weakly positive (+0.120) only when the target is conditioned on the bullet's +own line `b_bullet`, while the +0.566 proxy is inflated by being conditioned on +`b_our` (the realised arrival, i.e. where the recorded wave already was). Under +the task's own gate, **the veto fires and the live batch is not run.** + +### GATE B — state information under the EXACT label (MEASURED) + +Held-out per-candidate log-loss of the exact label, split BY BATTLE, 3 seeds: + +| model | log-loss (bits) | +|---|---:| +| state-free `P(label \| g)` | **0.1879** | +| state-conditional `P(label \| state, g)` | **0.3747** | +| Δ (state − state-free) | **+0.1868** | + +state conditioning is better in **0/3** splits. This **replicates j130 almost +exactly** (proxy: 0.3906 vs 0.1873, Δ +0.203, 0/3). Under the exact label the +coarse four-field state is still *worse* than the state-free model: the state +buys no held-out information, so it cannot be the thing the learned mover is +missing — **the observable state is still the binding constraint.** + +### GATE C — live panel (NOT RUN, by the pre-registered rule) + +Gate A's veto fired (exact correlation negative), so no live battles were +fought. Independently, the live batch would have been testing a label the module +**cannot construct** in the miss case (enemy wall endpoints are owner-private), +so a live "exact" arm would in practice be j130's proxy for ~90% of waves. + +### Direct answer + +**Does exact bullet geometry fix the label? NO — not on the measured metric and +not live.** The physically-exact map reads −0.230 against the j128 target +(still inverted; the proxy's +0.566 is the one that is inflated). And the +geometric endpoint **is not observable** by the dodger on this server for the +miss case: `BulletHitWallEvent` and `bulletStates` are owner-private, so the +only real enemy-bullet endpoints we get are the ~13% that hit us (and the rare +intercepts). The exact line therefore cannot be built live for the waves that +matter. + +**Is the binding constraint the STATE rather than the label or the learner? +YES — the same answer as j130, now measured for the third label.** Under the +exact label the state still loses to state-free on held-out log-loss (0.3747 vs +0.1879, 0/3 splits). j128 (histogram), j130 (outcome proxy) and j131 (exact +line) each change the label; none moves the live result and none makes the +state informative. The wave-crossing signal a 1v1 dodger needs is simply not in +the four-field observable state, and hand-tuned `strafe` remains hard to beat. + +### MEASURED vs INFERRED + +**MEASURED:** the event routing (decompiled the running 0.35.5 jar + +`TurnToTickEventForBotMapper`); the three-way Gate A correlation and the +exact-label Gate B log-loss on the recorded corpus; the module unit tests +(24/24, including the real-event and default-off parity checks); the env-report +guard (25/25); the clean-archive compile. **INFERRED:** that the offline +alignment transfers live — it cannot (open-loop corpus, see +`docs/offline_harness_trust.md`). + +**Status: the default is UNCHANGED (`TR_MOVEMENT=strafe`).** The real-event +resolution is default-off behind `TR_MOVEMENT=learned TR_LEARNED_REAL_EVENTS=1` +(combined with `TR_LEARNED_LABEL=outcome` for the dense readout). Revert = do +not set the env vars.