j131 learned movement: real bullet-endpoint resolution (TR_LEARNED_REAL_EVENTS, default off) + exact-geometry Gate A/B (inversion NOT fixed; state still the constraint)
This commit is contained in:
@@ -620,6 +620,15 @@ method onBulletHitWall*(bot: ModularBot, e: BulletHitWallEvent) =
|
||||
bot.strafeMover.removeBulletNear(e.bullet.x, e.bullet.y)
|
||||
bot.surfMover.removeBulletNear(e.bullet.x, e.bullet.y)
|
||||
bot.learnedMover.removeBulletNear(e.bullet.x, e.bullet.y)
|
||||
# Real wall endpoint of an ENEMY bullet. On the running server (0.35.5) a
|
||||
# BulletHitWallEvent is delivered only to the bullet's OWNER
|
||||
# (`addPrivateBotEvent(bullet.botId, ...)`), so this branch is a no-op live;
|
||||
# it is wired so a server that exposes enemy wall hit-points feeds the
|
||||
# exact-geometry wave resolution instead of the arrival-deadline proxy.
|
||||
if e.bullet.ownerId != getMyId():
|
||||
discard bot.learnedMover.resolveEnemyBullet(
|
||||
e.bullet.x, e.bullet.y, degToRad(e.bullet.direction),
|
||||
e.bullet.ownerId, bot.tick, false)
|
||||
|
||||
method onBulletHitBullet*(bot: ModularBot, e: BulletHitBulletEvent) =
|
||||
discard bot.resolveOwnBullet(e.bullet.bulletId) # bullet-vs-bullet: free the slot
|
||||
@@ -630,6 +639,13 @@ method onBulletHitBullet*(bot: ModularBot, e: BulletHitBulletEvent) =
|
||||
bot.strafeMover.removeBulletNear(e.bullet.x, e.bullet.y)
|
||||
bot.surfMover.removeBulletNear(e.bullet.x, e.bullet.y)
|
||||
bot.learnedMover.removeBulletNear(e.bullet.x, e.bullet.y)
|
||||
# `e.bullet` is OUR bullet; `e.hitBullet` is the bullet it struck. When that
|
||||
# is an ENEMY bullet, the event exposes its real endpoint + heading - a real
|
||||
# resolution for a wave that would otherwise wait for the deadline.
|
||||
if e.hitBullet.ownerId != getMyId() and e.hitBullet.bulletId != 0:
|
||||
discard bot.learnedMover.resolveEnemyBullet(
|
||||
e.hitBullet.x, e.hitBullet.y, degToRad(e.hitBullet.direction),
|
||||
e.hitBullet.ownerId, bot.tick, false)
|
||||
|
||||
method onHitByBullet*(bot: ModularBot, e: HitByBulletEvent) =
|
||||
# Feeds the ram bullet-rain abort window. Accumulate REAL ENERGY (the server's
|
||||
@@ -637,6 +653,11 @@ method onHitByBullet*(bot: ModularBot, e: HitByBulletEvent) =
|
||||
# energy/turn rate (see movements/ram_decision.bulletDamage).
|
||||
bot.ramDmgAccum += bulletDamage(e.bullet.power)
|
||||
bot.moveTracker.registerHit(e.bullet.power, e.bullet.direction, getX(), getY())
|
||||
# A real ENEMY bullet hit us: its endpoint (our impact point) + heading give
|
||||
# the wave's EXACT straight line, so resolve and drop that live wave now.
|
||||
discard bot.learnedMover.resolveEnemyBullet(
|
||||
e.bullet.x, e.bullet.y, degToRad(e.bullet.direction),
|
||||
e.bullet.ownerId, bot.tick, true)
|
||||
|
||||
proc enemyEnergyForReport(bot: ModularBot): float =
|
||||
## Last-scanned energy of the enemy the report is about (current target,
|
||||
|
||||
@@ -511,7 +511,7 @@ proc knownEnvNames*(): seq[string] =
|
||||
LearnedDecayEveryEnv, LearnedDecayShiftEnv, LearnedAlphaEnv,
|
||||
LearnedTravelEnv, LearnedReversalEnv, LearnedPrefDistEnv,
|
||||
LearnedDistBandEnv, LearnedRadialFracEnv, LearnedWallMarginEnv,
|
||||
LearnedGlobalEnv, LearnedLabelEnv, LearnedLogEnv,
|
||||
LearnedGlobalEnv, LearnedLabelEnv, LearnedRealEventsEnv, LearnedLogEnv,
|
||||
# harness vars (read by the test framework, inherited by the bot, so they
|
||||
# must NOT be reported as typos)
|
||||
"TR_SERVER_JAR", "TR_BATTLE_RUNNER", "TR_BATTLE_RUNNER_DIR",
|
||||
|
||||
@@ -88,6 +88,66 @@ proc counters(m: LearnedSurferModule): int =
|
||||
if c != 0'u8: inc n
|
||||
n
|
||||
|
||||
proc testRealEvents() =
|
||||
## `TR_LEARNED_REAL_EVENTS`: a wave resolves on the REAL bullet event (exact
|
||||
## origin->endpoint line) instead of the arrival deadline, and is dropped the
|
||||
## moment it resolves (ghost cleanup). Default-off parity is also pinned.
|
||||
const Ex = 400.0
|
||||
const Ey = 100.0
|
||||
const Ux = 400.0
|
||||
const Uy = 300.0
|
||||
|
||||
proc wsAt(tick: int; eEnergy: float): WorldState =
|
||||
WorldState(
|
||||
enemyX: Ex, enemyY: Ey, enemyEnergy: eEnergy,
|
||||
selfX: Ux, selfY: Uy, selfSpeed: 8.0, selfHeading: 0.0,
|
||||
arenaWidth: 800.0, arenaHeight: 600.0, tick: tick,
|
||||
enemies: @[EnemyInfo(id: 1, x: Ex, y: Ey,
|
||||
heading: 180.0, speed: 0.0, energy: eEnergy)])
|
||||
|
||||
# ── parity: with the knob off a real event changes nothing ──────────────
|
||||
putEnv(LearnedLabelEnv, "histogram")
|
||||
putEnv(LearnedRealEventsEnv, "")
|
||||
loadLearnedEnv()
|
||||
var mOff = initLearnedSurfer()
|
||||
discard mOff.computeMove(wsAt(0, 100.0)) # tick 0: baseline energy sample
|
||||
discard mOff.computeMove(wsAt(1, 99.0)) # tick 1: a 1.0 firepower drop
|
||||
check "RE default-off: the fire is detected as a live wave",
|
||||
mOff.liveWaves == 1
|
||||
check "RE default-off: resolveEnemyBullet is a no-op",
|
||||
(not mOff.resolveEnemyBullet(Ux, Uy, degToRad(90.0), 1, 13, true)) and
|
||||
mOff.liveWaves == 1
|
||||
|
||||
# ── ON: the exact origin->endpoint line resolves the wave immediately ────
|
||||
putEnv(LearnedRealEventsEnv, "1")
|
||||
loadLearnedEnv()
|
||||
var mOn = initLearnedSurfer()
|
||||
discard mOn.computeMove(wsAt(0, 100.0))
|
||||
discard mOn.computeMove(wsAt(1, 99.0))
|
||||
check "RE on: the fire is detected as a live wave", mOn.liveWaves == 1
|
||||
# endpoint = our position on the centre line (GF 0 -> bin 15) at ~nominal
|
||||
# (200 px at speed 17 -> ~12 ticks after the fire at tick 1)
|
||||
check "RE on: the real hit endpoint resolves the wave",
|
||||
mOn.resolveEnemyBullet(Ux, Uy, degToRad(90.0), 1, 13, true)
|
||||
check "RE on: the wave is dropped at once (ghost cleanup)", mOn.liveWaves == 0
|
||||
check "RE on: a real resolution is counted", mOn.resolvedReal == 1
|
||||
check "RE on: the exact straight line lands in the centre bin",
|
||||
mOn.glob[15] == 1
|
||||
check "RE on: the real flight time is recorded", mOn.lastFlightErr > -20.0
|
||||
|
||||
# ── ON: an unmatched wave still resolves (as a WALL MISS) at the deadline ─
|
||||
var mMiss = initLearnedSurfer()
|
||||
discard mMiss.computeMove(wsAt(0, 100.0))
|
||||
discard mMiss.computeMove(wsAt(1, 99.0))
|
||||
check "RE on: a wave is pending before any event", mMiss.liveWaves == 1
|
||||
for t in 2..<40: discard mMiss.computeMove(wsAt(t, 99.0))
|
||||
check "RE on: the unmatched wall wave eventually resolves",
|
||||
mMiss.liveWaves == 0 and mMiss.resolvedDead >= 1
|
||||
|
||||
putEnv(LearnedRealEventsEnv, "")
|
||||
putEnv(LearnedLabelEnv, "")
|
||||
loadLearnedEnv()
|
||||
|
||||
proc main() =
|
||||
# 1. concept
|
||||
check "MovementModule concept", isMovementModule(LearnedSurferModule)
|
||||
@@ -170,6 +230,8 @@ proc main() =
|
||||
putEnv(LearnedDecayEveryEnv, "")
|
||||
loadLearnedEnv()
|
||||
|
||||
echo ""
|
||||
testRealEvents()
|
||||
echo ""
|
||||
echo "checks=", checks, " failures=", failures
|
||||
if failures > 0: quit(1)
|
||||
|
||||
@@ -71,6 +71,13 @@
|
||||
## = the bot radius as an angle at the wave's
|
||||
## distance). See docs/movement_campaign.md,
|
||||
## "outcome label".
|
||||
## TR_LEARNED_REAL_EVENTS =1: resolve a wave on the REAL bullet event
|
||||
## (onHitByBullet / an enemy bullet intercepted by
|
||||
## ours) using the exact origin->endpoint line and
|
||||
## the real flight time, dropping the wave at once.
|
||||
## Default off (arrival-deadline proxy). Enemy WALL
|
||||
## hits are owner-private on server 0.35.5 and stay
|
||||
## deadline misses - see the const-block note.
|
||||
## TR_LEARNED_LOG per-decision log line
|
||||
|
||||
import std/[math, os]
|
||||
@@ -108,8 +115,30 @@ const
|
||||
LearnedWallMarginEnv* = "TR_LEARNED_WALL_MARGIN"
|
||||
LearnedGlobalEnv* = "TR_LEARNED_GLOBAL"
|
||||
LearnedLabelEnv* = "TR_LEARNED_LABEL"
|
||||
LearnedRealEventsEnv* = "TR_LEARNED_REAL_EVENTS"
|
||||
LearnedLogEnv* = "TR_LEARNED_LOG"
|
||||
|
||||
## Real-event matching (job j131).
|
||||
## A wave is normally resolved on the nominal arrival tick
|
||||
## `ceil(startDist/speed)`. With `TR_LEARNED_REAL_EVENTS=1` a wave is instead
|
||||
## resolved by the REAL server event that carries the bullet's endpoint:
|
||||
## * `onHitByBullet` -> the bullet HIT us; endpoint = our impact point;
|
||||
## * a bullet-vs-bullet intercept of an ENEMY bullet (our bullet hit theirs)
|
||||
## -> the enemy bullet's endpoint/heading are in the event.
|
||||
## The bullet's raw straight line (origin at fire -> endpoint) then gives the
|
||||
## EXACT GF bin and the real flight time (a cross-check on the energy-drop
|
||||
## speed inference), and the wave is dropped immediately (no ghost build-up).
|
||||
##
|
||||
## ON THE RUNNING SERVER (0.35.5, verified from the server bytecode +
|
||||
## `TurnToTickEventForBotMapper`) an ENEMY bullet that hits a WALL produces a
|
||||
## `BulletHitWallEvent` only for the bullet's OWNER (`addPrivateBotEvent(
|
||||
## bullet.botId, ...)`), and `bulletStates` is filtered to the bot's own
|
||||
## bullets. So a wall HIT is NOT observable by the dodger; those waves fall
|
||||
## back to the arrival deadline and are labelled a MISS. `resolveEnemyBullet`
|
||||
## accepts a wall endpoint anyway so a future/other server can feed it.
|
||||
RealEventsMatchTol = 12.0 ## max |real flight - nominal| to accept a match, ticks
|
||||
RealEventsGrace = 8 ## ticks past nominal before an unmatched wave resolves
|
||||
|
||||
## Number of joint (vlat,dist,room,turn) state codes = 4^4.
|
||||
LS_STATES = LS_Q * LS_Q * LS_Q * LS_Q
|
||||
## Prior-mix weight for the 2-class outcome readout.
|
||||
@@ -132,6 +161,7 @@ var
|
||||
LearnedWallMargin* = 48.0
|
||||
LearnedGlobal* = false
|
||||
LearnedLabel* = llHistogram
|
||||
LearnedRealEvents* = false
|
||||
LearnedLog* = false
|
||||
|
||||
proc getEnvFloat(name: string, default: float): float =
|
||||
@@ -164,6 +194,7 @@ proc loadLearnedEnv*() =
|
||||
LearnedRadialFrac = clamp(getEnvFloat(LearnedRadialFracEnv, 0.35), 0.0, 1.0)
|
||||
LearnedWallMargin = max(0.0, getEnvFloat(LearnedWallMarginEnv, 48.0))
|
||||
LearnedGlobal = envOn(LearnedGlobalEnv)
|
||||
LearnedRealEvents = envOn(LearnedRealEventsEnv)
|
||||
LearnedLog = envOn(LearnedLogEnv)
|
||||
LearnedLabel =
|
||||
case getEnv(LearnedLabelEnv, "").strip().toLowerAscii()
|
||||
@@ -219,6 +250,8 @@ proc roomToWall(px, py, dx, dy, arenaW, arenaH: float64): float64 =
|
||||
|
||||
type
|
||||
LSWave = object
|
||||
ownerId: int ## enemy that fired (energies are per-enemy)
|
||||
fireTick: int ## `ws.tick` at the fire tick (real flight time)
|
||||
originX, originY: float64
|
||||
bearing: float64 ## enemy -> us at the fire tick (centre line)
|
||||
speed: float64
|
||||
@@ -246,6 +279,9 @@ type
|
||||
prevHeading: float64
|
||||
debugGraphics*: bool
|
||||
decisions*: int ## decisions taken (diagnostic)
|
||||
resolvedReal*: int ## waves resolved by a real bullet event
|
||||
resolvedDead*: int ## waves resolved on the arrival deadline
|
||||
lastFlightErr*: float ## real flight - nominal flight, last resolution
|
||||
|
||||
proc resetRound*(m: var LearnedSurferModule) =
|
||||
## Per-ROUND reset: the waves and the smoothed global prior are per round, but
|
||||
@@ -261,6 +297,9 @@ proc resetRound*(m: var LearnedSurferModule) =
|
||||
m.prevY = 0.0
|
||||
m.prevHeading = 0.0
|
||||
m.decisions = 0
|
||||
m.resolvedReal = 0
|
||||
m.resolvedDead = 0
|
||||
m.lastFlightErr = 0.0
|
||||
for i in 0..<LS_BINS: m.glob[i] = 0
|
||||
m.glc = 0
|
||||
m.hitGlobal = 0
|
||||
@@ -292,6 +331,7 @@ proc resetBattle*(m: var LearnedSurferModule) =
|
||||
|
||||
proc clearGraphics*(m: var LearnedSurferModule) {.inline.} = discard
|
||||
proc removeBulletNear*(m: var LearnedSurferModule, x, y: float) {.inline.} = discard
|
||||
proc liveWaves*(m: LearnedSurferModule): int {.inline.} = m.waves.len
|
||||
|
||||
# ── the learner ─────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -339,6 +379,79 @@ proc learnWave(m: var LearnedSurferModule, w: LSWave, bin: int, hit: bool) =
|
||||
m.missGlobal = m.missGlobal - (m.missGlobal shr LearnedDecayShift)
|
||||
m.glc = 0
|
||||
|
||||
proc resolveWaveIdx(m: var LearnedSurferModule, idx, bin: int, hit: bool,
|
||||
currentTick: int) =
|
||||
## One live wave -> one training sample + removal. Shared by the arrival
|
||||
## deadline (unobserved wall misses) and the real-event path, so a wave is
|
||||
## ALWAYS trained and dropped exactly once - no ghost accumulation.
|
||||
let w = m.waves[idx]
|
||||
let nom = w.startDist / max(w.speed, 1e-9)
|
||||
let realFlight = float(currentTick - w.fireTick)
|
||||
m.lastFlightErr = realFlight - nom
|
||||
if LearnedLog:
|
||||
echo "[learned] resolve bin=", bin, " state=", w.stateRow, "/",
|
||||
w.stateCol, " d=", w.startDist.int, " e=", w.originX.int, ",",
|
||||
w.originY.int, " flight=", realFlight.int, " nominal=", nom.int,
|
||||
" hit=", hit
|
||||
m.learnWave(w, bin, hit)
|
||||
m.waves.del(idx)
|
||||
|
||||
proc missileLineBin(w: LSWave, x, y, headingRad: float): int =
|
||||
## GF bin of the bullet's real straight line through `(x,y)` (the endpoint),
|
||||
## falling back to the real heading when the endpoint is degenerate. This is
|
||||
## the EXACT geometry: origin at the fire tick + real endpoint, no timing
|
||||
## guess.
|
||||
let maxA = mea(w.speed)
|
||||
if maxA < 1e-9: return gfToBin(0.0)
|
||||
let ex = x - w.originX
|
||||
let ey = y - w.originY
|
||||
let lineDir =
|
||||
if hypot(ex, ey) > 1.0: arctan2(ey, ex)
|
||||
else: headingRad
|
||||
gfToBin(clamp(wrapPi(lineDir - w.bearing) / maxA, -1.0, 1.0))
|
||||
|
||||
proc resolveEnemyBullet*(m: var LearnedSurferModule, x, y, headingRad: float,
|
||||
ownerId, currentTick: int, hit: bool): bool =
|
||||
## Resolve (and DROP) the live wave matching a REAL enemy-bullet event.
|
||||
##
|
||||
## `x,y` the bullet's real endpoint (our impact point for a HIT, the
|
||||
## wall point for a wall hit, the intercept point for a
|
||||
## bullet-vs-bullet hit),
|
||||
## `headingRad` the bullet's real heading (fallback when the endpoint is
|
||||
## degenerate),
|
||||
## `hit` true only for a HIT on us.
|
||||
## The exact straight line origin->endpoint sets the label's GF bin and the
|
||||
## real flight time `currentTick - fireTick` is recorded, which cross-checks
|
||||
## the energy-drop speed inference. No-op unless `TR_LEARNED_REAL_EVENTS=1`.
|
||||
result = false
|
||||
if not LearnedRealEvents or m.waves.len == 0: return
|
||||
# The wave whose nominal arrival is closest to now is the one this bullet
|
||||
# belongs to; ownerId disambiguates when several enemies are firing.
|
||||
var best = -1
|
||||
var bestKey = Inf
|
||||
for i in 0..<m.waves.len:
|
||||
let w = m.waves[i]
|
||||
if ownerId >= 0 and w.ownerId != ownerId: continue
|
||||
let key = abs(float(currentTick - w.fireTick) -
|
||||
w.startDist / max(w.speed, 1e-9))
|
||||
if key < bestKey:
|
||||
bestKey = key
|
||||
best = i
|
||||
if best < 0: # no wave from that enemy: fall back to time-only matching
|
||||
for i in 0..<m.waves.len:
|
||||
let w = m.waves[i]
|
||||
let key = abs(float(currentTick - w.fireTick) -
|
||||
w.startDist / max(w.speed, 1e-9))
|
||||
if key < bestKey:
|
||||
bestKey = key
|
||||
best = i
|
||||
if best < 0 or bestKey > RealEventsMatchTol: return
|
||||
let w = m.waves[best]
|
||||
let bin = missileLineBin(w, x, y, headingRad)
|
||||
m.resolveWaveIdx(best, bin, hit, currentTick)
|
||||
inc m.resolvedReal
|
||||
result = true
|
||||
|
||||
proc predictHit*(m: LearnedSurferModule, row, col, g: int): float =
|
||||
## P(hit | state, candidate bin g) — the `outcome` danger (lower = safer),
|
||||
## from the 2-class counted SBC read out with the per-cell posterior and
|
||||
@@ -416,6 +529,7 @@ proc detectFire(m: var LearnedSurferModule, id: int, ex, ey, eenergy: float,
|
||||
let col = code(room, RoomEdges) * LS_Q + code(turn, TurnEdges)
|
||||
|
||||
m.waves.add LSWave(
|
||||
ownerId: id, fireTick: ws.tick,
|
||||
originX: ex, originY: ey, bearing: bearing, speed: bspeed,
|
||||
startDist: d, power: drop,
|
||||
ticksLeft: max(1, int(ceil(d / max(bspeed, 1e-9)))),
|
||||
@@ -445,24 +559,27 @@ proc computeMove*(m: var LearnedSurferModule, ws: WorldState): MoveCommand =
|
||||
m.waves[i].fresh = false # created this tick: not one tick old yet
|
||||
else:
|
||||
dec m.waves[i].ticksLeft
|
||||
if m.waves[i].ticksLeft <= 0:
|
||||
# With real events on, wait `RealEventsGrace` ticks past the nominal arrival
|
||||
# so a late HitByBullet can still claim the wave; a wave no event claims is
|
||||
# a WALL MISS (the wall event is owner-private - see the const block).
|
||||
let deadline = if LearnedRealEvents: -RealEventsGrace else: 0
|
||||
if m.waves[i].ticksLeft <= deadline:
|
||||
let w = m.waves[i]
|
||||
let maxA = mea(w.speed)
|
||||
if maxA >= 1e-9:
|
||||
let off = wrapPi(arctan2(botY - w.originY, botX - w.originX) - w.bearing)
|
||||
let bin = gfToBin(clamp(off / maxA, -1.0, 1.0))
|
||||
# llOutcome label: did THIS wave hit us? Our own energy dropped since
|
||||
# the fire tick. (One wave is live at a time in 1v1; ramming also drops
|
||||
# energy, so this is a proxy, not an oracle.)
|
||||
let hit = ws.selfEnergy < w.selfEnergyAtFire - 0.01
|
||||
m.learnWave(w, bin, hit)
|
||||
if LearnedLog:
|
||||
echo "[learned] resolve bin=", bin, " state=", w.stateRow, "/",
|
||||
w.stateCol, " d=", w.startDist.int, " e=", w.originX.int, ",",
|
||||
w.originY.int
|
||||
m.waves.del(i)
|
||||
else:
|
||||
inc i
|
||||
# llOutcome label: did THIS wave hit us? The energy drop is the proxy;
|
||||
# with real events on the HIT is taken from `onHitByBullet` instead, so
|
||||
# an unmatched wave is a wall MISS.
|
||||
let hit = (not LearnedRealEvents) and
|
||||
ws.selfEnergy < w.selfEnergyAtFire - 0.01
|
||||
m.resolveWaveIdx(i, bin, hit, ws.tick)
|
||||
inc m.resolvedDead
|
||||
else:
|
||||
m.waves.del(i)
|
||||
continue
|
||||
inc i
|
||||
|
||||
# ── 3. danger of every candidate bin, summed over every live wave ────────
|
||||
# llHistogram: precompute the predicted arrival-bin distribution per wave.
|
||||
|
||||
@@ -0,0 +1,224 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Exact-geometry Gate A/B for the learned movement (job j131).
|
||||
|
||||
Question: does labelling/resolving a wave by the REAL bullet endpoint (the
|
||||
exact origin->endpoint straight line, available live from `onHitByBullet` and
|
||||
from a bullet-vs-bullet intercept) fix the danger-map inversion that j128
|
||||
measured with the histogram label (`corr = -0.342`) and that j130 replaced with
|
||||
a suspicious live-computable proxy (`corr = +0.566`)?
|
||||
|
||||
This is the SAME corpus, SAME per-shot extraction and SAME metric as
|
||||
`outcome_label_gate.py` (which job j130 used), so the three danger maps are
|
||||
computed under ONE consistent computation and are directly comparable:
|
||||
|
||||
(a) histogram label danger(g) = P(arrival bin = g) (j128)
|
||||
(b) outcome proxy label danger(g) = P(hit and |g - b_our| <= w) (j130 live)
|
||||
(c) EXACT bullet line danger(g) = P(|g - b_bullet| <= w) (this job)
|
||||
|
||||
`b_our` is the GF of OUR position at the nominal arrival tick; `b_bullet` is
|
||||
the GF of the bullet's own straight line (from the recorded fire direction -
|
||||
exactly the line the real endpoint would give). `w` is the body half-width as an
|
||||
angle, in bins. Both correlations use the SAME realised per-bin hit rate
|
||||
`P(hit | b_our = g)` (the j128 metric), and a second, bullet-conditioned target
|
||||
is printed as a cross-check.
|
||||
|
||||
Gate B: held-out per-candidate log-loss of the EXACT (bullet-line) label,
|
||||
state-conditional vs state-free, the same measurement j130 ran for its proxy.
|
||||
|
||||
Run:
|
||||
python3 common_libs/tests/exact_geometry_gate.py \
|
||||
--corpus /tmp/tfil_ab2/out \
|
||||
--report common_libs/tests/fixtures/exact_geometry_gate_report.txt
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import statistics
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
import outcome_label_gate as olg # validated extraction + metric
|
||||
import analyze_drussgt_dodge_vs_power as adp
|
||||
|
||||
NBINS = olg.NBINS
|
||||
|
||||
|
||||
def p_hist(recs, b):
|
||||
return sum(1 for r in recs if r["b_our"] == b) / len(recs)
|
||||
|
||||
|
||||
def p_proxy(recs, b):
|
||||
"""j130 live label: P(hit and |b - b_our| <= w)."""
|
||||
return statistics.fmean(1 if (r["hit"] >= 0.5 and abs(b - r["b_our"]) <= r["w"])
|
||||
else 0 for r in recs)
|
||||
|
||||
|
||||
def p_exact(recs, b):
|
||||
"""Exact bullet-line label: P(|b - b_bullet| <= w)."""
|
||||
return statistics.fmean(1 if abs(b - r["b_bullet"]) <= r["w"] else 0
|
||||
for r in recs)
|
||||
|
||||
|
||||
def p_exact_hit(recs, b):
|
||||
"""Exact bullet-line AND hit: P(hit and |b - b_bullet| <= w)."""
|
||||
return statistics.fmean(1 if (r["hit"] >= 0.5 and abs(b - r["b_bullet"]) <= r["w"])
|
||||
else 0 for r in recs)
|
||||
|
||||
|
||||
def correlations(recs):
|
||||
n = [0] * NBINS
|
||||
h = [0] * NBINS
|
||||
for r in recs:
|
||||
n[r["b_our"]] += 1
|
||||
h[r["b_our"]] += r["hit"]
|
||||
used = [b for b in range(NBINS) if n[b] > 0]
|
||||
rate_our = [h[b] / n[b] for b in used]
|
||||
|
||||
nb = [0] * NBINS
|
||||
hb = [0] * NBINS
|
||||
for r in recs:
|
||||
nb[r["b_bullet"]] += 1
|
||||
hb[r["b_bullet"]] += r["hit"]
|
||||
usedb = [b for b in range(NBINS) if nb[b] > 0]
|
||||
rate_bullet = [hb[b] / nb[b] for b in usedb]
|
||||
|
||||
maps = {
|
||||
"histogram (j128): P(arrival = g)": [p_hist(recs, b) for b in used],
|
||||
"outcome proxy (j130 live): P(hit & |g-b_our|<=w)": [p_proxy(recs, b) for b in used],
|
||||
"EXACT bullet line: P(|g-b_bullet|<=w)": [p_exact(recs, b) for b in used],
|
||||
"EXACT bullet line & hit: P(hit & |g-b_bullet|<=w)": [p_exact_hit(recs, b) for b in used],
|
||||
}
|
||||
out = {}
|
||||
for name, d in maps.items():
|
||||
out[name] = (statistics.correlation(d, rate_our),
|
||||
statistics.correlation([d[used.index(b)] if b in used else 0.0
|
||||
for b in usedb], rate_bullet))
|
||||
return out, used, rate_our, usedb, rate_bullet
|
||||
|
||||
|
||||
def run_split_exact(recs, seed, decay=128, shift=1):
|
||||
tr_b, te_b = olg.split_battles({r["battle"] for r in recs}, seed)
|
||||
tr = [r for r in recs if r["battle"] in tr_b]
|
||||
te = [r for r in recs if r["battle"] in te_b]
|
||||
edges = dict(olg.CANON)
|
||||
om = olg.OutcomeModel(decay, shift, state_free=False)
|
||||
om0 = olg.OutcomeModel(decay, shift, state_free=True)
|
||||
for r in tr:
|
||||
st = olg.code_of(r, edges)
|
||||
for g in range(NBINS):
|
||||
lab = 1 if (r["hit"] >= 0.5 and abs(g - r["b_bullet"]) <= r["w"]) else 0
|
||||
om.learn(st, g, lab)
|
||||
om0.learn(st, g, lab)
|
||||
ll_s, ll_g, hit_out = [], [], []
|
||||
for r in te:
|
||||
st = olg.code_of(r, edges)
|
||||
go = min(range(NBINS), key=lambda g: om.predict_hit(st, g))
|
||||
real = lambda g: 1 if abs(g - r["b_bullet"]) <= r["w"] else 0
|
||||
hit_out.append(real(go))
|
||||
for g in range(NBINS):
|
||||
y = 1 if (r["hit"] >= 0.5 and abs(g - r["b_bullet"]) <= r["w"]) else 0
|
||||
ll_s.append(-olg.log2(om.predict_hit(st, g)) if y
|
||||
else -olg.log2(1.0 - om.predict_hit(st, g)))
|
||||
ll_g.append(-olg.log2(om0.predict_hit(st, g)) if y
|
||||
else -olg.log2(1.0 - om0.predict_hit(st, g)))
|
||||
return dict(seed=seed,
|
||||
ll_state=statistics.fmean(ll_s),
|
||||
ll_statefree=statistics.fmean(ll_g),
|
||||
delta=statistics.fmean([a - b for a, b in zip(ll_s, ll_g)]),
|
||||
hit_out=statistics.fmean(hit_out))
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--corpus", default="/tmp/tfil_ab2/out")
|
||||
ap.add_argument("--report", default=None)
|
||||
ap.add_argument("--seeds", type=int, default=3)
|
||||
args = ap.parse_args()
|
||||
|
||||
runs = adp.discover_tfil(args.corpus)
|
||||
recs = olg.extract(runs)
|
||||
lines = []
|
||||
|
||||
def out(s=""):
|
||||
print(s)
|
||||
lines.append(s)
|
||||
|
||||
out("# Exact-geometry Gate A/B — learned movement (job j131)")
|
||||
out()
|
||||
out(f"corpus : {args.corpus}")
|
||||
out(f"battles : {len(runs)}")
|
||||
out(f"shots : {len(recs)}")
|
||||
out(f"base hit : {statistics.fmean(r['hit'] for r in recs)*100:.2f}%")
|
||||
out("state : vlat, dist, room, turn (module's 4 fields, canonical edges)")
|
||||
out()
|
||||
|
||||
corr, used, rate_our, usedb, rate_bullet = correlations(recs)
|
||||
out("## A. danger-map alignment (ONE consistent computation)")
|
||||
out()
|
||||
out("corr( danger(g) , P(hit | b_our = g) ) [the j128 metric, = -0.342 hist]")
|
||||
out("corr( danger(g) , P(hit | b_bullet = g) ) [same danger, bullet-conditioned target]")
|
||||
out()
|
||||
out("| danger map | corr vs P(hit\\|b_our=g) | corr vs P(hit\\|b_bullet=g) |")
|
||||
out("|---|---:|---:|")
|
||||
for name, (c_our, c_bul) in corr.items():
|
||||
out(f"| {name} | {c_our:+.3f} | {c_bul:+.3f} |")
|
||||
out()
|
||||
out("Negative = minimising the danger steers INTO where the observed hits")
|
||||
out("happen (the j128 defect). The exact bullet line is the physically")
|
||||
out("correct 'would this wave hit me at g' map; if its correlation is still")
|
||||
out("negative, exact geometry does NOT fix the inversion.")
|
||||
out()
|
||||
|
||||
out("| bin | P(hit\\|b_our) | P(hit\\|b_bullet) | hist danger | proxy danger | exact danger |")
|
||||
out("|---:|---:|---:|---:|---:|---:|")
|
||||
rb = {b: rate_bullet[usedb.index(b)] for b in usedb}
|
||||
for b in used:
|
||||
rb_str = f"{rb[b]*100:.1f}%" if b in rb else "—"
|
||||
out(f"| {b} | {rate_our[used.index(b)]*100:.1f}% | "
|
||||
f"{rb_str} | "
|
||||
f"{p_hist(recs, b):.3f} | {p_proxy(recs, b):.3f} | {p_exact(recs, b):.3f} |")
|
||||
out()
|
||||
|
||||
per = [run_split_exact(recs, s) for s in range(args.seeds)]
|
||||
ll_s = statistics.fmean(p["ll_state"] for p in per)
|
||||
ll_g = statistics.fmean(p["ll_statefree"] for p in per)
|
||||
out("## B. state-conditional information under the EXACT bullet-line label")
|
||||
out()
|
||||
out("held-out per-candidate log-loss (bits) of the exact label, "
|
||||
"state-conditional vs state-free (same rows, same split):")
|
||||
out()
|
||||
out("| model | log-loss (bits) |")
|
||||
out("|---|---:|")
|
||||
out(f"| state-free P(label | g) | {ll_g:.4f} |")
|
||||
out(f"| state-conditional P(label | state, g) | {ll_s:.4f} |")
|
||||
out(f"| Δ (state − state-free) | {ll_s - ll_g:+.4f} |")
|
||||
out()
|
||||
neg = sum(1 for p in per if p["delta"] < 0)
|
||||
out(f"state conditioning is better in {neg}/{len(per)} splits "
|
||||
f"(negative Δ = better).")
|
||||
out()
|
||||
|
||||
out("## C. open-loop decision counterfactual (VETO ONLY)")
|
||||
out()
|
||||
out("argmin_g danger with the recorded bullet line as ground truth:")
|
||||
out()
|
||||
out(f"| exact-label argmin (j131) | "
|
||||
f"{statistics.fmean(p['hit_out'] for p in per)*100:.2f}% |")
|
||||
out()
|
||||
|
||||
out("## MEASURED vs INFERRED")
|
||||
out()
|
||||
out("* MEASURED: every number above, on the recorded corpus.")
|
||||
out("* INFERRED: that an offline alignment transfers live — it cannot, the")
|
||||
out(" corpus is open loop (`docs/offline_harness_trust.md`).")
|
||||
|
||||
if args.report:
|
||||
os.makedirs(os.path.dirname(args.report), exist_ok=True)
|
||||
with open(args.report, "w") as f:
|
||||
f.write("\n".join(lines) + "\n")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -2243,3 +2243,134 @@ pre-registered rule records as WORSE.
|
||||
**Status: the default is UNCHANGED (`TR_MOVEMENT=strafe`); the outcome mode is
|
||||
default-off behind `TR_MOVEMENT=learned TR_LEARNED_LABEL=outcome`.** Revert = do
|
||||
not set the env vars.
|
||||
|
||||
---
|
||||
|
||||
## Learned movement — real bullet endpoints (exact geometry)
|
||||
|
||||
**Job j131. The owner's request:** *"use real bullets: bullets that really hit
|
||||
me, bullets that hit the wall, both detectable. We ignore bullets that hit
|
||||
other bots, this movement is only for 1v1."* The task's premise was that
|
||||
`ModularBot.nim` already handles `onBulletHit`/`onBulletHitWall`, so the exact
|
||||
bullet line was available live and j130's rejection of the exact label ("needs
|
||||
bullet bodies the bot lacks") was wrong.
|
||||
|
||||
### THE PREMISE IS HALF WRONG — VERIFIED (MEASURED, not inferred)
|
||||
|
||||
The **fields** exist: `BulletState` has `x, y, direction, power, ownerId,
|
||||
bulletId`, and `BulletHitWallEvent`/`HitByBulletEvent` both expose
|
||||
`bullet: BulletState`. But **the events are not routed to the dodger**:
|
||||
|
||||
* `BulletHitWallEvent` is delivered **only to the bullet's owner**
|
||||
(`addPrivateBotEvent(outcome.bullet.botId, …)` — verified by decompiling the
|
||||
running server jar `robocode-tankroyale-server-0.35.5-all.jar`, and identical
|
||||
in the 1.1.0 source `CollisionDetector.applyBulletWallCollisions`). So an
|
||||
**enemy** bullet hitting a wall is **not observable** by us.
|
||||
* `TurnToTickEventForBotMapper` builds `bulletStates = turn.bullets.filter
|
||||
{ it.botId == bot.id }`, so `getBulletStates()` returns **only our own**
|
||||
bullets too.
|
||||
* The events the dodger **does** receive with a real enemy-bullet endpoint are:
|
||||
`onHitByBullet` (the bullet hit US — endpoint = our impact point) and a
|
||||
bullet-vs-bullet event where **our** bullet intercepted an enemy bullet
|
||||
(`e.hitBullet` is the enemy bullet, with its endpoint + heading).
|
||||
|
||||
**So the "exact straight line from a wall hit" cannot be built live.** In 1v1 a
|
||||
missed bullet does end on a wall, but the server keeps that observation private
|
||||
to the shooter. This is the second time the availability premise is the binding
|
||||
constraint, now for the exact label rather than the proxy.
|
||||
|
||||
### WHAT CHANGED (code)
|
||||
|
||||
* `common_libs/movements/learned_surfer.nim` — **default-off**
|
||||
`TR_LEARNED_REAL_EVENTS=1` (registered in `env_report.knownEnvNames()`). When
|
||||
on, a wave is resolved by the REAL event instead of the arrival deadline:
|
||||
the exact `origin → endpoint` straight line sets the label's GF bin, the real
|
||||
flight time `currentTick − fireTick` is recorded (`resolvedReal`, `lastFlightErr`
|
||||
— a cross-check on the energy-drop speed inference), and the wave is **dropped
|
||||
at once** (`resolveEnemyBullet`), so no ghost accumulates. A wave no event
|
||||
claims resolves `RealEventsGrace` ticks past nominal as a **wall MISS**. With
|
||||
the knob off the byte-for-byte j130 behaviour is preserved (tests pin it).
|
||||
* `ModularBot_garage/src/ModularBot.nim` — forwards `onHitByBullet` (hit on us),
|
||||
a bullet-vs-bullet intercept of an enemy bullet (`e.hitBullet`), and (guarded,
|
||||
dead on 0.35.5) an enemy `onBulletHitWall` to `learnedMover.resolveEnemyBullet`.
|
||||
* `ModularBot_garage/tests/test_learned_surfer.nim` — real-event unit checks
|
||||
(default-off parity, exact centre-bin resolution, ghost drop, wall-miss
|
||||
deadline). `common_libs/tests/exact_geometry_gate.py` — Gate A/B below.
|
||||
|
||||
### GATE A — danger-map alignment, ONE consistent computation (MEASURED)
|
||||
|
||||
`python3 common_libs/tests/exact_geometry_gate.py --corpus /tmp/tfil_ab2/out`
|
||||
(70 battles, 54 923 shots, the same extraction and the same
|
||||
`corr(danger(g), P(hit | b_our=g))` metric j128/j130 used):
|
||||
|
||||
| danger map | corr vs `P(hit\|b_our=g)` | corr vs `P(hit\|b_bullet=g)` |
|
||||
|---|---:|---:|
|
||||
| histogram P(arrival = g) (j128) | **−0.341** | −0.206 |
|
||||
| outcome proxy `P(hit & \|g−b_our\|≤w)` (j130 live) | **+0.566** | +0.604 |
|
||||
| **EXACT bullet line `P(\|g−b_bullet\|≤w)`** | **−0.230** | **+0.120** |
|
||||
| exact bullet line & hit | +0.465 | +0.684 |
|
||||
|
||||
**The exact-geometry label does NOT fix the inversion on the j128 metric** —
|
||||
−0.230 is still negative (minimising it still steers into where the observed
|
||||
hits happen). It is *less* negative than the histogram (−0.341) and turns
|
||||
weakly positive (+0.120) only when the target is conditioned on the bullet's
|
||||
own line `b_bullet`, while the +0.566 proxy is inflated by being conditioned on
|
||||
`b_our` (the realised arrival, i.e. where the recorded wave already was). Under
|
||||
the task's own gate, **the veto fires and the live batch is not run.**
|
||||
|
||||
### GATE B — state information under the EXACT label (MEASURED)
|
||||
|
||||
Held-out per-candidate log-loss of the exact label, split BY BATTLE, 3 seeds:
|
||||
|
||||
| model | log-loss (bits) |
|
||||
|---|---:|
|
||||
| state-free `P(label \| g)` | **0.1879** |
|
||||
| state-conditional `P(label \| state, g)` | **0.3747** |
|
||||
| Δ (state − state-free) | **+0.1868** |
|
||||
|
||||
state conditioning is better in **0/3** splits. This **replicates j130 almost
|
||||
exactly** (proxy: 0.3906 vs 0.1873, Δ +0.203, 0/3). Under the exact label the
|
||||
coarse four-field state is still *worse* than the state-free model: the state
|
||||
buys no held-out information, so it cannot be the thing the learned mover is
|
||||
missing — **the observable state is still the binding constraint.**
|
||||
|
||||
### GATE C — live panel (NOT RUN, by the pre-registered rule)
|
||||
|
||||
Gate A's veto fired (exact correlation negative), so no live battles were
|
||||
fought. Independently, the live batch would have been testing a label the module
|
||||
**cannot construct** in the miss case (enemy wall endpoints are owner-private),
|
||||
so a live "exact" arm would in practice be j130's proxy for ~90% of waves.
|
||||
|
||||
### Direct answer
|
||||
|
||||
**Does exact bullet geometry fix the label? NO — not on the measured metric and
|
||||
not live.** The physically-exact map reads −0.230 against the j128 target
|
||||
(still inverted; the proxy's +0.566 is the one that is inflated). And the
|
||||
geometric endpoint **is not observable** by the dodger on this server for the
|
||||
miss case: `BulletHitWallEvent` and `bulletStates` are owner-private, so the
|
||||
only real enemy-bullet endpoints we get are the ~13% that hit us (and the rare
|
||||
intercepts). The exact line therefore cannot be built live for the waves that
|
||||
matter.
|
||||
|
||||
**Is the binding constraint the STATE rather than the label or the learner?
|
||||
YES — the same answer as j130, now measured for the third label.** Under the
|
||||
exact label the state still loses to state-free on held-out log-loss (0.3747 vs
|
||||
0.1879, 0/3 splits). j128 (histogram), j130 (outcome proxy) and j131 (exact
|
||||
line) each change the label; none moves the live result and none makes the
|
||||
state informative. The wave-crossing signal a 1v1 dodger needs is simply not in
|
||||
the four-field observable state, and hand-tuned `strafe` remains hard to beat.
|
||||
|
||||
### MEASURED vs INFERRED
|
||||
|
||||
**MEASURED:** the event routing (decompiled the running 0.35.5 jar +
|
||||
`TurnToTickEventForBotMapper`); the three-way Gate A correlation and the
|
||||
exact-label Gate B log-loss on the recorded corpus; the module unit tests
|
||||
(24/24, including the real-event and default-off parity checks); the env-report
|
||||
guard (25/25); the clean-archive compile. **INFERRED:** that the offline
|
||||
alignment transfers live — it cannot (open-loop corpus, see
|
||||
`docs/offline_harness_trust.md`).
|
||||
|
||||
**Status: the default is UNCHANGED (`TR_MOVEMENT=strafe`).** The real-event
|
||||
resolution is default-off behind `TR_MOVEMENT=learned TR_LEARNED_REAL_EVENTS=1`
|
||||
(combined with `TR_LEARNED_LABEL=outcome` for the dense readout). Revert = do
|
||||
not set the env vars.
|
||||
|
||||
Reference in New Issue
Block a user