39c90fd930
The picker scored candidates on pathMaxHeat alone and then drew uniformly among the survivors, so a mirror-side tile was as likely as a straight-ahead one. Added a continuous turn cost as a DRAW WEIGHT applied only after the hard heat filter: w = max(1, round(1 + TR_TFIL_TURN_BIAS * (1 - max(0,|turn| - REF)/180))) - TR_TFIL_TURN_BIAS (default 0) is the odds ratio straight-ahead vs 180 deg; TR_TFIL_TURN_REF_DEG (default 45) is where the penalty starts. Both default-off-effect: the default-path golden in test_tfil_commit_env.nim is unchanged and still passes. - Turn cost is NEVER folded into the heat score. The filter stays hard. - The draw stays random (j51 measured an argmin worse); every weight is floored at 1, so the pool can never be emptied and bias 0 is exactly the shipped uniform draw. - |turn| now travels on the ScoredTile, and the commit log gained turn / minturn / promote so a caller can measure the regret of the draw. Guards: 51 -> 66 checks (an absurd 99:1 bias never rescues an over-threshold tile; mean |turn|, draw regret, >90 and mirror-side shares all fall; path heat does not rise). env_report + .env.example updated.
390 lines
19 KiB
Nim
390 lines
19 KiB
Nim
## j144 — the TFIL arrival-commitment mechanism ruler.
|
|
##
|
|
## Answers ONE question, offline, on the recorded DrussGT fixture
|
|
## (tools/fixtures/tr_drussgt_vs_modularbot.jsonl, 20026 ticks): does the j144
|
|
## fix stop the pathology the owner watched in the live GUI?
|
|
##
|
|
## *"when the tile to go is selected in just a few ticks, the bot is still
|
|
## accelerating and the target changes even if the path is still good, and
|
|
## choose a tile that is opposite way, in the meantime bullet arrived"*
|
|
##
|
|
## Decomposed into measurable quantities, per arm:
|
|
## * picks — how many times a target was chosen
|
|
## * mean hold — mean ticks a target is held (ticks/picks)
|
|
## * held < needed — fraction of commitments abandoned after
|
|
## FEWER ticks than the distance physically
|
|
## needs at MaxSpeed (the "still accelerating"
|
|
## tell: the bot was never going to get there)
|
|
## * rev / 100 ticks — picks >90 deg off the travel direction
|
|
## * rev while slow — those picks made at |speed| < MaxSpeed/2
|
|
## (THE OWNER'S FAILURE MODE)
|
|
## * opposite switch — a switch >90 deg off the travel direction
|
|
## * opposite while mid-flight — ... made while the previous target had NOT
|
|
## yet been reached (the target flips under a
|
|
## bot that is still building speed)
|
|
## * reached — fraction of commitments that ended with
|
|
## the bot actually on the committed tile
|
|
##
|
|
## This is a STATIC REPLAY of a veto, not a win claim: the live A/B decides
|
|
## damage/run and round wins (see docs/movement_campaign.md).
|
|
##
|
|
## nim c -r --path:common_libs common_libs/tests/measure_tfil_arrival.nim
|
|
|
|
import std/[os, json, random, math, strformat]
|
|
import std/strutils except fromHex # `fromHex` would clash with color.fromHex
|
|
import gun_harness/gun_interface
|
|
# Private-field access: include (do NOT import) the shipped mover.
|
|
include movements/the_floor_is_lava
|
|
|
|
const
|
|
repoRoot = currentSourcePath().parentDir.parentDir.parentDir
|
|
fixtureRel = "tr_drussgt_vs_modularbot.jsonl"
|
|
Seed = 20250923
|
|
ArenaW = 800.0
|
|
ArenaH = 600.0
|
|
HalfMax = MaxSpeed / 2.0 ## 4.0 px/tick: "still accelerating"
|
|
ArriveR = 18.0 ## must match ArriveRadius in the mover
|
|
|
|
# ── fixture ──────────────────────────────────────────────────────────────────
|
|
|
|
proc loadStates(): seq[WorldState] =
|
|
let path = repoRoot / "tools" / "fixtures" / fixtureRel
|
|
for rawLine in lines(path):
|
|
let line = rawLine.strip()
|
|
if line.len == 0: continue
|
|
let n = parseJson(line)
|
|
if n.hasKey("meta") or n.hasKey("end"): continue
|
|
let ex = n["ex"].getFloat()
|
|
let ey = n["ey"].getFloat()
|
|
result.add WorldState(
|
|
enemyX: ex, enemyY: ey,
|
|
enemyHeading: n["eh"].getFloat(), enemySpeed: n["es"].getFloat(),
|
|
enemyEnergy: n["ee"].getFloat(),
|
|
selfX: n["sx"].getFloat(), selfY: n["sy"].getFloat(),
|
|
selfHeading: n["sh"].getFloat(), selfSpeed: n["ss"].getFloat(),
|
|
selfEnergy: n["se"].getFloat(),
|
|
arenaWidth: ArenaW, arenaHeight: ArenaH,
|
|
tick: n["tick"].getInt(),
|
|
enemies: @[EnemyInfo(id: 1, x: ex, y: ey,
|
|
heading: n["eh"].getFloat(), speed: n["es"].getFloat(),
|
|
energy: n["ee"].getFloat())])
|
|
|
|
proc loadRoundStarts(): seq[int] =
|
|
let side = repoRoot / "tools" / "fixtures" / "drussgt_meta" /
|
|
(fixtureRel & ".rounds.json")
|
|
if not fileExists(side): return
|
|
for r in parseFile(side)["rounds"]:
|
|
result.add r["startTick"].getInt()
|
|
|
|
# ── per-arm accumulation ─────────────────────────────────────────────────────
|
|
|
|
type ArmStats = object
|
|
arm: string
|
|
env: string
|
|
ticks: int
|
|
picks: int
|
|
holdSum: int
|
|
rev: int ## >90 deg off travel direction
|
|
revSlow: int ## ... at |speed| < HalfMax
|
|
oppSwitch: int ## >90 deg off travel direction
|
|
oppMidFlight: int ## ... while the previous target was unreached
|
|
oppFlip: int ## ... AND more than 135 deg off: a true FLIP to
|
|
## the mirror side, which is what the owner watched
|
|
oppAngleSum: float ## sum |angle| over the mid-flight rearward
|
|
## switches — severity, not just count
|
|
oppMidSlow: int ## ... and made at |speed| < MaxSpeed/2: the owner's
|
|
## failure mode, exactly as the guard test counts it
|
|
reached: int ## commitments that ended on the committed tile
|
|
neededSum: int ## sum of ceil(distAtPick / MaxSpeed)
|
|
neededBeatsHold:int ## commitments held for FEWER ticks than needed
|
|
byReason: array[TfilReplanReason, int]
|
|
|
|
proc closeCommitment(s: var ArmStats, held: int, reached: bool) =
|
|
## `held` = ticks the target was held; `reached` = the bot ended inside the
|
|
## 18px arrival radius of the tile it was driving to.
|
|
s.holdSum += held
|
|
if reached: inc s.reached
|
|
|
|
# ── the replay ───────────────────────────────────────────────────────────────
|
|
|
|
proc replayArm(arm, envspec: string): ArmStats =
|
|
result.arm = arm
|
|
result.env = envspec
|
|
for tok in envspec.splitWhitespace():
|
|
let kv = tok.split('=', 1)
|
|
putEnv(kv[0], kv[1])
|
|
putEnv("TR_TFIL_COMMIT_LOG", "") # the module's own log, not this ruler's
|
|
loadTfilCommitEnv()
|
|
loadTfilHeatEnv()
|
|
|
|
randomize(Seed)
|
|
var m = initTFIL()
|
|
let states = loadStates()
|
|
let starts = loadRoundStarts()
|
|
var held = 0 # ticks the current target has been held
|
|
var distAtPick = 0.0 # distance to the target when it was chosen
|
|
var open = false # a commitment is currently being held
|
|
for i in 0..<states.len:
|
|
let ws = states[i]
|
|
if i == 0 or i in starts:
|
|
if open: result.closeCommitment(held, true) # round end: count the dwell
|
|
m.resetRound()
|
|
held = 0
|
|
open = false
|
|
let before = m.picks
|
|
let prevTarget = m.commitTarget
|
|
let prevHadPicks = before > 0
|
|
discard m.computeMove(ws)
|
|
inc result.ticks
|
|
if m.picks == before:
|
|
if open: inc held
|
|
continue
|
|
# ── a pick happened on this tick: close the previous commitment ──────────
|
|
let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0
|
|
else: ws.selfHeading
|
|
var rd = arctan2(m.commitTarget.y - ws.selfY, m.commitTarget.x - ws.selfX) *
|
|
180.0 / PI - travelDeg
|
|
while rd > 180.0: rd -= 360.0
|
|
while rd < -180.0: rd += 360.0
|
|
if open:
|
|
let remD = sqrt((ws.selfX - prevTarget.x)^2 + (ws.selfY - prevTarget.y)^2)
|
|
result.closeCommitment(held, remD < ArriveR)
|
|
let needed = int(ceil(distAtPick / MaxSpeed))
|
|
result.neededSum += needed
|
|
if held < needed: inc result.neededBeatsHold
|
|
inc result.picks
|
|
let d = sqrt((m.commitTarget.x - ws.selfX)^2 + (m.commitTarget.y - ws.selfY)^2)
|
|
distAtPick = d
|
|
held = 0
|
|
open = true
|
|
if abs(rd) > 90.0:
|
|
inc result.rev
|
|
if abs(ws.selfSpeed) < HalfMax: inc result.revSlow
|
|
if prevHadPicks:
|
|
inc result.oppSwitch
|
|
let remD = sqrt((ws.selfX - prevTarget.x)^2 + (ws.selfY - prevTarget.y)^2)
|
|
if remD >= ArriveR:
|
|
inc result.oppMidFlight
|
|
result.oppAngleSum += abs(rd)
|
|
if abs(rd) > 135.0: inc result.oppFlip
|
|
if abs(ws.selfSpeed) < HalfMax: inc result.oppMidSlow
|
|
# the reason is stashed on the module until the next pick is logged; read it
|
|
# from the live field the same way the mover's own JSONL log does
|
|
for rr in TfilReplanReason:
|
|
if m.replanReason == rr and rr != rrNone: inc result.byReason[rr]
|
|
if open: result.closeCommitment(held, true)
|
|
for tok in envspec.splitWhitespace():
|
|
putEnv(tok.split('=', 1)[0], "")
|
|
|
|
# ── reporting ────────────────────────────────────────────────────────────────
|
|
|
|
proc f(x: float, d = 2): string = formatFloat(x, ffDecimal, d)
|
|
proc pct(n, d: int): string =
|
|
if d == 0: return "-"
|
|
f(100.0 * n.float / d.float, 1) & "%"
|
|
|
|
let arms = [
|
|
("tfil (shipped)", ""),
|
|
("commit-only", "TR_TFIL_TILE_REPLAN=off"),
|
|
("arrive", "TR_TFIL_COMMIT_ARRIVAL=1"),
|
|
("arrive+hyst", "TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_COMMIT_MARGIN=10"),
|
|
("arrive+hyst+norev", "TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_COMMIT_MARGIN=10 TR_TFIL_NOREV_SPEED=4"),
|
|
("norev-alone", "TR_TFIL_NOREV_SPEED=4"),
|
|
]
|
|
|
|
echo "j144 TFIL arrival-commitment mechanism ruler — offline fixture replay\n"
|
|
echo "fixture: tools/fixtures/", fixtureRel, "\n"
|
|
var rows: seq[ArmStats]
|
|
for (name, envspec) in arms:
|
|
rows.add replayArm(name, envspec)
|
|
|
|
echo &"| arm | picks | mean hold (ticks) | held<needed | reached | rev/100 ticks | rev while slow | opposite switch | opposite mid-flight | mean abs(angle) | flips (>135 deg) | opp mid-flight while SLOW |"
|
|
echo "|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|"
|
|
for s in rows:
|
|
echo &"| {s.arm} | {s.picks} | {f(s.holdSum.float / max(1, s.picks).float)} | {pct(s.neededBeatsHold, s.picks)} | {pct(s.reached, s.picks)} | {f(100.0 * s.rev.float / max(1.0, s.ticks.float), 1)} | {s.revSlow} ({pct(s.revSlow, max(1, s.picks))}) | {s.oppSwitch} | {s.oppMidFlight} | {f(s.oppAngleSum.float / max(1.0, s.oppMidFlight.float), 1)} deg | {s.oppFlip} | {s.oppMidSlow} |"
|
|
|
|
echo "\n### ticks held vs ticks needed to reach the tile (mean needed = "
|
|
echo f(rows[0].neededSum.float / max(1, rows[0].picks).float, 1), " ticks on the shipped arm)\n"
|
|
echo &"| arm | mean needed | mean held | mean held - needed | abandoned early |"
|
|
echo "|---|---:|---:|---:|---:|"
|
|
for s in rows:
|
|
let need = s.neededSum.float / max(1, s.picks).float
|
|
let heldM = s.holdSum.float / max(1, s.picks).float
|
|
echo &"| {s.arm} | {f(need,1)} | {f(heldM,1)} | {f(heldM - need,1)} | {pct(s.neededBeatsHold, s.picks)} |"
|
|
|
|
echo "\n### why each commitment ended (picks only)\n"
|
|
echo "| arm | tile_self | tile_enemy | danger | expiry | arrival | hyst |"
|
|
echo "|---|---:|---:|---:|---:|---:|---:|"
|
|
for s in rows:
|
|
echo &"| {s.arm} | {s.byReason[rrTileSelf]} | {s.byReason[rrTileEnemy]} | {s.byReason[rrDanger]} | {s.byReason[rrExpiry]} | {s.byReason[rrArrival]} | {s.byReason[rrHyst]} |"
|
|
|
|
# ══════════════════════════════════════════════════════════════════════════════
|
|
# j145 — THE TURN-COST TIEBREAK, the mechanism ruler.
|
|
#
|
|
# Answers, on the SAME recorded fixture, whether `TR_TFIL_TURN_BIAS` does what
|
|
# it claims AMONG THE SAFE TILES and what it costs:
|
|
#
|
|
# mean |turn| — the |heading change| to the tile the picker chose
|
|
# mean regret — how many degrees worse than the SMALLEST-turn candidate
|
|
# actually available that choice was. The confound-free
|
|
# form: the arms draw from different candidate sets, so the
|
|
# raw mean alone can move without the mechanism biting.
|
|
# >90 / >135 — picks needing a real turn / a mirror-side pick
|
|
# path heat — mean max-lava on the straight path we were told to walk,
|
|
# and the share of picks over the hard threshold
|
|
# filter broken — the share of picks that had to promote a hot tile because
|
|
# FEWER THAN TWO tiles were safe (the shipped fallback)
|
|
# own-tile heat — mean lava under the bot's own wheels, every tick: THE
|
|
# SAFETY COST of the choice, measured where it is paid
|
|
# |turnRate| — the turn the bot actually EXECUTED, not just intended
|
|
# arrival — mean ticks to reach the committed tile, and the share of
|
|
# commitments that are actually reached
|
|
# ══════════════════════════════════════════════════════════════════════════════
|
|
|
|
const ArriveR2 = 18.0
|
|
|
|
type TurnStats = object
|
|
arm: string
|
|
env: string
|
|
ticks: int
|
|
picks: int
|
|
turnSum: float
|
|
minTurnSum: float
|
|
tookMin: int
|
|
bigTurn: int
|
|
flip: int
|
|
pathHeatSum: float
|
|
hot: int
|
|
broken: int
|
|
ownHeatSum: float
|
|
ownHot: int
|
|
turnRateSum: float
|
|
hardTurn: int
|
|
holdSum: int
|
|
reached: int
|
|
distSum: float
|
|
|
|
proc pathHeatAt(m: TFILModule, fx, fy, tx, ty: float): float =
|
|
## The mover's own sampler (PathSampleStep 18). The lava field is rebuilt from
|
|
## scratch every computeMove, so what is visible just after the call is the
|
|
## field the pick was actually made against.
|
|
let ddx = tx - fx
|
|
let ddy = ty - fy
|
|
let lineDist = sqrt(ddx*ddx + ddy*ddy)
|
|
if lineDist <= 0.1: return 0.0
|
|
let steps = max(1, int(lineDist / 18.0))
|
|
var h = 0.0
|
|
for si in 0..steps:
|
|
let fr = si.float / steps.float
|
|
let (sc, sr) = m.tileAt(fx + ddx * fr, fy + ddy * fr)
|
|
h = max(h, m.lavaAt(sc, sr))
|
|
h
|
|
|
|
proc replayTurn(arm, envspec: string): TurnStats =
|
|
result.arm = arm
|
|
result.env = envspec
|
|
for tok in envspec.splitWhitespace():
|
|
putEnv(tok.split('=', 1)[0], tok.split('=', 1)[1])
|
|
putEnv("TR_TFIL_COMMIT_LOG", "")
|
|
loadTfilCommitEnv()
|
|
loadTfilHeatEnv()
|
|
randomize(Seed)
|
|
var m = initTFIL()
|
|
let states = loadStates()
|
|
let starts = loadRoundStarts()
|
|
var prev = (x: 0.0, y: 0.0)
|
|
var hadPicks = false
|
|
var held = 0
|
|
for i in 0..<states.len:
|
|
let ws = states[i]
|
|
if i == 0 or i in starts:
|
|
m.resetRound()
|
|
hadPicks = false
|
|
held = 0
|
|
let before = m.picks
|
|
let cmd = m.computeMove(ws)
|
|
inc result.ticks
|
|
result.turnRateSum += abs(cmd.turnRate)
|
|
if abs(cmd.turnRate) >= 5.0: inc result.hardTurn
|
|
let (bc, br) = m.tileAt(ws.selfX, ws.selfY)
|
|
let own = m.lavaAt(bc, br)
|
|
result.ownHeatSum += own
|
|
if own > 10.0: inc result.ownHot
|
|
if m.picks != before:
|
|
if hadPicks: result.holdSum += held
|
|
let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0
|
|
else: ws.selfHeading
|
|
var rd = arctan2(m.commitTarget.y - ws.selfY, m.commitTarget.x - ws.selfX) *
|
|
180.0 / PI - travelDeg
|
|
while rd > 180.0: rd -= 360.0
|
|
while rd < -180.0: rd += 360.0
|
|
inc result.picks
|
|
result.turnSum += abs(rd)
|
|
result.minTurnSum += m.lastPickMinTurn
|
|
if abs(rd) <= m.lastPickMinTurn + 0.5: inc result.tookMin
|
|
if abs(rd) > 90.0: inc result.bigTurn
|
|
if abs(rd) > 135.0: inc result.flip
|
|
let ph = pathHeatAt(m, ws.selfX, ws.selfY, m.commitTarget.x, m.commitTarget.y)
|
|
result.pathHeatSum += ph
|
|
if ph > 10.0: inc result.hot
|
|
if m.lastPickPromoted: inc result.broken
|
|
if hadPicks and
|
|
sqrt((ws.selfX-prev.x)^2 + (ws.selfY-prev.y)^2) < ArriveR2: inc result.reached
|
|
result.distSum += sqrt((m.commitTarget.x - ws.selfX)^2 +
|
|
(m.commitTarget.y - ws.selfY)^2)
|
|
prev = m.commitTarget
|
|
hadPicks = true
|
|
held = 0
|
|
elif hadPicks:
|
|
inc held
|
|
if hadPicks: result.holdSum += held
|
|
for tok in envspec.splitWhitespace():
|
|
putEnv(tok.split('=', 1)[0], "")
|
|
|
|
proc mean(x: float, d: int): float =
|
|
if d == 0: return 0.0
|
|
x / d.float
|
|
|
|
let turnArms = [
|
|
("shipped", "TR_MOVEMENT=tfil"),
|
|
("arrive+norev", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4"),
|
|
("turn b=3", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=3"),
|
|
("turn b=9", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=9"),
|
|
("turn b=19", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19"),
|
|
("turn b=39", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=39"),
|
|
("turn b=9 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0"),
|
|
("turn b=19 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19 TR_TFIL_TURN_REF_DEG=0"),
|
|
("turn b=39 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=39 TR_TFIL_TURN_REF_DEG=0"),
|
|
("turn b=19 ref90", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19 TR_TFIL_TURN_REF_DEG=90"),
|
|
("arrive+norev b=9", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9"),
|
|
("arrive+norev b=19", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=19"),
|
|
("arrive+norev b=9 r90", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=90"),
|
|
("arrive+norev b=9 ref0", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0"),
|
|
]
|
|
|
|
echo "\n\n══ j145 TURN-COST TIEBREAK — mechanism ruler ══\n"
|
|
var trows: seq[TurnStats]
|
|
for (name, envspec) in turnArms:
|
|
trows.add replayTurn(name, envspec)
|
|
|
|
echo &"| arm | picks | mean \\|turn\\| | mean regret | took min-turn | >90 deg | opposite (>135) | mean path heat | path heat >10 | filter broken |"
|
|
echo "|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|"
|
|
for s in trows:
|
|
echo &"| {s.arm} | {s.picks} | {mean(s.turnSum, s.picks).formatFloat(ffDecimal,1)} | " &
|
|
&"{mean(s.turnSum - s.minTurnSum, s.picks).formatFloat(ffDecimal,1)} deg | " &
|
|
&"{pct(s.tookMin, s.picks)} | {pct(s.bigTurn, s.picks)} | {pct(s.flip, s.picks)} | " &
|
|
&"{mean(s.pathHeatSum, s.picks).formatFloat(ffDecimal,2)} | {pct(s.hot, s.picks)} | {pct(s.broken, s.picks)} |"
|
|
|
|
echo "\n### what the choice costs: the turn EXECUTED, the arrival time, the distance\n"
|
|
echo "### (the heat under the bot's own wheels is NOT here: this replay drives the"
|
|
echo "### RECORDED self states, so the travelled path is identical in every arm by"
|
|
echo "### construction — the pick varies, the trajectory does not. The safety cost"
|
|
echo "### of a bias that arrives later / dwells hotter is therefore NOT measurable"
|
|
echo "### offline here; `mean path heat` and `filter broken` are the honest proxies.)\n"
|
|
echo &"| arm | mean \\|turnRate\\| executed | hard turns (>=5 deg/tick) | mean arrival (ticks) | reached | mean distance to the pick |"
|
|
echo "|---|---:|---:|---:|---:|---:|"
|
|
for s in trows:
|
|
echo &"| {s.arm} | {mean(s.turnRateSum, s.ticks).formatFloat(ffDecimal,2)} | {pct(s.hardTurn, s.ticks)} | " &
|
|
&"{mean(s.holdSum.float, s.picks).formatFloat(ffDecimal,1)} | {pct(s.reached, s.picks)} | " &
|
|
&"{mean(s.distSum, s.picks).formatFloat(ffDecimal,0)} px |"
|