j145 TFIL: a turn-cost tiebreak among the SAFE tiles (default-off)
The picker scored candidates on pathMaxHeat alone and then drew uniformly among the survivors, so a mirror-side tile was as likely as a straight-ahead one. Added a continuous turn cost as a DRAW WEIGHT applied only after the hard heat filter: w = max(1, round(1 + TR_TFIL_TURN_BIAS * (1 - max(0,|turn| - REF)/180))) - TR_TFIL_TURN_BIAS (default 0) is the odds ratio straight-ahead vs 180 deg; TR_TFIL_TURN_REF_DEG (default 45) is where the penalty starts. Both default-off-effect: the default-path golden in test_tfil_commit_env.nim is unchanged and still passes. - Turn cost is NEVER folded into the heat score. The filter stays hard. - The draw stays random (j51 measured an argmin worse); every weight is floored at 1, so the pool can never be emptied and bias 0 is exactly the shipped uniform draw. - |turn| now travels on the ScoredTile, and the commit log gained turn / minturn / promote so a caller can measure the regret of the draw. Guards: 51 -> 66 checks (an absurd 99:1 bias never rescues an over-threshold tile; mean |turn|, draw regret, >90 and mirror-side shares all fall; path heat does not rise). env_report + .env.example updated.
This commit is contained in:
@@ -217,3 +217,173 @@ echo "| arm | tile_self | tile_enemy | danger | expiry | arrival | hyst |"
|
||||
echo "|---|---:|---:|---:|---:|---:|---:|"
|
||||
for s in rows:
|
||||
echo &"| {s.arm} | {s.byReason[rrTileSelf]} | {s.byReason[rrTileEnemy]} | {s.byReason[rrDanger]} | {s.byReason[rrExpiry]} | {s.byReason[rrArrival]} | {s.byReason[rrHyst]} |"
|
||||
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
# j145 — THE TURN-COST TIEBREAK, the mechanism ruler.
|
||||
#
|
||||
# Answers, on the SAME recorded fixture, whether `TR_TFIL_TURN_BIAS` does what
|
||||
# it claims AMONG THE SAFE TILES and what it costs:
|
||||
#
|
||||
# mean |turn| — the |heading change| to the tile the picker chose
|
||||
# mean regret — how many degrees worse than the SMALLEST-turn candidate
|
||||
# actually available that choice was. The confound-free
|
||||
# form: the arms draw from different candidate sets, so the
|
||||
# raw mean alone can move without the mechanism biting.
|
||||
# >90 / >135 — picks needing a real turn / a mirror-side pick
|
||||
# path heat — mean max-lava on the straight path we were told to walk,
|
||||
# and the share of picks over the hard threshold
|
||||
# filter broken — the share of picks that had to promote a hot tile because
|
||||
# FEWER THAN TWO tiles were safe (the shipped fallback)
|
||||
# own-tile heat — mean lava under the bot's own wheels, every tick: THE
|
||||
# SAFETY COST of the choice, measured where it is paid
|
||||
# |turnRate| — the turn the bot actually EXECUTED, not just intended
|
||||
# arrival — mean ticks to reach the committed tile, and the share of
|
||||
# commitments that are actually reached
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
const ArriveR2 = 18.0
|
||||
|
||||
type TurnStats = object
|
||||
arm: string
|
||||
env: string
|
||||
ticks: int
|
||||
picks: int
|
||||
turnSum: float
|
||||
minTurnSum: float
|
||||
tookMin: int
|
||||
bigTurn: int
|
||||
flip: int
|
||||
pathHeatSum: float
|
||||
hot: int
|
||||
broken: int
|
||||
ownHeatSum: float
|
||||
ownHot: int
|
||||
turnRateSum: float
|
||||
hardTurn: int
|
||||
holdSum: int
|
||||
reached: int
|
||||
distSum: float
|
||||
|
||||
proc pathHeatAt(m: TFILModule, fx, fy, tx, ty: float): float =
|
||||
## The mover's own sampler (PathSampleStep 18). The lava field is rebuilt from
|
||||
## scratch every computeMove, so what is visible just after the call is the
|
||||
## field the pick was actually made against.
|
||||
let ddx = tx - fx
|
||||
let ddy = ty - fy
|
||||
let lineDist = sqrt(ddx*ddx + ddy*ddy)
|
||||
if lineDist <= 0.1: return 0.0
|
||||
let steps = max(1, int(lineDist / 18.0))
|
||||
var h = 0.0
|
||||
for si in 0..steps:
|
||||
let fr = si.float / steps.float
|
||||
let (sc, sr) = m.tileAt(fx + ddx * fr, fy + ddy * fr)
|
||||
h = max(h, m.lavaAt(sc, sr))
|
||||
h
|
||||
|
||||
proc replayTurn(arm, envspec: string): TurnStats =
|
||||
result.arm = arm
|
||||
result.env = envspec
|
||||
for tok in envspec.splitWhitespace():
|
||||
putEnv(tok.split('=', 1)[0], tok.split('=', 1)[1])
|
||||
putEnv("TR_TFIL_COMMIT_LOG", "")
|
||||
loadTfilCommitEnv()
|
||||
loadTfilHeatEnv()
|
||||
randomize(Seed)
|
||||
var m = initTFIL()
|
||||
let states = loadStates()
|
||||
let starts = loadRoundStarts()
|
||||
var prev = (x: 0.0, y: 0.0)
|
||||
var hadPicks = false
|
||||
var held = 0
|
||||
for i in 0..<states.len:
|
||||
let ws = states[i]
|
||||
if i == 0 or i in starts:
|
||||
m.resetRound()
|
||||
hadPicks = false
|
||||
held = 0
|
||||
let before = m.picks
|
||||
let cmd = m.computeMove(ws)
|
||||
inc result.ticks
|
||||
result.turnRateSum += abs(cmd.turnRate)
|
||||
if abs(cmd.turnRate) >= 5.0: inc result.hardTurn
|
||||
let (bc, br) = m.tileAt(ws.selfX, ws.selfY)
|
||||
let own = m.lavaAt(bc, br)
|
||||
result.ownHeatSum += own
|
||||
if own > 10.0: inc result.ownHot
|
||||
if m.picks != before:
|
||||
if hadPicks: result.holdSum += held
|
||||
let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0
|
||||
else: ws.selfHeading
|
||||
var rd = arctan2(m.commitTarget.y - ws.selfY, m.commitTarget.x - ws.selfX) *
|
||||
180.0 / PI - travelDeg
|
||||
while rd > 180.0: rd -= 360.0
|
||||
while rd < -180.0: rd += 360.0
|
||||
inc result.picks
|
||||
result.turnSum += abs(rd)
|
||||
result.minTurnSum += m.lastPickMinTurn
|
||||
if abs(rd) <= m.lastPickMinTurn + 0.5: inc result.tookMin
|
||||
if abs(rd) > 90.0: inc result.bigTurn
|
||||
if abs(rd) > 135.0: inc result.flip
|
||||
let ph = pathHeatAt(m, ws.selfX, ws.selfY, m.commitTarget.x, m.commitTarget.y)
|
||||
result.pathHeatSum += ph
|
||||
if ph > 10.0: inc result.hot
|
||||
if m.lastPickPromoted: inc result.broken
|
||||
if hadPicks and
|
||||
sqrt((ws.selfX-prev.x)^2 + (ws.selfY-prev.y)^2) < ArriveR2: inc result.reached
|
||||
result.distSum += sqrt((m.commitTarget.x - ws.selfX)^2 +
|
||||
(m.commitTarget.y - ws.selfY)^2)
|
||||
prev = m.commitTarget
|
||||
hadPicks = true
|
||||
held = 0
|
||||
elif hadPicks:
|
||||
inc held
|
||||
if hadPicks: result.holdSum += held
|
||||
for tok in envspec.splitWhitespace():
|
||||
putEnv(tok.split('=', 1)[0], "")
|
||||
|
||||
proc mean(x: float, d: int): float =
|
||||
if d == 0: return 0.0
|
||||
x / d.float
|
||||
|
||||
let turnArms = [
|
||||
("shipped", "TR_MOVEMENT=tfil"),
|
||||
("arrive+norev", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4"),
|
||||
("turn b=3", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=3"),
|
||||
("turn b=9", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=9"),
|
||||
("turn b=19", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19"),
|
||||
("turn b=39", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=39"),
|
||||
("turn b=9 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0"),
|
||||
("turn b=19 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19 TR_TFIL_TURN_REF_DEG=0"),
|
||||
("turn b=39 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=39 TR_TFIL_TURN_REF_DEG=0"),
|
||||
("turn b=19 ref90", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19 TR_TFIL_TURN_REF_DEG=90"),
|
||||
("arrive+norev b=9", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9"),
|
||||
("arrive+norev b=19", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=19"),
|
||||
("arrive+norev b=9 r90", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=90"),
|
||||
("arrive+norev b=9 ref0", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0"),
|
||||
]
|
||||
|
||||
echo "\n\n══ j145 TURN-COST TIEBREAK — mechanism ruler ══\n"
|
||||
var trows: seq[TurnStats]
|
||||
for (name, envspec) in turnArms:
|
||||
trows.add replayTurn(name, envspec)
|
||||
|
||||
echo &"| arm | picks | mean \\|turn\\| | mean regret | took min-turn | >90 deg | opposite (>135) | mean path heat | path heat >10 | filter broken |"
|
||||
echo "|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|"
|
||||
for s in trows:
|
||||
echo &"| {s.arm} | {s.picks} | {mean(s.turnSum, s.picks).formatFloat(ffDecimal,1)} | " &
|
||||
&"{mean(s.turnSum - s.minTurnSum, s.picks).formatFloat(ffDecimal,1)} deg | " &
|
||||
&"{pct(s.tookMin, s.picks)} | {pct(s.bigTurn, s.picks)} | {pct(s.flip, s.picks)} | " &
|
||||
&"{mean(s.pathHeatSum, s.picks).formatFloat(ffDecimal,2)} | {pct(s.hot, s.picks)} | {pct(s.broken, s.picks)} |"
|
||||
|
||||
echo "\n### what the choice costs: the turn EXECUTED, the arrival time, the distance\n"
|
||||
echo "### (the heat under the bot's own wheels is NOT here: this replay drives the"
|
||||
echo "### RECORDED self states, so the travelled path is identical in every arm by"
|
||||
echo "### construction — the pick varies, the trajectory does not. The safety cost"
|
||||
echo "### of a bias that arrives later / dwells hotter is therefore NOT measurable"
|
||||
echo "### offline here; `mean path heat` and `filter broken` are the honest proxies.)\n"
|
||||
echo &"| arm | mean \\|turnRate\\| executed | hard turns (>=5 deg/tick) | mean arrival (ticks) | reached | mean distance to the pick |"
|
||||
echo "|---|---:|---:|---:|---:|---:|"
|
||||
for s in trows:
|
||||
echo &"| {s.arm} | {mean(s.turnRateSum, s.ticks).formatFloat(ffDecimal,2)} | {pct(s.hardTurn, s.ticks)} | " &
|
||||
&"{mean(s.holdSum.float, s.picks).formatFloat(ffDecimal,1)} | {pct(s.reached, s.picks)} | " &
|
||||
&"{mean(s.distSum, s.picks).formatFloat(ffDecimal,0)} px |"
|
||||
|
||||
Reference in New Issue
Block a user