j145 TFIL: a turn-cost tiebreak among the SAFE tiles (default-off)

The picker scored candidates on pathMaxHeat alone and then drew uniformly
among the survivors, so a mirror-side tile was as likely as a straight-ahead
one. Added a continuous turn cost as a DRAW WEIGHT applied only after the
hard heat filter:

  w = max(1, round(1 + TR_TFIL_TURN_BIAS * (1 - max(0,|turn| - REF)/180)))

- TR_TFIL_TURN_BIAS (default 0) is the odds ratio straight-ahead vs 180 deg;
  TR_TFIL_TURN_REF_DEG (default 45) is where the penalty starts. Both
  default-off-effect: the default-path golden in test_tfil_commit_env.nim is
  unchanged and still passes.
- Turn cost is NEVER folded into the heat score. The filter stays hard.
- The draw stays random (j51 measured an argmin worse); every weight is
  floored at 1, so the pool can never be emptied and bias 0 is exactly the
  shipped uniform draw.
- |turn| now travels on the ScoredTile, and the commit log gained turn /
  minturn / promote so a caller can measure the regret of the draw.

Guards: 51 -> 66 checks (an absurd 99:1 bias never rescues an over-threshold
tile; mean |turn|, draw regret, >90 and mirror-side shares all fall; path
heat does not rise). env_report + .env.example updated.
This commit is contained in:
2026-09-26 21:56:54 +02:00
parent 0e7e124c7f
commit 39c90fd930
7 changed files with 595 additions and 20 deletions
+98 -19
View File
@@ -128,6 +128,16 @@ var
TfilCommitArrival*: bool = false
TfilCommitMargin*: float = 0.0
TfilNoRevSpeed*: float = 0.0
## j145 — the turn-cost TIEBREAK, applied ONLY among tiles that already passed
## the hard heat filter. It is a WEIGHT on the draw, never a term in the heat
## score (see `turnWeights`).
## TR_TFIL_TURN_BIAS float the turn TIEBREAK's odds ratio: a
## straight-ahead safe tile is drawn
## `1 + bias` times as often as a 180 deg one
## (0 = off, today's uniform draw exactly)
## TR_TFIL_TURN_REF_DEG float turn below which there is no penalty (deg)
TfilTurnBias*: float = 0.0
TfilTurnRefDeg*: float = 45.0
## j134: the shared fire-detection correction (`TR_FIRE_FIX`, default on).
## Off = the shipped `prev - energy` detector byte-for-byte.
TfilFireFix*: bool = true
@@ -162,6 +172,8 @@ proc loadTfilCommitEnv*() =
TfilCommitArrival = getEnvBool("TR_TFIL_COMMIT_ARRIVAL", false)
TfilCommitMargin = max(0.0, getEnvFloat("TR_TFIL_COMMIT_MARGIN", 0.0))
TfilNoRevSpeed = max(0.0, getEnvFloat("TR_TFIL_NOREV_SPEED", 0.0))
TfilTurnBias = max(0.0, getEnvFloat("TR_TFIL_TURN_BIAS", 0.0))
TfilTurnRefDeg = max(0.0, getEnvFloat("TR_TFIL_TURN_REF_DEG", 45.0))
TfilFireFix = getEnvBool("TR_FIRE_FIX", true)
loadTfilCommitEnv()
@@ -289,6 +301,12 @@ type
replanReason: TfilReplanReason ## why the last commitment ended (log only)
lastPickCall: int ## callCount at the last pick (log only)
picks: int ## number of picks this round (log only)
lastPickPromoted: bool ## the last pick had to break the hard
## heat filter (fewer than 2 safe tiles)
lastPickMinTurn: float ## smallest |turn| available in the
## candidate set of the last pick (j145:
## lets a caller measure the REGRET of the
## draw instead of only the drawn value)
proc initTFIL*(): TFILModule = TFILModule(debugGraphics: false, fire: initFireTracker())
@@ -332,6 +350,8 @@ proc resetRound*(m: var TFILModule) =
m.replanReason = rrNone
m.lastPickCall = 0
m.picks = 0
m.lastPickPromoted = false
m.lastPickMinTurn = 0.0
# ── Commit diagnostics (TR_TFIL_COMMIT_LOG, off by default) ──────────────────
# One JSONL line per computeMove call, used by the A/B to prove the treatment
@@ -569,6 +589,27 @@ proc norevPool*(offs: openArray[float], threshold: float): seq[int] =
if abs(a) < abs(offs[best]): best = i
@[best]
proc turnWeights*(turns: openArray[float], bias, refDeg: float): seq[int] =
## j145: the turn-cost TIEBREAK, as an integer draw weight per candidate:
##
## w = max(1, round(1 + bias * (1 - max(0, |turn| - refDeg) / 180)))
##
## i.e. a straight-ahead (within `refDeg`) safe tile is drawn `1 + bias` times
## as often as a 180 deg one, falling LINEARLY in between. Read `bias` as that
## odds ratio: bias 0 = today's uniform draw, bias 9 = 10:1.
## Only tiles that already passed the hard heat filter reach this function, so
## it can never rescue a hot one.
## WHY A WEIGHT AND NOT `heat + k*turnDeg`: mixing the two trades dodging for
## smoothness, which is backwards in a bullet-dodging game — the safety filter
## stays hard and the turn only re-orders the survivors.
## WHY NOT AN ARGMIN: job j51 (3142b70) measured that randomness in this tie
## is LOAD-BEARING for this bot — a deterministic argmin scored worse. So the
## draw stays random and only its TILT is new.
## The weight is floored at 1, so the pool can never be starved.
for t in turns:
result.add max(1, int(round(1.0 + bias *
(1.0 - max(0.0, t - refDeg) / 180.0))))
proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
if m.cols == 0:
m.initGrid(ws.arenaWidth, ws.arenaHeight)
@@ -820,6 +861,8 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
var pickedRev = false
var pickedMidFlight = false
var pickedInterval = 0
var pickedTurn = 0.0 # log-only: |turn| to the tile that was chosen
var pickedPromoted = false ## log-only: the pick had to break the heat filter
# Hull + inside-tiles: only recompute on replan tick (commitTicks == 0)
type TileRef = tuple[col, row: int]
@@ -871,7 +914,10 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
# A single hot tile on the path (corridor, bullet core, enemy aura) makes the whole path unsafe.
const PathSampleStep = 18.0 # ~half a tile
const PathDangerThreshold = 10.0 # max lava on path; above this = unsafe
type ScoredTile = tuple[col, row: int; pathMaxHeat: float]
# j145: `turnDeg` is the |heading change| from the direction we are ALREADY
# travelling to the tile centre. It is carried on the candidate (never folded
# into `pathMaxHeat`) so the pick can bias among the safe tiles only.
type ScoredTile = tuple[col, row: int; pathMaxHeat: float; turnDeg: float]
proc pathMaxHeat(m: TFILModule, fx, fy, tx, ty: float): float =
## MAX lava on the straight-line segment (fx,fy) -> (tx,ty), sampled every
@@ -899,12 +945,20 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
while result > 180.0: result -= 360.0
while result < -180.0: result += 360.0
# j145: the committed travel direction, needed both to score the turn cost of
# every candidate and to draw from it. Computed here (not in the pick block)
# because the candidates are scored before the pick.
let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0
else: ws.selfHeading
var scoredTiles: seq[ScoredTile]
for t in coolTiles:
let tx = m.marginX + (t.col.float + 0.5) * GridSize
let ty = m.marginY + (t.row.float + 0.5) * GridSize
scoredTiles.add (col: t.col, row: t.row,
pathMaxHeat: pathMaxHeat(m, ws.selfX, ws.selfY, tx, ty))
pathMaxHeat: pathMaxHeat(m, ws.selfX, ws.selfY, tx, ty),
turnDeg: abs(tileOffTravel(m, t.col, t.row, ws.selfX,
ws.selfY, travelDeg)))
# Sort by pathMaxHeat ascending (insertion sort — small N)
for i in 1..<scoredTiles.len:
@@ -919,6 +973,8 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
# Fallback: if everything is hot, keep the 2 coolest paths anyway.
var safeTiles: seq[ScoredTile]
var blockedTiles: seq[ScoredTile]
m.lastPickPromoted = false # j145: set only by the promotion below, so it is
# always the flag OF THE PICK THAT JUST HAPPENED
for t in scoredTiles:
if t.pathMaxHeat <= PathDangerThreshold: safeTiles.add t
else: blockedTiles.add t
@@ -929,6 +985,7 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
let promote = min(needed, blockedTiles.len)
for i in 0..<promote:
safeTiles.add blockedTiles[i]
m.lastPickPromoted = true
blockedTiles = blockedTiles[promote ..< blockedTiles.len]
# Commitment logic. With every j144 knob at its default (all off) this is the
@@ -996,8 +1053,6 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
continue
candidates.add t
if candidates.len == 0: candidates = safeTiles # all blocked → ignore block
let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0
else: ws.selfHeading
# j144, no opposite-direction flip while still accelerating. Below the speed
# threshold the bot physically cannot complete a reversal before the bullet
# lands, so a mid-flight switch to the mirror side only destroys the dodge it
@@ -1013,23 +1068,40 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
for i in keep: narrowed.add candidates[i]
candidates = narrowed
var chosen = 0
if TfilNoRev and candidates.len >= 2:
# Soft no-reversal preference (arm C): down-weight — never filter — tiles
# that lie >90 deg from the current travel direction.
var dirs: seq[float]
for t in candidates:
let tx = m.marginX + (t.col.float + 0.5) * GridSize
let ty = m.marginY + (t.row.float + 0.5) * GridSize
dirs.add arctan2(ty - ws.selfY, tx - ws.selfX) * 180.0 / PI
let weights = noRevWeights(dirs, travelDeg)
var total = 0
var forward = 0
if (TfilNoRev or TfilTurnBias > 0.0) and candidates.len >= 2:
# Soft preferences — down-weight, never filter, and only ever among tiles
# that already passed the hard heat filter above:
# arm C (TR_TFIL_NO_REV, off by default) — 3:1 forward vs rearward,
# a BINARY cut: it cannot tell 20 deg from 90, nor 91 from 179.
# j145 (TR_TFIL_TURN_BIAS, off by default) — a CONTINUOUS turn cost,
# `turnWeights` = 1 + bias * (1 - excess/180).
# The two multiply, so turning one on never silently drops the other.
var weights = if TfilTurnBias > 0.0:
block:
var turns: seq[float]
for t in candidates: turns.add t.turnDeg
turnWeights(turns, TfilTurnBias, TfilTurnRefDeg)
else:
var w1 = newSeq[int](candidates.len)
for i in 0..<w1.len: w1[i] = 1
w1
if TfilNoRev:
var dirs: seq[float]
for t in candidates:
let tx = m.marginX + (t.col.float + 0.5) * GridSize
let ty = m.marginY + (t.row.float + 0.5) * GridSize
dirs.add arctan2(ty - ws.selfY, tx - ws.selfX) * 180.0 / PI
let nw = noRevWeights(dirs, travelDeg)
for i in 0..<nw.len: weights[i] = weights[i] * nw[i]
var total = 0
var wMax = 0
for w in weights:
total += w
if w > 1: inc forward
if forward == 0:
# Fallback: no forward tile exists → uniform draw, so the pool can
# never empty and the pick is identical to the shipped one.
wMax = max(wMax, w)
if wMax <= 1:
# Fallback: no candidate is preferred (every tile is equally bad, or all
# are rearward) → uniform draw, so the pool can never empty and the pick
# degrades to exactly the shipped one.
chosen = rand(candidates.high)
else:
let r = rand(total - 1)
@@ -1043,6 +1115,10 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
else:
chosen = rand(candidates.high)
let ct = candidates[chosen]
pickedPromoted = m.lastPickPromoted
var minTurn = Inf
for t in candidates: minTurn = min(minTurn, t.turnDeg)
m.lastPickMinTurn = if minTurn == Inf: ct.turnDeg else: minTurn
m.commitTarget = (x: m.marginX + (ct.col.float + 0.5) * GridSize,
y: m.marginY + (ct.row.float + 0.5) * GridSize)
m.commitTicks = TfilCommitTicks
@@ -1057,6 +1133,7 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
while rd < -180.0: rd += 360.0
pickedThisTick = true
pickedRev = abs(rd) > 90.0
pickedTurn = abs(rd)
pickedMidFlight = midFlight
pickedInterval = m.callCount - m.lastPickCall
m.lastPickCall = m.callCount
@@ -1129,6 +1206,8 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
tfilLogWrite("{\"tick\":" & $ws.tick & ",\"call\":" & $m.callCount &
",\"sp\":" & $ws.selfSpeed & ",\"ct\":" & $m.commitTicks &
",\"pick\":" & (if pickedThisTick: "1" else: "0") &
",\"turn\":" & $pickedTurn &
",\"promote\":" & (if pickedPromoted: "1" else: "0") &
",\"reason\":\"" & reasonName(reason) & "\",\"rev\":" &
(if pickedRev: "1" else: "0") & ",\"mid\":" &
(if pickedMidFlight: "1" else: "0") & ",\"interval\":" & $pickedInterval &