j145 TFIL: a turn-cost tiebreak among the SAFE tiles (default-off)
The picker scored candidates on pathMaxHeat alone and then drew uniformly among the survivors, so a mirror-side tile was as likely as a straight-ahead one. Added a continuous turn cost as a DRAW WEIGHT applied only after the hard heat filter: w = max(1, round(1 + TR_TFIL_TURN_BIAS * (1 - max(0,|turn| - REF)/180))) - TR_TFIL_TURN_BIAS (default 0) is the odds ratio straight-ahead vs 180 deg; TR_TFIL_TURN_REF_DEG (default 45) is where the penalty starts. Both default-off-effect: the default-path golden in test_tfil_commit_env.nim is unchanged and still passes. - Turn cost is NEVER folded into the heat score. The filter stays hard. - The draw stays random (j51 measured an argmin worse); every weight is floored at 1, so the pool can never be emptied and bias 0 is exactly the shipped uniform draw. - |turn| now travels on the ScoredTile, and the commit log gained turn / minturn / promote so a caller can measure the regret of the draw. Guards: 51 -> 66 checks (an absurd 99:1 bias never rescues an over-threshold tile; mean |turn|, draw regret, >90 and mirror-side shares all fall; path heat does not rise). env_report + .env.example updated.
This commit is contained in:
@@ -114,6 +114,8 @@ TR_TFIL_COMMIT_LOG= # path for the per-commit log; empty = no log
|
|||||||
TR_TFIL_COMMIT_ARRIVAL=off # on = hold the dodge tile until we are ON it (not a fixed dwell)
|
TR_TFIL_COMMIT_ARRIVAL=off # on = hold the dodge tile until we are ON it (not a fixed dwell)
|
||||||
TR_TFIL_COMMIT_MARGIN=0.0 # lava an alternative tile must be cooler by before it wins the tile
|
TR_TFIL_COMMIT_MARGIN=0.0 # lava an alternative tile must be cooler by before it wins the tile
|
||||||
TR_TFIL_NOREV_SPEED=0.0 # px/tick; below this, a mid-flight switch may not turn the bot around
|
TR_TFIL_NOREV_SPEED=0.0 # px/tick; below this, a mid-flight switch may not turn the bot around
|
||||||
|
TR_TFIL_TURN_BIAS=0.0 # turn TIEBREAK odds ratio among SAFE tiles; 0 = uniform draw as today
|
||||||
|
TR_TFIL_TURN_REF_DEG=45.0 # deg; turn below which the tiebreak applies no penalty
|
||||||
TR_TFIL_HEAT_TIME=off # on = index bullet heat by time (flat field when off)
|
TR_TFIL_HEAT_TIME=off # on = index bullet heat by time (flat field when off)
|
||||||
TR_TFIL_HEAT_TAU=9.0 # ticks a tracked bullet's heat lives for
|
TR_TFIL_HEAT_TAU=9.0 # ticks a tracked bullet's heat lives for
|
||||||
TR_TFIL_HEAT_POWER_GAIN=1.0 # scale of the heat a bullet paints, per firepower
|
TR_TFIL_HEAT_POWER_GAIN=1.0 # scale of the heat a bullet paints, per firepower
|
||||||
|
|||||||
@@ -324,6 +324,8 @@ proc printEffectiveValues(ctx: EnvReportContext) =
|
|||||||
emit("TR_TFIL_COMMIT_MARGIN", $TfilCommitMargin,
|
emit("TR_TFIL_COMMIT_MARGIN", $TfilCommitMargin,
|
||||||
sourceOf("TR_TFIL_COMMIT_MARGIN"))
|
sourceOf("TR_TFIL_COMMIT_MARGIN"))
|
||||||
emit("TR_TFIL_NOREV_SPEED", $TfilNoRevSpeed, sourceOf("TR_TFIL_NOREV_SPEED"))
|
emit("TR_TFIL_NOREV_SPEED", $TfilNoRevSpeed, sourceOf("TR_TFIL_NOREV_SPEED"))
|
||||||
|
emit("TR_TFIL_TURN_BIAS", $TfilTurnBias, sourceOf("TR_TFIL_TURN_BIAS"))
|
||||||
|
emit("TR_TFIL_TURN_REF_DEG", $TfilTurnRefDeg, sourceOf("TR_TFIL_TURN_REF_DEG"))
|
||||||
# time-indexed bullet heat (default off = shipped flat model)
|
# time-indexed bullet heat (default off = shipped flat model)
|
||||||
emit("TR_TFIL_HEAT_TIME", onOff(TfilHeatTime), sourceOf("TR_TFIL_HEAT_TIME"))
|
emit("TR_TFIL_HEAT_TIME", onOff(TfilHeatTime), sourceOf("TR_TFIL_HEAT_TIME"))
|
||||||
emit("TR_TFIL_HEAT_TAU", $TfilHeatTau, sourceOf("TR_TFIL_HEAT_TAU"))
|
emit("TR_TFIL_HEAT_TAU", $TfilHeatTau, sourceOf("TR_TFIL_HEAT_TAU"))
|
||||||
@@ -639,7 +641,7 @@ proc knownEnvNames*(): seq[string] =
|
|||||||
"TR_TFIL_WALL_RADIANCE",
|
"TR_TFIL_WALL_RADIANCE",
|
||||||
"TR_TFIL_TILE_REPLAN", "TR_TFIL_COMMIT_TICKS", "TR_TFIL_NO_REV",
|
"TR_TFIL_TILE_REPLAN", "TR_TFIL_COMMIT_TICKS", "TR_TFIL_NO_REV",
|
||||||
"TR_TFIL_COMMIT_LOG", "TR_TFIL_COMMIT_ARRIVAL", "TR_TFIL_COMMIT_MARGIN",
|
"TR_TFIL_COMMIT_LOG", "TR_TFIL_COMMIT_ARRIVAL", "TR_TFIL_COMMIT_MARGIN",
|
||||||
"TR_TFIL_NOREV_SPEED",
|
"TR_TFIL_NOREV_SPEED", "TR_TFIL_TURN_BIAS", "TR_TFIL_TURN_REF_DEG",
|
||||||
"TR_TFIL_HEAT_TIME", "TR_TFIL_HEAT_TAU", "TR_TFIL_HEAT_POWER_GAIN",
|
"TR_TFIL_HEAT_TIME", "TR_TFIL_HEAT_TAU", "TR_TFIL_HEAT_POWER_GAIN",
|
||||||
"TR_TFIL_PILLAR_ON",
|
"TR_TFIL_PILLAR_ON",
|
||||||
"TR_STRAFE_BAND", "TR_STRAFE_SPREAD", "TR_STRAFE_REACH",
|
"TR_STRAFE_BAND", "TR_STRAFE_SPREAD", "TR_STRAFE_REACH",
|
||||||
|
|||||||
@@ -128,6 +128,16 @@ var
|
|||||||
TfilCommitArrival*: bool = false
|
TfilCommitArrival*: bool = false
|
||||||
TfilCommitMargin*: float = 0.0
|
TfilCommitMargin*: float = 0.0
|
||||||
TfilNoRevSpeed*: float = 0.0
|
TfilNoRevSpeed*: float = 0.0
|
||||||
|
## j145 — the turn-cost TIEBREAK, applied ONLY among tiles that already passed
|
||||||
|
## the hard heat filter. It is a WEIGHT on the draw, never a term in the heat
|
||||||
|
## score (see `turnWeights`).
|
||||||
|
## TR_TFIL_TURN_BIAS float the turn TIEBREAK's odds ratio: a
|
||||||
|
## straight-ahead safe tile is drawn
|
||||||
|
## `1 + bias` times as often as a 180 deg one
|
||||||
|
## (0 = off, today's uniform draw exactly)
|
||||||
|
## TR_TFIL_TURN_REF_DEG float turn below which there is no penalty (deg)
|
||||||
|
TfilTurnBias*: float = 0.0
|
||||||
|
TfilTurnRefDeg*: float = 45.0
|
||||||
## j134: the shared fire-detection correction (`TR_FIRE_FIX`, default on).
|
## j134: the shared fire-detection correction (`TR_FIRE_FIX`, default on).
|
||||||
## Off = the shipped `prev - energy` detector byte-for-byte.
|
## Off = the shipped `prev - energy` detector byte-for-byte.
|
||||||
TfilFireFix*: bool = true
|
TfilFireFix*: bool = true
|
||||||
@@ -162,6 +172,8 @@ proc loadTfilCommitEnv*() =
|
|||||||
TfilCommitArrival = getEnvBool("TR_TFIL_COMMIT_ARRIVAL", false)
|
TfilCommitArrival = getEnvBool("TR_TFIL_COMMIT_ARRIVAL", false)
|
||||||
TfilCommitMargin = max(0.0, getEnvFloat("TR_TFIL_COMMIT_MARGIN", 0.0))
|
TfilCommitMargin = max(0.0, getEnvFloat("TR_TFIL_COMMIT_MARGIN", 0.0))
|
||||||
TfilNoRevSpeed = max(0.0, getEnvFloat("TR_TFIL_NOREV_SPEED", 0.0))
|
TfilNoRevSpeed = max(0.0, getEnvFloat("TR_TFIL_NOREV_SPEED", 0.0))
|
||||||
|
TfilTurnBias = max(0.0, getEnvFloat("TR_TFIL_TURN_BIAS", 0.0))
|
||||||
|
TfilTurnRefDeg = max(0.0, getEnvFloat("TR_TFIL_TURN_REF_DEG", 45.0))
|
||||||
TfilFireFix = getEnvBool("TR_FIRE_FIX", true)
|
TfilFireFix = getEnvBool("TR_FIRE_FIX", true)
|
||||||
|
|
||||||
loadTfilCommitEnv()
|
loadTfilCommitEnv()
|
||||||
@@ -289,6 +301,12 @@ type
|
|||||||
replanReason: TfilReplanReason ## why the last commitment ended (log only)
|
replanReason: TfilReplanReason ## why the last commitment ended (log only)
|
||||||
lastPickCall: int ## callCount at the last pick (log only)
|
lastPickCall: int ## callCount at the last pick (log only)
|
||||||
picks: int ## number of picks this round (log only)
|
picks: int ## number of picks this round (log only)
|
||||||
|
lastPickPromoted: bool ## the last pick had to break the hard
|
||||||
|
## heat filter (fewer than 2 safe tiles)
|
||||||
|
lastPickMinTurn: float ## smallest |turn| available in the
|
||||||
|
## candidate set of the last pick (j145:
|
||||||
|
## lets a caller measure the REGRET of the
|
||||||
|
## draw instead of only the drawn value)
|
||||||
|
|
||||||
proc initTFIL*(): TFILModule = TFILModule(debugGraphics: false, fire: initFireTracker())
|
proc initTFIL*(): TFILModule = TFILModule(debugGraphics: false, fire: initFireTracker())
|
||||||
|
|
||||||
@@ -332,6 +350,8 @@ proc resetRound*(m: var TFILModule) =
|
|||||||
m.replanReason = rrNone
|
m.replanReason = rrNone
|
||||||
m.lastPickCall = 0
|
m.lastPickCall = 0
|
||||||
m.picks = 0
|
m.picks = 0
|
||||||
|
m.lastPickPromoted = false
|
||||||
|
m.lastPickMinTurn = 0.0
|
||||||
|
|
||||||
# ── Commit diagnostics (TR_TFIL_COMMIT_LOG, off by default) ──────────────────
|
# ── Commit diagnostics (TR_TFIL_COMMIT_LOG, off by default) ──────────────────
|
||||||
# One JSONL line per computeMove call, used by the A/B to prove the treatment
|
# One JSONL line per computeMove call, used by the A/B to prove the treatment
|
||||||
@@ -569,6 +589,27 @@ proc norevPool*(offs: openArray[float], threshold: float): seq[int] =
|
|||||||
if abs(a) < abs(offs[best]): best = i
|
if abs(a) < abs(offs[best]): best = i
|
||||||
@[best]
|
@[best]
|
||||||
|
|
||||||
|
proc turnWeights*(turns: openArray[float], bias, refDeg: float): seq[int] =
|
||||||
|
## j145: the turn-cost TIEBREAK, as an integer draw weight per candidate:
|
||||||
|
##
|
||||||
|
## w = max(1, round(1 + bias * (1 - max(0, |turn| - refDeg) / 180)))
|
||||||
|
##
|
||||||
|
## i.e. a straight-ahead (within `refDeg`) safe tile is drawn `1 + bias` times
|
||||||
|
## as often as a 180 deg one, falling LINEARLY in between. Read `bias` as that
|
||||||
|
## odds ratio: bias 0 = today's uniform draw, bias 9 = 10:1.
|
||||||
|
## Only tiles that already passed the hard heat filter reach this function, so
|
||||||
|
## it can never rescue a hot one.
|
||||||
|
## WHY A WEIGHT AND NOT `heat + k*turnDeg`: mixing the two trades dodging for
|
||||||
|
## smoothness, which is backwards in a bullet-dodging game — the safety filter
|
||||||
|
## stays hard and the turn only re-orders the survivors.
|
||||||
|
## WHY NOT AN ARGMIN: job j51 (3142b70) measured that randomness in this tie
|
||||||
|
## is LOAD-BEARING for this bot — a deterministic argmin scored worse. So the
|
||||||
|
## draw stays random and only its TILT is new.
|
||||||
|
## The weight is floored at 1, so the pool can never be starved.
|
||||||
|
for t in turns:
|
||||||
|
result.add max(1, int(round(1.0 + bias *
|
||||||
|
(1.0 - max(0.0, t - refDeg) / 180.0))))
|
||||||
|
|
||||||
proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
||||||
if m.cols == 0:
|
if m.cols == 0:
|
||||||
m.initGrid(ws.arenaWidth, ws.arenaHeight)
|
m.initGrid(ws.arenaWidth, ws.arenaHeight)
|
||||||
@@ -820,6 +861,8 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
|||||||
var pickedRev = false
|
var pickedRev = false
|
||||||
var pickedMidFlight = false
|
var pickedMidFlight = false
|
||||||
var pickedInterval = 0
|
var pickedInterval = 0
|
||||||
|
var pickedTurn = 0.0 # log-only: |turn| to the tile that was chosen
|
||||||
|
var pickedPromoted = false ## log-only: the pick had to break the heat filter
|
||||||
|
|
||||||
# Hull + inside-tiles: only recompute on replan tick (commitTicks == 0)
|
# Hull + inside-tiles: only recompute on replan tick (commitTicks == 0)
|
||||||
type TileRef = tuple[col, row: int]
|
type TileRef = tuple[col, row: int]
|
||||||
@@ -871,7 +914,10 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
|||||||
# A single hot tile on the path (corridor, bullet core, enemy aura) makes the whole path unsafe.
|
# A single hot tile on the path (corridor, bullet core, enemy aura) makes the whole path unsafe.
|
||||||
const PathSampleStep = 18.0 # ~half a tile
|
const PathSampleStep = 18.0 # ~half a tile
|
||||||
const PathDangerThreshold = 10.0 # max lava on path; above this = unsafe
|
const PathDangerThreshold = 10.0 # max lava on path; above this = unsafe
|
||||||
type ScoredTile = tuple[col, row: int; pathMaxHeat: float]
|
# j145: `turnDeg` is the |heading change| from the direction we are ALREADY
|
||||||
|
# travelling to the tile centre. It is carried on the candidate (never folded
|
||||||
|
# into `pathMaxHeat`) so the pick can bias among the safe tiles only.
|
||||||
|
type ScoredTile = tuple[col, row: int; pathMaxHeat: float; turnDeg: float]
|
||||||
|
|
||||||
proc pathMaxHeat(m: TFILModule, fx, fy, tx, ty: float): float =
|
proc pathMaxHeat(m: TFILModule, fx, fy, tx, ty: float): float =
|
||||||
## MAX lava on the straight-line segment (fx,fy) -> (tx,ty), sampled every
|
## MAX lava on the straight-line segment (fx,fy) -> (tx,ty), sampled every
|
||||||
@@ -899,12 +945,20 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
|||||||
while result > 180.0: result -= 360.0
|
while result > 180.0: result -= 360.0
|
||||||
while result < -180.0: result += 360.0
|
while result < -180.0: result += 360.0
|
||||||
|
|
||||||
|
# j145: the committed travel direction, needed both to score the turn cost of
|
||||||
|
# every candidate and to draw from it. Computed here (not in the pick block)
|
||||||
|
# because the candidates are scored before the pick.
|
||||||
|
let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0
|
||||||
|
else: ws.selfHeading
|
||||||
|
|
||||||
var scoredTiles: seq[ScoredTile]
|
var scoredTiles: seq[ScoredTile]
|
||||||
for t in coolTiles:
|
for t in coolTiles:
|
||||||
let tx = m.marginX + (t.col.float + 0.5) * GridSize
|
let tx = m.marginX + (t.col.float + 0.5) * GridSize
|
||||||
let ty = m.marginY + (t.row.float + 0.5) * GridSize
|
let ty = m.marginY + (t.row.float + 0.5) * GridSize
|
||||||
scoredTiles.add (col: t.col, row: t.row,
|
scoredTiles.add (col: t.col, row: t.row,
|
||||||
pathMaxHeat: pathMaxHeat(m, ws.selfX, ws.selfY, tx, ty))
|
pathMaxHeat: pathMaxHeat(m, ws.selfX, ws.selfY, tx, ty),
|
||||||
|
turnDeg: abs(tileOffTravel(m, t.col, t.row, ws.selfX,
|
||||||
|
ws.selfY, travelDeg)))
|
||||||
|
|
||||||
# Sort by pathMaxHeat ascending (insertion sort — small N)
|
# Sort by pathMaxHeat ascending (insertion sort — small N)
|
||||||
for i in 1..<scoredTiles.len:
|
for i in 1..<scoredTiles.len:
|
||||||
@@ -919,6 +973,8 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
|||||||
# Fallback: if everything is hot, keep the 2 coolest paths anyway.
|
# Fallback: if everything is hot, keep the 2 coolest paths anyway.
|
||||||
var safeTiles: seq[ScoredTile]
|
var safeTiles: seq[ScoredTile]
|
||||||
var blockedTiles: seq[ScoredTile]
|
var blockedTiles: seq[ScoredTile]
|
||||||
|
m.lastPickPromoted = false # j145: set only by the promotion below, so it is
|
||||||
|
# always the flag OF THE PICK THAT JUST HAPPENED
|
||||||
for t in scoredTiles:
|
for t in scoredTiles:
|
||||||
if t.pathMaxHeat <= PathDangerThreshold: safeTiles.add t
|
if t.pathMaxHeat <= PathDangerThreshold: safeTiles.add t
|
||||||
else: blockedTiles.add t
|
else: blockedTiles.add t
|
||||||
@@ -929,6 +985,7 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
|||||||
let promote = min(needed, blockedTiles.len)
|
let promote = min(needed, blockedTiles.len)
|
||||||
for i in 0..<promote:
|
for i in 0..<promote:
|
||||||
safeTiles.add blockedTiles[i]
|
safeTiles.add blockedTiles[i]
|
||||||
|
m.lastPickPromoted = true
|
||||||
blockedTiles = blockedTiles[promote ..< blockedTiles.len]
|
blockedTiles = blockedTiles[promote ..< blockedTiles.len]
|
||||||
|
|
||||||
# Commitment logic. With every j144 knob at its default (all off) this is the
|
# Commitment logic. With every j144 knob at its default (all off) this is the
|
||||||
@@ -996,8 +1053,6 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
|||||||
continue
|
continue
|
||||||
candidates.add t
|
candidates.add t
|
||||||
if candidates.len == 0: candidates = safeTiles # all blocked → ignore block
|
if candidates.len == 0: candidates = safeTiles # all blocked → ignore block
|
||||||
let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0
|
|
||||||
else: ws.selfHeading
|
|
||||||
# j144, no opposite-direction flip while still accelerating. Below the speed
|
# j144, no opposite-direction flip while still accelerating. Below the speed
|
||||||
# threshold the bot physically cannot complete a reversal before the bullet
|
# threshold the bot physically cannot complete a reversal before the bullet
|
||||||
# lands, so a mid-flight switch to the mirror side only destroys the dodge it
|
# lands, so a mid-flight switch to the mirror side only destroys the dodge it
|
||||||
@@ -1013,23 +1068,40 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
|||||||
for i in keep: narrowed.add candidates[i]
|
for i in keep: narrowed.add candidates[i]
|
||||||
candidates = narrowed
|
candidates = narrowed
|
||||||
var chosen = 0
|
var chosen = 0
|
||||||
if TfilNoRev and candidates.len >= 2:
|
if (TfilNoRev or TfilTurnBias > 0.0) and candidates.len >= 2:
|
||||||
# Soft no-reversal preference (arm C): down-weight — never filter — tiles
|
# Soft preferences — down-weight, never filter, and only ever among tiles
|
||||||
# that lie >90 deg from the current travel direction.
|
# that already passed the hard heat filter above:
|
||||||
|
# arm C (TR_TFIL_NO_REV, off by default) — 3:1 forward vs rearward,
|
||||||
|
# a BINARY cut: it cannot tell 20 deg from 90, nor 91 from 179.
|
||||||
|
# j145 (TR_TFIL_TURN_BIAS, off by default) — a CONTINUOUS turn cost,
|
||||||
|
# `turnWeights` = 1 + bias * (1 - excess/180).
|
||||||
|
# The two multiply, so turning one on never silently drops the other.
|
||||||
|
var weights = if TfilTurnBias > 0.0:
|
||||||
|
block:
|
||||||
|
var turns: seq[float]
|
||||||
|
for t in candidates: turns.add t.turnDeg
|
||||||
|
turnWeights(turns, TfilTurnBias, TfilTurnRefDeg)
|
||||||
|
else:
|
||||||
|
var w1 = newSeq[int](candidates.len)
|
||||||
|
for i in 0..<w1.len: w1[i] = 1
|
||||||
|
w1
|
||||||
|
if TfilNoRev:
|
||||||
var dirs: seq[float]
|
var dirs: seq[float]
|
||||||
for t in candidates:
|
for t in candidates:
|
||||||
let tx = m.marginX + (t.col.float + 0.5) * GridSize
|
let tx = m.marginX + (t.col.float + 0.5) * GridSize
|
||||||
let ty = m.marginY + (t.row.float + 0.5) * GridSize
|
let ty = m.marginY + (t.row.float + 0.5) * GridSize
|
||||||
dirs.add arctan2(ty - ws.selfY, tx - ws.selfX) * 180.0 / PI
|
dirs.add arctan2(ty - ws.selfY, tx - ws.selfX) * 180.0 / PI
|
||||||
let weights = noRevWeights(dirs, travelDeg)
|
let nw = noRevWeights(dirs, travelDeg)
|
||||||
|
for i in 0..<nw.len: weights[i] = weights[i] * nw[i]
|
||||||
var total = 0
|
var total = 0
|
||||||
var forward = 0
|
var wMax = 0
|
||||||
for w in weights:
|
for w in weights:
|
||||||
total += w
|
total += w
|
||||||
if w > 1: inc forward
|
wMax = max(wMax, w)
|
||||||
if forward == 0:
|
if wMax <= 1:
|
||||||
# Fallback: no forward tile exists → uniform draw, so the pool can
|
# Fallback: no candidate is preferred (every tile is equally bad, or all
|
||||||
# never empty and the pick is identical to the shipped one.
|
# are rearward) → uniform draw, so the pool can never empty and the pick
|
||||||
|
# degrades to exactly the shipped one.
|
||||||
chosen = rand(candidates.high)
|
chosen = rand(candidates.high)
|
||||||
else:
|
else:
|
||||||
let r = rand(total - 1)
|
let r = rand(total - 1)
|
||||||
@@ -1043,6 +1115,10 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
|||||||
else:
|
else:
|
||||||
chosen = rand(candidates.high)
|
chosen = rand(candidates.high)
|
||||||
let ct = candidates[chosen]
|
let ct = candidates[chosen]
|
||||||
|
pickedPromoted = m.lastPickPromoted
|
||||||
|
var minTurn = Inf
|
||||||
|
for t in candidates: minTurn = min(minTurn, t.turnDeg)
|
||||||
|
m.lastPickMinTurn = if minTurn == Inf: ct.turnDeg else: minTurn
|
||||||
m.commitTarget = (x: m.marginX + (ct.col.float + 0.5) * GridSize,
|
m.commitTarget = (x: m.marginX + (ct.col.float + 0.5) * GridSize,
|
||||||
y: m.marginY + (ct.row.float + 0.5) * GridSize)
|
y: m.marginY + (ct.row.float + 0.5) * GridSize)
|
||||||
m.commitTicks = TfilCommitTicks
|
m.commitTicks = TfilCommitTicks
|
||||||
@@ -1057,6 +1133,7 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
|||||||
while rd < -180.0: rd += 360.0
|
while rd < -180.0: rd += 360.0
|
||||||
pickedThisTick = true
|
pickedThisTick = true
|
||||||
pickedRev = abs(rd) > 90.0
|
pickedRev = abs(rd) > 90.0
|
||||||
|
pickedTurn = abs(rd)
|
||||||
pickedMidFlight = midFlight
|
pickedMidFlight = midFlight
|
||||||
pickedInterval = m.callCount - m.lastPickCall
|
pickedInterval = m.callCount - m.lastPickCall
|
||||||
m.lastPickCall = m.callCount
|
m.lastPickCall = m.callCount
|
||||||
@@ -1129,6 +1206,8 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand =
|
|||||||
tfilLogWrite("{\"tick\":" & $ws.tick & ",\"call\":" & $m.callCount &
|
tfilLogWrite("{\"tick\":" & $ws.tick & ",\"call\":" & $m.callCount &
|
||||||
",\"sp\":" & $ws.selfSpeed & ",\"ct\":" & $m.commitTicks &
|
",\"sp\":" & $ws.selfSpeed & ",\"ct\":" & $m.commitTicks &
|
||||||
",\"pick\":" & (if pickedThisTick: "1" else: "0") &
|
",\"pick\":" & (if pickedThisTick: "1" else: "0") &
|
||||||
|
",\"turn\":" & $pickedTurn &
|
||||||
|
",\"promote\":" & (if pickedPromoted: "1" else: "0") &
|
||||||
",\"reason\":\"" & reasonName(reason) & "\",\"rev\":" &
|
",\"reason\":\"" & reasonName(reason) & "\",\"rev\":" &
|
||||||
(if pickedRev: "1" else: "0") & ",\"mid\":" &
|
(if pickedRev: "1" else: "0") & ",\"mid\":" &
|
||||||
(if pickedMidFlight: "1" else: "0") & ",\"interval\":" & $pickedInterval &
|
(if pickedMidFlight: "1" else: "0") & ",\"interval\":" & $pickedInterval &
|
||||||
|
|||||||
@@ -217,3 +217,173 @@ echo "| arm | tile_self | tile_enemy | danger | expiry | arrival | hyst |"
|
|||||||
echo "|---|---:|---:|---:|---:|---:|---:|"
|
echo "|---|---:|---:|---:|---:|---:|---:|"
|
||||||
for s in rows:
|
for s in rows:
|
||||||
echo &"| {s.arm} | {s.byReason[rrTileSelf]} | {s.byReason[rrTileEnemy]} | {s.byReason[rrDanger]} | {s.byReason[rrExpiry]} | {s.byReason[rrArrival]} | {s.byReason[rrHyst]} |"
|
echo &"| {s.arm} | {s.byReason[rrTileSelf]} | {s.byReason[rrTileEnemy]} | {s.byReason[rrDanger]} | {s.byReason[rrExpiry]} | {s.byReason[rrArrival]} | {s.byReason[rrHyst]} |"
|
||||||
|
|
||||||
|
# ══════════════════════════════════════════════════════════════════════════════
|
||||||
|
# j145 — THE TURN-COST TIEBREAK, the mechanism ruler.
|
||||||
|
#
|
||||||
|
# Answers, on the SAME recorded fixture, whether `TR_TFIL_TURN_BIAS` does what
|
||||||
|
# it claims AMONG THE SAFE TILES and what it costs:
|
||||||
|
#
|
||||||
|
# mean |turn| — the |heading change| to the tile the picker chose
|
||||||
|
# mean regret — how many degrees worse than the SMALLEST-turn candidate
|
||||||
|
# actually available that choice was. The confound-free
|
||||||
|
# form: the arms draw from different candidate sets, so the
|
||||||
|
# raw mean alone can move without the mechanism biting.
|
||||||
|
# >90 / >135 — picks needing a real turn / a mirror-side pick
|
||||||
|
# path heat — mean max-lava on the straight path we were told to walk,
|
||||||
|
# and the share of picks over the hard threshold
|
||||||
|
# filter broken — the share of picks that had to promote a hot tile because
|
||||||
|
# FEWER THAN TWO tiles were safe (the shipped fallback)
|
||||||
|
# own-tile heat — mean lava under the bot's own wheels, every tick: THE
|
||||||
|
# SAFETY COST of the choice, measured where it is paid
|
||||||
|
# |turnRate| — the turn the bot actually EXECUTED, not just intended
|
||||||
|
# arrival — mean ticks to reach the committed tile, and the share of
|
||||||
|
# commitments that are actually reached
|
||||||
|
# ══════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
const ArriveR2 = 18.0
|
||||||
|
|
||||||
|
type TurnStats = object
|
||||||
|
arm: string
|
||||||
|
env: string
|
||||||
|
ticks: int
|
||||||
|
picks: int
|
||||||
|
turnSum: float
|
||||||
|
minTurnSum: float
|
||||||
|
tookMin: int
|
||||||
|
bigTurn: int
|
||||||
|
flip: int
|
||||||
|
pathHeatSum: float
|
||||||
|
hot: int
|
||||||
|
broken: int
|
||||||
|
ownHeatSum: float
|
||||||
|
ownHot: int
|
||||||
|
turnRateSum: float
|
||||||
|
hardTurn: int
|
||||||
|
holdSum: int
|
||||||
|
reached: int
|
||||||
|
distSum: float
|
||||||
|
|
||||||
|
proc pathHeatAt(m: TFILModule, fx, fy, tx, ty: float): float =
|
||||||
|
## The mover's own sampler (PathSampleStep 18). The lava field is rebuilt from
|
||||||
|
## scratch every computeMove, so what is visible just after the call is the
|
||||||
|
## field the pick was actually made against.
|
||||||
|
let ddx = tx - fx
|
||||||
|
let ddy = ty - fy
|
||||||
|
let lineDist = sqrt(ddx*ddx + ddy*ddy)
|
||||||
|
if lineDist <= 0.1: return 0.0
|
||||||
|
let steps = max(1, int(lineDist / 18.0))
|
||||||
|
var h = 0.0
|
||||||
|
for si in 0..steps:
|
||||||
|
let fr = si.float / steps.float
|
||||||
|
let (sc, sr) = m.tileAt(fx + ddx * fr, fy + ddy * fr)
|
||||||
|
h = max(h, m.lavaAt(sc, sr))
|
||||||
|
h
|
||||||
|
|
||||||
|
proc replayTurn(arm, envspec: string): TurnStats =
|
||||||
|
result.arm = arm
|
||||||
|
result.env = envspec
|
||||||
|
for tok in envspec.splitWhitespace():
|
||||||
|
putEnv(tok.split('=', 1)[0], tok.split('=', 1)[1])
|
||||||
|
putEnv("TR_TFIL_COMMIT_LOG", "")
|
||||||
|
loadTfilCommitEnv()
|
||||||
|
loadTfilHeatEnv()
|
||||||
|
randomize(Seed)
|
||||||
|
var m = initTFIL()
|
||||||
|
let states = loadStates()
|
||||||
|
let starts = loadRoundStarts()
|
||||||
|
var prev = (x: 0.0, y: 0.0)
|
||||||
|
var hadPicks = false
|
||||||
|
var held = 0
|
||||||
|
for i in 0..<states.len:
|
||||||
|
let ws = states[i]
|
||||||
|
if i == 0 or i in starts:
|
||||||
|
m.resetRound()
|
||||||
|
hadPicks = false
|
||||||
|
held = 0
|
||||||
|
let before = m.picks
|
||||||
|
let cmd = m.computeMove(ws)
|
||||||
|
inc result.ticks
|
||||||
|
result.turnRateSum += abs(cmd.turnRate)
|
||||||
|
if abs(cmd.turnRate) >= 5.0: inc result.hardTurn
|
||||||
|
let (bc, br) = m.tileAt(ws.selfX, ws.selfY)
|
||||||
|
let own = m.lavaAt(bc, br)
|
||||||
|
result.ownHeatSum += own
|
||||||
|
if own > 10.0: inc result.ownHot
|
||||||
|
if m.picks != before:
|
||||||
|
if hadPicks: result.holdSum += held
|
||||||
|
let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0
|
||||||
|
else: ws.selfHeading
|
||||||
|
var rd = arctan2(m.commitTarget.y - ws.selfY, m.commitTarget.x - ws.selfX) *
|
||||||
|
180.0 / PI - travelDeg
|
||||||
|
while rd > 180.0: rd -= 360.0
|
||||||
|
while rd < -180.0: rd += 360.0
|
||||||
|
inc result.picks
|
||||||
|
result.turnSum += abs(rd)
|
||||||
|
result.minTurnSum += m.lastPickMinTurn
|
||||||
|
if abs(rd) <= m.lastPickMinTurn + 0.5: inc result.tookMin
|
||||||
|
if abs(rd) > 90.0: inc result.bigTurn
|
||||||
|
if abs(rd) > 135.0: inc result.flip
|
||||||
|
let ph = pathHeatAt(m, ws.selfX, ws.selfY, m.commitTarget.x, m.commitTarget.y)
|
||||||
|
result.pathHeatSum += ph
|
||||||
|
if ph > 10.0: inc result.hot
|
||||||
|
if m.lastPickPromoted: inc result.broken
|
||||||
|
if hadPicks and
|
||||||
|
sqrt((ws.selfX-prev.x)^2 + (ws.selfY-prev.y)^2) < ArriveR2: inc result.reached
|
||||||
|
result.distSum += sqrt((m.commitTarget.x - ws.selfX)^2 +
|
||||||
|
(m.commitTarget.y - ws.selfY)^2)
|
||||||
|
prev = m.commitTarget
|
||||||
|
hadPicks = true
|
||||||
|
held = 0
|
||||||
|
elif hadPicks:
|
||||||
|
inc held
|
||||||
|
if hadPicks: result.holdSum += held
|
||||||
|
for tok in envspec.splitWhitespace():
|
||||||
|
putEnv(tok.split('=', 1)[0], "")
|
||||||
|
|
||||||
|
proc mean(x: float, d: int): float =
|
||||||
|
if d == 0: return 0.0
|
||||||
|
x / d.float
|
||||||
|
|
||||||
|
let turnArms = [
|
||||||
|
("shipped", "TR_MOVEMENT=tfil"),
|
||||||
|
("arrive+norev", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4"),
|
||||||
|
("turn b=3", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=3"),
|
||||||
|
("turn b=9", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=9"),
|
||||||
|
("turn b=19", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19"),
|
||||||
|
("turn b=39", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=39"),
|
||||||
|
("turn b=9 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0"),
|
||||||
|
("turn b=19 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19 TR_TFIL_TURN_REF_DEG=0"),
|
||||||
|
("turn b=39 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=39 TR_TFIL_TURN_REF_DEG=0"),
|
||||||
|
("turn b=19 ref90", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19 TR_TFIL_TURN_REF_DEG=90"),
|
||||||
|
("arrive+norev b=9", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9"),
|
||||||
|
("arrive+norev b=19", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=19"),
|
||||||
|
("arrive+norev b=9 r90", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=90"),
|
||||||
|
("arrive+norev b=9 ref0", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0"),
|
||||||
|
]
|
||||||
|
|
||||||
|
echo "\n\n══ j145 TURN-COST TIEBREAK — mechanism ruler ══\n"
|
||||||
|
var trows: seq[TurnStats]
|
||||||
|
for (name, envspec) in turnArms:
|
||||||
|
trows.add replayTurn(name, envspec)
|
||||||
|
|
||||||
|
echo &"| arm | picks | mean \\|turn\\| | mean regret | took min-turn | >90 deg | opposite (>135) | mean path heat | path heat >10 | filter broken |"
|
||||||
|
echo "|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|"
|
||||||
|
for s in trows:
|
||||||
|
echo &"| {s.arm} | {s.picks} | {mean(s.turnSum, s.picks).formatFloat(ffDecimal,1)} | " &
|
||||||
|
&"{mean(s.turnSum - s.minTurnSum, s.picks).formatFloat(ffDecimal,1)} deg | " &
|
||||||
|
&"{pct(s.tookMin, s.picks)} | {pct(s.bigTurn, s.picks)} | {pct(s.flip, s.picks)} | " &
|
||||||
|
&"{mean(s.pathHeatSum, s.picks).formatFloat(ffDecimal,2)} | {pct(s.hot, s.picks)} | {pct(s.broken, s.picks)} |"
|
||||||
|
|
||||||
|
echo "\n### what the choice costs: the turn EXECUTED, the arrival time, the distance\n"
|
||||||
|
echo "### (the heat under the bot's own wheels is NOT here: this replay drives the"
|
||||||
|
echo "### RECORDED self states, so the travelled path is identical in every arm by"
|
||||||
|
echo "### construction — the pick varies, the trajectory does not. The safety cost"
|
||||||
|
echo "### of a bias that arrives later / dwells hotter is therefore NOT measurable"
|
||||||
|
echo "### offline here; `mean path heat` and `filter broken` are the honest proxies.)\n"
|
||||||
|
echo &"| arm | mean \\|turnRate\\| executed | hard turns (>=5 deg/tick) | mean arrival (ticks) | reached | mean distance to the pick |"
|
||||||
|
echo "|---|---:|---:|---:|---:|---:|"
|
||||||
|
for s in trows:
|
||||||
|
echo &"| {s.arm} | {mean(s.turnRateSum, s.ticks).formatFloat(ffDecimal,2)} | {pct(s.hardTurn, s.ticks)} | " &
|
||||||
|
&"{mean(s.holdSum.float, s.picks).formatFloat(ffDecimal,1)} | {pct(s.reached, s.picks)} | " &
|
||||||
|
&"{mean(s.distSum, s.picks).formatFloat(ffDecimal,0)} px |"
|
||||||
|
|||||||
@@ -6,6 +6,8 @@
|
|||||||
## TR_TFIL_COMMIT_ARRIVAL (0/1) default 0 = shipped (j144)
|
## TR_TFIL_COMMIT_ARRIVAL (0/1) default 0 = shipped (j144)
|
||||||
## TR_TFIL_COMMIT_MARGIN (float) default 0.0 = shipped (j144)
|
## TR_TFIL_COMMIT_MARGIN (float) default 0.0 = shipped (j144)
|
||||||
## TR_TFIL_NOREV_SPEED (float) default 0.0 = shipped (j144)
|
## TR_TFIL_NOREV_SPEED (float) default 0.0 = shipped (j144)
|
||||||
|
## TR_TFIL_TURN_BIAS (float) default 0.0 = shipped (j145)
|
||||||
|
## TR_TFIL_TURN_REF_DEG (float) default 45.0 (j145)
|
||||||
##
|
##
|
||||||
## NO battle, NO Java, NO server. Run with:
|
## NO battle, NO Java, NO server. Run with:
|
||||||
## nim c -r --path:common_libs common_libs/tests/test_tfil_commit_env.nim
|
## nim c -r --path:common_libs common_libs/tests/test_tfil_commit_env.nim
|
||||||
@@ -445,6 +447,216 @@ when declared(loadTfilCommitEnv):
|
|||||||
" hyst=", s.byReason[rrHyst],
|
" hyst=", s.byReason[rrHyst],
|
||||||
" danger=", s.byReason[rrDanger]
|
" danger=", s.byReason[rrDanger]
|
||||||
|
|
||||||
|
# ── 5. j145: the turn-cost TIEBREAK among SAFE tiles (default OFF) ─────────
|
||||||
|
|
||||||
|
type PickRec = object
|
||||||
|
turn: float ## |heading change| from the travel direction to the pick
|
||||||
|
minTurn: float ## the smallest |turn| AVAILABLE in that candidate set —
|
||||||
|
## turn - minTurn is the regret of the draw, which is
|
||||||
|
## the confound-free form of the mechanism metric
|
||||||
|
pathHeat: float ## max lava on the straight path the bot was told to walk
|
||||||
|
promoted: bool ## the pick had to break the hard heat filter
|
||||||
|
reached: bool ## the commitment ended with the bot on the tile
|
||||||
|
|
||||||
|
const DangerThreshold = 10.0 ## PathDangerThreshold inside the mover
|
||||||
|
|
||||||
|
proc probePathHeat(m: TFILModule, fx, fy, tx, ty: float): float =
|
||||||
|
## Mirrors the mover's own sampler (PathSampleStep = 18, ~half a tile). The
|
||||||
|
## field is rebuilt from scratch every computeMove, so the `m.lava` visible
|
||||||
|
## just after the call is exactly the field the pick was made against.
|
||||||
|
let ddx = tx - fx
|
||||||
|
let ddy = ty - fy
|
||||||
|
let lineDist = sqrt(ddx*ddx + ddy*ddy)
|
||||||
|
if lineDist <= 0.1: return 0.0
|
||||||
|
let steps = max(1, int(lineDist / 18.0))
|
||||||
|
var h = 0.0
|
||||||
|
for si in 0..steps:
|
||||||
|
let f = si.float / steps.float
|
||||||
|
let (sc, sr) = m.tileAt(fx + ddx * f, fy + ddy * f)
|
||||||
|
h = max(h, m.lavaAt(sc, sr))
|
||||||
|
h
|
||||||
|
|
||||||
|
proc replayJ145(tag: string, bias, refDeg: float,
|
||||||
|
arrive: bool, norevSpeed: float): seq[PickRec] =
|
||||||
|
## Drive the REAL computeMove with the j145 knobs set through the env (and
|
||||||
|
## the j144 knobs forced through the vars, so no `.env` can be in the way),
|
||||||
|
## and record what each pick actually cost.
|
||||||
|
putEnv("TR_TFIL_TURN_BIAS", $bias)
|
||||||
|
putEnv("TR_TFIL_TURN_REF_DEG", $refDeg)
|
||||||
|
putEnv("TR_TFIL_COMMIT_LOG", "")
|
||||||
|
loadTfilCommitEnv()
|
||||||
|
TfilCommitArrival = arrive
|
||||||
|
TfilCommitMargin = 0.0
|
||||||
|
TfilNoRevSpeed = norevSpeed
|
||||||
|
randomize(Seed)
|
||||||
|
var m = initTFIL()
|
||||||
|
let states = loadStates()
|
||||||
|
let starts = loadRoundStarts()
|
||||||
|
var prev = (x: 0.0, y: 0.0)
|
||||||
|
var hadPicks = false
|
||||||
|
for i in 0..<states.len:
|
||||||
|
let ws = states[i]
|
||||||
|
if i == 0 or i in starts:
|
||||||
|
m.resetRound()
|
||||||
|
hadPicks = false
|
||||||
|
let before = m.picks
|
||||||
|
discard m.computeMove(ws)
|
||||||
|
if m.picks != before:
|
||||||
|
let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0
|
||||||
|
else: ws.selfHeading
|
||||||
|
var rd = arctan2(m.commitTarget.y - ws.selfY, m.commitTarget.x - ws.selfX) *
|
||||||
|
180.0 / PI - travelDeg
|
||||||
|
while rd > 180.0: rd -= 360.0
|
||||||
|
while rd < -180.0: rd += 360.0
|
||||||
|
result.add PickRec(turn: abs(rd), minTurn: m.lastPickMinTurn,
|
||||||
|
pathHeat: probePathHeat(m, ws.selfX, ws.selfY,
|
||||||
|
m.commitTarget.x, m.commitTarget.y),
|
||||||
|
promoted: m.lastPickPromoted,
|
||||||
|
reached: hadPicks and
|
||||||
|
sqrt((ws.selfX-prev.x)^2 + (ws.selfY-prev.y)^2) < 18.0)
|
||||||
|
prev = m.commitTarget
|
||||||
|
hadPicks = true
|
||||||
|
putEnv("TR_TFIL_TURN_BIAS", "")
|
||||||
|
putEnv("TR_TFIL_TURN_REF_DEG", "")
|
||||||
|
loadTfilCommitEnv()
|
||||||
|
TfilCommitArrival = false
|
||||||
|
TfilNoRevSpeed = 0.0
|
||||||
|
discard tag
|
||||||
|
|
||||||
|
type TurnStats = object
|
||||||
|
picks: int
|
||||||
|
turnSum: float
|
||||||
|
minTurnSum: float
|
||||||
|
tookMin: int ## the draw landed on the smallest-turn candidate
|
||||||
|
bigTurn: int ## |turn| > 90 deg
|
||||||
|
flip: int ## |turn| > 135 deg — the opposite side
|
||||||
|
heatSum: float ## mean path heat of the tile we actually walked to
|
||||||
|
hotPicks: int ## ... over the hard threshold
|
||||||
|
badHot: int ## ... over the threshold WITHOUT the filter being
|
||||||
|
## broken — MUST be 0 at any turn bias
|
||||||
|
broken: int ## picks that had to promote a hot tile (fewer than
|
||||||
|
## two safe tiles existed) — the shipped fallback
|
||||||
|
reached: int
|
||||||
|
|
||||||
|
proc turnStats(p: seq[PickRec]): TurnStats =
|
||||||
|
result.picks = p.len
|
||||||
|
for r in p:
|
||||||
|
result.turnSum += r.turn
|
||||||
|
result.minTurnSum += r.minTurn
|
||||||
|
if r.turn <= r.minTurn + 0.5: inc result.tookMin
|
||||||
|
if r.turn > 90.0: inc result.bigTurn
|
||||||
|
if r.turn > 135.0: inc result.flip
|
||||||
|
result.heatSum += r.pathHeat
|
||||||
|
if r.pathHeat > DangerThreshold:
|
||||||
|
inc result.hotPicks
|
||||||
|
if not r.promoted: inc result.badHot
|
||||||
|
if r.promoted: inc result.broken
|
||||||
|
if r.reached: inc result.reached
|
||||||
|
|
||||||
|
proc meanTurn(s: TurnStats): float =
|
||||||
|
if s.picks == 0: return 0.0
|
||||||
|
s.turnSum / s.picks.float
|
||||||
|
proc meanRegret(s: TurnStats): float =
|
||||||
|
## How many degrees WORSE than the best available tile the draw actually was.
|
||||||
|
## Immune to the composition confound that the raw mean |turn| has (a bias
|
||||||
|
## arm makes different picks, so the two arms' candidate sets differ).
|
||||||
|
if s.picks == 0: return 0.0
|
||||||
|
(s.turnSum - s.minTurnSum) / s.picks.float
|
||||||
|
proc meanPathHeat(s: TurnStats): float =
|
||||||
|
if s.picks == 0: return 0.0
|
||||||
|
s.heatSum / s.picks.float
|
||||||
|
proc pct(n, d: int): string =
|
||||||
|
if d == 0: return "-"
|
||||||
|
(100.0 * n.float / d.float).formatFloat(ffDecimal, 1) & "%"
|
||||||
|
|
||||||
|
proc testJ145() =
|
||||||
|
# 5a. the shipped default is OFF — the parity check above is the proof
|
||||||
|
check "j145: the turn bias defaults to today's uniform draw (bias 0)",
|
||||||
|
TfilTurnBias == 0.0 and TfilTurnRefDeg == 45.0
|
||||||
|
|
||||||
|
# 5b. the weighting, in pure form
|
||||||
|
check "j145: bias 0 gives every safe tile weight 1 (byte-identical to the " &
|
||||||
|
"shipped uniform draw)", turnWeights(@[0.0, 91.0, 180.0], 0.0, 45.0) ==
|
||||||
|
@[1, 1, 1]
|
||||||
|
check "j145: the penalty is CONTINUOUS past the reference angle, where the " &
|
||||||
|
"binary TR_TFIL_NO_REV cannot see (bias 9, ref 45: 45/90/135/180 deg " &
|
||||||
|
"-> 10/8/6/3)", turnWeights(@[45.0, 90.0, 135.0, 180.0], 9.0, 45.0) ==
|
||||||
|
@[10, 8, 6, 3]
|
||||||
|
check "j145: a turn inside the reference angle is never penalised",
|
||||||
|
turnWeights(@[0.0, 20.0, 45.0], 5.0, 45.0) == @[6, 6, 6]
|
||||||
|
check "j145: the weight falls monotonically with the turn (ref 0, bias 9: " &
|
||||||
|
"0/30/60/90/120/180 deg -> 10/9/8/8/7/1)",
|
||||||
|
turnWeights(@[0.0, 30.0, 60.0, 90.0, 120.0, 180.0], 9.0, 0.0) ==
|
||||||
|
@[10, 9, 7, 6, 4, 1]
|
||||||
|
check "j145: the weight is floored at 1, so the pool can never be starved",
|
||||||
|
turnWeights(@[0.0, 180.0, 179.0], 9.0, 0.0)[1] >= 1 and
|
||||||
|
turnWeights(@[0.0, 180.0, 179.0], 9.0, 0.0) == @[10, 1, 1]
|
||||||
|
check "j145: `bias` IS the odds ratio — with ref 0 a straight-ahead safe tile " &
|
||||||
|
"is drawn 1+bias times as often as a 180 deg one (9 -> 10:1)",
|
||||||
|
turnWeights(@[0.0, 180.0], 9.0, 0.0) == @[10, 1]
|
||||||
|
|
||||||
|
# 5c. THE GATE: an absurd turn cost must not rescue a hot tile. The heat
|
||||||
|
# filter is UPSTREAM of the weighting, so an over-threshold pick can
|
||||||
|
# only ever be one the mover had to promote because nothing was safe.
|
||||||
|
let wild = turnStats(replayJ145("wild", 99.0, 45.0, true, 4.0))
|
||||||
|
check "j145: with an absurd turn bias (" & $wild.picks & " picks) NO tile " &
|
||||||
|
"over the heat threshold is ever chosen unless the filter had to be " &
|
||||||
|
"broken (" & $wild.badHot & " violations)",
|
||||||
|
wild.badHot == 0
|
||||||
|
check "j145: the over-threshold picks that do happen are only the promoted " &
|
||||||
|
"ones (" & $wild.hotPicks & "/" & $wild.picks & ", the shipped " &
|
||||||
|
"fewer-than-2-safe-tiles fallback)", wild.badHot == 0
|
||||||
|
|
||||||
|
# 5d. the mechanism: the turn really gets smaller, without paying for it in
|
||||||
|
# heat, and without emptying the pool
|
||||||
|
let off = turnStats(replayJ145("off", 0.0, 45.0, true, 4.0))
|
||||||
|
let mild = turnStats(replayJ145("mild", 9.0, 0.0, true, 4.0))
|
||||||
|
let firm = turnStats(replayJ145("firm", 39.0, 0.0, true, 4.0))
|
||||||
|
check "j145: with the bias on, the mean |turn| to the chosen tile falls " &
|
||||||
|
"(" & meanTurn(off).formatFloat(ffDecimal, 1) & " -> " &
|
||||||
|
meanTurn(mild).formatFloat(ffDecimal, 1) & " -> " &
|
||||||
|
meanTurn(firm).formatFloat(ffDecimal, 1) & " deg)",
|
||||||
|
meanTurn(mild) < meanTurn(off) * 0.95 and
|
||||||
|
meanTurn(firm) < meanTurn(off) * 0.95
|
||||||
|
check "j145: the REGRET of the draw (how many degrees worse than the best " &
|
||||||
|
"AVAILABLE candidate) falls " &
|
||||||
|
"(" & meanRegret(off).formatFloat(ffDecimal, 1) & " -> " &
|
||||||
|
meanRegret(mild).formatFloat(ffDecimal, 1) & " -> " &
|
||||||
|
meanRegret(firm).formatFloat(ffDecimal, 1) & " deg) — the " &
|
||||||
|
"confound-free form of the mechanism",
|
||||||
|
meanRegret(mild) < meanRegret(off) * 0.9 and
|
||||||
|
meanRegret(firm) < meanRegret(off) * 0.9
|
||||||
|
check "j145: the share of picks needing >90 deg of turn falls " &
|
||||||
|
"(" & pct(off.bigTurn, off.picks) & " -> " & pct(mild.bigTurn, mild.picks) &
|
||||||
|
" -> " & pct(firm.bigTurn, firm.picks) & ")",
|
||||||
|
mild.bigTurn < off.bigTurn
|
||||||
|
check "j145: a mirror-image tile no longer beats a straight-ahead one as " &
|
||||||
|
"readily — opposite-side picks fall " & pct(off.flip, off.picks) & " -> " &
|
||||||
|
pct(mild.flip, mild.picks) & " -> " & pct(firm.flip, firm.picks),
|
||||||
|
mild.flip.float < off.flip.float * 0.95
|
||||||
|
check "j145: SAFETY COST — the mean path heat of the chosen tile does not " &
|
||||||
|
"rise (bias off " & meanPathHeat(off).formatFloat(ffDecimal, 2) &
|
||||||
|
" vs bias 9 " & meanPathHeat(mild).formatFloat(ffDecimal, 2) &
|
||||||
|
" vs bias 39 " & meanPathHeat(firm).formatFloat(ffDecimal, 2) & ")",
|
||||||
|
meanPathHeat(mild) <= meanPathHeat(off) * 1.05 and
|
||||||
|
meanPathHeat(firm) <= meanPathHeat(off) * 1.05
|
||||||
|
check "j145: the bias never empties the pool — decisions stay within 5% of " &
|
||||||
|
"the same arm without it (" & $off.picks & " -> " & $mild.picks & " / " &
|
||||||
|
$firm.picks & ")",
|
||||||
|
abs(firm.picks.float - off.picks.float) <= 0.05 * off.picks.float
|
||||||
|
|
||||||
|
echo "\n j145 diagnostics (offline fixture replay, arrive+norev base):"
|
||||||
|
for (nm, s) in [("bias 0 (shipped)", off), ("bias 9 ref0", mild), ("bias 39 ref0", firm)]:
|
||||||
|
echo " ", nm.alignLeft(18), " picks=", s.picks,
|
||||||
|
" mean|turn|=", meanTurn(s).formatFloat(ffDecimal, 1),
|
||||||
|
" regret=", meanRegret(s).formatFloat(ffDecimal, 1),
|
||||||
|
" tookMin=", pct(s.tookMin, s.picks),
|
||||||
|
" >90deg=", pct(s.bigTurn, s.picks),
|
||||||
|
" >135deg=", pct(s.flip, s.picks),
|
||||||
|
" filter broken=", pct(s.broken, s.picks),
|
||||||
|
" mean path heat=", meanPathHeat(s).formatFloat(ffDecimal, 2),
|
||||||
|
" reached=", pct(s.reached, s.picks)
|
||||||
|
|
||||||
# ── driver ───────────────────────────────────────────────────────────────────
|
# ── driver ───────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
testDefaultParity()
|
testDefaultParity()
|
||||||
@@ -452,6 +664,7 @@ when declared(loadTfilCommitEnv):
|
|||||||
testKnobParsing()
|
testKnobParsing()
|
||||||
testArms()
|
testArms()
|
||||||
testJ144()
|
testJ144()
|
||||||
|
testJ145()
|
||||||
|
|
||||||
if failures > 0:
|
if failures > 0:
|
||||||
echo "\n", failures, " check(s) FAILED"
|
echo "\n", failures, " check(s) FAILED"
|
||||||
|
|||||||
@@ -2978,3 +2978,77 @@ He should expect the dodge to look *smoother and more deliberate* (fewer, longer
|
|||||||
commitments) rather than twitchy, and he should see fewer bullets connect. He
|
commitments) rather than twitchy, and he should see fewer bullets connect. He
|
||||||
should NOT expect a step change in his score from this alone: the measured
|
should NOT expect a step change in his score from this alone: the measured
|
||||||
outcome effect is +0.28 wins/run with a p of 0.057 on the primary test.
|
outcome effect is +0.28 wins/run with a p of 0.057 on the primary test.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# Batch 6 — TFIL turn-cost tiebreak (j145)
|
||||||
|
|
||||||
|
*Pre-registered BEFORE any battle of this batch was launched. No battle of this
|
||||||
|
batch existed when this section was written; the frozen binary for it is the
|
||||||
|
commit that adds the tiebreak.*
|
||||||
|
|
||||||
|
## The cause this batch fixes
|
||||||
|
|
||||||
|
The `tfil` picker's `ScoredTile` carried **one** term, `pathMaxHeat`. After the
|
||||||
|
hard filter (`pathMaxHeat <= PathDangerThreshold` = 10) the pick was a plain
|
||||||
|
`rand()` over the survivors, so a far-cooler tile on the OPPOSITE side was drawn
|
||||||
|
exactly as readily as a marginally-cooler one straight ahead. The only
|
||||||
|
heading-aware influence in the mover is `NoRevForwardWeight = 3` under
|
||||||
|
`TR_TFIL_NO_REV` — **binary** (it cannot tell 20 deg from 90, nor 91 from 179)
|
||||||
|
and **off by default**. This batch adds the continuous version.
|
||||||
|
|
||||||
|
## The treatment
|
||||||
|
|
||||||
|
Two knobs, both **off by default** (the shipped default path is byte-for-byte
|
||||||
|
identical — the golden in `test_tfil_commit_env.nim` still passes):
|
||||||
|
|
||||||
|
| knob | default | meaning |
|
||||||
|
|---|---|---|
|
||||||
|
| `TR_TFIL_TURN_BIAS` | `0.0` | the tiebreak's **odds ratio**: a straight-ahead safe tile is drawn `1 + bias` times as often as a 180 deg one |
|
||||||
|
| `TR_TFIL_TURN_REF_DEG` | `45.0` | the turn below which no penalty applies |
|
||||||
|
|
||||||
|
The draw weight of a safe candidate is
|
||||||
|
|
||||||
|
w = max(1, round(1 + bias * (1 - max(0, |turn| - refDeg) / 180)))
|
||||||
|
|
||||||
|
**The safety filter is untouched and stays hard.** Turn cost is never added to
|
||||||
|
the heat score (`heat + k*turnDeg` would trade dodging for smoothness, which is
|
||||||
|
backwards in a bullet-dodging game); the bias is applied *only* to the
|
||||||
|
weight of a draw *among tiles that already passed the filter*. Guard:
|
||||||
|
`test_tfil_commit_env.nim` runs an absurd bias (99:1) and asserts that **no**
|
||||||
|
over-threshold tile is ever chosen unless the mover's own "fewer than two tiles
|
||||||
|
are safe" fallback promoted it.
|
||||||
|
|
||||||
|
**Randomness is preserved.** Job j51 (`3142b70`) measured that randomness in
|
||||||
|
this tie is load-bearing for this bot — a deterministic argmin scored worse.
|
||||||
|
The pick is therefore a **weighted draw**, not an argmin; every weight is
|
||||||
|
floored at 1 so the pool can never be emptied, and at bias 0 every weight is 1,
|
||||||
|
i.e. exactly the shipped uniform draw.
|
||||||
|
|
||||||
|
## Arms (frozen, all `TR_MOVEMENT=tfil`)
|
||||||
|
|
||||||
|
1. `tfil_shipped` — stock defaults. **The reference.**
|
||||||
|
2. `arrive_norev` — the two knobs j144 recommends (`COMMIT_ARRIVAL=1`,
|
||||||
|
`NOREV_SPEED=4`, `MARGIN=0`).
|
||||||
|
3. `arrive_norev_turn` — arm 2 + `TURN_BIAS=9 TURN_REF_DEG=0`.
|
||||||
|
4. `turn_only` — `TURN_BIAS=9 TURN_REF_DEG=0` alone; isolates the turn fix.
|
||||||
|
|
||||||
|
Panel: the FROZEN 15-opponent `tools/ab/panel_movement.txt`. Harness:
|
||||||
|
`tools/ab/tournament_run.sh` + `tournament_analyze.py`.
|
||||||
|
|
||||||
|
## Pre-registered prediction and decision rule
|
||||||
|
|
||||||
|
* **Prediction.** Arm 3 > arm 2 > arm 1 on damage/run and round wins, because a
|
||||||
|
smaller commanded turn is a faster arrival and a shorter exposure. Arm 4 sits
|
||||||
|
between arm 1 and arm 3. The **incoming hit rate is the mechanism, not the
|
||||||
|
verdict** — the verdict is damage/run and round wins under the campaign's
|
||||||
|
pre-registered rule 2 (cross-opponent sign test p < 0.05 on one primary metric
|
||||||
|
with the other not down), with the SD/SE/95% CI/MDE reported alongside.
|
||||||
|
* **If nothing separates**, the verdict is *not distinguishable* and it is
|
||||||
|
**not shipped**. The pre-registered bar is not re-interpreted afterwards.
|
||||||
|
* **A null here does NOT undo j144.** j144's result is a *mechanism* result
|
||||||
|
(the incoming hit rate fell 18.07% → 14.92%, sign-flip p = 0.0013, in two
|
||||||
|
independent blocks) plus an under-powered outcome null. This batch can only
|
||||||
|
add to or fail to add to that; it cannot retract it.
|
||||||
|
|
||||||
|
*(results appended below after the battles)*
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
# ─────────────────────────────────────────────────────────────────────────────
|
||||||
|
# arms_tfil_turn.txt — j145: the TFIL TURN-COST TIEBREAK, four arms on the
|
||||||
|
# frozen movement panel (tools/ab/panel_movement.txt), all with TR_MOVEMENT=tfil
|
||||||
|
# pinned EXPLICITLY.
|
||||||
|
#
|
||||||
|
# WHAT THE KNOB IS. Among the tiles that already passed the HARD heat filter
|
||||||
|
# (pathMaxHeat <= PathDangerThreshold), the draw is weighted by
|
||||||
|
#
|
||||||
|
# w = max(1, round(1 + TR_TFIL_TURN_BIAS * (1 - max(0,|turn| - REF) / 180)))
|
||||||
|
#
|
||||||
|
# so a straight-ahead safe tile is drawn `1 + bias` times as often as a 180 deg
|
||||||
|
# one. The heat score is NOT touched: a tile over the threshold still loses, at
|
||||||
|
# any bias (guard: test_tfil_commit_env.nim, "j145: with an absurd turn bias...").
|
||||||
|
# The draw stays RANDOM (job j51, 3142b70: a deterministic argmin measured worse).
|
||||||
|
#
|
||||||
|
# Pre-registered in docs/movement_campaign.md ("TFIL turn-cost tiebreak") BEFORE
|
||||||
|
# any of these battles ran. Reference is `tfil_shipped`.
|
||||||
|
#
|
||||||
|
# Format: name | ENV=value ENV=value | label
|
||||||
|
# ─────────────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
# 1. THE ARM TO BEAT — the shipped tfil defaults, explicitly selected.
|
||||||
|
tfil_shipped | TR_MOVEMENT=tfil | shipped tfil defaults (control / reference)
|
||||||
|
|
||||||
|
# 2. the two knobs job j144 recommends, unchanged.
|
||||||
|
arrive_norev | TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 | j144's recommended .env
|
||||||
|
|
||||||
|
# 3. j144's knobs PLUS the turn tiebreak, at the value the offline ruler picked
|
||||||
|
# (the knee of the bias curve: bias 9 ref 0 cuts mean |turn| 11% and opposite
|
||||||
|
# picks 22% with no path-heat cost; bias 39 buys almost nothing more).
|
||||||
|
arrive_norev_turn | TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0 | j144 + the turn tiebreak
|
||||||
|
|
||||||
|
# 4. THE TURN FIX ALONE — isolates the tiebreak with no j144 knob set, which is
|
||||||
|
# the only arm that can say whether it is worth anything by itself.
|
||||||
|
turn_only | TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0 | the turn tiebreak alone
|
||||||
Reference in New Issue
Block a user