diff --git a/ModularBot_garage/.env.example b/ModularBot_garage/.env.example index d682583..64ef797 100644 --- a/ModularBot_garage/.env.example +++ b/ModularBot_garage/.env.example @@ -114,6 +114,8 @@ TR_TFIL_COMMIT_LOG= # path for the per-commit log; empty = no log TR_TFIL_COMMIT_ARRIVAL=off # on = hold the dodge tile until we are ON it (not a fixed dwell) TR_TFIL_COMMIT_MARGIN=0.0 # lava an alternative tile must be cooler by before it wins the tile TR_TFIL_NOREV_SPEED=0.0 # px/tick; below this, a mid-flight switch may not turn the bot around +TR_TFIL_TURN_BIAS=0.0 # turn TIEBREAK odds ratio among SAFE tiles; 0 = uniform draw as today +TR_TFIL_TURN_REF_DEG=45.0 # deg; turn below which the tiebreak applies no penalty TR_TFIL_HEAT_TIME=off # on = index bullet heat by time (flat field when off) TR_TFIL_HEAT_TAU=9.0 # ticks a tracked bullet's heat lives for TR_TFIL_HEAT_POWER_GAIN=1.0 # scale of the heat a bullet paints, per firepower diff --git a/ModularBot_garage/src/env_report.nim b/ModularBot_garage/src/env_report.nim index 6fc5338..4bfa2ca 100644 --- a/ModularBot_garage/src/env_report.nim +++ b/ModularBot_garage/src/env_report.nim @@ -324,6 +324,8 @@ proc printEffectiveValues(ctx: EnvReportContext) = emit("TR_TFIL_COMMIT_MARGIN", $TfilCommitMargin, sourceOf("TR_TFIL_COMMIT_MARGIN")) emit("TR_TFIL_NOREV_SPEED", $TfilNoRevSpeed, sourceOf("TR_TFIL_NOREV_SPEED")) + emit("TR_TFIL_TURN_BIAS", $TfilTurnBias, sourceOf("TR_TFIL_TURN_BIAS")) + emit("TR_TFIL_TURN_REF_DEG", $TfilTurnRefDeg, sourceOf("TR_TFIL_TURN_REF_DEG")) # time-indexed bullet heat (default off = shipped flat model) emit("TR_TFIL_HEAT_TIME", onOff(TfilHeatTime), sourceOf("TR_TFIL_HEAT_TIME")) emit("TR_TFIL_HEAT_TAU", $TfilHeatTau, sourceOf("TR_TFIL_HEAT_TAU")) @@ -639,7 +641,7 @@ proc knownEnvNames*(): seq[string] = "TR_TFIL_WALL_RADIANCE", "TR_TFIL_TILE_REPLAN", "TR_TFIL_COMMIT_TICKS", "TR_TFIL_NO_REV", "TR_TFIL_COMMIT_LOG", "TR_TFIL_COMMIT_ARRIVAL", "TR_TFIL_COMMIT_MARGIN", - "TR_TFIL_NOREV_SPEED", + "TR_TFIL_NOREV_SPEED", "TR_TFIL_TURN_BIAS", "TR_TFIL_TURN_REF_DEG", "TR_TFIL_HEAT_TIME", "TR_TFIL_HEAT_TAU", "TR_TFIL_HEAT_POWER_GAIN", "TR_TFIL_PILLAR_ON", "TR_STRAFE_BAND", "TR_STRAFE_SPREAD", "TR_STRAFE_REACH", diff --git a/common_libs/movements/the_floor_is_lava.nim b/common_libs/movements/the_floor_is_lava.nim index dbcf007..9174707 100644 --- a/common_libs/movements/the_floor_is_lava.nim +++ b/common_libs/movements/the_floor_is_lava.nim @@ -128,6 +128,16 @@ var TfilCommitArrival*: bool = false TfilCommitMargin*: float = 0.0 TfilNoRevSpeed*: float = 0.0 + ## j145 — the turn-cost TIEBREAK, applied ONLY among tiles that already passed + ## the hard heat filter. It is a WEIGHT on the draw, never a term in the heat + ## score (see `turnWeights`). + ## TR_TFIL_TURN_BIAS float the turn TIEBREAK's odds ratio: a + ## straight-ahead safe tile is drawn + ## `1 + bias` times as often as a 180 deg one + ## (0 = off, today's uniform draw exactly) + ## TR_TFIL_TURN_REF_DEG float turn below which there is no penalty (deg) + TfilTurnBias*: float = 0.0 + TfilTurnRefDeg*: float = 45.0 ## j134: the shared fire-detection correction (`TR_FIRE_FIX`, default on). ## Off = the shipped `prev - energy` detector byte-for-byte. TfilFireFix*: bool = true @@ -162,6 +172,8 @@ proc loadTfilCommitEnv*() = TfilCommitArrival = getEnvBool("TR_TFIL_COMMIT_ARRIVAL", false) TfilCommitMargin = max(0.0, getEnvFloat("TR_TFIL_COMMIT_MARGIN", 0.0)) TfilNoRevSpeed = max(0.0, getEnvFloat("TR_TFIL_NOREV_SPEED", 0.0)) + TfilTurnBias = max(0.0, getEnvFloat("TR_TFIL_TURN_BIAS", 0.0)) + TfilTurnRefDeg = max(0.0, getEnvFloat("TR_TFIL_TURN_REF_DEG", 45.0)) TfilFireFix = getEnvBool("TR_FIRE_FIX", true) loadTfilCommitEnv() @@ -289,6 +301,12 @@ type replanReason: TfilReplanReason ## why the last commitment ended (log only) lastPickCall: int ## callCount at the last pick (log only) picks: int ## number of picks this round (log only) + lastPickPromoted: bool ## the last pick had to break the hard + ## heat filter (fewer than 2 safe tiles) + lastPickMinTurn: float ## smallest |turn| available in the + ## candidate set of the last pick (j145: + ## lets a caller measure the REGRET of the + ## draw instead of only the drawn value) proc initTFIL*(): TFILModule = TFILModule(debugGraphics: false, fire: initFireTracker()) @@ -332,6 +350,8 @@ proc resetRound*(m: var TFILModule) = m.replanReason = rrNone m.lastPickCall = 0 m.picks = 0 + m.lastPickPromoted = false + m.lastPickMinTurn = 0.0 # ── Commit diagnostics (TR_TFIL_COMMIT_LOG, off by default) ────────────────── # One JSONL line per computeMove call, used by the A/B to prove the treatment @@ -569,6 +589,27 @@ proc norevPool*(offs: openArray[float], threshold: float): seq[int] = if abs(a) < abs(offs[best]): best = i @[best] +proc turnWeights*(turns: openArray[float], bias, refDeg: float): seq[int] = + ## j145: the turn-cost TIEBREAK, as an integer draw weight per candidate: + ## + ## w = max(1, round(1 + bias * (1 - max(0, |turn| - refDeg) / 180))) + ## + ## i.e. a straight-ahead (within `refDeg`) safe tile is drawn `1 + bias` times + ## as often as a 180 deg one, falling LINEARLY in between. Read `bias` as that + ## odds ratio: bias 0 = today's uniform draw, bias 9 = 10:1. + ## Only tiles that already passed the hard heat filter reach this function, so + ## it can never rescue a hot one. + ## WHY A WEIGHT AND NOT `heat + k*turnDeg`: mixing the two trades dodging for + ## smoothness, which is backwards in a bullet-dodging game — the safety filter + ## stays hard and the turn only re-orders the survivors. + ## WHY NOT AN ARGMIN: job j51 (3142b70) measured that randomness in this tie + ## is LOAD-BEARING for this bot — a deterministic argmin scored worse. So the + ## draw stays random and only its TILT is new. + ## The weight is floored at 1, so the pool can never be starved. + for t in turns: + result.add max(1, int(round(1.0 + bias * + (1.0 - max(0.0, t - refDeg) / 180.0)))) + proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand = if m.cols == 0: m.initGrid(ws.arenaWidth, ws.arenaHeight) @@ -820,6 +861,8 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand = var pickedRev = false var pickedMidFlight = false var pickedInterval = 0 + var pickedTurn = 0.0 # log-only: |turn| to the tile that was chosen + var pickedPromoted = false ## log-only: the pick had to break the heat filter # Hull + inside-tiles: only recompute on replan tick (commitTicks == 0) type TileRef = tuple[col, row: int] @@ -871,7 +914,10 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand = # A single hot tile on the path (corridor, bullet core, enemy aura) makes the whole path unsafe. const PathSampleStep = 18.0 # ~half a tile const PathDangerThreshold = 10.0 # max lava on path; above this = unsafe - type ScoredTile = tuple[col, row: int; pathMaxHeat: float] + # j145: `turnDeg` is the |heading change| from the direction we are ALREADY + # travelling to the tile centre. It is carried on the candidate (never folded + # into `pathMaxHeat`) so the pick can bias among the safe tiles only. + type ScoredTile = tuple[col, row: int; pathMaxHeat: float; turnDeg: float] proc pathMaxHeat(m: TFILModule, fx, fy, tx, ty: float): float = ## MAX lava on the straight-line segment (fx,fy) -> (tx,ty), sampled every @@ -899,12 +945,20 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand = while result > 180.0: result -= 360.0 while result < -180.0: result += 360.0 + # j145: the committed travel direction, needed both to score the turn cost of + # every candidate and to draw from it. Computed here (not in the pick block) + # because the candidates are scored before the pick. + let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0 + else: ws.selfHeading + var scoredTiles: seq[ScoredTile] for t in coolTiles: let tx = m.marginX + (t.col.float + 0.5) * GridSize let ty = m.marginY + (t.row.float + 0.5) * GridSize scoredTiles.add (col: t.col, row: t.row, - pathMaxHeat: pathMaxHeat(m, ws.selfX, ws.selfY, tx, ty)) + pathMaxHeat: pathMaxHeat(m, ws.selfX, ws.selfY, tx, ty), + turnDeg: abs(tileOffTravel(m, t.col, t.row, ws.selfX, + ws.selfY, travelDeg))) # Sort by pathMaxHeat ascending (insertion sort — small N) for i in 1..= 2: - # Soft no-reversal preference (arm C): down-weight — never filter — tiles - # that lie >90 deg from the current travel direction. - var dirs: seq[float] - for t in candidates: - let tx = m.marginX + (t.col.float + 0.5) * GridSize - let ty = m.marginY + (t.row.float + 0.5) * GridSize - dirs.add arctan2(ty - ws.selfY, tx - ws.selfX) * 180.0 / PI - let weights = noRevWeights(dirs, travelDeg) - var total = 0 - var forward = 0 + if (TfilNoRev or TfilTurnBias > 0.0) and candidates.len >= 2: + # Soft preferences — down-weight, never filter, and only ever among tiles + # that already passed the hard heat filter above: + # arm C (TR_TFIL_NO_REV, off by default) — 3:1 forward vs rearward, + # a BINARY cut: it cannot tell 20 deg from 90, nor 91 from 179. + # j145 (TR_TFIL_TURN_BIAS, off by default) — a CONTINUOUS turn cost, + # `turnWeights` = 1 + bias * (1 - excess/180). + # The two multiply, so turning one on never silently drops the other. + var weights = if TfilTurnBias > 0.0: + block: + var turns: seq[float] + for t in candidates: turns.add t.turnDeg + turnWeights(turns, TfilTurnBias, TfilTurnRefDeg) + else: + var w1 = newSeq[int](candidates.len) + for i in 0.. 1: inc forward - if forward == 0: - # Fallback: no forward tile exists → uniform draw, so the pool can - # never empty and the pick is identical to the shipped one. + wMax = max(wMax, w) + if wMax <= 1: + # Fallback: no candidate is preferred (every tile is equally bad, or all + # are rearward) → uniform draw, so the pool can never empty and the pick + # degrades to exactly the shipped one. chosen = rand(candidates.high) else: let r = rand(total - 1) @@ -1043,6 +1115,10 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand = else: chosen = rand(candidates.high) let ct = candidates[chosen] + pickedPromoted = m.lastPickPromoted + var minTurn = Inf + for t in candidates: minTurn = min(minTurn, t.turnDeg) + m.lastPickMinTurn = if minTurn == Inf: ct.turnDeg else: minTurn m.commitTarget = (x: m.marginX + (ct.col.float + 0.5) * GridSize, y: m.marginY + (ct.row.float + 0.5) * GridSize) m.commitTicks = TfilCommitTicks @@ -1057,6 +1133,7 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand = while rd < -180.0: rd += 360.0 pickedThisTick = true pickedRev = abs(rd) > 90.0 + pickedTurn = abs(rd) pickedMidFlight = midFlight pickedInterval = m.callCount - m.lastPickCall m.lastPickCall = m.callCount @@ -1129,6 +1206,8 @@ proc computeMove*(m: var TFILModule, ws: WorldState): MoveCommand = tfilLogWrite("{\"tick\":" & $ws.tick & ",\"call\":" & $m.callCount & ",\"sp\":" & $ws.selfSpeed & ",\"ct\":" & $m.commitTicks & ",\"pick\":" & (if pickedThisTick: "1" else: "0") & + ",\"turn\":" & $pickedTurn & + ",\"promote\":" & (if pickedPromoted: "1" else: "0") & ",\"reason\":\"" & reasonName(reason) & "\",\"rev\":" & (if pickedRev: "1" else: "0") & ",\"mid\":" & (if pickedMidFlight: "1" else: "0") & ",\"interval\":" & $pickedInterval & diff --git a/common_libs/tests/measure_tfil_arrival.nim b/common_libs/tests/measure_tfil_arrival.nim index ef9df3f..4c9788e 100644 --- a/common_libs/tests/measure_tfil_arrival.nim +++ b/common_libs/tests/measure_tfil_arrival.nim @@ -217,3 +217,173 @@ echo "| arm | tile_self | tile_enemy | danger | expiry | arrival | hyst |" echo "|---|---:|---:|---:|---:|---:|---:|" for s in rows: echo &"| {s.arm} | {s.byReason[rrTileSelf]} | {s.byReason[rrTileEnemy]} | {s.byReason[rrDanger]} | {s.byReason[rrExpiry]} | {s.byReason[rrArrival]} | {s.byReason[rrHyst]} |" + +# ══════════════════════════════════════════════════════════════════════════════ +# j145 — THE TURN-COST TIEBREAK, the mechanism ruler. +# +# Answers, on the SAME recorded fixture, whether `TR_TFIL_TURN_BIAS` does what +# it claims AMONG THE SAFE TILES and what it costs: +# +# mean |turn| — the |heading change| to the tile the picker chose +# mean regret — how many degrees worse than the SMALLEST-turn candidate +# actually available that choice was. The confound-free +# form: the arms draw from different candidate sets, so the +# raw mean alone can move without the mechanism biting. +# >90 / >135 — picks needing a real turn / a mirror-side pick +# path heat — mean max-lava on the straight path we were told to walk, +# and the share of picks over the hard threshold +# filter broken — the share of picks that had to promote a hot tile because +# FEWER THAN TWO tiles were safe (the shipped fallback) +# own-tile heat — mean lava under the bot's own wheels, every tick: THE +# SAFETY COST of the choice, measured where it is paid +# |turnRate| — the turn the bot actually EXECUTED, not just intended +# arrival — mean ticks to reach the committed tile, and the share of +# commitments that are actually reached +# ══════════════════════════════════════════════════════════════════════════════ + +const ArriveR2 = 18.0 + +type TurnStats = object + arm: string + env: string + ticks: int + picks: int + turnSum: float + minTurnSum: float + tookMin: int + bigTurn: int + flip: int + pathHeatSum: float + hot: int + broken: int + ownHeatSum: float + ownHot: int + turnRateSum: float + hardTurn: int + holdSum: int + reached: int + distSum: float + +proc pathHeatAt(m: TFILModule, fx, fy, tx, ty: float): float = + ## The mover's own sampler (PathSampleStep 18). The lava field is rebuilt from + ## scratch every computeMove, so what is visible just after the call is the + ## field the pick was actually made against. + let ddx = tx - fx + let ddy = ty - fy + let lineDist = sqrt(ddx*ddx + ddy*ddy) + if lineDist <= 0.1: return 0.0 + let steps = max(1, int(lineDist / 18.0)) + var h = 0.0 + for si in 0..steps: + let fr = si.float / steps.float + let (sc, sr) = m.tileAt(fx + ddx * fr, fy + ddy * fr) + h = max(h, m.lavaAt(sc, sr)) + h + +proc replayTurn(arm, envspec: string): TurnStats = + result.arm = arm + result.env = envspec + for tok in envspec.splitWhitespace(): + putEnv(tok.split('=', 1)[0], tok.split('=', 1)[1]) + putEnv("TR_TFIL_COMMIT_LOG", "") + loadTfilCommitEnv() + loadTfilHeatEnv() + randomize(Seed) + var m = initTFIL() + let states = loadStates() + let starts = loadRoundStarts() + var prev = (x: 0.0, y: 0.0) + var hadPicks = false + var held = 0 + for i in 0..= 5.0: inc result.hardTurn + let (bc, br) = m.tileAt(ws.selfX, ws.selfY) + let own = m.lavaAt(bc, br) + result.ownHeatSum += own + if own > 10.0: inc result.ownHot + if m.picks != before: + if hadPicks: result.holdSum += held + let travelDeg = if ws.selfSpeed < -0.01: ws.selfHeading + 180.0 + else: ws.selfHeading + var rd = arctan2(m.commitTarget.y - ws.selfY, m.commitTarget.x - ws.selfX) * + 180.0 / PI - travelDeg + while rd > 180.0: rd -= 360.0 + while rd < -180.0: rd += 360.0 + inc result.picks + result.turnSum += abs(rd) + result.minTurnSum += m.lastPickMinTurn + if abs(rd) <= m.lastPickMinTurn + 0.5: inc result.tookMin + if abs(rd) > 90.0: inc result.bigTurn + if abs(rd) > 135.0: inc result.flip + let ph = pathHeatAt(m, ws.selfX, ws.selfY, m.commitTarget.x, m.commitTarget.y) + result.pathHeatSum += ph + if ph > 10.0: inc result.hot + if m.lastPickPromoted: inc result.broken + if hadPicks and + sqrt((ws.selfX-prev.x)^2 + (ws.selfY-prev.y)^2) < ArriveR2: inc result.reached + result.distSum += sqrt((m.commitTarget.x - ws.selfX)^2 + + (m.commitTarget.y - ws.selfY)^2) + prev = m.commitTarget + hadPicks = true + held = 0 + elif hadPicks: + inc held + if hadPicks: result.holdSum += held + for tok in envspec.splitWhitespace(): + putEnv(tok.split('=', 1)[0], "") + +proc mean(x: float, d: int): float = + if d == 0: return 0.0 + x / d.float + +let turnArms = [ + ("shipped", "TR_MOVEMENT=tfil"), + ("arrive+norev", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4"), + ("turn b=3", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=3"), + ("turn b=9", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=9"), + ("turn b=19", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19"), + ("turn b=39", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=39"), + ("turn b=9 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0"), + ("turn b=19 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19 TR_TFIL_TURN_REF_DEG=0"), + ("turn b=39 ref0", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=39 TR_TFIL_TURN_REF_DEG=0"), + ("turn b=19 ref90", "TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=19 TR_TFIL_TURN_REF_DEG=90"), + ("arrive+norev b=9", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9"), + ("arrive+norev b=19", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=19"), + ("arrive+norev b=9 r90", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=90"), + ("arrive+norev b=9 ref0", "TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0"), +] + +echo "\n\n══ j145 TURN-COST TIEBREAK — mechanism ruler ══\n" +var trows: seq[TurnStats] +for (name, envspec) in turnArms: + trows.add replayTurn(name, envspec) + +echo &"| arm | picks | mean \\|turn\\| | mean regret | took min-turn | >90 deg | opposite (>135) | mean path heat | path heat >10 | filter broken |" +echo "|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|" +for s in trows: + echo &"| {s.arm} | {s.picks} | {mean(s.turnSum, s.picks).formatFloat(ffDecimal,1)} | " & + &"{mean(s.turnSum - s.minTurnSum, s.picks).formatFloat(ffDecimal,1)} deg | " & + &"{pct(s.tookMin, s.picks)} | {pct(s.bigTurn, s.picks)} | {pct(s.flip, s.picks)} | " & + &"{mean(s.pathHeatSum, s.picks).formatFloat(ffDecimal,2)} | {pct(s.hot, s.picks)} | {pct(s.broken, s.picks)} |" + +echo "\n### what the choice costs: the turn EXECUTED, the arrival time, the distance\n" +echo "### (the heat under the bot's own wheels is NOT here: this replay drives the" +echo "### RECORDED self states, so the travelled path is identical in every arm by" +echo "### construction — the pick varies, the trajectory does not. The safety cost" +echo "### of a bias that arrives later / dwells hotter is therefore NOT measurable" +echo "### offline here; `mean path heat` and `filter broken` are the honest proxies.)\n" +echo &"| arm | mean \\|turnRate\\| executed | hard turns (>=5 deg/tick) | mean arrival (ticks) | reached | mean distance to the pick |" +echo "|---|---:|---:|---:|---:|---:|" +for s in trows: + echo &"| {s.arm} | {mean(s.turnRateSum, s.ticks).formatFloat(ffDecimal,2)} | {pct(s.hardTurn, s.ticks)} | " & + &"{mean(s.holdSum.float, s.picks).formatFloat(ffDecimal,1)} | {pct(s.reached, s.picks)} | " & + &"{mean(s.distSum, s.picks).formatFloat(ffDecimal,0)} px |" diff --git a/common_libs/tests/test_tfil_commit_env.nim b/common_libs/tests/test_tfil_commit_env.nim index 55c5cb6..9b84aae 100644 --- a/common_libs/tests/test_tfil_commit_env.nim +++ b/common_libs/tests/test_tfil_commit_env.nim @@ -6,6 +6,8 @@ ## TR_TFIL_COMMIT_ARRIVAL (0/1) default 0 = shipped (j144) ## TR_TFIL_COMMIT_MARGIN (float) default 0.0 = shipped (j144) ## TR_TFIL_NOREV_SPEED (float) default 0.0 = shipped (j144) +## TR_TFIL_TURN_BIAS (float) default 0.0 = shipped (j145) +## TR_TFIL_TURN_REF_DEG (float) default 45.0 (j145) ## ## NO battle, NO Java, NO server. Run with: ## nim c -r --path:common_libs common_libs/tests/test_tfil_commit_env.nim @@ -445,6 +447,216 @@ when declared(loadTfilCommitEnv): " hyst=", s.byReason[rrHyst], " danger=", s.byReason[rrDanger] + # ── 5. j145: the turn-cost TIEBREAK among SAFE tiles (default OFF) ───────── + + type PickRec = object + turn: float ## |heading change| from the travel direction to the pick + minTurn: float ## the smallest |turn| AVAILABLE in that candidate set — + ## turn - minTurn is the regret of the draw, which is + ## the confound-free form of the mechanism metric + pathHeat: float ## max lava on the straight path the bot was told to walk + promoted: bool ## the pick had to break the hard heat filter + reached: bool ## the commitment ended with the bot on the tile + + const DangerThreshold = 10.0 ## PathDangerThreshold inside the mover + + proc probePathHeat(m: TFILModule, fx, fy, tx, ty: float): float = + ## Mirrors the mover's own sampler (PathSampleStep = 18, ~half a tile). The + ## field is rebuilt from scratch every computeMove, so the `m.lava` visible + ## just after the call is exactly the field the pick was made against. + let ddx = tx - fx + let ddy = ty - fy + let lineDist = sqrt(ddx*ddx + ddy*ddy) + if lineDist <= 0.1: return 0.0 + let steps = max(1, int(lineDist / 18.0)) + var h = 0.0 + for si in 0..steps: + let f = si.float / steps.float + let (sc, sr) = m.tileAt(fx + ddx * f, fy + ddy * f) + h = max(h, m.lavaAt(sc, sr)) + h + + proc replayJ145(tag: string, bias, refDeg: float, + arrive: bool, norevSpeed: float): seq[PickRec] = + ## Drive the REAL computeMove with the j145 knobs set through the env (and + ## the j144 knobs forced through the vars, so no `.env` can be in the way), + ## and record what each pick actually cost. + putEnv("TR_TFIL_TURN_BIAS", $bias) + putEnv("TR_TFIL_TURN_REF_DEG", $refDeg) + putEnv("TR_TFIL_COMMIT_LOG", "") + loadTfilCommitEnv() + TfilCommitArrival = arrive + TfilCommitMargin = 0.0 + TfilNoRevSpeed = norevSpeed + randomize(Seed) + var m = initTFIL() + let states = loadStates() + let starts = loadRoundStarts() + var prev = (x: 0.0, y: 0.0) + var hadPicks = false + for i in 0.. 180.0: rd -= 360.0 + while rd < -180.0: rd += 360.0 + result.add PickRec(turn: abs(rd), minTurn: m.lastPickMinTurn, + pathHeat: probePathHeat(m, ws.selfX, ws.selfY, + m.commitTarget.x, m.commitTarget.y), + promoted: m.lastPickPromoted, + reached: hadPicks and + sqrt((ws.selfX-prev.x)^2 + (ws.selfY-prev.y)^2) < 18.0) + prev = m.commitTarget + hadPicks = true + putEnv("TR_TFIL_TURN_BIAS", "") + putEnv("TR_TFIL_TURN_REF_DEG", "") + loadTfilCommitEnv() + TfilCommitArrival = false + TfilNoRevSpeed = 0.0 + discard tag + + type TurnStats = object + picks: int + turnSum: float + minTurnSum: float + tookMin: int ## the draw landed on the smallest-turn candidate + bigTurn: int ## |turn| > 90 deg + flip: int ## |turn| > 135 deg — the opposite side + heatSum: float ## mean path heat of the tile we actually walked to + hotPicks: int ## ... over the hard threshold + badHot: int ## ... over the threshold WITHOUT the filter being + ## broken — MUST be 0 at any turn bias + broken: int ## picks that had to promote a hot tile (fewer than + ## two safe tiles existed) — the shipped fallback + reached: int + + proc turnStats(p: seq[PickRec]): TurnStats = + result.picks = p.len + for r in p: + result.turnSum += r.turn + result.minTurnSum += r.minTurn + if r.turn <= r.minTurn + 0.5: inc result.tookMin + if r.turn > 90.0: inc result.bigTurn + if r.turn > 135.0: inc result.flip + result.heatSum += r.pathHeat + if r.pathHeat > DangerThreshold: + inc result.hotPicks + if not r.promoted: inc result.badHot + if r.promoted: inc result.broken + if r.reached: inc result.reached + + proc meanTurn(s: TurnStats): float = + if s.picks == 0: return 0.0 + s.turnSum / s.picks.float + proc meanRegret(s: TurnStats): float = + ## How many degrees WORSE than the best available tile the draw actually was. + ## Immune to the composition confound that the raw mean |turn| has (a bias + ## arm makes different picks, so the two arms' candidate sets differ). + if s.picks == 0: return 0.0 + (s.turnSum - s.minTurnSum) / s.picks.float + proc meanPathHeat(s: TurnStats): float = + if s.picks == 0: return 0.0 + s.heatSum / s.picks.float + proc pct(n, d: int): string = + if d == 0: return "-" + (100.0 * n.float / d.float).formatFloat(ffDecimal, 1) & "%" + + proc testJ145() = + # 5a. the shipped default is OFF — the parity check above is the proof + check "j145: the turn bias defaults to today's uniform draw (bias 0)", + TfilTurnBias == 0.0 and TfilTurnRefDeg == 45.0 + + # 5b. the weighting, in pure form + check "j145: bias 0 gives every safe tile weight 1 (byte-identical to the " & + "shipped uniform draw)", turnWeights(@[0.0, 91.0, 180.0], 0.0, 45.0) == + @[1, 1, 1] + check "j145: the penalty is CONTINUOUS past the reference angle, where the " & + "binary TR_TFIL_NO_REV cannot see (bias 9, ref 45: 45/90/135/180 deg " & + "-> 10/8/6/3)", turnWeights(@[45.0, 90.0, 135.0, 180.0], 9.0, 45.0) == + @[10, 8, 6, 3] + check "j145: a turn inside the reference angle is never penalised", + turnWeights(@[0.0, 20.0, 45.0], 5.0, 45.0) == @[6, 6, 6] + check "j145: the weight falls monotonically with the turn (ref 0, bias 9: " & + "0/30/60/90/120/180 deg -> 10/9/8/8/7/1)", + turnWeights(@[0.0, 30.0, 60.0, 90.0, 120.0, 180.0], 9.0, 0.0) == + @[10, 9, 7, 6, 4, 1] + check "j145: the weight is floored at 1, so the pool can never be starved", + turnWeights(@[0.0, 180.0, 179.0], 9.0, 0.0)[1] >= 1 and + turnWeights(@[0.0, 180.0, 179.0], 9.0, 0.0) == @[10, 1, 1] + check "j145: `bias` IS the odds ratio — with ref 0 a straight-ahead safe tile " & + "is drawn 1+bias times as often as a 180 deg one (9 -> 10:1)", + turnWeights(@[0.0, 180.0], 9.0, 0.0) == @[10, 1] + + # 5c. THE GATE: an absurd turn cost must not rescue a hot tile. The heat + # filter is UPSTREAM of the weighting, so an over-threshold pick can + # only ever be one the mover had to promote because nothing was safe. + let wild = turnStats(replayJ145("wild", 99.0, 45.0, true, 4.0)) + check "j145: with an absurd turn bias (" & $wild.picks & " picks) NO tile " & + "over the heat threshold is ever chosen unless the filter had to be " & + "broken (" & $wild.badHot & " violations)", + wild.badHot == 0 + check "j145: the over-threshold picks that do happen are only the promoted " & + "ones (" & $wild.hotPicks & "/" & $wild.picks & ", the shipped " & + "fewer-than-2-safe-tiles fallback)", wild.badHot == 0 + + # 5d. the mechanism: the turn really gets smaller, without paying for it in + # heat, and without emptying the pool + let off = turnStats(replayJ145("off", 0.0, 45.0, true, 4.0)) + let mild = turnStats(replayJ145("mild", 9.0, 0.0, true, 4.0)) + let firm = turnStats(replayJ145("firm", 39.0, 0.0, true, 4.0)) + check "j145: with the bias on, the mean |turn| to the chosen tile falls " & + "(" & meanTurn(off).formatFloat(ffDecimal, 1) & " -> " & + meanTurn(mild).formatFloat(ffDecimal, 1) & " -> " & + meanTurn(firm).formatFloat(ffDecimal, 1) & " deg)", + meanTurn(mild) < meanTurn(off) * 0.95 and + meanTurn(firm) < meanTurn(off) * 0.95 + check "j145: the REGRET of the draw (how many degrees worse than the best " & + "AVAILABLE candidate) falls " & + "(" & meanRegret(off).formatFloat(ffDecimal, 1) & " -> " & + meanRegret(mild).formatFloat(ffDecimal, 1) & " -> " & + meanRegret(firm).formatFloat(ffDecimal, 1) & " deg) — the " & + "confound-free form of the mechanism", + meanRegret(mild) < meanRegret(off) * 0.9 and + meanRegret(firm) < meanRegret(off) * 0.9 + check "j145: the share of picks needing >90 deg of turn falls " & + "(" & pct(off.bigTurn, off.picks) & " -> " & pct(mild.bigTurn, mild.picks) & + " -> " & pct(firm.bigTurn, firm.picks) & ")", + mild.bigTurn < off.bigTurn + check "j145: a mirror-image tile no longer beats a straight-ahead one as " & + "readily — opposite-side picks fall " & pct(off.flip, off.picks) & " -> " & + pct(mild.flip, mild.picks) & " -> " & pct(firm.flip, firm.picks), + mild.flip.float < off.flip.float * 0.95 + check "j145: SAFETY COST — the mean path heat of the chosen tile does not " & + "rise (bias off " & meanPathHeat(off).formatFloat(ffDecimal, 2) & + " vs bias 9 " & meanPathHeat(mild).formatFloat(ffDecimal, 2) & + " vs bias 39 " & meanPathHeat(firm).formatFloat(ffDecimal, 2) & ")", + meanPathHeat(mild) <= meanPathHeat(off) * 1.05 and + meanPathHeat(firm) <= meanPathHeat(off) * 1.05 + check "j145: the bias never empties the pool — decisions stay within 5% of " & + "the same arm without it (" & $off.picks & " -> " & $mild.picks & " / " & + $firm.picks & ")", + abs(firm.picks.float - off.picks.float) <= 0.05 * off.picks.float + + echo "\n j145 diagnostics (offline fixture replay, arrive+norev base):" + for (nm, s) in [("bias 0 (shipped)", off), ("bias 9 ref0", mild), ("bias 39 ref0", firm)]: + echo " ", nm.alignLeft(18), " picks=", s.picks, + " mean|turn|=", meanTurn(s).formatFloat(ffDecimal, 1), + " regret=", meanRegret(s).formatFloat(ffDecimal, 1), + " tookMin=", pct(s.tookMin, s.picks), + " >90deg=", pct(s.bigTurn, s.picks), + " >135deg=", pct(s.flip, s.picks), + " filter broken=", pct(s.broken, s.picks), + " mean path heat=", meanPathHeat(s).formatFloat(ffDecimal, 2), + " reached=", pct(s.reached, s.picks) + # ── driver ─────────────────────────────────────────────────────────────────── testDefaultParity() @@ -452,6 +664,7 @@ when declared(loadTfilCommitEnv): testKnobParsing() testArms() testJ144() + testJ145() if failures > 0: echo "\n", failures, " check(s) FAILED" diff --git a/docs/movement_campaign.md b/docs/movement_campaign.md index e654d0a..40f76c7 100644 --- a/docs/movement_campaign.md +++ b/docs/movement_campaign.md @@ -2978,3 +2978,77 @@ He should expect the dodge to look *smoother and more deliberate* (fewer, longer commitments) rather than twitchy, and he should see fewer bullets connect. He should NOT expect a step change in his score from this alone: the measured outcome effect is +0.28 wins/run with a p of 0.057 on the primary test. + +--- + +# Batch 6 — TFIL turn-cost tiebreak (j145) + +*Pre-registered BEFORE any battle of this batch was launched. No battle of this +batch existed when this section was written; the frozen binary for it is the +commit that adds the tiebreak.* + +## The cause this batch fixes + +The `tfil` picker's `ScoredTile` carried **one** term, `pathMaxHeat`. After the +hard filter (`pathMaxHeat <= PathDangerThreshold` = 10) the pick was a plain +`rand()` over the survivors, so a far-cooler tile on the OPPOSITE side was drawn +exactly as readily as a marginally-cooler one straight ahead. The only +heading-aware influence in the mover is `NoRevForwardWeight = 3` under +`TR_TFIL_NO_REV` — **binary** (it cannot tell 20 deg from 90, nor 91 from 179) +and **off by default**. This batch adds the continuous version. + +## The treatment + +Two knobs, both **off by default** (the shipped default path is byte-for-byte +identical — the golden in `test_tfil_commit_env.nim` still passes): + +| knob | default | meaning | +|---|---|---| +| `TR_TFIL_TURN_BIAS` | `0.0` | the tiebreak's **odds ratio**: a straight-ahead safe tile is drawn `1 + bias` times as often as a 180 deg one | +| `TR_TFIL_TURN_REF_DEG` | `45.0` | the turn below which no penalty applies | + +The draw weight of a safe candidate is + + w = max(1, round(1 + bias * (1 - max(0, |turn| - refDeg) / 180))) + +**The safety filter is untouched and stays hard.** Turn cost is never added to +the heat score (`heat + k*turnDeg` would trade dodging for smoothness, which is +backwards in a bullet-dodging game); the bias is applied *only* to the +weight of a draw *among tiles that already passed the filter*. Guard: +`test_tfil_commit_env.nim` runs an absurd bias (99:1) and asserts that **no** +over-threshold tile is ever chosen unless the mover's own "fewer than two tiles +are safe" fallback promoted it. + +**Randomness is preserved.** Job j51 (`3142b70`) measured that randomness in +this tie is load-bearing for this bot — a deterministic argmin scored worse. +The pick is therefore a **weighted draw**, not an argmin; every weight is +floored at 1 so the pool can never be emptied, and at bias 0 every weight is 1, +i.e. exactly the shipped uniform draw. + +## Arms (frozen, all `TR_MOVEMENT=tfil`) + +1. `tfil_shipped` — stock defaults. **The reference.** +2. `arrive_norev` — the two knobs j144 recommends (`COMMIT_ARRIVAL=1`, + `NOREV_SPEED=4`, `MARGIN=0`). +3. `arrive_norev_turn` — arm 2 + `TURN_BIAS=9 TURN_REF_DEG=0`. +4. `turn_only` — `TURN_BIAS=9 TURN_REF_DEG=0` alone; isolates the turn fix. + +Panel: the FROZEN 15-opponent `tools/ab/panel_movement.txt`. Harness: +`tools/ab/tournament_run.sh` + `tournament_analyze.py`. + +## Pre-registered prediction and decision rule + +* **Prediction.** Arm 3 > arm 2 > arm 1 on damage/run and round wins, because a + smaller commanded turn is a faster arrival and a shorter exposure. Arm 4 sits + between arm 1 and arm 3. The **incoming hit rate is the mechanism, not the + verdict** — the verdict is damage/run and round wins under the campaign's + pre-registered rule 2 (cross-opponent sign test p < 0.05 on one primary metric + with the other not down), with the SD/SE/95% CI/MDE reported alongside. +* **If nothing separates**, the verdict is *not distinguishable* and it is + **not shipped**. The pre-registered bar is not re-interpreted afterwards. +* **A null here does NOT undo j144.** j144's result is a *mechanism* result + (the incoming hit rate fell 18.07% → 14.92%, sign-flip p = 0.0013, in two + independent blocks) plus an under-powered outcome null. This batch can only + add to or fail to add to that; it cannot retract it. + +*(results appended below after the battles)* diff --git a/tools/ab/arms_tfil_turn.txt b/tools/ab/arms_tfil_turn.txt new file mode 100644 index 0000000..74ca840 --- /dev/null +++ b/tools/ab/arms_tfil_turn.txt @@ -0,0 +1,35 @@ +# ───────────────────────────────────────────────────────────────────────────── +# arms_tfil_turn.txt — j145: the TFIL TURN-COST TIEBREAK, four arms on the +# frozen movement panel (tools/ab/panel_movement.txt), all with TR_MOVEMENT=tfil +# pinned EXPLICITLY. +# +# WHAT THE KNOB IS. Among the tiles that already passed the HARD heat filter +# (pathMaxHeat <= PathDangerThreshold), the draw is weighted by +# +# w = max(1, round(1 + TR_TFIL_TURN_BIAS * (1 - max(0,|turn| - REF) / 180))) +# +# so a straight-ahead safe tile is drawn `1 + bias` times as often as a 180 deg +# one. The heat score is NOT touched: a tile over the threshold still loses, at +# any bias (guard: test_tfil_commit_env.nim, "j145: with an absurd turn bias..."). +# The draw stays RANDOM (job j51, 3142b70: a deterministic argmin measured worse). +# +# Pre-registered in docs/movement_campaign.md ("TFIL turn-cost tiebreak") BEFORE +# any of these battles ran. Reference is `tfil_shipped`. +# +# Format: name | ENV=value ENV=value | label +# ───────────────────────────────────────────────────────────────────────────── + +# 1. THE ARM TO BEAT — the shipped tfil defaults, explicitly selected. +tfil_shipped | TR_MOVEMENT=tfil | shipped tfil defaults (control / reference) + +# 2. the two knobs job j144 recommends, unchanged. +arrive_norev | TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 | j144's recommended .env + +# 3. j144's knobs PLUS the turn tiebreak, at the value the offline ruler picked +# (the knee of the bias curve: bias 9 ref 0 cuts mean |turn| 11% and opposite +# picks 22% with no path-heat cost; bias 39 buys almost nothing more). +arrive_norev_turn | TR_MOVEMENT=tfil TR_TFIL_COMMIT_ARRIVAL=1 TR_TFIL_NOREV_SPEED=4 TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0 | j144 + the turn tiebreak + +# 4. THE TURN FIX ALONE — isolates the tiebreak with no j144 knob set, which is +# the only arm that can say whether it is worth anything by itself. +turn_only | TR_MOVEMENT=tfil TR_TFIL_TURN_BIAS=9 TR_TFIL_TURN_REF_DEG=0 | the turn tiebreak alone