diff --git a/ModularBot_garage/src/ModularBot.nim b/ModularBot_garage/src/ModularBot.nim index 50540e9..42da05b 100644 --- a/ModularBot_garage/src/ModularBot.nim +++ b/ModularBot_garage/src/ModularBot.nim @@ -100,6 +100,11 @@ let MovementName* = getEnv("TR_MOVEMENT", "tfil").strip().toLowerAscii() ## Per-process output paths so concurrent A/B runs do not clobber each other. let GunStatsPath = getEnv("GUN_STATS_PATH", "/tmp/gun_stats.jsonl") let ShotLogPath = getEnv("GUN_SHOTLOG_PATH", "/tmp/shot_log.jsonl") +## Energy-aware power policy observability (TR_POWER_LOG=1): emit ONE line per +## CHANGE of the (power, cap, reason) decision — not per tick — so the GUI log +## shows why the cap moved. The policy itself lives in the shared gun harness +## (`applyPowerPolicy`), so both live and offline paths see the same rule. +let PowerLog = existsEnv("TR_POWER_LOG") const GunNames = ["HeadOn", "Linear", "Tsetlin", "Circular", "GuessFactor", "Pattern", "WallBounce", "Accel", "StopShot", "Displace", "AvgLead", "DecayGF", "KNN", "TMSelect"] const @@ -175,6 +180,7 @@ type bulletShot: Table[int, PendingShot] ## bulletId -> shot metadata (Task A shot log) pendingHitBullets: HashSet[int] ## hit bulletIds seen before their onBulletFired stamp gunSelectionCount: array[14, int] + lastPowerLogKey: string ## change detector for the TR_POWER_LOG line lastKnownTargetId: int ## persists through death, used for round-end stats # Radar measurement instrumentation (only touched when RadarScanLog is set). radarScanCounts: Table[int, int] ## onScannedBot calls per enemy id @@ -920,7 +926,18 @@ method run*(bot: ModularBot) = ) # Gun selection + fire. `bot.tick` drives the selector's dwell window. - let (selectedGun, _, power) = selectShot(bot.tracker, tid, bot.tick) + # The energy-aware power cap uses the current target distance and our own + # energy; `shouldRam` (the movement code's decision) exempts it. + let (selectedGun, _, power, pdec) = selectShotPolicy( + bot.tracker, tid, bot.tick, + dist = ramDist, selfEnergy = ws.selfEnergy, ramming = shouldRam) + if PowerLog: + let pkey = fmt"{power:.1f}|{pdec.cap:.1f}|{pdec.reason}" + if pkey != bot.lastPowerLogKey: + bot.lastPowerLogKey = pkey + echo fmt"[power] p={power:.1f} cap={pdec.cap:.1f} " & + fmt"reason={powerReasonName(pdec.reason)} " & + fmt"dist={ramDist:.0f} selfE={ws.selfEnergy:.0f} gun={GunNames[selectedGun]}" bot.gunSelectionCount[selectedGun] += 1 if selectedGun != bot.currentGun: bot.currentGun = selectedGun diff --git a/common_libs/gun_harness/selector.nim b/common_libs/gun_harness/selector.nim index 837d12b..f5946d8 100644 --- a/common_libs/gun_harness/selector.nim +++ b/common_libs/gun_harness/selector.nim @@ -66,10 +66,40 @@ proc shouldFire*(currentGunDir, targetAngle, gunHeat, distPx: float): bool = elif delta < -180.0: delta += 360.0 abs(delta) <= aimToleranceDeg(distPx) and gunHeat <= 0.0 -proc selectShot*(t: var VirtualTracker, targetId: int = -1, tick = 0): (GunId, int, float) = +proc selectShotPolicy*(t: var VirtualTracker, targetId = -1, tick = 0, + dist = 0.0, selfEnergy = 100.0, + ramming = false): (GunId, int, float, PowerCap) = + ## `selectShot` plus the energy-aware power-policy decision, so a caller can + ## log the cap and its reason (see `applyPowerPolicy` in virtual_bullets). + ## + ## `dist` is the current distance (px) to the target and `selfEnergy` our own + ## energy; `ramming` exempts the caps (the movement code's `shouldRam` is the + ## single source of truth). The policy is applied identically wherever this is + ## called, so live and any offline caller cannot diverge. + let gunId = t.selectGun(targetId, tick) + let (prefBin, preferred) = t.bestPower(gunId, targetId) + # pEst / pRef mirror `bestPower`'s own fitness source (per-target when data + # exists, else the deterministic aggregate). An empty bin carries no rate of + # its own, so it borrows the gun's aggregate — the same "no data" case the + # policy documents. + let fit = t.fitnessFor(targetId) + let pRef = if PowerRefFixed > 0.0: PowerRefFixed + else: gunRate(fit[gunId], pooled = true) + let pEst = + if fit[gunId].bins[prefBin].count == 0: pRef + else: fit[gunId].bins[prefBin].hitRate() + let dec = applyPowerPolicy(preferred, dist, selfEnergy, pEst, pRef, ramming) + result = (gunId, binIndexForPower(dec.power), dec.power, dec) + +proc selectShot*(t: var VirtualTracker, targetId = -1, tick = 0, + dist = 0.0, selfEnergy = 100.0, + ramming = false): (GunId, int, float) = ## Returns (gunId, powerBinIdx, power) — the shot to take this tick. ## Pass targetId to pick the best gun for that specific enemy. `tick` drives - ## the minimum-dwell hysteresis (see `selectGun`). - let gunId = t.selectGun(targetId, tick) - let (binIdx, power) = t.bestPower(gunId, targetId) + ## the minimum-dwell hysteresis (see `selectGun`). `dist`/`selfEnergy`/`ramming` + ## feed the energy-aware power cap (`TR_POWER_POLICY`); defaults keep every + ## existing caller compiling, and `TR_POWER_POLICY=0` reproduces the uncapped + ## `bestPower` preference. Use `selectShotPolicy` when the cap/reason is needed. + let (gunId, binIdx, power, _) = + t.selectShotPolicy(targetId, tick, dist, selfEnergy, ramming) result = (gunId, binIdx, power) diff --git a/common_libs/gun_harness/virtual_bullets.nim b/common_libs/gun_harness/virtual_bullets.nim index 9f25fcb..9c6b73f 100644 --- a/common_libs/gun_harness/virtual_bullets.nim +++ b/common_libs/gun_harness/virtual_bullets.nim @@ -599,6 +599,103 @@ proc bestPower*(t: VirtualTracker, gunId: GunId, targetId: int = -1): (int, floa if fw.count == 0 or fw.hitRate() >= bar: return (binIdx, PowerBins[binIdx]) +# ── energy-aware power policy (TR_POWER_*) ─────────────────────────────────── +# +# `bestPower` answers "which power bin does this gun's own virtual data prefer?" +# and is deliberately left untouched. The policy below CAPS that preference using +# only cheap, always-available state — range, our own energy, and the gun's own +# rate — so a long-range or low-energy shot trades single-hit damage for a +# faster bullet (speed = 20-3p, so LOW power is FASTER and needs less lead) and a +# shorter fire interval (10+2p, so LOW power = MORE shots). It never RAISES +# power, so the shipped behaviour is exactly the `cap = 3.0` case, which is also +# the control arm (`TR_POWER_POLICY=0`). +# +# Measured basis (real shots vs DrussGT, 8-16 runs): hit rate 21.6% at 0-100px, +# 27.1% at 100-200, then 19.3% at 200-300, 10.9% at 300-400, 6.8% at 400-600 and +# 5.4% at 600-800. Energy math: E[dE] = p(3P-1), so the break-even hit +# probability is 1/3 INDEPENDENT of power; at range/low energy the extra speed +# and shots of p=1.0 dominate. Damage is 4p (p<=1) / 6p-2 (p>1). +const + PowerPolicyEnvVar* = "TR_POWER_POLICY" ## 0 = control arm (uncapped) + PowerFarDistEnvVar* = "TR_POWER_FAR_DIST" ## px; beyond this = bad-chances zone + PowerLowEnergyEnvVar* = "TR_POWER_LOW_ENERGY" ## self energy below this = conserve + PowerFarCapEnvVar* = "TR_POWER_FAR_CAP" ## cap for far / low-energy + PowerMidCapEnvVar* = "TR_POWER_MID_CAP" ## cap when close+healthy but not above avg + PowerRefEnvVar* = "TR_POWER_REF" ## 0 = gun's own mean; >0 = fixed P_ref + +let PowerPolicyEnabled* = envBool(PowerPolicyEnvVar, true) +let PowerFarDist* = envFloat(PowerFarDistEnvVar, 200.0) +let PowerLowEnergy* = envFloat(PowerLowEnergyEnvVar, 50.0) +let PowerFarCap* = envFloat(PowerFarCapEnvVar, 1.0) +let PowerMidCap* = envFloat(PowerMidCapEnvVar, 2.0) +let PowerRefFixed* = envFloat(PowerRefEnvVar, 0.0) + +type + PowerReason* = enum + prFull ## above-average chances, close, healthy -> full power + prFar ## beyond TR_POWER_FAR_DIST -> bad-chances zone + prLowEnergy ## self energy below TR_POWER_LOW_ENERGY -> conserve + prBelowAvg ## chances not above the gun's own average -> no power 3.0 + prRam ## ramming: exempt (at contact P->1, so 3.0 is correct) + + PowerCap* = object + power*: float ## the power to fire (<= the gun's preference) + cap*: float ## the cap applied (3.0 = uncapped) + reason*: PowerReason ## why + +proc powerReasonName*(r: PowerReason): string = + case r + of prFull: "full" + of prFar: "far" + of prLowEnergy: "lowEnergy" + of prBelowAvg: "belowAvg" + of prRam: "ram" + +proc binIndexForPower*(power: float): int = + ## Index of `power` in `PowerBins`; if it is not an exact bin value, the + ## highest bin whose power does not exceed it (0 if none). Keeps the returned + ## bin index consistent with a capped power. + result = 0 + for i in 0.. far > low energy > below average > full. + ## + ## `pEst` is the gun's rate for the bin it chose (or its aggregate when that + ## bin is empty); `pRef` is the gun's aggregate mean (or the fixed + ## `TR_POWER_REF`). A cold gun has no data, so `pEst <= pRef` is vacuously + ## true and it gets the mid cap — deliberately conservative until it has + ## evidence its chances are above average. + if ramming: + return PowerCap(power: preferredPower, cap: 3.0, reason: prRam) + if not enabled: + return PowerCap(power: preferredPower, cap: 3.0, reason: prFull) + var cap: float + var reason: PowerReason + if dist > farDist: + cap = farCap + reason = prFar + elif selfEnergy < lowEnergy: + cap = farCap + reason = prLowEnergy + elif pEst <= pRef: + cap = midCap + reason = prBelowAvg + else: + cap = 3.0 + reason = prFull + PowerCap(power: min(preferredPower, cap), cap: cap, reason: reason) + proc chooseFromFit*(fit: seq[GunFitness], diag: ptr SelectorDiag = nil, mode: SelectorMode = smAbsolute, referenceRate = -1.0, diff --git a/common_libs/tests/test_power_policy.nim b/common_libs/tests/test_power_policy.nim new file mode 100644 index 0000000..5bb49ac --- /dev/null +++ b/common_libs/tests/test_power_policy.nim @@ -0,0 +1,207 @@ +## Unit guard for the energy-aware power policy (TR_POWER_*). +## +## The policy is a CAP on the gun's own preferred bin: at long range or low +## energy it trades single-hit damage for a faster bullet and more shots, and it +## withholds power 3.0 unless the gun's own chances for the chosen bin are above +## that gun's average. It must NEVER raise power, and `TR_POWER_POLICY=0` must +## reproduce the uncapped preference exactly (the control arm). +## +## Pure: no Java, no battle. Run: +## nim c -r common_libs/tests/test_power_policy.nim +## and once with the flag off: +## TR_POWER_POLICY=0 nim c -r common_libs/tests/test_power_policy.nim + +import std/[strformat, math, tables] +import gun_harness/virtual_bullets +import gun_harness/selector + +var failures = 0 +proc check(name: string, ok: bool) = + if ok: echo "PASS: ", name + else: echo "FAIL: ", name; inc failures + +proc recordHit(fw: var FitnessWindow, hit: bool) = + fw.hits[fw.head] = hit + fw.head = (fw.head + 1) mod WindowSize + inc fw.count + +proc seedWindow(t: var VirtualTracker, targetId, gunId, binIdx, hits, misses: int) = + if targetId notin t.fitness: + t.fitness[targetId] = newSeq[GunFitness](t.numGuns) + var fw = addr t.fitness[targetId][gunId].bins[binIdx] + for _ in 0.. TR_POWER_FAR_DIST -> cap 1.0 (far)", + d.power == 1.0 and d.cap == 1.0 and d.reason == prFar + +proc testLowEnergy() = + let d = applyPowerPolicy(3.0, 100.0, 30.0, 0.9, 0.1, false, enabled = true) + check "self energy < TR_POWER_LOW_ENERGY -> cap 1.0 (lowEnergy)", + d.power == 1.0 and d.cap == 1.0 and d.reason == prLowEnergy + +proc testBelowAverage() = + # pEst == pRef is "not above average": withhold power 3.0 -> cap 2.0. + let d = applyPowerPolicy(3.0, 100.0, 100.0, 0.2, 0.2, false, enabled = true) + check "chances not above average -> cap 2.0 (belowAvg)", + d.power == 2.0 and d.cap == 2.0 and d.reason == prBelowAvg + # Strictly below also caps. + let e = applyPowerPolicy(3.0, 100.0, 100.0, 0.1, 0.2, false, enabled = true) + check "chances strictly below average -> cap 2.0 (belowAvg)", + e.power == 2.0 and e.reason == prBelowAvg + +proc testAboveAverageFull() = + let d = applyPowerPolicy(3.0, 100.0, 100.0, 0.5, 0.2, false, enabled = true) + check "above average + close + healthy -> cap 3.0 (full)", + d.power == 3.0 and d.cap == 3.0 and d.reason == prFull + +proc testRamExempt() = + # Far, low energy, no chance data: still full power because we are ramming. + let d = applyPowerPolicy(3.0, 500.0, 5.0, 0.0, 0.9, true) + check "ramming exempts the caps even far + low energy", + d.power == 3.0 and d.reason == prRam + # Ram beats far even with the policy on. + let e = applyPowerPolicy(3.0, 500.0, 5.0, 0.0, 0.9, true, enabled = true) + check "ram exemption takes precedence over the far cap", e.power == 3.0 + +proc testCapNeverRaises() = + # Preferred below every cap must pass through untouched. + let a = applyPowerPolicy(1.0, 500.0, 5.0, 0.0, 0.9, false, enabled = true) + check "far cap does not raise a preferred p=1.0", a.power == 1.0 + let b = applyPowerPolicy(1.5, 100.0, 100.0, 0.1, 0.2, false, enabled = true) + check "mid cap does not raise a preferred p=1.5", b.power == 1.5 + let c = applyPowerPolicy(1.5, 100.0, 100.0, 0.9, 0.2, false, enabled = true) + check "full cap does not raise a preferred p=1.5", c.power == 1.5 + +proc testPrecedence() = + # far beats low energy, and low energy beats belowAvg. + let a = applyPowerPolicy(3.0, 500.0, 5.0, 0.0, 0.9, false, enabled = true) + check "far takes precedence over low energy", a.reason == prFar + let b = applyPowerPolicy(3.0, 100.0, 5.0, 0.0, 0.9, false, enabled = true) + check "low energy takes precedence over belowAvg", b.reason == prLowEnergy + +proc testDisabledUncapped() = + # enabled=false is the code path TR_POWER_POLICY=0 drives. + for p in [1.0, 1.5, 2.0, 3.0]: + let d = applyPowerPolicy(p, 500.0, 5.0, 0.0, 0.9, false, enabled = false) + check fmt"policy off reproduces the uncapped preference p={p:.1f}", + d.power == p and d.cap == 3.0 + +proc testEnvFlag() = + # The flag is read once at module init, so the active branch is selected by + # the process environment. Run the test twice (default and TR_POWER_POLICY=0). + if PowerPolicyEnabled: + check "TR_POWER_POLICY default (on): far shot is capped to 1.0", + applyPowerPolicy(3.0, 300.0, 100.0, 0.5, 0.2, false).power == 1.0 + else: + check "TR_POWER_POLICY=0: far shot is NOT capped (control arm)", + applyPowerPolicy(3.0, 300.0, 100.0, 0.5, 0.2, false).power == 3.0 + +proc testBinIndex() = + check "binIndexForPower maps the shipped bins exactly", + binIndexForPower(1.0) == 0 and binIndexForPower(1.5) == 1 and + binIndexForPower(2.0) == 2 and binIndexForPower(3.0) == 3 + check "binIndexForPower snaps a non-bin cap to the highest bin not above it", + binIndexForPower(2.5) == 2 and binIndexForPower(0.5) == 0 + +proc testReasonNames() = + check "reason names match the documented log vocabulary", + powerReasonName(prFar) == "far" and + powerReasonName(prLowEnergy) == "lowEnergy" and + powerReasonName(prBelowAvg) == "belowAvg" and + powerReasonName(prFull) == "full" and + powerReasonName(prRam) == "ram" + +# ── integration through the tracker (per-target fitness + empty bins) ───────── +# These depend on the process-wide flag, so each branch asserts the behaviour of +# the mode it is actually running in. + +proc testTrackerAboveAverage() = + var t = initTracker(1) + # bin3 is far above this gun's aggregate (50% vs 20%): above average. + seedWindow(t, 7, 0, 0, 10, 90) + seedWindow(t, 7, 0, 1, 10, 90) + seedWindow(t, 7, 0, 2, 10, 90) + seedWindow(t, 7, 0, 3, 50, 50) + let (g, _, p, d) = t.selectShotPolicy(7, tick = 0, dist = 100.0, + selfEnergy = 100.0, ramming = false) + if PowerPolicyEnabled: + check "tracker: above-average bin + close + healthy -> power 3.0", + g == 0 and p == 3.0 and d.reason == prFull + else: + check "tracker (control): above-average bin still fires power 3.0", + g == 0 and p == 3.0 + +proc testTrackerBelowAverage() = + var t = initTracker(1) + for b in 0.. cap 2.0 (belowAvg)", p == 2.0 and d.reason == prBelowAvg + else: + check "tracker (control): flat gun keeps its uncapped preference (power 3.0)", + p == 3.0 + +proc testTrackerFarAndLowEnergy() = + var t = initTracker(1) + seedWindow(t, 7, 0, 3, 50, 50) + let (_, _, pFar, dFar) = t.selectShotPolicy(7, 0, dist = 250.0, + selfEnergy = 100.0, ramming = false) + let (_, _, pLow, dLow) = t.selectShotPolicy(7, 0, dist = 100.0, + selfEnergy = 30.0, ramming = false) + if PowerPolicyEnabled: + check "tracker: far -> power 1.0", pFar == 1.0 and dFar.reason == prFar + check "tracker: low energy -> power 1.0", pLow == 1.0 and dLow.reason == prLowEnergy + else: + check "tracker (control): far does NOT cap (uncapped preference)", pFar == 3.0 + check "tracker (control): low energy does NOT cap (uncapped preference)", pLow == 3.0 + +proc testTrackerColdAndEmptyBin() = + var t = initTracker(1) + let (_, _, pCold, dCold) = t.selectShotPolicy(7, 0, dist = 100.0, + selfEnergy = 100.0, ramming = false) + if PowerPolicyEnabled: + # Cold gun: no data at all -> pEst <= pRef vacuously -> mid cap, but the + # preferred bin is already 1.0, so the fired power stays 1.0. + check "tracker: cold gun gets the mid cap but keeps its p=1.0 preference", + pCold == 1.0 and dCold.reason == prBelowAvg + else: + check "tracker (control): cold gun keeps its p=1.0 preference", pCold == 1.0 + +proc testTrackerRamExempt() = + var t = initTracker(1) + seedWindow(t, 7, 0, 3, 50, 50) + let (_, _, p, d) = t.selectShotPolicy(7, 0, dist = 500.0, + selfEnergy = 5.0, ramming = true) + check "tracker: ramming exempts the caps", p == 3.0 and d.reason == prRam + +# ── driver ─────────────────────────────────────────────────────────────────── + +testFar() +testLowEnergy() +testBelowAverage() +testAboveAverageFull() +testRamExempt() +testCapNeverRaises() +testPrecedence() +testDisabledUncapped() +testEnvFlag() +testBinIndex() +testReasonNames() +testTrackerAboveAverage() +testTrackerBelowAverage() +testTrackerFarAndLowEnergy() +testTrackerColdAndEmptyBin() +testTrackerRamExempt() + +if failures > 0: + echo "\n", failures, " check(s) FAILED" + quit(1) +echo "\nAll power-policy checks passed."