feat(guns): scale-aware power selection (+52% damage); TM classifier gun built, measured, DISABLED
TASK 2 - power selection, a clear win. bestPower used an ABSOLUTE MinHitRate = 0.40 bar. Measured per-bin virtual rates (rolling-100 fraction) show no bin ever clears 40%, so 11 of 14 guns were stuck at bin 0 (power 1.0) even where higher bins were comparable: Linear p1.0 44% p1.5 39% p2.0 30% p3.0 29% old bin 0 -> new bin 3 Accel p1.0 44% p1.5 40% p2.0 26% p3.0 29% old bin 1 -> new bin 3 Pattern p1.0 50% p1.5 40% p2.0 27% p3.0 12% old bin 1 -> new bin 2 Replaced with a scale-aware PowerBarFrac = 0.50 (a dimensionless FRACTION of the gun's own best bin rate). 13 of 14 selections now pick heavier bullets. Real effect vs DrussGT (8 rounds x 3 runs): hit rate unchanged (7.56% -> 7.47%) but damage dealt +52% (157 -> 239 per run) and rounds end faster. Same accuracy, half the shots, half again more damage. TASK 1 - the TM pattern-classifier gun does NOT earn its slot. It was built as a mixture of experts with a corrected-Granmo TM as a multi-class gate over HeadOn/Linear/Circular/WallBounce/Accel, labelled by which expert's prediction was closest to the actual enemy position (an exact, supervised, per-shot label - no delayed credit). Offline it loses to the best of its OWN experts on essentially every fixture, and against DrussGT it cost real performance: baseline (path+relative) 7.56% real hit rate, damage 157 + power fix 7.47%, damage 239 + power fix + TM gun 5.59%, damage 133 The gun was selected on 806 ticks and fired 24 real shots at 4.2%. So the tree ships with EnableTmSelector = false: code and wiring kept intact for re-enabling, but it is not in the active rack. Worth recording from the clause dump: the gate DOES latch onto meaningful structure. On energy-threshold-turner, HeadOn's clauses key on the energy bits (the rule's own driving variable) while Circular keys on distance/velocity. So the TM is learning something real and interpretable - it simply cannot beat 'always pick the best expert'. Root cause (INFERRED): the closest-expert label is noisy because several experts are near-tied, and under the path metric the winner varies by power bin while the gate sees one shared per-tick input, so a one-vs-rest gate over a saturated 870-bit clause space has no margin to exploit. (Zero-padding the 2-frame window was tried first and saturated every clause at 256-755 included literals; alternating the two real frames fixed that.) Also factors the corrected feedback into an exported tmLearnDir and exports the encoding/TM primitives; the Tsetlin tests still reproduce the documented mean=13.8 included literals, so the refactor is behaviour-preserving. Verified: 33/33 guard checks, tsetlin tests green, metric checks green, new power-selection guard green (13/14 selections change; relative bar still picks bin 1 and not bin 3 for a [30,25,12,5]% profile), 12/12 offline==online acceptance under the shipped default.
This commit is contained in:
@@ -34,6 +34,9 @@ const
|
||||
runnerJar = "/home/davide/Projects/tank-royale/runner/examples/lib/robocode-tankroyale-runner.jar"
|
||||
|
||||
const TsetlinId = 2
|
||||
const TmSelectorId = 13 ## also stochastic (rand() in Gate choose + TM feedback)
|
||||
|
||||
proc isStochastic(id: int): bool = id == TsetlinId or id == TmSelectorId
|
||||
|
||||
proc lastOnlineRound(path: string): JsonNode =
|
||||
result = nil
|
||||
@@ -78,12 +81,12 @@ proc main() =
|
||||
let reports = replayFixture(fx, buildAllGunDrivers(), liveActual = true)
|
||||
|
||||
# Map online stats by gun id.
|
||||
var onShots: array[13, int]
|
||||
var onHits: array[13, int]
|
||||
var onNames: array[13, string]
|
||||
var onShots: array[14, int]
|
||||
var onHits: array[14, int]
|
||||
var onNames: array[14, string]
|
||||
for g in online["guns"]:
|
||||
let id = g["id"].getInt()
|
||||
if id >= 0 and id < 13:
|
||||
if id >= 0 and id < 14:
|
||||
onShots[id] = g["vShots"].getInt()
|
||||
onHits[id] = g["vHits"].getInt()
|
||||
onNames[id] = g["name"].getStr()
|
||||
@@ -96,11 +99,11 @@ proc main() =
|
||||
echo "-----------------------------------------------------------------------"
|
||||
var matches = 0
|
||||
var deterministic = 0
|
||||
for id in 0..<13:
|
||||
for id in 0..<14:
|
||||
let r = reports[id]
|
||||
let match = r.hits == onHits[id] and r.shots == onShots[id]
|
||||
var verdict: string
|
||||
if id == TsetlinId:
|
||||
if isStochastic(id):
|
||||
verdict = if match: "MATCH (stochastic)" else: "differs (stochastic, expected)"
|
||||
else:
|
||||
inc deterministic
|
||||
@@ -117,7 +120,7 @@ proc main() =
|
||||
echo "VERDICT: FAIL — offline range does NOT reproduce the live metric."
|
||||
quit(1)
|
||||
echo "VERDICT: PASS — offline == online for all 12 deterministic guns."
|
||||
echo "(Tsetlin is stochastic and is allowed to differ.)"
|
||||
echo "(Tsetlin and TMSelect are stochastic and are allowed to differ.)"
|
||||
|
||||
when isMainModule:
|
||||
main()
|
||||
|
||||
@@ -17,12 +17,16 @@ import guns/displacement
|
||||
import guns/averaged_lead
|
||||
import guns/decay_gf
|
||||
import guns/knn_gun
|
||||
import guns/tm_selector
|
||||
|
||||
proc buildAllGunDrivers*(seed = -1): seq[GunDriver] =
|
||||
## seed >= 0 re-seeds the global RNG after constructing Tsetlin so the
|
||||
## stochastic gun's learning is reproducible for offline runs. (Its
|
||||
## constructor calls randomize(); we override that seed afterwards.)
|
||||
## seed >= 0 re-seeds the global RNG after constructing the stochastic guns
|
||||
## (Tsetlin and the TM selector both call randomize() in their constructors),
|
||||
## so their learning is reproducible for offline runs.
|
||||
##
|
||||
## Order matches ModularBot's gun ids exactly (TMSelect appended at 13).
|
||||
var tsetlin = initTsetlinGun()
|
||||
var tmSelector = initTmSelectorGun()
|
||||
if seed >= 0:
|
||||
randomize(seed)
|
||||
result = @[
|
||||
@@ -39,6 +43,7 @@ proc buildAllGunDrivers*(seed = -1): seq[GunDriver] =
|
||||
makeDriver("AvgLead", initAveragedLeadGun()),
|
||||
makeDriver("DecayGF", initDecayGFGun()),
|
||||
makeDriver("KNN", initKNNGun()),
|
||||
makeDriver("TMSelect", tmSelector),
|
||||
]
|
||||
|
||||
proc makeTsetlinDriver*(seed = -1): tuple[driver: GunDriver, gun: ref TsetlinGun] =
|
||||
@@ -57,3 +62,19 @@ proc makeTsetlinDriver*(seed = -1): tuple[driver: GunDriver, gun: ref TsetlinGun
|
||||
resultCb: proc(e: FeedbackEvent) = g[].onResult(e),
|
||||
readyCb: proc(): bool = g[].isWarmedUp(),
|
||||
)
|
||||
|
||||
proc makeTmSelectorDriver*(seed = -1): tuple[driver: GunDriver, gun: ref TmSelectorGun] =
|
||||
## Same as makeDriver("TMSelect", ...) but keeps a handle to the concrete gun
|
||||
## so a test can inspect its votes / clause interpretability after a replay.
|
||||
let g = new(TmSelectorGun)
|
||||
g[] = initTmSelectorGun()
|
||||
if seed >= 0:
|
||||
randomize(seed)
|
||||
result.gun = g
|
||||
result.driver = GunDriver(
|
||||
name: "TMSelect",
|
||||
predictCb: proc(state: WorldState, bulletSpeed: float): GunPrediction =
|
||||
g[].predict(state, bulletSpeed),
|
||||
resultCb: proc(e: FeedbackEvent) = g[].onResult(e),
|
||||
readyCb: proc(): bool = g[].isWarmedUp(),
|
||||
)
|
||||
|
||||
@@ -0,0 +1,105 @@
|
||||
## Guard + measurement for the scale-aware power bar in `bestPower`.
|
||||
##
|
||||
## Before the fix `bestPower` required an ABSOLUTE virtual hit rate >= MinHitRate
|
||||
## (0.40). On the shipped path-metric scale a gun's per-bin rates sit around
|
||||
## 3-40%, so once every bin had data no bin cleared the bar and bestPower
|
||||
## silently collapsed to bin 0 (power 1.0). The fix compares each bin's rate to
|
||||
## PowerBarFrac * (the same gun's best bin rate) — dimensionless, so it
|
||||
## discriminates at any scale.
|
||||
##
|
||||
## Run: nim c -r common_libs/tests/test_power_selection.nim
|
||||
|
||||
import std/[strformat, math, random, tables, os]
|
||||
import gun_harness/gun_interface
|
||||
import gun_harness/virtual_bullets
|
||||
import gun_harness/offline_range
|
||||
import range_guns
|
||||
|
||||
const fixturesDir = currentSourcePath().parentDir.parentDir.parentDir / "tools" / "fixtures"
|
||||
|
||||
var failures = 0
|
||||
proc check(name: string, ok: bool) =
|
||||
if ok: echo "PASS: ", name
|
||||
else: echo "FAIL: ", name; inc failures
|
||||
|
||||
proc recordHit(fw: var FitnessWindow, hit: bool) =
|
||||
fw.hits[fw.head] = hit
|
||||
fw.head = (fw.head + 1) mod WindowSize
|
||||
inc fw.count
|
||||
|
||||
proc seedFromReport(t: var VirtualTracker, targetId, gunId: int, r: GunReport) =
|
||||
if targetId notin t.fitness:
|
||||
t.fitness[targetId] = newSeq[GunFitness](t.numGuns)
|
||||
var fw = addr t.fitness[targetId][gunId].bins
|
||||
for b in 0..<len(PowerBins):
|
||||
for _ in 0..<r.bins[b].hits: recordHit(fw[][b], true)
|
||||
for _ in 0..<(r.bins[b].shots - r.bins[b].hits): recordHit(fw[][b], false)
|
||||
|
||||
proc oldBestPower(fit: GunFitness): int =
|
||||
## The pre-fix rule, reproduced for the A/B comparison only.
|
||||
var anyObs = false
|
||||
for b in 0..<len(PowerBins):
|
||||
if fit.bins[b].count > 0: anyObs = true
|
||||
if not anyObs: return 0
|
||||
for b in countdown(len(PowerBins) - 1, 0):
|
||||
if fit.bins[b].hitRate() >= MinHitRate or fit.bins[b].count == 0: return b
|
||||
0
|
||||
|
||||
proc main() =
|
||||
# A real closed-loop fixture where every gun's rates sit below the old 40%
|
||||
# absolute bar — the regime the bug lives in. Fall back to a synthetic fixture
|
||||
# if the committed capture is missing.
|
||||
let realPath = fixturesDir / "drussgt_vs_crazy.jsonl"
|
||||
let fx = if fileExists(realPath): loadFixture(realPath)
|
||||
else: synthesizeByName("energy-threshold-turner")
|
||||
echo "fixture: ", fx.meta.adversary, " (source=", fx.meta.source, ")"
|
||||
let reports = replayFixture(fx, buildAllGunDrivers(seed = 1), metric = bmPath)
|
||||
|
||||
var t = initTracker(reports.len)
|
||||
let tid = fx.enemyId
|
||||
for gi, r in reports:
|
||||
seedFromReport(t, tid, gi, r)
|
||||
|
||||
echo "per-gun per-bin virtual hit rate (measured) and selected bin (old | new):"
|
||||
var changed = 0
|
||||
var anyOldFellToZero = false
|
||||
for gi, r in reports:
|
||||
let fit = t.fitnessFor(tid)
|
||||
var rates = ""
|
||||
for b in 0..<len(PowerBins):
|
||||
rates.add fmt" p{PowerBins[b]:.1f}={fit[gi].bins[b].hitRate()*100:4.1f}%({r.bins[b].shots})"
|
||||
let ob = oldBestPower(fit[gi])
|
||||
let (nb, _) = t.bestPower(gi, tid)
|
||||
if ob != nb: inc changed
|
||||
# The bug: old rule returns bin 0 even though a higher bin is as good/better.
|
||||
if ob == 0 and nb > 0: anyOldFellToZero = true
|
||||
echo fmt" {r.name:<11}{rates} old=bin{ob} new=bin{nb}"
|
||||
|
||||
echo ""
|
||||
echo fmt"selections changed by the fix: {changed}/{reports.len}"
|
||||
check "the scale-aware bar changes at least one gun's power selection",
|
||||
changed > 0
|
||||
check "at least one gun the old absolute 0.40 bar sent to power 1.0 now uses a heavier bullet",
|
||||
anyOldFellToZero
|
||||
|
||||
# Relative bar still picks the best bin when it is the lowest power (no
|
||||
# pathological power-3 bias).
|
||||
block:
|
||||
var t2 = initTracker(1)
|
||||
# bin0 30%, bin1 25%, bin2 12%, bin3 5% -> bar=15%, only bins 0 and 1 clear
|
||||
# it; the HIGHEST acceptable is bin 1, not bin 0 and not bin 3.
|
||||
seedFromReport(t2, 7, 0, GunReport(name: "x", bins: [
|
||||
BinStat(shots: 100, hits: 30), BinStat(shots: 100, hits: 25),
|
||||
BinStat(shots: 100, hits: 12), BinStat(shots: 100, hits: 5)]))
|
||||
let (b, p) = t2.bestPower(0, 7)
|
||||
echo fmt" synthetic [30,25,12,5]% -> bin {b} (power {p})"
|
||||
check "relative bar picks the highest bin clearing 50% of the best (bin 1)",
|
||||
b == 1
|
||||
|
||||
if failures > 0:
|
||||
echo "\n", failures, " check(s) FAILED"
|
||||
quit(1)
|
||||
echo "\nAll power-selection checks passed."
|
||||
|
||||
when isMainModule:
|
||||
main()
|
||||
@@ -0,0 +1,37 @@
|
||||
## TM selector gun checks + interpretability dump.
|
||||
## Run: nim c -r common_libs/tests/test_tm_selector.nim
|
||||
|
||||
import std/[strformat, math, random]
|
||||
import gun_harness/offline_range
|
||||
import range_guns
|
||||
import guns/tm_selector
|
||||
import guns/head_on
|
||||
import guns/linear
|
||||
import guns/circular
|
||||
import guns/wall_bounce
|
||||
import guns/accel_predictor
|
||||
|
||||
proc main() =
|
||||
let names = ["decel-before-turn", "energy-threshold-turner", "oscillator", "random-walk"]
|
||||
for name in names:
|
||||
let fx = synthesizeByName(name)
|
||||
let (drv, gun) = makeTmSelectorDriver(seed = 1)
|
||||
let reps = replayFixture(fx, @[
|
||||
makeDriver("HeadOn", HeadOnGun()),
|
||||
makeDriver("Linear", LinearGun()),
|
||||
makeDriver("Circular", CircularGun()),
|
||||
makeDriver("WallBounce", initWallBounceGun()),
|
||||
makeDriver("Accel", initAccelGun()),
|
||||
drv], metric = bmPath)
|
||||
echo "=== ", name
|
||||
for r in reps: echo " ", formatReportRow(r)
|
||||
let st = gun[].selectorClauseStats()
|
||||
var wc = ""
|
||||
for c in 0..<N_EXPERTS: wc.add fmt" {ExpertNames[c]}={gun[].winCount[c]}"
|
||||
echo fmt" TMSelect trainCalls={gun[].trainCalls} traceMisses={gun[].traceMisses} " &
|
||||
fmt"clauses active={st.nActive}/{st.nClauses} meanIncl={st.meanIncluded:.1f}"
|
||||
echo " winner counts:", wc
|
||||
echo gun[].describeClauses(topN = 5)
|
||||
|
||||
when isMainModule:
|
||||
main()
|
||||
Reference in New Issue
Block a user