diff --git a/SNNBot_garage/src/SNNBot.nim b/SNNBot_garage/src/SNNBot.nim index 99f7f25..05f2dbf 100644 --- a/SNNBot_garage/src/SNNBot.nim +++ b/SNNBot_garage/src/SNNBot.nim @@ -9,12 +9,12 @@ # WAITING → aimTo() each tick; when error < 2° → EVALUATE # EVALUATE → measure error, compute SuperSpike/reservoir update, log, → DECIDE -import std/[math, random, os, strutils] +import std/[math, random, os, strutils, bitops] import robocode_tankroyale_botapi import radar_lock/radar_lock as radar_lock import reservoir -const USE_RESERVOIR* = true +const USE_RESERVOIR* = true # ponytail: kept for SNN path fallback; remove when binary aimer is validated # ── Constants ────────────────────────────────────────────────────────────────── @@ -198,7 +198,7 @@ type SNNBot = ref object of Bot snn: SNN - res: Reservoir + res: BinaryAimer phase: Phase targetAngle: float # SNN output (absolute bearing) enemyBearing: float # last known enemy bearing @@ -214,6 +214,7 @@ type velDirDeg: float64 # velocity direction (degrees) from last scan delta velSpeed: float64 # speed (units/tick) from last scan delta lastDecideGunDir: float # gun heading captured at DECIDE time for EVALUATE + lastBinInput: BitVec80 # DECIDE-time binary input, reused in EVALUATE # ── aimTo helper ────────────────────────────────────────────────────────────── @@ -375,7 +376,7 @@ method onRoundStarted*(bot: SNNBot, e: RoundStartedEvent) = method onGameStarted*(bot: SNNBot, e: GameStartedEventForBot) = initSNN(bot.snn) - bot.res = initReservoir(42) + bot.res = initAimer() # ── Reservoir helpers ───────────────────────────────────────────────────────── @@ -413,7 +414,8 @@ method run*(bot: SNNBot) = let binInput = toBitVec80(inputs) let winBin = bot.res.forward(binInput) let aimAngle = bot.res.interpolatedAngle(winBin) - bot.targetAngle = gunDir + aimAngle + bot.lastBinInput = binInput + bot.targetAngle = gunDir + aimAngle bot.lastDecideGunDir = gunDir echo "RES tick=" & $bot.tick & " bin=" & $winBin & " aim=" & formatFloat(aimAngle, ffDecimal, 1) else: @@ -460,18 +462,23 @@ method run*(bot: SNNBot) = let absBearing = directionTo(myX, myY, futureX, futureY) let targetRel = normalizeRelativeAngle(absBearing - bot.lastDecideGunDir) when USE_RESERVOIR: - # state/scores already set by DECIDE's forward(); re-forwarding would corrupt them let winBin = block: var best = 0; var bestS = -1 for k in 0 ..< N_BINS: if bot.res.scores[k] > bestS: bestS = bot.res.scores[k]; best = k best let aimAngle = bot.res.interpolatedAngle(winBin) - bot.res.learn(targetRel) + bot.res.learn(bot.lastBinInput, targetRel) + let correctBin = angleToBin(targetRel) + let readoutPopCorrect = block: + var c = 0 + for w in 0 ..< WORDS_IN: c += popcount(bot.res.readout[correctBin][w]).int + c echo "RES tick=" & $bot.tick & " bin=" & $winBin & " aim=" & formatFloat(aimAngle, ffDecimal, 1) & " target=" & formatFloat(targetRel, ffDecimal, 1) & - " err=" & formatFloat(abs(aimAngle - targetRel), ffDecimal, 1) + " err=" & formatFloat(abs(aimAngle - targetRel), ffDecimal, 1) & + " readout_pop_correct=" & $readoutPopCorrect else: let err = abs(normalizeRelativeAngle(gunDir - bot.enemyBearing)) bot.snn.superSpikeUpdate(bot.lastSpikes, bot.lastVSnap, bot.snn.preTrace, targetRel) diff --git a/SNNBot_garage/src/reservoir.nim b/SNNBot_garage/src/reservoir.nim index c7a0c86..e3a0361 100644 --- a/SNNBot_garage/src/reservoir.nim +++ b/SNNBot_garage/src/reservoir.nim @@ -1,157 +1,80 @@ -# ponytail: binary reservoir aimer — prototype; if reservoir projection is poor, scale RESERVOIR_SIZE to 4096 +# ponytail: direct binary readout; add reservoir back when temporal features matter (step 3) -import std/[algorithm, bitops, math] - -# ── Constants ────────────────────────────────────────────────────────────────── +import std/[bitops, math] const - RESERVOIR_SIZE* = 1024 - N_BINS* = 72 - BIN_WIDTH* = 5.0 - INPUT_BITS* = 80 - SPARSITY_IN = 0.1 - SPARSITY_REC = 0.05 - K_ACTIVE = 50 # ~5% of RESERVOIR_SIZE fire per tick - # ponytail: K_ACTIVE=50 gives ~5% sparsity; increase if readout can't discriminate, decrease if patterns overlap too much - -# ── Types ────────────────────────────────────────────────────────────────────── + INPUT_BITS* = 80 + N_BINS* = 72 + BIN_WIDTH* = 5.0 + WORDS_IN* = 2 # 80 bits → 2 × uint64 (128 bits, only 80 used) type - BitVec80* = array[2, uint64] # 128 bits allocated, lower 80 used - BitVec* = array[16, uint64] # 1024 bits + BitVec80* = array[WORDS_IN, uint64] - Reservoir* = object - wIn: array[RESERVOIR_SIZE, BitVec80] - wRec: array[RESERVOIR_SIZE, BitVec] - state: BitVec - readout: array[N_BINS, BitVec] + BinaryAimer* = object + readout*: array[N_BINS, BitVec80] # 72 bins × 80-bit weight vectors scores*: array[N_BINS, int] - rngState: uint64 -# ── PRNG ─────────────────────────────────────────────────────────────────────── - -proc nextRand(r: var Reservoir): uint64 {.inline.} = - var x = r.rngState - x = x xor (x shl 13) - x = x xor (x shr 7) - x = x xor (x shl 17) - r.rngState = x - return x - -proc sparseBits(r: var Reservoir, density: float): uint64 = - ## uint64 with approximately density*64 bits set via repeated AND. - ## density=0.1 → ~1 AND needed for 1/2, but we use a direct Bernoulli approach. - ## ponytail: simple loop over 64 bits; replace with table-lookup if init is hot - result = 0'u64 - let thresh = uint64(density * float(high(uint64))) - for bit in 0 ..< 64: - if r.nextRand() < thresh: - result = result or (1'u64 shl bit) - -# ── Init ─────────────────────────────────────────────────────────────────────── - -proc initReservoir*(seed: int): Reservoir = - result.rngState = uint64(seed) or 1'u64 # avoid zero state - - for i in 0 ..< RESERVOIR_SIZE: - # Input masks: 80 bits across two uint64s (word 0: bits 0-63, word 1: bits 64-79) - result.wIn[i][0] = sparseBits(result, SPARSITY_IN) - # Only bits 0-15 of word 1 are meaningful (global bits 64-79) - result.wIn[i][1] = sparseBits(result, SPARSITY_IN) and 0x0000_0000_0000_FFFF'u64 - - for w in 0 ..< 16: - result.wRec[i][w] = sparseBits(result, SPARSITY_REC) - - # readout and state are zero-initialized by default - -# ── Forward ──────────────────────────────────────────────────────────────────── - -proc forward*(r: var Reservoir, input: BitVec80): int = - # Compute activation scores for all neurons - var activations: array[RESERVOIR_SIZE, int] - for i in 0 ..< RESERVOIR_SIZE: - let inScore = popcount(input[0] and r.wIn[i][0]).int + - popcount(input[1] and r.wIn[i][1]).int - var recScore = 0 - for w in 0 ..< 16: - recScore += popcount(r.state[w] and r.wRec[i][w]).int - activations[i] = inScore + recScore - - # k-WTA: find K-th highest activation via sort on a copy - var sorted = activations - sort(sorted, order = SortOrder.Descending) - let kThreshold = sorted[min(K_ACTIVE - 1, RESERVOIR_SIZE - 1)] - - # Fire exactly K_ACTIVE neurons (tie-break: first in index order) - var newState: BitVec - var count = 0 - for i in 0 ..< RESERVOIR_SIZE: - if activations[i] >= kThreshold and count < K_ACTIVE: - newState[i shr 6] = newState[i shr 6] or (1'u64 shl (i and 63)) - inc count - - r.state = newState +proc initAimer*(): BinaryAimer = + # All readout weights start at zero — no bin preferred + result = BinaryAimer() +proc forward*(a: var BinaryAimer, input: BitVec80): int = + ## Score each bin via popcount(input AND weights), return best bin var bestBin = 0 var bestScore = -1 for k in 0 ..< N_BINS: var score = 0 - for w in 0 ..< 16: - score += popcount(r.state[w] and r.readout[k][w]) - r.scores[k] = score + for w in 0 ..< WORDS_IN: + score += popcount(input[w] and a.readout[k][w]).int + a.scores[k] = score if score > bestScore: bestScore = score bestBin = k - - return bestBin - -# ── Angle helpers ────────────────────────────────────────────────────────────── - -proc binToAngle*(bin: int): float = - ## Bin 0 = -180°, Bin 36 = 0°, Bin 71 = +175° - result = -180.0 + float(bin) * BIN_WIDTH + result = bestBin proc angleToBin*(angle: float): int = - ## angle in -180..+180, map to bin 0..71 var a = angle - if a < -180.0: a += 360.0 - if a >= 180.0: a -= 360.0 - result = int((a + 180.0) / BIN_WIDTH) mod N_BINS + while a < -180.0: a += 360.0 + while a >= 180.0: a -= 360.0 + result = clamp(int((a + 180.0) / BIN_WIDTH), 0, N_BINS - 1) -proc interpolatedAngle*(r: Reservoir, winnerBin: int): float = - ## Weighted circular centroid of winner ± 1 bins for sub-5° precision. +proc binToAngle*(bin: int): float = + result = -180.0 + (float(bin) + 0.5) * BIN_WIDTH # bin center + +proc interpolatedAngle*(a: BinaryAimer, winnerBin: int): float = + ## Weighted centroid of winner + neighbors for sub-bin precision let left = (winnerBin - 1 + N_BINS) mod N_BINS let right = (winnerBin + 1) mod N_BINS - let sW = float(r.scores[winnerBin]) - let sL = float(r.scores[left]) - let sR = float(r.scores[right]) + let sW = float(max(a.scores[winnerBin], 1)) + let sL = float(max(a.scores[left], 0)) + let sR = float(max(a.scores[right], 0)) let total = sW + sL + sR - if total == 0.0: return binToAngle(winnerBin) - let aW = degToRad(binToAngle(winnerBin)) - let aL = degToRad(binToAngle(left)) - let aR = degToRad(binToAngle(right)) - let sinAvg = (sW * sin(aW) + sL * sin(aL) + sR * sin(aR)) / total - let cosAvg = (sW * cos(aW) + sL * cos(aL) + sR * cos(aR)) / total + let aW = binToAngle(winnerBin) + let aL = binToAngle(left) + let aR = binToAngle(right) + # Circular mean + let sinAvg = (sW * sin(degToRad(aW)) + sL * sin(degToRad(aL)) + sR * sin(degToRad(aR))) / total + let cosAvg = (sW * cos(degToRad(aW)) + sL * cos(degToRad(aL)) + sR * cos(degToRad(aR))) / total result = radToDeg(arctan2(sinAvg, cosAvg)) -# ── Learning ─────────────────────────────────────────────────────────────────── - -proc learn*(r: var Reservoir, correctAngle: float) = +proc learn*(a: var BinaryAimer, input: BitVec80, correctAngle: float) = + ## WTA Hebbian: reinforce correct bin, punish worst wrong bin let correctBin = angleToBin(correctAngle) - # Reinforce correct bin - for w in 0 ..< 16: - r.readout[correctBin][w] = r.readout[correctBin][w] or r.state[w] + # Reinforce: OR input into correct bin + for w in 0 ..< WORDS_IN: + a.readout[correctBin][w] = a.readout[correctBin][w] or input[w] - # Punish highest-scoring wrong bin + # Find highest-scoring WRONG bin var worstBin = -1 var worstScore = -1 for k in 0 ..< N_BINS: - if k != correctBin and r.scores[k] > worstScore: - worstScore = r.scores[k] + if k != correctBin and a.scores[k] > worstScore: + worstScore = a.scores[k] worstBin = k - if worstBin >= 0: - for w in 0 ..< 16: - r.readout[worstBin][w] = r.readout[worstBin][w] and (not r.state[w]) - # ponytail: decay removed; add back if readout weights saturate (all scores converge to same value) + # Punish: AND NOT input from worst wrong bin + if worstBin >= 0: + for w in 0 ..< WORDS_IN: + a.readout[worstBin][w] = a.readout[worstBin][w] and (not input[w])