chore: rename libs→common_libs, all bot dirs to _garage suffix, fix all path refs

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-08-27 18:18:41 +02:00
parent f8c0c871c6
commit b509195ee9
832 changed files with 4967 additions and 368 deletions
BIN
View File
Binary file not shown.
+11
View File
@@ -0,0 +1,11 @@
{
"name": "QBot",
"version": "0.1.0",
"authors": ["Davide Cappellini"],
"description": "Tabular Q-learning bot with chained action decomposition",
"homepage": "",
"countryCodes": ["IT"],
"gameTypes": ["classic", "1v1"],
"platform": "Nim",
"programmingLang": "Nim"
}
+369
View File
@@ -0,0 +1,369 @@
# QBot — gun-only Q-learning prototype, v2.
# Chained action decomposition: T1 (aim offset, 7 actions) + T2 (fire, 3 actions).
# Bullet tracking ring buffer: Q-updates happen ONLY on bullet resolution (hit/wall).
# 120 states × (7 + 3) actions = 1,200 Q-table entries total.
# All debug output goes to stderr. printToStdOut is never called.
#
# Bullet tracking protocol:
# 1. run(): setFire succeeds → push BulletRecord with valid=false, id=-1 ("pending")
# 2. onBulletFired(): stamp the oldest pending record with the real bulletId, set valid=true
# 3. onBulletHit / onBulletHitWall(): resolveBullet() finds by id, does Q-update, clears slot
import std/[random, math, os, streams, strutils]
import tankroyale_botapi
import radar_lock/radar_lock as radar_lock
# ── Constants ──────────────────────────────────────────────────────────────────
const botJsonPath = currentSourcePath().parentDir / "QBot.json"
const qtablesDir = currentSourcePath().parentDir / "qtables"
const
# State features
N_AIM_ERR = 5 # gun aim error: far-left / left / on-target / right / far-right
N_DIST = 3 # close / mid / far
N_LAT_SPD = 4 # enemy lateral speed: fast-left / slow-left / slow-right / fast-right
N_HEAT = 2 # gun ready / cooling
N_STATES = N_AIM_ERR * N_DIST * N_LAT_SPD * N_HEAT # 120
# Action tables
N_AIM_ACT = 7 # aim offset actions
N_FIRE_ACT = 3 # fire decision actions
ALPHA = 0.3
# No GAMMA: contextual bandit (γ=0). Each shot is independent; no future state.
# Q[s,a] converges to expected reward of action a in state s.
EPS_INIT = 0.5
EPS_DECAY = 0.995
EPS_FLOOR = 0.03
Q_INIT = 0.0 # neutral init; signal from +1 hit / -0.2 miss drives learning
# Bullet tracking ring buffer — max bullets in flight simultaneously
BULLET_BUF_CAP = 5
# Aim offset per action (degrees relative to direct bearing to enemy)
const AIM_OFFSETS: array[N_AIM_ACT, float] = [-12.0, -6.0, -3.0, 0.0, 3.0, 6.0, 12.0]
# Fire power per action (0.0 = don't fire)
const FIRE_POWERS: array[N_FIRE_ACT, float] = [0.0, 1.0, 3.0]
# ── Bullet tracking ───────────────────────────────────────────────────────────
type BulletRecord = object
bulletId: int # -1 = pending (waiting for onBulletFired to stamp real ID)
state: int
aimAct: int
fireAct: int
valid: bool # true = slot occupied (either pending or id-stamped)
# ── Bot type ───────────────────────────────────────────────────────────────────
type QBot = ref object of Bot
# Enemy tracking
enemyX, enemyY: float
enemyDir: float
enemySpeed: float
hasContact: bool
ticksSinceScan: int
# Q-tables (flat arrays)
# T1: aim offset — index [state * N_AIM_ACT + aimAction]
qt1: array[N_STATES * N_AIM_ACT, float64]
# T2: fire decision — index [state * N_FIRE_ACT + fireAction]
qt2: array[N_STATES * N_FIRE_ACT, float64]
# Bullet tracking ring buffer
bullets: array[BULLET_BUF_CAP, BulletRecord]
bulletHead: int # next write position (ring)
# Round stats
totalRounds: int
epsilon: float
roundHits: int
roundMisses: int
roundShots: int
# ── State discretization ──────────────────────────────────────────────────────
proc discretizeState(gunAimError, dist, lateralSpeed, gunHeat: float): int {.inline.} =
# gunAimError: signed angle from gun to direct enemy bearing (negative = gun left of enemy)
let aimBin =
if gunAimError < -20.0: 0
elif gunAimError < -5.0: 1
elif gunAimError <= 5.0: 2
elif gunAimError <= 20.0: 3
else: 4
let distBin =
if dist < 200.0: 0 elif dist <= 500.0: 1 else: 2
# lateralSpeed: enemy velocity component perpendicular to our line of sight
let latBin =
if lateralSpeed < -3.0: 0
elif lateralSpeed < 0.0: 1
elif lateralSpeed <= 3.0: 2
else: 3
# heat ready: threshold slightly above 0 since fire 3.0 generates heat 1.6
let heatBin = if gunHeat <= 0.2: 0 else: 1
aimBin * (N_DIST * N_LAT_SPD * N_HEAT) +
distBin * (N_LAT_SPD * N_HEAT) +
latBin * N_HEAT +
heatBin
# ── Q-learning helpers ────────────────────────────────────────────────────────
proc selectAimAction(bot: QBot, state: int): int =
if rand(1.0) < bot.epsilon:
return rand(N_AIM_ACT - 1)
var best = -1e30
var bestA = 0
for a in 0 ..< N_AIM_ACT:
let v = bot.qt1[state * N_AIM_ACT + a]
if v > best:
best = v
bestA = a
bestA
proc selectFireAction(bot: QBot, state: int): int =
if rand(1.0) < bot.epsilon:
return rand(N_FIRE_ACT - 1)
var best = -1e30
var bestA = 0
for a in 0 ..< N_FIRE_ACT:
let v = bot.qt2[state * N_FIRE_ACT + a]
if v > best:
best = v
bestA = a
bestA
proc updateQT1(bot: QBot, state, action: int, reward: float) {.inline.} =
# γ=0 bandit update: Q ← Q + α*(r - Q)
let idx = state * N_AIM_ACT + action
bot.qt1[idx] += ALPHA * (reward - bot.qt1[idx])
proc updateQT2(bot: QBot, state, action: int, reward: float) {.inline.} =
let idx = state * N_FIRE_ACT + action
bot.qt2[idx] += ALPHA * (reward - bot.qt2[idx])
# ── Bullet ring buffer ────────────────────────────────────────────────────────
proc pushPendingBullet(bot: QBot, state, aimAct, fireAct: int) =
# Push a pending record; onBulletFired will stamp the real ID.
bot.bullets[bot.bulletHead] = BulletRecord(
bulletId: -1, state: state, aimAct: aimAct, fireAct: fireAct, valid: true)
bot.bulletHead = (bot.bulletHead + 1) mod BULLET_BUF_CAP
proc stampBulletId(bot: QBot, realId: int) =
# Find the oldest pending (id=-1) record and stamp it with the real bullet ID.
# Search backwards from bulletHead (most recent push is just before head).
for i in 1 .. BULLET_BUF_CAP:
let idx = (bot.bulletHead - i + BULLET_BUF_CAP) mod BULLET_BUF_CAP
if bot.bullets[idx].valid and bot.bullets[idx].bulletId == -1:
bot.bullets[idx].bulletId = realId
return
proc resolveBullet(bot: QBot, bulletId: int, reward: float) =
for i in 0 ..< BULLET_BUF_CAP:
if bot.bullets[i].valid and bot.bullets[i].bulletId == bulletId:
let rec = bot.bullets[i]
bot.bullets[i].valid = false
updateQT1(bot, rec.state, rec.aimAct, reward)
updateQT2(bot, rec.state, rec.fireAct, reward)
return
# Not found — bullet from previous round or missed stamp; silently ignore.
# ── Persistence ───────────────────────────────────────────────────────────────
proc saveQTable(bot: QBot) =
createDir(qtablesDir)
let s = newFileStream(qtablesDir / "gun_q.bin", fmWrite)
if s == nil: return
for v in bot.qt1: s.write(v)
for v in bot.qt2: s.write(v)
s.close()
writeFile(qtablesDir / "gun_meta.txt",
"rounds=" & $bot.totalRounds & "\n" &
"epsilon=" & $bot.epsilon & "\n")
proc loadQTable(bot: QBot) =
let path = qtablesDir / "gun_q.bin"
if not fileExists(path): return
let s = newFileStream(path, fmRead)
if s == nil: return
var i = 0
while not s.atEnd and i < bot.qt1.len:
bot.qt1[i] = s.readFloat64()
inc i
i = 0
while not s.atEnd and i < bot.qt2.len:
bot.qt2[i] = s.readFloat64()
inc i
s.close()
let mp = qtablesDir / "gun_meta.txt"
if not fileExists(mp): return
for line in lines(mp):
if line.startsWith("rounds="):
bot.totalRounds = parseInt(line[7..^1])
elif line.startsWith("epsilon="):
bot.epsilon = parseFloat(line[8..^1])
# ── Logging ───────────────────────────────────────────────────────────────────
proc logRoundEnd(bot: QBot) =
let hitRate = if bot.roundShots > 0:
float(bot.roundHits) / float(bot.roundShots)
else: 0.0
stderr.writeLine("[QBot] round=" & $bot.totalRounds &
" shots=" & $bot.roundShots &
" hits=" & $bot.roundHits &
" hitRate=" & formatFloat(hitRate, ffDecimal, 3) &
" eps=" & formatFloat(bot.epsilon, ffDecimal, 3))
# Top 5 T1 Q-values
type QEntry = tuple[v: float; s, a: int]
var top: array[5, QEntry]
var topN = 0
for s in 0 ..< N_STATES:
for a in 0 ..< N_AIM_ACT:
let v = bot.qt1[s * N_AIM_ACT + a]
if topN < 5:
top[topN] = (v, s, a)
inc topN
else:
var minIdx = 0
for j in 1 ..< 5:
if top[j].v < top[minIdx].v: minIdx = j
if v > top[minIdx].v:
top[minIdx] = (v, s, a)
# Sort descending (simple insertion sort over 5 elements)
for i in 0 ..< topN - 1:
for j in i + 1 ..< topN:
if top[j].v > top[i].v:
let tmp = top[i]; top[i] = top[j]; top[j] = tmp
stderr.writeLine("[QBot] top T1 Q-values:")
for i in 0 ..< topN:
stderr.writeLine(" s=" & $top[i].s & " aimAct=" & $top[i].a &
" offset=" & formatFloat(AIM_OFFSETS[top[i].a], ffDecimal, 1) & "deg" &
" q=" & formatFloat(top[i].v, ffDecimal, 4))
# ── Event handlers ────────────────────────────────────────────────────────────
method onScannedBot*(bot: QBot, e: ScannedBotEvent) =
bot.enemyX = e.x
bot.enemyY = e.y
bot.enemyDir = e.direction
bot.enemySpeed = e.speed
bot.hasContact = true
bot.ticksSinceScan = 0
method onBulletFired*(bot: QBot, e: BulletFiredEvent) =
# Stamp the pending bullet record with the real ID assigned by the server.
bot.stampBulletId(e.bullet.bulletId)
method onBulletHit*(bot: QBot, e: BulletHitBotEvent) =
if e.victimId != getMyId():
inc bot.roundHits
bot.resolveBullet(e.bullet.bulletId, +1.0)
method onBulletHitWall*(bot: QBot, e: BulletHitWallEvent) =
if e.bullet.ownerId == getMyId():
inc bot.roundMisses
bot.resolveBullet(e.bullet.bulletId, -0.2)
method onRoundStarted*(bot: QBot, e: RoundStartedEvent) =
setAdjustGunForBodyTurn(true)
setAdjustRadarForBodyTurn(true)
setAdjustRadarForGunTurn(true)
radar_lock.init()
bot.hasContact = false
bot.roundHits = 0
bot.roundMisses = 0
bot.roundShots = 0
bot.ticksSinceScan = 0
# Invalidate bullet buffer between rounds
for i in 0 ..< BULLET_BUF_CAP: bot.bullets[i].valid = false
bot.bulletHead = 0
setTargetSpeed(0.0)
setTurnRate(0.0)
bot.epsilon = max(EPS_FLOOR, bot.epsilon * EPS_DECAY)
method onRoundEnded*(bot: QBot, e: RoundEndedEventForBot) =
inc bot.totalRounds
# Bullets still in flight at round end → treat as misses
for i in 0 ..< BULLET_BUF_CAP:
if bot.bullets[i].valid:
updateQT1(bot, bot.bullets[i].state, bot.bullets[i].aimAct, -0.2)
updateQT2(bot, bot.bullets[i].state, bot.bullets[i].fireAct, -0.2)
bot.bullets[i].valid = false
logRoundEnd(bot)
method onGameStarted*(bot: QBot, e: GameStartedEventForBot) =
loadQTable(bot)
randomize()
method onGameEnded*(bot: QBot, e: GameEndedEventForBot) =
saveQTable(bot)
stderr.writeLine("[QBot] game ended rounds=" & $bot.totalRounds &
" eps=" & formatFloat(bot.epsilon, ffDecimal, 3))
# ── Main loop ─────────────────────────────────────────────────────────────────
method run*(bot: QBot) =
while isRunning():
setTargetSpeed(0.0)
setTurnRate(0.0)
inc bot.ticksSinceScan
if bot.ticksSinceScan > 5:
bot.hasContact = false
if not bot.hasContact:
setRadarTurnRate(45.0)
go()
continue
let myX = getX()
let myY = getY()
let gunDir = getGunDirection()
let dist = hypot(myX - bot.enemyX, myY - bot.enemyY)
# Absolute bearing from us to enemy
let bearingToEnemy = directionTo(myX, myY, bot.enemyX, bot.enemyY)
# Gun aim error: how far off our gun is from direct bearing to enemy
let gunAimError = normalizeRelativeAngle(bearingToEnemy - gunDir)
# Lateral speed: enemy velocity perpendicular to our line of sight
let lateralSpeed = bot.enemySpeed * sin(degToRad(bot.enemyDir - bearingToEnemy))
let state = discretizeState(gunAimError, dist, lateralSpeed, getGunHeat())
# Select actions (no per-tick Q-update — updates happen only on bullet resolution)
let aimAct = bot.selectAimAction(state)
let fireAct = bot.selectFireAction(state)
# Compute gun turn: aim at enemy bearing + selected offset, clamped to ±20 deg/tick
let aimOffset = AIM_OFFSETS[aimAct]
let desiredDelta = normalizeRelativeAngle(bearingToEnemy + aimOffset - gunDir)
setGunTurnRate(desiredDelta.clamp(-20.0, 20.0))
# Fire if the action calls for it and the gun is ready
let firePower = FIRE_POWERS[fireAct]
if firePower > 0.0 and getGunHeat() <= 0.0:
if setFire(firePower):
inc bot.roundShots
# Push a pending record; onBulletFired (fired next tick) stamps the real ID
bot.pushPendingBullet(state, aimAct, fireAct)
# Radar lock
setRadarTurnRate(radar_lock.doRadar(getRadarDirection(), bearingToEnemy))
go()
# ── Entry point ───────────────────────────────────────────────────────────────
when isMainModule:
createDir(qtablesDir)
var bot = QBot(epsilon: EPS_INIT)
for i in 0 ..< bot.qt1.len: bot.qt1[i] = Q_INIT
for i in 0 ..< bot.qt2.len: bot.qt2[i] = Q_INIT
start(bot, botJsonPath)
+11
View File
@@ -0,0 +1,11 @@
# Package
version = "0.1.0"
author = "Davide Cappellini"
description = "Tabular Q-learning Tank Royale bot — chained action decomposition prototype"
license = "MIT"
bin = @["QBot"]
# Dependencies
requires "nim >= 2.0.0"
# tankroyale_botapi and radar_lock are vendored in-tree (common_libs/) and wired via
# config.nims --path; no nimble dependency so builds never touch ~/.nimble/pkgs2.
+3
View File
@@ -0,0 +1,3 @@
#!/bin/sh
cd "$(dirname "$0")"
exec ./QBot 2>> /tmp/qbot_stderr.log
+1
View File
@@ -0,0 +1 @@
--path:"../common_libs"