chore: rename libs→common_libs, all bot dirs to _garage suffix, fix all path refs
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Executable
BIN
Binary file not shown.
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"name": "QBot",
|
||||
"version": "0.1.0",
|
||||
"authors": ["Davide Cappellini"],
|
||||
"description": "Tabular Q-learning bot with chained action decomposition",
|
||||
"homepage": "",
|
||||
"countryCodes": ["IT"],
|
||||
"gameTypes": ["classic", "1v1"],
|
||||
"platform": "Nim",
|
||||
"programmingLang": "Nim"
|
||||
}
|
||||
@@ -0,0 +1,369 @@
|
||||
# QBot — gun-only Q-learning prototype, v2.
|
||||
# Chained action decomposition: T1 (aim offset, 7 actions) + T2 (fire, 3 actions).
|
||||
# Bullet tracking ring buffer: Q-updates happen ONLY on bullet resolution (hit/wall).
|
||||
# 120 states × (7 + 3) actions = 1,200 Q-table entries total.
|
||||
# All debug output goes to stderr. printToStdOut is never called.
|
||||
#
|
||||
# Bullet tracking protocol:
|
||||
# 1. run(): setFire succeeds → push BulletRecord with valid=false, id=-1 ("pending")
|
||||
# 2. onBulletFired(): stamp the oldest pending record with the real bulletId, set valid=true
|
||||
# 3. onBulletHit / onBulletHitWall(): resolveBullet() finds by id, does Q-update, clears slot
|
||||
|
||||
import std/[random, math, os, streams, strutils]
|
||||
import tankroyale_botapi
|
||||
import radar_lock/radar_lock as radar_lock
|
||||
|
||||
# ── Constants ──────────────────────────────────────────────────────────────────
|
||||
|
||||
const botJsonPath = currentSourcePath().parentDir / "QBot.json"
|
||||
const qtablesDir = currentSourcePath().parentDir / "qtables"
|
||||
|
||||
const
|
||||
# State features
|
||||
N_AIM_ERR = 5 # gun aim error: far-left / left / on-target / right / far-right
|
||||
N_DIST = 3 # close / mid / far
|
||||
N_LAT_SPD = 4 # enemy lateral speed: fast-left / slow-left / slow-right / fast-right
|
||||
N_HEAT = 2 # gun ready / cooling
|
||||
N_STATES = N_AIM_ERR * N_DIST * N_LAT_SPD * N_HEAT # 120
|
||||
|
||||
# Action tables
|
||||
N_AIM_ACT = 7 # aim offset actions
|
||||
N_FIRE_ACT = 3 # fire decision actions
|
||||
|
||||
ALPHA = 0.3
|
||||
# No GAMMA: contextual bandit (γ=0). Each shot is independent; no future state.
|
||||
# Q[s,a] converges to expected reward of action a in state s.
|
||||
EPS_INIT = 0.5
|
||||
EPS_DECAY = 0.995
|
||||
EPS_FLOOR = 0.03
|
||||
Q_INIT = 0.0 # neutral init; signal from +1 hit / -0.2 miss drives learning
|
||||
|
||||
# Bullet tracking ring buffer — max bullets in flight simultaneously
|
||||
BULLET_BUF_CAP = 5
|
||||
|
||||
# Aim offset per action (degrees relative to direct bearing to enemy)
|
||||
const AIM_OFFSETS: array[N_AIM_ACT, float] = [-12.0, -6.0, -3.0, 0.0, 3.0, 6.0, 12.0]
|
||||
|
||||
# Fire power per action (0.0 = don't fire)
|
||||
const FIRE_POWERS: array[N_FIRE_ACT, float] = [0.0, 1.0, 3.0]
|
||||
|
||||
# ── Bullet tracking ───────────────────────────────────────────────────────────
|
||||
|
||||
type BulletRecord = object
|
||||
bulletId: int # -1 = pending (waiting for onBulletFired to stamp real ID)
|
||||
state: int
|
||||
aimAct: int
|
||||
fireAct: int
|
||||
valid: bool # true = slot occupied (either pending or id-stamped)
|
||||
|
||||
# ── Bot type ───────────────────────────────────────────────────────────────────
|
||||
|
||||
type QBot = ref object of Bot
|
||||
# Enemy tracking
|
||||
enemyX, enemyY: float
|
||||
enemyDir: float
|
||||
enemySpeed: float
|
||||
hasContact: bool
|
||||
ticksSinceScan: int
|
||||
|
||||
# Q-tables (flat arrays)
|
||||
# T1: aim offset — index [state * N_AIM_ACT + aimAction]
|
||||
qt1: array[N_STATES * N_AIM_ACT, float64]
|
||||
# T2: fire decision — index [state * N_FIRE_ACT + fireAction]
|
||||
qt2: array[N_STATES * N_FIRE_ACT, float64]
|
||||
|
||||
# Bullet tracking ring buffer
|
||||
bullets: array[BULLET_BUF_CAP, BulletRecord]
|
||||
bulletHead: int # next write position (ring)
|
||||
|
||||
# Round stats
|
||||
totalRounds: int
|
||||
epsilon: float
|
||||
roundHits: int
|
||||
roundMisses: int
|
||||
roundShots: int
|
||||
|
||||
# ── State discretization ──────────────────────────────────────────────────────
|
||||
|
||||
proc discretizeState(gunAimError, dist, lateralSpeed, gunHeat: float): int {.inline.} =
|
||||
# gunAimError: signed angle from gun to direct enemy bearing (negative = gun left of enemy)
|
||||
let aimBin =
|
||||
if gunAimError < -20.0: 0
|
||||
elif gunAimError < -5.0: 1
|
||||
elif gunAimError <= 5.0: 2
|
||||
elif gunAimError <= 20.0: 3
|
||||
else: 4
|
||||
|
||||
let distBin =
|
||||
if dist < 200.0: 0 elif dist <= 500.0: 1 else: 2
|
||||
|
||||
# lateralSpeed: enemy velocity component perpendicular to our line of sight
|
||||
let latBin =
|
||||
if lateralSpeed < -3.0: 0
|
||||
elif lateralSpeed < 0.0: 1
|
||||
elif lateralSpeed <= 3.0: 2
|
||||
else: 3
|
||||
|
||||
# heat ready: threshold slightly above 0 since fire 3.0 generates heat 1.6
|
||||
let heatBin = if gunHeat <= 0.2: 0 else: 1
|
||||
|
||||
aimBin * (N_DIST * N_LAT_SPD * N_HEAT) +
|
||||
distBin * (N_LAT_SPD * N_HEAT) +
|
||||
latBin * N_HEAT +
|
||||
heatBin
|
||||
|
||||
# ── Q-learning helpers ────────────────────────────────────────────────────────
|
||||
|
||||
proc selectAimAction(bot: QBot, state: int): int =
|
||||
if rand(1.0) < bot.epsilon:
|
||||
return rand(N_AIM_ACT - 1)
|
||||
var best = -1e30
|
||||
var bestA = 0
|
||||
for a in 0 ..< N_AIM_ACT:
|
||||
let v = bot.qt1[state * N_AIM_ACT + a]
|
||||
if v > best:
|
||||
best = v
|
||||
bestA = a
|
||||
bestA
|
||||
|
||||
proc selectFireAction(bot: QBot, state: int): int =
|
||||
if rand(1.0) < bot.epsilon:
|
||||
return rand(N_FIRE_ACT - 1)
|
||||
var best = -1e30
|
||||
var bestA = 0
|
||||
for a in 0 ..< N_FIRE_ACT:
|
||||
let v = bot.qt2[state * N_FIRE_ACT + a]
|
||||
if v > best:
|
||||
best = v
|
||||
bestA = a
|
||||
bestA
|
||||
|
||||
proc updateQT1(bot: QBot, state, action: int, reward: float) {.inline.} =
|
||||
# γ=0 bandit update: Q ← Q + α*(r - Q)
|
||||
let idx = state * N_AIM_ACT + action
|
||||
bot.qt1[idx] += ALPHA * (reward - bot.qt1[idx])
|
||||
|
||||
proc updateQT2(bot: QBot, state, action: int, reward: float) {.inline.} =
|
||||
let idx = state * N_FIRE_ACT + action
|
||||
bot.qt2[idx] += ALPHA * (reward - bot.qt2[idx])
|
||||
|
||||
# ── Bullet ring buffer ────────────────────────────────────────────────────────
|
||||
|
||||
proc pushPendingBullet(bot: QBot, state, aimAct, fireAct: int) =
|
||||
# Push a pending record; onBulletFired will stamp the real ID.
|
||||
bot.bullets[bot.bulletHead] = BulletRecord(
|
||||
bulletId: -1, state: state, aimAct: aimAct, fireAct: fireAct, valid: true)
|
||||
bot.bulletHead = (bot.bulletHead + 1) mod BULLET_BUF_CAP
|
||||
|
||||
proc stampBulletId(bot: QBot, realId: int) =
|
||||
# Find the oldest pending (id=-1) record and stamp it with the real bullet ID.
|
||||
# Search backwards from bulletHead (most recent push is just before head).
|
||||
for i in 1 .. BULLET_BUF_CAP:
|
||||
let idx = (bot.bulletHead - i + BULLET_BUF_CAP) mod BULLET_BUF_CAP
|
||||
if bot.bullets[idx].valid and bot.bullets[idx].bulletId == -1:
|
||||
bot.bullets[idx].bulletId = realId
|
||||
return
|
||||
|
||||
proc resolveBullet(bot: QBot, bulletId: int, reward: float) =
|
||||
for i in 0 ..< BULLET_BUF_CAP:
|
||||
if bot.bullets[i].valid and bot.bullets[i].bulletId == bulletId:
|
||||
let rec = bot.bullets[i]
|
||||
bot.bullets[i].valid = false
|
||||
updateQT1(bot, rec.state, rec.aimAct, reward)
|
||||
updateQT2(bot, rec.state, rec.fireAct, reward)
|
||||
return
|
||||
# Not found — bullet from previous round or missed stamp; silently ignore.
|
||||
|
||||
# ── Persistence ───────────────────────────────────────────────────────────────
|
||||
|
||||
proc saveQTable(bot: QBot) =
|
||||
createDir(qtablesDir)
|
||||
let s = newFileStream(qtablesDir / "gun_q.bin", fmWrite)
|
||||
if s == nil: return
|
||||
for v in bot.qt1: s.write(v)
|
||||
for v in bot.qt2: s.write(v)
|
||||
s.close()
|
||||
writeFile(qtablesDir / "gun_meta.txt",
|
||||
"rounds=" & $bot.totalRounds & "\n" &
|
||||
"epsilon=" & $bot.epsilon & "\n")
|
||||
|
||||
proc loadQTable(bot: QBot) =
|
||||
let path = qtablesDir / "gun_q.bin"
|
||||
if not fileExists(path): return
|
||||
let s = newFileStream(path, fmRead)
|
||||
if s == nil: return
|
||||
var i = 0
|
||||
while not s.atEnd and i < bot.qt1.len:
|
||||
bot.qt1[i] = s.readFloat64()
|
||||
inc i
|
||||
i = 0
|
||||
while not s.atEnd and i < bot.qt2.len:
|
||||
bot.qt2[i] = s.readFloat64()
|
||||
inc i
|
||||
s.close()
|
||||
let mp = qtablesDir / "gun_meta.txt"
|
||||
if not fileExists(mp): return
|
||||
for line in lines(mp):
|
||||
if line.startsWith("rounds="):
|
||||
bot.totalRounds = parseInt(line[7..^1])
|
||||
elif line.startsWith("epsilon="):
|
||||
bot.epsilon = parseFloat(line[8..^1])
|
||||
|
||||
# ── Logging ───────────────────────────────────────────────────────────────────
|
||||
|
||||
proc logRoundEnd(bot: QBot) =
|
||||
let hitRate = if bot.roundShots > 0:
|
||||
float(bot.roundHits) / float(bot.roundShots)
|
||||
else: 0.0
|
||||
stderr.writeLine("[QBot] round=" & $bot.totalRounds &
|
||||
" shots=" & $bot.roundShots &
|
||||
" hits=" & $bot.roundHits &
|
||||
" hitRate=" & formatFloat(hitRate, ffDecimal, 3) &
|
||||
" eps=" & formatFloat(bot.epsilon, ffDecimal, 3))
|
||||
|
||||
# Top 5 T1 Q-values
|
||||
type QEntry = tuple[v: float; s, a: int]
|
||||
var top: array[5, QEntry]
|
||||
var topN = 0
|
||||
for s in 0 ..< N_STATES:
|
||||
for a in 0 ..< N_AIM_ACT:
|
||||
let v = bot.qt1[s * N_AIM_ACT + a]
|
||||
if topN < 5:
|
||||
top[topN] = (v, s, a)
|
||||
inc topN
|
||||
else:
|
||||
var minIdx = 0
|
||||
for j in 1 ..< 5:
|
||||
if top[j].v < top[minIdx].v: minIdx = j
|
||||
if v > top[minIdx].v:
|
||||
top[minIdx] = (v, s, a)
|
||||
# Sort descending (simple insertion sort over 5 elements)
|
||||
for i in 0 ..< topN - 1:
|
||||
for j in i + 1 ..< topN:
|
||||
if top[j].v > top[i].v:
|
||||
let tmp = top[i]; top[i] = top[j]; top[j] = tmp
|
||||
stderr.writeLine("[QBot] top T1 Q-values:")
|
||||
for i in 0 ..< topN:
|
||||
stderr.writeLine(" s=" & $top[i].s & " aimAct=" & $top[i].a &
|
||||
" offset=" & formatFloat(AIM_OFFSETS[top[i].a], ffDecimal, 1) & "deg" &
|
||||
" q=" & formatFloat(top[i].v, ffDecimal, 4))
|
||||
|
||||
# ── Event handlers ────────────────────────────────────────────────────────────
|
||||
|
||||
method onScannedBot*(bot: QBot, e: ScannedBotEvent) =
|
||||
bot.enemyX = e.x
|
||||
bot.enemyY = e.y
|
||||
bot.enemyDir = e.direction
|
||||
bot.enemySpeed = e.speed
|
||||
bot.hasContact = true
|
||||
bot.ticksSinceScan = 0
|
||||
|
||||
method onBulletFired*(bot: QBot, e: BulletFiredEvent) =
|
||||
# Stamp the pending bullet record with the real ID assigned by the server.
|
||||
bot.stampBulletId(e.bullet.bulletId)
|
||||
|
||||
method onBulletHit*(bot: QBot, e: BulletHitBotEvent) =
|
||||
if e.victimId != getMyId():
|
||||
inc bot.roundHits
|
||||
bot.resolveBullet(e.bullet.bulletId, +1.0)
|
||||
|
||||
method onBulletHitWall*(bot: QBot, e: BulletHitWallEvent) =
|
||||
if e.bullet.ownerId == getMyId():
|
||||
inc bot.roundMisses
|
||||
bot.resolveBullet(e.bullet.bulletId, -0.2)
|
||||
|
||||
method onRoundStarted*(bot: QBot, e: RoundStartedEvent) =
|
||||
setAdjustGunForBodyTurn(true)
|
||||
setAdjustRadarForBodyTurn(true)
|
||||
setAdjustRadarForGunTurn(true)
|
||||
radar_lock.init()
|
||||
bot.hasContact = false
|
||||
bot.roundHits = 0
|
||||
bot.roundMisses = 0
|
||||
bot.roundShots = 0
|
||||
bot.ticksSinceScan = 0
|
||||
# Invalidate bullet buffer between rounds
|
||||
for i in 0 ..< BULLET_BUF_CAP: bot.bullets[i].valid = false
|
||||
bot.bulletHead = 0
|
||||
setTargetSpeed(0.0)
|
||||
setTurnRate(0.0)
|
||||
bot.epsilon = max(EPS_FLOOR, bot.epsilon * EPS_DECAY)
|
||||
|
||||
method onRoundEnded*(bot: QBot, e: RoundEndedEventForBot) =
|
||||
inc bot.totalRounds
|
||||
# Bullets still in flight at round end → treat as misses
|
||||
for i in 0 ..< BULLET_BUF_CAP:
|
||||
if bot.bullets[i].valid:
|
||||
updateQT1(bot, bot.bullets[i].state, bot.bullets[i].aimAct, -0.2)
|
||||
updateQT2(bot, bot.bullets[i].state, bot.bullets[i].fireAct, -0.2)
|
||||
bot.bullets[i].valid = false
|
||||
logRoundEnd(bot)
|
||||
|
||||
method onGameStarted*(bot: QBot, e: GameStartedEventForBot) =
|
||||
loadQTable(bot)
|
||||
randomize()
|
||||
|
||||
method onGameEnded*(bot: QBot, e: GameEndedEventForBot) =
|
||||
saveQTable(bot)
|
||||
stderr.writeLine("[QBot] game ended rounds=" & $bot.totalRounds &
|
||||
" eps=" & formatFloat(bot.epsilon, ffDecimal, 3))
|
||||
|
||||
# ── Main loop ─────────────────────────────────────────────────────────────────
|
||||
|
||||
method run*(bot: QBot) =
|
||||
while isRunning():
|
||||
setTargetSpeed(0.0)
|
||||
setTurnRate(0.0)
|
||||
|
||||
inc bot.ticksSinceScan
|
||||
if bot.ticksSinceScan > 5:
|
||||
bot.hasContact = false
|
||||
|
||||
if not bot.hasContact:
|
||||
setRadarTurnRate(45.0)
|
||||
go()
|
||||
continue
|
||||
|
||||
let myX = getX()
|
||||
let myY = getY()
|
||||
let gunDir = getGunDirection()
|
||||
let dist = hypot(myX - bot.enemyX, myY - bot.enemyY)
|
||||
# Absolute bearing from us to enemy
|
||||
let bearingToEnemy = directionTo(myX, myY, bot.enemyX, bot.enemyY)
|
||||
# Gun aim error: how far off our gun is from direct bearing to enemy
|
||||
let gunAimError = normalizeRelativeAngle(bearingToEnemy - gunDir)
|
||||
# Lateral speed: enemy velocity perpendicular to our line of sight
|
||||
let lateralSpeed = bot.enemySpeed * sin(degToRad(bot.enemyDir - bearingToEnemy))
|
||||
|
||||
let state = discretizeState(gunAimError, dist, lateralSpeed, getGunHeat())
|
||||
|
||||
# Select actions (no per-tick Q-update — updates happen only on bullet resolution)
|
||||
let aimAct = bot.selectAimAction(state)
|
||||
let fireAct = bot.selectFireAction(state)
|
||||
|
||||
# Compute gun turn: aim at enemy bearing + selected offset, clamped to ±20 deg/tick
|
||||
let aimOffset = AIM_OFFSETS[aimAct]
|
||||
let desiredDelta = normalizeRelativeAngle(bearingToEnemy + aimOffset - gunDir)
|
||||
setGunTurnRate(desiredDelta.clamp(-20.0, 20.0))
|
||||
|
||||
# Fire if the action calls for it and the gun is ready
|
||||
let firePower = FIRE_POWERS[fireAct]
|
||||
if firePower > 0.0 and getGunHeat() <= 0.0:
|
||||
if setFire(firePower):
|
||||
inc bot.roundShots
|
||||
# Push a pending record; onBulletFired (fired next tick) stamps the real ID
|
||||
bot.pushPendingBullet(state, aimAct, fireAct)
|
||||
|
||||
# Radar lock
|
||||
setRadarTurnRate(radar_lock.doRadar(getRadarDirection(), bearingToEnemy))
|
||||
|
||||
go()
|
||||
|
||||
# ── Entry point ───────────────────────────────────────────────────────────────
|
||||
|
||||
when isMainModule:
|
||||
createDir(qtablesDir)
|
||||
var bot = QBot(epsilon: EPS_INIT)
|
||||
for i in 0 ..< bot.qt1.len: bot.qt1[i] = Q_INIT
|
||||
for i in 0 ..< bot.qt2.len: bot.qt2[i] = Q_INIT
|
||||
start(bot, botJsonPath)
|
||||
@@ -0,0 +1,11 @@
|
||||
# Package
|
||||
version = "0.1.0"
|
||||
author = "Davide Cappellini"
|
||||
description = "Tabular Q-learning Tank Royale bot — chained action decomposition prototype"
|
||||
license = "MIT"
|
||||
bin = @["QBot"]
|
||||
|
||||
# Dependencies
|
||||
requires "nim >= 2.0.0"
|
||||
# tankroyale_botapi and radar_lock are vendored in-tree (common_libs/) and wired via
|
||||
# config.nims --path; no nimble dependency so builds never touch ~/.nimble/pkgs2.
|
||||
Executable
+3
@@ -0,0 +1,3 @@
|
||||
#!/bin/sh
|
||||
cd "$(dirname "$0")"
|
||||
exec ./QBot 2>> /tmp/qbot_stderr.log
|
||||
@@ -0,0 +1 @@
|
||||
--path:"../common_libs"
|
||||
Reference in New Issue
Block a user