Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| b7f49071ba |
+59
@@ -0,0 +1,59 @@
|
||||
# Evo_Bot
|
||||
|
||||
Robocode Tank Royale bot with a modular gun system where evolved neural networks learn to predict enemy dodge behavior.
|
||||
|
||||
## Language
|
||||
|
||||
**Evo_Bot**:
|
||||
The bot itself — handles movement and firing discipline. Guns are pluggable modules.
|
||||
_Avoid_: robot, tank
|
||||
|
||||
**Guess Factor (GF)**:
|
||||
A value from -1 to +1 representing where on the maximum escape angle arc the enemy is. 0 = directly ahead, -1 = full left dodge, +1 = full right dodge. The gun's prediction target.
|
||||
_Avoid_: aim offset, dodge index
|
||||
|
||||
**Max Escape Angle (MEA)**:
|
||||
The widest angle the enemy can reach before a bullet arrives, computed from distance and bullet speed. Guess factor is multiplied by MEA to get the aim offset.
|
||||
|
||||
**Lateral Velocity**:
|
||||
Enemy speed projected perpendicular to the line between you and them. The primary signal for guess factor prediction.
|
||||
_Avoid_: tangential speed, sideways velocity
|
||||
|
||||
**Sliding Window**:
|
||||
The last N ticks (default 30) of enemy state fed as input to the network. Each tick contains lateral velocity, heading delta, and wall distance ahead.
|
||||
_Avoid_: observation buffer, input history
|
||||
|
||||
**Replay Tape**:
|
||||
Rolling buffer of recorded enemy states (~2000 ticks). The evolution thread evaluates gun fitness against this tape.
|
||||
_Avoid_: experience buffer, replay buffer
|
||||
|
||||
**Virtual Gun**:
|
||||
A gun that runs in parallel without actually firing. It tracks where it would have aimed and whether a simulated bullet would have hit. Used to compare gun variants.
|
||||
|
||||
**Virtual Bullet**:
|
||||
A simulated bullet fired by a virtual gun. Never actually sent to the game engine.
|
||||
|
||||
**TOPO_Gun**:
|
||||
Fixed-topology ANN gun evolved by GA. Network shape is predetermined (e.g., 91-8-1), only weights are evolved.
|
||||
_Avoid_: static gun, fixed gun
|
||||
|
||||
**NEAT_Gun**:
|
||||
Variable-topology ANN gun where evolution can add/remove neurons and connections (NEAT algorithm). Deferred — only built if TOPO_Gun hits a ceiling.
|
||||
|
||||
**Population**:
|
||||
The set of candidate networks (default 64-200) being evolved. Each member is a complete set of ANN weights.
|
||||
|
||||
**Champion**:
|
||||
The best-performing network in the current GA population. The champion's weights are pushed to the inference side when it beats the current best.
|
||||
_Avoid_: best, winner, elite
|
||||
|
||||
**Fitness**:
|
||||
Hit count when a network's aim predictions are evaluated as virtual bullets against sampled ticks from the replay tape.
|
||||
_Avoid_: score, reward
|
||||
|
||||
**Cold Start**:
|
||||
The first-ever battle with no saved weights. The gun does not fire until the evolution thread produces its first champion. Subsequent battles load persisted weights.
|
||||
|
||||
**Weight Persistence**:
|
||||
Saving evolved weights to disk. Load order: per-opponent file, then global fallback, then random initialization.
|
||||
_Avoid_: model saving, checkpointing
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"name": "OscillatorBot",
|
||||
"version": "0.1.0",
|
||||
"authors": ["Davide Cappellini"],
|
||||
"description": "Predictable zigzag sparring partner for gun testing",
|
||||
"homepage": "",
|
||||
"countryCodes": ["IT"],
|
||||
"gameTypes": ["classic", "1v1"],
|
||||
"platform": "Nim",
|
||||
"programmingLang": "Nim"
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
# OscillatorBot — predictable zigzag sparring partner for GA gun testing.
|
||||
# Reverses direction + turn every PERIOD ticks. Head-on targeting only.
|
||||
|
||||
import std/[math, os]
|
||||
import tankroyale_botapi
|
||||
|
||||
const botJsonPath = currentSourcePath().parentDir / "OscillatorBot.json"
|
||||
|
||||
const
|
||||
SPEED = 7.0 # forward/backward speed
|
||||
PERIOD = 25 # ticks between direction reversals
|
||||
|
||||
type OscillatorBot = ref object of Bot
|
||||
tickCount: int
|
||||
moveSign: float # +1 forward, -1 backward
|
||||
turnSign: float # +1 right, -1 left
|
||||
|
||||
method onRoundStarted*(bot: OscillatorBot, e: RoundStartedEvent) =
|
||||
setAdjustGunForBodyTurn(true)
|
||||
setAdjustRadarForBodyTurn(true)
|
||||
setAdjustRadarForGunTurn(true)
|
||||
bot.tickCount = 0
|
||||
bot.moveSign = 1.0
|
||||
bot.turnSign = 1.0
|
||||
|
||||
method onScannedBot*(bot: OscillatorBot, e: ScannedBotEvent) =
|
||||
# Head-on targeting: aim gun directly at enemy, fire medium power
|
||||
let bearing = directionTo(getX(), getY(), e.x, e.y)
|
||||
let gunDelta = normalizeRelativeAngle(bearing - getGunDirection())
|
||||
setGunTurnRate(gunDelta.clamp(-MAX_GUN_TURN_RATE, MAX_GUN_TURN_RATE))
|
||||
if abs(gunDelta) < 10.0 and getGunHeat() <= 0.0:
|
||||
discard setFire(2.0)
|
||||
|
||||
method run*(bot: OscillatorBot) =
|
||||
while isRunning():
|
||||
inc bot.tickCount
|
||||
if bot.tickCount mod PERIOD == 0:
|
||||
bot.moveSign *= -1.0
|
||||
bot.turnSign *= -1.0
|
||||
|
||||
setTargetSpeed(bot.moveSign * SPEED)
|
||||
setTurnRate(bot.turnSign * 4.0)
|
||||
setRadarTurnRate(45.0) # spin radar to keep scanning
|
||||
go()
|
||||
|
||||
when isMainModule:
|
||||
var bot = OscillatorBot(moveSign: 1.0, turnSign: 1.0)
|
||||
start(bot, botJsonPath)
|
||||
Executable
+3
@@ -0,0 +1,3 @@
|
||||
#!/bin/sh
|
||||
cd "$(dirname "$0")"
|
||||
exec ./OscillatorBot 2>> /tmp/oscillatorbot_stderr.log
|
||||
@@ -0,0 +1 @@
|
||||
--path:"../libs"
|
||||
@@ -0,0 +1,46 @@
|
||||
# Neuroevolution gun with fixed-topology ANN evolved by GA
|
||||
|
||||
Evo_Bot needs a gun that adapts to each opponent's dodge patterns during a match, finds nonlinear movement patterns that histograms miss, and is original. We chose a fixed-topology feedforward ANN (91->8->1, 745 weights) whose weights are evolved by a mutation-only GA running on a parallel thread. This beats the alternatives (RL too slow to adapt in-match, Q-learning collapses to histogram for single-shot decisions, guess-factor histograms are unoriginal, transformer/LLM-style prediction is data-starved at ~14k ticks per match) while keeping implementation risk low by deferring topology evolution (NEAT) until the fixed network hits its ceiling.
|
||||
|
||||
## Considered Options
|
||||
|
||||
- **PPO / SAC (end-to-end RL):** Too slow -- thousands of rounds to converge, cannot adapt mid-match. Explored in other bots in this repo.
|
||||
- **Q-learning for aiming:** Collapses to a histogram. Single-shot aiming has no sequential decision structure for Q-learning to exploit.
|
||||
- **Guess-factor histogram:** Proven and fast to converge (~15 ticks), but unoriginal -- 20 years of community tuning.
|
||||
- **Transformer / LLM-style sequence prediction:** Data-starved. ~14k ticks per match vs billions needed for attention-based models.
|
||||
- **GA with crossover:** Literature uniformly shows crossover is harmful for ANN weight evolution -- it breaks co-adapted weight configurations. Every modern neuroevolution paper (Uber Deep GA, NRA, OpenAI ES) drops it.
|
||||
- **CMA-ES:** Ideal at d=750 weights (self-adapts sigma and covariance). More complex to implement; upgrade path from simple GA when needed.
|
||||
|
||||
## Decision
|
||||
|
||||
**Architecture:**
|
||||
- Evo_Bot (1v1) with modular gun interface: `feed(state)` / `aim() -> (angle, power)`
|
||||
- Gun owns its evolution thread (parallel, never blocks inference)
|
||||
|
||||
**TOPO_Gun (first gun implementation):**
|
||||
- Network: 91->8->1 (hidden size configurable), 745 weights
|
||||
- Input: 30 ticks x (lateral_vel, delta_heading, wall_distance_ahead) + current distance = 91
|
||||
- Output: guess factor (-1 to +1)
|
||||
- Bullet power: deterministic distance-based formula (not learned)
|
||||
- Evolution: population 300, clone loaded champion + small Gaussian mutations (ALL weights, sigma=0.005-0.01), NO crossover, single elite preserved unchanged
|
||||
- Fitness: virtual bullet hits on 100 randomly sampled replay tape ticks, using real distance-based power
|
||||
- Replay tape: rolling window ~2000 ticks (configurable)
|
||||
- Weight persistence: per-opponent -> global fallback -> random init (load order)
|
||||
- Cold start: first-ever run, don't fire until champion emerges; subsequent runs load weights, fire from tick 1
|
||||
- Push new champion weights to inference when it hits better than current
|
||||
|
||||
**Deferred:**
|
||||
- NEAT_Gun: deferred until TOPO_Gun hits its ceiling
|
||||
- Virtual Guns: run multiple guns in parallel, fire whichever has best virtual hit rate
|
||||
|
||||
**Boundaries:**
|
||||
- Bot controls firing discipline (when to shoot, energy management); gun always returns best aim
|
||||
|
||||
## Consequences
|
||||
|
||||
- GA+ANN finds nonlinear patterns histograms miss, but needs more data (~50+ ticks vs ~15 for guess-factor histogram) before it outperforms
|
||||
- Fixed topology before NEAT reduces implementation risk
|
||||
- Mutation-only evolution simplifies implementation (no crossover logic)
|
||||
- Parallel evolution thread reuses the pattern from SAC_LSTM_Bot (#48)
|
||||
- Per-opponent weight persistence eliminates cold start after first encounter
|
||||
- CMA-ES is the natural upgrade path if simple GA convergence is too slow (d=750 is CMA-ES sweet spot)
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,173 @@
|
||||
# GA/ES Parameters for ~750-Weight Neuroevolution
|
||||
|
||||
Research for issue #65. Concrete parameter recommendations extracted from 13 papers in `docs/papers/neuroevolution/`.
|
||||
|
||||
## Context
|
||||
|
||||
Evo_Bot's TOPO_Gun: fixed-topology feedforward ANN, 745-1489 weights depending on layer sizes. Task is predicting enemy dodge behavior from a 30-tick sliding window, outputting a guess factor. Online evolution during Robocode matches.
|
||||
|
||||
---
|
||||
|
||||
## 1. Population Size
|
||||
|
||||
**Recommendation: 64-256, not 300.**
|
||||
|
||||
| Source | Network size | Population | Notes |
|
||||
|--------|-------------|------------|-------|
|
||||
| Uber Deep GA (Such et al. 2017) | 4M params (Atari), 167k (Humanoid) | 1,000 | Massive networks, distributed; overkill for ~750 weights |
|
||||
| OpenAI ES (Salimans et al. 2017) | 1.7M params | 720-1,440 workers | NES-style, not a population GA |
|
||||
| Canonical ES (Chrabaszcz et al. 2018) | 1.7M params | 798 (lambda), mu=50 | mu=50 selected as best across games |
|
||||
| World Models CMA-ES (Ha & Schmidhuber 2018) | 867-1,088 params (controller) | 64 | CMA-ES with 16 evals per individual |
|
||||
| Challenges paper (Muller & Glasmachers 2018) | 1,352-2,349 weights | default CMA-ES lambda | LM-MA-ES for ~1k-2k weights |
|
||||
| Evolving Generalists (Triebold & Yaman 2023) | 246-728 weights | xNES default: 4+floor(3*ln(d)) | For d=745 -> ~24; for d=1489 -> ~26 |
|
||||
| NRA (Le Clei & Bellec 2022) | 3-322 params (dynamic) | 8-512 | 256+elitism best for ~300-param tasks; 16 enough for <100 params |
|
||||
| Playing Atari 6 Neurons (Cuccu et al. 2018) | ~3k connections, 6-18 neurons | xNES default | Small networks, 100 generations sufficient |
|
||||
|
||||
**Key finding:** For ~750 weights, CMA-ES default lambda = 4+floor(3*ln(745)) = ~24 is a starting floor. The World Models paper (867-1,088 params, closest to our size) used 64 with CMA-ES and solved CarRacing. NRA used 256+elitism for ~300-param Pendulum networks. For a simple truncation-selection GA (not CMA-ES), 100-200 is a reasonable population; 300 is slightly wasteful but not harmful if evaluation is cheap.
|
||||
|
||||
**Verdict: Start with 100-200 for a truncation GA. If using CMA-ES or xNES, use their defaults (~24-26). 300 is too large for the weight count but acceptable if per-evaluation cost is low (Robocode rounds are fast).**
|
||||
|
||||
---
|
||||
|
||||
## 2. Mutation Rate and Distribution
|
||||
|
||||
**Recommendation: Additive Gaussian on ALL weights, sigma=0.002-0.02, not 5% at sigma=0.1.**
|
||||
|
||||
| Source | Mutation scheme | Notes |
|
||||
|--------|----------------|-------|
|
||||
| Uber Deep GA (Such et al. 2017) | theta' = theta + sigma * epsilon, epsilon ~ N(0,I). Sigma determined empirically per task. | Mutates ALL weights every generation, not a fraction. No "mutation rate" — every weight gets noise. |
|
||||
| OpenAI ES (Salimans et al. 2017) | sigma = fixed hyperparameter (not adapted). Perturbation on full parameter vector. | Full-vector Gaussian perturbation, sigma tuned. |
|
||||
| Canonical ES (Chrabaszcz et al. 2018) | N(0, sigma^2) added to all params. Network init from N(0, 0.05). | sigma is the step-size, adapted or fixed. |
|
||||
| World Models (Ha & Schmidhuber 2018) | CMA-ES adapts sigma and full covariance matrix. | Self-adapting sigma — no manual sigma needed. |
|
||||
| Challenges paper (Muller & Glasmachers 2018) | CSA (cumulative step-size adaptation) essential. Fixed sigma converges as slowly as random search. | Step-size adaptation is critical; fixed sigma is a known failure mode. |
|
||||
| NRA (Le Clei & Bellec 2022) | N(0, 0.01) perturbation to all weights and biases. Top 50% selection. | sigma=0.01 for small networks. |
|
||||
|
||||
**Key finding:** No paper uses a "5% mutation rate" (mutating only 5% of weights). ALL papers mutate ALL weights simultaneously with small additive Gaussian noise. The "mutation rate" concept from traditional GAs (flip probability per gene) does not apply to real-valued neuroevolution. Instead, the noise magnitude (sigma) controls exploration.
|
||||
|
||||
For ~750 weights:
|
||||
- NRA uses sigma=0.01 for networks up to ~300 params
|
||||
- Uber GA uses sigma empirically per task (typical range 0.002-0.02 for Atari)
|
||||
- CMA-ES/xNES adapt sigma automatically
|
||||
|
||||
**Verdict: Mutate ALL weights every generation. Use sigma=0.005-0.01 as starting point. If using CMA-ES, sigma self-adapts. The "5% of weights mutated" approach is non-standard and likely harmful — it under-explores the search space.**
|
||||
|
||||
---
|
||||
|
||||
## 3. Selection Pressure
|
||||
|
||||
**Recommendation: Top 10-50% (truncation) or top mu out of lambda. 20% is reasonable.**
|
||||
|
||||
| Source | Selection | Notes |
|
||||
|--------|-----------|-------|
|
||||
| Uber Deep GA (Such et al. 2017) | Truncation selection, top T individuals become parents. T not specified as percentage — varies. | Parents chosen uniformly at random from top T. |
|
||||
| Canonical ES (Chrabaszcz et al. 2018) | Top mu=50 out of lambda=798 (~6%). Weighted mean of top mu. | mu=50 found optimal across games; tested mu in {10,20,50,100,200,400}. |
|
||||
| NRA (Le Clei & Bellec 2022) | Top 50% duplicated, bottom 50% replaced. | Simple and effective for small populations. |
|
||||
| Evolving Generalists (Triebold & Yaman 2023) | xNES default selection. | NES uses weighted rank-based update. |
|
||||
|
||||
**Key finding:** Selection pressure varies widely. Canonical ES uses ~6% (mu=50 out of 798). NRA uses 50%. Standard CMA-ES uses mu = lambda/2 (50%). Top 20% is in the middle range and is fine for a truncation GA.
|
||||
|
||||
**Verdict: 20% is reasonable. For small populations (64-100), 50% (top half) may work better. For larger populations (200+), stricter selection (10-20%) is appropriate. The Canonical ES result suggests mu=50 works well regardless of lambda for Atari-scale problems.**
|
||||
|
||||
---
|
||||
|
||||
## 4. Crossover
|
||||
|
||||
**Recommendation: No crossover. Mutation-only.**
|
||||
|
||||
| Source | Crossover? | Notes |
|
||||
|--------|-----------|-------|
|
||||
| Uber Deep GA (Such et al. 2017) | **No crossover.** "Historically, GAs often involve crossover, but for simplicity we did not include it." | Explicitly dropped crossover for DNN weights. |
|
||||
| OpenAI ES (Salimans et al. 2017) | No crossover. | ES-style: mean update, not recombination of individuals. |
|
||||
| NRA (Le Clei & Bellec 2022) | **No crossover.** "stripping down many mechanisms popular in traditional evolutionary methods, like agent crossover and speciation" | Crossover explicitly excluded. |
|
||||
| CMA-ES/xNES | No crossover in the traditional sense. | Weighted recombination of top individuals into distribution mean — not pairwise crossover. |
|
||||
| NEAT (Stanley & Miikkulainen 2011) | Has crossover via innovation numbers. | But NEAT is for topology evolution, not fixed-topology weight-only GA. |
|
||||
|
||||
**Key finding:** Every modern neuroevolution paper that works with fixed-topology networks drops crossover. Fogel & Stayton (1994, cited by Such et al.) showed crossover is often ineffective for simulated evolutionary optimization. For ANN weight vectors, crossover tends to be destructive because individual weights are not independent genes — they form functional units (layers, pathways) where mixing two different solutions creates non-functional hybrids.
|
||||
|
||||
**Verdict: No crossover. Mutation-only. Uniform crossover of ANN weights is harmful — it breaks co-adapted weight configurations. If recombination is desired, use CMA-ES/xNES-style weighted mean of top solutions, which is mathematically sound.**
|
||||
|
||||
---
|
||||
|
||||
## 5. Elitism
|
||||
|
||||
**Recommendation: Yes, keep top 1 unchanged (single elite).**
|
||||
|
||||
| Source | Elitism? | Notes |
|
||||
|--------|---------|-------|
|
||||
| Uber Deep GA (Such et al. 2017) | **Yes, 1 elite.** "The Nth individual is an unmodified copy of the best individual from the previous generation." Additionally, top 10 re-evaluated 30 times to find the true elite. | Single elite with robust re-evaluation. |
|
||||
| NRA (Le Clei & Bellec 2022) | **Yes, elitism tested and beneficial.** Population sizes labeled "(elite)" consistently outperform non-elite variants in all figures. | Elitism was the single most impactful improvement for small populations. |
|
||||
| CMA-ES | Elitist variants exist (mu+lambda). Standard CMA-ES is (mu,lambda) — non-elitist. | Non-elitist CMA-ES relies on distribution adaptation, not individual survival. |
|
||||
|
||||
**Key finding:** For simple truncation GAs, elitism (keeping top 1) prevents regression and is universally recommended. The NRA paper shows that adding elitism to even a population of 16 dramatically improves results. Uber's Deep GA uses elitism with robust re-evaluation (30 episodes to confirm the elite).
|
||||
|
||||
**Verdict: Keep top 1 elite unchanged. In noisy evaluation environments (Robocode), re-evaluate the top few candidates multiple times to find the true elite, following Uber's approach.**
|
||||
|
||||
---
|
||||
|
||||
## 6. Generations to Convergence
|
||||
|
||||
**Recommendation: 100-1,500 generations for ~750 weights, depending on the algorithm.**
|
||||
|
||||
| Source | Network size | Generations | Notes |
|
||||
|--------|-------------|-------------|-------|
|
||||
| Uber Deep GA (Such et al. 2017) | 4M params | 348-1,834 gens (at 1k pop) | Many games: best-in-run found in 1-29 gens |
|
||||
| World Models CMA-ES (Ha & Schmidhuber 2018) | 867 params | ~1,800 gens | CMA-ES, pop=64, solved CarRacing |
|
||||
| NRA (Le Clei & Bellec 2022) | 3-322 params (dynamic) | 100-5,000 gens | Simple tasks: <300 gens. Complex (Ant/Humanoid): 5,000+ |
|
||||
| Playing Atari 6 Neurons (Cuccu et al. 2018) | ~3k connections | 100 gens | Extremely tight budget, still achieved competitive results |
|
||||
| Challenges paper (Muller & Glasmachers 2018) | 1,352-2,349 weights | 100k-300k evals | LM-MA-ES, ~150k evals for bipedal walker convergence |
|
||||
| Evolving Generalists (Triebold & Yaman 2023) | 728 weights (Ant) | 5,000 gens max | xNES, some tasks solved in <100 gens |
|
||||
|
||||
**Key finding:** For ~750 weights with a simple truncation GA (pop=100), expect 200-500 generations for a well-tuned sigma. CMA-ES/xNES may converge faster in generations but each generation is more expensive. The Challenges paper warns that halving the distance to the optimum requires O(d) samples, so for d=750, expect ~750 evaluations per halving step.
|
||||
|
||||
For Evo_Bot's online evolution during matches: each Robocode round can evaluate one individual. With 35-round matches (typical), ~5 generations of pop=7 per match, or ~2 generations of pop=15. Convergence within a single match is unlikely; evolution must persist across matches via weight persistence.
|
||||
|
||||
**Verdict: Budget 500-2,000 generations. With pop=100, that's 50k-200k evaluations. Online evolution will need many matches to converge — weight persistence is essential.**
|
||||
|
||||
---
|
||||
|
||||
## 7. CMA-ES vs Simple GA vs Tournament Selection
|
||||
|
||||
**Recommendation: CMA-ES or xNES for ~750 weights. Simple GA as a simpler fallback.**
|
||||
|
||||
| Algorithm | Sweet spot | Pros | Cons | Source |
|
||||
|-----------|-----------|------|------|--------|
|
||||
| **CMA-ES** | d <= 1,000 (ideal), up to ~2,000 (practical) | Self-adapts sigma and covariance, best convergence rate, handles ill-conditioned landscapes | O(d^2) memory/time per generation, needs O(d^2) evals for full covariance learning | Muller & Glasmachers 2018, Ha & Schmidhuber 2018 |
|
||||
| **LM-MA-ES** | d = 1,000-10,000 | O(d) complexity, adapts fastest-evolving subspace, strong on ~2k weights | More complex to implement | Muller & Glasmachers 2018 |
|
||||
| **xNES** | d <= 1,000 | Natural gradient, self-adapting, elegant | Similar scaling limits to CMA-ES | Cuccu et al. 2018, Triebold & Yaman 2023 |
|
||||
| **Simple truncation GA** | Any d | Dead simple, trivially parallel, no internal state beyond population | Needs manual sigma tuning, no adaptation, converges slowly | Such et al. 2017, Le Clei & Bellec 2022 |
|
||||
| **Canonical (mu,lambda)-ES** | Any d | Step-size adaptation via CSA, simple | mu tuning matters; mu=50 worked well in Chrabaszcz 2018 | Chrabaszcz et al. 2018 |
|
||||
| **Tournament selection** | Traditional GA context | Tunable selection pressure | No advantage over truncation for ANN weights | Not specifically tested in any of the 13 papers |
|
||||
|
||||
**Key finding at d=750:** CMA-ES is in its sweet spot. The World Models paper (Ha & Schmidhuber 2018) used CMA-ES with pop=64 on 867-1,088 params and solved complex control tasks. The Evolving Generalists paper (Triebold & Yaman 2023) used xNES on 728 weights (Ant controller) with default population sizes. The Challenges paper (Muller & Glasmachers 2018) explicitly shows CMA-ES and LM-MA-ES outperforming simple ES on problems with 769-2,738 weights.
|
||||
|
||||
However, CMA-ES requires O(d^2) = O(560k) memory for the covariance matrix at d=750. This is manageable but not trivial for an online Robocode bot. A simpler option is a (mu,lambda)-ES with CSA for step-size adaptation.
|
||||
|
||||
**Verdict: CMA-ES or xNES is the best fit for 750 weights. If implementation complexity is a concern, a truncation GA with adaptive sigma (or even fixed sigma=0.005) is the pragmatic choice. Tournament selection offers no advantage.**
|
||||
|
||||
---
|
||||
|
||||
## Summary: Recommended Parameters for Evo_Bot TOPO_Gun
|
||||
|
||||
| Parameter | Current assumption | Recommendation | Rationale |
|
||||
|-----------|-------------------|----------------|-----------|
|
||||
| Population | 300 | 64-200 | 300 is oversized for ~750 weights; 64 (CMA-ES) to 200 (truncation GA) |
|
||||
| Mutation | 5% of weights, gaussian sigma=0.1 | ALL weights, sigma=0.005-0.01 | Every paper mutates all weights. sigma=0.1 is too large. |
|
||||
| Selection | Top 20% | Top 20-50% | 20% is fine; 50% if pop is small |
|
||||
| Crossover | Uniform | None | Uniformly dropped in all modern neuroevolution papers |
|
||||
| Elitism | Not specified | Top 1, re-evaluated | Single elite prevents regression; re-evaluate to handle noise |
|
||||
| Algorithm | Simple GA | CMA-ES or truncation GA + CSA | CMA-ES is in its sweet spot at d=750; simple GA works but converges slower |
|
||||
| Generations | Not specified | 500-2,000 | Online evolution needs many matches for convergence |
|
||||
|
||||
---
|
||||
|
||||
## Sources
|
||||
|
||||
1. Such et al. 2017 — "Deep Neuroevolution: Genetic Algorithms are a Competitive Alternative" (Uber AI Labs)
|
||||
2. Salimans et al. 2017 — "Evolution Strategies as a Scalable Alternative to Reinforcement Learning" (OpenAI)
|
||||
3. Chrabaszcz et al. 2018 — "Back to Basics: Benchmarking Canonical Evolution Strategies for Playing Atari"
|
||||
4. Muller & Glasmachers 2018 — "Challenges in High-dimensional Reinforcement Learning with Evolution Strategies"
|
||||
5. Ha & Schmidhuber 2018 — "Recurrent World Models Facilitate Policy Evolution" (World Models)
|
||||
6. Cuccu et al. 2018 — "Playing Atari with Six Neurons"
|
||||
7. Le Clei & Bellec 2022 — "Neuroevolution of Recurrent Architectures on Control Tasks"
|
||||
8. Triebold & Yaman 2023 — "Evolving Generalist Controllers to Handle a Wide Range of Morphological Variations"
|
||||
9. Stanley & Miikkulainen 2011 — "Competitive Coevolution through Evolutionary Complexification" (NEAT)
|
||||
@@ -0,0 +1,111 @@
|
||||
## ga_spike.nim — throwaway GA neuroevolution spike: evolve ANN to predict sin(x)
|
||||
## Validates: population init, fitness eval, truncation selection, gaussian
|
||||
## mutation on all weights, single-elite preservation, champion tracking.
|
||||
## Issue #66.
|
||||
|
||||
import std/[math, random, algorithm, strformat]
|
||||
|
||||
const
|
||||
InputDim = 10 # sliding window of 10 past sin values
|
||||
HiddenDim = 4
|
||||
OutputDim = 1
|
||||
NumWeights = (InputDim * HiddenDim + HiddenDim) + # W1 + b1
|
||||
(HiddenDim * OutputDim + OutputDim) # W2 + b2 = 49
|
||||
PopSize = 200
|
||||
Generations = 200
|
||||
TopFrac = 0.2 # keep top 20%
|
||||
Sigma = 0.01 # gaussian mutation sigma on ALL weights
|
||||
EvalPoints = 50 # fitness eval sample size
|
||||
|
||||
type Individual = object
|
||||
weights: seq[float64]
|
||||
fitness: float64
|
||||
|
||||
proc forward(w: seq[float64]; input: array[InputDim, float64]): float64 =
|
||||
## 10->4->1 tanh ANN, flat weight layout: W1[40], b1[4], W2[4], b2[1]
|
||||
var hidden: array[HiddenDim, float64]
|
||||
for h in 0..<HiddenDim:
|
||||
var s = w[InputDim * HiddenDim + h] # b1[h]
|
||||
for i in 0..<InputDim:
|
||||
s += w[h * InputDim + i] * input[i] # W1[h,i]
|
||||
hidden[h] = tanh(s)
|
||||
var s = w[InputDim * HiddenDim + HiddenDim + HiddenDim] # b2[0]
|
||||
for h in 0..<HiddenDim:
|
||||
s += w[InputDim * HiddenDim + HiddenDim + h] * hidden[h] # W2[h]
|
||||
result = tanh(s)
|
||||
|
||||
proc evaluate(ind: var Individual; rng: var Rand) =
|
||||
## Fitness = -MSE on EvalPoints random samples of sin prediction.
|
||||
var mse = 0.0
|
||||
for _ in 0..<EvalPoints:
|
||||
let x = rng.rand(0.0 .. 20.0 * PI)
|
||||
var input: array[InputDim, float64]
|
||||
for i in 0..<InputDim:
|
||||
input[i] = sin(x - float64(InputDim - 1 - i))
|
||||
let target = sin(x)
|
||||
let pred = forward(ind.weights, input)
|
||||
mse += (pred - target) * (pred - target)
|
||||
ind.fitness = -mse / EvalPoints.float64
|
||||
|
||||
proc mutate(ind: var Individual; rng: var Rand) =
|
||||
## Additive gaussian on ALL weights, per GA-parameters research.
|
||||
for i in 0..<ind.weights.len:
|
||||
ind.weights[i] += rng.gauss(0.0, Sigma)
|
||||
|
||||
when isMainModule:
|
||||
var rng = initRand(42)
|
||||
|
||||
# Init population: random weights in [-0.5, 0.5]
|
||||
var pop = newSeq[Individual](PopSize)
|
||||
for i in 0..<PopSize:
|
||||
pop[i].weights = newSeq[float64](NumWeights)
|
||||
for j in 0..<NumWeights:
|
||||
pop[i].weights[j] = rng.rand(-0.5 .. 0.5)
|
||||
|
||||
echo &"GA spike: {PopSize} individuals, {NumWeights} weights, {Generations} gens, sigma={Sigma}"
|
||||
|
||||
for gen in 0..<Generations:
|
||||
# Evaluate
|
||||
for i in 0..<PopSize:
|
||||
pop[i].evaluate(rng)
|
||||
|
||||
# Sort descending by fitness (higher = better)
|
||||
pop.sort(proc(a, b: Individual): int = cmp(b.fitness, a.fitness))
|
||||
|
||||
if gen mod 20 == 0 or gen == Generations - 1:
|
||||
echo &" gen {gen:3d} best_fitness={pop[0].fitness:.6f} (MSE={-pop[0].fitness:.6f})"
|
||||
|
||||
# Selection: keep top 20%
|
||||
let nParents = max(1, int(PopSize.float64 * TopFrac))
|
||||
|
||||
# Next generation: elite(1) + mutated children from parents
|
||||
var next = newSeq[Individual](PopSize)
|
||||
next[0] = pop[0] # single elite, unchanged
|
||||
for i in 1..<PopSize:
|
||||
let parent = rng.rand(0..<nParents)
|
||||
next[i].weights = pop[parent].weights # clone parent
|
||||
next[i].mutate(rng)
|
||||
pop = next
|
||||
|
||||
# Final evaluation of champion
|
||||
pop[0].evaluate(rng)
|
||||
echo &"\nChampion fitness: {pop[0].fitness:.6f} (MSE={-pop[0].fitness:.6f})"
|
||||
|
||||
# Print predictions vs actual
|
||||
echo "\nPredictions (sample):"
|
||||
var totalErr = 0.0
|
||||
let nSamples = 20
|
||||
for i in 0..<nSamples:
|
||||
let x = float64(i) * 1.0
|
||||
var input: array[InputDim, float64]
|
||||
for j in 0..<InputDim:
|
||||
input[j] = sin(x - float64(InputDim - 1 - j))
|
||||
let target = sin(x)
|
||||
let pred = forward(pop[0].weights, input)
|
||||
totalErr += abs(pred - target)
|
||||
echo &" x={x:5.1f} sin(x)={target:+.4f} pred={pred:+.4f} err={abs(pred-target):.4f}"
|
||||
|
||||
let avgErr = totalErr / nSamples.float64
|
||||
echo &"\nAverage absolute error: {avgErr:.4f}"
|
||||
assert avgErr < 0.1, &"Champion avg error {avgErr:.4f} >= 0.1 — evolution did not converge"
|
||||
echo "PASS: assert avgErr < 0.1"
|
||||
Reference in New Issue
Block a user