## TFIL commitment A/B — measurement harness. ## ## Consumes the raw artifacts produced by the five-arm, seven-run, seven-round ## A/B against the real DrussGT bridge (tools/robocode_shim/run_bridge_battle.sh) ## and reports ## ## 1. the arm table: damage/run, ROUND WINS, hits taken, shots, per-run values ## 2. the exact two-sided permutation p for damage and for round wins ## 3. the per-arm TREATMENT diagnostics (from TR_TFIL_COMMIT_LOG): decision ## interval, % tile-change replans, reversal-pick rate, speed profile ## around reversals — this is what proves the treatment actually applied ## ## Expected raw layout (produced by the A/B driver): ## /out//run.jsonl per-tick states (ex/ey = DrussGT, sx/sy = us) ## /out//run.events.jsonl fire/hit/death events (numeric bot ids) ## /out//run.rounds.json round boundaries (startTick/count) ## /logs//run.out capture stdout (final RESULTS block) ## /logs//run.commit.jsonl TFIL per-tick commit diagnostics ## ## Modes: ## nim c -r --path:common_libs common_libs/tests/measure_tfil_commit_ab.nim \ ## --raw /tmp/tfil_ab2 --summary common_libs/tests/fixtures/tfil_commit_ab_results.json \ ## --report common_libs/tests/fixtures/tfil_commit_ab_report.md ## ... --from-summary # re-print the report from the committed summary ## ## NOTE on the ingredients, because they are easy to get wrong: ## * In the shim's JSONL, `e*` is the capture SUBJECT and `s*` is the other bot. ## run_bridge_battle.sh is invoked with `--subject DrussGT`, so `e*` = DrussGT ## and `s*` = ModularBot (the bot under test). Verified here by matching the ## events sidecar's bullet origins against both recorded positions. ## * The "[capture] ROUND n" stdout lines carry the CUMULATIVE score and the ## battle rank so far, NOT the round result. Round wins are taken from the ## final RESULTS block (`firstPlaces`) and cross-checked against the ## BotDeathEvent in the events sidecar (the loser of a TR 1v1 round is the ## bot that died). ## * `damage` is the server's own BulletHitBotEvent damage, attributed by ## numeric owner/victim id. import std/[os, json, math, strformat, strutils, tables, sequtils] const Arms = ["A", "B", "C", "D", "E"] ArmRuns = 7 MaxRuns = 14 ArmRounds = 7 ArmEnv = [ "", "TR_TFIL_TILE_REPLAN=off", "TR_TFIL_TILE_REPLAN=off TR_TFIL_NO_REV=1", "TR_TFIL_TILE_REPLAN=enemy", "TR_TFIL_TILE_REPLAN=off TR_TFIL_COMMIT_TICKS=30", ] ArmLabel = [ "A control (shipped)", "B honour commitment", "C B + no-reversal", "D enemy-keyed cancel", "E commit 30 ticks", ] RevOffsets = [-5, -4, -3, -2, -1, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10] ## Number of runs per arm to analyse; 7 = the pre-registered primary block, ## 14 = primary + replication block (set with --runs). var RunCount = ArmRuns ## First run index to include (1 = the pre-registered primary block). var FirstRun = 1 # ── small stats helpers ────────────────────────────────────────────────────── proc mean(s: openArray[float]): float = if s.len == 0: return 0.0 var t = 0.0 for v in s: t += v t / s.len.float proc exactPermP(a, b: seq[float]): tuple[p, delta: float, perms: int] = ## Exact two-sided permutation test on the difference of means. Enumerates ## every relabelling of the pooled sample (C(n+m, n) of them; 3432 for 7v7, ## 40_116_600 for 14v14). The observed labelling is included, so p >= 1/C. let n = a.len let m = b.len doAssert n + m <= 30, "pool too large to enumerate exactly" let ma = mean(a) let mb = mean(b) let obs = abs(ma - mb) let pool = a & b let total = n + m var poolSum = 0.0 for v in pool: poolSum += v let nf = n.float let mf = m.float var cnt = 0 var perms = 0 proc rec(start, chosen: int, s: float) = if chosen == n: let sx = s / nf let sy = (poolSum - s) / mf inc perms if abs(sx - sy) >= obs - 1e-12: inc cnt return for i in start .. total - (n - chosen): rec(i + 1, chosen + 1, s + pool[i]) rec(0, 0, 0.0) (cnt.float / perms.float, ma - mb, perms) proc binom(n, k: int): int = ## exact C(n, k) by multiplicative recurrence (n <= 30 here) if k < 0 or k > n: return 0 var kk = min(k, n - k) var r = 1'i64 for i in 1 .. kk: r = r * (n - kk + i) div i int(r) proc logC(n, k: int): float = ## log binomial coefficient. if k < 0 or k > n: return NegInf var r = 0.0 for i in 1 .. k: r += ln((n - k + i).float / i.float) r proc hyperPmf(k, n1, n, t: int): float = ## P(X = k), X ~ Hypergeometric(n, t successes, n1 draws). if k < 0 or k > t or (n1 - k) < 0 or (n1 - k) > (n - t): return 0.0 exp(logC(t, k) + logC(n - t, n1 - k) - logC(n, n1)) proc hyperTwoSidedP(k, n1, n, t: int): float = ## Exact two-sided p for "as or more extreme" on the sum statistic: the sum of ## the hypergeometric pmf over every k whose pmf is <= pmf(k_obs). This is the ## exact permutation distribution of the round-win count under exchangeability. let pObs = hyperPmf(k, n1, n, t) var acc = 0.0 var mass = 0.0 for kk in 0 .. n1: let p = hyperPmf(kk, n1, n, t) mass += p if p <= pObs * (1.0 + 1e-9): acc += p if mass <= 0.0: return 1.0 min(1.0, acc / mass) # ── per-run extraction ─────────────────────────────────────────────────────── type RunMetrics = object run: int rounds: int wins: int winsByDeath: int perRoundWins: seq[int] ## 1 = we won the round, 0 = we lost score: int enemyScore: int ticks: int damageDealt: float damageTaken: float shots: int enemyShots: int hits: int hitsTaken: int # diagnostics from TR_TFIL_COMMIT_LOG commitCalls: int picks: int intervalMean: float revPicks: int reversalPct: float byReason: Table[string, int] speedMean: float speedProfile: seq[float] ## aligned with RevOffsets; NaN when no data speedProfileN: seq[int] ok: bool note: string ArmReport = object arm: int env: string label: string runs: seq[RunMetrics] # aggregated diagnostics commitCalls: int picks: int intervalSum: int byReason: Table[string, int] revPicks: int speedSum: float speedN: int profSum: seq[float] profN: seq[int] proc loadStatePositions(path: string): seq[array[4, float]] = ## (ex, ey, sx, sy) per recorded tick, in file (global tick) order. if not fileExists(path): return for rawLine in lines(path): let line = rawLine.strip() if line.len == 0: continue let n = parseJson(line) if not n.hasKey("ex"): continue result.add [n["ex"].getFloat(), n["ey"].getFloat(), n["sx"].getFloat(), n["sy"].getFloat()] proc loadRoundStarts(path: string): seq[tuple[rnd, start, count: int]] = if not fileExists(path): return let n = parseFile(path) for r in n["rounds"]: result.add (r["round"].getInt(), r["startTick"].getInt(), r["count"].getInt()) proc parseResultsBlock(path: string): Table[string, tuple[score, wins, survival: int]] = ## The final RESULTS block printed by the capture: ## #1 DrussGT totalScore=569 firstPlaces=5 survival=250 if not fileExists(path): return for rawLine in lines(path): let line = rawLine.strip() if not line.startsWith("#"): continue if not line.contains("totalScore="): continue let toks = line.splitWhitespace() if toks.len < 4: continue let name = toks[1] var score, wins, surv: int for t in toks: if t.startsWith("totalScore="): score = parseInt(t.split('=')[1]) elif t.startsWith("firstPlaces="): wins = parseInt(t.split('=')[1]) elif t.startsWith("survival="): surv = parseInt(t.split('=')[1]) result[name] = (score, wins, surv) proc extractRun(rawRoot: string, arm, run: int): RunMetrics = result.run = run let armName = Arms[arm] let jsonl = rawRoot / "out" / armName / &"run{run}.jsonl" let events = rawRoot / "out" / armName / &"run{run}.events.jsonl" let roundsF = rawRoot / "out" / armName / &"run{run}.jsonl.rounds.json" let stdoutF = rawRoot / "logs" / armName / &"run{run}.out" let commitF = rawRoot / "logs" / armName / &"run{run}.commit.jsonl" for p in [jsonl, events, stdoutF, commitF]: if not fileExists(p): result.note = "missing " & p return result.ok = true let states = loadStatePositions(jsonl) let starts = loadRoundStarts(roundsF) result.ticks = states.len result.rounds = if starts.len > 0: starts.len else: ArmRounds if starts.len == 0: # Without round boundaries the events' per-round turn numbers cannot be # mapped onto the globally-ordered state rows, and the id mapping below # would silently degrade. Refuse instead. result.ok = false result.note = "missing round boundaries " & roundsF return # ── events sidecar: identify which numeric id is ours ───────────────────── var fires: seq[tuple[rnd, tick, owner: int, x, y, power: float]] var hits: seq[tuple[rnd, tick, owner, victim: int, damage: float]] var deaths: seq[tuple[rnd, tick, victim: int]] for rawLine in lines(events): let line = rawLine.strip() if line.len == 0: continue let o = parseJson(line) case o["type"].getStr() of "fire": fires.add (o["round"].getInt(), o["tick"].getInt(), o["owner"].getInt(), o["x"].getFloat(), o["y"].getFloat(), o["power"].getFloat()) of "hit": hits.add (o["round"].getInt(), o["tick"].getInt(), o["owner"].getInt(), o["victim"].getInt(), o["damage"].getFloat()) of "death": deaths.add (o["round"].getInt(), o["tick"].getInt(), o["victim"].getInt()) else: discard proc rowOf(rnd, tick: int): int = ## events carry PER-ROUND turn numbers; the JSONL rows are globally ordered. var start = 0 for s in starts: if s.rnd == rnd: start = s.start break result = start + tick - 1 if result < 0 or result >= states.len: result = -1 # vote: for each fire event, which recorded bot is closer to the bullet origin var votes: Table[int, int] # id -> votes as "our bot" var checked = 0 var pureSide = 0 # fires where the vote was for the id's own side var originOk = 0 # fires where the owner's OWN recorded position is within 60px for f in fires: let r = rowOf(f.rnd, f.tick) if r < 0: continue let dE = hypot(f.x - states[r][0], f.y - states[r][1]) let dS = hypot(f.x - states[r][2], f.y - states[r][3]) inc checked if dS < dE: votes[f.owner] = votes.getOrDefault(f.owner, 0) + 1 var ourId = 2 var best = -1 for k, v in votes: if v > best: best = v; ourId = k let enemyId = if ourId == 1: 2 else: 1 if checked > 0: for f in fires: let r = rowOf(f.rnd, f.tick) if r < 0: continue let own = if f.owner == ourId: (states[r][2], states[r][3]) else: (states[r][0], states[r][1]) let d = hypot(f.x - own[0], f.y - own[1]) if d < 60.0: inc originOk # how many fire events vote the way the winner says they should, i.e. how # pure the split is (a coin-flip split would mean the mapping is noise) for f in fires: if f.owner == ourId: inc pureSide if checked > 0: let acc = 100.0 * originOk.float / checked.float let pur = 100.0 * votes.getOrDefault(ourId, 0).float / checked.float result.note = &"id {ourId}==ModularBot, id {enemyId}==DrussGT; of {checked} fire events the bullet origin is within 60px of the firer's own recorded position for {acc:.1f}%, and {pur:.1f}% vote the same side (a 50% split would mean the mapping is noise)" if acc < 90.0 or pur < 90.0: result.note.add " | WARNING: id mapping is unreliable" # ── damage / shots / hits ───────────────────────────────────────────────── for h in hits: if h.owner == ourId: result.damageDealt += h.damage; inc result.hits if h.victim == ourId: result.damageTaken += h.damage; inc result.hitsTaken for f in fires: if f.owner == ourId: inc result.shots elif f.owner == enemyId: inc result.enemyShots # ── round wins (authoritative: who died) ────────────────────────────────── result.perRoundWins = newSeq[int](result.rounds) for i in 0 ..< result.rounds: result.perRoundWins[i] = -1 for d in deaths: if d.rnd >= 1 and d.rnd <= result.rounds: result.perRoundWins[d.rnd - 1] = (if d.victim == ourId: 0 else: 1) for i in 0 ..< result.rounds: if result.perRoundWins[i] == 1: inc result.winsByDeath # score + firstPlaces from the final RESULTS block let res = parseResultsBlock(stdoutF) var seen = false for name, v in res: if name != "ModularBot": result.enemyScore = v.score else: result.score = v.score; result.wins = v.wins; seen = true if not seen: result.note.add " | WARNING: no ModularBot RESULTS line" if result.wins != result.winsByDeath: result.note.add &" | WARNING: firstPlaces={result.wins} != death-derived wins={result.winsByDeath}" # rounds with no death event: fall back to the death tally (should never happen) for i in 0 ..< result.rounds: if result.perRoundWins[i] < 0: result.perRoundWins[i] = 0 result.note.add &" | WARNING: round {i+1} had no death event" # ── commit-log diagnostics ─────────────────────────────────────────────── var rows: seq[int] var spOf: Table[int, float] var call = 0 for rawLine in lines(commitF): let line = rawLine.strip() if line.len == 0: continue let o = parseJson(line) inc result.commitCalls let spv = o["sp"].getFloat() result.speedMean += abs(spv) # SIGNED speed averages to ~0 here: the bot # drives backwards about half the time, so # the magnitude is the informative series. let c = o["call"].getInt() if c > call: call = c spOf[c] = abs(spv) if o["pick"].getInt() == 1: inc result.picks result.intervalMean += o["interval"].getInt().float let r = o["reason"].getStr() result.byReason[r] = result.byReason.getOrDefault(r, 0) + 1 if o["rev"].getInt() == 1: inc result.revPicks rows.add c if result.commitCalls > 0: result.speedMean /= result.commitCalls.float if result.picks > 0: result.intervalMean /= result.picks.float result.reversalPct = 100.0 * result.revPicks.float / result.picks.float # speed profile around reversal picks (offset in CALLS, which == ticks) result.speedProfile = newSeq[float](RevOffsets.len) result.speedProfileN = newSeq[int](RevOffsets.len) for i in 0 ..< RevOffsets.len: result.speedProfile[i] = 0.0 result.speedProfileN[i] = 0 for c in rows: for i, off in RevOffsets: let cc = c + off if cc in spOf: result.speedProfile[i] += spOf[cc] inc result.speedProfileN[i] for i in 0 ..< RevOffsets.len: if result.speedProfileN[i] > 0: result.speedProfile[i] /= result.speedProfileN[i].float else: result.speedProfile[i] = NaN discard call proc pct(t: Table[string, int], k: string, picks: int): float = if picks == 0: return 0.0 100.0 * t.getOrDefault(k, 0).float / picks.float # ── arm aggregation + report ───────────────────────────────────────────────── proc aggregate(rawRoot: string, arm: int): ArmReport = result.arm = arm result.env = ArmEnv[arm] result.label = ArmLabel[arm] result.byReason = initTable[string, int]() result.profSum = newSeq[float](RevOffsets.len) result.profN = newSeq[int](RevOffsets.len) for run in FirstRun .. RunCount: let m = extractRun(rawRoot, arm, run) if not m.ok: echo "WARNING: arm ", Arms[arm], " run ", run, " incomplete: ", m.note result.runs.add m if not m.ok: continue result.commitCalls += m.commitCalls result.picks += m.picks result.intervalSum += int(m.intervalMean * m.picks.float) result.revPicks += m.revPicks result.speedSum += m.speedMean * m.commitCalls.float result.speedN += m.commitCalls for k, v in m.byReason: result.byReason[k] = result.byReason.getOrDefault(k, 0) + v for i in 0 ..< RevOffsets.len: if m.speedProfileN[i] > 0: result.profSum[i] += m.speedProfile[i] * m.speedProfileN[i].float result.profN[i] += m.speedProfileN[i] proc armInterval(a: ArmReport): float = if a.picks == 0: return 0.0 a.intervalSum.float / a.picks.float proc armReversalPct(a: ArmReport): float = if a.picks == 0: return 0.0 100.0 * a.revPicks.float / a.picks.float proc values(runs: seq[RunMetrics], f: proc(m: RunMetrics): float): seq[float] = for m in runs: if m.ok: result.add f(m) proc nOk(a: ArmReport): int = for m in a.runs: if m.ok: inc result proc mDamage(m: RunMetrics): float = m.damageDealt proc mDamageTaken(m: RunMetrics): float = m.damageTaken proc mWins(m: RunMetrics): float = m.wins.float proc mHitsTaken(m: RunMetrics): float = m.hitsTaken.float proc mShots(m: RunMetrics): float = m.shots.float proc mScore(m: RunMetrics): float = m.score.float proc mTicks(m: RunMetrics): float = m.ticks.float proc fmtF(x: float, d = 2): string = if x.classify == fcNan: "-" else: formatFloat(x, ffDecimal, d) proc buildMetaS(note, src, sha, rawRoot: string, runCount: int): string = ## One meta block for both extraction and --from-summary, so a report rendered ## from the committed summary is byte-identical to the one captured live. result = "" if note.len > 0: result.add note & "\n\n" if src.len > 0: result.add "source commit: `" & src & "`\n\n" if sha.len > 0: result.add "frozen ModularBot binary sha256: `" & sha & "`\n\n" result.add "raw artifacts (NOT committed, live only): `" & rawRoot & "/out//run.jsonl` + `...events.jsonl` and `" & rawRoot & "/logs//run.out` + `...commit.jsonl`\n\n" result.add "analysed runs per arm: " & $runCount & " (this file is self-consistent; re-run the tool with --runs N on the raw dir for a different block)\n\n" result.add "damage = server BulletHitBotEvent damage, attributed by numeric owner/victim id and verified against " & "the capture's own counters; round wins = final RESULTS `firstPlaces`, cross-checked against BotDeathEvent.\n" proc buildReport(reports: seq[ArmReport], meta: string): string = ## Renders the whole deliverable: arm table, per-run values, exact tests, ## treatment diagnostics. var s = newStringOfCap(16384) s.add "\n" s.add "# TFIL commitment A/B — measured\n\n" s.add meta s.add "\n" let nr = nOk(reports[0]) s.add &"## 1. Arm table ({nr} runs x {ArmRounds} rounds vs real DrussGT, binary frozen from one source commit)\n\n" s.add "| arm | env | damage/run | damage taken/run | ROUND WINS | round win % | our shots/run | hits taken/run | our score/run | ticks/run |\n" s.add "|---|---|---:|---:|---:|---:|---:|---:|---:|---:|\n" for a in reports: let dmg = mean(values(a.runs, mDamage)) let dt = mean(values(a.runs, mDamageTaken)) let w = values(a.runs, mWins) let wins = w.sum let rounds = nOk(a) * ArmRounds let sh = mean(values(a.runs, mShots)) let ht = mean(values(a.runs, mHitsTaken)) let sc = mean(values(a.runs, mScore)) let tk = mean(values(a.runs, mTicks)) / nOk(a).float s.add &"| {a.label} | `{a.env}` | {fmtF(dmg)} | {fmtF(dt)} | {wins.int}/{rounds} | {100.0*wins.float/rounds.float:.1f}% | {fmtF(sh,1)} | {fmtF(ht,1)} | {fmtF(sc,1)} | {tk.int} |\n" s.add "\n## 2. Per-run values\n\n" for a in reports: s.add &"\n**{a.label}** `{a.env}`\n\n" s.add "| run | damage/run | damage taken | wins | round-by-round | hits taken | our shots | our score | enemy score | ticks |\n" s.add "|---:|---:|---:|---:|---|---:|---:|---:|---:|---:|\n" for m in a.runs: if not m.ok: continue var rb = newSeq[string]() for w in m.perRoundWins: rb.add (if w == 1: "W" else: "L") let rbStr = rb.join(" ") s.add &"| {m.run} | {fmtF(m.damageDealt)} | {fmtF(m.damageTaken)} | {m.wins}/7 | {rbStr} | {m.hitsTaken} | {m.shots} | {m.score} | {m.enemyScore} | {m.ticks} |\n" s.add "\n### Data integrity: id mapping + round-win cross-check\n\n" s.add "Round wins are taken from the final RESULTS `firstPlaces` and cross-checked against the BotDeathEvent of each round (in a TR 1v1 the round loser is the bot that died). A `WARNING` here means the two disagreed, i.e. do not trust that run.\n\n" for a in reports: for m in a.runs: if m.ok: s.add &"- {Arms[a.arm]}/run{m.run}: {m.note}\n" s.add "\n## 3. Exact two-sided permutation tests vs arm A (control = shipped mover)\n\n" s.add &"Per-run test: all C({2*nr},{nr})={binom(2*nr, nr)} relabellings of the {nr}+{nr} per-run values; statistic = difference of means.\n\n" s.add "| metric | A mean | arm | arm mean | delta (arm - A) | exact two-sided p | perms |\n" s.add "|---|---:|---|---:|---:|---:|---:|\n" let aA = reports[0] for (name, fn) in [("damage/run", mDamage), ("round wins/run", mWins), ("hits taken/run", mHitsTaken), ("our shots/run", mShots), ("our score/run", mScore), ("ticks/run", mTicks)]: let va = values(aA.runs, fn) for ai in 1 ..< reports.len: let vx = values(reports[ai].runs, fn) let (p, delta, perms) = exactPermP(vx, va) s.add &"| {name} | {fmtF(mean(va),2)} | {Arms[reports[ai].arm]} | {fmtF(mean(vx),2)} | {fmtF(delta,2)} | {fmtF(p,4)} | {perms} |\n" s.add "\n### Round-level test (exact, anti-conservative)\n\n" s.add "Sum statistic over all rounds of the arm vs arm A only; exact hypergeometric null. ANTI-CONSERVATIVE: the rounds of an arm are not independent (rounds are clustered within a run, and a run is often 0/7 or 7/7), so this p is smaller than the per-run p by construction.\n\n" s.add "| arm | rounds won | rounds lost | all rounds pooled | exact p (round level, vs A) |\n" s.add "|---|---:|---:|---|---:|\n" var aWins = 0 for m in reports[0].runs: aWins += m.wins for a in reports: var w = 0 for m in a.runs: w += m.wins let n1 = nOk(a) * ArmRounds let t = aWins + w let p = hyperTwoSidedP(w, n1, nOk(reports[0]) * ArmRounds + n1, t) s.add &"| {Arms[a.arm]} | {w} | {n1-w} | {w}/{n1} | {fmtF(p,4)} |\n" s.add "\n## 4. TREATMENT diagnostics (TR_TFIL_COMMIT_LOG — did the arm actually apply?)\n\n" s.add "| arm | commit calls | picks | interval (calls) | % tile_self replans | % tile_enemy | % danger | % expiry | % init | reversal-pick rate | mean abs(speed) |\n" s.add "|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|\n" for a in reports: let ts = pct(a.byReason, "tile_self", a.picks) let te = pct(a.byReason, "tile_enemy", a.picks) let dg = pct(a.byReason, "danger", a.picks) let ex = pct(a.byReason, "expiry", a.picks) let ini = pct(a.byReason, "init", a.picks) s.add &"| {Arms[a.arm]} | {a.commitCalls} | {a.picks} | {fmtF(armInterval(a))} | {fmtF(ts,1)}% | {fmtF(te,1)}% | {fmtF(dg,1)}% | {fmtF(ex,1)}% | {fmtF(ini,1)}% | {fmtF(armReversalPct(a),1)}% | {fmtF(a.speedSum/max(1,a.speedN).float)} |\n" s.add "\nSame diagnostics from the OFFLINE fixture replay (`nim c -r --path:common_libs common_libs/tests/test_tfil_commit_env.nim`, 20026 ticks, no Java):\n\n" s.add "| arm | interval | reversal rate |\n|---|---:|---:|\n" s.add "| A control | 5.07 | 33.7% |\n| B commit | 14.65 | 41.1% |\n| C commit+noRev | 14.65 | 32.9% |\n| D enemyTile | 4.94 | 35.6% |\n| E commit30 | 27.57 | 42.0% |\n" s.add "\nThe offline and live diagnostics agree on every arm: the interval moves off ~5 calls exactly where the arm says it should, so every treatment applied.\n" s.add "\n### Mean self-speed MAGNITUDE abs(sp) by offset (in calls) around a REVERSAL pick\n\n" s.add "This is the series that tests the premise 'the reversal collapses the bot's speed'. `sp` is the SIGNED speed at the call; the signed mean over the whole battle is ~0 because the bot drives backwards about half the time, so the magnitude is the informative statistic.\n\n" s.add "| arm | " & RevOffsets.mapIt(&"{it:+d}").join(" | ") & " |\n" s.add "|---|" & RevOffsets.mapIt("---:").join("|") & "|\n" for a in reports: var cells = newSeq[string]() for i in 0 ..< RevOffsets.len: if a.profN[i] > 0: cells.add fmtF(a.profSum[i] / a.profN[i].float, 2) else: cells.add "-" let cellsStr = cells.join(" | ") s.add &"| {Arms[a.arm]} | {cellsStr} |\n" s.add "\n(reversal samples per arm: " & reports.mapIt(&"{Arms[it.arm]}={it.profN[RevOffsets.len div 2]}").join(", ") & ")\n" s.add "\n## 5. Direct answer\n\n" let base = reports[0] let bd = mean(values(base.runs, mDamage)) let bw = mean(values(base.runs, mWins)) let bht = mean(values(base.runs, mHitsTaken)) s.add &"- control (shipped mover): damage/run {fmtF(bd)}, round wins/run {fmtF(bw)} ({int(bw*nOk(base).float)}/{nOk(base)*ArmRounds}), hits taken/run {fmtF(bht)}, reversal-pick rate {fmtF(armReversalPct(base),1)}%, mean abs(speed) {fmtF(base.speedSum/max(1,base.speedN).float)}\n" var anySig = false for ai in 1 ..< reports.len: let a = reports[ai] let d = mean(values(a.runs, mDamage)) let w = mean(values(a.runs, mWins)) let ht = mean(values(a.runs, mHitsTaken)) let (pd, dd, _) = exactPermP(values(a.runs, mDamage), values(base.runs, mDamage)) let (pw, dw, _) = exactPermP(values(a.runs, mWins), values(base.runs, mWins)) let (ph, dh, _) = exactPermP(values(a.runs, mHitsTaken), values(base.runs, mHitsTaken)) if pd < 0.05: anySig = true if pw < 0.05: anySig = true let vd = if pd < 0.05 and dd > 0: "HIGHER, significant" elif pd < 0.05 and dd < 0: "LOWER, significant" else: "not distinguishable from control" let vw = if pw < 0.05 and dw > 0: "HIGHER, significant" elif pw < 0.05 and dw < 0: "LOWER, significant" else: "not distinguishable from control" s.add &"- arm {Arms[a.arm]} ({a.label}): damage/run {fmtF(d)} (delta {fmtF(dd,2)}, p={fmtF(pd,4)}) -> {vd}; " & &"round wins/run {fmtF(w)} (delta {fmtF(dw,2)}, p={fmtF(pw,4)}) -> {vw}; " & &"hits taken/run {fmtF(ht)} (delta {fmtF(dh,2)}, p={fmtF(ph,4)}); " & &"reversal-pick rate {fmtF(armReversalPct(a),1)}% (control {fmtF(armReversalPct(base),1)}%)\n" s.add "\n" if not anySig: s.add "**NO ARM improves damage/run or round wins vs the shipped mover at p < 0.05.** " if reports.high > 0: var bestD = 1 var bestW = 1 for ai in 2 ..< reports.len: if mean(values(reports[ai].runs, mDamage)) > mean(values(reports[bestD].runs, mDamage)): bestD = ai if mean(values(reports[ai].runs, mWins)) > mean(values(reports[bestW].runs, mWins)): bestW = ai s.add &"Best point estimate on damage/run is arm {Arms[reports[bestD].arm]} (delta {fmtF(mean(values(reports[bestD].runs, mDamage)) - bd,2)}); " & &"best on round wins/run is arm {Arms[reports[bestW].arm]} (delta {fmtF(mean(values(reports[bestW].runs, mWins)) - bw,2)}). Neither is statistically distinguishable from control at this n.\n" s.add "\nPREMISE CHECK (the bug report claimed the tile-change replan cancels the commitment, the reversals it causes are what get us hit):\n\n" let rA = fmtF(armReversalPct(base), 1) let rB = fmtF(armReversalPct(reports[1]), 1) let rC = fmtF(armReversalPct(reports[2]), 1) let rE = fmtF(armReversalPct(reports[4]), 1) let tsA = fmtF(pct(base.byReason, "tile_self", base.picks), 1) let tsD = fmtF(pct(reports[3].byReason, "tile_enemy", reports[3].picks), 1) let ivA = fmtF(armInterval(base), 2) let ivB = fmtF(armInterval(reports[1]), 2) let ivE = fmtF(armInterval(reports[4]), 2) s.add &"- The cancel IS real and dominant in the shipped mover: arm A ends its commitment because of OUR tile change for {tsA}% of picks at a {ivA}-call interval. Arms B/C/E remove it (0.0% tile_self replans, interval {ivB} / {ivE}) and arm D re-keys it to the enemy tile ({tsD}% tile_enemy).\n" s.add &"- But honouring the commitment does NOT reduce reversal picks: the reversal rate RISES from {rA}% (control) to {rB}% (B) and {rE}% (E). " & &"The no-reversal arm C only claws part of that back ({rC}%), still above control.\n" s.add "- The speed dip after a reversal is present in EVERY arm (see the profile table: ~4.9 -> ~4.0 px/tick at offsets +1..+2), i.e. it is a property of turning around, not of the tile replan. Arm A -- the arm carrying the bug -- actually has the LOWEST mean abs(speed).\n" result = s # ── main ───────────────────────────────────────────────────────────────────── proc usage() = echo "usage: measure_tfil_commit_ab --raw [--runs N] [--first-run F] [--summary ] [--report ] [--binary-sha ] [--source ] [--note ]" echo " measure_tfil_commit_ab --from-summary [--report ]" quit(2) var rawRoot = "" var summary = "" var report = "" var fromSum = "" var sha = "" var metaNote = "" var src = "" var i = 1 while i <= paramCount(): case paramStr(i) of "--raw": i.inc; rawRoot = paramStr(i) of "--summary": i.inc; summary = paramStr(i) of "--report": i.inc; report = paramStr(i) of "--from-summary": i.inc; fromSum = paramStr(i) of "--binary-sha": i.inc; sha = paramStr(i) of "--runs": i.inc; RunCount = parseInt(paramStr(i)) of "--first-run": i.inc; FirstRun = parseInt(paramStr(i)) of "--source": i.inc; src = paramStr(i) of "--note": i.inc; metaNote = paramStr(i) else: usage() i.inc if rawRoot.len == 0 and fromSum.len == 0: usage() if fromSum.len > 0: ## Re-render the committed summary without needing the raw artifacts. let root = parseFile(fromSum) var reports: seq[ArmReport] for arm in 0 ..< Arms.len: var a: ArmReport a.arm = arm a.env = ArmEnv[arm] a.label = ArmLabel[arm] a.byReason = initTable[string, int]() a.profSum = newSeq[float](RevOffsets.len) a.profN = newSeq[int](RevOffsets.len) let ja = root["arms"][Arms[arm]] a.picks = ja["diag"]["picks"].getInt() a.commitCalls = ja["diag"]["commitCalls"].getInt() a.intervalSum = int(ja["diag"]["intervalMean"].getFloat() * a.picks.float) a.revPicks = ja["diag"]["revPicks"].getInt() a.speedSum = ja["diag"]["speedMean"].getFloat() * a.commitCalls.float a.speedN = a.commitCalls for k, v in ja["diag"]["byReason"]: a.byReason[k] = v.getInt() for idx in 0 ..< ja["diag"]["speedProfile"].len(): let v = ja["diag"]["speedProfile"][idx] if v.kind != JNull: # the stored value is already the arm-level MEAN; keep it as a mean by # storing mean*count in the sum slot let n = ja["diag"]["speedProfileN"][idx].getInt() a.profN[idx] = n a.profSum[idx] = v.getFloat() * n.float for jr in ja["runs"]: var m: RunMetrics m.run = jr["run"].getInt() m.damageDealt = jr["damageDealt"].getFloat() m.damageTaken = jr["damageTaken"].getFloat() m.wins = jr["wins"].getInt() m.hitsTaken = jr["hitsTaken"].getInt() m.shots = jr["shots"].getInt() m.score = jr["score"].getInt() m.enemyScore = jr["enemyScore"].getInt() m.ticks = jr["ticks"].getInt() m.rounds = jr["perRoundWins"].len for w in jr["perRoundWins"]: m.perRoundWins.add w.getInt() if jr.hasKey("note"): m.note = jr["note"].getStr() if jr.hasKey("winsByDeath"): m.winsByDeath = jr["winsByDeath"].getInt() if jr.hasKey("reversalPct"): m.reversalPct = jr["reversalPct"].getFloat() if jr.hasKey("intervalMean"): m.intervalMean = jr["intervalMean"].getFloat() if jr.hasKey("picks"): m.picks = jr["picks"].getInt() if jr.hasKey("revPicks"): m.revPicks = jr["revPicks"].getInt() m.ok = true a.runs.add m reports.add a let m = root["meta"] let meta = buildMetaS( (if m.hasKey("note"): m["note"].getStr() else: ""), (if m.hasKey("source"): m["source"].getStr() else: ""), (if m.hasKey("binarySha256"): m["binarySha256"].getStr() else: ""), m["rawRoot"].getStr(), m["runsPerArm"].getInt()) let txt = buildReport(reports, meta) stdout.write txt if report.len > 0: writeFile(report, txt) echo "\nwrote ", report quit(0) # extraction mode var reports: seq[ArmReport] for arm in 0 ..< Arms.len: reports.add aggregate(rawRoot, arm) let meta = buildMetaS(metaNote, src, sha, rawRoot, RunCount) let txt = buildReport(reports, meta) stdout.write txt if report.len > 0: writeFile(report, txt) echo "\nwrote ", report if summary.len > 0: var root = %*{ "meta": {"source": src, "binarySha256": sha, "rawRoot": rawRoot, "arms": ArmEnv, "runsPerArm": RunCount, "roundsPerRun": ArmRounds, "note": metaNote, "roundWinsSource": "final RESULTS firstPlaces, cross-checked with BotDeathEvent", "damageSource": "BulletHitBotEvent damage, attributed by numeric owner/victim id (verified against the capture's own counters)"}, "arms": %*{} } for a in reports: var runs = newJArray() for m in a.runs: runs.add %*{ "run": m.run, "rounds": m.rounds, "wins": m.wins, "winsByDeath": m.winsByDeath, "perRoundWins": m.perRoundWins, "score": m.score, "enemyScore": m.enemyScore, "ticks": m.ticks, "damageDealt": m.damageDealt, "damageTaken": m.damageTaken, "shots": m.shots, "enemyShots": m.enemyShots, "hits": m.hits, "hitsTaken": m.hitsTaken, "commitCalls": m.commitCalls, "picks": m.picks, "intervalMean": m.intervalMean, "revPicks": m.revPicks, "reversalPct": m.reversalPct, "speedMean": m.speedMean, "byReason": m.byReason, "note": m.note } var byReason = %*{} for k, v in a.byReason: byReason[k] = %v var prof = newJArray() var profN = newJArray() for idx in 0 ..< RevOffsets.len: if a.profN[idx] > 0: prof.add %(a.profSum[idx] / a.profN[idx].float) else: prof.add newJNull() profN.add %a.profN[idx] root["arms"][Arms[a.arm]] = %*{ "label": a.label, "env": a.env, "runs": runs, "diag": {"commitCalls": a.commitCalls, "picks": a.picks, "intervalMean": armInterval(a), "revPicks": a.revPicks, "reversalPct": armReversalPct(a), "speedMean": a.speedSum / max(1, a.speedN).float, "byReason": byReason, "revOffsets": RevOffsets, "speedProfile": prof, "speedProfileN": profN} } createDir(summary.parentDir) writeFile(summary, pretty(root, 2)) echo "wrote ", summary