/** * Component 18b — paired A/B measurement for bot heuristics. * * Dev-side only. Runs the current bot and one tweaked variant over the SAME deals and reports the * per-seed difference. * * WHY PAIRED, AND WHY IT IS NOT OPTIONAL. Revenue has a standard deviation of about 9 across games, * so two 100-game runs of the identical bot can differ by a point through nothing but the deal. Every * single-change claim in the changelog before the Interlocking work is inside that noise. Giving both * policies the same seed removes the deal from the comparison entirely: what is left is the change. * Measured here, the paired difference has σ ≈ 5.3 against ≈ 9 unpaired, and at 1600 seeds the * standard error is ±0.13 — so a +0.5 heuristic is resolvable in under two minutes, where unpaired it * would need tens of thousands of games. * * WHY THE SPLIT IS PRINTED. A mean carried by a skewed tail is a different claim from a mean carried * by broad improvement, and only the better/worse/identical counts tell them apart. The train cap is * the case in point: +0.60 overall, but 130 seeds better, 145 worse and 1325 unchanged — it removes a * rare catastrophe rather than making the bot play better. * * WHY THE FUNNEL IS PRINTED. Revenue can rise because a channel started working or because the bot * abandoned an expensive channel for a cheap one, and the two are identical in a single number. * * Run with: * node src/sim/compare.ts 1600 noTrainCap=1 — what the A/D cap is worth today * node src/sim/compare.ts 1600 noOperateFirst=1 — what operating before drawing is worth * * The flags are ABLATIONS: they turn off heuristics the bot already plays, so a negative delta is * the heuristic earning its place. That is what a measured bot needs going forward — the question * is no longer "does this help?" but "is this still true after the deck moved?". */ import { pump } from '../engine/advance.ts'; import { createGame } from '../engine/setup.ts'; import { DEFAULT_MAX_COLLISIONS_PER_DAY, DEFAULT_MAX_COLLISIONS_TOTAL, collectiveRevenueFloor, lengthProfile, } from '../engine/content.ts'; import type { GameLength } from '../engine/content.ts'; import type { GameConfig, GameMode } from '../engine/state.ts'; import type { BotPolicy, BotTweaks } from './bot.ts'; import { developerBot, makeDeveloperBot, playGame } from './bot.ts'; import type { Funnel, GameStats } from './stats.ts'; import { funnelReport, makeFunnelProbe, summarize } from './stats.ts'; export type PairedResult = { seeds: number; baseline: string; variant: string; /** Per-seed revenue difference, variant minus baseline. */ deltas: { seed: number; delta: number }[]; mean: number; /** Standard error of the mean difference — the number that decides whether this is real. */ stderr: number; sd: number; t: number; better: number; worse: number; identical: number; baseStats: GameStats[]; variantStats: GameStats[]; }; // Always one bot (`runOne` below), so the revenue floor is `collectiveRevenueFloor(1, days)`. const SOLO = (length: GameLength, mode: GameMode): GameConfig => { const days = lengthProfile(length).days; return { mode, days, minCombinedRevenue: collectiveRevenueFloor(1, days), maxCollisionsPerDay: DEFAULT_MAX_COLLISIONS_PER_DAY, maxCollisionsTotal: DEFAULT_MAX_COLLISIONS_TOTAL, pvpCardsAllowed: false, optionalRules: { reducedVisibility: false, sisterTrains: false, employeeRotation: false, emergencyToolbox: false, }, }; }; /** * One game, one seed, one policy. * * The seed stride matches `harness.ts` exactly, so a compare run and a harness run of the same size * are talking about the same games — otherwise two numbers that ought to agree would not, for a * reason nobody would find. */ function runOne(policy: BotPolicy, seed: number, length: GameLength, mode: GameMode): GameStats { const s = createGame({ id: `cmp-${seed}`, seed, config: SOLO(length, mode), playerNames: ['bot'] }); const probe = makeFunnelProbe(0); const r = playGame(s, policy, pump, 50_000, probe.onEvent, probe.onTurn); return summarize(seed, r.events, r.intents, s, probe.funnel); } export function compare( tweaks: BotTweaks, games: number, length: GameLength = 'standard', mode: GameMode = 'solitaire', ): PairedResult { const variantPolicy = makeDeveloperBot(tweaks); const baseStats: GameStats[] = []; const variantStats: GameStats[] = []; const deltas: { seed: number; delta: number }[] = []; for (let i = 0; i < games; i++) { const seed = 1000 + i * 7919; // the same prime stride the harness deals const a = runOne(developerBot, seed, length, mode); const b = runOne(variantPolicy, seed, length, mode); baseStats.push(a); variantStats.push(b); deltas.push({ seed, delta: b.revenue.net - a.revenue.net }); } const d = deltas.map((x) => x.delta); const mean = d.reduce((p, c) => p + c, 0) / Math.max(1, d.length); const sd = d.length > 1 ? Math.sqrt(d.reduce((p, c) => p + (c - mean) ** 2, 0) / (d.length - 1)) : 0; const stderr = d.length > 0 ? sd / Math.sqrt(d.length) : 0; return { seeds: games, baseline: developerBot.name, variant: variantPolicy.name, deltas, mean, sd, stderr, t: stderr === 0 ? 0 : mean / stderr, better: d.filter((x) => x > 0).length, worse: d.filter((x) => x < 0).length, identical: d.filter((x) => x === 0).length, baseStats, variantStats, }; } // --------------------------------------------------------------------------- // Reporting // --------------------------------------------------------------------------- const mean = (xs: number[]): number => (xs.length ? xs.reduce((p, c) => p + c, 0) / xs.length : 0); /** * The verdict, stated in the same terms every time. * * The threshold is t ≥ 3 rather than the conventional 2. This is a measurement taken repeatedly on * the same system while looking for something that works, so the conventional bar would have us keep * roughly one bad heuristic in twenty; and the cost of a wrong keep is not a wrong paper, it is a bot * that quietly plays worse and takes every later measurement with it. */ function verdict(t: number): string { const a = Math.abs(t); if (a >= 3) return t > 0 ? 'KEEP — clears the bar (t ≥ 3)' : 'REJECT — significantly worse'; if (a >= 2) return 'NOT PROVEN — suggestive, run more seeds before believing it'; return 'NO EFFECT MEASURED — inside the noise'; } export function formatPaired(r: PairedResult): string { const out: string[] = []; const num = (x: number, w = 8, dp = 2): string => x.toFixed(dp).padStart(w); out.push(`\n=== paired comparison · ${r.seeds} seeds · same deal to both ===\n`); out.push(` baseline ${r.baseline}`); out.push(` variant ${r.variant}\n`); const rows: [string, (g: GameStats) => number][] = [ ['revenue', (g) => g.revenue.net], ['collisions', (g) => g.collisions], ['trains scheduled', (g) => g.trains.scheduled], ['arrivals', (g) => g.trains.arrived], ['passengers on/off', (g) => g.revenue.passengerBoard + g.revenue.passengerDetrain], ['freight loads+unloads', (g) => g.revenue.freightLoad + g.revenue.freightUnload], ['cards played', (g) => g.development.cardsPlayed], ['final grid size', (g) => g.development.gridSize], ]; out.push(' baseline variant delta'); for (const [label, pick] of rows) { const a = mean(r.baseStats.map(pick)); const b = mean(r.variantStats.map(pick)); out.push(` ${label.padEnd(22)}${num(a)} ${num(b)} ${num(b - a)}`); } out.push( `\n REVENUE DELTA ${r.mean >= 0 ? '+' : ''}${r.mean.toFixed(2)} ± ${r.stderr.toFixed(2)}` + ` (t = ${r.t.toFixed(2)}, σ of the paired difference ${r.sd.toFixed(2)})`, ); out.push(` ${verdict(r.t)}`); out.push( `\n seeds better ${r.better} · worse ${r.worse} · identical ${r.identical}` + ` — ${((r.identical / Math.max(1, r.seeds)) * 100).toFixed(0)}% of games are untouched by this change`, ); // The tails, BY SEED, so a loss can be replayed rather than averaged away. const sorted = [...r.deltas].sort((a, b) => a.delta - b.delta); const show = (xs: { seed: number; delta: number }[]): string => xs.map((x) => `${x.seed} (${x.delta > 0 ? '+' : ''}${x.delta})`).join(', '); out.push(` worst seeds: ${show(sorted.slice(0, 3))}`); out.push(` best seeds: ${show(sorted.slice(-3).reverse())}`); // A change that raises revenue by abandoning a channel has to be visible as that. out.push('\n --- baseline funnel ---'); out.push(funnelReport(r.baseStats)); out.push('\n --- variant funnel ---'); out.push(funnelReport(r.variantStats)); return out.join('\n'); } // --------------------------------------------------------------------------- // CLI // --------------------------------------------------------------------------- /** * `noTrainCap=1` -> `{ noTrainCap: true }`. * * Unknown names THROW rather than being ignored: a typo parsed as "no tweaks" would compare the bot * against itself and report a confident zero, which is the most expensive way this tool could fail. */ export const NUMERIC_TWEAKS = new Set([]); export const BOOLEAN_TWEAKS = new Set(['noTrainCap', 'noOperateFirst']); export function parseTweaks(args: string[]): BotTweaks { const tweaks: Record = {}; for (const a of args) { const m = /^([A-Za-z]\w*)=(-?\d+(?:\.\d+)?)$/.exec(a); if (!m) continue; const name = m[1]!; if (NUMERIC_TWEAKS.has(name)) tweaks[name] = Number(m[2]); else if (BOOLEAN_TWEAKS.has(name)) tweaks[name] = Number(m[2]) !== 0; else { throw new Error( `unknown tweak "${name}" — known: ${[...NUMERIC_TWEAKS, ...BOOLEAN_TWEAKS].join(', ')}`, ); } } return tweaks as BotTweaks; } const isMain = process.argv[1]?.endsWith('compare.ts') ?? false; if (isMain) { const args = process.argv.slice(2); const games = Number(args.find((a) => /^\d+$/.test(a)) ?? 400); const lengthIdx = args.indexOf('--length'); const length = (lengthIdx >= 0 ? args[lengthIdx + 1] : 'standard') as GameLength; const tweaks = parseTweaks(args); if (Object.keys(tweaks).length === 0) { console.error('nothing to compare — pass at least one tweak, e.g. trainCapSlack=1'); process.exitCode = 1; } else { console.log(formatPaired(compare(tweaks, games, length))); } } export type { Funnel };