Files
station-master/src/sim/compare.ts
T

249 lines
9.9 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* Component 18b — paired A/B measurement for bot heuristics.
*
* Dev-side only. Runs the current bot and one tweaked variant over the SAME deals and reports the
* per-seed difference.
*
* WHY PAIRED, AND WHY IT IS NOT OPTIONAL. Revenue has a standard deviation of about 9 across games,
* so two 100-game runs of the identical bot can differ by a point through nothing but the deal. Every
* single-change claim in the changelog before the Interlocking work is inside that noise. Giving both
* policies the same seed removes the deal from the comparison entirely: what is left is the change.
* Measured here, the paired difference has σ ≈ 5.3 against ≈ 9 unpaired, and at 1600 seeds the
* standard error is ±0.13 — so a +0.5 heuristic is resolvable in under two minutes, where unpaired it
* would need tens of thousands of games.
*
* WHY THE SPLIT IS PRINTED. A mean carried by a skewed tail is a different claim from a mean carried
* by broad improvement, and only the better/worse/identical counts tell them apart. The train cap is
* the case in point: +0.60 overall, but 130 seeds better, 145 worse and 1325 unchanged — it removes a
* rare catastrophe rather than making the bot play better.
*
* WHY THE FUNNEL IS PRINTED. Revenue can rise because a channel started working or because the bot
* abandoned an expensive channel for a cheap one, and the two are identical in a single number.
*
* Run with:
* node src/sim/compare.ts 1600 noTrainCap=1 — what the A/D cap is worth today
* node src/sim/compare.ts 1600 noOperateFirst=1 — what operating before drawing is worth
*
* The flags are ABLATIONS: they turn off heuristics the bot already plays, so a negative delta is
* the heuristic earning its place. That is what a measured bot needs going forward — the question
* is no longer "does this help?" but "is this still true after the deck moved?".
*/
import { pump } from '../engine/advance.ts';
import { createGame } from '../engine/setup.ts';
import type { GameLength } from '../engine/content.ts';
import type { GameConfig, GameMode } from '../engine/state.ts';
import type { BotPolicy, BotTweaks } from './bot.ts';
import { developerBot, makeDeveloperBot, playGame } from './bot.ts';
import type { Funnel, GameStats } from './stats.ts';
import { funnelReport, makeFunnelProbe, summarize } from './stats.ts';
export type PairedResult = {
seeds: number;
baseline: string;
variant: string;
/** Per-seed revenue difference, variant minus baseline. */
deltas: { seed: number; delta: number }[];
mean: number;
/** Standard error of the mean difference — the number that decides whether this is real. */
stderr: number;
sd: number;
t: number;
better: number;
worse: number;
identical: number;
baseStats: GameStats[];
variantStats: GameStats[];
};
const SOLO = (length: GameLength, mode: GameMode): GameConfig => ({
mode,
victory: 'highestAfterDays',
length,
optionalRules: {
reducedVisibility: false,
sisterTrains: false,
employeeRotation: false,
emergencyToolbox: false,
},
});
/**
* One game, one seed, one policy.
*
* The seed stride matches `harness.ts` exactly, so a compare run and a harness run of the same size
* are talking about the same games — otherwise two numbers that ought to agree would not, for a
* reason nobody would find.
*/
function runOne(policy: BotPolicy, seed: number, length: GameLength, mode: GameMode): GameStats {
const s = createGame({ id: `cmp-${seed}`, seed, config: SOLO(length, mode), playerNames: ['bot'] });
const probe = makeFunnelProbe(0);
const r = playGame(s, policy, pump, 50_000, probe.onEvent, probe.onTurn);
return summarize(seed, r.events, r.intents, s, probe.funnel);
}
export function compare(
tweaks: BotTweaks,
games: number,
length: GameLength = 'standard',
mode: GameMode = 'solitaire',
): PairedResult {
const variantPolicy = makeDeveloperBot(tweaks);
const baseStats: GameStats[] = [];
const variantStats: GameStats[] = [];
const deltas: { seed: number; delta: number }[] = [];
for (let i = 0; i < games; i++) {
const seed = 1000 + i * 7919; // the same prime stride the harness deals
const a = runOne(developerBot, seed, length, mode);
const b = runOne(variantPolicy, seed, length, mode);
baseStats.push(a);
variantStats.push(b);
deltas.push({ seed, delta: b.revenue.net - a.revenue.net });
}
const d = deltas.map((x) => x.delta);
const mean = d.reduce((p, c) => p + c, 0) / Math.max(1, d.length);
const sd =
d.length > 1
? Math.sqrt(d.reduce((p, c) => p + (c - mean) ** 2, 0) / (d.length - 1))
: 0;
const stderr = d.length > 0 ? sd / Math.sqrt(d.length) : 0;
return {
seeds: games,
baseline: developerBot.name,
variant: variantPolicy.name,
deltas,
mean,
sd,
stderr,
t: stderr === 0 ? 0 : mean / stderr,
better: d.filter((x) => x > 0).length,
worse: d.filter((x) => x < 0).length,
identical: d.filter((x) => x === 0).length,
baseStats,
variantStats,
};
}
// ---------------------------------------------------------------------------
// Reporting
// ---------------------------------------------------------------------------
const mean = (xs: number[]): number => (xs.length ? xs.reduce((p, c) => p + c, 0) / xs.length : 0);
/**
* The verdict, stated in the same terms every time.
*
* The threshold is t ≥ 3 rather than the conventional 2. This is a measurement taken repeatedly on
* the same system while looking for something that works, so the conventional bar would have us keep
* roughly one bad heuristic in twenty; and the cost of a wrong keep is not a wrong paper, it is a bot
* that quietly plays worse and takes every later measurement with it.
*/
function verdict(t: number): string {
const a = Math.abs(t);
if (a >= 3) return t > 0 ? 'KEEP — clears the bar (t ≥ 3)' : 'REJECT — significantly worse';
if (a >= 2) return 'NOT PROVEN — suggestive, run more seeds before believing it';
return 'NO EFFECT MEASURED — inside the noise';
}
export function formatPaired(r: PairedResult): string {
const out: string[] = [];
const num = (x: number, w = 8, dp = 2): string => x.toFixed(dp).padStart(w);
out.push(`\n=== paired comparison · ${r.seeds} seeds · same deal to both ===\n`);
out.push(` baseline ${r.baseline}`);
out.push(` variant ${r.variant}\n`);
const rows: [string, (g: GameStats) => number][] = [
['revenue', (g) => g.revenue.net],
['collisions', (g) => g.collisions],
['trains scheduled', (g) => g.trains.scheduled],
['arrivals', (g) => g.trains.arrived],
['passengers on/off', (g) => g.revenue.passengerBoard + g.revenue.passengerDetrain],
['freight loads+unloads', (g) => g.revenue.freightLoad + g.revenue.freightUnload],
['cards played', (g) => g.development.cardsPlayed],
['final grid size', (g) => g.development.gridSize],
];
out.push(' baseline variant delta');
for (const [label, pick] of rows) {
const a = mean(r.baseStats.map(pick));
const b = mean(r.variantStats.map(pick));
out.push(` ${label.padEnd(22)}${num(a)} ${num(b)} ${num(b - a)}`);
}
out.push(
`\n REVENUE DELTA ${r.mean >= 0 ? '+' : ''}${r.mean.toFixed(2)} ± ${r.stderr.toFixed(2)}` +
` (t = ${r.t.toFixed(2)}, σ of the paired difference ${r.sd.toFixed(2)})`,
);
out.push(` ${verdict(r.t)}`);
out.push(
`\n seeds better ${r.better} · worse ${r.worse} · identical ${r.identical}` +
` — ${((r.identical / Math.max(1, r.seeds)) * 100).toFixed(0)}% of games are untouched by this change`,
);
// The tails, BY SEED, so a loss can be replayed rather than averaged away.
const sorted = [...r.deltas].sort((a, b) => a.delta - b.delta);
const show = (xs: { seed: number; delta: number }[]): string =>
xs.map((x) => `${x.seed} (${x.delta > 0 ? '+' : ''}${x.delta})`).join(', ');
out.push(` worst seeds: ${show(sorted.slice(0, 3))}`);
out.push(` best seeds: ${show(sorted.slice(-3).reverse())}`);
// A change that raises revenue by abandoning a channel has to be visible as that.
out.push('\n --- baseline funnel ---');
out.push(funnelReport(r.baseStats));
out.push('\n --- variant funnel ---');
out.push(funnelReport(r.variantStats));
return out.join('\n');
}
// ---------------------------------------------------------------------------
// CLI
// ---------------------------------------------------------------------------
/**
* `noTrainCap=1` -> `{ noTrainCap: true }`.
*
* Unknown names THROW rather than being ignored: a typo parsed as "no tweaks" would compare the bot
* against itself and report a confident zero, which is the most expensive way this tool could fail.
*/
export const NUMERIC_TWEAKS = new Set<string>([]);
export const BOOLEAN_TWEAKS = new Set(['noTrainCap', 'noOperateFirst']);
export function parseTweaks(args: string[]): BotTweaks {
const tweaks: Record<string, number | boolean> = {};
for (const a of args) {
const m = /^([A-Za-z]\w*)=(-?\d+(?:\.\d+)?)$/.exec(a);
if (!m) continue;
const name = m[1]!;
if (NUMERIC_TWEAKS.has(name)) tweaks[name] = Number(m[2]);
else if (BOOLEAN_TWEAKS.has(name)) tweaks[name] = Number(m[2]) !== 0;
else {
throw new Error(
`unknown tweak "${name}" — known: ${[...NUMERIC_TWEAKS, ...BOOLEAN_TWEAKS].join(', ')}`,
);
}
}
return tweaks as BotTweaks;
}
const isMain = process.argv[1]?.endsWith('compare.ts') ?? false;
if (isMain) {
const args = process.argv.slice(2);
const games = Number(args.find((a) => /^\d+$/.test(a)) ?? 400);
const lengthIdx = args.indexOf('--length');
const length = (lengthIdx >= 0 ? args[lengthIdx + 1] : 'standard') as GameLength;
const tweaks = parseTweaks(args);
if (Object.keys(tweaks).length === 0) {
console.error('nothing to compare — pass at least one tweak, e.g. trainCapSlack=1');
process.exitCode = 1;
} else {
console.log(formatPaired(compare(tweaks, games, length)));
}
}
export type { Funnel };