/** * The measurement machinery itself. * * These tests exist because a measurement tool that is quietly wrong is worse than no tool: it does * not fail, it just produces numbers that decide what the bot becomes. Each one guards a property * the paired method depends on. */ import { describe, it } from 'node:test'; import assert from 'node:assert/strict'; import { pump } from '../src/engine/advance.ts'; import { createGame } from '../src/engine/setup.ts'; import { adTrackCount } from '../src/engine/state.ts'; import type { GameConfig, GameState } from '../src/engine/state.ts'; import type { BotPolicy } from '../src/sim/bot.ts'; import { developerBot, makeDeveloperBot, playGame } from '../src/sim/bot.ts'; import { compare, parseTweaks } from '../src/sim/compare.ts'; import { makeFunnelProbe, summarize } from '../src/sim/stats.ts'; const config: GameConfig = { mode: 'solitaire', victory: 'highestAfterDays', length: 'standard', optionalRules: { reducedVisibility: false, sisterTrains: false, employeeRotation: false, emergencyToolbox: false, }, }; const game = (seed: number): GameState => createGame({ id: `h${seed}`, seed, config, playerNames: ['bot'] }); /** One game, reduced to a string: every event type in order, plus the final revenue. */ function fingerprint(policy: BotPolicy, seed: number): string { const s = game(seed); const r = playGame(s, policy, pump); return `${r.revenue[0]}|${r.events.map((e) => e.type).join(',')}`; } describe('the paired method rests on determinism', () => { it('plays the identical game twice from one seed', () => { /** * THE PROPERTY EVERYTHING ELSE DEPENDS ON. A paired comparison subtracts two games that share a * deal; if the same policy on the same seed can diverge, the difference is measuring the engine's * own noise and every heuristic verdict is worthless. * * It is also the engine's central claim — pure, seeded, no ambient randomness — and the failure * mode is silent: one `Math.random`, one Map iterated in a different order, and this is the only * thing that would notice. */ for (const seed of [1000, 8919, 775569289]) { assert.equal(fingerprint(developerBot, seed), fingerprint(developerBot, seed), `seed ${seed} diverged`); } }); it('gives the untweaked variant the identical game too', () => { // `developerBot` IS `makeDeveloperBot({})`, and the refactor that introduced tweaks must not // have changed a single decision. If this fails, every number measured before it is unusable. for (const seed of [1000, 8919, 775569289]) { assert.equal( fingerprint(developerBot, seed), fingerprint(makeDeveloperBot({}), seed), `seed ${seed}: the untweaked variant plays a different game`, ); } }); }); describe('a tweak has to actually do something', () => { it('actually holds the trains it says it holds', () => { /** * REGRESSION, and the reason this file exists. The first `trainCapSlack` gated the two branches * whose comments say they exist to play a train card — and measured **exactly zero difference * over 400 paired seeds**, because `followThrough` ends with a generic "play what is in hand" * fallback that played the card anyway. * * ASSERT THE CONTRACT, NOT MERELY A DIFFERENCE. A first version of this test only checked that * some decision changed somewhere, and it passed against the broken bot: with a tight cap the * gated branches do change which Local Operations option is chosen, they just fail to stop the * card being played two steps later. "Something moved" is not the promise. The promise is that * committed trains never exceed what the Office can hold. * * Office tiers only ever go up, so the final A/D count is the most generous the cap ever was. */ const committed = (s: GameState): number => s.timetable.filter((t) => t !== null).length + s.pendingExtras.length; for (const slack of [0, 1]) { const capped = makeDeveloperBot({ trainCapSlack: slack }); let cappedWorst = 0; let baseWorst = 0; for (const seed of [1000, 8919, 16838, 24757, 32676, 40595, 48514]) { const a = game(seed); playGame(a, capped, pump); const roomA = adTrackCount(a, 0) + slack; cappedWorst = Math.max(cappedWorst, committed(a) - roomA); const b = game(seed); playGame(b, developerBot, pump); baseWorst = Math.max(baseWorst, committed(b) - (adTrackCount(b, 0) + slack)); } assert.ok( cappedWorst <= 0, `trainCapSlack=${slack} let the bot commit ${cappedWorst} train(s) more than the Office can hold`, ); // And the cap is not vacuous: the uncapped bot really does overshoot on these seeds. assert.ok( baseWorst > 0, `slack=${slack}: the baseline never overshoots on these seeds, so the test proves nothing`, ); } }); it('names itself so a report cannot confuse two runs', () => { assert.equal(developerBot.name, 'developer'); assert.equal(makeDeveloperBot({}).name, 'developer'); assert.equal(makeDeveloperBot({ trainCapSlack: 1 }).name, 'developer+trainCapSlack=1'); }); it('refuses a tweak name it does not know', () => { // A typo silently parsed as "no tweaks" would compare the bot against itself and report a // confident zero — the most expensive possible failure of this tool. assert.deepEqual(parseTweaks(['400', 'trainCapSlack=2']), { trainCapSlack: 2 }); assert.throws(() => parseTweaks(['trainCapSlok=1']), /unknown tweak/); }); }); describe('the funnel counts what it claims to count', () => { it('keeps every gate inside the total it is a fraction of', () => { const s = game(430); const probe = makeFunnelProbe(0); const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn); const f = probe.funnel; assert.ok(f.arrivals > 0, 'no train reached an Office in a whole game'); for (const [name, n] of [ ['at a passenger office', f.arrivalsAtPassengerOffice], ['with an empty coach', f.arrivalsWithEmptyCoach], ['with a loaded coach', f.arrivalsWithLoadedCoach], ['with a wanted car', f.arrivalsWithWantedCar], ] as const) { assert.ok(n <= f.arrivals, `${name} (${n}) exceeds arrivals (${f.arrivals})`); } assert.ok(f.cargoPhases > 0, 'no Cargo phase was sampled'); for (const [name, n] of [ ['with a facility', f.cargoWithFacility], ['green stocked', f.cargoGreenStocked], ['car spotted', f.cargoCarSpotted], ['laborer free', f.cargoLaborerFree], ['ready', f.cargoReady], ] as const) { assert.ok(n <= f.cargoPhases, `${name} (${n}) exceeds Cargo phases (${f.cargoPhases})`); } // "All three at one industry" cannot exceed any of the three it is made of. assert.ok(f.cargoReady <= Math.min(f.cargoGreenStocked, f.cargoCarSpotted, f.cargoLaborerFree)); assert.ok(f.buriedWithDigAvailable <= f.buriedTurns); // Sampled once per Cargo phase, never once per decision: 5 Days x 12 Stages is the ceiling. assert.ok(f.cargoPhases <= 60, `${f.cargoPhases} Cargo phases in a 5-Day game`); // And the arrivals it counted are the arrivals that happened. const arrived = r.events.filter((e) => e.type === 'trainArrived').length; assert.equal(f.arrivals, arrived, 'the probe and the event log disagree about arrivals'); }); it('rides along on the harness summary', () => { const s = game(202); const probe = makeFunnelProbe(0); const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn); const stats = summarize(202, r.events, r.intents, s, probe.funnel); assert.ok(stats.funnel, 'summarize dropped the funnel'); assert.equal(stats.funnel!.arrivals, probe.funnel.arrivals); // Optional, so a caller with no probe still gets a summary. assert.equal(summarize(202, r.events, r.intents, s).funnel, undefined); }); }); describe('the comparison arithmetic', () => { it('reports a dead heat as a dead heat', () => { // Comparing the bot with itself must produce exactly zero, no games differing, and a t of 0 — // if the machinery has any asymmetry in it, this is where it shows. const r = compare({}, 12); assert.equal(r.mean, 0); assert.equal(r.identical, 12); assert.equal(r.better, 0); assert.equal(r.worse, 0); assert.equal(r.t, 0); }); it('pairs by seed, and deals the same seeds the harness does', () => { const r = compare({ trainCapSlack: 1 }, 5); assert.deepEqual( r.deltas.map((d) => d.seed), [0, 1, 2, 3, 4].map((i) => 1000 + i * 7919), 'compare and the harness must talk about the same games', ); for (let i = 0; i < r.deltas.length; i++) { assert.equal(r.baseStats[i]!.seed, r.deltas[i]!.seed); assert.equal(r.variantStats[i]!.seed, r.deltas[i]!.seed); assert.equal(r.deltas[i]!.delta, r.variantStats[i]!.revenue.net - r.baseStats[i]!.revenue.net); } }); it('computes the standard error from the PAIRED difference', () => { // The whole gain of pairing is that σ of the difference is smaller than σ of either side. Using // the level's σ by mistake would quietly restore the ±1.0 noise floor this exists to escape. const r = compare({ trainCapSlack: 1 }, 40); const d = r.deltas.map((x) => x.delta); const m = d.reduce((p, c) => p + c, 0) / d.length; const sd = Math.sqrt(d.reduce((p, c) => p + (c - m) ** 2, 0) / (d.length - 1)); assert.ok(Math.abs(r.mean - m) < 1e-9); assert.ok(Math.abs(r.sd - sd) < 1e-9); assert.ok(Math.abs(r.stderr - sd / Math.sqrt(d.length)) < 1e-9); }); });