/** * The measurement machinery itself. * * These tests exist because a measurement tool that is quietly wrong is worse than no tool: it does * not fail, it just produces numbers that decide what the bot becomes. Each one guards a property * the paired method depends on. */ import { describe, it } from 'node:test'; import assert from 'node:assert/strict'; import { existsSync, readFileSync, readdirSync } from 'node:fs'; import { dirname, join } from 'node:path'; import { fileURLToPath } from 'node:url'; import { pump } from '../src/engine/advance.ts'; import { createGame } from '../src/engine/setup.ts'; import { adTrackCount } from '../src/engine/state.ts'; import type { GameConfig, GameState } from '../src/engine/state.ts'; import type { BotPolicy } from '../src/sim/bot.ts'; import { developerBot, makeDeveloperBot, playGame } from '../src/sim/bot.ts'; import { compare, parseTweaks } from '../src/sim/compare.ts'; import { applyIntent } from '../src/engine/apply.ts'; import { legalActions } from '../src/engine/legal.ts'; import { ROLLING_STOCK_SUPPLY } from '../src/engine/content.ts'; import { fromSave } from '../src/web/game.ts'; import { makeFunnelProbe, summarize } from '../src/sim/stats.ts'; const root = join(dirname(fileURLToPath(import.meta.url)), '..'); const config: GameConfig = { mode: 'solitaire', days: 5, minCombinedRevenue: 0, maxCollisionsPerDay: 0, maxCollisionsTotal: 0, pvpCardsAllowed: false, optionalRules: { reducedVisibility: false, employeeRotation: false, emergencyToolbox: false, }, }; const game = (seed: number): GameState => createGame({ id: `h${seed}`, seed, config, playerNames: ['bot'] }); /** One game, reduced to a string: every event type in order, plus the final revenue. */ function fingerprint(policy: BotPolicy, seed: number): string { const s = game(seed); const r = playGame(s, policy, pump); return `${r.revenue[0]}|${r.events.map((e) => e.type).join(',')}`; } describe('the paired method rests on determinism', () => { it('plays the identical game twice from one seed', () => { /** * THE PROPERTY EVERYTHING ELSE DEPENDS ON. A paired comparison subtracts two games that share a * deal; if the same policy on the same seed can diverge, the difference is measuring the engine's * own noise and every heuristic verdict is worthless. * * It is also the engine's central claim — pure, seeded, no ambient randomness — and the failure * mode is silent: one `Math.random`, one Map iterated in a different order, and this is the only * thing that would notice. */ for (const seed of [1000, 8919, 775569289]) { assert.equal(fingerprint(developerBot, seed), fingerprint(developerBot, seed), `seed ${seed} diverged`); } }); it('gives the untweaked variant the identical game too', () => { // `developerBot` IS `makeDeveloperBot({})`, and the refactor that introduced tweaks must not // have changed a single decision. If this fails, every number measured before it is unusable. for (const seed of [1000, 8919, 775569289]) { assert.equal( fingerprint(developerBot, seed), fingerprint(makeDeveloperBot({}), seed), `seed ${seed}: the untweaked variant plays a different game`, ); } }); }); describe('a tweak has to actually do something', () => { it('actually holds the trains the Office cannot hold', () => { /** * REGRESSION, and the reason this file exists. The first `trainCapSlack` gated the two branches * whose comments say they exist to play a train card — and measured **exactly zero difference * over 400 paired seeds**, because `followThrough` ends with a generic "play what is in hand" * fallback that played the card anyway. * * ASSERT THE CONTRACT, NOT MERELY A DIFFERENCE. A first version of this test only checked that * some decision changed somewhere, and it passed against the broken bot: with a tight cap the * gated branches do change which Local Operations option is chosen, they just fail to stop the * card being played two steps later. "Something moved" is not the promise. The promise is that * committed trains never exceed what the Office can hold. * * Office tiers only ever go up, so the final A/D count is the most generous the cap ever was. */ const committed = (s: GameState): number => s.timetable.filter((t) => t !== null).length + s.pendingExtras.length; const uncapped = makeDeveloperBot({ noTrainCap: true }); let cappedWorst = 0; let ablatedWorst = 0; for (const seed of [1000, 8919, 16838, 24757, 32676, 40595, 48514]) { const a = game(seed); playGame(a, developerBot, pump); cappedWorst = Math.max(cappedWorst, committed(a) - adTrackCount(a, 0)); const b = game(seed); playGame(b, uncapped, pump); ablatedWorst = Math.max(ablatedWorst, committed(b) - adTrackCount(b, 0)); } assert.ok( cappedWorst <= 0, `the bot committed ${cappedWorst} train(s) more than the Office can hold`, ); // And the cap is not vacuous: with it ablated, the bot really does overshoot on these seeds. assert.ok(ablatedWorst > 0, 'the ablated bot never overshoots either, so this test proves nothing'); }); it('names itself so a report cannot confuse two runs', () => { assert.equal(developerBot.name, 'developer'); assert.equal(makeDeveloperBot({}).name, 'developer'); assert.equal(makeDeveloperBot({ noTrainCap: true }).name, 'developer+noTrainCap=true'); }); it('refuses a tweak name it does not know', () => { // A typo silently parsed as "no tweaks" would compare the bot against itself and report a // confident zero — the most expensive possible failure of this tool. assert.deepEqual(parseTweaks(['400', 'noTrainCap=1']), { noTrainCap: true }); assert.throws(() => parseTweaks(['noTrainCapp=1']), /unknown tweak/); }); }); describe('the funnel counts what it claims to count', () => { it('keeps every gate inside the total it is a fraction of', () => { const s = game(430); const probe = makeFunnelProbe(0); const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn); const f = probe.funnel; assert.ok(f.arrivals > 0, 'no train reached an Office in a whole game'); for (const [name, n] of [ ['at a passenger office', f.arrivalsAtPassengerOffice], ['with an empty coach', f.arrivalsWithEmptyCoach], ['with a loaded coach', f.arrivalsWithLoadedCoach], ['with a wanted car', f.arrivalsWithWantedCar], ] as const) { assert.ok(n <= f.arrivals, `${name} (${n}) exceeds arrivals (${f.arrivals})`); } assert.ok(f.cargoPhases > 0, 'no Cargo phase was sampled'); for (const [name, n] of [ ['with a facility', f.cargoWithFacility], ['green stocked', f.cargoGreenStocked], ['car spotted', f.cargoCarSpotted], ['laborer free', f.cargoLaborerFree], ['ready', f.cargoReady], ] as const) { assert.ok(n <= f.cargoPhases, `${name} (${n}) exceeds Cargo phases (${f.cargoPhases})`); } // "All three at one industry" cannot exceed any of the three it is made of. assert.ok(f.cargoReady <= Math.min(f.cargoGreenStocked, f.cargoCarSpotted, f.cargoLaborerFree)); assert.ok(f.buriedWithDigAvailable <= f.buriedTurns); // Sampled once per Cargo phase, never once per decision: 5 Days x 12 Stages is the ceiling. assert.ok(f.cargoPhases <= 60, `${f.cargoPhases} Cargo phases in a 5-Day game`); // And the arrivals it counted are the arrivals that happened. const arrived = r.events.filter((e) => e.type === 'trainArrived').length; assert.equal(f.arrivals, arrived, 'the probe and the event log disagree about arrivals'); }); it('rides along on the harness summary', () => { const s = game(202); const probe = makeFunnelProbe(0); const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn); const stats = summarize(202, r.events, r.intents, s, probe.funnel); assert.ok(stats.funnel, 'summarize dropped the funnel'); assert.equal(stats.funnel!.arrivals, probe.funnel.arrivals); // Optional, so a caller with no probe still gets a summary. assert.equal(summarize(202, r.events, r.intents, s).funnel, undefined); }); }); describe('the game conserves Rolling Stock', () => { it('creates no car that was not dealt, and destroys none but in a collision', () => { /** * REGRESSION, found by audit rather than by a failing test — which is why this one exists. * * `TODO.md` had it as an open question: a census of every holder came to 92 against the 80 cars * dealt, "not necessarily duplication, because some of those objects are cargo in transit". It * was duplication, and it came from two rules the engine had not implemented: * * §9.3 unload — "a load on the industry's track AND AN EMPTY CAR OF THAT TYPE IN THE * DIVISION YARD ... replaces the load with an empty car of that type" * §9.2 de-train — "a white empty coach IN THE DIVISION YARD ... replace the blue coach on the * train with the white one" * * Both requirements were unchecked and both replacement cars were conjured rather than taken, so * every unload and every de-training minted a car — 1.29 a game against a supply of 80, which is * the number `ROLLING_STOCK_SUPPLY` exists to control. * * Censused after every batch, because that is the only way this class of bug shows up at all. */ const supply = ROLLING_STOCK_SUPPLY.reduce((n, r) => n + r.loaded + r.empty, 0); const census = (st: GameState): number => { let n = st.yards.divisionYard.length + st.yards.classificationYard.length; for (const t of st.trays.values()) n += t.consist.length; for (const [, area] of st.officeAreas) { for (const card of area.grid.values()) { n += card.standing.length; const f = card.facility; if (!f) continue; n += f.industryTrack.cars.length + f.outboundBox.length + f.inboundBox.length; n += (f.menAtWork ?? []).filter((m) => m !== null).length; } } return n; }; for (const seed of [1000, 8919, 16838, 24757, 32676, 40595]) { const s = game(seed); assert.equal(census(s), supply, `seed ${seed}: the deal itself is short`); let expected = supply; for (let t = 0; t < 50_000; t++) { const before = census(s); const pumped = pump(s); /** * A COLLISION DESTROYS NO CAR, and this used to assume it destroyed all of them. * * The subtraction that stood here — `expected -= tr.consist.length` for every train in a * `trainsDestroyed` event — describes a rule the engine does not have. Gap 2c (`advance.ts`, * "TAKE THE WRECK OFF THE CARD") sends the wreck's cabooses back to the Division Yard and * everything else to Classification, so the stock is conserved through a collision like any * other move. The train is destroyed; its cars are not. * * It passed for as long as it did because none of the six seeds below ever collided, so the * branch never ran. Changing the deck to the sheet's counts (Gitea#14) moved the deals, seed * 24757 collided, and the test failed claiming the engine had CONJURED three cars — the * exact opposite of what had happened. * * So the census is now held flat, unconditionally, which is both the real invariant and a * stronger test than the one it replaces: there is no longer any event that excuses a * change, and `expected` cannot drift away from the supply it was dealt. */ assert.equal( census(s), before, `seed ${seed}: the engine changed the census by ${census(s) - before} while pumping`, ); if (s.status === 'finished') break; const actor = s.clock.pendingDecision !== null ? s.clock.superintendent : s.clock.currentActor; if (actor === null) break; const options = legalActions(s, actor); if (options.length === 0) break; const was = census(s); const r = applyIntent(s, actor, developerBot.choose(s, actor, options)); if (!r.ok) break; assert.equal( census(s), was, `seed ${seed}: a player action changed the Rolling Stock census by ${census(s) - was}`, ); } assert.equal(census(s), expected, `seed ${seed}: ended holding ${census(s)} of an expected ${expected}`); } }); }); describe('every published replay actually replays', () => { it('reaches the end of its own history', () => { /** * REGRESSION, and it had already bitten. `TODO.md` records both published replays going dead * without anyone noticing: `fromSave` stops at the first intent the rules no longer accept and * returns a SHORTER game, which on screen looks exactly like a game that ended early. Measured * when this test was written, all three published replays managed **2 intents of roughly 400** * — the site was serving three recordings of nothing. * * A replay is a save from a particular ruleset, so this is really a test that the rules have not * moved under the files in `public/replays`. When it fails, re-record rather than edit: * `node src/sim/save-replay.ts 400 --top 3`. */ const dir = join(root, 'public/replays'); const files = existsSync(dir) ? readdirSync(dir).filter((f) => f.endsWith('.json') && f !== 'manifest.json') : []; assert.ok(files.length > 0, 'no replays are published at all'); for (const f of files) { const save = JSON.parse(readFileSync(join(dir, f), 'utf8')) as { seed: number; history: unknown[]; rules?: unknown; }; // The WHOLE save, `rules` included. Rebuilding it from seed and history alone threw away the // one field that says which ruleset the file was recorded under, so this replayed every // published file under the pre-dialog defaults no matter what it said — a test that would // pass a genuinely dead replay the moment the defaults and the file disagreed. const back = fromSave(save as never); assert.equal( back.history.length, save.history.length, `${f} is dead — it replays ${back.history.length} of ${save.history.length} intents. ` + 'Re-record it with save-replay.ts rather than editing the file.', ); /** * `awaitingExtension` counts as the end since Gitea#11 — and for a published file it is the * EXPECTED end. These saves were recorded before extended play existed, so their histories * carry no `game.extend` vote: replaying one runs the timetable out and stops on the question * nobody was there to answer. That is a game that reached the end of its own history, which is * what this test is about. `active` would still be a dead replay. */ assert.ok( back.state.status === 'finished' || back.state.status === 'awaitingExtension', `${f} does not reach the end of its game — status ${back.state.status}`, ); assert.ok(back.state.official !== null, `${f} ends without recording a result`); } }); }); describe('the comparison arithmetic', () => { it('reports a dead heat as a dead heat', () => { // Comparing the bot with itself must produce exactly zero, no games differing, and a t of 0 — // if the machinery has any asymmetry in it, this is where it shows. const r = compare({}, 12); assert.equal(r.mean, 0); assert.equal(r.identical, 12); assert.equal(r.better, 0); assert.equal(r.worse, 0); assert.equal(r.t, 0); }); it('pairs by seed, and deals the same seeds the harness does', () => { const r = compare({ noTrainCap: true }, 5); assert.deepEqual( r.deltas.map((d) => d.seed), [0, 1, 2, 3, 4].map((i) => 1000 + i * 7919), 'compare and the harness must talk about the same games', ); for (let i = 0; i < r.deltas.length; i++) { assert.equal(r.baseStats[i]!.seed, r.deltas[i]!.seed); assert.equal(r.variantStats[i]!.seed, r.deltas[i]!.seed); assert.equal(r.deltas[i]!.delta, r.variantStats[i]!.revenue.net - r.baseStats[i]!.revenue.net); } }); it('computes the standard error from the PAIRED difference', () => { // The whole gain of pairing is that σ of the difference is smaller than σ of either side. Using // the level's σ by mistake would quietly restore the ±1.0 noise floor this exists to escape. const r = compare({ noTrainCap: true }, 40); const d = r.deltas.map((x) => x.delta); const m = d.reduce((p, c) => p + c, 0) / d.length; const sd = Math.sqrt(d.reduce((p, c) => p + (c - m) ** 2, 0) / (d.length - 1)); assert.ok(Math.abs(r.mean - m) < 1e-9); assert.ok(Math.abs(r.sd - sd) < 1e-9); assert.ok(Math.abs(r.stderr - sd / Math.sqrt(d.length)) < 1e-9); }); });