Files
station-master/test/harness.test.ts
T

220 lines
9.6 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* The measurement machinery itself.
*
* These tests exist because a measurement tool that is quietly wrong is worse than no tool: it does
* not fail, it just produces numbers that decide what the bot becomes. Each one guards a property
* the paired method depends on.
*/
import { describe, it } from 'node:test';
import assert from 'node:assert/strict';
import { pump } from '../src/engine/advance.ts';
import { createGame } from '../src/engine/setup.ts';
import { adTrackCount } from '../src/engine/state.ts';
import type { GameConfig, GameState } from '../src/engine/state.ts';
import type { BotPolicy } from '../src/sim/bot.ts';
import { developerBot, makeDeveloperBot, playGame } from '../src/sim/bot.ts';
import { compare, parseTweaks } from '../src/sim/compare.ts';
import { makeFunnelProbe, summarize } from '../src/sim/stats.ts';
const config: GameConfig = {
mode: 'solitaire',
victory: 'highestAfterDays',
length: 'standard',
optionalRules: {
reducedVisibility: false,
sisterTrains: false,
employeeRotation: false,
emergencyToolbox: false,
},
};
const game = (seed: number): GameState =>
createGame({ id: `h${seed}`, seed, config, playerNames: ['bot'] });
/** One game, reduced to a string: every event type in order, plus the final revenue. */
function fingerprint(policy: BotPolicy, seed: number): string {
const s = game(seed);
const r = playGame(s, policy, pump);
return `${r.revenue[0]}|${r.events.map((e) => e.type).join(',')}`;
}
describe('the paired method rests on determinism', () => {
it('plays the identical game twice from one seed', () => {
/**
* THE PROPERTY EVERYTHING ELSE DEPENDS ON. A paired comparison subtracts two games that share a
* deal; if the same policy on the same seed can diverge, the difference is measuring the engine's
* own noise and every heuristic verdict is worthless.
*
* It is also the engine's central claim — pure, seeded, no ambient randomness — and the failure
* mode is silent: one `Math.random`, one Map iterated in a different order, and this is the only
* thing that would notice.
*/
for (const seed of [1000, 8919, 775569289]) {
assert.equal(fingerprint(developerBot, seed), fingerprint(developerBot, seed), `seed ${seed} diverged`);
}
});
it('gives the untweaked variant the identical game too', () => {
// `developerBot` IS `makeDeveloperBot({})`, and the refactor that introduced tweaks must not
// have changed a single decision. If this fails, every number measured before it is unusable.
for (const seed of [1000, 8919, 775569289]) {
assert.equal(
fingerprint(developerBot, seed),
fingerprint(makeDeveloperBot({}), seed),
`seed ${seed}: the untweaked variant plays a different game`,
);
}
});
});
describe('a tweak has to actually do something', () => {
it('actually holds the trains it says it holds', () => {
/**
* REGRESSION, and the reason this file exists. The first `trainCapSlack` gated the two branches
* whose comments say they exist to play a train card — and measured **exactly zero difference
* over 400 paired seeds**, because `followThrough` ends with a generic "play what is in hand"
* fallback that played the card anyway.
*
* ASSERT THE CONTRACT, NOT MERELY A DIFFERENCE. A first version of this test only checked that
* some decision changed somewhere, and it passed against the broken bot: with a tight cap the
* gated branches do change which Local Operations option is chosen, they just fail to stop the
* card being played two steps later. "Something moved" is not the promise. The promise is that
* committed trains never exceed what the Office can hold.
*
* Office tiers only ever go up, so the final A/D count is the most generous the cap ever was.
*/
const committed = (s: GameState): number =>
s.timetable.filter((t) => t !== null).length + s.pendingExtras.length;
for (const slack of [0, 1]) {
const capped = makeDeveloperBot({ trainCapSlack: slack });
let cappedWorst = 0;
let baseWorst = 0;
for (const seed of [1000, 8919, 16838, 24757, 32676, 40595, 48514]) {
const a = game(seed);
playGame(a, capped, pump);
const roomA = adTrackCount(a, 0) + slack;
cappedWorst = Math.max(cappedWorst, committed(a) - roomA);
const b = game(seed);
playGame(b, developerBot, pump);
baseWorst = Math.max(baseWorst, committed(b) - (adTrackCount(b, 0) + slack));
}
assert.ok(
cappedWorst <= 0,
`trainCapSlack=${slack} let the bot commit ${cappedWorst} train(s) more than the Office can hold`,
);
// And the cap is not vacuous: the uncapped bot really does overshoot on these seeds.
assert.ok(
baseWorst > 0,
`slack=${slack}: the baseline never overshoots on these seeds, so the test proves nothing`,
);
}
});
it('names itself so a report cannot confuse two runs', () => {
assert.equal(developerBot.name, 'developer');
assert.equal(makeDeveloperBot({}).name, 'developer');
assert.equal(makeDeveloperBot({ trainCapSlack: 1 }).name, 'developer+trainCapSlack=1');
});
it('refuses a tweak name it does not know', () => {
// A typo silently parsed as "no tweaks" would compare the bot against itself and report a
// confident zero — the most expensive possible failure of this tool.
assert.deepEqual(parseTweaks(['400', 'trainCapSlack=2']), { trainCapSlack: 2 });
assert.throws(() => parseTweaks(['trainCapSlok=1']), /unknown tweak/);
});
});
describe('the funnel counts what it claims to count', () => {
it('keeps every gate inside the total it is a fraction of', () => {
const s = game(430);
const probe = makeFunnelProbe(0);
const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn);
const f = probe.funnel;
assert.ok(f.arrivals > 0, 'no train reached an Office in a whole game');
for (const [name, n] of [
['at a passenger office', f.arrivalsAtPassengerOffice],
['with an empty coach', f.arrivalsWithEmptyCoach],
['with a loaded coach', f.arrivalsWithLoadedCoach],
['with a wanted car', f.arrivalsWithWantedCar],
] as const) {
assert.ok(n <= f.arrivals, `${name} (${n}) exceeds arrivals (${f.arrivals})`);
}
assert.ok(f.cargoPhases > 0, 'no Cargo phase was sampled');
for (const [name, n] of [
['with a facility', f.cargoWithFacility],
['green stocked', f.cargoGreenStocked],
['car spotted', f.cargoCarSpotted],
['laborer free', f.cargoLaborerFree],
['ready', f.cargoReady],
] as const) {
assert.ok(n <= f.cargoPhases, `${name} (${n}) exceeds Cargo phases (${f.cargoPhases})`);
}
// "All three at one industry" cannot exceed any of the three it is made of.
assert.ok(f.cargoReady <= Math.min(f.cargoGreenStocked, f.cargoCarSpotted, f.cargoLaborerFree));
assert.ok(f.buriedWithDigAvailable <= f.buriedTurns);
// Sampled once per Cargo phase, never once per decision: 5 Days x 12 Stages is the ceiling.
assert.ok(f.cargoPhases <= 60, `${f.cargoPhases} Cargo phases in a 5-Day game`);
// And the arrivals it counted are the arrivals that happened.
const arrived = r.events.filter((e) => e.type === 'trainArrived').length;
assert.equal(f.arrivals, arrived, 'the probe and the event log disagree about arrivals');
});
it('rides along on the harness summary', () => {
const s = game(202);
const probe = makeFunnelProbe(0);
const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn);
const stats = summarize(202, r.events, r.intents, s, probe.funnel);
assert.ok(stats.funnel, 'summarize dropped the funnel');
assert.equal(stats.funnel!.arrivals, probe.funnel.arrivals);
// Optional, so a caller with no probe still gets a summary.
assert.equal(summarize(202, r.events, r.intents, s).funnel, undefined);
});
});
describe('the comparison arithmetic', () => {
it('reports a dead heat as a dead heat', () => {
// Comparing the bot with itself must produce exactly zero, no games differing, and a t of 0 —
// if the machinery has any asymmetry in it, this is where it shows.
const r = compare({}, 12);
assert.equal(r.mean, 0);
assert.equal(r.identical, 12);
assert.equal(r.better, 0);
assert.equal(r.worse, 0);
assert.equal(r.t, 0);
});
it('pairs by seed, and deals the same seeds the harness does', () => {
const r = compare({ trainCapSlack: 1 }, 5);
assert.deepEqual(
r.deltas.map((d) => d.seed),
[0, 1, 2, 3, 4].map((i) => 1000 + i * 7919),
'compare and the harness must talk about the same games',
);
for (let i = 0; i < r.deltas.length; i++) {
assert.equal(r.baseStats[i]!.seed, r.deltas[i]!.seed);
assert.equal(r.variantStats[i]!.seed, r.deltas[i]!.seed);
assert.equal(r.deltas[i]!.delta, r.variantStats[i]!.revenue.net - r.baseStats[i]!.revenue.net);
}
});
it('computes the standard error from the PAIRED difference', () => {
// The whole gain of pairing is that σ of the difference is smaller than σ of either side. Using
// the level's σ by mistake would quietly restore the ±1.0 noise floor this exists to escape.
const r = compare({ trainCapSlack: 1 }, 40);
const d = r.deltas.map((x) => x.delta);
const m = d.reduce((p, c) => p + c, 0) / d.length;
const sd = Math.sqrt(d.reduce((p, c) => p + (c - m) ** 2, 0) / (d.length - 1));
assert.ok(Math.abs(r.mean - m) < 1e-9);
assert.ok(Math.abs(r.sd - sd) < 1e-9);
assert.ok(Math.abs(r.stderr - sd / Math.sqrt(d.length)) < 1e-9);
});
});