334 lines
15 KiB
TypeScript
334 lines
15 KiB
TypeScript
/**
|
||
* The measurement machinery itself.
|
||
*
|
||
* These tests exist because a measurement tool that is quietly wrong is worse than no tool: it does
|
||
* not fail, it just produces numbers that decide what the bot becomes. Each one guards a property
|
||
* the paired method depends on.
|
||
*/
|
||
|
||
import { describe, it } from 'node:test';
|
||
import assert from 'node:assert/strict';
|
||
import { existsSync, readFileSync, readdirSync } from 'node:fs';
|
||
import { dirname, join } from 'node:path';
|
||
import { fileURLToPath } from 'node:url';
|
||
|
||
import { pump } from '../src/engine/advance.ts';
|
||
import { createGame } from '../src/engine/setup.ts';
|
||
import { adTrackCount } from '../src/engine/state.ts';
|
||
import type { GameConfig, GameState } from '../src/engine/state.ts';
|
||
import type { BotPolicy } from '../src/sim/bot.ts';
|
||
import { developerBot, makeDeveloperBot, playGame } from '../src/sim/bot.ts';
|
||
import { compare, parseTweaks } from '../src/sim/compare.ts';
|
||
import { applyIntent } from '../src/engine/apply.ts';
|
||
import { legalActions } from '../src/engine/legal.ts';
|
||
import { ROLLING_STOCK_SUPPLY } from '../src/engine/content.ts';
|
||
import { fromSave } from '../src/web/game.ts';
|
||
import { makeFunnelProbe, summarize } from '../src/sim/stats.ts';
|
||
|
||
const root = join(dirname(fileURLToPath(import.meta.url)), '..');
|
||
|
||
const config: GameConfig = {
|
||
mode: 'solitaire',
|
||
victory: 'highestAfterDays',
|
||
length: 'standard',
|
||
optionalRules: {
|
||
reducedVisibility: false,
|
||
sisterTrains: false,
|
||
employeeRotation: false,
|
||
emergencyToolbox: false,
|
||
},
|
||
};
|
||
|
||
const game = (seed: number): GameState =>
|
||
createGame({ id: `h${seed}`, seed, config, playerNames: ['bot'] });
|
||
|
||
/** One game, reduced to a string: every event type in order, plus the final revenue. */
|
||
function fingerprint(policy: BotPolicy, seed: number): string {
|
||
const s = game(seed);
|
||
const r = playGame(s, policy, pump);
|
||
return `${r.revenue[0]}|${r.events.map((e) => e.type).join(',')}`;
|
||
}
|
||
|
||
describe('the paired method rests on determinism', () => {
|
||
it('plays the identical game twice from one seed', () => {
|
||
/**
|
||
* THE PROPERTY EVERYTHING ELSE DEPENDS ON. A paired comparison subtracts two games that share a
|
||
* deal; if the same policy on the same seed can diverge, the difference is measuring the engine's
|
||
* own noise and every heuristic verdict is worthless.
|
||
*
|
||
* It is also the engine's central claim — pure, seeded, no ambient randomness — and the failure
|
||
* mode is silent: one `Math.random`, one Map iterated in a different order, and this is the only
|
||
* thing that would notice.
|
||
*/
|
||
for (const seed of [1000, 8919, 775569289]) {
|
||
assert.equal(fingerprint(developerBot, seed), fingerprint(developerBot, seed), `seed ${seed} diverged`);
|
||
}
|
||
});
|
||
|
||
it('gives the untweaked variant the identical game too', () => {
|
||
// `developerBot` IS `makeDeveloperBot({})`, and the refactor that introduced tweaks must not
|
||
// have changed a single decision. If this fails, every number measured before it is unusable.
|
||
for (const seed of [1000, 8919, 775569289]) {
|
||
assert.equal(
|
||
fingerprint(developerBot, seed),
|
||
fingerprint(makeDeveloperBot({}), seed),
|
||
`seed ${seed}: the untweaked variant plays a different game`,
|
||
);
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('a tweak has to actually do something', () => {
|
||
it('actually holds the trains the Office cannot hold', () => {
|
||
/**
|
||
* REGRESSION, and the reason this file exists. The first `trainCapSlack` gated the two branches
|
||
* whose comments say they exist to play a train card — and measured **exactly zero difference
|
||
* over 400 paired seeds**, because `followThrough` ends with a generic "play what is in hand"
|
||
* fallback that played the card anyway.
|
||
*
|
||
* ASSERT THE CONTRACT, NOT MERELY A DIFFERENCE. A first version of this test only checked that
|
||
* some decision changed somewhere, and it passed against the broken bot: with a tight cap the
|
||
* gated branches do change which Local Operations option is chosen, they just fail to stop the
|
||
* card being played two steps later. "Something moved" is not the promise. The promise is that
|
||
* committed trains never exceed what the Office can hold.
|
||
*
|
||
* Office tiers only ever go up, so the final A/D count is the most generous the cap ever was.
|
||
*/
|
||
const committed = (s: GameState): number =>
|
||
s.timetable.filter((t) => t !== null).length + s.pendingExtras.length;
|
||
|
||
const uncapped = makeDeveloperBot({ noTrainCap: true });
|
||
let cappedWorst = 0;
|
||
let ablatedWorst = 0;
|
||
for (const seed of [1000, 8919, 16838, 24757, 32676, 40595, 48514]) {
|
||
const a = game(seed);
|
||
playGame(a, developerBot, pump);
|
||
cappedWorst = Math.max(cappedWorst, committed(a) - adTrackCount(a, 0));
|
||
|
||
const b = game(seed);
|
||
playGame(b, uncapped, pump);
|
||
ablatedWorst = Math.max(ablatedWorst, committed(b) - adTrackCount(b, 0));
|
||
}
|
||
assert.ok(
|
||
cappedWorst <= 0,
|
||
`the bot committed ${cappedWorst} train(s) more than the Office can hold`,
|
||
);
|
||
// And the cap is not vacuous: with it ablated, the bot really does overshoot on these seeds.
|
||
assert.ok(ablatedWorst > 0, 'the ablated bot never overshoots either, so this test proves nothing');
|
||
});
|
||
|
||
it('names itself so a report cannot confuse two runs', () => {
|
||
assert.equal(developerBot.name, 'developer');
|
||
assert.equal(makeDeveloperBot({}).name, 'developer');
|
||
assert.equal(makeDeveloperBot({ noTrainCap: true }).name, 'developer+noTrainCap=true');
|
||
});
|
||
|
||
it('refuses a tweak name it does not know', () => {
|
||
// A typo silently parsed as "no tweaks" would compare the bot against itself and report a
|
||
// confident zero — the most expensive possible failure of this tool.
|
||
assert.deepEqual(parseTweaks(['400', 'noTrainCap=1']), { noTrainCap: true });
|
||
assert.throws(() => parseTweaks(['noTrainCapp=1']), /unknown tweak/);
|
||
});
|
||
});
|
||
|
||
describe('the funnel counts what it claims to count', () => {
|
||
it('keeps every gate inside the total it is a fraction of', () => {
|
||
const s = game(430);
|
||
const probe = makeFunnelProbe(0);
|
||
const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn);
|
||
const f = probe.funnel;
|
||
|
||
assert.ok(f.arrivals > 0, 'no train reached an Office in a whole game');
|
||
for (const [name, n] of [
|
||
['at a passenger office', f.arrivalsAtPassengerOffice],
|
||
['with an empty coach', f.arrivalsWithEmptyCoach],
|
||
['with a loaded coach', f.arrivalsWithLoadedCoach],
|
||
['with a wanted car', f.arrivalsWithWantedCar],
|
||
] as const) {
|
||
assert.ok(n <= f.arrivals, `${name} (${n}) exceeds arrivals (${f.arrivals})`);
|
||
}
|
||
|
||
assert.ok(f.cargoPhases > 0, 'no Cargo phase was sampled');
|
||
for (const [name, n] of [
|
||
['with a facility', f.cargoWithFacility],
|
||
['green stocked', f.cargoGreenStocked],
|
||
['car spotted', f.cargoCarSpotted],
|
||
['laborer free', f.cargoLaborerFree],
|
||
['ready', f.cargoReady],
|
||
] as const) {
|
||
assert.ok(n <= f.cargoPhases, `${name} (${n}) exceeds Cargo phases (${f.cargoPhases})`);
|
||
}
|
||
// "All three at one industry" cannot exceed any of the three it is made of.
|
||
assert.ok(f.cargoReady <= Math.min(f.cargoGreenStocked, f.cargoCarSpotted, f.cargoLaborerFree));
|
||
assert.ok(f.buriedWithDigAvailable <= f.buriedTurns);
|
||
|
||
// Sampled once per Cargo phase, never once per decision: 5 Days x 12 Stages is the ceiling.
|
||
assert.ok(f.cargoPhases <= 60, `${f.cargoPhases} Cargo phases in a 5-Day game`);
|
||
|
||
// And the arrivals it counted are the arrivals that happened.
|
||
const arrived = r.events.filter((e) => e.type === 'trainArrived').length;
|
||
assert.equal(f.arrivals, arrived, 'the probe and the event log disagree about arrivals');
|
||
});
|
||
|
||
it('rides along on the harness summary', () => {
|
||
const s = game(202);
|
||
const probe = makeFunnelProbe(0);
|
||
const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn);
|
||
const stats = summarize(202, r.events, r.intents, s, probe.funnel);
|
||
assert.ok(stats.funnel, 'summarize dropped the funnel');
|
||
assert.equal(stats.funnel!.arrivals, probe.funnel.arrivals);
|
||
// Optional, so a caller with no probe still gets a summary.
|
||
assert.equal(summarize(202, r.events, r.intents, s).funnel, undefined);
|
||
});
|
||
});
|
||
|
||
describe('the game conserves Rolling Stock', () => {
|
||
it('creates no car that was not dealt, and destroys none but in a collision', () => {
|
||
/**
|
||
* REGRESSION, found by audit rather than by a failing test — which is why this one exists.
|
||
*
|
||
* `TODO.md` had it as an open question: a census of every holder came to 92 against the 80 cars
|
||
* dealt, "not necessarily duplication, because some of those objects are cargo in transit". It
|
||
* was duplication, and it came from two rules the engine had not implemented:
|
||
*
|
||
* §9.3 unload — "a load on the industry's track AND AN EMPTY CAR OF THAT TYPE IN THE
|
||
* DIVISION YARD ... replaces the load with an empty car of that type"
|
||
* §9.2 de-train — "a white empty coach IN THE DIVISION YARD ... replace the blue coach on the
|
||
* train with the white one"
|
||
*
|
||
* Both requirements were unchecked and both replacement cars were conjured rather than taken, so
|
||
* every unload and every de-training minted a car — 1.29 a game against a supply of 80, which is
|
||
* the number `ROLLING_STOCK_SUPPLY` exists to control.
|
||
*
|
||
* Censused after every batch, because that is the only way this class of bug shows up at all.
|
||
*/
|
||
const supply = ROLLING_STOCK_SUPPLY.reduce((n, r) => n + r.loaded + r.empty, 0);
|
||
const census = (st: GameState): number => {
|
||
let n = st.yards.divisionYard.length + st.yards.classificationYard.length;
|
||
for (const t of st.trays.values()) n += t.consist.length;
|
||
for (const [, area] of st.officeAreas) {
|
||
for (const card of area.grid.values()) {
|
||
n += card.standing.length;
|
||
const f = card.facility;
|
||
if (!f) continue;
|
||
n += f.industryTrack.cars.length + f.outboundBox.length + f.inboundBox.length;
|
||
n += (f.menAtWork ?? []).filter((m) => m !== null).length;
|
||
}
|
||
}
|
||
return n;
|
||
};
|
||
|
||
for (const seed of [1000, 8919, 16838, 24757, 32676, 40595]) {
|
||
const s = game(seed);
|
||
assert.equal(census(s), supply, `seed ${seed}: the deal itself is short`);
|
||
let expected = supply;
|
||
|
||
for (let t = 0; t < 50_000; t++) {
|
||
const before = census(s);
|
||
const pumped = pump(s);
|
||
// §10 — a collision destroys both trains and everything they were carrying. That is the one
|
||
// legitimate way the count falls, so the expectation follows it down.
|
||
for (const e of pumped) {
|
||
if (e.type === 'trainsDestroyed') for (const tr of e.trains) expected -= tr.consist.length;
|
||
}
|
||
assert.ok(
|
||
census(s) === before || pumped.some((e) => e.type === 'trainsDestroyed'),
|
||
`seed ${seed}: the engine changed the census by ${census(s) - before} outside a collision`,
|
||
);
|
||
if (s.status === 'finished') break;
|
||
const actor = s.clock.pendingDecision !== null ? s.clock.superintendent : s.clock.currentActor;
|
||
if (actor === null) break;
|
||
const options = legalActions(s, actor);
|
||
if (options.length === 0) break;
|
||
const was = census(s);
|
||
const r = applyIntent(s, actor, developerBot.choose(s, actor, options));
|
||
if (!r.ok) break;
|
||
assert.equal(
|
||
census(s),
|
||
was,
|
||
`seed ${seed}: a player action changed the Rolling Stock census by ${census(s) - was}`,
|
||
);
|
||
}
|
||
assert.equal(census(s), expected, `seed ${seed}: ended holding ${census(s)} of an expected ${expected}`);
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('every published replay actually replays', () => {
|
||
it('reaches the end of its own history', () => {
|
||
/**
|
||
* REGRESSION, and it had already bitten. `TODO.md` records both published replays going dead
|
||
* without anyone noticing: `fromSave` stops at the first intent the rules no longer accept and
|
||
* returns a SHORTER game, which on screen looks exactly like a game that ended early. Measured
|
||
* when this test was written, all three published replays managed **2 intents of roughly 400**
|
||
* — the site was serving three recordings of nothing.
|
||
*
|
||
* A replay is a save from a particular ruleset, so this is really a test that the rules have not
|
||
* moved under the files in `public/replays`. When it fails, re-record rather than edit:
|
||
* `node src/sim/save-replay.ts 400 --top 3`.
|
||
*/
|
||
const dir = join(root, 'public/replays');
|
||
const files = existsSync(dir) ? readdirSync(dir).filter((f) => f.endsWith('.json') && f !== 'manifest.json') : [];
|
||
assert.ok(files.length > 0, 'no replays are published at all');
|
||
|
||
for (const f of files) {
|
||
const save = JSON.parse(readFileSync(join(dir, f), 'utf8')) as {
|
||
seed: number;
|
||
history: unknown[];
|
||
rules?: unknown;
|
||
};
|
||
// The WHOLE save, `rules` included. Rebuilding it from seed and history alone threw away the
|
||
// one field that says which ruleset the file was recorded under, so this replayed every
|
||
// published file under the pre-dialog defaults no matter what it said — a test that would
|
||
// pass a genuinely dead replay the moment the defaults and the file disagreed.
|
||
const back = fromSave(save as never);
|
||
assert.equal(
|
||
back.history.length,
|
||
save.history.length,
|
||
`${f} is dead — it replays ${back.history.length} of ${save.history.length} intents. ` +
|
||
'Re-record it with save-replay.ts rather than editing the file.',
|
||
);
|
||
assert.equal(back.state.status, 'finished', `${f} does not reach the end of its game`);
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('the comparison arithmetic', () => {
|
||
it('reports a dead heat as a dead heat', () => {
|
||
// Comparing the bot with itself must produce exactly zero, no games differing, and a t of 0 —
|
||
// if the machinery has any asymmetry in it, this is where it shows.
|
||
const r = compare({}, 12);
|
||
assert.equal(r.mean, 0);
|
||
assert.equal(r.identical, 12);
|
||
assert.equal(r.better, 0);
|
||
assert.equal(r.worse, 0);
|
||
assert.equal(r.t, 0);
|
||
});
|
||
|
||
it('pairs by seed, and deals the same seeds the harness does', () => {
|
||
const r = compare({ noTrainCap: true }, 5);
|
||
assert.deepEqual(
|
||
r.deltas.map((d) => d.seed),
|
||
[0, 1, 2, 3, 4].map((i) => 1000 + i * 7919),
|
||
'compare and the harness must talk about the same games',
|
||
);
|
||
for (let i = 0; i < r.deltas.length; i++) {
|
||
assert.equal(r.baseStats[i]!.seed, r.deltas[i]!.seed);
|
||
assert.equal(r.variantStats[i]!.seed, r.deltas[i]!.seed);
|
||
assert.equal(r.deltas[i]!.delta, r.variantStats[i]!.revenue.net - r.baseStats[i]!.revenue.net);
|
||
}
|
||
});
|
||
|
||
it('computes the standard error from the PAIRED difference', () => {
|
||
// The whole gain of pairing is that σ of the difference is smaller than σ of either side. Using
|
||
// the level's σ by mistake would quietly restore the ±1.0 noise floor this exists to escape.
|
||
const r = compare({ noTrainCap: true }, 40);
|
||
const d = r.deltas.map((x) => x.delta);
|
||
const m = d.reduce((p, c) => p + c, 0) / d.length;
|
||
const sd = Math.sqrt(d.reduce((p, c) => p + (c - m) ** 2, 0) / (d.length - 1));
|
||
assert.ok(Math.abs(r.mean - m) < 1e-9);
|
||
assert.ok(Math.abs(r.sd - sd) < 1e-9);
|
||
assert.ok(Math.abs(r.stderr - sd / Math.sqrt(d.length)) < 1e-9);
|
||
});
|
||
});
|