Gitea#17 — a 45° leg is an end of the west-to-east row, so backing into a cut through a curve's south leg no longer couples it back to front. The same assumption left a crew's own cut standing when it pulled out through a leg, which is the "cars left behind" report we had failed to reproduce. Gitea#14 — every count is docs/Deck cards5.xlsx. Track halved, and the Q12 office doubling and Gap 12 industry tripling both come out with it: they were measured against a deck with twice the track, and keeping them at the sheet's track count wipes out the reefer chain entirely. 84 rows now match card for card; the ten Safety, Event and Inspection cards it adds are not built and are held out. Cards the sheet no longer lists are dealt zero copies rather than deleted, so their rules stay implemented. Gitea#15 — RAR reversed it: a rail may stop dead against its neighbour and the placement is legal. What must hold is that no train crosses the gap, which was already true and is now pinned against the reported board. Gitea#3 — the printed speeds are scenery. A card costs one Stage per printed region and where a train STARTS is what varies; Fast/Slow is read on Hilly alone. Entering a one-region card behind another is a collision now, which is what ABS exists to prevent, and ABS no longer holds trains silently. Gitea#18 — the Division draws as one row, west to east, with no office-area detail. East is finally always to the right. Closes #3 Closes #14 Closes #15 Closes #17 Closes #18
350 lines
16 KiB
TypeScript
350 lines
16 KiB
TypeScript
/**
|
||
* The measurement machinery itself.
|
||
*
|
||
* These tests exist because a measurement tool that is quietly wrong is worse than no tool: it does
|
||
* not fail, it just produces numbers that decide what the bot becomes. Each one guards a property
|
||
* the paired method depends on.
|
||
*/
|
||
|
||
import { describe, it } from 'node:test';
|
||
import assert from 'node:assert/strict';
|
||
import { existsSync, readFileSync, readdirSync } from 'node:fs';
|
||
import { dirname, join } from 'node:path';
|
||
import { fileURLToPath } from 'node:url';
|
||
|
||
import { pump } from '../src/engine/advance.ts';
|
||
import { createGame } from '../src/engine/setup.ts';
|
||
import { adTrackCount } from '../src/engine/state.ts';
|
||
import type { GameConfig, GameState } from '../src/engine/state.ts';
|
||
import type { BotPolicy } from '../src/sim/bot.ts';
|
||
import { developerBot, makeDeveloperBot, playGame } from '../src/sim/bot.ts';
|
||
import { compare, parseTweaks } from '../src/sim/compare.ts';
|
||
import { applyIntent } from '../src/engine/apply.ts';
|
||
import { legalActions } from '../src/engine/legal.ts';
|
||
import { ROLLING_STOCK_SUPPLY } from '../src/engine/content.ts';
|
||
import { fromSave } from '../src/web/game.ts';
|
||
import { makeFunnelProbe, summarize } from '../src/sim/stats.ts';
|
||
|
||
const root = join(dirname(fileURLToPath(import.meta.url)), '..');
|
||
|
||
const config: GameConfig = {
|
||
mode: 'solitaire',
|
||
days: 5,
|
||
minCombinedRevenue: 0,
|
||
maxCollisionsPerDay: 0,
|
||
maxCollisionsTotal: 0,
|
||
pvpCardsAllowed: false,
|
||
optionalRules: {
|
||
reducedVisibility: false,
|
||
employeeRotation: false,
|
||
emergencyToolbox: false,
|
||
},
|
||
};
|
||
|
||
const game = (seed: number): GameState =>
|
||
createGame({ id: `h${seed}`, seed, config, playerNames: ['bot'] });
|
||
|
||
/** One game, reduced to a string: every event type in order, plus the final revenue. */
|
||
function fingerprint(policy: BotPolicy, seed: number): string {
|
||
const s = game(seed);
|
||
const r = playGame(s, policy, pump);
|
||
return `${r.revenue[0]}|${r.events.map((e) => e.type).join(',')}`;
|
||
}
|
||
|
||
describe('the paired method rests on determinism', () => {
|
||
it('plays the identical game twice from one seed', () => {
|
||
/**
|
||
* THE PROPERTY EVERYTHING ELSE DEPENDS ON. A paired comparison subtracts two games that share a
|
||
* deal; if the same policy on the same seed can diverge, the difference is measuring the engine's
|
||
* own noise and every heuristic verdict is worthless.
|
||
*
|
||
* It is also the engine's central claim — pure, seeded, no ambient randomness — and the failure
|
||
* mode is silent: one `Math.random`, one Map iterated in a different order, and this is the only
|
||
* thing that would notice.
|
||
*/
|
||
for (const seed of [1000, 8919, 775569289]) {
|
||
assert.equal(fingerprint(developerBot, seed), fingerprint(developerBot, seed), `seed ${seed} diverged`);
|
||
}
|
||
});
|
||
|
||
it('gives the untweaked variant the identical game too', () => {
|
||
// `developerBot` IS `makeDeveloperBot({})`, and the refactor that introduced tweaks must not
|
||
// have changed a single decision. If this fails, every number measured before it is unusable.
|
||
for (const seed of [1000, 8919, 775569289]) {
|
||
assert.equal(
|
||
fingerprint(developerBot, seed),
|
||
fingerprint(makeDeveloperBot({}), seed),
|
||
`seed ${seed}: the untweaked variant plays a different game`,
|
||
);
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('a tweak has to actually do something', () => {
|
||
it('actually holds the trains the Office cannot hold', () => {
|
||
/**
|
||
* REGRESSION, and the reason this file exists. The first `trainCapSlack` gated the two branches
|
||
* whose comments say they exist to play a train card — and measured **exactly zero difference
|
||
* over 400 paired seeds**, because `followThrough` ends with a generic "play what is in hand"
|
||
* fallback that played the card anyway.
|
||
*
|
||
* ASSERT THE CONTRACT, NOT MERELY A DIFFERENCE. A first version of this test only checked that
|
||
* some decision changed somewhere, and it passed against the broken bot: with a tight cap the
|
||
* gated branches do change which Local Operations option is chosen, they just fail to stop the
|
||
* card being played two steps later. "Something moved" is not the promise. The promise is that
|
||
* committed trains never exceed what the Office can hold.
|
||
*
|
||
* Office tiers only ever go up, so the final A/D count is the most generous the cap ever was.
|
||
*/
|
||
const committed = (s: GameState): number =>
|
||
s.timetable.filter((t) => t !== null).length + s.pendingExtras.length;
|
||
|
||
const uncapped = makeDeveloperBot({ noTrainCap: true });
|
||
let cappedWorst = 0;
|
||
let ablatedWorst = 0;
|
||
for (const seed of [1000, 8919, 16838, 24757, 32676, 40595, 48514]) {
|
||
const a = game(seed);
|
||
playGame(a, developerBot, pump);
|
||
cappedWorst = Math.max(cappedWorst, committed(a) - adTrackCount(a, 0));
|
||
|
||
const b = game(seed);
|
||
playGame(b, uncapped, pump);
|
||
ablatedWorst = Math.max(ablatedWorst, committed(b) - adTrackCount(b, 0));
|
||
}
|
||
assert.ok(
|
||
cappedWorst <= 0,
|
||
`the bot committed ${cappedWorst} train(s) more than the Office can hold`,
|
||
);
|
||
// And the cap is not vacuous: with it ablated, the bot really does overshoot on these seeds.
|
||
assert.ok(ablatedWorst > 0, 'the ablated bot never overshoots either, so this test proves nothing');
|
||
});
|
||
|
||
it('names itself so a report cannot confuse two runs', () => {
|
||
assert.equal(developerBot.name, 'developer');
|
||
assert.equal(makeDeveloperBot({}).name, 'developer');
|
||
assert.equal(makeDeveloperBot({ noTrainCap: true }).name, 'developer+noTrainCap=true');
|
||
});
|
||
|
||
it('refuses a tweak name it does not know', () => {
|
||
// A typo silently parsed as "no tweaks" would compare the bot against itself and report a
|
||
// confident zero — the most expensive possible failure of this tool.
|
||
assert.deepEqual(parseTweaks(['400', 'noTrainCap=1']), { noTrainCap: true });
|
||
assert.throws(() => parseTweaks(['noTrainCapp=1']), /unknown tweak/);
|
||
});
|
||
});
|
||
|
||
describe('the funnel counts what it claims to count', () => {
|
||
it('keeps every gate inside the total it is a fraction of', () => {
|
||
const s = game(430);
|
||
const probe = makeFunnelProbe(0);
|
||
const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn);
|
||
const f = probe.funnel;
|
||
|
||
assert.ok(f.arrivals > 0, 'no train reached an Office in a whole game');
|
||
for (const [name, n] of [
|
||
['at a passenger office', f.arrivalsAtPassengerOffice],
|
||
['with an empty coach', f.arrivalsWithEmptyCoach],
|
||
['with a loaded coach', f.arrivalsWithLoadedCoach],
|
||
['with a wanted car', f.arrivalsWithWantedCar],
|
||
] as const) {
|
||
assert.ok(n <= f.arrivals, `${name} (${n}) exceeds arrivals (${f.arrivals})`);
|
||
}
|
||
|
||
assert.ok(f.cargoPhases > 0, 'no Cargo phase was sampled');
|
||
for (const [name, n] of [
|
||
['with a facility', f.cargoWithFacility],
|
||
['green stocked', f.cargoGreenStocked],
|
||
['car spotted', f.cargoCarSpotted],
|
||
['laborer free', f.cargoLaborerFree],
|
||
['ready', f.cargoReady],
|
||
] as const) {
|
||
assert.ok(n <= f.cargoPhases, `${name} (${n}) exceeds Cargo phases (${f.cargoPhases})`);
|
||
}
|
||
// "All three at one industry" cannot exceed any of the three it is made of.
|
||
assert.ok(f.cargoReady <= Math.min(f.cargoGreenStocked, f.cargoCarSpotted, f.cargoLaborerFree));
|
||
assert.ok(f.buriedWithDigAvailable <= f.buriedTurns);
|
||
|
||
// Sampled once per Cargo phase, never once per decision: 5 Days x 12 Stages is the ceiling.
|
||
assert.ok(f.cargoPhases <= 60, `${f.cargoPhases} Cargo phases in a 5-Day game`);
|
||
|
||
// And the arrivals it counted are the arrivals that happened.
|
||
const arrived = r.events.filter((e) => e.type === 'trainArrived').length;
|
||
assert.equal(f.arrivals, arrived, 'the probe and the event log disagree about arrivals');
|
||
});
|
||
|
||
it('rides along on the harness summary', () => {
|
||
const s = game(202);
|
||
const probe = makeFunnelProbe(0);
|
||
const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn);
|
||
const stats = summarize(202, r.events, r.intents, s, probe.funnel);
|
||
assert.ok(stats.funnel, 'summarize dropped the funnel');
|
||
assert.equal(stats.funnel!.arrivals, probe.funnel.arrivals);
|
||
// Optional, so a caller with no probe still gets a summary.
|
||
assert.equal(summarize(202, r.events, r.intents, s).funnel, undefined);
|
||
});
|
||
});
|
||
|
||
describe('the game conserves Rolling Stock', () => {
|
||
it('creates no car that was not dealt, and destroys none but in a collision', () => {
|
||
/**
|
||
* REGRESSION, found by audit rather than by a failing test — which is why this one exists.
|
||
*
|
||
* `TODO.md` had it as an open question: a census of every holder came to 92 against the 80 cars
|
||
* dealt, "not necessarily duplication, because some of those objects are cargo in transit". It
|
||
* was duplication, and it came from two rules the engine had not implemented:
|
||
*
|
||
* §9.3 unload — "a load on the industry's track AND AN EMPTY CAR OF THAT TYPE IN THE
|
||
* DIVISION YARD ... replaces the load with an empty car of that type"
|
||
* §9.2 de-train — "a white empty coach IN THE DIVISION YARD ... replace the blue coach on the
|
||
* train with the white one"
|
||
*
|
||
* Both requirements were unchecked and both replacement cars were conjured rather than taken, so
|
||
* every unload and every de-training minted a car — 1.29 a game against a supply of 80, which is
|
||
* the number `ROLLING_STOCK_SUPPLY` exists to control.
|
||
*
|
||
* Censused after every batch, because that is the only way this class of bug shows up at all.
|
||
*/
|
||
const supply = ROLLING_STOCK_SUPPLY.reduce((n, r) => n + r.loaded + r.empty, 0);
|
||
const census = (st: GameState): number => {
|
||
let n = st.yards.divisionYard.length + st.yards.classificationYard.length;
|
||
for (const t of st.trays.values()) n += t.consist.length;
|
||
for (const [, area] of st.officeAreas) {
|
||
for (const card of area.grid.values()) {
|
||
n += card.standing.length;
|
||
const f = card.facility;
|
||
if (!f) continue;
|
||
n += f.industryTrack.cars.length + f.outboundBox.length + f.inboundBox.length;
|
||
n += (f.menAtWork ?? []).filter((m) => m !== null).length;
|
||
}
|
||
}
|
||
return n;
|
||
};
|
||
|
||
for (const seed of [1000, 8919, 16838, 24757, 32676, 40595]) {
|
||
const s = game(seed);
|
||
assert.equal(census(s), supply, `seed ${seed}: the deal itself is short`);
|
||
let expected = supply;
|
||
|
||
for (let t = 0; t < 50_000; t++) {
|
||
const before = census(s);
|
||
const pumped = pump(s);
|
||
/**
|
||
* A COLLISION DESTROYS NO CAR, and this used to assume it destroyed all of them.
|
||
*
|
||
* The subtraction that stood here — `expected -= tr.consist.length` for every train in a
|
||
* `trainsDestroyed` event — describes a rule the engine does not have. Gap 2c (`advance.ts`,
|
||
* "TAKE THE WRECK OFF THE CARD") sends the wreck's cabooses back to the Division Yard and
|
||
* everything else to Classification, so the stock is conserved through a collision like any
|
||
* other move. The train is destroyed; its cars are not.
|
||
*
|
||
* It passed for as long as it did because none of the six seeds below ever collided, so the
|
||
* branch never ran. Changing the deck to the sheet's counts (Gitea#14) moved the deals, seed
|
||
* 24757 collided, and the test failed claiming the engine had CONJURED three cars — the
|
||
* exact opposite of what had happened.
|
||
*
|
||
* So the census is now held flat, unconditionally, which is both the real invariant and a
|
||
* stronger test than the one it replaces: there is no longer any event that excuses a
|
||
* change, and `expected` cannot drift away from the supply it was dealt.
|
||
*/
|
||
assert.equal(
|
||
census(s),
|
||
before,
|
||
`seed ${seed}: the engine changed the census by ${census(s) - before} while pumping`,
|
||
);
|
||
if (s.status === 'finished') break;
|
||
const actor = s.clock.pendingDecision !== null ? s.clock.superintendent : s.clock.currentActor;
|
||
if (actor === null) break;
|
||
const options = legalActions(s, actor);
|
||
if (options.length === 0) break;
|
||
const was = census(s);
|
||
const r = applyIntent(s, actor, developerBot.choose(s, actor, options));
|
||
if (!r.ok) break;
|
||
assert.equal(
|
||
census(s),
|
||
was,
|
||
`seed ${seed}: a player action changed the Rolling Stock census by ${census(s) - was}`,
|
||
);
|
||
}
|
||
assert.equal(census(s), expected, `seed ${seed}: ended holding ${census(s)} of an expected ${expected}`);
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('every published replay actually replays', () => {
|
||
it('reaches the end of its own history', () => {
|
||
/**
|
||
* REGRESSION, and it had already bitten. `TODO.md` records both published replays going dead
|
||
* without anyone noticing: `fromSave` stops at the first intent the rules no longer accept and
|
||
* returns a SHORTER game, which on screen looks exactly like a game that ended early. Measured
|
||
* when this test was written, all three published replays managed **2 intents of roughly 400**
|
||
* — the site was serving three recordings of nothing.
|
||
*
|
||
* A replay is a save from a particular ruleset, so this is really a test that the rules have not
|
||
* moved under the files in `public/replays`. When it fails, re-record rather than edit:
|
||
* `node src/sim/save-replay.ts 400 --top 3`.
|
||
*/
|
||
const dir = join(root, 'public/replays');
|
||
const files = existsSync(dir) ? readdirSync(dir).filter((f) => f.endsWith('.json') && f !== 'manifest.json') : [];
|
||
assert.ok(files.length > 0, 'no replays are published at all');
|
||
|
||
for (const f of files) {
|
||
const save = JSON.parse(readFileSync(join(dir, f), 'utf8')) as {
|
||
seed: number;
|
||
history: unknown[];
|
||
rules?: unknown;
|
||
};
|
||
// The WHOLE save, `rules` included. Rebuilding it from seed and history alone threw away the
|
||
// one field that says which ruleset the file was recorded under, so this replayed every
|
||
// published file under the pre-dialog defaults no matter what it said — a test that would
|
||
// pass a genuinely dead replay the moment the defaults and the file disagreed.
|
||
const back = fromSave(save as never);
|
||
assert.equal(
|
||
back.history.length,
|
||
save.history.length,
|
||
`${f} is dead — it replays ${back.history.length} of ${save.history.length} intents. ` +
|
||
'Re-record it with save-replay.ts rather than editing the file.',
|
||
);
|
||
assert.equal(back.state.status, 'finished', `${f} does not reach the end of its game`);
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('the comparison arithmetic', () => {
|
||
it('reports a dead heat as a dead heat', () => {
|
||
// Comparing the bot with itself must produce exactly zero, no games differing, and a t of 0 —
|
||
// if the machinery has any asymmetry in it, this is where it shows.
|
||
const r = compare({}, 12);
|
||
assert.equal(r.mean, 0);
|
||
assert.equal(r.identical, 12);
|
||
assert.equal(r.better, 0);
|
||
assert.equal(r.worse, 0);
|
||
assert.equal(r.t, 0);
|
||
});
|
||
|
||
it('pairs by seed, and deals the same seeds the harness does', () => {
|
||
const r = compare({ noTrainCap: true }, 5);
|
||
assert.deepEqual(
|
||
r.deltas.map((d) => d.seed),
|
||
[0, 1, 2, 3, 4].map((i) => 1000 + i * 7919),
|
||
'compare and the harness must talk about the same games',
|
||
);
|
||
for (let i = 0; i < r.deltas.length; i++) {
|
||
assert.equal(r.baseStats[i]!.seed, r.deltas[i]!.seed);
|
||
assert.equal(r.variantStats[i]!.seed, r.deltas[i]!.seed);
|
||
assert.equal(r.deltas[i]!.delta, r.variantStats[i]!.revenue.net - r.baseStats[i]!.revenue.net);
|
||
}
|
||
});
|
||
|
||
it('computes the standard error from the PAIRED difference', () => {
|
||
// The whole gain of pairing is that σ of the difference is smaller than σ of either side. Using
|
||
// the level's σ by mistake would quietly restore the ±1.0 noise floor this exists to escape.
|
||
const r = compare({ noTrainCap: true }, 40);
|
||
const d = r.deltas.map((x) => x.delta);
|
||
const m = d.reduce((p, c) => p + c, 0) / d.length;
|
||
const sd = Math.sqrt(d.reduce((p, c) => p + (c - m) ** 2, 0) / (d.length - 1));
|
||
assert.ok(Math.abs(r.mean - m) < 1e-9);
|
||
assert.ok(Math.abs(r.sd - sd) < 1e-9);
|
||
assert.ok(Math.abs(r.stderr - sd / Math.sqrt(d.length)) < 1e-9);
|
||
});
|
||
});
|