created harness to better test bots. improved bot play

This commit is contained in:
Jesse
2026-08-09 19:24:54 -04:00
parent 35575147cd
commit 4a800a053a
13 changed files with 6263 additions and 5503 deletions
+137 -28
View File
@@ -8,6 +8,9 @@
import { describe, it } from 'node:test';
import assert from 'node:assert/strict';
import { existsSync, readFileSync, readdirSync } from 'node:fs';
import { dirname, join } from 'node:path';
import { fileURLToPath } from 'node:url';
import { pump } from '../src/engine/advance.ts';
import { createGame } from '../src/engine/setup.ts';
@@ -16,8 +19,14 @@ import type { GameConfig, GameState } from '../src/engine/state.ts';
import type { BotPolicy } from '../src/sim/bot.ts';
import { developerBot, makeDeveloperBot, playGame } from '../src/sim/bot.ts';
import { compare, parseTweaks } from '../src/sim/compare.ts';
import { applyIntent } from '../src/engine/apply.ts';
import { legalActions } from '../src/engine/legal.ts';
import { ROLLING_STOCK_SUPPLY } from '../src/engine/content.ts';
import { fromSave } from '../src/web/game.ts';
import { makeFunnelProbe, summarize } from '../src/sim/stats.ts';
const root = join(dirname(fileURLToPath(import.meta.url)), '..');
const config: GameConfig = {
mode: 'solitaire',
victory: 'highestAfterDays',
@@ -70,7 +79,7 @@ describe('the paired method rests on determinism', () => {
});
describe('a tweak has to actually do something', () => {
it('actually holds the trains it says it holds', () => {
it('actually holds the trains the Office cannot hold', () => {
/**
* REGRESSION, and the reason this file exists. The first `trainCapSlack` gated the two branches
* whose comments say they exist to play a train card — and measured **exactly zero difference
@@ -88,43 +97,37 @@ describe('a tweak has to actually do something', () => {
const committed = (s: GameState): number =>
s.timetable.filter((t) => t !== null).length + s.pendingExtras.length;
for (const slack of [0, 1]) {
const capped = makeDeveloperBot({ trainCapSlack: slack });
let cappedWorst = 0;
let baseWorst = 0;
for (const seed of [1000, 8919, 16838, 24757, 32676, 40595, 48514]) {
const a = game(seed);
playGame(a, capped, pump);
const roomA = adTrackCount(a, 0) + slack;
cappedWorst = Math.max(cappedWorst, committed(a) - roomA);
const uncapped = makeDeveloperBot({ noTrainCap: true });
let cappedWorst = 0;
let ablatedWorst = 0;
for (const seed of [1000, 8919, 16838, 24757, 32676, 40595, 48514]) {
const a = game(seed);
playGame(a, developerBot, pump);
cappedWorst = Math.max(cappedWorst, committed(a) - adTrackCount(a, 0));
const b = game(seed);
playGame(b, developerBot, pump);
baseWorst = Math.max(baseWorst, committed(b) - (adTrackCount(b, 0) + slack));
}
assert.ok(
cappedWorst <= 0,
`trainCapSlack=${slack} let the bot commit ${cappedWorst} train(s) more than the Office can hold`,
);
// And the cap is not vacuous: the uncapped bot really does overshoot on these seeds.
assert.ok(
baseWorst > 0,
`slack=${slack}: the baseline never overshoots on these seeds, so the test proves nothing`,
);
const b = game(seed);
playGame(b, uncapped, pump);
ablatedWorst = Math.max(ablatedWorst, committed(b) - adTrackCount(b, 0));
}
assert.ok(
cappedWorst <= 0,
`the bot committed ${cappedWorst} train(s) more than the Office can hold`,
);
// And the cap is not vacuous: with it ablated, the bot really does overshoot on these seeds.
assert.ok(ablatedWorst > 0, 'the ablated bot never overshoots either, so this test proves nothing');
});
it('names itself so a report cannot confuse two runs', () => {
assert.equal(developerBot.name, 'developer');
assert.equal(makeDeveloperBot({}).name, 'developer');
assert.equal(makeDeveloperBot({ trainCapSlack: 1 }).name, 'developer+trainCapSlack=1');
assert.equal(makeDeveloperBot({ noTrainCap: true }).name, 'developer+noTrainCap=true');
});
it('refuses a tweak name it does not know', () => {
// A typo silently parsed as "no tweaks" would compare the bot against itself and report a
// confident zero — the most expensive possible failure of this tool.
assert.deepEqual(parseTweaks(['400', 'trainCapSlack=2']), { trainCapSlack: 2 });
assert.throws(() => parseTweaks(['trainCapSlok=1']), /unknown tweak/);
assert.deepEqual(parseTweaks(['400', 'noTrainCap=1']), { noTrainCap: true });
assert.throws(() => parseTweaks(['noTrainCapp=1']), /unknown tweak/);
});
});
@@ -179,6 +182,112 @@ describe('the funnel counts what it claims to count', () => {
});
});
describe('the game conserves Rolling Stock', () => {
it('creates no car that was not dealt, and destroys none but in a collision', () => {
/**
* REGRESSION, found by audit rather than by a failing test — which is why this one exists.
*
* `TODO.md` had it as an open question: a census of every holder came to 92 against the 80 cars
* dealt, "not necessarily duplication, because some of those objects are cargo in transit". It
* was duplication, and it came from two rules the engine had not implemented:
*
* §9.3 unload — "a load on the industry's track AND AN EMPTY CAR OF THAT TYPE IN THE
* DIVISION YARD ... replaces the load with an empty car of that type"
* §9.2 de-train — "a white empty coach IN THE DIVISION YARD ... replace the blue coach on the
* train with the white one"
*
* Both requirements were unchecked and both replacement cars were conjured rather than taken, so
* every unload and every de-training minted a car — 1.29 a game against a supply of 80, which is
* the number `ROLLING_STOCK_SUPPLY` exists to control.
*
* Censused after every batch, because that is the only way this class of bug shows up at all.
*/
const supply = ROLLING_STOCK_SUPPLY.reduce((n, r) => n + r.loaded + r.empty, 0);
const census = (st: GameState): number => {
let n = st.yards.divisionYard.length + st.yards.classificationYard.length;
for (const t of st.trays.values()) n += t.consist.length;
for (const [, area] of st.officeAreas) {
for (const card of area.grid.values()) {
n += card.standing.length;
const f = card.facility;
if (!f) continue;
n += f.industryTrack.cars.length + f.outboundBox.length + f.inboundBox.length;
n += f.menAtWork.filter((m) => m !== null).length;
}
}
return n;
};
for (const seed of [1000, 8919, 16838, 24757, 32676, 40595]) {
const s = game(seed);
assert.equal(census(s), supply, `seed ${seed}: the deal itself is short`);
let expected = supply;
for (let t = 0; t < 50_000; t++) {
const before = census(s);
const pumped = pump(s);
// §10 — a collision destroys both trains and everything they were carrying. That is the one
// legitimate way the count falls, so the expectation follows it down.
for (const e of pumped) {
if (e.type === 'trainsDestroyed') for (const tr of e.trains) expected -= tr.consist.length;
}
assert.ok(
census(s) === before || pumped.some((e) => e.type === 'trainsDestroyed'),
`seed ${seed}: the engine changed the census by ${census(s) - before} outside a collision`,
);
if (s.status === 'finished') break;
const actor = s.clock.pendingDecision !== null ? s.clock.superintendent : s.clock.currentActor;
if (actor === null) break;
const options = legalActions(s, actor);
if (options.length === 0) break;
const was = census(s);
const r = applyIntent(s, actor, developerBot.choose(s, actor, options));
if (!r.ok) break;
assert.equal(
census(s),
was,
`seed ${seed}: a player action changed the Rolling Stock census by ${census(s) - was}`,
);
}
assert.equal(census(s), expected, `seed ${seed}: ended holding ${census(s)} of an expected ${expected}`);
}
});
});
describe('every published replay actually replays', () => {
it('reaches the end of its own history', () => {
/**
* REGRESSION, and it had already bitten. `TODO.md` records both published replays going dead
* without anyone noticing: `fromSave` stops at the first intent the rules no longer accept and
* returns a SHORTER game, which on screen looks exactly like a game that ended early. Measured
* when this test was written, all three published replays managed **2 intents of roughly 400**
* — the site was serving three recordings of nothing.
*
* A replay is a save from a particular ruleset, so this is really a test that the rules have not
* moved under the files in `public/replays`. When it fails, re-record rather than edit:
* `node src/sim/save-replay.ts 400 --top 3`.
*/
const dir = join(root, 'public/replays');
const files = existsSync(dir) ? readdirSync(dir).filter((f) => f.endsWith('.json') && f !== 'manifest.json') : [];
assert.ok(files.length > 0, 'no replays are published at all');
for (const f of files) {
const save = JSON.parse(readFileSync(join(dir, f), 'utf8')) as {
seed: number;
history: unknown[];
};
const back = fromSave({ seed: save.seed, history: save.history as never });
assert.equal(
back.history.length,
save.history.length,
`${f} is dead — it replays ${back.history.length} of ${save.history.length} intents. ` +
'Re-record it with save-replay.ts rather than editing the file.',
);
assert.equal(back.state.status, 'finished', `${f} does not reach the end of its game`);
}
});
});
describe('the comparison arithmetic', () => {
it('reports a dead heat as a dead heat', () => {
// Comparing the bot with itself must produce exactly zero, no games differing, and a t of 0 —
@@ -192,7 +301,7 @@ describe('the comparison arithmetic', () => {
});
it('pairs by seed, and deals the same seeds the harness does', () => {
const r = compare({ trainCapSlack: 1 }, 5);
const r = compare({ noTrainCap: true }, 5);
assert.deepEqual(
r.deltas.map((d) => d.seed),
[0, 1, 2, 3, 4].map((i) => 1000 + i * 7919),
@@ -208,7 +317,7 @@ describe('the comparison arithmetic', () => {
it('computes the standard error from the PAIRED difference', () => {
// The whole gain of pairing is that σ of the difference is smaller than σ of either side. Using
// the level's σ by mistake would quietly restore the ±1.0 noise floor this exists to escape.
const r = compare({ trainCapSlack: 1 }, 40);
const r = compare({ noTrainCap: true }, 40);
const d = r.deltas.map((x) => x.delta);
const m = d.reduce((p, c) => p + c, 0) / d.length;
const sd = Math.sqrt(d.reduce((p, c) => p + (c - m) ** 2, 0) / (d.length - 1));