added harness to support bots being able to run / test strategies.

This commit is contained in:
Jesse
2026-08-08 15:07:42 -04:00
parent 5d825b97d2
commit 35575147cd
9 changed files with 866 additions and 23 deletions
+219
View File
@@ -0,0 +1,219 @@
/**
* The measurement machinery itself.
*
* These tests exist because a measurement tool that is quietly wrong is worse than no tool: it does
* not fail, it just produces numbers that decide what the bot becomes. Each one guards a property
* the paired method depends on.
*/
import { describe, it } from 'node:test';
import assert from 'node:assert/strict';
import { pump } from '../src/engine/advance.ts';
import { createGame } from '../src/engine/setup.ts';
import { adTrackCount } from '../src/engine/state.ts';
import type { GameConfig, GameState } from '../src/engine/state.ts';
import type { BotPolicy } from '../src/sim/bot.ts';
import { developerBot, makeDeveloperBot, playGame } from '../src/sim/bot.ts';
import { compare, parseTweaks } from '../src/sim/compare.ts';
import { makeFunnelProbe, summarize } from '../src/sim/stats.ts';
const config: GameConfig = {
mode: 'solitaire',
victory: 'highestAfterDays',
length: 'standard',
optionalRules: {
reducedVisibility: false,
sisterTrains: false,
employeeRotation: false,
emergencyToolbox: false,
},
};
const game = (seed: number): GameState =>
createGame({ id: `h${seed}`, seed, config, playerNames: ['bot'] });
/** One game, reduced to a string: every event type in order, plus the final revenue. */
function fingerprint(policy: BotPolicy, seed: number): string {
const s = game(seed);
const r = playGame(s, policy, pump);
return `${r.revenue[0]}|${r.events.map((e) => e.type).join(',')}`;
}
describe('the paired method rests on determinism', () => {
it('plays the identical game twice from one seed', () => {
/**
* THE PROPERTY EVERYTHING ELSE DEPENDS ON. A paired comparison subtracts two games that share a
* deal; if the same policy on the same seed can diverge, the difference is measuring the engine's
* own noise and every heuristic verdict is worthless.
*
* It is also the engine's central claim — pure, seeded, no ambient randomness — and the failure
* mode is silent: one `Math.random`, one Map iterated in a different order, and this is the only
* thing that would notice.
*/
for (const seed of [1000, 8919, 775569289]) {
assert.equal(fingerprint(developerBot, seed), fingerprint(developerBot, seed), `seed ${seed} diverged`);
}
});
it('gives the untweaked variant the identical game too', () => {
// `developerBot` IS `makeDeveloperBot({})`, and the refactor that introduced tweaks must not
// have changed a single decision. If this fails, every number measured before it is unusable.
for (const seed of [1000, 8919, 775569289]) {
assert.equal(
fingerprint(developerBot, seed),
fingerprint(makeDeveloperBot({}), seed),
`seed ${seed}: the untweaked variant plays a different game`,
);
}
});
});
describe('a tweak has to actually do something', () => {
it('actually holds the trains it says it holds', () => {
/**
* REGRESSION, and the reason this file exists. The first `trainCapSlack` gated the two branches
* whose comments say they exist to play a train card — and measured **exactly zero difference
* over 400 paired seeds**, because `followThrough` ends with a generic "play what is in hand"
* fallback that played the card anyway.
*
* ASSERT THE CONTRACT, NOT MERELY A DIFFERENCE. A first version of this test only checked that
* some decision changed somewhere, and it passed against the broken bot: with a tight cap the
* gated branches do change which Local Operations option is chosen, they just fail to stop the
* card being played two steps later. "Something moved" is not the promise. The promise is that
* committed trains never exceed what the Office can hold.
*
* Office tiers only ever go up, so the final A/D count is the most generous the cap ever was.
*/
const committed = (s: GameState): number =>
s.timetable.filter((t) => t !== null).length + s.pendingExtras.length;
for (const slack of [0, 1]) {
const capped = makeDeveloperBot({ trainCapSlack: slack });
let cappedWorst = 0;
let baseWorst = 0;
for (const seed of [1000, 8919, 16838, 24757, 32676, 40595, 48514]) {
const a = game(seed);
playGame(a, capped, pump);
const roomA = adTrackCount(a, 0) + slack;
cappedWorst = Math.max(cappedWorst, committed(a) - roomA);
const b = game(seed);
playGame(b, developerBot, pump);
baseWorst = Math.max(baseWorst, committed(b) - (adTrackCount(b, 0) + slack));
}
assert.ok(
cappedWorst <= 0,
`trainCapSlack=${slack} let the bot commit ${cappedWorst} train(s) more than the Office can hold`,
);
// And the cap is not vacuous: the uncapped bot really does overshoot on these seeds.
assert.ok(
baseWorst > 0,
`slack=${slack}: the baseline never overshoots on these seeds, so the test proves nothing`,
);
}
});
it('names itself so a report cannot confuse two runs', () => {
assert.equal(developerBot.name, 'developer');
assert.equal(makeDeveloperBot({}).name, 'developer');
assert.equal(makeDeveloperBot({ trainCapSlack: 1 }).name, 'developer+trainCapSlack=1');
});
it('refuses a tweak name it does not know', () => {
// A typo silently parsed as "no tweaks" would compare the bot against itself and report a
// confident zero — the most expensive possible failure of this tool.
assert.deepEqual(parseTweaks(['400', 'trainCapSlack=2']), { trainCapSlack: 2 });
assert.throws(() => parseTweaks(['trainCapSlok=1']), /unknown tweak/);
});
});
describe('the funnel counts what it claims to count', () => {
it('keeps every gate inside the total it is a fraction of', () => {
const s = game(430);
const probe = makeFunnelProbe(0);
const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn);
const f = probe.funnel;
assert.ok(f.arrivals > 0, 'no train reached an Office in a whole game');
for (const [name, n] of [
['at a passenger office', f.arrivalsAtPassengerOffice],
['with an empty coach', f.arrivalsWithEmptyCoach],
['with a loaded coach', f.arrivalsWithLoadedCoach],
['with a wanted car', f.arrivalsWithWantedCar],
] as const) {
assert.ok(n <= f.arrivals, `${name} (${n}) exceeds arrivals (${f.arrivals})`);
}
assert.ok(f.cargoPhases > 0, 'no Cargo phase was sampled');
for (const [name, n] of [
['with a facility', f.cargoWithFacility],
['green stocked', f.cargoGreenStocked],
['car spotted', f.cargoCarSpotted],
['laborer free', f.cargoLaborerFree],
['ready', f.cargoReady],
] as const) {
assert.ok(n <= f.cargoPhases, `${name} (${n}) exceeds Cargo phases (${f.cargoPhases})`);
}
// "All three at one industry" cannot exceed any of the three it is made of.
assert.ok(f.cargoReady <= Math.min(f.cargoGreenStocked, f.cargoCarSpotted, f.cargoLaborerFree));
assert.ok(f.buriedWithDigAvailable <= f.buriedTurns);
// Sampled once per Cargo phase, never once per decision: 5 Days x 12 Stages is the ceiling.
assert.ok(f.cargoPhases <= 60, `${f.cargoPhases} Cargo phases in a 5-Day game`);
// And the arrivals it counted are the arrivals that happened.
const arrived = r.events.filter((e) => e.type === 'trainArrived').length;
assert.equal(f.arrivals, arrived, 'the probe and the event log disagree about arrivals');
});
it('rides along on the harness summary', () => {
const s = game(202);
const probe = makeFunnelProbe(0);
const r = playGame(s, developerBot, pump, 50_000, probe.onEvent, probe.onTurn);
const stats = summarize(202, r.events, r.intents, s, probe.funnel);
assert.ok(stats.funnel, 'summarize dropped the funnel');
assert.equal(stats.funnel!.arrivals, probe.funnel.arrivals);
// Optional, so a caller with no probe still gets a summary.
assert.equal(summarize(202, r.events, r.intents, s).funnel, undefined);
});
});
describe('the comparison arithmetic', () => {
it('reports a dead heat as a dead heat', () => {
// Comparing the bot with itself must produce exactly zero, no games differing, and a t of 0 —
// if the machinery has any asymmetry in it, this is where it shows.
const r = compare({}, 12);
assert.equal(r.mean, 0);
assert.equal(r.identical, 12);
assert.equal(r.better, 0);
assert.equal(r.worse, 0);
assert.equal(r.t, 0);
});
it('pairs by seed, and deals the same seeds the harness does', () => {
const r = compare({ trainCapSlack: 1 }, 5);
assert.deepEqual(
r.deltas.map((d) => d.seed),
[0, 1, 2, 3, 4].map((i) => 1000 + i * 7919),
'compare and the harness must talk about the same games',
);
for (let i = 0; i < r.deltas.length; i++) {
assert.equal(r.baseStats[i]!.seed, r.deltas[i]!.seed);
assert.equal(r.variantStats[i]!.seed, r.deltas[i]!.seed);
assert.equal(r.deltas[i]!.delta, r.variantStats[i]!.revenue.net - r.baseStats[i]!.revenue.net);
}
});
it('computes the standard error from the PAIRED difference', () => {
// The whole gain of pairing is that σ of the difference is smaller than σ of either side. Using
// the level's σ by mistake would quietly restore the ±1.0 noise floor this exists to escape.
const r = compare({ trainCapSlack: 1 }, 40);
const d = r.deltas.map((x) => x.delta);
const m = d.reduce((p, c) => p + c, 0) / d.length;
const sd = Math.sqrt(d.reduce((p, c) => p + (c - m) ** 2, 0) / (d.length - 1));
assert.ok(Math.abs(r.mean - m) < 1e-9);
assert.ok(Math.abs(r.sd - sd) < 1e-9);
assert.ok(Math.abs(r.stderr - sd / Math.sqrt(d.length)) < 1e-9);
});
});