Phases 0-1 shipped in v0.4.0 (seat/identity split, per-player turn state, the Session boundary). This lands Phase 2 (server core, one game, no lobby) and Phase 3 (persistence and resumption) per docs/architecture/multiplayer.md §12. Phases 4-6 (lobby/reconnection, the 22 opponent-directed cards, StartOS packaging) are still ahead. Phase 2: src/server/session.ts hosts a game in pure logic (no sockets) on top of game.ts's existing Game/submit/currentActor/actionMenu; it verifies seat === currentActor(game) itself before calling submit, since submit() trusts its caller and a server can't. src/server/http.ts and index.ts add POST /api/game, GET /api/stream (SSE, per-seat), POST /api/intent, and static serving of dist/. src/sim/frame-delta.ts is a purpose-built per-seat board delta for one live push at a time. Found and fixed along the way: actionMenu(game, seat) only used seat for the hand field, so a server computing every connected seat's Menu would have handed the acting player's legal moves to a waiting seat. Verified with a live end-to-end smoke test (2-player game, two SSE streams, a rejected intent from the wrong seat, an idempotent resend) plus test/server/session.test.ts and test/redaction.test.ts. Not verified: an actual browser (none available in this environment). Phase 3: src/server/persistence.ts writes game.json and turn-timings.json, atomic-rewrite-then- rename. game.ts gained fromMultiplayerSave, fixing a narration-attribution bug found while testing it (fromSave's replay loop drops the actor argument, invisible in solitaire, unreadable the moment there's more than one seat — fromSave itself still has this gap, deliberately untouched). Verified live: server killed and restarted mid-game, both seats reconnected exactly where they left off. Two rules bugs found while building this: the New Train phase never implemented its car-placement round (every car of every train was placed by the Superintendent alone, in every mode, all along — now reads the round position off tray.consist.length); and victory conditions are now one shared, configurable GameConfig set across solitaire/competitive/coop instead of a fixed length lookup and a dead firstToTarget condition. Also folds in the three fixes already released on the patch line as v0.4.9b/c/d: a switching train's crew badge failing to draw once it left the Office square, an unload that always took the westmost car regardless of which was picked, and a legal decision that could render with zero buttons. docs/testing/0.5.0-test-plan.md and three reported-bug save files (docs/station-master-seed*.json) included for reproducibility. tools/jitsi-harness/ deliberately left untracked — unrelated side-project work, not part of this release. 635 tests, 0 failures.
262 lines
10 KiB
TypeScript
262 lines
10 KiB
TypeScript
/**
|
||
* Component 18b — paired A/B measurement for bot heuristics.
|
||
*
|
||
* Dev-side only. Runs the current bot and one tweaked variant over the SAME deals and reports the
|
||
* per-seed difference.
|
||
*
|
||
* WHY PAIRED, AND WHY IT IS NOT OPTIONAL. Revenue has a standard deviation of about 9 across games,
|
||
* so two 100-game runs of the identical bot can differ by a point through nothing but the deal. Every
|
||
* single-change claim in the changelog before the Interlocking work is inside that noise. Giving both
|
||
* policies the same seed removes the deal from the comparison entirely: what is left is the change.
|
||
* Measured here, the paired difference has σ ≈ 5.3 against ≈ 9 unpaired, and at 1600 seeds the
|
||
* standard error is ±0.13 — so a +0.5 heuristic is resolvable in under two minutes, where unpaired it
|
||
* would need tens of thousands of games.
|
||
*
|
||
* WHY THE SPLIT IS PRINTED. A mean carried by a skewed tail is a different claim from a mean carried
|
||
* by broad improvement, and only the better/worse/identical counts tell them apart. The train cap is
|
||
* the case in point: +0.60 overall, but 130 seeds better, 145 worse and 1325 unchanged — it removes a
|
||
* rare catastrophe rather than making the bot play better.
|
||
*
|
||
* WHY THE FUNNEL IS PRINTED. Revenue can rise because a channel started working or because the bot
|
||
* abandoned an expensive channel for a cheap one, and the two are identical in a single number.
|
||
*
|
||
* Run with:
|
||
* node src/sim/compare.ts 1600 noTrainCap=1 — what the A/D cap is worth today
|
||
* node src/sim/compare.ts 1600 noOperateFirst=1 — what operating before drawing is worth
|
||
*
|
||
* The flags are ABLATIONS: they turn off heuristics the bot already plays, so a negative delta is
|
||
* the heuristic earning its place. That is what a measured bot needs going forward — the question
|
||
* is no longer "does this help?" but "is this still true after the deck moved?".
|
||
*/
|
||
|
||
import { pump } from '../engine/advance.ts';
|
||
import { createGame } from '../engine/setup.ts';
|
||
import {
|
||
DEFAULT_MAX_COLLISIONS_PER_DAY,
|
||
DEFAULT_MAX_COLLISIONS_TOTAL,
|
||
collectiveRevenueFloor,
|
||
lengthProfile,
|
||
} from '../engine/content.ts';
|
||
import type { GameLength } from '../engine/content.ts';
|
||
import type { GameConfig, GameMode } from '../engine/state.ts';
|
||
import type { BotPolicy, BotTweaks } from './bot.ts';
|
||
import { developerBot, makeDeveloperBot, playGame } from './bot.ts';
|
||
import type { Funnel, GameStats } from './stats.ts';
|
||
import { funnelReport, makeFunnelProbe, summarize } from './stats.ts';
|
||
|
||
export type PairedResult = {
|
||
seeds: number;
|
||
baseline: string;
|
||
variant: string;
|
||
/** Per-seed revenue difference, variant minus baseline. */
|
||
deltas: { seed: number; delta: number }[];
|
||
mean: number;
|
||
/** Standard error of the mean difference — the number that decides whether this is real. */
|
||
stderr: number;
|
||
sd: number;
|
||
t: number;
|
||
better: number;
|
||
worse: number;
|
||
identical: number;
|
||
baseStats: GameStats[];
|
||
variantStats: GameStats[];
|
||
};
|
||
|
||
// Always one bot (`runOne` below), so the revenue floor is `collectiveRevenueFloor(1, days)`.
|
||
const SOLO = (length: GameLength, mode: GameMode): GameConfig => {
|
||
const days = lengthProfile(length).days;
|
||
return {
|
||
mode,
|
||
days,
|
||
minCombinedRevenue: collectiveRevenueFloor(1, days),
|
||
maxCollisionsPerDay: DEFAULT_MAX_COLLISIONS_PER_DAY,
|
||
maxCollisionsTotal: DEFAULT_MAX_COLLISIONS_TOTAL,
|
||
pvpCardsAllowed: false,
|
||
optionalRules: {
|
||
reducedVisibility: false,
|
||
sisterTrains: false,
|
||
employeeRotation: false,
|
||
emergencyToolbox: false,
|
||
},
|
||
};
|
||
};
|
||
|
||
/**
|
||
* One game, one seed, one policy.
|
||
*
|
||
* The seed stride matches `harness.ts` exactly, so a compare run and a harness run of the same size
|
||
* are talking about the same games — otherwise two numbers that ought to agree would not, for a
|
||
* reason nobody would find.
|
||
*/
|
||
function runOne(policy: BotPolicy, seed: number, length: GameLength, mode: GameMode): GameStats {
|
||
const s = createGame({ id: `cmp-${seed}`, seed, config: SOLO(length, mode), playerNames: ['bot'] });
|
||
const probe = makeFunnelProbe(0);
|
||
const r = playGame(s, policy, pump, 50_000, probe.onEvent, probe.onTurn);
|
||
return summarize(seed, r.events, r.intents, s, probe.funnel);
|
||
}
|
||
|
||
export function compare(
|
||
tweaks: BotTweaks,
|
||
games: number,
|
||
length: GameLength = 'standard',
|
||
mode: GameMode = 'solitaire',
|
||
): PairedResult {
|
||
const variantPolicy = makeDeveloperBot(tweaks);
|
||
const baseStats: GameStats[] = [];
|
||
const variantStats: GameStats[] = [];
|
||
const deltas: { seed: number; delta: number }[] = [];
|
||
|
||
for (let i = 0; i < games; i++) {
|
||
const seed = 1000 + i * 7919; // the same prime stride the harness deals
|
||
const a = runOne(developerBot, seed, length, mode);
|
||
const b = runOne(variantPolicy, seed, length, mode);
|
||
baseStats.push(a);
|
||
variantStats.push(b);
|
||
deltas.push({ seed, delta: b.revenue.net - a.revenue.net });
|
||
}
|
||
|
||
const d = deltas.map((x) => x.delta);
|
||
const mean = d.reduce((p, c) => p + c, 0) / Math.max(1, d.length);
|
||
const sd =
|
||
d.length > 1
|
||
? Math.sqrt(d.reduce((p, c) => p + (c - mean) ** 2, 0) / (d.length - 1))
|
||
: 0;
|
||
const stderr = d.length > 0 ? sd / Math.sqrt(d.length) : 0;
|
||
|
||
return {
|
||
seeds: games,
|
||
baseline: developerBot.name,
|
||
variant: variantPolicy.name,
|
||
deltas,
|
||
mean,
|
||
sd,
|
||
stderr,
|
||
t: stderr === 0 ? 0 : mean / stderr,
|
||
better: d.filter((x) => x > 0).length,
|
||
worse: d.filter((x) => x < 0).length,
|
||
identical: d.filter((x) => x === 0).length,
|
||
baseStats,
|
||
variantStats,
|
||
};
|
||
}
|
||
|
||
// ---------------------------------------------------------------------------
|
||
// Reporting
|
||
// ---------------------------------------------------------------------------
|
||
|
||
const mean = (xs: number[]): number => (xs.length ? xs.reduce((p, c) => p + c, 0) / xs.length : 0);
|
||
|
||
/**
|
||
* The verdict, stated in the same terms every time.
|
||
*
|
||
* The threshold is t ≥ 3 rather than the conventional 2. This is a measurement taken repeatedly on
|
||
* the same system while looking for something that works, so the conventional bar would have us keep
|
||
* roughly one bad heuristic in twenty; and the cost of a wrong keep is not a wrong paper, it is a bot
|
||
* that quietly plays worse and takes every later measurement with it.
|
||
*/
|
||
function verdict(t: number): string {
|
||
const a = Math.abs(t);
|
||
if (a >= 3) return t > 0 ? 'KEEP — clears the bar (t ≥ 3)' : 'REJECT — significantly worse';
|
||
if (a >= 2) return 'NOT PROVEN — suggestive, run more seeds before believing it';
|
||
return 'NO EFFECT MEASURED — inside the noise';
|
||
}
|
||
|
||
export function formatPaired(r: PairedResult): string {
|
||
const out: string[] = [];
|
||
const num = (x: number, w = 8, dp = 2): string => x.toFixed(dp).padStart(w);
|
||
|
||
out.push(`\n=== paired comparison · ${r.seeds} seeds · same deal to both ===\n`);
|
||
out.push(` baseline ${r.baseline}`);
|
||
out.push(` variant ${r.variant}\n`);
|
||
|
||
const rows: [string, (g: GameStats) => number][] = [
|
||
['revenue', (g) => g.revenue.net],
|
||
['collisions', (g) => g.collisions],
|
||
['trains scheduled', (g) => g.trains.scheduled],
|
||
['arrivals', (g) => g.trains.arrived],
|
||
['passengers on/off', (g) => g.revenue.passengerBoard + g.revenue.passengerDetrain],
|
||
['freight loads+unloads', (g) => g.revenue.freightLoad + g.revenue.freightUnload],
|
||
['cards played', (g) => g.development.cardsPlayed],
|
||
['final grid size', (g) => g.development.gridSize],
|
||
];
|
||
out.push(' baseline variant delta');
|
||
for (const [label, pick] of rows) {
|
||
const a = mean(r.baseStats.map(pick));
|
||
const b = mean(r.variantStats.map(pick));
|
||
out.push(` ${label.padEnd(22)}${num(a)} ${num(b)} ${num(b - a)}`);
|
||
}
|
||
|
||
out.push(
|
||
`\n REVENUE DELTA ${r.mean >= 0 ? '+' : ''}${r.mean.toFixed(2)} ± ${r.stderr.toFixed(2)}` +
|
||
` (t = ${r.t.toFixed(2)}, σ of the paired difference ${r.sd.toFixed(2)})`,
|
||
);
|
||
out.push(` ${verdict(r.t)}`);
|
||
out.push(
|
||
`\n seeds better ${r.better} · worse ${r.worse} · identical ${r.identical}` +
|
||
` — ${((r.identical / Math.max(1, r.seeds)) * 100).toFixed(0)}% of games are untouched by this change`,
|
||
);
|
||
|
||
// The tails, BY SEED, so a loss can be replayed rather than averaged away.
|
||
const sorted = [...r.deltas].sort((a, b) => a.delta - b.delta);
|
||
const show = (xs: { seed: number; delta: number }[]): string =>
|
||
xs.map((x) => `${x.seed} (${x.delta > 0 ? '+' : ''}${x.delta})`).join(', ');
|
||
out.push(` worst seeds: ${show(sorted.slice(0, 3))}`);
|
||
out.push(` best seeds: ${show(sorted.slice(-3).reverse())}`);
|
||
|
||
// A change that raises revenue by abandoning a channel has to be visible as that.
|
||
out.push('\n --- baseline funnel ---');
|
||
out.push(funnelReport(r.baseStats));
|
||
out.push('\n --- variant funnel ---');
|
||
out.push(funnelReport(r.variantStats));
|
||
return out.join('\n');
|
||
}
|
||
|
||
// ---------------------------------------------------------------------------
|
||
// CLI
|
||
// ---------------------------------------------------------------------------
|
||
|
||
/**
|
||
* `noTrainCap=1` -> `{ noTrainCap: true }`.
|
||
*
|
||
* Unknown names THROW rather than being ignored: a typo parsed as "no tweaks" would compare the bot
|
||
* against itself and report a confident zero, which is the most expensive way this tool could fail.
|
||
*/
|
||
export const NUMERIC_TWEAKS = new Set<string>([]);
|
||
export const BOOLEAN_TWEAKS = new Set(['noTrainCap', 'noOperateFirst']);
|
||
|
||
export function parseTweaks(args: string[]): BotTweaks {
|
||
const tweaks: Record<string, number | boolean> = {};
|
||
for (const a of args) {
|
||
const m = /^([A-Za-z]\w*)=(-?\d+(?:\.\d+)?)$/.exec(a);
|
||
if (!m) continue;
|
||
const name = m[1]!;
|
||
if (NUMERIC_TWEAKS.has(name)) tweaks[name] = Number(m[2]);
|
||
else if (BOOLEAN_TWEAKS.has(name)) tweaks[name] = Number(m[2]) !== 0;
|
||
else {
|
||
throw new Error(
|
||
`unknown tweak "${name}" — known: ${[...NUMERIC_TWEAKS, ...BOOLEAN_TWEAKS].join(', ')}`,
|
||
);
|
||
}
|
||
}
|
||
return tweaks as BotTweaks;
|
||
}
|
||
|
||
const isMain = process.argv[1]?.endsWith('compare.ts') ?? false;
|
||
|
||
if (isMain) {
|
||
const args = process.argv.slice(2);
|
||
const games = Number(args.find((a) => /^\d+$/.test(a)) ?? 400);
|
||
const lengthIdx = args.indexOf('--length');
|
||
const length = (lengthIdx >= 0 ? args[lengthIdx + 1] : 'standard') as GameLength;
|
||
const tweaks = parseTweaks(args);
|
||
|
||
if (Object.keys(tweaks).length === 0) {
|
||
console.error('nothing to compare — pass at least one tweak, e.g. trainCapSlack=1');
|
||
process.exitCode = 1;
|
||
} else {
|
||
console.log(formatPaired(compare(tweaks, games, length)));
|
||
}
|
||
}
|
||
|
||
export type { Funnel };
|