From 76c6e103b392ab3dfccede10c041a8756f46a751 Mon Sep 17 00:00:00 2001 From: "Jesse.Markowitz" Date: Tue, 15 Sep 2026 15:30:42 -0400 Subject: [PATCH] =?UTF-8?q?v0.8.0.9=20=E2=80=94=20the=20bot=20plans=20its?= =?UTF-8?q?=20switching=20turn,=20stops=20wasting=20its=20draws,=20and=20t?= =?UTF-8?q?he=20engine=20walks=20each=20route=20once?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The developer bot, re-measured decision by decision against the bot before it, goes from about -0.3 revenue a game to about 4.8: - plans the whole switching turn before its first Move (sim/switch-planner.ts), +2.89 over 1600 paired seeds; closes TODO #53 - takes a face-up card only if it could play it, +1.52 over 1600 seeds - stops running Second Sections by accident in the New Train phase, +0.32 - lays track by what the district can do afterwards, +0.12 over 6400 seeds, run-arounds in 22 of 60 districts against 9 The engine is 2.8x faster with play proven identical: a route cache scoped to one unchanged position, applyIntent split into prepareIntent + commitEvents, and less allocation in exploreMoves. npm test now leaves out the bot simulations, which run as npm run test:sim. No rule changed; games in progress resume. Rejected candidates and the Second Section card question are in CHANGELOG.md and TODO.md (#104-#106). Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017nnuCv8UodHucFfx3LWEoX --- CHANGELOG.md | 319 +++++++++++++++++++++++++++++++++ README.md | 6 +- TODO.md | 281 +++++++++++++++++++++++++++-- package.json | 6 +- src/engine/advance.ts | 5 +- src/engine/apply.ts | 72 +++++++- src/engine/legal.ts | 134 ++++++++------ src/engine/track.ts | 47 +++-- src/sim/bot.ts | 254 ++++++++++++++++++++++++-- src/sim/compare.ts | 3 +- src/sim/switch-planner.ts | 347 ++++++++++++++++++++++++++++++++++++ test/route-cache.test.ts | 88 +++++++++ test/switch-planner.test.ts | 100 +++++++++++ 13 files changed, 1550 insertions(+), 112 deletions(-) create mode 100644 src/sim/switch-planner.ts create mode 100644 test/route-cache.test.ts create mode 100644 test/switch-planner.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index e7cc67e..efeec6c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -19,6 +19,325 @@ page as `v0.1.0 · · `, so what is deployed can always be identifie --- +## 0.8.0.9 — 2026-09-15 + +**The developer bot, re-measured decision by decision — and an engine 2.8× faster.** Across the changes +adopted below, each paired against the bot before it, the bot went from about −0.3 revenue a game to +about 4.8: the switching planner +2.89, taking a face-up card only if it could be played +1.52, the +deliberate New Train fallback +0.32, and laying track by what the district can do afterwards +0.12. + +### Games in progress + +**Resume.** No rule changed, and no legality check changes its answer. The engine's speed-ups were proven +to leave play identical — every event and intent of 32 seeded games hashed before and after each step — +and `test/route-cache.test.ts` pins the new check/commit split. Only the developer bot plays differently, +and a bot's past moves are already in the save. + +### The bot plans its switching turn instead of choosing one Move at a time + +The switching branch was a ladder of rules picking ONE Move, and its own comment named the gap: "a +strong player would use the six Moves to re-order the consist — that is the game's central switching +puzzle, and this bot does not attempt it." `sim/switch-planner.ts` attempts it, for one turn. + +**Why search is fair.** A switching turn draws no card and rolls no die, so trying sequences on a copy +of the game is what a player does by looking at the board. The score reads only what a player can see +— the district, the cars on the trains, the facilities — and never the deck. Jesse's line +(2026-09-14): no non-player advantage for the bot. + +**Why not every sequence.** Measured over 30 switching turns from bot games: a median turn reaches +229 distinct positions, but 11 of 30 passed 20,000, because **setting cars out costs no Move** and a +crew can leave them in a great many places. So the search keeps the best 48 positions at each step +and stops at 3,000 tried; small turns are searched completely inside that. + +**The score is of where the turn ends**, starting from Jesse's ruling that players deliver and pick up +cars even when it delays trains. A car an industry can work is +1 (+0.25 past its box count); a car it +cannot is −0.5 on its track; a finished car left on it −0.15; a wanted car aboard +0.35 or staged on +plain track +0.2; a coach kept with its train +0.3, stranded −0.3; anything fouling the Office −3; a +train away from the Office −0.25, −1 more if expedited; a train §8.2 would refuse (`badlyMadeUp`, now +exported rather than copied) −0.6. A load made in this district scores nothing at a receiver here. + +**The copy is partial.** `forkForSwitching` copies only the arrays a switching intent writes — cars +standing on cards, industry tracks, the district's A/D and held lists, the trays' consists, this turn +and the tally — and shares everything else. It began as a `structuredClone` of the district, which a +profile put at **44% of all planning time**; the targeted copy made a position about three times +cheaper to try. `test/switch-planner.test.ts` +proves across seeded games that planning leaves the real game byte-identical, and that every plan +replays through `applyIntent` on a FULL copy to the exact position it promised. + +**Measured, 1600 paired seeds: +2.89 ± 0.18 revenue a game (t = 15.79)**, 733 seeds better, 21 worse. + +| | rules | planner | +| --- | --- | --- | +| revenue | −0.05 | 2.83 | +| freight loads + unloads | 0.22 | 1.53 | +| collisions | 0.12 | 0.12 | +| Cargo phases with a car spotted | 5% | 18% | +| … with green box, car and Laborer at one industry | 3% | 11% | + +**Split by source over 400 seeds**, because the biggest wins were rescues: freight **+1.21** (t = 13.9), +passengers +0.26, expedite faults **+1.07** (19 of 400 games faulted under the rules, none under the +planner — TODO #53, closed by this), collisions unchanged. **Without the fault rescue it is still ++1.45 ± 0.12 (t = 12.2)** — the switching itself got better, not just the catastrophes rarer. It also +sets out half as many cars (3.6 against 7.0): the ladder was setting cars down and picking them up. + +**The worst seeds, traced.** Two of the three lost to "no free A/D track" collisions with both tracks +already held by trains standing at the Office — not trains left away, which is what the score's +`trainAway` weight would have explained. The third never had a coach train at the Office in a +Load/Unload phase at all: a different switching turn draws different cards, and the game diverges. +Neither is a scoring defect found; the full-Office collisions are worth watching. + +**How far to search.** The search size was measured rather than guessed, paired over 400 seeds against +the 3000-position, beam-48 search the gain above was measured with: + +| budget / beam | revenue against 3000/48 | median / worst per turn | +| --- | --- | --- | +| 3000 / 48 | — | 150 ms / 445 ms | +| **2000 / 32 (adopted)** | −0.02 ± 0.01 (t = −1.68), inside the noise | **69 ms / 222 ms** | +| 1000 / 24 | −0.06 ± 0.02 (t = −2.65) | 53 ms / 104 ms | + +Times are from 51 switching turns with other simulations sharing the CPU, so read them as relative. + +**A switching-only legal list.** `legal.ts` now exports `legalSwitchingActions`, the switching half of +the Local Operations candidates run through the same `check`, in the same order — so the planner stops +paying for every draw and Freight Agent candidate at each position it tries. A test asserts it equals +`legalActions`' switching subset across real games. It contains no rule; `check` still decides. + +**Nothing the simulation tests measure moved backwards.** `sim.test.ts` passes all 35 tests with the +planner as the default, floors unchanged. + +**Cost.** A turn is planned once and then played a step per decision, replanned if the position is +ever not the one expected. At a live table that is a synchronous pause inside `driveBots`, well inside +Jesse's bar of half a second before a switch. + +**The price is test time.** A standard solitaire game takes ~400 ms with the planner against ~108 ms +without, and `npm test` — which plays well over a thousand bot games in `sim.test.ts` — went from about +150 s to **8 min 6 s** (992 of 992 passing, measured with nothing else running). What is left of a +planned turn's time is the engine itself: applying a move is about half of it and `check` a third. + +**And the take rule tripled it again.** With both defaults in, `npm test` passes 992 of 992 in +**21 min 21 s** with nothing else running — the bigger districts the take rule builds (14 → 24 cards, +29 cards played a game) make every game longer to play and every switching turn wider to search. + +`noPlanSwitching=1` is the ablation. + +### Track goes where it lets the district do something + +Jesse's third area, after switching and industries: track for the run-around. With the draw no longer +wasted, ~13 of ~16 pieces a game were still being laid by the draw turn's fallback at the first legal +square, and both fixes to that fallback failed (holding −0.83, `bestTrackLay`'s own score +0.07). So the +limit was the scoring: `bestTrackLay` scores the PIECE, and cannot tell one that opens an industry site +or closes a run-around from one that fills a square. + +`bestValuedLay` scores the LAYOUT the piece would leave instead. Each legal lay is placed on a copy of +the district exactly as the reducer places it — `protoCard` and `extendLimitsIfNeeded`, now exported +rather than copied — and worth the change it makes to `layoutValue`: industry sites a crew can reach +(`canPlaceAt`), a closed run-around (the bot's own `descendFrom` walk), ways off the main and reachable +siding, a Running Track straight for Interlocking, and a penalty for a turnout on the main whose leg +joins nothing. `bestTrackLay`'s slot takes the best lay that gains something; the fallback still lays as +often as before — holding starved the district — but the best lay rather than the first. + +**+0.118 ± 0.029 revenue a game (t = 4.14) over 6400 paired seeds**, 1,667 better against 1,430 worse. +It needed that many: 400 seeds read +0.17 (t = 1.48) and 1600 read +0.14 (t = 2.58). What it builds is +the larger change — **closed run-arounds in 22 of 60 districts against 9**, industries 1.78 → 1.87, freight +2.54 → 2.65, collisions unchanged, the district 24.2 → 21.9 cards because squares stop being filled with +pieces that build nothing. No slower: a standard game measured faster, the smaller districts leaving +less to search. `noValueLays=1` is the ablation. + +**Why so many run-arounds buy so little.** Traced over 40 games: the planner uses the loop — a move ends +on it in 63 of 131 plans where one exists, 35 run it both ways — but its planned gain per turn is the +same with a run-around as without (0.28 against 0.27). A run-around is for putting a train's cars in a +different order, which pays off in the turns AFTER; a one-turn planner cannot value it. That is TODO #105. + +**How 6400 seeds were measured.** `compare.ts` keeps every game's full statistics for both sides, and the +run was stopped for low memory — as was a first lean attempt, by the machine's background-task guard +rather than by the process: a probe showed no leak (heap flat at 9 MB after GC across 300 paired games). +The figure above comes from a paired script with the same seeds and configuration that keeps only each +seed's revenue difference, run in resumable foreground chunks; its first 1600 seeds read +0.143, matching +`compare.ts`'s +0.14. + +### The engine walks each position's routes once — 2.8× faster, every game identical + +Jesse's call (2026-09-15): speed up the engine's move checking and applying first, because the live +server runs the same `legalActions` and `applyIntent` for every bot and every player. A CPU profile +with inlining off put the route walk (`reachableDestinations`) at a third of all time and garbage +collection at another third, and most of the walking was repeated work. + +- **A route cache scoped to one unchanged position** (`withRouteCache`, `apply.ts`). `legal.ts` walked a + tray's routes to list its moves, `check` walked them again for every one of those moves, and + `execute` walked the chosen move a third time. Now a legal-action listing, and the check-and-execute + of one intent, walk each route once. The cache is keyed to the state object and the walk's inputs + and dropped before `reduce` changes anything, so a hit is exactly what a fresh walk returns. +- **`applyIntent` = `prepareIntent` + `commitEvents`.** The planner decides every candidate against one + position inside one cache and commits each to its own copy; deciding on the copy had re-walked every + route the listing had just walked. +- **Less garbage in the walk itself** (`exploreMoves`): the blocked-square explanations are not built + when only destinations are wanted, the queue is read by index instead of `shift()`, and "has this route + been here" reads the route's own path instead of copying a Set at every step. + +**A standard solitaire bot game, 561 ms → 203 ms; a short one, 141 ms → 65 ms.** Proved identical by +hashing every event and intent of 32 seeded games (20 solo standard, 6 three-player standard, 6 +two-player short) before and after each change, and pinned by `test/route-cache.test.ts`, which checks +at every decision of a seeded game that `prepareIntent` writes nothing and that committing its events +to a copy equals `applyIntent` in place. + +### The test suite is split + +Jesse (2026-09-15): `npm test` runs after every change, so it has to stay fast; reducing the simulations' +game counts is not acceptable. `npm test` now runs every file except `test/sim.test.ts`, and +`npm run test:sim` runs the bot simulations on their own, typechecked first. + +**Measured with nothing else running, after the engine speed-up: `npm test` 958 of 958 in 1 min 45 s; +`npm run test:sim` 35 of 35 in 6 min 58 s** — against 21 min 21 s for the two together before it. + +### Where industries go was not the problem — there was nowhere, and the draw was going in circles + +The next thing Jesse named after switching was where industries and enhancements go, then track for the +run-around. Measured before touching either, 30 standard games, and it turned both round: + +- **The bot places an industry every time it legally can.** It held an industry card at 1,441 draw + decisions and a legal square existed at 24 of them (2%); it played the industry at all 24 and never + discarded one that had somewhere to go. 0.70 industries stand on a board at game end. +- **Why there was nowhere.** An industry is a plain east-west piece that must join existing track and + may not sit on the Running Track. Asked of `check` square by square, the closest a held card got was + NOT_CONNECTED 1,681 times, OUTSIDE_LIMITS 398, legal 28 — never locked out. Printed districts show why: + the row beside the main fills with 45° curves and turnouts whose east-west ends face occupied squares. +- **And track was not the lever either — its supply was.** At 1,334 decisions holding an industry with + no site, a track lay was legal at only 54; a lay that would open a site existed at 22 and the bot + already chose one at 17. The hand had no track in it. +- **Because the draw was cycling.** The bot took 20.1 timetabled trains and 11.9 industries a game off + the Departments and discarded 20.2 and 11.8: `takingRank` ranked a face-up train 2 whether or not the + A/D cap would let it be played, and a face-up industry 1 whether or not it had a site (one did at 0.17 + of those takes). The same card was taken again 28.8 times a game, against 19.6 blind draws. + +### The bot takes a face-up card only if it could play it + +`takingRank` now ranks a face-up train card 0 when the A/D cap would hold it, and a face-up industry 0 +when it is locked out or has no legal site — both asked of the rules the play itself is checked by +(`trainWouldOverfillTheOffice`, and `isLockedOut` with `canPlaceAt` in `industrySiteExists`). Nothing +here is hidden information: the Departments are face up. + +**+1.52 ± 0.10 revenue a game (t = 15.59)** over 1600 paired seeds, measured as the ablation +`noPlayableTakes=1` against the new default: 880 seeds worse without it, 211 better. (The first +400-seed read was +1.63 ± 0.19.) Cards played 15.8 → 29.0 a game and the district grows 14.2 → 24.2 +cards, because the draws it stopped wasting now bring in track: trains scheduled 1.74 → 2.30, freight +1.51 → 2.54, passengers 1.89 → 2.94. **Collisions rose, 0.12 → 0.24**, with the extra trains — see +TODO #106. + +### Rejected: holding track the run-around scoring declined + +Traced first: of ~16.4 track pieces a standard game lays, only ~3.3 came from `bestTrackLay`. The +draw turn's "play what is in hand" fallback laid the other ~13 at the FIRST legal square — pieces +`bestTrackLay` had just declined, including 4.0 turnouts and 4.3 straights a game on the main and 1.5 +turnouts a row off it. The candidate let the fallback lay track only when nothing else could bring the +hand under the limit. **−0.83 ± 0.16 (t = −5.17)**, 172 worse against 90 better: the district fell +24.1 → 11.7 cards and freight 2.41 → 1.77. `bestTrackLay` declines most pieces, so holding them +starves the district — #59's finding, again, with a fuller hand to hold them in. + +### Rejected: laying fallback track where it scores best instead of first + +The other half of the same trace: keep laying the ~13 fallback pieces a game, but at `bestTrackLay`'s +best-scoring square (bonus or not) rather than the first legal one. **+0.07 ± 0.13 (t = 0.60), inside +the noise**; freight 2.41 → 2.56 but the district 24.1 → 21.0 cards. With both halves measured, the +limit is `bestTrackLay`'s SCORING — it cannot tell a piece that opens an industry site or advances a +run-around from one that fills a square — not which branch places the piece. Code removed. + +### The bot stops running Second Sections by accident + +TODO #106 traced under today's defaults: of 20 "no free A/D track" collisions in 60 standard games, no +train had been held before any of them and only 8 of the destroyed were Extras. Six destroyed a train of +the same number as a Second Section run within two Stages, and **all 26 Second Sections the bot ran were +an accident**: when no car was on offer, the New Train phase fell back to `options[0]`, and `legalActions` +lists `newTrain.secondSection` ahead of the Extra starts — so a waiting Extra became a doubled train due +out, into an Office that never had an A/D track to spare. The fallback now takes a car, a pass or the +Extra's start, and never a Second Section or a Red Flag merely because it was listed first. + +**+0.32 ± 0.09 revenue a game (t = 3.64)** over 400 paired seeds, 30 better against 12 worse; collisions +0.24 → 0.19. `noDeliberateNewTrain=1` is the ablation. + +**Found on the way, for Jesse — a rules question, not changed.** Q9 defines the Second Section as a card +played on a train due out and `content.ts` gives `SECOND_SECTION` one copy, but `buildDeck` never deals it +and `check` asks for no card, so any player can run one for free on every train due out. + +### Rejected: planning two switching turns ahead (TODO #105) + +Built as `planTwoTurns`: keep the six best ends of a switching turn, remove the trains that highball in +the Mainline Phase in between (on the Office square and made up — their cars leave with them), reset the +Moves, plan the next turn from each, and choose by the position after departures plus 0.8 of what the +next turn adds. **−0.28 ± 0.11 (t = −2.58)**; corrected to charge the expedite fault the gap cannot undo +and a Stage for every train left away, **−0.14 ± 0.06 (t = −2.36)** — 400 paired seeds each. A second turn +has little to find: only 49% of switching turns keep their train for the next, a further turn could have +spotted just 0.7 of the 6.1 wanted cars a game that leave aboard departing trains, and trains left away +cost Stages walking crews home. Dropped at Jesse's call; the measurements are in TODO #105. + +### Rejected: starting an Extra the Office cannot take where its run never arrives + +TODO #106, re-measured under today's defaults: 20 collisions in 60 standard games (−1.67 revenue a +game), every one "no free A/D track", and 26 Extras forced out by a full hand of them. The bot also took +the first legal start for EVERY Extra — the western Division Point, 158 of 158 — so the candidate chose, +when the committed trains exceeded the A/D tracks, a start whose run never reaches the Office: an Extra +runs away from its start (`resolveExtraStart`), and at the Interchange one way may miss the Office. +**+0.10 ± 0.05 (t = 1.82), 395 of 400 seeds identical**: such a start was on offer at 1 of 25 over-cap +starts — most Divisions have no Interchange, and a Division Point always runs through the Office. Code +removed. + +### Rejected: refusing to draw into a forced Extra + +With the take rule in, the bot's worst seeds all lost to "no free A/D track". Traced: every one of 23 +train plays past the A/D cap in 40 games was an EXTRA, played because the hand held four of them and +the only legal action was `card.play` — §6.2 makes a player over the limit reduce the hand, and an Extra +may never be discarded (`keepReason`), so the cap in `choose` yields rather than leave nothing legal. +The candidate declined to choose the draw option with a full hand of such cards when switching or the +Freight Agent was on offer. **−0.03 ± 0.11 (t = −0.30), inside the noise**: collisions fell +0.26 → 0.20, and development fell with them (cards played 29.0 → 25.6) — the Stage spent avoiding the +draw was a Stage not spent building. Code removed; the trap itself is real and filed. + +### Rejected: discarding the card least likely to become playable + +When a discard was forced, the candidate picked WHICH card by a keep-value (next Office tier, then +Enhancements, track, trains, and industries with no site or Modifiers with no industry last), weighted +so no pile preference could overturn it. **−1.16 ± 0.12 (t = −9.27)**, 16 better against 119 worse, +cards played 15.7 → 10.4. Not traced; the likeliest reason is that the weight overrode `bestDiscard`'s +pile choice, which exists to avoid burying a face-up card the bot wants. Code removed. + +### Flying Switch is searched, and cannot be measured + +The planner now considers Flying Switch alongside Moves, set-outs and sorts — its partial copy of the +game carries the hand and the Salvage Yard, which `spendCard` writes, and spending the card costs a +tenth of a point so it is played only when it buys something. **Over 400 paired seeds it changed +nothing, because the card is dealt 0 copies** (not in sheet 5; Jesse, 2026-08-26): across 30 standard +games no Flying Switch was ever drawn. TODO #58's "never fires" is therefore a deck fact, not a bot +one. On by default, so the planner uses the card the day it is dealt again. + +### Rejected: letting the planned gain decide whether to switch at all + +Whether a Stage goes to switching is decided by `usefulSwitching`, a yes/no reading of the district. +The candidate replaced it with the planner's own answer — plan the turn on a fork before the option is +chosen, and switch only if it gains at least a threshold. Paired over 400 seeds against the planner: + +| threshold | revenue | seeds better / worse / identical | +| --- | --- | --- | +| 0.5 | **−0.33 ± 0.06 (t = −5.06)** | 15 / 70 / 315 | +| 0.25 | −0.01 ± 0.05 (t = −0.21) | 21 / 22 / 357 | +| 0.1 | +0.06 ± 0.03 (t = 1.79) | 20 / 13 / 367 | + +At 0.5 it refuses turns that only COLLECT — a wanted car picked up scores +0.35 — and those pickups feed +the deliveries after them. Below that it agrees with `usefulSwitching` almost everywhere, because that +rule already says yes exactly when there is a car to deliver or lift, which is when a plan gains. The +decision that would matter is switching against DRAWING or the Freight Agent, and the bot has no value +for either to compare with; that is the next piece of work, not a threshold. It also cost a full +search at nearly every Local Operations decision. Code removed. + +### Rejected: discounting a car its industry cannot work yet + +A spotted car scored half when its shipper had no load staged or its receiver's red box was full. +**−0.29 ± 0.05 (t = −5.52)** over 400 seeds, 47 worse against 4 better; green boxes stocked fell +15% → 11% of Cargo phases. The reason is the order the bot works in: the Freight Agent stocks a box +only once a car is spotted to receive it (`canStockProductively`), so an industry is "not ready" +precisely because nothing has been delivered — the discount withheld the delivery that makes it ready. +Code removed. + +--- + ## 0.8.0.8 — 2026-09-10 **A played train does not come back. Gitea#23, ruled and closed.** diff --git a/README.md b/README.md index 6bd7ade..7bc074e 100644 --- a/README.md +++ b/README.md @@ -56,7 +56,8 @@ deliberately no longer names one: it went stale for six releases. Balance is *not* where it should be, and this file no longer quotes a figure for it. It used to say "the developer bot averages 7.0 Revenue against a target of 20", which stopped being true the moment the transit rule it names was defaulted to off — that rule was worth ~5.4 of the 7.0, for traffic -nobody had to work. Measured at the current defaults the bot means about **zero**. +nobody had to work. Measured at the current defaults the bot meant about **zero** until it began +planning its switching turns (2026-09-14), which put it near **2.8**. The three rates — passenger per coach, freight per load, train per transit — are **settings fixed when the game is dealt**, along with the opening hand and where an Extra may start, so the economy can be @@ -102,7 +103,8 @@ separate thing: it assembles the static SITE into `dist/`.) ```sh npm install -npm test # node --test +npm test # node --test, everything except the bot simulations — run after every change +npm run test:sim # test/sim.test.ts, the bot simulations (~7 min) — run after a bot or balance change npm run typecheck # tsc --noEmit ``` diff --git a/TODO.md b/TODO.md index 2fcc149..1351fd3 100644 --- a/TODO.md +++ b/TODO.md @@ -81,7 +81,7 @@ Not items. Things that are true of every change, and that have gone wrong when s 5. **Replays and saved games** — #14 #47 #48 #49 #50 #51 #52 6. **Rules** — #12 #80 #82 #83 #85 7. **Play balance** — #61 #62 #63 #64 #67 #68 #69 #70 #71 #72 #73 #66 #65 #74 -8. **The bot** — #41 #57 #59 #53 #54 #58 #55 #56 #60 +8. **The bot** — #104 #105 #106 #41 #57 #59 #54 #58 #55 #56 #60 9. **Code health and housekeeping** — #46 #45 #84 #87 10. **Documentation and assets** — #15a #86 #88 @@ -100,20 +100,119 @@ Everything else in this file waits behind a release; this waits behind an aftern **Test runs WERE made across 0.7.4 through 0.7.9** (Jesse, 2026-09-07) and produced no change requests — the two bugs that did come out of them are Gitea#21 and #22, fixed in v0.7.9.1. So this section is not "nobody has touched it since 0.7.4"; it is the narrower and still-true claim that the -specific paths below have not been exercised at a table. **More testing is planned at the end of the -0.7.9 series, before 0.8.0 starts** — that is the moment to close these, not a separate errand. +specific paths below have not been exercised at a table. + +**The gate moved.** It was "before 0.8.0 starts"; 0.8.0 shipped anyway, through v0.8.0.8, so the +session now runs against that build and covers what it added as well. See **Preparing the session** +below — written 2026-09-10 because the measurement it rests on is the whole point: **three of the +four things this section is named for do not happen by themselves.** + +### Preparing the session + +**MEASURED, 2026-09-10, across ten full competitive games driven to completion.** What a table will +meet without trying, and what it will not: + +| interruption | fires in | so | +| --- | --- | --- | +| Superintendent clearance (§8.1) | **9/10 games** | you will meet it; just play | +| a train held at the Limits | 7/10 | ditto | +| Extras started and queued | 10/10 | ditto | +| collisions | 7/10 | ditto | +| Red Flags set / spent | 7/10, 6/10 | ditto | +| **the Yard Office offer** | **0/10** | must be set up | +| **the Red Flag hold and its prompt** | **0/10** | must be set up | +| **extended play (`dayExtended`)** | **0/10** | must be set up | + +Those last three are exactly what #39 and #35 are NAMED for. They are not broken — they are +conditional, and the conditions are these, read out of `advance.ts` rather than guessed: + +- **Yard Office** (`advance.ts` ~1290) needs the destination district to contain a card carrying the + `yardOffice` **enhancement**, AND an arriving train with **no coach** in its consist, AND a usable + route. The bot never builds one, so **somebody has to build a Yard Office and then let a freight + train arrive.** +- **Red Flag hold** (`advance.ts` ~1232) needs the destination player to be **holding the Red Flags + maneuver card**, AND an arrival that would genuinely collide — §8.3's own two ways: no free A/D + track, or cars fouling the Running Track. So: **hold that card and let your A/D tracks fill.** +- **Extended play** needs the timetable to RUN OUT, which a five-Day game does not do. Deal it with + **`days: 1`** — that is exactly what the 2026-08-29 API verification did, and why it got there. + +**What the session needs** + +- **Two people, two browsers, two devices.** #35's remaining gap is specifically what a SECOND + player sees while waiting on a first, and whether "waiting on Carol" still reads once Carol has + closed her laptop. That cannot be tested alone, and it is the half that has never been done. +- **Two games, not one.** A short `days: 1` game to reach the extension vote, and an ordinary game + for everything else — with somebody deliberately building a Yard Office and holding Red Flags. +- **#42a is separate and takes five minutes**, solitaire, one person: click every field on the setup + screen and confirm the dealt game matches what was chosen. + +**The caution this section exists because of.** #35's own Reference entry records that the +2026-08-29 verification passed over the HTTP API — **which renders no dialog** — and that is exactly +why the v0.7.9 bug survived: the vote sat underneath a modal results dialog whose only control was +Close. What was proven was that the SERVER supports extended play, not that a player can reach it. +Read that into every "verified on `phoenix.local`" line in this file, and into everything v0.8.0 +added, all of which is verified by test and simulation and none of it by eye. + +**What v0.8.0 added to this list**, none of it played by a person for a whole game and none with a +second human: the watchable board and its ordered steps, the speed control, the pile highlighting and +the Home Office deck tile, "Your Move" being put away while catching up, the Day-end collision line, +and the Salvage Yard naming its top card. + +### The checklist + +Grouped by what has to be set up, with the item each observation closes. Nothing here needs a +developer present; what it needs is somebody writing down what they saw. + +**Game A — `days: 1`, two humans, two browsers.** Reaches the extension vote in one Day. + +- [ ] The vote appears **in front of both players**, not underneath the results dialog (#35 — this is + the exact shape of the bug v0.7.9 fixed). +- [ ] While one player has not voted, the other's turn chart says **who** it is waiting on (#35). +- [ ] **Close the second laptop mid-vote.** Does the first player learn why nothing is happening, and + does "waiting on Carol" still read once Carol is gone? (#35 — never tested.) +- [ ] Reopen it. The history panel comes back **populated**, not empty, and the board is current + (the v0.7.9.5 reconnect fix, never seen by a person). +- [ ] Vote yes. The extra Day begins and the official result is **unchanged** from when the + timetable ran out (#35). + +**Game B — ordinary length, two humans, bots to fill.** Everything else. + +- [ ] Somebody **builds a Yard Office** and lets a freight train (no coach) arrive at it. The offer + interrupts the Mainline Phase and asks a question mid-thought — is it legible, and does it say + which train? (#39) +- [ ] Somebody **holds the Red Flags card** while their A/D tracks are full, so an arrival would + collide. The hold is offered out of phase (#39). +- [ ] A **loaded Extra** is made up and run (#39 — the third of its three). +- [ ] Watch a bot take a whole turn: does the district follow it, does the lit pile catch the eye, + does the caption say who and what? (v0.8.0) +- [ ] Find the speed that suits you and say what it is — it becomes the committed default. +- [ ] Let the board fall behind, then press **Skip**. Nothing is lost; the history has it all. +- [ ] End a Day with a collision on it: the summary reads "N on Day D, N in all" and cannot + contradict itself (v0.8.0.2). + +**Solitaire, five minutes, alone.** + +- [ ] Click through **every field** on the setup screen and confirm the dealt game matches what was + chosen (#42a). + +**Whatever else happens.** The two bugs that came out of the 0.7.4-0.7.9 runs were both things +nobody set out to test. Write down anything that reads wrong, even where the rule underneath is +right — most of this release's defects were legible-but-wrong rather than broken. - [ ] **#39** — **None of v0.7.4 has been played by a human.** The Yard Office offer, the Red Flag hold and its out-of-phase prompt, and the loaded-Extra make-up rules are tested end to end, packed, and running on `phoenix.local` — and nobody has met any of them at a board. **Two are interruptions that stop the Mainline Phase and put a question in front of somebody - mid-thought**, which is exactly the kind of thing only play reveals. See **Reference · #39**. + mid-thought**, which is exactly the kind of thing only play reveals. **Neither of those two + happens by itself — 0/10 games. See Preparing the session above for what to set up.** See + **Reference · #39**. - [ ] **#35** — **Extended play has never been played at a real table.** It was verified over the HTTP API, which renders no dialog — and when a human first reached it in a browser it was unusable (fixed in v0.7.9). The multiplayer vote has still never been driven through two browsers: what a second player sees while waiting, and whether "waiting on Carol" reads once Carol has closed her - laptop, are unanswered. See **Reference · #35**. + laptop, are unanswered. **Needs a `days: 1` game — the timetable does not run out in five + Days, so extended play fired in 0/10 measured games.** See **Reference · #35**. - [ ] **#42a** — **Nobody has clicked through the solitaire setup screen's own fields** and confirmed the dealt game matches what was chosen. It took three attempts to become reachable at all — @@ -368,21 +467,37 @@ v0.7.9's collision-floor change (#61). The developer bot exists to measure the game, not to be a good opponent — so a bot weakness matters when it stops a measurement being trustworthy. **Read #57 before tuning any weights.** +**Since 2026-09-14 the bot plans its whole switching turn** (`sim/switch-planner.ts`, +2.89 revenue a +game), **takes a face-up card only if it could play it** (+1.52), and **since 2026-09-15 lays track by +what the district can do afterwards** (`bestValuedLay`, +0.12 over 6400 seeds, run-arounds 9/60 → 22/60). Jesse's goal for it is better decisions in simulated runs AND at a real table, with no +non-player advantage — it reads the board, never the deck. + +- [ ] **#104** — Weigh a switching turn against drawing and the Freight Agent. Letting the planned gain + gate switching on its own measured nothing (0.1, 0.25) or worse (0.5): `usefulSwitching` already + says yes exactly when a plan gains. What would matter is a VALUE for the other two options to + compare against, which the bot does not have. See **Reference · #104**. + +- [ ] **#106** — The Extra trap: a full hand of Extras the A/D cap is holding back cannot be discarded, + so the next draw forces one into a full Office. All 23 train plays past the cap in 40 games were + this. Avoiding the draw measured nothing (−0.03) because it stalled development. See + **Reference · #106**. + +- [ ] **#105** — Plan across more than one turn. Jesse is in favour, one turn first to see the impact — + which is now measured. Deferred for a conversation, not declined. See **Reference · #105**. + - [ ] **#41** — The bot never plays Red Flags — zero in 200 games since Gitea#19, and that is deck luck rather than unwillingness. It takes the danger prompt unconditionally; what it never does is plant a flag ON PURPOSE to buy a Stage for switching, which needs it to know it wants time. See **Reference · #41**. - [ ] **#57** — The bot's priorities are not the problem — measured across ten heuristic variations. - **Read this before tuning weights**; it is the argument that the ceiling is elsewhere. See - **Reference · #57**. + **Read this before tuning weights**; it is the argument that the ceiling is elsewhere. **Part of + "elsewhere" was choosing one Move at a time**: planning the whole switching turn was worth + +2.89 (t = 15.8) in the 2026-09-14 bot-tuning round. See **Reference · #57**. - [ ] **#59** — The run-around is out of reach of any bot, and the deck is why — measured five ways. See **Reference · #59**. -- [ ] **#53** — The bot does not know to bring an expedited train back to the station. See **Reference - · #53**. - - [ ] **#54** — The bot cannot spot a car at a stub industry, and the cut-ordering rules made that visible. See **Reference · #54**. @@ -1424,6 +1539,12 @@ measuring deck luck rather than reachability, and its comment now says so. #### #57 — THE BOT'S PRIORITIES ARE NOT THE PROBLEM — measured. +**2026-09-14 — confirmed, and one ceiling found.** Reordering priorities still moves nothing; what +moved the bot was SEARCH. `sim/switch-planner.ts` plans the whole switching turn against a score of +where it ends, and measured +2.89 ± 0.18 (t = 15.79) over 1600 paired seeds — freight loads and +unloads 0.22 → 1.53, Cargo phases with a car spotted 5% → 18%. The "8% of Cargo phases" below was a +fact about how the bot switched, not only about the deck. See `CHANGELOG.md`, 0.8.0.9. + **THE BOT'S PRIORITIES ARE NOT THE PROBLEM — measured.** Ten heuristic variations, each paired over 400+ seeds. Every reordering of what the bot prefers came out inside the noise; the only @@ -1449,6 +1570,21 @@ prioritised better. What is left is the economy itself, which is a deck question #### #59 — THE RUN-AROUND IS OUT OF REACH OF ANY BOT, AND THE DECK IS WHY — measu… +**2026-09-14 — part of "the deck is why" was the bot's own draw.** It was taking ~32 face-up trains and +industries a game it could not play and discarding them again, so the hand rarely held track. Taking +only playable cards (now the default) grew districts 14 → 24 cards and run-arounds 3/60 → 9/60. And +~13 of the ~16 track pieces a game were being laid by the draw turn's "play what is in hand" fallback +at the first legal square, not by `bestTrackLay`; holding them (−0.83) and placing them by +`bestTrackLay`'s score (+0.07, noise) both failed, so the next limit is that SCORING — it cannot tell a +piece that opens an industry site or advances a run-around from one that fills a square. The deck +measurements below were taken before any of this and should be re-read with it in mind. + +**2026-09-15 — the scoring, fixed.** `bestValuedLay` scores the layout a lay leaves (reachable industry +sites, a closed run-around, ways off the main) instead of the piece: closed run-arounds in 22 of 60 +districts against 9, +0.118 ± 0.029 revenue (t = 4.14, 6400 seeds). The run-around is now reachable +without changing the deck; what is left is a bot that can USE one, which needs more than one turn of +planning (#105). + **THE RUN-AROUND IS OUT OF REACH OF ANY BOT, AND THE DECK IS WHY — measured, five ways.** "Teach the bot to plan across turns" was tried properly and does not work. Every attempt is @@ -1490,6 +1626,8 @@ the end of the game, drawing the fault **26 times**. Not an engine bug — the m exactly as designed — but a clear next bot heuristic: prefer ending a switching turn with any expedited crew back on the Office square, at least once it has finished the work it went out for. +**CLOSED 2026-09-14** — see **Done · 53**. + #### #54 — THE BOT CANNOT SPOT A CAR AT A STUB INDUSTRY, and the cut-ordering rul… **THE BOT CANNOT SPOT A CAR AT A STUB INDUSTRY, and the cut-ordering rules made that visible.** @@ -1511,6 +1649,12 @@ pass should not read the drop as a deck problem. #### #58 — The bot cannot get a crew next to an industry, so Flying Switch never… +**2026-09-14 — the premise is now a deck fact.** Flying Switch is dealt **0 copies** (not in sheet 5; +Jesse, 2026-08-26), so no bot can fire it: across 30 standard games none was ever drawn. The switching +planner searches the card by default, so it will be used the day it is dealt again. The reachability +sweep's exemption in `sim.test.ts` stays until then. + + **The bot cannot get a crew next to an industry, so Flying Switch never fires.** Industries are now stub-only and the bot places 2.23 a game (was 3.84), in districts averaging under two rows @@ -1558,6 +1702,114 @@ of zero" over 400 games — but at 400 games the standard error is ±0.33, so a have looked like nothing. They are nearly free to re-run now and at least one may have been discarded wrongly. +#### #104 — WEIGH SWITCHING AGAINST THE OTHER TWO OPTIONS, not against a threshold. + +Measured 2026-09-14, paired over 400 seeds against the planner: switching only when the planned gain +clears a threshold scored −0.33 at 0.5 (t = −5.06), −0.01 at 0.25, +0.06 at 0.1 (t = 1.79). At 0.5 it +refuses turns that only collect cars, which feed later deliveries; below that it agrees with +`usefulSwitching`. The choice that is still made by a fixed ladder is WHICH of §6's three options a +Stage goes to, and the planner can now put a number on one of them. The other two need numbers of +their own — what a draw is worth given the hand and the Departments, what stocking a box is worth given +the cars spotted — before the three can be compared. Also: planning at every Local Operations decision +costs a full search each time, so any version of this has to stay cheap. + +**2026-09-15 — measured the ceiling first: there is almost none.** At 907 real Local Operations choices +(75 standard games, seeds outside the usual measurement range), every legal option was tried and the rest +of the game played out by today's bot, 4 times each with the HIDDEN parts reshuffled — the Home Office +deck order and future rolls — and the same reshuffles for every option, so the comparison is paired. +Grouped by the rule that made the choice, the value of each alternative against it: + +| the ladder chose | times | switch instead | draw instead | Freight Agent instead | +| --- | --- | --- | --- | --- | +| draw — nothing urgent, develop | 494 | +0.08 ± 0.12 | — | +0.02 ± 0.04 | +| Freight Agent — feed the pipeline | 147 | +0.07 ± 0.06 | −0.01 ± 0.04 | — | +| switch — a train with work at the Office | 108 | — | −0.17 ± 0.10 | −0.25 ± 0.08 | +| draw — an Office upgrade in hand | 100 | +0.21 ± 0.16 (6) | — | −0.08 ± 0.06 | +| switch — walk the crew home | 47 | — | +0.22 ± 0.14 | −0.15 ± 0.08 | +| draw — a train card in hand | 11 | — | — | −0.45 ± 0.24 | + +No rule has an alternative that is significantly better; where the table leans, the ladder is usually +the one that is right. So given how the bot plays each option once chosen, the choice itself is close to +optimal, and value functions for draw and Freight Agent have little to find. The one lean worth a look if +this is reopened is walking a stranded crew home (+0.22, t ≈ 1.6). The rollout tool is analysis only — +the bot never sees a rollout. + +#### #106 — THE EXTRA TRAP — why the cap on committed trains still lets an Office overfill. + +**2026-09-15 — re-measured under today's defaults, and most of it is not the Extra trap.** 20 "no free +A/D track" collisions in 60 standard games, −1.67 revenue a game. No train was held (§8.2 or clearance) in +the Stage before any of them, and only 8 of the 20 trains destroyed were Extras. **Six destroyed a train +of the same number as a Second Section run within the previous two Stages — and all 26 Second Sections +the bot ran in those games were an accident:** the New Train phase's "no car on offer" fallback takes +`options[0]`, and `legalActions` lists `newTrain.secondSection` ahead of the Extra starts, so whenever an +Extra was waiting to start the bot doubled the train due out instead. The Office never had an A/D track +to spare for one. + +**A RULES QUESTION FOR JESSE, found on the way — not a bot matter.** Q9 (`implications.md`) defines the +Second Section as a CARD "played on a train that is due out", and `content.ts` defines `SECOND_SECTION` +with 1 copy — but `buildDeck` never deals it, and `check`'s `newTrain.secondSection` asks for no card in +hand. So any player may run a Second Section for free on every train due out. Either the card should be +dealt and required, or the free action is the intended rule and Q9's wording is stale. + +Also measured and removed: starting an over-cap Extra where its run never reaches the Office (+0.10, +t = 1.82) — such a start was on offer at 1 of 25 over-cap starts. + +**The accident is fixed** (2026-09-15, default): the New Train fallback takes a car, a pass or the +Extra's start, never `options[0]` — +0.32 ± 0.09 (t = 3.64), collisions 0.24 → 0.19. **Still open under +this item:** the forced Extra itself (a full, undiscardable hand of Extras), and the Second Section card +question above. + +`choose` removes train-card plays from the options when `trainWouldOverfillTheOffice`, but yields if +that would leave nothing legal. It does leave nothing legal in one ordinary position: the hand is over +the limit (§6.2 requires reducing it) and every card in it is an Extra, which `keepReason` forbids +discarding. Measured 2026-09-14 after the face-up take rule: 23 of 196 train plays in 40 games were past +the cap, every one "play what is in hand" with four Extras held, and the worst seeds each lost 3-4 +collisions to it. Declining the draw option in that position measured −0.03 ± 0.11 — collisions fell +0.26 → 0.20 but cards played fell 29.0 → 25.6. Better answers to try: play the Extra at the least +dangerous moment rather than the first, count WHEN each committed train is due at the Office instead of +how many there are, or keep the hand from filling with Extras in the first place. + +#### #105 — PLAN ACROSS TURNS — the evidence so far, for the conversation. + +For: one-turn planning already reaches most switching work (Cargo phases with a car spotted 5% → 18%), +and what it cannot do is exactly what spans a Stage — leave a car on a spur for the next crew, or start +a run-around and finish it later. The planner already scores staged wanted cars (+0.2), which is a +first, crude step in that direction. + +Against, for now: everything between two switching turns is not the player's — a Mainline Phase, trains +arriving, a Load/Unload phase — so a second turn cannot be searched the way the first is without either +simulating those phases (arrivals are on the public timetable, but cars on arriving trains are not +known) or scoring the position between turns more cleverly. And search cost is already what sets the +test suite's running time. Cheapest next step if taken up: a better score for "what the next turn can +still reach", not a deeper search. + +**2026-09-15 — the run-around measurement that bears on this.** A candidate that lays track by what the +district can do afterwards (`valueLays`) more than doubled closed run-arounds, 9/60 → 22/60, yet moved +revenue only +0.14 ± 0.06 (t = 2.58, 1600 seeds). Traced over 40 games: the one-turn planner DOES use +the loop — 63 of 131 plans in a district with one end a move on it, 35 run it both ways — but its planned +gain per turn is the same with a run-around as without (0.28 against 0.27). A run-around is for putting a +train's cars in a different order, which pays off in the turns after; a planner that looks one turn +ahead has no way to value it. + +**2026-09-15 — two-turn planning, built and measured: it does not pay.** `planTwoTurns` kept the six best +ends of a switching turn, removed the trains that would highball in the Mainline Phase in between (on the +Office square and made up — their cars leave with them), reset the Moves, planned the next turn from each, +and chose by the position after departures plus 0.8 of what the next turn adds. Nothing hidden is read. + +| version | revenue, 400 paired seeds | what went wrong | +| --- | --- | --- | +| first | −0.28 ± 0.11 (t = −2.58) | an expedited train left away drew 31 faults in one game: the fault is charged in the gap, which the discounted next turn "recovered"; trains left away 5.3 a game against 3.1 | +| with the gap fault charged in full and 0.5 a Stage per train left away | −0.14 ± 0.06 (t = −2.36) | trains still left away 4.8 a game; the "crew must get back to the Office" choice 4.2 a game against 2.7 | + +Why a second turn has so little to find, measured over 30 standard games: +- only **49%** of switching turns have the same train in the district at the next Local Operations choice; +- **6.1 wanted cars a game** do leave aboard departing trains — but a further switching turn from those + exact positions could have spotted only **0.7** of them: most were never deliverable; +- the one-turn planner already gains no more with a run-around than without (0.28 against 0.27). +And what a second turn COSTS is a Stage: trains left away have to be walked home, and those choices come +out of drawing and the Freight Agent (cards played 28.8 → 28.3). A multi-turn bot would have to weigh +switching against the other two options — which is #104, not a deeper search. + ### Code health and housekeeping #### #46 — tsc --noUnusedLocals finds 29 unused declarations across 14 files, and… @@ -1801,6 +2053,15 @@ where it belongs, and it is still open. Closed items, kept because several are the only record of a ruling or a lesson. Newest first within each group. +### Closed in the 2026-09-14 bot-tuning round (unreleased) + +53. ~~**The bot did not know to bring an expedited train back to the station.**~~ — done 2026-09-14, + not by a heuristic of its own but as a consequence of planning the switching turn: the planner's + score charges an expedited train left away from the Office a full Revenue point, which is what Q3 + charges. Over 400 paired seeds, 19 games drew expedite faults under the rule ladder and **none** + under the planner, worth +1.07 a game (t = 3.83) — the two worst cases had drawn 45 and 42 faults + in a single game. See `CHANGELOG.md`, 0.8.0.9. + ### Shipped through v0.7.9.8, from the queue Closed items, newest first. Kept because several of them are the only record of a ruling or a lesson; diff --git a/package.json b/package.json index 3687ae2..ab44cc7 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "station-master", - "version": "0.8.0.8", + "version": "0.8.0.9", "private": true, "type": "module", "description": "Station Master — a railroad operations game", @@ -10,7 +10,9 @@ "scripts": { "typecheck": "tsc --noEmit", "pretest": "tsc --noEmit && node scripts/build-web.ts", - "test": "node --test test/*.test.ts test/**/*.test.ts", + "test": "node --test $(ls test/*.test.ts test/**/*.test.ts | grep -v '^test/sim.test.ts$')", + "pretest:sim": "tsc --noEmit", + "test:sim": "node --test test/sim.test.ts", "build:web": "node scripts/build-web.ts", "serve:web": "node scripts/build-web.ts && npx --yes http-server dist -p 8080 -c-1", "deploy:web": "node scripts/deploy-web.ts", diff --git a/src/engine/advance.ts b/src/engine/advance.ts index 8a79f27..6bbc645 100644 --- a/src/engine/advance.ts +++ b/src/engine/advance.ts @@ -522,8 +522,11 @@ type MoveOutcome = 'moved' | 'held' | 'needsClearance'; * * This is reachable purely through switching. A train arrives made up, and only comes apart because * the player took cars onto the nose or picked up a cut in a run-around. + * + * Exported for the switching planner (`sim/switch-planner.ts`), which has to know whether a plan + * leaves a train unable to run — and must ask this rule rather than keep a copy of it. */ -function badlyMadeUp(tray: CrewTray): string | null { +export function badlyMadeUp(tray: CrewTray): string | null { const n = tray.consist.length; if (n === 0) return null; const pulling = tray.engineAt === 0; diff --git a/src/engine/apply.ts b/src/engine/apply.ts index 9e620ba..731dd17 100644 --- a/src/engine/apply.ts +++ b/src/engine/apply.ts @@ -1508,17 +1508,48 @@ export function selectDestination( return chosen ?? atTo[0]; } +/** + * ROUTES, WALKED ONCE PER POSITION. + * + * A route walk (`reachableDestinations`) was a third of all simulation time, and most of it was the + * same walk repeated: `legal.ts` walks a tray's routes to list its moves, then `check` walks them again + * for every one of those moves, and `applyIntent` walks the chosen one a third time in `execute`. + * Profiled 2026-09-14 with inlining off: `reachableDestinations` 34% inclusive, garbage collection 34%. + * + * So while one position is being examined — a legal-action listing, or the check and execute of one + * intent — a walk is kept and reused. Both scopes read the state and never write it, the key names + * everything the walk depends on besides that state, and the cache is keyed to the state OBJECT and + * cleared when the scope ends, so a hit returns exactly what a fresh walk would have. Nothing may + * mutate a returned route; nothing does. + */ +let routeCache: { state: GameState; routes: Map } | null = null; + +export function withRouteCache(s: GameState, fn: () => T): T { + if (routeCache) return fn(); + routeCache = { state: s, routes: new Map() }; + try { + return fn(); + } finally { + routeCache = null; + } +} + function destinationsFor( s: GameState, player: PlayerIndex, trayId: TrayId, from: GridCoord, reverse: boolean, -) { +): MoveDestination[] { + const cache = routeCache?.state === s ? routeCache.routes : null; + const key = cache ? `${player}|${trayId}|${from.row},${from.col}|${reverse ? 1 : 0}` : ''; + const hit = cache?.get(key); + if (hit) return hit; + const tray = s.trays.get(trayId)!; const facing = facingPort(s, trayId); const exit: Port = reverse ? reversePort(s, player, from, facing) : facing; - return reachableDestinations( + const routes = reachableDestinations( { area: areaOf(s, player), occupancy: occupancyFor(s, player, trayId), @@ -1528,6 +1559,8 @@ function destinationsFor( from, exit, ); + cache?.set(key, routes); + return routes; } /** @@ -2702,7 +2735,8 @@ export function reduce(s: GameState, e: GameEvent): void { * index is out of range, which `check` reports rather than silently defaulting — a wrong * orientation is a different card, not a detail. */ -function protoCard( +/** Exported for the same reason as `extendLimitsIfNeeded`: the bot builds the card a lay would place exactly as the reducer does. */ +export function protoCard( kind: { kind: string; geometry?: string; facility?: string; hand?: string }, variant: number | undefined, ): TrackCard | null { @@ -3093,7 +3127,8 @@ function applyModifier(area: OfficeArea, coord: GridCoord, modifier: ModifierKin * §8.1 and §10 both reason about "the track between the train and the Limits", Interlocking holds * an arrival AT the Limits, and running past a player's Limits is what makes a collision his fault. */ -function extendLimitsIfNeeded(area: OfficeArea, placed: GridCoord): void { +/** Exported so the bot can score a lay on a copy of the district by the engine's own rule, not a copy of it. */ +export function extendLimitsIfNeeded(area: OfficeArea, placed: GridCoord): void { if (placed.row !== area.runningRow) return; if (placed.col <= area.limitsWest.col) { @@ -3287,16 +3322,35 @@ function limitsCard(): TrackCard { // Public entry point // --------------------------------------------------------------------------- -export function applyIntent(s: GameState, player: PlayerIndex, i: Intent): ApplyResult { - const code = check(s, player, i); - if (code) return { ok: false, code, message: `${i.type} rejected: ${code}` }; +/** + * THE FIRST HALF OF `applyIntent`: decide, without changing anything. + * + * `check` and `execute` read the same unchanged position, so its routes are walked once between them + * (`withRouteCache`). Never writes `s`. Split out for a caller that decides many intents against ONE + * position and applies each to a COPY of it — the switching planner — which can then share that + * position's routes across every candidate instead of re-walking them on each copy. + */ +export function prepareIntent(s: GameState, player: PlayerIndex, i: Intent): ApplyResult { + const prepared = withRouteCache(s, (): { code: RejectionCode } | { events: GameEvent[] } => { + const code = check(s, player, i); + return code ? { code } : { events: execute(s, player, i) }; + }); + if ('code' in prepared) return { ok: false, code: prepared.code, message: `${i.type} rejected: ${prepared.code}` }; + return { ok: true, events: prepared.events }; +} - const events = execute(s, player, i); +/** THE SECOND HALF: fold events `prepareIntent` produced into a state equal to the one it read. */ +export function commitEvents(s: GameState, events: readonly GameEvent[]): void { for (const e of events) reduce(s, e); // Gitea#16 — the intent half of the fold; `advance` does the phase driver's half. See `tally.ts` // for why it cannot simply live inside `reduce`. for (const e of events) tallyEvent(s, e); - return { ok: true, events }; +} + +export function applyIntent(s: GameState, player: PlayerIndex, i: Intent): ApplyResult { + const r = prepareIntent(s, player, i); + if (r.ok) commitEvents(s, r.events); + return r; } export { isOperationalRail, destinationsFor }; diff --git a/src/engine/legal.ts b/src/engine/legal.ts index 99ef11a..b54edbd 100644 --- a/src/engine/legal.ts +++ b/src/engine/legal.ts @@ -14,7 +14,7 @@ import type { CarType, Hand, TrackGeometry } from './content.ts'; import { enhancementRule, mainlineProfile } from './content.ts'; -import { check, areaOf, destinationsFor } from './apply.ts'; +import { check, areaOf, destinationsFor, withRouteCache } from './apply.ts'; import type { Intent } from './intents.ts'; import type { GameState, GridCoord, PlayerIndex } from './state.ts'; import { coordKey, seatOf } from './state.ts'; @@ -53,7 +53,80 @@ const CAR_TYPES: readonly CarType[] = ['coach', 'boxcar', 'reefer', 'hopper', 't /** Every intent `player` may legally submit right now. */ export function legalActions(s: GameState, player: PlayerIndex): Intent[] { - return candidates(s, player).filter((i) => check(s, player, i) === null); + // One position, examined many times over: its routes are walked once (`withRouteCache`). + return withRouteCache(s, () => candidates(s, player).filter((i) => check(s, player, i) === null)); +} + +/** + * §6.1 — the switching half of the Local Operations candidates, in the order `legalActions` offers + * them. Split out so the switching planner (`sim/switch-planner.ts`) can ask for just these without + * `check` running over every draw and Freight Agent candidate at each of the thousands of positions it + * tries — that was about a quarter of all planning time. Still no rules here: `check` decides. + */ +function switchCandidates(s: GameState, player: PlayerIndex): Intent[] { + const out: Intent[] = []; + for (const [trayId, tray] of s.trays) { + if (tray.position.at !== 'grid' || tray.position.seat !== seatOf(s, player)) continue; + const from = tray.position.coord; + for (const reverse of [false, true]) { + const dests = destinationsFor(s, player, trayId, from, reverse); + // Grouped by destination square so `distinguishingVia` only ever compares routes that are + // actually racing for the same button — two routes to DIFFERENT squares need no `via` to + // tell apart, `to` already does that. + const byCoord = new Map(); + for (const d of dests) { + const k = coordKey(d.coord); + (byCoord.get(k) ?? byCoord.set(k, []).get(k)!).push(d); + } + for (const group of byCoord.values()) { + for (const d of group) { + const via = group.length > 1 ? distinguishingVia(d, group) : undefined; + out.push({ type: 'switch.move', trayId, to: d.coord, reverse, ...(via ? { via } : {}) }); + } + } + } + for (let n = 1; n <= tray.consist.length; n++) { + out.push({ type: 'switch.dropCars', trayId, count: n }); + // Off the nose as well as the tail — the only way to get cars back off the front of a train + // that shoved a cut, and therefore the only way an engine buried mid-train reaches an end. + out.push({ type: 'switch.dropCars', trayId, count: n, fromNose: true }); + } + // Small Yard: enumerating every permutation would explode, so offer the useful ones — + // bringing each car to the droppable end, plus a full reversal. `check` validates any order, + // so a UI may submit an arbitrary permutation. + const n = tray.consist.length; + if (n > 1) { + for (let k = 0; k < n; k++) { + const order = [...Array(n).keys()].filter((x) => x !== k); + order.push(k); + out.push({ type: 'switch.sortConsist', trayId, order }); + } + out.push({ type: 'switch.sortConsist', trayId, order: [...Array(n).keys()].reverse() }); + } + } + // Flying Switch — roll a cut into an ADJACENT industry without the engine entering it. + for (const cardId of s.decks.hands.get(player) ?? []) { + const k = s.cards.get(cardId)?.kind; + if (k?.kind !== 'maneuver' || k.key !== 'flyingSwitch') continue; + for (const [trayId, tray] of s.trays) { + if (tray.position.at !== 'grid' || tray.position.seat !== seatOf(s, player)) continue; + const from = tray.position.coord; + for (const reverse of [false, true]) { + for (const d of destinationsFor(s, player, trayId, from, reverse)) { + for (let count = 1; count <= tray.consist.length; count++) { + out.push({ type: 'maneuver.flyingSwitch', cardId, trayId, count, to: d.coord }); + } + } + } + } + } + out.push({ type: 'switch.end' }); + return out; +} + +/** The switching intents `player` may legally submit right now — exactly `legalActions`' switching subset. */ +export function legalSwitchingActions(s: GameState, player: PlayerIndex): Intent[] { + return withRouteCache(s, () => switchCandidates(s, player).filter((i) => check(s, player, i) === null)); } export function isLegal(s: GameState, player: PlayerIndex, i: Intent): boolean { @@ -138,62 +211,7 @@ function localOpsCandidates(s: GameState, player: PlayerIndex): Intent[] { const area = areaOf(s, player); // -- switch (§6.1) - for (const [trayId, tray] of s.trays) { - if (tray.position.at !== 'grid' || tray.position.seat !== seatOf(s, player)) continue; - const from = tray.position.coord; - for (const reverse of [false, true]) { - const dests = destinationsFor(s, player, trayId, from, reverse); - // Grouped by destination square so `distinguishingVia` only ever compares routes that are - // actually racing for the same button — two routes to DIFFERENT squares need no `via` to - // tell apart, `to` already does that. - const byCoord = new Map(); - for (const d of dests) { - const k = coordKey(d.coord); - (byCoord.get(k) ?? byCoord.set(k, []).get(k)!).push(d); - } - for (const group of byCoord.values()) { - for (const d of group) { - const via = group.length > 1 ? distinguishingVia(d, group) : undefined; - out.push({ type: 'switch.move', trayId, to: d.coord, reverse, ...(via ? { via } : {}) }); - } - } - } - for (let n = 1; n <= tray.consist.length; n++) { - out.push({ type: 'switch.dropCars', trayId, count: n }); - // Off the nose as well as the tail — the only way to get cars back off the front of a train - // that shoved a cut, and therefore the only way an engine buried mid-train reaches an end. - out.push({ type: 'switch.dropCars', trayId, count: n, fromNose: true }); - } - // Small Yard: enumerating every permutation would explode, so offer the useful ones — - // bringing each car to the droppable end, plus a full reversal. `check` validates any order, - // so a UI may submit an arbitrary permutation. - const n = tray.consist.length; - if (n > 1) { - for (let k = 0; k < n; k++) { - const order = [...Array(n).keys()].filter((x) => x !== k); - order.push(k); - out.push({ type: 'switch.sortConsist', trayId, order }); - } - out.push({ type: 'switch.sortConsist', trayId, order: [...Array(n).keys()].reverse() }); - } - } - // Flying Switch — roll a cut into an ADJACENT industry without the engine entering it. - for (const cardId of s.decks.hands.get(player) ?? []) { - const k = s.cards.get(cardId)?.kind; - if (k?.kind !== 'maneuver' || k.key !== 'flyingSwitch') continue; - for (const [trayId, tray] of s.trays) { - if (tray.position.at !== 'grid' || tray.position.seat !== seatOf(s, player)) continue; - const from = tray.position.coord; - for (const reverse of [false, true]) { - for (const d of destinationsFor(s, player, trayId, from, reverse)) { - for (let count = 1; count <= tray.consist.length; count++) { - out.push({ type: 'maneuver.flyingSwitch', cardId, trayId, count, to: d.coord }); - } - } - } - } - } - out.push({ type: 'switch.end' }); + out.push(...switchCandidates(s, player)); // -- draw (§6.2) out.push({ type: 'draw.fromHomeOffice' }); diff --git a/src/engine/track.ts b/src/engine/track.ts index 1d57b12..97f0a2a 100644 --- a/src/engine/track.ts +++ b/src/engine/track.ts @@ -377,7 +377,7 @@ export function reachableDestinations( start: GridCoord, initialExit: Port, ): MoveDestination[] { - return exploreMoves(ctx, start, initialExit).destinations; + return exploreMoves(ctx, start, initialExit, false).destinations; } /** @@ -434,12 +434,19 @@ export function exploreMoves( ctx: MoveContext, start: GridCoord, initialExit: Port, + /** + * False when only the destinations are wanted (`reachableDestinations`, every legality check): the + * rejections are then not recorded at all. They never change a destination, and building them was + * pure allocation on the hottest path in the engine. + */ + collectBlocks = true, ): { destinations: MoveDestination[]; blocked: MoveBlock[] } { const { area, occupancy } = ctx; const results: MoveDestination[] = []; const blocked: MoveBlock[] = []; const noted = new Set(); const block = (coord: GridCoord, kind: MoveBlockKind, why: string): void => { + if (!collectBlocks) return; const k = coordKey(coord); if (noted.has(k)) return; noted.add(k); @@ -459,10 +466,25 @@ export function exploreMoves( couples: RollingStock[]; /** Coord key each entry in `couples` came off, aligned by index — see `routeOutcomeKey`. */ origins: string[]; - /** Cards visited on THIS route, start included. A per-path set, not a global one — see the - * module doc comment on `MAX_ENUMERATED_FRONTIER` for why a global one would forbid the very - * routes this walk exists to find. */ - visited: Set; + }; + + /** + * Has THIS route already used `to`? Per-path, not global — see the doc comment on + * `MAX_ENUMERATED_FRONTIER` for why a global set would forbid the very routes this walk exists to + * find. + * + * Read off the route's own `path` instead of a Set copied at every step, which was a large share of + * the walk's garbage. It answers exactly as that Set did: the start square, then every square + * enqueued along the route AFTER the first hop, including this node's own — the first hop's square + * was never added, and `path[0]` is that square, so the scan begins at 1. + */ + const onRoute = (node: Frontier, to: GridCoord): boolean => { + if (sameCoord(to, start)) return true; + if (node.path.length > 0 && sameCoord(to, node.coord)) return true; + for (let k = 1; k < node.path.length; k++) { + if (sameCoord(node.path[k]!.coord, to)) return true; + } + return false; }; // The very first hop is checked here because `start`'s card is not itself enqueued; every later @@ -493,13 +515,14 @@ export function exploreMoves( path: [], couples: ownCut, origins: ownCut.map(() => startKey), - visited: new Set([startKey]), }, ]; let enumerated = 1; - while (queue.length > 0) { - const node = queue.shift()!; + // FIFO by index rather than `shift()`, which re-packs the array on every pop. Same order. + let head = 0; + while (head < queue.length) { + const node = queue[head++]!; const card = cardAt(area, node.coord); if (!card) continue; @@ -592,17 +615,15 @@ export function exploreMoves( // direction; it never says without repeating ground, but a train cannot occupy the same // track twice at once either). Per-path, not global — a DIFFERENT route may legitimately // pass through a card this one already used. - const toKey = coordKey(to); - if (node.visited.has(toKey)) continue; + if (onRoute(node, to)) continue; if (enumerated >= MAX_ENUMERATED_FRONTIER) break; enumerated++; const step: MoveStep = { coord: node.coord, entry: node.entry, exit }; - const visited = new Set(node.visited); - visited.add(toKey); - queue.push({ coord: to, entry: opposite(exit), path: [...node.path, step], couples, origins, visited }); + queue.push({ coord: to, entry: opposite(exit), path: [...node.path, step], couples, origins }); } } + if (!collectBlocks) return { destinations: results, blocked }; // A card that turned out to be reachable after all is not a blocker: the walk may meet a square // from a bad angle first and a good one later. const reached = new Set(results.map((r) => coordKey(r.coord))); diff --git a/src/sim/bot.ts b/src/sim/bot.ts index e097834..d84f533 100644 --- a/src/sim/bot.ts +++ b/src/sim/bot.ts @@ -22,6 +22,9 @@ import { applyIntent, areaOf, + extendLimitsIfNeeded, + isLockedOut, + protoCard, canAdvanceLoad, destinationsFor, facilityCarTypes, @@ -29,14 +32,15 @@ import { ownCutFor, } from '../engine/apply.ts'; import { MAX_CONSIST, nextOfficeTier, officeProfile } from '../engine/content.ts'; -import type { CarType, Hand, TrackGeometry } from '../engine/content.ts'; +import type { CarType, FreightKind, Hand, TrackGeometry } from '../engine/content.ts'; import type { GameEvent } from '../engine/events.ts'; import type { Intent } from '../engine/intents.ts'; import { legalActions } from '../engine/legal.ts'; -import { connectionsFor, exitsFrom, facilityVariants, hasPort, joins, neighbour, opposite, variantsFor } from '../engine/track.ts'; +import { canPlaceAt, connectionsFor, exitsFrom, facilityVariants, hasPort, joins, neighbour, opposite, variantsFor } from '../engine/track.ts'; import type { Port } from '../engine/track.ts'; import { actingPlayer, coordKey, turnOf } from '../engine/state.ts'; import type { Facility, GameState, GridCoord, OfficeArea, PlayerIndex, RollingStock, TrackCard } from '../engine/state.ts'; +import { planSwitchingTurn, switchFingerprint } from './switch-planner.ts'; export type BotPolicy = { name: string; @@ -106,7 +110,8 @@ function because(reason: string, intent: Intent): Intent { * Every flag here turns something OFF. That is the opposite of how this started — the tweaks were * candidates to switch on — and it is the right shape once a candidate has been adopted: what a * measured heuristic needs afterwards is a way to ask "is this still worth it?" when the deck or - * the rules move under it. Both of these were worth about +1.5 revenue together when adopted; if a + * the rules move under it. The first two were worth about +1.5 revenue together when adopted, and + * planning the switching turn (`noPlanSwitching`) +2.89 on its own; if a * rebalance changes the economy, that is a claim to re-test rather than to assume. * * The candidates that did NOT survive are gone rather than left switched off: preferring coaches at @@ -121,9 +126,78 @@ export type BotTweaks = { noTrainCap?: boolean; /** Draw whenever nothing is urgent, as the bot did before it preferred operating. */ noOperateFirst?: boolean; - + /** + * Choose switching Moves one at a time from the rule ladder, as the bot did before it planned the + * whole turn (`switch-planner.ts`). Measured at adoption, 2026-09-14: planning was worth + * +2.89 ± 0.18 revenue a game (t = 15.79) over 1600 paired seeds, 733 better against 21 worse. + */ + noPlanSwitching?: boolean; + /** + * Take a face-up train or industry card whether or not it could be played, as the bot did before + * 2026-09-14. It then took 20.1 trains and 11.9 industries a game off the Departments and discarded + * 20.2 and 11.8, retaking the same card 28.8 times a game. Asking first measured +1.52 ± 0.10 + * (t = 15.59) over 1600 paired seeds — this ablation was worse on 880 of them and better on 211. + */ + noPlayableTakes?: boolean; + /** + * Choose where track goes by `bestTrackLay`'s piece rules and the fallback's first legal square, as the + * bot did before 2026-09-15, instead of by what the district can DO afterwards (`bestValuedLay`). + * Scoring the layout measured +0.118 ± 0.029 (t = 4.14) over 6400 paired seeds, and closed run-arounds + * in 22 of 60 districts against 9. + */ + noValueLays?: boolean; + /** + * Let the New Train phase's fallback take `options[0]`, as the bot did before 2026-09-15. Because + * `legalActions` lists `newTrain.secondSection` ahead of the Extra starts, all 26 Second Sections the + * bot ran in 60 games were that accident, and 6 of the 20 collisions followed one. Taking a car, a pass + * or the Extra's start instead measured +0.32 ± 0.09 (t = 3.64) over 400 paired seeds. + */ + noDeliberateNewTrain?: boolean; }; +/** + * A switching turn planned once and then played a step per decision. + * + * Keyed by the tweaks object, because that is what one policy owns — the server shares a single + * `developerBot` across every bot seat, so the plan inside it is kept per player. Each step is + * submitted only while the position still matches the fingerprint the plan expected there; anything + * else replans. A switching turn has no randomness, so in practice a plan is made once a turn. + */ +type ActivePlan = { steps: Intent[]; keys: string[]; next: number; summary: string }; +const activePlans = new WeakMap>(); + +function plannedSwitch(s: GameState, player: PlayerIndex, options: Intent[], tweaks: BotTweaks): Intent | null { + let mine = activePlans.get(tweaks); + if (!mine) activePlans.set(tweaks, (mine = new Map())); + const here = switchFingerprint(s, player); + let active = mine.get(player); + if (!active || active.keys[active.next] !== here) { + const p = planSwitchingTurn(s, player); + active = { + steps: p.steps, + keys: p.keys, + next: 0, + summary: + `position ${p.rootScore.toFixed(2)} → ${p.score.toFixed(2)} over ${p.expanded} positions` + + (p.complete ? '' : ', search budget reached'), + }; + mine.set(player, active); + } + if (active.next >= active.steps.length) { + mine.delete(player); + const end = options.find((i) => i.type === 'switch.end'); + return end ? because(`planned switching turn complete — ${active.summary}`, end) : null; + } + const want = JSON.stringify(active.steps[active.next]); + const match = options.find((i) => JSON.stringify(i) === want); + if (!match) { + mine.delete(player); + return null; + } + active.next++; + return because(`step ${active.next} of ${active.steps.length} of a planned switching turn — ${active.summary}`, match); +} + /** The bot as it plays today. Every knob off. */ export const developerBot: BotPolicy = makeDeveloperBot({}); @@ -209,6 +283,13 @@ export function makeDeveloperBot(tweaks: BotTweaks): BotPolicy { ); if (match) return because(`the ${w.loaded ? 'loaded' : 'empty'} ${w.type} is what a facility is short of`, match); } + if (!tweaks.noDeliberateNewTrain) { + const move = + pickFirst(options, 'newTrain.placeCar', 'newTrain.passCar', 'newTrain.startExtra') ?? + options.find((i) => i.type !== 'newTrain.secondSection' && i.type !== 'maneuver.redFlags') ?? + options[0]!; + return because('no car on offer is one our facilities need', move); + } return because('no car on offer is one our facilities need', pickFirst(options, 'newTrain.placeCar', 'newTrain.passCar') ?? options[0]!); } @@ -438,7 +519,7 @@ function topOfDepartment(s: GameState, slot: number): string | undefined { * In a competitive game the same call reads the other way round — burying a card a rival wants is an * attack — which is why the choice belongs to the discarding player and not to the rules. */ -function bestDiscard(s: GameState, player: PlayerIndex, options: Intent[]): Intent | null { +function bestDiscard(s: GameState, player: PlayerIndex, options: Intent[], tweaks: BotTweaks): Intent | null { let best: Intent | null = null; let bestScore = -Infinity; for (const i of options) { @@ -448,7 +529,7 @@ function bestDiscard(s: GameState, player: PlayerIndex, options: Intent[]): Inte // two showing whatever they happened to start with. Measured over 100 games — spreading 2.87 // revenue, concentrating on the deepest 2.67, indifferent 2.67. const top = topOfDepartment(s, i.toSlot); - const wanted = isWorthTaking(s, player, i.toSlot); + const wanted = isWorthTaking(s, player, i.toSlot, tweaks); const depth = s.decks.departments[i.toSlot]?.length ?? 0; const score = (top === undefined ? 6 : wanted ? -10 : 2) - Math.min(depth, 4) * 0.5; if (score > bestScore) { @@ -460,10 +541,39 @@ function bestDiscard(s: GameState, player: PlayerIndex, options: Intent[]): Inte } /** A face-up card worth spending the draw on rather than gambling on the deck. */ -function isWorthTaking(s: GameState, player: PlayerIndex, slot: number): boolean { - return takingRank(s, player, slot) > 0; +function isWorthTaking(s: GameState, player: PlayerIndex, slot: number, tweaks: BotTweaks): boolean { + return takingRank(s, player, slot, tweaks) > 0; } +/** + * Could an industry of this kind be laid anywhere right now? Asked of the engine's own placement + * rule (`canPlaceAt`) and lockout (`isLockedOut`) rather than a copy: an industry is plain east-west + * track, so the only squares worth asking about are empty ones east or west of a card already down. + */ +function industrySiteExists(s: GameState, player: PlayerIndex, kind: FreightKind): boolean { + const area = areaOf(s, player); + if (isLockedOut(area, kind)) return false; + const probe = { + geometry: { kind: 'facility', facility: kind }, + baseOperationalRail: true, + standing: [], + standingWest: 0, + facility: null, + modifiers: [], + enhancements: [], + } as unknown as TrackCard; + for (const key of area.grid.keys()) { + const [row, col] = key.split(',').map(Number); + for (const dc of [1, -1]) { + const at = { row: row!, col: col! + dc }; + if (at.row === area.runningRow || area.grid.has(coordKey(at))) continue; + if (canPlaceAt(area, at, probe)) return true; + } + } + return false; +} + + /** * HOW BADLY the face-up card is wanted. 0 means not worth the draw. * @@ -471,14 +581,18 @@ function isWorthTaking(s: GameState, player: PlayerIndex, slot: number): boolean * happened to be scanned first — a coin flip on the card that decides whether the district ever * becomes a Passenger Facility at all. */ -function takingRank(s: GameState, player: PlayerIndex, slot: number): number { +function takingRank(s: GameState, player: PlayerIndex, slot: number, tweaks: BotTweaks): number { const id = topOfDepartment(s, slot); if (!id) return 0; const k = s.cards.get(id)?.kind; if (!k) return 0; if (k.kind === 'office') return nextOfficeTier(areaOf(s, player).tier) === k.tier ? 3 : 0; - if (k.kind === 'timetabledTrain' || k.kind === 'extraTrain') return 2; - if (k.kind === 'freightFacility') return 1; + if (k.kind === 'timetabledTrain' || k.kind === 'extraTrain') { + return !tweaks.noPlayableTakes && trainWouldOverfillTheOffice(s, player, tweaks) ? 0 : 2; + } + if (k.kind === 'freightFacility') { + return !tweaks.noPlayableTakes && !industrySiteExists(s, player, k.facility) ? 0 : 1; + } return 0; } @@ -741,6 +855,104 @@ function bestFacilityPlay(s: GameState, player: PlayerIndex, options: Intent[]): return best; } +/** + * WHAT A DISTRICT'S TRACK IS WORTH FOR WHAT IT LETS HAPPEN NEXT — the default since 2026-09-15; + * `noValueLays` turns it off. + * + * `bestTrackLay` scores the PIECE — its shape and where it sits — and cannot tell one that opens an + * industry site or closes a run-around from one that merely fills a square. This scores the LAYOUT the + * piece would leave, so a lay is worth the difference it makes. Every term is something the rules turn + * into play: a site is somewhere a held industry can go; a run-around lets a crew pass its own cars + * (§A.5); a way off the main is the only road to either; a Running Track straight is what Interlocking + * needs. The weights are a starting point to measure, not a result. + */ +function layoutValue(area: OfficeArea): number { + let v = 0; + const reachable = reachableOffMain(area); + v += Math.min(reachable.size, 12) * 0.2; + + let ways = 0; + let loops = 0; + for (const side of SIDES) { + for (const col of waysOff(area, side)) { + ways++; + if (descendFrom(area, col, side).rejoins.size > 0) loops++; + } + } + v += [0, 1.5, 2, 2.5][Math.min(ways, 3)]!; + if (loops > 0) v += 6 + Math.min(loops - 1, 1) * 2; + + // Squares an industry could legally be laid on, joined to track a crew can reach. + const probe = protoCard({ kind: 'freightFacility', facility: 'mineTipple' }, 0)!; + const tried = new Set(); + let sites = 0; + for (const key of reachable) { + const [row, col] = key.split(',').map(Number); + for (const dc of [1, -1]) { + const at = { row: row!, col: col! + dc }; + const k = coordKey(at); + if (tried.has(k) || area.grid.has(k) || at.row === area.runningRow) continue; + tried.add(k); + if (canPlaceAt(area, at, probe)) sites++; + } + } + v += [0, 2, 3, 3.5][Math.min(sites, 3)]!; + + let mainStraight = false; + for (const [key, card] of area.grid) { + if (Number(key.split(',')[0]) !== area.runningRow || card.geometry.kind !== 'track') continue; + if (card.geometry.geometry === 'straight') mainStraight = true; + // A turnout on the main whose leg joins nothing is a hole in the Running Track with no road behind it. + if (card.geometry.geometry === 'turnout') { + const col = Number(key.split(',')[1]); + for (const side of SIDES) { + if (!hasPort(card, legPort(side))) continue; + const beyond = area.grid.get(`${area.runningRow + side},${col}`); + if (!beyond || !joins(card, legPort(side), beyond)) v -= 0.5; + } + } + } + if (mainStraight) v += 1.5; + return v; +} + +/** + * The track lay worth most by `layoutValue`, placed on a copy of the district exactly as the reducer + * places it (`protoCard`, `extendLimitsIfNeeded`). With `mustBuild`, only a lay that gains something is + * offered, which is the slot `bestTrackLay` fills; without it, the best of whatever is legal, which is + * the slot the "play what is in hand" fallback fills. Ties go to the square nearer the Office. + */ +function bestValuedLay(s: GameState, player: PlayerIndex, options: Intent[], mustBuild: boolean): Intent | null { + const area = areaOf(s, player); + const base = layoutValue(area); + let best: Intent | null = null; + let bestScore = -Infinity; + for (const i of options) { + if (i.type !== 'card.play' || i.placement === undefined) continue; + const kind = s.cards.get(i.cardId)?.kind; + if (kind?.kind !== 'track') continue; + const built = protoCard(kind, i.variant); + if (!built) continue; + const after: OfficeArea = { + ...area, + grid: new Map(area.grid), + limitsWest: { ...area.limitsWest }, + limitsEast: { ...area.limitsEast }, + }; + after.grid.set(coordKey(i.placement), built); + extendLimitsIfNeeded(after, i.placement); + const gain = layoutValue(after) - base; + if (mustBuild && gain <= 0.1) continue; + const distance = Math.abs(i.placement.row - area.officeCoord.row) * 2 + Math.abs(i.placement.col - area.officeCoord.col); + const score = gain - distance * 0.01; + if (score > bestScore) { + bestScore = score; + best = i; + } + } + return best; +} + function bestTrackLay(s: GameState, player: PlayerIndex, options: Intent[]): Intent | null { const area = areaOf(s, player); @@ -1213,7 +1425,7 @@ function followThrough( if (!turnOf(s, player).drawnThisTurn) { const piles = options.filter( (i): i is Extract => - i.type === 'draw.fromDepartment' && isWorthTaking(s, player, i.slot), + i.type === 'draw.fromDepartment' && isWorthTaking(s, player, i.slot, tweaks), ); // Best-ranked pile rather than the first that qualifies: an Office card and a train card // both "qualify", and only one of them stops the collisions. @@ -1221,7 +1433,7 @@ function followThrough( // and a train card are face up together 1.6 decisions a game — but ranking them is what the // ranking function is for, and a coin flip on the card that decides whether the district // ever becomes a Passenger Facility is not worth keeping for its own sake. - const useful = piles.sort((a, b) => takingRank(s, player, b.slot) - takingRank(s, player, a.slot))[0]; + const useful = piles.sort((a, b) => takingRank(s, player, b.slot, tweaks) - takingRank(s, player, a.slot, tweaks))[0]; if (useful) return because('a face-up card is worth more than a blind draw right now', useful); const blind = options.find((i) => i.type === 'draw.fromHomeOffice'); if (blind) return because('no face-up card is worth taking — gamble on the deck', blind); @@ -1282,7 +1494,7 @@ function followThrough( // STRAIGHTS that Enhancements require, and no Freight Facility has anywhere to go until a // district exists. Measured with track absent, the hand held a playable Enhancement on 4,778 // turns and could legally place one on 33. - const track = bestTrackLay(s, player, options); + const track = !tweaks.noValueLays ? bestValuedLay(s, player, options, true) : bestTrackLay(s, player, options); if (track) return because('lay track — nothing else creates the straights Enhancements need or the spurs freight needs', track); // Then real development: a card actually laid into the grid. Freight facilities are scored — @@ -1307,17 +1519,27 @@ function followThrough( s.cards.get(i.cardId)?.kind.kind !== 'track', ); if (placed) return because('develop the district with a card that goes on the board', placed); - const play = options.find((i) => i.type === 'card.play'); + const play = !tweaks.noValueLays + ? options.find((i) => i.type === 'card.play' && s.cards.get(i.cardId)?.kind.kind !== 'track') ?? + bestValuedLay(s, player, options, false) + : options.find((i) => i.type === 'card.play'); if (play) return because('play what is in hand', play); const end = options.find((i) => i.type === 'draw.end'); if (end) return because('nothing in hand can be played anywhere legal', end); return because( 'nothing playable — discard onto the Department whose face-up card is least worth keeping reachable', - bestDiscard(s, player, options) ?? pickFirst(options, 'card.discard') ?? options[0]!, + bestDiscard(s, player, options, tweaks) ?? pickFirst(options, 'card.discard') ?? options[0]!, ); } case 'switch': { + // The planner decides the whole turn; the rules below are its fallback if the position is ever + // not the one it planned for, and the whole of switching under `noPlanSwitching`. + if (!tweaks.noPlanSwitching) { + const planned = plannedSwitch(s, player, options, tweaks); + if (planned) return planned; + } + /** * A MOVE THAT DRAGS THE CREW'S OWN CUT BACK ON IS A WASTED MOVE, so take those off the table * before any heuristic gets to choose one. diff --git a/src/sim/compare.ts b/src/sim/compare.ts index 84e6a22..291e97a 100644 --- a/src/sim/compare.ts +++ b/src/sim/compare.ts @@ -23,6 +23,7 @@ * Run with: * node src/sim/compare.ts 1600 noTrainCap=1 — what the A/D cap is worth today * node src/sim/compare.ts 1600 noOperateFirst=1 — what operating before drawing is worth + * node src/sim/compare.ts 1600 noPlanSwitching=1 — what planning the switching turn is worth * * The flags are ABLATIONS: they turn off heuristics the bot already plays, so a negative delta is * the heuristic earning its place. That is what a measured bot needs going forward — the question @@ -221,7 +222,7 @@ export function formatPaired(r: PairedResult): string { * against itself and report a confident zero, which is the most expensive way this tool could fail. */ export const NUMERIC_TWEAKS = new Set([]); -export const BOOLEAN_TWEAKS = new Set(['noTrainCap', 'noOperateFirst']); +export const BOOLEAN_TWEAKS = new Set(['noTrainCap', 'noOperateFirst', 'noPlanSwitching', 'noPlayableTakes', 'noValueLays', 'noDeliberateNewTrain']); export function parseTweaks(args: string[]): BotTweaks { const tweaks: Record = {}; diff --git a/src/sim/switch-planner.ts b/src/sim/switch-planner.ts new file mode 100644 index 0000000..448f17d --- /dev/null +++ b/src/sim/switch-planner.ts @@ -0,0 +1,347 @@ +/** + * Component 17b — planning a whole switching turn before making the first Move. + * + * Dev-side, like the rest of the bot. The developer bot's switching branch chooses ONE move at a time + * from a ladder of rules, and its own comment names what that cannot do: "a strong player would use + * the six Moves to re-order the consist — that is the game's central switching puzzle, and this bot + * does not attempt it." This attempts it, for one turn at a time. + * + * WHY SEARCH IS FAIR HERE. A switching turn draws no card and rolls no die, so trying sequences on a + * copy of the game is exactly what a player does by looking at the board. The score below reads only + * what a player can see — the district, the cars on the trains, the facilities — and never the deck. + * + * WHY NOT EVERY SEQUENCE. Measured 2026-09-14 over 30 switching turns from bot games: a median turn + * reaches 229 distinct positions, but 11 of 30 passed 20,000, because setting cars out is free and a + * crew can leave them in a great many places. So the search keeps the best `beam` positions at each + * step and stops at `budget` positions tried. Small turns are searched completely inside that. + * + * THE SCORE IS OF WHERE THE TURN ENDS, not of what it did, and it starts from Jesse's ruling + * (2026-09-14): "players will attempt to deliver / pick up cars even if it delays trains." So a car + * put where it can be worked is worth a point, and a train left away from the Office costs a quarter + * of one. The weights are a starting point to measure, not a result. + */ + +import { areaAtSeat, areaOf, commitEvents, facilityCarTypes, prepareIntent, withRouteCache } from '../engine/apply.ts'; +import { badlyMadeUp, isExpedited } from '../engine/advance.ts'; +import { MAX_CONSIST } from '../engine/content.ts'; +import type { Intent } from '../engine/intents.ts'; +import { legalSwitchingActions } from '../engine/legal.ts'; +import { cloneTally, coordKey, seatOf, turnOf } from '../engine/state.ts'; +import type { Facility, GameState, GridCoord, PlayerIndex, RollingStock, TrackCard } from '../engine/state.ts'; + +export const SWITCH_WEIGHTS = { + /** A car standing where its industry can load or unload it — the point of switching. */ + spot: 1.0, + /** The same, past what the industry's boxes can work at once. */ + spotBeyondCapacity: 0.25, + /** A car the industry cannot work, taking room on its track. */ + junkOnIndustry: -0.5, + /** A finished car — loaded at a shipper, emptied at a receiver — still waiting to be lifted. */ + finishedLeft: -0.15, + /** A car on one of this district's trains that some industry here would work. */ + carriedWanted: 0.35, + /** The same car left on ordinary track, where a later turn can fetch it. */ + stagedWanted: 0.2, + /** A coach kept with its train, or parked at the Office where §A.4 allows it. */ + coachWithTrain: 0.3, + /** A coach left anywhere else, where no Porter can work it. */ + coachStranded: -0.3, + /** Anything but a coach standing on the Office square — the next arrival collides (§8.3). */ + fouling: -3, + /** A train that ends the turn away from the Office and so cannot highball next Mainline Phase. */ + trainAway: -0.25, + /** On top of that, an expedited train — Q3 charges a Revenue point every Phase it is away. */ + expeditedAway: -1.0, + /** A train that could not leave even from the Office — engine buried, caboose mid-train (§8.2). */ + notMadeUp: -0.6, + /** Tie-breaks, so equal outcomes prefer the plan that does less. */ + perMove: -0.02, + perSetOut: -0.005, + /** A maneuver card spent — Flying Switch — so the planner plays one only when it buys something. */ + cardSpent: -0.1, +} as const; + +export type PlanOptions = { + budget: number; + beam: number; + /** + * Search Flying Switch alongside Moves, set-outs and sorts. On by default but UNMEASURED: the card + * is dealt 0 copies (Jesse, 2026-08-26), so over 400 paired seeds turning it on changed nothing — + * it is here so the planner can use the card the day it is dealt again. + */ + flyingSwitch?: boolean; +}; +/** + * Measured 2026-09-14, paired over 400 seeds against 3000/48: 2000/32 cost −0.02 ± 0.01 (t = −1.68, + * inside the noise) at half the time per turn; 1000/24 cost −0.06 ± 0.02 (t = −2.65) for little more. + */ +export const DEFAULT_PLAN: PlanOptions = { budget: 2000, beam: 32, flyingSwitch: true }; + +export type SwitchPlan = { + /** The intents to submit, in order. Empty when nothing beats stopping where the crew stands. */ + steps: Intent[]; + /** `switchFingerprint` before each step, and after the last — so a caller can tell it is on plan. */ + keys: string[]; + rootScore: number; + score: number; + /** Positions tried. */ + expanded: number; + /** False when the budget ran out before the search did. */ + complete: boolean; +}; + +/** + * A copy of the game that a switching intent can be applied to without touching the original. + * + * NOT `structuredClone`, of the state or even of the district. A switching intent writes only the + * cars standing on cards, the industry tracks, the district's A/D and held lists, the consist and + * position of the trays standing in it, this player's turn and the tally — so exactly those arrays are + * copied and everything else is shared by reference. Measured 2026-09-14, deep-cloning the district + * was 44% of all planning time. + * + * `test/switch-planner.test.ts` proves across real games that planning leaves the original + * byte-identical — which is what fails first if a reducer ever starts writing somewhere new, or + * starts mutating a car or a card in place instead of replacing it. + */ +export function forkForSwitching(s: GameState, player: PlayerIndex): GameState { + const seat = seatOf(s, player); + const area = areaAtSeat(s, seat); + const grid = new Map(); + for (const [key, card] of area.grid) { + const f = card.facility; + grid.set(key, { + ...card, + standing: [...card.standing], + facility: f ? { ...f, industryTrack: { cars: [...f.industryTrack.cars] } } : f, + }); + } + const officeAreas = new Map(s.officeAreas); + officeAreas.set(seat, { + ...area, + grid, + adOccupancy: [...area.adOccupancy], + heldAtLimits: [...area.heldAtLimits], + dispatchUsedToday: [...area.dispatchUsedToday], + }); + const trays = new Map(s.trays); + for (const [id, t] of s.trays) { + if (t.position.at === 'grid' && t.position.seat === seat) trays.set(id, { ...t, consist: [...t.consist] }); + } + const turns = new Map(s.turns); + const turn = s.turns.get(player)!; + turns.set(player, { ...turn, freightWorked: { ...turn.freightWorked } }); + // Flying Switch spends its card (`spendCard`): the hand map is rewritten and the Salvage Yard grows. + const decks = { ...s.decks, hands: new Map(s.decks.hands), salvageYard: [...s.decks.salvageYard] }; + return { ...s, officeAreas, trays, turns, decks, tally: cloneTally(s.tally) }; +} + +const carList = (xs: readonly RollingStock[]): string => + xs.map((c) => `${c.type}${c.loaded ? '+' : '-'}${c.origin ?? ''}`).join(','); + +/** + * Everything a switching intent can change, as a string — two positions with the same fingerprint + * are the same position as far as the rest of the turn is concerned. Identical cars are not told + * apart, which is right: no intent names a car. + */ +export function switchFingerprint(s: GameState, player: PlayerIndex): string { + const seat = seatOf(s, player); + const parts: string[] = []; + for (const [id, t] of s.trays) { + if (t.position.at !== 'grid' || t.position.seat !== seat) continue; + const { row, col } = t.position.coord; + parts.push(`${id}@${row},${col}/${t.facing}/${t.railFacing ?? ''}/${t.engineAt}:${carList(t.consist)}`); + } + for (const [key, card] of areaOf(s, player).grid) { + const track = card.facility?.kind === 'freight' ? card.facility.industryTrack.cars : null; + if (card.standing.length === 0 && card.standingWest === 0 && (track?.length ?? 0) === 0) continue; + parts.push(`${key}=${carList(card.standing)}|${card.standingWest}|${track ? carList(track) : ''}`); + } + const turn = turnOf(s, player); + parts.push(`m${turn.movesRemaining}`, JSON.stringify(turn.freightWorked), `h${(s.decks.hands.get(player) ?? []).join(',')}`); + return parts.join(';'); +} + +/** + * §9.3 — an outbound industry loads EMPTY cars of its commodity, an inbound one unloads LOADED ones — + * but never a load that was made in this same district (v0.4.9e, `LOADED_IN_THIS_DISTRICT`). + */ +function works(f: Facility, c: RollingStock, seat: number): boolean { + if (f.kind !== 'freight' || !facilityCarTypes(f).includes(c.type)) return false; + return (!c.loaded && f.allows.outbound) || (c.loaded && f.allows.inbound && c.origin !== seat); +} + +/** + * The car an industry has finished with. Only decidable at a one-way industry: at one that both + * ships and receives, a loaded car may be a delivery still waiting to be unloaded. + */ +function finished(f: Facility, c: RollingStock): boolean { + if (f.kind !== 'freight' || !facilityCarTypes(f).includes(c.type)) return false; + if (f.allows.outbound && !f.allows.inbound) return c.loaded; + if (f.allows.inbound && !f.allows.outbound) return !c.loaded; + return false; +} + +const same = (a: GridCoord, b: GridCoord): boolean => a.row === b.row && a.col === b.col; + +/** How good this district's position is for the rest of the game, in rough Revenue points. */ +export function evaluateSwitching(s: GameState, player: PlayerIndex): number { + const W = SWITCH_WEIGHTS; + const area = areaOf(s, player); + const seat = seatOf(s, player); + const officeKey = coordKey(area.officeCoord); + const passengerOffice = area.grid.get(officeKey)?.facility?.kind === 'passenger'; + let v = 0; + + const withRoom: Facility[] = []; + for (const card of area.grid.values()) { + const f = card.facility; + if (!f || f.kind !== 'freight') continue; + if (f.industryTrack.cars.length < MAX_CONSIST) withRoom.push(f); + const cap = Math.max(1, f.capacity.outbound + f.capacity.inbound); + let working = 0; + for (const c of f.industryTrack.cars) { + if (works(f, c, seat)) v += ++working <= cap ? W.spot : W.spotBeyondCapacity; + else if (finished(f, c)) v += W.finishedLeft; + else v += W.junkOnIndustry; + } + } + const wanted = (c: RollingStock): boolean => withRoom.some((f) => works(f, c, seat)); + + for (const [key, card] of area.grid) { + if (card.facility?.kind === 'freight') continue; + const atOffice = key === officeKey; + for (const c of card.standing) { + if (c.type === 'coach') v += atOffice && passengerOffice ? W.coachWithTrain : W.coachStranded; + else if (atOffice) v += W.fouling; + else if (wanted(c)) v += W.stagedWanted; + } + } + + for (const t of s.trays.values()) { + if (t.position.at !== 'grid' || t.position.seat !== seat) continue; + for (const c of t.consist) { + if (c.type === 'coach') v += passengerOffice ? W.coachWithTrain : 0; + else if (wanted(c)) v += W.carriedWanted; + } + if (t.trainNumber === null) continue; + if (!same(t.position.coord, area.officeCoord)) { + v += W.trainAway; + if (isExpedited(t)) v += W.expeditedAway; + } + if (badlyMadeUp(t)) v += W.notMadeUp; + } + return v; +} + +/** + * For ORDERING the beam only, never for choosing the plan: a Move toward an industry changes nothing + * the score can see until the car is set out, so without this the beam would drop the approach in + * favour of positions that merely look tidy. + */ +function approach(s: GameState, player: PlayerIndex): number { + const area = areaOf(s, player); + const seat = seatOf(s, player); + const targets: { at: GridCoord; f: Facility }[] = []; + for (const [key, card] of area.grid) { + const f = card.facility; + if (!f || f.kind !== 'freight' || f.industryTrack.cars.length >= MAX_CONSIST) continue; + const [row, col] = key.split(',').map(Number); + targets.push({ at: { row: row!, col: col! }, f }); + } + let bonus = 0; + for (const t of s.trays.values()) { + if (t.position.at !== 'grid' || t.position.seat !== seat) continue; + const here = t.position.coord; + for (const c of t.consist) { + let nearest = Infinity; + for (const { at, f } of targets) { + if (works(f, c, seat)) nearest = Math.min(nearest, Math.abs(at.row - here.row) + Math.abs(at.col - here.col)); + } + if (nearest !== Infinity) bonus += 0.1 / (1 + nearest); + } + } + return bonus; +} + +const SEARCHED = new Set(['switch.move', 'switch.dropCars', 'switch.sortConsist']); + +type Node = { + s: GameState; + steps: Intent[]; + keys: string[]; + moves: number; + setOuts: number; + cards: number; + score: number; + rank: number; +}; + +/** The best way found to spend what is left of this switching turn. Never mutates `s`. */ +export function planSwitchingTurn( + s: GameState, + player: PlayerIndex, + opts: PlanOptions = DEFAULT_PLAN, +): SwitchPlan { + const W = SWITCH_WEIGHTS; + const scoreOf = (st: GameState, moves: number, setOuts: number, cards: number): number => + evaluateSwitching(st, player) + moves * W.perMove + setOuts * W.perSetOut + cards * W.cardSpent; + const searched = (type: Intent['type']): boolean => + SEARCHED.has(type) || (opts.flyingSwitch === true && type === 'maneuver.flyingSwitch'); + + const rootKey = switchFingerprint(s, player); + const rootScore = scoreOf(s, 0, 0, 0); + const root: Node = { s, steps: [], keys: [rootKey], moves: 0, setOuts: 0, cards: 0, score: rootScore, rank: rootScore }; + let best = root; + const seen = new Set([rootKey]); + let frontier: Node[] = [root]; + let expanded = 0; + let complete = true; + + search: while (frontier.length > 0) { + const next: Node[] = []; + for (const node of frontier) { + const movesLeft = turnOf(node.s, player).movesRemaining; + // Every candidate is decided against THIS position, inside one route cache, and only then applied + // to its own copy: deciding on the copy would re-walk routes the listing had just walked. + const decided = withRouteCache(node.s, () => + legalSwitchingActions(node.s, player) + .filter((i) => searched(i.type) && (i.type === 'switch.dropCars' || movesLeft >= 1)) + .map((i) => ({ i, r: prepareIntent(node.s, player, i) })), + ); + for (const { i, r } of decided) { + if (expanded >= opts.budget) { + complete = false; + break search; + } + expanded++; + if (!r.ok) continue; + const f = forkForSwitching(node.s, player); + commitEvents(f, r.events); + const key = switchFingerprint(f, player); + if (seen.has(key)) continue; + seen.add(key); + const setOut = i.type === 'switch.dropCars'; + const moves = node.moves + (setOut ? 0 : 1); + const setOuts = node.setOuts + (setOut ? 1 : 0); + const cards = node.cards + (i.type === 'maneuver.flyingSwitch' ? 1 : 0); + const score = scoreOf(f, moves, setOuts, cards); + const child: Node = { + s: f, + steps: [...node.steps, i], + keys: [...node.keys, key], + moves, + setOuts, + cards, + score, + rank: score + approach(f, player), + }; + if (score > best.score + 1e-9) best = child; + next.push(child); + } + } + // A stable sort, so equal ranks keep `legalActions` order and the bot stays deterministic. + frontier = next.length > opts.beam ? next.sort((a, b) => b.rank - a.rank).slice(0, opts.beam) : next; + } + + return { steps: best.steps, keys: best.keys, rootScore, score: best.score, expanded, complete }; +} diff --git a/test/route-cache.test.ts b/test/route-cache.test.ts new file mode 100644 index 0000000..757e922 --- /dev/null +++ b/test/route-cache.test.ts @@ -0,0 +1,88 @@ +/** + * The engine's speed-ups must not change a single game (2026-09-15). + * + * `applyIntent` became `prepareIntent` (check and execute, sharing one walk of the position's routes) + * followed by `commitEvents` (reduce and tally), so the switching planner can decide every candidate + * against one position and apply each to a copy. Two properties hold that together: + * + * 1. `prepareIntent` never writes the state it reads — including through the route cache it opens. + * 2. Preparing on one state and committing to an EQUAL copy lands on exactly what `applyIntent` does. + * + * Checked at every decision of seeded bot games rather than on hand-built positions, so the intents + * exercised are the ones real play submits. + */ + +import { describe, it } from 'node:test'; +import assert from 'node:assert/strict'; + +import { pump } from '../src/engine/advance.ts'; +import { applyIntent, commitEvents, prepareIntent } from '../src/engine/apply.ts'; +import { + DEFAULT_MAX_COLLISIONS_PER_DAY, + DEFAULT_MAX_COLLISIONS_TOTAL, + collectiveRevenueFloor, + lengthProfile, +} from '../src/engine/content.ts'; +import { legalActions } from '../src/engine/legal.ts'; +import { createGame } from '../src/engine/setup.ts'; +import type { GameConfig, GameState } from '../src/engine/state.ts'; +import { developerBot, playGame } from '../src/sim/bot.ts'; + +const config = (): GameConfig => { + const days = lengthProfile('short').days; + return { + mode: 'solitaire', + days, + minCombinedRevenue: collectiveRevenueFloor(1, days), + maxCollisionsPerDay: DEFAULT_MAX_COLLISIONS_PER_DAY, + maxCollisionsTotal: DEFAULT_MAX_COLLISIONS_TOTAL, + pvpCardsAllowed: false, + optionalRules: { reducedVisibility: false, employeeRotation: false, emergencyToolbox: false }, + }; +}; + +const serialise = (s: GameState): string => + JSON.stringify(s, (_k, v) => (v instanceof Map ? [...v] : v instanceof Set ? [...v] : v)); + +describe('applyIntent split into prepareIntent and commitEvents', () => { + it('prepares without writing, and committing to a copy matches applying in place', () => { + let decisions = 0; + let rejectedSeen = 0; + const s = createGame({ id: 'split-8919', seed: 8919, config: config(), playerNames: ['bot'] }); + const policy = { + name: 'split-probe', + choose(st: GameState, player: number, options: ReturnType) { + const chosen = developerBot.choose(st, player, options); + if (decisions < 400) { + decisions++; + const before = serialise(st); + const prepared = prepareIntent(st, player, chosen); + assert.equal(serialise(st), before, `decision ${decisions}: prepareIntent wrote into the state it read`); + assert.ok(prepared.ok, `decision ${decisions}: a legal choice was refused by prepareIntent`); + + const viaCommit = structuredClone(st); + const viaApply = structuredClone(st); + commitEvents(viaCommit, prepared.events); + const applied = applyIntent(viaApply, player, chosen); + assert.ok(applied.ok); + assert.deepEqual(applied.events, prepared.events, `decision ${decisions}: the two paths produced different events`); + assert.equal(serialise(viaCommit), serialise(viaApply), `decision ${decisions}: committing to a copy diverged from applying`); + + // A refused intent must come back refused from both paths, with nothing written. + const refused = { type: 'switch.end' } as const; + const r = prepareIntent(st, player, refused); + if (!r.ok) { + rejectedSeen++; + assert.equal(serialise(st), before); + assert.equal(applyIntent(structuredClone(st), player, refused).ok, false); + } + } + return chosen; + }, + }; + const r = playGame(s, policy, pump); + assert.ok(r.finished, 'the probed game did not finish'); + assert.ok(decisions > 0, 'no decision was probed'); + assert.ok(rejectedSeen > 0, 'no refused intent was exercised'); + }); +}); diff --git a/test/switch-planner.test.ts b/test/switch-planner.test.ts new file mode 100644 index 0000000..ad3e1df --- /dev/null +++ b/test/switch-planner.test.ts @@ -0,0 +1,100 @@ +/** + * The switching planner (`sim/switch-planner.ts`) — the two properties it cannot be allowed to lose. + * + * 1. PLANNING TOUCHES NOTHING. The planner applies intents to a partial copy of the game + * (`forkForSwitching`) that shares everything a switching intent is not supposed to write. If a + * reducer ever starts writing somewhere new, the copy leaks into the real game, and this is where + * that shows: the game is serialised before and after planning and must not have changed. + * 2. A PLAN IS WHAT THE ENGINE WILL DO. Every step replays through `applyIntent` on a FULL copy, and + * lands on the fingerprint the planner promised for it. That is what makes the partial copy + * trustworthy, and it is what the bot relies on to know it is still on plan. + * + * Taken from real seeded bot games rather than hand-built positions, because a district that + * satisfies the track geometry by hand tests the builder as much as the planner. + */ + +import { describe, it } from 'node:test'; +import assert from 'node:assert/strict'; + +import { pump } from '../src/engine/advance.ts'; +import { applyIntent } from '../src/engine/apply.ts'; +import { legalActions, legalSwitchingActions } from '../src/engine/legal.ts'; +import { + DEFAULT_MAX_COLLISIONS_PER_DAY, + DEFAULT_MAX_COLLISIONS_TOTAL, + MOVES_PER_LOCAL_OPS, + collectiveRevenueFloor, + lengthProfile, +} from '../src/engine/content.ts'; +import { createGame } from '../src/engine/setup.ts'; +import type { GameConfig, GameState } from '../src/engine/state.ts'; +import { actingPlayer, turnOf } from '../src/engine/state.ts'; +import { developerBot, playGame } from '../src/sim/bot.ts'; +import { planSwitchingTurn, switchFingerprint } from '../src/sim/switch-planner.ts'; + +const config = (): GameConfig => { + const days = lengthProfile('short').days; + return { + mode: 'solitaire', + days, + minCombinedRevenue: collectiveRevenueFloor(1, days), + maxCollisionsPerDay: DEFAULT_MAX_COLLISIONS_PER_DAY, + maxCollisionsTotal: DEFAULT_MAX_COLLISIONS_TOTAL, + pvpCardsAllowed: false, + optionalRules: { reducedVisibility: false, employeeRotation: false, emergencyToolbox: false }, + }; +}; + +const serialise = (s: GameState): string => + JSON.stringify(s, (_k, v) => (v instanceof Map ? [...v] : v instanceof Set ? [...v] : v)); + +describe('switching planner', () => { + it('asks for switching intents that are exactly the switching subset of legalActions, in order', () => { + const SWITCHING = new Set(['switch.move', 'switch.dropCars', 'switch.sortConsist', 'maneuver.flyingSwitch', 'switch.end']); + let compared = 0; + for (const seed of [1000, 8919]) { + const s = createGame({ id: `legal-${seed}`, seed, config: config(), playerNames: ['bot'] }); + playGame(s, developerBot, pump, 50_000, undefined, (st) => { + const p = actingPlayer(st); + if (p === null || st.clock.phase !== 'localOps') return; + const all = legalActions(st, p).filter((i) => SWITCHING.has(i.type)); + assert.deepEqual(legalSwitchingActions(st, p), all); + compared++; + }); + } + assert.ok(compared > 0, 'no Local Operations decision was reached'); + }); + + it('never changes the game it plans for, and every plan replays to the position it promised', () => { + let checked = 0; + let withSteps = 0; + for (const seed of [1000, 8919, 16838]) { + const s = createGame({ id: `plan-${seed}`, seed, config: config(), playerNames: ['bot'] }); + const r = playGame(s, developerBot, pump, 50_000, undefined, (st) => { + const p = actingPlayer(st); + if (p === null || st.clock.phase !== 'localOps') return; + const turn = turnOf(st, p); + if (turn.option !== 'switch' || turn.movesRemaining !== MOVES_PER_LOCAL_OPS) return; + + const before = serialise(st); + const plan = planSwitchingTurn(st, p, { budget: 400, beam: 16 }); + assert.equal(serialise(st), before, `seed ${seed}: planning wrote into the real game`); + assert.ok(plan.score >= plan.rootScore, 'a plan is never worse than stopping where the crew stands'); + + const copy = structuredClone(st); + plan.steps.forEach((step, n) => { + assert.equal(switchFingerprint(copy, p), plan.keys[n], `seed ${seed}: step ${n} started off plan`); + const applied = applyIntent(copy, p, step); + assert.ok(applied.ok, `seed ${seed}: step ${n} (${step.type}) was refused by the engine`); + }); + assert.equal(switchFingerprint(copy, p), plan.keys.at(-1), `seed ${seed}: the plan did not end where it said`); + + checked++; + if (plan.steps.length > 0) withSteps++; + }); + assert.ok(r.finished, `seed ${seed}: a game with the planner switched on did not finish`); + } + assert.ok(checked > 0, 'no switching turn was reached, so nothing was tested'); + assert.ok(withSteps > 0, 'every plan was empty, so replay was never exercised'); + }); +});