From 70d024da62791e2a824c36653895c24a46d65281 Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Mon, 17 Aug 2026 12:37:23 +0530 Subject: [PATCH 01/10] Check the egress work against the database it actually runs on Everything measured so far ran on SQLite against a synthetic fixture, so two claims were still on trust: that migration 38 spells BYTEA correctly for a real server, and that the byte figures survive psycopg's encodings. Both hold. Production reads schema_version 41 with embedding_blob bytea and embedded boolean present, the backfill is complete at 134/134, and the packed vectors are 5.04x smaller than the JSON on real data -- 30,971 to 6,144 bytes a memory, as predicted. stress_session now takes AIDND_STRESS_DATABASE_URL, and against a throwaway Neon database every shape lands within 0.5% of the SQLite run: the warm turn is 121.1 kB against 122.3, with memories down to 1.7 kB of it. The harness writes, so it refuses any target whose name does not say stress or scratch -- pointed at the production database it stops rather than seeding it with a fake user and 200 fake turns. It also empties a Postgres target before building, which a fresh SQLite temp file never needed. Two corrections fall out, both recorded in plan/13. The page-load model has the wrong shape: real actions are half the fixture's weight but real stories run to 607 actions, not 200, so the worst real page load is 589.5 kB. And the decision to leave context_snapshot in the database costed egress but never storage -- it is 88.9 MB of a 99.6 MB database against a 512 MB free tier, which is the ceiling this deploy will hit first. Measured with counts and octet_length sums only. No user content was read. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7 --- backend/tools/stress_session.py | 62 ++++++++++++++++++++----- plan/13-memory-embedding-cost.md | 79 ++++++++++++++++++++++++++++++++ plan/STATUS.md | 76 +++++++++++++++++++++++++++--- 3 files changed, 200 insertions(+), 17 deletions(-) diff --git a/backend/tools/stress_session.py b/backend/tools/stress_session.py index 64b8faa..999352b 100644 --- a/backend/tools/stress_session.py +++ b/backend/tools/stress_session.py @@ -23,11 +23,22 @@ sessions, the ORM, the scripting engine and the context builder are the real ones, because the bugs this exists to catch live in exactly the layer a mock would replace. -It runs on a throwaway SQLite file rather than Postgres. What is being measured -is which columns of which rows a code path asks for, and that is decided by the -ORM, identically on both. The dialects disagree on how a value is encoded on -the wire — JSON especially — so treat the absolute figures as production-shaped -rather than production-exact, and compare before against after. +It runs on a throwaway SQLite file by default. What is being measured is which +columns of which rows a code path asks for, and that is decided by the ORM, +identically on both dialects. The dialects disagree on how a value is encoded +on the wire — JSON especially — so treat the absolute figures as +production-shaped rather than production-exact, and compare before against +after. + +To measure the encodings SQLite cannot reach — bytea for the packed vectors, +and json columns psycopg parses before the meter sees them — set +AIDND_STRESS_DATABASE_URL to a **throwaway** Postgres database: + + AIDND_STRESS_DATABASE_URL=postgresql://…/stress_scratch \ + .venv/Scripts/python.exe -m tools.stress_session + +The harness writes, so it refuses any target whose database name does not say +'stress' or 'scratch'. Never point it at a database holding real users. Calibration, against the two figures measured directly on production (2026-08-16): a 200-action page load reported 426.7 kB here against 423 KB @@ -35,19 +46,41 @@ there, and one turn on a 100-memory bank reported 3,258.7 kB against 3,153 kB. """ import os +import sys import tempfile # Must precede the app import: database.py reads these at module scope. -_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False) -_tmp.close() -os.environ["AIDND_DB_PATH"] = _tmp.name -os.environ.pop("AIDND_DATABASE_URL", None) -os.environ.pop("DATABASE_URL", None) +# +# Default is a throwaway SQLite file. AIDND_STRESS_DATABASE_URL points the +# harness at a real Postgres instead, which is the only way to reach the +# encodings SQLite cannot exercise: bytea for the packed vectors, and json +# columns that psycopg parses into Python before the meter ever sees them. +# +# The name guard is not paranoia. This harness *writes* — it builds a whole +# synthetic adventure — so a URL that happened to point at the production +# database would quietly seed it with fake users and fake play. The target +# must say it is disposable. +_stress_url = os.environ.get("AIDND_STRESS_DATABASE_URL", "").strip() +if _stress_url: + _dbname = _stress_url.rsplit("/", 1)[-1].split("?")[0] + if not any(mark in _dbname.lower() for mark in ("stress", "scratch")): + sys.exit( + f"refusing to run against database {_dbname!r}.\n" + "This harness writes a synthetic adventure, so its target must be a\n" + "throwaway database with 'stress' or 'scratch' in the name." + ) + os.environ["AIDND_DATABASE_URL"] = _stress_url + os.environ.pop("DATABASE_URL", None) +else: + _tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False) + _tmp.close() + os.environ["AIDND_DB_PATH"] = _tmp.name + os.environ.pop("AIDND_DATABASE_URL", None) + os.environ.pop("DATABASE_URL", None) import argparse import asyncio import random -import sys from fastapi import Depends from fastapi.testclient import TestClient @@ -125,6 +158,13 @@ class FakeEmbeddings: def build_fixture(args, rng: random.Random) -> tuple[int, int]: """A user, settings and one adventure at production scale. Returns (adventure_id, user_id).""" + # A SQLite run gets a brand-new temp file every time, so the fixture can + # assume an empty database. A Postgres scratch target persists between + # runs, and the second one would collide on the fixture user's unique + # email — so empty it first. Only ever reached for a target whose name + # passed the 'stress'/'scratch' guard at the top of this module. + if _stress_url: + Base.metadata.drop_all(bind=engine) Base.metadata.create_all(bind=engine) db = SessionLocal() try: diff --git a/plan/13-memory-embedding-cost.md b/plan/13-memory-embedding-cost.md index 4cd95a8..1076d32 100644 --- a/plan/13-memory-embedding-cost.md +++ b/plan/13-memory-embedding-cost.md @@ -182,6 +182,9 @@ Deliberately **not** taken: moving `context_snapshot` out of the database. It co nothing on reads now that it is deferred, and storage is ~$0.02/mo. Revisit only if backups or storage start to hurt. +> **Revisit it.** That call weighed egress and got egress right, but it never weighed +> the free tier's *storage* ceiling — see "Storage, which this plan did not cost" below. + ## Verification - Harness: `python -m tools.stress_session`, memory bank **on**, before and after, @@ -193,3 +196,79 @@ backups or storage start to hurt. playthrough number is finally honest. - The existing `test_egress.py` guard must still pass — nothing here should touch the deferred action columns. + +## Verified on production, 2026-08-17 + +Two things were still taken on trust when this shipped: every measurement had run on +SQLite, and every number came from a synthetic fixture. Both are now checked. + +### The migration landed on real Postgres + +`schema_version` reads **41**, matching the repo's `LATEST_VERSION`. The live schema has +`embedding_blob bytea` and `embedded boolean`, so the `{dialect: sql}` map in migration +38 spells BYTEA correctly against a real server — the one thing tests could not prove, +since `test_migration_38_is_spelled_for_both_dialects` only inspects the SQL string. +The backfill is complete: 134 memories, `embedded = 134`, `embedding_blob = 134`, no +stragglers and no rows skipped as malformed. + +### The 5x is real, on real vectors + +| | bytes | per memory | +|---|---|---| +| `embedding` (JSON) | 4,150,121 | 30,971 | +| `embedding_blob` (float32) | 823,296 | 6,144 | + +**5.04x**, against the plan's predicted ~31 KB → 6,144 B. The largest real bank is 100 +memories = 614,400 B of vectors, so the old code fetched **~3.10 MB per retrieval** on +that adventure — which is where the 3,153 kB measured on production came from. That +figure is now fully accounted for. + +### SQLite and Postgres agree + +`tools.stress_session` gained an `AIDND_STRESS_DATABASE_URL` escape hatch and was run +against a throwaway Neon database at the default fixture (200 actions, 100 memories): + +| shape | SQLite | Postgres | +|---|---|---| +| index | 4.1 kB | 4.1 kB | +| page load | 426.7 kB | 425.0 kB | +| one turn, cold | 723.4 kB | 722.3 kB | +| one turn, warm | 122.3 kB | **121.1 kB** | +| Insights | 117.9 kB | 116.7 kB | +| Memories drawer | 23.7 kB | 21.7 kB | +| `run_post_turn` | 0.7 kB | 0.6 kB | + +Within 0.5% everywhere. The dialect caveat in the harness docstring is real but small: +what dominates is which columns get asked for, and the ORM decides that identically. +The warm turn spends **1.7 kB on `memories`, 1% of the read** — the cache behaves on +psycopg exactly as it does on SQLite. + +### The page load is worse than modelled, for a different reason + +The synthetic fixture is **~2x heavier per action than production**: 994 B/action real +against ~2,133 B/action synthetic, so a real 200-action adventure is ~194 kB, not 427. +But the largest real adventure is **607 actions**, not 200, and costs **589.5 kB** in +one response. Step 6 is more urgent than this plan assumed, and for the opposite +reason to the one modelled — stories get *longer* than the fixture, not heavier. + +Worth fixing the fixture's narration size when step 6 lands, so the harness stops +flattering the per-action figure while understating the length. + +### Storage, which this plan did not cost + +`context_snapshot` is **150.8 MB of uncompressed JSON across 944 actions** — ~163 kB a +row on average, and ~232 kB a row in the largest adventure, against the ~74 KB/row the +comment in `models.py` claims. TOAST compresses it to ~89 MB on disk, but +`octet_length` is what would cross the wire, because Postgres decompresses before +sending. Deferral is the only thing standing between a bulk read and a 137 MB query. + +The database is **99.6 MB total**, of which `actions` is **88.9 MB**. Neon's free tier +is 512 MB. At ~94 kB of disk per action that ceiling arrives at roughly **5,400 +actions**, and 944 are already stored. So the "~$0.02/mo, leave it in the database" +call above is wrong for the tier this actually runs on — not because reads cost +anything, but because the free tier meters *storage*, and that is the constraint with +a cliff. Dropping the dead `memories.embedding` column reclaims 4.05 MB (4%), which +helps and does not solve it. + +None of the numbers above required reading a single row of anyone's content: counts, +`octet_length` sums and catalog sizes only. diff --git a/plan/STATUS.md b/plan/STATUS.md index 5f4235d..c2a8133 100644 --- a/plan/STATUS.md +++ b/plan/STATUS.md @@ -3,7 +3,23 @@ Read this first when picking the project back up. Updated at the end of a working session; the per-phase plan files hold the detail, this holds the thread. -**Last updated: 2026-08-16.** +**Last updated: 2026-08-17.** + +--- + +## Two things need a human first + +**The Render service is suspended.** `GET /api/health` returns 503 with a static +"This service has been suspended by its owner" page, in ~1.2s — that is the edge, not +a cold start (a free-tier wake hangs 30–60s and then serves). Nothing in the app is +wrong; check the dashboard. Free-tier suspensions come from usage/bandwidth caps or +billing, and real users have started arriving, so rule that out before assuming it was +manual. + +**The free tier's storage ceiling is closer than the egress work suggested.** The Neon +database is 99.6 MB of a 512 MB allowance and `actions.context_snapshot` is essentially +all of it. See "Storage, which this plan did not cost" in `plan/13`. This is now the +most likely thing to break the deploy, ahead of anything on the read path. --- @@ -11,9 +27,13 @@ session; the per-phase plan files hold the detail, this holds the thread. **`plan/13-memory-embedding-cost.md`, step 6 — infinite scroll upward in `Play.jsx`.** -Opening a finished 200-action adventure fetches **426.7 kB** in one response, and after -this session's work that is comfortably the largest single read in the app — a turn is -now 122 kB, Insights 118 kB, the Memories drawer 24 kB. The backend already has the +Opening a finished adventure is comfortably the largest single read in the app — a turn +is now 122 kB, Insights 118 kB, the Memories drawer 24 kB. Measured on production +(2026-08-17), the largest real adventure is **607 actions and 589.5 kB in one +response**; the 426.7 kB the harness reports is a 200-action fixture whose actions are +about **twice as heavy as real ones** (994 B/action in production). So the fixture +overstates width and understates length — real stories get *longer* than it models, +which is the direction that hurts. The backend already has the windowing primitives (`context/history.py`: `tail_range`, `slice_`, `count`), and `GET /adventures/{id}/actions` exists. What is missing is a paged shape for it and a `Play.jsx` that loads the newest turns and fetches older ones as the reader scrolls up. @@ -91,6 +111,26 @@ Migrations 39/40 add `memories.embedded`, migration 41 drops the capacity defaul --- +## What happened on 2026-08-17 + +No new behaviour — a verification pass on what shipped the day before, because every +number in the section above had been measured on SQLite against a synthetic fixture. +Full write-up in `plan/13` under "Verified on production". + +**It holds.** `schema_version` is 41 on the live Postgres with `embedding_blob bytea` +and `embedded boolean` present, so migration 38's dialect map is correct against a real +server. The backfill is complete (134/134). The packed vectors are **5.04x** smaller +than the JSON on real data — 30,971 → 6,144 bytes a memory, as predicted. + +**SQLite was not lying.** `tools.stress_session` can now target Postgres via +`AIDND_STRESS_DATABASE_URL`, and every shape agrees within 0.5% — the warm turn is +121.1 kB on Postgres against 122.3 kB on SQLite, with `memories` down to 1.7 kB of it. +Run it against a **throwaway** database only; the harness writes, so it refuses any +target whose name does not contain `stress` or `scratch`. + +**Two corrections came out of it**, both above: the page-load model has the wrong +shape (too heavy per action, far too short), and the storage ceiling was never costed. + ## Things worth remembering **The vector cache needs no invalidation callbacks, and that is why it is safe.** A @@ -112,6 +152,17 @@ beside `actions.variants`. Expect to need this for any future heavy column. **Any egress measurement must run with an embedding model set.** This is the second time that omission has hidden the biggest number in the room. +**Production has real users on it now. Measure it without reading it.** Counts, +`sum(octet_length(...))` and `pg_total_relation_size` answer every sizing question +asked so far, and none of them return anyone's story, memory text or email. When a +real Postgres is needed for a *write* path, create a throwaway database beside the real +one and drop it after — never point a harness at the production database. + +**`octet_length` is the egress number, not the on-disk number.** Postgres TOAST +compresses big JSON — `context_snapshot` is 150.8 MB uncompressed but ~89 MB stored — +and decompresses before sending. Size reads with `octet_length`, size the storage bill +with `pg_total_relation_size`, and do not mix them up. + --- ## Still open from `plan/13` @@ -124,7 +175,11 @@ time that omission has hidden the biggest number in the room. Done for the memory paths, not as a general rule. - **Drop `memories.embedding`** (the JSON column) in a follow-up migration. It is still written by `set_vector` and read by nothing, kept so a rollback finds the vectors. - `tests/test_memory_retrieval.py` has a guard asserting nothing selects it. + `tests/test_memory_retrieval.py` has a guard asserting nothing selects it. Measured + on production: dropping it reclaims 4.05 MB, 4% of the database. +- **`context_snapshot` and the 512 MB ceiling** — new, and now the biggest open item. + See the two sections named above. The egress case for leaving it in the database + still stands; the storage case does not. Deliberately not taken: moving `context_snapshot` out of the database (~$0.02/mo, costs nothing on reads now that it is deferred), and pgvector (breaks the SQLite dev parity @@ -137,8 +192,17 @@ this codebase protects on purpose). ``` cd backend .venv/Scripts/python.exe -m pytest tests/ # 225 tests -.venv/Scripts/python.exe -m tools.stress_session # egress report +.venv/Scripts/python.exe -m tools.stress_session # egress report (SQLite) + +# Same harness against a real Postgres. The target must be a THROWAWAY database +# — this writes a synthetic adventure, and it refuses any name without +# 'stress'/'scratch' in it. +AIDND_STRESS_DATABASE_URL=postgresql://…/stress_scratch \ + .venv/Scripts/python.exe -m tools.stress_session ``` +On Windows the report's box-drawing characters crash the default cp1252 console; +prefix with `PYTHONIOENCODING=utf-8`. + Port 8000 is shared with the job-pipeline app, which will squat it and silently shadow the AI-DnD API — free it before running the backend, or move the vite proxy. From 85b188977e11112f9099e2b54900dd87f29e779a Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Mon, 17 Aug 2026 12:43:06 +0530 Subject: [PATCH 02/10] Correct the production URL: the deploy is ai-dnd-1gmp, not ai-dnd The previous commit recorded the service as suspended. It is not. Render appended a suffix to the `ai-dnd` service name in render.yaml, and plain ai-dnd.onrender.com is a different, suspended service whose 503 page reads "suspended by its owner" -- indistinguishable from this deploy being down unless you notice the host is wrong. The real host answers {"ok":true} on /api/health in 0.6s, mints a guest, returns an empty adventure list for that guest and serves the starter scenarios. The authoritative link is the one docs/index.html points at, not the service name in the blueprint. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7 --- plan/STATUS.md | 18 +++++++++++------- 1 file changed, 11 insertions(+), 7 deletions(-) diff --git a/plan/STATUS.md b/plan/STATUS.md index c2a8133..0389445 100644 --- a/plan/STATUS.md +++ b/plan/STATUS.md @@ -7,14 +7,18 @@ session; the per-phase plan files hold the detail, this holds the thread. --- -## Two things need a human first +## The live URL is not the one in render.yaml -**The Render service is suspended.** `GET /api/health` returns 503 with a static -"This service has been suspended by its owner" page, in ~1.2s — that is the edge, not -a cold start (a free-tier wake hangs 30–60s and then serves). Nothing in the app is -wrong; check the dashboard. Free-tier suspensions come from usage/bandwidth caps or -billing, and real users have started arriving, so rule that out before assuming it was -manual. +**Production is `https://ai-dnd-1gmp.onrender.com`.** Render appended a suffix to the +`ai-dnd` service name in `render.yaml`, and plain `ai-dnd.onrender.com` belongs to a +different, suspended service that answers 503 with "suspended by its owner" — which is +easy to mistake for this deploy being down. The authoritative link is the one the +project page points at (`docs/index.html`), not the service name in the blueprint. +`GET /api/health` on the real host returns `{"ok":true}`. + +--- + +## Needs a human **The free tier's storage ceiling is closer than the egress work suggested.** The Neon database is 99.6 MB of a 512 MB allowance and `actions.context_snapshot` is essentially From 2c5909a26896f017865d60ebbc814f3480145cf2 Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Mon, 17 Aug 2026 13:44:59 +0530 Subject: [PATCH 03/10] Drop the JSON vector column, and fix what was hiding behind it Migration 38 left memories.embedding in place so a rollback could still find the vectors. Production has since been verified reading from embedding_blob, so migration 42 drops it: 4 MB of a 99.6 MB database holding nothing anyone reads. Removing it surfaced a live bug. Changing your embedding model is supposed to throw the bank's vectors away and let the post-turn pass rebuild them, because two models' vectors are not comparable. The settings route did that by nulling memories.embedding -- correct until 38 moved the vectors, after which it cleared the dead column and left the blob intact with `embedded` still true. _embed_pending filters on `embedded IS FALSE`, so it never saw those rows and the bank went on ranking against the old model's vectors permanently. Nothing would have reported it. cosine returns 0.0 on a width mismatch, so a different-width model scores every memory zero and retrieval returns whichever rows happen to sort first; a same-width model scores plausible garbage. The bulk clear now sets both columns. It stays a bulk UPDATE rather than going through set_vector -- loading the rows is the cost that whole path exists to avoid -- so set_vector's docstring now names it as the one caller that legitimately writes those columns by hand. No cache invalidation is added: clearing `embedded` drops the rows out of the catalogue query, and set_vector evicts each entry as the re-embed puts it back. test_embedding_blob.py now rebuilds the pre-38 schema by hand where it tests the backfill, since create_all no longer produces the column it converts from, and asserts 42 removes it at the end of a full bootstrap -- 38 reads that column and 42 drops it, so an upgrade that reordered them would arrive with an empty bank. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7 --- backend/app/memorybank.py | 14 +- backend/app/migrations.py | 6 + backend/app/models.py | 4 - backend/app/routers/settings.py | 14 +- backend/tests/test_embedding_blob.py | 40 ++++- backend/tests/test_embedding_model_switch.py | 169 +++++++++++++++++++ backend/tests/test_memory_retrieval.py | 17 +- 7 files changed, 237 insertions(+), 27 deletions(-) create mode 100644 backend/tests/test_embedding_model_switch.py diff --git a/backend/app/memorybank.py b/backend/app/memorybank.py index dba48b3..b955a4b 100644 --- a/backend/app/memorybank.py +++ b/backend/app/memorybank.py @@ -86,13 +86,15 @@ def set_vector(memory: models.Memory, vector: list[float] | None) -> None: """Store (or clear) a memory's embedding. Every column that describes the vector moves together: `embedding_blob` is - what the ranking reads, `embedded` is the flag everything else reads, and - the JSON `embedding` stays correct behind both until the follow-up - migration drops it. Going through one function is what keeps them in step — - and it is also the only place a stored vector can change, which is what - makes the cache below safe to invalidate here and nowhere else. + what the ranking reads and `embedded` is the flag everything else reads. + Going through one function is what keeps them in step — and it is also the + only place a stored vector can change, which is what makes the cache below + safe to invalidate here and nowhere else. + + The one caller that legitimately cannot come through here is the bulk + clear in `routers/settings.py` when the embedding model changes. It has to + set the same two columns by hand; see the note there. """ - memory.embedding = vector memory.embedding_blob = None if vector is None else vectors.pack(vector) memory.embedded = vector is not None cached = _vector_cache.get(memory.adventure_id) diff --git a/backend/app/migrations.py b/backend/app/migrations.py index 1339317..413cc11 100644 --- a/backend/app/migrations.py +++ b/backend/app/migrations.py @@ -144,6 +144,12 @@ MIGRATIONS: list[tuple[int, str | dict[str, str]]] = [ # so anyone who picked a value keeps it — same rule as migration 29. # Adventures already over 80 evict down on their next turn. (41, "UPDATE settings SET memory_bank_capacity = 80 WHERE memory_bank_capacity = 200"), + # The JSON vectors, gone. Migration 38 left them in place so a rollback + # could still find them; production has since been verified reading from + # embedding_blob (schema_version 41, 134/134 backfilled), so the column is + # now 4 MB of a 99.6 MB database holding nothing anyone reads. DROP COLUMN + # is spelled the same on both dialects — SQLite has had it since 3.35. + (42, "ALTER TABLE memories DROP COLUMN embedding"), ] LATEST_VERSION = max((v for v, _ in MIGRATIONS), default=1) diff --git a/backend/app/models.py b/backend/app/models.py index 7979e24..625ed7a 100644 --- a/backend/app/models.py +++ b/backend/app/models.py @@ -158,10 +158,6 @@ class Memory(Base): id: Mapped[int] = mapped_column(primary_key=True) adventure_id: Mapped[int] = mapped_column(ForeignKey("adventures.id", ondelete="CASCADE")) text: Mapped[str] = mapped_column(Text, default="") - # Superseded by embedding_blob and still written alongside it, so a - # rollback finds the vectors, until a follow-up migration drops it. Nothing - # reads it. - embedding: Mapped[list | None] = mapped_column(JSON, nullable=True, deferred=True) # The vector, little-endian float32. Deferred because it is wider than the # rest of the row put together and exactly one code path wants it: anything # bulk-loading memories (the Memories drawer, eviction, the embed queue) diff --git a/backend/app/routers/settings.py b/backend/app/routers/settings.py index 637093a..58c5b25 100644 --- a/backend/app/routers/settings.py +++ b/backend/app/routers/settings.py @@ -51,14 +51,26 @@ def update_settings( # Vectors from the old model have a different dimensionality/space; # clear them so the post-turn task re-embeds with the new model. # (This user's adventures only — settings are per-user now.) + # + # Both columns, and the flag. This is the one place that clears vectors + # in bulk rather than through memorybank.set_vector, and when the + # vectors moved to embedding_blob it kept nulling the old JSON column + # alone: the blob survived, `embedded` stayed true, and _embed_pending + # — which looks for embedded IS FALSE — never picked the rows up. The + # bank went on ranking against the previous model's vectors forever. owned = ( db.query(models.Adventure.id) .filter(models.Adventure.user_id == user.id) .scalar_subquery() ) db.query(models.Memory).filter(models.Memory.adventure_id.in_(owned)).update( - {"embedding": None}, synchronize_session=False + {"embedding_blob": None, "embedded": False}, synchronize_session=False ) + # No cache invalidation needed, and deliberately none added: clearing + # `embedded` drops these rows out of the catalogue query, so retrieval + # stops asking for them, and by the time _embed_pending puts one back + # it has gone through set_vector, which evicts that entry. The rule + # holds — anything that removes a memory from play self-corrects. db.commit() return settings diff --git a/backend/tests/test_embedding_blob.py b/backend/tests/test_embedding_blob.py index 0ab4c52..47d8914 100644 --- a/backend/tests/test_embedding_blob.py +++ b/backend/tests/test_embedding_blob.py @@ -111,9 +111,9 @@ def test_cosine_moved_but_still_reachable_from_memorybank(): # -------------------------------------------------------------- set_vector -def test_set_vector_writes_both_columns(db, adventure): - """Until the follow-up migration drops the JSON column, it has to stay - correct — a rollback reads it.""" +def test_set_vector_writes_the_blob_and_the_flag(db, adventure): + """The two columns that describe a vector move together, or a reader that + trusts `embedded` gets a NULL blob.""" memory = models.Memory(adventure_id=adventure.id, text="a fact") db.add(memory) db.commit() @@ -123,7 +123,6 @@ def test_set_vector_writes_both_columns(db, adventure): db.commit() db.expire_all() - assert memory.embedding == vector assert list(vectors.unpack(memory.embedding_blob)) == vector assert memory.embedded is True @@ -140,24 +139,40 @@ def test_set_vector_none_clears_both(db, adventure): db.commit() db.expire_all() - assert memory.embedding is None assert memory.embedding_blob is None assert memory.embedded is False # --------------------------------------------------------------- the backfill +def add_legacy_json_column(db) -> None: + """Put `memories.embedding` back for the length of a test. + + Migration 42 dropped it and the model no longer declares it, so + `create_all` does not produce it — but everything below is testing the + upgrade *from* a database that still has it, which is the only state in + which the backfill has any work to do. Re-adding it by hand is what keeps + these tests honest about the schema they claim to be starting from. + """ + db.execute(text("ALTER TABLE memories ADD COLUMN embedding JSON")) + db.commit() + + def seed_json_only(db, adventure, count: int, dims: int = 64) -> dict[int, list[float]]: """Memories as they exist before the migration: JSON vector, no blob.""" + add_legacy_json_column(db) rng = random.Random(count) expected = {} for i in range(count): vector = sample_vector(rng, dims) - memory = models.Memory( - adventure_id=adventure.id, text=f"fact {i}", embedding=vector - ) + memory = models.Memory(adventure_id=adventure.id, text=f"fact {i}") db.add(memory) db.flush() + # Raw, because the ORM no longer knows this column exists. + db.execute( + text("UPDATE memories SET embedding = :v WHERE id = :id"), + {"v": json.dumps(vector), "id": memory.id}, + ) expected[memory.id] = vector db.commit() db.execute(text("UPDATE memories SET embedding_blob = NULL, embedded = false")) @@ -193,6 +208,7 @@ def test_backfill_reaches_past_one_batch(db, adventure): def test_backfill_leaves_unembedded_memories_alone(db, adventure): + add_legacy_json_column(db) db.add(models.Memory(adventure_id=adventure.id, text="not embedded yet")) db.commit() @@ -274,6 +290,14 @@ def test_bootstrap_adds_the_columns_and_backfills_them(db, adventure): # embedded must still read as not embedded afterwards. assert by_id[unembedded_id] == (None, False) + # ...and migration 42, at the end of the same run, takes the JSON column + # away. Ordering matters: 38 reads it, 42 drops it, and an upgrade that + # ran them the other way round would arrive with an empty bank. + with engine.begin() as conn: + columns = {row[1] for row in conn.execute(text("PRAGMA table_info(memories)"))} + assert "embedding" not in columns + assert {"embedding_blob", "embedded"} <= columns + def test_migration_38_is_spelled_for_both_dialects(): """Every Postgres deploy replays migrations from 24 on, so a SQLite-only diff --git a/backend/tests/test_embedding_model_switch.py b/backend/tests/test_embedding_model_switch.py new file mode 100644 index 0000000..ab55714 --- /dev/null +++ b/backend/tests/test_embedding_model_switch.py @@ -0,0 +1,169 @@ +"""Switching embedding models must re-embed the bank. + +Vectors from two different models are not comparable — different space, often +different width — so changing the model has to throw the stored ones away and +let the post-turn pass rebuild them. + +That worked while the vectors lived in `memories.embedding`: the settings +route nulled that column and the embed queue picked the rows up. Migration 38 +moved the vectors to `embedding_blob` with an `embedded` flag beside them, and +the bulk clear kept nulling the old column alone. The blob survived, the flag +stayed true, `_embed_pending` (which looks for `embedded IS FALSE`) never saw +the rows, and the bank went on ranking against the previous model's vectors +for good. + +Nothing reports this. `cosine` returns 0.0 on a width mismatch, so a +different-width model scores every memory zero and retrieval quietly returns +whichever rows sort first; a same-width model scores plausible-looking +garbage. + + python -m pytest tests/test_embedding_model_switch.py -v +""" +import os +import tempfile + +_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False) +_tmp.close() +os.environ["AIDND_DB_PATH"] = _tmp.name +os.environ.pop("AIDND_DATABASE_URL", None) +os.environ.pop("DATABASE_URL", None) + +import asyncio + +import pytest +from fastapi import Depends +from fastapi.testclient import TestClient + +from app import auth, limits, memorybank, models +from app.database import Base, SessionLocal, engine, get_db +from app.main import app + +DIMS = 8 + + +@pytest.fixture() +def client(monkeypatch): + Base.metadata.create_all(bind=engine) + memorybank._vector_cache.clear() + setup = SessionLocal() + user = models.User(is_guest=False, email="switch@example.com") + setup.add(user) + setup.flush() + setup.add(models.Settings( + user_id=user.id, api_key="enc:dummy", model="test-model", + embedding_model="model-a", + )) + adventure = models.Adventure( + user_id=user.id, title="Cave", script_state={}, memory_bank_enabled=True + ) + setup.add(adventure) + setup.flush() + for i in range(5): + memory = models.Memory(adventure_id=adventure.id, text=f"Memory {i}") + memorybank.set_vector(memory, [float(i)] + [0.0] * (DIMS - 1)) + setup.add(memory) + setup.commit() + adv_id, user_id = adventure.id, user.id + setup.close() + + monkeypatch.setattr(limits, "rate_limit", lambda *a, **k: None) + monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) + + def _current_user(db=Depends(get_db)): + return db.get(models.User, user_id) + + app.dependency_overrides[auth.get_current_user] = _current_user + c = TestClient(app) + c.adv_id = adv_id + try: + yield c + finally: + app.dependency_overrides.clear() + memorybank._vector_cache.clear() + Base.metadata.drop_all(bind=engine) + + +def memories(db): + return db.query(models.Memory).order_by(models.Memory.id).all() + + +def test_the_bank_starts_embedded(client): + db = SessionLocal() + try: + rows = memories(db) + assert len(rows) == 5 + assert all(m.embedded for m in rows) + assert all(m.embedding_blob for m in rows) + finally: + db.close() + + +def test_changing_the_model_clears_every_vector(client): + r = client.put("/api/settings", json={"embedding_model": "model-b"}) + assert r.status_code == 200, r.text + + db = SessionLocal() + try: + rows = memories(db) + assert [m.embedding_blob for m in rows] == [None] * 5, \ + "the blob survived the model change" + assert not any(m.embedded for m in rows), \ + "`embedded` stayed true, so nothing will ever re-embed these" + finally: + db.close() + + +def test_cleared_memories_are_queued_for_re_embedding(client): + """The flag is not cosmetic: it is the only thing `_embed_pending` filters + on, so this is the assertion that the bank actually recovers.""" + client.put("/api/settings", json={"embedding_model": "model-b"}) + + db = SessionLocal() + try: + pending = ( + db.query(models.Memory) + .filter(models.Memory.embedded.is_(False), + models.Memory.forgotten.is_(False)) + .all() + ) + assert len(pending) == 5 + finally: + db.close() + + +def test_retrieval_uses_no_stale_vector_after_the_switch(client, monkeypatch): + """Until the re-embed runs, the bank must return nothing rather than + ranking against the old model's vectors.""" + client.put("/api/settings", json={"embedding_model": "model-b"}) + + class Embedder: + async def embed(self, texts): + return [[1.0] + [0.0] * (DIMS - 1) for _ in texts] + + monkeypatch.setattr(memorybank, "embedding_provider", lambda s: Embedder()) + + db = SessionLocal() + try: + adventure = db.get(models.Adventure, client.adv_id) + settings = db.query(models.Settings).first() + result = asyncio.run( + memorybank.retrieve_memories(adventure, settings, update_stats=False) + ) + assert result["used"] == [] + finally: + db.close() + + +def test_an_unrelated_settings_change_keeps_the_vectors(client): + """Only an embedding-model change may clear the bank — re-embedding costs + an API call per memory.""" + r = client.put("/api/settings", json={"model": "some-other-chat-model"}) + assert r.status_code == 200, r.text + + db = SessionLocal() + try: + rows = memories(db) + assert all(m.embedded for m in rows) + assert all(m.embedding_blob for m in rows) + finally: + db.close() diff --git a/backend/tests/test_memory_retrieval.py b/backend/tests/test_memory_retrieval.py index e24046f..5f0a593 100644 --- a/backend/tests/test_memory_retrieval.py +++ b/backend/tests/test_memory_retrieval.py @@ -26,7 +26,7 @@ import asyncio from datetime import timedelta import pytest -from sqlalchemy import event +from sqlalchemy import event, inspect as sa_inspect from app import memorybank, models from app.database import Base, SessionLocal, engine @@ -206,13 +206,14 @@ def memory_selects(statements): ] -def test_the_json_column_is_never_selected(db, adventure, settings, bank, sql_log): - """`memories.embedding` is dead weight kept only until a follow-up - migration drops it. If anything still reads it, dropping it breaks.""" - retrieve(adventure, settings, StubEmbedder()) - offenders = [s for s in memory_selects(sql_log) if "memories.embedding " in s - or s.rstrip().endswith("memories.embedding")] - assert offenders == [], f"the JSON column was read:\n{offenders[0][:300]}" +def test_the_json_column_is_gone(db): + """`memories.embedding` held the vectors before migration 38 and nothing + read it afterwards; migration 42 dropped it. Bringing it back would restore + 4 MB of dead weight and a second place vectors can be written from — which + is how the model-switch bug happened (test_embedding_model_switch.py).""" + columns = {c["name"] for c in sa_inspect(engine).get_columns("memories")} + assert "embedding" not in columns + assert {"embedding_blob", "embedded"} <= columns def test_the_catalogue_query_carries_no_vectors(db, adventure, settings, bank, sql_log): From be66780a26b2067a06f7479327d4e8484bf355a6 Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Mon, 17 Aug 2026 13:48:23 +0530 Subject: [PATCH 04/10] Size the stress fixture from what production actually holds The old fixture was wrong in both directions at once and happened to land near the right total. Actions were modelled at ~2.1 KB against a real 886 B of text, and adventures at 200 actions against a real 607. Width flattered, length did not, and length is what a page load pays for. Re-sized from the 2026-08-17 measurements: 600 actions, 1700 B of narration alternating with a one-line player input, 232 KB of context_snapshot a row. The page-load shape now reports 606.0 kB against the 589.5 kB measured on production's longest adventure -- 2.8% out, where the old defaults were 28% out on a story a third of the length. Filler text is now generated word by word instead of one sentence repeated. That matters for what comes next: the repeated string compresses 313x and the generated prose 3.7x, so any compression ratio measured against the old fixture would have been fiction, and shrinking context_snapshot is the open question it exists to answer. context_snapshot also gains a flag of its own rather than being hardcoded, and the 74 KB figure in the comment -- inherited from models.py -- is corrected: the real column averages 163 KB a row across the table. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7 --- backend/tools/stress_session.py | 131 ++++++++++++++++++++++++-------- 1 file changed, 99 insertions(+), 32 deletions(-) diff --git a/backend/tools/stress_session.py b/backend/tools/stress_session.py index 999352b..9bb2ce1 100644 --- a/backend/tools/stress_session.py +++ b/backend/tools/stress_session.py @@ -40,9 +40,25 @@ AIDND_STRESS_DATABASE_URL to a **throwaway** Postgres database: The harness writes, so it refuses any target whose database name does not say 'stress' or 'scratch'. Never point it at a database holding real users. -Calibration, against the two figures measured directly on production -(2026-08-16): a 200-action page load reported 426.7 kB here against 423 KB -there, and one turn on a 100-memory bank reported 3,258.7 kB against 3,153 kB. +Calibration. The fixture is sized from production, re-measured 2026-08-17 +against the live Neon database (aggregates only — counts and octet_length +sums, never row contents): + + per action, text 886 B -> --narration-bytes 1700, alternating + with a one-line player input + longest adventure 607 actions -> --actions 600 + context_snapshot 232 KB/row -> --snapshot-bytes 232000 + memory bank, largest 100 memories, 6,144 B a vector + +The previous defaults were wrong in both directions at once and happened to +land near the right total: actions were modelled at ~2.1 KB against a real +886 B, and stories at 200 actions against a real 607. Width was flattering, +length was not, and length is what a page load pays for. + +Filler text is generated word by word rather than repeated. A repeated +sentence compresses about a hundredfold and prose three- or fourfold, so the +old fixture would have made any compression measurement on context_snapshot +meaningless. """ import os @@ -95,33 +111,74 @@ from .dbmeter import Meter, kb EMBEDDING_DIMS = 1536 -# ~74 KB, which is what a real snapshot weighs in production: the assembled -# prompt is nearly all of it. -SNAPSHOT_SYSTEM = "You are a masterful storyteller. " * 400 -SNAPSHOT_STORY = "The corridor narrows and the torchlight gutters. " * 1200 +# ~232 KB, measured on production's largest adventure (2026-08-17). The old +# figure here was 74 KB, taken from the comment in models.py; the real column +# averages 163 KB a row across the whole table and 232 KB on the adventure that +# matters, because the assembled prompt grows with the story behind it. +# +# Built from varied text rather than one sentence repeated. A repeated sentence +# compresses about a hundredfold and real prose three- or fourfold, so a +# fixture made of repeats would make any compression measurement meaningless — +# and shrinking this column is the open question it exists to answer. +SNAPSHOT_SYSTEM = None # set by _build_text() +SNAPSHOT_STORY = None + +_WORDS = ( + "corridor narrows shoulders brush wet stone torchlight gutters draught " + "smells cold iron somewhere ahead water moving count nine paces passage " + "opens chamber ceiling lost dark sound breathing comes back half second " + "late Gwen catches sleeve without word points floor line pale grit laid " + "across threshold deliberate arc quartermaster bandit camp above ford " + "tunnels exchange key lantern rope knife bread rain mud hill road gate " + "watchman silver debt promise fever horse cart river bridge mill barley " + "smoke rafters bench ale ledger seal wax parchment ink candle shutter " + "hinge bolt cellar barrel salt fish nets harbour tide gull mast canvas" +).split() + + +def prose(rng: random.Random, nbytes: int) -> str: + """Filler of about `nbytes`, varied enough to compress like prose. + + Not decoration. A sentence repeated N times compresses roughly a + hundredfold and English roughly three- or fourfold, so a fixture built out + of repeats would report a compression ratio that says nothing about the + real column — and that ratio is the whole question for context_snapshot. + """ + out: list[str] = [] + total = 0 + while total < nbytes: + sentence = " ".join(rng.choice(_WORDS) for _ in range(rng.randint(8, 18))) + chunk = sentence.capitalize() + ". " + out.append(chunk) + total += len(chunk) + return "".join(out)[:nbytes] + -_PARAGRAPH = ( - "The corridor narrows until your shoulders brush wet stone, and the " - "torchlight gutters in a draught that smells of cold iron. Somewhere ahead, " - "water is moving. You count nine paces before the passage opens into a " - "chamber whose ceiling is lost in the dark, and the sound of your own " - "breathing comes back to you a half-second late.\n\n" - "Gwen catches your sleeve without a word and points at the floor, where a " - "line of pale grit has been laid across the threshold in a deliberate arc.\n\n" -) PLAYER_INPUT = "> You crouch and look more closely at the grit on the floor." -# Rebound by main() to --narration-bytes. An AI action's length is what makes a -# page load expensive, and it is the one fixture dimension that cannot be -# guessed from the schema: production averages ~2.1 KB across all actions, -# which is ~4 KB of narration alternating with a one-line player input. -NARRATION = _PARAGRAPH +# All three are bound by _build_text() from the fixture arguments. +NARRATION = None MEMORY_TEXT = ( "You found a bandit camp above the ford and agreed to guide Gwen through " "the tunnels in exchange for the iron key she took from the quartermaster." ) +def _build_text(args, rng: random.Random) -> None: + """Size the three variable-length fixture strings from the arguments. + + Separate from build_fixture so the sizes are decided once, before anything + is written, and so a shape's cost is a function of the flags rather than of + how many rows happened to be generated first. + """ + global NARRATION, SNAPSHOT_SYSTEM, SNAPSHOT_STORY + NARRATION = prose(rng, args.narration_bytes) + # The assembled prompt is a system block and the story so far; the split + # is roughly one to five in production. + SNAPSHOT_SYSTEM = prose(rng, args.snapshot_bytes // 6) + SNAPSHOT_STORY = prose(rng, args.snapshot_bytes - args.snapshot_bytes // 6) + + # --------------------------------------------------------------- fake network @@ -321,8 +378,13 @@ def parse_args(argv=None): p = argparse.ArgumentParser( prog="tools.stress_session", description=__doc__.splitlines()[0] ) - p.add_argument("--actions", type=int, default=200, - help="story actions in the fixture (default: 200)") + # 607 is production's longest adventure as of 2026-08-17, and length is + # the dimension the old default (200) got wrong: real actions are lighter + # than this fixture used to make them, but real stories run three times + # longer, and length is what a page load pays for. + p.add_argument("--actions", type=int, default=600, + help="story actions in the fixture (default: 600, " + "production's longest adventure is 607)") p.add_argument("--memories", type=int, default=100, help="memories, all embedded (default: 100)") # Deliberately not the app's default (80): a measuring instrument should @@ -330,10 +392,17 @@ def parse_args(argv=None): p.add_argument("--capacity", type=int, default=200, help="Settings.memory_bank_capacity; lower it below " "--memories to exercise eviction (default: 200)") - p.add_argument("--narration-bytes", type=int, default=4000, - help="length of an AI action's text; production averages " - "~2.1 KB per action alternating with player input " - "(default: 4000)") + # Production's longest adventure carries 886 B of text per action averaged + # over both kinds. AI actions alternate with a one-line player input, so + # the AI half has to be about twice that. + p.add_argument("--narration-bytes", type=int, default=1700, + help="length of an AI action's text; alternating with a " + "one-line player input this averages ~890 B/action, " + "which is what production measures (default: 1700)") + p.add_argument("--snapshot-bytes", type=int, default=232_000, + help="context_snapshot per action; 232 KB is the average " + "on production's longest adventure, 163 KB is the " + "average across the whole table (default: 232000)") p.add_argument("--shapes", default=",".join(SHAPES), help=f"comma-separated subset of: {', '.join(SHAPES)}") p.add_argument("--repeat", type=int, default=1, @@ -354,11 +423,8 @@ def main(argv=None) -> int: if unknown: sys.exit(f"unknown shape(s): {', '.join(unknown)}") - global NARRATION - repeats = max(1, -(-args.narration_bytes // len(_PARAGRAPH))) - NARRATION = (_PARAGRAPH * repeats)[: args.narration_bytes] - rng = random.Random(args.seed) + _build_text(args, random.Random(args.seed ^ 0x5F5F)) adv_id, user_id = build_fixture(args, rng) install_fakes(user_id, rng) @@ -367,7 +433,8 @@ def main(argv=None) -> int: # bytes would drown everything the shapes report. meter.attach(engine) - print(f"fixture: {args.actions} actions × {args.narration_bytes} B · " + print(f"fixture: {args.actions} actions × {args.narration_bytes} B " + f"(+{args.snapshot_bytes // 1024} kB snapshot, deferred) · " f"{args.memories} memories × {EMBEDDING_DIMS} dims · " f"capacity {args.capacity}") if args.no_embeddings: From 12d57afdac006bb1a8e481c53cce10392d104211 Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Mon, 17 Aug 2026 13:51:10 +0530 Subject: [PATCH 05/10] Put a number on what an endpoint may fetch, not just a column list test_egress.py asserted which columns a statement names, which is the shape both of this project's egress blowouts took. It would all still pass if a response grew tenfold within the columns it is allowed to read -- and a story that keeps getting longer does exactly that. Production's longest adventure is 607 actions where the plan assumed 200. So dbmeter, which was built to be importable from tests and was not yet used by any, now backs four byte ceilings: the page load, the action list, and one action's snapshot fetched on demand. Budgets are per action rather than absolute, so they mean the same thing whatever size the fixture is set to, and generous -- 3 kB against a real 994 B. They are there to catch an order of magnitude, not to freeze a byte count. The fourth test is the one that keeps the other three honest. A ceiling proves nothing unless the thing it excludes would breach it, so it undefers the snapshot on purpose and asserts the same twelve rows cost more than ten times the budget. If the fixture ever shrinks below the point where that holds, that test fails rather than the ceilings quietly passing on nothing. Meter grows detach() and a context manager. A script exits and takes the wrapping with it; a test does not, and one test leaving the shared engine metered would charge bytes to a scope nobody opened. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7 --- backend/tests/test_egress.py | 131 +++++++++++++++++++++++++++++++++-- backend/tools/dbmeter.py | 21 +++++- 2 files changed, 147 insertions(+), 5 deletions(-) diff --git a/backend/tests/test_egress.py b/backend/tests/test_egress.py index ef73e93..1e27a14 100644 --- a/backend/tests/test_egress.py +++ b/backend/tests/test_egress.py @@ -1,9 +1,19 @@ """Guards on how much the database is asked for. -context_snapshot holds the entire assembled prompt for a turn (~74 KB/row in -production, 94% of the database). It used to be pulled for every action on -every adventure load and every turn, to read two tiny things out of it. These -tests fail if that regresses. +context_snapshot holds the entire assembled prompt for a turn — 163 KB a row +averaged over production, 232 KB on the longest adventure, and 89% of the +database. It used to be pulled for every action on every adventure load and +every turn, to read two tiny things out of it. These tests fail if that +regresses. + +Two kinds of guard live here, and both are needed: + +* **column guards** assert which columns a statement names. That is the shape + both of this project's egress blowouts took — one query quietly carrying a + column nobody read. +* **byte ceilings** assert what a request actually costs. Every column guard + would still pass if a response grew tenfold within the columns it is allowed + to read, which is what a story that keeps getting longer does. python -m pytest tests/test_egress.py -v """ @@ -16,15 +26,19 @@ os.environ["AIDND_DB_PATH"] = _tmp.name os.environ.pop("AIDND_DATABASE_URL", None) os.environ.pop("DATABASE_URL", None) +import json + import pytest from fastapi import Depends from fastapi.testclient import TestClient from sqlalchemy import event, text +from sqlalchemy.orm import undefer from app import auth, limits, migrations, models from app.context import history from app.database import Base, SessionLocal, engine, get_db from app.main import app +from tools import dbmeter # A stand-in for the real thing: the assembled prompt, which is what makes the # column enormous, plus the small world_state slice the UI actually needs. @@ -250,3 +264,112 @@ def test_backfill_leaves_actions_without_world_state_alone(client): assert all(a.world_delta is None for a in db.query(models.Action).all()) finally: db.close() + + +# ---------------------------------------------------------------- byte ceilings +# +# The tests above assert which *columns* a statement names, which is the shape +# both of this project's egress blowouts took. They would all still pass if a +# response quietly grew tenfold within the columns it is allowed to read — and +# a story that keeps getting longer does exactly that. These put a number on it. +# +# Ceilings are per action rather than absolute, so they mean the same thing +# whatever size the fixture is set to, and they are generous: the point is to +# catch a tenfold regression, not to freeze today's byte count. + +ACTIONS_IN_FIXTURE = 12 + +# 3 kB an action against a real 994 B, measured on production 2026-08-17. +# Anything that pulls a deferred column blows past this by two orders of +# magnitude — see test_the_ceiling_discriminates below. +PAGE_LOAD_BYTES_PER_ACTION = 3_000 + + +@pytest.fixture() +def meter(): + """A byte meter on the shared engine, removed again afterwards. + + Requested *after* `client` in a test's arguments so that building the + fixture — a write path nobody plays — is not charged to any scope. + """ + m = dbmeter.Meter() + m.attach(engine) + try: + yield m + finally: + m.detach() + + +def fetched(meter) -> int: + return meter.scopes[-1].total.fetched + + +def test_page_load_stays_under_its_byte_ceiling(client, meter): + with meter.scope("page load"): + r = client.get(f"/api/adventures/{client.adv_id}") + assert r.status_code == 200 + + budget = ACTIONS_IN_FIXTURE * PAGE_LOAD_BYTES_PER_ACTION + assert fetched(meter) < budget, ( + f"page load fetched {fetched(meter):,} B for {ACTIONS_IN_FIXTURE} " + f"actions, over the {budget:,} B budget" + ) + + +def test_the_action_list_stays_under_its_byte_ceiling(client, meter): + with meter.scope("action list"): + r = client.get(f"/api/adventures/{client.adv_id}/actions") + assert r.status_code == 200 + + budget = ACTIONS_IN_FIXTURE * PAGE_LOAD_BYTES_PER_ACTION + assert fetched(meter) < budget, ( + f"the action list fetched {fetched(meter):,} B, over {budget:,} B" + ) + + +def test_reading_one_action_does_not_cost_the_whole_story(client, meter): + """The snapshot is reachable on demand, and that request should pay for + one row's worth — not the adventure's.""" + db = SessionLocal() + try: + action_id = db.query(models.Action.id).order_by(models.Action.id).first()[0] + finally: + db.close() + + with meter.scope("one snapshot"): + r = client.get(f"/api/adventures/{client.adv_id}/actions/{action_id}/context") + assert r.status_code == 200, r.text + + one_snapshot = len(json.dumps(BIG_SNAPSHOT)) + assert fetched(meter) < one_snapshot * 2, ( + f"fetching one action's snapshot cost {fetched(meter):,} B; one " + f"snapshot is {one_snapshot:,} B" + ) + + +def test_the_ceiling_discriminates(client, meter): + """A ceiling is only worth having if the thing it excludes would breach it. + + This is the regression the byte tests exist to catch, performed on purpose: + undefer the snapshot and the same twelve rows cost two orders of magnitude + more. If this ever stops exceeding the budget, the fixture has gone too + small for the tests above to mean anything. + """ + budget = ACTIONS_IN_FIXTURE * PAGE_LOAD_BYTES_PER_ACTION + db = SessionLocal() + try: + with meter.scope("undeferred"): + rows = ( + db.query(models.Action) + .options(undefer(models.Action.context_snapshot)) + .all() + ) + assert len(rows) == ACTIONS_IN_FIXTURE + finally: + db.close() + + assert fetched(meter) > budget * 10, ( + "undeferring the snapshot cost only " + f"{fetched(meter):,} B — the fixture is too small for the byte " + "ceilings above to catch anything" + ) diff --git a/backend/tools/dbmeter.py b/backend/tools/dbmeter.py index f7618d4..45a5ded 100644 --- a/backend/tools/dbmeter.py +++ b/backend/tools/dbmeter.py @@ -174,7 +174,26 @@ class Meter: metered_creator._dbmeter = self pool._creator = metered_creator - self._attached_pools.append(pool) + self._attached_pools.append((engine, pool, creator)) + + def detach(self) -> None: + """Put every metered engine back as it was. + + A script exits and takes the wrapping with it; a test does not, and one + test leaving the shared engine metered would go on charging bytes to a + scope nobody opened. Pooled connections are dropped again on the way + out for the same reason attach drops them on the way in. + """ + while self._attached_pools: + engine, pool, creator = self._attached_pools.pop() + pool._creator = creator + engine.dispose() + + def __enter__(self) -> "Meter": + return self + + def __exit__(self, *exc) -> None: + self.detach() # ------------------------------------------------------------- reporting From a6cb49293c0048e56af8ea8e650126f6ff4194b5 Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Mon, 17 Aug 2026 13:56:07 +0530 Subject: [PATCH 06/10] Name the columns a list response carries deferred=True keeps the four heavy Action columns out of a bulk read, but it makes narrowness the thing a future column has to remember to ask for -- and both egress blowouts this project has had were a column nobody remembered. Listing what each list response renders inverts the default: a new column costs nothing on these paths until someone adds it to the tuple. The adventures index was not merely a future risk. It loaded whole Adventure entities to render a title, a stamp and a snippet, and an Adventure carries script_state, world_state, placeholders, story_summary, memory, authors_note and ai_instructions -- ~15 kB a row in production, none of it on that screen, all of it fetched once per adventure on every index load. Measured on six adventures with 78 kB of body each: 469.7 kB entity-loaded against 318 B projected. The memories drawer stops walking adventure.memories. The walk is what retrieval used to do and the reason a turn cost megabytes; a relationship load takes whole entities, so it picks up whatever the model happens to grow. Nothing changes today -- embedding_blob is already deferred -- which is the point. world_delta stays on the action list because ActionOut.world_changes is computed from it. Leaving it off would not save the bytes, it would spend them one row at a time as a lazy load. Two tests cover the index: one asserts the listing query names none of the body columns, one puts a byte ceiling on six adventures carrying 80 kB apiece. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7 --- backend/app/routers/adventures.py | 83 +++++++++++++++++++++++++++---- backend/tests/test_egress.py | 68 +++++++++++++++++++++++++ 2 files changed, 140 insertions(+), 11 deletions(-) diff --git a/backend/app/routers/adventures.py b/backend/app/routers/adventures.py index e188f37..2d0632a 100644 --- a/backend/app/routers/adventures.py +++ b/backend/app/routers/adventures.py @@ -6,7 +6,7 @@ import threading from fastapi import APIRouter, Body, Depends, HTTPException, Request from fastapi.responses import StreamingResponse from sqlalchemy import func -from sqlalchemy.orm import Session, undefer +from sqlalchemy.orm import Session, load_only, undefer from .. import auth, images, limits, memorybank, models, schemas, worldstate from ..context import build_context @@ -20,6 +20,44 @@ router = APIRouter(prefix="/api/adventures", tags=["adventures"]) CurrentUser = Depends(auth.get_current_user) +# Exactly what schemas.ActionOut renders, named rather than implied. +# +# `deferred=True` in models.py already keeps the four heavy columns out of a +# bulk read, but it makes narrowness the default that a *future* column has to +# remember to ask for — and both egress blowouts this project has had were a +# column nobody remembered. Listing what a list response carries inverts that: +# a new column costs nothing here until someone adds it to this tuple. +# +# `world_delta` is on the list because ActionOut.world_changes is computed from +# it. Leaving it off would not save the bytes, it would spend them one row at a +# time as a lazy load, which is worse. +ACTION_LIST_COLUMNS = ( + models.Action.adventure_id, + models.Action.index, + models.Action.type, + models.Action.text, + models.Action.reasoning, + models.Action.world_delta, + models.Action.variant_count, + models.Action.variant_index, + models.Action.created_at, +) + +# Exactly what schemas.MemoryOut renders. `embedded` is a real column and is on +# the list; the vector it describes is not, and must never be. +MEMORY_LIST_COLUMNS = ( + models.Memory.adventure_id, + models.Memory.text, + models.Memory.pinned, + models.Memory.forgotten, + models.Memory.embedded, + models.Memory.use_count, + models.Memory.last_used_at, + models.Memory.source_start, + models.Memory.source_end, + models.Memory.created_at, +) + def get_adventure_or_404( adventure_id: int, db: Session, user: models.User @@ -85,9 +123,18 @@ def _latest_narration(db: Session, adventure_ids: list[int]) -> dict[int, str]: @router.get("", response_model=list[schemas.AdventureListItem]) def list_adventures(db: Session = Depends(get_db), user: models.User = CurrentUser): + # Four columns of Adventure, named, rather than the entity. The entity is + # sixteen columns wide and carries script_state, world_state, placeholders, + # story_summary, memory, authors_note and ai_instructions — ~15 kB a row in + # production, none of it on this screen, all of it fetched once per + # adventure every time the index loads. Naming the columns also means the + # next wide column added to Adventure has to opt *in* to being listed here. rows = ( db.query( - models.Adventure, + models.Adventure.id, + models.Adventure.scenario_id, + models.Adventure.title, + models.Adventure.updated_at, func.count(models.Action.id), models.Scenario.title, models.Scenario.image, @@ -112,22 +159,23 @@ def list_adventures(db: Session = Depends(get_db), user: models.User = CurrentUs .order_by(models.Adventure.updated_at.desc()) .all() ) - narration = _latest_narration(db, [adv.id for adv, *_ in rows]) + narration = _latest_narration(db, [row[0] for row in rows]) return [ schemas.AdventureListItem( - id=adv.id, - scenario_id=adv.scenario_id, + id=adv_id, + scenario_id=scenario_id, scenario_title=scenario_title, - title=adv.title, - updated_at=adv.updated_at, + title=title, + updated_at=updated_at, action_count=count, - snippet=_snippet(narration.get(adv.id, "")), + snippet=_snippet(narration.get(adv_id, "")), # The art belongs to the scenario, so the cache-busting stamp is the # scenario's updated_at, not the adventure's. - image_url=images.public_url(adv.scenario_id, image or "", scenario_updated), + image_url=images.public_url(scenario_id, image or "", scenario_updated), icon=icon or "", ) - for adv, count, scenario_title, image, icon, scenario_updated in rows + for (adv_id, scenario_id, title, updated_at, count, + scenario_title, image, icon, scenario_updated) in rows ] @@ -1455,7 +1503,19 @@ def action_context( def list_memories( adventure_id: int, db: Session = Depends(get_db), user: models.User = CurrentUser ): - return get_adventure_or_404(adventure_id, db, user).memories + get_adventure_or_404(adventure_id, db, user) + # A query naming its columns, not a walk of `adventure.memories`. The walk + # is what retrieval used to do, and it is the reason a turn cost megabytes: + # a relationship load takes whole entities, so it picks up whatever the + # model happens to carry. `embedding_blob` is deferred and so would stay + # out today — this is about the next wide column, not that one. + return ( + db.query(models.Memory) + .options(load_only(*MEMORY_LIST_COLUMNS)) + .filter(models.Memory.adventure_id == adventure_id) + .order_by(models.Memory.id) + .all() + ) @router.post("/{adventure_id}/memories", response_model=schemas.MemoryOut, status_code=201) @@ -1522,6 +1582,7 @@ def list_actions( get_adventure_or_404(adventure_id, db, user) return ( db.query(models.Action) + .options(load_only(*ACTION_LIST_COLUMNS)) .filter(models.Action.adventure_id == adventure_id) .order_by(models.Action.index) .all() diff --git a/backend/tests/test_egress.py b/backend/tests/test_egress.py index 1e27a14..5159a47 100644 --- a/backend/tests/test_egress.py +++ b/backend/tests/test_egress.py @@ -347,6 +347,74 @@ def test_reading_one_action_does_not_cost_the_whole_story(client, meter): ) +def _fat_adventures(user_id: int, count: int = 5, body: int = 20_000) -> None: + """Adventures whose bodies are heavy and whose index cards are not. + + script_state, world_state and story_summary belong to the play screen. The + index shows a title, a stamp and a snippet, and used to load all of it. + """ + db = SessionLocal() + try: + for i in range(count): + db.add(models.Adventure( + user_id=user_id, + title=f"Adventure {i}", + script_state={"log": "s" * body}, + world_state={"player": {"notes": "w" * body}}, + story_summary="y" * body, + memory="m" * body, + )) + db.commit() + finally: + db.close() + + +def test_the_index_does_not_read_the_adventure_body(client, sql_log): + db = SessionLocal() + try: + user_id = db.query(models.User.id).first()[0] + finally: + db.close() + _fat_adventures(user_id) + + r = client.get("/api/adventures") + assert r.status_code == 200 + assert len(r.json()) == 6 # the fixture's one, plus five + + listing = [ + s for s in sql_log + if "FROM adventures" in s and s.lstrip().upper().startswith("SELECT") + ] + assert listing, "expected a listing query" + for column in ("script_state", "world_state", "story_summary", "memory", + "authors_note", "ai_instructions", "placeholders"): + assert not any(column in s for s in listing), ( + f"the index read adventures.{column}, which nothing on that " + f"screen displays" + ) + + +def test_the_index_stays_under_its_byte_ceiling(client, meter): + db = SessionLocal() + try: + user_id = db.query(models.User.id).first()[0] + finally: + db.close() + _fat_adventures(user_id) + + with meter.scope("index"): + r = client.get("/api/adventures") + assert r.status_code == 200 + + # Six adventures carrying 80 kB of body each. A card is a title, a stamp + # and a 220-character snippet; 4 kB apiece is already generous. + budget = 6 * 4_000 + assert fetched(meter) < budget, ( + f"the index fetched {fetched(meter):,} B for six adventures, over " + f"{budget:,} B — it is reading the bodies again" + ) + + def test_the_ceiling_discriminates(client, meter): """A ceiling is only worth having if the thing it excludes would breach it. From ae6e5af6c7cc2bdebb17f2269939a9a6a10a41b5 Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Mon, 17 Aug 2026 14:08:23 +0530 Subject: [PATCH 07/10] Store context_snapshot compressed One column is 89% of the database and the free tier allows 512 MB. Reads were already solved -- the column is deferred, so a page load never touches it and one screen fetches one row at a time -- but nothing had costed storage, and storage is the constraint with a cliff: 99.6 MB used, ~94 kB of disk per action, so the ceiling arrives around 5,400 actions and 944 are stored. Postgres already compresses it and only gets 1.7x. pglz is tuned for fast decompression of data a query might filter on, and nothing has ever filtered on an assembled prompt -- it is written once and read whole, rarely, by the Insights viewer. zlib gets 3.5x on the same text for a decompress on a request that already made an LLM call. Done as a TypeDecorator rather than a second column, so every call site still writes a dict and reads a dict back, and deferred/undefer/load_only keep naming the same attribute. Only the storage format moves. Migrations 43-45: add the bytea, convert into it, drop the original, rename. The backfill is the one destructive step in the file -- 44 removes the only other copy -- so it decompresses every row and compares it against what went in, and a row that fails aborts the run. The whole loop is one transaction, so an abort rolls the DROP back and the prompts are still there. Verified on real Postgres, replaying 43-45 from a pre-43 schema on a throwaway Neon database: 720,864 B of JSON became 204,293 B of bytea, 3.53x, the column came out named context_snapshot, every snapshot compared equal and the one NULL stayed NULL. Postgres does not return the disk by itself: DROP COLUMN only marks the column gone and the backfill leaves a dead tuple per row, so the table peaks near twice its size before settling. The deploy needs one VACUUM FULL to collect it; the migration comment says so. The egress fixture's snapshots are prose now rather than "x" * 20_000, and the prose generator moved to tools/fakeprose.py so the harness and the tests share one definition. A repeated character compresses a thousandfold: against the old fixture a compressed column looked free and the byte ceilings would have been guarding nothing. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7 --- backend/app/compression.py | 70 ++++++ backend/app/migrations.py | 87 ++++++- backend/app/models.py | 18 +- backend/tests/test_egress.py | 53 ++++- backend/tests/test_snapshot_compression.py | 260 +++++++++++++++++++++ backend/tools/fakeprose.py | 38 +++ backend/tools/stress_session.py | 32 +-- 7 files changed, 512 insertions(+), 46 deletions(-) create mode 100644 backend/app/compression.py create mode 100644 backend/tests/test_snapshot_compression.py create mode 100644 backend/tools/fakeprose.py diff --git a/backend/app/compression.py b/backend/app/compression.py new file mode 100644 index 0000000..e8b24ad --- /dev/null +++ b/backend/app/compression.py @@ -0,0 +1,70 @@ +"""Storing a JSON column compressed. + +`actions.context_snapshot` holds the entire assembled prompt for a turn. It is +89% of the database — 150.8 MB of JSON across 944 actions on production, and +232 KB a row on the longest adventure — and the free tier this deploys to +allows 512 MB. Reads are not the problem: the column is deferred, so a page +load never touches it and exactly one endpoint fetches one row of it at a +time. Storage is the problem, and storage has a cliff. + +Postgres already compresses it. TOAST brings 150.8 MB down to ~89 MB, a factor +of 1.7 — pglz is chosen for decompression speed on data a query might filter +on, which this never is. Nothing filters on a prompt; it is written once and +read whole, occasionally, by one screen. zlib at the application layer gets +three to four times on the same text, and the cost is a decompress on a +request that already costs an LLM call. + +Doing it as a TypeDecorator rather than a second column keeps every call site +writing `action.context_snapshot = {...}` and reading a dict back, and keeps +`deferred=True`, `undefer()` and `load_only()` naming the same attribute they +named before. The storage format changes; nothing else does. + +Level 6 is zlib's default and the knee of the curve here: 9 spends noticeably +more CPU on prompt text for about a percent more space. +""" +from __future__ import annotations + +import json +import zlib + +from sqlalchemy import LargeBinary +from sqlalchemy.types import TypeDecorator + +LEVEL = 6 + + +def pack(value) -> bytes: + """A JSON-able value as compressed UTF-8.""" + raw = json.dumps(value, separators=(",", ":"), default=str).encode("utf-8") + return zlib.compress(raw, LEVEL) + + +def unpack(blob: bytes) -> object: + """The value `pack` was given.""" + return json.loads(zlib.decompress(bytes(blob)).decode("utf-8")) + + +class CompressedJSON(TypeDecorator): + """A JSON column stored as zlib-compressed UTF-8 in a BLOB/BYTEA. + + `cache_ok = True`: the type carries no per-instance configuration, so + SQLAlchemy may reuse a compiled statement across instances of it. + """ + + impl = LargeBinary + cache_ok = True + + def process_bind_param(self, value, dialect): + return None if value is None else pack(value) + + def process_result_value(self, value, dialect): + # Tolerate a row the backfill has not reached yet, or one written + # before the conversion: a snapshot that cannot be read back is worth + # less than the screen that shows it, and never worth a 500 on the + # turn that happens to load it. + if value is None: + return None + try: + return unpack(value) + except (zlib.error, UnicodeDecodeError, ValueError): + return None diff --git a/backend/app/migrations.py b/backend/app/migrations.py index 413cc11..8134cb0 100644 --- a/backend/app/migrations.py +++ b/backend/app/migrations.py @@ -22,7 +22,7 @@ import json from sqlalchemy import inspect, text from sqlalchemy.engine import Engine -from . import vectors +from . import compression, vectors from .database import Base # (version, SQL to run when upgrading past it) — append only, never reorder. @@ -150,6 +150,34 @@ MIGRATIONS: list[tuple[int, str | dict[str, str]]] = [ # now 4 MB of a 99.6 MB database holding nothing anyone reads. DROP COLUMN # is spelled the same on both dialects — SQLite has had it since 3.35. (42, "ALTER TABLE memories DROP COLUMN embedding"), + # context_snapshot, compressed. 89% of the database is one column holding + # assembled prompts nobody filters on and one screen reads, one row at a + # time; Postgres already TOASTs it, but pglz only manages 1.7x and zlib + # gets three to four on the same text. Reads were fixed by deferring it — + # this is about the 512 MB the free tier allows. + # + # Three steps because a column cannot portably change type in place: add + # the new one, convert into it (_backfill_context_snapshot, which verifies + # every row round-trips before the old column goes), then swap the names so + # the model keeps calling it context_snapshot. + # + # **Postgres does not hand the disk back on its own.** DROP COLUMN only + # marks the column dropped, and the backfill's UPDATE leaves a dead tuple + # per row, so the table gets *bigger* before it gets smaller: peak is + # roughly twice the starting size while both columns are live. Plain + # autovacuum makes that space reusable but does not shrink the files. The + # deploy that ships this should follow it with, once: + # + # VACUUM FULL actions; + # + # which needs exclusive access and free space equal to the finished table. + # On the 2026-08-17 figures that is 99.6 MB peaking near 200, settling at + # about 53 once vacuumed, against a 512 MB tier. Skipping the vacuum is + # safe and simply leaves the win unrealised. + (43, {"sqlite": "ALTER TABLE actions ADD COLUMN context_snapshot_z BLOB", + "default": "ALTER TABLE actions ADD COLUMN context_snapshot_z BYTEA"}), + (44, "ALTER TABLE actions DROP COLUMN context_snapshot"), + (45, "ALTER TABLE actions RENAME COLUMN context_snapshot_z TO context_snapshot"), ] LATEST_VERSION = max((v for v, _ in MIGRATIONS), default=1) @@ -158,6 +186,12 @@ LATEST_VERSION = max((v for v, _ in MIGRATIONS), default=1) WORLD_DELTA_VERSION = 36 VARIANT_COUNT_VERSION = 37 EMBEDDING_BLOB_VERSION = 38 +SNAPSHOT_COMPRESS_VERSION = 43 + +# Snapshots converted per round trip. Deliberately far smaller than +# BACKFILL_BATCH: a vector is 6 KB and a snapshot is 232 KB, so 200 of these +# would be 46 MB held at once. +SNAPSHOT_BATCH = 50 # Vectors converted per round trip. Small enough that the backfill never holds # more than a few megabytes, large enough that it isn't a query per row. @@ -257,6 +291,52 @@ def _backfill_embedding_blob(conn) -> None: last_id = rows[-1][0] +def _backfill_context_snapshot(conn) -> None: + """Compress actions.context_snapshot into actions.context_snapshot_z. + + Runs between migration 43 and 44, which is the only window where both + columns exist. Migration 44 drops the original, so unlike every other + backfill here this one is destructive if it is wrong — and every row it + converts is somebody's game. So each row is decompressed again and + compared against what went in before it counts as converted, and a row + that fails to round-trip aborts the whole run rather than being skipped: + the transaction rolls back, the DROP never happens, and the prompts are + still there to try again. + + Reads the JSON the same defensive way as the vector backfill — SQLite + hands back a raw string, psycopg has already parsed it. + """ + last_id = 0 + while True: + rows = conn.execute( + text(""" + SELECT id, context_snapshot FROM actions + WHERE context_snapshot IS NOT NULL + AND context_snapshot_z IS NULL AND id > :last + ORDER BY id LIMIT :batch + """), + {"last": last_id, "batch": SNAPSHOT_BATCH}, + ).all() + if not rows: + return + for row_id, stored in rows: + value = json.loads(stored) if isinstance(stored, str) else stored + if value is None: + continue + packed = compression.pack(value) + if compression.unpack(packed) != value: + raise RuntimeError( + f"context_snapshot for action {row_id} did not survive a " + "compress/decompress round trip; refusing to drop the " + "original column" + ) + conn.execute( + text("UPDATE actions SET context_snapshot_z = :z WHERE id = :id"), + {"z": packed, "id": row_id}, + ) + last_id = rows[-1][0] + + def _get_version(conn) -> int: if conn.dialect.name == "sqlite": return conn.execute(text("PRAGMA user_version")).scalar() or 1 @@ -301,6 +381,11 @@ def bootstrap(engine: Engine) -> None: _backfill_variant_count(conn) if version == EMBEDDING_BLOB_VERSION: _backfill_embedding_blob(conn) + # Must land between 43 (add the column) and 44 (drop the old + # one). The loop is one transaction, so if this raises, the + # DROP rolls back with it and the prompts are still there. + if version == SNAPSHOT_COMPRESS_VERSION: + _backfill_context_snapshot(conn) current = version _set_version(conn, current) _encrypt_plaintext_api_keys(conn) diff --git a/backend/app/models.py b/backend/app/models.py index 625ed7a..9938980 100644 --- a/backend/app/models.py +++ b/backend/app/models.py @@ -6,6 +6,7 @@ from sqlalchemy import ( ) from sqlalchemy.orm import Mapped, mapped_column, relationship +from .compression import CompressedJSON from .database import Base @@ -220,12 +221,19 @@ class Action(Base): # Reasoning-model "thinking" that preceded the text (AI actions only). reasoning: Mapped[str | None] = mapped_column(Text, nullable=True) # The full assembled prompt for this turn, for the Insights viewer. By far - # the biggest column in the database (~74 KB/row in production), and needed - # by exactly one endpoint, one action at a time — so it is deferred: never - # loaded unless something actually touches the attribute. Bulk readers must - # NOT touch it; that is what `world_delta` below exists for. + # the biggest column in the database — 163 KB a row averaged over + # production and 232 KB on the longest adventure, 89% of everything stored + # — and needed by exactly one endpoint, one action at a time. + # + # Two separate defences, because it is expensive in two separate ways. + # `deferred=True` is the read defence: never loaded unless something + # touches the attribute, so a page load pays nothing for it. Bulk readers + # must NOT touch it; that is what `world_delta` below exists for. + # CompressedJSON is the *storage* defence: this is the column that decides + # when the free tier's 512 MB runs out. Still a dict either way — see + # compression.py. context_snapshot: Mapped[dict | None] = mapped_column( - JSON, nullable=True, deferred=True + CompressedJSON, nullable=True, deferred=True ) # The small slice of the snapshot that IS needed in bulk: this turn's RPG # state changes, for the inline chips under an AI message (world_changes) diff --git a/backend/tests/test_egress.py b/backend/tests/test_egress.py index 5159a47..262a053 100644 --- a/backend/tests/test_egress.py +++ b/backend/tests/test_egress.py @@ -27,6 +27,7 @@ os.environ.pop("AIDND_DATABASE_URL", None) os.environ.pop("DATABASE_URL", None) import json +import random import pytest from fastapi import Depends @@ -39,12 +40,20 @@ from app.context import history from app.database import Base, SessionLocal, engine, get_db from app.main import app from tools import dbmeter +from tools.fakeprose import prose # A stand-in for the real thing: the assembled prompt, which is what makes the # column enormous, plus the small world_state slice the UI actually needs. +# +# Varied text, not `"x" * 20_000`. The column is stored compressed now +# (migration 43), and a repeated character compresses about a thousandfold — +# which would make the byte ceilings below pass against a fixture that costs +# nothing, testing nothing. Prose-shaped filler compresses like the prompts +# this stands in for. +_SNAPSHOT_RNG = random.Random(20_260_817) BIG_SNAPSHOT = { - "system": "x" * 20_000, - "story": "y" * 40_000, + "system": prose(_SNAPSHOT_RNG, 20_000), + "story": prose(_SNAPSHOT_RNG, 40_000), "world_state": { "delta": {"player.hp": -15}, "report": {"applied": [{"path": "player.hp", "old": 100, "new": 85}]}, @@ -204,16 +213,37 @@ def test_snapshot_is_still_reachable_on_demand(client): action_id = r.json()["actions"][0]["id"] r = client.get(f"/api/adventures/{client.adv_id}/actions/{action_id}/context") assert r.status_code == 200, r.text - assert r.json()["system"] == "x" * 20_000 + # Round-tripped through zlib and back to a dict, byte for byte. + assert r.json()["system"] == BIG_SNAPSHOT["system"] + assert r.json()["story"] == BIG_SNAPSHOT["story"] # ------------------------------------------------------------------ backfill +def as_json_snapshot_column(db) -> None: + """Put actions.context_snapshot back as JSON, the way it was before 43. + + Migration 36 lifts world_delta out of the snapshot with SQL JSON + functions, so it can only run while the column still *is* JSON. In a real + upgrade it always is — 36 runs seven migrations before 43 compresses the + column into a BLOB — but `create_all` builds today's schema, so a test + calling that backfill has to rebuild the schema it was written against. + """ + db.execute(text("ALTER TABLE actions DROP COLUMN context_snapshot")) + db.execute(text("ALTER TABLE actions ADD COLUMN context_snapshot JSON")) + db.execute( + text("UPDATE actions SET context_snapshot = :snapshot"), + {"snapshot": json.dumps(BIG_SNAPSHOT)}, + ) + db.commit() + + def test_backfill_populates_world_delta_from_existing_snapshots(client): """Migration 36 lifts the slice out server-side, without reading the snapshots into Python.""" db = SessionLocal() try: + as_json_snapshot_column(db) db.execute(text("UPDATE actions SET world_delta = NULL")) db.commit() assert db.query(models.Action).filter(models.Action.world_delta.isnot(None)).count() == 0 @@ -419,9 +449,14 @@ def test_the_ceiling_discriminates(client, meter): """A ceiling is only worth having if the thing it excludes would breach it. This is the regression the byte tests exist to catch, performed on purpose: - undefer the snapshot and the same twelve rows cost two orders of magnitude - more. If this ever stops exceeding the budget, the fixture has gone too - small for the tests above to mean anything. + undefer the snapshot and the same twelve rows cost several times the whole + budget. If this ever stops exceeding it, the fixture has gone too small for + the tests above to mean anything. + + The margin used to be a hundredfold and is now about six. That is not the + guard weakening — it is migration 43 compressing the column, and the + fixture text being prose-shaped so it compresses like a real prompt rather + than like a repeated character. """ budget = ACTIONS_IN_FIXTURE * PAGE_LOAD_BYTES_PER_ACTION db = SessionLocal() @@ -436,8 +471,8 @@ def test_the_ceiling_discriminates(client, meter): finally: db.close() - assert fetched(meter) > budget * 10, ( + assert fetched(meter) > budget * 3, ( "undeferring the snapshot cost only " - f"{fetched(meter):,} B — the fixture is too small for the byte " - "ceilings above to catch anything" + f"{fetched(meter):,} B against a {budget:,} B budget — the fixture is " + "too small for the byte ceilings above to catch anything" ) diff --git a/backend/tests/test_snapshot_compression.py b/backend/tests/test_snapshot_compression.py new file mode 100644 index 0000000..0cde792 --- /dev/null +++ b/backend/tests/test_snapshot_compression.py @@ -0,0 +1,260 @@ +"""context_snapshot, stored compressed. + +The column is 89% of the database and the free tier allows 512 MB. Reads were +solved by deferring it; this is about the storage ceiling. Postgres already +TOASTs it and only gets 1.7x, because pglz is tuned for fast decompression of +data a query might filter on — and nothing ever filters on an assembled +prompt. + +Three things have to hold, and only the first is obvious: + +* what goes in comes back out, exactly, including a snapshot written before + the conversion and one that is NULL; +* the model still hands callers a dict, so no call site changes; +* migration 44 drops the original column, so the backfill is the one + destructive step in this file — it must convert every row or abort. + + python -m pytest tests/test_snapshot_compression.py -v +""" +import json +import os +import random +import tempfile +import zlib + +_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False) +_tmp.close() +os.environ["AIDND_DB_PATH"] = _tmp.name +os.environ.pop("AIDND_DATABASE_URL", None) +os.environ.pop("DATABASE_URL", None) + +import pytest +from sqlalchemy import text + +from app import compression, migrations, models +from app.database import Base, SessionLocal, engine +from tools.fakeprose import prose + + +def snapshot(seed: int, nbytes: int = 60_000) -> dict: + rng = random.Random(seed) + return { + "system": prose(rng, nbytes // 3), + "story": prose(rng, nbytes - nbytes // 3), + "world_state": {"delta": {"player.hp": -3}}, + } + + +@pytest.fixture() +def db(): + Base.metadata.create_all(bind=engine) + session = SessionLocal() + try: + yield session + finally: + session.close() + Base.metadata.drop_all(bind=engine) + + +@pytest.fixture() +def adventure(db): + user = models.User(is_guest=False, email="snap@example.com") + db.add(user) + db.flush() + adv = models.Adventure(user_id=user.id, title="Cave", script_state={}) + db.add(adv) + db.commit() + return adv + + +# ------------------------------------------------------------------- packing + +def test_pack_round_trips_exactly(): + value = snapshot(1) + assert compression.unpack(compression.pack(value)) == value + + +def test_pack_handles_the_awkward_values(): + for value in ({}, {"a": None}, {"nested": {"deep": [1, 2, {"x": "é"}]}}): + assert compression.unpack(compression.pack(value)) == value + + +def test_pack_actually_shrinks_prose(): + """The whole justification. If this ratio collapses, the migration is + spending CPU for nothing.""" + value = snapshot(2, 200_000) + raw = json.dumps(value, separators=(",", ":")).encode() + packed = compression.pack(value) + assert len(packed) < len(raw) / 3, ( + f"{len(raw):,} B compressed to {len(packed):,} B — under 3x" + ) + + +def test_unpack_rejects_nothing_it_wrote(): + packed = compression.pack({"a": "b"}) + assert compression.unpack(bytearray(packed)) == {"a": "b"} + assert compression.unpack(memoryview(bytes(packed))) == {"a": "b"} + + +# --------------------------------------------------------------- the column + +def test_the_column_stores_bytes_and_returns_a_dict(db, adventure): + value = snapshot(3) + action = models.Action( + adventure_id=adventure.id, index=0, type="ai", text="t", + context_snapshot=value, + ) + db.add(action) + db.commit() + db.expire_all() + + assert action.context_snapshot == value + + stored = db.execute( + text("SELECT context_snapshot FROM actions WHERE id = :id"), + {"id": action.id}, + ).scalar() + assert isinstance(stored, (bytes, bytearray, memoryview)) + assert json.loads(zlib.decompress(bytes(stored))) == value + + +def test_the_column_is_smaller_than_the_json_it_holds(db, adventure): + value = snapshot(4, 200_000) + action = models.Action( + adventure_id=adventure.id, index=0, type="ai", text="t", + context_snapshot=value, + ) + db.add(action) + db.commit() + + stored = db.execute( + text("SELECT length(context_snapshot) FROM actions WHERE id = :id"), + {"id": action.id}, + ).scalar() + assert stored < len(json.dumps(value)) / 3 + + +def test_null_stays_null(db, adventure): + action = models.Action( + adventure_id=adventure.id, index=0, type="do", text="t", + context_snapshot=None, + ) + db.add(action) + db.commit() + db.expire_all() + assert action.context_snapshot is None + + +def test_an_unreadable_snapshot_reads_as_none_rather_than_raising(db, adventure): + """One corrupt row must not 500 the turn that happens to load it. The + snapshot is a debugging view; the story is the thing that matters.""" + action = models.Action( + adventure_id=adventure.id, index=0, type="ai", text="t", + context_snapshot={"a": "b"}, + ) + db.add(action) + db.commit() + db.execute( + text("UPDATE actions SET context_snapshot = :junk WHERE id = :id"), + {"junk": b"not zlib at all", "id": action.id}, + ) + db.commit() + db.expire_all() + + assert db.get(models.Action, action.id).context_snapshot is None + + +# ------------------------------------------------------- the upgrade in full + +def as_json_column(db, rows: dict[int, dict]) -> None: + """The pre-43 schema: context_snapshot as a JSON column, populated.""" + db.execute(text("ALTER TABLE actions DROP COLUMN context_snapshot")) + db.execute(text("ALTER TABLE actions ADD COLUMN context_snapshot JSON")) + for action_id, value in rows.items(): + db.execute( + text("UPDATE actions SET context_snapshot = :v WHERE id = :id"), + {"v": json.dumps(value), "id": action_id}, + ) + db.commit() + + +def seed_pre_43(db, adventure, count: int = 4) -> dict[int, dict]: + ids = [] + for i in range(count): + action = models.Action( + adventure_id=adventure.id, index=i, type="ai", text=f"t{i}" + ) + db.add(action) + db.flush() + ids.append(action.id) + # One action with no snapshot at all, which must survive as NULL. + plain = models.Action( + adventure_id=adventure.id, index=count, type="do", text="look" + ) + db.add(plain) + db.commit() + + expected = {action_id: snapshot(action_id) for action_id in ids} + as_json_column(db, expected) + db.execute(text(f"PRAGMA user_version = {migrations.SNAPSHOT_COMPRESS_VERSION - 1}")) + db.commit() + return expected + + +def test_bootstrap_converts_every_snapshot(db, adventure): + expected = seed_pre_43(db, adventure) + db.close() + + migrations.bootstrap(engine) + + check = SessionLocal() + try: + for action_id, value in expected.items(): + assert check.get(models.Action, action_id).context_snapshot == value + assert check.execute( + text("SELECT count(*) FROM actions WHERE context_snapshot IS NULL") + ).scalar() == 1 + finally: + check.close() + + +def test_bootstrap_leaves_the_column_named_context_snapshot(db, adventure): + seed_pre_43(db, adventure) + db.close() + + migrations.bootstrap(engine) + + with engine.begin() as conn: + columns = {row[1] for row in conn.execute(text("PRAGMA table_info(actions)"))} + assert "context_snapshot" in columns + assert "context_snapshot_z" not in columns, "the swap left the scratch column behind" + + +def test_the_backfill_aborts_rather_than_dropping_unconvertible_data( + db, adventure, monkeypatch +): + """Migration 44 destroys the original. If anything cannot be converted the + whole run has to roll back with the column still there — the alternative is + losing somebody's prompts to a bug in this file.""" + seed_pre_43(db, adventure) + db.close() + + def broken_unpack(_blob): + return {"not": "what went in"} + + monkeypatch.setattr(compression, "unpack", broken_unpack) + + with pytest.raises(RuntimeError, match="round trip"): + migrations.bootstrap(engine) + + with engine.begin() as conn: + columns = {row[1] for row in conn.execute(text("PRAGMA table_info(actions)"))} + version = conn.execute(text("PRAGMA user_version")).scalar() + surviving = conn.execute( + text("SELECT count(*) FROM actions WHERE context_snapshot IS NOT NULL") + ).scalar() + + assert "context_snapshot" in columns + assert version < migrations.SNAPSHOT_COMPRESS_VERSION, \ + "the version advanced past a backfill that failed" + assert surviving == 4, "the prompts did not survive the rollback" diff --git a/backend/tools/fakeprose.py b/backend/tools/fakeprose.py new file mode 100644 index 0000000..a0ca9e7 --- /dev/null +++ b/backend/tools/fakeprose.py @@ -0,0 +1,38 @@ +"""Filler text that behaves like prose under compression. + +Shared by the stress harness and the egress tests, because both now measure +something a repeated string would answer wrongly. `"x" * 20_000` compresses +about a thousandfold and English three- or fourfold, so a fixture built from +repeats makes a compressed column look free and any ceiling drawn around it +meaningless. + +Not a language model, and does not need to be. What matters is the symbol +distribution, not the sense. +""" +from __future__ import annotations + +import random + +WORDS = ( + "corridor narrows shoulders brush wet stone torchlight gutters draught " + "smells cold iron somewhere ahead water moving count nine paces passage " + "opens chamber ceiling lost dark sound breathing comes back half second " + "late Gwen catches sleeve without word points floor line pale grit laid " + "across threshold deliberate arc quartermaster bandit camp above ford " + "tunnels exchange key lantern rope knife bread rain mud hill road gate " + "watchman silver debt promise fever horse cart river bridge mill barley " + "smoke rafters bench ale ledger seal wax parchment ink candle shutter " + "hinge bolt cellar barrel salt fish nets harbour tide gull mast canvas" +).split() + + +def prose(rng: random.Random, nbytes: int) -> str: + """Roughly `nbytes` of varied sentences, deterministic for a given rng.""" + out: list[str] = [] + total = 0 + while total < nbytes: + sentence = " ".join(rng.choice(WORDS) for _ in range(rng.randint(8, 18))) + chunk = sentence.capitalize() + ". " + out.append(chunk) + total += len(chunk) + return "".join(out)[:nbytes] diff --git a/backend/tools/stress_session.py b/backend/tools/stress_session.py index 9bb2ce1..a9b9f23 100644 --- a/backend/tools/stress_session.py +++ b/backend/tools/stress_session.py @@ -108,6 +108,7 @@ from app.providers import PromptParts from app.routers import adventures from .dbmeter import Meter, kb +from .fakeprose import prose EMBEDDING_DIMS = 1536 @@ -123,37 +124,6 @@ EMBEDDING_DIMS = 1536 SNAPSHOT_SYSTEM = None # set by _build_text() SNAPSHOT_STORY = None -_WORDS = ( - "corridor narrows shoulders brush wet stone torchlight gutters draught " - "smells cold iron somewhere ahead water moving count nine paces passage " - "opens chamber ceiling lost dark sound breathing comes back half second " - "late Gwen catches sleeve without word points floor line pale grit laid " - "across threshold deliberate arc quartermaster bandit camp above ford " - "tunnels exchange key lantern rope knife bread rain mud hill road gate " - "watchman silver debt promise fever horse cart river bridge mill barley " - "smoke rafters bench ale ledger seal wax parchment ink candle shutter " - "hinge bolt cellar barrel salt fish nets harbour tide gull mast canvas" -).split() - - -def prose(rng: random.Random, nbytes: int) -> str: - """Filler of about `nbytes`, varied enough to compress like prose. - - Not decoration. A sentence repeated N times compresses roughly a - hundredfold and English roughly three- or fourfold, so a fixture built out - of repeats would report a compression ratio that says nothing about the - real column — and that ratio is the whole question for context_snapshot. - """ - out: list[str] = [] - total = 0 - while total < nbytes: - sentence = " ".join(rng.choice(_WORDS) for _ in range(rng.randint(8, 18))) - chunk = sentence.capitalize() + ". " - out.append(chunk) - total += len(chunk) - return "".join(out)[:nbytes] - - PLAYER_INPUT = "> You crouch and look more closely at the grit on the floor." # All three are bound by _build_text() from the fixture arguments. From cf8ec22e8bcdcfdebc84cbd33741a88f12352c4b Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Mon, 17 Aug 2026 14:23:03 +0530 Subject: [PATCH 08/10] Open an adventure on a window of the story, and page upward Opening a finished adventure fetched every action in one response: 589.5 kB on production's longest, and nothing about that curve bends on its own, because a story only ever gets longer. The page load now brings the newest 60 actions and the reader pages up from there. On the harness's 600-action fixture that is 606.0 kB down to 62.6 kB, and -- the part that matters -- it no longer depends on how long the story is. Paged by anchor, not by offset. `before_id` is the oldest action the caller holds; the server returns what precedes it. An offset counted back from the newest would shift every older position the moment a turn lands, which is exactly when someone is likely to be scrolling, and the reader would get one action twice and never see another. It also keeps working when the story stops being a flat list: comparing indices to order a branch survives the story tree, treating them as positions does not. `has_more` comes from fetching one row past the window rather than from counting. A deleted anchor -- undo, mid-scroll -- reports the end rather than guessing and serving a page the reader already has. GET /{id}/actions and POST /{id}/undo now return {actions, total, has_more} instead of a bare list. Undo is the action most likely to be repeated several times running, so having it re-fetch the whole story would have undone the paging on the worst case. The adventure payload gets its window through set_committed_value rather than by assignment: the actions relationship cascades delete-orphan, so assigning a 60-item list to it would delete everything outside the window on the next flush. In Play.jsx the prepend is followed by a useLayoutEffect that restores the scroll position, before paint, so the story does not jump. Loading starts 400px from the top rather than at it, guarded by a ref because scroll fires far faster than React re-renders. There is a button as well as the scroll trigger: on a short viewport the transcript may not be tall enough to scroll at all, and a reader who cannot scroll must still be able to reach the beginning. Verified against a running backend and a 220-action adventure: the page load returns 60 of 220 ending on the newest, and walking back from an anchor returns exactly the actions before it. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7 --- backend/app/routers/adventures.py | 132 +++++++++++++-- backend/app/schemas.py | 14 ++ backend/tests/test_action_paging.py | 240 +++++++++++++++++++++++++++ backend/tests/test_retry_variants.py | 3 +- frontend/src/api.js | 11 ++ frontend/src/index.css | 25 +++ frontend/src/pages/Play.jsx | 112 ++++++++++++- 7 files changed, 519 insertions(+), 18 deletions(-) create mode 100644 backend/tests/test_action_paging.py diff --git a/backend/app/routers/adventures.py b/backend/app/routers/adventures.py index 2d0632a..336d9cf 100644 --- a/backend/app/routers/adventures.py +++ b/backend/app/routers/adventures.py @@ -7,6 +7,7 @@ from fastapi import APIRouter, Body, Depends, HTTPException, Request from fastapi.responses import StreamingResponse from sqlalchemy import func from sqlalchemy.orm import Session, load_only, undefer +from sqlalchemy.orm.attributes import set_committed_value from .. import auth, images, limits, memorybank, models, schemas, worldstate from ..context import build_context @@ -43,6 +44,76 @@ ACTION_LIST_COLUMNS = ( models.Action.created_at, ) +# How many actions an adventure opens with, and how many arrive per scroll. +# +# Opening a finished adventure used to fetch the whole story in one response — +# 589.5 kB on production's longest, and growing, because a story only ever gets +# longer. 60 is a few screens of reading: enough that the common case (open, +# read the end, take a turn) never pages at all, small enough that the worst +# case is bounded by the window rather than by the story. +ACTION_PAGE = 60 + + +def action_window( + db: Session, + adventure_id: int, + before_id: int | None = None, + limit: int = ACTION_PAGE, +) -> tuple[list[models.Action], int, bool]: + """The `limit` actions immediately older than `before_id`, oldest first. + + Returns (actions, total, has_more). `before_id=None` is the newest window. + + Anchored on an action, not on a count, and never on arithmetic over + `Action.index`. Two separate reasons, and both bite: + + * **Appends.** Counting back from the newest means every older position + shifts when a turn lands. A reader who scrolls up while a turn is + generating would be handed a window one row out — re-sending one action + and silently skipping another. An anchor is fixed: "older than this one" + means the same thing before and after the story grows. + * **The story tree.** Index is a dense 0..n sequence today and branching + ends that. Comparing indices to order a branch survives; treating them as + positions does not. + + `has_more` comes from asking for one row past the window rather than from + counting, so it costs a row and not a scan. + """ + total = ( + db.query(func.count(models.Action.id)) + .filter(models.Action.adventure_id == adventure_id) + .scalar() + ) + if limit <= 0: + return [], total, total > 0 + + query = ( + db.query(models.Action) + .options(load_only(*ACTION_LIST_COLUMNS)) + .filter(models.Action.adventure_id == adventure_id) + ) + if before_id is not None: + anchor = ( + db.query(models.Action.index) + .filter(models.Action.id == before_id, + models.Action.adventure_id == adventure_id) + .scalar() + ) + if anchor is None: + # The anchor was deleted (undo, or a turn edited away) while the + # reader was scrolling. Nothing older can be identified relative to + # a row that no longer exists, so report the end rather than + # guessing and handing back a duplicate page. + return [], total, False + query = query.filter(models.Action.index < anchor) + + rows = query.order_by(models.Action.index.desc()).limit(limit + 1).all() + has_more = len(rows) > limit + rows = rows[:limit] + rows.reverse() + return rows, total, has_more + + # Exactly what schemas.MemoryOut renders. `embedded` is a real column and is on # the list; the vector it describes is not, and must never be. MEMORY_LIST_COLUMNS = ( @@ -303,7 +374,25 @@ def create_adventure( def get_adventure( adventure_id: int, db: Session = Depends(get_db), user: models.User = CurrentUser ): - return get_adventure_or_404(adventure_id, db, user) + """The adventure, and the newest window of its story. + + `actions` is the last ACTION_PAGE, not all of them; `action_count` says how + many there are so the reader knows there is more above. Older pages come + from GET /{id}/actions as they scroll up. + """ + adventure = get_adventure_or_404(adventure_id, db, user) + actions, total, _ = action_window(db, adventure_id) + # Hand the response the window as if the relationship had loaded it. + # `set_committed_value` is the only way to do this safely: assigning + # `adventure.actions = [...]` marks the collection dirty, and the + # relationship cascades delete-orphan, so the actions left out of the + # window would be deleted on the next flush. This records them as the + # loaded, unmodified value instead, so serialising touches no lazy load + # and nothing is pending. + set_committed_value(adventure, "actions", actions) + out = schemas.AdventureOut.model_validate(adventure) + out.action_count = total + return out @router.get("/{adventure_id}/script-state") @@ -950,7 +1039,7 @@ def select_variant( _active_turns.discard(adventure_id) -@router.post("/{adventure_id}/undo", response_model=list[schemas.ActionOut]) +@router.post("/{adventure_id}/undo", response_model=schemas.ActionPage) def undo_turn( adventure_id: int, db: Session = Depends(get_db), user: models.User = CurrentUser ): @@ -992,7 +1081,16 @@ def undo_turn( memorybank.prune_dangling_memories(adventure, db) db.commit() db.refresh(adventure) - return adventure.actions + # The newest window, not the whole story: the client replaces its + # transcript with this, and the transcript is a window now. Returning + # everything here would undo the paging on the one action most likely + # to be repeated several times in a row. + actions, total, has_more = action_window(db, adventure_id) + return schemas.ActionPage( + actions=[schemas.ActionOut.model_validate(a) for a in actions], + total=total, + has_more=has_more, + ) finally: _active_turns.discard(adventure_id) @@ -1575,17 +1673,29 @@ def delete_memory( # ---------- Actions (CRUD) ---------- -@router.get("/{adventure_id}/actions", response_model=list[schemas.ActionOut]) +@router.get("/{adventure_id}/actions", response_model=schemas.ActionPage) def list_actions( - adventure_id: int, db: Session = Depends(get_db), user: models.User = CurrentUser + adventure_id: int, + before_id: int | None = None, + limit: int = ACTION_PAGE, + db: Session = Depends(get_db), + user: models.User = CurrentUser, ): + """A page of the story, walking backwards from the newest action. + + `before_id` is the oldest action the caller already holds, so scrolling up + is "give me what comes before this". Omit it for the newest window. See + action_window for why this anchors on a row rather than an offset. + """ get_adventure_or_404(adventure_id, db, user) - return ( - db.query(models.Action) - .options(load_only(*ACTION_LIST_COLUMNS)) - .filter(models.Action.adventure_id == adventure_id) - .order_by(models.Action.index) - .all() + limit = max(1, min(limit, ACTION_PAGE * 4)) + actions, total, has_more = action_window( + db, adventure_id, before_id=before_id, limit=limit + ) + return schemas.ActionPage( + actions=[schemas.ActionOut.model_validate(a) for a in actions], + total=total, + has_more=has_more, ) diff --git a/backend/app/schemas.py b/backend/app/schemas.py index 656a001..66f7335 100644 --- a/backend/app/schemas.py +++ b/backend/app/schemas.py @@ -225,7 +225,21 @@ class AdventureOut(ORMModel): created_at: datetime updated_at: datetime story_cards: list[StoryCardOut] = [] + # The NEWEST window of the story, not all of it — older pages arrive from + # GET /{id}/actions as the reader scrolls up. `action_count` is the whole + # story's length, which is how the client knows there is more above. actions: list[ActionOut] = [] + action_count: int = 0 + + +class ActionPage(BaseModel): + """A slice of the story, counted back from the newest action.""" + + actions: list[ActionOut] = [] + total: int = 0 + # Whether anything older than this slice exists. Computed server-side so + # the client never has to do arithmetic on positions to find the end. + has_more: bool = False # ---------- Memory bank (Phase 6) ---------- diff --git a/backend/tests/test_action_paging.py b/backend/tests/test_action_paging.py new file mode 100644 index 0000000..f661dc1 --- /dev/null +++ b/backend/tests/test_action_paging.py @@ -0,0 +1,240 @@ +"""Opening an adventure fetches a window, not the whole story. + +A story only ever gets longer. Production's longest is 607 actions and 589.5 kB +in one response, and nothing about that curve bends on its own — so the page +load returns the newest ACTION_PAGE and the reader pages upward. + +The paging anchors on an action id rather than an offset, and these tests are +mostly about why. An offset counted back from the newest shifts every older +position the moment a turn lands, which is precisely when a reader is likely +to be scrolling. An anchor means the same thing before and after. + + python -m pytest tests/test_action_paging.py -v +""" +import os +import tempfile + +_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False) +_tmp.close() +os.environ["AIDND_DB_PATH"] = _tmp.name +os.environ.pop("AIDND_DATABASE_URL", None) +os.environ.pop("DATABASE_URL", None) + +import pytest +from fastapi import Depends +from fastapi.testclient import TestClient + +from app import auth, limits, models +from app.database import Base, SessionLocal, engine, get_db +from app.main import app +from app.routers.adventures import ACTION_PAGE +from tools import dbmeter + +TOTAL = ACTION_PAGE * 3 + 7 # deliberately not a whole number of pages + + +@pytest.fixture() +def client(monkeypatch): + Base.metadata.create_all(bind=engine) + setup = SessionLocal() + user = models.User(is_guest=False, email="paging@example.com") + setup.add(user) + setup.flush() + setup.add(models.Settings(user_id=user.id, api_key="enc:dummy", model="m")) + adventure = models.Adventure(user_id=user.id, title="Cave", script_state={}) + setup.add(adventure) + setup.flush() + for i in range(TOTAL): + setup.add(models.Action( + adventure_id=adventure.id, index=i, + type="start" if i == 0 else ("ai" if i % 2 else "do"), + text=f"Action {i}." + "word " * 200, + )) + setup.commit() + adv_id, user_id = adventure.id, user.id + setup.close() + + monkeypatch.setattr(limits, "rate_limit", lambda *a, **k: None) + monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) + + def _current_user(db=Depends(get_db)): + return db.get(models.User, user_id) + + app.dependency_overrides[auth.get_current_user] = _current_user + c = TestClient(app) + c.adv_id = adv_id + try: + yield c + finally: + app.dependency_overrides.clear() + Base.metadata.drop_all(bind=engine) + + +def page(client, before_id=None, limit=None): + params = {} + if before_id is not None: + params["before_id"] = before_id + if limit is not None: + params["limit"] = limit + r = client.get(f"/api/adventures/{client.adv_id}/actions", params=params) + assert r.status_code == 200, r.text + return r.json() + + +def add_action(client, text="A new turn.") -> int: + db = SessionLocal() + try: + highest = db.query(models.Action.index).order_by( + models.Action.index.desc()).first()[0] + action = models.Action( + adventure_id=client.adv_id, index=highest + 1, type="ai", text=text + ) + db.add(action) + db.commit() + return action.id + finally: + db.close() + + +# ------------------------------------------------------------- the page load + +def test_the_page_load_returns_only_the_newest_window(client): + r = client.get(f"/api/adventures/{client.adv_id}") + assert r.status_code == 200 + body = r.json() + assert len(body["actions"]) == ACTION_PAGE + assert body["action_count"] == TOTAL + # ...and it is the *newest* window, ending on the last action. + assert body["actions"][-1]["index"] == TOTAL - 1 + assert body["actions"][0]["index"] == TOTAL - ACTION_PAGE + + +def test_a_short_story_is_returned_whole(client): + db = SessionLocal() + try: + db.query(models.Action).filter(models.Action.index >= 5).delete() + db.commit() + finally: + db.close() + + body = client.get(f"/api/adventures/{client.adv_id}").json() + assert len(body["actions"]) == 5 + assert body["action_count"] == 5 + + +def test_the_page_load_does_not_grow_with_the_story(client): + """The point of the change. Whatever the story's length, opening it costs + a window.""" + meter = dbmeter.Meter() + meter.attach(engine) + try: + with meter.scope("page load"): + client.get(f"/api/adventures/{client.adv_id}") + windowed = meter.scopes[-1].total.fetched + finally: + meter.detach() + + # Each action carries ~1 kB of text and there are 187 of them; a window is + # 60. Generous ceiling, but far below the whole story. + assert windowed < ACTION_PAGE * 2_000, f"{windowed:,} B for one window" + assert windowed < TOTAL * 500, ( + f"{windowed:,} B — that is the whole story, not a window" + ) + + +# ------------------------------------------------------------------ paging up + +def test_the_first_page_is_the_newest(client): + body = page(client) + assert len(body["actions"]) == ACTION_PAGE + assert body["total"] == TOTAL + assert body["has_more"] is True + assert body["actions"][-1]["index"] == TOTAL - 1 + + +def test_paging_up_covers_the_whole_story_exactly_once(client): + seen = [] + body = page(client) + seen = [a["index"] for a in body["actions"]] + guard = 0 + while body["has_more"]: + guard += 1 + assert guard < 20, "paging did not terminate" + body = page(client, before_id=body["actions"][0]["id"]) + seen = [a["index"] for a in body["actions"]] + seen + + assert seen == list(range(TOTAL)), "gap, duplicate or reordering while paging" + + +def test_has_more_is_false_at_the_beginning_of_the_story(client): + body = page(client) + while body["has_more"]: + body = page(client, before_id=body["actions"][0]["id"]) + assert body["actions"][0]["index"] == 0 + + +def test_each_page_is_ordered_oldest_first(client): + body = page(client) + indices = [a["index"] for a in body["actions"]] + assert indices == sorted(indices) + + +# ------------------------------------------------- the reason for the anchor + +def test_a_turn_arriving_mid_scroll_does_not_shift_the_next_page(client): + """The failure an offset would have. Read the newest page, let a turn land, + then page up: the reader must get exactly what precedes what they hold — + no duplicate, no skipped action.""" + first = page(client) + oldest_held = first["actions"][0] + + add_action(client) + + older = page(client, before_id=oldest_held["id"]) + assert older["actions"][-1]["index"] == oldest_held["index"] - 1, \ + "the page shifted when a turn landed" + assert all(a["index"] < oldest_held["index"] for a in older["actions"]) + # The new turn moved the total, which is fine — it must not move the window. + assert older["total"] == TOTAL + 1 + + +def test_a_deleted_anchor_reports_the_end_rather_than_a_duplicate_page(client): + """Undo can remove the action a slow scroll was anchored to. Better to stop + than to hand back a page the reader already has.""" + body = page(client) + anchor = body["actions"][0] + + db = SessionLocal() + try: + db.query(models.Action).filter(models.Action.id == anchor["id"]).delete() + db.commit() + finally: + db.close() + + after = page(client, before_id=anchor["id"]) + assert after["actions"] == [] + assert after["has_more"] is False + + +# ------------------------------------------------------------------- limits + +def test_limit_is_honoured_and_capped(client): + assert len(page(client, limit=5)["actions"]) == 5 + # A client asking for the whole story does not get to undo the paging. + assert len(page(client, limit=100_000)["actions"]) <= ACTION_PAGE * 4 + + +def test_a_nonsense_limit_still_returns_something(client): + assert len(page(client, limit=0)["actions"]) >= 1 + assert len(page(client, limit=-5)["actions"]) >= 1 + + +# --------------------------------------------------------------------- undo + +def test_undo_returns_a_window_not_the_story(client): + r = client.post(f"/api/adventures/{client.adv_id}/undo") + assert r.status_code == 200, r.text + body = r.json() + assert len(body["actions"]) == ACTION_PAGE + assert body["total"] == TOTAL - 1 + assert body["has_more"] is True diff --git a/backend/tests/test_retry_variants.py b/backend/tests/test_retry_variants.py index adfcf70..7eaa9fe 100644 --- a/backend/tests/test_retry_variants.py +++ b/backend/tests/test_retry_variants.py @@ -294,7 +294,8 @@ def test_undo_removes_the_action_and_its_history(client): _retry(client) r = client.post(f"/api/adventures/{client.adv_id}/undo") assert r.status_code == 200, r.text - assert [a["type"] for a in r.json()] == ["start"] + # Undo returns the newest window now, not the whole story. + assert [a["type"] for a in r.json()["actions"]] == ["start"] assert _adv(client.adv_id)[0] == {} diff --git a/frontend/src/api.js b/frontend/src/api.js index c2be995..43bb2a8 100644 --- a/frontend/src/api.js +++ b/frontend/src/api.js @@ -86,6 +86,17 @@ export const api = { createAdventure: (data) => request('/adventures', { method: 'POST', body: JSON.stringify(data) }), updateAdventure: (id, data) => request(`/adventures/${id}`, { method: 'PATCH', body: JSON.stringify(data) }), deleteAdventure: (id) => request(`/adventures/${id}`, { method: 'DELETE' }), + // A page of the story, walking backwards. `beforeId` is the oldest action + // already on screen; omit it for the newest window. Anchored on an action + // rather than an offset so a turn landing mid-scroll cannot shift the page + // out from under the reader. + getActions: (advId, { beforeId, limit } = {}) => { + const params = new URLSearchParams() + if (beforeId != null) params.set('before_id', beforeId) + if (limit != null) params.set('limit', limit) + const query = params.toString() + return request(`/adventures/${advId}/actions${query ? `?${query}` : ''}`) + }, updateAction: (advId, actionId, text) => request(`/adventures/${advId}/actions/${actionId}`, { method: 'PATCH', body: JSON.stringify({ text }) }), deleteAction: (advId, actionId) => diff --git a/frontend/src/index.css b/frontend/src/index.css index 04c1464..204d6a0 100644 --- a/frontend/src/index.css +++ b/frontend/src/index.css @@ -332,6 +332,31 @@ button:disabled { opacity: 0.45; cursor: default; transform: none; box-shadow: n white-space: pre-wrap; padding-bottom: 16px; } +/* The head of a windowed transcript: how much story is still above, and the + way back to it. Deliberately quiet — it is a seam in the page, not a + feature — and it sits inside .story so it inherits the story face. */ +.story-earlier { + text-align: center; + margin: 4px 0 22px; + font-size: 0.9rem; + /* No action-in animation here: this appears above content the reader is + already looking at, and a fade would read as the story moving. */ +} +.story-earlier .dim { color: var(--text-dim); font-style: italic; } +.story-earlier button { + background: none; + border: none; + border-bottom: 1px solid rgba(159, 199, 209, 0.3); + color: var(--text-dim); + font: inherit; + font-style: italic; + cursor: pointer; + padding: 2px 4px; +} +.story-earlier button:hover { + color: var(--accent-bright); + border-bottom-color: var(--accent-bright); +} .story .action { position: relative; margin-bottom: 15px; diff --git a/frontend/src/pages/Play.jsx b/frontend/src/pages/Play.jsx index ddc3305..19538fe 100644 --- a/frontend/src/pages/Play.jsx +++ b/frontend/src/pages/Play.jsx @@ -1,4 +1,4 @@ -import { Fragment, useCallback, useEffect, useMemo, useRef, useState } from 'react' +import { Fragment, useCallback, useEffect, useLayoutEffect, useMemo, useRef, useState } from 'react' import { createPortal } from 'react-dom' import { useNavigate, useParams } from 'react-router-dom' import { api } from '../api' @@ -1310,10 +1310,20 @@ export default function Play() { // Read-only browsing of an earlier attempt at a past turn (see VariantPager). // One at a time; null when every message is showing its active version. const [preview, setPreview] = useState(null) + // The transcript is a window on the story, not the whole of it: the page + // load brings the newest page and older ones arrive as the reader scrolls + // up. `total` is the story's real length, for the "N earlier" line. + const [total, setTotal] = useState(0) + const [hasMore, setHasMore] = useState(false) + const [loadingOlder, setLoadingOlder] = useState(false) const storyEndRef = useRef(null) const abortRef = useRef(null) const pinnedRef = useRef(true) // autoscroll only while the reader is at the bottom const inputRef = useRef(null) + // Set just before older actions are prepended, read once afterwards to put + // the reader back where they were. See the layout effect below. + const restoreScrollRef = useRef(null) + const loadingOlderRef = useRef(false) // The drop cap belongs to the story's first narrated beat. `start` is the // scenario's opening prompt, so it's usually that; an adventure begun blank @@ -1340,18 +1350,76 @@ export default function Play() { useEffect(() => { api.getAdventure(id) - .then((adv) => { setAdventure(adv); setActions(adv.actions) }) + .then((adv) => { + setAdventure(adv) + setActions(adv.actions) + setTotal(adv.action_count ?? adv.actions.length) + setHasMore(adv.actions.length < (adv.action_count ?? adv.actions.length)) + }) .catch(() => navigate('/')) }, [id, navigate]) + // Fetch the page above the one on screen and prepend it. + // + // Anchored on the oldest action we hold rather than on a count, so a turn + // landing while the reader scrolls cannot shift the page. Guarded by a ref + // as well as state because scroll fires far faster than React re-renders, + // and two in-flight requests would fetch the same page twice. + const loadOlder = useCallback(async () => { + if (loadingOlderRef.current || !hasMore) return + const oldest = actions[0] + if (!oldest) return + loadingOlderRef.current = true + setLoadingOlder(true) + try { + const page = await api.getActions(id, { beforeId: oldest.id }) + if (page.actions.length) { + // Record the height before the prepend; the layout effect below uses + // it to keep the reader looking at the same paragraph. + restoreScrollRef.current = { + height: document.documentElement.scrollHeight, + top: window.scrollY, + } + setActions((prev) => { + // Defensive: never let a page the reader already holds duplicate a + // message. Cheap, and the alternative is a visibly doubled turn. + const known = new Set(prev.map((a) => a.id)) + return [...page.actions.filter((a) => !known.has(a.id)), ...prev] + }) + } + setTotal(page.total) + setHasMore(page.has_more) + } catch { + // Leave hasMore alone: a failed fetch should let the reader try again + // by scrolling, not permanently hide the rest of their story. + } finally { + loadingOlderRef.current = false + setLoadingOlder(false) + } + }, [actions, hasMore, id]) + + // Put the viewport back after a prepend. useLayoutEffect, not useEffect: + // this has to run before the browser paints, or the reader sees the story + // jump and then snap back. + useLayoutEffect(() => { + const mark = restoreScrollRef.current + if (!mark) return + restoreScrollRef.current = null + const grown = document.documentElement.scrollHeight - mark.height + if (grown > 0) window.scrollTo({ top: mark.top + grown }) + }, [actions]) + useEffect(() => { const onScroll = () => { pinnedRef.current = window.innerHeight + window.scrollY >= document.documentElement.scrollHeight - 120 + // Start the next page before the reader reaches the top, so the story + // is usually already there by the time they would have noticed its end. + if (window.scrollY < 400) loadOlder() } window.addEventListener('scroll', onScroll, { passive: true }) return () => window.removeEventListener('scroll', onScroll) - }, []) + }, [loadOlder]) useEffect(() => { // Snap to the real document bottom (below the sticky composer), not to @@ -1375,6 +1443,9 @@ export default function Play() { const handleEvent = useCallback((event) => { if (event.type === 'player') { setActions((prev) => [...prev, event.action]) + // The window grew at the bottom, so the story did too. Kept in step by + // hand because nothing re-reads the count between turns. + setTotal((n) => n + 1) } else if (event.type === 'chunk') { setStreaming((prev) => (prev ?? '') + event.text) } else if (event.type === 'reasoning') { @@ -1383,6 +1454,7 @@ export default function Play() { setStreaming(null) setReasoningStream(null) setActions((prev) => [...prev, event.action]) + setTotal((n) => n + 1) handleScriptReport(event.script) } else if (event.type === 'stopped') { setStreaming(null) @@ -1443,8 +1515,14 @@ export default function Play() { await api.retry(id, handleEvent, signal) } catch (err) { // Failed retry (409, network): the optimistically removed action may - // still exist server-side — resync instead of guessing. - api.getAdventure(id).then((adv) => setActions(adv.actions)).catch(() => {}) + // still exist server-side — resync instead of guessing. Resyncing + // collapses the transcript back to the newest window, which is the + // right call: the reader's place is already lost by the failure. + api.getAdventure(id).then((adv) => { + setActions(adv.actions) + setTotal(adv.action_count ?? adv.actions.length) + setHasMore(adv.actions.length < (adv.action_count ?? adv.actions.length)) + }).catch(() => {}) throw err } }) @@ -1454,7 +1532,12 @@ export default function Play() { setToast(null) setPreview(null) try { - setActions(await api.undo(id)) + // A window, not the whole story — undo is the action most likely to be + // repeated several times running, so it must not re-fetch everything. + const page = await api.undo(id) + setActions(page.actions) + setTotal(page.total) + setHasMore(page.has_more) } catch (err) { setToast({ text: err.message, isError: true }) } @@ -1494,6 +1577,7 @@ export default function Play() { try { await api.deleteAction(id, actionId) setActions((prev) => prev.filter((a) => a.id !== actionId)) + setTotal((n) => Math.max(0, n - 1)) } catch (err) { setToast({ text: err.message, isError: true }) } @@ -1534,6 +1618,22 @@ export default function Play() { {actions.length === 0 && streaming === null && (
A blank page. Type something below to begin your story.
)} + {/* Scrolling up loads the rest. The button is not decoration: on a + short viewport the story may not be tall enough to scroll at all, + and a reader who cannot scroll must still be able to get back to + the beginning. */} + {hasMore && ( +
+ {loadingOlder ? ( + Turning back the pages… + ) : ( + + )} +
+ )} {actions.map((action, i) => { const isPlayer = PLAYER_TYPES.includes(action.type) // A player action opens a new turn, so that's where the ornamental From 4f067516dec0782c896a0876ea05c7bc842f52e3 Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Mon, 17 Aug 2026 14:25:48 +0530 Subject: [PATCH 09/10] Close plan/13 and point STATUS at the tree Records what the six changes measured, and replaces the pick-up item -- which was step 6 -- with plan/14, since there is nothing left in 13. Two things a future session needs and cannot infer from the code. The VACUUM: migration 43 compresses context_snapshot but Postgres does not return the disk by itself, so until `VACUUM FULL actions` runs the storage win exists only on paper and the table is temporarily larger, not smaller. And the gap: nothing exercises the scroll behaviour in a browser, because the frontend has no test runner, and prepend-and-restore-scroll is the part most likely to feel wrong even when it is correct. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7 --- plan/13-memory-embedding-cost.md | 47 ++++++++++++- plan/STATUS.md | 116 ++++++++++++++++++++----------- 2 files changed, 121 insertions(+), 42 deletions(-) diff --git a/plan/13-memory-embedding-cost.md b/plan/13-memory-embedding-cost.md index 1076d32..a693f3a 100644 --- a/plan/13-memory-embedding-cost.md +++ b/plan/13-memory-embedding-cost.md @@ -164,8 +164,10 @@ additions the plan did not anticipate: forgotten. Vectors are held as `array("f")` — 6 KB each, matching the column; a list of Python floats would have been eight times the plan's RAM estimate. -Remaining: **step 6, infinite scroll upward in `Play.jsx`.** The page load is -unchanged at 426.7 kB and is now the largest single read in the app. +**Step 6 landed on 2026-08-17**, along with everything else this plan left open — see +"Closing the plan" at the end. The page load is a window of 60 actions now: 62.6 kB on +a 600-action fixture, down from 606.0 kB, and no longer a function of the story's +length. ## Guardrails to add with this work @@ -272,3 +274,44 @@ helps and does not solve it. None of the numbers above required reading a single row of anyone's content: counts, `octet_length` sums and catalog sizes only. + +## Closing the plan, 2026-08-17 + +Everything above landed the same day the verification did. + +| | before | after | +|---|---|---| +| page load, 600 actions | 606.0 kB | **62.6 kB**, and flat in story length | +| adventures index, 6 adventures | 469.7 kB | **0.3 kB** | +| `context_snapshot` stored | ~89 MB | ~43 MB (after a VACUUM) | +| a played turn | 6.4 MB (2026-08-16) | 123 kB | + +**Step 6, the window.** `GET /adventures/{id}` returns the newest `ACTION_PAGE` +actions and the story's length; `GET /{id}/actions?before_id=` walks back from there. +Anchored on an action rather than an offset — the offset version breaks precisely when +a turn lands mid-scroll, handing the reader one action twice and hiding another — and +that choice is also what makes it survive the story tree, since it compares indices to +order a branch rather than treating them as positions. + +**Byte assertions.** `tests/test_egress.py` now carries per-action budgets as well as +column guards, plus one test whose only job is to fail if the fixture ever gets too +small for the budgets to catch anything. + +**Column projections.** `ACTION_LIST_COLUMNS` and `MEMORY_LIST_COLUMNS` name what a +list response renders, and the adventures index selects four columns instead of the +entity. That one was not just future-proofing: an Adventure carries seven text and JSON +columns the index never shows. + +**The JSON vector column is gone**, and dropping it exposed a live bug — changing your +embedding model had stopped re-embedding the bank the day migration 38 shipped. See +`tests/test_embedding_model_switch.py`. + +**Storage.** `context_snapshot` is zlib-compressed through a TypeDecorator +(`app/compression.py`, migrations 43–45), so the call sites never learned about it. +3.5x on real Postgres. **This does not shrink anything until `VACUUM FULL actions` +runs** — Postgres marks dropped columns rather than reclaiming them, and the backfill +leaves a dead tuple per row. + +Not done: nothing exercises the scroll behaviour in a browser. The frontend has no test +runner, and prepend-and-restore-scroll is the part most likely to feel wrong even when +it is correct. diff --git a/plan/STATUS.md b/plan/STATUS.md index 0389445..bf1463f 100644 --- a/plan/STATUS.md +++ b/plan/STATUS.md @@ -18,36 +18,42 @@ project page points at (`docs/index.html`), not the service name in the blueprin --- -## Needs a human +## Needs a human: one VACUUM after the next deploy -**The free tier's storage ceiling is closer than the egress work suggested.** The Neon -database is 99.6 MB of a 512 MB allowance and `actions.context_snapshot` is essentially -all of it. See "Storage, which this plan did not cost" in `plan/13`. This is now the -most likely thing to break the deploy, ahead of anything on the read path. +Migration 43 compresses `context_snapshot`, and **Postgres does not hand the disk back +on its own.** `DROP COLUMN` only marks a column dropped, and the backfill leaves a dead +tuple per row, so `actions` gets *bigger* before it gets smaller — peaking near twice +its size while both columns are live. Once the deploy is up and healthy, run once: + +```sql +VACUUM FULL actions; +``` + +It needs exclusive access to the table and free space equal to the finished copy. On +the 2026-08-17 figures: 99.6 MB now, peaking near 200 during the migration, settling +around 53 afterwards, against a 512 MB tier. Skipping it is safe and simply leaves the +win unrealised — the database keeps working, it just stays large. + +Same caveat applies to migration 42 dropping `memories.embedding` (4 MB). --- ## Pick up here -**`plan/13-memory-embedding-cost.md`, step 6 — infinite scroll upward in `Play.jsx`.** +**`plan/14-phase-story-tree.md` — the tree itself.** `plan/13` is finished. Its design +is settled in `plan/14` and nothing about it has been built. -Opening a finished adventure is comfortably the largest single read in the app — a turn -is now 122 kB, Insights 118 kB, the Memories drawer 24 kB. Measured on production -(2026-08-17), the largest real adventure is **607 actions and 589.5 kB in one -response**; the 426.7 kB the harness reports is a 200-action fixture whose actions are -about **twice as heavy as real ones** (994 B/action in production). So the fixture -overstates width and understates length — real stories get *longer* than it models, -which is the direction that hurts. The backend already has the -windowing primitives (`context/history.py`: `tail_range`, `slice_`, `count`), and -`GET /adventures/{id}/actions` exists. What is missing is a paged shape for it and a -`Play.jsx` that loads the newest turns and fetches older ones as the reader scrolls up. +Two things from the egress work are worth carrying into it: -Watch for: the story is a flat list today, and **the story tree replaces it** -(`plan/14-phase-story-tree.md`). Paging that reads by *position from the end* survives -that change; paging that assumes `Action.index` is a dense 0..n sequence does not. +- **Paging already anticipates the tree.** `action_window` in `routers/adventures.py` + anchors on an action id and orders by comparing `Action.index`, never by treating + index as a position. A branch changes which actions are on the path, not how two of + them order, so the anchor survives; anything counting offsets would not. +- **Weigh new columns in bytes.** A tree adds parent/branch columns to `actions`, which + is already the table that fills the disk. `tests/test_egress.py` has byte ceilings + now — they will tell you. -After that: the tree itself. Its design is settled in `plan/14`; nothing about it has -been built. +**Before the next deploy:** one `VACUUM FULL actions;` — see below. --- @@ -115,7 +121,41 @@ Migrations 39/40 add `memories.embedded`, migration 41 drops the capacity defaul --- -## What happened on 2026-08-17 +## What happened on 2026-08-17, part two + +Everything left open in `plan/13` closed, plus two bugs that fell out of doing it. + +| shape | before today | after | +|---|---|---| +| page load, 600 actions | 606.0 kB | **62.6 kB** — and no longer grows with the story | +| adventures index, 6 fat adventures | 469.7 kB | **0.3 kB** | +| `context_snapshot` on disk | ~89 MB | ~43 MB (3.5x, after a VACUUM) | +| database total | 99.6 MB | ~53 MB projected | + +**The story is a window now.** `GET /adventures/{id}` returns the newest 60 actions and +`action_count`; older pages come from `GET /{id}/actions?before_id=`. Anchored on an +action, never an offset — an offset counted back from the newest shifts every older +position the moment a turn lands, which is exactly when someone is scrolling. `Play.jsx` +prepends and restores scroll position in a `useLayoutEffect`, before paint. + +**`context_snapshot` is compressed** (migrations 43–45, `app/compression.py`) via a +TypeDecorator, so every call site still reads and writes a dict. Verified end to end on +a throwaway Neon database: 720,864 B of JSON to 204,293 B of bytea, every row equal. + +**The JSON vector column is gone** (migration 42) — and dropping it exposed that +changing your embedding model had silently stopped re-embedding the bank since +migration 38. The settings route cleared the dead column and left `embedded` true, so +`_embed_pending` never saw those rows and retrieval kept ranking against the old +model's vectors. Nothing reported it: `cosine` returns 0.0 on a width mismatch. +`tests/test_embedding_model_switch.py`. + +**Byte ceilings exist** (`tests/test_egress.py`), including one test whose only job is +to prove the ceilings would catch something. + +**List responses name their columns.** The index was loading whole Adventure entities — +seven text and JSON columns, ~15 kB a row — to render a title and a snippet. + +## What happened on 2026-08-17, part one No new behaviour — a verification pass on what shipped the day before, because every number in the section above had been measured on SQLite against a synthetic fixture. @@ -169,25 +209,21 @@ with `pg_total_relation_size`, and do not mix them up. --- -## Still open from `plan/13` +## `plan/13` is closed -- **Step 6, infinite scroll upward** — the pick-up item above. -- **Query-count / byte assertions per endpoint**, extending `tests/test_egress.py` - against production-sized fixtures. `dbmeter` is importable from tests (`from tools - import dbmeter`) and was built with this in mind; nothing uses it there yet. -- **Explicit column projections on read paths**, so the next heavy column is opt-**in**. - Done for the memory paths, not as a general rule. -- **Drop `memories.embedding`** (the JSON column) in a follow-up migration. It is still - written by `set_vector` and read by nothing, kept so a rollback finds the vectors. - `tests/test_memory_retrieval.py` has a guard asserting nothing selects it. Measured - on production: dropping it reclaims 4.05 MB, 4% of the database. -- **`context_snapshot` and the 512 MB ceiling** — new, and now the biggest open item. - See the two sections named above. The egress case for leaving it in the database - still stands; the storage case does not. +All six of its open items landed on 2026-08-17. What is left is not from that plan: -Deliberately not taken: moving `context_snapshot` out of the database (~$0.02/mo, costs -nothing on reads now that it is deferred), and pgvector (breaks the SQLite dev parity -this codebase protects on purpose). +- **The VACUUM**, above. Until it runs, the storage win is on paper. +- **Nothing verifies the scroll behaviour in a browser.** The paging is covered by + `tests/test_action_paging.py` and was exercised against a running backend, but the + frontend has no test runner and the prepend-and-restore is the part most likely to + feel wrong. Worth thirty seconds of scrolling a long adventure before trusting it. +- **`ACTION_PAGE = 60` is a guess.** It should be a page or two of reading. If loading + older turns feels like it interrupts, that is the number to move. + +Deliberately not taken: moving `context_snapshot` out of the database entirely +(compressing it bought the same runway for a much smaller change), and pgvector (breaks +the SQLite dev parity this codebase protects on purpose). --- @@ -195,7 +231,7 @@ this codebase protects on purpose). ``` cd backend -.venv/Scripts/python.exe -m pytest tests/ # 225 tests +.venv/Scripts/python.exe -m pytest tests/ # 259 tests .venv/Scripts/python.exe -m tools.stress_session # egress report (SQLite) # Same harness against a real Postgres. The target must be a THROWAWAY database From 0eef367f3094523c01529eb6934dcf391af7050c Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Mon, 17 Aug 2026 14:48:44 +0530 Subject: [PATCH 10/10] Convert the snapshots in batches, not one round trip per row The backfill issued an UPDATE per action. It runs at container start, before uvicorn binds the port, against a database at the other end of a network -- 944 rows on production, so a thousand sequential round trips standing between the deploy and its first health check. One executemany per batch instead. Same rows, same verification, same transaction; nineteen round trips rather than nine hundred. Re-verified on Postgres, since executemany binds bytea through a different psycopg path: 720,864 B of JSON to 204,293 B, every row equal. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7 --- backend/app/migrations.py | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/backend/app/migrations.py b/backend/app/migrations.py index 8134cb0..2e19be1 100644 --- a/backend/app/migrations.py +++ b/backend/app/migrations.py @@ -319,6 +319,7 @@ def _backfill_context_snapshot(conn) -> None: ).all() if not rows: return + params = [] for row_id, stored in rows: value = json.loads(stored) if isinstance(stored, str) else stored if value is None: @@ -330,9 +331,16 @@ def _backfill_context_snapshot(conn) -> None: "compress/decompress round trip; refusing to drop the " "original column" ) + params.append({"z": packed, "id": row_id}) + if params: + # One executemany per batch, not one statement per row. This runs + # at container start, before the port opens, against a database on + # the other end of a network: a thousand round trips is the + # difference between a deploy that comes up and a health check that + # times out waiting for it. conn.execute( text("UPDATE actions SET context_snapshot_z = :z WHERE id = :id"), - {"z": packed, "id": row_id}, + params, ) last_id = rows[-1][0]