diff --git a/backend/tools/stress_session.py b/backend/tools/stress_session.py index 64b8faa..999352b 100644 --- a/backend/tools/stress_session.py +++ b/backend/tools/stress_session.py @@ -23,11 +23,22 @@ sessions, the ORM, the scripting engine and the context builder are the real ones, because the bugs this exists to catch live in exactly the layer a mock would replace. -It runs on a throwaway SQLite file rather than Postgres. What is being measured -is which columns of which rows a code path asks for, and that is decided by the -ORM, identically on both. The dialects disagree on how a value is encoded on -the wire — JSON especially — so treat the absolute figures as production-shaped -rather than production-exact, and compare before against after. +It runs on a throwaway SQLite file by default. What is being measured is which +columns of which rows a code path asks for, and that is decided by the ORM, +identically on both dialects. The dialects disagree on how a value is encoded +on the wire — JSON especially — so treat the absolute figures as +production-shaped rather than production-exact, and compare before against +after. + +To measure the encodings SQLite cannot reach — bytea for the packed vectors, +and json columns psycopg parses before the meter sees them — set +AIDND_STRESS_DATABASE_URL to a **throwaway** Postgres database: + + AIDND_STRESS_DATABASE_URL=postgresql://…/stress_scratch \ + .venv/Scripts/python.exe -m tools.stress_session + +The harness writes, so it refuses any target whose database name does not say +'stress' or 'scratch'. Never point it at a database holding real users. Calibration, against the two figures measured directly on production (2026-08-16): a 200-action page load reported 426.7 kB here against 423 KB @@ -35,19 +46,41 @@ there, and one turn on a 100-memory bank reported 3,258.7 kB against 3,153 kB. """ import os +import sys import tempfile # Must precede the app import: database.py reads these at module scope. -_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False) -_tmp.close() -os.environ["AIDND_DB_PATH"] = _tmp.name -os.environ.pop("AIDND_DATABASE_URL", None) -os.environ.pop("DATABASE_URL", None) +# +# Default is a throwaway SQLite file. AIDND_STRESS_DATABASE_URL points the +# harness at a real Postgres instead, which is the only way to reach the +# encodings SQLite cannot exercise: bytea for the packed vectors, and json +# columns that psycopg parses into Python before the meter ever sees them. +# +# The name guard is not paranoia. This harness *writes* — it builds a whole +# synthetic adventure — so a URL that happened to point at the production +# database would quietly seed it with fake users and fake play. The target +# must say it is disposable. +_stress_url = os.environ.get("AIDND_STRESS_DATABASE_URL", "").strip() +if _stress_url: + _dbname = _stress_url.rsplit("/", 1)[-1].split("?")[0] + if not any(mark in _dbname.lower() for mark in ("stress", "scratch")): + sys.exit( + f"refusing to run against database {_dbname!r}.\n" + "This harness writes a synthetic adventure, so its target must be a\n" + "throwaway database with 'stress' or 'scratch' in the name." + ) + os.environ["AIDND_DATABASE_URL"] = _stress_url + os.environ.pop("DATABASE_URL", None) +else: + _tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False) + _tmp.close() + os.environ["AIDND_DB_PATH"] = _tmp.name + os.environ.pop("AIDND_DATABASE_URL", None) + os.environ.pop("DATABASE_URL", None) import argparse import asyncio import random -import sys from fastapi import Depends from fastapi.testclient import TestClient @@ -125,6 +158,13 @@ class FakeEmbeddings: def build_fixture(args, rng: random.Random) -> tuple[int, int]: """A user, settings and one adventure at production scale. Returns (adventure_id, user_id).""" + # A SQLite run gets a brand-new temp file every time, so the fixture can + # assume an empty database. A Postgres scratch target persists between + # runs, and the second one would collide on the fixture user's unique + # email — so empty it first. Only ever reached for a target whose name + # passed the 'stress'/'scratch' guard at the top of this module. + if _stress_url: + Base.metadata.drop_all(bind=engine) Base.metadata.create_all(bind=engine) db = SessionLocal() try: diff --git a/plan/13-memory-embedding-cost.md b/plan/13-memory-embedding-cost.md index 4cd95a8..1076d32 100644 --- a/plan/13-memory-embedding-cost.md +++ b/plan/13-memory-embedding-cost.md @@ -182,6 +182,9 @@ Deliberately **not** taken: moving `context_snapshot` out of the database. It co nothing on reads now that it is deferred, and storage is ~$0.02/mo. Revisit only if backups or storage start to hurt. +> **Revisit it.** That call weighed egress and got egress right, but it never weighed +> the free tier's *storage* ceiling — see "Storage, which this plan did not cost" below. + ## Verification - Harness: `python -m tools.stress_session`, memory bank **on**, before and after, @@ -193,3 +196,79 @@ backups or storage start to hurt. playthrough number is finally honest. - The existing `test_egress.py` guard must still pass — nothing here should touch the deferred action columns. + +## Verified on production, 2026-08-17 + +Two things were still taken on trust when this shipped: every measurement had run on +SQLite, and every number came from a synthetic fixture. Both are now checked. + +### The migration landed on real Postgres + +`schema_version` reads **41**, matching the repo's `LATEST_VERSION`. The live schema has +`embedding_blob bytea` and `embedded boolean`, so the `{dialect: sql}` map in migration +38 spells BYTEA correctly against a real server — the one thing tests could not prove, +since `test_migration_38_is_spelled_for_both_dialects` only inspects the SQL string. +The backfill is complete: 134 memories, `embedded = 134`, `embedding_blob = 134`, no +stragglers and no rows skipped as malformed. + +### The 5x is real, on real vectors + +| | bytes | per memory | +|---|---|---| +| `embedding` (JSON) | 4,150,121 | 30,971 | +| `embedding_blob` (float32) | 823,296 | 6,144 | + +**5.04x**, against the plan's predicted ~31 KB → 6,144 B. The largest real bank is 100 +memories = 614,400 B of vectors, so the old code fetched **~3.10 MB per retrieval** on +that adventure — which is where the 3,153 kB measured on production came from. That +figure is now fully accounted for. + +### SQLite and Postgres agree + +`tools.stress_session` gained an `AIDND_STRESS_DATABASE_URL` escape hatch and was run +against a throwaway Neon database at the default fixture (200 actions, 100 memories): + +| shape | SQLite | Postgres | +|---|---|---| +| index | 4.1 kB | 4.1 kB | +| page load | 426.7 kB | 425.0 kB | +| one turn, cold | 723.4 kB | 722.3 kB | +| one turn, warm | 122.3 kB | **121.1 kB** | +| Insights | 117.9 kB | 116.7 kB | +| Memories drawer | 23.7 kB | 21.7 kB | +| `run_post_turn` | 0.7 kB | 0.6 kB | + +Within 0.5% everywhere. The dialect caveat in the harness docstring is real but small: +what dominates is which columns get asked for, and the ORM decides that identically. +The warm turn spends **1.7 kB on `memories`, 1% of the read** — the cache behaves on +psycopg exactly as it does on SQLite. + +### The page load is worse than modelled, for a different reason + +The synthetic fixture is **~2x heavier per action than production**: 994 B/action real +against ~2,133 B/action synthetic, so a real 200-action adventure is ~194 kB, not 427. +But the largest real adventure is **607 actions**, not 200, and costs **589.5 kB** in +one response. Step 6 is more urgent than this plan assumed, and for the opposite +reason to the one modelled — stories get *longer* than the fixture, not heavier. + +Worth fixing the fixture's narration size when step 6 lands, so the harness stops +flattering the per-action figure while understating the length. + +### Storage, which this plan did not cost + +`context_snapshot` is **150.8 MB of uncompressed JSON across 944 actions** — ~163 kB a +row on average, and ~232 kB a row in the largest adventure, against the ~74 KB/row the +comment in `models.py` claims. TOAST compresses it to ~89 MB on disk, but +`octet_length` is what would cross the wire, because Postgres decompresses before +sending. Deferral is the only thing standing between a bulk read and a 137 MB query. + +The database is **99.6 MB total**, of which `actions` is **88.9 MB**. Neon's free tier +is 512 MB. At ~94 kB of disk per action that ceiling arrives at roughly **5,400 +actions**, and 944 are already stored. So the "~$0.02/mo, leave it in the database" +call above is wrong for the tier this actually runs on — not because reads cost +anything, but because the free tier meters *storage*, and that is the constraint with +a cliff. Dropping the dead `memories.embedding` column reclaims 4.05 MB (4%), which +helps and does not solve it. + +None of the numbers above required reading a single row of anyone's content: counts, +`octet_length` sums and catalog sizes only. diff --git a/plan/STATUS.md b/plan/STATUS.md index 5f4235d..c2a8133 100644 --- a/plan/STATUS.md +++ b/plan/STATUS.md @@ -3,7 +3,23 @@ Read this first when picking the project back up. Updated at the end of a working session; the per-phase plan files hold the detail, this holds the thread. -**Last updated: 2026-08-16.** +**Last updated: 2026-08-17.** + +--- + +## Two things need a human first + +**The Render service is suspended.** `GET /api/health` returns 503 with a static +"This service has been suspended by its owner" page, in ~1.2s — that is the edge, not +a cold start (a free-tier wake hangs 30–60s and then serves). Nothing in the app is +wrong; check the dashboard. Free-tier suspensions come from usage/bandwidth caps or +billing, and real users have started arriving, so rule that out before assuming it was +manual. + +**The free tier's storage ceiling is closer than the egress work suggested.** The Neon +database is 99.6 MB of a 512 MB allowance and `actions.context_snapshot` is essentially +all of it. See "Storage, which this plan did not cost" in `plan/13`. This is now the +most likely thing to break the deploy, ahead of anything on the read path. --- @@ -11,9 +27,13 @@ session; the per-phase plan files hold the detail, this holds the thread. **`plan/13-memory-embedding-cost.md`, step 6 — infinite scroll upward in `Play.jsx`.** -Opening a finished 200-action adventure fetches **426.7 kB** in one response, and after -this session's work that is comfortably the largest single read in the app — a turn is -now 122 kB, Insights 118 kB, the Memories drawer 24 kB. The backend already has the +Opening a finished adventure is comfortably the largest single read in the app — a turn +is now 122 kB, Insights 118 kB, the Memories drawer 24 kB. Measured on production +(2026-08-17), the largest real adventure is **607 actions and 589.5 kB in one +response**; the 426.7 kB the harness reports is a 200-action fixture whose actions are +about **twice as heavy as real ones** (994 B/action in production). So the fixture +overstates width and understates length — real stories get *longer* than it models, +which is the direction that hurts. The backend already has the windowing primitives (`context/history.py`: `tail_range`, `slice_`, `count`), and `GET /adventures/{id}/actions` exists. What is missing is a paged shape for it and a `Play.jsx` that loads the newest turns and fetches older ones as the reader scrolls up. @@ -91,6 +111,26 @@ Migrations 39/40 add `memories.embedded`, migration 41 drops the capacity defaul --- +## What happened on 2026-08-17 + +No new behaviour — a verification pass on what shipped the day before, because every +number in the section above had been measured on SQLite against a synthetic fixture. +Full write-up in `plan/13` under "Verified on production". + +**It holds.** `schema_version` is 41 on the live Postgres with `embedding_blob bytea` +and `embedded boolean` present, so migration 38's dialect map is correct against a real +server. The backfill is complete (134/134). The packed vectors are **5.04x** smaller +than the JSON on real data — 30,971 → 6,144 bytes a memory, as predicted. + +**SQLite was not lying.** `tools.stress_session` can now target Postgres via +`AIDND_STRESS_DATABASE_URL`, and every shape agrees within 0.5% — the warm turn is +121.1 kB on Postgres against 122.3 kB on SQLite, with `memories` down to 1.7 kB of it. +Run it against a **throwaway** database only; the harness writes, so it refuses any +target whose name does not contain `stress` or `scratch`. + +**Two corrections came out of it**, both above: the page-load model has the wrong +shape (too heavy per action, far too short), and the storage ceiling was never costed. + ## Things worth remembering **The vector cache needs no invalidation callbacks, and that is why it is safe.** A @@ -112,6 +152,17 @@ beside `actions.variants`. Expect to need this for any future heavy column. **Any egress measurement must run with an embedding model set.** This is the second time that omission has hidden the biggest number in the room. +**Production has real users on it now. Measure it without reading it.** Counts, +`sum(octet_length(...))` and `pg_total_relation_size` answer every sizing question +asked so far, and none of them return anyone's story, memory text or email. When a +real Postgres is needed for a *write* path, create a throwaway database beside the real +one and drop it after — never point a harness at the production database. + +**`octet_length` is the egress number, not the on-disk number.** Postgres TOAST +compresses big JSON — `context_snapshot` is 150.8 MB uncompressed but ~89 MB stored — +and decompresses before sending. Size reads with `octet_length`, size the storage bill +with `pg_total_relation_size`, and do not mix them up. + --- ## Still open from `plan/13` @@ -124,7 +175,11 @@ time that omission has hidden the biggest number in the room. Done for the memory paths, not as a general rule. - **Drop `memories.embedding`** (the JSON column) in a follow-up migration. It is still written by `set_vector` and read by nothing, kept so a rollback finds the vectors. - `tests/test_memory_retrieval.py` has a guard asserting nothing selects it. + `tests/test_memory_retrieval.py` has a guard asserting nothing selects it. Measured + on production: dropping it reclaims 4.05 MB, 4% of the database. +- **`context_snapshot` and the 512 MB ceiling** — new, and now the biggest open item. + See the two sections named above. The egress case for leaving it in the database + still stands; the storage case does not. Deliberately not taken: moving `context_snapshot` out of the database (~$0.02/mo, costs nothing on reads now that it is deferred), and pgvector (breaks the SQLite dev parity @@ -137,8 +192,17 @@ this codebase protects on purpose). ``` cd backend .venv/Scripts/python.exe -m pytest tests/ # 225 tests -.venv/Scripts/python.exe -m tools.stress_session # egress report +.venv/Scripts/python.exe -m tools.stress_session # egress report (SQLite) + +# Same harness against a real Postgres. The target must be a THROWAWAY database +# — this writes a synthetic adventure, and it refuses any name without +# 'stress'/'scratch' in it. +AIDND_STRESS_DATABASE_URL=postgresql://…/stress_scratch \ + .venv/Scripts/python.exe -m tools.stress_session ``` +On Windows the report's box-drawing characters crash the default cp1252 console; +prefix with `PYTHONIOENCODING=utf-8`. + Port 8000 is shared with the job-pipeline app, which will squat it and silently shadow the AI-DnD API — free it before running the backend, or move the vite proxy.