diff --git a/backend/app/compression.py b/backend/app/compression.py new file mode 100644 index 0000000..e8b24ad --- /dev/null +++ b/backend/app/compression.py @@ -0,0 +1,70 @@ +"""Storing a JSON column compressed. + +`actions.context_snapshot` holds the entire assembled prompt for a turn. It is +89% of the database — 150.8 MB of JSON across 944 actions on production, and +232 KB a row on the longest adventure — and the free tier this deploys to +allows 512 MB. Reads are not the problem: the column is deferred, so a page +load never touches it and exactly one endpoint fetches one row of it at a +time. Storage is the problem, and storage has a cliff. + +Postgres already compresses it. TOAST brings 150.8 MB down to ~89 MB, a factor +of 1.7 — pglz is chosen for decompression speed on data a query might filter +on, which this never is. Nothing filters on a prompt; it is written once and +read whole, occasionally, by one screen. zlib at the application layer gets +three to four times on the same text, and the cost is a decompress on a +request that already costs an LLM call. + +Doing it as a TypeDecorator rather than a second column keeps every call site +writing `action.context_snapshot = {...}` and reading a dict back, and keeps +`deferred=True`, `undefer()` and `load_only()` naming the same attribute they +named before. The storage format changes; nothing else does. + +Level 6 is zlib's default and the knee of the curve here: 9 spends noticeably +more CPU on prompt text for about a percent more space. +""" +from __future__ import annotations + +import json +import zlib + +from sqlalchemy import LargeBinary +from sqlalchemy.types import TypeDecorator + +LEVEL = 6 + + +def pack(value) -> bytes: + """A JSON-able value as compressed UTF-8.""" + raw = json.dumps(value, separators=(",", ":"), default=str).encode("utf-8") + return zlib.compress(raw, LEVEL) + + +def unpack(blob: bytes) -> object: + """The value `pack` was given.""" + return json.loads(zlib.decompress(bytes(blob)).decode("utf-8")) + + +class CompressedJSON(TypeDecorator): + """A JSON column stored as zlib-compressed UTF-8 in a BLOB/BYTEA. + + `cache_ok = True`: the type carries no per-instance configuration, so + SQLAlchemy may reuse a compiled statement across instances of it. + """ + + impl = LargeBinary + cache_ok = True + + def process_bind_param(self, value, dialect): + return None if value is None else pack(value) + + def process_result_value(self, value, dialect): + # Tolerate a row the backfill has not reached yet, or one written + # before the conversion: a snapshot that cannot be read back is worth + # less than the screen that shows it, and never worth a 500 on the + # turn that happens to load it. + if value is None: + return None + try: + return unpack(value) + except (zlib.error, UnicodeDecodeError, ValueError): + return None diff --git a/backend/app/migrations.py b/backend/app/migrations.py index 413cc11..8134cb0 100644 --- a/backend/app/migrations.py +++ b/backend/app/migrations.py @@ -22,7 +22,7 @@ import json from sqlalchemy import inspect, text from sqlalchemy.engine import Engine -from . import vectors +from . import compression, vectors from .database import Base # (version, SQL to run when upgrading past it) — append only, never reorder. @@ -150,6 +150,34 @@ MIGRATIONS: list[tuple[int, str | dict[str, str]]] = [ # now 4 MB of a 99.6 MB database holding nothing anyone reads. DROP COLUMN # is spelled the same on both dialects — SQLite has had it since 3.35. (42, "ALTER TABLE memories DROP COLUMN embedding"), + # context_snapshot, compressed. 89% of the database is one column holding + # assembled prompts nobody filters on and one screen reads, one row at a + # time; Postgres already TOASTs it, but pglz only manages 1.7x and zlib + # gets three to four on the same text. Reads were fixed by deferring it — + # this is about the 512 MB the free tier allows. + # + # Three steps because a column cannot portably change type in place: add + # the new one, convert into it (_backfill_context_snapshot, which verifies + # every row round-trips before the old column goes), then swap the names so + # the model keeps calling it context_snapshot. + # + # **Postgres does not hand the disk back on its own.** DROP COLUMN only + # marks the column dropped, and the backfill's UPDATE leaves a dead tuple + # per row, so the table gets *bigger* before it gets smaller: peak is + # roughly twice the starting size while both columns are live. Plain + # autovacuum makes that space reusable but does not shrink the files. The + # deploy that ships this should follow it with, once: + # + # VACUUM FULL actions; + # + # which needs exclusive access and free space equal to the finished table. + # On the 2026-08-17 figures that is 99.6 MB peaking near 200, settling at + # about 53 once vacuumed, against a 512 MB tier. Skipping the vacuum is + # safe and simply leaves the win unrealised. + (43, {"sqlite": "ALTER TABLE actions ADD COLUMN context_snapshot_z BLOB", + "default": "ALTER TABLE actions ADD COLUMN context_snapshot_z BYTEA"}), + (44, "ALTER TABLE actions DROP COLUMN context_snapshot"), + (45, "ALTER TABLE actions RENAME COLUMN context_snapshot_z TO context_snapshot"), ] LATEST_VERSION = max((v for v, _ in MIGRATIONS), default=1) @@ -158,6 +186,12 @@ LATEST_VERSION = max((v for v, _ in MIGRATIONS), default=1) WORLD_DELTA_VERSION = 36 VARIANT_COUNT_VERSION = 37 EMBEDDING_BLOB_VERSION = 38 +SNAPSHOT_COMPRESS_VERSION = 43 + +# Snapshots converted per round trip. Deliberately far smaller than +# BACKFILL_BATCH: a vector is 6 KB and a snapshot is 232 KB, so 200 of these +# would be 46 MB held at once. +SNAPSHOT_BATCH = 50 # Vectors converted per round trip. Small enough that the backfill never holds # more than a few megabytes, large enough that it isn't a query per row. @@ -257,6 +291,52 @@ def _backfill_embedding_blob(conn) -> None: last_id = rows[-1][0] +def _backfill_context_snapshot(conn) -> None: + """Compress actions.context_snapshot into actions.context_snapshot_z. + + Runs between migration 43 and 44, which is the only window where both + columns exist. Migration 44 drops the original, so unlike every other + backfill here this one is destructive if it is wrong — and every row it + converts is somebody's game. So each row is decompressed again and + compared against what went in before it counts as converted, and a row + that fails to round-trip aborts the whole run rather than being skipped: + the transaction rolls back, the DROP never happens, and the prompts are + still there to try again. + + Reads the JSON the same defensive way as the vector backfill — SQLite + hands back a raw string, psycopg has already parsed it. + """ + last_id = 0 + while True: + rows = conn.execute( + text(""" + SELECT id, context_snapshot FROM actions + WHERE context_snapshot IS NOT NULL + AND context_snapshot_z IS NULL AND id > :last + ORDER BY id LIMIT :batch + """), + {"last": last_id, "batch": SNAPSHOT_BATCH}, + ).all() + if not rows: + return + for row_id, stored in rows: + value = json.loads(stored) if isinstance(stored, str) else stored + if value is None: + continue + packed = compression.pack(value) + if compression.unpack(packed) != value: + raise RuntimeError( + f"context_snapshot for action {row_id} did not survive a " + "compress/decompress round trip; refusing to drop the " + "original column" + ) + conn.execute( + text("UPDATE actions SET context_snapshot_z = :z WHERE id = :id"), + {"z": packed, "id": row_id}, + ) + last_id = rows[-1][0] + + def _get_version(conn) -> int: if conn.dialect.name == "sqlite": return conn.execute(text("PRAGMA user_version")).scalar() or 1 @@ -301,6 +381,11 @@ def bootstrap(engine: Engine) -> None: _backfill_variant_count(conn) if version == EMBEDDING_BLOB_VERSION: _backfill_embedding_blob(conn) + # Must land between 43 (add the column) and 44 (drop the old + # one). The loop is one transaction, so if this raises, the + # DROP rolls back with it and the prompts are still there. + if version == SNAPSHOT_COMPRESS_VERSION: + _backfill_context_snapshot(conn) current = version _set_version(conn, current) _encrypt_plaintext_api_keys(conn) diff --git a/backend/app/models.py b/backend/app/models.py index 625ed7a..9938980 100644 --- a/backend/app/models.py +++ b/backend/app/models.py @@ -6,6 +6,7 @@ from sqlalchemy import ( ) from sqlalchemy.orm import Mapped, mapped_column, relationship +from .compression import CompressedJSON from .database import Base @@ -220,12 +221,19 @@ class Action(Base): # Reasoning-model "thinking" that preceded the text (AI actions only). reasoning: Mapped[str | None] = mapped_column(Text, nullable=True) # The full assembled prompt for this turn, for the Insights viewer. By far - # the biggest column in the database (~74 KB/row in production), and needed - # by exactly one endpoint, one action at a time — so it is deferred: never - # loaded unless something actually touches the attribute. Bulk readers must - # NOT touch it; that is what `world_delta` below exists for. + # the biggest column in the database — 163 KB a row averaged over + # production and 232 KB on the longest adventure, 89% of everything stored + # — and needed by exactly one endpoint, one action at a time. + # + # Two separate defences, because it is expensive in two separate ways. + # `deferred=True` is the read defence: never loaded unless something + # touches the attribute, so a page load pays nothing for it. Bulk readers + # must NOT touch it; that is what `world_delta` below exists for. + # CompressedJSON is the *storage* defence: this is the column that decides + # when the free tier's 512 MB runs out. Still a dict either way — see + # compression.py. context_snapshot: Mapped[dict | None] = mapped_column( - JSON, nullable=True, deferred=True + CompressedJSON, nullable=True, deferred=True ) # The small slice of the snapshot that IS needed in bulk: this turn's RPG # state changes, for the inline chips under an AI message (world_changes) diff --git a/backend/tests/test_egress.py b/backend/tests/test_egress.py index 5159a47..262a053 100644 --- a/backend/tests/test_egress.py +++ b/backend/tests/test_egress.py @@ -27,6 +27,7 @@ os.environ.pop("AIDND_DATABASE_URL", None) os.environ.pop("DATABASE_URL", None) import json +import random import pytest from fastapi import Depends @@ -39,12 +40,20 @@ from app.context import history from app.database import Base, SessionLocal, engine, get_db from app.main import app from tools import dbmeter +from tools.fakeprose import prose # A stand-in for the real thing: the assembled prompt, which is what makes the # column enormous, plus the small world_state slice the UI actually needs. +# +# Varied text, not `"x" * 20_000`. The column is stored compressed now +# (migration 43), and a repeated character compresses about a thousandfold — +# which would make the byte ceilings below pass against a fixture that costs +# nothing, testing nothing. Prose-shaped filler compresses like the prompts +# this stands in for. +_SNAPSHOT_RNG = random.Random(20_260_817) BIG_SNAPSHOT = { - "system": "x" * 20_000, - "story": "y" * 40_000, + "system": prose(_SNAPSHOT_RNG, 20_000), + "story": prose(_SNAPSHOT_RNG, 40_000), "world_state": { "delta": {"player.hp": -15}, "report": {"applied": [{"path": "player.hp", "old": 100, "new": 85}]}, @@ -204,16 +213,37 @@ def test_snapshot_is_still_reachable_on_demand(client): action_id = r.json()["actions"][0]["id"] r = client.get(f"/api/adventures/{client.adv_id}/actions/{action_id}/context") assert r.status_code == 200, r.text - assert r.json()["system"] == "x" * 20_000 + # Round-tripped through zlib and back to a dict, byte for byte. + assert r.json()["system"] == BIG_SNAPSHOT["system"] + assert r.json()["story"] == BIG_SNAPSHOT["story"] # ------------------------------------------------------------------ backfill +def as_json_snapshot_column(db) -> None: + """Put actions.context_snapshot back as JSON, the way it was before 43. + + Migration 36 lifts world_delta out of the snapshot with SQL JSON + functions, so it can only run while the column still *is* JSON. In a real + upgrade it always is — 36 runs seven migrations before 43 compresses the + column into a BLOB — but `create_all` builds today's schema, so a test + calling that backfill has to rebuild the schema it was written against. + """ + db.execute(text("ALTER TABLE actions DROP COLUMN context_snapshot")) + db.execute(text("ALTER TABLE actions ADD COLUMN context_snapshot JSON")) + db.execute( + text("UPDATE actions SET context_snapshot = :snapshot"), + {"snapshot": json.dumps(BIG_SNAPSHOT)}, + ) + db.commit() + + def test_backfill_populates_world_delta_from_existing_snapshots(client): """Migration 36 lifts the slice out server-side, without reading the snapshots into Python.""" db = SessionLocal() try: + as_json_snapshot_column(db) db.execute(text("UPDATE actions SET world_delta = NULL")) db.commit() assert db.query(models.Action).filter(models.Action.world_delta.isnot(None)).count() == 0 @@ -419,9 +449,14 @@ def test_the_ceiling_discriminates(client, meter): """A ceiling is only worth having if the thing it excludes would breach it. This is the regression the byte tests exist to catch, performed on purpose: - undefer the snapshot and the same twelve rows cost two orders of magnitude - more. If this ever stops exceeding the budget, the fixture has gone too - small for the tests above to mean anything. + undefer the snapshot and the same twelve rows cost several times the whole + budget. If this ever stops exceeding it, the fixture has gone too small for + the tests above to mean anything. + + The margin used to be a hundredfold and is now about six. That is not the + guard weakening — it is migration 43 compressing the column, and the + fixture text being prose-shaped so it compresses like a real prompt rather + than like a repeated character. """ budget = ACTIONS_IN_FIXTURE * PAGE_LOAD_BYTES_PER_ACTION db = SessionLocal() @@ -436,8 +471,8 @@ def test_the_ceiling_discriminates(client, meter): finally: db.close() - assert fetched(meter) > budget * 10, ( + assert fetched(meter) > budget * 3, ( "undeferring the snapshot cost only " - f"{fetched(meter):,} B — the fixture is too small for the byte " - "ceilings above to catch anything" + f"{fetched(meter):,} B against a {budget:,} B budget — the fixture is " + "too small for the byte ceilings above to catch anything" ) diff --git a/backend/tests/test_snapshot_compression.py b/backend/tests/test_snapshot_compression.py new file mode 100644 index 0000000..0cde792 --- /dev/null +++ b/backend/tests/test_snapshot_compression.py @@ -0,0 +1,260 @@ +"""context_snapshot, stored compressed. + +The column is 89% of the database and the free tier allows 512 MB. Reads were +solved by deferring it; this is about the storage ceiling. Postgres already +TOASTs it and only gets 1.7x, because pglz is tuned for fast decompression of +data a query might filter on — and nothing ever filters on an assembled +prompt. + +Three things have to hold, and only the first is obvious: + +* what goes in comes back out, exactly, including a snapshot written before + the conversion and one that is NULL; +* the model still hands callers a dict, so no call site changes; +* migration 44 drops the original column, so the backfill is the one + destructive step in this file — it must convert every row or abort. + + python -m pytest tests/test_snapshot_compression.py -v +""" +import json +import os +import random +import tempfile +import zlib + +_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False) +_tmp.close() +os.environ["AIDND_DB_PATH"] = _tmp.name +os.environ.pop("AIDND_DATABASE_URL", None) +os.environ.pop("DATABASE_URL", None) + +import pytest +from sqlalchemy import text + +from app import compression, migrations, models +from app.database import Base, SessionLocal, engine +from tools.fakeprose import prose + + +def snapshot(seed: int, nbytes: int = 60_000) -> dict: + rng = random.Random(seed) + return { + "system": prose(rng, nbytes // 3), + "story": prose(rng, nbytes - nbytes // 3), + "world_state": {"delta": {"player.hp": -3}}, + } + + +@pytest.fixture() +def db(): + Base.metadata.create_all(bind=engine) + session = SessionLocal() + try: + yield session + finally: + session.close() + Base.metadata.drop_all(bind=engine) + + +@pytest.fixture() +def adventure(db): + user = models.User(is_guest=False, email="snap@example.com") + db.add(user) + db.flush() + adv = models.Adventure(user_id=user.id, title="Cave", script_state={}) + db.add(adv) + db.commit() + return adv + + +# ------------------------------------------------------------------- packing + +def test_pack_round_trips_exactly(): + value = snapshot(1) + assert compression.unpack(compression.pack(value)) == value + + +def test_pack_handles_the_awkward_values(): + for value in ({}, {"a": None}, {"nested": {"deep": [1, 2, {"x": "é"}]}}): + assert compression.unpack(compression.pack(value)) == value + + +def test_pack_actually_shrinks_prose(): + """The whole justification. If this ratio collapses, the migration is + spending CPU for nothing.""" + value = snapshot(2, 200_000) + raw = json.dumps(value, separators=(",", ":")).encode() + packed = compression.pack(value) + assert len(packed) < len(raw) / 3, ( + f"{len(raw):,} B compressed to {len(packed):,} B — under 3x" + ) + + +def test_unpack_rejects_nothing_it_wrote(): + packed = compression.pack({"a": "b"}) + assert compression.unpack(bytearray(packed)) == {"a": "b"} + assert compression.unpack(memoryview(bytes(packed))) == {"a": "b"} + + +# --------------------------------------------------------------- the column + +def test_the_column_stores_bytes_and_returns_a_dict(db, adventure): + value = snapshot(3) + action = models.Action( + adventure_id=adventure.id, index=0, type="ai", text="t", + context_snapshot=value, + ) + db.add(action) + db.commit() + db.expire_all() + + assert action.context_snapshot == value + + stored = db.execute( + text("SELECT context_snapshot FROM actions WHERE id = :id"), + {"id": action.id}, + ).scalar() + assert isinstance(stored, (bytes, bytearray, memoryview)) + assert json.loads(zlib.decompress(bytes(stored))) == value + + +def test_the_column_is_smaller_than_the_json_it_holds(db, adventure): + value = snapshot(4, 200_000) + action = models.Action( + adventure_id=adventure.id, index=0, type="ai", text="t", + context_snapshot=value, + ) + db.add(action) + db.commit() + + stored = db.execute( + text("SELECT length(context_snapshot) FROM actions WHERE id = :id"), + {"id": action.id}, + ).scalar() + assert stored < len(json.dumps(value)) / 3 + + +def test_null_stays_null(db, adventure): + action = models.Action( + adventure_id=adventure.id, index=0, type="do", text="t", + context_snapshot=None, + ) + db.add(action) + db.commit() + db.expire_all() + assert action.context_snapshot is None + + +def test_an_unreadable_snapshot_reads_as_none_rather_than_raising(db, adventure): + """One corrupt row must not 500 the turn that happens to load it. The + snapshot is a debugging view; the story is the thing that matters.""" + action = models.Action( + adventure_id=adventure.id, index=0, type="ai", text="t", + context_snapshot={"a": "b"}, + ) + db.add(action) + db.commit() + db.execute( + text("UPDATE actions SET context_snapshot = :junk WHERE id = :id"), + {"junk": b"not zlib at all", "id": action.id}, + ) + db.commit() + db.expire_all() + + assert db.get(models.Action, action.id).context_snapshot is None + + +# ------------------------------------------------------- the upgrade in full + +def as_json_column(db, rows: dict[int, dict]) -> None: + """The pre-43 schema: context_snapshot as a JSON column, populated.""" + db.execute(text("ALTER TABLE actions DROP COLUMN context_snapshot")) + db.execute(text("ALTER TABLE actions ADD COLUMN context_snapshot JSON")) + for action_id, value in rows.items(): + db.execute( + text("UPDATE actions SET context_snapshot = :v WHERE id = :id"), + {"v": json.dumps(value), "id": action_id}, + ) + db.commit() + + +def seed_pre_43(db, adventure, count: int = 4) -> dict[int, dict]: + ids = [] + for i in range(count): + action = models.Action( + adventure_id=adventure.id, index=i, type="ai", text=f"t{i}" + ) + db.add(action) + db.flush() + ids.append(action.id) + # One action with no snapshot at all, which must survive as NULL. + plain = models.Action( + adventure_id=adventure.id, index=count, type="do", text="look" + ) + db.add(plain) + db.commit() + + expected = {action_id: snapshot(action_id) for action_id in ids} + as_json_column(db, expected) + db.execute(text(f"PRAGMA user_version = {migrations.SNAPSHOT_COMPRESS_VERSION - 1}")) + db.commit() + return expected + + +def test_bootstrap_converts_every_snapshot(db, adventure): + expected = seed_pre_43(db, adventure) + db.close() + + migrations.bootstrap(engine) + + check = SessionLocal() + try: + for action_id, value in expected.items(): + assert check.get(models.Action, action_id).context_snapshot == value + assert check.execute( + text("SELECT count(*) FROM actions WHERE context_snapshot IS NULL") + ).scalar() == 1 + finally: + check.close() + + +def test_bootstrap_leaves_the_column_named_context_snapshot(db, adventure): + seed_pre_43(db, adventure) + db.close() + + migrations.bootstrap(engine) + + with engine.begin() as conn: + columns = {row[1] for row in conn.execute(text("PRAGMA table_info(actions)"))} + assert "context_snapshot" in columns + assert "context_snapshot_z" not in columns, "the swap left the scratch column behind" + + +def test_the_backfill_aborts_rather_than_dropping_unconvertible_data( + db, adventure, monkeypatch +): + """Migration 44 destroys the original. If anything cannot be converted the + whole run has to roll back with the column still there — the alternative is + losing somebody's prompts to a bug in this file.""" + seed_pre_43(db, adventure) + db.close() + + def broken_unpack(_blob): + return {"not": "what went in"} + + monkeypatch.setattr(compression, "unpack", broken_unpack) + + with pytest.raises(RuntimeError, match="round trip"): + migrations.bootstrap(engine) + + with engine.begin() as conn: + columns = {row[1] for row in conn.execute(text("PRAGMA table_info(actions)"))} + version = conn.execute(text("PRAGMA user_version")).scalar() + surviving = conn.execute( + text("SELECT count(*) FROM actions WHERE context_snapshot IS NOT NULL") + ).scalar() + + assert "context_snapshot" in columns + assert version < migrations.SNAPSHOT_COMPRESS_VERSION, \ + "the version advanced past a backfill that failed" + assert surviving == 4, "the prompts did not survive the rollback" diff --git a/backend/tools/fakeprose.py b/backend/tools/fakeprose.py new file mode 100644 index 0000000..a0ca9e7 --- /dev/null +++ b/backend/tools/fakeprose.py @@ -0,0 +1,38 @@ +"""Filler text that behaves like prose under compression. + +Shared by the stress harness and the egress tests, because both now measure +something a repeated string would answer wrongly. `"x" * 20_000` compresses +about a thousandfold and English three- or fourfold, so a fixture built from +repeats makes a compressed column look free and any ceiling drawn around it +meaningless. + +Not a language model, and does not need to be. What matters is the symbol +distribution, not the sense. +""" +from __future__ import annotations + +import random + +WORDS = ( + "corridor narrows shoulders brush wet stone torchlight gutters draught " + "smells cold iron somewhere ahead water moving count nine paces passage " + "opens chamber ceiling lost dark sound breathing comes back half second " + "late Gwen catches sleeve without word points floor line pale grit laid " + "across threshold deliberate arc quartermaster bandit camp above ford " + "tunnels exchange key lantern rope knife bread rain mud hill road gate " + "watchman silver debt promise fever horse cart river bridge mill barley " + "smoke rafters bench ale ledger seal wax parchment ink candle shutter " + "hinge bolt cellar barrel salt fish nets harbour tide gull mast canvas" +).split() + + +def prose(rng: random.Random, nbytes: int) -> str: + """Roughly `nbytes` of varied sentences, deterministic for a given rng.""" + out: list[str] = [] + total = 0 + while total < nbytes: + sentence = " ".join(rng.choice(WORDS) for _ in range(rng.randint(8, 18))) + chunk = sentence.capitalize() + ". " + out.append(chunk) + total += len(chunk) + return "".join(out)[:nbytes] diff --git a/backend/tools/stress_session.py b/backend/tools/stress_session.py index 9bb2ce1..a9b9f23 100644 --- a/backend/tools/stress_session.py +++ b/backend/tools/stress_session.py @@ -108,6 +108,7 @@ from app.providers import PromptParts from app.routers import adventures from .dbmeter import Meter, kb +from .fakeprose import prose EMBEDDING_DIMS = 1536 @@ -123,37 +124,6 @@ EMBEDDING_DIMS = 1536 SNAPSHOT_SYSTEM = None # set by _build_text() SNAPSHOT_STORY = None -_WORDS = ( - "corridor narrows shoulders brush wet stone torchlight gutters draught " - "smells cold iron somewhere ahead water moving count nine paces passage " - "opens chamber ceiling lost dark sound breathing comes back half second " - "late Gwen catches sleeve without word points floor line pale grit laid " - "across threshold deliberate arc quartermaster bandit camp above ford " - "tunnels exchange key lantern rope knife bread rain mud hill road gate " - "watchman silver debt promise fever horse cart river bridge mill barley " - "smoke rafters bench ale ledger seal wax parchment ink candle shutter " - "hinge bolt cellar barrel salt fish nets harbour tide gull mast canvas" -).split() - - -def prose(rng: random.Random, nbytes: int) -> str: - """Filler of about `nbytes`, varied enough to compress like prose. - - Not decoration. A sentence repeated N times compresses roughly a - hundredfold and English roughly three- or fourfold, so a fixture built out - of repeats would report a compression ratio that says nothing about the - real column — and that ratio is the whole question for context_snapshot. - """ - out: list[str] = [] - total = 0 - while total < nbytes: - sentence = " ".join(rng.choice(_WORDS) for _ in range(rng.randint(8, 18))) - chunk = sentence.capitalize() + ". " - out.append(chunk) - total += len(chunk) - return "".join(out)[:nbytes] - - PLAYER_INPUT = "> You crouch and look more closely at the grit on the floor." # All three are bound by _build_text() from the fixture arguments.