Cut database egress and storage
Cut database egress and storage: window the story, compress the prompts
This commit is contained in:
@@ -0,0 +1,70 @@
|
||||
"""Storing a JSON column compressed.
|
||||
|
||||
`actions.context_snapshot` holds the entire assembled prompt for a turn. It is
|
||||
89% of the database — 150.8 MB of JSON across 944 actions on production, and
|
||||
232 KB a row on the longest adventure — and the free tier this deploys to
|
||||
allows 512 MB. Reads are not the problem: the column is deferred, so a page
|
||||
load never touches it and exactly one endpoint fetches one row of it at a
|
||||
time. Storage is the problem, and storage has a cliff.
|
||||
|
||||
Postgres already compresses it. TOAST brings 150.8 MB down to ~89 MB, a factor
|
||||
of 1.7 — pglz is chosen for decompression speed on data a query might filter
|
||||
on, which this never is. Nothing filters on a prompt; it is written once and
|
||||
read whole, occasionally, by one screen. zlib at the application layer gets
|
||||
three to four times on the same text, and the cost is a decompress on a
|
||||
request that already costs an LLM call.
|
||||
|
||||
Doing it as a TypeDecorator rather than a second column keeps every call site
|
||||
writing `action.context_snapshot = {...}` and reading a dict back, and keeps
|
||||
`deferred=True`, `undefer()` and `load_only()` naming the same attribute they
|
||||
named before. The storage format changes; nothing else does.
|
||||
|
||||
Level 6 is zlib's default and the knee of the curve here: 9 spends noticeably
|
||||
more CPU on prompt text for about a percent more space.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import zlib
|
||||
|
||||
from sqlalchemy import LargeBinary
|
||||
from sqlalchemy.types import TypeDecorator
|
||||
|
||||
LEVEL = 6
|
||||
|
||||
|
||||
def pack(value) -> bytes:
|
||||
"""A JSON-able value as compressed UTF-8."""
|
||||
raw = json.dumps(value, separators=(",", ":"), default=str).encode("utf-8")
|
||||
return zlib.compress(raw, LEVEL)
|
||||
|
||||
|
||||
def unpack(blob: bytes) -> object:
|
||||
"""The value `pack` was given."""
|
||||
return json.loads(zlib.decompress(bytes(blob)).decode("utf-8"))
|
||||
|
||||
|
||||
class CompressedJSON(TypeDecorator):
|
||||
"""A JSON column stored as zlib-compressed UTF-8 in a BLOB/BYTEA.
|
||||
|
||||
`cache_ok = True`: the type carries no per-instance configuration, so
|
||||
SQLAlchemy may reuse a compiled statement across instances of it.
|
||||
"""
|
||||
|
||||
impl = LargeBinary
|
||||
cache_ok = True
|
||||
|
||||
def process_bind_param(self, value, dialect):
|
||||
return None if value is None else pack(value)
|
||||
|
||||
def process_result_value(self, value, dialect):
|
||||
# Tolerate a row the backfill has not reached yet, or one written
|
||||
# before the conversion: a snapshot that cannot be read back is worth
|
||||
# less than the screen that shows it, and never worth a 500 on the
|
||||
# turn that happens to load it.
|
||||
if value is None:
|
||||
return None
|
||||
try:
|
||||
return unpack(value)
|
||||
except (zlib.error, UnicodeDecodeError, ValueError):
|
||||
return None
|
||||
@@ -86,13 +86,15 @@ def set_vector(memory: models.Memory, vector: list[float] | None) -> None:
|
||||
"""Store (or clear) a memory's embedding.
|
||||
|
||||
Every column that describes the vector moves together: `embedding_blob` is
|
||||
what the ranking reads, `embedded` is the flag everything else reads, and
|
||||
the JSON `embedding` stays correct behind both until the follow-up
|
||||
migration drops it. Going through one function is what keeps them in step —
|
||||
and it is also the only place a stored vector can change, which is what
|
||||
makes the cache below safe to invalidate here and nowhere else.
|
||||
what the ranking reads and `embedded` is the flag everything else reads.
|
||||
Going through one function is what keeps them in step — and it is also the
|
||||
only place a stored vector can change, which is what makes the cache below
|
||||
safe to invalidate here and nowhere else.
|
||||
|
||||
The one caller that legitimately cannot come through here is the bulk
|
||||
clear in `routers/settings.py` when the embedding model changes. It has to
|
||||
set the same two columns by hand; see the note there.
|
||||
"""
|
||||
memory.embedding = vector
|
||||
memory.embedding_blob = None if vector is None else vectors.pack(vector)
|
||||
memory.embedded = vector is not None
|
||||
cached = _vector_cache.get(memory.adventure_id)
|
||||
|
||||
+100
-1
@@ -22,7 +22,7 @@ import json
|
||||
from sqlalchemy import inspect, text
|
||||
from sqlalchemy.engine import Engine
|
||||
|
||||
from . import vectors
|
||||
from . import compression, vectors
|
||||
from .database import Base
|
||||
|
||||
# (version, SQL to run when upgrading past it) — append only, never reorder.
|
||||
@@ -144,6 +144,40 @@ MIGRATIONS: list[tuple[int, str | dict[str, str]]] = [
|
||||
# so anyone who picked a value keeps it — same rule as migration 29.
|
||||
# Adventures already over 80 evict down on their next turn.
|
||||
(41, "UPDATE settings SET memory_bank_capacity = 80 WHERE memory_bank_capacity = 200"),
|
||||
# The JSON vectors, gone. Migration 38 left them in place so a rollback
|
||||
# could still find them; production has since been verified reading from
|
||||
# embedding_blob (schema_version 41, 134/134 backfilled), so the column is
|
||||
# now 4 MB of a 99.6 MB database holding nothing anyone reads. DROP COLUMN
|
||||
# is spelled the same on both dialects — SQLite has had it since 3.35.
|
||||
(42, "ALTER TABLE memories DROP COLUMN embedding"),
|
||||
# context_snapshot, compressed. 89% of the database is one column holding
|
||||
# assembled prompts nobody filters on and one screen reads, one row at a
|
||||
# time; Postgres already TOASTs it, but pglz only manages 1.7x and zlib
|
||||
# gets three to four on the same text. Reads were fixed by deferring it —
|
||||
# this is about the 512 MB the free tier allows.
|
||||
#
|
||||
# Three steps because a column cannot portably change type in place: add
|
||||
# the new one, convert into it (_backfill_context_snapshot, which verifies
|
||||
# every row round-trips before the old column goes), then swap the names so
|
||||
# the model keeps calling it context_snapshot.
|
||||
#
|
||||
# **Postgres does not hand the disk back on its own.** DROP COLUMN only
|
||||
# marks the column dropped, and the backfill's UPDATE leaves a dead tuple
|
||||
# per row, so the table gets *bigger* before it gets smaller: peak is
|
||||
# roughly twice the starting size while both columns are live. Plain
|
||||
# autovacuum makes that space reusable but does not shrink the files. The
|
||||
# deploy that ships this should follow it with, once:
|
||||
#
|
||||
# VACUUM FULL actions;
|
||||
#
|
||||
# which needs exclusive access and free space equal to the finished table.
|
||||
# On the 2026-08-17 figures that is 99.6 MB peaking near 200, settling at
|
||||
# about 53 once vacuumed, against a 512 MB tier. Skipping the vacuum is
|
||||
# safe and simply leaves the win unrealised.
|
||||
(43, {"sqlite": "ALTER TABLE actions ADD COLUMN context_snapshot_z BLOB",
|
||||
"default": "ALTER TABLE actions ADD COLUMN context_snapshot_z BYTEA"}),
|
||||
(44, "ALTER TABLE actions DROP COLUMN context_snapshot"),
|
||||
(45, "ALTER TABLE actions RENAME COLUMN context_snapshot_z TO context_snapshot"),
|
||||
]
|
||||
|
||||
LATEST_VERSION = max((v for v, _ in MIGRATIONS), default=1)
|
||||
@@ -152,6 +186,12 @@ LATEST_VERSION = max((v for v, _ in MIGRATIONS), default=1)
|
||||
WORLD_DELTA_VERSION = 36
|
||||
VARIANT_COUNT_VERSION = 37
|
||||
EMBEDDING_BLOB_VERSION = 38
|
||||
SNAPSHOT_COMPRESS_VERSION = 43
|
||||
|
||||
# Snapshots converted per round trip. Deliberately far smaller than
|
||||
# BACKFILL_BATCH: a vector is 6 KB and a snapshot is 232 KB, so 200 of these
|
||||
# would be 46 MB held at once.
|
||||
SNAPSHOT_BATCH = 50
|
||||
|
||||
# Vectors converted per round trip. Small enough that the backfill never holds
|
||||
# more than a few megabytes, large enough that it isn't a query per row.
|
||||
@@ -251,6 +291,60 @@ def _backfill_embedding_blob(conn) -> None:
|
||||
last_id = rows[-1][0]
|
||||
|
||||
|
||||
def _backfill_context_snapshot(conn) -> None:
|
||||
"""Compress actions.context_snapshot into actions.context_snapshot_z.
|
||||
|
||||
Runs between migration 43 and 44, which is the only window where both
|
||||
columns exist. Migration 44 drops the original, so unlike every other
|
||||
backfill here this one is destructive if it is wrong — and every row it
|
||||
converts is somebody's game. So each row is decompressed again and
|
||||
compared against what went in before it counts as converted, and a row
|
||||
that fails to round-trip aborts the whole run rather than being skipped:
|
||||
the transaction rolls back, the DROP never happens, and the prompts are
|
||||
still there to try again.
|
||||
|
||||
Reads the JSON the same defensive way as the vector backfill — SQLite
|
||||
hands back a raw string, psycopg has already parsed it.
|
||||
"""
|
||||
last_id = 0
|
||||
while True:
|
||||
rows = conn.execute(
|
||||
text("""
|
||||
SELECT id, context_snapshot FROM actions
|
||||
WHERE context_snapshot IS NOT NULL
|
||||
AND context_snapshot_z IS NULL AND id > :last
|
||||
ORDER BY id LIMIT :batch
|
||||
"""),
|
||||
{"last": last_id, "batch": SNAPSHOT_BATCH},
|
||||
).all()
|
||||
if not rows:
|
||||
return
|
||||
params = []
|
||||
for row_id, stored in rows:
|
||||
value = json.loads(stored) if isinstance(stored, str) else stored
|
||||
if value is None:
|
||||
continue
|
||||
packed = compression.pack(value)
|
||||
if compression.unpack(packed) != value:
|
||||
raise RuntimeError(
|
||||
f"context_snapshot for action {row_id} did not survive a "
|
||||
"compress/decompress round trip; refusing to drop the "
|
||||
"original column"
|
||||
)
|
||||
params.append({"z": packed, "id": row_id})
|
||||
if params:
|
||||
# One executemany per batch, not one statement per row. This runs
|
||||
# at container start, before the port opens, against a database on
|
||||
# the other end of a network: a thousand round trips is the
|
||||
# difference between a deploy that comes up and a health check that
|
||||
# times out waiting for it.
|
||||
conn.execute(
|
||||
text("UPDATE actions SET context_snapshot_z = :z WHERE id = :id"),
|
||||
params,
|
||||
)
|
||||
last_id = rows[-1][0]
|
||||
|
||||
|
||||
def _get_version(conn) -> int:
|
||||
if conn.dialect.name == "sqlite":
|
||||
return conn.execute(text("PRAGMA user_version")).scalar() or 1
|
||||
@@ -295,6 +389,11 @@ def bootstrap(engine: Engine) -> None:
|
||||
_backfill_variant_count(conn)
|
||||
if version == EMBEDDING_BLOB_VERSION:
|
||||
_backfill_embedding_blob(conn)
|
||||
# Must land between 43 (add the column) and 44 (drop the old
|
||||
# one). The loop is one transaction, so if this raises, the
|
||||
# DROP rolls back with it and the prompts are still there.
|
||||
if version == SNAPSHOT_COMPRESS_VERSION:
|
||||
_backfill_context_snapshot(conn)
|
||||
current = version
|
||||
_set_version(conn, current)
|
||||
_encrypt_plaintext_api_keys(conn)
|
||||
|
||||
+13
-9
@@ -6,6 +6,7 @@ from sqlalchemy import (
|
||||
)
|
||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||
|
||||
from .compression import CompressedJSON
|
||||
from .database import Base
|
||||
|
||||
|
||||
@@ -158,10 +159,6 @@ class Memory(Base):
|
||||
id: Mapped[int] = mapped_column(primary_key=True)
|
||||
adventure_id: Mapped[int] = mapped_column(ForeignKey("adventures.id", ondelete="CASCADE"))
|
||||
text: Mapped[str] = mapped_column(Text, default="")
|
||||
# Superseded by embedding_blob and still written alongside it, so a
|
||||
# rollback finds the vectors, until a follow-up migration drops it. Nothing
|
||||
# reads it.
|
||||
embedding: Mapped[list | None] = mapped_column(JSON, nullable=True, deferred=True)
|
||||
# The vector, little-endian float32. Deferred because it is wider than the
|
||||
# rest of the row put together and exactly one code path wants it: anything
|
||||
# bulk-loading memories (the Memories drawer, eviction, the embed queue)
|
||||
@@ -224,12 +221,19 @@ class Action(Base):
|
||||
# Reasoning-model "thinking" that preceded the text (AI actions only).
|
||||
reasoning: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
# The full assembled prompt for this turn, for the Insights viewer. By far
|
||||
# the biggest column in the database (~74 KB/row in production), and needed
|
||||
# by exactly one endpoint, one action at a time — so it is deferred: never
|
||||
# loaded unless something actually touches the attribute. Bulk readers must
|
||||
# NOT touch it; that is what `world_delta` below exists for.
|
||||
# the biggest column in the database — 163 KB a row averaged over
|
||||
# production and 232 KB on the longest adventure, 89% of everything stored
|
||||
# — and needed by exactly one endpoint, one action at a time.
|
||||
#
|
||||
# Two separate defences, because it is expensive in two separate ways.
|
||||
# `deferred=True` is the read defence: never loaded unless something
|
||||
# touches the attribute, so a page load pays nothing for it. Bulk readers
|
||||
# must NOT touch it; that is what `world_delta` below exists for.
|
||||
# CompressedJSON is the *storage* defence: this is the column that decides
|
||||
# when the free tier's 512 MB runs out. Still a dict either way — see
|
||||
# compression.py.
|
||||
context_snapshot: Mapped[dict | None] = mapped_column(
|
||||
JSON, nullable=True, deferred=True
|
||||
CompressedJSON, nullable=True, deferred=True
|
||||
)
|
||||
# The small slice of the snapshot that IS needed in bulk: this turn's RPG
|
||||
# state changes, for the inline chips under an AI message (world_changes)
|
||||
|
||||
@@ -6,7 +6,8 @@ import threading
|
||||
from fastapi import APIRouter, Body, Depends, HTTPException, Request
|
||||
from fastapi.responses import StreamingResponse
|
||||
from sqlalchemy import func
|
||||
from sqlalchemy.orm import Session, undefer
|
||||
from sqlalchemy.orm import Session, load_only, undefer
|
||||
from sqlalchemy.orm.attributes import set_committed_value
|
||||
|
||||
from .. import auth, images, limits, memorybank, models, schemas, worldstate
|
||||
from ..context import build_context
|
||||
@@ -20,6 +21,114 @@ router = APIRouter(prefix="/api/adventures", tags=["adventures"])
|
||||
|
||||
CurrentUser = Depends(auth.get_current_user)
|
||||
|
||||
# Exactly what schemas.ActionOut renders, named rather than implied.
|
||||
#
|
||||
# `deferred=True` in models.py already keeps the four heavy columns out of a
|
||||
# bulk read, but it makes narrowness the default that a *future* column has to
|
||||
# remember to ask for — and both egress blowouts this project has had were a
|
||||
# column nobody remembered. Listing what a list response carries inverts that:
|
||||
# a new column costs nothing here until someone adds it to this tuple.
|
||||
#
|
||||
# `world_delta` is on the list because ActionOut.world_changes is computed from
|
||||
# it. Leaving it off would not save the bytes, it would spend them one row at a
|
||||
# time as a lazy load, which is worse.
|
||||
ACTION_LIST_COLUMNS = (
|
||||
models.Action.adventure_id,
|
||||
models.Action.index,
|
||||
models.Action.type,
|
||||
models.Action.text,
|
||||
models.Action.reasoning,
|
||||
models.Action.world_delta,
|
||||
models.Action.variant_count,
|
||||
models.Action.variant_index,
|
||||
models.Action.created_at,
|
||||
)
|
||||
|
||||
# How many actions an adventure opens with, and how many arrive per scroll.
|
||||
#
|
||||
# Opening a finished adventure used to fetch the whole story in one response —
|
||||
# 589.5 kB on production's longest, and growing, because a story only ever gets
|
||||
# longer. 60 is a few screens of reading: enough that the common case (open,
|
||||
# read the end, take a turn) never pages at all, small enough that the worst
|
||||
# case is bounded by the window rather than by the story.
|
||||
ACTION_PAGE = 60
|
||||
|
||||
|
||||
def action_window(
|
||||
db: Session,
|
||||
adventure_id: int,
|
||||
before_id: int | None = None,
|
||||
limit: int = ACTION_PAGE,
|
||||
) -> tuple[list[models.Action], int, bool]:
|
||||
"""The `limit` actions immediately older than `before_id`, oldest first.
|
||||
|
||||
Returns (actions, total, has_more). `before_id=None` is the newest window.
|
||||
|
||||
Anchored on an action, not on a count, and never on arithmetic over
|
||||
`Action.index`. Two separate reasons, and both bite:
|
||||
|
||||
* **Appends.** Counting back from the newest means every older position
|
||||
shifts when a turn lands. A reader who scrolls up while a turn is
|
||||
generating would be handed a window one row out — re-sending one action
|
||||
and silently skipping another. An anchor is fixed: "older than this one"
|
||||
means the same thing before and after the story grows.
|
||||
* **The story tree.** Index is a dense 0..n sequence today and branching
|
||||
ends that. Comparing indices to order a branch survives; treating them as
|
||||
positions does not.
|
||||
|
||||
`has_more` comes from asking for one row past the window rather than from
|
||||
counting, so it costs a row and not a scan.
|
||||
"""
|
||||
total = (
|
||||
db.query(func.count(models.Action.id))
|
||||
.filter(models.Action.adventure_id == adventure_id)
|
||||
.scalar()
|
||||
)
|
||||
if limit <= 0:
|
||||
return [], total, total > 0
|
||||
|
||||
query = (
|
||||
db.query(models.Action)
|
||||
.options(load_only(*ACTION_LIST_COLUMNS))
|
||||
.filter(models.Action.adventure_id == adventure_id)
|
||||
)
|
||||
if before_id is not None:
|
||||
anchor = (
|
||||
db.query(models.Action.index)
|
||||
.filter(models.Action.id == before_id,
|
||||
models.Action.adventure_id == adventure_id)
|
||||
.scalar()
|
||||
)
|
||||
if anchor is None:
|
||||
# The anchor was deleted (undo, or a turn edited away) while the
|
||||
# reader was scrolling. Nothing older can be identified relative to
|
||||
# a row that no longer exists, so report the end rather than
|
||||
# guessing and handing back a duplicate page.
|
||||
return [], total, False
|
||||
query = query.filter(models.Action.index < anchor)
|
||||
|
||||
rows = query.order_by(models.Action.index.desc()).limit(limit + 1).all()
|
||||
has_more = len(rows) > limit
|
||||
rows = rows[:limit]
|
||||
rows.reverse()
|
||||
return rows, total, has_more
|
||||
|
||||
|
||||
# Exactly what schemas.MemoryOut renders. `embedded` is a real column and is on
|
||||
# the list; the vector it describes is not, and must never be.
|
||||
MEMORY_LIST_COLUMNS = (
|
||||
models.Memory.adventure_id,
|
||||
models.Memory.text,
|
||||
models.Memory.pinned,
|
||||
models.Memory.forgotten,
|
||||
models.Memory.embedded,
|
||||
models.Memory.use_count,
|
||||
models.Memory.last_used_at,
|
||||
models.Memory.source_start,
|
||||
models.Memory.source_end,
|
||||
models.Memory.created_at,
|
||||
)
|
||||
|
||||
|
||||
def get_adventure_or_404(
|
||||
adventure_id: int, db: Session, user: models.User
|
||||
@@ -85,9 +194,18 @@ def _latest_narration(db: Session, adventure_ids: list[int]) -> dict[int, str]:
|
||||
|
||||
@router.get("", response_model=list[schemas.AdventureListItem])
|
||||
def list_adventures(db: Session = Depends(get_db), user: models.User = CurrentUser):
|
||||
# Four columns of Adventure, named, rather than the entity. The entity is
|
||||
# sixteen columns wide and carries script_state, world_state, placeholders,
|
||||
# story_summary, memory, authors_note and ai_instructions — ~15 kB a row in
|
||||
# production, none of it on this screen, all of it fetched once per
|
||||
# adventure every time the index loads. Naming the columns also means the
|
||||
# next wide column added to Adventure has to opt *in* to being listed here.
|
||||
rows = (
|
||||
db.query(
|
||||
models.Adventure,
|
||||
models.Adventure.id,
|
||||
models.Adventure.scenario_id,
|
||||
models.Adventure.title,
|
||||
models.Adventure.updated_at,
|
||||
func.count(models.Action.id),
|
||||
models.Scenario.title,
|
||||
models.Scenario.image,
|
||||
@@ -112,22 +230,23 @@ def list_adventures(db: Session = Depends(get_db), user: models.User = CurrentUs
|
||||
.order_by(models.Adventure.updated_at.desc())
|
||||
.all()
|
||||
)
|
||||
narration = _latest_narration(db, [adv.id for adv, *_ in rows])
|
||||
narration = _latest_narration(db, [row[0] for row in rows])
|
||||
return [
|
||||
schemas.AdventureListItem(
|
||||
id=adv.id,
|
||||
scenario_id=adv.scenario_id,
|
||||
id=adv_id,
|
||||
scenario_id=scenario_id,
|
||||
scenario_title=scenario_title,
|
||||
title=adv.title,
|
||||
updated_at=adv.updated_at,
|
||||
title=title,
|
||||
updated_at=updated_at,
|
||||
action_count=count,
|
||||
snippet=_snippet(narration.get(adv.id, "")),
|
||||
snippet=_snippet(narration.get(adv_id, "")),
|
||||
# The art belongs to the scenario, so the cache-busting stamp is the
|
||||
# scenario's updated_at, not the adventure's.
|
||||
image_url=images.public_url(adv.scenario_id, image or "", scenario_updated),
|
||||
image_url=images.public_url(scenario_id, image or "", scenario_updated),
|
||||
icon=icon or "",
|
||||
)
|
||||
for adv, count, scenario_title, image, icon, scenario_updated in rows
|
||||
for (adv_id, scenario_id, title, updated_at, count,
|
||||
scenario_title, image, icon, scenario_updated) in rows
|
||||
]
|
||||
|
||||
|
||||
@@ -255,7 +374,25 @@ def create_adventure(
|
||||
def get_adventure(
|
||||
adventure_id: int, db: Session = Depends(get_db), user: models.User = CurrentUser
|
||||
):
|
||||
return get_adventure_or_404(adventure_id, db, user)
|
||||
"""The adventure, and the newest window of its story.
|
||||
|
||||
`actions` is the last ACTION_PAGE, not all of them; `action_count` says how
|
||||
many there are so the reader knows there is more above. Older pages come
|
||||
from GET /{id}/actions as they scroll up.
|
||||
"""
|
||||
adventure = get_adventure_or_404(adventure_id, db, user)
|
||||
actions, total, _ = action_window(db, adventure_id)
|
||||
# Hand the response the window as if the relationship had loaded it.
|
||||
# `set_committed_value` is the only way to do this safely: assigning
|
||||
# `adventure.actions = [...]` marks the collection dirty, and the
|
||||
# relationship cascades delete-orphan, so the actions left out of the
|
||||
# window would be deleted on the next flush. This records them as the
|
||||
# loaded, unmodified value instead, so serialising touches no lazy load
|
||||
# and nothing is pending.
|
||||
set_committed_value(adventure, "actions", actions)
|
||||
out = schemas.AdventureOut.model_validate(adventure)
|
||||
out.action_count = total
|
||||
return out
|
||||
|
||||
|
||||
@router.get("/{adventure_id}/script-state")
|
||||
@@ -902,7 +1039,7 @@ def select_variant(
|
||||
_active_turns.discard(adventure_id)
|
||||
|
||||
|
||||
@router.post("/{adventure_id}/undo", response_model=list[schemas.ActionOut])
|
||||
@router.post("/{adventure_id}/undo", response_model=schemas.ActionPage)
|
||||
def undo_turn(
|
||||
adventure_id: int, db: Session = Depends(get_db), user: models.User = CurrentUser
|
||||
):
|
||||
@@ -944,7 +1081,16 @@ def undo_turn(
|
||||
memorybank.prune_dangling_memories(adventure, db)
|
||||
db.commit()
|
||||
db.refresh(adventure)
|
||||
return adventure.actions
|
||||
# The newest window, not the whole story: the client replaces its
|
||||
# transcript with this, and the transcript is a window now. Returning
|
||||
# everything here would undo the paging on the one action most likely
|
||||
# to be repeated several times in a row.
|
||||
actions, total, has_more = action_window(db, adventure_id)
|
||||
return schemas.ActionPage(
|
||||
actions=[schemas.ActionOut.model_validate(a) for a in actions],
|
||||
total=total,
|
||||
has_more=has_more,
|
||||
)
|
||||
finally:
|
||||
_active_turns.discard(adventure_id)
|
||||
|
||||
@@ -1455,7 +1601,19 @@ def action_context(
|
||||
def list_memories(
|
||||
adventure_id: int, db: Session = Depends(get_db), user: models.User = CurrentUser
|
||||
):
|
||||
return get_adventure_or_404(adventure_id, db, user).memories
|
||||
get_adventure_or_404(adventure_id, db, user)
|
||||
# A query naming its columns, not a walk of `adventure.memories`. The walk
|
||||
# is what retrieval used to do, and it is the reason a turn cost megabytes:
|
||||
# a relationship load takes whole entities, so it picks up whatever the
|
||||
# model happens to carry. `embedding_blob` is deferred and so would stay
|
||||
# out today — this is about the next wide column, not that one.
|
||||
return (
|
||||
db.query(models.Memory)
|
||||
.options(load_only(*MEMORY_LIST_COLUMNS))
|
||||
.filter(models.Memory.adventure_id == adventure_id)
|
||||
.order_by(models.Memory.id)
|
||||
.all()
|
||||
)
|
||||
|
||||
|
||||
@router.post("/{adventure_id}/memories", response_model=schemas.MemoryOut, status_code=201)
|
||||
@@ -1515,16 +1673,29 @@ def delete_memory(
|
||||
|
||||
# ---------- Actions (CRUD) ----------
|
||||
|
||||
@router.get("/{adventure_id}/actions", response_model=list[schemas.ActionOut])
|
||||
@router.get("/{adventure_id}/actions", response_model=schemas.ActionPage)
|
||||
def list_actions(
|
||||
adventure_id: int, db: Session = Depends(get_db), user: models.User = CurrentUser
|
||||
adventure_id: int,
|
||||
before_id: int | None = None,
|
||||
limit: int = ACTION_PAGE,
|
||||
db: Session = Depends(get_db),
|
||||
user: models.User = CurrentUser,
|
||||
):
|
||||
"""A page of the story, walking backwards from the newest action.
|
||||
|
||||
`before_id` is the oldest action the caller already holds, so scrolling up
|
||||
is "give me what comes before this". Omit it for the newest window. See
|
||||
action_window for why this anchors on a row rather than an offset.
|
||||
"""
|
||||
get_adventure_or_404(adventure_id, db, user)
|
||||
return (
|
||||
db.query(models.Action)
|
||||
.filter(models.Action.adventure_id == adventure_id)
|
||||
.order_by(models.Action.index)
|
||||
.all()
|
||||
limit = max(1, min(limit, ACTION_PAGE * 4))
|
||||
actions, total, has_more = action_window(
|
||||
db, adventure_id, before_id=before_id, limit=limit
|
||||
)
|
||||
return schemas.ActionPage(
|
||||
actions=[schemas.ActionOut.model_validate(a) for a in actions],
|
||||
total=total,
|
||||
has_more=has_more,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -51,14 +51,26 @@ def update_settings(
|
||||
# Vectors from the old model have a different dimensionality/space;
|
||||
# clear them so the post-turn task re-embeds with the new model.
|
||||
# (This user's adventures only — settings are per-user now.)
|
||||
#
|
||||
# Both columns, and the flag. This is the one place that clears vectors
|
||||
# in bulk rather than through memorybank.set_vector, and when the
|
||||
# vectors moved to embedding_blob it kept nulling the old JSON column
|
||||
# alone: the blob survived, `embedded` stayed true, and _embed_pending
|
||||
# — which looks for embedded IS FALSE — never picked the rows up. The
|
||||
# bank went on ranking against the previous model's vectors forever.
|
||||
owned = (
|
||||
db.query(models.Adventure.id)
|
||||
.filter(models.Adventure.user_id == user.id)
|
||||
.scalar_subquery()
|
||||
)
|
||||
db.query(models.Memory).filter(models.Memory.adventure_id.in_(owned)).update(
|
||||
{"embedding": None}, synchronize_session=False
|
||||
{"embedding_blob": None, "embedded": False}, synchronize_session=False
|
||||
)
|
||||
# No cache invalidation needed, and deliberately none added: clearing
|
||||
# `embedded` drops these rows out of the catalogue query, so retrieval
|
||||
# stops asking for them, and by the time _embed_pending puts one back
|
||||
# it has gone through set_vector, which evicts that entry. The rule
|
||||
# holds — anything that removes a memory from play self-corrects.
|
||||
db.commit()
|
||||
return settings
|
||||
|
||||
|
||||
@@ -225,7 +225,21 @@ class AdventureOut(ORMModel):
|
||||
created_at: datetime
|
||||
updated_at: datetime
|
||||
story_cards: list[StoryCardOut] = []
|
||||
# The NEWEST window of the story, not all of it — older pages arrive from
|
||||
# GET /{id}/actions as the reader scrolls up. `action_count` is the whole
|
||||
# story's length, which is how the client knows there is more above.
|
||||
actions: list[ActionOut] = []
|
||||
action_count: int = 0
|
||||
|
||||
|
||||
class ActionPage(BaseModel):
|
||||
"""A slice of the story, counted back from the newest action."""
|
||||
|
||||
actions: list[ActionOut] = []
|
||||
total: int = 0
|
||||
# Whether anything older than this slice exists. Computed server-side so
|
||||
# the client never has to do arithmetic on positions to find the end.
|
||||
has_more: bool = False
|
||||
|
||||
|
||||
# ---------- Memory bank (Phase 6) ----------
|
||||
|
||||
@@ -0,0 +1,240 @@
|
||||
"""Opening an adventure fetches a window, not the whole story.
|
||||
|
||||
A story only ever gets longer. Production's longest is 607 actions and 589.5 kB
|
||||
in one response, and nothing about that curve bends on its own — so the page
|
||||
load returns the newest ACTION_PAGE and the reader pages upward.
|
||||
|
||||
The paging anchors on an action id rather than an offset, and these tests are
|
||||
mostly about why. An offset counted back from the newest shifts every older
|
||||
position the moment a turn lands, which is precisely when a reader is likely
|
||||
to be scrolling. An anchor means the same thing before and after.
|
||||
|
||||
python -m pytest tests/test_action_paging.py -v
|
||||
"""
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False)
|
||||
_tmp.close()
|
||||
os.environ["AIDND_DB_PATH"] = _tmp.name
|
||||
os.environ.pop("AIDND_DATABASE_URL", None)
|
||||
os.environ.pop("DATABASE_URL", None)
|
||||
|
||||
import pytest
|
||||
from fastapi import Depends
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from app import auth, limits, models
|
||||
from app.database import Base, SessionLocal, engine, get_db
|
||||
from app.main import app
|
||||
from app.routers.adventures import ACTION_PAGE
|
||||
from tools import dbmeter
|
||||
|
||||
TOTAL = ACTION_PAGE * 3 + 7 # deliberately not a whole number of pages
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
Base.metadata.create_all(bind=engine)
|
||||
setup = SessionLocal()
|
||||
user = models.User(is_guest=False, email="paging@example.com")
|
||||
setup.add(user)
|
||||
setup.flush()
|
||||
setup.add(models.Settings(user_id=user.id, api_key="enc:dummy", model="m"))
|
||||
adventure = models.Adventure(user_id=user.id, title="Cave", script_state={})
|
||||
setup.add(adventure)
|
||||
setup.flush()
|
||||
for i in range(TOTAL):
|
||||
setup.add(models.Action(
|
||||
adventure_id=adventure.id, index=i,
|
||||
type="start" if i == 0 else ("ai" if i % 2 else "do"),
|
||||
text=f"Action {i}." + "word " * 200,
|
||||
))
|
||||
setup.commit()
|
||||
adv_id, user_id = adventure.id, user.id
|
||||
setup.close()
|
||||
|
||||
monkeypatch.setattr(limits, "rate_limit", lambda *a, **k: None)
|
||||
monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None)
|
||||
|
||||
def _current_user(db=Depends(get_db)):
|
||||
return db.get(models.User, user_id)
|
||||
|
||||
app.dependency_overrides[auth.get_current_user] = _current_user
|
||||
c = TestClient(app)
|
||||
c.adv_id = adv_id
|
||||
try:
|
||||
yield c
|
||||
finally:
|
||||
app.dependency_overrides.clear()
|
||||
Base.metadata.drop_all(bind=engine)
|
||||
|
||||
|
||||
def page(client, before_id=None, limit=None):
|
||||
params = {}
|
||||
if before_id is not None:
|
||||
params["before_id"] = before_id
|
||||
if limit is not None:
|
||||
params["limit"] = limit
|
||||
r = client.get(f"/api/adventures/{client.adv_id}/actions", params=params)
|
||||
assert r.status_code == 200, r.text
|
||||
return r.json()
|
||||
|
||||
|
||||
def add_action(client, text="A new turn.") -> int:
|
||||
db = SessionLocal()
|
||||
try:
|
||||
highest = db.query(models.Action.index).order_by(
|
||||
models.Action.index.desc()).first()[0]
|
||||
action = models.Action(
|
||||
adventure_id=client.adv_id, index=highest + 1, type="ai", text=text
|
||||
)
|
||||
db.add(action)
|
||||
db.commit()
|
||||
return action.id
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
# ------------------------------------------------------------- the page load
|
||||
|
||||
def test_the_page_load_returns_only_the_newest_window(client):
|
||||
r = client.get(f"/api/adventures/{client.adv_id}")
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
assert len(body["actions"]) == ACTION_PAGE
|
||||
assert body["action_count"] == TOTAL
|
||||
# ...and it is the *newest* window, ending on the last action.
|
||||
assert body["actions"][-1]["index"] == TOTAL - 1
|
||||
assert body["actions"][0]["index"] == TOTAL - ACTION_PAGE
|
||||
|
||||
|
||||
def test_a_short_story_is_returned_whole(client):
|
||||
db = SessionLocal()
|
||||
try:
|
||||
db.query(models.Action).filter(models.Action.index >= 5).delete()
|
||||
db.commit()
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
body = client.get(f"/api/adventures/{client.adv_id}").json()
|
||||
assert len(body["actions"]) == 5
|
||||
assert body["action_count"] == 5
|
||||
|
||||
|
||||
def test_the_page_load_does_not_grow_with_the_story(client):
|
||||
"""The point of the change. Whatever the story's length, opening it costs
|
||||
a window."""
|
||||
meter = dbmeter.Meter()
|
||||
meter.attach(engine)
|
||||
try:
|
||||
with meter.scope("page load"):
|
||||
client.get(f"/api/adventures/{client.adv_id}")
|
||||
windowed = meter.scopes[-1].total.fetched
|
||||
finally:
|
||||
meter.detach()
|
||||
|
||||
# Each action carries ~1 kB of text and there are 187 of them; a window is
|
||||
# 60. Generous ceiling, but far below the whole story.
|
||||
assert windowed < ACTION_PAGE * 2_000, f"{windowed:,} B for one window"
|
||||
assert windowed < TOTAL * 500, (
|
||||
f"{windowed:,} B — that is the whole story, not a window"
|
||||
)
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ paging up
|
||||
|
||||
def test_the_first_page_is_the_newest(client):
|
||||
body = page(client)
|
||||
assert len(body["actions"]) == ACTION_PAGE
|
||||
assert body["total"] == TOTAL
|
||||
assert body["has_more"] is True
|
||||
assert body["actions"][-1]["index"] == TOTAL - 1
|
||||
|
||||
|
||||
def test_paging_up_covers_the_whole_story_exactly_once(client):
|
||||
seen = []
|
||||
body = page(client)
|
||||
seen = [a["index"] for a in body["actions"]]
|
||||
guard = 0
|
||||
while body["has_more"]:
|
||||
guard += 1
|
||||
assert guard < 20, "paging did not terminate"
|
||||
body = page(client, before_id=body["actions"][0]["id"])
|
||||
seen = [a["index"] for a in body["actions"]] + seen
|
||||
|
||||
assert seen == list(range(TOTAL)), "gap, duplicate or reordering while paging"
|
||||
|
||||
|
||||
def test_has_more_is_false_at_the_beginning_of_the_story(client):
|
||||
body = page(client)
|
||||
while body["has_more"]:
|
||||
body = page(client, before_id=body["actions"][0]["id"])
|
||||
assert body["actions"][0]["index"] == 0
|
||||
|
||||
|
||||
def test_each_page_is_ordered_oldest_first(client):
|
||||
body = page(client)
|
||||
indices = [a["index"] for a in body["actions"]]
|
||||
assert indices == sorted(indices)
|
||||
|
||||
|
||||
# ------------------------------------------------- the reason for the anchor
|
||||
|
||||
def test_a_turn_arriving_mid_scroll_does_not_shift_the_next_page(client):
|
||||
"""The failure an offset would have. Read the newest page, let a turn land,
|
||||
then page up: the reader must get exactly what precedes what they hold —
|
||||
no duplicate, no skipped action."""
|
||||
first = page(client)
|
||||
oldest_held = first["actions"][0]
|
||||
|
||||
add_action(client)
|
||||
|
||||
older = page(client, before_id=oldest_held["id"])
|
||||
assert older["actions"][-1]["index"] == oldest_held["index"] - 1, \
|
||||
"the page shifted when a turn landed"
|
||||
assert all(a["index"] < oldest_held["index"] for a in older["actions"])
|
||||
# The new turn moved the total, which is fine — it must not move the window.
|
||||
assert older["total"] == TOTAL + 1
|
||||
|
||||
|
||||
def test_a_deleted_anchor_reports_the_end_rather_than_a_duplicate_page(client):
|
||||
"""Undo can remove the action a slow scroll was anchored to. Better to stop
|
||||
than to hand back a page the reader already has."""
|
||||
body = page(client)
|
||||
anchor = body["actions"][0]
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
db.query(models.Action).filter(models.Action.id == anchor["id"]).delete()
|
||||
db.commit()
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
after = page(client, before_id=anchor["id"])
|
||||
assert after["actions"] == []
|
||||
assert after["has_more"] is False
|
||||
|
||||
|
||||
# ------------------------------------------------------------------- limits
|
||||
|
||||
def test_limit_is_honoured_and_capped(client):
|
||||
assert len(page(client, limit=5)["actions"]) == 5
|
||||
# A client asking for the whole story does not get to undo the paging.
|
||||
assert len(page(client, limit=100_000)["actions"]) <= ACTION_PAGE * 4
|
||||
|
||||
|
||||
def test_a_nonsense_limit_still_returns_something(client):
|
||||
assert len(page(client, limit=0)["actions"]) >= 1
|
||||
assert len(page(client, limit=-5)["actions"]) >= 1
|
||||
|
||||
|
||||
# --------------------------------------------------------------------- undo
|
||||
|
||||
def test_undo_returns_a_window_not_the_story(client):
|
||||
r = client.post(f"/api/adventures/{client.adv_id}/undo")
|
||||
assert r.status_code == 200, r.text
|
||||
body = r.json()
|
||||
assert len(body["actions"]) == ACTION_PAGE
|
||||
assert body["total"] == TOTAL - 1
|
||||
assert body["has_more"] is True
|
||||
@@ -1,9 +1,19 @@
|
||||
"""Guards on how much the database is asked for.
|
||||
|
||||
context_snapshot holds the entire assembled prompt for a turn (~74 KB/row in
|
||||
production, 94% of the database). It used to be pulled for every action on
|
||||
every adventure load and every turn, to read two tiny things out of it. These
|
||||
tests fail if that regresses.
|
||||
context_snapshot holds the entire assembled prompt for a turn — 163 KB a row
|
||||
averaged over production, 232 KB on the longest adventure, and 89% of the
|
||||
database. It used to be pulled for every action on every adventure load and
|
||||
every turn, to read two tiny things out of it. These tests fail if that
|
||||
regresses.
|
||||
|
||||
Two kinds of guard live here, and both are needed:
|
||||
|
||||
* **column guards** assert which columns a statement names. That is the shape
|
||||
both of this project's egress blowouts took — one query quietly carrying a
|
||||
column nobody read.
|
||||
* **byte ceilings** assert what a request actually costs. Every column guard
|
||||
would still pass if a response grew tenfold within the columns it is allowed
|
||||
to read, which is what a story that keeps getting longer does.
|
||||
|
||||
python -m pytest tests/test_egress.py -v
|
||||
"""
|
||||
@@ -16,21 +26,34 @@ os.environ["AIDND_DB_PATH"] = _tmp.name
|
||||
os.environ.pop("AIDND_DATABASE_URL", None)
|
||||
os.environ.pop("DATABASE_URL", None)
|
||||
|
||||
import json
|
||||
import random
|
||||
|
||||
import pytest
|
||||
from fastapi import Depends
|
||||
from fastapi.testclient import TestClient
|
||||
from sqlalchemy import event, text
|
||||
from sqlalchemy.orm import undefer
|
||||
|
||||
from app import auth, limits, migrations, models
|
||||
from app.context import history
|
||||
from app.database import Base, SessionLocal, engine, get_db
|
||||
from app.main import app
|
||||
from tools import dbmeter
|
||||
from tools.fakeprose import prose
|
||||
|
||||
# A stand-in for the real thing: the assembled prompt, which is what makes the
|
||||
# column enormous, plus the small world_state slice the UI actually needs.
|
||||
#
|
||||
# Varied text, not `"x" * 20_000`. The column is stored compressed now
|
||||
# (migration 43), and a repeated character compresses about a thousandfold —
|
||||
# which would make the byte ceilings below pass against a fixture that costs
|
||||
# nothing, testing nothing. Prose-shaped filler compresses like the prompts
|
||||
# this stands in for.
|
||||
_SNAPSHOT_RNG = random.Random(20_260_817)
|
||||
BIG_SNAPSHOT = {
|
||||
"system": "x" * 20_000,
|
||||
"story": "y" * 40_000,
|
||||
"system": prose(_SNAPSHOT_RNG, 20_000),
|
||||
"story": prose(_SNAPSHOT_RNG, 40_000),
|
||||
"world_state": {
|
||||
"delta": {"player.hp": -15},
|
||||
"report": {"applied": [{"path": "player.hp", "old": 100, "new": 85}]},
|
||||
@@ -190,16 +213,37 @@ def test_snapshot_is_still_reachable_on_demand(client):
|
||||
action_id = r.json()["actions"][0]["id"]
|
||||
r = client.get(f"/api/adventures/{client.adv_id}/actions/{action_id}/context")
|
||||
assert r.status_code == 200, r.text
|
||||
assert r.json()["system"] == "x" * 20_000
|
||||
# Round-tripped through zlib and back to a dict, byte for byte.
|
||||
assert r.json()["system"] == BIG_SNAPSHOT["system"]
|
||||
assert r.json()["story"] == BIG_SNAPSHOT["story"]
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ backfill
|
||||
|
||||
def as_json_snapshot_column(db) -> None:
|
||||
"""Put actions.context_snapshot back as JSON, the way it was before 43.
|
||||
|
||||
Migration 36 lifts world_delta out of the snapshot with SQL JSON
|
||||
functions, so it can only run while the column still *is* JSON. In a real
|
||||
upgrade it always is — 36 runs seven migrations before 43 compresses the
|
||||
column into a BLOB — but `create_all` builds today's schema, so a test
|
||||
calling that backfill has to rebuild the schema it was written against.
|
||||
"""
|
||||
db.execute(text("ALTER TABLE actions DROP COLUMN context_snapshot"))
|
||||
db.execute(text("ALTER TABLE actions ADD COLUMN context_snapshot JSON"))
|
||||
db.execute(
|
||||
text("UPDATE actions SET context_snapshot = :snapshot"),
|
||||
{"snapshot": json.dumps(BIG_SNAPSHOT)},
|
||||
)
|
||||
db.commit()
|
||||
|
||||
|
||||
def test_backfill_populates_world_delta_from_existing_snapshots(client):
|
||||
"""Migration 36 lifts the slice out server-side, without reading the
|
||||
snapshots into Python."""
|
||||
db = SessionLocal()
|
||||
try:
|
||||
as_json_snapshot_column(db)
|
||||
db.execute(text("UPDATE actions SET world_delta = NULL"))
|
||||
db.commit()
|
||||
assert db.query(models.Action).filter(models.Action.world_delta.isnot(None)).count() == 0
|
||||
@@ -250,3 +294,185 @@ def test_backfill_leaves_actions_without_world_state_alone(client):
|
||||
assert all(a.world_delta is None for a in db.query(models.Action).all())
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- byte ceilings
|
||||
#
|
||||
# The tests above assert which *columns* a statement names, which is the shape
|
||||
# both of this project's egress blowouts took. They would all still pass if a
|
||||
# response quietly grew tenfold within the columns it is allowed to read — and
|
||||
# a story that keeps getting longer does exactly that. These put a number on it.
|
||||
#
|
||||
# Ceilings are per action rather than absolute, so they mean the same thing
|
||||
# whatever size the fixture is set to, and they are generous: the point is to
|
||||
# catch a tenfold regression, not to freeze today's byte count.
|
||||
|
||||
ACTIONS_IN_FIXTURE = 12
|
||||
|
||||
# 3 kB an action against a real 994 B, measured on production 2026-08-17.
|
||||
# Anything that pulls a deferred column blows past this by two orders of
|
||||
# magnitude — see test_the_ceiling_discriminates below.
|
||||
PAGE_LOAD_BYTES_PER_ACTION = 3_000
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def meter():
|
||||
"""A byte meter on the shared engine, removed again afterwards.
|
||||
|
||||
Requested *after* `client` in a test's arguments so that building the
|
||||
fixture — a write path nobody plays — is not charged to any scope.
|
||||
"""
|
||||
m = dbmeter.Meter()
|
||||
m.attach(engine)
|
||||
try:
|
||||
yield m
|
||||
finally:
|
||||
m.detach()
|
||||
|
||||
|
||||
def fetched(meter) -> int:
|
||||
return meter.scopes[-1].total.fetched
|
||||
|
||||
|
||||
def test_page_load_stays_under_its_byte_ceiling(client, meter):
|
||||
with meter.scope("page load"):
|
||||
r = client.get(f"/api/adventures/{client.adv_id}")
|
||||
assert r.status_code == 200
|
||||
|
||||
budget = ACTIONS_IN_FIXTURE * PAGE_LOAD_BYTES_PER_ACTION
|
||||
assert fetched(meter) < budget, (
|
||||
f"page load fetched {fetched(meter):,} B for {ACTIONS_IN_FIXTURE} "
|
||||
f"actions, over the {budget:,} B budget"
|
||||
)
|
||||
|
||||
|
||||
def test_the_action_list_stays_under_its_byte_ceiling(client, meter):
|
||||
with meter.scope("action list"):
|
||||
r = client.get(f"/api/adventures/{client.adv_id}/actions")
|
||||
assert r.status_code == 200
|
||||
|
||||
budget = ACTIONS_IN_FIXTURE * PAGE_LOAD_BYTES_PER_ACTION
|
||||
assert fetched(meter) < budget, (
|
||||
f"the action list fetched {fetched(meter):,} B, over {budget:,} B"
|
||||
)
|
||||
|
||||
|
||||
def test_reading_one_action_does_not_cost_the_whole_story(client, meter):
|
||||
"""The snapshot is reachable on demand, and that request should pay for
|
||||
one row's worth — not the adventure's."""
|
||||
db = SessionLocal()
|
||||
try:
|
||||
action_id = db.query(models.Action.id).order_by(models.Action.id).first()[0]
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
with meter.scope("one snapshot"):
|
||||
r = client.get(f"/api/adventures/{client.adv_id}/actions/{action_id}/context")
|
||||
assert r.status_code == 200, r.text
|
||||
|
||||
one_snapshot = len(json.dumps(BIG_SNAPSHOT))
|
||||
assert fetched(meter) < one_snapshot * 2, (
|
||||
f"fetching one action's snapshot cost {fetched(meter):,} B; one "
|
||||
f"snapshot is {one_snapshot:,} B"
|
||||
)
|
||||
|
||||
|
||||
def _fat_adventures(user_id: int, count: int = 5, body: int = 20_000) -> None:
|
||||
"""Adventures whose bodies are heavy and whose index cards are not.
|
||||
|
||||
script_state, world_state and story_summary belong to the play screen. The
|
||||
index shows a title, a stamp and a snippet, and used to load all of it.
|
||||
"""
|
||||
db = SessionLocal()
|
||||
try:
|
||||
for i in range(count):
|
||||
db.add(models.Adventure(
|
||||
user_id=user_id,
|
||||
title=f"Adventure {i}",
|
||||
script_state={"log": "s" * body},
|
||||
world_state={"player": {"notes": "w" * body}},
|
||||
story_summary="y" * body,
|
||||
memory="m" * body,
|
||||
))
|
||||
db.commit()
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def test_the_index_does_not_read_the_adventure_body(client, sql_log):
|
||||
db = SessionLocal()
|
||||
try:
|
||||
user_id = db.query(models.User.id).first()[0]
|
||||
finally:
|
||||
db.close()
|
||||
_fat_adventures(user_id)
|
||||
|
||||
r = client.get("/api/adventures")
|
||||
assert r.status_code == 200
|
||||
assert len(r.json()) == 6 # the fixture's one, plus five
|
||||
|
||||
listing = [
|
||||
s for s in sql_log
|
||||
if "FROM adventures" in s and s.lstrip().upper().startswith("SELECT")
|
||||
]
|
||||
assert listing, "expected a listing query"
|
||||
for column in ("script_state", "world_state", "story_summary", "memory",
|
||||
"authors_note", "ai_instructions", "placeholders"):
|
||||
assert not any(column in s for s in listing), (
|
||||
f"the index read adventures.{column}, which nothing on that "
|
||||
f"screen displays"
|
||||
)
|
||||
|
||||
|
||||
def test_the_index_stays_under_its_byte_ceiling(client, meter):
|
||||
db = SessionLocal()
|
||||
try:
|
||||
user_id = db.query(models.User.id).first()[0]
|
||||
finally:
|
||||
db.close()
|
||||
_fat_adventures(user_id)
|
||||
|
||||
with meter.scope("index"):
|
||||
r = client.get("/api/adventures")
|
||||
assert r.status_code == 200
|
||||
|
||||
# Six adventures carrying 80 kB of body each. A card is a title, a stamp
|
||||
# and a 220-character snippet; 4 kB apiece is already generous.
|
||||
budget = 6 * 4_000
|
||||
assert fetched(meter) < budget, (
|
||||
f"the index fetched {fetched(meter):,} B for six adventures, over "
|
||||
f"{budget:,} B — it is reading the bodies again"
|
||||
)
|
||||
|
||||
|
||||
def test_the_ceiling_discriminates(client, meter):
|
||||
"""A ceiling is only worth having if the thing it excludes would breach it.
|
||||
|
||||
This is the regression the byte tests exist to catch, performed on purpose:
|
||||
undefer the snapshot and the same twelve rows cost several times the whole
|
||||
budget. If this ever stops exceeding it, the fixture has gone too small for
|
||||
the tests above to mean anything.
|
||||
|
||||
The margin used to be a hundredfold and is now about six. That is not the
|
||||
guard weakening — it is migration 43 compressing the column, and the
|
||||
fixture text being prose-shaped so it compresses like a real prompt rather
|
||||
than like a repeated character.
|
||||
"""
|
||||
budget = ACTIONS_IN_FIXTURE * PAGE_LOAD_BYTES_PER_ACTION
|
||||
db = SessionLocal()
|
||||
try:
|
||||
with meter.scope("undeferred"):
|
||||
rows = (
|
||||
db.query(models.Action)
|
||||
.options(undefer(models.Action.context_snapshot))
|
||||
.all()
|
||||
)
|
||||
assert len(rows) == ACTIONS_IN_FIXTURE
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
assert fetched(meter) > budget * 3, (
|
||||
"undeferring the snapshot cost only "
|
||||
f"{fetched(meter):,} B against a {budget:,} B budget — the fixture is "
|
||||
"too small for the byte ceilings above to catch anything"
|
||||
)
|
||||
|
||||
@@ -111,9 +111,9 @@ def test_cosine_moved_but_still_reachable_from_memorybank():
|
||||
|
||||
# -------------------------------------------------------------- set_vector
|
||||
|
||||
def test_set_vector_writes_both_columns(db, adventure):
|
||||
"""Until the follow-up migration drops the JSON column, it has to stay
|
||||
correct — a rollback reads it."""
|
||||
def test_set_vector_writes_the_blob_and_the_flag(db, adventure):
|
||||
"""The two columns that describe a vector move together, or a reader that
|
||||
trusts `embedded` gets a NULL blob."""
|
||||
memory = models.Memory(adventure_id=adventure.id, text="a fact")
|
||||
db.add(memory)
|
||||
db.commit()
|
||||
@@ -123,7 +123,6 @@ def test_set_vector_writes_both_columns(db, adventure):
|
||||
db.commit()
|
||||
db.expire_all()
|
||||
|
||||
assert memory.embedding == vector
|
||||
assert list(vectors.unpack(memory.embedding_blob)) == vector
|
||||
assert memory.embedded is True
|
||||
|
||||
@@ -140,24 +139,40 @@ def test_set_vector_none_clears_both(db, adventure):
|
||||
db.commit()
|
||||
db.expire_all()
|
||||
|
||||
assert memory.embedding is None
|
||||
assert memory.embedding_blob is None
|
||||
assert memory.embedded is False
|
||||
|
||||
|
||||
# --------------------------------------------------------------- the backfill
|
||||
|
||||
def add_legacy_json_column(db) -> None:
|
||||
"""Put `memories.embedding` back for the length of a test.
|
||||
|
||||
Migration 42 dropped it and the model no longer declares it, so
|
||||
`create_all` does not produce it — but everything below is testing the
|
||||
upgrade *from* a database that still has it, which is the only state in
|
||||
which the backfill has any work to do. Re-adding it by hand is what keeps
|
||||
these tests honest about the schema they claim to be starting from.
|
||||
"""
|
||||
db.execute(text("ALTER TABLE memories ADD COLUMN embedding JSON"))
|
||||
db.commit()
|
||||
|
||||
|
||||
def seed_json_only(db, adventure, count: int, dims: int = 64) -> dict[int, list[float]]:
|
||||
"""Memories as they exist before the migration: JSON vector, no blob."""
|
||||
add_legacy_json_column(db)
|
||||
rng = random.Random(count)
|
||||
expected = {}
|
||||
for i in range(count):
|
||||
vector = sample_vector(rng, dims)
|
||||
memory = models.Memory(
|
||||
adventure_id=adventure.id, text=f"fact {i}", embedding=vector
|
||||
)
|
||||
memory = models.Memory(adventure_id=adventure.id, text=f"fact {i}")
|
||||
db.add(memory)
|
||||
db.flush()
|
||||
# Raw, because the ORM no longer knows this column exists.
|
||||
db.execute(
|
||||
text("UPDATE memories SET embedding = :v WHERE id = :id"),
|
||||
{"v": json.dumps(vector), "id": memory.id},
|
||||
)
|
||||
expected[memory.id] = vector
|
||||
db.commit()
|
||||
db.execute(text("UPDATE memories SET embedding_blob = NULL, embedded = false"))
|
||||
@@ -193,6 +208,7 @@ def test_backfill_reaches_past_one_batch(db, adventure):
|
||||
|
||||
|
||||
def test_backfill_leaves_unembedded_memories_alone(db, adventure):
|
||||
add_legacy_json_column(db)
|
||||
db.add(models.Memory(adventure_id=adventure.id, text="not embedded yet"))
|
||||
db.commit()
|
||||
|
||||
@@ -274,6 +290,14 @@ def test_bootstrap_adds_the_columns_and_backfills_them(db, adventure):
|
||||
# embedded must still read as not embedded afterwards.
|
||||
assert by_id[unembedded_id] == (None, False)
|
||||
|
||||
# ...and migration 42, at the end of the same run, takes the JSON column
|
||||
# away. Ordering matters: 38 reads it, 42 drops it, and an upgrade that
|
||||
# ran them the other way round would arrive with an empty bank.
|
||||
with engine.begin() as conn:
|
||||
columns = {row[1] for row in conn.execute(text("PRAGMA table_info(memories)"))}
|
||||
assert "embedding" not in columns
|
||||
assert {"embedding_blob", "embedded"} <= columns
|
||||
|
||||
|
||||
def test_migration_38_is_spelled_for_both_dialects():
|
||||
"""Every Postgres deploy replays migrations from 24 on, so a SQLite-only
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
"""Switching embedding models must re-embed the bank.
|
||||
|
||||
Vectors from two different models are not comparable — different space, often
|
||||
different width — so changing the model has to throw the stored ones away and
|
||||
let the post-turn pass rebuild them.
|
||||
|
||||
That worked while the vectors lived in `memories.embedding`: the settings
|
||||
route nulled that column and the embed queue picked the rows up. Migration 38
|
||||
moved the vectors to `embedding_blob` with an `embedded` flag beside them, and
|
||||
the bulk clear kept nulling the old column alone. The blob survived, the flag
|
||||
stayed true, `_embed_pending` (which looks for `embedded IS FALSE`) never saw
|
||||
the rows, and the bank went on ranking against the previous model's vectors
|
||||
for good.
|
||||
|
||||
Nothing reports this. `cosine` returns 0.0 on a width mismatch, so a
|
||||
different-width model scores every memory zero and retrieval quietly returns
|
||||
whichever rows sort first; a same-width model scores plausible-looking
|
||||
garbage.
|
||||
|
||||
python -m pytest tests/test_embedding_model_switch.py -v
|
||||
"""
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False)
|
||||
_tmp.close()
|
||||
os.environ["AIDND_DB_PATH"] = _tmp.name
|
||||
os.environ.pop("AIDND_DATABASE_URL", None)
|
||||
os.environ.pop("DATABASE_URL", None)
|
||||
|
||||
import asyncio
|
||||
|
||||
import pytest
|
||||
from fastapi import Depends
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from app import auth, limits, memorybank, models
|
||||
from app.database import Base, SessionLocal, engine, get_db
|
||||
from app.main import app
|
||||
|
||||
DIMS = 8
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
Base.metadata.create_all(bind=engine)
|
||||
memorybank._vector_cache.clear()
|
||||
setup = SessionLocal()
|
||||
user = models.User(is_guest=False, email="switch@example.com")
|
||||
setup.add(user)
|
||||
setup.flush()
|
||||
setup.add(models.Settings(
|
||||
user_id=user.id, api_key="enc:dummy", model="test-model",
|
||||
embedding_model="model-a",
|
||||
))
|
||||
adventure = models.Adventure(
|
||||
user_id=user.id, title="Cave", script_state={}, memory_bank_enabled=True
|
||||
)
|
||||
setup.add(adventure)
|
||||
setup.flush()
|
||||
for i in range(5):
|
||||
memory = models.Memory(adventure_id=adventure.id, text=f"Memory {i}")
|
||||
memorybank.set_vector(memory, [float(i)] + [0.0] * (DIMS - 1))
|
||||
setup.add(memory)
|
||||
setup.commit()
|
||||
adv_id, user_id = adventure.id, user.id
|
||||
setup.close()
|
||||
|
||||
monkeypatch.setattr(limits, "rate_limit", lambda *a, **k: None)
|
||||
monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None)
|
||||
|
||||
def _current_user(db=Depends(get_db)):
|
||||
return db.get(models.User, user_id)
|
||||
|
||||
app.dependency_overrides[auth.get_current_user] = _current_user
|
||||
c = TestClient(app)
|
||||
c.adv_id = adv_id
|
||||
try:
|
||||
yield c
|
||||
finally:
|
||||
app.dependency_overrides.clear()
|
||||
memorybank._vector_cache.clear()
|
||||
Base.metadata.drop_all(bind=engine)
|
||||
|
||||
|
||||
def memories(db):
|
||||
return db.query(models.Memory).order_by(models.Memory.id).all()
|
||||
|
||||
|
||||
def test_the_bank_starts_embedded(client):
|
||||
db = SessionLocal()
|
||||
try:
|
||||
rows = memories(db)
|
||||
assert len(rows) == 5
|
||||
assert all(m.embedded for m in rows)
|
||||
assert all(m.embedding_blob for m in rows)
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def test_changing_the_model_clears_every_vector(client):
|
||||
r = client.put("/api/settings", json={"embedding_model": "model-b"})
|
||||
assert r.status_code == 200, r.text
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
rows = memories(db)
|
||||
assert [m.embedding_blob for m in rows] == [None] * 5, \
|
||||
"the blob survived the model change"
|
||||
assert not any(m.embedded for m in rows), \
|
||||
"`embedded` stayed true, so nothing will ever re-embed these"
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def test_cleared_memories_are_queued_for_re_embedding(client):
|
||||
"""The flag is not cosmetic: it is the only thing `_embed_pending` filters
|
||||
on, so this is the assertion that the bank actually recovers."""
|
||||
client.put("/api/settings", json={"embedding_model": "model-b"})
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
pending = (
|
||||
db.query(models.Memory)
|
||||
.filter(models.Memory.embedded.is_(False),
|
||||
models.Memory.forgotten.is_(False))
|
||||
.all()
|
||||
)
|
||||
assert len(pending) == 5
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def test_retrieval_uses_no_stale_vector_after_the_switch(client, monkeypatch):
|
||||
"""Until the re-embed runs, the bank must return nothing rather than
|
||||
ranking against the old model's vectors."""
|
||||
client.put("/api/settings", json={"embedding_model": "model-b"})
|
||||
|
||||
class Embedder:
|
||||
async def embed(self, texts):
|
||||
return [[1.0] + [0.0] * (DIMS - 1) for _ in texts]
|
||||
|
||||
monkeypatch.setattr(memorybank, "embedding_provider", lambda s: Embedder())
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
adventure = db.get(models.Adventure, client.adv_id)
|
||||
settings = db.query(models.Settings).first()
|
||||
result = asyncio.run(
|
||||
memorybank.retrieve_memories(adventure, settings, update_stats=False)
|
||||
)
|
||||
assert result["used"] == []
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def test_an_unrelated_settings_change_keeps_the_vectors(client):
|
||||
"""Only an embedding-model change may clear the bank — re-embedding costs
|
||||
an API call per memory."""
|
||||
r = client.put("/api/settings", json={"model": "some-other-chat-model"})
|
||||
assert r.status_code == 200, r.text
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
rows = memories(db)
|
||||
assert all(m.embedded for m in rows)
|
||||
assert all(m.embedding_blob for m in rows)
|
||||
finally:
|
||||
db.close()
|
||||
@@ -26,7 +26,7 @@ import asyncio
|
||||
from datetime import timedelta
|
||||
|
||||
import pytest
|
||||
from sqlalchemy import event
|
||||
from sqlalchemy import event, inspect as sa_inspect
|
||||
|
||||
from app import memorybank, models
|
||||
from app.database import Base, SessionLocal, engine
|
||||
@@ -206,13 +206,14 @@ def memory_selects(statements):
|
||||
]
|
||||
|
||||
|
||||
def test_the_json_column_is_never_selected(db, adventure, settings, bank, sql_log):
|
||||
"""`memories.embedding` is dead weight kept only until a follow-up
|
||||
migration drops it. If anything still reads it, dropping it breaks."""
|
||||
retrieve(adventure, settings, StubEmbedder())
|
||||
offenders = [s for s in memory_selects(sql_log) if "memories.embedding " in s
|
||||
or s.rstrip().endswith("memories.embedding")]
|
||||
assert offenders == [], f"the JSON column was read:\n{offenders[0][:300]}"
|
||||
def test_the_json_column_is_gone(db):
|
||||
"""`memories.embedding` held the vectors before migration 38 and nothing
|
||||
read it afterwards; migration 42 dropped it. Bringing it back would restore
|
||||
4 MB of dead weight and a second place vectors can be written from — which
|
||||
is how the model-switch bug happened (test_embedding_model_switch.py)."""
|
||||
columns = {c["name"] for c in sa_inspect(engine).get_columns("memories")}
|
||||
assert "embedding" not in columns
|
||||
assert {"embedding_blob", "embedded"} <= columns
|
||||
|
||||
|
||||
def test_the_catalogue_query_carries_no_vectors(db, adventure, settings, bank, sql_log):
|
||||
|
||||
@@ -294,7 +294,8 @@ def test_undo_removes_the_action_and_its_history(client):
|
||||
_retry(client)
|
||||
r = client.post(f"/api/adventures/{client.adv_id}/undo")
|
||||
assert r.status_code == 200, r.text
|
||||
assert [a["type"] for a in r.json()] == ["start"]
|
||||
# Undo returns the newest window now, not the whole story.
|
||||
assert [a["type"] for a in r.json()["actions"]] == ["start"]
|
||||
assert _adv(client.adv_id)[0] == {}
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,260 @@
|
||||
"""context_snapshot, stored compressed.
|
||||
|
||||
The column is 89% of the database and the free tier allows 512 MB. Reads were
|
||||
solved by deferring it; this is about the storage ceiling. Postgres already
|
||||
TOASTs it and only gets 1.7x, because pglz is tuned for fast decompression of
|
||||
data a query might filter on — and nothing ever filters on an assembled
|
||||
prompt.
|
||||
|
||||
Three things have to hold, and only the first is obvious:
|
||||
|
||||
* what goes in comes back out, exactly, including a snapshot written before
|
||||
the conversion and one that is NULL;
|
||||
* the model still hands callers a dict, so no call site changes;
|
||||
* migration 44 drops the original column, so the backfill is the one
|
||||
destructive step in this file — it must convert every row or abort.
|
||||
|
||||
python -m pytest tests/test_snapshot_compression.py -v
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import random
|
||||
import tempfile
|
||||
import zlib
|
||||
|
||||
_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False)
|
||||
_tmp.close()
|
||||
os.environ["AIDND_DB_PATH"] = _tmp.name
|
||||
os.environ.pop("AIDND_DATABASE_URL", None)
|
||||
os.environ.pop("DATABASE_URL", None)
|
||||
|
||||
import pytest
|
||||
from sqlalchemy import text
|
||||
|
||||
from app import compression, migrations, models
|
||||
from app.database import Base, SessionLocal, engine
|
||||
from tools.fakeprose import prose
|
||||
|
||||
|
||||
def snapshot(seed: int, nbytes: int = 60_000) -> dict:
|
||||
rng = random.Random(seed)
|
||||
return {
|
||||
"system": prose(rng, nbytes // 3),
|
||||
"story": prose(rng, nbytes - nbytes // 3),
|
||||
"world_state": {"delta": {"player.hp": -3}},
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def db():
|
||||
Base.metadata.create_all(bind=engine)
|
||||
session = SessionLocal()
|
||||
try:
|
||||
yield session
|
||||
finally:
|
||||
session.close()
|
||||
Base.metadata.drop_all(bind=engine)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def adventure(db):
|
||||
user = models.User(is_guest=False, email="snap@example.com")
|
||||
db.add(user)
|
||||
db.flush()
|
||||
adv = models.Adventure(user_id=user.id, title="Cave", script_state={})
|
||||
db.add(adv)
|
||||
db.commit()
|
||||
return adv
|
||||
|
||||
|
||||
# ------------------------------------------------------------------- packing
|
||||
|
||||
def test_pack_round_trips_exactly():
|
||||
value = snapshot(1)
|
||||
assert compression.unpack(compression.pack(value)) == value
|
||||
|
||||
|
||||
def test_pack_handles_the_awkward_values():
|
||||
for value in ({}, {"a": None}, {"nested": {"deep": [1, 2, {"x": "é"}]}}):
|
||||
assert compression.unpack(compression.pack(value)) == value
|
||||
|
||||
|
||||
def test_pack_actually_shrinks_prose():
|
||||
"""The whole justification. If this ratio collapses, the migration is
|
||||
spending CPU for nothing."""
|
||||
value = snapshot(2, 200_000)
|
||||
raw = json.dumps(value, separators=(",", ":")).encode()
|
||||
packed = compression.pack(value)
|
||||
assert len(packed) < len(raw) / 3, (
|
||||
f"{len(raw):,} B compressed to {len(packed):,} B — under 3x"
|
||||
)
|
||||
|
||||
|
||||
def test_unpack_rejects_nothing_it_wrote():
|
||||
packed = compression.pack({"a": "b"})
|
||||
assert compression.unpack(bytearray(packed)) == {"a": "b"}
|
||||
assert compression.unpack(memoryview(bytes(packed))) == {"a": "b"}
|
||||
|
||||
|
||||
# --------------------------------------------------------------- the column
|
||||
|
||||
def test_the_column_stores_bytes_and_returns_a_dict(db, adventure):
|
||||
value = snapshot(3)
|
||||
action = models.Action(
|
||||
adventure_id=adventure.id, index=0, type="ai", text="t",
|
||||
context_snapshot=value,
|
||||
)
|
||||
db.add(action)
|
||||
db.commit()
|
||||
db.expire_all()
|
||||
|
||||
assert action.context_snapshot == value
|
||||
|
||||
stored = db.execute(
|
||||
text("SELECT context_snapshot FROM actions WHERE id = :id"),
|
||||
{"id": action.id},
|
||||
).scalar()
|
||||
assert isinstance(stored, (bytes, bytearray, memoryview))
|
||||
assert json.loads(zlib.decompress(bytes(stored))) == value
|
||||
|
||||
|
||||
def test_the_column_is_smaller_than_the_json_it_holds(db, adventure):
|
||||
value = snapshot(4, 200_000)
|
||||
action = models.Action(
|
||||
adventure_id=adventure.id, index=0, type="ai", text="t",
|
||||
context_snapshot=value,
|
||||
)
|
||||
db.add(action)
|
||||
db.commit()
|
||||
|
||||
stored = db.execute(
|
||||
text("SELECT length(context_snapshot) FROM actions WHERE id = :id"),
|
||||
{"id": action.id},
|
||||
).scalar()
|
||||
assert stored < len(json.dumps(value)) / 3
|
||||
|
||||
|
||||
def test_null_stays_null(db, adventure):
|
||||
action = models.Action(
|
||||
adventure_id=adventure.id, index=0, type="do", text="t",
|
||||
context_snapshot=None,
|
||||
)
|
||||
db.add(action)
|
||||
db.commit()
|
||||
db.expire_all()
|
||||
assert action.context_snapshot is None
|
||||
|
||||
|
||||
def test_an_unreadable_snapshot_reads_as_none_rather_than_raising(db, adventure):
|
||||
"""One corrupt row must not 500 the turn that happens to load it. The
|
||||
snapshot is a debugging view; the story is the thing that matters."""
|
||||
action = models.Action(
|
||||
adventure_id=adventure.id, index=0, type="ai", text="t",
|
||||
context_snapshot={"a": "b"},
|
||||
)
|
||||
db.add(action)
|
||||
db.commit()
|
||||
db.execute(
|
||||
text("UPDATE actions SET context_snapshot = :junk WHERE id = :id"),
|
||||
{"junk": b"not zlib at all", "id": action.id},
|
||||
)
|
||||
db.commit()
|
||||
db.expire_all()
|
||||
|
||||
assert db.get(models.Action, action.id).context_snapshot is None
|
||||
|
||||
|
||||
# ------------------------------------------------------- the upgrade in full
|
||||
|
||||
def as_json_column(db, rows: dict[int, dict]) -> None:
|
||||
"""The pre-43 schema: context_snapshot as a JSON column, populated."""
|
||||
db.execute(text("ALTER TABLE actions DROP COLUMN context_snapshot"))
|
||||
db.execute(text("ALTER TABLE actions ADD COLUMN context_snapshot JSON"))
|
||||
for action_id, value in rows.items():
|
||||
db.execute(
|
||||
text("UPDATE actions SET context_snapshot = :v WHERE id = :id"),
|
||||
{"v": json.dumps(value), "id": action_id},
|
||||
)
|
||||
db.commit()
|
||||
|
||||
|
||||
def seed_pre_43(db, adventure, count: int = 4) -> dict[int, dict]:
|
||||
ids = []
|
||||
for i in range(count):
|
||||
action = models.Action(
|
||||
adventure_id=adventure.id, index=i, type="ai", text=f"t{i}"
|
||||
)
|
||||
db.add(action)
|
||||
db.flush()
|
||||
ids.append(action.id)
|
||||
# One action with no snapshot at all, which must survive as NULL.
|
||||
plain = models.Action(
|
||||
adventure_id=adventure.id, index=count, type="do", text="look"
|
||||
)
|
||||
db.add(plain)
|
||||
db.commit()
|
||||
|
||||
expected = {action_id: snapshot(action_id) for action_id in ids}
|
||||
as_json_column(db, expected)
|
||||
db.execute(text(f"PRAGMA user_version = {migrations.SNAPSHOT_COMPRESS_VERSION - 1}"))
|
||||
db.commit()
|
||||
return expected
|
||||
|
||||
|
||||
def test_bootstrap_converts_every_snapshot(db, adventure):
|
||||
expected = seed_pre_43(db, adventure)
|
||||
db.close()
|
||||
|
||||
migrations.bootstrap(engine)
|
||||
|
||||
check = SessionLocal()
|
||||
try:
|
||||
for action_id, value in expected.items():
|
||||
assert check.get(models.Action, action_id).context_snapshot == value
|
||||
assert check.execute(
|
||||
text("SELECT count(*) FROM actions WHERE context_snapshot IS NULL")
|
||||
).scalar() == 1
|
||||
finally:
|
||||
check.close()
|
||||
|
||||
|
||||
def test_bootstrap_leaves_the_column_named_context_snapshot(db, adventure):
|
||||
seed_pre_43(db, adventure)
|
||||
db.close()
|
||||
|
||||
migrations.bootstrap(engine)
|
||||
|
||||
with engine.begin() as conn:
|
||||
columns = {row[1] for row in conn.execute(text("PRAGMA table_info(actions)"))}
|
||||
assert "context_snapshot" in columns
|
||||
assert "context_snapshot_z" not in columns, "the swap left the scratch column behind"
|
||||
|
||||
|
||||
def test_the_backfill_aborts_rather_than_dropping_unconvertible_data(
|
||||
db, adventure, monkeypatch
|
||||
):
|
||||
"""Migration 44 destroys the original. If anything cannot be converted the
|
||||
whole run has to roll back with the column still there — the alternative is
|
||||
losing somebody's prompts to a bug in this file."""
|
||||
seed_pre_43(db, adventure)
|
||||
db.close()
|
||||
|
||||
def broken_unpack(_blob):
|
||||
return {"not": "what went in"}
|
||||
|
||||
monkeypatch.setattr(compression, "unpack", broken_unpack)
|
||||
|
||||
with pytest.raises(RuntimeError, match="round trip"):
|
||||
migrations.bootstrap(engine)
|
||||
|
||||
with engine.begin() as conn:
|
||||
columns = {row[1] for row in conn.execute(text("PRAGMA table_info(actions)"))}
|
||||
version = conn.execute(text("PRAGMA user_version")).scalar()
|
||||
surviving = conn.execute(
|
||||
text("SELECT count(*) FROM actions WHERE context_snapshot IS NOT NULL")
|
||||
).scalar()
|
||||
|
||||
assert "context_snapshot" in columns
|
||||
assert version < migrations.SNAPSHOT_COMPRESS_VERSION, \
|
||||
"the version advanced past a backfill that failed"
|
||||
assert surviving == 4, "the prompts did not survive the rollback"
|
||||
@@ -174,7 +174,26 @@ class Meter:
|
||||
|
||||
metered_creator._dbmeter = self
|
||||
pool._creator = metered_creator
|
||||
self._attached_pools.append(pool)
|
||||
self._attached_pools.append((engine, pool, creator))
|
||||
|
||||
def detach(self) -> None:
|
||||
"""Put every metered engine back as it was.
|
||||
|
||||
A script exits and takes the wrapping with it; a test does not, and one
|
||||
test leaving the shared engine metered would go on charging bytes to a
|
||||
scope nobody opened. Pooled connections are dropped again on the way
|
||||
out for the same reason attach drops them on the way in.
|
||||
"""
|
||||
while self._attached_pools:
|
||||
engine, pool, creator = self._attached_pools.pop()
|
||||
pool._creator = creator
|
||||
engine.dispose()
|
||||
|
||||
def __enter__(self) -> "Meter":
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc) -> None:
|
||||
self.detach()
|
||||
|
||||
# ------------------------------------------------------------- reporting
|
||||
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
"""Filler text that behaves like prose under compression.
|
||||
|
||||
Shared by the stress harness and the egress tests, because both now measure
|
||||
something a repeated string would answer wrongly. `"x" * 20_000` compresses
|
||||
about a thousandfold and English three- or fourfold, so a fixture built from
|
||||
repeats makes a compressed column look free and any ceiling drawn around it
|
||||
meaningless.
|
||||
|
||||
Not a language model, and does not need to be. What matters is the symbol
|
||||
distribution, not the sense.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import random
|
||||
|
||||
WORDS = (
|
||||
"corridor narrows shoulders brush wet stone torchlight gutters draught "
|
||||
"smells cold iron somewhere ahead water moving count nine paces passage "
|
||||
"opens chamber ceiling lost dark sound breathing comes back half second "
|
||||
"late Gwen catches sleeve without word points floor line pale grit laid "
|
||||
"across threshold deliberate arc quartermaster bandit camp above ford "
|
||||
"tunnels exchange key lantern rope knife bread rain mud hill road gate "
|
||||
"watchman silver debt promise fever horse cart river bridge mill barley "
|
||||
"smoke rafters bench ale ledger seal wax parchment ink candle shutter "
|
||||
"hinge bolt cellar barrel salt fish nets harbour tide gull mast canvas"
|
||||
).split()
|
||||
|
||||
|
||||
def prose(rng: random.Random, nbytes: int) -> str:
|
||||
"""Roughly `nbytes` of varied sentences, deterministic for a given rng."""
|
||||
out: list[str] = []
|
||||
total = 0
|
||||
while total < nbytes:
|
||||
sentence = " ".join(rng.choice(WORDS) for _ in range(rng.randint(8, 18)))
|
||||
chunk = sentence.capitalize() + ". "
|
||||
out.append(chunk)
|
||||
total += len(chunk)
|
||||
return "".join(out)[:nbytes]
|
||||
+120
-43
@@ -23,31 +23,80 @@ sessions, the ORM, the scripting engine and the context builder are the real
|
||||
ones, because the bugs this exists to catch live in exactly the layer a mock
|
||||
would replace.
|
||||
|
||||
It runs on a throwaway SQLite file rather than Postgres. What is being measured
|
||||
is which columns of which rows a code path asks for, and that is decided by the
|
||||
ORM, identically on both. The dialects disagree on how a value is encoded on
|
||||
the wire — JSON especially — so treat the absolute figures as production-shaped
|
||||
rather than production-exact, and compare before against after.
|
||||
It runs on a throwaway SQLite file by default. What is being measured is which
|
||||
columns of which rows a code path asks for, and that is decided by the ORM,
|
||||
identically on both dialects. The dialects disagree on how a value is encoded
|
||||
on the wire — JSON especially — so treat the absolute figures as
|
||||
production-shaped rather than production-exact, and compare before against
|
||||
after.
|
||||
|
||||
Calibration, against the two figures measured directly on production
|
||||
(2026-08-16): a 200-action page load reported 426.7 kB here against 423 KB
|
||||
there, and one turn on a 100-memory bank reported 3,258.7 kB against 3,153 kB.
|
||||
To measure the encodings SQLite cannot reach — bytea for the packed vectors,
|
||||
and json columns psycopg parses before the meter sees them — set
|
||||
AIDND_STRESS_DATABASE_URL to a **throwaway** Postgres database:
|
||||
|
||||
AIDND_STRESS_DATABASE_URL=postgresql://…/stress_scratch \
|
||||
.venv/Scripts/python.exe -m tools.stress_session
|
||||
|
||||
The harness writes, so it refuses any target whose database name does not say
|
||||
'stress' or 'scratch'. Never point it at a database holding real users.
|
||||
|
||||
Calibration. The fixture is sized from production, re-measured 2026-08-17
|
||||
against the live Neon database (aggregates only — counts and octet_length
|
||||
sums, never row contents):
|
||||
|
||||
per action, text 886 B -> --narration-bytes 1700, alternating
|
||||
with a one-line player input
|
||||
longest adventure 607 actions -> --actions 600
|
||||
context_snapshot 232 KB/row -> --snapshot-bytes 232000
|
||||
memory bank, largest 100 memories, 6,144 B a vector
|
||||
|
||||
The previous defaults were wrong in both directions at once and happened to
|
||||
land near the right total: actions were modelled at ~2.1 KB against a real
|
||||
886 B, and stories at 200 actions against a real 607. Width was flattering,
|
||||
length was not, and length is what a page load pays for.
|
||||
|
||||
Filler text is generated word by word rather than repeated. A repeated
|
||||
sentence compresses about a hundredfold and prose three- or fourfold, so the
|
||||
old fixture would have made any compression measurement on context_snapshot
|
||||
meaningless.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
# Must precede the app import: database.py reads these at module scope.
|
||||
_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False)
|
||||
_tmp.close()
|
||||
os.environ["AIDND_DB_PATH"] = _tmp.name
|
||||
os.environ.pop("AIDND_DATABASE_URL", None)
|
||||
os.environ.pop("DATABASE_URL", None)
|
||||
#
|
||||
# Default is a throwaway SQLite file. AIDND_STRESS_DATABASE_URL points the
|
||||
# harness at a real Postgres instead, which is the only way to reach the
|
||||
# encodings SQLite cannot exercise: bytea for the packed vectors, and json
|
||||
# columns that psycopg parses into Python before the meter ever sees them.
|
||||
#
|
||||
# The name guard is not paranoia. This harness *writes* — it builds a whole
|
||||
# synthetic adventure — so a URL that happened to point at the production
|
||||
# database would quietly seed it with fake users and fake play. The target
|
||||
# must say it is disposable.
|
||||
_stress_url = os.environ.get("AIDND_STRESS_DATABASE_URL", "").strip()
|
||||
if _stress_url:
|
||||
_dbname = _stress_url.rsplit("/", 1)[-1].split("?")[0]
|
||||
if not any(mark in _dbname.lower() for mark in ("stress", "scratch")):
|
||||
sys.exit(
|
||||
f"refusing to run against database {_dbname!r}.\n"
|
||||
"This harness writes a synthetic adventure, so its target must be a\n"
|
||||
"throwaway database with 'stress' or 'scratch' in the name."
|
||||
)
|
||||
os.environ["AIDND_DATABASE_URL"] = _stress_url
|
||||
os.environ.pop("DATABASE_URL", None)
|
||||
else:
|
||||
_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False)
|
||||
_tmp.close()
|
||||
os.environ["AIDND_DB_PATH"] = _tmp.name
|
||||
os.environ.pop("AIDND_DATABASE_URL", None)
|
||||
os.environ.pop("DATABASE_URL", None)
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import random
|
||||
import sys
|
||||
|
||||
from fastapi import Depends
|
||||
from fastapi.testclient import TestClient
|
||||
@@ -59,36 +108,47 @@ from app.providers import PromptParts
|
||||
from app.routers import adventures
|
||||
|
||||
from .dbmeter import Meter, kb
|
||||
from .fakeprose import prose
|
||||
|
||||
EMBEDDING_DIMS = 1536
|
||||
|
||||
# ~74 KB, which is what a real snapshot weighs in production: the assembled
|
||||
# prompt is nearly all of it.
|
||||
SNAPSHOT_SYSTEM = "You are a masterful storyteller. " * 400
|
||||
SNAPSHOT_STORY = "The corridor narrows and the torchlight gutters. " * 1200
|
||||
# ~232 KB, measured on production's largest adventure (2026-08-17). The old
|
||||
# figure here was 74 KB, taken from the comment in models.py; the real column
|
||||
# averages 163 KB a row across the whole table and 232 KB on the adventure that
|
||||
# matters, because the assembled prompt grows with the story behind it.
|
||||
#
|
||||
# Built from varied text rather than one sentence repeated. A repeated sentence
|
||||
# compresses about a hundredfold and real prose three- or fourfold, so a
|
||||
# fixture made of repeats would make any compression measurement meaningless —
|
||||
# and shrinking this column is the open question it exists to answer.
|
||||
SNAPSHOT_SYSTEM = None # set by _build_text()
|
||||
SNAPSHOT_STORY = None
|
||||
|
||||
_PARAGRAPH = (
|
||||
"The corridor narrows until your shoulders brush wet stone, and the "
|
||||
"torchlight gutters in a draught that smells of cold iron. Somewhere ahead, "
|
||||
"water is moving. You count nine paces before the passage opens into a "
|
||||
"chamber whose ceiling is lost in the dark, and the sound of your own "
|
||||
"breathing comes back to you a half-second late.\n\n"
|
||||
"Gwen catches your sleeve without a word and points at the floor, where a "
|
||||
"line of pale grit has been laid across the threshold in a deliberate arc.\n\n"
|
||||
)
|
||||
PLAYER_INPUT = "> You crouch and look more closely at the grit on the floor."
|
||||
|
||||
# Rebound by main() to --narration-bytes. An AI action's length is what makes a
|
||||
# page load expensive, and it is the one fixture dimension that cannot be
|
||||
# guessed from the schema: production averages ~2.1 KB across all actions,
|
||||
# which is ~4 KB of narration alternating with a one-line player input.
|
||||
NARRATION = _PARAGRAPH
|
||||
# All three are bound by _build_text() from the fixture arguments.
|
||||
NARRATION = None
|
||||
MEMORY_TEXT = (
|
||||
"You found a bandit camp above the ford and agreed to guide Gwen through "
|
||||
"the tunnels in exchange for the iron key she took from the quartermaster."
|
||||
)
|
||||
|
||||
|
||||
def _build_text(args, rng: random.Random) -> None:
|
||||
"""Size the three variable-length fixture strings from the arguments.
|
||||
|
||||
Separate from build_fixture so the sizes are decided once, before anything
|
||||
is written, and so a shape's cost is a function of the flags rather than of
|
||||
how many rows happened to be generated first.
|
||||
"""
|
||||
global NARRATION, SNAPSHOT_SYSTEM, SNAPSHOT_STORY
|
||||
NARRATION = prose(rng, args.narration_bytes)
|
||||
# The assembled prompt is a system block and the story so far; the split
|
||||
# is roughly one to five in production.
|
||||
SNAPSHOT_SYSTEM = prose(rng, args.snapshot_bytes // 6)
|
||||
SNAPSHOT_STORY = prose(rng, args.snapshot_bytes - args.snapshot_bytes // 6)
|
||||
|
||||
|
||||
# --------------------------------------------------------------- fake network
|
||||
|
||||
|
||||
@@ -125,6 +185,13 @@ class FakeEmbeddings:
|
||||
def build_fixture(args, rng: random.Random) -> tuple[int, int]:
|
||||
"""A user, settings and one adventure at production scale. Returns
|
||||
(adventure_id, user_id)."""
|
||||
# A SQLite run gets a brand-new temp file every time, so the fixture can
|
||||
# assume an empty database. A Postgres scratch target persists between
|
||||
# runs, and the second one would collide on the fixture user's unique
|
||||
# email — so empty it first. Only ever reached for a target whose name
|
||||
# passed the 'stress'/'scratch' guard at the top of this module.
|
||||
if _stress_url:
|
||||
Base.metadata.drop_all(bind=engine)
|
||||
Base.metadata.create_all(bind=engine)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
@@ -281,8 +348,13 @@ def parse_args(argv=None):
|
||||
p = argparse.ArgumentParser(
|
||||
prog="tools.stress_session", description=__doc__.splitlines()[0]
|
||||
)
|
||||
p.add_argument("--actions", type=int, default=200,
|
||||
help="story actions in the fixture (default: 200)")
|
||||
# 607 is production's longest adventure as of 2026-08-17, and length is
|
||||
# the dimension the old default (200) got wrong: real actions are lighter
|
||||
# than this fixture used to make them, but real stories run three times
|
||||
# longer, and length is what a page load pays for.
|
||||
p.add_argument("--actions", type=int, default=600,
|
||||
help="story actions in the fixture (default: 600, "
|
||||
"production's longest adventure is 607)")
|
||||
p.add_argument("--memories", type=int, default=100,
|
||||
help="memories, all embedded (default: 100)")
|
||||
# Deliberately not the app's default (80): a measuring instrument should
|
||||
@@ -290,10 +362,17 @@ def parse_args(argv=None):
|
||||
p.add_argument("--capacity", type=int, default=200,
|
||||
help="Settings.memory_bank_capacity; lower it below "
|
||||
"--memories to exercise eviction (default: 200)")
|
||||
p.add_argument("--narration-bytes", type=int, default=4000,
|
||||
help="length of an AI action's text; production averages "
|
||||
"~2.1 KB per action alternating with player input "
|
||||
"(default: 4000)")
|
||||
# Production's longest adventure carries 886 B of text per action averaged
|
||||
# over both kinds. AI actions alternate with a one-line player input, so
|
||||
# the AI half has to be about twice that.
|
||||
p.add_argument("--narration-bytes", type=int, default=1700,
|
||||
help="length of an AI action's text; alternating with a "
|
||||
"one-line player input this averages ~890 B/action, "
|
||||
"which is what production measures (default: 1700)")
|
||||
p.add_argument("--snapshot-bytes", type=int, default=232_000,
|
||||
help="context_snapshot per action; 232 KB is the average "
|
||||
"on production's longest adventure, 163 KB is the "
|
||||
"average across the whole table (default: 232000)")
|
||||
p.add_argument("--shapes", default=",".join(SHAPES),
|
||||
help=f"comma-separated subset of: {', '.join(SHAPES)}")
|
||||
p.add_argument("--repeat", type=int, default=1,
|
||||
@@ -314,11 +393,8 @@ def main(argv=None) -> int:
|
||||
if unknown:
|
||||
sys.exit(f"unknown shape(s): {', '.join(unknown)}")
|
||||
|
||||
global NARRATION
|
||||
repeats = max(1, -(-args.narration_bytes // len(_PARAGRAPH)))
|
||||
NARRATION = (_PARAGRAPH * repeats)[: args.narration_bytes]
|
||||
|
||||
rng = random.Random(args.seed)
|
||||
_build_text(args, random.Random(args.seed ^ 0x5F5F))
|
||||
adv_id, user_id = build_fixture(args, rng)
|
||||
install_fakes(user_id, rng)
|
||||
|
||||
@@ -327,7 +403,8 @@ def main(argv=None) -> int:
|
||||
# bytes would drown everything the shapes report.
|
||||
meter.attach(engine)
|
||||
|
||||
print(f"fixture: {args.actions} actions × {args.narration_bytes} B · "
|
||||
print(f"fixture: {args.actions} actions × {args.narration_bytes} B "
|
||||
f"(+{args.snapshot_bytes // 1024} kB snapshot, deferred) · "
|
||||
f"{args.memories} memories × {EMBEDDING_DIMS} dims · "
|
||||
f"capacity {args.capacity}")
|
||||
if args.no_embeddings:
|
||||
|
||||
@@ -86,6 +86,17 @@ export const api = {
|
||||
createAdventure: (data) => request('/adventures', { method: 'POST', body: JSON.stringify(data) }),
|
||||
updateAdventure: (id, data) => request(`/adventures/${id}`, { method: 'PATCH', body: JSON.stringify(data) }),
|
||||
deleteAdventure: (id) => request(`/adventures/${id}`, { method: 'DELETE' }),
|
||||
// A page of the story, walking backwards. `beforeId` is the oldest action
|
||||
// already on screen; omit it for the newest window. Anchored on an action
|
||||
// rather than an offset so a turn landing mid-scroll cannot shift the page
|
||||
// out from under the reader.
|
||||
getActions: (advId, { beforeId, limit } = {}) => {
|
||||
const params = new URLSearchParams()
|
||||
if (beforeId != null) params.set('before_id', beforeId)
|
||||
if (limit != null) params.set('limit', limit)
|
||||
const query = params.toString()
|
||||
return request(`/adventures/${advId}/actions${query ? `?${query}` : ''}`)
|
||||
},
|
||||
updateAction: (advId, actionId, text) =>
|
||||
request(`/adventures/${advId}/actions/${actionId}`, { method: 'PATCH', body: JSON.stringify({ text }) }),
|
||||
deleteAction: (advId, actionId) =>
|
||||
|
||||
@@ -332,6 +332,31 @@ button:disabled { opacity: 0.45; cursor: default; transform: none; box-shadow: n
|
||||
white-space: pre-wrap;
|
||||
padding-bottom: 16px;
|
||||
}
|
||||
/* The head of a windowed transcript: how much story is still above, and the
|
||||
way back to it. Deliberately quiet — it is a seam in the page, not a
|
||||
feature — and it sits inside .story so it inherits the story face. */
|
||||
.story-earlier {
|
||||
text-align: center;
|
||||
margin: 4px 0 22px;
|
||||
font-size: 0.9rem;
|
||||
/* No action-in animation here: this appears above content the reader is
|
||||
already looking at, and a fade would read as the story moving. */
|
||||
}
|
||||
.story-earlier .dim { color: var(--text-dim); font-style: italic; }
|
||||
.story-earlier button {
|
||||
background: none;
|
||||
border: none;
|
||||
border-bottom: 1px solid rgba(159, 199, 209, 0.3);
|
||||
color: var(--text-dim);
|
||||
font: inherit;
|
||||
font-style: italic;
|
||||
cursor: pointer;
|
||||
padding: 2px 4px;
|
||||
}
|
||||
.story-earlier button:hover {
|
||||
color: var(--accent-bright);
|
||||
border-bottom-color: var(--accent-bright);
|
||||
}
|
||||
.story .action {
|
||||
position: relative;
|
||||
margin-bottom: 15px;
|
||||
|
||||
+106
-6
@@ -1,4 +1,4 @@
|
||||
import { Fragment, useCallback, useEffect, useMemo, useRef, useState } from 'react'
|
||||
import { Fragment, useCallback, useEffect, useLayoutEffect, useMemo, useRef, useState } from 'react'
|
||||
import { createPortal } from 'react-dom'
|
||||
import { useNavigate, useParams } from 'react-router-dom'
|
||||
import { api } from '../api'
|
||||
@@ -1310,10 +1310,20 @@ export default function Play() {
|
||||
// Read-only browsing of an earlier attempt at a past turn (see VariantPager).
|
||||
// One at a time; null when every message is showing its active version.
|
||||
const [preview, setPreview] = useState(null)
|
||||
// The transcript is a window on the story, not the whole of it: the page
|
||||
// load brings the newest page and older ones arrive as the reader scrolls
|
||||
// up. `total` is the story's real length, for the "N earlier" line.
|
||||
const [total, setTotal] = useState(0)
|
||||
const [hasMore, setHasMore] = useState(false)
|
||||
const [loadingOlder, setLoadingOlder] = useState(false)
|
||||
const storyEndRef = useRef(null)
|
||||
const abortRef = useRef(null)
|
||||
const pinnedRef = useRef(true) // autoscroll only while the reader is at the bottom
|
||||
const inputRef = useRef(null)
|
||||
// Set just before older actions are prepended, read once afterwards to put
|
||||
// the reader back where they were. See the layout effect below.
|
||||
const restoreScrollRef = useRef(null)
|
||||
const loadingOlderRef = useRef(false)
|
||||
|
||||
// The drop cap belongs to the story's first narrated beat. `start` is the
|
||||
// scenario's opening prompt, so it's usually that; an adventure begun blank
|
||||
@@ -1340,18 +1350,76 @@ export default function Play() {
|
||||
|
||||
useEffect(() => {
|
||||
api.getAdventure(id)
|
||||
.then((adv) => { setAdventure(adv); setActions(adv.actions) })
|
||||
.then((adv) => {
|
||||
setAdventure(adv)
|
||||
setActions(adv.actions)
|
||||
setTotal(adv.action_count ?? adv.actions.length)
|
||||
setHasMore(adv.actions.length < (adv.action_count ?? adv.actions.length))
|
||||
})
|
||||
.catch(() => navigate('/'))
|
||||
}, [id, navigate])
|
||||
|
||||
// Fetch the page above the one on screen and prepend it.
|
||||
//
|
||||
// Anchored on the oldest action we hold rather than on a count, so a turn
|
||||
// landing while the reader scrolls cannot shift the page. Guarded by a ref
|
||||
// as well as state because scroll fires far faster than React re-renders,
|
||||
// and two in-flight requests would fetch the same page twice.
|
||||
const loadOlder = useCallback(async () => {
|
||||
if (loadingOlderRef.current || !hasMore) return
|
||||
const oldest = actions[0]
|
||||
if (!oldest) return
|
||||
loadingOlderRef.current = true
|
||||
setLoadingOlder(true)
|
||||
try {
|
||||
const page = await api.getActions(id, { beforeId: oldest.id })
|
||||
if (page.actions.length) {
|
||||
// Record the height before the prepend; the layout effect below uses
|
||||
// it to keep the reader looking at the same paragraph.
|
||||
restoreScrollRef.current = {
|
||||
height: document.documentElement.scrollHeight,
|
||||
top: window.scrollY,
|
||||
}
|
||||
setActions((prev) => {
|
||||
// Defensive: never let a page the reader already holds duplicate a
|
||||
// message. Cheap, and the alternative is a visibly doubled turn.
|
||||
const known = new Set(prev.map((a) => a.id))
|
||||
return [...page.actions.filter((a) => !known.has(a.id)), ...prev]
|
||||
})
|
||||
}
|
||||
setTotal(page.total)
|
||||
setHasMore(page.has_more)
|
||||
} catch {
|
||||
// Leave hasMore alone: a failed fetch should let the reader try again
|
||||
// by scrolling, not permanently hide the rest of their story.
|
||||
} finally {
|
||||
loadingOlderRef.current = false
|
||||
setLoadingOlder(false)
|
||||
}
|
||||
}, [actions, hasMore, id])
|
||||
|
||||
// Put the viewport back after a prepend. useLayoutEffect, not useEffect:
|
||||
// this has to run before the browser paints, or the reader sees the story
|
||||
// jump and then snap back.
|
||||
useLayoutEffect(() => {
|
||||
const mark = restoreScrollRef.current
|
||||
if (!mark) return
|
||||
restoreScrollRef.current = null
|
||||
const grown = document.documentElement.scrollHeight - mark.height
|
||||
if (grown > 0) window.scrollTo({ top: mark.top + grown })
|
||||
}, [actions])
|
||||
|
||||
useEffect(() => {
|
||||
const onScroll = () => {
|
||||
pinnedRef.current =
|
||||
window.innerHeight + window.scrollY >= document.documentElement.scrollHeight - 120
|
||||
// Start the next page before the reader reaches the top, so the story
|
||||
// is usually already there by the time they would have noticed its end.
|
||||
if (window.scrollY < 400) loadOlder()
|
||||
}
|
||||
window.addEventListener('scroll', onScroll, { passive: true })
|
||||
return () => window.removeEventListener('scroll', onScroll)
|
||||
}, [])
|
||||
}, [loadOlder])
|
||||
|
||||
useEffect(() => {
|
||||
// Snap to the real document bottom (below the sticky composer), not to
|
||||
@@ -1375,6 +1443,9 @@ export default function Play() {
|
||||
const handleEvent = useCallback((event) => {
|
||||
if (event.type === 'player') {
|
||||
setActions((prev) => [...prev, event.action])
|
||||
// The window grew at the bottom, so the story did too. Kept in step by
|
||||
// hand because nothing re-reads the count between turns.
|
||||
setTotal((n) => n + 1)
|
||||
} else if (event.type === 'chunk') {
|
||||
setStreaming((prev) => (prev ?? '') + event.text)
|
||||
} else if (event.type === 'reasoning') {
|
||||
@@ -1383,6 +1454,7 @@ export default function Play() {
|
||||
setStreaming(null)
|
||||
setReasoningStream(null)
|
||||
setActions((prev) => [...prev, event.action])
|
||||
setTotal((n) => n + 1)
|
||||
handleScriptReport(event.script)
|
||||
} else if (event.type === 'stopped') {
|
||||
setStreaming(null)
|
||||
@@ -1443,8 +1515,14 @@ export default function Play() {
|
||||
await api.retry(id, handleEvent, signal)
|
||||
} catch (err) {
|
||||
// Failed retry (409, network): the optimistically removed action may
|
||||
// still exist server-side — resync instead of guessing.
|
||||
api.getAdventure(id).then((adv) => setActions(adv.actions)).catch(() => {})
|
||||
// still exist server-side — resync instead of guessing. Resyncing
|
||||
// collapses the transcript back to the newest window, which is the
|
||||
// right call: the reader's place is already lost by the failure.
|
||||
api.getAdventure(id).then((adv) => {
|
||||
setActions(adv.actions)
|
||||
setTotal(adv.action_count ?? adv.actions.length)
|
||||
setHasMore(adv.actions.length < (adv.action_count ?? adv.actions.length))
|
||||
}).catch(() => {})
|
||||
throw err
|
||||
}
|
||||
})
|
||||
@@ -1454,7 +1532,12 @@ export default function Play() {
|
||||
setToast(null)
|
||||
setPreview(null)
|
||||
try {
|
||||
setActions(await api.undo(id))
|
||||
// A window, not the whole story — undo is the action most likely to be
|
||||
// repeated several times running, so it must not re-fetch everything.
|
||||
const page = await api.undo(id)
|
||||
setActions(page.actions)
|
||||
setTotal(page.total)
|
||||
setHasMore(page.has_more)
|
||||
} catch (err) {
|
||||
setToast({ text: err.message, isError: true })
|
||||
}
|
||||
@@ -1494,6 +1577,7 @@ export default function Play() {
|
||||
try {
|
||||
await api.deleteAction(id, actionId)
|
||||
setActions((prev) => prev.filter((a) => a.id !== actionId))
|
||||
setTotal((n) => Math.max(0, n - 1))
|
||||
} catch (err) {
|
||||
setToast({ text: err.message, isError: true })
|
||||
}
|
||||
@@ -1534,6 +1618,22 @@ export default function Play() {
|
||||
{actions.length === 0 && streaming === null && (
|
||||
<div className="empty">A blank page. Type something below to begin your story.</div>
|
||||
)}
|
||||
{/* Scrolling up loads the rest. The button is not decoration: on a
|
||||
short viewport the story may not be tall enough to scroll at all,
|
||||
and a reader who cannot scroll must still be able to get back to
|
||||
the beginning. */}
|
||||
{hasMore && (
|
||||
<div className="story-earlier">
|
||||
{loadingOlder ? (
|
||||
<span className="dim">Turning back the pages…</span>
|
||||
) : (
|
||||
<button type="button" onClick={loadOlder}>
|
||||
{Math.max(total - actions.length, 0)} earlier
|
||||
{total - actions.length === 1 ? ' moment' : ' moments'}
|
||||
</button>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
{actions.map((action, i) => {
|
||||
const isPlayer = PLAYER_TYPES.includes(action.type)
|
||||
// A player action opens a new turn, so that's where the ornamental
|
||||
|
||||
@@ -164,8 +164,10 @@ additions the plan did not anticipate:
|
||||
forgotten. Vectors are held as `array("f")` — 6 KB each, matching the column;
|
||||
a list of Python floats would have been eight times the plan's RAM estimate.
|
||||
|
||||
Remaining: **step 6, infinite scroll upward in `Play.jsx`.** The page load is
|
||||
unchanged at 426.7 kB and is now the largest single read in the app.
|
||||
**Step 6 landed on 2026-08-17**, along with everything else this plan left open — see
|
||||
"Closing the plan" at the end. The page load is a window of 60 actions now: 62.6 kB on
|
||||
a 600-action fixture, down from 606.0 kB, and no longer a function of the story's
|
||||
length.
|
||||
|
||||
## Guardrails to add with this work
|
||||
|
||||
@@ -182,6 +184,9 @@ Deliberately **not** taken: moving `context_snapshot` out of the database. It co
|
||||
nothing on reads now that it is deferred, and storage is ~$0.02/mo. Revisit only if
|
||||
backups or storage start to hurt.
|
||||
|
||||
> **Revisit it.** That call weighed egress and got egress right, but it never weighed
|
||||
> the free tier's *storage* ceiling — see "Storage, which this plan did not cost" below.
|
||||
|
||||
## Verification
|
||||
|
||||
- Harness: `python -m tools.stress_session`, memory bank **on**, before and after,
|
||||
@@ -193,3 +198,120 @@ backups or storage start to hurt.
|
||||
playthrough number is finally honest.
|
||||
- The existing `test_egress.py` guard must still pass — nothing here should touch the
|
||||
deferred action columns.
|
||||
|
||||
## Verified on production, 2026-08-17
|
||||
|
||||
Two things were still taken on trust when this shipped: every measurement had run on
|
||||
SQLite, and every number came from a synthetic fixture. Both are now checked.
|
||||
|
||||
### The migration landed on real Postgres
|
||||
|
||||
`schema_version` reads **41**, matching the repo's `LATEST_VERSION`. The live schema has
|
||||
`embedding_blob bytea` and `embedded boolean`, so the `{dialect: sql}` map in migration
|
||||
38 spells BYTEA correctly against a real server — the one thing tests could not prove,
|
||||
since `test_migration_38_is_spelled_for_both_dialects` only inspects the SQL string.
|
||||
The backfill is complete: 134 memories, `embedded = 134`, `embedding_blob = 134`, no
|
||||
stragglers and no rows skipped as malformed.
|
||||
|
||||
### The 5x is real, on real vectors
|
||||
|
||||
| | bytes | per memory |
|
||||
|---|---|---|
|
||||
| `embedding` (JSON) | 4,150,121 | 30,971 |
|
||||
| `embedding_blob` (float32) | 823,296 | 6,144 |
|
||||
|
||||
**5.04x**, against the plan's predicted ~31 KB → 6,144 B. The largest real bank is 100
|
||||
memories = 614,400 B of vectors, so the old code fetched **~3.10 MB per retrieval** on
|
||||
that adventure — which is where the 3,153 kB measured on production came from. That
|
||||
figure is now fully accounted for.
|
||||
|
||||
### SQLite and Postgres agree
|
||||
|
||||
`tools.stress_session` gained an `AIDND_STRESS_DATABASE_URL` escape hatch and was run
|
||||
against a throwaway Neon database at the default fixture (200 actions, 100 memories):
|
||||
|
||||
| shape | SQLite | Postgres |
|
||||
|---|---|---|
|
||||
| index | 4.1 kB | 4.1 kB |
|
||||
| page load | 426.7 kB | 425.0 kB |
|
||||
| one turn, cold | 723.4 kB | 722.3 kB |
|
||||
| one turn, warm | 122.3 kB | **121.1 kB** |
|
||||
| Insights | 117.9 kB | 116.7 kB |
|
||||
| Memories drawer | 23.7 kB | 21.7 kB |
|
||||
| `run_post_turn` | 0.7 kB | 0.6 kB |
|
||||
|
||||
Within 0.5% everywhere. The dialect caveat in the harness docstring is real but small:
|
||||
what dominates is which columns get asked for, and the ORM decides that identically.
|
||||
The warm turn spends **1.7 kB on `memories`, 1% of the read** — the cache behaves on
|
||||
psycopg exactly as it does on SQLite.
|
||||
|
||||
### The page load is worse than modelled, for a different reason
|
||||
|
||||
The synthetic fixture is **~2x heavier per action than production**: 994 B/action real
|
||||
against ~2,133 B/action synthetic, so a real 200-action adventure is ~194 kB, not 427.
|
||||
But the largest real adventure is **607 actions**, not 200, and costs **589.5 kB** in
|
||||
one response. Step 6 is more urgent than this plan assumed, and for the opposite
|
||||
reason to the one modelled — stories get *longer* than the fixture, not heavier.
|
||||
|
||||
Worth fixing the fixture's narration size when step 6 lands, so the harness stops
|
||||
flattering the per-action figure while understating the length.
|
||||
|
||||
### Storage, which this plan did not cost
|
||||
|
||||
`context_snapshot` is **150.8 MB of uncompressed JSON across 944 actions** — ~163 kB a
|
||||
row on average, and ~232 kB a row in the largest adventure, against the ~74 KB/row the
|
||||
comment in `models.py` claims. TOAST compresses it to ~89 MB on disk, but
|
||||
`octet_length` is what would cross the wire, because Postgres decompresses before
|
||||
sending. Deferral is the only thing standing between a bulk read and a 137 MB query.
|
||||
|
||||
The database is **99.6 MB total**, of which `actions` is **88.9 MB**. Neon's free tier
|
||||
is 512 MB. At ~94 kB of disk per action that ceiling arrives at roughly **5,400
|
||||
actions**, and 944 are already stored. So the "~$0.02/mo, leave it in the database"
|
||||
call above is wrong for the tier this actually runs on — not because reads cost
|
||||
anything, but because the free tier meters *storage*, and that is the constraint with
|
||||
a cliff. Dropping the dead `memories.embedding` column reclaims 4.05 MB (4%), which
|
||||
helps and does not solve it.
|
||||
|
||||
None of the numbers above required reading a single row of anyone's content: counts,
|
||||
`octet_length` sums and catalog sizes only.
|
||||
|
||||
## Closing the plan, 2026-08-17
|
||||
|
||||
Everything above landed the same day the verification did.
|
||||
|
||||
| | before | after |
|
||||
|---|---|---|
|
||||
| page load, 600 actions | 606.0 kB | **62.6 kB**, and flat in story length |
|
||||
| adventures index, 6 adventures | 469.7 kB | **0.3 kB** |
|
||||
| `context_snapshot` stored | ~89 MB | ~43 MB (after a VACUUM) |
|
||||
| a played turn | 6.4 MB (2026-08-16) | 123 kB |
|
||||
|
||||
**Step 6, the window.** `GET /adventures/{id}` returns the newest `ACTION_PAGE`
|
||||
actions and the story's length; `GET /{id}/actions?before_id=` walks back from there.
|
||||
Anchored on an action rather than an offset — the offset version breaks precisely when
|
||||
a turn lands mid-scroll, handing the reader one action twice and hiding another — and
|
||||
that choice is also what makes it survive the story tree, since it compares indices to
|
||||
order a branch rather than treating them as positions.
|
||||
|
||||
**Byte assertions.** `tests/test_egress.py` now carries per-action budgets as well as
|
||||
column guards, plus one test whose only job is to fail if the fixture ever gets too
|
||||
small for the budgets to catch anything.
|
||||
|
||||
**Column projections.** `ACTION_LIST_COLUMNS` and `MEMORY_LIST_COLUMNS` name what a
|
||||
list response renders, and the adventures index selects four columns instead of the
|
||||
entity. That one was not just future-proofing: an Adventure carries seven text and JSON
|
||||
columns the index never shows.
|
||||
|
||||
**The JSON vector column is gone**, and dropping it exposed a live bug — changing your
|
||||
embedding model had stopped re-embedding the bank the day migration 38 shipped. See
|
||||
`tests/test_embedding_model_switch.py`.
|
||||
|
||||
**Storage.** `context_snapshot` is zlib-compressed through a TypeDecorator
|
||||
(`app/compression.py`, migrations 43–45), so the call sites never learned about it.
|
||||
3.5x on real Postgres. **This does not shrink anything until `VACUUM FULL actions`
|
||||
runs** — Postgres marks dropped columns rather than reclaiming them, and the backfill
|
||||
leaves a dead tuple per row.
|
||||
|
||||
Not done: nothing exercises the scroll behaviour in a browser. The frontend has no test
|
||||
runner, and prepend-and-restore-scroll is the part most likely to feel wrong even when
|
||||
it is correct.
|
||||
|
||||
+132
-28
@@ -3,27 +3,57 @@
|
||||
Read this first when picking the project back up. Updated at the end of a working
|
||||
session; the per-phase plan files hold the detail, this holds the thread.
|
||||
|
||||
**Last updated: 2026-08-16.**
|
||||
**Last updated: 2026-08-17.**
|
||||
|
||||
---
|
||||
|
||||
## The live URL is not the one in render.yaml
|
||||
|
||||
**Production is `https://ai-dnd-1gmp.onrender.com`.** Render appended a suffix to the
|
||||
`ai-dnd` service name in `render.yaml`, and plain `ai-dnd.onrender.com` belongs to a
|
||||
different, suspended service that answers 503 with "suspended by its owner" — which is
|
||||
easy to mistake for this deploy being down. The authoritative link is the one the
|
||||
project page points at (`docs/index.html`), not the service name in the blueprint.
|
||||
`GET /api/health` on the real host returns `{"ok":true}`.
|
||||
|
||||
---
|
||||
|
||||
## Needs a human: one VACUUM after the next deploy
|
||||
|
||||
Migration 43 compresses `context_snapshot`, and **Postgres does not hand the disk back
|
||||
on its own.** `DROP COLUMN` only marks a column dropped, and the backfill leaves a dead
|
||||
tuple per row, so `actions` gets *bigger* before it gets smaller — peaking near twice
|
||||
its size while both columns are live. Once the deploy is up and healthy, run once:
|
||||
|
||||
```sql
|
||||
VACUUM FULL actions;
|
||||
```
|
||||
|
||||
It needs exclusive access to the table and free space equal to the finished copy. On
|
||||
the 2026-08-17 figures: 99.6 MB now, peaking near 200 during the migration, settling
|
||||
around 53 afterwards, against a 512 MB tier. Skipping it is safe and simply leaves the
|
||||
win unrealised — the database keeps working, it just stays large.
|
||||
|
||||
Same caveat applies to migration 42 dropping `memories.embedding` (4 MB).
|
||||
|
||||
---
|
||||
|
||||
## Pick up here
|
||||
|
||||
**`plan/13-memory-embedding-cost.md`, step 6 — infinite scroll upward in `Play.jsx`.**
|
||||
**`plan/14-phase-story-tree.md` — the tree itself.** `plan/13` is finished. Its design
|
||||
is settled in `plan/14` and nothing about it has been built.
|
||||
|
||||
Opening a finished 200-action adventure fetches **426.7 kB** in one response, and after
|
||||
this session's work that is comfortably the largest single read in the app — a turn is
|
||||
now 122 kB, Insights 118 kB, the Memories drawer 24 kB. The backend already has the
|
||||
windowing primitives (`context/history.py`: `tail_range`, `slice_`, `count`), and
|
||||
`GET /adventures/{id}/actions` exists. What is missing is a paged shape for it and a
|
||||
`Play.jsx` that loads the newest turns and fetches older ones as the reader scrolls up.
|
||||
Two things from the egress work are worth carrying into it:
|
||||
|
||||
Watch for: the story is a flat list today, and **the story tree replaces it**
|
||||
(`plan/14-phase-story-tree.md`). Paging that reads by *position from the end* survives
|
||||
that change; paging that assumes `Action.index` is a dense 0..n sequence does not.
|
||||
- **Paging already anticipates the tree.** `action_window` in `routers/adventures.py`
|
||||
anchors on an action id and orders by comparing `Action.index`, never by treating
|
||||
index as a position. A branch changes which actions are on the path, not how two of
|
||||
them order, so the anchor survives; anything counting offsets would not.
|
||||
- **Weigh new columns in bytes.** A tree adds parent/branch columns to `actions`, which
|
||||
is already the table that fills the disk. `tests/test_egress.py` has byte ceilings
|
||||
now — they will tell you.
|
||||
|
||||
After that: the tree itself. Its design is settled in `plan/14`; nothing about it has
|
||||
been built.
|
||||
**Before the next deploy:** one `VACUUM FULL actions;` — see below.
|
||||
|
||||
---
|
||||
|
||||
@@ -91,6 +121,60 @@ Migrations 39/40 add `memories.embedded`, migration 41 drops the capacity defaul
|
||||
|
||||
---
|
||||
|
||||
## What happened on 2026-08-17, part two
|
||||
|
||||
Everything left open in `plan/13` closed, plus two bugs that fell out of doing it.
|
||||
|
||||
| shape | before today | after |
|
||||
|---|---|---|
|
||||
| page load, 600 actions | 606.0 kB | **62.6 kB** — and no longer grows with the story |
|
||||
| adventures index, 6 fat adventures | 469.7 kB | **0.3 kB** |
|
||||
| `context_snapshot` on disk | ~89 MB | ~43 MB (3.5x, after a VACUUM) |
|
||||
| database total | 99.6 MB | ~53 MB projected |
|
||||
|
||||
**The story is a window now.** `GET /adventures/{id}` returns the newest 60 actions and
|
||||
`action_count`; older pages come from `GET /{id}/actions?before_id=`. Anchored on an
|
||||
action, never an offset — an offset counted back from the newest shifts every older
|
||||
position the moment a turn lands, which is exactly when someone is scrolling. `Play.jsx`
|
||||
prepends and restores scroll position in a `useLayoutEffect`, before paint.
|
||||
|
||||
**`context_snapshot` is compressed** (migrations 43–45, `app/compression.py`) via a
|
||||
TypeDecorator, so every call site still reads and writes a dict. Verified end to end on
|
||||
a throwaway Neon database: 720,864 B of JSON to 204,293 B of bytea, every row equal.
|
||||
|
||||
**The JSON vector column is gone** (migration 42) — and dropping it exposed that
|
||||
changing your embedding model had silently stopped re-embedding the bank since
|
||||
migration 38. The settings route cleared the dead column and left `embedded` true, so
|
||||
`_embed_pending` never saw those rows and retrieval kept ranking against the old
|
||||
model's vectors. Nothing reported it: `cosine` returns 0.0 on a width mismatch.
|
||||
`tests/test_embedding_model_switch.py`.
|
||||
|
||||
**Byte ceilings exist** (`tests/test_egress.py`), including one test whose only job is
|
||||
to prove the ceilings would catch something.
|
||||
|
||||
**List responses name their columns.** The index was loading whole Adventure entities —
|
||||
seven text and JSON columns, ~15 kB a row — to render a title and a snippet.
|
||||
|
||||
## What happened on 2026-08-17, part one
|
||||
|
||||
No new behaviour — a verification pass on what shipped the day before, because every
|
||||
number in the section above had been measured on SQLite against a synthetic fixture.
|
||||
Full write-up in `plan/13` under "Verified on production".
|
||||
|
||||
**It holds.** `schema_version` is 41 on the live Postgres with `embedding_blob bytea`
|
||||
and `embedded boolean` present, so migration 38's dialect map is correct against a real
|
||||
server. The backfill is complete (134/134). The packed vectors are **5.04x** smaller
|
||||
than the JSON on real data — 30,971 → 6,144 bytes a memory, as predicted.
|
||||
|
||||
**SQLite was not lying.** `tools.stress_session` can now target Postgres via
|
||||
`AIDND_STRESS_DATABASE_URL`, and every shape agrees within 0.5% — the warm turn is
|
||||
121.1 kB on Postgres against 122.3 kB on SQLite, with `memories` down to 1.7 kB of it.
|
||||
Run it against a **throwaway** database only; the harness writes, so it refuses any
|
||||
target whose name does not contain `stress` or `scratch`.
|
||||
|
||||
**Two corrections came out of it**, both above: the page-load model has the wrong
|
||||
shape (too heavy per action, far too short), and the storage ceiling was never costed.
|
||||
|
||||
## Things worth remembering
|
||||
|
||||
**The vector cache needs no invalidation callbacks, and that is why it is safe.** A
|
||||
@@ -112,23 +196,34 @@ beside `actions.variants`. Expect to need this for any future heavy column.
|
||||
**Any egress measurement must run with an embedding model set.** This is the second
|
||||
time that omission has hidden the biggest number in the room.
|
||||
|
||||
**Production has real users on it now. Measure it without reading it.** Counts,
|
||||
`sum(octet_length(...))` and `pg_total_relation_size` answer every sizing question
|
||||
asked so far, and none of them return anyone's story, memory text or email. When a
|
||||
real Postgres is needed for a *write* path, create a throwaway database beside the real
|
||||
one and drop it after — never point a harness at the production database.
|
||||
|
||||
**`octet_length` is the egress number, not the on-disk number.** Postgres TOAST
|
||||
compresses big JSON — `context_snapshot` is 150.8 MB uncompressed but ~89 MB stored —
|
||||
and decompresses before sending. Size reads with `octet_length`, size the storage bill
|
||||
with `pg_total_relation_size`, and do not mix them up.
|
||||
|
||||
---
|
||||
|
||||
## Still open from `plan/13`
|
||||
## `plan/13` is closed
|
||||
|
||||
- **Step 6, infinite scroll upward** — the pick-up item above.
|
||||
- **Query-count / byte assertions per endpoint**, extending `tests/test_egress.py`
|
||||
against production-sized fixtures. `dbmeter` is importable from tests (`from tools
|
||||
import dbmeter`) and was built with this in mind; nothing uses it there yet.
|
||||
- **Explicit column projections on read paths**, so the next heavy column is opt-**in**.
|
||||
Done for the memory paths, not as a general rule.
|
||||
- **Drop `memories.embedding`** (the JSON column) in a follow-up migration. It is still
|
||||
written by `set_vector` and read by nothing, kept so a rollback finds the vectors.
|
||||
`tests/test_memory_retrieval.py` has a guard asserting nothing selects it.
|
||||
All six of its open items landed on 2026-08-17. What is left is not from that plan:
|
||||
|
||||
Deliberately not taken: moving `context_snapshot` out of the database (~$0.02/mo, costs
|
||||
nothing on reads now that it is deferred), and pgvector (breaks the SQLite dev parity
|
||||
this codebase protects on purpose).
|
||||
- **The VACUUM**, above. Until it runs, the storage win is on paper.
|
||||
- **Nothing verifies the scroll behaviour in a browser.** The paging is covered by
|
||||
`tests/test_action_paging.py` and was exercised against a running backend, but the
|
||||
frontend has no test runner and the prepend-and-restore is the part most likely to
|
||||
feel wrong. Worth thirty seconds of scrolling a long adventure before trusting it.
|
||||
- **`ACTION_PAGE = 60` is a guess.** It should be a page or two of reading. If loading
|
||||
older turns feels like it interrupts, that is the number to move.
|
||||
|
||||
Deliberately not taken: moving `context_snapshot` out of the database entirely
|
||||
(compressing it bought the same runway for a much smaller change), and pgvector (breaks
|
||||
the SQLite dev parity this codebase protects on purpose).
|
||||
|
||||
---
|
||||
|
||||
@@ -136,9 +231,18 @@ this codebase protects on purpose).
|
||||
|
||||
```
|
||||
cd backend
|
||||
.venv/Scripts/python.exe -m pytest tests/ # 225 tests
|
||||
.venv/Scripts/python.exe -m tools.stress_session # egress report
|
||||
.venv/Scripts/python.exe -m pytest tests/ # 259 tests
|
||||
.venv/Scripts/python.exe -m tools.stress_session # egress report (SQLite)
|
||||
|
||||
# Same harness against a real Postgres. The target must be a THROWAWAY database
|
||||
# — this writes a synthetic adventure, and it refuses any name without
|
||||
# 'stress'/'scratch' in it.
|
||||
AIDND_STRESS_DATABASE_URL=postgresql://…/stress_scratch \
|
||||
.venv/Scripts/python.exe -m tools.stress_session
|
||||
```
|
||||
|
||||
On Windows the report's box-drawing characters crash the default cp1252 console;
|
||||
prefix with `PYTHONIOENCODING=utf-8`.
|
||||
|
||||
Port 8000 is shared with the job-pipeline app, which will squat it and silently shadow
|
||||
the AI-DnD API — free it before running the backend, or move the vite proxy.
|
||||
|
||||
Reference in New Issue
Block a user