Cut database egress 189x by deferring the prompt snapshot

The free-tier 5 GB/month network transfer allowance ran out, which blocks
connections outright. The database is only ~55 MB, so 5 GB meant the whole
thing was being pulled roughly 90 times over.

Cause: actions is 39 MB of that 55 MB -- 541 rows at ~74 KB each, almost
entirely context_snapshot, which stores the whole assembled prompt for a
turn. Every adventure load and every turn fetched all of it in order to
read two small things out of it: the world-change chips under an AI
message (Action.world_changes) and the emit block re-attached when
replaying history to the model (_history_text). The Insights viewer is
the only consumer that wants the whole snapshot, and it asks for one
action at a time.

Lifts that slice into its own small actions.world_delta column
(migration 36) and marks context_snapshot, state_before and
world_state_before deferred, so they load only when something touches
the attribute -- Insights, undo and retry, all single-action paths.
The backfill runs server-side, dialect-specific (json_extract on SQLite,
#> on Postgres), because pulling 39 MB of snapshots into Python to
rewrite a slice of each would defeat the purpose.

Measured at production shape (541 actions, 72 KB snapshots), one
adventure load goes from 38.46 MB to 0.20 MB. The traffic that consumed
5 GB would now be about 27 MB.

Deliberately not included: limiting the history query to recent actions,
and removing the redundant db.refresh(adventure) calls. Both were sized
against the old numbers; against a 0.20 MB load they would take ~27 MB a
month down to ~10 MB, which is not worth the complexity.

tests/test_egress.py hooks before_cursor_execute and asserts the emitted
SQL never names the deferred columns during a bulk load, so this cannot
regress silently. 123 tests pass.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01UeQVy5bEjLhfgWNc27Efet
This commit is contained in:
parththakkar106
2026-08-03 20:35:54 +05:30
co-authored by Claude Opus 5
parent 7c538c8235
commit f1bd099ec8
5 changed files with 276 additions and 19 deletions
+176
View File
@@ -0,0 +1,176 @@
"""Guards on how much the database is asked for.
context_snapshot holds the entire assembled prompt for a turn (~74 KB/row in
production, 94% of the database). It used to be pulled for every action on
every adventure load and every turn, to read two tiny things out of it. These
tests fail if that regresses.
python -m pytest tests/test_egress.py -v
"""
import os
import tempfile
_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False)
_tmp.close()
os.environ["AIDND_DB_PATH"] = _tmp.name
os.environ.pop("AIDND_DATABASE_URL", None)
os.environ.pop("DATABASE_URL", None)
import pytest
from fastapi import Depends
from fastapi.testclient import TestClient
from sqlalchemy import event, text
from app import auth, limits, migrations, models
from app.database import Base, SessionLocal, engine, get_db
from app.main import app
# A stand-in for the real thing: the assembled prompt, which is what makes the
# column enormous, plus the small world_state slice the UI actually needs.
BIG_SNAPSHOT = {
"system": "x" * 20_000,
"story": "y" * 40_000,
"world_state": {
"delta": {"player.hp": -15},
"report": {"applied": [{"path": "player.hp", "old": 100, "new": 85}]},
"state": {"player": {"hp": 85}},
},
}
@pytest.fixture()
def sql_log():
"""Every statement the ORM sends, for asserting on what was selected."""
statements: list[str] = []
def record(conn, cursor, statement, parameters, context, executemany):
statements.append(statement)
event.listen(engine, "before_cursor_execute", record)
try:
yield statements
finally:
event.remove(engine, "before_cursor_execute", record)
@pytest.fixture()
def client(monkeypatch):
Base.metadata.create_all(bind=engine)
setup = SessionLocal()
user = models.User(is_guest=False, email="egress@example.com")
setup.add(user)
setup.flush()
setup.add(models.Settings(user_id=user.id, api_key="enc:dummy", model="test-model"))
adventure = models.Adventure(user_id=user.id, title="Cave", script_state={})
setup.add(adventure)
setup.flush()
for i in range(12):
setup.add(models.Action(
adventure_id=adventure.id, index=i,
type="ai" if i % 2 else "do", text=f"Action {i}.",
context_snapshot=BIG_SNAPSHOT,
world_delta={"delta": {"player.hp": -15},
"applied": [{"path": "player.hp", "old": 100, "new": 85}]},
))
setup.commit()
adv_id, user_id = adventure.id, user.id
setup.close()
monkeypatch.setattr(limits, "rate_limit", lambda *a, **k: None)
monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None)
def _current_user(db=Depends(get_db)):
return db.get(models.User, user_id)
app.dependency_overrides[auth.get_current_user] = _current_user
c = TestClient(app)
c.adv_id = adv_id
try:
yield c
finally:
app.dependency_overrides.clear()
Base.metadata.drop_all(bind=engine)
def action_selects(statements: list[str]) -> list[str]:
return [s for s in statements if "FROM actions" in s and s.lstrip().upper().startswith("SELECT")]
# ------------------------------------------------------- the deferred columns
def test_loading_an_adventure_does_not_fetch_context_snapshot(client, sql_log):
r = client.get(f"/api/adventures/{client.adv_id}")
assert r.status_code == 200, r.text
assert len(r.json()["actions"]) == 12
selects = action_selects(sql_log)
assert selects, "expected at least one SELECT against actions"
offenders = [s for s in selects if "context_snapshot" in s]
assert offenders == [], f"context_snapshot was fetched in bulk:\n{offenders[0][:400]}"
def test_state_before_and_world_state_before_are_not_fetched_in_bulk(client, sql_log):
"""Both are rollback snapshots, only ever needed for the single action
being undone or retried."""
client.get(f"/api/adventures/{client.adv_id}")
selects = action_selects(sql_log)
for column in ("state_before", "world_state_before"):
offenders = [s for s in selects if column in s]
assert offenders == [], f"{column} was fetched in bulk"
def test_world_changes_still_works_without_the_snapshot(client):
"""The chips under an AI message must survive the snapshot being deferred."""
r = client.get(f"/api/adventures/{client.adv_id}")
ai = [a for a in r.json()["actions"] if a["type"] == "ai"]
assert ai, "fixture should have AI actions"
assert ai[0]["world_changes"] == [
{"kind": "stat", "label": "hp", "delta": -15, "value": 85}
]
def test_snapshot_is_still_reachable_on_demand(client):
"""Deferred means lazy, not gone — Insights still gets the full thing."""
r = client.get(f"/api/adventures/{client.adv_id}")
action_id = r.json()["actions"][0]["id"]
r = client.get(f"/api/adventures/{client.adv_id}/actions/{action_id}/context")
assert r.status_code == 200, r.text
assert r.json()["system"] == "x" * 20_000
# ------------------------------------------------------------------ backfill
def test_backfill_populates_world_delta_from_existing_snapshots(client):
"""Migration 36 lifts the slice out server-side, without reading the
snapshots into Python."""
db = SessionLocal()
try:
db.execute(text("UPDATE actions SET world_delta = NULL"))
db.commit()
assert db.query(models.Action).filter(models.Action.world_delta.isnot(None)).count() == 0
with engine.begin() as conn:
migrations._backfill_world_delta(conn)
db.expire_all()
actions = db.query(models.Action).all()
assert all(a.world_delta is not None for a in actions)
assert actions[0].world_delta["delta"] == {"player.hp": -15}
assert actions[0].world_delta["applied"] == [
{"path": "player.hp", "old": 100, "new": 85}
]
finally:
db.close()
def test_backfill_leaves_actions_without_world_state_alone(client):
db = SessionLocal()
try:
db.execute(text("UPDATE actions SET world_delta = NULL, context_snapshot = '{\"story\": \"s\"}'"))
db.commit()
with engine.begin() as conn:
migrations._backfill_world_delta(conn)
db.expire_all()
assert all(a.world_delta is None for a in db.query(models.Action).all())
finally:
db.close()