"""How a campaign bundle grows with the campaign, measured rather than reasoned. python -m tools.m9_scale_report [--turns 120] [--budget 16384] Run from `backend/`. Plays a campaign of `--turns` turns against a scripted narrator with the real prompt builder and a realistic context budget, then exports it and reports where the bytes are. The question it exists to answer is the one M9's decision to carry historical prompts raises: **a per-turn prompt contains the story so far, so storing one per turn is quadratic in campaign length.** That is already true of the database — `compression.py` records the column as 89% of production storage — and M9 makes it true of the export as well. Reasoning about it gives the wrong number, because the prompt is bounded by the context budget rather than by the transcript: once the history window is full, each turn's snapshot stops growing and the total becomes linear again. Where that knee falls is a measurement. It also watches for the accidental costs §26 names: a query per row, a duplicated body of knowledge content, or a snapshot written more than once per turn. """ from __future__ import annotations import argparse import json import os import sys import tempfile import time from pathlib import Path _HERE = Path(__file__).resolve().parent sys.path.insert(0, str(_HERE.parent / "tests")) _DB = tempfile.NamedTemporaryFile(suffix="-m9-scale.db", delete=False) _DB.close() os.environ["AIDND_DB_PATH"] = _DB.name os.environ.pop("AIDND_DATABASE_URL", None) os.environ.pop("DATABASE_URL", None) from fastapi import Depends # noqa: E402 from fastapi.testclient import TestClient # noqa: E402 import m9_fixture # noqa: E402 from app import auth, limits, memorybank, models # noqa: E402 from app.database import Base, SessionLocal, engine, get_db # noqa: E402 from app.main import app # noqa: E402 from app.routers import adventures # noqa: E402 from fakes import ScriptedProvider, state_block # noqa: E402 #: Prose long enough that a turn is a turn rather than a sentence. The history #: window is what fills the prompt, so a fixture of three-word replies would #: measure a campaign nobody plays. PROSE = ( "The rain came harder off the fen and the lantern light shivered on the wet " "boards. Mara set down the cloth she had been folding and looked at him for " "a while without saying anything, the way she did when the answer was going " "to cost her something. Outside, somebody crossed the yard and did not stop." ) class _Stub: async def complete(self, system, prompt, **kwargs): return "The story had established a good deal by this point." async def embed(self, texts): return [[1.0, 0.5, 0.25, 0.125] for _ in texts] def _setup(budget: int) -> tuple[TestClient, int]: adventures.turns.OpenAICompatibleProvider = ScriptedProvider memorybank.embedding_provider = lambda s: _Stub() memorybank.summary_provider = lambda s: _Stub() limits.check_row_cap = lambda *a, **k: None Base.metadata.create_all(bind=engine) db = SessionLocal() try: user = models.User(is_guest=False, email="scale@example.com") db.add(user) db.flush() db.add(models.Settings( user_id=user.id, model="scale-model", embedding_model="stub", context_token_budget=budget, max_output_tokens=800, )) adventure = models.Adventure( user_id=user.id, title="Scale", auto_summarize=True, memory_bank_enabled=True, campaign_canon=m9_fixture.CAMPAIGN_CANON, ) db.add(adventure) db.flush() db.add(models.Action( adventure_id=adventure.id, type="start", text=m9_fixture.OPENING, )) db.commit() adv_id, user_id = adventure.id, user.id finally: db.close() app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) return TestClient(app), adv_id def _bundle_bytes(client, adv_id) -> tuple[int, dict, float]: started = time.perf_counter() response = client.get(f"/api/adventures/{adv_id}/export") seconds = time.perf_counter() - started response.raise_for_status() payload = response.json() return len(json.dumps(payload).encode("utf-8")), payload, seconds def _snapshot_bytes(payload: dict) -> int: """What the stored prompts cost **in the file**, which is the encoded size. Measured as they appear rather than decoded first: the question this report answers is how large the file gets and how close it comes to the import ceiling, so what counts is the bytes that actually travel. """ return sum( len(json.dumps(action[key]).encode("utf-8")) for action in payload["actions"] for key in ("contextSnapshotZ", "contextSnapshot") if action.get(key) ) def _decoded_snapshot_bytes(payload: dict) -> int: """What the same prompts would cost uncompressed, for the ratio.""" from app import bundle as bundle_module total = 0 for action in payload["actions"]: snapshot = ( action.get("contextSnapshot") if isinstance(action.get("contextSnapshot"), dict) else bundle_module._unpacked(action.get("contextSnapshotZ")) ) if isinstance(snapshot, dict): total += len(json.dumps(snapshot).encode("utf-8")) return total def _section_bytes(payload: dict) -> dict: """What each v3 addition costs in the file, separately. Needed because "the snapshots are 21% of the file" does not answer "what did M9 add": the state events, the proposals and the summaries are v3 additions too, and a claim about M9's cost that counted only the prompts would be understating it. """ def size(value) -> int: return len(json.dumps(value).encode("utf-8")) per_node = {"contextSnapshotZ": 0, "id": 0, "parentId": 0} for action in payload["actions"]: for key in per_node: if key in action: per_node[key] += size(action[key]) + len(key) + 4 return { "prompts (contextSnapshotZ)": per_node["contextSnapshotZ"], "state events": size(payload.get("stateEvents") or []), "state proposals": size(payload.get("stateProposals") or []), "summaries": size(payload.get("summaries") or []), "node ids + parentage": per_node["id"] + per_node["parentId"], "per-position state (v2 already)": sum( size(a["narrativeStateAfter"]) for a in payload["actions"] if "narrativeStateAfter" in a ), } def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--turns", type=int, default=120) parser.add_argument("--budget", type=int, default=16384) parser.add_argument("--every", type=int, default=20, help="report the running size every N turns") args = parser.parse_args() client, adv_id = _setup(args.budget) for name, body, kind in ( ("canon.md", m9_fixture.CANON_MD, "canon"), ("reference.md", m9_fixture.REFERENCE_MD, "reference"), ("secret.md", m9_fixture.SECRET_MD, "canon"), ): m9_fixture.upload(client, adv_id, name, body, kind) print(f"budget {args.budget} tokens, {args.turns} turns\n") print(f"{'turns':>6} {'actions':>8} {'bundle B':>12} {'snapshots B':>13} " f"{'B/turn':>9} {'export s':>9} {'import s':>9}") rows = [] sections: dict[int, dict] = {} for turn in range(1, args.turns + 1): ScriptedProvider.replies = [ f"{PROSE} [{turn}]\n" + state_block([{"type": "add_fact", "predicate": "tally", "value": turn * 10, "fact_id": f"tally-{turn}"}]) ] response = client.post(f"/api/adventures/{adv_id}/actions", json={"type": "do", "text": f"press on, {turn}"}) assert response.status_code == 200, response.text[:300] if turn % args.every and turn != args.turns: continue m9_fixture.settle_derived(adv_id) size, payload, export_seconds = _bundle_bytes(client, adv_id) started = time.perf_counter() imported = client.post("/api/adventures/import", json=payload) import_seconds = time.perf_counter() - started assert imported.status_code == 201, imported.text[:300] client.delete(f"/api/adventures/{imported.json()['id']}") snapshots = _snapshot_bytes(payload) plain = _decoded_snapshot_bytes(payload) rows.append((turn, size, snapshots, plain)) sections[turn] = _section_bytes(payload) print(f"{turn:>6} {len(payload['actions']):>8} {size:>12,} " f"{snapshots:>13,} {snapshots // turn:>9,} " f"{export_seconds:>9.3f} {import_seconds:>9.3f}") db_bytes = Path(_DB.name).stat().st_size last_turn, last_size, last_snapshots, last_plain = rows[-1] per_turn = last_snapshots // last_turn cap = limits.MAX_IMPORT_BODY_BYTES print() print(f"database on disk: {db_bytes:,} bytes") print(f"snapshot share of file: {100 * last_snapshots // last_size}%") print(f"stored uncompressed: {last_plain:,} bytes " f"({last_plain / max(last_snapshots, 1):.1f}x the encoded size)") print(f"import body cap: {cap:,} bytes") print(f"turns before the cap: ~{cap // max(per_turn, 1):,} " f"at the marginal rate above") # Growth between the last two samples says whether the per-turn cost has # settled. It should: once the history window fills the budget, a prompt # stops growing with the transcript and the total becomes linear. if len(rows) >= 2: (t0, _, s0, _p0), (t1, _, s1, _p1) = rows[-2], rows[-1] print(f"marginal cost, last {t1 - t0} turns: " f"{(s1 - s0) // max(t1 - t0, 1):,} bytes/turn") print() print("where the bytes are, at the last sample:") last = sections[last_turn] added = sum(v for k, v in last.items() if not k.endswith("(v2 already)")) for name, value in sorted(last.items(), key=lambda kv: -kv[1]): print(f" {name:34} {value:>12,} {100 * value / last_size:5.1f}%") print(f" {'--- everything v3 added':34} {added:>12,} " f"{100 * added / last_size:5.1f}%") without = last_size - added print(f" a v2 file of the same campaign {without:>12,}") print(f" ceiling with v3 additions: ~{int((cap / (last_size / last_turn**2)) ** 0.5):,} turns") print(f" ceiling without them: ~{int((cap / (without / last_turn**2)) ** 0.5):,} turns") app.dependency_overrides.clear() return 0 if __name__ == "__main__": try: raise SystemExit(main()) finally: try: os.unlink(_DB.name) except OSError: pass