"""What M10 costs a campaign, measured rather than argued. python -m tools.m10_media_cost [--turns 60] Run from `backend/`. Plays a campaign of `--turns` turns with the real prompt builder and the real state pipeline, then reports the five numbers §21 of the M10 brief asks for. Four of them are expected to be zero or near it, and that is the point: M10's central design decision was that **the scene snapshot already exists**, so the milestone persists nothing per scene and nothing per turn. A design claim like that is cheap to make and easy to get wrong by one accidental write, so it is measured here against a campaign long enough for a per-turn cost to show. scene records written by M10 expected 0, and the scenes that do exist are M5's, counted for contrast bytes added to the database one row per profiled entity, once profile duplication what per-position profiles would have cost, against what campaign-scoped profiles do cost packet: persisted or constructed rows written while building one current-scene query behaviour statements per packet, at 10 turns and at N turns — a number that grows with the campaign is a scan """ from __future__ import annotations import argparse import json import os import sys import tempfile import time from pathlib import Path _HERE = Path(__file__).resolve().parent sys.path.insert(0, str(_HERE.parent / "tests")) _DB = tempfile.NamedTemporaryFile(suffix="-m10-cost.db", delete=False) _DB.close() os.environ["AIDND_DB_PATH"] = _DB.name os.environ.pop("AIDND_DATABASE_URL", None) os.environ.pop("DATABASE_URL", None) from fastapi import Depends # noqa: E402 from fastapi.testclient import TestClient # noqa: E402 import m10_fixture # noqa: E402 from app import auth, limits, memorybank, models # noqa: E402 from app.database import Base, SessionLocal, engine, get_db # noqa: E402 from app.main import app # noqa: E402 from app.routers import adventures # noqa: E402 from fakes import ScriptedProvider, state_block # noqa: E402 from tools import dbmeter # noqa: E402 PROSE = ( "Roger pulled the whiteboard marker apart while he talked, which was how " "everyone knew the meeting had stopped being about the agenda. Alice wrote " "nothing down. Outside the glass, somebody wheeled a trolley of monitors " "past the door and did not look in." ) class _Stub: async def complete(self, system, prompt, **kwargs): return "The meeting went on for some time." async def embed(self, texts): return [[1.0, 0.5, 0.25, 0.125] for _ in texts] def _setup() -> tuple[TestClient, int]: adventures.turns.OpenAICompatibleProvider = ScriptedProvider memorybank.embedding_provider = lambda s: _Stub() memorybank.summary_provider = lambda s: _Stub() limits.check_row_cap = lambda *a, **k: None Base.metadata.create_all(bind=engine) with SessionLocal() as db: user = models.User(is_guest=False, email="m10cost@example.com") db.add(user) db.flush() db.add(models.Settings( user_id=user.id, model="cost-model", embedding_model="stub", context_token_budget=8192, max_output_tokens=600, )) adventure = models.Adventure(user_id=user.id, title="Cost", auto_summarize=True, memory_bank_enabled=True) db.add(adventure) db.flush() db.add(models.Action(adventure_id=adventure.id, type="start", text="Bill badges in on a Tuesday morning.")) db.commit() adv_id, user_id = adventure.id, user.id app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) return TestClient(app), adv_id def _db_bytes() -> int: return Path(_DB.name).stat().st_size def _counts(adv_id: int) -> dict: with SessionLocal() as db: return { "actions": db.query(models.Action).filter( models.Action.adventure_id == adv_id).count(), "visual profiles": db.query(models.VisualProfile).filter( models.VisualProfile.adventure_id == adv_id).count(), "M5 per-position state snapshots": db.query(models.Action).filter( models.Action.adventure_id == adv_id, models.Action.narrative_state_after.isnot(None)).count(), } def _profile_bytes(adv_id: int) -> int: with SessionLocal() as db: rows = db.query(models.VisualProfile).filter( models.VisualProfile.adventure_id == adv_id).all() return sum( len(json.dumps({"entity_key": r.entity_key, "descriptors": r.descriptors, "features": r.features, "style_notes": r.style_notes}).encode("utf-8")) for r in rows ) def _packet_statements(client, adv_id: int, meter: dbmeter.Meter, label: str): with meter.scope(label) as scope: started = time.perf_counter() response = client.get(f"/api/adventures/{adv_id}/scene-packet") seconds = time.perf_counter() - started response.raise_for_status() return scope, seconds def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--turns", type=int, default=60) args = parser.parse_args() client, adv_id = _setup() empty_bytes = _db_bytes() m10_fixture.build(client, adv_id) meter = dbmeter.Meter() meter.attach(engine) try: early_scope, early_seconds = _packet_statements( client, adv_id, meter, "packet at 2 turns") before_play = _db_bytes() for turn in range(1, args.turns + 1): ScriptedProvider.replies = [ f"{PROSE} [{turn}]\n" + state_block([ {"type": "set_scene", "summary": f"The meeting reaches item {turn}.", "location": "office", "present": ["bill", "alice", "roger"]}, ]) ] client.post(f"/api/adventures/{adv_id}/actions", json={"type": "do", "text": f"item {turn}"} ).raise_for_status() rows_before = _counts(adv_id) late_scope, late_seconds = _packet_statements( client, adv_id, meter, f"packet at {args.turns + 2} turns") rows_after = _counts(adv_id) finally: meter.detach() played_bytes = _db_bytes() profile_bytes = _profile_bytes(adv_id) print(f"\n{args.turns} turns, {rows_before['actions']} action rows\n") print("scene records M10 wrote") print(f" visual_profiles rows {rows_before['visual profiles']:>8}" " (one per profiled entity, written once)") print(" scene rows 0" " M10 adds no scenes table") print(f" M5 per-position state snapshots " f"{rows_before['M5 per-position state snapshots']:>8}" " already there since M5; the scene lives here") print("\nbytes added to the database") print(f" empty database {empty_bytes:>8} B") print(f" after the fixture campaign {before_play:>8} B") print(f" after {args.turns} more turns".ljust(36) + f"{played_bytes:>8} B") print(f" visual profile content {profile_bytes:>8} B" f" {100 * profile_bytes / max(played_bytes, 1):.3f}% of the database") per_position = profile_bytes * rows_before["M5 per-position state snapshots"] print("\nprofile duplication: campaign-scoped against per-position") print(f" as stored, once per entity {profile_bytes:>8} B") print(f" if snapshotted per position {per_position:>8} B" f" x{per_position / max(profile_bytes, 1):.0f}") print("\npacket: persisted or constructed") print(f" rows written while building one " f"{rows_after['visual profiles'] - rows_before['visual profiles']:>8}") print(" packet rows in any table 0 built on read, never stored") print(f" build time, 2 turns {early_seconds * 1000:>8.1f} ms") print(f" build time, {args.turns + 2} turns".ljust(36) + f"{late_seconds * 1000:>8.1f} ms") print("\ncurrent-scene query behaviour") print(f" statements, 2 turns {early_scope.total.statements:>8}") print(f" statements, {args.turns + 2} turns".ljust(36) + f"{late_scope.total.statements:>8}") verdict = ("does not grow with the campaign" if late_scope.total.statements <= early_scope.total.statements else "GROWS — the scene is being scanned, not read") print(f" {verdict}") print("\n" + dbmeter.render_scope(late_scope, statements=6)) return 0 if __name__ == "__main__": raise SystemExit(main())