"""Measures what a campaign bundle preserves, omits and rebuilds. python -m tools.m9_portability_report # human-readable python -m tools.m9_portability_report --json # machine-readable Run from `backend/`, with the virtualenv on the path. The script builds the M9 portability fixture in a throwaway database, exports it, imports it into a second throwaway database, and then compares the two campaigns family by family. It exists because the M9 brief asks for the baseline to be **measured** rather than assumed. Running it on the M8 commit produces the inventory M9 started from; running it on the M9 tree produces the one M9 finished with, and the difference between the two files is the milestone's portability claim in a form a reviewer can reproduce rather than take on trust. The comparison is by data family rather than by row count. "12 actions in, 12 actions out" is the check that misses a bundle carrying every turn and none of its state, so each family below reports what a reader could still see afterwards. Nothing here touches the developer's own database: two temporary files are created and removed, and no network call is made — the narrator, the summariser and the embedder are all local fakes. """ from __future__ import annotations import argparse import json import os import sys import tempfile import time from pathlib import Path # The test harness owns the fixture and the fakes. Both live under `tests/`, # which is not a package, so the path is extended rather than imported from. _HERE = Path(__file__).resolve().parent sys.path.insert(0, str(_HERE.parent / "tests")) # `app.database` reads this at import and builds the engine once, exactly as # `tests/conftest.py` explains. It has to be set before the first `app` import. _SOURCE_DB = tempfile.NamedTemporaryFile(suffix="-m9-source.db", delete=False) _SOURCE_DB.close() os.environ["AIDND_DB_PATH"] = _SOURCE_DB.name os.environ.pop("AIDND_DATABASE_URL", None) os.environ.pop("DATABASE_URL", None) from fastapi import Depends # noqa: E402 from fastapi.testclient import TestClient # noqa: E402 import m9_fixture # noqa: E402 from app import auth, limits, memorybank, models # noqa: E402 from app.database import Base, SessionLocal, engine, get_db # noqa: E402 from app.main import app # noqa: E402 from app.routers import adventures # noqa: E402 from fakes import ScriptedProvider # noqa: E402 class _StubDerivedProvider: """Deterministic vectors and prose, so the report needs no model at all. One object serves as both the embedder and the summariser, because the memory pass builds each from the same factory and stubbing only one of them is the M6 finding M6-F3 mistake: the unstubbed factory opens a socket against the default endpoint on every turn. """ _written = 0 async def complete(self, system, prompt, **kwargs): _StubDerivedProvider._written += 1 return ( f"Memory {_StubDerivedProvider._written}: what the story had " f"established by this point." ) async def embed(self, texts): out = [] for text in texts: lowered = text.lower() out.append([ 1.0, 1.0 if "abbey" in lowered or "crypt" in lowered else 0.0, 1.0 if "tavern" in lowered or "lantern" in lowered else 0.0, 1.0 if "rain" in lowered else 0.0, ]) return out def _install_fakes() -> None: adventures.turns.OpenAICompatibleProvider = ScriptedProvider memorybank.embedding_provider = lambda s: _StubDerivedProvider() memorybank.summary_provider = lambda s: _StubDerivedProvider() limits.check_row_cap = lambda *a, **k: None def _new_user_and_campaign(title: str) -> tuple[int, int]: db = SessionLocal() try: user = models.User(is_guest=False, email=f"m9-{title}@example.com") db.add(user) db.flush() db.add(models.Settings( user_id=user.id, model="report-model", embedding_model="stub-embed", context_token_budget=4000, max_output_tokens=400, )) adventure = models.Adventure( user_id=user.id, title=title, campaign_canon=m9_fixture.CAMPAIGN_CANON, ) db.add(adventure) db.flush() db.add(models.Action( adventure_id=adventure.id, type="start", text=m9_fixture.OPENING, )) # A neighbour, so a bundle that reached past its own campaign would # bring back rows this report can see. neighbour = models.Adventure(user_id=user.id, title="Neighbour") db.add(neighbour) db.flush() db.add(models.Action( adventure_id=neighbour.id, type="start", text="A different story.", )) db.commit() return adventure.id, user.id finally: db.close() def _client(user_id: int) -> TestClient: app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) return TestClient(app) # ----------------------------------------------------------------- the families # One entry per data family the M9 brief asks the baseline to classify. Each # `present` function answers "did this survive into the file?" from the bundle # alone, because that is the question the classification is about. def _actions(bundle: dict) -> list[dict]: return [a for a in (bundle.get("actions") or []) if isinstance(a, dict)] def _snapshots(bundle: dict) -> list[dict]: """Every stored prompt in the file, decoded. The export compresses them (`bundle._packed`), so a report that looked for a plain dict would say the evidence was omitted when it is merely encoded — which is the mistake this whole tool exists to avoid making about anything. """ from app import bundle as bundle_module out = [] for action in _actions(bundle): snapshot = ( action.get("contextSnapshot") if isinstance(action.get("contextSnapshot"), dict) else bundle_module._unpacked(action.get("contextSnapshotZ")) ) if isinstance(snapshot, dict): out.append(snapshot) return out FAMILIES: list[tuple[str, str, callable]] = [ ("campaign identity", "title, instructions, persona, canon, the campaign's own settings", lambda b: bool(b.get("title"))), ("transcript", "every accepted player and narrator action, live and superseded", lambda b: bool(_actions(b))), ("branches", "the retained tree, its fork points and its names", lambda b: bool(b.get("branches"))), ("branch disposition", "which lines the story left behind, and where", lambda b: any("supersededAt" in x for x in (b.get("branches") or []))), ("active head", "the branch and depth the campaign is being read at", lambda b: b.get("headDepth") is not None), ("alternate takes", "every attempt at a turn, and which one is the story", lambda b: any(not a.get("live", True) for a in _actions(b))), ("take grouping", "which attempts belong to the same turn across a fork (SP9 parentage)", lambda b: any("parentId" in a for a in _actions(b))), ("save points", "named coordinates, their notes and their positions", lambda b: bool(b.get("checkpoints"))), ("narrative state (current)", "the authoritative document at the exported head", lambda b: b.get("narrativeState") is not None), ("narrative state (per position)", "the snapshot every position restores from", lambda b: any("narrativeStateAfter" in a for a in _actions(b))), ("state events", "the accepted typed events: the audit half of the hybrid", lambda b: bool(b.get("stateEvents"))), ("state proposals", "what the model proposed and what the application did about it", lambda b: bool(b.get("stateProposals"))), ("manual corrections", "state the user asserted, distinguishable from state the story did", lambda b: any(e.get("source") == "manual_correction" for e in (b.get("stateEvents") or []))), ("historical prompt/context", "the exact prompt each turn was given", lambda b: bool(_snapshots(b))), ("retrieval provenance", "which passages a historical turn was shown, and their text", lambda b: any((s.get("knowledge") or {}).get("used") for s in _snapshots(b))), ("per-turn model settings", "the model and generation settings a historical turn ran under", lambda b: any(s.get("settings") for s in _snapshots(b))), ("imported knowledge", "source content, class, lifecycle, visibility and hash", lambda b: bool(b.get("knowledge"))), ("knowledge parser versions", "what produced the chunks the source last had", lambda b: any("parserVersion" in k for k in (b.get("knowledge") or []))), ("summaries", "the generated rolling summaries and the story they cover", lambda b: bool(b.get("summaries"))), ("memories", "long-term memories and the coordinate each hangs off", lambda b: bool(b.get("memories"))), ("memory authority", "whether a memory is accepted story or a heuristic reading of it", lambda b: any("authority" in m for m in (b.get("memories") or []))), ("scene metadata", "the scene section of the authoritative state document", lambda b: isinstance(b.get("narrativeState"), dict) and "scene" in b["narrativeState"]), ("story cards (legacy)", "the inherited lore primitive, which has no v1 browser surface", lambda b: "storyCards" in b), ] #: Families that are deliberately rebuilt rather than carried, with the reason. REBUILDABLE = { "knowledge passages": "a deterministic function of the source content", "lexical (FTS) index": "rebuilt from the passages on import", "knowledge embeddings": "belong to the importing machine's embedding model", "memory embeddings": "the same, for the memory bank", "branch lineage cache": "computed from parent plus fork depth", "derived status": "describes the last run of a background pass, not the story", } def measure(json_out: bool) -> dict: _install_fakes() Base.metadata.create_all(bind=engine) adv_id, user_id = _new_user_and_campaign("M9 Portability Fixture") client = _client(user_id) built = time.perf_counter() source = m9_fixture.build(client, adv_id) build_seconds = time.perf_counter() - built started = time.perf_counter() response = client.get(f"/api/adventures/{adv_id}/export") export_seconds = time.perf_counter() - started response.raise_for_status() bundle = response.json() encoded = json.dumps(bundle, ensure_ascii=False).encode("utf-8") started = time.perf_counter() imported = client.post("/api/adventures/import", json=bundle) import_seconds = time.perf_counter() - started import_status = imported.status_code copy = ( m9_fixture.snapshot_of(client, imported.json()["id"]) if import_status == 201 else None ) source_db_bytes = Path(_SOURCE_DB.name).stat().st_size app.dependency_overrides.clear() families = [ {"family": name, "what": what, "verdict": "PRESERVED" if present(bundle) else "OMITTED"} for name, what, present in FAMILIES ] report = { "format": bundle.get("format"), "families": families, "rebuildable": REBUILDABLE, "sizes": { "source_database_bytes": source_db_bytes, "bundle_bytes": len(encoded), "bundle_actions": len(_actions(bundle)), "bundle_keys": sorted(bundle), }, "timings_seconds": { "fixture_build": round(build_seconds, 3), "export": round(export_seconds, 3), "import": round(import_seconds, 3), }, "round_trip": { "import_status": import_status, "agrees": _agreement(source, copy) if copy else None, }, } return report def _agreement(source: dict, copy: dict) -> dict: """Which of the reader-visible families match between original and copy.""" keys = ("title", "canon_rules", "transcript", "branch_count", "checkpoints", "knowledge", "state", "state_events", "memories", "summaries", "can_undo", "can_redo") return {key: source.get(key) == copy.get(key) for key in keys} def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--json", action="store_true", help="print the report as JSON") args = parser.parse_args() try: report = measure(args.json) finally: for path in (_SOURCE_DB.name,): try: os.unlink(path) except OSError: pass if args.json: print(json.dumps(report, indent=2, sort_keys=True)) return 0 print(f"bundle format: {report['format']}") print(f"bundle size: {report['sizes']['bundle_bytes']:,} bytes " f"across {report['sizes']['bundle_actions']} actions") print(f"source db: {report['sizes']['source_database_bytes']:,} bytes") print(f"timings: {report['timings_seconds']}") print() width = max(len(name) for name, _, _ in FAMILIES) for row in report["families"]: print(f" {row['verdict']:<10} {row['family']:<{width}} {row['what']}") print() print(" DERIVED/REBUILDABLE (deliberately not carried)") for name, why in REBUILDABLE.items(): print(f" {name:<24} {why}") print() print(f"round trip: HTTP {report['round_trip']['import_status']}") for key, agreed in (report["round_trip"]["agrees"] or {}).items(): print(f" {'same' if agreed else 'DIFFERS':<8} {key}") return 0 if __name__ == "__main__": raise SystemExit(main())