"""M6 section 13: context assembly and derived work against a real model. The M6 equivalent of `test_narrative_realistic.py`, and it exists for the same reason: memory and summary extraction can look correct against a tiny synthetic prompt and behave differently under a full application context — a real narrator instruction, real authoritative state, enough recent story to exercise budgeting, a summary, and several memories. **What is asserted, and what is not.** These tests do not assert that the model writes a good summary or picks the right memory. No test can, and a threshold would fail when a model is swapped rather than when the code breaks. They assert that the *application* stays correct around whatever the model produces: * the prompt stays inside its budget and keeps the reply reserve; * a summary the model generates is anchored to the story it covers; * a failure is recorded rather than swallowed; * nothing from an abandoned line reaches the prompt. Model behaviour is recorded as evidence and printed, not asserted. ## Running it Skipped unless an endpoint is configured, so the ordinary suite stays local, deterministic and offline: AIDND_TEST_ENDPOINT=http://127.0.0.1:11434/v1 \\ AIDND_TEST_MODEL=qwen2.5:3b-instruct \\ AIDND_TEST_EMBED_MODEL=nomic-embed-text \\ python -m pytest tests/test_context_realistic.py -v -s The endpoint is read from the environment and never written down here, and the same endpoint policy the rest of the product enforces applies: loopback or a trusted-LAN address, TLS verified, no cloud. """ import asyncio import json import os import pytest from fastapi import Depends from fastapi.testclient import TestClient from app import auth, limits, memorybank, models, summaries from app.database import Base, SessionLocal, engine, get_db from app.main import app from app.routers import adventures from fakes import ScriptedProvider, state_block ENDPOINT = os.environ.get("AIDND_TEST_ENDPOINT", "") MODEL = os.environ.get("AIDND_TEST_MODEL", "") EMBED_MODEL = os.environ.get("AIDND_TEST_EMBED_MODEL", "nomic-embed-text") pytestmark = pytest.mark.skipif( not (ENDPOINT and MODEL), reason="set AIDND_TEST_ENDPOINT and AIDND_TEST_MODEL to run against a real model", ) CANON = { "rules": ["The Crooked Lantern is the only inn in the valley."], "forbidden": ["No character may use magic."], } @pytest.fixture() def client(monkeypatch): Base.metadata.create_all(bind=engine) memorybank._vector_cache.clear() setup = SessionLocal() user = models.User(is_guest=False, email="m6live@example.com") setup.add(user) setup.flush() setup.add(models.Settings( user_id=user.id, api_key="enc:dummy", endpoint_url=ENDPOINT, model=MODEL, summary_model=MODEL, embedding_model=EMBED_MODEL, context_token_budget=8192, max_output_tokens=700, memory_top_k=4, model_timeout_seconds=600, )) adventure = models.Adventure( user_id=user.id, title="The Crooked Lantern", memory_bank_enabled=True, auto_summarize=True, campaign_canon=CANON, ) setup.add(adventure) setup.flush() setup.add(models.Action(adventure_id=adventure.id, type="start", text="Rain hammers the road outside the Crooked Lantern.")) setup.commit() adv_id, user_id = adventure.id, user.id setup.close() monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) test_client = TestClient(app) test_client.adv_id = adv_id test_client.user_id = user_id try: yield test_client finally: app.dependency_overrides.clear() memorybank._vector_cache.clear() Base.metadata.drop_all(bind=engine) def play_scripted(client, text, prose, events=None): """A turn with a known outcome, so the fixture is deterministic.""" real = adventures.turns.OpenAICompatibleProvider adventures.turns.OpenAICompatibleProvider = ScriptedProvider ScriptedProvider.replies = [f"{prose}\n{state_block(events or [])}"] try: r = client.post(f"/api/adventures/{client.adv_id}/actions", json={"type": "do", "text": text}) assert r.status_code == 200, r.text[:300] finally: adventures.turns.OpenAICompatibleProvider = real def context(client) -> dict: r = client.get(f"/api/adventures/{client.adv_id}/context") assert r.status_code == 200, r.text[:300] return r.json() def test_a_real_summary_is_generated_and_anchored(client): """The summariser runs against the real model, and what it writes is anchored to the story it read rather than to a column.""" play_scripted(client, "step inside", "Aldric shakes the rain from his coat.", [ {"type": "create_entity", "entity": "aldric", "entity_type": "character", "name": "Aldric"}, {"type": "create_entity", "entity": "mara", "entity_type": "character", "name": "Mara"}, ]) for i in range(18): play_scripted(client, f"talk on {i}", f"Mara pours another measure and tells him about the road north. " f"The lantern gutters. [{i}]") asyncio.run(memorybank.run_post_turn(client.adv_id)) with SessionLocal() as db: adventure = db.get(models.Adventure, client.adv_id) rows = summaries.all_for(db, adventure) eligible = summaries.current(db, adventure) status = {r["kind"]: r["status"] for r in __import__("app.derived", fromlist=["report"]).report(db, client.adv_id)} print(json.dumps({ "model": MODEL, "embedding_model": EMBED_MODEL, "summaries_written": len(rows), "derived_status": status, "summary_preview": (eligible.text[:300] if eligible else None), }, indent=2, sort_keys=True)) assert status.get("summary") == "ok", f"the summariser failed: {status}" assert rows, "no summary was written" assert eligible is not None # Anchored, not floating: it names the stretch of story it covers. assert eligible.depth is not None assert eligible.branch_id is not None assert eligible.model_name == MODEL # And it reaches the prompt. assert eligible.text[:40] in "\n".join(s["text"] for s in context(client)["sections"]) def test_the_prompt_stays_bounded_and_reserves_the_reply_under_real_context(client): """Budgeting, measured on a realistic prompt rather than a synthetic one.""" play_scripted(client, "step inside", "Aldric shakes the rain from his coat.", [ {"type": "create_entity", "entity": "aldric", "entity_type": "character", "name": "Aldric"}, ]) for i in range(40): play_scripted(client, f"on {i}", f"[{i}] " + "The lantern swings and the rain keeps on. " * 20) asyncio.run(memorybank.run_post_turn(client.adv_id)) report = context(client) print(json.dumps({ "model": MODEL, "budget": report["tokens"]["budget"], "input_tokens": report["tokens"]["total"], "output_reserve": report["tokens"]["output_reserve"], "protected": report["tokens"]["protected"], "available_for_history": report["tokens"]["available_for_history"], "actions_included": report["history"]["included"], "actions_total": report["history"]["total"], "memories_used": len(report["memories"]["used"]) if report["memories"] else 0, }, indent=2, sort_keys=True)) assert report["tokens"]["total"] <= report["tokens"]["budget"] assert report["tokens"]["total"] + 700 <= report["tokens"]["budget"], ( "the real prompt left no room for the configured reply" ) assert report["history"]["included"] < report["history"]["total"], ( "the whole transcript was sent" ) def test_a_real_turn_still_generates_with_memory_and_summary_present(client): """The end-to-end shape: a real narrator turn on a campaign that has a generated summary, retrieved memories and authoritative state.""" play_scripted(client, "step inside", "Aldric shakes the rain from his coat.", [ {"type": "create_entity", "entity": "aldric", "entity_type": "character", "name": "Aldric"}, ]) for i in range(18): play_scripted(client, f"talk {i}", f"They talk of the road north while the fire burns down. [{i}]") asyncio.run(memorybank.run_post_turn(client.adv_id)) # A real turn, through the real provider. r = client.post(f"/api/adventures/{client.adv_id}/actions", json={"type": "do", "text": "ask Mara what lies north"}) assert r.status_code == 200, r.text[:300] assert '"error"' not in r.text, r.text[:400] with SessionLocal() as db: action = (db.query(models.Action) .filter_by(adventure_id=client.adv_id, type="ai") .order_by(models.Action.id.desc()).first()) snapshot = action.context_snapshot text = action.text labels = [s["label"] for s in snapshot["sections"]] print(json.dumps({ "model": MODEL, "sections": labels, "input_tokens": snapshot["tokens"]["total"], "output_reserve": snapshot["tokens"]["output_reserve"], "reply_chars": len(text), }, indent=2, sort_keys=True)) assert "narrator" in labels assert "history" in labels # The reply is a story, not protocol. assert "```state" not in text assert '"events"' not in text