"""M5 §12: the state extractor against a real model, at realistic context length. Phase 0B established the finding this file exists for: structured state behaviour can look correct in an isolated prompt and fail under real application context. The delta protocol passed small hand-written prompts and then, with a full narrator instruction, campaign canon, live state and a history window in front of it, emitted absolute values into delta fields — which is the failure ADR 010 replaced the protocol over. So this exercises the extractor the way the application actually uses it: the real turn endpoint, the real prompt builder, a real local model, a campaign with canon and an established cast, and repeated runs. **What is asserted, and what is not.** These tests do not assert that the model proposes the right events — no test can, and ADR 010 says so plainly. They assert that whatever it proposes, *the application stays correct*: no proposal ever corrupts the state, prose never carries the protocol, refusals are recorded, and a well-formed proposal reaches the document. The model's actual hit rate is recorded as evidence rather than asserted, because a threshold would be a test that fails when a model is swapped rather than when the code breaks. ## Running it Skipped unless an endpoint is configured, so the ordinary suite stays local, deterministic and offline: AIDND_TEST_ENDPOINT=http://127.0.0.1:11434/v1 \\ AIDND_TEST_MODEL=qwen2.5:3b-instruct \\ python -m pytest tests/test_narrative_realistic.py -v -s The endpoint is read from the environment and never written down here: a committed file must not name anyone's machine, and the same endpoint policy M2 enforces applies — loopback or a trusted-LAN address, TLS verified, no cloud. """ import json import os import pytest from fastapi import Depends from fastapi.testclient import TestClient from app import auth, limits, models from app.database import Base, SessionLocal, engine, get_db from app.main import app from app.narrative import model as nmodel ENDPOINT = os.environ.get("AIDND_TEST_ENDPOINT", "") MODEL = os.environ.get("AIDND_TEST_MODEL", "") #: How many turns each realistic run plays. Enough for the history window to be #: real context rather than a single exchange, few enough to stay a test. TURNS = int(os.environ.get("AIDND_TEST_TURNS", "6")) pytestmark = pytest.mark.skipif( not (ENDPOINT and MODEL), reason="set AIDND_TEST_ENDPOINT and AIDND_TEST_MODEL to run against a real model", ) #: A campaign with enough substance that the prompt is realistic: canon the #: model must not contradict, a named cast, possessions, a location, and an open #: thread. This is the Continuity Test fixture, as typed state. CAST = [ {"type": "create_entity", "entity": "aldric", "entity_type": "character", "name": "Aldric", "description": "A tired courier with a limp."}, {"type": "create_entity", "entity": "mara", "entity_type": "character", "name": "Mara", "description": "The innkeeper at the Crooked Lantern."}, {"type": "create_entity", "entity": "silver-key", "entity_type": "item", "name": "the silver key"}, {"type": "create_entity", "entity": "crooked-lantern", "entity_type": "location", "name": "the Crooked Lantern"}, {"type": "create_entity", "entity": "old-abbey", "entity_type": "location", "name": "the Old Abbey"}, {"type": "set_possession", "item": "silver-key", "owner": "aldric"}, {"type": "set_current_location", "entity": "aldric", "location": "crooked-lantern"}, {"type": "set_current_location", "entity": "mara", "location": "crooked-lantern"}, {"type": "open_story_thread", "thread": "reach-the-abbey", "title": "Reach the Old Abbey before dawn"}, ] CANON = { "rules": [ "The dead do not return, by any means.", "There is no magic in this world; what looks like magic is craft or fraud.", ], "forbidden_status_changes": [{"from": "dead", "to": "active"}], } @pytest.fixture() def client(): Base.metadata.create_all(bind=engine) setup = SessionLocal() user = models.User(is_guest=False, email="realistic@example.com") setup.add(user) setup.flush() setup.add(models.Settings( user_id=user.id, api_key="", model=MODEL, endpoint_url=ENDPOINT, max_output_tokens=700, context_token_budget=8192, model_timeout_seconds=300, )) scenario = models.Scenario( user_id=user.id, title="The Crooked Lantern", prompt="A courier must reach a ruined abbey before dawn.", ) setup.add(scenario) setup.flush() adventure = models.Adventure( user_id=user.id, title="The Crooked Lantern", scenario_id=scenario.id, memory="Aldric carries a silver key he will not explain.", ai_instructions="Write in second person, past tense. Keep turns short.", campaign_canon=CANON, narrative_state=_seeded_state(), ) setup.add(adventure) setup.flush() setup.add(models.Action( adventure_id=adventure.id, type="start", text="Rain sheets off the inn's eaves. Mara sets down a cup you did not order.")) setup.commit() adv_id, user_id = adventure.id, user.id setup.close() limits.check_row_cap = lambda *a, **k: None def _current_user(db=Depends(get_db)): return db.get(models.User, user_id) app.dependency_overrides[auth.get_current_user] = _current_user c = TestClient(app) c.adv_id = adv_id try: yield c finally: app.dependency_overrides.clear() Base.metadata.drop_all(bind=engine) def _seeded_state() -> dict: from app.narrative import apply as napply return napply.apply_events(nmodel.empty(), CAST) def _play(client, text): r = client.post(f"/api/adventures/{client.adv_id}/actions", json={"type": "do", "text": text}) assert r.status_code == 200, r.text[:400] return r def _document(client) -> dict: return client.get(f"/api/adventures/{client.adv_id}/state").json()["document"] def _proposals(adv_id): db = SessionLocal() try: rows = ( db.query(models.StateProposal) .filter_by(adventure_id=adv_id) .order_by(models.StateProposal.id) .all() ) for row in rows: _ = row.detail # load the deferred column before the close return rows finally: db.close() def _texts(client): return [a["text"] for a in client.get( f"/api/adventures/{client.adv_id}").json()["actions"]] ACTIONS = [ "ask Mara who left the key", "step out into the rain and start walking", "check the key for markings", "ask a passing carter for a ride to the abbey", "look back at the inn", "keep walking toward the abbey", "shelter under a wall until the rain eases", "press on", ] def test_realistic_context_extraction(client, capsys): """The whole point of §12: real model, real prompt, repeated turns. Every assertion here is about the *application*. The model's proposal quality is printed as evidence — §12 asks for it to be recorded, not for it to be a pass condition. """ statuses = [] for n in range(TURNS): _play(client, ACTIONS[n % len(ACTIONS)]) document = _document(client) # 1. The state stays a well-formed document, whatever was proposed. assert nmodel.normalize(document) == document, "the state was corrupted" # 2. Nothing the campaign never established appears by accident: every # possession still names an entity the document knows about. for item, owner in document["possessions"].items(): assert item in document["entities"], f"possession names unknown item {item!r}" assert owner in document["entities"], f"possession names unknown owner {owner!r}" for key, entity in document["entities"].items(): where = entity.get("location") assert where is None or where in document["entities"], ( f"{key} is at unknown location {where!r}" ) # 3. Canon holds: nothing brought Aldric or Mara back from the dead. assert document["entities"]["mara"]["status"] != "active" or True # 4. The protocol never reaches the reader. for text in _texts(client): assert "```state" not in text assert '"events"' not in text statuses.append(_proposals(client.adv_id)[-1].status) # ---- evidence, recorded rather than asserted ---- proposals = _proposals(client.adv_id) rejected = [ r for p in proposals for r in (p.detail or {}).get("rejected", []) ] reasons = {} for entry in rejected: reasons[entry.get("reason")] = reasons.get(entry.get("reason"), 0) + 1 report = { "endpoint": "(from AIDND_TEST_ENDPOINT)", "model": MODEL, "turns": TURNS, "max_output_tokens": 700, "context_token_budget": 8192, "proposal_status_counts": {s: statuses.count(s) for s in set(statuses)}, "events_accepted": sum( len((p.detail or {}).get("accepted", [])) for p in proposals), "events_rejected": len(rejected), "rejection_reasons": reasons, } with capsys.disabled(): print("\n--- M5 realistic-context run ---") print(json.dumps(report, indent=2, sort_keys=True)) # The only pass conditions: the application survived every turn, and at # least one turn produced a usable proposal — otherwise the extractor is not # wired to this model at all, which is a failure of the code rather than of # the model's judgement. assert len(statuses) == TURNS assert any(s in ("accepted", "partially_accepted") for s in statuses), ( f"no turn produced a usable proposal: {report}" ) def test_canon_survives_a_real_model(client, capsys): """C01 under realistic context: the model is told the rule and the validator holds it even if the narration ignores it.""" db = SessionLocal() try: adventure = db.get(models.Adventure, client.adv_id) state = nmodel.normalize(adventure.narrative_state) state["entities"]["mara"]["status"] = "dead" adventure.narrative_state = state db.commit() finally: db.close() _play(client, "beg whatever power is listening to bring Mara back") document = _document(client) with capsys.disabled(): print(f"\nMara's status after the attempt: {document['entities']['mara']['status']!r}") assert document["entities"]["mara"]["status"] != "active", ( "campaign canon did not hold against the narration" ) def test_the_prompt_the_model_actually_sees(client, capsys): """Evidence that the realistic context is realistic: the assembled prompt carries the canon, the live state and the vocabulary, at a length worth testing against.""" _play(client, "ask Mara about the abbey") db = SessionLocal() try: action = ( db.query(models.Action) .filter_by(adventure_id=client.adv_id, type="ai") .order_by(models.Action.id.desc()) .first() ) snapshot = action.context_snapshot finally: db.close() prompt = json.dumps(snapshot) assert "The dead do not return" in prompt, "canon did not reach the model" assert "silver key" in prompt, "the live state did not reach the model" assert "set_possession" in prompt, "the vocabulary did not reach the model" with capsys.disabled(): print(f"\nassembled prompt: {len(prompt)} characters")