"""M11: the multi-character identity diagnostic (post-M8 finding D). python -m tools.m11_identity # against a real model python -m tools.m11_identity --scripted # harness self-test, no model Run from `backend/`. Reads `AIDND_TEST_ENDPOINT` and `AIDND_TEST_MODEL`. ## What this is for A hands-on session against accepted M8 put four people in one scene — a protagonist and three others — and later narration treated one of them as two different people. The campaign was a disposable database and was destroyed, so **the root cause was never established and cannot be**. What M11 owes the finding is not a fix for an unknown defect; it is a diagnostic that can tell the candidate causes apart the *next* time, and evidence about whether the product does the things it can be blamed for. `BUILD-MILESTONES.md` names four candidate causes and asks for a classification: STATE DEFECT the state itself is wrong or ambiguous CONTEXT ASSEMBLY DEFECT the state is right, the prompt is not DERIVED MEMORY-SUMMARY DEFECT a summary or memory carried the error in MODEL FAILURE WITH CORRECT CONTEXT the prompt was right and the model was not AMBIGUOUS the evidence does not separate them ## How it decides The objective checks are the ones a program can make honestly, and they are made against the **stored prompt snapshot** and the **authoritative state**, before and after every turn: duplicate entity keys the state model refuses these; a breach is a STATE DEFECT shared display names permitted by design, reported by `narrative.model.duplicate_names`; a new one appearing mid-scene is a STATE DEFECT for this scene's purposes protagonist drift the persona's entity key changing, or the protagonist disappearing from `present` state/context disagreement a name in the prompt's state block that the document does not have, or vice versa derived contamination the same name appearing under two keys inside a summary or memory that reached the prompt Prose-level judgements — did the narrator misattribute this line of dialogue, did it have a character refer to itself as someone else — are **not** graded automatically. A regex cannot read dialogue, and a diagnostic that pretended to would produce exactly the confident wrong answer this finding is about. Every turn's narration is written out for a person to read, next to the prompt that produced it, and the tool's verdict says plainly when the objective checks are clean and the question is therefore about the prose. ## What it preserves On any signal, everything the finding lists is written to the run directory: the pre-turn state, the exact stored prompt snapshot, the narration, the history, summaries, memories, imported knowledge and the model settings. The campaign is also exported as an M9 bundle, so the whole failing case is portable and can be replayed on another machine. """ from __future__ import annotations import argparse import json import os import sys import tempfile from datetime import datetime from pathlib import Path _HERE = Path(__file__).resolve().parent sys.path.insert(0, str(_HERE.parent / "tests")) _DB = tempfile.NamedTemporaryFile(suffix="-m11-identity.db", delete=False) _DB.close() os.environ["AIDND_DB_PATH"] = _DB.name os.environ.pop("AIDND_DATABASE_URL", None) os.environ.pop("DATABASE_URL", None) from fastapi import Depends # noqa: E402 from fastapi.testclient import TestClient # noqa: E402 from sqlalchemy.orm import undefer # noqa: E402 from app import auth, limits, memorybank, models # noqa: E402 from app.database import Base, SessionLocal, engine, get_db # noqa: E402 from app.main import app # noqa: E402 from app.narrative import model as nmodel # noqa: E402 from app.routers import adventures # noqa: E402 ENDPOINT = os.environ.get("AIDND_TEST_ENDPOINT", "") MODEL = os.environ.get("AIDND_TEST_MODEL", "") #: v1.1 release Gate 7 asks for this diagnostic with **memory on**. It shipped #: with no embedding model and the bank switched off, so a release run of it #: would have reported a clean identity result without memory ever taking part. #: Empty keeps the old behaviour, which is what `--scripted` wants. EMBED_MODEL = os.environ.get("AIDND_TEST_EMBED_MODEL", "") #: The cast the finding describes: a protagonist and three others, all on stage. CAST = [ ("bill", "character", "Bill"), ("alice", "character", "Alice"), ("roger", "character", "Roger"), ("john", "character", "John"), ("office", "location", "The meeting room"), ] PROTAGONIST = "bill" #: The sequence, built to stress exactly what the finding names. Each entry is #: (what the reader writes, what it is meant to stress). BEATS = [ ("Alice asks Roger what he thinks of the proposal.", "dialogue attribution between two non-protagonists"), ("I ask her to say that again.", "pronoun reference to the last speaker"), ("John comes in and sits down without saying anything.", "entrance mid-scene"), ("I ask the newcomer what he wants.", "reference by role rather than by name"), ("Alice tells John what Roger just said.", "one character speaking about another"), ("Roger leaves the room.", "exit mid-scene"), ("I ask Alice whether she agrees with the man who just left.", "reference to an absent character by role"), ("Alice and John talk about me as if I were not here.", "the protagonist referred to in the third person"), ("I remind them all who called this meeting.", "protagonist self-reference"), ("Alice says one last thing to Roger.", "reference to an absent character by name"), ] def _setup(scripted: bool): if scripted: from fakes import ScriptedProvider adventures.turns.OpenAICompatibleProvider = ScriptedProvider limits.check_row_cap = lambda *a, **k: None Base.metadata.create_all(bind=engine) with SessionLocal() as db: user = models.User(is_guest=False, email="identity@example.com") db.add(user) db.flush() db.add(models.Settings( user_id=user.id, model=MODEL or "scripted", endpoint_url=ENDPOINT or "http://127.0.0.1:11434/v1", embedding_model=EMBED_MODEL, context_token_budget=16384, max_output_tokens=500, model_timeout_seconds=300, )) db.commit() user_id = user.id app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) return TestClient(app) def _campaign(client) -> int: created = client.post("/api/adventures", json={ "title": "Multi-Character Identity Test", "opening": ( "A Tuesday morning meeting. Bill has called it. Alice and Roger are " "already at the table; John has not arrived yet." ), "canon_rules": [ "Bill, Alice, Roger and John are four different people.", "Bill is the protagonist and the one the reader plays.", ], "persona_name": "Bill", "narration_length": "brief", }) created.raise_for_status() adv = created.json()["id"] # Memory and summaries are per-campaign switches defaulting to off. Gate 7 # asks for this diagnostic with memory on, and the ten beats below write # twenty actions — past `MEMORY_START` — so the bank has something to do. client.patch(f"/api/adventures/{adv}", json={"memory_bank_enabled": True, "auto_summarize": True} ).raise_for_status() answer = client.post(f"/api/adventures/{adv}/state/corrections", json={ "events": [ {"type": "create_entity", "entity": key, "entity_type": kind, "name": name} for key, kind, name in CAST ] + [ {"type": "set_scene", "summary": "Bill, Alice and Roger at the table; John not yet arrived.", "location": "office", "present": ["bill", "alice", "roger"]}, ], "note": "the cast, before anything is narrated", }) answer.raise_for_status() # The first run of this diagnostic set a scene whose location entity did not # exist. The event was correctly refused and — before M11 fixed it — the 201 # said nothing, so the whole run happened with an empty scene and no list of # who was in the room. That is a fixture defect that would have been read as # a model failure, which is exactly what this diagnostic exists not to do. refused = answer.json().get("refused") or [] if refused: raise SystemExit(f"the fixture itself was refused: {refused}") scene = client.get(f"/api/adventures/{adv}/state").json()["document"].get("scene") if not (scene or {}).get("present"): raise SystemExit("the fixture did not establish a scene; the run would be void") return adv def _state(adv: int) -> dict: with SessionLocal() as db: adventure = db.get(models.Adventure, adv) from app import narrative return narrative.store.current(adventure) def _last_ai(adv: int): with SessionLocal() as db: return ( db.query(models.Action) .filter(models.Action.adventure_id == adv, models.Action.type == "ai") .options(undefer(models.Action.context_snapshot)) .order_by(models.Action.id.desc()).first() ) def _signals(before: dict, after: dict, snapshot: dict, narration: str) -> list[dict]: """Every objective thing that is wrong, as a list. Empty means clean.""" found: list[dict] = [] entities = (after.get("entities") or {}) # 1. Duplicate keys are structurally impossible; a breach is a state defect. if len(entities) != len({k.lower() for k in entities}): found.append({"kind": "duplicate_entity_key", "class": "STATE DEFECT", "detail": sorted(entities)}) # 2. A shared display name appearing that was not there before. was = nmodel.duplicate_names(before) now = nmodel.duplicate_names(after) new_clashes = {n: keys for n, keys in now.items() if n not in was} if new_clashes: found.append({"kind": "shared_display_name", "class": "STATE DEFECT", "detail": new_clashes}) # 3. A new character invented mid-scene with a name the cast already has. known = {k for k, _, _ in CAST} invented = { key: value.get("name") for key, value in entities.items() if key not in known and value.get("type") == "character" } cast_names = {name.lower() for _, _, name in CAST} shadowing = {k: n for k, n in invented.items() if str(n or "").strip().lower() in cast_names} if shadowing: found.append({"kind": "duplicate_character_creation", "class": "STATE DEFECT", "detail": shadowing}) # 4. Protagonist drift: the persona's entity gone, or dropped from the scene # while the narration still speaks in second person. scene = after.get("scene") or {} present = scene.get("present") or [] if PROTAGONIST not in entities: found.append({"kind": "protagonist_missing", "class": "STATE DEFECT", "detail": PROTAGONIST}) elif present and PROTAGONIST not in present and " you " in f" {narration.lower()} ": found.append({"kind": "protagonist_dropped_from_scene", "class": "STATE DEFECT", "detail": present}) # 5. State/context disagreement: a name the prompt's state section shows that # the document does not have. sections = {s["label"]: s["text"] for s in (snapshot.get("sections") or [])} state_text = sections.get("narrative_state", "") or sections.get("world_state", "") document_names = {str(v.get("name") or "").strip() for v in entities.values() if v.get("name")} for _, _, name in CAST: in_prompt = name in state_text in_document = name in document_names if in_prompt != in_document: found.append({"kind": "state_context_disagreement", "class": "CONTEXT ASSEMBLY DEFECT", "detail": {"name": name, "in_prompt": in_prompt, "in_document": in_document}}) # 6. Derived contamination: a summary or memory in the prompt that names one # cast member as two people. derived_text = " ".join( sections.get(label, "") for label in ("story_summary", "memories") ) for _, _, name in CAST: if derived_text.count(f"{name} and {name}") or derived_text.count( f"the other {name}"): found.append({"kind": "derived_identity_contamination", "class": "DERIVED MEMORY-SUMMARY DEFECT", "detail": name}) return found def _preserve(root: Path, client, adv: int, index: int, payload: dict) -> Path: """Everything the finding says to keep, for one turn.""" directory = root / f"turn-{index:02d}" directory.mkdir(parents=True, exist_ok=True) (directory / "evidence.json").write_text(json.dumps(payload, indent=2, default=str)) for name, url in ( ("state.json", f"/api/adventures/{adv}/state"), ("context.json", f"/api/adventures/{adv}/context"), ("memories.json", f"/api/adventures/{adv}/memories"), ("knowledge.json", f"/api/adventures/{adv}/knowledge"), ("settings.json", "/api/settings"), ("bundle.json", f"/api/adventures/{adv}/export"), ): response = client.get(url) if response.status_code == 200: (directory / name).write_text(json.dumps(response.json(), indent=2)) return directory def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--scripted", action="store_true", help="run the harness against a scripted narrator") parser.add_argument("--inject", action="store_true", help=("scripted mode only: have the narrator commit the " "exact confusion the finding describes, to prove " "the detectors fire. A diagnostic that has only " "ever returned 'clean' has not been tested.")) parser.add_argument("--out", default="", help="where to preserve evidence (default: a temp dir)") args = parser.parse_args() if not args.scripted and not (ENDPOINT and MODEL): print("set AIDND_TEST_ENDPOINT and AIDND_TEST_MODEL, or pass --scripted") return 2 root = Path(args.out or tempfile.mkdtemp(prefix="m11-identity-")) root.mkdir(parents=True, exist_ok=True) client = _setup(args.scripted) adv = _campaign(client) print(f"\nMulti-Character Identity Test — {'scripted' if args.scripted else MODEL}") print(f"evidence: {root}\n") print(f"{'#':>3} {'stresses':40} {'signals':>7} narration") print("-" * 100) all_signals: list[dict] = [] for index, (text, stresses) in enumerate(BEATS, start=1): before = _state(adv) if args.scripted: from fakes import ScriptedProvider, state_block events = [] if args.inject and index == 5: # The finding's own failure mode: a second Alice, created # because the narrator lost track of the first one. events = [{"type": "create_entity", "entity": "alice_2", "entity_type": "character", "name": "Alice"}] ScriptedProvider.replies = [ f"Alice answers, and Roger nods.\n{state_block(events)}" ] response = client.post(f"/api/adventures/{adv}/actions", json={"type": "do", "text": text}) if response.status_code != 200: print(f"{index:>3} {stresses:40} {'ERROR':>7} {response.text[:60]}") continue action = _last_ai(adv) narration = action.text if action else "" snapshot = action.context_snapshot if action else {} after = _state(adv) signals = _signals(before, after, snapshot, narration) all_signals += [dict(s, turn=index) for s in signals] first_line = " ".join(narration.split())[:56] print(f"{index:>3} {stresses:40} {len(signals):>7} {first_line}") if signals: where = _preserve(root, client, adv, index, { "beat": text, "stresses": stresses, "signals": signals, "state_before": before, "state_after": after, "narration": narration, "prompt_snapshot": snapshot, }) for signal in signals: print(f" -> {signal['class']}: {signal['kind']} {signal['detail']}") print(f" -> preserved in {where}") # Always preserve the final campaign, signals or not: a clean run is # evidence too, and the bundle makes it replayable. _preserve(root, client, adv, 99, {"note": "final state", "signals": all_signals}) print("\n" + "=" * 100) classes = sorted({s["class"] for s in all_signals}) if not all_signals: print("VERDICT: no objective identity defect detected.") print(" The state kept four distinct people, no name was shared, the") print(" protagonist did not drift, the prompt agreed with the document,") print(" and no summary or memory carried a confusion into the prompt.") print(" Whether the *prose* misattributed anything is a question for a") print(" person reading the narration beside its prompt — both are in") print(f" {root}. If the prose is wrong and these checks are clean, the") print(" classification is MODEL FAILURE WITH CORRECT CONTEXT.") else: print(f"VERDICT: {len(all_signals)} signal(s): {', '.join(classes)}") for signal in all_signals: print(f" turn {signal['turn']:>2} {signal['class']:34} {signal['kind']}") print("=" * 100) return 0 if __name__ == "__main__": raise SystemExit(main())