"""v1.1 WP-B.2: the memory summariser, after the rejected B2.4 prompt experiment. B2.4 tried a memory prompt instructing the model to keep named facts and objects. Measured against the reference model, it did not correct the creation failure it was for, and it was not shipped (`V1.1-WP-B2-REPORT.md` §T). The shipped prompt is v1.0.0's. This file keeps two kinds of test apart. **Acceptance tests** gate the tree: - the shipped memory prompt is exactly v1.0.0's, so the experiment is gone; - every fidelity fixture reaches the summariser whole, through the application's own prompt assembly; - a long memory is stored as the model wrote it, never cut; - the memory the attempt-2 block should have produced ranks first under B2.1. **Diagnostic-measurement tests** check only that `tools/memory_fidelity.py` measures correctly: fact retention, attribution, invention, word count, a leading "Memory:", second person and promise retention, on hand-written memories whose answers are known. What a real model scores on those measurements is nondeterministic, is taken with inference, and is reported. It is never a gate here. python -m pytest tests/test_v11_b2_summarizer_fidelity.py -v """ import asyncio import re import subprocess import pytest from app import memorybank, models, tree from app.database import Base, SessionLocal, engine from tools import memory_diagnostic as md from tools import memory_fidelity as mf # ==================================================================== acceptance def test_the_shipped_memory_prompt_is_v1_0_0s(): """The B2.4 experiment is reverted: production sends the prompt v1.0.0 and WP-B.1 shipped, unchanged.""" try: source = subprocess.run(["git", "show", "beb17ad:backend/app/memorybank.py"], capture_output=True, text=True, check=True).stdout except (OSError, subprocess.CalledProcessError): pytest.skip("git history not available") block = re.search(r"^MEMORY_SYSTEM_PROMPT = \((.*?)^\)$", source, re.S | re.M).group(1) shipped = eval(f"({block})", {"MEMORY_MAX_WORDS": 50}) # noqa: S307 - our own source assert memorybank.MEMORY_SYSTEM_PROMPT == shipped assert memorybank.MEMORY_MAX_WORDS == 50 def test_the_rejected_experiment_is_not_what_ships(): assert mf.B24_EXPERIMENT_PROMPT != memorybank.MEMORY_SYSTEM_PROMPT assert "Keep each fact with the person it belongs to" not in memorybank.MEMORY_SYSTEM_PROMPT @pytest.mark.parametrize("fixture", mf.FIXTURES, ids=lambda f: f.fixture_id) def test_every_fixture_reaches_the_summariser_whole(fixture): """Creation can only fail at the model if the fact was sent. Each fixture fits the excerpt budget, so the whole block is the excerpt.""" user = mf.user_prompt_for(fixture) assert memorybank.count_tokens(fixture.raw) <= memorybank.MEMORY_EXCERPT_TOKENS assert f"Story excerpt:\n\n{fixture.raw}\n\nMemory:" in user assert user.startswith("Cast:\n- " + fixture.protagonist + " — the protagonist.") class Scripted: def __init__(self, reply): self.reply = reply self.calls: list[tuple[str, str]] = [] async def complete(self, system, user, *, temperature=0.3, max_tokens=400): self.calls.append((system, user)) return self.reply @pytest.fixture() def db(): Base.metadata.create_all(bind=engine) session = SessionLocal() try: yield session finally: session.close() Base.metadata.drop_all(bind=engine) def campaign(db, fixture): user = models.User(is_guest=False, email="b2-fidelity@example.com") db.add(user) db.flush() db.add(models.Settings(user_id=user.id, model="m", embedding_model="")) adventure = models.Adventure(user_id=user.id, title="Fidelity", script_state={}, auto_summarize=True, persona_name=fixture.protagonist) db.add(adventure) db.flush() nodes = [] for kind, text in fixture.actions + (("ai", "The story moves on."),): action = models.Action(adventure_id=adventure.id, type=kind, text=text) tree.place_action(db, adventure, action) db.add(action) db.flush() nodes.append(action) db.commit() return adventure, nodes def write_memory(db, adventure, monkeypatch, provider): monkeypatch.setattr(memorybank, "MEMORY_START", 0) monkeypatch.setattr(memorybank, "MAX_MEMORIES_PER_RUN", 1) monkeypatch.setattr(memorybank, "summary_provider", lambda s: provider) settings = db.query(models.Settings).first() asyncio.run(memorybank._create_due_memories(adventure, settings, db)) return db.query(models.Memory).filter_by(adventure_id=adventure.id).all() def test_the_application_sends_the_shipped_prompt_and_the_whole_planting_block(db, monkeypatch): fixture = mf.FIXTURES_BY_ID["regression_attempt_2"] adventure, nodes = campaign(db, fixture) provider = Scripted(fixture.faithful) [memory] = write_memory(db, adventure, monkeypatch, provider) system, user = provider.calls[0] assert system == memorybank.MEMORY_SYSTEM_PROMPT assert "> I watch Mara slip the amber sundial inside the cracked teapot" in user assert (memory.source_start, memory.source_end) == (nodes[0].depth, nodes[5].depth) def test_an_over_long_memory_is_stored_as_written_never_cut(db, monkeypatch): """The word target is an instruction, not a truncation: cutting a memory after the fact can split or drop exactly the fact it was written to keep.""" fixture = mf.FIXTURES_BY_ID["regression_attempt_2"] adventure, _ = campaign(db, fixture) long_reply = fixture.faithful + " " + " ".join(["They advanced cautiously through the dark."] * 12) [memory] = write_memory(db, adventure, monkeypatch, Scripted(long_reply)) assert memory.text == long_reply assert len(memory.text.split()) > 2 * memorybank.MEMORY_MAX_WORDS def test_a_faithful_regression_memory_ranks_first_for_its_question(): """If the summariser keeps the fact, B2.1 finds it: the memory the attempt-2 block should have produced, among the memories its bank really held for that stretch, under production scoring.""" fixture = mf.FIXTURES_BY_ID["regression_attempt_2"] stored, _ = fixture.unfaithful[0] bank = { 1: fixture.faithful, 2: stored, 3: "Aldric, Mara and Edrin advanced through the cold crypt, the silver key heavy in Aldric's hands.", 4: "Aldric told Mara the silver key opens the crypt beneath the Old Abbey.", 5: "Rain kept falling on Westhaven as the travellers walked toward the abbey grounds.", } embed = md.ConceptEmbedder.vector query = {"input": "> I ask Mara quietly where she hid the amber sundial.", "context": "Aldric and Mara in the Crooked Lantern, rain outside."} held = {i: embed(t) for i, t in bank.items()} terms = {i: memorybank.lexical_terms(t) for i, t in bank.items()} rows = memorybank.score_candidates(list(bank), held, terms, embed(query["input"]), embed(query["context"]), sorted(memorybank.lexical_terms(query["input"]))) assert rows[0][1] == 1 assert rows[0][3] > 0 # ======================================================= diagnostic measurements # These prove the measuring instrument. They say nothing about any model. def test_the_fixtures_cover_each_measurement_in_more_than_one_genre(): requirements = {f.requirement for f in mf.FIXTURES} assert {"distinctive object and place", "player-established concrete fact", "promise / commitment", "attribution", "clutter pressure", "no invention", "multiple concrete facts", "the actual failed-run block"} <= requirements assert {"office", "contemporary", "science-fiction-neutral"} <= {f.genre for f in mf.FIXTURES} @pytest.mark.parametrize("fixture", mf.FIXTURES, ids=lambda f: f.fixture_id) def test_the_checker_passes_a_faithful_memory(fixture): result = mf.evaluate(fixture, fixture.faithful) assert result["passed"], result assert not result["over_target"] and not result["memory_prefix"] and not result["second_person"] @pytest.mark.parametrize("fixture, memory, reason", [ (f, memory, reason) for f in mf.FIXTURES for memory, reason in f.unfaithful ], ids=lambda v: v.fixture_id if isinstance(v, mf.Fixture) else None) def test_the_checker_fails_each_failure_shape(fixture, memory, reason): result = mf.evaluate(fixture, memory) assert not result["passed"], result if reason == "not retained": assert not result["retained"] elif reason == "misattributed": assert result["misattributed"] elif reason == "invented": assert result["inventions"] def test_the_checker_reads_the_stored_attempt_2_memory_as_the_real_failure(): fixture = mf.FIXTURES_BY_ID["regression_attempt_2"] stored, _ = fixture.unfaithful[0] result = mf.evaluate(fixture, stored) assert result["retained"] is False and result["words"] == 102 and result["over_target"] @pytest.mark.parametrize("memory, prefix, you", [ ("Memory: Dana promised Marcus the lease by Friday.", True, False), (" memory: Dana promised the lease.", True, False), ("You thanked Marcus and left.", False, True), ("Dana thanked Marcus; your lease is due.", False, True), ("Dana promised Marcus she would bring the signed lease by Friday.", False, False), ]) def test_the_checker_measures_framing(memory, prefix, you): result = mf.evaluate(mf.FIXTURES_BY_ID["promise_contemporary"], memory) assert result["memory_prefix"] is prefix assert result["second_person"] is you def test_the_checker_measures_promise_retention(): fixture = mf.FIXTURES_BY_ID["promise_contemporary"] kept = mf.evaluate(fixture, "Dana promised to bring Marcus the signed lease by Friday.") scenery = mf.evaluate(fixture, "Memory: Dana looked around the empty living room while a dog barked.") assert kept["facts"]["lease by Friday"]["kept"] and kept["passed"] assert not scenery["facts"]["lease by Friday"]["kept"] and scenery["memory_prefix"]