"""v1.1 WP-B.1: the memory-retention diagnostic, deterministically. B.1 changes no memory behaviour. These tests prove two things about the diagnostic in `tools/memory_diagnostic.py`: 1. **It measures what it claims.** - The fixture keeps the planted fact out of every layer except memory. - Each stage (created, retained, ranked, injected) is reported from the rows and the recall turn's own stored context. - Its ranking agrees with the selection production stored. 2. **What it finds on this tree.** The scenarios run with a best-case summariser, one that keeps a fact if and only if the fact reached it. Any failure is therefore the application's mechanism, not a model's writing. - WP-B.1 marked the criteria v1.0.0 did not meet `xfail(strict=True)`. WP-B.2 fixed ranking, eviction and the creation excerpt, and those tests are now ordinary passes; the v1.0.0 results are recorded in the WP-B.2 report. - The same file is run unchanged against v1.0.0 for the baseline. python -m pytest tests/test_v11_b1_memory_diagnostic.py -v """ import asyncio import pytest from fastapi import Depends from fastapi.testclient import TestClient from sqlalchemy.orm import undefer from app import auth, limits, memorybank, models from app.database import Base, SessionLocal, engine, get_db from app.main import app from app.routers import adventures from tools import memory_diagnostic as md _results: dict = {} def scenario(name: str) -> dict: """Runs a named scenario once per session and keeps the result.""" if name not in _results: _results[name] = md.run_scenario(md.SCENARIOS[name]) return _results[name] # ------------------------------------------------------- fixture preconditions def test_the_fact_is_planted_early_and_recalled_past_depth_one_hundred(): result = scenario("independent_default") assert result["plant_depth"] is not None and result["plant_depth"] <= 3 assert result["recall_depth"] >= 100 @pytest.mark.parametrize("check", ["state_document", "state_snapshots", "later_narration", "summary", "knowledge", "recent_history", "state_section"]) def test_no_layer_but_memory_carries_the_fact(check): """A test where another layer carries F is not evidence about memory.""" isolation = scenario("independent_default")["isolation"] assert isolation["checks"][check]["ok"], isolation["checks"][check] assert isolation["ok"] def test_the_isolation_check_fails_when_another_layer_carries_the_fact(): """The negative control for the precondition itself: a state fact naming F.""" fact = md.FACT_F with SessionLocal() as db: Base.metadata.create_all(bind=engine) try: user = models.User(is_guest=False, email="b1-iso@example.com") db.add(user) db.flush() adventure = models.Adventure(user_id=user.id, title="iso") adventure.narrative_state = {"facts": [{"id": "x", "predicate": "hidden", "value": "the amber sundial is in the teapot"}]} db.add(adventure) db.commit() result = md.isolation(db, adventure, fact, 1) assert result["ok"] is False assert result["checks"]["state_document"]["ok"] is False finally: db.close() Base.metadata.drop_all(bind=engine) # ------------------------------------------------------------------- stages def test_creation_is_reported_with_the_covering_memory_and_what_the_summariser_saw(): created = scenario("independent_default")["diagnosis"]["created"] assert created["yes"] is True assert created["source_start"] <= scenario("independent_default")["plant_depth"] <= created["source_end"] assert md.FACT_F.carried_by(created["memory_text"]) covering = [c for c in created["covering_memories"] if c["memory_id"] == created["memory_id"]] assert covering and covering[0]["fact_in_block"] and covering[0]["fact_in_summariser_excerpt"] def test_retention_is_reported_with_the_bank_and_its_eviction_order(): retained = scenario("independent_default")["diagnosis"]["retained"] assert retained["yes"] is True and retained["forgotten"] is False assert retained["on_active_lineage"] is True assert retained["active_memories"] <= retained["memory_bank_capacity"] assert retained["eviction_position"] is not None def test_ranking_is_production_ranking_and_agrees_with_the_stored_selection(): ranked = scenario("independent_default")["diagnosis"]["ranked"] assert ranked["replica_matches_stored_selection"] is True assert ranked["top_k_cutoff"] == 5 assert ranked["yes"] is True and ranked["selected"] is True assert 1 <= ranked["rank"] <= ranked["top_k_cutoff"] # v1.1 WP-B.2: every part of the score is reported, and they add up. assert 0.0 <= ranked["lexical_score"] <= 1.0 assert ranked["final_score"] == pytest.approx( ranked["semantic_score"] + memorybank.LEXICAL_WEIGHT * ranked["lexical_score"], abs=2e-4) assert ranked["query"]["input"].endswith(md.SCENARIOS["independent_default"].recall_text) def test_injection_is_read_from_the_recall_turns_own_context(): diagnosis = scenario("independent_default")["diagnosis"] assert diagnosis["injected"]["yes"] is True assert diagnosis["injected"]["context_component"] == md.MEMORIES_LABEL assert diagnosis["injected"]["token_count"] > 0 assert diagnosis["verdict"] == "injected" def test_ranking_variants_direct_paraphrase_and_unrelated(): variants = scenario("independent_default")["ranking_variants"] assert variants["direct"]["rank"] == 1 and variants["direct"]["selected"] assert variants["paraphrase"]["rank"] == 1 and variants["paraphrase"]["selected"] assert (variants["direct"]["final_score"] > variants["paraphrase"]["final_score"] > variants["unrelated"]["final_score"]) def test_retrieval_still_fills_top_k_whatever_the_similarity(): """There is still no relevance floor: an unrelated question selects a full `memory_top_k`. B.2 changed which memories those are, not how many — the early fact is no longer carried along by an unrelated question.""" variants = scenario("independent_default")["ranking_variants"] assert variants["unrelated"]["selected_count"] == 5 assert variants["unrelated"]["selected"] is False assert variants["unrelated"]["rank"] > 5 # ------------------------------------------------- WP-B.2 ranking acceptance @pytest.mark.parametrize("name", ["ranking_crowded", "ranking_context_dependent"]) def test_acceptance_the_early_memory_is_ranked_and_injected_below_capacity(name): """The B.1 ranking failure, made deterministic. On v1.0.0 both fixtures are `retained_but_not_ranked` (ranks 7 and 6 of 17 against a top-k of 4).""" result = scenario(name) assert result["isolation"]["ok"], result["isolation"] diagnosis = result["diagnosis"] assert diagnosis["created"]["yes"] and diagnosis["retained"]["yes"] assert diagnosis["retained"]["active_memories"] <= diagnosis["retained"]["memory_bank_capacity"] assert diagnosis["ranked"]["yes"] and diagnosis["ranked"]["rank"] <= 4 assert diagnosis["ranked"]["replica_matches_stored_selection"] is True assert diagnosis["injected"]["yes"] is True assert diagnosis["verdict"] == "injected" def test_the_crowded_fixture_is_won_by_the_players_question(): ranked = scenario("ranking_crowded")["diagnosis"]["ranked"] assert ranked["rank"] == 1 assert ranked["lexical_score"] > 0 # "sundial" and "amber" are in the question def test_a_paraphrase_is_found_by_meaning_not_by_shared_words(): """Lexical matching must not replace semantic retrieval. The paraphrase shares none of F's distinctive words, yet ranks first.""" for name in ("ranking_crowded", "independent_default"): paraphrase = scenario(name)["ranking_variants"]["paraphrase"] assert paraphrase["rank"] == 1 and paraphrase["selected"] # Only "Mara" is shared, which is far less than the direct question holds. direct = scenario(name)["ranking_variants"]["direct"] assert paraphrase["lexical_score"] < direct["lexical_score"] / 2 def test_a_context_dependent_question_needs_the_scene(): """"I ask her what she keeps up there" names nothing F's memory holds. The scene the last narration set up (Mara, the top shelf, a kettle) is what finds it; without that context it ranks last.""" result = scenario("ranking_context_dependent") ranked = result["diagnosis"]["ranked"] assert ranked["lexical_score"] == 0.0 assert ranked["rank"] <= 4 assert "top shelf" in ranked["query"]["context"] assert result["ranking_variants"]["input_only"]["rank"] > 4 def test_an_unrelated_rare_word_does_not_outrank_the_relevant_memory(): """Negative control: the paraphrase plus a place only one other memory holds. The decoy gains lexical score, and still ranks below F.""" for name in ("ranking_crowded", "independent_default"): control = scenario(name)["ranking_variants"]["rare_word_with_paraphrase"] assert control["decoy_lexical_score"] > control["lexical_score"] assert control["rank"] == 1 assert control["decoy_rank"] > control["rank"] def test_common_words_contribute_nothing(): common = scenario("independent_default")["ranking_variants"]["common_words"] assert common["lexical_score"] == 0.0 # ---------------------------------------------------------- capacity/eviction @pytest.mark.parametrize("name", ["past_capacity", "past_capacity_pinned", "past_capacity_low_top_k"]) def test_past_capacity_the_early_memory_is_retained(name): """v1.1 WP-B.2. On v1.0.0 all three are `created_but_evicted`: F was the least recently used row once recent narration stopped retrieving it, and went first (turns 21, 21 and 36). Coverage-first eviction keeps the only memory of the opening, so it stays active and is recalled at depth 106.""" result = scenario(name) assert result["isolation"]["ok"], result["isolation"] diagnosis = result["diagnosis"] assert diagnosis["created"]["yes"] and diagnosis["retained"]["yes"] assert result["eviction"]["f_evicted_at_turn"] is None # The bank really was past capacity, and stayed bounded. assert result["eviction"]["first_eviction_turn"] is not None assert all(t["active"] <= result["scenario"]["capacity"] + (1 if result["scenario"]["pin_first_memory"] else 0) for t in result["trace"]) assert result["trace"][-1]["total"] > result["scenario"]["capacity"] def test_past_capacity_the_bank_still_describes_the_whole_story(): """What the rule buys in general, not only for F: the active bank reaches from the opening to the newest block, and no stretch between them goes undescribed for more than twice the average spacing a bank of this capacity can afford (story span / capacity). On v1.0.0 these banks began at depths 36 and 18: the opening was simply gone.""" for name in ("past_capacity", "past_capacity_low_top_k"): result = scenario(name) cover = result["trace"][-1]["coverage"] assert cover["first_start"] == 0 assert cover["last_end"] >= result["recall_depth"] - 2 * memorybank.MEMORY_INTERVAL assert cover["largest_gap"] <= 2 * (cover["last_end"] + 1) / result["scenario"]["capacity"] def test_no_memory_is_evicted_by_the_same_pass_that_created_it(): """The frozen-bank regression the v1.0.0 rule fixed, still holding.""" for name in ("past_capacity", "past_capacity_pinned", "past_capacity_low_top_k"): assert scenario(name)["eviction"]["created_and_evicted_same_turn"] == [] def test_a_pinned_memory_survives_capacity(): eviction = scenario("past_capacity_pinned")["eviction"] assert eviction["pinned_memory_id"] is not None assert eviction["pinned_memory_forgotten"] is False def test_acceptance_an_early_fact_is_recalled_from_memory_past_capacity(): """Was `xfail(strict=True)` in WP-B.1; B.2 fixed the eviction rule.""" for name in ("past_capacity", "past_capacity_pinned", "past_capacity_low_top_k"): assert scenario(name)["diagnosis"]["verdict"] == "injected" # ---------------------------------------------------------- creation window def test_a_fact_early_in_a_long_block_now_reaches_the_summariser(): """v1.1 WP-B.2. On v1.0.0 this block (2,079 tokens) was cut to its last 2,000, the fact at its start was never seen, and the stage was `not_created`. The excerpt is now the block's opening and end.""" result = scenario("long_block_fact_early") created = result["diagnosis"]["created"] covering = created["covering_memories"] assert covering, "the long block must have been summarised" assert covering[0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS assert covering[0]["fact_in_block"] is True assert covering[0]["fact_in_summariser_excerpt"] is True assert created["yes"] is True assert created["source_start"] <= result["plant_depth"] <= created["source_end"] assert memorybank.EXCERPT_OMISSION_MARKER not in created["memory_text"] def test_the_same_fact_late_in_the_same_sized_block_does(): result = scenario("long_block_fact_late") covering = result["diagnosis"]["created"]["covering_memories"] assert covering[0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS assert covering[0]["fact_in_summariser_excerpt"] is True assert result["diagnosis"]["created"]["yes"] is True def test_acceptance_a_fact_early_in_a_long_block_is_remembered(): """Was `xfail(strict=True)` in WP-B.1; B.2 changed the excerpt.""" assert scenario("long_block_fact_early")["diagnosis"]["created"]["yes"] is True assert scenario("long_block_fact_early")["diagnosis"]["verdict"] == "injected" # ------------------------------------------ WP-B.2 full deterministic acceptance def test_acceptance_full_isolation_holds_on_every_turn(): """`independent_full`: long blocks, a crowded query, a bank past capacity. F must be carried by memory alone for the whole run, not only at recall.""" result = scenario("independent_full") assert result["plant_depth"] <= 3 and result["recall_depth"] >= 100 assert not any(t["f_in_state"] for t in result["trace"]) assert not any(t["f_in_summary"] for t in result["trace"]) assert result["isolation"]["ok"], result["isolation"] for check in ("state_document", "state_snapshots", "later_narration", "summary", "knowledge", "recent_history", "state_section"): assert result["isolation"]["checks"][check]["ok"], check def test_acceptance_full_every_stage_passes_past_capacity_with_long_blocks(): """On v1.0.0 this fixture fails at creation: every block is over 2,000 tokens, and the fact at the start of the first one is never summarised.""" result = scenario("independent_full") diagnosis = result["diagnosis"] assert result["trace"][-1]["total"] > result["scenario"]["capacity"] assert diagnosis["created"]["covering_memories"][0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS assert diagnosis["created"]["yes"] and diagnosis["retained"]["yes"] assert diagnosis["ranked"]["yes"] and diagnosis["ranked"]["replica_matches_stored_selection"] assert diagnosis["injected"]["yes"] assert diagnosis["verdict"] == "injected" def test_acceptance_full_provenance_resolves_to_the_planting_turn(): provenance = scenario("independent_full")["provenance"] assert provenance["recorded"] is not None assert provenance["range_covers_plant"] and provenance["matches_row"] assert provenance["source_block_holds_planting"] is True assert provenance["recorded"]["authority"] == memorybank.ACCEPTED_STORY def test_acceptance_full_is_the_long_run_independent_memory_verdict(): """The same measurements, judged by the long-run tool's own verdict.""" from tools import m11_long_run as lr result = scenario("independent_full") checks = result["isolation"]["checks"] diagnosis = result["diagnosis"] verdict = lr._independent_memory_verdict({ "independent_planted_depth": result["plant_depth"], "planted_turn_outside_history": checks["recent_history"]["ok"], "absent_from_state": checks["state_document"]["ok"] and checks["state_snapshots"]["ok"] and not any(t["f_in_state"] for t in result["trace"]), "absent_from_summary": checks["summary"]["ok"] and not any(t["f_in_summary"] for t in result["trace"]), "absent_from_knowledge": checks["knowledge"]["ok"], "absent_from_later_narration": checks["later_narration"]["ok"], "memory_covering_planting_carries_fact": diagnosis["created"]["yes"], "memory_forgotten": not diagnosis["retained"]["yes"], "memory_injected": diagnosis["injected"]["yes"], }) assert verdict == "recovered_through_memory_independent" # ------------------------------------------------------- lineage control (G) def test_an_abandoned_lines_memory_is_stored_but_never_eligible_or_injected(): g = scenario("lineage_control")["lineage_control"] assert g["memory_ids"], "G's memory must exist on line A before it is abandoned" assert sorted(g["stored"]) == sorted(g["memory_ids"]) assert g["eligible_on_active_line"] == [] assert g["g_text_ever_in_used_memories"] is False # Any turn that did name G's memory was on line A, before the divergence. assert g["eligible_after_returning_to_line_a"] == g["memory_ids"] def test_the_lineage_scenario_still_diagnoses_f_on_the_active_line(): result = scenario("lineage_control") assert result["isolation"]["ok"], result["isolation"] assert result["diagnosis"]["verdict"] == "injected" # ----------------------------------------------------- authority control @pytest.fixture() def authority_client(monkeypatch): embedder = md.ConceptEmbedder() Base.metadata.create_all(bind=engine) memorybank._vector_cache.clear() with SessionLocal() as db: user = models.User(is_guest=False, email="b1-auth@example.com") db.add(user) db.flush() db.add(models.Settings(user_id=user.id, model="script", endpoint_url="http://127.0.0.1:9/v1", embedding_model="concept-embed", memory_top_k=5)) adventure = models.Adventure(user_id=user.id, title="auth", memory_bank_enabled=True, auto_summarize=True) db.add(adventure) db.flush() db.add(models.Action(adventure_id=adventure.id, type="start", text="The tavern at dusk.")) db.commit() adv, user_id = adventure.id, user.id monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", md.ScriptNarrator) monkeypatch.setattr(memorybank, "embedding_provider", lambda s: embedder) monkeypatch.setattr(memorybank, "summary_provider", lambda s: md.BestCaseSummariser()) monkeypatch.setattr(memorybank, "schedule_post_turn", lambda a: None) app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id)) client = TestClient(app) client.adv = adv try: yield client finally: app.dependency_overrides.clear() adventures.turns._active_turns.clear() memorybank._vector_cache.clear() Base.metadata.drop_all(bind=engine) def test_a_memory_that_contradicts_state_loses_and_changes_nothing(authority_client): client, adv = authority_client, authority_client.adv corrected = client.post(f"/api/adventures/{adv}/state/corrections", json={"events": [ {"type": "add_fact", "predicate": "the tavern lamp is lit", "fact_id": "lamp-lit"}]}) assert corrected.status_code in (200, 201), corrected.text[:300] made = client.post(f"/api/adventures/{adv}/memories", json={"text": "The tavern lamp was never lit that night."}) assert made.status_code == 201, made.text[:300] client.patch(f"/api/adventures/{adv}/memories/{made.json()['id']}", json={"pinned": True}) asyncio.run(memorybank.run_post_turn(adv)) # embed it before = client.get(f"/api/adventures/{adv}/state").json()["document"] md.ScriptNarrator.next_reply = 'The fire crackles.\n```state\n{"events": []}\n```' played = client.post(f"/api/adventures/{adv}/actions", json={"type": "do", "text": "I look at the lamp."}) assert played.status_code == 200 and '"type": "error"' not in played.text after = client.get(f"/api/adventures/{adv}/state").json()["document"] assert after == before # retrieval mutated no state with SessionLocal() as db: action = (db.query(models.Action).filter_by(adventure_id=adv, type="ai") .options(undefer(models.Action.context_snapshot)) .order_by(models.Action.id.desc()).first()) snapshot = action.context_snapshot state_text = md._section(snapshot, md.STATE_LABEL) memory_text = md._section(snapshot, md.MEMORIES_LABEL) assert "the tavern lamp is lit" in state_text assert "never lit" in memory_text assert memory_text.startswith("Memories from earlier in the story") labels = [s["label"] for s in snapshot["sections"]] # State is read last of the live sections: it settles the conflict. assert labels.index(md.STATE_LABEL) > labels.index(md.MEMORIES_LABEL)