From beb17ada10331570b87fb4fcb101f644f0b02b71 Mon Sep 17 00:00:00 2001 From: JesseMarkowitz Date: Mon, 14 Sep 2026 20:50:05 -0400 Subject: [PATCH] v1.1 WP-B.1: diagnose independent long-term memory retention Diagnostic only; no memory behaviour changes. - tools/memory_diagnostic.py: planted-fact isolation checks, the four-stage diagnosis (created / retained / ranked / injected) with a verdict, a production-ranking replica, deterministic summariser/embedder/narrator stubs and seven scenarios (default, past capacity, pinned, low top_k, long-block early/late, lineage control) - tools/v11_b1_memory.py: CLI for the scenarios and for diagnosing a copy of a finished real campaign - tools/m11_long_run.py: opt-in --independent-fact mode with per-turn isolation tracking and the recovered_through_memory_independent verdict; M04 verdicts unchanged - tests: diagnostic stages, eviction, creation window, ranking, lineage and authority controls; two strict xfails record the diagnosed retention and creation defects for WP-B.2 to flip - planning/reports/v1.1/V1.1-WP-B1-REPORT.md First failing stage: ranking (real model); retention past capacity and creation for early facts in long blocks (deterministic, same on v1.0.0). Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY --- backend/tests/test_v11_b1_long_run_verdict.py | 86 ++ .../tests/test_v11_b1_memory_diagnostic.py | 308 +++++++ backend/tools/m11_long_run.py | 206 +++++ backend/tools/memory_diagnostic.py | 839 ++++++++++++++++++ backend/tools/v11_b1_memory.py | 122 +++ planning/reports/v1.1/V1.1-WP-B1-REPORT.md | 623 +++++++++++++ 6 files changed, 2184 insertions(+) create mode 100644 backend/tests/test_v11_b1_long_run_verdict.py create mode 100644 backend/tests/test_v11_b1_memory_diagnostic.py create mode 100644 backend/tools/memory_diagnostic.py create mode 100644 backend/tools/v11_b1_memory.py create mode 100644 planning/reports/v1.1/V1.1-WP-B1-REPORT.md diff --git a/backend/tests/test_v11_b1_long_run_verdict.py b/backend/tests/test_v11_b1_long_run_verdict.py new file mode 100644 index 0000000..b9f7886 --- /dev/null +++ b/backend/tests/test_v11_b1_long_run_verdict.py @@ -0,0 +1,86 @@ +"""v1.1 WP-B.1: the long run's `recovered_through_memory_independent` verdict. + +The new verdict must never be reported when anything other than memory could +have carried the fact. Each precondition is named when it fails. The existing M04 +verdicts keep their meaning exactly. + + python -m pytest tests/test_v11_b1_long_run_verdict.py -v +""" + +import pytest + +from tools import m11_long_run as lr + +GOOD = { + "independent_planted_depth": 3, + "planted_turn_outside_history": True, + "absent_from_state": True, + "absent_from_summary": True, + "absent_from_knowledge": True, + "absent_from_later_narration": True, + "memory_covering_planting_carries_fact": True, + "memory_forgotten": False, + "memory_injected": True, +} + + +def test_every_precondition_and_an_injected_memory_is_the_new_verdict(): + assert lr._independent_memory_verdict(GOOD) == "recovered_through_memory_independent" + + +@pytest.mark.parametrize("name", lr.INDEPENDENT_PRECONDITIONS) +def test_a_failed_precondition_is_named_and_never_a_recovery(name): + assert lr._independent_memory_verdict({**GOOD, name: False}) == f"precondition_failed:{name}" + + +@pytest.mark.parametrize("name", lr.INDEPENDENT_PRECONDITIONS) +def test_an_unmeasured_precondition_is_unknown_not_a_pass(name): + assert lr._independent_memory_verdict({**GOOD, name: None}) == f"precondition_unknown:{name}" + + +def test_no_planted_depth_is_unknown(): + assert lr._independent_memory_verdict({**GOOD, "independent_planted_depth": None}) == \ + "precondition_unknown:planted_depth" + + +@pytest.mark.parametrize("change, verdict", [ + ({"memory_covering_planting_carries_fact": False}, "not_recovered:not_created"), + ({"memory_forgotten": True}, "not_recovered:evicted"), + ({"memory_injected": False}, "not_recovered:not_injected"), +]) +def test_the_failing_memory_stage_is_named(change, verdict): + assert lr._independent_memory_verdict({**GOOD, **change}) == verdict + + +def test_preconditions_are_judged_before_memory(): + """A carried fact disqualifies the run even when memory also failed.""" + both = {**GOOD, "absent_from_state": False, "memory_covering_planting_carries_fact": False} + assert lr._independent_memory_verdict(both) == "precondition_failed:absent_from_state" + + +def test_the_fact_is_matched_as_whole_words(): + assert lr._mentions_fact("She hid the amber Sundial.") + assert lr._mentions_fact("a cracked TEAPOT on the shelf") + assert not lr._mentions_fact("teapots") # a different word, not the fact's + assert not lr._mentions_fact("the sun dialled down") + + +def test_the_m04_verdicts_are_unchanged(): + base = {"planted_turn_in_history_window": False, "in_memories_section": False, + "in_summary_section": False, "in_state_section": False} + assert lr._m04_verdict(base) == "not_recovered" + assert lr._m04_verdict({**base, "in_state_section": True}) == "recovered_through_state_only" + assert lr._m04_verdict({**base, "in_memories_section": True}) == \ + "recovered_through_memory_or_summary" + assert lr._m04_verdict({**base, "planted_turn_in_history_window": True}) == \ + "precondition_not_met" + + +def test_the_independent_fact_is_not_in_any_imported_knowledge_file(): + for text in (lr.CANON_MD, lr.REFERENCE_MD, lr.INSPIRATION_MD, *lr.BEATS): + assert not lr._mentions_fact(text) + + +def test_the_planting_text_and_recall_carry_the_fact(): + assert lr._mentions_fact(lr.INDEPENDENT_FACT_TEXT) + assert lr._mentions_fact(lr.INDEPENDENT_RECALL_TEXT) diff --git a/backend/tests/test_v11_b1_memory_diagnostic.py b/backend/tests/test_v11_b1_memory_diagnostic.py new file mode 100644 index 0000000..a9fb2a0 --- /dev/null +++ b/backend/tests/test_v11_b1_memory_diagnostic.py @@ -0,0 +1,308 @@ +"""v1.1 WP-B.1: the memory-retention diagnostic, deterministically. + +B.1 changes no memory behaviour. These tests prove two things about the +diagnostic in `tools/memory_diagnostic.py`: + +1. **It measures what it claims.** + - The fixture keeps the planted fact out of every layer except memory. + - Each stage (created, retained, ranked, injected) is reported from the rows + and the recall turn's own stored context. + - Its ranking agrees with the selection production stored. +2. **What it finds on this tree.** The scenarios run with a best-case summariser, + one that keeps a fact if and only if the fact reached it. Any failure is + therefore the application's mechanism, not a model's writing. + - The criteria the current code does not meet are marked `xfail(strict=True)`, + so B.2 has to flip them deliberately. + - The same file is run unchanged against v1.0.0 for the baseline. + + python -m pytest tests/test_v11_b1_memory_diagnostic.py -v +""" + +import asyncio + +import pytest +from fastapi import Depends +from fastapi.testclient import TestClient +from sqlalchemy.orm import undefer + +from app import auth, limits, memorybank, models +from app.database import Base, SessionLocal, engine, get_db +from app.main import app +from app.routers import adventures +from tools import memory_diagnostic as md + +_results: dict = {} + + +def scenario(name: str) -> dict: + """Runs a named scenario once per session and keeps the result.""" + if name not in _results: + _results[name] = md.run_scenario(md.SCENARIOS[name]) + return _results[name] + + +# ------------------------------------------------------- fixture preconditions + +def test_the_fact_is_planted_early_and_recalled_past_depth_one_hundred(): + result = scenario("independent_default") + assert result["plant_depth"] is not None and result["plant_depth"] <= 3 + assert result["recall_depth"] >= 100 + + +@pytest.mark.parametrize("check", ["state_document", "state_snapshots", "later_narration", + "summary", "knowledge", "recent_history", "state_section"]) +def test_no_layer_but_memory_carries_the_fact(check): + """A test where another layer carries F is not evidence about memory.""" + isolation = scenario("independent_default")["isolation"] + assert isolation["checks"][check]["ok"], isolation["checks"][check] + assert isolation["ok"] + + +def test_the_isolation_check_fails_when_another_layer_carries_the_fact(): + """The negative control for the precondition itself: a state fact naming F.""" + fact = md.FACT_F + with SessionLocal() as db: + Base.metadata.create_all(bind=engine) + try: + user = models.User(is_guest=False, email="b1-iso@example.com") + db.add(user) + db.flush() + adventure = models.Adventure(user_id=user.id, title="iso") + adventure.narrative_state = {"facts": [{"id": "x", "predicate": "hidden", + "value": "the amber sundial is in the teapot"}]} + db.add(adventure) + db.commit() + result = md.isolation(db, adventure, fact, 1) + assert result["ok"] is False + assert result["checks"]["state_document"]["ok"] is False + finally: + db.close() + Base.metadata.drop_all(bind=engine) + + +# ------------------------------------------------------------------- stages + +def test_creation_is_reported_with_the_covering_memory_and_what_the_summariser_saw(): + created = scenario("independent_default")["diagnosis"]["created"] + assert created["yes"] is True + assert created["source_start"] <= scenario("independent_default")["plant_depth"] <= created["source_end"] + assert md.FACT_F.carried_by(created["memory_text"]) + covering = [c for c in created["covering_memories"] if c["memory_id"] == created["memory_id"]] + assert covering and covering[0]["fact_in_block"] and covering[0]["fact_in_summariser_excerpt"] + + +def test_retention_is_reported_with_the_bank_and_its_eviction_order(): + retained = scenario("independent_default")["diagnosis"]["retained"] + assert retained["yes"] is True and retained["forgotten"] is False + assert retained["on_active_lineage"] is True + assert retained["active_memories"] <= retained["memory_bank_capacity"] + assert retained["eviction_position"] is not None + + +def test_ranking_is_production_ranking_and_agrees_with_the_stored_selection(): + ranked = scenario("independent_default")["diagnosis"]["ranked"] + assert ranked["replica_matches_stored_selection"] is True + assert ranked["lexical_score"] is None # memory ranking has no lexical term + assert ranked["top_k_cutoff"] == 5 + assert ranked["yes"] is True and ranked["selected"] is True + assert 1 <= ranked["rank"] <= ranked["top_k_cutoff"] + # The production query is the newest four actions, cut to 600 tokens, and the + # one-line question is diluted by the narration around it. + variants = scenario("independent_default")["ranking_variants"] + assert ranked["semantic_score"] < variants["direct"]["similarity"] + + +def test_injection_is_read_from_the_recall_turns_own_context(): + diagnosis = scenario("independent_default")["diagnosis"] + assert diagnosis["injected"]["yes"] is True + assert diagnosis["injected"]["context_component"] == md.MEMORIES_LABEL + assert diagnosis["injected"]["token_count"] > 0 + assert diagnosis["verdict"] == "injected" + + +def test_ranking_variants_direct_paraphrase_and_unrelated(): + variants = scenario("independent_default")["ranking_variants"] + assert variants["direct"]["rank"] == 1 and variants["direct"]["selected"] + assert variants["paraphrase"]["rank"] == 1 and variants["paraphrase"]["selected"] + assert (variants["direct"]["similarity"] > variants["paraphrase"]["similarity"] + > 5 * variants["unrelated"]["similarity"]) + + +def test_retrieval_fills_top_k_whatever_the_similarity(): + """Diagnosis: there is no relevance floor. With more memories than + `memory_top_k`, an unrelated query still selects five, and the early fact + rides along at a similarity near zero.""" + variants = scenario("independent_default")["ranking_variants"] + assert variants["unrelated"]["similarity"] < 0.1 + assert variants["unrelated"]["selected"] is True + + +# ---------------------------------------------------------- capacity/eviction + +def test_past_capacity_the_early_memory_is_evicted_and_the_stage_says_so(): + """Diagnosis, not a requirement: what the current eviction rule does to F.""" + result = scenario("past_capacity") + assert result["diagnosis"]["created"]["yes"] is True + assert result["diagnosis"]["verdict"] == "created_but_evicted" + eviction = result["eviction"] + assert eviction["f_evicted_at_turn"] is not None + # It was retrieved while the bank was small, stopped being retrieved once + # recent narration filled the top-k, and was then the least recently used. + assert eviction["f_use_count_when_evicted"] > 0 + assert result["f_last_use_increase_turn"] < eviction["f_evicted_at_turn"] + assert eviction["f_memory_was_first_evicted"] is True + + +def test_at_a_lower_top_k_the_early_memory_ages_out_after_it_stops_being_retrieved(): + """Diagnosis with most of the bank unretrieved on any turn, nearer the + shipped 5-in-80 ratio. F is not simply the oldest row: it is evicted some + turns after recent narration stopped pulling it into the top-k, which is + what ordering by last use does to a fact nothing recent mentions.""" + result = scenario("past_capacity_low_top_k") + eviction = result["eviction"] + assert result["diagnosis"]["created"]["yes"] is True + assert result["diagnosis"]["verdict"] == "created_but_evicted" + assert eviction["f_use_count_when_evicted"] > 0 + assert result["f_last_use_increase_turn"] < eviction["f_evicted_at_turn"] + assert eviction["first_eviction_turn"] <= eviction["f_evicted_at_turn"] + assert eviction["created_and_evicted_same_turn"] == [] + + +def test_no_memory_is_evicted_by_the_same_pass_that_created_it(): + """The frozen-bank regression the current rule fixed, still holding.""" + for name in ("past_capacity", "past_capacity_pinned"): + assert scenario(name)["eviction"]["created_and_evicted_same_turn"] == [] + + +def test_a_pinned_memory_survives_capacity(): + eviction = scenario("past_capacity_pinned")["eviction"] + assert eviction["pinned_memory_id"] is not None + assert eviction["pinned_memory_forgotten"] is False + + +@pytest.mark.xfail(strict=True, reason=( + "WP-B.1 diagnosis on this tree: past memory_bank_capacity the planting-era " + "memory is evicted first, because it was never retrieved and eviction orders " + "by last use, then creation. B.2 must flip this deliberately.")) +def test_acceptance_an_early_fact_is_recalled_from_memory_past_capacity(): + assert scenario("past_capacity")["diagnosis"]["verdict"] == "injected" + + +# ---------------------------------------------------------- creation window + +def test_a_fact_early_in_a_long_block_never_reaches_the_summariser(): + result = scenario("long_block_fact_early") + created = result["diagnosis"]["created"] + covering = created["covering_memories"] + assert covering, "the long block must have been summarised" + assert covering[0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS + assert covering[0]["fact_in_block"] is True + assert covering[0]["fact_in_summariser_excerpt"] is False + assert result["diagnosis"]["verdict"] == "not_created" + + +def test_the_same_fact_late_in_the_same_sized_block_does(): + result = scenario("long_block_fact_late") + covering = result["diagnosis"]["created"]["covering_memories"] + assert covering[0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS + assert covering[0]["fact_in_summariser_excerpt"] is True + assert result["diagnosis"]["created"]["yes"] is True + + +@pytest.mark.xfail(strict=True, reason=( + "WP-B.1 diagnosis on this tree: the summariser reads only the last " + f"{memorybank.MEMORY_EXCERPT_TOKENS} tokens of a block, so a fact early in a " + "long block is never seen. B.2 must flip this deliberately.")) +def test_acceptance_a_fact_early_in_a_long_block_is_remembered(): + assert scenario("long_block_fact_early")["diagnosis"]["created"]["yes"] is True + + +# ------------------------------------------------------- lineage control (G) + +def test_an_abandoned_lines_memory_is_stored_but_never_eligible_or_injected(): + g = scenario("lineage_control")["lineage_control"] + assert g["memory_ids"], "G's memory must exist on line A before it is abandoned" + assert sorted(g["stored"]) == sorted(g["memory_ids"]) + assert g["eligible_on_active_line"] == [] + assert g["g_text_ever_in_used_memories"] is False + # Any turn that did name G's memory was on line A, before the divergence. + assert g["eligible_after_returning_to_line_a"] == g["memory_ids"] + + +def test_the_lineage_scenario_still_diagnoses_f_on_the_active_line(): + result = scenario("lineage_control") + assert result["isolation"]["ok"], result["isolation"] + assert result["diagnosis"]["verdict"] == "injected" + + +# ----------------------------------------------------- authority control + +@pytest.fixture() +def authority_client(monkeypatch): + embedder = md.ConceptEmbedder() + Base.metadata.create_all(bind=engine) + memorybank._vector_cache.clear() + with SessionLocal() as db: + user = models.User(is_guest=False, email="b1-auth@example.com") + db.add(user) + db.flush() + db.add(models.Settings(user_id=user.id, model="script", + endpoint_url="http://127.0.0.1:9/v1", + embedding_model="concept-embed", memory_top_k=5)) + adventure = models.Adventure(user_id=user.id, title="auth", memory_bank_enabled=True, + auto_summarize=True) + db.add(adventure) + db.flush() + db.add(models.Action(adventure_id=adventure.id, type="start", text="The tavern at dusk.")) + db.commit() + adv, user_id = adventure.id, user.id + monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) + monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", md.ScriptNarrator) + monkeypatch.setattr(memorybank, "embedding_provider", lambda s: embedder) + monkeypatch.setattr(memorybank, "summary_provider", lambda s: md.BestCaseSummariser()) + monkeypatch.setattr(memorybank, "schedule_post_turn", lambda a: None) + app.dependency_overrides[auth.get_current_user] = ( + lambda db=Depends(get_db): db.get(models.User, user_id)) + client = TestClient(app) + client.adv = adv + try: + yield client + finally: + app.dependency_overrides.clear() + adventures.turns._active_turns.clear() + memorybank._vector_cache.clear() + Base.metadata.drop_all(bind=engine) + + +def test_a_memory_that_contradicts_state_loses_and_changes_nothing(authority_client): + client, adv = authority_client, authority_client.adv + corrected = client.post(f"/api/adventures/{adv}/state/corrections", json={"events": [ + {"type": "add_fact", "predicate": "the tavern lamp is lit", "fact_id": "lamp-lit"}]}) + assert corrected.status_code in (200, 201), corrected.text[:300] + made = client.post(f"/api/adventures/{adv}/memories", + json={"text": "The tavern lamp was never lit that night."}) + assert made.status_code == 201, made.text[:300] + client.patch(f"/api/adventures/{adv}/memories/{made.json()['id']}", json={"pinned": True}) + asyncio.run(memorybank.run_post_turn(adv)) # embed it + + before = client.get(f"/api/adventures/{adv}/state").json()["document"] + md.ScriptNarrator.next_reply = 'The fire crackles.\n```state\n{"events": []}\n```' + played = client.post(f"/api/adventures/{adv}/actions", + json={"type": "do", "text": "I look at the lamp."}) + assert played.status_code == 200 and '"type": "error"' not in played.text + after = client.get(f"/api/adventures/{adv}/state").json()["document"] + assert after == before # retrieval mutated no state + + with SessionLocal() as db: + action = (db.query(models.Action).filter_by(adventure_id=adv, type="ai") + .options(undefer(models.Action.context_snapshot)) + .order_by(models.Action.id.desc()).first()) + snapshot = action.context_snapshot + state_text = md._section(snapshot, md.STATE_LABEL) + memory_text = md._section(snapshot, md.MEMORIES_LABEL) + assert "the tavern lamp is lit" in state_text + assert "never lit" in memory_text + assert memory_text.startswith("Memories from earlier in the story") + labels = [s["label"] for s in snapshot["sections"]] + # State is read last of the live sections: it settles the conflict. + assert labels.index(md.STATE_LABEL) > labels.index(md.MEMORIES_LABEL) diff --git a/backend/tools/m11_long_run.py b/backend/tools/m11_long_run.py index 6dbc4ba..f8e2fdd 100644 --- a/backend/tools/m11_long_run.py +++ b/backend/tools/m11_long_run.py @@ -190,6 +190,31 @@ CLUE_FACT = { "fact_id": "silver-key-opens-crypt", } +#: v1.1 WP-B.1: a second planted fact, established in the **story only**. +#: +#: The M04 clue above is planted as accepted state, and memories are written +#: from story text, so no memory could ever carry it on its own. That is why +#: every M04 recovery so far ran through state. This fact is told to the reader +#: in narration and never corrected into state, so memory is the only layer that +#: is meant to carry it. `--independent-fact` plants it and reports +#: `recovered_through_memory_independent` only when every other layer is proven +#: not to carry it. The words are copied from `tools/memory_diagnostic.FACT_F`, +#: for the reason `HISTORY_LABELS` is copied. +INDEPENDENT_FACT_TEXT = ("I watch Mara slip the amber sundial inside the cracked teapot on " + "the tavern's top shelf, and she makes me promise to tell no one.") +INDEPENDENT_FACT_TERMS = ("sundial", "teapot") +INDEPENDENT_RECALL_TEXT = "I ask Mara quietly where she hid the amber sundial." +#: How far past the planting turn its memory block can reach. Narration inside +#: that block may repeat the fact; narration after it may not. +INDEPENDENT_BLOCK_SLACK = 6 +INDEPENDENT_PRECONDITIONS = ( + "planted_turn_outside_history", + "absent_from_state", + "absent_from_summary", + "absent_from_knowledge", + "absent_from_later_narration", +) + CANON = [ "The dead do not return. No rite, relic or bargain has ever returned anyone.", "The abbey crypt has been sealed since the founding.", @@ -365,6 +390,14 @@ class Run: #: The depth of the player turn that planted the clue. M04's #: precondition is that this turn has left the history window. self.planted_depth: int | None = None + #: v1.1 WP-B.1, with --independent-fact: where the story-only fact was + #: planted, and the accepted-turn count at which each isolation + #: precondition first failed. + self.independent_fact = False + self.independent_depth: int | None = None + self.independent_violations: dict[str, int] = {} + self.last_done: dict = {} + self.last_report: dict = {} # ------------------------------------------------------------ recording @@ -393,6 +426,8 @@ class Run: "turns_target": self.turns_target, "log_offset": self.log_offset, "planted_depth": self.planted_depth, + "independent_depth": self.independent_depth, + "independent_violations": self.independent_violations, "written": datetime.now().isoformat(timespec="seconds"), } tmp = self.out / (RESUME_FILE + ".tmp") @@ -414,6 +449,8 @@ class Run: self.elapsed_before = prior.get("elapsed_seconds", 0) self.log_offset = prior.get("log_offset", 0) self.planted_depth = prior.get("planted_depth") + self.independent_depth = prior.get("independent_depth") + self.independent_violations = dict(prior.get("independent_violations") or {}) self.resumed = True def reattach(self) -> None: @@ -570,15 +607,44 @@ class Run: "observed_margin": accounting.get("observed_margin"), "safety_reserve": accounting.get("safety_reserve"), }) + self.last_done = done + if self.independent_fact and self.independent_depth is not None: + self._check_independent_isolation(done, sample) self.note("turn", text=text, seconds=round(seconds, 1), **sample) return {"accepted": True, "seconds": seconds, **sample} + def _check_independent_isolation(self, done: dict, sample: dict) -> None: + """v1.1 WP-B.1: does anything but memory carry the story-only fact yet? + + Checked on every accepted turn, so a run knows the first turn at which + the experiment stopped being about memory, instead of finding out at + recall. Each precondition records only its first failure. + """ + depth = sample.get("total_actions", 0) - 1 + text = (done.get("action") or {}).get("text") or "" + found = {} + if depth > self.independent_depth + INDEPENDENT_BLOCK_SLACK and _mentions_fact(text): + found["absent_from_later_narration"] = f"narration at depth {depth}" + document = self.state().get("document") or {} + if _mentions_fact(json.dumps(document)): + found["absent_from_state"] = "the narrative state names the fact" + summary = next((sec.get("text", "") for sec in (self.last_report.get("sections") or []) + if sec.get("label") == SUMMARY_LABEL), "") + if _mentions_fact(summary): + found["absent_from_summary"] = "the active summary names the fact" + for name, detail in found.items(): + if name not in self.independent_violations: + self.independent_violations[name] = self.accepted + self.note("independent_precondition_failed", precondition=name, detail=detail) + sample["independent_violations"] = dict(self.independent_violations) + def count_actions(self) -> int: return self.server.call("GET", f"/adventures/{self.adv}/actions?limit=1")["total"] def measure(self) -> dict: """M03's numbers, read from the prompt the app would send right now.""" report = self.server.call("GET", f"/adventures/{self.adv}/context") + self.last_report = report tokens = report["tokens"] sections = {s["label"]: s["tokens"] for s in report["sections"]} window = report.get("window") or {} @@ -714,6 +780,10 @@ def main() -> int: "--max-consecutive-failures", type=int, default=DEFAULT_MAX_CONSECUTIVE_FAILURES, help="stop and write the evidence after this many unaccepted turns") + parser.add_argument( + "--independent-fact", action="store_true", + help=("v1.1 WP-B.1: also plant a story-only fact at depth 3 and report " + "whether memory alone recovers it")) args = parser.parse_args() if not (ENDPOINT and MODEL and EMBED_MODEL): @@ -746,6 +816,7 @@ def main() -> int: server.start() run = Run(server, out, turns_target=args.turns, turn_timeout=args.turn_timeout) + run.independent_fact = args.independent_fact if prior: run.adopt(prior) @@ -782,6 +853,18 @@ def main() -> int: "the planted clue is not in accepted state, so M04 cannot " "be measured from this run. Stopping before the campaign " "starts rather than reporting a recall failure later.") + if args.independent_fact: + # v1.1 WP-B.1: the story-only fact, told in the next turn and + # never corrected into state. Depth 3: the opening, the clue turn + # and its reply come first. + if any(_mentions_fact(md) for md in (CANON_MD, REFERENCE_MD, INSPIRATION_MD)): + raise SystemExit("the imported knowledge names the independent fact") + planting_f = run.turn(INDEPENDENT_FACT_TEXT) + if not planting_f.get("accepted"): + raise SystemExit("the turn that plants the independent fact was not accepted") + run.independent_depth = planting_f["total_actions"] - 2 + run.note("independent_fact_planted", depth=run.independent_depth, + terms=list(INDEPENDENT_FACT_TERMS)) # The first checkpoint, and the point from which --resume works: the # campaign exists and its clue is planted. run.save_resume() @@ -846,10 +929,15 @@ def main() -> int: # Skipped on an aborted run: it asks the narrator a question, and the # reason the run stopped is that the narrator does not answer. recall = None + independent = None if aborted is None: run.note("recall_begin") recall = _recall(run) (out / "recall.json").write_text(json.dumps(recall, indent=2)) + if args.independent_fact and run.independent_depth is not None: + independent = _independent_recall(run, out / "campaign.db") + (out / "recall-independent.json").write_text(json.dumps(independent, indent=2)) + run.note("independent_recall", verdict=independent["verdict"]) # ---- Export whatever exists, for the recovery evidence. ---- # Attempted even for an aborted run: the recovery check and the storage @@ -897,6 +985,7 @@ def main() -> int: "elapsed_seconds": run.elapsed(), "turn_timeout_seconds": args.turn_timeout, "recall": recall, + "independent_recall": independent, "final_state": _or_none(lambda: run.state()["document"]), "final_measurement": _or_none(run.measure), "db_bytes": db_path.stat().st_size, @@ -1224,6 +1313,123 @@ def _m04_verdict(recall: dict) -> str: return "not_recovered" +def _mentions_fact(text: str | None) -> bool: + """v1.1 WP-B.1: whether `text` names the independent fact, as a whole word.""" + low = (text or "").lower() + return any(re.search(rf"(? str: + """v1.1 WP-B.1: whether memory alone recovered the story-only fact. + + `recovered_through_memory_independent` requires every precondition, so no + other layer could have carried the fact. It also requires that a memory + covering the planting turn carries the fact and was injected into the recall + turn. A failed precondition is named and is never a recovery, and the M04 + verdicts above are untouched. + """ + if check.get("independent_planted_depth") is None: + return "precondition_unknown:planted_depth" + for name in INDEPENDENT_PRECONDITIONS: + value = check.get(name) + if value is None: + return f"precondition_unknown:{name}" + if not value: + return f"precondition_failed:{name}" + if not check.get("memory_covering_planting_carries_fact"): + return "not_recovered:not_created" + if check.get("memory_forgotten"): + return "not_recovered:evicted" + if not check.get("memory_injected"): + return "not_recovered:not_injected" + return "recovered_through_memory_independent" + + +def _independent_recall(run: "Run", db_path: Path) -> dict: + """v1.1 WP-B.1: ask for the story-only fact, and find out which layer answered. + + The prompt-level facts come from the recall turn's own stored context. The + memory rows come from the campaign database, read-only. Ranking is not + recomputed here, because that needs the embedding model; + `tools/v11_b1_memory.py diagnose` does it afterwards against a copy of the + database. + """ + import sqlite3 + import zlib + + result = run.turn(INDEPENDENT_RECALL_TEXT) + action_id = (run.last_done.get("action") or {}).get("id") + snapshot = (run.server.call("GET", f"/adventures/{run.adv}/actions/{action_id}/context") + if result.get("accepted") and action_id else {}) or {} + sections = {} + for sec in snapshot.get("sections") or []: + sections.setdefault(sec.get("label"), []).append(sec.get("text", "")) + text_of = {label: "\n".join(parts) for label, parts in sections.items()} + floor = (snapshot.get("history") or {}).get("floor_depth") + depth = run.independent_depth + used = [m.get("id") for m in (snapshot.get("memories") or {}).get("used") or []] + + covering = [] + connection = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True) + try: + rows = connection.execute( + "SELECT id, text, source_start, source_end, forgotten, pinned, use_count, " + "branch_id, depth FROM memories WHERE adventure_id = ? AND source_start <= ? " + "AND source_end >= ? ORDER BY id", (run.adv, depth, depth)).fetchall() + blob = connection.execute( + "SELECT context_snapshot FROM actions WHERE id = ?", (action_id or -1,)).fetchone() + finally: + connection.close() + for row in rows: + memory_id, text, start, end, forgotten, pinned, use_count, branch_id, node_depth = row + covering.append({ + "memory_id": memory_id, "text": text, "source_start": start, "source_end": end, + "forgotten": bool(forgotten), "pinned": bool(pinned), "use_count": use_count, + "branch_id": branch_id, "depth": node_depth, + "carries_fact": all(re.search(rf"(? bool: + low = (text or "").lower() + return all(any(_has_word(low, term) for term in group) for group in self.carry_groups) + + def mentioned_by(self, text: str | None) -> bool: + low = (text or "").lower() + return any(_has_word(low, term) for term in self.leak_terms) + + +def _has_word(low: str, term: str) -> bool: + return re.search(rf"(?= depth, + ) + if not any_branch: + query = query.where(lineage.path_of(db, adventure).clause(models.Memory)) + return db.execute(query.order_by(models.Memory.id)).scalars().all() + + +def planting_block_end(db, adventure, plant_depth: int) -> int: + """The last depth of the memory block holding the planted turn. + + Taken from the memory that covers it where one exists. Before one exists it + is the furthest a block could reach, so a later-narration check never counts + a turn inside the planting block as a repetition. + """ + rows = covering_memories(db, adventure, plant_depth) + if rows: + return max(row.source_end for row in rows) + return plant_depth + memorybank.MEMORY_INTERVAL + + +# ---------------------------------------------------------------- isolation + +def isolation(db, adventure, fact: Fact, plant_depth: int, *, + recall_snapshot: dict | None = None, + recall_depth: int | None = None) -> dict: + """Every layer other than memory that could carry F, checked. + + Returns `{check: {"ok": bool, "detail": str}}` and `ok` over all of them. + With `recall_snapshot`, the recall turn's stored context, the prompt-level + checks (history window, summary section, knowledge sections) are made + against what the narrator was actually given. + """ + checks: dict[str, dict] = {} + + document = adventure.narrative_state or {} + hits = [key for key in ("entities", "facts", "relationships", "threads", "scene", + "possessions") + if fact.mentioned_by(json.dumps(document.get(key), default=str))] + checks["state_document"] = { + "ok": not hits and not fact.mentioned_by(json.dumps(document, default=str)), + "detail": f"mentioned in {hits}" if hits else "absent", + } + + snapshot_hits = [] + later_hits = [] + block_end = planting_block_end(db, adventure, plant_depth) + for action in _lineage_actions(db, adventure): + if fact.mentioned_by(json.dumps(action.narrative_state_after, default=str)): + snapshot_hits.append(action.depth) + if (action.type == "ai" and action.depth is not None and action.depth > block_end + and (recall_depth is None or action.depth < recall_depth) + and fact.mentioned_by(action.text)): + later_hits.append(action.depth) + checks["state_snapshots"] = { + "ok": not snapshot_hits, + "detail": f"mentioned in snapshots at depths {snapshot_hits[:10]}" if snapshot_hits + else "absent from every node's narrative_state_after on the active lineage", + } + checks["later_narration"] = { + "ok": not later_hits, + "detail": (f"narration after the planting block (ends at depth {block_end}) " + f"mentions the fact at depths {later_hits[:10]}") if later_hits + else f"no narrator turn after depth {block_end} mentions the fact", + } + + active = summaries.current(db, adventure) + summary_text = active.text if active is not None else "" + if recall_snapshot is not None: + summary_text += "\n" + _section(recall_snapshot, SUMMARY_LABEL) + checks["summary"] = { + "ok": not fact.mentioned_by(summary_text), + "detail": "the active summary mentions the fact" if fact.mentioned_by(summary_text) + else ("absent from the active summary" if active is not None else "no summary yet"), + } + + sources = db.execute( + select(models.KnowledgeSource.content).where( + models.KnowledgeSource.adventure_id == adventure.id) + ).scalars().all() + knowledge_text = "\n".join(s or "" for s in sources) + if recall_snapshot is not None: + knowledge_text += "\n" + "\n".join(_section(recall_snapshot, l) for l in KNOWLEDGE_LABELS) + checks["knowledge"] = { + "ok": not fact.mentioned_by(knowledge_text), + "detail": "imported knowledge mentions the fact" if fact.mentioned_by(knowledge_text) + else f"absent from {len(sources)} imported source(s)", + } + + if recall_snapshot is not None: + hist = recall_snapshot.get("history") or {} + floor = hist.get("floor_depth") + history_text = "\n".join(_section(recall_snapshot, l) for l in HISTORY_LABELS) + outside = floor is not None and plant_depth < floor + checks["recent_history"] = { + "ok": outside and not fact.carried_by(history_text), + "detail": (f"history window starts at depth {floor}; planted at {plant_depth}; " + f"fact text in history sections: {fact.carried_by(history_text)}"), + } + checks["state_section"] = { + "ok": not fact.mentioned_by(_section(recall_snapshot, STATE_LABEL)), + "detail": "the recall prompt's narrative_state section " + + ("mentions the fact" if fact.mentioned_by(_section(recall_snapshot, STATE_LABEL)) + else "does not mention the fact"), + } + + return {"ok": all(c["ok"] for c in checks.values()), "checks": checks} + + +def _section(snapshot: dict, label: str) -> str: + return "\n".join(s.get("text", "") for s in (snapshot.get("sections") or []) + if s.get("label") == label) + + +# ------------------------------------------------------------------- stages + +async def rank_bank(db, adventure, settings, query: str, embed) -> dict: + """Production's ranking, recomputed for `query`, for every eligible memory. + + The same catalogue clause, the same cosine, the same pin rule, the same + redundancy suppression helper. Returns every scored row, not just the top-k, + because "where did F rank" is the question. + """ + catalogue = db.execute( + select(models.Memory.id, models.Memory.pinned, models.Memory.authority, + models.Memory.embedding_blob).where( + models.Memory.adventure_id == adventure.id, + lineage.path_of(db, adventure).clause(models.Memory), + models.Memory.forgotten.is_(False), + models.Memory.embedded.is_(True), + ) + ).all() + if not catalogue or not query.strip(): + return {"query": query, "scored": [], "selected": [], "top_k": settings.memory_top_k} + [query_vec] = await embed([query]) + held = {row.id: vectors.unpack(row.embedding_blob) for row in catalogue if row.embedding_blob} + authority_of = {row.id: row.authority for row in catalogue} + scored = sorted( + ((vectors.cosine(query_vec, held[row.id]), row.id, row.pinned) + for row in catalogue if row.id in held), + key=lambda r: r[0], reverse=True, + ) + top_k = max(1, settings.memory_top_k) + used = [r for r in scored if r[2]] + remaining = max(0, top_k - len(used)) + candidates = [r for r in scored if not r[2]] + kept, suppressed = memorybank._drop_redundant(candidates, held, authority_of, remaining) + selected = {r[1] for r in used + kept} + suppressed_by = dict(suppressed) + return { + "query": query, + "top_k": top_k, + "scored": [ + {"rank": i + 1, "memory_id": memory_id, "similarity": round(score, 4), + "pinned": pinned, "selected": memory_id in selected, + "suppressed_as_duplicate_of": suppressed_by.get(memory_id)} + for i, (score, memory_id, pinned) in enumerate(scored) + ], + "selected": sorted(selected), + } + + +def production_query(adventure, exclude_action_id: int | None) -> str: + """The retrieval query a turn used: its newest actions, as `retrieve_memories` builds it.""" + recent = history.tail(adventure, memorybank.RETRIEVAL_WINDOW_ACTIONS, exclude_action_id) + return builder.truncate_to_last_tokens( + "\n\n".join(a.text for a in recent), memorybank.RETRIEVAL_WINDOW_TOKENS) + + +def eviction_order(db, adventure) -> list[int]: + """The order `_evict_over_capacity` would take unpinned active memories in.""" + from sqlalchemy import func + return db.execute( + select(models.Memory.id).where( + models.Memory.adventure_id == adventure.id, + models.Memory.forgotten.is_(False), + models.Memory.pinned.is_(False), + ).order_by(func.coalesce(models.Memory.last_used_at, models.Memory.created_at), + models.Memory.use_count) + ).scalars().all() + + +async def diagnose(db, adventure, settings, fact: Fact, plant_depth: int, *, + recall_action: models.Action, embed) -> dict: + """The four stages for `fact`, judged at `recall_action`, the recall turn's AI node. + + Ranking is recomputed with the query that turn used, and checked against the + turn's own stored `memories.used`. Injection is read from that snapshot, so + it reports what the narrator was actually given, not a re-run. + """ + snapshot = recall_action.context_snapshot or {} + out: dict = {"fact_id": fact.fact_id, "plant_depth": plant_depth, + "recall_depth": recall_action.depth} + + covering = covering_memories(db, adventure, plant_depth) + carrying = [m for m in covering if fact.carried_by(m.text)] + elsewhere = [m for m in db.execute(select(models.Memory).where( + models.Memory.adventure_id == adventure.id)).scalars().all() + if fact.carried_by(m.text) and m not in carrying] + creation_input = [] + for memory in covering: + block = memorybank.source_block(db, memory) + raw = "\n\n".join(a.text for a in block) + excerpt = builder.truncate_to_last_tokens(raw, memorybank.MEMORY_EXCERPT_TOKENS) + creation_input.append({ + "memory_id": memory.id, "source_start": memory.source_start, + "source_end": memory.source_end, "block_tokens": builder.count_tokens(raw), + "fact_in_block": fact.carried_by(raw), + "fact_in_summariser_excerpt": fact.carried_by(excerpt), + "memory_text": memory.text, + }) + memory = carrying[0] if carrying else None + out["created"] = { + "yes": memory is not None, + "memory_id": getattr(memory, "id", None), + "source_start": getattr(memory, "source_start", None), + "source_end": getattr(memory, "source_end", None), + "memory_text": getattr(memory, "text", None), + "covering_memories": creation_input, + "no_covering_memory": not covering, + "carried_by_other_memories": [ + {"memory_id": m.id, "source_start": m.source_start, "source_end": m.source_end} + for m in elsewhere], + } + + if memory is None: + out["verdict"] = "not_created" + return out + + order = eviction_order(db, adventure) + active = db.execute(select(models.Memory.id).where( + models.Memory.adventure_id == adventure.id, + models.Memory.forgotten.is_(False))).scalars().all() + on_lineage = db.execute(select(models.Memory.id).where( + models.Memory.id == memory.id, + lineage.path_of(db, adventure).clause(models.Memory))).scalar() is not None + out["retained"] = { + "yes": not memory.forgotten, + "forgotten": memory.forgotten, + "pinned": memory.pinned, + "embedded": memory.embedded, + "on_active_lineage": on_lineage, + "use_count": memory.use_count, + "last_used_at": str(memory.last_used_at) if memory.last_used_at else None, + "created_at": str(memory.created_at), + "active_memories": len(active), + "memory_bank_capacity": settings.memory_bank_capacity, + "eviction_position": (order.index(memory.id) + 1) if memory.id in order else None, + "reason": ("evicted: marked forgotten by capacity eviction" if memory.forgotten + else "active"), + } + if memory.forgotten: + out["verdict"] = "created_but_evicted" + return out + + query = production_query(adventure, recall_action.id) + ranking = await rank_bank(db, adventure, settings, query, embed) + row = next((r for r in ranking["scored"] if r["memory_id"] == memory.id), None) + stored_used = [m.get("id") for m in (snapshot.get("memories") or {}).get("used") or []] + out["ranked"] = { + "yes": row is not None and row["rank"] <= ranking["top_k"], + "eligible": row is not None, + "lexical_score": None, # memory ranking has no lexical term (CONTEXT-AND-MEMORY §20) + "semantic_score": row["similarity"] if row else None, + "final_score": row["similarity"] if row else None, + "pin_effect": "always selected" if memory.pinned else "none", + "rank": row["rank"] if row else None, + "of": len(ranking["scored"]), + "top_k_cutoff": ranking["top_k"], + "selected": bool(row and row["selected"]), + "suppressed_as_duplicate_of": row["suppressed_as_duplicate_of"] if row else None, + "query": query, + "replica_matches_stored_selection": sorted(stored_used) == ranking["selected"], + } + if row is None or row["rank"] > ranking["top_k"] and not row["selected"]: + out["verdict"] = "retained_but_not_ranked" + return out + if not row["selected"]: + out["verdict"] = "ranked_but_not_selected" + return out + + section = _section(snapshot, MEMORIES_LABEL) + injected = memory.id in stored_used and memory.text in section + out["injected"] = { + "yes": injected, + "context_component": MEMORIES_LABEL, + "in_stored_memories_used": memory.id in stored_used, + "text_in_section": memory.text in section, + "token_count": builder.count_tokens(section) if section else 0, + } + out["verdict"] = "injected" if injected else "selected_but_not_injected" + return out + + +# ---------------------------------------------------------------- the stubs + +@dataclass +class BestCaseSummariser: + """The ideal memory writer: F survives if, and only if, F reached it. + + A memory keeps every sentence of the excerpt that carries a planted fact, and + adds one sentence naming the block's own distinct detail so memories differ. + Summary updates never repeat a planted fact, so the summary layer stays out of + the experiment. Every excerpt it was given is kept, for the creation-window + diagnostic. + """ + + facts: tuple[Fact, ...] = (FACT_F, FACT_G) + excerpts: list = field(default_factory=list) + + async def complete(self, system, user, *, temperature=0.3, max_tokens=400): + if "Current story summary:" in user: + return "The travellers kept moving through the country around Westhaven." + excerpt = user.split("Story excerpt:\n\n", 1)[-1].rsplit("\n\nMemory:", 1)[0] + self.excerpts.append(excerpt) + kept = [s.strip() for s in re.split(r"(?<=[.!?])\s+", excerpt) + if any(f.carried_by(s) for f in self.facts)] + detail = re.findall(r"\bat the ([a-z]+ [a-z]+)\b", excerpt.lower()) + tail = f"The travellers spent time at the {detail[-1]}." if detail else \ + "The travellers pressed on." + return " ".join(dict.fromkeys(kept + [tail])) + + +#: Words that mean the same thing to `ConceptEmbedder`. The point is only that a +#: paraphrase lands near the original; the table is the model of that. +CONCEPTS = { + "timepiece": ("sundial", "dial", "hour", "hours", "clock", "timepiece"), + "vessel": ("teapot", "pot", "kettle", "tea", "jar"), + "hid": ("hid", "hide", "hidden", "slipped", "tucked", "put", "stashed"), + "weathervane": ("weathervane", "vane"), + "waterwheel": ("waterwheel", "wheel", "mill"), +} +_WORD_TO_CONCEPT = {w: c for c, words in CONCEPTS.items() for w in words} +DIMENSIONS = 96 + + +@dataclass +class ConceptEmbedder: + """A deterministic embedding: concepts in fixed dimensions, other words hashed.""" + + calls: int = 0 + + async def embed(self, texts): + self.calls += 1 + return [self.vector(t) for t in texts] + + @staticmethod + def vector(text: str) -> list[float]: + v = [0.0] * DIMENSIONS + v[0] = 0.2 # every text shares a little, as real embeddings do + concept_names = list(CONCEPTS) + for word in re.findall(r"[a-z]+", text.lower()): + concept = _WORD_TO_CONCEPT.get(word) + if concept is not None: + v[1 + concept_names.index(concept)] += 3.0 + elif len(word) > 3: + bucket = int(hashlib.sha256(word.encode()).hexdigest(), 16) + v[1 + len(concept_names) + bucket % (DIMENSIONS - 1 - len(concept_names))] += 1.0 + norm = math.sqrt(sum(x * x for x in v)) or 1.0 + return [x / norm for x in v] + + +# ---------------------------------------------------------------- scenarios + +#: Filler places. No word here is in `CONCEPTS`, and none names a planted fact. +PLACES = ( + "north gate", "salt market", "ferry landing", "chapel steps", "rope walk", + "fish stalls", "old bridge", "tanner yard", "lamp street", "weir path", + "grain store", "boat yard", "watch house", "cloth hall", "eel traps", + "sheep fold", "smith forge", "stone quay", "reed beds", "toll booth", +) +PARAPHRASE_QUERY = "I ask Mara where she tucked the little brass dial that tells the hour." +UNRELATED_QUERY = "I ask the ferryman what rope costs at the landing this season." + + +def filler_prose(index: int, words: int) -> str: + """Narration that moves on and never touches a planted fact.""" + place = PLACES[index % len(PLACES)] + sentence = (f"At the {place} the travellers stopped, listened to the gulls over the " + f"grey water, and talked about the long road north.") + reps = max(1, round(words / len(sentence.split()))) + return " ".join([sentence] * reps) + + +@dataclass +class Scenario: + """One deterministic campaign. Depths: the opening is 0, turn *n*'s player + action is 2n-1 and its reply 2n.""" + + name: str + turns: int = 52 + capacity: int = 80 + top_k: int = 5 + budget: int = 4096 + prose_words: int = 60 + plant_turn: int = 1 + recall_text: str = "I ask Mara where she hid the amber sundial." + pin_first_memory: bool = False + lineage_control: bool = False + diagnose_recall: bool = True + + +SCENARIOS = { + "independent_default": Scenario("independent_default"), + "past_capacity": Scenario("past_capacity", capacity=6), + "past_capacity_pinned": Scenario("past_capacity_pinned", capacity=6, pin_first_memory=True), + # Closer to the shipped ratio (memory_top_k 5 against capacity 80): most of + # the bank is not retrieved on a given turn. + "past_capacity_low_top_k": Scenario("past_capacity_low_top_k", capacity=8, top_k=2), + "long_block_fact_early": Scenario("long_block_fact_early", turns=10, prose_words=850, + plant_turn=1, budget=16384), + "long_block_fact_late": Scenario("long_block_fact_late", turns=10, prose_words=850, + plant_turn=3, budget=16384), + "lineage_control": Scenario("lineage_control", lineage_control=True), +} + + +class ScriptNarrator: + """Stands in for the narrator: returns `next_reply`, with an empty state block.""" + + next_reply = "" + last_usage = None + prompts: list = [] + + def __init__(self, *a, **k): + pass + + async def generate(self, parts, *, temperature, max_tokens): + ScriptNarrator.prompts.append((parts.system, parts.story)) + yield ("text", ScriptNarrator.next_reply) + + +def run_scenario(scenario: Scenario) -> dict: + """Plays `scenario` through the real turn route and returns everything measured. + + Uses the database `app.database` is already bound to, creating and dropping + its tables, the way the suite's fixtures do. Patches are applied here and + removed before returning, so this runs the same under pytest and from the CLI. + """ + import asyncio + + from fastapi import Depends + from fastapi.testclient import TestClient + from sqlalchemy.orm import undefer + + from app import auth, limits + from app.database import Base, SessionLocal, engine, get_db + from app.main import app + from app.routers import adventures as adventure_routes + + summariser = BestCaseSummariser() + embedder = ConceptEmbedder() + patches = [ + (memorybank, "summary_provider", lambda s: summariser), + (memorybank, "embedding_provider", lambda s: embedder), + # Post-turn work is settled explicitly after each turn, so eviction + # happens at a known point rather than whenever a background task runs. + (memorybank, "schedule_post_turn", lambda adventure: None), + (adventure_routes.turns, "OpenAICompatibleProvider", ScriptNarrator), + (limits, "check_row_cap", lambda *a, **k: None), + ] + saved = [(obj, name, getattr(obj, name)) for obj, name, _ in patches] + for obj, name, value in patches: + setattr(obj, name, value) + ScriptNarrator.prompts = [] + + Base.metadata.create_all(bind=engine) + memorybank._vector_cache.clear() + with SessionLocal() as db: + user = models.User(is_guest=False, email=f"b1-{scenario.name}@example.com") + db.add(user) + db.flush() + db.add(models.Settings( + user_id=user.id, model="script", endpoint_url="http://127.0.0.1:9/v1", + embedding_model="concept-embed", context_token_budget=scenario.budget, + max_output_tokens=500, memory_bank_capacity=scenario.capacity, + memory_top_k=scenario.top_k, + )) + adventure = models.Adventure( + user_id=user.id, title=f"B.1 {scenario.name}", memory_bank_enabled=True, + auto_summarize=True, persona_name="Aldric", + ) + db.add(adventure) + db.flush() + db.add(models.Action(adventure_id=adventure.id, type="start", + text="Rain over Westhaven, and the tavern door banging in the wind.")) + db.commit() + adv, user_id = adventure.id, user.id + app.dependency_overrides[auth.get_current_user] = ( + lambda db=Depends(get_db): db.get(models.User, user_id) + ) + client = TestClient(app) + result: dict = {"scenario": scenario.__dict__.copy(), "trace": []} + + def call(method, path, body=None, expect=200): + response = client.request(method, f"/api/adventures/{adv}{path}", json=body) + assert response.status_code == expect, (path, response.status_code, response.text[:300]) + return response.json() if response.content else None + + def marks(): + with SessionLocal() as db: + rows = db.execute(select(models.Memory.id, models.Memory.forgotten, + models.Memory.embedded).where( + models.Memory.adventure_id == adv)).all() + summaries_n = db.query(models.Summary).filter_by(adventure_id=adv).count() + return tuple(sorted(rows)), summaries_n + + def settle(): + for _ in range(12): + before = marks() + asyncio.run(memorybank.run_post_turn(adv)) + if marks() == before: + return + + def memories(): + with SessionLocal() as db: + return [dict(row._mapping) for row in db.execute(select( + models.Memory.id, models.Memory.text, models.Memory.source_start, + models.Memory.source_end, models.Memory.forgotten, models.Memory.pinned, + models.Memory.use_count, models.Memory.last_used_at, models.Memory.branch_id, + models.Memory.created_at).where(models.Memory.adventure_id == adv) + .order_by(models.Memory.id)).all()] + + def turn(kind, text, reply): + ScriptNarrator.next_reply = f"{reply}\n```state\n{{\"events\": []}}\n```" + response = client.post(f"/api/adventures/{adv}/actions", json={"type": kind, "text": text}) + assert response.status_code == 200, response.text[:300] + assert '"type": "error"' not in response.text, response.text[-300:] + + plant_depth = None + f_memory_id = None + pinned_id = None + known: dict[int, dict] = {} + g: dict = {} + try: + for n in range(1, scenario.turns + 1): + if n == scenario.plant_turn: + turn("story", FACT_F.sentence, filler_prose(n, scenario.prose_words)) + with SessionLocal() as db: + plant_depth = db.query(models.Action.depth).filter_by( + adventure_id=adv, text=FACT_F.sentence).scalar() + elif scenario.lineage_control and n == 21: + call("POST", "/checkpoints", {"name": "before the mill"}, expect=201) + turn("story", FACT_G.sentence, filler_prose(n, scenario.prose_words)) + with SessionLocal() as db: + g["plant_depth"] = db.query(models.Action.depth).filter_by( + adventure_id=adv, text=FACT_G.sentence).scalar() + elif scenario.lineage_control and n == 30: + # Line A carries G's memory. Mark it, then abandon it: Undo back + # to before G was planted and write something else. + g["line_a"] = call("POST", "/checkpoints", {"name": "line A, after the mill"}, + expect=201)["id"] + with SessionLocal() as db: + g_rows = [m for m in db.execute(select(models.Memory).where( + models.Memory.adventure_id == adv)).scalars() if FACT_G.carried_by(m.text)] + g["memory_ids"] = [m.id for m in g_rows] + with SessionLocal() as db: + g["last_action_id_before_divergence"] = db.query(models.Action.id).filter_by( + adventure_id=adv).order_by(models.Action.id.desc()).limit(1).scalar() + for _ in range(9): + call("POST", "/undo") + turn("do", f"I turn away from the mill and walk to the {PLACES[n % len(PLACES)]}.", + filler_prose(n + 100, scenario.prose_words)) + g["diverged_at_turn"] = n + else: + turn("do", f"I walk on to the {PLACES[n % len(PLACES)]}.", + filler_prose(n, scenario.prose_words)) + settle() + + rows = memories() + created = [r["id"] for r in rows if r["id"] not in known] + newly_forgotten = [r["id"] for r in rows + if r["forgotten"] and not known.get(r["id"], {}).get("forgotten")] + for r in rows: + known[r["id"]] = r + if f_memory_id is None and plant_depth is not None: + for r in rows: + if (r["source_start"] is not None and r["source_start"] <= plant_depth + <= r["source_end"] and FACT_F.carried_by(r["text"])): + f_memory_id = r["id"] + if scenario.pin_first_memory and pinned_id is None: + candidate = next((r for r in rows if r["id"] != f_memory_id), None) + if candidate is not None: + call("PATCH", f"/memories/{candidate['id']}", {"pinned": True}) + pinned_id = candidate["id"] + f_row = known.get(f_memory_id) if f_memory_id else None + result["trace"].append({ + "turn": n, + "active": sum(1 for r in rows if not r["forgotten"]), + "total": len(rows), + "created": created, + "evicted": newly_forgotten, + "created_and_evicted_same_turn": sorted(set(created) & set(newly_forgotten)), + "f_memory_id": f_memory_id, + "f_forgotten": bool(f_row and f_row["forgotten"]), + "f_use_count": f_row["use_count"] if f_row else None, + }) + + turn("do", scenario.recall_text, filler_prose(999, scenario.prose_words)) + + with SessionLocal() as db: + adventure = db.get(models.Adventure, adv) + settings = db.query(models.Settings).filter_by(user_id=user_id).first() + recall_action = (db.query(models.Action) + .filter(models.Action.adventure_id == adv, + models.Action.type == "ai") + .options(undefer(models.Action.context_snapshot)) + .order_by(models.Action.id.desc()).first()) + result["plant_depth"] = plant_depth + result["recall_depth"] = recall_action.depth + result["isolation"] = isolation( + db, adventure, FACT_F, plant_depth, + recall_snapshot=recall_action.context_snapshot, + recall_depth=recall_action.depth) + result["diagnosis"] = asyncio.run(diagnose( + db, adventure, settings, FACT_F, plant_depth, + recall_action=recall_action, embed=embedder.embed)) + result["summariser_excerpts"] = len(summariser.excerpts) + + memory_id = result["diagnosis"]["created"]["memory_id"] + if memory_id is not None and not result["diagnosis"]["retained"]["forgotten"]: + variants = {} + for label, query in (("direct", scenario.recall_text), + ("paraphrase", PARAPHRASE_QUERY), + ("unrelated", UNRELATED_QUERY)): + ranking = asyncio.run(rank_bank(db, adventure, settings, query, embedder.embed)) + row = next((r for r in ranking["scored"] if r["memory_id"] == memory_id), None) + variants[label] = {"query": query, "rank": row and row["rank"], + "of": len(ranking["scored"]), + "similarity": row and row["similarity"], + "selected": bool(row and row["selected"]), + "top_k": ranking["top_k"]} + result["ranking_variants"] = variants + if memory_id is not None: + result["f_first_used_turn"] = next( + (t["turn"] for t in result["trace"] if (t["f_use_count"] or 0) > 0), None) + result["f_last_use_increase_turn"] = max( + (b["turn"] for a, b in zip(result["trace"], result["trace"][1:]) + if (b["f_use_count"] or 0) > (a["f_use_count"] or 0)), default=None) + + evicted_turn = next((t["turn"] for t in result["trace"] if t["f_forgotten"]), None) + first_evictions = next((t["evicted"] for t in result["trace"] if t["evicted"]), []) + result["eviction"] = { + "capacity": scenario.capacity, + "f_evicted_at_turn": evicted_turn, + "f_use_count_when_evicted": next( + (t["f_use_count"] for t in result["trace"] if t["f_forgotten"]), None), + "first_eviction_turn": next( + (t["turn"] for t in result["trace"] if t["evicted"]), None), + "first_evicted_ids": first_evictions, + "f_memory_was_first_evicted": bool(f_memory_id and f_memory_id in first_evictions), + "created_and_evicted_same_turn": sorted( + {i for t in result["trace"] for i in t["created_and_evicted_same_turn"]}), + "pinned_memory_id": pinned_id, + "pinned_memory_forgotten": bool(pinned_id and known[pinned_id]["forgotten"]), + } + + if scenario.lineage_control: + path_clause = lineage.path_of(db, adventure).clause(models.Memory) + stored = db.execute(select(models.Memory.id).where( + models.Memory.id.in_(g.get("memory_ids") or [-1]))).scalars().all() + eligible = db.execute(select(models.Memory.id).where( + models.Memory.id.in_(g.get("memory_ids") or [-1]), path_clause)).scalars().all() + used_after = set() + injected_text = False + # Only turns played after the divergence. Before it, G was on the + # active line, and a memory of it being used then is correct. + for action in (db.query(models.Action) + .filter(models.Action.adventure_id == adv, + models.Action.type == "ai", + models.Action.id > g["last_action_id_before_divergence"]) + .options(undefer(models.Action.context_snapshot))): + snap = action.context_snapshot or {} + for m in (snap.get("memories") or {}).get("used") or []: + if m.get("id") in (g.get("memory_ids") or []): + used_after.add(action.id) + if FACT_G.mentioned_by(_section(snap, MEMORIES_LABEL)): + injected_text = True + g.update(stored=stored, eligible_on_active_line=eligible, + turns_whose_memories_used_named_g=sorted(used_after), + g_text_ever_in_used_memories=injected_text) + if scenario.lineage_control: + call("POST", f"/checkpoints/{g['line_a']}/restore") + with SessionLocal() as db: + adventure = db.get(models.Adventure, adv) + eligible = db.execute(select(models.Memory.id).where( + models.Memory.id.in_(g.get("memory_ids") or [-1]), + lineage.path_of(db, adventure).clause(models.Memory))).scalars().all() + g["eligible_after_returning_to_line_a"] = eligible + result["lineage_control"] = g + return result + finally: + for obj, name, value in saved: + setattr(obj, name, value) + app.dependency_overrides.clear() + adventure_routes.turns._active_turns.clear() + memorybank._vector_cache.clear() + Base.metadata.drop_all(bind=engine) diff --git a/backend/tools/v11_b1_memory.py b/backend/tools/v11_b1_memory.py new file mode 100644 index 0000000..329411e --- /dev/null +++ b/backend/tools/v11_b1_memory.py @@ -0,0 +1,122 @@ +"""v1.1 WP-B.1: run the deterministic memory-retention scenarios, or diagnose a real campaign. + + # the deterministic scenarios, against an isolated database in --out + .venv/bin/python -m tools.v11_b1_memory scenarios --out "$HOME/v11-evidence/b1/