diff --git a/backend/tests/test_m11_long_run_memory.py b/backend/tests/test_m11_long_run_memory.py new file mode 100644 index 0000000..bc2cca7 --- /dev/null +++ b/backend/tests/test_m11_long_run_memory.py @@ -0,0 +1,184 @@ +"""The long run turns the memory bank and the rolling summary on, and proves it. + +M01's step list asks for "summary/memory activation". Both are per-campaign +switches that default to off (`models.Adventure`), and the harness that ran the +first complete hundred-turn campaign never touched them: the bank stayed empty, +no summary was written, and M04's recall succeeded through narrative state alone. +Nothing in that run's evidence said so except a row of zeros nobody was looking +for. + +These tests drive `Run.setup` against the real application in-process, so the +switch is proved by the application accepting it rather than by the harness +sending it. What they cannot prove is that a hundred turns then fill the bank — +that is what the run itself proves, and `memories_in_bank` in its timeline is +where it shows. +""" +import pytest +from fastapi import Depends +from fastapi.testclient import TestClient + +from app import auth, models +from app.database import Base, SessionLocal, engine, get_db +from app.main import app +from tools import m11_long_run as lr + + +UNREACHABLE = "http://127.0.0.1:1/v1" + + +class InProcess: + """`Storyteller.call`, spoken to the application through the test client.""" + + starts = 1 + + def __init__(self, client: TestClient): + self.client = client + + def call(self, method, path, payload=None, timeout=600): + response = self.client.request(method, f"/api{path}", json=payload) + response.raise_for_status() + return response.json() if response.content else None + + +class IgnoresThePatch: + """A server that answers the PATCH and changes nothing, which is exactly the + failure `setup` must refuse rather than record.""" + + starts = 1 + + def __init__(self): + self.calls = [] + + def call(self, method, path, payload=None, timeout=600): + self.calls.append((method, path)) + if path == "/settings": + return {"model": "m", "context_token_budget": 16384, + "model_timeout_seconds": 1800} + if method == "POST" and path == "/adventures": + return {"id": 1} + if method == "GET" and path == "/adventures/1": + return {"memory_bank_enabled": False, "auto_summarize": False} + return {} + + +@pytest.fixture() +def client(monkeypatch): + monkeypatch.setattr(lr, "ENDPOINT", UNREACHABLE) + monkeypatch.setattr(lr, "MODEL", "some-model") + monkeypatch.setattr(lr, "EMBED_MODEL", "some-embedder") + Base.metadata.create_all(bind=engine) + setup = SessionLocal() + user = models.User(is_guest=False, email="m11mem@example.com") + setup.add(user) + setup.commit() + user_id = user.id + setup.close() + + app.dependency_overrides[auth.get_current_user] = ( + lambda db=Depends(get_db): db.get(models.User, user_id) + ) + try: + yield TestClient(app) + finally: + app.dependency_overrides.clear() + Base.metadata.drop_all(bind=engine) + + +@pytest.fixture() +def run_for(tmp_path, monkeypatch): + """A `Run` on the given server. Uploads are skipped: `upload` builds its own + multipart request to a port, and the knowledge library is not what these + tests are about.""" + made = [] + + def make(server): + run = lr.Run(server, tmp_path, turns_target=100) + monkeypatch.setattr(run, "upload", lambda *a, **k: None) + made.append(run) + return run + + yield make + for run in made: + run.timeline.close() + + +def test_a_fresh_campaign_starts_with_both_switches_off(client): + """The premise. If this ever changes, the harness's PATCH is redundant but + harmless; while it holds, a harness without the PATCH measures nothing.""" + created = client.post("/api/adventures", json={"title": "untouched"}).json() + assert created["memory_bank_enabled"] is False + assert created["auto_summarize"] is False + + +def test_setup_leaves_the_campaign_with_memory_and_summary_on(client, run_for): + run = run_for(InProcess(client)) + run.setup() + + stored = client.get(f"/api/adventures/{run.adv}").json() + assert stored["memory_bank_enabled"] is True + assert stored["auto_summarize"] is True + + activated = [e for e in run.events if e["kind"] == "memory_activated"] + assert len(activated) == 1 + assert activated[0]["memory_bank_enabled"] is True + assert activated[0]["auto_summarize"] is True + + +def test_setup_refuses_a_campaign_that_did_not_take_the_switches(run_for): + """Hours of turns against a campaign with the bank off is the run that was + already had. It must stop before the first one, not report silence after.""" + server = IgnoresThePatch() + run = run_for(server) + + with pytest.raises(SystemExit, match="summary/memory clause"): + run.setup() + + # It asked, it read back, and it went no further. + assert ("PATCH", "/adventures/1") in server.calls + assert not any("/state/corrections" in path for _, path in server.calls) + recorded = [e for e in run.events if e["kind"] == "memory_activated"] + assert recorded and recorded[0]["memory_bank_enabled"] is False + + +def test_a_run_without_an_embedding_model_is_refused(monkeypatch, tmp_path, capsys): + """With the bank on and no embedder, memories are written and never + retrieved: `memorybank.retrieve` answers "No embedding model configured". + That is the same unexercised path in a fuller bank, so it is refused before + a server is started or a directory is claimed.""" + monkeypatch.setattr(lr, "ENDPOINT", UNREACHABLE) + monkeypatch.setattr(lr, "MODEL", "some-model") + monkeypatch.setattr(lr, "EMBED_MODEL", "") + out = tmp_path / "never-made" + monkeypatch.setattr("sys.argv", ["m11_long_run", "--out", str(out)]) + + assert lr.main() == 2 + assert "AIDND_TEST_EMBED_MODEL" in capsys.readouterr().out + assert not out.exists() + + +def test_the_bank_is_counted_from_the_application(client, run_for): + run = run_for(InProcess(client)) + run.setup() + assert run.bank_size() == 0 + + db = SessionLocal() + try: + db.add(models.Memory(adventure_id=run.adv, text="the key opens the crypt")) + db.commit() + finally: + db.close() + assert run.bank_size() == 1 + + +def test_a_count_that_cannot_be_read_is_minus_one_not_an_exception(run_for): + """Measurement never fails a turn; -1 is distinguishable from an empty bank.""" + + class Down: + starts = 1 + + def call(self, *a, **k): + raise ConnectionError("gone") + + run = run_for(Down()) + run.adv = 7 + assert run.bank_size() == -1 diff --git a/backend/tools/m11_long_run.py b/backend/tools/m11_long_run.py index a1a9027..bf628ac 100644 --- a/backend/tools/m11_long_run.py +++ b/backend/tools/m11_long_run.py @@ -3,7 +3,7 @@ python -m tools.m11_long_run --turns 100 --out Run from `backend/`. Reads `AIDND_TEST_ENDPOINT`, `AIDND_TEST_MODEL` and -optionally `AIDND_TEST_EMBED_MODEL`. +`AIDND_TEST_EMBED_MODEL`, all three required. ## Why this is a script that spawns servers rather than a test @@ -26,6 +26,17 @@ at planned points. Everything that survives crosses as bytes on disk. window and then retrieved, so the run measures the window at intervals and the recall check at the end asks the application what it would actually send. +M01's step list also asks for **"summary/memory activation"**, and both are +per-campaign switches that default to off. An earlier version of this harness +never turned them on, so a hundred turns ran with an empty memory bank and no +summaries: six of M01's seven clauses were exercised and the seventh was +reported by silence. Worse for M04, whose whole question is whether a fact +planted at turn one is still reachable at turn a hundred — with the bank off it +was reachable through narrative state alone, and the retrieval path M6 built was +never asked. `setup` now turns both on and proves it, and `measure` records how +many memories and summaries exist so that "the bank stayed empty" is a number in +the evidence rather than an absence nobody looked for. + ## What it records A JSON line per turn (`timeline.jsonl`) carrying the context measurements M03 @@ -401,6 +412,22 @@ class Run: self.adv = created["id"] self.note("campaign", id=self.adv) + # M01 asks for "summary/memory activation". Both are per-campaign and + # both default to False (`models.Adventure`), so a campaign created and + # played without touching them never writes a memory or a summary. + # Enabled here, and read back rather than assumed: a PATCH that silently + # did nothing would leave the same hole this closes. + self.server.call("PATCH", f"/adventures/{self.adv}", + {"memory_bank_enabled": True, "auto_summarize": True}) + back = self.server.call("GET", f"/adventures/{self.adv}") + self.note("memory_activated", + memory_bank_enabled=back.get("memory_bank_enabled"), + auto_summarize=back.get("auto_summarize")) + if not (back.get("memory_bank_enabled") and back.get("auto_summarize")): + raise SystemExit( + "the campaign did not accept memory-bank and auto-summarize; " + "M01's summary/memory clause cannot be measured from this run.") + for name, body, kind in (("canon.md", CANON_MD, "canon"), ("reference.md", REFERENCE_MD, "reference"), ("inspiration.md", INSPIRATION_MD, "inspiration")): @@ -505,6 +532,10 @@ class Run: # campaign resumed against an older build has neither. "history_floor_depth": report["history"].get("floor_depth"), "history_trim_block": report["history"].get("trim_block"), + # M01's summary/memory clause, as a count on every turn. A bank that + # stays at zero is then visible in the evidence while the run is + # still going, instead of being discovered afterwards. + "memories_in_bank": self.bank_size(), "prompt_tokens": tokens["total"], "budget": tokens["budget"], "configured_budget": tokens.get("configured_budget"), @@ -529,6 +560,15 @@ class Run: "clue_in_prompt": CLUE_SENTINEL in json.dumps(report["sections"]), } + def bank_size(self) -> int: + """How many memories the bank holds. Never raises: this is measurement, + and a turn is not worth failing over a count.""" + try: + return len(self.server.call( + "GET", f"/adventures/{self.adv}/memories") or []) + except Exception: # noqa: BLE001 + return -1 + def state(self) -> dict: return self.server.call("GET", f"/adventures/{self.adv}/state") @@ -554,8 +594,12 @@ def main() -> int: help="stop and write the evidence after this many unaccepted turns") args = parser.parse_args() - if not (ENDPOINT and MODEL): - print("set AIDND_TEST_ENDPOINT and AIDND_TEST_MODEL") + if not (ENDPOINT and MODEL and EMBED_MODEL): + # The embedding model is not optional. Without one the summary pass + # still writes memories, but `memorybank.retrieve` answers "No embedding + # model configured" and returns none — so the bank fills and M04's + # memory path is never asked, which is the hole `setup` exists to close. + print("set AIDND_TEST_ENDPOINT, AIDND_TEST_MODEL and AIDND_TEST_EMBED_MODEL") return 2 if not 30 <= args.turn_timeout <= 3600: # The settings schema's own bound, checked here so a mistyped timeout @@ -687,6 +731,8 @@ def main() -> int: "final_measurement": _or_none(run.measure), "db_bytes": db_path.stat().st_size, "bundle_bytes": bundle_bytes, + # Whether the seventh clause of M01's step list actually happened. + "memories_in_bank": _or_none(run.bank_size), } (out / "summary.json").write_text(json.dumps(summary, indent=2, default=str)) print(json.dumps({k: v for k, v in summary.items() @@ -939,6 +985,13 @@ def _recall(run: Run) -> dict: "memories_used": [ m.get("text", "")[:120] for m in (after.get("memories") or {}).get("used", []) ], + # M01's summary/memory clause, stated rather than implied. A recall that + # succeeds only through narrative state, with an empty bank, has proved + # one of the two paths the design has — and the reader of this report + # should be able to see which. + "memories_in_bank": run.bank_size(), + "summary_present": bool( + (server.call("GET", f"/adventures/{adv}") or {}).get("story_summary")), "prompt_tokens": after["tokens"]["total"], }