diff --git a/backend/tests/test_m11_long_run_memory.py b/backend/tests/test_m11_long_run_memory.py new file mode 100644 index 0000000..bc2cca7 --- /dev/null +++ b/backend/tests/test_m11_long_run_memory.py @@ -0,0 +1,184 @@ +"""The long run turns the memory bank and the rolling summary on, and proves it. + +M01's step list asks for "summary/memory activation". Both are per-campaign +switches that default to off (`models.Adventure`), and the harness that ran the +first complete hundred-turn campaign never touched them: the bank stayed empty, +no summary was written, and M04's recall succeeded through narrative state alone. +Nothing in that run's evidence said so except a row of zeros nobody was looking +for. + +These tests drive `Run.setup` against the real application in-process, so the +switch is proved by the application accepting it rather than by the harness +sending it. What they cannot prove is that a hundred turns then fill the bank — +that is what the run itself proves, and `memories_in_bank` in its timeline is +where it shows. +""" +import pytest +from fastapi import Depends +from fastapi.testclient import TestClient + +from app import auth, models +from app.database import Base, SessionLocal, engine, get_db +from app.main import app +from tools import m11_long_run as lr + + +UNREACHABLE = "http://127.0.0.1:1/v1" + + +class InProcess: + """`Storyteller.call`, spoken to the application through the test client.""" + + starts = 1 + + def __init__(self, client: TestClient): + self.client = client + + def call(self, method, path, payload=None, timeout=600): + response = self.client.request(method, f"/api{path}", json=payload) + response.raise_for_status() + return response.json() if response.content else None + + +class IgnoresThePatch: + """A server that answers the PATCH and changes nothing, which is exactly the + failure `setup` must refuse rather than record.""" + + starts = 1 + + def __init__(self): + self.calls = [] + + def call(self, method, path, payload=None, timeout=600): + self.calls.append((method, path)) + if path == "/settings": + return {"model": "m", "context_token_budget": 16384, + "model_timeout_seconds": 1800} + if method == "POST" and path == "/adventures": + return {"id": 1} + if method == "GET" and path == "/adventures/1": + return {"memory_bank_enabled": False, "auto_summarize": False} + return {} + + +@pytest.fixture() +def client(monkeypatch): + monkeypatch.setattr(lr, "ENDPOINT", UNREACHABLE) + monkeypatch.setattr(lr, "MODEL", "some-model") + monkeypatch.setattr(lr, "EMBED_MODEL", "some-embedder") + Base.metadata.create_all(bind=engine) + setup = SessionLocal() + user = models.User(is_guest=False, email="m11mem@example.com") + setup.add(user) + setup.commit() + user_id = user.id + setup.close() + + app.dependency_overrides[auth.get_current_user] = ( + lambda db=Depends(get_db): db.get(models.User, user_id) + ) + try: + yield TestClient(app) + finally: + app.dependency_overrides.clear() + Base.metadata.drop_all(bind=engine) + + +@pytest.fixture() +def run_for(tmp_path, monkeypatch): + """A `Run` on the given server. Uploads are skipped: `upload` builds its own + multipart request to a port, and the knowledge library is not what these + tests are about.""" + made = [] + + def make(server): + run = lr.Run(server, tmp_path, turns_target=100) + monkeypatch.setattr(run, "upload", lambda *a, **k: None) + made.append(run) + return run + + yield make + for run in made: + run.timeline.close() + + +def test_a_fresh_campaign_starts_with_both_switches_off(client): + """The premise. If this ever changes, the harness's PATCH is redundant but + harmless; while it holds, a harness without the PATCH measures nothing.""" + created = client.post("/api/adventures", json={"title": "untouched"}).json() + assert created["memory_bank_enabled"] is False + assert created["auto_summarize"] is False + + +def test_setup_leaves_the_campaign_with_memory_and_summary_on(client, run_for): + run = run_for(InProcess(client)) + run.setup() + + stored = client.get(f"/api/adventures/{run.adv}").json() + assert stored["memory_bank_enabled"] is True + assert stored["auto_summarize"] is True + + activated = [e for e in run.events if e["kind"] == "memory_activated"] + assert len(activated) == 1 + assert activated[0]["memory_bank_enabled"] is True + assert activated[0]["auto_summarize"] is True + + +def test_setup_refuses_a_campaign_that_did_not_take_the_switches(run_for): + """Hours of turns against a campaign with the bank off is the run that was + already had. It must stop before the first one, not report silence after.""" + server = IgnoresThePatch() + run = run_for(server) + + with pytest.raises(SystemExit, match="summary/memory clause"): + run.setup() + + # It asked, it read back, and it went no further. + assert ("PATCH", "/adventures/1") in server.calls + assert not any("/state/corrections" in path for _, path in server.calls) + recorded = [e for e in run.events if e["kind"] == "memory_activated"] + assert recorded and recorded[0]["memory_bank_enabled"] is False + + +def test_a_run_without_an_embedding_model_is_refused(monkeypatch, tmp_path, capsys): + """With the bank on and no embedder, memories are written and never + retrieved: `memorybank.retrieve` answers "No embedding model configured". + That is the same unexercised path in a fuller bank, so it is refused before + a server is started or a directory is claimed.""" + monkeypatch.setattr(lr, "ENDPOINT", UNREACHABLE) + monkeypatch.setattr(lr, "MODEL", "some-model") + monkeypatch.setattr(lr, "EMBED_MODEL", "") + out = tmp_path / "never-made" + monkeypatch.setattr("sys.argv", ["m11_long_run", "--out", str(out)]) + + assert lr.main() == 2 + assert "AIDND_TEST_EMBED_MODEL" in capsys.readouterr().out + assert not out.exists() + + +def test_the_bank_is_counted_from_the_application(client, run_for): + run = run_for(InProcess(client)) + run.setup() + assert run.bank_size() == 0 + + db = SessionLocal() + try: + db.add(models.Memory(adventure_id=run.adv, text="the key opens the crypt")) + db.commit() + finally: + db.close() + assert run.bank_size() == 1 + + +def test_a_count_that_cannot_be_read_is_minus_one_not_an_exception(run_for): + """Measurement never fails a turn; -1 is distinguishable from an empty bank.""" + + class Down: + starts = 1 + + def call(self, *a, **k): + raise ConnectionError("gone") + + run = run_for(Down()) + run.adv = 7 + assert run.bank_size() == -1 diff --git a/backend/tools/m11_long_run.py b/backend/tools/m11_long_run.py index a1a9027..bf628ac 100644 --- a/backend/tools/m11_long_run.py +++ b/backend/tools/m11_long_run.py @@ -3,7 +3,7 @@ python -m tools.m11_long_run --turns 100 --out