The first complete hundred-turn campaign did not exercise M01's "summary/memory activation" step. Memory bank and auto-summarize are per-campaign switches that default to off, and m11_long_run never turned them on: summary_tokens and memory_tokens were 0 on every turn, memories_used was empty, and M04's clue was recalled through narrative state alone. The retrieval path M6 built was never asked, and nothing in the evidence said so except a row of zeros. setup now PATCHes both switches on, reads the campaign back, and stops before the first turn if either did not take. memories_in_bank is recorded on every turn, in the final summary and in the recall, and the recall also says whether a summary exists, so which of the two recall paths succeeded is stated rather than implied. The embedding model is now required. Without one the summary pass still writes memories, but memorybank.retrieve answers "No embedding model configured" and returns none -- the same unexercised path in a fuller bank. The harness refuses before it starts a server or claims --out. tests/test_m11_long_run_memory.py drives setup against the real application in-process: the switches are on afterwards, a server that ignores the PATCH is refused before any state is written, the bank count comes from the application and reads -1 rather than raising when it cannot, and a run with no embedding model is refused. The four that exercise setup and the bank count were run against the previous harness and fail there; the premise test (a fresh campaign has both switches off) passes on both, as it should. The 2026-09-10 run in ~/m11-evidence/m01 therefore does not count as M01. It has to be run again on this harness. Backend 1,382 passed, 18 skipped, 0 failed. The eighteenth skip is test_built_spa_fetches_no_fonts_remotely, which wants a built frontend/dist this worktree does not have; it is an environment condition, not a change here. The frontend is untouched. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01XKWHt2DXuvqP83cAk6Zq88
185 lines
6.3 KiB
Python
185 lines
6.3 KiB
Python
"""The long run turns the memory bank and the rolling summary on, and proves it.
|
|
|
|
M01's step list asks for "summary/memory activation". Both are per-campaign
|
|
switches that default to off (`models.Adventure`), and the harness that ran the
|
|
first complete hundred-turn campaign never touched them: the bank stayed empty,
|
|
no summary was written, and M04's recall succeeded through narrative state alone.
|
|
Nothing in that run's evidence said so except a row of zeros nobody was looking
|
|
for.
|
|
|
|
These tests drive `Run.setup` against the real application in-process, so the
|
|
switch is proved by the application accepting it rather than by the harness
|
|
sending it. What they cannot prove is that a hundred turns then fill the bank —
|
|
that is what the run itself proves, and `memories_in_bank` in its timeline is
|
|
where it shows.
|
|
"""
|
|
import pytest
|
|
from fastapi import Depends
|
|
from fastapi.testclient import TestClient
|
|
|
|
from app import auth, models
|
|
from app.database import Base, SessionLocal, engine, get_db
|
|
from app.main import app
|
|
from tools import m11_long_run as lr
|
|
|
|
|
|
UNREACHABLE = "http://127.0.0.1:1/v1"
|
|
|
|
|
|
class InProcess:
|
|
"""`Storyteller.call`, spoken to the application through the test client."""
|
|
|
|
starts = 1
|
|
|
|
def __init__(self, client: TestClient):
|
|
self.client = client
|
|
|
|
def call(self, method, path, payload=None, timeout=600):
|
|
response = self.client.request(method, f"/api{path}", json=payload)
|
|
response.raise_for_status()
|
|
return response.json() if response.content else None
|
|
|
|
|
|
class IgnoresThePatch:
|
|
"""A server that answers the PATCH and changes nothing, which is exactly the
|
|
failure `setup` must refuse rather than record."""
|
|
|
|
starts = 1
|
|
|
|
def __init__(self):
|
|
self.calls = []
|
|
|
|
def call(self, method, path, payload=None, timeout=600):
|
|
self.calls.append((method, path))
|
|
if path == "/settings":
|
|
return {"model": "m", "context_token_budget": 16384,
|
|
"model_timeout_seconds": 1800}
|
|
if method == "POST" and path == "/adventures":
|
|
return {"id": 1}
|
|
if method == "GET" and path == "/adventures/1":
|
|
return {"memory_bank_enabled": False, "auto_summarize": False}
|
|
return {}
|
|
|
|
|
|
@pytest.fixture()
|
|
def client(monkeypatch):
|
|
monkeypatch.setattr(lr, "ENDPOINT", UNREACHABLE)
|
|
monkeypatch.setattr(lr, "MODEL", "some-model")
|
|
monkeypatch.setattr(lr, "EMBED_MODEL", "some-embedder")
|
|
Base.metadata.create_all(bind=engine)
|
|
setup = SessionLocal()
|
|
user = models.User(is_guest=False, email="m11mem@example.com")
|
|
setup.add(user)
|
|
setup.commit()
|
|
user_id = user.id
|
|
setup.close()
|
|
|
|
app.dependency_overrides[auth.get_current_user] = (
|
|
lambda db=Depends(get_db): db.get(models.User, user_id)
|
|
)
|
|
try:
|
|
yield TestClient(app)
|
|
finally:
|
|
app.dependency_overrides.clear()
|
|
Base.metadata.drop_all(bind=engine)
|
|
|
|
|
|
@pytest.fixture()
|
|
def run_for(tmp_path, monkeypatch):
|
|
"""A `Run` on the given server. Uploads are skipped: `upload` builds its own
|
|
multipart request to a port, and the knowledge library is not what these
|
|
tests are about."""
|
|
made = []
|
|
|
|
def make(server):
|
|
run = lr.Run(server, tmp_path, turns_target=100)
|
|
monkeypatch.setattr(run, "upload", lambda *a, **k: None)
|
|
made.append(run)
|
|
return run
|
|
|
|
yield make
|
|
for run in made:
|
|
run.timeline.close()
|
|
|
|
|
|
def test_a_fresh_campaign_starts_with_both_switches_off(client):
|
|
"""The premise. If this ever changes, the harness's PATCH is redundant but
|
|
harmless; while it holds, a harness without the PATCH measures nothing."""
|
|
created = client.post("/api/adventures", json={"title": "untouched"}).json()
|
|
assert created["memory_bank_enabled"] is False
|
|
assert created["auto_summarize"] is False
|
|
|
|
|
|
def test_setup_leaves_the_campaign_with_memory_and_summary_on(client, run_for):
|
|
run = run_for(InProcess(client))
|
|
run.setup()
|
|
|
|
stored = client.get(f"/api/adventures/{run.adv}").json()
|
|
assert stored["memory_bank_enabled"] is True
|
|
assert stored["auto_summarize"] is True
|
|
|
|
activated = [e for e in run.events if e["kind"] == "memory_activated"]
|
|
assert len(activated) == 1
|
|
assert activated[0]["memory_bank_enabled"] is True
|
|
assert activated[0]["auto_summarize"] is True
|
|
|
|
|
|
def test_setup_refuses_a_campaign_that_did_not_take_the_switches(run_for):
|
|
"""Hours of turns against a campaign with the bank off is the run that was
|
|
already had. It must stop before the first one, not report silence after."""
|
|
server = IgnoresThePatch()
|
|
run = run_for(server)
|
|
|
|
with pytest.raises(SystemExit, match="summary/memory clause"):
|
|
run.setup()
|
|
|
|
# It asked, it read back, and it went no further.
|
|
assert ("PATCH", "/adventures/1") in server.calls
|
|
assert not any("/state/corrections" in path for _, path in server.calls)
|
|
recorded = [e for e in run.events if e["kind"] == "memory_activated"]
|
|
assert recorded and recorded[0]["memory_bank_enabled"] is False
|
|
|
|
|
|
def test_a_run_without_an_embedding_model_is_refused(monkeypatch, tmp_path, capsys):
|
|
"""With the bank on and no embedder, memories are written and never
|
|
retrieved: `memorybank.retrieve` answers "No embedding model configured".
|
|
That is the same unexercised path in a fuller bank, so it is refused before
|
|
a server is started or a directory is claimed."""
|
|
monkeypatch.setattr(lr, "ENDPOINT", UNREACHABLE)
|
|
monkeypatch.setattr(lr, "MODEL", "some-model")
|
|
monkeypatch.setattr(lr, "EMBED_MODEL", "")
|
|
out = tmp_path / "never-made"
|
|
monkeypatch.setattr("sys.argv", ["m11_long_run", "--out", str(out)])
|
|
|
|
assert lr.main() == 2
|
|
assert "AIDND_TEST_EMBED_MODEL" in capsys.readouterr().out
|
|
assert not out.exists()
|
|
|
|
|
|
def test_the_bank_is_counted_from_the_application(client, run_for):
|
|
run = run_for(InProcess(client))
|
|
run.setup()
|
|
assert run.bank_size() == 0
|
|
|
|
db = SessionLocal()
|
|
try:
|
|
db.add(models.Memory(adventure_id=run.adv, text="the key opens the crypt"))
|
|
db.commit()
|
|
finally:
|
|
db.close()
|
|
assert run.bank_size() == 1
|
|
|
|
|
|
def test_a_count_that_cannot_be_read_is_minus_one_not_an_exception(run_for):
|
|
"""Measurement never fails a turn; -1 is distinguishable from an empty bank."""
|
|
|
|
class Down:
|
|
starts = 1
|
|
|
|
def call(self, *a, **k):
|
|
raise ConnectionError("gone")
|
|
|
|
run = run_for(Down())
|
|
run.adv = 7
|
|
assert run.bank_size() == -1
|