Files
interactive-story/backend/tests/test_m11_long_run_memory.py
T
JesseMarkowitzandClaude Opus 5 0c7316f951 Keep the state section and its proposal out of the story
The first complete M01 run with the memory bank on (f8d4010, 101 turns on
a GPU host) reported "complete". It still did not prove M04. The planted
clue was found at turn 100 only because the narrator had pasted the
narrative-state section into its own prose, and the paste was still in
recent history. No memory and no summary carried the clue.

The narrator is a small local model. It wrote protocol into its stored
narration on 42 of 104 turns, starting at depth 2, in four shapes:

- a copy of the state section: `Scene:`, `Who and what exists:`, `Held:`,
  `Established:`, `Still open:`
- that copy above a correct ```state block, which was stripped while the
  copy stayed
- the copy, then a bare `State` heading, then a `> {"events": ...}`
  proposal quoted like a player turn, sometimes with story after it
- the same block cut off by the output-token limit, on 10 turns

Stored text is replayed verbatim as history, so each leak also put a second,
older account of the state into the next prompt. That is what M5 review
Finding 4 removed from history replay, and every leak gave the model another
example to copy.

The extractor now removes:

- a pasted state section, recognised by at least two of the renderer's own
  headings as whole lines. The headings are constants in `render.py`, so the
  renderer and the extractor cannot drift apart. One heading alone, or a
  `Scene:` line of prose, is left.
- an unfenced proposal that starts a line, quoted or not, when it parses and
  is a proposal. With no fence it becomes the turn's proposal. A `State`
  heading directly above goes with it. Candidates are taken outermost first,
  so a finished event line inside an unfinished block is never taken as a
  proposal by itself.
- an unfinished unfenced proposal at the end that reads as protocol.
- whatever is left at the end: a `State` heading, a bare `>`, a parroted
  reminder or continue hint (closed or not), and a ```json fence cut off
  before it names its events. These are cut repeatedly until nothing more
  comes off.

This also fixes an older bug. `_STATE_FENCE_RE` read "a ```state block"
inside a parroted reminder as a fence opening and cut out the middle of the
reminder. The label must now end its line or run straight into the payload.

A reply whose only removal is a pasted state section records no raw block,
so the turn is not marked unparseable for a block it never started.

Every AI turn in four real runs was replayed through the new extractor:
draco M01, the two 26-turn GPU trials, and this M01 run. 339 turns in all.
No turn the old extractor had left clean changed. Every leak of our own
protocol is gone: 42 of 42 in this M01 run, 5 in trial 2, 3 on draco.
Trial 1 still has model-invented headings ("Identifiers established:",
"Set of events made true:") on 10 turns. They paraphrase the instruction and
are not our renderer's text, so they are left, not guessed at.

The long-run harness now records an explicit M04 verdict, which is never a
recovery while the clue is still in recent history. It also counts the AI
turns in the export that still carry protocol.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0136VBTMUKWYeU6G9HgbDbND
2026-09-13 21:18:18 -04:00

301 lines
11 KiB
Python

"""The long run turns the memory bank and the rolling summary on, and proves it.
M01's step list asks for "summary/memory activation". Both are per-campaign
switches that default to off (`models.Adventure`), and the harness that ran the
first complete hundred-turn campaign never touched them: the bank stayed empty,
no summary was written, and M04's recall succeeded through narrative state alone.
Nothing in that run's evidence said so except a row of zeros nobody was looking
for.
These tests drive `Run.setup` against the real application in-process, so the
switch is proved by the application accepting it rather than by the harness
sending it. What they cannot prove is that a hundred turns then fill the bank —
that is what the run itself proves, and `memories_in_bank` in its timeline is
where it shows.
"""
import json
import pytest
from fastapi import Depends
from fastapi.testclient import TestClient
from app import auth, models
from app.database import Base, SessionLocal, engine, get_db
from app.main import app
from tools import m11_long_run as lr
UNREACHABLE = "http://127.0.0.1:1/v1"
class InProcess:
"""`Storyteller.call`, spoken to the application through the test client."""
starts = 1
def __init__(self, client: TestClient):
self.client = client
def call(self, method, path, payload=None, timeout=600):
response = self.client.request(method, f"/api{path}", json=payload)
response.raise_for_status()
return response.json() if response.content else None
class IgnoresThePatch:
"""A server that answers the PATCH and changes nothing, which is exactly the
failure `setup` must refuse rather than record."""
starts = 1
def __init__(self):
self.calls = []
def call(self, method, path, payload=None, timeout=600):
self.calls.append((method, path))
if path == "/settings":
return {"model": "m", "context_token_budget": 16384,
"model_timeout_seconds": 1800}
if method == "POST" and path == "/adventures":
return {"id": 1}
if method == "GET" and path == "/adventures/1":
return {"memory_bank_enabled": False, "auto_summarize": False}
return {}
@pytest.fixture()
def client(monkeypatch):
monkeypatch.setattr(lr, "ENDPOINT", UNREACHABLE)
monkeypatch.setattr(lr, "MODEL", "some-model")
monkeypatch.setattr(lr, "EMBED_MODEL", "some-embedder")
Base.metadata.create_all(bind=engine)
setup = SessionLocal()
user = models.User(is_guest=False, email="m11mem@example.com")
setup.add(user)
setup.commit()
user_id = user.id
setup.close()
app.dependency_overrides[auth.get_current_user] = (
lambda db=Depends(get_db): db.get(models.User, user_id)
)
try:
yield TestClient(app)
finally:
app.dependency_overrides.clear()
Base.metadata.drop_all(bind=engine)
@pytest.fixture()
def run_for(tmp_path, monkeypatch):
"""A `Run` on the given server. Uploads are skipped: `upload` builds its own
multipart request to a port, and the knowledge library is not what these
tests are about."""
made = []
def make(server):
run = lr.Run(server, tmp_path, turns_target=100)
monkeypatch.setattr(run, "upload", lambda *a, **k: None)
made.append(run)
return run
yield make
for run in made:
run.timeline.close()
def test_a_fresh_campaign_starts_with_both_switches_off(client):
"""The premise. If this ever changes, the harness's PATCH is redundant but
harmless; while it holds, a harness without the PATCH measures nothing."""
created = client.post("/api/adventures", json={"title": "untouched"}).json()
assert created["memory_bank_enabled"] is False
assert created["auto_summarize"] is False
def test_setup_leaves_the_campaign_with_memory_and_summary_on(client, run_for):
run = run_for(InProcess(client))
run.setup()
stored = client.get(f"/api/adventures/{run.adv}").json()
assert stored["memory_bank_enabled"] is True
assert stored["auto_summarize"] is True
activated = [e for e in run.events if e["kind"] == "memory_activated"]
assert len(activated) == 1
assert activated[0]["memory_bank_enabled"] is True
assert activated[0]["auto_summarize"] is True
def test_setup_refuses_a_campaign_that_did_not_take_the_switches(run_for):
"""Hours of turns against a campaign with the bank off is the run that was
already had. It must stop before the first one, not report silence after."""
server = IgnoresThePatch()
run = run_for(server)
with pytest.raises(SystemExit, match="summary/memory clause"):
run.setup()
# It asked, it read back, and it went no further.
assert ("PATCH", "/adventures/1") in server.calls
assert not any("/state/corrections" in path for _, path in server.calls)
recorded = [e for e in run.events if e["kind"] == "memory_activated"]
assert recorded and recorded[0]["memory_bank_enabled"] is False
def test_a_run_without_an_embedding_model_is_refused(monkeypatch, tmp_path, capsys):
"""With the bank on and no embedder, memories are written and never
retrieved: `memorybank.retrieve` answers "No embedding model configured".
That is the same unexercised path in a fuller bank, so it is refused before
a server is started or a directory is claimed."""
monkeypatch.setattr(lr, "ENDPOINT", UNREACHABLE)
monkeypatch.setattr(lr, "MODEL", "some-model")
monkeypatch.setattr(lr, "EMBED_MODEL", "")
out = tmp_path / "never-made"
monkeypatch.setattr("sys.argv", ["m11_long_run", "--out", str(out)])
assert lr.main() == 2
assert "AIDND_TEST_EMBED_MODEL" in capsys.readouterr().out
assert not out.exists()
def test_the_bank_is_counted_from_the_application(client, run_for):
run = run_for(InProcess(client))
run.setup()
assert run.bank_size() == 0
db = SessionLocal()
try:
db.add(models.Memory(adventure_id=run.adv, text="the key opens the crypt"))
db.commit()
finally:
db.close()
assert run.bank_size() == 1
def test_a_count_that_cannot_be_read_is_minus_one_not_an_exception(run_for):
"""Measurement never fails a turn; -1 is distinguishable from an empty bank."""
class Down:
starts = 1
def call(self, *a, **k):
raise ConnectionError("gone")
run = run_for(Down())
run.adv = 7
assert run.bank_size() == -1
# ----------------------------------------------- failed post-turn work stops a run
class Reports:
"""A server whose derived status, summaries and log say what the test sets."""
starts = 1
def __init__(self, log_path, *, status=None, summaries=0, memories=0):
self.log_path = log_path
self.status = status or []
self.summaries = summaries
self.memories = memories
def call(self, method, path, payload=None, timeout=600):
if path.endswith("/derived"):
return {"status": self.status,
"failing": [r["kind"] for r in self.status if r["status"] == "failed"],
"summaries": [{"id": i} for i in range(self.summaries)]}
if path.endswith("/memories"):
return [{"id": i} for i in range(self.memories)]
return {}
def test_a_failed_pass_in_derived_status_stops_the_run(run_for, tmp_path):
server = Reports(tmp_path / "server.log", status=[
{"kind": "summary", "status": "failed", "detail": "ProviderError: gone"},
{"kind": "memory", "status": "idle", "detail": ""},
])
run = run_for(server)
run.adv = 1
found = run.background_failures()
assert found == ["summary: ProviderError: gone"]
def test_a_failure_the_application_could_not_record_is_found_in_the_log_once(run_for, tmp_path):
"""The failure that hid the first GPU trial: derived status said `idle` and
the only record was in the server log."""
log = tmp_path / "server.log"
log.write_text("INFO: 200 OK\nERROR:app.memorybank:could not record derived-work failure for 1\n")
run = run_for(Reports(log))
run.adv = 1
assert len(run.background_failures()) == 1
assert run.background_failures() == [], "the same line was reported twice"
with log.open("a") as handle:
handle.write("ERROR:app.derived:derived summary work failed for adventure 1\n")
assert len(run.background_failures()) == 1
def test_healthy_status_and_a_quiet_log_find_nothing(run_for, tmp_path):
log = tmp_path / "server.log"
log.write_text('INFO: "POST /api/adventures/1/actions HTTP/1.1" 200 OK\n')
run = run_for(Reports(log, status=[{"kind": "memory", "status": "ok", "detail": ""}]))
run.adv = 1
assert run.background_failures() == []
def test_the_log_position_survives_a_resume(run_for, tmp_path):
"""Otherwise a resumed run would find the failure that stopped it again, and
stop again, however healthy the application now is."""
first = run_for(Reports(tmp_path / "server.log"))
first.adv, first.log_offset = 1, 4096
first.save_resume()
second = run_for(Reports(tmp_path / "server.log"))
second.adopt(json.loads((tmp_path / lr.RESUME_FILE).read_text()))
assert second.log_offset == 4096
def test_a_run_with_no_summary_or_no_memory_is_not_complete(run_for, tmp_path):
log = tmp_path / "server.log"
assert "summaries=0" in lr._activation_shortfall(
_with_adv(run_for(Reports(log, memories=3, summaries=0))))
assert "memories_in_bank=0" in lr._activation_shortfall(
_with_adv(run_for(Reports(log, memories=0, summaries=2))))
assert lr._activation_shortfall(
_with_adv(run_for(Reports(log, memories=3, summaries=1)))) is None
def _with_adv(run):
run.adv = 1
return run
# ------------------------------------------------------- what M04 actually proved
def test_the_m04_verdict_never_credits_a_clue_still_in_recent_history():
"""The first long run with the bank on found the clue at turn 100 only
because the narrator had pasted the state into recent history."""
base = {"clue_in_recent_history_window": False, "in_memories_section": False,
"in_summary_section": False, "in_state_section": False}
assert lr._m04_verdict({**base, "clue_in_recent_history_window": True,
"in_memories_section": True}) == "precondition_not_met"
assert lr._m04_verdict({**base, "in_memories_section": True}) == \
"recovered_through_memory_or_summary"
assert lr._m04_verdict({**base, "in_summary_section": True}) == \
"recovered_through_memory_or_summary"
assert lr._m04_verdict({**base, "in_state_section": True}) == \
"recovered_through_state_only"
assert lr._m04_verdict(base) == "not_recovered"
def test_protocol_left_in_stored_narration_is_counted():
bundle = {"actions": [
{"id": 1, "type": "do", "text": '> You say {"events": []}'},
{"id": 2, "type": "ai", "text": "The rain eases."},
{"id": 3, "type": "ai", "text": "Beat.\n\nWho and what exists:\n mara: Mara"},
{"id": 4, "type": "ai", "text": 'Beat.\n\n> {"events": []}'},
]}
assert lr._protocol_leaks(bundle) == {
"ai_actions": 3, "leaking": 2, "example_ids": [3, 4]}