The first M01 trial with the memory bank on was 26 turns on a GPU host. It
accepted every turn and reported "complete". It also wrote two memories and
no summary, and logged 180 `database is locked` errors, while derived status
still read `idle`.
The cause was a single uncommitted UPDATE. Retrieval bumped each used
memory's counter before the model call, and the turn commits only after the
reply has streamed. SQLite has one writer, so the turn held the write lock for
the whole reply. Every post-turn memory, summary and status write in that
window waited out the five-second timeout and failed. Recording the failure
needed a write as well, and without a rollback first it raised
PendingRollbackError. The loss therefore reached the log and never reached
the status the Insights panel reads, which F08 forbids. The draco run never
hit this because the bank was off there.
- `retrieve_memories` now only reads. `record_use` writes the counters in the
turn's single commit, so a turn that never lands counts nothing.
- The post-turn task's outer handler rolls back before it records a failure.
The harness could not have caught any of this. It read three prompt sections
under names the builder does not use: `memories` (really `used_memories`),
`story_history` (really `history`/`recent_history`), and a `knowledge` prefix
that matched the fixed instruction section instead of the imported passages.
Memory tokens read 0 whatever the prompt held, and the in-history and
in-memories recall checks could never come out true. The labels are now
constants, pinned by a test against a prompt the real builder assembled.
The harness also stops at the first sign of failed post-turn work. It checks
/derived and new server.log lines after every turn, keeps its log position
across --resume, and waits for background work to settle before its final
checks. A run with no memories or no summaries now ends "failed", not
"complete".
Both new application tests fail on fec46f6: the lock probe sees
`database is locked`, and memory status stays `idle`. The full backend suite
passes (1392 passed, 17 skipped). A 26-turn re-run against the same host had
0 lock errors, wrote 7 memories and 2 summaries, and used them in the prompt
from turn 8.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0136VBTMUKWYeU6G9HgbDbND
958 lines
43 KiB
Python
958 lines
43 KiB
Python
"""M6: branch-safe context, summaries and long-term story memory.
|
|
|
|
The acceptance contract for this milestone is F01-F08 plus the E-series lineage
|
|
tests that own the memory and summary consequences of branching. Each test below
|
|
names the criterion it carries.
|
|
|
|
Two things are asserted throughout rather than assumed:
|
|
|
|
* **The assembled prompt, not the narration.** A model that fails to mention a
|
|
leaked memory is not evidence that the memory did not leak, so every leak test
|
|
reads the context the builder actually produced.
|
|
* **The lineage chokepoint, not a reimplementation.** Memories and summaries are
|
|
filtered by `lineage.Path.clause`, the same clause every read of the story
|
|
goes through. A test that walked the tree itself could pass while the product
|
|
leaked.
|
|
|
|
python -m pytest tests/test_context_memory.py -v
|
|
"""
|
|
|
|
import asyncio
|
|
import sqlite3
|
|
|
|
import pytest
|
|
from fastapi import Depends
|
|
from fastapi.testclient import TestClient
|
|
from sqlalchemy import select
|
|
|
|
from app import auth, derived, limits, memorybank, models, summaries
|
|
from app.context import builder, lineage
|
|
from app.database import DB_PATH, Base, SessionLocal, engine, get_db
|
|
from app.knowledge import classes
|
|
from app.main import app
|
|
from app.providers import ProviderError
|
|
from app.routers import adventures
|
|
|
|
from fakes import ScriptedProvider, state_block
|
|
from tools import m11_long_run
|
|
|
|
|
|
class StubEmbedder:
|
|
"""A deterministic embedder. Distinct texts get distinguishable vectors."""
|
|
|
|
def __init__(self):
|
|
self.calls = 0
|
|
|
|
async def embed(self, texts):
|
|
self.calls += 1
|
|
out = []
|
|
for text in texts:
|
|
lowered = text.lower()
|
|
out.append([
|
|
1.0,
|
|
1.0 if "ledger" in lowered or "flagstone" in lowered else 0.0,
|
|
1.0 if "chapel" in lowered else 0.0,
|
|
])
|
|
return out
|
|
|
|
|
|
class StubSummariser:
|
|
"""Stands in for the summariser so this file opens no sockets."""
|
|
|
|
async def complete(self, system, user, *, max_tokens=600):
|
|
return "A summary of what has happened so far."
|
|
|
|
|
|
@pytest.fixture()
|
|
def client(monkeypatch):
|
|
Base.metadata.create_all(bind=engine)
|
|
memorybank._vector_cache.clear()
|
|
setup = SessionLocal()
|
|
user = models.User(is_guest=False, email="m6@example.com")
|
|
setup.add(user)
|
|
setup.flush()
|
|
setup.add(models.Settings(
|
|
user_id=user.id, api_key="enc:dummy", model="test-model",
|
|
embedding_model="embed-test", context_token_budget=4000,
|
|
max_output_tokens=400, memory_top_k=3,
|
|
))
|
|
adventure = models.Adventure(
|
|
user_id=user.id, title="M6", memory_bank_enabled=True, auto_summarize=True,
|
|
)
|
|
setup.add(adventure)
|
|
setup.flush()
|
|
setup.add(models.Action(adventure_id=adventure.id, type="start",
|
|
text="The road forks at the Crooked Lantern."))
|
|
setup.commit()
|
|
adv_id, user_id = adventure.id, user.id
|
|
setup.close()
|
|
|
|
monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None)
|
|
monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", ScriptedProvider)
|
|
# Both derived providers are stubbed, not just the embedder (M6 review
|
|
# finding M6-F3). With only the embedder replaced, the post-turn pass built
|
|
# a real summariser against the default endpoint and every turn in this file
|
|
# opened a socket to localhost:11434 — slow, dependent on what happens to be
|
|
# listening, and the source of an abandoned-coroutine RuntimeWarning when
|
|
# the TestClient event loop closed under it. Tests that deliberately
|
|
# exercise real provider construction live in `test_provider_wiring.py`.
|
|
monkeypatch.setattr(memorybank, "embedding_provider", lambda s: StubEmbedder())
|
|
monkeypatch.setattr(memorybank, "summary_provider", lambda s: StubSummariser())
|
|
app.dependency_overrides[auth.get_current_user] = (
|
|
lambda db=Depends(get_db): db.get(models.User, user_id)
|
|
)
|
|
test_client = TestClient(app)
|
|
test_client.adv_id = adv_id
|
|
test_client.user_id = user_id
|
|
try:
|
|
yield test_client
|
|
finally:
|
|
app.dependency_overrides.clear()
|
|
memorybank._vector_cache.clear()
|
|
Base.metadata.drop_all(bind=engine)
|
|
|
|
|
|
# ----------------------------------------------------------------- helpers
|
|
|
|
def play(client, text, prose="The road bends onward past the treeline.", events=None):
|
|
ScriptedProvider.replies = [f"{prose}\n{state_block(events or [])}"]
|
|
r = client.post(f"/api/adventures/{client.adv_id}/actions",
|
|
json={"type": "do", "text": text})
|
|
assert r.status_code == 200, r.text[:300]
|
|
assert '"error"' not in r.text, r.text[:300]
|
|
|
|
|
|
def head_of(client):
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
return adventure.head_branch_id, adventure.head_depth
|
|
|
|
|
|
def context_report(client) -> dict:
|
|
"""The prompt the app would send now, assembled through the real builder."""
|
|
r = client.get(f"/api/adventures/{client.adv_id}/context")
|
|
assert r.status_code == 200, r.text[:300]
|
|
return r.json()
|
|
|
|
|
|
def prompt_text(report: dict) -> str:
|
|
return "\n".join(s["text"] for s in report["sections"])
|
|
|
|
|
|
def plant_memory(client, text, *, authority=None, at_depth=None):
|
|
"""Attaches an embedded memory to a live node, as the real pass would."""
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
depth = adventure.head_depth if at_depth is None else at_depth
|
|
node = db.execute(
|
|
select(models.Action).where(
|
|
models.Action.adventure_id == client.adv_id,
|
|
lineage.path_of(db, adventure).uncapped().clause(models.Action),
|
|
models.Action.depth == depth,
|
|
)
|
|
).scalars().first()
|
|
assert node is not None, f"no live node at depth {depth}"
|
|
memory = models.Memory(
|
|
adventure_id=client.adv_id, text=text,
|
|
branch_id=node.branch_id, depth=node.depth,
|
|
source_start=node.depth, source_end=node.depth,
|
|
authority=authority or memorybank.classify_authority(text),
|
|
)
|
|
memorybank.set_vector(memory, asyncio.run(StubEmbedder().embed([text]))[0])
|
|
db.add(memory)
|
|
db.commit()
|
|
return memory.id
|
|
|
|
|
|
def eligible_memory_texts(client) -> list[str]:
|
|
"""What the retrieval filter would consider, through the real clause."""
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
return list(db.execute(
|
|
select(models.Memory.text).where(
|
|
models.Memory.adventure_id == client.adv_id,
|
|
lineage.path_of(db, adventure).clause(models.Memory),
|
|
models.Memory.forgotten.is_(False),
|
|
)
|
|
).scalars().all())
|
|
|
|
|
|
# --------------------------------------------------------------------- F01
|
|
|
|
def test_f01_recent_turns_stay_in_the_prompt(client):
|
|
"""F01. The immediately preceding turns are what conversational coherence
|
|
is made of, so they have to actually be there."""
|
|
play(client, "ask Mara about the key", prose="Mara turns the silver key over.")
|
|
play(client, "wait for her answer", prose="'I found it at the chapel,' she says.")
|
|
|
|
story = prompt_text(context_report(client))
|
|
|
|
assert "Mara turns the silver key over." in story
|
|
assert "'I found it at the chapel,' she says." in story
|
|
assert "ask Mara about the key" in story
|
|
|
|
|
|
# --------------------------------------------------------------------- F02
|
|
|
|
def test_f02_an_old_clue_survives_outside_recent_history(client):
|
|
"""F02. A distinctive clue is planted, the story runs on past it, and the
|
|
clue comes back through memory rather than through the whole transcript."""
|
|
play(client, "search the floor",
|
|
prose="Aldric pries up the third flagstone and hides the ledger beneath it.")
|
|
plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
|
|
for i in range(22):
|
|
play(client, f"walk on {i}", prose=f"[{i}] " + "The road runs on. " * 60)
|
|
|
|
report = context_report(client)
|
|
story = prompt_text(report)
|
|
|
|
# It has fallen out of the verbatim history.
|
|
history_text = "\n".join(
|
|
s["text"] for s in report["sections"]
|
|
if s["label"] in ("history", "recent_history")
|
|
)
|
|
assert "third flagstone" not in history_text, (
|
|
"the fixture did not push the clue out of recent history"
|
|
)
|
|
# But it is still available to the narrator, through memory.
|
|
assert "third flagstone" in story
|
|
assert any("flagstone" in m["text"] for m in report["memories"]["used"])
|
|
# And not by sending the whole story.
|
|
assert report["history"]["included"] < report["history"]["total"]
|
|
|
|
|
|
# --------------------------------------------------------------------- F03
|
|
|
|
def test_f03_the_prompt_stays_bounded_as_the_story_grows(client):
|
|
"""F03. Input must not grow with the transcript."""
|
|
play(client, "begin", prose="The road bends. " * 40)
|
|
for i in range(6):
|
|
play(client, f"on {i}", prose=f"[{i}] " + "The road bends. " * 40)
|
|
short = context_report(client)
|
|
for i in range(24):
|
|
play(client, f"further {i}", prose=f"[{i}] " + "The road bends. " * 40)
|
|
long = context_report(client)
|
|
|
|
assert long["history"]["total"] > short["history"]["total"] * 2, "fixture too small"
|
|
budget = long["tokens"]["budget"]
|
|
assert long["tokens"]["total"] <= budget
|
|
# Four times the story must not be four times the prompt.
|
|
assert long["tokens"]["total"] < short["tokens"]["total"] * 2
|
|
|
|
|
|
# --------------------------------------------------------------------- F04
|
|
|
|
def test_f04_the_reply_budget_is_reserved(client):
|
|
"""F04. The configured reply length stays available whatever the story."""
|
|
for i in range(20):
|
|
play(client, f"on {i}", prose=f"[{i}] " + "The road bends. " * 40)
|
|
|
|
report = context_report(client)
|
|
with SessionLocal() as db:
|
|
settings = db.query(models.Settings).filter_by(user_id=client.user_id).first()
|
|
max_output = settings.max_output_tokens
|
|
|
|
assert report["tokens"]["output_reserve"] >= max_output
|
|
assert report["tokens"]["total"] + max_output <= report["tokens"]["budget"], (
|
|
"the assembled input left no room for the reply"
|
|
)
|
|
|
|
|
|
def test_f04_a_budget_too_small_for_the_reply_is_refused(client):
|
|
"""Section 9: fail clearly rather than build a prompt known to overflow."""
|
|
play(client, "begin")
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
settings = db.query(models.Settings).filter_by(user_id=client.user_id).first()
|
|
settings.context_token_budget = 200
|
|
settings.max_output_tokens = 4000
|
|
db.commit()
|
|
with pytest.raises(builder.ContextOverflow) as exc:
|
|
builder.build_context(adventure, settings)
|
|
# The message has to say what to change.
|
|
assert "context budget" in str(exc.value)
|
|
assert "reserved for the reply" in str(exc.value)
|
|
|
|
|
|
def test_f04_an_impossible_budget_fails_the_turn_without_losing_the_story(client):
|
|
"""The refusal reaches the reader as a failed turn, not a 500."""
|
|
play(client, "begin", prose="The lantern swings.")
|
|
with SessionLocal() as db:
|
|
settings = db.query(models.Settings).filter_by(user_id=client.user_id).first()
|
|
settings.context_token_budget = 200
|
|
settings.max_output_tokens = 4000
|
|
db.commit()
|
|
|
|
ScriptedProvider.replies = ["should never be reached"]
|
|
r = client.post(f"/api/adventures/{client.adv_id}/actions",
|
|
json={"type": "do", "text": "carry on"})
|
|
assert r.status_code == 200
|
|
assert "context budget" in r.text
|
|
# The story that already existed is untouched.
|
|
actions = client.get(f"/api/adventures/{client.adv_id}").json()["actions"]
|
|
assert any("The lantern swings." in a["text"] for a in actions)
|
|
|
|
|
|
# --------------------------------------------------------------------- F05
|
|
|
|
def test_f05_the_inspector_shows_every_component_m6_owns(client):
|
|
"""F05, for the components this milestone owns."""
|
|
play(client, "begin", prose="Aldric sets the key down.",
|
|
events=[{"type": "create_entity", "entity": "aldric",
|
|
"entity_type": "character", "name": "Aldric"}])
|
|
plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
summaries.record(db, adventure, "The party reached the Crooked Lantern.")
|
|
db.commit()
|
|
play(client, "carry on")
|
|
|
|
report = context_report(client)
|
|
labels = {s["label"] for s in report["sections"]}
|
|
|
|
assert "narrator" in labels, "narrator/system rules"
|
|
assert "narrative_state" in labels, "current authoritative state"
|
|
assert "story_summary" in labels, "the summary used"
|
|
assert "used_memories" in labels, "retrieved memories"
|
|
assert "history" in labels, "recent history"
|
|
# Model and settings.
|
|
assert report["settings"]["model"] == "test-model"
|
|
assert report["settings"]["max_output_tokens"] == 400
|
|
# Token accounting, per component and in total.
|
|
assert all(isinstance(s["tokens"], int) for s in report["sections"])
|
|
for key in ("total", "budget", "output_reserve", "protected", "available_for_history"):
|
|
assert key in report["tokens"], key
|
|
# Summary provenance.
|
|
assert report["summary"]["depth"] is not None
|
|
# Derived-work health.
|
|
assert isinstance(report["derived"], list)
|
|
|
|
|
|
# --------------------------------------------------------------------- F06
|
|
|
|
def test_f06_a_retrieved_memory_is_traceable_to_its_source(client):
|
|
"""F06. "Where did this memory come from?" must be answerable."""
|
|
play(client, "search the floor", prose="Aldric hides the ledger.")
|
|
memory_id = plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
|
|
play(client, "carry on")
|
|
|
|
used = context_report(client)["memories"]["used"]
|
|
entry = next(m for m in used if m["id"] == memory_id)
|
|
|
|
assert entry["source"]["branch_id"] is not None
|
|
assert entry["source"]["depth"] is not None
|
|
assert entry["source"]["source_start"] is not None
|
|
# And the coordinate names a real node of this campaign's accepted history.
|
|
with SessionLocal() as db:
|
|
node = db.execute(
|
|
select(models.Action).where(
|
|
models.Action.adventure_id == client.adv_id,
|
|
models.Action.branch_id == entry["source"]["branch_id"],
|
|
models.Action.depth == entry["source"]["depth"],
|
|
)
|
|
).scalars().first()
|
|
assert node is not None, "the memory's provenance points at no action"
|
|
|
|
|
|
# --------------------------------------------------------------------- F07
|
|
|
|
def test_f07_a_heuristic_memory_is_labelled_and_is_not_state(client):
|
|
"""F07. An inference may be recalled; it may not become canon."""
|
|
play(client, "watch her", prose="Mara glances at the door.",
|
|
events=[{"type": "create_entity", "entity": "mara",
|
|
"entity_type": "character", "name": "Mara"}])
|
|
plant_memory(client, "Mara seemed nervous around Captain Vale.")
|
|
play(client, "carry on")
|
|
|
|
report = context_report(client)
|
|
used = report["memories"]["used"]
|
|
entry = next(m for m in used if "Captain Vale" in m["text"])
|
|
assert entry["authority"] == "heuristic"
|
|
|
|
story = prompt_text(report)
|
|
assert "[inferred]" in story, "the prompt does not mark the inference"
|
|
assert "interpretation, not established fact" in story
|
|
|
|
# And it did not become authoritative state.
|
|
document = client.get(f"/api/adventures/{client.adv_id}/state").json()["document"]
|
|
facts = [f["predicate"] for f in document["facts"]]
|
|
assert not any("Vale" in f for f in facts), "a heuristic memory became a fact"
|
|
|
|
|
|
def test_the_application_classifies_authority_not_the_model(client):
|
|
"""The classifier is the application's, and it is inspectable."""
|
|
assert memorybank.classify_authority(
|
|
"Aldric promised Mara he would return before dawn.") == "accepted_story"
|
|
assert memorybank.classify_authority(
|
|
"Mara seemed uneasy when Captain Vale was mentioned.") == "heuristic"
|
|
|
|
|
|
# --------------------------------------------------------------------- F08
|
|
|
|
def test_f08_a_failing_memory_pass_keeps_the_story_and_is_visible(client):
|
|
"""F08. Derived work fails softly, and audibly."""
|
|
play(client, "begin", prose="The lantern swings.",
|
|
events=[{"type": "create_entity", "entity": "aldric",
|
|
"entity_type": "character", "name": "Aldric"}])
|
|
before_state = client.get(f"/api/adventures/{client.adv_id}/state").json()["document"]
|
|
|
|
class Broken:
|
|
async def complete(self, *a, **k):
|
|
raise ProviderError("the summariser is unreachable")
|
|
|
|
async def embed(self, texts):
|
|
raise ProviderError("the embedder is unreachable")
|
|
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
# Enough uncovered story that the memory pass is genuinely due.
|
|
for depth in range(20):
|
|
db.add(models.Action(adventure_id=adventure.id, type="do",
|
|
text=f"filler {depth}"))
|
|
db.commit()
|
|
# Read the head *after* the fixture's own writes, so what this test measures
|
|
# is the effect of the failing derived pass and nothing else.
|
|
before_head = head_of(client)
|
|
|
|
import app.memorybank as mb
|
|
real_summary, real_embed = mb.summary_provider, mb.embedding_provider
|
|
mb.summary_provider = lambda s: Broken()
|
|
mb.embedding_provider = lambda s: Broken()
|
|
try:
|
|
asyncio.run(mb.run_post_turn(client.adv_id))
|
|
finally:
|
|
mb.summary_provider, mb.embedding_provider = real_summary, real_embed
|
|
|
|
# The accepted story, its state and the head all survived.
|
|
actions = client.get(f"/api/adventures/{client.adv_id}").json()["actions"]
|
|
assert any("The lantern swings." in a["text"] for a in actions)
|
|
after_state = client.get(f"/api/adventures/{client.adv_id}/state").json()["document"]
|
|
assert after_state["entities"].keys() == before_state["entities"].keys()
|
|
assert head_of(client) == before_head
|
|
|
|
# The failure is findable.
|
|
status = client.get(f"/api/adventures/{client.adv_id}/derived").json()
|
|
assert "memory" in status["failing"], status
|
|
detail = next(r for r in status["status"] if r["kind"] == "memory")
|
|
assert "unreachable" in detail["detail"]
|
|
assert detail["failures"] >= 1
|
|
|
|
# And the story continues.
|
|
play(client, "carry on", prose="The door opens.")
|
|
assert any("The door opens." in a["text"]
|
|
for a in client.get(f"/api/adventures/{client.adv_id}").json()["actions"])
|
|
|
|
|
|
def test_f08_a_recovered_pass_clears_the_failure(client):
|
|
"""Derived work can be retried: the next healthy run clears the record."""
|
|
with SessionLocal() as db:
|
|
derived.failed(db, client.adv_id, derived.SUMMARY,
|
|
ProviderError("the summariser is unreachable"))
|
|
db.commit()
|
|
assert client.get(f"/api/adventures/{client.adv_id}/derived").json()["failing"] \
|
|
== ["summary"]
|
|
|
|
with SessionLocal() as db:
|
|
derived.succeeded(db, client.adv_id, derived.SUMMARY)
|
|
db.commit()
|
|
|
|
status = client.get(f"/api/adventures/{client.adv_id}/derived").json()
|
|
assert status["failing"] == []
|
|
row = next(r for r in status["status"] if r["kind"] == "summary")
|
|
assert row["status"] == "ok" and row["failures"] == 0
|
|
|
|
|
|
# ------------------------------------------------------- E02 / E03 lineage
|
|
#
|
|
# The memory half of this was already correct at the M5 baseline: memories carry
|
|
# a `(branch_id, depth)` coordinate and retrieval filters them through the
|
|
# capped lineage. These tests pin that behaviour so a later change cannot lose
|
|
# it. The summary half was not: before M6 the rolling summary was one column
|
|
# with no coordinate, and it leaked across a divergence. That is what
|
|
# `app/summaries.py` fixes, and what E03 below measures.
|
|
|
|
SECRET_A = "Aldric hid the ledger beneath the third flagstone."
|
|
SECRET_B = "The party swore an oath in the drowned chapel."
|
|
|
|
|
|
def test_e02_the_ten_step_memory_negative_control(client):
|
|
"""E02, exactly as the milestone brief numbers it."""
|
|
# 1-2. Establish the fact and let a memory be made from it.
|
|
play(client, "search the floor", prose="Aldric pries up the flagstone.")
|
|
plant_memory(client, SECRET_A)
|
|
|
|
# 3. Retrievable on that valid line.
|
|
assert SECRET_A in eligible_memory_texts(client)
|
|
assert SECRET_A in prompt_text(context_report(client))
|
|
|
|
# 4-5. Undo to before it: no longer eligible.
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
assert SECRET_A not in eligible_memory_texts(client)
|
|
assert SECRET_A not in prompt_text(context_report(client))
|
|
|
|
# 6-7. Redo: eligible again, and no re-embedding was needed.
|
|
client.post(f"/api/adventures/{client.adv_id}/redo")
|
|
client.post(f"/api/adventures/{client.adv_id}/redo")
|
|
assert SECRET_A in eligible_memory_texts(client)
|
|
with SessionLocal() as db:
|
|
assert db.execute(
|
|
select(models.Memory.embedded).where(
|
|
models.Memory.adventure_id == client.adv_id)
|
|
).scalars().first() is True, "the memory was re-embedded rather than reused"
|
|
|
|
# 8-9. Undo again and diverge onto a new continuation.
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
play(client, "take the other road", prose="A different road opens.")
|
|
|
|
# 10. Still stored, never in the active prompt.
|
|
with SessionLocal() as db:
|
|
assert db.query(models.Memory).filter_by(adventure_id=client.adv_id).count() == 1
|
|
assert SECRET_A not in eligible_memory_texts(client)
|
|
assert SECRET_A not in prompt_text(context_report(client))
|
|
|
|
|
|
def test_e02_the_same_control_through_a_save_point_restore(client):
|
|
"""E02 again, reached by restoring a Save Point rather than by Undo."""
|
|
play(client, "begin", prose="The lantern swings.")
|
|
r = client.post(f"/api/adventures/{client.adv_id}/checkpoints",
|
|
json={"name": "Before the ledger"})
|
|
assert r.status_code in (200, 201), r.text[:200]
|
|
save_point = r.json()
|
|
|
|
play(client, "search the floor", prose="Aldric pries up the flagstone.")
|
|
plant_memory(client, SECRET_A)
|
|
assert SECRET_A in prompt_text(context_report(client))
|
|
|
|
r = client.post(
|
|
f"/api/adventures/{client.adv_id}/checkpoints/{save_point['id']}/restore")
|
|
assert r.status_code == 200, r.text[:200]
|
|
|
|
assert SECRET_A not in eligible_memory_texts(client)
|
|
assert SECRET_A not in prompt_text(context_report(client))
|
|
|
|
# Diverging from the restored position keeps it out for good.
|
|
play(client, "a different road", prose="A different road opens.")
|
|
assert SECRET_A not in prompt_text(context_report(client))
|
|
with SessionLocal() as db:
|
|
assert db.query(models.Memory).filter_by(adventure_id=client.adv_id).count() == 1
|
|
|
|
|
|
def test_e03_an_abandoned_summary_is_retained_but_never_used(client):
|
|
"""E03. The failure this milestone fixes, measured in the prompt.
|
|
|
|
Before M6 the summary was a single column with a lineage cursor but no
|
|
lineage of its own, and the builder injected it unconditionally. Undo plus a
|
|
divergence therefore left the narrator reading sentences about a story the
|
|
reader was no longer on.
|
|
"""
|
|
play(client, "begin", prose="The lantern swings.")
|
|
play(client, "go to the chapel", prose="The chapel door gives.")
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
summaries.record(db, adventure, SECRET_B, trigger="interval",
|
|
model_name="test-model")
|
|
db.commit()
|
|
|
|
# Eligible on the line that produced it.
|
|
assert SECRET_B in prompt_text(context_report(client))
|
|
assert context_report(client)["summary"]["trigger"] == "interval"
|
|
|
|
# Undo before the summarized stretch, then diverge.
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
play(client, "take the other road", prose="A different road opens.")
|
|
|
|
report = context_report(client)
|
|
assert SECRET_B not in prompt_text(report), "an abandoned summary reached the prompt"
|
|
assert report["summary"] is None or SECRET_B not in report["summary"].get("preview", "")
|
|
|
|
# Retained, not deleted — and visible as retained.
|
|
status = client.get(f"/api/adventures/{client.adv_id}/derived").json()
|
|
stored = [row for row in status["summaries"] if SECRET_B in row["preview"]]
|
|
assert stored, "the abandoned summary was deleted rather than retained"
|
|
assert stored[0]["eligible"] is False
|
|
|
|
|
|
def test_e03_a_summary_becomes_eligible_again_on_redo(client):
|
|
"""The negative control needs its positive half: Redo restores the line, so
|
|
the summary written on it is usable again."""
|
|
play(client, "begin", prose="The lantern swings.")
|
|
play(client, "go to the chapel", prose="The chapel door gives.")
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
summaries.record(db, adventure, SECRET_B)
|
|
db.commit()
|
|
assert SECRET_B in prompt_text(context_report(client))
|
|
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
assert SECRET_B not in prompt_text(context_report(client))
|
|
|
|
client.post(f"/api/adventures/{client.adv_id}/redo")
|
|
client.post(f"/api/adventures/{client.adv_id}/redo")
|
|
assert SECRET_B in prompt_text(context_report(client))
|
|
|
|
|
|
def test_a_summary_the_reader_typed_is_anchored_too(client):
|
|
"""A hand-written summary is still a summary. It would otherwise survive a
|
|
divergence that its generated equivalent correctly does not."""
|
|
play(client, "begin", prose="The lantern swings.")
|
|
play(client, "go to the chapel", prose="The chapel door gives.")
|
|
r = client.patch(f"/api/adventures/{client.adv_id}",
|
|
json={"story_summary": SECRET_B})
|
|
assert r.status_code == 200, r.text[:200]
|
|
assert SECRET_B in prompt_text(context_report(client))
|
|
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
play(client, "the other road", prose="A different road opens.")
|
|
|
|
assert SECRET_B not in prompt_text(context_report(client))
|
|
|
|
|
|
def test_e01_and_e04_state_and_scene_are_unchanged_by_m6(client):
|
|
"""M5's lineage behaviour must not regress while context selection changes."""
|
|
play(client, "establish", prose="Mara arrives.", events=[
|
|
{"type": "create_entity", "entity": "mara", "entity_type": "character",
|
|
"name": "Mara"}])
|
|
play(client, "she learns", prose="Mara learns the code.", events=[
|
|
{"type": "add_fact", "subject": "mara", "predicate": "knows the vault code",
|
|
"fact_id": "vault"}])
|
|
document = client.get(f"/api/adventures/{client.adv_id}/state").json()["document"]
|
|
assert "knows the vault code" in [f["predicate"] for f in document["facts"]]
|
|
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
client.post(f"/api/adventures/{client.adv_id}/undo")
|
|
play(client, "a different road", prose="A different road opens.")
|
|
|
|
document = client.get(f"/api/adventures/{client.adv_id}/state").json()["document"]
|
|
assert "knows the vault code" not in [f["predicate"] for f in document["facts"]]
|
|
assert "vault code" not in prompt_text(context_report(client))
|
|
|
|
|
|
# ------------------------------------------- authority conflicts (section 8)
|
|
|
|
def test_a_memory_cannot_outrank_a_manual_correction(client):
|
|
"""Section 8. A withdrawn assertion may survive as history; it may not be
|
|
presented as current truth, whatever a memory says about it."""
|
|
play(client, "establish", prose="Mara arrives.", events=[
|
|
{"type": "create_entity", "entity": "mara", "entity_type": "character",
|
|
"name": "Mara"}])
|
|
play(client, "she learns", prose="Mara learns where the key was found.", events=[
|
|
{"type": "add_fact", "subject": "mara",
|
|
"predicate": "knows where the key was found", "fact_id": "mara-knows"}])
|
|
# A memory that records the same thing, written before the correction.
|
|
plant_memory(client, "Mara knows where the key was found.")
|
|
|
|
r = client.post(f"/api/adventures/{client.adv_id}/state/corrections", json={
|
|
"events": [{"type": "invalidate_fact", "fact_id": "mara-knows",
|
|
"reason": "Mara never learned where the silver key was found."}],
|
|
"note": "Mara never learned where the silver key was found.",
|
|
})
|
|
assert r.status_code in (200, 201), r.text[:200]
|
|
|
|
report = context_report(client)
|
|
sections = {s["label"]: s["text"] for s in report["sections"]}
|
|
|
|
# The authoritative state says it is withdrawn, in the prompt itself.
|
|
assert "No longer true" in sections["narrative_state"]
|
|
assert "Mara never learned" in sections["narrative_state"]
|
|
# The state section does not carry it among the facts that stand.
|
|
established = sections["narrative_state"].split("No longer true")[0]
|
|
assert "knows where the key was found" not in established
|
|
# The memory is subordinate: it is not state, and it is not canon.
|
|
document = client.get(f"/api/adventures/{client.adv_id}/state").json()["document"]
|
|
active = [f["predicate"] for f in document["facts"]
|
|
if f.get("status") != "invalidated"]
|
|
assert "knows where the key was found" not in active
|
|
|
|
|
|
# ---------------------------------------------------- derived rebuildability
|
|
|
|
def test_derived_data_can_be_deleted_and_rebuilt(client):
|
|
"""Section 16. Authoritative history must not depend on derived rows."""
|
|
play(client, "begin", prose="The lantern swings.", events=[
|
|
{"type": "create_entity", "entity": "aldric", "entity_type": "character",
|
|
"name": "Aldric"}])
|
|
plant_memory(client, SECRET_A)
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
summaries.record(db, adventure, SECRET_B)
|
|
db.commit()
|
|
|
|
before_actions = [a["text"] for a in
|
|
client.get(f"/api/adventures/{client.adv_id}").json()["actions"]]
|
|
before_state = client.get(f"/api/adventures/{client.adv_id}/state").json()["document"]
|
|
before_head = head_of(client)
|
|
|
|
# Remove every derived row.
|
|
with SessionLocal() as db:
|
|
db.query(models.Memory).filter_by(adventure_id=client.adv_id).delete()
|
|
db.query(models.Summary).filter_by(adventure_id=client.adv_id).delete()
|
|
db.commit()
|
|
|
|
after_actions = [a["text"] for a in
|
|
client.get(f"/api/adventures/{client.adv_id}").json()["actions"]]
|
|
after_state = client.get(f"/api/adventures/{client.adv_id}/state").json()["document"]
|
|
assert after_actions == before_actions, "deleting derived data changed the transcript"
|
|
assert after_state == before_state, "deleting derived data changed the state"
|
|
assert head_of(client) == before_head
|
|
# The story still plays with no derived data at all.
|
|
play(client, "carry on", prose="The door opens.")
|
|
|
|
# And derived data can be written again.
|
|
plant_memory(client, SECRET_A)
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
summaries.record(db, adventure, SECRET_B)
|
|
db.commit()
|
|
assert SECRET_A in prompt_text(context_report(client))
|
|
assert SECRET_B in prompt_text(context_report(client))
|
|
|
|
|
|
# ------------------------------------------- E03 regenerated after divergence
|
|
#
|
|
# M6 review finding M6-F1. The original E03 test proved only that the *old*
|
|
# summary row becomes ineligible after a divergence, and passed while the defect
|
|
# was live: the summariser seeded itself from `adventures.story_summary`, a
|
|
# campaign-global mirror with no lineage, so the summary it generated on the new
|
|
# line inherited the abandoned line's prose. The row was correctly anchored; its
|
|
# contents were not.
|
|
#
|
|
# The regression below plays far enough on the new line to force a *new* summary
|
|
# to be generated, which is the step that was missing.
|
|
|
|
E03_SENTINEL = "ABANDONED-CHAPEL-OATH-9930"
|
|
|
|
|
|
class CarryingSummariser:
|
|
"""A summariser that behaves like a real one.
|
|
|
|
It carries the summary it was given forward and folds in the new events, so
|
|
"did abandoned content reach this summary?" has an exact answer. The
|
|
per-block memory prompt is answered separately, echoing the sentinel only
|
|
for blocks that genuinely contain it.
|
|
"""
|
|
|
|
def __init__(self):
|
|
self.summary_seeds = []
|
|
|
|
async def complete(self, system, user, *, max_tokens=600):
|
|
if "Current story summary:" not in user:
|
|
return f"MEM[{E03_SENTINEL}]" if E03_SENTINEL in user else "MEM[dry road]"
|
|
current = user.split("Current story summary:\n", 1)[1].split("\n\nNew events")[0]
|
|
events = user.split("New events since the last update:\n", 1)[1].split(
|
|
"\n\nUpdated summary:")[0]
|
|
self.summary_seeds.append(current.strip())
|
|
carried = "" if current.strip() == "(none yet)" else current.strip() + " "
|
|
return (carried + events.strip().replace("\n", " "))[:1500]
|
|
|
|
async def embed(self, texts):
|
|
return [[1.0, 0.0, 0.0] for _ in texts]
|
|
|
|
|
|
def test_e03_a_summary_generated_after_divergence_carries_no_abandoned_content(client):
|
|
"""M6-F1. The failure the original E03 test could not see.
|
|
|
|
Every step of the review's reproduction, in order, with the positive control
|
|
first — a summary that does not exist proves nothing about what it omits.
|
|
"""
|
|
summariser = CarryingSummariser()
|
|
import app.memorybank as mb
|
|
real_summary, real_embed = mb.summary_provider, mb.embedding_provider
|
|
mb.summary_provider = lambda s: summariser
|
|
mb.embedding_provider = lambda s: summariser
|
|
try:
|
|
# 1-2. Path A, long enough to generate a summary, with the sentinel on it.
|
|
for i in range(20):
|
|
play(client, f"a{i}", prose=f"They swear the {E03_SENTINEL}. [{i}]")
|
|
asyncio.run(mb.run_post_turn(client.adv_id))
|
|
|
|
# 3. POSITIVE CONTROL: the sentinel really is in the path-A summary.
|
|
report_a = context_report(client)
|
|
summary_a = next((s["text"] for s in report_a["sections"]
|
|
if s["label"] == "story_summary"), "")
|
|
assert summary_a, "no summary was generated on path A; the rest proves nothing"
|
|
assert E03_SENTINEL in summary_a, "the fixture did not put the sentinel in the summary"
|
|
assert E03_SENTINEL in prompt_text(report_a)
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
path_a_summary_id = summaries.current(db, adventure).id
|
|
|
|
# 4. Move the head below every turn that mentions the sentinel.
|
|
while head_of(client)[1] > 0:
|
|
if client.post(f"/api/adventures/{client.adv_id}/undo").status_code != 200:
|
|
break
|
|
|
|
# 5-6. Diverge, and play far enough that a NEW summary is generated.
|
|
# Seeds recorded from here on are the ones that matter: on path A the
|
|
# summariser is *supposed* to be seeded with the sentinel, because the
|
|
# sentinel is on path A.
|
|
summariser.summary_seeds.clear()
|
|
for i in range(20):
|
|
play(client, f"b{i}", prose=f"A dry road, nothing sworn. [{i}]")
|
|
asyncio.run(mb.run_post_turn(client.adv_id))
|
|
|
|
report_b = context_report(client)
|
|
summary_b_row = report_b["summary"]
|
|
summary_b = next((s["text"] for s in report_b["sections"]
|
|
if s["label"] == "story_summary"), "")
|
|
|
|
# 7. A new summary really was generated on the new line.
|
|
assert summary_b_row is not None, "no summary is eligible on path B"
|
|
assert summary_b_row["id"] != path_a_summary_id, (
|
|
"path B reused path A's summary row rather than generating one"
|
|
)
|
|
|
|
# 8. No path-A story is on path B's lineage, so anything from it is a leak.
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
carried_over = db.query(models.Action).filter(
|
|
models.Action.adventure_id == client.adv_id,
|
|
lineage.path_of(db, adventure).clause(models.Action),
|
|
models.Action.text.like(f"%{E03_SENTINEL}%"),
|
|
).count()
|
|
assert carried_over == 0, "the fixture left path-A story on path B's lineage"
|
|
|
|
# 9-10. The sentinel is in neither the new summary nor the whole prompt.
|
|
assert E03_SENTINEL not in summary_b, (
|
|
"the summary generated on path B carries the abandoned line's content"
|
|
)
|
|
assert E03_SENTINEL not in prompt_text(report_b), (
|
|
"abandoned content reached the active narrator prompt"
|
|
)
|
|
|
|
# And it was never even *offered* the abandoned prose: the fix is at the
|
|
# input, not a filter over the output.
|
|
assert summariser.summary_seeds, "no summary was generated on path B"
|
|
assert not any(E03_SENTINEL in seed for seed in summariser.summary_seeds), (
|
|
"the summariser was seeded with content from the abandoned line"
|
|
)
|
|
|
|
# 11. The old summary is retained, and reported as retained-but-ineligible.
|
|
listing = client.get(f"/api/adventures/{client.adv_id}/derived").json()
|
|
old = [row for row in listing["summaries"] if row["id"] == path_a_summary_id]
|
|
assert old, "the abandoned summary row was deleted rather than retained"
|
|
assert old[0]["eligible"] is False
|
|
finally:
|
|
mb.summary_provider, mb.embedding_provider = real_summary, real_embed
|
|
|
|
|
|
# ------------------------------------------- M11: post-turn work and the write lock
|
|
#
|
|
# Found by the first 26-turn M01 trial on a GPU host. Every turn was accepted,
|
|
# and the run reported "complete" with two memories, no summary and 180
|
|
# `database is locked` errors. A turn that used a memory wrote its use counter
|
|
# before the model call and committed only after the reply. That held SQLite's
|
|
# single write lock for the whole reply. Post-turn memory and summary writes
|
|
# timed out behind it, and the record of each failure timed out the same way.
|
|
|
|
|
|
class LockProbe(ScriptedProvider):
|
|
"""A narrator that checks, mid-reply, whether any other writer could get in."""
|
|
|
|
seen: list = []
|
|
|
|
async def generate(self, parts, *, temperature, max_tokens):
|
|
# Its own connection, as a post-turn task's session would have. The
|
|
# short timeout turns "would wait five seconds and fail" into an
|
|
# immediate answer.
|
|
probe = sqlite3.connect(DB_PATH, timeout=0.1)
|
|
try:
|
|
probe.execute("BEGIN IMMEDIATE")
|
|
probe.rollback()
|
|
LockProbe.seen.append("free")
|
|
except sqlite3.OperationalError as exc:
|
|
LockProbe.seen.append(str(exc))
|
|
finally:
|
|
probe.close()
|
|
async for item in super().generate(parts, temperature=temperature,
|
|
max_tokens=max_tokens):
|
|
yield item
|
|
|
|
|
|
def test_no_write_lock_is_held_while_the_narrator_is_talking(client, monkeypatch):
|
|
play(client, "begin", prose="Aldric sets the key down.")
|
|
memory_id = plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
|
|
LockProbe.seen = []
|
|
monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", LockProbe)
|
|
|
|
play(client, "I lift the flagstone and look for the ledger.")
|
|
|
|
with SessionLocal() as db:
|
|
# The premise. A turn that retrieved no memory never took the lock, so
|
|
# the probe below would pass for the wrong reason.
|
|
assert db.get(models.Memory, memory_id).use_count == 1, (
|
|
"the turn did not use the planted memory, so this proves nothing")
|
|
assert LockProbe.seen == ["free"], (
|
|
"a write transaction was open during the model call, so every "
|
|
f"post-turn write in that window is locked out: {LockProbe.seen}")
|
|
|
|
|
|
def test_a_failed_turn_counts_no_memory_as_used(client, monkeypatch):
|
|
"""The counter is written with the turn now, so a turn that never landed
|
|
used nothing."""
|
|
play(client, "begin", prose="Aldric sets the key down.")
|
|
memory_id = plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
|
|
ScriptedProvider.replies = [ProviderError("the narrator is gone")]
|
|
r = client.post(f"/api/adventures/{client.adv_id}/actions",
|
|
json={"type": "do", "text": "I look for the ledger."})
|
|
assert '"error"' in r.text
|
|
|
|
with SessionLocal() as db:
|
|
assert db.get(models.Memory, memory_id).use_count == 0
|
|
|
|
|
|
def test_a_failure_that_breaks_the_session_is_still_recorded(client, monkeypatch):
|
|
"""Recording a failure needs a working session. Without a rollback first,
|
|
the recorder raised `PendingRollbackError`, the failure went only to the
|
|
log, and derived status kept reporting a healthy bank."""
|
|
play(client, "begin", prose="Aldric sets the key down.")
|
|
existing = plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
|
|
|
|
def collide(adventure, settings, db):
|
|
# A primary key that already exists: the flush fails and leaves the
|
|
# session needing a rollback, which is the state a lock timeout on
|
|
# commit leaves it in.
|
|
db.add(models.Memory(id=existing, adventure_id=client.adv_id,
|
|
text="a second row with the same key"))
|
|
db.flush()
|
|
|
|
monkeypatch.setattr(memorybank, "_evict_over_capacity", collide)
|
|
asyncio.run(memorybank.run_post_turn(client.adv_id))
|
|
|
|
with SessionLocal() as db:
|
|
rows = {row["kind"]: row for row in derived.report(db, client.adv_id)}
|
|
assert rows[derived.MEMORY]["status"] == "failed", rows.get(derived.MEMORY)
|
|
assert "PendingRollbackError" not in rows[derived.MEMORY]["detail"]
|
|
|
|
|
|
def test_the_long_run_harness_reads_sections_by_their_real_names(client):
|
|
"""`tools/m11_long_run.py` finds prompt sections by label, and a wrong label
|
|
is silent: it measured 0 memory tokens and could never find the clue in
|
|
history or in memories. These are the names the real builder uses."""
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
adventure.authors_note = "Keep the rain in every scene."
|
|
db.commit()
|
|
play(client, "begin", prose="Aldric sets the key down.",
|
|
events=[{"type": "create_entity", "entity": "aldric",
|
|
"entity_type": "character", "name": "Aldric"}])
|
|
for step in range(6):
|
|
play(client, f"walk on {step}")
|
|
plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
|
|
with SessionLocal() as db:
|
|
adventure = db.get(models.Adventure, client.adv_id)
|
|
summaries.record(db, adventure, "The party reached the Crooked Lantern.")
|
|
db.commit()
|
|
play(client, "I look for the ledger.")
|
|
|
|
labels = {s["label"] for s in context_report(client)["sections"]}
|
|
for label in (m11_long_run.MEMORIES_LABEL, m11_long_run.SUMMARY_LABEL,
|
|
m11_long_run.STATE_LABEL, *m11_long_run.HISTORY_LABELS):
|
|
assert label in labels, f"the harness reads {label!r}; the prompt has {sorted(labels)}"
|
|
assert set(m11_long_run.IMPORTED_KNOWLEDGE_LABELS) == {
|
|
classes.SECTION_ALWAYS_CANON, *classes.CLASS_SECTIONS.values()}
|