Stop a turn locking out its own memory bank, and let the long run notice

The first M01 trial with the memory bank on was 26 turns on a GPU host. It
accepted every turn and reported "complete". It also wrote two memories and
no summary, and logged 180 `database is locked` errors, while derived status
still read `idle`.

The cause was a single uncommitted UPDATE. Retrieval bumped each used
memory's counter before the model call, and the turn commits only after the
reply has streamed. SQLite has one writer, so the turn held the write lock for
the whole reply. Every post-turn memory, summary and status write in that
window waited out the five-second timeout and failed. Recording the failure
needed a write as well, and without a rollback first it raised
PendingRollbackError. The loss therefore reached the log and never reached
the status the Insights panel reads, which F08 forbids. The draco run never
hit this because the bank was off there.

- `retrieve_memories` now only reads. `record_use` writes the counters in the
  turn's single commit, so a turn that never lands counts nothing.
- The post-turn task's outer handler rolls back before it records a failure.

The harness could not have caught any of this. It read three prompt sections
under names the builder does not use: `memories` (really `used_memories`),
`story_history` (really `history`/`recent_history`), and a `knowledge` prefix
that matched the fixed instruction section instead of the imported passages.
Memory tokens read 0 whatever the prompt held, and the in-history and
in-memories recall checks could never come out true. The labels are now
constants, pinned by a test against a prompt the real builder assembled.

The harness also stops at the first sign of failed post-turn work. It checks
/derived and new server.log lines after every turn, keeps its log position
across --resume, and waits for background work to settle before its final
checks. A run with no memories or no summaries now ends "failed", not
"complete".

Both new application tests fail on fec46f6: the lock probe sees
`database is locked`, and memory status stays `idle`. The full backend suite
passes (1392 passed, 17 skipped). A 26-turn re-run against the same host had
0 lock errors, wrote 7 memories and 2 summaries, and used them in the prompt
from turn 8.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0136VBTMUKWYeU6G9HgbDbND
This commit is contained in:
JesseMarkowitz
2026-09-13 20:21:21 -04:00
co-authored by Claude Opus 5
parent fec46f66bb
commit f8d401029f
9 changed files with 431 additions and 37 deletions
+121 -1
View File
@@ -18,6 +18,7 @@ Two things are asserted throughout rather than assumed:
"""
import asyncio
import sqlite3
import pytest
from fastapi import Depends
@@ -26,12 +27,14 @@ from sqlalchemy import select
from app import auth, derived, limits, memorybank, models, summaries
from app.context import builder, lineage
from app.database import Base, SessionLocal, engine, get_db
from app.database import DB_PATH, Base, SessionLocal, engine, get_db
from app.knowledge import classes
from app.main import app
from app.providers import ProviderError
from app.routers import adventures
from fakes import ScriptedProvider, state_block
from tools import m11_long_run
class StubEmbedder:
@@ -835,3 +838,120 @@ def test_e03_a_summary_generated_after_divergence_carries_no_abandoned_content(c
assert old[0]["eligible"] is False
finally:
mb.summary_provider, mb.embedding_provider = real_summary, real_embed
# ------------------------------------------- M11: post-turn work and the write lock
#
# Found by the first 26-turn M01 trial on a GPU host. Every turn was accepted,
# and the run reported "complete" with two memories, no summary and 180
# `database is locked` errors. A turn that used a memory wrote its use counter
# before the model call and committed only after the reply. That held SQLite's
# single write lock for the whole reply. Post-turn memory and summary writes
# timed out behind it, and the record of each failure timed out the same way.
class LockProbe(ScriptedProvider):
"""A narrator that checks, mid-reply, whether any other writer could get in."""
seen: list = []
async def generate(self, parts, *, temperature, max_tokens):
# Its own connection, as a post-turn task's session would have. The
# short timeout turns "would wait five seconds and fail" into an
# immediate answer.
probe = sqlite3.connect(DB_PATH, timeout=0.1)
try:
probe.execute("BEGIN IMMEDIATE")
probe.rollback()
LockProbe.seen.append("free")
except sqlite3.OperationalError as exc:
LockProbe.seen.append(str(exc))
finally:
probe.close()
async for item in super().generate(parts, temperature=temperature,
max_tokens=max_tokens):
yield item
def test_no_write_lock_is_held_while_the_narrator_is_talking(client, monkeypatch):
play(client, "begin", prose="Aldric sets the key down.")
memory_id = plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
LockProbe.seen = []
monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", LockProbe)
play(client, "I lift the flagstone and look for the ledger.")
with SessionLocal() as db:
# The premise. A turn that retrieved no memory never took the lock, so
# the probe below would pass for the wrong reason.
assert db.get(models.Memory, memory_id).use_count == 1, (
"the turn did not use the planted memory, so this proves nothing")
assert LockProbe.seen == ["free"], (
"a write transaction was open during the model call, so every "
f"post-turn write in that window is locked out: {LockProbe.seen}")
def test_a_failed_turn_counts_no_memory_as_used(client, monkeypatch):
"""The counter is written with the turn now, so a turn that never landed
used nothing."""
play(client, "begin", prose="Aldric sets the key down.")
memory_id = plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
ScriptedProvider.replies = [ProviderError("the narrator is gone")]
r = client.post(f"/api/adventures/{client.adv_id}/actions",
json={"type": "do", "text": "I look for the ledger."})
assert '"error"' in r.text
with SessionLocal() as db:
assert db.get(models.Memory, memory_id).use_count == 0
def test_a_failure_that_breaks_the_session_is_still_recorded(client, monkeypatch):
"""Recording a failure needs a working session. Without a rollback first,
the recorder raised `PendingRollbackError`, the failure went only to the
log, and derived status kept reporting a healthy bank."""
play(client, "begin", prose="Aldric sets the key down.")
existing = plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
def collide(adventure, settings, db):
# A primary key that already exists: the flush fails and leaves the
# session needing a rollback, which is the state a lock timeout on
# commit leaves it in.
db.add(models.Memory(id=existing, adventure_id=client.adv_id,
text="a second row with the same key"))
db.flush()
monkeypatch.setattr(memorybank, "_evict_over_capacity", collide)
asyncio.run(memorybank.run_post_turn(client.adv_id))
with SessionLocal() as db:
rows = {row["kind"]: row for row in derived.report(db, client.adv_id)}
assert rows[derived.MEMORY]["status"] == "failed", rows.get(derived.MEMORY)
assert "PendingRollbackError" not in rows[derived.MEMORY]["detail"]
def test_the_long_run_harness_reads_sections_by_their_real_names(client):
"""`tools/m11_long_run.py` finds prompt sections by label, and a wrong label
is silent: it measured 0 memory tokens and could never find the clue in
history or in memories. These are the names the real builder uses."""
with SessionLocal() as db:
adventure = db.get(models.Adventure, client.adv_id)
adventure.authors_note = "Keep the rain in every scene."
db.commit()
play(client, "begin", prose="Aldric sets the key down.",
events=[{"type": "create_entity", "entity": "aldric",
"entity_type": "character", "name": "Aldric"}])
for step in range(6):
play(client, f"walk on {step}")
plant_memory(client, "Aldric hid the ledger beneath the third flagstone.")
with SessionLocal() as db:
adventure = db.get(models.Adventure, client.adv_id)
summaries.record(db, adventure, "The party reached the Crooked Lantern.")
db.commit()
play(client, "I look for the ledger.")
labels = {s["label"] for s in context_report(client)["sections"]}
for label in (m11_long_run.MEMORIES_LABEL, m11_long_run.SUMMARY_LABEL,
m11_long_run.STATE_LABEL, *m11_long_run.HISTORY_LABELS):
assert label in labels, f"the harness reads {label!r}; the prompt has {sorted(labels)}"
assert set(m11_long_run.IMPORTED_KNOWLEDGE_LABELS) == {
classes.SECTION_ALWAYS_CANON, *classes.CLASS_SECTIONS.values()}