Stop a turn locking out its own memory bank, and let the long run notice
The first M01 trial with the memory bank on was 26 turns on a GPU host. It
accepted every turn and reported "complete". It also wrote two memories and
no summary, and logged 180 `database is locked` errors, while derived status
still read `idle`.
The cause was a single uncommitted UPDATE. Retrieval bumped each used
memory's counter before the model call, and the turn commits only after the
reply has streamed. SQLite has one writer, so the turn held the write lock for
the whole reply. Every post-turn memory, summary and status write in that
window waited out the five-second timeout and failed. Recording the failure
needed a write as well, and without a rollback first it raised
PendingRollbackError. The loss therefore reached the log and never reached
the status the Insights panel reads, which F08 forbids. The draco run never
hit this because the bank was off there.
- `retrieve_memories` now only reads. `record_use` writes the counters in the
turn's single commit, so a turn that never lands counts nothing.
- The post-turn task's outer handler rolls back before it records a failure.
The harness could not have caught any of this. It read three prompt sections
under names the builder does not use: `memories` (really `used_memories`),
`story_history` (really `history`/`recent_history`), and a `knowledge` prefix
that matched the fixed instruction section instead of the imported passages.
Memory tokens read 0 whatever the prompt held, and the in-history and
in-memories recall checks could never come out true. The labels are now
constants, pinned by a test against a prompt the real builder assembled.
The harness also stops at the first sign of failed post-turn work. It checks
/derived and new server.log lines after every turn, keeps its log position
across --resume, and waits for background work to settle before its final
checks. A run with no memories or no summaries now ends "failed", not
"complete".
Both new application tests fail on fec46f6: the lock probe sees
`database is locked`, and memory status stays `idle`. The full backend suite
passes (1392 passed, 17 skipped). A 26-turn re-run against the same host had
0 lock errors, wrote 7 memories and 2 summaries, and used them in the prompt
from turn 8.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0136VBTMUKWYeU6G9HgbDbND
This commit is contained in:
co-authored by
Claude Opus 5
parent
fec46f66bb
commit
f8d401029f
@@ -13,6 +13,8 @@ sending it. What they cannot prove is that a hundred turns then fill the bank
|
||||
that is what the run itself proves, and `memories_in_bank` in its timeline is
|
||||
where it shows.
|
||||
"""
|
||||
import json
|
||||
|
||||
import pytest
|
||||
from fastapi import Depends
|
||||
from fastapi.testclient import TestClient
|
||||
@@ -182,3 +184,88 @@ def test_a_count_that_cannot_be_read_is_minus_one_not_an_exception(run_for):
|
||||
run = run_for(Down())
|
||||
run.adv = 7
|
||||
assert run.bank_size() == -1
|
||||
|
||||
|
||||
# ----------------------------------------------- failed post-turn work stops a run
|
||||
|
||||
class Reports:
|
||||
"""A server whose derived status, summaries and log say what the test sets."""
|
||||
|
||||
starts = 1
|
||||
|
||||
def __init__(self, log_path, *, status=None, summaries=0, memories=0):
|
||||
self.log_path = log_path
|
||||
self.status = status or []
|
||||
self.summaries = summaries
|
||||
self.memories = memories
|
||||
|
||||
def call(self, method, path, payload=None, timeout=600):
|
||||
if path.endswith("/derived"):
|
||||
return {"status": self.status,
|
||||
"failing": [r["kind"] for r in self.status if r["status"] == "failed"],
|
||||
"summaries": [{"id": i} for i in range(self.summaries)]}
|
||||
if path.endswith("/memories"):
|
||||
return [{"id": i} for i in range(self.memories)]
|
||||
return {}
|
||||
|
||||
|
||||
def test_a_failed_pass_in_derived_status_stops_the_run(run_for, tmp_path):
|
||||
server = Reports(tmp_path / "server.log", status=[
|
||||
{"kind": "summary", "status": "failed", "detail": "ProviderError: gone"},
|
||||
{"kind": "memory", "status": "idle", "detail": ""},
|
||||
])
|
||||
run = run_for(server)
|
||||
run.adv = 1
|
||||
found = run.background_failures()
|
||||
assert found == ["summary: ProviderError: gone"]
|
||||
|
||||
|
||||
def test_a_failure_the_application_could_not_record_is_found_in_the_log_once(run_for, tmp_path):
|
||||
"""The failure that hid the first GPU trial: derived status said `idle` and
|
||||
the only record was in the server log."""
|
||||
log = tmp_path / "server.log"
|
||||
log.write_text("INFO: 200 OK\nERROR:app.memorybank:could not record derived-work failure for 1\n")
|
||||
run = run_for(Reports(log))
|
||||
run.adv = 1
|
||||
|
||||
assert len(run.background_failures()) == 1
|
||||
assert run.background_failures() == [], "the same line was reported twice"
|
||||
|
||||
with log.open("a") as handle:
|
||||
handle.write("ERROR:app.derived:derived summary work failed for adventure 1\n")
|
||||
assert len(run.background_failures()) == 1
|
||||
|
||||
|
||||
def test_healthy_status_and_a_quiet_log_find_nothing(run_for, tmp_path):
|
||||
log = tmp_path / "server.log"
|
||||
log.write_text('INFO: "POST /api/adventures/1/actions HTTP/1.1" 200 OK\n')
|
||||
run = run_for(Reports(log, status=[{"kind": "memory", "status": "ok", "detail": ""}]))
|
||||
run.adv = 1
|
||||
assert run.background_failures() == []
|
||||
|
||||
|
||||
def test_the_log_position_survives_a_resume(run_for, tmp_path):
|
||||
"""Otherwise a resumed run would find the failure that stopped it again, and
|
||||
stop again, however healthy the application now is."""
|
||||
first = run_for(Reports(tmp_path / "server.log"))
|
||||
first.adv, first.log_offset = 1, 4096
|
||||
first.save_resume()
|
||||
|
||||
second = run_for(Reports(tmp_path / "server.log"))
|
||||
second.adopt(json.loads((tmp_path / lr.RESUME_FILE).read_text()))
|
||||
assert second.log_offset == 4096
|
||||
|
||||
|
||||
def test_a_run_with_no_summary_or_no_memory_is_not_complete(run_for, tmp_path):
|
||||
log = tmp_path / "server.log"
|
||||
assert "summaries=0" in lr._activation_shortfall(
|
||||
_with_adv(run_for(Reports(log, memories=3, summaries=0))))
|
||||
assert "memories_in_bank=0" in lr._activation_shortfall(
|
||||
_with_adv(run_for(Reports(log, memories=0, summaries=2))))
|
||||
assert lr._activation_shortfall(
|
||||
_with_adv(run_for(Reports(log, memories=3, summaries=1)))) is None
|
||||
|
||||
|
||||
def _with_adv(run):
|
||||
run.adv = 1
|
||||
return run
|
||||
|
||||
Reference in New Issue
Block a user