The M04 re-run on0c7316fran 101 turns with no failures, and the verdict still came out `precondition_not_met`. That verdict was wrong. The planted player turn (depth 1) was 65 actions outside the history window, whose floor was 66. The sentinel's text was in recent history for two other reasons. - On 5 turns the narrator wrote a section of its own, `## Established:` over indented facts, with the planted clue copied into it from the state section. The extractor passed it: it was a single heading, and the `## ` meant it did not match. The harness leak count passed it the same way, and read 1 where 5 turns leaked. - On 4 turns the narrator used the sentinel as a name inside ordinary sentences ("the SILVER-KEY-CRYPT-OLD-ABBEY, flickers with latent power"). That is story text and cannot be stripped. So the sentinel's text in history can never be the precondition. M04's written pass is "Fact/event remains recoverable without entire transcript in prompt" (V1-ACCEPTANCE-TESTS.md). The owner agreed on 2026-09-13 that the precondition is positional, and that recovery through authoritative state counts; memory is not required. Extractor: - A state-section heading is recognised with any markdown the model wrapped it in (`## Established:`, `**Held:**`, `> Held:`). - One heading with an indented entry under it now qualifies as protocol. Before, a block needed two headings, or one heading and the scene line. A heading followed by unindented prose is still story. The earlier guard test "Held:" with an indented line is now protocol, and the case was rewritten unindented. Harness: - It records `planted_depth` when the clue is planted, and carries it across --resume. No endpoint reports an action's depth, but a fresh campaign's path is the opening, the planted turn and its reply, so the depth is `total_actions - 2`. That was checked against the database in three runs. - `_recall` reports `planted_depth`, `history_floor_depth` and `planted_turn_in_history_window`. The verdict is `precondition_not_met` only when the planted turn is still in the window, and `precondition_unknown` when its depth was never recorded. `clue_in_recent_history_window` stays as a fact about the prompt. - The protocol-leak count matches a state heading with an indented entry under it, in any markdown. Every AI turn in five real runs was replayed through the new extractor: draco M01, the two 26-turn GPU trials, thef8d4010M01 run and the M04 re-run. 443 turns in all. No turn the old extractor had left clean changed, and the M04 re-run lost 7 more leaks. The sentinel-as-a-name turns are story and remain. Trial 1's model-invented headings remain, as before. The M04 re-run's own evidence, reclassified under the new precondition from its database and its recall-turn snapshot, reads `recovered_through_state_only`. The original recall.json is kept unchanged beside `recall-reclassified.json`. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0136VBTMUKWYeU6G9HgbDbND
325 lines
12 KiB
Python
325 lines
12 KiB
Python
"""The long run turns the memory bank and the rolling summary on, and proves it.
|
|
|
|
M01's step list asks for "summary/memory activation". Both are per-campaign
|
|
switches that default to off (`models.Adventure`), and the harness that ran the
|
|
first complete hundred-turn campaign never touched them: the bank stayed empty,
|
|
no summary was written, and M04's recall succeeded through narrative state alone.
|
|
Nothing in that run's evidence said so except a row of zeros nobody was looking
|
|
for.
|
|
|
|
These tests drive `Run.setup` against the real application in-process, so the
|
|
switch is proved by the application accepting it rather than by the harness
|
|
sending it. What they cannot prove is that a hundred turns then fill the bank —
|
|
that is what the run itself proves, and `memories_in_bank` in its timeline is
|
|
where it shows.
|
|
"""
|
|
import json
|
|
|
|
import pytest
|
|
from fastapi import Depends
|
|
from fastapi.testclient import TestClient
|
|
|
|
from app import auth, models
|
|
from app.database import Base, SessionLocal, engine, get_db
|
|
from app.main import app
|
|
from tools import m11_long_run as lr
|
|
|
|
|
|
UNREACHABLE = "http://127.0.0.1:1/v1"
|
|
|
|
|
|
class InProcess:
|
|
"""`Storyteller.call`, spoken to the application through the test client."""
|
|
|
|
starts = 1
|
|
|
|
def __init__(self, client: TestClient):
|
|
self.client = client
|
|
|
|
def call(self, method, path, payload=None, timeout=600):
|
|
response = self.client.request(method, f"/api{path}", json=payload)
|
|
response.raise_for_status()
|
|
return response.json() if response.content else None
|
|
|
|
|
|
class IgnoresThePatch:
|
|
"""A server that answers the PATCH and changes nothing, which is exactly the
|
|
failure `setup` must refuse rather than record."""
|
|
|
|
starts = 1
|
|
|
|
def __init__(self):
|
|
self.calls = []
|
|
|
|
def call(self, method, path, payload=None, timeout=600):
|
|
self.calls.append((method, path))
|
|
if path == "/settings":
|
|
return {"model": "m", "context_token_budget": 16384,
|
|
"model_timeout_seconds": 1800}
|
|
if method == "POST" and path == "/adventures":
|
|
return {"id": 1}
|
|
if method == "GET" and path == "/adventures/1":
|
|
return {"memory_bank_enabled": False, "auto_summarize": False}
|
|
return {}
|
|
|
|
|
|
@pytest.fixture()
|
|
def client(monkeypatch):
|
|
monkeypatch.setattr(lr, "ENDPOINT", UNREACHABLE)
|
|
monkeypatch.setattr(lr, "MODEL", "some-model")
|
|
monkeypatch.setattr(lr, "EMBED_MODEL", "some-embedder")
|
|
Base.metadata.create_all(bind=engine)
|
|
setup = SessionLocal()
|
|
user = models.User(is_guest=False, email="m11mem@example.com")
|
|
setup.add(user)
|
|
setup.commit()
|
|
user_id = user.id
|
|
setup.close()
|
|
|
|
app.dependency_overrides[auth.get_current_user] = (
|
|
lambda db=Depends(get_db): db.get(models.User, user_id)
|
|
)
|
|
try:
|
|
yield TestClient(app)
|
|
finally:
|
|
app.dependency_overrides.clear()
|
|
Base.metadata.drop_all(bind=engine)
|
|
|
|
|
|
@pytest.fixture()
|
|
def run_for(tmp_path, monkeypatch):
|
|
"""A `Run` on the given server. Uploads are skipped: `upload` builds its own
|
|
multipart request to a port, and the knowledge library is not what these
|
|
tests are about."""
|
|
made = []
|
|
|
|
def make(server):
|
|
run = lr.Run(server, tmp_path, turns_target=100)
|
|
monkeypatch.setattr(run, "upload", lambda *a, **k: None)
|
|
made.append(run)
|
|
return run
|
|
|
|
yield make
|
|
for run in made:
|
|
run.timeline.close()
|
|
|
|
|
|
def test_a_fresh_campaign_starts_with_both_switches_off(client):
|
|
"""The premise. If this ever changes, the harness's PATCH is redundant but
|
|
harmless; while it holds, a harness without the PATCH measures nothing."""
|
|
created = client.post("/api/adventures", json={"title": "untouched"}).json()
|
|
assert created["memory_bank_enabled"] is False
|
|
assert created["auto_summarize"] is False
|
|
|
|
|
|
def test_setup_leaves_the_campaign_with_memory_and_summary_on(client, run_for):
|
|
run = run_for(InProcess(client))
|
|
run.setup()
|
|
|
|
stored = client.get(f"/api/adventures/{run.adv}").json()
|
|
assert stored["memory_bank_enabled"] is True
|
|
assert stored["auto_summarize"] is True
|
|
|
|
activated = [e for e in run.events if e["kind"] == "memory_activated"]
|
|
assert len(activated) == 1
|
|
assert activated[0]["memory_bank_enabled"] is True
|
|
assert activated[0]["auto_summarize"] is True
|
|
|
|
|
|
def test_setup_refuses_a_campaign_that_did_not_take_the_switches(run_for):
|
|
"""Hours of turns against a campaign with the bank off is the run that was
|
|
already had. It must stop before the first one, not report silence after."""
|
|
server = IgnoresThePatch()
|
|
run = run_for(server)
|
|
|
|
with pytest.raises(SystemExit, match="summary/memory clause"):
|
|
run.setup()
|
|
|
|
# It asked, it read back, and it went no further.
|
|
assert ("PATCH", "/adventures/1") in server.calls
|
|
assert not any("/state/corrections" in path for _, path in server.calls)
|
|
recorded = [e for e in run.events if e["kind"] == "memory_activated"]
|
|
assert recorded and recorded[0]["memory_bank_enabled"] is False
|
|
|
|
|
|
def test_a_run_without_an_embedding_model_is_refused(monkeypatch, tmp_path, capsys):
|
|
"""With the bank on and no embedder, memories are written and never
|
|
retrieved: `memorybank.retrieve` answers "No embedding model configured".
|
|
That is the same unexercised path in a fuller bank, so it is refused before
|
|
a server is started or a directory is claimed."""
|
|
monkeypatch.setattr(lr, "ENDPOINT", UNREACHABLE)
|
|
monkeypatch.setattr(lr, "MODEL", "some-model")
|
|
monkeypatch.setattr(lr, "EMBED_MODEL", "")
|
|
out = tmp_path / "never-made"
|
|
monkeypatch.setattr("sys.argv", ["m11_long_run", "--out", str(out)])
|
|
|
|
assert lr.main() == 2
|
|
assert "AIDND_TEST_EMBED_MODEL" in capsys.readouterr().out
|
|
assert not out.exists()
|
|
|
|
|
|
def test_the_bank_is_counted_from_the_application(client, run_for):
|
|
run = run_for(InProcess(client))
|
|
run.setup()
|
|
assert run.bank_size() == 0
|
|
|
|
db = SessionLocal()
|
|
try:
|
|
db.add(models.Memory(adventure_id=run.adv, text="the key opens the crypt"))
|
|
db.commit()
|
|
finally:
|
|
db.close()
|
|
assert run.bank_size() == 1
|
|
|
|
|
|
def test_a_count_that_cannot_be_read_is_minus_one_not_an_exception(run_for):
|
|
"""Measurement never fails a turn; -1 is distinguishable from an empty bank."""
|
|
|
|
class Down:
|
|
starts = 1
|
|
|
|
def call(self, *a, **k):
|
|
raise ConnectionError("gone")
|
|
|
|
run = run_for(Down())
|
|
run.adv = 7
|
|
assert run.bank_size() == -1
|
|
|
|
|
|
# ----------------------------------------------- failed post-turn work stops a run
|
|
|
|
class Reports:
|
|
"""A server whose derived status, summaries and log say what the test sets."""
|
|
|
|
starts = 1
|
|
|
|
def __init__(self, log_path, *, status=None, summaries=0, memories=0):
|
|
self.log_path = log_path
|
|
self.status = status or []
|
|
self.summaries = summaries
|
|
self.memories = memories
|
|
|
|
def call(self, method, path, payload=None, timeout=600):
|
|
if path.endswith("/derived"):
|
|
return {"status": self.status,
|
|
"failing": [r["kind"] for r in self.status if r["status"] == "failed"],
|
|
"summaries": [{"id": i} for i in range(self.summaries)]}
|
|
if path.endswith("/memories"):
|
|
return [{"id": i} for i in range(self.memories)]
|
|
return {}
|
|
|
|
|
|
def test_a_failed_pass_in_derived_status_stops_the_run(run_for, tmp_path):
|
|
server = Reports(tmp_path / "server.log", status=[
|
|
{"kind": "summary", "status": "failed", "detail": "ProviderError: gone"},
|
|
{"kind": "memory", "status": "idle", "detail": ""},
|
|
])
|
|
run = run_for(server)
|
|
run.adv = 1
|
|
found = run.background_failures()
|
|
assert found == ["summary: ProviderError: gone"]
|
|
|
|
|
|
def test_a_failure_the_application_could_not_record_is_found_in_the_log_once(run_for, tmp_path):
|
|
"""The failure that hid the first GPU trial: derived status said `idle` and
|
|
the only record was in the server log."""
|
|
log = tmp_path / "server.log"
|
|
log.write_text("INFO: 200 OK\nERROR:app.memorybank:could not record derived-work failure for 1\n")
|
|
run = run_for(Reports(log))
|
|
run.adv = 1
|
|
|
|
assert len(run.background_failures()) == 1
|
|
assert run.background_failures() == [], "the same line was reported twice"
|
|
|
|
with log.open("a") as handle:
|
|
handle.write("ERROR:app.derived:derived summary work failed for adventure 1\n")
|
|
assert len(run.background_failures()) == 1
|
|
|
|
|
|
def test_healthy_status_and_a_quiet_log_find_nothing(run_for, tmp_path):
|
|
log = tmp_path / "server.log"
|
|
log.write_text('INFO: "POST /api/adventures/1/actions HTTP/1.1" 200 OK\n')
|
|
run = run_for(Reports(log, status=[{"kind": "memory", "status": "ok", "detail": ""}]))
|
|
run.adv = 1
|
|
assert run.background_failures() == []
|
|
|
|
|
|
def test_the_log_position_survives_a_resume(run_for, tmp_path):
|
|
"""Otherwise a resumed run would find the failure that stopped it again, and
|
|
stop again, however healthy the application now is."""
|
|
first = run_for(Reports(tmp_path / "server.log"))
|
|
first.adv, first.log_offset = 1, 4096
|
|
first.save_resume()
|
|
|
|
second = run_for(Reports(tmp_path / "server.log"))
|
|
second.adopt(json.loads((tmp_path / lr.RESUME_FILE).read_text()))
|
|
assert second.log_offset == 4096
|
|
|
|
|
|
def test_a_run_with_no_summary_or_no_memory_is_not_complete(run_for, tmp_path):
|
|
log = tmp_path / "server.log"
|
|
assert "summaries=0" in lr._activation_shortfall(
|
|
_with_adv(run_for(Reports(log, memories=3, summaries=0))))
|
|
assert "memories_in_bank=0" in lr._activation_shortfall(
|
|
_with_adv(run_for(Reports(log, memories=0, summaries=2))))
|
|
assert lr._activation_shortfall(
|
|
_with_adv(run_for(Reports(log, memories=3, summaries=1)))) is None
|
|
|
|
|
|
def _with_adv(run):
|
|
run.adv = 1
|
|
return run
|
|
|
|
|
|
# ------------------------------------------------------- what M04 actually proved
|
|
|
|
def test_the_m04_verdict_never_credits_a_planted_turn_still_in_the_window():
|
|
base = {"planted_turn_in_history_window": False, "in_memories_section": False,
|
|
"in_summary_section": False, "in_state_section": False,
|
|
"clue_in_recent_history_window": False}
|
|
assert lr._m04_verdict({**base, "planted_turn_in_history_window": True,
|
|
"in_memories_section": True}) == "precondition_not_met"
|
|
assert lr._m04_verdict({**base, "planted_turn_in_history_window": None,
|
|
"in_state_section": True}) == "precondition_unknown"
|
|
assert lr._m04_verdict({**base, "in_memories_section": True}) == \
|
|
"recovered_through_memory_or_summary"
|
|
assert lr._m04_verdict({**base, "in_summary_section": True}) == \
|
|
"recovered_through_memory_or_summary"
|
|
assert lr._m04_verdict({**base, "in_state_section": True}) == \
|
|
"recovered_through_state_only"
|
|
assert lr._m04_verdict(base) == "not_recovered"
|
|
|
|
|
|
def test_the_sentinel_in_recent_history_does_not_decide_the_precondition():
|
|
"""The M04 re-run: the narrator reused the sentinel in its own prose while
|
|
the planted turn was 65 actions outside the window."""
|
|
recall = {"planted_turn_in_history_window": False,
|
|
"clue_in_recent_history_window": True,
|
|
"in_memories_section": False, "in_summary_section": False,
|
|
"in_state_section": True}
|
|
assert lr._m04_verdict(recall) == "recovered_through_state_only"
|
|
|
|
|
|
def test_the_planted_depth_survives_a_resume(run_for, tmp_path):
|
|
first = run_for(Reports(tmp_path / "server.log"))
|
|
first.adv, first.planted_depth = 1, 1
|
|
first.save_resume()
|
|
|
|
second = run_for(Reports(tmp_path / "server.log"))
|
|
second.adopt(json.loads((tmp_path / lr.RESUME_FILE).read_text()))
|
|
assert second.planted_depth == 1
|
|
|
|
|
|
def test_protocol_left_in_stored_narration_is_counted():
|
|
bundle = {"actions": [
|
|
{"id": 1, "type": "do", "text": '> You say {"events": []}'},
|
|
{"id": 2, "type": "ai", "text": "The rain eases."},
|
|
{"id": 3, "type": "ai", "text": "Beat.\n\nWho and what exists:\n mara: Mara"},
|
|
{"id": 4, "type": "ai", "text": 'Beat.\n\n> {"events": []}'},
|
|
{"id": 5, "type": "ai",
|
|
"text": "Rain.\n\n## Established:\n the crypt is sealed (SENTINEL)"},
|
|
{"id": 6, "type": "ai", "text": "The notice read:\n\nHeld:\nnothing at all."},
|
|
]}
|
|
assert lr._protocol_leaks(bundle) == {
|
|
"ai_actions": 5, "leaking": 3, "example_ids": [3, 4, 5]}
|