"""The long run's resume checkpoint, and the refusals that protect its evidence. M11's release campaign was lost twice over: once to a host crash at turn 97, and again to the fact that starting the harness a second time began a new campaign rather than continuing the old one. `tools/m11_long_run.py` now checkpoints `resume.json` and can be pointed back at it. These tests exercise that logic without a narrator, a server or a database, because none of it needs one: the checkpoint is a file, and the decisions made around it are decisions about files. What they cannot prove is that a resumed campaign continues correctly against a real application — that is what the run itself proves, and §G of the M11 report is where it is reported. """ import json import pytest from tools import m11_long_run as lr class FakeServer: """Enough of `Storyteller` for the checkpoint: it records process starts.""" def __init__(self, starts=1): self.starts = starts @pytest.fixture def run(tmp_path): made = lr.Run(FakeServer(), tmp_path, turns_target=100) yield made made.timeline.close() # --------------------------------------------------------------- checkpoint def test_a_checkpoint_carries_what_a_resume_needs(run, tmp_path): run.adv = 7 run.accepted = 41 run.beat = 44 run.completed_steps = {7, 14, 21} run.save_resume() saved = json.loads((tmp_path / lr.RESUME_FILE).read_text()) assert saved["adventure"] == 7 assert saved["accepted"] == 41 assert saved["beat"] == 44 assert saved["completed_steps"] == [7, 14, 21] assert saved["server_starts"] == 1 assert saved["turns_target"] == 100 def test_the_checkpoint_is_replaced_rather_than_appended(run, tmp_path): run.adv = 7 run.accepted = 1 run.save_resume() run.accepted = 2 run.save_resume() assert json.loads((tmp_path / lr.RESUME_FILE).read_text())["accepted"] == 2 # The temporary name it is written under must not survive the rename. assert not (tmp_path / (lr.RESUME_FILE + ".tmp")).exists() def test_a_checkpoint_round_trips_into_a_later_session(run, tmp_path): run.adv = 7 run.accepted = 41 run.beat = 44 run.completed_steps = {7, 14} run.elapsed_before = 100 run.save_resume() saved = json.loads((tmp_path / lr.RESUME_FILE).read_text()) later = lr.Run(FakeServer(starts=3), tmp_path, turns_target=100) try: later.adopt(saved) assert later.adv == 7 assert later.accepted == 41 assert later.beat == 44 assert later.completed_steps == {7, 14} assert later.resumed is True # Run time accumulates across sessions rather than restarting. assert later.elapsed_before >= 100 assert later.elapsed() >= 100 finally: later.timeline.close() def test_a_resumed_session_appends_to_the_existing_timeline(run, tmp_path): run.adv = 7 run.note("turn", text="the first session") run.timeline.close() later = lr.Run(FakeServer(), tmp_path, turns_target=100) try: later.note("resumed", adventure=7) finally: later.timeline.close() lines = (tmp_path / "timeline.jsonl").read_text().strip().splitlines() assert [json.loads(line)["kind"] for line in lines] == ["turn", "resumed"] # ------------------------------------------------------------- the decision def test_a_clean_directory_starts_a_run(tmp_path): assert lr._resume_state( tmp_path / lr.RESUME_FILE, tmp_path / "timeline.jsonl", False) is None def test_an_unfinished_run_is_not_overwritten(tmp_path): resume_path = tmp_path / lr.RESUME_FILE resume_path.write_text(json.dumps({"adventure": 7, "accepted": 41})) refusal = lr._resume_state(resume_path, tmp_path / "timeline.jsonl", False) assert isinstance(refusal, str) assert "--resume" in refusal def test_a_recorded_run_with_no_checkpoint_is_not_reused(tmp_path): """A run that recorded something and then died before its first checkpoint. Starting here would put a second campaign in the same timeline.""" timeline = tmp_path / "timeline.jsonl" timeline.write_text(json.dumps({"kind": "settings"}) + "\n") refusal = lr._resume_state(tmp_path / lr.RESUME_FILE, timeline, False) assert isinstance(refusal, str) assert "second campaign" in refusal def test_a_run_that_recorded_nothing_leaves_the_directory_usable(tmp_path): """A server that never came up opens the timeline and writes no line to it. Nothing was written that a fresh run could collide with.""" (tmp_path / "timeline.jsonl").write_text("") assert lr._resume_state( tmp_path / lr.RESUME_FILE, tmp_path / "timeline.jsonl", False) is None def test_resuming_returns_the_checkpoint(tmp_path): resume_path = tmp_path / lr.RESUME_FILE resume_path.write_text(json.dumps({"adventure": 7, "accepted": 41})) prior = lr._resume_state(resume_path, tmp_path / "timeline.jsonl", True) assert prior["adventure"] == 7 assert prior["accepted"] == 41 def test_resuming_nothing_is_refused_rather_than_started_fresh(tmp_path): refusal = lr._resume_state( tmp_path / lr.RESUME_FILE, tmp_path / "timeline.jsonl", True) assert isinstance(refusal, str) assert "no resume.json" in refusal def test_an_unreadable_checkpoint_is_refused(tmp_path): resume_path = tmp_path / lr.RESUME_FILE resume_path.write_text("{not json") refusal = lr._resume_state(resume_path, tmp_path / "timeline.jsonl", True) assert isinstance(refusal, str) assert "cannot read" in refusal def test_a_checkpoint_naming_no_campaign_is_refused(tmp_path): resume_path = tmp_path / lr.RESUME_FILE resume_path.write_text(json.dumps({"accepted": 41})) refusal = lr._resume_state(resume_path, tmp_path / "timeline.jsonl", True) assert isinstance(refusal, str) assert "names no campaign" in refusal # ----------------------------------------------------------------- timeouts def test_the_harness_waits_longer_than_the_application_does(tmp_path): """Otherwise the socket closes before the application can report the failure inside the stream, and a real error is recorded as a transport one.""" server = lr.Storyteller(tmp_path / "campaign.db", tmp_path / "server.log", turn_timeout=1800) assert server.stream_timeout > server.turn_timeout def test_the_default_timeout_is_inside_the_settings_bound(): """`app/schemas.py` bounds model_timeout_seconds at 30..3600.""" assert 30 <= lr.DEFAULT_TURN_TIMEOUT <= 3600 # ----------------------------------------------------------------- schedule def test_every_scheduled_operation_has_its_own_turn(tmp_path): """The steps are keyed by turn number, which is what lets a completed one be remembered across a resume: two are called `retry` and two `restart`, so a name does not identify one.""" plan = lr._schedule(100) assert len(plan) == len(set(plan)) == 12 assert sorted(plan)[0] >= 1 assert max(plan) < 100 names = list(plan.values()) assert names.count("restart") == 2 assert names.count("retry") == 2 # --------------------------------------------------------------- the clue def test_the_planted_clue_uses_a_field_add_fact_actually_carries(): """M04's state half turns on the clue text reaching the stored fact. `add_fact` requires `predicate` and accepts `subject`, `object`, `value` and `fact_id`. A key it does not define is dropped, and the correction still succeeds — so a clue planted into the wrong key leaves a fact asserting nothing, and `_recall` reports a recall failure the application did not cause. This test fails against the `detail` key that used to be sent. """ from app.narrative.events import SPECS spec = SPECS["add_fact"] allowed = {"type"} | set(spec["required"]) | set(spec["optional"]) assert set(lr.CLUE_FACT) <= allowed, ( f"{set(lr.CLUE_FACT) - allowed} is not carried by add_fact") def test_the_planted_clue_carries_the_sentinel_recall_looks_for(): assert lr.CLUE_SENTINEL in lr.CLUE_FACT["value"] assert lr.CLUE in lr.CLUE_FACT["value"]