diff --git a/backend/app/narrative/extract.py b/backend/app/narrative/extract.py index 8ba6622..7e20eb2 100644 --- a/backend/app/narrative/extract.py +++ b/backend/app/narrative/extract.py @@ -226,21 +226,36 @@ def _strip_dangling_object(text: str) -> str: return text +# The markdown a model wraps a heading in: `## Established:`, `**Held:**`, +# `> Held:`. +_HEADING_DECORATION_RE = re.compile(r"^[\s#>*_]+|[\s*_]+$") + + +def _section_heading(line: str) -> str | None: + """The state-section heading this line is, markdown aside, or None.""" + bare = _HEADING_DECORATION_RE.sub("", line) + return bare if bare in render.SECTION_HEADINGS else None + + def _strip_echoed_state(text: str) -> str: """Removes a copy of the narrative-state section pasted into the prose. - Judged by the section's own headings (`render.SECTION_HEADINGS`), as whole - lines. A block qualifies only when it carries two of them, or one and the - scene line directly above it. A story may contain a line reading "Held:", - and one heading on its own is left there. The block runs over the headings, - their indented entries and the blank lines between them, and stops at the - first line of ordinary prose. + Judged by the section's own headings (`render.SECTION_HEADINGS`) as whole + lines, with any markdown the model wrapped them in taken off. A block + qualifies when it carries two headings, or one and the scene line directly + above it, or one heading with an indented entry under it. That last case + is the model writing a section of its own: the M04 re-run found + `## Established:` over two indented facts on 5 turns, one of them copying + the planted clue out of the state section. A lone "Held:" with prose after + it is still somebody's story. The block runs over the headings, their + indented entries and the blank lines between them, and stops at the first + line of ordinary prose. """ lines = text.split("\n") drop = [False] * len(lines) index = 0 while index < len(lines): - if lines[index].strip() not in render.SECTION_HEADINGS: + if _section_heading(lines[index]) is None: index += 1 continue start = index @@ -254,20 +269,23 @@ def _strip_echoed_state(text: str) -> str: if scene: start = above headings: set[str] = set() + entries = 0 end = index cursor = index while cursor < len(lines): line = lines[cursor] stripped = line.strip() - if stripped in render.SECTION_HEADINGS: - headings.add(stripped) + heading = _section_heading(line) + if heading is not None: + headings.add(heading) end = cursor elif stripped and line[:1] in (" ", "\t"): + entries += 1 end = cursor elif stripped: break cursor += 1 - if len(headings) + (1 if scene else 0) >= 2: + if len(headings) + (1 if scene else 0) >= 2 or (headings and entries): for position in range(start, end + 1): drop[position] = True index = end + 1 diff --git a/backend/tests/test_m11_long_run_memory.py b/backend/tests/test_m11_long_run_memory.py index 504d994..aa00ef6 100644 --- a/backend/tests/test_m11_long_run_memory.py +++ b/backend/tests/test_m11_long_run_memory.py @@ -273,13 +273,14 @@ def _with_adv(run): # ------------------------------------------------------- what M04 actually proved -def test_the_m04_verdict_never_credits_a_clue_still_in_recent_history(): - """The first long run with the bank on found the clue at turn 100 only - because the narrator had pasted the state into recent history.""" - base = {"clue_in_recent_history_window": False, "in_memories_section": False, - "in_summary_section": False, "in_state_section": False} - assert lr._m04_verdict({**base, "clue_in_recent_history_window": True, +def test_the_m04_verdict_never_credits_a_planted_turn_still_in_the_window(): + base = {"planted_turn_in_history_window": False, "in_memories_section": False, + "in_summary_section": False, "in_state_section": False, + "clue_in_recent_history_window": False} + assert lr._m04_verdict({**base, "planted_turn_in_history_window": True, "in_memories_section": True}) == "precondition_not_met" + assert lr._m04_verdict({**base, "planted_turn_in_history_window": None, + "in_state_section": True}) == "precondition_unknown" assert lr._m04_verdict({**base, "in_memories_section": True}) == \ "recovered_through_memory_or_summary" assert lr._m04_verdict({**base, "in_summary_section": True}) == \ @@ -289,12 +290,35 @@ def test_the_m04_verdict_never_credits_a_clue_still_in_recent_history(): assert lr._m04_verdict(base) == "not_recovered" +def test_the_sentinel_in_recent_history_does_not_decide_the_precondition(): + """The M04 re-run: the narrator reused the sentinel in its own prose while + the planted turn was 65 actions outside the window.""" + recall = {"planted_turn_in_history_window": False, + "clue_in_recent_history_window": True, + "in_memories_section": False, "in_summary_section": False, + "in_state_section": True} + assert lr._m04_verdict(recall) == "recovered_through_state_only" + + +def test_the_planted_depth_survives_a_resume(run_for, tmp_path): + first = run_for(Reports(tmp_path / "server.log")) + first.adv, first.planted_depth = 1, 1 + first.save_resume() + + second = run_for(Reports(tmp_path / "server.log")) + second.adopt(json.loads((tmp_path / lr.RESUME_FILE).read_text())) + assert second.planted_depth == 1 + + def test_protocol_left_in_stored_narration_is_counted(): bundle = {"actions": [ {"id": 1, "type": "do", "text": '> You say {"events": []}'}, {"id": 2, "type": "ai", "text": "The rain eases."}, {"id": 3, "type": "ai", "text": "Beat.\n\nWho and what exists:\n mara: Mara"}, {"id": 4, "type": "ai", "text": 'Beat.\n\n> {"events": []}'}, + {"id": 5, "type": "ai", + "text": "Rain.\n\n## Established:\n the crypt is sealed (SENTINEL)"}, + {"id": 6, "type": "ai", "text": "The notice read:\n\nHeld:\nnothing at all."}, ]} assert lr._protocol_leaks(bundle) == { - "ai_actions": 3, "leaking": 2, "example_ids": [3, 4]} + "ai_actions": 5, "leaking": 3, "example_ids": [3, 4, 5]} diff --git a/backend/tests/test_narrative_state.py b/backend/tests/test_narrative_state.py index ad4a415..cbf6e28 100644 --- a/backend/tests/test_narrative_state.py +++ b/backend/tests/test_narrative_state.py @@ -1767,6 +1767,39 @@ def test_a_parroted_continue_hint_cut_off_mid_sentence_is_not_story(): assert prose == "Aldric's steps were firm." +def test_a_section_of_its_own_under_a_markdown_heading_is_not_story(): + """The M04 re-run: the narrator wrote its own `## Established:` with the + planted clue copied into it, which kept the clue in recent history on a + turn the extractor had passed as clean.""" + reply = ( + "The rain outside seems to echo their uncertainty.\n\n" + "## Established:\n" + " the silver key has an enchantment that unlocks the sealed crypt of the Old Abbey\n" + " The Old Abbey's crypt is sealed (SILVER-KEY-CRYPT-OLD-ABBEY)" + ) + prose, _parsed, _raw = extract.split(reply) + assert prose == "The rain outside seems to echo their uncertainty." + assert "SILVER-KEY" not in prose + + +def test_a_bold_heading_mid_story_goes_and_the_story_either_side_stays(): + reply = ( + "Mara takes the key.\n\n**Held:**\n the silver key — Mara\n\n" + "They step out into the rain." + ) + prose, _parsed, _raw = extract.split(reply) + assert prose == "Mara takes the key.\n\nThey step out into the rain." + + +def test_decorated_headings_count_toward_a_pasted_section(): + reply = ( + "Beat.\n\n### Who and what exists:\n mara: Mara (character)\n\n" + "### Held:\n the key — Mara" + ) + prose, _parsed, _raw = extract.split(reply) + assert prose == "Beat." + + def test_a_pasted_state_section_with_no_block_records_no_block(): """A paste is not a proposal, so the turn is not marked unparseable.""" prose, parsed, raw = extract.split("The rain eases.\n\n" + PASTED_STATE) @@ -1776,7 +1809,7 @@ def test_a_pasted_state_section_with_no_block_records_no_block(): @pytest.mark.parametrize("reply", [ - "The notice on the door read:\n\nHeld:\n nothing, by order of the Watch.", + "The notice on the door read:\n\nHeld:\nnothing, by order of the Watch.", "He chalked the first mark of a sum on the wall:\n{", "The sign said only [continue at your own risk", 'She typed it out:\n```json\n{"name":', diff --git a/backend/tools/m11_long_run.py b/backend/tools/m11_long_run.py index 1c1f051..2f50289 100644 --- a/backend/tools/m11_long_run.py +++ b/backend/tools/m11_long_run.py @@ -98,6 +98,7 @@ from __future__ import annotations import argparse import json import os +import re import socket import subprocess import sys @@ -361,6 +362,9 @@ class Run: #: How far `background_failures` has read `server.log`. Carried across a #: resume, so a failure is reported once, not again every session. self.log_offset = 0 + #: The depth of the player turn that planted the clue. M04's + #: precondition is that this turn has left the history window. + self.planted_depth: int | None = None # ------------------------------------------------------------ recording @@ -388,6 +392,7 @@ class Run: "elapsed_seconds": self.elapsed(), "turns_target": self.turns_target, "log_offset": self.log_offset, + "planted_depth": self.planted_depth, "written": datetime.now().isoformat(timespec="seconds"), } tmp = self.out / (RESUME_FILE + ".tmp") @@ -408,6 +413,7 @@ class Run: self.completed_steps = set(prior.get("completed_steps") or []) self.elapsed_before = prior.get("elapsed_seconds", 0) self.log_offset = prior.get("log_offset", 0) + self.planted_depth = prior.get("planted_depth") self.resumed = True def reattach(self) -> None: @@ -739,7 +745,17 @@ def main() -> int: run.setup() # ---- The planted clue, at the very beginning. ---- - run.turn(f"I tell Mara quietly that {CLUE} — {CLUE_SENTINEL}.") + planting = run.turn(f"I tell Mara quietly that {CLUE} — {CLUE_SENTINEL}.") + if not planting.get("accepted"): + raise SystemExit( + "the turn that plants the clue was not accepted, so M04 has " + "no planted turn to measure. Stopping before the campaign starts.") + # Where the planted turn sits, which is what M04's precondition is + # measured against. No endpoint reports an action's depth, but a + # fresh campaign's path is the opening (0), this player turn (1) and + # its reply (2), and the history report counts that path. It is + # checked against the database in the M04 re-run evidence. + run.planted_depth = planting["total_actions"] - 2 run.correct([CLUE_FACT], note="the planted clue, as accepted state") # Proved here, at turn one, where it costs a single request. M04 # asks whether the clue is still recoverable a hundred turns later, @@ -749,7 +765,7 @@ def main() -> int: planted = any(CLUE_SENTINEL in json.dumps(fact) for fact in run.state()["document"].get("facts") or []) run.note("clue_planted", sentinel=CLUE_SENTINEL, - verified_in_state=planted) + verified_in_state=planted, planted_depth=run.planted_depth) if not planted: raise SystemExit( "the planted clue is not in accepted state, so M04 cannot " @@ -1119,6 +1135,16 @@ def _recall(run: Run) -> dict: history_text = " ".join( s["text"] for s in report["sections"] if s["label"] in HISTORY_LABELS) in_history = CLUE_SENTINEL in history_text + # The precondition proper: has the planted turn itself left the window? + # The sentinel's text is no guide. The narrator reuses it in its own prose, + # once copied from the state section and once as a name, so it can be in + # recent history long after the planted turn is gone. `floor_depth` is None + # when the whole path is included. + floor = report["history"].get("floor_depth") + if run.planted_depth is None: + planted_in_window = None + else: + planted_in_window = floor is None or floor <= run.planted_depth # 2. Ask about the subject, and see what the application assembles. run.turn("I think back to what I told Mara about the key, that first night.") @@ -1131,7 +1157,12 @@ def _recall(run: Run) -> dict: fact_present = any( CLUE_SENTINEL in json.dumps(f) for f in document.get("facts") or []) result = { + # The sentinel's text in recent history, from any source. A fact about + # the prompt, kept for the report, and no longer the precondition. "clue_in_recent_history_window": in_history, + "planted_depth": run.planted_depth, + "history_floor_depth": floor, + "planted_turn_in_history_window": planted_in_window, "clue_in_prompt": CLUE_SENTINEL in whole_prompt, "in_state_section": CLUE_SENTINEL in sections.get(STATE_LABEL, ""), "in_summary_section": CLUE_SENTINEL in sections.get(SUMMARY_LABEL, ""), @@ -1161,12 +1192,19 @@ def _recall(run: Run) -> dict: def _m04_verdict(recall: dict) -> str: """What the recall check proved, in one word the report can quote. - The first long run with the bank on found the clue "in the prompt" at turn - 100, but only because the narrator had pasted the state section into its - prose and the paste was still in recent history. A clue in recent history is - no evidence about memory, so that case gets its own verdict and never counts - as recovery.""" - if recall["clue_in_recent_history_window"]: + M04 passes when the planted fact is recoverable "without entire transcript + in prompt" (`V1-ACCEPTANCE-TESTS.md`). So the precondition is that the + planted turn has left the history window. Any recovery path then counts: + memory, summary or authoritative state. The owner accepted state-only + recovery on 2026-09-13. + + An earlier version used the sentinel's text in recent history as the + precondition. The narrator reuses that text in its own prose, so a run whose + planted turn was 65 actions out of the window read `precondition_not_met`.""" + in_window = recall.get("planted_turn_in_history_window") + if in_window is None: + return "precondition_unknown" + if in_window: return "precondition_not_met" if recall["in_memories_section"] or recall["in_summary_section"]: return "recovered_through_memory_or_summary" @@ -1175,18 +1213,28 @@ def _m04_verdict(recall: dict) -> str: return "not_recovered" -#: Signs the application stored protocol as story: the state section's own -#: headings, and a proposal's event list. Copied rather than imported from -#: `app.narrative.render`, for the reason `HISTORY_LABELS` is copied. -PROTOCOL_LEAK_MARKERS = ("Who and what exists:", "\nHeld:\n", "\nEstablished:\n", - '"events"') +#: Signs the application stored protocol as story. The first is a state-section +#: heading with an indented entry under it, in any markdown, because +#: `## Established:` got past a plain substring match and the count read 1 +#: where 5 turns leaked. The second is a proposal's event list. Both are copied +#: rather than imported from `app.narrative`, for the reason `HISTORY_LABELS` is. +PROTOCOL_LEAK_HEADING_RE = re.compile( + r"^[ \t#>*_]*(?:Who and what exists|Held|Established" + r"|No longer true — do not treat these as established|Between them|Still open)" + r":[ \t*_]*\n[ \t]+\S", + re.MULTILINE, +) +PROTOCOL_LEAK_EVENTS = '"events"' + + +def _leaks_protocol(text: str) -> bool: + return bool(PROTOCOL_LEAK_HEADING_RE.search(text)) or PROTOCOL_LEAK_EVENTS in text def _protocol_leaks(bundle: dict) -> dict: """How many stored AI turns still carry protocol, on every branch.""" ai = [a for a in bundle.get("actions") or [] if a.get("type") == "ai"] - leaking = [a for a in ai - if any(marker in (a.get("text") or "") for marker in PROTOCOL_LEAK_MARKERS)] + leaking = [a for a in ai if _leaks_protocol(a.get("text") or "")] return {"ai_actions": len(ai), "leaking": len(leaking), "example_ids": [a.get("id") for a in leaking[:10]]}