Files
interactive-story/backend/tests/test_v11_b1_long_run_verdict.py
JesseMarkowitzandClaude Opus 5 beb17ada10 v1.1 WP-B.1: diagnose independent long-term memory retention
Diagnostic only; no memory behaviour changes.

- tools/memory_diagnostic.py: planted-fact isolation checks, the four-stage
  diagnosis (created / retained / ranked / injected) with a verdict, a
  production-ranking replica, deterministic summariser/embedder/narrator
  stubs and seven scenarios (default, past capacity, pinned, low top_k,
  long-block early/late, lineage control)
- tools/v11_b1_memory.py: CLI for the scenarios and for diagnosing a copy of
  a finished real campaign
- tools/m11_long_run.py: opt-in --independent-fact mode with per-turn
  isolation tracking and the recovered_through_memory_independent verdict;
  M04 verdicts unchanged
- tests: diagnostic stages, eviction, creation window, ranking, lineage and
  authority controls; two strict xfails record the diagnosed retention and
  creation defects for WP-B.2 to flip
- planning/reports/v1.1/V1.1-WP-B1-REPORT.md

First failing stage: ranking (real model); retention past capacity and
creation for early facts in long blocks (deterministic, same on v1.0.0).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY
2026-09-14 20:50:05 -04:00

87 lines
3.4 KiB
Python

"""v1.1 WP-B.1: the long run's `recovered_through_memory_independent` verdict.
The new verdict must never be reported when anything other than memory could
have carried the fact. Each precondition is named when it fails. The existing M04
verdicts keep their meaning exactly.
python -m pytest tests/test_v11_b1_long_run_verdict.py -v
"""
import pytest
from tools import m11_long_run as lr
GOOD = {
"independent_planted_depth": 3,
"planted_turn_outside_history": True,
"absent_from_state": True,
"absent_from_summary": True,
"absent_from_knowledge": True,
"absent_from_later_narration": True,
"memory_covering_planting_carries_fact": True,
"memory_forgotten": False,
"memory_injected": True,
}
def test_every_precondition_and_an_injected_memory_is_the_new_verdict():
assert lr._independent_memory_verdict(GOOD) == "recovered_through_memory_independent"
@pytest.mark.parametrize("name", lr.INDEPENDENT_PRECONDITIONS)
def test_a_failed_precondition_is_named_and_never_a_recovery(name):
assert lr._independent_memory_verdict({**GOOD, name: False}) == f"precondition_failed:{name}"
@pytest.mark.parametrize("name", lr.INDEPENDENT_PRECONDITIONS)
def test_an_unmeasured_precondition_is_unknown_not_a_pass(name):
assert lr._independent_memory_verdict({**GOOD, name: None}) == f"precondition_unknown:{name}"
def test_no_planted_depth_is_unknown():
assert lr._independent_memory_verdict({**GOOD, "independent_planted_depth": None}) == \
"precondition_unknown:planted_depth"
@pytest.mark.parametrize("change, verdict", [
({"memory_covering_planting_carries_fact": False}, "not_recovered:not_created"),
({"memory_forgotten": True}, "not_recovered:evicted"),
({"memory_injected": False}, "not_recovered:not_injected"),
])
def test_the_failing_memory_stage_is_named(change, verdict):
assert lr._independent_memory_verdict({**GOOD, **change}) == verdict
def test_preconditions_are_judged_before_memory():
"""A carried fact disqualifies the run even when memory also failed."""
both = {**GOOD, "absent_from_state": False, "memory_covering_planting_carries_fact": False}
assert lr._independent_memory_verdict(both) == "precondition_failed:absent_from_state"
def test_the_fact_is_matched_as_whole_words():
assert lr._mentions_fact("She hid the amber Sundial.")
assert lr._mentions_fact("a cracked TEAPOT on the shelf")
assert not lr._mentions_fact("teapots") # a different word, not the fact's
assert not lr._mentions_fact("the sun dialled down")
def test_the_m04_verdicts_are_unchanged():
base = {"planted_turn_in_history_window": False, "in_memories_section": False,
"in_summary_section": False, "in_state_section": False}
assert lr._m04_verdict(base) == "not_recovered"
assert lr._m04_verdict({**base, "in_state_section": True}) == "recovered_through_state_only"
assert lr._m04_verdict({**base, "in_memories_section": True}) == \
"recovered_through_memory_or_summary"
assert lr._m04_verdict({**base, "planted_turn_in_history_window": True}) == \
"precondition_not_met"
def test_the_independent_fact_is_not_in_any_imported_knowledge_file():
for text in (lr.CANON_MD, lr.REFERENCE_MD, lr.INSPIRATION_MD, *lr.BEATS):
assert not lr._mentions_fact(text)
def test_the_planting_text_and_recall_carry_the_fact():
assert lr._mentions_fact(lr.INDEPENDENT_FACT_TEXT)
assert lr._mentions_fact(lr.INDEPENDENT_RECALL_TEXT)