Diagnostic only; no memory behaviour changes. - tools/memory_diagnostic.py: planted-fact isolation checks, the four-stage diagnosis (created / retained / ranked / injected) with a verdict, a production-ranking replica, deterministic summariser/embedder/narrator stubs and seven scenarios (default, past capacity, pinned, low top_k, long-block early/late, lineage control) - tools/v11_b1_memory.py: CLI for the scenarios and for diagnosing a copy of a finished real campaign - tools/m11_long_run.py: opt-in --independent-fact mode with per-turn isolation tracking and the recovered_through_memory_independent verdict; M04 verdicts unchanged - tests: diagnostic stages, eviction, creation window, ranking, lineage and authority controls; two strict xfails record the diagnosed retention and creation defects for WP-B.2 to flip - planning/reports/v1.1/V1.1-WP-B1-REPORT.md First failing stage: ranking (real model); retention past capacity and creation for early facts in long blocks (deterministic, same on v1.0.0). Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY
87 lines
3.4 KiB
Python
87 lines
3.4 KiB
Python
"""v1.1 WP-B.1: the long run's `recovered_through_memory_independent` verdict.
|
|
|
|
The new verdict must never be reported when anything other than memory could
|
|
have carried the fact. Each precondition is named when it fails. The existing M04
|
|
verdicts keep their meaning exactly.
|
|
|
|
python -m pytest tests/test_v11_b1_long_run_verdict.py -v
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from tools import m11_long_run as lr
|
|
|
|
GOOD = {
|
|
"independent_planted_depth": 3,
|
|
"planted_turn_outside_history": True,
|
|
"absent_from_state": True,
|
|
"absent_from_summary": True,
|
|
"absent_from_knowledge": True,
|
|
"absent_from_later_narration": True,
|
|
"memory_covering_planting_carries_fact": True,
|
|
"memory_forgotten": False,
|
|
"memory_injected": True,
|
|
}
|
|
|
|
|
|
def test_every_precondition_and_an_injected_memory_is_the_new_verdict():
|
|
assert lr._independent_memory_verdict(GOOD) == "recovered_through_memory_independent"
|
|
|
|
|
|
@pytest.mark.parametrize("name", lr.INDEPENDENT_PRECONDITIONS)
|
|
def test_a_failed_precondition_is_named_and_never_a_recovery(name):
|
|
assert lr._independent_memory_verdict({**GOOD, name: False}) == f"precondition_failed:{name}"
|
|
|
|
|
|
@pytest.mark.parametrize("name", lr.INDEPENDENT_PRECONDITIONS)
|
|
def test_an_unmeasured_precondition_is_unknown_not_a_pass(name):
|
|
assert lr._independent_memory_verdict({**GOOD, name: None}) == f"precondition_unknown:{name}"
|
|
|
|
|
|
def test_no_planted_depth_is_unknown():
|
|
assert lr._independent_memory_verdict({**GOOD, "independent_planted_depth": None}) == \
|
|
"precondition_unknown:planted_depth"
|
|
|
|
|
|
@pytest.mark.parametrize("change, verdict", [
|
|
({"memory_covering_planting_carries_fact": False}, "not_recovered:not_created"),
|
|
({"memory_forgotten": True}, "not_recovered:evicted"),
|
|
({"memory_injected": False}, "not_recovered:not_injected"),
|
|
])
|
|
def test_the_failing_memory_stage_is_named(change, verdict):
|
|
assert lr._independent_memory_verdict({**GOOD, **change}) == verdict
|
|
|
|
|
|
def test_preconditions_are_judged_before_memory():
|
|
"""A carried fact disqualifies the run even when memory also failed."""
|
|
both = {**GOOD, "absent_from_state": False, "memory_covering_planting_carries_fact": False}
|
|
assert lr._independent_memory_verdict(both) == "precondition_failed:absent_from_state"
|
|
|
|
|
|
def test_the_fact_is_matched_as_whole_words():
|
|
assert lr._mentions_fact("She hid the amber Sundial.")
|
|
assert lr._mentions_fact("a cracked TEAPOT on the shelf")
|
|
assert not lr._mentions_fact("teapots") # a different word, not the fact's
|
|
assert not lr._mentions_fact("the sun dialled down")
|
|
|
|
|
|
def test_the_m04_verdicts_are_unchanged():
|
|
base = {"planted_turn_in_history_window": False, "in_memories_section": False,
|
|
"in_summary_section": False, "in_state_section": False}
|
|
assert lr._m04_verdict(base) == "not_recovered"
|
|
assert lr._m04_verdict({**base, "in_state_section": True}) == "recovered_through_state_only"
|
|
assert lr._m04_verdict({**base, "in_memories_section": True}) == \
|
|
"recovered_through_memory_or_summary"
|
|
assert lr._m04_verdict({**base, "planted_turn_in_history_window": True}) == \
|
|
"precondition_not_met"
|
|
|
|
|
|
def test_the_independent_fact_is_not_in_any_imported_knowledge_file():
|
|
for text in (lr.CANON_MD, lr.REFERENCE_MD, lr.INSPIRATION_MD, *lr.BEATS):
|
|
assert not lr._mentions_fact(text)
|
|
|
|
|
|
def test_the_planting_text_and_recall_carry_the_fact():
|
|
assert lr._mentions_fact(lr.INDEPENDENT_FACT_TEXT)
|
|
assert lr._mentions_fact(lr.INDEPENDENT_RECALL_TEXT)
|