Diagnostic only; no memory behaviour changes. - tools/memory_diagnostic.py: planted-fact isolation checks, the four-stage diagnosis (created / retained / ranked / injected) with a verdict, a production-ranking replica, deterministic summariser/embedder/narrator stubs and seven scenarios (default, past capacity, pinned, low top_k, long-block early/late, lineage control) - tools/v11_b1_memory.py: CLI for the scenarios and for diagnosing a copy of a finished real campaign - tools/m11_long_run.py: opt-in --independent-fact mode with per-turn isolation tracking and the recovered_through_memory_independent verdict; M04 verdicts unchanged - tests: diagnostic stages, eviction, creation window, ranking, lineage and authority controls; two strict xfails record the diagnosed retention and creation defects for WP-B.2 to flip - planning/reports/v1.1/V1.1-WP-B1-REPORT.md First failing stage: ranking (real model); retention past capacity and creation for early facts in long blocks (deterministic, same on v1.0.0). Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY
309 lines
14 KiB
Python
309 lines
14 KiB
Python
"""v1.1 WP-B.1: the memory-retention diagnostic, deterministically.
|
|
|
|
B.1 changes no memory behaviour. These tests prove two things about the
|
|
diagnostic in `tools/memory_diagnostic.py`:
|
|
|
|
1. **It measures what it claims.**
|
|
- The fixture keeps the planted fact out of every layer except memory.
|
|
- Each stage (created, retained, ranked, injected) is reported from the rows
|
|
and the recall turn's own stored context.
|
|
- Its ranking agrees with the selection production stored.
|
|
2. **What it finds on this tree.** The scenarios run with a best-case summariser,
|
|
one that keeps a fact if and only if the fact reached it. Any failure is
|
|
therefore the application's mechanism, not a model's writing.
|
|
- The criteria the current code does not meet are marked `xfail(strict=True)`,
|
|
so B.2 has to flip them deliberately.
|
|
- The same file is run unchanged against v1.0.0 for the baseline.
|
|
|
|
python -m pytest tests/test_v11_b1_memory_diagnostic.py -v
|
|
"""
|
|
|
|
import asyncio
|
|
|
|
import pytest
|
|
from fastapi import Depends
|
|
from fastapi.testclient import TestClient
|
|
from sqlalchemy.orm import undefer
|
|
|
|
from app import auth, limits, memorybank, models
|
|
from app.database import Base, SessionLocal, engine, get_db
|
|
from app.main import app
|
|
from app.routers import adventures
|
|
from tools import memory_diagnostic as md
|
|
|
|
_results: dict = {}
|
|
|
|
|
|
def scenario(name: str) -> dict:
|
|
"""Runs a named scenario once per session and keeps the result."""
|
|
if name not in _results:
|
|
_results[name] = md.run_scenario(md.SCENARIOS[name])
|
|
return _results[name]
|
|
|
|
|
|
# ------------------------------------------------------- fixture preconditions
|
|
|
|
def test_the_fact_is_planted_early_and_recalled_past_depth_one_hundred():
|
|
result = scenario("independent_default")
|
|
assert result["plant_depth"] is not None and result["plant_depth"] <= 3
|
|
assert result["recall_depth"] >= 100
|
|
|
|
|
|
@pytest.mark.parametrize("check", ["state_document", "state_snapshots", "later_narration",
|
|
"summary", "knowledge", "recent_history", "state_section"])
|
|
def test_no_layer_but_memory_carries_the_fact(check):
|
|
"""A test where another layer carries F is not evidence about memory."""
|
|
isolation = scenario("independent_default")["isolation"]
|
|
assert isolation["checks"][check]["ok"], isolation["checks"][check]
|
|
assert isolation["ok"]
|
|
|
|
|
|
def test_the_isolation_check_fails_when_another_layer_carries_the_fact():
|
|
"""The negative control for the precondition itself: a state fact naming F."""
|
|
fact = md.FACT_F
|
|
with SessionLocal() as db:
|
|
Base.metadata.create_all(bind=engine)
|
|
try:
|
|
user = models.User(is_guest=False, email="b1-iso@example.com")
|
|
db.add(user)
|
|
db.flush()
|
|
adventure = models.Adventure(user_id=user.id, title="iso")
|
|
adventure.narrative_state = {"facts": [{"id": "x", "predicate": "hidden",
|
|
"value": "the amber sundial is in the teapot"}]}
|
|
db.add(adventure)
|
|
db.commit()
|
|
result = md.isolation(db, adventure, fact, 1)
|
|
assert result["ok"] is False
|
|
assert result["checks"]["state_document"]["ok"] is False
|
|
finally:
|
|
db.close()
|
|
Base.metadata.drop_all(bind=engine)
|
|
|
|
|
|
# ------------------------------------------------------------------- stages
|
|
|
|
def test_creation_is_reported_with_the_covering_memory_and_what_the_summariser_saw():
|
|
created = scenario("independent_default")["diagnosis"]["created"]
|
|
assert created["yes"] is True
|
|
assert created["source_start"] <= scenario("independent_default")["plant_depth"] <= created["source_end"]
|
|
assert md.FACT_F.carried_by(created["memory_text"])
|
|
covering = [c for c in created["covering_memories"] if c["memory_id"] == created["memory_id"]]
|
|
assert covering and covering[0]["fact_in_block"] and covering[0]["fact_in_summariser_excerpt"]
|
|
|
|
|
|
def test_retention_is_reported_with_the_bank_and_its_eviction_order():
|
|
retained = scenario("independent_default")["diagnosis"]["retained"]
|
|
assert retained["yes"] is True and retained["forgotten"] is False
|
|
assert retained["on_active_lineage"] is True
|
|
assert retained["active_memories"] <= retained["memory_bank_capacity"]
|
|
assert retained["eviction_position"] is not None
|
|
|
|
|
|
def test_ranking_is_production_ranking_and_agrees_with_the_stored_selection():
|
|
ranked = scenario("independent_default")["diagnosis"]["ranked"]
|
|
assert ranked["replica_matches_stored_selection"] is True
|
|
assert ranked["lexical_score"] is None # memory ranking has no lexical term
|
|
assert ranked["top_k_cutoff"] == 5
|
|
assert ranked["yes"] is True and ranked["selected"] is True
|
|
assert 1 <= ranked["rank"] <= ranked["top_k_cutoff"]
|
|
# The production query is the newest four actions, cut to 600 tokens, and the
|
|
# one-line question is diluted by the narration around it.
|
|
variants = scenario("independent_default")["ranking_variants"]
|
|
assert ranked["semantic_score"] < variants["direct"]["similarity"]
|
|
|
|
|
|
def test_injection_is_read_from_the_recall_turns_own_context():
|
|
diagnosis = scenario("independent_default")["diagnosis"]
|
|
assert diagnosis["injected"]["yes"] is True
|
|
assert diagnosis["injected"]["context_component"] == md.MEMORIES_LABEL
|
|
assert diagnosis["injected"]["token_count"] > 0
|
|
assert diagnosis["verdict"] == "injected"
|
|
|
|
|
|
def test_ranking_variants_direct_paraphrase_and_unrelated():
|
|
variants = scenario("independent_default")["ranking_variants"]
|
|
assert variants["direct"]["rank"] == 1 and variants["direct"]["selected"]
|
|
assert variants["paraphrase"]["rank"] == 1 and variants["paraphrase"]["selected"]
|
|
assert (variants["direct"]["similarity"] > variants["paraphrase"]["similarity"]
|
|
> 5 * variants["unrelated"]["similarity"])
|
|
|
|
|
|
def test_retrieval_fills_top_k_whatever_the_similarity():
|
|
"""Diagnosis: there is no relevance floor. With more memories than
|
|
`memory_top_k`, an unrelated query still selects five, and the early fact
|
|
rides along at a similarity near zero."""
|
|
variants = scenario("independent_default")["ranking_variants"]
|
|
assert variants["unrelated"]["similarity"] < 0.1
|
|
assert variants["unrelated"]["selected"] is True
|
|
|
|
|
|
# ---------------------------------------------------------- capacity/eviction
|
|
|
|
def test_past_capacity_the_early_memory_is_evicted_and_the_stage_says_so():
|
|
"""Diagnosis, not a requirement: what the current eviction rule does to F."""
|
|
result = scenario("past_capacity")
|
|
assert result["diagnosis"]["created"]["yes"] is True
|
|
assert result["diagnosis"]["verdict"] == "created_but_evicted"
|
|
eviction = result["eviction"]
|
|
assert eviction["f_evicted_at_turn"] is not None
|
|
# It was retrieved while the bank was small, stopped being retrieved once
|
|
# recent narration filled the top-k, and was then the least recently used.
|
|
assert eviction["f_use_count_when_evicted"] > 0
|
|
assert result["f_last_use_increase_turn"] < eviction["f_evicted_at_turn"]
|
|
assert eviction["f_memory_was_first_evicted"] is True
|
|
|
|
|
|
def test_at_a_lower_top_k_the_early_memory_ages_out_after_it_stops_being_retrieved():
|
|
"""Diagnosis with most of the bank unretrieved on any turn, nearer the
|
|
shipped 5-in-80 ratio. F is not simply the oldest row: it is evicted some
|
|
turns after recent narration stopped pulling it into the top-k, which is
|
|
what ordering by last use does to a fact nothing recent mentions."""
|
|
result = scenario("past_capacity_low_top_k")
|
|
eviction = result["eviction"]
|
|
assert result["diagnosis"]["created"]["yes"] is True
|
|
assert result["diagnosis"]["verdict"] == "created_but_evicted"
|
|
assert eviction["f_use_count_when_evicted"] > 0
|
|
assert result["f_last_use_increase_turn"] < eviction["f_evicted_at_turn"]
|
|
assert eviction["first_eviction_turn"] <= eviction["f_evicted_at_turn"]
|
|
assert eviction["created_and_evicted_same_turn"] == []
|
|
|
|
|
|
def test_no_memory_is_evicted_by_the_same_pass_that_created_it():
|
|
"""The frozen-bank regression the current rule fixed, still holding."""
|
|
for name in ("past_capacity", "past_capacity_pinned"):
|
|
assert scenario(name)["eviction"]["created_and_evicted_same_turn"] == []
|
|
|
|
|
|
def test_a_pinned_memory_survives_capacity():
|
|
eviction = scenario("past_capacity_pinned")["eviction"]
|
|
assert eviction["pinned_memory_id"] is not None
|
|
assert eviction["pinned_memory_forgotten"] is False
|
|
|
|
|
|
@pytest.mark.xfail(strict=True, reason=(
|
|
"WP-B.1 diagnosis on this tree: past memory_bank_capacity the planting-era "
|
|
"memory is evicted first, because it was never retrieved and eviction orders "
|
|
"by last use, then creation. B.2 must flip this deliberately."))
|
|
def test_acceptance_an_early_fact_is_recalled_from_memory_past_capacity():
|
|
assert scenario("past_capacity")["diagnosis"]["verdict"] == "injected"
|
|
|
|
|
|
# ---------------------------------------------------------- creation window
|
|
|
|
def test_a_fact_early_in_a_long_block_never_reaches_the_summariser():
|
|
result = scenario("long_block_fact_early")
|
|
created = result["diagnosis"]["created"]
|
|
covering = created["covering_memories"]
|
|
assert covering, "the long block must have been summarised"
|
|
assert covering[0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS
|
|
assert covering[0]["fact_in_block"] is True
|
|
assert covering[0]["fact_in_summariser_excerpt"] is False
|
|
assert result["diagnosis"]["verdict"] == "not_created"
|
|
|
|
|
|
def test_the_same_fact_late_in_the_same_sized_block_does():
|
|
result = scenario("long_block_fact_late")
|
|
covering = result["diagnosis"]["created"]["covering_memories"]
|
|
assert covering[0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS
|
|
assert covering[0]["fact_in_summariser_excerpt"] is True
|
|
assert result["diagnosis"]["created"]["yes"] is True
|
|
|
|
|
|
@pytest.mark.xfail(strict=True, reason=(
|
|
"WP-B.1 diagnosis on this tree: the summariser reads only the last "
|
|
f"{memorybank.MEMORY_EXCERPT_TOKENS} tokens of a block, so a fact early in a "
|
|
"long block is never seen. B.2 must flip this deliberately."))
|
|
def test_acceptance_a_fact_early_in_a_long_block_is_remembered():
|
|
assert scenario("long_block_fact_early")["diagnosis"]["created"]["yes"] is True
|
|
|
|
|
|
# ------------------------------------------------------- lineage control (G)
|
|
|
|
def test_an_abandoned_lines_memory_is_stored_but_never_eligible_or_injected():
|
|
g = scenario("lineage_control")["lineage_control"]
|
|
assert g["memory_ids"], "G's memory must exist on line A before it is abandoned"
|
|
assert sorted(g["stored"]) == sorted(g["memory_ids"])
|
|
assert g["eligible_on_active_line"] == []
|
|
assert g["g_text_ever_in_used_memories"] is False
|
|
# Any turn that did name G's memory was on line A, before the divergence.
|
|
assert g["eligible_after_returning_to_line_a"] == g["memory_ids"]
|
|
|
|
|
|
def test_the_lineage_scenario_still_diagnoses_f_on_the_active_line():
|
|
result = scenario("lineage_control")
|
|
assert result["isolation"]["ok"], result["isolation"]
|
|
assert result["diagnosis"]["verdict"] == "injected"
|
|
|
|
|
|
# ----------------------------------------------------- authority control
|
|
|
|
@pytest.fixture()
|
|
def authority_client(monkeypatch):
|
|
embedder = md.ConceptEmbedder()
|
|
Base.metadata.create_all(bind=engine)
|
|
memorybank._vector_cache.clear()
|
|
with SessionLocal() as db:
|
|
user = models.User(is_guest=False, email="b1-auth@example.com")
|
|
db.add(user)
|
|
db.flush()
|
|
db.add(models.Settings(user_id=user.id, model="script",
|
|
endpoint_url="http://127.0.0.1:9/v1",
|
|
embedding_model="concept-embed", memory_top_k=5))
|
|
adventure = models.Adventure(user_id=user.id, title="auth", memory_bank_enabled=True,
|
|
auto_summarize=True)
|
|
db.add(adventure)
|
|
db.flush()
|
|
db.add(models.Action(adventure_id=adventure.id, type="start", text="The tavern at dusk."))
|
|
db.commit()
|
|
adv, user_id = adventure.id, user.id
|
|
monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None)
|
|
monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", md.ScriptNarrator)
|
|
monkeypatch.setattr(memorybank, "embedding_provider", lambda s: embedder)
|
|
monkeypatch.setattr(memorybank, "summary_provider", lambda s: md.BestCaseSummariser())
|
|
monkeypatch.setattr(memorybank, "schedule_post_turn", lambda a: None)
|
|
app.dependency_overrides[auth.get_current_user] = (
|
|
lambda db=Depends(get_db): db.get(models.User, user_id))
|
|
client = TestClient(app)
|
|
client.adv = adv
|
|
try:
|
|
yield client
|
|
finally:
|
|
app.dependency_overrides.clear()
|
|
adventures.turns._active_turns.clear()
|
|
memorybank._vector_cache.clear()
|
|
Base.metadata.drop_all(bind=engine)
|
|
|
|
|
|
def test_a_memory_that_contradicts_state_loses_and_changes_nothing(authority_client):
|
|
client, adv = authority_client, authority_client.adv
|
|
corrected = client.post(f"/api/adventures/{adv}/state/corrections", json={"events": [
|
|
{"type": "add_fact", "predicate": "the tavern lamp is lit", "fact_id": "lamp-lit"}]})
|
|
assert corrected.status_code in (200, 201), corrected.text[:300]
|
|
made = client.post(f"/api/adventures/{adv}/memories",
|
|
json={"text": "The tavern lamp was never lit that night."})
|
|
assert made.status_code == 201, made.text[:300]
|
|
client.patch(f"/api/adventures/{adv}/memories/{made.json()['id']}", json={"pinned": True})
|
|
asyncio.run(memorybank.run_post_turn(adv)) # embed it
|
|
|
|
before = client.get(f"/api/adventures/{adv}/state").json()["document"]
|
|
md.ScriptNarrator.next_reply = 'The fire crackles.\n```state\n{"events": []}\n```'
|
|
played = client.post(f"/api/adventures/{adv}/actions",
|
|
json={"type": "do", "text": "I look at the lamp."})
|
|
assert played.status_code == 200 and '"type": "error"' not in played.text
|
|
after = client.get(f"/api/adventures/{adv}/state").json()["document"]
|
|
assert after == before # retrieval mutated no state
|
|
|
|
with SessionLocal() as db:
|
|
action = (db.query(models.Action).filter_by(adventure_id=adv, type="ai")
|
|
.options(undefer(models.Action.context_snapshot))
|
|
.order_by(models.Action.id.desc()).first())
|
|
snapshot = action.context_snapshot
|
|
state_text = md._section(snapshot, md.STATE_LABEL)
|
|
memory_text = md._section(snapshot, md.MEMORIES_LABEL)
|
|
assert "the tavern lamp is lit" in state_text
|
|
assert "never lit" in memory_text
|
|
assert memory_text.startswith("Memories from earlier in the story")
|
|
labels = [s["label"] for s in snapshot["sections"]]
|
|
# State is read last of the live sections: it settles the conflict.
|
|
assert labels.index(md.STATE_LABEL) > labels.index(md.MEMORIES_LABEL)
|