Files
interactive-story/backend/tests/test_v11_b2_memory_excerpt.py
JesseMarkowitzandClaude Opus 5 0c1ba836ba v1.1 WP-B.2: independent long-term memory retention
Corrects the memory mechanisms WP-B.1 diagnosed, one at a time, each
verified before the next. Accepted by the owner with a documented
reference-model limitation. No schema, bundle format, setting default,
lineage, authority or protocol-cleanup change.

- B2.1 ranking: the retrieval query is the player's input plus a bounded
  scene context (state scene + end of the newest narration), embedded in
  one call. final = semantic (0.6 input / 0.4 context) + 0.15 x lexical,
  where lexical is a rarity-weighted share of the input's words, computed
  per turn over the candidates with no index. Scores and the query are
  recorded per used memory; pins and redundancy suppression unchanged.
- B2.2 coverage-aware eviction (memorybank.eviction_order): the earliest
  and newest memories are kept, the smallest coverage hole goes first,
  least-recently-used breaks ties and remains the fallback. Bounded; pins
  never evicted; frozen-bank protection kept; reads no text or vectors.
- B2.3 bounded memory creation: a block longer than 2,000 tokens is shown
  to the summariser as head + tail with an omission marker, inside the
  same budget; shorter blocks unchanged; the marker is never stored.
- The memory summariser prompt is unchanged from v1.0.0. A B2.4 prompt
  experiment was measured on the reference model, showed no reliable
  improvement for the target failure (0/5 under both prompts, with new
  "Memory:"-prefix, second-person and length regressions), and was
  reverted. memorybank.memory_user_prompt is kept as a behaviour-neutral
  helper.
- tools/memory_fidelity.py (diagnostic only): genre-neutral fixtures plus
  the failed block, a deterministic fidelity checker, and a real-model
  shipped-vs-experiment measurement.
- tools/memory_diagnostic.py: ranking replica uses production scoring;
  ranking_crowded, ranking_context_dependent and independent_full
  fixtures; per-turn isolation and provenance.
- tests: B.1's two strict xfails are now ordinary passes; ranking,
  eviction and excerpt tests; summariser acceptance tests kept apart from
  diagnostic-measurement tests.
- DEVELOPMENT.md: the GPU-host kernel/Ollama watch used `-k -u ollama`,
  which matches nothing; now the OR form.
- docs: CONTEXT-AND-MEMORY 15/18/20/21 as shipped, V1.1-PLAN (status and
  release criteria 12-13), planning README, VERSION v4.3,
  reports/v1.1/V1.1-WP-B2-REPORT.md.

Deterministic independent-memory recovery: PASS (independent_full fails
on v1.0.0 at creation and returns recovered_through_memory_independent
here). Reference-model independent recovery: FAILED on the
precondition-valid attempt, at memory creation: the summariser omitted a
player-established fact from a block it received whole. Accepted as a
documented v1.1 residual and carried into the release gate.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY
2026-09-15 11:21:53 -04:00

190 lines
7.6 KiB
Python

"""v1.1 WP-B.2 (B2.3): what the memory summariser is shown of a long block.
v1.0.0 sent the last 2,000 tokens of a block, so a fact early in a longer block
never reached the summariser (B.1 §E). A block that fits is still sent whole. A
longer one is now sent as its opening and its end, with a marker between them,
inside the same 2,000-token budget.
The scenario-level test (the planted fact early in a long block, remembered) is
in `test_v11_b1_memory_diagnostic.py`.
python -m pytest tests/test_v11_b2_memory_excerpt.py -v
"""
import asyncio
import random
import pytest
from app import memorybank, models, tree
from app.context import builder
from app.database import Base, SessionLocal, engine
BUDGET = memorybank.MEMORY_EXCERPT_TOKENS
MARKER = memorybank.EXCERPT_OMISSION_MARKER
FILLER = "The travellers walked the long grey road north past the salt market and the reed beds. "
def words_to_tokens(tokens: int) -> str:
"""Filler at least `tokens` long."""
text = FILLER
while builder.count_tokens(text) < tokens:
text += FILLER
return text
# --------------------------------------------------------------- the excerpt
def test_a_block_that_fits_is_sent_whole_and_unchanged():
raw = words_to_tokens(BUDGET - 200)
assert builder.count_tokens(raw) <= BUDGET
assert memorybank.memory_excerpt(raw) == raw
def test_a_block_of_exactly_the_budget_is_unchanged():
raw = words_to_tokens(BUDGET)
tokens = memorybank._excerpt_encoding().encode(raw)[:BUDGET]
exact = memorybank._excerpt_encoding().decode(tokens)
if builder.count_tokens(exact) == BUDGET:
assert memorybank.memory_excerpt(exact) == exact
def test_a_long_block_keeps_its_opening_and_its_end_in_order():
opening = "Mara slipped the amber sundial inside the cracked teapot. "
ending = "Aldric finally reached the north gate at dawn."
raw = opening + words_to_tokens(3 * BUDGET) + ending
excerpt = memorybank.memory_excerpt(raw)
assert excerpt.startswith(opening)
assert excerpt.endswith(ending)
head, _, tail = excerpt.partition(f"\n\n{MARKER}\n\n")
assert tail, "the marker must sit between the two parts"
assert excerpt.index(opening) < excerpt.index(MARKER) < excerpt.index(ending)
def test_the_split_is_even_and_documented():
raw = words_to_tokens(4 * BUDGET)
head_budget, tail_budget = memorybank.excerpt_split(BUDGET)
marker_tokens = builder.count_tokens(f"\n\n{MARKER}\n\n")
assert head_budget + tail_budget + marker_tokens == BUDGET
assert abs(head_budget - tail_budget) <= 1
excerpt = memorybank.memory_excerpt(raw)
head, _, tail = excerpt.partition(f"\n\n{MARKER}\n\n")
# Each part is cut as a run of `head_budget` / `tail_budget` tokens. Measured
# on its own, a cut run can come to one token more, because the text either
# side of the cut tokenises differently once it is separated; the hard limit
# is the whole excerpt, tested below.
assert builder.count_tokens(head) <= head_budget + 1
assert builder.count_tokens(tail) <= tail_budget + 1
assert builder.count_tokens(excerpt) <= BUDGET
@pytest.mark.parametrize("extra", [1, 7, 500, BUDGET, 9 * BUDGET])
def test_the_excerpt_never_exceeds_the_budget(extra):
raw = words_to_tokens(BUDGET + extra)
excerpt = memorybank.memory_excerpt(raw)
assert builder.count_tokens(excerpt) <= BUDGET
assert memorybank.memory_excerpt(raw) == excerpt # deterministic
@pytest.mark.parametrize("seed", range(4))
def test_the_budget_holds_for_awkward_text(seed):
"""Token boundaries can merge differently once the parts are rejoined, and
text that is not plain English tokenises unevenly. The budget still holds."""
rng = random.Random(seed)
alphabet = "abcdefghij ÄÖÜ ßé漢字かな 🙂🐉 \n\t.,;:—'\""
raw = "".join(rng.choice(alphabet) for _ in range(12000))
assert builder.count_tokens(memorybank.memory_excerpt(raw)) <= BUDGET
def test_a_fact_in_the_middle_of_a_very_long_block_is_still_omitted():
"""The documented limit of a bounded excerpt: head and tail, not everything."""
half = words_to_tokens(3 * BUDGET)
raw = half + "Mara slipped the amber sundial inside the cracked teapot. " + half
assert "sundial" not in memorybank.memory_excerpt(raw)
# ------------------------------------------------------------ memory creation
class EchoSummariser:
"""Returns the whole excerpt it was given as the memory: the worst case for
a marker leaking into stored text."""
def __init__(self):
self.users: list[str] = []
async def complete(self, system, user, *, temperature=0.3, max_tokens=400):
self.users.append(user)
return user.split("Story excerpt:\n\n", 1)[-1].rsplit("\n\nMemory:", 1)[0]
@pytest.fixture()
def db():
Base.metadata.create_all(bind=engine)
session = SessionLocal()
try:
yield session
finally:
session.close()
Base.metadata.drop_all(bind=engine)
def campaign(db, texts):
user = models.User(is_guest=False, email="b2-excerpt@example.com")
db.add(user)
db.flush()
db.add(models.Settings(user_id=user.id, model="m", embedding_model=""))
adventure = models.Adventure(user_id=user.id, title="Excerpt", script_state={},
auto_summarize=True, memory_bank_enabled=True)
db.add(adventure)
db.flush()
nodes = []
for i, text in enumerate(texts):
action = models.Action(adventure_id=adventure.id, type="ai" if i % 2 else "do", text=text)
tree.place_action(db, adventure, action)
db.add(action)
db.flush()
nodes.append(action)
db.commit()
return adventure, nodes
def write_memory(db, adventure, monkeypatch, summariser):
monkeypatch.setattr(memorybank, "MEMORY_START", 0)
monkeypatch.setattr(memorybank, "MAX_MEMORIES_PER_RUN", 1)
monkeypatch.setattr(memorybank, "summary_provider", lambda s: summariser)
settings = db.query(models.Settings).first()
asyncio.run(memorybank._create_due_memories(adventure, settings, db))
return db.query(models.Memory).filter_by(adventure_id=adventure.id).all()
def test_the_marker_is_never_stored_as_part_of_a_memory(db, monkeypatch):
long = words_to_tokens(800)
adventure, _ = campaign(db, [long] * (memorybank.MEMORY_INTERVAL + memorybank.SETTLE_SLACK))
summariser = EchoSummariser()
[memory] = write_memory(db, adventure, monkeypatch, summariser)
assert MARKER in summariser.users[0] # the summariser was told
assert MARKER not in memory.text # and the memory does not repeat it
assert "[…" not in memory.text and "omitted" not in memory.text
def test_a_short_block_is_prompted_exactly_as_before(db, monkeypatch):
texts = [f"Short action {i}." for i in range(memorybank.MEMORY_INTERVAL + memorybank.SETTLE_SLACK)]
adventure, _ = campaign(db, texts)
summariser = EchoSummariser()
write_memory(db, adventure, monkeypatch, summariser)
block = "\n\n".join(texts[:memorybank.MEMORY_INTERVAL])
assert summariser.users[0] == f"Story excerpt:\n\n{block}\n\nMemory:"
def test_a_long_blocks_memory_keeps_its_source_provenance(db, monkeypatch):
early = "Mara slipped the amber sundial inside the cracked teapot. " + words_to_tokens(900)
texts = [early] + [words_to_tokens(900)] * (memorybank.MEMORY_INTERVAL + memorybank.SETTLE_SLACK - 1)
adventure, nodes = campaign(db, texts)
[memory] = write_memory(db, adventure, monkeypatch, EchoSummariser())
block = nodes[:memorybank.MEMORY_INTERVAL]
assert "sundial" in memory.text # the early fact reached the summariser
assert (memory.source_start, memory.source_end) == (block[0].depth, block[-1].depth)
assert (memory.branch_id, memory.depth) == (block[-1].branch_id, block[-1].depth)