Corrects the memory mechanisms WP-B.1 diagnosed, one at a time, each verified before the next. Accepted by the owner with a documented reference-model limitation. No schema, bundle format, setting default, lineage, authority or protocol-cleanup change. - B2.1 ranking: the retrieval query is the player's input plus a bounded scene context (state scene + end of the newest narration), embedded in one call. final = semantic (0.6 input / 0.4 context) + 0.15 x lexical, where lexical is a rarity-weighted share of the input's words, computed per turn over the candidates with no index. Scores and the query are recorded per used memory; pins and redundancy suppression unchanged. - B2.2 coverage-aware eviction (memorybank.eviction_order): the earliest and newest memories are kept, the smallest coverage hole goes first, least-recently-used breaks ties and remains the fallback. Bounded; pins never evicted; frozen-bank protection kept; reads no text or vectors. - B2.3 bounded memory creation: a block longer than 2,000 tokens is shown to the summariser as head + tail with an omission marker, inside the same budget; shorter blocks unchanged; the marker is never stored. - The memory summariser prompt is unchanged from v1.0.0. A B2.4 prompt experiment was measured on the reference model, showed no reliable improvement for the target failure (0/5 under both prompts, with new "Memory:"-prefix, second-person and length regressions), and was reverted. memorybank.memory_user_prompt is kept as a behaviour-neutral helper. - tools/memory_fidelity.py (diagnostic only): genre-neutral fixtures plus the failed block, a deterministic fidelity checker, and a real-model shipped-vs-experiment measurement. - tools/memory_diagnostic.py: ranking replica uses production scoring; ranking_crowded, ranking_context_dependent and independent_full fixtures; per-turn isolation and provenance. - tests: B.1's two strict xfails are now ordinary passes; ranking, eviction and excerpt tests; summariser acceptance tests kept apart from diagnostic-measurement tests. - DEVELOPMENT.md: the GPU-host kernel/Ollama watch used `-k -u ollama`, which matches nothing; now the OR form. - docs: CONTEXT-AND-MEMORY 15/18/20/21 as shipped, V1.1-PLAN (status and release criteria 12-13), planning README, VERSION v4.3, reports/v1.1/V1.1-WP-B2-REPORT.md. Deterministic independent-memory recovery: PASS (independent_full fails on v1.0.0 at creation and returns recovered_through_memory_independent here). Reference-model independent recovery: FAILED on the precondition-valid attempt, at memory creation: the summariser omitted a player-established fact from a block it received whole. Accepted as a documented v1.1 residual and carried into the release gate. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY
190 lines
7.6 KiB
Python
190 lines
7.6 KiB
Python
"""v1.1 WP-B.2 (B2.3): what the memory summariser is shown of a long block.
|
|
|
|
v1.0.0 sent the last 2,000 tokens of a block, so a fact early in a longer block
|
|
never reached the summariser (B.1 §E). A block that fits is still sent whole. A
|
|
longer one is now sent as its opening and its end, with a marker between them,
|
|
inside the same 2,000-token budget.
|
|
|
|
The scenario-level test (the planted fact early in a long block, remembered) is
|
|
in `test_v11_b1_memory_diagnostic.py`.
|
|
|
|
python -m pytest tests/test_v11_b2_memory_excerpt.py -v
|
|
"""
|
|
|
|
import asyncio
|
|
import random
|
|
|
|
import pytest
|
|
|
|
from app import memorybank, models, tree
|
|
from app.context import builder
|
|
from app.database import Base, SessionLocal, engine
|
|
|
|
BUDGET = memorybank.MEMORY_EXCERPT_TOKENS
|
|
MARKER = memorybank.EXCERPT_OMISSION_MARKER
|
|
FILLER = "The travellers walked the long grey road north past the salt market and the reed beds. "
|
|
|
|
|
|
def words_to_tokens(tokens: int) -> str:
|
|
"""Filler at least `tokens` long."""
|
|
text = FILLER
|
|
while builder.count_tokens(text) < tokens:
|
|
text += FILLER
|
|
return text
|
|
|
|
|
|
# --------------------------------------------------------------- the excerpt
|
|
|
|
|
|
def test_a_block_that_fits_is_sent_whole_and_unchanged():
|
|
raw = words_to_tokens(BUDGET - 200)
|
|
assert builder.count_tokens(raw) <= BUDGET
|
|
assert memorybank.memory_excerpt(raw) == raw
|
|
|
|
|
|
def test_a_block_of_exactly_the_budget_is_unchanged():
|
|
raw = words_to_tokens(BUDGET)
|
|
tokens = memorybank._excerpt_encoding().encode(raw)[:BUDGET]
|
|
exact = memorybank._excerpt_encoding().decode(tokens)
|
|
if builder.count_tokens(exact) == BUDGET:
|
|
assert memorybank.memory_excerpt(exact) == exact
|
|
|
|
|
|
def test_a_long_block_keeps_its_opening_and_its_end_in_order():
|
|
opening = "Mara slipped the amber sundial inside the cracked teapot. "
|
|
ending = "Aldric finally reached the north gate at dawn."
|
|
raw = opening + words_to_tokens(3 * BUDGET) + ending
|
|
excerpt = memorybank.memory_excerpt(raw)
|
|
assert excerpt.startswith(opening)
|
|
assert excerpt.endswith(ending)
|
|
head, _, tail = excerpt.partition(f"\n\n{MARKER}\n\n")
|
|
assert tail, "the marker must sit between the two parts"
|
|
assert excerpt.index(opening) < excerpt.index(MARKER) < excerpt.index(ending)
|
|
|
|
|
|
def test_the_split_is_even_and_documented():
|
|
raw = words_to_tokens(4 * BUDGET)
|
|
head_budget, tail_budget = memorybank.excerpt_split(BUDGET)
|
|
marker_tokens = builder.count_tokens(f"\n\n{MARKER}\n\n")
|
|
assert head_budget + tail_budget + marker_tokens == BUDGET
|
|
assert abs(head_budget - tail_budget) <= 1
|
|
excerpt = memorybank.memory_excerpt(raw)
|
|
head, _, tail = excerpt.partition(f"\n\n{MARKER}\n\n")
|
|
# Each part is cut as a run of `head_budget` / `tail_budget` tokens. Measured
|
|
# on its own, a cut run can come to one token more, because the text either
|
|
# side of the cut tokenises differently once it is separated; the hard limit
|
|
# is the whole excerpt, tested below.
|
|
assert builder.count_tokens(head) <= head_budget + 1
|
|
assert builder.count_tokens(tail) <= tail_budget + 1
|
|
assert builder.count_tokens(excerpt) <= BUDGET
|
|
|
|
|
|
@pytest.mark.parametrize("extra", [1, 7, 500, BUDGET, 9 * BUDGET])
|
|
def test_the_excerpt_never_exceeds_the_budget(extra):
|
|
raw = words_to_tokens(BUDGET + extra)
|
|
excerpt = memorybank.memory_excerpt(raw)
|
|
assert builder.count_tokens(excerpt) <= BUDGET
|
|
assert memorybank.memory_excerpt(raw) == excerpt # deterministic
|
|
|
|
|
|
@pytest.mark.parametrize("seed", range(4))
|
|
def test_the_budget_holds_for_awkward_text(seed):
|
|
"""Token boundaries can merge differently once the parts are rejoined, and
|
|
text that is not plain English tokenises unevenly. The budget still holds."""
|
|
rng = random.Random(seed)
|
|
alphabet = "abcdefghij ÄÖÜ ßé漢字かな 🙂🐉 \n\t.,;:—'\""
|
|
raw = "".join(rng.choice(alphabet) for _ in range(12000))
|
|
assert builder.count_tokens(memorybank.memory_excerpt(raw)) <= BUDGET
|
|
|
|
|
|
def test_a_fact_in_the_middle_of_a_very_long_block_is_still_omitted():
|
|
"""The documented limit of a bounded excerpt: head and tail, not everything."""
|
|
half = words_to_tokens(3 * BUDGET)
|
|
raw = half + "Mara slipped the amber sundial inside the cracked teapot. " + half
|
|
assert "sundial" not in memorybank.memory_excerpt(raw)
|
|
|
|
|
|
# ------------------------------------------------------------ memory creation
|
|
|
|
|
|
class EchoSummariser:
|
|
"""Returns the whole excerpt it was given as the memory: the worst case for
|
|
a marker leaking into stored text."""
|
|
|
|
def __init__(self):
|
|
self.users: list[str] = []
|
|
|
|
async def complete(self, system, user, *, temperature=0.3, max_tokens=400):
|
|
self.users.append(user)
|
|
return user.split("Story excerpt:\n\n", 1)[-1].rsplit("\n\nMemory:", 1)[0]
|
|
|
|
|
|
@pytest.fixture()
|
|
def db():
|
|
Base.metadata.create_all(bind=engine)
|
|
session = SessionLocal()
|
|
try:
|
|
yield session
|
|
finally:
|
|
session.close()
|
|
Base.metadata.drop_all(bind=engine)
|
|
|
|
|
|
def campaign(db, texts):
|
|
user = models.User(is_guest=False, email="b2-excerpt@example.com")
|
|
db.add(user)
|
|
db.flush()
|
|
db.add(models.Settings(user_id=user.id, model="m", embedding_model=""))
|
|
adventure = models.Adventure(user_id=user.id, title="Excerpt", script_state={},
|
|
auto_summarize=True, memory_bank_enabled=True)
|
|
db.add(adventure)
|
|
db.flush()
|
|
nodes = []
|
|
for i, text in enumerate(texts):
|
|
action = models.Action(adventure_id=adventure.id, type="ai" if i % 2 else "do", text=text)
|
|
tree.place_action(db, adventure, action)
|
|
db.add(action)
|
|
db.flush()
|
|
nodes.append(action)
|
|
db.commit()
|
|
return adventure, nodes
|
|
|
|
|
|
def write_memory(db, adventure, monkeypatch, summariser):
|
|
monkeypatch.setattr(memorybank, "MEMORY_START", 0)
|
|
monkeypatch.setattr(memorybank, "MAX_MEMORIES_PER_RUN", 1)
|
|
monkeypatch.setattr(memorybank, "summary_provider", lambda s: summariser)
|
|
settings = db.query(models.Settings).first()
|
|
asyncio.run(memorybank._create_due_memories(adventure, settings, db))
|
|
return db.query(models.Memory).filter_by(adventure_id=adventure.id).all()
|
|
|
|
|
|
def test_the_marker_is_never_stored_as_part_of_a_memory(db, monkeypatch):
|
|
long = words_to_tokens(800)
|
|
adventure, _ = campaign(db, [long] * (memorybank.MEMORY_INTERVAL + memorybank.SETTLE_SLACK))
|
|
summariser = EchoSummariser()
|
|
[memory] = write_memory(db, adventure, monkeypatch, summariser)
|
|
assert MARKER in summariser.users[0] # the summariser was told
|
|
assert MARKER not in memory.text # and the memory does not repeat it
|
|
assert "[…" not in memory.text and "omitted" not in memory.text
|
|
|
|
|
|
def test_a_short_block_is_prompted_exactly_as_before(db, monkeypatch):
|
|
texts = [f"Short action {i}." for i in range(memorybank.MEMORY_INTERVAL + memorybank.SETTLE_SLACK)]
|
|
adventure, _ = campaign(db, texts)
|
|
summariser = EchoSummariser()
|
|
write_memory(db, adventure, monkeypatch, summariser)
|
|
block = "\n\n".join(texts[:memorybank.MEMORY_INTERVAL])
|
|
assert summariser.users[0] == f"Story excerpt:\n\n{block}\n\nMemory:"
|
|
|
|
|
|
def test_a_long_blocks_memory_keeps_its_source_provenance(db, monkeypatch):
|
|
early = "Mara slipped the amber sundial inside the cracked teapot. " + words_to_tokens(900)
|
|
texts = [early] + [words_to_tokens(900)] * (memorybank.MEMORY_INTERVAL + memorybank.SETTLE_SLACK - 1)
|
|
adventure, nodes = campaign(db, texts)
|
|
[memory] = write_memory(db, adventure, monkeypatch, EchoSummariser())
|
|
block = nodes[:memorybank.MEMORY_INTERVAL]
|
|
assert "sundial" in memory.text # the early fact reached the summariser
|
|
assert (memory.source_start, memory.source_end) == (block[0].depth, block[-1].depth)
|
|
assert (memory.branch_id, memory.depth) == (block[-1].branch_id, block[-1].depth)
|