diff --git a/DEVELOPMENT.md b/DEVELOPMENT.md index 7123762..fd2e122 100644 --- a/DEVELOPMENT.md +++ b/DEVELOPMENT.md @@ -473,8 +473,11 @@ nvidia-smi --query-gpu=timestamp,pcie.link.gen.current,pcie.link.width.current,p # The driver's own sampling: power, utilisation, clocks, memory, ECC and throttling nvidia-smi dmon -s pucvmet -d 5 | tee "$HOME/gpu-dmon-$(date +%F-%H%M).log" -# Kernel and Ollama messages, live -journalctl -f -k -u ollama | tee "$HOME/ollama-kernel-watch-$(date +%F-%H%M).log" +# Kernel and Ollama messages, live. The `+` is an OR: `journalctl -k -u ollama` +# asks for messages that are both kernel messages and the ollama unit's, which +# is none, and writes an empty log. +journalctl -f -o short-iso _TRANSPORT=kernel + _SYSTEMD_UNIT=ollama.service \ + | tee "$HOME/ollama-kernel-watch-$(date +%F-%H%M).log" ``` If the GPU faults, find the moment and then read what the card was doing just diff --git a/backend/app/memorybank.py b/backend/app/memorybank.py index 12f3a66..289d0c6 100644 --- a/backend/app/memorybank.py +++ b/backend/app/memorybank.py @@ -17,9 +17,10 @@ database session. It does three things: then evicts the bank down to its capacity. Evicted memories are marked as forgotten and kept so that the UI can still show them. -When the app generates a turn, `retrieve_memories` embeds the recent story text -and ranks the bank by cosine similarity. The highest-ranked memories become the -Memories section of the context. +When the app generates a turn, `retrieve_memories` embeds the player's input +and the current scene, and ranks the bank by a fixed mix of cosine similarity +and rarity-weighted word overlap with the input (v1.1 WP-B.2). The +highest-ranked memories become the Memories section of the context. Every AI call in this module is best-effort. A failure is logged to the debug page and retried on a later turn, because the cursors advance only after a call @@ -28,6 +29,7 @@ succeeds. import asyncio import logging +import math from array import array from collections import OrderedDict @@ -36,6 +38,7 @@ from sqlalchemy.orm import Session, defer, object_session from . import derived, models, summaries, tree, vectors from .context import ( + count_tokens, cursors, history, lineage, @@ -43,8 +46,11 @@ from .context import ( story_actions, truncate_to_last_tokens, ) +from .context.builder import _encoding as _token_encoding from .database import SessionLocal from .knowledge import embeddings as knowledge_embeddings +from .knowledge import fts +from .narrative import model as narrative_model from .providers import OpenAICompatibleProvider, ProviderError from .vectors import cosine # re-exported: the ranking lives here, the maths there @@ -55,10 +61,14 @@ MEMORY_START = 12 # first memory once the adventure reaches this many actions SUMMARY_INTERVAL = 15 # actions between Story Summary updates MAX_MEMORIES_PER_RUN = 5 # cap catch-up work (e.g. imported adventures) per turn MAX_EMBED_BATCH = 32 -RETRIEVAL_WINDOW_TOKENS = 600 # recent story text used as the similarity query -RETRIEVAL_WINDOW_ACTIONS = 4 # ...taken from this many of the newest actions SUMMARY_MAX_WORDS = 250 -MEMORY_EXCERPT_TOKENS = 2000 # of the block, when a block is longer than this +MEMORY_EXCERPT_TOKENS = 2000 # the most of a block the summariser is shown + +# v1.1 WP-B.2: what stands between the two parts of a block too long to send +# whole. It says a part is missing, so the summariser does not read the end as +# following straight on from the opening, and `summarize_block` removes it from +# anything the model repeats back. +EXCERPT_OMISSION_MARKER = "[… the middle of this stretch of story is left out here …]" # How much story has to sit past a block before that block is summarized. # @@ -278,9 +288,10 @@ def set_vector(memory: models.Memory, vector: list[float] | None) -> None: """ memory.embedding_blob = None if vector is None else vectors.pack(vector) memory.embedded = vector is not None - cached = _vector_cache.get(memory.adventure_id) - if cached is not None: - cached.pop(memory.id, None) + for cache in (_vector_cache, _terms_cache): + cached = cache.get(memory.adventure_id) + if cached is not None: + cached.pop(memory.id, None) # ---------- The vector cache ---------- @@ -309,6 +320,37 @@ _vector_cache: OrderedDict[int, dict[int, array]] = OrderedDict() VECTOR_CACHE_ADVENTURES = 8 # ~600 KB each at a 100-memory bank +# v1.1 WP-B.2: each memory's lexical terms, held the same way and by the same +# two rules as its vector. `set_vector` is also where a memory's text changes +# (an edit clears the vector to re-embed it), so dropping the entry there covers +# a rewritten text as well as a rewritten vector. Text is read only for memories +# not already held, and only on a turn whose input has words to match. +_terms_cache: OrderedDict[int, dict[int, frozenset[str]]] = OrderedDict() + + +def _terms_for(db: Session, adventure_id: int, ids: list[int]) -> dict[int, frozenset[str]]: + """The lexical terms for `ids`, reading text only for the ones not already held.""" + cached = _terms_cache.get(adventure_id) + if cached is None: + cached = _terms_cache[adventure_id] = {} + _terms_cache.move_to_end(adventure_id) + while len(_terms_cache) > VECTOR_CACHE_ADVENTURES: + _terms_cache.popitem(last=False) + + wanted = set(ids) + for gone in set(cached) - wanted: + del cached[gone] + missing = [memory_id for memory_id in ids if memory_id not in cached] + if missing: + rows = db.execute( + select(models.Memory.id, models.Memory.text) + .where(models.Memory.id.in_(missing)) + ).all() + for memory_id, text in rows: + cached[memory_id] = lexical_terms(text or "") + return cached + + def forget_cached_vectors(adventure_id: int) -> None: """Drops an adventure's cached vectors. @@ -316,6 +358,7 @@ def forget_cached_vectors(adventure_id: int) -> None: corrects itself, as described in the comment above. """ _vector_cache.pop(adventure_id, None) + _terms_cache.pop(adventure_id, None) def _vectors_for(db: Session, adventure_id: int, ids: list[int]) -> dict[int, array]: @@ -535,6 +578,199 @@ def cast_brief(adventure: models.Adventure, text: str) -> str: # ---------- Retrieval (runs inside the turn, before build_context) ---------- +# v1.1 WP-B.2: what the retrieval query is made of, and how a memory is scored +# against it (CONTEXT-AND-MEMORY §18, §20). +# +# WP-B.1 measured the v1.0.0 query, the newest four actions cut to 600 tokens, +# against a planted early fact. The player's one-line question arrived after +# three turns of narration, so the embedding mostly described the narration: a +# direct question about the fact fell from cosine 0.708 on its own to 0.241 in +# that query, and a real 100-turn campaign ranked the only memory of the fact +# 10th of 19 against a `memory_top_k` of 4. +# +# The query is now two short texts, embedded in one call: +# +# input the player's own action this turn, when there is one +# context the current scene from the authoritative state (summary, location, +# who is present), then the end of the newest narration +# +# The context is still there because a question often cannot be read without +# it ("I ask her where she hid it"), and §18 says retrieval must not rely on raw +# input alone. It is bounded so it can resolve a reference but cannot outweigh +# the question by sheer length. +# +# A memory's score is +# +# semantic_score = INPUT_WEIGHT * cos(input, memory) +# + (1 - INPUT_WEIGHT) * cos(context, memory) +# lexical_score = rarity-weighted share of the input's words the memory holds +# final_score = semantic_score + LEXICAL_WEIGHT * lexical_score +# +# With no player input (a continue, or a dry run from Insights) the semantic +# score is the context cosine alone and the lexical score is 0. Pins are +# unchanged: a pinned memory is always used and counts toward `memory_top_k`. +INPUT_TYPES = ("do", "say", "story") # player actions that carry words to search for +QUERY_INPUT_TOKENS = 200 # of the player's action; a long `story` entry is cut +QUERY_SCENE_TOKENS = 60 # of the state's scene line +QUERY_NARRATION_TOKENS = 120 # from the end of the newest narration +INPUT_WEIGHT = 0.6 +# Chosen by sweep (0, 0.05, 0.1, 0.15, 0.2, 0.3, 0.5) over the deterministic +# ranking fixtures, recorded in the WP-B.2 report (§C, §D). The two-part query +# alone already ranks the planting-era memory first; 0.15 is the smallest weight +# at which the lexical term by itself also lifts it into `memory_top_k` against +# the v1.0.0 narration-filled query, and no rare-word negative control put an +# unrelated memory above it. At 0.5 an incidental shared word was enough to +# select it for an unrelated question, which is the failure a larger weight buys. +LEXICAL_WEIGHT = 0.15 +# `fts.terms` drops these already; the plural fold below is the only stemming. +_MIN_FOLD_LENGTH = 5 + + +def _fold(word: str) -> str: + """One term, reduced so "shelves'" and "shelf" do not meet, but "teapots" + and "teapot" do. Possessives lose their `'s`, and a trailing `s` goes from a + word long enough to be a plural and not ending in `ss`. Deliberately no more + than that: a stemmer is a dependency, and a wrong fold merges two words.""" + word = word.split("'", 1)[0] + if len(word) >= _MIN_FOLD_LENGTH and word.endswith("s") and not word.endswith("ss"): + word = word[:-1] + return word + + +def lexical_terms(text: str) -> frozenset[str]: + """The words of `text` that lexical matching compares, folded. + + The tokenizer and stop list are imported knowledge's (`knowledge.fts`), so + the two retrieval paths agree on what a word is. + """ + return frozenset(t for t in (_fold(w) for w in fts.terms(text)) if len(t) >= fts.MIN_TERM_LENGTH) + + +def lexical_scores(input_terms: frozenset[str], terms_of: dict[int, frozenset[str]]) -> dict[int, float]: + """Each candidate's share of the input's rarity, in [0, 1]. + + A term's weight is `ln((N + 1) / (df + 1))`: N candidates, df of them holding + it. A word every candidate holds weighs exactly 0, so a protagonist's name or + a word the whole bank shares moves nothing, and a word no candidate holds + weighs the most. The share is taken over **all** the input's terms, so a + memory that happens to hold one rare word of a longer question gets that + word's part of the question, not the whole of it. The weights live only for + this call, over this candidate set: no index, no stored field. + """ + if not input_terms or not terms_of: + return {memory_id: 0.0 for memory_id in terms_of} + n = len(terms_of) + weight = { + term: math.log((n + 1) / (sum(1 for terms in terms_of.values() if term in terms) + 1)) + for term in input_terms + } + total = sum(weight.values()) + if total <= 0: + return {memory_id: 0.0 for memory_id in terms_of} + return { + memory_id: min(1.0, sum(w for term, w in weight.items() if term in terms) / total) + for memory_id, terms in terms_of.items() + } + + +def _scene_text(state) -> str: + """The scene as the authoritative state has it: summary, location, who is present. + + Names only, read straight off the document. The full entity list is left + out on purpose: a campaign with a large cast would turn every query into a + search for everyone. + """ + if not isinstance(state, dict): + return "" + scene = state.get("scene") + if not isinstance(scene, dict): + return "" + pieces: list[str] = [] + summary = scene.get("summary") + if isinstance(summary, str) and summary.strip(): + pieces.append(summary.strip()) + location = scene.get("location") + if isinstance(location, str) and location.strip(): + pieces.append(narrative_model.entity_name(state, location.strip())) + present = scene.get("present") + if isinstance(present, list): + names = [narrative_model.entity_name(state, key) for key in present[:8] + if isinstance(key, str) and key.strip()] + if names: + pieces.append(", ".join(names)) + return truncate_to_last_tokens(". ".join(pieces), QUERY_SCENE_TOKENS) + + +def retrieval_query(adventure: models.Adventure, exclude_action_id: int | None = None) -> dict: + """The two texts a turn's memory retrieval embeds, and the words it matches. + + Returns `{"input", "context", "input_terms"}`. `input` is empty when the + newest action is not a player action with text, which is a continue turn or a + dry run. `context` is empty only for a story with no scene and no narration. + """ + recent = history.tail(adventure, 2, exclude_action_id) + newest = recent[-1] if recent else None + player_input = "" + if newest is not None and newest.type in INPUT_TYPES: + player_input = truncate_to_last_tokens(newest.text.strip(), QUERY_INPUT_TOKENS) + narration = recent[0].text if len(recent) > 1 else "" + else: + narration = newest.text if newest is not None else "" + context = "\n".join(part for part in ( + _scene_text(adventure.narrative_state), + truncate_to_last_tokens(narration.strip(), QUERY_NARRATION_TOKENS), + ) if part.strip()) + return { + "input": player_input, + "context": context, + "input_terms": sorted(lexical_terms(player_input)), + } + + +def score_candidates( + ids: list[int], + held: dict, + terms_of: dict[int, frozenset[str]], + input_vec, + context_vec, + input_terms, +) -> list[tuple[float, int, float, float]]: + """`(final_score, memory_id, semantic_score, lexical_score)`, best first. + + Ties on the final score are broken by id, so the order never depends on the + order the database returned rows in. + """ + lexical = lexical_scores(frozenset(input_terms), {i: terms_of.get(i, frozenset()) for i in ids}) + rows = [] + for memory_id in ids: + vector = held[memory_id] + if input_vec is not None and context_vec is not None: + semantic = (INPUT_WEIGHT * cosine(input_vec, vector) + + (1.0 - INPUT_WEIGHT) * cosine(context_vec, vector)) + else: + semantic = cosine(input_vec if input_vec is not None else context_vec, vector) + lex = lexical.get(memory_id, 0.0) + rows.append((semantic + LEXICAL_WEIGHT * lex, memory_id, semantic, lex)) + rows.sort(key=lambda row: (-row[0], row[1])) + return rows + + +def select_memories(scored, pinned_of, held, authority_of, top_k): + """Pins first, then the best-scoring rest, skipping repeats (§22). + + Returns `(used, suppressed)`. `used` is `(final_score, memory_id, pinned)` + rows, best first. + """ + rows = [(final, memory_id, pinned_of[memory_id]) for final, memory_id, _, _ in scored] + used = [row for row in rows if row[2]] + remaining = max(0, top_k - len(used)) + candidates = [row for row in rows if not row[2]] + kept, suppressed = _drop_redundant(candidates, held, authority_of, remaining) + used += kept + used.sort(key=lambda row: (-row[0], row[1])) + return used, suppressed + + async def retrieve_memories( adventure: models.Adventure, settings: models.Settings, @@ -544,16 +780,17 @@ async def retrieve_memories( """Returns the memories to inject, or None when the bank is off. The result is a dict of the form - `{"used": [{id, text, similarity, pinned}], "error": str | None}`. It is - None when the memory bank is disabled for this adventure. + `{"used": [{id, text, similarity, semantic_score, lexical_score, + final_score, pinned, authority, source}], "query": {...}, "error": str | None}`. + `similarity` is the semantic score, under the name the inspector has always + shown. It is None when the memory bank is disabled for this adventure. This only reads. A turn counts the memories it used with `record_use`, just before the commit that saves the turn; see that function for why the count cannot be written here. - `exclude_action_id` removes the action being retried from the similarity - query, so that a discarded attempt cannot influence which memories are - returned. + `exclude_action_id` removes the action being retried from the query, so that + a discarded attempt cannot influence which memories are returned. """ if not adventure.memory_bank_enabled: return None @@ -585,39 +822,33 @@ async def retrieve_memories( if not catalogue: return {"used": [], "error": None} - recent = history.tail(adventure, RETRIEVAL_WINDOW_ACTIONS, exclude_action_id) - query = truncate_to_last_tokens( - "\n\n".join(a.text for a in recent), RETRIEVAL_WINDOW_TOKENS - ) - if not query.strip(): + query = retrieval_query(adventure, exclude_action_id) + texts = [t for t in (query["input"], query["context"]) if t.strip()] + if not texts: return {"used": [], "error": None} try: - [query_vec] = await embedding_provider(settings).embed([query]) + embedded = await embedding_provider(settings).embed(texts) except ProviderError as exc: return {"used": [], "error": str(exc)} + vectors_by_text = dict(zip(texts, embedded)) + input_vec = vectors_by_text.get(query["input"]) if query["input"].strip() else None + context_vec = vectors_by_text.get(query["context"]) if query["context"].strip() else None - held = _vectors_for(db, adventure.id, [memory_id for memory_id, _, _ in catalogue]) + ids = [memory_id for memory_id, _, _ in catalogue] + held = _vectors_for(db, adventure.id, ids) + # Memory text is read only when there are input words to match against, and + # then only for memories not already held (see `_terms_for`). + terms_of = _terms_for(db, adventure.id, ids) if query["input_terms"] else {} authority_of = {memory_id: authority for memory_id, _, authority in catalogue} - scored = sorted( - ( - (cosine(query_vec, held[memory_id]), memory_id, pinned) - for memory_id, pinned, _ in catalogue - if memory_id in held - ), - key=lambda row: row[0], - reverse=True, + pinned_of = {memory_id: pinned for memory_id, pinned, _ in catalogue} + scored = score_candidates( + [memory_id for memory_id in ids if memory_id in held], + held, terms_of, input_vec, context_vec, query["input_terms"], ) - # Pinned memories are always used, and they count toward `top_k`, so the - # injected set stays within the budget unless the pinned memories alone - # exceed it. - top_k = max(1, settings.memory_top_k) - used = [row for row in scored if row[2]] - remaining = max(0, top_k - len(used)) - candidates = [row for row in scored if not row[2]] - kept, suppressed = _drop_redundant(candidates, held, authority_of, remaining) - used += kept - used.sort(key=lambda row: row[0], reverse=True) + components = {memory_id: (semantic, lex) for _, memory_id, semantic, lex in scored} + used, suppressed = select_memories( + scored, pinned_of, held, authority_of, max(1, settings.memory_top_k)) if not used: return {"used": [], "error": None} @@ -638,14 +869,19 @@ async def retrieve_memories( ).where(models.Memory.id.in_(used_ids)) ).all() } - texts = {memory_id: row.text for memory_id, row in detail.items()} + texts_of = {memory_id: row.text for memory_id, row in detail.items()} return { "used": [ { "id": memory_id, - "text": texts.get(memory_id, ""), - "similarity": round(score, 4), + "text": texts_of.get(memory_id, ""), + "similarity": round(components[memory_id][0], 4), + # v1.1 WP-B.2: the parts of the score, so an inspector can see + # why this memory beat the ones below it. + "semantic_score": round(components[memory_id][0], 4), + "lexical_score": round(components[memory_id][1], 4), + "final_score": round(final, 4), "pinned": pinned, # M6: what weight this carries, and where it came from. "authority": getattr(detail.get(memory_id), "authority", ACCEPTED_STORY), @@ -656,7 +892,7 @@ async def retrieve_memories( "source_end": getattr(detail.get(memory_id), "source_end", None), }, } - for score, memory_id, pinned in used + for final, memory_id, pinned in used ], "considered": len(catalogue), # M6: how many candidates were set aside as repeating one already @@ -665,6 +901,15 @@ async def retrieve_memories( {"id": memory_id, "duplicate_of": kept_id} for memory_id, kept_id in suppressed ], + # v1.1 WP-B.2: what was searched for. Recorded per turn, like the rest. + "query": { + "input": query["input"], + "context": query["context"], + "input_terms": query["input_terms"], + "input_weight": INPUT_WEIGHT if input_vec is not None and context_vec is not None + else (1.0 if input_vec is not None else 0.0), + "lexical_weight": LEXICAL_WEIGHT, + }, "error": None, } @@ -822,6 +1067,61 @@ async def _guarded(db: Session, adventure_id: int, kind: str, coro) -> None: db.commit() +def _excerpt_encoding(): + return _token_encoding() + + +def excerpt_split(budget: int = MEMORY_EXCERPT_TOKENS) -> tuple[int, int]: + """`(head_tokens, tail_tokens)` for a block longer than `budget`. + + The marker and the blank lines around it are paid for first; what is left is + halved, and an odd token goes to the tail, the most recent part. So the two + parts plus the marker come to exactly `budget`. + """ + room = max(0, budget - count_tokens(f"\n\n{EXCERPT_OMISSION_MARKER}\n\n")) + head = room // 2 + return head, room - head + + +def memory_excerpt(raw: str, budget: int = MEMORY_EXCERPT_TOKENS) -> str: + """What the summariser is shown of one block. + + v1.1 WP-B.2. A block that fits in `budget` tokens is sent whole, exactly as + before. A longer block used to be cut to its last `budget` tokens, and B.1 + showed that a fact near its start then never reached the summariser at all. + It is now sent as its opening and its end, in order, with + `EXCERPT_OMISSION_MARKER` between them, still inside `budget`. + + Rejoining two token runs can tokenise a little differently at the seams, so + the result is measured, and the head gives up tokens until it fits. A fact in + the middle of a very long block is still left out: this bounds the input, it + does not summarise everything. + """ + enc = _excerpt_encoding() + tokens = enc.encode(raw) + if len(tokens) <= budget: + return raw + head_n, tail_n = excerpt_split(budget) + while True: + excerpt = (f"{enc.decode(tokens[:head_n]).rstrip()}\n\n{EXCERPT_OMISSION_MARKER}\n\n" + f"{enc.decode(tokens[-tail_n:]).lstrip()}" if tail_n else + enc.decode(tokens[:head_n])) + over = count_tokens(excerpt) - budget + if over <= 0 or head_n == 0: + return excerpt + head_n = max(0, head_n - over) + + +def memory_user_prompt(brief: str, excerpt: str) -> str: + """The user message of a memory call: the cast brief, then the excerpt. + + Kept apart from `summarize_block` so an evaluation can send a model exactly + what the application sends (v1.1 WP-B.2, `tools/memory_fidelity.py`). + """ + prompt = f"Story excerpt:\n\n{excerpt}\n\nMemory:" + return f"{brief}\n\n{prompt}" if brief else prompt + + async def summarize_block( adventure: models.Adventure, provider: OpenAICompatibleProvider, @@ -840,15 +1140,16 @@ async def summarize_block( old text in place and moves on. """ raw = "\n\n".join(a.text for a in block) - excerpt = truncate_to_last_tokens(raw, MEMORY_EXCERPT_TOKENS) + excerpt = memory_excerpt(raw) # Match the cast against the untruncated block. The excerpt is what the - # model reads, but a character named in the part that was trimmed is still + # model reads, but a character named in the part that was left out is still # one the memory may have to name. brief = cast_brief(adventure, raw) - prompt = f"Story excerpt:\n\n{excerpt}\n\nMemory:" - return await provider.complete( - MEMORY_SYSTEM_PROMPT, f"{brief}\n\n{prompt}" if brief else prompt - ) + text = await provider.complete(MEMORY_SYSTEM_PROMPT, memory_user_prompt(brief, excerpt)) + # The marker is an instruction to the summariser, never a fact of the story. + if text and EXCERPT_OMISSION_MARKER in text: + text = " ".join(text.replace(EXCERPT_OMISSION_MARKER, " ").split()) + return text async def _create_due_memories( @@ -1028,11 +1329,93 @@ async def _embed_pending( return len(pending) +def eviction_order(rows, limit: int) -> list[int]: + """The ids eviction would take, first to last, at most `limit` of them. + + v1.1 WP-B.2. `rows` are the active memories of one adventure, each with + `id`, `pinned`, `source_start`, `source_end`, `last_used_at`, `created_at` + and `use_count`. Nothing here reads a vector or the database, so the same + function is what the eviction pass runs and what a diagnostic reports. + + WP-B.1 showed what pure least-recently-used order does to a long campaign. + Retrieval is steered by the present scene, so a memory of an early stretch + nothing recent resembles stops being used. It then becomes the least + recently used row, and it goes first, while the bank keeps several memories + of the last few scenes that the history window still holds in full. The + rule below keeps the bank spread over the whole story instead. + + **Coverage.** Memories with a source range say which stretch of the story + they describe. A memory is judged by the hole its removal would leave: the + number of depths between the end of the nearest memory before it and the + start of the nearest memory after it. The smallest hole goes first, so the + bank thins where it is densest. A memory whose start another memory shares + (a retried or re-played stretch, or a sibling line) leaves no hole, and is + the first kind to go. Pinned memories count as coverage, since they stay. + + **Boundaries.** The earliest and the latest memory by position leave a hole + with no memory on one side: removing the first loses the only record of the + opening, and removing the last loses the only record of the most recent + stretch, which is also what keeps a memory written this turn from being + evicted by the pass that wrote it (the frozen bank, below). Boundaries are + not coverage candidates. + + **Recency.** Among memories whose removal leaves the same hole, the least + recently used goes first (`coalesce(last_used_at, created_at)`), then the + less used, then the lower id. Ties are therefore never left to the order the + database returned rows in. + + **Fallback.** When no memory is a coverage candidate — memories typed by the + player or migrated from before coordinates have no range, and a bank can be + all boundaries — the rest are taken least recently used first, exactly as + v1.0.0 did. The bank stays bounded either way. Pinned memories are never + taken; if every active memory is pinned, capacity yields to the pins. + + Recomputed after each pick, because removing one memory widens the holes + of its neighbours. + """ + remaining = {row.id: row for row in rows if not row.pinned} + coverers = {row.id: row for row in rows + if row.source_start is not None and row.source_end is not None} + + def recency(row): + return (row.last_used_at or row.created_at, row.use_count or 0, row.id) + + order: list[int] = [] + while remaining and len(order) < limit: + spans = sorted(coverers.values(), key=lambda r: (r.source_start, r.source_end, r.id)) + starts: dict[int, int] = {} + for row in spans: + starts[row.source_start] = starts.get(row.source_start, 0) + 1 + best = None + furthest_end = None # the largest source_end before index i + for i, row in enumerate(spans): + if row.id in remaining: + if starts[row.source_start] > 1: + cost = 0 + elif i == 0 or i == len(spans) - 1: + cost = None # a boundary + else: + cost = max(0, spans[i + 1].source_start - furthest_end - 1) + if cost is not None: + key = (cost, *recency(row)) + if best is None or key < best[0]: + best = (key, row.id) + furthest_end = row.source_end if furthest_end is None else max(furthest_end, row.source_end) + if best is None: + victim = min(remaining.values(), key=recency).id + else: + victim = best[1] + order.append(victim) + del remaining[victim] + coverers.pop(victim, None) + return order + + def _evict_over_capacity( adventure: models.Adventure, settings: models.Settings, db: Session ) -> None: - # The database performs both the count and the ranking, and returns neither - # the rows nor the vectors. Counting by walking `adventure.memories` fetched + # The database performs the count, and the rows read for ordering carry + # neither text nor vectors. Counting by walking `adventure.memories` fetched # every vector in the bank on every turn, whether or not the bank was over # capacity. in_this_bank = (models.Memory.adventure_id == adventure.id, @@ -1043,32 +1426,23 @@ def _evict_over_capacity( overflow = active - max(1, settings.memory_bank_capacity) if overflow <= 0: return - # Evict the least recently used memory first, and use the use count only to - # break ties. + # v1.1 WP-B.2: the order is `eviction_order`, coverage first and recency + # second. It replaces least recently used alone; see that function. # - # Ordering by use count first froze the bank. A memory written on this turn - # has never been used, so once every other memory had been retrieved at - # least once, the new memory held the lowest count in the bank. The same - # post-turn run that wrote it then evicted it, one pass after embedding it. - # Use counts only increase, so the bank never recovered. An adventure kept - # whatever memories it held when the bank first filled, and every later - # memory was summarized, marked as forgotten, and never ranked. - # - # Ordering by recency avoids that. A new memory carries the newest - # timestamp, so it is the last row to be evicted rather than the first, and - # it remains until other memories are used. Demoting the use count costs - # little, because retrieving a useful memory also makes it recent. The two - # orderings differ only for memories that were used once and have not been - # retrieved since, which are the rows a full bank should evict. - doomed = db.execute( - select(models.Memory.id) - .where(*in_this_bank, models.Memory.pinned.is_(False)) - .order_by( - func.coalesce(models.Memory.last_used_at, models.Memory.created_at), - models.Memory.use_count, - ) - .limit(overflow) - ).scalars().all() + # What the old ordering fixed still holds. Ordering by use count first froze + # the bank: a memory written on this turn has never been used, so once every + # other memory had been retrieved at least once, the new memory held the + # lowest count in the bank, and the same post-turn run that wrote it evicted + # it. Counts only increase, so the bank never recovered. Under the coverage + # rule the newest memory is the latest boundary, so it is not a coverage + # candidate, and in the fallback it carries the newest timestamp. + rows = db.execute( + select(models.Memory.id, models.Memory.pinned, models.Memory.source_start, + models.Memory.source_end, models.Memory.last_used_at, + models.Memory.created_at, models.Memory.use_count) + .where(*in_this_bank) + ).all() + doomed = eviction_order(rows, overflow) if not doomed: return # Every active memory is pinned, so the pins override capacity. db.execute( diff --git a/backend/tests/test_v11_b1_memory_diagnostic.py b/backend/tests/test_v11_b1_memory_diagnostic.py index a9fb2a0..8dfb1bb 100644 --- a/backend/tests/test_v11_b1_memory_diagnostic.py +++ b/backend/tests/test_v11_b1_memory_diagnostic.py @@ -11,8 +11,9 @@ diagnostic in `tools/memory_diagnostic.py`: 2. **What it finds on this tree.** The scenarios run with a best-case summariser, one that keeps a fact if and only if the fact reached it. Any failure is therefore the application's mechanism, not a model's writing. - - The criteria the current code does not meet are marked `xfail(strict=True)`, - so B.2 has to flip them deliberately. + - WP-B.1 marked the criteria v1.0.0 did not meet `xfail(strict=True)`. WP-B.2 + fixed ranking, eviction and the creation excerpt, and those tests are now + ordinary passes; the v1.0.0 results are recorded in the WP-B.2 report. - The same file is run unchanged against v1.0.0 for the baseline. python -m pytest tests/test_v11_b1_memory_diagnostic.py -v @@ -102,14 +103,14 @@ def test_retention_is_reported_with_the_bank_and_its_eviction_order(): def test_ranking_is_production_ranking_and_agrees_with_the_stored_selection(): ranked = scenario("independent_default")["diagnosis"]["ranked"] assert ranked["replica_matches_stored_selection"] is True - assert ranked["lexical_score"] is None # memory ranking has no lexical term assert ranked["top_k_cutoff"] == 5 assert ranked["yes"] is True and ranked["selected"] is True assert 1 <= ranked["rank"] <= ranked["top_k_cutoff"] - # The production query is the newest four actions, cut to 600 tokens, and the - # one-line question is diluted by the narration around it. - variants = scenario("independent_default")["ranking_variants"] - assert ranked["semantic_score"] < variants["direct"]["similarity"] + # v1.1 WP-B.2: every part of the score is reported, and they add up. + assert 0.0 <= ranked["lexical_score"] <= 1.0 + assert ranked["final_score"] == pytest.approx( + ranked["semantic_score"] + memorybank.LEXICAL_WEIGHT * ranked["lexical_score"], abs=2e-4) + assert ranked["query"]["input"].endswith(md.SCENARIOS["independent_default"].recall_text) def test_injection_is_read_from_the_recall_turns_own_context(): @@ -124,53 +125,118 @@ def test_ranking_variants_direct_paraphrase_and_unrelated(): variants = scenario("independent_default")["ranking_variants"] assert variants["direct"]["rank"] == 1 and variants["direct"]["selected"] assert variants["paraphrase"]["rank"] == 1 and variants["paraphrase"]["selected"] - assert (variants["direct"]["similarity"] > variants["paraphrase"]["similarity"] - > 5 * variants["unrelated"]["similarity"]) + assert (variants["direct"]["final_score"] > variants["paraphrase"]["final_score"] + > variants["unrelated"]["final_score"]) -def test_retrieval_fills_top_k_whatever_the_similarity(): - """Diagnosis: there is no relevance floor. With more memories than - `memory_top_k`, an unrelated query still selects five, and the early fact - rides along at a similarity near zero.""" +def test_retrieval_still_fills_top_k_whatever_the_similarity(): + """There is still no relevance floor: an unrelated question selects a full + `memory_top_k`. B.2 changed which memories those are, not how many — the + early fact is no longer carried along by an unrelated question.""" variants = scenario("independent_default")["ranking_variants"] - assert variants["unrelated"]["similarity"] < 0.1 - assert variants["unrelated"]["selected"] is True + assert variants["unrelated"]["selected_count"] == 5 + assert variants["unrelated"]["selected"] is False + assert variants["unrelated"]["rank"] > 5 + + +# ------------------------------------------------- WP-B.2 ranking acceptance + +@pytest.mark.parametrize("name", ["ranking_crowded", "ranking_context_dependent"]) +def test_acceptance_the_early_memory_is_ranked_and_injected_below_capacity(name): + """The B.1 ranking failure, made deterministic. On v1.0.0 both fixtures are + `retained_but_not_ranked` (ranks 7 and 6 of 17 against a top-k of 4).""" + result = scenario(name) + assert result["isolation"]["ok"], result["isolation"] + diagnosis = result["diagnosis"] + assert diagnosis["created"]["yes"] and diagnosis["retained"]["yes"] + assert diagnosis["retained"]["active_memories"] <= diagnosis["retained"]["memory_bank_capacity"] + assert diagnosis["ranked"]["yes"] and diagnosis["ranked"]["rank"] <= 4 + assert diagnosis["ranked"]["replica_matches_stored_selection"] is True + assert diagnosis["injected"]["yes"] is True + assert diagnosis["verdict"] == "injected" + + +def test_the_crowded_fixture_is_won_by_the_players_question(): + ranked = scenario("ranking_crowded")["diagnosis"]["ranked"] + assert ranked["rank"] == 1 + assert ranked["lexical_score"] > 0 # "sundial" and "amber" are in the question + + +def test_a_paraphrase_is_found_by_meaning_not_by_shared_words(): + """Lexical matching must not replace semantic retrieval. The paraphrase + shares none of F's distinctive words, yet ranks first.""" + for name in ("ranking_crowded", "independent_default"): + paraphrase = scenario(name)["ranking_variants"]["paraphrase"] + assert paraphrase["rank"] == 1 and paraphrase["selected"] + # Only "Mara" is shared, which is far less than the direct question holds. + direct = scenario(name)["ranking_variants"]["direct"] + assert paraphrase["lexical_score"] < direct["lexical_score"] / 2 + + +def test_a_context_dependent_question_needs_the_scene(): + """"I ask her what she keeps up there" names nothing F's memory holds. The + scene the last narration set up (Mara, the top shelf, a kettle) is what + finds it; without that context it ranks last.""" + result = scenario("ranking_context_dependent") + ranked = result["diagnosis"]["ranked"] + assert ranked["lexical_score"] == 0.0 + assert ranked["rank"] <= 4 + assert "top shelf" in ranked["query"]["context"] + assert result["ranking_variants"]["input_only"]["rank"] > 4 + + +def test_an_unrelated_rare_word_does_not_outrank_the_relevant_memory(): + """Negative control: the paraphrase plus a place only one other memory + holds. The decoy gains lexical score, and still ranks below F.""" + for name in ("ranking_crowded", "independent_default"): + control = scenario(name)["ranking_variants"]["rare_word_with_paraphrase"] + assert control["decoy_lexical_score"] > control["lexical_score"] + assert control["rank"] == 1 + assert control["decoy_rank"] > control["rank"] + + +def test_common_words_contribute_nothing(): + common = scenario("independent_default")["ranking_variants"]["common_words"] + assert common["lexical_score"] == 0.0 # ---------------------------------------------------------- capacity/eviction -def test_past_capacity_the_early_memory_is_evicted_and_the_stage_says_so(): - """Diagnosis, not a requirement: what the current eviction rule does to F.""" - result = scenario("past_capacity") - assert result["diagnosis"]["created"]["yes"] is True - assert result["diagnosis"]["verdict"] == "created_but_evicted" - eviction = result["eviction"] - assert eviction["f_evicted_at_turn"] is not None - # It was retrieved while the bank was small, stopped being retrieved once - # recent narration filled the top-k, and was then the least recently used. - assert eviction["f_use_count_when_evicted"] > 0 - assert result["f_last_use_increase_turn"] < eviction["f_evicted_at_turn"] - assert eviction["f_memory_was_first_evicted"] is True +@pytest.mark.parametrize("name", ["past_capacity", "past_capacity_pinned", "past_capacity_low_top_k"]) +def test_past_capacity_the_early_memory_is_retained(name): + """v1.1 WP-B.2. On v1.0.0 all three are `created_but_evicted`: F was the + least recently used row once recent narration stopped retrieving it, and + went first (turns 21, 21 and 36). Coverage-first eviction keeps the only + memory of the opening, so it stays active and is recalled at depth 106.""" + result = scenario(name) + assert result["isolation"]["ok"], result["isolation"] + diagnosis = result["diagnosis"] + assert diagnosis["created"]["yes"] and diagnosis["retained"]["yes"] + assert result["eviction"]["f_evicted_at_turn"] is None + # The bank really was past capacity, and stayed bounded. + assert result["eviction"]["first_eviction_turn"] is not None + assert all(t["active"] <= result["scenario"]["capacity"] + (1 if result["scenario"]["pin_first_memory"] else 0) + for t in result["trace"]) + assert result["trace"][-1]["total"] > result["scenario"]["capacity"] -def test_at_a_lower_top_k_the_early_memory_ages_out_after_it_stops_being_retrieved(): - """Diagnosis with most of the bank unretrieved on any turn, nearer the - shipped 5-in-80 ratio. F is not simply the oldest row: it is evicted some - turns after recent narration stopped pulling it into the top-k, which is - what ordering by last use does to a fact nothing recent mentions.""" - result = scenario("past_capacity_low_top_k") - eviction = result["eviction"] - assert result["diagnosis"]["created"]["yes"] is True - assert result["diagnosis"]["verdict"] == "created_but_evicted" - assert eviction["f_use_count_when_evicted"] > 0 - assert result["f_last_use_increase_turn"] < eviction["f_evicted_at_turn"] - assert eviction["first_eviction_turn"] <= eviction["f_evicted_at_turn"] - assert eviction["created_and_evicted_same_turn"] == [] +def test_past_capacity_the_bank_still_describes_the_whole_story(): + """What the rule buys in general, not only for F: the active bank reaches + from the opening to the newest block, and no stretch between them goes + undescribed for more than twice the average spacing a bank of this capacity + can afford (story span / capacity). On v1.0.0 these banks began at depths 36 + and 18: the opening was simply gone.""" + for name in ("past_capacity", "past_capacity_low_top_k"): + result = scenario(name) + cover = result["trace"][-1]["coverage"] + assert cover["first_start"] == 0 + assert cover["last_end"] >= result["recall_depth"] - 2 * memorybank.MEMORY_INTERVAL + assert cover["largest_gap"] <= 2 * (cover["last_end"] + 1) / result["scenario"]["capacity"] def test_no_memory_is_evicted_by_the_same_pass_that_created_it(): - """The frozen-bank regression the current rule fixed, still holding.""" - for name in ("past_capacity", "past_capacity_pinned"): + """The frozen-bank regression the v1.0.0 rule fixed, still holding.""" + for name in ("past_capacity", "past_capacity_pinned", "past_capacity_low_top_k"): assert scenario(name)["eviction"]["created_and_evicted_same_turn"] == [] @@ -180,25 +246,28 @@ def test_a_pinned_memory_survives_capacity(): assert eviction["pinned_memory_forgotten"] is False -@pytest.mark.xfail(strict=True, reason=( - "WP-B.1 diagnosis on this tree: past memory_bank_capacity the planting-era " - "memory is evicted first, because it was never retrieved and eviction orders " - "by last use, then creation. B.2 must flip this deliberately.")) def test_acceptance_an_early_fact_is_recalled_from_memory_past_capacity(): - assert scenario("past_capacity")["diagnosis"]["verdict"] == "injected" + """Was `xfail(strict=True)` in WP-B.1; B.2 fixed the eviction rule.""" + for name in ("past_capacity", "past_capacity_pinned", "past_capacity_low_top_k"): + assert scenario(name)["diagnosis"]["verdict"] == "injected" # ---------------------------------------------------------- creation window -def test_a_fact_early_in_a_long_block_never_reaches_the_summariser(): +def test_a_fact_early_in_a_long_block_now_reaches_the_summariser(): + """v1.1 WP-B.2. On v1.0.0 this block (2,079 tokens) was cut to its last + 2,000, the fact at its start was never seen, and the stage was + `not_created`. The excerpt is now the block's opening and end.""" result = scenario("long_block_fact_early") created = result["diagnosis"]["created"] covering = created["covering_memories"] assert covering, "the long block must have been summarised" assert covering[0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS assert covering[0]["fact_in_block"] is True - assert covering[0]["fact_in_summariser_excerpt"] is False - assert result["diagnosis"]["verdict"] == "not_created" + assert covering[0]["fact_in_summariser_excerpt"] is True + assert created["yes"] is True + assert created["source_start"] <= result["plant_depth"] <= created["source_end"] + assert memorybank.EXCERPT_OMISSION_MARKER not in created["memory_text"] def test_the_same_fact_late_in_the_same_sized_block_does(): @@ -209,12 +278,69 @@ def test_the_same_fact_late_in_the_same_sized_block_does(): assert result["diagnosis"]["created"]["yes"] is True -@pytest.mark.xfail(strict=True, reason=( - "WP-B.1 diagnosis on this tree: the summariser reads only the last " - f"{memorybank.MEMORY_EXCERPT_TOKENS} tokens of a block, so a fact early in a " - "long block is never seen. B.2 must flip this deliberately.")) def test_acceptance_a_fact_early_in_a_long_block_is_remembered(): + """Was `xfail(strict=True)` in WP-B.1; B.2 changed the excerpt.""" assert scenario("long_block_fact_early")["diagnosis"]["created"]["yes"] is True + assert scenario("long_block_fact_early")["diagnosis"]["verdict"] == "injected" + + +# ------------------------------------------ WP-B.2 full deterministic acceptance + +def test_acceptance_full_isolation_holds_on_every_turn(): + """`independent_full`: long blocks, a crowded query, a bank past capacity. + F must be carried by memory alone for the whole run, not only at recall.""" + result = scenario("independent_full") + assert result["plant_depth"] <= 3 and result["recall_depth"] >= 100 + assert not any(t["f_in_state"] for t in result["trace"]) + assert not any(t["f_in_summary"] for t in result["trace"]) + assert result["isolation"]["ok"], result["isolation"] + for check in ("state_document", "state_snapshots", "later_narration", "summary", + "knowledge", "recent_history", "state_section"): + assert result["isolation"]["checks"][check]["ok"], check + + +def test_acceptance_full_every_stage_passes_past_capacity_with_long_blocks(): + """On v1.0.0 this fixture fails at creation: every block is over 2,000 + tokens, and the fact at the start of the first one is never summarised.""" + result = scenario("independent_full") + diagnosis = result["diagnosis"] + assert result["trace"][-1]["total"] > result["scenario"]["capacity"] + assert diagnosis["created"]["covering_memories"][0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS + assert diagnosis["created"]["yes"] and diagnosis["retained"]["yes"] + assert diagnosis["ranked"]["yes"] and diagnosis["ranked"]["replica_matches_stored_selection"] + assert diagnosis["injected"]["yes"] + assert diagnosis["verdict"] == "injected" + + +def test_acceptance_full_provenance_resolves_to_the_planting_turn(): + provenance = scenario("independent_full")["provenance"] + assert provenance["recorded"] is not None + assert provenance["range_covers_plant"] and provenance["matches_row"] + assert provenance["source_block_holds_planting"] is True + assert provenance["recorded"]["authority"] == memorybank.ACCEPTED_STORY + + +def test_acceptance_full_is_the_long_run_independent_memory_verdict(): + """The same measurements, judged by the long-run tool's own verdict.""" + from tools import m11_long_run as lr + + result = scenario("independent_full") + checks = result["isolation"]["checks"] + diagnosis = result["diagnosis"] + verdict = lr._independent_memory_verdict({ + "independent_planted_depth": result["plant_depth"], + "planted_turn_outside_history": checks["recent_history"]["ok"], + "absent_from_state": checks["state_document"]["ok"] and checks["state_snapshots"]["ok"] + and not any(t["f_in_state"] for t in result["trace"]), + "absent_from_summary": checks["summary"]["ok"] + and not any(t["f_in_summary"] for t in result["trace"]), + "absent_from_knowledge": checks["knowledge"]["ok"], + "absent_from_later_narration": checks["later_narration"]["ok"], + "memory_covering_planting_carries_fact": diagnosis["created"]["yes"], + "memory_forgotten": not diagnosis["retained"]["yes"], + "memory_injected": diagnosis["injected"]["yes"], + }) + assert verdict == "recovered_through_memory_independent" # ------------------------------------------------------- lineage control (G) diff --git a/backend/tests/test_v11_b2_memory_eviction.py b/backend/tests/test_v11_b2_memory_eviction.py new file mode 100644 index 0000000..4d2b226 --- /dev/null +++ b/backend/tests/test_v11_b2_memory_eviction.py @@ -0,0 +1,277 @@ +"""v1.1 WP-B.2 (B2.2): which memory a full bank lets go of. + +v1.0.0 evicted the least recently used memory. B.1 showed that this discards +the only memory of an early stretch first, because retrieval follows the present +scene and nothing recent resembles it. `memorybank.eviction_order` now thins the +bank where it is densest and keeps the opening and the newest stretch, with +recency as the tie-break and least-recently-used as the fallback. + +The scenario-level tests (the planted fact kept past capacity) are in +`test_v11_b1_memory_diagnostic.py`; the v1.0.0 eviction tests in +`test_memory_retrieval.py` still pass unchanged, because their memories carry +no source range and take the fallback. + + python -m pytest tests/test_v11_b2_memory_eviction.py -v +""" + +import random +from collections import namedtuple +from datetime import datetime, timedelta + +import pytest +from sqlalchemy import select + +from app import memorybank, models, tree +from app.context import lineage +from app.database import Base, SessionLocal, engine + +T0 = datetime(2026, 1, 1, 12, 0, 0) +Row = namedtuple("Row", "id pinned source_start source_end last_used_at created_at use_count") + + +def row(id, start, end=None, *, pinned=False, used=None, created=None, uses=0): + """A memory as eviction sees it. Times are minutes after T0.""" + return Row(id, pinned, start, (start + 5) if end is None and start is not None else end, + None if used is None else T0 + timedelta(minutes=used), + T0 + timedelta(minutes=id if created is None else created), uses) + + +def blocks(n, *, first_id=1): + return [row(first_id + i, 6 * i) for i in range(n)] + + +def largest_gap_from_opening(rows): + """The longest uncovered run of depths from depth 0 to the last memory.""" + ordered = sorted((r.source_start, r.source_end) for r in rows) + gaps = [ordered[0][0]] + reach = ordered[0][1] + for start, end in ordered[1:]: + gaps.append(max(0, start - reach - 1)) + reach = max(reach, end) + return max(gaps) + + +# ------------------------------------------------------------ the pure order + + +def test_the_opening_and_the_newest_memory_are_kept(): + bank = blocks(7) + doomed = memorybank.eviction_order(bank, 5) + assert bank[0].id not in doomed and bank[-1].id not in doomed + assert len(doomed) == 5 + + +def test_the_densest_stretch_is_thinned_first(): + # Memories every 6 depths to 30, then a sparse stretch. Removing one of the + # dense ones leaves a 6-depth hole; removing a sparse one leaves far more. + bank = [row(1, 0), row(2, 6), row(3, 12), row(4, 18), row(5, 60), row(6, 120), row(7, 180)] + assert memorybank.eviction_order(bank, 1)[0] in {2, 3, 4} + assert set(memorybank.eviction_order(bank, 2)) <= {2, 3, 4} + + +def test_a_stretch_two_memories_describe_loses_one_of_them_first(): + """A shared start (a re-played stretch, or a sibling line) leaves no hole. + Of the two, the less recently used goes, even though a unique memory + elsewhere is older and less used than both.""" + bank = [row(1, 0), row(2, 6, used=5), row(3, 12, used=50), row(4, 12, used=40), row(5, 18)] + assert memorybank.eviction_order(bank, 1) == [4] + + +def test_equal_holes_fall_to_the_least_recently_used(): + bank = [row(1, 0), row(2, 6, used=30), row(3, 12, used=10), row(4, 18, used=20), row(5, 24)] + assert memorybank.eviction_order(bank, 1) == [3] + + +def test_a_newborn_can_be_the_legitimate_first_to_go(): + """The frozen bank is about a newborn losing to a count it cannot have yet. + A newborn that only repeats a stretch another memory describes, one used + after it was written, is legitimately the first to go.""" + bank = [row(1, 0), row(2, 6, used=100), row(3, 12), row(4, 6, created=90)] + assert memorybank.eviction_order(bank, 1) == [4] + + +def test_the_newest_memory_is_not_evicted_by_the_bank_it_joins(): + """The frozen-bank regression under the new rule: every older memory has + been used, the newborn never has, and it still stays.""" + bank = [row(i, 6 * (i - 1), used=200 + i, uses=3) for i in range(1, 6)] + newborn = row(6, 30, created=300) + assert newborn.id not in memorybank.eviction_order(bank + [newborn], 1) + + +def test_pins_are_never_taken_but_still_count_as_coverage(): + bank = [row(1, 0), row(2, 6, pinned=True), row(3, 12), row(4, 18, pinned=True), row(5, 24)] + doomed = memorybank.eviction_order(bank, 10) + assert not {2, 4} & set(doomed) + # With the pins covering 6 and 18, memory 3's hole is only its own block. + assert doomed[0] == 3 + + +def test_memories_without_a_range_take_the_least_recently_used_fallback(): + hand_written = [row(1, None, None, used=30), row(2, None, None, used=10), + row(3, None, None, used=20)] + assert memorybank.eviction_order(hand_written, 3) == [2, 3, 1] + + +def test_the_fallback_is_used_only_once_no_interior_memory_remains(): + bank = [row(1, 0, used=1), row(2, 6, used=90), row(3, 12, used=2), + row(10, None, None, used=0)] + order = memorybank.eviction_order(bank, 4) + assert order[0] == 2 # the interior memory, although recently used + assert order[1:] == [10, 1, 3] # then least recently used + + +def test_the_order_does_not_depend_on_row_order(): + bank = [row(i, 6 * (i - 1), used=(i * 37) % 11, uses=i % 3) for i in range(1, 30)] + bank += [row(40, 12), row(41, 12)] # a shared start with identical timestamps + expected = memorybank.eviction_order(bank, 20) + for seed in range(5): + shuffled = bank[:] + random.Random(seed).shuffle(shuffled) + assert memorybank.eviction_order(shuffled, 20) == expected + + +def test_a_tie_on_every_signal_is_broken_by_id(): + bank = [row(1, 0), row(9, 6, created=0), row(4, 12, created=0), row(20, 18)] + assert memorybank.eviction_order(bank, 1) == [4] + + +@pytest.mark.parametrize("seed", range(8)) +def test_capacity_holds_and_pins_survive_for_any_bank(seed): + rng = random.Random(seed) + bank = [] + for i in range(1, rng.randint(2, 60)): + start = None if rng.random() < 0.15 else rng.randrange(0, 400) + bank.append(row(i, start, None if start is None else start + rng.choice([3, 5, 8]), + pinned=rng.random() < 0.1, used=rng.choice([None, rng.randrange(500)]), + uses=rng.randrange(4))) + capacity = rng.randint(1, 30) + overflow = len(bank) - capacity + doomed = memorybank.eviction_order(bank, max(0, overflow)) + pinned = {r.id for r in bank if r.pinned} + assert not pinned & set(doomed) + assert len(set(doomed)) == len(doomed) + remaining = len(bank) - len(doomed) + assert remaining == max(capacity, len(pinned)) if overflow > 0 else remaining == len(bank) + + +@pytest.mark.parametrize("n, capacity, irregular", [(60, 10, False), (500, 80, False), (500, 80, True)]) +def test_a_long_bank_keeps_describing_the_whole_story(n, capacity, irregular): + """The general property, with nothing ever retrieved: memories arrive one + block at a time and the bank is kept at capacity. The opening stays, and no + stretch goes undescribed for more than twice the average spacing. Least + recently used order, on the same arrivals, keeps only the newest stretch.""" + rng = random.Random(n) + kept, lru = [], [] + depth = 0 + for i in range(1, n + 1): + size = rng.choice([4, 6, 6, 9]) if irregular else 6 + memory = row(i, depth, depth + size - 1) + depth += size + kept.append(memory) + lru.append(memory) + if len(kept) > capacity: + doomed = set(memorybank.eviction_order(kept, len(kept) - capacity)) + kept = [m for m in kept if m.id not in doomed] + lru = sorted(lru, key=lambda m: (m.created_at, m.id))[len(lru) - capacity:] + assert len(kept) == capacity + assert min(m.source_start for m in kept) == 0 + assert max(m.id for m in kept) == n + assert largest_gap_from_opening(kept) <= 2 * depth / capacity + assert largest_gap_from_opening(lru) > depth / 2 # v1.0.0 order: the opening is gone + + +# --------------------------------------------------------- on real rows + + +@pytest.fixture() +def db(): + Base.metadata.create_all(bind=engine) + memorybank._vector_cache.clear() + session = SessionLocal() + try: + yield session + finally: + session.close() + memorybank._vector_cache.clear() + Base.metadata.drop_all(bind=engine) + + +@pytest.fixture() +def adventure(db): + user = models.User(is_guest=False, email="b2-evict@example.com") + db.add(user) + db.flush() + settings = models.Settings(user_id=user.id, model="m", embedding_model="e", + memory_bank_capacity=3) + adv = models.Adventure(user_id=user.id, title="Evict", script_state={}, memory_bank_enabled=True) + db.add_all([settings, adv]) + db.commit() + adv.settings_row = settings + return adv + + +def test_the_pass_changes_nothing_but_forgotten(db, adventure): + for i in range(6): + memory = models.Memory(adventure_id=adventure.id, text=f"block {i}", + source_start=6 * i, source_end=6 * i + 5, branch_id=None, depth=6 * i + 5) + db.add(memory) + db.commit() + columns = (models.Memory.id, models.Memory.text, models.Memory.source_start, + models.Memory.source_end, models.Memory.branch_id, models.Memory.depth, + models.Memory.pinned, models.Memory.use_count, models.Memory.last_used_at) + before = {r.id: tuple(r) for r in db.execute(select(*columns)).all()} + memorybank._evict_over_capacity(adventure, adventure.settings_row, db) + after = {r.id: tuple(r) for r in db.execute(select(*columns)).all()} + assert before == after + active = db.execute(select(models.Memory.id).where(models.Memory.forgotten.is_(False))).scalars().all() + assert len(active) == 3 + assert min(active) == min(before) and max(active) == max(before) # the boundaries + + +def test_eviction_does_not_make_an_abandoned_lines_memory_eligible(db, adventure): + """Eviction and lineage are separate: the pass decides only `forgotten`, so + a memory on a line the story left is exactly as ineligible afterwards.""" + trunk = [] + for i in range(4): + action = models.Action(adventure_id=adventure.id, type="ai", text=f"trunk {i}") + tree.place_action(db, adventure, action) + db.add(action) + db.flush() + trunk.append(action) + abandoned_node = models.Action(adventure_id=adventure.id, type="ai", text="the abandoned line") + tree.place_action(db, adventure, abandoned_node) + db.add(abandoned_node) + db.flush() + abandoned = models.Memory(adventure_id=adventure.id, text="on the abandoned line", + source_start=4, source_end=4) + tree.attach_memory(abandoned, abandoned_node) + db.add(abandoned) + db.commit() + # Move the head back and diverge, so the abandoned node is off the path. + adventure.head_depth = trunk[-1].depth + db.commit() + from app import head + head.fork_if_behind_head(db, adventure) + divergent = models.Action(adventure_id=adventure.id, type="ai", text="the new line") + tree.place_action(db, adventure, divergent) + db.add(divergent) + db.flush() + for i, node in enumerate(trunk + [divergent]): + memory = models.Memory(adventure_id=adventure.id, text=f"active {i}", + source_start=node.depth, source_end=node.depth) + tree.attach_memory(memory, node) + db.add(memory) + db.commit() + + def eligible(): + return set(db.execute(select(models.Memory.id).where( + models.Memory.adventure_id == adventure.id, + lineage.path_of(db, adventure).clause(models.Memory), + models.Memory.forgotten.is_(False))).scalars().all()) + + assert abandoned.id not in eligible() + memorybank._evict_over_capacity(adventure, adventure.settings_row, db) + db.expire_all() + assert abandoned.id not in eligible() + assert len(db.execute(select(models.Memory.id).where( + models.Memory.forgotten.is_(False))).scalars().all()) == 3 diff --git a/backend/tests/test_v11_b2_memory_excerpt.py b/backend/tests/test_v11_b2_memory_excerpt.py new file mode 100644 index 0000000..ca65e03 --- /dev/null +++ b/backend/tests/test_v11_b2_memory_excerpt.py @@ -0,0 +1,189 @@ +"""v1.1 WP-B.2 (B2.3): what the memory summariser is shown of a long block. + +v1.0.0 sent the last 2,000 tokens of a block, so a fact early in a longer block +never reached the summariser (B.1 §E). A block that fits is still sent whole. A +longer one is now sent as its opening and its end, with a marker between them, +inside the same 2,000-token budget. + +The scenario-level test (the planted fact early in a long block, remembered) is +in `test_v11_b1_memory_diagnostic.py`. + + python -m pytest tests/test_v11_b2_memory_excerpt.py -v +""" + +import asyncio +import random + +import pytest + +from app import memorybank, models, tree +from app.context import builder +from app.database import Base, SessionLocal, engine + +BUDGET = memorybank.MEMORY_EXCERPT_TOKENS +MARKER = memorybank.EXCERPT_OMISSION_MARKER +FILLER = "The travellers walked the long grey road north past the salt market and the reed beds. " + + +def words_to_tokens(tokens: int) -> str: + """Filler at least `tokens` long.""" + text = FILLER + while builder.count_tokens(text) < tokens: + text += FILLER + return text + + +# --------------------------------------------------------------- the excerpt + + +def test_a_block_that_fits_is_sent_whole_and_unchanged(): + raw = words_to_tokens(BUDGET - 200) + assert builder.count_tokens(raw) <= BUDGET + assert memorybank.memory_excerpt(raw) == raw + + +def test_a_block_of_exactly_the_budget_is_unchanged(): + raw = words_to_tokens(BUDGET) + tokens = memorybank._excerpt_encoding().encode(raw)[:BUDGET] + exact = memorybank._excerpt_encoding().decode(tokens) + if builder.count_tokens(exact) == BUDGET: + assert memorybank.memory_excerpt(exact) == exact + + +def test_a_long_block_keeps_its_opening_and_its_end_in_order(): + opening = "Mara slipped the amber sundial inside the cracked teapot. " + ending = "Aldric finally reached the north gate at dawn." + raw = opening + words_to_tokens(3 * BUDGET) + ending + excerpt = memorybank.memory_excerpt(raw) + assert excerpt.startswith(opening) + assert excerpt.endswith(ending) + head, _, tail = excerpt.partition(f"\n\n{MARKER}\n\n") + assert tail, "the marker must sit between the two parts" + assert excerpt.index(opening) < excerpt.index(MARKER) < excerpt.index(ending) + + +def test_the_split_is_even_and_documented(): + raw = words_to_tokens(4 * BUDGET) + head_budget, tail_budget = memorybank.excerpt_split(BUDGET) + marker_tokens = builder.count_tokens(f"\n\n{MARKER}\n\n") + assert head_budget + tail_budget + marker_tokens == BUDGET + assert abs(head_budget - tail_budget) <= 1 + excerpt = memorybank.memory_excerpt(raw) + head, _, tail = excerpt.partition(f"\n\n{MARKER}\n\n") + # Each part is cut as a run of `head_budget` / `tail_budget` tokens. Measured + # on its own, a cut run can come to one token more, because the text either + # side of the cut tokenises differently once it is separated; the hard limit + # is the whole excerpt, tested below. + assert builder.count_tokens(head) <= head_budget + 1 + assert builder.count_tokens(tail) <= tail_budget + 1 + assert builder.count_tokens(excerpt) <= BUDGET + + +@pytest.mark.parametrize("extra", [1, 7, 500, BUDGET, 9 * BUDGET]) +def test_the_excerpt_never_exceeds_the_budget(extra): + raw = words_to_tokens(BUDGET + extra) + excerpt = memorybank.memory_excerpt(raw) + assert builder.count_tokens(excerpt) <= BUDGET + assert memorybank.memory_excerpt(raw) == excerpt # deterministic + + +@pytest.mark.parametrize("seed", range(4)) +def test_the_budget_holds_for_awkward_text(seed): + """Token boundaries can merge differently once the parts are rejoined, and + text that is not plain English tokenises unevenly. The budget still holds.""" + rng = random.Random(seed) + alphabet = "abcdefghij ÄÖÜ ßé漢字かな 🙂🐉 \n\t.,;:—'\"" + raw = "".join(rng.choice(alphabet) for _ in range(12000)) + assert builder.count_tokens(memorybank.memory_excerpt(raw)) <= BUDGET + + +def test_a_fact_in_the_middle_of_a_very_long_block_is_still_omitted(): + """The documented limit of a bounded excerpt: head and tail, not everything.""" + half = words_to_tokens(3 * BUDGET) + raw = half + "Mara slipped the amber sundial inside the cracked teapot. " + half + assert "sundial" not in memorybank.memory_excerpt(raw) + + +# ------------------------------------------------------------ memory creation + + +class EchoSummariser: + """Returns the whole excerpt it was given as the memory: the worst case for + a marker leaking into stored text.""" + + def __init__(self): + self.users: list[str] = [] + + async def complete(self, system, user, *, temperature=0.3, max_tokens=400): + self.users.append(user) + return user.split("Story excerpt:\n\n", 1)[-1].rsplit("\n\nMemory:", 1)[0] + + +@pytest.fixture() +def db(): + Base.metadata.create_all(bind=engine) + session = SessionLocal() + try: + yield session + finally: + session.close() + Base.metadata.drop_all(bind=engine) + + +def campaign(db, texts): + user = models.User(is_guest=False, email="b2-excerpt@example.com") + db.add(user) + db.flush() + db.add(models.Settings(user_id=user.id, model="m", embedding_model="")) + adventure = models.Adventure(user_id=user.id, title="Excerpt", script_state={}, + auto_summarize=True, memory_bank_enabled=True) + db.add(adventure) + db.flush() + nodes = [] + for i, text in enumerate(texts): + action = models.Action(adventure_id=adventure.id, type="ai" if i % 2 else "do", text=text) + tree.place_action(db, adventure, action) + db.add(action) + db.flush() + nodes.append(action) + db.commit() + return adventure, nodes + + +def write_memory(db, adventure, monkeypatch, summariser): + monkeypatch.setattr(memorybank, "MEMORY_START", 0) + monkeypatch.setattr(memorybank, "MAX_MEMORIES_PER_RUN", 1) + monkeypatch.setattr(memorybank, "summary_provider", lambda s: summariser) + settings = db.query(models.Settings).first() + asyncio.run(memorybank._create_due_memories(adventure, settings, db)) + return db.query(models.Memory).filter_by(adventure_id=adventure.id).all() + + +def test_the_marker_is_never_stored_as_part_of_a_memory(db, monkeypatch): + long = words_to_tokens(800) + adventure, _ = campaign(db, [long] * (memorybank.MEMORY_INTERVAL + memorybank.SETTLE_SLACK)) + summariser = EchoSummariser() + [memory] = write_memory(db, adventure, monkeypatch, summariser) + assert MARKER in summariser.users[0] # the summariser was told + assert MARKER not in memory.text # and the memory does not repeat it + assert "[…" not in memory.text and "omitted" not in memory.text + + +def test_a_short_block_is_prompted_exactly_as_before(db, monkeypatch): + texts = [f"Short action {i}." for i in range(memorybank.MEMORY_INTERVAL + memorybank.SETTLE_SLACK)] + adventure, _ = campaign(db, texts) + summariser = EchoSummariser() + write_memory(db, adventure, monkeypatch, summariser) + block = "\n\n".join(texts[:memorybank.MEMORY_INTERVAL]) + assert summariser.users[0] == f"Story excerpt:\n\n{block}\n\nMemory:" + + +def test_a_long_blocks_memory_keeps_its_source_provenance(db, monkeypatch): + early = "Mara slipped the amber sundial inside the cracked teapot. " + words_to_tokens(900) + texts = [early] + [words_to_tokens(900)] * (memorybank.MEMORY_INTERVAL + memorybank.SETTLE_SLACK - 1) + adventure, nodes = campaign(db, texts) + [memory] = write_memory(db, adventure, monkeypatch, EchoSummariser()) + block = nodes[:memorybank.MEMORY_INTERVAL] + assert "sundial" in memory.text # the early fact reached the summariser + assert (memory.source_start, memory.source_end) == (block[0].depth, block[-1].depth) + assert (memory.branch_id, memory.depth) == (block[-1].branch_id, block[-1].depth) diff --git a/backend/tests/test_v11_b2_memory_ranking.py b/backend/tests/test_v11_b2_memory_ranking.py new file mode 100644 index 0000000..fbc7582 --- /dev/null +++ b/backend/tests/test_v11_b2_memory_ranking.py @@ -0,0 +1,326 @@ +"""v1.1 WP-B.2 (B2.1): what memory retrieval searches for, and how it scores. + +B.1 found the planting-era memory created and retained but ranked out of +`memory_top_k`, because the query was three turns of narration with the player's +question at the end. The query is now the player's input plus a short scene +context, and the score adds one transparent lexical term over the input. + +These tests pin the pieces. The end-to-end fixture tests (crowded bank, +context-dependent question, negative controls) are in +`test_v11_b1_memory_diagnostic.py`, beside the diagnostic they use. + + python -m pytest tests/test_v11_b2_memory_ranking.py -v +""" + +import asyncio +import math + +import pytest +from sqlalchemy import event + +from app import memorybank, models, tree +from app.context import builder +from app.database import Base, SessionLocal, engine + +# ------------------------------------------------------------------ lexical + + +def test_terms_are_folded_but_not_stemmed(): + terms = memorybank.lexical_terms("The tavern's teapots, the glass and the SUNDIAL") + assert {"tavern", "teapot", "glass", "sundial"} <= terms + assert "the" not in terms # the knowledge path's stop list + assert "glas" not in terms # a word ending in "ss" is not a plural + + +def test_a_word_every_candidate_holds_weighs_nothing(): + scores = memorybank.lexical_scores( + frozenset({"travellers"}), {1: frozenset({"travellers", "road"}), 2: frozenset({"travellers"})}) + assert scores == {1: 0.0, 2: 0.0} + + +def test_a_rarer_word_weighs_more_than_a_common_one(): + scores = memorybank.lexical_scores( + frozenset({"sundial", "road"}), + {1: frozenset({"sundial"}), 2: frozenset({"road"}), 3: frozenset({"road"}), + 4: frozenset({"gate"})}) + assert scores[1] > scores[2] == scores[3] > scores[4] == 0.0 + + +def test_a_single_rare_word_of_a_longer_question_is_only_its_share(): + """The share is over the whole question, so one incidental word match + cannot score like a memory that answers it.""" + question = frozenset({"where", "amber", "sundial", "fish"}) + scores = memorybank.lexical_scores( + question, {1: frozenset({"fish"}), 2: frozenset({"amber", "sundial"}), 3: frozenset({"road"})}) + assert 0.0 < scores[1] < scores[2] <= 1.0 + assert scores[1] < 0.5 + + +def test_scores_are_bounded_and_empty_inputs_score_zero(): + candidates = {1: frozenset({"a1", "b2"}), 2: frozenset({"a1"})} + assert all(0.0 <= v <= 1.0 for v in memorybank.lexical_scores(frozenset({"a1", "b2"}), candidates).values()) + assert memorybank.lexical_scores(frozenset(), candidates) == {1: 0.0, 2: 0.0} + assert memorybank.lexical_scores(frozenset({"a1"}), {}) == {} + + +# ------------------------------------------------------------------ scoring + + +def _unit(angle): + return [math.cos(angle), math.sin(angle)] + + +def test_ties_are_broken_by_id_not_by_row_order(): + held = {7: [1.0, 0.0], 3: [1.0, 0.0], 5: [1.0, 0.0]} + rows = memorybank.score_candidates([7, 3, 5], held, {}, [1.0, 0.0], None, []) + assert [row[1] for row in rows] == [3, 5, 7] + + +def test_the_semantic_score_mixes_input_and_context_by_the_fixed_weight(): + held = {1: [1.0, 0.0]} + [(final, _, semantic, lexical)] = memorybank.score_candidates( + [1], held, {}, [1.0, 0.0], [0.0, 1.0], []) + assert semantic == pytest.approx(memorybank.INPUT_WEIGHT) + assert final == semantic and lexical == 0.0 + # Either part alone is used as it is. + [(_, _, only_context, _)] = memorybank.score_candidates([1], held, {}, None, [0.0, 1.0], []) + assert only_context == pytest.approx(0.0) + + +@pytest.mark.parametrize("margin, relevant_first", [(0.01, True), (-0.01, False)]) +def test_a_lexical_match_moves_a_memory_at_most_the_lexical_weight(margin, relevant_first): + """The bound that keeps rarity from overruling meaning: a memory more than + `LEXICAL_WEIGHT` behind semantically cannot pass one ahead of it, however + rare the word it shares.""" + decoy_cos = 1.0 - memorybank.LEXICAL_WEIGHT - margin + held = {1: [1.0, 0.0], 2: _unit(math.acos(decoy_cos))} + terms = {1: frozenset(), 2: frozenset({"zeppelin"})} + rows = memorybank.score_candidates([1, 2], held, terms, [1.0, 0.0], None, ["zeppelin"]) + order = [row[1] for row in rows] + assert (order[0] == 1) is relevant_first + + +# ---------------------------------------------------- the query, on real rows + + +@pytest.fixture() +def db(): + Base.metadata.create_all(bind=engine) + memorybank._vector_cache.clear() + memorybank._terms_cache.clear() + session = SessionLocal() + try: + yield session + finally: + session.close() + memorybank._vector_cache.clear() + memorybank._terms_cache.clear() + Base.metadata.drop_all(bind=engine) + + +@pytest.fixture(autouse=True) +def restore_embedding_provider(): + real = memorybank.embedding_provider + try: + yield + finally: + memorybank.embedding_provider = real + + +@pytest.fixture() +def settings(db): + user = models.User(is_guest=False, email="b2-rank@example.com") + db.add(user) + db.flush() + row = models.Settings(user_id=user.id, model="m", embedding_model="stub-embed", + memory_top_k=2, memory_bank_capacity=80) + db.add(row) + db.commit() + return row + + +SCENE_STATE = { + "entities": {"mara": {"type": "character", "name": "Mara"}, + "tavern": {"type": "location", "name": "The Crooked Lantern"}}, + "scene": {"summary": "Closing time", "location": "tavern", "present": ["mara"]}, +} + + +def make_adventure(db, settings, texts, state=None): + """`texts` is `[(type, text)]`, oldest first, each placed on the tree.""" + adventure = models.Adventure(user_id=settings.user_id, title="Rank", script_state={}, + memory_bank_enabled=True, narrative_state=state or {}) + db.add(adventure) + db.flush() + placed = [] + for kind, text in texts: + action = models.Action(adventure_id=adventure.id, type=kind, text=text) + tree.place_action(db, adventure, action) + db.add(action) + db.flush() + placed.append(action) + db.commit() + return adventure, placed + + +def test_the_query_is_the_players_input_and_the_scene(db, settings): + adventure, _ = make_adventure(db, settings, [ + ("start", "Rain over the harbour."), + ("ai", "Mara wipes down the counter and glances up at the shelf."), + ("do", "> You ask Mara about the brass dial."), + ], state=SCENE_STATE) + query = memorybank.retrieval_query(adventure) + assert query["input"] == "> You ask Mara about the brass dial." + assert "The Crooked Lantern" in query["context"] and "Mara" in query["context"] + assert "glances up at the shelf" in query["context"] + assert "brass" in query["input_terms"] and "dial" in query["input_terms"] + + +def test_a_continue_turn_has_no_input_and_searches_by_the_scene(db, settings): + adventure, _ = make_adventure(db, settings, [ + ("do", "> You sit down."), + ("ai", "The fire burns low in the grate."), + ]) + query = memorybank.retrieval_query(adventure) + assert query["input"] == "" and query["input_terms"] == [] + assert "fire burns low" in query["context"] + + +def test_a_retry_searches_with_the_input_it_is_retrying(db, settings): + adventure, placed = make_adventure(db, settings, [ + ("ai", "The market is quiet."), + ("do", "> You ask about the sundial."), + ("ai", "A discarded attempt about lanterns."), + ]) + query = memorybank.retrieval_query(adventure, exclude_action_id=placed[-1].id) + assert query["input"] == "> You ask about the sundial." + assert "lanterns" not in query["context"] + assert "market is quiet" in query["context"] + + +def test_the_query_is_bounded_however_long_the_story(db, settings): + long = "The travellers walked the long grey road north past the salt market. " * 400 + adventure, _ = make_adventure(db, settings, [ + ("ai", long), ("story", long)], state=SCENE_STATE) + query = memorybank.retrieval_query(adventure) + assert builder.count_tokens(query["input"]) <= memorybank.QUERY_INPUT_TOKENS + assert builder.count_tokens(query["context"]) <= ( + memorybank.QUERY_SCENE_TOKENS + memorybank.QUERY_NARRATION_TOKENS + 2) + + +# ------------------------------------------------ retrieval, end to end + + +class SameVector: + """Every text embeds the same, so only the lexical term separates memories.""" + + async def embed(self, texts): + return [[1.0, 0.0, 0.0] for _ in texts] + + +def add_memory(db, adventure, text, **kwargs): + memory = models.Memory(adventure_id=adventure.id, text=text, **kwargs) + db.add(memory) + db.flush() + memorybank.set_vector(memory, [1.0, 0.0, 0.0]) + db.commit() + return memory + + +def retrieve(adventure, settings, **kwargs): + memorybank.embedding_provider = lambda s: SameVector() + return asyncio.run(memorybank.retrieve_memories(adventure, settings, **kwargs)) + + +@pytest.fixture() +def played(db, settings): + adventure, _ = make_adventure(db, settings, [ + ("ai", "The tavern is warm."), + ("do", "> You ask Mara where the amber sundial went."), + ], state=SCENE_STATE) + bank = { + "road": add_memory(db, adventure, "Aldric walked the north road."), + "sundial": add_memory(db, adventure, "Mara hid the amber sundial in the teapot."), + "gate": add_memory(db, adventure, "The gate guard asked for a toll."), + } + return adventure, bank + + +def test_every_used_memory_reports_the_parts_of_its_score(db, settings, played): + adventure, bank = played + result = retrieve(adventure, settings) + first = result["used"][0] + assert first["id"] == bank["sundial"].id + assert first["similarity"] == first["semantic_score"] + assert first["lexical_score"] > 0 + assert first["final_score"] == pytest.approx( + first["semantic_score"] + memorybank.LEXICAL_WEIGHT * first["lexical_score"], abs=2e-4) + assert result["query"]["input"] == "> You ask Mara where the amber sundial went." + assert result["query"]["lexical_weight"] == memorybank.LEXICAL_WEIGHT + assert result["query"]["input_weight"] == memorybank.INPUT_WEIGHT + + +def test_a_pin_is_still_always_used_and_counts_toward_top_k(db, settings, played): + adventure, bank = played + settings.memory_top_k = 1 + bank["gate"].pinned = True + db.commit() + used = retrieve(adventure, settings)["used"] + assert [m["id"] for m in used] == [bank["gate"].id] + assert used[0]["pinned"] is True + + +def memory_text_reads(statements): + return [s for s in statements + if s.lstrip().upper().startswith("SELECT") and "FROM memories" in s + and "memories.text" in s] + + +@pytest.fixture() +def sql_log(): + statements: list[str] = [] + + def record(conn, cursor, statement, parameters, context, executemany): + statements.append(statement) + + event.listen(engine, "before_cursor_execute", record) + try: + yield statements + finally: + event.remove(engine, "before_cursor_execute", record) + + +def test_memory_text_is_read_once_and_then_held(db, settings, played, sql_log): + adventure, _ = played + retrieve(adventure, settings) + sql_log.clear() + result = retrieve(adventure, settings) + reads = memory_text_reads(sql_log) + # Only the detail read of the memories chosen remains. (Every memory here + # embeds identically, so redundancy suppression keeps just one of them.) + assert len(reads) == 1 and reads[0].count("?") == len(result["used"]) + + +def test_a_continue_turn_reads_no_memory_text_to_rank(db, settings, sql_log): + adventure, _ = make_adventure(db, settings, [("do", "> You wait."), ("ai", "Night falls.")]) + for text in ("one", "two", "three"): + add_memory(db, adventure, f"memory {text}") + sql_log.clear() + result = retrieve(adventure, settings) + assert all(m["lexical_score"] == 0.0 for m in result["used"]) + assert len(memory_text_reads(sql_log)) == 1 # the top-k detail read only + + +def test_an_edited_memory_is_matched_on_its_new_text(db, settings, played): + adventure, bank = played + assert retrieve(adventure, settings)["used"][0]["id"] == bank["sundial"].id + # An edit clears the vector (the route calls set_vector(None)); re-embedding + # sets it again. Both go through set_vector, which drops the held terms. + bank["road"].text = "The amber sundial was traded for the road toll." + memorybank.set_vector(bank["road"], None) + memorybank.set_vector(bank["road"], [1.0, 0.0, 0.0]) + bank["sundial"].text = "Mara hid a bottle in the cellar." + memorybank.set_vector(bank["sundial"], None) + memorybank.set_vector(bank["sundial"], [1.0, 0.0, 0.0]) + db.commit() + assert retrieve(adventure, settings)["used"][0]["id"] == bank["road"].id diff --git a/backend/tests/test_v11_b2_summarizer_fidelity.py b/backend/tests/test_v11_b2_summarizer_fidelity.py new file mode 100644 index 0000000..8121de4 --- /dev/null +++ b/backend/tests/test_v11_b2_summarizer_fidelity.py @@ -0,0 +1,225 @@ +"""v1.1 WP-B.2: the memory summariser, after the rejected B2.4 prompt experiment. + +B2.4 tried a memory prompt instructing the model to keep named facts and objects. +Measured against the reference model, it did not correct the creation failure it +was for, and it was not shipped (`V1.1-WP-B2-REPORT.md` §T). The shipped prompt +is v1.0.0's. + +This file keeps two kinds of test apart. + +**Acceptance tests** gate the tree: +- the shipped memory prompt is exactly v1.0.0's, so the experiment is gone; +- every fidelity fixture reaches the summariser whole, through the application's + own prompt assembly; +- a long memory is stored as the model wrote it, never cut; +- the memory the attempt-2 block should have produced ranks first under B2.1. + +**Diagnostic-measurement tests** check only that `tools/memory_fidelity.py` +measures correctly: fact retention, attribution, invention, word count, a leading +"Memory:", second person and promise retention, on hand-written memories whose +answers are known. What a real model scores on those measurements is +nondeterministic, is taken with inference, and is reported. It is never a gate +here. + + python -m pytest tests/test_v11_b2_summarizer_fidelity.py -v +""" + +import asyncio +import re +import subprocess + +import pytest + +from app import memorybank, models, tree +from app.database import Base, SessionLocal, engine +from tools import memory_diagnostic as md +from tools import memory_fidelity as mf + +# ==================================================================== acceptance + + +def test_the_shipped_memory_prompt_is_v1_0_0s(): + """The B2.4 experiment is reverted: production sends the prompt v1.0.0 and + WP-B.1 shipped, unchanged.""" + try: + source = subprocess.run(["git", "show", "beb17ad:backend/app/memorybank.py"], + capture_output=True, text=True, check=True).stdout + except (OSError, subprocess.CalledProcessError): + pytest.skip("git history not available") + block = re.search(r"^MEMORY_SYSTEM_PROMPT = \((.*?)^\)$", source, re.S | re.M).group(1) + shipped = eval(f"({block})", {"MEMORY_MAX_WORDS": 50}) # noqa: S307 - our own source + assert memorybank.MEMORY_SYSTEM_PROMPT == shipped + assert memorybank.MEMORY_MAX_WORDS == 50 + + +def test_the_rejected_experiment_is_not_what_ships(): + assert mf.B24_EXPERIMENT_PROMPT != memorybank.MEMORY_SYSTEM_PROMPT + assert "Keep each fact with the person it belongs to" not in memorybank.MEMORY_SYSTEM_PROMPT + + +@pytest.mark.parametrize("fixture", mf.FIXTURES, ids=lambda f: f.fixture_id) +def test_every_fixture_reaches_the_summariser_whole(fixture): + """Creation can only fail at the model if the fact was sent. Each fixture + fits the excerpt budget, so the whole block is the excerpt.""" + user = mf.user_prompt_for(fixture) + assert memorybank.count_tokens(fixture.raw) <= memorybank.MEMORY_EXCERPT_TOKENS + assert f"Story excerpt:\n\n{fixture.raw}\n\nMemory:" in user + assert user.startswith("Cast:\n- " + fixture.protagonist + " — the protagonist.") + + +class Scripted: + def __init__(self, reply): + self.reply = reply + self.calls: list[tuple[str, str]] = [] + + async def complete(self, system, user, *, temperature=0.3, max_tokens=400): + self.calls.append((system, user)) + return self.reply + + +@pytest.fixture() +def db(): + Base.metadata.create_all(bind=engine) + session = SessionLocal() + try: + yield session + finally: + session.close() + Base.metadata.drop_all(bind=engine) + + +def campaign(db, fixture): + user = models.User(is_guest=False, email="b2-fidelity@example.com") + db.add(user) + db.flush() + db.add(models.Settings(user_id=user.id, model="m", embedding_model="")) + adventure = models.Adventure(user_id=user.id, title="Fidelity", script_state={}, auto_summarize=True, + persona_name=fixture.protagonist) + db.add(adventure) + db.flush() + nodes = [] + for kind, text in fixture.actions + (("ai", "The story moves on."),): + action = models.Action(adventure_id=adventure.id, type=kind, text=text) + tree.place_action(db, adventure, action) + db.add(action) + db.flush() + nodes.append(action) + db.commit() + return adventure, nodes + + +def write_memory(db, adventure, monkeypatch, provider): + monkeypatch.setattr(memorybank, "MEMORY_START", 0) + monkeypatch.setattr(memorybank, "MAX_MEMORIES_PER_RUN", 1) + monkeypatch.setattr(memorybank, "summary_provider", lambda s: provider) + settings = db.query(models.Settings).first() + asyncio.run(memorybank._create_due_memories(adventure, settings, db)) + return db.query(models.Memory).filter_by(adventure_id=adventure.id).all() + + +def test_the_application_sends_the_shipped_prompt_and_the_whole_planting_block(db, monkeypatch): + fixture = mf.FIXTURES_BY_ID["regression_attempt_2"] + adventure, nodes = campaign(db, fixture) + provider = Scripted(fixture.faithful) + [memory] = write_memory(db, adventure, monkeypatch, provider) + system, user = provider.calls[0] + assert system == memorybank.MEMORY_SYSTEM_PROMPT + assert "> I watch Mara slip the amber sundial inside the cracked teapot" in user + assert (memory.source_start, memory.source_end) == (nodes[0].depth, nodes[5].depth) + + +def test_an_over_long_memory_is_stored_as_written_never_cut(db, monkeypatch): + """The word target is an instruction, not a truncation: cutting a memory + after the fact can split or drop exactly the fact it was written to keep.""" + fixture = mf.FIXTURES_BY_ID["regression_attempt_2"] + adventure, _ = campaign(db, fixture) + long_reply = fixture.faithful + " " + " ".join(["They advanced cautiously through the dark."] * 12) + [memory] = write_memory(db, adventure, monkeypatch, Scripted(long_reply)) + assert memory.text == long_reply + assert len(memory.text.split()) > 2 * memorybank.MEMORY_MAX_WORDS + + +def test_a_faithful_regression_memory_ranks_first_for_its_question(): + """If the summariser keeps the fact, B2.1 finds it: the memory the attempt-2 + block should have produced, among the memories its bank really held for that + stretch, under production scoring.""" + fixture = mf.FIXTURES_BY_ID["regression_attempt_2"] + stored, _ = fixture.unfaithful[0] + bank = { + 1: fixture.faithful, + 2: stored, + 3: "Aldric, Mara and Edrin advanced through the cold crypt, the silver key heavy in Aldric's hands.", + 4: "Aldric told Mara the silver key opens the crypt beneath the Old Abbey.", + 5: "Rain kept falling on Westhaven as the travellers walked toward the abbey grounds.", + } + embed = md.ConceptEmbedder.vector + query = {"input": "> I ask Mara quietly where she hid the amber sundial.", + "context": "Aldric and Mara in the Crooked Lantern, rain outside."} + held = {i: embed(t) for i, t in bank.items()} + terms = {i: memorybank.lexical_terms(t) for i, t in bank.items()} + rows = memorybank.score_candidates(list(bank), held, terms, embed(query["input"]), + embed(query["context"]), + sorted(memorybank.lexical_terms(query["input"]))) + assert rows[0][1] == 1 + assert rows[0][3] > 0 + + +# ======================================================= diagnostic measurements +# These prove the measuring instrument. They say nothing about any model. + + +def test_the_fixtures_cover_each_measurement_in_more_than_one_genre(): + requirements = {f.requirement for f in mf.FIXTURES} + assert {"distinctive object and place", "player-established concrete fact", "promise / commitment", + "attribution", "clutter pressure", "no invention", "multiple concrete facts", + "the actual failed-run block"} <= requirements + assert {"office", "contemporary", "science-fiction-neutral"} <= {f.genre for f in mf.FIXTURES} + + +@pytest.mark.parametrize("fixture", mf.FIXTURES, ids=lambda f: f.fixture_id) +def test_the_checker_passes_a_faithful_memory(fixture): + result = mf.evaluate(fixture, fixture.faithful) + assert result["passed"], result + assert not result["over_target"] and not result["memory_prefix"] and not result["second_person"] + + +@pytest.mark.parametrize("fixture, memory, reason", [ + (f, memory, reason) for f in mf.FIXTURES for memory, reason in f.unfaithful +], ids=lambda v: v.fixture_id if isinstance(v, mf.Fixture) else None) +def test_the_checker_fails_each_failure_shape(fixture, memory, reason): + result = mf.evaluate(fixture, memory) + assert not result["passed"], result + if reason == "not retained": + assert not result["retained"] + elif reason == "misattributed": + assert result["misattributed"] + elif reason == "invented": + assert result["inventions"] + + +def test_the_checker_reads_the_stored_attempt_2_memory_as_the_real_failure(): + fixture = mf.FIXTURES_BY_ID["regression_attempt_2"] + stored, _ = fixture.unfaithful[0] + result = mf.evaluate(fixture, stored) + assert result["retained"] is False and result["words"] == 102 and result["over_target"] + + +@pytest.mark.parametrize("memory, prefix, you", [ + ("Memory: Dana promised Marcus the lease by Friday.", True, False), + (" memory: Dana promised the lease.", True, False), + ("You thanked Marcus and left.", False, True), + ("Dana thanked Marcus; your lease is due.", False, True), + ("Dana promised Marcus she would bring the signed lease by Friday.", False, False), +]) +def test_the_checker_measures_framing(memory, prefix, you): + result = mf.evaluate(mf.FIXTURES_BY_ID["promise_contemporary"], memory) + assert result["memory_prefix"] is prefix + assert result["second_person"] is you + + +def test_the_checker_measures_promise_retention(): + fixture = mf.FIXTURES_BY_ID["promise_contemporary"] + kept = mf.evaluate(fixture, "Dana promised to bring Marcus the signed lease by Friday.") + scenery = mf.evaluate(fixture, "Memory: Dana looked around the empty living room while a dog barked.") + assert kept["facts"]["lease by Friday"]["kept"] and kept["passed"] + assert not scenery["facts"]["lease by Friday"]["kept"] and scenery["memory_prefix"] diff --git a/backend/tools/memory_diagnostic.py b/backend/tools/memory_diagnostic.py index 34b2115..09c482b 100644 --- a/backend/tools/memory_diagnostic.py +++ b/backend/tools/memory_diagnostic.py @@ -255,70 +255,101 @@ def _section(snapshot: dict, label: str) -> str: # ------------------------------------------------------------------- stages -async def rank_bank(db, adventure, settings, query: str, embed) -> dict: +async def rank_bank(db, adventure, settings, query: dict, embed) -> dict: """Production's ranking, recomputed for `query`, for every eligible memory. - The same catalogue clause, the same cosine, the same pin rule, the same - redundancy suppression helper. Returns every scored row, not just the top-k, - because "where did F rank" is the question. + `query` is a `memorybank.retrieval_query` dict. The catalogue clause, the + scoring (`memorybank.score_candidates`) and the selection with its pins and + redundancy rule (`memorybank.select_memories`) are production's own + functions, so this is production's ranking, not a second opinion. Returns + every scored row, not just the top-k, because "where did F rank" is the + question. """ catalogue = db.execute( select(models.Memory.id, models.Memory.pinned, models.Memory.authority, - models.Memory.embedding_blob).where( + models.Memory.embedding_blob, models.Memory.text).where( models.Memory.adventure_id == adventure.id, lineage.path_of(db, adventure).clause(models.Memory), models.Memory.forgotten.is_(False), models.Memory.embedded.is_(True), ) ).all() - if not catalogue or not query.strip(): - return {"query": query, "scored": [], "selected": [], "top_k": settings.memory_top_k} - [query_vec] = await embed([query]) - held = {row.id: vectors.unpack(row.embedding_blob) for row in catalogue if row.embedding_blob} - authority_of = {row.id: row.authority for row in catalogue} - scored = sorted( - ((vectors.cosine(query_vec, held[row.id]), row.id, row.pinned) - for row in catalogue if row.id in held), - key=lambda r: r[0], reverse=True, - ) top_k = max(1, settings.memory_top_k) - used = [r for r in scored if r[2]] - remaining = max(0, top_k - len(used)) - candidates = [r for r in scored if not r[2]] - kept, suppressed = memorybank._drop_redundant(candidates, held, authority_of, remaining) - selected = {r[1] for r in used + kept} + texts = [t for t in (query["input"], query["context"]) if t.strip()] + if not catalogue or not texts: + return {"query": query, "scored": [], "selected": [], "top_k": top_k} + vectors_by_text = dict(zip(texts, await embed(texts))) + input_vec = vectors_by_text.get(query["input"]) if query["input"].strip() else None + context_vec = vectors_by_text.get(query["context"]) if query["context"].strip() else None + held = {row.id: vectors.unpack(row.embedding_blob) for row in catalogue if row.embedding_blob} + terms_of = ({row.id: memorybank.lexical_terms(row.text or "") for row in catalogue} + if query["input_terms"] else {}) + authority_of = {row.id: row.authority for row in catalogue} + pinned_of = {row.id: row.pinned for row in catalogue} + scored = memorybank.score_candidates( + [row.id for row in catalogue if row.id in held], held, terms_of, + input_vec, context_vec, query["input_terms"]) + used, suppressed = memorybank.select_memories(scored, pinned_of, held, authority_of, top_k) + selected = {row[1] for row in used} suppressed_by = dict(suppressed) return { "query": query, "top_k": top_k, "scored": [ - {"rank": i + 1, "memory_id": memory_id, "similarity": round(score, 4), - "pinned": pinned, "selected": memory_id in selected, + {"rank": i + 1, "memory_id": memory_id, "similarity": round(semantic, 4), + "semantic_score": round(semantic, 4), "lexical_score": round(lexical, 4), + "final_score": round(final, 4), + "pinned": pinned_of[memory_id], "selected": memory_id in selected, "suppressed_as_duplicate_of": suppressed_by.get(memory_id)} - for i, (score, memory_id, pinned) in enumerate(scored) + for i, (final, memory_id, semantic, lexical) in enumerate(scored) ], "selected": sorted(selected), } -def production_query(adventure, exclude_action_id: int | None) -> str: - """The retrieval query a turn used: its newest actions, as `retrieve_memories` builds it.""" - recent = history.tail(adventure, memorybank.RETRIEVAL_WINDOW_ACTIONS, exclude_action_id) - return builder.truncate_to_last_tokens( - "\n\n".join(a.text for a in recent), memorybank.RETRIEVAL_WINDOW_TOKENS) +def production_query(adventure, exclude_action_id: int | None) -> dict: + """The retrieval query a turn used, built by production's own `retrieval_query`.""" + return memorybank.retrieval_query(adventure, exclude_action_id) + + +def variant_query(base: dict, player_input: str) -> dict: + """`base` with a different player input: "what if the player had asked this + here", with the scene and narration context the recall turn really had.""" + return {"input": player_input, "context": base["context"], + "input_terms": sorted(memorybank.lexical_terms(player_input))} def eviction_order(db, adventure) -> list[int]: - """The order `_evict_over_capacity` would take unpinned active memories in.""" - from sqlalchemy import func - return db.execute( - select(models.Memory.id).where( + """The order `_evict_over_capacity` would take unpinned active memories in. + + Production's own `memorybank.eviction_order`, run to the end of the bank.""" + rows = db.execute( + select(models.Memory.id, models.Memory.pinned, models.Memory.source_start, + models.Memory.source_end, models.Memory.last_used_at, + models.Memory.created_at, models.Memory.use_count).where( models.Memory.adventure_id == adventure.id, models.Memory.forgotten.is_(False), - models.Memory.pinned.is_(False), - ).order_by(func.coalesce(models.Memory.last_used_at, models.Memory.created_at), - models.Memory.use_count) - ).scalars().all() + ) + ).all() + return memorybank.eviction_order(rows, len(rows)) + + +def coverage(ranges: list[tuple[int, int]], tip: int | None) -> dict: + """How much of the story `ranges` (active memories' source ranges) describe. + + `largest_gap` is the longest run of depths, between the first memory's start + and `tip`, that no memory covers. It is how the eviction rule is judged in + general, not only for the planted fact.""" + if not ranges: + return {"first_start": None, "last_end": None, "largest_gap": None} + ordered = sorted(ranges) + gaps = [] + reach = ordered[0][1] + for start, end in ordered[1:]: + gaps.append(max(0, start - reach - 1)) + reach = max(reach, end) + return {"first_start": ordered[0][0], "last_end": reach, + "largest_gap": max(gaps, default=0)} async def diagnose(db, adventure, settings, fact: Fact, plant_depth: int, *, @@ -342,7 +373,7 @@ async def diagnose(db, adventure, settings, fact: Fact, plant_depth: int, *, for memory in covering: block = memorybank.source_block(db, memory) raw = "\n\n".join(a.text for a in block) - excerpt = builder.truncate_to_last_tokens(raw, memorybank.MEMORY_EXCERPT_TOKENS) + excerpt = memorybank.memory_excerpt(raw) # what `summarize_block` sends creation_input.append({ "memory_id": memory.id, "source_start": memory.source_start, "source_end": memory.source_end, "block_tokens": builder.count_tokens(raw), @@ -394,6 +425,8 @@ async def diagnose(db, adventure, settings, fact: Fact, plant_depth: int, *, out["verdict"] = "created_but_evicted" return out + # The recall turn's AI node is excluded, so the newest action is the recall + # player action, exactly as the turn saw it before it wrote its reply. query = production_query(adventure, recall_action.id) ranking = await rank_bank(db, adventure, settings, query, embed) row = next((r for r in ranking["scored"] if r["memory_id"] == memory.id), None) @@ -401,9 +434,10 @@ async def diagnose(db, adventure, settings, fact: Fact, plant_depth: int, *, out["ranked"] = { "yes": row is not None and row["rank"] <= ranking["top_k"], "eligible": row is not None, - "lexical_score": None, # memory ranking has no lexical term (CONTEXT-AND-MEMORY §20) - "semantic_score": row["similarity"] if row else None, - "final_score": row["similarity"] if row else None, + "lexical_score": row["lexical_score"] if row else None, + "semantic_score": row["semantic_score"] if row else None, + "final_score": row["final_score"] if row else None, + "selected_top_k": ranking["selected"], "pin_effect": "always selected" if memory.pinned else "none", "rank": row["rank"] if row else None, "of": len(ranking["scored"]), @@ -512,6 +546,9 @@ PLACES = ( ) PARAPHRASE_QUERY = "I ask Mara where she tucked the little brass dial that tells the hour." UNRELATED_QUERY = "I ask the ferryman what rope costs at the landing this season." +#: Built only from words every fixture memory holds ("travellers", "spent", +#: "time"), so its rarity weight is zero everywhere. +COMMON_WORDS_QUERY = "The travellers spent time." def filler_prose(index: int, words: int) -> str: @@ -539,6 +576,10 @@ class Scenario: pin_first_memory: bool = False lineage_control: bool = False diagnose_recall: bool = True + #: The narration of the last turn before recall, when a fixture needs the + #: scene to say something (WP-B.2's context-dependent question). Must not + #: name a planted fact. + pre_recall_reply: str = "" SCENARIOS = { @@ -553,6 +594,23 @@ SCENARIOS = { "long_block_fact_late": Scenario("long_block_fact_late", turns=10, prose_words=850, plant_turn=3, budget=16384), "lineage_control": Scenario("lineage_control", lineage_control=True), + # v1.1 WP-B.2: the ranking failure B.1 saw on the real model, made + # deterministic. Longer narration fills the v1.0.0 query, and `memory_top_k` + # is the real run's 4. Below capacity, isolation valid. + "ranking_crowded": Scenario("ranking_crowded", prose_words=150, top_k=4), + # v1.1 WP-B.2: all three B.1 failures at once. Every block is longer than the + # summariser's excerpt, narration crowds the query at `memory_top_k` 4, and + # the bank passes a capacity of 8 long before recall at depth 106. + "independent_full": Scenario("independent_full", prose_words=850, top_k=4, + capacity=8, budget=16384), + # v1.1 WP-B.2: a question that names neither the sundial nor the teapot and + # cannot be answered without the scene. The last narration puts Mara at the + # tavern's top shelf; the player asks "her" what she put "up there". + "ranking_context_dependent": Scenario( + "ranking_context_dependent", prose_words=150, top_k=4, + recall_text="I ask her what she keeps up there.", + pre_recall_reply=("Mara stands on a stool at the tavern's top shelf, running a cloth " + "around the old kettle up there, and she will not meet your eye.")), } @@ -662,6 +720,16 @@ def run_scenario(scenario: Scenario) -> dict: models.Memory.created_at).where(models.Memory.adventure_id == adv) .order_by(models.Memory.id)).all()] + def per_turn_isolation(adventure_id): + with SessionLocal() as db: + adventure = db.get(models.Adventure, adventure_id) + active = summaries.current(db, adventure) + return { + "f_in_state": FACT_F.mentioned_by(json.dumps(adventure.narrative_state or {}, + default=str)), + "f_in_summary": FACT_F.mentioned_by(active.text if active is not None else ""), + } + def turn(kind, text, reply): ScriptNarrator.next_reply = f"{reply}\n```state\n{{\"events\": []}}\n```" response = client.post(f"/api/adventures/{adv}/actions", json={"type": kind, "text": text}) @@ -703,6 +771,8 @@ def run_scenario(scenario: Scenario) -> dict: turn("do", f"I turn away from the mill and walk to the {PLACES[n % len(PLACES)]}.", filler_prose(n + 100, scenario.prose_words)) g["diverged_at_turn"] = n + elif n == scenario.turns and scenario.pre_recall_reply: + turn("do", "I head back to the tavern.", scenario.pre_recall_reply) else: turn("do", f"I walk on to the {PLACES[n % len(PLACES)]}.", filler_prose(n, scenario.prose_words)) @@ -735,6 +805,11 @@ def run_scenario(scenario: Scenario) -> dict: "f_memory_id": f_memory_id, "f_forgotten": bool(f_row and f_row["forgotten"]), "f_use_count": f_row["use_count"] if f_row else None, + "coverage": coverage([(r["source_start"], r["source_end"]) for r in rows + if not r["forgotten"] and r["source_start"] is not None], + None), + # Isolation on every turn, not only at recall (WP-B.2 acceptance). + **per_turn_isolation(adv), }) turn("do", scenario.recall_text, filler_prose(999, scenario.prose_words)) @@ -757,18 +832,64 @@ def run_scenario(scenario: Scenario) -> dict: db, adventure, settings, FACT_F, plant_depth, recall_action=recall_action, embed=embedder.embed)) result["summariser_excerpts"] = len(summariser.excerpts) + # Provenance as the recall turn recorded it, resolved back to rows. + used = (recall_action.context_snapshot.get("memories") or {}).get("used") or [] + f_entry = next((m for m in used + if m.get("id") == result["diagnosis"]["created"]["memory_id"]), None) + f_memory = db.get(models.Memory, f_entry["id"]) if f_entry else None + block = memorybank.source_block(db, f_memory) if f_memory is not None else [] + result["provenance"] = { + "recorded": f_entry and {k: f_entry.get(k) for k in + ("id", "source", "semantic_score", "lexical_score", + "final_score", "authority")}, + "range_covers_plant": bool(f_entry and f_entry["source"]["source_start"] + <= plant_depth <= f_entry["source"]["source_end"]), + "matches_row": bool(f_memory is not None and f_entry["source"] == { + "branch_id": f_memory.branch_id, "depth": f_memory.depth, + "source_start": f_memory.source_start, "source_end": f_memory.source_end}), + "source_block_depths": [a.depth for a in block], + "source_block_holds_planting": any(a.text == FACT_F.sentence for a in block), + } memory_id = result["diagnosis"]["created"]["memory_id"] if memory_id is not None and not result["diagnosis"]["retained"]["forgotten"]: variants = {} - for label, query in (("direct", scenario.recall_text), - ("paraphrase", PARAPHRASE_QUERY), - ("unrelated", UNRELATED_QUERY)): + base = production_query(adventure, recall_action.id) + # WP-B.2's negative control needs a decoy: another memory that + # holds a word the question adds, and nothing about F. + decoy = next(((m.id, found.group(1)) for m in db.execute( + select(models.Memory).where(models.Memory.adventure_id == adventure.id, + models.Memory.forgotten.is_(False), + models.Memory.id != memory_id) + .order_by(models.Memory.id)).scalars() + if (found := re.search(r"at the ([a-z]+ [a-z]+)\.", m.text or ""))), None) + queries = [ + ("direct", variant_query(base, scenario.recall_text)), + ("paraphrase", variant_query(base, PARAPHRASE_QUERY)), + ("unrelated", variant_query(base, UNRELATED_QUERY)), + # The player's words with the context taken away. + ("input_only", {**variant_query(base, scenario.recall_text), "context": ""}), + # Words every memory in these fixtures holds, and nothing else. + ("common_words", variant_query(base, COMMON_WORDS_QUERY)), + ] + if decoy is not None: + queries.append(("rare_word_with_paraphrase", variant_query( + base, PARAPHRASE_QUERY[:-1] + f", out by the {decoy[1]}."))) + for label, query in queries: ranking = asyncio.run(rank_bank(db, adventure, settings, query, embedder.embed)) row = next((r for r in ranking["scored"] if r["memory_id"] == memory_id), None) - variants[label] = {"query": query, "rank": row and row["rank"], + decoy_row = next((r for r in ranking["scored"] + if decoy is not None and r["memory_id"] == decoy[0]), None) + text = query["input"] + variants[label] = {"query": text, "rank": row and row["rank"], + "selected_count": len(ranking["selected"]), + "decoy_memory_id": decoy and decoy[0], + "decoy_rank": decoy_row and decoy_row["rank"], + "decoy_lexical_score": decoy_row and decoy_row["lexical_score"], "of": len(ranking["scored"]), "similarity": row and row["similarity"], + "lexical_score": row and row["lexical_score"], + "final_score": row and row["final_score"], "selected": bool(row and row["selected"]), "top_k": ranking["top_k"]} result["ranking_variants"] = variants diff --git a/backend/tools/memory_fidelity.py b/backend/tools/memory_fidelity.py new file mode 100644 index 0000000..badd2e3 --- /dev/null +++ b/backend/tools/memory_fidelity.py @@ -0,0 +1,551 @@ +"""v1.1 WP-B.2: how faithfully a real model's memories keep the facts of their block. + +**Diagnostic only. Nothing here is imported by the application, and nothing here +is a release gate.** + +Why it exists: the first isolation-valid real-model WP-B run failed at creation. +The whole planting block reached the summariser, and the memory it wrote left the +planted fact out (`V1.1-WP-B2-REPORT.md` §L.2). B2.4 tried the plan's bounded +remedy, a memory prompt instructing the model to keep named facts and objects. It +was measured with this module, did not correct the failure, and was **not +shipped** (§T). The module stays so the limitation can be measured again, on this +model or a different one. + +- **Fixtures.** Short story blocks, each built around a fact a later scene could + turn on, with the ordinary texture a real block carries around it. They are + genre-neutral (office, contemporary, a science-fiction-neutral station), plus + the attempt-2 planting block itself. Each names the facts a memory must keep, + whom each belongs to, and what it must not invent. +- **The checker** (`evaluate`) is deterministic and reads only the memory text. + It is a heuristic, and says so: + - a fact counts as kept when one sentence names every part of it; + - attribution is the nearest named character before the fact's verb; + - it also reports word count, a leading "Memory:" and second-person "you". +- **The comparison** (`compare`) sends each fixture to a real model through the + application's own provider and `memorybank.memory_user_prompt`. It scores + memories under the shipped prompt and under the rejected B2.4 experiment. + +A scripted summariser cannot show what a prompt makes a model do. So the +deterministic tests prove only that the checker is right and the fixtures reach +the summariser; model quality is measured here, with inference, and reported. + + # the real-model measurement (inference: ask first) + .venv/bin/python -m tools.memory_fidelity --endpoint \\ + --model qwen2.5:3b-instruct-16k --samples 5 --out "$HOME/v11-evidence/