Mark the story with a node, not with a count

The memory bank and the story summary each kept a cursor: how many story
actions they had already covered. A count is a position in a list, and this
list moves — delete an action in front of the mark and every later one slides
down a slot, so the mark now covers one it has never read. All the cursor
bookkeeping existed to patch that up.

Both marks are now (branch_id, depth): the node up to and including which the
work is done. A depth is a coordinate along a path, not an offset into a list,
so nothing in front of it can move it. That deletes rather than rewrites
`position_of_index`, `note_action_removed`, `_rewind_cursors_to_index`,
`prune_dangling_memories` and the every-pass clamp in `run_post_turn`.

A memory hangs off the node its block ends on, so a fork inherits its
ancestors' memories without copying any, and retrieval selects through the
branch clause over the *whole* lineage — recall is long-range by definition and
cannot be windowed. Measured: 1,807 B on a story forked twenty times against
1,823 B on a flat one of the same length.

Migrations 53-56 translate the old counts into nodes. They rewrite `adventures`
and not `actions`, so this one needs no VACUUM FULL.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7
This commit is contained in:
parththakkar106
2026-08-18 19:14:07 +05:30
committed by Parth
co-authored by Claude Opus 5
parent c7b6a46a8a
commit c51531709d
16 changed files with 1334 additions and 260 deletions
+30 -14
View File
@@ -190,7 +190,7 @@ def test_window_is_ordered_and_free_of_duplicates(story):
assert ids == sorted(ids), "window must be oldest-first"
# --------------------------------------------- the cursor arithmetic agrees
# ------------------------------------------- the node-anchored reads agree
def test_helpers_agree_with_the_full_list(story):
db, adventure, settings = story
@@ -204,30 +204,46 @@ def test_helpers_agree_with_the_full_list(story):
assert [a.id for a in history.tail_range(adventure, 5, 3)] == \
[a.id for a in actions[-8:-5]]
assert memorybank.settled_count(adventure) == len(actions) - 1
assert history.newest_settled(adventure).id == actions[-2].id
for probe in (0, 1, ACTION_COUNT // 2, ACTION_COUNT - 1):
target = actions[probe]
expected = next(i for i, a in enumerate(actions) if a.index >= target.index)
assert history.position_of_index(adventure, target.index) == expected
boundary = actions[probe].depth
assert history.count_after(adventure, boundary) == ACTION_COUNT - probe - 1
assert [a.id for a in history.after(adventure, boundary, 3)] == \
[a.id for a in actions[probe + 1:probe + 4]]
def test_positions_still_line_up_after_a_middle_action_is_deleted(story):
"""The gap in Action.index is exactly what makes positions and indexes
diverge — the case that has broken the cursors twice before."""
def test_a_depth_boundary_survives_a_middle_action_being_deleted(story):
"""The case that has broken the cursors twice before, and the reason they
are depths now.
A *position* answers "how much story is past this point?" by counting from
the start, so deleting anything in front of the mark changes which action
the mark names. A depth names the same node either way — the only thing
that changes is the count of what comes after, which is what did change.
"""
db, adventure, settings = story
actions = history.story_actions(adventure)
victim = actions[50]
mark = actions[30].depth
before = history.count_after(adventure, mark)
next_three = [a.id for a in history.after(adventure, mark, 3)]
victim = actions[10] # in front of the mark
db.delete(victim)
db.commit()
db.expire(adventure)
remaining = history.story_actions(adventure)
assert len(remaining) == ACTION_COUNT - 1
assert history.count(adventure) == ACTION_COUNT - 1
for probe in (0, 49, 50, 51, ACTION_COUNT - 2):
target = remaining[probe]
expected = next(i for i, a in enumerate(remaining) if a.index >= target.index)
assert history.position_of_index(adventure, target.index) == expected, probe
assert history.count_after(adventure, mark) == before, "the mark moved"
assert [a.id for a in history.after(adventure, mark, 3)] == next_three
# ...and deleting something *after* it is the one thing that does change
# the count, because that is a fact about the story rather than about the
# coordinate system.
db.delete(history.after(adventure, mark, 1)[0])
db.commit()
db.expire(adventure)
assert history.count_after(adventure, mark) == before - 1
def test_blank_actions_are_excluded_the_same_way_in_sql_and_python(story):
+416
View File
@@ -0,0 +1,416 @@
"""Phase 14 SP3 — memories hang off nodes, and the marks are nodes too.
Two claims, and neither of them fails loudly if it is wrong:
* **A memory belongs to the path that produced it.** A memory made on branch B
must be invisible from A, and the memories of a shared ancestor must be
visible from both — without anything being copied when a fork happens. The
failure mode is a prompt quietly carrying a summary of a story the player
abandoned.
* **Retrieval reads the *whole* lineage, and that stays affordable.** The story
is read through a window, but recall is long-range by definition and cannot
be — so the clause names every ancestor, and the bet is that memories are
sparse enough (one per six actions) for that to be tens of small rows even
twenty forks deep. Measured below rather than asserted.
Nothing in the product forks yet, so the fork is built by hand, exactly as
`test_branch_clause.py` builds it.
python -m pytest tests/test_memory_nodes.py -v
"""
import os
import tempfile
_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False)
_tmp.close()
os.environ["AIDND_DB_PATH"] = _tmp.name
os.environ.pop("AIDND_DATABASE_URL", None)
os.environ.pop("DATABASE_URL", None)
import asyncio
import pytest
from app import memorybank, models, tree
from app.context import cursors, lineage
from app.database import Base, SessionLocal, engine
from tools import dbmeter
class StubEmbedder:
"""Returns whatever vector the test set, for any text."""
def __init__(self, vector=(1.0, 0.0, 0.0)):
self.vector = list(vector)
async def embed(self, texts):
return [list(self.vector) for _ in texts]
# --------------------------------------------------------------- the fixture
def make_branch(db, adventure, parent=None, fork_depth=None):
"""A branch row whose lineage is its parent's, capped, plus itself — the
computation SP5 will do at fork time, written out so the fixture cannot
pass by agreeing with a bug in the code under test."""
branch = models.Branch(
adventure_id=adventure.id,
parent_branch_id=parent.id if parent else None,
fork_depth=fork_depth,
lineage=[],
)
db.add(branch)
db.flush()
inherited = []
if parent is not None:
for ancestor_id, cap in lineage.entries_of(parent):
capped = fork_depth if cap is None else min(cap, fork_depth)
inherited.append([ancestor_id, capped])
branch.lineage = [[branch.id, None]] + inherited
db.flush()
return branch
def add_node(db, adventure, branch, depth, label, index=None):
action = models.Action(
adventure_id=adventure.id,
index=depth if index is None else index,
branch_id=branch.id,
depth=depth,
type="ai" if depth % 2 else "do",
text=f"{label}{depth}",
)
db.add(action)
return action
def add_memory(db, adventure, text, node, vector=(1.0, 0.0, 0.0), **kwargs):
"""A memory of the block ending on `node`, attached the way the post-turn
pass attaches one."""
memory = models.Memory(
adventure_id=adventure.id, text=text,
source_start=None if node is None else node.depth,
source_end=None if node is None else node.depth,
**kwargs,
)
if node is not None:
tree.attach_memory(memory, node)
else:
tree.place_memory(db, adventure, memory)
db.add(memory)
db.flush()
memorybank.set_vector(memory, list(vector))
db.commit()
return memory
@pytest.fixture()
def forked():
"""A0..A3, then B4 B5 off A3, then C6 C7 off B5 — with a memory hung off
one node of each branch, and A playing on past the fork it was left at.
The head is C, so the story is A0 A1 A2 A3 B4 B5 C6 C7 and the memories in
play are A's and B's and C's — but not the one on A5, which is on a sibling
of B4 and belongs to a story nobody is reading.
"""
Base.metadata.create_all(bind=engine)
db = SessionLocal()
user = models.User(is_guest=False, email="nodes@example.com")
db.add(user)
db.flush()
settings = models.Settings(
user_id=user.id, api_key="enc:dummy", model="m",
embedding_model="text-embedding-3-small", memory_top_k=10,
memory_bank_capacity=80,
)
db.add(settings)
adventure = models.Adventure(
user_id=user.id, title="Forked", script_state={}, memory_bank_enabled=True,
auto_summarize=True,
)
db.add(adventure)
db.flush()
a = make_branch(db, adventure)
b = make_branch(db, adventure, parent=a, fork_depth=3)
c = make_branch(db, adventure, parent=b, fork_depth=5)
nodes = {}
for depth in range(4):
nodes[f"A{depth}"] = add_node(db, adventure, a, depth, "A")
for depth in (4, 5): # A kept playing: siblings of B4/B5
nodes[f"A{depth}"] = add_node(db, adventure, a, depth, "A", index=100 + depth)
for depth in (4, 5):
nodes[f"B{depth}"] = add_node(db, adventure, b, depth, "B")
for depth in (6, 7):
nodes[f"C{depth}"] = add_node(db, adventure, c, depth, "C")
db.flush()
memories = {
"shared": add_memory(db, adventure, "on the shared trunk", nodes["A3"]),
"sibling": add_memory(db, adventure, "on A's own continuation", nodes["A5"]),
"b": add_memory(db, adventure, "on B", nodes["B5"]),
"c": add_memory(db, adventure, "on C", nodes["C7"]),
}
adventure.head_branch_id = c.id
adventure.head_depth = 7
db.commit()
ids = {"a": a.id, "b": b.id, "c": c.id, "nodes": nodes, "memories": memories}
try:
yield db, adventure, settings, ids
finally:
db.close()
Base.metadata.drop_all(bind=engine)
def switch_to(db, adventure, branch_id, tip):
adventure.head_branch_id = branch_id
adventure.head_depth = tip
db.commit()
def retrieved(adventure, settings) -> set[str]:
memorybank.embedding_provider = lambda s: StubEmbedder()
result = asyncio.run(
memorybank.retrieve_memories(adventure, settings, update_stats=False)
)
assert result["error"] is None, result["error"]
return {m["text"] for m in result["used"]}
# ------------------------------------------------------------- the isolation
def test_a_memory_on_a_sibling_is_not_retrieved(forked):
"""The whole point. A5 is a node of the story that was abandoned when B
forked, and the memory hanging off it must not reach a prompt on C."""
db, adventure, settings, ids = forked
assert retrieved(adventure, settings) == {
"on the shared trunk", "on B", "on C"
}
def test_a_shared_ancestor_is_visible_from_both_branches(forked):
"""Nothing is copied at a fork, so the trunk's memories are shared by
construction rather than by duplication."""
db, adventure, settings, ids = forked
switch_to(db, adventure, ids["a"], 5)
from_a = retrieved(adventure, settings)
assert "on the shared trunk" in from_a
# ...and from A, the branches taken off it are the ones out of reach.
assert from_a == {"on the shared trunk", "on A's own continuation"}
def test_the_lineage_is_read_whole_not_windowed(forked):
"""The story is read through a window; recall is not. The trunk memory is
four nodes and two forks back, and is still a candidate."""
db, adventure, settings, ids = forked
path = lineage.path_of(db, adventure)
assert len(path) == 3
# The window a *story* read would use here names one entry. Retrieval names
# all three, which is the difference this test exists to pin.
assert path.prefix_covering(2) == 1
assert "on the shared trunk" in retrieved(adventure, settings)
def test_a_hand_written_memory_is_not_lost_at_the_first_fork(forked):
"""A memory nobody derived summarises no node, so it has a branch but no
depth. A capped `depth <= fork` would drop it the moment its branch stopped
being the newest entry — a memory vanishing some turns after it was typed,
which is exactly the kind of thing nothing reports."""
db, adventure, settings, ids = forked
switch_to(db, adventure, ids["a"], 5)
typed = add_memory(db, adventure, "typed by hand", None)
assert (typed.branch_id, typed.depth) == (ids["a"], None)
switch_to(db, adventure, ids["c"], 7) # fork away from where it was written
assert "typed by hand" in retrieved(adventure, settings)
# ------------------------------------------------------------------ the marks
def test_a_mark_moves_to_the_node_the_memory_covers(forked):
"""The mark and the memory are one statement about where the pass got to,
so they are written from the same row."""
db, adventure, settings, ids = forked
cursors.MEMORY.anchor_at(adventure, ids["nodes"]["B5"])
db.commit()
assert cursors.MEMORY.stored(adventure) == (ids["b"], 5)
assert cursors.MEMORY.depth(db, adventure) == 5
def test_a_mark_from_a_sibling_reads_as_nothing_covered(forked):
"""A mark is a node, so moving to another story has to be answered rather
than assumed. Ground this path never travelled is not covered ground, and
the fallback for 'I don't know' has to be redoing the work, not skipping
it."""
db, adventure, settings, ids = forked
cursors.MEMORY.anchor_at(adventure, ids["nodes"]["C7"])
db.commit()
switch_to(db, adventure, ids["a"], 5)
assert cursors.MEMORY.depth(db, adventure) == cursors.NO_DEPTH
def test_a_mark_on_an_ancestor_is_capped_at_the_fork(forked):
"""A6 and A7 are past where this path left A, so a mark deeper than the
fork cannot mean 'covered' for anything on this story."""
db, adventure, settings, ids = forked
cursors.MEMORY.anchor_at(adventure, ids["nodes"]["A5"])
db.commit()
assert cursors.MEMORY.depth(db, adventure) == 3 # C forks off B forks off A@3
def test_a_mark_never_moves_forward_on_a_rewind(forked):
db, adventure, settings, ids = forked
cursors.MEMORY.anchor_at(adventure, ids["nodes"]["A3"])
cursors.rewind_all(adventure, ids["c"], 6)
assert cursors.MEMORY.stored(adventure) == (ids["a"], 3)
# ---------------------------------------------------- what the passes read
def test_the_summary_folds_in_only_the_path_it_is_on(forked, monkeypatch):
"""`_update_story_summary` gathers the memories past its mark. On C that is
B's and C's — never the one on A's own continuation, whose depth would
otherwise put it squarely inside the range."""
db, adventure, settings, ids = forked
monkeypatch.setattr(memorybank, "SUMMARY_INTERVAL", 1)
class Stub:
def __init__(self):
self.prompts = []
async def complete(self, system, user, **kwargs):
self.prompts.append(user)
return "A summary."
stub = Stub()
monkeypatch.setattr(memorybank, "summary_provider", lambda s: stub)
cursors.SUMMARY.anchor_at(adventure, ids["nodes"]["A3"])
db.commit()
asyncio.run(memorybank._update_story_summary(adventure, settings, db))
[prompt] = stub.prompts
assert "on B" in prompt and "on C" in prompt
assert "on A's own continuation" not in prompt
assert "on the shared trunk" not in prompt # behind the mark
# Caught up to the settled end of the story: C7 is retryable, C6 is not.
assert cursors.SUMMARY.stored(adventure) == (ids["c"], 6)
def test_a_block_is_summarized_from_the_path_and_hung_off_its_last_node(
forked, monkeypatch
):
db, adventure, settings, ids = forked
monkeypatch.setattr(memorybank, "MEMORY_START", 0)
monkeypatch.setattr(memorybank, "MEMORY_INTERVAL", 4)
class Stub:
def __init__(self):
self.excerpts = []
async def complete(self, system, user, **kwargs):
self.excerpts.append(user)
return f"Memory {len(self.excerpts)}."
stub = Stub()
monkeypatch.setattr(memorybank, "summary_provider", lambda s: stub)
asyncio.run(memorybank._create_due_memories(adventure, settings, db))
# Two blocks of four from a path of eight, minus the held-back newest: one.
[excerpt] = stub.excerpts
assert "A5" not in excerpt, "a sibling's narration reached the summarizer"
assert ["A0", "A1", "A2", "A3"] == [line for line in excerpt.split() if line[0] in "ABC"]
made = db.query(models.Memory).filter_by(text="Memory 1.").one()
assert (made.branch_id, made.depth) == (ids["a"], 3)
assert cursors.MEMORY.stored(adventure) == (ids["a"], 3)
# ------------------------------------------------------ the cost of forking
@pytest.fixture()
def deeply_forked():
"""A story forked twenty times, with a memory every six actions — the
density the post-turn pass actually produces."""
Base.metadata.create_all(bind=engine)
db = SessionLocal()
user = models.User(is_guest=False, email="deepmem@example.com")
db.add(user)
db.flush()
db.add(models.Settings(
user_id=user.id, api_key="enc:dummy", model="m",
embedding_model="text-embedding-3-small",
# Every candidate is injected, so the measurement covers fetching the
# texts too and not only ranking them.
memory_top_k=50,
))
def story(title, forks):
adventure = models.Adventure(
user_id=user.id, title=title, script_state={}, memory_bank_enabled=True,
)
db.add(adventure)
db.flush()
branch = make_branch(db, adventure)
depth = 0
nodes = []
for _ in range(4):
nodes.append(add_node(db, adventure, branch, depth, "n"))
depth += 1
for _ in range(forks):
branch = make_branch(db, adventure, parent=branch, fork_depth=depth - 1)
for _ in range(2):
nodes.append(add_node(db, adventure, branch, depth, "n"))
depth += 1
if forks:
branch = make_branch(db, adventure, parent=branch, fork_depth=depth - 1)
for _ in range(84 - depth):
nodes.append(add_node(db, adventure, branch, depth, "n"))
depth += 1
db.flush()
for node in nodes[5::6]: # one memory per six actions, as the pass makes them
add_memory(db, adventure, f"memory at {node.depth}", node)
adventure.head_branch_id = branch.id
adventure.head_depth = depth - 1
return adventure
forked_story = story("Forked", 20)
flat_story = story("Flat", 0)
db.commit()
try:
yield db, flat_story, forked_story
finally:
db.close()
Base.metadata.drop_all(bind=engine)
def test_retrieving_from_a_deep_fork_costs_what_a_flat_story_costs(deeply_forked):
"""The bet, in bytes. Retrieval names all twenty-two branches instead of
one — but it is fetching an id and a flag per memory, and there are the
same fourteen either way, so the clause is where the difference is and the
clause is not what crosses the wire."""
db, flat_story, forked_story = deeply_forked
settings = db.query(models.Settings).one()
flat_id, forked_id = flat_story.id, forked_story.id
db.commit()
db.expire_all()
meter = dbmeter.Meter()
meter.attach(engine)
try:
with meter.scope("flat"):
assert len(retrieved(db.get(models.Adventure, flat_id), settings)) == 14
flat_bytes = meter.scopes[-1].total.fetched
with meter.scope("forked"):
assert len(retrieved(db.get(models.Adventure, forked_id), settings)) == 14
forked_bytes = meter.scopes[-1].total.fetched
finally:
meter.detach()
# Measured 2026-08-18: 1,807 B against 1,823 B — the same fourteen rows,
# named through twenty-two branch terms instead of one.
assert flat_bytes > 0, "the meter saw nothing; it is measuring the wrong connection"
assert forked_bytes < flat_bytes * 1.5, (
f"retrieval on a 20-fork story cost {forked_bytes:,} B against the "
f"{flat_bytes:,} B a flat story of the same length cost"
)
+116 -65
View File
@@ -1,10 +1,19 @@
"""Memories must never describe an attempt the player can still retry away.
"""Memories must never describe an attempt the player can still retry away,
and must never skip a stretch of story.
Only the last action is retryable, so summarization holds the newest action
back one turn (memorybank.settled_story_actions). Without that, a memory could
cover the just-generated AI turn; retrying it rewrites Action.text but the
memory cursor has already advanced, so the memory is never regenerated and goes
on describing narration that is no longer in the story.
cover the just-generated AI turn; retrying it rewrites Action.text but the mark
has already moved past it, so the memory is never regenerated and goes on
describing narration that is no longer in the story.
Phase 14 SP3 changed what that mark *is*. It used to be a count of covered
story actions, and the second half of this file is the price of that: deleting
an action from in front of a position slid a never-summarized action into the
covered range, so every delete had to slide the cursors too. The mark is a node
now — `(branch_id, depth)` — and a node does not move when something in front
of it is deleted, so those tests assert that nothing happens where they used to
assert that the right correction happened.
python -m pytest tests/test_memory_settling.py -v
"""
@@ -20,7 +29,8 @@ os.environ.pop("DATABASE_URL", None)
import pytest
from app import memorybank, models
from app import memorybank, models, tree
from app.context import cursors
from app.database import Base, SessionLocal, engine
@@ -68,6 +78,26 @@ def make_adventure(db, action_count: int) -> models.Adventure:
return adventure
def cover(db, adventure, position: int) -> None:
"""Mark the first `position` story actions as already summarized.
Written as a position and translated to the node it names, because that is
what every adventure in the database looked like before SP3 and what a v1
bundle still carries. `memory_cursor` keeps the old number so the two
coordinate systems can be compared where a test cares.
"""
adventure.memory_cursor = position
adventure.summary_cursor = position
cursors.anchor_at_position(adventure, cursors.MEMORY, position)
cursors.anchor_at_position(adventure, cursors.SUMMARY, position)
db.commit()
def covered_depth(db, adventure) -> int:
"""The memory mark, as a depth on the story being played."""
return cursors.MEMORY.depth(db, adventure)
def run_memories(db, adventure, stub, monkeypatch):
monkeypatch.setattr(memorybank, "summary_provider", lambda s: stub)
settings = db.query(models.Settings).first()
@@ -99,25 +129,23 @@ def test_settled_actions_on_a_one_action_story(db):
# ------------------------------------------------------- the bug this prevents
def test_memory_never_covers_the_newest_retryable_action(db, monkeypatch):
"""cursor=6 with 12 actions is exactly the case that used to bite: the
6-action block ends on the newest action, which is still retryable."""
"""Covered up to action 5 with 12 actions is exactly the case that used to
bite: the 6-action block ends on the newest action, still retryable."""
adventure = make_adventure(db, 12)
adventure.memory_cursor = 6
db.commit()
cover(db, adventure, 6)
stub = StubSummarizer()
run_memories(db, adventure, stub, monkeypatch)
assert stub.excerpts == [] # only 11 settled — one short of a block
assert db.query(models.Memory).count() == 0
assert adventure.memory_cursor == 6
assert covered_depth(db, adventure) == 5
def test_the_block_lands_a_turn_later_without_the_newest_action(db, monkeypatch):
"""One more action and the same block is summarized — minus the new one."""
adventure = make_adventure(db, 13)
adventure.memory_cursor = 6
db.commit()
cover(db, adventure, 6)
stub = StubSummarizer()
run_memories(db, adventure, stub, monkeypatch)
@@ -128,7 +156,10 @@ def test_the_block_lands_a_turn_later_without_the_newest_action(db, monkeypatch)
assert "Action 12." not in excerpt # the newest, still retryable
memory = db.query(models.Memory).one()
assert (memory.source_start, memory.source_end) == (6, 11)
assert adventure.memory_cursor == 12
# The mark and the memory name the same node — that is what keeps them from
# drifting apart however gappy the depths underneath are.
assert (memory.branch_id, memory.depth) == cursors.MEMORY.stored(adventure)
assert covered_depth(db, adventure) == 11
def test_first_memory_waits_one_action_past_memory_start(db, monkeypatch):
@@ -150,21 +181,20 @@ def test_first_memory_waits_one_action_past_memory_start(db, monkeypatch):
def test_legacy_caught_up_adventure_is_not_rewound(db, monkeypatch):
"""An adventure summarized under the OLD rule can have memory_cursor equal
to its action count. The run_post_turn clamp must use the FULL count, not
the settled one — clamping to settled would rewind the cursor a step and
re-cover an already-summarized action in the next block."""
"""An adventure summarized under the OLD rule carries a cursor equal to its
action count — one past the settled end. That used to need a clamp on every
post-turn pass, and clamping it to the *settled* count re-covered an action.
A mark that names a node has no such edge: the newest action is the node,
and "everything after it" is empty until the story grows.
"""
adventure = make_adventure(db, 12)
db.add(models.Memory(adventure_id=adventure.id, text="A", source_start=0, source_end=5))
db.add(models.Memory(adventure_id=adventure.id, text="B", source_start=6, source_end=11))
adventure.memory_cursor = 12
adventure.summary_cursor = 12
db.commit()
cover(db, adventure, 12)
# The clamp as run_post_turn applies it.
count = len(memorybank.story_actions(adventure))
adventure.memory_cursor = min(adventure.memory_cursor, count)
assert adventure.memory_cursor == 12 # not rewound to 11
assert covered_depth(db, adventure) == 11 # the newest action, not one past it
assert memorybank.settled_after(adventure, covered_depth(db, adventure)) == -1
# Grow the story and let the next block form.
for i in range(12, 25):
@@ -191,93 +221,114 @@ def test_no_memories_before_memory_start(db, monkeypatch):
# ------------------------------------------- deleting already-summarized ground
def orphans(db, adventure) -> list[int]:
"""Action indices the cursor calls summarized that no memory describes."""
"""Depths the mark calls summarized that no memory describes.
The failure this whole section is about, stated once: an action behind the
mark with nothing covering it is never summarized again, and nothing ever
reports it.
"""
covered: set[int] = set()
for m in db.query(models.Memory).filter_by(adventure_id=adventure.id):
covered |= set(range(m.source_start, m.source_end + 1))
actions = memorybank.story_actions(adventure)
return [a.index for a in actions[: adventure.memory_cursor] if a.index not in covered]
mark = cursors.MEMORY.depth(db, adventure)
return [
a.depth for a in memorybank.story_actions(adventure)
if a.depth <= mark and a.depth not in covered
]
def summarized_adventure(db):
"""13 actions with two memories covering indices 0-11, cursor at 12."""
"""13 actions with two memories covering depths 0-11, the mark on node 11."""
adventure = make_adventure(db, 13)
db.add(models.Memory(adventure_id=adventure.id, text="A", source_start=0, source_end=5))
db.add(models.Memory(adventure_id=adventure.id, text="B", source_start=6, source_end=11))
adventure.memory_cursor = 12
adventure.summary_cursor = 12
db.commit()
for text, start, end in (("A", 0, 5), ("B", 6, 11)):
node = db.query(models.Action).filter_by(
adventure_id=adventure.id, index=end
).one()
memory = models.Memory(
adventure_id=adventure.id, text=text, source_start=start, source_end=end
)
tree.attach_memory(memory, node)
db.add(memory)
cover(db, adventure, 12)
db.refresh(adventure)
return adventure
def test_deleting_a_middle_action_does_not_skip_a_later_one(db):
"""memory_cursor counts positions, so removing an earlier action slides a
never-summarized one into the covered range unless the cursor slides too."""
adventure = summarized_adventure(db)
victim = db.query(models.Action).filter_by(adventure_id=adventure.id, index=5).one()
def test_deleting_a_middle_action_leaves_the_mark_where_it_was(db):
"""The bug that motivated the old machinery, and the reason it is gone.
memorybank.note_action_removed(adventure, victim)
A position cursor counted actions from the start, so deleting an earlier
one slid a never-summarized action into the covered range and every delete
had to correct for it. A depth is not a count: node 11 is still node 11
with node 5 gone.
"""
adventure = summarized_adventure(db)
# Node 4 is inside memory A's block but is not the node it hangs off, so
# nothing is withdrawn — the same reading the old code had, where only a
# memory whose *end* had fallen off the story was pruned.
victim = db.query(models.Action).filter_by(adventure_id=adventure.id, index=4).one()
assert memorybank.forget_node(db, adventure, victim) == 0
db.delete(victim)
db.commit()
db.refresh(adventure)
assert adventure.memory_cursor == 11 # slid down by one
assert covered_depth(db, adventure) == 11
assert [m.text for m in db.query(models.Memory).all()] == ["A", "B"]
assert orphans(db, adventure) == []
def test_deleting_a_later_action_leaves_cursors_alone(db):
"""Only actions *before* the cursor shift it."""
def test_deleting_a_later_action_leaves_the_mark_alone(db):
adventure = summarized_adventure(db)
victim = db.query(models.Action).filter_by(adventure_id=adventure.id, index=12).one()
memorybank.note_action_removed(adventure, victim)
memorybank.forget_node(db, adventure, victim)
db.delete(victim)
db.commit()
db.refresh(adventure)
assert adventure.memory_cursor == 12
assert covered_depth(db, adventure) == 11
assert orphans(db, adventure) == []
def test_pruning_a_memory_rewinds_to_where_it_started(db):
"""Discarding a memory isn't enough — the actions it covered are still
behind the cursor, so they must be handed back to the summarizer."""
adventure = summarized_adventure(db)
# Delete back past index 11, so memory B (6..11) covers a missing action.
for index in (12, 11):
victim = db.query(models.Action).filter_by(adventure_id=adventure.id, index=index).one()
memorybank.note_action_removed(adventure, victim)
db.delete(victim)
db.flush()
db.expire(adventure, ["actions"])
def test_deleting_a_summarized_node_withdraws_its_memory(db):
"""Discarding the memory isn't enough — the story it covered is still
behind the mark, so the mark has to come back to where that block began.
assert memorybank.prune_dangling_memories(adventure, db) == 1
Memory B ends on node 11, so deleting node 11 is what withdraws it. The old
code found this by scanning for a memory whose covered range had fallen off
the end of the story; the memory hangs off the node now, so it is a lookup.
"""
adventure = summarized_adventure(db)
victim = db.query(models.Action).filter_by(adventure_id=adventure.id, index=11).one()
assert memorybank.forget_node(db, adventure, victim) == 1
db.delete(victim)
db.commit()
db.refresh(adventure)
assert [m.text for m in db.query(models.Memory).all()] == ["A"]
assert adventure.memory_cursor == 6 # back to where the discarded memory began
assert covered_depth(db, adventure) == 5 # back to where the discarded memory began
assert cursors.SUMMARY.depth(db, adventure) == 5 # and the summary with it
assert orphans(db, adventure) == []
def test_repeated_deletes_never_orphan_an_action(db):
"""The scenario that motivated this: undo/delete-last, over and over."""
"""The scenario that motivated this: undo/delete-last, over and over.
No clamp in the loop any more, and no bookkeeping call per delete beyond
withdrawing what the node produced.
"""
adventure = summarized_adventure(db)
for _ in range(6):
actions = memorybank.story_actions(adventure)
if not actions:
break
victim = max(actions, key=lambda a: a.index)
memorybank.note_action_removed(adventure, victim)
victim = max(actions, key=lambda a: a.depth)
memorybank.forget_node(db, adventure, victim)
db.delete(victim)
db.flush()
db.expire(adventure, ["actions"])
memorybank.prune_dangling_memories(adventure, db)
count = len(memorybank.story_actions(adventure))
adventure.memory_cursor = min(adventure.memory_cursor, count)
adventure.summary_cursor = min(adventure.summary_cursor, count)
db.commit()
db.refresh(adventure)
assert orphans(db, adventure) == []
assert adventure.memory_cursor <= len(memorybank.story_actions(adventure))
+12 -7
View File
@@ -140,24 +140,29 @@ def test_undo_prunes_memory_covering_removed_actions(db):
assert texts == {"k"}
# ---------------------------------------------------------------- prune helper
# -------------------------------------------------------- withdrawing a node
def test_prune_dangling_memories_counts_and_removes(db):
def test_forget_node_withdraws_only_what_that_node_produced(db):
"""Phase 14 SP3: a memory hangs off the node its block ends on, so removing
a node is a lookup rather than a scan for memories that have fallen off the
end of the story."""
user, adv = _make_adventure(db, {})
_add(db, adv, 0, "do")
_add(db, adv, 1, "ai")
second = _add(db, adv, 1, "ai")
db.add_all([
models.Memory(adventure_id=adv.id, text="live", source_start=0, source_end=1),
models.Memory(adventure_id=adv.id, text="dead", source_start=2, source_end=5),
models.Memory(adventure_id=adv.id, text="hangs off node 1",
source_start=0, source_end=1),
models.Memory(adventure_id=adv.id, text="hangs off node 0",
source_start=0, source_end=0),
])
db.commit()
removed = memorybank.prune_dangling_memories(adv, db)
removed = memorybank.forget_node(db, adv, second)
db.commit()
db.refresh(adv) # expire_on_commit=False: reload the memories collection
assert removed == 1
assert {m.text for m in adv.memories} == {"live"}
assert {m.text for m in adv.memories} == {"hangs off node 0"}
# ---------------------------------------------------------------- snapshot
+78 -7
View File
@@ -99,6 +99,13 @@ PRE_TREE_DDL = (
# action never renumbered the ones after it. The gap has to survive as a gap.
GAPPED_INDEXES = (0, 1, 2, 4)
STRAIGHT_INDEXES = (0, 1)
# "Blank" holds an action whose text is nothing but whitespace. It is a row of
# the adventure but not of the *story*, so a cursor counting covered actions
# never counted it — and migration 56 has to skip it the same way, using a
# frozen copy of the story-text predicate. This is the one duplicated
# definition in the change, so it gets the one case that can tell.
BLANK_INDEXES = (0, 1, 2, 3)
BLANK_AT = 2
@pytest.fixture()
@@ -123,11 +130,20 @@ def pre_tree():
"demo_turns_date) VALUES (1, 'v45@example.com', 0, CURRENT_TIMESTAMP, 0, '')"
))
# The cursors as schema 45 held them: counts of covered story actions.
# Gapped's story is 0,1,2,4 — so "3 covered" is the node at depth 2 and
# "4 covered" is the node at depth 4, which is the whole reason a count
# and a depth are not the same number. Straight is caught up past its
# own end (5 covered, 2 actions), which is a state the older rule left
# behind and the clamp used to paper over every post-turn pass.
cursors_at = {"Gapped": (3, 4), "Straight": (5, 0), "Empty": (0, 0),
"Blank": (3, 0)}
ids = {}
for name in ("Gapped", "Straight", "Empty"):
for name in ("Gapped", "Straight", "Empty", "Blank"):
conn.execute(text(
"INSERT INTO adventures (user_id, title) VALUES (1, :title)"
), {"title": name})
"INSERT INTO adventures (user_id, title, memory_cursor, summary_cursor) "
"VALUES (1, :title, :mc, :sc)"
), {"title": name, "mc": cursors_at[name][0], "sc": cursors_at[name][1]})
ids[name] = conn.execute(text(
"SELECT id FROM adventures WHERE title = :title"
), {"title": name}).scalar()
@@ -135,14 +151,16 @@ def pre_tree():
for adventure_id, indexes in (
(ids["Gapped"], GAPPED_INDEXES),
(ids["Straight"], STRAIGHT_INDEXES),
(ids["Blank"], BLANK_INDEXES),
):
for index in indexes:
blank = adventure_id == ids["Blank"] and index == BLANK_AT
conn.execute(text(
'INSERT INTO actions (adventure_id, "index", type, text) '
"VALUES (:a, :i, :t, :x)"
), {"a": adventure_id, "i": index,
"t": "start" if index == 0 else "do",
"x": f"Turn {index}."})
"x": " \n\t " if blank else f"Turn {index}."})
# One memory that summarised a block of story, and one written by hand,
# which summarised nothing and so belongs to no node.
@@ -218,7 +236,7 @@ def test_one_root_branch_per_adventure_with_its_own_lineage(pre_tree):
branches = rows(
"SELECT id, adventure_id, parent_branch_id, fork_depth, lineage FROM branches"
)
assert len(branches) == 3, "one branch per adventure, including the empty one"
assert len(branches) == 4, "one branch per adventure, including the empty one"
for branch_id, _adventure_id, parent, fork_depth, lineage in branches:
assert parent is None, "a migrated branch is a root; nothing forked yet"
assert fork_depth is None
@@ -263,6 +281,54 @@ def test_memories_attach_to_the_node_they_summarised(pre_tree):
assert manual and all(depth is None and branch is not None for depth, branch in manual)
def test_the_cursors_become_the_nodes_they_named(pre_tree):
"""SP3, migration 56. A count of covered actions and a depth are different
numbers the moment the story has a gap in it, which every adventure anyone
has ever deleted from does."""
migrations.bootstrap(engine)
def marks(title):
[row] = rows(
"SELECT memory_cursor_depth, summary_cursor_depth, "
"memory_cursor_branch_id, summary_cursor_branch_id "
"FROM adventures WHERE title = :t", t=title
)
return row
# Gapped's story is 0,1,2,4. "3 covered" is the *third* action, at depth 2 —
# reading the count as a depth would have handed the summarizer node 3,
# which does not exist, and quietly skipped node 4 forever.
memory_depth, summary_depth, memory_branch, summary_branch = marks("Gapped")
assert (memory_depth, summary_depth) == (2, 4)
root = scalar(
"SELECT id FROM branches WHERE adventure_id = "
"(SELECT id FROM adventures WHERE title = 'Gapped')"
)
assert memory_branch == summary_branch == root
# Straight was caught up under the older rule: 5 covered, 2 actions. There
# is no fifth node to name, and the number meant "caught up", so it lands
# on the tip rather than on nothing.
memory_depth, summary_depth, _, summary_branch = marks("Straight")
assert memory_depth == 1
assert (summary_depth, summary_branch) == (migrations.NO_DEPTH, None)
# Nothing covered stays nothing covered, and names no branch.
assert marks("Empty") == (migrations.NO_DEPTH, migrations.NO_DEPTH, None, None)
# A whitespace-only action is a row but not a story action, so it was never
# counted — "3 covered" of 0,1,[blank],3 is the node at depth 3, not 2. The
# migration's copy of the story-text predicate is the only place that rule
# is written twice, so this is the case that catches it drifting.
assert marks("Blank")[0] == 3
# The legacy columns are left exactly as they were: a rolled-back build
# reads them, and this migration is not the one that drops them.
assert rows(
"SELECT memory_cursor, summary_cursor FROM adventures ORDER BY title"
) == [(3, 0), (0, 0), (3, 4), (5, 0)] # Blank, Empty, Gapped, Straight
def test_the_branch_clause_index_exists(pre_tree):
"""SP2's reads are only cheap if this exists — and `create_all` does not add
an index to a table it did not create, which is what migration 52 is for."""
@@ -279,7 +345,9 @@ def test_running_it_again_changes_nothing(pre_tree):
snapshot = (
rows("SELECT id, branch_id, depth FROM actions ORDER BY id"),
rows("SELECT id, adventure_id, lineage FROM branches ORDER BY id"),
rows("SELECT id, head_branch_id, head_depth FROM adventures ORDER BY id"),
rows("SELECT id, head_branch_id, head_depth, memory_cursor_branch_id, "
"memory_cursor_depth, summary_cursor_branch_id, summary_cursor_depth "
"FROM adventures ORDER BY id"),
rows("SELECT id, branch_id, depth FROM memories ORDER BY id"),
)
@@ -290,11 +358,14 @@ def test_running_it_again_changes_nothing(pre_tree):
migrations.bootstrap(engine)
with engine.begin() as conn:
migrations._backfill_tree(conn)
migrations._backfill_cursor_anchors(conn)
assert (
rows("SELECT id, branch_id, depth FROM actions ORDER BY id"),
rows("SELECT id, adventure_id, lineage FROM branches ORDER BY id"),
rows("SELECT id, head_branch_id, head_depth FROM adventures ORDER BY id"),
rows("SELECT id, head_branch_id, head_depth, memory_cursor_branch_id, "
"memory_cursor_depth, summary_cursor_branch_id, summary_cursor_depth "
"FROM adventures ORDER BY id"),
rows("SELECT id, branch_id, depth FROM memories ORDER BY id"),
) == snapshot