Read a window of the story per turn instead of all of it
Two reads still grew without bound after the snapshot fix. `Action.variants` holds every discarded retry attempt, but a list response only needs how many there are — so each retry permanently added ~5 KB to every later load of that adventure. Defer the column and keep the count beside it (migration 37, backfilled server-side), with set_variants() as the one write path that keeps the two in step. `story_actions()` walked adventure.actions, then every caller threw almost all of it away: the builder concatenates the story and immediately cuts it back to the token budget, the NPC check looks at the last 6, retrieval at the last 4, the cursor clamp only wants a count. A turn on a 200-action adventure read 839 KB to use ~70 KB, and grew with every turn played. app/context/ history.py serves those shapes from SQL; window_covering() measures the actions it fetched and projects how many more it needs, fetching only the part it does not already hold. Memorybank cursors move to position_of_index() and settled_count()/settled_slice() — same arithmetic, no full list. The scripting pipeline still receives the whole history per AI Dungeon's API, and every helper reuses adventure.actions when it is already loaded, so a scripted adventure pays what it always did and never twice. Measured at production shape: retry tax 5.1 KB -> 0; turn 200 839 KB -> 129 KB and flat from ~turn 50; a 200-turn playthrough 84.5 MB -> 23.0 MB; a delete 115 KB -> 5 KB. Verified the window builds a byte-identical prompt to the full story across budgets from 1K to 100K tokens, with and without the retry exclusion - this is a cost change and nothing else. Cursor helpers checked against the old list arithmetic, including after deleting a middle action. Counts are real SELECT count(...): Query.count() wraps the entity select in a subquery, so the SQL named every deferred column and the egress guard could not tell it apart from a bulk fetch. 139 tests pass; the four new guards verified by sabotage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UeQVy5bEjLhfgWNc27Efet
This commit is contained in:
co-authored by
Claude Opus 5
parent
f1bd099ec8
commit
47e33fa311
@@ -17,9 +17,11 @@ from dataclasses import dataclass
|
||||
import tiktoken
|
||||
|
||||
from .. import models, worldstate
|
||||
from . import history
|
||||
|
||||
AUTHORS_NOTE_DEPTH = 3 # actions from the end of history
|
||||
CARD_BUDGET_SHARE = 0.4 # max share of non-reserved budget that story cards may take
|
||||
NPC_WINDOW = 6 # actions of story searched for NPC trigger words ("in scene")
|
||||
SEPARATOR = "\n\n"
|
||||
|
||||
|
||||
@@ -76,25 +78,12 @@ def _history_text(action: models.Action) -> str:
|
||||
return text
|
||||
|
||||
|
||||
def story_actions(
|
||||
adventure: models.Adventure, exclude_action_id: int | None = None
|
||||
) -> list[models.Action]:
|
||||
"""The adventure's non-empty actions, as the model should see them.
|
||||
|
||||
`exclude_action_id` drops one action from the story — used by retry, where
|
||||
the row being regenerated is still attached to the adventure (it holds the
|
||||
variant history) but must not appear in the context assembled to replace it.
|
||||
"""
|
||||
return [
|
||||
a for a in adventure.actions
|
||||
if a.text.strip() and (exclude_action_id is None or a.id != exclude_action_id)
|
||||
]
|
||||
|
||||
|
||||
def _visible_npcs(actions: list[models.Action], stat_schema: dict) -> dict[str, str]:
|
||||
"""Defined NPCs whose trigger words appear in the recent story — the ones
|
||||
"in scene", so only their stats get injected. Maps npc id -> display name."""
|
||||
recent = SEPARATOR.join(a.text for a in actions[-6:]).lower()
|
||||
"in scene", so only their stats get injected. Maps npc id -> display name.
|
||||
|
||||
`actions` is already the last handful (see NPC_WINDOW)."""
|
||||
recent = SEPARATOR.join(a.text for a in actions).lower()
|
||||
visible: dict[str, str] = {}
|
||||
for npc_key, ndef in (stat_schema.get("npcs") or {}).items():
|
||||
if not isinstance(ndef, dict):
|
||||
@@ -127,9 +116,8 @@ def build_context(
|
||||
) -> tuple[str, str, dict]:
|
||||
"""Returns (system_text, story_text, context_report). `memory_bank` is the
|
||||
result of memorybank.retrieve_memories (None when the bank is off);
|
||||
`exclude_action_id` omits one action from the story (see story_actions)."""
|
||||
`exclude_action_id` omits one action from the story (see history.py)."""
|
||||
script_mem = _script_memory(adventure)
|
||||
actions = story_actions(adventure, exclude_action_id)
|
||||
|
||||
# ----- Always-included components -----
|
||||
system_sections: list[Section] = [Section("narrator", settings.narrator_prompt.strip())]
|
||||
@@ -142,7 +130,10 @@ def build_context(
|
||||
if guide:
|
||||
system_sections.append(Section("world_state_guide", guide))
|
||||
block = worldstate.render_state_section(
|
||||
adventure.world_state, stat_schema, _visible_npcs(actions, stat_schema)
|
||||
adventure.world_state, stat_schema,
|
||||
_visible_npcs(
|
||||
history.tail(adventure, NPC_WINDOW, exclude_action_id), stat_schema
|
||||
),
|
||||
)
|
||||
if block:
|
||||
system_sections.append(Section("world_state", block))
|
||||
@@ -181,6 +172,14 @@ def build_context(
|
||||
)
|
||||
available = max(256, settings.context_token_budget - reserved)
|
||||
|
||||
# Only the newest actions can reach the prompt: everything below is either
|
||||
# truncated to `available` tokens or stops at the budget. Fetch a window
|
||||
# that is provably larger than that and no more — a long adventure would
|
||||
# otherwise read its entire history every turn to use the tail of it.
|
||||
actions = history.window_covering(
|
||||
adventure, available, count_tokens, exclude_action_id
|
||||
)
|
||||
|
||||
# ----- Story cards: triggered by recent story text (the window history could fill) -----
|
||||
trigger_window = truncate_to_last_tokens(SEPARATOR.join(a.text for a in actions), available)
|
||||
triggered = _match_cards(adventure.story_cards, trigger_window)
|
||||
@@ -268,7 +267,9 @@ def build_context(
|
||||
"memories": memory_bank,
|
||||
"history": {
|
||||
"included": len(included_actions),
|
||||
"total": len(actions),
|
||||
# The whole story, not just the window fetched above — Insights
|
||||
# reports "N of M actions included" and M is the real total.
|
||||
"total": history.count(adventure, exclude_action_id),
|
||||
"oldest_truncated": oldest_truncated,
|
||||
},
|
||||
"settings": {
|
||||
|
||||
Reference in New Issue
Block a user