Prompt caching bills on a shared prefix: the endpoint reuses the request up to the first byte that differs from last time and no further. The live world-state block sat third from the top of the system message, so every turn re-priced the instructions, the plot essentials and the whole story history underneath it. The retrieved memories and the rewritten summary did it again. Everything fixed is emitted first now, and everything that moves goes after the history, ordered least-volatile first — which is also where recency serves it best, the reasoning that already put the emit reminder last. The three tail sections that are last for their own reasons stay last. The moved sections are still charged to the token budget; only their position changed. Two smaller halves of the same problem. OpenRouter serves a model from whichever upstream is free and each upstream holds its own cache, so a deepseek model now names deepseek as its preferred upstream — a preference, not a restriction, so a turn still runs if that upstream is down. And the endpoint's usage block is read back off the response and kept per attempt, so the hit rate shows up in Insights and the debug log instead of being assumed. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01DfMCsN1KBLsTqMkj5hSgrY
404 lines
19 KiB
Python
404 lines
19 KiB
Python
"""Context assembly per AI Dungeon's memory system
|
|
(help.aidungeon.com/faq/the-memory-system):
|
|
|
|
[AI Instructions] always included
|
|
[Plot Essentials] always included (classic "Memory")
|
|
[Story Summary] always included (manual in Phase 3, auto in Phase 6)
|
|
[Used Memories] top-K memory-bank retrievals (Phase 6, when enabled)
|
|
[Triggered Story Cards] "World Lore: <entry>", conditional; first dropped when over budget
|
|
[Story history] newest actions that fit the remaining token budget
|
|
[Author's Note] injected AUTHORS_NOTE_DEPTH actions before the end of history
|
|
[Latest player action] (+ script frontMemory right after it, Phase 4)
|
|
|
|
Which components are present is AI Dungeon's design, above. The *order* they
|
|
are laid out in is not: everything fixed is emitted first and everything that
|
|
moves after the history, because prompt caching bills on a shared prefix and
|
|
one mutable section high up re-prices the whole prompt under it. See the two
|
|
"static block" / "live sections" comments in `build_context`.
|
|
"""
|
|
|
|
import functools
|
|
from dataclasses import dataclass
|
|
|
|
import tiktoken
|
|
|
|
from .. import models, worldstate
|
|
from . import history
|
|
|
|
AUTHORS_NOTE_DEPTH = 3 # actions from the end of history
|
|
CARD_BUDGET_SHARE = 0.4 # max share of non-reserved budget that story cards may take
|
|
NPC_WINDOW = 6 # actions of story searched for NPC trigger words ("in scene")
|
|
SEPARATOR = "\n\n"
|
|
|
|
# Output-length guidance. max_output_tokens is a hard wall the endpoint enforces
|
|
# mid-sentence: hitting it truncates whatever is being written, and since the
|
|
# state block is emitted last, it is what gets lost. Asking the model to land
|
|
# just inside the wall keeps the cut from happening in the first place.
|
|
LENGTH_HEADROOM = 50 # tokens held back from the cap for the state block itself
|
|
# Models cannot count their own tokens, but they do follow a word budget, so the
|
|
# reserved budget is stated in words. ~0.75 words per token for English prose.
|
|
WORDS_PER_TOKEN = 0.75
|
|
# A word budget is a suggestion the model routinely overshoots, and the cap it is
|
|
# protecting is a hard wall — so aim 10% short of the real ceiling and let the
|
|
# overshoot land in the slack instead of in the state block.
|
|
LENGTH_BUFFER = 0.90
|
|
MIN_LENGTH_HINT_WORDS = 40 # below this the hint is noise; a tiny cap speaks for itself
|
|
# A ceiling alone is a one-sided instruction, and models read it very differently:
|
|
# a verbose one is held back by it, while a terse one has nothing to act on except
|
|
# the "write only as much as the moment needs" clause and collapses to two
|
|
# paragraphs. Stating a floor as well turns the guidance into a band, so the same
|
|
# prompt lands in the same place regardless of which way the model leans. Set as a
|
|
# share of the ceiling so the floor can never approach it.
|
|
LENGTH_FLOOR_SHARE = 0.35
|
|
# Below this a floor is meaningless — at a tight cap a short turn is the correct
|
|
# turn — and the tight-cap wording is the one measured to keep the state block
|
|
# alive, so it is left exactly as it was.
|
|
MIN_LENGTH_FLOOR_WORDS = 60
|
|
# The floor exists to stop a collapse to two paragraphs, not to demand an essay:
|
|
# at a 2400-token cap the share alone would ask for 555 words *minimum*. Past this
|
|
# point a reader wanting more length can say so in the author's note.
|
|
MAX_LENGTH_FLOOR_WORDS = 300
|
|
|
|
|
|
@functools.lru_cache(maxsize=1)
|
|
def _encoding() -> tiktoken.Encoding:
|
|
return tiktoken.get_encoding("cl100k_base")
|
|
|
|
|
|
def count_tokens(text: str) -> int:
|
|
return len(_encoding().encode(text))
|
|
|
|
|
|
def truncate_to_last_tokens(text: str, budget: int) -> str:
|
|
tokens = _encoding().encode(text)
|
|
if len(tokens) <= budget:
|
|
return text
|
|
return _encoding().decode(tokens[-budget:])
|
|
|
|
|
|
@dataclass
|
|
class Section:
|
|
label: str
|
|
text: str
|
|
|
|
@property
|
|
def tokens(self) -> int:
|
|
return count_tokens(self.text)
|
|
|
|
|
|
def length_hint(max_output_tokens: int, *, has_ws: bool) -> str:
|
|
"""Ask for a turn that fits inside the output cap, stated as a word budget.
|
|
|
|
Returns "" when the cap is too small to phrase usefully — the hint is a
|
|
suggestion the model can drift past, so it only earns its tokens when there
|
|
is enough room for the drift to still land inside the wall.
|
|
"""
|
|
words = int((max_output_tokens - LENGTH_HEADROOM) * WORDS_PER_TOKEN * LENGTH_BUFFER)
|
|
if words < MIN_LENGTH_HINT_WORDS:
|
|
return ""
|
|
tail = (
|
|
" Finish the narration and append the state block well inside the limit."
|
|
if has_ws
|
|
else " Bring the turn to a close well inside the limit rather than "
|
|
"stopping mid-sentence."
|
|
)
|
|
# Phrased as a ceiling, never as a budget. Measured against this model, "keep
|
|
# this turn under about N words" reads as a target to fill: it moved a 174-word
|
|
# average to 246 (n=5, every run longer than every unhinted one), i.e. the hint
|
|
# pushed turns toward the very wall it exists to keep them away from. Naming
|
|
# the number as a limit, plus saying a typical turn is far shorter, left the
|
|
# average at 170 while still rescuing the state block at tight caps.
|
|
floor = min(int(words * LENGTH_FLOOR_SHARE), MAX_LENGTH_FLOOR_WORDS)
|
|
if floor < MIN_LENGTH_FLOOR_WORDS:
|
|
return (
|
|
f"[Hard limit: this turn must not exceed {words} words. Write only as "
|
|
f"much as the moment needs — a typical turn is much shorter.{tail}]"
|
|
)
|
|
# Both numbers are bounds, and deliberately asymmetric ones: "must not exceed"
|
|
# for the wall the endpoint enforces, "should not stop short of" for the floor.
|
|
# Neither is a target, which is what the measurement above says matters. The
|
|
# "prefer the lower end" clause does the job the old "a typical turn is much
|
|
# shorter" line did — holding a verbose model off the wall — but now with a
|
|
# number under it, so a terse model reading the same clause lands on the floor
|
|
# instead of at forty words.
|
|
return (
|
|
f"[Hard limit: this turn must not exceed {words} words, and it should not "
|
|
f"stop short of about {floor}. Prefer the lower end of that range unless "
|
|
f"the scene genuinely needs more.{tail}]"
|
|
)
|
|
|
|
|
|
def _script_memory(adventure: models.Adventure) -> dict:
|
|
"""Script-provided memory overrides (populated by Phase 4 scripting)."""
|
|
state = adventure.script_state if isinstance(adventure.script_state, dict) else {}
|
|
memory = state.get("memory")
|
|
return memory if isinstance(memory, dict) else {}
|
|
|
|
|
|
def _history_text(action: models.Action) -> str:
|
|
"""An AI turn as the model should see it in replayed history: its narration
|
|
with the state block it emitted re-appended (reconstructed from the stored
|
|
delta). The block is stripped before storage/UI, so without this every past
|
|
AI turn would look like one that emitted nothing — biasing the model, by
|
|
imitation, to stop emitting too. Player turns and blockless turns are
|
|
returned unchanged.
|
|
|
|
Reads `world_delta`, not `context_snapshot`: this runs for every action in
|
|
the replayed history, and the snapshot is deferred precisely so a turn
|
|
never drags the prompt archive out of the database."""
|
|
text = action.text
|
|
wd = action.world_delta if isinstance(action.world_delta, dict) else None
|
|
if wd:
|
|
block = worldstate.render_delta_block(wd.get("delta") or {})
|
|
if block:
|
|
text = f"{text}\n{block}"
|
|
return text
|
|
|
|
|
|
def _visible_npcs(actions: list[models.Action], stat_schema: dict) -> dict[str, str]:
|
|
"""Defined NPCs whose trigger words appear in the recent story — the ones
|
|
"in scene", so only their stats get injected. Maps npc id -> display name.
|
|
|
|
`actions` is already the last handful (see NPC_WINDOW)."""
|
|
recent = SEPARATOR.join(a.text for a in actions).lower()
|
|
visible: dict[str, str] = {}
|
|
for npc_key, ndef in (stat_schema.get("npcs") or {}).items():
|
|
if not isinstance(ndef, dict):
|
|
continue
|
|
if any(trigger in recent for trigger in worldstate.npc_triggers(ndef, npc_key)):
|
|
visible[npc_key] = worldstate.npc_name(ndef, npc_key)
|
|
return visible
|
|
|
|
|
|
def _match_cards(cards: list[models.StoryCard], window_text: str) -> list[dict]:
|
|
"""AI Dungeon trigger rules: case-insensitive, space-sensitive, partial-word
|
|
('boat' triggers on 'boats'). Returns one record per card with the keyword that fired."""
|
|
haystack = window_text.lower()
|
|
matched = []
|
|
for card in cards:
|
|
for key in (k.strip().lower() for k in card.keys.split(",")):
|
|
if key and key in haystack:
|
|
matched.append(
|
|
{"id": card.id, "name": card.name, "keyword": key, "entry": card.entry}
|
|
)
|
|
break
|
|
return matched
|
|
|
|
|
|
def build_context(
|
|
adventure: models.Adventure,
|
|
settings: models.Settings,
|
|
memory_bank: dict | None = None,
|
|
exclude_action_id: int | None = None,
|
|
) -> tuple[str, str, dict]:
|
|
"""Returns (system_text, story_text, context_report). `memory_bank` is the
|
|
result of memorybank.retrieve_memories (None when the bank is off);
|
|
`exclude_action_id` omits one action from the story (see history.py)."""
|
|
script_mem = _script_memory(adventure)
|
|
|
|
# ----- The static block: byte-for-byte the same prompt every turn -----
|
|
# The ordering here is a billing decision, not a stylistic one. Prompt
|
|
# caching matches a *prefix*: an endpoint reuses the prompt up to the first
|
|
# byte that differs from last time and no further. So one mutable section
|
|
# near the top re-prices everything below it, and what is below it is the
|
|
# story history, which is the bulk of the prompt. Anything that changes
|
|
# turn to turn therefore goes *after* the history, in the live sections —
|
|
# which is also where recency serves it best, the same reasoning that
|
|
# already puts EMIT_REMINDER last.
|
|
system_sections: list[Section] = [Section("narrator", settings.narrator_prompt.strip())]
|
|
|
|
# RPG world state (Phase 12): how to report changes. The live values are a
|
|
# live section below; the schema-derived guide and the emit rule are fixed
|
|
# for as long as the scenario is.
|
|
stat_schema = adventure.scenario.stat_schema if adventure.scenario else None
|
|
has_ws = worldstate.has_schema(stat_schema)
|
|
if has_ws:
|
|
guide = worldstate.render_reference(stat_schema)
|
|
if guide:
|
|
system_sections.append(Section("world_state_guide", guide))
|
|
system_sections.append(Section("world_state_rule", worldstate.EMIT_RULE))
|
|
|
|
if isinstance(script_mem.get("context"), str) and script_mem["context"].strip():
|
|
system_sections.append(Section("script_context", script_mem["context"].strip()))
|
|
if adventure.ai_instructions.strip():
|
|
system_sections.append(Section("ai_instructions", adventure.ai_instructions.strip()))
|
|
if adventure.memory.strip():
|
|
system_sections.append(
|
|
Section("plot_essentials", f"Plot essentials:\n{adventure.memory.strip()}")
|
|
)
|
|
|
|
# ----- Live sections: everything that moves, built here, placed after the
|
|
# history further down. Ordered least-volatile first, so a turn that
|
|
# changes only the fastest-moving one keeps the others cached too: the
|
|
# summary is rewritten every few turns, lore turns over with the scene, the
|
|
# retrieved memories change on most turns, the stat values on nearly all.
|
|
# `world_lore` joins them below — it is the history window that triggers
|
|
# the cards, so it cannot be known yet.
|
|
summary_section = (
|
|
Section("story_summary", f"Story summary:\n{adventure.story_summary.strip()}")
|
|
if adventure.story_summary.strip()
|
|
else None
|
|
)
|
|
memories_section = None
|
|
if memory_bank and memory_bank.get("used"):
|
|
lines_text = "\n".join(f"- {m['text']}" for m in memory_bank["used"])
|
|
memories_section = Section("used_memories", f"Memories:\n{lines_text}")
|
|
world_state_section = None
|
|
if has_ws:
|
|
block = worldstate.render_state_section(
|
|
adventure.world_state, stat_schema,
|
|
_visible_npcs(
|
|
history.tail(adventure, NPC_WINDOW, exclude_action_id), stat_schema
|
|
),
|
|
)
|
|
if block:
|
|
world_state_section = Section("world_state", block)
|
|
|
|
authors_note_text = adventure.authors_note.strip()
|
|
if isinstance(script_mem.get("authorsNote"), str) and script_mem["authorsNote"].strip():
|
|
authors_note_text = script_mem["authorsNote"].strip()
|
|
authors_note = f"[Author's note: {authors_note_text}]" if authors_note_text else ""
|
|
|
|
front_memory = ""
|
|
if isinstance(script_mem.get("frontMemory"), str):
|
|
front_memory = script_mem["frontMemory"].strip()
|
|
|
|
length_note = length_hint(settings.max_output_tokens, has_ws=has_ws)
|
|
|
|
# The live sections moved below the history but they are still in the
|
|
# prompt, so they are still reserved against the budget. (`world_lore` is
|
|
# not: it is budgeted out of `available` further down, as it always was.)
|
|
reserved = (
|
|
sum(s.tokens for s in system_sections)
|
|
+ sum(
|
|
s.tokens
|
|
for s in (summary_section, memories_section, world_state_section)
|
|
if s is not None
|
|
)
|
|
+ count_tokens(authors_note)
|
|
+ count_tokens(front_memory)
|
|
+ count_tokens(length_note)
|
|
+ (count_tokens(worldstate.EMIT_REMINDER) if has_ws else 0)
|
|
)
|
|
available = max(256, settings.context_token_budget - reserved)
|
|
|
|
# Only the newest actions can reach the prompt: everything below is either
|
|
# truncated to `available` tokens or stops at the budget. Fetch a window
|
|
# that is provably larger than that and no more — a long adventure would
|
|
# otherwise read its entire history every turn to use the tail of it.
|
|
actions = history.window_covering(
|
|
adventure, available, count_tokens, exclude_action_id
|
|
)
|
|
|
|
# ----- Story cards: triggered by recent story text (the window history could fill) -----
|
|
trigger_window = truncate_to_last_tokens(SEPARATOR.join(a.text for a in actions), available)
|
|
triggered = _match_cards(adventure.story_cards, trigger_window)
|
|
|
|
card_budget = int(available * CARD_BUDGET_SHARE)
|
|
card_records = []
|
|
lore_lines: list[str] = []
|
|
used = 0
|
|
for match in triggered:
|
|
line = f"World Lore: {match['entry'].strip()}"
|
|
tokens = count_tokens(line)
|
|
included = used + tokens <= card_budget
|
|
if included:
|
|
lore_lines.append(line)
|
|
used += tokens
|
|
card_records.append(
|
|
{"id": match["id"], "name": match["name"], "keyword": match["keyword"],
|
|
"included": included}
|
|
)
|
|
lore_section = (
|
|
Section("world_lore", "\n".join(lore_lines)) if lore_lines else None
|
|
)
|
|
|
|
# ----- Story history: newest first until the remaining budget is spent -----
|
|
history_budget = available - used
|
|
included_actions: list[models.Action] = []
|
|
spent = 0
|
|
oldest_truncated = False
|
|
for action in reversed(actions):
|
|
# Budget on the text as it will actually appear — with the re-attached
|
|
# state block (B) when this adventure tracks world state.
|
|
rendered = _history_text(action) if has_ws else action.text
|
|
tokens = count_tokens(rendered) + count_tokens(SEPARATOR)
|
|
if spent + tokens > history_budget:
|
|
if not included_actions:
|
|
# Even the newest action alone is over budget: hard-truncate it.
|
|
included_actions.append(
|
|
models.Action(
|
|
adventure_id=action.adventure_id, index=action.index,
|
|
type=action.type,
|
|
text=truncate_to_last_tokens(action.text, history_budget),
|
|
)
|
|
)
|
|
oldest_truncated = True
|
|
break
|
|
included_actions.append(action)
|
|
spent += tokens
|
|
included_actions.reverse()
|
|
|
|
# ----- Assemble story text with author's note near the end -----
|
|
# Re-attach each AI turn's state block (stripped before storage) so recent
|
|
# history shows the model its own emit pattern to imitate.
|
|
texts = [_history_text(a) if has_ws else a.text for a in included_actions]
|
|
note_sections: list[Section] = []
|
|
if authors_note:
|
|
pos = max(0, len(texts) - AUTHORS_NOTE_DEPTH)
|
|
before, after = texts[:pos], texts[pos:]
|
|
if before:
|
|
note_sections.append(Section("history", SEPARATOR.join(before)))
|
|
note_sections.append(Section("authors_note", authors_note))
|
|
note_sections.append(Section("recent_history", SEPARATOR.join(after)))
|
|
else:
|
|
note_sections.append(Section("history", SEPARATOR.join(texts)))
|
|
# The live sections, least volatile first (see where they are built). They
|
|
# sit below the history so the history caches, and above the tail so the
|
|
# three sections that are last for a reason stay last.
|
|
for live in (summary_section, lore_section, memories_section, world_state_section):
|
|
if live is not None:
|
|
note_sections.append(live)
|
|
if front_memory:
|
|
note_sections.append(Section("front_memory", front_memory))
|
|
# Sits just above the emit reminder, which keeps the strongest recency slot:
|
|
# the length budget is about the narration, the reminder about the block that
|
|
# comes after it, so this is also the order the model has to act in.
|
|
note_sections.append(Section("length_hint", length_note))
|
|
if has_ws:
|
|
# Terminal reminder: the emit rule sits up in the system block, far from
|
|
# where the model generates; repeat it last, in the strongest recency slot.
|
|
note_sections.append(Section("world_state_reminder", worldstate.EMIT_REMINDER))
|
|
|
|
story_sections = [s for s in note_sections if s.text]
|
|
system_text = SEPARATOR.join(s.text for s in system_sections if s.text)
|
|
story_text = SEPARATOR.join(s.text for s in story_sections)
|
|
|
|
all_sections = [s for s in system_sections if s.text] + story_sections
|
|
report = {
|
|
"sections": [
|
|
{"label": s.label, "text": s.text, "tokens": s.tokens} for s in all_sections
|
|
],
|
|
"prompt": {"system": system_text, "story": story_text},
|
|
"tokens": {
|
|
"total": count_tokens(system_text) + count_tokens(story_text),
|
|
"budget": settings.context_token_budget,
|
|
},
|
|
"cards": card_records,
|
|
"memories": memory_bank,
|
|
"history": {
|
|
"included": len(included_actions),
|
|
# The whole story, not just the window fetched above — Insights
|
|
# reports "N of M actions included" and M is the real total.
|
|
"total": history.count(adventure, exclude_action_id),
|
|
"oldest_truncated": oldest_truncated,
|
|
},
|
|
"settings": {
|
|
"model": settings.model,
|
|
"api_mode": settings.api_mode,
|
|
"temperature": settings.temperature,
|
|
"max_output_tokens": settings.max_output_tokens,
|
|
},
|
|
}
|
|
return system_text, story_text, report
|