Files
interactive-story/backend/app/context/builder.py
T
parththakkar106andClaude Opus 5 f295893204 Ask the model for a turn that fits inside the output cap
max_output_tokens is a hard wall the endpoint enforces mid-sentence. The
```state block is emitted after the narration, so a long turn hits the wall
partway through the block and the deltas are lost — silently, since nothing
reads finish_reason.

builder.length_hint() derives a word limit from the cap ((cap - 50 headroom)
* 0.75 words/token * 0.90 buffer) and injects it just above EMIT_REMINDER,
which keeps the last slot it needs. Reserved in build_context like the
reminder is.

Phrased as a ceiling, not a budget. Measured against gemma-4-26b at cap 800,
n=5 per arm: no hint 174 words, "keep this turn under about N words" 246,
"hard limit ... a typical turn is much shorter" 170. A budget reads as a
target to fill — every budget run was longer than every unhinted one, pushing
turns toward the wall the hint exists to avoid. Ceiling phrasing still works
at tight caps: at 250, unhinted hit finish_reason=length 2/6, hinted 0/6.

tests/test_length_hint.py, 11 tests; each mechanism verified by sabotage.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01UeQVy5bEjLhfgWNc27Efet
2026-08-08 12:20:26 +05:30

332 lines
14 KiB
Python

"""Context assembly per AI Dungeon's memory system
(help.aidungeon.com/faq/the-memory-system):
[AI Instructions] always included
[Plot Essentials] always included (classic "Memory")
[Story Summary] always included (manual in Phase 3, auto in Phase 6)
[Used Memories] top-K memory-bank retrievals (Phase 6, when enabled)
[Triggered Story Cards] "World Lore: <entry>", conditional; first dropped when over budget
[Story history] newest actions that fit the remaining token budget
[Author's Note] injected AUTHORS_NOTE_DEPTH actions before the end of history
[Latest player action] (+ script frontMemory right after it, Phase 4)
"""
import functools
from dataclasses import dataclass
import tiktoken
from .. import models, worldstate
from . import history
AUTHORS_NOTE_DEPTH = 3 # actions from the end of history
CARD_BUDGET_SHARE = 0.4 # max share of non-reserved budget that story cards may take
NPC_WINDOW = 6 # actions of story searched for NPC trigger words ("in scene")
SEPARATOR = "\n\n"
# Output-length guidance. max_output_tokens is a hard wall the endpoint enforces
# mid-sentence: hitting it truncates whatever is being written, and since the
# state block is emitted last, it is what gets lost. Asking the model to land
# just inside the wall keeps the cut from happening in the first place.
LENGTH_HEADROOM = 50 # tokens held back from the cap for the state block itself
# Models cannot count their own tokens, but they do follow a word budget, so the
# reserved budget is stated in words. ~0.75 words per token for English prose.
WORDS_PER_TOKEN = 0.75
# A word budget is a suggestion the model routinely overshoots, and the cap it is
# protecting is a hard wall — so aim 10% short of the real ceiling and let the
# overshoot land in the slack instead of in the state block.
LENGTH_BUFFER = 0.90
MIN_LENGTH_HINT_WORDS = 40 # below this the hint is noise; a tiny cap speaks for itself
@functools.lru_cache(maxsize=1)
def _encoding() -> tiktoken.Encoding:
return tiktoken.get_encoding("cl100k_base")
def count_tokens(text: str) -> int:
return len(_encoding().encode(text))
def truncate_to_last_tokens(text: str, budget: int) -> str:
tokens = _encoding().encode(text)
if len(tokens) <= budget:
return text
return _encoding().decode(tokens[-budget:])
@dataclass
class Section:
label: str
text: str
@property
def tokens(self) -> int:
return count_tokens(self.text)
def length_hint(max_output_tokens: int, *, has_ws: bool) -> str:
"""Ask for a turn that fits inside the output cap, stated as a word budget.
Returns "" when the cap is too small to phrase usefully — the hint is a
suggestion the model can drift past, so it only earns its tokens when there
is enough room for the drift to still land inside the wall.
"""
words = int((max_output_tokens - LENGTH_HEADROOM) * WORDS_PER_TOKEN * LENGTH_BUFFER)
if words < MIN_LENGTH_HINT_WORDS:
return ""
tail = (
" Finish the narration and append the state block well inside the limit."
if has_ws
else " Bring the turn to a close well inside the limit rather than "
"stopping mid-sentence."
)
# Phrased as a ceiling, never as a budget. Measured against this model, "keep
# this turn under about N words" reads as a target to fill: it moved a 174-word
# average to 246 (n=5, every run longer than every unhinted one), i.e. the hint
# pushed turns toward the very wall it exists to keep them away from. Naming
# the number as a limit, plus saying a typical turn is far shorter, left the
# average at 170 while still rescuing the state block at tight caps.
return (
f"[Hard limit: this turn must not exceed {words} words. Write only as "
f"much as the moment needs — a typical turn is much shorter.{tail}]"
)
def _script_memory(adventure: models.Adventure) -> dict:
"""Script-provided memory overrides (populated by Phase 4 scripting)."""
state = adventure.script_state if isinstance(adventure.script_state, dict) else {}
memory = state.get("memory")
return memory if isinstance(memory, dict) else {}
def _history_text(action: models.Action) -> str:
"""An AI turn as the model should see it in replayed history: its narration
with the state block it emitted re-appended (reconstructed from the stored
delta). The block is stripped before storage/UI, so without this every past
AI turn would look like one that emitted nothing — biasing the model, by
imitation, to stop emitting too. Player turns and blockless turns are
returned unchanged.
Reads `world_delta`, not `context_snapshot`: this runs for every action in
the replayed history, and the snapshot is deferred precisely so a turn
never drags the prompt archive out of the database."""
text = action.text
wd = action.world_delta if isinstance(action.world_delta, dict) else None
if wd:
block = worldstate.render_delta_block(wd.get("delta") or {})
if block:
text = f"{text}\n{block}"
return text
def _visible_npcs(actions: list[models.Action], stat_schema: dict) -> dict[str, str]:
"""Defined NPCs whose trigger words appear in the recent story — the ones
"in scene", so only their stats get injected. Maps npc id -> display name.
`actions` is already the last handful (see NPC_WINDOW)."""
recent = SEPARATOR.join(a.text for a in actions).lower()
visible: dict[str, str] = {}
for npc_key, ndef in (stat_schema.get("npcs") or {}).items():
if not isinstance(ndef, dict):
continue
if any(trigger in recent for trigger in worldstate.npc_triggers(ndef, npc_key)):
visible[npc_key] = worldstate.npc_name(ndef, npc_key)
return visible
def _match_cards(cards: list[models.StoryCard], window_text: str) -> list[dict]:
"""AI Dungeon trigger rules: case-insensitive, space-sensitive, partial-word
('boat' triggers on 'boats'). Returns one record per card with the keyword that fired."""
haystack = window_text.lower()
matched = []
for card in cards:
for key in (k.strip().lower() for k in card.keys.split(",")):
if key and key in haystack:
matched.append(
{"id": card.id, "name": card.name, "keyword": key, "entry": card.entry}
)
break
return matched
def build_context(
adventure: models.Adventure,
settings: models.Settings,
memory_bank: dict | None = None,
exclude_action_id: int | None = None,
) -> tuple[str, str, dict]:
"""Returns (system_text, story_text, context_report). `memory_bank` is the
result of memorybank.retrieve_memories (None when the bank is off);
`exclude_action_id` omits one action from the story (see history.py)."""
script_mem = _script_memory(adventure)
# ----- Always-included components -----
system_sections: list[Section] = [Section("narrator", settings.narrator_prompt.strip())]
# RPG world state (Phase 12): current stats/milestones + how to report changes.
stat_schema = adventure.scenario.stat_schema if adventure.scenario else None
has_ws = worldstate.has_schema(stat_schema)
if has_ws:
guide = worldstate.render_reference(stat_schema)
if guide:
system_sections.append(Section("world_state_guide", guide))
block = worldstate.render_state_section(
adventure.world_state, stat_schema,
_visible_npcs(
history.tail(adventure, NPC_WINDOW, exclude_action_id), stat_schema
),
)
if block:
system_sections.append(Section("world_state", block))
system_sections.append(Section("world_state_rule", worldstate.EMIT_RULE))
if isinstance(script_mem.get("context"), str) and script_mem["context"].strip():
system_sections.append(Section("script_context", script_mem["context"].strip()))
if adventure.ai_instructions.strip():
system_sections.append(Section("ai_instructions", adventure.ai_instructions.strip()))
if adventure.memory.strip():
system_sections.append(
Section("plot_essentials", f"Plot essentials:\n{adventure.memory.strip()}")
)
if adventure.story_summary.strip():
system_sections.append(
Section("story_summary", f"Story summary:\n{adventure.story_summary.strip()}")
)
if memory_bank and memory_bank.get("used"):
lines = "\n".join(f"- {m['text']}" for m in memory_bank["used"])
system_sections.append(Section("used_memories", f"Memories:\n{lines}"))
authors_note_text = adventure.authors_note.strip()
if isinstance(script_mem.get("authorsNote"), str) and script_mem["authorsNote"].strip():
authors_note_text = script_mem["authorsNote"].strip()
authors_note = f"[Author's note: {authors_note_text}]" if authors_note_text else ""
front_memory = ""
if isinstance(script_mem.get("frontMemory"), str):
front_memory = script_mem["frontMemory"].strip()
length_note = length_hint(settings.max_output_tokens, has_ws=has_ws)
reserved = (
sum(s.tokens for s in system_sections)
+ count_tokens(authors_note)
+ count_tokens(front_memory)
+ count_tokens(length_note)
+ (count_tokens(worldstate.EMIT_REMINDER) if has_ws else 0)
)
available = max(256, settings.context_token_budget - reserved)
# Only the newest actions can reach the prompt: everything below is either
# truncated to `available` tokens or stops at the budget. Fetch a window
# that is provably larger than that and no more — a long adventure would
# otherwise read its entire history every turn to use the tail of it.
actions = history.window_covering(
adventure, available, count_tokens, exclude_action_id
)
# ----- Story cards: triggered by recent story text (the window history could fill) -----
trigger_window = truncate_to_last_tokens(SEPARATOR.join(a.text for a in actions), available)
triggered = _match_cards(adventure.story_cards, trigger_window)
card_budget = int(available * CARD_BUDGET_SHARE)
card_records = []
lore_lines: list[str] = []
used = 0
for match in triggered:
line = f"World Lore: {match['entry'].strip()}"
tokens = count_tokens(line)
included = used + tokens <= card_budget
if included:
lore_lines.append(line)
used += tokens
card_records.append(
{"id": match["id"], "name": match["name"], "keyword": match["keyword"],
"included": included}
)
if lore_lines:
system_sections.append(Section("world_lore", "\n".join(lore_lines)))
# ----- Story history: newest first until the remaining budget is spent -----
history_budget = available - used
included_actions: list[models.Action] = []
spent = 0
oldest_truncated = False
for action in reversed(actions):
# Budget on the text as it will actually appear — with the re-attached
# state block (B) when this adventure tracks world state.
rendered = _history_text(action) if has_ws else action.text
tokens = count_tokens(rendered) + count_tokens(SEPARATOR)
if spent + tokens > history_budget:
if not included_actions:
# Even the newest action alone is over budget: hard-truncate it.
included_actions.append(
models.Action(
adventure_id=action.adventure_id, index=action.index,
type=action.type,
text=truncate_to_last_tokens(action.text, history_budget),
)
)
oldest_truncated = True
break
included_actions.append(action)
spent += tokens
included_actions.reverse()
# ----- Assemble story text with author's note near the end -----
# Re-attach each AI turn's state block (stripped before storage) so recent
# history shows the model its own emit pattern to imitate.
texts = [_history_text(a) if has_ws else a.text for a in included_actions]
note_sections: list[Section] = []
if authors_note:
pos = max(0, len(texts) - AUTHORS_NOTE_DEPTH)
before, after = texts[:pos], texts[pos:]
if before:
note_sections.append(Section("history", SEPARATOR.join(before)))
note_sections.append(Section("authors_note", authors_note))
note_sections.append(Section("recent_history", SEPARATOR.join(after)))
else:
note_sections.append(Section("history", SEPARATOR.join(texts)))
if front_memory:
note_sections.append(Section("front_memory", front_memory))
# Sits just above the emit reminder, which keeps the strongest recency slot:
# the length budget is about the narration, the reminder about the block that
# comes after it, so this is also the order the model has to act in.
note_sections.append(Section("length_hint", length_note))
if has_ws:
# Terminal reminder: the emit rule sits up in the system block, far from
# where the model generates; repeat it last, in the strongest recency slot.
note_sections.append(Section("world_state_reminder", worldstate.EMIT_REMINDER))
story_sections = [s for s in note_sections if s.text]
system_text = SEPARATOR.join(s.text for s in system_sections if s.text)
story_text = SEPARATOR.join(s.text for s in story_sections)
all_sections = [s for s in system_sections if s.text] + story_sections
report = {
"sections": [
{"label": s.label, "text": s.text, "tokens": s.tokens} for s in all_sections
],
"prompt": {"system": system_text, "story": story_text},
"tokens": {
"total": count_tokens(system_text) + count_tokens(story_text),
"budget": settings.context_token_budget,
},
"cards": card_records,
"memories": memory_bank,
"history": {
"included": len(included_actions),
# The whole story, not just the window fetched above — Insights
# reports "N of M actions included" and M is the real total.
"total": history.count(adventure, exclude_action_id),
"oldest_truncated": oldest_truncated,
},
"settings": {
"model": settings.model,
"api_mode": settings.api_mode,
"temperature": settings.temperature,
"max_output_tokens": settings.max_output_tokens,
},
}
return system_text, story_text, report