"""Context assembly per AI Dungeon's memory system (help.aidungeon.com/faq/the-memory-system): [AI Instructions] always included [Plot Essentials] always included (classic "Memory") [Story Summary] always included (manual in Phase 3, auto in Phase 6) [Used Memories] top-K memory-bank retrievals (Phase 6, when enabled) [Triggered Story Cards] "World Lore: ", conditional; first dropped when over budget [Story history] newest actions that fit the remaining token budget [Author's Note] injected AUTHORS_NOTE_DEPTH actions before the end of history [Latest player action] (+ script frontMemory right after it, Phase 4) Which components are present is AI Dungeon's design, above. The *order* they are laid out in is not: everything fixed is emitted first and everything that moves after the history, because prompt caching bills on a shared prefix and one mutable section high up re-prices the whole prompt under it. See the two "static block" / "live sections" comments in `build_context`. """ import functools from dataclasses import dataclass import tiktoken from .. import models, worldstate from . import history AUTHORS_NOTE_DEPTH = 3 # actions from the end of history CARD_BUDGET_SHARE = 0.4 # max share of non-reserved budget that story cards may take NPC_WINDOW = 6 # actions of story searched for NPC trigger words ("in scene") SEPARATOR = "\n\n" # Output-length guidance. max_output_tokens is a hard wall the endpoint enforces # mid-sentence: hitting it truncates whatever is being written, and since the # state block is emitted last, it is what gets lost. Asking the model to land # just inside the wall keeps the cut from happening in the first place. LENGTH_HEADROOM = 50 # tokens held back from the cap for the state block itself # Models cannot count their own tokens, but they do follow a word budget, so the # reserved budget is stated in words. ~0.75 words per token for English prose. WORDS_PER_TOKEN = 0.75 # A word budget is a suggestion the model routinely overshoots, and the cap it is # protecting is a hard wall — so aim 10% short of the real ceiling and let the # overshoot land in the slack instead of in the state block. LENGTH_BUFFER = 0.90 MIN_LENGTH_HINT_WORDS = 40 # below this the hint is noise; a tiny cap speaks for itself # A ceiling alone is a one-sided instruction, and models read it very differently: # a verbose one is held back by it, while a terse one has nothing to act on except # the "write only as much as the moment needs" clause and collapses to two # paragraphs. Stating a floor as well turns the guidance into a band, so the same # prompt lands in the same place regardless of which way the model leans. Set as a # share of the ceiling so the floor can never approach it. LENGTH_FLOOR_SHARE = 0.35 # Below this a floor is meaningless — at a tight cap a short turn is the correct # turn — and the tight-cap wording is the one measured to keep the state block # alive, so it is left exactly as it was. MIN_LENGTH_FLOOR_WORDS = 60 # The floor exists to stop a collapse to two paragraphs, not to demand an essay: # at a 2400-token cap the share alone would ask for 555 words *minimum*. Past this # point a reader wanting more length can say so in the author's note. MAX_LENGTH_FLOOR_WORDS = 300 @functools.lru_cache(maxsize=1) def _encoding() -> tiktoken.Encoding: return tiktoken.get_encoding("cl100k_base") def count_tokens(text: str) -> int: return len(_encoding().encode(text)) def truncate_to_last_tokens(text: str, budget: int) -> str: tokens = _encoding().encode(text) if len(tokens) <= budget: return text return _encoding().decode(tokens[-budget:]) @dataclass class Section: label: str text: str @property def tokens(self) -> int: return count_tokens(self.text) def length_hint(max_output_tokens: int, *, has_ws: bool) -> str: """Ask for a turn that fits inside the output cap, stated as a word budget. Returns "" when the cap is too small to phrase usefully — the hint is a suggestion the model can drift past, so it only earns its tokens when there is enough room for the drift to still land inside the wall. """ words = int((max_output_tokens - LENGTH_HEADROOM) * WORDS_PER_TOKEN * LENGTH_BUFFER) if words < MIN_LENGTH_HINT_WORDS: return "" tail = ( " Finish the narration and append the state block well inside the limit." if has_ws else " Bring the turn to a close well inside the limit rather than " "stopping mid-sentence." ) # Phrased as a ceiling, never as a budget. Measured against this model, "keep # this turn under about N words" reads as a target to fill: it moved a 174-word # average to 246 (n=5, every run longer than every unhinted one), i.e. the hint # pushed turns toward the very wall it exists to keep them away from. Naming # the number as a limit, plus saying a typical turn is far shorter, left the # average at 170 while still rescuing the state block at tight caps. floor = min(int(words * LENGTH_FLOOR_SHARE), MAX_LENGTH_FLOOR_WORDS) if floor < MIN_LENGTH_FLOOR_WORDS: return ( f"[Hard limit: this turn must not exceed {words} words. Write only as " f"much as the moment needs — a typical turn is much shorter.{tail}]" ) # Both numbers are bounds, and deliberately asymmetric ones: "must not exceed" # for the wall the endpoint enforces, "should not stop short of" for the floor. # Neither is a target, which is what the measurement above says matters. The # "prefer the lower end" clause does the job the old "a typical turn is much # shorter" line did — holding a verbose model off the wall — but now with a # number under it, so a terse model reading the same clause lands on the floor # instead of at forty words. return ( f"[Hard limit: this turn must not exceed {words} words, and it should not " f"stop short of about {floor}. Prefer the lower end of that range unless " f"the scene genuinely needs more.{tail}]" ) def _script_memory(adventure: models.Adventure) -> dict: """Script-provided memory overrides (populated by Phase 4 scripting).""" state = adventure.script_state if isinstance(adventure.script_state, dict) else {} memory = state.get("memory") return memory if isinstance(memory, dict) else {} def _history_text(action: models.Action) -> str: """An AI turn as the model should see it in replayed history: its narration with the state block it emitted re-appended (reconstructed from the stored delta). The block is stripped before storage/UI, so without this every past AI turn would look like one that emitted nothing — biasing the model, by imitation, to stop emitting too. Player turns and blockless turns are returned unchanged. Reads `world_delta`, not `context_snapshot`: this runs for every action in the replayed history, and the snapshot is deferred precisely so a turn never drags the prompt archive out of the database.""" text = action.text wd = action.world_delta if isinstance(action.world_delta, dict) else None if wd: block = worldstate.render_delta_block(wd.get("delta") or {}) if block: text = f"{text}\n{block}" return text def _visible_npcs(actions: list[models.Action], stat_schema: dict) -> dict[str, str]: """Defined NPCs whose trigger words appear in the recent story — the ones "in scene", so only their stats get injected. Maps npc id -> display name. `actions` is already the last handful (see NPC_WINDOW).""" recent = SEPARATOR.join(a.text for a in actions).lower() visible: dict[str, str] = {} for npc_key, ndef in (stat_schema.get("npcs") or {}).items(): if not isinstance(ndef, dict): continue if any(trigger in recent for trigger in worldstate.npc_triggers(ndef, npc_key)): visible[npc_key] = worldstate.npc_name(ndef, npc_key) return visible def _match_cards(cards: list[models.StoryCard], window_text: str) -> list[dict]: """AI Dungeon trigger rules: case-insensitive, space-sensitive, partial-word ('boat' triggers on 'boats'). Returns one record per card with the keyword that fired.""" haystack = window_text.lower() matched = [] for card in cards: for key in (k.strip().lower() for k in card.keys.split(",")): if key and key in haystack: matched.append( {"id": card.id, "name": card.name, "keyword": key, "entry": card.entry} ) break return matched def build_context( adventure: models.Adventure, settings: models.Settings, memory_bank: dict | None = None, exclude_action_id: int | None = None, ) -> tuple[str, str, dict]: """Returns (system_text, story_text, context_report). `memory_bank` is the result of memorybank.retrieve_memories (None when the bank is off); `exclude_action_id` omits one action from the story (see history.py).""" script_mem = _script_memory(adventure) # ----- The static block: byte-for-byte the same prompt every turn ----- # The ordering here is a billing decision, not a stylistic one. Prompt # caching matches a *prefix*: an endpoint reuses the prompt up to the first # byte that differs from last time and no further. So one mutable section # near the top re-prices everything below it, and what is below it is the # story history, which is the bulk of the prompt. Anything that changes # turn to turn therefore goes *after* the history, in the live sections — # which is also where recency serves it best, the same reasoning that # already puts EMIT_REMINDER last. system_sections: list[Section] = [Section("narrator", settings.narrator_prompt.strip())] # RPG world state (Phase 12): how to report changes. The live values are a # live section below; the schema-derived guide and the emit rule are fixed # for as long as the scenario is. stat_schema = adventure.scenario.stat_schema if adventure.scenario else None has_ws = worldstate.has_schema(stat_schema) if has_ws: guide = worldstate.render_reference(stat_schema) if guide: system_sections.append(Section("world_state_guide", guide)) system_sections.append(Section("world_state_rule", worldstate.EMIT_RULE)) if isinstance(script_mem.get("context"), str) and script_mem["context"].strip(): system_sections.append(Section("script_context", script_mem["context"].strip())) if adventure.ai_instructions.strip(): system_sections.append(Section("ai_instructions", adventure.ai_instructions.strip())) if adventure.memory.strip(): system_sections.append( Section("plot_essentials", f"Plot essentials:\n{adventure.memory.strip()}") ) # ----- Live sections: everything that moves, built here, placed after the # history further down. Ordered least-volatile first, so a turn that # changes only the fastest-moving one keeps the others cached too: the # summary is rewritten every few turns, lore turns over with the scene, the # retrieved memories change on most turns, the stat values on nearly all. # `world_lore` joins them below — it is the history window that triggers # the cards, so it cannot be known yet. summary_section = ( Section("story_summary", f"Story summary:\n{adventure.story_summary.strip()}") if adventure.story_summary.strip() else None ) memories_section = None if memory_bank and memory_bank.get("used"): lines_text = "\n".join(f"- {m['text']}" for m in memory_bank["used"]) memories_section = Section("used_memories", f"Memories:\n{lines_text}") world_state_section = None if has_ws: block = worldstate.render_state_section( adventure.world_state, stat_schema, _visible_npcs( history.tail(adventure, NPC_WINDOW, exclude_action_id), stat_schema ), ) if block: world_state_section = Section("world_state", block) authors_note_text = adventure.authors_note.strip() if isinstance(script_mem.get("authorsNote"), str) and script_mem["authorsNote"].strip(): authors_note_text = script_mem["authorsNote"].strip() authors_note = f"[Author's note: {authors_note_text}]" if authors_note_text else "" front_memory = "" if isinstance(script_mem.get("frontMemory"), str): front_memory = script_mem["frontMemory"].strip() length_note = length_hint(settings.max_output_tokens, has_ws=has_ws) # The live sections moved below the history but they are still in the # prompt, so they are still reserved against the budget. (`world_lore` is # not: it is budgeted out of `available` further down, as it always was.) reserved = ( sum(s.tokens for s in system_sections) + sum( s.tokens for s in (summary_section, memories_section, world_state_section) if s is not None ) + count_tokens(authors_note) + count_tokens(front_memory) + count_tokens(length_note) + (count_tokens(worldstate.EMIT_REMINDER) if has_ws else 0) ) available = max(256, settings.context_token_budget - reserved) # Only the newest actions can reach the prompt: everything below is either # truncated to `available` tokens or stops at the budget. Fetch a window # that is provably larger than that and no more — a long adventure would # otherwise read its entire history every turn to use the tail of it. actions = history.window_covering( adventure, available, count_tokens, exclude_action_id ) # ----- Story cards: triggered by recent story text (the window history could fill) ----- trigger_window = truncate_to_last_tokens(SEPARATOR.join(a.text for a in actions), available) triggered = _match_cards(adventure.story_cards, trigger_window) card_budget = int(available * CARD_BUDGET_SHARE) card_records = [] lore_lines: list[str] = [] used = 0 for match in triggered: line = f"World Lore: {match['entry'].strip()}" tokens = count_tokens(line) included = used + tokens <= card_budget if included: lore_lines.append(line) used += tokens card_records.append( {"id": match["id"], "name": match["name"], "keyword": match["keyword"], "included": included} ) lore_section = ( Section("world_lore", "\n".join(lore_lines)) if lore_lines else None ) # ----- Story history: newest first until the remaining budget is spent ----- history_budget = available - used included_actions: list[models.Action] = [] spent = 0 oldest_truncated = False for action in reversed(actions): # Budget on the text as it will actually appear — with the re-attached # state block (B) when this adventure tracks world state. rendered = _history_text(action) if has_ws else action.text tokens = count_tokens(rendered) + count_tokens(SEPARATOR) if spent + tokens > history_budget: if not included_actions: # Even the newest action alone is over budget: hard-truncate it. included_actions.append( models.Action( adventure_id=action.adventure_id, index=action.index, type=action.type, text=truncate_to_last_tokens(action.text, history_budget), ) ) oldest_truncated = True break included_actions.append(action) spent += tokens included_actions.reverse() # ----- Assemble story text with author's note near the end ----- # Re-attach each AI turn's state block (stripped before storage) so recent # history shows the model its own emit pattern to imitate. texts = [_history_text(a) if has_ws else a.text for a in included_actions] note_sections: list[Section] = [] if authors_note: pos = max(0, len(texts) - AUTHORS_NOTE_DEPTH) before, after = texts[:pos], texts[pos:] if before: note_sections.append(Section("history", SEPARATOR.join(before))) note_sections.append(Section("authors_note", authors_note)) note_sections.append(Section("recent_history", SEPARATOR.join(after))) else: note_sections.append(Section("history", SEPARATOR.join(texts))) # The live sections, least volatile first (see where they are built). They # sit below the history so the history caches, and above the tail so the # three sections that are last for a reason stay last. for live in (summary_section, lore_section, memories_section, world_state_section): if live is not None: note_sections.append(live) if front_memory: note_sections.append(Section("front_memory", front_memory)) # Sits just above the emit reminder, which keeps the strongest recency slot: # the length budget is about the narration, the reminder about the block that # comes after it, so this is also the order the model has to act in. note_sections.append(Section("length_hint", length_note)) if has_ws: # Terminal reminder: the emit rule sits up in the system block, far from # where the model generates; repeat it last, in the strongest recency slot. note_sections.append(Section("world_state_reminder", worldstate.EMIT_REMINDER)) story_sections = [s for s in note_sections if s.text] system_text = SEPARATOR.join(s.text for s in system_sections if s.text) story_text = SEPARATOR.join(s.text for s in story_sections) all_sections = [s for s in system_sections if s.text] + story_sections report = { "sections": [ {"label": s.label, "text": s.text, "tokens": s.tokens} for s in all_sections ], "prompt": {"system": system_text, "story": story_text}, "tokens": { "total": count_tokens(system_text) + count_tokens(story_text), "budget": settings.context_token_budget, }, "cards": card_records, "memories": memory_bank, "history": { "included": len(included_actions), # The whole story, not just the window fetched above — Insights # reports "N of M actions included" and M is the real total. "total": history.count(adventure, exclude_action_id), "oldest_truncated": oldest_truncated, }, "settings": { "model": settings.model, "api_mode": settings.api_mode, "temperature": settings.temperature, "max_output_tokens": settings.max_output_tokens, }, } return system_text, story_text, report