Ask the model for a turn that fits inside the output cap
max_output_tokens is a hard wall the endpoint enforces mid-sentence. The ```state block is emitted after the narration, so a long turn hits the wall partway through the block and the deltas are lost — silently, since nothing reads finish_reason. builder.length_hint() derives a word limit from the cap ((cap - 50 headroom) * 0.75 words/token * 0.90 buffer) and injects it just above EMIT_REMINDER, which keeps the last slot it needs. Reserved in build_context like the reminder is. Phrased as a ceiling, not a budget. Measured against gemma-4-26b at cap 800, n=5 per arm: no hint 174 words, "keep this turn under about N words" 246, "hard limit ... a typical turn is much shorter" 170. A budget reads as a target to fill — every budget run was longer than every unhinted one, pushing turns toward the wall the hint exists to avoid. Ceiling phrasing still works at tight caps: at 250, unhinted hit finish_reason=length 2/6, hinted 0/6. tests/test_length_hint.py, 11 tests; each mechanism verified by sabotage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UeQVy5bEjLhfgWNc27Efet
This commit is contained in:
co-authored by
Claude Opus 5
parent
23398c4ac0
commit
f295893204
@@ -24,6 +24,20 @@ CARD_BUDGET_SHARE = 0.4 # max share of non-reserved budget that story cards may
|
||||
NPC_WINDOW = 6 # actions of story searched for NPC trigger words ("in scene")
|
||||
SEPARATOR = "\n\n"
|
||||
|
||||
# Output-length guidance. max_output_tokens is a hard wall the endpoint enforces
|
||||
# mid-sentence: hitting it truncates whatever is being written, and since the
|
||||
# state block is emitted last, it is what gets lost. Asking the model to land
|
||||
# just inside the wall keeps the cut from happening in the first place.
|
||||
LENGTH_HEADROOM = 50 # tokens held back from the cap for the state block itself
|
||||
# Models cannot count their own tokens, but they do follow a word budget, so the
|
||||
# reserved budget is stated in words. ~0.75 words per token for English prose.
|
||||
WORDS_PER_TOKEN = 0.75
|
||||
# A word budget is a suggestion the model routinely overshoots, and the cap it is
|
||||
# protecting is a hard wall — so aim 10% short of the real ceiling and let the
|
||||
# overshoot land in the slack instead of in the state block.
|
||||
LENGTH_BUFFER = 0.90
|
||||
MIN_LENGTH_HINT_WORDS = 40 # below this the hint is noise; a tiny cap speaks for itself
|
||||
|
||||
|
||||
@functools.lru_cache(maxsize=1)
|
||||
def _encoding() -> tiktoken.Encoding:
|
||||
@@ -51,6 +65,34 @@ class Section:
|
||||
return count_tokens(self.text)
|
||||
|
||||
|
||||
def length_hint(max_output_tokens: int, *, has_ws: bool) -> str:
|
||||
"""Ask for a turn that fits inside the output cap, stated as a word budget.
|
||||
|
||||
Returns "" when the cap is too small to phrase usefully — the hint is a
|
||||
suggestion the model can drift past, so it only earns its tokens when there
|
||||
is enough room for the drift to still land inside the wall.
|
||||
"""
|
||||
words = int((max_output_tokens - LENGTH_HEADROOM) * WORDS_PER_TOKEN * LENGTH_BUFFER)
|
||||
if words < MIN_LENGTH_HINT_WORDS:
|
||||
return ""
|
||||
tail = (
|
||||
" Finish the narration and append the state block well inside the limit."
|
||||
if has_ws
|
||||
else " Bring the turn to a close well inside the limit rather than "
|
||||
"stopping mid-sentence."
|
||||
)
|
||||
# Phrased as a ceiling, never as a budget. Measured against this model, "keep
|
||||
# this turn under about N words" reads as a target to fill: it moved a 174-word
|
||||
# average to 246 (n=5, every run longer than every unhinted one), i.e. the hint
|
||||
# pushed turns toward the very wall it exists to keep them away from. Naming
|
||||
# the number as a limit, plus saying a typical turn is far shorter, left the
|
||||
# average at 170 while still rescuing the state block at tight caps.
|
||||
return (
|
||||
f"[Hard limit: this turn must not exceed {words} words. Write only as "
|
||||
f"much as the moment needs — a typical turn is much shorter.{tail}]"
|
||||
)
|
||||
|
||||
|
||||
def _script_memory(adventure: models.Adventure) -> dict:
|
||||
"""Script-provided memory overrides (populated by Phase 4 scripting)."""
|
||||
state = adventure.script_state if isinstance(adventure.script_state, dict) else {}
|
||||
@@ -164,10 +206,13 @@ def build_context(
|
||||
if isinstance(script_mem.get("frontMemory"), str):
|
||||
front_memory = script_mem["frontMemory"].strip()
|
||||
|
||||
length_note = length_hint(settings.max_output_tokens, has_ws=has_ws)
|
||||
|
||||
reserved = (
|
||||
sum(s.tokens for s in system_sections)
|
||||
+ count_tokens(authors_note)
|
||||
+ count_tokens(front_memory)
|
||||
+ count_tokens(length_note)
|
||||
+ (count_tokens(worldstate.EMIT_REMINDER) if has_ws else 0)
|
||||
)
|
||||
available = max(256, settings.context_token_budget - reserved)
|
||||
@@ -244,6 +289,10 @@ def build_context(
|
||||
note_sections.append(Section("history", SEPARATOR.join(texts)))
|
||||
if front_memory:
|
||||
note_sections.append(Section("front_memory", front_memory))
|
||||
# Sits just above the emit reminder, which keeps the strongest recency slot:
|
||||
# the length budget is about the narration, the reminder about the block that
|
||||
# comes after it, so this is also the order the model has to act in.
|
||||
note_sections.append(Section("length_hint", length_note))
|
||||
if has_ws:
|
||||
# Terminal reminder: the emit rule sits up in the system block, far from
|
||||
# where the model generates; repeat it last, in the strongest recency slot.
|
||||
|
||||
Reference in New Issue
Block a user