Ask the model for a turn that fits inside the output cap
max_output_tokens is a hard wall the endpoint enforces mid-sentence. The ```state block is emitted after the narration, so a long turn hits the wall partway through the block and the deltas are lost — silently, since nothing reads finish_reason. builder.length_hint() derives a word limit from the cap ((cap - 50 headroom) * 0.75 words/token * 0.90 buffer) and injects it just above EMIT_REMINDER, which keeps the last slot it needs. Reserved in build_context like the reminder is. Phrased as a ceiling, not a budget. Measured against gemma-4-26b at cap 800, n=5 per arm: no hint 174 words, "keep this turn under about N words" 246, "hard limit ... a typical turn is much shorter" 170. A budget reads as a target to fill — every budget run was longer than every unhinted one, pushing turns toward the wall the hint exists to avoid. Ceiling phrasing still works at tight caps: at 250, unhinted hit finish_reason=length 2/6, hinted 0/6. tests/test_length_hint.py, 11 tests; each mechanism verified by sabotage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UeQVy5bEjLhfgWNc27Efet
This commit is contained in:
co-authored by
Claude Opus 5
parent
23398c4ac0
commit
f295893204
@@ -24,6 +24,20 @@ CARD_BUDGET_SHARE = 0.4 # max share of non-reserved budget that story cards may
|
|||||||
NPC_WINDOW = 6 # actions of story searched for NPC trigger words ("in scene")
|
NPC_WINDOW = 6 # actions of story searched for NPC trigger words ("in scene")
|
||||||
SEPARATOR = "\n\n"
|
SEPARATOR = "\n\n"
|
||||||
|
|
||||||
|
# Output-length guidance. max_output_tokens is a hard wall the endpoint enforces
|
||||||
|
# mid-sentence: hitting it truncates whatever is being written, and since the
|
||||||
|
# state block is emitted last, it is what gets lost. Asking the model to land
|
||||||
|
# just inside the wall keeps the cut from happening in the first place.
|
||||||
|
LENGTH_HEADROOM = 50 # tokens held back from the cap for the state block itself
|
||||||
|
# Models cannot count their own tokens, but they do follow a word budget, so the
|
||||||
|
# reserved budget is stated in words. ~0.75 words per token for English prose.
|
||||||
|
WORDS_PER_TOKEN = 0.75
|
||||||
|
# A word budget is a suggestion the model routinely overshoots, and the cap it is
|
||||||
|
# protecting is a hard wall — so aim 10% short of the real ceiling and let the
|
||||||
|
# overshoot land in the slack instead of in the state block.
|
||||||
|
LENGTH_BUFFER = 0.90
|
||||||
|
MIN_LENGTH_HINT_WORDS = 40 # below this the hint is noise; a tiny cap speaks for itself
|
||||||
|
|
||||||
|
|
||||||
@functools.lru_cache(maxsize=1)
|
@functools.lru_cache(maxsize=1)
|
||||||
def _encoding() -> tiktoken.Encoding:
|
def _encoding() -> tiktoken.Encoding:
|
||||||
@@ -51,6 +65,34 @@ class Section:
|
|||||||
return count_tokens(self.text)
|
return count_tokens(self.text)
|
||||||
|
|
||||||
|
|
||||||
|
def length_hint(max_output_tokens: int, *, has_ws: bool) -> str:
|
||||||
|
"""Ask for a turn that fits inside the output cap, stated as a word budget.
|
||||||
|
|
||||||
|
Returns "" when the cap is too small to phrase usefully — the hint is a
|
||||||
|
suggestion the model can drift past, so it only earns its tokens when there
|
||||||
|
is enough room for the drift to still land inside the wall.
|
||||||
|
"""
|
||||||
|
words = int((max_output_tokens - LENGTH_HEADROOM) * WORDS_PER_TOKEN * LENGTH_BUFFER)
|
||||||
|
if words < MIN_LENGTH_HINT_WORDS:
|
||||||
|
return ""
|
||||||
|
tail = (
|
||||||
|
" Finish the narration and append the state block well inside the limit."
|
||||||
|
if has_ws
|
||||||
|
else " Bring the turn to a close well inside the limit rather than "
|
||||||
|
"stopping mid-sentence."
|
||||||
|
)
|
||||||
|
# Phrased as a ceiling, never as a budget. Measured against this model, "keep
|
||||||
|
# this turn under about N words" reads as a target to fill: it moved a 174-word
|
||||||
|
# average to 246 (n=5, every run longer than every unhinted one), i.e. the hint
|
||||||
|
# pushed turns toward the very wall it exists to keep them away from. Naming
|
||||||
|
# the number as a limit, plus saying a typical turn is far shorter, left the
|
||||||
|
# average at 170 while still rescuing the state block at tight caps.
|
||||||
|
return (
|
||||||
|
f"[Hard limit: this turn must not exceed {words} words. Write only as "
|
||||||
|
f"much as the moment needs — a typical turn is much shorter.{tail}]"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def _script_memory(adventure: models.Adventure) -> dict:
|
def _script_memory(adventure: models.Adventure) -> dict:
|
||||||
"""Script-provided memory overrides (populated by Phase 4 scripting)."""
|
"""Script-provided memory overrides (populated by Phase 4 scripting)."""
|
||||||
state = adventure.script_state if isinstance(adventure.script_state, dict) else {}
|
state = adventure.script_state if isinstance(adventure.script_state, dict) else {}
|
||||||
@@ -164,10 +206,13 @@ def build_context(
|
|||||||
if isinstance(script_mem.get("frontMemory"), str):
|
if isinstance(script_mem.get("frontMemory"), str):
|
||||||
front_memory = script_mem["frontMemory"].strip()
|
front_memory = script_mem["frontMemory"].strip()
|
||||||
|
|
||||||
|
length_note = length_hint(settings.max_output_tokens, has_ws=has_ws)
|
||||||
|
|
||||||
reserved = (
|
reserved = (
|
||||||
sum(s.tokens for s in system_sections)
|
sum(s.tokens for s in system_sections)
|
||||||
+ count_tokens(authors_note)
|
+ count_tokens(authors_note)
|
||||||
+ count_tokens(front_memory)
|
+ count_tokens(front_memory)
|
||||||
|
+ count_tokens(length_note)
|
||||||
+ (count_tokens(worldstate.EMIT_REMINDER) if has_ws else 0)
|
+ (count_tokens(worldstate.EMIT_REMINDER) if has_ws else 0)
|
||||||
)
|
)
|
||||||
available = max(256, settings.context_token_budget - reserved)
|
available = max(256, settings.context_token_budget - reserved)
|
||||||
@@ -244,6 +289,10 @@ def build_context(
|
|||||||
note_sections.append(Section("history", SEPARATOR.join(texts)))
|
note_sections.append(Section("history", SEPARATOR.join(texts)))
|
||||||
if front_memory:
|
if front_memory:
|
||||||
note_sections.append(Section("front_memory", front_memory))
|
note_sections.append(Section("front_memory", front_memory))
|
||||||
|
# Sits just above the emit reminder, which keeps the strongest recency slot:
|
||||||
|
# the length budget is about the narration, the reminder about the block that
|
||||||
|
# comes after it, so this is also the order the model has to act in.
|
||||||
|
note_sections.append(Section("length_hint", length_note))
|
||||||
if has_ws:
|
if has_ws:
|
||||||
# Terminal reminder: the emit rule sits up in the system block, far from
|
# Terminal reminder: the emit rule sits up in the system block, far from
|
||||||
# where the model generates; repeat it last, in the strongest recency slot.
|
# where the model generates; repeat it last, in the strongest recency slot.
|
||||||
|
|||||||
@@ -0,0 +1,211 @@
|
|||||||
|
"""The turn prompt asks for a turn that fits inside `max_output_tokens`.
|
||||||
|
|
||||||
|
`max_output_tokens` is a hard wall the endpoint enforces mid-sentence. The state
|
||||||
|
block is emitted *after* the narration, so a long turn hits the wall partway
|
||||||
|
through the block and the deltas are lost — silently, since nothing reads
|
||||||
|
`finish_reason`. The prompt now carries a word budget derived from the cap so
|
||||||
|
the model lands just inside it.
|
||||||
|
|
||||||
|
Two things are easy to break here:
|
||||||
|
|
||||||
|
* the hint must be stated in **words**, not tokens — a model cannot count its
|
||||||
|
own tokens, and a hint it cannot follow is just wasted budget;
|
||||||
|
* it must not displace `EMIT_REMINDER` from the last position, which is the
|
||||||
|
whole mechanism keeping the state block emitted at all (see
|
||||||
|
test_worldstate.py and the emit-reliability fix).
|
||||||
|
|
||||||
|
python -m pytest tests/test_length_hint.py -v
|
||||||
|
"""
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
_tmp = tempfile.NamedTemporaryFile(suffix=".db", delete=False)
|
||||||
|
_tmp.close()
|
||||||
|
os.environ["AIDND_DB_PATH"] = _tmp.name
|
||||||
|
os.environ.pop("AIDND_DATABASE_URL", None)
|
||||||
|
os.environ.pop("DATABASE_URL", None)
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from app import models, worldstate
|
||||||
|
from app.context import builder
|
||||||
|
from app.database import Base, SessionLocal, engine
|
||||||
|
|
||||||
|
SCHEMA = {
|
||||||
|
"player": {"hp": {"min": 0, "max": 100, "initial": 100, "desc": "Health"}},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture()
|
||||||
|
def story():
|
||||||
|
"""A short adventure, with and without a stat schema on demand."""
|
||||||
|
Base.metadata.create_all(bind=engine)
|
||||||
|
db = SessionLocal()
|
||||||
|
user = models.User(is_guest=False, email="length@example.com")
|
||||||
|
db.add(user)
|
||||||
|
db.flush()
|
||||||
|
settings = models.Settings(user_id=user.id, api_key="enc:dummy", model="m")
|
||||||
|
db.add(settings)
|
||||||
|
scenario = models.Scenario(user_id=user.id, title="S", prompt="A road.")
|
||||||
|
db.add(scenario)
|
||||||
|
db.flush()
|
||||||
|
adventure = models.Adventure(
|
||||||
|
user_id=user.id, title="A", scenario_id=scenario.id, script_state={},
|
||||||
|
memory="The hero hunts bandits.",
|
||||||
|
)
|
||||||
|
db.add(adventure)
|
||||||
|
db.flush()
|
||||||
|
for i in range(4):
|
||||||
|
db.add(models.Action(adventure_id=adventure.id, index=i,
|
||||||
|
type="ai" if i % 2 else "do", text=f"[{i}] Onward."))
|
||||||
|
db.commit()
|
||||||
|
db.expire_all()
|
||||||
|
adventure = db.get(models.Adventure, adventure.id)
|
||||||
|
settings = db.get(models.Settings, settings.id)
|
||||||
|
try:
|
||||||
|
yield db, adventure, settings, scenario.id
|
||||||
|
finally:
|
||||||
|
db.close()
|
||||||
|
Base.metadata.drop_all(bind=engine)
|
||||||
|
|
||||||
|
|
||||||
|
def with_schema(db, scenario_id, adventure):
|
||||||
|
scenario = db.get(models.Scenario, scenario_id)
|
||||||
|
scenario.stat_schema = SCHEMA
|
||||||
|
adventure.world_state = worldstate.instantiate(SCHEMA)
|
||||||
|
db.commit()
|
||||||
|
db.expire_all()
|
||||||
|
|
||||||
|
|
||||||
|
# ----------------------------------------------------- the hint itself
|
||||||
|
|
||||||
|
def test_budget_is_the_cap_minus_headroom_and_buffer_in_words():
|
||||||
|
"""800-token cap → 750 after headroom → ~562 words → 506 after the buffer."""
|
||||||
|
hint = builder.length_hint(800, has_ws=True)
|
||||||
|
assert "506" in hint
|
||||||
|
assert "token" not in hint.lower(), "a model cannot count its own tokens"
|
||||||
|
|
||||||
|
|
||||||
|
def test_budget_tracks_the_setting():
|
||||||
|
small = builder.length_hint(800, has_ws=True)
|
||||||
|
large = builder.length_hint(2400, has_ws=True)
|
||||||
|
assert small != large
|
||||||
|
assert "1586" in large
|
||||||
|
|
||||||
|
|
||||||
|
def asked_words(cap):
|
||||||
|
return int(re.search(r"(\d+) words", builder.length_hint(cap, has_ws=True)).group(1))
|
||||||
|
|
||||||
|
|
||||||
|
def test_buffer_leaves_room_for_overshoot():
|
||||||
|
"""The stated number must sit meaningfully under the real ceiling, or an
|
||||||
|
on-target-but-slightly-long turn still hits the wall."""
|
||||||
|
for cap in (400, 800, 1500, 2400):
|
||||||
|
asked = asked_words(cap)
|
||||||
|
ceiling = (cap - builder.LENGTH_HEADROOM) * builder.WORDS_PER_TOKEN
|
||||||
|
assert asked < ceiling
|
||||||
|
assert asked >= ceiling * 0.85, "buffer so large the hint wastes the cap"
|
||||||
|
|
||||||
|
|
||||||
|
def test_hint_is_phrased_as_a_ceiling_not_a_budget():
|
||||||
|
"""Measured: budget phrasing ("keep this turn under about N words") reads as a
|
||||||
|
target to fill and moved the mean turn from 174 to 246 words — toward the wall
|
||||||
|
it exists to avoid. The limit framing must survive future prompt edits."""
|
||||||
|
hint = builder.length_hint(800, has_ws=True)
|
||||||
|
assert "must not exceed" in hint
|
||||||
|
assert "much shorter" in hint, "without this the number still reads as a target"
|
||||||
|
assert "under about" not in hint
|
||||||
|
|
||||||
|
|
||||||
|
def test_no_hint_when_the_cap_is_too_small_to_phrase():
|
||||||
|
"""Under the floor the hint is noise the model pays for in context."""
|
||||||
|
assert builder.length_hint(100, has_ws=True) == ""
|
||||||
|
assert builder.length_hint(builder.LENGTH_HEADROOM, has_ws=True) == ""
|
||||||
|
assert builder.length_hint(0, has_ws=True) == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_no_negative_word_budget():
|
||||||
|
"""A cap below the headroom must not ask for a negative number of words."""
|
||||||
|
for cap in (1, 10, 49, 51):
|
||||||
|
assert builder.length_hint(cap, has_ws=True) == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_reason_given_matches_whether_state_is_tracked():
|
||||||
|
assert "state block" in builder.length_hint(800, has_ws=True)
|
||||||
|
assert "state block" not in builder.length_hint(800, has_ws=False)
|
||||||
|
|
||||||
|
|
||||||
|
# ----------------------------------------------------- in the assembled prompt
|
||||||
|
|
||||||
|
def test_hint_reaches_the_story_prompt(story):
|
||||||
|
db, adventure, settings, _ = story
|
||||||
|
settings.max_output_tokens = 800
|
||||||
|
_, story_text, report = builder.build_context(adventure, settings)
|
||||||
|
|
||||||
|
assert "506" in story_text
|
||||||
|
labels = [s["label"] for s in report["sections"]]
|
||||||
|
assert "length_hint" in labels
|
||||||
|
|
||||||
|
|
||||||
|
def test_emit_reminder_keeps_the_last_word(story):
|
||||||
|
"""The hint sits above the emit reminder: the reminder's whole value is the
|
||||||
|
recency slot, and the model has to write the narration before the block."""
|
||||||
|
db, adventure, settings, scenario_id = story
|
||||||
|
with_schema(db, scenario_id, adventure)
|
||||||
|
settings.max_output_tokens = 800
|
||||||
|
|
||||||
|
_, story_text, report = builder.build_context(adventure, settings)
|
||||||
|
|
||||||
|
assert story_text.rstrip().endswith(worldstate.EMIT_REMINDER.rstrip())
|
||||||
|
labels = [s["label"] for s in report["sections"]]
|
||||||
|
assert labels.index("length_hint") < labels.index("world_state_reminder")
|
||||||
|
|
||||||
|
|
||||||
|
def test_prompt_stays_inside_the_budget_on_a_long_story(story):
|
||||||
|
"""Regression guard: the hint is appended after history has already spent
|
||||||
|
the budget, so it must be reserved up front like EMIT_REMINDER is.
|
||||||
|
|
||||||
|
Weak on purpose — the history loop stops *before* crossing its budget, so it
|
||||||
|
leaves about one action of slack and the ~30-token hint hides inside it.
|
||||||
|
This catches a hint that grows large, not a missing reservation; the
|
||||||
|
reservation itself is not observable from the outside."""
|
||||||
|
db, adventure, settings, _ = story
|
||||||
|
for i in range(4, 120):
|
||||||
|
db.add(models.Action(
|
||||||
|
adventure_id=adventure.id, index=i, type="ai" if i % 2 else "do",
|
||||||
|
text=f"[{i}] " + "The road bends past the burnt mill and the smoke. " * 12,
|
||||||
|
))
|
||||||
|
db.commit()
|
||||||
|
db.expire_all()
|
||||||
|
adventure = db.get(models.Adventure, adventure.id)
|
||||||
|
|
||||||
|
settings.max_output_tokens = 2400
|
||||||
|
settings.context_token_budget = 2048
|
||||||
|
|
||||||
|
_, _, report = builder.build_context(adventure, settings)
|
||||||
|
assert report["history"]["included"] < 120, "budget was never actually filled"
|
||||||
|
assert report["tokens"]["total"] <= report["tokens"]["budget"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_hint_is_counted_in_the_reported_totals(story):
|
||||||
|
"""Insights reports what the turn actually costs; a section that reaches the
|
||||||
|
model but not the accounting makes that number a lie."""
|
||||||
|
db, adventure, settings, _ = story
|
||||||
|
settings.max_output_tokens = 800
|
||||||
|
|
||||||
|
_, _, report = builder.build_context(adventure, settings)
|
||||||
|
hint = next(s for s in report["sections"] if s["label"] == "length_hint")
|
||||||
|
assert hint["tokens"] > 0
|
||||||
|
assert hint["text"] in report["prompt"]["story"]
|
||||||
|
assert builder.count_tokens(report["prompt"]["story"]) <= report["tokens"]["total"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_no_hint_section_when_the_cap_is_tiny(story):
|
||||||
|
"""An empty hint drops out entirely rather than leaving a blank section."""
|
||||||
|
db, adventure, settings, _ = story
|
||||||
|
settings.max_output_tokens = 60
|
||||||
|
|
||||||
|
_, story_text, report = builder.build_context(adventure, settings)
|
||||||
|
assert "length_hint" not in [s["label"] for s in report["sections"]]
|
||||||
|
assert "Keep this turn" not in story_text
|
||||||
Reference in New Issue
Block a user