From 3b9e6b3d50a4e472233654b93e7c12188adff96c Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Fri, 21 Aug 2026 00:14:56 +0530 Subject: [PATCH] Give the length hint a floor, not just a wall A ceiling alone is a one-sided instruction, and models read it in opposite directions. A verbose one is held back by it; a terse one has nothing to act on except "write only as much as the moment needs -- a typical turn is much shorter" and collapses to two paragraphs. Same prompt, wildly different turn lengths depending on which model is behind it. State a floor as well, so the guidance is a band. The two bounds are deliberately asymmetric -- "must not exceed" for the wall the endpoint enforces, "should not stop short of" for the floor -- so neither reads as a number to hit, which is the property the earlier A/B says decides whether this hint helps or hurts. "Prefer the lower end" inherits the anti-overshoot job the deleted "much shorter" line was doing, but now with a number under it, so a terse model lands on the floor instead of at forty words. Below MIN_LENGTH_FLOOR_WORDS the floor is dropped and the tight-cap wording is left byte-identical: at a tight cap a short turn is the correct turn, and that phrasing is the one measured to keep the state block alive (0/6 truncations at cap 250 against 2/6 unhinted). So this only moves loose caps. MAX_LENGTH_FLOOR_WORDS keeps the share from demanding 555 words minimum at cap 2400 -- a big cap means long turns are allowed, not compulsory. Shipped without an A/B run, deliberately. Two things to watch live: whether a stated range invites landing mid-range on verbose models (drop the share to ~0.25 if so), and whether the state block still survives -- nothing reads finish_reason yet, so truncation is silent. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01DfMCsN1KBLsTqMkj5hSgrY --- backend/app/context/builder.py | 33 +++++++++++++++++++++++++-- backend/tests/test_length_hint.py | 38 ++++++++++++++++++++++++++++++- 2 files changed, 68 insertions(+), 3 deletions(-) diff --git a/backend/app/context/builder.py b/backend/app/context/builder.py index 7078904..75fb1eb 100644 --- a/backend/app/context/builder.py +++ b/backend/app/context/builder.py @@ -37,6 +37,21 @@ WORDS_PER_TOKEN = 0.75 # overshoot land in the slack instead of in the state block. LENGTH_BUFFER = 0.90 MIN_LENGTH_HINT_WORDS = 40 # below this the hint is noise; a tiny cap speaks for itself +# A ceiling alone is a one-sided instruction, and models read it very differently: +# a verbose one is held back by it, while a terse one has nothing to act on except +# the "write only as much as the moment needs" clause and collapses to two +# paragraphs. Stating a floor as well turns the guidance into a band, so the same +# prompt lands in the same place regardless of which way the model leans. Set as a +# share of the ceiling so the floor can never approach it. +LENGTH_FLOOR_SHARE = 0.35 +# Below this a floor is meaningless — at a tight cap a short turn is the correct +# turn — and the tight-cap wording is the one measured to keep the state block +# alive, so it is left exactly as it was. +MIN_LENGTH_FLOOR_WORDS = 60 +# The floor exists to stop a collapse to two paragraphs, not to demand an essay: +# at a 2400-token cap the share alone would ask for 555 words *minimum*. Past this +# point a reader wanting more length can say so in the author's note. +MAX_LENGTH_FLOOR_WORDS = 300 @functools.lru_cache(maxsize=1) @@ -87,9 +102,23 @@ def length_hint(max_output_tokens: int, *, has_ws: bool) -> str: # pushed turns toward the very wall it exists to keep them away from. Naming # the number as a limit, plus saying a typical turn is far shorter, left the # average at 170 while still rescuing the state block at tight caps. + floor = min(int(words * LENGTH_FLOOR_SHARE), MAX_LENGTH_FLOOR_WORDS) + if floor < MIN_LENGTH_FLOOR_WORDS: + return ( + f"[Hard limit: this turn must not exceed {words} words. Write only as " + f"much as the moment needs — a typical turn is much shorter.{tail}]" + ) + # Both numbers are bounds, and deliberately asymmetric ones: "must not exceed" + # for the wall the endpoint enforces, "should not stop short of" for the floor. + # Neither is a target, which is what the measurement above says matters. The + # "prefer the lower end" clause does the job the old "a typical turn is much + # shorter" line did — holding a verbose model off the wall — but now with a + # number under it, so a terse model reading the same clause lands on the floor + # instead of at forty words. return ( - f"[Hard limit: this turn must not exceed {words} words. Write only as " - f"much as the moment needs — a typical turn is much shorter.{tail}]" + f"[Hard limit: this turn must not exceed {words} words, and it should not " + f"stop short of about {floor}. Prefer the lower end of that range unless " + f"the scene genuinely needs more.{tail}]" ) diff --git a/backend/tests/test_length_hint.py b/backend/tests/test_length_hint.py index 71fff5c..b3faafa 100644 --- a/backend/tests/test_length_hint.py +++ b/backend/tests/test_length_hint.py @@ -114,8 +114,44 @@ def test_hint_is_phrased_as_a_ceiling_not_a_budget(): it exists to avoid. The limit framing must survive future prompt edits.""" hint = builder.length_hint(800, has_ws=True) assert "must not exceed" in hint - assert "much shorter" in hint, "without this the number still reads as a target" assert "under about" not in hint + assert "lower end" in hint, "without this the number still reads as a target" + + +def test_hint_states_a_floor_as_well_as_a_ceiling(): + """A ceiling alone is one-sided: a terse model has nothing to act on but the + "only as much as the moment needs" clause and collapses to two paragraphs. + The floor is what makes the same prompt land in the same place across models + that lean opposite ways.""" + hint = builder.length_hint(800, has_ws=True) + assert "506" in hint and "177" in hint + assert "should not stop short of" in hint + # Asymmetric on purpose: the wall is a wall, the floor is a floor, and neither + # is phrased as a number to hit. + assert hint.index("must not exceed") < hint.index("should not stop short of") + + +def test_floor_stays_well_under_the_ceiling(): + for cap in (400, 800, 1500, 2400): + hint = builder.length_hint(cap, has_ws=True) + ceiling, floor = (int(n) for n in re.findall(r"(\d+)", hint)[:2]) + assert floor < ceiling * 0.5 + + +def test_floor_is_dropped_when_the_cap_is_too_tight_for_one(): + """At a tight cap a short turn is the correct turn, and the tight-cap wording + is the one measured to keep the state block alive (0/6 truncations at cap 250 + against 2/6 unhinted) — so it is left exactly as it was.""" + hint = builder.length_hint(250, has_ws=True) + assert "should not stop short of" not in hint + assert "much shorter" in hint + + +def test_floor_does_not_grow_without_bound(): + """A big cap means "long turns are allowed", not "every turn must be an essay": + the share alone would demand 555 words minimum at cap 2400.""" + hint = builder.length_hint(2400, has_ws=True) + assert str(builder.MAX_LENGTH_FLOOR_WORDS) in hint def test_no_hint_when_the_cap_is_too_small_to_phrase():