Add a reasoning-off setting (-1 reasoning budget)
Models like DeepSeek V4 Flash reason by default, and the reasoning budget
setting could only ever add thinking tokens - there was no value that turned
thinking off. A negative budget now sends `reasoning: {effort: "none"}`.
Uses effort:none rather than exclude:true deliberately - exclude still thinks
and still bills, it only hides the trace.
Zero keeps its old meaning (send no `reasoning` field at all) so endpoints that
reject unknown fields, like the default Ollama one, are unaffected. Reusing the
existing int column this way avoids a migration.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01UeQVy5bEjLhfgWNc27Efet
This commit is contained in:
co-authored by
Claude Opus 5
parent
aed28d9295
commit
d66fd1d6a2
@@ -0,0 +1,45 @@
|
||||
"""What the provider puts in the `reasoning` request field for each budget
|
||||
setting: a positive budget asks for thinking, 0 stays silent, -1 turns it off.
|
||||
|
||||
python -m pytest tests/test_reasoning_param.py -v
|
||||
"""
|
||||
from app.providers.openai_compatible import OpenAICompatibleProvider
|
||||
|
||||
|
||||
def _body(reasoning_max_tokens, api_mode="chat", max_tokens=1000):
|
||||
provider = OpenAICompatibleProvider(
|
||||
"https://openrouter.ai/api/v1", "k", "deepseek/deepseek-v4-flash-0731",
|
||||
api_mode, reasoning_max_tokens,
|
||||
)
|
||||
body = {"max_tokens": max_tokens}
|
||||
provider._apply_reasoning_budget(body)
|
||||
return body
|
||||
|
||||
|
||||
def test_zero_sends_nothing():
|
||||
"""Ollama and friends reject unknown fields — 0 must stay silent."""
|
||||
assert "reasoning" not in _body(0)
|
||||
|
||||
|
||||
def test_positive_budget_adds_thinking_tokens():
|
||||
body = _body(500)
|
||||
assert body["reasoning"] == {"max_tokens": 500}
|
||||
# the story output keeps its own full budget on top of the thinking budget
|
||||
assert body["max_tokens"] == 1500
|
||||
|
||||
|
||||
def test_negative_turns_reasoning_off():
|
||||
body = _body(-1)
|
||||
assert body["reasoning"] == {"effort": "none"}
|
||||
# "off" must not inflate the output budget
|
||||
assert body["max_tokens"] == 1000
|
||||
|
||||
|
||||
def test_off_is_not_merely_excluded():
|
||||
"""`exclude: true` still thinks and still bills; we want it actually off."""
|
||||
assert _body(-1)["reasoning"].get("exclude") is None
|
||||
|
||||
|
||||
def test_completion_mode_never_sends_reasoning():
|
||||
for budget in (-1, 0, 500):
|
||||
assert "reasoning" not in _body(budget, api_mode="completion")
|
||||
Reference in New Issue
Block a user