Add a reasoning-off setting (-1 reasoning budget)
Models like DeepSeek V4 Flash reason by default, and the reasoning budget
setting could only ever add thinking tokens - there was no value that turned
thinking off. A negative budget now sends `reasoning: {effort: "none"}`.
Uses effort:none rather than exclude:true deliberately - exclude still thinks
and still bills, it only hides the trace.
Zero keeps its old meaning (send no `reasoning` field at all) so endpoints that
reject unknown fields, like the default Ollama one, are unaffected. Reusing the
existing int column this way avoids a migration.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01UeQVy5bEjLhfgWNc27Efet
This commit is contained in:
co-authored by
Claude Opus 5
parent
aed28d9295
commit
d66fd1d6a2
@@ -307,7 +307,8 @@ class Settings(Base):
|
|||||||
# and left reasoning models with nothing after their thinking.
|
# and left reasoning models with nothing after their thinking.
|
||||||
max_output_tokens: Mapped[int] = mapped_column(Integer, default=800)
|
max_output_tokens: Mapped[int] = mapped_column(Integer, default=800)
|
||||||
# Separate thinking budget for reasoning models (OpenRouter-style
|
# Separate thinking budget for reasoning models (OpenRouter-style
|
||||||
# `reasoning: {max_tokens}`); 0 = param not sent. Added on top of
|
# `reasoning: {max_tokens}`); 0 = param not sent, -1 = reasoning explicitly
|
||||||
|
# off (`reasoning: {effort: none}`). Added on top of
|
||||||
# max_output_tokens so story output keeps its full budget.
|
# max_output_tokens so story output keeps its full budget.
|
||||||
reasoning_max_tokens: Mapped[int] = mapped_column(Integer, default=0)
|
reasoning_max_tokens: Mapped[int] = mapped_column(Integer, default=0)
|
||||||
context_token_budget: Mapped[int] = mapped_column(Integer, default=16384)
|
context_token_budget: Mapped[int] = mapped_column(Integer, default=16384)
|
||||||
|
|||||||
@@ -39,7 +39,8 @@ class OpenAICompatibleProvider(Provider):
|
|||||||
self.api_mode = api_mode # "chat" | "completion"
|
self.api_mode = api_mode # "chat" | "completion"
|
||||||
# Thinking budget for reasoning models, on top of max_tokens. 0 = the
|
# Thinking budget for reasoning models, on top of max_tokens. 0 = the
|
||||||
# `reasoning` param is not sent (endpoints that don't know it may
|
# `reasoning` param is not sent (endpoints that don't know it may
|
||||||
# reject unknown fields).
|
# reject unknown fields); negative = explicitly ask the endpoint to
|
||||||
|
# turn reasoning off.
|
||||||
self.reasoning_max_tokens = reasoning_max_tokens
|
self.reasoning_max_tokens = reasoning_max_tokens
|
||||||
|
|
||||||
def _headers(self) -> dict:
|
def _headers(self) -> dict:
|
||||||
@@ -50,8 +51,18 @@ class OpenAICompatibleProvider(Provider):
|
|||||||
|
|
||||||
def _apply_reasoning_budget(self, body: dict) -> None:
|
def _apply_reasoning_budget(self, body: dict) -> None:
|
||||||
"""Give reasoning models their own thinking budget (OpenRouter-style),
|
"""Give reasoning models their own thinking budget (OpenRouter-style),
|
||||||
raising max_tokens so the actual output keeps its full budget."""
|
raising max_tokens so the actual output keeps its full budget.
|
||||||
if self.reasoning_max_tokens > 0 and self.api_mode == "chat":
|
|
||||||
|
A negative budget means the opposite: send `effort: "none"` to switch
|
||||||
|
reasoning off on models that do it by default (DeepSeek V4 Flash, say).
|
||||||
|
That's distinct from `exclude: true`, which still thinks — and bills —
|
||||||
|
but hides the trace. Zero stays "send nothing at all" so endpoints that
|
||||||
|
reject unknown fields (Ollama) keep working."""
|
||||||
|
if self.api_mode != "chat":
|
||||||
|
return
|
||||||
|
if self.reasoning_max_tokens < 0:
|
||||||
|
body["reasoning"] = {"effort": "none"}
|
||||||
|
elif self.reasoning_max_tokens > 0:
|
||||||
body["reasoning"] = {"max_tokens": self.reasoning_max_tokens}
|
body["reasoning"] = {"max_tokens": self.reasoning_max_tokens}
|
||||||
body["max_tokens"] += self.reasoning_max_tokens
|
body["max_tokens"] += self.reasoning_max_tokens
|
||||||
|
|
||||||
|
|||||||
@@ -375,7 +375,8 @@ class SettingsUpdate(BaseModel):
|
|||||||
api_mode: Annotated[str, Field(max_length=20)] | None = None
|
api_mode: Annotated[str, Field(max_length=20)] | None = None
|
||||||
temperature: Annotated[float, Field(ge=0, le=5)] | None = None
|
temperature: Annotated[float, Field(ge=0, le=5)] | None = None
|
||||||
max_output_tokens: Annotated[int, Field(ge=1, le=100_000)] | None = None
|
max_output_tokens: Annotated[int, Field(ge=1, le=100_000)] | None = None
|
||||||
reasoning_max_tokens: Annotated[int, Field(ge=0, le=100_000)] | None = None
|
# -1 = explicitly off (sends `reasoning: {effort: none}`); 0 = send nothing.
|
||||||
|
reasoning_max_tokens: Annotated[int, Field(ge=-1, le=100_000)] | None = None
|
||||||
context_token_budget: Annotated[int, Field(ge=256, le=200_000)] | None = None
|
context_token_budget: Annotated[int, Field(ge=256, le=200_000)] | None = None
|
||||||
narrator_prompt: Prose | None = None
|
narrator_prompt: Prose | None = None
|
||||||
stream: bool | None = None
|
stream: bool | None = None
|
||||||
|
|||||||
@@ -0,0 +1,45 @@
|
|||||||
|
"""What the provider puts in the `reasoning` request field for each budget
|
||||||
|
setting: a positive budget asks for thinking, 0 stays silent, -1 turns it off.
|
||||||
|
|
||||||
|
python -m pytest tests/test_reasoning_param.py -v
|
||||||
|
"""
|
||||||
|
from app.providers.openai_compatible import OpenAICompatibleProvider
|
||||||
|
|
||||||
|
|
||||||
|
def _body(reasoning_max_tokens, api_mode="chat", max_tokens=1000):
|
||||||
|
provider = OpenAICompatibleProvider(
|
||||||
|
"https://openrouter.ai/api/v1", "k", "deepseek/deepseek-v4-flash-0731",
|
||||||
|
api_mode, reasoning_max_tokens,
|
||||||
|
)
|
||||||
|
body = {"max_tokens": max_tokens}
|
||||||
|
provider._apply_reasoning_budget(body)
|
||||||
|
return body
|
||||||
|
|
||||||
|
|
||||||
|
def test_zero_sends_nothing():
|
||||||
|
"""Ollama and friends reject unknown fields — 0 must stay silent."""
|
||||||
|
assert "reasoning" not in _body(0)
|
||||||
|
|
||||||
|
|
||||||
|
def test_positive_budget_adds_thinking_tokens():
|
||||||
|
body = _body(500)
|
||||||
|
assert body["reasoning"] == {"max_tokens": 500}
|
||||||
|
# the story output keeps its own full budget on top of the thinking budget
|
||||||
|
assert body["max_tokens"] == 1500
|
||||||
|
|
||||||
|
|
||||||
|
def test_negative_turns_reasoning_off():
|
||||||
|
body = _body(-1)
|
||||||
|
assert body["reasoning"] == {"effort": "none"}
|
||||||
|
# "off" must not inflate the output budget
|
||||||
|
assert body["max_tokens"] == 1000
|
||||||
|
|
||||||
|
|
||||||
|
def test_off_is_not_merely_excluded():
|
||||||
|
"""`exclude: true` still thinks and still bills; we want it actually off."""
|
||||||
|
assert _body(-1)["reasoning"].get("exclude") is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_completion_mode_never_sends_reasoning():
|
||||||
|
for budget in (-1, 0, 500):
|
||||||
|
assert "reasoning" not in _body(budget, api_mode="completion")
|
||||||
@@ -167,12 +167,15 @@ export default function Settings() {
|
|||||||
</div>
|
</div>
|
||||||
<label className="field">
|
<label className="field">
|
||||||
<span className="label">Reasoning budget (tokens)</span>
|
<span className="label">Reasoning budget (tokens)</span>
|
||||||
<input type="number" min="0" value={settings.reasoning_max_tokens}
|
<input type="number" min="-1" value={settings.reasoning_max_tokens}
|
||||||
onChange={(e) => setField('reasoning_max_tokens', Number(e.target.value))} />
|
onChange={(e) => setField(
|
||||||
|
'reasoning_max_tokens', Math.max(-1, Number(e.target.value)))} />
|
||||||
<span className="label" style={{ marginTop: 4 }}>
|
<span className="label" style={{ marginTop: 4 }}>
|
||||||
For reasoning models: separate thinking budget on top of max output tokens,
|
For reasoning models: separate thinking budget on top of max output tokens,
|
||||||
and thinking is shown collapsed above each response. 0 = off (nothing extra
|
and thinking is shown collapsed above each response. 0 = nothing extra is
|
||||||
is sent — keep 0 for endpoints/models without reasoning support).
|
sent (keep 0 for endpoints without reasoning support, e.g. Ollama).
|
||||||
|
−1 = actively turn reasoning off, for models that think by default
|
||||||
|
(DeepSeek V4 Flash) — saves the thinking tokens rather than just hiding them.
|
||||||
</span>
|
</span>
|
||||||
</label>
|
</label>
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user