From d66fd1d6a2a312feb8eb2c4a272475f7236a8537 Mon Sep 17 00:00:00 2001 From: parththakkar106 Date: Sun, 2 Aug 2026 11:10:21 +0530 Subject: [PATCH] Add a reasoning-off setting (-1 reasoning budget) Models like DeepSeek V4 Flash reason by default, and the reasoning budget setting could only ever add thinking tokens - there was no value that turned thinking off. A negative budget now sends `reasoning: {effort: "none"}`. Uses effort:none rather than exclude:true deliberately - exclude still thinks and still bills, it only hides the trace. Zero keeps its old meaning (send no `reasoning` field at all) so endpoints that reject unknown fields, like the default Ollama one, are unaffected. Reusing the existing int column this way avoids a migration. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01UeQVy5bEjLhfgWNc27Efet --- backend/app/models.py | 3 +- backend/app/providers/openai_compatible.py | 17 ++++++-- backend/app/schemas.py | 3 +- backend/tests/test_reasoning_param.py | 45 ++++++++++++++++++++++ frontend/src/pages/Settings.jsx | 11 ++++-- 5 files changed, 70 insertions(+), 9 deletions(-) create mode 100644 backend/tests/test_reasoning_param.py diff --git a/backend/app/models.py b/backend/app/models.py index f27dfec..8634d96 100644 --- a/backend/app/models.py +++ b/backend/app/models.py @@ -307,7 +307,8 @@ class Settings(Base): # and left reasoning models with nothing after their thinking. max_output_tokens: Mapped[int] = mapped_column(Integer, default=800) # Separate thinking budget for reasoning models (OpenRouter-style - # `reasoning: {max_tokens}`); 0 = param not sent. Added on top of + # `reasoning: {max_tokens}`); 0 = param not sent, -1 = reasoning explicitly + # off (`reasoning: {effort: none}`). Added on top of # max_output_tokens so story output keeps its full budget. reasoning_max_tokens: Mapped[int] = mapped_column(Integer, default=0) context_token_budget: Mapped[int] = mapped_column(Integer, default=16384) diff --git a/backend/app/providers/openai_compatible.py b/backend/app/providers/openai_compatible.py index 2c2c3dd..3dc28c4 100644 --- a/backend/app/providers/openai_compatible.py +++ b/backend/app/providers/openai_compatible.py @@ -39,7 +39,8 @@ class OpenAICompatibleProvider(Provider): self.api_mode = api_mode # "chat" | "completion" # Thinking budget for reasoning models, on top of max_tokens. 0 = the # `reasoning` param is not sent (endpoints that don't know it may - # reject unknown fields). + # reject unknown fields); negative = explicitly ask the endpoint to + # turn reasoning off. self.reasoning_max_tokens = reasoning_max_tokens def _headers(self) -> dict: @@ -50,8 +51,18 @@ class OpenAICompatibleProvider(Provider): def _apply_reasoning_budget(self, body: dict) -> None: """Give reasoning models their own thinking budget (OpenRouter-style), - raising max_tokens so the actual output keeps its full budget.""" - if self.reasoning_max_tokens > 0 and self.api_mode == "chat": + raising max_tokens so the actual output keeps its full budget. + + A negative budget means the opposite: send `effort: "none"` to switch + reasoning off on models that do it by default (DeepSeek V4 Flash, say). + That's distinct from `exclude: true`, which still thinks — and bills — + but hides the trace. Zero stays "send nothing at all" so endpoints that + reject unknown fields (Ollama) keep working.""" + if self.api_mode != "chat": + return + if self.reasoning_max_tokens < 0: + body["reasoning"] = {"effort": "none"} + elif self.reasoning_max_tokens > 0: body["reasoning"] = {"max_tokens": self.reasoning_max_tokens} body["max_tokens"] += self.reasoning_max_tokens diff --git a/backend/app/schemas.py b/backend/app/schemas.py index 186d1e0..02e3009 100644 --- a/backend/app/schemas.py +++ b/backend/app/schemas.py @@ -375,7 +375,8 @@ class SettingsUpdate(BaseModel): api_mode: Annotated[str, Field(max_length=20)] | None = None temperature: Annotated[float, Field(ge=0, le=5)] | None = None max_output_tokens: Annotated[int, Field(ge=1, le=100_000)] | None = None - reasoning_max_tokens: Annotated[int, Field(ge=0, le=100_000)] | None = None + # -1 = explicitly off (sends `reasoning: {effort: none}`); 0 = send nothing. + reasoning_max_tokens: Annotated[int, Field(ge=-1, le=100_000)] | None = None context_token_budget: Annotated[int, Field(ge=256, le=200_000)] | None = None narrator_prompt: Prose | None = None stream: bool | None = None diff --git a/backend/tests/test_reasoning_param.py b/backend/tests/test_reasoning_param.py new file mode 100644 index 0000000..4f461be --- /dev/null +++ b/backend/tests/test_reasoning_param.py @@ -0,0 +1,45 @@ +"""What the provider puts in the `reasoning` request field for each budget +setting: a positive budget asks for thinking, 0 stays silent, -1 turns it off. + + python -m pytest tests/test_reasoning_param.py -v +""" +from app.providers.openai_compatible import OpenAICompatibleProvider + + +def _body(reasoning_max_tokens, api_mode="chat", max_tokens=1000): + provider = OpenAICompatibleProvider( + "https://openrouter.ai/api/v1", "k", "deepseek/deepseek-v4-flash-0731", + api_mode, reasoning_max_tokens, + ) + body = {"max_tokens": max_tokens} + provider._apply_reasoning_budget(body) + return body + + +def test_zero_sends_nothing(): + """Ollama and friends reject unknown fields — 0 must stay silent.""" + assert "reasoning" not in _body(0) + + +def test_positive_budget_adds_thinking_tokens(): + body = _body(500) + assert body["reasoning"] == {"max_tokens": 500} + # the story output keeps its own full budget on top of the thinking budget + assert body["max_tokens"] == 1500 + + +def test_negative_turns_reasoning_off(): + body = _body(-1) + assert body["reasoning"] == {"effort": "none"} + # "off" must not inflate the output budget + assert body["max_tokens"] == 1000 + + +def test_off_is_not_merely_excluded(): + """`exclude: true` still thinks and still bills; we want it actually off.""" + assert _body(-1)["reasoning"].get("exclude") is None + + +def test_completion_mode_never_sends_reasoning(): + for budget in (-1, 0, 500): + assert "reasoning" not in _body(budget, api_mode="completion") diff --git a/frontend/src/pages/Settings.jsx b/frontend/src/pages/Settings.jsx index 4ba667b..58a2164 100644 --- a/frontend/src/pages/Settings.jsx +++ b/frontend/src/pages/Settings.jsx @@ -167,12 +167,15 @@ export default function Settings() {