From 930dd17086a1b75317f9e1d20de39a134cbb82fe Mon Sep 17 00:00:00 2001 From: maattm <139474836+maattm@users.noreply.github.com> Date: Mon, 15 Jun 2026 10:35:33 +0100 Subject: [PATCH] feat: add Anthropic-compatible thinking parameter (#5856) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat: add Anthropic-compatible thinking parameter Add `thinking` parameter using Anthropic's format ({type: 'disabled'} / {type: 'enabled'}) alongside the existing `enable_thinking` boolean for backward compatibility. The new parameter is mapped internally to `enable_thinking` at the route layer so all downstream templates and backends continue to work unchanged. Changes: - Add ThinkingConfig model and `thinking` field to ChatCompletionRequest - Add mapping logic in routes: thinking.type -> enable_thinking - Add `thinking` field to frontend TypeScript types - Update frontend request building to send thinking parameter - Add tests for new thinking parameter * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * fix: move thinking→enable_thinking mapping to model_validator The Gemini review correctly identified that the route-level mapping bypasses normalization for external provider requests. Moving the mapping into a @model_validator on ChatCompletionRequest ensures it runs during Pydantic validation regardless of routing path. * Document ThinkingConfig scope and thinking validation behavior --------- Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> Co-authored-by: Daniel Han Co-authored-by: Lee Jackson <130007945+Imagineer99@users.noreply.github.com> --- studio/backend/models/inference.py | 29 +++++ .../backend/tests/test_thinking_parameter.py | 106 ++++++++++++++++++ .../src/features/chat/api/chat-adapter.ts | 4 +- .../frontend/src/features/chat/types/api.ts | 1 + 4 files changed, 138 insertions(+), 2 deletions(-) create mode 100644 studio/backend/tests/test_thinking_parameter.py diff --git a/studio/backend/models/inference.py b/studio/backend/models/inference.py index 4e6a5937d8..5b68364687 100644 --- a/studio/backend/models/inference.py +++ b/studio/backend/models/inference.py @@ -581,6 +581,16 @@ class ChatMessage(BaseModel): return self +class ThinkingConfig(BaseModel): + """Anthropic-compatible thinking/reasoning configuration. + Use type='disabled' to turn off thinking, or type='enabled' to turn it on. + Only type is read; extra fields (e.g. budget_tokens) are ignored, since + Studio sets provider thinking budgets itself. + """ + + type: Literal["disabled", "enabled"] = "disabled" + + class ChatCompletionRequest(BaseModel): """OpenAI-compatible chat completion request. @@ -694,6 +704,11 @@ class ChatCompletionRequest(BaseModel): None, description = "[x-unsloth] When true, keep historical blocks from past assistant turns in the prompt (Qwen3.6 templates). Independent of enable_thinking / reasoning_effort.", ) + thinking: Optional[ThinkingConfig] = Field( + None, + description = "[Anthropic-compatible] Thinking configuration. " + "Use {type: 'disabled'} to disable thinking, {type: 'enabled'} to enable.", + ) enable_tools: Optional[bool] = Field( None, description = "[x-unsloth] Enable tool calling for supported models", @@ -952,6 +967,20 @@ class ChatCompletionRequest(BaseModel): msg.tool_call_id = picked return self + @model_validator(mode = "after") + def _map_thinking_to_enable_thinking(self) -> "ChatCompletionRequest": + """Map Anthropic-style ``thinking`` parameter to internal ``enable_thinking``. + + ``thinking: {type: 'enabled'}`` sets ``enable_thinking = True`` and + ``thinking: {type: 'disabled'}`` sets ``enable_thinking = False``. + ``enable_thinking`` takes precedence when both are provided so that + callers who already use the internal field are unaffected. Invalid + ``thinking`` shapes are rejected at validation time (422). + """ + if self.thinking is not None and self.enable_thinking is None: + self.enable_thinking = self.thinking.type == "enabled" + return self + class ToolConfirmRequest(BaseModel): session_id: Optional[str] = None diff --git a/studio/backend/tests/test_thinking_parameter.py b/studio/backend/tests/test_thinking_parameter.py new file mode 100644 index 0000000000..92e08912cb --- /dev/null +++ b/studio/backend/tests/test_thinking_parameter.py @@ -0,0 +1,106 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 + +""" +Unit tests for the Anthropic-compatible thinking parameter. + +Covers: +- ThinkingConfig model validation +- ChatCompletionRequest with thinking parameter +- Mapping logic: thinking.type -> enable_thinking +""" + +import os +import sys + +_backend = os.path.join(os.path.dirname(__file__), "..") +sys.path.insert(0, _backend) + +from models.inference import ChatCompletionRequest, ThinkingConfig + + +def test_thinking_config_defaults_to_disabled(): + """ThinkingConfig should default to type='disabled'.""" + config = ThinkingConfig() + assert config.type == "disabled" + + +def test_thinking_config_explicit_disabled(): + """ThinkingConfig should accept type='disabled'.""" + config = ThinkingConfig(type = "disabled") + assert config.type == "disabled" + + +def test_thinking_config_explicit_enabled(): + """ThinkingConfig should accept type='enabled'.""" + config = ThinkingConfig(type = "enabled") + assert config.type == "enabled" + + +def test_chat_completion_request_with_thinking_disabled(): + """thinking.type='disabled' should map to enable_thinking=False.""" + req = ChatCompletionRequest.model_validate( + { + "model": "test-model", + "messages": [{"role": "user", "content": "hello"}], + "thinking": {"type": "disabled"}, + } + ) + assert req.thinking is not None + assert req.thinking.type == "disabled" + assert req.enable_thinking is False + + +def test_chat_completion_request_with_thinking_enabled(): + """thinking.type='enabled' should map to enable_thinking=True.""" + req = ChatCompletionRequest.model_validate( + { + "model": "test-model", + "messages": [{"role": "user", "content": "hello"}], + "thinking": {"type": "enabled"}, + } + ) + assert req.thinking is not None + assert req.thinking.type == "enabled" + assert req.enable_thinking is True + + +def test_chat_completion_request_without_thinking(): + """ChatCompletionRequest should work without thinking parameter.""" + req = ChatCompletionRequest.model_validate( + { + "model": "test-model", + "messages": [{"role": "user", "content": "hello"}], + } + ) + assert req.thinking is None + assert req.enable_thinking is None + + +def test_chat_completion_request_backward_compatible_enable_thinking(): + """ChatCompletionRequest should still support enable_thinking.""" + req = ChatCompletionRequest.model_validate( + { + "model": "test-model", + "messages": [{"role": "user", "content": "hello"}], + "enable_thinking": True, + } + ) + assert req.enable_thinking is True + assert req.thinking is None + + +def test_thinking_overrides_enable_thinking_when_both_provided(): + """When both thinking and enable_thinking are provided, + enable_thinking takes precedence (no override).""" + req = ChatCompletionRequest.model_validate( + { + "model": "test-model", + "messages": [{"role": "user", "content": "hello"}], + "thinking": {"type": "enabled"}, + "enable_thinking": False, + } + ) + # enable_thinking is explicitly set, so it takes precedence + assert req.enable_thinking is False + assert req.thinking.type == "enabled" diff --git a/studio/frontend/src/features/chat/api/chat-adapter.ts b/studio/frontend/src/features/chat/api/chat-adapter.ts index 7d9ee20a5f..6122b51dfd 100644 --- a/studio/frontend/src/features/chat/api/chat-adapter.ts +++ b/studio/frontend/src/features/chat/api/chat-adapter.ts @@ -2428,7 +2428,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter { : { reasoning_effort: fallbackExternalEffort, } - : { enable_thinking: reasoningEnabled } + : { thinking: { type: reasoningEnabled ? "enabled" : "disabled" } } : {}), }; } @@ -2457,7 +2457,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter { ? reasoningEnabled ? { reasoning_effort: localReasoningEffort } : {} - : { enable_thinking: reasoningEnabled } + : { thinking: { type: reasoningEnabled ? "enabled" : "disabled" } } : {}), ...(supportsPreserveThinking ? { preserve_thinking: preserveThinking } diff --git a/studio/frontend/src/features/chat/types/api.ts b/studio/frontend/src/features/chat/types/api.ts index 022624de60..2f0e30fdcf 100644 --- a/studio/frontend/src/features/chat/types/api.ts +++ b/studio/frontend/src/features/chat/types/api.ts @@ -289,6 +289,7 @@ export interface OpenAIChatCompletionsRequest { | "xhigh" | null; preserve_thinking?: boolean | null; + thinking?: {type: "disabled" | "enabled";} | null; enable_tools?: boolean | null; enabled_tools?: string[]; /** Local models + enable_tools only. */