feat: add Anthropic-compatible thinking parameter (#5856)

* feat: add Anthropic-compatible thinking parameter

Add `thinking` parameter using Anthropic's format ({type: 'disabled'} /
{type: 'enabled'}) alongside the existing `enable_thinking` boolean for
backward compatibility.

The new parameter is mapped internally to `enable_thinking` at the route
layer so all downstream templates and backends continue to work unchanged.

Changes:
- Add ThinkingConfig model and `thinking` field to ChatCompletionRequest
- Add mapping logic in routes: thinking.type -> enable_thinking
- Add `thinking` field to frontend TypeScript types
- Update frontend request building to send thinking parameter
- Add tests for new thinking parameter

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* fix: move thinking→enable_thinking mapping to model_validator

The Gemini review correctly identified that the route-level mapping
bypasses normalization for external provider requests. Moving the
mapping into a @model_validator on ChatCompletionRequest ensures it
runs during Pydantic validation regardless of routing path.

* Document ThinkingConfig scope and thinking validation behavior

---------

Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
Co-authored-by: Daniel Han <danielhanchen@gmail.com>
Co-authored-by: Lee Jackson <130007945+Imagineer99@users.noreply.github.com>
This commit is contained in:
maattm 2026-06-15 10:35:33 +01:00 committed by GitHub
commit 930dd17086
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 138 additions and 2 deletions

View file

@ -581,6 +581,16 @@ class ChatMessage(BaseModel):
return self
class ThinkingConfig(BaseModel):
"""Anthropic-compatible thinking/reasoning configuration.
Use type='disabled' to turn off thinking, or type='enabled' to turn it on.
Only type is read; extra fields (e.g. budget_tokens) are ignored, since
Studio sets provider thinking budgets itself.
"""
type: Literal["disabled", "enabled"] = "disabled"
class ChatCompletionRequest(BaseModel):
"""OpenAI-compatible chat completion request.
@ -694,6 +704,11 @@ class ChatCompletionRequest(BaseModel):
None,
description = "[x-unsloth] When true, keep historical <think> blocks from past assistant turns in the prompt (Qwen3.6 templates). Independent of enable_thinking / reasoning_effort.",
)
thinking: Optional[ThinkingConfig] = Field(
None,
description = "[Anthropic-compatible] Thinking configuration. "
"Use {type: 'disabled'} to disable thinking, {type: 'enabled'} to enable.",
)
enable_tools: Optional[bool] = Field(
None,
description = "[x-unsloth] Enable tool calling for supported models",
@ -952,6 +967,20 @@ class ChatCompletionRequest(BaseModel):
msg.tool_call_id = picked
return self
@model_validator(mode = "after")
def _map_thinking_to_enable_thinking(self) -> "ChatCompletionRequest":
"""Map Anthropic-style ``thinking`` parameter to internal ``enable_thinking``.
``thinking: {type: 'enabled'}`` sets ``enable_thinking = True`` and
``thinking: {type: 'disabled'}`` sets ``enable_thinking = False``.
``enable_thinking`` takes precedence when both are provided so that
callers who already use the internal field are unaffected. Invalid
``thinking`` shapes are rejected at validation time (422).
"""
if self.thinking is not None and self.enable_thinking is None:
self.enable_thinking = self.thinking.type == "enabled"
return self
class ToolConfirmRequest(BaseModel):
session_id: Optional[str] = None

View file

@ -0,0 +1,106 @@
# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
"""
Unit tests for the Anthropic-compatible thinking parameter.
Covers:
- ThinkingConfig model validation
- ChatCompletionRequest with thinking parameter
- Mapping logic: thinking.type -> enable_thinking
"""
import os
import sys
_backend = os.path.join(os.path.dirname(__file__), "..")
sys.path.insert(0, _backend)
from models.inference import ChatCompletionRequest, ThinkingConfig
def test_thinking_config_defaults_to_disabled():
"""ThinkingConfig should default to type='disabled'."""
config = ThinkingConfig()
assert config.type == "disabled"
def test_thinking_config_explicit_disabled():
"""ThinkingConfig should accept type='disabled'."""
config = ThinkingConfig(type = "disabled")
assert config.type == "disabled"
def test_thinking_config_explicit_enabled():
"""ThinkingConfig should accept type='enabled'."""
config = ThinkingConfig(type = "enabled")
assert config.type == "enabled"
def test_chat_completion_request_with_thinking_disabled():
"""thinking.type='disabled' should map to enable_thinking=False."""
req = ChatCompletionRequest.model_validate(
{
"model": "test-model",
"messages": [{"role": "user", "content": "hello"}],
"thinking": {"type": "disabled"},
}
)
assert req.thinking is not None
assert req.thinking.type == "disabled"
assert req.enable_thinking is False
def test_chat_completion_request_with_thinking_enabled():
"""thinking.type='enabled' should map to enable_thinking=True."""
req = ChatCompletionRequest.model_validate(
{
"model": "test-model",
"messages": [{"role": "user", "content": "hello"}],
"thinking": {"type": "enabled"},
}
)
assert req.thinking is not None
assert req.thinking.type == "enabled"
assert req.enable_thinking is True
def test_chat_completion_request_without_thinking():
"""ChatCompletionRequest should work without thinking parameter."""
req = ChatCompletionRequest.model_validate(
{
"model": "test-model",
"messages": [{"role": "user", "content": "hello"}],
}
)
assert req.thinking is None
assert req.enable_thinking is None
def test_chat_completion_request_backward_compatible_enable_thinking():
"""ChatCompletionRequest should still support enable_thinking."""
req = ChatCompletionRequest.model_validate(
{
"model": "test-model",
"messages": [{"role": "user", "content": "hello"}],
"enable_thinking": True,
}
)
assert req.enable_thinking is True
assert req.thinking is None
def test_thinking_overrides_enable_thinking_when_both_provided():
"""When both thinking and enable_thinking are provided,
enable_thinking takes precedence (no override)."""
req = ChatCompletionRequest.model_validate(
{
"model": "test-model",
"messages": [{"role": "user", "content": "hello"}],
"thinking": {"type": "enabled"},
"enable_thinking": False,
}
)
# enable_thinking is explicitly set, so it takes precedence
assert req.enable_thinking is False
assert req.thinking.type == "enabled"

View file

@ -2428,7 +2428,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
: {
reasoning_effort: fallbackExternalEffort,
}
: { enable_thinking: reasoningEnabled }
: { thinking: { type: reasoningEnabled ? "enabled" : "disabled" } }
: {}),
};
}
@ -2457,7 +2457,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
? reasoningEnabled
? { reasoning_effort: localReasoningEffort }
: {}
: { enable_thinking: reasoningEnabled }
: { thinking: { type: reasoningEnabled ? "enabled" : "disabled" } }
: {}),
...(supportsPreserveThinking
? { preserve_thinking: preserveThinking }

View file

@ -289,6 +289,7 @@ export interface OpenAIChatCompletionsRequest {
| "xhigh"
| null;
preserve_thinking?: boolean | null;
thinking?: {type: "disabled" | "enabled";} | null;
enable_tools?: boolean | null;
enabled_tools?: string[];
/** Local models + enable_tools only. */