feat: add Anthropic-compatible thinking parameter (#5856)
* feat: add Anthropic-compatible thinking parameter
Add `thinking` parameter using Anthropic's format ({type: 'disabled'} /
{type: 'enabled'}) alongside the existing `enable_thinking` boolean for
backward compatibility.
The new parameter is mapped internally to `enable_thinking` at the route
layer so all downstream templates and backends continue to work unchanged.
Changes:
- Add ThinkingConfig model and `thinking` field to ChatCompletionRequest
- Add mapping logic in routes: thinking.type -> enable_thinking
- Add `thinking` field to frontend TypeScript types
- Update frontend request building to send thinking parameter
- Add tests for new thinking parameter
* [pre-commit.ci] auto fixes from pre-commit.com hooks
for more information, see https://pre-commit.ci
* fix: move thinking→enable_thinking mapping to model_validator
The Gemini review correctly identified that the route-level mapping
bypasses normalization for external provider requests. Moving the
mapping into a @model_validator on ChatCompletionRequest ensures it
runs during Pydantic validation regardless of routing path.
* Document ThinkingConfig scope and thinking validation behavior
---------
Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
Co-authored-by: Daniel Han <danielhanchen@gmail.com>
Co-authored-by: Lee Jackson <130007945+Imagineer99@users.noreply.github.com>
This commit is contained in:
parent
9f694ab750
commit
930dd17086
4 changed files with 138 additions and 2 deletions
|
|
@ -581,6 +581,16 @@ class ChatMessage(BaseModel):
|
|||
return self
|
||||
|
||||
|
||||
class ThinkingConfig(BaseModel):
|
||||
"""Anthropic-compatible thinking/reasoning configuration.
|
||||
Use type='disabled' to turn off thinking, or type='enabled' to turn it on.
|
||||
Only type is read; extra fields (e.g. budget_tokens) are ignored, since
|
||||
Studio sets provider thinking budgets itself.
|
||||
"""
|
||||
|
||||
type: Literal["disabled", "enabled"] = "disabled"
|
||||
|
||||
|
||||
class ChatCompletionRequest(BaseModel):
|
||||
"""OpenAI-compatible chat completion request.
|
||||
|
||||
|
|
@ -694,6 +704,11 @@ class ChatCompletionRequest(BaseModel):
|
|||
None,
|
||||
description = "[x-unsloth] When true, keep historical <think> blocks from past assistant turns in the prompt (Qwen3.6 templates). Independent of enable_thinking / reasoning_effort.",
|
||||
)
|
||||
thinking: Optional[ThinkingConfig] = Field(
|
||||
None,
|
||||
description = "[Anthropic-compatible] Thinking configuration. "
|
||||
"Use {type: 'disabled'} to disable thinking, {type: 'enabled'} to enable.",
|
||||
)
|
||||
enable_tools: Optional[bool] = Field(
|
||||
None,
|
||||
description = "[x-unsloth] Enable tool calling for supported models",
|
||||
|
|
@ -952,6 +967,20 @@ class ChatCompletionRequest(BaseModel):
|
|||
msg.tool_call_id = picked
|
||||
return self
|
||||
|
||||
@model_validator(mode = "after")
|
||||
def _map_thinking_to_enable_thinking(self) -> "ChatCompletionRequest":
|
||||
"""Map Anthropic-style ``thinking`` parameter to internal ``enable_thinking``.
|
||||
|
||||
``thinking: {type: 'enabled'}`` sets ``enable_thinking = True`` and
|
||||
``thinking: {type: 'disabled'}`` sets ``enable_thinking = False``.
|
||||
``enable_thinking`` takes precedence when both are provided so that
|
||||
callers who already use the internal field are unaffected. Invalid
|
||||
``thinking`` shapes are rejected at validation time (422).
|
||||
"""
|
||||
if self.thinking is not None and self.enable_thinking is None:
|
||||
self.enable_thinking = self.thinking.type == "enabled"
|
||||
return self
|
||||
|
||||
|
||||
class ToolConfirmRequest(BaseModel):
|
||||
session_id: Optional[str] = None
|
||||
|
|
|
|||
106
studio/backend/tests/test_thinking_parameter.py
Normal file
106
studio/backend/tests/test_thinking_parameter.py
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
# SPDX-License-Identifier: AGPL-3.0-only
|
||||
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
||||
|
||||
"""
|
||||
Unit tests for the Anthropic-compatible thinking parameter.
|
||||
|
||||
Covers:
|
||||
- ThinkingConfig model validation
|
||||
- ChatCompletionRequest with thinking parameter
|
||||
- Mapping logic: thinking.type -> enable_thinking
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
_backend = os.path.join(os.path.dirname(__file__), "..")
|
||||
sys.path.insert(0, _backend)
|
||||
|
||||
from models.inference import ChatCompletionRequest, ThinkingConfig
|
||||
|
||||
|
||||
def test_thinking_config_defaults_to_disabled():
|
||||
"""ThinkingConfig should default to type='disabled'."""
|
||||
config = ThinkingConfig()
|
||||
assert config.type == "disabled"
|
||||
|
||||
|
||||
def test_thinking_config_explicit_disabled():
|
||||
"""ThinkingConfig should accept type='disabled'."""
|
||||
config = ThinkingConfig(type = "disabled")
|
||||
assert config.type == "disabled"
|
||||
|
||||
|
||||
def test_thinking_config_explicit_enabled():
|
||||
"""ThinkingConfig should accept type='enabled'."""
|
||||
config = ThinkingConfig(type = "enabled")
|
||||
assert config.type == "enabled"
|
||||
|
||||
|
||||
def test_chat_completion_request_with_thinking_disabled():
|
||||
"""thinking.type='disabled' should map to enable_thinking=False."""
|
||||
req = ChatCompletionRequest.model_validate(
|
||||
{
|
||||
"model": "test-model",
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
"thinking": {"type": "disabled"},
|
||||
}
|
||||
)
|
||||
assert req.thinking is not None
|
||||
assert req.thinking.type == "disabled"
|
||||
assert req.enable_thinking is False
|
||||
|
||||
|
||||
def test_chat_completion_request_with_thinking_enabled():
|
||||
"""thinking.type='enabled' should map to enable_thinking=True."""
|
||||
req = ChatCompletionRequest.model_validate(
|
||||
{
|
||||
"model": "test-model",
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
"thinking": {"type": "enabled"},
|
||||
}
|
||||
)
|
||||
assert req.thinking is not None
|
||||
assert req.thinking.type == "enabled"
|
||||
assert req.enable_thinking is True
|
||||
|
||||
|
||||
def test_chat_completion_request_without_thinking():
|
||||
"""ChatCompletionRequest should work without thinking parameter."""
|
||||
req = ChatCompletionRequest.model_validate(
|
||||
{
|
||||
"model": "test-model",
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
}
|
||||
)
|
||||
assert req.thinking is None
|
||||
assert req.enable_thinking is None
|
||||
|
||||
|
||||
def test_chat_completion_request_backward_compatible_enable_thinking():
|
||||
"""ChatCompletionRequest should still support enable_thinking."""
|
||||
req = ChatCompletionRequest.model_validate(
|
||||
{
|
||||
"model": "test-model",
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
"enable_thinking": True,
|
||||
}
|
||||
)
|
||||
assert req.enable_thinking is True
|
||||
assert req.thinking is None
|
||||
|
||||
|
||||
def test_thinking_overrides_enable_thinking_when_both_provided():
|
||||
"""When both thinking and enable_thinking are provided,
|
||||
enable_thinking takes precedence (no override)."""
|
||||
req = ChatCompletionRequest.model_validate(
|
||||
{
|
||||
"model": "test-model",
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
"thinking": {"type": "enabled"},
|
||||
"enable_thinking": False,
|
||||
}
|
||||
)
|
||||
# enable_thinking is explicitly set, so it takes precedence
|
||||
assert req.enable_thinking is False
|
||||
assert req.thinking.type == "enabled"
|
||||
|
|
@ -2428,7 +2428,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
: {
|
||||
reasoning_effort: fallbackExternalEffort,
|
||||
}
|
||||
: { enable_thinking: reasoningEnabled }
|
||||
: { thinking: { type: reasoningEnabled ? "enabled" : "disabled" } }
|
||||
: {}),
|
||||
};
|
||||
}
|
||||
|
|
@ -2457,7 +2457,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
? reasoningEnabled
|
||||
? { reasoning_effort: localReasoningEffort }
|
||||
: {}
|
||||
: { enable_thinking: reasoningEnabled }
|
||||
: { thinking: { type: reasoningEnabled ? "enabled" : "disabled" } }
|
||||
: {}),
|
||||
...(supportsPreserveThinking
|
||||
? { preserve_thinking: preserveThinking }
|
||||
|
|
|
|||
|
|
@ -289,6 +289,7 @@ export interface OpenAIChatCompletionsRequest {
|
|||
| "xhigh"
|
||||
| null;
|
||||
preserve_thinking?: boolean | null;
|
||||
thinking?: {type: "disabled" | "enabled";} | null;
|
||||
enable_tools?: boolean | null;
|
||||
enabled_tools?: string[];
|
||||
/** Local models + enable_tools only. */
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue