diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 0776f0fdfe..5fb7395c00 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -4248,6 +4248,7 @@ class LlamaCppBackend: frequency_penalty: Optional[float] = None, seed: Optional[int] = None, parallel_tool_calls: Optional[bool] = None, + typical_p: Optional[float] = None, ) -> Generator[str | dict, None, None]: """ Send a chat completion request to llama-server and stream tokens back. @@ -4306,6 +4307,11 @@ class LlamaCppBackend: payload["seed"] = seed if parallel_tool_calls is not None: payload["parallel_tool_calls"] = parallel_tool_calls + # Locally typical sampling. llama-server default 1.0 disables it; + # the field is llama.cpp-specific (no cloud provider accepts it), + # so we only forward it when the caller explicitly sets one. + if typical_p is not None: + payload["typical_p"] = typical_p payload["stream_options"] = {"include_usage": True} url = f"{self.base_url}/v1/chat/completions" @@ -4451,6 +4457,7 @@ class LlamaCppBackend: frequency_penalty: Optional[float] = None, seed: Optional[int] = None, parallel_tool_calls: Optional[bool] = None, + typical_p: Optional[float] = None, ) -> Generator[dict, None, None]: """ Agentic loop: let the model call tools, execute them, and continue. @@ -4548,6 +4555,8 @@ class LlamaCppBackend: payload["seed"] = seed if parallel_tool_calls is not None: payload["parallel_tool_calls"] = parallel_tool_calls + if typical_p is not None: + payload["typical_p"] = typical_p try: _auth_headers = ( @@ -5240,6 +5249,8 @@ class LlamaCppBackend: stream_payload["seed"] = seed if parallel_tool_calls is not None: stream_payload["parallel_tool_calls"] = parallel_tool_calls + if typical_p is not None: + stream_payload["typical_p"] = typical_p stream_payload["stream_options"] = {"include_usage": True} cumulative = "" diff --git a/studio/backend/models/inference.py b/studio/backend/models/inference.py index 040bb6ce6c..b2c12b82c4 100644 --- a/studio/backend/models/inference.py +++ b/studio/backend/models/inference.py @@ -828,6 +828,18 @@ class ChatCompletionRequest(BaseModel): "default (which is `true` everywhere today)." ), ) + typical_p: Optional[float] = Field( + None, + ge = 0.0, + le = 1.0, + description = ( + "Locally typical sampling (llama.cpp `typ_p`). 1.0 disables. " + "Local llama-server only — no SaaS provider currently accepts " + "this field, so the frontend capability map gates it off for " + "every external provider and the local path forwards it on " + "/v1/chat/completions." + ), + ) fast_mode: Optional[bool] = Field( None, description = ( diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 72ab4afed8..5d71b446b5 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -2517,6 +2517,7 @@ async def openai_chat_completions( frequency_penalty = payload.frequency_penalty, seed = payload.seed, parallel_tool_calls = payload.parallel_tool_calls, + typical_p = payload.typical_p, ) _tool_sentinel = object() @@ -2690,6 +2691,7 @@ async def openai_chat_completions( frequency_penalty = payload.frequency_penalty, seed = payload.seed, parallel_tool_calls = payload.parallel_tool_calls, + typical_p = payload.typical_p, ) _gguf_sentinel = object() @@ -4963,6 +4965,7 @@ def _build_passthrough_payload( frequency_penalty = None, seed = None, parallel_tool_calls = None, + typical_p = None, tool_choice = "auto", response_format = None, chat_template_kwargs = None, @@ -5013,6 +5016,11 @@ def _build_passthrough_payload( body["seed"] = seed if parallel_tool_calls is not None: body["parallel_tool_calls"] = parallel_tool_calls + # llama.cpp-specific locally-typical sampling (typ_p in the sampler + # chain). No SaaS provider accepts this; the frontend capability map + # gates it to local only. + if typical_p is not None: + body["typical_p"] = typical_p if response_format is not None: # llama-server applies a GBNF grammar derived from the JSON schema # when response_format is present. Field is documented flat at the @@ -5447,6 +5455,7 @@ def _build_openai_passthrough_body(payload, backend_ctx = None) -> dict: frequency_penalty = payload.frequency_penalty, seed = payload.seed, parallel_tool_calls = payload.parallel_tool_calls, + typical_p = payload.typical_p, tool_choice = tool_choice, response_format = _extract_response_format(payload), chat_template_kwargs = tpl_kwargs, diff --git a/studio/backend/tests/test_sampling_params_routing.py b/studio/backend/tests/test_sampling_params_routing.py index 1727d01028..50714fcac4 100644 --- a/studio/backend/tests/test_sampling_params_routing.py +++ b/studio/backend/tests/test_sampling_params_routing.py @@ -1001,6 +1001,40 @@ def test_local_anthropic_passthrough_helpers_accept_parallel_tool_calls(): assert body.get("parallel_tool_calls") is False, body +def test_local_passthrough_forwards_typical_p_when_set(): + """`typical_p` is a llama.cpp-specific sampler (`typ_p` in the + sampler chain). The local-llama-cpp passthrough payload builder must + forward it when set so the chat-adapter can opt in for local + backends without the field bleeding into external providers (whose + capability map gates it off).""" + from routes import inference as route_mod + + body = route_mod._build_passthrough_payload( + openai_messages = [{"role": "user", "content": "hi"}], + openai_tools = None, + temperature = 0.6, + top_p = 0.95, + top_k = 20, + max_tokens = 64, + stream = True, + typical_p = 0.7, + ) + assert body.get("typical_p") == 0.7, body + + # When unset, the field is omitted entirely so llama-server falls + # back to its 1.0 default. + body2 = route_mod._build_passthrough_payload( + openai_messages = [{"role": "user", "content": "hi"}], + openai_tools = None, + temperature = 0.6, + top_p = 0.95, + top_k = 20, + max_tokens = 64, + stream = True, + ) + assert "typical_p" not in body2, body2 + + def test_anthropic_4_7_sampling_removed_regex_matches_expected_ids(): """Pin the canonical Claude 4.7 model-id shape so the frontend ANTHROPIC_4_7_SAMPLING_REMOVED_REGEX in diff --git a/studio/frontend/src/features/chat/api/chat-adapter.ts b/studio/frontend/src/features/chat/api/chat-adapter.ts index c6550d96e8..52e2b87d2d 100644 --- a/studio/frontend/src/features/chat/api/chat-adapter.ts +++ b/studio/frontend/src/features/chat/api/chat-adapter.ts @@ -1587,6 +1587,15 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter { params.parallelToolCalls === false ? { parallel_tool_calls: false } : {}), + // llama.cpp `typ_p`. External providers all have + // capabilities.typicalP=false; only the permissive local + // buckets (custom/vllm/ollama/llama_cpp) opt in. `null` + // (unset) or `1.0` (llama-server default) is a no-op. + ...(externalCapabilities?.typicalP && + params.typicalP !== null && + params.typicalP !== 1 + ? { typical_p: params.typicalP } + : {}), // Built-in tools: Search pill maps to provider-side // web_search (currently OpenAI / Anthropic / OpenRouter / // Kimi); Code pill maps to Anthropic's server-side @@ -1707,6 +1716,12 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter { : {}), ...(params.seed !== null ? { seed: params.seed } : {}), ...(params.stop.length > 0 ? { stop: params.stop } : {}), + // llama.cpp `typ_p`. Local only — external providers gate + // it off via capability map. `null` (unset) or 1.0 (server + // default) is a no-op so we forward only meaningful values. + ...(params.typicalP !== null && params.typicalP !== 1 + ? { typical_p: params.typicalP } + : {}), parallel_tool_calls: params.parallelToolCalls, image_base64: imageBase64, audio_base64: audioBase64, diff --git a/studio/frontend/src/features/chat/provider-capabilities.ts b/studio/frontend/src/features/chat/provider-capabilities.ts index e97d419b45..5a48aa8fc6 100644 --- a/studio/frontend/src/features/chat/provider-capabilities.ts +++ b/studio/frontend/src/features/chat/provider-capabilities.ts @@ -13,6 +13,10 @@ * a null capability — every knob renders for them. */ +// NB: when adding a new sampling knob, default it to `false` on every +// SaaS provider in PROVIDER_CAPABILITIES below (only local backends +// + the permissive {custom, vllm, ollama, llama_cpp, openrouter} +// providers should expose llama.cpp-specific samplers). export interface ProviderCapabilities { /** * Temperature sampling. Reasoning-class models (OpenAI's gpt-5.x / o3 via @@ -58,6 +62,13 @@ export interface ProviderCapabilities { * `disable_parallel_tool_use: true` on Anthropic (inverted). */ parallelToolCalls: boolean; + /** + * llama.cpp `typ_p` (locally typical sampling). Local llama-server + * only — no SaaS provider currently accepts this field. Default is + * `false` for every external provider and `true` only for the local + * permissive {custom, vllm, ollama, llama_cpp} buckets. + */ + typicalP: boolean; } /** @@ -401,6 +412,7 @@ const OPENAI_COMPAT_BASE: ProviderCapabilities = { stop: true, serviceTier: false, parallelToolCalls: true, + typicalP: false, }; const ALL_SUPPORTED: ProviderCapabilities = { @@ -415,6 +427,7 @@ const ALL_SUPPORTED: ProviderCapabilities = { stop: true, serviceTier: false, parallelToolCalls: true, + typicalP: true, }; // Reasoning-class OpenAI models served via /v1/responses fix temperature @@ -437,6 +450,7 @@ const OPENAI_REASONING_CAPABILITIES: ProviderCapabilities = { stop: false, serviceTier: true, parallelToolCalls: true, + typicalP: false, }; const OPENAI_CHAT_CAPABILITIES: ProviderCapabilities = { temperature: true, @@ -452,6 +466,7 @@ const OPENAI_CHAT_CAPABILITIES: ProviderCapabilities = { stop: false, serviceTier: true, parallelToolCalls: true, + typicalP: false, }; // Prefix list for OpenAI reasoning-class model ids. Kept in sync with @@ -493,6 +508,23 @@ function isClaude47SamplingRemoved(modelId: string | null | undefined): boolean return ANTHROPIC_4_7_SAMPLING_REMOVED_REGEX.test(normalized); } +// DeepSeek reasoning-class models silently ignore temperature, top_p, +// presence_penalty, frequency_penalty and 400 on logprobs/top_logprobs. +// `deepseek-reasoner` is the dedicated thinking model; +// `deepseek-v4-flash` runs reasoning-mode under the same flag as well. +// Match by prefix so future revisions (deepseek-reasoner-2027 etc.) +// continue to gate correctly. +const DEEPSEEK_REASONING_MODEL_PREFIXES = [ + "deepseek-reasoner", + "deepseek-r1", +] as const; + +function isDeepSeekReasoningModelId(modelId: string | null | undefined): boolean { + const normalized = modelId?.trim().toLowerCase() ?? ""; + if (!normalized) return false; + return DEEPSEEK_REASONING_MODEL_PREFIXES.some((p) => normalized.startsWith(p)); +} + const PROVIDER_CAPABILITIES: Record = { // Default OpenAI bucket is reasoning-class (current registry only ships // gpt-5.x / o3 ids), but per-model resolution in getProviderCapabilities @@ -520,6 +552,7 @@ const PROVIDER_CAPABILITIES: Record = { stop: true, serviceTier: true, parallelToolCalls: true, + typicalP: false, }, mistral: OPENAI_COMPAT_BASE, gemini: OPENAI_COMPAT_BASE, @@ -547,8 +580,17 @@ const PROVIDER_CAPABILITIES: Record = { stop: true, serviceTier: false, parallelToolCalls: false, + typicalP: false, }, // DeepSeek deprecated presence/frequency penalty in their current docs. + // Chat-class defaults (deepseek-chat / deepseek-v4-flash non-thinking): + // accept temperature, top_p, seed, stop. Reasoning class + // (deepseek-reasoner / deepseek-v4-flash thinking-mode) ignores + // temperature, top_p, presence_penalty, frequency_penalty entirely and + // 400s on logprobs — see + // https://api-docs.deepseek.com/guides/reasoning_model. Per-model + // resolution in getProviderCapabilities downshifts reasoner ids onto + // DEEPSEEK_REASONING_CAPABILITIES. deepseek: { temperature: true, topP: true, @@ -561,6 +603,7 @@ const PROVIDER_CAPABILITIES: Record = { stop: true, serviceTier: false, parallelToolCalls: true, + typicalP: false, }, qwen: OPENAI_COMPAT_BASE, huggingface: OPENAI_COMPAT_BASE, @@ -591,6 +634,8 @@ const DEFAULT_EXTERNAL_CAPABILITIES = OPENAI_COMPAT_BASE; * (OPENAI_REASONING_CAPABILITIES). * - anthropic + claude-*-4-7: temperature/top_p/top_k stripped to * match the backend 400-avoidance regex. + * - deepseek + reasoning model (deepseek-reasoner / r1): hides + * temperature/top_p (silently ignored upstream). */ export function getProviderCapabilities( providerType: string | null | undefined, @@ -604,6 +649,9 @@ export function getProviderCapabilities( if (providerType === "anthropic" && isClaude47SamplingRemoved(modelId)) { return { ...base, temperature: false, topP: false, topK: false }; } + if (providerType === "deepseek" && isDeepSeekReasoningModelId(modelId)) { + return { ...base, temperature: false, topP: false }; + } return base; } diff --git a/studio/frontend/src/features/chat/types/api.ts b/studio/frontend/src/features/chat/types/api.ts index 3cf4d83a48..0db72ac5c2 100644 --- a/studio/frontend/src/features/chat/types/api.ts +++ b/studio/frontend/src/features/chat/types/api.ts @@ -304,6 +304,14 @@ export interface OpenAIChatCompletionsRequest { * keeps each provider's upstream default. */ parallel_tool_calls?: boolean; + /** + * llama.cpp `typ_p` (locally typical sampling). Local llama-server + * only — no SaaS provider currently accepts this. 1.0 disables + * (llama-server default). External-provider capability map already + * gates this off, so on the wire it only appears for local + the + * permissive {custom, vllm, ollama, llama_cpp} buckets. + */ + typical_p?: number; /** * Anthropic fast-mode toggle. Opus 4.6 / 4.7 only; backend drops * silently on every other model + provider. See diff --git a/studio/frontend/src/features/chat/types/runtime.ts b/studio/frontend/src/features/chat/types/runtime.ts index 7fac74979b..0793573583 100644 --- a/studio/frontend/src/features/chat/types/runtime.ts +++ b/studio/frontend/src/features/chat/types/runtime.ts @@ -44,6 +44,12 @@ export interface InferenceParams { * upstream defaults across all three. */ parallelToolCalls: boolean; + /** + * Locally typical sampling (llama.cpp `typ_p`). Local llama-server + * only — no SaaS provider currently accepts this. 1.0 disables (and + * is the llama-server default). `null` = unset (not forwarded). + */ + typicalP: number | null; maxSeqLength: number; maxTokens: number; systemPrompt: string; @@ -70,6 +76,7 @@ export const DEFAULT_INFERENCE_PARAMS: InferenceParams = { stop: [], serviceTier: null, parallelToolCalls: true, + typicalP: null, maxSeqLength: 4096, maxTokens: 8192, systemPrompt: "", diff --git a/studio/frontend/src/features/chat/utils/chat-settings-storage.ts b/studio/frontend/src/features/chat/utils/chat-settings-storage.ts index 1b186db727..ede79c7098 100644 --- a/studio/frontend/src/features/chat/utils/chat-settings-storage.ts +++ b/studio/frontend/src/features/chat/utils/chat-settings-storage.ts @@ -182,6 +182,16 @@ function sanitizeInferenceParams( if (typeof value.parallelToolCalls === "boolean") { params.parallelToolCalls = value.parallelToolCalls; } + // typicalP: nullable float (null = no typ_p on the wire, matching + // llama-server's default 1.0). Mirror seed's nullable-float handling. + if (value.typicalP === null) { + params.typicalP = null; + } else if ( + typeof value.typicalP === "number" && + Number.isFinite(value.typicalP) + ) { + params.typicalP = value.typicalP; + } // Mirror trustRemoteCode handling so the toggle survives reload // and the /api/chat/settings round-trip. if (typeof value.fastMode === "boolean") {