From c22f6e48ff6f3c77adab22d0e8637fdce4c0404f Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Wed, 27 May 2026 06:49:15 +0000 Subject: [PATCH] Apply round-2 audit fixes: per-model OpenAI caps + o-series effort + Ollama bucket (PR #5711) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Second 5-Opus reviewer round. Applying high-confidence fixes; speculative items (gpt-5.5-pro effort restriction, o3 image_generation gating, o-series parallel_tool_calls per-model, gpt-5.x new model prefixes, Anthropic fast-mode + Priority exclusion UI gate, Gemini service_tier, Kimi k2.5 toggleable thinking) deferred to follow-up because they need type-system changes, more verification, or backend wire work. OpenAI max-output caps — replace the 3-line table with one driven by direct dev.openai.com per-model fetches (cross-checked against the Azure Foundry reasoning table): - gpt-5.4 / gpt-5.4-pro / gpt-5.4-mini / gpt-5.4-nano: 65536 -> 128000 (https://developers.openai.com/api/docs/models/gpt-5.4 "128,000 max output tokens"; Azure table same). - gpt-5.3-codex: 16384 -> 128000 (https://developers.openai.com/api/docs/models/gpt-5.3-codex). - gpt-5 / gpt-5.1 / gpt-5.2: 32k default -> 128000 (https://developers.openai.com/api/docs/models/gpt-5.2 confirms 128k; Azure table extends to gpt-5/5.1). - gpt-5.3-chat-latest and gpt-5.1-chat keep 16384 (chat-class variants per Azure context table row). - o1 / o3 / o3-mini / o3-pro / o4-mini / codex-mini: 32k default -> 100000 (https://developers.openai.com/api/docs/models/o3 "100,000 max output tokens"; Azure o-series table same). Implementation: list the two 16k chat-latest ids first so the broader `gpt-5` 128k entry doesn't shadow them. OpenAI reasoning_effort levels: - gpt-5.3-codex: drop "none" from levels + flip supportsOff to false. Dev page lists the enum as low/medium/high/xhigh only — `none` is not in the codex variant. - o-series bucket: change prefix from ["o3"] to ["o1","o3","o4","codex-mini"]. Previously o1 / o4-mini / codex-mini fell into NO_REASONING_CAPS so the panel HID the effort slider for them — real UX regression for users on those ids. Azure o-series table confirms all four accept low/medium/high reasoning_effort. DeepSeek default_models: - Add deepseek-v4-pro + deepseek-v4-flash alongside the legacy deepseek-chat / deepseek-reasoner aliases. The latter retire on 2026-07-24 per https://api-docs.deepseek.com/updates; surfacing both lets the picker keep working on cutover. Local backend bucket split (Ollama-stricter): - Splits the round-1 VLLM_OLLAMA_CAPABILITIES into a vLLM-specific bucket (keeps top_k / min_p / repetition_penalty / seed on; vLLM's SamplingParams supports all four) and an Ollama-specific bucket that ALSO hides top_k / min_p / repetition_penalty. Ollama's OAI translator (ollama/openai/openai.go FromChatRequest) only copies the documented OpenAI subset on the /v1/chat/completions path that Studio uses; the three knobs are silently dropped even though native /api/chat would forward them via `options`. Hiding them is the smaller fix vs adding a backend /api/chat rewrite path. Reviewer claims verified wrong, skipped: - _ANTHROPIC_NEW_CODE_EXEC_PREFIXES already lists opus-4-7, opus-4-6, sonnet-4-6 (external_provider.py:337-339). No-op. - Mistral `seed` already renamed to `random_seed` by backend at external_provider.py:772. No-op. - OpenRouter `isOpenRouterMandatoryReasoningModel` uses `Set.has()` exact match, not prefix match, so deepseek/deepseek-r1-distill-* cannot accidentally hit the always-on guard. No-op. Tests: 63/63 sampling_params_routing tests pass; frontend tsc clean. --- studio/backend/core/inference/providers.py | 8 ++- .../features/chat/provider-capabilities.ts | 69 +++++++++++++------ 2 files changed, 55 insertions(+), 22 deletions(-) diff --git a/studio/backend/core/inference/providers.py b/studio/backend/core/inference/providers.py index c0bd0afdc0..5d4d11059a 100644 --- a/studio/backend/core/inference/providers.py +++ b/studio/backend/core/inference/providers.py @@ -97,7 +97,13 @@ PROVIDER_REGISTRY: dict[str, dict[str, Any]] = { "deepseek": { "display_name": "DeepSeek", "base_url": "https://api.deepseek.com/v1", + # Legacy aliases (deepseek-chat / deepseek-reasoner) retire + # 2026-07-24 per https://api-docs.deepseek.com/updates. Surface + # the new canonical ids (deepseek-v4-flash / deepseek-v4-pro) + # alongside so the picker keeps working on cutover. "default_models": [ + "deepseek-v4-pro", + "deepseek-v4-flash", "deepseek-chat", "deepseek-reasoner", ], @@ -106,7 +112,7 @@ PROVIDER_REGISTRY: dict[str, dict[str, Any]] = { "supports_tool_calling": True, "auth_header": "Authorization", "auth_prefix": "Bearer ", - "notes": "OpenAI-compatible API. deepseek-chat = V3, deepseek-reasoner = R1 thinking mode.", + "notes": "OpenAI-compatible API. deepseek-v4-pro / deepseek-v4-flash are the new canonical ids; deepseek-chat / deepseek-reasoner remain as legacy aliases until 2026-07-24.", }, "mistral": { "display_name": "Mistral AI", diff --git a/studio/frontend/src/features/chat/provider-capabilities.ts b/studio/frontend/src/features/chat/provider-capabilities.ts index 936ce6a7ea..45647b970b 100644 --- a/studio/frontend/src/features/chat/provider-capabilities.ts +++ b/studio/frontend/src/features/chat/provider-capabilities.ts @@ -230,10 +230,17 @@ const EXTERNAL_MAX_OUTPUT_TOKENS_BY_MODEL: Array<{ prefixes: readonly string[]; cap: number; }> = [ - // OpenAI - { providerType: "openai", prefixes: ["gpt-5.5-pro", "gpt-5.5"], cap: 128000 }, - { providerType: "openai", prefixes: ["gpt-5.4-pro", "gpt-5.4"], cap: 65536 }, - { providerType: "openai", prefixes: ["gpt-5.3"], cap: 16384 }, + // OpenAI per-model output caps from developers.openai.com per-model + // pages (cross-checked against the Azure Foundry reasoning table). + // Order matters: list the 16k chat-latest variants first so the + // broader gpt-5 / gpt-4 entries don't shadow them. + // gpt-5.3-chat-latest / gpt-5.1-chat = 16384 (chat-class) + // gpt-5.5* / gpt-5.4* / gpt-5.3-codex / gpt-5.2 / gpt-5.1 / gpt-5 + // / gpt-5-codex / gpt-5-pro = 128000 + // o1 / o3 / o3-pro / o4-mini / codex-mini = 100000 + { providerType: "openai", prefixes: ["gpt-5.3-chat-latest", "gpt-5.1-chat"], cap: 16384 }, + { providerType: "openai", prefixes: ["gpt-5"], cap: 128000 }, + { providerType: "openai", prefixes: ["o1", "o3", "o4", "codex-mini"], cap: 100000 }, // Anthropic — overview table at // platform.claude.com/docs/en/about-claude/models/overview. Opus 4.7 // and Opus 4.6 BOTH ship 128k Max output (the legacy-table row for @@ -581,15 +588,11 @@ const LLAMA_CPP_CAPABILITIES: ProviderCapabilities = { topA: false, }; -// vLLM and Ollama's OpenAI-compat endpoints accept the OpenAI subset -// plus top_k / min_p / repetition_penalty / seed, but neither forwards -// the llama.cpp-only extended samplers (typical_p, top_n_sigma, -// repeat_last_n, dynatemp_*, mirostat*). vLLM's SamplingParams has no -// fields for them (vllm/sampling_params.py) and Ollama's OAI -// translator (ollama/openai/openai.go FromChatRequest) only copies the -// OpenAI subset. Surfacing the eight extra sliders here would be the -// silent-drop UX the file header warns against, so we hide them. -const VLLM_OLLAMA_CAPABILITIES: ProviderCapabilities = { +// vLLM's OpenAI-compat endpoint accepts the OpenAI subset plus top_k / +// min_p / repetition_penalty / seed, but not the 8 llama.cpp-only +// extended samplers (vLLM's SamplingParams has no fields for them — +// vllm/sampling_params.py). +const VLLM_CAPABILITIES: ProviderCapabilities = { ...LLAMA_CPP_CAPABILITIES, typicalP: false, topNSigma: false, @@ -601,6 +604,20 @@ const VLLM_OLLAMA_CAPABILITIES: ProviderCapabilities = { mirostatEta: false, }; +// Ollama is stricter than vLLM. Studio reaches Ollama via the OpenAI- +// compat /v1/chat/completions transport, and Ollama's translator +// (ollama/openai/openai.go FromChatRequest) only copies the documented +// OpenAI subset — top_k / min_p / repetition_penalty are silently +// DROPPED on that path even though native /api/chat would forward them +// through the `options` bag. Hide them so users don't move a slider +// the wire never carries. +const OLLAMA_CAPABILITIES: ProviderCapabilities = { + ...VLLM_CAPABILITIES, + topK: false, + minP: false, + repetitionPenalty: false, +}; + // OpenRouter is a router-of-routers: the gateway accepts a wider set // of OpenAI-style sampling fields than any single upstream supports // and silently drops what the chosen route does not, per @@ -861,13 +878,14 @@ const PROVIDER_CAPABILITIES: Record = { openrouter: OPENROUTER_CAPABILITIES, // `llama_cpp` and the permissive `custom` preset terminate at the // first-party llama-server runtime, so the full sampler chain is - // available. vLLM and Ollama only surface the OpenAI subset - // (+ top_k/min_p/rep_penalty/seed) — the 8 extended llama.cpp - // samplers are hidden to avoid the silent-drop UX. + // available. vLLM surfaces the OpenAI subset + top_k/min_p/ + // repetition_penalty/seed (no extended llama.cpp samplers). Ollama + // is stricter: its OAI translator drops top_k/min_p/repetition_penalty + // too on the /v1 path. custom: LLAMA_CPP_CAPABILITIES, llama_cpp: LLAMA_CPP_CAPABILITIES, - vllm: VLLM_OLLAMA_CAPABILITIES, - ollama: VLLM_OLLAMA_CAPABILITIES, + vllm: VLLM_CAPABILITIES, + ollama: OLLAMA_CAPABILITIES, }; const DEFAULT_EXTERNAL_CAPABILITIES = OPENAI_COMPAT_BASE; @@ -1008,9 +1026,12 @@ const OPENAI_REASONING_MODELS = [ levels: ["medium"], }, { + // gpt-5.3-codex per dev page lists ONLY low/medium/high/xhigh + // (https://developers.openai.com/api/docs/models/gpt-5.3-codex); + // `none` is not in the codex enum so supportsOff stays false. prefixes: ["gpt-5.3-codex"], - supportsOff: true, - levels: ["none", "low", "medium", "high", "xhigh"], + supportsOff: false, + levels: ["low", "medium", "high", "xhigh"], }, { // Original gpt-5: minimal is supported, but per Azure footnote ^7^ @@ -1027,7 +1048,13 @@ const OPENAI_REASONING_MODELS = [ levels: ["minimal", "low", "medium", "high"], }, { - prefixes: ["o3"], + // o-series reasoning models: o1, o3, o3-mini, o3-pro, o4-mini, + // codex-mini all expose low/medium/high reasoning_effort per + // developers.openai.com/api/docs/models/o3 and the Azure Foundry + // o-series table. Without this entry o1/o4/codex-mini fell into + // NO_REASONING_CAPS and the panel hid the effort slider — a real + // UX regression for users on those ids. + prefixes: ["o1", "o3", "o4", "codex-mini"], supportsOff: false, levels: DEFAULT_EFFORT_LEVELS, },