Tighten comments across PR 5711 (no behaviour change)

Comments-only pass. Drops verbose docstrings to single-line form,
removes repetitive "null = unset" / "Local only" tails (already
encoded by the type signature and capability map), keeps every
authoritative source URL but cuts surrounding prose, and removes
fully-redundant per-field comments where the field name already
says what the comment says.

Touches:
  - types/runtime.ts (InferenceParams)
  - types/api.ts (OpenAIChatCompletionsRequest wire shape)
  - provider-capabilities.ts (ProviderCapabilities interface + bucket
    inline blocks + per-model resolvers + reasoning helpers)
  - api/chat-adapter.ts (external + local forwarding stanzas)
  - backend models/inference.py (Field descriptions)
  - backend llama_cpp.py (3rd payload-builder inline comments)
  - backend routes/inference.py (_build_passthrough_payload)
  - backend external_provider.py (4.7 sampling-removed header,
    _is_openai_family_cloud docstring)

Net 508 lines deleted across 8 files; 65/65 sampling_params_routing
tests still pass; frontend tsc clean.
This commit is contained in:
Daniel Han 2026-05-27 11:14:32 +00:00
commit b60b0740c2
8 changed files with 304 additions and 812 deletions

View file

@ -71,32 +71,23 @@ def _normalize_stop_for_provider(
return None
# Claude 4.7 Opus removed temperature, top_p, and top_k — the API
# returns 400 "<param> is deprecated for this model" if any of them is
# set to a non-default value. The "Sampling parameters removed" section
# of the 4.7 release notes is the authoritative reference:
# https://platform.claude.com/docs/en/about-claude/models/whats-new-claude-4-7
# Only Opus shipped in the 4.7 generation (Sonnet stops at 4.6, Haiku at
# 4.5 per https://platform.claude.com/docs/en/about-claude/models/overview),
# so the regex is anchored to opus-4-7 only. 3.x and 4.5/4.6 still accept
# all three knobs; the trailing -4-7[-.]/EOL anchor keeps future versions
# (e.g. claude-opus-5) unaffected.
# Opus 4.7 removed temperature/top_p/top_k (400s on any non-default).
# Only Opus shipped in 4.7; 3.x and 4.5/4.6 still accept all three.
# Trailing -4-7[-.]/EOL anchor keeps future families (claude-opus-5
# etc) unaffected.
# https://platform.claude.com/docs/en/about-claude/models/whats-new-claude-4-7
def _is_openai_family_cloud(base_url: Optional[str]) -> bool:
"""True iff ``base_url`` points at OpenAI cloud or Azure OpenAI Foundry.
Anchored to the URL host so an attacker can't bypass the gate with a
path or subdomain like ``https://evil.com/api.openai.com/v1`` or
``https://api.openai.com.attacker.com/v1`` (CodeQL py/incomplete-url-
substring-sanitization). Used to scope cloud-only Responses-API
extensions (prompt_cache_retention, context_management compaction,
container shell tool) that 400 on non-cloud OpenAI-compatible
servers (ollama / llama.cpp / vLLM).
Host-anchored to avoid subdomain-injection bypass
(https://evil.com/api.openai.com/v1, https://api.openai.com.attacker.com/v1).
Used to gate cloud-only Responses-API extensions
(prompt_cache_retention, context_management compaction, container
shell tool) that 400 on non-cloud OAI-compat servers.
Azure Foundry resources are scoped to
``<resource-name>.openai.azure.com``; match any subdomain via an
`endswith` on the lowercased hostname, with the leading dot so
`openai.azure.com` itself doesn't slip through (there is no
apex-hosted Azure Foundry endpoint).
Azure Foundry uses <resource>.openai.azure.com; match via endswith
with the leading dot so the apex `openai.azure.com` can't slip
through (no apex Foundry endpoint exists).
"""
if not base_url:
return False

View file

@ -4327,22 +4327,17 @@ class LlamaCppBackend:
_cleaned = [s for s in stop if isinstance(s, str) and s]
if _cleaned:
payload["stop"] = _cleaned
# Optional sampling extensions, gated on `is not None` so 0,
# 0.0, and False all reach the wire.
# Each field gated `is not None` so explicit 0 / 0.0 / False
# values reach the wire. llama-server silently ignores fields
# it doesn't recognise.
if frequency_penalty is not None:
payload["frequency_penalty"] = frequency_penalty
if seed is not None:
payload["seed"] = seed
if parallel_tool_calls is not None:
payload["parallel_tool_calls"] = parallel_tool_calls
# Locally typical sampling. llama-server default 1.0 disables it;
# the field is llama.cpp-specific (no cloud provider accepts it),
# so we only forward it when the caller explicitly sets one.
if typical_p is not None:
payload["typical_p"] = typical_p
# Extended llama.cpp sampler chain (top_n_sigma, repeat_last_n,
# dynatemp_*, mirostat_*). All llama.cpp-specific; the frontend
# capability map gates them to local backends only.
if top_n_sigma is not None:
payload["top_n_sigma"] = top_n_sigma
if repeat_last_n is not None:
@ -4357,9 +4352,6 @@ class LlamaCppBackend:
payload["mirostat_tau"] = mirostat_tau
if mirostat_eta is not None:
payload["mirostat_eta"] = mirostat_eta
# DRY / XTC / min_keep / ignore_eos / min_tokens — same llama.cpp-
# only fields as above. Each is gated `is not None` so explicit
# 0 / False values still reach the wire.
if dry_multiplier is not None:
payload["dry_multiplier"] = dry_multiplier
if dry_base is not None:
@ -4378,8 +4370,6 @@ class LlamaCppBackend:
payload["ignore_eos"] = ignore_eos
if min_tokens is not None:
payload["min_tokens"] = min_tokens
# vLLM output-shape knobs — forwarded `is not None` so user
# opt-outs (skip_special_tokens=False etc) still reach the wire.
if skip_special_tokens is not None:
payload["skip_special_tokens"] = skip_special_tokens
if spaces_between_special_tokens is not None:
@ -4388,7 +4378,6 @@ class LlamaCppBackend:
payload["include_stop_str_in_output"] = include_stop_str_in_output
if truncate_prompt_tokens is not None:
payload["truncate_prompt_tokens"] = truncate_prompt_tokens
# llama.cpp context / KV-cache / instrumentation knobs.
if n_keep is not None:
payload["n_keep"] = n_keep
if n_probs is not None:

View file

@ -872,231 +872,147 @@ class ChatCompletionRequest(BaseModel):
None,
ge = 0.0,
le = 1.0,
description = (
"Locally typical sampling (llama.cpp `typ_p`). 1.0 disables. "
"Local llama-server only — no SaaS provider currently accepts "
"this field, so the frontend capability map gates it off for "
"every external provider and the local path forwards it on "
"/v1/chat/completions."
),
description = "llama.cpp `typ_p`. 1.0 disables. Local only.",
)
top_n_sigma: Optional[float] = Field(
None,
description = (
"llama.cpp `top_n_sigma` sampler. -1.0 disables (server "
"default). Local only — no SaaS provider accepts it."
),
description = "llama.cpp `top_n_sigma`. -1 disables. Local only.",
)
repeat_last_n: Optional[int] = Field(
None,
description = (
"llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. "
"Pairs with repetition_penalty. Local only."
),
description = "llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. Local only.",
)
dynatemp_range: Optional[float] = Field(
None,
ge = 0.0,
description = ("llama.cpp `dynatemp_range`. 0.0 disables. Local only."),
description = "llama.cpp `dynatemp_range`. 0 disables. Local only.",
)
dynatemp_exponent: Optional[float] = Field(
None,
ge = 0.0,
description = (
"llama.cpp `dynatemp_exponent`. Local only; pairs with " "dynatemp_range."
),
description = "llama.cpp `dynatemp_exponent`. Pairs with dynatemp_range. Local only.",
)
mirostat: Optional[int] = Field(
None,
ge = 0,
le = 2,
description = (
"llama.cpp `mirostat` mode. 0 = disabled, 1 = Mirostat, "
"2 = Mirostat 2.0. Local only."
),
description = "llama.cpp `mirostat` (0=off, 1=Mirostat, 2=Mirostat 2.0). Local only.",
)
mirostat_tau: Optional[float] = Field(
None,
ge = 0.0,
description = "llama.cpp `mirostat_tau` target entropy. Local only.",
description = "llama.cpp `mirostat_tau`. Local only.",
)
mirostat_eta: Optional[float] = Field(
None,
ge = 0.0,
description = "llama.cpp `mirostat_eta` learning rate. Local only.",
description = "llama.cpp `mirostat_eta`. Local only.",
)
top_a: Optional[float] = Field(
None,
ge = 0.0,
le = 1.0,
description = (
"OpenRouter `top_a` alternate dynamic-top-P. Documented at "
"https://openrouter.ai/docs/api/reference/parameters. "
"OpenRouter-only; other gateways silently drop it."
"OpenRouter `top_a`. OpenRouter-only. "
"https://openrouter.ai/docs/api/reference/parameters"
),
)
dry_multiplier: Optional[float] = Field(
None,
ge = 0.0,
description = (
"llama.cpp DRY (Don't Repeat Yourself) penalty multiplier. "
"0.0 disables (server default). Master switch for the 4-field "
"DRY family — backend only forwards dry_base / dry_allowed_"
"length / dry_penalty_last_n when multiplier > 0. Local only."
"llama.cpp DRY multiplier. 0 disables the 4-field chain "
"(dry_base / dry_allowed_length / dry_penalty_last_n). Local only."
),
)
dry_base: Optional[float] = Field(
None,
ge = 1.0,
description = (
"llama.cpp DRY base value (exponential growth base). Default "
"1.75. Local only; only meaningful when dry_multiplier > 0."
),
description = "llama.cpp DRY base. Default 1.75. Local only.",
)
dry_allowed_length: Optional[int] = Field(
None,
ge = 0,
description = (
"llama.cpp DRY allowed-length threshold. Default 2. Local "
"only; only meaningful when dry_multiplier > 0."
),
description = "llama.cpp DRY allowed-length. Default 2. Local only.",
)
dry_penalty_last_n: Optional[int] = Field(
None,
description = (
"llama.cpp DRY penalty scan window. 0 disables, -1 = ctx-size. "
"Local only; only meaningful when dry_multiplier > 0."
),
description = "llama.cpp DRY scan window. 0 disables, -1 = ctx-size. Local only.",
)
xtc_probability: Optional[float] = Field(
None,
ge = 0.0,
le = 1.0,
description = (
"llama.cpp XTC (eXclude Top Choice) sampler probability. "
"0.0 disables. Master switch for xtc_threshold. Local only."
),
description = "llama.cpp XTC probability. 0 disables; pairs with xtc_threshold. Local only.",
)
xtc_threshold: Optional[float] = Field(
None,
ge = 0.0,
le = 1.0,
description = (
"llama.cpp XTC sampler probability threshold. Default 0.1. "
"Local only; only meaningful when xtc_probability > 0."
),
description = "llama.cpp XTC threshold. Default 0.1. Local only.",
)
min_keep: Optional[int] = Field(
None,
ge = 0,
description = (
"llama.cpp `min_keep` — force min N tokens past every "
"sampler filter. 0 disables (server default). Local only."
),
description = "llama.cpp `min_keep` (force min N past every filter). Local only.",
)
ignore_eos: Optional[bool] = Field(
None,
description = (
"Continue generation past the model's EOS token. Accepted by "
"llama.cpp + vLLM; Ollama's OAI translator drops it. False "
"matches each backend's upstream default."
),
description = "Continue past EOS. llama.cpp + vLLM only.",
)
min_tokens: Optional[int] = Field(
None,
ge = 0,
description = (
"Minimum output tokens before stop sequences / EOS can fire. "
"Accepted by llama.cpp + vLLM; Ollama's OAI translator drops "
"it. 0 disables (server default)."
),
description = "Min output tokens before stop / EOS. llama.cpp + vLLM only.",
)
skip_special_tokens: Optional[bool] = Field(
None,
description = (
"vLLM `skip_special_tokens` (default true). Forward only when "
"false — i.e. user wants raw special tokens in the output. "
"vLLM only; llama-server / Ollama do not document this field."
),
description = "vLLM `skip_special_tokens` (default true). vLLM only.",
)
spaces_between_special_tokens: Optional[bool] = Field(
None,
description = (
"vLLM `spaces_between_special_tokens` (default true). Forward "
"only when false. vLLM only."
),
description = "vLLM `spaces_between_special_tokens` (default true). vLLM only.",
)
include_stop_str_in_output: Optional[bool] = Field(
None,
description = (
"vLLM `include_stop_str_in_output` (default false). Forward "
"only when true — useful for agentic tools that need the "
"matched stop string echoed back. vLLM only."
),
description = "vLLM `include_stop_str_in_output`. Useful for agentic tools. vLLM only.",
)
truncate_prompt_tokens: Optional[int] = Field(
None,
ge = 1,
description = (
"vLLM `truncate_prompt_tokens` — left-truncate the prompt to "
"this many tokens. Useful for long-context overflow. vLLM "
"only; llama-server / Ollama drop this on the OAI path."
),
description = "vLLM `truncate_prompt_tokens` (left-truncate prompt). vLLM only.",
)
n_keep: Optional[int] = Field(
None,
description = (
"llama.cpp `n_keep` — tokens to retain when context overflows. "
"0 disables (server default), -1 keeps the whole prompt. "
"Local llama-server only."
),
description = "llama.cpp `n_keep`. 0 disables, -1 = keep all. Local only.",
)
n_probs: Optional[int] = Field(
None,
ge = 0,
description = (
"llama.cpp `n_probs` — return top-N token probabilities per "
"generated token. 0 disables (server default). Local only."
),
description = "llama.cpp `n_probs` (top-N token probs). 0 disables. Local only.",
)
cache_prompt: Optional[bool] = Field(
None,
description = (
"llama.cpp `cache_prompt` — reuse KV cache across requests "
"with a shared prefix. Default true upstream; forward only "
"when explicitly false (e.g. deterministic benchmarks). "
"Local llama-server only."
),
description = "llama.cpp `cache_prompt` (default true upstream). Local only.",
)
return_tokens: Optional[bool] = Field(
None,
description = (
"llama.cpp `return_tokens` — include raw token IDs in the "
"response. Debug. Local only."
),
description = "llama.cpp `return_tokens` (debug). Local only.",
)
timings_per_token: Optional[bool] = Field(
None,
description = (
"llama.cpp `timings_per_token` — include per-token speed "
"metrics in the streaming response. Local only."
),
description = "llama.cpp `timings_per_token` (perf debug). Local only.",
)
post_sampling_probs: Optional[bool] = Field(
None,
description = (
"llama.cpp `post_sampling_probs` — return token probabilities "
"AFTER the sampler chain runs (useful for sampler-tuning). "
"Local only."
),
description = "llama.cpp `post_sampling_probs` (sampler debug). Local only.",
)
fast_mode: Optional[bool] = Field(
None,
description = (
"[x-unsloth] Anthropic fast-mode toggle. On Claude Opus 4.6 / "
"4.7 adds the `fast-mode-2026-02-01` beta header and sends "
"`speed: 'fast'` for higher OTPS at premium pricing. Silently "
"ignored on every other model + provider. See "
"[x-unsloth] Anthropic fast-mode on Opus 4.6 / 4.7. Adds the "
"fast-mode-2026-02-01 beta header + speed:'fast' for higher "
"OTPS at premium pricing. Silently dropped elsewhere. "
"https://platform.claude.com/docs/en/build-with-claude/fast-mode"
),
)

View file

@ -5148,21 +5148,18 @@ def _build_passthrough_payload(
body["presence_penalty"] = presence_penalty
# llama-server's /v1/chat/completions accepts the standard OpenAI
# fields. parallel_tool_calls is a no-op on llama-server today but
# is forwarded so a future release picks it up automatically.
# forwarded so a future release picks it up automatically.
# Each field below gated `is not None` so explicit 0 / False reach
# the wire; llama-server silently ignores unknown fields, Ollama's
# OAI translator drops everything outside the OAI subset.
if frequency_penalty is not None:
body["frequency_penalty"] = frequency_penalty
if seed is not None:
body["seed"] = seed
if parallel_tool_calls is not None:
body["parallel_tool_calls"] = parallel_tool_calls
# llama.cpp-specific locally-typical sampling (typ_p in the sampler
# chain). No SaaS provider accepts this; the frontend capability map
# gates it to local only.
if typical_p is not None:
body["typical_p"] = typical_p
# Extended llama.cpp sampler chain. All llama.cpp-specific; the
# frontend capability map gates them to local backends only. Server
# silently ignores fields it doesn't recognise.
if top_n_sigma is not None:
body["top_n_sigma"] = top_n_sigma
if repeat_last_n is not None:
@ -5177,10 +5174,6 @@ def _build_passthrough_payload(
body["mirostat_tau"] = mirostat_tau
if mirostat_eta is not None:
body["mirostat_eta"] = mirostat_eta
# DRY / XTC / min_keep / ignore_eos / min_tokens — llama-server
# specific (DRY+XTC+min_keep) plus vLLM-shared (ignore_eos+min_tokens).
# Forwarded `is not None` so explicit 0 / False values still reach
# the wire; the OAI translator on Ollama drops these silently.
if dry_multiplier is not None:
body["dry_multiplier"] = dry_multiplier
if dry_base is not None:
@ -5199,10 +5192,6 @@ def _build_passthrough_payload(
body["ignore_eos"] = ignore_eos
if min_tokens is not None:
body["min_tokens"] = min_tokens
# vLLM output-shape knobs + llama.cpp context / KV / instrumentation
# knobs. Per-backend capability gating on the frontend prevents these
# from being forwarded to wires that don't recognise them; here we
# only enforce the `is not None` rule so explicit defaults still pass.
if skip_special_tokens is not None:
body["skip_special_tokens"] = skip_special_tokens
if spaces_between_special_tokens is not None:
@ -5224,15 +5213,12 @@ def _build_passthrough_payload(
if post_sampling_probs is not None:
body["post_sampling_probs"] = post_sampling_probs
if response_format is not None:
# llama-server applies a GBNF grammar derived from the JSON schema
# when response_format is present. Field is documented flat at the
# request root (tools/server/README.md), which is also what the
# OpenAI SDK produces by spreading extra_body into the body top.
# llama-server applies a GBNF grammar from the JSON schema.
# Field is documented flat at the request root.
body["response_format"] = response_format
if chat_template_kwargs is not None:
# Propagate reasoning / template overrides (e.g. enable_thinking)
# so llama-server renders the Jinja template in the mode the caller
# asked for instead of whatever default the model was loaded with.
# Reasoning / template overrides (e.g. enable_thinking) so
# llama-server renders the Jinja template in the requested mode.
body["chat_template_kwargs"] = chat_template_kwargs
return body

View file

@ -1748,29 +1748,24 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
params.parallelToolCalls === false
? { parallel_tool_calls: false }
: {}),
// llama.cpp `typ_p`. External providers all have
// capabilities.typicalP=false; only the permissive local
// buckets (custom/vllm/ollama/llama_cpp) opt in. `null`
// (unset) or `1.0` (llama-server default) is a no-op.
// llama.cpp / vLLM / OpenRouter extras. Each is gated by
// (a) the active provider's capability flag and (b) a
// non-default value, so only meaningful knobs hit the wire.
...(externalCapabilities?.typicalP &&
params.typicalP !== null &&
params.typicalP !== 1
? { typical_p: params.typicalP }
: {}),
// llama.cpp `top_n_sigma`. -1 disables (server default);
// only forward meaningful values.
...(externalCapabilities?.topNSigma &&
params.topNSigma !== null &&
params.topNSigma !== -1
? { top_n_sigma: params.topNSigma }
: {}),
// llama.cpp `repeat_last_n`. Pairs with repetition_penalty.
...(externalCapabilities?.repeatLastN &&
params.repeatLastN !== null
? { repeat_last_n: params.repeatLastN }
: {}),
// llama.cpp dynamic-temperature. Only forward when the
// user opted in (range > 0).
// Dynatemp: range>0 unlocks both fields.
...(externalCapabilities?.dynatempRange &&
params.dynatempRange !== null &&
params.dynatempRange > 0
@ -1782,8 +1777,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
: {}),
}
: {}),
// llama.cpp Mirostat. Mode 0 disables; only forward the
// sub-params when mode is enabled.
// Mirostat: mode!=0 unlocks tau + eta.
...(externalCapabilities?.mirostat &&
params.mirostat !== null &&
params.mirostat !== 0
@ -1799,16 +1793,12 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
: {}),
}
: {}),
// OpenRouter `top_a` (alternate dynamic top-P). Documented
// range [0, 1]; 0 disables.
...(externalCapabilities?.topA &&
params.topA !== null &&
params.topA > 0
? { top_a: params.topA }
: {}),
// llama.cpp DRY sampler. dry_multiplier=0 disables the
// whole chain; only forward the paired fields when the
// master is set to a meaningful value.
// DRY: multiplier>0 unlocks the 4-field chain.
...(externalCapabilities?.dryMultiplier &&
params.dryMultiplier !== null &&
params.dryMultiplier > 0
@ -1828,7 +1818,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
: {}),
}
: {}),
// llama.cpp XTC sampler. xtc_probability=0 disables.
// XTC: probability>0 unlocks threshold.
...(externalCapabilities?.xtcProbability &&
params.xtcProbability !== null &&
params.xtcProbability > 0
@ -1840,29 +1830,21 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
: {}),
}
: {}),
// llama.cpp `min_keep` (force min N tokens past filters).
// 0 is the upstream default; only forward when set higher.
...(externalCapabilities?.minKeep &&
params.minKeep !== null &&
params.minKeep > 0
? { min_keep: params.minKeep }
: {}),
// Continue past EOS. llama.cpp + vLLM only; forward only
// when explicitly true (false matches upstream default).
...(externalCapabilities?.ignoreEos && params.ignoreEos === true
? { ignore_eos: true }
: {}),
// Minimum output tokens before stop / EOS can fire.
// 0 = upstream default; only forward when set higher.
...(externalCapabilities?.minTokens &&
params.minTokens !== null &&
params.minTokens > 0
? { min_tokens: params.minTokens }
: {}),
// vLLM output-shape knobs. Upstream defaults:
// skip_special_tokens=true, spaces_between_special_tokens=true,
// include_stop_str_in_output=false. Forward only when user
// opted away from the default to avoid no-op wire bloat.
// vLLM output-shape: default true for skip/spaces, false
// for include-stop. Forward only on user opt-out.
...(externalCapabilities?.skipSpecialTokens &&
params.skipSpecialTokens === false
? { skip_special_tokens: false }
@ -1880,8 +1862,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
params.truncatePromptTokens > 0
? { truncate_prompt_tokens: params.truncatePromptTokens }
: {}),
// llama.cpp-only context / KV-cache / instrumentation knobs.
// n_keep accepts -1 (= keep all) so the gate is != 0.
// n_keep accepts -1 (keep all), so the gate is != 0.
...(externalCapabilities?.nKeep &&
params.nKeep !== null &&
params.nKeep !== 0
@ -1924,14 +1905,10 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
enable_tools: true,
enabled_tools: [
...(webSearchEnabledForThisTurn ? ["web_search"] : []),
// web_fetch has its own Fetch pill, independent
// of Search. Anthropic-only today.
// web_fetch has its own pill (Anthropic-only).
...(webFetchEnabledForThisTurn ? ["web_fetch"] : []),
...(codeExecEnabledForThisTurn ? ["code_execution"] : []),
// OpenAI Responses-API only: `image_generation`
// returns inline image_generation_call output
// items; the backend's _stream_openai_responses
// path translates them to assistant tool events.
// image_generation: OpenAI Responses-API only.
...(imageGenerationEnabledForThisTurn
? ["image_generation"]
: []),
@ -1967,11 +1944,8 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
externalProvider.enablePromptCaching ?? true,
}
: {}),
// Anthropic-only: pass the cache TTL the user picked in
// Configuration → Provider. Omitted = inherit the default
// 5-minute pool. The backend's `_stream_anthropic` only
// attaches `cache_control.ttl` when the value is one of
// "5m" / "1h" (see external_provider.py near line 1375),
// Anthropic-only cache TTL. Backend's _stream_anthropic
// only attaches cache_control.ttl when value is "5m"/"1h",
// so unknown values are a no-op end-to-end.
...(supportsProviderPromptCacheTtl(
externalProvider.providerType,
@ -1980,9 +1954,8 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
isPromptCacheTtl(externalProvider.promptCacheTtl)
? { prompt_cache_ttl: externalProvider.promptCacheTtl }
: {}),
// Anthropic fast mode (Opus 4.6 / 4.7 only); backend
// silently drops on unsupported models as a second
// line of defence.
// Fast mode (Anthropic Opus 4.6 / 4.7). Backend drops on
// unsupported models as second defence.
...(params.fastMode &&
providerSupportsFastMode(
externalProvider.providerType,
@ -2015,24 +1988,15 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
min_p: params.minP,
repetition_penalty: params.repetitionPenalty,
presence_penalty: params.presencePenalty,
// Optional sampling extensions; local llama-server already
// accepts `stop` / `seed` / `frequency_penalty` via
// _build_passthrough_payload (routes/inference.py:4884) and
// silently ignores fields it does not recognise. llama-server
// documents `parallel_tool_calls` defaulting to FALSE
// (https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md);
// forward the user's preference unconditionally so the
// default-on UI state actually enables parallel tool calls
// there. External providers default to true everywhere; the
// external branch above keeps its opt-in-on-false shape.
// llama-server accepts the standard OAI extensions via
// _build_passthrough_payload and silently ignores unknown
// fields. parallel_tool_calls defaults to false upstream so
// we forward unconditionally to honour the default-on UI.
...(params.frequencyPenalty !== 0
? { frequency_penalty: params.frequencyPenalty }
: {}),
...(params.seed !== null ? { seed: params.seed } : {}),
...(params.stop.length > 0 ? { stop: params.stop } : {}),
// llama.cpp `typ_p`. Local only — external providers gate
// it off via capability map. `null` (unset) or 1.0 (server
// default) is a no-op so we forward only meaningful values.
...(params.typicalP !== null && params.typicalP !== 1
? { typical_p: params.typicalP }
: {}),
@ -2061,7 +2025,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
: {}),
}
: {}),
// llama.cpp DRY sampler — dry_multiplier=0 disables the chain.
// DRY: multiplier>0 unlocks the 4-field chain.
...(params.dryMultiplier !== null && params.dryMultiplier > 0
? {
dry_multiplier: params.dryMultiplier,
@ -2076,7 +2040,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
: {}),
}
: {}),
// llama.cpp XTC sampler — xtc_probability=0 disables.
// XTC: probability>0 unlocks threshold.
...(params.xtcProbability !== null && params.xtcProbability > 0
? {
xtc_probability: params.xtcProbability,
@ -2088,15 +2052,13 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
...(params.minKeep !== null && params.minKeep > 0
? { min_keep: params.minKeep }
: {}),
// ignore_eos / min_tokens are shared with vLLM but local
// llama-server accepts them too.
...(params.ignoreEos === true ? { ignore_eos: true } : {}),
...(params.minTokens !== null && params.minTokens > 0
? { min_tokens: params.minTokens }
: {}),
// Local llama-server / vLLM / Ollama route. Per-backend
// capability gating handles the silent-drop story; here we
// forward only when the value diverges from upstream default.
// Forward only when value diverges from upstream default;
// per-backend capability gating decides whether the wire
// even sees these.
...(params.skipSpecialTokens === false
? { skip_special_tokens: false }
: {}),

View file

@ -1,163 +1,77 @@
// SPDX-License-Identifier: AGPL-3.0-only
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
/**
* Per-provider sampling parameter capability matrix.
*
* Values are derived from each provider's published chat-completion docs as of
* 2026-05. They describe which of our UI knobs map cleanly onto the provider's
* request body; the panel hides params a provider does not accept so users
* cannot dial a value that gets silently dropped or rejected.
*
* "Local" models (anything that is not an external provider) are represented by
* a null capability every knob renders for them.
*/
// NB: when adding a new sampling knob, default it to `false` on every
// SaaS provider in PROVIDER_CAPABILITIES below (only local backends
// + the permissive {custom, vllm, ollama, llama_cpp, openrouter}
// providers should expose llama.cpp-specific samplers).
// Per-provider sampling capability matrix. Sourced from each
// provider's chat-completion docs (2026-05). The panel hides params
// the active provider does not accept so users never move a knob that
// would be silently dropped or rejected.
// When adding a new knob: default it to false on every SaaS bucket;
// only local backends + the permissive openrouter bucket should
// expose llama.cpp-specific samplers.
export interface ProviderCapabilities {
/**
* Temperature sampling. Reasoning-class models (OpenAI's gpt-5.x / o3 via
* /v1/responses) reject this with `Unsupported parameter`.
*/
/** OpenAI gpt-5.x / o-series reject via /v1/responses. */
temperature: boolean;
/** Nucleus (top_p) sampling. Same restriction as `temperature` on OpenAI. */
topP: boolean;
/** top-k token sampling (only Anthropic on the providers we ship). */
/** Anthropic only among SaaS providers. */
topK: boolean;
/** min-p token cutoff (no SaaS provider currently exposes this). */
minP: boolean;
/** Repetition penalty (no SaaS provider currently exposes this). */
repetitionPenalty: boolean;
/** OpenAI-style presence penalty. */
presencePenalty: boolean;
/**
* OpenAI-style frequency penalty. Accepted by Chat Completions only.
* Anthropic and the OpenAI Responses family both reject it (the latter
* with `Unsupported parameter`).
*/
/** OAI Chat only; rejected by Responses + Anthropic. */
frequencyPenalty: boolean;
/**
* Best-effort determinism seed. Accepted by OpenAI Chat Completions and
* most OpenAI-compatible local backends (vLLM, llama.cpp). Rejected by
* the Responses family and silently dropped by Anthropic.
*/
/** OAI Chat + OAI-compat. Responses + Anthropic drop. */
seed: boolean;
/**
* Custom stop sequences. Maps to `stop` (OpenAI Chat) or `stop_sequences`
* (Anthropic). Not accepted by the Responses family.
*/
/** Not accepted by Responses; mapped to `stop_sequences` on Anthropic. */
stop: boolean;
/**
* Provider service tier (`auto` / `standard_only` for Anthropic,
* `auto`/`default`/`flex`/`priority`(+`scale`) for OpenAI). See
* {@link getServiceTierOptions} for the legal values per provider.
*/
/** Per-provider enum, see getServiceTierOptions. */
serviceTier: boolean;
/**
* Whether the provider supports turning off parallel tool dispatch.
* Maps to `parallel_tool_calls: false` on both OpenAI APIs and
* `disable_parallel_tool_use: true` on Anthropic (inverted).
*/
/** Anthropic inverts to `disable_parallel_tool_use`. */
parallelToolCalls: boolean;
/**
* llama.cpp `typ_p` (locally typical sampling). Local llama-server
* only no SaaS provider currently accepts this field. Default is
* `false` for every external provider and `true` only for the local
* permissive {custom, vllm, ollama, llama_cpp} buckets.
*/
/** llama.cpp `typ_p`. */
typicalP: boolean;
/**
* llama.cpp `top_n_sigma` sampler (newer top-sigma cutoff). Local
* only; -1 disables.
* https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md
*/
/** llama.cpp `top_n_sigma`. */
topNSigma: boolean;
/**
* llama.cpp repetition window (`repeat_last_n`). Pairs with
* `repeat_penalty`. Local only; 0 disables, -1 = ctx-size.
*/
/** llama.cpp `repeat_last_n`. */
repeatLastN: boolean;
/**
* llama.cpp dynamic temperature range (`dynatemp_range`). Local
* only; 0.0 disables.
*/
/** llama.cpp `dynatemp_range`. */
dynatempRange: boolean;
/**
* llama.cpp dynamic temperature exponent (`dynatemp_exponent`).
* Local only. Paired with dynatempRange.
*/
/** llama.cpp `dynatemp_exponent`. */
dynatempExponent: boolean;
/**
* llama.cpp Mirostat sampling mode (`mirostat`). Local only.
* 0 = disabled, 1 = Mirostat, 2 = Mirostat 2.0.
*/
/** llama.cpp `mirostat` (0/1/2). */
mirostat: boolean;
/**
* llama.cpp Mirostat target entropy (`mirostat_tau`). Local only.
* Only meaningful when mirostat != 0.
*/
mirostatTau: boolean;
/**
* llama.cpp Mirostat learning rate (`mirostat_eta`). Local only.
* Only meaningful when mirostat != 0.
*/
mirostatEta: boolean;
/**
* OpenRouter `top_a` (alternate dynamic-top-P). Documented at
* https://openrouter.ai/docs/api/reference/parameters. Other
* gateways silently drop it; we surface it only for openrouter.
*/
/** OpenRouter `top_a`. https://openrouter.ai/docs/api/reference/parameters */
topA: boolean;
/**
* llama.cpp DRY (Don't Repeat Yourself) repetition multiplier.
* Master switch for the 4-field DRY sampler family. Local llama-
* server only vLLM / Ollama do not implement DRY.
*/
/** llama.cpp DRY (4 fields). dryMultiplier is the master switch. */
dryMultiplier: boolean;
/** llama.cpp DRY base value (exponential growth base). Local only. */
dryBase: boolean;
/** llama.cpp DRY allowed token-extension threshold. Local only. */
dryAllowedLength: boolean;
/** llama.cpp DRY penalty scan window. Local only. */
dryPenaltyLastN: boolean;
/** llama.cpp XTC (eXclude Top Choice) sampler probability. Local only. */
/** llama.cpp XTC (2 fields). xtcProbability is the master switch. */
xtcProbability: boolean;
/** llama.cpp XTC sampler probability threshold. Local only. */
xtcThreshold: boolean;
/** llama.cpp `min_keep` (force min N tokens past every filter). Local only. */
/** llama.cpp `min_keep`. */
minKeep: boolean;
/**
* Continue generating past EOS. llama.cpp + vLLM accept this on the
* /v1/chat/completions surface; Ollama's OAI translator drops it.
*/
/** llama.cpp + vLLM. Ollama OAI translator drops it. */
ignoreEos: boolean;
/**
* Minimum output tokens before stop sequences / EOS can fire.
* llama.cpp + vLLM accept this; Ollama's OAI translator drops it.
*/
/** llama.cpp + vLLM. Ollama OAI translator drops it. */
minTokens: boolean;
/** vLLM `skip_special_tokens` (vLLM SamplingParams). vLLM only. */
/** vLLM only. */
skipSpecialTokens: boolean;
/** vLLM `spaces_between_special_tokens`. vLLM only. */
spacesBetweenSpecialTokens: boolean;
/** vLLM `include_stop_str_in_output`. vLLM only — useful for agentic tools. */
/** vLLM only. Useful for agentic tools. */
includeStopStrInOutput: boolean;
/** vLLM `truncate_prompt_tokens` — left-truncate the prompt. vLLM only. */
/** vLLM only. Left-truncate the prompt. */
truncatePromptTokens: boolean;
/** llama.cpp `n_keep` — tokens to retain on context overflow. llama.cpp only. */
/** llama.cpp `n_keep` / `n_probs`. */
nKeep: boolean;
/** llama.cpp `n_probs` — return top-N token probabilities. llama.cpp only. */
nProbs: boolean;
/** llama.cpp `cache_prompt` — KV-cache reuse. llama.cpp only. */
/** llama.cpp `cache_prompt`. */
cachePrompt: boolean;
/** llama.cpp `return_tokens` — debug. llama.cpp only. */
/** llama.cpp debug flags. */
returnTokens: boolean;
/** llama.cpp `timings_per_token` — performance debug. llama.cpp only. */
timingsPerToken: boolean;
/** llama.cpp `post_sampling_probs` — sampling-chain debug. llama.cpp only. */
postSamplingProbs: boolean;
}
@ -265,36 +179,24 @@ export function clampReasoningEffortToLevels(
*/
export const EXTERNAL_MAX_OUTPUT_TOKENS = 32768;
/**
* Per-model max-output caps from each provider's docs:
* OpenAI: developers.openai.com/api/docs/models/gpt-5.5
* Anthropic: platform.claude.com/docs/en/about-claude/models
* Gemini: ai.google.dev/gemini-api/docs/models/gemini-3.1-pro-preview
* DeepSeek: api-docs.deepseek.com/quick_start/pricing (V4 family)
* Local-model path is unaffected.
*/
// Per-model max-output caps from each provider's docs (verified May 2026):
// OpenAI: developers.openai.com/api/docs/models/<model>
// Anthropic: platform.claude.com/docs/en/about-claude/models/overview
// Gemini: ai.google.dev/gemini-api/docs/models
// DeepSeek: api-docs.deepseek.com/quick_start/pricing
// Order matters: list specific chat-class ids before broader gpt-5 /
// claude-opus-4 entries so the longer prefix wins via .startsWith().
const EXTERNAL_MAX_OUTPUT_TOKENS_BY_MODEL: Array<{
providerType: string;
prefixes: readonly string[];
cap: number;
}> = [
// OpenAI per-model output caps from developers.openai.com per-model
// pages (cross-checked against the Azure Foundry reasoning table).
// Order matters: list the 16k chat-latest variants first so the
// broader gpt-5 / gpt-4 entries don't shadow them.
// gpt-5.3-chat-latest / gpt-5.1-chat = 16384 (chat-class)
// gpt-5.5* / gpt-5.4* / gpt-5.3-codex / gpt-5.2 / gpt-5.1 / gpt-5
// / gpt-5-codex / gpt-5-pro = 128000
// o1 / o3 / o3-pro / o4-mini / codex-mini = 100000
{ providerType: "openai", prefixes: ["gpt-5.3-chat-latest", "gpt-5.1-chat"], cap: 16384 },
{ providerType: "openai", prefixes: ["gpt-5"], cap: 128000 },
{ providerType: "openai", prefixes: ["o1", "o3", "o4", "codex-mini"], cap: 100000 },
// Anthropic — overview table at
// platform.claude.com/docs/en/about-claude/models/overview. Opus 4.7
// and Opus 4.6 BOTH ship 128k Max output (the legacy-table row for
// 4.6 reads "128k tokens"); Sonnet 4.6 / Sonnet 4.5 / Sonnet 4 / Opus
// 4.5 / Haiku 4.5 ship 64k; Opus 4.1 / Opus 4 ship 32k (covered by
// the 32k default below).
// Anthropic Opus 4.6 + 4.7 ship 128k Max output; Sonnet 4.5/4.6/4 +
// Opus 4.5 + Haiku 4.5 ship 64k; Opus 4.1 / Opus 4 fall through to
// the 32k EXTERNAL_MAX_OUTPUT_TOKENS default.
{
providerType: "anthropic",
prefixes: ["claude-opus-4-7", "claude-opus-4-6"],
@ -311,13 +213,12 @@ const EXTERNAL_MAX_OUTPUT_TOKENS_BY_MODEL: Array<{
],
cap: 64000,
},
// Gemini
{
providerType: "gemini",
prefixes: ["gemini-3", "gemini-pro", "gemini-flash"],
cap: 65536,
},
// DeepSeek (V4: deepseek-chat / deepseek-reasoner alias V4-flash).
// V4: deepseek-chat / deepseek-reasoner alias V4-flash.
{ providerType: "deepseek", prefixes: ["deepseek"], cap: 384000 },
];
@ -361,33 +262,16 @@ function _inferProviderFromOpenrouterId(
return null;
}
/**
* Whether the external provider offers a built-in web-search tool that the
* model invokes server-side. When `true`, the chat composer's Search button
* is available for that provider and the chat-adapter forwards
* `enable_tools: true, enabled_tools: ["web_search"]` on the request the
* backend routes the call through the provider's tool schema:
* - OpenAI: `tools: [{type: "web_search"}]` on /v1/responses
* - Anthropic: `tools: [{type: "web_search_20250305", name: "web_search",
* max_uses: 5}]` on /v1/messages
* - OpenRouter: `plugins: [{id: "web"}]` on /v1/chat/completions (the
* router's universal web-search shape; works for every
* underlying model including the `openrouter/free` router).
* - Kimi: `tools: [{type: "builtin_function", function: {name:
* "$web_search"}}]` with `thinking: {type:
* "disabled"}`. Requires a client round-trip:
* the first call returns the search args; the backend
* echoes them back as a role=tool message; the second
* call streams the answer. Handled in
* _stream_kimi_web_search on the backend.
*
* Mistral is intentionally excluded: their `web_search` connector lives on
* the Agents API (`/v1/agents` + `/v1/conversations`), not chat completions,
* and returns `"WebSearchTool connector is not supported"` if injected into
* /v1/chat/completions. Wiring it would require a dedicated Agents streaming
* path. Gemini's grounded-search can be added with the same pattern when
* matching backend translation lands.
*/
// Gates the composer's Search button. Backend translates
// enable_tools:["web_search"] into each provider's tool schema:
// OpenAI: tools:[{type:"web_search"}] on /v1/responses
// Anthropic: tools:[{type:"web_search_20250305", max_uses:5}] on /v1/messages
// OpenRouter: plugins:[{id:"web"}] (router's universal shape)
// Kimi: $web_search builtin (two-call round trip via
// _stream_kimi_web_search)
// Mistral excluded: their web_search is on the Agents API, not chat
// completions, and 400s if injected. Gemini grounded-search needs
// matching backend translation first.
export function providerSupportsBuiltinWebSearch(
providerType: string | null | undefined,
): boolean {
@ -399,24 +283,18 @@ export function providerSupportsBuiltinWebSearch(
);
}
/**
* Whether the external provider exposes a server-side web_fetch tool
* (single URL, text or PDF) emitting a document block. Anthropic-only
* today (`web_fetch_20250910` / `web_fetch_20260209`). Gates the
* composer's standalone Fetch pill, independent of Search.
*/
// Anthropic-only server-side web_fetch tool
// (web_fetch_20250910 / _20260209). Gates the composer's Fetch pill.
export function providerSupportsBuiltinWebFetch(
providerType: string | null | undefined,
): boolean {
return providerType === "anthropic";
}
/**
* Whether the active provider + model supports Anthropic fast-mode
* (`speed: "fast"` + `fast-mode-2026-02-01` header). Opus 4.6 / 4.7
* only per https://platform.claude.com/docs/en/build-with-claude/fast-mode.
* Backend silently drops on unsupported models as a second defence.
*/
// Anthropic fast-mode (`speed:"fast"` + fast-mode-2026-02-01 header).
// Opus 4.6 / 4.7 only per
// https://platform.claude.com/docs/en/build-with-claude/fast-mode.
// Backend silently drops on unsupported models as a second defence.
const ANTHROPIC_FAST_MODE_MODEL_PREFIXES = [
"claude-opus-4-7",
"claude-opus-4-6",
@ -428,38 +306,21 @@ export function providerSupportsFastMode(
): boolean {
if (providerType !== "anthropic") return false;
if (!modelId) return false;
// Family boundary ("" or "-") required so IDs like "claude-opus-4-70"
// / "claude-opus-4-7b" do not match.
// Family boundary required so "claude-opus-4-70" doesn't match.
return ANTHROPIC_FAST_MODE_MODEL_PREFIXES.some(
(prefix) => modelId === prefix || modelId.startsWith(`${prefix}-`),
);
}
/**
* Whether the selected external provider/model exposes a server-side
* code-execution tool. Two providers ship one today:
*
* - **Anthropic** (`code_execution_20250825`): Python + bash +
* str_replace-based file edits inside a 5 GB sandboxed container
* per request. Documented at
* https://platform.claude.com/docs/en/agents-and-tools/tool-use/code-execution-tool
*
* - **OpenAI cloud** (`shell` on /v1/responses): bash inside a
* reusable container; we auto-create one on the first turn of a
* chat thread and reference it on subsequent turns via the
* thread's stored `openaiCodeExecContainerId`. Documented at
* https://developers.openai.com/api/docs/guides/tools-shell
*
* Returns false for every other provider. The backend additionally
* gates the OpenAI shell tool on `is_openai_cloud` so custom
* OpenAI-compat servers (ollama / llama.cpp / vLLM) that also report
* `provider_type="openai"` never receive the tool but in practice
* none of those catalogs surface the `gpt-5.5` ids anyway, so the
* frontend prefix match is enough.
*
* v1 wires the tools themselves; file uploads (Anthropic
* `container_upload` / OpenAI `input_file`) are a deliberate follow-up.
*/
// Server-side code-execution tools:
// Anthropic code_execution_20250825 (Python + bash + str_replace in
// a 5 GB sandbox).
// OpenAI cloud `shell` on /v1/responses (bash in a reusable container
// referenced via openaiCodeExecContainerId across turns).
// Backend also gates OpenAI on is_openai_cloud so custom OAI-compat
// servers reporting provider_type="openai" can't accidentally get the
// shell tool. File uploads (container_upload / input_file) are
// follow-up work.
const ANTHROPIC_CODE_EXECUTION_MODEL_PREFIXES = [
"claude-opus-4-7",
"claude-opus-4-6",
@ -523,18 +384,9 @@ export function providerSupportsBuiltinCodeExecution(
return false;
}
/**
* Whether the selected external provider/model exposes OpenAI's
* Responses-API server-side image_generation tool. Lit on for OpenAI
* cloud (`api.openai.com`) when the picked model is a Responses-API
* family id (gpt-5.x today). The backend additionally gates on
* `is_openai_cloud`; mirror that here so the pill is hidden on custom
* OpenAI-compat backends (ollama / llama.cpp / vLLM) that report
* `provider_type="openai"` but would 400 on a `{type:"image_generation"}`
* tool. See backend/core/inference/external_provider.py near line 2770
* for the dispatch and backend/tests/test_openai_image_generation.py
* for the round-trip coverage.
*/
// OpenAI Responses-API image_generation tool. OpenAI cloud +
// Responses-family ids only; backend mirrors via is_openai_cloud so
// custom OAI-compat servers reporting provider_type="openai" don't 400.
const OPENAI_IMAGE_GENERATION_MODEL_PREFIXES = [
"gpt-5.5-pro",
"gpt-5.5",
@ -561,19 +413,10 @@ export function providerSupportsBuiltinImageGeneration(
);
}
/**
* Per-provider minimum on the outbound max_tokens. Kimi's docs require
* `max_tokens >= 16000` whenever a thinking model is in use so the
* reasoning_content and final answer both fit in the budget anything
* lower truncates the response mid-stream. Other providers don't have a
* documented floor, so they fall through to the generic min of 64 in
* the slider.
*
* The chat-adapter resolves the effective floor on send and bumps the
* outbound max_tokens up to this value if the user's stored maxTokens
* sits below it. The settings panel reflects the same floor as the
* slider min so the displayed value never drifts from what's sent.
*/
// Per-provider min on outbound max_tokens. Kimi thinking models need
// >=16000 or the response truncates mid-stream. Other providers fall
// through to the generic 64. chat-adapter bumps the user's stored
// maxTokens up to the floor on send; the slider min mirrors the same.
const EXTERNAL_MIN_OUTPUT_TOKENS_BY_PROVIDER: Record<string, number> = {
kimi: 16000,
};
@ -662,12 +505,10 @@ const LLAMA_CPP_CAPABILITIES: ProviderCapabilities = {
minKeep: true,
ignoreEos: true,
minTokens: true,
// vLLM-only output-shape knobs — llama-server does not document them.
skipSpecialTokens: false,
spacesBetweenSpecialTokens: false,
includeStopStrInOutput: false,
truncatePromptTokens: false,
// llama.cpp-only context / KV-cache / instrumentation knobs.
nKeep: true,
nProbs: true,
cachePrompt: true,
@ -676,10 +517,10 @@ const LLAMA_CPP_CAPABILITIES: ProviderCapabilities = {
postSamplingProbs: true,
};
// vLLM's OpenAI-compat endpoint accepts the OpenAI subset plus top_k /
// min_p / repetition_penalty / seed, but not the 8 llama.cpp-only
// extended samplers (vLLM's SamplingParams has no fields for them —
// vllm/sampling_params.py).
// vLLM SamplingParams: OAI subset + top_k/min_p/repetition_penalty/seed
// + the 4 vLLM-only output-shape knobs. No DRY / XTC / mirostat /
// dynatemp / typical_p / min_keep / n_keep / n_probs / cache_prompt /
// debug flags (none in SamplingParams).
const VLLM_CAPABILITIES: ProviderCapabilities = {
...LLAMA_CPP_CAPABILITIES,
typicalP: false,
@ -690,9 +531,6 @@ const VLLM_CAPABILITIES: ProviderCapabilities = {
mirostat: false,
mirostatTau: false,
mirostatEta: false,
// vLLM's SamplingParams has no DRY / XTC / min_keep fields (only
// llama-server implements them). Keep ignoreEos + minTokens on:
// both are documented vLLM SamplingParams fields.
dryMultiplier: false,
dryBase: false,
dryAllowedLength: false,
@ -700,12 +538,10 @@ const VLLM_CAPABILITIES: ProviderCapabilities = {
xtcProbability: false,
xtcThreshold: false,
minKeep: false,
// vLLM-only output-shape knobs — flip the LLAMA_CPP defaults.
skipSpecialTokens: true,
spacesBetweenSpecialTokens: true,
includeStopStrInOutput: true,
truncatePromptTokens: true,
// llama.cpp-only instrumentation knobs — vLLM has no analog.
nKeep: false,
nProbs: false,
cachePrompt: false,
@ -714,38 +550,29 @@ const VLLM_CAPABILITIES: ProviderCapabilities = {
postSamplingProbs: false,
};
// Ollama is stricter than vLLM. Studio reaches Ollama via the OpenAI-
// compat /v1/chat/completions transport, and Ollama's translator
// (ollama/openai/openai.go FromChatRequest) only copies the documented
// OpenAI subset — top_k / min_p / repetition_penalty are silently
// DROPPED on that path even though native /api/chat would forward them
// through the `options` bag. Hide them so users don't move a slider
// the wire never carries.
// Ollama OAI translator (openai/openai.go FromChatRequest) only copies
// the documented OpenAI subset on /v1/chat/completions — top_k / min_p
// / repetition_penalty / ignore_eos / min_tokens / the 4 vLLM output
// knobs all silently drop on this path. (Native /api/chat would forward
// them via `options`, but Studio uses /v1.)
const OLLAMA_CAPABILITIES: ProviderCapabilities = {
...VLLM_CAPABILITIES,
topK: false,
minP: false,
repetitionPenalty: false,
// Ollama's OAI translator (openai/openai.go FromChatRequest) doesn't
// forward ignore_eos or min_tokens either — both fields silently drop
// on the /v1/chat/completions path Studio uses.
ignoreEos: false,
minTokens: false,
// The vLLM-specific output-shape knobs are not recognised by the
// Ollama OAI translator; flip them back to false.
skipSpecialTokens: false,
spacesBetweenSpecialTokens: false,
includeStopStrInOutput: false,
truncatePromptTokens: false,
};
// OpenRouter is a router-of-routers: the gateway accepts a wider set
// of OpenAI-style sampling fields than any single upstream supports
// and silently drops what the chosen route does not, per
// https://openrouter.ai/docs/api/reference/parameters. Surface the
// router's full documented set (incl. top_a) and leave the
// llama.cpp-only knobs off (the docs don't list them, so we don't
// either even though many openrouter routes terminate at llama.cpp).
// OpenRouter is a router-of-routers: gateway accepts a wider set of
// OAI-style fields than any single upstream and silently drops what
// the chosen route doesn't. Surface the full documented set (incl.
// top_a) and leave llama.cpp-only knobs off.
// https://openrouter.ai/docs/api/reference/parameters
const OPENROUTER_CAPABILITIES: ProviderCapabilities = {
temperature: true,
topP: true,
@ -788,14 +615,11 @@ const OPENROUTER_CAPABILITIES: ProviderCapabilities = {
postSamplingProbs: false,
};
// Reasoning-class OpenAI models served via /v1/responses fix temperature
// at 1, ignore top_p, and 400 on presence/frequency_penalty / seed. Non
// reasoning models (gpt-4o, gpt-4-turbo, gpt-4, gpt-3.5-turbo) keep the
// full sampling surface even when routed through /v1/responses. See
// https://platform.openai.com/docs/guides/reasoning and the GPT-5 release
// notes; backend dispatch is external_provider._stream_openai_responses.
// The Responses API itself drops `stop`, so we leave that off for all
// OpenAI models regardless of family.
// OpenAI reasoning class via /v1/responses: temperature fixed at 1,
// top_p ignored, 400s on presence/frequency_penalty/seed. Chat-class
// (gpt-4o etc) keeps the full surface even via /v1/responses. Both
// drop `stop` (Responses doesn't surface it).
// https://platform.openai.com/docs/guides/reasoning
const OPENAI_REASONING_CAPABILITIES: ProviderCapabilities = {
temperature: false,
topP: false,
@ -906,14 +730,9 @@ function isOpenAIReasoningModelId(modelId: string | null | undefined): boolean {
return OPENAI_REASONING_MODEL_PREFIXES.some((p) => normalized.startsWith(p));
}
// Mirror of backend _ANTHROPIC_4_7_SAMPLING_REMOVED in
// studio/backend/core/inference/external_provider.py:110. Claude Opus
// 4.7 removed temperature, top_p, and top_k entirely; surfacing the
// sliders would let the user move a control that the backend silently
// strips. Only Opus shipped in the 4.7 generation (Sonnet stops at 4.6,
// Haiku at 4.5 per platform.claude.com/docs/en/about-claude/models/
// overview), so the regex is opus-only. The trailing -4-7[-.]/EOL
// anchor keeps future families (claude-opus-5 etc.) unaffected.
// Mirror of backend _ANTHROPIC_4_7_SAMPLING_REMOVED. Opus 4.7 removed
// temperature/top_p/top_k; only Opus shipped in 4.7. The -4-7[-.]/EOL
// anchor keeps future families (claude-opus-5 etc) unaffected.
const ANTHROPIC_4_7_SAMPLING_REMOVED_REGEX = /^claude-opus-4-7(?:[-.]|$)/i;
function isClaude47SamplingRemoved(modelId: string | null | undefined): boolean {
@ -922,12 +741,9 @@ function isClaude47SamplingRemoved(modelId: string | null | undefined): boolean
return ANTHROPIC_4_7_SAMPLING_REMOVED_REGEX.test(normalized);
}
// DeepSeek reasoning-class models silently ignore temperature, top_p,
// presence_penalty, frequency_penalty and 400 on logprobs/top_logprobs.
// `deepseek-reasoner` is the dedicated thinking model;
// `deepseek-v4-flash` runs reasoning-mode under the same flag as well.
// Match by prefix so future revisions (deepseek-reasoner-2027 etc.)
// continue to gate correctly.
// DeepSeek reasoner ids silently ignore temperature/top_p/presence/
// frequency and 400 on logprobs per the reasoning_model guide. Prefix
// match covers future revisions (deepseek-reasoner-2027 etc).
const DEEPSEEK_REASONING_MODEL_PREFIXES = [
"deepseek-reasoner",
"deepseek-r1",
@ -940,20 +756,13 @@ function isDeepSeekReasoningModelId(modelId: string | null | undefined): boolean
}
const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
// Default OpenAI bucket is reasoning-class (current registry only ships
// gpt-5.x / o3 ids), but per-model resolution in getProviderCapabilities
// upgrades non-reasoning ids (gpt-4o etc.) to OPENAI_CHAT_CAPABILITIES.
// Default to reasoning-class; getProviderCapabilities upgrades
// non-reasoning ids (gpt-4o etc) to OPENAI_CHAT_CAPABILITIES.
openai: OPENAI_REASONING_CAPABILITIES,
// Anthropic's Messages API accepts top_k on 3.x and 4.5/4.6, but Claude
// 4.7 (Opus/Sonnet/Haiku) deprecated it and returns 400 if it is set.
// We surface top_k in the panel for all Anthropic providers and let the
// backend strip it per-model — see _stream_anthropic in
// studio/backend/core/inference/external_provider.py.
// Presence/frequency penalty / seed / logprobs are not part of the
// Messages API on any Claude generation. stop_sequences (Anthropic name
// for `stop`), service_tier (auto|standard_only), and
// disable_parallel_tool_use (inverse of parallel_tool_calls) ARE
// supported.
// Messages API: temperature/top_p/top_k/stop_sequences/service_tier
// (auto|standard_only)/disable_parallel_tool_use. Opus 4.7 strips
// temperature/top_p/top_k via the regex above. No presence/frequency
// penalty / seed / logprobs on any Claude generation.
anthropic: {
temperature: true,
topP: true,
@ -997,15 +806,9 @@ const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
},
mistral: OPENAI_COMPAT_BASE,
gemini: OPENAI_COMPAT_BASE,
// Kimi k2.5/k2.6 are reasoning-class; the API locks temperature
// and top_p to fixed defaults and 400s on any other value:
// "invalid temperature: only 1 is allowed for this model".
// Hide both sliders so the user is not offered knobs the model
// silently overrides. Backend additionally strips these fields via
// PROVIDER_REGISTRY['kimi']['body_omit']. seed and parallel_tool_
// calls are not in Kimi's documented Chat Completion schema
// (https://platform.kimi.ai/docs/api/chat); hide them so users are
// not offered controls that the upstream may silently drop or 400.
// Kimi K2.x locks temperature + top_p ("only 1 is allowed for this
// model"); seed + parallel_tool_calls aren't in the Chat schema
// (platform.kimi.ai/docs/api/chat). Backend strips via body_omit.
kimi: {
temperature: false,
topP: false,
@ -1013,9 +816,7 @@ const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
minP: false,
repetitionPenalty: false,
presencePenalty: true,
// K2.5/K2.6 lock sampling the same way temperature/top_p are
// locked; reviewers report non-default frequency_penalty 400s
// upstream, so hide the slider and strip the field in body_omit.
// K2.x 400s on non-default frequency_penalty; backend strips too.
frequencyPenalty: false,
seed: false,
stop: true,
@ -1050,18 +851,10 @@ const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
timingsPerToken: false,
postSamplingProbs: false,
},
// DeepSeek deprecated presence/frequency penalty and never published
// `seed` or `parallel_tool_calls` in the current chat-completion
// schema — see https://api-docs.deepseek.com/api/create-chat-completion
// (body fields: messages, model, thinking, max_tokens, response_format,
// stop, stream, stream_options, temperature, top_p, tools, tool_choice,
// logprobs, top_logprobs, user_id). Chat-class (deepseek-chat /
// deepseek-v4-flash non-thinking) accepts temperature, top_p, stop;
// reasoning class (deepseek-reasoner / deepseek-v4-flash thinking-mode)
// additionally ignores temperature, top_p, presence_penalty,
// frequency_penalty per
// https://api-docs.deepseek.com/guides/reasoning_model. Per-model
// resolution in getProviderCapabilities downshifts reasoner ids.
// DeepSeek schema (api-docs.deepseek.com/api/create-chat-completion)
// lists temperature/top_p/stop only — no seed or parallel_tool_calls.
// Presence/frequency are deprecated. Reasoner ids additionally ignore
// temperature/top_p; getProviderCapabilities downshifts them.
deepseek: {
temperature: true,
topP: true,
@ -1105,18 +898,10 @@ const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
},
qwen: OPENAI_COMPAT_BASE,
huggingface: OPENAI_COMPAT_BASE,
// OpenRouter surfaces the gateway's documented sampling field set
// (incl. top_a). llama.cpp-specific knobs (typical_p, mirostat,
// dynatemp, top_n_sigma, repeat_last_n) are gated off because the
// OpenRouter API docs do not list them; they would be silently
// dropped on most underlying models.
openrouter: OPENROUTER_CAPABILITIES,
// `llama_cpp` and the permissive `custom` preset terminate at the
// first-party llama-server runtime, so the full sampler chain is
// available. vLLM surfaces the OpenAI subset + top_k/min_p/
// repetition_penalty/seed (no extended llama.cpp samplers). Ollama
// is stricter: its OAI translator drops top_k/min_p/repetition_penalty
// too on the /v1 path.
// llama_cpp + custom: first-party llama-server, full chain.
// vllm: OAI subset + top_k/min_p/repetition_penalty/seed.
// ollama: stricter — OAI translator drops top_k/min_p/rep_pen too.
custom: LLAMA_CPP_CAPABILITIES,
llama_cpp: LLAMA_CPP_CAPABILITIES,
vllm: VLLM_CAPABILITIES,
@ -1125,23 +910,14 @@ const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
const DEFAULT_EXTERNAL_CAPABILITIES = OPENAI_COMPAT_BASE;
/**
* Resolve the capability set for an external provider, optionally
* specialised by model id. Returns `null` for a local model (i.e. when
* `providerType` is null/undefined), which callers should treat as
* "every knob applies".
*
* Per-model specialisations:
* - openai + non-reasoning model (gpt-4o, gpt-4-turbo, gpt-4,
* gpt-3.5-turbo): full sampling surface (OPENAI_CHAT_CAPABILITIES).
* - openai + reasoning model (gpt-5.x, o1, o3, o4): restrictive
* (OPENAI_REASONING_CAPABILITIES).
* - anthropic + claude-opus-4-7: temperature/top_p/top_k stripped to
* match the backend 400-avoidance regex (Sonnet/Haiku 4.7 do not
* ship; only Opus does in the 4.7 generation).
* - deepseek + reasoning model (deepseek-reasoner / r1): hides
* temperature/top_p (silently ignored upstream).
*/
// Per-model specialisations:
// openai + chat-class (gpt-4o, gpt-4-turbo, gpt-4, gpt-3.5):
// full sampling surface (OPENAI_CHAT_CAPABILITIES).
// openai + reasoning (gpt-5.x, o1, o3, o4): OPENAI_REASONING_CAPABILITIES.
// anthropic + claude-opus-4-7: strips temp/top_p/top_k (Opus only
// in 4.7; Sonnet/Haiku don't ship).
// deepseek reasoner: hides temp/top_p (silently ignored upstream).
// Returns null for local models (caller treats as "every knob applies").
export function getProviderCapabilities(
providerType: string | null | undefined,
modelId?: string | null | undefined,
@ -1161,10 +937,9 @@ export function getProviderCapabilities(
}
const DEFAULT_EFFORT_LEVELS = ["low", "medium", "high"] as const;
// OpenRouter ids that have NO non-reasoning mode. `google/gemini-pro-latest`
// used to live here but the gateway 404s the id today
// (https://openrouter.ai/google/gemini-pro-latest); drop it rather than
// re-pin to a versioned id that may rotate again.
// OpenRouter ids with no non-reasoning mode. (google/gemini-pro-latest
// was dropped — gateway 404s; don't re-pin to a versioned id that
// may rotate again.)
const OPENROUTER_MANDATORY_REASONING_MODELS = new Set([
"baidu/cobuddy:free",
"inclusionai/ring-2.6-1t:free",
@ -1196,9 +971,12 @@ const NO_REASONING_CAPS: ReasoningCaps = {
reasoningEffortLevels: DEFAULT_EFFORT_LEVELS,
};
// Order matters: longest/most-specific prefixes first so the find() loop
// in resolveAnthropicReasoningEffortCapabilities lands the right bucket
// before the bare-family fallback ("claude-opus-4") sweeps an id.
// Order matters: longest prefixes first so find() picks the right
// bucket before the bare-family fallback ("claude-opus-4") sweeps.
// Levels per platform.claude.com/docs/en/about-claude/models/overview;
// 4.5 line uses budget_tokens mapped by the backend. Legacy 4.x
// (opus-4-1 / opus-4 / sonnet-4) supports Extended thinking per the
// overview table; sonnet-4 / opus-4 retire 2026-06-15.
const ANTHROPIC_REASONING_MODELS = [
{
prefixes: ["claude-opus-4-7"],
@ -1210,13 +988,9 @@ const ANTHROPIC_REASONING_MODELS = [
},
{
prefixes: ["claude-opus-4-5", "claude-sonnet-4-5", "claude-haiku-4-5"],
// Backend maps semantic levels to manual budget_tokens.
levels: ["none", "low", "medium", "high"],
},
{
// Legacy 4.x models. Live overview lists "Extended thinking = Yes"
// for opus-4-1, sonnet-4, opus-4 (the latter two retire 2026-06-15
// but the registry still surfaces them).
prefixes: ["claude-opus-4-1", "claude-opus-4", "claude-sonnet-4"],
levels: ["none", "low", "medium", "high"],
},
@ -1261,18 +1035,14 @@ const OPENAI_REASONING_MODELS = [
levels: ["medium"],
},
{
// gpt-5.3-codex per dev page lists ONLY low/medium/high/xhigh
// (https://developers.openai.com/api/docs/models/gpt-5.3-codex);
// `none` is not in the codex enum so supportsOff stays false.
// gpt-5.3-codex enum is low/medium/high/xhigh only per dev page.
prefixes: ["gpt-5.3-codex"],
supportsOff: false,
levels: ["low", "medium", "high", "xhigh"],
},
{
// Original gpt-5: minimal is supported, but per Azure footnote ^7^
// "minimal is only supported with the original GPT-5 reasoning
// models. minimal is not supported with gpt-5.1 or greater".
// Listed before the gpt-5.1/5.2 entry so the longer match wins.
// Azure footnote ^7^: minimal supported only on original gpt-5.
// Listed before the bare gpt-5 entry so the longer match wins.
prefixes: ["gpt-5.1", "gpt-5.2"],
supportsOff: true,
levels: ["none", "low", "medium", "high", "xhigh"],
@ -1283,12 +1053,7 @@ const OPENAI_REASONING_MODELS = [
levels: ["minimal", "low", "medium", "high"],
},
{
// o-series reasoning models: o1, o3, o3-mini, o3-pro, o4-mini,
// codex-mini all expose low/medium/high reasoning_effort per
// developers.openai.com/api/docs/models/o3 and the Azure Foundry
// o-series table. Without this entry o1/o4/codex-mini fell into
// NO_REASONING_CAPS and the panel hid the effort slider — a real
// UX regression for users on those ids.
// o-series all accept low/medium/high per dev pages + Azure table.
prefixes: ["o1", "o3", "o4", "codex-mini"],
supportsOff: false,
levels: DEFAULT_EFFORT_LEVELS,
@ -1352,11 +1117,10 @@ function resolveKimiReasoningCapabilities(modelId: string): ExternalReasoningCap
}
function resolveMistralReasoningCapabilities(modelId: string): ExternalReasoningCapabilities {
// Native always-on reasoning family: magistral-* per
// https://mistral.ai/news/magistral and
// https://docs.mistral.ai/studio-api/conversations/reasoning .
// "Always reasons; no parameter needed" — injecting reasoning_effort
// returns 422 upstream. Treat like an OpenAI o-series always-on.
// magistral-* is native always-on (no reasoning_effort param; 422 if
// injected). mistral-{small,medium,vibe-cli}-latest is adjustable
// none/low/medium/high. See docs.mistral.ai/studio-api/conversations/
// reasoning + mistral.ai/news/magistral.
if (
modelId === "magistral-medium-latest" ||
modelId === "magistral-small-latest"
@ -1366,9 +1130,6 @@ function resolveMistralReasoningCapabilities(modelId: string): ExternalReasoning
reasoningAlwaysOn: true,
});
}
// Adjustable reasoning family: three documented levels low/medium/high
// plus the "none" off-switch (Mistral Studio conversations doc). The
// earlier two-level ["none","high"] ladder was wrong.
if (
modelId === "mistral-small-latest" ||
modelId === "mistral-medium-latest" ||
@ -1402,11 +1163,9 @@ function resolveConnectionLevelReasoning(
return null;
}
/**
* resolve external-model thinking capabilities.
* provider-specific matching lives in the OpenAI/Anthropic resolvers.
* other providers default to no reasoning controls.
*/
// Provider-specific matching lives in the per-provider resolvers
// (resolveOpenAI / Anthropic / Kimi / Mistral...). Unknown providers
// default to no reasoning controls.
export function getExternalReasoningCapabilities(
providerType: string | null | undefined,
modelId: string | null | undefined,
@ -1448,9 +1207,8 @@ export function getExternalReasoningCapabilities(
const isOpenRouterProvider = normalizedProvider === "openrouter";
if (isOpenRouterProvider) {
// OpenRouter's unified `reasoning` parameter is accepted on every
// chat-completion request; the gateway silently no-ops for models
// that don't reason. Mandatory-reasoning ids are handled by the
// early guard above; everything else exposes a toggleable control.
// request; gateway no-ops for non-reasoning models. Mandatory ids
// already handled above; everything else exposes a toggle.
return {
supportsReasoning: true,
reasoningStyle: "enable_thinking",

View file

@ -282,33 +282,16 @@ export interface OpenAIChatCompletionsRequest {
* the Anthropic provider with `code_execution` in `enabled_tools`.
*/
anthropic_code_exec_container_id?: string | null;
/**
* OpenAI Chat Completions only; rejected by the Responses family and
* silently dropped by Anthropic. Range -2.0 .. 2.0.
*/
/** OpenAI Chat only. Range -2..2. */
frequency_penalty?: number;
/**
* Best-effort determinism seed. OpenAI Chat / OpenAI-compat backends
* forward it; Responses + Anthropic drop it server-side.
*/
/** OAI Chat + most OAI-compat. Responses + Anthropic drop. */
seed?: number;
/**
* Custom stop sequences. Backend translates to `stop_sequences` for
* Anthropic; OpenAI Chat caps at 4 entries (server-side truncates
* with a warning). Empty arrays are omitted.
*/
/** OAI Chat caps at 4; Anthropic mapped to `stop_sequences`. */
stop?: string[];
/**
* Provider service tier. Anthropic accepts `auto|standard_only`;
* OpenAI Chat + Responses both accept
* `auto|default|flex|scale|priority` per the live `openai-python`
* SDK (`src/openai/types/responses/response_create_params.py`
* declares `Optional[Literal["auto", "default", "flex", "scale",
* "priority"]]`). The wire-side helper in
* `studio/backend/core/inference/external_provider.py` drops values
* that a given provider does not accept; this union stays permissive
* so the request-builder typechecks against
* `InferenceParams.serviceTier` without per-provider narrowing.
* Per-provider enum (see getServiceTierOptions). Union stays
* permissive; external_provider.py drops values the active provider
* doesn't accept.
*/
service_tier?:
| "auto"
@ -317,94 +300,66 @@ export interface OpenAIChatCompletionsRequest {
| "priority"
| "scale"
| "standard_only";
/**
* Whether the provider may dispatch tool calls in parallel.
* OpenAI: forwarded as `parallel_tool_calls`. Anthropic: inverted
* into `disable_parallel_tool_use` server-side. Default `undefined`
* keeps each provider's upstream default.
*/
/** Anthropic inverts to `disable_parallel_tool_use`. */
parallel_tool_calls?: boolean;
/**
* llama.cpp `typ_p` (locally typical sampling). Local llama-server
* only no SaaS provider currently accepts this. 1.0 disables
* (llama-server default). External-provider capability map already
* gates this off, so on the wire it only appears for local + the
* permissive {custom, vllm, ollama, llama_cpp} buckets.
*/
/** llama.cpp `typ_p`. 1.0 disables. */
typical_p?: number;
/** llama.cpp `top_n_sigma`. -1 disables. Local only. */
/** llama.cpp `top_n_sigma`. -1 disables. */
top_n_sigma?: number;
/** llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. Local only. */
/** llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. */
repeat_last_n?: number;
/** llama.cpp `dynatemp_range`. 0 disables. Local only. */
/** llama.cpp `dynatemp_range`. 0 disables. */
dynatemp_range?: number;
/** llama.cpp `dynatemp_exponent`. Local only, paired with dynatemp_range. */
/** llama.cpp `dynatemp_exponent`. Pairs with dynatemp_range. */
dynatemp_exponent?: number;
/** llama.cpp `mirostat` (0/1/2). 0 disables. Local only. */
/** llama.cpp `mirostat` (0/1/2). 0 disables. */
mirostat?: number;
/** llama.cpp `mirostat_tau` target entropy. Local only. */
mirostat_tau?: number;
/** llama.cpp `mirostat_eta` learning rate. Local only. */
mirostat_eta?: number;
/**
* OpenRouter `top_a` (alternate dynamic-top-P).
* https://openrouter.ai/docs/api/reference/parameters — gateway-only.
*/
/** OpenRouter `top_a`. https://openrouter.ai/docs/api/reference/parameters */
top_a?: number;
/**
* Anthropic fast-mode toggle. Opus 4.6 / 4.7 only; backend drops
* silently on every other model + provider. See
* https://platform.claude.com/docs/en/build-with-claude/fast-mode
*/
/** Anthropic Opus 4.6 / 4.7 only. https://platform.claude.com/docs/en/build-with-claude/fast-mode */
fast_mode?: boolean | null;
/**
* llama.cpp DRY (Don't Repeat Yourself) sampler family. All four
* fields documented at
* llama.cpp DRY sampler (4 fields). `dry_multiplier=0` disables.
* https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md
* 0.0 / null on `dry_multiplier` disables the whole chain. Local only.
*/
dry_multiplier?: number;
/** llama.cpp DRY base. Default 1.75. Local only. */
/** Default 1.75. */
dry_base?: number;
/** llama.cpp DRY allowed length threshold. Default 2. Local only. */
/** Default 2. */
dry_allowed_length?: number;
/** llama.cpp DRY penalty scan window. 0 disables, -1 = ctx-size. Local only. */
/** 0 disables, -1 = ctx-size. */
dry_penalty_last_n?: number;
/** llama.cpp XTC sampler probability. 0.0 disables. Local only. */
/** llama.cpp XTC. 0 disables. */
xtc_probability?: number;
/** llama.cpp XTC sampler threshold. Default 0.1. Local only. */
/** Default 0.1. */
xtc_threshold?: number;
/** llama.cpp `min_keep` (force min N tokens past filters). Local only. */
/** llama.cpp `min_keep`. */
min_keep?: number;
/**
* Continue generating past the model's EOS token. llama.cpp + vLLM only.
* `false` matches each backend's upstream default.
*/
/** Continue past EOS. llama.cpp + vLLM. */
ignore_eos?: boolean;
/**
* Minimum output tokens before stop / EOS can fire. vLLM + llama.cpp only.
* 0 disables.
*/
/** Min tokens before stop / EOS. llama.cpp + vLLM. */
min_tokens?: number;
/** vLLM `skip_special_tokens` — default true; forward only when false. */
/** vLLM only. */
skip_special_tokens?: boolean;
/** vLLM `spaces_between_special_tokens` — default true; forward only when false. */
/** vLLM only. */
spaces_between_special_tokens?: boolean;
/** vLLM `include_stop_str_in_output` — default false; forward only when true. */
/** vLLM only. Useful for agentic tools. */
include_stop_str_in_output?: boolean;
/** vLLM `truncate_prompt_tokens` — left-truncate the prompt. > 0 only. */
/** vLLM only. Left-truncate the prompt. */
truncate_prompt_tokens?: number;
/** llama.cpp `n_keep` — tokens to retain on context overflow. -1 = all. */
/** llama.cpp `n_keep`. -1 = keep all. */
n_keep?: number;
/** llama.cpp `n_probs` — return top-N token probabilities. > 0 only. */
/** llama.cpp `n_probs`. */
n_probs?: number;
/** llama.cpp `cache_prompt` — KV-cache reuse. Default true upstream; forward only when false. */
/** llama.cpp `cache_prompt`. */
cache_prompt?: boolean;
/** llama.cpp `return_tokens` — include raw token IDs in response. Default false. */
/** llama.cpp `return_tokens` (debug). */
return_tokens?: boolean;
/** llama.cpp `timings_per_token` — include per-token speed metrics. Default false. */
/** llama.cpp `timings_per_token` (perf debug). */
timings_per_token?: boolean;
/** llama.cpp `post_sampling_probs` — token probs after the sampler chain. Default false. */
/** llama.cpp `post_sampling_probs` (sampler debug). */
post_sampling_probs?: boolean;
}

View file

@ -9,6 +9,11 @@ export type ServiceTier =
| "scale"
| "standard_only";
// All `number | null` / `boolean | null` fields below follow the same
// convention: `null` = field omitted from the wire request (provider
// uses its own default). Per-provider capability gating lives in
// provider-capabilities.ts; the chat-adapter forwards only when the
// active provider's bucket has the matching flag set true.
export interface InferenceParams {
temperature: number;
topP: number;
@ -16,150 +21,80 @@ export interface InferenceParams {
minP: number;
repetitionPenalty: number;
presencePenalty: number;
/** OpenAI Chat Completions only; rejected by Responses + Anthropic. */
/** OpenAI Chat only; rejected by Responses + Anthropic. */
frequencyPenalty: number;
/**
* Best-effort determinism seed. OpenAI Chat Completions only; the
* Responses family and Anthropic reject it (silently dropped server-side).
* `null` = unset (no `seed` field on the wire).
*/
/** Determinism seed. OpenAI Chat + most OAI-compat backends only. */
seed: number | null;
/**
* Custom stop sequences. Maps to `stop` on OpenAI Chat Completions and
* `stop_sequences` on Anthropic Messages. OpenAI caps the array at 4
* entries; backend truncates with a warning. Empty array = unset.
*/
/** OAI Chat `stop` / Anthropic `stop_sequences`. OAI caps at 4. */
stop: string[];
/**
* Provider service tier. Each provider accepts a different enum set;
* `getServiceTierOptions(providerType)` resolves the legal values. `null`
* means "let the provider pick its default" and is the safe choice on
* provider switch.
*/
/** Per-provider enum via `getServiceTierOptions`. `null` = provider default. */
serviceTier: ServiceTier | null;
/**
* Whether the provider may dispatch tool calls in parallel. Maps to
* `parallel_tool_calls` on both OpenAI APIs and is inverted into
* `disable_parallel_tool_use` for Anthropic. Default true matches the
* upstream defaults across all three.
*/
/** Anthropic inverts to `disable_parallel_tool_use`. */
parallelToolCalls: boolean;
/**
* Locally typical sampling (llama.cpp `typ_p`). Local llama-server
* only no SaaS provider currently accepts this. 1.0 disables (and
* is the llama-server default). `null` = unset (not forwarded).
*/
/** llama.cpp `typ_p`. 1.0 disables. */
typicalP: number | null;
/** llama.cpp `top_n_sigma`. -1 disables. `null` = unset. */
/** llama.cpp `top_n_sigma`. -1 disables. */
topNSigma: number | null;
/** llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. `null` = unset. */
/** llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. */
repeatLastN: number | null;
/** llama.cpp `dynatemp_range`. 0.0 disables. `null` = unset. */
/** llama.cpp `dynatemp_range`. 0 disables. */
dynatempRange: number | null;
/** llama.cpp `dynatemp_exponent`. `null` = unset. */
/** llama.cpp `dynatemp_exponent`. Pairs with dynatempRange. */
dynatempExponent: number | null;
/** llama.cpp `mirostat` mode (0/1/2). 0 disables. `null` = unset. */
/** llama.cpp `mirostat` (0/1/2). 0 disables. */
mirostat: number | null;
/** llama.cpp `mirostat_tau` target entropy. `null` = unset. */
mirostatTau: number | null;
/** llama.cpp `mirostat_eta` learning rate. `null` = unset. */
mirostatEta: number | null;
/**
* OpenRouter `top_a` alternate dynamic-top-P. OpenRouter-only.
* Range [0, 1]. `null` = unset.
*/
/** OpenRouter `top_a`. Range [0, 1]. */
topA: number | null;
/**
* llama.cpp DRY (Don't Repeat Yourself) penalty multiplier.
* 0.0 disables (server default). `null` = unset.
* https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md
* llama.cpp DRY sampler multiplier is the master switch (0 disables
* the 4-field chain). See llama.cpp/tools/server/README.md.
*/
dryMultiplier: number | null;
/** llama.cpp DRY base value (exponential growth base). Default 1.75. `null` = unset. */
/** Default 1.75. */
dryBase: number | null;
/** llama.cpp DRY allowed token-extension threshold. Default 2. `null` = unset. */
/** Default 2. */
dryAllowedLength: number | null;
/** llama.cpp DRY penalty scan window. 0 disables, -1 = ctx-size. `null` = unset. */
/** 0 disables, -1 = ctx-size. */
dryPenaltyLastN: number | null;
/** llama.cpp XTC sampler probability. 0.0 disables. `null` = unset. */
/** llama.cpp XTC — probability is the master switch (0 disables). */
xtcProbability: number | null;
/** llama.cpp XTC sampler threshold. Default 0.1. `null` = unset. */
/** Default 0.1. */
xtcThreshold: number | null;
/** llama.cpp `min_keep` (force min N tokens past filters). 0 disables. `null` = unset. */
/** llama.cpp `min_keep` — min tokens past all filters. 0 disables. */
minKeep: number | null;
/**
* Force generation past the EOS token. llama.cpp + vLLM accept this.
* `null` = unset; `false` matches upstream default.
*/
/** Continue past EOS. llama.cpp + vLLM. */
ignoreEos: boolean | null;
/**
* Minimum output tokens before stop sequences / EOS can fire.
* vLLM + llama.cpp accept this. 0 disables. `null` = unset.
*/
/** Min tokens before stop / EOS can fire. llama.cpp + vLLM. */
minTokens: number | null;
/**
* vLLM `skip_special_tokens`. Default true. Forward only when false
* (i.e. user wants to see raw special tokens in the output).
* https://docs.vllm.ai/en/latest/api/vllm/sampling_params/
*/
/** vLLM only. Default true; forward only when false. */
skipSpecialTokens: boolean | null;
/**
* vLLM `spaces_between_special_tokens`. Default true. Forward only
* when false.
*/
/** vLLM only. Default true; forward only when false. */
spacesBetweenSpecialTokens: boolean | null;
/**
* vLLM `include_stop_str_in_output`. Default false. Useful for
* agentic tools that need the matched stop string echoed back.
*/
/** vLLM only. Useful for agentic tools needing the matched stop string echoed. */
includeStopStrInOutput: boolean | null;
/**
* vLLM `truncate_prompt_tokens` left-truncate the prompt to this
* many tokens. Useful for long-context overflow. `null` = unset.
*/
/** vLLM only. Left-truncate the prompt. */
truncatePromptTokens: number | null;
/**
* llama.cpp `n_keep` tokens to retain when context overflows.
* 0 disables, -1 keeps all. `null` = unset.
*/
/** llama.cpp `n_keep`. 0 disables, -1 = keep all. */
nKeep: number | null;
/**
* llama.cpp `n_probs` return top-N token probabilities per
* generated token. 0 disables. `null` = unset.
*/
/** llama.cpp `n_probs` — top-N token probabilities per token. */
nProbs: number | null;
/**
* llama.cpp `cache_prompt` reuse KV cache from previous prompts
* with a shared prefix. Default true upstream. Forward only when
* explicitly false (e.g. for deterministic benchmarks).
*/
/** llama.cpp `cache_prompt`. Default true; forward only when false. */
cachePrompt: boolean | null;
/**
* llama.cpp `return_tokens` include raw token IDs in the response.
* Debug. Default false.
*/
/** llama.cpp `return_tokens` (debug). */
returnTokens: boolean | null;
/**
* llama.cpp `timings_per_token` include per-token speed metrics.
* Default false.
*/
/** llama.cpp `timings_per_token` (perf debug). */
timingsPerToken: boolean | null;
/**
* llama.cpp `post_sampling_probs` return token probabilities AFTER
* the sampler chain runs. Debug. Default false.
*/
/** llama.cpp `post_sampling_probs` (sampler debug). */
postSamplingProbs: boolean | null;
maxSeqLength: number;
maxTokens: number;
systemPrompt: string;
checkpoint: string;
/** Allow loading models with custom code (e.g. NVIDIA Nemotron). Only enable for repos you trust. */
/** Trust custom model code (e.g. NVIDIA Nemotron). Only for trusted repos. */
trustRemoteCode?: boolean;
/**
* Anthropic fast-mode toggle. Opus 4.6 / 4.7 only; higher OTPS at
* 6x standard Opus pricing. Default false.
* https://platform.claude.com/docs/en/build-with-claude/fast-mode
*/
/** Anthropic Opus 4.6 / 4.7 only. 6x pricing for higher OTPS. */
fastMode?: boolean;
}