Tighten comments across PR 5711 (no behaviour change)
Comments-only pass. Drops verbose docstrings to single-line form,
removes repetitive "null = unset" / "Local only" tails (already
encoded by the type signature and capability map), keeps every
authoritative source URL but cuts surrounding prose, and removes
fully-redundant per-field comments where the field name already
says what the comment says.
Touches:
- types/runtime.ts (InferenceParams)
- types/api.ts (OpenAIChatCompletionsRequest wire shape)
- provider-capabilities.ts (ProviderCapabilities interface + bucket
inline blocks + per-model resolvers + reasoning helpers)
- api/chat-adapter.ts (external + local forwarding stanzas)
- backend models/inference.py (Field descriptions)
- backend llama_cpp.py (3rd payload-builder inline comments)
- backend routes/inference.py (_build_passthrough_payload)
- backend external_provider.py (4.7 sampling-removed header,
_is_openai_family_cloud docstring)
Net 508 lines deleted across 8 files; 65/65 sampling_params_routing
tests still pass; frontend tsc clean.
This commit is contained in:
parent
eaaf7142f6
commit
b60b0740c2
8 changed files with 304 additions and 812 deletions
|
|
@ -71,32 +71,23 @@ def _normalize_stop_for_provider(
|
|||
return None
|
||||
|
||||
|
||||
# Claude 4.7 Opus removed temperature, top_p, and top_k — the API
|
||||
# returns 400 "<param> is deprecated for this model" if any of them is
|
||||
# set to a non-default value. The "Sampling parameters removed" section
|
||||
# of the 4.7 release notes is the authoritative reference:
|
||||
# https://platform.claude.com/docs/en/about-claude/models/whats-new-claude-4-7
|
||||
# Only Opus shipped in the 4.7 generation (Sonnet stops at 4.6, Haiku at
|
||||
# 4.5 per https://platform.claude.com/docs/en/about-claude/models/overview),
|
||||
# so the regex is anchored to opus-4-7 only. 3.x and 4.5/4.6 still accept
|
||||
# all three knobs; the trailing -4-7[-.]/EOL anchor keeps future versions
|
||||
# (e.g. claude-opus-5) unaffected.
|
||||
# Opus 4.7 removed temperature/top_p/top_k (400s on any non-default).
|
||||
# Only Opus shipped in 4.7; 3.x and 4.5/4.6 still accept all three.
|
||||
# Trailing -4-7[-.]/EOL anchor keeps future families (claude-opus-5
|
||||
# etc) unaffected.
|
||||
# https://platform.claude.com/docs/en/about-claude/models/whats-new-claude-4-7
|
||||
def _is_openai_family_cloud(base_url: Optional[str]) -> bool:
|
||||
"""True iff ``base_url`` points at OpenAI cloud or Azure OpenAI Foundry.
|
||||
|
||||
Anchored to the URL host so an attacker can't bypass the gate with a
|
||||
path or subdomain like ``https://evil.com/api.openai.com/v1`` or
|
||||
``https://api.openai.com.attacker.com/v1`` (CodeQL py/incomplete-url-
|
||||
substring-sanitization). Used to scope cloud-only Responses-API
|
||||
extensions (prompt_cache_retention, context_management compaction,
|
||||
container shell tool) that 400 on non-cloud OpenAI-compatible
|
||||
servers (ollama / llama.cpp / vLLM).
|
||||
Host-anchored to avoid subdomain-injection bypass
|
||||
(https://evil.com/api.openai.com/v1, https://api.openai.com.attacker.com/v1).
|
||||
Used to gate cloud-only Responses-API extensions
|
||||
(prompt_cache_retention, context_management compaction, container
|
||||
shell tool) that 400 on non-cloud OAI-compat servers.
|
||||
|
||||
Azure Foundry resources are scoped to
|
||||
``<resource-name>.openai.azure.com``; match any subdomain via an
|
||||
`endswith` on the lowercased hostname, with the leading dot so
|
||||
`openai.azure.com` itself doesn't slip through (there is no
|
||||
apex-hosted Azure Foundry endpoint).
|
||||
Azure Foundry uses <resource>.openai.azure.com; match via endswith
|
||||
with the leading dot so the apex `openai.azure.com` can't slip
|
||||
through (no apex Foundry endpoint exists).
|
||||
"""
|
||||
if not base_url:
|
||||
return False
|
||||
|
|
|
|||
|
|
@ -4327,22 +4327,17 @@ class LlamaCppBackend:
|
|||
_cleaned = [s for s in stop if isinstance(s, str) and s]
|
||||
if _cleaned:
|
||||
payload["stop"] = _cleaned
|
||||
# Optional sampling extensions, gated on `is not None` so 0,
|
||||
# 0.0, and False all reach the wire.
|
||||
# Each field gated `is not None` so explicit 0 / 0.0 / False
|
||||
# values reach the wire. llama-server silently ignores fields
|
||||
# it doesn't recognise.
|
||||
if frequency_penalty is not None:
|
||||
payload["frequency_penalty"] = frequency_penalty
|
||||
if seed is not None:
|
||||
payload["seed"] = seed
|
||||
if parallel_tool_calls is not None:
|
||||
payload["parallel_tool_calls"] = parallel_tool_calls
|
||||
# Locally typical sampling. llama-server default 1.0 disables it;
|
||||
# the field is llama.cpp-specific (no cloud provider accepts it),
|
||||
# so we only forward it when the caller explicitly sets one.
|
||||
if typical_p is not None:
|
||||
payload["typical_p"] = typical_p
|
||||
# Extended llama.cpp sampler chain (top_n_sigma, repeat_last_n,
|
||||
# dynatemp_*, mirostat_*). All llama.cpp-specific; the frontend
|
||||
# capability map gates them to local backends only.
|
||||
if top_n_sigma is not None:
|
||||
payload["top_n_sigma"] = top_n_sigma
|
||||
if repeat_last_n is not None:
|
||||
|
|
@ -4357,9 +4352,6 @@ class LlamaCppBackend:
|
|||
payload["mirostat_tau"] = mirostat_tau
|
||||
if mirostat_eta is not None:
|
||||
payload["mirostat_eta"] = mirostat_eta
|
||||
# DRY / XTC / min_keep / ignore_eos / min_tokens — same llama.cpp-
|
||||
# only fields as above. Each is gated `is not None` so explicit
|
||||
# 0 / False values still reach the wire.
|
||||
if dry_multiplier is not None:
|
||||
payload["dry_multiplier"] = dry_multiplier
|
||||
if dry_base is not None:
|
||||
|
|
@ -4378,8 +4370,6 @@ class LlamaCppBackend:
|
|||
payload["ignore_eos"] = ignore_eos
|
||||
if min_tokens is not None:
|
||||
payload["min_tokens"] = min_tokens
|
||||
# vLLM output-shape knobs — forwarded `is not None` so user
|
||||
# opt-outs (skip_special_tokens=False etc) still reach the wire.
|
||||
if skip_special_tokens is not None:
|
||||
payload["skip_special_tokens"] = skip_special_tokens
|
||||
if spaces_between_special_tokens is not None:
|
||||
|
|
@ -4388,7 +4378,6 @@ class LlamaCppBackend:
|
|||
payload["include_stop_str_in_output"] = include_stop_str_in_output
|
||||
if truncate_prompt_tokens is not None:
|
||||
payload["truncate_prompt_tokens"] = truncate_prompt_tokens
|
||||
# llama.cpp context / KV-cache / instrumentation knobs.
|
||||
if n_keep is not None:
|
||||
payload["n_keep"] = n_keep
|
||||
if n_probs is not None:
|
||||
|
|
|
|||
|
|
@ -872,231 +872,147 @@ class ChatCompletionRequest(BaseModel):
|
|||
None,
|
||||
ge = 0.0,
|
||||
le = 1.0,
|
||||
description = (
|
||||
"Locally typical sampling (llama.cpp `typ_p`). 1.0 disables. "
|
||||
"Local llama-server only — no SaaS provider currently accepts "
|
||||
"this field, so the frontend capability map gates it off for "
|
||||
"every external provider and the local path forwards it on "
|
||||
"/v1/chat/completions."
|
||||
),
|
||||
description = "llama.cpp `typ_p`. 1.0 disables. Local only.",
|
||||
)
|
||||
top_n_sigma: Optional[float] = Field(
|
||||
None,
|
||||
description = (
|
||||
"llama.cpp `top_n_sigma` sampler. -1.0 disables (server "
|
||||
"default). Local only — no SaaS provider accepts it."
|
||||
),
|
||||
description = "llama.cpp `top_n_sigma`. -1 disables. Local only.",
|
||||
)
|
||||
repeat_last_n: Optional[int] = Field(
|
||||
None,
|
||||
description = (
|
||||
"llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. "
|
||||
"Pairs with repetition_penalty. Local only."
|
||||
),
|
||||
description = "llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. Local only.",
|
||||
)
|
||||
dynatemp_range: Optional[float] = Field(
|
||||
None,
|
||||
ge = 0.0,
|
||||
description = ("llama.cpp `dynatemp_range`. 0.0 disables. Local only."),
|
||||
description = "llama.cpp `dynatemp_range`. 0 disables. Local only.",
|
||||
)
|
||||
dynatemp_exponent: Optional[float] = Field(
|
||||
None,
|
||||
ge = 0.0,
|
||||
description = (
|
||||
"llama.cpp `dynatemp_exponent`. Local only; pairs with " "dynatemp_range."
|
||||
),
|
||||
description = "llama.cpp `dynatemp_exponent`. Pairs with dynatemp_range. Local only.",
|
||||
)
|
||||
mirostat: Optional[int] = Field(
|
||||
None,
|
||||
ge = 0,
|
||||
le = 2,
|
||||
description = (
|
||||
"llama.cpp `mirostat` mode. 0 = disabled, 1 = Mirostat, "
|
||||
"2 = Mirostat 2.0. Local only."
|
||||
),
|
||||
description = "llama.cpp `mirostat` (0=off, 1=Mirostat, 2=Mirostat 2.0). Local only.",
|
||||
)
|
||||
mirostat_tau: Optional[float] = Field(
|
||||
None,
|
||||
ge = 0.0,
|
||||
description = "llama.cpp `mirostat_tau` target entropy. Local only.",
|
||||
description = "llama.cpp `mirostat_tau`. Local only.",
|
||||
)
|
||||
mirostat_eta: Optional[float] = Field(
|
||||
None,
|
||||
ge = 0.0,
|
||||
description = "llama.cpp `mirostat_eta` learning rate. Local only.",
|
||||
description = "llama.cpp `mirostat_eta`. Local only.",
|
||||
)
|
||||
top_a: Optional[float] = Field(
|
||||
None,
|
||||
ge = 0.0,
|
||||
le = 1.0,
|
||||
description = (
|
||||
"OpenRouter `top_a` alternate dynamic-top-P. Documented at "
|
||||
"https://openrouter.ai/docs/api/reference/parameters. "
|
||||
"OpenRouter-only; other gateways silently drop it."
|
||||
"OpenRouter `top_a`. OpenRouter-only. "
|
||||
"https://openrouter.ai/docs/api/reference/parameters"
|
||||
),
|
||||
)
|
||||
dry_multiplier: Optional[float] = Field(
|
||||
None,
|
||||
ge = 0.0,
|
||||
description = (
|
||||
"llama.cpp DRY (Don't Repeat Yourself) penalty multiplier. "
|
||||
"0.0 disables (server default). Master switch for the 4-field "
|
||||
"DRY family — backend only forwards dry_base / dry_allowed_"
|
||||
"length / dry_penalty_last_n when multiplier > 0. Local only."
|
||||
"llama.cpp DRY multiplier. 0 disables the 4-field chain "
|
||||
"(dry_base / dry_allowed_length / dry_penalty_last_n). Local only."
|
||||
),
|
||||
)
|
||||
dry_base: Optional[float] = Field(
|
||||
None,
|
||||
ge = 1.0,
|
||||
description = (
|
||||
"llama.cpp DRY base value (exponential growth base). Default "
|
||||
"1.75. Local only; only meaningful when dry_multiplier > 0."
|
||||
),
|
||||
description = "llama.cpp DRY base. Default 1.75. Local only.",
|
||||
)
|
||||
dry_allowed_length: Optional[int] = Field(
|
||||
None,
|
||||
ge = 0,
|
||||
description = (
|
||||
"llama.cpp DRY allowed-length threshold. Default 2. Local "
|
||||
"only; only meaningful when dry_multiplier > 0."
|
||||
),
|
||||
description = "llama.cpp DRY allowed-length. Default 2. Local only.",
|
||||
)
|
||||
dry_penalty_last_n: Optional[int] = Field(
|
||||
None,
|
||||
description = (
|
||||
"llama.cpp DRY penalty scan window. 0 disables, -1 = ctx-size. "
|
||||
"Local only; only meaningful when dry_multiplier > 0."
|
||||
),
|
||||
description = "llama.cpp DRY scan window. 0 disables, -1 = ctx-size. Local only.",
|
||||
)
|
||||
xtc_probability: Optional[float] = Field(
|
||||
None,
|
||||
ge = 0.0,
|
||||
le = 1.0,
|
||||
description = (
|
||||
"llama.cpp XTC (eXclude Top Choice) sampler probability. "
|
||||
"0.0 disables. Master switch for xtc_threshold. Local only."
|
||||
),
|
||||
description = "llama.cpp XTC probability. 0 disables; pairs with xtc_threshold. Local only.",
|
||||
)
|
||||
xtc_threshold: Optional[float] = Field(
|
||||
None,
|
||||
ge = 0.0,
|
||||
le = 1.0,
|
||||
description = (
|
||||
"llama.cpp XTC sampler probability threshold. Default 0.1. "
|
||||
"Local only; only meaningful when xtc_probability > 0."
|
||||
),
|
||||
description = "llama.cpp XTC threshold. Default 0.1. Local only.",
|
||||
)
|
||||
min_keep: Optional[int] = Field(
|
||||
None,
|
||||
ge = 0,
|
||||
description = (
|
||||
"llama.cpp `min_keep` — force min N tokens past every "
|
||||
"sampler filter. 0 disables (server default). Local only."
|
||||
),
|
||||
description = "llama.cpp `min_keep` (force min N past every filter). Local only.",
|
||||
)
|
||||
ignore_eos: Optional[bool] = Field(
|
||||
None,
|
||||
description = (
|
||||
"Continue generation past the model's EOS token. Accepted by "
|
||||
"llama.cpp + vLLM; Ollama's OAI translator drops it. False "
|
||||
"matches each backend's upstream default."
|
||||
),
|
||||
description = "Continue past EOS. llama.cpp + vLLM only.",
|
||||
)
|
||||
min_tokens: Optional[int] = Field(
|
||||
None,
|
||||
ge = 0,
|
||||
description = (
|
||||
"Minimum output tokens before stop sequences / EOS can fire. "
|
||||
"Accepted by llama.cpp + vLLM; Ollama's OAI translator drops "
|
||||
"it. 0 disables (server default)."
|
||||
),
|
||||
description = "Min output tokens before stop / EOS. llama.cpp + vLLM only.",
|
||||
)
|
||||
skip_special_tokens: Optional[bool] = Field(
|
||||
None,
|
||||
description = (
|
||||
"vLLM `skip_special_tokens` (default true). Forward only when "
|
||||
"false — i.e. user wants raw special tokens in the output. "
|
||||
"vLLM only; llama-server / Ollama do not document this field."
|
||||
),
|
||||
description = "vLLM `skip_special_tokens` (default true). vLLM only.",
|
||||
)
|
||||
spaces_between_special_tokens: Optional[bool] = Field(
|
||||
None,
|
||||
description = (
|
||||
"vLLM `spaces_between_special_tokens` (default true). Forward "
|
||||
"only when false. vLLM only."
|
||||
),
|
||||
description = "vLLM `spaces_between_special_tokens` (default true). vLLM only.",
|
||||
)
|
||||
include_stop_str_in_output: Optional[bool] = Field(
|
||||
None,
|
||||
description = (
|
||||
"vLLM `include_stop_str_in_output` (default false). Forward "
|
||||
"only when true — useful for agentic tools that need the "
|
||||
"matched stop string echoed back. vLLM only."
|
||||
),
|
||||
description = "vLLM `include_stop_str_in_output`. Useful for agentic tools. vLLM only.",
|
||||
)
|
||||
truncate_prompt_tokens: Optional[int] = Field(
|
||||
None,
|
||||
ge = 1,
|
||||
description = (
|
||||
"vLLM `truncate_prompt_tokens` — left-truncate the prompt to "
|
||||
"this many tokens. Useful for long-context overflow. vLLM "
|
||||
"only; llama-server / Ollama drop this on the OAI path."
|
||||
),
|
||||
description = "vLLM `truncate_prompt_tokens` (left-truncate prompt). vLLM only.",
|
||||
)
|
||||
n_keep: Optional[int] = Field(
|
||||
None,
|
||||
description = (
|
||||
"llama.cpp `n_keep` — tokens to retain when context overflows. "
|
||||
"0 disables (server default), -1 keeps the whole prompt. "
|
||||
"Local llama-server only."
|
||||
),
|
||||
description = "llama.cpp `n_keep`. 0 disables, -1 = keep all. Local only.",
|
||||
)
|
||||
n_probs: Optional[int] = Field(
|
||||
None,
|
||||
ge = 0,
|
||||
description = (
|
||||
"llama.cpp `n_probs` — return top-N token probabilities per "
|
||||
"generated token. 0 disables (server default). Local only."
|
||||
),
|
||||
description = "llama.cpp `n_probs` (top-N token probs). 0 disables. Local only.",
|
||||
)
|
||||
cache_prompt: Optional[bool] = Field(
|
||||
None,
|
||||
description = (
|
||||
"llama.cpp `cache_prompt` — reuse KV cache across requests "
|
||||
"with a shared prefix. Default true upstream; forward only "
|
||||
"when explicitly false (e.g. deterministic benchmarks). "
|
||||
"Local llama-server only."
|
||||
),
|
||||
description = "llama.cpp `cache_prompt` (default true upstream). Local only.",
|
||||
)
|
||||
return_tokens: Optional[bool] = Field(
|
||||
None,
|
||||
description = (
|
||||
"llama.cpp `return_tokens` — include raw token IDs in the "
|
||||
"response. Debug. Local only."
|
||||
),
|
||||
description = "llama.cpp `return_tokens` (debug). Local only.",
|
||||
)
|
||||
timings_per_token: Optional[bool] = Field(
|
||||
None,
|
||||
description = (
|
||||
"llama.cpp `timings_per_token` — include per-token speed "
|
||||
"metrics in the streaming response. Local only."
|
||||
),
|
||||
description = "llama.cpp `timings_per_token` (perf debug). Local only.",
|
||||
)
|
||||
post_sampling_probs: Optional[bool] = Field(
|
||||
None,
|
||||
description = (
|
||||
"llama.cpp `post_sampling_probs` — return token probabilities "
|
||||
"AFTER the sampler chain runs (useful for sampler-tuning). "
|
||||
"Local only."
|
||||
),
|
||||
description = "llama.cpp `post_sampling_probs` (sampler debug). Local only.",
|
||||
)
|
||||
fast_mode: Optional[bool] = Field(
|
||||
None,
|
||||
description = (
|
||||
"[x-unsloth] Anthropic fast-mode toggle. On Claude Opus 4.6 / "
|
||||
"4.7 adds the `fast-mode-2026-02-01` beta header and sends "
|
||||
"`speed: 'fast'` for higher OTPS at premium pricing. Silently "
|
||||
"ignored on every other model + provider. See "
|
||||
"[x-unsloth] Anthropic fast-mode on Opus 4.6 / 4.7. Adds the "
|
||||
"fast-mode-2026-02-01 beta header + speed:'fast' for higher "
|
||||
"OTPS at premium pricing. Silently dropped elsewhere. "
|
||||
"https://platform.claude.com/docs/en/build-with-claude/fast-mode"
|
||||
),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -5148,21 +5148,18 @@ def _build_passthrough_payload(
|
|||
body["presence_penalty"] = presence_penalty
|
||||
# llama-server's /v1/chat/completions accepts the standard OpenAI
|
||||
# fields. parallel_tool_calls is a no-op on llama-server today but
|
||||
# is forwarded so a future release picks it up automatically.
|
||||
# forwarded so a future release picks it up automatically.
|
||||
# Each field below gated `is not None` so explicit 0 / False reach
|
||||
# the wire; llama-server silently ignores unknown fields, Ollama's
|
||||
# OAI translator drops everything outside the OAI subset.
|
||||
if frequency_penalty is not None:
|
||||
body["frequency_penalty"] = frequency_penalty
|
||||
if seed is not None:
|
||||
body["seed"] = seed
|
||||
if parallel_tool_calls is not None:
|
||||
body["parallel_tool_calls"] = parallel_tool_calls
|
||||
# llama.cpp-specific locally-typical sampling (typ_p in the sampler
|
||||
# chain). No SaaS provider accepts this; the frontend capability map
|
||||
# gates it to local only.
|
||||
if typical_p is not None:
|
||||
body["typical_p"] = typical_p
|
||||
# Extended llama.cpp sampler chain. All llama.cpp-specific; the
|
||||
# frontend capability map gates them to local backends only. Server
|
||||
# silently ignores fields it doesn't recognise.
|
||||
if top_n_sigma is not None:
|
||||
body["top_n_sigma"] = top_n_sigma
|
||||
if repeat_last_n is not None:
|
||||
|
|
@ -5177,10 +5174,6 @@ def _build_passthrough_payload(
|
|||
body["mirostat_tau"] = mirostat_tau
|
||||
if mirostat_eta is not None:
|
||||
body["mirostat_eta"] = mirostat_eta
|
||||
# DRY / XTC / min_keep / ignore_eos / min_tokens — llama-server
|
||||
# specific (DRY+XTC+min_keep) plus vLLM-shared (ignore_eos+min_tokens).
|
||||
# Forwarded `is not None` so explicit 0 / False values still reach
|
||||
# the wire; the OAI translator on Ollama drops these silently.
|
||||
if dry_multiplier is not None:
|
||||
body["dry_multiplier"] = dry_multiplier
|
||||
if dry_base is not None:
|
||||
|
|
@ -5199,10 +5192,6 @@ def _build_passthrough_payload(
|
|||
body["ignore_eos"] = ignore_eos
|
||||
if min_tokens is not None:
|
||||
body["min_tokens"] = min_tokens
|
||||
# vLLM output-shape knobs + llama.cpp context / KV / instrumentation
|
||||
# knobs. Per-backend capability gating on the frontend prevents these
|
||||
# from being forwarded to wires that don't recognise them; here we
|
||||
# only enforce the `is not None` rule so explicit defaults still pass.
|
||||
if skip_special_tokens is not None:
|
||||
body["skip_special_tokens"] = skip_special_tokens
|
||||
if spaces_between_special_tokens is not None:
|
||||
|
|
@ -5224,15 +5213,12 @@ def _build_passthrough_payload(
|
|||
if post_sampling_probs is not None:
|
||||
body["post_sampling_probs"] = post_sampling_probs
|
||||
if response_format is not None:
|
||||
# llama-server applies a GBNF grammar derived from the JSON schema
|
||||
# when response_format is present. Field is documented flat at the
|
||||
# request root (tools/server/README.md), which is also what the
|
||||
# OpenAI SDK produces by spreading extra_body into the body top.
|
||||
# llama-server applies a GBNF grammar from the JSON schema.
|
||||
# Field is documented flat at the request root.
|
||||
body["response_format"] = response_format
|
||||
if chat_template_kwargs is not None:
|
||||
# Propagate reasoning / template overrides (e.g. enable_thinking)
|
||||
# so llama-server renders the Jinja template in the mode the caller
|
||||
# asked for instead of whatever default the model was loaded with.
|
||||
# Reasoning / template overrides (e.g. enable_thinking) so
|
||||
# llama-server renders the Jinja template in the requested mode.
|
||||
body["chat_template_kwargs"] = chat_template_kwargs
|
||||
return body
|
||||
|
||||
|
|
|
|||
|
|
@ -1748,29 +1748,24 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
params.parallelToolCalls === false
|
||||
? { parallel_tool_calls: false }
|
||||
: {}),
|
||||
// llama.cpp `typ_p`. External providers all have
|
||||
// capabilities.typicalP=false; only the permissive local
|
||||
// buckets (custom/vllm/ollama/llama_cpp) opt in. `null`
|
||||
// (unset) or `1.0` (llama-server default) is a no-op.
|
||||
// llama.cpp / vLLM / OpenRouter extras. Each is gated by
|
||||
// (a) the active provider's capability flag and (b) a
|
||||
// non-default value, so only meaningful knobs hit the wire.
|
||||
...(externalCapabilities?.typicalP &&
|
||||
params.typicalP !== null &&
|
||||
params.typicalP !== 1
|
||||
? { typical_p: params.typicalP }
|
||||
: {}),
|
||||
// llama.cpp `top_n_sigma`. -1 disables (server default);
|
||||
// only forward meaningful values.
|
||||
...(externalCapabilities?.topNSigma &&
|
||||
params.topNSigma !== null &&
|
||||
params.topNSigma !== -1
|
||||
? { top_n_sigma: params.topNSigma }
|
||||
: {}),
|
||||
// llama.cpp `repeat_last_n`. Pairs with repetition_penalty.
|
||||
...(externalCapabilities?.repeatLastN &&
|
||||
params.repeatLastN !== null
|
||||
? { repeat_last_n: params.repeatLastN }
|
||||
: {}),
|
||||
// llama.cpp dynamic-temperature. Only forward when the
|
||||
// user opted in (range > 0).
|
||||
// Dynatemp: range>0 unlocks both fields.
|
||||
...(externalCapabilities?.dynatempRange &&
|
||||
params.dynatempRange !== null &&
|
||||
params.dynatempRange > 0
|
||||
|
|
@ -1782,8 +1777,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
: {}),
|
||||
}
|
||||
: {}),
|
||||
// llama.cpp Mirostat. Mode 0 disables; only forward the
|
||||
// sub-params when mode is enabled.
|
||||
// Mirostat: mode!=0 unlocks tau + eta.
|
||||
...(externalCapabilities?.mirostat &&
|
||||
params.mirostat !== null &&
|
||||
params.mirostat !== 0
|
||||
|
|
@ -1799,16 +1793,12 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
: {}),
|
||||
}
|
||||
: {}),
|
||||
// OpenRouter `top_a` (alternate dynamic top-P). Documented
|
||||
// range [0, 1]; 0 disables.
|
||||
...(externalCapabilities?.topA &&
|
||||
params.topA !== null &&
|
||||
params.topA > 0
|
||||
? { top_a: params.topA }
|
||||
: {}),
|
||||
// llama.cpp DRY sampler. dry_multiplier=0 disables the
|
||||
// whole chain; only forward the paired fields when the
|
||||
// master is set to a meaningful value.
|
||||
// DRY: multiplier>0 unlocks the 4-field chain.
|
||||
...(externalCapabilities?.dryMultiplier &&
|
||||
params.dryMultiplier !== null &&
|
||||
params.dryMultiplier > 0
|
||||
|
|
@ -1828,7 +1818,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
: {}),
|
||||
}
|
||||
: {}),
|
||||
// llama.cpp XTC sampler. xtc_probability=0 disables.
|
||||
// XTC: probability>0 unlocks threshold.
|
||||
...(externalCapabilities?.xtcProbability &&
|
||||
params.xtcProbability !== null &&
|
||||
params.xtcProbability > 0
|
||||
|
|
@ -1840,29 +1830,21 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
: {}),
|
||||
}
|
||||
: {}),
|
||||
// llama.cpp `min_keep` (force min N tokens past filters).
|
||||
// 0 is the upstream default; only forward when set higher.
|
||||
...(externalCapabilities?.minKeep &&
|
||||
params.minKeep !== null &&
|
||||
params.minKeep > 0
|
||||
? { min_keep: params.minKeep }
|
||||
: {}),
|
||||
// Continue past EOS. llama.cpp + vLLM only; forward only
|
||||
// when explicitly true (false matches upstream default).
|
||||
...(externalCapabilities?.ignoreEos && params.ignoreEos === true
|
||||
? { ignore_eos: true }
|
||||
: {}),
|
||||
// Minimum output tokens before stop / EOS can fire.
|
||||
// 0 = upstream default; only forward when set higher.
|
||||
...(externalCapabilities?.minTokens &&
|
||||
params.minTokens !== null &&
|
||||
params.minTokens > 0
|
||||
? { min_tokens: params.minTokens }
|
||||
: {}),
|
||||
// vLLM output-shape knobs. Upstream defaults:
|
||||
// skip_special_tokens=true, spaces_between_special_tokens=true,
|
||||
// include_stop_str_in_output=false. Forward only when user
|
||||
// opted away from the default to avoid no-op wire bloat.
|
||||
// vLLM output-shape: default true for skip/spaces, false
|
||||
// for include-stop. Forward only on user opt-out.
|
||||
...(externalCapabilities?.skipSpecialTokens &&
|
||||
params.skipSpecialTokens === false
|
||||
? { skip_special_tokens: false }
|
||||
|
|
@ -1880,8 +1862,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
params.truncatePromptTokens > 0
|
||||
? { truncate_prompt_tokens: params.truncatePromptTokens }
|
||||
: {}),
|
||||
// llama.cpp-only context / KV-cache / instrumentation knobs.
|
||||
// n_keep accepts -1 (= keep all) so the gate is != 0.
|
||||
// n_keep accepts -1 (keep all), so the gate is != 0.
|
||||
...(externalCapabilities?.nKeep &&
|
||||
params.nKeep !== null &&
|
||||
params.nKeep !== 0
|
||||
|
|
@ -1924,14 +1905,10 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
enable_tools: true,
|
||||
enabled_tools: [
|
||||
...(webSearchEnabledForThisTurn ? ["web_search"] : []),
|
||||
// web_fetch has its own Fetch pill, independent
|
||||
// of Search. Anthropic-only today.
|
||||
// web_fetch has its own pill (Anthropic-only).
|
||||
...(webFetchEnabledForThisTurn ? ["web_fetch"] : []),
|
||||
...(codeExecEnabledForThisTurn ? ["code_execution"] : []),
|
||||
// OpenAI Responses-API only: `image_generation`
|
||||
// returns inline image_generation_call output
|
||||
// items; the backend's _stream_openai_responses
|
||||
// path translates them to assistant tool events.
|
||||
// image_generation: OpenAI Responses-API only.
|
||||
...(imageGenerationEnabledForThisTurn
|
||||
? ["image_generation"]
|
||||
: []),
|
||||
|
|
@ -1967,11 +1944,8 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
externalProvider.enablePromptCaching ?? true,
|
||||
}
|
||||
: {}),
|
||||
// Anthropic-only: pass the cache TTL the user picked in
|
||||
// Configuration → Provider. Omitted = inherit the default
|
||||
// 5-minute pool. The backend's `_stream_anthropic` only
|
||||
// attaches `cache_control.ttl` when the value is one of
|
||||
// "5m" / "1h" (see external_provider.py near line 1375),
|
||||
// Anthropic-only cache TTL. Backend's _stream_anthropic
|
||||
// only attaches cache_control.ttl when value is "5m"/"1h",
|
||||
// so unknown values are a no-op end-to-end.
|
||||
...(supportsProviderPromptCacheTtl(
|
||||
externalProvider.providerType,
|
||||
|
|
@ -1980,9 +1954,8 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
isPromptCacheTtl(externalProvider.promptCacheTtl)
|
||||
? { prompt_cache_ttl: externalProvider.promptCacheTtl }
|
||||
: {}),
|
||||
// Anthropic fast mode (Opus 4.6 / 4.7 only); backend
|
||||
// silently drops on unsupported models as a second
|
||||
// line of defence.
|
||||
// Fast mode (Anthropic Opus 4.6 / 4.7). Backend drops on
|
||||
// unsupported models as second defence.
|
||||
...(params.fastMode &&
|
||||
providerSupportsFastMode(
|
||||
externalProvider.providerType,
|
||||
|
|
@ -2015,24 +1988,15 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
min_p: params.minP,
|
||||
repetition_penalty: params.repetitionPenalty,
|
||||
presence_penalty: params.presencePenalty,
|
||||
// Optional sampling extensions; local llama-server already
|
||||
// accepts `stop` / `seed` / `frequency_penalty` via
|
||||
// _build_passthrough_payload (routes/inference.py:4884) and
|
||||
// silently ignores fields it does not recognise. llama-server
|
||||
// documents `parallel_tool_calls` defaulting to FALSE
|
||||
// (https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md);
|
||||
// forward the user's preference unconditionally so the
|
||||
// default-on UI state actually enables parallel tool calls
|
||||
// there. External providers default to true everywhere; the
|
||||
// external branch above keeps its opt-in-on-false shape.
|
||||
// llama-server accepts the standard OAI extensions via
|
||||
// _build_passthrough_payload and silently ignores unknown
|
||||
// fields. parallel_tool_calls defaults to false upstream so
|
||||
// we forward unconditionally to honour the default-on UI.
|
||||
...(params.frequencyPenalty !== 0
|
||||
? { frequency_penalty: params.frequencyPenalty }
|
||||
: {}),
|
||||
...(params.seed !== null ? { seed: params.seed } : {}),
|
||||
...(params.stop.length > 0 ? { stop: params.stop } : {}),
|
||||
// llama.cpp `typ_p`. Local only — external providers gate
|
||||
// it off via capability map. `null` (unset) or 1.0 (server
|
||||
// default) is a no-op so we forward only meaningful values.
|
||||
...(params.typicalP !== null && params.typicalP !== 1
|
||||
? { typical_p: params.typicalP }
|
||||
: {}),
|
||||
|
|
@ -2061,7 +2025,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
: {}),
|
||||
}
|
||||
: {}),
|
||||
// llama.cpp DRY sampler — dry_multiplier=0 disables the chain.
|
||||
// DRY: multiplier>0 unlocks the 4-field chain.
|
||||
...(params.dryMultiplier !== null && params.dryMultiplier > 0
|
||||
? {
|
||||
dry_multiplier: params.dryMultiplier,
|
||||
|
|
@ -2076,7 +2040,7 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
: {}),
|
||||
}
|
||||
: {}),
|
||||
// llama.cpp XTC sampler — xtc_probability=0 disables.
|
||||
// XTC: probability>0 unlocks threshold.
|
||||
...(params.xtcProbability !== null && params.xtcProbability > 0
|
||||
? {
|
||||
xtc_probability: params.xtcProbability,
|
||||
|
|
@ -2088,15 +2052,13 @@ export function createOpenAIStreamAdapter(): ChatModelAdapter {
|
|||
...(params.minKeep !== null && params.minKeep > 0
|
||||
? { min_keep: params.minKeep }
|
||||
: {}),
|
||||
// ignore_eos / min_tokens are shared with vLLM but local
|
||||
// llama-server accepts them too.
|
||||
...(params.ignoreEos === true ? { ignore_eos: true } : {}),
|
||||
...(params.minTokens !== null && params.minTokens > 0
|
||||
? { min_tokens: params.minTokens }
|
||||
: {}),
|
||||
// Local llama-server / vLLM / Ollama route. Per-backend
|
||||
// capability gating handles the silent-drop story; here we
|
||||
// forward only when the value diverges from upstream default.
|
||||
// Forward only when value diverges from upstream default;
|
||||
// per-backend capability gating decides whether the wire
|
||||
// even sees these.
|
||||
...(params.skipSpecialTokens === false
|
||||
? { skip_special_tokens: false }
|
||||
: {}),
|
||||
|
|
|
|||
|
|
@ -1,163 +1,77 @@
|
|||
// SPDX-License-Identifier: AGPL-3.0-only
|
||||
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
||||
|
||||
/**
|
||||
* Per-provider sampling parameter capability matrix.
|
||||
*
|
||||
* Values are derived from each provider's published chat-completion docs as of
|
||||
* 2026-05. They describe which of our UI knobs map cleanly onto the provider's
|
||||
* request body; the panel hides params a provider does not accept so users
|
||||
* cannot dial a value that gets silently dropped or rejected.
|
||||
*
|
||||
* "Local" models (anything that is not an external provider) are represented by
|
||||
* a null capability — every knob renders for them.
|
||||
*/
|
||||
|
||||
// NB: when adding a new sampling knob, default it to `false` on every
|
||||
// SaaS provider in PROVIDER_CAPABILITIES below (only local backends
|
||||
// + the permissive {custom, vllm, ollama, llama_cpp, openrouter}
|
||||
// providers should expose llama.cpp-specific samplers).
|
||||
// Per-provider sampling capability matrix. Sourced from each
|
||||
// provider's chat-completion docs (2026-05). The panel hides params
|
||||
// the active provider does not accept so users never move a knob that
|
||||
// would be silently dropped or rejected.
|
||||
// When adding a new knob: default it to false on every SaaS bucket;
|
||||
// only local backends + the permissive openrouter bucket should
|
||||
// expose llama.cpp-specific samplers.
|
||||
export interface ProviderCapabilities {
|
||||
/**
|
||||
* Temperature sampling. Reasoning-class models (OpenAI's gpt-5.x / o3 via
|
||||
* /v1/responses) reject this with `Unsupported parameter`.
|
||||
*/
|
||||
/** OpenAI gpt-5.x / o-series reject via /v1/responses. */
|
||||
temperature: boolean;
|
||||
/** Nucleus (top_p) sampling. Same restriction as `temperature` on OpenAI. */
|
||||
topP: boolean;
|
||||
/** top-k token sampling (only Anthropic on the providers we ship). */
|
||||
/** Anthropic only among SaaS providers. */
|
||||
topK: boolean;
|
||||
/** min-p token cutoff (no SaaS provider currently exposes this). */
|
||||
minP: boolean;
|
||||
/** Repetition penalty (no SaaS provider currently exposes this). */
|
||||
repetitionPenalty: boolean;
|
||||
/** OpenAI-style presence penalty. */
|
||||
presencePenalty: boolean;
|
||||
/**
|
||||
* OpenAI-style frequency penalty. Accepted by Chat Completions only.
|
||||
* Anthropic and the OpenAI Responses family both reject it (the latter
|
||||
* with `Unsupported parameter`).
|
||||
*/
|
||||
/** OAI Chat only; rejected by Responses + Anthropic. */
|
||||
frequencyPenalty: boolean;
|
||||
/**
|
||||
* Best-effort determinism seed. Accepted by OpenAI Chat Completions and
|
||||
* most OpenAI-compatible local backends (vLLM, llama.cpp). Rejected by
|
||||
* the Responses family and silently dropped by Anthropic.
|
||||
*/
|
||||
/** OAI Chat + OAI-compat. Responses + Anthropic drop. */
|
||||
seed: boolean;
|
||||
/**
|
||||
* Custom stop sequences. Maps to `stop` (OpenAI Chat) or `stop_sequences`
|
||||
* (Anthropic). Not accepted by the Responses family.
|
||||
*/
|
||||
/** Not accepted by Responses; mapped to `stop_sequences` on Anthropic. */
|
||||
stop: boolean;
|
||||
/**
|
||||
* Provider service tier (`auto` / `standard_only` for Anthropic,
|
||||
* `auto`/`default`/`flex`/`priority`(+`scale`) for OpenAI). See
|
||||
* {@link getServiceTierOptions} for the legal values per provider.
|
||||
*/
|
||||
/** Per-provider enum, see getServiceTierOptions. */
|
||||
serviceTier: boolean;
|
||||
/**
|
||||
* Whether the provider supports turning off parallel tool dispatch.
|
||||
* Maps to `parallel_tool_calls: false` on both OpenAI APIs and
|
||||
* `disable_parallel_tool_use: true` on Anthropic (inverted).
|
||||
*/
|
||||
/** Anthropic inverts to `disable_parallel_tool_use`. */
|
||||
parallelToolCalls: boolean;
|
||||
/**
|
||||
* llama.cpp `typ_p` (locally typical sampling). Local llama-server
|
||||
* only — no SaaS provider currently accepts this field. Default is
|
||||
* `false` for every external provider and `true` only for the local
|
||||
* permissive {custom, vllm, ollama, llama_cpp} buckets.
|
||||
*/
|
||||
/** llama.cpp `typ_p`. */
|
||||
typicalP: boolean;
|
||||
/**
|
||||
* llama.cpp `top_n_sigma` sampler (newer top-sigma cutoff). Local
|
||||
* only; -1 disables.
|
||||
* https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md
|
||||
*/
|
||||
/** llama.cpp `top_n_sigma`. */
|
||||
topNSigma: boolean;
|
||||
/**
|
||||
* llama.cpp repetition window (`repeat_last_n`). Pairs with
|
||||
* `repeat_penalty`. Local only; 0 disables, -1 = ctx-size.
|
||||
*/
|
||||
/** llama.cpp `repeat_last_n`. */
|
||||
repeatLastN: boolean;
|
||||
/**
|
||||
* llama.cpp dynamic temperature range (`dynatemp_range`). Local
|
||||
* only; 0.0 disables.
|
||||
*/
|
||||
/** llama.cpp `dynatemp_range`. */
|
||||
dynatempRange: boolean;
|
||||
/**
|
||||
* llama.cpp dynamic temperature exponent (`dynatemp_exponent`).
|
||||
* Local only. Paired with dynatempRange.
|
||||
*/
|
||||
/** llama.cpp `dynatemp_exponent`. */
|
||||
dynatempExponent: boolean;
|
||||
/**
|
||||
* llama.cpp Mirostat sampling mode (`mirostat`). Local only.
|
||||
* 0 = disabled, 1 = Mirostat, 2 = Mirostat 2.0.
|
||||
*/
|
||||
/** llama.cpp `mirostat` (0/1/2). */
|
||||
mirostat: boolean;
|
||||
/**
|
||||
* llama.cpp Mirostat target entropy (`mirostat_tau`). Local only.
|
||||
* Only meaningful when mirostat != 0.
|
||||
*/
|
||||
mirostatTau: boolean;
|
||||
/**
|
||||
* llama.cpp Mirostat learning rate (`mirostat_eta`). Local only.
|
||||
* Only meaningful when mirostat != 0.
|
||||
*/
|
||||
mirostatEta: boolean;
|
||||
/**
|
||||
* OpenRouter `top_a` (alternate dynamic-top-P). Documented at
|
||||
* https://openrouter.ai/docs/api/reference/parameters. Other
|
||||
* gateways silently drop it; we surface it only for openrouter.
|
||||
*/
|
||||
/** OpenRouter `top_a`. https://openrouter.ai/docs/api/reference/parameters */
|
||||
topA: boolean;
|
||||
/**
|
||||
* llama.cpp DRY (Don't Repeat Yourself) repetition multiplier.
|
||||
* Master switch for the 4-field DRY sampler family. Local llama-
|
||||
* server only — vLLM / Ollama do not implement DRY.
|
||||
*/
|
||||
/** llama.cpp DRY (4 fields). dryMultiplier is the master switch. */
|
||||
dryMultiplier: boolean;
|
||||
/** llama.cpp DRY base value (exponential growth base). Local only. */
|
||||
dryBase: boolean;
|
||||
/** llama.cpp DRY allowed token-extension threshold. Local only. */
|
||||
dryAllowedLength: boolean;
|
||||
/** llama.cpp DRY penalty scan window. Local only. */
|
||||
dryPenaltyLastN: boolean;
|
||||
/** llama.cpp XTC (eXclude Top Choice) sampler probability. Local only. */
|
||||
/** llama.cpp XTC (2 fields). xtcProbability is the master switch. */
|
||||
xtcProbability: boolean;
|
||||
/** llama.cpp XTC sampler probability threshold. Local only. */
|
||||
xtcThreshold: boolean;
|
||||
/** llama.cpp `min_keep` (force min N tokens past every filter). Local only. */
|
||||
/** llama.cpp `min_keep`. */
|
||||
minKeep: boolean;
|
||||
/**
|
||||
* Continue generating past EOS. llama.cpp + vLLM accept this on the
|
||||
* /v1/chat/completions surface; Ollama's OAI translator drops it.
|
||||
*/
|
||||
/** llama.cpp + vLLM. Ollama OAI translator drops it. */
|
||||
ignoreEos: boolean;
|
||||
/**
|
||||
* Minimum output tokens before stop sequences / EOS can fire.
|
||||
* llama.cpp + vLLM accept this; Ollama's OAI translator drops it.
|
||||
*/
|
||||
/** llama.cpp + vLLM. Ollama OAI translator drops it. */
|
||||
minTokens: boolean;
|
||||
/** vLLM `skip_special_tokens` (vLLM SamplingParams). vLLM only. */
|
||||
/** vLLM only. */
|
||||
skipSpecialTokens: boolean;
|
||||
/** vLLM `spaces_between_special_tokens`. vLLM only. */
|
||||
spacesBetweenSpecialTokens: boolean;
|
||||
/** vLLM `include_stop_str_in_output`. vLLM only — useful for agentic tools. */
|
||||
/** vLLM only. Useful for agentic tools. */
|
||||
includeStopStrInOutput: boolean;
|
||||
/** vLLM `truncate_prompt_tokens` — left-truncate the prompt. vLLM only. */
|
||||
/** vLLM only. Left-truncate the prompt. */
|
||||
truncatePromptTokens: boolean;
|
||||
/** llama.cpp `n_keep` — tokens to retain on context overflow. llama.cpp only. */
|
||||
/** llama.cpp `n_keep` / `n_probs`. */
|
||||
nKeep: boolean;
|
||||
/** llama.cpp `n_probs` — return top-N token probabilities. llama.cpp only. */
|
||||
nProbs: boolean;
|
||||
/** llama.cpp `cache_prompt` — KV-cache reuse. llama.cpp only. */
|
||||
/** llama.cpp `cache_prompt`. */
|
||||
cachePrompt: boolean;
|
||||
/** llama.cpp `return_tokens` — debug. llama.cpp only. */
|
||||
/** llama.cpp debug flags. */
|
||||
returnTokens: boolean;
|
||||
/** llama.cpp `timings_per_token` — performance debug. llama.cpp only. */
|
||||
timingsPerToken: boolean;
|
||||
/** llama.cpp `post_sampling_probs` — sampling-chain debug. llama.cpp only. */
|
||||
postSamplingProbs: boolean;
|
||||
}
|
||||
|
||||
|
|
@ -265,36 +179,24 @@ export function clampReasoningEffortToLevels(
|
|||
*/
|
||||
export const EXTERNAL_MAX_OUTPUT_TOKENS = 32768;
|
||||
|
||||
/**
|
||||
* Per-model max-output caps from each provider's docs:
|
||||
* OpenAI: developers.openai.com/api/docs/models/gpt-5.5
|
||||
* Anthropic: platform.claude.com/docs/en/about-claude/models
|
||||
* Gemini: ai.google.dev/gemini-api/docs/models/gemini-3.1-pro-preview
|
||||
* DeepSeek: api-docs.deepseek.com/quick_start/pricing (V4 family)
|
||||
* Local-model path is unaffected.
|
||||
*/
|
||||
// Per-model max-output caps from each provider's docs (verified May 2026):
|
||||
// OpenAI: developers.openai.com/api/docs/models/<model>
|
||||
// Anthropic: platform.claude.com/docs/en/about-claude/models/overview
|
||||
// Gemini: ai.google.dev/gemini-api/docs/models
|
||||
// DeepSeek: api-docs.deepseek.com/quick_start/pricing
|
||||
// Order matters: list specific chat-class ids before broader gpt-5 /
|
||||
// claude-opus-4 entries so the longer prefix wins via .startsWith().
|
||||
const EXTERNAL_MAX_OUTPUT_TOKENS_BY_MODEL: Array<{
|
||||
providerType: string;
|
||||
prefixes: readonly string[];
|
||||
cap: number;
|
||||
}> = [
|
||||
// OpenAI per-model output caps from developers.openai.com per-model
|
||||
// pages (cross-checked against the Azure Foundry reasoning table).
|
||||
// Order matters: list the 16k chat-latest variants first so the
|
||||
// broader gpt-5 / gpt-4 entries don't shadow them.
|
||||
// gpt-5.3-chat-latest / gpt-5.1-chat = 16384 (chat-class)
|
||||
// gpt-5.5* / gpt-5.4* / gpt-5.3-codex / gpt-5.2 / gpt-5.1 / gpt-5
|
||||
// / gpt-5-codex / gpt-5-pro = 128000
|
||||
// o1 / o3 / o3-pro / o4-mini / codex-mini = 100000
|
||||
{ providerType: "openai", prefixes: ["gpt-5.3-chat-latest", "gpt-5.1-chat"], cap: 16384 },
|
||||
{ providerType: "openai", prefixes: ["gpt-5"], cap: 128000 },
|
||||
{ providerType: "openai", prefixes: ["o1", "o3", "o4", "codex-mini"], cap: 100000 },
|
||||
// Anthropic — overview table at
|
||||
// platform.claude.com/docs/en/about-claude/models/overview. Opus 4.7
|
||||
// and Opus 4.6 BOTH ship 128k Max output (the legacy-table row for
|
||||
// 4.6 reads "128k tokens"); Sonnet 4.6 / Sonnet 4.5 / Sonnet 4 / Opus
|
||||
// 4.5 / Haiku 4.5 ship 64k; Opus 4.1 / Opus 4 ship 32k (covered by
|
||||
// the 32k default below).
|
||||
// Anthropic Opus 4.6 + 4.7 ship 128k Max output; Sonnet 4.5/4.6/4 +
|
||||
// Opus 4.5 + Haiku 4.5 ship 64k; Opus 4.1 / Opus 4 fall through to
|
||||
// the 32k EXTERNAL_MAX_OUTPUT_TOKENS default.
|
||||
{
|
||||
providerType: "anthropic",
|
||||
prefixes: ["claude-opus-4-7", "claude-opus-4-6"],
|
||||
|
|
@ -311,13 +213,12 @@ const EXTERNAL_MAX_OUTPUT_TOKENS_BY_MODEL: Array<{
|
|||
],
|
||||
cap: 64000,
|
||||
},
|
||||
// Gemini
|
||||
{
|
||||
providerType: "gemini",
|
||||
prefixes: ["gemini-3", "gemini-pro", "gemini-flash"],
|
||||
cap: 65536,
|
||||
},
|
||||
// DeepSeek (V4: deepseek-chat / deepseek-reasoner alias V4-flash).
|
||||
// V4: deepseek-chat / deepseek-reasoner alias V4-flash.
|
||||
{ providerType: "deepseek", prefixes: ["deepseek"], cap: 384000 },
|
||||
];
|
||||
|
||||
|
|
@ -361,33 +262,16 @@ function _inferProviderFromOpenrouterId(
|
|||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the external provider offers a built-in web-search tool that the
|
||||
* model invokes server-side. When `true`, the chat composer's Search button
|
||||
* is available for that provider and the chat-adapter forwards
|
||||
* `enable_tools: true, enabled_tools: ["web_search"]` on the request — the
|
||||
* backend routes the call through the provider's tool schema:
|
||||
* - OpenAI: `tools: [{type: "web_search"}]` on /v1/responses
|
||||
* - Anthropic: `tools: [{type: "web_search_20250305", name: "web_search",
|
||||
* max_uses: 5}]` on /v1/messages
|
||||
* - OpenRouter: `plugins: [{id: "web"}]` on /v1/chat/completions (the
|
||||
* router's universal web-search shape; works for every
|
||||
* underlying model including the `openrouter/free` router).
|
||||
* - Kimi: `tools: [{type: "builtin_function", function: {name:
|
||||
* "$web_search"}}]` with `thinking: {type:
|
||||
* "disabled"}`. Requires a client round-trip:
|
||||
* the first call returns the search args; the backend
|
||||
* echoes them back as a role=tool message; the second
|
||||
* call streams the answer. Handled in
|
||||
* _stream_kimi_web_search on the backend.
|
||||
*
|
||||
* Mistral is intentionally excluded: their `web_search` connector lives on
|
||||
* the Agents API (`/v1/agents` + `/v1/conversations`), not chat completions,
|
||||
* and returns `"WebSearchTool connector is not supported"` if injected into
|
||||
* /v1/chat/completions. Wiring it would require a dedicated Agents streaming
|
||||
* path. Gemini's grounded-search can be added with the same pattern when
|
||||
* matching backend translation lands.
|
||||
*/
|
||||
// Gates the composer's Search button. Backend translates
|
||||
// enable_tools:["web_search"] into each provider's tool schema:
|
||||
// OpenAI: tools:[{type:"web_search"}] on /v1/responses
|
||||
// Anthropic: tools:[{type:"web_search_20250305", max_uses:5}] on /v1/messages
|
||||
// OpenRouter: plugins:[{id:"web"}] (router's universal shape)
|
||||
// Kimi: $web_search builtin (two-call round trip via
|
||||
// _stream_kimi_web_search)
|
||||
// Mistral excluded: their web_search is on the Agents API, not chat
|
||||
// completions, and 400s if injected. Gemini grounded-search needs
|
||||
// matching backend translation first.
|
||||
export function providerSupportsBuiltinWebSearch(
|
||||
providerType: string | null | undefined,
|
||||
): boolean {
|
||||
|
|
@ -399,24 +283,18 @@ export function providerSupportsBuiltinWebSearch(
|
|||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the external provider exposes a server-side web_fetch tool
|
||||
* (single URL, text or PDF) emitting a document block. Anthropic-only
|
||||
* today (`web_fetch_20250910` / `web_fetch_20260209`). Gates the
|
||||
* composer's standalone Fetch pill, independent of Search.
|
||||
*/
|
||||
// Anthropic-only server-side web_fetch tool
|
||||
// (web_fetch_20250910 / _20260209). Gates the composer's Fetch pill.
|
||||
export function providerSupportsBuiltinWebFetch(
|
||||
providerType: string | null | undefined,
|
||||
): boolean {
|
||||
return providerType === "anthropic";
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the active provider + model supports Anthropic fast-mode
|
||||
* (`speed: "fast"` + `fast-mode-2026-02-01` header). Opus 4.6 / 4.7
|
||||
* only per https://platform.claude.com/docs/en/build-with-claude/fast-mode.
|
||||
* Backend silently drops on unsupported models as a second defence.
|
||||
*/
|
||||
// Anthropic fast-mode (`speed:"fast"` + fast-mode-2026-02-01 header).
|
||||
// Opus 4.6 / 4.7 only per
|
||||
// https://platform.claude.com/docs/en/build-with-claude/fast-mode.
|
||||
// Backend silently drops on unsupported models as a second defence.
|
||||
const ANTHROPIC_FAST_MODE_MODEL_PREFIXES = [
|
||||
"claude-opus-4-7",
|
||||
"claude-opus-4-6",
|
||||
|
|
@ -428,38 +306,21 @@ export function providerSupportsFastMode(
|
|||
): boolean {
|
||||
if (providerType !== "anthropic") return false;
|
||||
if (!modelId) return false;
|
||||
// Family boundary ("" or "-") required so IDs like "claude-opus-4-70"
|
||||
// / "claude-opus-4-7b" do not match.
|
||||
// Family boundary required so "claude-opus-4-70" doesn't match.
|
||||
return ANTHROPIC_FAST_MODE_MODEL_PREFIXES.some(
|
||||
(prefix) => modelId === prefix || modelId.startsWith(`${prefix}-`),
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the selected external provider/model exposes a server-side
|
||||
* code-execution tool. Two providers ship one today:
|
||||
*
|
||||
* - **Anthropic** (`code_execution_20250825`): Python + bash +
|
||||
* str_replace-based file edits inside a 5 GB sandboxed container
|
||||
* per request. Documented at
|
||||
* https://platform.claude.com/docs/en/agents-and-tools/tool-use/code-execution-tool
|
||||
*
|
||||
* - **OpenAI cloud** (`shell` on /v1/responses): bash inside a
|
||||
* reusable container; we auto-create one on the first turn of a
|
||||
* chat thread and reference it on subsequent turns via the
|
||||
* thread's stored `openaiCodeExecContainerId`. Documented at
|
||||
* https://developers.openai.com/api/docs/guides/tools-shell
|
||||
*
|
||||
* Returns false for every other provider. The backend additionally
|
||||
* gates the OpenAI shell tool on `is_openai_cloud` so custom
|
||||
* OpenAI-compat servers (ollama / llama.cpp / vLLM) that also report
|
||||
* `provider_type="openai"` never receive the tool — but in practice
|
||||
* none of those catalogs surface the `gpt-5.5` ids anyway, so the
|
||||
* frontend prefix match is enough.
|
||||
*
|
||||
* v1 wires the tools themselves; file uploads (Anthropic
|
||||
* `container_upload` / OpenAI `input_file`) are a deliberate follow-up.
|
||||
*/
|
||||
// Server-side code-execution tools:
|
||||
// Anthropic code_execution_20250825 (Python + bash + str_replace in
|
||||
// a 5 GB sandbox).
|
||||
// OpenAI cloud `shell` on /v1/responses (bash in a reusable container
|
||||
// referenced via openaiCodeExecContainerId across turns).
|
||||
// Backend also gates OpenAI on is_openai_cloud so custom OAI-compat
|
||||
// servers reporting provider_type="openai" can't accidentally get the
|
||||
// shell tool. File uploads (container_upload / input_file) are
|
||||
// follow-up work.
|
||||
const ANTHROPIC_CODE_EXECUTION_MODEL_PREFIXES = [
|
||||
"claude-opus-4-7",
|
||||
"claude-opus-4-6",
|
||||
|
|
@ -523,18 +384,9 @@ export function providerSupportsBuiltinCodeExecution(
|
|||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the selected external provider/model exposes OpenAI's
|
||||
* Responses-API server-side image_generation tool. Lit on for OpenAI
|
||||
* cloud (`api.openai.com`) when the picked model is a Responses-API
|
||||
* family id (gpt-5.x today). The backend additionally gates on
|
||||
* `is_openai_cloud`; mirror that here so the pill is hidden on custom
|
||||
* OpenAI-compat backends (ollama / llama.cpp / vLLM) that report
|
||||
* `provider_type="openai"` but would 400 on a `{type:"image_generation"}`
|
||||
* tool. See backend/core/inference/external_provider.py near line 2770
|
||||
* for the dispatch and backend/tests/test_openai_image_generation.py
|
||||
* for the round-trip coverage.
|
||||
*/
|
||||
// OpenAI Responses-API image_generation tool. OpenAI cloud +
|
||||
// Responses-family ids only; backend mirrors via is_openai_cloud so
|
||||
// custom OAI-compat servers reporting provider_type="openai" don't 400.
|
||||
const OPENAI_IMAGE_GENERATION_MODEL_PREFIXES = [
|
||||
"gpt-5.5-pro",
|
||||
"gpt-5.5",
|
||||
|
|
@ -561,19 +413,10 @@ export function providerSupportsBuiltinImageGeneration(
|
|||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-provider minimum on the outbound max_tokens. Kimi's docs require
|
||||
* `max_tokens >= 16000` whenever a thinking model is in use so the
|
||||
* reasoning_content and final answer both fit in the budget — anything
|
||||
* lower truncates the response mid-stream. Other providers don't have a
|
||||
* documented floor, so they fall through to the generic min of 64 in
|
||||
* the slider.
|
||||
*
|
||||
* The chat-adapter resolves the effective floor on send and bumps the
|
||||
* outbound max_tokens up to this value if the user's stored maxTokens
|
||||
* sits below it. The settings panel reflects the same floor as the
|
||||
* slider min so the displayed value never drifts from what's sent.
|
||||
*/
|
||||
// Per-provider min on outbound max_tokens. Kimi thinking models need
|
||||
// >=16000 or the response truncates mid-stream. Other providers fall
|
||||
// through to the generic 64. chat-adapter bumps the user's stored
|
||||
// maxTokens up to the floor on send; the slider min mirrors the same.
|
||||
const EXTERNAL_MIN_OUTPUT_TOKENS_BY_PROVIDER: Record<string, number> = {
|
||||
kimi: 16000,
|
||||
};
|
||||
|
|
@ -662,12 +505,10 @@ const LLAMA_CPP_CAPABILITIES: ProviderCapabilities = {
|
|||
minKeep: true,
|
||||
ignoreEos: true,
|
||||
minTokens: true,
|
||||
// vLLM-only output-shape knobs — llama-server does not document them.
|
||||
skipSpecialTokens: false,
|
||||
spacesBetweenSpecialTokens: false,
|
||||
includeStopStrInOutput: false,
|
||||
truncatePromptTokens: false,
|
||||
// llama.cpp-only context / KV-cache / instrumentation knobs.
|
||||
nKeep: true,
|
||||
nProbs: true,
|
||||
cachePrompt: true,
|
||||
|
|
@ -676,10 +517,10 @@ const LLAMA_CPP_CAPABILITIES: ProviderCapabilities = {
|
|||
postSamplingProbs: true,
|
||||
};
|
||||
|
||||
// vLLM's OpenAI-compat endpoint accepts the OpenAI subset plus top_k /
|
||||
// min_p / repetition_penalty / seed, but not the 8 llama.cpp-only
|
||||
// extended samplers (vLLM's SamplingParams has no fields for them —
|
||||
// vllm/sampling_params.py).
|
||||
// vLLM SamplingParams: OAI subset + top_k/min_p/repetition_penalty/seed
|
||||
// + the 4 vLLM-only output-shape knobs. No DRY / XTC / mirostat /
|
||||
// dynatemp / typical_p / min_keep / n_keep / n_probs / cache_prompt /
|
||||
// debug flags (none in SamplingParams).
|
||||
const VLLM_CAPABILITIES: ProviderCapabilities = {
|
||||
...LLAMA_CPP_CAPABILITIES,
|
||||
typicalP: false,
|
||||
|
|
@ -690,9 +531,6 @@ const VLLM_CAPABILITIES: ProviderCapabilities = {
|
|||
mirostat: false,
|
||||
mirostatTau: false,
|
||||
mirostatEta: false,
|
||||
// vLLM's SamplingParams has no DRY / XTC / min_keep fields (only
|
||||
// llama-server implements them). Keep ignoreEos + minTokens on:
|
||||
// both are documented vLLM SamplingParams fields.
|
||||
dryMultiplier: false,
|
||||
dryBase: false,
|
||||
dryAllowedLength: false,
|
||||
|
|
@ -700,12 +538,10 @@ const VLLM_CAPABILITIES: ProviderCapabilities = {
|
|||
xtcProbability: false,
|
||||
xtcThreshold: false,
|
||||
minKeep: false,
|
||||
// vLLM-only output-shape knobs — flip the LLAMA_CPP defaults.
|
||||
skipSpecialTokens: true,
|
||||
spacesBetweenSpecialTokens: true,
|
||||
includeStopStrInOutput: true,
|
||||
truncatePromptTokens: true,
|
||||
// llama.cpp-only instrumentation knobs — vLLM has no analog.
|
||||
nKeep: false,
|
||||
nProbs: false,
|
||||
cachePrompt: false,
|
||||
|
|
@ -714,38 +550,29 @@ const VLLM_CAPABILITIES: ProviderCapabilities = {
|
|||
postSamplingProbs: false,
|
||||
};
|
||||
|
||||
// Ollama is stricter than vLLM. Studio reaches Ollama via the OpenAI-
|
||||
// compat /v1/chat/completions transport, and Ollama's translator
|
||||
// (ollama/openai/openai.go FromChatRequest) only copies the documented
|
||||
// OpenAI subset — top_k / min_p / repetition_penalty are silently
|
||||
// DROPPED on that path even though native /api/chat would forward them
|
||||
// through the `options` bag. Hide them so users don't move a slider
|
||||
// the wire never carries.
|
||||
// Ollama OAI translator (openai/openai.go FromChatRequest) only copies
|
||||
// the documented OpenAI subset on /v1/chat/completions — top_k / min_p
|
||||
// / repetition_penalty / ignore_eos / min_tokens / the 4 vLLM output
|
||||
// knobs all silently drop on this path. (Native /api/chat would forward
|
||||
// them via `options`, but Studio uses /v1.)
|
||||
const OLLAMA_CAPABILITIES: ProviderCapabilities = {
|
||||
...VLLM_CAPABILITIES,
|
||||
topK: false,
|
||||
minP: false,
|
||||
repetitionPenalty: false,
|
||||
// Ollama's OAI translator (openai/openai.go FromChatRequest) doesn't
|
||||
// forward ignore_eos or min_tokens either — both fields silently drop
|
||||
// on the /v1/chat/completions path Studio uses.
|
||||
ignoreEos: false,
|
||||
minTokens: false,
|
||||
// The vLLM-specific output-shape knobs are not recognised by the
|
||||
// Ollama OAI translator; flip them back to false.
|
||||
skipSpecialTokens: false,
|
||||
spacesBetweenSpecialTokens: false,
|
||||
includeStopStrInOutput: false,
|
||||
truncatePromptTokens: false,
|
||||
};
|
||||
|
||||
// OpenRouter is a router-of-routers: the gateway accepts a wider set
|
||||
// of OpenAI-style sampling fields than any single upstream supports
|
||||
// and silently drops what the chosen route does not, per
|
||||
// https://openrouter.ai/docs/api/reference/parameters. Surface the
|
||||
// router's full documented set (incl. top_a) and leave the
|
||||
// llama.cpp-only knobs off (the docs don't list them, so we don't
|
||||
// either even though many openrouter routes terminate at llama.cpp).
|
||||
// OpenRouter is a router-of-routers: gateway accepts a wider set of
|
||||
// OAI-style fields than any single upstream and silently drops what
|
||||
// the chosen route doesn't. Surface the full documented set (incl.
|
||||
// top_a) and leave llama.cpp-only knobs off.
|
||||
// https://openrouter.ai/docs/api/reference/parameters
|
||||
const OPENROUTER_CAPABILITIES: ProviderCapabilities = {
|
||||
temperature: true,
|
||||
topP: true,
|
||||
|
|
@ -788,14 +615,11 @@ const OPENROUTER_CAPABILITIES: ProviderCapabilities = {
|
|||
postSamplingProbs: false,
|
||||
};
|
||||
|
||||
// Reasoning-class OpenAI models served via /v1/responses fix temperature
|
||||
// at 1, ignore top_p, and 400 on presence/frequency_penalty / seed. Non
|
||||
// reasoning models (gpt-4o, gpt-4-turbo, gpt-4, gpt-3.5-turbo) keep the
|
||||
// full sampling surface even when routed through /v1/responses. See
|
||||
// https://platform.openai.com/docs/guides/reasoning and the GPT-5 release
|
||||
// notes; backend dispatch is external_provider._stream_openai_responses.
|
||||
// The Responses API itself drops `stop`, so we leave that off for all
|
||||
// OpenAI models regardless of family.
|
||||
// OpenAI reasoning class via /v1/responses: temperature fixed at 1,
|
||||
// top_p ignored, 400s on presence/frequency_penalty/seed. Chat-class
|
||||
// (gpt-4o etc) keeps the full surface even via /v1/responses. Both
|
||||
// drop `stop` (Responses doesn't surface it).
|
||||
// https://platform.openai.com/docs/guides/reasoning
|
||||
const OPENAI_REASONING_CAPABILITIES: ProviderCapabilities = {
|
||||
temperature: false,
|
||||
topP: false,
|
||||
|
|
@ -906,14 +730,9 @@ function isOpenAIReasoningModelId(modelId: string | null | undefined): boolean {
|
|||
return OPENAI_REASONING_MODEL_PREFIXES.some((p) => normalized.startsWith(p));
|
||||
}
|
||||
|
||||
// Mirror of backend _ANTHROPIC_4_7_SAMPLING_REMOVED in
|
||||
// studio/backend/core/inference/external_provider.py:110. Claude Opus
|
||||
// 4.7 removed temperature, top_p, and top_k entirely; surfacing the
|
||||
// sliders would let the user move a control that the backend silently
|
||||
// strips. Only Opus shipped in the 4.7 generation (Sonnet stops at 4.6,
|
||||
// Haiku at 4.5 per platform.claude.com/docs/en/about-claude/models/
|
||||
// overview), so the regex is opus-only. The trailing -4-7[-.]/EOL
|
||||
// anchor keeps future families (claude-opus-5 etc.) unaffected.
|
||||
// Mirror of backend _ANTHROPIC_4_7_SAMPLING_REMOVED. Opus 4.7 removed
|
||||
// temperature/top_p/top_k; only Opus shipped in 4.7. The -4-7[-.]/EOL
|
||||
// anchor keeps future families (claude-opus-5 etc) unaffected.
|
||||
const ANTHROPIC_4_7_SAMPLING_REMOVED_REGEX = /^claude-opus-4-7(?:[-.]|$)/i;
|
||||
|
||||
function isClaude47SamplingRemoved(modelId: string | null | undefined): boolean {
|
||||
|
|
@ -922,12 +741,9 @@ function isClaude47SamplingRemoved(modelId: string | null | undefined): boolean
|
|||
return ANTHROPIC_4_7_SAMPLING_REMOVED_REGEX.test(normalized);
|
||||
}
|
||||
|
||||
// DeepSeek reasoning-class models silently ignore temperature, top_p,
|
||||
// presence_penalty, frequency_penalty and 400 on logprobs/top_logprobs.
|
||||
// `deepseek-reasoner` is the dedicated thinking model;
|
||||
// `deepseek-v4-flash` runs reasoning-mode under the same flag as well.
|
||||
// Match by prefix so future revisions (deepseek-reasoner-2027 etc.)
|
||||
// continue to gate correctly.
|
||||
// DeepSeek reasoner ids silently ignore temperature/top_p/presence/
|
||||
// frequency and 400 on logprobs per the reasoning_model guide. Prefix
|
||||
// match covers future revisions (deepseek-reasoner-2027 etc).
|
||||
const DEEPSEEK_REASONING_MODEL_PREFIXES = [
|
||||
"deepseek-reasoner",
|
||||
"deepseek-r1",
|
||||
|
|
@ -940,20 +756,13 @@ function isDeepSeekReasoningModelId(modelId: string | null | undefined): boolean
|
|||
}
|
||||
|
||||
const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
|
||||
// Default OpenAI bucket is reasoning-class (current registry only ships
|
||||
// gpt-5.x / o3 ids), but per-model resolution in getProviderCapabilities
|
||||
// upgrades non-reasoning ids (gpt-4o etc.) to OPENAI_CHAT_CAPABILITIES.
|
||||
// Default to reasoning-class; getProviderCapabilities upgrades
|
||||
// non-reasoning ids (gpt-4o etc) to OPENAI_CHAT_CAPABILITIES.
|
||||
openai: OPENAI_REASONING_CAPABILITIES,
|
||||
// Anthropic's Messages API accepts top_k on 3.x and 4.5/4.6, but Claude
|
||||
// 4.7 (Opus/Sonnet/Haiku) deprecated it and returns 400 if it is set.
|
||||
// We surface top_k in the panel for all Anthropic providers and let the
|
||||
// backend strip it per-model — see _stream_anthropic in
|
||||
// studio/backend/core/inference/external_provider.py.
|
||||
// Presence/frequency penalty / seed / logprobs are not part of the
|
||||
// Messages API on any Claude generation. stop_sequences (Anthropic name
|
||||
// for `stop`), service_tier (auto|standard_only), and
|
||||
// disable_parallel_tool_use (inverse of parallel_tool_calls) ARE
|
||||
// supported.
|
||||
// Messages API: temperature/top_p/top_k/stop_sequences/service_tier
|
||||
// (auto|standard_only)/disable_parallel_tool_use. Opus 4.7 strips
|
||||
// temperature/top_p/top_k via the regex above. No presence/frequency
|
||||
// penalty / seed / logprobs on any Claude generation.
|
||||
anthropic: {
|
||||
temperature: true,
|
||||
topP: true,
|
||||
|
|
@ -997,15 +806,9 @@ const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
|
|||
},
|
||||
mistral: OPENAI_COMPAT_BASE,
|
||||
gemini: OPENAI_COMPAT_BASE,
|
||||
// Kimi k2.5/k2.6 are reasoning-class; the API locks temperature
|
||||
// and top_p to fixed defaults and 400s on any other value:
|
||||
// "invalid temperature: only 1 is allowed for this model".
|
||||
// Hide both sliders so the user is not offered knobs the model
|
||||
// silently overrides. Backend additionally strips these fields via
|
||||
// PROVIDER_REGISTRY['kimi']['body_omit']. seed and parallel_tool_
|
||||
// calls are not in Kimi's documented Chat Completion schema
|
||||
// (https://platform.kimi.ai/docs/api/chat); hide them so users are
|
||||
// not offered controls that the upstream may silently drop or 400.
|
||||
// Kimi K2.x locks temperature + top_p ("only 1 is allowed for this
|
||||
// model"); seed + parallel_tool_calls aren't in the Chat schema
|
||||
// (platform.kimi.ai/docs/api/chat). Backend strips via body_omit.
|
||||
kimi: {
|
||||
temperature: false,
|
||||
topP: false,
|
||||
|
|
@ -1013,9 +816,7 @@ const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
|
|||
minP: false,
|
||||
repetitionPenalty: false,
|
||||
presencePenalty: true,
|
||||
// K2.5/K2.6 lock sampling the same way temperature/top_p are
|
||||
// locked; reviewers report non-default frequency_penalty 400s
|
||||
// upstream, so hide the slider and strip the field in body_omit.
|
||||
// K2.x 400s on non-default frequency_penalty; backend strips too.
|
||||
frequencyPenalty: false,
|
||||
seed: false,
|
||||
stop: true,
|
||||
|
|
@ -1050,18 +851,10 @@ const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
|
|||
timingsPerToken: false,
|
||||
postSamplingProbs: false,
|
||||
},
|
||||
// DeepSeek deprecated presence/frequency penalty and never published
|
||||
// `seed` or `parallel_tool_calls` in the current chat-completion
|
||||
// schema — see https://api-docs.deepseek.com/api/create-chat-completion
|
||||
// (body fields: messages, model, thinking, max_tokens, response_format,
|
||||
// stop, stream, stream_options, temperature, top_p, tools, tool_choice,
|
||||
// logprobs, top_logprobs, user_id). Chat-class (deepseek-chat /
|
||||
// deepseek-v4-flash non-thinking) accepts temperature, top_p, stop;
|
||||
// reasoning class (deepseek-reasoner / deepseek-v4-flash thinking-mode)
|
||||
// additionally ignores temperature, top_p, presence_penalty,
|
||||
// frequency_penalty per
|
||||
// https://api-docs.deepseek.com/guides/reasoning_model. Per-model
|
||||
// resolution in getProviderCapabilities downshifts reasoner ids.
|
||||
// DeepSeek schema (api-docs.deepseek.com/api/create-chat-completion)
|
||||
// lists temperature/top_p/stop only — no seed or parallel_tool_calls.
|
||||
// Presence/frequency are deprecated. Reasoner ids additionally ignore
|
||||
// temperature/top_p; getProviderCapabilities downshifts them.
|
||||
deepseek: {
|
||||
temperature: true,
|
||||
topP: true,
|
||||
|
|
@ -1105,18 +898,10 @@ const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
|
|||
},
|
||||
qwen: OPENAI_COMPAT_BASE,
|
||||
huggingface: OPENAI_COMPAT_BASE,
|
||||
// OpenRouter surfaces the gateway's documented sampling field set
|
||||
// (incl. top_a). llama.cpp-specific knobs (typical_p, mirostat,
|
||||
// dynatemp, top_n_sigma, repeat_last_n) are gated off because the
|
||||
// OpenRouter API docs do not list them; they would be silently
|
||||
// dropped on most underlying models.
|
||||
openrouter: OPENROUTER_CAPABILITIES,
|
||||
// `llama_cpp` and the permissive `custom` preset terminate at the
|
||||
// first-party llama-server runtime, so the full sampler chain is
|
||||
// available. vLLM surfaces the OpenAI subset + top_k/min_p/
|
||||
// repetition_penalty/seed (no extended llama.cpp samplers). Ollama
|
||||
// is stricter: its OAI translator drops top_k/min_p/repetition_penalty
|
||||
// too on the /v1 path.
|
||||
// llama_cpp + custom: first-party llama-server, full chain.
|
||||
// vllm: OAI subset + top_k/min_p/repetition_penalty/seed.
|
||||
// ollama: stricter — OAI translator drops top_k/min_p/rep_pen too.
|
||||
custom: LLAMA_CPP_CAPABILITIES,
|
||||
llama_cpp: LLAMA_CPP_CAPABILITIES,
|
||||
vllm: VLLM_CAPABILITIES,
|
||||
|
|
@ -1125,23 +910,14 @@ const PROVIDER_CAPABILITIES: Record<string, ProviderCapabilities> = {
|
|||
|
||||
const DEFAULT_EXTERNAL_CAPABILITIES = OPENAI_COMPAT_BASE;
|
||||
|
||||
/**
|
||||
* Resolve the capability set for an external provider, optionally
|
||||
* specialised by model id. Returns `null` for a local model (i.e. when
|
||||
* `providerType` is null/undefined), which callers should treat as
|
||||
* "every knob applies".
|
||||
*
|
||||
* Per-model specialisations:
|
||||
* - openai + non-reasoning model (gpt-4o, gpt-4-turbo, gpt-4,
|
||||
* gpt-3.5-turbo): full sampling surface (OPENAI_CHAT_CAPABILITIES).
|
||||
* - openai + reasoning model (gpt-5.x, o1, o3, o4): restrictive
|
||||
* (OPENAI_REASONING_CAPABILITIES).
|
||||
* - anthropic + claude-opus-4-7: temperature/top_p/top_k stripped to
|
||||
* match the backend 400-avoidance regex (Sonnet/Haiku 4.7 do not
|
||||
* ship; only Opus does in the 4.7 generation).
|
||||
* - deepseek + reasoning model (deepseek-reasoner / r1): hides
|
||||
* temperature/top_p (silently ignored upstream).
|
||||
*/
|
||||
// Per-model specialisations:
|
||||
// openai + chat-class (gpt-4o, gpt-4-turbo, gpt-4, gpt-3.5):
|
||||
// full sampling surface (OPENAI_CHAT_CAPABILITIES).
|
||||
// openai + reasoning (gpt-5.x, o1, o3, o4): OPENAI_REASONING_CAPABILITIES.
|
||||
// anthropic + claude-opus-4-7: strips temp/top_p/top_k (Opus only
|
||||
// in 4.7; Sonnet/Haiku don't ship).
|
||||
// deepseek reasoner: hides temp/top_p (silently ignored upstream).
|
||||
// Returns null for local models (caller treats as "every knob applies").
|
||||
export function getProviderCapabilities(
|
||||
providerType: string | null | undefined,
|
||||
modelId?: string | null | undefined,
|
||||
|
|
@ -1161,10 +937,9 @@ export function getProviderCapabilities(
|
|||
}
|
||||
|
||||
const DEFAULT_EFFORT_LEVELS = ["low", "medium", "high"] as const;
|
||||
// OpenRouter ids that have NO non-reasoning mode. `google/gemini-pro-latest`
|
||||
// used to live here but the gateway 404s the id today
|
||||
// (https://openrouter.ai/google/gemini-pro-latest); drop it rather than
|
||||
// re-pin to a versioned id that may rotate again.
|
||||
// OpenRouter ids with no non-reasoning mode. (google/gemini-pro-latest
|
||||
// was dropped — gateway 404s; don't re-pin to a versioned id that
|
||||
// may rotate again.)
|
||||
const OPENROUTER_MANDATORY_REASONING_MODELS = new Set([
|
||||
"baidu/cobuddy:free",
|
||||
"inclusionai/ring-2.6-1t:free",
|
||||
|
|
@ -1196,9 +971,12 @@ const NO_REASONING_CAPS: ReasoningCaps = {
|
|||
reasoningEffortLevels: DEFAULT_EFFORT_LEVELS,
|
||||
};
|
||||
|
||||
// Order matters: longest/most-specific prefixes first so the find() loop
|
||||
// in resolveAnthropicReasoningEffortCapabilities lands the right bucket
|
||||
// before the bare-family fallback ("claude-opus-4") sweeps an id.
|
||||
// Order matters: longest prefixes first so find() picks the right
|
||||
// bucket before the bare-family fallback ("claude-opus-4") sweeps.
|
||||
// Levels per platform.claude.com/docs/en/about-claude/models/overview;
|
||||
// 4.5 line uses budget_tokens mapped by the backend. Legacy 4.x
|
||||
// (opus-4-1 / opus-4 / sonnet-4) supports Extended thinking per the
|
||||
// overview table; sonnet-4 / opus-4 retire 2026-06-15.
|
||||
const ANTHROPIC_REASONING_MODELS = [
|
||||
{
|
||||
prefixes: ["claude-opus-4-7"],
|
||||
|
|
@ -1210,13 +988,9 @@ const ANTHROPIC_REASONING_MODELS = [
|
|||
},
|
||||
{
|
||||
prefixes: ["claude-opus-4-5", "claude-sonnet-4-5", "claude-haiku-4-5"],
|
||||
// Backend maps semantic levels to manual budget_tokens.
|
||||
levels: ["none", "low", "medium", "high"],
|
||||
},
|
||||
{
|
||||
// Legacy 4.x models. Live overview lists "Extended thinking = Yes"
|
||||
// for opus-4-1, sonnet-4, opus-4 (the latter two retire 2026-06-15
|
||||
// but the registry still surfaces them).
|
||||
prefixes: ["claude-opus-4-1", "claude-opus-4", "claude-sonnet-4"],
|
||||
levels: ["none", "low", "medium", "high"],
|
||||
},
|
||||
|
|
@ -1261,18 +1035,14 @@ const OPENAI_REASONING_MODELS = [
|
|||
levels: ["medium"],
|
||||
},
|
||||
{
|
||||
// gpt-5.3-codex per dev page lists ONLY low/medium/high/xhigh
|
||||
// (https://developers.openai.com/api/docs/models/gpt-5.3-codex);
|
||||
// `none` is not in the codex enum so supportsOff stays false.
|
||||
// gpt-5.3-codex enum is low/medium/high/xhigh only per dev page.
|
||||
prefixes: ["gpt-5.3-codex"],
|
||||
supportsOff: false,
|
||||
levels: ["low", "medium", "high", "xhigh"],
|
||||
},
|
||||
{
|
||||
// Original gpt-5: minimal is supported, but per Azure footnote ^7^
|
||||
// "minimal is only supported with the original GPT-5 reasoning
|
||||
// models. minimal is not supported with gpt-5.1 or greater".
|
||||
// Listed before the gpt-5.1/5.2 entry so the longer match wins.
|
||||
// Azure footnote ^7^: minimal supported only on original gpt-5.
|
||||
// Listed before the bare gpt-5 entry so the longer match wins.
|
||||
prefixes: ["gpt-5.1", "gpt-5.2"],
|
||||
supportsOff: true,
|
||||
levels: ["none", "low", "medium", "high", "xhigh"],
|
||||
|
|
@ -1283,12 +1053,7 @@ const OPENAI_REASONING_MODELS = [
|
|||
levels: ["minimal", "low", "medium", "high"],
|
||||
},
|
||||
{
|
||||
// o-series reasoning models: o1, o3, o3-mini, o3-pro, o4-mini,
|
||||
// codex-mini all expose low/medium/high reasoning_effort per
|
||||
// developers.openai.com/api/docs/models/o3 and the Azure Foundry
|
||||
// o-series table. Without this entry o1/o4/codex-mini fell into
|
||||
// NO_REASONING_CAPS and the panel hid the effort slider — a real
|
||||
// UX regression for users on those ids.
|
||||
// o-series all accept low/medium/high per dev pages + Azure table.
|
||||
prefixes: ["o1", "o3", "o4", "codex-mini"],
|
||||
supportsOff: false,
|
||||
levels: DEFAULT_EFFORT_LEVELS,
|
||||
|
|
@ -1352,11 +1117,10 @@ function resolveKimiReasoningCapabilities(modelId: string): ExternalReasoningCap
|
|||
}
|
||||
|
||||
function resolveMistralReasoningCapabilities(modelId: string): ExternalReasoningCapabilities {
|
||||
// Native always-on reasoning family: magistral-* per
|
||||
// https://mistral.ai/news/magistral and
|
||||
// https://docs.mistral.ai/studio-api/conversations/reasoning .
|
||||
// "Always reasons; no parameter needed" — injecting reasoning_effort
|
||||
// returns 422 upstream. Treat like an OpenAI o-series always-on.
|
||||
// magistral-* is native always-on (no reasoning_effort param; 422 if
|
||||
// injected). mistral-{small,medium,vibe-cli}-latest is adjustable
|
||||
// none/low/medium/high. See docs.mistral.ai/studio-api/conversations/
|
||||
// reasoning + mistral.ai/news/magistral.
|
||||
if (
|
||||
modelId === "magistral-medium-latest" ||
|
||||
modelId === "magistral-small-latest"
|
||||
|
|
@ -1366,9 +1130,6 @@ function resolveMistralReasoningCapabilities(modelId: string): ExternalReasoning
|
|||
reasoningAlwaysOn: true,
|
||||
});
|
||||
}
|
||||
// Adjustable reasoning family: three documented levels low/medium/high
|
||||
// plus the "none" off-switch (Mistral Studio conversations doc). The
|
||||
// earlier two-level ["none","high"] ladder was wrong.
|
||||
if (
|
||||
modelId === "mistral-small-latest" ||
|
||||
modelId === "mistral-medium-latest" ||
|
||||
|
|
@ -1402,11 +1163,9 @@ function resolveConnectionLevelReasoning(
|
|||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* resolve external-model thinking capabilities.
|
||||
* provider-specific matching lives in the OpenAI/Anthropic resolvers.
|
||||
* other providers default to no reasoning controls.
|
||||
*/
|
||||
// Provider-specific matching lives in the per-provider resolvers
|
||||
// (resolveOpenAI / Anthropic / Kimi / Mistral...). Unknown providers
|
||||
// default to no reasoning controls.
|
||||
export function getExternalReasoningCapabilities(
|
||||
providerType: string | null | undefined,
|
||||
modelId: string | null | undefined,
|
||||
|
|
@ -1448,9 +1207,8 @@ export function getExternalReasoningCapabilities(
|
|||
const isOpenRouterProvider = normalizedProvider === "openrouter";
|
||||
if (isOpenRouterProvider) {
|
||||
// OpenRouter's unified `reasoning` parameter is accepted on every
|
||||
// chat-completion request; the gateway silently no-ops for models
|
||||
// that don't reason. Mandatory-reasoning ids are handled by the
|
||||
// early guard above; everything else exposes a toggleable control.
|
||||
// request; gateway no-ops for non-reasoning models. Mandatory ids
|
||||
// already handled above; everything else exposes a toggle.
|
||||
return {
|
||||
supportsReasoning: true,
|
||||
reasoningStyle: "enable_thinking",
|
||||
|
|
|
|||
|
|
@ -282,33 +282,16 @@ export interface OpenAIChatCompletionsRequest {
|
|||
* the Anthropic provider with `code_execution` in `enabled_tools`.
|
||||
*/
|
||||
anthropic_code_exec_container_id?: string | null;
|
||||
/**
|
||||
* OpenAI Chat Completions only; rejected by the Responses family and
|
||||
* silently dropped by Anthropic. Range -2.0 .. 2.0.
|
||||
*/
|
||||
/** OpenAI Chat only. Range -2..2. */
|
||||
frequency_penalty?: number;
|
||||
/**
|
||||
* Best-effort determinism seed. OpenAI Chat / OpenAI-compat backends
|
||||
* forward it; Responses + Anthropic drop it server-side.
|
||||
*/
|
||||
/** OAI Chat + most OAI-compat. Responses + Anthropic drop. */
|
||||
seed?: number;
|
||||
/**
|
||||
* Custom stop sequences. Backend translates to `stop_sequences` for
|
||||
* Anthropic; OpenAI Chat caps at 4 entries (server-side truncates
|
||||
* with a warning). Empty arrays are omitted.
|
||||
*/
|
||||
/** OAI Chat caps at 4; Anthropic mapped to `stop_sequences`. */
|
||||
stop?: string[];
|
||||
/**
|
||||
* Provider service tier. Anthropic accepts `auto|standard_only`;
|
||||
* OpenAI Chat + Responses both accept
|
||||
* `auto|default|flex|scale|priority` per the live `openai-python`
|
||||
* SDK (`src/openai/types/responses/response_create_params.py`
|
||||
* declares `Optional[Literal["auto", "default", "flex", "scale",
|
||||
* "priority"]]`). The wire-side helper in
|
||||
* `studio/backend/core/inference/external_provider.py` drops values
|
||||
* that a given provider does not accept; this union stays permissive
|
||||
* so the request-builder typechecks against
|
||||
* `InferenceParams.serviceTier` without per-provider narrowing.
|
||||
* Per-provider enum (see getServiceTierOptions). Union stays
|
||||
* permissive; external_provider.py drops values the active provider
|
||||
* doesn't accept.
|
||||
*/
|
||||
service_tier?:
|
||||
| "auto"
|
||||
|
|
@ -317,94 +300,66 @@ export interface OpenAIChatCompletionsRequest {
|
|||
| "priority"
|
||||
| "scale"
|
||||
| "standard_only";
|
||||
/**
|
||||
* Whether the provider may dispatch tool calls in parallel.
|
||||
* OpenAI: forwarded as `parallel_tool_calls`. Anthropic: inverted
|
||||
* into `disable_parallel_tool_use` server-side. Default `undefined`
|
||||
* keeps each provider's upstream default.
|
||||
*/
|
||||
/** Anthropic inverts to `disable_parallel_tool_use`. */
|
||||
parallel_tool_calls?: boolean;
|
||||
/**
|
||||
* llama.cpp `typ_p` (locally typical sampling). Local llama-server
|
||||
* only — no SaaS provider currently accepts this. 1.0 disables
|
||||
* (llama-server default). External-provider capability map already
|
||||
* gates this off, so on the wire it only appears for local + the
|
||||
* permissive {custom, vllm, ollama, llama_cpp} buckets.
|
||||
*/
|
||||
/** llama.cpp `typ_p`. 1.0 disables. */
|
||||
typical_p?: number;
|
||||
/** llama.cpp `top_n_sigma`. -1 disables. Local only. */
|
||||
/** llama.cpp `top_n_sigma`. -1 disables. */
|
||||
top_n_sigma?: number;
|
||||
/** llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. Local only. */
|
||||
/** llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. */
|
||||
repeat_last_n?: number;
|
||||
/** llama.cpp `dynatemp_range`. 0 disables. Local only. */
|
||||
/** llama.cpp `dynatemp_range`. 0 disables. */
|
||||
dynatemp_range?: number;
|
||||
/** llama.cpp `dynatemp_exponent`. Local only, paired with dynatemp_range. */
|
||||
/** llama.cpp `dynatemp_exponent`. Pairs with dynatemp_range. */
|
||||
dynatemp_exponent?: number;
|
||||
/** llama.cpp `mirostat` (0/1/2). 0 disables. Local only. */
|
||||
/** llama.cpp `mirostat` (0/1/2). 0 disables. */
|
||||
mirostat?: number;
|
||||
/** llama.cpp `mirostat_tau` target entropy. Local only. */
|
||||
mirostat_tau?: number;
|
||||
/** llama.cpp `mirostat_eta` learning rate. Local only. */
|
||||
mirostat_eta?: number;
|
||||
/**
|
||||
* OpenRouter `top_a` (alternate dynamic-top-P).
|
||||
* https://openrouter.ai/docs/api/reference/parameters — gateway-only.
|
||||
*/
|
||||
/** OpenRouter `top_a`. https://openrouter.ai/docs/api/reference/parameters */
|
||||
top_a?: number;
|
||||
/**
|
||||
* Anthropic fast-mode toggle. Opus 4.6 / 4.7 only; backend drops
|
||||
* silently on every other model + provider. See
|
||||
* https://platform.claude.com/docs/en/build-with-claude/fast-mode
|
||||
*/
|
||||
/** Anthropic Opus 4.6 / 4.7 only. https://platform.claude.com/docs/en/build-with-claude/fast-mode */
|
||||
fast_mode?: boolean | null;
|
||||
/**
|
||||
* llama.cpp DRY (Don't Repeat Yourself) sampler family. All four
|
||||
* fields documented at
|
||||
* llama.cpp DRY sampler (4 fields). `dry_multiplier=0` disables.
|
||||
* https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md
|
||||
* 0.0 / null on `dry_multiplier` disables the whole chain. Local only.
|
||||
*/
|
||||
dry_multiplier?: number;
|
||||
/** llama.cpp DRY base. Default 1.75. Local only. */
|
||||
/** Default 1.75. */
|
||||
dry_base?: number;
|
||||
/** llama.cpp DRY allowed length threshold. Default 2. Local only. */
|
||||
/** Default 2. */
|
||||
dry_allowed_length?: number;
|
||||
/** llama.cpp DRY penalty scan window. 0 disables, -1 = ctx-size. Local only. */
|
||||
/** 0 disables, -1 = ctx-size. */
|
||||
dry_penalty_last_n?: number;
|
||||
/** llama.cpp XTC sampler probability. 0.0 disables. Local only. */
|
||||
/** llama.cpp XTC. 0 disables. */
|
||||
xtc_probability?: number;
|
||||
/** llama.cpp XTC sampler threshold. Default 0.1. Local only. */
|
||||
/** Default 0.1. */
|
||||
xtc_threshold?: number;
|
||||
/** llama.cpp `min_keep` (force min N tokens past filters). Local only. */
|
||||
/** llama.cpp `min_keep`. */
|
||||
min_keep?: number;
|
||||
/**
|
||||
* Continue generating past the model's EOS token. llama.cpp + vLLM only.
|
||||
* `false` matches each backend's upstream default.
|
||||
*/
|
||||
/** Continue past EOS. llama.cpp + vLLM. */
|
||||
ignore_eos?: boolean;
|
||||
/**
|
||||
* Minimum output tokens before stop / EOS can fire. vLLM + llama.cpp only.
|
||||
* 0 disables.
|
||||
*/
|
||||
/** Min tokens before stop / EOS. llama.cpp + vLLM. */
|
||||
min_tokens?: number;
|
||||
/** vLLM `skip_special_tokens` — default true; forward only when false. */
|
||||
/** vLLM only. */
|
||||
skip_special_tokens?: boolean;
|
||||
/** vLLM `spaces_between_special_tokens` — default true; forward only when false. */
|
||||
/** vLLM only. */
|
||||
spaces_between_special_tokens?: boolean;
|
||||
/** vLLM `include_stop_str_in_output` — default false; forward only when true. */
|
||||
/** vLLM only. Useful for agentic tools. */
|
||||
include_stop_str_in_output?: boolean;
|
||||
/** vLLM `truncate_prompt_tokens` — left-truncate the prompt. > 0 only. */
|
||||
/** vLLM only. Left-truncate the prompt. */
|
||||
truncate_prompt_tokens?: number;
|
||||
/** llama.cpp `n_keep` — tokens to retain on context overflow. -1 = all. */
|
||||
/** llama.cpp `n_keep`. -1 = keep all. */
|
||||
n_keep?: number;
|
||||
/** llama.cpp `n_probs` — return top-N token probabilities. > 0 only. */
|
||||
/** llama.cpp `n_probs`. */
|
||||
n_probs?: number;
|
||||
/** llama.cpp `cache_prompt` — KV-cache reuse. Default true upstream; forward only when false. */
|
||||
/** llama.cpp `cache_prompt`. */
|
||||
cache_prompt?: boolean;
|
||||
/** llama.cpp `return_tokens` — include raw token IDs in response. Default false. */
|
||||
/** llama.cpp `return_tokens` (debug). */
|
||||
return_tokens?: boolean;
|
||||
/** llama.cpp `timings_per_token` — include per-token speed metrics. Default false. */
|
||||
/** llama.cpp `timings_per_token` (perf debug). */
|
||||
timings_per_token?: boolean;
|
||||
/** llama.cpp `post_sampling_probs` — token probs after the sampler chain. Default false. */
|
||||
/** llama.cpp `post_sampling_probs` (sampler debug). */
|
||||
post_sampling_probs?: boolean;
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -9,6 +9,11 @@ export type ServiceTier =
|
|||
| "scale"
|
||||
| "standard_only";
|
||||
|
||||
// All `number | null` / `boolean | null` fields below follow the same
|
||||
// convention: `null` = field omitted from the wire request (provider
|
||||
// uses its own default). Per-provider capability gating lives in
|
||||
// provider-capabilities.ts; the chat-adapter forwards only when the
|
||||
// active provider's bucket has the matching flag set true.
|
||||
export interface InferenceParams {
|
||||
temperature: number;
|
||||
topP: number;
|
||||
|
|
@ -16,150 +21,80 @@ export interface InferenceParams {
|
|||
minP: number;
|
||||
repetitionPenalty: number;
|
||||
presencePenalty: number;
|
||||
/** OpenAI Chat Completions only; rejected by Responses + Anthropic. */
|
||||
/** OpenAI Chat only; rejected by Responses + Anthropic. */
|
||||
frequencyPenalty: number;
|
||||
/**
|
||||
* Best-effort determinism seed. OpenAI Chat Completions only; the
|
||||
* Responses family and Anthropic reject it (silently dropped server-side).
|
||||
* `null` = unset (no `seed` field on the wire).
|
||||
*/
|
||||
/** Determinism seed. OpenAI Chat + most OAI-compat backends only. */
|
||||
seed: number | null;
|
||||
/**
|
||||
* Custom stop sequences. Maps to `stop` on OpenAI Chat Completions and
|
||||
* `stop_sequences` on Anthropic Messages. OpenAI caps the array at 4
|
||||
* entries; backend truncates with a warning. Empty array = unset.
|
||||
*/
|
||||
/** OAI Chat `stop` / Anthropic `stop_sequences`. OAI caps at 4. */
|
||||
stop: string[];
|
||||
/**
|
||||
* Provider service tier. Each provider accepts a different enum set;
|
||||
* `getServiceTierOptions(providerType)` resolves the legal values. `null`
|
||||
* means "let the provider pick its default" and is the safe choice on
|
||||
* provider switch.
|
||||
*/
|
||||
/** Per-provider enum via `getServiceTierOptions`. `null` = provider default. */
|
||||
serviceTier: ServiceTier | null;
|
||||
/**
|
||||
* Whether the provider may dispatch tool calls in parallel. Maps to
|
||||
* `parallel_tool_calls` on both OpenAI APIs and is inverted into
|
||||
* `disable_parallel_tool_use` for Anthropic. Default true matches the
|
||||
* upstream defaults across all three.
|
||||
*/
|
||||
/** Anthropic inverts to `disable_parallel_tool_use`. */
|
||||
parallelToolCalls: boolean;
|
||||
/**
|
||||
* Locally typical sampling (llama.cpp `typ_p`). Local llama-server
|
||||
* only — no SaaS provider currently accepts this. 1.0 disables (and
|
||||
* is the llama-server default). `null` = unset (not forwarded).
|
||||
*/
|
||||
/** llama.cpp `typ_p`. 1.0 disables. */
|
||||
typicalP: number | null;
|
||||
/** llama.cpp `top_n_sigma`. -1 disables. `null` = unset. */
|
||||
/** llama.cpp `top_n_sigma`. -1 disables. */
|
||||
topNSigma: number | null;
|
||||
/** llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. `null` = unset. */
|
||||
/** llama.cpp `repeat_last_n`. 0 disables, -1 = ctx-size. */
|
||||
repeatLastN: number | null;
|
||||
/** llama.cpp `dynatemp_range`. 0.0 disables. `null` = unset. */
|
||||
/** llama.cpp `dynatemp_range`. 0 disables. */
|
||||
dynatempRange: number | null;
|
||||
/** llama.cpp `dynatemp_exponent`. `null` = unset. */
|
||||
/** llama.cpp `dynatemp_exponent`. Pairs with dynatempRange. */
|
||||
dynatempExponent: number | null;
|
||||
/** llama.cpp `mirostat` mode (0/1/2). 0 disables. `null` = unset. */
|
||||
/** llama.cpp `mirostat` (0/1/2). 0 disables. */
|
||||
mirostat: number | null;
|
||||
/** llama.cpp `mirostat_tau` target entropy. `null` = unset. */
|
||||
mirostatTau: number | null;
|
||||
/** llama.cpp `mirostat_eta` learning rate. `null` = unset. */
|
||||
mirostatEta: number | null;
|
||||
/**
|
||||
* OpenRouter `top_a` alternate dynamic-top-P. OpenRouter-only.
|
||||
* Range [0, 1]. `null` = unset.
|
||||
*/
|
||||
/** OpenRouter `top_a`. Range [0, 1]. */
|
||||
topA: number | null;
|
||||
/**
|
||||
* llama.cpp DRY (Don't Repeat Yourself) penalty multiplier.
|
||||
* 0.0 disables (server default). `null` = unset.
|
||||
* https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md
|
||||
* llama.cpp DRY sampler — multiplier is the master switch (0 disables
|
||||
* the 4-field chain). See llama.cpp/tools/server/README.md.
|
||||
*/
|
||||
dryMultiplier: number | null;
|
||||
/** llama.cpp DRY base value (exponential growth base). Default 1.75. `null` = unset. */
|
||||
/** Default 1.75. */
|
||||
dryBase: number | null;
|
||||
/** llama.cpp DRY allowed token-extension threshold. Default 2. `null` = unset. */
|
||||
/** Default 2. */
|
||||
dryAllowedLength: number | null;
|
||||
/** llama.cpp DRY penalty scan window. 0 disables, -1 = ctx-size. `null` = unset. */
|
||||
/** 0 disables, -1 = ctx-size. */
|
||||
dryPenaltyLastN: number | null;
|
||||
/** llama.cpp XTC sampler probability. 0.0 disables. `null` = unset. */
|
||||
/** llama.cpp XTC — probability is the master switch (0 disables). */
|
||||
xtcProbability: number | null;
|
||||
/** llama.cpp XTC sampler threshold. Default 0.1. `null` = unset. */
|
||||
/** Default 0.1. */
|
||||
xtcThreshold: number | null;
|
||||
/** llama.cpp `min_keep` (force min N tokens past filters). 0 disables. `null` = unset. */
|
||||
/** llama.cpp `min_keep` — min tokens past all filters. 0 disables. */
|
||||
minKeep: number | null;
|
||||
/**
|
||||
* Force generation past the EOS token. llama.cpp + vLLM accept this.
|
||||
* `null` = unset; `false` matches upstream default.
|
||||
*/
|
||||
/** Continue past EOS. llama.cpp + vLLM. */
|
||||
ignoreEos: boolean | null;
|
||||
/**
|
||||
* Minimum output tokens before stop sequences / EOS can fire.
|
||||
* vLLM + llama.cpp accept this. 0 disables. `null` = unset.
|
||||
*/
|
||||
/** Min tokens before stop / EOS can fire. llama.cpp + vLLM. */
|
||||
minTokens: number | null;
|
||||
/**
|
||||
* vLLM `skip_special_tokens`. Default true. Forward only when false
|
||||
* (i.e. user wants to see raw special tokens in the output).
|
||||
* https://docs.vllm.ai/en/latest/api/vllm/sampling_params/
|
||||
*/
|
||||
/** vLLM only. Default true; forward only when false. */
|
||||
skipSpecialTokens: boolean | null;
|
||||
/**
|
||||
* vLLM `spaces_between_special_tokens`. Default true. Forward only
|
||||
* when false.
|
||||
*/
|
||||
/** vLLM only. Default true; forward only when false. */
|
||||
spacesBetweenSpecialTokens: boolean | null;
|
||||
/**
|
||||
* vLLM `include_stop_str_in_output`. Default false. Useful for
|
||||
* agentic tools that need the matched stop string echoed back.
|
||||
*/
|
||||
/** vLLM only. Useful for agentic tools needing the matched stop string echoed. */
|
||||
includeStopStrInOutput: boolean | null;
|
||||
/**
|
||||
* vLLM `truncate_prompt_tokens` — left-truncate the prompt to this
|
||||
* many tokens. Useful for long-context overflow. `null` = unset.
|
||||
*/
|
||||
/** vLLM only. Left-truncate the prompt. */
|
||||
truncatePromptTokens: number | null;
|
||||
/**
|
||||
* llama.cpp `n_keep` — tokens to retain when context overflows.
|
||||
* 0 disables, -1 keeps all. `null` = unset.
|
||||
*/
|
||||
/** llama.cpp `n_keep`. 0 disables, -1 = keep all. */
|
||||
nKeep: number | null;
|
||||
/**
|
||||
* llama.cpp `n_probs` — return top-N token probabilities per
|
||||
* generated token. 0 disables. `null` = unset.
|
||||
*/
|
||||
/** llama.cpp `n_probs` — top-N token probabilities per token. */
|
||||
nProbs: number | null;
|
||||
/**
|
||||
* llama.cpp `cache_prompt` — reuse KV cache from previous prompts
|
||||
* with a shared prefix. Default true upstream. Forward only when
|
||||
* explicitly false (e.g. for deterministic benchmarks).
|
||||
*/
|
||||
/** llama.cpp `cache_prompt`. Default true; forward only when false. */
|
||||
cachePrompt: boolean | null;
|
||||
/**
|
||||
* llama.cpp `return_tokens` — include raw token IDs in the response.
|
||||
* Debug. Default false.
|
||||
*/
|
||||
/** llama.cpp `return_tokens` (debug). */
|
||||
returnTokens: boolean | null;
|
||||
/**
|
||||
* llama.cpp `timings_per_token` — include per-token speed metrics.
|
||||
* Default false.
|
||||
*/
|
||||
/** llama.cpp `timings_per_token` (perf debug). */
|
||||
timingsPerToken: boolean | null;
|
||||
/**
|
||||
* llama.cpp `post_sampling_probs` — return token probabilities AFTER
|
||||
* the sampler chain runs. Debug. Default false.
|
||||
*/
|
||||
/** llama.cpp `post_sampling_probs` (sampler debug). */
|
||||
postSamplingProbs: boolean | null;
|
||||
maxSeqLength: number;
|
||||
maxTokens: number;
|
||||
systemPrompt: string;
|
||||
checkpoint: string;
|
||||
/** Allow loading models with custom code (e.g. NVIDIA Nemotron). Only enable for repos you trust. */
|
||||
/** Trust custom model code (e.g. NVIDIA Nemotron). Only for trusted repos. */
|
||||
trustRemoteCode?: boolean;
|
||||
/**
|
||||
* Anthropic fast-mode toggle. Opus 4.6 / 4.7 only; higher OTPS at
|
||||
* 6x standard Opus pricing. Default false.
|
||||
* https://platform.claude.com/docs/en/build-with-claude/fast-mode
|
||||
*/
|
||||
/** Anthropic Opus 4.6 / 4.7 only. 6x pricing for higher OTPS. */
|
||||
fastMode?: boolean;
|
||||
}
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue