Studio: surface Gemini 3.5/3.1/3 + Nano Banana 2/Pro and plumb thinking budget
`gemini-2.0-flash` / `gemini-2.0-flash-exp` were retired by Google in 2026
(`/v1beta/models/gemini-2.0-flash:streamGenerateContent` returns HTTP 404
"no longer available to new users"), and the picker had nothing past the
2.x family. Verified against the live ListModels catalog: drop the retired
ids from `default_models` + allowlist and surface the chat-capable
3.5 / 3.1 / 3 families plus the Nano Banana image trio.
Also plumb `enable_thinking` / `reasoning_effort` into Gemini's
`generationConfig.thinkingConfig`. Without this, Gemini 3.5 Flash,
gemini-pro-latest, and the 3.x previews silently spend the caller's
`max_tokens` budget on hidden "thoughts" before emitting any visible
answer -- the chat shows a truncated stub like "The capital of" and
streams stop. Mapping:
- enable_thinking=False / reasoning_effort=none -> thinkingBudget=0
(Flash tier; Pro tier coerces to a small positive budget because
the API 400s on 0 with "This model only works in thinking mode")
- minimal/low/medium/high -> 512/2048/8192/24576 budget tokens
- max/xhigh -> -1 (dynamic)
- default (neither knob set) -> thinkingConfig omitted, model decides
Frontend `getExternalReasoningCapabilities` now surfaces a
`reasoning_effort` picker for every Gemini chat id (Pro tier hides the
"none" option; image-tier ids stay knob-less). Adds 6 unit tests
covering Flash/Pro effort mapping, the off-toggle coercion on Pro,
default omission, and the nano-banana-pro-preview alias routing
through the image modalities path. 28 -> 34 tests in
`test_gemini_provider.py`, all green; full backend suite still passes
(1459/1460; the unrelated test_help_output flake is pre-existing and
not in any file this PR touches).
Live verification against generativelanguage.googleapis.com on
2026-05-24 with `_stream_gemini` directly:
text gemini-3.5-flash single PASS multi PASS
text gemini-3.1-pro-preview single PASS multi PASS
text gemini-3.1-flash-lite single PASS multi PASS
text gemini-3-pro-preview single PASS multi PASS
text gemini-3-flash-preview single PASS multi PASS
text gemini-2.5-pro single PASS multi PASS
text gemini-2.5-flash single PASS multi PASS
text gemini-2.5-flash-lite single PASS multi PASS
text gemini-flash-latest single PASS multi PASS
text gemini-flash-lite-latest single PASS multi PASS
text gemini-pro-latest single PASS multi PASS
image gemini-2.5-flash-image PASS (1082 KB png returned)
image gemini-3.1-flash-image-preview PASS (Nano Banana 2)
image gemini-3-pro-image-preview PASS (Nano Banana Pro)
tool web_search PASS
tool code_execution PASS
-> 16/16 e2e through the actual ExternalProviderClient code path.
This commit is contained in:
parent
696728d53f
commit
c6724dbdc7
4 changed files with 230 additions and 9 deletions
|
|
@ -377,6 +377,8 @@ class ExternalProviderClient:
|
|||
presence_penalty,
|
||||
enabled_tools,
|
||||
enable_prompt_caching,
|
||||
enable_thinking,
|
||||
reasoning_effort,
|
||||
):
|
||||
yield line
|
||||
return
|
||||
|
|
@ -2574,6 +2576,8 @@ class ExternalProviderClient:
|
|||
presence_penalty: float = 0.0,
|
||||
enabled_tools: Optional[list[str]] = None,
|
||||
enable_prompt_caching: Optional[Any] = None,
|
||||
enable_thinking: Optional[bool] = None,
|
||||
reasoning_effort: Optional[str] = None,
|
||||
) -> AsyncGenerator[str, None]:
|
||||
"""
|
||||
Call Google's native Gemini API and translate its streaming
|
||||
|
|
@ -2797,12 +2801,58 @@ class ExternalProviderClient:
|
|||
# we translate to a tool_end with image_b64/image_mime so the
|
||||
# chat UI renders the picture inline. See
|
||||
# https://ai.google.dev/gemini-api/docs/image-generation.
|
||||
is_image_model = "-image" in model.lower() or bool(
|
||||
is_image_model = "-image" in model.lower() or "nano-banana" in model.lower() or bool(
|
||||
enabled_tools and "image_generation" in enabled_tools
|
||||
)
|
||||
if is_image_model:
|
||||
gen_config["responseModalities"] = ["TEXT", "IMAGE"]
|
||||
|
||||
# Thinking budget plumbing. Gemini 3.x, 3.5 Flash, gemini-pro-latest
|
||||
# and gemini-flash-latest spend hidden "thoughts" tokens before they
|
||||
# produce the streamed answer. With a tight `max_tokens` budget the
|
||||
# answer can be entirely consumed by thoughts, so the caller's
|
||||
# `enable_thinking` / `reasoning_effort` knobs need to actually
|
||||
# reach `generationConfig.thinkingConfig`. Per
|
||||
# https://ai.google.dev/gemini-api/docs/thinking:
|
||||
# thinkingBudget = 0 -> disable thinking (Flash-tier only;
|
||||
# Pro-tier 400s with "only works in
|
||||
# thinking mode")
|
||||
# thinkingBudget = -1 -> dynamic / let the model decide
|
||||
# thinkingBudget = N>0 -> hard cap of N thought tokens
|
||||
# Pro-tier models cannot be turned off; we silently coerce
|
||||
# an "off" request to a small budget on those so they stay
|
||||
# responsive instead of 400ing the whole turn.
|
||||
_PRO_THINKING_ONLY = (
|
||||
"gemini-pro-latest",
|
||||
"gemini-3.1-pro",
|
||||
"gemini-3-pro",
|
||||
"gemini-2.5-pro",
|
||||
)
|
||||
_is_pro_thinking_only = any(p in model for p in _PRO_THINKING_ONLY)
|
||||
# Effort -> budget tokens. Mirrors the OpenAI ladder so the
|
||||
# frontend's existing "minimal/low/medium/high/max" picker maps
|
||||
# to sensible Gemini budgets. Match the keys the rest of Studio
|
||||
# uses (see `stream_chat_completion`'s reasoning_effort docstring
|
||||
# and the OpenAI / Anthropic helpers).
|
||||
_EFFORT_TO_BUDGET: dict[str, int] = {
|
||||
"minimal": 512,
|
||||
"low": 2048,
|
||||
"medium": 8192,
|
||||
"high": 24576,
|
||||
"xhigh": -1, # dynamic
|
||||
"max": -1, # dynamic
|
||||
}
|
||||
thinking_budget: Optional[int] = None
|
||||
effort_lc = (reasoning_effort or "").strip().lower()
|
||||
if effort_lc == "none" or enable_thinking is False:
|
||||
thinking_budget = 128 if _is_pro_thinking_only else 0
|
||||
elif effort_lc in _EFFORT_TO_BUDGET:
|
||||
thinking_budget = _EFFORT_TO_BUDGET[effort_lc]
|
||||
elif enable_thinking is True:
|
||||
thinking_budget = -1
|
||||
if thinking_budget is not None:
|
||||
gen_config["thinkingConfig"] = {"thinkingBudget": thinking_budget}
|
||||
|
||||
if gen_config:
|
||||
body["generationConfig"] = gen_config
|
||||
|
||||
|
|
|
|||
|
|
@ -72,14 +72,23 @@ PROVIDER_REGISTRY: dict[str, dict[str, Any]] = {
|
|||
"base_url": "https://generativelanguage.googleapis.com/v1beta",
|
||||
# Curated lineup -- the live ListModels response returns dozens
|
||||
# of historical / experimental / embedding ids. Cap to the
|
||||
# current 2.5/2.0 family plus the Nano Banana image model and
|
||||
# the rolling `*-latest` aliases.
|
||||
# current chat-capable Gemini families (3.5 / 3.1 / 3 / 2.5)
|
||||
# plus the Nano Banana image trio and the rolling `*-latest`
|
||||
# aliases. `gemini-2.0-flash*` were retired by Google in 2026
|
||||
# and are intentionally excluded; the allowlist below blocks
|
||||
# them from re-appearing through the live ListModels fetch.
|
||||
# Verified against the live `/v1beta/models` catalog 2026-05-24.
|
||||
"default_models": [
|
||||
"gemini-3.5-flash",
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3.1-flash-lite",
|
||||
"gemini-3-pro-preview",
|
||||
"gemini-3-flash-preview",
|
||||
"gemini-2.5-pro",
|
||||
"gemini-2.5-flash",
|
||||
"gemini-2.5-flash-lite",
|
||||
"gemini-2.0-flash",
|
||||
"gemini-2.0-flash-exp",
|
||||
"gemini-3-pro-image-preview",
|
||||
"gemini-3.1-flash-image-preview",
|
||||
"gemini-2.5-flash-image",
|
||||
],
|
||||
"supports_streaming": True,
|
||||
|
|
@ -95,11 +104,25 @@ PROVIDER_REGISTRY: dict[str, dict[str, Any]] = {
|
|||
"API key from https://aistudio.google.com/apikey. "
|
||||
"See https://ai.google.dev/gemini-api/docs for endpoint shapes."
|
||||
),
|
||||
# Matches the chat-capable 3.5 / 3.1 / 3 / 2.5 families plus the
|
||||
# rolling *-latest aliases (which Google rolls forward as new
|
||||
# generations ship). Image-tier ids (`-image`, `-image-preview`,
|
||||
# `nano-banana-pro-preview`) flow through the Nano Banana
|
||||
# `responseModalities` path in `_stream_gemini`. Retired 2.0
|
||||
# ids ARE NOT in this regex on purpose -- Google's ListModels
|
||||
# would otherwise re-surface them and they 404 on use.
|
||||
"model_id_allowlist": re.compile(
|
||||
r"^(gemini-2\.5-pro|gemini-2\.5-flash|gemini-2\.5-flash-lite|"
|
||||
r"gemini-2\.5-flash-image|gemini-2\.0-flash|"
|
||||
r"gemini-2\.0-flash-exp|gemini-pro-latest|"
|
||||
r"gemini-flash-latest|gemini-flash-lite-latest)$"
|
||||
r"^("
|
||||
r"gemini-3\.5-(?:flash|pro)(?:-preview)?|"
|
||||
r"gemini-3\.1-(?:flash|pro|flash-lite)(?:-preview)?(?:-customtools)?|"
|
||||
r"gemini-3\.1-flash-image-preview|"
|
||||
r"gemini-3-(?:flash|pro)(?:-preview)?|"
|
||||
r"gemini-3-pro-image-preview|"
|
||||
r"nano-banana-pro-preview|"
|
||||
r"gemini-2\.5-pro|gemini-2\.5-flash|gemini-2\.5-flash-lite|"
|
||||
r"gemini-2\.5-flash-image|"
|
||||
r"gemini-pro-latest|gemini-flash-latest|gemini-flash-lite-latest"
|
||||
r")$"
|
||||
),
|
||||
},
|
||||
"deepseek": {
|
||||
|
|
|
|||
|
|
@ -249,6 +249,90 @@ def test_presence_penalty_forwarded_to_generation_config(monkeypatch):
|
|||
assert "presencePenalty" not in captured["body"]["generationConfig"]
|
||||
|
||||
|
||||
# ── thinkingConfig translation ────────────────────────────────────────
|
||||
|
||||
|
||||
def test_thinking_disabled_sets_budget_zero_on_flash(monkeypatch):
|
||||
"""enable_thinking=False on Flash-tier sets thinkingBudget=0."""
|
||||
captured = _capture_body(
|
||||
monkeypatch,
|
||||
model = "gemini-3.5-flash",
|
||||
enable_thinking = False,
|
||||
)
|
||||
tc = captured["body"]["generationConfig"].get("thinkingConfig")
|
||||
assert tc == {"thinkingBudget": 0}, tc
|
||||
|
||||
|
||||
def test_thinking_disabled_pro_tier_uses_small_budget(monkeypatch):
|
||||
"""Pro-tier ids 400 on thinkingBudget=0 ("only works in thinking mode");
|
||||
a small positive budget is forwarded instead so the turn doesn't fail.
|
||||
"""
|
||||
for model in (
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3-pro-preview",
|
||||
"gemini-2.5-pro",
|
||||
"gemini-pro-latest",
|
||||
):
|
||||
captured = _capture_body(
|
||||
monkeypatch,
|
||||
model = model,
|
||||
enable_thinking = False,
|
||||
)
|
||||
tc = captured["body"]["generationConfig"].get("thinkingConfig")
|
||||
assert tc is not None, f"missing thinkingConfig for {model}: {captured}"
|
||||
assert tc["thinkingBudget"] > 0, (model, tc)
|
||||
|
||||
|
||||
def test_reasoning_effort_levels_map_to_budgets(monkeypatch):
|
||||
"""The OpenAI/Anthropic effort ladder must translate to Gemini budgets."""
|
||||
cases = {
|
||||
"minimal": 512,
|
||||
"low": 2048,
|
||||
"medium": 8192,
|
||||
"high": 24576,
|
||||
"max": -1, # dynamic
|
||||
"xhigh": -1,
|
||||
}
|
||||
for effort, expected in cases.items():
|
||||
captured = _capture_body(
|
||||
monkeypatch,
|
||||
model = "gemini-3.5-flash",
|
||||
reasoning_effort = effort,
|
||||
)
|
||||
tc = captured["body"]["generationConfig"].get("thinkingConfig")
|
||||
assert tc == {"thinkingBudget": expected}, (effort, tc)
|
||||
|
||||
|
||||
def test_reasoning_effort_none_disables_on_flash(monkeypatch):
|
||||
"""`reasoning_effort='none'` is shorthand for thinking off (Flash)."""
|
||||
captured = _capture_body(
|
||||
monkeypatch,
|
||||
model = "gemini-3.5-flash",
|
||||
reasoning_effort = "none",
|
||||
)
|
||||
tc = captured["body"]["generationConfig"].get("thinkingConfig")
|
||||
assert tc == {"thinkingBudget": 0}, tc
|
||||
|
||||
|
||||
def test_thinking_default_omits_thinking_config(monkeypatch):
|
||||
"""When neither knob is supplied, thinkingConfig is omitted entirely
|
||||
(Google's server-side default applies)."""
|
||||
captured = _capture_body(monkeypatch, model = "gemini-3.5-flash")
|
||||
gc = captured["body"]["generationConfig"]
|
||||
assert "thinkingConfig" not in gc, gc
|
||||
|
||||
|
||||
def test_nano_banana_alias_routes_through_image_modalities(monkeypatch):
|
||||
"""`nano-banana-pro-preview` is an alias for the Pro image model and
|
||||
must set responseModalities=[TEXT,IMAGE] same as the `*-image` ids."""
|
||||
captured = _capture_body(
|
||||
monkeypatch,
|
||||
model = "nano-banana-pro-preview",
|
||||
)
|
||||
gc = captured["body"]["generationConfig"]
|
||||
assert gc.get("responseModalities") == ["TEXT", "IMAGE"], gc
|
||||
|
||||
|
||||
# ── web_search forwarded as googleSearch tool ────────────────────────
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -577,6 +577,67 @@ function resolveKimiReasoningCapabilities(modelId: string): ExternalReasoningCap
|
|||
return withEnableThinkingStyle();
|
||||
}
|
||||
|
||||
// Gemini's thinking ladder.
|
||||
// - 3.5 / 3.1 / 3 Flash + Flash-Lite + 2.5 Flash + *-latest aliases:
|
||||
// toggleable thinking with effort levels (backend maps to
|
||||
// `thinkingConfig.thinkingBudget`).
|
||||
// - 3.x Pro + gemini-pro-latest + 2.5 Pro: "thinking only" -- the
|
||||
// API 400s on `thinkingBudget=0` ("This model only works in
|
||||
// thinking mode"), so the UI hides the off switch.
|
||||
// - 2.5 Flash-Lite: no native thinking surfaced; leave it off.
|
||||
// - Image-tier ids (`*-image*`, `nano-banana-pro-preview`): image
|
||||
// generation path -- no reasoning controls.
|
||||
const GEMINI_THINKING_PRO_PREFIXES = [
|
||||
"gemini-3.5-pro",
|
||||
"gemini-3.1-pro",
|
||||
"gemini-3-pro",
|
||||
"gemini-2.5-pro",
|
||||
"gemini-pro-latest",
|
||||
];
|
||||
const GEMINI_THINKING_FLASH_PREFIXES = [
|
||||
"gemini-3.5-flash",
|
||||
"gemini-3.1-flash",
|
||||
"gemini-3-flash",
|
||||
"gemini-2.5-flash",
|
||||
"gemini-flash-latest",
|
||||
"gemini-flash-lite-latest",
|
||||
];
|
||||
const GEMINI_IMAGE_HINTS = [
|
||||
"-image",
|
||||
"nano-banana",
|
||||
];
|
||||
function resolveGeminiReasoningCapabilities(
|
||||
modelId: string,
|
||||
): ExternalReasoningCapabilities {
|
||||
const m = modelId.toLowerCase();
|
||||
if (GEMINI_IMAGE_HINTS.some((h) => m.includes(h))) {
|
||||
// Image generation; no thinking knob.
|
||||
return withEnableThinkingStyle();
|
||||
}
|
||||
if (GEMINI_THINKING_PRO_PREFIXES.some((p) => m.startsWith(p))) {
|
||||
return withReasoningEffortStyle({
|
||||
supportsReasoning: true,
|
||||
// Pro tier: cannot turn thinking fully off.
|
||||
supportsReasoningOff: false,
|
||||
reasoningEffortLevels: ["low", "medium", "high", "max"] as const,
|
||||
});
|
||||
}
|
||||
if (GEMINI_THINKING_FLASH_PREFIXES.some((p) => m.startsWith(p))) {
|
||||
return withReasoningEffortStyle({
|
||||
supportsReasoning: true,
|
||||
supportsReasoningOff: true,
|
||||
reasoningEffortLevels: [
|
||||
"none",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"max",
|
||||
] as const,
|
||||
});
|
||||
}
|
||||
return withEnableThinkingStyle();
|
||||
}
|
||||
|
||||
function resolveMistralReasoningCapabilities(modelId: string): ExternalReasoningCapabilities {
|
||||
if (modelId === "magistral-medium-latest") {
|
||||
return withReasoningEffortStyle({
|
||||
|
|
@ -674,6 +735,9 @@ export function getExternalReasoningCapabilities(
|
|||
}
|
||||
if (isKimiProvider) return resolveKimiReasoningCapabilities(modelForMatching);
|
||||
if (isMistralProvider) return resolveMistralReasoningCapabilities(modelForMatching);
|
||||
if (normalizedProvider === "gemini") {
|
||||
return resolveGeminiReasoningCapabilities(modelForMatching);
|
||||
}
|
||||
if (!isOpenAIProvider && !isAnthropicProvider) {
|
||||
return withEnableThinkingStyle();
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue