Studio: drop non-streaming gpt-realtime + gpt-audio-mini from picker
Codex P1 round 2 on this PR: by allowing the full gpt-realtime and gpt-audio families through the denylist we let the chat picker surface IDs that OpenAI's model cards mark Streaming: Not supported (gpt- realtime, gpt-realtime-mini, gpt-audio-mini -- Realtime-API/WebSocket only). Studio's path always sends stream:true, so selecting them 4xx'd at request time instead of producing a reply. Narrow the regex to drop ^gpt-realtime(?:$|-) and ^gpt-audio-mini\b; gpt-audio (chat-streams over /v1/chat/completions) and gpt-4o-realtime- preview (chat-streams via the /v1/responses adapter) still pass. Updated test_provider_registry_filters.py to pin the new chat/non-chat split: 9/9 filter tests pass.
This commit is contained in:
parent
3661243e35
commit
1a7238c753
2 changed files with 36 additions and 53 deletions
|
|
@ -68,17 +68,19 @@ PROVIDER_REGISTRY: dict[str, dict[str, Any]] = {
|
|||
# Verified against the live /v1/models listing 2026-05-22.
|
||||
"model_id_denylist": re.compile(
|
||||
# Feature suffixes that mark a non-chat variant on any base.
|
||||
# Note: `audio` and `realtime` are omitted on purpose --
|
||||
# see docstring above. `translate` is here so the audio
|
||||
# translation variants `gpt-realtime-translate*` /
|
||||
# `gpt-audio-translate*` are dropped even though we keep
|
||||
# the parent `gpt-realtime` / `gpt-audio` chat families.
|
||||
# `instruct` catches the legacy completion-only suffix
|
||||
# used by `gpt-3.5-turbo-instruct*`; that family is wired
|
||||
# to /v1/completions and would 4xx on our chat/responses
|
||||
# transport, so it should not surface in the chat picker.
|
||||
# `audio`/`realtime` are NOT in this group so `gpt-audio`
|
||||
# (chat-capable, streaming) stays; specific non-streaming
|
||||
# IDs are caught further down.
|
||||
r"(?:^|-)(?:embedding|tts|whisper|moderation|image|"
|
||||
r"transcribe|translate|instruct|sora)\b"
|
||||
# OpenAI realtime/audio variants that don't support chat
|
||||
# streaming. Studio always sends `stream: true`, and OpenAI
|
||||
# marks `gpt-realtime*` and `gpt-audio-mini` as Streaming:
|
||||
# Not supported (Realtime API only); surfacing them in the
|
||||
# picker would 4xx at request time. `gpt-audio` and
|
||||
# `gpt-4o-realtime-preview` still stream and are kept.
|
||||
r"|^gpt-realtime(?:$|-)"
|
||||
r"|^gpt-audio-mini\b"
|
||||
# Legacy completion bases -- ^-anchored to avoid false
|
||||
# positives on hypothetical future chat ids containing
|
||||
# those words mid-string.
|
||||
|
|
|
|||
|
|
@ -56,12 +56,10 @@ def test_openai_keeps_every_known_chat_family():
|
|||
# search-api suffix.
|
||||
"gpt-4o-search-preview",
|
||||
"gpt-4o-mini-search-preview",
|
||||
# gpt-audio family: chat-completion-capable (text in / text or
|
||||
# audio out) per OpenAI's audio guide. Earlier revisions of
|
||||
# this denylist dropped them via `(?:^|-)audio`; the regex
|
||||
# comment block now keeps them.
|
||||
# gpt-audio is chat-completion-capable with streaming. The
|
||||
# `-mini` variant does NOT stream (Realtime API only) and is
|
||||
# asserted dropped in the realtime/audio split test below.
|
||||
"gpt-audio",
|
||||
"gpt-audio-mini",
|
||||
"gpt-audio-1.5",
|
||||
# Hypothetical future families that the old allowlist would have
|
||||
# silently dropped -- they MUST surface under the new denylist.
|
||||
|
|
@ -77,39 +75,35 @@ def test_openai_keeps_every_known_chat_family():
|
|||
assert surviving == live, surviving
|
||||
|
||||
|
||||
def test_openai_audio_and_realtime_families_are_chat_capable_and_kept():
|
||||
# Pin the chat/non-chat split that lives in the regex comment:
|
||||
# gpt-audio AND gpt-realtime are kept (both accept text in via
|
||||
# /v1/chat/completions and /v1/responses, not only their
|
||||
# specialised transports); gpt-4o-transcribe etc. are dropped
|
||||
# (audio-only input via /v1/audio/transcriptions).
|
||||
def test_openai_audio_and_realtime_families_split_on_streaming_support():
|
||||
# Pin the chat/non-chat split: Studio always sends `stream: true`,
|
||||
# so the picker keeps gpt-audio* (chat streams over /v1/chat/
|
||||
# completions) and gpt-4o-realtime-preview (chat streams via the
|
||||
# /v1/responses adapter), but drops the new gpt-realtime* family
|
||||
# and gpt-audio-mini, which OpenAI's model cards mark Streaming:
|
||||
# Not supported (Realtime API / WebSocket only).
|
||||
kept = _apply(
|
||||
"openai",
|
||||
[
|
||||
"gpt-audio",
|
||||
"gpt-audio-1.5",
|
||||
"gpt-audio-mini",
|
||||
"gpt-realtime",
|
||||
"gpt-realtime-mini",
|
||||
"gpt-realtime-1.5",
|
||||
"gpt-realtime-2",
|
||||
"gpt-4o-realtime-preview",
|
||||
],
|
||||
)
|
||||
assert kept == [
|
||||
"gpt-audio",
|
||||
"gpt-audio-1.5",
|
||||
"gpt-audio-mini",
|
||||
"gpt-realtime",
|
||||
"gpt-realtime-mini",
|
||||
"gpt-realtime-1.5",
|
||||
"gpt-realtime-2",
|
||||
"gpt-4o-realtime-preview",
|
||||
], kept
|
||||
|
||||
dropped = _apply(
|
||||
"openai",
|
||||
[
|
||||
"gpt-audio-mini",
|
||||
"gpt-realtime",
|
||||
"gpt-realtime-mini",
|
||||
"gpt-realtime-1.5",
|
||||
"gpt-realtime-2",
|
||||
"gpt-4o-transcribe",
|
||||
"gpt-4o-mini-transcribe",
|
||||
],
|
||||
|
|
@ -120,9 +114,9 @@ def test_openai_audio_and_realtime_families_are_chat_capable_and_kept():
|
|||
def test_openai_drops_non_chat_ids():
|
||||
noise = [
|
||||
# Embeddings / TTS / image / moderation / whisper / etc.
|
||||
# gpt-audio* and gpt-realtime* are intentionally OMITTED from
|
||||
# this list -- both are chat-capable. See
|
||||
# test_openai_audio_and_realtime_families_are_chat_capable_and_kept.
|
||||
# gpt-audio (no -mini) and gpt-4o-realtime-preview are
|
||||
# intentionally OMITTED from this list -- they DO stream chat.
|
||||
# See test_openai_audio_and_realtime_families_split_on_streaming_support.
|
||||
"text-embedding-3-small",
|
||||
"text-embedding-3-large",
|
||||
"text-embedding-ada-002",
|
||||
|
|
@ -169,26 +163,13 @@ def test_openai_drops_non_chat_ids():
|
|||
|
||||
|
||||
def test_openai_realtime_translate_variants_are_dropped_but_parent_chat_family_survives():
|
||||
"""gpt-realtime / gpt-realtime-mini are chat-capable and stay,
|
||||
but the audio-translation variants share the chat picker's
|
||||
transport and would 4xx, so they must be filtered. Pin both
|
||||
behaviours so a future regex tweak cannot collapse one into
|
||||
the other."""
|
||||
kept = _apply(
|
||||
"openai",
|
||||
[
|
||||
"gpt-realtime",
|
||||
"gpt-realtime-mini",
|
||||
"gpt-audio",
|
||||
"gpt-audio-mini",
|
||||
],
|
||||
)
|
||||
assert kept == [
|
||||
"gpt-realtime",
|
||||
"gpt-realtime-mini",
|
||||
"gpt-audio",
|
||||
"gpt-audio-mini",
|
||||
], kept
|
||||
"""gpt-audio is chat-capable (streams over /v1/chat/completions)
|
||||
and stays, but the audio-translation variants share the chat
|
||||
picker's transport and would 4xx, so they must be filtered. The
|
||||
new gpt-realtime* family and gpt-audio-mini are Realtime-only and
|
||||
are dropped by a sibling assertion."""
|
||||
kept = _apply("openai", ["gpt-audio"])
|
||||
assert kept == ["gpt-audio"], kept
|
||||
|
||||
dropped = _apply(
|
||||
"openai",
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue