unsloth/studio/backend/tests/test_codex_provider.py
Daniel Han f97a800d5b Studio: round 5e Codex hardening (4 follow-ups)
1. parallel_calls validator now clamps instead of 422-rejecting.
   The Pydantic schema was `int Field(ge=1, le=20)`, which was a
   regression from the pre-PR OpenAI-extra behaviour: a non-Codex
   client that sent the field with a legacy value like 0 (or a
   stray string from a misconfigured wrapper) now got a 422 even
   though the route silently ignores the field on every non-Codex
   provider. Replaced with a `field_validator(mode="before")` that
   coerces any input to the [1, 20] range, keeping the schema docs
   self-documenting while accepting legacy inputs.

2. Buffered Codex result with `final_response=None` no longer
   leaks `TurnResult(...)` Python object repr into the chat. The
   upstream SDK documents `TurnResult.final_response` as nullable
   for turns that perform tool work without producing a final
   assistant message; the previous `... or str(result)` fallback
   would render the repr as visible assistant text. New
   `_buffered_result_text` helper returns the empty string in that
   case so the stream finishes cleanly with no extra content
   chunk. Same fix applied to `_run_codex_synthesis`.

3. Device-login SSE no longer forwards arbitrary subprocess output
   to the browser. The previous code yielded every CLI line under
   `{type:"log"}`, which on a shimmed binary could leak refresh
   tokens, auth JSON, or local config paths into the authenticated
   stream. Filtered to a known-safe vocabulary ("Welcome to
   Codex", "Initializing", "Successfully logged in", etc.).
   `device_url` and `device_code` events still fire as before.

4. CodexLoginButton no longer calls `window.open` from inside an
   awaited SSE handler. Browser popup blockers (Firefox, Safari,
   Chrome strict) silently block popups triggered outside a fresh
   user gesture, so the auto-open was unreliable. Replaced with a
   prominent "Open verification page" button styled as an anchor;
   the click handler is a real user gesture and is never blocked.
   The URL string is still shown below the button for copy/paste.

Tests: 46 cases total (was 43). New regressions cover the
parallel_calls clamp path on three garbage inputs, the buffered
TurnResult-with-None-final repr leak guard, and the device-login
log filter (asserts refresh tokens / auth.json paths are dropped
while known-safe progress lines pass through).
2026-05-24 16:59:46 +00:00

1605 lines
60 KiB
Python

# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
"""
Unit tests for the Codex SDK provider integration.
Covers:
* Availability probe: codex missing, codex present but logged out,
codex present + logged in, plus the empty-output / non-zero rc
edge cases the CLI has shipped over time.
* ``stream_codex`` event translation: a fake codex_app_server module
is dropped into ``sys.modules`` so the production import path runs
without the real SDK installed. Verifies an OpenAI Chat Completions
shape (content chunk, stop chunk, [DONE]).
* Parallel-calls fan-out: ``parallel_calls > 1`` spawns N async tasks
and emits ``codex_tab_open`` / ``codex_tab_chunk`` / ``codex_tab_close``
events plus a final ``codex_gather`` synthesis event.
* Request validator: ``parallel_calls`` is silently clamped to
[1, 20] by a Pydantic field validator (not by ``ge=1, le=20``) so
non-Codex clients that send legacy values like ``0`` continue to
be accepted instead of getting a 422.
"""
from __future__ import annotations
import asyncio
import json
import os
import sys
import types
from typing import Any
import pytest
_backend = os.path.join(os.path.dirname(__file__), "..")
if _backend not in sys.path:
sys.path.insert(0, _backend)
# Resolved relative to this file so the source-inspection tests work in any
# checkout location (CI, dev machines, the review worker, etc.).
_BACKEND_DIR = os.path.realpath(os.path.join(os.path.dirname(__file__), ".."))
def _backend_file(rel: str) -> str:
"""Return an absolute path inside the backend tree, regardless of cwd."""
return os.path.join(_BACKEND_DIR, rel)
# ── Helpers ─────────────────────────────────────────────────────────
class _FakeStream:
"""Async iterator that yields predetermined string text events.
The Codex SDK's ``thread.run_streaming`` returns an async iterable
of events. ``_stream_thread_run`` converts those into raw text via
``_coerce_text``; passing in plain strings exercises the simplest
coercion path.
"""
def __init__(self, chunks: list[str]):
self._chunks = list(chunks)
self._i = 0
def __aiter__(self):
return self
async def __anext__(self):
if self._i >= len(self._chunks):
raise StopAsyncIteration
text = self._chunks[self._i]
self._i += 1
return text
class _FakeThread:
def __init__(self, chunks: list[str], final: str | None = None):
self._chunks = chunks
self._final = final if final is not None else "".join(chunks)
def run_streaming(self, prompt: str):
# ``run_streaming`` may return either an async iterable or a
# coroutine that resolves to one; cover the direct-return
# shape here, the coroutine shape is covered in a separate
# test below.
return _FakeStream(self._chunks)
async def run(self, prompt: str):
return self._final
class _FakeAsyncCodex:
"""Async-context-manager facade matching codex_app_server.AsyncCodex."""
def __init__(
self,
chunks: list[str] | None = None,
final: str | None = None,
raise_on_start: Exception | None = None,
):
self._chunks = chunks or []
self._final = final
self._raise = raise_on_start
async def __aenter__(self):
return self
async def __aexit__(self, exc_type, exc, tb):
return False
async def thread_start(self, **kwargs):
if self._raise is not None:
raise self._raise
return _FakeThread(self._chunks, self._final)
def _install_fake_codex_sdk(monkeypatch, async_codex_cls):
"""Drop a fake ``codex_app_server`` module into sys.modules so the
production lazy-import path picks it up without the real SDK
being installed.
"""
fake_mod = types.ModuleType("codex_app_server")
fake_mod.AsyncCodex = async_codex_cls # type: ignore[attr-defined]
monkeypatch.setitem(sys.modules, "codex_app_server", fake_mod)
# importlib.util.find_spec walks finders, not sys.modules; patch
# it directly so the lazy-import gate accepts the fake.
import importlib.util as _iu
real_find_spec = _iu.find_spec
def _shim(name: str, *args, **kwargs):
if name == "codex_app_server":
return types.SimpleNamespace()
return real_find_spec(name, *args, **kwargs)
monkeypatch.setattr("importlib.util.find_spec", _shim)
# ── Availability probe ─────────────────────────────────────────────
class TestCodexAvailability:
def test_absent_when_cli_missing(self, monkeypatch):
from core.inference import codex_availability as ca
monkeypatch.setattr(ca, "_which_codex", lambda: None)
monkeypatch.setattr(ca, "_sdk_importable", lambda: False)
payload = asyncio.run(ca.probe_codex_availability())
assert payload["installed"] is False
assert payload["cli_path"] is None
assert payload["sdk_importable"] is False
# supported_models is a sensible default even when nothing is
# installed so the picker has something to render IF the user
# forces the entry on a future status flip.
assert isinstance(payload["supported_models"], list)
assert len(payload["supported_models"]) > 0
def test_present_but_sdk_missing(self, monkeypatch):
from core.inference import codex_availability as ca
monkeypatch.setattr(ca, "_which_codex", lambda: "/usr/local/bin/codex")
monkeypatch.setattr(ca, "_sdk_importable", lambda: False)
async def fake_version():
return "codex-cli 0.133.0"
async def fake_logged_in():
return True
monkeypatch.setattr(ca, "_detect_version", fake_version)
monkeypatch.setattr(ca, "_detect_logged_in", fake_logged_in)
payload = asyncio.run(ca.probe_codex_availability())
# The SDK is what backs `AsyncCodex(...)`, so installed=False
# when the SDK is missing -- even if a standalone CLI is on
# PATH there is no way for Studio to drive it without the
# Python bindings.
assert payload["installed"] is False
assert payload["cli_path"] == "/usr/local/bin/codex"
assert payload["sdk_importable"] is False
assert payload["version"] == "codex-cli 0.133.0"
def test_present_and_logged_out(self, monkeypatch):
from core.inference import codex_availability as ca
monkeypatch.setattr(ca, "_which_codex", lambda: "/usr/local/bin/codex")
monkeypatch.setattr(ca, "_sdk_importable", lambda: True)
async def fake_version():
return "codex-cli 0.133.0"
async def fake_logged_in():
return False
monkeypatch.setattr(ca, "_detect_version", fake_version)
monkeypatch.setattr(ca, "_detect_logged_in", fake_logged_in)
payload = asyncio.run(ca.probe_codex_availability())
assert payload["installed"] is True
assert payload["logged_in"] is False
assert payload["version"] == "codex-cli 0.133.0"
def test_present_and_logged_in(self, monkeypatch):
from core.inference import codex_availability as ca
monkeypatch.setattr(ca, "_which_codex", lambda: "/usr/local/bin/codex")
monkeypatch.setattr(ca, "_sdk_importable", lambda: True)
async def fake_version():
return "codex-cli 0.133.0"
async def fake_logged_in():
return True
monkeypatch.setattr(ca, "_detect_version", fake_version)
monkeypatch.setattr(ca, "_detect_logged_in", fake_logged_in)
payload = asyncio.run(ca.probe_codex_availability())
assert payload["installed"] is True
assert payload["logged_in"] is True
# ── _stream_codex translation ──────────────────────────────────────
def _collect_stream(gen) -> list[str]:
async def run():
out: list[str] = []
async for line in gen:
out.append(line)
return out
return asyncio.run(run())
def _parse_sse_chunks(lines: list[str]) -> list[dict[str, Any]]:
"""Decode SSE ``data: {...}`` lines into the chunk dicts. Skips the
sentinel ``data: [DONE]`` line and anything that isn't valid JSON.
"""
out: list[dict[str, Any]] = []
for raw in lines:
if not raw.startswith("data:"):
continue
body = raw[len("data:") :].strip()
if not body or body == "[DONE]":
continue
try:
out.append(json.loads(body))
except json.JSONDecodeError:
continue
return out
class TestStreamCodexSingle:
def test_streaming_chunks_translate_into_openai_shape(self, monkeypatch):
_install_fake_codex_sdk(
monkeypatch,
lambda: _FakeAsyncCodex(chunks = ["Hello", ", ", "world"]),
)
from core.inference.codex_provider import stream_codex
lines = _collect_stream(
stream_codex(
messages = [{"role": "user", "content": "Say hello in 3 chunks."}],
model = "gpt-5.4",
)
)
chunks = _parse_sse_chunks(lines)
# Three content deltas + one usage chunk + one stop chunk.
content_chunks = [
c
for c in chunks
if c.get("choices")
and isinstance(c["choices"], list)
and c["choices"]
and c["choices"][0].get("delta", {}).get("content")
]
assert [c["choices"][0]["delta"]["content"] for c in content_chunks] == [
"Hello",
", ",
"world",
]
# Usage chunk (OpenAI include_usage shape) is a choices=[] entry
# with a populated usage block.
usage_chunks = [c for c in chunks if c.get("choices") == [] and c.get("usage")]
assert len(usage_chunks) == 1
usage = usage_chunks[0]["usage"]
assert usage["prompt_tokens"] > 0
assert usage["completion_tokens"] >= 0
# Final stop chunk with finish_reason=stop.
stop_chunks = [
c
for c in chunks
if c.get("choices")
and c["choices"]
and c["choices"][0].get("finish_reason") == "stop"
]
assert len(stop_chunks) == 1
# And the trailing [DONE] sentinel.
assert any(line.strip() == "data: [DONE]" for line in lines)
def test_empty_user_prompt_emits_helpful_message(self, monkeypatch):
_install_fake_codex_sdk(monkeypatch, lambda: _FakeAsyncCodex(chunks = []))
from core.inference.codex_provider import stream_codex
lines = _collect_stream(
stream_codex(
messages = [{"role": "system", "content": "you are helpful"}],
model = "gpt-5.4",
)
)
text = "\n".join(lines)
assert "no user prompt" in text.lower()
class TestStreamCodexParallel:
def test_parallel_calls_spawn_tabs_and_synthesise(self, monkeypatch):
# The fake SDK returns the same canned chunks for every spawned
# AsyncCodex instance; we just need to verify the orchestrator
# emits N tab_open events, per-tab chunk events keyed by
# tab_id, and a final codex_gather summary event.
_install_fake_codex_sdk(
monkeypatch,
lambda: _FakeAsyncCodex(
chunks = ["alpha"],
final = "synthesised answer",
),
)
from core.inference.codex_provider import stream_codex
n = 3
lines = _collect_stream(
stream_codex(
messages = [{"role": "user", "content": "Test"}],
model = "gpt-5.4",
parallel_calls = n,
)
)
chunks = _parse_sse_chunks(lines)
tool_events = [c["_toolEvent"] for c in chunks if "_toolEvent" in c]
tab_opens = [e for e in tool_events if e.get("type") == "codex_tab_open"]
tab_chunks = [e for e in tool_events if e.get("type") == "codex_tab_chunk"]
tab_closes = [e for e in tool_events if e.get("type") == "codex_tab_close"]
gather = [e for e in tool_events if e.get("type") == "codex_gather"]
# Each tab opens once -- the N tabs are pre-emitted so the
# UI can paint the strip before content arrives.
assert len(tab_opens) == n
assert sorted(e["tab_id"] for e in tab_opens) == list(range(1, n + 1))
# Per-tab chunks may interleave in any order but every tab id
# must produce at least one chunk before its close event.
seen_tabs = {e["tab_id"] for e in tab_chunks}
assert seen_tabs == set(range(1, n + 1))
# Each tab emits exactly one close marker.
assert sorted(e["tab_id"] for e in tab_closes) == list(range(1, n + 1))
# Exactly one synthesis event with the unified summary.
assert len(gather) == 1
assert gather[0]["tab_count"] == n
# The summary text comes from the final synthesis Codex call;
# our fake returns "synthesised answer" via .run().
assert "synth" in gather[0]["summary"].lower()
def test_parallel_calls_clamped_to_maximum(self, monkeypatch):
"""Passing parallel_calls=500 must NOT spawn 500 tasks; the
clamp at MAX_PARALLEL_CALLS keeps the local CLI safe.
"""
from core.inference import codex_provider as cp
_install_fake_codex_sdk(
monkeypatch,
lambda: _FakeAsyncCodex(chunks = ["x"], final = "synth"),
)
lines = _collect_stream(
cp.stream_codex(
messages = [{"role": "user", "content": "x"}],
model = "gpt-5.4",
parallel_calls = 500,
)
)
chunks = _parse_sse_chunks(lines)
tab_opens = [
c["_toolEvent"]
for c in chunks
if c.get("_toolEvent", {}).get("type") == "codex_tab_open"
]
assert len(tab_opens) == cp.MAX_PARALLEL_CALLS
def test_parallel_calls_one_takes_single_path(self, monkeypatch):
"""parallel_calls=1 must not emit any tab tool-events -- it's the
regular single-call shape.
"""
_install_fake_codex_sdk(
monkeypatch,
lambda: _FakeAsyncCodex(chunks = ["one"]),
)
from core.inference.codex_provider import stream_codex
lines = _collect_stream(
stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.4",
parallel_calls = 1,
)
)
chunks = _parse_sse_chunks(lines)
tool_events = [c.get("_toolEvent") for c in chunks if c.get("_toolEvent")]
for event in tool_events:
assert not (event.get("type") or "").startswith("codex_tab")
assert event.get("type") != "codex_gather"
# ── Request validator ──────────────────────────────────────────────
class TestParallelCallsValidator:
def test_request_accepts_valid_range(self):
from models.inference import ChatCompletionRequest
for n in (1, 5, 10, 20):
req = ChatCompletionRequest(
model = "gpt-5.4",
messages = [{"role": "user", "content": "hi"}],
parallel_calls = n,
)
assert req.parallel_calls == n
def test_request_clamps_below_one(self):
"""Pre-PR clients sometimes sent `parallel_calls=0` as a stray
OpenAI extra and the request was silently accepted; rejecting
with 422 would regress that. The validator now clamps to 1.
"""
from models.inference import ChatCompletionRequest
for n in (0, -1, -100):
req = ChatCompletionRequest(
model = "gpt-5.4",
messages = [{"role": "user", "content": "hi"}],
parallel_calls = n,
)
assert req.parallel_calls == 1, f"clamp failed for {n}"
def test_request_clamps_above_twenty(self):
"""A runaway value (1000, etc.) is clamped to the 20 cap so it
cannot saturate the local CLI even when the client misbehaves.
"""
from models.inference import ChatCompletionRequest
for n in (21, 100, 1000):
req = ChatCompletionRequest(
model = "gpt-5.4",
messages = [{"role": "user", "content": "hi"}],
parallel_calls = n,
)
assert req.parallel_calls == 20, f"clamp failed for {n}"
def test_request_coerces_garbage_to_one(self):
"""Strings / floats / None coerce to 1 instead of 422 so a
legacy or misconfigured client cannot break chat for everyone."""
from models.inference import ChatCompletionRequest
for value in (None, "garbage", float("nan")):
req = ChatCompletionRequest(
model = "gpt-5.4",
messages = [{"role": "user", "content": "hi"}],
parallel_calls = value,
)
assert req.parallel_calls == 1
def test_request_default_is_one(self):
"""Default = 1 so the field matches the single-call code path
and the schema documentation. Non-codex providers ignore the
field regardless of its value, so backwards compat is
preserved.
"""
from models.inference import ChatCompletionRequest
req = ChatCompletionRequest(
model = "gpt-5.4",
messages = [{"role": "user", "content": "hi"}],
)
assert req.parallel_calls == 1
# ── Codex unavailable surfacing ────────────────────────────────────
class TestCodexUnavailable:
def test_missing_sdk_raises_typed_error(self, monkeypatch):
# Force find_spec to return None so the lazy import fails.
# The provider probes both the canonical upstream name
# ``openai_codex`` and the legacy alias ``codex_app_server``,
# so we have to suppress both for the import to fail.
import importlib.util as _iu
real = _iu.find_spec
_SDK_NAMES = {"openai_codex", "codex_app_server"}
def _shim(name, *args, **kwargs):
if name in _SDK_NAMES:
return None
return real(name, *args, **kwargs)
monkeypatch.setattr("importlib.util.find_spec", _shim)
# Also drop any cached fakes from prior tests.
for _name in _SDK_NAMES:
monkeypatch.delitem(sys.modules, _name, raising = False)
from core.inference.codex_provider import (
CodexUnavailableError,
stream_codex,
)
with pytest.raises(CodexUnavailableError):
asyncio.run(
_consume_first(
stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.4",
)
)
)
class TestCodexHardenedRegressions:
"""Tests covering the post-review hardening pass.
Each test pins a specific regression: the wrong subcommand
(``codex auth login`` → ``codex login``), the wrong SDK package
name (``codex_app_server`` → ``openai_codex`` with legacy alias),
the ``not logged in`` substring footgun, the ANSI-wrapped device
URL, and the fan-out cancellation contract.
"""
def test_sdk_probes_openai_codex_first(self, monkeypatch):
"""The canonical upstream name must be tried before the alias."""
import importlib.util as _iu
real = _iu.find_spec
calls: list[str] = []
def _shim(name, *args, **kwargs):
if name in ("openai_codex", "codex_app_server"):
calls.append(name)
return None
return real(name, *args, **kwargs)
monkeypatch.setattr("importlib.util.find_spec", _shim)
from core.inference.codex_availability import _sdk_importable
assert _sdk_importable() is False
assert (
calls and calls[0] == "openai_codex"
), f"availability probe must check openai_codex first; saw {calls}"
def test_login_status_uses_login_subcommand(self):
"""Upstream is `codex login status`, NOT `codex auth status`."""
src = _backend_file("core/inference/codex_availability.py")
text = open(src).read()
assert (
'"auth", "status"' not in text
), "_detect_logged_in must use `codex login status`, not `codex auth status`"
assert '"login", "status"' in text
def test_device_login_uses_login_subcommand(self):
src = _backend_file("core/inference/codex_provider.py")
text = open(src).read()
assert (
'"auth", "login", "--device-auth"' not in text
), "stream_codex_device_login must use `codex login --device-auth`"
assert '"login", "--device-auth"' in text
def test_not_logged_in_not_misparsed_as_logged_in(self):
"""The substring "logged in" inside "not logged in" must not
flip the detection to True."""
import asyncio
from core.inference import codex_availability as av
async def _fake_run_cli(args, **kw):
return (0, "Not logged in. Run `codex login` to authenticate.", "")
orig = av._run_cli
av._run_cli = _fake_run_cli # type: ignore[assignment]
try:
result = asyncio.run(av._detect_logged_in())
assert result is False, "'Not logged in' was misparsed as logged_in=True"
finally:
av._run_cli = orig # type: ignore[assignment]
def test_logged_in_is_detected(self):
import asyncio
from core.inference import codex_availability as av
async def _fake_run_cli(args, **kw):
return (0, "Logged in using ChatGPT", "")
orig = av._run_cli
av._run_cli = _fake_run_cli # type: ignore[assignment]
try:
result = asyncio.run(av._detect_logged_in())
assert result is True
finally:
av._run_cli = orig # type: ignore[assignment]
def test_multi_turn_prompt_includes_prior_turns(self):
"""The Codex prompt MUST contain prior assistant turns."""
from core.inference.codex_provider import _last_user_prompt
msgs = [
{"role": "user", "content": "what is the capital of france?"},
{"role": "assistant", "content": "Paris."},
{"role": "user", "content": "and germany?"},
]
prompt = _last_user_prompt(msgs)
assert "and germany?" in prompt
assert (
"Paris" in prompt
), f"PRIOR ASSISTANT TURN DROPPED — multi-turn broken. Prompt:\n{prompt}"
assert "capital of france" in prompt.lower()
def test_single_turn_prompt_unchanged(self):
"""Single-turn case must not get the User:/Assistant: framing."""
from core.inference.codex_provider import _last_user_prompt
prompt = _last_user_prompt([{"role": "user", "content": "hi"}])
assert prompt == "hi"
def test_default_models_no_o3(self):
"""The Codex registry must not advertise `o3` (not in upstream)."""
from core.inference.providers import PROVIDER_REGISTRY
codex = PROVIDER_REGISTRY["codex"]
assert (
"o3" not in codex["default_models"]
), "o3 is not a Codex model; remove from default_models"
assert "gpt-5.5" in codex["default_models"]
def test_inference_route_no_raw_exc_leak(self):
"""SSE error frame must NOT echo str(exc) verbatim (CodeQL)."""
import re
src = _backend_file("routes/inference.py")
text = open(src).read()
bad = re.findall(r'f["\']Codex error:\s*\{exc\}["\']', text)
assert not bad, f"raw exception in SSE: {bad}"
def test_codex_route_no_raw_exc_leak(self):
"""codex.py SSE stream wrapping must also not leak str(exc)."""
import re
src = _backend_file("routes/codex.py")
text = open(src).read()
for line in text.splitlines():
ls = line.strip()
if ls.startswith("yield ") and re.search(r"\{exc\}|\{e\}", ls):
assert False, f"raw exception leaked: {ls}"
def test_parallel_tab_error_sanitised(self, monkeypatch):
"""A worker that raises with a path-leaking message must NOT
send that text to the client; the SSE codex_tab_error event
must carry a generic message + exception_type.
"""
fake = _FakeAsyncCodex(
raise_on_start = RuntimeError(
"secret /home/alice/.codex/config.json token=abc"
)
)
_install_fake_codex_sdk(monkeypatch, lambda: fake)
from core.inference.codex_provider import stream_codex
chunks: list[str] = []
async def _collect():
async for c in stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.5",
parallel_calls = 2,
):
chunks.append(c)
asyncio.run(_collect())
body = "".join(chunks)
assert (
"secret /home/alice" not in body
), "raw exception text leaked into codex_tab_error SSE frame"
assert "Codex tab failed" in body or "exception_type" in body
def test_codex_subprocess_env_scrubbed(self, monkeypatch):
"""The codex subprocess env must not include other-provider secrets.
OPENAI_API_KEY is intentionally excluded too: a shimmed `codex`
binary on PATH must not receive Studio's stored OpenAI provider
key. Users wire Codex auth via `codex login` or the
codex-specific CODEX_OPENAI_API_KEY override instead.
"""
from core.inference.codex_availability import _codex_subprocess_env
monkeypatch.setenv("HF_TOKEN", "hf_should_not_leak")
monkeypatch.setenv("GH_TOKEN", "gh_should_not_leak")
monkeypatch.setenv("WANDB_API_KEY", "wandb_should_not_leak")
monkeypatch.setenv("ANTHROPIC_API_KEY", "anthropic_should_not_leak")
monkeypatch.setenv("OPENAI_API_KEY", "openai_provider_key_not_for_codex")
monkeypatch.setenv("CODEX_OPENAI_API_KEY", "codex_specific_key")
monkeypatch.setenv("CODEX_HOME", "/custom/.codex")
monkeypatch.setenv("PATH", "/usr/bin")
env = _codex_subprocess_env()
for secret in (
"HF_TOKEN",
"GH_TOKEN",
"WANDB_API_KEY",
"ANTHROPIC_API_KEY",
# OPENAI_API_KEY belongs to the OpenAI provider, not Codex.
"OPENAI_API_KEY",
):
assert secret not in env, f"{secret} leaked into codex env"
# Codex-relevant keys must be preserved.
assert env.get("CODEX_OPENAI_API_KEY") == "codex_specific_key"
assert env.get("CODEX_HOME") == "/custom/.codex"
assert env.get("PATH") == "/usr/bin"
def test_partial_stream_failure_does_not_replay_turn(self, monkeypatch):
"""If turn.stream() fails after emitting some text, the buffered
run() fallback must NOT fire -- replaying would duplicate side
effects (file writes, shell commands).
"""
run_calls = {"n": 0}
class _PartialStreamTurn:
async def stream(self):
yield {"text": "partial output "}
raise RuntimeError("network glitch mid-stream")
class _ThreadPartialFail:
def turn(self, prompt):
return _PartialStreamTurn()
async def run(self, prompt):
run_calls["n"] += 1
return "REPLAYED -- BAD"
class _Async:
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
return _ThreadPartialFail()
_install_fake_codex_sdk(monkeypatch, _Async)
from core.inference.codex_provider import stream_codex
chunks: list[str] = []
async def _collect():
async for c in stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.5",
parallel_calls = 1,
):
chunks.append(c)
asyncio.run(_collect())
assert run_calls["n"] == 0, (
"buffered run() fired after partial stream emission -- "
"would replay side effects"
)
body = "".join(chunks)
assert "partial output" in body
assert "REPLAYED" not in body
def test_not_signed_in_wording_also_handled(self):
"""`Not signed in` (alternative localisation) must also be
treated as logged-out, not as positive match.
"""
import asyncio
from core.inference import codex_availability as av
async def _fake_run_cli(args, **kw):
return (0, "Not signed in.", "")
orig = av._run_cli
av._run_cli = _fake_run_cli # type: ignore[assignment]
try:
assert asyncio.run(av._detect_logged_in()) is False
finally:
av._run_cli = orig # type: ignore[assignment]
def test_device_url_accepts_generic_verification_url(self):
"""The login parser must accept upstream's chatgpt.com/activate
URL as well as the canonical /codex/device shape.
"""
import re
src = _backend_file("core/inference/codex_provider.py")
text = open(src).read()
# Find the url_re pattern literal and compile it.
m = re.search(r"url_re\s*=\s*re\.compile\(\s*\n?\s*r\"([^\"]+)\"", text)
assert m, "url_re definition not found"
pattern = re.compile(m.group(1), re.IGNORECASE)
# Upstream device URLs we expect to match.
for u in (
"https://auth.openai.com/codex/device",
"https://chatgpt.com/activate",
"https://auth.openai.com/device/verify?code=ABCD",
):
assert pattern.search(u), f"device URL regex missed: {u}"
def test_synthesis_call_forwards_system_prompt(self, monkeypatch):
"""`_run_codex_synthesis` must pass the system prompt so a
fan-out style instruction ("Always answer in Spanish") survives
the unification step.
"""
seen_kwargs: list[dict] = []
seen_prompts: list[str] = []
class _SynThread:
async def run(self, prompt):
seen_prompts.append(prompt)
return "synth ok"
def turn(self, prompt):
# Force buffered path via no `stream` attr.
class _T:
pass
return _T()
class _Async:
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
seen_kwargs.append(kw)
return _SynThread()
_install_fake_codex_sdk(monkeypatch, _Async)
from core.inference.codex_provider import _run_codex_synthesis
out = asyncio.run(
_run_codex_synthesis(
model = "gpt-5.5",
system = "Always answer in Spanish.",
prompt = "What is the capital of France?",
tab_outputs = ["Paris", "Paris."],
)
)
# The upstream openai_codex SDK uses `base_instructions` for the
# system prompt; the legacy alias accepts `system`; the last-resort
# fallback inlines the system text into the user prompt. Accept
# any of those paths.
system_seen = (
any("Spanish" in (kw.get("base_instructions") or "") for kw in seen_kwargs)
or any("Spanish" in (kw.get("system") or "") for kw in seen_kwargs)
or any("Always answer in Spanish" in p for p in seen_prompts)
)
assert system_seen, (
f"system prompt dropped in synthesis. kwargs={seen_kwargs} "
f"prompts={seen_prompts}"
)
# And the synthesis still returned the model's text.
assert "synth" in out.lower()
def test_sdk_env_scrubbed_via_appserverconfig(self, monkeypatch):
"""The SDK construction path must wire AppServerConfig(env=...)
when the SDK exposes it, so HF_TOKEN / GH_TOKEN are not leaked
to the codex app-server subprocess.
"""
monkeypatch.setenv("HF_TOKEN", "should_be_scrubbed")
monkeypatch.setenv("GH_TOKEN", "should_be_scrubbed")
# OPENAI_API_KEY is now ALSO scrubbed -- it belongs to the
# OpenAI provider, not Codex. CODEX_OPENAI_API_KEY is the
# codex-specific override that survives.
monkeypatch.setenv("OPENAI_API_KEY", "openai_provider_key_not_for_codex")
monkeypatch.setenv("CODEX_OPENAI_API_KEY", "codex_specific_key")
seen_configs: list[Any] = []
class _FakeAppServerConfig:
def __init__(self, env = None, **kw):
self.env = env or {}
class _Async:
def __init__(self, config = None):
seen_configs.append(config)
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
return _FakeThread(chunks = ["ok"])
# Inject a fake openai_codex module exposing AppServerConfig.
import importlib.util as _iu
import types as _types
fake_mod = _types.ModuleType("openai_codex")
fake_mod.AsyncCodex = _Async # type: ignore[attr-defined]
fake_mod.AppServerConfig = _FakeAppServerConfig # type: ignore[attr-defined]
monkeypatch.setitem(sys.modules, "openai_codex", fake_mod)
real_find_spec = _iu.find_spec
monkeypatch.setattr(
"importlib.util.find_spec",
lambda n, *a, **kw: (
_types.SimpleNamespace()
if n in ("openai_codex", "codex_app_server")
else real_find_spec(n, *a, **kw)
),
)
from core.inference.codex_provider import stream_codex
asyncio.run(
_consume_first(
stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.5",
parallel_calls = 1,
)
)
)
assert seen_configs, "AsyncCodex was never instantiated"
cfg = seen_configs[0]
assert cfg is not None, "AppServerConfig was not passed to AsyncCodex"
assert (
"HF_TOKEN" in cfg.env and cfg.env["HF_TOKEN"] == ""
), "HF_TOKEN not overridden to empty in SDK env"
assert "GH_TOKEN" in cfg.env and cfg.env["GH_TOKEN"] == ""
# OPENAI_API_KEY is intentionally overridden to empty in the
# SDK env so the app-server cannot use it as a Codex credential
# by accident. The OpenAI provider still reads its own key from
# Studio's storage; nothing in this path needs the env var.
assert cfg.env.get("OPENAI_API_KEY") == ""
# CODEX_OPENAI_API_KEY is the Codex-specific override and must
# survive untouched so users can wire that key into Codex.
assert "CODEX_OPENAI_API_KEY" not in cfg.env
def test_thread_turn_stream_path_taken(self, monkeypatch):
"""The canonical openai_codex API uses thread.turn(prompt).stream();
the provider must prefer that over the legacy run_streaming hook.
"""
events_seen = {"turn_called": False, "run_streaming_called": False}
class _TurnEvent:
def __init__(self, txt):
self.payload = {"text": txt}
class _TurnHandle:
def __init__(self, prompt):
self.prompt = prompt
async def stream(self):
yield _TurnEvent("hello ")
yield _TurnEvent("from turn.stream")
class _ThreadWithTurn:
def turn(self, prompt):
events_seen["turn_called"] = True
return _TurnHandle(prompt)
def run_streaming(self, prompt):
events_seen["run_streaming_called"] = True
raise AssertionError("should not be called when turn().stream() works")
async def run(self, prompt):
raise AssertionError("should not fall through to buffered run()")
class _Async:
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
return _ThreadWithTurn()
_install_fake_codex_sdk(monkeypatch, _Async)
from core.inference.codex_provider import stream_codex
chunks: list[str] = []
async def _collect():
async for c in stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.5",
parallel_calls = 1,
):
chunks.append(c)
asyncio.run(_collect())
assert events_seen["turn_called"], "thread.turn() never called"
assert not events_seen["run_streaming_called"]
body = "".join(chunks)
# Each text chunk wraps in its own SSE delta, so check both pieces.
assert '"content": "hello "' in body
assert '"content": "from turn.stream"' in body
def test_installed_true_when_sdk_only(self, monkeypatch):
"""SDK alone is sufficient: openai-codex-cli-bin ships the
runtime that backs `AsyncCodex(...)`, so the picker must be
shown even when no standalone `codex` lives on PATH.
"""
from core.inference import codex_availability as ca
monkeypatch.setattr(ca, "_which_codex", lambda: None)
monkeypatch.setattr(ca, "_sdk_importable", lambda: True)
payload = asyncio.run(ca.probe_codex_availability())
assert payload["installed"] is True
assert payload["cli_path"] is None
assert payload["sdk_importable"] is True
# logged_in stays False because the version/login probes only
# run when a CLI is present (they shell out to it). That is the
# expected behaviour, not a bug.
assert payload["logged_in"] is False
def test_base_instructions_kwarg_preferred(self, monkeypatch):
"""The upstream openai_codex SDK uses `base_instructions` for
the system prompt. The provider must try that name first; only
if the SDK rejects it should it fall back to `system`.
"""
seen_kwargs: list[dict] = []
class _Async:
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
seen_kwargs.append(dict(kw))
return _FakeThread(chunks = ["ok"])
_install_fake_codex_sdk(monkeypatch, _Async)
from core.inference.codex_provider import stream_codex
async def _collect():
async for _ in stream_codex(
messages = [
{"role": "system", "content": "You are helpful."},
{"role": "user", "content": "hi"},
],
model = "gpt-5.5",
parallel_calls = 1,
):
pass
asyncio.run(_collect())
# The first (and only, since this fake accepts any kwargs)
# call must use base_instructions, not the legacy `system`.
assert seen_kwargs, "thread_start was never called"
assert (
"base_instructions" in seen_kwargs[0]
), f"upstream-canonical kwarg not used: {seen_kwargs[0]}"
assert seen_kwargs[0]["base_instructions"] == "You are helpful."
assert (
"system" not in seen_kwargs[0]
), "legacy `system` kwarg was sent even though base_instructions worked"
def test_base_instructions_falls_back_to_system(self, monkeypatch):
"""When the SDK rejects `base_instructions` with TypeError the
helper must retry with the legacy `system` kwarg before giving
up and inlining the system text in the prompt.
"""
call_log: list[dict] = []
class _StrictSDK:
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
call_log.append(dict(kw))
if "base_instructions" in kw:
raise TypeError(
"thread_start() got an unexpected keyword 'base_instructions'"
)
return _FakeThread(chunks = ["ok"])
_install_fake_codex_sdk(monkeypatch, _StrictSDK)
from core.inference.codex_provider import stream_codex
async def _collect():
async for _ in stream_codex(
messages = [
{"role": "system", "content": "You are helpful."},
{"role": "user", "content": "hi"},
],
model = "gpt-5.5",
parallel_calls = 1,
):
pass
asyncio.run(_collect())
assert len(call_log) >= 2, "fallback to `system` kwarg never tried"
assert "base_instructions" in call_log[0]
assert "system" in call_log[1] and call_log[1]["system"] == "You are helpful."
def test_scrubbed_env_wrapper_strips_secrets_before_construction(self, monkeypatch):
"""When AppServerConfig is missing the fail-closed wrapper must
remove secret env vars BEFORE the SDK constructor runs (the
SDK starts its app-server with `env = os.environ.copy()`).
"""
observed_env_during_init: dict[str, str | None] = {}
class _NoConfigAsync:
def __init__(self):
# Capture the environment exactly as the SDK would see
# it at construction time.
observed_env_during_init["HF_TOKEN"] = os.environ.get("HF_TOKEN")
observed_env_during_init["GH_TOKEN"] = os.environ.get("GH_TOKEN")
observed_env_during_init["WANDB_API_KEY"] = os.environ.get(
"WANDB_API_KEY"
)
observed_env_during_init["PATH"] = os.environ.get("PATH")
observed_env_during_init["CODEX_HOME"] = os.environ.get("CODEX_HOME")
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
return _FakeThread(chunks = ["ok"])
monkeypatch.setenv("HF_TOKEN", "should_be_gone")
monkeypatch.setenv("GH_TOKEN", "should_be_gone")
monkeypatch.setenv("WANDB_API_KEY", "should_be_gone")
monkeypatch.setenv("PATH", "/usr/bin")
monkeypatch.setenv("CODEX_HOME", "/home/u/.codex")
# No AppServerConfig in the fake module -- forces the wrapper path.
_install_fake_codex_sdk(monkeypatch, _NoConfigAsync)
from core.inference.codex_provider import stream_codex
async def _collect():
async for _ in stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.5",
parallel_calls = 1,
):
pass
asyncio.run(_collect())
# Secrets must have been removed from os.environ BEFORE the
# SDK constructor captured the env.
assert (
observed_env_during_init["HF_TOKEN"] is None
), "HF_TOKEN visible to SDK constructor -- env scrub failed"
assert observed_env_during_init["GH_TOKEN"] is None
assert observed_env_during_init["WANDB_API_KEY"] is None
# Safe-listed keys must survive.
assert observed_env_during_init["PATH"] == "/usr/bin"
assert observed_env_during_init["CODEX_HOME"] == "/home/u/.codex"
# And the wrapper must restore them after exit.
assert os.environ.get("HF_TOKEN") == "should_be_gone"
assert os.environ.get("GH_TOKEN") == "should_be_gone"
def test_coerce_text_drops_non_answer_event_types(self):
"""Tool / command / plan deltas have their own `delta` fields
that must NOT be rendered as assistant text -- otherwise local
stdout, file paths, or tool-call arguments would leak into the
Chat Completions reply.
"""
from core.inference.codex_provider import _coerce_text
# Allowed answer-bearing event types contribute text.
assert _coerce_text({"type": "message.delta", "delta": "hello"}) == "hello"
assert _coerce_text({"type": "completed", "text": "done"}) == "done"
assert _coerce_text({"type": "text_delta", "delta": "x"}) == "x"
# Non-answer event types are silenced even when they expose a
# delta string that looks like prose.
for ev_type in (
"command.delta",
"command_output",
"file_write.delta",
"tool_call.delta",
"plan.update",
"exec.stdout",
"exec.stderr",
"patch.apply",
"thread.tool_call",
"agent_reasoning",
):
payload = {"type": ev_type, "delta": "this should NOT leak"}
assert _coerce_text(payload) == "", (
f"{ev_type} leaked text into assistant reply: "
f"{_coerce_text(payload)!r}"
)
# Plain strings and untyped dicts still pass through (legacy path).
assert _coerce_text("raw text") == "raw text"
assert _coerce_text({"text": "no type tag"}) == "no type tag"
def test_authenticated_yes_wording_is_detected(self):
"""An `Authenticated: Yes` line (a wording the CLI ships in
some locales / versions) must be parsed as logged-in.
"""
from core.inference import codex_availability as av
async def _fake_run_cli(args, **kw):
return (0, "Authenticated: Yes\nuser@example.com", "")
orig = av._run_cli
av._run_cli = _fake_run_cli # type: ignore[assignment]
try:
assert asyncio.run(av._detect_logged_in()) is True
finally:
av._run_cli = orig # type: ignore[assignment]
def test_codex_openai_api_key_overrides_openai_provider_key(self, monkeypatch):
"""Studio's `OPENAI_API_KEY` must NOT reach codex -- but the
codex-specific `CODEX_OPENAI_API_KEY` MUST be forwarded so
users can deliberately wire a key into Codex.
"""
from core.inference.codex_availability import _codex_subprocess_env
monkeypatch.setenv("OPENAI_API_KEY", "belongs_to_openai_provider")
monkeypatch.setenv("CODEX_OPENAI_API_KEY", "explicit_codex_key")
env = _codex_subprocess_env()
assert (
"OPENAI_API_KEY" not in env
), "OpenAI provider key leaked into codex subprocess env"
assert env.get("CODEX_OPENAI_API_KEY") == "explicit_codex_key"
def test_thread_start_uses_safe_approval_and_sandbox(self, monkeypatch):
"""When the SDK exposes ApprovalMode + SandboxMode, the
provider MUST pin approval to `deny_all` and sandbox to
`read_only`. The upstream SDK default
(`auto_review` approvals, unspecified sandbox) would let the
model auto-execute commands and write files on the server.
"""
seen_kwargs: list[dict] = []
class _Async:
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
seen_kwargs.append(dict(kw))
return _FakeThread(chunks = ["ok"])
# Drop a fake openai_codex with ApprovalMode + SandboxMode enums.
import importlib.util as _iu
fake_mod = types.ModuleType("openai_codex")
fake_mod.AsyncCodex = _Async # type: ignore[attr-defined]
fake_mod.ApprovalMode = types.SimpleNamespace( # type: ignore[attr-defined]
deny_all = "DENY_ALL_SENTINEL",
auto_review = "AUTO_REVIEW_SENTINEL",
)
fake_mod.SandboxMode = types.SimpleNamespace( # type: ignore[attr-defined]
read_only = "READ_ONLY_SENTINEL",
workspace_write = "WS_WRITE_SENTINEL",
danger_full_access = "DANGER_SENTINEL",
)
monkeypatch.setitem(sys.modules, "openai_codex", fake_mod)
real_find_spec = _iu.find_spec
monkeypatch.setattr(
"importlib.util.find_spec",
lambda n, *a, **kw: (
types.SimpleNamespace()
if n in ("openai_codex", "codex_app_server")
else real_find_spec(n, *a, **kw)
),
)
from core.inference.codex_provider import stream_codex
async def _collect():
async for _ in stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.5",
parallel_calls = 1,
):
pass
asyncio.run(_collect())
assert seen_kwargs, "thread_start never called"
kw = seen_kwargs[0]
assert (
kw.get("approval_mode") == "DENY_ALL_SENTINEL"
), f"approval_mode not pinned to deny_all: {kw}"
assert (
kw.get("sandbox") == "READ_ONLY_SENTINEL"
), f"sandbox not pinned to read_only: {kw}"
def test_thread_start_skips_safety_kwargs_on_old_sdk(self, monkeypatch):
"""If the installed SDK does not expose ApprovalMode or
SandboxMode (older rev / alias), thread_start must still run
-- failing closed would brick anyone on a pre-release build.
The provider logs a warning and proceeds without the kwargs.
"""
seen_kwargs: list[dict] = []
class _Async:
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
seen_kwargs.append(dict(kw))
return _FakeThread(chunks = ["ok"])
# Fake SDK without ApprovalMode / SandboxMode.
_install_fake_codex_sdk(monkeypatch, _Async)
from core.inference.codex_provider import stream_codex
async def _collect():
async for _ in stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.5",
parallel_calls = 1,
):
pass
asyncio.run(_collect())
assert seen_kwargs, "thread_start never called"
kw = seen_kwargs[0]
assert "approval_mode" not in kw, (
"should not pass an unknown approval_mode value on an "
"SDK that does not expose the enum"
)
assert "sandbox" not in kw
# Model still passed so the request is well-formed.
assert kw.get("model") == "gpt-5.5"
def test_device_login_log_filter_drops_unknown_lines(self, monkeypatch):
"""The login stream's `log` events must not forward arbitrary
subprocess output. Only an allow-list of known progress
strings reaches the browser; anything else (auth JSON,
tokens, paths, error tails) stays in backend logs.
"""
# Build a synthetic stdout stream with one safe line and one
# unsafe line, then drive the login generator against it.
from core.inference import codex_provider as cp
class _FakeStdout:
def __init__(self, lines: list[bytes]):
self._lines = list(lines)
async def readline(self) -> bytes:
if not self._lines:
return b""
return self._lines.pop(0)
class _FakeProc:
pid = 99999
returncode = None
stdout = _FakeStdout(
[
b"Welcome to Codex\n",
b"Open: https://auth.openai.com/codex/device\n",
b"Enter this one-time code: ABCD-EFGH\n",
b'{"refresh_token": "rt_LEAK_LEAK_LEAK"}\n',
b"/home/u/.codex/auth.json saved\n",
b"Successfully logged in\n",
],
)
async def wait(self):
self.returncode = 0
return 0
def kill(self):
self.returncode = -9
def terminate(self):
self.returncode = -15
async def _fake_create_subprocess_exec(*a, **kw):
return _FakeProc()
monkeypatch.setattr(
cp.asyncio, "create_subprocess_exec", _fake_create_subprocess_exec
)
events: list[dict] = []
async def _collect():
async for ev in cp.stream_codex_device_login():
events.append(ev)
asyncio.run(_collect())
log_lines = [ev.get("line", "") for ev in events if ev.get("type") == "log"]
joined = "\n".join(log_lines)
# Sensitive content must not have been forwarded.
assert "refresh_token" not in joined, f"token leaked: {joined!r}"
assert "rt_LEAK_LEAK_LEAK" not in joined
assert "auth.json" not in joined, f"local config path leaked: {joined!r}"
# The known-safe progress lines must be present so the UI can
# show the user what is happening.
assert any("Welcome to Codex" in line for line in log_lines)
assert any("Successfully logged in" in line for line in log_lines)
# device_url + device_code events must still fire.
url_events = [ev for ev in events if ev.get("type") == "device_url"]
code_events = [ev for ev in events if ev.get("type") == "device_code"]
assert url_events and url_events[0]["url"].endswith("/codex/device")
assert code_events and code_events[0]["code"] == "ABCD-EFGH"
def test_buffered_result_none_final_does_not_emit_repr(self, monkeypatch):
"""A buffered TurnResult whose final_response is None must NOT
send a Python object repr (``TurnResult(...)``) to the user.
Returning an empty content chunk is the right shape: the
stream still finishes with the usage + stop + [DONE] frames,
but no garbage assistant text appears.
"""
class _ResultNoFinal:
final_response = None # explicit None
def __repr__(self):
return "TurnResult(internal=should_not_leak)"
class _ThreadBuffered:
async def run(self, prompt):
return _ResultNoFinal()
class _Async:
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
return _ThreadBuffered()
_install_fake_codex_sdk(monkeypatch, _Async)
from core.inference.codex_provider import stream_codex
chunks: list[str] = []
async def _collect():
async for c in stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.5",
parallel_calls = 1,
):
chunks.append(c)
asyncio.run(_collect())
body = "".join(chunks)
assert (
"TurnResult" not in body
), f"Python object repr leaked to user content: {body!r}"
assert "should_not_leak" not in body
# Stream still terminated cleanly.
assert "[DONE]" in body
def test_empty_stream_falls_back_to_completed_agent_message(self, monkeypatch):
"""A successful turn that emits zero ``message.delta`` events
but DOES emit a final ``ItemCompletedNotification`` with an
agent message must surface that text. Without the fallback the
Chat Completions reply would be empty even though Codex
produced a complete answer.
"""
class _CompletedEvent:
payload = {
"type": "item.completed",
"item": {
"root": {
"type": "agentMessage",
"text": "final answer from completion",
},
},
}
class _Turn:
async def stream(self):
yield _CompletedEvent()
class _ThreadEmptyDeltas:
def turn(self, prompt):
return _Turn()
async def run(self, prompt):
raise AssertionError(
"must not fall through to buffered run() when "
"the stream completes successfully"
)
class _Async:
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
return _ThreadEmptyDeltas()
_install_fake_codex_sdk(monkeypatch, _Async)
from core.inference.codex_provider import stream_codex
chunks: list[str] = []
async def _collect():
async for c in stream_codex(
messages = [{"role": "user", "content": "hi"}],
model = "gpt-5.5",
parallel_calls = 1,
):
chunks.append(c)
asyncio.run(_collect())
body = "".join(chunks)
assert (
"final answer from completion" in body
), f"agent message text from completion event was dropped; body={body!r}"
def test_synthesis_also_pins_safety_kwargs(self, monkeypatch):
"""The synthesis turn that unifies parallel fan-out outputs
must use the same safety pins -- a fan-out tab could otherwise
sneak an unsafe approval into the final synthesis prompt.
"""
seen_kwargs: list[dict] = []
class _SynThread:
async def run(self, prompt):
return "synth ok"
class _Async:
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def thread_start(self, **kw):
seen_kwargs.append(dict(kw))
return _SynThread()
import importlib.util as _iu
fake_mod = types.ModuleType("openai_codex")
fake_mod.AsyncCodex = _Async # type: ignore[attr-defined]
fake_mod.ApprovalMode = types.SimpleNamespace( # type: ignore[attr-defined]
deny_all = "DENY_ALL_SENTINEL",
)
fake_mod.SandboxMode = types.SimpleNamespace( # type: ignore[attr-defined]
read_only = "READ_ONLY_SENTINEL",
)
monkeypatch.setitem(sys.modules, "openai_codex", fake_mod)
real_find_spec = _iu.find_spec
monkeypatch.setattr(
"importlib.util.find_spec",
lambda n, *a, **kw: (
types.SimpleNamespace()
if n in ("openai_codex", "codex_app_server")
else real_find_spec(n, *a, **kw)
),
)
from core.inference.codex_provider import _run_codex_synthesis
asyncio.run(
_run_codex_synthesis(
model = "gpt-5.5",
system = "Always answer in Spanish.",
prompt = "What is the capital of France?",
tab_outputs = ["Paris", "Paris."],
)
)
assert seen_kwargs, "synthesis thread_start never called"
kw = seen_kwargs[0]
assert kw.get("approval_mode") == "DENY_ALL_SENTINEL"
assert kw.get("sandbox") == "READ_ONLY_SENTINEL"
async def _consume_first(gen):
"""Drive an async generator until it raises or yields its first
value. Used to surface lazy-import errors that fire on the first
SDK touch -- otherwise the generator would swallow them on
``__aiter__`` and the test couldn't see them.
"""
async for _ in gen:
return