# SPDX-License-Identifier: AGPL-3.0-only # Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 """ Unit tests for the Codex SDK provider integration. Covers: * Availability probe: codex missing, codex present but logged out, codex present + logged in, plus the empty-output / non-zero rc edge cases the CLI has shipped over time. * ``stream_codex`` event translation: a fake codex_app_server module is dropped into ``sys.modules`` so the production import path runs without the real SDK installed. Verifies an OpenAI Chat Completions shape (content chunk, stop chunk, [DONE]). * Parallel-calls fan-out: ``parallel_calls > 1`` spawns N async tasks and emits ``codex_tab_open`` / ``codex_tab_chunk`` / ``codex_tab_close`` events plus a final ``codex_gather`` synthesis event. * Request validator: ``parallel_calls`` is silently clamped to [1, 20] by a Pydantic field validator (not by ``ge=1, le=20``) so non-Codex clients that send legacy values like ``0`` continue to be accepted instead of getting a 422. """ from __future__ import annotations import asyncio import json import os import sys import types from typing import Any import pytest _backend = os.path.join(os.path.dirname(__file__), "..") if _backend not in sys.path: sys.path.insert(0, _backend) # Resolved relative to this file so the source-inspection tests work in any # checkout location (CI, dev machines, the review worker, etc.). _BACKEND_DIR = os.path.realpath(os.path.join(os.path.dirname(__file__), "..")) def _backend_file(rel: str) -> str: """Return an absolute path inside the backend tree, regardless of cwd.""" return os.path.join(_BACKEND_DIR, rel) # ── Helpers ───────────────────────────────────────────────────────── class _FakeStream: """Async iterator that yields predetermined string text events. The Codex SDK's ``thread.run_streaming`` returns an async iterable of events. ``_stream_thread_run`` converts those into raw text via ``_coerce_text``; passing in plain strings exercises the simplest coercion path. """ def __init__(self, chunks: list[str]): self._chunks = list(chunks) self._i = 0 def __aiter__(self): return self async def __anext__(self): if self._i >= len(self._chunks): raise StopAsyncIteration text = self._chunks[self._i] self._i += 1 return text class _FakeThread: def __init__(self, chunks: list[str], final: str | None = None): self._chunks = chunks self._final = final if final is not None else "".join(chunks) def run_streaming(self, prompt: str): # ``run_streaming`` may return either an async iterable or a # coroutine that resolves to one; cover the direct-return # shape here, the coroutine shape is covered in a separate # test below. return _FakeStream(self._chunks) async def run(self, prompt: str): return self._final class _FakeAsyncCodex: """Async-context-manager facade matching codex_app_server.AsyncCodex.""" def __init__( self, chunks: list[str] | None = None, final: str | None = None, raise_on_start: Exception | None = None, ): self._chunks = chunks or [] self._final = final self._raise = raise_on_start async def __aenter__(self): return self async def __aexit__(self, exc_type, exc, tb): return False async def thread_start(self, **kwargs): if self._raise is not None: raise self._raise return _FakeThread(self._chunks, self._final) def _install_fake_codex_sdk(monkeypatch, async_codex_cls, *, with_safety_enums = True): """Drop a fake ``codex_app_server`` module into sys.modules so the production lazy-import path picks it up without the real SDK being installed. ``with_safety_enums=True`` (the default) also injects fake ``ApprovalMode`` + ``SandboxMode`` so the round 6b fail-closed path in ``_safe_thread_safety_kwargs`` is not triggered for every test that just wants to exercise stream translation. The two dedicated round 6b tests (fail_closed / explicit_opt_in) pass ``with_safety_enums=False`` so they can prove the fail-closed branch fires when those enums are missing. """ fake_mod = types.ModuleType("codex_app_server") fake_mod.AsyncCodex = async_codex_cls # type: ignore[attr-defined] if with_safety_enums: fake_mod.ApprovalMode = types.SimpleNamespace( # type: ignore[attr-defined] deny_all = "DENY_ALL", auto_review = "AUTO_REVIEW", ) fake_mod.SandboxMode = types.SimpleNamespace( # type: ignore[attr-defined] read_only = "READ_ONLY", workspace_write = "WORKSPACE_WRITE", danger_full_access = "DANGER_FULL_ACCESS", ) # Inject the fake under BOTH module names the production importer # checks. ``openai_codex`` is the canonical upstream name and is # preferred by the lazy-import gate; ``codex_app_server`` is the # legacy / Rust-crate alias. Hosts that have ``openai_codex`` # actually installed (developer venvs, CI runners after the PR's # `pip install openai-codex`) would otherwise bypass the fake and # exercise the real SDK -- the same fake must be reachable under # both names for the test to be deterministic. monkeypatch.setitem(sys.modules, "codex_app_server", fake_mod) monkeypatch.setitem(sys.modules, "openai_codex", fake_mod) # importlib.util.find_spec walks finders, not sys.modules; patch # it directly so the lazy-import gate accepts the fake. import importlib.util as _iu real_find_spec = _iu.find_spec def _shim(name: str, *args, **kwargs): if name in ("codex_app_server", "openai_codex"): return types.SimpleNamespace() return real_find_spec(name, *args, **kwargs) monkeypatch.setattr("importlib.util.find_spec", _shim) # ── Availability probe ───────────────────────────────────────────── class TestCodexAvailability: def test_absent_when_cli_missing(self, monkeypatch): from core.inference import codex_availability as ca monkeypatch.setattr(ca, "_which_codex", lambda: None) monkeypatch.setattr(ca, "_sdk_importable", lambda: False) payload = asyncio.run(ca.probe_codex_availability()) assert payload["installed"] is False assert payload["cli_path"] is None assert payload["sdk_importable"] is False # supported_models is a sensible default even when nothing is # installed so the picker has something to render IF the user # forces the entry on a future status flip. assert isinstance(payload["supported_models"], list) assert len(payload["supported_models"]) > 0 def test_present_but_sdk_missing(self, monkeypatch): from core.inference import codex_availability as ca monkeypatch.setattr(ca, "_which_codex", lambda: "/usr/local/bin/codex") monkeypatch.setattr(ca, "_sdk_importable", lambda: False) async def fake_version(): return "codex-cli 0.133.0" async def fake_logged_in(): return True monkeypatch.setattr(ca, "_detect_version", fake_version) monkeypatch.setattr(ca, "_detect_logged_in", fake_logged_in) payload = asyncio.run(ca.probe_codex_availability()) # The SDK is what backs `AsyncCodex(...)`, so installed=False # when the SDK is missing -- even if a standalone CLI is on # PATH there is no way for Studio to drive it without the # Python bindings. assert payload["installed"] is False assert payload["cli_path"] == "/usr/local/bin/codex" assert payload["sdk_importable"] is False assert payload["version"] == "codex-cli 0.133.0" def test_present_and_logged_out(self, monkeypatch): from core.inference import codex_availability as ca monkeypatch.setattr(ca, "_which_codex", lambda: "/usr/local/bin/codex") monkeypatch.setattr(ca, "_sdk_importable", lambda: True) async def fake_version(): return "codex-cli 0.133.0" async def fake_logged_in(): return False monkeypatch.setattr(ca, "_detect_version", fake_version) monkeypatch.setattr(ca, "_detect_logged_in", fake_logged_in) payload = asyncio.run(ca.probe_codex_availability()) assert payload["installed"] is True assert payload["logged_in"] is False assert payload["version"] == "codex-cli 0.133.0" def test_present_and_logged_in(self, monkeypatch): from core.inference import codex_availability as ca monkeypatch.setattr(ca, "_which_codex", lambda: "/usr/local/bin/codex") monkeypatch.setattr(ca, "_sdk_importable", lambda: True) async def fake_version(): return "codex-cli 0.133.0" async def fake_logged_in(): return True monkeypatch.setattr(ca, "_detect_version", fake_version) monkeypatch.setattr(ca, "_detect_logged_in", fake_logged_in) payload = asyncio.run(ca.probe_codex_availability()) assert payload["installed"] is True assert payload["logged_in"] is True # ── _stream_codex translation ────────────────────────────────────── def _collect_stream(gen) -> list[str]: async def run(): out: list[str] = [] async for line in gen: out.append(line) return out return asyncio.run(run()) def _parse_sse_chunks(lines: list[str]) -> list[dict[str, Any]]: """Decode SSE ``data: {...}`` lines into the chunk dicts. Skips the sentinel ``data: [DONE]`` line and anything that isn't valid JSON. """ out: list[dict[str, Any]] = [] for raw in lines: if not raw.startswith("data:"): continue body = raw[len("data:") :].strip() if not body or body == "[DONE]": continue try: out.append(json.loads(body)) except json.JSONDecodeError: continue return out class TestStreamCodexSingle: def test_streaming_chunks_translate_into_openai_shape(self, monkeypatch): _install_fake_codex_sdk( monkeypatch, lambda: _FakeAsyncCodex(chunks = ["Hello", ", ", "world"]), ) from core.inference.codex_provider import stream_codex lines = _collect_stream( stream_codex( messages = [{"role": "user", "content": "Say hello in 3 chunks."}], model = "gpt-5.4", ) ) chunks = _parse_sse_chunks(lines) # Three content deltas + one usage chunk + one stop chunk. content_chunks = [ c for c in chunks if c.get("choices") and isinstance(c["choices"], list) and c["choices"] and c["choices"][0].get("delta", {}).get("content") ] assert [c["choices"][0]["delta"]["content"] for c in content_chunks] == [ "Hello", ", ", "world", ] # Usage chunk (OpenAI include_usage shape) is a choices=[] entry # with a populated usage block. usage_chunks = [c for c in chunks if c.get("choices") == [] and c.get("usage")] assert len(usage_chunks) == 1 usage = usage_chunks[0]["usage"] assert usage["prompt_tokens"] > 0 assert usage["completion_tokens"] >= 0 # Final stop chunk with finish_reason=stop. stop_chunks = [ c for c in chunks if c.get("choices") and c["choices"] and c["choices"][0].get("finish_reason") == "stop" ] assert len(stop_chunks) == 1 # And the trailing [DONE] sentinel. assert any(line.strip() == "data: [DONE]" for line in lines) def test_empty_user_prompt_emits_helpful_message(self, monkeypatch): _install_fake_codex_sdk(monkeypatch, lambda: _FakeAsyncCodex(chunks = [])) from core.inference.codex_provider import stream_codex lines = _collect_stream( stream_codex( messages = [{"role": "system", "content": "you are helpful"}], model = "gpt-5.4", ) ) text = "\n".join(lines) assert "no user prompt" in text.lower() class TestStreamCodexParallel: def test_parallel_calls_spawn_tabs_and_synthesise(self, monkeypatch): # The fake SDK returns the same canned chunks for every spawned # AsyncCodex instance; we just need to verify the orchestrator # emits N tab_open events, per-tab chunk events keyed by # tab_id, and a final codex_gather summary event. _install_fake_codex_sdk( monkeypatch, lambda: _FakeAsyncCodex( chunks = ["alpha"], final = "synthesised answer", ), ) from core.inference.codex_provider import stream_codex n = 3 lines = _collect_stream( stream_codex( messages = [{"role": "user", "content": "Test"}], model = "gpt-5.4", parallel_calls = n, ) ) chunks = _parse_sse_chunks(lines) tool_events = [c["_toolEvent"] for c in chunks if "_toolEvent" in c] tab_opens = [e for e in tool_events if e.get("type") == "codex_tab_open"] tab_chunks = [e for e in tool_events if e.get("type") == "codex_tab_chunk"] tab_closes = [e for e in tool_events if e.get("type") == "codex_tab_close"] gather = [e for e in tool_events if e.get("type") == "codex_gather"] # Each tab opens once -- the N tabs are pre-emitted so the # UI can paint the strip before content arrives. assert len(tab_opens) == n assert sorted(e["tab_id"] for e in tab_opens) == list(range(1, n + 1)) # Per-tab chunks may interleave in any order but every tab id # must produce at least one chunk before its close event. seen_tabs = {e["tab_id"] for e in tab_chunks} assert seen_tabs == set(range(1, n + 1)) # Each tab emits exactly one close marker. assert sorted(e["tab_id"] for e in tab_closes) == list(range(1, n + 1)) # Exactly one synthesis event with the unified summary. assert len(gather) == 1 assert gather[0]["tab_count"] == n # The summary text comes from the final synthesis Codex call; # our fake returns "synthesised answer" via .run(). assert "synth" in gather[0]["summary"].lower() def test_parallel_calls_clamped_to_maximum(self, monkeypatch): """Passing parallel_calls=500 must NOT spawn 500 tasks; the clamp at MAX_PARALLEL_CALLS keeps the local CLI safe. """ from core.inference import codex_provider as cp _install_fake_codex_sdk( monkeypatch, lambda: _FakeAsyncCodex(chunks = ["x"], final = "synth"), ) lines = _collect_stream( cp.stream_codex( messages = [{"role": "user", "content": "x"}], model = "gpt-5.4", parallel_calls = 500, ) ) chunks = _parse_sse_chunks(lines) tab_opens = [ c["_toolEvent"] for c in chunks if c.get("_toolEvent", {}).get("type") == "codex_tab_open" ] assert len(tab_opens) == cp.MAX_PARALLEL_CALLS def test_parallel_calls_one_takes_single_path(self, monkeypatch): """parallel_calls=1 must not emit any tab tool-events -- it's the regular single-call shape. """ _install_fake_codex_sdk( monkeypatch, lambda: _FakeAsyncCodex(chunks = ["one"]), ) from core.inference.codex_provider import stream_codex lines = _collect_stream( stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.4", parallel_calls = 1, ) ) chunks = _parse_sse_chunks(lines) tool_events = [c.get("_toolEvent") for c in chunks if c.get("_toolEvent")] for event in tool_events: assert not (event.get("type") or "").startswith("codex_tab") assert event.get("type") != "codex_gather" # ── Request validator ────────────────────────────────────────────── class TestParallelCallsValidator: def test_request_accepts_valid_range(self): from models.inference import ChatCompletionRequest for n in (1, 5, 10, 20): req = ChatCompletionRequest( model = "gpt-5.4", messages = [{"role": "user", "content": "hi"}], parallel_calls = n, ) assert req.parallel_calls == n def test_request_clamps_below_one(self): """Pre-PR clients sometimes sent `parallel_calls=0` as a stray OpenAI extra and the request was silently accepted; rejecting with 422 would regress that. The validator now clamps to 1. """ from models.inference import ChatCompletionRequest for n in (0, -1, -100): req = ChatCompletionRequest( model = "gpt-5.4", messages = [{"role": "user", "content": "hi"}], parallel_calls = n, ) assert req.parallel_calls == 1, f"clamp failed for {n}" def test_request_clamps_above_twenty(self): """A runaway value (1000, etc.) is clamped to the 20 cap so it cannot saturate the local CLI even when the client misbehaves. """ from models.inference import ChatCompletionRequest for n in (21, 100, 1000): req = ChatCompletionRequest( model = "gpt-5.4", messages = [{"role": "user", "content": "hi"}], parallel_calls = n, ) assert req.parallel_calls == 20, f"clamp failed for {n}" def test_request_coerces_garbage_to_one(self): """Strings / floats / None coerce to 1 instead of 422 so a legacy or misconfigured client cannot break chat for everyone.""" from models.inference import ChatCompletionRequest for value in (None, "garbage", float("nan")): req = ChatCompletionRequest( model = "gpt-5.4", messages = [{"role": "user", "content": "hi"}], parallel_calls = value, ) assert req.parallel_calls == 1 def test_request_default_is_one(self): """Default = 1 so the field matches the single-call code path and the schema documentation. Non-codex providers ignore the field regardless of its value, so backwards compat is preserved. """ from models.inference import ChatCompletionRequest req = ChatCompletionRequest( model = "gpt-5.4", messages = [{"role": "user", "content": "hi"}], ) assert req.parallel_calls == 1 # ── Codex unavailable surfacing ──────────────────────────────────── class TestCodexUnavailable: def test_missing_sdk_raises_typed_error(self, monkeypatch): # Force find_spec to return None so the lazy import fails. # The provider probes both the canonical upstream name # ``openai_codex`` and the legacy alias ``codex_app_server``, # so we have to suppress both for the import to fail. import importlib.util as _iu real = _iu.find_spec _SDK_NAMES = {"openai_codex", "codex_app_server"} def _shim(name, *args, **kwargs): if name in _SDK_NAMES: return None return real(name, *args, **kwargs) monkeypatch.setattr("importlib.util.find_spec", _shim) # Also drop any cached fakes from prior tests. for _name in _SDK_NAMES: monkeypatch.delitem(sys.modules, _name, raising = False) from core.inference.codex_provider import ( CodexUnavailableError, stream_codex, ) with pytest.raises(CodexUnavailableError): asyncio.run( _consume_first( stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.4", ) ) ) class TestCodexHardenedRegressions: """Tests covering the post-review hardening pass. Each test pins a specific regression: the wrong subcommand (``codex auth login`` → ``codex login``), the wrong SDK package name (``codex_app_server`` → ``openai_codex`` with legacy alias), the ``not logged in`` substring footgun, the ANSI-wrapped device URL, and the fan-out cancellation contract. """ def test_sdk_probes_openai_codex_first(self, monkeypatch): """The canonical upstream name must be tried before the alias.""" import importlib.util as _iu real = _iu.find_spec calls: list[str] = [] def _shim(name, *args, **kwargs): if name in ("openai_codex", "codex_app_server"): calls.append(name) return None return real(name, *args, **kwargs) monkeypatch.setattr("importlib.util.find_spec", _shim) from core.inference.codex_availability import _sdk_importable assert _sdk_importable() is False assert ( calls and calls[0] == "openai_codex" ), f"availability probe must check openai_codex first; saw {calls}" def test_login_status_uses_login_subcommand(self): """Upstream is `codex login status`, NOT `codex auth status`.""" src = _backend_file("core/inference/codex_availability.py") text = open(src).read() assert ( '"auth", "status"' not in text ), "_detect_logged_in must use `codex login status`, not `codex auth status`" assert '"login", "status"' in text def test_device_login_uses_login_subcommand(self): src = _backend_file("core/inference/codex_provider.py") text = open(src).read() assert ( '"auth", "login", "--device-auth"' not in text ), "stream_codex_device_login must use `codex login --device-auth`" assert '"login", "--device-auth"' in text def test_not_logged_in_not_misparsed_as_logged_in(self): """The substring "logged in" inside "not logged in" must not flip the detection to True.""" import asyncio from core.inference import codex_availability as av async def _fake_run_cli(args, **kw): return (0, "Not logged in. Run `codex login` to authenticate.", "") orig = av._run_cli av._run_cli = _fake_run_cli # type: ignore[assignment] try: result = asyncio.run(av._detect_logged_in()) assert result is False, "'Not logged in' was misparsed as logged_in=True" finally: av._run_cli = orig # type: ignore[assignment] def test_logged_in_is_detected(self): import asyncio from core.inference import codex_availability as av async def _fake_run_cli(args, **kw): return (0, "Logged in using ChatGPT", "") orig = av._run_cli av._run_cli = _fake_run_cli # type: ignore[assignment] try: result = asyncio.run(av._detect_logged_in()) assert result is True finally: av._run_cli = orig # type: ignore[assignment] def test_multi_turn_prompt_includes_prior_turns(self): """The Codex prompt MUST contain prior assistant turns.""" from core.inference.codex_provider import _last_user_prompt msgs = [ {"role": "user", "content": "what is the capital of france?"}, {"role": "assistant", "content": "Paris."}, {"role": "user", "content": "and germany?"}, ] prompt = _last_user_prompt(msgs) assert "and germany?" in prompt assert ( "Paris" in prompt ), f"PRIOR ASSISTANT TURN DROPPED — multi-turn broken. Prompt:\n{prompt}" assert "capital of france" in prompt.lower() def test_single_turn_prompt_unchanged(self): """Single-turn case must not get the User:/Assistant: framing.""" from core.inference.codex_provider import _last_user_prompt prompt = _last_user_prompt([{"role": "user", "content": "hi"}]) assert prompt == "hi" def test_default_models_no_o3(self): """The Codex registry must not advertise `o3` (not in upstream).""" from core.inference.providers import PROVIDER_REGISTRY codex = PROVIDER_REGISTRY["codex"] assert ( "o3" not in codex["default_models"] ), "o3 is not a Codex model; remove from default_models" assert "gpt-5.5" in codex["default_models"] def test_inference_route_no_raw_exc_leak(self): """SSE error frame must NOT echo str(exc) verbatim (CodeQL).""" import re src = _backend_file("routes/inference.py") text = open(src).read() bad = re.findall(r'f["\']Codex error:\s*\{exc\}["\']', text) assert not bad, f"raw exception in SSE: {bad}" def test_codex_route_no_raw_exc_leak(self): """codex.py SSE stream wrapping must also not leak str(exc).""" import re src = _backend_file("routes/codex.py") text = open(src).read() for line in text.splitlines(): ls = line.strip() if ls.startswith("yield ") and re.search(r"\{exc\}|\{e\}", ls): assert False, f"raw exception leaked: {ls}" def test_parallel_tab_error_sanitised(self, monkeypatch): """A worker that raises with a path-leaking message must NOT send that text to the client; the SSE codex_tab_error event must carry a generic message + exception_type. """ fake = _FakeAsyncCodex( raise_on_start = RuntimeError( "secret /home/alice/.codex/config.json token=abc" ) ) _install_fake_codex_sdk(monkeypatch, lambda: fake) from core.inference.codex_provider import stream_codex chunks: list[str] = [] async def _collect(): async for c in stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 2, ): chunks.append(c) asyncio.run(_collect()) body = "".join(chunks) assert ( "secret /home/alice" not in body ), "raw exception text leaked into codex_tab_error SSE frame" assert "Codex tab failed" in body or "exception_type" in body def test_codex_subprocess_env_scrubbed(self, monkeypatch): """The codex subprocess env must not include other-provider secrets. OPENAI_API_KEY is intentionally excluded too: a shimmed `codex` binary on PATH must not receive Studio's stored OpenAI provider key. Users wire Codex auth via `codex login` or the codex-specific CODEX_OPENAI_API_KEY override instead. """ from core.inference.codex_availability import _codex_subprocess_env monkeypatch.setenv("HF_TOKEN", "hf_should_not_leak") monkeypatch.setenv("GH_TOKEN", "gh_should_not_leak") monkeypatch.setenv("WANDB_API_KEY", "wandb_should_not_leak") monkeypatch.setenv("ANTHROPIC_API_KEY", "anthropic_should_not_leak") monkeypatch.setenv("OPENAI_API_KEY", "openai_provider_key_not_for_codex") monkeypatch.setenv("CODEX_OPENAI_API_KEY", "codex_specific_key") monkeypatch.setenv("CODEX_HOME", "/custom/.codex") monkeypatch.setenv("PATH", "/usr/bin") env = _codex_subprocess_env() for secret in ( "HF_TOKEN", "GH_TOKEN", "WANDB_API_KEY", "ANTHROPIC_API_KEY", # OPENAI_API_KEY belongs to the OpenAI provider, not Codex. "OPENAI_API_KEY", ): assert secret not in env, f"{secret} leaked into codex env" # Codex-relevant keys must be preserved. assert env.get("CODEX_OPENAI_API_KEY") == "codex_specific_key" assert env.get("CODEX_HOME") == "/custom/.codex" assert env.get("PATH") == "/usr/bin" def test_partial_stream_failure_does_not_replay_turn(self, monkeypatch): """If turn.stream() fails after emitting some text, the buffered run() fallback must NOT fire -- replaying would duplicate side effects (file writes, shell commands). """ run_calls = {"n": 0} class _PartialStreamTurn: async def stream(self): yield {"text": "partial output "} raise RuntimeError("network glitch mid-stream") class _ThreadPartialFail: def turn(self, prompt): return _PartialStreamTurn() async def run(self, prompt): run_calls["n"] += 1 return "REPLAYED -- BAD" class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): return _ThreadPartialFail() _install_fake_codex_sdk(monkeypatch, _Async) from core.inference.codex_provider import stream_codex chunks: list[str] = [] async def _collect(): async for c in stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 1, ): chunks.append(c) asyncio.run(_collect()) assert run_calls["n"] == 0, ( "buffered run() fired after partial stream emission -- " "would replay side effects" ) body = "".join(chunks) assert "partial output" in body assert "REPLAYED" not in body def test_not_signed_in_wording_also_handled(self): """`Not signed in` (alternative localisation) must also be treated as logged-out, not as positive match. """ import asyncio from core.inference import codex_availability as av async def _fake_run_cli(args, **kw): return (0, "Not signed in.", "") orig = av._run_cli av._run_cli = _fake_run_cli # type: ignore[assignment] try: assert asyncio.run(av._detect_logged_in()) is False finally: av._run_cli = orig # type: ignore[assignment] def test_device_url_accepts_generic_verification_url(self): """The login parser must accept upstream's chatgpt.com/activate URL as well as the canonical /codex/device shape. """ import re src = _backend_file("core/inference/codex_provider.py") text = open(src).read() # Find the url_re pattern literal and compile it. m = re.search(r"url_re\s*=\s*re\.compile\(\s*\n?\s*r\"([^\"]+)\"", text) assert m, "url_re definition not found" pattern = re.compile(m.group(1), re.IGNORECASE) # Upstream device URLs we expect to match. for u in ( "https://auth.openai.com/codex/device", "https://chatgpt.com/activate", "https://auth.openai.com/device/verify?code=ABCD", ): assert pattern.search(u), f"device URL regex missed: {u}" def test_synthesis_call_forwards_system_prompt(self, monkeypatch): """`_run_codex_synthesis` must pass the system prompt so a fan-out style instruction ("Always answer in Spanish") survives the unification step. """ seen_kwargs: list[dict] = [] seen_prompts: list[str] = [] class _SynThread: async def run(self, prompt): seen_prompts.append(prompt) return "synth ok" def turn(self, prompt): # Force buffered path via no `stream` attr. class _T: pass return _T() class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): seen_kwargs.append(kw) return _SynThread() _install_fake_codex_sdk(monkeypatch, _Async) from core.inference.codex_provider import _run_codex_synthesis out = asyncio.run( _run_codex_synthesis( model = "gpt-5.5", system = "Always answer in Spanish.", prompt = "What is the capital of France?", tab_outputs = ["Paris", "Paris."], ) ) # The upstream openai_codex SDK uses `base_instructions` for the # system prompt; the legacy alias accepts `system`; the last-resort # fallback inlines the system text into the user prompt. Accept # any of those paths. system_seen = ( any("Spanish" in (kw.get("base_instructions") or "") for kw in seen_kwargs) or any("Spanish" in (kw.get("system") or "") for kw in seen_kwargs) or any("Always answer in Spanish" in p for p in seen_prompts) ) assert system_seen, ( f"system prompt dropped in synthesis. kwargs={seen_kwargs} " f"prompts={seen_prompts}" ) # And the synthesis still returned the model's text. assert "synth" in out.lower() def test_sdk_env_scrubbed_via_appserverconfig(self, monkeypatch): """The SDK construction path must wire AppServerConfig(env=...) when the SDK exposes it, so HF_TOKEN / GH_TOKEN are not leaked to the codex app-server subprocess. """ monkeypatch.setenv("HF_TOKEN", "should_be_scrubbed") monkeypatch.setenv("GH_TOKEN", "should_be_scrubbed") # OPENAI_API_KEY is now ALSO scrubbed -- it belongs to the # OpenAI provider, not Codex. CODEX_OPENAI_API_KEY is the # codex-specific override that survives. monkeypatch.setenv("OPENAI_API_KEY", "openai_provider_key_not_for_codex") monkeypatch.setenv("CODEX_OPENAI_API_KEY", "codex_specific_key") seen_configs: list[Any] = [] class _FakeAppServerConfig: def __init__(self, env = None, **kw): self.env = env or {} class _Async: def __init__(self, config = None): seen_configs.append(config) async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): return _FakeThread(chunks = ["ok"]) # Inject a fake openai_codex module exposing AppServerConfig. import importlib.util as _iu import types as _types fake_mod = _types.ModuleType("openai_codex") fake_mod.AsyncCodex = _Async # type: ignore[attr-defined] fake_mod.AppServerConfig = _FakeAppServerConfig # type: ignore[attr-defined] # Round 6b: safety enums must be present or the fail-closed # path raises before AppServerConfig ever gets consulted. fake_mod.ApprovalMode = _types.SimpleNamespace( # type: ignore[attr-defined] deny_all = "DENY_ALL", auto_review = "AUTO", ) fake_mod.SandboxMode = _types.SimpleNamespace( # type: ignore[attr-defined] read_only = "READ_ONLY", workspace_write = "WW", ) monkeypatch.setitem(sys.modules, "openai_codex", fake_mod) real_find_spec = _iu.find_spec monkeypatch.setattr( "importlib.util.find_spec", lambda n, *a, **kw: ( _types.SimpleNamespace() if n in ("openai_codex", "codex_app_server") else real_find_spec(n, *a, **kw) ), ) from core.inference.codex_provider import stream_codex asyncio.run( _consume_first( stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 1, ) ) ) assert seen_configs, "AsyncCodex was never instantiated" cfg = seen_configs[0] assert cfg is not None, "AppServerConfig was not passed to AsyncCodex" assert ( "HF_TOKEN" in cfg.env and cfg.env["HF_TOKEN"] == "" ), "HF_TOKEN not overridden to empty in SDK env" assert "GH_TOKEN" in cfg.env and cfg.env["GH_TOKEN"] == "" # OPENAI_API_KEY is intentionally overridden to empty in the # SDK env so the app-server cannot use it as a Codex credential # by accident. The OpenAI provider still reads its own key from # Studio's storage; nothing in this path needs the env var. assert cfg.env.get("OPENAI_API_KEY") == "" # CODEX_OPENAI_API_KEY is the Codex-specific override and must # survive untouched so users can wire that key into Codex. assert "CODEX_OPENAI_API_KEY" not in cfg.env def test_thread_turn_stream_path_taken(self, monkeypatch): """The canonical openai_codex API uses thread.turn(prompt).stream(); the provider must prefer that over the legacy run_streaming hook. """ events_seen = {"turn_called": False, "run_streaming_called": False} class _TurnEvent: def __init__(self, txt): self.payload = {"text": txt} class _TurnHandle: def __init__(self, prompt): self.prompt = prompt async def stream(self): yield _TurnEvent("hello ") yield _TurnEvent("from turn.stream") class _ThreadWithTurn: def turn(self, prompt): events_seen["turn_called"] = True return _TurnHandle(prompt) def run_streaming(self, prompt): events_seen["run_streaming_called"] = True raise AssertionError("should not be called when turn().stream() works") async def run(self, prompt): raise AssertionError("should not fall through to buffered run()") class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): return _ThreadWithTurn() _install_fake_codex_sdk(monkeypatch, _Async) from core.inference.codex_provider import stream_codex chunks: list[str] = [] async def _collect(): async for c in stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 1, ): chunks.append(c) asyncio.run(_collect()) assert events_seen["turn_called"], "thread.turn() never called" assert not events_seen["run_streaming_called"] body = "".join(chunks) # Each text chunk wraps in its own SSE delta, so check both pieces. assert '"content": "hello "' in body assert '"content": "from turn.stream"' in body def test_installed_requires_both_cli_and_sdk(self, monkeypatch): """Round 6 revert: the login route shells out to `codex`, so marking `installed=True` on SDK-only would surface a Codex provider row whose Sign-in button immediately fails. The canonical `openai-codex` package installs `openai-codex-cli-bin` which puts the `codex` shim on PATH, so common installs still light up correctly; the gate just refuses to advertise a provider Studio cannot actually drive. """ from core.inference import codex_availability as ca # SDK present, no CLI -> hidden (cannot complete login). monkeypatch.setattr(ca, "_which_codex", lambda: None) monkeypatch.setattr(ca, "_sdk_importable", lambda: True) payload = asyncio.run(ca.probe_codex_availability()) assert payload["installed"] is False assert payload["cli_path"] is None assert payload["sdk_importable"] is True # CLI present, SDK missing -> still hidden (cannot drive chat). monkeypatch.setattr(ca, "_which_codex", lambda: "/usr/bin/codex") monkeypatch.setattr(ca, "_sdk_importable", lambda: False) async def fake_version(): return "codex-cli 0.133.0" async def fake_logged_in(): return True monkeypatch.setattr(ca, "_detect_version", fake_version) monkeypatch.setattr(ca, "_detect_logged_in", fake_logged_in) payload2 = asyncio.run(ca.probe_codex_availability()) assert payload2["installed"] is False def test_base_instructions_kwarg_preferred(self, monkeypatch): """The upstream openai_codex SDK uses `base_instructions` for the system prompt. The provider must try that name first; only if the SDK rejects it should it fall back to `system`. """ seen_kwargs: list[dict] = [] class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): seen_kwargs.append(dict(kw)) return _FakeThread(chunks = ["ok"]) _install_fake_codex_sdk(monkeypatch, _Async) from core.inference.codex_provider import stream_codex async def _collect(): async for _ in stream_codex( messages = [ {"role": "system", "content": "You are helpful."}, {"role": "user", "content": "hi"}, ], model = "gpt-5.5", parallel_calls = 1, ): pass asyncio.run(_collect()) # The first (and only, since this fake accepts any kwargs) # call must use base_instructions, not the legacy `system`. assert seen_kwargs, "thread_start was never called" assert ( "base_instructions" in seen_kwargs[0] ), f"upstream-canonical kwarg not used: {seen_kwargs[0]}" assert seen_kwargs[0]["base_instructions"] == "You are helpful." assert ( "system" not in seen_kwargs[0] ), "legacy `system` kwarg was sent even though base_instructions worked" def test_base_instructions_falls_back_to_system(self, monkeypatch): """When the SDK rejects `base_instructions` with TypeError the helper must retry with the legacy `system` kwarg before giving up and inlining the system text in the prompt. """ call_log: list[dict] = [] class _StrictSDK: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): call_log.append(dict(kw)) if "base_instructions" in kw: raise TypeError( "thread_start() got an unexpected keyword 'base_instructions'" ) return _FakeThread(chunks = ["ok"]) _install_fake_codex_sdk(monkeypatch, _StrictSDK) from core.inference.codex_provider import stream_codex async def _collect(): async for _ in stream_codex( messages = [ {"role": "system", "content": "You are helpful."}, {"role": "user", "content": "hi"}, ], model = "gpt-5.5", parallel_calls = 1, ): pass asyncio.run(_collect()) assert len(call_log) >= 2, "fallback to `system` kwarg never tried" assert "base_instructions" in call_log[0] assert "system" in call_log[1] and call_log[1]["system"] == "You are helpful." def test_scrubbed_env_wrapper_strips_secrets_before_construction(self, monkeypatch): """When AppServerConfig is missing the fail-closed wrapper must remove secret env vars BEFORE the SDK constructor runs (the SDK starts its app-server with `env = os.environ.copy()`). """ observed_env_during_init: dict[str, str | None] = {} class _NoConfigAsync: def __init__(self): # Capture the environment exactly as the SDK would see # it at construction time. observed_env_during_init["HF_TOKEN"] = os.environ.get("HF_TOKEN") observed_env_during_init["GH_TOKEN"] = os.environ.get("GH_TOKEN") observed_env_during_init["WANDB_API_KEY"] = os.environ.get( "WANDB_API_KEY" ) observed_env_during_init["PATH"] = os.environ.get("PATH") observed_env_during_init["CODEX_HOME"] = os.environ.get("CODEX_HOME") async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): return _FakeThread(chunks = ["ok"]) monkeypatch.setenv("HF_TOKEN", "should_be_gone") monkeypatch.setenv("GH_TOKEN", "should_be_gone") monkeypatch.setenv("WANDB_API_KEY", "should_be_gone") monkeypatch.setenv("PATH", "/usr/bin") monkeypatch.setenv("CODEX_HOME", "/home/u/.codex") # No AppServerConfig in the fake module -- forces the wrapper path. _install_fake_codex_sdk(monkeypatch, _NoConfigAsync) from core.inference.codex_provider import stream_codex async def _collect(): async for _ in stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 1, ): pass asyncio.run(_collect()) # Secrets must have been removed from os.environ BEFORE the # SDK constructor captured the env. assert ( observed_env_during_init["HF_TOKEN"] is None ), "HF_TOKEN visible to SDK constructor -- env scrub failed" assert observed_env_during_init["GH_TOKEN"] is None assert observed_env_during_init["WANDB_API_KEY"] is None # Safe-listed keys must survive. assert observed_env_during_init["PATH"] == "/usr/bin" assert observed_env_during_init["CODEX_HOME"] == "/home/u/.codex" # And the wrapper must restore them after exit. assert os.environ.get("HF_TOKEN") == "should_be_gone" assert os.environ.get("GH_TOKEN") == "should_be_gone" def test_coerce_text_drops_non_answer_event_types(self): """Tool / command / plan deltas have their own `delta` fields that must NOT be rendered as assistant text -- otherwise local stdout, file paths, or tool-call arguments would leak into the Chat Completions reply. Round 6 also requires the object-shape path to gate on type and class name; the upstream SDK emits typed notification objects (CommandExecutionOutputDelta, FileChangeDelta, etc.) with `.delta` strings that would otherwise leak. """ from core.inference.codex_provider import _coerce_text # Allowed answer-bearing event types contribute text. assert _coerce_text({"type": "message.delta", "delta": "hello"}) == "hello" assert _coerce_text({"type": "completed", "text": "done"}) == "done" assert _coerce_text({"type": "text_delta", "delta": "x"}) == "x" # Non-answer dict event types are silenced. for ev_type in ( "command.delta", "command_output", "file_write.delta", "tool_call.delta", "plan.update", "exec.stdout", "exec.stderr", "patch.apply", "thread.tool_call", "agent_reasoning", ): payload = {"type": ev_type, "delta": "this should NOT leak"} assert _coerce_text(payload) == "", ( f"{ev_type} leaked text into assistant reply: " f"{_coerce_text(payload)!r}" ) # Object-shape gate: typed payloads whose class name contains # a tool/command/file/patch/plan marker drop the .delta too. class CommandExecutionOutputDelta: delta = "SECRET_STDOUT" class FileChangeDelta: delta = "secret/file/path" class ToolCallDelta: text = "tool_arg_payload" class PatchApplyDelta: delta = "diff --git a/secret" class PlanUpdateDelta: delta = "plan content" class AgentReasoningDelta: delta = "internal CoT" for obj in ( CommandExecutionOutputDelta(), FileChangeDelta(), ToolCallDelta(), PatchApplyDelta(), PlanUpdateDelta(), AgentReasoningDelta(), ): assert _coerce_text(obj) == "", ( f"object-shape {obj.__class__.__name__} leaked: " f"{_coerce_text(obj)!r}" ) # Object with explicit type attr also drops if not in allow-list. class _WithType: type = "command.delta" delta = "leak" assert _coerce_text(_WithType()) == "" # Object-shape answer events DO pass through. class AgentMessageDelta: delta = "real assistant text" assert _coerce_text(AgentMessageDelta()) == "real assistant text" # Plain strings and untyped dicts still pass through (legacy path). assert _coerce_text("raw text") == "raw text" assert _coerce_text({"text": "no type tag"}) == "no type tag" def test_authenticated_yes_wording_is_detected(self): """An `Authenticated: Yes` line (a wording the CLI ships in some locales / versions) must be parsed as logged-in. """ from core.inference import codex_availability as av async def _fake_run_cli(args, **kw): return (0, "Authenticated: Yes\nuser@example.com", "") orig = av._run_cli av._run_cli = _fake_run_cli # type: ignore[assignment] try: assert asyncio.run(av._detect_logged_in()) is True finally: av._run_cli = orig # type: ignore[assignment] def test_codex_openai_api_key_overrides_openai_provider_key(self, monkeypatch): """Studio's `OPENAI_API_KEY` must NOT reach codex -- but the codex-specific `CODEX_OPENAI_API_KEY` MUST be forwarded so users can deliberately wire a key into Codex. """ from core.inference.codex_availability import _codex_subprocess_env monkeypatch.setenv("OPENAI_API_KEY", "belongs_to_openai_provider") monkeypatch.setenv("CODEX_OPENAI_API_KEY", "explicit_codex_key") env = _codex_subprocess_env() assert ( "OPENAI_API_KEY" not in env ), "OpenAI provider key leaked into codex subprocess env" assert env.get("CODEX_OPENAI_API_KEY") == "explicit_codex_key" def test_thread_start_uses_safe_approval_and_sandbox(self, monkeypatch): """When the SDK exposes ApprovalMode + SandboxMode, the provider MUST pin approval to `deny_all` and sandbox to `read_only`. The upstream SDK default (`auto_review` approvals, unspecified sandbox) would let the model auto-execute commands and write files on the server. """ seen_kwargs: list[dict] = [] class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): seen_kwargs.append(dict(kw)) return _FakeThread(chunks = ["ok"]) # Drop a fake openai_codex with ApprovalMode + SandboxMode enums. import importlib.util as _iu fake_mod = types.ModuleType("openai_codex") fake_mod.AsyncCodex = _Async # type: ignore[attr-defined] fake_mod.ApprovalMode = types.SimpleNamespace( # type: ignore[attr-defined] deny_all = "DENY_ALL_SENTINEL", auto_review = "AUTO_REVIEW_SENTINEL", ) fake_mod.SandboxMode = types.SimpleNamespace( # type: ignore[attr-defined] read_only = "READ_ONLY_SENTINEL", workspace_write = "WS_WRITE_SENTINEL", danger_full_access = "DANGER_SENTINEL", ) monkeypatch.setitem(sys.modules, "openai_codex", fake_mod) real_find_spec = _iu.find_spec monkeypatch.setattr( "importlib.util.find_spec", lambda n, *a, **kw: ( types.SimpleNamespace() if n in ("openai_codex", "codex_app_server") else real_find_spec(n, *a, **kw) ), ) from core.inference.codex_provider import stream_codex async def _collect(): async for _ in stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 1, ): pass asyncio.run(_collect()) assert seen_kwargs, "thread_start never called" kw = seen_kwargs[0] assert ( kw.get("approval_mode") == "DENY_ALL_SENTINEL" ), f"approval_mode not pinned to deny_all: {kw}" assert ( kw.get("sandbox") == "READ_ONLY_SENTINEL" ), f"sandbox not pinned to read_only: {kw}" def test_safety_kwargs_finds_sandbox_mode_in_submodule(self, monkeypatch): """Round 6 caught that `SandboxMode` is exported by the upstream SDK from `openai_codex.generated.v2_all`, NOT from the top-level `openai_codex` package. The previous lookup used `getattr(sdk_mod, 'SandboxMode', None)` only and returned None for the canonical SDK install, silently degrading to the unsafe auto_review default. """ seen_kwargs: list[dict] = [] class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): seen_kwargs.append(dict(kw)) return _FakeThread(chunks = ["ok"]) # Build a fake openai_codex that DOES NOT expose SandboxMode # at the top level -- only inside `.generated.v2_all`. import importlib.util as _iu fake_root = types.ModuleType("openai_codex") fake_root.AsyncCodex = _Async # type: ignore[attr-defined] fake_root.ApprovalMode = types.SimpleNamespace( # type: ignore[attr-defined] deny_all = "DENY_ALL", auto_review = "AUTO", ) # Submodule chain `.generated.v2_all` fake_generated = types.ModuleType("openai_codex.generated") fake_v2 = types.ModuleType("openai_codex.generated.v2_all") fake_v2.SandboxMode = types.SimpleNamespace( # type: ignore[attr-defined] read_only = "READ_ONLY", workspace_write = "WW", ) fake_generated.v2_all = fake_v2 # type: ignore[attr-defined] fake_root.generated = fake_generated # type: ignore[attr-defined] monkeypatch.setitem(sys.modules, "openai_codex", fake_root) monkeypatch.setitem(sys.modules, "openai_codex.generated", fake_generated) monkeypatch.setitem(sys.modules, "openai_codex.generated.v2_all", fake_v2) real_find_spec = _iu.find_spec monkeypatch.setattr( "importlib.util.find_spec", lambda n, *a, **kw: ( types.SimpleNamespace() if n in ("openai_codex", "codex_app_server") else real_find_spec(n, *a, **kw) ), ) from core.inference.codex_provider import stream_codex async def _collect(): async for _ in stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 1, ): pass asyncio.run(_collect()) assert seen_kwargs, "thread_start never called" kw = seen_kwargs[0] assert ( kw.get("approval_mode") == "DENY_ALL" ), f"approval_mode not pinned even with submodule SandboxMode: {kw}" assert ( kw.get("sandbox") == "READ_ONLY" ), f"sandbox not pinned via submodule lookup: {kw}" def test_scrubbed_env_construction_failure_restores_env(self, monkeypatch): """Round 6: if the SDK constructor raises before __aenter__ returns, the previous wrapper never called __aexit__ so the scrubbed env vars leaked permanently. Now the scrub is rolled back on failure. """ from core.inference.codex_provider import _ScrubbedEnvAsyncCodex monkeypatch.setenv("HF_TOKEN", "must_survive") class _FailingAsync: def __init__(self): raise RuntimeError("SDK construction failed") async def _run(): wrapper = _ScrubbedEnvAsyncCodex(_FailingAsync) try: async with wrapper: pass except RuntimeError: pass asyncio.run(_run()) # HF_TOKEN must be restored even though __aexit__ never fired # for the failed construction. assert ( os.environ.get("HF_TOKEN") == "must_survive" ), "scrubbed env leaked permanently when SDK construction failed" def test_thread_start_fails_closed_when_safety_unavailable(self, monkeypatch): """Round 6b: if the installed SDK cannot expose ApprovalMode or SandboxMode, the provider MUST fail closed rather than silently fall through to the SDK's `auto_review` default. A server-side chat surface with no per-action approval UI cannot tolerate the model deciding on its own to run shell commands. The error surfaces as a typed CodexUnavailableError the route layer translates to 503. """ # Make sure the override env var is NOT set. monkeypatch.delenv("UNSLOTH_CODEX_ALLOW_UNSAFE_DEFAULTS", raising = False) seen_kwargs: list[dict] = [] class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): seen_kwargs.append(dict(kw)) return _FakeThread(chunks = ["ok"]) _install_fake_codex_sdk(monkeypatch, _Async, with_safety_enums = False) from core.inference.codex_provider import ( CodexUnavailableError, stream_codex, ) async def _collect(): async for _ in stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 1, ): pass with pytest.raises(CodexUnavailableError) as exc_info: asyncio.run(_collect()) assert "ApprovalMode" in str(exc_info.value) or "SandboxMode" in str( exc_info.value ) assert not seen_kwargs, ( "thread_start must NOT have been called when safety pins " "could not be applied" ) def test_thread_start_allows_unsafe_defaults_with_explicit_opt_in( self, monkeypatch ): """When the operator deliberately sets the UNSLOTH_CODEX_ALLOW_UNSAFE_DEFAULTS escape hatch, the provider proceeds without the safety pins (logs a warning) instead of raising. This is the dev-only override for pre-release alpha SDK builds that have not yet exposed ApprovalMode/SandboxMode. """ monkeypatch.setenv("UNSLOTH_CODEX_ALLOW_UNSAFE_DEFAULTS", "1") seen_kwargs: list[dict] = [] class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): seen_kwargs.append(dict(kw)) return _FakeThread(chunks = ["ok"]) _install_fake_codex_sdk(monkeypatch, _Async, with_safety_enums = False) from core.inference.codex_provider import stream_codex async def _collect(): async for _ in stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 1, ): pass asyncio.run(_collect()) assert seen_kwargs, "thread_start never called under override" kw = seen_kwargs[0] assert "approval_mode" not in kw assert "sandbox" not in kw assert kw.get("model") == "gpt-5.5" def test_device_login_log_filter_drops_unknown_lines(self, monkeypatch): """The login stream's `log` events must not forward arbitrary subprocess output. Only an allow-list of known progress strings reaches the browser; anything else (auth JSON, tokens, paths, error tails) stays in backend logs. """ # Build a synthetic stdout stream with one safe line and one # unsafe line, then drive the login generator against it. from core.inference import codex_provider as cp class _FakeStdout: def __init__(self, lines: list[bytes]): self._lines = list(lines) async def readline(self) -> bytes: if not self._lines: return b"" return self._lines.pop(0) class _FakeProc: pid = 99999 returncode = None stdout = _FakeStdout( [ b"Welcome to Codex\n", b"Open: https://auth.openai.com/codex/device\n", b"Enter this one-time code: ABCD-EFGH\n", b'{"refresh_token": "rt_LEAK_LEAK_LEAK"}\n', b"/home/u/.codex/auth.json saved\n", b"Successfully logged in\n", ], ) async def wait(self): self.returncode = 0 return 0 def kill(self): self.returncode = -9 def terminate(self): self.returncode = -15 async def _fake_create_subprocess_exec(*a, **kw): return _FakeProc() monkeypatch.setattr( cp.asyncio, "create_subprocess_exec", _fake_create_subprocess_exec ) events: list[dict] = [] async def _collect(): async for ev in cp.stream_codex_device_login(): events.append(ev) asyncio.run(_collect()) log_lines = [ev.get("line", "") for ev in events if ev.get("type") == "log"] joined = "\n".join(log_lines) # Sensitive content must not have been forwarded. assert "refresh_token" not in joined, f"token leaked: {joined!r}" assert "rt_LEAK_LEAK_LEAK" not in joined assert "auth.json" not in joined, f"local config path leaked: {joined!r}" # The known-safe progress lines must be present so the UI can # show the user what is happening. assert any("Welcome to Codex" in line for line in log_lines) assert any("Successfully logged in" in line for line in log_lines) # device_url + device_code events must still fire. url_events = [ev for ev in events if ev.get("type") == "device_url"] code_events = [ev for ev in events if ev.get("type") == "device_code"] assert url_events and url_events[0]["url"].endswith("/codex/device") assert code_events and code_events[0]["code"] == "ABCD-EFGH" def test_parallel_usage_accounts_for_all_calls(self, monkeypatch): """The fan-out path runs N worker calls + 1 synthesis call. The reported usage must reflect that, not just one call's worth, otherwise the cost / context display is off by the fan-out factor. """ _install_fake_codex_sdk( monkeypatch, lambda: _FakeAsyncCodex( chunks = ["AAAAAAAAAA"], # 10 chars per tab final = "SYNTHESISED" * 10, # 110 chars synthesis ), ) from core.inference.codex_provider import stream_codex n = 4 long_prompt = "a" * 200 # 200 chars lines = _collect_stream( stream_codex( messages = [{"role": "user", "content": long_prompt}], model = "gpt-5.4", parallel_calls = n, ) ) chunks = _parse_sse_chunks(lines) usage_chunks = [c for c in chunks if c.get("choices") == [] and c.get("usage")] assert len(usage_chunks) == 1 usage = usage_chunks[0]["usage"] # Single-call prompt would be ~200/4 = 50 tokens. For n=4 with # synthesis, prompt should be much larger: n*200 + (n*10 + 200) # = 800 + 240 = 1040 chars ~= 260 tokens. assert ( usage["prompt_tokens"] >= 200 ), f"prompt_tokens not scaled for fan-out: {usage['prompt_tokens']}" # Completion = n*10 (tab outputs) + 110 (synthesis) = 150 chars # ~= 37 tokens. Definitely > the synthesis-only count of 27. assert ( usage["completion_tokens"] >= 30 ), f"completion_tokens not scaled for fan-out: {usage['completion_tokens']}" def test_buffered_result_none_final_does_not_emit_repr(self, monkeypatch): """A buffered TurnResult whose final_response is None must NOT send a Python object repr (``TurnResult(...)``) to the user. Returning an empty content chunk is the right shape: the stream still finishes with the usage + stop + [DONE] frames, but no garbage assistant text appears. """ class _ResultNoFinal: final_response = None # explicit None def __repr__(self): return "TurnResult(internal=should_not_leak)" class _ThreadBuffered: async def run(self, prompt): return _ResultNoFinal() class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): return _ThreadBuffered() _install_fake_codex_sdk(monkeypatch, _Async) from core.inference.codex_provider import stream_codex chunks: list[str] = [] async def _collect(): async for c in stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 1, ): chunks.append(c) asyncio.run(_collect()) body = "".join(chunks) assert ( "TurnResult" not in body ), f"Python object repr leaked to user content: {body!r}" assert "should_not_leak" not in body # Stream still terminated cleanly. assert "[DONE]" in body def test_empty_stream_falls_back_to_completed_agent_message(self, monkeypatch): """A successful turn that emits zero ``message.delta`` events but DOES emit a final ``ItemCompletedNotification`` with an agent message must surface that text. Without the fallback the Chat Completions reply would be empty even though Codex produced a complete answer. """ class _CompletedEvent: payload = { "type": "item.completed", "item": { "root": { "type": "agentMessage", "text": "final answer from completion", }, }, } class _Turn: async def stream(self): yield _CompletedEvent() class _ThreadEmptyDeltas: def turn(self, prompt): return _Turn() async def run(self, prompt): raise AssertionError( "must not fall through to buffered run() when " "the stream completes successfully" ) class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): return _ThreadEmptyDeltas() _install_fake_codex_sdk(monkeypatch, _Async) from core.inference.codex_provider import stream_codex chunks: list[str] = [] async def _collect(): async for c in stream_codex( messages = [{"role": "user", "content": "hi"}], model = "gpt-5.5", parallel_calls = 1, ): chunks.append(c) asyncio.run(_collect()) body = "".join(chunks) assert ( "final answer from completion" in body ), f"agent message text from completion event was dropped; body={body!r}" def test_synthesis_also_pins_safety_kwargs(self, monkeypatch): """The synthesis turn that unifies parallel fan-out outputs must use the same safety pins -- a fan-out tab could otherwise sneak an unsafe approval into the final synthesis prompt. """ seen_kwargs: list[dict] = [] class _SynThread: async def run(self, prompt): return "synth ok" class _Async: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def thread_start(self, **kw): seen_kwargs.append(dict(kw)) return _SynThread() import importlib.util as _iu fake_mod = types.ModuleType("openai_codex") fake_mod.AsyncCodex = _Async # type: ignore[attr-defined] fake_mod.ApprovalMode = types.SimpleNamespace( # type: ignore[attr-defined] deny_all = "DENY_ALL_SENTINEL", ) fake_mod.SandboxMode = types.SimpleNamespace( # type: ignore[attr-defined] read_only = "READ_ONLY_SENTINEL", ) monkeypatch.setitem(sys.modules, "openai_codex", fake_mod) real_find_spec = _iu.find_spec monkeypatch.setattr( "importlib.util.find_spec", lambda n, *a, **kw: ( types.SimpleNamespace() if n in ("openai_codex", "codex_app_server") else real_find_spec(n, *a, **kw) ), ) from core.inference.codex_provider import _run_codex_synthesis asyncio.run( _run_codex_synthesis( model = "gpt-5.5", system = "Always answer in Spanish.", prompt = "What is the capital of France?", tab_outputs = ["Paris", "Paris."], ) ) assert seen_kwargs, "synthesis thread_start never called" kw = seen_kwargs[0] assert kw.get("approval_mode") == "DENY_ALL_SENTINEL" assert kw.get("sandbox") == "READ_ONLY_SENTINEL" async def _consume_first(gen): """Drive an async generator until it raises or yields its first value. Used to surface lazy-import errors that fire on the first SDK touch -- otherwise the generator would swallow them on ``__aiter__`` and the test couldn't see them. """ async for _ in gen: return # ── Round 7: _ScrubbedEnvAsyncCodex cross-wrapper concurrency ────── class TestScrubbedEnvConcurrency: """Reproduce the cross-wrapper concurrency hole the round 7 review surfaced and lock in the fix: when wrapper B enters AFTER wrapper A has already deleted ``HF_TOKEN`` from ``os.environ``, B must still increment the refcount for that key so A's exit does not restore the secret while B is mid-session. """ def test_overlapping_wrappers_keep_keys_scrubbed_until_last_release( self, monkeypatch ): import os from core.inference.codex_provider import ( _SCRUBBED_ENV_REFCOUNT, _ScrubbedEnvAsyncCodex, ) # Reset module-level state in case prior tests left residue. _SCRUBBED_ENV_REFCOUNT.clear() # _SCRUBBED_ENV_ORIGINALS is the round 7 fix's shared snapshot # store; older codex_provider builds tracked originals per- # instance under _restored_via_us. Reset whichever store the # current build exposes so prior tests cannot leak state in. from core.inference import codex_provider as _cp _orig = getattr(_cp, "_SCRUBBED_ENV_ORIGINALS", None) if isinstance(_orig, dict): _orig.clear() monkeypatch.setenv("HF_TOKEN", "sekret-hf") monkeypatch.setenv("GH_TOKEN", "sekret-gh") # Keys NOT on the safe-list end up in _codex_sdk_env_override(). class _FakeInner: async def __aenter__(self_inner): return self_inner async def __aexit__(self_inner, *a): return False def _fake_async_codex(): return _FakeInner() async def scenario(): wrapper_a = _ScrubbedEnvAsyncCodex(_fake_async_codex) wrapper_b = _ScrubbedEnvAsyncCodex(_fake_async_codex) # Wrapper A enters first and scrubs both secrets. await wrapper_a.__aenter__() assert "HF_TOKEN" not in os.environ assert "GH_TOKEN" not in os.environ # Wrapper B enters while A is still active. Even though # os.environ no longer contains HF_TOKEN/GH_TOKEN (A already # deleted them), B must pick them up from the live refcount # table so A's later exit does not restore them prematurely. await wrapper_b.__aenter__() assert _SCRUBBED_ENV_REFCOUNT.get("HF_TOKEN") == 2 assert _SCRUBBED_ENV_REFCOUNT.get("GH_TOKEN") == 2 # A exits first -- B is still active so the keys MUST remain # absent from os.environ. await wrapper_a.__aexit__(None, None, None) assert "HF_TOKEN" not in os.environ, ( "HF_TOKEN leaked back into os.environ while wrapper B " "is still active" ) assert "GH_TOKEN" not in os.environ assert _SCRUBBED_ENV_REFCOUNT.get("HF_TOKEN") == 1 assert _SCRUBBED_ENV_REFCOUNT.get("GH_TOKEN") == 1 # B exits -- now the keys must be restored from the saved # originals. await wrapper_b.__aexit__(None, None, None) assert os.environ.get("HF_TOKEN") == "sekret-hf" assert os.environ.get("GH_TOKEN") == "sekret-gh" assert "HF_TOKEN" not in _SCRUBBED_ENV_REFCOUNT asyncio.run(scenario()) # ── Round 7: device-auth URL allowlisting ─────────────────────────── class TestDeviceUrlAllowlist: """Lock in the device-auth URL allowlist: only `auth.openai.com` and `chatgpt.com` over https are accepted as `device_url` events. A shimmed codex earlier on PATH could otherwise print `https://evil.example/activate?code=ABCD` and Studio would render a phishing CTA. """ def test_known_good_urls_allowed(self): from core.inference.codex_provider import _is_allowed_device_url assert _is_allowed_device_url( "https://auth.openai.com/codex/device?user_code=ABCD-EFGH" ) assert _is_allowed_device_url( "https://chatgpt.com/activate?user_code=WXYZ-1234" ) def test_attacker_hosts_rejected(self): from core.inference.codex_provider import _is_allowed_device_url for evil in [ "https://evil.example/activate?code=ABCD", "https://auth-openai-com.evil.example/codex/device", "https://chatgpt.com.evil.example/activate", "https://login.openai.com/codex/device", ]: assert not _is_allowed_device_url(evil), evil def test_http_downgrade_rejected(self): from core.inference.codex_provider import _is_allowed_device_url assert not _is_allowed_device_url( "http://auth.openai.com/codex/device?user_code=ABCD-EFGH" ) def test_garbage_url_rejected(self): from core.inference.codex_provider import _is_allowed_device_url assert not _is_allowed_device_url("not a url") assert not _is_allowed_device_url("") assert not _is_allowed_device_url("javascript:alert(1)") # ── Round 7: tightened device-login log filter ────────────────────── class TestDeviceLoginLogFilter: """The login-output filter must not forward sensitive lines a malicious codex shim could print -- including 'Not logged in:' leaks that match the old loose 'logged in' substring test, plus refresh tokens, auth.json paths, and the codex config dir. """ def _safe_to_forward(self): # _safe_to_forward is defined inside stream_codex_device_login; # re-extracting it requires us to import it through the source # module path. Easier: replicate the production regex set in # the test directly so a regression in the source list is # caught when the production source is loaded. import importlib mod = importlib.reload(importlib.import_module("core.inference.codex_provider")) # Walk the source string to find the patterns; they live inside # the generator. Use a stable proxy: read the regex literals. import re src = open(mod.__file__).read() # Smoke check: the source has anchored regex (^) for the safe # phrases AND an unsafe-content blocklist. assert "safe_log_res" in src assert "unsafe_log_re" in src assert ( "not\\s+(?:logged|signed)\\s+in" in src or "not\\\\s+(?:logged|signed)\\\\s+in" in src ) return None def test_safe_log_source_has_anchored_patterns_and_blocklist(self): self._safe_to_forward() def test_blocklist_rejects_known_leaks(self): # Reconstruct the production regex set the same way stream_codex # _device_login does, then assert each attacker string is dropped. import re unsafe_log_re = re.compile( r"\bnot\s+(?:logged|signed)\s+in\b|" r"\bnot\s+authenticated\b|" r"refresh[_-]?token|access[_-]?token|" r"\bapi[_-]?key\b|\bsecret\b|" r"\bauth\.json\b|" r"/\.codex/|\\\.codex\\", re.IGNORECASE, ) for line in [ "Not logged in: refresh_token=rt_LEAK auth.json=/home/u/.codex/auth.json", "logged in (refresh_token=abc)", "Open this: https://auth.openai.com/codex/device but access_token=hunter2", "Logged in - secret=hunter2", "API_KEY=sk-x logged in", "Reading /home/u/.codex/auth.json", ]: assert unsafe_log_re.search(line), f"line should match unsafe: {line!r}" def test_safe_phrases_pass_when_clean(self): import re safe_log_res = ( re.compile(r"^welcome to codex\b", re.IGNORECASE), re.compile(r"^initializing\b", re.IGNORECASE), re.compile(r"^open (?:this|the verification)", re.IGNORECASE), re.compile(r"^open:\s*https?://", re.IGNORECASE), re.compile(r"^enter (?:this one-time code|the code)\b", re.IGNORECASE), re.compile(r"^waiting\b", re.IGNORECASE), re.compile(r"^successfully (?:logged|signed) in\b", re.IGNORECASE), re.compile(r"^(?:logged|signed) in\b", re.IGNORECASE), re.compile(r"^browser opened\b", re.IGNORECASE), re.compile(r"^press ctrl", re.IGNORECASE), ) for clean in [ "Welcome to codex", "Initializing device auth...", "Open this URL: https://auth.openai.com/codex/device", "Open: https://auth.openai.com/codex/device", "Enter this one-time code:", "Waiting for authentication...", "Successfully logged in", "Logged in using ChatGPT", "Browser opened", "Press Ctrl+C to cancel", ]: assert any( pat.search(clean) for pat in safe_log_res ), f"clean line should match safe: {clean!r}" # ── Round 8: stream replay protection on non-visible events ────────── class TestStreamReplayProtection: """Lock in the round 8 fix: a turn that fired non-rendered events (command/file/tool deltas) before crashing MUST NOT replay via the buffered `thread.run(prompt)` fallback even though no visible text was yielded. The earlier guard only tracked `emitted_any` (visible text), missing the case where shell commands or file writes already happened upstream. """ def test_buffered_run_not_called_after_non_visible_event_crash(self): """Stream raises after a tool event with no visible text. The buffered ``thread.run`` MUST NOT be called -- the Codex turn has already started running side-effects upstream. """ from core.inference.codex_provider import _stream_thread_run class _Stream: def __init__(self): self._i = 0 def __aiter__(self): return self async def __anext__(self): self._i += 1 if self._i == 1: # An event with no answer text -- _coerce_text # returns "" but the turn has demonstrably run. return {"type": "command.delta", "delta": "rm -rf"} raise RuntimeError("upstream stream died mid-turn") class _Turn: def stream(self): return _Stream() class _Thread: run_calls: int = 0 def turn(self_inner, prompt): return _Turn() async def run(self_inner, prompt): self_inner.run_calls += 1 return "REPLAY-WOULD-RETURN-THIS" thread = _Thread() async def collect(): chunks = [] async for c in _stream_thread_run(thread, "hello"): chunks.append(c) return chunks chunks = asyncio.run(collect()) # No visible text was emitted (the only event was filtered), # but thread.run MUST NOT have been called because the turn # already started. assert thread.run_calls == 0, ( "thread.run was called after a partial-turn crash; this " "would replay shell commands / file writes" ) assert chunks == [] def test_buffered_run_called_when_no_streaming_helper(self): """Threads that expose neither .turn nor .run_streaming still fall through to the buffered .run -- that is the ONLY path the buffered fallback is allowed to execute. """ from core.inference.codex_provider import _stream_thread_run class _Thread: run_calls: int = 0 async def run(self_inner, prompt): self_inner.run_calls += 1 return "answer" thread = _Thread() async def collect(): chunks = [] async for c in _stream_thread_run(thread, "hello"): chunks.append(c) return chunks chunks = asyncio.run(collect()) assert thread.run_calls == 1 assert chunks == ["answer"] # ── Round 9: P2 fixes from latest Codex bot review ────────────────── class TestRunStreamingCompletionFallback: """Round 9 fix: legacy SDK exposes ``thread.run_streaming`` but the stream only emits completion-style events (no message deltas). We must still emit the agentMessage text, otherwise the request returns 200 with an empty assistant reply. """ def test_run_streaming_only_completion_emits_final_text(self): from core.inference.codex_provider import _stream_thread_run # Dict shape matching _completed_agent_message_text's accepted # form: type=thread.item.completed, item.type=agentMessage, # item.text=. The legacy run_streaming path now # extracts ``.payload`` first (matching the canonical path), so # a plain dict event is the simplest faithful fixture. completed_event = { "type": "thread.item.completed", "item": {"type": "agentMessage", "text": "FINAL_ANSWER"}, } class _Stream: def __init__(self): self._sent = False def __aiter__(self): return self async def __anext__(self): if self._sent: raise StopAsyncIteration self._sent = True return completed_event class _Thread: # No .turn so the canonical path is skipped; only legacy # run_streaming exists, and it yields a completion event # with no streaming deltas. def run_streaming(self_inner, prompt): return _Stream() async def run(self_inner, prompt): # pragma: no cover # Should never be called -- the legacy stream # completed cleanly via the completion event. raise AssertionError("buffered run must not fire") thread = _Thread() async def collect(): return [c async for c in _stream_thread_run(thread, "hi")] chunks = asyncio.run(collect()) assert chunks == ["FINAL_ANSWER"], chunks class TestParallelSetupErrorPropagation: """Round 9 fix: when CodexUnavailableError takes out every worker in a parallel-calls fan-out, the function re-raises so the route layer can return a proper 503. Per-tab runtime failures (timeout etc.) still get swallowed into codex_tab_error events as before. """ def test_unavailable_in_every_worker_reraises(self, monkeypatch): from core.inference import codex_provider as cp # Force _import_codex to raise CodexUnavailableError. Every # worker hits this on entry so per_tab_texts stays empty and # the function MUST re-raise. def boom(): raise cp.CodexUnavailableError("SDK not installed (test)") monkeypatch.setattr(cp, "_import_codex", boom) async def collect_lines(): lines = [] try: async for line in cp._stream_codex_parallel( model = "gpt-5.4-mini", system = "", prompt = "hello", n = 3, completion_id = "test-completion", ): lines.append(line) except cp.CodexUnavailableError as exc: return lines, exc return lines, None lines, exc = asyncio.run(collect_lines()) assert exc is not None, ( "CodexUnavailableError did not propagate -- the stream " "returned a 200 with no assistant content" ) assert "SDK not installed (test)" in str(exc) class TestCodexDoneSentinelExactMatch: """Round 9 fix: the Codex SSE wrapper's `sent_done` detection now requires an EXACT `data: [DONE]` line match. The substring check was firing on `delta.content` payloads that happened to contain the literal text `[DONE]`. The route source is the canonical reference -- this test asserts the source uses an anchored comparison, not a substring `in` check, so the fix is locked in even if the route is restructured. """ def test_route_uses_exact_done_match(self): with open(_backend_file("routes/inference.py"), "r", encoding = "utf-8") as f: src = f.read() # The Codex SSE wrapper is the only place we expect this # comparison style; allow either single or double quotes # around the canonical line for forward compatibility. assert ( 'line.strip() == "data: [DONE]"' in src or "line.strip() == 'data: [DONE]'" in src ), ( "Codex SSE wrapper must terminate on an exact `data: [DONE]` " "line, not on a substring containing `[DONE]`." ) # Inspect the Codex stream block specifically. The old # substring check `if "[DONE]" in line: sent_done = True` # must NOT appear as an active comparison. Ignore matches # inside comments (lines starting with `#` or inside string # literals describing the old behavior) by scanning for the # exact statement form. codex_block_start = src.find("async def _codex_stream():") if codex_block_start != -1: window = src[codex_block_start : codex_block_start + 4000] for line in window.splitlines(): stripped = line.strip() if stripped.startswith("#"): continue assert 'if "[DONE]" in line' not in stripped, ( "Codex SSE wrapper still uses substring [DONE] check: " + stripped ) class TestCodexSpoofModule: """Ensure the in-process Codex SDK spoof installs cleanly under the env flag, satisfies _import_codex, exposes the SDK surface the provider uses, and streams deterministic per-tab text. The spoof is the credit-free path that the rest of the test suite (and the UI demo) rides on, so a regression here would silently break every downstream consumer. """ def test_install_swaps_in_module_when_flag_set(self, monkeypatch): # Ensure clean import state. import sys for name in ("openai_codex", "codex_app_server"): sys.modules.pop(name, None) from core.inference import codex_spoof monkeypatch.setenv(codex_spoof.SPOOF_ENV_VAR, "1") assert codex_spoof.is_spoof_enabled() codex_spoof.install_as_openai_codex() import openai_codex # type: ignore[import-not-found] assert getattr(openai_codex, "__spoof__", False) is True assert hasattr(openai_codex, "AsyncCodex") assert hasattr(openai_codex, "AppServerConfig") assert openai_codex.ApprovalMode.deny_all assert openai_codex.SandboxMode.read_only def test_flag_disabled_does_not_install(self, monkeypatch): import sys for name in ("openai_codex", "codex_app_server"): sys.modules.pop(name, None) from core.inference import codex_spoof monkeypatch.delenv(codex_spoof.SPOOF_ENV_VAR, raising = False) assert not codex_spoof.is_spoof_enabled() def test_spoof_stream_emits_deltas_and_completion(self): import asyncio from core.inference import codex_spoof async def run(): codex = codex_spoof.AsyncCodex(config = codex_spoof.AppServerConfig(env = {})) thread = await codex.thread_start( model = "gpt-5.4-mini", base_instructions = None, approval_mode = codex_spoof.ApprovalMode.deny_all, sandbox = codex_spoof.SandboxMode.read_only, ) events = [] async for ev in thread.turn("hello").stream(): events.append(ev) return events events = asyncio.run(run()) deltas = [ e for e in events if isinstance(e, dict) and e.get("type") == "message.delta" ] assert len(deltas) >= 3, "spoof should stream multiple deltas" last = events[-1] assert type(last).__name__ == "_ItemCompletedNotification" # Must carry the worker tag when no [tab N] marker is in the system prompt. full_text = last.item.root.text assert "spoof reply from gpt-5.4-mini" in full_text assert "no upstream tokens" in full_text def test_spoof_tags_per_tab(self): import asyncio from core.inference import codex_spoof async def reply_for_tab(idx: int) -> str: codex = codex_spoof.AsyncCodex() thread = await codex.thread_start( model = "gpt-5.4-mini", base_instructions = f"[tab {idx}/3]", ) result = await thread.run("explain LoRA") return result.text async def _gather(): return await asyncio.gather( reply_for_tab(1), reply_for_tab(2), reply_for_tab(3), ) tab_replies = asyncio.run(_gather()) # Each tab must mention its own worker index, so the UI tabs # show visibly distinct text when clicked. for i, reply in enumerate(tab_replies, start = 1): assert f"worker {i}" in reply, f"tab {i} missing its tag: {reply}" def test_provider_picks_up_spoof_via_import(self, monkeypatch): import sys for name in ("openai_codex", "codex_app_server"): sys.modules.pop(name, None) from core.inference import codex_provider, codex_spoof monkeypatch.setenv(codex_spoof.SPOOF_ENV_VAR, "1") mod = codex_provider._import_codex() assert getattr(mod, "__spoof__", False) is True def test_safety_kwargs_resolve_against_spoof(self, monkeypatch): import sys for name in ("openai_codex", "codex_app_server"): sys.modules.pop(name, None) from core.inference import codex_provider, codex_spoof monkeypatch.setenv(codex_spoof.SPOOF_ENV_VAR, "1") codex_provider._import_codex() # ensures install kwargs = codex_provider._safe_thread_safety_kwargs() # Spoof exports ApprovalMode.deny_all + SandboxMode.read_only, # so the provider must be able to pin both without falling # through to the unsafe-defaults gate. assert kwargs, "safety kwargs missing -- provider would fail closed" assert str(kwargs["approval_mode"]).endswith("deny_all") assert str(kwargs["sandbox"]).endswith("read_only")