Use UTF-8 for Python code-execution subprocess I/O (#6489 class) (#6548)

* Use UTF-8 for Python code-execution subprocess I/O

Studio's code-execution tool already tells the child to emit UTF-8
(PYTHONIOENCODING=utf-8 in _build_safe_env), but _python_exec writes the
temp script and decodes the subprocess pipe with the OS default codec.
On Windows (cp1252), non-ASCII in model-written code or its output --
arrows, CJK, emoji -- raises UnicodeEncodeError / UnicodeDecodeError and
breaks execution.

Complete the UTF-8 wiring in core/inference/tools.py:
- write the temp script with encoding="utf-8"
- decode _python_exec stdout as utf-8, errors="replace"
- set PYTHONIOENCODING=utf-8 in _build_bypass_env too (matches
  _build_safe_env, so the bypass path's child also emits utf-8)

The child is python with PYTHONIOENCODING=utf-8, so it emits UTF-8
regardless of the console code page and the decode is always correct.
Shell execution via cmd.exe has a separate console-code-page story and
is left to a follow-up.

Refs unslothai/unsloth#6489

* Scope Python exec UTF-8 env to Python tool

* Make bash bypass test robust to a host-set PYTHONIOENCODING for PR #6548

Bypass mode preserves benign host env vars, so a host-set PYTHONIOENCODING was
inherited into the bash bypass env and tripped the new assertion even though
_bash_exec never adds it. Clear it in the test so the assertion checks _bash_exec,
not the runner environment.

---------

Co-authored-by: Lee Jackson <130007945+Imagineer99@users.noreply.github.com>
Co-authored-by: Daniel Han <danielhanchen@gmail.com>
This commit is contained in:
Saicharan Ramineni 2026-06-22 12:06:03 -04:00 committed by GitHub
commit 7ecbf5a770
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 49 additions and 2 deletions

View file

@ -2545,14 +2545,24 @@ def _python_exec(
pass
try:
fd, tmp_path = tempfile.mkstemp(suffix = ".py", prefix = "studio_exec_", dir = workdir)
with os.fdopen(fd, "w") as f:
# utf-8 so non-ASCII in model-written code survives the OS default codec
# (Windows cp1252 would otherwise raise UnicodeEncodeError).
with os.fdopen(fd, "w", encoding = "utf-8") as f:
f.write(code)
safe_env = _build_bypass_env(workdir) if disable_sandbox else _build_safe_env(workdir)
if disable_sandbox:
# Match the sandboxed Python path without changing bypass shell I/O.
safe_env = dict(safe_env)
safe_env["PYTHONIOENCODING"] = "utf-8"
popen_kwargs = dict(
stdout = subprocess.PIPE,
stderr = subprocess.STDOUT,
text = True,
# Decode child output as utf-8 (it emits utf-8 via PYTHONIOENCODING);
# replace so non-ASCII output never crashes the read on Windows.
encoding = "utf-8",
errors = "replace",
cwd = workdir,
env = safe_env,
)

View file

@ -135,6 +135,7 @@ def test_python_bypass_uses_bypass_preexec_and_bypass_env(captured_popen, monkey
assert captured_popen["kwargs"]["preexec_fn"] is tools._bypass_preexec
env = captured_popen["kwargs"]["env"]
assert env.get("HOSTVAR") == "benign-xyz"
assert env.get("PYTHONIOENCODING") == "utf-8"
assert "HF_TOKEN" not in env
@ -151,9 +152,12 @@ def test_bash_blocklist_skipped_when_bypassed(captured_popen):
@_POSIX_ONLY
def test_bash_bypass_uses_bypass_preexec(captured_popen):
def test_bash_bypass_uses_bypass_preexec(captured_popen, monkeypatch):
# bypass inherits benign host vars; clear so we assert _bash_exec adds none.
monkeypatch.delenv("PYTHONIOENCODING", raising = False)
_bash_exec("echo hi", None, 5, "t", disable_sandbox = True)
assert captured_popen["kwargs"]["preexec_fn"] is tools._bypass_preexec
assert "PYTHONIOENCODING" not in captured_popen["kwargs"]["env"]
# ── real end-to-end python execution under bypass ───────────────────

View file

@ -0,0 +1,33 @@
# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved.
"""_python_exec must round-trip non-ASCII output end to end.
Model-written code routinely contains non-ASCII (arrows, CJK, emoji). The temp
script and the child's stdout pipe both have to be UTF-8 or it crashes/garbles
on Windows, whose default codec is cp1252. Mirrors the report in
unslothai/unsloth#6489. The child is ``python`` with PYTHONIOENCODING=utf-8, so
it emits UTF-8 on every OS; this proves the round-trip on a UTF-8 host and
guards against a regression to the OS default codec.
"""
import sys
from pathlib import Path
import pytest
_BACKEND_ROOT = Path(__file__).resolve().parents[1]
if str(_BACKEND_ROOT) not in sys.path:
sys.path.insert(0, str(_BACKEND_ROOT))
from core.inference.tools import _python_exec
# Arrow, em-dash, accent, CJK, check mark, astral-plane emoji -- none encodable
# in cp1252, so the OS default codec would raise on write or read.
_UNICODE = "café — 数字 → ✓ 😀"
@pytest.mark.parametrize("disable_sandbox", [False, True])
def test_python_exec_round_trips_non_ascii(disable_sandbox):
out = _python_exec(f"print({_UNICODE!r})", disable_sandbox = disable_sandbox)
assert _UNICODE in out, repr(out)