diff --git a/install.ps1 b/install.ps1
index b26566cc3d..3911236d87 100644
--- a/install.ps1
+++ b/install.ps1
@@ -1300,15 +1300,11 @@ shell.Run cmd, 0, False
if ($SkipTorch) {
# No-torch: install unsloth + unsloth-zoo with --no-deps, then
# runtime deps (typer, safetensors, transformers, etc.) with --no-deps.
- $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --no-deps --reinstall-package unsloth --reinstall-package unsloth-zoo "unsloth>=2026.5.6" unsloth-zoo }
+ $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --no-deps --reinstall-package unsloth --reinstall-package unsloth-zoo "unsloth>=2026.5.7" unsloth-zoo }
if ($baseInstallExit -eq 0) {
- # Install pydantic WITH deps so pip pins pydantic-core to
- # the exact version pydantic's metadata requires. The
- # --no-deps install of no-torch-runtime.txt below would
- # otherwise pick the latest of each independently and
- # trip pydantic's _ensure_pydantic_core_version check.
- # pydantic's deps (annotated-types, pydantic-core,
- # typing-extensions, typing-inspection) are torch-free.
+ # Resolve pydantic WITH deps so pip pins pydantic-core
+ # to the matching version (no-torch-runtime.txt below
+ # is --no-deps). All transitive deps are torch-free.
$baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython pydantic }
}
if ($baseInstallExit -eq 0) {
@@ -1318,7 +1314,7 @@ shell.Run cmd, 0, False
}
}
} else {
- $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --reinstall-package unsloth --reinstall-package unsloth-zoo "unsloth>=2026.5.6" unsloth-zoo }
+ $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --reinstall-package unsloth --reinstall-package unsloth-zoo "unsloth>=2026.5.7" unsloth-zoo }
}
if ($baseInstallExit -ne 0) {
Write-Host "[ERROR] Failed to install unsloth (exit code $baseInstallExit)" -ForegroundColor Red
@@ -1356,10 +1352,9 @@ shell.Run cmd, 0, False
if ($SkipTorch) {
# No-torch: install unsloth + unsloth-zoo with --no-deps, then
# runtime deps (typer, safetensors, transformers, etc.) with --no-deps.
- $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --no-deps --upgrade-package unsloth --upgrade-package unsloth-zoo "unsloth>=2026.5.6" unsloth-zoo }
+ $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --no-deps --upgrade-package unsloth --upgrade-package unsloth-zoo "unsloth>=2026.5.7" unsloth-zoo }
if ($baseInstallExit -eq 0) {
- # Install pydantic WITH deps so pip pins pydantic-core to
- # the matching version (see migrated branch above).
+ # Same pydantic-with-deps trick as the migrated branch.
$baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython pydantic }
}
if ($baseInstallExit -eq 0) {
@@ -1369,7 +1364,7 @@ shell.Run cmd, 0, False
}
}
} elseif ($StudioLocalInstall) {
- $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --upgrade-package unsloth "unsloth>=2026.5.6" unsloth-zoo }
+ $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --upgrade-package unsloth "unsloth>=2026.5.7" unsloth-zoo }
} else {
$baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --upgrade-package unsloth -- "$PackageName" }
}
@@ -1397,7 +1392,7 @@ shell.Run cmd, 0, False
Write-TauriLog "STEP" "Installing unsloth"
substep "installing unsloth (this may take a few minutes)..."
if ($StudioLocalInstall) {
- $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython unsloth-zoo "unsloth>=2026.5.6" --torch-backend=auto }
+ $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython unsloth-zoo "unsloth>=2026.5.7" --torch-backend=auto }
if ($baseInstallExit -ne 0) {
Write-Host "[ERROR] Failed to install unsloth (exit code $baseInstallExit)" -ForegroundColor Red
return (Exit-InstallFailure "Failed to install unsloth (exit code $baseInstallExit)" $baseInstallExit)
diff --git a/install.sh b/install.sh
index 9bdd935171..cc92fd52c2 100755
--- a/install.sh
+++ b/install.sh
@@ -1865,14 +1865,10 @@ if [ "$_MIGRATED" = true ]; then
# to prevent transitive torch resolution.
run_install_cmd "install unsloth (migrated no-torch)" uv pip install --python "$_VENV_PY" --no-deps \
--reinstall-package unsloth --reinstall-package unsloth-zoo \
- "unsloth>=2026.5.6" unsloth-zoo
- # Install pydantic WITH deps so pip pins pydantic-core to the
- # exact version pydantic's own metadata requires. The --no-deps
- # install below would otherwise pick the latest of each
- # independently and trip pydantic's _ensure_pydantic_core_version
- # check on the next import. pydantic's deps (annotated-types,
- # pydantic-core, typing-extensions, typing-inspection) are
- # torch-free, so this is safe on the no-torch path.
+ "unsloth>=2026.5.7" unsloth-zoo
+ # Resolve pydantic WITH deps so pip pins pydantic-core to the
+ # matching version (no-torch-runtime.txt below is --no-deps).
+ # All transitive deps are torch-free.
run_install_cmd "install pydantic (with deps for compatible core)" \
uv pip install --python "$_VENV_PY" pydantic
_NO_TORCH_RT="$(_find_no_torch_runtime)"
@@ -1882,7 +1878,7 @@ if [ "$_MIGRATED" = true ]; then
else
run_install_cmd "install unsloth (migrated)" uv pip install --python "$_VENV_PY" \
--reinstall-package unsloth --reinstall-package unsloth-zoo \
- "unsloth>=2026.5.6" unsloth-zoo
+ "unsloth>=2026.5.7" unsloth-zoo
fi
if [ "$STUDIO_LOCAL_INSTALL" = true ]; then
substep "overlaying local repo (editable)..."
@@ -2050,9 +2046,8 @@ elif [ -n "$TORCH_INDEX_URL" ]; then
# runtime deps (typer, safetensors, transformers, etc.) with --no-deps.
run_install_cmd "install unsloth (no-torch)" uv pip install --python "$_VENV_PY" --no-deps \
--upgrade-package unsloth --upgrade-package unsloth-zoo \
- "unsloth>=2026.5.6" unsloth-zoo
- # Install pydantic WITH deps so pip pins pydantic-core to the
- # exact version pydantic requires (see migrated branch above).
+ "unsloth>=2026.5.7" unsloth-zoo
+ # Same pydantic-with-deps trick as the migrated branch.
run_install_cmd "install pydantic (with deps for compatible core)" \
uv pip install --python "$_VENV_PY" pydantic
_NO_TORCH_RT="$(_find_no_torch_runtime)"
@@ -2069,7 +2064,7 @@ elif [ -n "$TORCH_INDEX_URL" ]; then
fi
elif [ "$STUDIO_LOCAL_INSTALL" = true ]; then
run_install_cmd "install unsloth (local)" uv pip install --python "$_VENV_PY" \
- --upgrade-package unsloth "unsloth>=2026.5.6" unsloth-zoo
+ --upgrade-package unsloth "unsloth>=2026.5.7" unsloth-zoo
substep "overlaying local repo (editable)..."
run_install_cmd "overlay local repo" uv pip install --python "$_VENV_PY" -e "$_REPO_ROOT" --no-deps
substep "overlaying unsloth-zoo from git main..."
@@ -2101,7 +2096,7 @@ else
tauri_log "STEP" "Installing Unsloth"
substep "installing unsloth (this may take a few minutes)..."
if [ "$STUDIO_LOCAL_INSTALL" = true ]; then
- run_install_cmd "install unsloth (auto torch backend)" uv pip install --python "$_VENV_PY" unsloth-zoo "unsloth>=2026.5.6" --torch-backend=auto
+ run_install_cmd "install unsloth (auto torch backend)" uv pip install --python "$_VENV_PY" unsloth-zoo "unsloth>=2026.5.7" --torch-backend=auto
substep "overlaying local repo (editable)..."
run_install_cmd "overlay local repo" uv pip install --python "$_VENV_PY" -e "$_REPO_ROOT" --no-deps
substep "overlaying unsloth-zoo from git main..."
diff --git a/studio/backend/core/export/export.py b/studio/backend/core/export/export.py
index 4ab95d896f..7cabd382eb 100644
--- a/studio/backend/core/export/export.py
+++ b/studio/backend/core/export/export.py
@@ -475,6 +475,7 @@ class ExportBackend:
self.current_model.save_pretrained_merged(
save_directory,
self.current_tokenizer,
+ save_method = "merged_16bit",
)
else:
self.current_model.save_pretrained(save_directory)
@@ -510,6 +511,7 @@ class ExportBackend:
self.current_model.save_pretrained_merged(
tmp_dir,
self.current_tokenizer,
+ save_method = "merged_16bit",
)
self.current_model.push_to_hub_merged(
repo_id,
diff --git a/studio/backend/requirements/no-torch-runtime.txt b/studio/backend/requirements/no-torch-runtime.txt
index c33ebf4d94..85294114b1 100644
--- a/studio/backend/requirements/no-torch-runtime.txt
+++ b/studio/backend/requirements/no-torch-runtime.txt
@@ -22,19 +22,11 @@ rich>=13.0
markdown-it-py>=3.0
mdurl>=0.1
pygments>=2.0
-# pydantic is intentionally NOT installed via this --no-deps file.
-# install.sh / install.ps1 / install_python_stack.py run a separate
-# `pip install pydantic` (with deps) just before this file is
-# applied, so pip resolves `pydantic-core` to the exact version
-# pydantic's `_ensure_pydantic_core_version` check expects. Listing
-# pydantic + pydantic-core unpinned here and resolving them under
-# --no-deps used to pick the latest of each independently and trip
-# `SystemError: pydantic-core 2.X.Y is incompatible with the current
-# pydantic version` on the first import (Windows fresh-venv repro
-# was the canonical case). pydantic's transitive deps
-# (annotated-types, pydantic-core, typing-extensions,
-# typing-inspection) are torch-free, so installing it WITH deps
-# does not pull torch.
+# pydantic is intentionally NOT pinned here. install.sh / install.ps1
+# / install_python_stack.py run `pip install pydantic` WITH deps just
+# before this --no-deps file is applied, so pip resolves pydantic-core
+# to the exact version pydantic's _ensure_pydantic_core_version check
+# expects. Pinning both under --no-deps used to drift them apart.
pyyaml
nest-asyncio
diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py
index 02270ab405..bf92055929 100644
--- a/studio/backend/routes/inference.py
+++ b/studio/backend/routes/inference.py
@@ -427,9 +427,17 @@ _TOOL_ACTION_NUDGE = (
" Do NOT output code blocks -- use the python tool instead."
)
-# Regex for stripping leaked tool-call XML from assistant messages/stream
+# Strip tool-call XML the speculative buffer in core/inference/llama_cpp.py
+# split across the visible/DRAIN boundary. Four leak shapes:
+# 1. well-formed `...` / `...`
+# 2. orphan opening to EOF (close was DRAINED)
+# 3. bare orphan close (open was DRAINED)
+# 4. tail-only `` (outer close truncated by EOS); anchored to
+# `\Z` so mid-text `` in user code samples survives.
_TOOL_XML_RE = _re.compile(
- r".*?|.*?",
+ r"<(?:tool_call|function=\w+)>.*?(?:(?:tool_call|function)>|\Z)"
+ r"|(?:tool_call|function)>"
+ r"|\s*\Z",
_re.DOTALL,
)
logger = get_logger(__name__)
diff --git a/studio/backend/tests/test_tool_xml_strip.py b/studio/backend/tests/test_tool_xml_strip.py
new file mode 100644
index 0000000000..8b90a46d5a
--- /dev/null
+++ b/studio/backend/tests/test_tool_xml_strip.py
@@ -0,0 +1,263 @@
+# SPDX-License-Identifier: AGPL-3.0-only
+# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+
+"""Tests for `_TOOL_XML_RE` (routes/inference.py) -- strips tool-call
+XML that leaks past the speculative buffer in core/inference/llama_cpp.py
+when the open/close pair is split across the visible/DRAIN boundary.
+"""
+
+from __future__ import annotations
+
+import sys
+import types as _types
+from pathlib import Path
+
+import pytest
+
+_BACKEND_DIR = str(Path(__file__).resolve().parent.parent)
+if _BACKEND_DIR not in sys.path:
+ sys.path.insert(0, _BACKEND_DIR)
+
+# Extract the regex from source (routes module needs heavy stubbing to import).
+import re as _re
+
+_src = (Path(_BACKEND_DIR) / "routes" / "inference.py").read_text()
+_m = _re.search(r"_TOOL_XML_RE = _re\.compile\((.*?)\n\)", _src, _re.DOTALL)
+assert _m, "could not extract _TOOL_XML_RE source"
+_ns = {"_re": _re}
+exec(f"_TOOL_XML_RE = _re.compile({_m.group(1)})", _ns)
+_TOOL_XML_RE = _ns["_TOOL_XML_RE"]
+
+
+# ── Well-formed pairs ─────────────────────────────────────────────
+
+
+def test_strips_well_formed_tool_call():
+ text = (
+ "Let me search.\n"
+ "\n"
+ "\n"
+ "\nBillboard 2015\n\n"
+ "\n"
+ "\n"
+ "Here are the songs:"
+ )
+ cleaned = _TOOL_XML_RE.sub("", text)
+ assert "" not in cleaned
+ assert "" not in cleaned
+ assert "" not in cleaned
+ assert "Here are the songs:" in cleaned, "non-XML content must survive"
+ assert "Let me search." in cleaned
+
+
+def test_strips_function_only_well_formed():
+ text = "Setup.\n\n\nprint(1)\n\n\nDone."
+ cleaned = _TOOL_XML_RE.sub("", text)
+ assert ""
+ "\n"
+ "\n"
+ "\nBillboard 2015\n\n"
+ "" not in cleaned
+ assert "\n\nprint(1)\n"
+ )
+ cleaned = _TOOL_XML_RE.sub("", text)
+ assert "")
+ assert "" not in cleaned
+ assert "Search starting." in cleaned
+
+
+def test_strips_multiple_orphans():
+ text = (
+ "First call:\n\n\n\nx=1\n"
+ "Second call:\n\n\nhi\n"
+ )
+ cleaned = _TOOL_XML_RE.sub("", text)
+ assert "" not in cleaned
+ assert "" not in cleaned
+ assert "" not in cleaned
+ # Mid-string intentionally preserved (see preserve test).
+
+
+# ── Tail-only (PR #5735 follow-up) ───────────────────
+
+
+def test_strips_tail_only_parameter_orphan():
+ # Outer truncated by EOS, inner DRAINED.
+ cleaned = _TOOL_XML_RE.sub("", "and the text is not readable.\n\n\n")
+ assert "" not in cleaned
+ assert "and the text is not readable." in cleaned
+
+
+def test_strips_tail_only_parameter_orphan_single_newline():
+ cleaned = _TOOL_XML_RE.sub("", "Global Economic Prospects\n\n")
+ assert "" not in cleaned
+ assert "Global Economic Prospects" in cleaned
+
+
+def test_strips_tail_only_parameter_orphan_no_trailing_ws():
+ cleaned = _TOOL_XML_RE.sub("", "Final answer.")
+ assert "" not in cleaned
+ assert "Final answer." in cleaned
+
+
+def test_preserves_mid_string_parameter_in_code_sample():
+ # Tail-anchor on `` is required so doc/example prose survives.
+ text = (
+ "Here is the Qwen tool-call format:\n"
+ "```xml\n"
+ "value\n"
+ "```\n"
+ "Note the closing sits inside ."
+ )
+ cleaned = _TOOL_XML_RE.sub("", text)
+ assert "Note the closing sits inside" in cleaned
+
+
+def test_strips_well_formed_then_orphan():
+ text = (
+ "Round one:\n\n\n\n1\n"
+ "\n\n\n"
+ "Now round two:\n\n\n\n"
+ "what is X\n\n" not in cleaned
+ assert "\n\n\n"Billboard Hot 100" "2015" "weekly" "chart" "position" "3"\n\n\n\n\n"peaked at number 3" Billboard Hot 100 2015 list\n\n\n\n\n"List of Billboard Hot 100 top-ten singles in 2015" wikipedia\n\n\n\nThe user wants me to list and categorize all songs that charted #3 on the Billboard Hot 100 in 2015. I have been trying to get this data",
+ # Qwen3.6-35B-A3B Q8_0 billboard s21 -- orphan close
+ "parse it more carefully.\n\n\nThe user wants a list of songs that charted #3 on the Billboard Hot 100 in 2015, categorized.",
+]
+
+
+@pytest.mark.parametrize(
+ "leak", REAL_LEAKS, ids = [f"sweep_sample_{i}" for i in range(len(REAL_LEAKS))]
+)
+def test_real_world_sweep_leaks_get_stripped(leak):
+ cleaned = _TOOL_XML_RE.sub("", leak)
+ assert "" not in cleaned, f"leak survived: {cleaned!r}"
+ assert " from gdpval sweep ──────────
+
+
+# All end-anchored: outer truncated by EOS,
+# inner open DRAINED, leaving bare tail.
+GDPVAL_PARAMETER_LEAKS = [
+ # Qwen3.5-27B Q8_0 / worldbank s00
+ "the page contains image data and the text is not readable.\n\n\n",
+ # Qwen3.5-27B Q8_0 / worldbank s42 (preceded by mojibake)
+ "...some mojibake content here...\n\n\n",
+ # Qwen3.5-27B UD-Q4_K_XL / coppa s07
+ "blocked, while others may still be in effect. The law is currently under further review by the Ninth Circuit.\n\n\n",
+ # Qwen3.5-27B UD-Q4_K_XL / police_training s00
+ "comprehensive training report\n\n\n",
+ # Qwen3.5-27B UD-Q4_K_XL / worldbank s00
+ "Global Economic Prospects\nJune 2025\nGlobal Economic Prospects\n\n",
+ # Qwen3.6-27B Q8_0 / overpass s07
+ "Let me create a comprehensive query and instructions document.\n\n\n",
+]
+
+
+@pytest.mark.parametrize(
+ "leak",
+ GDPVAL_PARAMETER_LEAKS,
+ ids = [f"gdpval_param_orphan_{i}" for i in range(len(GDPVAL_PARAMETER_LEAKS))],
+)
+def test_gdpval_parameter_orphans_get_stripped(leak):
+ cleaned = _TOOL_XML_RE.sub("", leak)
+ assert "" not in cleaned, f"leak survived: {cleaned!r}"
+
+
+# ── Backtracking guards ──────────────────────────────────────────
+
+
+def test_no_catastrophic_backtracking_on_open_bracket_spam():
+ # 256KB of '<' must fail fast (literal mismatch char 2), not backtrack.
+ import time
+
+ adv = "<" * (1024 * 256) + "X"
+ t0 = time.perf_counter()
+ _TOOL_XML_RE.sub("", adv)
+ elapsed = time.perf_counter() - t0
+ assert elapsed < 0.5, f"regex took {elapsed*1000:.0f}ms on 256KB '<' spam"
+
+
+def test_no_catastrophic_backtracking_on_orphan_opening_spam():
+ # 1000 unclosed openings: first alt must consume them all greedily.
+ import time
+
+ adv = "X" * 1000
+ t0 = time.perf_counter()
+ cleaned = _TOOL_XML_RE.sub("", adv)
+ elapsed = time.perf_counter() - t0
+ assert elapsed < 0.1, f"regex took {elapsed*1000:.0f}ms on 1000x orphan opens"
+ assert "" not in cleaned
diff --git a/studio/install_llama_prebuilt.py b/studio/install_llama_prebuilt.py
index 91076a4743..7672af6630 100644
--- a/studio/install_llama_prebuilt.py
+++ b/studio/install_llama_prebuilt.py
@@ -3827,37 +3827,18 @@ def paired_runtime_dll_patterns(choice: AssetChoice) -> list[str]:
def runtime_patterns_for_choice(choice: AssetChoice) -> list[str]:
+ # Broad shared-library glob + explicit binary names. Lets upstream
+ # repackage the SO/DLL set (e.g. ggml-org/llama.cpp#23462 split the
+ # per-binary entry code into paired ``lib-impl.so`` shared
+ # libraries between b9279 and b9283) without us re-enumerating
+ # every new file. Studio only invokes llama-server and llama-quantize;
+ # other CLIs upstream ships (llama-cli, llama-bench, ...) are skipped.
if choice.install_kind in {"linux-cpu", "linux-cuda", "linux-rocm"}:
- return [
- "llama-server",
- "llama-quantize",
- "libllama-common.so*",
- "libllama.so*",
- # Upstream llama.cpp split the per-binary entry code into
- # paired ``libllama--impl.so`` shared libraries
- # around release b9261. ``llama-server`` and
- # ``llama-quantize`` are NEEDED-linked against
- # ``libllama-server-impl.so`` / ``libllama-quantize-impl.so``
- # respectively, with RUNPATH ``$ORIGIN``. Without copying
- # the impl ``.so`` files alongside the binaries, ldd
- # reports them missing, preflight rejects the install, and
- # the installer falls back to a source build on a fresh
- # Linux install. Glob the whole family so future bundles
- # that split additional binaries (e.g. ``llama-cli``,
- # ``llama-bench``) keep working.
- "libllama-*-impl.so*",
- "libggml.so*",
- "libggml-base.so*",
- "libmtmd.so*",
- "libggml-cpu-*.so*",
- "libggml-cuda.so*",
- "libggml-hip.so*",
- "libggml-rpc.so*",
- ]
+ return ["llama-server", "llama-quantize", "lib*.so*"]
if choice.install_kind in {"macos-arm64", "macos-x64"}:
return ["llama-server", "llama-quantize", "lib*.dylib"]
if choice.install_kind in {"windows-cpu", "windows-cuda", "windows-hip"}:
- return ["*.exe", "*.dll"]
+ return ["llama-server.exe", "llama-quantize.exe", "*.dll"]
raise PrebuiltFallback(
f"unsupported install kind for runtime overlay: {choice.install_kind}"
)
diff --git a/studio/install_python_stack.py b/studio/install_python_stack.py
index 4dfa20032b..9166d35ce3 100644
--- a/studio/install_python_stack.py
+++ b/studio/install_python_stack.py
@@ -979,22 +979,11 @@ def install_python_stack() -> int:
package_name,
"unsloth-zoo",
)
- # Pydantic ships its core as a separate compiled wheel
- # (pydantic-core), and pydantic's ``_ensure_pydantic_core_version``
- # checks the installed core matches the exact version pinned in
- # its own metadata. With ``--no-deps`` plus an unpinned
- # ``pydantic`` / ``pydantic-core`` pair in no-torch-runtime.txt,
- # pip resolved each to the newest available version and the two
- # drifted (pydantic 2.13.4 pins pydantic-core==2.46.4 today, but
- # pydantic-core 2.47.0 was the latest). On a fresh Windows venv
- # the next ``import pydantic`` raised ``SystemError: ...
- # incompatible with the current pydantic version``.
- #
- # Resolve them WITH deps in a focused pip call so pip picks a
- # compatible pair. pydantic's own deps are
- # ``annotated-types``, ``pydantic-core``, ``typing-extensions``,
- # ``typing-inspection`` -- none of which transitively pull
- # torch, so this is safe for the no-torch path.
+ # Resolve pydantic WITH deps so pip pins pydantic-core to the
+ # exact version pydantic's metadata declares. Under --no-deps
+ # alone pip picks the latest of each and trips pydantic's
+ # _ensure_pydantic_core_version check. Transitive deps are
+ # torch-free.
pip_install(
"Installing pydantic (with deps for compatible core)",
"--no-cache-dir",
diff --git a/tests/python/test_construct_chat_template_validation.py b/tests/python/test_construct_chat_template_validation.py
new file mode 100644
index 0000000000..9ab68639c4
--- /dev/null
+++ b/tests/python/test_construct_chat_template_validation.py
@@ -0,0 +1,77 @@
+"""Negative-path validation tests for unsloth.chat_templates.construct_chat_template.
+
+Regression coverage for the str.find() / regex no-match guards added in
+PR #5763 follow-up: missing placeholders or unrecoverable two-example
+structures must raise RuntimeError with a clear message, not IndexError
+or AttributeError, and must never silently drop the last character via
+s[:-1].
+
+Uses a minimal fake tokenizer so the cases run on CPU-only CI without
+HF_TOKEN and without downloading a gated model. The validation paths
+exercised here fail before construct_chat_template reaches any heavy
+tokenizer interaction, so the stub stays small.
+"""
+
+import pytest
+
+from unsloth.chat_templates import construct_chat_template
+
+
+class _FakeTokenizer:
+ """Minimum surface construct_chat_template touches before the
+ validation guards fire."""
+
+ name_or_path = "fake/tokenizer"
+ eos_token = ""
+
+ def get_vocab(self):
+ return {"": 0}
+
+
+@pytest.mark.parametrize(
+ "template, expected_in_message",
+ [
+ ("only {INPUT} here, no output marker", "{OUTPUT}"),
+ ("only {OUTPUT} here, no input marker", "{INPUT}"),
+ ("neither sentinel here, just literal text", "{INPUT}"),
+ ("neither sentinel here, just literal text", "{OUTPUT}"),
+ ],
+)
+def test_missing_placeholder_in_chat_template_raises(template, expected_in_message):
+ with pytest.raises(RuntimeError) as exc_info:
+ construct_chat_template(
+ tokenizer = _FakeTokenizer(),
+ chat_template = template,
+ extra_eos_tokens = [""],
+ )
+ assert expected_in_message in str(exc_info.value)
+
+
+def test_single_pair_template_raises_clear_error_not_attribute_error():
+ """One {INPUT}/{OUTPUT} pair (rather than the required two) used to
+ crash with AttributeError on `found.group(1)` after the for-loop
+ broke without setting `found`. Must raise RuntimeError now."""
+ template = "user: {INPUT}\nassistant: {OUTPUT}\n"
+ with pytest.raises(RuntimeError):
+ construct_chat_template(
+ tokenizer = _FakeTokenizer(),
+ chat_template = template,
+ extra_eos_tokens = [""],
+ )
+
+
+def test_error_message_excerpt_is_bounded():
+ """Error messages must include a bounded excerpt of the offending
+ template, not dump arbitrarily large content into the traceback."""
+ huge = ("garbage " * 5000) + "{INPUT}" # ~40 KB, missing {OUTPUT}
+ with pytest.raises(RuntimeError) as exc_info:
+ construct_chat_template(
+ tokenizer = _FakeTokenizer(),
+ chat_template = huge,
+ extra_eos_tokens = [""],
+ )
+ msg = str(exc_info.value)
+ # Excerpt is repr-quoted and capped; total message should stay well
+ # under the template length.
+ assert len(msg) < 1000
+ assert "{OUTPUT}" in msg
diff --git a/tests/studio/install/test_rocm_support.py b/tests/studio/install/test_rocm_support.py
index 698625e59b..e6f1ae1c65 100644
--- a/tests/studio/install/test_rocm_support.py
+++ b/tests/studio/install/test_rocm_support.py
@@ -346,28 +346,26 @@ class TestRuntimePatterns:
patterns = runtime_patterns_for_choice(choice)
assert "llama-server" in patterns
assert "llama-quantize" in patterns
- # Upstream split entry code into ``libllama--impl.so``
- # shared libraries (b9261+). llama-server and llama-quantize
- # are NEEDED-linked against ``libllama-server-impl.so`` and
- # ``libllama-quantize-impl.so`` respectively with RUNPATH
- # ``$ORIGIN``, so the prebuilt overlay MUST copy them
- # alongside the binaries or ldd reports them missing and
- # preflight forces a source-build fallback.
- assert "libllama-*-impl.so*" in patterns
+ # Broad lib*.so* covers libllama, libggml, libmtmd, libggml-cpu-*,
+ # plus the libllama--impl.so split that ggml-org/llama.cpp
+ # #23462 introduced between b9279 and b9283.
+ assert "lib*.so*" in patterns
def test_linux_cuda_patterns(self):
choice = AssetChoice(
repo = "", tag = "", name = "", url = "", source_label = "", install_kind = "linux-cuda"
)
patterns = runtime_patterns_for_choice(choice)
- assert "libggml-cuda.so*" in patterns
+ # libggml-cuda.so is matched by lib*.so* now.
+ assert "lib*.so*" in patterns
def test_linux_rocm_patterns(self):
choice = AssetChoice(
repo = "", tag = "", name = "", url = "", source_label = "", install_kind = "linux-rocm"
)
patterns = runtime_patterns_for_choice(choice)
- assert "libggml-hip.so*" in patterns
+ # libggml-hip.so is matched by lib*.so* now.
+ assert "lib*.so*" in patterns
assert "llama-server" in patterns
def test_windows_hip_patterns(self):
@@ -380,7 +378,10 @@ class TestRuntimePatterns:
install_kind = "windows-hip",
)
patterns = runtime_patterns_for_choice(choice)
- assert "*.exe" in patterns
+ # Narrowed from "*.exe" to the two binaries Studio actually
+ # invokes, mirroring the Linux/macOS pattern style.
+ assert "llama-server.exe" in patterns
+ assert "llama-quantize.exe" in patterns
assert "*.dll" in patterns
def test_macos_patterns(self):
diff --git a/unsloth/chat_templates.py b/unsloth/chat_templates.py
index e8a34cbc60..956fcb2392 100644
--- a/unsloth/chat_templates.py
+++ b/unsloth/chat_templates.py
@@ -2461,17 +2461,40 @@ extra_eos_tokens = None,
f"{left_changed}"
)
except:
- ending = chat_template[chat_template.find("{OUTPUT}") + len("{OUTPUT}"):]
+ output_pos = chat_template.find("{OUTPUT}")
+ input_pos = chat_template.find("{INPUT}")
+ if output_pos == -1 or input_pos == -1:
+ missing = []
+ if input_pos == -1: missing.append("{INPUT}")
+ if output_pos == -1: missing.append("{OUTPUT}")
+ raise RuntimeError(
+ f"Unsloth: chat_template must contain {' and '.join(missing)} "
+ f"placeholder(s). Got: {chat_template[:200]!r}"
+ )
+ ending = chat_template[output_pos + len("{OUTPUT}"):]
ending = re.escape(ending)
find_text = "{INPUT}" + ending + "(.+?{OUTPUT}" + ending + ")"
response_part = re.findall(find_text, chat_template, flags = re.DOTALL | re.MULTILINE)
+ if len(response_part) == 0:
+ raise RuntimeError(
+ "Unsloth: Could not recover a two-example structure from chat_template. "
+ "Provide exactly two {INPUT}/{OUTPUT} pairs (and optionally {SYSTEM}). "
+ f"Got: {chat_template[:200]!r}"
+ )
response_part = response_part[0]
+ found = None
for j in range(1, len(response_part)):
try_find = re.escape(response_part[:j])
try: found = next(re.finditer("(" + try_find + ").+?\\{INPUT\\}", chat_template, flags = re.DOTALL | re.MULTILINE))
except: break
+ if found is None:
+ raise RuntimeError(
+ "Unsloth: Could not locate a separator between examples in chat_template. "
+ "Provide exactly two {INPUT}/{OUTPUT} pairs (and optionally {SYSTEM}). "
+ f"Got: {chat_template[:200]!r}"
+ )
separator = found.group(1)
response_start = chat_template.find(response_part)
@@ -2607,8 +2630,20 @@ extra_eos_tokens = None,
jinja_template = "{{ bos_token }}" + jinja_template
# Get instruction and output parts for train_on_inputs = False
- input_part = input_part [:input_part .find("{INPUT}")]
- output_part = output_part[:output_part.find("{OUTPUT}")]
+ input_idx = input_part .find("{INPUT}")
+ output_idx = output_part.find("{OUTPUT}")
+ if input_idx == -1:
+ raise RuntimeError(
+ f"Unsloth: The instruction section of the template must contain the "
+ f"'{{INPUT}}' placeholder. Section: {input_part[:200]!r}"
+ )
+ if output_idx == -1:
+ raise RuntimeError(
+ f"Unsloth: The response section of the template must contain the "
+ f"'{{OUTPUT}}' placeholder. Section: {output_part[:200]!r}"
+ )
+ input_part = input_part [:input_idx ]
+ output_part = output_part[:output_idx]
return modelfile, jinja_template, input_part, output_part
diff --git a/unsloth/models/_utils.py b/unsloth/models/_utils.py
index 3b67c0f487..b940fdf35a 100644
--- a/unsloth/models/_utils.py
+++ b/unsloth/models/_utils.py
@@ -12,7 +12,7 @@
# See the License for the specific language governing permissions and
# limitations under the License.
-__version__ = "2026.5.6"
+__version__ = "2026.5.7"
__all__ = [
"SUPPORTS_BFLOAT16",
diff --git a/unsloth/models/rl.py b/unsloth/models/rl.py
index 7b7c3ac1a4..c82c8364b3 100644
--- a/unsloth/models/rl.py
+++ b/unsloth/models/rl.py
@@ -1312,7 +1312,9 @@ def _patch_trl_rl_trainers_impl(trainer_file = "grpo_trainer"):
"logging_nan_inf_filter": False,
"per_device_train_batch_size": 4,
"gradient_accumulation_steps": 2,
- "weight_decay": 0.01,
+ # LoRA decays A and B toward 0 so effective W = W_init + (alpha/r) * B @ A is pulled toward W_init, not 0 as in full FT.
+ # 0.001 keeps a small Frobenius prior |A|_F^2 + |B|_F^2 without measurably dragging the merged adapter back to base.
+ "weight_decay": 0.001,
"seed": 3407,
"optim": "adamw_8bit",
"learning_rate": 5e-05,