diff --git a/install.ps1 b/install.ps1 index b26566cc3d..3911236d87 100644 --- a/install.ps1 +++ b/install.ps1 @@ -1300,15 +1300,11 @@ shell.Run cmd, 0, False if ($SkipTorch) { # No-torch: install unsloth + unsloth-zoo with --no-deps, then # runtime deps (typer, safetensors, transformers, etc.) with --no-deps. - $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --no-deps --reinstall-package unsloth --reinstall-package unsloth-zoo "unsloth>=2026.5.6" unsloth-zoo } + $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --no-deps --reinstall-package unsloth --reinstall-package unsloth-zoo "unsloth>=2026.5.7" unsloth-zoo } if ($baseInstallExit -eq 0) { - # Install pydantic WITH deps so pip pins pydantic-core to - # the exact version pydantic's metadata requires. The - # --no-deps install of no-torch-runtime.txt below would - # otherwise pick the latest of each independently and - # trip pydantic's _ensure_pydantic_core_version check. - # pydantic's deps (annotated-types, pydantic-core, - # typing-extensions, typing-inspection) are torch-free. + # Resolve pydantic WITH deps so pip pins pydantic-core + # to the matching version (no-torch-runtime.txt below + # is --no-deps). All transitive deps are torch-free. $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython pydantic } } if ($baseInstallExit -eq 0) { @@ -1318,7 +1314,7 @@ shell.Run cmd, 0, False } } } else { - $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --reinstall-package unsloth --reinstall-package unsloth-zoo "unsloth>=2026.5.6" unsloth-zoo } + $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --reinstall-package unsloth --reinstall-package unsloth-zoo "unsloth>=2026.5.7" unsloth-zoo } } if ($baseInstallExit -ne 0) { Write-Host "[ERROR] Failed to install unsloth (exit code $baseInstallExit)" -ForegroundColor Red @@ -1356,10 +1352,9 @@ shell.Run cmd, 0, False if ($SkipTorch) { # No-torch: install unsloth + unsloth-zoo with --no-deps, then # runtime deps (typer, safetensors, transformers, etc.) with --no-deps. - $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --no-deps --upgrade-package unsloth --upgrade-package unsloth-zoo "unsloth>=2026.5.6" unsloth-zoo } + $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --no-deps --upgrade-package unsloth --upgrade-package unsloth-zoo "unsloth>=2026.5.7" unsloth-zoo } if ($baseInstallExit -eq 0) { - # Install pydantic WITH deps so pip pins pydantic-core to - # the matching version (see migrated branch above). + # Same pydantic-with-deps trick as the migrated branch. $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython pydantic } } if ($baseInstallExit -eq 0) { @@ -1369,7 +1364,7 @@ shell.Run cmd, 0, False } } } elseif ($StudioLocalInstall) { - $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --upgrade-package unsloth "unsloth>=2026.5.6" unsloth-zoo } + $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --upgrade-package unsloth "unsloth>=2026.5.7" unsloth-zoo } } else { $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython --upgrade-package unsloth -- "$PackageName" } } @@ -1397,7 +1392,7 @@ shell.Run cmd, 0, False Write-TauriLog "STEP" "Installing unsloth" substep "installing unsloth (this may take a few minutes)..." if ($StudioLocalInstall) { - $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython unsloth-zoo "unsloth>=2026.5.6" --torch-backend=auto } + $baseInstallExit = Invoke-InstallCommand { uv pip install --python $VenvPython unsloth-zoo "unsloth>=2026.5.7" --torch-backend=auto } if ($baseInstallExit -ne 0) { Write-Host "[ERROR] Failed to install unsloth (exit code $baseInstallExit)" -ForegroundColor Red return (Exit-InstallFailure "Failed to install unsloth (exit code $baseInstallExit)" $baseInstallExit) diff --git a/install.sh b/install.sh index 9bdd935171..cc92fd52c2 100755 --- a/install.sh +++ b/install.sh @@ -1865,14 +1865,10 @@ if [ "$_MIGRATED" = true ]; then # to prevent transitive torch resolution. run_install_cmd "install unsloth (migrated no-torch)" uv pip install --python "$_VENV_PY" --no-deps \ --reinstall-package unsloth --reinstall-package unsloth-zoo \ - "unsloth>=2026.5.6" unsloth-zoo - # Install pydantic WITH deps so pip pins pydantic-core to the - # exact version pydantic's own metadata requires. The --no-deps - # install below would otherwise pick the latest of each - # independently and trip pydantic's _ensure_pydantic_core_version - # check on the next import. pydantic's deps (annotated-types, - # pydantic-core, typing-extensions, typing-inspection) are - # torch-free, so this is safe on the no-torch path. + "unsloth>=2026.5.7" unsloth-zoo + # Resolve pydantic WITH deps so pip pins pydantic-core to the + # matching version (no-torch-runtime.txt below is --no-deps). + # All transitive deps are torch-free. run_install_cmd "install pydantic (with deps for compatible core)" \ uv pip install --python "$_VENV_PY" pydantic _NO_TORCH_RT="$(_find_no_torch_runtime)" @@ -1882,7 +1878,7 @@ if [ "$_MIGRATED" = true ]; then else run_install_cmd "install unsloth (migrated)" uv pip install --python "$_VENV_PY" \ --reinstall-package unsloth --reinstall-package unsloth-zoo \ - "unsloth>=2026.5.6" unsloth-zoo + "unsloth>=2026.5.7" unsloth-zoo fi if [ "$STUDIO_LOCAL_INSTALL" = true ]; then substep "overlaying local repo (editable)..." @@ -2050,9 +2046,8 @@ elif [ -n "$TORCH_INDEX_URL" ]; then # runtime deps (typer, safetensors, transformers, etc.) with --no-deps. run_install_cmd "install unsloth (no-torch)" uv pip install --python "$_VENV_PY" --no-deps \ --upgrade-package unsloth --upgrade-package unsloth-zoo \ - "unsloth>=2026.5.6" unsloth-zoo - # Install pydantic WITH deps so pip pins pydantic-core to the - # exact version pydantic requires (see migrated branch above). + "unsloth>=2026.5.7" unsloth-zoo + # Same pydantic-with-deps trick as the migrated branch. run_install_cmd "install pydantic (with deps for compatible core)" \ uv pip install --python "$_VENV_PY" pydantic _NO_TORCH_RT="$(_find_no_torch_runtime)" @@ -2069,7 +2064,7 @@ elif [ -n "$TORCH_INDEX_URL" ]; then fi elif [ "$STUDIO_LOCAL_INSTALL" = true ]; then run_install_cmd "install unsloth (local)" uv pip install --python "$_VENV_PY" \ - --upgrade-package unsloth "unsloth>=2026.5.6" unsloth-zoo + --upgrade-package unsloth "unsloth>=2026.5.7" unsloth-zoo substep "overlaying local repo (editable)..." run_install_cmd "overlay local repo" uv pip install --python "$_VENV_PY" -e "$_REPO_ROOT" --no-deps substep "overlaying unsloth-zoo from git main..." @@ -2101,7 +2096,7 @@ else tauri_log "STEP" "Installing Unsloth" substep "installing unsloth (this may take a few minutes)..." if [ "$STUDIO_LOCAL_INSTALL" = true ]; then - run_install_cmd "install unsloth (auto torch backend)" uv pip install --python "$_VENV_PY" unsloth-zoo "unsloth>=2026.5.6" --torch-backend=auto + run_install_cmd "install unsloth (auto torch backend)" uv pip install --python "$_VENV_PY" unsloth-zoo "unsloth>=2026.5.7" --torch-backend=auto substep "overlaying local repo (editable)..." run_install_cmd "overlay local repo" uv pip install --python "$_VENV_PY" -e "$_REPO_ROOT" --no-deps substep "overlaying unsloth-zoo from git main..." diff --git a/studio/backend/core/export/export.py b/studio/backend/core/export/export.py index 4ab95d896f..7cabd382eb 100644 --- a/studio/backend/core/export/export.py +++ b/studio/backend/core/export/export.py @@ -475,6 +475,7 @@ class ExportBackend: self.current_model.save_pretrained_merged( save_directory, self.current_tokenizer, + save_method = "merged_16bit", ) else: self.current_model.save_pretrained(save_directory) @@ -510,6 +511,7 @@ class ExportBackend: self.current_model.save_pretrained_merged( tmp_dir, self.current_tokenizer, + save_method = "merged_16bit", ) self.current_model.push_to_hub_merged( repo_id, diff --git a/studio/backend/requirements/no-torch-runtime.txt b/studio/backend/requirements/no-torch-runtime.txt index c33ebf4d94..85294114b1 100644 --- a/studio/backend/requirements/no-torch-runtime.txt +++ b/studio/backend/requirements/no-torch-runtime.txt @@ -22,19 +22,11 @@ rich>=13.0 markdown-it-py>=3.0 mdurl>=0.1 pygments>=2.0 -# pydantic is intentionally NOT installed via this --no-deps file. -# install.sh / install.ps1 / install_python_stack.py run a separate -# `pip install pydantic` (with deps) just before this file is -# applied, so pip resolves `pydantic-core` to the exact version -# pydantic's `_ensure_pydantic_core_version` check expects. Listing -# pydantic + pydantic-core unpinned here and resolving them under -# --no-deps used to pick the latest of each independently and trip -# `SystemError: pydantic-core 2.X.Y is incompatible with the current -# pydantic version` on the first import (Windows fresh-venv repro -# was the canonical case). pydantic's transitive deps -# (annotated-types, pydantic-core, typing-extensions, -# typing-inspection) are torch-free, so installing it WITH deps -# does not pull torch. +# pydantic is intentionally NOT pinned here. install.sh / install.ps1 +# / install_python_stack.py run `pip install pydantic` WITH deps just +# before this --no-deps file is applied, so pip resolves pydantic-core +# to the exact version pydantic's _ensure_pydantic_core_version check +# expects. Pinning both under --no-deps used to drift them apart. pyyaml nest-asyncio diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 02270ab405..bf92055929 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -427,9 +427,17 @@ _TOOL_ACTION_NUDGE = ( " Do NOT output code blocks -- use the python tool instead." ) -# Regex for stripping leaked tool-call XML from assistant messages/stream +# Strip tool-call XML the speculative buffer in core/inference/llama_cpp.py +# split across the visible/DRAIN boundary. Four leak shapes: +# 1. well-formed `...` / `...` +# 2. orphan opening to EOF (close was DRAINED) +# 3. bare orphan close (open was DRAINED) +# 4. tail-only `` (outer close truncated by EOS); anchored to +# `\Z` so mid-text `` in user code samples survives. _TOOL_XML_RE = _re.compile( - r".*?|.*?", + r"<(?:tool_call|function=\w+)>.*?(?:|\Z)" + r"|" + r"|\s*\Z", _re.DOTALL, ) logger = get_logger(__name__) diff --git a/studio/backend/tests/test_tool_xml_strip.py b/studio/backend/tests/test_tool_xml_strip.py new file mode 100644 index 0000000000..8b90a46d5a --- /dev/null +++ b/studio/backend/tests/test_tool_xml_strip.py @@ -0,0 +1,263 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 + +"""Tests for `_TOOL_XML_RE` (routes/inference.py) -- strips tool-call +XML that leaks past the speculative buffer in core/inference/llama_cpp.py +when the open/close pair is split across the visible/DRAIN boundary. +""" + +from __future__ import annotations + +import sys +import types as _types +from pathlib import Path + +import pytest + +_BACKEND_DIR = str(Path(__file__).resolve().parent.parent) +if _BACKEND_DIR not in sys.path: + sys.path.insert(0, _BACKEND_DIR) + +# Extract the regex from source (routes module needs heavy stubbing to import). +import re as _re + +_src = (Path(_BACKEND_DIR) / "routes" / "inference.py").read_text() +_m = _re.search(r"_TOOL_XML_RE = _re\.compile\((.*?)\n\)", _src, _re.DOTALL) +assert _m, "could not extract _TOOL_XML_RE source" +_ns = {"_re": _re} +exec(f"_TOOL_XML_RE = _re.compile({_m.group(1)})", _ns) +_TOOL_XML_RE = _ns["_TOOL_XML_RE"] + + +# ── Well-formed pairs ───────────────────────────────────────────── + + +def test_strips_well_formed_tool_call(): + text = ( + "Let me search.\n" + "\n" + "\n" + "\nBillboard 2015\n\n" + "\n" + "\n" + "Here are the songs:" + ) + cleaned = _TOOL_XML_RE.sub("", text) + assert "" not in cleaned + assert "" not in cleaned + assert "" not in cleaned + assert "Here are the songs:" in cleaned, "non-XML content must survive" + assert "Let me search." in cleaned + + +def test_strips_function_only_well_formed(): + text = "Setup.\n\n\nprint(1)\n\n\nDone." + cleaned = _TOOL_XML_RE.sub("", text) + assert "" + "\n" + "\n" + "\nBillboard 2015\n\n" + "" not in cleaned + assert "\n\nprint(1)\n" + ) + cleaned = _TOOL_XML_RE.sub("", text) + assert "") + assert "" not in cleaned + assert "Search starting." in cleaned + + +def test_strips_multiple_orphans(): + text = ( + "First call:\n\n\n\nx=1\n" + "Second call:\n\n\nhi\n" + ) + cleaned = _TOOL_XML_RE.sub("", text) + assert "" not in cleaned + assert "" not in cleaned + assert "" not in cleaned + # Mid-string intentionally preserved (see preserve test). + + +# ── Tail-only (PR #5735 follow-up) ─────────────────── + + +def test_strips_tail_only_parameter_orphan(): + # Outer truncated by EOS, inner DRAINED. + cleaned = _TOOL_XML_RE.sub("", "and the text is not readable.\n\n\n") + assert "" not in cleaned + assert "and the text is not readable." in cleaned + + +def test_strips_tail_only_parameter_orphan_single_newline(): + cleaned = _TOOL_XML_RE.sub("", "Global Economic Prospects\n\n") + assert "" not in cleaned + assert "Global Economic Prospects" in cleaned + + +def test_strips_tail_only_parameter_orphan_no_trailing_ws(): + cleaned = _TOOL_XML_RE.sub("", "Final answer.") + assert "" not in cleaned + assert "Final answer." in cleaned + + +def test_preserves_mid_string_parameter_in_code_sample(): + # Tail-anchor on `` is required so doc/example prose survives. + text = ( + "Here is the Qwen tool-call format:\n" + "```xml\n" + "value\n" + "```\n" + "Note the closing sits inside ." + ) + cleaned = _TOOL_XML_RE.sub("", text) + assert "Note the closing sits inside" in cleaned + + +def test_strips_well_formed_then_orphan(): + text = ( + "Round one:\n\n\n\n1\n" + "\n\n\n" + "Now round two:\n\n\n\n" + "what is X\n\n" not in cleaned + assert "\n\n\n"Billboard Hot 100" "2015" "weekly" "chart" "position" "3"\n\n\n\n\n"peaked at number 3" Billboard Hot 100 2015 list\n\n\n\n\n"List of Billboard Hot 100 top-ten singles in 2015" wikipedia\n\n\n\nThe user wants me to list and categorize all songs that charted #3 on the Billboard Hot 100 in 2015. I have been trying to get this data", + # Qwen3.6-35B-A3B Q8_0 billboard s21 -- orphan close + "parse it more carefully.\n\n\nThe user wants a list of songs that charted #3 on the Billboard Hot 100 in 2015, categorized.", +] + + +@pytest.mark.parametrize( + "leak", REAL_LEAKS, ids = [f"sweep_sample_{i}" for i in range(len(REAL_LEAKS))] +) +def test_real_world_sweep_leaks_get_stripped(leak): + cleaned = _TOOL_XML_RE.sub("", leak) + assert "" not in cleaned, f"leak survived: {cleaned!r}" + assert " from gdpval sweep ────────── + + +# All end-anchored: outer truncated by EOS, +# inner open DRAINED, leaving bare tail. +GDPVAL_PARAMETER_LEAKS = [ + # Qwen3.5-27B Q8_0 / worldbank s00 + "the page contains image data and the text is not readable.\n\n\n", + # Qwen3.5-27B Q8_0 / worldbank s42 (preceded by mojibake) + "...some mojibake content here...\n\n\n", + # Qwen3.5-27B UD-Q4_K_XL / coppa s07 + "blocked, while others may still be in effect. The law is currently under further review by the Ninth Circuit.\n\n\n", + # Qwen3.5-27B UD-Q4_K_XL / police_training s00 + "comprehensive training report\n\n\n", + # Qwen3.5-27B UD-Q4_K_XL / worldbank s00 + "Global Economic Prospects\nJune 2025\nGlobal Economic Prospects\n\n", + # Qwen3.6-27B Q8_0 / overpass s07 + "Let me create a comprehensive query and instructions document.\n\n\n", +] + + +@pytest.mark.parametrize( + "leak", + GDPVAL_PARAMETER_LEAKS, + ids = [f"gdpval_param_orphan_{i}" for i in range(len(GDPVAL_PARAMETER_LEAKS))], +) +def test_gdpval_parameter_orphans_get_stripped(leak): + cleaned = _TOOL_XML_RE.sub("", leak) + assert "" not in cleaned, f"leak survived: {cleaned!r}" + + +# ── Backtracking guards ────────────────────────────────────────── + + +def test_no_catastrophic_backtracking_on_open_bracket_spam(): + # 256KB of '<' must fail fast (literal mismatch char 2), not backtrack. + import time + + adv = "<" * (1024 * 256) + "X" + t0 = time.perf_counter() + _TOOL_XML_RE.sub("", adv) + elapsed = time.perf_counter() - t0 + assert elapsed < 0.5, f"regex took {elapsed*1000:.0f}ms on 256KB '<' spam" + + +def test_no_catastrophic_backtracking_on_orphan_opening_spam(): + # 1000 unclosed openings: first alt must consume them all greedily. + import time + + adv = "X" * 1000 + t0 = time.perf_counter() + cleaned = _TOOL_XML_RE.sub("", adv) + elapsed = time.perf_counter() - t0 + assert elapsed < 0.1, f"regex took {elapsed*1000:.0f}ms on 1000x orphan opens" + assert "" not in cleaned diff --git a/studio/install_llama_prebuilt.py b/studio/install_llama_prebuilt.py index 91076a4743..7672af6630 100644 --- a/studio/install_llama_prebuilt.py +++ b/studio/install_llama_prebuilt.py @@ -3827,37 +3827,18 @@ def paired_runtime_dll_patterns(choice: AssetChoice) -> list[str]: def runtime_patterns_for_choice(choice: AssetChoice) -> list[str]: + # Broad shared-library glob + explicit binary names. Lets upstream + # repackage the SO/DLL set (e.g. ggml-org/llama.cpp#23462 split the + # per-binary entry code into paired ``lib-impl.so`` shared + # libraries between b9279 and b9283) without us re-enumerating + # every new file. Studio only invokes llama-server and llama-quantize; + # other CLIs upstream ships (llama-cli, llama-bench, ...) are skipped. if choice.install_kind in {"linux-cpu", "linux-cuda", "linux-rocm"}: - return [ - "llama-server", - "llama-quantize", - "libllama-common.so*", - "libllama.so*", - # Upstream llama.cpp split the per-binary entry code into - # paired ``libllama--impl.so`` shared libraries - # around release b9261. ``llama-server`` and - # ``llama-quantize`` are NEEDED-linked against - # ``libllama-server-impl.so`` / ``libllama-quantize-impl.so`` - # respectively, with RUNPATH ``$ORIGIN``. Without copying - # the impl ``.so`` files alongside the binaries, ldd - # reports them missing, preflight rejects the install, and - # the installer falls back to a source build on a fresh - # Linux install. Glob the whole family so future bundles - # that split additional binaries (e.g. ``llama-cli``, - # ``llama-bench``) keep working. - "libllama-*-impl.so*", - "libggml.so*", - "libggml-base.so*", - "libmtmd.so*", - "libggml-cpu-*.so*", - "libggml-cuda.so*", - "libggml-hip.so*", - "libggml-rpc.so*", - ] + return ["llama-server", "llama-quantize", "lib*.so*"] if choice.install_kind in {"macos-arm64", "macos-x64"}: return ["llama-server", "llama-quantize", "lib*.dylib"] if choice.install_kind in {"windows-cpu", "windows-cuda", "windows-hip"}: - return ["*.exe", "*.dll"] + return ["llama-server.exe", "llama-quantize.exe", "*.dll"] raise PrebuiltFallback( f"unsupported install kind for runtime overlay: {choice.install_kind}" ) diff --git a/studio/install_python_stack.py b/studio/install_python_stack.py index 4dfa20032b..9166d35ce3 100644 --- a/studio/install_python_stack.py +++ b/studio/install_python_stack.py @@ -979,22 +979,11 @@ def install_python_stack() -> int: package_name, "unsloth-zoo", ) - # Pydantic ships its core as a separate compiled wheel - # (pydantic-core), and pydantic's ``_ensure_pydantic_core_version`` - # checks the installed core matches the exact version pinned in - # its own metadata. With ``--no-deps`` plus an unpinned - # ``pydantic`` / ``pydantic-core`` pair in no-torch-runtime.txt, - # pip resolved each to the newest available version and the two - # drifted (pydantic 2.13.4 pins pydantic-core==2.46.4 today, but - # pydantic-core 2.47.0 was the latest). On a fresh Windows venv - # the next ``import pydantic`` raised ``SystemError: ... - # incompatible with the current pydantic version``. - # - # Resolve them WITH deps in a focused pip call so pip picks a - # compatible pair. pydantic's own deps are - # ``annotated-types``, ``pydantic-core``, ``typing-extensions``, - # ``typing-inspection`` -- none of which transitively pull - # torch, so this is safe for the no-torch path. + # Resolve pydantic WITH deps so pip pins pydantic-core to the + # exact version pydantic's metadata declares. Under --no-deps + # alone pip picks the latest of each and trips pydantic's + # _ensure_pydantic_core_version check. Transitive deps are + # torch-free. pip_install( "Installing pydantic (with deps for compatible core)", "--no-cache-dir", diff --git a/tests/python/test_construct_chat_template_validation.py b/tests/python/test_construct_chat_template_validation.py new file mode 100644 index 0000000000..9ab68639c4 --- /dev/null +++ b/tests/python/test_construct_chat_template_validation.py @@ -0,0 +1,77 @@ +"""Negative-path validation tests for unsloth.chat_templates.construct_chat_template. + +Regression coverage for the str.find() / regex no-match guards added in +PR #5763 follow-up: missing placeholders or unrecoverable two-example +structures must raise RuntimeError with a clear message, not IndexError +or AttributeError, and must never silently drop the last character via +s[:-1]. + +Uses a minimal fake tokenizer so the cases run on CPU-only CI without +HF_TOKEN and without downloading a gated model. The validation paths +exercised here fail before construct_chat_template reaches any heavy +tokenizer interaction, so the stub stays small. +""" + +import pytest + +from unsloth.chat_templates import construct_chat_template + + +class _FakeTokenizer: + """Minimum surface construct_chat_template touches before the + validation guards fire.""" + + name_or_path = "fake/tokenizer" + eos_token = "" + + def get_vocab(self): + return {"": 0} + + +@pytest.mark.parametrize( + "template, expected_in_message", + [ + ("only {INPUT} here, no output marker", "{OUTPUT}"), + ("only {OUTPUT} here, no input marker", "{INPUT}"), + ("neither sentinel here, just literal text", "{INPUT}"), + ("neither sentinel here, just literal text", "{OUTPUT}"), + ], +) +def test_missing_placeholder_in_chat_template_raises(template, expected_in_message): + with pytest.raises(RuntimeError) as exc_info: + construct_chat_template( + tokenizer = _FakeTokenizer(), + chat_template = template, + extra_eos_tokens = [""], + ) + assert expected_in_message in str(exc_info.value) + + +def test_single_pair_template_raises_clear_error_not_attribute_error(): + """One {INPUT}/{OUTPUT} pair (rather than the required two) used to + crash with AttributeError on `found.group(1)` after the for-loop + broke without setting `found`. Must raise RuntimeError now.""" + template = "user: {INPUT}\nassistant: {OUTPUT}\n" + with pytest.raises(RuntimeError): + construct_chat_template( + tokenizer = _FakeTokenizer(), + chat_template = template, + extra_eos_tokens = [""], + ) + + +def test_error_message_excerpt_is_bounded(): + """Error messages must include a bounded excerpt of the offending + template, not dump arbitrarily large content into the traceback.""" + huge = ("garbage " * 5000) + "{INPUT}" # ~40 KB, missing {OUTPUT} + with pytest.raises(RuntimeError) as exc_info: + construct_chat_template( + tokenizer = _FakeTokenizer(), + chat_template = huge, + extra_eos_tokens = [""], + ) + msg = str(exc_info.value) + # Excerpt is repr-quoted and capped; total message should stay well + # under the template length. + assert len(msg) < 1000 + assert "{OUTPUT}" in msg diff --git a/tests/studio/install/test_rocm_support.py b/tests/studio/install/test_rocm_support.py index 698625e59b..e6f1ae1c65 100644 --- a/tests/studio/install/test_rocm_support.py +++ b/tests/studio/install/test_rocm_support.py @@ -346,28 +346,26 @@ class TestRuntimePatterns: patterns = runtime_patterns_for_choice(choice) assert "llama-server" in patterns assert "llama-quantize" in patterns - # Upstream split entry code into ``libllama--impl.so`` - # shared libraries (b9261+). llama-server and llama-quantize - # are NEEDED-linked against ``libllama-server-impl.so`` and - # ``libllama-quantize-impl.so`` respectively with RUNPATH - # ``$ORIGIN``, so the prebuilt overlay MUST copy them - # alongside the binaries or ldd reports them missing and - # preflight forces a source-build fallback. - assert "libllama-*-impl.so*" in patterns + # Broad lib*.so* covers libllama, libggml, libmtmd, libggml-cpu-*, + # plus the libllama--impl.so split that ggml-org/llama.cpp + # #23462 introduced between b9279 and b9283. + assert "lib*.so*" in patterns def test_linux_cuda_patterns(self): choice = AssetChoice( repo = "", tag = "", name = "", url = "", source_label = "", install_kind = "linux-cuda" ) patterns = runtime_patterns_for_choice(choice) - assert "libggml-cuda.so*" in patterns + # libggml-cuda.so is matched by lib*.so* now. + assert "lib*.so*" in patterns def test_linux_rocm_patterns(self): choice = AssetChoice( repo = "", tag = "", name = "", url = "", source_label = "", install_kind = "linux-rocm" ) patterns = runtime_patterns_for_choice(choice) - assert "libggml-hip.so*" in patterns + # libggml-hip.so is matched by lib*.so* now. + assert "lib*.so*" in patterns assert "llama-server" in patterns def test_windows_hip_patterns(self): @@ -380,7 +378,10 @@ class TestRuntimePatterns: install_kind = "windows-hip", ) patterns = runtime_patterns_for_choice(choice) - assert "*.exe" in patterns + # Narrowed from "*.exe" to the two binaries Studio actually + # invokes, mirroring the Linux/macOS pattern style. + assert "llama-server.exe" in patterns + assert "llama-quantize.exe" in patterns assert "*.dll" in patterns def test_macos_patterns(self): diff --git a/unsloth/chat_templates.py b/unsloth/chat_templates.py index e8a34cbc60..956fcb2392 100644 --- a/unsloth/chat_templates.py +++ b/unsloth/chat_templates.py @@ -2461,17 +2461,40 @@ extra_eos_tokens = None, f"{left_changed}" ) except: - ending = chat_template[chat_template.find("{OUTPUT}") + len("{OUTPUT}"):] + output_pos = chat_template.find("{OUTPUT}") + input_pos = chat_template.find("{INPUT}") + if output_pos == -1 or input_pos == -1: + missing = [] + if input_pos == -1: missing.append("{INPUT}") + if output_pos == -1: missing.append("{OUTPUT}") + raise RuntimeError( + f"Unsloth: chat_template must contain {' and '.join(missing)} " + f"placeholder(s). Got: {chat_template[:200]!r}" + ) + ending = chat_template[output_pos + len("{OUTPUT}"):] ending = re.escape(ending) find_text = "{INPUT}" + ending + "(.+?{OUTPUT}" + ending + ")" response_part = re.findall(find_text, chat_template, flags = re.DOTALL | re.MULTILINE) + if len(response_part) == 0: + raise RuntimeError( + "Unsloth: Could not recover a two-example structure from chat_template. " + "Provide exactly two {INPUT}/{OUTPUT} pairs (and optionally {SYSTEM}). " + f"Got: {chat_template[:200]!r}" + ) response_part = response_part[0] + found = None for j in range(1, len(response_part)): try_find = re.escape(response_part[:j]) try: found = next(re.finditer("(" + try_find + ").+?\\{INPUT\\}", chat_template, flags = re.DOTALL | re.MULTILINE)) except: break + if found is None: + raise RuntimeError( + "Unsloth: Could not locate a separator between examples in chat_template. " + "Provide exactly two {INPUT}/{OUTPUT} pairs (and optionally {SYSTEM}). " + f"Got: {chat_template[:200]!r}" + ) separator = found.group(1) response_start = chat_template.find(response_part) @@ -2607,8 +2630,20 @@ extra_eos_tokens = None, jinja_template = "{{ bos_token }}" + jinja_template # Get instruction and output parts for train_on_inputs = False - input_part = input_part [:input_part .find("{INPUT}")] - output_part = output_part[:output_part.find("{OUTPUT}")] + input_idx = input_part .find("{INPUT}") + output_idx = output_part.find("{OUTPUT}") + if input_idx == -1: + raise RuntimeError( + f"Unsloth: The instruction section of the template must contain the " + f"'{{INPUT}}' placeholder. Section: {input_part[:200]!r}" + ) + if output_idx == -1: + raise RuntimeError( + f"Unsloth: The response section of the template must contain the " + f"'{{OUTPUT}}' placeholder. Section: {output_part[:200]!r}" + ) + input_part = input_part [:input_idx ] + output_part = output_part[:output_idx] return modelfile, jinja_template, input_part, output_part diff --git a/unsloth/models/_utils.py b/unsloth/models/_utils.py index 3b67c0f487..b940fdf35a 100644 --- a/unsloth/models/_utils.py +++ b/unsloth/models/_utils.py @@ -12,7 +12,7 @@ # See the License for the specific language governing permissions and # limitations under the License. -__version__ = "2026.5.6" +__version__ = "2026.5.7" __all__ = [ "SUPPORTS_BFLOAT16", diff --git a/unsloth/models/rl.py b/unsloth/models/rl.py index 7b7c3ac1a4..c82c8364b3 100644 --- a/unsloth/models/rl.py +++ b/unsloth/models/rl.py @@ -1312,7 +1312,9 @@ def _patch_trl_rl_trainers_impl(trainer_file = "grpo_trainer"): "logging_nan_inf_filter": False, "per_device_train_batch_size": 4, "gradient_accumulation_steps": 2, - "weight_decay": 0.01, + # LoRA decays A and B toward 0 so effective W = W_init + (alpha/r) * B @ A is pulled toward W_init, not 0 as in full FT. + # 0.001 keeps a small Frobenius prior |A|_F^2 + |B|_F^2 without measurably dragging the merged adapter back to base. + "weight_decay": 0.001, "seed": 3407, "optim": "adamw_8bit", "learning_rate": 5e-05,