unsloth/tests/studio/install/test_hf_auth.py
Daniel Han 2e29363ad9
Studio CI: stop HF 429 rate limits from sinking the llama.cpp prebuilt path (#6199)
* Stop HF 429 rate limits from sinking the llama.cpp prebuilt path in Studio CI

The Windows Studio API smoke job failed when anonymous huggingface.co
fetches of the tiny GGUF validation model (stories260K.gguf) hit HTTP 429
on the shared runner IP. The installer correctly refused the unvalidated
prebuilt and fell back to a source build, which the prebuilt assert then
flags. Three layers fix this:

1. Installer: auth_headers sends HF_TOKEN (or HUGGING_FACE_HUB_TOKEN) to
   huggingface.co hosts, mirroring the existing GH_TOKEN handling for the
   GitHub API rate limit. A redirect handler strips Authorization when a
   download is redirected off-host (CDN signed URLs reject foreign auth;
   urllib forwards headers on redirect, unlike requests/huggingface_hub).

2. Workflows: the HF_HOME prime steps also prefetch the validation model
   so the install's hf_hub_download resolves from the local cache even
   when the Hub is rate limiting; cache keys bumped v1 to v2 to repopulate.
   This also covers fork PRs, which cannot see secrets.

3. Workflows: every Install Studio / update step that already passes
   GH_TOKEN now also passes HF_TOKEN, so both the huggingface_hub path and
   the direct URL fallback are authenticated.

Tests: tests/studio/install/test_hf_auth.py covers token-to-host routing,
the cross-host redirect strip, and the download_bytes wiring (offline).
Verified live: authenticated download of the validation model through the
new opener (CDN redirect exercised, pinned sha matches) and an offline
hf_hub_download cache hit against an HF_HOME primed by the new step.

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

---------

Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
2026-06-11 06:57:48 -07:00

114 lines
4.4 KiB
Python

"""Tests for Hugging Face auth on the llama.cpp prebuilt installer's fetches.
Anonymous huggingface.co downloads (tiny GGUF validation model) share a
per-IP rate limit that CI fleets exhaust (HTTP 429), forcing the prebuilt
path into a source build. auth_headers now sends HF_TOKEN to huggingface.co
hosts, and a redirect handler strips Authorization when the download is
redirected to a different host (CDN signed URLs). All tests are offline.
"""
import importlib.util
import sys
import urllib.request
from pathlib import Path
from unittest.mock import MagicMock, patch
import pytest
PACKAGE_ROOT = Path(__file__).resolve().parents[3]
_MODULE_PATH = PACKAGE_ROOT / "studio" / "install_llama_prebuilt.py"
_SPEC = importlib.util.spec_from_file_location(
"studio_install_llama_prebuilt_hf_auth", _MODULE_PATH
)
assert _SPEC is not None and _SPEC.loader is not None
mod = importlib.util.module_from_spec(_SPEC)
sys.modules[_SPEC.name] = mod
_SPEC.loader.exec_module(mod)
_TOKEN_VARS = ("GH_TOKEN", "GITHUB_TOKEN", "HF_TOKEN", "HUGGING_FACE_HUB_TOKEN")
HF_URL = "https://huggingface.co/ggml-org/models/resolve/main/tinyllamas/stories260K.gguf"
GH_URL = "https://api.github.com/repos/unslothai/llama.cpp/releases"
def _headers(url, env):
"""auth_headers under a fully controlled token environment."""
with patch.dict(mod.os.environ, env, clear = False):
for var in _TOKEN_VARS:
if var not in env:
mod.os.environ.pop(var, None)
return mod.auth_headers(url)
class TestAuthHeaderRouting:
def test_hf_token_sent_to_huggingface(self):
headers = _headers(HF_URL, {"HF_TOKEN": "hf_x"})
assert headers.get("Authorization") == "Bearer hf_x"
def test_hub_token_fallback(self):
headers = _headers(HF_URL, {"HUGGING_FACE_HUB_TOKEN": "hf_y"})
assert headers.get("Authorization") == "Bearer hf_y"
def test_hf_token_not_sent_to_github(self):
headers = _headers(GH_URL, {"HF_TOKEN": "hf_x"})
assert "Authorization" not in headers
def test_hf_token_not_sent_to_other_hosts(self):
headers = _headers("https://cdn-lfs.huggingface.co/x", {"HF_TOKEN": "hf_x"})
assert "Authorization" not in headers
def test_gh_token_not_sent_to_huggingface(self):
headers = _headers(HF_URL, {"GH_TOKEN": "gh_x"})
assert "Authorization" not in headers
def test_gh_token_still_wins_on_github(self):
headers = _headers(GH_URL, {"GH_TOKEN": "gh_x", "HF_TOKEN": "hf_x"})
assert headers.get("Authorization") == "Bearer gh_x"
def test_no_tokens_no_auth(self):
assert "Authorization" not in _headers(HF_URL, {})
def test_validation_model_url_is_hf(self):
assert mod.should_send_hf_auth(mod.TEST_MODEL_URL) is True
class TestCrossHostRedirectStripsAuth:
def _redirect(self, newurl):
req = urllib.request.Request(HF_URL, headers = {"Authorization": "Bearer hf_x"})
handler = mod._CrossHostAuthStrippingRedirectHandler()
return handler.redirect_request(req, None, 302, "Found", {}, newurl)
def test_cross_host_redirect_drops_authorization(self):
new_request = self._redirect("https://cdn-lfs.huggingface.co/signed/blob")
assert new_request is not None
assert "Authorization" not in new_request.headers
assert "Authorization" not in new_request.unredirected_hdrs
def test_same_host_redirect_keeps_authorization(self):
new_request = self._redirect("https://huggingface.co/elsewhere/blob")
assert new_request is not None
assert new_request.headers.get("Authorization") == "Bearer hf_x"
class TestDownloadBytesWiring:
def test_download_bytes_sends_hf_auth(self):
response = MagicMock()
response.__enter__ = lambda s: s
response.__exit__ = lambda s, *a: False
response.headers.get.return_value = None
response.read.side_effect = [b"data", b""]
with (
patch.object(mod._URL_OPENER, "open", return_value = response) as opened,
patch.dict(mod.os.environ, {"HF_TOKEN": "hf_x"}, clear = False),
):
for var in ("GH_TOKEN", "GITHUB_TOKEN"):
mod.os.environ.pop(var, None)
data = mod.download_bytes(HF_URL)
assert data == b"data"
request = opened.call_args.args[0]
assert request.headers.get("Authorization") == "Bearer hf_x"
if __name__ == "__main__":
sys.exit(pytest.main([__file__, "-q"]))