Merge remote-tracking branch 'origin/main' into feature/deep-research
# Conflicts: # studio/frontend/src/features/chat/chat-page.tsx
This commit is contained in:
commit
d65c9520cb
43 changed files with 4215 additions and 391 deletions
82
tests/python/test_get_lora_parameters_bias_fp8_block_size.py
Normal file
82
tests/python/test_get_lora_parameters_bias_fp8_block_size.py
Normal file
|
|
@ -0,0 +1,82 @@
|
|||
import ast
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def _load_function(name):
|
||||
# Extract a function from kernels/utils.py without importing unsloth (which
|
||||
# needs a GPU / torch / bitsandbytes). The functions exercised here only use
|
||||
# getattr and the _FP8_WEIGHT_DTYPES name on the paths under test.
|
||||
source = Path(__file__).parents[2] / "unsloth" / "kernels" / "utils.py"
|
||||
tree = ast.parse(source.read_text(encoding = "utf-8"))
|
||||
funcs = [
|
||||
node for node in ast.walk(tree) if isinstance(node, ast.FunctionDef) and node.name == name
|
||||
]
|
||||
assert len(funcs) == 1, (name, funcs)
|
||||
namespace = {"getattr": getattr, "_FP8_WEIGHT_DTYPES": ()}
|
||||
module = ast.Module(body = funcs, type_ignores = [])
|
||||
ast.fix_missing_locations(module)
|
||||
exec(compile(module, str(source), "exec"), namespace)
|
||||
return namespace[name]
|
||||
|
||||
|
||||
class _Obj:
|
||||
pass
|
||||
|
||||
|
||||
def _make_disabled_block_fp8_proj(block_size):
|
||||
# A merged/disabled projection whose base layer is a block-fp8 weight that
|
||||
# ships a non-default block size on its checkpoint.
|
||||
weight = _Obj()
|
||||
weight.quant_state = _Obj()
|
||||
base_layer = _Obj()
|
||||
base_layer.weight = weight
|
||||
base_layer.quant_method = "fp8"
|
||||
base_layer.block_size = block_size
|
||||
base_layer.bias = None
|
||||
proj = _Obj()
|
||||
proj.base_layer = base_layer
|
||||
proj.merged = True
|
||||
proj.disable_adapters = True
|
||||
return proj, weight.quant_state
|
||||
|
||||
|
||||
def test_bias_variant_propagates_fp8_block_size_on_disabled_path():
|
||||
# Downstream fp8 kernels read getattr(weight_scale, "block_size", [128, 128]),
|
||||
# so the checkpoint's real block size must survive the merged/disabled path,
|
||||
# exactly as it does for the non-bias sibling get_lora_parameters.
|
||||
get_lora_parameters_bias = _load_function("get_lora_parameters_bias")
|
||||
|
||||
proj, weight_scale = _make_disabled_block_fp8_proj([64, 128])
|
||||
get_lora_parameters_bias(proj)
|
||||
|
||||
assert getattr(weight_scale, "block_size", [128, 128]) == [64, 128]
|
||||
|
||||
|
||||
def _make_decompressed_merged_proj():
|
||||
# A merged compressed-tensors layer that was decompressed back to bf16. It keeps
|
||||
# quant_method == "fp8" from the checkpoint metadata, but the live weight is bf16
|
||||
# so there is no quant state to attach a block size to.
|
||||
weight = _Obj()
|
||||
weight.dtype = "bfloat16"
|
||||
base_layer = _Obj()
|
||||
base_layer.weight = weight
|
||||
base_layer.quant_method = "fp8"
|
||||
base_layer.block_size = [128, 128]
|
||||
base_layer.bias = None
|
||||
proj = _Obj()
|
||||
proj.base_layer = base_layer
|
||||
proj.merged = True
|
||||
proj.disable_adapters = True
|
||||
return proj
|
||||
|
||||
|
||||
def test_bias_variant_keeps_none_quant_state_for_decompressed_layer():
|
||||
# Such a layer has no quant state, and fast_linear_forward relies on getting
|
||||
# W_quant None back so it can fall back to a plain matmul, so setting the block
|
||||
# size must not assume a quant state is present.
|
||||
get_lora_parameters_bias = _load_function("get_lora_parameters_bias")
|
||||
|
||||
W, W_quant = get_lora_parameters_bias(_make_decompressed_merged_proj())[:2]
|
||||
|
||||
assert W_quant is None
|
||||
assert getattr(W, "block_size", None) == [128, 128]
|
||||
81
tests/python/test_get_lora_parameters_fp8_block_size.py
Normal file
81
tests/python/test_get_lora_parameters_fp8_block_size.py
Normal file
|
|
@ -0,0 +1,81 @@
|
|||
import ast
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def _load_function(name):
|
||||
# Extract a function from kernels/utils.py without importing unsloth (which
|
||||
# needs a GPU / torch / bitsandbytes). get_lora_parameters only uses getattr,
|
||||
# hasattr and the _FP8_WEIGHT_DTYPES name on the paths under test.
|
||||
source = Path(__file__).parents[2] / "unsloth" / "kernels" / "utils.py"
|
||||
tree = ast.parse(source.read_text(encoding = "utf-8"))
|
||||
funcs = [
|
||||
node for node in ast.walk(tree) if isinstance(node, ast.FunctionDef) and node.name == name
|
||||
]
|
||||
assert len(funcs) == 1, (name, funcs)
|
||||
namespace = {"getattr": getattr, "hasattr": hasattr, "_FP8_WEIGHT_DTYPES": ()}
|
||||
module = ast.Module(body = funcs, type_ignores = [])
|
||||
ast.fix_missing_locations(module)
|
||||
exec(compile(module, str(source), "exec"), namespace)
|
||||
return namespace[name]
|
||||
|
||||
|
||||
class _Obj:
|
||||
pass
|
||||
|
||||
|
||||
def _make_disabled_block_fp8_proj(block_size):
|
||||
# A merged/disabled projection whose base layer is a block-fp8 weight that
|
||||
# ships a non-default block size on its checkpoint.
|
||||
weight = _Obj()
|
||||
weight.quant_state = _Obj()
|
||||
base_layer = _Obj()
|
||||
base_layer.weight = weight
|
||||
base_layer.quant_method = "fp8"
|
||||
base_layer.block_size = block_size
|
||||
proj = _Obj()
|
||||
proj.base_layer = base_layer
|
||||
proj.merged = True
|
||||
proj.disable_adapters = True
|
||||
return proj, weight.quant_state
|
||||
|
||||
|
||||
def test_propagates_fp8_block_size_on_disabled_path():
|
||||
# get_lora_parameters already sets block_size before its early return; downstream
|
||||
# fp8 kernels read getattr(weight_scale, "block_size", [128, 128]), so the
|
||||
# checkpoint's real block size must survive the merged/disabled path.
|
||||
get_lora_parameters = _load_function("get_lora_parameters")
|
||||
|
||||
proj, weight_scale = _make_disabled_block_fp8_proj([64, 128])
|
||||
get_lora_parameters(proj)
|
||||
|
||||
assert getattr(weight_scale, "block_size", [128, 128]) == [64, 128]
|
||||
|
||||
|
||||
def _make_decompressed_merged_proj():
|
||||
# A merged compressed-tensors layer that was decompressed back to bf16. It keeps
|
||||
# quant_method == "fp8" from the checkpoint metadata, but the live weight is bf16
|
||||
# so there is no quant state to attach a block size to.
|
||||
weight = _Obj()
|
||||
weight.dtype = "bfloat16"
|
||||
base_layer = _Obj()
|
||||
base_layer.weight = weight
|
||||
base_layer.quant_method = "fp8"
|
||||
base_layer.block_size = [128, 128]
|
||||
proj = _Obj()
|
||||
proj.base_layer = base_layer
|
||||
proj.merged = True
|
||||
proj.disable_adapters = True
|
||||
return proj
|
||||
|
||||
|
||||
def test_keeps_none_quant_state_for_decompressed_layer():
|
||||
# Mirrors the get_lora_parameters_bias guard: with no quant state, assigning
|
||||
# W_quant.block_size must not assume one is present, or it raises AttributeError
|
||||
# on None. fast_lora relies on getting W_quant None back to fall back to a plain
|
||||
# matmul, so this path must stay crash-free.
|
||||
get_lora_parameters = _load_function("get_lora_parameters")
|
||||
|
||||
W, W_quant = get_lora_parameters(_make_decompressed_merged_proj())[:2]
|
||||
|
||||
assert W_quant is None
|
||||
assert getattr(W, "block_size", None) == [128, 128]
|
||||
307
tests/saving/test_fix_sentencepiece_tokenizer_guard.py
Normal file
307
tests/saving/test_fix_sentencepiece_tokenizer_guard.py
Normal file
|
|
@ -0,0 +1,307 @@
|
|||
# SPDX-License-Identifier: AGPL-3.0-only
|
||||
import gc
|
||||
import os
|
||||
|
||||
os.environ.setdefault("PROTOCOL_BUFFERS_PYTHON_IMPLEMENTATION", "python")
|
||||
|
||||
import transformers
|
||||
from transformers.utils import sentencepiece_model_pb2
|
||||
|
||||
from unsloth.tokenizer_utils import fix_sentencepiece_tokenizer
|
||||
|
||||
|
||||
NORMAL, CONTROL = 1, 3
|
||||
|
||||
|
||||
def _spm_bytes(pieces):
|
||||
m = sentencepiece_model_pb2.ModelProto()
|
||||
for piece, score, typ in pieces:
|
||||
p = m.pieces.add()
|
||||
p.piece = piece
|
||||
p.score = score
|
||||
p.type = typ
|
||||
return m.SerializeToString()
|
||||
|
||||
|
||||
def _read_pieces(path):
|
||||
m = sentencepiece_model_pb2.ModelProto()
|
||||
with open(path, "rb") as f:
|
||||
m.ParseFromString(f.read())
|
||||
return [p.piece for p in m.pieces]
|
||||
|
||||
|
||||
class _FakeTokenizer:
|
||||
"""Minimal stand-in for a sentencepiece-backed slow tokenizer.
|
||||
|
||||
``save_pretrained`` writes a tokenizer.model, which is what the real slow
|
||||
tokenizers do and what fix_sentencepiece_tokenizer reads back.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
name,
|
||||
spm_bytes = None,
|
||||
vocab = None,
|
||||
):
|
||||
self.name = name
|
||||
self.eos_token = "</s>"
|
||||
self.pad_token = "<pad>"
|
||||
self._spm_bytes = spm_bytes
|
||||
self._vocab = vocab or {}
|
||||
self.saved_to = []
|
||||
|
||||
def save_pretrained(self, location):
|
||||
self.saved_to.append(location)
|
||||
os.makedirs(location, exist_ok = True)
|
||||
if self._spm_bytes is not None:
|
||||
with open(os.path.join(location, "tokenizer.model"), "wb") as f:
|
||||
f.write(self._spm_bytes)
|
||||
|
||||
def __call__(
|
||||
self,
|
||||
texts,
|
||||
add_special_tokens = False,
|
||||
):
|
||||
class _Encoded:
|
||||
pass
|
||||
|
||||
encoded = _Encoded()
|
||||
encoded.input_ids = [[self._vocab[text]] for text in texts]
|
||||
return encoded
|
||||
|
||||
|
||||
def _tokenizers():
|
||||
pieces = [("<s>", 0.0, CONTROL), ("a", -1.0, NORMAL), ("</s>", 0.0, CONTROL)]
|
||||
old = _FakeTokenizer("old", spm_bytes = _spm_bytes(pieces), vocab = {"</s>": 2})
|
||||
new = _FakeTokenizer("new")
|
||||
return old, new
|
||||
|
||||
|
||||
class _ReloadedTokenizer:
|
||||
"""Weakref-able stand-in for the tokenizer AutoTokenizer.from_pretrained returns."""
|
||||
|
||||
def __init__(self, location):
|
||||
self.location = location
|
||||
|
||||
|
||||
def _stub_auto_tokenizer(monkeypatch):
|
||||
"""fix_sentencepiece_tokenizer reloads the patched directory through
|
||||
AutoTokenizer at the end; that needs a full tokenizer on disk, which is
|
||||
out of scope here. Record the reload location and hand back a sentinel.
|
||||
"""
|
||||
loaded = []
|
||||
|
||||
class _StubAutoTokenizer:
|
||||
@staticmethod
|
||||
def from_pretrained(location, **kwargs):
|
||||
loaded.append(location)
|
||||
return _ReloadedTokenizer(location)
|
||||
|
||||
monkeypatch.setattr(transformers, "AutoTokenizer", _StubAutoTokenizer)
|
||||
return loaded
|
||||
|
||||
|
||||
def test_old_tokenizer_is_saved_so_its_model_can_be_read(tmp_path, monkeypatch):
|
||||
"""The guard must not skip the body on a fresh temporary directory.
|
||||
|
||||
fix_sentencepiece_tokenizer creates its scratch directory itself and then
|
||||
checks for a tokenizer.model inside it, but that file only appears once
|
||||
old_tokenizer.save_pretrained() has run.
|
||||
"""
|
||||
_stub_auto_tokenizer(monkeypatch)
|
||||
old, new = _tokenizers()
|
||||
location = str(tmp_path / "_unsloth_sentencepiece_temp")
|
||||
|
||||
fix_sentencepiece_tokenizer(old, new, {"</s>": "<|im_end|>"}, temporary_location = location)
|
||||
|
||||
assert old.saved_to, "old tokenizer was never saved: the body did not run"
|
||||
|
||||
|
||||
def test_token_mapping_is_applied_to_the_sentencepiece_model(tmp_path, monkeypatch):
|
||||
loaded = _stub_auto_tokenizer(monkeypatch)
|
||||
old, new = _tokenizers()
|
||||
location = str(tmp_path / "_unsloth_sentencepiece_temp")
|
||||
|
||||
# Hold the returned tokenizer so its scratch dir survives until we read it.
|
||||
tok = fix_sentencepiece_tokenizer(old, new, {"</s>": "<|im_end|>"}, temporary_location = location)
|
||||
|
||||
assert "<|im_end|>" in _read_pieces(f"{loaded[-1]}/tokenizer.model")
|
||||
assert tok is not None
|
||||
|
||||
|
||||
def test_tokenizer_without_a_sentencepiece_model_is_returned_untouched(tmp_path, monkeypatch):
|
||||
"""A fast-only tokenizer writes no tokenizer.model, so the guard still
|
||||
short-circuits and the caller gets new_tokenizer back unchanged. Its scratch
|
||||
dir is unreferenced and reclaimed immediately.
|
||||
"""
|
||||
_stub_auto_tokenizer(monkeypatch)
|
||||
old = _FakeTokenizer("old", spm_bytes = None)
|
||||
new = _FakeTokenizer("new")
|
||||
location = str(tmp_path / "_unsloth_sentencepiece_temp")
|
||||
|
||||
result = fix_sentencepiece_tokenizer(
|
||||
old, new, {"</s>": "<|im_end|>"}, temporary_location = location
|
||||
)
|
||||
|
||||
assert result is new
|
||||
assert not any(
|
||||
name.startswith("tokenizer_") for name in os.listdir(location)
|
||||
), "the fast-only scratch dir was not reclaimed"
|
||||
|
||||
|
||||
def test_each_call_uses_a_fresh_isolated_subdirectory(tmp_path, monkeypatch):
|
||||
"""Each call must work in its own unique subdirectory, so concurrent or
|
||||
repeated calls never share scratch files, stale artifacts never leak into
|
||||
the reload, and nothing the caller left in the scratch location is deleted.
|
||||
"""
|
||||
loaded = _stub_auto_tokenizer(monkeypatch)
|
||||
location = str(tmp_path / "_unsloth_sentencepiece_temp")
|
||||
os.makedirs(location, exist_ok = True)
|
||||
|
||||
# A pre-existing artifact in the shared scratch location.
|
||||
marker = os.path.join(location, "leftover.json")
|
||||
with open(marker, "w") as f:
|
||||
f.write("{}")
|
||||
|
||||
old1, new1 = _tokenizers()
|
||||
old2, new2 = _tokenizers()
|
||||
# Hold both returned tokenizers so their scratch dirs stay alive.
|
||||
tok1 = fix_sentencepiece_tokenizer(
|
||||
old1, new1, {"</s>": "<|im_end|>"}, temporary_location = location
|
||||
)
|
||||
tok2 = fix_sentencepiece_tokenizer(
|
||||
old2, new2, {"</s>": "<|im_end|>"}, temporary_location = location
|
||||
)
|
||||
|
||||
work1, work2 = loaded[0], loaded[1]
|
||||
assert work1 != work2, "two calls reused the same directory"
|
||||
assert os.path.dirname(work1) == location and os.path.dirname(work2) == location
|
||||
assert os.path.isdir(work1) and os.path.isdir(work2)
|
||||
# Nothing the caller left behind is deleted, and it never leaks into a work dir.
|
||||
assert os.path.isfile(marker), "a pre-existing scratch file was deleted"
|
||||
assert not os.path.isfile(os.path.join(work1, "leftover.json"))
|
||||
assert not os.path.isfile(os.path.join(work2, "leftover.json"))
|
||||
assert tok1 is not None and tok2 is not None
|
||||
|
||||
|
||||
def test_sentencepiece_scratch_dir_is_reclaimed_once_the_tokenizer_is_gone(tmp_path, monkeypatch):
|
||||
"""The scratch dir must live as long as the returned tokenizer (its vocab_file
|
||||
points there), then be reclaimed when the tokenizer is garbage collected.
|
||||
"""
|
||||
loaded = _stub_auto_tokenizer(monkeypatch)
|
||||
old, new = _tokenizers()
|
||||
location = str(tmp_path / "_unsloth_sentencepiece_temp")
|
||||
|
||||
tok = fix_sentencepiece_tokenizer(old, new, {"</s>": "<|im_end|>"}, temporary_location = location)
|
||||
work = loaded[-1]
|
||||
assert os.path.isdir(work), "scratch dir vanished while the tokenizer was alive"
|
||||
|
||||
del tok
|
||||
gc.collect()
|
||||
assert not os.path.isdir(work), "scratch dir was not reclaimed after the tokenizer was freed"
|
||||
|
||||
|
||||
class _CopyFromSubdirTokenizer:
|
||||
"""A slow tokenizer whose sentencepiece source lives elsewhere (like the
|
||||
tokenizers convert_to_fast_tokenizer produces under {location}/{name}).
|
||||
save_pretrained copies that source into the destination, as HF slow
|
||||
tokenizers copy their vocab_file.
|
||||
"""
|
||||
|
||||
def __init__(self, source_model_path):
|
||||
self.eos_token = "</s>"
|
||||
self.pad_token = "<pad>"
|
||||
self._source_model_path = source_model_path
|
||||
|
||||
def save_pretrained(self, location):
|
||||
os.makedirs(location, exist_ok = True)
|
||||
if os.path.isfile(self._source_model_path):
|
||||
with open(self._source_model_path, "rb") as src:
|
||||
data = src.read()
|
||||
with open(os.path.join(location, "tokenizer.model"), "wb") as dst:
|
||||
dst.write(data)
|
||||
|
||||
def __call__(
|
||||
self,
|
||||
texts,
|
||||
add_special_tokens = False,
|
||||
):
|
||||
class _Encoded:
|
||||
pass
|
||||
|
||||
encoded = _Encoded()
|
||||
encoded.input_ids = [[2] for _ in texts]
|
||||
return encoded
|
||||
|
||||
|
||||
def test_source_vocab_outside_the_work_directory_is_not_disturbed(tmp_path, monkeypatch):
|
||||
"""A tokenizer whose sentencepiece source lives elsewhere (e.g. the subtree
|
||||
convert_to_fast_tokenizer created) is copied into the fresh work directory
|
||||
and patched there; the original source is left untouched.
|
||||
"""
|
||||
loaded = _stub_auto_tokenizer(monkeypatch)
|
||||
location = str(tmp_path / "_unsloth_sentencepiece_temp")
|
||||
subdir = os.path.join(location, "some_model")
|
||||
os.makedirs(subdir, exist_ok = True)
|
||||
|
||||
pieces = [("<s>", 0.0, CONTROL), ("a", -1.0, NORMAL), ("</s>", 0.0, CONTROL)]
|
||||
source_model = os.path.join(subdir, "tokenizer.model")
|
||||
with open(source_model, "wb") as f:
|
||||
f.write(_spm_bytes(pieces))
|
||||
|
||||
old = _CopyFromSubdirTokenizer(source_model)
|
||||
new = _FakeTokenizer("new")
|
||||
tok = fix_sentencepiece_tokenizer(old, new, {"</s>": "<|im_end|>"}, temporary_location = location)
|
||||
|
||||
assert _read_pieces(source_model) == [
|
||||
"<s>",
|
||||
"a",
|
||||
"</s>",
|
||||
], "the original source vocab was modified"
|
||||
assert "<|im_end|>" in _read_pieces(f"{loaded[-1]}/tokenizer.model")
|
||||
assert tok is not None
|
||||
|
||||
|
||||
def test_swap_mapping_swaps_both_pieces_without_duplicating(tmp_path, monkeypatch):
|
||||
"""When the caller swaps eos and stop_word in the fast JSON it must pass both
|
||||
directions here; a one-way mapping would leave two stop_word pieces and no eos.
|
||||
"""
|
||||
loaded = _stub_auto_tokenizer(monkeypatch)
|
||||
location = str(tmp_path / "_unsloth_sentencepiece_temp")
|
||||
|
||||
pieces = [("<s>", 0.0, CONTROL), ("<|im_end|>", -1.0, NORMAL), ("</s>", 0.0, CONTROL)]
|
||||
old = _FakeTokenizer("old", spm_bytes = _spm_bytes(pieces), vocab = {"</s>": 2, "<|im_end|>": 1})
|
||||
new = _FakeTokenizer("new")
|
||||
|
||||
tok = fix_sentencepiece_tokenizer(
|
||||
old, new, {"</s>": "<|im_end|>", "<|im_end|>": "</s>"}, temporary_location = location
|
||||
)
|
||||
|
||||
result = _read_pieces(f"{loaded[-1]}/tokenizer.model")
|
||||
assert result.count("<|im_end|>") == 1 and result.count("</s>") == 1, result
|
||||
assert tok is not None
|
||||
|
||||
|
||||
def test_only_applied_mappings_are_patched(tmp_path, monkeypatch):
|
||||
"""When the caller skips a mapping whose target already exists, it must not
|
||||
pass that mapping here, or the skipped source token gets renamed anyway and
|
||||
duplicates the existing target in the model.
|
||||
"""
|
||||
loaded = _stub_auto_tokenizer(monkeypatch)
|
||||
location = str(tmp_path / "_unsloth_sentencepiece_temp")
|
||||
|
||||
pieces = [
|
||||
("<s>", 0.0, CONTROL),
|
||||
("aa", -1.0, NORMAL),
|
||||
("bb", -1.0, NORMAL),
|
||||
("X", -1.0, NORMAL),
|
||||
]
|
||||
old = _FakeTokenizer("old", spm_bytes = _spm_bytes(pieces), vocab = {"aa": 1, "bb": 2})
|
||||
new = _FakeTokenizer("new")
|
||||
|
||||
# Caller skipped aa->X (X already exists) and applied bb->Y, so only bb->Y is passed.
|
||||
tok = fix_sentencepiece_tokenizer(old, new, {"bb": "Y"}, temporary_location = location)
|
||||
|
||||
result = _read_pieces(f"{loaded[-1]}/tokenizer.model")
|
||||
assert result.count("X") == 1 and "Y" in result and "aa" in result, result
|
||||
assert tok is not None
|
||||
|
|
@ -34,17 +34,22 @@ def test_model_selector_trigger_label_uses_leading_tight():
|
|||
|
||||
def test_sidebar_account_block_uses_leading_tight():
|
||||
src = _read(APP_SIDEBAR)
|
||||
# Match the account-block parent div regardless of its gap utility; this
|
||||
# guard is about the leading-* class, not the spacing.
|
||||
pattern = re.compile(
|
||||
r'<div\s+className="flex\s+flex-col\s+gap-\S+\s+(\S+)\s+group-data-\[collapsible=icon\]:hidden">',
|
||||
)
|
||||
matches = pattern.findall(src)
|
||||
class_names = re.findall(r'<div\s+className="([^"]+)"', src)
|
||||
required = {
|
||||
"flex",
|
||||
"flex-1",
|
||||
"flex-col",
|
||||
"group-data-[collapsible=icon]:hidden",
|
||||
}
|
||||
matches = [classes for classes in class_names if required <= set(classes.split())]
|
||||
assert matches, "could not find sidebar account-block parent div"
|
||||
leading_classes = [m for m in matches if m.startswith("leading-")]
|
||||
assert leading_classes, f"no leading-* class on sidebar account-block parent: {matches}"
|
||||
for cls in leading_classes:
|
||||
assert cls == "leading-tight", f"sidebar account-block must use leading-tight, got: {cls}"
|
||||
for classes in matches:
|
||||
leading_classes = [cls for cls in classes.split() if cls.startswith("leading-")]
|
||||
assert leading_classes, f"no leading-* class on sidebar account-block parent: {classes}"
|
||||
for cls in leading_classes:
|
||||
assert (
|
||||
cls == "leading-tight"
|
||||
), f"sidebar account-block must use leading-tight, got: {cls}"
|
||||
|
||||
|
||||
def test_no_truncate_plus_leading_none_in_changed_files():
|
||||
|
|
|
|||
|
|
@ -295,7 +295,25 @@ def test_smart_chunk_text_single_chunk_no_eos_returns_plain_list():
|
|||
return True
|
||||
|
||||
|
||||
def test_load_from_file_skips_non_object_json_lines():
|
||||
"""Non-object .jsonl lines (valid JSON, not dicts) are skipped, not fatal."""
|
||||
# "context" contains "text", ["text"] holds it, 42 isn't iterable -- each
|
||||
# would reach data[field] and raise TypeError without the isinstance guard.
|
||||
with tempfile.NamedTemporaryFile("w", suffix = ".jsonl", delete = False) as f:
|
||||
f.write('"context"\n["text", "x"]\n42\n{"text": "keep this"}\n')
|
||||
path = f.name
|
||||
try:
|
||||
text = RawTextDataLoader(None)._read_file_by_format(path, "json_lines")
|
||||
assert text == "keep this", text
|
||||
finally:
|
||||
os.unlink(path)
|
||||
|
||||
print("test_load_from_file_skips_non_object_json_lines passed")
|
||||
return True
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
success = test_raw_text_loader()
|
||||
success = test_smart_chunk_text_single_chunk_no_eos_returns_plain_list() and success
|
||||
success = test_load_from_file_skips_non_object_json_lines() and success
|
||||
sys.exit(0 if success else 1)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue