From 2b3642e365cbd3c404fed80f69855945a6ed6fd7 Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Fri, 26 Jun 2026 18:25:07 +0000 Subject: [PATCH] Address review: main-process guard, calibration subsampling, tokenizer + dtype handling - Route the 16bit merge through unsloth_generic_save for both LoRA and full finetuned models, so non-PEFT models are written in 16bit consistently instead of saving the original (possibly quantized) weights directly. - Honor is_main_process: only the main process quantizes and writes the compressed output, so distributed ranks do not race on the same dirs. - Subsample an in-memory calibration Dataset before save_to_disk so large training sets are not fully copied to a temp dir. - Tolerate a missing tokenizer in the converter (data-free exports); still require one for calibration based schemes. - Open config.json via a context manager in both files. - Drop the redundant nvfp4 entry from the unsupported-name check (fp4 covers it). --- unsloth/_compressed_quantize.py | 19 ++++++++-- unsloth/save.py | 63 +++++++++++++++++++-------------- 2 files changed, 52 insertions(+), 30 deletions(-) diff --git a/unsloth/_compressed_quantize.py b/unsloth/_compressed_quantize.py index 69e947f3ea..015a23a975 100644 --- a/unsloth/_compressed_quantize.py +++ b/unsloth/_compressed_quantize.py @@ -146,7 +146,16 @@ def main(): model = _from_pretrained(auto_model, args.model, args.trust_remote_code) model.eval() - tokenizer = auto_proc.from_pretrained(args.model, trust_remote_code = args.trust_remote_code) + # A tokenizer may be absent if the caller saved it separately; only calibration needs one. + try: + tokenizer = auto_proc.from_pretrained(args.model, trust_remote_code = args.trust_remote_code) + except Exception: + if args.needs_calibration: + raise RuntimeError( + f"Unsloth: calibration export needs a tokenizer but none was found in {args.model}. " + "Pass tokenizer=... to save_pretrained_merged." + ) + tokenizer = None recipe = QuantizationModifier(targets = "Linear", scheme = args.scheme, ignore = ["lm_head"]) if args.needs_calibration: @@ -171,10 +180,14 @@ def main(): os.makedirs(args.out, exist_ok = True) model.save_pretrained(args.out, save_compressed = True) - tokenizer.save_pretrained(args.out) + if tokenizer is not None: + tokenizer.save_pretrained(args.out) cfg_path = os.path.join(args.out, "config.json") - cfg = json.load(open(cfg_path)) if os.path.exists(cfg_path) else {} + cfg = {} + if os.path.exists(cfg_path): + with open(cfg_path, "r", encoding="utf-8") as f: + cfg = json.load(f) if "quantization_config" not in cfg: print(f"Unsloth: ERROR - no quantization_config written to {cfg_path}", flush = True) sys.exit(2) diff --git a/unsloth/save.py b/unsloth/save.py index ecb1629fc9..0ea734950e 100644 --- a/unsloth/save.py +++ b/unsloth/save.py @@ -177,7 +177,7 @@ def _normalize_compressed_method(save_method): key = save_method.lower().strip().replace("-", "_").replace(" ", "_") if key in COMPRESSED_EXPORT_SCHEMES: return COMPRESSED_EXPORT_SCHEMES[key] - if any(tag in key for tag in ("fp8", "fp4", "nvfp4", "mxfp")): + if any(tag in key for tag in ("fp8", "fp4", "mxfp")): supported = ", ".join(sorted(COMPRESSED_EXPORT_SCHEMES.keys())) raise RuntimeError( f"Unsloth: save_method='{save_method}' is not a supported compressed export.\n" @@ -1845,6 +1845,7 @@ def unsloth_save_pretrained_merged( suffix = suffix, push_to_hub = push_to_hub, token = token, + is_main_process = is_main_process, calibration_dataset = calibration_dataset, num_calibration_samples = num_calibration_samples, max_seq_length = max_seq_length, @@ -3397,6 +3398,7 @@ def unsloth_generic_save_pretrained_merged( suffix = suffix, push_to_hub = push_to_hub, token = token, + is_main_process = is_main_process, calibration_dataset = calibration_dataset, num_calibration_samples = num_calibration_samples, max_seq_length = max_seq_length, @@ -3672,6 +3674,7 @@ def _unsloth_save_compressed_tensors( suffix: str, push_to_hub: bool = False, token: Optional[Union[str, bool]] = None, + is_main_process: bool = True, calibration_dataset = None, num_calibration_samples: int = 512, max_seq_length: int = 2048, @@ -3692,35 +3695,31 @@ def _unsloth_save_compressed_tensors( # Accept os.PathLike save directories (string concatenation below needs a str). save_directory = os.fspath(save_directory) - # 1) Merge LoRA -> 16bit on disk and KEEP it. Must be a local save: merge_and_overwrite_lora - # deletes save_directory when push_to_hub=True, which would remove it before reload. - is_peft = isinstance(model, PeftModelForCausalLM) or isinstance(model, PeftModel) - if is_peft: - print(f"Unsloth: Merging LoRA weights to 16bit before {scheme} quantization...") - merge_args = dict(merge_kwargs) - merge_args.update( - dict( - model = model, - tokenizer = tokenizer, - save_directory = save_directory, - save_method = "merged_16bit", - push_to_hub = False, - token = token, - ) - ) - unsloth_generic_save(**merge_args) - else: - print(f"Unsloth: Saving base model to 16bit before {scheme} quantization...") - os.makedirs(save_directory, exist_ok = True) - model.save_pretrained(save_directory) - if tokenizer is not None: - tokenizer.save_pretrained(save_directory) + # 1) Merge to 16bit on disk and KEEP it, via unsloth_generic_save so LoRA adapters are merged + # and full-finetuned models are written in 16bit consistently. Must be a local save: + # merge_and_overwrite_lora deletes save_directory when push_to_hub=True. + print(f"Unsloth: Merging to 16bit before {scheme} quantization...") + merge_args = dict(merge_kwargs) + merge_args.update(dict( + model = model, + tokenizer = tokenizer, + save_directory = save_directory, + save_method = "merged_16bit", + push_to_hub = False, + token = token, + is_main_process = is_main_process, + )) + unsloth_generic_save(**merge_args) for _ in range(3): gc.collect() if torch.cuda.is_available(): torch.cuda.empty_cache() + # Only the main process quantizes and writes the compressed output (avoids ranks racing). + if not is_main_process: + return None + # 2) Lazily import / install llm-compressor (without breaking torch / transformers) and # gate on scheme availability up-front so the user gets a clear error fast. install_llm_compressor() @@ -3765,11 +3764,18 @@ def _unsloth_save_compressed_tensors( calib_kind = "disk" if os.path.isdir(calib_value) else "hfid" elif hasattr(calibration_dataset, "save_to_disk"): import tempfile - + # Only persist the samples we need, so multi-GB training sets are not fully copied. + ds_to_save = calibration_dataset + try: + if (num_calibration_samples and hasattr(ds_to_save, "select") + and len(ds_to_save) > num_calibration_samples): + ds_to_save = ds_to_save.shuffle(seed = 42).select(range(num_calibration_samples)) + except Exception: + ds_to_save = calibration_dataset parent = os.path.dirname(os.path.abspath(save_directory)) or None calib_tmp = tempfile.mkdtemp(prefix = "unsloth-calib-", dir = parent) shutil.rmtree(calib_tmp, ignore_errors = True) # save_to_disk wants a fresh path - calibration_dataset.save_to_disk(calib_tmp) + ds_to_save.save_to_disk(calib_tmp) calib_kind, calib_value = "disk", calib_tmp else: raise TypeError( @@ -3831,7 +3837,10 @@ def _unsloth_save_compressed_tensors( # 6) Validate the artifact. cfg_path = os.path.join(out_dir, "config.json") - cfg = json.load(open(cfg_path)) if os.path.exists(cfg_path) else {} + cfg = {} + if os.path.exists(cfg_path): + with open(cfg_path, "r", encoding = "utf-8") as f: + cfg = json.load(f) if "quantization_config" not in cfg: raise RuntimeError( f"Unsloth: {scheme} export failed - no quantization_config written to {cfg_path}"