From 03be1d30f45ed86c4b61ff7af7feaf001d779efd Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Sat, 27 Jun 2026 08:23:42 +0000 Subject: [PATCH] Harden compressed export: explicit sequential pipeline, base-tokenizer calibration, GPU memory - nvfp4 calibration now passes pipeline="sequential" explicitly (layer-by-layer onloading) instead of relying on the inferred default, with a "basic" fallback - Calibration datasets with a messages column no longer require a chat template: base / non-chat tokenizers fall back to concatenating message contents - Free the in-memory model's CUDA memory before the quantize subprocess loads its own copy from disk (best-effort, single-device non-quantized only; restored afterward), so a single GPU need not hold two copies at once - Create the calibration temp dir in the system temp location instead of next to the save directory, avoiding stray dirs in the workspace --- unsloth/_compressed_quantize.py | 16 +++++++++++----- unsloth/save.py | 33 ++++++++++++++++++++++++++++++--- 2 files changed, 41 insertions(+), 8 deletions(-) diff --git a/unsloth/_compressed_quantize.py b/unsloth/_compressed_quantize.py index a61c9a5d0d..52976d4b68 100644 --- a/unsloth/_compressed_quantize.py +++ b/unsloth/_compressed_quantize.py @@ -71,9 +71,14 @@ def _build_calibration_dataset(tokenizer, kind, value, num_samples, max_seq_leng if "input_ids" in cols: return ds if "messages" in cols: + # Base / non-chat tokenizers have no chat template; concatenate message contents instead + # of calling apply_chat_template (which would raise). + has_chat_template = bool(getattr(_tok, "chat_template", None)) def _prep(ex): - return {"text": _tok.apply_chat_template(ex["messages"], tokenize = False)} + if has_chat_template: + return {"text": _tok.apply_chat_template(ex["messages"], tokenize = False)} + return {"text": "\n".join(m.get("content", "") for m in ex["messages"])} ds = ds.map(_prep) elif "text" not in cols: @@ -176,10 +181,10 @@ def main(): args.num_calibration_samples, args.max_seq_length, ) - # Let llm-compressor pick its default (sequential) pipeline: it onloads layer-by-layer, - # so models that do not fit in memory at once can still calibrate. Running here in a clean - # process (Unsloth's attention patches are absent) means tracing works; fall back to the - # memory-hungry "basic" pipeline only if tracing fails. + # Use the sequential pipeline: it onloads layer-by-layer, so models that do not fit in + # memory at once can still calibrate. Running here in a clean process (Unsloth's attention + # patches are absent) means tracing works; fall back to the memory-hungry "basic" pipeline + # only if tracing fails. try: oneshot( model = model, @@ -187,6 +192,7 @@ def main(): recipe = _make_recipe(), max_seq_length = args.max_seq_length, num_calibration_samples = args.num_calibration_samples, + pipeline = "sequential", ) except Exception as e: print( diff --git a/unsloth/save.py b/unsloth/save.py index b3e1ab5386..fc2633315f 100644 --- a/unsloth/save.py +++ b/unsloth/save.py @@ -3847,7 +3847,7 @@ def _unsloth_save_compressed_tensors( # 2) Pick the local working dir. For a hub push, save_directory is a repo id, so merge and # quantize inside an isolated temp dir instead of writing ./ into the cwd. - repo_id, work_tmp, calib_tmp = None, None, None + repo_id, work_tmp, calib_tmp, model_dev = None, None, None, None if push_to_hub: repo_id = os.fspath(save_directory) work_tmp = tempfile.mkdtemp(prefix = "unsloth-compressed-") @@ -3929,8 +3929,7 @@ def _unsloth_save_compressed_tensors( ) except Exception: ds_to_save = calibration_dataset - parent = os.path.dirname(os.path.abspath(local_dir)) or None - calib_tmp = tempfile.mkdtemp(prefix = "unsloth-calib-", dir = parent) + calib_tmp = tempfile.mkdtemp(prefix = "unsloth-calib-") shutil.rmtree(calib_tmp, ignore_errors = True) # save_to_disk wants a fresh path ds_to_save.save_to_disk(calib_tmp) calib_kind, calib_value = "disk", calib_tmp @@ -3975,6 +3974,29 @@ def _unsloth_save_compressed_tensors( if trust_remote_code: cmd.append("--trust-remote-code") + # Free the in-memory model's CUDA memory before the subprocess loads its own copy from + # disk, so a single GPU need not hold both at once. Best-effort and restored in finally; + # skipped for quantized or multi-device models where moving is unsafe. + try: + if ( + torch.cuda.is_available() + and hasattr(model, "parameters") + and not getattr(model, "is_loaded_in_4bit", False) + and not getattr(model, "is_loaded_in_8bit", False) + and not getattr(model, "is_quantized", False) + ): + _devs = {str(p.device) for p in model.parameters()} + if len(_devs) == 1 and next(iter(_devs)).startswith("cuda"): + _dev = next(model.parameters()).device + model.to("cpu") + model_dev = _dev # set only after a successful move, so finally can restore + except Exception: + model_dev = None + for _ in range(3): + gc.collect() + if torch.cuda.is_available(): + torch.cuda.empty_cache() + print( f"Unsloth: Quantizing the merged model to {scheme} with llm-compressor " "(in a separate process)..." @@ -4017,6 +4039,11 @@ def _unsloth_save_compressed_tensors( _print_compressed_hw_note(scheme, result) return result finally: + if model_dev is not None: + try: + model.to(model_dev) # restore the model to its original device + except Exception: + pass if calib_tmp is not None and os.path.isdir(calib_tmp): shutil.rmtree(calib_tmp, ignore_errors = True) if work_tmp is not None: