add code for intel qlora (#3370)

* add code for intel qlora

* add specified code for xpu device
This commit is contained in:
Lei Zhenyuan 2025-10-27 12:44:29 +08:00 committed by GitHub
commit 24cf1f88e0
3 changed files with 22 additions and 37 deletions

View file

@ -202,7 +202,7 @@ elif DEVICE_TYPE == "hip":
# NO-OP for rocm device
pass
elif DEVICE_TYPE == "xpu":
# currently intel xpu will not support bnb, will add support in the future
import bitsandbytes as bnb
# TODO: check triton for intel installed properly.
pass

View file

@ -90,19 +90,13 @@ def calculate_settings(n : int) -> (int, int,):
pass
HAS_CUDA_STREAM = False
# INTEL GPU specific logic
if DEVICE_TYPE == "xpu":
# TODO: Changed here after adding XPU BNB support
HAS_XPU_STREAM = True
def get_ptr(x: Optional[torch.Tensor]):
raise RuntimeError("XPU BNB support is not implemented yet. This function should not be called.")
else:
# NVIDIA-GPU logic here as default
import bitsandbytes as bnb
# https://github.com/bitsandbytes-foundation/bitsandbytes/pull/1330/files
HAS_CUDA_STREAM = Version(bnb.__version__) > Version("0.43.3")
get_ptr = bnb.functional.get_ptr
import bitsandbytes as bnb
# https://github.com/bitsandbytes-foundation/bitsandbytes/pull/1330/files
HAS_CUDA_STREAM = Version(bnb.__version__) > Version("0.43.3")
get_ptr = bnb.functional.get_ptr
if DEVICE_TYPE == "xpu":
HAS_XPU_STREAM = True
if DEVICE_COUNT > 1:
if DEVICE_TYPE in ("cuda", "hip"):
@ -163,31 +157,19 @@ pass
# Bitsandbytes operations
ctypes_c_int = ctypes.c_int
ctypes_c_int32 = ctypes.c_int32
# INTEL GPU Specific Logic
cdequantize_blockwise_fp32 = bnb.functional.lib.cdequantize_blockwise_fp32
cdequantize_blockwise_fp16_nf4 = bnb.functional.lib.cdequantize_blockwise_fp16_nf4
cdequantize_blockwise_bf16_nf4 = bnb.functional.lib.cdequantize_blockwise_bf16_nf4
if DEVICE_TYPE == "xpu":
# TODO: After adding XPU BNB support, this function should be implemented
def cdequantize_blockwise_fp32(*args, **kwargs):
raise RuntimeError("XPU BNB support is not implemented yet. cdequantize_blockwise_fp32 should not be called now.")
def cdequantize_blockwise_fp16_nf4(*args, **kwargs):
raise RuntimeError("XPU BNB support is not implemented yet. cdequantize_blockwise_fp16_nf4 should not be called now.")
def cdequantize_blockwise_bf16_nf4(*args, **kwargs):
raise RuntimeError("XPU BNB support is not implemented yet. cdequantize_blockwise_bf16_nf4 should not be called now.")
def cgemm_4bit_inference_naive_fp16(*args, **kwargs):
raise RuntimeError("XPU BNB support is not implemented yet. cgemm_4bit_inference_naive_fp16 should not be called now.")
def cgemm_4bit_inference_naive_bf16(*args, **kwargs):
raise RuntimeError("XPU BNB support is not implemented yet. cgemm_4bit_inference_naive_bf16 should not be called now.")
# https://github.com/bitsandbytes-foundation/bitsandbytes/blob/c3b8de268fdb55a88f92feada23fc811a1e6877a/bitsandbytes/backends/xpu/ops.py#L115
# for xpu, inference gemv using above link
cgemm_4bit_inference_naive_fp16 = bnb.functional.lib.cgemv_4bit_inference_fp16
cgemm_4bit_inference_naive_bf16 = bnb.functional.lib.cgemv_4bit_inference_bf16
else:
# NVIDIA GPU Default Logic
cdequantize_blockwise_fp32 = bnb.functional.lib.cdequantize_blockwise_fp32
cdequantize_blockwise_fp16_nf4 = bnb.functional.lib.cdequantize_blockwise_fp16_nf4
cdequantize_blockwise_bf16_nf4 = bnb.functional.lib.cdequantize_blockwise_bf16_nf4
cgemm_4bit_inference_naive_fp16 = bnb.functional.lib.cgemm_4bit_inference_naive_fp16
cgemm_4bit_inference_naive_bf16 = bnb.functional.lib.cgemm_4bit_inference_naive_bf16
pass
torch_device_stream = torch.xpu.current_stream if DEVICE_TYPE == "xpu" else torch.cuda.current_stream
@ -562,8 +544,12 @@ if DEVICE_TYPE == "xpu" and HAS_XPU_STREAM:
# assert(out.shape == (1, 1, bout,))
# pass
n = 1
m = shape[0]
if DEVICE_TYPE == "xpu":
m = 1
n = shape[0]
else:
n = 1
m = shape[0]
k = shape[1]
lda = shape[0]
ldc = shape[0]

View file

@ -553,8 +553,7 @@ pass
# =============================================
# Get Flash Attention v2 if Ampere (RTX 30xx, A100)
if DEVICE_TYPE in ("cuda", "hip"):
import bitsandbytes as bnb
import bitsandbytes as bnb
from transformers import AutoTokenizer
from transformers.utils.import_utils import _is_package_available