Disable flex attention on Blackwell+ GPUs (sm_120+) at startup
This commit is contained in:
parent
346d7cd13a
commit
5a02ed4f0f
1 changed files with 14 additions and 1 deletions
|
|
@ -1,6 +1,7 @@
|
|||
"""
|
||||
Main FastAPI application for Unsloth UI Backend
|
||||
"""
|
||||
import os
|
||||
import secrets
|
||||
import shutil
|
||||
from contextlib import asynccontextmanager
|
||||
|
|
@ -15,7 +16,7 @@ from datetime import datetime
|
|||
# Import routers
|
||||
from routes import training_router, models_router, inference_router, datasets_router, auth_router, export_router
|
||||
from auth import storage
|
||||
from utils.hardware import detect_hardware
|
||||
from utils.hardware import detect_hardware, get_device, DeviceType
|
||||
import utils.hardware.hardware as _hw_module
|
||||
|
||||
UNSLOTH_CACHE_DIR = Path(__file__).parent / "unsloth_compiled_cache"
|
||||
|
|
@ -27,6 +28,18 @@ async def lifespan(app: FastAPI):
|
|||
# Detect hardware first — sets DEVICE global used everywhere
|
||||
detect_hardware()
|
||||
|
||||
# Disable flex attention on Blackwell+ GPUs (sm_120 and above)
|
||||
if get_device() == DeviceType.CUDA:
|
||||
import torch
|
||||
props = torch.cuda.get_device_properties(0)
|
||||
sm_version = props.major * 10 + props.minor
|
||||
if sm_version >= 120:
|
||||
os.environ["UNSLOTH_FLEX_ATTENTION"] = "0"
|
||||
import logging
|
||||
logging.getLogger(__name__).info(
|
||||
f"GPU sm_{sm_version} detected — setting UNSLOTH_FLEX_ATTENTION=0"
|
||||
)
|
||||
|
||||
if not storage.is_initialized():
|
||||
setup_token = secrets.token_urlsafe(32)
|
||||
storage.save_setup_token(setup_token)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue