* Add Qwen3.6 inference defaults for Studio Add qwen3.6 family entry to inference_defaults.json with the recommended sampling parameters from Qwen's documentation: temperature=0.7, top_p=0.8, top_k=20, min_p=0.0, presence_penalty=1.5, repetition_penalty=1.0. Without this, Qwen3.6 models fall through to the generic qwen3 pattern which uses different defaults (temperature=0.6, top_p=0.95, no presence_penalty). * Add Qwen3.6-35B-A3B-GGUF to default model lists * Add Qwen3.5/3.6 presence_penalty to thinking toggle and small-model disable logic - Thinking toggle (on-load + button click) now sets presencePenalty: 1.5 for Qwen3.5 and Qwen3.6 models (both thinking-ON and thinking-OFF states) - Small-model thinking-disable check (<9B defaults to no-thinking) extended from Qwen3.5-only to also cover Qwen3.6, in all 3 locations: frontend on-load, frontend refresh, backend llama_cpp.py
53 lines
1.7 KiB
Python
53 lines
1.7 KiB
Python
# SPDX-License-Identifier: AGPL-3.0-only
|
|
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
|
|
|
"""Default model lists for inference, split by platform."""
|
|
|
|
import utils.hardware.hardware as hw
|
|
|
|
DEFAULT_MODELS_GGUF = [
|
|
"unsloth/gemma-4-E2B-it-GGUF",
|
|
"unsloth/gemma-4-E4B-it-GGUF",
|
|
"unsloth/gemma-4-31B-it-GGUF",
|
|
"unsloth/gemma-4-26B-A4B-it-GGUF",
|
|
"unsloth/Qwen3.6-35B-A3B-GGUF",
|
|
"unsloth/Qwen3.5-4B-GGUF",
|
|
"unsloth/Qwen3.5-9B-GGUF",
|
|
"unsloth/Qwen3.5-35B-A3B-GGUF",
|
|
"unsloth/Qwen3.5-0.8B-GGUF",
|
|
"unsloth/Llama-3.2-1B-Instruct-GGUF",
|
|
"unsloth/Llama-3.2-3B-Instruct-GGUF",
|
|
"unsloth/Llama-3.1-8B-Instruct-GGUF",
|
|
"unsloth/gemma-3-1b-it-GGUF",
|
|
"unsloth/gemma-3-4b-it-GGUF",
|
|
"unsloth/Qwen3-4B-GGUF",
|
|
]
|
|
|
|
DEFAULT_MODELS_STANDARD = [
|
|
"unsloth/gemma-4-E2B-it-GGUF",
|
|
"unsloth/gemma-4-E4B-it-GGUF",
|
|
"unsloth/gemma-4-31B-it-GGUF",
|
|
"unsloth/gemma-4-26B-A4B-it-GGUF",
|
|
"unsloth/Qwen3.6-35B-A3B-GGUF",
|
|
"unsloth/Qwen3.5-4B-GGUF",
|
|
"unsloth/Qwen3.5-9B-GGUF",
|
|
"unsloth/Qwen3.5-35B-A3B-GGUF",
|
|
"unsloth/Qwen3.5-0.8B-GGUF",
|
|
"unsloth/gemma-4-E2B-it",
|
|
"unsloth/gemma-4-E4B-it",
|
|
"unsloth/gemma-4-31B-it",
|
|
"unsloth/gemma-4-26B-A4B-it",
|
|
"unsloth/Qwen3-4B-Instruct-2507",
|
|
"unsloth/Meta-Llama-3.1-8B-Instruct-bnb-4bit",
|
|
"unsloth/Mistral-Nemo-Instruct-2407-bnb-4bit",
|
|
"unsloth/Phi-3.5-mini-instruct",
|
|
"unsloth/Gemma-3-4B-it",
|
|
"unsloth/Qwen2-VL-2B-Instruct-bnb-4bit",
|
|
]
|
|
|
|
|
|
def get_default_models() -> list[str]:
|
|
hw.get_device() # ensure detect_hardware() has run
|
|
if hw.CHAT_ONLY:
|
|
return list(DEFAULT_MODELS_GGUF)
|
|
return list(DEFAULT_MODELS_STANDARD)
|