Merge branch 'main' into fix/studio-tool-chat-respawn

This commit is contained in:
danielhanchen 2026-07-25 06:41:00 +00:00
commit 3bfe71b4c7
25 changed files with 1122 additions and 171 deletions

View file

@ -307,6 +307,26 @@ def _native_linux_system_rocm_lib_dirs(binary_dir: str = "") -> "list[str]":
_DEFAULT_MAX_TOKENS_FLOOR = 32768
_DEFAULT_FIRST_TOKEN_TIMEOUT_S = 1200.0 # 20 min
def _finalize_reasoning_only_cumulative(
cumulative: str, reasoning_text: str, finish_reason: Optional[str], promote_reasoning_only: bool
) -> str:
"""Close a live thinking block and promote it only after a clean stop.
Local inference streams cumulative snapshots. Replacing ``<think>...`` with
bare reasoning at EOF makes the final snapshot shorter, so suffix-based
route consumers drop the intended fallback. Keep the snapshot append-only.
A length-truncated thought is not a final answer, so close it without
promotion and let the client surface the ``length`` terminal state. Raw
consumers that do not split reasoning from visible content can disable the
fallback to avoid returning the same reasoning twice.
"""
visible_fallback = (
reasoning_text if promote_reasoning_only and finish_reason != "length" else ""
)
return cumulative + "</think>" + visible_fallback
# Only large streamed tool payloads get an early provisional card; render_html
# is exempt because it needs immediate artifact feedback.
_PROVISIONAL_ARGS_MIN_CHARS = 256
@ -10586,6 +10606,7 @@ class LlamaCppBackend:
reasoning_effort: Optional[str] = None,
preserve_thinking: Optional[bool] = None,
seed: Optional[int] = None,
promote_reasoning_only: bool = True,
_allow_respawn_retry: bool = True,
) -> Generator[Union[str, dict], None, None]:
"""
@ -10668,7 +10689,12 @@ class LlamaCppBackend:
# model put its whole reply in reasoning
# (e.g. Qwen3 always-think). Show it as
# the main response, not a thinking block.
cumulative = reasoning_text
cumulative = _finalize_reasoning_only_cumulative(
cumulative,
reasoning_text,
_metadata_finish_reason,
promote_reasoning_only,
)
yield cumulative
_stream_done = True
break # exit inner while
@ -10765,6 +10791,7 @@ class LlamaCppBackend:
reasoning_effort = reasoning_effort,
preserve_thinking = preserve_thinking,
seed = seed,
promote_reasoning_only = promote_reasoning_only,
_allow_respawn_retry = False,
)
return
@ -10806,6 +10833,7 @@ class LlamaCppBackend:
confirm_tool_calls: bool = False,
bypass_permissions: bool = False,
permission_mode: Optional[str] = None,
promote_reasoning_only: bool = True,
) -> Generator[dict, None, None]:
"""
Agentic loop: let the model call tools, execute them, and continue.
@ -11147,7 +11175,12 @@ class LlamaCppBackend:
),
}
else:
cumulative_display = reasoning_accum
cumulative_display = _finalize_reasoning_only_cumulative(
cumulative_display,
reasoning_accum,
_iter_finish_reason,
promote_reasoning_only,
)
if not _suppress_visible_output:
yield {
"type": "content",
@ -11611,7 +11644,12 @@ class LlamaCppBackend:
if _reasoning_started_at is not None and not _reasoning_summary_emitted:
_reasoning_summary_emitted = True
yield _reasoning_summary_event(_reasoning_started_at)
cumulative_display = reasoning_accum
cumulative_display = _finalize_reasoning_only_cumulative(
cumulative_display,
reasoning_accum,
_iter_finish_reason,
promote_reasoning_only,
)
if not _suppress_visible_output:
yield {
"type": "content",
@ -12175,7 +12213,12 @@ class LlamaCppBackend:
"text": _strip_tool_markup(cumulative, final = True),
}
else:
cumulative = reasoning_text
cumulative = _finalize_reasoning_only_cumulative(
cumulative,
reasoning_text,
_metadata_finish_reason,
promote_reasoning_only,
)
yield {"type": "content", "text": cumulative}
_stream_done = True
break # exit inner while

View file

@ -1794,7 +1794,16 @@ router = APIRouter()
studio_router = APIRouter()
_ARTIFACT_PREVIEW_FRAME_ANCESTORS = "'self' tauri://localhost http://tauri.localhost"
# Packaged desktop runs at tauri://localhost (macOS/Linux) or http://tauri.localhost
# (Windows WebView2); the web build is same-origin ('self'). The `tauri dev` shell,
# however, serves the frontend from the Vite dev origin (http://localhost:5173),
# so the packaged allowlist alone leaves the preview blocked in dev with an
# "ancestor violates frame-ancestors" error. This shell exposes no server resource
# (it only renders postMessage'd HTML in a no-same-origin sandbox), so also allowing
# any localhost/127.0.0.1 dev origin to frame it is safe and unblocks the dev shell.
_ARTIFACT_PREVIEW_FRAME_ANCESTORS = (
"'self' tauri://localhost http://tauri.localhost http://localhost:* http://127.0.0.1:*"
)
_ARTIFACT_PREVIEW_FRAME_STRICT_CSP = (
"default-src 'none'; "
"script-src 'unsafe-inline'; "
@ -13355,6 +13364,7 @@ async def anthropic_messages(
disable_parallel_tool_use = _disable_parallel,
bypass_permissions = bool(payload.bypass_permissions),
permission_mode = getattr(payload, "permission_mode", None),
promote_reasoning_only = False,
)
if payload.stream:
@ -13394,6 +13404,7 @@ async def anthropic_messages(
max_tokens = payload.max_tokens,
stop = stop,
cancel_event = cancel_event,
promote_reasoning_only = False,
)
if payload.stream:

View file

@ -68,16 +68,15 @@ def _emitter_client_text(events: list[str]) -> str:
def test_anthropic_emitter_closes_reasoning_only_think_block():
# A reasoning-only reply streams <think>X live then shrinks to bare X at EOF.
# This emitter diffs cumulative snapshots and drops the shrink, so without a
# closing pass the client text would end on an unclosed <think>. finish()
# must balance it.
# Anthropic asks the GGUF generator not to promote reasoning into a duplicate
# visible fallback, so its final cumulative snapshot only balances the block.
emitter = AnthropicStreamEmitter()
events = emitter.start("msg_1", "m")
events += emitter.feed({"type": "content", "text": "<think>The capital"})
events += emitter.feed({"type": "content", "text": "<think>The capital of France is Paris."})
# The generator's final bare-text shrink (dropped by the cumulative diff).
events += emitter.feed({"type": "content", "text": "The capital of France is Paris."})
events += emitter.feed(
{"type": "content", "text": "<think>The capital of France is Paris.</think>"}
)
events += emitter.finish()
assert _emitter_client_text(events) == "<think>The capital of France is Paris.</think>"
@ -1563,6 +1562,44 @@ class TestAnthropicMessagesToolRouting:
assert entry["context_length"] == 2048
assert monitor.active_count() == 0
@pytest.mark.parametrize("stream", [False, True])
@pytest.mark.parametrize("with_tools", [False, True])
def test_reasoning_only_output_is_not_duplicated(self, monkeypatch, stream, with_tools):
reasoning = "The capital of France is Paris."
def _gen_plain(**kwargs):
assert kwargs["promote_reasoning_only"] is False
yield f"<think>{reasoning}"
yield f"<think>{reasoning}</think>"
def _gen_tools(**kwargs):
assert kwargs["promote_reasoning_only"] is False
yield {"type": "content", "text": f"<think>{reasoning}"}
yield {"type": "content", "text": f"<think>{reasoning}</think>"}
_mock_backend(
monkeypatch,
generate_chat_completion = _gen_plain,
generate_chat_completion_with_tools = _gen_tools,
)
payload_fields = {"stream": stream}
if with_tools:
payload_fields.update(
{
"enable_tools": True,
"tools": [{"type": "web_search_20250305", "name": "web_search"}],
}
)
payload = _basic_payload(**payload_fields)
response = _drive(anthropic_messages(payload, request = self._Request(), current_subject = "t"))
if stream:
body = self._sse_blob(self._consume_response(response))
assert body.count(reasoning) == 1
else:
body = json.loads(response.body)
assert body["content"][0]["text"] == f"<think>{reasoning}</think>"
def test_tool_use_non_streaming_records_api_monitor_reply(self, monkeypatch):
import routes.inference as inf_mod

View file

@ -789,9 +789,12 @@ class TestLoadHubDownloadExclusion:
# The gguf_load_in_flight marker must be entered before the hub-download
# guard and the unload so a concurrent load can't race the download
# manager. The llama_extra_args inheritance that used to sit between the
# marker and the guard now runs in _guard_chat_load_against_training, ahead
# of the GGUF branch, so it is no longer a landmark inside this slice.
# manager. The llama_extra_args inheritance moved out of the branch into
# _resolve_inherited_extra_args, which must still run BEFORE it: the
# inherited value (e.g. a carried --no-mmproj) shapes the guard's
# require_mmproj. Anchor on the call form so the assertion pins the
# endpoint's call site, not the function definition.
assert source.index("= _resolve_inherited_extra_args(") < source.index("if config.is_gguf:")
assert (
gguf_branch.index("enter_context(gguf_load_in_flight")
< gguf_branch.index("_hub_download_blocks_gguf_load")

View file

@ -37,6 +37,24 @@ def _done() -> str:
return "data: [DONE]\n"
def _finish(reason: str) -> str:
return (
"data: "
+ json.dumps(
{
"choices": [
{
"index": 0,
"delta": {},
"finish_reason": reason,
}
]
}
)
+ "\n"
)
def _make_backend(
monkeypatch,
streams: list[object],
@ -327,9 +345,8 @@ def test_reasoning_streams_incrementally_with_tools(monkeypatch):
def test_reasoning_only_reply_matches_no_tool_path_with_tools(monkeypatch):
# A reasoning-only turn (whole answer in reasoning_content, no content, no
# tool) with a tool active streams the reasoning live, then resolves to the
# bare reasoning text -- identical to the no-tool generate_chat_completion
# path -- so the non-streaming drain still returns it as `content`, not an
# empty answer.
# same text on the visible channel. The final cumulative snapshot stays
# append-only so route suffix extraction cannot drop that fallback.
stream = [
_sse({"reasoning_content": "The capital of France is Paris."}),
_done(),
@ -349,8 +366,49 @@ def test_reasoning_only_reply_matches_no_tool_path_with_tools(monkeypatch):
content_texts = [e["text"] for e in events if e["type"] == "content"]
# Reasoning streamed live during BUFFERING (the fix).
assert content_texts[0] == "<think>The capital of France is Paris."
# Resolves to bare reasoning, matching the no-tool sibling.
assert content_texts[-1] == "The capital of France is Paris."
assert content_texts[-1] == (
"<think>The capital of France is Paris.</think>The capital of France is Paris."
)
def _assert_reasoning_only_raw_consumer_gets_one_balanced_think_block(monkeypatch, with_tools):
stream = [
_sse({"reasoning_content": "The capital of France is Paris."}),
_done(),
]
backend = _make_backend(monkeypatch, [stream], [])
if with_tools:
items = list(
backend.generate_chat_completion_with_tools(
messages = [{"role": "user", "content": "capital of France?"}],
tools = [{"type": "function", "function": {"name": "web_search"}}],
max_tool_iterations = 1,
promote_reasoning_only = False,
)
)
cumulatives = [item["text"] for item in items if item.get("type") == "content"]
else:
items = list(
backend.generate_chat_completion(
messages = [{"role": "user", "content": "capital of France?"}],
promote_reasoning_only = False,
)
)
cumulatives = [item for item in items if isinstance(item, str)]
assert cumulatives[-1] == "<think>The capital of France is Paris.</think>"
assert all(
current.startswith(previous) for previous, current in zip([""] + cumulatives, cumulatives)
)
def test_reasoning_only_raw_consumer_without_tools_gets_one_balanced_think_block(monkeypatch):
_assert_reasoning_only_raw_consumer_gets_one_balanced_think_block(monkeypatch, False)
def test_reasoning_only_raw_consumer_with_tools_gets_one_balanced_think_block(monkeypatch):
_assert_reasoning_only_raw_consumer_gets_one_balanced_think_block(monkeypatch, True)
def test_reasoning_before_structured_tool_closes_think_block(monkeypatch):
@ -420,8 +478,8 @@ def _replay_route_reasoning_extractor(cumulatives: list[str]) -> tuple[str, str]
def test_reasoning_only_route_output_matches_no_tool_path(monkeypatch):
# Parity contract: a reasoning-only reply must reach the client identically
# whether tools are on or off. Both generators stream <think> live then
# resolve to the bare reasoning text; the route's suffix-diff + extractor
# must therefore produce the same (visible, reasoning) split for both.
# append a balanced close plus visible fallback; the route's suffix-diff +
# extractor must therefore produce the same split for both.
stream = [
_sse({"reasoning_content": "The capital"}),
_sse({"reasoning_content": " of France is Paris."}),
@ -458,10 +516,37 @@ def test_reasoning_only_route_output_matches_no_tool_path(monkeypatch):
no_tool_out = _replay_route_reasoning_extractor(no_tool_cumulatives)
assert tool_out == no_tool_out
# Pin the shared contract so a change to either path shows up here.
_visible, reasoning = tool_out
visible, reasoning = tool_out
assert visible == "The capital of France is Paris."
assert reasoning == "The capital of France is Paris."
def test_length_truncated_reasoning_stays_append_only_without_visible_promotion(monkeypatch):
stream = [
_sse({"reasoning_content": "The proof begins by assuming finitely many primes."}),
_finish("length"),
_done(),
]
backend = _make_backend(monkeypatch, [stream], [])
items = list(
backend.generate_chat_completion(
messages = [{"role": "user", "content": "Prove infinitely many primes"}],
max_tokens = 16,
)
)
cumulatives = [item for item in items if isinstance(item, str)]
assert all(
current.startswith(previous) for previous, current in zip([""] + cumulatives, cumulatives)
)
assert cumulatives[-1] == ("<think>The proof begins by assuming finitely many primes.</think>")
visible, reasoning = _replay_route_reasoning_extractor(cumulatives)
assert visible == ""
assert reasoning == "The proof begins by assuming finitely many primes."
assert items[-1]["finish_reason"] == "length"
def test_reasoning_before_bare_json_tool_closes_think_block(monkeypatch):
# _drain_silently sibling of the structured-tool close: a bare-JSON tool call
# with a live reasoning prefix must also close </think> before draining, and

View file

@ -524,8 +524,10 @@ export function AppProvider({ children }: AppProviderProps) {
visibleToasts={2}
expand={true}
closeButton={true}
// Clear the chat header buttons on the right.
offset={{ top: 12, right: 64 }}
// Clear the chat header buttons on the right. On desktop, also drop
// below the ~34px custom window titlebar so toasts don't cover the
// minimize / maximize / close controls.
offset={{ top: isTauri ? 46 : 12, right: 64 }}
/>
</TooltipProvider>
</MotionConfig>

View file

@ -89,6 +89,7 @@ import {
import { resolveLoadMaxSeqLength } from "../presets/preset-policy";
import {
generateAudio,
GenerationLengthError,
listCachedGguf,
listCachedModels,
listGgufVariants,
@ -4093,7 +4094,15 @@ export function createOpenAIStreamAdapter(
);
if (!abortSignal.aborted) {
const msg = err instanceof Error ? err.message : String(err);
if (err instanceof StreamInterruptedError) {
if (err instanceof GenerationLengthError) {
toast.error("Response ran out of tokens", {
description:
"The model used the full Max Tokens budget while thinking " +
"and did not produce a final answer. Increase Max Tokens in " +
"chat Settings or turn off thinking, then retry.",
duration: 8000,
});
} else if (err instanceof StreamInterruptedError) {
// Connection dropped mid-turn: surface it explicitly (the rethrow
// below also marks the message with an inline error + Retry).
toast.error("Response interrupted", {

View file

@ -50,6 +50,21 @@ export class StreamInterruptedError extends Error {
}
}
/**
* Thrown when a reasoning model consumes its output budget before emitting any
* standard content. Keeping this distinct from a dropped connection lets the
* chat UI explain why a completed stream contains only a thinking panel.
*/
export class GenerationLengthError extends Error {
constructor() {
super(
"The model reached the Max Tokens limit before producing a final answer. " +
"Increase Max Tokens or disable thinking, then retry.",
);
this.name = "GenerationLengthError";
}
}
export function notifyChatHistoryUpdated(): void {
if (typeof window !== "undefined") {
window.dispatchEvent(new Event(CHAT_HISTORY_UPDATED_EVENT));
@ -982,6 +997,61 @@ function parseSseEvent(rawEvent: string): string[] {
return dataLines;
}
function hasNonWhitespaceText(value: unknown): boolean {
if (typeof value === "string") {
return value.trim().length > 0;
}
if (Array.isArray(value)) {
return value.some((item) => hasNonWhitespaceText(item));
}
if (!value || typeof value !== "object") {
return false;
}
const record = value as Record<string, unknown>;
return ["thinking", "text", "content", "reasoning", "summary"].some(
(key) => key in record && hasNonWhitespaceText(record[key]),
);
}
function classifyStructuredDeltaContent(content: unknown): {
hasAssistantContent: boolean;
hasReasoningContent: boolean;
} {
if (typeof content === "string") {
return {
hasAssistantContent: hasNonWhitespaceText(content),
hasReasoningContent: false,
};
}
if (!Array.isArray(content)) {
return {
hasAssistantContent: false,
hasReasoningContent: false,
};
}
let hasAssistantContent = false;
let hasReasoningContent = false;
for (const part of content) {
if (typeof part === "string") {
hasAssistantContent ||= hasNonWhitespaceText(part);
continue;
}
if (!part || typeof part !== "object") {
continue;
}
const record = part as Record<string, unknown>;
if (record.type === "thinking" || record.type === "reasoning") {
hasReasoningContent ||= hasNonWhitespaceText(record);
} else if (record.type === "text" || record.type === "output_text") {
const text =
typeof record.text === "string" ? record.text : record.content;
hasAssistantContent ||= hasNonWhitespaceText(text);
}
}
return { hasAssistantContent, hasReasoningContent };
}
export async function* streamChatCompletions(
payload: OpenAIChatCompletionsRequest,
signal: AbortSignal,
@ -1009,6 +1079,19 @@ export async function* streamChatCompletions(
// EOF without `[DONE]` or a finish_reason chunk means the stream was cut
// mid-generation: surface as interrupted, not silent success.
let sawTerminalSignal = false;
let terminalFinishReason: string | null = null;
let sawAssistantContent = false;
let sawReasoningContent = false;
const throwIfReasoningOnlyLength = () => {
if (
terminalFinishReason === "length" &&
sawReasoningContent &&
!sawAssistantContent
) {
throw new GenerationLengthError();
}
};
try {
while (true) {
@ -1018,6 +1101,7 @@ export async function* streamChatCompletions(
if (!sawTerminalSignal) {
throw new StreamInterruptedError();
}
throwIfReasoningOnlyLength();
break;
}
@ -1039,6 +1123,7 @@ export async function* streamChatCompletions(
if (dataText === "[DONE]") {
completed = true;
sawTerminalSignal = true;
throwIfReasoningOnlyLength();
return;
}
@ -1094,11 +1179,31 @@ export async function* streamChatCompletions(
}
// finish_reason is a valid terminal signal for providers that close
// the stream without an explicit [DONE] sentinel.
const finishReason = (
const parsedChoices = (
parsed as {
choices?: Array<{ finish_reason?: string | null }>;
choices?: Array<{
delta?: Record<string, unknown>;
finish_reason?: string | null;
}>;
}
).choices?.[0]?.finish_reason;
).choices;
for (const choice of parsedChoices ?? []) {
const delta = choice.delta;
if (delta) {
const contentState = classifyStructuredDeltaContent(delta.content);
sawAssistantContent ||= contentState.hasAssistantContent;
sawReasoningContent ||= contentState.hasReasoningContent;
const reasoning =
delta.reasoning_content ??
delta.reasoning ??
delta.reasoning_details;
sawReasoningContent ||= hasNonWhitespaceText(reasoning);
}
if (choice.finish_reason) {
terminalFinishReason = choice.finish_reason;
}
}
const finishReason = parsedChoices?.[0]?.finish_reason;
if (finishReason) {
sawTerminalSignal = true;
}

View file

@ -12,6 +12,8 @@ import {
import { MascotImg } from "@/components/mascot-img";
import { Button } from "@/components/ui/button";
import { copyToClipboard } from "@/lib/copy-to-clipboard";
import { downloadFile, isDownloadCancelled } from "@/lib/native-files";
import { toast } from "@/lib/toast";
import { cn } from "@/lib/utils";
import { CopyIcon, EyeIcon, Maximize2Icon, XIcon } from "lucide-react";
import { Download01Icon } from "@hugeicons/core-free-icons";
@ -91,18 +93,6 @@ function ArtifactGeneratingPanel() {
);
}
function downloadTextFile(filename: string, text: string): void {
const blob = new Blob([text], { type: "text/html;charset=utf-8" });
const url = URL.createObjectURL(blob);
const anchor = document.createElement("a");
anchor.href = url;
anchor.download = filename;
document.body.appendChild(anchor);
anchor.click();
document.body.removeChild(anchor);
window.setTimeout(() => URL.revokeObjectURL(url), 0);
}
export function ArtifactSurface({
artifact,
variant,
@ -205,7 +195,7 @@ export function ArtifactSurface({
className={cn(
"relative flex min-h-0 flex-col bg-background",
variant === "panel"
? "artifact-panel-shell mx-2 mt-[72px] mb-8 h-[calc(100%_-_104px)] overflow-visible rounded-[28px] border-t border-border/70 bg-card/95"
? "artifact-panel-shell mx-2 mt-[90px] mb-8 h-[calc(100%_-_122px)] overflow-visible rounded-[28px] border-t border-border/70 bg-card/95"
: "h-[min(92vh,900px)] w-[min(96vw,1200px)] overflow-hidden rounded-2xl border border-border shadow-xl",
)}
aria-label={`${artifact.title} canvas`}
@ -265,7 +255,19 @@ export function ArtifactSurface({
size="icon"
className="size-8"
disabled={isLoadingArtifact || !hasArtifactCode}
onClick={() => downloadTextFile(filename, artifact.code)}
onClick={() => {
// Route through the native save dialog on desktop; the plain
// blob-anchor download is silently dropped by the Tauri WebView2.
void downloadFile(
artifact.code,
filename,
"text/html;charset=utf-8",
).catch((err) => {
if (!isDownloadCancelled(err)) {
toast.error("Failed to save canvas HTML");
}
});
}}
aria-label="Download canvas HTML"
>
<HugeiconsIcon icon={Download01Icon} className="size-4" />

View file

@ -2676,7 +2676,7 @@ export function ChatPage({
config: meta?.config,
nativePathToken: meta?.nativePathToken,
nativePathExpiresAtMs: meta?.nativePathExpiresAtMs,
forceReload: isSameLoadedModel || undefined,
forceReload: meta?.forceReload ?? (isSameLoadedModel || undefined),
};
await stageOrLoad(selection);
})();

View file

@ -966,7 +966,7 @@ export function ChatSettingsPanel({
Delete
</Button>
</div>
<p className="text-[11px] leading-relaxed text-muted-foreground">
<p className="text-ui-11 leading-relaxed text-muted-foreground">
Saving a preset also stores current load settings (context length,
KV cache dtype, speculative decoding, GPU layers).
{currentLoadSummary ? (

View file

@ -62,6 +62,7 @@ import {
import { isExternalModelId } from "../external-providers";
import {
applyPerModelConfigToRuntime,
normalizeMaxSeqLength,
type PerModelConfig,
} from "@/features/model-picker";
import type {
@ -604,12 +605,19 @@ export function useChatModelRuntime() {
async function performLoad(): Promise<void> {
if (abortCtrl.signal.aborted) throw new Error("Cancelled");
let previousWasUnloaded = false;
const pendingLoadConfig =
typeof selection !== "string" ? selection.config : undefined;
if (pendingLoadConfig) {
applyPerModelConfigToRuntime(pendingLoadConfig);
}
const currentCheckpoint =
useChatRuntimeStore.getState().params.checkpoint;
const stateBeforeUnload = useChatRuntimeStore.getState();
let trustRemoteCode = stateBeforeUnload.params.trustRemoteCode ?? false;
let approvedRemoteCodeFingerprint: string | null = null;
const maxSeqLength = stateBeforeUnload.params.maxSeqLength;
const maxSeqLength =
normalizeMaxSeqLength(pendingLoadConfig?.maxSeqLength) ??
stateBeforeUnload.params.maxSeqLength;
const previousActiveNativePathToken =
stateBeforeUnload.activeNativePathToken;
const previousIsGguf =
@ -643,34 +651,54 @@ export function useChatModelRuntime() {
const previousActiveNativePathExpiresAtMs =
stateBeforeUnload.activeNativePathExpiresAtMs;
// Snapshot the load settings at click time, before the awaits below
// (validation, the trust dialog, unload).
const loadChatTemplateOverride = stateBeforeUnload.chatTemplateOverride;
const loadKvCacheDtype = stateBeforeUnload.kvCacheDtype;
// (validation, the trust dialog, unload). When the picker staged a
// config payload, prefer it over the store: React may not have
// flushed NumericValueInput's blur commit into state yet.
const loadChatTemplateOverride =
pendingLoadConfig?.chatTemplateOverride?.trim()
? pendingLoadConfig.chatTemplateOverride
: stateBeforeUnload.chatTemplateOverride;
const loadKvCacheDtype =
pendingLoadConfig?.kvCacheDtype ?? stateBeforeUnload.kvCacheDtype;
// gpuMemoryMode is a standing preference (kept across a model switch);
// the rest are per-model knobs the reset below clears, so they are
// re-baselined there in lock-step with the store.
let loadCustomContextLength = stateBeforeUnload.customContextLength;
let loadCustomContextLength =
pendingLoadConfig?.customContextLength ??
stateBeforeUnload.customContextLength;
const loadGgufContextLength = stateBeforeUnload.ggufContextLength;
const loadTensorParallel = stateBeforeUnload.tensorParallel;
const loadTensorParallel =
pendingLoadConfig?.tensorParallel ?? stateBeforeUnload.tensorParallel;
const loadActivePresetSource = stateBeforeUnload.activePresetSource;
const loadActiveGgufVariant = stateBeforeUnload.activeGgufVariant;
const loadGpuMemoryMode = stateBeforeUnload.gpuMemoryMode;
let loadGpuLayers = stateBeforeUnload.gpuLayers;
let loadNCpuMoe = stateBeforeUnload.nCpuMoe;
const loadGpuMemoryMode =
pendingLoadConfig?.gpuMemoryMode ?? stateBeforeUnload.gpuMemoryMode;
let loadGpuLayers =
pendingLoadConfig?.gpuLayers ?? stateBeforeUnload.gpuLayers;
let loadNCpuMoe =
pendingLoadConfig?.nCpuMoe ?? stateBeforeUnload.nCpuMoe;
let loadSplitRatio = stateBeforeUnload.splitRatio;
// Reconcile the persisted pick against the GPUs present now, so a stale
// cross-host / now-hidden pick is dropped before /load rather than
// rejected there. Warm the device cache first: load-on-selection can
// run before any GPU hook mounted, and a cold cache would pass the
// pick through unvalidated. validateGpuIds derives from this too.
if (stateBeforeUnload.selectedGpuIds != null) {
if (
pendingLoadConfig?.selectedGpuIds !== undefined ||
stateBeforeUnload.selectedGpuIds != null
) {
await ensureGpuDeviceCache();
}
let loadSelectedGpuIds = reconcilePersistedGpuIds(
stateBeforeUnload.selectedGpuIds,
);
let loadSpeculativeType = stateBeforeUnload.speculativeType;
let loadSpecDraftNMax = stateBeforeUnload.specDraftNMax;
let loadSelectedGpuIds =
pendingLoadConfig?.selectedGpuIds !== undefined
? reconcilePersistedGpuIds(pendingLoadConfig.selectedGpuIds)
: reconcilePersistedGpuIds(stateBeforeUnload.selectedGpuIds);
let loadSpeculativeType =
pendingLoadConfig?.speculativeType != null
? normalizeSpeculativeType(pendingLoadConfig.speculativeType)
: stateBeforeUnload.speculativeType;
let loadSpecDraftNMax =
pendingLoadConfig?.specDraftNMax ?? stateBeforeUnload.specDraftNMax;
try {
// Lightweight pre-flight validation: avoid unloading a working model
// if the new identifier is clearly invalid (e.g. bad HF id / path).
@ -810,15 +838,23 @@ export function useChatModelRuntime() {
// model loads at Auto/native, not the previous model's pin.
customContextLength: null,
});
loadSpeculativeType = persistedSpeculativeType;
loadSpecDraftNMax = null;
loadSpeculativeType =
pendingLoadConfig?.speculativeType != null
? normalizeSpeculativeType(pendingLoadConfig.speculativeType)
: persistedSpeculativeType;
loadSpecDraftNMax = pendingLoadConfig?.specDraftNMax ?? null;
// Keep the click-time snapshot in lock-step with the store reset so
// the load below sizes against the cleared per-model knobs, not the
// previous model's (gpuMemoryMode is standing, so left as captured).
loadCustomContextLength = null;
loadSelectedGpuIds = null;
loadGpuLayers = GPU_LAYERS_AUTO;
loadNCpuMoe = 0;
// An explicit staged config from run-settings still wins.
loadCustomContextLength =
pendingLoadConfig?.customContextLength ?? null;
loadSelectedGpuIds =
pendingLoadConfig?.selectedGpuIds !== undefined
? reconcilePersistedGpuIds(pendingLoadConfig.selectedGpuIds)
: null;
loadGpuLayers = pendingLoadConfig?.gpuLayers ?? GPU_LAYERS_AUTO;
loadNCpuMoe = pendingLoadConfig?.nCpuMoe ?? 0;
loadSplitRatio = null;
}
@ -1271,12 +1307,19 @@ export function useChatModelRuntime() {
prog.expected_bytes,
dlSamples,
);
setLoadProgress({
percent: pct,
label: progressLabel,
phase: "downloading",
});
if (loadToastDismissedRef.current) return;
// loadProgress state is only read by the dismissed-toast inline
// status. Writing it while the toast is visible re-renders the
// whole chat page every poll — cheap in Chrome, janky in the
// desktop WebView2 (laggy typing). Feed the toast directly and
// only touch state when the inline view is actually live.
if (loadToastDismissedRef.current) {
setLoadProgress({
percent: pct,
label: progressLabel,
phase: "downloading",
});
return;
}
toast(null, {
id: toastId,
...modelLoadToastOptions(
@ -1298,19 +1341,23 @@ export function useChatModelRuntime() {
const est = estimate(dlSamples, prog.downloaded_bytes, 0);
const rateSuffix =
est.stable ? `${formatRate(est.rate)}` : "";
setLoadProgress({
percent: null,
label: `${dlGb.toFixed(1)} GB downloaded${rateSuffix}`,
phase: "downloading",
});
// Inline-status-only state; skip the chat-page re-render unless it's shown.
if (loadToastDismissedRef.current) {
setLoadProgress({
percent: null,
label: `${dlGb.toFixed(1)} GB downloaded${rateSuffix}`,
phase: "downloading",
});
}
} else if (prog.progress >= 1 && hasShownProgress) {
downloadComplete = true;
setLoadProgress({
percent: 100,
label: "Download complete",
phase: "starting",
});
if (!loadToastDismissedRef.current) {
if (loadToastDismissedRef.current) {
setLoadProgress({
percent: 100,
label: "Download complete",
phase: "starting",
});
} else {
toast(null, {
id: toastId,
...modelLoadToastOptions(
@ -1364,12 +1411,17 @@ export function useChatModelRuntime() {
formatEta(est.eta) !== "--" ? `${formatEta(est.eta)} left` : ""
}`
: base;
setLoadProgress({
percent: pct,
label,
phase: "starting",
});
if (loadToastDismissedRef.current) return;
// Inline-status-only state (see pollDownload): while the toast is
// up, skip the state write so the chat page doesn't re-render every
// poll during "Starting model" — the desktop WebView2 typing-lag fix.
if (loadToastDismissedRef.current) {
setLoadProgress({
percent: pct,
label,
phase: "starting",
});
return;
}
toast(null, {
id: toastId,
...modelLoadToastOptions(

View file

@ -24,7 +24,14 @@ import { ChevronDownStandardIcon } from "@/lib/chevron-icons";
import { toast } from "@/lib/toast";
import { ArrowLeft01Icon } from "@hugeicons/core-free-icons";
import { HugeiconsIcon } from "@hugeicons/react";
import { type ReactNode, useEffect, useId, useState } from "react";
import {
type ReactNode,
type Ref,
useEffect,
useId,
useRef,
useState,
} from "react";
import {
useDefaultChatTemplate,
useModelMaxPositionEmbeddings,
@ -50,7 +57,10 @@ import {
} from "../model-config/per-model-config";
import { ChatTemplateEditorDialog } from "./chat-template-editor-dialog";
import type { ModelPickTarget } from "./model-selector/types";
import { NumericValueInput } from "./numeric-value-input";
import {
NumericValueInput,
type NumericValueInputHandle,
} from "./numeric-value-input";
const ROW_CLASS = "flex min-h-8 items-center justify-between gap-3";
const LABEL_CLASS =
@ -130,11 +140,13 @@ function MaxSeqLengthSetting({
max,
inputMax,
onChange,
inputRef,
}: {
value: number;
max: number;
inputMax: number;
onChange: (value: number) => void;
inputRef?: Ref<NumericValueInputHandle>;
}) {
return (
<div className="space-y-3">
@ -146,6 +158,7 @@ function MaxSeqLengthSetting({
</InfoHint>
</div>
<NumericValueInput
ref={inputRef}
value={value}
min={MAX_SEQ_LENGTH_MIN}
max={inputMax}
@ -182,6 +195,7 @@ function AdvancedGpuSlider({
onChange,
displayValue,
info,
inputRef,
}: {
label: string;
value: number;
@ -190,6 +204,7 @@ function AdvancedGpuSlider({
onChange: (value: number) => void;
displayValue?: string;
info?: ReactNode;
inputRef?: Ref<NumericValueInputHandle>;
}) {
return (
<div className="space-y-3">
@ -199,6 +214,7 @@ function AdvancedGpuSlider({
{info && <InfoHint>{info}</InfoHint>}
</div>
<NumericValueInput
ref={inputRef}
value={value}
min={min}
max={max}
@ -231,11 +247,15 @@ function GpuMemorySettings({
update,
layerCount,
moeLayerCount,
gpuLayersInputRef,
moeLayersInputRef,
}: {
config: PerModelConfig;
update: (patch: Partial<PerModelConfig>) => void;
layerCount: number | null;
moeLayerCount: number | null;
gpuLayersInputRef?: Ref<NumericValueInputHandle>;
moeLayersInputRef?: Ref<NumericValueInputHandle>;
}) {
const gpuDevices = useGpuDevices();
const mode = config.gpuMemoryMode ?? "auto";
@ -322,6 +342,7 @@ function GpuMemorySettings({
<>
<AdvancedGpuSlider
label="GPU Layers"
inputRef={gpuLayersInputRef}
value={Math.max(GPU_LAYERS_AUTO, Math.min(gpuLayers, gpuLayersMax))}
min={GPU_LAYERS_AUTO}
max={gpuLayersMax}
@ -338,6 +359,7 @@ function GpuMemorySettings({
{showMoeSlider && (
<AdvancedGpuSlider
label="MoE Layers on CPU"
inputRef={moeLayersInputRef}
value={Math.min(nCpuMoe, moeLayersMax)}
min={0}
max={moeLayersMax}
@ -399,6 +421,8 @@ function GgufAdvancedSettings({
onEditTemplate,
layerCount,
moeLayerCount,
gpuLayersInputRef,
moeLayersInputRef,
}: {
config: PerModelConfig;
update: (patch: Partial<PerModelConfig>) => void;
@ -407,6 +431,8 @@ function GgufAdvancedSettings({
onEditTemplate: () => void;
layerCount: number | null;
moeLayerCount: number | null;
gpuLayersInputRef?: Ref<NumericValueInputHandle>;
moeLayersInputRef?: Ref<NumericValueInputHandle>;
}) {
return (
<>
@ -535,6 +561,8 @@ function GgufAdvancedSettings({
update={update}
layerCount={layerCount}
moeLayerCount={moeLayerCount}
gpuLayersInputRef={gpuLayersInputRef}
moeLayersInputRef={moeLayersInputRef}
/>
<ChatTemplateSetting config={config} onEditTemplate={onEditTemplate} />
@ -597,6 +625,10 @@ export function ModelConfigPage({
const [showAdvanced, setShowAdvanced] = useState(() =>
hasNonDefaultAdvanced(config),
);
const contextInputRef = useRef<NumericValueInputHandle>(null);
const maxSeqLengthInputRef = useRef<NumericValueInputHandle>(null);
const gpuLayersInputRef = useRef<NumericValueInputHandle>(null);
const moeLayersInputRef = useRef<NumericValueInputHandle>(null);
const nativePathToken =
target.meta.nativePathToken ??
(isActiveModel ? activeNativePathToken : null);
@ -744,11 +776,6 @@ export function ModelConfigPage({
? { ...config, customContextLength: activeLoadedContext }
: config
: config;
// Load request needs a concrete max length; substitute the fallback here only,
// never in the persisted runtimeConfig.
const loadConfig = target.isGguf
? runtimeConfig
: { ...runtimeConfig, maxSeqLength: maxSeqLengthValue };
const rememberChanged = remember !== savedRemember;
const persistenceOnly = isActiveModel && atBaseline && rememberChanged;
const primaryActionLabel = persistenceOnly
@ -760,18 +787,90 @@ export function ModelConfigPage({
: "Load model";
const handleRun = () => {
const defaultConfig = isDefaultConfig(runtimeConfig);
// Same-click Load/Reload: a numeric draft the user just typed is flushed only
// by that input's blur handler, which updates the parent config after this
// click closure already captured the stale value. Commit every numeric input
// imperatively so the staged load honors what the user just typed, not just
// the Context field.
const committedContext = target.isGguf
? contextInputRef.current?.commit()
: undefined;
const committedMaxSeqLength = target.isGguf
? undefined
: maxSeqLengthInputRef.current?.commit();
const committedGpuLayers = target.isGguf
? gpuLayersInputRef.current?.commit()
: undefined;
const committedMoeLayers = target.isGguf
? moeLayersInputRef.current?.commit()
: undefined;
const pendingPatch: Partial<PerModelConfig> = {};
if (committedContext != null) {
pendingPatch.customContextLength = committedContext;
}
if (committedMaxSeqLength != null) {
pendingPatch.maxSeqLength = clampMaxSeqLength(
committedMaxSeqLength,
MAX_SEQ_LENGTH_MAX,
);
}
if (committedGpuLayers != null) {
pendingPatch.gpuLayers = committedGpuLayers;
}
if (committedMoeLayers != null) {
pendingPatch.nCpuMoe = committedMoeLayers;
}
const hasPending =
committedContext != null ||
committedMaxSeqLength != null ||
committedGpuLayers != null ||
committedMoeLayers != null;
const effectiveConfig = hasPending
? { ...config, ...pendingPatch }
: config;
// pinFixedLayerContext above was computed from the render-time config, before
// the same-click GPU Layers draft was committed. Recompute it from
// effectiveConfig so committing a positive fixed-layer value still pins the
// fitted context; otherwise the saved config carries customContextLength: null
// and a later fresh load sends the native context with fixed layers (the OOM
// the pin exists to avoid).
const effectivePinFixedLayerContext =
target.isGguf &&
effectiveConfig.gpuMemoryMode === "manual" &&
effectiveConfig.gpuLayers != null &&
effectiveConfig.gpuLayers >= 0 &&
effectiveConfig.customContextLength == null &&
activeLoadedContext != null;
const effectiveRuntimeConfig = hasPending
? effectivePinFixedLayerContext
? { ...effectiveConfig, customContextLength: activeLoadedContext }
: effectiveConfig
: runtimeConfig;
// Non-GGUF load substitutes the resolved max sequence length; recompute it
// from the committed draft so a same-click Max Seq Length edit is not lost.
const effectiveMaxSeqLengthValue =
committedMaxSeqLength == null
? maxSeqLengthValue
: (normalizeMaxSeqLength(effectiveConfig.maxSeqLength) ??
clampMaxSeqLength(DEFAULT_MAX_SEQ_LENGTH, nativeMaxSeqLength));
// Recheck the committed draft so Save/Forget reloads when needed.
const effectiveAtBaseline = perModelConfigsEqual(effectiveConfig, baseline);
const effectivePersistenceOnly =
isActiveModel && effectiveAtBaseline && rememberChanged;
const defaultConfig = isDefaultConfig(effectiveRuntimeConfig);
let saveFailed = false;
if (remember) {
saveFailed = !savePerModelConfig(
target.id,
target.ggufVariant,
runtimeConfig,
effectiveRuntimeConfig,
);
} else {
saveFailed = !deletePerModelConfig(target.id, target.ggufVariant);
}
if (persistenceOnly) {
if (effectivePersistenceOnly) {
if (saveFailed) {
toast.error("Couldn't save settings for this model.");
return;
@ -791,7 +890,10 @@ export function ModelConfigPage({
if (saveFailed) {
toast.error("Couldn't save these settings, loading with them anyway.");
}
onRun(loadConfig);
const effectiveLoadConfig = target.isGguf
? effectiveRuntimeConfig
: { ...effectiveRuntimeConfig, maxSeqLength: effectiveMaxSeqLengthValue };
onRun(effectiveLoadConfig);
};
return (
@ -838,6 +940,7 @@ export function ModelConfigPage({
</InfoHint>
</div>
<NumericValueInput
ref={contextInputRef}
value={contextValue}
min={minContext}
max={maxContext}
@ -886,6 +989,8 @@ export function ModelConfigPage({
onEditTemplate={() => setTemplateOpen(true)}
layerCount={stagedDims?.layerCount ?? null}
moeLayerCount={stagedDims?.moeLayerCount ?? null}
gpuLayersInputRef={gpuLayersInputRef}
moeLayersInputRef={moeLayersInputRef}
/>
)}
@ -914,6 +1019,7 @@ export function ModelConfigPage({
value={maxSeqLengthValue}
max={maxSeqLengthMax}
inputMax={MAX_SEQ_LENGTH_MAX}
inputRef={maxSeqLengthInputRef}
onChange={(value) =>
update({
maxSeqLength: clampMaxSeqLength(value, MAX_SEQ_LENGTH_MAX),

View file

@ -2,7 +2,13 @@
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
import { cn } from "@/lib/utils";
import { useRef, useState } from "react";
import {
forwardRef,
useEffect,
useImperativeHandle,
useRef,
useState,
} from "react";
export function snapToStep(
value: number,
@ -28,44 +34,109 @@ function sanitizeNumeric(raw: string, allowNegative: boolean): string {
return `${sign}${head}${tail}`;
}
export function NumericValueInput({
value,
min,
max,
step,
onChange,
displayValue,
className,
ariaLabel,
size: sizeAttr,
disabled = false,
}: {
value: number;
min?: number;
max?: number;
step: number;
onChange: (v: number) => void;
displayValue?: string;
className?: string;
ariaLabel?: string;
size?: number;
disabled?: boolean;
}) {
export type NumericValueInputHandle = {
/** Commit a valid focused/same-click draft; null when none is pending. */
commit: () => number | null;
};
export const NumericValueInput = forwardRef<
NumericValueInputHandle,
{
value: number;
min?: number;
max?: number;
step: number;
onChange: (v: number) => void;
displayValue?: string;
className?: string;
ariaLabel?: string;
size?: number;
disabled?: boolean;
}
>(function NumericValueInput(
{
value,
min,
max,
step,
onChange,
displayValue,
className,
ariaLabel,
size: sizeAttr,
disabled = false,
},
ref,
) {
const [focused, setFocused] = useState(false);
const [draft, setDraft] = useState("");
const cancelBlurCommitRef = useRef(false);
const draftRef = useRef("");
const dirtyRef = useRef(false);
// Same-click Load: blur commits via onChange and clears dirtyRef before the
// button onClick runs, while parent `value` is still stale. Keep the blur
// result for one imperative commit(); clear when `value` catches up or on
// focus / external edits (Reset, slider).
const lastBlurCommittedRef = useRef<number | null>(null);
const commit = (raw: string) => {
// The blur bridge is only valid across the single synchronous gesture that set
// it: blur commits during a button's mousedown and that button's onClick
// consumes it via commit() before React re-renders. Any settled render means the
// gesture is over, so drop the cache on every commit. Keying this on [value]
// alone missed a Reset (or other external edit) that restores the shown value
// unchanged when the blur did dispatch onChange (final !== value): value nets
// back to its prior number, so the effect never re-ran, the stale pin survived,
// and the next Load/Save replayed the override Reset had removed.
useEffect(() => {
lastBlurCommittedRef.current = null;
});
const commitDraft = (raw: string): number | null => {
const parsed = Number.parseFloat(raw);
if (!Number.isFinite(parsed)) {
return;
return null;
}
const final = snapToStep(parsed, step, min, max);
if (final !== value) {
onChange(final);
}
return final;
};
useImperativeHandle(
ref,
() => ({
commit: () => {
if (dirtyRef.current) {
const raw = draftRef.current;
const final = commitDraft(raw);
dirtyRef.current = false;
lastBlurCommittedRef.current = null;
if (final == null) {
draftRef.current = String(value);
}
if (focused) {
setFocused(false);
}
return final;
}
const blurCommitted = lastBlurCommittedRef.current;
if (blurCommitted != null) {
lastBlurCommittedRef.current = null;
if (focused) {
setFocused(false);
}
return blurCommitted;
}
if (focused) {
setFocused(false);
}
return null;
},
}),
[draft, focused, max, min, onChange, step, value],
);
const displayed = focused ? draft : (displayValue ?? String(value));
return (
@ -82,7 +153,11 @@ export function NumericValueInput({
aria-label={ariaLabel}
onFocus={(e) => {
cancelBlurCommitRef.current = false;
setDraft(String(value));
dirtyRef.current = false;
lastBlurCommittedRef.current = null;
const next = String(value);
draftRef.current = next;
setDraft(next);
setFocused(true);
const target = e.currentTarget;
requestAnimationFrame(() => target.select());
@ -90,24 +165,47 @@ export function NumericValueInput({
onBlur={() => {
if (cancelBlurCommitRef.current) {
cancelBlurCommitRef.current = false;
} else {
commit(draft);
lastBlurCommittedRef.current = null;
} else if (dirtyRef.current) {
const final = commitDraft(draftRef.current);
dirtyRef.current = false;
if (final == null) {
draftRef.current = String(value);
lastBlurCommittedRef.current = null;
} else {
draftRef.current = String(final);
// Only bridge the still-stale parent value when the blur actually
// dispatched onChange (final !== value). When final === value the
// parent is already current, so there is nothing to bridge; caching
// here would leave a stale pin that a later Reset or external edit
// (which doesn't change the displayed value) can never clear, so a
// following Load/Save would recreate the override Reset removed.
lastBlurCommittedRef.current = final !== value ? final : null;
}
}
setFocused(false);
}}
onChange={(e) =>
setDraft(sanitizeNumeric(e.target.value, (min ?? 0) < 0))
}
onChange={(e) => {
dirtyRef.current = true;
lastBlurCommittedRef.current = null;
const next = sanitizeNumeric(e.target.value, (min ?? 0) < 0);
draftRef.current = next;
setDraft(next);
}}
onKeyDown={(e) => {
if (e.key === "Enter") {
e.currentTarget.blur();
} else if (e.key === "Escape") {
cancelBlurCommitRef.current = true;
setDraft(String(value));
dirtyRef.current = false;
lastBlurCommittedRef.current = null;
const next = String(value);
draftRef.current = next;
setDraft(next);
e.currentTarget.blur();
}
}}
className={cn(className)}
/>
);
}
});

View file

@ -12,6 +12,7 @@ export {
export { hfModelFitsDevice } from "./components/model-selector/recommended-fit";
export {
NumericValueInput,
type NumericValueInputHandle,
snapToStep,
} from "./components/numeric-value-input";
export { SidebarModelConfig } from "./components/sidebar-model-config";

View file

@ -1,10 +1,18 @@
// SPDX-License-Identifier: AGPL-3.0-only
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
import { type ReactElement, useCallback } from "react";
import { Lock, LockOpen, Maximize2, Minus, Plus } from "lucide-react";
import { Panel, useReactFlow } from "@xyflow/react";
import { Button } from "@/components/ui/button";
import { Panel, useReactFlow } from "@xyflow/react";
import {
Focus,
Lock,
LockOpen,
Maximize2,
Minimize2,
Minus,
Plus,
} from "lucide-react";
import { type ReactElement, useCallback } from "react";
import { buildFitViewOptions } from "../../utils/graph/fit-view";
import { RECIPE_FLOATING_ICON_BUTTON_CLASS } from "../recipe-floating-icon-button-class";
@ -12,12 +20,16 @@ type ViewportControlsProps = {
interactive: boolean;
lockDisabled?: boolean;
onToggleInteractive: () => void;
maximized: boolean;
onToggleMaximize: () => void;
};
export function ViewportControls({
interactive,
lockDisabled = false,
onToggleInteractive,
maximized,
onToggleMaximize,
}: ViewportControlsProps): ReactElement {
const { zoomIn, zoomOut, fitView, getNodes } = useReactFlow();
@ -61,9 +73,23 @@ export function ViewportControls({
size="icon"
className={RECIPE_FLOATING_ICON_BUTTON_CLASS}
onClick={handleFitView}
aria-label="Fit view"
aria-label="Center view"
>
<Maximize2 className="size-4" />
<Focus className="size-4" />
</Button>
<Button
type="button"
variant="ghost"
size="icon"
className={RECIPE_FLOATING_ICON_BUTTON_CLASS}
onClick={onToggleMaximize}
aria-label={maximized ? "Exit full view" : "Expand to full view"}
>
{maximized ? (
<Minimize2 className="size-4" />
) : (
<Maximize2 className="size-4" />
)}
</Button>
<Button
type="button"
@ -74,7 +100,11 @@ export function ViewportControls({
onClick={onToggleInteractive}
aria-label={interactive ? "Lock interaction" : "Unlock interaction"}
>
{interactive ? <LockOpen className="size-4" /> : <Lock className="size-4" />}
{interactive ? (
<LockOpen className="size-4" />
) : (
<Lock className="size-4" />
)}
</Button>
</Panel>
);

View file

@ -237,6 +237,7 @@ export function RecipeStudioPage({
}, [setActiveView]);
const [processorsOpen, setProcessorsOpen] = useState(false);
const [interactive, setInteractive] = useState(true);
const [maximized, setMaximized] = useState(false);
const [runtimeIslandMinimized, setRuntimeIslandMinimized] = useState(false);
const [recentCompletedExecution, setRecentCompletedExecution] =
useState<RecipeExecutionRecord | null>(null);
@ -569,6 +570,16 @@ export function RecipeStudioPage({
[reactFlowInstance],
);
const toggleMaximize = useCallback(() => {
// The maximized surface is a fixed z-50 overlay that already covers the
// app sidebar (z-10/z-20), so we don't touch the sidebar's own state — that
// state is persisted in pin mode and mutating it here would leak the
// temporary collapse into the next page/session.
setMaximized((prev) => !prev);
// Container size changes; refit once the layout settles.
scheduleFitView({ delayMs: TAB_SWITCH_FIT_DELAY_MS });
}, [scheduleFitView]);
useEffect(() => {
if (
previousActiveViewRef.current !== activeView &&
@ -587,6 +598,15 @@ export function RecipeStudioPage({
}
}, [activeView, reactFlowInstance]);
// The "Exit full view" control lives inside the editor canvas, which unmounts
// on other tabs. Drop full-view mode when leaving the editor so Easy/Runs
// aren't left under the fixed overlay.
useEffect(() => {
if (activeView !== "editor" && maximized) {
setMaximized(false);
}
}, [activeView, maximized]);
useEffect(() => {
if (
!reactFlowInstance ||
@ -732,6 +752,8 @@ export function RecipeStudioPage({
interactive={canvasInteractive}
lockDisabled={executionLocked}
onToggleInteractive={toggleInteractive}
maximized={maximized}
onToggleMaximize={toggleMaximize}
/>
{islandExecution &&
(isExecutionInProgress(islandExecution.status) ||
@ -773,10 +795,25 @@ export function RecipeStudioPage({
}
return (
<div className="min-h-[calc(100dvh-var(--studio-titlebar-height,0px))] bg-background">
<main className="w-full px-6 py-8">
<div
className={
maximized
? "fixed inset-x-0 bottom-0 z-50 flex flex-col bg-background"
: "flex h-full min-h-0 flex-1 flex-col bg-background"
}
style={
maximized
? {
// Start below the custom/mac window titlebar so the header and
// its controls aren't hidden under (or click-blocked by) it.
top: "var(--studio-non-chat-content-top-inset, var(--studio-content-top-inset, 0px))",
}
: undefined
}
>
<main className="flex min-h-0 w-full flex-1 flex-col">
<div
className="relative w-full overflow-hidden rounded-2xl corner-squircle border"
className="relative flex min-h-0 w-full flex-1 flex-col overflow-hidden border"
ref={setSheetContainer}
>
<RecipeStudioHeader
@ -794,7 +831,7 @@ export function RecipeStudioPage({
}}
/>
<div
className="h-[75vh] w-full rounded-t-none"
className="flex min-h-0 w-full flex-1 rounded-t-none"
ref={flowContainerRef}
>
{activeView === "easy" ? (

View file

@ -282,8 +282,13 @@
/* Standard interactive-icon size for nav, menus, action bars, and
in-message code-block actions. Sized one step above body text so
icons read as minimally larger than adjacent labels (~14px text).
Follows the UI font size preference; 18px at the default.
Theme-independent declared once in :root. */
--icon-size: 18px;
/* Standard icon size follows the UI font size itself: matches it below
the 16px default, grows at half the change above it (setting 20 ->
18px), so icons read slightly smaller than enlarged text. */
--ui-icon-size: min(calc(1rem * var(--ui-font-scale, 1)), calc(0.5rem + 0.5rem * var(--ui-font-scale, 1)));
--icon-size: var(--ui-icon-size);
/* Inset of a centered .size-icon glyph within a 2rem (size-8) action
button i.e. (32px icon-size) / 2. Use as a negative margin on a
chat-message action bar so the leftmost icon's visual edge aligns
@ -1265,8 +1270,8 @@ html[data-chat-font] .aui-root {
}
.app-user-menu [data-slot="dropdown-menu-item"] svg,
.app-user-menu [data-slot="dropdown-menu-sub-trigger"] svg {
width: 19px !important;
height: 19px !important;
width: var(--ui-icon-size) !important;
height: var(--ui-icon-size) !important;
flex-shrink: 0;
}
.app-user-menu [data-slot="dropdown-menu-item"]:focus,
@ -1561,20 +1566,20 @@ html[data-chat-font] .aui-root {
/* Fixed-width icon slot so every pill's icon occupies the same space and
the labels line up on an even rhythm, regardless of icon size. */
.composer-pill-glyph {
@apply relative inline-flex w-[19px] shrink-0 items-center justify-center transition-opacity;
@apply relative inline-flex w-[var(--ui-icon-size)] shrink-0 items-center justify-center transition-opacity;
}
/* On hover the icon swaps for an X inside a soft circle (ChatGPT-style),
filling the icon slot so every pill's X is identical and centered. */
.composer-pill-x {
@apply pointer-events-none absolute inset-0 m-auto size-[19px] rounded-full bg-primary/15 p-[3px] opacity-0 transition-opacity dark:bg-white/[0.14];
@apply pointer-events-none absolute inset-0 m-auto size-[var(--ui-icon-size)] rounded-full bg-primary/15 p-[3px] opacity-0 transition-opacity dark:bg-white/[0.14];
}
/* Icon-only (compact) pills are too small for the circle, so show a bare x. */
[data-pill-compact="true"]
.composer-pill-btn:not([data-keep-label])
.composer-pill-x {
@apply size-[15px] bg-transparent p-0 dark:bg-transparent;
@apply size-[min(calc(15px*var(--ui-font-scale,1)),calc(7.5px+7.5px*var(--ui-font-scale,1)))] bg-transparent p-0 dark:bg-transparent;
}
/* Compact pills hide their labels, so surface the name as a hover
@ -1853,8 +1858,8 @@ html[data-chat-font] .aui-root {
/* Smaller tick for selected Thinking options. */
.unsloth-tick {
width: 0.8rem !important;
height: 0.8rem !important;
width: min(calc(0.8rem * var(--ui-font-scale, 1)), calc(0.4rem + 0.4rem * var(--ui-font-scale, 1))) !important;
height: min(calc(0.8rem * var(--ui-font-scale, 1)), calc(0.4rem + 0.4rem * var(--ui-font-scale, 1))) !important;
}
/* Soft elevation; [data-slot] outranks the component ring-1, dropping the border. */
@ -1943,9 +1948,9 @@ html[data-chat-font] .aui-root {
[data-slot="dropdown-menu-item"],
[data-slot="dropdown-menu-sub-trigger"]
)
svg {
width: 1.15rem;
height: 1.15rem;
svg:not(.unsloth-tick) {
width: var(--ui-icon-size) !important;
height: var(--ui-icon-size) !important;
}
/* Destructive items keep red text and a red-tinted hover, not the grey one. */
@ -2770,3 +2775,96 @@ html[data-chat-font] .aui-root {
display: block !important;
width: 8px;
}
/* Icons that sit beside scaled labels follow the UI font size itself:
glyphs at or above a 16px base render at --ui-icon-size (12 -> 12px,
16 -> 16px, 20 -> 18px), so icons track the text below the default and
read slightly smaller than it above. Sub-16px glyphs keep their
proportions through the same curve as a factor. Menu, select and closed select trigger surfaces, popovers, toasts,
the chat thread and both composers. Only glyphs scale; hit targets,
paddings and surface geometry stay fixed. Identity at the default. */
:is(
[data-slot='dropdown-menu-content'],
[data-slot='dropdown-menu-sub-content'],
[data-slot='select-content'],
[data-slot='select-trigger'],
[data-slot='combobox-content'],
[data-slot='combobox-trigger'],
[data-slot='context-menu-content'],
[data-slot='context-menu-sub-content'],
[data-slot='menubar-content'],
[data-slot='popover-content'],
[data-slot='command'],
[data-sonner-toast],
.composer-action-wrapper,
.aui-composer-action-wrapper,
.aui-action-bar-more-content,
.aui-root
) {
& svg.size-2\.5 { width: min(calc(0.625rem * var(--ui-font-scale, 1)), calc(0.3125rem + 0.3125rem * var(--ui-font-scale, 1))); height: min(calc(0.625rem * var(--ui-font-scale, 1)), calc(0.3125rem + 0.3125rem * var(--ui-font-scale, 1))); }
& svg.size-3 { width: min(calc(0.75rem * var(--ui-font-scale, 1)), calc(0.375rem + 0.375rem * var(--ui-font-scale, 1))); height: min(calc(0.75rem * var(--ui-font-scale, 1)), calc(0.375rem + 0.375rem * var(--ui-font-scale, 1))); }
& svg.size-3\.5 { width: min(calc(0.875rem * var(--ui-font-scale, 1)), calc(0.4375rem + 0.4375rem * var(--ui-font-scale, 1))); height: min(calc(0.875rem * var(--ui-font-scale, 1)), calc(0.4375rem + 0.4375rem * var(--ui-font-scale, 1))); }
& svg.size-4 { width: var(--ui-icon-size); height: var(--ui-icon-size); }
& svg.size-4\.5 { width: var(--ui-icon-size); height: var(--ui-icon-size); }
& svg.size-5 { width: var(--ui-icon-size); height: var(--ui-icon-size); }
& svg.size-6 { width: min(calc(1.5rem * var(--ui-font-scale, 1)), calc(0.75rem + 0.75rem * var(--ui-font-scale, 1))); height: min(calc(1.5rem * var(--ui-font-scale, 1)), calc(0.75rem + 0.75rem * var(--ui-font-scale, 1))); }
& svg.size-\[5px\] { width: min(calc(5px * var(--ui-font-scale, 1)), calc(2.5px + 2.5px * var(--ui-font-scale, 1))); height: min(calc(5px * var(--ui-font-scale, 1)), calc(2.5px + 2.5px * var(--ui-font-scale, 1))); }
& svg.size-\[6px\] { width: min(calc(6px * var(--ui-font-scale, 1)), calc(3px + 3px * var(--ui-font-scale, 1))); height: min(calc(6px * var(--ui-font-scale, 1)), calc(3px + 3px * var(--ui-font-scale, 1))); }
& svg.size-\[10px\] { width: min(calc(10px * var(--ui-font-scale, 1)), calc(5px + 5px * var(--ui-font-scale, 1))); height: min(calc(10px * var(--ui-font-scale, 1)), calc(5px + 5px * var(--ui-font-scale, 1))); }
& svg.size-\[11px\] { width: min(calc(11px * var(--ui-font-scale, 1)), calc(5.5px + 5.5px * var(--ui-font-scale, 1))); height: min(calc(11px * var(--ui-font-scale, 1)), calc(5.5px + 5.5px * var(--ui-font-scale, 1))); }
& svg.size-\[12px\] { width: min(calc(12px * var(--ui-font-scale, 1)), calc(6px + 6px * var(--ui-font-scale, 1))); height: min(calc(12px * var(--ui-font-scale, 1)), calc(6px + 6px * var(--ui-font-scale, 1))); }
& svg.size-\[13px\] { width: min(calc(13px * var(--ui-font-scale, 1)), calc(6.5px + 6.5px * var(--ui-font-scale, 1))); height: min(calc(13px * var(--ui-font-scale, 1)), calc(6.5px + 6.5px * var(--ui-font-scale, 1))); }
& svg.size-\[14px\] { width: min(calc(14px * var(--ui-font-scale, 1)), calc(7px + 7px * var(--ui-font-scale, 1))); height: min(calc(14px * var(--ui-font-scale, 1)), calc(7px + 7px * var(--ui-font-scale, 1))); }
& svg.size-\[15px\] { width: min(calc(15px * var(--ui-font-scale, 1)), calc(7.5px + 7.5px * var(--ui-font-scale, 1))); height: min(calc(15px * var(--ui-font-scale, 1)), calc(7.5px + 7.5px * var(--ui-font-scale, 1))); }
& svg.size-\[15\.5px\] { width: min(calc(15.5px * var(--ui-font-scale, 1)), calc(7.75px + 7.75px * var(--ui-font-scale, 1))); height: min(calc(15.5px * var(--ui-font-scale, 1)), calc(7.75px + 7.75px * var(--ui-font-scale, 1))); }
& svg.size-\[16px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
& svg.size-\[17px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
& svg.size-\[18px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
& svg.size-\[18\.5px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
& svg.size-\[20px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
& svg.size-\[21px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
& svg.size-\[22px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
& svg.size-\[36px\] { width: min(calc(36px * var(--ui-font-scale, 1)), calc(18px + 18px * var(--ui-font-scale, 1))); height: min(calc(36px * var(--ui-font-scale, 1)), calc(18px + 18px * var(--ui-font-scale, 1))); }
& svg.w-3 { width: min(calc(0.75rem * var(--ui-font-scale, 1)), calc(0.375rem + 0.375rem * var(--ui-font-scale, 1))); }
& svg.h-3 { height: min(calc(0.75rem * var(--ui-font-scale, 1)), calc(0.375rem + 0.375rem * var(--ui-font-scale, 1))); }
& svg.w-3\.5 { width: min(calc(0.875rem * var(--ui-font-scale, 1)), calc(0.4375rem + 0.4375rem * var(--ui-font-scale, 1))); }
& svg.h-3\.5 { height: min(calc(0.875rem * var(--ui-font-scale, 1)), calc(0.4375rem + 0.4375rem * var(--ui-font-scale, 1))); }
& svg.w-4 { width: var(--ui-icon-size); }
& svg.h-4 { height: var(--ui-icon-size); }
& svg.w-5 { width: var(--ui-icon-size); }
& svg.h-5 { height: var(--ui-icon-size); }
/* Buttons default un-classed icons to size-4 the same way. Sonner's
close button keeps its compact 12px glyph inside a fixed control. */
& button:not([class*=':size-3'], [data-close-button]) svg:not([class*='size-'], [class*='w-'], [class*='h-'], .unsloth-tick) {
width: var(--ui-icon-size);
height: var(--ui-icon-size);
}
/* Menu items default un-classed icons to size-4. */
& [data-slot*='item'] svg:not([class*='size-'], [class*='w-'], [class*='h-']) {
width: var(--ui-icon-size);
height: var(--ui-icon-size);
}
}
/* Sonner injects fixed 13px toast text and 12px action labels at runtime;
text follows the preference at full rate. Line heights are unitless so
they track automatically. */
[data-sonner-toast][data-styled='true'] {
font-size: calc(13px * var(--ui-font-scale, 1)) !important;
}
[data-sonner-toast][data-styled='true'] [data-description] {
font-size: calc(13px * var(--ui-font-scale, 1)) !important;
}
[data-sonner-toast][data-styled='true'] [data-button] {
font-size: calc(12px * var(--ui-font-scale, 1)) !important;
}
/* Sonner's icon well is a fixed 16px box; track the glyph. */
[data-sonner-toast][data-styled='true'] [data-icon] {
width: var(--ui-icon-size) !important;
height: var(--ui-icon-size) !important;
}
/* Defensive: the built-in loader is unused (a custom loading icon is always
passed) but keep its fixed --size on the scale in case that changes. */
[data-sonner-toast] .sonner-loading-wrapper {
--size: var(--ui-icon-size) !important;
}

View file

@ -46,10 +46,13 @@ fn save_filter(file_name: &str) -> (&'static str, Vec<&'static str>) {
Some("jsonl") | Some("ndjson") => ("JSON Lines", vec!["jsonl", "ndjson"]),
Some("csv") => ("CSV", vec!["csv"]),
Some("md") | Some("markdown") => ("Markdown", vec!["md", "markdown"]),
Some("html") | Some("htm") => ("HTML", vec!["html", "htm"]),
Some("zip") => ("ZIP archive", vec!["zip"]),
_ => (
"Export files",
vec!["json", "jsonl", "ndjson", "csv", "md", "markdown", "zip"],
vec![
"json", "jsonl", "ndjson", "csv", "md", "markdown", "html", "htm", "zip",
],
),
}
}
@ -252,6 +255,12 @@ mod tests {
);
}
#[test]
fn html_canvas_exports_use_an_html_save_filter() {
assert_eq!(save_filter("canvas.html"), ("HTML", vec!["html", "htm"]));
assert_eq!(save_filter("canvas.HTM"), ("HTML", vec!["html", "htm"]));
}
#[test]
fn reads_supported_import_and_rejects_other_extensions() {
let jsonl_path = temp_path("allowed").with_extension("JSONL");

View file

@ -16,7 +16,7 @@
"app": {
"withGlobalTauri": true,
"security": {
"csp": "default-src 'self'; connect-src 'self' http://localhost:* ws://localhost:* ws://127.0.0.1:* http://127.0.0.1:* https://huggingface.co https://*.huggingface.co https://datasets-server.huggingface.co; img-src 'self' data: blob: https:; media-src 'self' data: blob: https:; style-src 'self' 'unsafe-inline'; font-src 'self' data:"
"csp": "default-src 'self'; connect-src 'self' http://localhost:* ws://localhost:* ws://127.0.0.1:* http://127.0.0.1:* https://huggingface.co https://*.huggingface.co https://datasets-server.huggingface.co; img-src 'self' data: blob: https:; media-src 'self' data: blob: https:; style-src 'self' 'unsafe-inline'; font-src 'self' data:; frame-src 'self' http://localhost:* http://127.0.0.1:*"
},
"windows": [
{

View file

@ -70,6 +70,8 @@ ART_DIR = os.environ.get("PW_ART_DIR", "logs/playwright_modelcfg")
ART = Path(ART_DIR)
ART.mkdir(parents = True, exist_ok = True)
STRICT = os.environ.get("STUDIO_UI_STRICT", "0") == "1"
PLAYWRIGHT_BROWSER = os.environ.get("STUDIO_PLAYWRIGHT_BROWSER", "chromium").lower()
PLAYWRIGHT_CHANNEL = os.environ.get("STUDIO_PLAYWRIGHT_CHANNEL") or None
TURN_TIMEOUT_MS = int(os.environ.get("STUDIO_UI_TURN_TIMEOUT_MS", "180000"))
WALL_TIMEOUT_S = float(os.environ.get("STUDIO_UI_WALL_TIMEOUT_S", "720"))
FETCH_TIMEOUT_MS = int(os.environ.get("STUDIO_UI_FETCH_TIMEOUT_MS", "30000"))
@ -145,10 +147,19 @@ with sync_playwright() as p:
)
# Health pre-flight: bash-side health wait can pass before the auth DB migrates.
wait_for_health(BASE, timeout = 30.0, info = info)
browser = p.chromium.launch(
headless = True,
args = chromium_launch_args(),
)
if PLAYWRIGHT_BROWSER not in ("chromium", "firefox", "webkit"):
fail(f"unsupported STUDIO_PLAYWRIGHT_BROWSER={PLAYWRIGHT_BROWSER!r}")
sys.exit(1)
browser_type = getattr(p, PLAYWRIGHT_BROWSER)
launch_kwargs = {"headless": True}
if PLAYWRIGHT_BROWSER == "chromium":
launch_kwargs["args"] = chromium_launch_args()
if PLAYWRIGHT_CHANNEL:
launch_kwargs["channel"] = PLAYWRIGHT_CHANNEL
elif PLAYWRIGHT_CHANNEL:
fail("STUDIO_PLAYWRIGHT_CHANNEL requires chromium")
sys.exit(1)
browser = browser_type.launch(**launch_kwargs)
ctx = browser.new_context(
viewport = {"width": 1280, "height": 900},
reduced_motion = "reduce",
@ -464,11 +475,6 @@ with sync_playwright() as p:
else:
default_ctx = ctx_in.input_value()
info(f"default Context Length shown: {default_ctx!r}")
ctx_in.click()
ctx_in.fill(str(DISTINCT_CTX))
page.wait_for_timeout(300)
page.keyboard.press("Tab") # blur to commit
page.wait_for_timeout(300)
remember = popover.get_by_label("Remember for this model").first
if _count(remember):
try:
@ -477,12 +483,16 @@ with sync_playwright() as p:
remember.click()
else:
fail("'Remember for this model' checkbox not found")
ctx_in.click()
ctx_in.fill(str(DISTINCT_CTX))
page.wait_for_timeout(300)
shoot("05-ctx-set")
btn = primary_button(popover)
if btn is None:
fail("primary Load/Save button not found in run-settings")
else:
# Keep the input focused. The button click must commit the draft
# and use it in the same load request.
btn.click()
page.wait_for_timeout(2500)
shoot("06-after-load")
@ -570,6 +580,47 @@ with sync_playwright() as p:
else:
info("OK reset: distinctive context cleared from unsloth_model_configs")
shoot("08-after-reset")
# ─────────────────────────────────────────────────────
# 3b. Re-typing the value already shown must not pin an override (HARD).
# Entering the currently displayed native/default context commits no
# onChange (the value is unchanged), so the cached blur value must not be
# replayed into a stored override on Load. Otherwise re-typing the shown
# number, or doing so before a Reset, recreates a phantom context pin.
# ─────────────────────────────────────────────────────
step("re-typing the shown context does not pin an override")
ctx_in = context_input(popover)
native_default = _as_int(ctx_in.input_value()) if ctx_in else None
if ctx_in is None or native_default is None:
info("skip re-type-shown: Context Length input has no numeric default")
else:
remember = popover.get_by_label("Remember for this model").first
if _count(remember):
try:
remember.check()
except Exception:
remember.click()
ctx_in.click()
ctx_in.fill(str(native_default))
page.wait_for_timeout(200)
btn = primary_button(popover)
if btn is not None and btn.is_enabled():
# Same-click Load: the button click must commit the draft, but a draft
# equal to the shown value carries no override.
btn.click()
page.wait_for_timeout(1500)
cfg = read_configs()
pinned = any(
_as_int(e.get("customContextLength")) == native_default for e in config_entries(cfg)
)
if pinned:
fail(
"re-typing the shown context pinned it as an override "
f"(customContextLength={native_default})"
)
else:
info("OK re-type-shown: shown context not stored as an override")
shoot("08b-after-retype-shown")
close_picker()
# ─────────────────────────────────────────────────────

View file

@ -208,6 +208,14 @@ def main():
# text-ui-12p5 at scale 0.75; 16px means twMerge dropped the token.
if not near(tab_font, 12.5 * 12 / 16):
fail(f"hub tab font did not scale (twMerge drop?): {tab_font}")
icon_w = page.evaluate(
"() => { const el = document.querySelector('.size-icon');"
" return el ? parseFloat(getComputedStyle(el).width) : null; }"
)
# Standard icons render at the UI font size itself below the
# default, so setting 12 gives 12px glyphs.
if not near(icon_w, 12):
fail(f"size-icon did not match the UI font size below 16: {icon_w}")
page.goto(BASE, wait_until = "domcontentloaded")
page.wait_for_timeout(1500)
open_appearance(page)

View file

@ -0,0 +1,26 @@
# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved.
from pathlib import Path
CHAT_API = (
Path(__file__).resolve().parents[2]
/ "studio"
/ "frontend"
/ "src"
/ "features"
/ "chat"
/ "api"
/ "chat-api.ts"
)
def test_length_detection_classifies_visible_and_reasoning_content():
source = CHAT_API.read_text(encoding = "utf-8")
assert "return value.trim().length > 0;" in source
assert 'record.type === "thinking" || record.type === "reasoning"' in source
assert 'record.type === "text" || record.type === "output_text"' in source
assert "sawAssistantContent ||= contentState.hasAssistantContent;" in source
assert "sawReasoningContent ||= contentState.hasReasoningContent;" in source

View file

@ -346,6 +346,36 @@ def test_fixed_layer_gguf_pins_displayed_context():
assert "customContextLength: activeLoadedContext" in src
def test_fixed_layer_pin_recomputed_after_committing_gpu_layers():
"""pinFixedLayerContext is computed from the render-time config, before a
same-click GPU Layers draft is committed. handleRun must recompute it from the
committed effectiveConfig; otherwise typing a positive GPU Layers value on an
auto-fit GGUF and clicking Reload saves customContextLength: null, so a later
fresh load sends the native context with fixed layers (the OOM the pin avoids)."""
src = _read("features/model-picker/components/model-config-page.tsx")
assert "const effectivePinFixedLayerContext =" in src
assert 'effectiveConfig.gpuMemoryMode === "manual"' in src
assert "effectiveConfig.gpuLayers != null" in src
assert "effectiveConfig.customContextLength == null" in src
assert "{ ...effectiveConfig, customContextLength: activeLoadedContext }" in src
def test_blur_cache_cleared_on_every_settled_render():
"""The lastBlurCommittedRef bridge is valid only across the single synchronous
same-click gesture that set it. Keying its clear on [value] missed a Reset (or
external edit) that restores the shown value unchanged after the blur dispatched
onChange: value nets back to its prior number, the effect never re-ran, and a
later Load/Save replayed the override Reset removed. Clear it on every settled
render instead."""
src = _read("features/model-picker/components/numeric-value-input.tsx")
# The clearing effect must run on every commit, not be gated on [value] alone.
assert not re.search(r"lastBlurCommittedRef\.current = null;\s*\}, \[value\]\);", src)
assert re.search(
r"useEffect\(\(\) => \{\s*lastBlurCommittedRef\.current = null;\s*\}\);",
src,
)
def test_auto_defaults_not_persisted_as_overrides():
"""Auto GPU memory mode and Auto/default speculative type are follow-global
defaults; normalization must not persist them as per-model overrides, else a
@ -382,15 +412,90 @@ def test_reset_persists_null_max_length_and_substitutes_only_for_load():
Reset) so isDefaultConfig can clear a remembered override; the concrete
fallback is substituted only into the load request, not the saved record."""
src = _read("features/model-picker/components/model-config-page.tsx")
# Load-only substitution of the resolved value.
assert "maxSeqLength: maxSeqLengthValue" in src
assert "const loadConfig" in src
# The persisted record is loaded via onRun(loadConfig), and save uses the
# untouched runtimeConfig (so a reset/default config stays default).
assert "onRun(loadConfig)" in src
# Load-only substitution of the resolved value (recomputed from any committed
# same-click Max Seq Length draft, so it is never dropped).
assert "maxSeqLength: effectiveMaxSeqLengthValue" in src
assert "const effectiveLoadConfig" in src
# The persisted record is saved from effectiveRuntimeConfig; the load request
# carries effectiveLoadConfig (with any committed context input).
assert "onRun(effectiveLoadConfig)" in src
assert "savePerModelConfig(" in src
def test_initial_load_uses_staged_config_payload():
"""Run-settings Load must pass the staged config through to /load even when
React has not flushed NumericValueInput blur commits into the store yet."""
runtime = _read("features/chat/hooks/use-chat-model-runtime.ts")
assert "const pendingLoadConfig =" in runtime
assert "pendingLoadConfig?.kvCacheDtype" in runtime
assert "pendingLoadConfig?.customContextLength" in runtime
page = _read("features/model-picker/components/model-config-page.tsx")
assert "contextInputRef" in page
assert "contextInputRef.current?.commit()" in page
numeric = _read("features/model-picker/components/numeric-value-input.tsx")
assert "export type NumericValueInputHandle" in numeric
assert "commit:" in numeric
# P1: commit returns null unless the user actually edited the field,
# so Load/Save with untouched Auto does not pin native context.
assert "dirtyRef.current" in numeric
assert "return null;" in numeric
# P2: blur clears dirtyRef after commit so Reset/slider cannot be
# overwritten by a stale draft on a later Load.
assert "dirtyRef.current = false;" in numeric
assert "draftRef.current = String(final);" in numeric
# Same-click Load after blur still sees the committed draft.
assert "lastBlurCommittedRef" in numeric
# Invalid drafts must not turn Auto into an explicit pin.
assert "const commitDraft = (raw: string): number | null" in numeric
assert re.search(r"if \(!Number\.isFinite\(parsed\)\) \{\s*return null;", numeric)
assert re.search(
r"if \(final == null\) \{\s*"
r"draftRef\.current = String\(value\);\s*"
r"lastBlurCommittedRef\.current = null;",
numeric,
)
# handleRun only promotes commit() when non-null.
assert "committedContext != null" in page
assert "pendingPatch.customContextLength = committedContext;" in page
def test_same_click_commit_covers_all_numeric_inputs():
"""The same-click blur bridge must flush every NumericValueInput-backed
setting, not just Context Length. Max Seq Length (non-GGUF), GPU Layers and
MoE Layers (GGUF) also stage their draft only on blur, so handleRun must
imperatively commit each and fold the value into the staged load config;
otherwise a value the user typed right before clicking Load/Reload is lost."""
page = _read("features/model-picker/components/model-config-page.tsx")
# Each numeric input owns an imperative handle that handleRun commits, and the
# handle is forwarded down to the actual NumericValueInput.
for ref in ("maxSeqLengthInputRef", "gpuLayersInputRef", "moeLayersInputRef"):
assert f"const {ref} = useRef<NumericValueInputHandle>(null);" in page
assert f"{ref}.current?.commit()" in page
assert f"inputRef={{{ref}}}" in page
# The leaf sub-components accept and forward the handle as a ref.
assert page.count("inputRef?: Ref<NumericValueInputHandle>;") >= 2
assert "ref={inputRef}" in page
# Committed drafts are folded into the staged config, gated on non-null so an
# untouched field never fabricates an override.
assert "committedMaxSeqLength != null" in page
assert "committedGpuLayers != null" in page
assert "committedMoeLayers != null" in page
assert "pendingPatch.gpuLayers = committedGpuLayers;" in page
assert "pendingPatch.nCpuMoe = committedMoeLayers;" in page
# The non-GGUF load path substitutes the committed Max Seq Length draft.
assert "const effectiveMaxSeqLengthValue =" in page
assert "maxSeqLength: effectiveMaxSeqLengthValue" in page
def test_context_commit_rechecks_persistence_only_shortcut():
"""Committed context changes must bypass persistence-only saves."""
src = _read("features/model-picker/components/model-config-page.tsx")
assert "const effectiveConfig =" in src
assert "perModelConfigsEqual(effectiveConfig, baseline)" in src
assert "const effectivePersistenceOnly =" in src
assert "if (effectivePersistenceOnly)" in src
def test_reset_enabled_for_explicit_context_pin_at_native():
"""An explicit customContextLength that equals the native ceiling is still a
user override, so contextAtDefault must require customContextLength == null.

View file

@ -98,6 +98,39 @@ def test_cn_knows_the_ui_typography_tokens():
assert "/^ui-\\d+(p5)?$/.test(value)" in UTILS
def test_icons_follow_the_ui_font_size_itself():
"""Standard glyphs render at --ui-icon-size, which follows the UI font
size itself: matches it below the 16px default and grows at half the
change above it (setting 20 gives 18px icons), so icons track the text
when shrinking and read slightly smaller than it when growing. Sub 16px
glyphs keep their proportions through the same curve as a factor.
Sonner toast text and action labels are text, so they follow at full
rate everywhere."""
assert (
"--ui-icon-size: min(calc(1rem * var(--ui-font-scale, 1)), "
"calc(0.5rem + 0.5rem * var(--ui-font-scale, 1)));"
) in INDEX_CSS
assert "--icon-size: var(--ui-icon-size);" in INDEX_CSS
assert "& svg.size-4 { width: var(--ui-icon-size); height: var(--ui-icon-size); }" in INDEX_CSS
assert "font-size: calc(13px * var(--ui-font-scale, 1)) !important;" in INDEX_CSS
assert "font-size: calc(12px * var(--ui-font-scale, 1)) !important;" in INDEX_CSS
# Menu rules that outrank the scoped block must carry the token too,
# without flattening the smaller thinking ticks.
assert "width: var(--ui-icon-size) !important;" in INDEX_CSS
assert "svg:not(.unsloth-tick) {" in INDEX_CSS
# Oversized art glyphs stay proportional instead of uniform.
assert "& svg.size-6 { width: min(calc(1.5rem" in INDEX_CSS
for scope in (
"[data-slot='dropdown-menu-content']",
"[data-slot='select-content']",
"[data-slot='select-trigger']",
"[data-slot='combobox-content']",
"[data-sonner-toast]",
".aui-root",
):
assert scope in INDEX_CSS
def test_no_raw_pixel_text_utilities():
offenders = []
for path in _frontend_sources():