Merge branch 'main' into fix/studio-tool-chat-respawn
This commit is contained in:
commit
3bfe71b4c7
25 changed files with 1122 additions and 171 deletions
|
|
@ -307,6 +307,26 @@ def _native_linux_system_rocm_lib_dirs(binary_dir: str = "") -> "list[str]":
|
|||
_DEFAULT_MAX_TOKENS_FLOOR = 32768
|
||||
_DEFAULT_FIRST_TOKEN_TIMEOUT_S = 1200.0 # 20 min
|
||||
|
||||
|
||||
def _finalize_reasoning_only_cumulative(
|
||||
cumulative: str, reasoning_text: str, finish_reason: Optional[str], promote_reasoning_only: bool
|
||||
) -> str:
|
||||
"""Close a live thinking block and promote it only after a clean stop.
|
||||
|
||||
Local inference streams cumulative snapshots. Replacing ``<think>...`` with
|
||||
bare reasoning at EOF makes the final snapshot shorter, so suffix-based
|
||||
route consumers drop the intended fallback. Keep the snapshot append-only.
|
||||
A length-truncated thought is not a final answer, so close it without
|
||||
promotion and let the client surface the ``length`` terminal state. Raw
|
||||
consumers that do not split reasoning from visible content can disable the
|
||||
fallback to avoid returning the same reasoning twice.
|
||||
"""
|
||||
visible_fallback = (
|
||||
reasoning_text if promote_reasoning_only and finish_reason != "length" else ""
|
||||
)
|
||||
return cumulative + "</think>" + visible_fallback
|
||||
|
||||
|
||||
# Only large streamed tool payloads get an early provisional card; render_html
|
||||
# is exempt because it needs immediate artifact feedback.
|
||||
_PROVISIONAL_ARGS_MIN_CHARS = 256
|
||||
|
|
@ -10586,6 +10606,7 @@ class LlamaCppBackend:
|
|||
reasoning_effort: Optional[str] = None,
|
||||
preserve_thinking: Optional[bool] = None,
|
||||
seed: Optional[int] = None,
|
||||
promote_reasoning_only: bool = True,
|
||||
_allow_respawn_retry: bool = True,
|
||||
) -> Generator[Union[str, dict], None, None]:
|
||||
"""
|
||||
|
|
@ -10668,7 +10689,12 @@ class LlamaCppBackend:
|
|||
# model put its whole reply in reasoning
|
||||
# (e.g. Qwen3 always-think). Show it as
|
||||
# the main response, not a thinking block.
|
||||
cumulative = reasoning_text
|
||||
cumulative = _finalize_reasoning_only_cumulative(
|
||||
cumulative,
|
||||
reasoning_text,
|
||||
_metadata_finish_reason,
|
||||
promote_reasoning_only,
|
||||
)
|
||||
yield cumulative
|
||||
_stream_done = True
|
||||
break # exit inner while
|
||||
|
|
@ -10765,6 +10791,7 @@ class LlamaCppBackend:
|
|||
reasoning_effort = reasoning_effort,
|
||||
preserve_thinking = preserve_thinking,
|
||||
seed = seed,
|
||||
promote_reasoning_only = promote_reasoning_only,
|
||||
_allow_respawn_retry = False,
|
||||
)
|
||||
return
|
||||
|
|
@ -10806,6 +10833,7 @@ class LlamaCppBackend:
|
|||
confirm_tool_calls: bool = False,
|
||||
bypass_permissions: bool = False,
|
||||
permission_mode: Optional[str] = None,
|
||||
promote_reasoning_only: bool = True,
|
||||
) -> Generator[dict, None, None]:
|
||||
"""
|
||||
Agentic loop: let the model call tools, execute them, and continue.
|
||||
|
|
@ -11147,7 +11175,12 @@ class LlamaCppBackend:
|
|||
),
|
||||
}
|
||||
else:
|
||||
cumulative_display = reasoning_accum
|
||||
cumulative_display = _finalize_reasoning_only_cumulative(
|
||||
cumulative_display,
|
||||
reasoning_accum,
|
||||
_iter_finish_reason,
|
||||
promote_reasoning_only,
|
||||
)
|
||||
if not _suppress_visible_output:
|
||||
yield {
|
||||
"type": "content",
|
||||
|
|
@ -11611,7 +11644,12 @@ class LlamaCppBackend:
|
|||
if _reasoning_started_at is not None and not _reasoning_summary_emitted:
|
||||
_reasoning_summary_emitted = True
|
||||
yield _reasoning_summary_event(_reasoning_started_at)
|
||||
cumulative_display = reasoning_accum
|
||||
cumulative_display = _finalize_reasoning_only_cumulative(
|
||||
cumulative_display,
|
||||
reasoning_accum,
|
||||
_iter_finish_reason,
|
||||
promote_reasoning_only,
|
||||
)
|
||||
if not _suppress_visible_output:
|
||||
yield {
|
||||
"type": "content",
|
||||
|
|
@ -12175,7 +12213,12 @@ class LlamaCppBackend:
|
|||
"text": _strip_tool_markup(cumulative, final = True),
|
||||
}
|
||||
else:
|
||||
cumulative = reasoning_text
|
||||
cumulative = _finalize_reasoning_only_cumulative(
|
||||
cumulative,
|
||||
reasoning_text,
|
||||
_metadata_finish_reason,
|
||||
promote_reasoning_only,
|
||||
)
|
||||
yield {"type": "content", "text": cumulative}
|
||||
_stream_done = True
|
||||
break # exit inner while
|
||||
|
|
|
|||
|
|
@ -1794,7 +1794,16 @@ router = APIRouter()
|
|||
studio_router = APIRouter()
|
||||
|
||||
|
||||
_ARTIFACT_PREVIEW_FRAME_ANCESTORS = "'self' tauri://localhost http://tauri.localhost"
|
||||
# Packaged desktop runs at tauri://localhost (macOS/Linux) or http://tauri.localhost
|
||||
# (Windows WebView2); the web build is same-origin ('self'). The `tauri dev` shell,
|
||||
# however, serves the frontend from the Vite dev origin (http://localhost:5173),
|
||||
# so the packaged allowlist alone leaves the preview blocked in dev with an
|
||||
# "ancestor violates frame-ancestors" error. This shell exposes no server resource
|
||||
# (it only renders postMessage'd HTML in a no-same-origin sandbox), so also allowing
|
||||
# any localhost/127.0.0.1 dev origin to frame it is safe and unblocks the dev shell.
|
||||
_ARTIFACT_PREVIEW_FRAME_ANCESTORS = (
|
||||
"'self' tauri://localhost http://tauri.localhost http://localhost:* http://127.0.0.1:*"
|
||||
)
|
||||
_ARTIFACT_PREVIEW_FRAME_STRICT_CSP = (
|
||||
"default-src 'none'; "
|
||||
"script-src 'unsafe-inline'; "
|
||||
|
|
@ -13355,6 +13364,7 @@ async def anthropic_messages(
|
|||
disable_parallel_tool_use = _disable_parallel,
|
||||
bypass_permissions = bool(payload.bypass_permissions),
|
||||
permission_mode = getattr(payload, "permission_mode", None),
|
||||
promote_reasoning_only = False,
|
||||
)
|
||||
|
||||
if payload.stream:
|
||||
|
|
@ -13394,6 +13404,7 @@ async def anthropic_messages(
|
|||
max_tokens = payload.max_tokens,
|
||||
stop = stop,
|
||||
cancel_event = cancel_event,
|
||||
promote_reasoning_only = False,
|
||||
)
|
||||
|
||||
if payload.stream:
|
||||
|
|
|
|||
|
|
@ -68,16 +68,15 @@ def _emitter_client_text(events: list[str]) -> str:
|
|||
|
||||
|
||||
def test_anthropic_emitter_closes_reasoning_only_think_block():
|
||||
# A reasoning-only reply streams <think>X live then shrinks to bare X at EOF.
|
||||
# This emitter diffs cumulative snapshots and drops the shrink, so without a
|
||||
# closing pass the client text would end on an unclosed <think>. finish()
|
||||
# must balance it.
|
||||
# Anthropic asks the GGUF generator not to promote reasoning into a duplicate
|
||||
# visible fallback, so its final cumulative snapshot only balances the block.
|
||||
emitter = AnthropicStreamEmitter()
|
||||
events = emitter.start("msg_1", "m")
|
||||
events += emitter.feed({"type": "content", "text": "<think>The capital"})
|
||||
events += emitter.feed({"type": "content", "text": "<think>The capital of France is Paris."})
|
||||
# The generator's final bare-text shrink (dropped by the cumulative diff).
|
||||
events += emitter.feed({"type": "content", "text": "The capital of France is Paris."})
|
||||
events += emitter.feed(
|
||||
{"type": "content", "text": "<think>The capital of France is Paris.</think>"}
|
||||
)
|
||||
events += emitter.finish()
|
||||
|
||||
assert _emitter_client_text(events) == "<think>The capital of France is Paris.</think>"
|
||||
|
|
@ -1563,6 +1562,44 @@ class TestAnthropicMessagesToolRouting:
|
|||
assert entry["context_length"] == 2048
|
||||
assert monitor.active_count() == 0
|
||||
|
||||
@pytest.mark.parametrize("stream", [False, True])
|
||||
@pytest.mark.parametrize("with_tools", [False, True])
|
||||
def test_reasoning_only_output_is_not_duplicated(self, monkeypatch, stream, with_tools):
|
||||
reasoning = "The capital of France is Paris."
|
||||
|
||||
def _gen_plain(**kwargs):
|
||||
assert kwargs["promote_reasoning_only"] is False
|
||||
yield f"<think>{reasoning}"
|
||||
yield f"<think>{reasoning}</think>"
|
||||
|
||||
def _gen_tools(**kwargs):
|
||||
assert kwargs["promote_reasoning_only"] is False
|
||||
yield {"type": "content", "text": f"<think>{reasoning}"}
|
||||
yield {"type": "content", "text": f"<think>{reasoning}</think>"}
|
||||
|
||||
_mock_backend(
|
||||
monkeypatch,
|
||||
generate_chat_completion = _gen_plain,
|
||||
generate_chat_completion_with_tools = _gen_tools,
|
||||
)
|
||||
payload_fields = {"stream": stream}
|
||||
if with_tools:
|
||||
payload_fields.update(
|
||||
{
|
||||
"enable_tools": True,
|
||||
"tools": [{"type": "web_search_20250305", "name": "web_search"}],
|
||||
}
|
||||
)
|
||||
payload = _basic_payload(**payload_fields)
|
||||
|
||||
response = _drive(anthropic_messages(payload, request = self._Request(), current_subject = "t"))
|
||||
if stream:
|
||||
body = self._sse_blob(self._consume_response(response))
|
||||
assert body.count(reasoning) == 1
|
||||
else:
|
||||
body = json.loads(response.body)
|
||||
assert body["content"][0]["text"] == f"<think>{reasoning}</think>"
|
||||
|
||||
def test_tool_use_non_streaming_records_api_monitor_reply(self, monkeypatch):
|
||||
import routes.inference as inf_mod
|
||||
|
||||
|
|
|
|||
|
|
@ -789,9 +789,12 @@ class TestLoadHubDownloadExclusion:
|
|||
|
||||
# The gguf_load_in_flight marker must be entered before the hub-download
|
||||
# guard and the unload so a concurrent load can't race the download
|
||||
# manager. The llama_extra_args inheritance that used to sit between the
|
||||
# marker and the guard now runs in _guard_chat_load_against_training, ahead
|
||||
# of the GGUF branch, so it is no longer a landmark inside this slice.
|
||||
# manager. The llama_extra_args inheritance moved out of the branch into
|
||||
# _resolve_inherited_extra_args, which must still run BEFORE it: the
|
||||
# inherited value (e.g. a carried --no-mmproj) shapes the guard's
|
||||
# require_mmproj. Anchor on the call form so the assertion pins the
|
||||
# endpoint's call site, not the function definition.
|
||||
assert source.index("= _resolve_inherited_extra_args(") < source.index("if config.is_gguf:")
|
||||
assert (
|
||||
gguf_branch.index("enter_context(gguf_load_in_flight")
|
||||
< gguf_branch.index("_hub_download_blocks_gguf_load")
|
||||
|
|
|
|||
|
|
@ -37,6 +37,24 @@ def _done() -> str:
|
|||
return "data: [DONE]\n"
|
||||
|
||||
|
||||
def _finish(reason: str) -> str:
|
||||
return (
|
||||
"data: "
|
||||
+ json.dumps(
|
||||
{
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {},
|
||||
"finish_reason": reason,
|
||||
}
|
||||
]
|
||||
}
|
||||
)
|
||||
+ "\n"
|
||||
)
|
||||
|
||||
|
||||
def _make_backend(
|
||||
monkeypatch,
|
||||
streams: list[object],
|
||||
|
|
@ -327,9 +345,8 @@ def test_reasoning_streams_incrementally_with_tools(monkeypatch):
|
|||
def test_reasoning_only_reply_matches_no_tool_path_with_tools(monkeypatch):
|
||||
# A reasoning-only turn (whole answer in reasoning_content, no content, no
|
||||
# tool) with a tool active streams the reasoning live, then resolves to the
|
||||
# bare reasoning text -- identical to the no-tool generate_chat_completion
|
||||
# path -- so the non-streaming drain still returns it as `content`, not an
|
||||
# empty answer.
|
||||
# same text on the visible channel. The final cumulative snapshot stays
|
||||
# append-only so route suffix extraction cannot drop that fallback.
|
||||
stream = [
|
||||
_sse({"reasoning_content": "The capital of France is Paris."}),
|
||||
_done(),
|
||||
|
|
@ -349,8 +366,49 @@ def test_reasoning_only_reply_matches_no_tool_path_with_tools(monkeypatch):
|
|||
content_texts = [e["text"] for e in events if e["type"] == "content"]
|
||||
# Reasoning streamed live during BUFFERING (the fix).
|
||||
assert content_texts[0] == "<think>The capital of France is Paris."
|
||||
# Resolves to bare reasoning, matching the no-tool sibling.
|
||||
assert content_texts[-1] == "The capital of France is Paris."
|
||||
assert content_texts[-1] == (
|
||||
"<think>The capital of France is Paris.</think>The capital of France is Paris."
|
||||
)
|
||||
|
||||
|
||||
def _assert_reasoning_only_raw_consumer_gets_one_balanced_think_block(monkeypatch, with_tools):
|
||||
stream = [
|
||||
_sse({"reasoning_content": "The capital of France is Paris."}),
|
||||
_done(),
|
||||
]
|
||||
backend = _make_backend(monkeypatch, [stream], [])
|
||||
|
||||
if with_tools:
|
||||
items = list(
|
||||
backend.generate_chat_completion_with_tools(
|
||||
messages = [{"role": "user", "content": "capital of France?"}],
|
||||
tools = [{"type": "function", "function": {"name": "web_search"}}],
|
||||
max_tool_iterations = 1,
|
||||
promote_reasoning_only = False,
|
||||
)
|
||||
)
|
||||
cumulatives = [item["text"] for item in items if item.get("type") == "content"]
|
||||
else:
|
||||
items = list(
|
||||
backend.generate_chat_completion(
|
||||
messages = [{"role": "user", "content": "capital of France?"}],
|
||||
promote_reasoning_only = False,
|
||||
)
|
||||
)
|
||||
cumulatives = [item for item in items if isinstance(item, str)]
|
||||
|
||||
assert cumulatives[-1] == "<think>The capital of France is Paris.</think>"
|
||||
assert all(
|
||||
current.startswith(previous) for previous, current in zip([""] + cumulatives, cumulatives)
|
||||
)
|
||||
|
||||
|
||||
def test_reasoning_only_raw_consumer_without_tools_gets_one_balanced_think_block(monkeypatch):
|
||||
_assert_reasoning_only_raw_consumer_gets_one_balanced_think_block(monkeypatch, False)
|
||||
|
||||
|
||||
def test_reasoning_only_raw_consumer_with_tools_gets_one_balanced_think_block(monkeypatch):
|
||||
_assert_reasoning_only_raw_consumer_gets_one_balanced_think_block(monkeypatch, True)
|
||||
|
||||
|
||||
def test_reasoning_before_structured_tool_closes_think_block(monkeypatch):
|
||||
|
|
@ -420,8 +478,8 @@ def _replay_route_reasoning_extractor(cumulatives: list[str]) -> tuple[str, str]
|
|||
def test_reasoning_only_route_output_matches_no_tool_path(monkeypatch):
|
||||
# Parity contract: a reasoning-only reply must reach the client identically
|
||||
# whether tools are on or off. Both generators stream <think> live then
|
||||
# resolve to the bare reasoning text; the route's suffix-diff + extractor
|
||||
# must therefore produce the same (visible, reasoning) split for both.
|
||||
# append a balanced close plus visible fallback; the route's suffix-diff +
|
||||
# extractor must therefore produce the same split for both.
|
||||
stream = [
|
||||
_sse({"reasoning_content": "The capital"}),
|
||||
_sse({"reasoning_content": " of France is Paris."}),
|
||||
|
|
@ -458,10 +516,37 @@ def test_reasoning_only_route_output_matches_no_tool_path(monkeypatch):
|
|||
no_tool_out = _replay_route_reasoning_extractor(no_tool_cumulatives)
|
||||
assert tool_out == no_tool_out
|
||||
# Pin the shared contract so a change to either path shows up here.
|
||||
_visible, reasoning = tool_out
|
||||
visible, reasoning = tool_out
|
||||
assert visible == "The capital of France is Paris."
|
||||
assert reasoning == "The capital of France is Paris."
|
||||
|
||||
|
||||
def test_length_truncated_reasoning_stays_append_only_without_visible_promotion(monkeypatch):
|
||||
stream = [
|
||||
_sse({"reasoning_content": "The proof begins by assuming finitely many primes."}),
|
||||
_finish("length"),
|
||||
_done(),
|
||||
]
|
||||
backend = _make_backend(monkeypatch, [stream], [])
|
||||
|
||||
items = list(
|
||||
backend.generate_chat_completion(
|
||||
messages = [{"role": "user", "content": "Prove infinitely many primes"}],
|
||||
max_tokens = 16,
|
||||
)
|
||||
)
|
||||
cumulatives = [item for item in items if isinstance(item, str)]
|
||||
|
||||
assert all(
|
||||
current.startswith(previous) for previous, current in zip([""] + cumulatives, cumulatives)
|
||||
)
|
||||
assert cumulatives[-1] == ("<think>The proof begins by assuming finitely many primes.</think>")
|
||||
visible, reasoning = _replay_route_reasoning_extractor(cumulatives)
|
||||
assert visible == ""
|
||||
assert reasoning == "The proof begins by assuming finitely many primes."
|
||||
assert items[-1]["finish_reason"] == "length"
|
||||
|
||||
|
||||
def test_reasoning_before_bare_json_tool_closes_think_block(monkeypatch):
|
||||
# _drain_silently sibling of the structured-tool close: a bare-JSON tool call
|
||||
# with a live reasoning prefix must also close </think> before draining, and
|
||||
|
|
|
|||
|
|
@ -524,8 +524,10 @@ export function AppProvider({ children }: AppProviderProps) {
|
|||
visibleToasts={2}
|
||||
expand={true}
|
||||
closeButton={true}
|
||||
// Clear the chat header buttons on the right.
|
||||
offset={{ top: 12, right: 64 }}
|
||||
// Clear the chat header buttons on the right. On desktop, also drop
|
||||
// below the ~34px custom window titlebar so toasts don't cover the
|
||||
// minimize / maximize / close controls.
|
||||
offset={{ top: isTauri ? 46 : 12, right: 64 }}
|
||||
/>
|
||||
</TooltipProvider>
|
||||
</MotionConfig>
|
||||
|
|
|
|||
|
|
@ -89,6 +89,7 @@ import {
|
|||
import { resolveLoadMaxSeqLength } from "../presets/preset-policy";
|
||||
import {
|
||||
generateAudio,
|
||||
GenerationLengthError,
|
||||
listCachedGguf,
|
||||
listCachedModels,
|
||||
listGgufVariants,
|
||||
|
|
@ -4093,7 +4094,15 @@ export function createOpenAIStreamAdapter(
|
|||
);
|
||||
if (!abortSignal.aborted) {
|
||||
const msg = err instanceof Error ? err.message : String(err);
|
||||
if (err instanceof StreamInterruptedError) {
|
||||
if (err instanceof GenerationLengthError) {
|
||||
toast.error("Response ran out of tokens", {
|
||||
description:
|
||||
"The model used the full Max Tokens budget while thinking " +
|
||||
"and did not produce a final answer. Increase Max Tokens in " +
|
||||
"chat Settings or turn off thinking, then retry.",
|
||||
duration: 8000,
|
||||
});
|
||||
} else if (err instanceof StreamInterruptedError) {
|
||||
// Connection dropped mid-turn: surface it explicitly (the rethrow
|
||||
// below also marks the message with an inline error + Retry).
|
||||
toast.error("Response interrupted", {
|
||||
|
|
|
|||
|
|
@ -50,6 +50,21 @@ export class StreamInterruptedError extends Error {
|
|||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Thrown when a reasoning model consumes its output budget before emitting any
|
||||
* standard content. Keeping this distinct from a dropped connection lets the
|
||||
* chat UI explain why a completed stream contains only a thinking panel.
|
||||
*/
|
||||
export class GenerationLengthError extends Error {
|
||||
constructor() {
|
||||
super(
|
||||
"The model reached the Max Tokens limit before producing a final answer. " +
|
||||
"Increase Max Tokens or disable thinking, then retry.",
|
||||
);
|
||||
this.name = "GenerationLengthError";
|
||||
}
|
||||
}
|
||||
|
||||
export function notifyChatHistoryUpdated(): void {
|
||||
if (typeof window !== "undefined") {
|
||||
window.dispatchEvent(new Event(CHAT_HISTORY_UPDATED_EVENT));
|
||||
|
|
@ -982,6 +997,61 @@ function parseSseEvent(rawEvent: string): string[] {
|
|||
return dataLines;
|
||||
}
|
||||
|
||||
function hasNonWhitespaceText(value: unknown): boolean {
|
||||
if (typeof value === "string") {
|
||||
return value.trim().length > 0;
|
||||
}
|
||||
if (Array.isArray(value)) {
|
||||
return value.some((item) => hasNonWhitespaceText(item));
|
||||
}
|
||||
if (!value || typeof value !== "object") {
|
||||
return false;
|
||||
}
|
||||
const record = value as Record<string, unknown>;
|
||||
return ["thinking", "text", "content", "reasoning", "summary"].some(
|
||||
(key) => key in record && hasNonWhitespaceText(record[key]),
|
||||
);
|
||||
}
|
||||
|
||||
function classifyStructuredDeltaContent(content: unknown): {
|
||||
hasAssistantContent: boolean;
|
||||
hasReasoningContent: boolean;
|
||||
} {
|
||||
if (typeof content === "string") {
|
||||
return {
|
||||
hasAssistantContent: hasNonWhitespaceText(content),
|
||||
hasReasoningContent: false,
|
||||
};
|
||||
}
|
||||
if (!Array.isArray(content)) {
|
||||
return {
|
||||
hasAssistantContent: false,
|
||||
hasReasoningContent: false,
|
||||
};
|
||||
}
|
||||
|
||||
let hasAssistantContent = false;
|
||||
let hasReasoningContent = false;
|
||||
for (const part of content) {
|
||||
if (typeof part === "string") {
|
||||
hasAssistantContent ||= hasNonWhitespaceText(part);
|
||||
continue;
|
||||
}
|
||||
if (!part || typeof part !== "object") {
|
||||
continue;
|
||||
}
|
||||
const record = part as Record<string, unknown>;
|
||||
if (record.type === "thinking" || record.type === "reasoning") {
|
||||
hasReasoningContent ||= hasNonWhitespaceText(record);
|
||||
} else if (record.type === "text" || record.type === "output_text") {
|
||||
const text =
|
||||
typeof record.text === "string" ? record.text : record.content;
|
||||
hasAssistantContent ||= hasNonWhitespaceText(text);
|
||||
}
|
||||
}
|
||||
return { hasAssistantContent, hasReasoningContent };
|
||||
}
|
||||
|
||||
export async function* streamChatCompletions(
|
||||
payload: OpenAIChatCompletionsRequest,
|
||||
signal: AbortSignal,
|
||||
|
|
@ -1009,6 +1079,19 @@ export async function* streamChatCompletions(
|
|||
// EOF without `[DONE]` or a finish_reason chunk means the stream was cut
|
||||
// mid-generation: surface as interrupted, not silent success.
|
||||
let sawTerminalSignal = false;
|
||||
let terminalFinishReason: string | null = null;
|
||||
let sawAssistantContent = false;
|
||||
let sawReasoningContent = false;
|
||||
|
||||
const throwIfReasoningOnlyLength = () => {
|
||||
if (
|
||||
terminalFinishReason === "length" &&
|
||||
sawReasoningContent &&
|
||||
!sawAssistantContent
|
||||
) {
|
||||
throw new GenerationLengthError();
|
||||
}
|
||||
};
|
||||
|
||||
try {
|
||||
while (true) {
|
||||
|
|
@ -1018,6 +1101,7 @@ export async function* streamChatCompletions(
|
|||
if (!sawTerminalSignal) {
|
||||
throw new StreamInterruptedError();
|
||||
}
|
||||
throwIfReasoningOnlyLength();
|
||||
break;
|
||||
}
|
||||
|
||||
|
|
@ -1039,6 +1123,7 @@ export async function* streamChatCompletions(
|
|||
if (dataText === "[DONE]") {
|
||||
completed = true;
|
||||
sawTerminalSignal = true;
|
||||
throwIfReasoningOnlyLength();
|
||||
return;
|
||||
}
|
||||
|
||||
|
|
@ -1094,11 +1179,31 @@ export async function* streamChatCompletions(
|
|||
}
|
||||
// finish_reason is a valid terminal signal for providers that close
|
||||
// the stream without an explicit [DONE] sentinel.
|
||||
const finishReason = (
|
||||
const parsedChoices = (
|
||||
parsed as {
|
||||
choices?: Array<{ finish_reason?: string | null }>;
|
||||
choices?: Array<{
|
||||
delta?: Record<string, unknown>;
|
||||
finish_reason?: string | null;
|
||||
}>;
|
||||
}
|
||||
).choices?.[0]?.finish_reason;
|
||||
).choices;
|
||||
for (const choice of parsedChoices ?? []) {
|
||||
const delta = choice.delta;
|
||||
if (delta) {
|
||||
const contentState = classifyStructuredDeltaContent(delta.content);
|
||||
sawAssistantContent ||= contentState.hasAssistantContent;
|
||||
sawReasoningContent ||= contentState.hasReasoningContent;
|
||||
const reasoning =
|
||||
delta.reasoning_content ??
|
||||
delta.reasoning ??
|
||||
delta.reasoning_details;
|
||||
sawReasoningContent ||= hasNonWhitespaceText(reasoning);
|
||||
}
|
||||
if (choice.finish_reason) {
|
||||
terminalFinishReason = choice.finish_reason;
|
||||
}
|
||||
}
|
||||
const finishReason = parsedChoices?.[0]?.finish_reason;
|
||||
if (finishReason) {
|
||||
sawTerminalSignal = true;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -12,6 +12,8 @@ import {
|
|||
import { MascotImg } from "@/components/mascot-img";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { copyToClipboard } from "@/lib/copy-to-clipboard";
|
||||
import { downloadFile, isDownloadCancelled } from "@/lib/native-files";
|
||||
import { toast } from "@/lib/toast";
|
||||
import { cn } from "@/lib/utils";
|
||||
import { CopyIcon, EyeIcon, Maximize2Icon, XIcon } from "lucide-react";
|
||||
import { Download01Icon } from "@hugeicons/core-free-icons";
|
||||
|
|
@ -91,18 +93,6 @@ function ArtifactGeneratingPanel() {
|
|||
);
|
||||
}
|
||||
|
||||
function downloadTextFile(filename: string, text: string): void {
|
||||
const blob = new Blob([text], { type: "text/html;charset=utf-8" });
|
||||
const url = URL.createObjectURL(blob);
|
||||
const anchor = document.createElement("a");
|
||||
anchor.href = url;
|
||||
anchor.download = filename;
|
||||
document.body.appendChild(anchor);
|
||||
anchor.click();
|
||||
document.body.removeChild(anchor);
|
||||
window.setTimeout(() => URL.revokeObjectURL(url), 0);
|
||||
}
|
||||
|
||||
export function ArtifactSurface({
|
||||
artifact,
|
||||
variant,
|
||||
|
|
@ -205,7 +195,7 @@ export function ArtifactSurface({
|
|||
className={cn(
|
||||
"relative flex min-h-0 flex-col bg-background",
|
||||
variant === "panel"
|
||||
? "artifact-panel-shell mx-2 mt-[72px] mb-8 h-[calc(100%_-_104px)] overflow-visible rounded-[28px] border-t border-border/70 bg-card/95"
|
||||
? "artifact-panel-shell mx-2 mt-[90px] mb-8 h-[calc(100%_-_122px)] overflow-visible rounded-[28px] border-t border-border/70 bg-card/95"
|
||||
: "h-[min(92vh,900px)] w-[min(96vw,1200px)] overflow-hidden rounded-2xl border border-border shadow-xl",
|
||||
)}
|
||||
aria-label={`${artifact.title} canvas`}
|
||||
|
|
@ -265,7 +255,19 @@ export function ArtifactSurface({
|
|||
size="icon"
|
||||
className="size-8"
|
||||
disabled={isLoadingArtifact || !hasArtifactCode}
|
||||
onClick={() => downloadTextFile(filename, artifact.code)}
|
||||
onClick={() => {
|
||||
// Route through the native save dialog on desktop; the plain
|
||||
// blob-anchor download is silently dropped by the Tauri WebView2.
|
||||
void downloadFile(
|
||||
artifact.code,
|
||||
filename,
|
||||
"text/html;charset=utf-8",
|
||||
).catch((err) => {
|
||||
if (!isDownloadCancelled(err)) {
|
||||
toast.error("Failed to save canvas HTML");
|
||||
}
|
||||
});
|
||||
}}
|
||||
aria-label="Download canvas HTML"
|
||||
>
|
||||
<HugeiconsIcon icon={Download01Icon} className="size-4" />
|
||||
|
|
|
|||
|
|
@ -2676,7 +2676,7 @@ export function ChatPage({
|
|||
config: meta?.config,
|
||||
nativePathToken: meta?.nativePathToken,
|
||||
nativePathExpiresAtMs: meta?.nativePathExpiresAtMs,
|
||||
forceReload: isSameLoadedModel || undefined,
|
||||
forceReload: meta?.forceReload ?? (isSameLoadedModel || undefined),
|
||||
};
|
||||
await stageOrLoad(selection);
|
||||
})();
|
||||
|
|
|
|||
|
|
@ -966,7 +966,7 @@ export function ChatSettingsPanel({
|
|||
Delete
|
||||
</Button>
|
||||
</div>
|
||||
<p className="text-[11px] leading-relaxed text-muted-foreground">
|
||||
<p className="text-ui-11 leading-relaxed text-muted-foreground">
|
||||
Saving a preset also stores current load settings (context length,
|
||||
KV cache dtype, speculative decoding, GPU layers).
|
||||
{currentLoadSummary ? (
|
||||
|
|
|
|||
|
|
@ -62,6 +62,7 @@ import {
|
|||
import { isExternalModelId } from "../external-providers";
|
||||
import {
|
||||
applyPerModelConfigToRuntime,
|
||||
normalizeMaxSeqLength,
|
||||
type PerModelConfig,
|
||||
} from "@/features/model-picker";
|
||||
import type {
|
||||
|
|
@ -604,12 +605,19 @@ export function useChatModelRuntime() {
|
|||
async function performLoad(): Promise<void> {
|
||||
if (abortCtrl.signal.aborted) throw new Error("Cancelled");
|
||||
let previousWasUnloaded = false;
|
||||
const pendingLoadConfig =
|
||||
typeof selection !== "string" ? selection.config : undefined;
|
||||
if (pendingLoadConfig) {
|
||||
applyPerModelConfigToRuntime(pendingLoadConfig);
|
||||
}
|
||||
const currentCheckpoint =
|
||||
useChatRuntimeStore.getState().params.checkpoint;
|
||||
const stateBeforeUnload = useChatRuntimeStore.getState();
|
||||
let trustRemoteCode = stateBeforeUnload.params.trustRemoteCode ?? false;
|
||||
let approvedRemoteCodeFingerprint: string | null = null;
|
||||
const maxSeqLength = stateBeforeUnload.params.maxSeqLength;
|
||||
const maxSeqLength =
|
||||
normalizeMaxSeqLength(pendingLoadConfig?.maxSeqLength) ??
|
||||
stateBeforeUnload.params.maxSeqLength;
|
||||
const previousActiveNativePathToken =
|
||||
stateBeforeUnload.activeNativePathToken;
|
||||
const previousIsGguf =
|
||||
|
|
@ -643,34 +651,54 @@ export function useChatModelRuntime() {
|
|||
const previousActiveNativePathExpiresAtMs =
|
||||
stateBeforeUnload.activeNativePathExpiresAtMs;
|
||||
// Snapshot the load settings at click time, before the awaits below
|
||||
// (validation, the trust dialog, unload).
|
||||
const loadChatTemplateOverride = stateBeforeUnload.chatTemplateOverride;
|
||||
const loadKvCacheDtype = stateBeforeUnload.kvCacheDtype;
|
||||
// (validation, the trust dialog, unload). When the picker staged a
|
||||
// config payload, prefer it over the store: React may not have
|
||||
// flushed NumericValueInput's blur commit into state yet.
|
||||
const loadChatTemplateOverride =
|
||||
pendingLoadConfig?.chatTemplateOverride?.trim()
|
||||
? pendingLoadConfig.chatTemplateOverride
|
||||
: stateBeforeUnload.chatTemplateOverride;
|
||||
const loadKvCacheDtype =
|
||||
pendingLoadConfig?.kvCacheDtype ?? stateBeforeUnload.kvCacheDtype;
|
||||
// gpuMemoryMode is a standing preference (kept across a model switch);
|
||||
// the rest are per-model knobs the reset below clears, so they are
|
||||
// re-baselined there in lock-step with the store.
|
||||
let loadCustomContextLength = stateBeforeUnload.customContextLength;
|
||||
let loadCustomContextLength =
|
||||
pendingLoadConfig?.customContextLength ??
|
||||
stateBeforeUnload.customContextLength;
|
||||
const loadGgufContextLength = stateBeforeUnload.ggufContextLength;
|
||||
const loadTensorParallel = stateBeforeUnload.tensorParallel;
|
||||
const loadTensorParallel =
|
||||
pendingLoadConfig?.tensorParallel ?? stateBeforeUnload.tensorParallel;
|
||||
const loadActivePresetSource = stateBeforeUnload.activePresetSource;
|
||||
const loadActiveGgufVariant = stateBeforeUnload.activeGgufVariant;
|
||||
const loadGpuMemoryMode = stateBeforeUnload.gpuMemoryMode;
|
||||
let loadGpuLayers = stateBeforeUnload.gpuLayers;
|
||||
let loadNCpuMoe = stateBeforeUnload.nCpuMoe;
|
||||
const loadGpuMemoryMode =
|
||||
pendingLoadConfig?.gpuMemoryMode ?? stateBeforeUnload.gpuMemoryMode;
|
||||
let loadGpuLayers =
|
||||
pendingLoadConfig?.gpuLayers ?? stateBeforeUnload.gpuLayers;
|
||||
let loadNCpuMoe =
|
||||
pendingLoadConfig?.nCpuMoe ?? stateBeforeUnload.nCpuMoe;
|
||||
let loadSplitRatio = stateBeforeUnload.splitRatio;
|
||||
// Reconcile the persisted pick against the GPUs present now, so a stale
|
||||
// cross-host / now-hidden pick is dropped before /load rather than
|
||||
// rejected there. Warm the device cache first: load-on-selection can
|
||||
// run before any GPU hook mounted, and a cold cache would pass the
|
||||
// pick through unvalidated. validateGpuIds derives from this too.
|
||||
if (stateBeforeUnload.selectedGpuIds != null) {
|
||||
if (
|
||||
pendingLoadConfig?.selectedGpuIds !== undefined ||
|
||||
stateBeforeUnload.selectedGpuIds != null
|
||||
) {
|
||||
await ensureGpuDeviceCache();
|
||||
}
|
||||
let loadSelectedGpuIds = reconcilePersistedGpuIds(
|
||||
stateBeforeUnload.selectedGpuIds,
|
||||
);
|
||||
let loadSpeculativeType = stateBeforeUnload.speculativeType;
|
||||
let loadSpecDraftNMax = stateBeforeUnload.specDraftNMax;
|
||||
let loadSelectedGpuIds =
|
||||
pendingLoadConfig?.selectedGpuIds !== undefined
|
||||
? reconcilePersistedGpuIds(pendingLoadConfig.selectedGpuIds)
|
||||
: reconcilePersistedGpuIds(stateBeforeUnload.selectedGpuIds);
|
||||
let loadSpeculativeType =
|
||||
pendingLoadConfig?.speculativeType != null
|
||||
? normalizeSpeculativeType(pendingLoadConfig.speculativeType)
|
||||
: stateBeforeUnload.speculativeType;
|
||||
let loadSpecDraftNMax =
|
||||
pendingLoadConfig?.specDraftNMax ?? stateBeforeUnload.specDraftNMax;
|
||||
try {
|
||||
// Lightweight pre-flight validation: avoid unloading a working model
|
||||
// if the new identifier is clearly invalid (e.g. bad HF id / path).
|
||||
|
|
@ -810,15 +838,23 @@ export function useChatModelRuntime() {
|
|||
// model loads at Auto/native, not the previous model's pin.
|
||||
customContextLength: null,
|
||||
});
|
||||
loadSpeculativeType = persistedSpeculativeType;
|
||||
loadSpecDraftNMax = null;
|
||||
loadSpeculativeType =
|
||||
pendingLoadConfig?.speculativeType != null
|
||||
? normalizeSpeculativeType(pendingLoadConfig.speculativeType)
|
||||
: persistedSpeculativeType;
|
||||
loadSpecDraftNMax = pendingLoadConfig?.specDraftNMax ?? null;
|
||||
// Keep the click-time snapshot in lock-step with the store reset so
|
||||
// the load below sizes against the cleared per-model knobs, not the
|
||||
// previous model's (gpuMemoryMode is standing, so left as captured).
|
||||
loadCustomContextLength = null;
|
||||
loadSelectedGpuIds = null;
|
||||
loadGpuLayers = GPU_LAYERS_AUTO;
|
||||
loadNCpuMoe = 0;
|
||||
// An explicit staged config from run-settings still wins.
|
||||
loadCustomContextLength =
|
||||
pendingLoadConfig?.customContextLength ?? null;
|
||||
loadSelectedGpuIds =
|
||||
pendingLoadConfig?.selectedGpuIds !== undefined
|
||||
? reconcilePersistedGpuIds(pendingLoadConfig.selectedGpuIds)
|
||||
: null;
|
||||
loadGpuLayers = pendingLoadConfig?.gpuLayers ?? GPU_LAYERS_AUTO;
|
||||
loadNCpuMoe = pendingLoadConfig?.nCpuMoe ?? 0;
|
||||
loadSplitRatio = null;
|
||||
}
|
||||
|
||||
|
|
@ -1271,12 +1307,19 @@ export function useChatModelRuntime() {
|
|||
prog.expected_bytes,
|
||||
dlSamples,
|
||||
);
|
||||
setLoadProgress({
|
||||
percent: pct,
|
||||
label: progressLabel,
|
||||
phase: "downloading",
|
||||
});
|
||||
if (loadToastDismissedRef.current) return;
|
||||
// loadProgress state is only read by the dismissed-toast inline
|
||||
// status. Writing it while the toast is visible re-renders the
|
||||
// whole chat page every poll — cheap in Chrome, janky in the
|
||||
// desktop WebView2 (laggy typing). Feed the toast directly and
|
||||
// only touch state when the inline view is actually live.
|
||||
if (loadToastDismissedRef.current) {
|
||||
setLoadProgress({
|
||||
percent: pct,
|
||||
label: progressLabel,
|
||||
phase: "downloading",
|
||||
});
|
||||
return;
|
||||
}
|
||||
toast(null, {
|
||||
id: toastId,
|
||||
...modelLoadToastOptions(
|
||||
|
|
@ -1298,19 +1341,23 @@ export function useChatModelRuntime() {
|
|||
const est = estimate(dlSamples, prog.downloaded_bytes, 0);
|
||||
const rateSuffix =
|
||||
est.stable ? ` • ${formatRate(est.rate)}` : "";
|
||||
setLoadProgress({
|
||||
percent: null,
|
||||
label: `${dlGb.toFixed(1)} GB downloaded${rateSuffix}`,
|
||||
phase: "downloading",
|
||||
});
|
||||
// Inline-status-only state; skip the chat-page re-render unless it's shown.
|
||||
if (loadToastDismissedRef.current) {
|
||||
setLoadProgress({
|
||||
percent: null,
|
||||
label: `${dlGb.toFixed(1)} GB downloaded${rateSuffix}`,
|
||||
phase: "downloading",
|
||||
});
|
||||
}
|
||||
} else if (prog.progress >= 1 && hasShownProgress) {
|
||||
downloadComplete = true;
|
||||
setLoadProgress({
|
||||
percent: 100,
|
||||
label: "Download complete",
|
||||
phase: "starting",
|
||||
});
|
||||
if (!loadToastDismissedRef.current) {
|
||||
if (loadToastDismissedRef.current) {
|
||||
setLoadProgress({
|
||||
percent: 100,
|
||||
label: "Download complete",
|
||||
phase: "starting",
|
||||
});
|
||||
} else {
|
||||
toast(null, {
|
||||
id: toastId,
|
||||
...modelLoadToastOptions(
|
||||
|
|
@ -1364,12 +1411,17 @@ export function useChatModelRuntime() {
|
|||
formatEta(est.eta) !== "--" ? ` • ${formatEta(est.eta)} left` : ""
|
||||
}`
|
||||
: base;
|
||||
setLoadProgress({
|
||||
percent: pct,
|
||||
label,
|
||||
phase: "starting",
|
||||
});
|
||||
if (loadToastDismissedRef.current) return;
|
||||
// Inline-status-only state (see pollDownload): while the toast is
|
||||
// up, skip the state write so the chat page doesn't re-render every
|
||||
// poll during "Starting model" — the desktop WebView2 typing-lag fix.
|
||||
if (loadToastDismissedRef.current) {
|
||||
setLoadProgress({
|
||||
percent: pct,
|
||||
label,
|
||||
phase: "starting",
|
||||
});
|
||||
return;
|
||||
}
|
||||
toast(null, {
|
||||
id: toastId,
|
||||
...modelLoadToastOptions(
|
||||
|
|
|
|||
|
|
@ -24,7 +24,14 @@ import { ChevronDownStandardIcon } from "@/lib/chevron-icons";
|
|||
import { toast } from "@/lib/toast";
|
||||
import { ArrowLeft01Icon } from "@hugeicons/core-free-icons";
|
||||
import { HugeiconsIcon } from "@hugeicons/react";
|
||||
import { type ReactNode, useEffect, useId, useState } from "react";
|
||||
import {
|
||||
type ReactNode,
|
||||
type Ref,
|
||||
useEffect,
|
||||
useId,
|
||||
useRef,
|
||||
useState,
|
||||
} from "react";
|
||||
import {
|
||||
useDefaultChatTemplate,
|
||||
useModelMaxPositionEmbeddings,
|
||||
|
|
@ -50,7 +57,10 @@ import {
|
|||
} from "../model-config/per-model-config";
|
||||
import { ChatTemplateEditorDialog } from "./chat-template-editor-dialog";
|
||||
import type { ModelPickTarget } from "./model-selector/types";
|
||||
import { NumericValueInput } from "./numeric-value-input";
|
||||
import {
|
||||
NumericValueInput,
|
||||
type NumericValueInputHandle,
|
||||
} from "./numeric-value-input";
|
||||
|
||||
const ROW_CLASS = "flex min-h-8 items-center justify-between gap-3";
|
||||
const LABEL_CLASS =
|
||||
|
|
@ -130,11 +140,13 @@ function MaxSeqLengthSetting({
|
|||
max,
|
||||
inputMax,
|
||||
onChange,
|
||||
inputRef,
|
||||
}: {
|
||||
value: number;
|
||||
max: number;
|
||||
inputMax: number;
|
||||
onChange: (value: number) => void;
|
||||
inputRef?: Ref<NumericValueInputHandle>;
|
||||
}) {
|
||||
return (
|
||||
<div className="space-y-3">
|
||||
|
|
@ -146,6 +158,7 @@ function MaxSeqLengthSetting({
|
|||
</InfoHint>
|
||||
</div>
|
||||
<NumericValueInput
|
||||
ref={inputRef}
|
||||
value={value}
|
||||
min={MAX_SEQ_LENGTH_MIN}
|
||||
max={inputMax}
|
||||
|
|
@ -182,6 +195,7 @@ function AdvancedGpuSlider({
|
|||
onChange,
|
||||
displayValue,
|
||||
info,
|
||||
inputRef,
|
||||
}: {
|
||||
label: string;
|
||||
value: number;
|
||||
|
|
@ -190,6 +204,7 @@ function AdvancedGpuSlider({
|
|||
onChange: (value: number) => void;
|
||||
displayValue?: string;
|
||||
info?: ReactNode;
|
||||
inputRef?: Ref<NumericValueInputHandle>;
|
||||
}) {
|
||||
return (
|
||||
<div className="space-y-3">
|
||||
|
|
@ -199,6 +214,7 @@ function AdvancedGpuSlider({
|
|||
{info && <InfoHint>{info}</InfoHint>}
|
||||
</div>
|
||||
<NumericValueInput
|
||||
ref={inputRef}
|
||||
value={value}
|
||||
min={min}
|
||||
max={max}
|
||||
|
|
@ -231,11 +247,15 @@ function GpuMemorySettings({
|
|||
update,
|
||||
layerCount,
|
||||
moeLayerCount,
|
||||
gpuLayersInputRef,
|
||||
moeLayersInputRef,
|
||||
}: {
|
||||
config: PerModelConfig;
|
||||
update: (patch: Partial<PerModelConfig>) => void;
|
||||
layerCount: number | null;
|
||||
moeLayerCount: number | null;
|
||||
gpuLayersInputRef?: Ref<NumericValueInputHandle>;
|
||||
moeLayersInputRef?: Ref<NumericValueInputHandle>;
|
||||
}) {
|
||||
const gpuDevices = useGpuDevices();
|
||||
const mode = config.gpuMemoryMode ?? "auto";
|
||||
|
|
@ -322,6 +342,7 @@ function GpuMemorySettings({
|
|||
<>
|
||||
<AdvancedGpuSlider
|
||||
label="GPU Layers"
|
||||
inputRef={gpuLayersInputRef}
|
||||
value={Math.max(GPU_LAYERS_AUTO, Math.min(gpuLayers, gpuLayersMax))}
|
||||
min={GPU_LAYERS_AUTO}
|
||||
max={gpuLayersMax}
|
||||
|
|
@ -338,6 +359,7 @@ function GpuMemorySettings({
|
|||
{showMoeSlider && (
|
||||
<AdvancedGpuSlider
|
||||
label="MoE Layers on CPU"
|
||||
inputRef={moeLayersInputRef}
|
||||
value={Math.min(nCpuMoe, moeLayersMax)}
|
||||
min={0}
|
||||
max={moeLayersMax}
|
||||
|
|
@ -399,6 +421,8 @@ function GgufAdvancedSettings({
|
|||
onEditTemplate,
|
||||
layerCount,
|
||||
moeLayerCount,
|
||||
gpuLayersInputRef,
|
||||
moeLayersInputRef,
|
||||
}: {
|
||||
config: PerModelConfig;
|
||||
update: (patch: Partial<PerModelConfig>) => void;
|
||||
|
|
@ -407,6 +431,8 @@ function GgufAdvancedSettings({
|
|||
onEditTemplate: () => void;
|
||||
layerCount: number | null;
|
||||
moeLayerCount: number | null;
|
||||
gpuLayersInputRef?: Ref<NumericValueInputHandle>;
|
||||
moeLayersInputRef?: Ref<NumericValueInputHandle>;
|
||||
}) {
|
||||
return (
|
||||
<>
|
||||
|
|
@ -535,6 +561,8 @@ function GgufAdvancedSettings({
|
|||
update={update}
|
||||
layerCount={layerCount}
|
||||
moeLayerCount={moeLayerCount}
|
||||
gpuLayersInputRef={gpuLayersInputRef}
|
||||
moeLayersInputRef={moeLayersInputRef}
|
||||
/>
|
||||
|
||||
<ChatTemplateSetting config={config} onEditTemplate={onEditTemplate} />
|
||||
|
|
@ -597,6 +625,10 @@ export function ModelConfigPage({
|
|||
const [showAdvanced, setShowAdvanced] = useState(() =>
|
||||
hasNonDefaultAdvanced(config),
|
||||
);
|
||||
const contextInputRef = useRef<NumericValueInputHandle>(null);
|
||||
const maxSeqLengthInputRef = useRef<NumericValueInputHandle>(null);
|
||||
const gpuLayersInputRef = useRef<NumericValueInputHandle>(null);
|
||||
const moeLayersInputRef = useRef<NumericValueInputHandle>(null);
|
||||
const nativePathToken =
|
||||
target.meta.nativePathToken ??
|
||||
(isActiveModel ? activeNativePathToken : null);
|
||||
|
|
@ -744,11 +776,6 @@ export function ModelConfigPage({
|
|||
? { ...config, customContextLength: activeLoadedContext }
|
||||
: config
|
||||
: config;
|
||||
// Load request needs a concrete max length; substitute the fallback here only,
|
||||
// never in the persisted runtimeConfig.
|
||||
const loadConfig = target.isGguf
|
||||
? runtimeConfig
|
||||
: { ...runtimeConfig, maxSeqLength: maxSeqLengthValue };
|
||||
const rememberChanged = remember !== savedRemember;
|
||||
const persistenceOnly = isActiveModel && atBaseline && rememberChanged;
|
||||
const primaryActionLabel = persistenceOnly
|
||||
|
|
@ -760,18 +787,90 @@ export function ModelConfigPage({
|
|||
: "Load model";
|
||||
|
||||
const handleRun = () => {
|
||||
const defaultConfig = isDefaultConfig(runtimeConfig);
|
||||
// Same-click Load/Reload: a numeric draft the user just typed is flushed only
|
||||
// by that input's blur handler, which updates the parent config after this
|
||||
// click closure already captured the stale value. Commit every numeric input
|
||||
// imperatively so the staged load honors what the user just typed, not just
|
||||
// the Context field.
|
||||
const committedContext = target.isGguf
|
||||
? contextInputRef.current?.commit()
|
||||
: undefined;
|
||||
const committedMaxSeqLength = target.isGguf
|
||||
? undefined
|
||||
: maxSeqLengthInputRef.current?.commit();
|
||||
const committedGpuLayers = target.isGguf
|
||||
? gpuLayersInputRef.current?.commit()
|
||||
: undefined;
|
||||
const committedMoeLayers = target.isGguf
|
||||
? moeLayersInputRef.current?.commit()
|
||||
: undefined;
|
||||
|
||||
const pendingPatch: Partial<PerModelConfig> = {};
|
||||
if (committedContext != null) {
|
||||
pendingPatch.customContextLength = committedContext;
|
||||
}
|
||||
if (committedMaxSeqLength != null) {
|
||||
pendingPatch.maxSeqLength = clampMaxSeqLength(
|
||||
committedMaxSeqLength,
|
||||
MAX_SEQ_LENGTH_MAX,
|
||||
);
|
||||
}
|
||||
if (committedGpuLayers != null) {
|
||||
pendingPatch.gpuLayers = committedGpuLayers;
|
||||
}
|
||||
if (committedMoeLayers != null) {
|
||||
pendingPatch.nCpuMoe = committedMoeLayers;
|
||||
}
|
||||
const hasPending =
|
||||
committedContext != null ||
|
||||
committedMaxSeqLength != null ||
|
||||
committedGpuLayers != null ||
|
||||
committedMoeLayers != null;
|
||||
|
||||
const effectiveConfig = hasPending
|
||||
? { ...config, ...pendingPatch }
|
||||
: config;
|
||||
// pinFixedLayerContext above was computed from the render-time config, before
|
||||
// the same-click GPU Layers draft was committed. Recompute it from
|
||||
// effectiveConfig so committing a positive fixed-layer value still pins the
|
||||
// fitted context; otherwise the saved config carries customContextLength: null
|
||||
// and a later fresh load sends the native context with fixed layers (the OOM
|
||||
// the pin exists to avoid).
|
||||
const effectivePinFixedLayerContext =
|
||||
target.isGguf &&
|
||||
effectiveConfig.gpuMemoryMode === "manual" &&
|
||||
effectiveConfig.gpuLayers != null &&
|
||||
effectiveConfig.gpuLayers >= 0 &&
|
||||
effectiveConfig.customContextLength == null &&
|
||||
activeLoadedContext != null;
|
||||
const effectiveRuntimeConfig = hasPending
|
||||
? effectivePinFixedLayerContext
|
||||
? { ...effectiveConfig, customContextLength: activeLoadedContext }
|
||||
: effectiveConfig
|
||||
: runtimeConfig;
|
||||
// Non-GGUF load substitutes the resolved max sequence length; recompute it
|
||||
// from the committed draft so a same-click Max Seq Length edit is not lost.
|
||||
const effectiveMaxSeqLengthValue =
|
||||
committedMaxSeqLength == null
|
||||
? maxSeqLengthValue
|
||||
: (normalizeMaxSeqLength(effectiveConfig.maxSeqLength) ??
|
||||
clampMaxSeqLength(DEFAULT_MAX_SEQ_LENGTH, nativeMaxSeqLength));
|
||||
// Recheck the committed draft so Save/Forget reloads when needed.
|
||||
const effectiveAtBaseline = perModelConfigsEqual(effectiveConfig, baseline);
|
||||
const effectivePersistenceOnly =
|
||||
isActiveModel && effectiveAtBaseline && rememberChanged;
|
||||
const defaultConfig = isDefaultConfig(effectiveRuntimeConfig);
|
||||
let saveFailed = false;
|
||||
if (remember) {
|
||||
saveFailed = !savePerModelConfig(
|
||||
target.id,
|
||||
target.ggufVariant,
|
||||
runtimeConfig,
|
||||
effectiveRuntimeConfig,
|
||||
);
|
||||
} else {
|
||||
saveFailed = !deletePerModelConfig(target.id, target.ggufVariant);
|
||||
}
|
||||
if (persistenceOnly) {
|
||||
if (effectivePersistenceOnly) {
|
||||
if (saveFailed) {
|
||||
toast.error("Couldn't save settings for this model.");
|
||||
return;
|
||||
|
|
@ -791,7 +890,10 @@ export function ModelConfigPage({
|
|||
if (saveFailed) {
|
||||
toast.error("Couldn't save these settings, loading with them anyway.");
|
||||
}
|
||||
onRun(loadConfig);
|
||||
const effectiveLoadConfig = target.isGguf
|
||||
? effectiveRuntimeConfig
|
||||
: { ...effectiveRuntimeConfig, maxSeqLength: effectiveMaxSeqLengthValue };
|
||||
onRun(effectiveLoadConfig);
|
||||
};
|
||||
|
||||
return (
|
||||
|
|
@ -838,6 +940,7 @@ export function ModelConfigPage({
|
|||
</InfoHint>
|
||||
</div>
|
||||
<NumericValueInput
|
||||
ref={contextInputRef}
|
||||
value={contextValue}
|
||||
min={minContext}
|
||||
max={maxContext}
|
||||
|
|
@ -886,6 +989,8 @@ export function ModelConfigPage({
|
|||
onEditTemplate={() => setTemplateOpen(true)}
|
||||
layerCount={stagedDims?.layerCount ?? null}
|
||||
moeLayerCount={stagedDims?.moeLayerCount ?? null}
|
||||
gpuLayersInputRef={gpuLayersInputRef}
|
||||
moeLayersInputRef={moeLayersInputRef}
|
||||
/>
|
||||
)}
|
||||
|
||||
|
|
@ -914,6 +1019,7 @@ export function ModelConfigPage({
|
|||
value={maxSeqLengthValue}
|
||||
max={maxSeqLengthMax}
|
||||
inputMax={MAX_SEQ_LENGTH_MAX}
|
||||
inputRef={maxSeqLengthInputRef}
|
||||
onChange={(value) =>
|
||||
update({
|
||||
maxSeqLength: clampMaxSeqLength(value, MAX_SEQ_LENGTH_MAX),
|
||||
|
|
|
|||
|
|
@ -2,7 +2,13 @@
|
|||
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
||||
|
||||
import { cn } from "@/lib/utils";
|
||||
import { useRef, useState } from "react";
|
||||
import {
|
||||
forwardRef,
|
||||
useEffect,
|
||||
useImperativeHandle,
|
||||
useRef,
|
||||
useState,
|
||||
} from "react";
|
||||
|
||||
export function snapToStep(
|
||||
value: number,
|
||||
|
|
@ -28,44 +34,109 @@ function sanitizeNumeric(raw: string, allowNegative: boolean): string {
|
|||
return `${sign}${head}${tail}`;
|
||||
}
|
||||
|
||||
export function NumericValueInput({
|
||||
value,
|
||||
min,
|
||||
max,
|
||||
step,
|
||||
onChange,
|
||||
displayValue,
|
||||
className,
|
||||
ariaLabel,
|
||||
size: sizeAttr,
|
||||
disabled = false,
|
||||
}: {
|
||||
value: number;
|
||||
min?: number;
|
||||
max?: number;
|
||||
step: number;
|
||||
onChange: (v: number) => void;
|
||||
displayValue?: string;
|
||||
className?: string;
|
||||
ariaLabel?: string;
|
||||
size?: number;
|
||||
disabled?: boolean;
|
||||
}) {
|
||||
export type NumericValueInputHandle = {
|
||||
/** Commit a valid focused/same-click draft; null when none is pending. */
|
||||
commit: () => number | null;
|
||||
};
|
||||
|
||||
export const NumericValueInput = forwardRef<
|
||||
NumericValueInputHandle,
|
||||
{
|
||||
value: number;
|
||||
min?: number;
|
||||
max?: number;
|
||||
step: number;
|
||||
onChange: (v: number) => void;
|
||||
displayValue?: string;
|
||||
className?: string;
|
||||
ariaLabel?: string;
|
||||
size?: number;
|
||||
disabled?: boolean;
|
||||
}
|
||||
>(function NumericValueInput(
|
||||
{
|
||||
value,
|
||||
min,
|
||||
max,
|
||||
step,
|
||||
onChange,
|
||||
displayValue,
|
||||
className,
|
||||
ariaLabel,
|
||||
size: sizeAttr,
|
||||
disabled = false,
|
||||
},
|
||||
ref,
|
||||
) {
|
||||
const [focused, setFocused] = useState(false);
|
||||
const [draft, setDraft] = useState("");
|
||||
const cancelBlurCommitRef = useRef(false);
|
||||
const draftRef = useRef("");
|
||||
const dirtyRef = useRef(false);
|
||||
// Same-click Load: blur commits via onChange and clears dirtyRef before the
|
||||
// button onClick runs, while parent `value` is still stale. Keep the blur
|
||||
// result for one imperative commit(); clear when `value` catches up or on
|
||||
// focus / external edits (Reset, slider).
|
||||
const lastBlurCommittedRef = useRef<number | null>(null);
|
||||
|
||||
const commit = (raw: string) => {
|
||||
// The blur bridge is only valid across the single synchronous gesture that set
|
||||
// it: blur commits during a button's mousedown and that button's onClick
|
||||
// consumes it via commit() before React re-renders. Any settled render means the
|
||||
// gesture is over, so drop the cache on every commit. Keying this on [value]
|
||||
// alone missed a Reset (or other external edit) that restores the shown value
|
||||
// unchanged when the blur did dispatch onChange (final !== value): value nets
|
||||
// back to its prior number, so the effect never re-ran, the stale pin survived,
|
||||
// and the next Load/Save replayed the override Reset had removed.
|
||||
useEffect(() => {
|
||||
lastBlurCommittedRef.current = null;
|
||||
});
|
||||
|
||||
const commitDraft = (raw: string): number | null => {
|
||||
const parsed = Number.parseFloat(raw);
|
||||
if (!Number.isFinite(parsed)) {
|
||||
return;
|
||||
return null;
|
||||
}
|
||||
const final = snapToStep(parsed, step, min, max);
|
||||
if (final !== value) {
|
||||
onChange(final);
|
||||
}
|
||||
return final;
|
||||
};
|
||||
|
||||
useImperativeHandle(
|
||||
ref,
|
||||
() => ({
|
||||
commit: () => {
|
||||
if (dirtyRef.current) {
|
||||
const raw = draftRef.current;
|
||||
const final = commitDraft(raw);
|
||||
dirtyRef.current = false;
|
||||
lastBlurCommittedRef.current = null;
|
||||
if (final == null) {
|
||||
draftRef.current = String(value);
|
||||
}
|
||||
if (focused) {
|
||||
setFocused(false);
|
||||
}
|
||||
return final;
|
||||
}
|
||||
const blurCommitted = lastBlurCommittedRef.current;
|
||||
if (blurCommitted != null) {
|
||||
lastBlurCommittedRef.current = null;
|
||||
if (focused) {
|
||||
setFocused(false);
|
||||
}
|
||||
return blurCommitted;
|
||||
}
|
||||
if (focused) {
|
||||
setFocused(false);
|
||||
}
|
||||
return null;
|
||||
},
|
||||
}),
|
||||
[draft, focused, max, min, onChange, step, value],
|
||||
);
|
||||
|
||||
const displayed = focused ? draft : (displayValue ?? String(value));
|
||||
|
||||
return (
|
||||
|
|
@ -82,7 +153,11 @@ export function NumericValueInput({
|
|||
aria-label={ariaLabel}
|
||||
onFocus={(e) => {
|
||||
cancelBlurCommitRef.current = false;
|
||||
setDraft(String(value));
|
||||
dirtyRef.current = false;
|
||||
lastBlurCommittedRef.current = null;
|
||||
const next = String(value);
|
||||
draftRef.current = next;
|
||||
setDraft(next);
|
||||
setFocused(true);
|
||||
const target = e.currentTarget;
|
||||
requestAnimationFrame(() => target.select());
|
||||
|
|
@ -90,24 +165,47 @@ export function NumericValueInput({
|
|||
onBlur={() => {
|
||||
if (cancelBlurCommitRef.current) {
|
||||
cancelBlurCommitRef.current = false;
|
||||
} else {
|
||||
commit(draft);
|
||||
lastBlurCommittedRef.current = null;
|
||||
} else if (dirtyRef.current) {
|
||||
const final = commitDraft(draftRef.current);
|
||||
dirtyRef.current = false;
|
||||
if (final == null) {
|
||||
draftRef.current = String(value);
|
||||
lastBlurCommittedRef.current = null;
|
||||
} else {
|
||||
draftRef.current = String(final);
|
||||
// Only bridge the still-stale parent value when the blur actually
|
||||
// dispatched onChange (final !== value). When final === value the
|
||||
// parent is already current, so there is nothing to bridge; caching
|
||||
// here would leave a stale pin that a later Reset or external edit
|
||||
// (which doesn't change the displayed value) can never clear, so a
|
||||
// following Load/Save would recreate the override Reset removed.
|
||||
lastBlurCommittedRef.current = final !== value ? final : null;
|
||||
}
|
||||
}
|
||||
setFocused(false);
|
||||
}}
|
||||
onChange={(e) =>
|
||||
setDraft(sanitizeNumeric(e.target.value, (min ?? 0) < 0))
|
||||
}
|
||||
onChange={(e) => {
|
||||
dirtyRef.current = true;
|
||||
lastBlurCommittedRef.current = null;
|
||||
const next = sanitizeNumeric(e.target.value, (min ?? 0) < 0);
|
||||
draftRef.current = next;
|
||||
setDraft(next);
|
||||
}}
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === "Enter") {
|
||||
e.currentTarget.blur();
|
||||
} else if (e.key === "Escape") {
|
||||
cancelBlurCommitRef.current = true;
|
||||
setDraft(String(value));
|
||||
dirtyRef.current = false;
|
||||
lastBlurCommittedRef.current = null;
|
||||
const next = String(value);
|
||||
draftRef.current = next;
|
||||
setDraft(next);
|
||||
e.currentTarget.blur();
|
||||
}
|
||||
}}
|
||||
className={cn(className)}
|
||||
/>
|
||||
);
|
||||
}
|
||||
});
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ export {
|
|||
export { hfModelFitsDevice } from "./components/model-selector/recommended-fit";
|
||||
export {
|
||||
NumericValueInput,
|
||||
type NumericValueInputHandle,
|
||||
snapToStep,
|
||||
} from "./components/numeric-value-input";
|
||||
export { SidebarModelConfig } from "./components/sidebar-model-config";
|
||||
|
|
|
|||
|
|
@ -1,10 +1,18 @@
|
|||
// SPDX-License-Identifier: AGPL-3.0-only
|
||||
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
||||
|
||||
import { type ReactElement, useCallback } from "react";
|
||||
import { Lock, LockOpen, Maximize2, Minus, Plus } from "lucide-react";
|
||||
import { Panel, useReactFlow } from "@xyflow/react";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { Panel, useReactFlow } from "@xyflow/react";
|
||||
import {
|
||||
Focus,
|
||||
Lock,
|
||||
LockOpen,
|
||||
Maximize2,
|
||||
Minimize2,
|
||||
Minus,
|
||||
Plus,
|
||||
} from "lucide-react";
|
||||
import { type ReactElement, useCallback } from "react";
|
||||
import { buildFitViewOptions } from "../../utils/graph/fit-view";
|
||||
import { RECIPE_FLOATING_ICON_BUTTON_CLASS } from "../recipe-floating-icon-button-class";
|
||||
|
||||
|
|
@ -12,12 +20,16 @@ type ViewportControlsProps = {
|
|||
interactive: boolean;
|
||||
lockDisabled?: boolean;
|
||||
onToggleInteractive: () => void;
|
||||
maximized: boolean;
|
||||
onToggleMaximize: () => void;
|
||||
};
|
||||
|
||||
export function ViewportControls({
|
||||
interactive,
|
||||
lockDisabled = false,
|
||||
onToggleInteractive,
|
||||
maximized,
|
||||
onToggleMaximize,
|
||||
}: ViewportControlsProps): ReactElement {
|
||||
const { zoomIn, zoomOut, fitView, getNodes } = useReactFlow();
|
||||
|
||||
|
|
@ -61,9 +73,23 @@ export function ViewportControls({
|
|||
size="icon"
|
||||
className={RECIPE_FLOATING_ICON_BUTTON_CLASS}
|
||||
onClick={handleFitView}
|
||||
aria-label="Fit view"
|
||||
aria-label="Center view"
|
||||
>
|
||||
<Maximize2 className="size-4" />
|
||||
<Focus className="size-4" />
|
||||
</Button>
|
||||
<Button
|
||||
type="button"
|
||||
variant="ghost"
|
||||
size="icon"
|
||||
className={RECIPE_FLOATING_ICON_BUTTON_CLASS}
|
||||
onClick={onToggleMaximize}
|
||||
aria-label={maximized ? "Exit full view" : "Expand to full view"}
|
||||
>
|
||||
{maximized ? (
|
||||
<Minimize2 className="size-4" />
|
||||
) : (
|
||||
<Maximize2 className="size-4" />
|
||||
)}
|
||||
</Button>
|
||||
<Button
|
||||
type="button"
|
||||
|
|
@ -74,7 +100,11 @@ export function ViewportControls({
|
|||
onClick={onToggleInteractive}
|
||||
aria-label={interactive ? "Lock interaction" : "Unlock interaction"}
|
||||
>
|
||||
{interactive ? <LockOpen className="size-4" /> : <Lock className="size-4" />}
|
||||
{interactive ? (
|
||||
<LockOpen className="size-4" />
|
||||
) : (
|
||||
<Lock className="size-4" />
|
||||
)}
|
||||
</Button>
|
||||
</Panel>
|
||||
);
|
||||
|
|
|
|||
|
|
@ -237,6 +237,7 @@ export function RecipeStudioPage({
|
|||
}, [setActiveView]);
|
||||
const [processorsOpen, setProcessorsOpen] = useState(false);
|
||||
const [interactive, setInteractive] = useState(true);
|
||||
const [maximized, setMaximized] = useState(false);
|
||||
const [runtimeIslandMinimized, setRuntimeIslandMinimized] = useState(false);
|
||||
const [recentCompletedExecution, setRecentCompletedExecution] =
|
||||
useState<RecipeExecutionRecord | null>(null);
|
||||
|
|
@ -569,6 +570,16 @@ export function RecipeStudioPage({
|
|||
[reactFlowInstance],
|
||||
);
|
||||
|
||||
const toggleMaximize = useCallback(() => {
|
||||
// The maximized surface is a fixed z-50 overlay that already covers the
|
||||
// app sidebar (z-10/z-20), so we don't touch the sidebar's own state — that
|
||||
// state is persisted in pin mode and mutating it here would leak the
|
||||
// temporary collapse into the next page/session.
|
||||
setMaximized((prev) => !prev);
|
||||
// Container size changes; refit once the layout settles.
|
||||
scheduleFitView({ delayMs: TAB_SWITCH_FIT_DELAY_MS });
|
||||
}, [scheduleFitView]);
|
||||
|
||||
useEffect(() => {
|
||||
if (
|
||||
previousActiveViewRef.current !== activeView &&
|
||||
|
|
@ -587,6 +598,15 @@ export function RecipeStudioPage({
|
|||
}
|
||||
}, [activeView, reactFlowInstance]);
|
||||
|
||||
// The "Exit full view" control lives inside the editor canvas, which unmounts
|
||||
// on other tabs. Drop full-view mode when leaving the editor so Easy/Runs
|
||||
// aren't left under the fixed overlay.
|
||||
useEffect(() => {
|
||||
if (activeView !== "editor" && maximized) {
|
||||
setMaximized(false);
|
||||
}
|
||||
}, [activeView, maximized]);
|
||||
|
||||
useEffect(() => {
|
||||
if (
|
||||
!reactFlowInstance ||
|
||||
|
|
@ -732,6 +752,8 @@ export function RecipeStudioPage({
|
|||
interactive={canvasInteractive}
|
||||
lockDisabled={executionLocked}
|
||||
onToggleInteractive={toggleInteractive}
|
||||
maximized={maximized}
|
||||
onToggleMaximize={toggleMaximize}
|
||||
/>
|
||||
{islandExecution &&
|
||||
(isExecutionInProgress(islandExecution.status) ||
|
||||
|
|
@ -773,10 +795,25 @@ export function RecipeStudioPage({
|
|||
}
|
||||
|
||||
return (
|
||||
<div className="min-h-[calc(100dvh-var(--studio-titlebar-height,0px))] bg-background">
|
||||
<main className="w-full px-6 py-8">
|
||||
<div
|
||||
className={
|
||||
maximized
|
||||
? "fixed inset-x-0 bottom-0 z-50 flex flex-col bg-background"
|
||||
: "flex h-full min-h-0 flex-1 flex-col bg-background"
|
||||
}
|
||||
style={
|
||||
maximized
|
||||
? {
|
||||
// Start below the custom/mac window titlebar so the header and
|
||||
// its controls aren't hidden under (or click-blocked by) it.
|
||||
top: "var(--studio-non-chat-content-top-inset, var(--studio-content-top-inset, 0px))",
|
||||
}
|
||||
: undefined
|
||||
}
|
||||
>
|
||||
<main className="flex min-h-0 w-full flex-1 flex-col">
|
||||
<div
|
||||
className="relative w-full overflow-hidden rounded-2xl corner-squircle border"
|
||||
className="relative flex min-h-0 w-full flex-1 flex-col overflow-hidden border"
|
||||
ref={setSheetContainer}
|
||||
>
|
||||
<RecipeStudioHeader
|
||||
|
|
@ -794,7 +831,7 @@ export function RecipeStudioPage({
|
|||
}}
|
||||
/>
|
||||
<div
|
||||
className="h-[75vh] w-full rounded-t-none"
|
||||
className="flex min-h-0 w-full flex-1 rounded-t-none"
|
||||
ref={flowContainerRef}
|
||||
>
|
||||
{activeView === "easy" ? (
|
||||
|
|
|
|||
|
|
@ -282,8 +282,13 @@
|
|||
/* Standard interactive-icon size for nav, menus, action bars, and
|
||||
in-message code-block actions. Sized one step above body text so
|
||||
icons read as minimally larger than adjacent labels (~14px text).
|
||||
Follows the UI font size preference; 18px at the default.
|
||||
Theme-independent — declared once in :root. */
|
||||
--icon-size: 18px;
|
||||
/* Standard icon size follows the UI font size itself: matches it below
|
||||
the 16px default, grows at half the change above it (setting 20 ->
|
||||
18px), so icons read slightly smaller than enlarged text. */
|
||||
--ui-icon-size: min(calc(1rem * var(--ui-font-scale, 1)), calc(0.5rem + 0.5rem * var(--ui-font-scale, 1)));
|
||||
--icon-size: var(--ui-icon-size);
|
||||
/* Inset of a centered .size-icon glyph within a 2rem (size-8) action
|
||||
button — i.e. (32px − icon-size) / 2. Use as a negative margin on a
|
||||
chat-message action bar so the leftmost icon's visual edge aligns
|
||||
|
|
@ -1265,8 +1270,8 @@ html[data-chat-font] .aui-root {
|
|||
}
|
||||
.app-user-menu [data-slot="dropdown-menu-item"] svg,
|
||||
.app-user-menu [data-slot="dropdown-menu-sub-trigger"] svg {
|
||||
width: 19px !important;
|
||||
height: 19px !important;
|
||||
width: var(--ui-icon-size) !important;
|
||||
height: var(--ui-icon-size) !important;
|
||||
flex-shrink: 0;
|
||||
}
|
||||
.app-user-menu [data-slot="dropdown-menu-item"]:focus,
|
||||
|
|
@ -1561,20 +1566,20 @@ html[data-chat-font] .aui-root {
|
|||
/* Fixed-width icon slot so every pill's icon occupies the same space and
|
||||
the labels line up on an even rhythm, regardless of icon size. */
|
||||
.composer-pill-glyph {
|
||||
@apply relative inline-flex w-[19px] shrink-0 items-center justify-center transition-opacity;
|
||||
@apply relative inline-flex w-[var(--ui-icon-size)] shrink-0 items-center justify-center transition-opacity;
|
||||
}
|
||||
|
||||
/* On hover the icon swaps for an X inside a soft circle (ChatGPT-style),
|
||||
filling the icon slot so every pill's X is identical and centered. */
|
||||
.composer-pill-x {
|
||||
@apply pointer-events-none absolute inset-0 m-auto size-[19px] rounded-full bg-primary/15 p-[3px] opacity-0 transition-opacity dark:bg-white/[0.14];
|
||||
@apply pointer-events-none absolute inset-0 m-auto size-[var(--ui-icon-size)] rounded-full bg-primary/15 p-[3px] opacity-0 transition-opacity dark:bg-white/[0.14];
|
||||
}
|
||||
|
||||
/* Icon-only (compact) pills are too small for the circle, so show a bare x. */
|
||||
[data-pill-compact="true"]
|
||||
.composer-pill-btn:not([data-keep-label])
|
||||
.composer-pill-x {
|
||||
@apply size-[15px] bg-transparent p-0 dark:bg-transparent;
|
||||
@apply size-[min(calc(15px*var(--ui-font-scale,1)),calc(7.5px+7.5px*var(--ui-font-scale,1)))] bg-transparent p-0 dark:bg-transparent;
|
||||
}
|
||||
|
||||
/* Compact pills hide their labels, so surface the name as a hover
|
||||
|
|
@ -1853,8 +1858,8 @@ html[data-chat-font] .aui-root {
|
|||
|
||||
/* Smaller tick for selected Thinking options. */
|
||||
.unsloth-tick {
|
||||
width: 0.8rem !important;
|
||||
height: 0.8rem !important;
|
||||
width: min(calc(0.8rem * var(--ui-font-scale, 1)), calc(0.4rem + 0.4rem * var(--ui-font-scale, 1))) !important;
|
||||
height: min(calc(0.8rem * var(--ui-font-scale, 1)), calc(0.4rem + 0.4rem * var(--ui-font-scale, 1))) !important;
|
||||
}
|
||||
|
||||
/* Soft elevation; [data-slot] outranks the component ring-1, dropping the border. */
|
||||
|
|
@ -1943,9 +1948,9 @@ html[data-chat-font] .aui-root {
|
|||
[data-slot="dropdown-menu-item"],
|
||||
[data-slot="dropdown-menu-sub-trigger"]
|
||||
)
|
||||
svg {
|
||||
width: 1.15rem;
|
||||
height: 1.15rem;
|
||||
svg:not(.unsloth-tick) {
|
||||
width: var(--ui-icon-size) !important;
|
||||
height: var(--ui-icon-size) !important;
|
||||
}
|
||||
|
||||
/* Destructive items keep red text and a red-tinted hover, not the grey one. */
|
||||
|
|
@ -2770,3 +2775,96 @@ html[data-chat-font] .aui-root {
|
|||
display: block !important;
|
||||
width: 8px;
|
||||
}
|
||||
|
||||
/* Icons that sit beside scaled labels follow the UI font size itself:
|
||||
glyphs at or above a 16px base render at --ui-icon-size (12 -> 12px,
|
||||
16 -> 16px, 20 -> 18px), so icons track the text below the default and
|
||||
read slightly smaller than it above. Sub-16px glyphs keep their
|
||||
proportions through the same curve as a factor. Menu, select and closed select trigger surfaces, popovers, toasts,
|
||||
the chat thread and both composers. Only glyphs scale; hit targets,
|
||||
paddings and surface geometry stay fixed. Identity at the default. */
|
||||
:is(
|
||||
[data-slot='dropdown-menu-content'],
|
||||
[data-slot='dropdown-menu-sub-content'],
|
||||
[data-slot='select-content'],
|
||||
[data-slot='select-trigger'],
|
||||
[data-slot='combobox-content'],
|
||||
[data-slot='combobox-trigger'],
|
||||
[data-slot='context-menu-content'],
|
||||
[data-slot='context-menu-sub-content'],
|
||||
[data-slot='menubar-content'],
|
||||
[data-slot='popover-content'],
|
||||
[data-slot='command'],
|
||||
[data-sonner-toast],
|
||||
.composer-action-wrapper,
|
||||
.aui-composer-action-wrapper,
|
||||
.aui-action-bar-more-content,
|
||||
.aui-root
|
||||
) {
|
||||
& svg.size-2\.5 { width: min(calc(0.625rem * var(--ui-font-scale, 1)), calc(0.3125rem + 0.3125rem * var(--ui-font-scale, 1))); height: min(calc(0.625rem * var(--ui-font-scale, 1)), calc(0.3125rem + 0.3125rem * var(--ui-font-scale, 1))); }
|
||||
& svg.size-3 { width: min(calc(0.75rem * var(--ui-font-scale, 1)), calc(0.375rem + 0.375rem * var(--ui-font-scale, 1))); height: min(calc(0.75rem * var(--ui-font-scale, 1)), calc(0.375rem + 0.375rem * var(--ui-font-scale, 1))); }
|
||||
& svg.size-3\.5 { width: min(calc(0.875rem * var(--ui-font-scale, 1)), calc(0.4375rem + 0.4375rem * var(--ui-font-scale, 1))); height: min(calc(0.875rem * var(--ui-font-scale, 1)), calc(0.4375rem + 0.4375rem * var(--ui-font-scale, 1))); }
|
||||
& svg.size-4 { width: var(--ui-icon-size); height: var(--ui-icon-size); }
|
||||
& svg.size-4\.5 { width: var(--ui-icon-size); height: var(--ui-icon-size); }
|
||||
& svg.size-5 { width: var(--ui-icon-size); height: var(--ui-icon-size); }
|
||||
& svg.size-6 { width: min(calc(1.5rem * var(--ui-font-scale, 1)), calc(0.75rem + 0.75rem * var(--ui-font-scale, 1))); height: min(calc(1.5rem * var(--ui-font-scale, 1)), calc(0.75rem + 0.75rem * var(--ui-font-scale, 1))); }
|
||||
& svg.size-\[5px\] { width: min(calc(5px * var(--ui-font-scale, 1)), calc(2.5px + 2.5px * var(--ui-font-scale, 1))); height: min(calc(5px * var(--ui-font-scale, 1)), calc(2.5px + 2.5px * var(--ui-font-scale, 1))); }
|
||||
& svg.size-\[6px\] { width: min(calc(6px * var(--ui-font-scale, 1)), calc(3px + 3px * var(--ui-font-scale, 1))); height: min(calc(6px * var(--ui-font-scale, 1)), calc(3px + 3px * var(--ui-font-scale, 1))); }
|
||||
& svg.size-\[10px\] { width: min(calc(10px * var(--ui-font-scale, 1)), calc(5px + 5px * var(--ui-font-scale, 1))); height: min(calc(10px * var(--ui-font-scale, 1)), calc(5px + 5px * var(--ui-font-scale, 1))); }
|
||||
& svg.size-\[11px\] { width: min(calc(11px * var(--ui-font-scale, 1)), calc(5.5px + 5.5px * var(--ui-font-scale, 1))); height: min(calc(11px * var(--ui-font-scale, 1)), calc(5.5px + 5.5px * var(--ui-font-scale, 1))); }
|
||||
& svg.size-\[12px\] { width: min(calc(12px * var(--ui-font-scale, 1)), calc(6px + 6px * var(--ui-font-scale, 1))); height: min(calc(12px * var(--ui-font-scale, 1)), calc(6px + 6px * var(--ui-font-scale, 1))); }
|
||||
& svg.size-\[13px\] { width: min(calc(13px * var(--ui-font-scale, 1)), calc(6.5px + 6.5px * var(--ui-font-scale, 1))); height: min(calc(13px * var(--ui-font-scale, 1)), calc(6.5px + 6.5px * var(--ui-font-scale, 1))); }
|
||||
& svg.size-\[14px\] { width: min(calc(14px * var(--ui-font-scale, 1)), calc(7px + 7px * var(--ui-font-scale, 1))); height: min(calc(14px * var(--ui-font-scale, 1)), calc(7px + 7px * var(--ui-font-scale, 1))); }
|
||||
& svg.size-\[15px\] { width: min(calc(15px * var(--ui-font-scale, 1)), calc(7.5px + 7.5px * var(--ui-font-scale, 1))); height: min(calc(15px * var(--ui-font-scale, 1)), calc(7.5px + 7.5px * var(--ui-font-scale, 1))); }
|
||||
& svg.size-\[15\.5px\] { width: min(calc(15.5px * var(--ui-font-scale, 1)), calc(7.75px + 7.75px * var(--ui-font-scale, 1))); height: min(calc(15.5px * var(--ui-font-scale, 1)), calc(7.75px + 7.75px * var(--ui-font-scale, 1))); }
|
||||
& svg.size-\[16px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
|
||||
& svg.size-\[17px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
|
||||
& svg.size-\[18px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
|
||||
& svg.size-\[18\.5px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
|
||||
& svg.size-\[20px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
|
||||
& svg.size-\[21px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
|
||||
& svg.size-\[22px\] { width: var(--ui-icon-size); height: var(--ui-icon-size); }
|
||||
& svg.size-\[36px\] { width: min(calc(36px * var(--ui-font-scale, 1)), calc(18px + 18px * var(--ui-font-scale, 1))); height: min(calc(36px * var(--ui-font-scale, 1)), calc(18px + 18px * var(--ui-font-scale, 1))); }
|
||||
& svg.w-3 { width: min(calc(0.75rem * var(--ui-font-scale, 1)), calc(0.375rem + 0.375rem * var(--ui-font-scale, 1))); }
|
||||
& svg.h-3 { height: min(calc(0.75rem * var(--ui-font-scale, 1)), calc(0.375rem + 0.375rem * var(--ui-font-scale, 1))); }
|
||||
& svg.w-3\.5 { width: min(calc(0.875rem * var(--ui-font-scale, 1)), calc(0.4375rem + 0.4375rem * var(--ui-font-scale, 1))); }
|
||||
& svg.h-3\.5 { height: min(calc(0.875rem * var(--ui-font-scale, 1)), calc(0.4375rem + 0.4375rem * var(--ui-font-scale, 1))); }
|
||||
& svg.w-4 { width: var(--ui-icon-size); }
|
||||
& svg.h-4 { height: var(--ui-icon-size); }
|
||||
& svg.w-5 { width: var(--ui-icon-size); }
|
||||
& svg.h-5 { height: var(--ui-icon-size); }
|
||||
/* Buttons default un-classed icons to size-4 the same way. Sonner's
|
||||
close button keeps its compact 12px glyph inside a fixed control. */
|
||||
& button:not([class*=':size-3'], [data-close-button]) svg:not([class*='size-'], [class*='w-'], [class*='h-'], .unsloth-tick) {
|
||||
width: var(--ui-icon-size);
|
||||
height: var(--ui-icon-size);
|
||||
}
|
||||
/* Menu items default un-classed icons to size-4. */
|
||||
& [data-slot*='item'] svg:not([class*='size-'], [class*='w-'], [class*='h-']) {
|
||||
width: var(--ui-icon-size);
|
||||
height: var(--ui-icon-size);
|
||||
}
|
||||
}
|
||||
|
||||
/* Sonner injects fixed 13px toast text and 12px action labels at runtime;
|
||||
text follows the preference at full rate. Line heights are unitless so
|
||||
they track automatically. */
|
||||
[data-sonner-toast][data-styled='true'] {
|
||||
font-size: calc(13px * var(--ui-font-scale, 1)) !important;
|
||||
}
|
||||
[data-sonner-toast][data-styled='true'] [data-description] {
|
||||
font-size: calc(13px * var(--ui-font-scale, 1)) !important;
|
||||
}
|
||||
[data-sonner-toast][data-styled='true'] [data-button] {
|
||||
font-size: calc(12px * var(--ui-font-scale, 1)) !important;
|
||||
}
|
||||
/* Sonner's icon well is a fixed 16px box; track the glyph. */
|
||||
[data-sonner-toast][data-styled='true'] [data-icon] {
|
||||
width: var(--ui-icon-size) !important;
|
||||
height: var(--ui-icon-size) !important;
|
||||
}
|
||||
/* Defensive: the built-in loader is unused (a custom loading icon is always
|
||||
passed) but keep its fixed --size on the scale in case that changes. */
|
||||
[data-sonner-toast] .sonner-loading-wrapper {
|
||||
--size: var(--ui-icon-size) !important;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -46,10 +46,13 @@ fn save_filter(file_name: &str) -> (&'static str, Vec<&'static str>) {
|
|||
Some("jsonl") | Some("ndjson") => ("JSON Lines", vec!["jsonl", "ndjson"]),
|
||||
Some("csv") => ("CSV", vec!["csv"]),
|
||||
Some("md") | Some("markdown") => ("Markdown", vec!["md", "markdown"]),
|
||||
Some("html") | Some("htm") => ("HTML", vec!["html", "htm"]),
|
||||
Some("zip") => ("ZIP archive", vec!["zip"]),
|
||||
_ => (
|
||||
"Export files",
|
||||
vec!["json", "jsonl", "ndjson", "csv", "md", "markdown", "zip"],
|
||||
vec![
|
||||
"json", "jsonl", "ndjson", "csv", "md", "markdown", "html", "htm", "zip",
|
||||
],
|
||||
),
|
||||
}
|
||||
}
|
||||
|
|
@ -252,6 +255,12 @@ mod tests {
|
|||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn html_canvas_exports_use_an_html_save_filter() {
|
||||
assert_eq!(save_filter("canvas.html"), ("HTML", vec!["html", "htm"]));
|
||||
assert_eq!(save_filter("canvas.HTM"), ("HTML", vec!["html", "htm"]));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reads_supported_import_and_rejects_other_extensions() {
|
||||
let jsonl_path = temp_path("allowed").with_extension("JSONL");
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@
|
|||
"app": {
|
||||
"withGlobalTauri": true,
|
||||
"security": {
|
||||
"csp": "default-src 'self'; connect-src 'self' http://localhost:* ws://localhost:* ws://127.0.0.1:* http://127.0.0.1:* https://huggingface.co https://*.huggingface.co https://datasets-server.huggingface.co; img-src 'self' data: blob: https:; media-src 'self' data: blob: https:; style-src 'self' 'unsafe-inline'; font-src 'self' data:"
|
||||
"csp": "default-src 'self'; connect-src 'self' http://localhost:* ws://localhost:* ws://127.0.0.1:* http://127.0.0.1:* https://huggingface.co https://*.huggingface.co https://datasets-server.huggingface.co; img-src 'self' data: blob: https:; media-src 'self' data: blob: https:; style-src 'self' 'unsafe-inline'; font-src 'self' data:; frame-src 'self' http://localhost:* http://127.0.0.1:*"
|
||||
},
|
||||
"windows": [
|
||||
{
|
||||
|
|
|
|||
|
|
@ -70,6 +70,8 @@ ART_DIR = os.environ.get("PW_ART_DIR", "logs/playwright_modelcfg")
|
|||
ART = Path(ART_DIR)
|
||||
ART.mkdir(parents = True, exist_ok = True)
|
||||
STRICT = os.environ.get("STUDIO_UI_STRICT", "0") == "1"
|
||||
PLAYWRIGHT_BROWSER = os.environ.get("STUDIO_PLAYWRIGHT_BROWSER", "chromium").lower()
|
||||
PLAYWRIGHT_CHANNEL = os.environ.get("STUDIO_PLAYWRIGHT_CHANNEL") or None
|
||||
TURN_TIMEOUT_MS = int(os.environ.get("STUDIO_UI_TURN_TIMEOUT_MS", "180000"))
|
||||
WALL_TIMEOUT_S = float(os.environ.get("STUDIO_UI_WALL_TIMEOUT_S", "720"))
|
||||
FETCH_TIMEOUT_MS = int(os.environ.get("STUDIO_UI_FETCH_TIMEOUT_MS", "30000"))
|
||||
|
|
@ -145,10 +147,19 @@ with sync_playwright() as p:
|
|||
)
|
||||
# Health pre-flight: bash-side health wait can pass before the auth DB migrates.
|
||||
wait_for_health(BASE, timeout = 30.0, info = info)
|
||||
browser = p.chromium.launch(
|
||||
headless = True,
|
||||
args = chromium_launch_args(),
|
||||
)
|
||||
if PLAYWRIGHT_BROWSER not in ("chromium", "firefox", "webkit"):
|
||||
fail(f"unsupported STUDIO_PLAYWRIGHT_BROWSER={PLAYWRIGHT_BROWSER!r}")
|
||||
sys.exit(1)
|
||||
browser_type = getattr(p, PLAYWRIGHT_BROWSER)
|
||||
launch_kwargs = {"headless": True}
|
||||
if PLAYWRIGHT_BROWSER == "chromium":
|
||||
launch_kwargs["args"] = chromium_launch_args()
|
||||
if PLAYWRIGHT_CHANNEL:
|
||||
launch_kwargs["channel"] = PLAYWRIGHT_CHANNEL
|
||||
elif PLAYWRIGHT_CHANNEL:
|
||||
fail("STUDIO_PLAYWRIGHT_CHANNEL requires chromium")
|
||||
sys.exit(1)
|
||||
browser = browser_type.launch(**launch_kwargs)
|
||||
ctx = browser.new_context(
|
||||
viewport = {"width": 1280, "height": 900},
|
||||
reduced_motion = "reduce",
|
||||
|
|
@ -464,11 +475,6 @@ with sync_playwright() as p:
|
|||
else:
|
||||
default_ctx = ctx_in.input_value()
|
||||
info(f"default Context Length shown: {default_ctx!r}")
|
||||
ctx_in.click()
|
||||
ctx_in.fill(str(DISTINCT_CTX))
|
||||
page.wait_for_timeout(300)
|
||||
page.keyboard.press("Tab") # blur to commit
|
||||
page.wait_for_timeout(300)
|
||||
remember = popover.get_by_label("Remember for this model").first
|
||||
if _count(remember):
|
||||
try:
|
||||
|
|
@ -477,12 +483,16 @@ with sync_playwright() as p:
|
|||
remember.click()
|
||||
else:
|
||||
fail("'Remember for this model' checkbox not found")
|
||||
ctx_in.click()
|
||||
ctx_in.fill(str(DISTINCT_CTX))
|
||||
page.wait_for_timeout(300)
|
||||
shoot("05-ctx-set")
|
||||
btn = primary_button(popover)
|
||||
if btn is None:
|
||||
fail("primary Load/Save button not found in run-settings")
|
||||
else:
|
||||
# Keep the input focused. The button click must commit the draft
|
||||
# and use it in the same load request.
|
||||
btn.click()
|
||||
page.wait_for_timeout(2500)
|
||||
shoot("06-after-load")
|
||||
|
|
@ -570,6 +580,47 @@ with sync_playwright() as p:
|
|||
else:
|
||||
info("OK reset: distinctive context cleared from unsloth_model_configs")
|
||||
shoot("08-after-reset")
|
||||
|
||||
# ─────────────────────────────────────────────────────
|
||||
# 3b. Re-typing the value already shown must not pin an override (HARD).
|
||||
# Entering the currently displayed native/default context commits no
|
||||
# onChange (the value is unchanged), so the cached blur value must not be
|
||||
# replayed into a stored override on Load. Otherwise re-typing the shown
|
||||
# number, or doing so before a Reset, recreates a phantom context pin.
|
||||
# ─────────────────────────────────────────────────────
|
||||
step("re-typing the shown context does not pin an override")
|
||||
ctx_in = context_input(popover)
|
||||
native_default = _as_int(ctx_in.input_value()) if ctx_in else None
|
||||
if ctx_in is None or native_default is None:
|
||||
info("skip re-type-shown: Context Length input has no numeric default")
|
||||
else:
|
||||
remember = popover.get_by_label("Remember for this model").first
|
||||
if _count(remember):
|
||||
try:
|
||||
remember.check()
|
||||
except Exception:
|
||||
remember.click()
|
||||
ctx_in.click()
|
||||
ctx_in.fill(str(native_default))
|
||||
page.wait_for_timeout(200)
|
||||
btn = primary_button(popover)
|
||||
if btn is not None and btn.is_enabled():
|
||||
# Same-click Load: the button click must commit the draft, but a draft
|
||||
# equal to the shown value carries no override.
|
||||
btn.click()
|
||||
page.wait_for_timeout(1500)
|
||||
cfg = read_configs()
|
||||
pinned = any(
|
||||
_as_int(e.get("customContextLength")) == native_default for e in config_entries(cfg)
|
||||
)
|
||||
if pinned:
|
||||
fail(
|
||||
"re-typing the shown context pinned it as an override "
|
||||
f"(customContextLength={native_default})"
|
||||
)
|
||||
else:
|
||||
info("OK re-type-shown: shown context not stored as an override")
|
||||
shoot("08b-after-retype-shown")
|
||||
close_picker()
|
||||
|
||||
# ─────────────────────────────────────────────────────
|
||||
|
|
|
|||
|
|
@ -208,6 +208,14 @@ def main():
|
|||
# text-ui-12p5 at scale 0.75; 16px means twMerge dropped the token.
|
||||
if not near(tab_font, 12.5 * 12 / 16):
|
||||
fail(f"hub tab font did not scale (twMerge drop?): {tab_font}")
|
||||
icon_w = page.evaluate(
|
||||
"() => { const el = document.querySelector('.size-icon');"
|
||||
" return el ? parseFloat(getComputedStyle(el).width) : null; }"
|
||||
)
|
||||
# Standard icons render at the UI font size itself below the
|
||||
# default, so setting 12 gives 12px glyphs.
|
||||
if not near(icon_w, 12):
|
||||
fail(f"size-icon did not match the UI font size below 16: {icon_w}")
|
||||
page.goto(BASE, wait_until = "domcontentloaded")
|
||||
page.wait_for_timeout(1500)
|
||||
open_appearance(page)
|
||||
|
|
|
|||
26
tests/studio/test_generation_length_ui_contract.py
Normal file
26
tests/studio/test_generation_length_ui_contract.py
Normal file
|
|
@ -0,0 +1,26 @@
|
|||
# SPDX-License-Identifier: AGPL-3.0-only
|
||||
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved.
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
CHAT_API = (
|
||||
Path(__file__).resolve().parents[2]
|
||||
/ "studio"
|
||||
/ "frontend"
|
||||
/ "src"
|
||||
/ "features"
|
||||
/ "chat"
|
||||
/ "api"
|
||||
/ "chat-api.ts"
|
||||
)
|
||||
|
||||
|
||||
def test_length_detection_classifies_visible_and_reasoning_content():
|
||||
source = CHAT_API.read_text(encoding = "utf-8")
|
||||
|
||||
assert "return value.trim().length > 0;" in source
|
||||
assert 'record.type === "thinking" || record.type === "reasoning"' in source
|
||||
assert 'record.type === "text" || record.type === "output_text"' in source
|
||||
assert "sawAssistantContent ||= contentState.hasAssistantContent;" in source
|
||||
assert "sawReasoningContent ||= contentState.hasReasoningContent;" in source
|
||||
|
|
@ -346,6 +346,36 @@ def test_fixed_layer_gguf_pins_displayed_context():
|
|||
assert "customContextLength: activeLoadedContext" in src
|
||||
|
||||
|
||||
def test_fixed_layer_pin_recomputed_after_committing_gpu_layers():
|
||||
"""pinFixedLayerContext is computed from the render-time config, before a
|
||||
same-click GPU Layers draft is committed. handleRun must recompute it from the
|
||||
committed effectiveConfig; otherwise typing a positive GPU Layers value on an
|
||||
auto-fit GGUF and clicking Reload saves customContextLength: null, so a later
|
||||
fresh load sends the native context with fixed layers (the OOM the pin avoids)."""
|
||||
src = _read("features/model-picker/components/model-config-page.tsx")
|
||||
assert "const effectivePinFixedLayerContext =" in src
|
||||
assert 'effectiveConfig.gpuMemoryMode === "manual"' in src
|
||||
assert "effectiveConfig.gpuLayers != null" in src
|
||||
assert "effectiveConfig.customContextLength == null" in src
|
||||
assert "{ ...effectiveConfig, customContextLength: activeLoadedContext }" in src
|
||||
|
||||
|
||||
def test_blur_cache_cleared_on_every_settled_render():
|
||||
"""The lastBlurCommittedRef bridge is valid only across the single synchronous
|
||||
same-click gesture that set it. Keying its clear on [value] missed a Reset (or
|
||||
external edit) that restores the shown value unchanged after the blur dispatched
|
||||
onChange: value nets back to its prior number, the effect never re-ran, and a
|
||||
later Load/Save replayed the override Reset removed. Clear it on every settled
|
||||
render instead."""
|
||||
src = _read("features/model-picker/components/numeric-value-input.tsx")
|
||||
# The clearing effect must run on every commit, not be gated on [value] alone.
|
||||
assert not re.search(r"lastBlurCommittedRef\.current = null;\s*\}, \[value\]\);", src)
|
||||
assert re.search(
|
||||
r"useEffect\(\(\) => \{\s*lastBlurCommittedRef\.current = null;\s*\}\);",
|
||||
src,
|
||||
)
|
||||
|
||||
|
||||
def test_auto_defaults_not_persisted_as_overrides():
|
||||
"""Auto GPU memory mode and Auto/default speculative type are follow-global
|
||||
defaults; normalization must not persist them as per-model overrides, else a
|
||||
|
|
@ -382,15 +412,90 @@ def test_reset_persists_null_max_length_and_substitutes_only_for_load():
|
|||
Reset) so isDefaultConfig can clear a remembered override; the concrete
|
||||
fallback is substituted only into the load request, not the saved record."""
|
||||
src = _read("features/model-picker/components/model-config-page.tsx")
|
||||
# Load-only substitution of the resolved value.
|
||||
assert "maxSeqLength: maxSeqLengthValue" in src
|
||||
assert "const loadConfig" in src
|
||||
# The persisted record is loaded via onRun(loadConfig), and save uses the
|
||||
# untouched runtimeConfig (so a reset/default config stays default).
|
||||
assert "onRun(loadConfig)" in src
|
||||
# Load-only substitution of the resolved value (recomputed from any committed
|
||||
# same-click Max Seq Length draft, so it is never dropped).
|
||||
assert "maxSeqLength: effectiveMaxSeqLengthValue" in src
|
||||
assert "const effectiveLoadConfig" in src
|
||||
# The persisted record is saved from effectiveRuntimeConfig; the load request
|
||||
# carries effectiveLoadConfig (with any committed context input).
|
||||
assert "onRun(effectiveLoadConfig)" in src
|
||||
assert "savePerModelConfig(" in src
|
||||
|
||||
|
||||
def test_initial_load_uses_staged_config_payload():
|
||||
"""Run-settings Load must pass the staged config through to /load even when
|
||||
React has not flushed NumericValueInput blur commits into the store yet."""
|
||||
runtime = _read("features/chat/hooks/use-chat-model-runtime.ts")
|
||||
assert "const pendingLoadConfig =" in runtime
|
||||
assert "pendingLoadConfig?.kvCacheDtype" in runtime
|
||||
assert "pendingLoadConfig?.customContextLength" in runtime
|
||||
page = _read("features/model-picker/components/model-config-page.tsx")
|
||||
assert "contextInputRef" in page
|
||||
assert "contextInputRef.current?.commit()" in page
|
||||
numeric = _read("features/model-picker/components/numeric-value-input.tsx")
|
||||
assert "export type NumericValueInputHandle" in numeric
|
||||
assert "commit:" in numeric
|
||||
# P1: commit returns null unless the user actually edited the field,
|
||||
# so Load/Save with untouched Auto does not pin native context.
|
||||
assert "dirtyRef.current" in numeric
|
||||
assert "return null;" in numeric
|
||||
# P2: blur clears dirtyRef after commit so Reset/slider cannot be
|
||||
# overwritten by a stale draft on a later Load.
|
||||
assert "dirtyRef.current = false;" in numeric
|
||||
assert "draftRef.current = String(final);" in numeric
|
||||
# Same-click Load after blur still sees the committed draft.
|
||||
assert "lastBlurCommittedRef" in numeric
|
||||
# Invalid drafts must not turn Auto into an explicit pin.
|
||||
assert "const commitDraft = (raw: string): number | null" in numeric
|
||||
assert re.search(r"if \(!Number\.isFinite\(parsed\)\) \{\s*return null;", numeric)
|
||||
assert re.search(
|
||||
r"if \(final == null\) \{\s*"
|
||||
r"draftRef\.current = String\(value\);\s*"
|
||||
r"lastBlurCommittedRef\.current = null;",
|
||||
numeric,
|
||||
)
|
||||
# handleRun only promotes commit() when non-null.
|
||||
assert "committedContext != null" in page
|
||||
assert "pendingPatch.customContextLength = committedContext;" in page
|
||||
|
||||
|
||||
def test_same_click_commit_covers_all_numeric_inputs():
|
||||
"""The same-click blur bridge must flush every NumericValueInput-backed
|
||||
setting, not just Context Length. Max Seq Length (non-GGUF), GPU Layers and
|
||||
MoE Layers (GGUF) also stage their draft only on blur, so handleRun must
|
||||
imperatively commit each and fold the value into the staged load config;
|
||||
otherwise a value the user typed right before clicking Load/Reload is lost."""
|
||||
page = _read("features/model-picker/components/model-config-page.tsx")
|
||||
# Each numeric input owns an imperative handle that handleRun commits, and the
|
||||
# handle is forwarded down to the actual NumericValueInput.
|
||||
for ref in ("maxSeqLengthInputRef", "gpuLayersInputRef", "moeLayersInputRef"):
|
||||
assert f"const {ref} = useRef<NumericValueInputHandle>(null);" in page
|
||||
assert f"{ref}.current?.commit()" in page
|
||||
assert f"inputRef={{{ref}}}" in page
|
||||
# The leaf sub-components accept and forward the handle as a ref.
|
||||
assert page.count("inputRef?: Ref<NumericValueInputHandle>;") >= 2
|
||||
assert "ref={inputRef}" in page
|
||||
# Committed drafts are folded into the staged config, gated on non-null so an
|
||||
# untouched field never fabricates an override.
|
||||
assert "committedMaxSeqLength != null" in page
|
||||
assert "committedGpuLayers != null" in page
|
||||
assert "committedMoeLayers != null" in page
|
||||
assert "pendingPatch.gpuLayers = committedGpuLayers;" in page
|
||||
assert "pendingPatch.nCpuMoe = committedMoeLayers;" in page
|
||||
# The non-GGUF load path substitutes the committed Max Seq Length draft.
|
||||
assert "const effectiveMaxSeqLengthValue =" in page
|
||||
assert "maxSeqLength: effectiveMaxSeqLengthValue" in page
|
||||
|
||||
|
||||
def test_context_commit_rechecks_persistence_only_shortcut():
|
||||
"""Committed context changes must bypass persistence-only saves."""
|
||||
src = _read("features/model-picker/components/model-config-page.tsx")
|
||||
assert "const effectiveConfig =" in src
|
||||
assert "perModelConfigsEqual(effectiveConfig, baseline)" in src
|
||||
assert "const effectivePersistenceOnly =" in src
|
||||
assert "if (effectivePersistenceOnly)" in src
|
||||
|
||||
|
||||
def test_reset_enabled_for_explicit_context_pin_at_native():
|
||||
"""An explicit customContextLength that equals the native ceiling is still a
|
||||
user override, so contextAtDefault must require customContextLength == null.
|
||||
|
|
|
|||
|
|
@ -98,6 +98,39 @@ def test_cn_knows_the_ui_typography_tokens():
|
|||
assert "/^ui-\\d+(p5)?$/.test(value)" in UTILS
|
||||
|
||||
|
||||
def test_icons_follow_the_ui_font_size_itself():
|
||||
"""Standard glyphs render at --ui-icon-size, which follows the UI font
|
||||
size itself: matches it below the 16px default and grows at half the
|
||||
change above it (setting 20 gives 18px icons), so icons track the text
|
||||
when shrinking and read slightly smaller than it when growing. Sub 16px
|
||||
glyphs keep their proportions through the same curve as a factor.
|
||||
Sonner toast text and action labels are text, so they follow at full
|
||||
rate everywhere."""
|
||||
assert (
|
||||
"--ui-icon-size: min(calc(1rem * var(--ui-font-scale, 1)), "
|
||||
"calc(0.5rem + 0.5rem * var(--ui-font-scale, 1)));"
|
||||
) in INDEX_CSS
|
||||
assert "--icon-size: var(--ui-icon-size);" in INDEX_CSS
|
||||
assert "& svg.size-4 { width: var(--ui-icon-size); height: var(--ui-icon-size); }" in INDEX_CSS
|
||||
assert "font-size: calc(13px * var(--ui-font-scale, 1)) !important;" in INDEX_CSS
|
||||
assert "font-size: calc(12px * var(--ui-font-scale, 1)) !important;" in INDEX_CSS
|
||||
# Menu rules that outrank the scoped block must carry the token too,
|
||||
# without flattening the smaller thinking ticks.
|
||||
assert "width: var(--ui-icon-size) !important;" in INDEX_CSS
|
||||
assert "svg:not(.unsloth-tick) {" in INDEX_CSS
|
||||
# Oversized art glyphs stay proportional instead of uniform.
|
||||
assert "& svg.size-6 { width: min(calc(1.5rem" in INDEX_CSS
|
||||
for scope in (
|
||||
"[data-slot='dropdown-menu-content']",
|
||||
"[data-slot='select-content']",
|
||||
"[data-slot='select-trigger']",
|
||||
"[data-slot='combobox-content']",
|
||||
"[data-sonner-toast]",
|
||||
".aui-root",
|
||||
):
|
||||
assert scope in INDEX_CSS
|
||||
|
||||
|
||||
def test_no_raw_pixel_text_utilities():
|
||||
offenders = []
|
||||
for path in _frontend_sources():
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue