Refresh Hub inference status after API-driven model switches
The Hub read /api/inference/status once, on mount. An OpenAI-compatible request can auto-switch the resident model at any time, and nothing else on /hub reads status, so every 'loaded' marker and, more importantly, the per-model settings page kept describing the model that was resident when the Hub opened. The cost is concrete: with a stale checkpoint the newly loaded model fails settingsTargetIsResident, its settings page is handed loadedConfig=null, ModelConfigPage seeds the editor from saved or default values, and Apply reloads the model with them over the launch settings the API selected. Re-read on the moments this tab could have missed a switch -- regaining focus or visibility, and opening a model's settings page -- rather than polling on a timer. adoptResidentModelStatus already stands down for an external selection and for a load this tab started, so re-running it never fights the model the user is switching to, and applying an unchanged status leaves the live config identical, so the settings editor is not remounted under a draft.
This commit is contained in:
parent
c81fac2c20
commit
41b40bff71
3 changed files with 249 additions and 5 deletions
|
|
@ -74,6 +74,7 @@ import { useHubInventory } from "./inventory";
|
|||
import { LOCAL_MODEL_SOURCE } from "./inventory/constants";
|
||||
import { settingsGgufVariantForRow } from "./inventory/settings-identity";
|
||||
import { adoptResidentModelStatus } from "./lib/adopt-inference-status";
|
||||
import { subscribeResidentStatusRefresh } from "./lib/resident-status-refresh";
|
||||
import {
|
||||
CHANNEL_TO_SECTION,
|
||||
type ChannelId,
|
||||
|
|
@ -384,11 +385,13 @@ export function ModelsPage() {
|
|||
const fitOnDeviceOnly = useChatRuntimeStore((s) => s.fitOnDeviceOnly);
|
||||
const setFitOnDeviceOnly = useChatRuntimeStore((s) => s.setFitOnDeviceOnly);
|
||||
|
||||
useEffect(() => {
|
||||
let cancelled = false;
|
||||
// Drops a response that lands after a newer read started, or after unmount.
|
||||
const residentStatusSeq = useRef(0);
|
||||
const refreshResidentModelStatus = useCallback(() => {
|
||||
const seq = ++residentStatusSeq.current;
|
||||
void getInferenceStatus()
|
||||
.then((status) => {
|
||||
if (cancelled) return;
|
||||
if (seq !== residentStatusSeq.current) return;
|
||||
const store = useChatRuntimeStore.getState();
|
||||
adoptResidentModelStatus(
|
||||
{
|
||||
|
|
@ -424,11 +427,24 @@ export function ModelsPage() {
|
|||
);
|
||||
})
|
||||
.catch(() => undefined);
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, []);
|
||||
|
||||
// Mount, then again whenever this tab could have missed an API-driven switch.
|
||||
// adoptResidentModelStatus is what makes re-reading safe: it stands down for an
|
||||
// external selection and for a load this tab started, so a refresh never fights
|
||||
// the model the user is switching to.
|
||||
useEffect(() => {
|
||||
refreshResidentModelStatus();
|
||||
const unsubscribe = subscribeResidentStatusRefresh(
|
||||
refreshResidentModelStatus,
|
||||
);
|
||||
return () => {
|
||||
// A response still in flight adopts nothing once the Hub is gone.
|
||||
residentStatusSeq.current += 1;
|
||||
unsubscribe();
|
||||
};
|
||||
}, [refreshResidentModelStatus]);
|
||||
|
||||
const { tab, setTab: setModelsTab } = useModelsTabState();
|
||||
const [query, setQuery] = useState("");
|
||||
const [sortBy, setSortBy] = useState<HfSortKey>(
|
||||
|
|
@ -1282,6 +1298,13 @@ export function ModelsPage() {
|
|||
const [settingsTarget, setSettingsTarget] = useState<ModelPickTarget | null>(
|
||||
null,
|
||||
);
|
||||
// Opening a model's settings is where a stale read costs something: the editor
|
||||
// is seeded once, from saved/default values when the store says this model is
|
||||
// not the resident one, and Apply reloads with them. Reading status here covers
|
||||
// a switch that landed while this window kept focus the whole time.
|
||||
useEffect(() => {
|
||||
if (settingsTarget) refreshResidentModelStatus();
|
||||
}, [settingsTarget, refreshResidentModelStatus]);
|
||||
// Bumped per open so a slow variant lookup for an abandoned row cannot land
|
||||
// on top of the row actually chosen.
|
||||
const settingsOpenSeq = useRef(0);
|
||||
|
|
|
|||
|
|
@ -0,0 +1,50 @@
|
|||
// SPDX-License-Identifier: AGPL-3.0-only
|
||||
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
||||
|
||||
// When the Hub has to re-read /api/inference/status.
|
||||
//
|
||||
// An OpenAI-compatible request can auto-switch the resident model at any moment,
|
||||
// and nothing else on /hub reads status: the chat runtime hook has no mount sync
|
||||
// and the chat page is a different route. A mount-only read therefore leaves
|
||||
// every "loaded" marker -- and, worse, the settings page's live config -- pinned
|
||||
// to whatever was resident when the Hub opened, so the newly loaded model's
|
||||
// editor seeds from saved/default values and Apply reloads it with them.
|
||||
//
|
||||
// A background timer would keep asking a server that has usually not changed, so
|
||||
// re-read only on the moments this tab could have missed a switch instead.
|
||||
|
||||
/** The event targets to listen on; injected so this is testable off a browser. */
|
||||
export interface ResidentStatusRefreshTargets {
|
||||
window: Pick<EventTarget, "addEventListener" | "removeEventListener">;
|
||||
document: Pick<EventTarget, "addEventListener" | "removeEventListener"> & {
|
||||
readonly hidden: boolean;
|
||||
};
|
||||
}
|
||||
|
||||
function browserTargets(): ResidentStatusRefreshTargets {
|
||||
return { window, document };
|
||||
}
|
||||
|
||||
/**
|
||||
* Call ``refresh`` whenever this tab comes back to the foreground: returning from
|
||||
* the terminal or client that made the API call is exactly when what is resident
|
||||
* may have moved. Returns the unsubscribe.
|
||||
*/
|
||||
export function subscribeResidentStatusRefresh(
|
||||
refresh: () => void,
|
||||
targets: ResidentStatusRefreshTargets = browserTargets(),
|
||||
): () => void {
|
||||
const onFocus = () => refresh();
|
||||
// Focus alone misses a tab that was merely backgrounded, and visibility alone
|
||||
// misses a window that never went hidden; a redundant pair of reads is cheaper
|
||||
// than a missed switch.
|
||||
const onVisibility = () => {
|
||||
if (!targets.document.hidden) refresh();
|
||||
};
|
||||
targets.window.addEventListener("focus", onFocus);
|
||||
targets.document.addEventListener("visibilitychange", onVisibility);
|
||||
return () => {
|
||||
targets.window.removeEventListener("focus", onFocus);
|
||||
targets.document.removeEventListener("visibilitychange", onVisibility);
|
||||
};
|
||||
}
|
||||
171
studio/frontend/tests/hub-resident-status-refresh.test.ts
Normal file
171
studio/frontend/tests/hub-resident-status-refresh.test.ts
Normal file
|
|
@ -0,0 +1,171 @@
|
|||
// SPDX-License-Identifier: AGPL-3.0-only
|
||||
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
||||
|
||||
import assert from "node:assert/strict";
|
||||
import test from "node:test";
|
||||
|
||||
import { adoptResidentModelStatus } from "../src/features/hub/lib/adopt-inference-status.ts";
|
||||
import {
|
||||
ggufVariantsMatch,
|
||||
residentModelIdMatches,
|
||||
} from "../src/features/hub/lib/model-identity.ts";
|
||||
import {
|
||||
type ResidentStatusRefreshTargets,
|
||||
subscribeResidentStatusRefresh,
|
||||
} from "../src/features/hub/lib/resident-status-refresh.ts";
|
||||
|
||||
function fakeTargets(): ResidentStatusRefreshTargets & {
|
||||
hidden: boolean;
|
||||
fire: (target: "window" | "document", type: string) => void;
|
||||
listenerCount: () => number;
|
||||
} {
|
||||
const listeners = new Map<string, Set<EventListenerOrEventListenerObject>>();
|
||||
const key = (target: string, type: string) => `${target}:${type}`;
|
||||
const make = (target: "window" | "document") => ({
|
||||
addEventListener(type: string, fn: EventListenerOrEventListenerObject) {
|
||||
const set = listeners.get(key(target, type)) ?? new Set();
|
||||
set.add(fn);
|
||||
listeners.set(key(target, type), set);
|
||||
},
|
||||
removeEventListener(type: string, fn: EventListenerOrEventListenerObject) {
|
||||
listeners.get(key(target, type))?.delete(fn);
|
||||
},
|
||||
});
|
||||
const visibility = { hidden: false };
|
||||
const state = {
|
||||
get hidden() {
|
||||
return visibility.hidden;
|
||||
},
|
||||
set hidden(next: boolean) {
|
||||
visibility.hidden = next;
|
||||
},
|
||||
window: make("window"),
|
||||
document: {
|
||||
...make("document"),
|
||||
get hidden() {
|
||||
return visibility.hidden;
|
||||
},
|
||||
},
|
||||
fire(target: "window" | "document", type: string) {
|
||||
for (const fn of listeners.get(key(target, type)) ?? []) {
|
||||
(fn as EventListener)(new Event(type));
|
||||
}
|
||||
},
|
||||
listenerCount() {
|
||||
let total = 0;
|
||||
for (const set of listeners.values()) total += set.size;
|
||||
return total;
|
||||
},
|
||||
};
|
||||
return state as never;
|
||||
}
|
||||
|
||||
test("coming back to the window re-reads inference status", () => {
|
||||
// An OpenAI-compatible request auto-switches the resident model whenever it
|
||||
// likes. The Hub's only other status read is its mount effect, so without this
|
||||
// the catalog and the settings page keep describing the previous model for as
|
||||
// long as the Hub stays mounted.
|
||||
const targets = fakeTargets();
|
||||
let reads = 0;
|
||||
subscribeResidentStatusRefresh(() => {
|
||||
reads += 1;
|
||||
}, targets);
|
||||
|
||||
assert.equal(reads, 0, "subscribing must not read on its own");
|
||||
targets.fire("window", "focus");
|
||||
assert.equal(reads, 1);
|
||||
targets.fire("document", "visibilitychange");
|
||||
assert.equal(reads, 2);
|
||||
});
|
||||
|
||||
test("a tab going hidden does not read", () => {
|
||||
// visibilitychange fires on the way out too, and a hidden tab has no settings
|
||||
// page to correct.
|
||||
const targets = fakeTargets();
|
||||
let reads = 0;
|
||||
subscribeResidentStatusRefresh(() => {
|
||||
reads += 1;
|
||||
}, targets);
|
||||
|
||||
targets.hidden = true;
|
||||
targets.fire("document", "visibilitychange");
|
||||
assert.equal(reads, 0);
|
||||
|
||||
targets.hidden = false;
|
||||
targets.fire("document", "visibilitychange");
|
||||
assert.equal(reads, 1);
|
||||
});
|
||||
|
||||
test("an auto-switch under a mounted Hub stops hiding the live config", () => {
|
||||
// The whole point, end to end: while the Hub is mounted an OpenAI-compatible
|
||||
// request swaps the resident model. Without a second read the store still names
|
||||
// the old one, so hub-page's settingsTargetIsResident says the newly loaded
|
||||
// model is not resident, its settings page is handed loadedConfig=null, and
|
||||
// ModelConfigPage seeds the editor from saved/default values -- which Apply then
|
||||
// reloads the model with, over what the API actually selected.
|
||||
const store = {
|
||||
checkpoint: "unsloth/Qwen3-8B-GGUF" as string | null,
|
||||
checkpointIsExternal: false,
|
||||
activeGgufVariant: "Q4_K_M" as string | null,
|
||||
modelLoading: false,
|
||||
};
|
||||
// What the server reports once the API request has switched it.
|
||||
let serverStatus = {
|
||||
checkpointId: "unsloth/Llama-3.1-8B-Instruct-GGUF",
|
||||
ggufVariant: "Q8_0",
|
||||
};
|
||||
const readStatusAndAdopt = () => {
|
||||
adoptResidentModelStatus(
|
||||
serverStatus,
|
||||
{ ...store },
|
||||
{
|
||||
setCheckpoint: (checkpointId, ggufVariant) => {
|
||||
store.checkpoint = checkpointId;
|
||||
store.activeGgufVariant = ggufVariant;
|
||||
},
|
||||
applyStatus: () => undefined,
|
||||
},
|
||||
);
|
||||
};
|
||||
|
||||
// hub-page.tsx's settingsTargetIsResident, for the model the API just loaded.
|
||||
const settingsTargetIsResident = () =>
|
||||
residentModelIdMatches(store.checkpoint, serverStatus.checkpointId) &&
|
||||
ggufVariantsMatch(store.activeGgufVariant, serverStatus.ggufVariant);
|
||||
|
||||
const targets = fakeTargets();
|
||||
subscribeResidentStatusRefresh(readStatusAndAdopt, targets);
|
||||
|
||||
assert.equal(
|
||||
settingsTargetIsResident(),
|
||||
false,
|
||||
"precondition: the mount-time read predates the switch",
|
||||
);
|
||||
targets.fire("window", "focus");
|
||||
assert.equal(settingsTargetIsResident(), true);
|
||||
|
||||
// A load this tab started owns the store until it settles, so a refresh landing
|
||||
// mid-switch must not re-pin the model the user is moving away from.
|
||||
store.modelLoading = true;
|
||||
serverStatus = {
|
||||
checkpointId: "unsloth/Qwen3-8B-GGUF",
|
||||
ggufVariant: "Q4_K_M",
|
||||
};
|
||||
targets.fire("window", "focus");
|
||||
assert.equal(store.checkpoint, "unsloth/Llama-3.1-8B-Instruct-GGUF");
|
||||
});
|
||||
|
||||
test("unsubscribing stops the reads and leaves no listener behind", () => {
|
||||
const targets = fakeTargets();
|
||||
let reads = 0;
|
||||
const unsubscribe = subscribeResidentStatusRefresh(() => {
|
||||
reads += 1;
|
||||
}, targets);
|
||||
|
||||
assert.equal(targets.listenerCount(), 2);
|
||||
unsubscribe();
|
||||
assert.equal(targets.listenerCount(), 0);
|
||||
targets.fire("window", "focus");
|
||||
targets.fire("document", "visibilitychange");
|
||||
assert.equal(reads, 0);
|
||||
});
|
||||
Loading…
Add table
Add a link
Reference in a new issue