diff --git a/tools/ui/src/lib/components/app/chat/ChatForm/ChatFormContextGauge/ContextGaugePopup.svelte b/tools/ui/src/lib/components/app/chat/ChatForm/ChatFormContextGauge/ContextGaugePopup.svelte index 8fa09cf704..cdd6821e38 100644 --- a/tools/ui/src/lib/components/app/chat/ChatForm/ChatFormContextGauge/ContextGaugePopup.svelte +++ b/tools/ui/src/lib/components/app/chat/ChatForm/ChatFormContextGauge/ContextGaugePopup.svelte @@ -73,7 +73,7 @@ - {#if gauge.activeModelId !== null && !gauge.isActiveModelLoaded} + {#if gauge.canLoadActiveModel && gauge.activeModelId !== null && !gauge.isActiveModelLoaded} (null); @@ -28,11 +28,14 @@ { + onModelChange={async (modelId: string, modelName: string, backendId?: string) => { + // capability of the picked model's own backend, not the active one + const loadsOnRequest = getBackendCapabilities(getBackend(backendId)).loadUnload; const status = modelsStore.getModelStatus(modelId); - // external backends load the model implicitly on the request - if (canLoadModels && status !== ServerModelStatus.LOADED) { + // only a llama.cpp server loads up front; remote backends load the + // model with the request itself + if (loadsOnRequest && status !== ServerModelStatus.LOADED) { pendingModel = modelId; try { diff --git a/tools/ui/src/lib/components/app/models/ModelsSelector/ModelsSelectorDropdown.svelte b/tools/ui/src/lib/components/app/models/ModelsSelector/ModelsSelectorDropdown.svelte index 9f0abc6685..934fcdab46 100644 --- a/tools/ui/src/lib/components/app/models/ModelsSelector/ModelsSelectorDropdown.svelte +++ b/tools/ui/src/lib/components/app/models/ModelsSelector/ModelsSelectorDropdown.svelte @@ -31,7 +31,11 @@ currentModel?: string | null; disabled?: boolean; forceForegroundText?: boolean; - onModelChange?: (modelId: string, modelName: string) => Promise | boolean | void; + onModelChange?: ( + modelId: string, + modelName: string, + backendId?: string + ) => Promise | boolean | void; useGlobalSelection?: boolean; } diff --git a/tools/ui/src/lib/components/app/models/ModelsSelector/ModelsSelectorSheet.svelte b/tools/ui/src/lib/components/app/models/ModelsSelector/ModelsSelectorSheet.svelte index 5735eaf39d..956abeab77 100644 --- a/tools/ui/src/lib/components/app/models/ModelsSelector/ModelsSelectorSheet.svelte +++ b/tools/ui/src/lib/components/app/models/ModelsSelector/ModelsSelectorSheet.svelte @@ -22,7 +22,11 @@ class?: string; currentModel?: string | null; /** Callback when model changes. Return false to keep menu open (e.g., for validation failures) */ - onModelChange?: (modelId: string, modelName: string) => Promise | boolean | void; + onModelChange?: ( + modelId: string, + modelName: string, + backendId?: string + ) => Promise | boolean | void; disabled?: boolean; forceForegroundText?: boolean; /** When true, user's global selection takes priority over currentModel (for form selector) */ diff --git a/tools/ui/src/lib/hooks/use-context-gauge.svelte.ts b/tools/ui/src/lib/hooks/use-context-gauge.svelte.ts index c6d55e3935..2ee9cb7260 100644 --- a/tools/ui/src/lib/hooks/use-context-gauge.svelte.ts +++ b/tools/ui/src/lib/hooks/use-context-gauge.svelte.ts @@ -8,10 +8,12 @@ import { useProcessingState } from './use-processing-state.svelte'; import { colorLevelFromPercent } from '$lib/components/app/chat/ChatForm/ChatFormContextGauge/context-gauge'; import { STATS_UNITS } from '$lib/constants'; import { ColorLevel } from '$lib/enums'; -import { contextStatsStore, modelsStore } from '$lib/stores'; +import { contextStatsStore, modelsStore, serverStore } from '$lib/stores'; export interface UseContextGaugeReturn { readonly activeModelId: string | null; + /** Whether the active backend can load the model it is serving. */ + readonly canLoadActiveModel: boolean; readonly isActiveModelLoaded: boolean; readonly isActiveModelLoading: boolean; readonly contextTotal: number | null; @@ -73,6 +75,9 @@ export function useContextGauge(): UseContextGaugeReturn { contextStatsStore.averageTokensPerSecond !== null || transientDetails.length > 0 ); + // loading is a llama.cpp router feature, and the gauge tracks the active model, + // so the active backend decides whether a load is possible at all + const canLoadActiveModel = $derived(serverStore.isRouterMode); async function loadModel() { const modelId = contextStatsStore.activeModelId; @@ -93,6 +98,9 @@ export function useContextGauge(): UseContextGaugeReturn { get averageTokensPerSecond() { return contextStatsStore.averageTokensPerSecond; }, + get canLoadActiveModel() { + return canLoadActiveModel; + }, get colorLevel() { return colorLevel; }, diff --git a/tools/ui/src/lib/hooks/use-models-selector.svelte.ts b/tools/ui/src/lib/hooks/use-models-selector.svelte.ts index f4c4b6b21e..38c9502536 100644 --- a/tools/ui/src/lib/hooks/use-models-selector.svelte.ts +++ b/tools/ui/src/lib/hooks/use-models-selector.svelte.ts @@ -26,7 +26,11 @@ export interface UseModelsSelectorOptions { currentModel: () => string | null; useGlobalSelection?: () => boolean; onModelChange?: () => - | ((modelId: string, modelName: string) => Promise | boolean | void) + | (( + modelId: string, + modelName: string, + backendId?: string + ) => Promise | boolean | void) | undefined; onOpenChange?: (open: boolean) => void; } @@ -266,7 +270,7 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele let shouldCloseMenu = true; if (onModelChange) { - const result = await onModelChange(rawModelId(option.id), option.model); + const result = await onModelChange(rawModelId(option.id), option.model, option.backendId); if (result === false) { shouldCloseMenu = false; @@ -285,7 +289,10 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele }); } - if (!onModelChange && isRouter && !modelsStore.isModelLoaded(option.model)) { + // only the built-in server loads on request, and only in router mode + const canLoadHere = option.backendId === LOCAL_BACKEND_ID && isRouter; + + if (!onModelChange && canLoadHere && !modelsStore.isModelLoaded(option.model)) { isLoadingModel = true; modelsStore.status diff --git a/tools/ui/src/lib/utils/backend.ts b/tools/ui/src/lib/utils/backend.ts index d578648684..4dcb281602 100644 --- a/tools/ui/src/lib/utils/backend.ts +++ b/tools/ui/src/lib/utils/backend.ts @@ -57,9 +57,12 @@ function normalizeBaseUrl(url: string): string | null { } } -/** Features a backend supports, derived from its protocol. */ -export function getBackendCapabilities(backend: Backend): BackendCapabilities { - return BACKEND_CAPABILITIES[backend.protocol] ?? BACKEND_CAPABILITIES.openai; +/** + * Features a backend supports, derived from its protocol. A missing backend + * (unknown model, early startup) gets the plain compatible defaults. + */ +export function getBackendCapabilities(backend?: Backend): BackendCapabilities { + return BACKEND_CAPABILITIES[backend?.protocol ?? 'openai'] ?? BACKEND_CAPABILITIES.openai; } /** Wire quirks for a backend: protocol defaults overridden by the backend. */