ui : load a model only on backends that load it

Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
This commit is contained in:
Aleksander Grygier
2026-09-28 12:10:11 +02:00
parent 849526c6bc
commit d2bbf81fc7
7 changed files with 44 additions and 15 deletions
@@ -73,7 +73,7 @@
</span>
</div>
{#if gauge.activeModelId !== null && !gauge.isActiveModelLoaded}
{#if gauge.canLoadActiveModel && gauge.activeModelId !== null && !gauge.isActiveModelLoaded}
<ContextGaugeLoadModel
isLoading={gauge.isActiveModelLoading}
modelId={gauge.activeModelId}
@@ -2,7 +2,8 @@
import { ModelBadge, ModelsSelectorDropdown } from '$lib/components/app';
import { ServerModelStatus } from '$lib/enums';
import { modelsStore, serverStore } from '$lib/stores';
import { copyToClipboard } from '$lib/utils';
import { copyToClipboard, getBackendCapabilities } from '$lib/utils';
import { getBackend } from '$lib/utils/api-base';
interface Props {
displayedModel: string | null;
@@ -15,7 +16,6 @@
// same selectability rule as the form selector: router mode, or any backend
// that exposes a selectable model list
let isSelectable = $derived(serverStore.isRouterMode || !serverStore.capabilities.props);
let canLoadModels = $derived(serverStore.capabilities.loadUnload);
let pendingModel = $state<string | null>(null);
@@ -28,11 +28,14 @@
<ModelsSelectorDropdown
currentModel={pendingModel ?? displayedModel}
disabled={isLoading}
onModelChange={async (modelId: string, modelName: string) => {
onModelChange={async (modelId: string, modelName: string, backendId?: string) => {
// capability of the picked model's own backend, not the active one
const loadsOnRequest = getBackendCapabilities(getBackend(backendId)).loadUnload;
const status = modelsStore.getModelStatus(modelId);
// external backends load the model implicitly on the request
if (canLoadModels && status !== ServerModelStatus.LOADED) {
// only a llama.cpp server loads up front; remote backends load the
// model with the request itself
if (loadsOnRequest && status !== ServerModelStatus.LOADED) {
pendingModel = modelId;
try {
@@ -31,7 +31,11 @@
currentModel?: string | null;
disabled?: boolean;
forceForegroundText?: boolean;
onModelChange?: (modelId: string, modelName: string) => Promise<boolean> | boolean | void;
onModelChange?: (
modelId: string,
modelName: string,
backendId?: string
) => Promise<boolean> | boolean | void;
useGlobalSelection?: boolean;
}
@@ -22,7 +22,11 @@
class?: string;
currentModel?: string | null;
/** Callback when model changes. Return false to keep menu open (e.g., for validation failures) */
onModelChange?: (modelId: string, modelName: string) => Promise<boolean> | boolean | void;
onModelChange?: (
modelId: string,
modelName: string,
backendId?: string
) => Promise<boolean> | boolean | void;
disabled?: boolean;
forceForegroundText?: boolean;
/** When true, user's global selection takes priority over currentModel (for form selector) */
@@ -8,10 +8,12 @@ import { useProcessingState } from './use-processing-state.svelte';
import { colorLevelFromPercent } from '$lib/components/app/chat/ChatForm/ChatFormContextGauge/context-gauge';
import { STATS_UNITS } from '$lib/constants';
import { ColorLevel } from '$lib/enums';
import { contextStatsStore, modelsStore } from '$lib/stores';
import { contextStatsStore, modelsStore, serverStore } from '$lib/stores';
export interface UseContextGaugeReturn {
readonly activeModelId: string | null;
/** Whether the active backend can load the model it is serving. */
readonly canLoadActiveModel: boolean;
readonly isActiveModelLoaded: boolean;
readonly isActiveModelLoading: boolean;
readonly contextTotal: number | null;
@@ -73,6 +75,9 @@ export function useContextGauge(): UseContextGaugeReturn {
contextStatsStore.averageTokensPerSecond !== null ||
transientDetails.length > 0
);
// loading is a llama.cpp router feature, and the gauge tracks the active model,
// so the active backend decides whether a load is possible at all
const canLoadActiveModel = $derived(serverStore.isRouterMode);
async function loadModel() {
const modelId = contextStatsStore.activeModelId;
@@ -93,6 +98,9 @@ export function useContextGauge(): UseContextGaugeReturn {
get averageTokensPerSecond() {
return contextStatsStore.averageTokensPerSecond;
},
get canLoadActiveModel() {
return canLoadActiveModel;
},
get colorLevel() {
return colorLevel;
},
@@ -26,7 +26,11 @@ export interface UseModelsSelectorOptions {
currentModel: () => string | null;
useGlobalSelection?: () => boolean;
onModelChange?: () =>
| ((modelId: string, modelName: string) => Promise<boolean> | boolean | void)
| ((
modelId: string,
modelName: string,
backendId?: string
) => Promise<boolean> | boolean | void)
| undefined;
onOpenChange?: (open: boolean) => void;
}
@@ -266,7 +270,7 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele
let shouldCloseMenu = true;
if (onModelChange) {
const result = await onModelChange(rawModelId(option.id), option.model);
const result = await onModelChange(rawModelId(option.id), option.model, option.backendId);
if (result === false) {
shouldCloseMenu = false;
@@ -285,7 +289,10 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele
});
}
if (!onModelChange && isRouter && !modelsStore.isModelLoaded(option.model)) {
// only the built-in server loads on request, and only in router mode
const canLoadHere = option.backendId === LOCAL_BACKEND_ID && isRouter;
if (!onModelChange && canLoadHere && !modelsStore.isModelLoaded(option.model)) {
isLoadingModel = true;
modelsStore.status
+6 -3
View File
@@ -57,9 +57,12 @@ function normalizeBaseUrl(url: string): string | null {
}
}
/** Features a backend supports, derived from its protocol. */
export function getBackendCapabilities(backend: Backend): BackendCapabilities {
return BACKEND_CAPABILITIES[backend.protocol] ?? BACKEND_CAPABILITIES.openai;
/**
* Features a backend supports, derived from its protocol. A missing backend
* (unknown model, early startup) gets the plain compatible defaults.
*/
export function getBackendCapabilities(backend?: Backend): BackendCapabilities {
return BACKEND_CAPABILITIES[backend?.protocol ?? 'openai'] ?? BACKEND_CAPABILITIES.openai;
}
/** Wire quirks for a backend: protocol defaults overridden by the backend. */