mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 19:07:25 -05:00
ui : load a model only on backends that load it
Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
This commit is contained in:
+1
-1
@@ -73,7 +73,7 @@
|
||||
</span>
|
||||
</div>
|
||||
|
||||
{#if gauge.activeModelId !== null && !gauge.isActiveModelLoaded}
|
||||
{#if gauge.canLoadActiveModel && gauge.activeModelId !== null && !gauge.isActiveModelLoaded}
|
||||
<ContextGaugeLoadModel
|
||||
isLoading={gauge.isActiveModelLoading}
|
||||
modelId={gauge.activeModelId}
|
||||
|
||||
+8
-5
@@ -2,7 +2,8 @@
|
||||
import { ModelBadge, ModelsSelectorDropdown } from '$lib/components/app';
|
||||
import { ServerModelStatus } from '$lib/enums';
|
||||
import { modelsStore, serverStore } from '$lib/stores';
|
||||
import { copyToClipboard } from '$lib/utils';
|
||||
import { copyToClipboard, getBackendCapabilities } from '$lib/utils';
|
||||
import { getBackend } from '$lib/utils/api-base';
|
||||
|
||||
interface Props {
|
||||
displayedModel: string | null;
|
||||
@@ -15,7 +16,6 @@
|
||||
// same selectability rule as the form selector: router mode, or any backend
|
||||
// that exposes a selectable model list
|
||||
let isSelectable = $derived(serverStore.isRouterMode || !serverStore.capabilities.props);
|
||||
let canLoadModels = $derived(serverStore.capabilities.loadUnload);
|
||||
|
||||
let pendingModel = $state<string | null>(null);
|
||||
|
||||
@@ -28,11 +28,14 @@
|
||||
<ModelsSelectorDropdown
|
||||
currentModel={pendingModel ?? displayedModel}
|
||||
disabled={isLoading}
|
||||
onModelChange={async (modelId: string, modelName: string) => {
|
||||
onModelChange={async (modelId: string, modelName: string, backendId?: string) => {
|
||||
// capability of the picked model's own backend, not the active one
|
||||
const loadsOnRequest = getBackendCapabilities(getBackend(backendId)).loadUnload;
|
||||
const status = modelsStore.getModelStatus(modelId);
|
||||
|
||||
// external backends load the model implicitly on the request
|
||||
if (canLoadModels && status !== ServerModelStatus.LOADED) {
|
||||
// only a llama.cpp server loads up front; remote backends load the
|
||||
// model with the request itself
|
||||
if (loadsOnRequest && status !== ServerModelStatus.LOADED) {
|
||||
pendingModel = modelId;
|
||||
|
||||
try {
|
||||
|
||||
@@ -31,7 +31,11 @@
|
||||
currentModel?: string | null;
|
||||
disabled?: boolean;
|
||||
forceForegroundText?: boolean;
|
||||
onModelChange?: (modelId: string, modelName: string) => Promise<boolean> | boolean | void;
|
||||
onModelChange?: (
|
||||
modelId: string,
|
||||
modelName: string,
|
||||
backendId?: string
|
||||
) => Promise<boolean> | boolean | void;
|
||||
useGlobalSelection?: boolean;
|
||||
}
|
||||
|
||||
|
||||
@@ -22,7 +22,11 @@
|
||||
class?: string;
|
||||
currentModel?: string | null;
|
||||
/** Callback when model changes. Return false to keep menu open (e.g., for validation failures) */
|
||||
onModelChange?: (modelId: string, modelName: string) => Promise<boolean> | boolean | void;
|
||||
onModelChange?: (
|
||||
modelId: string,
|
||||
modelName: string,
|
||||
backendId?: string
|
||||
) => Promise<boolean> | boolean | void;
|
||||
disabled?: boolean;
|
||||
forceForegroundText?: boolean;
|
||||
/** When true, user's global selection takes priority over currentModel (for form selector) */
|
||||
|
||||
@@ -8,10 +8,12 @@ import { useProcessingState } from './use-processing-state.svelte';
|
||||
import { colorLevelFromPercent } from '$lib/components/app/chat/ChatForm/ChatFormContextGauge/context-gauge';
|
||||
import { STATS_UNITS } from '$lib/constants';
|
||||
import { ColorLevel } from '$lib/enums';
|
||||
import { contextStatsStore, modelsStore } from '$lib/stores';
|
||||
import { contextStatsStore, modelsStore, serverStore } from '$lib/stores';
|
||||
|
||||
export interface UseContextGaugeReturn {
|
||||
readonly activeModelId: string | null;
|
||||
/** Whether the active backend can load the model it is serving. */
|
||||
readonly canLoadActiveModel: boolean;
|
||||
readonly isActiveModelLoaded: boolean;
|
||||
readonly isActiveModelLoading: boolean;
|
||||
readonly contextTotal: number | null;
|
||||
@@ -73,6 +75,9 @@ export function useContextGauge(): UseContextGaugeReturn {
|
||||
contextStatsStore.averageTokensPerSecond !== null ||
|
||||
transientDetails.length > 0
|
||||
);
|
||||
// loading is a llama.cpp router feature, and the gauge tracks the active model,
|
||||
// so the active backend decides whether a load is possible at all
|
||||
const canLoadActiveModel = $derived(serverStore.isRouterMode);
|
||||
|
||||
async function loadModel() {
|
||||
const modelId = contextStatsStore.activeModelId;
|
||||
@@ -93,6 +98,9 @@ export function useContextGauge(): UseContextGaugeReturn {
|
||||
get averageTokensPerSecond() {
|
||||
return contextStatsStore.averageTokensPerSecond;
|
||||
},
|
||||
get canLoadActiveModel() {
|
||||
return canLoadActiveModel;
|
||||
},
|
||||
get colorLevel() {
|
||||
return colorLevel;
|
||||
},
|
||||
|
||||
@@ -26,7 +26,11 @@ export interface UseModelsSelectorOptions {
|
||||
currentModel: () => string | null;
|
||||
useGlobalSelection?: () => boolean;
|
||||
onModelChange?: () =>
|
||||
| ((modelId: string, modelName: string) => Promise<boolean> | boolean | void)
|
||||
| ((
|
||||
modelId: string,
|
||||
modelName: string,
|
||||
backendId?: string
|
||||
) => Promise<boolean> | boolean | void)
|
||||
| undefined;
|
||||
onOpenChange?: (open: boolean) => void;
|
||||
}
|
||||
@@ -266,7 +270,7 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele
|
||||
let shouldCloseMenu = true;
|
||||
|
||||
if (onModelChange) {
|
||||
const result = await onModelChange(rawModelId(option.id), option.model);
|
||||
const result = await onModelChange(rawModelId(option.id), option.model, option.backendId);
|
||||
|
||||
if (result === false) {
|
||||
shouldCloseMenu = false;
|
||||
@@ -285,7 +289,10 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele
|
||||
});
|
||||
}
|
||||
|
||||
if (!onModelChange && isRouter && !modelsStore.isModelLoaded(option.model)) {
|
||||
// only the built-in server loads on request, and only in router mode
|
||||
const canLoadHere = option.backendId === LOCAL_BACKEND_ID && isRouter;
|
||||
|
||||
if (!onModelChange && canLoadHere && !modelsStore.isModelLoaded(option.model)) {
|
||||
isLoadingModel = true;
|
||||
|
||||
modelsStore.status
|
||||
|
||||
@@ -57,9 +57,12 @@ function normalizeBaseUrl(url: string): string | null {
|
||||
}
|
||||
}
|
||||
|
||||
/** Features a backend supports, derived from its protocol. */
|
||||
export function getBackendCapabilities(backend: Backend): BackendCapabilities {
|
||||
return BACKEND_CAPABILITIES[backend.protocol] ?? BACKEND_CAPABILITIES.openai;
|
||||
/**
|
||||
* Features a backend supports, derived from its protocol. A missing backend
|
||||
* (unknown model, early startup) gets the plain compatible defaults.
|
||||
*/
|
||||
export function getBackendCapabilities(backend?: Backend): BackendCapabilities {
|
||||
return BACKEND_CAPABILITIES[backend?.protocol ?? 'openai'] ?? BACKEND_CAPABILITIES.openai;
|
||||
}
|
||||
|
||||
/** Wire quirks for a backend: protocol defaults overridden by the backend. */
|
||||
|
||||
Reference in New Issue
Block a user