diff --git a/tools/ui/src/lib/components/app/backends/BackendCard.svelte b/tools/ui/src/lib/components/app/backends/BackendCard.svelte index 2067f3f46e..b46bf77772 100644 --- a/tools/ui/src/lib/components/app/backends/BackendCard.svelte +++ b/tools/ui/src/lib/components/app/backends/BackendCard.svelte @@ -7,12 +7,13 @@ import * as Card from '$lib/components/ui/card'; import { Switch } from '$lib/components/ui/switch'; import * as Tooltip from '$lib/components/ui/tooltip'; + import { BackendProtocol } from '$lib/constants'; import { backendsModelsStore } from '$lib/stores/backendsModels.svelte'; - import type { Backend, BackendProtocol } from '$lib/types'; + import type { Backend } from '$lib/types'; const PROTOCOL_LABELS: Record = { - 'llama.cpp': 'Llama-compatible', - openai: 'OpenAI-compatible' + [BackendProtocol.COMPAT]: 'Llama-compatible', + [BackendProtocol.OPENAI]: 'OpenAI-compatible' }; const CARD_ICON_CLASS = 'h-5 w-5'; diff --git a/tools/ui/src/lib/components/app/backends/BackendForm.svelte b/tools/ui/src/lib/components/app/backends/BackendForm.svelte index 28f2114e84..274cb3a00f 100644 --- a/tools/ui/src/lib/components/app/backends/BackendForm.svelte +++ b/tools/ui/src/lib/components/app/backends/BackendForm.svelte @@ -3,12 +3,16 @@ import * as Collapsible from '$lib/components/ui/collapsible'; import { Input } from '$lib/components/ui/input'; import * as Select from '$lib/components/ui/select'; - import { DEFAULT_BACKEND_CHAT_PATH, DEFAULT_BACKEND_MODELS_PATH } from '$lib/constants'; - import type { Backend, BackendProtocol } from '$lib/types'; + import { + BackendProtocol, + DEFAULT_BACKEND_CHAT_PATH, + DEFAULT_BACKEND_MODELS_PATH + } from '$lib/constants'; + import type { Backend } from '$lib/types'; const PROTOCOL_OPTIONS: Array<{ label: string; value: BackendProtocol }> = [ - { label: 'OpenAI-compatible', value: 'openai' }, - { label: 'Llama-compatible (llama-server)', value: 'llama.cpp' } + { label: 'OpenAI-compatible', value: BackendProtocol.OPENAI }, + { label: 'Llama-compatible (llama-server)', value: BackendProtocol.COMPAT } ]; interface Props { @@ -41,13 +45,13 @@ let detectionLabel = $derived.by(() => { if (detecting) return 'Checking what the endpoint speaks...'; - if (detected === 'llama.cpp') { + if (detected === BackendProtocol.COMPAT) { return apiKeyRequired ? 'llama.cpp server detected. It asks for an API key.' : 'llama.cpp server detected.'; } - if (detected === 'openai') return 'OpenAI-compatible endpoint detected.'; + if (detected === BackendProtocol.OPENAI) return 'OpenAI-compatible endpoint detected.'; return null; }); diff --git a/tools/ui/src/lib/components/app/backends/DialogBackendForm.svelte b/tools/ui/src/lib/components/app/backends/DialogBackendForm.svelte index 9cf2b627f3..dff9821c72 100644 --- a/tools/ui/src/lib/components/app/backends/DialogBackendForm.svelte +++ b/tools/ui/src/lib/components/app/backends/DialogBackendForm.svelte @@ -8,13 +8,14 @@ import { BACKEND_ID_PREFIX, BACKEND_PRESETS, + BackendProtocol, DISMISSED_RECOMMENDED_BACKENDS_LOCALSTORAGE_KEY } from '$lib/constants'; import { BooleanString } from '$lib/enums'; import { BackendsService } from '$lib/services'; import type { BackendTestResult } from '$lib/services/backends.service'; import { backendsStore } from '$lib/stores'; - import type { Backend, BackendPreset, BackendProtocol } from '$lib/types'; + import type { Backend, BackendPreset } from '$lib/types'; import { findBackendPreset, uuid } from '$lib/utils'; import { untrack } from 'svelte'; @@ -28,7 +29,7 @@ let { backend = null, - defaultProtocol = 'openai', + defaultProtocol = BackendProtocol.OPENAI, onOpenChange, onSaved, open = $bindable(false) diff --git a/tools/ui/src/lib/components/app/chat/ChatForm/ChatFormActions/ChatFormActionModels.svelte b/tools/ui/src/lib/components/app/chat/ChatForm/ChatFormActions/ChatFormActionModels.svelte index a4baa40bf9..434db0cf4f 100644 --- a/tools/ui/src/lib/components/app/chat/ChatForm/ChatFormActions/ChatFormActionModels.svelte +++ b/tools/ui/src/lib/components/app/chat/ChatForm/ChatFormActions/ChatFormActionModels.svelte @@ -1,6 +1,14 @@ {#if hasError || isLoadingModel} diff --git a/tools/ui/src/lib/components/app/dialogs/DialogManageModels.svelte b/tools/ui/src/lib/components/app/dialogs/DialogManageModels.svelte index a466ad0533..be67178fbd 100644 --- a/tools/ui/src/lib/components/app/dialogs/DialogManageModels.svelte +++ b/tools/ui/src/lib/components/app/dialogs/DialogManageModels.svelte @@ -5,7 +5,7 @@ import ModelsManagerModelProviders from '$lib/components/app/models/ModelsManager/ModelsManagerModelProviders.svelte'; import { Button } from '$lib/components/ui/button'; import * as Dialog from '$lib/components/ui/dialog'; - import { MODEL_ICON } from '$lib/constants'; + import { MODEL_ICON, MODELS_DIALOG_VIEW, type ModelsDialogView } from '$lib/constants'; import { uiStore } from '$lib/stores'; import { untrack } from 'svelte'; @@ -16,15 +16,13 @@ let { onOpenChange, open = $bindable(false) }: Props = $props(); - type View = 'discover' | 'manage' | 'providers'; - /** How long a view takes to fade out before the next one fades in. */ const VIEW_FADE_MS = 150; // what the user asked for, and what is actually rendered: a view change fades the // current one out, swaps, then fades the next one in - let view = $state('manage'); - let shownView = $state('manage'); + let view = $state(MODELS_DIALOG_VIEW.MANAGE); + let shownView = $state(MODELS_DIALOG_VIEW.MANAGE); let isSwapping = $state(false); $effect(() => { @@ -52,7 +50,11 @@ }); let title = $derived( - view === 'discover' ? 'Discover' : view === 'providers' ? 'Providers' : 'Models' + view === MODELS_DIALOG_VIEW.DISCOVER + ? 'Discover' + : view === MODELS_DIALOG_VIEW.PROVIDERS + ? 'Providers' + : 'Models' ); // the sidebar's Discover entry opens this dialog on its Discover view @@ -61,7 +63,7 @@ untrack(() => { uiStore.discoverModelsOpen = false; - view = 'discover'; + view = MODELS_DIALOG_VIEW.DISCOVER; handleOpenChange(true); }); }); @@ -87,10 +89,10 @@ {/snippet} - {:else if shownView === 'discover'} + {:else if shownView === MODELS_DIALOG_VIEW.DISCOVER}
diff --git a/tools/ui/src/lib/components/app/models/ModelAvatar.svelte b/tools/ui/src/lib/components/app/models/ModelAvatar.svelte index 13c0931672..f352bed0ab 100644 --- a/tools/ui/src/lib/components/app/models/ModelAvatar.svelte +++ b/tools/ui/src/lib/components/app/models/ModelAvatar.svelte @@ -1,9 +1,13 @@ -{#if orgName && hubEnabled} +{#if useProviderIcon} + + + {#snippet fallback()} + {#if isLocal} + + {:else} + + {/if} + {/snippet} + + +{:else if orgName && hubEnabled} (isNearViewport = true)} class={['inline-flex shrink-0', className]} diff --git a/tools/ui/src/lib/components/app/models/ModelCapabilities.svelte b/tools/ui/src/lib/components/app/models/ModelCapabilities.svelte index c854a8be51..e5040081c0 100644 --- a/tools/ui/src/lib/components/app/models/ModelCapabilities.svelte +++ b/tools/ui/src/lib/components/app/models/ModelCapabilities.svelte @@ -1,5 +1,6 @@
- {#if isLoading} + {#if !canLoad} + {#if showBackendMark} + + {:else if showRemoteMark} + + + + {#if isBackendFailed} + + {/if} + + {/if} + {:else if isLoading} {:else} diff --git a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManager.svelte b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManager.svelte index 148fb839d8..d5f65c3245 100644 --- a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManager.svelte +++ b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManager.svelte @@ -3,22 +3,31 @@ import ModelsManagerModelsTable from './ModelsManagerModelsTable.svelte'; import { groupModelQuants, + loadExtraArgs, + loadOverrides, modelContextLength, + modelDraftBadges, + type ModelOverride, type ModelQuantGroup, type ModelsTableGroup, - modelSupports + modelSupports, + saveOverrides } from './utils'; import { LOCAL_BACKEND_ID, type ModalityKey, MODELS_TABLE_GROUP_LABELS, - ModelsTableGroupKind + ModelsTableGroupKind, + ModelsTableProviderKind } from '$lib/constants'; import { ModelCapability } from '$lib/enums'; - import { conversationsStore, modelsStore, uiStore } from '$lib/stores'; + import { backendsStore, conversationsStore, modelsStore, uiStore } from '$lib/stores'; import type { ModelOption } from '$lib/types/models'; + import { getBackend } from '$lib/utils/api-base'; + import { getBackendCapabilities } from '$lib/utils/backend'; import { type Snippet, untrack } from 'svelte'; import { SvelteMap, SvelteSet } from 'svelte/reactivity'; + import { toast } from 'svelte-sonner'; interface Props { class?: string; @@ -29,10 +38,13 @@ let { class: className, toolbarEnd }: Props = $props(); let filter = $state(''); + let providerFilter = $state([]); let contextLimit = $state(0); let modalityFilter = $state([]); let capabilityFilter = $state([]); + let draftFilter = $state(false); let selectedId = $state(null); + let overrides = $state>(loadOverrides()); let allModels = $derived(modelsStore.models); @@ -100,6 +112,8 @@ let isFavorite = $derived((option: ModelOption) => modelsStore.favoriteModelIds.has(option.model) ); + // every filter but the provider one, so a provider count does not fall to zero + // the moment that provider is the one being looked at let matching = $derived.by(() => { const term = filter.trim().toLowerCase(); @@ -115,6 +129,13 @@ return false; } + if ( + draftFilter && + modelDraftBadges(option, overrides[option.id]?.load?.speculativeDecoding).length === 0 + ) { + return false; + } + // a model whose modalities are unknown cannot be shown to match if (modalityFilter.length > 0 && !modalityFilter.some((key) => option.modalities?.[key])) { return false; @@ -123,6 +144,24 @@ return contextLimit === 0 || modelContextLength(option) >= contextLimit; }); }); + let visible = $derived.by(() => + providerFilter.length === 0 + ? matching + : matching.filter((option) => providerFilter.includes(option.backendId ?? LOCAL_BACKEND_ID)) + ); + // the rail counts follow the active view and filter, so it always says how many + // repos each provider contributes to what the table is showing + let providerCounts = $derived.by(() => { + const counts: Record = {}; + + for (const entry of groupModelQuants(matching)) { + const backendId = entry.base.backendId ?? LOCAL_BACKEND_ID; + + counts[backendId] = (counts[backendId] ?? 0) + 1; + } + + return counts; + }); // recently used models lead their section, the rest keep the server's order let rank = $derived.by(() => { @@ -137,11 +176,15 @@ Math.min(...entry.quants.map((quant) => rank.get(quant.id) ?? Number.MAX_SAFE_INTEGER)); const byRecency = (list: ModelQuantGroup[]) => rank.size === 0 ? list : [...list].sort((a, b) => rankOf(a) - rankOf(b)); - // one entry per repo, so a model with several quants takes a single table row - let entries = $derived(byRecency(groupModelQuants(matching))); + // one entry per repo, so a model with several quants takes a single table row; + // loaded models lead the table, then favorites, then one block per provider + let entries = $derived(byRecency(groupModelQuants(visible))); let groups = $derived.by(() => { - // a loaded quant is a model of its own: its repo keeps the quants left behind - const isLoaded = (option: ModelOption) => modelsStore.isModelLoaded(option.model); + // a loaded quant is a model of its own: its repo keeps the quants left behind. + // Only llama-compat servers report a load state. + const isLoaded = (option: ModelOption) => + getBackendCapabilities(getBackend(option.backendId)).loadUnload && + modelsStore.isModelLoaded(option.model); const loaded: ModelQuantGroup[] = []; const rest: ModelQuantGroup[] = []; @@ -168,23 +211,58 @@ (entry) => !claimed.has(entry.key) && entry.quants.some((q) => modelsStore.isHidden(q.id)) ); const hiddenKeys = new SvelteSet(hidden.map((entry) => entry.key)); - const local = rest.filter((entry) => !claimed.has(entry.key) && !hiddenKeys.has(entry.key)); + const byBackend = new SvelteMap(); + + for (const entry of rest) { + if (claimed.has(entry.key) || hiddenKeys.has(entry.key)) continue; + + const backendId = entry.base.backendId ?? LOCAL_BACKEND_ID; + + if (!byBackend.has(backendId)) byBackend.set(backendId, []); + + byBackend.get(backendId)!.push(entry); + } + const ordered: ModelsTableGroup[] = []; - // loaded models lead the table, then favorites, then the local block - const pushSection = (kind: ModelsTableGroupKind, items: ModelQuantGroup[]): void => { + // loaded models lead the table, then favorites, then one block per backend + const pushSection = ( + kind: ModelsTableGroup['kind'], + items: ModelQuantGroup[], + provider?: { id: string; label: string } + ): void => { if (items.length === 0) return; + const isLocal = kind === ModelsTableGroupKind.LOCAL; + ordered.push({ + backendId: provider?.id ?? (isLocal ? LOCAL_BACKEND_ID : null), + isLocal, items, - key: kind === ModelsTableGroupKind.LOCAL ? LOCAL_BACKEND_ID : kind, + key: provider?.id ?? (isLocal ? LOCAL_BACKEND_ID : kind), kind, - label: MODELS_TABLE_GROUP_LABELS[kind] + label: provider?.label ?? MODELS_TABLE_GROUP_LABELS[kind as ModelsTableGroupKind] }); }; pushSection(ModelsTableGroupKind.LOADED, loaded); pushSection(ModelsTableGroupKind.FAVORITES, favorites); - pushSection(ModelsTableGroupKind.LOCAL, local); + + const localItems = byBackend.get(LOCAL_BACKEND_ID); + + if (localItems?.length) pushSection(ModelsTableGroupKind.LOCAL, localItems); + + for (const backend of backendsStore.enabled) { + if (backend.id === LOCAL_BACKEND_ID) continue; + + const items = byBackend.get(backend.id); + + if (items?.length) + pushSection(ModelsTableProviderKind.PROVIDER, items, { + id: backend.id, + label: backend.name + }); + } + pushSection(ModelsTableGroupKind.HIDDEN, hidden); return ordered; @@ -209,7 +287,7 @@ return; } - await modelsStore.status.load(option.model); + await modelsStore.status.load(option.model, loadExtraArgs(overrides[option.id])); } async function useInNewChat(option: ModelOption): Promise { @@ -218,6 +296,24 @@ // the chat is behind the dialog, so it takes focus once the dialog is out of the way uiStore.requestComposerFocus(); } + + /** Point the selected model's load settings at another model as its draft. */ + function useAsDraft(draft: ModelOption, targetId: string): void { + const target = modelsStore.models.find((option) => option.id === targetId); + + if (!target) return; + + saveOverride(target, { + ...overrides[target.id], + load: { ...overrides[target.id]?.load, speculativeDecoding: draft.id } + }); + } + + function saveOverride(option: ModelOption, override: ModelOverride): void { + overrides = { ...overrides, [option.id]: override }; + saveOverrides(overrides); + toast.success(`Saved settings for ${option.name}`); + }
@@ -225,11 +321,16 @@ (selectedId = option.id)} + onUseAsDraft={useAsDraft} + {overrides} + {providerCounts} {selectedId} {toolbarEnd} /> diff --git a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerFilters.svelte b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerFilters.svelte index be3cbe6f4a..743885fd6f 100644 --- a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerFilters.svelte +++ b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerFilters.svelte @@ -1,12 +1,17 @@ +{#snippet providerMark(backend: Backend)} + {#if backend.id === LOCAL_BACKEND_ID} + + {#snippet fallback()} + + {/snippet} + + {:else} + + {/if} +{/snippet} + + {#if backends.length > 1} + + + {#snippet child({ props })} + + {/snippet} + + + + + Providers + + {#each backends as backend (backend.id)} + toggleProvider(backend.id, checked)} + > + {@render providerMark(backend)} + + {backend.name} + + {providerCounts[backend.id] ?? 0} + + {/each} + + + {#if providers.length > 0} + + + (providers = [])} + >Show every provider + {/if} + + + {/if} + (contextLimit = Number(value))} type="single" @@ -86,6 +174,26 @@ + + + + + Has draft sidecar + + void backendsModelsStore.loadAll()} /> diff --git a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerModelRow.svelte b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerModelRow.svelte index e23362e2d3..6cc8848021 100644 --- a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerModelRow.svelte +++ b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerModelRow.svelte @@ -4,8 +4,8 @@ import ModelContext from '../ModelContext.svelte'; import ModelId from '../ModelId.svelte'; import ModelsManagerStatusCell from './ModelsManagerStatusCell.svelte'; - import { modelRowActions } from './row-actions'; - import { configuredContext } from './utils'; + import { modelRowActions, type ModelRowDraftTarget } from './row-actions'; + import { canLoadOption, configuredContext, modelDraftBadges, type ModelOverride } from './utils'; import { MoreHorizontal } from '@lucide/svelte'; import { DropdownMenuActions } from '$lib/components/app'; import { MODEL_ROW_GRID_CLASS } from '$lib/constants'; @@ -14,19 +14,37 @@ import type { ModelOption } from '$lib/types/models'; interface Props { + /** Model the pane has open, when this row can be set as its draft. */ + draftTarget?: ModelRowDraftTarget | null; isFavorite: (option: ModelOption) => boolean; + /** Stored per-model overrides, for the drafts and context the row reports. */ + overrides?: Record; option: ModelOption; onDelete: (option: ModelOption) => void; onSelect: (option: ModelOption) => void; + onUseAsDraft?: (draft: ModelOption, targetId: string) => void; selected: boolean; /** Left padding in px, from the nesting depth. */ indent?: number; } - let { indent = 0, isFavorite, onDelete, onSelect, option, selected }: Props = $props(); + let { + draftTarget = null, + indent = 0, + isFavorite, + onDelete, + onSelect, + onUseAsDraft, + option, + overrides, + selected + }: Props = $props(); let favorite = $derived(isFavorite(option)); let isHidden = $derived(modelsStore.isHidden(option.id)); + let draftBadges = $derived( + modelDraftBadges(option, overrides?.[option.id]?.load?.speculativeDecoding) + ); function handleKeydown(event: KeyboardEvent): void { if (event.key === KeyboardKey.SPACE) event.preventDefault(); @@ -59,7 +77,7 @@ - +
boolean; onSelect: (option: ModelOption) => void; /** Modalities a model must support at least one of. */ modalities?: ModalityKey[]; + /** Per-model load and inference overrides, keyed by backend-qualified id. */ + overrides: Record; + /** Backend ids to keep; empty keeps every provider. */ + providers?: string[]; + /** Repos each provider contributes to the current search, for the filter menu. */ + providerCounts?: Record; + /** Called when a row is set as the draft of the selected model. */ + onUseAsDraft?: (draft: ModelOption, targetId: string) => void; selectedId: string | null; /** Rendered at the toolbar's right end, past the filters. */ toolbarEnd?: Snippet; @@ -64,17 +77,35 @@ let { capabilities = $bindable([]), contextLimit = $bindable(0), + draft = $bindable(false), filter = $bindable(''), groups, isFavorite, modalities = $bindable([]), onSelect, + onUseAsDraft, + overrides, + providerCounts = {}, + providers = $bindable([]), selectedId, toolbarEnd }: Props = $props(); let isEmpty = $derived(groups.every((group) => group.items.length === 0)); - let hasFilters = $derived(contextLimit > 0 || modalities.length > 0 || capabilities.length > 0); + let hasFilters = $derived( + providers.length > 0 || + contextLimit > 0 || + modalities.length > 0 || + capabilities.length > 0 || + draft + ); + + /** Model the configuration pane has open, when a row can be set as its draft. */ + let draftTarget = $derived.by(() => { + const selected = modelsStore.models.find((option) => option.id === selectedId); + + return selected ? { id: selected.id, label: selected.name } : null; + }); /** In-flight and paused downloads, tracked by the status feed. */ let downloadEntries = $derived(modelsStore.status.getDownloadEntries()); @@ -226,27 +257,35 @@ expanded={!collapsedQuants.has(entry.key)} {indent} onToggle={() => toggleQuants(entry.key)} + {overrides} /> {#each entry.quants as quant (quant.id)} {/each} {:else} {/if} @@ -343,8 +382,11 @@ @@ -391,12 +433,19 @@ {/if} {/snippet} + {@const backendState = group.backendId ? backendsModelsStore.get(group.backendId) : null} + boolean; + /** Stored per-model overrides, for the drafts and context the row reports. */ + overrides?: Record; option: ModelOption; onDelete: (option: ModelOption) => void; onSelect: (option: ModelOption) => void; + onUseAsDraft?: (draft: ModelOption, targetId: string) => void; selected: boolean; + /** Badge the backend name instead of the quant, for a provider group. */ + showProvider?: boolean; /** Left padding in px, from the nesting depth. */ indent?: number; } - let { indent = 0, isFavorite, onDelete, onSelect, option, selected }: Props = $props(); + let { + draftTarget = null, + indent = 0, + isFavorite, + onDelete, + onSelect, + onUseAsDraft, + option, + overrides, + selected, + showProvider = false + }: Props = $props(); let favorite = $derived(isFavorite(option)); let isHidden = $derived(modelsStore.isHidden(option.id)); @@ -48,20 +67,30 @@ tabindex="0" > - {quant} + + {showProvider ? (getBackend(option.backendId)?.name ?? quant) : quant} + - + {option.model} - +
void; + /** Stored per-model overrides, for the drafts and context the row reports. */ + overrides?: Record; /** Left padding in px, from the nesting depth. */ indent?: number; } - let { entry, expanded, indent = 0, onToggle }: Props = $props(); + let { entry, expanded, indent = 0, onToggle, overrides }: Props = $props(); + let providerCount = $derived(new Set(entry.quants.map((option) => option.backendId ?? '')).size); let groupLabel = $derived( - entry.kind === 'variants' - ? `${entry.quants.length} variants` - : `${entry.quants.length} quants available` + entry.kind === 'providers' + ? `${providerCount} provider${providerCount === 1 ? '' : 's'}` + : entry.kind === 'variants' + ? `${entry.quants.length} variants` + : `${entry.quants.length} quants available` ); let anyLoaded = $derived(entry.quants.some((quant) => isModelRunning(quant))); // a repo row stands for its quants, so it reports what they agree on @@ -78,7 +83,7 @@ diff --git a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerStatusCell.svelte b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerStatusCell.svelte index 9f3d2873a4..004e4950c1 100644 --- a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerStatusCell.svelte +++ b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerStatusCell.svelte @@ -1,5 +1,6 @@ diff --git a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerTableToolbar.svelte b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerTableToolbar.svelte index 2fd4403846..f44a8be923 100644 --- a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerTableToolbar.svelte +++ b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerTableToolbar.svelte @@ -5,7 +5,7 @@ import { Button } from '$lib/components/ui/button'; import { type ModalityKey } from '$lib/constants'; import { ModelCapability } from '$lib/enums'; - import { uiStore } from '$lib/stores'; + import { backendsStore, uiStore } from '$lib/stores'; import type { Snippet } from 'svelte'; interface Props { @@ -13,9 +13,15 @@ capabilities?: ModelCapability[]; /** Smallest context a model must support; 0 keeps every model. */ contextLimit?: number; + /** Keep only models that have a draft sidecar to speculate with. */ + draft?: boolean; filter?: string; /** Modalities a model must support at least one of. */ modalities?: ModalityKey[]; + /** Repos each provider contributes to the current search, for the filter menu. */ + providerCounts?: Record; + /** Backend ids to keep; empty keeps every provider. */ + providers?: string[]; /** Rendered at the toolbar's right end, past the filters. */ toolbarEnd?: Snippet; } @@ -23,12 +29,21 @@ let { capabilities = $bindable([]), contextLimit = $bindable(0), + draft = $bindable(false), filter = $bindable(''), modalities = $bindable([]), + providerCounts = {}, + providers = $bindable([]), toolbarEnd }: Props = $props(); - let hasFilters = $derived(contextLimit > 0 || modalities.length > 0 || capabilities.length > 0); + let hasFilters = $derived( + providers.length > 0 || + contextLimit > 0 || + modalities.length > 0 || + capabilities.length > 0 || + draft + ); let filterInput = $state(null); // the dialog hands focus to its first control, so the filter takes it instead @@ -59,15 +74,25 @@ size="sm" /> - + {#if hasFilters} {:else} {@const selectedOption = ms.getDisplayOption()} {@const triggerModel = selectedOption?.model} @@ -91,13 +113,15 @@
+ + void ms.showBackendModels(backend.id)} +/> diff --git a/tools/ui/src/lib/components/app/navigation/utils.ts b/tools/ui/src/lib/components/app/navigation/utils.ts index b62f427f3d..6987114714 100644 --- a/tools/ui/src/lib/components/app/navigation/utils.ts +++ b/tools/ui/src/lib/components/app/navigation/utils.ts @@ -12,9 +12,25 @@ export interface OrgGroup { items: ModelItem[]; } +/** One remote backend's section on the remote view. */ +export interface ProviderGroup { + backendId: string; + /** Models the provider lists without any search filtering. */ + catalog: number; + error: string | null; + /** Rows to render, capped by the display limit. */ + items: ModelItem[]; + loading: boolean; + /** Rows left after the search filter, before the cap. */ + matched: number; + name: string; +} + export interface GroupedModelOptions { available: OrgGroup[]; loaded: ModelItem[]; + /** Remote backends, one section each. */ + providers: ProviderGroup[]; } function matchesModality(option: ModelOption, term: string): boolean { @@ -133,5 +149,48 @@ export function groupModelOptions( available.push({ items, orgName: orgName || null }); } - return { available, loaded }; + return { available, loaded, providers: [] }; +} + +/** + * Remote backends as sections, one per backend, in the given order. Each section + * keeps at most `limit` rows; `matched` carries the full count so the caller can + * offer the rest. + */ +export function groupProviderOptions( + options: ModelOption[], + providers: { + backendId: string; + catalog: number; + error: string | null; + loading: boolean; + name: string; + }[], + limit = Infinity, + recentIds: readonly string[] = [] +): ProviderGroup[] { + const byBackend = new SvelteMap(); + const rank = new SvelteMap(); + + recentIds.forEach((id, index) => rank.set(id, index)); + + const rankOf = (id: string) => rank.get(id) ?? Number.MAX_SAFE_INTEGER; + // recently used models lead their section, the rest keep the backend's order + const byRecency = (items: ModelItem[]) => + rank.size === 0 ? items : [...items].sort((a, b) => rankOf(a.option.id) - rankOf(b.option.id)); + + for (let i = 0; i < options.length; i++) { + const option = options[i]; + const backendId = option.backendId ?? ''; + + if (!byBackend.has(backendId)) byBackend.set(backendId, []); + + byBackend.get(backendId)!.push({ flatIndex: i, option }); + } + + return providers.map((provider) => { + const items = byRecency(byBackend.get(provider.backendId) ?? []); + + return { ...provider, items: items.slice(0, limit), matched: items.length }; + }); } diff --git a/tools/ui/src/lib/constants/backend.constants.ts b/tools/ui/src/lib/constants/backend.constants.ts index 100ee17dcd..bd752ead00 100644 --- a/tools/ui/src/lib/constants/backend.constants.ts +++ b/tools/ui/src/lib/constants/backend.constants.ts @@ -1,15 +1,29 @@ -import type { - BackendCapabilities, - BackendCompat, - BackendPreset, - BackendProtocol -} from '$lib/types'; +import type { BackendCapabilities, BackendCompat, BackendPreset } from '$lib/types'; + +/** Request/response shape a backend speaks. */ +export const BackendProtocol = { + COMPAT: 'llama.cpp', + OPENAI: 'openai' +} as const; + +export type BackendProtocol = (typeof BackendProtocol)[keyof typeof BackendProtocol]; + +/** Field carrying the output token cap, per OpenAI dialect. */ +export const MaxTokensField = { + CHAT_COMPLETION: 'max_completion_tokens', + COMPLETION: 'max_tokens' +} as const; + +export type MaxTokensField = (typeof MaxTokensField)[keyof typeof MaxTokensField]; /** Prefix for generated ids of user-added backends. */ export const BACKEND_ID_PREFIX = 'backend'; /** Protocols a configured backend can speak, in display order. */ -export const BACKEND_PROTOCOLS: readonly BackendProtocol[] = ['llama.cpp', 'openai']; +export const BACKEND_PROTOCOLS: readonly BackendProtocol[] = [ + BackendProtocol.COMPAT, + BackendProtocol.OPENAI +]; /** Chat completions path used when a backend does not override it. */ export const DEFAULT_BACKEND_CHAT_PATH = '/v1/chat/completions'; @@ -45,15 +59,21 @@ const COMPATIBLE_CAPABILITIES: BackendCapabilities = { /** Capabilities per backend protocol. */ export const BACKEND_CAPABILITIES: Record = { - 'llama.cpp': LLAMA_CPP_CAPABILITIES, - openai: COMPATIBLE_CAPABILITIES + [BackendProtocol.COMPAT]: LLAMA_CPP_CAPABILITIES, + [BackendProtocol.OPENAI]: COMPATIBLE_CAPABILITIES }; /** Default wire quirks per protocol. */ export const BACKEND_COMPAT: Record = { // llama-server reports its own timings, so it needs no usage chunk - 'llama.cpp': { maxTokensField: 'max_tokens', supportsUsageInStreaming: false }, - openai: { maxTokensField: 'max_tokens', supportsUsageInStreaming: true } + [BackendProtocol.COMPAT]: { + maxTokensField: MaxTokensField.COMPLETION, + supportsUsageInStreaming: false + }, + [BackendProtocol.OPENAI]: { + maxTokensField: MaxTokensField.COMPLETION, + supportsUsageInStreaming: true + } }; /** @@ -84,7 +104,7 @@ export const BACKEND_PRESETS: readonly BackendPreset[] = [ iconUrl: '/backend-presets/huggingface.svg', id: 'huggingface', name: 'Hugging Face', - protocol: 'openai' + protocol: BackendProtocol.OPENAI }, { baseUrl: 'https://openrouter.ai/api', diff --git a/tools/ui/src/lib/constants/models-manager.constants.ts b/tools/ui/src/lib/constants/models-manager.constants.ts index 97fd528542..9537ae3af1 100644 --- a/tools/ui/src/lib/constants/models-manager.constants.ts +++ b/tools/ui/src/lib/constants/models-manager.constants.ts @@ -2,6 +2,7 @@ /** What a repo group of the table folds, which decides its label. */ export const ModelGroupKind = { + PROVIDERS: 'providers', QUANTS: 'quants', VARIANTS: 'variants' } as const; @@ -18,6 +19,15 @@ export const ModelsTableGroupKind = { export type ModelsTableGroupKind = (typeof ModelsTableGroupKind)[keyof typeof ModelsTableGroupKind]; +/** Kinds a provider block carries on top of the manager's own sections. */ +export const ModelsTableProviderKind = { + COMPAT: 'compat', + PROVIDER: 'provider' +} as const; + +export type ModelsTableProviderKind = + (typeof ModelsTableProviderKind)[keyof typeof ModelsTableProviderKind]; + /** Header label of each manager section. */ export const MODELS_TABLE_GROUP_LABELS: Record = { [ModelsTableGroupKind.FAVORITES]: 'Favorites', @@ -34,3 +44,12 @@ export const ModelsTableSortKey = { } as const; export type ModelsTableSortKey = (typeof ModelsTableSortKey)[keyof typeof ModelsTableSortKey]; + +/** Panel the models dialog shows. */ +export const MODELS_DIALOG_VIEW = { + DISCOVER: 'discover', + MANAGE: 'manage', + PROVIDERS: 'providers' +} as const; + +export type ModelsDialogView = (typeof MODELS_DIALOG_VIEW)[keyof typeof MODELS_DIALOG_VIEW]; diff --git a/tools/ui/src/lib/hooks/use-models-selector.svelte.ts b/tools/ui/src/lib/hooks/use-models-selector.svelte.ts index b8863f3cf3..19daacce94 100644 --- a/tools/ui/src/lib/hooks/use-models-selector.svelte.ts +++ b/tools/ui/src/lib/hooks/use-models-selector.svelte.ts @@ -2,19 +2,34 @@ import type { ModelItem } from '$lib/components/app/navigation/utils'; import { filterModelOptions, groupFavoriteOptions, - groupModelOptions + groupModelOptions, + groupProviderOptions } from '$lib/components/app/navigation/utils'; -import { CHAT_INPUT_FOCUS_SELECTOR } from '$lib/constants'; -import { modelsStore, serverStore, uiStore } from '$lib/stores'; +import { + CHAT_INPUT_FOCUS_SELECTOR, + LOCAL_BACKEND_ID, + REMOTE_PROVIDER_MODEL_LIMIT +} from '$lib/constants'; +import { backendsModelsStore, backendsStore, modelsStore, serverStore, uiStore } from '$lib/stores'; import type { ModelOption } from '$lib/types/models'; +import { getBackend } from '$lib/utils/api-base'; +import { getBackendCapabilities } from '$lib/utils/backend'; +import { rawModelId } from '$lib/utils/model-option-id'; import { onMount } from 'svelte'; import { SvelteSet } from 'svelte/reactivity'; +/** Groups of the favorites tab, which lists favorites only. */ +const EMPTY_GROUPS = { available: [], loaded: [], providers: [] }; + export interface UseModelsSelectorOptions { currentModel: () => string | null; useGlobalSelection?: () => boolean; onModelChange?: () => - | ((modelId: string, modelName: string) => Promise | boolean | void) + | (( + modelId: string, + modelName: string, + backendId?: string + ) => Promise | boolean | void) | undefined; onOpenChange?: (open: boolean) => void; } @@ -34,10 +49,14 @@ export interface UseModelsSelectorReturn { readonly loadedItems: ModelItem[]; readonly filteredOptions: ModelOption[]; readonly isEmpty: boolean; + readonly isProviderView: boolean; readonly groupedFilteredOptions: ReturnType; readonly isLoadingModel: boolean; readonly searchTerm: string; + closeProvider(): void; + openProvider(backendId: string): void; setSearchTerm(value: string): void; + showBackendModels(backendId: string): Promise; handleSelect(modelId: string): Promise; handleOpenChange(open: boolean): void; isFavorite(model: string): boolean; @@ -52,21 +71,36 @@ export interface UseModelsSelectorReturn { * duplicating store derivations, selection handling, and model loading. */ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSelectorReturn { - let isLoadingModel = $state(false); - let searchTerm = $state(''); + /** + * Current view: the favorites of every backend, the local server's models, or + * the remote backends'. Favorites are the default while there is at least one. + */ + /** Remote backend drilled into from its section; null while browsing. */ + let providerViewId = $state(null); - const options = $derived( + const isProviderView = $derived(providerViewId !== null); + const isLocalOption = (option: ModelOption) => option.backendId === LOCAL_BACKEND_ID; + // every enabled backend's models are one list: favorites, then the local + // server, then one section per remote provider + const allOptions = $derived( modelsStore.models.filter((option) => { const modelProps = modelsStore.props.getModelProps(option.model); return modelProps?.ui !== false; }) ); + const options = $derived( + providerViewId ? allOptions.filter((option) => option.backendId === providerViewId) : allOptions + ); const loading = $derived(modelsStore.loading); const updating = $derived(modelsStore.updating); const activeId = $derived(modelsStore.selectedModelId); - // a lone llama.cpp server without a router has nothing to choose from - const isMultiModel = $derived(serverStore.isRouterMode); + // Router mode and external backends both expose a selectable model list; only + // a lone llama.cpp server without a router has nothing to choose from. + const isMultiModel = $derived( + serverStore.isRouterMode || + backendsStore.enabled.some((backend) => backend.id !== LOCAL_BACKEND_ID) + ); const isRouter = $derived(serverStore.isRouterMode); const serverModel = $derived(modelsStore.singleModelName); const currentModel = $derived(opts.currentModel()); @@ -74,28 +108,54 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele const isHighlightedCurrentModelActive = $derived.by(() => { if (!isRouter || !currentModel) return false; - const currentOption = options.find((option) => option.model === currentModel); + const currentOption = allOptions.find((option) => option.model === currentModel); return currentOption ? currentOption.id === activeId : false; }); const isCurrentModelInCache = $derived.by(() => { if (!isRouter || !currentModel) return true; - return options.some((option) => option.model === currentModel); + return allOptions.some((option) => option.model === currentModel); }); - const visibleOptions = $derived(options.filter((option) => !modelsStore.isHidden(option.id))); + + let isLoadingModel = $state(false); + let searchTerm = $state(''); + + const visibleOptions = $derived(allOptions.filter((option) => !modelsStore.isHidden(option.id))); const filteredOptions = $derived(filterModelOptions(options, searchTerm)); + // favorites span every backend, so they come from the full option list const favoriteItems = $derived( groupFavoriteOptions( filterModelOptions(visibleOptions, searchTerm), modelsStore.favoriteModelIds ) ); - const loadedItems = $derived( - filterModelOptions(visibleOptions, searchTerm) - .map((option, flatIndex) => ({ flatIndex, option })) - .filter(({ option }) => modelsStore.isModelLoaded(option.model)) + const remoteProviders = $derived( + backendsStore.enabled + .filter((backend) => backend.id !== LOCAL_BACKEND_ID) + .map((backend) => { + const state = backendsModelsStore.get(backend.id); + + return { + backendId: backend.id, + catalog: state.models.length, + error: state.error, + loading: state.loading, + name: backend.name + }; + }) ); + // loaded models lead the list, from any llama-compat backend + const isLoadedLlamaCompat = (option: ModelOption) => + modelsStore.isModelLoaded(option.model) && + getBackendCapabilities(getBackend(option.backendId)).loadUnload; + const loadedItems = $derived.by(() => { + if (isProviderView) return []; + + return filterModelOptions(visibleOptions, searchTerm) + .map((option, flatIndex) => ({ flatIndex, option })) + .filter(({ option }) => isLoadedLlamaCompat(option)); + }); const loadedIds = $derived(new SvelteSet(loadedItems.map((item) => item.option.id))); // loaded models and favorites are listed once, at the top: the sections skip both const sectionOptions = $derived( @@ -103,9 +163,28 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele (option) => !modelsStore.favoriteModelIds.has(option.model) && !loadedIds.has(option.id) ) ); - const groupedFilteredOptions = $derived( - groupModelOptions(sectionOptions, (model) => modelsStore.isModelLoaded(model)) + const providerSections = $derived( + groupProviderOptions( + sectionOptions, + remoteProviders, + // a drill-in or a search reaches every model, the sections stay short + providerViewId || searchTerm ? Infinity : REMOTE_PROVIDER_MODEL_LIMIT, + modelsStore.recentModelIds + ) ); + const groupedFilteredOptions = $derived.by(() => { + if (isProviderView) { + const sections = providerSections.filter((section) => section.backendId === providerViewId); + + return { ...EMPTY_GROUPS, providers: sections }; + } + + const local = groupModelOptions(sectionOptions.filter(isLocalOption), (m) => + modelsStore.isModelLoaded(m) + ); + + return { ...local, providers: providerSections }; + }); const isEmpty = $derived( filteredOptions.length === 0 && favoriteItems.length === 0 && loadedItems.length === 0 ); @@ -135,6 +214,7 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele } searchTerm = ''; + providerViewId = null; if (open && isRouter) { modelsStore.props.fetchModalitiesForLoadedModels(); @@ -143,15 +223,37 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele opts.onOpenChange?.(open); } + /** + * Switch the rendered view. Views are display only: the backend that serves + * requests follows the selected model, not the view. + */ + /** Drill into one remote backend's full model list. */ + function openProvider(backendId: string) { + providerViewId = backendId; + searchTerm = ''; + } + + function closeProvider() { + providerViewId = null; + searchTerm = ''; + } + + /** Refresh a backend's models, e.g. right after it was added. */ + async function showBackendModels(backendId: string): Promise { + await backendsModelsStore.ensureLoaded(backendId); + } + async function handleSelect(modelId: string) { - const option = options.find((opt) => opt.id === modelId); + // favorites live above the tabs and may belong to another backend, so the + // lookup spans every enabled backend + const option = allOptions.find((opt) => opt.id === modelId); if (!option) return; let shouldCloseMenu = true; if (onModelChange) { - const result = await onModelChange(option.id, option.model); + const result = await onModelChange(rawModelId(option.id), option.model, option.backendId); if (result === false) { shouldCloseMenu = false; @@ -171,7 +273,9 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele } // only the built-in server loads on request, and only in router mode - if (!onModelChange && isRouter && !modelsStore.isModelLoaded(option.model)) { + const canLoadHere = option.backendId === LOCAL_BACKEND_ID && isRouter; + + if (!onModelChange && canLoadHere && !modelsStore.isModelLoaded(option.model)) { isLoadingModel = true; modelsStore.status @@ -183,6 +287,18 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele function getDisplayOption(): ModelOption | undefined { if (!isRouter) { + // External backend: the selection is backend-scoped, so it wins over + // the conversation's model, which may belong to another backend. + if (!serverStore.capabilities.props) { + const selected = activeId ? allOptions.find((option) => option.id === activeId) : undefined; + + if (selected) return selected; + + return currentModel + ? allOptions.find((option) => option.model === currentModel) + : undefined; + } + const displayModel = serverModel || currentModel; if (displayModel) { @@ -207,11 +323,11 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele }; } - return options.find((option) => option.model === currentModel); + return allOptions.find((option) => option.model === currentModel); } if (activeId) { - return options.find((option) => option.id === activeId); + return allOptions.find((option) => option.id === activeId); } return undefined; @@ -222,6 +338,8 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele return activeId; }, + closeProvider, + get emptyMessage() { return emptyMessage; }, @@ -267,6 +385,10 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele return isMultiModel; }, + get isProviderView() { + return isProviderView; + }, + get isRouter() { return isRouter; }, @@ -279,6 +401,8 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele return loading; }, + openProvider, + get options() { return options; }, @@ -295,6 +419,8 @@ export function useModelsSelector(opts: UseModelsSelectorOptions): UseModelsSele searchTerm = value; }, + showBackendModels, + get updating() { return updating; } diff --git a/tools/ui/src/lib/services/backends.service.ts b/tools/ui/src/lib/services/backends.service.ts index 5d660f600f..3945d6142f 100644 --- a/tools/ui/src/lib/services/backends.service.ts +++ b/tools/ui/src/lib/services/backends.service.ts @@ -6,9 +6,9 @@ * consumed by the backends settings UI and the per-backend model cache. */ -import { API_MODELS, LOCAL_BACKEND_ID } from '$lib/constants'; +import { API_MODELS, BackendProtocol, LOCAL_BACKEND_ID } from '$lib/constants'; import { ModelsService } from '$lib/services/models.service'; -import type { ApiModelsListResponse, Backend, BackendProtocol, ModelOption } from '$lib/types'; +import type { ApiModelsListResponse, Backend, ModelOption } from '$lib/types'; import { isAbortError } from '$lib/utils/abort'; import { apiUrl } from '$lib/utils/api-base'; import { getAuthHeadersForBackend } from '$lib/utils/api-headers'; @@ -49,7 +49,7 @@ export class BackendsService { static async detectProtocol(backend: Backend): Promise { const base = backend.baseUrl.trim().replace(/\/+$/, ''); - if (!base) return { authRequired: false, protocol: 'openai' }; + if (!base) return { authRequired: false, protocol: BackendProtocol.OPENAI }; try { // llama-server answers /props with its build and generation defaults; a @@ -62,10 +62,10 @@ export class BackendsService { // a llama-server behind a key refuses before it says anything else, while // an OpenAI-compatible endpoint has no /props to guard in the first place if (response.status === 401) { - return { authRequired: true, protocol: 'llama.cpp' }; + return { authRequired: true, protocol: BackendProtocol.COMPAT }; } - if (!response.ok) return { authRequired: false, protocol: 'openai' }; + if (!response.ok) return { authRequired: false, protocol: BackendProtocol.OPENAI }; const body = (await response.json()) as Record; const isLlamaCpp = @@ -73,10 +73,10 @@ export class BackendsService { return { authRequired: false, - protocol: isLlamaCpp ? 'llama.cpp' : 'openai' + protocol: isLlamaCpp ? BackendProtocol.COMPAT : BackendProtocol.OPENAI }; } catch { - return { authRequired: false, protocol: 'openai' }; + return { authRequired: false, protocol: BackendProtocol.OPENAI }; } } diff --git a/tools/ui/src/lib/services/protocols/index.ts b/tools/ui/src/lib/services/protocols/index.ts index db1f5668a5..f24379abf9 100644 --- a/tools/ui/src/lib/services/protocols/index.ts +++ b/tools/ui/src/lib/services/protocols/index.ts @@ -8,11 +8,12 @@ import { openaiAdapter } from './openai'; import type { ChatProtocolAdapter } from './types'; -import type { Backend, BackendProtocol } from '$lib/types'; +import { BackendProtocol } from '$lib/constants'; +import type { Backend } from '$lib/types'; const ADAPTERS: Record = { - 'llama.cpp': openaiAdapter, - openai: openaiAdapter + [BackendProtocol.COMPAT]: openaiAdapter, + [BackendProtocol.OPENAI]: openaiAdapter }; export function getProtocolAdapter(backend?: Backend): ChatProtocolAdapter { diff --git a/tools/ui/src/lib/services/protocols/openai.ts b/tools/ui/src/lib/services/protocols/openai.ts index 99fd12af40..0eadc580a3 100644 --- a/tools/ui/src/lib/services/protocols/openai.ts +++ b/tools/ui/src/lib/services/protocols/openai.ts @@ -5,7 +5,7 @@ */ import type { ChatProtocolAdapter, ChatStreamEvent, ChatStreamReader } from './types'; -import { HEADERS } from '$lib/constants'; +import { BackendProtocol, HEADERS, MaxTokensField } from '$lib/constants'; import type { Backend } from '$lib/types'; import type { ApiChatCompletionStreamChunk } from '$lib/types/api'; import { getBackendCompat } from '$lib/utils/backend'; @@ -60,7 +60,7 @@ function buildChatRequest( body: Record, backend: Backend ): Record { - if (backend.protocol === 'llama.cpp') return body; + if (backend.protocol === BackendProtocol.COMPAT) return body; const compat = getBackendCompat(backend); const request: Record = { ...body }; @@ -83,7 +83,10 @@ function buildChatRequest( // newer OpenAI models require max_completion_tokens, most compatible // endpoints only understand max_tokens - if (compat.maxTokensField === 'max_completion_tokens' && request.max_tokens !== undefined) { + if ( + compat.maxTokensField === MaxTokensField.CHAT_COMPLETION && + request.max_tokens !== undefined + ) { request.max_completion_tokens = request.max_tokens; delete request.max_tokens; } diff --git a/tools/ui/src/lib/stores/models/index.svelte.ts b/tools/ui/src/lib/stores/models/index.svelte.ts index 5a35c34931..21a53708be 100644 --- a/tools/ui/src/lib/stores/models/index.svelte.ts +++ b/tools/ui/src/lib/stores/models/index.svelte.ts @@ -9,6 +9,7 @@ import { browser } from '$app/environment'; import { + BackendProtocol, FAVORITE_MODELS_LOCALSTORAGE_KEY, HIDDEN_MODELS_LOCALSTORAGE_KEY, LOCAL_BACKEND_ID, @@ -649,7 +650,7 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost { // local props describe the server the UI is served from; keep them while // an external backend is active instead of dropping and refetching - if (backend.protocol === 'llama.cpp') { + if (backend.protocol === BackendProtocol.COMPAT) { serverStore.restoreLocalState(); } else { serverStore.cacheLocalState(); @@ -665,7 +666,7 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost { return; } - if (backend.protocol === 'llama.cpp' && !serverStore.props) { + if (backend.protocol === BackendProtocol.COMPAT && !serverStore.props) { // first visit to the local tab in this session await serverStore.fetch({ background: true }); } @@ -675,7 +676,11 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost { // the local router rows carry the load statuses; the startup prefetch // already returned them, so a tab switch rebuilds the list from memory - if (backend.protocol === 'llama.cpp' && this.routerModels.length === 0 && cached.raw) { + if ( + backend.protocol === BackendProtocol.COMPAT && + this.routerModels.length === 0 && + cached.raw + ) { this.routerModels = cached.raw.data; this.activeModels = this.buildModelOptions(cached.raw); } diff --git a/tools/ui/src/lib/stores/server.svelte.ts b/tools/ui/src/lib/stores/server.svelte.ts index 23131d0d45..d4b16481a1 100644 --- a/tools/ui/src/lib/stores/server.svelte.ts +++ b/tools/ui/src/lib/stores/server.svelte.ts @@ -6,7 +6,7 @@ * PropsService for the /props fetch. */ -import { BACKEND_CAPABILITIES, LOCAL_BACKEND_ID } from '$lib/constants'; +import { BACKEND_CAPABILITIES, BackendProtocol, LOCAL_BACKEND_ID } from '$lib/constants'; import { ServerRole } from '$lib/enums'; import { PropsService } from '$lib/services/props.service'; import type { BackendCapabilities } from '$lib/types'; @@ -38,7 +38,7 @@ class ServerStore { get capabilities(): BackendCapabilities { const backend = getBackend(); - return backend ? getBackendCapabilities(backend) : BACKEND_CAPABILITIES['llama.cpp']; + return backend ? getBackendCapabilities(backend) : BACKEND_CAPABILITIES[BackendProtocol.COMPAT]; } get contextSize(): number | null { diff --git a/tools/ui/src/lib/types/backend.d.ts b/tools/ui/src/lib/types/backend.d.ts index 019fce3610..e4516e6873 100644 --- a/tools/ui/src/lib/types/backend.d.ts +++ b/tools/ui/src/lib/types/backend.d.ts @@ -7,8 +7,9 @@ * protocol. */ -/** Request/response shape a backend speaks. */ -export type BackendProtocol = 'llama.cpp' | 'openai'; +import type { BackendProtocol, MaxTokensField } from '$lib/constants'; + +export type { BackendProtocol }; /** * Wire-level quirks of a backend's protocol. Capabilities gate llama.cpp @@ -16,7 +17,7 @@ export type BackendProtocol = 'llama.cpp' | 'openai'; */ export interface BackendCompat { /** Field carrying the output token cap. */ - maxTokensField: 'max_completion_tokens' | 'max_tokens'; + maxTokensField: MaxTokensField; /** Whether the endpoint accepts stream_options.include_usage. */ supportsUsageInStreaming: boolean; } diff --git a/tools/ui/src/lib/utils/backend.ts b/tools/ui/src/lib/utils/backend.ts index 29c68558f7..e2fd53880c 100644 --- a/tools/ui/src/lib/utils/backend.ts +++ b/tools/ui/src/lib/utils/backend.ts @@ -12,19 +12,15 @@ import { BACKEND_ID_PREFIX, BACKEND_PRESETS, BACKEND_PROTOCOLS, + BackendProtocol, DEFAULT_BACKEND_CHAT_PATH, DEFAULT_BACKEND_MODELS_PATH, FAVICON_SERVICE_URL, LOCAL_BACKEND_ID, + MaxTokensField, MODEL_CONTEXT_LENGTH_FIELDS } from '$lib/constants'; -import type { - Backend, - BackendCapabilities, - BackendCompat, - BackendPreset, - BackendProtocol -} from '$lib/types'; +import type { Backend, BackendCapabilities, BackendCompat, BackendPreset } from '$lib/types'; /** Absolute chat completions URL for a backend. */ export function backendChatUrl(backend: Backend): string { @@ -84,12 +80,18 @@ function normalizeBaseUrl(url: string): string | null { * (unknown model, early startup) gets the plain compatible defaults. */ export function getBackendCapabilities(backend?: Backend): BackendCapabilities { - return BACKEND_CAPABILITIES[backend?.protocol ?? 'openai'] ?? BACKEND_CAPABILITIES.openai; + return ( + BACKEND_CAPABILITIES[backend?.protocol ?? BackendProtocol.OPENAI] ?? + BACKEND_CAPABILITIES[BackendProtocol.OPENAI] + ); } /** Wire quirks for a backend: protocol defaults overridden by the backend. */ export function getBackendCompat(backend: Backend): BackendCompat { - return { ...(BACKEND_COMPAT[backend.protocol] ?? BACKEND_COMPAT.openai), ...backend.compat }; + return { + ...(BACKEND_COMPAT[backend.protocol] ?? BACKEND_COMPAT[BackendProtocol.OPENAI]), + ...backend.compat + }; } /** The built-in backend pointing at the server that serves this UI. */ @@ -100,7 +102,7 @@ export function createLocalBackend(apiKey?: string, enabled = true): Backend { enabled, id: LOCAL_BACKEND_ID, name: 'Local', - protocol: 'llama.cpp' + protocol: BackendProtocol.COMPAT }; } @@ -155,7 +157,7 @@ function parseBackendEntry(entry: unknown, index: number): Backend | null { const protocol = BACKEND_PROTOCOLS.includes(raw.protocol as BackendProtocol) ? (raw.protocol as BackendProtocol) - : 'openai'; + : BackendProtocol.OPENAI; const id = typeof raw.id === 'string' && raw.id.trim() ? raw.id.trim() @@ -190,7 +192,10 @@ function parseBackendCompat( const defaults = BACKEND_COMPAT[protocol] ?? BACKEND_COMPAT.openai; const overrides: Partial = {}; - if (entry.maxTokensField === 'max_tokens' || entry.maxTokensField === 'max_completion_tokens') { + if ( + entry.maxTokensField === MaxTokensField.COMPLETION || + entry.maxTokensField === MaxTokensField.CHAT_COMPLETION + ) { overrides.maxTokensField = entry.maxTokensField; } diff --git a/tools/ui/tests/stories/ModelsSelector.stories.svelte b/tools/ui/tests/stories/ModelsSelector.stories.svelte index 1a2c643c7f..263d1ed010 100644 --- a/tools/ui/tests/stories/ModelsSelector.stories.svelte +++ b/tools/ui/tests/stories/ModelsSelector.stories.svelte @@ -103,7 +103,8 @@ orgName: 'intel' } ], - loaded: loadedModels + loaded: loadedModels, + providers: [] }; function handleSelect(modelId: string) { @@ -137,7 +138,8 @@ currentModel={null} groups={{ available: [], - loaded: [loadedModels[0]] + loaded: [loadedModels[0]], + providers: [] }} onSelect={handleSelect} /> @@ -152,7 +154,8 @@ favorites={favoriteModels} groups={{ available: [], - loaded: [] + loaded: [], + providers: [] }} onSelect={handleSelect} />