ui : filter models by capability and by draft sidecar

The manager could narrow by provider, context and modality only. It now
also asks for tool use or for reasoning, which the Hub chat template
answers for a local model, and for models that have a draft sidecar to
speculate with. The provider menu says how many repos each provider
contributes and steps aside when there is only one to choose between.

The Hub details cache becomes a SvelteMap so a filter re-runs when a
template arrives after the row mounts, and the discover entry's effect
consumes its flag untracked instead of invalidating itself.

Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
This commit is contained in:
Aleksander Grygier
2026-09-28 12:10:14 +02:00
parent d5e81d004b
commit 3044a73ce5
6 changed files with 195 additions and 67 deletions
@@ -58,9 +58,11 @@
$effect(() => {
if (!uiStore.discoverModelsOpen) return;
uiStore.discoverModelsOpen = false;
view = 'discover';
handleOpenChange(true);
untrack(() => {
uiStore.discoverModelsOpen = false;
view = 'discover';
handleOpenChange(true);
});
});
function handleOpenChange(value: boolean) {
@@ -8,12 +8,15 @@
loadOverrides,
type ModalityKey,
modelContextLength,
modelDraftsFor,
type ModelOverride,
type ModelQuantGroup,
type ModelsTableGroup,
modelSupports,
saveOverrides
} from './utils';
import { LOCAL_BACKEND_ID } from '$lib/constants';
import { ModelCapability } from '$lib/enums';
import { backendsStore, conversationsStore, modelsStore, uiStore } from '$lib/stores';
import type { ModelOption } from '$lib/types/models';
import { getBackend } from '$lib/utils/api-base';
@@ -35,6 +38,8 @@
let providerFilter = $state<string[]>([]);
let contextLimit = $state(0);
let modalityFilter = $state<ModalityKey[]>([]);
let capabilityFilter = $state<ModelCapability[]>([]);
let draftFilter = $state(false);
let selectedId = $state<string | null>(null);
let overrides = $state<Record<string, ModelOverride>>(loadOverrides());
@@ -42,18 +47,28 @@
let isFavorite = $derived((option: ModelOption) =>
modelsStore.favoriteModelIds.has(option.model)
);
// the rail counts follow the active view and filter, so it always says how many
// models each provider contributes to what the table is showing
let visible = $derived.by(() => {
// every filter but the provider one, so a provider count does not fall to zero
// the moment that provider is the one being looked at
let matching = $derived.by(() => {
const term = filter.trim().toLowerCase();
return allModels.filter((option) => {
if (term && !`${option.name} ${option.model}`.toLowerCase().includes(term)) return false;
if (providerFilter.length > 0) {
const backendId = option.backendId ?? LOCAL_BACKEND_ID;
// every capability asked for has to be there: tools and reasoning are
// features a model either has or does not, unlike modalities
if (
capabilityFilter.length > 0 &&
!capabilityFilter.every((capability) => modelSupports(option, capability))
) {
return false;
}
if (!providerFilter.includes(backendId)) return false;
if (
draftFilter &&
modelDraftsFor(option, overrides[option.id]?.load?.speculativeDecoding).length === 0
) {
return false;
}
// a model whose modalities are unknown cannot be shown to match
@@ -64,6 +79,24 @@
return contextLimit === 0 || modelContextLength(option) >= contextLimit;
});
});
let visible = $derived.by(() =>
providerFilter.length === 0
? matching
: matching.filter((option) => providerFilter.includes(option.backendId ?? LOCAL_BACKEND_ID))
);
// the rail counts follow the active view and filter, so it always says how many
// repos each provider contributes to what the table is showing
let providerCounts = $derived.by(() => {
const counts: Record<string, number> = {};
for (const entry of groupModelQuants(matching)) {
const backendId = entry.base.backendId ?? LOCAL_BACKEND_ID;
counts[backendId] = (counts[backendId] ?? 0) + 1;
}
return counts;
});
// recently used models lead their section, the rest keep the server's order
const rank = new SvelteMap<string, number>();
@@ -356,7 +389,9 @@
<div class={['flex min-h-0 flex-1', className]}>
<div class="min-h-0 min-w-0 flex-1">
<ModelsManagerModelsTable
bind:capabilities={capabilityFilter}
bind:contextLimit
bind:draft={draftFilter}
bind:filter
bind:modalities={modalityFilter}
bind:providers={providerFilter}
@@ -365,6 +400,7 @@
onSelect={(option) => (selectedId = option.id)}
onUseAsDraft={useAsDraft}
{overrides}
{providerCounts}
{selectedId}
toolbarEnd={toolbarEndRegion}
/>
@@ -1,29 +1,40 @@
<script lang="ts">
import type { ModalityKey } from './utils';
import { ChevronDown, Image, Mic, Server, Video } from '@lucide/svelte';
import { Check, ChevronDown, Image, Lightbulb, Mic, Server, Video, Wrench } from '@lucide/svelte';
import { Logo } from '$lib/components/app';
import { BackendIcon } from '$lib/components/app/backends';
import * as DropdownMenu from '$lib/components/ui/dropdown-menu';
import * as Select from '$lib/components/ui/select';
import { Toggle } from '$lib/components/ui/toggle';
import * as ToggleGroup from '$lib/components/ui/toggle-group';
import { FILTER_TRIGGER_CLASS, LOCAL_BACKEND_ID } from '$lib/constants';
import { ModelCapability } from '$lib/enums';
import type { Backend } from '$lib/types/backend';
interface Props {
backends: Backend[];
/** Capabilities a model must have every one of. */
capabilities?: ModelCapability[];
/** Smallest context a model must support; 0 keeps every model. */
contextLimit?: number;
/** Keep only models that have a draft sidecar to speculate with. */
draft?: boolean;
/** Modalities a model must support at least one of. */
modalities?: ModalityKey[];
/** Backend ids to keep; empty keeps every provider. */
providers?: string[];
/** Repos each provider contributes to the current search, shown in the menu. */
providerCounts?: Record<string, number>;
}
let {
backends,
capabilities = $bindable<ModelCapability[]>([]),
contextLimit = $bindable(0),
modalities = $bindable([]),
providers = $bindable([])
draft = $bindable(false),
modalities = $bindable<ModalityKey[]>([]),
providerCounts = {},
providers = $bindable<string[]>([])
}: Props = $props();
const CONTEXT_STEPS: { label: string; value: number }[] = [
@@ -34,12 +45,18 @@
{ label: '256K or more', value: 262_144 },
{ label: '1M or more', value: 1_048_576 }
];
const CAPABILITIES: { icon: typeof Wrench; label: string; value: ModelCapability }[] = [
{ icon: Wrench, label: 'Tool use', value: ModelCapability.TOOL_USE },
{ icon: Lightbulb, label: 'Reasoning', value: ModelCapability.REASONING }
];
const MODALITIES: { icon: typeof Image; key: ModalityKey; label: string }[] = [
{ icon: Image, key: 'vision', label: 'Vision' },
{ icon: Video, key: 'video', label: 'Video' },
{ icon: Mic, key: 'audio', label: 'Audio' }
];
const TOGGLE_ITEM_CLASS = 'bg-transparent! border-border/30! dark:border-border/20!';
// none selected means every provider, so the label names the selection
let providerLabel = $derived(
providers.length === 0
@@ -70,46 +87,52 @@
{/snippet}
<div class="flex flex-wrap items-center gap-2">
<DropdownMenu.Root>
<DropdownMenu.Trigger>
{#snippet child({ props })}
<button
{...props}
class="inline-flex items-center whitespace-nowrap {FILTER_TRIGGER_CLASS}"
type="button"
>
<Server class="h-3.5 w-3.5" />
{providerLabel}
<ChevronDown class="h-3.5 w-3.5 opacity-60" />
</button>
{/snippet}
</DropdownMenu.Trigger>
<DropdownMenu.Content align="start" class="min-w-48">
<DropdownMenu.Group>
<DropdownMenu.GroupHeading>Providers</DropdownMenu.GroupHeading>
{#each backends as backend (backend.id)}
<DropdownMenu.CheckboxItem
checked={providers.includes(backend.id)}
onCheckedChange={(checked) => toggleProvider(backend.id, checked)}
{#if backends.length > 1}
<DropdownMenu.Root>
<DropdownMenu.Trigger>
{#snippet child({ props })}
<button
{...props}
class="inline-flex items-center whitespace-nowrap {FILTER_TRIGGER_CLASS}"
type="button"
>
{@render providerMark(backend)}
<Server class="h-3.5 w-3.5" />
{backend.name}
</DropdownMenu.CheckboxItem>
{/each}
</DropdownMenu.Group>
{providerLabel}
{#if providers.length > 0}
<DropdownMenu.Separator />
<ChevronDown class="h-3.5 w-3.5 opacity-60" />
</button>
{/snippet}
</DropdownMenu.Trigger>
<DropdownMenu.Item onSelect={() => (providers = [])}>Show every provider</DropdownMenu.Item>
{/if}
</DropdownMenu.Content>
</DropdownMenu.Root>
<DropdownMenu.Content align="start" class="min-w-48">
<DropdownMenu.Group>
<DropdownMenu.GroupHeading>Providers</DropdownMenu.GroupHeading>
{#each backends as backend (backend.id)}
<DropdownMenu.CheckboxItem
checked={providers.includes(backend.id)}
onCheckedChange={(checked) => toggleProvider(backend.id, checked)}
>
{@render providerMark(backend)}
{backend.name}
<DropdownMenu.Shortcut>{providerCounts[backend.id] ?? 0}</DropdownMenu.Shortcut>
</DropdownMenu.CheckboxItem>
{/each}
</DropdownMenu.Group>
{#if providers.length > 0}
<DropdownMenu.Separator />
<DropdownMenu.Item onSelect={() => (providers = [])}
>Show every provider</DropdownMenu.Item
>
{/if}
</DropdownMenu.Content>
</DropdownMenu.Root>
{/if}
<Select.Root
onValueChange={(value) => (contextLimit = Number(value))}
@@ -129,6 +152,44 @@
</Select.Content>
</Select.Root>
<ToggleGroup.Root
bind:value={capabilities}
class="bg-muted/60 dark:bg-muted/75"
type="multiple"
variant="outline"
>
{#each CAPABILITIES as capability (capability.value)}
<ToggleGroup.Item
aria-label={capability.label}
class={TOGGLE_ITEM_CLASS}
title={capability.label}
value={capability.value}
>
<capability.icon class="h-3.5 w-3.5" />
</ToggleGroup.Item>
{/each}
</ToggleGroup.Root>
<!-- a checkbox chip: the whole pill is the control, the box is its indicator -->
<Toggle
bind:pressed={draft}
class="inline-flex items-center whitespace-nowrap {FILTER_TRIGGER_CLASS}"
variant="outline"
>
<span
aria-hidden="true"
class="flex size-4 shrink-0 items-center justify-center rounded-[4px] border transition-shadow {draft
? 'border-primary bg-primary text-primary-foreground'
: 'border-input bg-background dark:bg-input/30'}"
>
{#if draft}
<Check class="size-3" />
{/if}
</span>
Has draft sidecar
</Toggle>
<ToggleGroup.Root
bind:value={modalities}
class="bg-muted/60 dark:bg-muted/75"
@@ -138,7 +199,7 @@
{#each MODALITIES as modality (modality.key)}
<ToggleGroup.Item
aria-label={modality.label}
class="bg-transparent! border-border/30! dark:border-border/20!"
class={TOGGLE_ITEM_CLASS}
title={modality.label}
value={modality.key}
>
@@ -1,17 +1,14 @@
<script lang="ts">
import ModelsManagerFilters from './ModelsManagerFilters.svelte';
import {
draftFromArgs,
draftFromSetting,
isLocalOption,
type ModalityKey,
modelContextLength,
type ModelDraft,
modelDrafts,
modelDraftsFor,
type ModelOverride,
type ModelQuantGroup,
type ModelsTableGroup,
sidecarFilesFor
type ModelsTableGroup
} from './utils';
import {
ArrowDown,
@@ -48,7 +45,7 @@
import { Badge } from '$lib/components/ui/badge';
import { Button } from '$lib/components/ui/button';
import { FAMILY_ROW_WINDOW, MODEL_ROW_WINDOW } from '$lib/constants';
import { ModelDownloadConfirmAction, ServerModelStatus } from '$lib/enums';
import { ModelCapability, ModelDownloadConfirmAction, ServerModelStatus } from '$lib/enums';
import {
backendsModelsStore,
backendsStore,
@@ -64,8 +61,12 @@
import { SvelteSet } from 'svelte/reactivity';
interface Props {
/** Capabilities a model must have every one of. */
capabilities?: ModelCapability[];
/** Smallest context a model must support; 0 keeps every model. */
contextLimit?: number;
/** Keep only models that have a draft sidecar to speculate with. */
draft?: boolean;
filter?: string;
groups: ModelsTableGroup[];
isFavorite: (option: ModelOption) => boolean;
@@ -76,6 +77,8 @@
overrides: Record<string, ModelOverride>;
/** Backend ids to keep; empty keeps every provider. */
providers?: string[];
/** Repos each provider contributes to the current search, for the filter menu. */
providerCounts?: Record<string, number>;
/** Called when a row is set as the draft of the selected model. */
onUseAsDraft?: (draft: ModelOption, targetId: string) => void;
selectedId: string | null;
@@ -84,7 +87,9 @@
}
let {
capabilities = $bindable<ModelCapability[]>([]),
contextLimit = $bindable(0),
draft = $bindable(false),
filter = $bindable(''),
groups,
isFavorite,
@@ -92,13 +97,20 @@
onSelect,
onUseAsDraft,
overrides,
providerCounts = {},
providers = $bindable<string[]>([]),
selectedId,
toolbarEnd
}: Props = $props();
let isEmpty = $derived(groups.every((group) => group.items.length === 0));
let hasFilters = $derived(providers.length > 0 || contextLimit > 0 || modalities.length > 0);
let hasFilters = $derived(
providers.length > 0 ||
contextLimit > 0 ||
modalities.length > 0 ||
capabilities.length > 0 ||
draft
);
let filterInput = $state<HTMLInputElement | null>(null);
// the dialog hands focus to its first control, so the filter takes it instead
@@ -224,16 +236,7 @@
/** Drafts to show on a row: the configured one first, then this repo's own sidecars. */
function draftsFor(option: ModelOption): ModelDraft[] {
// speculative decoding is a llama.cpp feature
if (!getBackendCapabilities(getBackend(option.backendId)).loadUnload) return [];
// the router reports the arguments a model loads with, draft included
const args = modelsStore.routerModels.find((model) => model.id === option.model)?.status?.args;
const configured =
draftFromArgs(args, option) ??
draftFromSetting(option, overrides[option.id]?.load?.speculativeDecoding);
return modelDrafts(option, sidecarFilesFor(option), configured);
return modelDraftsFor(option, overrides[option.id]?.load?.speculativeDecoding);
}
/** Name of the model the configuration pane has open, for the draft action label. */
@@ -646,10 +649,13 @@
/>
<ModelsManagerFilters
bind:capabilities
bind:contextLimit
bind:draft
bind:modalities
bind:providers
backends={backendsStore.enabled}
{providerCounts}
/>
{#if hasFilters}
@@ -659,6 +665,8 @@
providers = [];
contextLimit = 0;
modalities = [];
capabilities = [];
draft = false;
}}
size="sm"
variant="ghost"
@@ -6,6 +6,7 @@ import {
SETTINGS_KEYS,
SPEC_TYPE
} from '$lib/constants';
import { ModelCapability } from '$lib/enums';
import { HuggingFaceService, ModelsService } from '$lib/services';
import { backendsModelsStore, modelsStore, settingsStore } from '$lib/stores';
import type {
@@ -14,7 +15,9 @@ import type {
ModelOption,
ModelSidecarFile
} from '$lib/types/models';
import { detectThinkingSupport, detectToolUseSupport } from '$lib/utils';
import { getBackend } from '$lib/utils/api-base';
import { getBackendCapabilities } from '$lib/utils/backend';
import { formatFileSize, formatParameters } from '$lib/utils/formatters';
import { rawModelId } from '$lib/utils/model-option-id';
import { SvelteMap } from 'svelte/reactivity';
@@ -217,6 +220,9 @@ export function draftFromSetting(option: ModelOption, value?: string | null): Mo
* settings name, then any other sidecar the model's own repo ships.
*/
export function modelDraftsFor(option: ModelOption, settingValue?: string | null): ModelDraft[] {
// speculative decoding is a llama.cpp feature
if (!getBackendCapabilities(getBackend(option.backendId)).loadUnload) return [];
const args = modelsStore.routerModels.find((model) => model.id === option.model)?.status?.args;
const configured = draftFromArgs(args, option) ?? draftFromSetting(option, settingValue);
@@ -261,6 +267,20 @@ export function modelDrafts(
return drafts;
}
/**
* Whether a model has a capability. A listing that declares nothing is not a listing
* that lacks it: the chat template the Hub carries says whether tools or reasoning work.
*/
export function modelSupports(option: ModelOption, capability: ModelCapability): boolean {
if (option.capabilities.includes(capability)) return true;
const template = HuggingFaceService.cachedDetails(option.model)?.gguf?.chat_template ?? '';
return capability === ModelCapability.TOOL_USE
? detectToolUseSupport(template)
: detectThinkingSupport(template);
}
export function servedByLabel(option: ModelOption): string {
const backend = getBackend(option.backendId);
@@ -72,6 +72,7 @@ import type {
HfModelSibling
} from '$lib/types/huggingface';
import { sidecarFromFileToken } from '$lib/utils';
import { SvelteMap } from 'svelte/reactivity';
/**
* HuggingFaceService - Service for browsing and searching GGUF models on Hugging Face Hub
@@ -91,7 +92,7 @@ export class HuggingFaceService {
// Model details and file trees fetched this session, keyed by repo id. The
// Hub rate limits aggressively, so each repo costs at most one request per
// app load; failed fetches are not cached so the next mount can retry.
private static detailsCache = new Map<string, HfModelDetailInfo | null>();
private static detailsCache = new SvelteMap<string, HfModelDetailInfo | null>();
private static detailsPending = new Map<string, Promise<HfModelDetailInfo | null>>();