mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-28 17:07:31 -05:00
ui : show a model's info from what its provider and load state allow
Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
This commit is contained in:
+6
-1
@@ -30,7 +30,12 @@
|
||||
let edits = $state<ModelOverride | null>(null);
|
||||
let draft = $derived(edits ?? override ?? {});
|
||||
|
||||
let serverProps = $derived(modelsStore.props.getModelProps(option.model));
|
||||
// the props cache is a plain Map, so the read has to name its version to stay reactive
|
||||
let serverProps = $derived.by(() => {
|
||||
void modelsStore.props.cacheVersion;
|
||||
|
||||
return modelsStore.props.getModelProps(option.model);
|
||||
});
|
||||
let status = $derived.by(() => {
|
||||
const model = modelsStore.routerModels.find((m) => m.id === option.model);
|
||||
|
||||
|
||||
+26
-10
@@ -7,6 +7,7 @@
|
||||
import { ModelCapability, ServerModelStatus } from '$lib/enums';
|
||||
import type { ModelOption } from '$lib/types/models';
|
||||
import { getBackend } from '$lib/utils/api-base';
|
||||
import { getBackendCapabilities } from '$lib/utils/backend';
|
||||
|
||||
interface Props {
|
||||
isCustomized: boolean;
|
||||
@@ -24,7 +25,16 @@
|
||||
|
||||
let supportsToolUse = $derived(option.capabilities.includes(ModelCapability.TOOL_USE));
|
||||
let supportsThinking = $derived(option.capabilities.includes(ModelCapability.REASONING));
|
||||
let backendName = $derived(getBackend(option.backendId)?.name ?? null);
|
||||
let backend = $derived(getBackend(option.backendId));
|
||||
let backendName = $derived(backend?.name ?? null);
|
||||
let capabilities = $derived(getBackendCapabilities(backend));
|
||||
// what the provider speaks decides which server endpoints the pane can read
|
||||
let compatLabel = $derived(backend?.protocol === 'llama.cpp' ? 'Llama-compat' : 'OAI-compat');
|
||||
let compatTitle = $derived(
|
||||
capabilities.props
|
||||
? 'llama.cpp server: reads /props and /slots'
|
||||
: 'OpenAI-compatible: no /props, /slots, load or unload'
|
||||
);
|
||||
|
||||
// the listing usually carries the size; a local repo falls back to its tree
|
||||
let size = $state<string | null>(null);
|
||||
@@ -134,6 +144,10 @@
|
||||
{backendName}
|
||||
</span>
|
||||
{/if}
|
||||
|
||||
<Badge class="h-5 shrink-0 px-1.5 text-[10px]" title={compatTitle} variant="secondary">
|
||||
{compatLabel}
|
||||
</Badge>
|
||||
</div>
|
||||
|
||||
<div class="flex gap-2">
|
||||
@@ -143,16 +157,18 @@
|
||||
Start a new chat
|
||||
</Button>
|
||||
|
||||
<Button class="flex-1 gap-1.5" onclick={onToggleLoad} variant="outline">
|
||||
{#if isLoaded}
|
||||
<Eject class="h-3.5 w-3.5" />
|
||||
{#if capabilities.loadUnload}
|
||||
<Button class="flex-1 gap-1.5" onclick={onToggleLoad} variant="outline">
|
||||
{#if isLoaded}
|
||||
<Eject class="h-3.5 w-3.5" />
|
||||
|
||||
Unload model
|
||||
{:else}
|
||||
<Power class="h-3.5 w-3.5" />
|
||||
Unload model
|
||||
{:else}
|
||||
<Power class="h-3.5 w-3.5" />
|
||||
|
||||
Load model
|
||||
{/if}
|
||||
</Button>
|
||||
Load model
|
||||
{/if}
|
||||
</Button>
|
||||
{/if}
|
||||
</div>
|
||||
</header>
|
||||
|
||||
+49
-17
@@ -17,6 +17,8 @@
|
||||
import type { ApiLlamaCppServerProps } from '$lib/types/api';
|
||||
import type { HfModelDetailInfo } from '$lib/types/huggingface';
|
||||
import type { ModelOption } from '$lib/types/models';
|
||||
import { getBackend } from '$lib/utils/api-base';
|
||||
import { getBackendCapabilities } from '$lib/utils/backend';
|
||||
import { formatFileSize, formatNumber, formatParameters } from '$lib/utils/formatters';
|
||||
|
||||
interface Props {
|
||||
@@ -68,17 +70,32 @@
|
||||
let activeDraft = $derived(drafts.find((draft) => draft.active) ?? null);
|
||||
let idleDrafts = $derived(drafts.filter((draft) => !draft.active));
|
||||
let meta = $derived(option.meta);
|
||||
// a plain OpenAI-compatible endpoint has no /props or /slots to read
|
||||
let reportsServerInfo = $derived(getBackendCapabilities(getBackend(option.backendId)).props);
|
||||
// the server reports these once the model is loaded; the Hub knows them anyway
|
||||
let gguf = $derived(hub?.gguf ?? null);
|
||||
let modalities = $derived(modelsStore.props.getModelModalitiesArray(option.id));
|
||||
let modalities = $derived.by(() => {
|
||||
void modelsStore.props.cacheVersion;
|
||||
|
||||
return modelsStore.props.getModelModalitiesArray(option.id);
|
||||
});
|
||||
let rows = $derived([
|
||||
{ isCopyable: true, isMono: true, label: 'File Path', value: serverProps?.model_path ?? null },
|
||||
...(reportsServerInfo
|
||||
? [
|
||||
{
|
||||
isCopyable: true,
|
||||
isMono: true,
|
||||
label: 'File Path',
|
||||
value: serverProps?.model_path ?? null
|
||||
}
|
||||
]
|
||||
: []),
|
||||
{
|
||||
label: 'Context Size',
|
||||
// the server reports the context it runs with once the model is loaded; until then
|
||||
// the listing, or the Hub, still says what the model can take
|
||||
value: serverProps
|
||||
? `${formatNumber(serverProps.default_generation_settings.n_ctx)} tokens`
|
||||
? `${formatNumber(serverProps.default_generation_settings?.n_ctx ?? 0)} tokens`
|
||||
: (contextLabel ?? null)
|
||||
},
|
||||
{ label: 'Training Context', value: contextLabel },
|
||||
@@ -121,8 +138,15 @@
|
||||
label: 'Other sidecars',
|
||||
value: idleDrafts.length > 0 ? idleDrafts.map((draft) => draft.kind).join(', ') : null
|
||||
},
|
||||
{ label: 'Parallel Slots', value: serverProps ? String(serverProps.total_slots) : null },
|
||||
{ isMono: true, label: 'Build Info', value: serverProps?.build_info ?? null }
|
||||
...(reportsServerInfo
|
||||
? [
|
||||
{
|
||||
label: 'Parallel Slots',
|
||||
value: serverProps?.total_slots != null ? String(serverProps.total_slots) : null
|
||||
},
|
||||
{ isMono: true, label: 'Build Info', value: serverProps?.build_info ?? null }
|
||||
]
|
||||
: [])
|
||||
] satisfies Array<{
|
||||
isBadge?: boolean;
|
||||
isCapitalized?: boolean;
|
||||
@@ -202,9 +226,15 @@
|
||||
{/if}
|
||||
</div>
|
||||
|
||||
{#if hub}
|
||||
{#if !reportsServerInfo}
|
||||
<p class="pt-2 text-xs text-muted-foreground">
|
||||
Some values come from the Hugging Face Hub. Load the model to read them from the server.
|
||||
OpenAI-compatible provider: there is no /props or /slots, so only what the listing and the
|
||||
model's repo report is shown.
|
||||
</p>
|
||||
{:else if !serverProps}
|
||||
<p class="pt-2 text-xs text-muted-foreground">
|
||||
/props values - file path, slots, build info - appear once the model is loaded. Reading this
|
||||
page never loads a model.
|
||||
</p>
|
||||
{/if}
|
||||
|
||||
@@ -220,15 +250,17 @@
|
||||
'Not reported by the server.'}</pre>
|
||||
</CollapsibleSection>
|
||||
|
||||
<CollapsibleSection triggerClass={sectionTrigger}>
|
||||
{#snippet trigger()}
|
||||
<span class="text-sm font-medium">Source File</span>
|
||||
{/snippet}
|
||||
{#if reportsServerInfo}
|
||||
<CollapsibleSection triggerClass={sectionTrigger}>
|
||||
{#snippet trigger()}
|
||||
<span class="text-sm font-medium">Source File</span>
|
||||
{/snippet}
|
||||
|
||||
<div class="space-y-2 pt-1">
|
||||
<p class="text-xs break-all text-muted-foreground">
|
||||
{serverProps?.model_path ?? 'Path is reported once the model is loaded.'}
|
||||
</p>
|
||||
</div>
|
||||
</CollapsibleSection>
|
||||
<div class="space-y-2 pt-1">
|
||||
<p class="text-xs break-all text-muted-foreground">
|
||||
{serverProps?.model_path ?? 'Path is reported once the model is loaded.'}
|
||||
</p>
|
||||
</div>
|
||||
</CollapsibleSection>
|
||||
{/if}
|
||||
</div>
|
||||
|
||||
Reference in New Issue
Block a user