mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-30 01:47:39 -05:00
ui : drop the size column and resolve the size in the detail pane
Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
This commit is contained in:
@@ -1,81 +0,0 @@
|
||||
<script lang="ts">
|
||||
import { modelSizeLabel } from './ModelsManager/utils';
|
||||
import { HuggingFaceService } from '$lib/services';
|
||||
import type { ModelOption } from '$lib/types/models';
|
||||
import { formatFileSize } from '$lib/utils/formatters';
|
||||
|
||||
interface Props {
|
||||
option: ModelOption;
|
||||
}
|
||||
|
||||
let { option }: Props = $props();
|
||||
|
||||
// the router reports a size for some backends, that wins over a lookup
|
||||
let reported = $derived(modelSizeLabel(option));
|
||||
let el = $state<HTMLElement | null>(null);
|
||||
let isNearViewport = $state(false);
|
||||
let fetchedBytes = $state<number | null>(null);
|
||||
|
||||
// a repo tree costs one Hugging Face request, so wait until the row is near
|
||||
// the viewport and only look up local GGUF quants, which are files on disk
|
||||
let repo = $derived(option.model.split(':')[0] ?? '');
|
||||
let quant = $derived(option.model.split(':')[1] ?? '');
|
||||
|
||||
$effect(() => {
|
||||
if (isNearViewport || !el) return;
|
||||
|
||||
if (typeof IntersectionObserver === 'undefined') {
|
||||
isNearViewport = true;
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
const observer = new IntersectionObserver(
|
||||
(entries) => {
|
||||
if (entries.some((entry) => entry.isIntersecting)) {
|
||||
isNearViewport = true;
|
||||
observer.disconnect();
|
||||
}
|
||||
},
|
||||
{ rootMargin: '200px' }
|
||||
);
|
||||
|
||||
observer.observe(el);
|
||||
|
||||
return () => observer.disconnect();
|
||||
});
|
||||
|
||||
$effect(() => {
|
||||
fetchedBytes = null;
|
||||
|
||||
if (reported || !isNearViewport || !repo || !quant) return;
|
||||
|
||||
let cancelled = false;
|
||||
|
||||
void HuggingFaceService.getTree(repo)
|
||||
.then((tree) => {
|
||||
if (cancelled) return;
|
||||
|
||||
// shards of one quant collapse into a single entry carrying the total
|
||||
const file = HuggingFaceService.collapseGgufShards(
|
||||
HuggingFaceService.filterByExtension(tree, '.gguf')
|
||||
).find((entry) => {
|
||||
const meta = HuggingFaceService.extractQuantMeta(entry.path);
|
||||
|
||||
return meta?.quant === quant && !meta.sidecar;
|
||||
});
|
||||
|
||||
if (file?.size) fetchedBytes = file.size;
|
||||
})
|
||||
// best-effort: offline or a repo we cannot read keeps the dash
|
||||
.catch(() => {});
|
||||
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
});
|
||||
|
||||
let label = $derived(reported ?? (fetchedBytes ? formatFileSize(fetchedBytes) : '—'));
|
||||
</script>
|
||||
|
||||
<span bind:this={el} class="text-sm text-muted-foreground">{label}</span>
|
||||
+22
-1
@@ -5,6 +5,7 @@
|
||||
type ModelOverride,
|
||||
modelQuantLabel,
|
||||
modelSizeLabel,
|
||||
resolveModelSize,
|
||||
SAMPLING_DEFAULTS,
|
||||
SPECULATIVE_OPTIONS
|
||||
} from './utils';
|
||||
@@ -79,6 +80,26 @@
|
||||
let quant = $derived(modelQuantLabel(option));
|
||||
let size = $derived(modelSizeLabel(option));
|
||||
let stopStrings = $derived(draft.stopStrings ?? []);
|
||||
let resolvedSize = $state<string | null>(null);
|
||||
|
||||
// the router rarely reports a size, the repo tree does
|
||||
$effect(() => {
|
||||
const target = option;
|
||||
|
||||
let cancelled = false;
|
||||
|
||||
resolvedSize = null;
|
||||
|
||||
void resolveModelSize(target)
|
||||
.then((value) => {
|
||||
if (!cancelled) resolvedSize = value;
|
||||
})
|
||||
.catch(() => {});
|
||||
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
});
|
||||
let infoRows = $derived([
|
||||
{ label: 'Model', value: option.model },
|
||||
{ label: 'File Path', value: serverProps?.model_path ?? null },
|
||||
@@ -92,7 +113,7 @@
|
||||
label: 'Training Context',
|
||||
value: option.contextLength ? `${formatParameters(option.contextLength)} tokens` : null
|
||||
},
|
||||
{ label: 'Model Size', value: size },
|
||||
{ label: 'Model Size', value: resolvedSize ?? size },
|
||||
{ label: 'Parameters', value: option.parsedId?.params ?? null },
|
||||
{ isBadge: true, label: 'Quantization', value: quant },
|
||||
{ isBadge: true, label: 'Architecture', value: (option.meta?.architecture as string) ?? null },
|
||||
|
||||
+1
-6
@@ -7,7 +7,6 @@
|
||||
ModelId,
|
||||
ModelLoadControl,
|
||||
ModelRowActions,
|
||||
ModelSize,
|
||||
ModelsSection
|
||||
} from '$lib/components/app';
|
||||
import { Input } from '$lib/components/ui/input';
|
||||
@@ -36,7 +35,7 @@
|
||||
}: Props = $props();
|
||||
|
||||
let isEmpty = $derived(groups.every((group) => group.items.length === 0));
|
||||
const rowGrid = 'grid grid-cols-[minmax(0,1fr)_5.5rem_3rem_4.5rem] items-center gap-3';
|
||||
const rowGrid = 'grid grid-cols-[minmax(0,1fr)_3rem_4.5rem] items-center gap-3';
|
||||
|
||||
function stateOf(option: ModelOption): ServerModelStatus | null {
|
||||
const model = modelsStore.routerModels.find((m) => m.id === option.model);
|
||||
@@ -83,8 +82,6 @@
|
||||
/>
|
||||
</span>
|
||||
|
||||
<ModelSize {option} />
|
||||
|
||||
<ModelLoadControl
|
||||
canLoad={getBackendCapabilities(getBackend(option.backendId)).loadUnload}
|
||||
{isFailed}
|
||||
@@ -114,8 +111,6 @@
|
||||
>
|
||||
<span>Model</span>
|
||||
|
||||
<span>Size</span>
|
||||
|
||||
<span class="text-center">State</span>
|
||||
|
||||
<span class="text-right">Actions</span>
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { LOCAL_BACKEND_ID, MODEL_OVERRIDES_LOCALSTORAGE_KEY } from '$lib/constants';
|
||||
import { HuggingFaceService } from '$lib/services';
|
||||
import type { ModelOption } from '$lib/types/models';
|
||||
import { getBackend } from '$lib/utils/api-base';
|
||||
import { formatFileSize, formatParameters } from '$lib/utils/formatters';
|
||||
@@ -129,6 +130,34 @@ export function modelQuantLabel(option: ModelOption): string | null {
|
||||
return option.parsedId?.quantization ?? null;
|
||||
}
|
||||
|
||||
/**
|
||||
* File size of the model's own quant. The router reports one for some backends;
|
||||
* otherwise a local GGUF reads it from its repo tree, the same source the
|
||||
* discovery details use. Returns null when neither knows.
|
||||
*/
|
||||
export async function resolveModelSize(option: ModelOption): Promise<string | null> {
|
||||
const reported = modelSizeLabel(option);
|
||||
|
||||
if (reported) return reported;
|
||||
|
||||
if (!isLocalOption(option)) return null;
|
||||
|
||||
const [repo, quant] = option.model.split(':');
|
||||
|
||||
if (!repo || !quant) return null;
|
||||
|
||||
const tree = await HuggingFaceService.getTree(repo);
|
||||
const file = HuggingFaceService.collapseGgufShards(
|
||||
HuggingFaceService.filterByExtension(tree, '.gguf')
|
||||
).find((entry) => {
|
||||
const meta = HuggingFaceService.extractQuantMeta(entry.path);
|
||||
|
||||
return meta?.quant === quant && !meta.sidecar;
|
||||
});
|
||||
|
||||
return file?.size ? formatFileSize(file.size) : null;
|
||||
}
|
||||
|
||||
/** Extra args the router applies when this model is loaded. */
|
||||
export function loadExtraArgs(override?: ModelOverride): string[] {
|
||||
const load = override?.load;
|
||||
|
||||
@@ -51,7 +51,6 @@ export { default as ModelId } from './ModelId.svelte';
|
||||
export { default as ModelAvatar } from './ModelAvatar.svelte';
|
||||
export { default as ModelLoadControl } from './ModelLoadControl.svelte';
|
||||
export { default as ModelRowActions } from './ModelRowActions.svelte';
|
||||
export { default as ModelSize } from './ModelSize.svelte';
|
||||
export { default as ModelsSection } from './ModelsSection.svelte';
|
||||
|
||||
/**
|
||||
|
||||
Reference in New Issue
Block a user