mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-29 17:37:39 -05:00
ui : add the providers data layer
A provider is a server entry with a base url, an optional key, a protocol and the paths its API lives at; the local llama.cpp server is the built-in one. Backends persist in settings and the active one is restored on load. Requests resolve against the active backend, model ids become backend-qualified, and every backend's model list is fetched and cached in the background. The manager's helpers learn to read a model's drafts, context and the provider that serves it. Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
This commit is contained in:
Vendored
+2
@@ -12,6 +12,7 @@ import type {
|
||||
ApiChatCompletionStreamChunk,
|
||||
ApiChatCompletionToolCall,
|
||||
ApiChatCompletionToolCallDelta,
|
||||
ApiChatCompletionUsage,
|
||||
ApiChatMessageContentPart,
|
||||
ApiChatMessageData,
|
||||
ApiContextSizeError,
|
||||
@@ -74,6 +75,7 @@ declare global {
|
||||
ApiChatCompletionResponse,
|
||||
ApiChatCompletionStreamChunk,
|
||||
ApiChatCompletionToolCall,
|
||||
ApiChatCompletionUsage,
|
||||
ApiChatCompletionToolCallDelta,
|
||||
ApiChatMessageData,
|
||||
ApiChatMessageContentPart,
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
<script lang="ts">
|
||||
import ModelCapabilityIcons from './ModelCapabilityIcons.svelte';
|
||||
import type { ModelDraft } from './ModelsManager/utils';
|
||||
import { Database, ScrollText } from '@lucide/svelte';
|
||||
import { TruncatedText } from '$lib/components/app';
|
||||
import * as Tooltip from '$lib/components/ui/tooltip';
|
||||
@@ -37,6 +38,8 @@
|
||||
/** Min/max GGUF file size (main + draft) across quants; renders a range when set. */
|
||||
sizeRange?: { min: number; max: number } | null;
|
||||
draftSidecars?: ModelSidecar[];
|
||||
/** Drafts a load would use, rendered as `+ [KIND] [QUANT]` next to the id. */
|
||||
drafts?: ModelDraft[];
|
||||
/** Allow badges to wrap onto new lines instead of truncating. */
|
||||
wrap?: boolean;
|
||||
class?: string;
|
||||
@@ -46,6 +49,7 @@
|
||||
aliases,
|
||||
class: className = '',
|
||||
contextLength,
|
||||
drafts = [],
|
||||
draftSidecars = [],
|
||||
hideCapabilities = false,
|
||||
hideModalities = false,
|
||||
@@ -88,6 +92,7 @@
|
||||
let uniqueAliases = $derived([...new Set(aliases ?? [])]);
|
||||
let uniqueTags = $derived([...new Set([...(parsed.tags ?? []), ...(tags ?? [])])]);
|
||||
let uniqueDraftSidecars = $derived([...new Set(draftSidecars)].filter((s) => !isAuxSidecar(s)));
|
||||
let activeDrafts = $derived(drafts.filter((draft) => draft.active && draft.kind));
|
||||
|
||||
let primaryAlias = $derived(uniqueAliases.length === 1 ? uniqueAliases[0] : null);
|
||||
let displayName = $derived(primaryAlias ?? parsed.modelName ?? modelId);
|
||||
@@ -137,6 +142,18 @@
|
||||
</span>
|
||||
{/if}
|
||||
|
||||
{#each activeDrafts as draft (draft.kind ?? 'draft')}
|
||||
<span class="flex shrink-0 items-center gap-1">
|
||||
<span class="text-[10px] text-muted-foreground">+</span>
|
||||
|
||||
<span class={variantBadgeClass} title="Speculative draft in use">{draft.kind}</span>
|
||||
|
||||
{#if draft.quant}
|
||||
<span class={badgeClass}>{draft.quant}</span>
|
||||
{/if}
|
||||
</span>
|
||||
{/each}
|
||||
|
||||
{#each uniqueDraftSidecars as sidecar (sidecar)}
|
||||
<span class={variantBadgeClass} title={`${sidecar.toUpperCase()} draft model available`}>
|
||||
{sidecar}
|
||||
|
||||
@@ -1,10 +1,25 @@
|
||||
import { MODEL_OVERRIDES_LOCALSTORAGE_KEY, SETTINGS_KEYS } from '$lib/constants';
|
||||
import {
|
||||
LOCAL_BACKEND_ID,
|
||||
MODEL_ID,
|
||||
MODEL_OVERRIDES_LOCALSTORAGE_KEY,
|
||||
type ModelSidecar,
|
||||
SETTINGS_KEYS,
|
||||
SPEC_TYPE
|
||||
} from '$lib/constants';
|
||||
import { ModelCapability } from '$lib/enums';
|
||||
import { HuggingFaceService, ModelsService } from '$lib/services';
|
||||
import { settingsStore } from '$lib/stores';
|
||||
import type { ModelLoadProgress, ModelModalities, ModelOption } from '$lib/types/models';
|
||||
import { backendsModelsStore, modelsStore, settingsStore } from '$lib/stores';
|
||||
import type {
|
||||
ModelLoadProgress,
|
||||
ModelModalities,
|
||||
ModelOption,
|
||||
ModelSidecarFile
|
||||
} from '$lib/types/models';
|
||||
import { detectThinkingSupport, detectToolUseSupport } from '$lib/utils';
|
||||
import { getBackend } from '$lib/utils/api-base';
|
||||
import { getBackendCapabilities } from '$lib/utils/backend';
|
||||
import { formatFileSize, formatParameters } from '$lib/utils/formatters';
|
||||
import { rawModelId } from '$lib/utils/model-option-id';
|
||||
import { SvelteMap } from 'svelte/reactivity';
|
||||
|
||||
/** Load parameters a model can override before it is loaded. */
|
||||
@@ -113,6 +128,149 @@ export function saveOverrides(overrides: ModelOverrideMap): void {
|
||||
}
|
||||
}
|
||||
|
||||
/** Backend a model is served by, the local server reads as "This server". */
|
||||
/** A draft a model can speculate with, and whether a load would use it. */
|
||||
export interface ModelDraft {
|
||||
/** The draft a load would use, from the server's own arguments or from the settings. */
|
||||
active: boolean;
|
||||
kind: ModelSidecar | null;
|
||||
/** Repo the draft comes from; null when the file sits in the model's own repo. */
|
||||
model: string | null;
|
||||
params: string | null;
|
||||
quant: string | null;
|
||||
}
|
||||
|
||||
/** Repo an id belongs to: the id without its quant tag. */
|
||||
function repoOf(modelId: string): string | null {
|
||||
const raw = rawModelId(modelId).split(MODEL_ID.QUANTIZATION_SEPARATOR)[0] ?? '';
|
||||
|
||||
return raw || null;
|
||||
}
|
||||
|
||||
/** Sidecar a `--spec-type` value names, e.g. `draft-mtp` -> mtp. */
|
||||
function sidecarFromSpecType(specType: string | null | undefined): ModelSidecar | null {
|
||||
if (!specType) return null;
|
||||
|
||||
const entry = Object.entries(SPEC_TYPE).find(([, value]) => value === specType);
|
||||
|
||||
return (entry?.[0] as ModelSidecar | undefined) ?? null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Repo a draft path names. A Hub cache path carries it (`models--org--name`), a plain
|
||||
* file next to the model does not, so that falls back to the file name.
|
||||
*/
|
||||
function repoFromDraftPath(path: string): string {
|
||||
const cached = /models--([^/]+)[/]/.exec(path);
|
||||
|
||||
if (cached) {
|
||||
const [org, ...rest] = cached[1].split('--');
|
||||
|
||||
if (rest.length > 0) return `${org}/${rest.join('--')}`;
|
||||
}
|
||||
|
||||
return path.split(/[/]/).pop() ?? path;
|
||||
}
|
||||
|
||||
/**
|
||||
* Draft the server's own launch arguments point at. The router reports the arguments a
|
||||
* model loads with, so this is what a load would really speculate with.
|
||||
*/
|
||||
export function draftFromArgs(args: string[] | undefined, option: ModelOption): ModelDraft | null {
|
||||
const flag = args?.indexOf('--model-draft') ?? -1;
|
||||
const path = flag === -1 ? null : (args?.[flag + 1] ?? null);
|
||||
|
||||
if (!path) return null;
|
||||
|
||||
const parsed = ModelsService.parseModelId(path.split(/[/\\]/).pop() ?? path);
|
||||
const repo = repoFromDraftPath(path);
|
||||
|
||||
return {
|
||||
active: true,
|
||||
kind: sidecarFromSpecType(args?.[(args?.indexOf('--spec-type') ?? -1) + 1]) ?? parsed.sidecar,
|
||||
model: repo === repoOf(option.model) ? null : repo,
|
||||
params: parsed.params
|
||||
? `${parsed.params}${parsed.activatedParams ? `-${parsed.activatedParams}` : ''}`
|
||||
: null,
|
||||
quant: parsed.quantization
|
||||
};
|
||||
}
|
||||
|
||||
/** Draft the load settings name, resolved against the model's own repo. */
|
||||
export function draftFromSetting(option: ModelOption, value?: string | null): ModelDraft | null {
|
||||
const id = value?.trim();
|
||||
|
||||
if (!id || id === 'off') return null;
|
||||
|
||||
const parsed = ModelsService.parseModelId(id);
|
||||
|
||||
return {
|
||||
active: true,
|
||||
kind: parsed.sidecar,
|
||||
model: repoOf(id) === repoOf(option.model) ? null : id,
|
||||
params: parsed.params
|
||||
? `${parsed.params}${parsed.activatedParams ? `-${parsed.activatedParams}` : ''}`
|
||||
: null,
|
||||
quant: parsed.quantization
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Drafts of a model, in the order they matter: what the server loads with, else what the
|
||||
* settings name, then any other sidecar the model's own repo ships.
|
||||
*/
|
||||
export function modelDraftsFor(option: ModelOption, settingValue?: string | null): ModelDraft[] {
|
||||
// speculative decoding is a llama.cpp feature
|
||||
if (!getBackendCapabilities(getBackend(option.backendId)).loadUnload) return [];
|
||||
|
||||
const args = modelsStore.routerModels.find((model) => model.id === option.model)?.status?.args;
|
||||
const configured = draftFromArgs(args, option) ?? draftFromSetting(option, settingValue);
|
||||
|
||||
return modelDrafts(option, sidecarFilesFor(option), configured);
|
||||
}
|
||||
|
||||
/** Draft sidecars a listing reported for the model's repo. */
|
||||
export function sidecarFilesFor(option: ModelOption): ModelSidecarFile[] {
|
||||
const repo = option.model.split(':')[0] ?? '';
|
||||
const state = backendsModelsStore.get(option.backendId ?? LOCAL_BACKEND_ID);
|
||||
|
||||
return state.drafts?.[repo] ?? [];
|
||||
}
|
||||
|
||||
/**
|
||||
* Drafts to show for a model: the one the load settings name, plus any draft sidecar
|
||||
* the model's own repo ships. The configured one is the active draft; a sidecar that
|
||||
* is merely on disk stays visible but idle.
|
||||
*/
|
||||
export function modelDrafts(
|
||||
option: ModelOption,
|
||||
available: ModelSidecarFile[] = [],
|
||||
configured?: ModelDraft | null
|
||||
): ModelDraft[] {
|
||||
const drafts: ModelDraft[] = [];
|
||||
|
||||
if (configured) drafts.push(configured);
|
||||
|
||||
for (const file of available) {
|
||||
// a sidecar a load already points at is the active draft, not a second entry
|
||||
if (drafts.some((draft) => draft.kind === file.kind)) continue;
|
||||
|
||||
drafts.push({
|
||||
active: false,
|
||||
kind: file.kind,
|
||||
model: null,
|
||||
params: file.params,
|
||||
quant: file.quant
|
||||
});
|
||||
}
|
||||
|
||||
return drafts;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a model has a capability. A listing that declares nothing is not a listing
|
||||
* that lacks it: the chat template the Hub carries says whether tools or reasoning work.
|
||||
*/
|
||||
export function modelSupports(option: ModelOption, capability: ModelCapability): boolean {
|
||||
if (option.capabilities.includes(capability)) return true;
|
||||
|
||||
@@ -123,6 +281,14 @@ export function modelSupports(option: ModelOption, capability: ModelCapability):
|
||||
: detectThinkingSupport(template);
|
||||
}
|
||||
|
||||
export function servedByLabel(option: ModelOption): string {
|
||||
const backend = getBackend(option.backendId);
|
||||
|
||||
if (!backend || backend.id === LOCAL_BACKEND_ID) return 'This server';
|
||||
|
||||
return backend.name;
|
||||
}
|
||||
|
||||
/** Modalities a model can accept, as the manager filter offers them. */
|
||||
export type ModalityKey = keyof ModelModalities;
|
||||
|
||||
@@ -138,6 +304,10 @@ export function modelContextLength(option: ModelOption): number {
|
||||
);
|
||||
}
|
||||
|
||||
export function isLocalOption(option: ModelOption): boolean {
|
||||
return (option.backendId ?? LOCAL_BACKEND_ID) === LOCAL_BACKEND_ID;
|
||||
}
|
||||
|
||||
/** File size of a local GGUF, when the router reported one. */
|
||||
export function modelSizeLabel(option: ModelOption): string | null {
|
||||
const bytes = option.meta?.size;
|
||||
@@ -170,7 +340,7 @@ export async function resolveModelSize(option: ModelOption): Promise<string | nu
|
||||
if (reported) return reported;
|
||||
|
||||
// the repo tree lookup only happens for installs that opted into the Hub
|
||||
if (!settingsStore.config[SETTINGS_KEYS.ENABLE_DISCOVER_MODELS]) {
|
||||
if (!isLocalOption(option) || !settingsStore.config[SETTINGS_KEYS.ENABLE_DISCOVER_MODELS]) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -198,12 +368,15 @@ export function modelRepoKey(model: string): string {
|
||||
}
|
||||
|
||||
/** Fold the quants of one repo into a single entry, so the table shows one row per model. */
|
||||
export function groupModelQuants(models: ModelOption[]): ModelQuantGroup[] {
|
||||
export function groupModelQuants(models: ModelOption[], mergeProviders = false): ModelQuantGroup[] {
|
||||
const groups = new SvelteMap<string, ModelQuantGroup>();
|
||||
|
||||
for (const option of models) {
|
||||
const repo = modelRepoKey(option.model);
|
||||
const key = repo;
|
||||
// groups stay within one backend, so the same repo served by two providers
|
||||
// is not read as two quants of one model. The OAI-compat block asks for the
|
||||
// opposite: one repo, one row per provider that serves it.
|
||||
const key = mergeProviders ? repo : `${option.backendId ?? ''}::${repo}`;
|
||||
const group = groups.get(key);
|
||||
|
||||
if (group) {
|
||||
@@ -219,7 +392,8 @@ export function groupModelQuants(models: ModelOption[]): ModelQuantGroup[] {
|
||||
const kind = groupKind(group.quants);
|
||||
const modelIds = group.quants.map((option) => option.model);
|
||||
|
||||
// the very same id twice is not a quant set; keep those rows apart
|
||||
// the very same id twice is not a quant set; keep those rows apart, unless
|
||||
// the group exists to list the providers that serve it
|
||||
if (
|
||||
kind !== 'providers' &&
|
||||
modelIds.length > 1 &&
|
||||
@@ -238,8 +412,12 @@ export function groupModelQuants(models: ModelOption[]): ModelQuantGroup[] {
|
||||
});
|
||||
}
|
||||
|
||||
/** What a group folds: quants or variants of one repo. */
|
||||
/** What a group folds: providers, quants, or variants of one repo. */
|
||||
function groupKind(quants: ModelOption[]): ModelGroupKind {
|
||||
const backends = new Set(quants.map((option) => option.backendId ?? ''));
|
||||
|
||||
if (backends.size > 1) return 'providers';
|
||||
|
||||
const isQuant = quants.every(
|
||||
(option) => (option.parsedId ?? ModelsService.parseModelId(option.model)).quantization
|
||||
);
|
||||
|
||||
@@ -1,2 +1,66 @@
|
||||
/** Marks the built-in local llama.cpp backend. */
|
||||
import type { BackendCapabilities, BackendCompat, BackendProtocol } from '$lib/types';
|
||||
|
||||
/** Prefix for generated ids of user-added backends. */
|
||||
export const BACKEND_ID_PREFIX = 'backend';
|
||||
|
||||
/** Protocols a configured backend can speak, in display order. */
|
||||
export const BACKEND_PROTOCOLS: readonly BackendProtocol[] = ['llama.cpp', 'openai'];
|
||||
|
||||
/** Chat completions path used when a backend does not override it. */
|
||||
export const DEFAULT_BACKEND_CHAT_PATH = '/v1/chat/completions';
|
||||
|
||||
/** Models listing path used when a backend does not override it. */
|
||||
export const DEFAULT_BACKEND_MODELS_PATH = '/v1/models';
|
||||
|
||||
/** Id of the built-in backend that points at the server serving this UI. */
|
||||
export const LOCAL_BACKEND_ID = 'local';
|
||||
|
||||
/** Capabilities of a full llama.cpp server. */
|
||||
const LLAMA_CPP_CAPABILITIES: BackendCapabilities = {
|
||||
corsProxy: true,
|
||||
loadUnload: true,
|
||||
props: true,
|
||||
resumableStreams: true,
|
||||
router: true,
|
||||
slots: true,
|
||||
statusFeed: true,
|
||||
tools: true
|
||||
};
|
||||
/** Capabilities of a plain OpenAI-compatible endpoint. */
|
||||
const COMPATIBLE_CAPABILITIES: BackendCapabilities = {
|
||||
corsProxy: false,
|
||||
loadUnload: false,
|
||||
props: false,
|
||||
resumableStreams: false,
|
||||
router: false,
|
||||
slots: false,
|
||||
statusFeed: false,
|
||||
tools: false
|
||||
};
|
||||
|
||||
/** Capabilities per backend protocol. */
|
||||
export const BACKEND_CAPABILITIES: Record<BackendProtocol, BackendCapabilities> = {
|
||||
'llama.cpp': LLAMA_CPP_CAPABILITIES,
|
||||
openai: COMPATIBLE_CAPABILITIES
|
||||
};
|
||||
|
||||
/** Default wire quirks per protocol. */
|
||||
export const BACKEND_COMPAT: Record<BackendProtocol, BackendCompat> = {
|
||||
// llama-server reports its own timings, so it needs no usage chunk
|
||||
'llama.cpp': { maxTokensField: 'max_tokens', supportsUsageInStreaming: false },
|
||||
openai: { maxTokensField: 'max_tokens', supportsUsageInStreaming: true }
|
||||
};
|
||||
|
||||
/**
|
||||
* Fields that may carry a model's context size in an OpenAI-compatible model
|
||||
* listing. Providers pick their own name, and most report nothing at all.
|
||||
*/
|
||||
export const MODEL_CONTEXT_LENGTH_FIELDS = [
|
||||
'context_length',
|
||||
'context_window',
|
||||
'max_context_length',
|
||||
'max_position_embeddings'
|
||||
] as const;
|
||||
|
||||
/** Favicon extract keyed by domain, for backends with no bundled mark. */
|
||||
export const FAVICON_SERVICE_URL = 'https://www.google.com/s2/favicons?domain=';
|
||||
|
||||
@@ -11,6 +11,7 @@ export const SETTINGS_KEYS = {
|
||||
API_KEY: 'apiKey',
|
||||
AUTO_MIC_ON_EMPTY: 'autoMicOnEmpty',
|
||||
BACKEND_SAMPLING: 'backend_sampling',
|
||||
BACKENDS: 'backends',
|
||||
CONVERSATION_TABS: 'conversationTabs',
|
||||
COPY_TEXT_ATTACHMENTS_AS_PLAIN_TEXT: 'copyTextAttachmentsAsPlainText',
|
||||
CUSTOM_CSS: 'customCss',
|
||||
@@ -32,6 +33,7 @@ export const SETTINGS_KEYS = {
|
||||
FULL_HEIGHT_CODE_BLOCKS: 'fullHeightCodeBlocks',
|
||||
GROUP_MODELS_BY_FAMILY: 'groupModelsByFamily',
|
||||
JS_SANDBOX_ENABLED: 'jsSandboxEnabled',
|
||||
LOCAL_BACKEND_ENABLED: 'localBackendEnabled',
|
||||
MAX_IMAGE_RESOLUTION: 'maxImageMPixels',
|
||||
MAX_TOKENS: 'max_tokens',
|
||||
MCP_REQUEST_TIMEOUT_SECONDS: 'mcpRequestTimeoutSeconds',
|
||||
|
||||
@@ -16,6 +16,9 @@ export const DB_APP_NAME_DEPRECATED = 'LlamacppWebui';
|
||||
|
||||
export const ALWAYS_ALLOWED_TOOLS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.alwaysAllowedTools`;
|
||||
|
||||
/** Id of the backend the selector and new requests target, restored on page load. */
|
||||
export const ACTIVE_BACKEND_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.activeBackend`;
|
||||
|
||||
/** Paused model download ids (`<repo>:<tag>`), restored on the next page load. */
|
||||
export const PAUSED_MODEL_DOWNLOADS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.pausedModelDownloads`;
|
||||
export const CONFIG_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.config`;
|
||||
@@ -34,6 +37,9 @@ export const MODEL_GROUP_OPEN_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.modelGroup
|
||||
/** Per-model load and inference overrides, keyed by backend-qualified model id. */
|
||||
export const MODEL_OVERRIDES_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.modelOverrides`;
|
||||
|
||||
/** Model the user picked last, kept across reloads. Stores `{ id, model }`. */
|
||||
export const SELECTED_MODEL_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.selectedModel`;
|
||||
|
||||
/** Recently used model ids, most recent first, backend-qualified. */
|
||||
export const RECENT_MODELS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.recentModels`;
|
||||
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
// while the tab was hidden. covers brief background pauses without thrashing live streams
|
||||
export const STREAM_VISIBILITY_KICK_MS = 3000;
|
||||
|
||||
// minimum gap between synthesized live timing updates for backends that do not
|
||||
// stream their own, keeps the per-chunk state updates cheap
|
||||
export const STREAM_LIVE_TIMINGS_INTERVAL_MS = 500;
|
||||
|
||||
// separator joining a conversation id and its per-model stream identity
|
||||
// suffix (conv::model) used by the server side replay buffer
|
||||
export const CONVERSATION_ID_SEPARATOR = '::';
|
||||
|
||||
@@ -0,0 +1,188 @@
|
||||
/**
|
||||
* BackendsService - Stateless backend connectivity checks and model listing
|
||||
*
|
||||
* Probes a backend's models endpoint to validate its URL and credentials, and
|
||||
* normalizes the response into the UI model shape. No reactive state;
|
||||
* consumed by the backends settings UI and the per-backend model cache.
|
||||
*/
|
||||
|
||||
import { API_MODELS, LOCAL_BACKEND_ID } from '$lib/constants';
|
||||
import { ModelsService } from '$lib/services/models.service';
|
||||
import type { ApiModelsListResponse, Backend, BackendProtocol, ModelOption } from '$lib/types';
|
||||
import { isAbortError } from '$lib/utils/abort';
|
||||
import { apiUrl } from '$lib/utils/api-base';
|
||||
import { getAuthHeadersForBackend } from '$lib/utils/api-headers';
|
||||
import { backendModelsUrl, readModelContextLength } from '$lib/utils/backend';
|
||||
|
||||
/** Models returned by a backend, plus the failure detail when the call fails. */
|
||||
export interface BackendModelsResult {
|
||||
error?: string;
|
||||
models: ModelOption[];
|
||||
ok: boolean;
|
||||
status: number | null;
|
||||
/** Untouched list payload of the local backend, kept so the router rows and their load statuses can be rebuilt without asking again. */
|
||||
raw?: ApiModelsListResponse;
|
||||
}
|
||||
|
||||
/** What probing a backend's endpoint said about it. */
|
||||
export interface BackendProbe {
|
||||
/** The endpoint refused the request for want of a key, so it wants one. */
|
||||
authRequired: boolean;
|
||||
protocol: BackendProtocol;
|
||||
}
|
||||
|
||||
/** Outcome of a backend connectivity check. */
|
||||
export interface BackendTestResult {
|
||||
error?: string;
|
||||
modelCount?: number;
|
||||
ok: boolean;
|
||||
status: number | null;
|
||||
}
|
||||
|
||||
export class BackendsService {
|
||||
/**
|
||||
* List the models a backend exposes on its models endpoint.
|
||||
*
|
||||
* @param backend - Backend to query. Does not need to be registered yet.
|
||||
* @param signal - Optional abort signal for a cancelled request.
|
||||
*/
|
||||
static async detectProtocol(backend: Backend): Promise<BackendProbe> {
|
||||
const base = backend.baseUrl.trim().replace(/\/+$/, '');
|
||||
|
||||
if (!base) return { authRequired: false, protocol: 'openai' };
|
||||
|
||||
try {
|
||||
// llama-server answers /props with its build and generation defaults; a
|
||||
// plain OpenAI-compatible endpoint answers 404 there, or not at all
|
||||
const response = await fetch(`${base}/props`, {
|
||||
headers: getAuthHeadersForBackend(backend),
|
||||
signal: AbortSignal.timeout(5000)
|
||||
});
|
||||
|
||||
// a llama-server behind a key refuses before it says anything else, while
|
||||
// an OpenAI-compatible endpoint has no /props to guard in the first place
|
||||
if (response.status === 401) {
|
||||
return { authRequired: true, protocol: 'llama.cpp' };
|
||||
}
|
||||
|
||||
if (!response.ok) return { authRequired: false, protocol: 'openai' };
|
||||
|
||||
const body = (await response.json()) as Record<string, unknown>;
|
||||
const isLlamaCpp =
|
||||
'default_generation_settings' in body || 'build_info' in body || body.role === 'router';
|
||||
|
||||
return {
|
||||
authRequired: false,
|
||||
protocol: isLlamaCpp ? 'llama.cpp' : 'openai'
|
||||
};
|
||||
} catch {
|
||||
return { authRequired: false, protocol: 'openai' };
|
||||
}
|
||||
}
|
||||
|
||||
static async listModels(backend: Backend, signal?: AbortSignal): Promise<BackendModelsResult> {
|
||||
// the local backend has no base URL; its models endpoint is base relative
|
||||
const url = backend.baseUrl.trim()
|
||||
? backendModelsUrl(backend)
|
||||
: apiUrl(API_MODELS.LIST, LOCAL_BACKEND_ID);
|
||||
|
||||
if (!backend.baseUrl.trim() && backend.id !== LOCAL_BACKEND_ID) {
|
||||
return { error: 'Backend URL is required', models: [], ok: false, status: null };
|
||||
}
|
||||
|
||||
try {
|
||||
const response = await fetch(url, {
|
||||
headers: getAuthHeadersForBackend(backend),
|
||||
signal
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
return {
|
||||
error: await describeFailure(response),
|
||||
models: [],
|
||||
ok: false,
|
||||
status: response.status
|
||||
};
|
||||
}
|
||||
|
||||
const body = (await response.json()) as { data?: unknown };
|
||||
const entries = Array.isArray(body?.data) ? body.data : [];
|
||||
const models = entries.flatMap((entry) => normalizeBackendModel(entry));
|
||||
// the local rows carry load status, external ones carry the context size
|
||||
const raw = body as ApiModelsListResponse;
|
||||
|
||||
return { models, ok: true, raw, status: response.status };
|
||||
} catch (error) {
|
||||
if (isAbortError(error)) {
|
||||
return { models: [], ok: false, status: null };
|
||||
}
|
||||
|
||||
return {
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
models: [],
|
||||
ok: false,
|
||||
status: null
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check that a backend answers on its models endpoint.
|
||||
*
|
||||
* @param backend - Backend to probe. Does not need to be registered yet.
|
||||
* @param signal - Optional abort signal for a cancelled test.
|
||||
*/
|
||||
static async test(backend: Backend, signal?: AbortSignal): Promise<BackendTestResult> {
|
||||
const result = await BackendsService.listModels(backend, signal);
|
||||
|
||||
return {
|
||||
error: result.error,
|
||||
modelCount: result.models.length,
|
||||
ok: result.ok,
|
||||
status: result.status
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/** Build a human-readable message from a non-OK response. */
|
||||
async function describeFailure(response: Response): Promise<string> {
|
||||
const status = `${response.status} ${response.statusText}`.trim();
|
||||
|
||||
try {
|
||||
const body = (await response.json()) as { error?: { message?: string }; message?: string };
|
||||
const message = body?.error?.message ?? body?.message;
|
||||
|
||||
if (message) return `${status}: ${message}`;
|
||||
} catch {
|
||||
// non-JSON error body, fall back to the status line
|
||||
}
|
||||
|
||||
return status;
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize one entry of an OpenAI-compatible `/v1/models` response. External
|
||||
* backends only guarantee an id, so that doubles as the display name.
|
||||
*/
|
||||
function normalizeBackendModel(entry: unknown): ModelOption[] {
|
||||
if (!entry || typeof entry !== 'object') return [];
|
||||
|
||||
const raw = entry as Record<string, unknown>;
|
||||
const id = typeof raw.id === 'string' ? raw.id.trim() : '';
|
||||
|
||||
if (!id) return [];
|
||||
|
||||
// a llama-compat server lists its projector and draft sidecars as models too
|
||||
if (ModelsService.isSidecarEntry(id)) return [];
|
||||
|
||||
return [
|
||||
{
|
||||
capabilities: [],
|
||||
contextLength: readModelContextLength(raw),
|
||||
id,
|
||||
model: id,
|
||||
name: id,
|
||||
status: raw.status as ApiModelDataEntry['status']
|
||||
}
|
||||
];
|
||||
}
|
||||
@@ -19,6 +19,14 @@
|
||||
*
|
||||
*/
|
||||
|
||||
/**
|
||||
* **BackendsService** - Backend connectivity checks
|
||||
*
|
||||
* Probes an external backend's models endpoint to validate its URL and
|
||||
* credentials before it is saved. Stateless.
|
||||
*/
|
||||
export { BackendsService } from './backends.service';
|
||||
|
||||
/**
|
||||
* **ChatService** - Chat Completions API communication layer
|
||||
*
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
/**
|
||||
* backendsStore - API endpoints the UI can talk to.
|
||||
*
|
||||
* The built-in local backend is the llama-server serving this UI. External
|
||||
* backends are user-configured endpoints persisted in settings. The store
|
||||
* registers the resolved list with the api-base registry, which services use
|
||||
* to build request URLs.
|
||||
*/
|
||||
|
||||
import { browser } from '$app/environment';
|
||||
import { ACTIVE_BACKEND_LOCALSTORAGE_KEY, LOCAL_BACKEND_ID, SETTINGS_KEYS } from '$lib/constants';
|
||||
import { serverStore } from '$lib/stores/server.svelte';
|
||||
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
||||
import type { Backend } from '$lib/types';
|
||||
import { setBackendsResolver } from '$lib/utils/api-base';
|
||||
import { createLocalBackend, parseBackendsSettings } from '$lib/utils/backend';
|
||||
|
||||
function loadActiveBackendId(): string {
|
||||
if (!browser) return LOCAL_BACKEND_ID;
|
||||
|
||||
try {
|
||||
return localStorage.getItem(ACTIVE_BACKEND_LOCALSTORAGE_KEY) ?? LOCAL_BACKEND_ID;
|
||||
} catch {
|
||||
return LOCAL_BACKEND_ID;
|
||||
}
|
||||
}
|
||||
|
||||
function persistActiveBackendId(backendId: string): void {
|
||||
if (!browser) return;
|
||||
|
||||
try {
|
||||
localStorage.setItem(ACTIVE_BACKEND_LOCALSTORAGE_KEY, backendId);
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
|
||||
class BackendsStore {
|
||||
activeId = $state<string>(loadActiveBackendId());
|
||||
|
||||
get active(): Backend {
|
||||
const active = this.enabled.find((backend) => backend.id === this.activeId);
|
||||
|
||||
return active ?? this.enabled[0] ?? this.local;
|
||||
}
|
||||
|
||||
get enabled(): Backend[] {
|
||||
return this.list.filter((backend) => backend.enabled && !this.isMissingLocal(backend));
|
||||
}
|
||||
|
||||
get external(): Backend[] {
|
||||
return parseBackendsSettings(settingsStore.config[SETTINGS_KEYS.BACKENDS]);
|
||||
}
|
||||
|
||||
get list(): Backend[] {
|
||||
return [this.local, ...this.external];
|
||||
}
|
||||
|
||||
get local(): Backend {
|
||||
return createLocalBackend(
|
||||
settingsStore.config.apiKey?.toString().trim() || undefined,
|
||||
settingsStore.config[SETTINGS_KEYS.LOCAL_BACKEND_ENABLED] !== false
|
||||
);
|
||||
}
|
||||
|
||||
addBackend(backend: Backend): void {
|
||||
this.saveExternal([...this.external, backend]);
|
||||
}
|
||||
|
||||
initialize(): void {
|
||||
if (!browser) return;
|
||||
|
||||
setBackendsResolver(() => ({ activeId: this.active.id, backends: this.list }));
|
||||
}
|
||||
|
||||
removeBackend(backendId: string): void {
|
||||
this.saveExternal(this.external.filter((backend) => backend.id !== backendId));
|
||||
|
||||
if (this.activeId === backendId) {
|
||||
this.setActive(LOCAL_BACKEND_ID);
|
||||
}
|
||||
}
|
||||
|
||||
setActive(backendId: string): void {
|
||||
const id = this.list.some((backend) => backend.id === backendId) ? backendId : LOCAL_BACKEND_ID;
|
||||
|
||||
this.activeId = id;
|
||||
persistActiveBackendId(id);
|
||||
}
|
||||
|
||||
setLocalEnabled(enabled: boolean): void {
|
||||
settingsStore.updateConfig(SETTINGS_KEYS.LOCAL_BACKEND_ENABLED, enabled);
|
||||
}
|
||||
|
||||
updateBackend(backendId: string, updates: Partial<Backend>): void {
|
||||
this.saveExternal(
|
||||
this.external.map((backend) =>
|
||||
backend.id === backendId ? { ...backend, ...updates } : backend
|
||||
)
|
||||
);
|
||||
}
|
||||
|
||||
/** The built-in backend counts only when a local server answered this session. */
|
||||
private isMissingLocal(backend: Backend): boolean {
|
||||
return backend.id === LOCAL_BACKEND_ID && serverStore.localServerMissing;
|
||||
}
|
||||
|
||||
private saveExternal(backends: Backend[]): void {
|
||||
settingsStore.updateConfig(SETTINGS_KEYS.BACKENDS, JSON.stringify(backends));
|
||||
}
|
||||
}
|
||||
|
||||
export const backendsStore = new BackendsStore();
|
||||
@@ -0,0 +1,103 @@
|
||||
/**
|
||||
* backendsModelsStore - Per-backend model catalog cache.
|
||||
*
|
||||
* Prefetches every enabled backend's model list, so switching backends is
|
||||
* instant and the switcher can show load state. The active backend's list
|
||||
* still lives in modelsStore, which owns selection and chat wiring; this
|
||||
* cache is the prefetch layer the switches start from.
|
||||
*/
|
||||
|
||||
import { BackendsService } from '$lib/services/backends.service';
|
||||
import { ModelsService } from '$lib/services/models.service';
|
||||
import { backendsStore } from '$lib/stores/backends.svelte';
|
||||
import type { Backend } from '$lib/types';
|
||||
import type { ApiModelsListResponse } from '$lib/types';
|
||||
import type { ModelSidecarFile } from '$lib/types/models';
|
||||
import type { ModelOption } from '$lib/types/models';
|
||||
|
||||
export interface BackendModelsState {
|
||||
/** Draft sidecars the listing carries, keyed by the repo they belong to. */
|
||||
drafts?: Record<string, ModelSidecarFile[]>;
|
||||
error: string | null;
|
||||
loaded: boolean;
|
||||
loading: boolean;
|
||||
models: ModelOption[];
|
||||
/** Untouched list payload, kept for the local backend so its router rows survive a tab switch. */
|
||||
raw?: ApiModelsListResponse;
|
||||
}
|
||||
|
||||
const EMPTY_STATE: BackendModelsState = {
|
||||
error: null,
|
||||
loaded: false,
|
||||
loading: false,
|
||||
models: []
|
||||
};
|
||||
|
||||
class BackendsModelsStore {
|
||||
private states = $state<Record<string, BackendModelsState>>({});
|
||||
|
||||
clear(backendId: string): void {
|
||||
delete this.states[backendId];
|
||||
}
|
||||
|
||||
/**
|
||||
* Load a backend's models once.
|
||||
*/
|
||||
async ensureLoaded(backendId: string): Promise<void> {
|
||||
const backend = backendsStore.enabled.find((candidate) => candidate.id === backendId);
|
||||
|
||||
if (!backend) return;
|
||||
|
||||
const state = this.states[backendId];
|
||||
|
||||
if (state?.loaded || state?.loading) return;
|
||||
|
||||
this.states[backendId] = { error: null, loaded: false, loading: true, models: [] };
|
||||
await this.fetch(backend);
|
||||
}
|
||||
|
||||
get(backendId: string): BackendModelsState {
|
||||
return this.states[backendId] ?? EMPTY_STATE;
|
||||
}
|
||||
|
||||
/** Prefetch every enabled backend's model list. */
|
||||
async loadAll(): Promise<void> {
|
||||
const enabled = backendsStore.enabled;
|
||||
const ids = new Set(enabled.map((backend) => backend.id));
|
||||
|
||||
for (const id of Object.keys(this.states)) {
|
||||
if (!ids.has(id)) {
|
||||
delete this.states[id];
|
||||
}
|
||||
}
|
||||
|
||||
await Promise.all(enabled.map((backend) => this.ensureLoaded(backend.id)));
|
||||
}
|
||||
|
||||
/**
|
||||
* Ask a backend for its list again, keeping what is already known. Used while
|
||||
* a remote load settles, since its status never reaches the local feed.
|
||||
*/
|
||||
async refresh(backendId: string): Promise<void> {
|
||||
const backend = backendsStore.enabled.find((candidate) => candidate.id === backendId);
|
||||
|
||||
if (!backend) return;
|
||||
|
||||
await this.fetch(backend);
|
||||
}
|
||||
|
||||
private async fetch(backend: Backend): Promise<void> {
|
||||
const result = await BackendsService.listModels(backend);
|
||||
|
||||
this.states[backend.id] = {
|
||||
drafts: result.raw ? ModelsService.draftSidecarsByRepo(result.raw) : undefined,
|
||||
error: result.error ?? null,
|
||||
loaded: result.ok,
|
||||
loading: false,
|
||||
models: result.models,
|
||||
raw: result.raw
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export const backendsModelsStore = new BackendsModelsStore();
|
||||
Vendored
+13
@@ -358,7 +358,9 @@ export interface ApiChatCompletionStreamChunk {
|
||||
metadata?: { model?: string };
|
||||
delta: {
|
||||
content?: string;
|
||||
reasoning?: string;
|
||||
reasoning_content?: string;
|
||||
reasoning_text?: string;
|
||||
model?: string;
|
||||
tool_calls?: ApiChatCompletionToolCallDelta[];
|
||||
};
|
||||
@@ -372,6 +374,17 @@ export interface ApiChatCompletionStreamChunk {
|
||||
cache_n?: number;
|
||||
};
|
||||
prompt_progress?: ChatMessagePromptProgress;
|
||||
/** Token counts, sent by OpenAI-compatible servers on the final chunk. */
|
||||
usage?: ApiChatCompletionUsage;
|
||||
}
|
||||
|
||||
export interface ApiChatCompletionUsage {
|
||||
cached_tokens?: number;
|
||||
completion_tokens?: number;
|
||||
prompt_cache_hit_tokens?: number;
|
||||
prompt_tokens?: number;
|
||||
prompt_tokens_details?: { cached_tokens?: number; cache_write_tokens?: number };
|
||||
total_tokens?: number;
|
||||
}
|
||||
|
||||
export interface ApiChatCompletionResponse {
|
||||
|
||||
Vendored
+98
@@ -0,0 +1,98 @@
|
||||
/**
|
||||
* Backend types.
|
||||
*
|
||||
* A backend is one API endpoint the UI can talk to. The built-in `local`
|
||||
* backend is the llama-server serving the UI. External backends are
|
||||
* user-configured endpoints that speak an OpenAI-compatible
|
||||
* protocol.
|
||||
*/
|
||||
|
||||
/** Request/response shape a backend speaks. */
|
||||
export type BackendProtocol = 'llama.cpp' | 'openai';
|
||||
|
||||
/**
|
||||
* Wire-level quirks of a backend's protocol. Capabilities gate llama.cpp
|
||||
* features; compat describes how the request and stream payloads differ.
|
||||
*/
|
||||
export interface BackendCompat {
|
||||
/** Field carrying the output token cap. */
|
||||
maxTokensField: 'max_completion_tokens' | 'max_tokens';
|
||||
/** Whether the endpoint accepts stream_options.include_usage. */
|
||||
supportsUsageInStreaming: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Features a backend supports. A llama.cpp server exposes extra endpoints on
|
||||
* top of the OpenAI-compatible API; plain OpenAI-compatible
|
||||
* endpoints only provide chat and model listing.
|
||||
*/
|
||||
export interface BackendCapabilities {
|
||||
/** llama-server's /cors-proxy endpoint for cross-origin MCP requests. */
|
||||
corsProxy: boolean;
|
||||
/** Router-mode model load/unload. */
|
||||
loadUnload: boolean;
|
||||
/** The /props endpoint with server role and generation defaults. */
|
||||
props: boolean;
|
||||
/** Resumable stream sessions (/v1/stream, /v1/streams/lookup). */
|
||||
resumableStreams: boolean;
|
||||
/** Multi-model router mode. */
|
||||
router: boolean;
|
||||
/** The /slots introspection endpoint. */
|
||||
slots: boolean;
|
||||
/** The /models/sse load and download progress feed. */
|
||||
statusFeed: boolean;
|
||||
/** The /tools listing and execution endpoint. */
|
||||
tools: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* One configured API endpoint.
|
||||
*
|
||||
* TODO: a backend paired by QR code is reached over WebRTC instead of plain
|
||||
* HTTP, so it needs a transport discriminator and its peer description here.
|
||||
*/
|
||||
export interface Backend {
|
||||
/** Bearer token / API key used for this backend. */
|
||||
apiKey?: string;
|
||||
/**
|
||||
* API root the endpoint paths are appended to, e.g. https://api.example.com.
|
||||
* Empty for the local backend, which resolves against the UI origin instead.
|
||||
*/
|
||||
baseUrl: string;
|
||||
/** Chat completions path override, e.g. /v1/messages. */
|
||||
chatPath?: string;
|
||||
/** Wire quirks overriding the protocol defaults. */
|
||||
compat?: Partial<BackendCompat>;
|
||||
/** Disabled backends stay configured but are not queried. */
|
||||
enabled: boolean;
|
||||
/** Extra headers merged into every request to this backend. */
|
||||
headers?: Record<string, string>;
|
||||
/** Stable identity. The local backend id is reserved. */
|
||||
id: string;
|
||||
/** Models listing path override, e.g. /models. */
|
||||
modelsPath?: string;
|
||||
name: string;
|
||||
protocol: BackendProtocol;
|
||||
}
|
||||
|
||||
/** A ready-made backend configuration offered when adding a backend. */
|
||||
export interface BackendPreset {
|
||||
/** Optional help text shown under the API key field. */
|
||||
apiKeyHelp?: string;
|
||||
baseUrl: string;
|
||||
/** One line describing the endpoint, shown on the preset card. */
|
||||
description?: string;
|
||||
chatPath?: string;
|
||||
/** Brand mark used in both themes, for logos that carry their own background. */
|
||||
iconUrl?: string;
|
||||
/** Brand mark for the dark theme. Preferred over `iconUrl` when paired with `iconUrlLight`. */
|
||||
iconUrlDark?: string;
|
||||
/** Brand mark for the light theme. Preferred over `iconUrl` when paired with `iconUrlDark`. */
|
||||
iconUrlLight?: string;
|
||||
/** Wire quirks this preset needs on top of the protocol defaults. */
|
||||
compat?: Partial<BackendCompat>;
|
||||
id: string;
|
||||
modelsPath?: string;
|
||||
name: string;
|
||||
protocol: BackendProtocol;
|
||||
}
|
||||
Vendored
+1
-1
@@ -334,6 +334,6 @@ export interface ChatFormActionsContext {
|
||||
readonly hasVideoModality: boolean;
|
||||
readonly hasVisionModality: boolean;
|
||||
onFileUpload?: () => void;
|
||||
onSystemPromptClick?: () => void;
|
||||
onMcpSettingsClick?: () => void;
|
||||
onSystemPromptClick?: () => void;
|
||||
}
|
||||
|
||||
@@ -25,6 +25,7 @@ export type {
|
||||
ApiChatCompletionToolCallDelta,
|
||||
ApiChatCompletionToolCall,
|
||||
ApiChatCompletionStreamChunk,
|
||||
ApiChatCompletionUsage,
|
||||
ApiChatCompletionResponse,
|
||||
ApiSlotData,
|
||||
ApiProcessingState,
|
||||
@@ -35,6 +36,15 @@ export type {
|
||||
ApiStreamSession
|
||||
} from './api';
|
||||
|
||||
// Backend types
|
||||
export type {
|
||||
Backend,
|
||||
BackendCapabilities,
|
||||
BackendCompat,
|
||||
BackendPreset,
|
||||
BackendProtocol
|
||||
} from './backend';
|
||||
|
||||
// HuggingFace types
|
||||
export type {
|
||||
HfCatalogBuild,
|
||||
|
||||
Vendored
+19
-15
@@ -16,6 +16,8 @@ export interface ModelOption {
|
||||
id: string;
|
||||
name: string;
|
||||
model: string;
|
||||
/** Backend that serves this model; set on the aggregated option list. */
|
||||
backendId?: string;
|
||||
description?: string;
|
||||
capabilities: string[];
|
||||
/** Context size reported by the provider's model listing, when it reports one. */
|
||||
@@ -23,6 +25,8 @@ export interface ModelOption {
|
||||
modalities?: ModelModalities;
|
||||
details?: ApiModelDetails['details'];
|
||||
meta?: ApiModelDataEntry['meta'];
|
||||
/** Load state the provider's own listing reports, when it reports one. */
|
||||
status?: ApiModelDataEntry['status'];
|
||||
parsedId?: ParsedModelId;
|
||||
aliases?: string[];
|
||||
tags?: string[];
|
||||
@@ -49,6 +53,21 @@ export interface ModelDownloadProgress {
|
||||
files: Record<string, ModelDownloadFileProgress>;
|
||||
}
|
||||
|
||||
/**
|
||||
* A draft sidecar file a model listing reports as its own entry. The router lists a
|
||||
* downloaded sidecar as a model, so this is what pairs it back with its model.
|
||||
*/
|
||||
export interface ModelSidecarFile {
|
||||
id: string;
|
||||
kind: ModelSidecar;
|
||||
/** Repo the sidecar belongs to, e.g. `ggml-org/Qwen3.6-35B-A3B-GGUF`. */
|
||||
model: string;
|
||||
/** Parameter count the sidecar reports, e.g. `35B-A3B`. */
|
||||
params: string | null;
|
||||
/** Quantization of the sidecar file, e.g. `Q4_0`. */
|
||||
quant: string | null;
|
||||
}
|
||||
|
||||
export interface ParsedModelId {
|
||||
raw: string;
|
||||
orgName: string | null;
|
||||
@@ -66,18 +85,3 @@ export interface ModalityCapabilities {
|
||||
hasAudio: boolean;
|
||||
hasVideo: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* A draft sidecar file a model listing reports as its own entry. The router lists a
|
||||
* downloaded sidecar as a model, so this is what pairs it back with its model.
|
||||
*/
|
||||
export interface ModelSidecarFile {
|
||||
id: string;
|
||||
kind: ModelSidecar;
|
||||
/** Repo the sidecar belongs to, e.g. `ggml-org/Qwen3.6-35B-A3B-GGUF`. */
|
||||
model: string;
|
||||
/** Parameter count the sidecar reports, e.g. `35B-A3B`. */
|
||||
params: string | null;
|
||||
/** Quantization of the sidecar file, e.g. `Q4_0`. */
|
||||
quant: string | null;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
/**
|
||||
* API base resolution for backends.
|
||||
*
|
||||
* The UI can talk to more than one backend endpoint. Services build request
|
||||
* URLs through {@link apiUrl} so a request always targets the right backend.
|
||||
* The backends store registers a resolver here; this module never imports the
|
||||
* store, which keeps URL resolution free of store dependencies.
|
||||
*/
|
||||
|
||||
import { backendChatUrl, backendModelsUrl } from './backend';
|
||||
import { base } from '$app/paths';
|
||||
import { API_ABSOLUTE_URL_PROTOCOLS, API_CHAT, API_MODELS } from '$lib/constants';
|
||||
import type { Backend } from '$lib/types';
|
||||
|
||||
/** Backend list and active selection as exposed to URL resolution. */
|
||||
export interface BackendsSnapshot {
|
||||
activeId: string;
|
||||
backends: Backend[];
|
||||
}
|
||||
|
||||
type BackendsResolver = () => BackendsSnapshot;
|
||||
|
||||
let resolveBackends: BackendsResolver | null = null;
|
||||
|
||||
/** Registered once by the backends store. */
|
||||
export function setBackendsResolver(resolver: BackendsResolver | null): void {
|
||||
resolveBackends = resolver;
|
||||
}
|
||||
|
||||
/**
|
||||
* Look up a backend by id, defaulting to the active one.
|
||||
*/
|
||||
export function getBackend(backendId?: string): Backend | undefined {
|
||||
const snapshot = resolveBackends?.();
|
||||
|
||||
if (!snapshot) return undefined;
|
||||
|
||||
const id = backendId ?? snapshot.activeId;
|
||||
|
||||
return snapshot.backends.find((backend) => backend.id === id);
|
||||
}
|
||||
|
||||
/** API root for a backend, or an empty string for the local backend. */
|
||||
export function getBackendBaseUrl(backendId?: string): string {
|
||||
return getBackend(backendId)?.baseUrl.trim() ?? '';
|
||||
}
|
||||
|
||||
/**
|
||||
* Request target for a backend's chat completions endpoint. Returns a relative
|
||||
* path for the local backend and an absolute URL for external ones, so callers
|
||||
* can pass the result straight to `fetch` (or `apiFetch`, which resolves
|
||||
* relative paths against the base).
|
||||
*/
|
||||
export function apiChatUrl(backendId?: string): string {
|
||||
const backend = getBackend(backendId);
|
||||
|
||||
if (backend?.baseUrl.trim()) {
|
||||
return backendChatUrl(backend);
|
||||
}
|
||||
|
||||
return apiUrl(API_CHAT.COMPLETIONS, backendId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Request target for a backend's models listing. Returns a plain path for the
|
||||
* local backend (so `apiFetch` applies the base) and an absolute URL otherwise.
|
||||
*/
|
||||
export function apiModelsUrl(backendId?: string): string {
|
||||
const backend = getBackend(backendId);
|
||||
|
||||
if (backend?.baseUrl.trim()) {
|
||||
return backendModelsUrl(backend);
|
||||
}
|
||||
|
||||
return API_MODELS.LIST;
|
||||
}
|
||||
|
||||
/**
|
||||
* Absolute URL for an API path on a backend.
|
||||
*
|
||||
* Absolute URLs pass through untouched. Paths on the local backend keep the
|
||||
* existing base-path-relative form, so serving under a subpath still works.
|
||||
* Paths on external backends resolve against the backend's API root.
|
||||
*/
|
||||
export function apiUrl(path: string, backendId?: string): string {
|
||||
if (API_ABSOLUTE_URL_PROTOCOLS.some((protocol) => path.startsWith(protocol))) {
|
||||
return path;
|
||||
}
|
||||
|
||||
const baseUrl = getBackendBaseUrl(backendId);
|
||||
|
||||
if (!baseUrl) {
|
||||
return `${base}${path}`;
|
||||
}
|
||||
|
||||
const root = baseUrl.endsWith('/') ? baseUrl : `${baseUrl}/`;
|
||||
|
||||
return new URL(path.replace(/^\.?\//, ''), root).toString();
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
import { apiUrl } from './api-base';
|
||||
import { getAuthHeaders, getJsonHeaders } from './api-headers';
|
||||
import { base } from '$app/paths';
|
||||
import { API_ABSOLUTE_URL_PROTOCOLS, ERROR_MESSAGES, HTTP_CODE_TO_STRING } from '$lib/constants';
|
||||
import { ERROR_MESSAGES, HTTP_CODE_TO_STRING } from '$lib/constants';
|
||||
|
||||
/**
|
||||
* API Fetch Utilities
|
||||
@@ -32,6 +32,8 @@ export interface ApiFetchOptions extends Omit<RequestInit, 'headers'> {
|
||||
* Default: false (uses JSON headers with Content-Type: application/json)
|
||||
*/
|
||||
authOnly?: boolean;
|
||||
/** Backend to target; defaults to the active one. */
|
||||
backendId?: string;
|
||||
/**
|
||||
* Additional headers to merge with default headers.
|
||||
*/
|
||||
@@ -59,11 +61,10 @@ export interface ApiFetchOptions extends Omit<RequestInit, 'headers'> {
|
||||
* ```
|
||||
*/
|
||||
export async function apiFetch<T>(path: string, options: ApiFetchOptions = {}): Promise<T> {
|
||||
const { authOnly = false, headers: customHeaders, ...fetchOptions } = options;
|
||||
const baseHeaders = authOnly ? getAuthHeaders() : getJsonHeaders();
|
||||
const { authOnly = false, backendId, headers: customHeaders, ...fetchOptions } = options;
|
||||
const baseHeaders = authOnly ? getAuthHeaders(backendId) : getJsonHeaders(backendId);
|
||||
const headers = { ...baseHeaders, ...customHeaders };
|
||||
// absolute URLs with an allowed protocol pass through untouched; relative paths get the base prefix
|
||||
const url = API_ABSOLUTE_URL_PROTOCOLS.some((p) => path.startsWith(p)) ? path : `${base}${path}`;
|
||||
const url = apiUrl(path, backendId);
|
||||
|
||||
let response;
|
||||
|
||||
@@ -106,7 +107,7 @@ export async function apiFetchWithParams<T>(
|
||||
params: Record<string, string>,
|
||||
options: ApiFetchOptions = {}
|
||||
): Promise<T> {
|
||||
const url = new URL(basePath, window.location.href);
|
||||
const url = new URL(apiUrl(basePath, options.backendId), window.location.href);
|
||||
|
||||
for (const [key, value] of Object.entries(params)) {
|
||||
if (value !== undefined && value !== null) {
|
||||
|
||||
@@ -1,26 +1,42 @@
|
||||
import { getBackend } from './api-base';
|
||||
import { redactValue } from './redact';
|
||||
import { CORS_PROXY, HEADERS } from '$lib/constants';
|
||||
import { MimeTypeApplication } from '$lib/enums';
|
||||
import { getProtocolAdapter } from '$lib/services/protocols';
|
||||
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
||||
import type { Backend } from '$lib/types';
|
||||
|
||||
/**
|
||||
* Get authorization headers for API requests
|
||||
* Includes Bearer token if API key is configured
|
||||
* Get authorization headers for API requests to a backend.
|
||||
*/
|
||||
export function getAuthHeaders(): Record<string, string> {
|
||||
const currentConfig = settingsStore.config;
|
||||
const apiKey = currentConfig.apiKey?.toString().trim();
|
||||
export function getAuthHeaders(backendId?: string): Record<string, string> {
|
||||
const backend = getBackend(backendId);
|
||||
|
||||
if (backend) return getAuthHeadersForBackend(backend);
|
||||
|
||||
// no backends resolver yet (early startup, or a non-browser call): keep the
|
||||
// pre-backends behaviour and authenticate against the serving origin
|
||||
const apiKey = settingsStore.config.apiKey?.toString().trim();
|
||||
|
||||
return apiKey ? { [HEADERS.AUTHORIZATION]: `${HEADERS.BEARER}${apiKey}` } : {};
|
||||
}
|
||||
|
||||
/**
|
||||
* Get authorization headers for a backend object, including one that is not
|
||||
* registered yet (used by the connection test on the add-backend form).
|
||||
* The protocol adapter owns the credential scheme and any required headers.
|
||||
*/
|
||||
export function getAuthHeadersForBackend(backend: Backend): Record<string, string> {
|
||||
return getProtocolAdapter(backend).authHeaders(backend);
|
||||
}
|
||||
|
||||
/**
|
||||
* Get standard JSON headers with optional authorization
|
||||
*/
|
||||
export function getJsonHeaders(): Record<string, string> {
|
||||
export function getJsonHeaders(backendId?: string): Record<string, string> {
|
||||
return {
|
||||
[HEADERS.CONTENT_TYPE]: MimeTypeApplication.JSON,
|
||||
...getAuthHeaders()
|
||||
...getAuthHeaders(backendId)
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
import { error } from '@sveltejs/kit';
|
||||
import { browser } from '$app/environment';
|
||||
import { base } from '$app/paths';
|
||||
import { HEADERS } from '$lib/constants';
|
||||
import { MimeTypeApplication } from '$lib/enums';
|
||||
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
||||
import { apiUrl, getBackend } from '$lib/utils/api-base';
|
||||
import { getBackendCapabilities } from '$lib/utils/backend';
|
||||
|
||||
/**
|
||||
* Validates API key by making a request to the server props endpoint
|
||||
@@ -14,6 +15,13 @@ export async function validateApiKey(fetch: typeof globalThis.fetch): Promise<vo
|
||||
return;
|
||||
}
|
||||
|
||||
// /props only exists on llama.cpp servers; external backends carry their own key
|
||||
const backend = getBackend();
|
||||
|
||||
if (backend && !getBackendCapabilities(backend).props) {
|
||||
return;
|
||||
}
|
||||
|
||||
const apiKey = settingsStore.config.apiKey;
|
||||
|
||||
try {
|
||||
@@ -28,7 +36,7 @@ export async function validateApiKey(fetch: typeof globalThis.fetch): Promise<vo
|
||||
headers[HEADERS.AUTHORIZATION] = `${HEADERS.BEARER}${apiKey}`;
|
||||
}
|
||||
|
||||
const response = await fetch(`${base}/props`, { headers });
|
||||
const response = await fetch(apiUrl('/props'), { headers });
|
||||
|
||||
if (!response.ok) {
|
||||
if (response.status === 401 || response.status === 403) {
|
||||
|
||||
@@ -0,0 +1,243 @@
|
||||
/**
|
||||
* Backend list parsing, defaults and endpoint URLs.
|
||||
*
|
||||
* External backends are persisted in settings as a JSON list. Malformed
|
||||
* entries are dropped instead of throwing so a corrupted settings value can
|
||||
* never break URL resolution.
|
||||
*/
|
||||
|
||||
import {
|
||||
BACKEND_CAPABILITIES,
|
||||
BACKEND_COMPAT,
|
||||
BACKEND_ID_PREFIX,
|
||||
BACKEND_PROTOCOLS,
|
||||
DEFAULT_BACKEND_CHAT_PATH,
|
||||
DEFAULT_BACKEND_MODELS_PATH,
|
||||
FAVICON_SERVICE_URL,
|
||||
LOCAL_BACKEND_ID,
|
||||
MODEL_CONTEXT_LENGTH_FIELDS
|
||||
} from '$lib/constants';
|
||||
import type { Backend, BackendCapabilities, BackendCompat, BackendProtocol } from '$lib/types';
|
||||
|
||||
/** Absolute chat completions URL for a backend. */
|
||||
export function backendChatUrl(backend: Backend): string {
|
||||
return joinBackendUrl(backend.baseUrl, backend.chatPath ?? DEFAULT_BACKEND_CHAT_PATH);
|
||||
}
|
||||
|
||||
/** Absolute models listing URL for a backend. */
|
||||
export function backendModelsUrl(backend: Backend): string {
|
||||
return joinBackendUrl(backend.baseUrl, backend.modelsPath ?? DEFAULT_BACKEND_MODELS_PATH);
|
||||
}
|
||||
|
||||
/** Icon size requested from the favicon service, shown at 16px. */
|
||||
const FAVICON_SIZE = 64;
|
||||
|
||||
/**
|
||||
* Favicon of a backend's root domain, used when no bundled mark matches. API
|
||||
* hosts rarely serve a favicon themselves, so the subdomain is dropped
|
||||
* (`api.z.ai` -> `z.ai`) and the icon is requested through a favicon service.
|
||||
* Returns null when the URL carries no usable host.
|
||||
*/
|
||||
export function backendFaviconUrl(baseUrl: string): string | null {
|
||||
try {
|
||||
const labels = new URL(baseUrl).hostname.split('.');
|
||||
const root = labels.length > 2 ? labels.slice(-2).join('.') : labels.join('.');
|
||||
|
||||
return root ? `${FAVICON_SERVICE_URL}${root}&sz=${FAVICON_SIZE}` : null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Features a backend supports, derived from its protocol. A missing backend
|
||||
* (unknown model, early startup) gets the plain compatible defaults.
|
||||
*/
|
||||
export function getBackendCapabilities(backend?: Backend): BackendCapabilities {
|
||||
return BACKEND_CAPABILITIES[backend?.protocol ?? 'openai'] ?? BACKEND_CAPABILITIES.openai;
|
||||
}
|
||||
|
||||
/** Wire quirks for a backend: protocol defaults overridden by the backend. */
|
||||
export function getBackendCompat(backend: Backend): BackendCompat {
|
||||
return { ...(BACKEND_COMPAT[backend.protocol] ?? BACKEND_COMPAT.openai), ...backend.compat };
|
||||
}
|
||||
|
||||
/** The built-in backend pointing at the server that serves this UI. */
|
||||
export function createLocalBackend(apiKey?: string, enabled = true): Backend {
|
||||
return {
|
||||
apiKey,
|
||||
baseUrl: '',
|
||||
enabled,
|
||||
id: LOCAL_BACKEND_ID,
|
||||
name: 'Local',
|
||||
protocol: 'llama.cpp'
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the persisted backends JSON into backend entries.
|
||||
*/
|
||||
export function parseBackendsSettings(rawBackends: unknown): Backend[] {
|
||||
if (!rawBackends) return [];
|
||||
|
||||
let parsed: unknown;
|
||||
|
||||
if (typeof rawBackends === 'string') {
|
||||
const trimmed = rawBackends.trim();
|
||||
|
||||
if (!trimmed) return [];
|
||||
|
||||
try {
|
||||
parsed = JSON.parse(trimmed);
|
||||
} catch (error) {
|
||||
console.warn('[backends] Failed to parse backends JSON, ignoring value:', error);
|
||||
|
||||
return [];
|
||||
}
|
||||
} else {
|
||||
parsed = rawBackends;
|
||||
}
|
||||
|
||||
if (!Array.isArray(parsed)) return [];
|
||||
|
||||
return parsed.flatMap((entry, index) => {
|
||||
const backend = parseBackendEntry(entry, index);
|
||||
|
||||
return backend ? [backend] : [];
|
||||
});
|
||||
}
|
||||
|
||||
function joinBackendUrl(baseUrl: string, path: string): string {
|
||||
const base = baseUrl.replace(/\/+$/, '');
|
||||
const suffix = path.startsWith('/') ? path : `/${path}`;
|
||||
|
||||
return `${base}${suffix}`;
|
||||
}
|
||||
|
||||
function parseBackendEntry(entry: unknown, index: number): Backend | null {
|
||||
if (!entry || typeof entry !== 'object') return null;
|
||||
|
||||
const raw = entry as Record<string, unknown>;
|
||||
const baseUrl = typeof raw.baseUrl === 'string' ? raw.baseUrl.trim() : '';
|
||||
|
||||
// the local backend is built in and never persisted
|
||||
if (!baseUrl || raw.id === LOCAL_BACKEND_ID) return null;
|
||||
|
||||
const protocol = BACKEND_PROTOCOLS.includes(raw.protocol as BackendProtocol)
|
||||
? (raw.protocol as BackendProtocol)
|
||||
: 'openai';
|
||||
const id =
|
||||
typeof raw.id === 'string' && raw.id.trim()
|
||||
? raw.id.trim()
|
||||
: `${BACKEND_ID_PREFIX}-${index + 1}`;
|
||||
const name = typeof raw.name === 'string' && raw.name.trim() ? raw.name.trim() : baseUrl;
|
||||
const apiKey =
|
||||
typeof raw.apiKey === 'string' && raw.apiKey.trim() ? raw.apiKey.trim() : undefined;
|
||||
|
||||
return {
|
||||
apiKey,
|
||||
baseUrl,
|
||||
chatPath: parseOptionalPath(raw.chatPath),
|
||||
compat: parseBackendCompat(raw.compat, protocol),
|
||||
enabled: raw.enabled !== false,
|
||||
headers: parseBackendHeaders(raw.headers),
|
||||
id,
|
||||
modelsPath: parseOptionalPath(raw.modelsPath),
|
||||
name,
|
||||
protocol
|
||||
};
|
||||
}
|
||||
|
||||
// only keep the override keys the protocol understands, so a stale persisted
|
||||
// value can never inject an unknown field into a request
|
||||
function parseBackendCompat(
|
||||
raw: unknown,
|
||||
protocol: BackendProtocol
|
||||
): Partial<BackendCompat> | undefined {
|
||||
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) return undefined;
|
||||
|
||||
const entry = raw as Record<string, unknown>;
|
||||
const defaults = BACKEND_COMPAT[protocol] ?? BACKEND_COMPAT.openai;
|
||||
const overrides: Partial<BackendCompat> = {};
|
||||
|
||||
if (entry.maxTokensField === 'max_tokens' || entry.maxTokensField === 'max_completion_tokens') {
|
||||
overrides.maxTokensField = entry.maxTokensField;
|
||||
}
|
||||
|
||||
if (typeof entry.supportsUsageInStreaming === 'boolean') {
|
||||
overrides.supportsUsageInStreaming = entry.supportsUsageInStreaming;
|
||||
}
|
||||
|
||||
// drop a no-op override so an unmodified backend stays undefined
|
||||
const isDefault =
|
||||
(overrides.maxTokensField === undefined ||
|
||||
overrides.maxTokensField === defaults.maxTokensField) &&
|
||||
(overrides.supportsUsageInStreaming === undefined ||
|
||||
overrides.supportsUsageInStreaming === defaults.supportsUsageInStreaming);
|
||||
|
||||
return isDefault ? undefined : overrides;
|
||||
}
|
||||
|
||||
function parseBackendHeaders(raw: unknown): Record<string, string> | undefined {
|
||||
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) return undefined;
|
||||
|
||||
const entries = Object.entries(raw as Record<string, unknown>)
|
||||
.filter(([, value]) => typeof value === 'string' && value.trim() !== '')
|
||||
.map(([key, value]) => [key.trim(), (value as string).trim()] as const);
|
||||
|
||||
return entries.length > 0 ? Object.fromEntries(entries) : undefined;
|
||||
}
|
||||
|
||||
function parseOptionalPath(raw: unknown): string | undefined {
|
||||
return typeof raw === 'string' && raw.trim() ? raw.trim() : undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Read a model's context size out of one `/v1/models` entry. Providers use
|
||||
* different field names, and OpenRouter nests the authoritative value under
|
||||
* `top_provider`, so try the flat fields first and the nested one after.
|
||||
*/
|
||||
export function readModelContextLength(value: unknown): number | undefined {
|
||||
if (!value || typeof value !== 'object') return undefined;
|
||||
|
||||
const entry = value as Record<string, unknown>;
|
||||
const flat = readContextField(entry);
|
||||
|
||||
if (flat !== undefined) return flat;
|
||||
|
||||
const topProvider = entry.top_provider;
|
||||
|
||||
if (topProvider && typeof topProvider === 'object') {
|
||||
const nested = readContextField(topProvider as Record<string, unknown>);
|
||||
|
||||
if (nested !== undefined) return nested;
|
||||
}
|
||||
|
||||
// Hugging Face lists one entry per inference provider, and they disagree on
|
||||
// the budget; take the largest so the gauge does not undersell the model.
|
||||
const providers = entry.providers;
|
||||
|
||||
if (Array.isArray(providers)) {
|
||||
const sizes = providers
|
||||
.map((provider) =>
|
||||
provider && typeof provider === 'object'
|
||||
? readContextField(provider as Record<string, unknown>)
|
||||
: undefined
|
||||
)
|
||||
.filter((size): size is number => size !== undefined);
|
||||
|
||||
if (sizes.length > 0) return Math.max(...sizes);
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function readContextField(entry: Record<string, unknown>): number | undefined {
|
||||
for (const field of MODEL_CONTEXT_LENGTH_FIELDS) {
|
||||
const value = entry[field];
|
||||
|
||||
if (typeof value === 'number' && value > 0) return value;
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
@@ -2,7 +2,7 @@
|
||||
* CORS Proxy utility for routing requests through llama-server's CORS proxy.
|
||||
*/
|
||||
|
||||
import { base } from '$app/paths';
|
||||
import { apiUrl } from './api-base';
|
||||
import { CORS_PROXY, CORS_PROXY_ENDPOINT } from '$lib/constants';
|
||||
|
||||
/**
|
||||
@@ -11,7 +11,7 @@ import { CORS_PROXY, CORS_PROXY_ENDPOINT } from '$lib/constants';
|
||||
* @returns URL pointing to the CORS proxy with target encoded
|
||||
*/
|
||||
export function buildProxiedUrl(targetUrl: string): URL {
|
||||
const proxyPath = `${base}${CORS_PROXY_ENDPOINT}`;
|
||||
const proxyPath = apiUrl(CORS_PROXY_ENDPOINT);
|
||||
const proxyUrl = new URL(proxyPath, window.location.origin);
|
||||
|
||||
proxyUrl.searchParams.set(CORS_PROXY.URL_PARAM, targetUrl);
|
||||
|
||||
@@ -8,9 +8,30 @@
|
||||
*/
|
||||
|
||||
// API utilities
|
||||
export { getAuthHeaders, getJsonHeaders, sanitizeHeaders } from './api-headers';
|
||||
export {
|
||||
apiChatUrl,
|
||||
apiModelsUrl,
|
||||
apiUrl,
|
||||
getBackend,
|
||||
getBackendBaseUrl,
|
||||
type BackendsSnapshot
|
||||
} from './api-base';
|
||||
export {
|
||||
getAuthHeaders,
|
||||
getAuthHeadersForBackend,
|
||||
getJsonHeaders,
|
||||
sanitizeHeaders
|
||||
} from './api-headers';
|
||||
export { ApiError, apiDelete, apiFetch, apiFetchWithParams, apiPost } from './api-fetch';
|
||||
export { validateApiKey } from './api-key-validation';
|
||||
export {
|
||||
backendChatUrl,
|
||||
backendFaviconUrl,
|
||||
backendModelsUrl,
|
||||
createLocalBackend,
|
||||
getBackendCapabilities,
|
||||
parseBackendsSettings
|
||||
} from './backend';
|
||||
|
||||
// Attachment utilities
|
||||
export { getAttachmentDisplayItems, isMcpPrompt, isMcpResource } from './attachment-display';
|
||||
@@ -107,6 +128,10 @@ export {
|
||||
// Model name utilities
|
||||
export { isValidModelName, normalizeModelName, orgOf } from './model-names';
|
||||
|
||||
// Backend-qualified model option ids
|
||||
export { backendIdFromModelId, qualifyModelId, rawModelId } from './model-option-id';
|
||||
export { groupModelFamilies, modelFamilyKey, type ModelFamilyGroup } from './model-families';
|
||||
|
||||
// Sidecar token utilities
|
||||
export { isAuxSidecar, isDraftSidecar, sidecarFromFileToken, sidecarFromTag } from './sidecars';
|
||||
|
||||
@@ -138,6 +163,8 @@ export { extractSseDataPayload, parseSseJsonStream, splitSseRecords } from './ss
|
||||
// Stream session identity (conversation-id based)
|
||||
export { streamIdentity } from './stream-identity';
|
||||
|
||||
export { buildTimingsFromUsage, usageTokenCounts } from './timings';
|
||||
|
||||
// MCP utilities
|
||||
export {
|
||||
detectMcpTransportFromUrl,
|
||||
@@ -360,4 +387,3 @@ export { remToPx } from './css';
|
||||
|
||||
// Audio format helper (used by agentic store and chat service)
|
||||
export { getAudioInputFormat } from './audio-format';
|
||||
export { groupModelFamilies, modelFamilyKey, type ModelFamilyGroup } from './model-families';
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
/**
|
||||
* Backend-qualified model option ids.
|
||||
*
|
||||
* The selector lists models from every enabled backend, so a bare model id can
|
||||
* collide. Option ids are qualified as `<backendId>::<rawModelId>`; selection
|
||||
* strips the prefix to find the row in the active backend's list.
|
||||
*/
|
||||
|
||||
const MODEL_OPTION_ID_SEPARATOR = '::';
|
||||
|
||||
/** Prefix a raw model id with the backend that serves it. */
|
||||
export function qualifyModelId(backendId: string, modelId: string): string {
|
||||
return `${backendId}${MODEL_OPTION_ID_SEPARATOR}${modelId}`;
|
||||
}
|
||||
|
||||
/** Backend id of a qualified model option id, or null when unqualified. */
|
||||
export function backendIdFromModelId(qualifiedId: string): string | null {
|
||||
const index = qualifiedId.indexOf(MODEL_OPTION_ID_SEPARATOR);
|
||||
|
||||
return index === -1 ? null : qualifiedId.slice(0, index);
|
||||
}
|
||||
|
||||
/** Strip the backend prefix from a qualified model option id. */
|
||||
export function rawModelId(qualifiedId: string): string {
|
||||
const index = qualifiedId.indexOf(MODEL_OPTION_ID_SEPARATOR);
|
||||
|
||||
return index === -1 ? qualifiedId : qualifiedId.slice(index + MODEL_OPTION_ID_SEPARATOR.length);
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
/**
|
||||
* Client side timing fallback for backends that do not report their own.
|
||||
*
|
||||
* llama.cpp streams per-token timings; OpenAI-compatible servers
|
||||
* do not. Token counts come from the usage block of the final chunk (or the
|
||||
* count of streamed deltas as a fallback), times are measured locally: the wait
|
||||
* for the first token is attributed to prompt processing, the rest to
|
||||
* generation. Wall clock, so network and queueing are part of the numbers.
|
||||
*/
|
||||
|
||||
import type { ApiChatCompletionUsage } from '$lib/types/api';
|
||||
import type { ChatMessageTimings } from '$lib/types/chat';
|
||||
|
||||
export interface StreamClock {
|
||||
startedAt: number;
|
||||
firstTokenAt: number | null;
|
||||
lastTokenAt: number | null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Prompt/output/cache token counts. `promptTokens` excludes the cache read
|
||||
* tokens, which are returned separately as `cacheTokens`, so the two always
|
||||
* add up to the prompt size.
|
||||
*/
|
||||
export function usageTokenCounts(usage: ApiChatCompletionUsage | undefined): {
|
||||
cacheTokens: number;
|
||||
completionTokens: number;
|
||||
promptTokens: number;
|
||||
} {
|
||||
// a total that includes the cache reads, which are reported separately
|
||||
const cacheTokens =
|
||||
usage?.prompt_tokens_details?.cached_tokens ??
|
||||
usage?.prompt_cache_hit_tokens ??
|
||||
usage?.cached_tokens ??
|
||||
0;
|
||||
const promptTotal = usage?.prompt_tokens ?? 0;
|
||||
|
||||
return {
|
||||
cacheTokens,
|
||||
completionTokens: usage?.completion_tokens ?? 0,
|
||||
promptTokens: Math.max(0, promptTotal - cacheTokens)
|
||||
};
|
||||
}
|
||||
|
||||
export function buildTimingsFromUsage(
|
||||
usage: ApiChatCompletionUsage | undefined,
|
||||
clock: StreamClock,
|
||||
fallbackTokens = 0
|
||||
): ChatMessageTimings | null {
|
||||
const { cacheTokens, completionTokens, promptTokens } = usageTokenCounts(usage);
|
||||
const predictedN = completionTokens || fallbackTokens;
|
||||
|
||||
if (promptTokens === 0 && predictedN === 0) return null;
|
||||
|
||||
const { firstTokenAt, startedAt } = clock;
|
||||
const lastTokenAt = clock.lastTokenAt ?? firstTokenAt;
|
||||
|
||||
return {
|
||||
cache_n: cacheTokens,
|
||||
// clamp so a one-token reply still reports a positive duration
|
||||
predicted_ms: firstTokenAt && lastTokenAt ? Math.max(1, lastTokenAt - firstTokenAt) : undefined,
|
||||
predicted_n: predictedN,
|
||||
prompt_ms: firstTokenAt ? Math.max(1, firstTokenAt - startedAt) : undefined,
|
||||
prompt_n: promptTokens
|
||||
};
|
||||
}
|
||||
Reference in New Issue
Block a user