diff --git a/tools/ui/src/app.d.ts b/tools/ui/src/app.d.ts index 5c039063ad..c3d210c9e6 100644 --- a/tools/ui/src/app.d.ts +++ b/tools/ui/src/app.d.ts @@ -12,6 +12,7 @@ import type { ApiChatCompletionStreamChunk, ApiChatCompletionToolCall, ApiChatCompletionToolCallDelta, + ApiChatCompletionUsage, ApiChatMessageContentPart, ApiChatMessageData, ApiContextSizeError, @@ -74,6 +75,7 @@ declare global { ApiChatCompletionResponse, ApiChatCompletionStreamChunk, ApiChatCompletionToolCall, + ApiChatCompletionUsage, ApiChatCompletionToolCallDelta, ApiChatMessageData, ApiChatMessageContentPart, diff --git a/tools/ui/src/lib/components/app/models/ModelId.svelte b/tools/ui/src/lib/components/app/models/ModelId.svelte index 88140cfe6d..b833a591e2 100644 --- a/tools/ui/src/lib/components/app/models/ModelId.svelte +++ b/tools/ui/src/lib/components/app/models/ModelId.svelte @@ -106,6 +106,7 @@ parsed.sidecar || uniqueDraftKinds.length > 0 || uniqueDraftSidecars.length > 0 || + uniqueDraftKinds.length > 0 || (parsed.params && !hideParameters) || (parsed.quantization && !resolvedHideQuantization) || primaryAlias || diff --git a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerModelsTable.svelte b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerModelsTable.svelte index f05ca8a423..7193d1544b 100644 --- a/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerModelsTable.svelte +++ b/tools/ui/src/lib/components/app/models/ModelsManager/ModelsManagerModelsTable.svelte @@ -85,7 +85,7 @@ /** Repos whose quants are folded away; the rest show them. */ const collapsedQuants = new SvelteSet(); /** Sections that list their models straight, without folding them into families. */ - const FLAT_SECTIONS = new Set([ + const FLAT_SECTIONS = new Set([ ModelsTableGroupKind.DOWNLOADING, ModelsTableGroupKind.FAVORITES, ModelsTableGroupKind.LOADED diff --git a/tools/ui/src/lib/components/app/models/ModelsManager/utils.ts b/tools/ui/src/lib/components/app/models/ModelsManager/utils.ts index 1bb9eb696c..7cad8bd52f 100644 --- a/tools/ui/src/lib/components/app/models/ModelsManager/utils.ts +++ b/tools/ui/src/lib/components/app/models/ModelsManager/utils.ts @@ -1,9 +1,63 @@ -import { LOCAL_BACKEND_ID, ModelGroupKind, ModelsTableGroupKind } from '$lib/constants'; +import { + LOCAL_BACKEND_ID, + MODEL_ID, + MODEL_OVERRIDES_LOCALSTORAGE_KEY, + ModelGroupKind, + type ModelSidecar, + ModelsTableGroupKind, + SETTINGS_KEYS, + SPEC_TYPE +} from '$lib/constants'; import { ModelCapability, ServerModelStatus } from '$lib/enums'; import { HuggingFaceService, ModelsService } from '$lib/services'; -import { modelsStore } from '$lib/stores'; -import type { ModelDownloadProgress, ModelOption } from '$lib/types/models'; +import { backendsModelsStore, modelsStore, settingsStore } from '$lib/stores'; +import type { + ModelDownloadProgress, + ModelLoadProgress, + ModelOption, + ModelSidecarFile +} from '$lib/types/models'; import { detectThinkingSupport, detectToolUseSupport } from '$lib/utils'; +import { getBackend } from '$lib/utils/api-base'; +import { getBackendCapabilities } from '$lib/utils/backend'; +import { formatFileSize, formatParameters } from '$lib/utils/formatters'; +import { rawModelId } from '$lib/utils/model-option-id'; +import { SvelteMap } from 'svelte/reactivity'; + +/** Load parameters a model can override before it is loaded. */ +export interface ModelLoadOverride { + batchSize?: number; + contextLength?: number; + cpuThreads?: number; + flashAttention?: boolean; + gpuOffload?: number; + keepInMemory?: boolean; + speculativeDecoding?: string; + ubatchSize?: number; + useMmap?: boolean; +} + +/** Sampling parameters. A null value means the server default stays in charge. */ +export interface ModelSamplingOverride { + minP?: number | null; + repeatPenalty?: number | null; + temperature?: number | null; + topK?: number | null; + topP?: number | null; +} + +export interface ModelOverride { + load?: ModelLoadOverride; + reasoning?: { budget: string; enabled: boolean }; + sampling?: ModelSamplingOverride; + stopStrings?: string[]; + structuredOutput?: { enabled: boolean; schema: string }; + systemPrompt?: string; +} + +export type ModelOverrideMap = Record; + +export type { ModelLoadProgress }; /** One repo of the table, with the rows it ships as. */ export interface ModelQuantGroup { @@ -54,13 +108,207 @@ export function downloadGroups(entries: DownloadEntry[], models: ModelOption[]): /** One collapsible block of the manager's table. */ export interface ModelsTableGroup { + /** Backend the block belongs to, when it is tied to one. */ + backendId?: string | null; /** One entry per repo, its quants hanging off it. */ items: ModelQuantGroup[]; + isLocal?: boolean; key: string; - kind: ModelsTableGroupKind; + /** Manager sections use the kind constants, provider blocks their own kinds. */ + kind: ModelsTableGroupKind | 'compat' | 'provider'; label: string; } +/** Values the load form falls back to when the server reports nothing. */ +export const LOAD_DEFAULTS = { + batchSize: 2048, + contextLength: 8192, + cpuThreads: 13, + gpuOffload: 42, + speculativeDecoding: 'off', + ubatchSize: 512 +}; + +export const SAMPLING_DEFAULTS = { + minP: 0.05, + repeatPenalty: 1.1, + temperature: 1, + topK: 64, + topP: 0.95 +}; + +export const SPECULATIVE_OPTIONS = ['off', 'draft-model']; + +/** True when the user saved anything for this model. */ +export function isCustomized(override?: ModelOverride): boolean { + return override !== undefined && Object.keys(override).length > 0; +} + +export function loadOverrides(): ModelOverrideMap { + try { + const raw = localStorage.getItem(MODEL_OVERRIDES_LOCALSTORAGE_KEY); + + if (!raw) return {}; + + const parsed = JSON.parse(raw) as unknown; + + return parsed && typeof parsed === 'object' ? (parsed as ModelOverrideMap) : {}; + } catch { + return {}; + } +} + +export function saveOverrides(overrides: ModelOverrideMap): void { + try { + localStorage.setItem(MODEL_OVERRIDES_LOCALSTORAGE_KEY, JSON.stringify(overrides)); + } catch { + console.warn('[ModelsManager] Failed to persist model overrides'); + } +} + +/** Backend a model is served by, the local server reads as "This server". */ +/** A draft a model can speculate with, and whether a load would use it. */ +export interface ModelDraft { + /** The draft a load would use, from the server's own arguments or from the settings. */ + active: boolean; + kind: ModelSidecar | null; + /** Repo the draft comes from; null when the file sits in the model's own repo. */ + model: string | null; + params: string | null; + quant: string | null; +} + +/** Repo an id belongs to: the id without its quant tag. */ +function repoOf(modelId: string): string | null { + const raw = rawModelId(modelId).split(MODEL_ID.QUANTIZATION_SEPARATOR)[0] ?? ''; + + return raw || null; +} + +/** Sidecar a `--spec-type` value names, e.g. `draft-mtp` -> mtp. */ +function sidecarFromSpecType(specType: string | null | undefined): ModelSidecar | null { + if (!specType) return null; + + const entry = Object.entries(SPEC_TYPE).find(([, value]) => value === specType); + + return (entry?.[0] as ModelSidecar | undefined) ?? null; +} + +/** + * Repo a draft path names. A Hub cache path carries it (`models--org--name`), a plain + * file next to the model does not, so that falls back to the file name. + */ +function repoFromDraftPath(path: string): string { + const cached = /models--([^/]+)[/]/.exec(path); + + if (cached) { + const [org, ...rest] = cached[1].split('--'); + + if (rest.length > 0) return `${org}/${rest.join('--')}`; + } + + return path.split(/[/]/).pop() ?? path; +} + +/** + * Draft the server's own launch arguments point at. The router reports the arguments a + * model loads with, so this is what a load would really speculate with. + */ +export function draftFromArgs(args: string[] | undefined, option: ModelOption): ModelDraft | null { + const flag = args?.indexOf('--model-draft') ?? -1; + const path = flag === -1 ? null : (args?.[flag + 1] ?? null); + + if (!path) return null; + + const parsed = ModelsService.parseModelId(path.split(/[/\\]/).pop() ?? path); + const repo = repoFromDraftPath(path); + + return { + active: true, + kind: sidecarFromSpecType(args?.[(args?.indexOf('--spec-type') ?? -1) + 1]) ?? parsed.sidecar, + model: repo === repoOf(option.model) ? null : repo, + params: parsed.params + ? `${parsed.params}${parsed.activatedParams ? `-${parsed.activatedParams}` : ''}` + : null, + quant: parsed.quantization + }; +} + +/** Draft the load settings name, resolved against the model's own repo. */ +export function draftFromSetting(option: ModelOption, value?: string | null): ModelDraft | null { + const id = value?.trim(); + + if (!id || id === 'off') return null; + + const parsed = ModelsService.parseModelId(id); + + return { + active: true, + kind: parsed.sidecar, + model: repoOf(id) === repoOf(option.model) ? null : id, + params: parsed.params + ? `${parsed.params}${parsed.activatedParams ? `-${parsed.activatedParams}` : ''}` + : null, + quant: parsed.quantization + }; +} + +/** + * Drafts of a model, in the order they matter: what the server loads with, else what the + * settings name, then any other sidecar the model's own repo ships. + */ +export function modelDraftsFor(option: ModelOption, settingValue?: string | null): ModelDraft[] { + // speculative decoding is a llama.cpp feature + if (!getBackendCapabilities(getBackend(option.backendId)).loadUnload) return []; + + const args = modelsStore.routerModels.find((model) => model.id === option.model)?.status?.args; + const configured = draftFromArgs(args, option) ?? draftFromSetting(option, settingValue); + + return modelDrafts(option, sidecarFilesFor(option), configured); +} + +/** Draft sidecars a listing reported for the model's repo. */ +export function sidecarFilesFor(option: ModelOption): ModelSidecarFile[] { + const repo = option.model.split(':')[0] ?? ''; + const state = backendsModelsStore.get(option.backendId ?? LOCAL_BACKEND_ID); + + return state.drafts?.[repo] ?? []; +} + +/** + * Drafts to show for a model: the one the load settings name, plus any draft sidecar + * the model's own repo ships. The configured one is the active draft; a sidecar that + * is merely on disk stays visible but idle. + */ +export function modelDrafts( + option: ModelOption, + available: ModelSidecarFile[] = [], + configured?: ModelDraft | null +): ModelDraft[] { + const drafts: ModelDraft[] = []; + + if (configured) drafts.push(configured); + + for (const file of available) { + // a sidecar a load already points at is the active draft, not a second entry + if (drafts.some((draft) => draft.kind === file.kind)) continue; + + drafts.push({ + active: false, + kind: file.kind, + model: null, + params: file.params, + quant: file.quant + }); + } + + return drafts; +} + +/** + * Whether a model has a capability. A listing that declares nothing is not a listing + * that lacks it: the chat template the Hub carries says whether tools or reasoning work. + */ /** True when the model is loaded (or sleeping) and not mid-operation. */ export function isModelRunning(option: ModelOption): boolean { const status = modelsStore.getModelStatus(option.model); @@ -86,6 +334,14 @@ export function modelSupports(option: ModelOption, capability: ModelCapability): : detectThinkingSupport(template); } +export function servedByLabel(option: ModelOption): string { + const backend = getBackend(option.backendId); + + if (!backend || backend.id === LOCAL_BACKEND_ID) return 'This server'; + + return backend.name; +} + /** Context a model reports: the provider listing first, then the cached Hub record. */ export function modelContextLength(option: ModelOption): number { return ( @@ -95,6 +351,62 @@ export function modelContextLength(option: ModelOption): number { ); } +export function isLocalOption(option: ModelOption): boolean { + return (option.backendId ?? LOCAL_BACKEND_ID) === LOCAL_BACKEND_ID; +} + +/** File size of a local GGUF, when the router reported one. */ +export function modelSizeLabel(option: ModelOption): string | null { + const bytes = option.meta?.size; + + if (typeof bytes === 'number' && bytes > 0) return formatFileSize(bytes); + + return null; +} + +export function modelParamsLabel(option: ModelOption): string | null { + if (option.parsedId?.params) return option.parsedId.params; + + const params = option.meta?.n_params; + + return typeof params === 'number' ? formatParameters(params) : null; +} + +export function modelQuantLabel(option: ModelOption): string | null { + return option.parsedId?.quantization ?? null; +} + +/** + * File size of the model's own quant. The router reports one for some backends; + * otherwise a local GGUF reads it from its repo tree, the same source the + * discovery details use. Returns null when neither knows. + */ +export async function resolveModelSize(option: ModelOption): Promise { + const reported = modelSizeLabel(option); + + if (reported) return reported; + + // the repo tree lookup only happens for installs that opted into the Hub + if (!isLocalOption(option) || !settingsStore.config[SETTINGS_KEYS.ENABLE_DISCOVER_MODELS]) { + return null; + } + + const [repo, quant] = option.model.split(':'); + + if (!repo || !quant) return null; + + const tree = await HuggingFaceService.getTree(repo); + const file = HuggingFaceService.collapseGgufShards( + HuggingFaceService.filterByExtension(tree, '.gguf') + ).find((entry) => { + const meta = HuggingFaceService.extractQuantMeta(entry.path); + + return meta?.quant === quant && !meta.sidecar; + }); + + return file?.size ? formatFileSize(file.size) : null; +} + /** Repo a `repo:quant` id belongs to, the id itself when it carries no quant. */ export function modelRepoKey(model: string): string { const quant = ModelsService.parseModelId(model).quantization; @@ -103,11 +415,15 @@ export function modelRepoKey(model: string): string { } /** Fold the quants of one repo into a single entry, so the table shows one row per model. */ -export function groupModelQuants(models: ModelOption[]): ModelQuantGroup[] { - const groups = new Map(); +export function groupModelQuants(models: ModelOption[], mergeProviders = false): ModelQuantGroup[] { + const groups = new SvelteMap(); for (const option of models) { - const key = modelRepoKey(option.model); + const repo = modelRepoKey(option.model); + // groups stay within one backend, so the same repo served by two providers + // is not read as two quants of one model. The OAI-compat block asks for the + // opposite: one repo, one row per provider that serves it. + const key = mergeProviders ? repo : `${option.backendId ?? ''}::${repo}`; const group = groups.get(key); if (group) { @@ -123,8 +439,13 @@ export function groupModelQuants(models: ModelOption[]): ModelQuantGroup[] { const kind = groupKind(group.quants); const modelIds = group.quants.map((option) => option.model); - // the very same id twice is not a quant set; keep those rows apart - if (modelIds.length > 1 && modelIds.every((model) => model === modelIds[0])) { + // the very same id twice is not a quant set; keep those rows apart, unless + // the group exists to list the providers that serve it + if ( + kind !== 'providers' && + modelIds.length > 1 && + modelIds.every((model) => model === modelIds[0]) + ) { return group.quants.map((option) => ({ ...group, base: option, @@ -138,11 +459,59 @@ export function groupModelQuants(models: ModelOption[]): ModelQuantGroup[] { }); } -/** What a group folds: quants or variants of one repo. */ +/** What a group folds: providers, quants, or variants of one repo. */ function groupKind(quants: ModelOption[]): ModelGroupKind { + const backends = new Set(quants.map((option) => option.backendId ?? '')); + + if (backends.size > 1) return 'providers'; + const isQuant = quants.every( (option) => (option.parsedId ?? ModelsService.parseModelId(option.model)).quantization ); return isQuant ? ModelGroupKind.QUANTS : ModelGroupKind.VARIANTS; } + +/** Compact "last used" label: minutes, hours, then days. */ +export function formatLastUsed(timestamp?: number): string { + if (!timestamp) return '—'; + + const minutes = Math.floor((Date.now() - timestamp) / 60_000); + + if (minutes < 1) return 'just now'; + + if (minutes < 60) return `${minutes}m`; + + const hours = Math.floor(minutes / 60); + + if (hours < 24) return `${hours}h`; + + return `${Math.floor(hours / 24)}d`; +} + +/** Extra args the router applies when this model is loaded. */ +export function loadExtraArgs(override?: ModelOverride): string[] { + const load = override?.load; + + if (!load) return []; + + const args: string[] = []; + + if (load.contextLength) args.push('--ctx-size', String(load.contextLength)); + + if (load.gpuOffload !== undefined) args.push('--n-gpu-layers', String(load.gpuOffload)); + + if (load.cpuThreads) args.push('--threads', String(load.cpuThreads)); + + if (load.batchSize) args.push('--batch-size', String(load.batchSize)); + + if (load.ubatchSize) args.push('--ubatch-size', String(load.ubatchSize)); + + if (load.flashAttention) args.push('--flash-attn', 'on'); + + if (load.useMmap === false) args.push('--no-mmap'); + + if (load.keepInMemory === false) args.push('--no-kv-offload'); + + return args; +} diff --git a/tools/ui/src/lib/constants/backend.constants.ts b/tools/ui/src/lib/constants/backend.constants.ts index 57f5c42132..1f3ab2e394 100644 --- a/tools/ui/src/lib/constants/backend.constants.ts +++ b/tools/ui/src/lib/constants/backend.constants.ts @@ -1,2 +1,66 @@ -/** Marks the built-in local llama.cpp backend. */ +import type { BackendCapabilities, BackendCompat, BackendProtocol } from '$lib/types'; + +/** Prefix for generated ids of user-added backends. */ +export const BACKEND_ID_PREFIX = 'backend'; + +/** Protocols a configured backend can speak, in display order. */ +export const BACKEND_PROTOCOLS: readonly BackendProtocol[] = ['llama.cpp', 'openai']; + +/** Chat completions path used when a backend does not override it. */ +export const DEFAULT_BACKEND_CHAT_PATH = '/v1/chat/completions'; + +/** Models listing path used when a backend does not override it. */ +export const DEFAULT_BACKEND_MODELS_PATH = '/v1/models'; + +/** Id of the built-in backend that points at the server serving this UI. */ export const LOCAL_BACKEND_ID = 'local'; + +/** Capabilities of a full llama.cpp server. */ +const LLAMA_CPP_CAPABILITIES: BackendCapabilities = { + corsProxy: true, + loadUnload: true, + props: true, + resumableStreams: true, + router: true, + slots: true, + statusFeed: true, + tools: true +}; +/** Capabilities of a plain OpenAI-compatible endpoint. */ +const COMPATIBLE_CAPABILITIES: BackendCapabilities = { + corsProxy: false, + loadUnload: false, + props: false, + resumableStreams: false, + router: false, + slots: false, + statusFeed: false, + tools: false +}; + +/** Capabilities per backend protocol. */ +export const BACKEND_CAPABILITIES: Record = { + 'llama.cpp': LLAMA_CPP_CAPABILITIES, + openai: COMPATIBLE_CAPABILITIES +}; + +/** Default wire quirks per protocol. */ +export const BACKEND_COMPAT: Record = { + // llama-server reports its own timings, so it needs no usage chunk + 'llama.cpp': { maxTokensField: 'max_tokens', supportsUsageInStreaming: false }, + openai: { maxTokensField: 'max_tokens', supportsUsageInStreaming: true } +}; + +/** + * Fields that may carry a model's context size in an OpenAI-compatible model + * listing. Providers pick their own name, and most report nothing at all. + */ +export const MODEL_CONTEXT_LENGTH_FIELDS = [ + 'context_length', + 'context_window', + 'max_context_length', + 'max_position_embeddings' +] as const; + +/** Favicon extract keyed by domain, for backends with no bundled mark. */ +export const FAVICON_SERVICE_URL = 'https://www.google.com/s2/favicons?domain='; diff --git a/tools/ui/src/lib/constants/settings-keys.constants.ts b/tools/ui/src/lib/constants/settings-keys.constants.ts index d4ae8318c5..b0a8995393 100644 --- a/tools/ui/src/lib/constants/settings-keys.constants.ts +++ b/tools/ui/src/lib/constants/settings-keys.constants.ts @@ -11,6 +11,7 @@ export const SETTINGS_KEYS = { API_KEY: 'apiKey', AUTO_MIC_ON_EMPTY: 'autoMicOnEmpty', BACKEND_SAMPLING: 'backend_sampling', + BACKENDS: 'backends', CONVERSATION_TABS: 'conversationTabs', COPY_TEXT_ATTACHMENTS_AS_PLAIN_TEXT: 'copyTextAttachmentsAsPlainText', CUSTOM_CSS: 'customCss', @@ -32,6 +33,7 @@ export const SETTINGS_KEYS = { FULL_HEIGHT_CODE_BLOCKS: 'fullHeightCodeBlocks', GROUP_MODELS_BY_FAMILY: 'groupModelsByFamily', JS_SANDBOX_ENABLED: 'jsSandboxEnabled', + LOCAL_BACKEND_ENABLED: 'localBackendEnabled', MAX_IMAGE_RESOLUTION: 'maxImageMPixels', MAX_TOKENS: 'max_tokens', MCP_REQUEST_TIMEOUT_SECONDS: 'mcpRequestTimeoutSeconds', diff --git a/tools/ui/src/lib/constants/storage.constants.ts b/tools/ui/src/lib/constants/storage.constants.ts index e8612c5cc9..f66a0a0112 100644 --- a/tools/ui/src/lib/constants/storage.constants.ts +++ b/tools/ui/src/lib/constants/storage.constants.ts @@ -16,6 +16,9 @@ export const DB_APP_NAME_DEPRECATED = 'LlamacppWebui'; export const ALWAYS_ALLOWED_TOOLS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.alwaysAllowedTools`; +/** Id of the backend the selector and new requests target, restored on page load. */ +export const ACTIVE_BACKEND_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.activeBackend`; + /** Paused model download ids (`:`), restored on the next page load. */ export const PAUSED_MODEL_DOWNLOADS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.pausedModelDownloads`; export const CONFIG_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.config`; @@ -31,6 +34,12 @@ export const FAVORITE_MODELS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.favoriteMod /** Open state the user set for a model list section or one of its families, by id. */ export const MODEL_GROUP_OPEN_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.modelGroupOpen`; +/** Per-model load and inference overrides, keyed by backend-qualified model id. */ +export const MODEL_OVERRIDES_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.modelOverrides`; + +/** Model the user picked last, kept across reloads. Stores `{ id, model }`. */ +export const SELECTED_MODEL_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.selectedModel`; + /** Recently used model ids, most recent first, backend-qualified. */ export const RECENT_MODELS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.recentModels`; diff --git a/tools/ui/src/lib/constants/stream.constants.ts b/tools/ui/src/lib/constants/stream.constants.ts index 64f67243c2..d68ffbe4bf 100644 --- a/tools/ui/src/lib/constants/stream.constants.ts +++ b/tools/ui/src/lib/constants/stream.constants.ts @@ -2,6 +2,10 @@ // while the tab was hidden. covers brief background pauses without thrashing live streams export const STREAM_VISIBILITY_KICK_MS = 3000; +// minimum gap between synthesized live timing updates for backends that do not +// stream their own, keeps the per-chunk state updates cheap +export const STREAM_LIVE_TIMINGS_INTERVAL_MS = 500; + // separator joining a conversation id and its per-model stream identity // suffix (conv::model) used by the server side replay buffer export const CONVERSATION_ID_SEPARATOR = '::'; diff --git a/tools/ui/src/lib/services/backends.service.ts b/tools/ui/src/lib/services/backends.service.ts new file mode 100644 index 0000000000..5d660f600f --- /dev/null +++ b/tools/ui/src/lib/services/backends.service.ts @@ -0,0 +1,188 @@ +/** + * BackendsService - Stateless backend connectivity checks and model listing + * + * Probes a backend's models endpoint to validate its URL and credentials, and + * normalizes the response into the UI model shape. No reactive state; + * consumed by the backends settings UI and the per-backend model cache. + */ + +import { API_MODELS, LOCAL_BACKEND_ID } from '$lib/constants'; +import { ModelsService } from '$lib/services/models.service'; +import type { ApiModelsListResponse, Backend, BackendProtocol, ModelOption } from '$lib/types'; +import { isAbortError } from '$lib/utils/abort'; +import { apiUrl } from '$lib/utils/api-base'; +import { getAuthHeadersForBackend } from '$lib/utils/api-headers'; +import { backendModelsUrl, readModelContextLength } from '$lib/utils/backend'; + +/** Models returned by a backend, plus the failure detail when the call fails. */ +export interface BackendModelsResult { + error?: string; + models: ModelOption[]; + ok: boolean; + status: number | null; + /** Untouched list payload of the local backend, kept so the router rows and their load statuses can be rebuilt without asking again. */ + raw?: ApiModelsListResponse; +} + +/** What probing a backend's endpoint said about it. */ +export interface BackendProbe { + /** The endpoint refused the request for want of a key, so it wants one. */ + authRequired: boolean; + protocol: BackendProtocol; +} + +/** Outcome of a backend connectivity check. */ +export interface BackendTestResult { + error?: string; + modelCount?: number; + ok: boolean; + status: number | null; +} + +export class BackendsService { + /** + * List the models a backend exposes on its models endpoint. + * + * @param backend - Backend to query. Does not need to be registered yet. + * @param signal - Optional abort signal for a cancelled request. + */ + static async detectProtocol(backend: Backend): Promise { + const base = backend.baseUrl.trim().replace(/\/+$/, ''); + + if (!base) return { authRequired: false, protocol: 'openai' }; + + try { + // llama-server answers /props with its build and generation defaults; a + // plain OpenAI-compatible endpoint answers 404 there, or not at all + const response = await fetch(`${base}/props`, { + headers: getAuthHeadersForBackend(backend), + signal: AbortSignal.timeout(5000) + }); + + // a llama-server behind a key refuses before it says anything else, while + // an OpenAI-compatible endpoint has no /props to guard in the first place + if (response.status === 401) { + return { authRequired: true, protocol: 'llama.cpp' }; + } + + if (!response.ok) return { authRequired: false, protocol: 'openai' }; + + const body = (await response.json()) as Record; + const isLlamaCpp = + 'default_generation_settings' in body || 'build_info' in body || body.role === 'router'; + + return { + authRequired: false, + protocol: isLlamaCpp ? 'llama.cpp' : 'openai' + }; + } catch { + return { authRequired: false, protocol: 'openai' }; + } + } + + static async listModels(backend: Backend, signal?: AbortSignal): Promise { + // the local backend has no base URL; its models endpoint is base relative + const url = backend.baseUrl.trim() + ? backendModelsUrl(backend) + : apiUrl(API_MODELS.LIST, LOCAL_BACKEND_ID); + + if (!backend.baseUrl.trim() && backend.id !== LOCAL_BACKEND_ID) { + return { error: 'Backend URL is required', models: [], ok: false, status: null }; + } + + try { + const response = await fetch(url, { + headers: getAuthHeadersForBackend(backend), + signal + }); + + if (!response.ok) { + return { + error: await describeFailure(response), + models: [], + ok: false, + status: response.status + }; + } + + const body = (await response.json()) as { data?: unknown }; + const entries = Array.isArray(body?.data) ? body.data : []; + const models = entries.flatMap((entry) => normalizeBackendModel(entry)); + // the local rows carry load status, external ones carry the context size + const raw = body as ApiModelsListResponse; + + return { models, ok: true, raw, status: response.status }; + } catch (error) { + if (isAbortError(error)) { + return { models: [], ok: false, status: null }; + } + + return { + error: error instanceof Error ? error.message : String(error), + models: [], + ok: false, + status: null + }; + } + } + + /** + * Check that a backend answers on its models endpoint. + * + * @param backend - Backend to probe. Does not need to be registered yet. + * @param signal - Optional abort signal for a cancelled test. + */ + static async test(backend: Backend, signal?: AbortSignal): Promise { + const result = await BackendsService.listModels(backend, signal); + + return { + error: result.error, + modelCount: result.models.length, + ok: result.ok, + status: result.status + }; + } +} + +/** Build a human-readable message from a non-OK response. */ +async function describeFailure(response: Response): Promise { + const status = `${response.status} ${response.statusText}`.trim(); + + try { + const body = (await response.json()) as { error?: { message?: string }; message?: string }; + const message = body?.error?.message ?? body?.message; + + if (message) return `${status}: ${message}`; + } catch { + // non-JSON error body, fall back to the status line + } + + return status; +} + +/** + * Normalize one entry of an OpenAI-compatible `/v1/models` response. External + * backends only guarantee an id, so that doubles as the display name. + */ +function normalizeBackendModel(entry: unknown): ModelOption[] { + if (!entry || typeof entry !== 'object') return []; + + const raw = entry as Record; + const id = typeof raw.id === 'string' ? raw.id.trim() : ''; + + if (!id) return []; + + // a llama-compat server lists its projector and draft sidecars as models too + if (ModelsService.isSidecarEntry(id)) return []; + + return [ + { + capabilities: [], + contextLength: readModelContextLength(raw), + id, + model: id, + name: id, + status: raw.status as ApiModelDataEntry['status'] + } + ]; +} diff --git a/tools/ui/src/lib/services/index.ts b/tools/ui/src/lib/services/index.ts index a8072e55bc..6a7009ebaf 100644 --- a/tools/ui/src/lib/services/index.ts +++ b/tools/ui/src/lib/services/index.ts @@ -19,6 +19,14 @@ * */ +/** + * **BackendsService** - Backend connectivity checks + * + * Probes an external backend's models endpoint to validate its URL and + * credentials before it is saved. Stateless. + */ +export { BackendsService } from './backends.service'; + /** * **ChatService** - Chat Completions API communication layer * diff --git a/tools/ui/src/lib/stores/backends.svelte.ts b/tools/ui/src/lib/stores/backends.svelte.ts new file mode 100644 index 0000000000..7b9d9b0962 --- /dev/null +++ b/tools/ui/src/lib/stores/backends.svelte.ts @@ -0,0 +1,113 @@ +/** + * backendsStore - API endpoints the UI can talk to. + * + * The built-in local backend is the llama-server serving this UI. External + * backends are user-configured endpoints persisted in settings. The store + * registers the resolved list with the api-base registry, which services use + * to build request URLs. + */ + +import { browser } from '$app/environment'; +import { ACTIVE_BACKEND_LOCALSTORAGE_KEY, LOCAL_BACKEND_ID, SETTINGS_KEYS } from '$lib/constants'; +import { serverStore } from '$lib/stores/server.svelte'; +import { settingsStore } from '$lib/stores/settings/index.svelte'; +import type { Backend } from '$lib/types'; +import { setBackendsResolver } from '$lib/utils/api-base'; +import { createLocalBackend, parseBackendsSettings } from '$lib/utils/backend'; + +function loadActiveBackendId(): string { + if (!browser) return LOCAL_BACKEND_ID; + + try { + return localStorage.getItem(ACTIVE_BACKEND_LOCALSTORAGE_KEY) ?? LOCAL_BACKEND_ID; + } catch { + return LOCAL_BACKEND_ID; + } +} + +function persistActiveBackendId(backendId: string): void { + if (!browser) return; + + try { + localStorage.setItem(ACTIVE_BACKEND_LOCALSTORAGE_KEY, backendId); + } catch { + /* ignore */ + } +} + +class BackendsStore { + activeId = $state(loadActiveBackendId()); + + get active(): Backend { + const active = this.enabled.find((backend) => backend.id === this.activeId); + + return active ?? this.enabled[0] ?? this.local; + } + + get enabled(): Backend[] { + return this.list.filter((backend) => backend.enabled && !this.isMissingLocal(backend)); + } + + get external(): Backend[] { + return parseBackendsSettings(settingsStore.config[SETTINGS_KEYS.BACKENDS]); + } + + get list(): Backend[] { + return [this.local, ...this.external]; + } + + get local(): Backend { + return createLocalBackend( + settingsStore.config.apiKey?.toString().trim() || undefined, + settingsStore.config[SETTINGS_KEYS.LOCAL_BACKEND_ENABLED] !== false + ); + } + + addBackend(backend: Backend): void { + this.saveExternal([...this.external, backend]); + } + + initialize(): void { + if (!browser) return; + + setBackendsResolver(() => ({ activeId: this.active.id, backends: this.list })); + } + + removeBackend(backendId: string): void { + this.saveExternal(this.external.filter((backend) => backend.id !== backendId)); + + if (this.activeId === backendId) { + this.setActive(LOCAL_BACKEND_ID); + } + } + + setActive(backendId: string): void { + const id = this.list.some((backend) => backend.id === backendId) ? backendId : LOCAL_BACKEND_ID; + + this.activeId = id; + persistActiveBackendId(id); + } + + setLocalEnabled(enabled: boolean): void { + settingsStore.updateConfig(SETTINGS_KEYS.LOCAL_BACKEND_ENABLED, enabled); + } + + updateBackend(backendId: string, updates: Partial): void { + this.saveExternal( + this.external.map((backend) => + backend.id === backendId ? { ...backend, ...updates } : backend + ) + ); + } + + /** The built-in backend counts only when a local server answered this session. */ + private isMissingLocal(backend: Backend): boolean { + return backend.id === LOCAL_BACKEND_ID && serverStore.localServerMissing; + } + + private saveExternal(backends: Backend[]): void { + settingsStore.updateConfig(SETTINGS_KEYS.BACKENDS, JSON.stringify(backends)); + } +} + +export const backendsStore = new BackendsStore(); diff --git a/tools/ui/src/lib/stores/backendsModels.svelte.ts b/tools/ui/src/lib/stores/backendsModels.svelte.ts new file mode 100644 index 0000000000..3b4ef86d01 --- /dev/null +++ b/tools/ui/src/lib/stores/backendsModels.svelte.ts @@ -0,0 +1,103 @@ +/** + * backendsModelsStore - Per-backend model catalog cache. + * + * Prefetches every enabled backend's model list, so switching backends is + * instant and the switcher can show load state. The active backend's list + * still lives in modelsStore, which owns selection and chat wiring; this + * cache is the prefetch layer the switches start from. + */ + +import { BackendsService } from '$lib/services/backends.service'; +import { ModelsService } from '$lib/services/models.service'; +import { backendsStore } from '$lib/stores/backends.svelte'; +import type { Backend } from '$lib/types'; +import type { ApiModelsListResponse } from '$lib/types'; +import type { ModelSidecarFile } from '$lib/types/models'; +import type { ModelOption } from '$lib/types/models'; + +export interface BackendModelsState { + /** Draft sidecars the listing carries, keyed by the repo they belong to. */ + drafts?: Record; + error: string | null; + loaded: boolean; + loading: boolean; + models: ModelOption[]; + /** Untouched list payload, kept for the local backend so its router rows survive a tab switch. */ + raw?: ApiModelsListResponse; +} + +const EMPTY_STATE: BackendModelsState = { + error: null, + loaded: false, + loading: false, + models: [] +}; + +class BackendsModelsStore { + private states = $state>({}); + + clear(backendId: string): void { + delete this.states[backendId]; + } + + /** + * Load a backend's models once. + */ + async ensureLoaded(backendId: string): Promise { + const backend = backendsStore.enabled.find((candidate) => candidate.id === backendId); + + if (!backend) return; + + const state = this.states[backendId]; + + if (state?.loaded || state?.loading) return; + + this.states[backendId] = { error: null, loaded: false, loading: true, models: [] }; + await this.fetch(backend); + } + + get(backendId: string): BackendModelsState { + return this.states[backendId] ?? EMPTY_STATE; + } + + /** Prefetch every enabled backend's model list. */ + async loadAll(): Promise { + const enabled = backendsStore.enabled; + const ids = new Set(enabled.map((backend) => backend.id)); + + for (const id of Object.keys(this.states)) { + if (!ids.has(id)) { + delete this.states[id]; + } + } + + await Promise.all(enabled.map((backend) => this.ensureLoaded(backend.id))); + } + + /** + * Ask a backend for its list again, keeping what is already known. Used while + * a remote load settles, since its status never reaches the local feed. + */ + async refresh(backendId: string): Promise { + const backend = backendsStore.enabled.find((candidate) => candidate.id === backendId); + + if (!backend) return; + + await this.fetch(backend); + } + + private async fetch(backend: Backend): Promise { + const result = await BackendsService.listModels(backend); + + this.states[backend.id] = { + drafts: result.raw ? ModelsService.draftSidecarsByRepo(result.raw) : undefined, + error: result.error ?? null, + loaded: result.ok, + loading: false, + models: result.models, + raw: result.raw + }; + } +} + +export const backendsModelsStore = new BackendsModelsStore(); diff --git a/tools/ui/src/lib/types/api.d.ts b/tools/ui/src/lib/types/api.d.ts index b52485ade8..ca724ee389 100644 --- a/tools/ui/src/lib/types/api.d.ts +++ b/tools/ui/src/lib/types/api.d.ts @@ -358,7 +358,9 @@ export interface ApiChatCompletionStreamChunk { metadata?: { model?: string }; delta: { content?: string; + reasoning?: string; reasoning_content?: string; + reasoning_text?: string; model?: string; tool_calls?: ApiChatCompletionToolCallDelta[]; }; @@ -372,6 +374,17 @@ export interface ApiChatCompletionStreamChunk { cache_n?: number; }; prompt_progress?: ChatMessagePromptProgress; + /** Token counts, sent by OpenAI-compatible servers on the final chunk. */ + usage?: ApiChatCompletionUsage; +} + +export interface ApiChatCompletionUsage { + cached_tokens?: number; + completion_tokens?: number; + prompt_cache_hit_tokens?: number; + prompt_tokens?: number; + prompt_tokens_details?: { cached_tokens?: number; cache_write_tokens?: number }; + total_tokens?: number; } export interface ApiChatCompletionResponse { diff --git a/tools/ui/src/lib/types/backend.d.ts b/tools/ui/src/lib/types/backend.d.ts new file mode 100644 index 0000000000..019fce3610 --- /dev/null +++ b/tools/ui/src/lib/types/backend.d.ts @@ -0,0 +1,98 @@ +/** + * Backend types. + * + * A backend is one API endpoint the UI can talk to. The built-in `local` + * backend is the llama-server serving the UI. External backends are + * user-configured endpoints that speak an OpenAI-compatible + * protocol. + */ + +/** Request/response shape a backend speaks. */ +export type BackendProtocol = 'llama.cpp' | 'openai'; + +/** + * Wire-level quirks of a backend's protocol. Capabilities gate llama.cpp + * features; compat describes how the request and stream payloads differ. + */ +export interface BackendCompat { + /** Field carrying the output token cap. */ + maxTokensField: 'max_completion_tokens' | 'max_tokens'; + /** Whether the endpoint accepts stream_options.include_usage. */ + supportsUsageInStreaming: boolean; +} + +/** + * Features a backend supports. A llama.cpp server exposes extra endpoints on + * top of the OpenAI-compatible API; plain OpenAI-compatible + * endpoints only provide chat and model listing. + */ +export interface BackendCapabilities { + /** llama-server's /cors-proxy endpoint for cross-origin MCP requests. */ + corsProxy: boolean; + /** Router-mode model load/unload. */ + loadUnload: boolean; + /** The /props endpoint with server role and generation defaults. */ + props: boolean; + /** Resumable stream sessions (/v1/stream, /v1/streams/lookup). */ + resumableStreams: boolean; + /** Multi-model router mode. */ + router: boolean; + /** The /slots introspection endpoint. */ + slots: boolean; + /** The /models/sse load and download progress feed. */ + statusFeed: boolean; + /** The /tools listing and execution endpoint. */ + tools: boolean; +} + +/** + * One configured API endpoint. + * + * TODO: a backend paired by QR code is reached over WebRTC instead of plain + * HTTP, so it needs a transport discriminator and its peer description here. + */ +export interface Backend { + /** Bearer token / API key used for this backend. */ + apiKey?: string; + /** + * API root the endpoint paths are appended to, e.g. https://api.example.com. + * Empty for the local backend, which resolves against the UI origin instead. + */ + baseUrl: string; + /** Chat completions path override, e.g. /v1/messages. */ + chatPath?: string; + /** Wire quirks overriding the protocol defaults. */ + compat?: Partial; + /** Disabled backends stay configured but are not queried. */ + enabled: boolean; + /** Extra headers merged into every request to this backend. */ + headers?: Record; + /** Stable identity. The local backend id is reserved. */ + id: string; + /** Models listing path override, e.g. /models. */ + modelsPath?: string; + name: string; + protocol: BackendProtocol; +} + +/** A ready-made backend configuration offered when adding a backend. */ +export interface BackendPreset { + /** Optional help text shown under the API key field. */ + apiKeyHelp?: string; + baseUrl: string; + /** One line describing the endpoint, shown on the preset card. */ + description?: string; + chatPath?: string; + /** Brand mark used in both themes, for logos that carry their own background. */ + iconUrl?: string; + /** Brand mark for the dark theme. Preferred over `iconUrl` when paired with `iconUrlLight`. */ + iconUrlDark?: string; + /** Brand mark for the light theme. Preferred over `iconUrl` when paired with `iconUrlDark`. */ + iconUrlLight?: string; + /** Wire quirks this preset needs on top of the protocol defaults. */ + compat?: Partial; + id: string; + modelsPath?: string; + name: string; + protocol: BackendProtocol; +} diff --git a/tools/ui/src/lib/types/chat.d.ts b/tools/ui/src/lib/types/chat.d.ts index 86a868c33a..835f2081d2 100644 --- a/tools/ui/src/lib/types/chat.d.ts +++ b/tools/ui/src/lib/types/chat.d.ts @@ -334,6 +334,6 @@ export interface ChatFormActionsContext { readonly hasVideoModality: boolean; readonly hasVisionModality: boolean; onFileUpload?: () => void; - onSystemPromptClick?: () => void; onMcpSettingsClick?: () => void; + onSystemPromptClick?: () => void; } diff --git a/tools/ui/src/lib/types/index.ts b/tools/ui/src/lib/types/index.ts index 7f35a13633..3723e8d5f9 100644 --- a/tools/ui/src/lib/types/index.ts +++ b/tools/ui/src/lib/types/index.ts @@ -25,6 +25,7 @@ export type { ApiChatCompletionToolCallDelta, ApiChatCompletionToolCall, ApiChatCompletionStreamChunk, + ApiChatCompletionUsage, ApiChatCompletionResponse, ApiSlotData, ApiProcessingState, @@ -35,6 +36,15 @@ export type { ApiStreamSession } from './api'; +// Backend types +export type { + Backend, + BackendCapabilities, + BackendCompat, + BackendPreset, + BackendProtocol +} from './backend'; + // HuggingFace types export type { HfCatalogBuild, diff --git a/tools/ui/src/lib/types/models.d.ts b/tools/ui/src/lib/types/models.d.ts index 9d1f974931..f152fe047a 100644 --- a/tools/ui/src/lib/types/models.d.ts +++ b/tools/ui/src/lib/types/models.d.ts @@ -16,6 +16,8 @@ export interface ModelOption { id: string; name: string; model: string; + /** Backend that serves this model; set on the aggregated option list. */ + backendId?: string; description?: string; capabilities: string[]; /** Context size reported by the provider's model listing, when it reports one. */ @@ -25,6 +27,8 @@ export interface ModelOption { modalities?: ModelModalities; details?: ApiModelDetails['details']; meta?: ApiModelDataEntry['meta']; + /** Load state the provider's own listing reports, when it reports one. */ + status?: ApiModelDataEntry['status']; parsedId?: ParsedModelId; aliases?: string[]; tags?: string[]; @@ -51,6 +55,21 @@ export interface ModelDownloadProgress { files: Record; } +/** + * A draft sidecar file a model listing reports as its own entry. The router lists a + * downloaded sidecar as a model, so this is what pairs it back with its model. + */ +export interface ModelSidecarFile { + id: string; + kind: ModelSidecar; + /** Repo the sidecar belongs to, e.g. `ggml-org/Qwen3.6-35B-A3B-GGUF`. */ + model: string; + /** Parameter count the sidecar reports, e.g. `35B-A3B`. */ + params: string | null; + /** Quantization of the sidecar file, e.g. `Q4_0`. */ + quant: string | null; +} + export interface ParsedModelId { raw: string; orgName: string | null; diff --git a/tools/ui/src/lib/utils/api-base.ts b/tools/ui/src/lib/utils/api-base.ts new file mode 100644 index 0000000000..d88d536531 --- /dev/null +++ b/tools/ui/src/lib/utils/api-base.ts @@ -0,0 +1,99 @@ +/** + * API base resolution for backends. + * + * The UI can talk to more than one backend endpoint. Services build request + * URLs through {@link apiUrl} so a request always targets the right backend. + * The backends store registers a resolver here; this module never imports the + * store, which keeps URL resolution free of store dependencies. + */ + +import { backendChatUrl, backendModelsUrl } from './backend'; +import { base } from '$app/paths'; +import { API_ABSOLUTE_URL_PROTOCOLS, API_CHAT, API_MODELS } from '$lib/constants'; +import type { Backend } from '$lib/types'; + +/** Backend list and active selection as exposed to URL resolution. */ +export interface BackendsSnapshot { + activeId: string; + backends: Backend[]; +} + +type BackendsResolver = () => BackendsSnapshot; + +let resolveBackends: BackendsResolver | null = null; + +/** Registered once by the backends store. */ +export function setBackendsResolver(resolver: BackendsResolver | null): void { + resolveBackends = resolver; +} + +/** + * Look up a backend by id, defaulting to the active one. + */ +export function getBackend(backendId?: string): Backend | undefined { + const snapshot = resolveBackends?.(); + + if (!snapshot) return undefined; + + const id = backendId ?? snapshot.activeId; + + return snapshot.backends.find((backend) => backend.id === id); +} + +/** API root for a backend, or an empty string for the local backend. */ +export function getBackendBaseUrl(backendId?: string): string { + return getBackend(backendId)?.baseUrl.trim() ?? ''; +} + +/** + * Request target for a backend's chat completions endpoint. Returns a relative + * path for the local backend and an absolute URL for external ones, so callers + * can pass the result straight to `fetch` (or `apiFetch`, which resolves + * relative paths against the base). + */ +export function apiChatUrl(backendId?: string): string { + const backend = getBackend(backendId); + + if (backend?.baseUrl.trim()) { + return backendChatUrl(backend); + } + + return apiUrl(API_CHAT.COMPLETIONS, backendId); +} + +/** + * Request target for a backend's models listing. Returns a plain path for the + * local backend (so `apiFetch` applies the base) and an absolute URL otherwise. + */ +export function apiModelsUrl(backendId?: string): string { + const backend = getBackend(backendId); + + if (backend?.baseUrl.trim()) { + return backendModelsUrl(backend); + } + + return API_MODELS.LIST; +} + +/** + * Absolute URL for an API path on a backend. + * + * Absolute URLs pass through untouched. Paths on the local backend keep the + * existing base-path-relative form, so serving under a subpath still works. + * Paths on external backends resolve against the backend's API root. + */ +export function apiUrl(path: string, backendId?: string): string { + if (API_ABSOLUTE_URL_PROTOCOLS.some((protocol) => path.startsWith(protocol))) { + return path; + } + + const baseUrl = getBackendBaseUrl(backendId); + + if (!baseUrl) { + return `${base}${path}`; + } + + const root = baseUrl.endsWith('/') ? baseUrl : `${baseUrl}/`; + + return new URL(path.replace(/^\.?\//, ''), root).toString(); +} diff --git a/tools/ui/src/lib/utils/api-fetch.ts b/tools/ui/src/lib/utils/api-fetch.ts index fefea2fe1f..d5fce0fd49 100644 --- a/tools/ui/src/lib/utils/api-fetch.ts +++ b/tools/ui/src/lib/utils/api-fetch.ts @@ -1,6 +1,6 @@ +import { apiUrl } from './api-base'; import { getAuthHeaders, getJsonHeaders } from './api-headers'; -import { base } from '$app/paths'; -import { API_ABSOLUTE_URL_PROTOCOLS, ERROR_MESSAGES, HTTP_CODE_TO_STRING } from '$lib/constants'; +import { ERROR_MESSAGES, HTTP_CODE_TO_STRING } from '$lib/constants'; /** * API Fetch Utilities @@ -32,6 +32,8 @@ export interface ApiFetchOptions extends Omit { * Default: false (uses JSON headers with Content-Type: application/json) */ authOnly?: boolean; + /** Backend to target; defaults to the active one. */ + backendId?: string; /** * Additional headers to merge with default headers. */ @@ -59,11 +61,10 @@ export interface ApiFetchOptions extends Omit { * ``` */ export async function apiFetch(path: string, options: ApiFetchOptions = {}): Promise { - const { authOnly = false, headers: customHeaders, ...fetchOptions } = options; - const baseHeaders = authOnly ? getAuthHeaders() : getJsonHeaders(); + const { authOnly = false, backendId, headers: customHeaders, ...fetchOptions } = options; + const baseHeaders = authOnly ? getAuthHeaders(backendId) : getJsonHeaders(backendId); const headers = { ...baseHeaders, ...customHeaders }; - // absolute URLs with an allowed protocol pass through untouched; relative paths get the base prefix - const url = API_ABSOLUTE_URL_PROTOCOLS.some((p) => path.startsWith(p)) ? path : `${base}${path}`; + const url = apiUrl(path, backendId); let response; @@ -106,7 +107,7 @@ export async function apiFetchWithParams( params: Record, options: ApiFetchOptions = {} ): Promise { - const url = new URL(basePath, window.location.href); + const url = new URL(apiUrl(basePath, options.backendId), window.location.href); for (const [key, value] of Object.entries(params)) { if (value !== undefined && value !== null) { diff --git a/tools/ui/src/lib/utils/api-headers.ts b/tools/ui/src/lib/utils/api-headers.ts index 49d56d0619..4938e01aad 100644 --- a/tools/ui/src/lib/utils/api-headers.ts +++ b/tools/ui/src/lib/utils/api-headers.ts @@ -1,26 +1,42 @@ +import { getBackend } from './api-base'; import { redactValue } from './redact'; import { CORS_PROXY, HEADERS } from '$lib/constants'; import { MimeTypeApplication } from '$lib/enums'; +import { getProtocolAdapter } from '$lib/services/protocols'; import { settingsStore } from '$lib/stores/settings/index.svelte'; +import type { Backend } from '$lib/types'; /** - * Get authorization headers for API requests - * Includes Bearer token if API key is configured + * Get authorization headers for API requests to a backend. */ -export function getAuthHeaders(): Record { - const currentConfig = settingsStore.config; - const apiKey = currentConfig.apiKey?.toString().trim(); +export function getAuthHeaders(backendId?: string): Record { + const backend = getBackend(backendId); + + if (backend) return getAuthHeadersForBackend(backend); + + // no backends resolver yet (early startup, or a non-browser call): keep the + // pre-backends behaviour and authenticate against the serving origin + const apiKey = settingsStore.config.apiKey?.toString().trim(); return apiKey ? { [HEADERS.AUTHORIZATION]: `${HEADERS.BEARER}${apiKey}` } : {}; } +/** + * Get authorization headers for a backend object, including one that is not + * registered yet (used by the connection test on the add-backend form). + * The protocol adapter owns the credential scheme and any required headers. + */ +export function getAuthHeadersForBackend(backend: Backend): Record { + return getProtocolAdapter(backend).authHeaders(backend); +} + /** * Get standard JSON headers with optional authorization */ -export function getJsonHeaders(): Record { +export function getJsonHeaders(backendId?: string): Record { return { [HEADERS.CONTENT_TYPE]: MimeTypeApplication.JSON, - ...getAuthHeaders() + ...getAuthHeaders(backendId) }; } diff --git a/tools/ui/src/lib/utils/api-key-validation.ts b/tools/ui/src/lib/utils/api-key-validation.ts index 187199afc2..22f84c604c 100644 --- a/tools/ui/src/lib/utils/api-key-validation.ts +++ b/tools/ui/src/lib/utils/api-key-validation.ts @@ -1,9 +1,10 @@ import { error } from '@sveltejs/kit'; import { browser } from '$app/environment'; -import { base } from '$app/paths'; import { HEADERS } from '$lib/constants'; import { MimeTypeApplication } from '$lib/enums'; import { settingsStore } from '$lib/stores/settings/index.svelte'; +import { apiUrl, getBackend } from '$lib/utils/api-base'; +import { getBackendCapabilities } from '$lib/utils/backend'; /** * Validates API key by making a request to the server props endpoint @@ -14,6 +15,13 @@ export async function validateApiKey(fetch: typeof globalThis.fetch): Promise `z.ai`) and the icon is requested through a favicon service. + * Returns null when the URL carries no usable host. + */ +export function backendFaviconUrl(baseUrl: string): string | null { + try { + const labels = new URL(baseUrl).hostname.split('.'); + const root = labels.length > 2 ? labels.slice(-2).join('.') : labels.join('.'); + + return root ? `${FAVICON_SERVICE_URL}${root}&sz=${FAVICON_SIZE}` : null; + } catch { + return null; + } +} + +/** + * Features a backend supports, derived from its protocol. A missing backend + * (unknown model, early startup) gets the plain compatible defaults. + */ +export function getBackendCapabilities(backend?: Backend): BackendCapabilities { + return BACKEND_CAPABILITIES[backend?.protocol ?? 'openai'] ?? BACKEND_CAPABILITIES.openai; +} + +/** Wire quirks for a backend: protocol defaults overridden by the backend. */ +export function getBackendCompat(backend: Backend): BackendCompat { + return { ...(BACKEND_COMPAT[backend.protocol] ?? BACKEND_COMPAT.openai), ...backend.compat }; +} + +/** The built-in backend pointing at the server that serves this UI. */ +export function createLocalBackend(apiKey?: string, enabled = true): Backend { + return { + apiKey, + baseUrl: '', + enabled, + id: LOCAL_BACKEND_ID, + name: 'Local', + protocol: 'llama.cpp' + }; +} + +/** + * Parse the persisted backends JSON into backend entries. + */ +export function parseBackendsSettings(rawBackends: unknown): Backend[] { + if (!rawBackends) return []; + + let parsed: unknown; + + if (typeof rawBackends === 'string') { + const trimmed = rawBackends.trim(); + + if (!trimmed) return []; + + try { + parsed = JSON.parse(trimmed); + } catch (error) { + console.warn('[backends] Failed to parse backends JSON, ignoring value:', error); + + return []; + } + } else { + parsed = rawBackends; + } + + if (!Array.isArray(parsed)) return []; + + return parsed.flatMap((entry, index) => { + const backend = parseBackendEntry(entry, index); + + return backend ? [backend] : []; + }); +} + +function joinBackendUrl(baseUrl: string, path: string): string { + const base = baseUrl.replace(/\/+$/, ''); + const suffix = path.startsWith('/') ? path : `/${path}`; + + return `${base}${suffix}`; +} + +function parseBackendEntry(entry: unknown, index: number): Backend | null { + if (!entry || typeof entry !== 'object') return null; + + const raw = entry as Record; + const baseUrl = typeof raw.baseUrl === 'string' ? raw.baseUrl.trim() : ''; + + // the local backend is built in and never persisted + if (!baseUrl || raw.id === LOCAL_BACKEND_ID) return null; + + const protocol = BACKEND_PROTOCOLS.includes(raw.protocol as BackendProtocol) + ? (raw.protocol as BackendProtocol) + : 'openai'; + const id = + typeof raw.id === 'string' && raw.id.trim() + ? raw.id.trim() + : `${BACKEND_ID_PREFIX}-${index + 1}`; + const name = typeof raw.name === 'string' && raw.name.trim() ? raw.name.trim() : baseUrl; + const apiKey = + typeof raw.apiKey === 'string' && raw.apiKey.trim() ? raw.apiKey.trim() : undefined; + + return { + apiKey, + baseUrl, + chatPath: parseOptionalPath(raw.chatPath), + compat: parseBackendCompat(raw.compat, protocol), + enabled: raw.enabled !== false, + headers: parseBackendHeaders(raw.headers), + id, + modelsPath: parseOptionalPath(raw.modelsPath), + name, + protocol + }; +} + +// only keep the override keys the protocol understands, so a stale persisted +// value can never inject an unknown field into a request +function parseBackendCompat( + raw: unknown, + protocol: BackendProtocol +): Partial | undefined { + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) return undefined; + + const entry = raw as Record; + const defaults = BACKEND_COMPAT[protocol] ?? BACKEND_COMPAT.openai; + const overrides: Partial = {}; + + if (entry.maxTokensField === 'max_tokens' || entry.maxTokensField === 'max_completion_tokens') { + overrides.maxTokensField = entry.maxTokensField; + } + + if (typeof entry.supportsUsageInStreaming === 'boolean') { + overrides.supportsUsageInStreaming = entry.supportsUsageInStreaming; + } + + // drop a no-op override so an unmodified backend stays undefined + const isDefault = + (overrides.maxTokensField === undefined || + overrides.maxTokensField === defaults.maxTokensField) && + (overrides.supportsUsageInStreaming === undefined || + overrides.supportsUsageInStreaming === defaults.supportsUsageInStreaming); + + return isDefault ? undefined : overrides; +} + +function parseBackendHeaders(raw: unknown): Record | undefined { + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) return undefined; + + const entries = Object.entries(raw as Record) + .filter(([, value]) => typeof value === 'string' && value.trim() !== '') + .map(([key, value]) => [key.trim(), (value as string).trim()] as const); + + return entries.length > 0 ? Object.fromEntries(entries) : undefined; +} + +function parseOptionalPath(raw: unknown): string | undefined { + return typeof raw === 'string' && raw.trim() ? raw.trim() : undefined; +} + +/** + * Read a model's context size out of one `/v1/models` entry. Providers use + * different field names, and OpenRouter nests the authoritative value under + * `top_provider`, so try the flat fields first and the nested one after. + */ +export function readModelContextLength(value: unknown): number | undefined { + if (!value || typeof value !== 'object') return undefined; + + const entry = value as Record; + const flat = readContextField(entry); + + if (flat !== undefined) return flat; + + const topProvider = entry.top_provider; + + if (topProvider && typeof topProvider === 'object') { + const nested = readContextField(topProvider as Record); + + if (nested !== undefined) return nested; + } + + // Hugging Face lists one entry per inference provider, and they disagree on + // the budget; take the largest so the gauge does not undersell the model. + const providers = entry.providers; + + if (Array.isArray(providers)) { + const sizes = providers + .map((provider) => + provider && typeof provider === 'object' + ? readContextField(provider as Record) + : undefined + ) + .filter((size): size is number => size !== undefined); + + if (sizes.length > 0) return Math.max(...sizes); + } + + return undefined; +} + +function readContextField(entry: Record): number | undefined { + for (const field of MODEL_CONTEXT_LENGTH_FIELDS) { + const value = entry[field]; + + if (typeof value === 'number' && value > 0) return value; + } + + return undefined; +} diff --git a/tools/ui/src/lib/utils/cors-proxy.ts b/tools/ui/src/lib/utils/cors-proxy.ts index 58423b7e7d..147b6126f2 100644 --- a/tools/ui/src/lib/utils/cors-proxy.ts +++ b/tools/ui/src/lib/utils/cors-proxy.ts @@ -2,7 +2,7 @@ * CORS Proxy utility for routing requests through llama-server's CORS proxy. */ -import { base } from '$app/paths'; +import { apiUrl } from './api-base'; import { CORS_PROXY, CORS_PROXY_ENDPOINT } from '$lib/constants'; /** @@ -11,7 +11,7 @@ import { CORS_PROXY, CORS_PROXY_ENDPOINT } from '$lib/constants'; * @returns URL pointing to the CORS proxy with target encoded */ export function buildProxiedUrl(targetUrl: string): URL { - const proxyPath = `${base}${CORS_PROXY_ENDPOINT}`; + const proxyPath = apiUrl(CORS_PROXY_ENDPOINT); const proxyUrl = new URL(proxyPath, window.location.origin); proxyUrl.searchParams.set(CORS_PROXY.URL_PARAM, targetUrl); diff --git a/tools/ui/src/lib/utils/index.ts b/tools/ui/src/lib/utils/index.ts index 1e328d5348..3c8a3caf8c 100644 --- a/tools/ui/src/lib/utils/index.ts +++ b/tools/ui/src/lib/utils/index.ts @@ -8,9 +8,30 @@ */ // API utilities -export { getAuthHeaders, getJsonHeaders, sanitizeHeaders } from './api-headers'; +export { + apiChatUrl, + apiModelsUrl, + apiUrl, + getBackend, + getBackendBaseUrl, + type BackendsSnapshot +} from './api-base'; +export { + getAuthHeaders, + getAuthHeadersForBackend, + getJsonHeaders, + sanitizeHeaders +} from './api-headers'; export { ApiError, apiDelete, apiFetch, apiFetchWithParams, apiPost } from './api-fetch'; export { validateApiKey } from './api-key-validation'; +export { + backendChatUrl, + backendFaviconUrl, + backendModelsUrl, + createLocalBackend, + getBackendCapabilities, + parseBackendsSettings +} from './backend'; // Attachment utilities export { getAttachmentDisplayItems, isMcpPrompt, isMcpResource } from './attachment-display'; @@ -107,6 +128,10 @@ export { // Model name utilities export { isValidModelName, normalizeModelName, orgOf } from './model-names'; +// Backend-qualified model option ids +export { backendIdFromModelId, qualifyModelId, rawModelId } from './model-option-id'; +export { groupModelFamilies, modelFamilyKey, type ModelFamilyGroup } from './model-families'; + // Sidecar token utilities export { isAuxSidecar, isDraftSidecar, sidecarFromFileToken, sidecarFromTag } from './sidecars'; @@ -138,6 +163,8 @@ export { extractSseDataPayload, parseSseJsonStream, splitSseRecords } from './ss // Stream session identity (conversation-id based) export { streamIdentity } from './stream-identity'; +export { buildTimingsFromUsage, usageTokenCounts } from './timings'; + // MCP utilities export { detectMcpTransportFromUrl, @@ -360,7 +387,6 @@ export { remToPx } from './css'; // Audio format helper (used by agentic store and chat service) export { getAudioInputFormat } from './audio-format'; -export { groupModelFamilies, modelFamilyKey, type ModelFamilyGroup } from './model-families'; // Svelte actions export { nearViewport } from './near-viewport'; diff --git a/tools/ui/src/lib/utils/model-option-id.ts b/tools/ui/src/lib/utils/model-option-id.ts new file mode 100644 index 0000000000..b856cb39dc --- /dev/null +++ b/tools/ui/src/lib/utils/model-option-id.ts @@ -0,0 +1,28 @@ +/** + * Backend-qualified model option ids. + * + * The selector lists models from every enabled backend, so a bare model id can + * collide. Option ids are qualified as `::`; selection + * strips the prefix to find the row in the active backend's list. + */ + +const MODEL_OPTION_ID_SEPARATOR = '::'; + +/** Prefix a raw model id with the backend that serves it. */ +export function qualifyModelId(backendId: string, modelId: string): string { + return `${backendId}${MODEL_OPTION_ID_SEPARATOR}${modelId}`; +} + +/** Backend id of a qualified model option id, or null when unqualified. */ +export function backendIdFromModelId(qualifiedId: string): string | null { + const index = qualifiedId.indexOf(MODEL_OPTION_ID_SEPARATOR); + + return index === -1 ? null : qualifiedId.slice(0, index); +} + +/** Strip the backend prefix from a qualified model option id. */ +export function rawModelId(qualifiedId: string): string { + const index = qualifiedId.indexOf(MODEL_OPTION_ID_SEPARATOR); + + return index === -1 ? qualifiedId : qualifiedId.slice(index + MODEL_OPTION_ID_SEPARATOR.length); +} diff --git a/tools/ui/src/lib/utils/timings.ts b/tools/ui/src/lib/utils/timings.ts new file mode 100644 index 0000000000..a6478e8441 --- /dev/null +++ b/tools/ui/src/lib/utils/timings.ts @@ -0,0 +1,66 @@ +/** + * Client side timing fallback for backends that do not report their own. + * + * llama.cpp streams per-token timings; OpenAI-compatible servers + * do not. Token counts come from the usage block of the final chunk (or the + * count of streamed deltas as a fallback), times are measured locally: the wait + * for the first token is attributed to prompt processing, the rest to + * generation. Wall clock, so network and queueing are part of the numbers. + */ + +import type { ApiChatCompletionUsage } from '$lib/types/api'; +import type { ChatMessageTimings } from '$lib/types/chat'; + +export interface StreamClock { + startedAt: number; + firstTokenAt: number | null; + lastTokenAt: number | null; +} + +/** + * Prompt/output/cache token counts. `promptTokens` excludes the cache read + * tokens, which are returned separately as `cacheTokens`, so the two always + * add up to the prompt size. + */ +export function usageTokenCounts(usage: ApiChatCompletionUsage | undefined): { + cacheTokens: number; + completionTokens: number; + promptTokens: number; +} { + // a total that includes the cache reads, which are reported separately + const cacheTokens = + usage?.prompt_tokens_details?.cached_tokens ?? + usage?.prompt_cache_hit_tokens ?? + usage?.cached_tokens ?? + 0; + const promptTotal = usage?.prompt_tokens ?? 0; + + return { + cacheTokens, + completionTokens: usage?.completion_tokens ?? 0, + promptTokens: Math.max(0, promptTotal - cacheTokens) + }; +} + +export function buildTimingsFromUsage( + usage: ApiChatCompletionUsage | undefined, + clock: StreamClock, + fallbackTokens = 0 +): ChatMessageTimings | null { + const { cacheTokens, completionTokens, promptTokens } = usageTokenCounts(usage); + const predictedN = completionTokens || fallbackTokens; + + if (promptTokens === 0 && predictedN === 0) return null; + + const { firstTokenAt, startedAt } = clock; + const lastTokenAt = clock.lastTokenAt ?? firstTokenAt; + + return { + cache_n: cacheTokens, + // clamp so a one-token reply still reports a positive duration + predicted_ms: firstTokenAt && lastTokenAt ? Math.max(1, lastTokenAt - firstTokenAt) : undefined, + predicted_n: predictedN, + prompt_ms: firstTokenAt ? Math.max(1, firstTokenAt - startedAt) : undefined, + prompt_n: promptTokens + }; +}