mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-29 01:17:36 -05:00
ui : route chat through the active provider
Chat goes through protocol adapters, so an OpenAI-compatible endpoint speaks its own wire format: per-backend paths and headers, the model on the request, tools kept on the local server, and token counts synthesized for endpoints that do not stream their own timings. The server store keeps the local server's props while another provider is active, a conversation resolves the provider its model belongs to before sending, and the chat screen never blocks on the local probe when the install has none. Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
This commit is contained in:
+1
-1
@@ -73,7 +73,7 @@
|
||||
</span>
|
||||
</div>
|
||||
|
||||
{#if gauge.activeModelId !== null && !gauge.isActiveModelLoaded}
|
||||
{#if gauge.canLoadActiveModel && gauge.activeModelId !== null && !gauge.isActiveModelLoaded}
|
||||
<ContextGaugeLoadModel
|
||||
isLoading={gauge.isActiveModelLoading}
|
||||
modelId={gauge.activeModelId}
|
||||
|
||||
-1
@@ -166,7 +166,6 @@
|
||||
<ChatMessageAssistantModel
|
||||
{displayedModel}
|
||||
isLoading={chatStore.isLoading}
|
||||
{isRouter}
|
||||
{onRegenerate}
|
||||
/>
|
||||
|
||||
|
||||
+15
-7
@@ -1,17 +1,21 @@
|
||||
<script lang="ts">
|
||||
import { ModelBadge, ModelsSelectorDropdown } from '$lib/components/app';
|
||||
import { ServerModelStatus } from '$lib/enums';
|
||||
import { modelsStore } from '$lib/stores';
|
||||
import { copyToClipboard } from '$lib/utils';
|
||||
import { modelsStore, serverStore } from '$lib/stores';
|
||||
import { copyToClipboard, getBackendCapabilities } from '$lib/utils';
|
||||
import { getBackend } from '$lib/utils/api-base';
|
||||
|
||||
interface Props {
|
||||
displayedModel: string | null;
|
||||
isRouter: boolean;
|
||||
isLoading: boolean;
|
||||
onRegenerate: (modelOverride?: string) => void;
|
||||
}
|
||||
|
||||
let { displayedModel, isLoading, isRouter, onRegenerate }: Props = $props();
|
||||
let { displayedModel, isLoading, onRegenerate }: Props = $props();
|
||||
|
||||
// same selectability rule as the form selector: router mode, or any backend
|
||||
// that exposes a selectable model list
|
||||
let isSelectable = $derived(serverStore.isRouterMode || !serverStore.capabilities.props);
|
||||
|
||||
let pendingModel = $state<string | null>(null);
|
||||
|
||||
@@ -20,14 +24,18 @@
|
||||
}
|
||||
</script>
|
||||
|
||||
{#if isRouter}
|
||||
{#if isSelectable}
|
||||
<ModelsSelectorDropdown
|
||||
currentModel={pendingModel ?? displayedModel}
|
||||
disabled={isLoading}
|
||||
onModelChange={async (modelId: string, modelName: string) => {
|
||||
onModelChange={async (modelId: string, modelName: string, backendId?: string) => {
|
||||
// capability of the picked model's own backend, not the active one
|
||||
const loadsOnRequest = getBackendCapabilities(getBackend(backendId)).loadUnload;
|
||||
const status = modelsStore.getModelStatus(modelId);
|
||||
|
||||
if (status !== ServerModelStatus.LOADED) {
|
||||
// only a llama.cpp server loads up front; remote backends load the
|
||||
// model with the request itself
|
||||
if (loadsOnRequest && status !== ServerModelStatus.LOADED) {
|
||||
pendingModel = modelId;
|
||||
|
||||
try {
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
} from '$lib/components/app';
|
||||
import { LANDING_SETTLE_MAX_MS, LANDING_STABLE_FRAMES, ROUTES } from '$lib/constants';
|
||||
import { createAutoScrollController } from '$lib/hooks/use-auto-scroll.svelte';
|
||||
import { useBackendAvailability } from '$lib/hooks/use-backend-availability.svelte';
|
||||
import { useChatScreenActiveModel } from '$lib/hooks/use-chat-screen-active-model.svelte';
|
||||
import { useChatScreenDragAndDrop } from '$lib/hooks/use-chat-screen-drag-and-drop.svelte';
|
||||
import { useChatScreenFileUpload } from '$lib/hooks/use-chat-screen-file-upload.svelte';
|
||||
@@ -44,8 +45,10 @@
|
||||
showCenteredEmpty && conversationsStore.activeMessages.length === 0 && !chatStore.isLoading
|
||||
);
|
||||
let activeErrorDialog = $derived(chatStore.errorDialogState);
|
||||
let isServerLoading = $derived(serverStore.loading);
|
||||
let hasPropsError = $derived(!!serverStore.error);
|
||||
// no local server in this deployment: never block on its probe
|
||||
let isServerLoading = $derived(serverStore.loading && !serverStore.localServerMissing);
|
||||
const availability = useBackendAvailability();
|
||||
let hasPropsError = $derived(availability.isOffline);
|
||||
let isCurrentConversationLoading = $derived(chatStore.isLoading || chatStore.isStreaming());
|
||||
let chatFormBottomPosition = $derived.by(() => {
|
||||
if (!deviceStore.isMobile) return '1rem';
|
||||
|
||||
@@ -17,7 +17,7 @@
|
||||
<h1 class="mb-2 text-2xl font-semibold tracking-tight md:text-3xl">Hello there</h1>
|
||||
|
||||
<p class="text-muted-foreground md:text-lg">
|
||||
{serverStore.props?.modalities?.audio ? 'Record audio, type a message ' : 'Type a message'} or upload
|
||||
files to get started
|
||||
{serverStore.localProps?.modalities?.audio ? 'Record audio, type a message ' : 'Type a message'} or
|
||||
upload files to get started
|
||||
</p>
|
||||
</div>
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
<script lang="ts">
|
||||
import { AlertTriangle, CheckCircle, Key, RefreshCw, XCircle } from '@lucide/svelte';
|
||||
import { goto } from '$app/navigation';
|
||||
import { base } from '$app/paths';
|
||||
import { Button } from '$lib/components/ui/button';
|
||||
import { Input } from '$lib/components/ui/input';
|
||||
import Label from '$lib/components/ui/label/label.svelte';
|
||||
import { HEADERS, ICON_CLASS_DEFAULT, ROUTES, SETTINGS_KEYS } from '$lib/constants';
|
||||
import { KeyboardKey } from '$lib/enums';
|
||||
import { serverStore, settingsStore } from '$lib/stores';
|
||||
import { apiUrl } from '$lib/utils/api-base';
|
||||
import { fade, fly, scale } from 'svelte/transition';
|
||||
|
||||
interface Props {
|
||||
@@ -67,7 +67,7 @@
|
||||
settingsStore.updateConfig(SETTINGS_KEYS.API_KEY, apiKeyInput.trim());
|
||||
|
||||
// Test the API key by making a real request to the server
|
||||
const response = await fetch(`${base}/props`, {
|
||||
const response = await fetch(apiUrl('/props'), {
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
[HEADERS.AUTHORIZATION]: `${HEADERS.BEARER}${apiKeyInput.trim()}`
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
let error = $derived(serverStore.error);
|
||||
let loading = $derived(serverStore.loading);
|
||||
let model = $derived(modelsStore.singleModelName);
|
||||
let serverData = $derived(serverStore.props);
|
||||
let serverData = $derived(serverStore.localProps);
|
||||
|
||||
function getStatusColor() {
|
||||
if (loading) return 'bg-yellow-500';
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
/**
|
||||
* Backend availability for connection error states.
|
||||
*
|
||||
* The server error banner and the offline affordances should only appear when
|
||||
* no backend can serve requests. The active backend failing is not enough:
|
||||
* with an external backend configured, the UI stays usable.
|
||||
*/
|
||||
|
||||
import { LOCAL_BACKEND_ID } from '$lib/constants';
|
||||
import { backendsModelsStore, backendsStore, serverStore } from '$lib/stores';
|
||||
|
||||
export interface BackendAvailability {
|
||||
readonly hasAvailableBackend: boolean;
|
||||
readonly isOffline: boolean;
|
||||
}
|
||||
|
||||
export function useBackendAvailability(): BackendAvailability {
|
||||
const hasAvailableBackend = $derived.by(() => {
|
||||
// no connection error at all: nothing to report
|
||||
if (!serverStore.error) return true;
|
||||
|
||||
// the local server failed, but an external backend may still be usable
|
||||
return backendsStore.enabled.some(
|
||||
(backend) =>
|
||||
backend.id !== LOCAL_BACKEND_ID && backendsModelsStore.get(backend.id).error === null
|
||||
);
|
||||
});
|
||||
|
||||
return {
|
||||
get hasAvailableBackend() {
|
||||
return hasAvailableBackend;
|
||||
},
|
||||
|
||||
get isOffline() {
|
||||
return !hasAvailableBackend;
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -8,10 +8,12 @@ import { useProcessingState } from './use-processing-state.svelte';
|
||||
import { colorLevelFromPercent } from '$lib/components/app/chat/ChatForm/ChatFormContextGauge/context-gauge';
|
||||
import { STATS_UNITS } from '$lib/constants';
|
||||
import { ColorLevel } from '$lib/enums';
|
||||
import { contextStatsStore, modelsStore } from '$lib/stores';
|
||||
import { contextStatsStore, modelsStore, serverStore } from '$lib/stores';
|
||||
|
||||
export interface UseContextGaugeReturn {
|
||||
readonly activeModelId: string | null;
|
||||
/** Whether the active backend can load the model it is serving. */
|
||||
readonly canLoadActiveModel: boolean;
|
||||
readonly isActiveModelLoaded: boolean;
|
||||
readonly isActiveModelLoading: boolean;
|
||||
readonly contextTotal: number | null;
|
||||
@@ -73,6 +75,9 @@ export function useContextGauge(): UseContextGaugeReturn {
|
||||
contextStatsStore.averageTokensPerSecond !== null ||
|
||||
transientDetails.length > 0
|
||||
);
|
||||
// loading is a llama.cpp router feature, and the gauge tracks the active model,
|
||||
// so the active backend decides whether a load is possible at all
|
||||
const canLoadActiveModel = $derived(serverStore.isRouterMode);
|
||||
|
||||
async function loadModel() {
|
||||
const modelId = contextStatsStore.activeModelId;
|
||||
@@ -93,6 +98,9 @@ export function useContextGauge(): UseContextGaugeReturn {
|
||||
get averageTokensPerSecond() {
|
||||
return contextStatsStore.averageTokensPerSecond;
|
||||
},
|
||||
get canLoadActiveModel() {
|
||||
return canLoadActiveModel;
|
||||
},
|
||||
get colorLevel() {
|
||||
return colorLevel;
|
||||
},
|
||||
|
||||
@@ -20,6 +20,7 @@ import {
|
||||
SSE_DATA_PREFIX,
|
||||
SSE_DONE_MARKER,
|
||||
SSE_LINE_SEPARATOR,
|
||||
STREAM_LIVE_TIMINGS_INTERVAL_MS,
|
||||
STREAM_QUERY_PARAMS,
|
||||
STREAM_RESUME_LOCALSTORAGE_KEY_PREFIX,
|
||||
STREAM_VISIBILITY_KICK_MS
|
||||
@@ -32,7 +33,10 @@ import {
|
||||
ReasoningFormat,
|
||||
StreamConnectionState
|
||||
} from '$lib/enums';
|
||||
import { getProtocolAdapter } from '$lib/services/protocols';
|
||||
import { extractModelName } from '$lib/services/protocols/openai';
|
||||
import { modelsStore } from '$lib/stores/models/index.svelte';
|
||||
import { serverStore } from '$lib/stores/server.svelte';
|
||||
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
||||
import type { DatabaseMessageExtraMcpPrompt, DatabaseMessageExtraMcpResource } from '$lib/types';
|
||||
import type {
|
||||
@@ -42,10 +46,12 @@ import type {
|
||||
ApiStreamSession
|
||||
} from '$lib/types/api';
|
||||
import { isAbortError } from '$lib/utils/abort';
|
||||
import { apiChatUrl, apiUrl, getBackend } from '$lib/utils/api-base';
|
||||
import { ApiError } from '$lib/utils/api-fetch';
|
||||
import { getAuthHeaders, getJsonHeaders } from '$lib/utils/api-headers';
|
||||
import { formatAttachmentText } from '$lib/utils/formatters';
|
||||
import { streamIdentity } from '$lib/utils/stream-identity';
|
||||
import { buildTimingsFromUsage } from '$lib/utils/timings';
|
||||
|
||||
interface ResumableStreamState {
|
||||
bytesReceived: number;
|
||||
@@ -86,9 +92,12 @@ export class ChatService {
|
||||
* @returns {Promise<boolean>} Promise that resolves to true if all slots are idle, false if any is processing
|
||||
*/
|
||||
static async areAllSlotsIdle(model?: string | null, signal?: AbortSignal): Promise<boolean> {
|
||||
// the /slots endpoint only exists on llama.cpp servers
|
||||
if (!serverStore.capabilities.slots) return true;
|
||||
|
||||
try {
|
||||
const url = model ? `${API_SLOTS.LIST}?model=${encodeURIComponent(model)}` : API_SLOTS.LIST;
|
||||
const res = await fetch(url, { signal });
|
||||
const res = await fetch(apiUrl(url), { signal });
|
||||
|
||||
if (!res.ok) return true;
|
||||
|
||||
@@ -106,6 +115,8 @@ export class ChatService {
|
||||
static async cancelServerStream(conversationId: string, model?: string | null): Promise<void> {
|
||||
if (!conversationId) return;
|
||||
|
||||
if (!serverStore.capabilities.resumableStreams) return;
|
||||
|
||||
try {
|
||||
const id = streamIdentity(conversationId, model);
|
||||
|
||||
@@ -339,6 +350,10 @@ export class ChatService {
|
||||
* caller can pipe it through the SSE parser like a fresh stream.
|
||||
*/
|
||||
static async fetchStreamReplay(streamId: string): Promise<Response> {
|
||||
if (!serverStore.capabilities.resumableStreams) {
|
||||
return new Response(null, { status: 501, statusText: 'Not Implemented' });
|
||||
}
|
||||
|
||||
const resp = await fetch(ChatService.buildStreamUrl(streamId, 0), {
|
||||
headers: getAuthHeaders()
|
||||
});
|
||||
@@ -482,6 +497,17 @@ export class ChatService {
|
||||
let toolCallIndexOffset = 0;
|
||||
let hasOpenToolCallBatch = false;
|
||||
|
||||
// client side clock for backends that do not stream their own timings
|
||||
const startedAt = Date.now();
|
||||
|
||||
let firstTokenAt: number | null = null;
|
||||
let lastTokenAt: number | null = null;
|
||||
let streamedTokens = 0;
|
||||
let liveTimingsAt = 0;
|
||||
let usage: ApiChatCompletionUsage | undefined;
|
||||
|
||||
// the protocol decides how payloads map onto canonical events
|
||||
const streamReader = getProtocolAdapter(getBackend()).createStreamReader();
|
||||
const finalizeOpenToolCallBatch = () => {
|
||||
if (!hasOpenToolCallBatch) {
|
||||
return;
|
||||
@@ -521,6 +547,32 @@ export class ChatService {
|
||||
onToolCallChunk?.(serializedToolCalls);
|
||||
}
|
||||
};
|
||||
// backends that do not stream their own timings report progress from wall
|
||||
// clock time and the streamed delta count, throttled to keep updates cheap
|
||||
const markToken = () => {
|
||||
firstTokenAt ??= Date.now();
|
||||
lastTokenAt = Date.now();
|
||||
streamedTokens++;
|
||||
|
||||
if (
|
||||
serverStore.capabilities.props ||
|
||||
Date.now() - liveTimingsAt < STREAM_LIVE_TIMINGS_INTERVAL_MS
|
||||
) {
|
||||
return;
|
||||
}
|
||||
|
||||
liveTimingsAt = Date.now();
|
||||
|
||||
const liveTimings = buildTimingsFromUsage(
|
||||
usage,
|
||||
{ firstTokenAt, lastTokenAt, startedAt },
|
||||
streamedTokens
|
||||
);
|
||||
|
||||
if (liveTimings) {
|
||||
ChatService.notifyTimings(liveTimings, undefined, onTimings);
|
||||
}
|
||||
};
|
||||
const onVisibilityChange = () => {
|
||||
if (typeof document === 'undefined') return;
|
||||
|
||||
@@ -622,56 +674,89 @@ export class ChatService {
|
||||
continue;
|
||||
}
|
||||
|
||||
let parsed: unknown;
|
||||
|
||||
try {
|
||||
const parsed: ApiChatCompletionStreamChunk = JSON.parse(data);
|
||||
const choice = parsed.choices?.[0];
|
||||
const content = choice?.delta?.content;
|
||||
const reasoningContent = choice?.delta?.reasoning_content;
|
||||
const toolCalls = choice?.delta?.tool_calls;
|
||||
const timings = parsed.timings;
|
||||
const promptProgress = parsed.prompt_progress;
|
||||
const chunkModel = ChatService.extractModelName(parsed);
|
||||
parsed = JSON.parse(data);
|
||||
} catch (parseError) {
|
||||
console.error('Error parsing JSON chunk:', parseError);
|
||||
|
||||
if (chunkModel && !modelEmitted) {
|
||||
modelEmitted = true;
|
||||
onModel?.(chunkModel);
|
||||
continue;
|
||||
}
|
||||
|
||||
for (const event of streamReader.readChunk(parsed)) {
|
||||
switch (event.type) {
|
||||
case 'done':
|
||||
streamFinished = true;
|
||||
|
||||
break;
|
||||
|
||||
case 'error':
|
||||
throw new Error(event.message);
|
||||
|
||||
case 'id':
|
||||
if (!idEmitted) {
|
||||
idEmitted = true;
|
||||
onCompletionId?.(event.id);
|
||||
}
|
||||
|
||||
break;
|
||||
|
||||
case 'model':
|
||||
if (!modelEmitted) {
|
||||
modelEmitted = true;
|
||||
onModel?.(event.model);
|
||||
}
|
||||
|
||||
break;
|
||||
|
||||
case 'prompt_progress':
|
||||
ChatService.notifyTimings(undefined, event.progress, onTimings);
|
||||
|
||||
break;
|
||||
|
||||
case 'text':
|
||||
finalizeOpenToolCallBatch();
|
||||
aggregatedContent += event.text;
|
||||
|
||||
if (!abortSignal?.aborted) {
|
||||
onChunk?.(event.text);
|
||||
}
|
||||
|
||||
markToken();
|
||||
|
||||
break;
|
||||
|
||||
case 'thinking':
|
||||
finalizeOpenToolCallBatch();
|
||||
fullReasoningContent += event.text;
|
||||
|
||||
if (!abortSignal?.aborted) {
|
||||
onReasoningChunk?.(event.text);
|
||||
}
|
||||
|
||||
markToken();
|
||||
|
||||
break;
|
||||
|
||||
case 'timings':
|
||||
ChatService.notifyTimings(event.timings, event.promptProgress, onTimings);
|
||||
lastTimings = event.timings;
|
||||
|
||||
break;
|
||||
|
||||
case 'tool_calls':
|
||||
processToolCallDelta(event.deltas);
|
||||
|
||||
break;
|
||||
|
||||
case 'usage':
|
||||
// providers may split usage across chunks (some report input
|
||||
// tokens on message_start and output tokens on message_delta)
|
||||
usage = { ...usage, ...event.usage };
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
if (parsed.id && !idEmitted) {
|
||||
idEmitted = true;
|
||||
onCompletionId?.(parsed.id);
|
||||
}
|
||||
|
||||
if (promptProgress) {
|
||||
ChatService.notifyTimings(undefined, promptProgress, onTimings);
|
||||
}
|
||||
|
||||
if (timings) {
|
||||
ChatService.notifyTimings(timings, promptProgress, onTimings);
|
||||
lastTimings = timings;
|
||||
}
|
||||
|
||||
if (content) {
|
||||
finalizeOpenToolCallBatch();
|
||||
aggregatedContent += content;
|
||||
|
||||
if (!abortSignal?.aborted) {
|
||||
onChunk?.(content);
|
||||
}
|
||||
}
|
||||
|
||||
if (reasoningContent) {
|
||||
finalizeOpenToolCallBatch();
|
||||
fullReasoningContent += reasoningContent;
|
||||
|
||||
if (!abortSignal?.aborted) {
|
||||
onReasoningChunk?.(reasoningContent);
|
||||
}
|
||||
}
|
||||
|
||||
processToolCallDelta(toolCalls);
|
||||
} catch (e) {
|
||||
console.error('Error parsing JSON chunk:', e);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -742,6 +827,20 @@ export class ChatService {
|
||||
if (streamFinished) {
|
||||
finalizeOpenToolCallBatch();
|
||||
|
||||
// external backends report token counts only in the final usage chunk
|
||||
if (!lastTimings && !serverStore.capabilities.props) {
|
||||
lastTimings =
|
||||
buildTimingsFromUsage(
|
||||
usage,
|
||||
{ firstTokenAt, lastTokenAt, startedAt },
|
||||
streamedTokens
|
||||
) ?? undefined;
|
||||
|
||||
if (lastTimings) {
|
||||
ChatService.notifyTimings(lastTimings, undefined, onTimings);
|
||||
}
|
||||
}
|
||||
|
||||
if (conversationId) {
|
||||
ChatService.clearStreamState(conversationId);
|
||||
}
|
||||
@@ -781,7 +880,9 @@ export class ChatService {
|
||||
* conv::model identity when a model was bound at POST time.
|
||||
*/
|
||||
static async lookupStreamSessions(conversationIds: string[]): Promise<ApiStreamSession[]> {
|
||||
const resp = await fetch(API_STREAM.LOOKUP, {
|
||||
if (!serverStore.capabilities.resumableStreams) return [];
|
||||
|
||||
const resp = await fetch(apiUrl(API_STREAM.LOOKUP), {
|
||||
body: JSON.stringify({ conversation_ids: conversationIds }),
|
||||
headers: getJsonHeaders(),
|
||||
method: 'POST'
|
||||
@@ -843,6 +944,9 @@ export class ChatService {
|
||||
excludeReasoning?: boolean,
|
||||
signal?: AbortSignal
|
||||
): Promise<void> {
|
||||
// pre-encode warms the llama.cpp KV cache and posts llama.cpp-only fields
|
||||
if (!serverStore.capabilities.props) return;
|
||||
|
||||
const normalizedMessages: ApiChatMessageData[] =
|
||||
await ChatService.normalizeMessagesForApi(messages);
|
||||
const requestBody: Record<string, unknown> = {
|
||||
@@ -869,7 +973,7 @@ export class ChatService {
|
||||
}
|
||||
|
||||
try {
|
||||
await fetch(API_CHAT.COMPLETIONS, {
|
||||
await fetch(apiChatUrl(), {
|
||||
body: JSON.stringify(requestBody),
|
||||
headers: getJsonHeaders(),
|
||||
method: 'POST',
|
||||
@@ -887,6 +991,8 @@ export class ChatService {
|
||||
static async probeResumeStatus(streamId: string): Promise<number> {
|
||||
if (!streamId) return 0;
|
||||
|
||||
if (!serverStore.capabilities.resumableStreams) return 0;
|
||||
|
||||
const ac = new AbortController();
|
||||
|
||||
try {
|
||||
@@ -910,6 +1016,8 @@ export class ChatService {
|
||||
): Promise<Response | null> {
|
||||
if (!conversationId) return null;
|
||||
|
||||
if (!serverStore.capabilities.resumableStreams) return null;
|
||||
|
||||
const state = ChatService.getStreamState(conversationId);
|
||||
const from = state?.bytesReceived ?? 0;
|
||||
const id = streamIdentity(conversationId, model);
|
||||
@@ -1222,16 +1330,26 @@ export class ChatService {
|
||||
|
||||
// tag streaming requests with the conversation id, this single header is the opt in for the
|
||||
// server side replay buffer and powers discoverActiveStream on tab reopen. with an explicit
|
||||
// model the ::model suffix keeps the per model session distinct
|
||||
if (stream && conversationId) {
|
||||
// model the ::model suffix keeps the per model session distinct. external providers do not
|
||||
// know the header and their CORS preflight rejects it, so only llama.cpp gets it
|
||||
if (stream && conversationId && serverStore.capabilities.resumableStreams) {
|
||||
headers[HEADERS.X_CONVERSATION_ID_HEADER] = streamIdentity(conversationId, options.model);
|
||||
// persist the pending stream before the fetch: a reload during the model load or
|
||||
// the prompt processing must still find its way back to the session once it exists
|
||||
ChatService.saveStreamState(conversationId, 0, options.model ?? null);
|
||||
}
|
||||
|
||||
const response = await fetch(API_CHAT.COMPLETIONS, {
|
||||
body: JSON.stringify(requestBody),
|
||||
// the protocol adapter owns the wire format: it strips llama.cpp-only
|
||||
// fields, applies the backend's token cap field and adds the usage chunk
|
||||
const backend = getBackend();
|
||||
const wireBody = backend
|
||||
? getProtocolAdapter(backend).buildChatRequest(
|
||||
requestBody as unknown as Record<string, unknown>,
|
||||
backend
|
||||
)
|
||||
: (requestBody as unknown as Record<string, unknown>);
|
||||
const response = await fetch(apiChatUrl(), {
|
||||
body: JSON.stringify(wireBody),
|
||||
headers,
|
||||
method: 'POST',
|
||||
signal
|
||||
@@ -1342,7 +1460,7 @@ export class ChatService {
|
||||
if (model) body.model = model;
|
||||
|
||||
try {
|
||||
const res = await fetch(API_CHAT.CONTROL, {
|
||||
const res = await fetch(apiUrl(API_CHAT.CONTROL), {
|
||||
body: JSON.stringify(body),
|
||||
headers: getJsonHeaders(),
|
||||
method: 'POST'
|
||||
@@ -1373,62 +1491,7 @@ export class ChatService {
|
||||
const query = `${STREAM_QUERY_PARAMS.CONV_ID}=${encodeURIComponent(streamId)}`;
|
||||
const offset = from === undefined ? '' : `&${STREAM_QUERY_PARAMS.FROM}=${from}`;
|
||||
|
||||
return `${API_STREAM.BASE}?${query}${offset}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extracts model name from Chat Completions API response data.
|
||||
* Handles various response formats including streaming chunks and final responses.
|
||||
*
|
||||
* WORKAROUND: In single model mode, llama-server returns a default/incorrect model name
|
||||
* in the response. We override it with the actual model name from serverStore.
|
||||
*
|
||||
* @param data - Raw response data from the Chat Completions API
|
||||
* @returns Model name string if found, undefined otherwise
|
||||
* @private
|
||||
*/
|
||||
private static extractModelName(data: unknown): string | undefined {
|
||||
const asRecord = (value: unknown): Record<string, unknown> | undefined => {
|
||||
return typeof value === 'object' && value !== null
|
||||
? (value as Record<string, unknown>)
|
||||
: undefined;
|
||||
};
|
||||
const getTrimmedString = (value: unknown): string | undefined => {
|
||||
return typeof value === 'string' && value.trim() ? value.trim() : undefined;
|
||||
};
|
||||
const root = asRecord(data);
|
||||
|
||||
if (!root) return undefined;
|
||||
|
||||
// 1) root (some implementations provide `model` at the top level)
|
||||
const rootModel = getTrimmedString(root.model);
|
||||
|
||||
if (rootModel) {
|
||||
return rootModel;
|
||||
}
|
||||
|
||||
// 2) streaming choice (delta) or final response (message)
|
||||
const firstChoice = Array.isArray(root.choices) ? asRecord(root.choices[0]) : undefined;
|
||||
|
||||
if (!firstChoice) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
// priority: delta.model (first chunk) else message.model (final response)
|
||||
const deltaModel = getTrimmedString(asRecord(firstChoice.delta)?.model);
|
||||
|
||||
if (deltaModel) {
|
||||
return deltaModel;
|
||||
}
|
||||
|
||||
const messageModel = getTrimmedString(asRecord(firstChoice.message)?.model);
|
||||
|
||||
if (messageModel) {
|
||||
return messageModel;
|
||||
}
|
||||
|
||||
// avoid guessing from non-standard locations (metadata, etc.)
|
||||
return undefined;
|
||||
return apiUrl(`${API_STREAM.BASE}?${query}${offset}`);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1463,7 +1526,7 @@ export class ChatService {
|
||||
}
|
||||
|
||||
const data: ApiChatCompletionResponse = JSON.parse(responseText);
|
||||
const responseModel = ChatService.extractModelName(data);
|
||||
const responseModel = extractModelName(data);
|
||||
|
||||
if (responseModel) {
|
||||
onModel?.(responseModel);
|
||||
|
||||
@@ -6,20 +6,46 @@
|
||||
* modelsStore and its status manager.
|
||||
*/
|
||||
|
||||
import { base } from '$app/paths';
|
||||
import { API_MODELS, MODEL_ID, type ModelSidecar, SIDECAR_TOKENS } from '$lib/constants';
|
||||
import {
|
||||
API_MODELS,
|
||||
LOCAL_BACKEND_ID,
|
||||
MODEL_ID,
|
||||
type ModelSidecar,
|
||||
SIDECAR_TOKENS
|
||||
} from '$lib/constants';
|
||||
import { ServerModelStatus } from '$lib/enums';
|
||||
import type { ParsedModelId } from '$lib/types/models';
|
||||
import type { ModelSidecarFile, ParsedModelId } from '$lib/types/models';
|
||||
import {
|
||||
apiDelete,
|
||||
apiFetch,
|
||||
apiModelsUrl,
|
||||
apiPost,
|
||||
apiUrl,
|
||||
extractSseDataPayload,
|
||||
normalizeModelName,
|
||||
sidecarFromFileToken,
|
||||
splitSseRecords
|
||||
} from '$lib/utils';
|
||||
import { getAuthHeaders } from '$lib/utils/api-headers';
|
||||
import { isAuxSidecar } from '$lib/utils/sidecars';
|
||||
|
||||
/** Sidecar token a file name carries, for the forms parsing an id as a model misses. */
|
||||
function sidecarTokenInFilename(modelId: string): ModelSidecar | null {
|
||||
// the token can sit after a colon (`org/model:mtp`), a dash or an underscore
|
||||
const name = modelId.toLowerCase();
|
||||
const match = SIDECAR_TOKENS.find((token) =>
|
||||
new RegExp(`(^|[-_:])${token}([-_.:]|$)`).test(name)
|
||||
);
|
||||
|
||||
return (match as ModelSidecar | undefined) ?? null;
|
||||
}
|
||||
|
||||
/** Parameter label a parsed id reports, e.g. `35B-A3B`. */
|
||||
function paramsLabel(parsed: ParsedModelId): string | null {
|
||||
if (!parsed.params) return null;
|
||||
|
||||
return `${parsed.params}${parsed.activatedParams ? `-${parsed.activatedParams}` : ''}`;
|
||||
}
|
||||
|
||||
export class ModelsService {
|
||||
private static readonly SSE_RECONNECT_MS = 1000;
|
||||
@@ -88,6 +114,50 @@ export class ModelsService {
|
||||
return apiPost<ApiModelsDownloadResponse>(API_MODELS.DOWNLOAD, payload);
|
||||
}
|
||||
|
||||
/**
|
||||
* True when a router entry id is a sidecar-only entry, e.g. `org/model:Q4_0-mtp`
|
||||
* or `org/model:mmproj`. Such entries mark a downloaded sidecar file, not a
|
||||
* loadable model, so the selector skips them.
|
||||
*/
|
||||
/**
|
||||
* Draft sidecars a listing carries as their own entries, keyed by the repo they
|
||||
* belong to. The router lists a downloaded sidecar as a model of its own, so this
|
||||
* keeps the pairing that the model list itself is filtered to drop.
|
||||
*/
|
||||
static draftSidecarsByRepo(response: ApiModelsListResponse): Record<string, ModelSidecarFile[]> {
|
||||
const byRepo: Record<string, ModelSidecarFile[]> = {};
|
||||
|
||||
for (const entry of response.data ?? []) {
|
||||
const parsed = ModelsService.parseModelId(entry.id);
|
||||
const sidecar = parsed.sidecar ?? sidecarTokenInFilename(entry.id);
|
||||
|
||||
if (!sidecar || isAuxSidecar(sidecar)) continue;
|
||||
|
||||
// the repo is the id with its quant tag and sidecar token taken off; naming
|
||||
// parts of the id are not touched, a model name may carry `-4b` for instance
|
||||
const model = entry.id
|
||||
.split(MODEL_ID.QUANTIZATION_SEPARATOR)[0]
|
||||
.replace(MODEL_ID.WEIGHT_EXTENSION_REGEX, '')
|
||||
.replace(new RegExp(`[-_ ]?${sidecar}([-_ ]?draft)?$`, 'i'), '');
|
||||
|
||||
if (!model) continue;
|
||||
|
||||
const files = (byRepo[model] ??= []);
|
||||
|
||||
if (files.some((file) => file.kind === sidecar)) continue;
|
||||
|
||||
files.push({
|
||||
id: entry.id,
|
||||
kind: sidecar,
|
||||
model,
|
||||
params: paramsLabel(parsed),
|
||||
quant: parsed.quantization
|
||||
});
|
||||
}
|
||||
|
||||
return byRepo;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a model is loaded based on its metadata.
|
||||
*
|
||||
@@ -98,16 +168,6 @@ export class ModelsService {
|
||||
return model.status.value === ServerModelStatus.LOADED;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a model is currently loading.
|
||||
*
|
||||
* @param model - Model data entry from the API response
|
||||
* @returns True if the model status is LOADING
|
||||
*/
|
||||
static isModelLoading(model: ApiModelDataEntry): boolean {
|
||||
return model.status.value === ServerModelStatus.LOADING;
|
||||
}
|
||||
|
||||
/**
|
||||
*
|
||||
*
|
||||
@@ -117,20 +177,35 @@ export class ModelsService {
|
||||
*/
|
||||
|
||||
/**
|
||||
* True when a router entry id is a sidecar-only entry, e.g. `org/model:Q4_0-mtp`
|
||||
* or `org/model:mmproj`. Such entries mark a downloaded sidecar file, not a
|
||||
* loadable model, so the selector skips them.
|
||||
* Check if a model is currently loading.
|
||||
*
|
||||
* @param model - Model data entry from the API response
|
||||
* @returns True if the model status is LOADING
|
||||
*/
|
||||
static isModelLoading(model: ApiModelDataEntry): boolean {
|
||||
return model.status.value === ServerModelStatus.LOADING;
|
||||
}
|
||||
|
||||
static isSidecarEntry(modelId: string): boolean {
|
||||
const idx = modelId.indexOf(MODEL_ID.QUANTIZATION_SEPARATOR);
|
||||
|
||||
if (idx === MODEL_ID.NOT_FOUND) return false;
|
||||
if (idx !== MODEL_ID.NOT_FOUND) {
|
||||
const tag = modelId.slice(idx + 1).toLowerCase();
|
||||
const dash = tag.lastIndexOf(MODEL_ID.SEGMENT_SEPARATOR);
|
||||
const token = dash === -1 ? tag : tag.slice(dash + 1);
|
||||
|
||||
const tag = modelId.slice(idx + 1).toLowerCase();
|
||||
const dash = tag.lastIndexOf(MODEL_ID.SEGMENT_SEPARATOR);
|
||||
const token = dash === -1 ? tag : tag.slice(dash + 1);
|
||||
if (SIDECAR_TOKENS.includes(token)) return true;
|
||||
}
|
||||
|
||||
return SIDECAR_TOKENS.includes(token);
|
||||
// the router also lists projector and draft files by filename, e.g.
|
||||
// `org/model-mmproj-F16.gguf` or `mmproj-model.gguf`
|
||||
const name = modelId.split('/').pop() ?? modelId;
|
||||
|
||||
return (
|
||||
MODEL_ID.SIDECAR_INFIX_REGEX.test(name) ||
|
||||
MODEL_ID.SIDECAR_PREFIX_REGEX.test(name) ||
|
||||
MODEL_ID.SIDECAR_SUFFIX_REGEX.test(name)
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -140,7 +215,7 @@ export class ModelsService {
|
||||
* @returns List of available models with basic metadata
|
||||
*/
|
||||
static async list(): Promise<ApiModelsListResponse> {
|
||||
return apiFetch<ApiModelsListResponse>(API_MODELS.LIST);
|
||||
return apiFetch<ApiModelsListResponse>(apiModelsUrl());
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -150,16 +225,21 @@ export class ModelsService {
|
||||
*
|
||||
* @param modelId - Model identifier to load
|
||||
* @param extraArgs - Optional additional arguments to pass to the model instance
|
||||
* @param backendId - Backend serving the model; the active one when omitted
|
||||
* @returns Load response from the server
|
||||
*/
|
||||
static async load(modelId: string, extraArgs?: string[]): Promise<ApiModelsLoadResponse> {
|
||||
static async load(
|
||||
modelId: string,
|
||||
extraArgs?: string[],
|
||||
backendId?: string
|
||||
): Promise<ApiModelsLoadResponse> {
|
||||
const payload: { model: string; extra_args?: string[] } = { model: modelId };
|
||||
|
||||
if (extraArgs && extraArgs.length > 0) {
|
||||
payload.extra_args = extraArgs;
|
||||
}
|
||||
|
||||
return apiPost<ApiModelsLoadResponse>(API_MODELS.LOAD, payload);
|
||||
return apiPost<ApiModelsLoadResponse>(API_MODELS.LOAD, payload, { backendId });
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -325,10 +405,11 @@ export class ModelsService {
|
||||
* before unloading completes — use polling to await actual unload status.
|
||||
*
|
||||
* @param modelId - Model identifier to unload
|
||||
* @param backendId - Backend serving the model; the active one when omitted
|
||||
* @returns Unload response from the server
|
||||
*/
|
||||
static async unload(modelId: string): Promise<ApiModelsUnloadResponse> {
|
||||
return apiPost<ApiModelsUnloadResponse>(API_MODELS.UNLOAD, { model: modelId });
|
||||
static async unload(modelId: string, backendId?: string): Promise<ApiModelsUnloadResponse> {
|
||||
return apiPost<ApiModelsUnloadResponse>(API_MODELS.UNLOAD, { model: modelId }, { backendId });
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -345,8 +426,10 @@ export class ModelsService {
|
||||
|
||||
while (!signal.aborted) {
|
||||
try {
|
||||
const response = await fetch(`${base}${API_MODELS.SSE}`, {
|
||||
headers: getAuthHeaders(),
|
||||
// the status feed only exists on the local llama.cpp server; pin the
|
||||
// request so an active external backend cannot redirect it
|
||||
const response = await fetch(apiUrl(API_MODELS.SSE, LOCAL_BACKEND_ID), {
|
||||
headers: getAuthHeaders(LOCAL_BACKEND_ID),
|
||||
signal
|
||||
});
|
||||
|
||||
|
||||
@@ -15,17 +15,21 @@ export class PropsService {
|
||||
* In ROUTER mode, returns server-wide settings without model-specific modalities.
|
||||
*
|
||||
* @param autoload - If false, prevents automatic model loading (default: false)
|
||||
* @param backendId - Backend to ask; defaults to the active one.
|
||||
* @returns Server properties including default generation settings and capabilities
|
||||
* @throws {Error} If the request fails or returns invalid data
|
||||
*/
|
||||
static async fetch(autoload = false): Promise<ApiLlamaCppServerProps> {
|
||||
static async fetch(autoload = false, backendId?: string): Promise<ApiLlamaCppServerProps> {
|
||||
const params: Record<string, string> = {};
|
||||
|
||||
if (!autoload) {
|
||||
params.autoload = 'false';
|
||||
}
|
||||
|
||||
return apiFetchWithParams<ApiLlamaCppServerProps>('./props', params, { authOnly: true });
|
||||
return apiFetchWithParams<ApiLlamaCppServerProps>('./props', params, {
|
||||
authOnly: true,
|
||||
backendId
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
/**
|
||||
* Protocol adapter registry.
|
||||
*
|
||||
* Resolves the wire mapping for a backend by its protocol. Unknown or
|
||||
* not-yet-loaded backends fall back to the OpenAI-compatible adapter, which is
|
||||
* what the local llama-server speaks.
|
||||
*/
|
||||
|
||||
import { openaiAdapter } from './openai';
|
||||
import type { ChatProtocolAdapter } from './types';
|
||||
import type { Backend, BackendProtocol } from '$lib/types';
|
||||
|
||||
const ADAPTERS: Record<BackendProtocol, ChatProtocolAdapter> = {
|
||||
'llama.cpp': openaiAdapter,
|
||||
openai: openaiAdapter
|
||||
};
|
||||
|
||||
export function getProtocolAdapter(backend?: Backend): ChatProtocolAdapter {
|
||||
if (!backend) return openaiAdapter;
|
||||
|
||||
return ADAPTERS[backend.protocol] ?? openaiAdapter;
|
||||
}
|
||||
|
||||
export type { ChatProtocolAdapter, ChatStreamEvent, ChatStreamReader } from './types';
|
||||
@@ -0,0 +1,183 @@
|
||||
/**
|
||||
* OpenAI-compatible protocol: llama-server, hosted OpenAI endpoints and any
|
||||
* compatible gateway. llama-server accepts a superset of the wire format, so
|
||||
* its only specialization is that nothing gets stripped.
|
||||
*/
|
||||
|
||||
import type { ChatProtocolAdapter, ChatStreamEvent, ChatStreamReader } from './types';
|
||||
import { HEADERS } from '$lib/constants';
|
||||
import type { Backend } from '$lib/types';
|
||||
import type { ApiChatCompletionStreamChunk } from '$lib/types/api';
|
||||
import { getBackendCompat } from '$lib/utils/backend';
|
||||
|
||||
/**
|
||||
* llama.cpp-only chat request fields. Strict OpenAI-compatible endpoints
|
||||
* reject unknown parameters, so they are dropped for those backends.
|
||||
*/
|
||||
const COMPAT_ONLY_OMIT_REQUEST_FIELDS = [
|
||||
'add_generation_prompt',
|
||||
'backend_sampling',
|
||||
'cache_prompt',
|
||||
'chat_template_kwargs',
|
||||
'continue_final_message',
|
||||
'dry_allowed_length',
|
||||
'dry_base',
|
||||
'dry_multiplier',
|
||||
'dry_penalty_last_n',
|
||||
'dynatemp_exponent',
|
||||
'dynatemp_range',
|
||||
'id_slot',
|
||||
'min_p',
|
||||
'n_keep',
|
||||
'n_predict',
|
||||
'reasoning_control',
|
||||
'reasoning_format',
|
||||
'repeat_last_n',
|
||||
'repeat_penalty',
|
||||
'return_progress',
|
||||
'samplers',
|
||||
'sse_ping_interval',
|
||||
'thinking_budget_tokens',
|
||||
'timings_per_token',
|
||||
'top_k',
|
||||
'typ_p',
|
||||
'xtc_probability',
|
||||
'xtc_threshold'
|
||||
];
|
||||
|
||||
function authHeaders(backend: Backend): Record<string, string> {
|
||||
const headers: Record<string, string> = { ...(backend.headers ?? {}) };
|
||||
const apiKey = backend.apiKey?.trim();
|
||||
|
||||
if (apiKey) {
|
||||
headers[HEADERS.AUTHORIZATION] = `${HEADERS.BEARER}${apiKey}`;
|
||||
}
|
||||
|
||||
return headers;
|
||||
}
|
||||
|
||||
function buildChatRequest(
|
||||
body: Record<string, unknown>,
|
||||
backend: Backend
|
||||
): Record<string, unknown> {
|
||||
if (backend.protocol === 'llama.cpp') return body;
|
||||
|
||||
const compat = getBackendCompat(backend);
|
||||
const request: Record<string, unknown> = { ...body };
|
||||
|
||||
for (const field of COMPAT_ONLY_OMIT_REQUEST_FIELDS) {
|
||||
delete request[field];
|
||||
}
|
||||
|
||||
// compatible endpoints reject the reasoning_content message extension
|
||||
const messages = request.messages as { reasoning_content?: string }[] | undefined;
|
||||
|
||||
for (const message of messages ?? []) {
|
||||
delete message.reasoning_content;
|
||||
}
|
||||
|
||||
// -1 is llama.cpp's "no limit" sentinel; compatible endpoints reject it
|
||||
if (typeof request.max_tokens === 'number' && request.max_tokens <= 0) {
|
||||
delete request.max_tokens;
|
||||
}
|
||||
|
||||
// newer OpenAI models require max_completion_tokens, most compatible
|
||||
// endpoints only understand max_tokens
|
||||
if (compat.maxTokensField === 'max_completion_tokens' && request.max_tokens !== undefined) {
|
||||
request.max_completion_tokens = request.max_tokens;
|
||||
delete request.max_tokens;
|
||||
}
|
||||
|
||||
// a final usage chunk is what the client side timing fallback reads
|
||||
if (request.stream && compat.supportsUsageInStreaming && request.stream_options === undefined) {
|
||||
request.stream_options = { include_usage: true };
|
||||
}
|
||||
|
||||
return request;
|
||||
}
|
||||
|
||||
/**
|
||||
* Model name a payload reports. Streaming chunks carry it on the delta, final
|
||||
* responses on the message, and some gateways on metadata or the choice itself.
|
||||
*/
|
||||
export function extractModelName(data: unknown): string | undefined {
|
||||
const asRecord = (value: unknown): Record<string, unknown> | undefined =>
|
||||
typeof value === 'object' && value !== null ? (value as Record<string, unknown>) : undefined;
|
||||
const getTrimmedString = (value: unknown): string | undefined =>
|
||||
typeof value === 'string' && value.trim() ? value.trim() : undefined;
|
||||
const root = asRecord(data);
|
||||
|
||||
if (!root) return undefined;
|
||||
|
||||
const rootModel = getTrimmedString(root.model);
|
||||
|
||||
if (rootModel) return rootModel;
|
||||
|
||||
const firstChoice = Array.isArray(root.choices) ? asRecord(root.choices[0]) : undefined;
|
||||
|
||||
if (!firstChoice) return undefined;
|
||||
|
||||
const metadataModel = getTrimmedString(asRecord(firstChoice.metadata)?.model);
|
||||
|
||||
if (metadataModel) return metadataModel;
|
||||
|
||||
const deltaModel = getTrimmedString(asRecord(firstChoice.delta)?.model);
|
||||
|
||||
if (deltaModel) return deltaModel;
|
||||
|
||||
const messageModel = getTrimmedString(asRecord(firstChoice.message)?.model);
|
||||
|
||||
if (messageModel) return messageModel;
|
||||
|
||||
return getTrimmedString(firstChoice.model);
|
||||
}
|
||||
|
||||
function readChunk(payload: unknown): ChatStreamEvent[] {
|
||||
if (!payload || typeof payload !== 'object') return [];
|
||||
|
||||
const chunk = payload as ApiChatCompletionStreamChunk;
|
||||
const events: ChatStreamEvent[] = [];
|
||||
const model = extractModelName(chunk);
|
||||
|
||||
if (chunk.id) events.push({ id: chunk.id, type: 'id' });
|
||||
|
||||
if (model) events.push({ model, type: 'model' });
|
||||
|
||||
if (chunk.prompt_progress) {
|
||||
events.push({ progress: chunk.prompt_progress, type: 'prompt_progress' });
|
||||
}
|
||||
|
||||
if (chunk.timings) {
|
||||
events.push({
|
||||
promptProgress: chunk.prompt_progress,
|
||||
timings: chunk.timings,
|
||||
type: 'timings'
|
||||
});
|
||||
}
|
||||
|
||||
if (chunk.usage) events.push({ type: 'usage', usage: chunk.usage });
|
||||
|
||||
const choice = chunk.choices?.[0];
|
||||
const delta = choice?.delta;
|
||||
|
||||
if (!delta) return events;
|
||||
|
||||
if (delta.content) events.push({ text: delta.content, type: 'text' });
|
||||
|
||||
// endpoints disagree on the reasoning field name; take the first one set
|
||||
const reasoning = delta.reasoning_content ?? delta.reasoning ?? delta.reasoning_text;
|
||||
|
||||
if (reasoning) events.push({ text: reasoning, type: 'thinking' });
|
||||
|
||||
if (delta.tool_calls) events.push({ deltas: delta.tool_calls, type: 'tool_calls' });
|
||||
|
||||
return events;
|
||||
}
|
||||
|
||||
export const openaiAdapter: ChatProtocolAdapter = {
|
||||
authHeaders,
|
||||
buildChatRequest,
|
||||
createStreamReader(): ChatStreamReader {
|
||||
return { readChunk };
|
||||
}
|
||||
};
|
||||
@@ -0,0 +1,42 @@
|
||||
/**
|
||||
* Backend protocol adapters.
|
||||
*
|
||||
* A backend speaks one wire protocol. ChatService owns the transport (fetch,
|
||||
* SSE framing, resume offsets) and delegates the parts that differ per
|
||||
* protocol here: credential headers, request shaping and stream decoding.
|
||||
*
|
||||
* Decoding is per-stream: a reader keeps per-stream state, so it must not be
|
||||
* shared between concurrent requests.
|
||||
*/
|
||||
|
||||
import type { Backend } from '$lib/types';
|
||||
import type { ApiChatCompletionToolCallDelta, ApiChatCompletionUsage } from '$lib/types/api';
|
||||
import type { ChatMessagePromptProgress, ChatMessageTimings } from '$lib/types/chat';
|
||||
|
||||
/** One canonical delta decoded from a backend's stream payload. */
|
||||
export type ChatStreamEvent =
|
||||
| { type: 'done' }
|
||||
| { type: 'error'; message: string }
|
||||
| { type: 'id'; id: string }
|
||||
| { type: 'model'; model: string }
|
||||
| { type: 'prompt_progress'; progress: ChatMessagePromptProgress }
|
||||
| { type: 'text'; text: string }
|
||||
| { type: 'thinking'; text: string }
|
||||
| { type: 'timings'; timings: ChatMessageTimings; promptProgress?: ChatMessagePromptProgress }
|
||||
| { type: 'tool_calls'; deltas: ApiChatCompletionToolCallDelta[] }
|
||||
| { type: 'usage'; usage: ApiChatCompletionUsage };
|
||||
|
||||
/** Decodes one stream's payloads. Create one per request. */
|
||||
export interface ChatStreamReader {
|
||||
/** Canonical deltas for one parsed SSE payload. */
|
||||
readChunk(payload: unknown): ChatStreamEvent[];
|
||||
}
|
||||
|
||||
/** Wire mapping for one protocol. */
|
||||
export interface ChatProtocolAdapter {
|
||||
/** Credential and protocol-required headers for a backend. */
|
||||
authHeaders(backend: Backend): Record<string, string>;
|
||||
/** Rewrite the canonical request body into this protocol's wire body. */
|
||||
buildChatRequest(body: Record<string, unknown>, backend: Backend): Record<string, unknown>;
|
||||
createStreamReader(): ChatStreamReader;
|
||||
}
|
||||
@@ -5,11 +5,10 @@
|
||||
* No reactive state; consumed by toolsStore.
|
||||
*/
|
||||
|
||||
import { base } from '$app/paths';
|
||||
import { API_TOOLS, HEADERS } from '$lib/constants';
|
||||
import { API_TOOLS, HEADERS, LOCAL_BACKEND_ID } from '$lib/constants';
|
||||
import { ToolResponseField } from '$lib/enums';
|
||||
import type { ServerToolInfo, ToolExecutionResult } from '$lib/types';
|
||||
import { apiFetch } from '$lib/utils';
|
||||
import { apiFetch, apiUrl } from '$lib/utils';
|
||||
import { getJsonHeaders } from '$lib/utils/api-headers';
|
||||
import { parseSseJsonStream, type SseJsonEvent } from '$lib/utils/sse';
|
||||
|
||||
@@ -28,6 +27,7 @@ export class ToolsService {
|
||||
cwd?: string
|
||||
): Promise<ToolExecutionResult> {
|
||||
const result = await apiFetch<Record<string, unknown>>(API_TOOLS.EXECUTE, {
|
||||
backendId: LOCAL_BACKEND_ID,
|
||||
body: JSON.stringify({ params, tool: toolName }),
|
||||
headers: cwd ? { [HEADERS.X_TOOL_CWD_HEADER]: cwd } : undefined,
|
||||
method: 'POST',
|
||||
@@ -67,6 +67,7 @@ export class ToolsService {
|
||||
if (respType) headers[HEADERS.X_RESP_TYPE_HEADER] = respType;
|
||||
|
||||
return apiFetch<Record<string, unknown>>(API_TOOLS.EXECUTE, {
|
||||
backendId: LOCAL_BACKEND_ID,
|
||||
body: JSON.stringify({ params, tool: toolName }),
|
||||
headers: Object.keys(headers).length > 0 ? headers : undefined,
|
||||
method: 'POST',
|
||||
@@ -80,7 +81,8 @@ export class ToolsService {
|
||||
* @returns Array of tool definitions in OpenAI-compatible format
|
||||
*/
|
||||
static async list(): Promise<ServerToolInfo[]> {
|
||||
return apiFetch<ServerToolInfo[]>(API_TOOLS.LIST);
|
||||
// pinned to the local server: server tools do not exist on other backends
|
||||
return apiFetch<ServerToolInfo[]>(API_TOOLS.LIST, { backendId: LOCAL_BACKEND_ID });
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -104,11 +106,11 @@ export class ToolsService {
|
||||
signal?: AbortSignal,
|
||||
cwd?: string
|
||||
): AsyncGenerator<ToolStreamEvent> {
|
||||
const headers = getJsonHeaders();
|
||||
const headers = getJsonHeaders(LOCAL_BACKEND_ID);
|
||||
|
||||
if (cwd) headers[HEADERS.X_TOOL_CWD_HEADER] = cwd;
|
||||
|
||||
const response = await fetch(`${base}${API_TOOLS.EXECUTE}`, {
|
||||
const response = await fetch(apiUrl(API_TOOLS.EXECUTE, LOCAL_BACKEND_ID), {
|
||||
body: JSON.stringify({ params, stream: true, tool: toolName }),
|
||||
headers,
|
||||
method: 'POST',
|
||||
|
||||
@@ -374,6 +374,11 @@ class ChatStore implements ChatStreamHost, ChatFlowsHost {
|
||||
if (serverStore.isRouterMode) {
|
||||
const modelName = modelsStore.selectedModelName;
|
||||
|
||||
if (modelName) apiOptions.model = modelName;
|
||||
} else if (!serverStore.capabilities.props) {
|
||||
// external backends need an explicit model on every request
|
||||
const modelName = modelsStore.activeModelId;
|
||||
|
||||
if (modelName) apiOptions.model = modelName;
|
||||
}
|
||||
|
||||
@@ -804,6 +809,15 @@ class ChatStore implements ChatStreamHost, ChatFlowsHost {
|
||||
modelOverride?: string | null,
|
||||
firstUserMessageContent?: string
|
||||
): Promise<void> {
|
||||
// a conversation keeps the model that generated it, which can belong to
|
||||
// another backend; make that backend active so the request is not sent to
|
||||
// a server that does not serve the model
|
||||
const requestedModel = modelOverride ?? getConversationModel(allMessages);
|
||||
|
||||
if (requestedModel) {
|
||||
await modelsStore.ensureModelBackend(requestedModel);
|
||||
}
|
||||
|
||||
// the ::model suffix in the stream identity is only for router mode, where it routes to the
|
||||
// owning child. in single-model mode the identity stays the bare conv id so that attach, stop
|
||||
// and reattach all agree, regardless of fresh send vs regenerate passing a resolved model
|
||||
@@ -813,6 +827,9 @@ class ChatStore implements ChatStreamHost, ChatFlowsHost {
|
||||
const conversationModel = getConversationModel(allMessages);
|
||||
|
||||
effectiveModel = modelOverride || modelsStore.selectedModelName || conversationModel;
|
||||
} else if (!serverStore.capabilities.props) {
|
||||
// external backends are not router mode but still require a model
|
||||
effectiveModel = modelOverride || modelsStore.activeModelId;
|
||||
}
|
||||
|
||||
if (serverStore.isRouterMode && effectiveModel) {
|
||||
@@ -1277,8 +1294,8 @@ class ChatStore implements ChatStreamHost, ChatFlowsHost {
|
||||
convId: string
|
||||
): Promise<void> {
|
||||
const effectiveModel =
|
||||
serverStore.isRouterMode && modelsStore.selectedModelName
|
||||
? modelsStore.selectedModelName
|
||||
serverStore.isRouterMode || !serverStore.capabilities.props
|
||||
? (modelsStore.activeModelId ?? undefined)
|
||||
: undefined;
|
||||
const configValue = settingsStore.config;
|
||||
const titlePromptTemplate =
|
||||
|
||||
@@ -37,6 +37,11 @@ export { conversationsStore } from './conversations/index.svelte';
|
||||
// MCP
|
||||
export { mcpStore } from './mcp/index.svelte';
|
||||
|
||||
// BACKENDS
|
||||
export { backendsStore } from './backends.svelte';
|
||||
|
||||
export { backendsModelsStore } from './backendsModels.svelte';
|
||||
|
||||
// MODELS
|
||||
export { modelsStore } from './models/index.svelte';
|
||||
|
||||
|
||||
@@ -1,6 +1,10 @@
|
||||
// direct imports, not via the barrel, to avoid circular deps
|
||||
import { backendsStore } from './backends.svelte';
|
||||
import { backendsModelsStore } from './backendsModels.svelte';
|
||||
import { conversationsStore } from './conversations/index.svelte';
|
||||
import { modelsStore } from './models/index.svelte';
|
||||
import { permissionsStore } from './permissions.svelte';
|
||||
import { serverStore } from './server.svelte';
|
||||
import { settingsStore } from './settings/index.svelte';
|
||||
import { tabsStore } from './tabs.svelte';
|
||||
import { toolsStore } from './tools.svelte';
|
||||
@@ -8,17 +12,47 @@ import { versionStore } from './version.svelte';
|
||||
import { browser } from '$app/environment';
|
||||
import { MigrationService } from '$lib/services/migration.service';
|
||||
|
||||
let hydration: Promise<void> | null = null;
|
||||
let startup: Promise<void> | null = null;
|
||||
|
||||
/**
|
||||
* Read the stored state the first request needs: migrations, the settings that
|
||||
* carry the API key, and the backend that request goes to.
|
||||
*
|
||||
* Separate from {@link initStores} so a `load` can await it. The requests
|
||||
* `initStores` starts belong after a load, where plain `window.fetch` is the
|
||||
* intended one and SvelteKit raises no warning.
|
||||
*/
|
||||
export function hydrateStores(): Promise<void> {
|
||||
if (!browser) return Promise.resolve();
|
||||
|
||||
hydration ??= (async () => {
|
||||
await MigrationService.runAllMigrations();
|
||||
|
||||
settingsStore.initialize();
|
||||
backendsStore.initialize();
|
||||
})();
|
||||
|
||||
return hydration;
|
||||
}
|
||||
|
||||
export function initStores(): Promise<void> {
|
||||
if (!browser) return Promise.resolve();
|
||||
|
||||
startup ??= (async () => {
|
||||
await MigrationService.runAllMigrations();
|
||||
await hydrateStores();
|
||||
|
||||
// prefetch every backend's model list in the background; failures are
|
||||
// per-backend and never block startup
|
||||
void backendsModelsStore.loadAll().then(() => modelsStore.warmHubDetails());
|
||||
|
||||
settingsStore.initialize();
|
||||
permissionsStore.initialize();
|
||||
toolsStore.initialize();
|
||||
|
||||
// the local server state backs the installation facts and decides whether
|
||||
// /tools exists at all, so probe it first and only then list the tools;
|
||||
// otherwise they stay empty until the tools menu is opened
|
||||
void serverStore.prefetchLocalState().then(() => toolsStore.fetchServerTools());
|
||||
void versionStore.initialize();
|
||||
|
||||
// the full conversation list loads in the background; once it is back,
|
||||
|
||||
@@ -11,26 +11,53 @@ import { browser } from '$app/environment';
|
||||
import {
|
||||
FAVORITE_MODELS_LOCALSTORAGE_KEY,
|
||||
HIDDEN_MODELS_LOCALSTORAGE_KEY,
|
||||
LOCAL_BACKEND_ID,
|
||||
MODEL_GROUP_OPEN_LOCALSTORAGE_KEY,
|
||||
MODEL_ROW_WINDOW,
|
||||
RECENT_MODEL_LIMIT,
|
||||
RECENT_MODEL_USAGE_LOCALSTORAGE_KEY,
|
||||
RECENT_MODELS_LOCALSTORAGE_KEY,
|
||||
SELECTED_MODEL_LOCALSTORAGE_KEY,
|
||||
SETTINGS_KEYS
|
||||
} from '$lib/constants';
|
||||
import { ServerModelStatus } from '$lib/enums';
|
||||
import { HuggingFaceService } from '$lib/services/huggingface.service';
|
||||
import { ModelsService } from '$lib/services/models.service';
|
||||
// direct imports between stores, not via the barrel, to avoid circular deps
|
||||
import { backendsStore } from '$lib/stores/backends.svelte';
|
||||
import { backendsModelsStore } from '$lib/stores/backendsModels.svelte';
|
||||
import { conversationsStore } from '$lib/stores/conversations/index.svelte';
|
||||
import { type ModelPropsHost, ModelPropsManager } from '$lib/stores/models/props.svelte';
|
||||
import { type ModelStatusHost, ModelStatusManager } from '$lib/stores/models/status.svelte';
|
||||
import { serverStore } from '$lib/stores/server.svelte';
|
||||
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
||||
import { getBackendCapabilities, readModelContextLength } from '$lib/utils/backend';
|
||||
import { getConversationModel } from '$lib/utils/conversation-utils';
|
||||
import { backendIdFromModelId, qualifyModelId, rawModelId } from '$lib/utils/model-option-id';
|
||||
import { SvelteMap, SvelteSet } from 'svelte/reactivity';
|
||||
import { toast } from 'svelte-sonner';
|
||||
|
||||
/** Selection kept from the last session, so a reload does not drop the picked model. */
|
||||
function loadStoredSelection(): { id: string; model: string | null } | null {
|
||||
if (!browser) return null;
|
||||
|
||||
try {
|
||||
const raw = localStorage.getItem(SELECTED_MODEL_LOCALSTORAGE_KEY);
|
||||
|
||||
if (!raw) return null;
|
||||
|
||||
const parsed = JSON.parse(raw) as { id?: unknown; model?: unknown };
|
||||
|
||||
if (typeof parsed?.id !== 'string' || !parsed.id) return null;
|
||||
|
||||
return { id: parsed.id, model: typeof parsed.model === 'string' ? parsed.model : null };
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
const storedSelection = loadStoredSelection();
|
||||
|
||||
/** Last use timestamp per backend-qualified model id. */
|
||||
function loadRecentModelUsage(): Record<string, number> {
|
||||
if (!browser) return {};
|
||||
@@ -99,17 +126,60 @@ function loadRecentModels(): string[] {
|
||||
}
|
||||
|
||||
class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
activeModels = $state<ModelOption[]>([]);
|
||||
error = $state<string | null>(null);
|
||||
favoriteModelIds = $state<Set<string>>(this.loadFavoritesFromStorage());
|
||||
groupOpenState = $state<SvelteMap<string, boolean>>(loadGroupOpenState());
|
||||
hiddenModelIds = $state<Set<string>>(loadHiddenModels());
|
||||
loading = $state(false);
|
||||
models = $state<ModelOption[]>([]);
|
||||
/**
|
||||
* Every selectable model across enabled backends. The active backend's
|
||||
* models come from {@link activeModels}; the rest come from the background
|
||||
* prefetch cache. Ids are backend-qualified so the same model name on two
|
||||
* backends stays distinct.
|
||||
*/
|
||||
/**
|
||||
* Computed once per state change. Rows reach this through per-model props
|
||||
* lookups, so a getter that rebuilt the list on every read made opening the
|
||||
* selector quadratic in the size of the catalog.
|
||||
*/
|
||||
models = $derived.by((): ModelOption[] => {
|
||||
const activeBackendId = backendsStore.active.id;
|
||||
const merged: ModelOption[] = [];
|
||||
const seen = new SvelteSet<string>();
|
||||
const push = (option: ModelOption, backendId: string) => {
|
||||
const id = qualifyModelId(backendId, rawModelId(option.id));
|
||||
|
||||
// a backend can be listed twice while a switch is in flight: the rows
|
||||
// of the previous backend are still in activeModels
|
||||
if (seen.has(id)) return;
|
||||
|
||||
seen.add(id);
|
||||
merged.push({ ...option, backendId, id });
|
||||
};
|
||||
|
||||
for (const option of this.activeModels) {
|
||||
// keep the backend an option was built for: rows from the previous
|
||||
// backend must not be relabelled while a switch is in flight
|
||||
push(option, option.backendId ?? activeBackendId);
|
||||
}
|
||||
|
||||
for (const backend of backendsStore.enabled) {
|
||||
if (backend.id === activeBackendId) continue;
|
||||
|
||||
for (const option of backendsModelsStore.get(backend.id).models) {
|
||||
push(option, option.backendId ?? backend.id);
|
||||
}
|
||||
}
|
||||
|
||||
return merged;
|
||||
});
|
||||
recentModelIds = $state<string[]>(loadRecentModels());
|
||||
recentModelUsage = $state<Record<string, number>>(loadRecentModelUsage());
|
||||
routerModels = $state<ApiModelDataEntry[]>([]);
|
||||
selectedModelId = $state<string | null>(null);
|
||||
selectedModelName = $state<string | null>(null);
|
||||
selectedModelId = $state<string | null>(storedSelection?.id ?? null);
|
||||
|
||||
selectedModelName = $state<string | null>(storedSelection?.model ?? null);
|
||||
|
||||
updating = $state(false);
|
||||
|
||||
@@ -123,6 +193,9 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
// Without this, ?model=<name> URL handler races an in-progress fetch and sees an empty list.
|
||||
private inflightFetch: Promise<void> | null = null;
|
||||
|
||||
/** A restored selection loses to the active conversation's model, a fresh pick does not. */
|
||||
private selectionFromStorage = storedSelection !== null;
|
||||
|
||||
/**
|
||||
* Model the active conversation view resolves to. Router mode: the user's
|
||||
* selection first, then the conversation's own model. Otherwise the single
|
||||
@@ -130,14 +203,20 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
*/
|
||||
get activeModelId(): string | null {
|
||||
if (!serverStore.isRouterMode) {
|
||||
// external backends expose a selectable list; prefer the user's pick
|
||||
const selected = this.selectedModelId
|
||||
? this.models.find((m) => m.id === this.selectedModelId)
|
||||
: undefined;
|
||||
|
||||
if (selected) return selected.model;
|
||||
|
||||
return this.models.length > 0 ? this.models[0].model : this.singleModelName;
|
||||
}
|
||||
|
||||
if (this.selectedModelId) {
|
||||
const selected = this.models.find((m) => m.id === this.selectedModelId);
|
||||
const picked = this.selectedModelId && !this.selectionFromStorage ? this.selectedModelId : null;
|
||||
const selected = picked ? this.models.find((m) => m.id === picked) : undefined;
|
||||
|
||||
if (selected) return selected.model;
|
||||
}
|
||||
if (selected) return selected.model;
|
||||
|
||||
const conversationModel = getConversationModel(conversationsStore.activeMessages);
|
||||
|
||||
@@ -147,7 +226,11 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
if (model) return model.model;
|
||||
}
|
||||
|
||||
return null;
|
||||
const restored = this.selectedModelId
|
||||
? this.models.find((m) => m.id === this.selectedModelId)
|
||||
: undefined;
|
||||
|
||||
return restored?.model ?? null;
|
||||
}
|
||||
|
||||
get loadedModelIds(): string[] {
|
||||
@@ -200,6 +283,7 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
clearSelection(): void {
|
||||
this.selectedModelId = null;
|
||||
this.selectedModelName = null;
|
||||
this.persistSelection();
|
||||
}
|
||||
|
||||
/** Family keys folded away under one section, for a list that restores them. */
|
||||
@@ -271,6 +355,19 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
await this.selectModelById(availableModels[0].id);
|
||||
}
|
||||
|
||||
/**
|
||||
* Make the backend serving `modelName` active when it is not already.
|
||||
* A conversation keeps the model that generated it, which can belong to a
|
||||
* backend other than the active one.
|
||||
*/
|
||||
async ensureModelBackend(modelName: string): Promise<void> {
|
||||
const option = this.models.find((model) => model.model === modelName);
|
||||
|
||||
if (!option?.backendId || option.backendId === backendsStore.active.id) return;
|
||||
|
||||
await this.selectModelById(option.id);
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch list of models from server and detect server role.
|
||||
* Also fetches modalities for MODEL mode (single model).
|
||||
@@ -278,7 +375,7 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
async fetch(force = false): Promise<void> {
|
||||
if (this.inflightFetch) return this.inflightFetch;
|
||||
|
||||
if (this.models.length > 0 && !force) return;
|
||||
if (this.activeModels.length > 0 && !force) return;
|
||||
|
||||
this.inflightFetch = this.runFetch();
|
||||
try {
|
||||
@@ -302,7 +399,7 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
this.routerModels = response.data;
|
||||
// keep the selector options in sync: a downloaded / deleted model shows
|
||||
// up here too, not only in the router model rows
|
||||
this.models = this.buildModelOptions(response);
|
||||
this.activeModels = this.buildModelOptions(response);
|
||||
this.warmHubDetails();
|
||||
await this.props.fetchModalitiesForLoadedModels();
|
||||
|
||||
@@ -348,7 +445,19 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Load state a model's own backend reports. Only the local server has a status
|
||||
* feed, so an external backend answers from its own model listing.
|
||||
*/
|
||||
getModelStatus(modelId: string): ServerModelStatus | null {
|
||||
const backendId = this.models.find((model) => model.model === modelId)?.backendId;
|
||||
|
||||
if (backendId && backendId !== LOCAL_BACKEND_ID) {
|
||||
const option = backendsModelsStore.get(backendId).models.find((m) => m.model === modelId);
|
||||
|
||||
return (option?.status?.value as ServerModelStatus) ?? null;
|
||||
}
|
||||
|
||||
const model = this.routerModels.find((m) => m.id === modelId);
|
||||
|
||||
return (model?.status?.value as ServerModelStatus) ?? null;
|
||||
@@ -383,13 +492,26 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
async selectModelById(modelId: string, options?: { recordRecent?: boolean }): Promise<void> {
|
||||
if (!modelId || this.updating) return;
|
||||
|
||||
if (this.selectedModelId === modelId) {
|
||||
if (options?.recordRecent) this.recordRecentModel(modelId);
|
||||
const backendId = backendIdFromModelId(modelId) ?? backendsStore.active.id;
|
||||
const rawId = rawModelId(modelId);
|
||||
// the selection is stored backend-qualified, matching the aggregated
|
||||
// model list, no matter which form the caller passed
|
||||
const qualifiedId = qualifyModelId(backendId, rawId);
|
||||
|
||||
// a model from another backend makes that backend active first
|
||||
if (backendId !== backendsStore.active.id) {
|
||||
backendsStore.setActive(backendId);
|
||||
await backendsModelsStore.ensureLoaded(backendId);
|
||||
await this.switchBackend();
|
||||
}
|
||||
|
||||
if (this.selectedModelId === qualifiedId) {
|
||||
if (options?.recordRecent) this.recordRecentModel(qualifiedId);
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
const option = this.models.find((model) => model.id === modelId);
|
||||
const option = this.activeModels.find((model) => model.id === rawId);
|
||||
|
||||
if (!option) throw new Error('Selected model is not available');
|
||||
|
||||
@@ -397,10 +519,12 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
this.error = null;
|
||||
|
||||
try {
|
||||
this.selectedModelId = option.id;
|
||||
this.selectedModelId = qualifiedId;
|
||||
this.selectedModelName = option.model;
|
||||
this.selectionFromStorage = false;
|
||||
this.persistSelection();
|
||||
|
||||
if (options?.recordRecent) this.recordRecentModel(modelId);
|
||||
if (options?.recordRecent) this.recordRecentModel(qualifiedId);
|
||||
} finally {
|
||||
this.updating = false;
|
||||
}
|
||||
@@ -479,6 +603,53 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Activate a backend for the selector tabs. Everything comes from memory:
|
||||
* the model list and router rows are prefetched at startup and the local
|
||||
* server state is kept while an external backend is active. The selection
|
||||
* is left alone, switching tabs must not pick a model.
|
||||
*/
|
||||
async switchBackend(): Promise<void> {
|
||||
this.error = null;
|
||||
|
||||
const backend = backendsStore.active;
|
||||
|
||||
// local props describe the server the UI is served from; keep them while
|
||||
// an external backend is active instead of dropping and refetching
|
||||
if (backend.protocol === 'llama.cpp') {
|
||||
serverStore.restoreLocalState();
|
||||
} else {
|
||||
serverStore.cacheLocalState();
|
||||
serverStore.clear();
|
||||
}
|
||||
|
||||
const cached = backendsModelsStore.get(backend.id);
|
||||
|
||||
if (!cached.loaded) {
|
||||
// nothing prefetched for this backend (startup prefetch failed): load it once
|
||||
await this.fetch(true);
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
if (backend.protocol === 'llama.cpp' && !serverStore.props) {
|
||||
// first visit to the local tab in this session
|
||||
await serverStore.fetch({ background: true });
|
||||
}
|
||||
|
||||
this.activeModels = cached.models;
|
||||
this.loading = false;
|
||||
|
||||
// the local router rows carry the load statuses; the startup prefetch
|
||||
// already returned them, so a tab switch rebuilds the list from memory
|
||||
if (backend.protocol === 'llama.cpp' && this.routerModels.length === 0 && cached.raw) {
|
||||
this.routerModels = cached.raw.data;
|
||||
this.activeModels = this.buildModelOptions(cached.raw);
|
||||
}
|
||||
|
||||
this.warmHubDetails();
|
||||
}
|
||||
|
||||
toDisplayName(id: string): string {
|
||||
const segments = id.split(/\\|\//);
|
||||
const candidate = segments.pop();
|
||||
@@ -519,10 +690,14 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
|
||||
const repos: string[] = [];
|
||||
|
||||
for (const option of this.models) {
|
||||
const repo = option.model.split(':')[0];
|
||||
for (const backend of backendsStore.enabled) {
|
||||
if (!getBackendCapabilities(backend).props) continue;
|
||||
|
||||
if (repo?.includes('/') && !repos.includes(repo)) repos.push(repo);
|
||||
for (const option of backendsModelsStore.get(backend.id).models) {
|
||||
const repo = option.model.split(':')[0];
|
||||
|
||||
if (repo?.includes('/') && !repos.includes(repo)) repos.push(repo);
|
||||
}
|
||||
}
|
||||
|
||||
// a large catalog would fire one request per repo on every load, so this warms
|
||||
@@ -566,9 +741,15 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
|
||||
return {
|
||||
aliases: item.aliases ?? [],
|
||||
// stamp the backend here so the option keeps its origin even
|
||||
// after another backend becomes active
|
||||
backendId: backendsStore.active.id,
|
||||
capabilities: rawCapabilities.filter((value: unknown): value is string =>
|
||||
Boolean(value)
|
||||
),
|
||||
// external backends report the context in their listing, so the
|
||||
// gauge keeps working when the list is rebuilt on reload
|
||||
contextLength: readModelContextLength(item),
|
||||
description: details?.description,
|
||||
details: details?.details,
|
||||
id: item.id,
|
||||
@@ -582,18 +763,20 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
})
|
||||
);
|
||||
}
|
||||
|
||||
/** Fetch models in MODEL mode (single model, standard OpenAI-compatible). */
|
||||
private async fetchModelModeInternal(): Promise<ModelOption[]> {
|
||||
const response = await ModelsService.list();
|
||||
|
||||
return this.buildModelOptions(response);
|
||||
}
|
||||
|
||||
/**
|
||||
* Filter to models visible in the UI (ui !== false).
|
||||
*/
|
||||
private getVisibleModels(): ModelOption[] {
|
||||
return this.models.filter((option) => this.props.getModelProps(option.model)?.ui !== false);
|
||||
return this.activeModels.filter(
|
||||
(option) => this.props.getModelProps(option.model)?.ui !== false
|
||||
);
|
||||
}
|
||||
|
||||
private loadFavoritesFromStorage(): Set<string> {
|
||||
@@ -608,6 +791,23 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
}
|
||||
}
|
||||
|
||||
private persistSelection(): void {
|
||||
if (!browser) return;
|
||||
|
||||
try {
|
||||
if (!this.selectedModelId) {
|
||||
localStorage.removeItem(SELECTED_MODEL_LOCALSTORAGE_KEY);
|
||||
} else {
|
||||
localStorage.setItem(
|
||||
SELECTED_MODEL_LOCALSTORAGE_KEY,
|
||||
JSON.stringify({ id: this.selectedModelId, model: this.selectedModelName })
|
||||
);
|
||||
}
|
||||
} catch {
|
||||
console.warn('[ModelsStore] Failed to persist the model selection');
|
||||
}
|
||||
}
|
||||
|
||||
/** Move a model to the front of the recently used list. */
|
||||
private recordRecentModel(qualifiedId: string): void {
|
||||
this.recentModelIds = [
|
||||
@@ -644,7 +844,7 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
const response = await ModelsService.list();
|
||||
|
||||
this.routerModels = response.data;
|
||||
this.models = this.buildModelOptions(response);
|
||||
this.activeModels = this.buildModelOptions(response);
|
||||
|
||||
await this.props.fetchModalitiesForLoadedModels();
|
||||
|
||||
@@ -654,10 +854,16 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
|
||||
this.selectModelById(visible[0].id);
|
||||
}
|
||||
} else {
|
||||
this.models = await this.fetchModelModeInternal();
|
||||
this.activeModels = await this.fetchModelModeInternal();
|
||||
|
||||
// external backends expose a selectable list; pick a default so the
|
||||
// first send and title generation have a model to target
|
||||
if (!serverStore.capabilities.props && !this.selectedModelName) {
|
||||
await this.ensureFirstModelSelected();
|
||||
}
|
||||
}
|
||||
} catch (error) {
|
||||
this.models = [];
|
||||
this.activeModels = [];
|
||||
this.error = error instanceof Error ? error.message : 'Failed to load models';
|
||||
|
||||
throw error;
|
||||
|
||||
@@ -14,12 +14,17 @@ import { MODEL_PROPS_CACHE } from '$lib/constants';
|
||||
import { FileTypeCategory, ModelModality } from '$lib/enums';
|
||||
import { PropsService } from '$lib/services/props.service';
|
||||
// direct imports between stores, not via the barrel, to avoid circular deps
|
||||
import { backendsStore } from '$lib/stores/backends.svelte';
|
||||
import { serverStore } from '$lib/stores/server.svelte';
|
||||
// deep imports, not the '$lib/utils' barrel: it re-exports modules that reach back
|
||||
// into the stores, and going through it here would read a half-built module
|
||||
import { TTLCache } from '$lib/utils/cache-ttl';
|
||||
import { detectThinkingSupport } from '$lib/utils/chat-template-thinking-detector';
|
||||
import { SvelteSet } from 'svelte/reactivity';
|
||||
import { rawModelId } from '$lib/utils/model-option-id';
|
||||
import { SvelteMap, SvelteSet } from 'svelte/reactivity';
|
||||
|
||||
/** Grace period before /props is retried for a model after a failed fetch. */
|
||||
const PROPS_FETCH_RETRY_COOLDOWN_MS = 30_000;
|
||||
|
||||
/**
|
||||
* The slice of modelsStore the manager reads. Kept narrow on purpose so it
|
||||
@@ -28,7 +33,9 @@ import { SvelteSet } from 'svelte/reactivity';
|
||||
*/
|
||||
export interface ModelPropsHost {
|
||||
/** Model rows the manager mirrors fetched modalities onto. */
|
||||
models: ModelOption[];
|
||||
activeModels: ModelOption[];
|
||||
/** Every enabled backend's models, used to resolve a model's owner. */
|
||||
readonly models: ModelOption[];
|
||||
readonly selectedModelName: string | null;
|
||||
readonly loadedModelIds: string[];
|
||||
isModelLoaded(modelId: string): boolean;
|
||||
@@ -45,6 +52,8 @@ export class ModelPropsManager {
|
||||
maxEntries: MODEL_PROPS_CACHE.MAX_ENTRIES,
|
||||
ttlMs: MODEL_PROPS_CACHE.TTL_MS
|
||||
});
|
||||
/** Last failed /props fetch per model; requests stay on cooldown after it. */
|
||||
private failedPropsAt = new SvelteMap<string, number>();
|
||||
private fetching = new SvelteSet<string>();
|
||||
|
||||
/**
|
||||
@@ -123,7 +132,7 @@ export class ModelPropsManager {
|
||||
try {
|
||||
const results = await Promise.all(propsPromises);
|
||||
|
||||
this.host.models = this.host.models.map((model) => {
|
||||
this.host.activeModels = this.host.activeModels.map((model) => {
|
||||
const modelIndex = loadedModelIds.indexOf(model.model);
|
||||
|
||||
if (modelIndex === -1) return model;
|
||||
@@ -152,14 +161,36 @@ export class ModelPropsManager {
|
||||
* @returns Props data or null if fetch failed or model not loaded
|
||||
*/
|
||||
async fetchModelProps(modelId: string): Promise<ApiLlamaCppServerProps | null> {
|
||||
// /props only exists on llama.cpp servers
|
||||
if (!serverStore.capabilities.props) return null;
|
||||
|
||||
// props not loaded yet (mid backend switch): the server role and model
|
||||
// load state are not trustworthy, so do not guess
|
||||
if (!serverStore.props) return null;
|
||||
|
||||
const cached = this.cache.get(modelId);
|
||||
|
||||
if (cached) return cached;
|
||||
|
||||
// never ask a server about a model another backend serves; a name that
|
||||
// exists on two backends resolves to the active one's entry first
|
||||
const option = this.host.models.find(
|
||||
(m) => m.model === modelId || m.id === modelId || rawModelId(m.id) === modelId
|
||||
);
|
||||
|
||||
if (option?.backendId && option.backendId !== backendsStore.active.id) return null;
|
||||
|
||||
if (serverStore.isRouterMode && !this.host.isModelLoaded(modelId)) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// a failing model would otherwise be refetched on every reactive update
|
||||
const failedAt = this.failedPropsAt.get(modelId);
|
||||
|
||||
if (failedAt !== undefined && Date.now() - failedAt < PROPS_FETCH_RETRY_COOLDOWN_MS) {
|
||||
return null;
|
||||
}
|
||||
|
||||
if (this.fetching.has(modelId)) return null;
|
||||
|
||||
this.fetching.add(modelId);
|
||||
@@ -169,9 +200,11 @@ export class ModelPropsManager {
|
||||
|
||||
this.cache.set(modelId, props);
|
||||
this.cacheVersion++;
|
||||
this.failedPropsAt.delete(modelId);
|
||||
|
||||
return props;
|
||||
} catch (error) {
|
||||
this.failedPropsAt.set(modelId, Date.now());
|
||||
console.warn(`Failed to fetch props for model ${modelId}:`, error);
|
||||
|
||||
return null;
|
||||
@@ -184,7 +217,12 @@ export class ModelPropsManager {
|
||||
const props = this.getModelProps(modelId);
|
||||
const nCtx = props?.default_generation_settings?.n_ctx;
|
||||
|
||||
return typeof nCtx === 'number' ? nCtx : null;
|
||||
if (typeof nCtx === 'number') return nCtx;
|
||||
|
||||
// external backends report the context in their model listing, when they report one
|
||||
const model = this.host.activeModels.find((m) => m.model === modelId || m.id === modelId);
|
||||
|
||||
return model?.contextLength ?? null;
|
||||
}
|
||||
|
||||
getModelModalities(modelId: string): ModelModalities | null {
|
||||
@@ -192,7 +230,7 @@ export class ModelPropsManager {
|
||||
return this.buildModalities(serverStore.props.modalities);
|
||||
}
|
||||
|
||||
const model = this.host.models.find((m) => m.model === modelId || m.id === modelId);
|
||||
const model = this.host.activeModels.find((m) => m.model === modelId || m.id === modelId);
|
||||
|
||||
if (model?.modalities) {
|
||||
return model.modalities;
|
||||
@@ -252,7 +290,7 @@ export class ModelPropsManager {
|
||||
|
||||
if (!props?.modalities) return;
|
||||
|
||||
this.host.models = this.host.models.map((model) =>
|
||||
this.host.activeModels = this.host.activeModels.map((model) =>
|
||||
model.model === modelId
|
||||
? { ...model, modalities: this.buildModalities(props.modalities!) }
|
||||
: model
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
import {
|
||||
CLI_FLAGS,
|
||||
HF_UD_QUANT_PREFIX_REGEX,
|
||||
LOCAL_BACKEND_ID,
|
||||
MODEL_ID,
|
||||
PATH_SEPARATOR,
|
||||
PAUSED_MODEL_DOWNLOADS_LOCALSTORAGE_KEY
|
||||
@@ -17,11 +18,14 @@ import {
|
||||
import { ModelDownloadStopRequest, ServerModelsSseEventType, ServerModelStatus } from '$lib/enums';
|
||||
import { HuggingFaceService } from '$lib/services/huggingface.service';
|
||||
import { ModelsService } from '$lib/services/models.service';
|
||||
import type { ModelPropsManager } from '$lib/stores/models/props.svelte';
|
||||
import { backendsStore } from '$lib/stores/backends.svelte';
|
||||
// direct imports between stores, not via the barrel, to avoid circular deps
|
||||
import { backendsModelsStore } from '$lib/stores/backendsModels.svelte';
|
||||
import type { ModelPropsManager } from '$lib/stores/models/props.svelte';
|
||||
import { serverStore } from '$lib/stores/server.svelte';
|
||||
// explicit type imports: the app.d.ts globals resolve to `any`, so import the real types
|
||||
import type { ApiModelsSseDownloadProgressData, ModelDownloadProgress } from '$lib/types';
|
||||
import { backendIdFromModelId } from '$lib/utils/model-option-id';
|
||||
import { SvelteMap, SvelteSet } from 'svelte/reactivity';
|
||||
import { toast } from 'svelte-sonner';
|
||||
|
||||
@@ -30,13 +34,20 @@ import { toast } from 'svelte-sonner';
|
||||
* cannot reach around the host's full surface; modelsStore implements this
|
||||
* structurally.
|
||||
*/
|
||||
/** How long a remote operation is watched before its last answer is kept. */
|
||||
const REMOTE_STATUS_TIMEOUT_MS = 10 * 60 * 1000;
|
||||
const REMOTE_STATUS_POLL_MS = 2000;
|
||||
|
||||
export interface ModelStatusHost {
|
||||
error: string | null;
|
||||
/** Load state a model's own backend reports, local or external. */
|
||||
getModelStatus(modelId: string): ServerModelStatus | null;
|
||||
readonly props: ModelPropsManager;
|
||||
/** Router model rows the status feed updates. */
|
||||
routerModels: ApiModelDataEntry[];
|
||||
fetchRouterModels(): Promise<void>;
|
||||
isModelLoaded(modelId: string): boolean;
|
||||
switchBackend(): Promise<void>;
|
||||
toDisplayName(id: string): string;
|
||||
}
|
||||
|
||||
@@ -55,6 +66,10 @@ function downloadIdKey(repoWithTag: string): string {
|
||||
}
|
||||
|
||||
export class ModelStatusManager {
|
||||
/** How long a remote operation is watched before its last answer is kept. */
|
||||
REMOTE_STATUS_POLL_MS = 2000;
|
||||
/** How long a remote operation is watched before the last answer is kept. */
|
||||
REMOTE_STATUS_TIMEOUT_MS = 10 * 60 * 1000;
|
||||
/**
|
||||
* Sidecar files pulled by registered models, as `<repo>/<file>` keys.
|
||||
* Sidecars are not separate /v1/models entries - the router pulls them as
|
||||
@@ -96,10 +111,12 @@ export class ModelStatusManager {
|
||||
// /models/sse feed state, the single source of truth for status and load progress
|
||||
private statusAbort: AbortController | null = null;
|
||||
private statusReaderActive = false;
|
||||
|
||||
private statusWaiters = new SvelteMap<
|
||||
string,
|
||||
{ target: ServerModelStatus; resolve: () => void; reject: (e: Error) => void }
|
||||
>();
|
||||
|
||||
/** Tags the user asked to stop (pause or cancel); the download_failed the stop triggers is intentional, not a failure. */
|
||||
private stopRequests = new SvelteMap<string, ModelDownloadStopRequest>();
|
||||
|
||||
@@ -109,6 +126,8 @@ export class ModelStatusManager {
|
||||
* feed's model_remove event.
|
||||
*/
|
||||
async cancelDownload(repoWithTag: string): Promise<boolean> {
|
||||
await this.ensureLocalTarget();
|
||||
|
||||
if (!serverStore.isRouterMode) {
|
||||
toast.error('Model downloads are only available in router mode');
|
||||
|
||||
@@ -154,12 +173,14 @@ export class ModelStatusManager {
|
||||
* waiter is registered here.
|
||||
*/
|
||||
async cancelLoad(modelId: string): Promise<void> {
|
||||
const backendId = this.backendIdFor(modelId);
|
||||
|
||||
if (!serverStore.isRouterMode) return;
|
||||
|
||||
this.subscribe();
|
||||
|
||||
try {
|
||||
await ModelsService.unload(modelId);
|
||||
await ModelsService.unload(modelId, backendId);
|
||||
toast.info(`Load cancelled: ${this.host.toDisplayName(modelId)}`);
|
||||
} catch (error) {
|
||||
toast.error(`Failed to cancel load: ${this.host.toDisplayName(modelId)}`);
|
||||
@@ -190,6 +211,8 @@ export class ModelStatusManager {
|
||||
* (same tag) continues from the partial files the pause kept on disk.
|
||||
*/
|
||||
async downloadModel(repoWithTag: string): Promise<void> {
|
||||
await this.ensureLocalTarget();
|
||||
|
||||
if (!serverStore.isRouterMode) {
|
||||
toast.error('Model downloads are only available in router mode');
|
||||
|
||||
@@ -319,7 +342,10 @@ export class ModelStatusManager {
|
||||
return this.downloadedSidecars.has(`${repoId}/${filePath}`);
|
||||
}
|
||||
|
||||
async load(modelId: string): Promise<void> {
|
||||
async load(modelId: string, extraArgs?: string[]): Promise<void> {
|
||||
const backendId = this.backendIdFor(modelId);
|
||||
const isLocal = backendId === LOCAL_BACKEND_ID;
|
||||
|
||||
if (this.host.isModelLoaded(modelId)) return;
|
||||
|
||||
if (this.loadingStates.get(modelId)) return;
|
||||
@@ -330,13 +356,21 @@ export class ModelStatusManager {
|
||||
// the feed drives completion, so it must be live before the request
|
||||
this.subscribe();
|
||||
|
||||
const reachedLoaded = this.waitForStatus(modelId, ServerModelStatus.LOADED);
|
||||
const reachedLoaded = isLocal
|
||||
? this.waitForStatus(modelId, ServerModelStatus.LOADED)
|
||||
: Promise.resolve();
|
||||
|
||||
reachedLoaded.catch(() => {});
|
||||
|
||||
try {
|
||||
await ModelsService.load(modelId);
|
||||
await reachedLoaded;
|
||||
await ModelsService.load(modelId, extraArgs, backendId);
|
||||
|
||||
if (isLocal) {
|
||||
await reachedLoaded;
|
||||
} else {
|
||||
await this.waitForRemoteStatus(backendId, modelId, ServerModelStatus.LOADED);
|
||||
}
|
||||
|
||||
toast.success(`Model loaded: ${this.host.toDisplayName(modelId)}`);
|
||||
} catch (error) {
|
||||
this.rejectStatus(modelId, error instanceof Error ? error : new Error('load failed'));
|
||||
@@ -356,6 +390,8 @@ export class ModelStatusManager {
|
||||
* reports the stop as download_failed; a 'pause' stop request marks it as such.
|
||||
*/
|
||||
async pauseDownload(repoWithTag: string): Promise<void> {
|
||||
await this.ensureLocalTarget();
|
||||
|
||||
if (!serverStore.isRouterMode) {
|
||||
toast.error('Model downloads are only available in router mode');
|
||||
|
||||
@@ -374,10 +410,6 @@ export class ModelStatusManager {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Open the /models/sse feed and keep it live with auto reconnect.
|
||||
* Idempotent and router mode only.
|
||||
*/
|
||||
subscribe(): void {
|
||||
if (this.statusReaderActive) return;
|
||||
|
||||
@@ -389,6 +421,9 @@ export class ModelStatusManager {
|
||||
}
|
||||
|
||||
async unload(modelId: string): Promise<void> {
|
||||
const backendId = this.backendIdFor(modelId);
|
||||
const isLocal = backendId === LOCAL_BACKEND_ID;
|
||||
|
||||
if (!this.host.isModelLoaded(modelId)) return;
|
||||
|
||||
if (this.loadingStates.get(modelId)) return;
|
||||
@@ -398,13 +433,21 @@ export class ModelStatusManager {
|
||||
|
||||
this.subscribe();
|
||||
|
||||
const reachedUnloaded = this.waitForStatus(modelId, ServerModelStatus.UNLOADED);
|
||||
const reachedUnloaded = isLocal
|
||||
? this.waitForStatus(modelId, ServerModelStatus.UNLOADED)
|
||||
: Promise.resolve();
|
||||
|
||||
reachedUnloaded.catch(() => {});
|
||||
|
||||
try {
|
||||
await ModelsService.unload(modelId);
|
||||
await reachedUnloaded;
|
||||
await ModelsService.unload(modelId, backendId);
|
||||
|
||||
if (isLocal) {
|
||||
await reachedUnloaded;
|
||||
} else {
|
||||
await this.waitForRemoteStatus(backendId, modelId, ServerModelStatus.UNLOADED);
|
||||
}
|
||||
|
||||
toast.info(`Model unloaded: ${this.host.toDisplayName(modelId)}`);
|
||||
} catch (error) {
|
||||
this.rejectStatus(modelId, error instanceof Error ? error : new Error('unload failed'));
|
||||
@@ -578,6 +621,34 @@ export class ModelStatusManager {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Open the /models/sse feed and keep it live with auto reconnect.
|
||||
* Idempotent and router mode only.
|
||||
*/
|
||||
/**
|
||||
* Load, unload and download only exist on the local server. Make it the
|
||||
* target first when the action comes from a row while a remote backend is
|
||||
* the selected one.
|
||||
*/
|
||||
/**
|
||||
* Backend a model belongs to: the one its qualified id names, else whichever
|
||||
* enabled backend lists it. Loading an external llama-server's model must not
|
||||
* become a request to the server this UI is served from.
|
||||
*/
|
||||
private backendIdFor(modelId: string): string {
|
||||
const qualified = backendIdFromModelId(modelId);
|
||||
|
||||
if (qualified) return qualified;
|
||||
|
||||
for (const backend of backendsStore.enabled) {
|
||||
const state = backendsModelsStore.get(backend.id);
|
||||
|
||||
if (state.models.some((option) => option.model === modelId)) return backend.id;
|
||||
}
|
||||
|
||||
return backendsStore.active.id;
|
||||
}
|
||||
|
||||
private deletePausedDownload(repoWithTag: string): boolean {
|
||||
if (!this.pausedDownloads.delete(repoWithTag)) return false;
|
||||
|
||||
@@ -585,6 +656,13 @@ export class ModelStatusManager {
|
||||
|
||||
return true;
|
||||
}
|
||||
private async ensureLocalTarget(): Promise<void> {
|
||||
if (backendsStore.active.id === LOCAL_BACKEND_ID) return;
|
||||
|
||||
backendsStore.setActive(LOCAL_BACKEND_ID);
|
||||
await backendsModelsStore.ensureLoaded(LOCAL_BACKEND_ID);
|
||||
await this.host.switchBackend();
|
||||
}
|
||||
|
||||
private persistPausedDownloads(): void {
|
||||
try {
|
||||
@@ -670,6 +748,42 @@ export class ModelStatusManager {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Wait for an external backend to report a status. Its changes never reach the
|
||||
* local feed, so its own listing is asked again until the status lands or the
|
||||
* deadline passes, which leaves the last answer in place.
|
||||
*/
|
||||
private async waitForRemoteStatus(
|
||||
backendId: string,
|
||||
modelId: string,
|
||||
target: ServerModelStatus
|
||||
): Promise<void> {
|
||||
const deadline = Date.now() + REMOTE_STATUS_TIMEOUT_MS;
|
||||
|
||||
while (Date.now() < deadline) {
|
||||
await backendsModelsStore.refresh(backendId);
|
||||
|
||||
const status = this.host.getModelStatus(modelId);
|
||||
|
||||
if (status === ServerModelStatus.FAILED) {
|
||||
throw new Error(
|
||||
target === ServerModelStatus.LOADED
|
||||
? 'the server failed to load it'
|
||||
: 'the server failed to unload it'
|
||||
);
|
||||
}
|
||||
|
||||
const reached =
|
||||
target === ServerModelStatus.LOADED
|
||||
? status === ServerModelStatus.LOADED || status === ServerModelStatus.SLEEPING
|
||||
: status === ServerModelStatus.UNLOADED || status === null;
|
||||
|
||||
if (reached) return;
|
||||
|
||||
await new Promise((resolve) => setTimeout(resolve, REMOTE_STATUS_POLL_MS));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Register an awaiter that resolves when the feed reports target status.
|
||||
* One operation runs per model at a time, so one awaiter per model is kept.
|
||||
|
||||
@@ -6,9 +6,13 @@
|
||||
* PropsService for the /props fetch.
|
||||
*/
|
||||
|
||||
import { BACKEND_CAPABILITIES, LOCAL_BACKEND_ID } from '$lib/constants';
|
||||
import { ServerRole } from '$lib/enums';
|
||||
import { PropsService } from '$lib/services/props.service';
|
||||
import type { BackendCapabilities } from '$lib/types';
|
||||
import { ApiError } from '$lib/utils';
|
||||
import { getBackend } from '$lib/utils/api-base';
|
||||
import { getBackendCapabilities } from '$lib/utils/backend';
|
||||
|
||||
const LOADING_RETRY_INTERVAL_MS = 1000;
|
||||
|
||||
@@ -18,9 +22,25 @@ class ServerStore {
|
||||
props = $state<ApiLlamaCppServerProps | null>(null);
|
||||
role = $state<ServerRole | null>(null);
|
||||
status = $state<number | null>(null);
|
||||
private fetchBackendId: string | undefined;
|
||||
private fetchPromise: Promise<void> | null = null;
|
||||
/**
|
||||
* Whether a local llama.cpp server answered. Null until the first probe
|
||||
* resolves: the bundled build is served by one, the hosted PWA is not.
|
||||
*/
|
||||
private localAvailable: boolean | null = null;
|
||||
/** Local server state kept alive while an external backend is active. */
|
||||
private localState: { props: ApiLlamaCppServerProps | null; role: ServerRole | null } | null =
|
||||
null;
|
||||
private retryTimer: ReturnType<typeof setTimeout> | null = null;
|
||||
|
||||
/** Features of the active backend. Defaults to full llama.cpp support. */
|
||||
get capabilities(): BackendCapabilities {
|
||||
const backend = getBackend();
|
||||
|
||||
return backend ? getBackendCapabilities(backend) : BACKEND_CAPABILITIES['llama.cpp'];
|
||||
}
|
||||
|
||||
get contextSize(): number | null {
|
||||
const nCtx = this.props?.default_generation_settings?.n_ctx;
|
||||
|
||||
@@ -28,7 +48,12 @@ class ServerStore {
|
||||
}
|
||||
|
||||
get defaultParams(): ApiLlamaCppServerProps['default_generation_settings']['params'] | null {
|
||||
return this.props?.default_generation_settings?.params || null;
|
||||
return this.localProps?.default_generation_settings?.params || null;
|
||||
}
|
||||
|
||||
/** A local llama.cpp server answered, either now or earlier this session. */
|
||||
get hasLocalServer(): boolean {
|
||||
return this.localAvailable === true;
|
||||
}
|
||||
|
||||
get isModelMode(): boolean {
|
||||
@@ -39,10 +64,52 @@ class ServerStore {
|
||||
return this.role === ServerRole.ROUTER;
|
||||
}
|
||||
|
||||
/**
|
||||
* Local server state. The local server is the base of the installation, so
|
||||
* its props and role come from the mirror while another backend is active
|
||||
* instead of following the selected model.
|
||||
*/
|
||||
private get localFacts(): { props: ApiLlamaCppServerProps | null; role: ServerRole | null } {
|
||||
// no resolvable backend (early startup, non-browser call): props belong to
|
||||
// the origin serving this UI, which is the local server
|
||||
const backendId = getBackend()?.id ?? LOCAL_BACKEND_ID;
|
||||
|
||||
if (backendId === LOCAL_BACKEND_ID) return { props: this.props, role: this.role };
|
||||
|
||||
return this.localState ?? { props: null, role: null };
|
||||
}
|
||||
|
||||
/** The local server runs as a router, independent of the active backend. */
|
||||
get localIsRouter(): boolean {
|
||||
return this.localFacts.role === ServerRole.ROUTER;
|
||||
}
|
||||
|
||||
/** Props of the local server, from the active backend or the cache. */
|
||||
get localProps(): ApiLlamaCppServerProps | null {
|
||||
return this.localFacts.props;
|
||||
}
|
||||
|
||||
/** The local server is known to be unreachable: local-only UI stays out. */
|
||||
get localServerMissing(): boolean {
|
||||
return this.localAvailable === false;
|
||||
}
|
||||
|
||||
get uiSettings(): Record<string, string | number | boolean> | undefined {
|
||||
return this.props?.ui_settings ?? this.props?.webui_settings;
|
||||
}
|
||||
|
||||
/**
|
||||
* Keep the local server state before switching to an external backend, so
|
||||
* switching back restores it instead of asking the server again.
|
||||
*/
|
||||
cacheLocalState(): void {
|
||||
// props only exist while a llama.cpp server is active; an external to
|
||||
// external switch must not overwrite the kept local state with blanks
|
||||
if (!this.props) return;
|
||||
|
||||
this.localState = { props: this.props, role: this.role };
|
||||
}
|
||||
|
||||
clear(): void {
|
||||
this.clearRetryTimer();
|
||||
this.props = null;
|
||||
@@ -51,6 +118,7 @@ class ServerStore {
|
||||
this.loading = false;
|
||||
this.role = null;
|
||||
this.fetchPromise = null;
|
||||
this.fetchBackendId = undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -59,10 +127,24 @@ class ServerStore {
|
||||
* splash and the chat screen every retry tick.
|
||||
*/
|
||||
async fetch({ background = false }: { background?: boolean } = {}): Promise<void> {
|
||||
if (this.fetchPromise) return this.fetchPromise;
|
||||
// props and role describe one server. a fetch started for another backend
|
||||
// must not be reused, and its response must not commit once the active
|
||||
// backend has changed while it was in flight
|
||||
const backendId = getBackend()?.id;
|
||||
|
||||
if (this.fetchPromise && this.fetchBackendId === backendId) return this.fetchPromise;
|
||||
|
||||
this.clearRetryTimer();
|
||||
|
||||
// External backends expose no /props endpoint. Keep MODEL-mode defaults so
|
||||
// role detection and generation defaults degrade instead of failing.
|
||||
if (!this.capabilities.props) {
|
||||
this.clear();
|
||||
this.role = ServerRole.MODEL;
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
if (!background) {
|
||||
this.loading = true;
|
||||
}
|
||||
@@ -73,19 +155,33 @@ class ServerStore {
|
||||
this.error = null;
|
||||
}
|
||||
|
||||
const fetchPromise = (async () => {
|
||||
const promise = (async () => {
|
||||
try {
|
||||
const props = await PropsService.fetch();
|
||||
|
||||
// the active backend changed while this request was in flight
|
||||
if (getBackend()?.id !== backendId) return;
|
||||
|
||||
this.props = props;
|
||||
this.error = null;
|
||||
this.status = null;
|
||||
this.detectRole(props);
|
||||
|
||||
if (backendId === LOCAL_BACKEND_ID) {
|
||||
this.localAvailable = true;
|
||||
// mirror the local facts as they arrive: a later clear() must not
|
||||
// be able to wipe what the local server told us
|
||||
this.localState = { props, role: this.role };
|
||||
}
|
||||
} catch (error: unknown) {
|
||||
if (getBackend()?.id !== backendId) return;
|
||||
|
||||
this.error = error instanceof Error ? error.message : String(error);
|
||||
this.status = error instanceof ApiError ? error.status : null;
|
||||
console.error('Error fetching server properties:', error);
|
||||
|
||||
if (backendId === LOCAL_BACKEND_ID) this.localAvailable = false;
|
||||
|
||||
if (this.status === 503) {
|
||||
this.scheduleRetry();
|
||||
}
|
||||
@@ -93,13 +189,56 @@ class ServerStore {
|
||||
if (!background) {
|
||||
this.loading = false;
|
||||
}
|
||||
|
||||
this.fetchPromise = null;
|
||||
}
|
||||
})();
|
||||
|
||||
this.fetchPromise = fetchPromise;
|
||||
await fetchPromise;
|
||||
this.fetchPromise = promise;
|
||||
this.fetchBackendId = backendId;
|
||||
|
||||
// a backend switch clears the in-flight handle; only the fetch that is
|
||||
// still the current one may release it
|
||||
void promise
|
||||
.catch(() => {})
|
||||
.finally(() => {
|
||||
if (this.fetchPromise === promise) {
|
||||
this.fetchPromise = null;
|
||||
this.fetchBackendId = undefined;
|
||||
}
|
||||
});
|
||||
|
||||
await promise;
|
||||
}
|
||||
|
||||
/**
|
||||
* Load the local server state in the background at startup. The local tab
|
||||
* then opens from memory instead of asking for props on the first click.
|
||||
*/
|
||||
async prefetchLocalState(): Promise<void> {
|
||||
if (this.localState) return;
|
||||
|
||||
try {
|
||||
const props = await PropsService.fetch(false, LOCAL_BACKEND_ID);
|
||||
|
||||
this.localState = {
|
||||
props,
|
||||
role: props?.role === ServerRole.ROUTER ? ServerRole.ROUTER : ServerRole.MODEL
|
||||
};
|
||||
this.localAvailable = true;
|
||||
} catch {
|
||||
// no local server in this deployment: mark it missing so the local
|
||||
// backend drops out of the enabled list
|
||||
this.localAvailable = false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Restore the state kept by {@link cacheLocalState}; no request is made. */
|
||||
restoreLocalState(): void {
|
||||
if (!this.localState) return;
|
||||
|
||||
this.props = this.localState.props;
|
||||
this.role = this.localState.role;
|
||||
this.error = null;
|
||||
this.status = null;
|
||||
}
|
||||
|
||||
private clearRetryTimer(): void {
|
||||
|
||||
@@ -28,8 +28,10 @@ import {
|
||||
} from '$lib/enums';
|
||||
import { ToolsService } from '$lib/services/tools.service';
|
||||
// direct imports between stores, not via the barrel, to avoid circular deps
|
||||
import { backendsStore } from '$lib/stores/backends.svelte';
|
||||
import { mcpStore } from '$lib/stores/mcp/index.svelte';
|
||||
import { modelsStore } from '$lib/stores/models/index.svelte';
|
||||
import { serverStore } from '$lib/stores/server.svelte';
|
||||
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
||||
import type { OpenAIToolDefinition, ToolEntry, ToolGroup } from '$lib/types';
|
||||
import { ApiError, buildSandboxToolDefinition } from '$lib/utils';
|
||||
@@ -232,6 +234,15 @@ class ToolsStore {
|
||||
}
|
||||
|
||||
async fetchServerTools(): Promise<void> {
|
||||
// the /tools endpoint only exists on the local llama.cpp server, so the
|
||||
// list follows the built-in backend rather than the selected model
|
||||
if (!serverStore.hasLocalServer || !backendsStore.local.enabled) {
|
||||
this._serverTools = [];
|
||||
this.cwdAwareTools = new SvelteSet();
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
if (this._loading) return;
|
||||
|
||||
this._loading = true;
|
||||
|
||||
@@ -11,7 +11,6 @@ class UiStore {
|
||||
composerFocusRequested = $state(false);
|
||||
|
||||
discoverModelsOpen = $state(false);
|
||||
|
||||
/** Whether the desktop sidebar is expanded (open). */
|
||||
isSidebarExpanded = $state(false);
|
||||
/** Model the manager reveals when it opens, a qualified id or a raw model name. */
|
||||
|
||||
@@ -1,10 +1,11 @@
|
||||
import type { PageLoad } from './$types';
|
||||
import { initStores } from '$lib/stores/init';
|
||||
import { hydrateStores } from '$lib/stores/init';
|
||||
import { validateApiKey } from '$lib/utils';
|
||||
|
||||
export const load: PageLoad = async ({ fetch }) => {
|
||||
// loads run before the root layout script, so the stored API key reaches
|
||||
// the probe only once the settings store has read localStorage
|
||||
await initStores();
|
||||
// the probe only once the settings store has read localStorage; the rest of
|
||||
// the startup belongs to the layout, after this load
|
||||
await hydrateStores();
|
||||
await validateApiKey(fetch);
|
||||
};
|
||||
|
||||
@@ -1,10 +1,11 @@
|
||||
import type { PageLoad } from './$types';
|
||||
import { initStores } from '$lib/stores/init';
|
||||
import { hydrateStores } from '$lib/stores/init';
|
||||
import { validateApiKey } from '$lib/utils';
|
||||
|
||||
export const load: PageLoad = async ({ fetch }) => {
|
||||
// loads run before the root layout script, so the stored API key reaches
|
||||
// the probe only once the settings store has read localStorage
|
||||
await initStores();
|
||||
// the probe only once the settings store has read localStorage; the rest of
|
||||
// the startup belongs to the layout, after this load
|
||||
await hydrateStores();
|
||||
await validateApiKey(fetch);
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user