diff --git a/tools/ui/src/app.d.ts b/tools/ui/src/app.d.ts index 5c039063ad..c3d210c9e6 100644 --- a/tools/ui/src/app.d.ts +++ b/tools/ui/src/app.d.ts @@ -12,6 +12,7 @@ import type { ApiChatCompletionStreamChunk, ApiChatCompletionToolCall, ApiChatCompletionToolCallDelta, + ApiChatCompletionUsage, ApiChatMessageContentPart, ApiChatMessageData, ApiContextSizeError, @@ -74,6 +75,7 @@ declare global { ApiChatCompletionResponse, ApiChatCompletionStreamChunk, ApiChatCompletionToolCall, + ApiChatCompletionUsage, ApiChatCompletionToolCallDelta, ApiChatMessageData, ApiChatMessageContentPart, diff --git a/tools/ui/src/lib/constants/stream.constants.ts b/tools/ui/src/lib/constants/stream.constants.ts index 64f67243c2..d68ffbe4bf 100644 --- a/tools/ui/src/lib/constants/stream.constants.ts +++ b/tools/ui/src/lib/constants/stream.constants.ts @@ -2,6 +2,10 @@ // while the tab was hidden. covers brief background pauses without thrashing live streams export const STREAM_VISIBILITY_KICK_MS = 3000; +// minimum gap between synthesized live timing updates for backends that do not +// stream their own, keeps the per-chunk state updates cheap +export const STREAM_LIVE_TIMINGS_INTERVAL_MS = 500; + // separator joining a conversation id and its per-model stream identity // suffix (conv::model) used by the server side replay buffer export const CONVERSATION_ID_SEPARATOR = '::'; diff --git a/tools/ui/src/lib/services/chat.service.ts b/tools/ui/src/lib/services/chat.service.ts index 2801a902d1..a96b3568fe 100644 --- a/tools/ui/src/lib/services/chat.service.ts +++ b/tools/ui/src/lib/services/chat.service.ts @@ -20,6 +20,7 @@ import { SSE_DATA_PREFIX, SSE_DONE_MARKER, SSE_LINE_SEPARATOR, + STREAM_LIVE_TIMINGS_INTERVAL_MS, STREAM_QUERY_PARAMS, STREAM_RESUME_LOCALSTORAGE_KEY_PREFIX, STREAM_VISIBILITY_KICK_MS @@ -48,6 +49,7 @@ import { ApiError } from '$lib/utils/api-fetch'; import { getAuthHeaders, getJsonHeaders } from '$lib/utils/api-headers'; import { formatAttachmentText } from '$lib/utils/formatters'; import { streamIdentity } from '$lib/utils/stream-identity'; +import { buildTimingsFromUsage } from '$lib/utils/timings'; /** * llama.cpp-only chat request fields. Strict OpenAI-compatible endpoints @@ -528,6 +530,15 @@ export class ChatService { let toolCallIndexOffset = 0; let hasOpenToolCallBatch = false; + // client side clock for backends that do not stream their own timings + const startedAt = Date.now(); + + let firstTokenAt: number | null = null; + let lastTokenAt: number | null = null; + let streamedTokens = 0; + let liveTimingsAt = 0; + let usage: ApiChatCompletionUsage | undefined; + const finalizeOpenToolCallBatch = () => { if (!hasOpenToolCallBatch) { return; @@ -676,8 +687,11 @@ export class ChatService { const toolCalls = choice?.delta?.tool_calls; const timings = parsed.timings; const promptProgress = parsed.prompt_progress; + const chunkUsage = parsed.usage; const chunkModel = ChatService.extractModelName(parsed); + if (chunkUsage) usage = chunkUsage; + if (chunkModel && !modelEmitted) { modelEmitted = true; onModel?.(chunkModel); @@ -716,6 +730,29 @@ export class ChatService { } processToolCallDelta(toolCalls); + + if (content || reasoningContent) { + firstTokenAt ??= Date.now(); + lastTokenAt = Date.now(); + streamedTokens++; + + if ( + !serverStore.capabilities.props && + Date.now() - liveTimingsAt >= STREAM_LIVE_TIMINGS_INTERVAL_MS + ) { + liveTimingsAt = Date.now(); + + const liveTimings = buildTimingsFromUsage( + usage, + { firstTokenAt, lastTokenAt, startedAt }, + streamedTokens + ); + + if (liveTimings) { + ChatService.notifyTimings(liveTimings, undefined, onTimings); + } + } + } } catch (e) { console.error('Error parsing JSON chunk:', e); } @@ -788,6 +825,20 @@ export class ChatService { if (streamFinished) { finalizeOpenToolCallBatch(); + // external backends report token counts only in the final usage chunk + if (!lastTimings && !serverStore.capabilities.props) { + lastTimings = + buildTimingsFromUsage( + usage, + { firstTokenAt, lastTokenAt, startedAt }, + streamedTokens + ) ?? undefined; + + if (lastTimings) { + ChatService.notifyTimings(lastTimings, undefined, onTimings); + } + } + if (conversationId) { ChatService.clearStreamState(conversationId); } @@ -1262,6 +1313,12 @@ export class ChatService { if (timings_per_token !== undefined) requestBody.timings_per_token = timings_per_token; + // OpenAI-compatible servers report token counts in a final usage chunk, which + // the client side timing fallback in handleStreamResponse relies on + if (stream && !serverStore.capabilities.props && requestBody.stream_options === undefined) { + requestBody.stream_options = { include_usage: true }; + } + if (custom) { try { const customParams = typeof custom === 'string' ? JSON.parse(custom) : custom; diff --git a/tools/ui/src/lib/types/api.d.ts b/tools/ui/src/lib/types/api.d.ts index 9621249492..d1fc52ad20 100644 --- a/tools/ui/src/lib/types/api.d.ts +++ b/tools/ui/src/lib/types/api.d.ts @@ -372,6 +372,16 @@ export interface ApiChatCompletionStreamChunk { cache_n?: number; }; prompt_progress?: ChatMessagePromptProgress; + /** Token counts, sent by OpenAI-compatible servers on the final chunk. */ + usage?: ApiChatCompletionUsage; +} + +export interface ApiChatCompletionUsage { + completion_tokens?: number; + input_tokens?: number; + output_tokens?: number; + prompt_tokens?: number; + total_tokens?: number; } export interface ApiChatCompletionResponse { diff --git a/tools/ui/src/lib/types/index.ts b/tools/ui/src/lib/types/index.ts index 22ac655299..56758d957c 100644 --- a/tools/ui/src/lib/types/index.ts +++ b/tools/ui/src/lib/types/index.ts @@ -25,6 +25,7 @@ export type { ApiChatCompletionToolCallDelta, ApiChatCompletionToolCall, ApiChatCompletionStreamChunk, + ApiChatCompletionUsage, ApiChatCompletionResponse, ApiSlotData, ApiProcessingState, diff --git a/tools/ui/src/lib/utils/index.ts b/tools/ui/src/lib/utils/index.ts index 6c41198b45..31478440c3 100644 --- a/tools/ui/src/lib/utils/index.ts +++ b/tools/ui/src/lib/utils/index.ts @@ -161,6 +161,8 @@ export { extractSseDataPayload, parseSseJsonStream, splitSseRecords } from './ss // Stream session identity (conversation-id based) export { streamIdentity } from './stream-identity'; +export { buildTimingsFromUsage, usageTokenCounts } from './timings'; + // MCP utilities export { detectMcpTransportFromUrl, diff --git a/tools/ui/src/lib/utils/timings.ts b/tools/ui/src/lib/utils/timings.ts new file mode 100644 index 0000000000..bd0179f8b7 --- /dev/null +++ b/tools/ui/src/lib/utils/timings.ts @@ -0,0 +1,51 @@ +/** + * Client side timing fallback for backends that do not report their own. + * + * llama.cpp streams per-token timings; OpenAI and Anthropic compatible servers + * do not. Token counts come from the usage block of the final chunk (or the + * count of streamed deltas as a fallback), times are measured locally: the wait + * for the first token is attributed to prompt processing, the rest to + * generation. Wall clock, so network and queueing are part of the numbers. + */ + +import type { ApiChatCompletionUsage } from '$lib/types/api'; +import type { ChatMessageTimings } from '$lib/types/chat'; + +export interface StreamClock { + startedAt: number; + firstTokenAt: number | null; + lastTokenAt: number | null; +} + +/** Prompt/output token counts, accepting OpenAI and Anthropic usage fields. */ +export function usageTokenCounts(usage: ApiChatCompletionUsage | undefined): { + promptTokens: number; + completionTokens: number; +} { + return { + completionTokens: usage?.completion_tokens ?? usage?.output_tokens ?? 0, + promptTokens: usage?.prompt_tokens ?? usage?.input_tokens ?? 0 + }; +} + +export function buildTimingsFromUsage( + usage: ApiChatCompletionUsage | undefined, + clock: StreamClock, + fallbackTokens = 0 +): ChatMessageTimings | null { + const { completionTokens, promptTokens } = usageTokenCounts(usage); + const predictedN = completionTokens || fallbackTokens; + + if (promptTokens === 0 && predictedN === 0) return null; + + const { firstTokenAt, startedAt } = clock; + const lastTokenAt = clock.lastTokenAt ?? firstTokenAt; + + return { + // clamp so a one-token reply still reports a positive duration + predicted_ms: firstTokenAt && lastTokenAt ? Math.max(1, lastTokenAt - firstTokenAt) : undefined, + predicted_n: predictedN, + prompt_ms: firstTokenAt ? Math.max(1, firstTokenAt - startedAt) : undefined, + prompt_n: promptTokens + }; +}