ui : report token stats for external backends

Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
This commit is contained in:
Aleksander Grygier
2026-09-28 12:10:10 +02:00
parent d145c15199
commit 5dd920ebe0
7 changed files with 127 additions and 0 deletions
+2
View File
@@ -12,6 +12,7 @@ import type {
ApiChatCompletionStreamChunk,
ApiChatCompletionToolCall,
ApiChatCompletionToolCallDelta,
ApiChatCompletionUsage,
ApiChatMessageContentPart,
ApiChatMessageData,
ApiContextSizeError,
@@ -74,6 +75,7 @@ declare global {
ApiChatCompletionResponse,
ApiChatCompletionStreamChunk,
ApiChatCompletionToolCall,
ApiChatCompletionUsage,
ApiChatCompletionToolCallDelta,
ApiChatMessageData,
ApiChatMessageContentPart,
@@ -2,6 +2,10 @@
// while the tab was hidden. covers brief background pauses without thrashing live streams
export const STREAM_VISIBILITY_KICK_MS = 3000;
// minimum gap between synthesized live timing updates for backends that do not
// stream their own, keeps the per-chunk state updates cheap
export const STREAM_LIVE_TIMINGS_INTERVAL_MS = 500;
// separator joining a conversation id and its per-model stream identity
// suffix (conv::model) used by the server side replay buffer
export const CONVERSATION_ID_SEPARATOR = '::';
+57
View File
@@ -20,6 +20,7 @@ import {
SSE_DATA_PREFIX,
SSE_DONE_MARKER,
SSE_LINE_SEPARATOR,
STREAM_LIVE_TIMINGS_INTERVAL_MS,
STREAM_QUERY_PARAMS,
STREAM_RESUME_LOCALSTORAGE_KEY_PREFIX,
STREAM_VISIBILITY_KICK_MS
@@ -48,6 +49,7 @@ import { ApiError } from '$lib/utils/api-fetch';
import { getAuthHeaders, getJsonHeaders } from '$lib/utils/api-headers';
import { formatAttachmentText } from '$lib/utils/formatters';
import { streamIdentity } from '$lib/utils/stream-identity';
import { buildTimingsFromUsage } from '$lib/utils/timings';
/**
* llama.cpp-only chat request fields. Strict OpenAI-compatible endpoints
@@ -528,6 +530,15 @@ export class ChatService {
let toolCallIndexOffset = 0;
let hasOpenToolCallBatch = false;
// client side clock for backends that do not stream their own timings
const startedAt = Date.now();
let firstTokenAt: number | null = null;
let lastTokenAt: number | null = null;
let streamedTokens = 0;
let liveTimingsAt = 0;
let usage: ApiChatCompletionUsage | undefined;
const finalizeOpenToolCallBatch = () => {
if (!hasOpenToolCallBatch) {
return;
@@ -676,8 +687,11 @@ export class ChatService {
const toolCalls = choice?.delta?.tool_calls;
const timings = parsed.timings;
const promptProgress = parsed.prompt_progress;
const chunkUsage = parsed.usage;
const chunkModel = ChatService.extractModelName(parsed);
if (chunkUsage) usage = chunkUsage;
if (chunkModel && !modelEmitted) {
modelEmitted = true;
onModel?.(chunkModel);
@@ -716,6 +730,29 @@ export class ChatService {
}
processToolCallDelta(toolCalls);
if (content || reasoningContent) {
firstTokenAt ??= Date.now();
lastTokenAt = Date.now();
streamedTokens++;
if (
!serverStore.capabilities.props &&
Date.now() - liveTimingsAt >= STREAM_LIVE_TIMINGS_INTERVAL_MS
) {
liveTimingsAt = Date.now();
const liveTimings = buildTimingsFromUsage(
usage,
{ firstTokenAt, lastTokenAt, startedAt },
streamedTokens
);
if (liveTimings) {
ChatService.notifyTimings(liveTimings, undefined, onTimings);
}
}
}
} catch (e) {
console.error('Error parsing JSON chunk:', e);
}
@@ -788,6 +825,20 @@ export class ChatService {
if (streamFinished) {
finalizeOpenToolCallBatch();
// external backends report token counts only in the final usage chunk
if (!lastTimings && !serverStore.capabilities.props) {
lastTimings =
buildTimingsFromUsage(
usage,
{ firstTokenAt, lastTokenAt, startedAt },
streamedTokens
) ?? undefined;
if (lastTimings) {
ChatService.notifyTimings(lastTimings, undefined, onTimings);
}
}
if (conversationId) {
ChatService.clearStreamState(conversationId);
}
@@ -1262,6 +1313,12 @@ export class ChatService {
if (timings_per_token !== undefined) requestBody.timings_per_token = timings_per_token;
// OpenAI-compatible servers report token counts in a final usage chunk, which
// the client side timing fallback in handleStreamResponse relies on
if (stream && !serverStore.capabilities.props && requestBody.stream_options === undefined) {
requestBody.stream_options = { include_usage: true };
}
if (custom) {
try {
const customParams = typeof custom === 'string' ? JSON.parse(custom) : custom;
+10
View File
@@ -372,6 +372,16 @@ export interface ApiChatCompletionStreamChunk {
cache_n?: number;
};
prompt_progress?: ChatMessagePromptProgress;
/** Token counts, sent by OpenAI-compatible servers on the final chunk. */
usage?: ApiChatCompletionUsage;
}
export interface ApiChatCompletionUsage {
completion_tokens?: number;
input_tokens?: number;
output_tokens?: number;
prompt_tokens?: number;
total_tokens?: number;
}
export interface ApiChatCompletionResponse {
+1
View File
@@ -25,6 +25,7 @@ export type {
ApiChatCompletionToolCallDelta,
ApiChatCompletionToolCall,
ApiChatCompletionStreamChunk,
ApiChatCompletionUsage,
ApiChatCompletionResponse,
ApiSlotData,
ApiProcessingState,
+2
View File
@@ -161,6 +161,8 @@ export { extractSseDataPayload, parseSseJsonStream, splitSseRecords } from './ss
// Stream session identity (conversation-id based)
export { streamIdentity } from './stream-identity';
export { buildTimingsFromUsage, usageTokenCounts } from './timings';
// MCP utilities
export {
detectMcpTransportFromUrl,
+51
View File
@@ -0,0 +1,51 @@
/**
* Client side timing fallback for backends that do not report their own.
*
* llama.cpp streams per-token timings; OpenAI and Anthropic compatible servers
* do not. Token counts come from the usage block of the final chunk (or the
* count of streamed deltas as a fallback), times are measured locally: the wait
* for the first token is attributed to prompt processing, the rest to
* generation. Wall clock, so network and queueing are part of the numbers.
*/
import type { ApiChatCompletionUsage } from '$lib/types/api';
import type { ChatMessageTimings } from '$lib/types/chat';
export interface StreamClock {
startedAt: number;
firstTokenAt: number | null;
lastTokenAt: number | null;
}
/** Prompt/output token counts, accepting OpenAI and Anthropic usage fields. */
export function usageTokenCounts(usage: ApiChatCompletionUsage | undefined): {
promptTokens: number;
completionTokens: number;
} {
return {
completionTokens: usage?.completion_tokens ?? usage?.output_tokens ?? 0,
promptTokens: usage?.prompt_tokens ?? usage?.input_tokens ?? 0
};
}
export function buildTimingsFromUsage(
usage: ApiChatCompletionUsage | undefined,
clock: StreamClock,
fallbackTokens = 0
): ChatMessageTimings | null {
const { completionTokens, promptTokens } = usageTokenCounts(usage);
const predictedN = completionTokens || fallbackTokens;
if (promptTokens === 0 && predictedN === 0) return null;
const { firstTokenAt, startedAt } = clock;
const lastTokenAt = clock.lastTokenAt ?? firstTokenAt;
return {
// clamp so a one-token reply still reports a positive duration
predicted_ms: firstTokenAt && lastTokenAt ? Math.max(1, lastTokenAt - firstTokenAt) : undefined,
predicted_n: predictedN,
prompt_ms: firstTokenAt ? Math.max(1, firstTokenAt - startedAt) : undefined,
prompt_n: promptTokens
};
}