mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-29 09:27:33 -05:00
ui : report token stats for external backends
Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
This commit is contained in:
Vendored
+2
@@ -12,6 +12,7 @@ import type {
|
||||
ApiChatCompletionStreamChunk,
|
||||
ApiChatCompletionToolCall,
|
||||
ApiChatCompletionToolCallDelta,
|
||||
ApiChatCompletionUsage,
|
||||
ApiChatMessageContentPart,
|
||||
ApiChatMessageData,
|
||||
ApiContextSizeError,
|
||||
@@ -74,6 +75,7 @@ declare global {
|
||||
ApiChatCompletionResponse,
|
||||
ApiChatCompletionStreamChunk,
|
||||
ApiChatCompletionToolCall,
|
||||
ApiChatCompletionUsage,
|
||||
ApiChatCompletionToolCallDelta,
|
||||
ApiChatMessageData,
|
||||
ApiChatMessageContentPart,
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
// while the tab was hidden. covers brief background pauses without thrashing live streams
|
||||
export const STREAM_VISIBILITY_KICK_MS = 3000;
|
||||
|
||||
// minimum gap between synthesized live timing updates for backends that do not
|
||||
// stream their own, keeps the per-chunk state updates cheap
|
||||
export const STREAM_LIVE_TIMINGS_INTERVAL_MS = 500;
|
||||
|
||||
// separator joining a conversation id and its per-model stream identity
|
||||
// suffix (conv::model) used by the server side replay buffer
|
||||
export const CONVERSATION_ID_SEPARATOR = '::';
|
||||
|
||||
@@ -20,6 +20,7 @@ import {
|
||||
SSE_DATA_PREFIX,
|
||||
SSE_DONE_MARKER,
|
||||
SSE_LINE_SEPARATOR,
|
||||
STREAM_LIVE_TIMINGS_INTERVAL_MS,
|
||||
STREAM_QUERY_PARAMS,
|
||||
STREAM_RESUME_LOCALSTORAGE_KEY_PREFIX,
|
||||
STREAM_VISIBILITY_KICK_MS
|
||||
@@ -48,6 +49,7 @@ import { ApiError } from '$lib/utils/api-fetch';
|
||||
import { getAuthHeaders, getJsonHeaders } from '$lib/utils/api-headers';
|
||||
import { formatAttachmentText } from '$lib/utils/formatters';
|
||||
import { streamIdentity } from '$lib/utils/stream-identity';
|
||||
import { buildTimingsFromUsage } from '$lib/utils/timings';
|
||||
|
||||
/**
|
||||
* llama.cpp-only chat request fields. Strict OpenAI-compatible endpoints
|
||||
@@ -528,6 +530,15 @@ export class ChatService {
|
||||
let toolCallIndexOffset = 0;
|
||||
let hasOpenToolCallBatch = false;
|
||||
|
||||
// client side clock for backends that do not stream their own timings
|
||||
const startedAt = Date.now();
|
||||
|
||||
let firstTokenAt: number | null = null;
|
||||
let lastTokenAt: number | null = null;
|
||||
let streamedTokens = 0;
|
||||
let liveTimingsAt = 0;
|
||||
let usage: ApiChatCompletionUsage | undefined;
|
||||
|
||||
const finalizeOpenToolCallBatch = () => {
|
||||
if (!hasOpenToolCallBatch) {
|
||||
return;
|
||||
@@ -676,8 +687,11 @@ export class ChatService {
|
||||
const toolCalls = choice?.delta?.tool_calls;
|
||||
const timings = parsed.timings;
|
||||
const promptProgress = parsed.prompt_progress;
|
||||
const chunkUsage = parsed.usage;
|
||||
const chunkModel = ChatService.extractModelName(parsed);
|
||||
|
||||
if (chunkUsage) usage = chunkUsage;
|
||||
|
||||
if (chunkModel && !modelEmitted) {
|
||||
modelEmitted = true;
|
||||
onModel?.(chunkModel);
|
||||
@@ -716,6 +730,29 @@ export class ChatService {
|
||||
}
|
||||
|
||||
processToolCallDelta(toolCalls);
|
||||
|
||||
if (content || reasoningContent) {
|
||||
firstTokenAt ??= Date.now();
|
||||
lastTokenAt = Date.now();
|
||||
streamedTokens++;
|
||||
|
||||
if (
|
||||
!serverStore.capabilities.props &&
|
||||
Date.now() - liveTimingsAt >= STREAM_LIVE_TIMINGS_INTERVAL_MS
|
||||
) {
|
||||
liveTimingsAt = Date.now();
|
||||
|
||||
const liveTimings = buildTimingsFromUsage(
|
||||
usage,
|
||||
{ firstTokenAt, lastTokenAt, startedAt },
|
||||
streamedTokens
|
||||
);
|
||||
|
||||
if (liveTimings) {
|
||||
ChatService.notifyTimings(liveTimings, undefined, onTimings);
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (e) {
|
||||
console.error('Error parsing JSON chunk:', e);
|
||||
}
|
||||
@@ -788,6 +825,20 @@ export class ChatService {
|
||||
if (streamFinished) {
|
||||
finalizeOpenToolCallBatch();
|
||||
|
||||
// external backends report token counts only in the final usage chunk
|
||||
if (!lastTimings && !serverStore.capabilities.props) {
|
||||
lastTimings =
|
||||
buildTimingsFromUsage(
|
||||
usage,
|
||||
{ firstTokenAt, lastTokenAt, startedAt },
|
||||
streamedTokens
|
||||
) ?? undefined;
|
||||
|
||||
if (lastTimings) {
|
||||
ChatService.notifyTimings(lastTimings, undefined, onTimings);
|
||||
}
|
||||
}
|
||||
|
||||
if (conversationId) {
|
||||
ChatService.clearStreamState(conversationId);
|
||||
}
|
||||
@@ -1262,6 +1313,12 @@ export class ChatService {
|
||||
|
||||
if (timings_per_token !== undefined) requestBody.timings_per_token = timings_per_token;
|
||||
|
||||
// OpenAI-compatible servers report token counts in a final usage chunk, which
|
||||
// the client side timing fallback in handleStreamResponse relies on
|
||||
if (stream && !serverStore.capabilities.props && requestBody.stream_options === undefined) {
|
||||
requestBody.stream_options = { include_usage: true };
|
||||
}
|
||||
|
||||
if (custom) {
|
||||
try {
|
||||
const customParams = typeof custom === 'string' ? JSON.parse(custom) : custom;
|
||||
|
||||
Vendored
+10
@@ -372,6 +372,16 @@ export interface ApiChatCompletionStreamChunk {
|
||||
cache_n?: number;
|
||||
};
|
||||
prompt_progress?: ChatMessagePromptProgress;
|
||||
/** Token counts, sent by OpenAI-compatible servers on the final chunk. */
|
||||
usage?: ApiChatCompletionUsage;
|
||||
}
|
||||
|
||||
export interface ApiChatCompletionUsage {
|
||||
completion_tokens?: number;
|
||||
input_tokens?: number;
|
||||
output_tokens?: number;
|
||||
prompt_tokens?: number;
|
||||
total_tokens?: number;
|
||||
}
|
||||
|
||||
export interface ApiChatCompletionResponse {
|
||||
|
||||
@@ -25,6 +25,7 @@ export type {
|
||||
ApiChatCompletionToolCallDelta,
|
||||
ApiChatCompletionToolCall,
|
||||
ApiChatCompletionStreamChunk,
|
||||
ApiChatCompletionUsage,
|
||||
ApiChatCompletionResponse,
|
||||
ApiSlotData,
|
||||
ApiProcessingState,
|
||||
|
||||
@@ -161,6 +161,8 @@ export { extractSseDataPayload, parseSseJsonStream, splitSseRecords } from './ss
|
||||
// Stream session identity (conversation-id based)
|
||||
export { streamIdentity } from './stream-identity';
|
||||
|
||||
export { buildTimingsFromUsage, usageTokenCounts } from './timings';
|
||||
|
||||
// MCP utilities
|
||||
export {
|
||||
detectMcpTransportFromUrl,
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
/**
|
||||
* Client side timing fallback for backends that do not report their own.
|
||||
*
|
||||
* llama.cpp streams per-token timings; OpenAI and Anthropic compatible servers
|
||||
* do not. Token counts come from the usage block of the final chunk (or the
|
||||
* count of streamed deltas as a fallback), times are measured locally: the wait
|
||||
* for the first token is attributed to prompt processing, the rest to
|
||||
* generation. Wall clock, so network and queueing are part of the numbers.
|
||||
*/
|
||||
|
||||
import type { ApiChatCompletionUsage } from '$lib/types/api';
|
||||
import type { ChatMessageTimings } from '$lib/types/chat';
|
||||
|
||||
export interface StreamClock {
|
||||
startedAt: number;
|
||||
firstTokenAt: number | null;
|
||||
lastTokenAt: number | null;
|
||||
}
|
||||
|
||||
/** Prompt/output token counts, accepting OpenAI and Anthropic usage fields. */
|
||||
export function usageTokenCounts(usage: ApiChatCompletionUsage | undefined): {
|
||||
promptTokens: number;
|
||||
completionTokens: number;
|
||||
} {
|
||||
return {
|
||||
completionTokens: usage?.completion_tokens ?? usage?.output_tokens ?? 0,
|
||||
promptTokens: usage?.prompt_tokens ?? usage?.input_tokens ?? 0
|
||||
};
|
||||
}
|
||||
|
||||
export function buildTimingsFromUsage(
|
||||
usage: ApiChatCompletionUsage | undefined,
|
||||
clock: StreamClock,
|
||||
fallbackTokens = 0
|
||||
): ChatMessageTimings | null {
|
||||
const { completionTokens, promptTokens } = usageTokenCounts(usage);
|
||||
const predictedN = completionTokens || fallbackTokens;
|
||||
|
||||
if (promptTokens === 0 && predictedN === 0) return null;
|
||||
|
||||
const { firstTokenAt, startedAt } = clock;
|
||||
const lastTokenAt = clock.lastTokenAt ?? firstTokenAt;
|
||||
|
||||
return {
|
||||
// clamp so a one-token reply still reports a positive duration
|
||||
predicted_ms: firstTokenAt && lastTokenAt ? Math.max(1, lastTokenAt - firstTokenAt) : undefined,
|
||||
predicted_n: predictedN,
|
||||
prompt_ms: firstTokenAt ? Math.max(1, firstTokenAt - startedAt) : undefined,
|
||||
prompt_n: promptTokens
|
||||
};
|
||||
}
|
||||
Reference in New Issue
Block a user