From 74cbaa2ce1f49c7f6cc2033ffd14be8a2fe0e598 Mon Sep 17 00:00:00 2001 From: Aleksander Grygier Date: Wed, 16 Sep 2026 18:40:47 +0200 Subject: [PATCH] ui : only send the conversation id header to llama.cpp Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash --- tools/ui/src/lib/services/chat.service.ts | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/tools/ui/src/lib/services/chat.service.ts b/tools/ui/src/lib/services/chat.service.ts index 8ff2558763..030e48f19b 100644 --- a/tools/ui/src/lib/services/chat.service.ts +++ b/tools/ui/src/lib/services/chat.service.ts @@ -1330,8 +1330,9 @@ export class ChatService { // tag streaming requests with the conversation id, this single header is the opt in for the // server side replay buffer and powers discoverActiveStream on tab reopen. with an explicit - // model the ::model suffix keeps the per model session distinct - if (stream && conversationId) { + // model the ::model suffix keeps the per model session distinct. external providers do not + // know the header and their CORS preflight rejects it, so only llama.cpp gets it + if (stream && conversationId && serverStore.capabilities.resumableStreams) { headers[HEADERS.X_CONVERSATION_ID_HEADER] = streamIdentity(conversationId, options.model); // persist the pending stream before the fetch: a reload during the model load or // the prompt processing must still find its way back to the session once it exists