mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-29 09:27:33 -05:00
fix header for pcm
This commit is contained in:
@@ -806,7 +806,7 @@ Returns raw audio bytes (`audio/wav` by default) rather than JSON. For more info
|
||||
|
||||
`n_predict`: Max number of audio frames to generate. Defaults to `512`; generation normally stops earlier once the model emits an end-of-speech token.
|
||||
|
||||
`response_format`: `wav` (default) or `pcm` (raw `float32` samples, no header).
|
||||
`response_format`: `wav` (default) or `pcm` (raw `float32` little-endian mono samples, no header; the response `Content-Type` carries the sample rate, e.g. `audio/pcm;rate=24000;encoding=float;bits=32`).
|
||||
|
||||
`stream`: If `true`, the response is streamed as audio becomes available instead of waiting for the full generation to finish. WAV streaming writes an RFC-noncompliant header with an unknown (`0xFFFFFFFF`) size field, since the final length isn't known up front; most players and decoders (ffmpeg, VLC, ...) handle this by reading until EOF.
|
||||
|
||||
|
||||
@@ -5216,6 +5216,8 @@ void server_routes::init_routes() {
|
||||
return res;
|
||||
}
|
||||
|
||||
const auto info = mtmd_gen_audio_get_info(ctx_server.mctx);
|
||||
|
||||
const json body = json::parse(req.body);
|
||||
|
||||
std::string prompt = json_value(body, "input", json_value(body, "prompt", std::string()));
|
||||
@@ -5283,7 +5285,10 @@ void server_routes::init_routes() {
|
||||
task.id = rd.get_new_id();
|
||||
rd.post_task(std::move(task));
|
||||
|
||||
const std::string content_type = response_format == "pcm" ? "audio/L16" : "audio/wav";
|
||||
// raw float32 LE mono samples; audio/L16 would imply 16-bit big-endian (RFC 2586)
|
||||
const std::string content_type = response_format == "pcm"
|
||||
? "audio/pcm;rate=" + std::to_string(info.sample_rate) + ";encoding=float;bits=32"
|
||||
: "audio/wav";
|
||||
|
||||
if (!stream) {
|
||||
auto result = rd.next(req.should_stop);
|
||||
|
||||
Reference in New Issue
Block a user