mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-03 03:17:32 -05:00
onyx: bring the chat parser onto the onyx branch
common/chat.cpp on this branch has no Onyx handling, so a converted model
serves malformed chat: the assistant preamble leaks into content
("to=self<|message|>...") and tool calls fail with
HTTP 500 "The model produced output that does not match the expected
peg-native format"
common_chat_params_init_onyx exists on onyx-fair-patch, added there by
8bb73dd3d. It was never on this branch, so this is not a regression --
the two lines developed independently.
The code here is taken verbatim from that commit. It is the clean side of
`git merge origin/onyx-fair-patch`: chat.cpp is one of the files that
merges without conflict. The full merge is not viable -- it produces 13
conflicts, including add/add on conversion/onyx.py and src/models/onyx.cpp
where the q_norm-folding and metadata-scale approaches contradict each
other, and #4/#7 are stacked on this branch's side of that.
Verified on this branch: builds with 0 errors, converts an Onyx checkpoint,
and serving it gives "4" for "What is 2+2?" plus a correct
get_weather {"city":"Paris"} tool call, where the unported branch gives the
two failures above.
No converter or runtime changes are included, so this should not interact
with the q_norm work.
Co-authored-by: Beto de Paola <betodepaola@meta.com>
This commit is contained in:
co-authored by
Beto de Paola
parent
7c5abdae0f
commit
f4370c741b
+206
@@ -2895,6 +2895,204 @@ static common_chat_params common_chat_params_init_minicpm5(const common_chat_tem
|
||||
return data;
|
||||
}
|
||||
|
||||
// An assistant turn is rendered as one or more messages, each
|
||||
// "<|start|>assistant to=<recipient><|message|>{content}{END}" where END is
|
||||
// <|eom|> (more messages follow) or <|eot|> (end of turn):
|
||||
// - chain-of-thought: to=self, terminated by <|eom|>
|
||||
// - final answer: to=user, terminated by <|eot|>
|
||||
// The generation prompt is just "<|start|>assistant"; the model emits its own
|
||||
// " to=...<|message|>".
|
||||
static common_chat_params common_chat_params_init_onyx(const common_chat_template & tmpl,
|
||||
const autoparser::generation_params & inputs) {
|
||||
common_chat_params data;
|
||||
|
||||
data.prompt = common_chat_template_direct_apply_impl(tmpl, inputs);
|
||||
// generation_prompt is the exact assistant-turn prefix the server strips when recovering
|
||||
// the model's output from the full decoded text. Hardcode it to "<|start|>assistant" (the
|
||||
// model emits its own " to=<recipient><|message|>" after it). Re-rendering the template
|
||||
// with add_generation_prompt would also inject a default system message, which must not
|
||||
// become part of this stripped prefix.
|
||||
data.generation_prompt = "<|start|>assistant";
|
||||
data.format = COMMON_CHAT_FORMAT_PEG_NATIVE;
|
||||
data.supports_thinking = true;
|
||||
|
||||
data.preserved_tokens = {
|
||||
"<|start|>", "<|message|>", "<|eom|>", "<|eot|>",
|
||||
// ATEM tool-call markup emitted on " to=<tool>" turns.
|
||||
"<atem:function_calls>", "<atem:invoke", "<atem:parameter", "</atem:parameter>",
|
||||
"</atem:invoke>", "</atem:function_calls>",
|
||||
};
|
||||
|
||||
data.message_delimiters = {
|
||||
{ COMMON_CHAT_ROLE_ASSISTANT, "<|start|>assistant" },
|
||||
{ COMMON_CHAT_ROLE_USER, "<|start|>user" },
|
||||
{ COMMON_CHAT_ROLE_SYSTEM, "<|start|>system" },
|
||||
{ COMMON_CHAT_ROLE_TOOL, "<|start|>tool" },
|
||||
};
|
||||
|
||||
if (inputs.has_continuation()) {
|
||||
const auto & msg = inputs.continue_msg;
|
||||
|
||||
data.generation_prompt = "<|start|>assistant to=self<|message|>" + msg.reasoning_content;
|
||||
if (inputs.continue_final_message == COMMON_CHAT_CONTINUATION_CONTENT) {
|
||||
data.generation_prompt += "<|eom|><|start|>assistant to=user<|message|>" + msg.render_content();
|
||||
}
|
||||
|
||||
data.prompt += data.generation_prompt;
|
||||
}
|
||||
|
||||
auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
||||
|
||||
auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();
|
||||
// Constrained grammar whenever tools are offered. For tool_choice=REQUIRED the grammar is
|
||||
// non-lazy (forces a valid ATEM tool call from the first token). For the auto path it is lazy
|
||||
// and only activates once the model commits to a tool recipient (" to=<tool><|message|>"), so
|
||||
// the free-text reasoning (to=self) and final (to=user) turns generate unconstrained. See the
|
||||
// grammar_triggers below.
|
||||
auto include_grammar = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto start = p.rule("start", p.literal("<|start|>assistant"));
|
||||
|
||||
// No reasoning to split out and no tools: emit the raw stream as content.
|
||||
// Tool parsing does not depend on reasoning_format. include_grammar (above) forces a tool
|
||||
// call whenever tools are offered, so the parser runs the tool rules even under NONE.
|
||||
// reasoning_format only decides whether the to=self block is tagged reasoning or content.
|
||||
if (!extract_reasoning && !include_grammar) {
|
||||
return start + p.content(p.rest());
|
||||
}
|
||||
|
||||
// chain-of-thought message: " to=self<|message|>{body}<|eom|>". The literals stay
|
||||
// structural so a partial " to=" prefix does not emit a content delta before it resolves
|
||||
// to a tool recipient. Under NONE the body is tagged content instead of reasoning.
|
||||
if (extract_reasoning) {
|
||||
p.rule("analysis", p.literal(" to=self<|message|>") + p.reasoning(p.until("<|eom|>")) + p.literal("<|eom|>"));
|
||||
} else {
|
||||
p.rule("analysis", p.literal(" to=self<|message|>") + p.content(p.until("<|eom|>")) + p.literal("<|eom|>"));
|
||||
}
|
||||
auto analysis = p.ref("analysis");
|
||||
|
||||
// final answer: "[ to=user]<|message|>{content}<|eot|>". The recipient is kept fixed
|
||||
// (not a wildcard) so a partial chain-of-thought (" to=self...") never matches this
|
||||
// branch and leaks into content before its <|eom|> terminator arrives.
|
||||
auto recipient = p.optional(p.literal(" to=user"));
|
||||
auto final_msg = p.rule("final", recipient + p.literal("<|message|>") + p.content(p.until("<|eot|>")));
|
||||
|
||||
// Tool-call turn: " to=<tool><|message|><atem:function_calls>...</atem:function_calls>{term}".
|
||||
// Ported from the ATEM tool-call parser. Only added when tools are offered; otherwise the
|
||||
// grammar and top-level rule are exactly the reasoning/final structure above.
|
||||
//
|
||||
// These tool-call rules assume grammar-constrained input, which is what the
|
||||
// server produces during generation: canonical whitespace and in-schema params,
|
||||
// with no inline preamble because onyx puts preamble in the to=self message.
|
||||
// Parsing unconstrained or historical assistant text is strict on purpose and
|
||||
// will reject the whole tool call rather than guess.
|
||||
if (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE) {
|
||||
// A tool-call message closes with "</atem:function_calls>", terminated by <|eom|>
|
||||
// (more calls this turn) or <|eot|> (end of turn). We do NOT consume the trailing
|
||||
// <|eot|> -- it fires as the server's end-of-turn EOG stop; the <|eom|> between
|
||||
// parallel calls IS consumed (see tool_calls below).
|
||||
|
||||
// Schema-declared string args are captured verbatim up to the closing tag; anything
|
||||
// else is parsed as JSON (via the per-arg schema rule) before its closing tag.
|
||||
auto string_value = p.ac(
|
||||
p.tool_arg_string_value(p.until("</atem:parameter>")) + p.tool_arg_close(p.literal("</atem:parameter>")),
|
||||
"</atem:parameter>");
|
||||
|
||||
auto tool_choice = p.choice();
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
const std::string name = function.at("name");
|
||||
auto params = function.contains("parameters") ? function.at("parameters") : json::object();
|
||||
|
||||
auto args = p.eps();
|
||||
if (params.contains("properties") && params.at("properties").is_object() && !params.at("properties").empty()) {
|
||||
auto schema_info = common_schema_info();
|
||||
schema_info.resolve_refs(params);
|
||||
|
||||
auto arg_choice = p.choice();
|
||||
for (const auto & [prop_name, prop_schema] : params.at("properties").items()) {
|
||||
auto value_parser = p.eps();
|
||||
if (schema_info.resolves_to_string(prop_schema)) {
|
||||
value_parser = string_value;
|
||||
} else {
|
||||
value_parser = p.tool_arg_json_value(
|
||||
p.schema(p.json(), "tool-" + name + "-arg-" + prop_name + "-schema", prop_schema, false))
|
||||
+ p.tool_arg_close(p.literal("</atem:parameter>"));
|
||||
}
|
||||
|
||||
auto arg_rule = p.tool_arg(
|
||||
p.tool_arg_open(p.literal("<atem:parameter name=\"") + p.tool_arg_name(p.literal(prop_name)) + p.literal("\">")) +
|
||||
value_parser);
|
||||
|
||||
arg_choice |= arg_rule;
|
||||
}
|
||||
args = p.zero_or_more(arg_choice + p.space());
|
||||
}
|
||||
|
||||
// The tool identity is keyed on the <atem:invoke name="..."> (the
|
||||
// actual call), NOT the " to=<recipient>". Onyx models sometimes emit
|
||||
// a recipient that disagrees with the invoke name (e.g. " to=
|
||||
// container.exec" with <atem:invoke name="bash.exec">); accept any
|
||||
// recipient here so such calls are parsed rather than dropped. The
|
||||
// "<atem:function_calls>" literal disambiguates this from the to=user
|
||||
// final turn.
|
||||
auto tool_parser = p.tool(
|
||||
p.tool_open(p.literal(" to=") + p.until("<|message|>") +
|
||||
p.literal("<|message|><atem:function_calls>") + p.space() +
|
||||
p.literal("<atem:invoke name=\"") + p.tool_name(p.literal(name)) + p.literal("\">") + p.space())
|
||||
<< p.tool_args(args)
|
||||
<< p.tool_close(p.literal("</atem:invoke>") + p.space() + p.literal("</atem:function_calls>")));
|
||||
|
||||
tool_choice |= p.rule("tool-" + name, tool_parser);
|
||||
});
|
||||
|
||||
// Parallel tool calls are consecutive assistant messages separated by <|eom|>, the turn
|
||||
// ending with <|eot|> after the last. Gate the loop on parallel_tool_calls like the other
|
||||
// formats: disabled -> single call; enabled -> multiple. NOTE: this model is unreliable at
|
||||
// single-turn parallel (it often ends the turn after the first call), so
|
||||
// parallel_tool_calls=false (sequential, driven by the agent loop) is recommended; true is
|
||||
// honored but may under-emit.
|
||||
auto tool_calls = inputs.parallel_tool_calls
|
||||
? p.trigger_rule("tool-call", tool_choice + p.zero_or_more(p.literal("<|eom|>") + start + tool_choice))
|
||||
: p.trigger_rule("tool-call", tool_choice);
|
||||
|
||||
|
||||
if (inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED) {
|
||||
return p.zero_or_more(start + analysis) + start + tool_calls;
|
||||
}
|
||||
return p.zero_or_more(start + analysis) + start + (tool_calls | final_msg);
|
||||
}
|
||||
|
||||
return p.zero_or_more(start + analysis) + start + final_msg;
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
auto schema = function.contains("parameters") ? function.at("parameters") : json::object();
|
||||
builder.resolve_refs(schema);
|
||||
});
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
// Trigger only on a TOOL recipient, not the to=self (reasoning) or to=user (final) turns,
|
||||
// which are free text and would empty the grammar stack. The capture group starts at " to="
|
||||
// (matching where the grammar's tool rule begins) so the replayed input lines up with the
|
||||
// grammar; requiring the trailing "<|message|>" (and excluding self/user before it) means
|
||||
// the pattern only fires once the full tool recipient is known.
|
||||
data.grammar_triggers = {
|
||||
{ COMMON_GRAMMAR_TRIGGER_TYPE_PATTERN,
|
||||
"<\\|start\\|>assistant( to=(?!self<\\|message\\|>)(?!user<\\|message\\|>)[^<]*?<\\|message\\|>)" },
|
||||
};
|
||||
}
|
||||
|
||||
return data;
|
||||
}
|
||||
|
||||
static json common_chat_extra_context() {
|
||||
json ctx = json::object();
|
||||
std::chrono::system_clock::time_point now = std::chrono::system_clock::now();
|
||||
@@ -2923,6 +3121,14 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
return common_chat_params_init_gpt_oss(tmpl, params);
|
||||
}
|
||||
|
||||
// Onyx format using " to=<recipient>" recipients and <|eom|>/<|eot|> message
|
||||
// terminators. The ATEM tool-call markup "<atem:function_calls>" is unique to
|
||||
// this template.
|
||||
if (src.find("<atem:function_calls>") != std::string::npos && src.find("<|eom|>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Onyx\n");
|
||||
return common_chat_params_init_onyx(tmpl, params);
|
||||
}
|
||||
|
||||
// Functionary v3.2 - uses recipient-based format with >>>recipient\n{content}
|
||||
// Detection: template has ">>>all" for content and ">>>" prefix for tool calls
|
||||
if (src.find(">>>all") != std::string::npos && src.find(">>>${recipient}") != std::string::npos) {
|
||||
|
||||
Reference in New Issue
Block a user