mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-11 07:20:33 +02:00
chat : fix Muse Glimmer ignoring response_format json_schema with --jinja (#29615)
* chat : fix Muse Glimmer ignoring response_format json_schema with --jinja Fixes #29613 * chat : accept json fences and clean up * chat : fix choice parenthesis --------- Co-authored-by: Alde Rojas <hello@alde.dev>
This commit is contained in:
co-authored by
Alde Rojas
parent
76a5bc86d1
commit
139997d8e7
@@ -43,9 +43,10 @@ common_chat_params common_chat_params_init_muse_glimmer(const common_chat_templa
|
||||
|
||||
auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
||||
|
||||
auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();
|
||||
// Constrained grammar whenever tools are offered.
|
||||
auto include_grammar = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;
|
||||
auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();
|
||||
auto has_response_format = !inputs.json_schema.is_null() && inputs.json_schema.is_object();
|
||||
// Constrained grammar whenever tools are offered or a response format is requested.
|
||||
auto include_grammar = has_response_format || (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE);
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto start = p.rule("start", p.literal("<|start|>assistant"));
|
||||
@@ -65,6 +66,15 @@ common_chat_params common_chat_params_init_muse_glimmer(const common_chat_templa
|
||||
auto final_msg = p.rule("final", recipient + p.literal("<|message|>") +
|
||||
p.content(p.until_one_of({ "<|eot|>", "<|eom|>" })));
|
||||
|
||||
if (has_response_format) {
|
||||
auto response_json = p.content(p.schema(p.json(), "response-format-schema", inputs.json_schema));
|
||||
auto response_format = p.rule("response-format",
|
||||
recipient + p.literal("<|message|>") +
|
||||
((p.literal("```json") + p.space() + response_json + p.space() + p.literal("```")) | response_json));
|
||||
|
||||
return p.zero_or_more(start + analysis) + start + response_format;
|
||||
}
|
||||
|
||||
if (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE) {
|
||||
auto string_value = p.ac(
|
||||
p.tool_arg_string_value(p.until("</atem:parameter>")) + p.tool_arg_close(p.literal("</atem:parameter>")),
|
||||
@@ -124,7 +134,7 @@ common_chat_params common_chat_params_init_muse_glimmer(const common_chat_templa
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;
|
||||
data.grammar_lazy = !(has_response_format || (has_tools && inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED));
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
@@ -1217,6 +1217,16 @@ static void test_peg_parser(common_chat_templates * tmpls,
|
||||
}
|
||||
assert_msg_equals(tc.expect, msg_accum, true);
|
||||
|
||||
// A response format must be enforced by an eager grammar
|
||||
if (!tc.params.json_schema.empty()) {
|
||||
if (parser.params_.grammar.empty()) {
|
||||
throw std::runtime_error("json_schema is set but no grammar was produced");
|
||||
}
|
||||
if (parser.params_.grammar_lazy) {
|
||||
throw std::runtime_error("json_schema is set but the grammar is lazy");
|
||||
}
|
||||
}
|
||||
|
||||
// Test grammar if present in params
|
||||
if (!parser.params_.grammar.empty()) {
|
||||
auto grammar = build_grammar(parser.params_.grammar);
|
||||
@@ -6384,6 +6394,22 @@ static void test_template_output_peg_parsers(bool detailed_debug) {
|
||||
.expect_content("You invoke it like this:\n" + call_markup)
|
||||
.run();
|
||||
|
||||
// Structured output, straight to the final answer
|
||||
tst.test(" to=user<|message|>" R"({"amount": 123.45, "date": "2025-12-03"})")
|
||||
.reasoning_format(COMMON_REASONING_FORMAT_AUTO)
|
||||
.json_schema(invoice_schema)
|
||||
.expect_content(R"({"amount": 123.45, "date": "2025-12-03"})")
|
||||
.run();
|
||||
|
||||
// Structured output after a reasoning message: reasoning stays free-form
|
||||
tst.test(" to=self<|message|>I need to output the invoice details in JSON<|eom|>"
|
||||
"<|start|>assistant to=user<|message|>" R"({"amount": 123.45, "date": "2025-12-03"})")
|
||||
.reasoning_format(COMMON_REASONING_FORMAT_AUTO)
|
||||
.json_schema(invoice_schema)
|
||||
.expect_reasoning("I need to output the invoice details in JSON")
|
||||
.expect_content(R"({"amount": 123.45, "date": "2025-12-03"})")
|
||||
.run();
|
||||
|
||||
// Tool markup inside the analysis channel is reasoning, not a call
|
||||
tst.test(" to=self<|message|>I could use " + call_markup + " here<|eom|>"
|
||||
"<|start|>assistant to=user<|message|>Hello!<|eot|>")
|
||||
|
||||
Reference in New Issue
Block a user