Commit 139997d8e for llama.cpp
commit 139997d8e74255f7c7a64759f02e3ac42c2e595c
Author: vaibhavdedhia <vaibhavdedhia@gmail.com>
Date: Tue Sep 29 11:48:58 2026 +0530
chat : fix Muse Glimmer ignoring response_format json_schema with --jinja (#29615)
* chat : fix Muse Glimmer ignoring response_format json_schema with --jinja
Fixes #29613
* chat : accept json fences and clean up
* chat : fix choice parenthesis
---------
Co-authored-by: Alde Rojas <hello@alde.dev>
diff --git a/common/parsers/muse-glimmer.cpp b/common/parsers/muse-glimmer.cpp
index d03bf2d58..784928269 100644
--- a/common/parsers/muse-glimmer.cpp
+++ b/common/parsers/muse-glimmer.cpp
@@ -43,9 +43,10 @@ common_chat_params common_chat_params_init_muse_glimmer(const common_chat_templa
auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
- auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();
- // Constrained grammar whenever tools are offered.
- auto include_grammar = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;
+ auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();
+ auto has_response_format = !inputs.json_schema.is_null() && inputs.json_schema.is_object();
+ // Constrained grammar whenever tools are offered or a response format is requested.
+ auto include_grammar = has_response_format || (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE);
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
auto start = p.rule("start", p.literal("<|start|>assistant"));
@@ -65,6 +66,15 @@ common_chat_params common_chat_params_init_muse_glimmer(const common_chat_templa
auto final_msg = p.rule("final", recipient + p.literal("<|message|>") +
p.content(p.until_one_of({ "<|eot|>", "<|eom|>" })));
+ if (has_response_format) {
+ auto response_json = p.content(p.schema(p.json(), "response-format-schema", inputs.json_schema));
+ auto response_format = p.rule("response-format",
+ recipient + p.literal("<|message|>") +
+ ((p.literal("```json") + p.space() + response_json + p.space() + p.literal("```")) | response_json));
+
+ return p.zero_or_more(start + analysis) + start + response_format;
+ }
+
if (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE) {
auto string_value = p.ac(
p.tool_arg_string_value(p.until("</atem:parameter>")) + p.tool_arg_close(p.literal("</atem:parameter>")),
@@ -124,7 +134,7 @@ common_chat_params common_chat_params_init_muse_glimmer(const common_chat_templa
data.parser = parser.save();
if (include_grammar) {
- data.grammar_lazy = inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;
+ data.grammar_lazy = !(has_response_format || (has_tools && inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED));
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
parser.build_grammar(builder, data.grammar_lazy);
});
diff --git a/tests/test-chat.cpp b/tests/test-chat.cpp
index f2728f7cc..6b8b85616 100644
--- a/tests/test-chat.cpp
+++ b/tests/test-chat.cpp
@@ -1217,6 +1217,16 @@ static void test_peg_parser(common_chat_templates * tmpls,
}
assert_msg_equals(tc.expect, msg_accum, true);
+ // A response format must be enforced by an eager grammar
+ if (!tc.params.json_schema.empty()) {
+ if (parser.params_.grammar.empty()) {
+ throw std::runtime_error("json_schema is set but no grammar was produced");
+ }
+ if (parser.params_.grammar_lazy) {
+ throw std::runtime_error("json_schema is set but the grammar is lazy");
+ }
+ }
+
// Test grammar if present in params
if (!parser.params_.grammar.empty()) {
auto grammar = build_grammar(parser.params_.grammar);
@@ -6384,6 +6394,22 @@ static void test_template_output_peg_parsers(bool detailed_debug) {
.expect_content("You invoke it like this:\n" + call_markup)
.run();
+ // Structured output, straight to the final answer
+ tst.test(" to=user<|message|>" R"({"amount": 123.45, "date": "2025-12-03"})")
+ .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+ .json_schema(invoice_schema)
+ .expect_content(R"({"amount": 123.45, "date": "2025-12-03"})")
+ .run();
+
+ // Structured output after a reasoning message: reasoning stays free-form
+ tst.test(" to=self<|message|>I need to output the invoice details in JSON<|eom|>"
+ "<|start|>assistant to=user<|message|>" R"({"amount": 123.45, "date": "2025-12-03"})")
+ .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+ .json_schema(invoice_schema)
+ .expect_reasoning("I need to output the invoice details in JSON")
+ .expect_content(R"({"amount": 123.45, "date": "2025-12-03"})")
+ .run();
+
// Tool markup inside the analysis channel is reasoning, not a call
tst.test(" to=self<|message|>I could use " + call_markup + " here<|eom|>"
"<|start|>assistant to=user<|message|>Hello!<|eot|>")