Commit 9bf55f4a3 for llama.cpp

commit 9bf55f4a3677af697d914d959eaa70f93cfdc494
Author: Yash Raj Pandey <55940078+devYRPauli@users.noreply.github.com>
Date:   Sat Oct 3 09:40:50 2026 -0400

    chat : honor json_schema in Ling 3.0 parser (#29813)

    * chat : honor json_schema in Ling 3.0 parser

    Ling 3.0 only built a grammar for tool calls and did not handle inputs.json_schema, so response_format requests were left unconstrained.

    Add an eager response-format grammar path with precedence over tools, following the existing parser patterns. Require </think> before JSON when thinking is enabled and do not allow trailing prose after the JSON response.

    Fixes #29652.

    Assisted-by: Claude Opus 5.5

    * chat : require Ling 3.0 think block for response formats

diff --git a/common/parsers/ling3.cpp b/common/parsers/ling3.cpp
index 8b49847e2..8b60e40f4 100644
--- a/common/parsers/ling3.cpp
+++ b/common/parsers/ling3.cpp
@@ -75,9 +75,10 @@ common_chat_params common_chat_params_init_ling3(const common_chat_template &
                      (last_close == std::string::npos || last_open > last_close);
     }

-    auto has_tools         = inputs.tools.is_array() && !inputs.tools.empty();
-    auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
-    auto include_grammar   = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;
+    auto has_tools           = inputs.tools.is_array() && !inputs.tools.empty();
+    auto has_response_format = inputs.json_schema.is_object() && !inputs.json_schema.empty();
+    auto extract_reasoning   = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
+    auto include_grammar     = has_response_format || (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE);

     auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
         auto end = p.end();
@@ -101,6 +102,13 @@ common_chat_params common_chat_params_init_ling3(const common_chat_template &
         // a trailing end-of-turn token is consumed instead of leaking into content
         auto tail = p.optional(p.content(p.until(ROLE_END))) + p.optional(p.literal(ROLE_END));

+        // the think block must close before the JSON, so the turn cannot end inside the reasoning
+        if (has_response_format) {
+            auto closed_reasoning = p.literal(THINK_START) + think_body + p.literal(THINK_END);
+            auto response_format  = p.content(p.schema(p.json(), "response-format", inputs.json_schema));
+            return opener + (closed_reasoning << response_format) + end;
+        }
+
         if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {
             return opener + reasoning + tail + end;
         }
@@ -180,7 +188,7 @@ common_chat_params common_chat_params_init_ling3(const common_chat_template &
     data.parser = parser.save();

     if (include_grammar) {
-        data.grammar_lazy = inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;
+        data.grammar_lazy = !has_response_format && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;
         data.grammar      = build_grammar([&](const common_grammar_builder & builder) {
             parser.build_grammar(builder, data.grammar_lazy);
         });
diff --git a/tests/test-chat.cpp b/tests/test-chat.cpp
index d0fe08585..c7edf0c2d 100644
--- a/tests/test-chat.cpp
+++ b/tests/test-chat.cpp
@@ -4835,6 +4835,22 @@ static void test_template_output_peg_parsers(bool detailed_debug) {
             .expect_tool_calls({ { "set_union", R"({"value": "plain text", "amount": "1abc"})", "" } })
             .run();

+        // A response format is enforced with thinking on and off.
+        tst.test("I need to output the invoice details in JSON\n</think>\n"
+                 R"({"amount": 123.45, "date": "2025-12-03"})")
+            .reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
+            .json_schema(invoice_schema)
+            .expect_reasoning("I need to output the invoice details in JSON\n")
+            .expect_content(R"({"amount": 123.45, "date": "2025-12-03"})")
+            .run();
+
+        tst.test(R"({"amount": 123.45, "date": "2025-12-03"})")
+            .reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
+            .enable_thinking(false)
+            .json_schema(invoice_schema)
+            .expect_content(R"({"amount": 123.45, "date": "2025-12-03"})")
+            .run();
+
         // Continuation: the partial assistant turn is spliced back into the prompt.
         common_chat_msg prefill = simple_assist_msg("", "I'm thinking");