Commit 462524043 for llama.cpp

commit 4625240437c6821ee5c2da99e12d4817c6a8f07f
Author: bitalov <60072763+bitalov@users.noreply.github.com>
Date:   Tue Oct 6 20:42:24 2026 +0400

    model : add K2 Horizon dense and MoVA support (#29535)

    * model: K2 Horizon gguf conversion code

    * model: loading hparams and tensors in k2-horizon.cpp

    * model: K2 Horizon compute graph

    * model: K2 Horizon compute graph adjustment and registering tokenizers

    * model: K2 Horizon chat template and accomodate safetensors naming

    * unicode : add the K2-Horizon pre-tokenizer splitter

    The K2-Horizon regex had no arm in unicode_regex_split_custom and fell through to the
    general std::regex fallback, which fails two ways.

    On MSVC std::regex rejects \p{...}, so no K2-Horizon GGUF loads on Windows at all:
    llama-quantize, llama-imatrix and llama-perplexity all abort with
    regex_error(error_escape) before a token is produced.

    Where the fallback does compile it is still wrong. unicode_regex_split collapses each
    codepoint to a single byte naming its Unicode category before matching, and U+200C/U+200D
    are category Control, which has no entry in k_ucat_cpt, so both become the 0xD0 fallback
    byte. The literal ‌ and ‍ alternatives in K2's regex can then never match and
    every ZWNJ or ZWJ ends a letter run.

    The splitter is the existing llama3 one with a single rule widened, since K2's regex
    differs from llama3's only in that a letter run also takes marks, ZWNJ and ZWJ.

    tests/test-unicode.cpp gains a case for this: it fails before the change with
    [Amy] [ZWNJ khaham] and passes after with the run intact.

    * tests: expand K2 Horizon unicode splitter coverage

    * unicode: handle K2 Horizon case folding and empty input

    Assisted-by: Codex

    * jinja : support sequence indices in selectattr and rejectattr

    Assisted-by: Codex

    * model : add K2 Horizon dense and MoVA support

    Includes the K2 Horizon implementation from ifm-ai/llama.cpp with converter, tensor-parallel and model save/reload fixes.

    Assisted-by: Codex

    * chat : support K2 Horizon reasoning and tool calls

    Assisted-by: Codex

    * conversion: remove obsolete K2 Aurora alias

    Assisted-by: Codex

    * k2-horizon: enforce response schemas and load YaRN betas

    Constrain final JSON after reasoning, accept flexible JSON tool envelopes,
    enforce XML dialects, and handle repeated or alternate thinking markers.
    Load YaRN beta metadata instead of retaining the default values.

    Add schema, streaming, continuation, and model reload regressions. Validate
    CUDA and CPU builds and 0.9B, 4B, and MoVA conversation/tool round trips.

    Assisted-by: Codex

    * renaming template fixture

    * adressing cisc follows ups

    * desloppify the parser / adress aldehir comments

    * clean test-chat

    * remove fallback : model trained mostly on high anyway

    * fix k2 attn_v_exp tn splitting and metal fusion baseline

    * k2-horizon : forward expand views before sums

    * k2-horizon: copy embds before group norm to fix TP

    * disable tesnor parallelism

    ---------

    Co-authored-by: Ryandito Diandaru <ryandito.diandaru@mbzuai.ac.ae>
    Co-authored-by: WestWaters <mario.papaleo2013@gmail.com>
    Co-authored-by: Natani L. Mayday <71436458+TaskPuppyNatani@users.noreply.github.com>
    Co-authored-by: West <100190545+WestWaters@users.noreply.github.com>
    Co-authored-by: aaryamonvikram <aaryamonvikram@gmail.com>
    Co-authored-by: aaryamonvikram <96529820+aaryamonvikram@users.noreply.github.com>

diff --git a/common/chat.cpp b/common/chat.cpp
index 3502b96eb..1c1533325 100644
--- a/common/chat.cpp
+++ b/common/chat.cpp
@@ -1139,6 +1139,14 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
         return common_chat_params_init_kimi_k3(tmpl, params);
     }

+    // K2 Horizon - <|ifm|im_start|> turns, <ifm|think*> reasoning picked by reasoning_effort and
+    // <ifm|tool_calls> sections; the three think tag pairs defeat the autoparser's reasoning detection
+    if (src.find("<|ifm|im_start|>") != std::string::npos &&
+        src.find("<ifm|tool_calls>") != std::string::npos) {
+        LOG_DBG("Using specialized template: K2 Horizon\n");
+        return common_chat_params_init_k2_horizon(tmpl, params);
+    }
+
     // Ling 3.0 / Bailing V3 - <role>X</role> sections with <arg_key>/<arg_value> tagged
     // tool calls. <role> sections are unique to this family among the tagged-arg templates.
     if (src.find("<role>ASSISTANT</role>") != std::string::npos &&
diff --git a/common/jinja/caps.cpp b/common/jinja/caps.cpp
index c5962ab77..6ff17d10b 100644
--- a/common/jinja/caps.cpp
+++ b/common/jinja/caps.cpp
@@ -37,38 +37,57 @@ static void caps_try_execute(jinja::program & prog,
                              const caps_ctx_fn & ctx_fn,
                              const caps_json_fn & tools_fn,
                              const caps_analyze_fn & analyze_fn) {
-    context ctx;
-    ctx.is_get_stats = true;
-    jinja::global_from_json(ctx, json{
-        {"messages", messages_fn()},
-        {"tools", tools_fn ? tools_fn() : json::array()},
-        {"bos_token", ""},
-        {"eos_token", ""},
-        {"add_generation_prompt", true}
-    }, true);
-
-    if (ctx_fn) {
-        ctx_fn(ctx);
-    }
+    json msgs = messages_fn();
+    for (int attempt = 0; attempt < 2; attempt++) {
+        context ctx;
+        ctx.is_get_stats = true;
+        jinja::global_from_json(ctx, json{
+            {"messages", msgs},
+            {"tools", tools_fn ? tools_fn() : json::array()},
+            {"bos_token", ""},
+            {"eos_token", ""},
+            {"add_generation_prompt", true}
+        }, true);
+
+        if (ctx_fn) {
+            ctx_fn(ctx);
+        }

-    auto messages = ctx.get_val("messages");
-    auto tools = ctx.get_val("tools");
-
-    bool success = false;
-    std::string result;
-    try {
-        jinja::runtime runtime(ctx);
-        auto results = runtime.execute(prog);
-        auto parts = jinja::runtime::gather_string_parts(results);
-        result = parts->as_string().str();
-        success = true;
-    } catch (const std::exception & e) {
-        JJ_DEBUG("Exception during execution: %s", e.what());
-        result = "";
-        // ignore exceptions during capability analysis
-    }
+        auto messages = ctx.get_val("messages");
+        auto tools = ctx.get_val("tools");
+
+        bool success = false;
+        std::string result;
+        try {
+            jinja::runtime runtime(ctx);
+            auto results = runtime.execute(prog);
+            auto parts = jinja::runtime::gather_string_parts(results);
+            result = parts->as_string().str();
+            success = true;
+        } catch (const std::exception & e) {
+            JJ_DEBUG("Exception during execution: %s", e.what());
+            result = "";
+            // ignore exceptions during capability analysis
+        }
+
+        // some templates require a thinking field on every assistant turn (e.g. K2 Horizon):
+        // retry once with an empty reasoning_content on the assistant turns that lack one
+        if (!success && attempt == 0) {
+            bool added = false;
+            for (auto & msg : msgs) {
+                if (msg.is_object() && msg.value("role", "") == "assistant" && !msg.contains("reasoning_content")) {
+                    msg["reasoning_content"] = "";
+                    added = true;
+                }
+            }
+            if (added) {
+                continue;
+            }
+        }

-    analyze_fn(ctx, success, messages, tools, result);
+        analyze_fn(ctx, success, messages, tools, result);
+        return;
+    }
 }

 // for debugging only
diff --git a/common/parsers/k2-horizon.cpp b/common/parsers/k2-horizon.cpp
new file mode 100644
index 000000000..687a893ce
--- /dev/null
+++ b/common/parsers/k2-horizon.cpp
@@ -0,0 +1,193 @@
+#include "parsers.h"
+
+// K2 Horizon format:
+// - Reasoning: <ifm|think>...</ifm|think>, or <ifm|think_fast>/<ifm|think_faster> for medium/low reasoning_effort
+// - Tool calls: <ifm|tool_calls><ifm|tool_call>...</ifm|tool_call>...</ifm|tool_calls>, one call per <ifm|tool_call>:
+//   xml (default): name <ifm|arg_key>k</ifm|arg_key> [<ifm|arg_type>t</ifm|arg_type>] <ifm|arg_value>v</ifm|arg_value> ...
+//   json:          {"name": "...", "arguments": {...}}
+common_chat_params common_chat_params_init_k2_horizon(const common_chat_template &          tmpl,
+                                                      const autoparser::generation_params & inputs) {
+    common_chat_params data;
+
+    // The template requires a thinking field on every assistant message
+    auto messages = inputs.messages;
+    for (auto & msg : messages) {
+        if (msg.value("role", "") == "assistant" && !msg.contains("reasoning_content")) {
+            msg["reasoning_content"] = "";
+        }
+    }
+
+    data.prompt            = common_chat_template_direct_apply_impl(tmpl, inputs, messages);
+    data.generation_prompt = common_chat_template_generation_prompt_impl(tmpl, inputs, messages);
+    data.format            = COMMON_CHAT_FORMAT_PEG_NATIVE;
+    data.supports_thinking = true;
+
+    const std::string effort      = inputs.extra_context.value("reasoning_effort", "high");
+    const std::string call_format = inputs.extra_context.value("tool_call_format", "xml");
+
+    // Templates that handle enable_thinking disable it with an empty <ifm|think></ifm|think> block for every effort
+    const bool thinking_off = !inputs.enable_thinking && tmpl.source().find("enable_thinking") != std::string::npos;
+    const std::string think = thinking_off       ? "ifm|think"        :
+                              effort == "medium" ? "ifm|think_fast"   :
+                              effort == "low"    ? "ifm|think_faster" : "ifm|think";
+
+    const std::string GEN_PREFIX    = "<|ifm|im_start|>assistant\n";
+    const std::string THINK_START   = "<" + think + ">";
+    const std::string THINK_END     = "</" + think + ">";
+    const std::string SECTION_START = "<ifm|tool_calls>";
+    const std::string SECTION_END   = "</ifm|tool_calls>";
+    const std::string CALL_START    = "<ifm|tool_call>";
+    const std::string CALL_END      = "</ifm|tool_call>";
+    const std::string ARG_KEY       = "<ifm|arg_key>";
+    const std::string ARG_KEY_END   = "</ifm|arg_key>";
+    const std::string ARG_TYPE      = "<ifm|arg_type>";
+    const std::string ARG_TYPE_END  = "</ifm|arg_type>";
+    const std::string ARG_VAL       = "<ifm|arg_value>";
+    const std::string ARG_VAL_END   = "</ifm|arg_value>";
+
+    data.thinking_start_tag = THINK_START;
+    data.thinking_end_tags  = { THINK_END };
+
+    data.preserved_tokens = data.thinking_end_tags;
+    data.preserved_tokens.insert(data.preserved_tokens.end(), {
+        THINK_START, SECTION_START, SECTION_END, CALL_START, CALL_END,
+        ARG_KEY, ARG_KEY_END, ARG_TYPE, ARG_TYPE_END, ARG_VAL, ARG_VAL_END,
+    });
+
+    data.message_delimiters = {
+        { COMMON_CHAT_ROLE_ASSISTANT, "<|ifm|im_start|>assistant" },
+        { COMMON_CHAT_ROLE_USER,      "<|ifm|im_start|>user"      },
+        { COMMON_CHAT_ROLE_TOOL,      "<|ifm|im_start|>tool"      },
+        { COMMON_CHAT_ROLE_SYSTEM,    "<|ifm|im_start|>system"    },
+    };
+
+    auto has_tools           = inputs.tools.is_array() && !inputs.tools.empty();
+    auto has_response_format = inputs.json_schema.is_object() && !inputs.json_schema.empty();
+    auto extract_reasoning   = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
+    auto include_grammar     = has_response_format || (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE);
+
+    if (inputs.has_continuation()) {
+        const auto & msg = inputs.continue_msg;
+
+        data.generation_prompt = GEN_PREFIX + THINK_START + "\n" + msg.reasoning_content;
+        if (inputs.continue_final_message == COMMON_CHAT_CONTINUATION_CONTENT) {
+            data.generation_prompt += THINK_END + msg.render_content();
+        }
+
+        data.prompt += data.generation_prompt;
+    }
+
+    auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
+        auto generation_prompt = p.literal(GEN_PREFIX);
+
+        auto think_end = p.choice();
+        for (const auto & tag : data.thinking_end_tags) {
+            think_end |= p.literal(tag);
+        }
+        auto think_body  = p.until_one_of(data.thinking_end_tags);
+        auto think_block = [&](const common_peg_parser & body) {
+            return p.optional(THINK_START + p.space() + p.ac(body + think_end, data.thinking_end_tags));
+        };
+        auto reasoning = extract_reasoning ? think_block(p.reasoning(think_body)) : p.eps();
+
+        if (has_response_format) {
+            // The answer must be bare JSON, so the think block is consumed even when it is not extracted
+            auto thoughts = extract_reasoning ? reasoning : think_block(think_body);
+            return generation_prompt + (thoughts << p.content(p.schema(p.json(), "response-format", inputs.json_schema)));
+        }
+
+        if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {
+            return generation_prompt + (reasoning << p.content(p.rest()));
+        }
+
+        auto tool_choice = p.choice();
+        if (call_format == "json") {
+            tool_choice = p.standard_json_tools(CALL_START, CALL_END, inputs.tools, false, true);
+        } else {
+            auto arg_close  = p.tool_arg_close(p.literal(ARG_VAL_END));
+            auto arg_string = p.rule("xml-arg-string", p.ac(p.tool_arg_string_value(p.until(ARG_VAL_END)) + arg_close, ARG_VAL_END));
+
+            // The models leave out <ifm|arg_type> even when asked for xml_typed
+            auto arg_type = call_format == "xml_typed" ? p.optional(ARG_TYPE + p.until(ARG_TYPE_END) + ARG_TYPE_END + p.space()) : p.eps();
+
+            foreach_function(inputs.tools, [&](const json & tool) {
+                const auto & function = tool.at("function");
+                std::string  name     = function.at("name");
+
+                std::vector<common_peg_parser> required_args;
+                std::vector<common_peg_parser> optional_args;
+                foreach_parameter(function, [&](const common_chat_schema_property & param, const common_chat_schema_document_ptr & doc) {
+                    auto rule_name = "tool-" + name + "-arg-" + param.name;
+                    auto types     = param.schema->value_types();
+                    auto arg_value = arg_string;
+                    if (!types.has(common_chat_schema::TYPE_STRING)) {
+                        arg_value = p.tool_arg_json_value(p.schema(p.json(), rule_name + "-schema", doc, *param.schema)) + arg_close;
+                    }
+                    if (types.has(common_chat_schema::TYPE_STRING) && !types.is_only(common_chat_schema::TYPE_STRING)) {
+                        // The string alternative accepts any text, so only the parser needs the JSON alternatives.
+                        auto json_value = p.choice();
+                        if (types.has(common_chat_schema::TYPE_OBJECT)) {
+                            json_value |= p.json_object();
+                        }
+                        if (types.has(common_chat_schema::TYPE_ARRAY)) {
+                            json_value |= p.json_array();
+                        }
+                        if (types.has(common_chat_schema::TYPE_NUMBER) || types.has(common_chat_schema::TYPE_INTEGER)) {
+                            json_value |= p.json_number();
+                        }
+                        if (types.has(common_chat_schema::TYPE_BOOLEAN)) {
+                            json_value |= p.json_bool();
+                        }
+                        if (types.has(common_chat_schema::TYPE_NULL)) {
+                            json_value |= p.json_null();
+                        }
+                        arg_value = p.gbnf(p.atomic(p.tool_arg_json_value(json_value) + arg_close) | arg_string, "xml-arg-string");
+                    }
+
+                    auto arg = p.space() + p.tool_arg(p.tool_arg_open(ARG_KEY + p.tool_arg_name(p.literal(param.name)) + ARG_KEY_END) <<
+                                                      arg_type + ARG_VAL + arg_value);
+                    (param.required ? required_args : optional_args).push_back(p.rule(rule_name, arg));
+                });
+
+                auto args = p.permute("tool-" + name + "-args", required_args);
+                if (!optional_args.empty()) {
+                    args = args + p.zero_or_more(p.choice(optional_args));
+                }
+
+                tool_choice |= p.rule("tool-" + name, p.tool(
+                    p.tool_open(CALL_START + p.tool_name(p.literal(name)) + "\n") + p.tool_args(args) << p.tool_close(p.literal(CALL_END))));
+            });
+        }
+
+        auto required   = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED;
+        auto calls      = inputs.parallel_tool_calls ? tool_choice + p.zero_or_more(p.space() + tool_choice) : tool_choice;
+        auto tool_calls = p.trigger_rule("tool-calls", p.repeat(SECTION_START << calls << SECTION_END, required ? 1 : 0, 1));
+
+        // Keep thinking inline when required calls bypass the content parser.
+        if (required && !extract_reasoning) {
+            reasoning = p.content(think_block(think_body));
+        }
+
+        // A required call follows the reasoning directly, the models otherwise keep writing content
+        auto content = required ? p.eps() : p.content(p.until(SECTION_START));
+
+        return generation_prompt + (reasoning << content << tool_calls);
+    });
+
+    data.parser = parser.save();
+
+    if (include_grammar) {
+        data.grammar_lazy = !(has_response_format || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED);
+        data.grammar      = build_grammar([&](const common_grammar_builder & builder) {
+            parser.build_grammar(builder, data.grammar_lazy);
+        });
+
+        if (data.grammar_lazy) {
+            data.grammar_triggers = {
+                { COMMON_GRAMMAR_TRIGGER_TYPE_WORD, SECTION_START },
+            };
+        }
+    }
+
+    return data;
+}
diff --git a/common/parsers/parsers.h b/common/parsers/parsers.h
index 0caf62d8c..7786b82f4 100644
--- a/common/parsers/parsers.h
+++ b/common/parsers/parsers.h
@@ -59,6 +59,8 @@ common_chat_params common_chat_params_init_gigachat_v3(const common_chat_templat

 common_chat_params common_chat_params_init_gpt_oss(const common_chat_template & tmpl, const autoparser::generation_params & inputs);

+common_chat_params common_chat_params_init_k2_horizon(const common_chat_template & tmpl, const autoparser::generation_params & inputs);
+
 common_chat_params common_chat_params_init_kimi_k2(const common_chat_template & tmpl, const autoparser::generation_params & inputs);

 common_chat_params common_chat_params_init_kimi_k3(const common_chat_template & tmpl, const autoparser::generation_params & inputs);
diff --git a/common/parsers/sources.cmake b/common/parsers/sources.cmake
index 5815939d6..af0c68019 100644
--- a/common/parsers/sources.cmake
+++ b/common/parsers/sources.cmake
@@ -9,6 +9,7 @@ set(LLAMA_CHAT_PARSERS_SOURCES
     ${CMAKE_CURRENT_LIST_DIR}/gemma4.cpp
     ${CMAKE_CURRENT_LIST_DIR}/gigachat-v3.cpp
     ${CMAKE_CURRENT_LIST_DIR}/gpt-oss.cpp
+    ${CMAKE_CURRENT_LIST_DIR}/k2-horizon.cpp
     ${CMAKE_CURRENT_LIST_DIR}/kimi-k2.cpp
     ${CMAKE_CURRENT_LIST_DIR}/kimi-k3.cpp
     ${CMAKE_CURRENT_LIST_DIR}/ling3.cpp
diff --git a/conversion/__init__.py b/conversion/__init__.py
index d8af76a27..96b018adf 100644
--- a/conversion/__init__.py
+++ b/conversion/__init__.py
@@ -142,6 +142,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
     "JinaBertForMaskedLM": "bert",
     "JinaBertModel": "bert",
     "JinaEmbeddingsV5Model": "bert",
+    "K2HorizonForCausalLM": "k2_horizon",
     "KORMoForCausalLM": "qwen",
     "KimiK25ForConditionalGeneration": "deepseek",
     "KimiK3ForConditionalGeneration": "kimi_k3",
diff --git a/conversion/base.py b/conversion/base.py
index 0f3bd9a7b..e28fad479 100644
--- a/conversion/base.py
+++ b/conversion/base.py
@@ -1529,7 +1529,7 @@ class TextModel(ModelBase):
             self.gguf_writer.add_expert_group_used_count(n_group_used)
             logger.info(f"gguf: expert groups used count = {n_group_used}")

-        if (score_func := self.find_hparam(["score_function", "scoring_func", "score_func", "moe_router_activation", "moe_router_activation_func", "expert_selection_fn"], optional=True)) is not None:
+        if (score_func := self.find_hparam(["score_function", "scoring_func", "score_func", "moe_router_activation", "moe_router_activation_func", "expert_selection_fn", "router_score_func"], optional=True)) is not None:
             if score_func == "sigmoid":
                 self.gguf_writer.add_expert_gating_func(gguf.ExpertGatingFuncType.SIGMOID)
             elif score_func == "softmax":
@@ -1713,6 +1713,9 @@ class TextModel(ModelBase):
         if chkhsh == "0a766d034107bc736a3f2dc4968fd62e54a3570f1454443e0c5a4cc6bd7941ed":
             # ref: https://huggingface.co/XHToken/Spark-X2.5-1.7B
             res = "spark2_5"
+        if chkhsh == "1f9825a388f700a6b591722f17d470cbbcf10973ece35d2fd14239a14110ae1a":
+            # ref: https://huggingface.co/IFM/K2-Horizon-0.9B
+            res = "k2-horizon"
         if chkhsh == "0ef9807a4087ebef797fc749390439009c3b9eda9ad1a097abbe738f486c01e5":
             # ref: https://huggingface.co/meta-llama/Meta-Llama-3-8B
             res = "llama-bpe"
@@ -1941,6 +1944,9 @@ class TextModel(ModelBase):
         if chkhsh == "4b05e02dad1c5ae07d266fd3342ddb644c6f6be058d728bc0a33af31a1d6ee66":
             # ref: https://huggingface.co/jhu-clsp/mmBERT-base
             res = "mmbert"
+        if chkhsh == "a9af07a84191f55098b248ae6f3dfe9e32d3190bebe8eafd91c1ddec9bc3449f":
+            # ref: https://huggingface.co/IFM/K2-Horizon-36B
+            res = "k2-horizon"

         if res is None:
             logger.warning("\n")
diff --git a/conversion/k2_horizon.py b/conversion/k2_horizon.py
new file mode 100644
index 000000000..8122005f5
--- /dev/null
+++ b/conversion/k2_horizon.py
@@ -0,0 +1,105 @@
+from __future__ import annotations
+
+import re
+from collections.abc import Iterable
+from typing import TYPE_CHECKING
+
+import torch
+
+if TYPE_CHECKING:
+    from torch import Tensor
+
+from .base import ModelBase, TextModel, gguf
+
+
+@ModelBase.register("K2HorizonForCausalLM")
+@ModelBase.example("IFM/K2-Horizon-0.9B", "IFM/K2-Horizon-36B")
+class K2HorizonModel(TextModel):
+    model_arch = gguf.MODEL_ARCH.K2HORIZON
+
+    _experts: list[dict[str, Tensor]] | None = None
+
+    def set_gguf_parameters(self):
+        super().set_gguf_parameters()
+        hparams = self.hparams
+
+        self.gguf_writer.add_group_norm_groups(int(hparams.get("layernorm_num_groups", 1)))
+        if (rope_head_dim := hparams.get("rope_head_dim")) is not None:
+            self.gguf_writer.add_rope_dimension_count(int(rope_head_dim))
+
+        if int(hparams.get("num_experts", 0)) > 0:
+            n_ff_exp = int(hparams["moe_intermediate_size"])
+            n_shared = int(hparams.get("num_shared_experts", 0))
+
+            # the leading dense layers are the prefix of mlp_only_layers, unless given explicitly
+            n_dense = hparams.get("num_dense_layers")
+            if n_dense is None:
+                mlp_only_layers = {int(il) for il in hparams.get("mlp_only_layers", [])}
+                n_dense = 0
+                while n_dense in mlp_only_layers:
+                    n_dense += 1
+
+            self.gguf_writer.add_expert_feed_forward_length(n_ff_exp)
+            self.gguf_writer.add_leading_dense_block_count(n_dense)
+            self.gguf_writer.add_moe_every_n_layers(int(hparams.get("decoder_sparse_step", 1)))
+            self.gguf_writer.add_expert_shared_count(n_shared)
+            self.gguf_writer.add_expert_weights_norm(bool(hparams.get("norm_topk_prob", False)))
+            if n_shared > 0:
+                self.gguf_writer.add_expert_shared_feed_forward_length(n_ff_exp * n_shared)
+            if (router_scale := hparams.get("router_scaling_factor")) is not None:
+                self.gguf_writer.add_expert_weights_scale(float(router_scale))
+
+        # MoVA
+        n_value_expert      = int(hparams.get("mova_num_experts", 0))
+        n_value_expert_used = int(hparams.get("mova_num_experts_per_tok", 0))
+        if n_value_expert > 0 and n_value_expert_used > 0:
+            assert n_value_expert_used <= n_value_expert
+            self.gguf_writer.add_attention_value_expert_count(n_value_expert)
+            self.gguf_writer.add_attention_value_expert_used_count(n_value_expert_used)
+
+        if (gate_func := hparams.get("attention_gate_func")) not in (None, "softplus"):
+            raise ValueError(f"Unsupported attention_gate_func: {gate_func!r}")
+
+    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
+        # the MoE router bias only selects experts
+        if name.endswith(".mlp.gate.bias"):
+            assert bid is not None
+            yield self.format_tensor_name(gguf.MODEL_TENSOR.FFN_EXP_PROBS_B, bid, ".bias"), data_torch
+            return
+
+        if re.fullmatch(r"model\.layers\.\d+\.mlp\.experts\.\d+\.(down|gate|up)_proj\.weight", name):
+            yield from self._stack_experts(data_torch, name, bid, int(self.hparams["num_experts"]),
+                                           "model.layers.{bid}.mlp.experts.{xid}.{w}.weight", ("down_proj", "gate_proj", "up_proj"))
+            return
+
+        if re.fullmatch(r"model\.layers\.\d+\.self_attn\.v_experts\.\d+\.weight", name):
+            yield from self._stack_experts(data_torch, name, bid, int(self.hparams["mova_num_experts"]),
+                                           "model.layers.{bid}.self_attn.v_experts.{xid}{w}.weight", ("",))
+            return
+
+        yield from super().modify_tensors(data_torch, name, bid)
+
+    # collect the per-expert weights of a layer, then emit one stacked 3D tensor per projection
+    def _stack_experts(self, data_torch: Tensor, name: str, bid: int | None, n_experts: int,
+                       fmt: str, projs: tuple[str, ...]) -> Iterable[tuple[str, Tensor]]:
+        assert bid is not None
+        if self._experts is None:
+            self._experts = [{} for _ in range(self.block_count)]
+        self._experts[bid][name] = data_torch
+
+        names = {w: [fmt.format(bid=bid, xid=xid, w=w) for xid in range(n_experts)] for w in projs}
+        if not all(n in self._experts[bid] for ns in names.values() for n in ns):
+            return
+
+        for w, ns in names.items():
+            merged = torch.stack([self._experts[bid].pop(n) for n in ns], dim=0)
+            yield from super().modify_tensors(merged, fmt.replace(".{xid}", "").format(bid=bid, w=w), bid)
+
+    def prepare_tensors(self):
+        super().prepare_tensors()
+
+        if self._experts is not None:
+            # flatten the list of dicts
+            experts = [k for d in self._experts for k in d.keys()]
+            if len(experts) > 0:
+                raise ValueError(f"Unprocessed experts: {experts}")
diff --git a/convert_hf_to_gguf_update.py b/convert_hf_to_gguf_update.py
index d39d2f6fe..0afbcc7e9 100755
--- a/convert_hf_to_gguf_update.py
+++ b/convert_hf_to_gguf_update.py
@@ -165,6 +165,7 @@ models = [
     {"name": "laguna",           "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/poolside/Laguna-XS.2", },
     {"name": "ufakzeka",         "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/ufakai/ufakzeka-1", },
     {"name": "mmbert",           "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/jhu-clsp/mmBERT-base", },
+    {"name": "k2-horizon",       "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/IFM/K2-Horizon-36B", },
 ]

 # some models are known to be broken upstream, so we will skip them as exceptions
@@ -198,6 +199,8 @@ pre_computed_hashes = [
     # no-op here); the gemma4 pre (escape ws, split on newlines only) matches it.
     {"name": "gemma4", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/danish-foundation-models/DFM-Mimir", "chkhsh": "846deafc5b0fa786186fa4ae6c7b49903cf2f1d1895bdb80b9120d60be135252"},
     {"name": "spark2_5", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/XHToken/Spark-X2.5-1.7B", "chkhsh": "0a766d034107bc736a3f2dc4968fd62e54a3570f1454443e0c5a4cc6bd7941ed"},
+    # k2-horizon variants
+    {"name": "k2-horizon", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/IFM/K2-Horizon-0.9B", "chkhsh": "1f9825a388f700a6b591722f17d470cbbcf10973ece35d2fd14239a14110ae1a"},
 ]


diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py
index ea7ddfa57..9434e4120 100644
--- a/gguf-py/gguf/constants.py
+++ b/gguf-py/gguf/constants.py
@@ -222,6 +222,8 @@ class Keys:
         RECURRENT_LAYERS             = "{arch}.attention.recurrent_layers"
         TEMPERATURE_SCALE            = "{arch}.attention.temperature_scale"
         ROPE_PATTERN                 = "{arch}.attention.rope_pattern"
+        VALUE_EXPERT_COUNT           = "{arch}.attention.value_expert_count"
+        VALUE_EXPERT_USED_COUNT      = "{arch}.attention.value_expert_used_count"

         class Indexer:
             HEAD_COUNT = "{arch}.attention.indexer.head_count"
@@ -658,6 +660,7 @@ class MODEL_ARCH(IntEnum):
     NANBEIGE         = auto()
     QWEN3TTS         = auto()
     POCKETTTS        = auto()
+    K2HORIZON        = auto()


 class VISION_PROJECTOR_TYPE(IntEnum):
@@ -940,6 +943,8 @@ class MODEL_TENSOR(IntEnum):
     INDEXER_COMPRESSOR_NORM = auto()
     INDEXER_KPOOL_GATE   = auto()
     INDEXER_KPOOL_APE    = auto()
+    ATTN_V_GATE          = auto() # k2-horizon MoVA router
+    ATTN_V_EXP           = auto() # k2-horizon MoVA value experts
     # vision
     V_MMPROJ             = auto()
     V_MMPROJ_FC          = auto()
@@ -1439,6 +1444,7 @@ MODEL_ARCH_NAMES: dict[MODEL_ARCH, str] = {
     MODEL_ARCH.NANBEIGE:         "nanbeige",
     MODEL_ARCH.QWEN3TTS:         "qwen3tts",
     MODEL_ARCH.POCKETTTS:        "pockettts",
+    MODEL_ARCH.K2HORIZON:        "k2-horizon",
 }

 VISION_PROJECTOR_TYPE_NAMES: dict[VISION_PROJECTOR_TYPE, str] = {
@@ -2054,6 +2060,8 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
     MODEL_TENSOR.DFLASH_SELECTOR_NEXT:      "selector_successor",
     MODEL_TENSOR.DFLASH_SELECTOR_HIDDEN:    "selector_hidden",
     MODEL_TENSOR.D2T:                       "d2t",
+    MODEL_TENSOR.ATTN_V_GATE:               "blk.{bid}.attn_v_gate",
+    MODEL_TENSOR.ATTN_V_EXP:                "blk.{bid}.attn_v_exps",
 }

 MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
@@ -5789,6 +5797,33 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
         MODEL_TENSOR.FFN_DOWN,
         MODEL_TENSOR.FFN_UP,
     ],
+    MODEL_ARCH.K2HORIZON: [
+        MODEL_TENSOR.TOKEN_EMBD,
+        MODEL_TENSOR.OUTPUT_NORM,
+        MODEL_TENSOR.OUTPUT,
+        MODEL_TENSOR.ATTN_NORM,
+        MODEL_TENSOR.ATTN_Q,
+        MODEL_TENSOR.ATTN_Q_NORM,
+        MODEL_TENSOR.ATTN_K,
+        MODEL_TENSOR.ATTN_K_NORM,
+        MODEL_TENSOR.ATTN_V,
+        MODEL_TENSOR.ATTN_V_GATE,
+        MODEL_TENSOR.ATTN_V_EXP,
+        MODEL_TENSOR.ATTN_OUT,
+        MODEL_TENSOR.ATTN_GATE,
+        MODEL_TENSOR.FFN_NORM,
+        MODEL_TENSOR.FFN_GATE,
+        MODEL_TENSOR.FFN_UP,
+        MODEL_TENSOR.FFN_DOWN,
+        MODEL_TENSOR.FFN_GATE_INP,
+        MODEL_TENSOR.FFN_EXP_PROBS_B,
+        MODEL_TENSOR.FFN_GATE_EXP,
+        MODEL_TENSOR.FFN_UP_EXP,
+        MODEL_TENSOR.FFN_DOWN_EXP,
+        MODEL_TENSOR.FFN_GATE_SHEXP,
+        MODEL_TENSOR.FFN_UP_SHEXP,
+        MODEL_TENSOR.FFN_DOWN_SHEXP,
+    ],
 }

 # tensors that will not be serialized
diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py
index 4f1f8e5a5..e814691d7 100644
--- a/gguf-py/gguf/gguf_writer.py
+++ b/gguf-py/gguf/gguf_writer.py
@@ -1621,6 +1621,12 @@ class GGUFWriter:
     def add_xielu_eps(self, values: Sequence[float]):
         self.add_array(Keys.xIELU.EPS, values)

+    def add_attention_value_expert_count(self, count: int):
+        self.add_uint32(Keys.Attention.VALUE_EXPERT_COUNT.format(arch=self.arch), count)
+
+    def add_attention_value_expert_used_count(self, count: int):
+        self.add_uint32(Keys.Attention.VALUE_EXPERT_USED_COUNT.format(arch=self.arch), count)
+
     # diffusion models

     def add_diffusion_shift_logits(self, value: bool) -> None:
diff --git a/gguf-py/gguf/tensor_mapping.py b/gguf-py/gguf/tensor_mapping.py
index 5ac11cb46..156ac3a50 100644
--- a/gguf-py/gguf/tensor_mapping.py
+++ b/gguf-py/gguf/tensor_mapping.py
@@ -400,6 +400,7 @@ class TensorNameMap:
             "model.layers.{bid}.self_attn.g_proj",    # step3.5 head-wise attention gate
             "model.layers.{bid}.self_attn.output_gate",  # minimax-01
             "model.layers.{bid}.self_attn.linear_gate",  # hy-v4
+            "model.layers.{bid}.self_attn.attn_gate_proj",  # k2-horizon
         ),

         # Feed-forward norm
@@ -2862,6 +2863,14 @@ class TensorNameMap:
         MODEL_TENSOR.NEXTN_SHARED_HEAD_NORM: (
             "model.layers.{bid}.shared_head.norm",
         ),
+
+        MODEL_TENSOR.ATTN_V_GATE: (
+            "model.layers.{bid}.self_attn.v_router",  # k2-horizon
+        ),
+
+        MODEL_TENSOR.ATTN_V_EXP: (
+            "model.layers.{bid}.self_attn.v_experts",  # k2-horizon
+        ),
     }

     # architecture-specific block mappings
diff --git a/models/templates/IFM-K2-Horizon.jinja b/models/templates/IFM-K2-Horizon.jinja
new file mode 100644
index 000000000..41dde1f4b
--- /dev/null
+++ b/models/templates/IFM-K2-Horizon.jinja
@@ -0,0 +1,994 @@
+{{- bos_token }}
+{%- if tool_presentation is defined -%}
+    {{- raise_exception("Unsupported argument: tool_presentation. Use tool_presentation_format with one of: json, xml, markdown.") -}}
+{%- endif -%}
+{%- if tool_calling_format is defined -%}
+    {{- raise_exception("Unsupported argument: tool_calling_format. Use tool_call_format with one of: json, xml, xml_typed.") -}}
+{%- endif -%}
+{%- if tool_format is defined -%}
+    {{- raise_exception("Unsupported argument: tool_format. Use tool_call_format with one of: json, xml, xml_typed.") -}}
+{%- endif -%}
+{%- set tool_presentation_fmt = tool_presentation_format | default('markdown') -%}
+{%- set tool_call_fmt = tool_call_format | default('xml') -%}
+{%- if tool_presentation_fmt != 'json' and tool_presentation_fmt != 'xml' and tool_presentation_fmt != 'markdown' -%}
+    {{- raise_exception("Unsupported tool_presentation_format: '" ~ tool_presentation_fmt ~ "'. Supported formats: json, xml, markdown.") -}}
+{%- endif -%}
+{%- if tool_call_fmt != 'json' and tool_call_fmt != 'xml' and tool_call_fmt != 'xml_typed' -%}
+    {{- raise_exception("Unsupported tool_call_format: '" ~ tool_call_fmt ~ "'. Supported formats: json, xml, xml_typed.") -}}
+{%- endif -%}
+
+{#- Renderability state, computed during validate_tools (single walk, no extra -#}
+{#- traversal at render time): ok = working flag for the tool being validated; -#}
+{#- bad = pipe-delimited indices of tools that must render as verbatim JSON. -#}
+{%- set RB = namespace(ok=true, bad='|') -%}
+
+{%- macro value_contains_mapping(v) -%}
+{%- if v is mapping -%}
+true
+{%- elif v is sequence and v is not string -%}
+{%- set f = namespace(x='false') -%}
+{%- for c in v -%}{%- if value_contains_mapping(c) == 'true' -%}{%- set f.x = 'true' -%}{%- endif -%}{%- endfor -%}
+{{- f.x -}}
+{%- else -%}
+false
+{%- endif -%}
+{%- endmacro -%}
+
+{#- $ref inlining state: defs = local $defs of the tool being rendered; seen = -#}
+{#- pipe-delimited names already expanded for this tool (each def inlines at most -#}
+{#- once; later references render by def name; cycles terminate immediately). -#}
+{#- $ref-sibling annotations (description/default/...) merge OVER the def at -#}
+{#- the inline site, so use-site annotations win and are never dropped. -#}
+{%- set REFS = namespace(defs={}, seen='|') -%}
+
+{%- macro render_compact_type_name(type_name, spec) -%}
+{%- if type_name == "array" -%}
+array[{%- if 'items' in spec -%}{{ render_compact_type(spec['items']) }}{%- else -%}any{%- endif -%}]
+{%- elif type_name -%}
+{{- type_name -}}
+{%- else -%}
+any
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_compact_type(spec) -%}
+{%- if spec is not mapping -%}
+any
+{%- elif spec.type is defined and spec.type is sequence and spec.type is not string and spec.type | length > 0 -%}
+{%- for type_name in spec.type -%}{{ render_compact_type_name(type_name, spec) }}{%- if not loop.last -%}|{%- endif -%}{%- endfor -%}
+{%- elif spec.type is defined and spec.type is sequence and spec.type is not string -%}
+any
+{%- elif spec.type -%}
+{{- render_compact_type_name(spec.type, spec) -}}
+{%- elif spec['$ref'] is string -%}
+{{- spec['$ref'].split('/') | last -}}
+{%- elif spec.oneOf -%}
+oneOf[{%- for variant in spec.oneOf -%}{{ render_compact_type(variant) }}{%- if not loop.last -%}|{%- endif -%}{%- endfor -%}]
+{%- elif spec.anyOf -%}
+anyOf[{%- for variant in spec.anyOf -%}{{ render_compact_type(variant) }}{%- if not loop.last -%}|{%- endif -%}{%- endfor -%}]
+{%- elif spec.properties -%}
+object
+{%- elif 'items' in spec -%}
+array[{{ render_compact_type(spec['items']) }}]
+{%- else -%}
+any
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_type_name(type_name, spec) -%}
+{%- if type_name == "array" -%}
+array of {% if 'items' in spec %}{{ render_markdown_type(spec['items']) }}{% else %}any{% endif %}
+{%- elif type_name -%}
+{{- type_name -}}
+{%- else -%}
+any
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_type(spec) -%}
+{%- if spec is sameas true -%}
+True
+{%- elif spec is sameas false -%}
+False
+{%- elif spec is not mapping -%}
+any
+{%- elif spec.type is defined and spec.type is sequence and spec.type is not string and spec.type | length > 0 -%}
+{%- for type_name in spec.type -%}{{ render_markdown_type_name(type_name, spec) }}{% if not loop.last %} or {% endif %}{%- endfor -%}
+{%- elif spec.type is defined and spec.type is sequence and spec.type is not string -%}
+any
+{%- elif spec.type -%}
+{{- render_markdown_type_name(spec.type, spec) -}}
+{%- elif spec['$ref'] is string -%}
+{{- spec['$ref'].split('/') | last -}}
+{%- elif spec.oneOf -%}
+oneOf[{%- for variant in spec.oneOf -%}{{ render_markdown_type(variant) }}{% if not loop.last %} or {% endif %}{%- endfor -%}]
+{%- elif spec.anyOf -%}
+anyOf[{%- for variant in spec.anyOf -%}{{ render_markdown_type(variant) }}{% if not loop.last %} or {% endif %}{%- endfor -%}]
+{%- elif spec.properties -%}
+object
+{%- elif 'items' in spec -%}
+array of {{ render_markdown_type(spec['items']) }}
+{%- else -%}
+any
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_xml_text(value) -%}
+{{- value.split() | join(" ") -}}
+{%- endmacro -%}
+
+{%- macro render_python_string(value) -%}
+'{{- value.split() | join(" ") | replace("\\", "\\\\") | replace("'", "\\'") -}}'
+{%- endmacro -%}
+
+{%- macro render_python_repr(value) -%}
+{%- if value is string -%}
+{{ render_python_string(value) }}
+{%- elif value is sameas true -%}
+True
+{%- elif value is sameas false -%}
+False
+{%- elif value is none -%}
+None
+{%- elif value is mapping -%}
+{{- "{" -}}
+{%- for key, child in value | items -%}
+{{ render_python_repr(key) }}: {{ render_python_repr(child) }}{%- if not loop.last -%}, {% endif -%}
+{%- endfor -%}
+{{- "}" -}}
+{%- elif value is sequence -%}
+{{- "[" -}}
+{%- for child in value -%}
+{{ render_python_repr(child) }}{%- if not loop.last -%}, {% endif -%}
+{%- endfor -%}
+{{- "]" -}}
+{%- else -%}
+{{- value -}}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_xml_value(value) -%}
+{%- if value is string -%}{{ render_xml_text(value) }}{%- else -%}{{ render_python_repr(value) }}{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_xml_enum_value(value) -%}
+{%- if value is string -%}"{{- value | replace("\\", "\\\\") | replace("\"", "\\\"") -}}"{%- else -%}"{{- render_python_repr(value) | replace("\\", "\\\\") | replace("\"", "\\\"") -}}"{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_xml_enum(values) -%}
+{%- for value in values -%}{{ render_xml_enum_value(value) }}{%- if not loop.last -%}|{%- endif -%}{%- endfor -%}
+{%- endmacro -%}
+
+{%- macro render_xml_default_attr(value) -%}
+{{- " default=" }}{%- if value is string -%}"{{- value | replace("\\", "\\\\") | replace("\"", "\\\"") -}}"{%- else -%}{{ render_xml_value(value) }}{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_xml_attr(name, value) -%}
+{{- " " + name + "=" }}{%- if value == "" -%}""{%- else -%}{{ render_xml_value(value) }}{%- endif -%}
+{%- endmacro -%}
+
+{%- macro validate_schema(spec, path, lenient=false, classify=true, in_variant=false) -%}
+{%- if spec is mapping -%}
+    {%- if not lenient -%}
+    {%- if spec.required is defined -%}
+        {%- if spec.required is string or spec.required is not sequence -%}
+            {{- raise_exception("Schema '" + path + "' has 'required' but it is not a list.") -}}
+        {%- endif -%}
+        {%- if spec.required | length > 0 and not spec.properties and not in_variant -%}
+            {{- raise_exception("Schema '" + path + "' has required fields but no properties object to define them.") -}}
+        {%- endif -%}
+        {%- if spec.properties -%}
+            {%- for required_name in spec.required -%}
+                {%- if required_name not in spec.properties -%}
+                    {{- raise_exception("Schema '" + path + "' marks '" + required_name + "' as required, but that property is not defined in properties.") -}}
+                {%- endif -%}
+            {%- endfor -%}
+        {%- endif -%}
+    {%- endif -%}
+    {%- endif -%}
+    {#- renderability classification, piggybacking on this walk (no raises here): -#}
+    {#- constructs the pretty renderer does not fully handle flip RB.ok so the -#}
+    {#- tool falls back to verbatim JSON. Skipped entirely for json presentation. -#}
+    {%- if classify -%}
+    {%- for key, value in spec | items -%}
+        {%- if key == '$ref' -%}
+            {%- if value is not string -%}{%- set RB.ok = false -%}
+            {%- elif not (value.startswith('#/$defs/') or value.startswith('#/definitions/')) -%}{%- set RB.ok = false -%}{%- endif -%}
+        {%- elif key == '$defs' or key == 'definitions' -%}
+            {%- if value is mapping -%}
+                {%- for dk, dv in value | items -%}
+                    {{- validate_schema(dv, path + ".$defs." + dk, true) -}}
+                {%- endfor -%}
+            {%- else -%}{%- set RB.ok = false -%}{%- endif -%}
+        {%- elif key == 'type' -%}
+            {%- if value is mapping -%}{%- set RB.ok = false -%}{%- endif -%}
+        {%- elif key == 'enum' -%}
+            {%- if value is string or value is mapping or value is not sequence -%}{%- set RB.ok = false -%}{%- endif -%}
+        {%- elif key == 'items' -%}
+            {#- any items shape renders: mapping structurally, others via repr detail -#}
+        {%- elif key == 'oneOf' or key == 'anyOf' -%}
+            {%- if value is mapping or value is string or value is not sequence -%}{%- set RB.ok = false -%}{%- endif -%}
+        {%- elif key == 'required' -%}
+            {%- if value and not spec.properties -%}{%- set RB.ok = false -%}{%- endif -%}
+        {%- elif ('|' ~ key ~ '|') in '|description|default|title|examples|properties|patternProperties|additionalProperties|returns|' -%}
+        {%- elif value is mapping -%}
+            {%- for uk, uv in value | items -%}
+                {%- if value_contains_mapping(uv) == 'true' -%}{%- set RB.ok = false -%}{%- endif -%}
+            {%- endfor -%}
+        {%- elif value is sequence and value is not string -%}
+            {%- if value_contains_mapping(value) == 'true' -%}{%- set RB.ok = false -%}{%- endif -%}
+        {%- endif -%}
+    {%- endfor -%}
+    {%- endif -%}
+    {%- if spec.properties -%}
+        {%- for child_name, child_spec in spec.properties | items -%}
+            {{- validate_schema(child_spec, path + "." + child_name, lenient, classify) -}}
+        {%- endfor -%}
+    {%- endif -%}
+    {%- if 'items' in spec -%}{{- validate_schema(spec['items'], path + "[]", lenient, classify) -}}{%- endif -%}
+    {%- if spec.oneOf -%}
+        {%- for variant in spec.oneOf -%}{{- validate_schema(variant, path + ".oneOf[" + (loop.index0 | string) + "]", lenient, classify, true) -}}{%- endfor -%}
+    {%- endif -%}
+    {%- if spec.anyOf -%}
+        {%- for variant in spec.anyOf -%}{{- validate_schema(variant, path + ".anyOf[" + (loop.index0 | string) + "]", lenient, classify, true) -}}{%- endfor -%}
+    {%- endif -%}
+    {%- if spec.additionalProperties is mapping -%}{{- validate_schema(spec.additionalProperties, path + ".additionalProperties", lenient, classify) -}}{%- endif -%}
+    {%- if spec.patternProperties is mapping -%}
+        {%- for pattern, pattern_spec in spec.patternProperties | items -%}
+            {{- validate_schema(pattern_spec, path + ".patternProperties[" + pattern + "]", lenient, classify) -}}
+        {%- endfor -%}
+    {%- endif -%}
+    {%- if spec.returns is mapping -%}{{- validate_schema(spec.returns, path + ".returns", lenient, classify) -}}{%- endif -%}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro validate_tools(tools_list, classify=true) -%}
+{%- set RB.bad = '|' -%}
+{%- for tool in tools_list -%}
+    {%- set fn = tool.function if tool.function is defined else tool -%}
+    {%- set RB.ok = true -%}
+    {%- if fn.parameters is defined and fn.parameters is string -%}
+        {{- raise_exception("tool.function.parameters must be a dict, not a JSON string. Parse it before passing to the template.") -}}
+    {%- endif -%}
+    {%- if fn.parameters is not defined or fn.parameters is none -%}
+        {%- if fn.arguments is defined -%}
+            {{- raise_exception("Tool '" + fn.name + "' has 'arguments' instead of 'parameters'. Rename 'arguments' to 'parameters'.") -}}
+        {%- else -%}
+            {{- raise_exception("Tool '" + fn.name + "' is missing required 'parameters' field. Each tool must have a 'parameters' dict with 'type', 'properties', and 'required' keys.") -}}
+        {%- endif -%}
+    {%- endif -%}
+    {{- validate_schema(fn.parameters, "tool." + fn.name + ".parameters", false, classify) -}}
+    {%- if classify -%}
+    {%- if fn.parameters is mapping -%}
+        {#- unknown container-valued keys at the parameters ROOT are never rendered -#}
+        {#- by the pretty path (root extras are dropped) -> verbatim fallback. -#}
+        {%- for rk, rv in fn.parameters | items -%}
+            {%- if rk not in ['type', 'description', 'enum', 'default', 'properties', 'required', 'optional', 'title', 'items', 'oneOf', 'anyOf', 'additionalProperties', 'patternProperties', 'returns', 'examples', '$defs', 'definitions', '$ref'] -%}
+                {%- if rv is mapping or (rv is sequence and rv is not string) -%}{%- set RB.ok = false -%}{%- endif -%}
+            {%- endif -%}
+        {%- endfor -%}
+    {%- else -%}
+        {%- set RB.ok = false -%}
+    {%- endif -%}
+    {%- endif -%}
+    {%- if fn.returns is mapping -%}{{- validate_schema(fn.returns, "tool." + fn.name + ".returns", false, classify) -}}{%- endif -%}
+    {%- if classify and fn.returns is not defined and fn.response is mapping -%}{{- validate_schema(fn.response, "tool." + fn.name + ".response", true) -}}{%- endif -%}
+    {#- unknown container-valued keys at the FUNCTION level are never rendered -> fallback. -#}
+    {%- if classify -%}
+    {%- for fk, fv in fn | items -%}
+        {%- if fk not in ['name', 'description', 'parameters', 'returns', 'response', 'type', 'function'] -%}
+            {%- if fv is mapping or (fv is sequence and fv is not string) -%}{%- set RB.ok = false -%}{%- endif -%}
+        {%- endif -%}
+    {%- endfor -%}
+    {%- endif -%}
+    {%- if not RB.ok -%}{%- set RB.bad = RB.bad ~ loop.index0 ~ '|' -%}{%- endif -%}
+{%- endfor -%}
+{%- endmacro -%}
+
+{%- macro render_tools_json(tools_list) -%}
+{{- "<ifm|tools>" }}
+{%- for tool in tools_list %}
+{{- "\n" }}
+{{- tool | tojson }}
+{%- endfor %}
+{{- "\n</ifm|tools>" }}
+{%- endmacro -%}
+
+{%- macro render_xml_schema_attrs(spec, include_value_attrs) -%}
+{%- if spec is mapping -%}
+{%- set structural_keys = ["type", "description", "enum", "default", "properties", "required", "items", "oneOf", "anyOf", "additionalProperties", "patternProperties", "returns"] -%}
+{%- if include_value_attrs and spec.enum -%}{{- " enum=" }}{{ render_xml_enum(spec.enum) }}{%- endif -%}
+{%- if include_value_attrs and spec.default is defined -%}{{ render_xml_default_attr(spec.default) }}{%- endif -%}
+{%- if spec.additionalProperties is defined and spec.additionalProperties is not mapping -%}{{ render_xml_attr("additionalProperties", spec.additionalProperties) }}{%- endif -%}
+{%- if spec.patternProperties is defined and spec.patternProperties is not mapping -%}{{ render_xml_attr("patternProperties", spec.patternProperties) }}{%- endif -%}
+{%- for key, value in spec | items -%}
+    {%- if key not in structural_keys -%}
+{{ render_xml_attr(key, value) }}
+    {%- endif -%}
+{%- endfor -%}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro xml_schema_has_children(spec, include_properties, include_description) -%}
+{%- if spec is not mapping -%}
+false
+{%- elif (include_description and spec.description is defined) or (include_properties and spec.properties) or 'items' in spec or spec.oneOf or spec.anyOf or spec.additionalProperties is mapping or spec.patternProperties is mapping or spec.returns is defined -%}
+true
+{%- else -%}
+false
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_xml_schema_node(tag, spec, include_properties) -%}
+{%- if spec is mapping and spec['$ref'] is string -%}
+    {%- set _r = spec['$ref'] -%}
+    {%- set _k = _r[8:] if _r.startswith('#/$defs/') else (_r[14:] if _r.startswith('#/definitions/') else none) -%}
+    {%- if _k is not none and ('|' + _k + '|') not in REFS.seen and REFS.defs[_k] is mapping -%}
+        {%- set spec = dict((REFS.defs[_k] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+        {%- set REFS.seen = REFS.seen + _k + '|' -%}
+        {%- if spec['$ref'] is string -%}
+            {%- set _r2 = spec['$ref'] -%}
+            {%- set _k2 = _r2[8:] if _r2.startswith('#/$defs/') else (_r2[14:] if _r2.startswith('#/definitions/') else none) -%}
+            {%- if _k2 is not none and REFS.defs[_k2] is mapping -%}
+                {%- set spec = dict((REFS.defs[_k2] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+                {%- set REFS.seen = REFS.seen + _k2 + '|' -%}
+            {%- endif -%}
+        {%- endif -%}
+    {%- endif -%}
+{%- endif -%}
+{%- if spec is mapping -%}
+{{- "<" + tag + " type=" + render_compact_type(spec) }}{{ render_xml_schema_attrs(spec, true) }}
+{%- if xml_schema_has_children(spec, include_properties, true) == 'true' -%}
+{{- ">" }}{{ render_xml_schema_children(spec, include_properties, true) }}{{- "</" + tag + ">" }}
+{%- else -%}
+{{- "/>" }}
+{%- endif -%}
+{%- else -%}
+{{- "<" + tag + ">" }}{{ render_xml_value(spec) }}{{- "</" + tag + ">" }}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_xml_pattern_property(pattern, spec) -%}
+{%- if spec is mapping and spec['$ref'] is string -%}
+    {%- set _r = spec['$ref'] -%}
+    {%- set _k = _r[8:] if _r.startswith('#/$defs/') else (_r[14:] if _r.startswith('#/definitions/') else none) -%}
+    {%- if _k is not none and ('|' + _k + '|') not in REFS.seen and REFS.defs[_k] is mapping -%}
+        {%- set spec = dict((REFS.defs[_k] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+        {%- set REFS.seen = REFS.seen + _k + '|' -%}
+        {%- if spec['$ref'] is string -%}
+            {%- set _r2 = spec['$ref'] -%}
+            {%- set _k2 = _r2[8:] if _r2.startswith('#/$defs/') else (_r2[14:] if _r2.startswith('#/definitions/') else none) -%}
+            {%- if _k2 is not none and REFS.defs[_k2] is mapping -%}
+                {%- set spec = dict((REFS.defs[_k2] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+                {%- set REFS.seen = REFS.seen + _k2 + '|' -%}
+            {%- endif -%}
+        {%- endif -%}
+    {%- endif -%}
+{%- endif -%}
+{%- if spec is mapping -%}
+{{- "<patternProperty" }}{{ render_xml_attr("pattern", pattern) }}{{- " type=" + render_compact_type(spec) }}{{ render_xml_schema_attrs(spec, true) }}
+{%- if xml_schema_has_children(spec, true, true) == 'true' -%}
+{{- ">" }}{{ render_xml_schema_children(spec, true, true) }}{{- "</patternProperty>" }}
+{%- else -%}
+{{- "/>" }}
+{%- endif -%}
+{%- else -%}
+{{- "<patternProperty" }}{{ render_xml_attr("pattern", pattern) }}{{- ">" }}{{ render_xml_value(spec) }}{{- "</patternProperty>" }}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_xml_schema_children(spec, include_properties, include_description) -%}
+{%- if include_description and spec.description is defined -%}{{- "<description>" }}{{ spec.description }}{{- "</description>" }}{%- endif -%}
+{%- if include_properties and spec.properties -%}
+{%- for child_name, child_spec in spec.properties | items -%}
+{{- render_xml_param(child_name, child_spec, spec.required or []) }}
+{%- endfor -%}
+{%- endif -%}
+{%- if 'items' in spec -%}{{ render_xml_schema_node("items", spec['items'], true) }}{%- endif -%}
+{%- if spec.oneOf -%}
+{{- "<oneOf>" }}
+{%- for variant in spec.oneOf -%}{{ render_xml_schema_node("variant", variant, true) }}{%- endfor -%}
+{{- "</oneOf>" }}
+{%- endif -%}
+{%- if spec.anyOf -%}
+{{- "<anyOf>" }}
+{%- for variant in spec.anyOf -%}{{ render_xml_schema_node("variant", variant, true) }}{%- endfor -%}
+{{- "</anyOf>" }}
+{%- endif -%}
+{%- if spec.additionalProperties is mapping -%}{{ render_xml_schema_node("additionalProperties", spec.additionalProperties, true) }}{%- endif -%}
+{%- if spec.patternProperties is mapping -%}
+{{- "<patternProperties>" }}
+{%- for pattern, pattern_spec in spec.patternProperties | items -%}{{ render_xml_pattern_property(pattern, pattern_spec) }}{%- endfor -%}
+{{- "</patternProperties>" }}
+{%- elif spec.patternProperties is defined -%}<patternProperties>{{ render_xml_value(spec.patternProperties) }}</patternProperties>{%- endif -%}
+{%- if spec.returns is mapping -%}{{ render_xml_schema_node("returns", spec.returns, true) }}{%- elif spec.returns is defined -%}<returns>{{ render_xml_value(spec.returns) }}</returns>{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_xml_param(name, spec, required_list) -%}
+{%- if spec is mapping and spec['$ref'] is string -%}
+    {%- set _r = spec['$ref'] -%}
+    {%- set _k = _r[8:] if _r.startswith('#/$defs/') else (_r[14:] if _r.startswith('#/definitions/') else none) -%}
+    {%- if _k is not none and ('|' + _k + '|') not in REFS.seen and REFS.defs[_k] is mapping -%}
+        {%- set spec = dict((REFS.defs[_k] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+        {%- set REFS.seen = REFS.seen + _k + '|' -%}
+        {%- if spec['$ref'] is string -%}
+            {%- set _r2 = spec['$ref'] -%}
+            {%- set _k2 = _r2[8:] if _r2.startswith('#/$defs/') else (_r2[14:] if _r2.startswith('#/definitions/') else none) -%}
+            {%- if _k2 is not none and REFS.defs[_k2] is mapping -%}
+                {%- set spec = dict((REFS.defs[_k2] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+                {%- set REFS.seen = REFS.seen + _k2 + '|' -%}
+            {%- endif -%}
+        {%- endif -%}
+    {%- endif -%}
+{%- endif -%}
+{{- "<param name=" + name + " type=" + render_compact_type(spec) }}
+{%- if name in (required_list or []) -%}{{- " required=true" }}{%- endif -%}
+{%- if spec.enum -%}{{- " enum=" }}{{ render_xml_enum(spec.enum) }}{%- endif -%}
+{%- if spec.default is defined -%}{{ render_xml_default_attr(spec.default) }}{%- endif -%}
+{{- render_xml_schema_attrs(spec, false) }}
+{%- if spec.description or xml_schema_has_children(spec, true, false) == 'true' -%}
+{{- ">" }}
+{%- if spec.description -%}{{ spec.description }}{%- endif -%}
+{{- render_xml_schema_children(spec, true, false) }}
+{{- "</param>" }}
+{%- else -%}
+{{- "/>" }}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_tools_xml(tools_list) -%}
+{{- "<ifm|tools>" }}
+{%- for tool in tools_list -%}
+    {%- set fn = tool.function if tool.function is defined else tool -%}
+    {%- set REFS.defs = fn.parameters['$defs'] if (fn.parameters is mapping and fn.parameters['$defs'] is mapping) else (fn.parameters['definitions'] if (fn.parameters is mapping and fn.parameters['definitions'] is mapping) else {}) -%}
+    {%- set REFS.seen = '|' -%}
+    {%- set fnp = namespace(p=fn.parameters) -%}
+    {%- if fnp.p is mapping and fnp.p['$ref'] is string -%}
+        {%- set _r = fnp.p['$ref'] -%}
+        {%- set _k = _r[8:] if _r.startswith('#/$defs/') else (_r[14:] if _r.startswith('#/definitions/') else none) -%}
+        {%- if _k is not none and REFS.defs[_k] is mapping -%}
+            {%- set fnp.p = dict((REFS.defs[_k] | items | list) + (fnp.p | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+            {%- set REFS.seen = REFS.seen + _k + '|' -%}
+        {%- endif -%}
+    {%- endif -%}
+{{- "\n<function name=" + fn.name + ">" }}
+{%- if fn.description -%}
+{{- "<description>" }}{{ fn.description }}{{- "</description>" }}
+{%- endif -%}
+{{- "<parameters>" }}
+{%- if fnp.p and fnp.p.properties -%}
+    {%- for pname, pspec in fnp.p.properties | items -%}
+{{- render_xml_param(pname, pspec, fnp.p.required or []) }}
+    {%- endfor -%}
+{%- elif fnp.p is mapping and (fnp.p.oneOf or fnp.p.anyOf or 'items' in fnp.p) -%}
+{{- render_xml_schema_children(fnp.p, true, false) }}
+{%- endif -%}
+{{- "</parameters>" }}
+{%- set fn_ret = fn.returns if fn.returns is defined else fn.response -%}
+{%- if fn_ret is mapping -%}{{ render_xml_schema_node("returns", fn_ret, true) }}{%- elif fn_ret is defined -%}<returns>{{ render_xml_value(fn_ret) }}</returns>{%- endif -%}
+{{- "</function>" }}
+{%- endfor -%}
+{{- "\n</ifm|tools>" }}
+{%- endmacro -%}
+
+{%- macro render_markdown_literal(value) -%}
+{%- if value is string and value == "" -%}""
+{%- elif value is string -%}`{{ value | replace("\n", "\\n") }}`
+{%- else -%}`{{ render_python_repr(value) }}`
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_allowed_values(values) -%}
+{%- for value in values -%}{{ render_markdown_literal(value) }}{% if not loop.last %}, {% endif %}{%- endfor -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_value(value) -%}
+{%- if value is string and value == "" -%}""{%- elif value is string -%}{{ value }}{%- else -%}{{ render_python_repr(value) }}{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_detail(indent, label, value) -%}
+{{- "\n" + indent + "  - " + label + ": " }}{{ render_markdown_value(value) }}
+{%- endmacro -%}
+
+{%- macro render_markdown_metadata_detail(label, value) -%}
+{{- "\n- " + label + ": " }}{{ render_markdown_value(value) }}
+{%- endmacro -%}
+
+{%- macro render_markdown_schema_annotations(spec, indent, include_value_details) -%}
+{%- if include_value_details and spec.description is defined -%}{{ render_markdown_detail(indent, "Description", spec.description | replace("\n", "\n" + indent + "    ")) }}{%- endif -%}
+{%- if include_value_details and spec.enum is defined -%}{{- "\n" + indent + "  - Allowed values: " }}{{ render_allowed_values(spec.enum) }}{%- endif -%}
+{%- if include_value_details and spec.default is defined -%}{{- "\n" + indent + "  - Default: " }}{{ render_markdown_literal(spec.default) }}{%- endif -%}
+{%- if spec.additionalProperties is defined -%}
+    {%- if spec.additionalProperties is mapping -%}
+{{- "\n" + indent + "  - Additional properties *(" + render_markdown_type(spec.additionalProperties) + ")*" }}
+{{- render_markdown_schema_details(spec.additionalProperties, indent + "  ", true) }}
+    {%- else -%}
+{{ render_markdown_detail(indent, "Additional properties", spec.additionalProperties) }}
+    {%- endif -%}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_metadata_annotations(spec) -%}
+{%- if spec.description is defined -%}{{ render_markdown_metadata_detail("Description", spec.description | replace("\n", "\n    ")) }}{%- endif -%}
+{%- if spec.enum is defined -%}{{- "\n- Allowed values: " }}{{ render_allowed_values(spec.enum) }}{%- endif -%}
+{%- if spec.default is defined -%}{{- "\n- Default: " }}{{ render_markdown_literal(spec.default) }}{%- endif -%}
+{%- if spec.additionalProperties is defined -%}
+    {%- if spec.additionalProperties is mapping -%}
+{{- "\n- Additional properties *(" + render_markdown_type(spec.additionalProperties) + ")*" }}
+{{- render_markdown_schema_details(spec.additionalProperties, "", true) }}
+    {%- else -%}
+{{ render_markdown_metadata_detail("Additional properties", spec.additionalProperties) }}
+    {%- endif -%}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_schema_extras(spec, indent) -%}
+{%- set rendered_keys = ["type", "description", "enum", "default", "properties", "required", "items", "oneOf", "anyOf", "additionalProperties", "patternProperties", "returns"] -%}
+{%- for key, value in spec | items -%}
+    {%- if key not in rendered_keys -%}
+{{- "\n" + indent + "  - " + key + ": " }}{{ render_markdown_value(value) }}
+    {%- endif -%}
+{%- endfor -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_metadata_extras(spec) -%}
+{%- set rendered_keys = ["type", "description", "enum", "default", "properties", "required", "items", "oneOf", "anyOf", "additionalProperties", "patternProperties", "returns"] -%}
+{%- for key, value in spec | items -%}
+    {%- if key not in rendered_keys -%}
+{{- "\n- " + key + ": " }}{{ render_markdown_value(value) }}
+    {%- endif -%}
+{%- endfor -%}
+{%- endmacro -%}
+
+{%- macro markdown_schema_has_extra(spec) -%}
+{%- set rendered_keys = ["type", "description", "enum", "default", "properties", "required", "items", "oneOf", "anyOf", "additionalProperties", "patternProperties", "returns"] -%}
+{%- set found = namespace(value='false') -%}
+{%- for key, value in spec | items -%}
+    {%- if key not in rendered_keys -%}{%- set found.value = 'true' -%}{%- endif -%}
+{%- endfor -%}
+{{- found.value -}}
+{%- endmacro -%}
+
+{%- macro markdown_parameter_schema_has_details(spec) -%}
+{%- if spec.description is defined or spec.enum is defined or spec.default is defined or spec.additionalProperties is defined or spec.patternProperties is defined or 'items' in spec or spec.oneOf or spec.anyOf or spec.returns is defined or markdown_schema_has_extra(spec) == 'true' -%}
+true
+{%- else -%}
+false
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_schema_structure(spec, indent, include_properties) -%}
+{%- if include_properties and spec.properties -%}
+    {%- for child_name, child_spec in spec.properties | items -%}
+{{- render_markdown_param(child_name, child_spec, spec.required or [], indent + "  ") }}
+    {%- endfor -%}
+{%- endif -%}
+{%- if 'items' in spec and spec['items'] is mapping -%}
+{{- "\n" + indent + "  - Items *(" + render_markdown_type(spec['items']) + ")*" }}
+{{- render_markdown_schema_details(spec['items'], indent + "  ", true) }}
+{%- elif 'items' in spec -%}
+{{ render_markdown_detail(indent, "Items", spec['items']) }}
+{%- endif -%}
+{%- if spec.oneOf -%}
+{{- "\n" + indent + "  - oneOf:" }}
+    {%- for variant in spec.oneOf -%}
+{{- "\n" + indent + "    - Variant " }}{{ loop.index }}{{- " *(" + render_markdown_type(variant) + ")*" }}
+{{- render_markdown_schema_details(variant, indent + "    ", true) }}
+    {%- endfor -%}
+{%- endif -%}
+{%- if spec.anyOf -%}
+{{- "\n" + indent + "  - anyOf:" }}
+    {%- for variant in spec.anyOf -%}
+{{- "\n" + indent + "    - Variant " }}{{ loop.index }}{{- " *(" + render_markdown_type(variant) + ")*" }}
+{{- render_markdown_schema_details(variant, indent + "    ", true) }}
+    {%- endfor -%}
+{%- endif -%}
+{%- if spec.patternProperties is mapping -%}
+{{- "\n" + indent + "  - Pattern properties:" }}
+    {%- for pattern, pattern_spec in spec.patternProperties | items -%}
+        {%- if pattern_spec is mapping -%}
+{{- "\n" + indent + "    - `" + pattern + "` *(" + render_markdown_type(pattern_spec) + ")*" }}
+{{- render_markdown_schema_details(pattern_spec, indent + "    ", true) }}
+        {%- else -%}
+{{- "\n" + indent + "    - `" + pattern + "`: " }}{{ render_markdown_value(pattern_spec) }}
+        {%- endif -%}
+    {%- endfor -%}
+{%- elif spec.patternProperties is defined -%}
+{{ render_markdown_detail(indent, "Pattern properties", spec.patternProperties) }}
+{%- endif -%}
+{%- if spec.returns is mapping -%}
+{{- "\n" + indent + "  - Returns *(" + render_markdown_type(spec.returns) + ")*" }}
+{{- render_markdown_schema_details(spec.returns, indent + "  ", true) }}
+{%- elif spec.returns is defined -%}
+{{ render_markdown_detail(indent, "Returns", spec.returns) }}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_schema_details(spec, indent, include_value_details) -%}
+{%- if spec is mapping and spec['$ref'] is string -%}
+    {%- set _r = spec['$ref'] -%}
+    {%- set _k = _r[8:] if _r.startswith('#/$defs/') else (_r[14:] if _r.startswith('#/definitions/') else none) -%}
+    {%- if _k is not none and ('|' + _k + '|') not in REFS.seen and REFS.defs[_k] is mapping -%}
+        {%- set spec = dict((REFS.defs[_k] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+        {%- set REFS.seen = REFS.seen + _k + '|' -%}
+        {%- if spec['$ref'] is string -%}
+            {%- set _r2 = spec['$ref'] -%}
+            {%- set _k2 = _r2[8:] if _r2.startswith('#/$defs/') else (_r2[14:] if _r2.startswith('#/definitions/') else none) -%}
+            {%- if _k2 is not none and REFS.defs[_k2] is mapping -%}
+                {%- set spec = dict((REFS.defs[_k2] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+                {%- set REFS.seen = REFS.seen + _k2 + '|' -%}
+            {%- endif -%}
+        {%- endif -%}
+    {%- endif -%}
+{%- endif -%}
+{%- if spec is mapping -%}
+{{- render_markdown_schema_annotations(spec, indent, include_value_details) }}
+{{- render_markdown_schema_structure(spec, indent, true) }}
+{{- render_markdown_schema_extras(spec, indent) }}
+{%- elif spec is not sameas true and spec is not sameas false -%}
+{{- "\n" + indent + "  - Value: " }}{{ render_markdown_literal(spec) }}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_parameter_schema(spec) -%}
+{%- if spec is mapping and spec['$ref'] is string -%}
+    {%- set _r = spec['$ref'] -%}
+    {%- set _k = _r[8:] if _r.startswith('#/$defs/') else (_r[14:] if _r.startswith('#/definitions/') else none) -%}
+    {%- if _k is not none and ('|' + _k + '|') not in REFS.seen and REFS.defs[_k] is mapping -%}
+        {%- set spec = dict((REFS.defs[_k] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+        {%- set REFS.seen = REFS.seen + _k + '|' -%}
+        {%- if spec['$ref'] is string -%}
+            {%- set _r2 = spec['$ref'] -%}
+            {%- set _k2 = _r2[8:] if _r2.startswith('#/$defs/') else (_r2[14:] if _r2.startswith('#/definitions/') else none) -%}
+            {%- if _k2 is not none and REFS.defs[_k2] is mapping -%}
+                {%- set spec = dict((REFS.defs[_k2] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+                {%- set REFS.seen = REFS.seen + _k2 + '|' -%}
+            {%- endif -%}
+        {%- endif -%}
+    {%- endif -%}
+{%- endif -%}
+{%- if spec is mapping -%}
+{{- render_markdown_metadata_annotations(spec) }}
+{%- if 'items' in spec and spec['items'] is mapping -%}
+{{- "\n- Items *(" + render_markdown_type(spec['items']) + ")*" }}
+{{- render_markdown_schema_details(spec['items'], "", true) }}
+{%- elif 'items' in spec -%}
+{{ render_markdown_metadata_detail("Items", spec['items']) }}
+{%- endif -%}
+{%- if spec.oneOf -%}
+{{- "\n- oneOf:" }}
+    {%- for variant in spec.oneOf -%}
+{{- "\n  - Variant " }}{{ loop.index }}{{- " *(" + render_markdown_type(variant) + ")*" }}
+{{- render_markdown_schema_details(variant, "  ", true) }}
+    {%- endfor -%}
+{%- endif -%}
+{%- if spec.anyOf -%}
+{{- "\n- anyOf:" }}
+    {%- for variant in spec.anyOf -%}
+{{- "\n  - Variant " }}{{ loop.index }}{{- " *(" + render_markdown_type(variant) + ")*" }}
+{{- render_markdown_schema_details(variant, "  ", true) }}
+    {%- endfor -%}
+{%- endif -%}
+{%- if spec.patternProperties is mapping -%}
+{{- "\n- Pattern properties:" }}
+    {%- for pattern, pattern_spec in spec.patternProperties | items -%}
+        {%- if pattern_spec is mapping -%}
+{{- "\n  - `" + pattern + "` *(" + render_markdown_type(pattern_spec) + ")*" }}
+{{- render_markdown_schema_details(pattern_spec, "  ", true) }}
+        {%- else -%}
+{{- "\n  - `" + pattern + "`: " }}{{ render_markdown_value(pattern_spec) }}
+        {%- endif -%}
+    {%- endfor -%}
+{%- elif spec.patternProperties is defined -%}
+{{ render_markdown_metadata_detail("Pattern properties", spec.patternProperties) }}
+{%- endif -%}
+{%- if spec.returns is mapping -%}
+{{- "\n- Returns *(" + render_markdown_type(spec.returns) + ")*" }}
+{{- render_markdown_schema_details(spec.returns, "", true) }}
+{%- elif spec.returns is defined -%}
+{{ render_markdown_metadata_detail("Returns", spec.returns) }}
+{%- endif -%}
+{{- render_markdown_metadata_extras(spec) }}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_markdown_param(name, spec, required_list, indent) -%}
+{%- if spec is mapping and spec['$ref'] is string -%}
+    {%- set _r = spec['$ref'] -%}
+    {%- set _k = _r[8:] if _r.startswith('#/$defs/') else (_r[14:] if _r.startswith('#/definitions/') else none) -%}
+    {%- if _k is not none and ('|' + _k + '|') not in REFS.seen and REFS.defs[_k] is mapping -%}
+        {%- set spec = dict((REFS.defs[_k] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+        {%- set REFS.seen = REFS.seen + _k + '|' -%}
+        {%- if spec['$ref'] is string -%}
+            {%- set _r2 = spec['$ref'] -%}
+            {%- set _k2 = _r2[8:] if _r2.startswith('#/$defs/') else (_r2[14:] if _r2.startswith('#/definitions/') else none) -%}
+            {%- if _k2 is not none and REFS.defs[_k2] is mapping -%}
+                {%- set spec = dict((REFS.defs[_k2] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+                {%- set REFS.seen = REFS.seen + _k2 + '|' -%}
+            {%- endif -%}
+        {%- endif -%}
+    {%- endif -%}
+{%- endif -%}
+{{- "\n" + indent + "- `" + name + "` *(" + render_markdown_type(spec) }}
+{%- if name in (required_list or []) -%}{{- ", required" }}{%- endif -%}
+{{- ")*" }}
+{%- if spec.description -%}{{- " - " + spec.description | replace("\n", "\n" + indent + "  ") }}{%- endif -%}
+{%- if spec.enum -%}
+{{- "\n" + indent + "  - Allowed values: " }}{{ render_allowed_values(spec.enum) }}
+{%- endif -%}
+{%- if spec.default is defined -%}
+{{- "\n" + indent + "  - Default: " }}{{ render_markdown_literal(spec.default) }}
+{%- endif -%}
+{{- render_markdown_schema_details(spec, indent, false) }}
+{%- endmacro -%}
+
+{%- macro render_tools_markdown(tools_list) -%}
+{{- "<ifm|tools>" }}
+{%- for tool in tools_list -%}
+    {%- set fn = tool.function if tool.function is defined else tool -%}
+    {%- set REFS.defs = fn.parameters['$defs'] if (fn.parameters is mapping and fn.parameters['$defs'] is mapping) else (fn.parameters['definitions'] if (fn.parameters is mapping and fn.parameters['definitions'] is mapping) else {}) -%}
+    {%- set REFS.seen = '|' -%}
+    {%- set fnp = namespace(p=fn.parameters) -%}
+    {%- if fnp.p is mapping and fnp.p['$ref'] is string -%}
+        {%- set _r = fnp.p['$ref'] -%}
+        {%- set _k = _r[8:] if _r.startswith('#/$defs/') else (_r[14:] if _r.startswith('#/definitions/') else none) -%}
+        {%- if _k is not none and REFS.defs[_k] is mapping -%}
+            {%- set fnp.p = dict((REFS.defs[_k] | items | list) + (fnp.p | items | rejectattr('0', 'equalto', '$ref') | list)) -%}
+            {%- set REFS.seen = REFS.seen + _k + '|' -%}
+        {%- endif -%}
+    {%- endif -%}
+{{- "\n## " + fn.name }}
+{%- if fn.description -%}
+{{- "\n" + fn.description }}
+{%- endif -%}
+{{- "\n\n**Parameters**" }}
+{%- if fnp.p and fnp.p.properties -%}
+    {%- for pname, pspec in fnp.p.properties | items -%}
+{{- render_markdown_param(pname, pspec, fnp.p.required or [], "") }}
+    {%- endfor -%}
+{%- elif fnp.p is mapping and (fnp.p.oneOf or fnp.p.anyOf or 'items' in fnp.p) -%}
+{{- render_markdown_parameter_schema(fnp.p) }}
+{%- else -%}
+{{- "\n- None" }}
+{%- endif -%}
+{%- set fn_ret = fn.returns if fn.returns is defined else fn.response -%}
+{%- if fn_ret is mapping -%}
+{{- "\n\n**Returns**" }}
+{{- "\n- Return *(" + render_markdown_type(fn_ret) + ")*" }}
+{{- render_markdown_schema_details(fn_ret, "", true) }}
+{%- elif fn_ret is defined -%}
+{{- "\n\n**Returns**\n- " }}{{ render_markdown_value(fn_ret) }}
+{%- endif -%}
+{%- if not loop.last -%}{{- "\n" }}{%- endif -%}
+{%- endfor -%}
+{{- "\n</ifm|tools>" }}
+{%- endmacro -%}
+
+{%- macro render_tool_presentation(tools_list, fmt) -%}
+{%- if fmt == 'json' -%}
+{{- render_tools_json(tools_list) }}
+{%- elif RB.bad != '|' -%}
+{#- some tool uses constructs the pretty renderers cannot represent (verdicts -#}
+{#- computed during validate_tools): render the WHOLE toolset exactly as the -#}
+{#- json presentation would, so the block stays uniform and model-familiar. -#}
+{{- render_tools_json(tools_list) }}
+{%- elif fmt == 'xml' -%}
+{{- render_tools_xml(tools_list) }}
+{%- elif fmt == 'markdown' -%}
+{{- render_tools_markdown(tools_list) }}
+{%- else -%}
+{{- raise_exception("Unsupported tool_presentation_format: '" + fmt + "'. Supported formats: json, xml, markdown.") }}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_call_instructions(fmt) -%}
+{%- if fmt == 'json' -%}
+{{- "Wrap all tool calls in a single <ifm|tool_calls></ifm|tool_calls> block. For each call, emit one JSON object with the function name and arguments on the same line inside <ifm|tool_call></ifm|tool_call> tags:\n\n<ifm|tool_calls>\n<ifm|tool_call>{\"name\": <function-name>, \"arguments\": <args-json-object>}</ifm|tool_call>\n</ifm|tool_calls>" }}
+{%- elif fmt == 'xml' -%}
+{{- "Wrap all tool calls in a single <ifm|tool_calls></ifm|tool_calls> block. For each call, write the function name at the start of <ifm|tool_call>, followed by paired <ifm|arg_key> and <ifm|arg_value> tags for each argument:\n\n<ifm|tool_calls>\n<ifm|tool_call>$FUNCTION_NAME\n<ifm|arg_key>$PARAMETER_NAME</ifm|arg_key>\n<ifm|arg_value>$PARAMETER_VALUE</ifm|arg_value>\n...\n</ifm|tool_call>\n</ifm|tool_calls>\n\nString and scalar parameters should be written as plain text. Array and object parameters should be written as JSON literals." }}
+{%- elif fmt == 'xml_typed' -%}
+{{- "Wrap all tool calls in a single <ifm|tool_calls></ifm|tool_calls> block. For each call, write the function name at the start of <ifm|tool_call>, followed by <ifm|arg_key>, <ifm|arg_type>, and <ifm|arg_value> tags for each argument:\n\n<ifm|tool_calls>\n<ifm|tool_call>$FUNCTION_NAME\n<ifm|arg_key>$PARAMETER_NAME</ifm|arg_key>\n<ifm|arg_type>$ARGUMENT_TYPE</ifm|arg_type>\n<ifm|arg_value>$PARAMETER_VALUE</ifm|arg_value>\n...\n</ifm|tool_call>\n</ifm|tool_calls>\n\nUse the parameter type shown in the tool definition. If that type contains anyOf or oneOf, use the actual argument value type instead. String and scalar parameters should be written as plain text. Array and object parameters should be written as JSON literals." }}
+{%- else -%}
+{{- raise_exception("Unsupported tool_call_format: '" + fmt + "'. Supported formats: json, xml, xml_typed.") }}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_system_with_tools(tools_list, system_content, presentation_fmt, call_fmt) -%}
+{{- "<|ifm|im_start|>system\n# Tools\nYou may call one or more tools to assist with the user query.\n\nAvailable tools are:\n\n" }}
+{{- render_tool_presentation(tools_list, presentation_fmt) }}
+{{- "\n\nWhen calling tools, you MUST follow the tool-call format below:\n\n" }}
+{{- render_call_instructions(call_fmt) }}
+{%- if system_content -%}
+{{- "\n\n" + system_content }}
+{%- endif -%}
+{{- "<|ifm|im_end|>" }}
+{%- endmacro -%}
+
+{%- macro render_argument_value(value) -%}
+{%- if value is string -%}{{- value -}}{%- else -%}{{- value | tojson -}}{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_value_type(value) -%}
+{%- if value is none -%}null
+{%- elif value is boolean -%}boolean
+{%- elif value is integer -%}integer
+{%- elif value is number -%}number
+{%- elif value is string -%}string
+{%- elif value is mapping -%}object
+{%- elif value is sequence -%}array
+{%- else -%}any
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro schema_has_combinator(spec) -%}
+{%- if spec.oneOf or spec.anyOf -%}
+true
+{%- elif spec.type is defined and spec.type is sequence and spec.type is not string and spec.type | length > 1 -%}
+true
+{%- elif spec.type == "array" and 'items' in spec -%}
+{{- schema_has_combinator(spec['items']) -}}
+{%- elif spec.properties -%}
+    {%- set found = namespace(value='false') -%}
+    {%- for child_name, child_spec in spec.properties | items -%}
+        {%- if schema_has_combinator(child_spec) == 'true' -%}
+            {%- set found.value = 'true' -%}
+        {%- endif -%}
+    {%- endfor -%}
+{{- found.value -}}
+{%- else -%}
+false
+{%- endif -%}
+{%- endmacro -%}
+
+{%- macro render_arg_type(tools_list, tool_name, arg_name, value) -%}
+{%- set found = namespace(type='any') -%}
+{%- for tool in tools_list -%}
+    {%- set fn = tool.function if tool.function is defined else tool -%}
+    {%- if fn.name == tool_name and fn.parameters and fn.parameters.properties and arg_name in fn.parameters.properties -%}
+        {%- set spec = fn.parameters.properties[arg_name] -%}
+        {%- if spec is mapping and spec['$ref'] is string -%}
+            {%- set _r = spec['$ref'] -%}
+            {%- set _k = _r[8:] if _r.startswith('#/$defs/') else (_r[14:] if _r.startswith('#/definitions/') else none) -%}
+            {%- set _d = fn.parameters['$defs'] if fn.parameters['$defs'] is mapping else fn.parameters['definitions'] -%}
+            {%- set spec = dict((_d[_k] | items | list) + (spec | items | rejectattr('0', 'equalto', '$ref') | list)) if (_k is not none and _d is mapping and _d[_k] is mapping) else spec -%}
+        {%- endif -%}
+        {%- if schema_has_combinator(spec) == 'true' -%}
+            {%- set found.type = render_value_type(value) -%}
+        {%- else -%}
+            {%- set found.type = render_compact_type(spec) -%}
+        {%- endif -%}
+    {%- endif -%}
+{%- endfor -%}
+{{- found.type -}}
+{%- endmacro -%}
+
+{%- macro render_tool_calls_block(tool_calls, fmt, tools_list) -%}
+{{- "<ifm|tool_calls>" }}
+{%- for raw_tool_call in tool_calls -%}
+    {%- set tool_call = raw_tool_call.function if raw_tool_call.function else raw_tool_call -%}
+    {%- if tool_call.arguments is string -%}
+        {{- raise_exception("tool_call.arguments must be a dict, not a JSON string. Parse it before passing to the template.") -}}
+    {%- endif -%}
+    {%- if fmt == 'json' -%}
+{{- "\n<ifm|tool_call>{\"name\": \"" + tool_call.name + "\", \"arguments\": " }}{{ tool_call.arguments | tojson }}{{- "}</ifm|tool_call>" }}
+    {%- elif fmt == 'xml' or fmt == 'xml_typed' -%}
+{{- "\n<ifm|tool_call>" + tool_call.name + "\n" }}
+        {%- for key, value in tool_call.arguments | items -%}
+{{- "<ifm|arg_key>" + key + "</ifm|arg_key>\n" }}
+{%- if fmt == 'xml_typed' -%}
+{{- "<ifm|arg_type>" + render_arg_type(tools_list, tool_call.name, key, value) + "</ifm|arg_type>\n" }}
+{%- endif -%}
+{{- "<ifm|arg_value>" }}{{ render_argument_value(value) }}{{- "</ifm|arg_value>\n" }}
+        {%- endfor -%}
+{{- "</ifm|tool_call>" }}
+    {%- else -%}
+        {{- raise_exception("Unsupported tool_call_format: '" + fmt + "'. Supported formats: json, xml, xml_typed.") -}}
+    {%- endif -%}
+{%- endfor -%}
+{{- "\n</ifm|tool_calls>" }}
+{%- endmacro -%}
+
+{%- macro render_tool_response_messages(raw_content) -%}
+{%- if raw_content is string -%}
+{{- '<|ifm|im_start|>tool\n' + raw_content + '<|ifm|im_end|>' }}
+{%- elif raw_content is sequence and raw_content is not string and raw_content is not mapping -%}
+    {%- if raw_content | length == 0 -%}
+        {{- raise_exception("tool message content list must not be empty.") -}}
+    {%- endif -%}
+{{- '<|ifm|im_start|>tool\n' -}}
+    {%- for item in raw_content -%}
+        {%- if not loop.first -%}{{- '\n' -}}{%- endif -%}
+        {%- if item is string -%}
+{{- item -}}
+        {%- elif item is mapping and item.text is string -%}
+{{- item.text -}}
+        {%- else -%}
+{{- (item | tojson) -}}
+        {%- endif -%}
+    {%- endfor -%}
+{{- '<|ifm|im_end|>' -}}
+{%- else -%}
+{{- '<|ifm|im_start|>tool\n' }}{{ raw_content | tojson }}{{- '<|ifm|im_end|>' }}
+{%- endif -%}
+{%- endmacro -%}
+
+{%- set available_tools = tools if tools else [] -%}
+{%- if (not available_tools) and messages[0].role == 'system' and messages[0].get('tools') -%}
+    {%- set available_tools = messages[0]['tools'] -%}
+{%- endif -%}
+{%- if available_tools -%}
+    {{- validate_tools(available_tools, tool_presentation_fmt != 'json') }}
+    {%- set system_content = '' -%}
+    {%- if messages[0].role == 'system' and messages[0].content -%}
+        {%- set system_content = messages[0].content -%}
+    {%- endif -%}
+    {{- render_system_with_tools(available_tools, system_content, tool_presentation_fmt, tool_call_fmt) }}
+{%- else -%}
+    {%- if messages[0].role == 'system' -%}
+        {{- '<|ifm|im_start|>system\n' + messages[0].content + '<|ifm|im_end|>' }}
+    {%- endif -%}
+{%- endif -%}
+
+{%- for message in messages -%}
+    {%- if message.content is string -%}
+        {%- set content = message.content -%}
+    {%- else -%}
+        {%- set content = '' -%}
+    {%- endif -%}
+    {%- if (message.role == "user") or (message.role == "system" and not loop.first) -%}
+        {{- '<|ifm|im_start|>' + message.role + '\n' + content + '<|ifm|im_end|>' }}
+    {%- elif message.role == "assistant" -%}
+        {%- set thinking_content = '' -%}
+        {%- set think_tag = 'ifm|think' -%}
+        {%- if message.think is defined and message.think is string -%}
+            {%- set thinking_content = message.think -%}
+            {%- set think_tag = 'ifm|think' -%}
+        {%- elif message.think_fast is defined and message.think_fast is string -%}
+            {%- set thinking_content = message.think_fast -%}
+            {%- set think_tag = 'ifm|think_fast' -%}
+        {%- elif message.think_faster is defined and message.think_faster is string -%}
+            {%- set thinking_content = message.think_faster -%}
+            {%- set think_tag = 'ifm|think_faster' -%}
+        {%- elif message.reasoning_content is defined and message.reasoning_content is string -%}
+            {%- set thinking_content = message.reasoning_content -%}
+            {%- set think_tag = 'ifm|think' -%}
+        {%- elif message.reasoning is defined and message.reasoning is string -%}
+            {%- set thinking_content = message.reasoning -%}
+            {%- set think_tag = 'ifm|think' -%}
+        {%- elif message.think is not defined and message.reasoning is not defined and message.reasoning_content is not defined and message.think_fast is not defined and message.think_faster is not defined -%}
+            {{- raise_exception("Assistant message is missing a thinking field. Provide one of: think, reasoning, reasoning_content, think_fast, think_faster.") -}}
+        {%- else -%}
+            {{- raise_exception("Assistant thinking fields must be strings. Provide one of: think, reasoning, reasoning_content, think_fast, think_faster as a string.") -}}
+        {%- endif -%}
+        {{- '<|ifm|im_start|>' + message.role }}
+        {% generation %}
+        {%- if think_tag -%}
+            {%- if thinking_content -%}
+                {{- '<' + think_tag + '>\n' + thinking_content + '</' + think_tag + '>' + content }}
+            {%- else -%}
+                {{- '<' + think_tag + '>\n</' + think_tag + '>' + content }}
+            {%- endif -%}
+        {%- else -%}
+            {{- content }}
+        {%- endif -%}
+        {%- if message.tool_calls -%}
+            {{- render_tool_calls_block(message.tool_calls, tool_call_fmt, available_tools) }}
+        {%- endif -%}
+        {{- '<|ifm|im_end|>' -}}
+        {%- endgeneration -%}
+    {%- elif message.role == "tool" -%}
+        {{- render_tool_response_messages(message.content) }}
+    {%- endif -%}
+{%- endfor -%}
+{%- if add_generation_prompt -%}
+    {%- set effort = reasoning_effort | default('high') -%}
+    {%- if effort == 'high' -%}
+        {{- '<|ifm|im_start|>assistant\n<ifm|think>\n' }}
+    {%- elif effort == 'medium' -%}
+        {{- '<|ifm|im_start|>assistant\n<ifm|think_fast>\n' }}
+    {%- elif effort == 'low' -%}
+        {{- '<|ifm|im_start|>assistant\n<ifm|think_faster>\n' }}
+    {%- else -%}
+        {{- raise_exception("Unsupported reasoning_effort: '" + effort + "'. Supported values: high, medium, low.") -}}
+    {%- endif -%}
+{%- endif -%}
diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp
index e1014ed38..eea2ef589 100644
--- a/src/llama-arch.cpp
+++ b/src/llama-arch.cpp
@@ -161,6 +161,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
     { LLM_ARCH_NANBEIGE,         "nanbeige"         },
     { LLM_ARCH_QWEN3TTS,         "qwen3tts"         },
     { LLM_ARCH_POCKETTTS,        "pockettts"        },
+    { LLM_ARCH_K2_HORIZON,       "k2-horizon"       },
     { LLM_ARCH_UNKNOWN,          "(unknown)"        },
 };

@@ -278,6 +279,8 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
     { LLM_KV_ATTENTION_SLIDING_WINDOW,               "%s.attention.sliding_window"               },
     { LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN,       "%s.attention.sliding_window_pattern"       },
     { LLM_KV_ATTENTION_ROPE_PATTERN,                 "%s.attention.rope_pattern"                 },
+    { LLM_KV_ATTENTION_VALUE_EXPERT_COUNT,           "%s.attention.value_expert_count"           },
+    { LLM_KV_ATTENTION_VALUE_EXPERT_USED_COUNT,      "%s.attention.value_expert_used_count"      },

     { LLM_KV_ATTENTION_SCALE,                        "%s.attention.scale"                        },
     { LLM_KV_ATTENTION_OUTPUT_SCALE,                 "%s.attention.output_scale"                 },
@@ -735,6 +738,8 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
     { LLM_TENSOR_DFLASH_SELECTOR_PREV,                   "selector_predecessor" },
     { LLM_TENSOR_DFLASH_SELECTOR_NEXT,                   "selector_successor" },
     { LLM_TENSOR_DFLASH_SELECTOR_HIDDEN,                 "selector_hidden" },
+    { LLM_TENSOR_ATTN_V_GATE,                            "blk.%d.attn_v_gate" },
+    { LLM_TENSOR_ATTN_V_EXPS,                            "blk.%d.attn_v_exps" },
 };

 // declare information about the model weight tensors:
@@ -1045,6 +1050,8 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
     {LLM_TENSOR_DFLASH_SELECTOR_PREV,       {LLM_TENSOR_LAYER_OUTPUT,    GGML_OP_GET_ROWS}},
     {LLM_TENSOR_DFLASH_SELECTOR_NEXT,       {LLM_TENSOR_LAYER_OUTPUT,    GGML_OP_GET_ROWS}},
     {LLM_TENSOR_DFLASH_SELECTOR_HIDDEN,     {LLM_TENSOR_LAYER_OUTPUT,    GGML_OP_MUL_MAT}},
+    {LLM_TENSOR_ATTN_V_GATE,                {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
+    {LLM_TENSOR_ATTN_V_EXPS,                {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
 };

 LLM_KV::LLM_KV(llm_arch arch, const char * suffix) : arch(arch), suffix(suffix) {}
@@ -1226,6 +1233,7 @@ bool llm_arch_supports_sm_tensor(const llm_arch & arch) {
         case LLM_ARCH_KIMI_K3:
         case LLM_ARCH_GLM5_NEXT:
         case LLM_ARCH_QWEN3TTS:
+        case LLM_ARCH_K2_HORIZON:
             return false;
         default:
             return true;
diff --git a/src/llama-arch.h b/src/llama-arch.h
index 80344d328..c8abee234 100644
--- a/src/llama-arch.h
+++ b/src/llama-arch.h
@@ -166,6 +166,7 @@ enum llm_arch {
     LLM_ARCH_POCKETTTS,
     LLM_ARCH_MINIMAX_01,
     LLM_ARCH_HRM_TEXT,
+    LLM_ARCH_K2_HORIZON,
     LLM_ARCH_UNKNOWN,
 };

@@ -284,6 +285,8 @@ enum llm_kv {
     LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN,
     LLM_KV_ATTENTION_SCALE,
     LLM_KV_ATTENTION_ROPE_PATTERN,
+    LLM_KV_ATTENTION_VALUE_EXPERT_COUNT,
+    LLM_KV_ATTENTION_VALUE_EXPERT_USED_COUNT,

     LLM_KV_ATTENTION_OUTPUT_SCALE,
     LLM_KV_ATTENTION_VALUE_SCALE,
@@ -742,6 +745,8 @@ enum llm_tensor {
     LLM_TENSOR_DFLASH_SELECTOR_PREV,
     LLM_TENSOR_DFLASH_SELECTOR_NEXT,
     LLM_TENSOR_DFLASH_SELECTOR_HIDDEN,
+    LLM_TENSOR_ATTN_V_GATE,
+    LLM_TENSOR_ATTN_V_EXPS,
 };


diff --git a/src/llama-hparams.h b/src/llama-hparams.h
index 6c504c5dd..848428706 100644
--- a/src/llama-hparams.h
+++ b/src/llama-hparams.h
@@ -72,6 +72,8 @@ struct llama_hparams {
     int32_t  router_layer = -1;
     uint32_t n_expert = 0;
     uint32_t n_rel_attn_bkts = 0;
+    uint32_t n_value_expert      = 0; // MoVA value experts (K2 Horizon)
+    uint32_t n_value_expert_used = 0;

     // TODO: this needs to be reworked
     int32_t  n_layer_kv_from_start = -1; // if non-negative, the first n_layer_kv_from_start layers have KV cache
diff --git a/src/llama-model-saver.cpp b/src/llama-model-saver.cpp
index e4fd45ec4..8d9728319 100644
--- a/src/llama-model-saver.cpp
+++ b/src/llama-model-saver.cpp
@@ -279,6 +279,8 @@ void llama_model_saver::add_kv_from_model() {
     add_kv(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS,       hparams.f_norm_rms_eps);
     add_kv(LLM_KV_ATTENTION_GROUPNORM_EPS,           hparams.f_norm_group_eps);
     add_kv(LLM_KV_ATTENTION_GROUPNORM_GROUPS,        hparams.n_norm_groups);
+    add_kv(LLM_KV_ATTENTION_VALUE_EXPERT_COUNT,      hparams.n_value_expert);
+    add_kv(LLM_KV_ATTENTION_VALUE_EXPERT_USED_COUNT, hparams.n_value_expert_used);
     add_kv(LLM_KV_ATTENTION_CAUSAL,                  hparams.causal_attn);
     add_kv(LLM_KV_ATTENTION_Q_LORA_RANK,             hparams.n_lora_q);
     add_kv(LLM_KV_ATTENTION_KV_LORA_RANK,            hparams.n_lora_kv);
diff --git a/src/llama-model.cpp b/src/llama-model.cpp
index fa379bae8..1f3b80b08 100644
--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
@@ -350,6 +350,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params
             return new llama_model_step35(params);
         case LLM_ARCH_SPARK2_5:
             return new llama_model_spark2_5(params);
+        case LLM_ARCH_K2_HORIZON:
+            return new llama_model_k2_horizon(params);
         default:
             throw std::runtime_error(std::string("unsupported model architecture: '") + llm_arch_name(arch) + "'");
     }
@@ -387,6 +389,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str

     static const std::regex pattern_q_weight        ("blk\\.\\d*\\.attn_q.weight");
     static const std::regex pattern_kv_weight       ("blk\\.\\d*\\.attn_(k|v).weight");
+    static const std::regex pattern_v_exps_weight   ("blk\\.\\d*\\.attn_v_exps.weight"); // K2 Horizon MoVA
     static const std::regex pattern_qkv_weight      ("blk\\.\\d*\\.attn_qkv.weight");
     static const std::regex pattern_q_bias          ("blk\\.\\d*\\.attn_q\\.bias");
     static const std::regex pattern_kv_bias         ("blk\\.\\d*\\.attn_(k|v)\\.bias");
@@ -525,6 +528,10 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str
         if (std::regex_match(tensor_name, pattern_q_weight) || std::regex_match(tensor_name, pattern_kv_weight)) {
             return get_tensor_config_impl(GGML_BACKEND_SPLIT_AXIS_1, "attn_output.weight", "ssm_out.weight");
         }
+        // routed value experts {n_embd, n_embd_v_gqa, n_expert} produce V, so they split like attn_v
+        if (std::regex_match(tensor_name, pattern_v_exps_weight)) {
+            return get_tensor_config_impl(GGML_BACKEND_SPLIT_AXIS_1, "attn_output.weight");
+        }
         if (std::regex_match(tensor_name, pattern_q_bias) || std::regex_match(tensor_name, pattern_kv_bias)) {
             return get_tensor_config_impl(GGML_BACKEND_SPLIT_AXIS_0, "attn_output.weight", "ssm_out.weight");
         }
@@ -790,6 +797,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str
             // three stay in lockstep per device
             const int64_t granularity_v  = (granularity_kv / hparams.n_embd_head_k(il)) * hparams.n_embd_head_v(il);
             if (std::regex_match(tensor_name, pattern_kv_weight) ||
+                std::regex_match(tensor_name, pattern_v_exps_weight) ||
                 std::regex_match(tensor_name, pattern_kv_bias) ||
                 std::regex_match(tensor_name, pattern_kv_cache)) {
                 GGML_ASSERT(segments.size() == 1);
@@ -3191,6 +3199,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
         case LLM_ARCH_STEP35:
         case LLM_ARCH_SPARK2_5:
         case LLM_ARCH_TALKIE:
+        case LLM_ARCH_K2_HORIZON:
         case LLM_ARCH_MELLUM:
         case LLM_ARCH_MAPLE:
         case LLM_ARCH_HRM_TEXT:
diff --git a/src/llama-model.h b/src/llama-model.h
index 2243377cc..a86cc5693 100644
--- a/src/llama-model.h
+++ b/src/llama-model.h
@@ -308,6 +308,10 @@ struct llama_layer {
     struct ggml_tensor * wv_enc    = nullptr;
     struct ggml_tensor * wo_enc    = nullptr;
     struct ggml_tensor * wqkv_gate = nullptr;
+    // K2 Horizon MoVA
+    struct ggml_tensor * attn_v_gate   = nullptr;
+    struct ggml_tensor * attn_v_gate_b = nullptr;
+    struct ggml_tensor * attn_v_exps   = nullptr;

     // relative position bias
     struct ggml_tensor * attn_rel_b       = nullptr;
diff --git a/src/llama-vocab.cpp b/src/llama-vocab.cpp
index 015e98374..dddb6bab8 100644
--- a/src/llama-vocab.cpp
+++ b/src/llama-vocab.cpp
@@ -558,6 +558,11 @@ struct llm_tokenizer_bpe : llm_tokenizer {
                     "(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])|[^\\r\\n\\p{L}\\p{N}]?\\p{L}+|\\p{N}+| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
                 };
                 break;
+            case LLAMA_VOCAB_PRE_TYPE_K2_HORIZON:
+                regex_exprs = {
+                    "(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])|[^\\r\\n\\p{L}\\p{N}]?(?:\\p{L}|\\p{M}|\\u200C|\\u200D)+|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
+                };
+                break;
             case LLAMA_VOCAB_PRE_TYPE_WHITESPACE:
                 // whitespace pre-tokenizer (jinaai/jina-embeddings-v2-base-zh)
                 regex_exprs = {
@@ -2458,6 +2463,10 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
             } else if (
                 tokenizer_pre == "mellum2") {
                 pre_type = LLAMA_VOCAB_PRE_TYPE_MELLUM2;
+            } else if (
+                tokenizer_pre == "k2-horizon") {
+                pre_type = LLAMA_VOCAB_PRE_TYPE_K2_HORIZON;
+                clean_spaces = false;
             } else {
                 throw std::runtime_error(format("unknown pre-tokenizer type: '%s'", tokenizer_pre.c_str()));
             }
@@ -2945,6 +2954,7 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
                     || t.first == "<|tool_response>" // gemma4
                     || t.first == "<|end▁of▁sentence|>" // deepseek-ocr
                     || t.first == "[e~[" // minimax-m2/m3
+                    || t.first == "<|ifm|im_end|>" // k2-horizon
                ) {
                 special_eog_ids.insert(t.second);
                 if ((attr & LLAMA_TOKEN_ATTR_CONTROL) == 0) {
diff --git a/src/llama-vocab.h b/src/llama-vocab.h
index 033fa6208..2d40c6271 100644
--- a/src/llama-vocab.h
+++ b/src/llama-vocab.h
@@ -69,6 +69,7 @@ enum llama_vocab_pre_type {
     LLAMA_VOCAB_PRE_TYPE_SPARK2_5          = 58,
     LLAMA_VOCAB_PRE_TYPE_UFAKZEKA          = 59,
     LLAMA_VOCAB_PRE_TYPE_MMBERT            = 60,
+    LLAMA_VOCAB_PRE_TYPE_K2_HORIZON        = 61,
 };

 struct LLM_KV;
diff --git a/src/models/k2-horizon.cpp b/src/models/k2-horizon.cpp
new file mode 100644
index 000000000..222952738
--- /dev/null
+++ b/src/models/k2-horizon.cpp
@@ -0,0 +1,362 @@
+#include "models.h"
+
+void llama_model_k2_horizon::load_arch_hparams(llama_model_loader & ml) {
+    ml.get_key(LLM_KV_ROPE_SCALING_YARN_BETA_FAST, hparams.yarn_beta_fast, false);
+    ml.get_key(LLM_KV_ROPE_SCALING_YARN_BETA_SLOW, hparams.yarn_beta_slow, false);
+    ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
+    ml.get_key(LLM_KV_ATTENTION_GROUPNORM_GROUPS,  hparams.n_norm_groups, false);
+    if (hparams.n_norm_groups == 0) {
+        hparams.n_norm_groups = 1;
+    }
+
+    if (hparams.n_expert > 0) {
+        ml.get_key_or_arr(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp_arr, hparams.n_layer_all);
+        ml.get_key(LLM_KV_LEADING_DENSE_BLOCK_COUNT,         hparams.n_layer_dense_lead,   false);
+        ml.get_key(LLM_KV_MOE_EVERY_N_LAYERS,                hparams.moe_every_n_layers,   false);
+        ml.get_key(LLM_KV_EXPERT_SHARED_COUNT,               hparams.n_expert_shared,      false);
+        ml.get_key(LLM_KV_EXPERT_SHARED_FEED_FORWARD_LENGTH, hparams.n_ff_shexp,           false);
+        ml.get_key(LLM_KV_EXPERT_WEIGHTS_SCALE,              hparams.expert_weights_scale, false);
+        ml.get_key(LLM_KV_EXPERT_WEIGHTS_NORM,               hparams.expert_weights_norm,  false);
+        ml.get_key(LLM_KV_EXPERT_GATING_FUNC,                hparams.expert_gating_func,   false);
+        if (hparams.expert_gating_func == LLAMA_EXPERT_GATING_FUNC_TYPE_NONE) {
+            hparams.expert_gating_func = LLAMA_EXPERT_GATING_FUNC_TYPE_SIGMOID;
+        }
+    }
+
+    // MoVA
+    ml.get_key(LLM_KV_ATTENTION_VALUE_EXPERT_COUNT,      hparams.n_value_expert,      false);
+    ml.get_key(LLM_KV_ATTENTION_VALUE_EXPERT_USED_COUNT, hparams.n_value_expert_used, false);
+    if (hparams.n_value_expert > 0) {
+        GGML_ASSERT(hparams.n_value_expert <= LLAMA_MAX_EXPERTS);
+        GGML_ASSERT(hparams.n_value_expert_used > 0);
+        GGML_ASSERT(hparams.n_value_expert_used <= hparams.n_value_expert);
+    } else {
+        GGML_ASSERT(hparams.n_value_expert_used == 0);
+    }
+
+    switch (hparams.n_layer()) {
+        case 28: type = LLM_TYPE_1B; break;
+        case 36:
+            switch (hparams.n_embd) {
+                case 2560: type = LLM_TYPE_4B; break;
+                case 4096: type = LLM_TYPE_7B; break;
+                default:   type = LLM_TYPE_UNKNOWN;
+            } break;
+        case 48: type = LLM_TYPE_36B; break;
+        case 64: type = LLM_TYPE_32B; break;
+        default: type = LLM_TYPE_UNKNOWN;
+    }
+}
+
+void llama_model_k2_horizon::load_arch_tensors(llama_model_loader &) {
+    LLAMA_LOAD_LOCALS;
+
+    tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);
+
+    // output
+    output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);
+    output      = create_tensor(tn(LLM_TENSOR_OUTPUT,      "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED);
+    // if output is NULL, init from the input tok embed
+    if (output == NULL) {
+        output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED);
+    }
+
+    for (int i = 0; i < n_layer; ++i) {
+        auto & layer = layers[i];
+
+        const bool is_moe_layer  = n_expert > 0 && (uint32_t) i >= hparams.n_layer_dense_lead;
+        const bool is_mova_layer = is_moe_layer && hparams.n_value_expert > 0;
+
+        layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
+
+        layer.wq          = create_tensor(tn(LLM_TENSOR_ATTN_Q,      "weight", i), {n_embd, n_embd_head_k * n_head}, 0);
+        layer.wk          = create_tensor(tn(LLM_TENSOR_ATTN_K,      "weight", i), {n_embd, n_embd_k_gqa}, 0);
+        // one norm weight per head, stored flat; viewed as {head_dim, n_head} so it splits by head like Q/K
+        layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", i), {n_embd_head_k, n_head},    TENSOR_NOT_REQUIRED | TENSOR_ALLOW_RESHAPE);
+        layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", i), {n_embd_head_k, n_head_kv}, TENSOR_NOT_REQUIRED | TENSOR_ALLOW_RESHAPE);
+
+        if (is_mova_layer) {
+            layer.attn_v_gate   = create_tensor(tn(LLM_TENSOR_ATTN_V_GATE, "weight", i), {n_embd, hparams.n_value_expert}, 0);
+            layer.attn_v_gate_b = create_tensor(tn(LLM_TENSOR_ATTN_V_GATE, "bias",   i), {hparams.n_value_expert}, TENSOR_NOT_REQUIRED);
+            layer.attn_v_exps   = create_tensor(tn(LLM_TENSOR_ATTN_V_EXPS, "weight", i), {n_embd, n_embd_v_gqa, hparams.n_value_expert}, 0);
+        } else {
+            layer.wv = create_tensor(tn(LLM_TENSOR_ATTN_V, "weight", i), {n_embd, n_embd_v_gqa}, 0);
+        }
+
+        layer.wo        = create_tensor(tn(LLM_TENSOR_ATTN_OUT,  "weight", i), {n_embd_head_v * n_head, n_embd}, 0);
+        layer.wqkv_gate = create_tensor(tn(LLM_TENSOR_ATTN_GATE, "weight", i), {n_embd, n_embd_head_v * n_head}, TENSOR_NOT_REQUIRED);
+
+        layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
+
+        if (is_moe_layer) {
+            const int64_t n_ff_exp = hparams.n_ff_exp(i);
+            if (n_ff_exp == 0) {
+                throw std::runtime_error("K2 Horizon MoE layer requires expert_feed_forward_length");
+            }
+
+            layer.ffn_gate_inp    = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP,    "weight", i), {n_embd, n_expert}, 0);
+            layer.ffn_exp_probs_b = create_tensor(tn(LLM_TENSOR_FFN_EXP_PROBS_B, "bias",   i), {n_expert}, TENSOR_NOT_REQUIRED);
+
+            layer.ffn_up_exps   = create_tensor(tn(LLM_TENSOR_FFN_UP_EXPS,   "weight", i), {n_embd,   n_ff_exp, n_expert}, 0);
+            layer.ffn_gate_exps = create_tensor(tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", i), {n_embd,   n_ff_exp, n_expert}, 0);
+            layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i), {n_ff_exp, n_embd,   n_expert}, 0);
+
+            if (hparams.n_expert_shared > 0) {
+                const int64_t n_ff_shexp = hparams.n_ff_shexp > 0 ? hparams.n_ff_shexp : n_ff_exp * hparams.n_expert_shared;
+
+                layer.ffn_up_shexp   = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP,   "weight", i), {n_embd,     n_ff_shexp}, 0);
+                layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", i), {n_embd,     n_ff_shexp}, 0);
+                layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", i), {n_ff_shexp, n_embd},     0);
+            }
+        } else {
+            layer.ffn_up   = create_tensor(tn(LLM_TENSOR_FFN_UP,   "weight", i), {n_embd, n_ff}, 0);
+            layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff}, 0);
+            layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff,   n_embd}, 0);
+        }
+    }
+}
+
+std::unique_ptr<llm_graph_context> llama_model_k2_horizon::build_arch_graph(const llm_graph_params & params) const {
+    return std::make_unique<graph>(*this, params);
+}
+
+// RMS norm over n_groups equal slices of ne[0], then one full-width weight
+static ggml_tensor * k2_horizon_group_rms_norm(ggml_context * ctx, ggml_tensor * cur, ggml_tensor * weight, int64_t n_groups, float eps) {
+    GGML_ASSERT(n_groups > 0 && cur->ne[0] % n_groups == 0);
+
+    const int64_t n_embd   = cur->ne[0];
+    const int64_t n_tokens = cur->ne[1];
+
+    cur = ggml_reshape_3d(ctx, cur, n_embd / n_groups, n_groups, n_tokens);
+    cur = ggml_rms_norm(ctx, cur, eps);
+    cur = ggml_reshape_2d(ctx, cur, n_embd, n_tokens);
+
+    return weight ? ggml_mul(ctx, cur, weight) : cur;
+}
+
+// MoVA: route each token to n_value_expert_used value experts, V = sum_k w_k * silu(W_k x)
+ggml_tensor * llama_model_k2_horizon::graph::build_routed_value(const llama_layer & layer, ggml_tensor * cur, int il) const {
+    const int64_t n_embd     = cur->ne[0];
+    const int64_t n_tokens   = cur->ne[1];
+    const int64_t n_embd_gqa = hparams.n_embd_v_gqa(il);
+    const int64_t n_values   = hparams.n_value_expert;
+    const int64_t n_used     = hparams.n_value_expert_used;
+
+    ggml_tensor * logits = build_lora_mm(layer.attn_v_gate, cur);
+    ggml_tensor * probs  = nullptr;
+
+    switch ((llama_expert_gating_func_type) hparams.expert_gating_func) {
+        case LLAMA_EXPERT_GATING_FUNC_TYPE_SOFTMAX: probs = ggml_soft_max(ctx0, logits); break;
+        case LLAMA_EXPERT_GATING_FUNC_TYPE_SIGMOID: probs = ggml_sigmoid(ctx0, logits);  break;
+        default: GGML_ABORT("unsupported K2 Horizon value-router gating function");
+    }
+
+    // the bias only affects which experts are selected, not their weights
+    ggml_tensor * selection_probs = probs;
+    if (layer.attn_v_gate_b) {
+        selection_probs = ggml_add(ctx0, probs, layer.attn_v_gate_b);
+        cb(selection_probs, "v_moe_probs_biased", il);
+    }
+
+    ggml_tensor * selected_experts = ggml_argsort_top_k(ctx0, selection_probs, n_used);
+
+    probs = ggml_reshape_3d(ctx0, probs, 1, n_values, n_tokens);
+    ggml_tensor * weights = ggml_get_rows(ctx0, probs, selected_experts);
+
+    if (hparams.expert_weights_norm) {
+        weights = ggml_reshape_2d(ctx0, weights, n_used, n_tokens);
+        ggml_tensor * weights_sum = ggml_sum_rows(ctx0, weights);
+        weights_sum = ggml_clamp(ctx0, weights_sum, 6.103515625e-5f, INFINITY);
+        weights = ggml_div(ctx0, weights, weights_sum);
+        weights = ggml_reshape_3d(ctx0, weights, 1, n_used, n_tokens);
+        cb(weights, "v_moe_weights_norm", il);
+    }
+
+    if (hparams.expert_weights_scale != 0.0f && hparams.expert_weights_scale != 1.0f) {
+        weights = ggml_scale(ctx0, weights, hparams.expert_weights_scale);
+        cb(weights, "v_moe_weights_scaled", il);
+    }
+
+    cb(logits, "v_moe_logits", il);
+    cb(probs,  "v_moe_probs",  il);
+    cb(selected_experts->src[0], "v_moe_argsort", il);
+    cb(selected_experts,         "v_moe_topk",    il);
+    cb(weights, "v_moe_weights", il);
+
+    ggml_tensor * values = build_lora_mm_id(layer.attn_v_exps, ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens), selected_experts);
+    values = ggml_silu(ctx0, values);
+    values = ggml_mul(ctx0, values, weights);
+    cb(values, "v_moe_weighted", il);
+
+    // sum the selected experts; 3D views of {n_embd_gqa, 1, n_tokens} keep the strides of values,
+    // which lets the tensor-parallel backend follow its split through the views
+    // order the views before the adds so backends can fuse the sum
+    ggml_tensor * value_views[LLAMA_MAX_EXPERTS] = { nullptr };
+    for (int64_t i = 0; i < n_used; ++i) {
+        value_views[i] = ggml_view_3d(ctx0, values, n_embd_gqa, 1, n_tokens, values->nb[1], values->nb[2], i * values->nb[1]);
+        ggml_build_forward_expand(gf, value_views[i]);
+    }
+
+    ggml_tensor * value_out = value_views[0];
+    for (int64_t i = 1; i < n_used; ++i) {
+        value_out = ggml_add(ctx0, value_out, value_views[i]);
+        ggml_build_forward_expand(gf, value_out);
+    }
+    if (n_used == 1) {
+        value_out = ggml_cont(ctx0, value_out);
+    }
+    cb(value_out, "Vcur_routed", il);
+
+    return value_out;
+}
+
+llama_model_k2_horizon::graph::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) {
+    const int64_t n_embd_head = hparams.n_embd_head_v();
+    GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());
+
+    ggml_tensor * cur;
+    ggml_tensor * inpL;
+
+    inpL = build_inp_embd(model.tok_embd);
+
+    // inp_pos - contains the positions
+    ggml_tensor * inp_pos = build_inp_pos();
+
+    auto * inp_attn = build_attn_inp_kv();
+
+    ggml_tensor * inp_out_ids = build_inp_out_ids();
+
+    const float kq_scale = 1.0f / sqrtf(float(n_embd_head));
+
+    for (int il = 0; il < n_layer; ++il) {
+        const auto & layer = model.layers[il];
+
+        res->t_layer_inp[il] = inpL;
+
+        ggml_tensor * inpSA = inpL;
+
+        const bool is_moe_layer  = n_expert > 0 && (uint32_t) il >= hparams.n_layer_dense_lead;
+        const bool is_mova_layer = is_moe_layer && hparams.n_value_expert > 0;
+
+        cur = k2_horizon_group_rms_norm(ctx0, inpL, layer.attn_norm, hparams.n_norm_groups, hparams.f_norm_rms_eps);
+        cb(cur, "attn_norm", il);
+
+        // self-attention
+        {
+            ggml_tensor * attn_inp = cur; // saved for the output gate
+
+            ggml_tensor * Qcur = build_lora_mm(layer.wq, cur, layer.wq_s);
+            ggml_tensor * Kcur = build_lora_mm(layer.wk, cur, layer.wk_s);
+            ggml_tensor * Vcur = is_mova_layer ? build_routed_value(layer, cur, il) : build_lora_mm(layer.wv, cur, layer.wv_s);
+
+            Qcur = ggml_reshape_3d(ctx0, Qcur, n_embd_head, n_head,    n_tokens);
+            Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);
+            Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);
+
+            // per-head RMS norm with a separate weight for every head
+            if (layer.attn_q_norm) {
+                Qcur = build_norm(Qcur, layer.attn_q_norm, NULL, LLM_NORM_RMS, il);
+            }
+            if (layer.attn_k_norm) {
+                Kcur = build_norm(Kcur, layer.attn_k_norm, NULL, LLM_NORM_RMS, il);
+            }
+
+            Qcur = ggml_rope_ext(ctx0, Qcur, inp_pos, nullptr,
+                    n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
+                    ext_factor, attn_factor, beta_fast, beta_slow);
+            Kcur = ggml_rope_ext(ctx0, Kcur, inp_pos, nullptr,
+                    n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
+                    ext_factor, attn_factor, beta_fast, beta_slow);
+
+            cb(Qcur, "Qcur", il);
+            cb(Kcur, "Kcur", il);
+            cb(Vcur, "Vcur", il);
+
+            // with an output gate, o_proj is applied after gating
+            const bool gated = layer.wqkv_gate != nullptr;
+
+            cur = build_attn(inp_attn,
+                    gated ? nullptr : layer.wo, gated ? nullptr : layer.wo_b, gated ? nullptr : layer.wo_s,
+                    Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il);
+
+            if (gated) {
+                // softplus with beta = ln(2): log2(1 + 2^x)
+                constexpr float ln2 = 0.6931471805599453f;
+                ggml_tensor * gate = build_lora_mm(layer.wqkv_gate, attn_inp, layer.wqkv_gate_s);
+                gate = ggml_scale(ctx0, gate, ln2);
+                gate = ggml_softplus(ctx0, gate);
+                gate = ggml_scale(ctx0, gate, 1.4426950408889634f); // 1 / ln(2)
+
+                cur = ggml_mul(ctx0, cur, gate);
+                cur = build_lora_mm(layer.wo, cur, layer.wo_s);
+                if (layer.wo_b) {
+                    cur = ggml_add(ctx0, cur, layer.wo_b);
+                }
+            }
+        }
+
+        if (il == n_layer - 1 && inp_out_ids) {
+            cur   = ggml_get_rows(ctx0,   cur, inp_out_ids);
+            inpSA = ggml_get_rows(ctx0, inpSA, inp_out_ids);
+        }
+
+        ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpSA);
+        cb(ffn_inp, "ffn_inp", il);
+
+        cur = k2_horizon_group_rms_norm(ctx0, ffn_inp, layer.ffn_norm, hparams.n_norm_groups, hparams.f_norm_rms_eps);
+        cb(cur, "ffn_norm", il);
+
+        if (is_moe_layer) {
+            ggml_tensor * moe_out = build_moe_ffn(cur,
+                    layer.ffn_gate_inp,
+                    layer.ffn_up_exps,
+                    layer.ffn_gate_exps,
+                    layer.ffn_down_exps,
+                    layer.ffn_exp_probs_b,
+                    n_expert, n_expert_used,
+                    LLM_FFN_SILU,
+                    hparams.expert_weights_norm,
+                    hparams.expert_weights_scale,
+                    (llama_expert_gating_func_type) hparams.expert_gating_func,
+                    il);
+
+            if (layer.ffn_gate_shexp) {
+                ggml_tensor * ffn_shexp = build_ffn(cur,
+                        layer.ffn_up_shexp,   NULL, NULL,
+                        layer.ffn_gate_shexp, NULL, NULL,
+                        layer.ffn_down_shexp, NULL, NULL,
+                        NULL,
+                        LLM_FFN_SILU, LLM_FFN_PAR, il);
+                cur = ggml_add(ctx0, moe_out, ffn_shexp);
+            } else {
+                cur = moe_out;
+            }
+        } else {
+            cur = build_ffn(cur,
+                    layer.ffn_up,   NULL, NULL,
+                    layer.ffn_gate, NULL, NULL,
+                    layer.ffn_down, NULL, NULL,
+                    NULL,
+                    LLM_FFN_SILU, LLM_FFN_PAR, il);
+        }
+        cb(cur, "ffn_out", il);
+
+        cur = ggml_add(ctx0, cur, ffn_inp);
+        cur = build_cvec(cur, il);
+        cb(cur, "l_out", il);
+
+        // input for next layer
+        inpL = cur;
+    }
+
+    cur = k2_horizon_group_rms_norm(ctx0, inpL, model.output_norm, hparams.n_norm_groups, hparams.f_norm_rms_eps);
+    cb(cur, "result_norm", -1);
+    res->t_embd = cur;
+
+    // lm_head
+    cur = build_lora_mm(model.output, cur, model.output_s);
+    cb(cur, "result_output", -1);
+    res->t_logits = cur;
+
+    ggml_build_forward_expand(gf, cur);
+}
diff --git a/src/models/models.h b/src/models/models.h
index 1b589ddcf..023ed3021 100644
--- a/src/models/models.h
+++ b/src/models/models.h
@@ -2856,3 +2856,17 @@ struct llama_model_spark2_5 : public llama_model_base {

     std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
 };
+
+struct llama_model_k2_horizon : public llama_model_base {
+    llama_model_k2_horizon(const struct llama_model_params & params) : llama_model_base(params) {}
+    void load_arch_hparams(llama_model_loader & ml) override;
+    void load_arch_tensors(llama_model_loader & ml) override;
+
+    struct graph : public llm_graph_context {
+        graph(const llama_model & model, const llm_graph_params & params);
+
+        ggml_tensor * build_routed_value(const llama_layer & layer, ggml_tensor * cur, int il) const;
+    };
+
+    std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
+};
diff --git a/src/unicode.cpp b/src/unicode.cpp
index 93996f9dd..07b425f27 100644
--- a/src/unicode.cpp
+++ b/src/unicode.cpp
@@ -470,6 +470,153 @@ static std::vector<size_t> unicode_regex_split_custom_llama3(const std::string &
     return bpe_offsets;
 }

+static std::vector<size_t> unicode_regex_split_custom_k2_horizon(const std::string & text, const std::vector<size_t> & offsets) {
+    std::vector<size_t> bpe_offsets; // store the offset of each word
+    bpe_offsets.reserve(offsets.size()); // Reserve memory for the approximate size
+
+    const auto cpts = unicode_cpts_from_utf8(text);
+
+    size_t start = 0;
+    for (auto offset : offsets) {
+        const size_t offset_ini = start;
+        const size_t offset_end = start + offset;
+        assert(offset_end <= cpts.size());
+        start = offset_end;
+
+        static const uint32_t OUT_OF_RANGE = 0xFFFFFFFF;
+        auto _get_cpt = [&] (const size_t pos) -> uint32_t {
+            return (offset_ini <= pos && pos < offset_end) ? cpts[pos] : OUT_OF_RANGE;
+        };
+
+        auto _get_flags = [&] (const size_t pos) -> unicode_cpt_flags {
+            return (offset_ini <= pos && pos < offset_end) ? unicode_cpt_flags_from_cpt(cpts[pos]) : unicode_cpt_flags{};
+        };
+
+        // K2-Horizon: letter runs are (?:\p{L}|\p{M}|\u200C|\u200D)+
+        auto _is_k2_letter = [&] (const size_t pos) -> bool {
+            const uint32_t c = _get_cpt(pos);
+            if (c == 0x200C || c == 0x200D) {
+                return true;
+            }
+            const auto f = _get_flags(pos);
+            return f.is_letter || f.is_accent_mark;
+        };
+
+        size_t _prev_end = offset_ini;
+        auto _add_token = [&] (const size_t end) -> size_t {
+            assert(_prev_end <= end && end <= offset_end);
+            size_t len = end - _prev_end;
+            if (len > 0) {
+                bpe_offsets.push_back(len);
+            }
+            _prev_end = end;
+            return len;
+        };
+
+        for (size_t pos = offset_ini; pos < offset_end; /*pos++*/ ) {
+            const uint32_t cpt = _get_cpt(pos);
+            const auto flags = _get_flags(pos);
+
+            // regex: (?i:'s|'t|'re|'ve|'m|'ll|'d) // case insensitive
+            if (cpt == '\'' && pos+1 < offset_end) {
+                uint32_t cpt_next = unicode_tolower(_get_cpt(pos+1));
+                if (cpt_next == 0x017F) {
+                    cpt_next = 's'; // Unicode case-folding of long s
+                }
+                if (cpt_next == 's' || cpt_next == 't' || cpt_next == 'm' || cpt_next == 'd') {
+                    pos += _add_token(pos+2);
+                    continue;
+                }
+                if (pos+2 < offset_end) {
+                    uint32_t cpt_next_next = unicode_tolower(_get_cpt(pos+2));
+                    if ((cpt_next == 'r' && cpt_next_next == 'e') ||
+                        (cpt_next == 'v' && cpt_next_next == 'e') ||
+                        (cpt_next == 'l' && cpt_next_next == 'l')) {
+                        pos += _add_token(pos+3);
+                        continue;
+                    }
+                }
+            }
+
+            // regex: [^\r\n\p{L}\p{N}]?(?:\p{L}|\p{M}|\u200C|\u200D)+
+            if (!(cpt == '\r' || cpt == '\n' || flags.is_number)) {
+                if (_is_k2_letter(pos) || _is_k2_letter(pos+1)) {  // one or more letters/marks/ZWNJ/ZWJ
+                    pos++;
+                    while (_is_k2_letter(pos)) {
+                        pos++;
+                    }
+                    _add_token(pos);
+                    continue;
+                }
+            }
+
+            // regex: \p{N}{1,3}
+            if (flags.is_number) {
+                size_t ini = pos;
+                while (_get_flags(pos).is_number) {
+                    if (++pos - ini >= 3 ) {
+                        _add_token(pos);
+                        ini = pos;
+                    }
+                }
+                _add_token(pos);
+                continue;
+            }
+
+            // regex: <space>?[^\s\p{L}\p{N}]+[\r\n]*
+            auto flags2 = (cpt == ' ' ? _get_flags(pos+1) : flags);
+            if (!(flags2.is_whitespace | flags2.is_letter | flags2.is_number) && flags.as_uint()) {
+                pos += (cpt == ' ');
+                while (!(flags2.is_whitespace | flags2.is_letter | flags2.is_number) && flags2.as_uint()) {
+                    flags2 = _get_flags(++pos);
+                }
+                uint32_t cpt2 = _get_cpt(pos);
+                while (cpt2 == '\r' || cpt2 == '\n') {
+                    cpt2 = _get_cpt(++pos);
+                }
+                _add_token(pos);
+                continue;
+            }
+
+            size_t num_whitespaces = 0;
+            size_t last_end_r_or_n = 0;
+            while (_get_flags(pos+num_whitespaces).is_whitespace) {
+                uint32_t cpt2 = _get_cpt(pos+num_whitespaces);
+                if (cpt2 == '\r' || cpt2 == '\n') {
+                    last_end_r_or_n = pos + num_whitespaces + 1;
+                }
+                num_whitespaces++;
+            }
+
+            // regex: \s*[\r\n]+
+            if (last_end_r_or_n > 0) {
+                pos = last_end_r_or_n;
+                _add_token(pos);
+                continue;
+            }
+
+            // regex: \s+(?!\S)
+            if (num_whitespaces > 1 && _get_cpt(pos+num_whitespaces) != OUT_OF_RANGE) {
+                pos += num_whitespaces - 1;
+                _add_token(pos);
+                continue;
+            }
+
+            // regex: \s+
+            if (num_whitespaces > 0) {
+                pos += num_whitespaces;
+                _add_token(pos);
+                continue;
+            }
+
+            // no matches
+            _add_token(++pos);
+        }
+    }
+
+    return bpe_offsets;
+}
+
 // Qwen2 system regex: "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?\\p{L}+|\\p{N}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+"
 static std::vector<size_t> unicode_regex_split_custom_qwen2(const std::string & text, const std::vector<size_t> & offsets) {
     std::vector<size_t> bpe_offsets; // store the offset of each word
@@ -1062,6 +1209,11 @@ static std::vector<size_t> unicode_regex_split_custom(const std::string & text,
     } else if (
            regex_expr == "(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+") {
         bpe_offsets = unicode_regex_split_custom_qwen35(text, offsets);
+    } else if (
+            regex_expr == "(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])|[^\\r\\n\\p{L}\\p{N}]?(?:\\p{L}|\\p{M}|\\u200C|\\u200D)+|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+") {
+        // K2-Horizon: llama3 splitter with marks + ZWNJ/ZWJ inside letter runs
+        // (the generic std::regex fallback cannot parse \p{..} on MSVC)
+        bpe_offsets = unicode_regex_split_custom_k2_horizon(text, offsets);
     } else if (regex_expr == "\\p{Han}+") {
         // K2's first pattern - handle all K2 patterns together
         bpe_offsets = unicode_regex_split_custom_kimi_k2(text, offsets);
@@ -1214,6 +1366,10 @@ bool unicode_cpt_is_han(uint32_t cpt) {
 }

 std::vector<std::string> unicode_regex_split(const std::string & text, const std::vector<std::string> & regex_exprs, bool byte_encode) {
+    if (text.empty()) {
+        return {};
+    }
+
     // unicode categories
     static const std::map<std::string, int> k_ucat_enum = {
         { "\\p{N}", unicode_cpt_flags::NUMBER },
diff --git a/tests/fusion/MTL.csv b/tests/fusion/MTL.csv
index 9d771ddce..e839e20ce 100644
--- a/tests/fusion/MTL.csv
+++ b/tests/fusion/MTL.csv
@@ -126,6 +126,10 @@ internlm2           ,0   ,any     ,RMS_NORM+MUL                ,      5
 jais                ,0   ,any     ,NORM+MUL+ADD                ,      5
 jais2               ,0   ,any     ,NORM+MUL+ADD                ,      5
 jamba               ,0   ,any     ,RMS_NORM+MUL                ,      8
+k2-horizon          ,0   ,any     ,RMS_NORM+MUL                ,      4
+k2-horizon          ,0   ,any     ,ADD+ADD                     ,      1
+k2-horizon          ,0   ,any     ,MUL+ADD                     ,      2
+k2-horizon          ,0   ,any     ,RMS_NORM+MUL                ,      4
 kimi-k3             ,0   ,any     ,GATED_DELTA_NET+CPY         ,      1
 kimi-k3             ,0   ,any     ,MUL+ADD                     ,      2
 kimi-k3             ,0   ,any     ,RMS_NORM+MUL                ,     17
diff --git a/tests/test-chat.cpp b/tests/test-chat.cpp
index c7edf0c2d..9f96e1084 100644
--- a/tests/test-chat.cpp
+++ b/tests/test-chat.cpp
@@ -1537,6 +1537,11 @@ class peg_test_builder {
         return *this;
     }

+    peg_test_builder & chat_template_kwargs(const std::map<std::string, std::string> & kwargs) {
+        tc_.params.chat_template_kwargs = kwargs;
+        return *this;
+    }
+
     peg_test_builder & is_partial(bool val) {
         tc_.is_partial = val;
         return *this;
@@ -4875,6 +4880,207 @@ static void test_template_output_peg_parsers(bool detailed_debug) {
             .run();
     }

+    // K2 Horizon
+    {
+        auto tst = peg_tester("models/templates/IFM-K2-Horizon.jinja", detailed_debug);
+
+        tst.test("I'm\nthinking</ifm|think>Hello, world!\nWhat's up?")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .expect(message_assist_thoughts)
+            .expect_reconstruction()
+            .run();
+
+        tst.test("I'm\nthinking")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .expect_reasoning("I'm\nthinking")
+            .run();
+
+        tst.test("I'm\nthinking</ifm|think>Hello, world!\nWhat's up?")
+            .reasoning_format(COMMON_REASONING_FORMAT_NONE)
+            .expect_content("<ifm|think>\nI'm\nthinking</ifm|think>Hello, world!\nWhat's up?")
+            .run();
+
+        tst.test(
+               "I'm\nthinking</ifm|think><ifm|tool_calls>\n"
+               "<ifm|tool_call>special_function\n"
+               "<ifm|arg_key>arg1</ifm|arg_key>\n"
+               "<ifm|arg_value>1</ifm|arg_value>\n"
+               "</ifm|tool_call>\n"
+               "</ifm|tool_calls>")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .tools({ special_function_tool })
+            .expect(message_assist_call_thoughts)
+            .expect_reconstruction()
+            .run();
+
+        tst.test(
+               "I'm\nthinking</ifm|think><ifm|tool_calls>\n"
+               "<ifm|tool_call>special_function\n"
+               "<ifm|arg_key>arg1</ifm|arg_key>\n"
+               "<ifm|arg_type>integer</ifm|arg_type>\n"
+               "<ifm|arg_value>1</ifm|arg_value>\n"
+               "</ifm|tool_call>\n"
+               "</ifm|tool_calls>")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .tools({ special_function_tool })
+            .chat_template_kwargs({ { "tool_call_format", R"("xml_typed")" } })
+            .expect(message_assist_call_thoughts)
+            .expect_reconstruction()
+            .run();
+
+        tst.test(
+               "I'm\nthinking</ifm|think><ifm|tool_calls>\n"
+               "<ifm|tool_call>{\"name\": \"special_function\", \"arguments\": {\"arg1\": 1}}</ifm|tool_call>\n"
+               "</ifm|tool_calls>")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .tools({ special_function_tool })
+            .chat_template_kwargs({ { "tool_call_format", R"("json")" } })
+            .expect(message_assist_call_thoughts)
+            .expect_reconstruction()
+            .run();
+
+        tst.test(
+               "</ifm|think><ifm|tool_calls>\n"
+               "<ifm|tool_call>empty_args\n"
+               "</ifm|tool_call>\n"
+               "</ifm|tool_calls>")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .tools({ empty_args_tool })
+            .expect(simple_assist_msg("", "", "empty_args", "{}"))
+            .run();
+
+        tst.test(
+               "</ifm|think><ifm|tool_calls>\n"
+               "<ifm|tool_call>get_time\n"
+               "<ifm|arg_key>city</ifm|arg_key>\n"
+               "<ifm|arg_value>Paris</ifm|arg_value>\n"
+               "</ifm|tool_call>\n"
+               "<ifm|tool_call>get_time\n"
+               "<ifm|arg_key>city</ifm|arg_key>\n"
+               "<ifm|arg_value>Rome</ifm|arg_value>\n"
+               "</ifm|tool_call>\n"
+               "</ifm|tool_calls>")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .parallel_tool_calls(true)
+            .tools({ get_time_tool })
+            .expect_tool_calls({
+                { "get_time", R"({"city": "Paris"})", {} },
+                { "get_time", R"({"city": "Rome"})", {} },
+            })
+            .run();
+
+        tst.test(
+               "</ifm|think><ifm|tool_calls>\n"
+               "<ifm|tool_call>tool_2req_4opt\n"
+               "<ifm|arg_key>req2</ifm|arg_key>\n"
+               "<ifm|arg_value>7</ifm|arg_value>\n"
+               "<ifm|arg_key>req1</ifm|arg_key>\n"
+               "<ifm|arg_value>hello</ifm|arg_value>\n"
+               "</ifm|tool_call>\n"
+               "</ifm|tool_calls>")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .tools({ tool_2req_4opt })
+            .expect_tool_calls({ { "tool_2req_4opt", R"({"req2": 7, "req1": "hello"})", {} } })
+            .run();
+
+        for (const std::string value : { "true", "42", "null", "[]", R"("quoted")", "{not valid json" }) {
+            tst.test(
+                   "</ifm|think><ifm|tool_calls>\n"
+                   "<ifm|tool_call>set_union\n"
+                   "<ifm|arg_key>value</ifm|arg_key>\n"
+                   "<ifm|arg_value>" + value + "</ifm|arg_value>\n"
+                   "<ifm|arg_key>amount</ifm|arg_key>\n"
+                   "<ifm|arg_value>42</ifm|arg_value>\n"
+                   "</ifm|tool_call>\n"
+                   "</ifm|tool_calls>")
+                .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+                .tools({ string_union_tool })
+                .expect_tool_calls({ { "set_union", json({ { "value", value }, { "amount", 42 } }).dump(), {} } })
+                .run();
+        }
+
+        tst.test(
+               "</ifm|think><ifm|tool_calls>\n"
+               "<ifm|tool_call>set_union\n"
+               "<ifm|arg_key>value</ifm|arg_key>\n"
+               "<ifm|arg_value>{\"a\": 1}</ifm|arg_value>\n"
+               "<ifm|arg_key>amount</ifm|arg_key>\n"
+               "<ifm|arg_value>2 dollars</ifm|arg_value>\n"
+               "</ifm|tool_call>\n"
+               "</ifm|tool_calls>")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .tools({ string_union_tool })
+            .expect_tool_calls({ { "set_union", R"({"value": {"a": 1}, "amount": "2 dollars"})", {} } })
+            .run();
+
+        tst.test(
+               "I'm\nthinking</ifm|think><ifm|tool_calls>\n"
+               "<ifm|tool_call>special_function\n"
+               "<ifm|arg_key>arg1</ifm|arg_key>\n"
+               "<ifm|arg_value>1</ifm|arg_value>\n"
+               "</ifm|tool_call>\n"
+               "</ifm|tool_calls>")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .tools({ special_function_tool })
+            .tool_choice(COMMON_CHAT_TOOL_CHOICE_REQUIRED)
+            .expect(message_assist_call_thoughts)
+            .run();
+
+        tst.test(
+               "I'm\nthinking</ifm|think><ifm|tool_calls>\n"
+               "<ifm|tool_call>special_function\n"
+               "<ifm|arg_key>arg1</ifm|arg_key>\n"
+               "<ifm|arg_value>1</ifm|arg_value>\n"
+               "</ifm|tool_call>\n"
+               "</ifm|tool_calls>")
+            .reasoning_format(COMMON_REASONING_FORMAT_NONE)
+            .tools({ special_function_tool })
+            .tool_choice(COMMON_CHAT_TOOL_CHOICE_REQUIRED)
+            .expect_content("<ifm|think>\nI'm\nthinking</ifm|think>")
+            .expect_tool_calls({ { "special_function", R"({"arg1": 1})", {} } })
+            .run();
+
+        tst.test("I'm\nthinking</ifm|think>Hello, world!\nWhat's up?")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .tools({ special_function_tool })
+            .expect(message_assist_thoughts)
+            .run();
+
+        const std::string answer_schema = R"({"type":"object","properties":{"answer":{"type":"integer"}},"required":["answer"]})";
+
+        tst.test("Let me calculate.</ifm|think>{\"answer\":42}")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .json_schema(answer_schema)
+            .expect_reasoning("Let me calculate.")
+            .expect_content(R"({"answer":42})")
+            .run();
+
+        tst.test("Let me calculate.</ifm|think>{\"answer\":42}")
+            .reasoning_format(COMMON_REASONING_FORMAT_NONE)
+            .json_schema(answer_schema)
+            .expect_content(R"({"answer":42})")
+            .run();
+
+        tst.test("42}")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .json_schema(answer_schema)
+            .messages({ message_user, simple_assist_msg("{\"answer\":", "Calculated.") })
+            .add_generation_prompt(false)
+            .continue_final_message(COMMON_CHAT_CONTINUATION_CONTENT)
+            .expect_reasoning("Calculated.")
+            .expect_content(R"({"answer":42})")
+            .run();
+
+        tst.test(" thinking</ifm|think>Hello, world!\nWhat's up?")
+            .reasoning_format(COMMON_REASONING_FORMAT_AUTO)
+            .messages({ message_user, message_assist_prefill_reasoning })
+            .add_generation_prompt(false)
+            .continue_final_message(COMMON_CHAT_CONTINUATION_REASONING)
+            .expect_reasoning("I'm thinking")
+            .expect_content("Hello, world!\nWhat's up?")
+            .run();
+    }
+
     // Kimi-K3 tests - custom parser
     // Unique feature: XTML tags built from <|open|>/<|close|>/<|sep|>, and a
     // generation prompt that leaves the think section already open.
diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp
index e74d8767c..54122629a 100644
--- a/tests/test-llama-archs.cpp
+++ b/tests/test-llama-archs.cpp
@@ -195,6 +195,11 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
     ms.add_kv(LLM_KV_BLOCK_COUNT,               n_layer);
     ms.add_kv(LLM_KV_LEADING_DENSE_BLOCK_COUNT, uint32_t(1));

+    if (arch == LLM_ARCH_K2_HORIZON) {
+        ms.add_kv(LLM_KV_ROPE_SCALING_YARN_BETA_FAST, 128.0f);
+        ms.add_kv(LLM_KV_ROPE_SCALING_YARN_BETA_SLOW,   4.0f);
+    }
+
     if (arch == LLM_ARCH_NEMOTRON_H || arch == LLM_ARCH_NEMOTRON_H_MOE) {
         std::vector<uint32_t> n_ff_per_layer;
         n_ff_per_layer.reserve(n_layer);
@@ -432,6 +437,10 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
         ms.add_kv(LLM_KV_EXPERT_GATING_FUNC,         arch == LLM_ARCH_DEEPSEEK4 ? uint32_t(4) : uint32_t(2)); // sqrtsoftplus : sigmoid
         ms.add_kv(LLM_KV_EXPERT_GROUP_SCALE,         1.0f);
         ms.add_kv(LLM_KV_EXPERTS_PER_GROUP,          uint32_t(1));
+        if (arch == LLM_ARCH_K2_HORIZON) {
+            ms.add_kv(LLM_KV_ATTENTION_VALUE_EXPERT_COUNT,      uint32_t(2));
+            ms.add_kv(LLM_KV_ATTENTION_VALUE_EXPERT_USED_COUNT, uint32_t(2));
+        }
     }

     ms.add_kv(LLM_KV_POSNET_EMBEDDING_LENGTH,   n_embd);
@@ -707,6 +716,7 @@ static bool moe_implemented(const llm_arch arch) {
         case LLM_ARCH_GRANITE_MOE:
         case LLM_ARCH_MISTRAL3:
         case LLM_ARCH_LLAMA_EMBED:
+        case LLM_ARCH_K2_HORIZON:
             return true;
         default:
             return false;