Commit 25747b08e for llama.cpp

commit 25747b08e7a0f9a59a2089ce6b98d2229b76042a
Author: Pascal <admin@serveurperso.com>
Date:   Wed Sep 30 11:17:02 2026 +0200

    openvino: serve GET_ROWS on a weight view from the base Constant (#28381)

    * openvino: serve GET_ROWS on a weight view from the base Constant

    Resolve view_src when collecting weight Constants so a view over a
    quantized weight no longer becomes a dynamic typed Parameter, and fold
    the row offset of the view into the gather indices instead of slicing
    the dequantization subgraph.

    * openvino: lift the quantized GET_ROWS view rejection

    The supports_op rejection of a quantized src0 view with a nonzero
    offset keeps the vs0 GET_ROWS cases of #28253 away from OpenVINO.
    The weight view now resolves to the base Constant with the row offset
    folded into the gather indices, so the rejection goes away.

diff --git a/ggml/src/ggml-openvino/ggml-decoder.cpp b/ggml/src/ggml-openvino/ggml-decoder.cpp
index cd06b22e8..cec32f6df 100644
--- a/ggml/src/ggml-openvino/ggml-decoder.cpp
+++ b/ggml/src/ggml-openvino/ggml-decoder.cpp
@@ -1165,6 +1165,11 @@ void GgmlOvDecoder::compute_model_inputs() {
             if (m_model_weights.find(src_name) != m_model_weights.end()) {
                 continue;
             }
+            // A view over a weight is served by the base tensor's Constant, never by a Parameter.
+            if (src->view_src != nullptr &&
+                m_model_weights.find(get_tensor_ov_name(m_cgraph, src->view_src)) != m_model_weights.end()) {
+                continue;
+            }

             bool is_intermediate_node = false;
             for (const auto & node_info : m_node_info_list) {
@@ -1300,19 +1305,19 @@ std::map<std::string, std::shared_ptr<ov::Node>> GgmlOvDecoder::create_weight_no
                 continue;
             }

-            std::string src_name = get_tensor_ov_name(cgraph, src);
-            if (is_rope_freqs_weight(src, node)) {
+            // A view over a weight is served by the base tensor's Constant.
+            ggml_tensor * base = src->view_src ? src->view_src : src;
+            std::string src_name = get_tensor_ov_name(cgraph, base);
+            if (is_rope_freqs_weight(base, node)) {
                 src_name = "rope_freqs.weight";
             }
-            if (!src->view_src) {
-                ggml_backend_buffer * buffer = src->buffer;
-                if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(src->type) ||
-                    is_mul_mat_id_expert_weight(node, i)) {
-                    if (model_weights.find(src_name) == model_weights.end()) {
-                        auto weight_node = create_weight_node(src, naive);
-                        weight_node->set_friendly_name(src_name);
-                        model_weights[src_name] = weight_node;
-                    }
+            ggml_backend_buffer * buffer = base->buffer;
+            if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(base->type) ||
+                is_mul_mat_id_expert_weight(node, i)) {
+                if (model_weights.find(src_name) == model_weights.end()) {
+                    auto weight_node = create_weight_node(base, naive);
+                    weight_node->set_friendly_name(src_name);
+                    model_weights[src_name] = weight_node;
                 }
             }
         }
@@ -1341,15 +1346,14 @@ std::set<std::string> GgmlOvDecoder::collect_weight_names(ggml_cgraph * cgraph)
             if (src == nullptr) {
                 continue;
             }
-            std::string src_name(src->name);
-            if (is_rope_freqs_weight(src, node)) {
+            const ggml_tensor * base = src->view_src ? src->view_src : src;
+            std::string src_name(base->name);
+            if (is_rope_freqs_weight(base, node)) {
                 src_name = "rope_freqs.weight";
             }
-            if (!src->view_src) {
-                ggml_backend_buffer * buffer = src->buffer;
-                if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(src->type)) {
-                    names.insert(src_name);
-                }
+            ggml_backend_buffer * buffer = base->buffer;
+            if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(base->type)) {
+                names.insert(src_name);
             }
         }
     }
diff --git a/ggml/src/ggml-openvino/ggml-openvino.cpp b/ggml/src/ggml-openvino/ggml-openvino.cpp
index b2d81df97..e15f1c1ab 100644
--- a/ggml/src/ggml-openvino/ggml-openvino.cpp
+++ b/ggml/src/ggml-openvino/ggml-openvino.cpp
@@ -1179,10 +1179,6 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
         if (op->ne[3] != 1) {
             return {false, "GET_ROWS/SET_ROWS with ne[3] != 1 (ne[3]=" + std::to_string(op->ne[3]) + ") is not supported"};
         }
-        if (op->op == GGML_OP_GET_ROWS && ggml_is_quantized(op->src[0]->type) &&
-            op->src[0]->view_src != nullptr && op->src[0]->view_offs != 0) {
-            return {false, "GET_ROWS with a nonzero quantized src0 view offset is not supported"};
-        }
         if (op->op == GGML_OP_GET_ROWS && ggml_openvino_get_device_name() == "GPU" &&
             op->src[0]->type == GGML_TYPE_BF16) {
             return {false, "GET_ROWS with BF16 src0 is not supported on GPU"};
diff --git a/ggml/src/ggml-openvino/openvino/op/get_rows.cpp b/ggml/src/ggml-openvino/openvino/op/get_rows.cpp
index 2ac8ec0ba..d122722b7 100644
--- a/ggml/src/ggml-openvino/openvino/op/get_rows.cpp
+++ b/ggml/src/ggml-openvino/openvino/op/get_rows.cpp
@@ -5,6 +5,7 @@
 #include <climits>
 #include <openvino/core/node.hpp>
 #include <openvino/core/node_output.hpp>
+#include <openvino/op/add.hpp>
 #include <openvino/op/broadcast.hpp>
 #include <openvino/op/concat.hpp>
 #include <openvino/op/constant.hpp>
@@ -24,7 +25,23 @@ OutputVector translate_get_rows(const NodeContext & context) {
     num_inputs_check(context, 2, 2);

     Output<Node> res;
-    auto data = process_view_input_new(context, 0);
+    Output<Node> data;
+    int64_t row_offset = 0;
+    if (context.get_view_input_size(0) > 0 && context.get_input(0).get_partial_shape().rank() == 2) {
+        // A row range view over a 2D weight Constant folds into the gather indices,
+        // which keeps the dequantization subgraph intact for the plugins.
+        const auto view_shape = context.get_view_input_ggml_shape(0, 0);
+        const auto view_stride = context.get_view_input_stride(0, 0);
+        const size_t view_offset = context.get_view_input_offset(0, 0);
+        const size_t row_bytes = view_stride[2];
+        data = context.get_input(0);
+        FRONT_END_OP_CONVERSION_CHECK(row_bytes > 0 && view_offset % row_bytes == 0 &&
+                                          data.get_partial_shape()[1].compatible(view_shape[3]),
+                                      "GET_ROWS: view over a weight must be a row range");
+        row_offset = static_cast<int64_t>(view_offset / row_bytes);
+    } else {
+        data = process_view_input_new(context, 0);
+    }

     auto op_case = context.get_op_case();
     ov::Output<ov::Node> indices;
@@ -51,6 +68,10 @@ OutputVector translate_get_rows(const NodeContext & context) {
     // data[x,y] ind[1,1,1,x'] normal case
     indices =
         std::make_shared<ov::op::v0::Squeeze>(indices, ov::op::v0::Constant::create(ov::element::i64, {2}, {0, 1}));
+    if (row_offset != 0) {
+        indices = std::make_shared<ov::op::v1::Add>(
+            indices, ov::op::v0::Constant::create(indices.get_element_type(), {}, {row_offset}));
+    }
     if (data.get_partial_shape().rank() == 4) {
         if (!(data.get_partial_shape()[1].is_dynamic()) && data.get_partial_shape()[1].get_length() == 1) {
             // Work-around for a bug in ov cpu plugin for test-backend-ops