Commit 25747b08e for llama.cpp
commit 25747b08e7a0f9a59a2089ce6b98d2229b76042a
Author: Pascal <admin@serveurperso.com>
Date: Wed Sep 30 11:17:02 2026 +0200
openvino: serve GET_ROWS on a weight view from the base Constant (#28381)
* openvino: serve GET_ROWS on a weight view from the base Constant
Resolve view_src when collecting weight Constants so a view over a
quantized weight no longer becomes a dynamic typed Parameter, and fold
the row offset of the view into the gather indices instead of slicing
the dequantization subgraph.
* openvino: lift the quantized GET_ROWS view rejection
The supports_op rejection of a quantized src0 view with a nonzero
offset keeps the vs0 GET_ROWS cases of #28253 away from OpenVINO.
The weight view now resolves to the base Constant with the row offset
folded into the gather indices, so the rejection goes away.
diff --git a/ggml/src/ggml-openvino/ggml-decoder.cpp b/ggml/src/ggml-openvino/ggml-decoder.cpp
index cd06b22e8..cec32f6df 100644
--- a/ggml/src/ggml-openvino/ggml-decoder.cpp
+++ b/ggml/src/ggml-openvino/ggml-decoder.cpp
@@ -1165,6 +1165,11 @@ void GgmlOvDecoder::compute_model_inputs() {
if (m_model_weights.find(src_name) != m_model_weights.end()) {
continue;
}
+ // A view over a weight is served by the base tensor's Constant, never by a Parameter.
+ if (src->view_src != nullptr &&
+ m_model_weights.find(get_tensor_ov_name(m_cgraph, src->view_src)) != m_model_weights.end()) {
+ continue;
+ }
bool is_intermediate_node = false;
for (const auto & node_info : m_node_info_list) {
@@ -1300,19 +1305,19 @@ std::map<std::string, std::shared_ptr<ov::Node>> GgmlOvDecoder::create_weight_no
continue;
}
- std::string src_name = get_tensor_ov_name(cgraph, src);
- if (is_rope_freqs_weight(src, node)) {
+ // A view over a weight is served by the base tensor's Constant.
+ ggml_tensor * base = src->view_src ? src->view_src : src;
+ std::string src_name = get_tensor_ov_name(cgraph, base);
+ if (is_rope_freqs_weight(base, node)) {
src_name = "rope_freqs.weight";
}
- if (!src->view_src) {
- ggml_backend_buffer * buffer = src->buffer;
- if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(src->type) ||
- is_mul_mat_id_expert_weight(node, i)) {
- if (model_weights.find(src_name) == model_weights.end()) {
- auto weight_node = create_weight_node(src, naive);
- weight_node->set_friendly_name(src_name);
- model_weights[src_name] = weight_node;
- }
+ ggml_backend_buffer * buffer = base->buffer;
+ if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(base->type) ||
+ is_mul_mat_id_expert_weight(node, i)) {
+ if (model_weights.find(src_name) == model_weights.end()) {
+ auto weight_node = create_weight_node(base, naive);
+ weight_node->set_friendly_name(src_name);
+ model_weights[src_name] = weight_node;
}
}
}
@@ -1341,15 +1346,14 @@ std::set<std::string> GgmlOvDecoder::collect_weight_names(ggml_cgraph * cgraph)
if (src == nullptr) {
continue;
}
- std::string src_name(src->name);
- if (is_rope_freqs_weight(src, node)) {
+ const ggml_tensor * base = src->view_src ? src->view_src : src;
+ std::string src_name(base->name);
+ if (is_rope_freqs_weight(base, node)) {
src_name = "rope_freqs.weight";
}
- if (!src->view_src) {
- ggml_backend_buffer * buffer = src->buffer;
- if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(src->type)) {
- names.insert(src_name);
- }
+ ggml_backend_buffer * buffer = base->buffer;
+ if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(base->type)) {
+ names.insert(src_name);
}
}
}
diff --git a/ggml/src/ggml-openvino/ggml-openvino.cpp b/ggml/src/ggml-openvino/ggml-openvino.cpp
index b2d81df97..e15f1c1ab 100644
--- a/ggml/src/ggml-openvino/ggml-openvino.cpp
+++ b/ggml/src/ggml-openvino/ggml-openvino.cpp
@@ -1179,10 +1179,6 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
if (op->ne[3] != 1) {
return {false, "GET_ROWS/SET_ROWS with ne[3] != 1 (ne[3]=" + std::to_string(op->ne[3]) + ") is not supported"};
}
- if (op->op == GGML_OP_GET_ROWS && ggml_is_quantized(op->src[0]->type) &&
- op->src[0]->view_src != nullptr && op->src[0]->view_offs != 0) {
- return {false, "GET_ROWS with a nonzero quantized src0 view offset is not supported"};
- }
if (op->op == GGML_OP_GET_ROWS && ggml_openvino_get_device_name() == "GPU" &&
op->src[0]->type == GGML_TYPE_BF16) {
return {false, "GET_ROWS with BF16 src0 is not supported on GPU"};
diff --git a/ggml/src/ggml-openvino/openvino/op/get_rows.cpp b/ggml/src/ggml-openvino/openvino/op/get_rows.cpp
index 2ac8ec0ba..d122722b7 100644
--- a/ggml/src/ggml-openvino/openvino/op/get_rows.cpp
+++ b/ggml/src/ggml-openvino/openvino/op/get_rows.cpp
@@ -5,6 +5,7 @@
#include <climits>
#include <openvino/core/node.hpp>
#include <openvino/core/node_output.hpp>
+#include <openvino/op/add.hpp>
#include <openvino/op/broadcast.hpp>
#include <openvino/op/concat.hpp>
#include <openvino/op/constant.hpp>
@@ -24,7 +25,23 @@ OutputVector translate_get_rows(const NodeContext & context) {
num_inputs_check(context, 2, 2);
Output<Node> res;
- auto data = process_view_input_new(context, 0);
+ Output<Node> data;
+ int64_t row_offset = 0;
+ if (context.get_view_input_size(0) > 0 && context.get_input(0).get_partial_shape().rank() == 2) {
+ // A row range view over a 2D weight Constant folds into the gather indices,
+ // which keeps the dequantization subgraph intact for the plugins.
+ const auto view_shape = context.get_view_input_ggml_shape(0, 0);
+ const auto view_stride = context.get_view_input_stride(0, 0);
+ const size_t view_offset = context.get_view_input_offset(0, 0);
+ const size_t row_bytes = view_stride[2];
+ data = context.get_input(0);
+ FRONT_END_OP_CONVERSION_CHECK(row_bytes > 0 && view_offset % row_bytes == 0 &&
+ data.get_partial_shape()[1].compatible(view_shape[3]),
+ "GET_ROWS: view over a weight must be a row range");
+ row_offset = static_cast<int64_t>(view_offset / row_bytes);
+ } else {
+ data = process_view_input_new(context, 0);
+ }
auto op_case = context.get_op_case();
ov::Output<ov::Node> indices;
@@ -51,6 +68,10 @@ OutputVector translate_get_rows(const NodeContext & context) {
// data[x,y] ind[1,1,1,x'] normal case
indices =
std::make_shared<ov::op::v0::Squeeze>(indices, ov::op::v0::Constant::create(ov::element::i64, {2}, {0, 1}));
+ if (row_offset != 0) {
+ indices = std::make_shared<ov::op::v1::Add>(
+ indices, ov::op::v0::Constant::create(indices.get_element_type(), {}, {row_offset}));
+ }
if (data.get_partial_shape().rank() == 4) {
if (!(data.get_partial_shape()[1].is_dynamic()) && data.get_partial_shape()[1].get_length() == 1) {
// Work-around for a bug in ov cpu plugin for test-backend-ops