Commit d4d82d67f for llama.cpp
commit d4d82d67f48bb8f51e9a3518063e49d9ee54f58b
Author: Mark Todorovich <mtodorov@qti.qualcomm.com>
Date: Thu Oct 8 23:04:09 2026 -0700
hexagon: fix IM2COL patch-embed DMA ring overflow (#30189)
* hexagon: fix IM2COL patch-embed DMA ring overflow
The exact-tiling (stride == kernel, no pad/dilation) IM2COL DMA kernel
issues IC*KH DDR->VTCM descriptors per output row without checking the
return value of dma_queue_push(), and then pops IC*KH times. The per-thread
DMA ring holds 256 entries and a push into a full ring returns false
and drops the transfer, so for IC*KH > 255 the remaining rows of the
VTCM staging buffer were never written and stale data (often NaN/inf)
leaked into the output.
Solution is to retire the oldest descriptor when the ring is full, just as the blocked
kernel in the same file already does, and wait with dma_queue_flush().
For testing, added exact-tiling test cases with IC*KH > 256 (2D 1x1, 2D 2x2 patch
embed and 1D, F16 and F32 dst), which fail on HTP without this fix.
* Apply suggestion from @max-krasnyansky
---------
Co-authored-by: Max Krasnyansky <maxk@qti.qualcomm.com>
diff --git a/ggml/src/ggml-hexagon/htp/im2col-ops.c b/ggml/src/ggml-hexagon/htp/im2col-ops.c
index 2e05cf3e1..b99d711c9 100644
--- a/ggml/src/ggml-hexagon/htp/im2col-ops.c
+++ b/ggml/src/ggml-hexagon/htp/im2col-ops.c
@@ -351,12 +351,15 @@ IM2COL_BLOCKED_DMA_BODY(im2col_blocked_dma_f32_thread, float, hvx_copy_f32_uu,
const dma_addr_t vsrc = ok \
? (src_data + (size_t) ((in * IC + iic) * IH + iih) * IW * sizeof(float)) \
: src_data; \
- dma_queue_push(dma_q, dma_make_data(vdst, vsrc), \
- IW * sizeof(float), IW * sizeof(float), IW * sizeof(float), ok ? 1 : 0); \
+ /* IC*KH descriptors per row can exceed the ring capacity: retire the oldest when full */ \
+ while (!dma_queue_push(dma_q, dma_make_data(vdst, vsrc), \
+ IW * sizeof(float), IW * sizeof(float), IW * sizeof(float), \
+ ok ? 1 : 0)) { \
+ dma_queue_pop(dma_q); \
+ } \
} \
} \
- for (uint32_t i = 0; i < IC * KH; i++) \
- dma_queue_pop(dma_q); \
+ dma_queue_flush(dma_q); \
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, r); \
for (uint32_t iow = 0; iow < OW; iow++) { \
DST_CTYPE * dst_patch = dstb + (uint64_t) iow * patch_stride; \
diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp
index 7b7e6cae1..9a8d46bfd 100644
--- a/tests/test-backend-ops.cpp
+++ b/tests/test-backend-ops.cpp
@@ -9789,6 +9789,14 @@ static std::vector<std::unique_ptr<test_case>> make_test_cases_eval() {
test_cases.emplace_back(new test_im2col(GGML_TYPE_F32, GGML_TYPE_F32, GGML_TYPE_F32, {2, 2, 1536, 729}, {2, 2, 1536, 4096}, 1, 1, 0, 0, 1, 1, true));
test_cases.emplace_back(new test_im2col(GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_F16, {128, 128, 1, 2}, {32, 33, 1, 2}, 1, 1, 1, 1, 1, 1, true));
test_cases.emplace_back(new test_im2col(GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_F16, {128, 128, 2, 1}, {33, 34, 2, 1}, 1, 1, 1, 1, 1, 1, true));
+ // non-overlapping (stride == kernel, no pad/dilation) with IC*KH well above 256, e.g. SD1.5 1x1 convs
+ for (ggml_type dst_type : {GGML_TYPE_F16, GGML_TYPE_F32}) {
+ test_cases.emplace_back(new test_im2col(GGML_TYPE_F32, GGML_TYPE_F16, dst_type, {64, 64, 320, 1}, {1, 1, 320, 1}, 1, 1, 0, 0, 1, 1, true));
+ test_cases.emplace_back(new test_im2col(GGML_TYPE_F32, GGML_TYPE_F16, dst_type, {16, 16, 1280, 2}, {1, 1, 1280, 1}, 1, 1, 0, 0, 1, 1, true));
+ test_cases.emplace_back(new test_im2col(GGML_TYPE_F32, GGML_TYPE_F16, dst_type, {8, 8, 2560, 1}, {1, 1, 2560, 1}, 1, 1, 0, 0, 1, 1, true));
+ test_cases.emplace_back(new test_im2col(GGML_TYPE_F32, GGML_TYPE_F16, dst_type, {32, 32, 192, 1}, {2, 2, 192, 1}, 2, 2, 0, 0, 1, 1, true));
+ test_cases.emplace_back(new test_im2col(GGML_TYPE_F32, GGML_TYPE_F16, dst_type, {64, 512, 1, 1}, {2, 512, 1, 1}, 2, 0, 0, 0, 1, 0, false));
+ }
// im2col 3D
test_cases.emplace_back(new test_im2col_3d(GGML_TYPE_F32, GGML_TYPE_F32, GGML_TYPE_F32));