Commit 70815103c for llama.cpp

commit 70815103c8b251bdf5abba4eca12749eab1de686
Author: kurquhar <kurquhar@qti.qualcomm.com>
Date:   Wed Oct 7 16:57:09 2026 -0700

    hexagon: improved GELU accuracy (#30104)

    Assisted-by: OpenCode

diff --git a/ggml/src/ggml-hexagon/ggml-hexagon.cpp b/ggml/src/ggml-hexagon/ggml-hexagon.cpp
index 8c650b85c..f0d768388 100644
--- a/ggml/src/ggml-hexagon/ggml-hexagon.cpp
+++ b/ggml/src/ggml-hexagon/ggml-hexagon.cpp
@@ -7677,7 +7677,7 @@ static htp_op_code op_remap_to_htp(const ggml_tensor * t) {
             switch (ggml_get_unary_op(t)) {
                 case GGML_UNARY_OP_SILU:       return HTP_OP_UNARY_SILU;
                 case GGML_UNARY_OP_GELU:       return HTP_OP_UNARY_GELU;
-                case GGML_UNARY_OP_GELU_QUICK: return HTP_OP_UNARY_GELU;
+                case GGML_UNARY_OP_GELU_QUICK: return HTP_OP_UNARY_GELU_QUICK;
                 case GGML_UNARY_OP_GELU_ERF:   return HTP_OP_UNARY_GELU_ERF;
                 case GGML_UNARY_OP_SIGMOID:    return HTP_OP_UNARY_SIGMOID;
                 case GGML_UNARY_OP_NEG:        return HTP_OP_UNARY_NEG;
diff --git a/ggml/src/ggml-hexagon/htp/htp-ops.h b/ggml/src/ggml-hexagon/htp/htp-ops.h
index e47b42e64..5d2571fef 100644
--- a/ggml/src/ggml-hexagon/htp/htp-ops.h
+++ b/ggml/src/ggml-hexagon/htp/htp-ops.h
@@ -117,6 +117,7 @@ enum htp_op_code {
     HTP_OP_GLU_GEGLU_ERF,
     HTP_OP_POOL_2D,
     HTP_OP_POOL_1D,
+    HTP_OP_UNARY_GELU_QUICK,

     HTP_OP_INVALID
 };
diff --git a/ggml/src/ggml-hexagon/htp/hvx-erf.h b/ggml/src/ggml-hexagon/htp/hvx-erf.h
index 6eefc300f..b17c657a8 100644
--- a/ggml/src/ggml-hexagon/htp/hvx-erf.h
+++ b/ggml/src/ggml-hexagon/htp/hvx-erf.h
@@ -12,7 +12,7 @@ static __attribute__((noinline)) HVX_Vector hvx_vec_erf_f32(HVX_Vector x) {
     HVX_Vector t = hvx_vec_inverse_f32(hvx_vec_add_f32_f32(
         hvx_vec_splat_f32(1.0f), hvx_vec_mul_f32_f32(hvx_vec_splat_f32(0.3275911f), ax)));

-    HVX_Vector poly = hvx_vec_mul_f32_f32(hvx_vec_splat_f32(1.061405429f), t);
+    HVX_Vector poly = hvx_vec_splat_f32(1.061405429f);
     poly = hvx_vec_add_f32_f32(hvx_vec_splat_f32(-1.453152027f), hvx_vec_mul_f32_f32(poly, t));
     poly = hvx_vec_add_f32_f32(hvx_vec_splat_f32(1.421413741f), hvx_vec_mul_f32_f32(poly, t));
     poly = hvx_vec_add_f32_f32(hvx_vec_splat_f32(-0.284496736f), hvx_vec_mul_f32_f32(poly, t));
@@ -27,6 +27,7 @@ static __attribute__((noinline)) HVX_Vector hvx_vec_erf_f32(HVX_Vector x) {
     return result;
 }

+// GELU_ERF uses the erf definition.
 static inline HVX_Vector hvx_vec_gelu_erf_f32(HVX_Vector x) {
     const HVX_Vector scale = hvx_vec_splat_f32(0.7071067811865475f);
     const HVX_Vector half  = hvx_vec_splat_f32(0.5f);
diff --git a/ggml/src/ggml-hexagon/htp/main.c b/ggml/src/ggml-hexagon/htp/main.c
index 41f78b387..2dd1b3fc1 100644
--- a/ggml/src/ggml-hexagon/htp/main.c
+++ b/ggml/src/ggml-hexagon/htp/main.c
@@ -851,6 +851,7 @@ static int execute_op(struct htp_ops_context * octx) {
         case HTP_OP_UNARY_SIGMOID:
         case HTP_OP_UNARY_SILU:
         case HTP_OP_UNARY_GELU:
+        case HTP_OP_UNARY_GELU_QUICK:
         case HTP_OP_UNARY_GELU_ERF:
         case HTP_OP_UNARY_NEG:
         case HTP_OP_UNARY_EXP:
diff --git a/ggml/src/ggml-hexagon/htp/unary-ops.c b/ggml/src/ggml-hexagon/htp/unary-ops.c
index b63fd4c29..c557b0907 100644
--- a/ggml/src/ggml-hexagon/htp/unary-ops.c
+++ b/ggml/src/ggml-hexagon/htp/unary-ops.c
@@ -497,7 +497,72 @@ static void silu_f32(const void * restrict src,
     }
 }

-// gelu(x) = x * sigmoid(1.702 * x)  (quick/sigmoid approximation, matches CPU GELU_QUICK reference)
+// GELU uses the tanh approximation.
+static __attribute__((noinline)) HVX_Vector hvx_vec_gelu_f32(HVX_Vector x) {
+    const HVX_Vector half = hvx_vec_splat_f32(0.5f);
+    const HVX_Vector one  = hvx_vec_splat_f32(1.0f);
+
+    HVX_Vector inner = hvx_vec_mul_f32_f32(x, x);
+    inner = hvx_vec_mul_f32_f32(inner, hvx_vec_splat_f32(0.044715f));
+    inner = hvx_vec_add_f32_f32(inner, one);
+    inner = hvx_vec_mul_f32_f32(inner, x);
+    inner = hvx_vec_mul_f32_f32(inner, hvx_vec_splat_f32(0.7978845608028654f));
+
+    return hvx_vec_mul_f32_f32(hvx_vec_mul_f32_f32(half, x),
+                               hvx_vec_add_f32_f32(one, hvx_vec_tanh_f32(inner)));
+}
+
+static inline void hvx_gelu_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) {
+    assert((unsigned long) dst % 128 == 0);
+    assert((unsigned long) src % 128 == 0);
+
+    HVX_Vector * restrict vdst = (HVX_Vector *) dst;
+    HVX_Vector * restrict vsrc = (HVX_Vector *) src;
+
+    const uint32_t nvec = n / VLEN_FP32;
+    const uint32_t nloe = n % VLEN_FP32;
+
+    uint32_t i = 0;
+    _Pragma("unroll(4)")
+    for (; i < nvec; i++) {
+        vdst[i] = hvx_vec_gelu_f32(vsrc[i]);
+    }
+    if (nloe) {
+        hvx_vec_store_a(&vdst[i], nloe * sizeof(float), hvx_vec_gelu_f32(vsrc[i]));
+    }
+}
+
+// GELU_QUICK uses x * sigmoid(1.702 * x).
+static __attribute__((noinline)) HVX_Vector hvx_vec_gelu_quick_f32(HVX_Vector x) {
+    const HVX_Vector one     = hvx_vec_splat_f32(1.0f);
+    const HVX_Vector max_exp = hvx_vec_splat_f32(87.0f);
+    const HVX_Vector min_exp = hvx_vec_splat_f32(-87.0f);
+    const HVX_Vector scaled  = hvx_vec_mul_f32_f32(x, hvx_vec_splat_f32(1.702f));
+    const HVX_Vector sigmoid = hvx_vec_fast_sigmoid_f32_guard(scaled, one, max_exp, min_exp);
+
+    return hvx_vec_mul_f32_f32(x, sigmoid);
+}
+
+static inline void hvx_gelu_quick_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) {
+    assert((unsigned long) dst % 128 == 0);
+    assert((unsigned long) src % 128 == 0);
+
+    HVX_Vector * restrict vdst = (HVX_Vector *) dst;
+    HVX_Vector * restrict vsrc = (HVX_Vector *) src;
+
+    const uint32_t nvec = n / VLEN_FP32;
+    const uint32_t nloe = n % VLEN_FP32;
+
+    uint32_t i = 0;
+    _Pragma("unroll(4)")
+    for (; i < nvec; i++) {
+        vdst[i] = hvx_vec_gelu_quick_f32(vsrc[i]);
+    }
+    if (nloe) {
+        hvx_vec_store_a(&vdst[i], nloe * sizeof(float), hvx_vec_gelu_quick_f32(vsrc[i]));
+    }
+}
+
 static void gelu_f32(const void * restrict src,
                      void * restrict dst,
                      const uint32_t num_rows,
@@ -508,9 +573,21 @@ static void gelu_f32(const void * restrict src,
         const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned);
         uint8_t * restrict dst_local       = (uint8_t *)dst + (ir * dst_row_size_aligned);

-        hvx_mul_scalar_f32(dst_local, src_local, 1.702f, ne0);
-        hvx_sigmoid_f32_aa(dst_local, dst_local, ne0);
-        hvx_mul_f32_aaa(dst_local, src_local, dst_local, ne0);
+        hvx_gelu_f32_aa(dst_local, src_local, ne0);
+    }
+}
+
+static void gelu_quick_f32(const void * restrict src,
+                           void * restrict dst,
+                           const uint32_t num_rows,
+                           const struct htp_unary_context * uctx) {
+    htp_unary_op_preamble;
+
+    for (uint32_t ir = 0; ir < num_rows; ir++) {
+        const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned);
+        uint8_t * restrict dst_local       = (uint8_t *)dst + (ir * dst_row_size_aligned);
+
+        hvx_gelu_quick_f32_aa(dst_local, src_local, ne0);
     }
 }

@@ -774,9 +851,12 @@ static void tile_silu_f32(void * restrict dst, const void * restrict src, uint32

 static void tile_gelu_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) {
     (void) uctx;
-    hvx_mul_scalar_f32((uint8_t *) dst, (const uint8_t *) src, 1.702f, tw);
-    hvx_sigmoid_f32_aa((uint8_t *) dst, (uint8_t *) dst, tw);
-    hvx_mul_f32_aaa((uint8_t *) dst, (const uint8_t *) src, (uint8_t *) dst, tw);
+    hvx_gelu_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw);
+}
+
+static void tile_gelu_quick_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) {
+    (void) uctx;
+    hvx_gelu_quick_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw);
 }

 static void tile_gelu_erf_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) {
@@ -1533,6 +1613,7 @@ static int execute_op_unary(struct htp_ops_context * octx) {
         case HTP_OP_UNARY_SIGMOID:   op_type = "sigmoid-f32";                                break;
         case HTP_OP_UNARY_SILU:      op_type = "silu-f32";                                   break;
         case HTP_OP_UNARY_GELU:      op_type = "gelu-f32";                                   break;
+        case HTP_OP_UNARY_GELU_QUICK: op_type = "gelu-quick-f32";                             break;
         case HTP_OP_UNARY_GELU_ERF:  op_type = "gelu-erf-f32";                               break;
         case HTP_OP_UNARY_SOFTPLUS:  op_type = "softplus-f32";                               break;
         case HTP_OP_UNARY_TANH:      op_type = "tanh-f32";                                   break;
@@ -1684,6 +1765,7 @@ static int execute_op_unary(struct htp_ops_context * octx) {
             case HTP_OP_UNARY_SIGMOID:   compute_func = (void *) tile_sigmoid_f32;        break;
             case HTP_OP_UNARY_SILU:      compute_func = (void *) tile_silu_f32;           break;
             case HTP_OP_UNARY_GELU:      compute_func = (void *) tile_gelu_f32;           break;
+            case HTP_OP_UNARY_GELU_QUICK: compute_func = (void *) tile_gelu_quick_f32;     break;
             case HTP_OP_UNARY_GELU_ERF:  compute_func = (void *) tile_gelu_erf_f32;       break;
             case HTP_OP_UNARY_SOFTPLUS:  compute_func = (void *) tile_softplus_f32;       break;
             case HTP_OP_UNARY_TANH:      compute_func = (void *) tile_tanh_f32;           break;
@@ -1731,6 +1813,7 @@ static int execute_op_unary(struct htp_ops_context * octx) {
             case HTP_OP_UNARY_SIGMOID:   compute_func = (void *) sigmoid_f32;             break;
             case HTP_OP_UNARY_SILU:      compute_func = (void *) silu_f32;                break;
             case HTP_OP_UNARY_GELU:      compute_func = (void *) gelu_f32;                break;
+            case HTP_OP_UNARY_GELU_QUICK: compute_func = (void *) gelu_quick_f32;          break;
             case HTP_OP_UNARY_GELU_ERF:  compute_func = (void *) gelu_erf_f32;            break;
             case HTP_OP_UNARY_SOFTPLUS:  compute_func = (void *) softplus_f32;            break;
             case HTP_OP_UNARY_TANH:      compute_func = (void *) tanh_f32;                break;
diff --git a/ggml/src/ggml-hexagon/htp/unary-ops.h b/ggml/src/ggml-hexagon/htp/unary-ops.h
index 7b73cf10f..7e5779e5a 100644
--- a/ggml/src/ggml-hexagon/htp/unary-ops.h
+++ b/ggml/src/ggml-hexagon/htp/unary-ops.h
@@ -54,6 +54,7 @@ static inline bool htp_op_is_unary(uint32_t opcode) {
         case HTP_OP_UNARY_SIGMOID:
         case HTP_OP_UNARY_SILU:
         case HTP_OP_UNARY_GELU:
+        case HTP_OP_UNARY_GELU_QUICK:
         case HTP_OP_UNARY_GELU_ERF:
         case HTP_OP_UNARY_SOFTPLUS:
         case HTP_OP_UNARY_TANH: