Commit e904318a2 for llama.cpp

commit e904318a2d3e2382e78c98b4353c5dd569c079f6
Author: Aparna M P <aparmp@qti.qualcomm.com>
Date:   Tue Sep 29 22:05:47 2026 +0530

    hexagon: add FP32 GELU_ERF and GEGLU_ERF support (#29631)

    * hexagon: add FP32 GELU_ERF and GEGLU_ERF support

    * hex-erf: reduce register pressure in kernels

diff --git a/ggml/src/ggml-hexagon/ggml-hexagon.cpp b/ggml/src/ggml-hexagon/ggml-hexagon.cpp
index 521a97a73..161eee248 100644
--- a/ggml/src/ggml-hexagon/ggml-hexagon.cpp
+++ b/ggml/src/ggml-hexagon/ggml-hexagon.cpp
@@ -6621,6 +6621,7 @@ static htp_op_code op_remap_to_htp(const ggml_tensor * t) {
                 case GGML_UNARY_OP_SILU:       return HTP_OP_UNARY_SILU;
                 case GGML_UNARY_OP_GELU:       return HTP_OP_UNARY_GELU;
                 case GGML_UNARY_OP_GELU_QUICK: return HTP_OP_UNARY_GELU;
+                case GGML_UNARY_OP_GELU_ERF:   return HTP_OP_UNARY_GELU_ERF;
                 case GGML_UNARY_OP_SIGMOID:    return HTP_OP_UNARY_SIGMOID;
                 case GGML_UNARY_OP_NEG:        return HTP_OP_UNARY_NEG;
                 case GGML_UNARY_OP_EXP:        return HTP_OP_UNARY_EXP;
@@ -6641,6 +6642,7 @@ static htp_op_code op_remap_to_htp(const ggml_tensor * t) {
                 case GGML_GLU_OP_SWIGLU_CLAMP: return HTP_OP_GLU_SWIGLU_CLAMP;
                 case GGML_GLU_OP_GEGLU:      return HTP_OP_GLU_GEGLU;
                 case GGML_GLU_OP_GEGLU_QUICK: return HTP_OP_GLU_GEGLU_QUICK;
+                case GGML_GLU_OP_GEGLU_ERF:  return HTP_OP_GLU_GEGLU_ERF;
                 default: break;
             }
             break;
@@ -7697,6 +7699,7 @@ static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, cons
                 case GGML_UNARY_OP_SILU:
                 case GGML_UNARY_OP_GELU:
                 case GGML_UNARY_OP_GELU_QUICK:
+                case GGML_UNARY_OP_GELU_ERF:
                 case GGML_UNARY_OP_RELU:
                 case GGML_UNARY_OP_STEP:
                     supp = ggml_hexagon_supported_unary(sess, op);
@@ -7714,6 +7717,7 @@ static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, cons
                 case GGML_GLU_OP_SWIGLU_CLAMP:
                 case GGML_GLU_OP_GEGLU:
                 case GGML_GLU_OP_GEGLU_QUICK:
+                case GGML_GLU_OP_GEGLU_ERF:
                     supp = ggml_hexagon_supported_activations(sess, op);
                     break;
                 default:
diff --git a/ggml/src/ggml-hexagon/htp/act-ops.c b/ggml/src/ggml-hexagon/htp/act-ops.c
index d59ac0770..28c581df4 100644
--- a/ggml/src/ggml-hexagon/htp/act-ops.c
+++ b/ggml/src/ggml-hexagon/htp/act-ops.c
@@ -362,6 +362,36 @@ static inline void hvx_geglu_quick_f32_aa(uint8_t * restrict dst, const uint8_t
     }
 }

+static inline void hvx_geglu_erf_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) {
+    assert((unsigned long) dst  % 128 == 0);
+    assert((unsigned long) src0 % 128 == 0);
+    assert((unsigned long) src1 % 128 == 0);
+
+    HVX_Vector * restrict vdst        = (HVX_Vector *) dst;
+    const HVX_Vector * restrict vsrc0 = (const HVX_Vector *) src0;
+    const HVX_Vector * restrict vsrc1 = (const HVX_Vector *) src1;
+
+    const uint32_t epv  = 128 / sizeof(float);
+    const uint32_t nvec = n / epv;
+    const uint32_t nloe = n % epv;
+
+    uint32_t i = 0;
+
+    _Pragma("unroll(4)")
+    for (; i < nvec; i++) {
+        HVX_Vector x = vsrc0[i];
+        HVX_Vector g = vsrc1[i];
+        vdst[i] = hvx_vec_mul_f32_f32(hvx_vec_gelu_erf_f32(x), g);
+    }
+
+    if (nloe) {
+        HVX_Vector x = vsrc0[i];
+        HVX_Vector g = vsrc1[i];
+        HVX_Vector result = hvx_vec_mul_f32_f32(hvx_vec_gelu_erf_f32(x), g);
+        hvx_vec_store_a((void *) &vdst[i], nloe * sizeof(float), result);
+    }
+}
+
 // geglu(x, g) = gelu(x) * g
 static void geglu_f32(const float * restrict src0,
                                                 const float * restrict src1,
@@ -396,6 +426,23 @@ static void geglu_quick_f32(const float * restrict src0,
     }
 }

+// geglu_erf(x, g) = gelu_erf(x) * g
+static void geglu_erf_f32(const float * restrict src0,
+                          const float * restrict src1,
+                          float * restrict dst,
+                          const uint32_t num_rows,
+                          const struct htp_act_context * actx) {
+    htp_glu_op_preamble;
+
+    for (uint32_t ib = 0; ib < num_rows; ib++) {
+        const uint8_t * restrict src0_ptr = (const uint8_t *) src0 + (ib * src0_row_size_aligned);
+        const uint8_t * restrict src1_ptr = (const uint8_t *) src1 + (ib * src1_row_size_aligned);
+        uint8_t * restrict dst_ptr        = (uint8_t *) dst + (ib * dst_row_size_aligned);
+
+        hvx_geglu_erf_f32_aa(dst_ptr, src0_ptr, src1_ptr, nc);
+    }
+}
+
 static void glu_f32_per_thread(unsigned int nth, unsigned int ith, void * data) {
     struct htp_act_context * actx = (struct htp_act_context *) data;
     htp_act_preamble;
@@ -529,6 +576,11 @@ static int execute_op_activations_f32(struct htp_ops_context * octx) {
             compute_fn = geglu_quick_f32;
             op_type    = "geglu-quick-f32";
             break;
+
+        case HTP_OP_GLU_GEGLU_ERF:
+            compute_fn = geglu_erf_f32;
+            op_type    = "geglu-erf-f32";
+            break;
         default:
             FARF(ERROR, "Unsupported activations Op %u\n", octx->op);
             return HTP_STATUS_NO_SUPPORT;
@@ -634,7 +686,8 @@ static int execute_op_activations_f32(struct htp_ops_context * octx) {
                   octx->op == HTP_OP_GLU_SWIGLU_OAI ||
                   octx->op == HTP_OP_GLU_SWIGLU_CLAMP ||
                   octx->op == HTP_OP_GLU_GEGLU ||
-                  octx->op == HTP_OP_GLU_GEGLU_QUICK)) {
+                  octx->op == HTP_OP_GLU_GEGLU_QUICK ||
+                  octx->op == HTP_OP_GLU_GEGLU_ERF)) {
          const int32_t swapped = octx->op_params[1];
          data_src1 = data_src0;
          actx.src1_row_size = actx.src0_row_size;
diff --git a/ggml/src/ggml-hexagon/htp/htp-ops.h b/ggml/src/ggml-hexagon/htp/htp-ops.h
index 674d52c78..3dbde5adf 100644
--- a/ggml/src/ggml-hexagon/htp/htp-ops.h
+++ b/ggml/src/ggml-hexagon/htp/htp-ops.h
@@ -111,6 +111,8 @@ enum htp_op_code {
     HTP_OP_MDEV_GROUP,
     HTP_OP_ROLL,
     HTP_OP_ARGMAX,
+    HTP_OP_UNARY_GELU_ERF,
+    HTP_OP_GLU_GEGLU_ERF,

     HTP_OP_INVALID
 };
diff --git a/ggml/src/ggml-hexagon/htp/hvx-erf.h b/ggml/src/ggml-hexagon/htp/hvx-erf.h
new file mode 100644
index 000000000..6eefc300f
--- /dev/null
+++ b/ggml/src/ggml-hexagon/htp/hvx-erf.h
@@ -0,0 +1,73 @@
+#ifndef HVX_ERF_H
+#define HVX_ERF_H
+
+#include "hvx-base.h"
+#include "hvx-exp.h"
+#include "hvx-inverse.h"
+
+// Maximum error is about 1.5e-7 for the Abramowitz-Stegun approximation.
+static __attribute__((noinline)) HVX_Vector hvx_vec_erf_f32(HVX_Vector x) {
+    const HVX_Vector zero = hvx_vec_splat_f32(0.0f);
+    const HVX_Vector ax = hvx_vec_abs_f32(x);
+    HVX_Vector t = hvx_vec_inverse_f32(hvx_vec_add_f32_f32(
+        hvx_vec_splat_f32(1.0f), hvx_vec_mul_f32_f32(hvx_vec_splat_f32(0.3275911f), ax)));
+
+    HVX_Vector poly = hvx_vec_mul_f32_f32(hvx_vec_splat_f32(1.061405429f), t);
+    poly = hvx_vec_add_f32_f32(hvx_vec_splat_f32(-1.453152027f), hvx_vec_mul_f32_f32(poly, t));
+    poly = hvx_vec_add_f32_f32(hvx_vec_splat_f32(1.421413741f), hvx_vec_mul_f32_f32(poly, t));
+    poly = hvx_vec_add_f32_f32(hvx_vec_splat_f32(-0.284496736f), hvx_vec_mul_f32_f32(poly, t));
+    poly = hvx_vec_add_f32_f32(hvx_vec_splat_f32(0.254829592f), hvx_vec_mul_f32_f32(poly, t));
+
+    const HVX_Vector exp_term = hvx_vec_exp_f32(hvx_vec_neg_f32(hvx_vec_mul_f32_f32(ax, ax)));
+    HVX_Vector result = hvx_vec_sub_f32_f32(hvx_vec_splat_f32(1.0f),
+        hvx_vec_mul_f32_f32(hvx_vec_mul_f32_f32(poly, t), exp_term));
+
+    const HVX_VectorPred neg = Q6_Q_vcmp_gt_VsfVsf(zero, x);
+    result = Q6_V_vmux_QVV(neg, hvx_vec_neg_f32(result), result);
+    return result;
+}
+
+static inline HVX_Vector hvx_vec_gelu_erf_f32(HVX_Vector x) {
+    const HVX_Vector scale = hvx_vec_splat_f32(0.7071067811865475f);
+    const HVX_Vector half  = hvx_vec_splat_f32(0.5f);
+    const HVX_Vector one   = hvx_vec_splat_f32(1.0f);
+    const HVX_Vector max_x = hvx_vec_splat_f32(10.0f);
+    const HVX_Vector min_x = hvx_vec_splat_f32(-10.0f);
+    const HVX_VectorPred neg_large = Q6_Q_vcmp_gt_VsfVsf(min_x, x);
+    const HVX_VectorPred pos_large = Q6_Q_vcmp_gt_VsfVsf(x, max_x);
+    HVX_Vector x_calc = Q6_V_vmux_QVV(neg_large, min_x, x);
+    x_calc = Q6_V_vmux_QVV(pos_large, max_x, x_calc);
+    const HVX_Vector erf = hvx_vec_erf_f32(hvx_vec_mul_f32_f32(x_calc, scale));
+
+    HVX_Vector result = hvx_vec_mul_f32_f32(hvx_vec_mul_f32_f32(half, x_calc), hvx_vec_add_f32_f32(one, erf));
+
+    result = Q6_V_vmux_QVV(neg_large, hvx_vec_splat_f32(0.0f), result);
+    result = Q6_V_vmux_QVV(pos_large, x, result);
+    return result;
+}
+
+static inline void hvx_gelu_erf_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) {
+    assert((unsigned long) dst % 128 == 0);
+    assert((unsigned long) src % 128 == 0);
+
+    HVX_Vector * restrict vdst = (HVX_Vector *) dst;
+    HVX_Vector * restrict vsrc = (HVX_Vector *) src;
+
+    const uint32_t elem_size = sizeof(float);
+    const uint32_t epv       = 128 / elem_size;
+    const uint32_t nvec      = n / epv;
+    const uint32_t nloe      = n % epv;
+
+    uint32_t i = 0;
+
+    _Pragma("unroll(4)")
+    for (; i < nvec; i++) {
+        vdst[i] = hvx_vec_gelu_erf_f32(vsrc[i]);
+    }
+    if (nloe) {
+        HVX_Vector v = hvx_vec_gelu_erf_f32(vsrc[i]);
+        hvx_vec_store_a((void *) &vdst[i], nloe * elem_size, v);
+    }
+}
+
+#endif /* HVX_ERF_H */
diff --git a/ggml/src/ggml-hexagon/htp/hvx-utils.h b/ggml/src/ggml-hexagon/htp/hvx-utils.h
index 706a64f3a..dbdd7e09e 100644
--- a/ggml/src/ggml-hexagon/htp/hvx-utils.h
+++ b/ggml/src/ggml-hexagon/htp/hvx-utils.h
@@ -8,6 +8,7 @@
 #include "hvx-repl.h"
 #include "hvx-scale.h"
 #include "hvx-exp.h"
+#include "hvx-erf.h"
 #include "hvx-inverse.h"
 #include "hvx-reduce.h"
 #include "hvx-sigmoid.h"
diff --git a/ggml/src/ggml-hexagon/htp/main.c b/ggml/src/ggml-hexagon/htp/main.c
index 6d1c47dd0..fcb0a7202 100644
--- a/ggml/src/ggml-hexagon/htp/main.c
+++ b/ggml/src/ggml-hexagon/htp/main.c
@@ -851,6 +851,7 @@ static int execute_op(struct htp_ops_context * octx) {
         case HTP_OP_UNARY_SIGMOID:
         case HTP_OP_UNARY_SILU:
         case HTP_OP_UNARY_GELU:
+        case HTP_OP_UNARY_GELU_ERF:
         case HTP_OP_UNARY_NEG:
         case HTP_OP_UNARY_EXP:
         case HTP_OP_UNARY_TANH:
@@ -866,6 +867,7 @@ static int execute_op(struct htp_ops_context * octx) {
         case HTP_OP_GLU_SWIGLU_CLAMP:
         case HTP_OP_GLU_GEGLU:
         case HTP_OP_GLU_GEGLU_QUICK:
+        case HTP_OP_GLU_GEGLU_ERF:
             return op_activations(octx);

         case HTP_OP_SOFTMAX:
diff --git a/ggml/src/ggml-hexagon/htp/unary-ops.c b/ggml/src/ggml-hexagon/htp/unary-ops.c
index 8b64a9eaa..b63fd4c29 100644
--- a/ggml/src/ggml-hexagon/htp/unary-ops.c
+++ b/ggml/src/ggml-hexagon/htp/unary-ops.c
@@ -514,6 +514,20 @@ static void gelu_f32(const void * restrict src,
     }
 }

+static void gelu_erf_f32(const void * restrict src,
+                         void * restrict dst,
+                         const uint32_t num_rows,
+                         const struct htp_unary_context * uctx) {
+    htp_unary_op_preamble;
+
+    for (uint32_t ir = 0; ir < num_rows; ir++) {
+        const uint8_t * restrict src_local = (const uint8_t *) src + (ir * src0_row_size_aligned);
+        uint8_t * restrict dst_local       = (uint8_t *) dst + (ir * dst_row_size_aligned);
+
+        hvx_gelu_erf_f32_aa(dst_local, src_local, ne0);
+    }
+}
+
 static void tri_f32(const void * restrict src,
                     void * restrict dst,
                     const uint32_t num_rows,
@@ -765,6 +779,11 @@ static void tile_gelu_f32(void * restrict dst, const void * restrict src, uint32
     hvx_mul_f32_aaa((uint8_t *) dst, (const uint8_t *) src, (uint8_t *) dst, tw);
 }

+static void tile_gelu_erf_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) {
+    (void) uctx;
+    hvx_gelu_erf_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw);
+}
+
 static void tile_softplus_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) {
     (void) uctx;
     const float * restrict sf = (const float *) src;
@@ -1514,6 +1533,7 @@ static int execute_op_unary(struct htp_ops_context * octx) {
         case HTP_OP_UNARY_SIGMOID:   op_type = "sigmoid-f32";                                break;
         case HTP_OP_UNARY_SILU:      op_type = "silu-f32";                                   break;
         case HTP_OP_UNARY_GELU:      op_type = "gelu-f32";                                   break;
+        case HTP_OP_UNARY_GELU_ERF:  op_type = "gelu-erf-f32";                               break;
         case HTP_OP_UNARY_SOFTPLUS:  op_type = "softplus-f32";                               break;
         case HTP_OP_UNARY_TANH:      op_type = "tanh-f32";                                   break;
         case HTP_OP_UNARY_ABS:       op_type = is_f16 ? "abs-f16"      : "abs-f32";          break;
@@ -1664,6 +1684,7 @@ static int execute_op_unary(struct htp_ops_context * octx) {
             case HTP_OP_UNARY_SIGMOID:   compute_func = (void *) tile_sigmoid_f32;        break;
             case HTP_OP_UNARY_SILU:      compute_func = (void *) tile_silu_f32;           break;
             case HTP_OP_UNARY_GELU:      compute_func = (void *) tile_gelu_f32;           break;
+            case HTP_OP_UNARY_GELU_ERF:  compute_func = (void *) tile_gelu_erf_f32;       break;
             case HTP_OP_UNARY_SOFTPLUS:  compute_func = (void *) tile_softplus_f32;       break;
             case HTP_OP_UNARY_TANH:      compute_func = (void *) tile_tanh_f32;           break;
             case HTP_OP_UNARY_ABS:       compute_func = (void *) tile_abs_f32;            break;
@@ -1710,6 +1731,7 @@ static int execute_op_unary(struct htp_ops_context * octx) {
             case HTP_OP_UNARY_SIGMOID:   compute_func = (void *) sigmoid_f32;             break;
             case HTP_OP_UNARY_SILU:      compute_func = (void *) silu_f32;                break;
             case HTP_OP_UNARY_GELU:      compute_func = (void *) gelu_f32;                break;
+            case HTP_OP_UNARY_GELU_ERF:  compute_func = (void *) gelu_erf_f32;            break;
             case HTP_OP_UNARY_SOFTPLUS:  compute_func = (void *) softplus_f32;            break;
             case HTP_OP_UNARY_TANH:      compute_func = (void *) tanh_f32;                break;
             case HTP_OP_UNARY_ABS:       compute_func = (void *) abs_f32;                 break;
diff --git a/ggml/src/ggml-hexagon/htp/unary-ops.h b/ggml/src/ggml-hexagon/htp/unary-ops.h
index 67d813cbc..7b73cf10f 100644
--- a/ggml/src/ggml-hexagon/htp/unary-ops.h
+++ b/ggml/src/ggml-hexagon/htp/unary-ops.h
@@ -54,6 +54,7 @@ static inline bool htp_op_is_unary(uint32_t opcode) {
         case HTP_OP_UNARY_SIGMOID:
         case HTP_OP_UNARY_SILU:
         case HTP_OP_UNARY_GELU:
+        case HTP_OP_UNARY_GELU_ERF:
         case HTP_OP_UNARY_SOFTPLUS:
         case HTP_OP_UNARY_TANH:
         case HTP_OP_UNARY_ABS: