Commit 2bbca8f20 for llama.cpp

commit 2bbca8f202e76b25dac9116755bf1713a84e2753
Author: Aaron Teo <aaron.teo1@ibm.com>
Date:   Sat Oct 10 14:08:52 2026 +0800

    ggml-cpu: vectorize fp32 to fp16 conversion (#30157)

    ggml-cpu(s390x): rename ulong to uint64_t sized types



    ggml-cpu(s390x): rm comment

    Signed-off-by: Aaron Teo <aaron.teo1@ibm.com>

diff --git a/ggml/src/ggml-cpu/ggml-cpu-impl.h b/ggml/src/ggml-cpu/ggml-cpu-impl.h
index 5dd9ec8e6..82345b206 100644
--- a/ggml/src/ggml-cpu/ggml-cpu-impl.h
+++ b/ggml/src/ggml-cpu/ggml-cpu-impl.h
@@ -390,17 +390,16 @@ typedef unsigned char uchar8x16_t __attribute__((vector_size(16)));
 typedef int8_t  int8x16_t __attribute__((vector_size(16)));
 typedef int16_t int16x8_t __attribute__((vector_size(16)));
 typedef int32_t int32x4_t __attribute__((vector_size(16)));
+typedef int64_t int64x2_t __attribute__((vector_size(16)));

 typedef uint8_t  uint8x16_t __attribute__((vector_size(16)));
 typedef uint16_t uint16x8_t __attribute__((vector_size(16)));
 typedef uint32_t uint32x4_t __attribute__((vector_size(16)));
+typedef uint64_t uint64x2_t __attribute__((vector_size(16)));

 typedef float  float32x4_t  __attribute__((vector_size(16)));
 typedef double double64x2_t __attribute__((vector_size(16)));

-typedef signed   long long long64x2_t  __attribute__((vector_size(16)));
-typedef unsigned long long ulong64x2_t __attribute__((vector_size(16)));
-
 typedef struct ggml_uint8x16x2_t {
     uint8x16_t val[2];
 } ggml_uint8x16x2_t;
diff --git a/ggml/src/ggml-cpu/ggml-cpu.c b/ggml/src/ggml-cpu/ggml-cpu.c
index 7c274c03b..5428d5164 100644
--- a/ggml/src/ggml-cpu/ggml-cpu.c
+++ b/ggml/src/ggml-cpu/ggml-cpu.c
@@ -3504,6 +3504,12 @@ void ggml_cpu_fp32_to_fp16(const float * x, ggml_fp16_t * y, int64_t n) {
         vfloat16m1_t vy = __riscv_vfncvt_f_f_w_f16m1(vx, vl);
         __riscv_vse16_v_f16m1((_Float16 *)&y[i], vy, vl);
     }
+#elif defined(__VXE__) || defined(__VXE2__)
+    for (; i + 7 < n; i += 8) {
+        const uint32x4_t v_yl = __lzs_f32cx4_to_f16(vec_xl(0, x + i + 0));
+        const uint32x4_t v_yh = __lzs_f32cx4_to_f16(vec_xl(0, x + i + 4));
+        vec_xst(vec_pack(v_yl, v_yh), 0, (uint16_t *)(y + i));
+    }
 #endif
     for (; i < n; ++i) {
         y[i] = GGML_CPU_FP32_TO_FP16(x[i]);
diff --git a/ggml/src/ggml-cpu/simd-mappings.h b/ggml/src/ggml-cpu/simd-mappings.h
index 89a5afa9c..f351e8127 100644
--- a/ggml/src/ggml-cpu/simd-mappings.h
+++ b/ggml/src/ggml-cpu/simd-mappings.h
@@ -1223,6 +1223,24 @@ static inline void __lsx_f16x4_store(ggml_fp16_t * x, __m128 y) {
 #define GGML_F16_STEP GGML_F32_STEP
 #define GGML_F16_EPR  GGML_F32_EPR

+static inline uint32x4_t __lzs_f32cx4_to_f16(float32x4_t v_f) {
+    float32x4_t v_base = vec_mul(vec_mul(vec_abs(v_f), vec_splats(0x1.0p+112f)), vec_splats(0x1.0p-110f));
+
+    const uint32x4_t v_w      = (uint32x4_t)v_f;
+    const uint32x4_t v_shl1_w = vec_add(v_w, v_w);
+    const uint32x4_t v_sign   = vec_and(v_w, vec_splats(UINT32_C(0x80000000)));
+    const uint32x4_t v_bias   = vec_max(vec_and(v_shl1_w, vec_splats(UINT32_C(0xFF000000))), vec_splats(UINT32_C(0x71000000)));
+
+    v_base = vec_add((float32x4_t)vec_add(vec_sr(v_bias, 1), vec_splats(UINT32_C(0x07800000))), v_base);
+
+    const uint32x4_t v_bits    = (uint32x4_t)v_base;
+    const uint32x4_t v_nonsign = vec_add(vec_and(vec_sr(v_bits, 13), vec_splats(UINT32_C(0x00007C00))),
+                                         vec_and(v_bits, vec_splats(UINT32_C(0x00000FFF))));
+    const uint32x4_t v_is_nan  = (uint32x4_t)vec_cmpgt(v_shl1_w, vec_splats(UINT32_C(0xFF000000)));
+
+    return vec_or(vec_sr(v_sign, 16), vec_sel(v_nonsign, vec_splats(UINT32_C(0x7E00)), v_is_nan));
+}
+
 static inline float32x4_t __lzs_f16cx4_load(const ggml_fp16_t * x) {
     float tmp[4];

@@ -1236,15 +1254,9 @@ static inline float32x4_t __lzs_f16cx4_load(const ggml_fp16_t * x) {
 }

 static inline void __lzs_f16cx4_store(ggml_fp16_t * x, float32x4_t v_y) {
-    float arr[4];
-
-    // note: keep type-cast here to prevent compiler bugs
-    // see: https://github.com/ggml-org/llama.cpp/issues/12846
-    vec_xst(v_y, 0, (float *)(arr));
-
-    for (int i = 0; i < 4; i++) {
-        x[i] = GGML_CPU_FP32_TO_FP16(arr[i]);
-    }
+    const uint32x4_t v_h = __lzs_f32cx4_to_f16(v_y);
+    const uint64_t   tmp = ((uint64x2_t)vec_pack(v_h, v_h))[0];
+    memcpy(x, &tmp, sizeof(tmp));
 }

 #define GGML_F16_VEC                GGML_F32x4