Commit 1c87d7d6e3 for openssl.org

commit 1c87d7d6e3450bad2416236675975a9677689ab7
Author: Timo Keller <tkeller@linux.ibm.com>
Date:   Wed Aug 19 09:36:48 2026 +0200

    Add s390x cross-compile workflow

    Add s390x cross-compile workflow and change Configure for the s390x-
    optimized implementations of ML-KEM and ML-DSA.

    Signed-off-by: Timo Keller <tkeller@linux.ibm.com>
    Assisted-by: IBM Bob:2.0.3
    Reviewed-by: Mounir Idrassi <mounir.idrassi@idrix.fr>
    Reviewed-by: Tomas Mraz <tomas@openssl.foundation>
    Merge-date: Tue Sep 29 16:47:14 2026
    Merged-from: https://github.com/openssl/openssl/pull/31929

diff --git a/.github/workflows/cross-compiles.yml b/.github/workflows/cross-compiles.yml
index 9f32de5232..25a8486e13 100644
--- a/.github/workflows/cross-compiles.yml
+++ b/.github/workflows/cross-compiles.yml
@@ -63,6 +63,14 @@ jobs:
         #                this string will be used as content for the OpenSSL
         #                capabilities variable.
         #   ppa:   Launchpad PPA repository to download packages from.
+        #   compiler: optional; set to "clang" to build with Clang instead of
+        #             GCC.  The cross-compile prefix is still used for the
+        #             binutils (assembler, linker, nm …).
+        #   nm_check: optional; when set, a post-build step verifies that the
+        #             six ML-KEM VX vector-entry-point symbols are present as
+        #             text symbols in the ml_kem_vec128.o object file.
+        #   label: optional; used as the artifact name suffix in place of
+        #          arch, to disambiguate matrix jobs that share the same arch.
         platform: [
           {
             arch: i386-pc-msdosdjgpp,
@@ -138,10 +146,143 @@ jobs:
             target: linux64-riscv64,
             fips: no
           }, {
+            # Baseline s390x job.  No qemucpu override — QEMU's default TCG
+            # model avoids the "CPU model features not available" admission
+            # failure that named models (z13, z13-base, …) trigger on x86
+            # GitHub Actions runners.  Named z/Architecture models include
+            # mandatory CPACF sub-features (kimd, km, klmd, …) that require
+            # real s390x hardware or KVM; the TCG emulator cannot satisfy them
+            # and QEMU 8.x exits fatally before OpenSSL even starts.
+            # z900 zeros every CPACF vector and all stfle facility bits (no VX),
+            # so OpenSSL uses the pure-software scalar fallback paths.
             arch: s390x-linux-gnu,
             libs: libc6-dev-s390x-cross,
             target: linux64-s390x,
-            fips: no
+            fips: no,
+            opensslcapsname: s390xcap,
+            opensslcaps: z900
+          }, {
+            # Build-only: verify no-asm does not produce undefined references to
+            # OPENSSL_s390xcap_P (which is only compiled when asm is enabled).
+            arch: s390x-linux-gnu,
+            libs: libc6-dev-s390x-cross,
+            target: no-shared no-asm linux64-s390x,
+            fips: no,
+            tests: none
+          }, {
+            # GCC / VX enabled (FIPS).
+            # The base linux64-s390x target carries no -march flag, so the
+            # whole library is safe to execute on any s390x CPU.
+            # ml_kem_vec128.c and ml_dsa_ntt_vec128.c use "#pragma GCC target"
+            # to scope VX codegen to those TUs only.
+            #
+            # Why qemucpu: max instead of z13:
+            #   QEMU 8.x named z/Architecture CPU models (z13, z13-base, …)
+            #   include mandatory CPACF sub-features (kimd-sha-1, km-aes-*, …)
+            #   that the TCG softmmu cannot provide on a non-s390x host.  QEMU
+            #   prints a fatal "Some features … not available" warning and exits
+            #   with a non-zero status *before* loading the OpenSSL binary.  The
+            #   Perl test harness sees the non-zero exit code and marks every
+            #   test as failed — even though the OpenSSL code is correct.
+            #   "max" enables all features the TCG emulator actually supports
+            #   (which includes VX vector-register instructions) without
+            #   requiring hardware CPACF, so QEMU starts cleanly.
+            #   The OPENSSL_s390xcap override below controls what capability
+            #   bits OpenSSL reads at runtime, so VX detection is correct
+            #   regardless of which CPU model QEMU uses.
+            #
+            # The nm_check steps assert VX symbols are present and that the
+            # dispatcher (ml_dsa_ntt.o) has unresolved references to them.
+            arch: s390x-linux-gnu,
+            libs: libc6-dev-s390x-cross,
+            target: linux64-s390x,
+            qemucpu: max,
+            opensslcapsname: s390xcap,
+            # Enable the VX facility bit (stfle[2] bit 129 = 0x4000000000000000)
+            # while zeroing every CPACF instruction vector (kimd/klmd/km/…).
+            # QEMU emulates the VX vector-register instructions but cannot
+            # execute CPACF instructions; this lets the VX code paths run
+            # under QEMU without triggering illegal-instruction traps.
+            opensslcaps: "stfle:0:0:4000000000000000;kimd:0:0;klmd:0:0;km:0:0;kmc:0:0;kmac:0:0;kmctr:0:0;kmo:0:0;kmf:0:0;prno:0:0;kma:0:0;pcc:0:0;kdsa:0:0",
+            tests: test_internal_ml_kem test_evp_extra_ml_kem test_ml_kem_codecs test_internal_ml_dsa test_ml_dsa test_ml_dsa_codecs test_evp*,
+            nm_check: "yes",
+            label: s390x-gcc-z13-vx-on-fips
+          }, {
+            # GCC / VX masked off (no FIPS).
+            # Same base build (linux64-s390x, no -march) as the VX-on job
+            # above, but OPENSSL_s390xcap=z900 zeros all stfle facility bits
+            # (including VX) so S390X_VX_CAPABLE is false and the generic
+            # scalar path is exercised on the same binary.
+            # qemucpu: max — see the VX-on job comment for why z13 is avoided.
+            arch: s390x-linux-gnu,
+            libs: libc6-dev-s390x-cross,
+            target: linux64-s390x,
+            fips: no,
+            qemucpu: max,
+            opensslcapsname: s390xcap,
+            # z900 zeros all CPACF vectors and all stfle facility bits including
+            # VX, exercising the scalar fallback path without QEMU CPACF traps.
+            opensslcaps: z900,
+            tests: test_internal_ml_kem test_evp_extra_ml_kem test_ml_kem_codecs test_internal_ml_dsa test_ml_dsa test_ml_dsa_codecs test_evp*,
+            label: s390x-gcc-z13-vx-off
+          }, {
+            # Clang / z13 / VX enabled (FIPS).
+            # Clang does not honour #pragma GCC target for preprocessor-level
+            # macros, so -march=z13 must be passed at compile time to make
+            # Clang define __VX__ and enable the VX code paths.  -march=z13
+            # is safe to run on any z13+ CPU and does not imply z14 instructions.
+            # Configure also adds -fzvector when Clang needs it to access
+            # <vecintrin.h>; that flag only unlocks the header, not codegen.
+            # Verifies that Clang emits all six ML-KEM and three ML-DSA VX
+            # entry points and that the ML-DSA dispatcher references them.
+            # qemucpu: max — see the GCC VX-on job comment for why z13 is avoided.
+            arch: s390x-linux-gnu,
+            libs: libc6-dev-s390x-cross,
+            target: -march=z13 linux64-s390x,
+            compiler: clang,
+            qemucpu: max,
+            opensslcapsname: s390xcap,
+            opensslcaps: "stfle:0:0:4000000000000000;kimd:0:0;klmd:0:0;km:0:0;kmc:0:0;kmac:0:0;kmctr:0:0;kmo:0:0;kmf:0:0;prno:0:0;kma:0:0;pcc:0:0;kdsa:0:0",
+            tests: test_internal_ml_kem test_evp_extra_ml_kem test_ml_kem_codecs test_internal_ml_dsa test_ml_dsa test_ml_dsa_codecs test_evp*,
+            nm_check: "yes",
+            label: s390x-clang-z13-vx-on-fips
+          }, {
+            # Clang / z13 / VX masked off (no FIPS).
+            # Same -march=z13 build as the Clang VX-on job but
+            # OPENSSL_s390xcap=z900 zeros all stfle bits (no VX, no CPACF) so
+            # the generic scalar path is exercised on the same binary.
+            # qemucpu: max — see the GCC VX-on job comment for why z13 is avoided.
+            arch: s390x-linux-gnu,
+            libs: libc6-dev-s390x-cross,
+            target: -march=z13 linux64-s390x,
+            fips: no,
+            compiler: clang,
+            qemucpu: max,
+            opensslcapsname: s390xcap,
+            opensslcaps: z900,
+            tests: test_internal_ml_kem test_evp_extra_ml_kem test_ml_kem_codecs test_internal_ml_dsa test_ml_dsa test_ml_dsa_codecs test_evp*,
+            label: s390x-clang-z13-vx-off
+          }, {
+            # Clang / z10 baseline / VX enabled (FIPS).
+            # Tests the intended per-TU isolation design: the whole library is
+            # compiled with -march=z10 (no VX), but Configure detects the s390x
+            # target via $target{asm_arch} and runs the feature probe with the
+            # effective compiler flags plus -march=z13 -mvx.  Only the two vec128
+            # TUs (ml_kem_vec128.c, ml_dsa_ntt_vec128.c) receive -march=z13 -mvx
+            # (and -fzvector if Clang needs it); the rest of the library is compiled
+            # to the z10 baseline.  This verifies that the scalar fallback objects
+            # are correct z10 code while the two vector TUs are valid z13+VX code.
+            # qemucpu: max — see the GCC VX-on job comment for why z13 is avoided.
+            arch: s390x-linux-gnu,
+            libs: libc6-dev-s390x-cross,
+            target: -march=z10 linux64-s390x,
+            compiler: clang,
+            qemucpu: max,
+            opensslcapsname: s390xcap,
+            opensslcaps: "stfle:0:0:4000000000000000;kimd:0:0;klmd:0:0;km:0:0;kmc:0:0;kmac:0:0;kmctr:0:0;kmo:0:0;kmf:0:0;prno:0:0;kma:0:0;pcc:0:0;kdsa:0:0",
+            tests: test_internal_ml_kem test_evp_extra_ml_kem test_ml_kem_codecs test_internal_ml_dsa test_ml_dsa test_ml_dsa_codecs test_evp*,
+            nm_check: "yes",
+            label: s390x-clang-z10-vx-on-fips
           }, {
             arch: sh4-linux-gnu,
             libs: libc6-dev-sh4-cross,
@@ -189,36 +330,127 @@ jobs:
       if: matrix.platform.ppa != ''
       run: |
         sudo add-apt-repository ppa:${{ matrix.platform.ppa }}
-    - name: install packages
+    - name: install packages (GCC)
+      if: matrix.platform.compiler != 'clang'
       run: |
         sudo apt-get update
         sudo apt-get -yq --allow-unauthenticated --allow-downgrades --allow-remove-essential --allow-change-held-packages install \
             gcc-${{ matrix.platform.arch }} \
             ${{ matrix.platform.libs }}
+    - name: install packages (Clang)
+      if: matrix.platform.compiler == 'clang'
+      run: |
+        sudo apt-get update
+        sudo apt-get -yq --allow-unauthenticated --allow-downgrades --allow-remove-essential --allow-change-held-packages install \
+            clang \
+            binutils-${{ matrix.platform.arch }} \
+            gcc-${{ matrix.platform.arch }} \
+            ${{ matrix.platform.libs }}
+        # Detect sysroot: Ubuntu <=22.04 uses .../sys-root, 24.04+ installs directly.
+        SYSROOT=/usr/${{ matrix.platform.arch }}/sys-root
+        [ -d "$SYSROOT" ] || SYSROOT=/usr/${{ matrix.platform.arch }}
+        echo "CROSS_SYSROOT=$SYSROOT" >> $GITHUB_ENV
     - uses: actions/checkout@v6
       with:
         persist-credentials: false
     - name: checkout fuzz/corpora submodule
       run: git submodule update --init --depth 1 fuzz/corpora

-    - name: config with FIPS
-      if: matrix.platform.fips != 'no'
+    - name: config with FIPS (GCC)
+      if: matrix.platform.fips != 'no' && matrix.platform.compiler != 'clang'
       run: |
         ./config --banner=Configured --strict-warnings enable-fips enable-lms \
                  --cross-compile-prefix=${{ matrix.platform.arch }}- \
                  ${{ matrix.platform.target }}
-    - name: config without FIPS
-      if: matrix.platform.fips == 'no'
+    - name: config without FIPS (GCC)
+      if: matrix.platform.fips == 'no' && matrix.platform.compiler != 'clang'
       run: |
         ./config --banner=Configured --strict-warnings enable-lms \
                  --cross-compile-prefix=${{ matrix.platform.arch }}- \
                  ${{ matrix.platform.target }}
+    - name: config with FIPS (Clang)
+      if: matrix.platform.fips != 'no' && matrix.platform.compiler == 'clang'
+      env:
+        CC: clang --target=${{ matrix.platform.arch }} --sysroot=${{ env.CROSS_SYSROOT }}
+        AR: ${{ matrix.platform.arch }}-ar
+        RANLIB: ${{ matrix.platform.arch }}-ranlib
+        AS: ${{ matrix.platform.arch }}-as
+        # Reset the linker sysroot to / so that absolute-path linker scripts
+        # shipped by libc6-dev-*-cross (e.g. libc.so referencing
+        # /usr/<arch>/lib/libc.so.6) are resolved against the real root
+        # rather than being doubled-prefixed under CROSS_SYSROOT.
+        # LDFLAGS is picked up by OpenSSL's Configure and forwarded to the
+        # linker only, leaving compiler-detection probes unaffected.
+        LDFLAGS: -Wl,--sysroot=/
+      run: |
+        ./config --banner=Configured --strict-warnings enable-fips enable-lms \
+                 ${{ matrix.platform.target }}
+    - name: config without FIPS (Clang)
+      if: matrix.platform.fips == 'no' && matrix.platform.compiler == 'clang'
+      env:
+        CC: clang --target=${{ matrix.platform.arch }} --sysroot=${{ env.CROSS_SYSROOT }}
+        AR: ${{ matrix.platform.arch }}-ar
+        RANLIB: ${{ matrix.platform.arch }}-ranlib
+        AS: ${{ matrix.platform.arch }}-as
+        LDFLAGS: -Wl,--sysroot=/
+      run: |
+        ./config --banner=Configured --strict-warnings enable-lms \
+                 ${{ matrix.platform.target }}
     - name: config dump
       run: ./configdata.pm --dump

     - name: make
       run: make -s -j4

+    - name: check ML-KEM VX vector symbols
+      if: matrix.platform.nm_check != ''
+      run: |
+        ${{ matrix.platform.arch }}-nm crypto/ml_kem/libcrypto-lib-ml_kem_vec128.o \
+          | awk '$2 == "T"' \
+          | grep -cE ' (ossl_ml_kem_scalar_ntt_vec128|ossl_ml_kem_scalar_inverse_ntt_vec128|ossl_ml_kem_scalar_inverse_ntt_demontgomerize_vec128|ossl_ml_kem_scalar_mult_add_vec128|ossl_ml_kem_inner_product_montgomery_vec128|ossl_ml_kem_matrix_mult_intt_vec128)$' \
+          | grep -q '^6$'
+
+    - name: check ML-KEM dispatcher references VX symbols
+      # Verify that ml_kem.o (the dispatcher) has unresolved external
+      # references (U-type) to all six vec128 functions and also references
+      # OPENSSL_s390xcap_P.  This proves that the dispatch block in
+      # ml_kem_ntt_init() was compiled in — the standalone vec128 object
+      # check above only verifies that the symbols exist in the vec128 file,
+      # not that the dispatcher actually calls them at runtime.
+      # Note: 'nm' omits the address column for undefined symbols, so
+      # undefined entries look like "                 U symbol_name" and
+      # $1 == "U" is the correct awk field (not $2).
+      if: matrix.platform.nm_check != ''
+      run: |
+        ${{ matrix.platform.arch }}-nm crypto/ml_kem/libcrypto-lib-ml_kem.o \
+          | awk '$1 == "U"' \
+          | grep -cE ' (ossl_ml_kem_scalar_ntt_vec128|ossl_ml_kem_scalar_inverse_ntt_vec128|ossl_ml_kem_scalar_inverse_ntt_demontgomerize_vec128|ossl_ml_kem_scalar_mult_add_vec128|ossl_ml_kem_inner_product_montgomery_vec128|ossl_ml_kem_matrix_mult_intt_vec128)$' \
+          | grep -q '^6$'
+        ${{ matrix.platform.arch }}-nm crypto/ml_kem/libcrypto-lib-ml_kem.o \
+          | awk '$1 == "U"' \
+          | grep -q ' OPENSSL_s390xcap_P$'
+
+    - name: check ML-DSA VX vector symbols
+      if: matrix.platform.nm_check != ''
+      run: |
+        ${{ matrix.platform.arch }}-nm crypto/ml_dsa/libcrypto-lib-ml_dsa_ntt_vec128.o \
+          | awk '$2 == "T"' \
+          | grep -cE ' (ossl_ml_dsa_poly_ntt_vec128|ossl_ml_dsa_poly_ntt_inverse_vec128|ossl_poly_ntt_mult_scalar_vec128)$' \
+          | grep -q '^3$'
+
+    - name: check ML-DSA dispatcher references VX symbols
+      # Verify that ml_dsa_ntt.o (the dispatcher) has unresolved external
+      # references (U-type) to all three vec128 functions.  This proves the
+      # dispatch block was compiled in — the standalone object check above
+      # only verifies that the symbols exist in the vec128 file, not that
+      # the dispatcher actually calls them.
+      if: matrix.platform.nm_check != ''
+      run: |
+        ${{ matrix.platform.arch }}-nm crypto/ml_dsa/libcrypto-lib-ml_dsa_ntt.o \
+          | awk '$1 == "U"' \
+          | grep -cE ' (ossl_ml_dsa_poly_ntt_vec128|ossl_ml_dsa_poly_ntt_inverse_vec128|ossl_poly_ntt_mult_scalar_vec128)$' \
+          | grep -q '^3$'
+
     - name: install qemu
       if: matrix.platform.tests != 'none'
       run: sudo apt-get -yq --allow-unauthenticated --allow-downgrades --allow-remove-essential --allow-change-held-packages install qemu-user
@@ -244,13 +476,19 @@ jobs:
                   TESTS="-test_afalg" \
                   QEMU_LD_PREFIX=/usr/${{ matrix.platform.arch }}
     - name: make some tests
-      if: env.EXTENDED == 'true' && matrix.platform.tests != 'none' && matrix.platform.tests != ''
+      if: >-
+        matrix.platform.tests != 'none' &&
+        matrix.platform.tests != '' &&
+        (env.EXTENDED == 'true' || matrix.platform.arch == 's390x-linux-gnu')
       run: |
         .github/workflows/make-test \
                   TESTS="${{ matrix.platform.tests }} -test_afalg" \
                   QEMU_LD_PREFIX=/usr/${{ matrix.platform.arch }}
     - name: make evp tests
-      if: env.EXTENDED != 'true' && matrix.platform.tests != 'none'
+      if: >-
+        env.EXTENDED != 'true' &&
+        matrix.platform.tests != 'none' &&
+        (matrix.platform.tests == '' || matrix.platform.arch != 's390x-linux-gnu')
       run: |
         .github/workflows/make-test \
                   TESTS="test_evp*" \
@@ -259,6 +497,6 @@ jobs:
       if: success() || failure()
       uses: actions/upload-artifact@v5
       with:
-        name: "cross-compiles@${{ matrix.platform.arch }}"
+        name: "cross-compiles@${{ matrix.platform.label || matrix.platform.arch }}"
         path: artifacts.tar.gz
         if-no-files-found: ignore
diff --git a/Configurations/unix-Makefile.tmpl b/Configurations/unix-Makefile.tmpl
index f8a5b1612e..e2f4d54e4c 100644
--- a/Configurations/unix-Makefile.tmpl
+++ b/Configurations/unix-Makefile.tmpl
@@ -529,6 +529,15 @@ BIN_LDFLAGS={- join(' ', $target{bin_lflags} || (),
                          '$(CNF_LDFLAGS)', '$(LDFLAGS)') -}
 BIN_EX_LIBS=$(CNF_EX_LIBS) $(EX_LIBS)

+# Extra flags appended only to the two s390x VX vector source files
+# (ml_kem_vec128.c and ml_dsa_ntt_vec128.c).  For GCC this is empty
+# (the #pragma GCC target inside each file sets arch=z13,vx for that TU).
+# For Clang this contains -fzvector when the probe determined it is needed
+# to make <vecintrin.h> available.  The per-object rule in src2obj also
+# appends -march=z13 -mvx unconditionally for these two files so that
+# Clang defines __VX__ and emits the vector entry points.
+S390X_VEC_CFLAGS={- $config{s390x_vector_cflags} // '' -}
+
 CMOCKA_LIBS={- $config{cmocka_libs} // '' -}
 DETOURS_LIBS={- $config{detours_libs} // '' -}

@@ -1360,7 +1369,9 @@ providers/fips.module.sources.new: configdata.pm
 		   crypto/sha/asm/*.pl \
 		   crypto/slh_dsa/asm/*.pl \
 		   crypto/*cpuid.pl crypto/*cpuid.S \
-		   crypto/*cap.c; do \
+		   crypto/*cap.c \
+		   crypto/ml_kem/ml_kem_vec128.c \
+		   crypto/ml_dsa/ml_dsa_ntt_vec128.c; do \
 	    test -e "$$x" && echo "$$x"; \
 	  done \
 	) | grep -v sm2p256 | sort | uniq > providers/fips.module.sources.new
@@ -1797,6 +1808,15 @@ EOF
       my $deps = join(" ", @srcs, @{$args{deps}});
       my $incs = join("", map { " -I".$_ } @{$args{incs}});
       my $defs = join("", map { " -D".$_ } @{$args{defs}});
+      # Per-file extra flags for the two s390x VX vector translation units.
+      # GCC uses #pragma GCC target inside each file so only needs an empty
+      # suffix.  Clang needs -march=z13 -mvx (to define __VX__ and generate
+      # VX instructions) plus the optional $(S390X_VEC_CFLAGS) (-fzvector
+      # for older Clang to unlock <vecintrin.h>).
+      my $extra_cflags = "";
+      if (grep m{(?:^|/)(ml_kem_vec128|ml_dsa_ntt_vec128)\.c$}, @srcs) {
+          $extra_cflags = " -march=z13 -mvx \$(S390X_VEC_CFLAGS)";
+      }
       my $cmd;
       my $cmdflags;
       my $cmdcompile;
@@ -1843,7 +1863,7 @@ EOF
       } elsif ($makedep_scheme eq 'gcc' && !grep /\.rc$/, @srcs) {
           $recipe .= <<"EOF";
 $obj: $deps
-	$cmd $incs $defs $cmdflags -MMD -MF $dep.tmp -c -o \$\@ $srcs
+	$cmd $incs $defs $cmdflags$extra_cflags -MMD -MF $dep.tmp -c -o \$\@ $srcs
 	\@touch $dep.tmp
 	\@if cmp $dep.tmp $dep > /dev/null 2> /dev/null; then \\
 		rm -f $dep.tmp; \\
@@ -1854,11 +1874,11 @@ EOF
       } else {
           $recipe .= <<"EOF";
 $obj: $deps
-	$cmd $incs $defs $cmdflags $cmdcompile -o \$\@ $srcs
+	$cmd $incs $defs $cmdflags$extra_cflags $cmdcompile -o \$\@ $srcs
 EOF
           if ($makedep_scheme eq 'makedepend') {
               $recipe .= <<"EOF";
-	\$(MAKEDEPEND) -f- -Y -- $incs $cmdflags -- $srcs 2>/dev/null \\
+	\$(MAKEDEPEND) -f- -Y -- $incs $cmdflags$extra_cflags -- $srcs 2>/dev/null \\
 	    > $dep
 EOF
           }
diff --git a/Configure b/Configure
index 9ab450b153..b4292d3136 100755
--- a/Configure
+++ b/Configure
@@ -1846,6 +1846,115 @@ if (!$disabled{asm} && !$predefined_C{__MACH__} && $^O ne 'VMS' && !$predefined_
     }
 }

+# On s390x, <vecintrin.h> (included by ml_kem_vec128.c and
+# ml_dsa_ntt_vec128.c) requires -fzvector with Clang; GCC accepts it
+# without that flag.  We probe for the required intrinsics (vec_perm,
+# vec_mulh, vec_min) and store the result in $config{s390x_vector_cflags}:
+#   undef  — probe failed; vec128 back-end will be excluded from the build
+#   ""     — probe passed without extra flags (GCC)
+#   "-fzvector" — probe passed only with -fzvector (older Clang)
+# The value is applied ONLY to the two vec128 translation units.
+#
+# GCC: "#pragma GCC target("arch=z13,vx")" inside each vec128 file scopes
+#      VX code generation to that TU; no global -march is needed.
+#      GCC also does not require -fzvector, so $config{s390x_vector_cflags}
+#      will be empty ("") for GCC builds.
+# Clang: does not honour #pragma GCC target for preprocessor macros; it
+#        requires -march=z13 (passed per-file via the Makefile wrapper rule
+#        generated from $config{s390x_vector_cflags} and the explicit
+#        -march=z13 -mvx flags).  -fzvector is additionally needed on older
+#        Clang versions to make <vecintrin.h> available.
+#
+# $config{s390x_vector_cflags} is consumed by Configurations/unix-Makefile.tmpl
+# to emit a S390X_VEC_CFLAGS Makefile variable and to append it (alongside
+# -march=z13 -mvx) to the per-object compile command for the two vec128 files.
+# The build.info files for ml_kem and ml_dsa gate the vec128 source files on
+# defined($config{s390x_vector_cflags}), so a failed probe cleanly excludes
+# them from the build.
+if (!$disabled{asm} && (($target{asm_arch} // '') eq 's390x')) {
+    my $cc  = $config{CROSS_COMPILE} . $config{CC};
+    my $tf  = "vx_probe_$$.c";
+    my $obj = "vx_probe_$$.o";
+    my @all_cflags = (
+        @{$config{CPPFLAGS} // []},
+        @{$config{cppflags} // []},
+        (ref($target{cflags}) ? @{$target{cflags}} : ($target{cflags} // ())),
+        @{$config{cflags}},
+        @{$config{CFLAGS}   // []},
+    );
+    # --ossl-strict-warnings is a Configure-internal placeholder that is
+    # expanded to real warning flags later (after this probe runs).  Drop it
+    # here so the probe compiler invocation does not see an unknown option.
+    my $flagstr = join(' ', grep { $_ ne '--ossl-strict-warnings' } @all_cflags);
+
+    open(my $fh, '>', $tf) or die "Cannot write $tf: $!";
+    # The probe exercises all intrinsics used by the vec128 back-end:
+    #   vec_perm      – used for even/odd lane extraction (ML-KEM, ML-DSA)
+    #   vec_mulh i16  – signed 16-bit multiply-high (ML-KEM Montgomery multiply)
+    #   vec_mulh i32  – signed 32-bit multiply-high (ML-DSA Montgomery multiply)
+    #   vec_min       – unsigned 16-bit lane-wise minimum (ML-KEM reduce_once_vec128)
+    # The probe uses plain GCC vector types (no __may_alias__) because the
+    # s390x built-in signatures for vec_perm/vec_mulh/vec_min require their
+    # native types.  The production source files handle aliasing via explicit
+    # pointer casts and the separate vec_int16_noalias_t / vec_uint16_t types
+    # defined in ml_kem_vec128.c, so the probe correctly reflects what
+    # the compiler must accept when building those TUs.
+    print $fh <<'END_PROBE';
+#include <vecintrin.h>
+#include <stdint.h>
+typedef __attribute__((vector_size(16))) int16_t  vec_i16;
+typedef __attribute__((vector_size(16))) uint16_t vec_u16;
+typedef __attribute__((vector_size(16))) uint8_t  vec_u8;
+typedef __attribute__((vector_size(16))) int32_t  vec_i32;
+static vec_i16 probe(int16_t a, int16_t b, uint8_t c) {
+    vec_u8  perm    = vec_splats(c);
+    vec_i16 va      = vec_splats(a);
+    vec_i16 vb      = vec_splats(b);
+    vec_i32 va32    = vec_splats((int32_t)a);
+    vec_i32 vb32    = vec_splats((int32_t)b);
+    /* vec_perm: even/odd lane extraction */
+    vec_i16 vperm   = (vec_i16)vec_perm(va, vb, perm);
+    /* vec_mulh: signed 16-bit multiply-high (ML-KEM) */
+    vec_i16 vmulh   = vec_mulh(va, vb);
+    /* vec_mulh: signed 32-bit multiply-high (ML-DSA) */
+    vec_i32 vmulh32 = vec_mulh(va32, vb32);
+    /* vec_min: unsigned 16-bit lane-wise minimum (ML-KEM) */
+    vec_u16 vmin    = vec_min((vec_u16)va, (vec_u16)vb);
+    /* reinterpret-cast vec_i32 → vec_i16 to combine all results */
+    return vperm + vmulh + (vec_i16)vmin + (vec_i16)vmulh32;
+}
+int main(void) { vec_i16 v = probe(1, 2, 0); return (int)v[0]; }
+END_PROBE
+    close($fh);
+
+    # The vec128 TUs are always compiled with -march=z13 -mvx regardless of
+    # the global baseline march.  The probe must use the same target flags so
+    # that the result is valid for those TUs.  With an older baseline (e.g.
+    # -march=z10) Clang rejects <vecintrin.h> even with -fzvector unless the
+    # target already includes z13/VX.
+    #
+    # First try without -fzvector (GCC works without it).
+    # If that fails, retry with -fzvector (Clang needs it for <vecintrin.h>).
+    # Store the result in s390x_vector_cflags — NOT in the global cflags —
+    # so only the two vec128 TUs see it.
+    # If both attempts fail the compiler lacks the required support; disable
+    # the s390x vec128 back-end so the build can still complete using the
+    # generic implementation.
+    $config{s390x_vector_cflags} = undef;   # undef == probe failed / disabled
+    my $vec_target = "-march=z13 -mvx";
+    my $rc = system("$cc $flagstr $vec_target -c -o $obj $tf 2>/dev/null");
+    if ($rc != 0) {
+        $rc = system("$cc $flagstr $vec_target -fzvector -c -o $obj $tf 2>/dev/null");
+        if ($rc == 0) {
+            $config{s390x_vector_cflags} = "-fzvector";
+        }
+        # else: both attempts failed — leave s390x_vector_cflags as undef
+    } else {
+        $config{s390x_vector_cflags} = "";
+    }
+    unlink($tf, $obj);
+}
+
 # Deal with bn_ops ###################################################

 $config{bn_ll}                  =0;