Commit 1c87d7d6e3 for openssl.org
commit 1c87d7d6e3450bad2416236675975a9677689ab7
Author: Timo Keller <tkeller@linux.ibm.com>
Date: Wed Aug 19 09:36:48 2026 +0200
Add s390x cross-compile workflow
Add s390x cross-compile workflow and change Configure for the s390x-
optimized implementations of ML-KEM and ML-DSA.
Signed-off-by: Timo Keller <tkeller@linux.ibm.com>
Assisted-by: IBM Bob:2.0.3
Reviewed-by: Mounir Idrassi <mounir.idrassi@idrix.fr>
Reviewed-by: Tomas Mraz <tomas@openssl.foundation>
Merge-date: Tue Sep 29 16:47:14 2026
Merged-from: https://github.com/openssl/openssl/pull/31929
diff --git a/.github/workflows/cross-compiles.yml b/.github/workflows/cross-compiles.yml
index 9f32de5232..25a8486e13 100644
--- a/.github/workflows/cross-compiles.yml
+++ b/.github/workflows/cross-compiles.yml
@@ -63,6 +63,14 @@ jobs:
# this string will be used as content for the OpenSSL
# capabilities variable.
# ppa: Launchpad PPA repository to download packages from.
+ # compiler: optional; set to "clang" to build with Clang instead of
+ # GCC. The cross-compile prefix is still used for the
+ # binutils (assembler, linker, nm …).
+ # nm_check: optional; when set, a post-build step verifies that the
+ # six ML-KEM VX vector-entry-point symbols are present as
+ # text symbols in the ml_kem_vec128.o object file.
+ # label: optional; used as the artifact name suffix in place of
+ # arch, to disambiguate matrix jobs that share the same arch.
platform: [
{
arch: i386-pc-msdosdjgpp,
@@ -138,10 +146,143 @@ jobs:
target: linux64-riscv64,
fips: no
}, {
+ # Baseline s390x job. No qemucpu override — QEMU's default TCG
+ # model avoids the "CPU model features not available" admission
+ # failure that named models (z13, z13-base, …) trigger on x86
+ # GitHub Actions runners. Named z/Architecture models include
+ # mandatory CPACF sub-features (kimd, km, klmd, …) that require
+ # real s390x hardware or KVM; the TCG emulator cannot satisfy them
+ # and QEMU 8.x exits fatally before OpenSSL even starts.
+ # z900 zeros every CPACF vector and all stfle facility bits (no VX),
+ # so OpenSSL uses the pure-software scalar fallback paths.
arch: s390x-linux-gnu,
libs: libc6-dev-s390x-cross,
target: linux64-s390x,
- fips: no
+ fips: no,
+ opensslcapsname: s390xcap,
+ opensslcaps: z900
+ }, {
+ # Build-only: verify no-asm does not produce undefined references to
+ # OPENSSL_s390xcap_P (which is only compiled when asm is enabled).
+ arch: s390x-linux-gnu,
+ libs: libc6-dev-s390x-cross,
+ target: no-shared no-asm linux64-s390x,
+ fips: no,
+ tests: none
+ }, {
+ # GCC / VX enabled (FIPS).
+ # The base linux64-s390x target carries no -march flag, so the
+ # whole library is safe to execute on any s390x CPU.
+ # ml_kem_vec128.c and ml_dsa_ntt_vec128.c use "#pragma GCC target"
+ # to scope VX codegen to those TUs only.
+ #
+ # Why qemucpu: max instead of z13:
+ # QEMU 8.x named z/Architecture CPU models (z13, z13-base, …)
+ # include mandatory CPACF sub-features (kimd-sha-1, km-aes-*, …)
+ # that the TCG softmmu cannot provide on a non-s390x host. QEMU
+ # prints a fatal "Some features … not available" warning and exits
+ # with a non-zero status *before* loading the OpenSSL binary. The
+ # Perl test harness sees the non-zero exit code and marks every
+ # test as failed — even though the OpenSSL code is correct.
+ # "max" enables all features the TCG emulator actually supports
+ # (which includes VX vector-register instructions) without
+ # requiring hardware CPACF, so QEMU starts cleanly.
+ # The OPENSSL_s390xcap override below controls what capability
+ # bits OpenSSL reads at runtime, so VX detection is correct
+ # regardless of which CPU model QEMU uses.
+ #
+ # The nm_check steps assert VX symbols are present and that the
+ # dispatcher (ml_dsa_ntt.o) has unresolved references to them.
+ arch: s390x-linux-gnu,
+ libs: libc6-dev-s390x-cross,
+ target: linux64-s390x,
+ qemucpu: max,
+ opensslcapsname: s390xcap,
+ # Enable the VX facility bit (stfle[2] bit 129 = 0x4000000000000000)
+ # while zeroing every CPACF instruction vector (kimd/klmd/km/…).
+ # QEMU emulates the VX vector-register instructions but cannot
+ # execute CPACF instructions; this lets the VX code paths run
+ # under QEMU without triggering illegal-instruction traps.
+ opensslcaps: "stfle:0:0:4000000000000000;kimd:0:0;klmd:0:0;km:0:0;kmc:0:0;kmac:0:0;kmctr:0:0;kmo:0:0;kmf:0:0;prno:0:0;kma:0:0;pcc:0:0;kdsa:0:0",
+ tests: test_internal_ml_kem test_evp_extra_ml_kem test_ml_kem_codecs test_internal_ml_dsa test_ml_dsa test_ml_dsa_codecs test_evp*,
+ nm_check: "yes",
+ label: s390x-gcc-z13-vx-on-fips
+ }, {
+ # GCC / VX masked off (no FIPS).
+ # Same base build (linux64-s390x, no -march) as the VX-on job
+ # above, but OPENSSL_s390xcap=z900 zeros all stfle facility bits
+ # (including VX) so S390X_VX_CAPABLE is false and the generic
+ # scalar path is exercised on the same binary.
+ # qemucpu: max — see the VX-on job comment for why z13 is avoided.
+ arch: s390x-linux-gnu,
+ libs: libc6-dev-s390x-cross,
+ target: linux64-s390x,
+ fips: no,
+ qemucpu: max,
+ opensslcapsname: s390xcap,
+ # z900 zeros all CPACF vectors and all stfle facility bits including
+ # VX, exercising the scalar fallback path without QEMU CPACF traps.
+ opensslcaps: z900,
+ tests: test_internal_ml_kem test_evp_extra_ml_kem test_ml_kem_codecs test_internal_ml_dsa test_ml_dsa test_ml_dsa_codecs test_evp*,
+ label: s390x-gcc-z13-vx-off
+ }, {
+ # Clang / z13 / VX enabled (FIPS).
+ # Clang does not honour #pragma GCC target for preprocessor-level
+ # macros, so -march=z13 must be passed at compile time to make
+ # Clang define __VX__ and enable the VX code paths. -march=z13
+ # is safe to run on any z13+ CPU and does not imply z14 instructions.
+ # Configure also adds -fzvector when Clang needs it to access
+ # <vecintrin.h>; that flag only unlocks the header, not codegen.
+ # Verifies that Clang emits all six ML-KEM and three ML-DSA VX
+ # entry points and that the ML-DSA dispatcher references them.
+ # qemucpu: max — see the GCC VX-on job comment for why z13 is avoided.
+ arch: s390x-linux-gnu,
+ libs: libc6-dev-s390x-cross,
+ target: -march=z13 linux64-s390x,
+ compiler: clang,
+ qemucpu: max,
+ opensslcapsname: s390xcap,
+ opensslcaps: "stfle:0:0:4000000000000000;kimd:0:0;klmd:0:0;km:0:0;kmc:0:0;kmac:0:0;kmctr:0:0;kmo:0:0;kmf:0:0;prno:0:0;kma:0:0;pcc:0:0;kdsa:0:0",
+ tests: test_internal_ml_kem test_evp_extra_ml_kem test_ml_kem_codecs test_internal_ml_dsa test_ml_dsa test_ml_dsa_codecs test_evp*,
+ nm_check: "yes",
+ label: s390x-clang-z13-vx-on-fips
+ }, {
+ # Clang / z13 / VX masked off (no FIPS).
+ # Same -march=z13 build as the Clang VX-on job but
+ # OPENSSL_s390xcap=z900 zeros all stfle bits (no VX, no CPACF) so
+ # the generic scalar path is exercised on the same binary.
+ # qemucpu: max — see the GCC VX-on job comment for why z13 is avoided.
+ arch: s390x-linux-gnu,
+ libs: libc6-dev-s390x-cross,
+ target: -march=z13 linux64-s390x,
+ fips: no,
+ compiler: clang,
+ qemucpu: max,
+ opensslcapsname: s390xcap,
+ opensslcaps: z900,
+ tests: test_internal_ml_kem test_evp_extra_ml_kem test_ml_kem_codecs test_internal_ml_dsa test_ml_dsa test_ml_dsa_codecs test_evp*,
+ label: s390x-clang-z13-vx-off
+ }, {
+ # Clang / z10 baseline / VX enabled (FIPS).
+ # Tests the intended per-TU isolation design: the whole library is
+ # compiled with -march=z10 (no VX), but Configure detects the s390x
+ # target via $target{asm_arch} and runs the feature probe with the
+ # effective compiler flags plus -march=z13 -mvx. Only the two vec128
+ # TUs (ml_kem_vec128.c, ml_dsa_ntt_vec128.c) receive -march=z13 -mvx
+ # (and -fzvector if Clang needs it); the rest of the library is compiled
+ # to the z10 baseline. This verifies that the scalar fallback objects
+ # are correct z10 code while the two vector TUs are valid z13+VX code.
+ # qemucpu: max — see the GCC VX-on job comment for why z13 is avoided.
+ arch: s390x-linux-gnu,
+ libs: libc6-dev-s390x-cross,
+ target: -march=z10 linux64-s390x,
+ compiler: clang,
+ qemucpu: max,
+ opensslcapsname: s390xcap,
+ opensslcaps: "stfle:0:0:4000000000000000;kimd:0:0;klmd:0:0;km:0:0;kmc:0:0;kmac:0:0;kmctr:0:0;kmo:0:0;kmf:0:0;prno:0:0;kma:0:0;pcc:0:0;kdsa:0:0",
+ tests: test_internal_ml_kem test_evp_extra_ml_kem test_ml_kem_codecs test_internal_ml_dsa test_ml_dsa test_ml_dsa_codecs test_evp*,
+ nm_check: "yes",
+ label: s390x-clang-z10-vx-on-fips
}, {
arch: sh4-linux-gnu,
libs: libc6-dev-sh4-cross,
@@ -189,36 +330,127 @@ jobs:
if: matrix.platform.ppa != ''
run: |
sudo add-apt-repository ppa:${{ matrix.platform.ppa }}
- - name: install packages
+ - name: install packages (GCC)
+ if: matrix.platform.compiler != 'clang'
run: |
sudo apt-get update
sudo apt-get -yq --allow-unauthenticated --allow-downgrades --allow-remove-essential --allow-change-held-packages install \
gcc-${{ matrix.platform.arch }} \
${{ matrix.platform.libs }}
+ - name: install packages (Clang)
+ if: matrix.platform.compiler == 'clang'
+ run: |
+ sudo apt-get update
+ sudo apt-get -yq --allow-unauthenticated --allow-downgrades --allow-remove-essential --allow-change-held-packages install \
+ clang \
+ binutils-${{ matrix.platform.arch }} \
+ gcc-${{ matrix.platform.arch }} \
+ ${{ matrix.platform.libs }}
+ # Detect sysroot: Ubuntu <=22.04 uses .../sys-root, 24.04+ installs directly.
+ SYSROOT=/usr/${{ matrix.platform.arch }}/sys-root
+ [ -d "$SYSROOT" ] || SYSROOT=/usr/${{ matrix.platform.arch }}
+ echo "CROSS_SYSROOT=$SYSROOT" >> $GITHUB_ENV
- uses: actions/checkout@v6
with:
persist-credentials: false
- name: checkout fuzz/corpora submodule
run: git submodule update --init --depth 1 fuzz/corpora
- - name: config with FIPS
- if: matrix.platform.fips != 'no'
+ - name: config with FIPS (GCC)
+ if: matrix.platform.fips != 'no' && matrix.platform.compiler != 'clang'
run: |
./config --banner=Configured --strict-warnings enable-fips enable-lms \
--cross-compile-prefix=${{ matrix.platform.arch }}- \
${{ matrix.platform.target }}
- - name: config without FIPS
- if: matrix.platform.fips == 'no'
+ - name: config without FIPS (GCC)
+ if: matrix.platform.fips == 'no' && matrix.platform.compiler != 'clang'
run: |
./config --banner=Configured --strict-warnings enable-lms \
--cross-compile-prefix=${{ matrix.platform.arch }}- \
${{ matrix.platform.target }}
+ - name: config with FIPS (Clang)
+ if: matrix.platform.fips != 'no' && matrix.platform.compiler == 'clang'
+ env:
+ CC: clang --target=${{ matrix.platform.arch }} --sysroot=${{ env.CROSS_SYSROOT }}
+ AR: ${{ matrix.platform.arch }}-ar
+ RANLIB: ${{ matrix.platform.arch }}-ranlib
+ AS: ${{ matrix.platform.arch }}-as
+ # Reset the linker sysroot to / so that absolute-path linker scripts
+ # shipped by libc6-dev-*-cross (e.g. libc.so referencing
+ # /usr/<arch>/lib/libc.so.6) are resolved against the real root
+ # rather than being doubled-prefixed under CROSS_SYSROOT.
+ # LDFLAGS is picked up by OpenSSL's Configure and forwarded to the
+ # linker only, leaving compiler-detection probes unaffected.
+ LDFLAGS: -Wl,--sysroot=/
+ run: |
+ ./config --banner=Configured --strict-warnings enable-fips enable-lms \
+ ${{ matrix.platform.target }}
+ - name: config without FIPS (Clang)
+ if: matrix.platform.fips == 'no' && matrix.platform.compiler == 'clang'
+ env:
+ CC: clang --target=${{ matrix.platform.arch }} --sysroot=${{ env.CROSS_SYSROOT }}
+ AR: ${{ matrix.platform.arch }}-ar
+ RANLIB: ${{ matrix.platform.arch }}-ranlib
+ AS: ${{ matrix.platform.arch }}-as
+ LDFLAGS: -Wl,--sysroot=/
+ run: |
+ ./config --banner=Configured --strict-warnings enable-lms \
+ ${{ matrix.platform.target }}
- name: config dump
run: ./configdata.pm --dump
- name: make
run: make -s -j4
+ - name: check ML-KEM VX vector symbols
+ if: matrix.platform.nm_check != ''
+ run: |
+ ${{ matrix.platform.arch }}-nm crypto/ml_kem/libcrypto-lib-ml_kem_vec128.o \
+ | awk '$2 == "T"' \
+ | grep -cE ' (ossl_ml_kem_scalar_ntt_vec128|ossl_ml_kem_scalar_inverse_ntt_vec128|ossl_ml_kem_scalar_inverse_ntt_demontgomerize_vec128|ossl_ml_kem_scalar_mult_add_vec128|ossl_ml_kem_inner_product_montgomery_vec128|ossl_ml_kem_matrix_mult_intt_vec128)$' \
+ | grep -q '^6$'
+
+ - name: check ML-KEM dispatcher references VX symbols
+ # Verify that ml_kem.o (the dispatcher) has unresolved external
+ # references (U-type) to all six vec128 functions and also references
+ # OPENSSL_s390xcap_P. This proves that the dispatch block in
+ # ml_kem_ntt_init() was compiled in — the standalone vec128 object
+ # check above only verifies that the symbols exist in the vec128 file,
+ # not that the dispatcher actually calls them at runtime.
+ # Note: 'nm' omits the address column for undefined symbols, so
+ # undefined entries look like " U symbol_name" and
+ # $1 == "U" is the correct awk field (not $2).
+ if: matrix.platform.nm_check != ''
+ run: |
+ ${{ matrix.platform.arch }}-nm crypto/ml_kem/libcrypto-lib-ml_kem.o \
+ | awk '$1 == "U"' \
+ | grep -cE ' (ossl_ml_kem_scalar_ntt_vec128|ossl_ml_kem_scalar_inverse_ntt_vec128|ossl_ml_kem_scalar_inverse_ntt_demontgomerize_vec128|ossl_ml_kem_scalar_mult_add_vec128|ossl_ml_kem_inner_product_montgomery_vec128|ossl_ml_kem_matrix_mult_intt_vec128)$' \
+ | grep -q '^6$'
+ ${{ matrix.platform.arch }}-nm crypto/ml_kem/libcrypto-lib-ml_kem.o \
+ | awk '$1 == "U"' \
+ | grep -q ' OPENSSL_s390xcap_P$'
+
+ - name: check ML-DSA VX vector symbols
+ if: matrix.platform.nm_check != ''
+ run: |
+ ${{ matrix.platform.arch }}-nm crypto/ml_dsa/libcrypto-lib-ml_dsa_ntt_vec128.o \
+ | awk '$2 == "T"' \
+ | grep -cE ' (ossl_ml_dsa_poly_ntt_vec128|ossl_ml_dsa_poly_ntt_inverse_vec128|ossl_poly_ntt_mult_scalar_vec128)$' \
+ | grep -q '^3$'
+
+ - name: check ML-DSA dispatcher references VX symbols
+ # Verify that ml_dsa_ntt.o (the dispatcher) has unresolved external
+ # references (U-type) to all three vec128 functions. This proves the
+ # dispatch block was compiled in — the standalone object check above
+ # only verifies that the symbols exist in the vec128 file, not that
+ # the dispatcher actually calls them.
+ if: matrix.platform.nm_check != ''
+ run: |
+ ${{ matrix.platform.arch }}-nm crypto/ml_dsa/libcrypto-lib-ml_dsa_ntt.o \
+ | awk '$1 == "U"' \
+ | grep -cE ' (ossl_ml_dsa_poly_ntt_vec128|ossl_ml_dsa_poly_ntt_inverse_vec128|ossl_poly_ntt_mult_scalar_vec128)$' \
+ | grep -q '^3$'
+
- name: install qemu
if: matrix.platform.tests != 'none'
run: sudo apt-get -yq --allow-unauthenticated --allow-downgrades --allow-remove-essential --allow-change-held-packages install qemu-user
@@ -244,13 +476,19 @@ jobs:
TESTS="-test_afalg" \
QEMU_LD_PREFIX=/usr/${{ matrix.platform.arch }}
- name: make some tests
- if: env.EXTENDED == 'true' && matrix.platform.tests != 'none' && matrix.platform.tests != ''
+ if: >-
+ matrix.platform.tests != 'none' &&
+ matrix.platform.tests != '' &&
+ (env.EXTENDED == 'true' || matrix.platform.arch == 's390x-linux-gnu')
run: |
.github/workflows/make-test \
TESTS="${{ matrix.platform.tests }} -test_afalg" \
QEMU_LD_PREFIX=/usr/${{ matrix.platform.arch }}
- name: make evp tests
- if: env.EXTENDED != 'true' && matrix.platform.tests != 'none'
+ if: >-
+ env.EXTENDED != 'true' &&
+ matrix.platform.tests != 'none' &&
+ (matrix.platform.tests == '' || matrix.platform.arch != 's390x-linux-gnu')
run: |
.github/workflows/make-test \
TESTS="test_evp*" \
@@ -259,6 +497,6 @@ jobs:
if: success() || failure()
uses: actions/upload-artifact@v5
with:
- name: "cross-compiles@${{ matrix.platform.arch }}"
+ name: "cross-compiles@${{ matrix.platform.label || matrix.platform.arch }}"
path: artifacts.tar.gz
if-no-files-found: ignore
diff --git a/Configurations/unix-Makefile.tmpl b/Configurations/unix-Makefile.tmpl
index f8a5b1612e..e2f4d54e4c 100644
--- a/Configurations/unix-Makefile.tmpl
+++ b/Configurations/unix-Makefile.tmpl
@@ -529,6 +529,15 @@ BIN_LDFLAGS={- join(' ', $target{bin_lflags} || (),
'$(CNF_LDFLAGS)', '$(LDFLAGS)') -}
BIN_EX_LIBS=$(CNF_EX_LIBS) $(EX_LIBS)
+# Extra flags appended only to the two s390x VX vector source files
+# (ml_kem_vec128.c and ml_dsa_ntt_vec128.c). For GCC this is empty
+# (the #pragma GCC target inside each file sets arch=z13,vx for that TU).
+# For Clang this contains -fzvector when the probe determined it is needed
+# to make <vecintrin.h> available. The per-object rule in src2obj also
+# appends -march=z13 -mvx unconditionally for these two files so that
+# Clang defines __VX__ and emits the vector entry points.
+S390X_VEC_CFLAGS={- $config{s390x_vector_cflags} // '' -}
+
CMOCKA_LIBS={- $config{cmocka_libs} // '' -}
DETOURS_LIBS={- $config{detours_libs} // '' -}
@@ -1360,7 +1369,9 @@ providers/fips.module.sources.new: configdata.pm
crypto/sha/asm/*.pl \
crypto/slh_dsa/asm/*.pl \
crypto/*cpuid.pl crypto/*cpuid.S \
- crypto/*cap.c; do \
+ crypto/*cap.c \
+ crypto/ml_kem/ml_kem_vec128.c \
+ crypto/ml_dsa/ml_dsa_ntt_vec128.c; do \
test -e "$$x" && echo "$$x"; \
done \
) | grep -v sm2p256 | sort | uniq > providers/fips.module.sources.new
@@ -1797,6 +1808,15 @@ EOF
my $deps = join(" ", @srcs, @{$args{deps}});
my $incs = join("", map { " -I".$_ } @{$args{incs}});
my $defs = join("", map { " -D".$_ } @{$args{defs}});
+ # Per-file extra flags for the two s390x VX vector translation units.
+ # GCC uses #pragma GCC target inside each file so only needs an empty
+ # suffix. Clang needs -march=z13 -mvx (to define __VX__ and generate
+ # VX instructions) plus the optional $(S390X_VEC_CFLAGS) (-fzvector
+ # for older Clang to unlock <vecintrin.h>).
+ my $extra_cflags = "";
+ if (grep m{(?:^|/)(ml_kem_vec128|ml_dsa_ntt_vec128)\.c$}, @srcs) {
+ $extra_cflags = " -march=z13 -mvx \$(S390X_VEC_CFLAGS)";
+ }
my $cmd;
my $cmdflags;
my $cmdcompile;
@@ -1843,7 +1863,7 @@ EOF
} elsif ($makedep_scheme eq 'gcc' && !grep /\.rc$/, @srcs) {
$recipe .= <<"EOF";
$obj: $deps
- $cmd $incs $defs $cmdflags -MMD -MF $dep.tmp -c -o \$\@ $srcs
+ $cmd $incs $defs $cmdflags$extra_cflags -MMD -MF $dep.tmp -c -o \$\@ $srcs
\@touch $dep.tmp
\@if cmp $dep.tmp $dep > /dev/null 2> /dev/null; then \\
rm -f $dep.tmp; \\
@@ -1854,11 +1874,11 @@ EOF
} else {
$recipe .= <<"EOF";
$obj: $deps
- $cmd $incs $defs $cmdflags $cmdcompile -o \$\@ $srcs
+ $cmd $incs $defs $cmdflags$extra_cflags $cmdcompile -o \$\@ $srcs
EOF
if ($makedep_scheme eq 'makedepend') {
$recipe .= <<"EOF";
- \$(MAKEDEPEND) -f- -Y -- $incs $cmdflags -- $srcs 2>/dev/null \\
+ \$(MAKEDEPEND) -f- -Y -- $incs $cmdflags$extra_cflags -- $srcs 2>/dev/null \\
> $dep
EOF
}
diff --git a/Configure b/Configure
index 9ab450b153..b4292d3136 100755
--- a/Configure
+++ b/Configure
@@ -1846,6 +1846,115 @@ if (!$disabled{asm} && !$predefined_C{__MACH__} && $^O ne 'VMS' && !$predefined_
}
}
+# On s390x, <vecintrin.h> (included by ml_kem_vec128.c and
+# ml_dsa_ntt_vec128.c) requires -fzvector with Clang; GCC accepts it
+# without that flag. We probe for the required intrinsics (vec_perm,
+# vec_mulh, vec_min) and store the result in $config{s390x_vector_cflags}:
+# undef — probe failed; vec128 back-end will be excluded from the build
+# "" — probe passed without extra flags (GCC)
+# "-fzvector" — probe passed only with -fzvector (older Clang)
+# The value is applied ONLY to the two vec128 translation units.
+#
+# GCC: "#pragma GCC target("arch=z13,vx")" inside each vec128 file scopes
+# VX code generation to that TU; no global -march is needed.
+# GCC also does not require -fzvector, so $config{s390x_vector_cflags}
+# will be empty ("") for GCC builds.
+# Clang: does not honour #pragma GCC target for preprocessor macros; it
+# requires -march=z13 (passed per-file via the Makefile wrapper rule
+# generated from $config{s390x_vector_cflags} and the explicit
+# -march=z13 -mvx flags). -fzvector is additionally needed on older
+# Clang versions to make <vecintrin.h> available.
+#
+# $config{s390x_vector_cflags} is consumed by Configurations/unix-Makefile.tmpl
+# to emit a S390X_VEC_CFLAGS Makefile variable and to append it (alongside
+# -march=z13 -mvx) to the per-object compile command for the two vec128 files.
+# The build.info files for ml_kem and ml_dsa gate the vec128 source files on
+# defined($config{s390x_vector_cflags}), so a failed probe cleanly excludes
+# them from the build.
+if (!$disabled{asm} && (($target{asm_arch} // '') eq 's390x')) {
+ my $cc = $config{CROSS_COMPILE} . $config{CC};
+ my $tf = "vx_probe_$$.c";
+ my $obj = "vx_probe_$$.o";
+ my @all_cflags = (
+ @{$config{CPPFLAGS} // []},
+ @{$config{cppflags} // []},
+ (ref($target{cflags}) ? @{$target{cflags}} : ($target{cflags} // ())),
+ @{$config{cflags}},
+ @{$config{CFLAGS} // []},
+ );
+ # --ossl-strict-warnings is a Configure-internal placeholder that is
+ # expanded to real warning flags later (after this probe runs). Drop it
+ # here so the probe compiler invocation does not see an unknown option.
+ my $flagstr = join(' ', grep { $_ ne '--ossl-strict-warnings' } @all_cflags);
+
+ open(my $fh, '>', $tf) or die "Cannot write $tf: $!";
+ # The probe exercises all intrinsics used by the vec128 back-end:
+ # vec_perm – used for even/odd lane extraction (ML-KEM, ML-DSA)
+ # vec_mulh i16 – signed 16-bit multiply-high (ML-KEM Montgomery multiply)
+ # vec_mulh i32 – signed 32-bit multiply-high (ML-DSA Montgomery multiply)
+ # vec_min – unsigned 16-bit lane-wise minimum (ML-KEM reduce_once_vec128)
+ # The probe uses plain GCC vector types (no __may_alias__) because the
+ # s390x built-in signatures for vec_perm/vec_mulh/vec_min require their
+ # native types. The production source files handle aliasing via explicit
+ # pointer casts and the separate vec_int16_noalias_t / vec_uint16_t types
+ # defined in ml_kem_vec128.c, so the probe correctly reflects what
+ # the compiler must accept when building those TUs.
+ print $fh <<'END_PROBE';
+#include <vecintrin.h>
+#include <stdint.h>
+typedef __attribute__((vector_size(16))) int16_t vec_i16;
+typedef __attribute__((vector_size(16))) uint16_t vec_u16;
+typedef __attribute__((vector_size(16))) uint8_t vec_u8;
+typedef __attribute__((vector_size(16))) int32_t vec_i32;
+static vec_i16 probe(int16_t a, int16_t b, uint8_t c) {
+ vec_u8 perm = vec_splats(c);
+ vec_i16 va = vec_splats(a);
+ vec_i16 vb = vec_splats(b);
+ vec_i32 va32 = vec_splats((int32_t)a);
+ vec_i32 vb32 = vec_splats((int32_t)b);
+ /* vec_perm: even/odd lane extraction */
+ vec_i16 vperm = (vec_i16)vec_perm(va, vb, perm);
+ /* vec_mulh: signed 16-bit multiply-high (ML-KEM) */
+ vec_i16 vmulh = vec_mulh(va, vb);
+ /* vec_mulh: signed 32-bit multiply-high (ML-DSA) */
+ vec_i32 vmulh32 = vec_mulh(va32, vb32);
+ /* vec_min: unsigned 16-bit lane-wise minimum (ML-KEM) */
+ vec_u16 vmin = vec_min((vec_u16)va, (vec_u16)vb);
+ /* reinterpret-cast vec_i32 → vec_i16 to combine all results */
+ return vperm + vmulh + (vec_i16)vmin + (vec_i16)vmulh32;
+}
+int main(void) { vec_i16 v = probe(1, 2, 0); return (int)v[0]; }
+END_PROBE
+ close($fh);
+
+ # The vec128 TUs are always compiled with -march=z13 -mvx regardless of
+ # the global baseline march. The probe must use the same target flags so
+ # that the result is valid for those TUs. With an older baseline (e.g.
+ # -march=z10) Clang rejects <vecintrin.h> even with -fzvector unless the
+ # target already includes z13/VX.
+ #
+ # First try without -fzvector (GCC works without it).
+ # If that fails, retry with -fzvector (Clang needs it for <vecintrin.h>).
+ # Store the result in s390x_vector_cflags — NOT in the global cflags —
+ # so only the two vec128 TUs see it.
+ # If both attempts fail the compiler lacks the required support; disable
+ # the s390x vec128 back-end so the build can still complete using the
+ # generic implementation.
+ $config{s390x_vector_cflags} = undef; # undef == probe failed / disabled
+ my $vec_target = "-march=z13 -mvx";
+ my $rc = system("$cc $flagstr $vec_target -c -o $obj $tf 2>/dev/null");
+ if ($rc != 0) {
+ $rc = system("$cc $flagstr $vec_target -fzvector -c -o $obj $tf 2>/dev/null");
+ if ($rc == 0) {
+ $config{s390x_vector_cflags} = "-fzvector";
+ }
+ # else: both attempts failed — leave s390x_vector_cflags as undef
+ } else {
+ $config{s390x_vector_cflags} = "";
+ }
+ unlink($tf, $obj);
+}
+
# Deal with bn_ops ###################################################
$config{bn_ll} =0;