Commit 146feef481 for openssl.org

commit 146feef481212287596f0bf9d8a34251354ccec4
Author: Julian Zhu <julian.oerv@isrc.iscas.ac.cn>
Date:   Mon Sep 7 12:34:28 2026 +0800

    RISC-V: handle misaligned input in the SM3 core

    The input is loaded with ld, which is very slow on cores that trap
    misaligned accesses. Build each doubleword from two aligned loads when the
    input is misaligned.

    Signed-off-by: Julian Zhu <julian.oerv@isrc.iscas.ac.cn>
    Reviewed-by: Paul Dale <paul.dale@oracle.com>
    Reviewed-by: Tomas Mraz <tomas@openssl.foundation>
    Merge-date: Fri Oct  2 08:24:17 2026
    Merged-from: https://github.com/openssl/openssl/pull/33007

diff --git a/crypto/sm3/asm/sm3-riscv64-zbb.pl b/crypto/sm3/asm/sm3-riscv64-zbb.pl
index 75ea72f360..83b62ade86 100644
--- a/crypto/sm3/asm/sm3-riscv64-zbb.pl
+++ b/crypto/sm3/asm/sm3-riscv64-zbb.pl
@@ -67,6 +67,9 @@ my ($A, $B, $C, $D ,$E ,$F ,$G ,$H) = ("s2", "s3", "s4", "s5", "s6", "s7", "s8",
 my ($W9, $W10, $W11, $W12, $W13 ,$W14 ,$W15) = ("s0", "s1", "a5", "a6", "a7", "s10", "s11");
 my @W = (undef, undef, undef, undef, undef, undef, undef, undef, undef,
         $W9, $W10, $W11, $W12, $W13, $W14, $W15);
+# Misaligned input only; they reuse round temporaries, so are set up per block
+my ($BASE, $SHL, $SHR) = ($T3, $T4, $T5);
+my ($MISALIGNED_INPUT, $ALIGNED_INPUT) = (0, 1);

 # W[9..15] live in registers, the rest on the stack; Wload/Wstore skip registers
 sub Wreg {
@@ -233,48 +236,64 @@ ___
     return $code;
 }

+# Misaligned: $dst = (lo >> SHL) | (hi << SHR) from two aligned loads
+sub loadDword {
+    my ($ALIGNED, $dst, $off) = @_;
+    if ($ALIGNED) {
+        return "ld $dst, $off($INP)";
+    }
+    my $code=<<___;
+    ld $TMP0, $off($BASE)
+    ld $TMP1, ($off+8)($BASE)
+    srl $dst, $TMP0, $SHL
+    sll $TMP1, $TMP1, $SHR
+    or $dst, $dst, $TMP1
+___
+    return $code;
+}
+
 # One ld plus rev8 yields two message words, the first in the top half
 sub loadMsgRev32 {
+    my ($ALIGNED) = @_;
     my $code=<<___;
-
-    ld $T1, 0($INP)
+    @{[loadDword $ALIGNED, $T1, 0]}
     @{[rev8 $T1, $T1]}
     srli $T2, $T1, 32
     sw $T2, 0($ADDR)
     sw $T1, 4($ADDR)

-    ld $T1, 8($INP)
+    @{[loadDword $ALIGNED, $T1, 8]}
     @{[rev8 $T1, $T1]}
     srli $T2, $T1, 32
     sw $T2, 8($ADDR)
     sw $T1, 12($ADDR)

-    ld $T1, 16($INP)
+    @{[loadDword $ALIGNED, $T1, 16]}
     @{[rev8 $T1, $T1]}
     srli $T2, $T1, 32
     sw $T2, 16($ADDR)
     sw $T1, 20($ADDR)

-    ld $T1, 24($INP)
+    @{[loadDword $ALIGNED, $T1, 24]}
     @{[rev8 $T1, $T1]}
     srli $T2, $T1, 32
     sw $T2, 24($ADDR)
     sw $T1, 28($ADDR)

-    ld $W9, 32($INP)
+    @{[loadDword $ALIGNED, $W9, 32]}
     @{[rev8 $W9, $W9]}
     srli $T2, $W9, 32
     sw $T2, 32($ADDR)

-    ld $W11, 40($INP)
+    @{[loadDword $ALIGNED, $W11, 40]}
     @{[rev8 $W11, $W11]}
     srli $W10, $W11, 32

-    ld $W13, 48($INP)
+    @{[loadDword $ALIGNED, $W13, 48]}
     @{[rev8 $W13, $W13]}
     srli $W12, $W13, 32

-    ld $W15, 56($INP)
+    @{[loadDword $ALIGNED, $W15, 56]}
     @{[rev8 $W15, $W15]}
     srli $W14, $W15, 32
 ___
@@ -322,7 +341,20 @@ L_round_loop:
     # Decrement length by 1
     addi $LEN, $LEN, -1

-    @{[loadMsgRev32]}
+    andi $T1, $INP, 7
+    bnez $T1, L_load_misaligned
+    @{[loadMsgRev32 $ALIGNED_INPUT]}
+    j L_rounds
+
+L_load_misaligned:
+    andi $BASE, $INP, -8
+    andi $SHL, $INP, 7
+    slli $SHL, $SHL, 3
+    li $SHR, 64
+    sub $SHR, $SHR, $SHL
+    @{[loadMsgRev32 $MISALIGNED_INPUT]}
+
+L_rounds:
 ___

 for (my $i = 0; $i < 16; $i += 4) {