[llvm] 4d5bc90 - [SelectionDAG] Improve `CLMUL` lowering for `Promote` types wider than register width (#209265)

via llvm-commits llvm-commits at lists.llvm.org
Mon Jul 13 23:04:57 PDT 2026


Author: Sean Clarke
Date: 2026-07-13T23:04:50-07:00
New Revision: 4d5bc903c8a79fbadb62d8dd1e7cc93ad7a394fd

URL: https://github.com/llvm/llvm-project/commit/4d5bc903c8a79fbadb62d8dd1e7cc93ad7a394fd
DIFF: https://github.com/llvm/llvm-project/commit/4d5bc903c8a79fbadb62d8dd1e7cc93ad7a394fd.diff

LOG: [SelectionDAG] Improve `CLMUL` lowering for `Promote` types wider than register width (#209265)

For `ISD::CLMUL`, add a check to `PromoteIntRes_CLMUL` to take advantage
of the cross-product expansion in `ExpandIntRes_CLMUL` when possible for
`Promote` types which are wider than the register width (e.g. `i96` for
64-bit registers).

Added: 
    

Modified: 
    llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
    llvm/test/CodeGen/AArch64/clmul.ll
    llvm/test/CodeGen/RISCV/clmul.ll
    llvm/test/CodeGen/X86/clmul.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
index 815ef9f74a9c4..2fee0b280ed9b 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
@@ -1791,7 +1791,12 @@ SDValue DAGTypeLegalizer::PromoteIntRes_CLMUL(SDNode *N) {
   EVT VT = TLI.getTypeToTransformTo(*DAG.getContext(), OldVT);
 
   if (Opcode == ISD::CLMUL) {
-    if (!TLI.isOperationLegalOrCustomOrPromote(ISD::CLMUL, VT)) {
+    // Avoid the generic expansion if the cross-product expansion in
+    // ExpandIntRes_CLMUL would produce a better result.
+    if (!TLI.isOperationLegalOrCustomOrPromote(ISD::CLMUL, VT) &&
+        !(getTypeAction(VT) == TargetLowering::TypeExpandInteger &&
+          TLI.isOperationLegalOrCustom(
+              ISD::CLMUL, TLI.getRegisterType(*DAG.getContext(), VT)))) {
       if (SDValue Res = TLI.expandCLMUL(N, DAG))
         return DAG.getNode(ISD::ANY_EXTEND, DL, VT, Res);
     }

diff  --git a/llvm/test/CodeGen/AArch64/clmul.ll b/llvm/test/CodeGen/AArch64/clmul.ll
index bfa4b73b4e677..48cfae3eaef0d 100644
--- a/llvm/test/CodeGen/AArch64/clmul.ll
+++ b/llvm/test/CodeGen/AArch64/clmul.ll
@@ -195,6 +195,705 @@ define i64 @clmul_i64(i64 %x, i64 %y) {
   ret i64 %a
 }
 
+define i96 @clmul_i96(i96 %x, i96 %y) {
+; CHECK-NEON-LABEL: clmul_i96:
+; CHECK-NEON:       // %bb.0:
+; CHECK-NEON-NEXT:    stp x29, x30, [sp, #-96]! // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    stp x28, x27, [sp, #16] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    stp x26, x25, [sp, #32] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    stp x24, x23, [sp, #48] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    stp x22, x21, [sp, #64] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    stp x20, x19, [sp, #80] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    sub sp, sp, #592
+; CHECK-NEON-NEXT:    .cfi_def_cfa_offset 688
+; CHECK-NEON-NEXT:    .cfi_offset w19, -8
+; CHECK-NEON-NEXT:    .cfi_offset w20, -16
+; CHECK-NEON-NEXT:    .cfi_offset w21, -24
+; CHECK-NEON-NEXT:    .cfi_offset w22, -32
+; CHECK-NEON-NEXT:    .cfi_offset w23, -40
+; CHECK-NEON-NEXT:    .cfi_offset w24, -48
+; CHECK-NEON-NEXT:    .cfi_offset w25, -56
+; CHECK-NEON-NEXT:    .cfi_offset w26, -64
+; CHECK-NEON-NEXT:    .cfi_offset w27, -72
+; CHECK-NEON-NEXT:    .cfi_offset w28, -80
+; CHECK-NEON-NEXT:    .cfi_offset w30, -88
+; CHECK-NEON-NEXT:    .cfi_offset w29, -96
+; CHECK-NEON-NEXT:    sbfx x9, x2, #1, #1
+; CHECK-NEON-NEXT:    lsl x8, x0, #1
+; CHECK-NEON-NEXT:    sbfx x16, x2, #0, #1
+; CHECK-NEON-NEXT:    lsl x10, x0, #2
+; CHECK-NEON-NEXT:    sbfx x13, x2, #2, #1
+; CHECK-NEON-NEXT:    lsl x11, x0, #3
+; CHECK-NEON-NEXT:    sbfx x14, x2, #3, #1
+; CHECK-NEON-NEXT:    lsl x12, x0, #4
+; CHECK-NEON-NEXT:    sbfx x15, x2, #4, #1
+; CHECK-NEON-NEXT:    str x9, [sp, #200] // 8-byte Spill
+; CHECK-NEON-NEXT:    sbfx x17, x2, #8, #1
+; CHECK-NEON-NEXT:    sbfx x18, x2, #12, #1
+; CHECK-NEON-NEXT:    str x8, [sp, #512] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x8, x9, x8
+; CHECK-NEON-NEXT:    and x9, x16, x0
+; CHECK-NEON-NEXT:    str x10, [sp, #568] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x10, x13, x10
+; CHECK-NEON-NEXT:    eor x8, x9, x8
+; CHECK-NEON-NEXT:    str x11, [sp, #552] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x11, x14, x11
+; CHECK-NEON-NEXT:    sbfx x4, x2, #10, #1
+; CHECK-NEON-NEXT:    stp x14, x13, [sp, #272] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    eor x9, x10, x11
+; CHECK-NEON-NEXT:    and x10, x15, x12
+; CHECK-NEON-NEXT:    str x12, [sp, #536] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x11, x0, #5
+; CHECK-NEON-NEXT:    sbfx x14, x2, #5, #1
+; CHECK-NEON-NEXT:    lsl x12, x0, #6
+; CHECK-NEON-NEXT:    str x16, [sp, #152] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x13, x0, #7
+; CHECK-NEON-NEXT:    str x15, [sp, #288] // 8-byte Spill
+; CHECK-NEON-NEXT:    sbfx x15, x2, #6, #1
+; CHECK-NEON-NEXT:    sbfx x16, x2, #7, #1
+; CHECK-NEON-NEXT:    str x14, [sp, #192] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x8, x8, x9
+; CHECK-NEON-NEXT:    sbfx x20, x2, #20, #1
+; CHECK-NEON-NEXT:    stp x11, x12, [sp, #480] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    and x11, x14, x11
+; CHECK-NEON-NEXT:    lsl x14, x0, #8
+; CHECK-NEON-NEXT:    str x13, [sp, #520] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    and x11, x15, x12
+; CHECK-NEON-NEXT:    stp x16, x15, [sp, #240] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    and x12, x16, x13
+; CHECK-NEON-NEXT:    and x13, x17, x14
+; CHECK-NEON-NEXT:    str x14, [sp, #544] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x14, x0, #9
+; CHECK-NEON-NEXT:    sbfx x15, x2, #9, #1
+; CHECK-NEON-NEXT:    str x17, [sp, #232] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x9, x10, x11
+; CHECK-NEON-NEXT:    eor x10, x12, x13
+; CHECK-NEON-NEXT:    str x14, [sp, #528] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x11, x15, x14
+; CHECK-NEON-NEXT:    lsl x12, x0, #11
+; CHECK-NEON-NEXT:    sbfx x17, x2, #11, #1
+; CHECK-NEON-NEXT:    lsl x14, x0, #12
+; CHECK-NEON-NEXT:    str x15, [sp, #256] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x15, x0, #10
+; CHECK-NEON-NEXT:    lsl x13, x0, #13
+; CHECK-NEON-NEXT:    sbfx x16, x2, #13, #1
+; CHECK-NEON-NEXT:    str x12, [sp, #464] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    and x11, x17, x12
+; CHECK-NEON-NEXT:    and x12, x18, x14
+; CHECK-NEON-NEXT:    str x13, [sp, #576] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x13, x16, x13
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    and x12, x4, x15
+; CHECK-NEON-NEXT:    str x14, [sp, #496] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x8, x8, x9
+; CHECK-NEON-NEXT:    eor x9, x10, x12
+; CHECK-NEON-NEXT:    eor x10, x11, x13
+; CHECK-NEON-NEXT:    lsl x11, x0, #14
+; CHECK-NEON-NEXT:    sbfx x14, x2, #14, #1
+; CHECK-NEON-NEXT:    lsl x12, x0, #16
+; CHECK-NEON-NEXT:    str x15, [sp, #416] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x13, x0, #15
+; CHECK-NEON-NEXT:    sbfx x15, x2, #16, #1
+; CHECK-NEON-NEXT:    stp x18, x17, [sp, #168] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    sbfx x17, x2, #15, #1
+; CHECK-NEON-NEXT:    eor x8, x8, x9
+; CHECK-NEON-NEXT:    str x16, [sp, #216] // 8-byte Spill
+; CHECK-NEON-NEXT:    sbfx x16, x2, #17, #1
+; CHECK-NEON-NEXT:    sbfx x24, x2, #26, #1
+; CHECK-NEON-NEXT:    str x14, [sp, #136] // 8-byte Spill
+; CHECK-NEON-NEXT:    sbfx x7, x2, #30, #1
+; CHECK-NEON-NEXT:    cmn w2, #1
+; CHECK-NEON-NEXT:    stp x12, x11, [sp, #448] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    and x11, x14, x11
+; CHECK-NEON-NEXT:    lsl x14, x0, #17
+; CHECK-NEON-NEXT:    str x13, [sp, #424] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    and x11, x17, x13
+; CHECK-NEON-NEXT:    str x15, [sp, #224] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x12, x15, x12
+; CHECK-NEON-NEXT:    and x13, x16, x14
+; CHECK-NEON-NEXT:    str x14, [sp, #504] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x14, x0, #18
+; CHECK-NEON-NEXT:    sbfx x15, x2, #18, #1
+; CHECK-NEON-NEXT:    eor x9, x10, x11
+; CHECK-NEON-NEXT:    eor x10, x12, x13
+; CHECK-NEON-NEXT:    lsl x12, x0, #19
+; CHECK-NEON-NEXT:    str x14, [sp, #472] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x11, x15, x14
+; CHECK-NEON-NEXT:    sbfx x14, x2, #19, #1
+; CHECK-NEON-NEXT:    str x17, [sp, #88] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    lsl x13, x0, #22
+; CHECK-NEON-NEXT:    str x16, [sp, #208] // 8-byte Spill
+; CHECK-NEON-NEXT:    sbfx x16, x2, #22, #1
+; CHECK-NEON-NEXT:    and x11, x14, x12
+; CHECK-NEON-NEXT:    str x14, [sp, #48] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x14, x0, #23
+; CHECK-NEON-NEXT:    sbfx x17, x2, #23, #1
+; CHECK-NEON-NEXT:    str x15, [sp, #184] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x15, x0, #20
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    str x12, [sp, #392] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x12, x16, x13
+; CHECK-NEON-NEXT:    eor x8, x8, x9
+; CHECK-NEON-NEXT:    stp x13, x14, [sp, #432] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    and x13, x17, x14
+; CHECK-NEON-NEXT:    lsl x14, x0, #24
+; CHECK-NEON-NEXT:    stp x17, x16, [sp, #120] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    sbfx x16, x2, #24, #1
+; CHECK-NEON-NEXT:    eor x11, x12, x13
+; CHECK-NEON-NEXT:    str x15, [sp, #352] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x12, x20, x15
+; CHECK-NEON-NEXT:    sbfx x15, x2, #25, #1
+; CHECK-NEON-NEXT:    str x14, [sp, #584] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x13, x16, x14
+; CHECK-NEON-NEXT:    lsl x14, x0, #25
+; CHECK-NEON-NEXT:    str x16, [sp, #144] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x9, x10, x12
+; CHECK-NEON-NEXT:    eor x10, x11, x13
+; CHECK-NEON-NEXT:    lsl x13, x0, #21
+; CHECK-NEON-NEXT:    sbfx x16, x2, #21, #1
+; CHECK-NEON-NEXT:    and x11, x15, x14
+; CHECK-NEON-NEXT:    lsl x12, x0, #26
+; CHECK-NEON-NEXT:    str x14, [sp, #560] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    str x16, [sp, #40] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x11, x16, x13
+; CHECK-NEON-NEXT:    lsl x14, x0, #27
+; CHECK-NEON-NEXT:    sbfx x16, x2, #27, #1
+; CHECK-NEON-NEXT:    str x15, [sp, #104] // 8-byte Spill
+; CHECK-NEON-NEXT:    sbfx x15, x2, #28, #1
+; CHECK-NEON-NEXT:    str x13, [sp, #344] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x13, x0, #28
+; CHECK-NEON-NEXT:    eor x9, x9, x11
+; CHECK-NEON-NEXT:    str x12, [sp, #408] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x12, x24, x12
+; CHECK-NEON-NEXT:    and x11, x16, x14
+; CHECK-NEON-NEXT:    eor x10, x10, x12
+; CHECK-NEON-NEXT:    str x14, [sp, #384] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x14, x0, #30
+; CHECK-NEON-NEXT:    str x13, [sp, #400] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    and x11, x15, x13
+; CHECK-NEON-NEXT:    str x15, [sp, #264] // 8-byte Spill
+; CHECK-NEON-NEXT:    lsl x13, x0, #29
+; CHECK-NEON-NEXT:    sbfx x15, x2, #29, #1
+; CHECK-NEON-NEXT:    eor x8, x8, x9
+; CHECK-NEON-NEXT:    eor x9, x10, x11
+; CHECK-NEON-NEXT:    lsl x12, x0, #31
+; CHECK-NEON-NEXT:    stp x13, x14, [sp, #368] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    and x10, x15, x13
+; CHECK-NEON-NEXT:    sbfx x13, x2, #32, #1
+; CHECK-NEON-NEXT:    and x11, x7, x14
+; CHECK-NEON-NEXT:    sbfx x21, x2, #33, #1
+; CHECK-NEON-NEXT:    str x12, [sp, #360] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    csel x11, xzr, x12, gt
+; CHECK-NEON-NEXT:    and x12, x13, x0, lsl #32
+; CHECK-NEON-NEXT:    sbfx x14, x2, #34, #1
+; CHECK-NEON-NEXT:    str x13, [sp, #32] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    and x11, x21, x0, lsl #33
+; CHECK-NEON-NEXT:    sbfx x13, x2, #35, #1
+; CHECK-NEON-NEXT:    eor x8, x8, x9
+; CHECK-NEON-NEXT:    eor x9, x10, x12
+; CHECK-NEON-NEXT:    and x10, x14, x0, lsl #34
+; CHECK-NEON-NEXT:    sbfx x12, x2, #36, #1
+; CHECK-NEON-NEXT:    eor x9, x9, x11
+; CHECK-NEON-NEXT:    and x11, x13, x0, lsl #35
+; CHECK-NEON-NEXT:    sbfx x25, x2, #37, #1
+; CHECK-NEON-NEXT:    sbfx x22, x2, #38, #1
+; CHECK-NEON-NEXT:    eor x9, x9, x10
+; CHECK-NEON-NEXT:    and x10, x12, x0, lsl #36
+; CHECK-NEON-NEXT:    sbfx x19, x2, #39, #1
+; CHECK-NEON-NEXT:    str x12, [sp, #312] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x9, x9, x11
+; CHECK-NEON-NEXT:    and x11, x25, x0, lsl #37
+; CHECK-NEON-NEXT:    and x12, x22, x0, lsl #38
+; CHECK-NEON-NEXT:    sbfx x5, x2, #40, #1
+; CHECK-NEON-NEXT:    eor x9, x9, x10
+; CHECK-NEON-NEXT:    and x10, x19, x0, lsl #39
+; CHECK-NEON-NEXT:    sbfx x26, x2, #41, #1
+; CHECK-NEON-NEXT:    str x13, [sp, #296] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    and x12, x5, x0, lsl #40
+; CHECK-NEON-NEXT:    sbfx x13, x2, #42, #1
+; CHECK-NEON-NEXT:    eor x10, x11, x10
+; CHECK-NEON-NEXT:    and x11, x26, x0, lsl #41
+; CHECK-NEON-NEXT:    sbfx x27, x2, #43, #1
+; CHECK-NEON-NEXT:    eor x8, x8, x9
+; CHECK-NEON-NEXT:    eor x9, x10, x12
+; CHECK-NEON-NEXT:    str x13, [sp, #24] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x10, x13, x0, lsl #42
+; CHECK-NEON-NEXT:    sbfx x13, x2, #44, #1
+; CHECK-NEON-NEXT:    stp x14, x15, [sp, #8] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    eor x9, x9, x11
+; CHECK-NEON-NEXT:    and x11, x27, x0, lsl #43
+; CHECK-NEON-NEXT:    sbfx x12, x2, #45, #1
+; CHECK-NEON-NEXT:    sbfx x14, x2, #46, #1
+; CHECK-NEON-NEXT:    eor x9, x9, x10
+; CHECK-NEON-NEXT:    and x10, x13, x0, lsl #44
+; CHECK-NEON-NEXT:    sbfx x28, x2, #47, #1
+; CHECK-NEON-NEXT:    eor x9, x9, x11
+; CHECK-NEON-NEXT:    str x12, [sp, #304] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x11, x12, x0, lsl #45
+; CHECK-NEON-NEXT:    and x12, x14, x0, lsl #46
+; CHECK-NEON-NEXT:    sbfx x6, x2, #48, #1
+; CHECK-NEON-NEXT:    eor x9, x9, x10
+; CHECK-NEON-NEXT:    and x10, x28, x0, lsl #47
+; CHECK-NEON-NEXT:    sbfx x29, x2, #49, #1
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    str x13, [sp, #96] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x12, x6, x0, lsl #48
+; CHECK-NEON-NEXT:    eor x13, x8, x9
+; CHECK-NEON-NEXT:    eor x9, x11, x10
+; CHECK-NEON-NEXT:    and x10, x29, x0, lsl #49
+; CHECK-NEON-NEXT:    str x14, [sp, #72] // 8-byte Spill
+; CHECK-NEON-NEXT:    sbfx x14, x2, #50, #1
+; CHECK-NEON-NEXT:    eor x9, x9, x12
+; CHECK-NEON-NEXT:    sbfx x8, x2, #55, #1
+; CHECK-NEON-NEXT:    sbfx x30, x2, #51, #1
+; CHECK-NEON-NEXT:    eor x9, x9, x10
+; CHECK-NEON-NEXT:    sbfx x10, x2, #54, #1
+; CHECK-NEON-NEXT:    and x11, x14, x0, lsl #50
+; CHECK-NEON-NEXT:    str x16, [sp, #160] // 8-byte Spill
+; CHECK-NEON-NEXT:    sbfx x16, x2, #52, #1
+; CHECK-NEON-NEXT:    and x18, x8, x0, lsl #55
+; CHECK-NEON-NEXT:    stp x10, x8, [sp, #320] // 16-byte Folded Spill
+; CHECK-NEON-NEXT:    sbfx x8, x2, #56, #1
+; CHECK-NEON-NEXT:    eor x9, x9, x11
+; CHECK-NEON-NEXT:    and x11, x30, x0, lsl #51
+; CHECK-NEON-NEXT:    and x12, x10, x0, lsl #54
+; CHECK-NEON-NEXT:    sbfx x17, x2, #53, #1
+; CHECK-NEON-NEXT:    sbfx x10, x2, #57, #1
+; CHECK-NEON-NEXT:    str x4, [sp, #112] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x4, x16, x0, lsl #52
+; CHECK-NEON-NEXT:    str x8, [sp, #80] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x8, x8, x0, lsl #56
+; CHECK-NEON-NEXT:    eor x11, x9, x11
+; CHECK-NEON-NEXT:    str x10, [sp, #56] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x12, x12, x18
+; CHECK-NEON-NEXT:    and x9, x17, x0, lsl #53
+; CHECK-NEON-NEXT:    and x10, x10, x0, lsl #57
+; CHECK-NEON-NEXT:    sbfx x23, x2, #58, #1
+; CHECK-NEON-NEXT:    eor x11, x11, x4
+; CHECK-NEON-NEXT:    eor x8, x12, x8
+; CHECK-NEON-NEXT:    eor x9, x11, x9
+; CHECK-NEON-NEXT:    sbfx x4, x2, #59, #1
+; CHECK-NEON-NEXT:    eor x12, x8, x10
+; CHECK-NEON-NEXT:    and x10, x23, x0, lsl #58
+; CHECK-NEON-NEXT:    eor x9, x13, x9
+; CHECK-NEON-NEXT:    sbfx x18, x2, #60, #1
+; CHECK-NEON-NEXT:    extr x8, x1, x0, #63
+; CHECK-NEON-NEXT:    str x9, [sp, #336] // 8-byte Spill
+; CHECK-NEON-NEXT:    eor x9, x12, x10
+; CHECK-NEON-NEXT:    ldr x10, [sp, #200] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x11, x4, x0, lsl #59
+; CHECK-NEON-NEXT:    extr x12, x1, x0, #62
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #61
+; CHECK-NEON-NEXT:    str x14, [sp, #64] // 8-byte Spill
+; CHECK-NEON-NEXT:    and x10, x10, x8
+; CHECK-NEON-NEXT:    and x8, x18, x0, lsl #60
+; CHECK-NEON-NEXT:    eor x9, x9, x11
+; CHECK-NEON-NEXT:    ldr x11, [sp, #152] // 8-byte Reload
+; CHECK-NEON-NEXT:    cmn w3, #1
+; CHECK-NEON-NEXT:    eor x15, x9, x8
+; CHECK-NEON-NEXT:    ldp x8, x9, [sp, #272] // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    and x11, x11, x1
+; CHECK-NEON-NEXT:    eor x14, x11, x10
+; CHECK-NEON-NEXT:    extr x10, x1, x0, #60
+; CHECK-NEON-NEXT:    and x11, x9, x12
+; CHECK-NEON-NEXT:    and x12, x8, x13
+; CHECK-NEON-NEXT:    ldr x8, [sp, #288] // 8-byte Reload
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #59
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    extr x9, x1, x0, #57
+; CHECK-NEON-NEXT:    and x10, x8, x10
+; CHECK-NEON-NEXT:    ldr x8, [sp, #192] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x11, x14, x11
+; CHECK-NEON-NEXT:    extr x12, x1, x0, #58
+; CHECK-NEON-NEXT:    and x13, x8, x13
+; CHECK-NEON-NEXT:    extr x8, x1, x0, #56
+; CHECK-NEON-NEXT:    eor x10, x10, x13
+; CHECK-NEON-NEXT:    ldp x13, x14, [sp, #240] // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    and x9, x13, x9
+; CHECK-NEON-NEXT:    ldr x13, [sp, #232] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x12, x14, x12
+; CHECK-NEON-NEXT:    eor x10, x10, x12
+; CHECK-NEON-NEXT:    ldr x12, [sp, #256] // 8-byte Reload
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #52
+; CHECK-NEON-NEXT:    and x8, x13, x8
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #55
+; CHECK-NEON-NEXT:    eor x10, x11, x10
+; CHECK-NEON-NEXT:    eor x8, x9, x8
+; CHECK-NEON-NEXT:    extr x9, x1, x0, #54
+; CHECK-NEON-NEXT:    ldr x11, [sp, #112] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x12, x12, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #53
+; CHECK-NEON-NEXT:    eor x8, x8, x12
+; CHECK-NEON-NEXT:    and x9, x11, x9
+; CHECK-NEON-NEXT:    ldp x12, x11, [sp, #168] // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    eor x8, x8, x9
+; CHECK-NEON-NEXT:    ldr x9, [sp, #216] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x8, x10, x8
+; CHECK-NEON-NEXT:    and x11, x11, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #51
+; CHECK-NEON-NEXT:    and x12, x12, x14
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    extr x12, x1, x0, #50
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #49
+; CHECK-NEON-NEXT:    and x13, x9, x13
+; CHECK-NEON-NEXT:    sbfx x9, x2, #61, #1
+; CHECK-NEON-NEXT:    eor x10, x11, x13
+; CHECK-NEON-NEXT:    ldr x11, [sp, #136] // 8-byte Reload
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #47
+; CHECK-NEON-NEXT:    and x11, x11, x12
+; CHECK-NEON-NEXT:    and x12, x9, x0, lsl #61
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    ldr x11, [sp, #88] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x12, x15, x12
+; CHECK-NEON-NEXT:    extr x15, x1, x0, #39
+; CHECK-NEON-NEXT:    and x11, x11, x14
+; CHECK-NEON-NEXT:    str x12, [sp, #288] // 8-byte Spill
+; CHECK-NEON-NEXT:    ldr x12, [sp, #224] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    extr x11, x1, x0, #48
+; CHECK-NEON-NEXT:    eor x8, x8, x10
+; CHECK-NEON-NEXT:    extr x10, x1, x0, #46
+; CHECK-NEON-NEXT:    and x11, x12, x11
+; CHECK-NEON-NEXT:    ldr x12, [sp, #208] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x12, x12, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #45
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    ldr x12, [sp, #184] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x10, x12, x10
+; CHECK-NEON-NEXT:    extr x12, x1, x0, #44
+; CHECK-NEON-NEXT:    eor x10, x11, x10
+; CHECK-NEON-NEXT:    ldr x11, [sp, #48] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x11, x11, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #42
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    and x11, x20, x12
+; CHECK-NEON-NEXT:    extr x12, x1, x0, #41
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    ldp x14, x11, [sp, #120] // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    ldr x20, [sp, #40] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x11, x11, x13
+; CHECK-NEON-NEXT:    and x12, x14, x12
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #40
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    ldr x12, [sp, #144] // 8-byte Reload
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #43
+; CHECK-NEON-NEXT:    and x12, x12, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #38
+; CHECK-NEON-NEXT:    and x14, x20, x14
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    ldr x12, [sp, #104] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x10, x10, x14
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #35
+; CHECK-NEON-NEXT:    lsl x20, x2, #32
+; CHECK-NEON-NEXT:    eor x8, x8, x10
+; CHECK-NEON-NEXT:    and x12, x12, x15
+; CHECK-NEON-NEXT:    extr x15, x1, x0, #34
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    and x12, x24, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #37
+; CHECK-NEON-NEXT:    ldr x24, [sp, #160] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    extr x12, x1, x0, #33
+; CHECK-NEON-NEXT:    and x15, x7, x15
+; CHECK-NEON-NEXT:    extr x7, x1, x0, #36
+; CHECK-NEON-NEXT:    and x13, x24, x13
+; CHECK-NEON-NEXT:    ldr x24, [sp, #16] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x12, x12, x20, asr #63
+; CHECK-NEON-NEXT:    eor x11, x11, x13
+; CHECK-NEON-NEXT:    extr x20, x1, x0, #32
+; CHECK-NEON-NEXT:    and x14, x24, x14
+; CHECK-NEON-NEXT:    eor x13, x14, x15
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #31
+; CHECK-NEON-NEXT:    ldr x15, [sp, #264] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x12, x13, x12
+; CHECK-NEON-NEXT:    ldr x13, [sp, #32] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x15, x15, x7
+; CHECK-NEON-NEXT:    and x13, x13, x20
+; CHECK-NEON-NEXT:    eor x10, x11, x15
+; CHECK-NEON-NEXT:    ldr x15, [sp, #296] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x12, x12, x13
+; CHECK-NEON-NEXT:    and x13, x21, x14
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #30
+; CHECK-NEON-NEXT:    eor x11, x12, x13
+; CHECK-NEON-NEXT:    ldr x12, [sp, #8] // 8-byte Reload
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #27
+; CHECK-NEON-NEXT:    eor x8, x8, x10
+; CHECK-NEON-NEXT:    and x12, x12, x14
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #26
+; CHECK-NEON-NEXT:    eor x10, x11, x12
+; CHECK-NEON-NEXT:    extr x11, x1, x0, #25
+; CHECK-NEON-NEXT:    and x12, x25, x13
+; CHECK-NEON-NEXT:    and x13, x22, x14
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #24
+; CHECK-NEON-NEXT:    eor x12, x12, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #29
+; CHECK-NEON-NEXT:    and x11, x19, x11
+; CHECK-NEON-NEXT:    eor x11, x12, x11
+; CHECK-NEON-NEXT:    and x12, x5, x14
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #23
+; CHECK-NEON-NEXT:    and x13, x15, x13
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    extr x12, x1, x0, #22
+; CHECK-NEON-NEXT:    eor x10, x10, x13
+; CHECK-NEON-NEXT:    ldr x13, [sp, #24] // 8-byte Reload
+; CHECK-NEON-NEXT:    extr x15, x1, x0, #28
+; CHECK-NEON-NEXT:    and x14, x26, x14
+; CHECK-NEON-NEXT:    extr x5, x1, x0, #21
+; CHECK-NEON-NEXT:    and x12, x13, x12
+; CHECK-NEON-NEXT:    ldr x13, [sp, #312] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x11, x11, x14
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #20
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    and x12, x27, x5
+; CHECK-NEON-NEXT:    and x13, x13, x15
+; CHECK-NEON-NEXT:    extr x15, x1, x0, #18
+; CHECK-NEON-NEXT:    extr x5, x1, x0, #17
+; CHECK-NEON-NEXT:    eor x10, x10, x13
+; CHECK-NEON-NEXT:    ldr x13, [sp, #96] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    extr x12, x1, x0, #16
+; CHECK-NEON-NEXT:    eor x8, x8, x10
+; CHECK-NEON-NEXT:    and x13, x13, x14
+; CHECK-NEON-NEXT:    ldr x14, [sp, #72] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x12, x6, x12
+; CHECK-NEON-NEXT:    eor x11, x11, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #14
+; CHECK-NEON-NEXT:    and x14, x14, x15
+; CHECK-NEON-NEXT:    and x15, x28, x5
+; CHECK-NEON-NEXT:    extr x5, x1, x0, #13
+; CHECK-NEON-NEXT:    eor x14, x14, x15
+; CHECK-NEON-NEXT:    extr x15, x1, x0, #15
+; CHECK-NEON-NEXT:    ldr x6, [sp, #304] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x12, x14, x12
+; CHECK-NEON-NEXT:    and x14, x29, x15
+; CHECK-NEON-NEXT:    extr x15, x1, x0, #19
+; CHECK-NEON-NEXT:    eor x12, x12, x14
+; CHECK-NEON-NEXT:    ldr x14, [sp, #64] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x15, x6, x15
+; CHECK-NEON-NEXT:    and x13, x14, x13
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #12
+; CHECK-NEON-NEXT:    eor x10, x11, x15
+; CHECK-NEON-NEXT:    eor x12, x12, x13
+; CHECK-NEON-NEXT:    and x13, x30, x5
+; CHECK-NEON-NEXT:    eor x10, x8, x10
+; CHECK-NEON-NEXT:    eor x12, x12, x13
+; CHECK-NEON-NEXT:    and x13, x16, x14
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #11
+; CHECK-NEON-NEXT:    eor x11, x12, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #8
+; CHECK-NEON-NEXT:    ldr x15, [sp, #320] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x12, x17, x14
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #7
+; CHECK-NEON-NEXT:    ldr x16, [sp, #288] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x8, x11, x12
+; CHECK-NEON-NEXT:    ldr x12, [sp, #80] // 8-byte Reload
+; CHECK-NEON-NEXT:    extr x11, x1, x0, #6
+; CHECK-NEON-NEXT:    ldr x17, [sp, #568] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x12, x12, x13
+; CHECK-NEON-NEXT:    ldr x13, [sp, #56] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x11, x23, x11
+; CHECK-NEON-NEXT:    and x13, x13, x14
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #5
+; CHECK-NEON-NEXT:    eor x12, x12, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #4
+; CHECK-NEON-NEXT:    eor x11, x12, x11
+; CHECK-NEON-NEXT:    and x12, x4, x14
+; CHECK-NEON-NEXT:    extr x14, x1, x0, #10
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    and x12, x18, x13
+; CHECK-NEON-NEXT:    extr x13, x1, x0, #3
+; CHECK-NEON-NEXT:    and x14, x15, x14
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    extr x12, x1, x0, #9
+; CHECK-NEON-NEXT:    and x9, x9, x13
+; CHECK-NEON-NEXT:    sbfx x13, x2, #62, #1
+; CHECK-NEON-NEXT:    extr x15, x1, x0, #2
+; CHECK-NEON-NEXT:    eor x8, x8, x14
+; CHECK-NEON-NEXT:    ldr x14, [sp, #328] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x9, x11, x9
+; CHECK-NEON-NEXT:    asr x11, x2, #63
+; CHECK-NEON-NEXT:    and x12, x14, x12
+; CHECK-NEON-NEXT:    and x14, x13, x15
+; CHECK-NEON-NEXT:    extr x15, x1, x0, #1
+; CHECK-NEON-NEXT:    eor x9, x9, x14
+; CHECK-NEON-NEXT:    sbfx x14, x3, #0, #1
+; CHECK-NEON-NEXT:    eor x12, x8, x12
+; CHECK-NEON-NEXT:    and x8, x13, x0, lsl #62
+; CHECK-NEON-NEXT:    and x13, x11, x15
+; CHECK-NEON-NEXT:    lsl x15, x3, #62
+; CHECK-NEON-NEXT:    eor x10, x10, x12
+; CHECK-NEON-NEXT:    eor x12, x9, x13
+; CHECK-NEON-NEXT:    and x9, x11, x0, lsl #63
+; CHECK-NEON-NEXT:    and x11, x14, x0
+; CHECK-NEON-NEXT:    ldr x13, [sp, #512] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x8, x16, x8
+; CHECK-NEON-NEXT:    eor x11, x12, x11
+; CHECK-NEON-NEXT:    lsl x12, x3, #60
+; CHECK-NEON-NEXT:    ldr x16, [sp, #552] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x13, x13, x15, asr #63
+; CHECK-NEON-NEXT:    lsl x14, x3, #59
+; CHECK-NEON-NEXT:    lsl x15, x3, #58
+; CHECK-NEON-NEXT:    and x12, x16, x12, asr #63
+; CHECK-NEON-NEXT:    ldr x16, [sp, #536] // 8-byte Reload
+; CHECK-NEON-NEXT:    ldr x1, [sp, #584] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x11, x11, x13
+; CHECK-NEON-NEXT:    lsl x13, x3, #61
+; CHECK-NEON-NEXT:    lsl x0, x3, #33
+; CHECK-NEON-NEXT:    and x14, x16, x14, asr #63
+; CHECK-NEON-NEXT:    lsl x16, x3, #57
+; CHECK-NEON-NEXT:    eor x8, x8, x9
+; CHECK-NEON-NEXT:    and x13, x17, x13, asr #63
+; CHECK-NEON-NEXT:    ldr x17, [sp, #480] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x12, x12, x14
+; CHECK-NEON-NEXT:    ldr x14, [sp, #488] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x15, x17, x15, asr #63
+; CHECK-NEON-NEXT:    lsl x17, x3, #56
+; CHECK-NEON-NEXT:    eor x11, x11, x13
+; CHECK-NEON-NEXT:    and x14, x14, x16, asr #63
+; CHECK-NEON-NEXT:    lsl x16, x3, #55
+; CHECK-NEON-NEXT:    ldr x13, [sp, #464] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x12, x12, x15
+; CHECK-NEON-NEXT:    ldr x15, [sp, #520] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    eor x12, x12, x14
+; CHECK-NEON-NEXT:    ldr x14, [sp, #544] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x15, x15, x17, asr #63
+; CHECK-NEON-NEXT:    lsl x17, x3, #54
+; CHECK-NEON-NEXT:    and x14, x14, x16, asr #63
+; CHECK-NEON-NEXT:    lsl x16, x3, #53
+; CHECK-NEON-NEXT:    eor x12, x12, x15
+; CHECK-NEON-NEXT:    ldr x15, [sp, #528] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x12, x12, x14
+; CHECK-NEON-NEXT:    ldr x14, [sp, #416] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x15, x15, x17, asr #63
+; CHECK-NEON-NEXT:    lsl x17, x3, #52
+; CHECK-NEON-NEXT:    and x14, x14, x16, asr #63
+; CHECK-NEON-NEXT:    lsl x16, x3, #51
+; CHECK-NEON-NEXT:    eor x12, x12, x15
+; CHECK-NEON-NEXT:    and x13, x13, x17, asr #63
+; CHECK-NEON-NEXT:    ldr x17, [sp, #576] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x11, x12, x14
+; CHECK-NEON-NEXT:    ldr x12, [sp, #496] // 8-byte Reload
+; CHECK-NEON-NEXT:    lsl x14, x3, #48
+; CHECK-NEON-NEXT:    eor x11, x11, x13
+; CHECK-NEON-NEXT:    lsl x13, x3, #49
+; CHECK-NEON-NEXT:    lsl x15, x3, #47
+; CHECK-NEON-NEXT:    and x12, x12, x16, asr #63
+; CHECK-NEON-NEXT:    ldr x16, [sp, #456] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x13, x16, x13, asr #63
+; CHECK-NEON-NEXT:    ldr x16, [sp, #424] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    lsl x12, x3, #50
+; CHECK-NEON-NEXT:    and x14, x16, x14, asr #63
+; CHECK-NEON-NEXT:    lsl x16, x3, #46
+; CHECK-NEON-NEXT:    and x12, x17, x12, asr #63
+; CHECK-NEON-NEXT:    ldr x17, [sp, #448] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x13, x13, x14
+; CHECK-NEON-NEXT:    ldr x14, [sp, #504] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x15, x17, x15, asr #63
+; CHECK-NEON-NEXT:    lsl x17, x3, #45
+; CHECK-NEON-NEXT:    eor x11, x11, x12
+; CHECK-NEON-NEXT:    and x14, x14, x16, asr #63
+; CHECK-NEON-NEXT:    lsl x16, x3, #44
+; CHECK-NEON-NEXT:    eor x10, x10, x11
+; CHECK-NEON-NEXT:    eor x13, x13, x15
+; CHECK-NEON-NEXT:    ldr x15, [sp, #472] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x13, x13, x14
+; CHECK-NEON-NEXT:    ldr x14, [sp, #392] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x15, x15, x17, asr #63
+; CHECK-NEON-NEXT:    lsl x17, x3, #43
+; CHECK-NEON-NEXT:    and x14, x14, x16, asr #63
+; CHECK-NEON-NEXT:    lsl x16, x3, #42
+; CHECK-NEON-NEXT:    eor x13, x13, x15
+; CHECK-NEON-NEXT:    ldr x15, [sp, #352] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x13, x13, x14
+; CHECK-NEON-NEXT:    ldr x14, [sp, #344] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x15, x15, x17, asr #63
+; CHECK-NEON-NEXT:    lsl x17, x3, #41
+; CHECK-NEON-NEXT:    and x14, x14, x16, asr #63
+; CHECK-NEON-NEXT:    lsl x16, x3, #40
+; CHECK-NEON-NEXT:    eor x12, x13, x15
+; CHECK-NEON-NEXT:    ldr x13, [sp, #432] // 8-byte Reload
+; CHECK-NEON-NEXT:    lsl x15, x3, #37
+; CHECK-NEON-NEXT:    eor x12, x12, x14
+; CHECK-NEON-NEXT:    ldr x14, [sp, #440] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x13, x13, x17, asr #63
+; CHECK-NEON-NEXT:    ldr x17, [sp, #384] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x14, x14, x16, asr #63
+; CHECK-NEON-NEXT:    lsl x16, x3, #39
+; CHECK-NEON-NEXT:    eor x12, x12, x13
+; CHECK-NEON-NEXT:    lsl x13, x3, #36
+; CHECK-NEON-NEXT:    eor x12, x12, x14
+; CHECK-NEON-NEXT:    ldp x18, x14, [sp, #400] // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    and x13, x17, x13, asr #63
+; CHECK-NEON-NEXT:    lsl x17, x3, #34
+; CHECK-NEON-NEXT:    and x16, x1, x16, asr #63
+; CHECK-NEON-NEXT:    and x14, x14, x15, asr #63
+; CHECK-NEON-NEXT:    lsl x15, x3, #35
+; CHECK-NEON-NEXT:    eor x11, x12, x16
+; CHECK-NEON-NEXT:    eor x13, x14, x13
+; CHECK-NEON-NEXT:    ldr x14, [sp, #368] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x15, x18, x15, asr #63
+; CHECK-NEON-NEXT:    lsl x18, x3, #38
+; CHECK-NEON-NEXT:    and x14, x14, x17, asr #63
+; CHECK-NEON-NEXT:    ldr x17, [sp, #560] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x13, x13, x15
+; CHECK-NEON-NEXT:    ldr x15, [sp, #376] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x17, x17, x18, asr #63
+; CHECK-NEON-NEXT:    eor x12, x13, x14
+; CHECK-NEON-NEXT:    ldr x13, [sp, #360] // 8-byte Reload
+; CHECK-NEON-NEXT:    and x15, x15, x0, asr #63
+; CHECK-NEON-NEXT:    eor x11, x11, x17
+; CHECK-NEON-NEXT:    csel x13, xzr, x13, gt
+; CHECK-NEON-NEXT:    eor x12, x12, x15
+; CHECK-NEON-NEXT:    eor x9, x10, x11
+; CHECK-NEON-NEXT:    ldr x11, [sp, #336] // 8-byte Reload
+; CHECK-NEON-NEXT:    eor x10, x12, x13
+; CHECK-NEON-NEXT:    eor x0, x11, x8
+; CHECK-NEON-NEXT:    eor x1, x9, x10
+; CHECK-NEON-NEXT:    add sp, sp, #592
+; CHECK-NEON-NEXT:    ldp x20, x19, [sp, #80] // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    ldp x22, x21, [sp, #64] // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    ldp x24, x23, [sp, #48] // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    ldp x26, x25, [sp, #32] // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    ldp x28, x27, [sp, #16] // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    ldp x29, x30, [sp], #96 // 16-byte Folded Reload
+; CHECK-NEON-NEXT:    ret
+;
+; CHECK-AES-LABEL: clmul_i96:
+; CHECK-AES:       // %bb.0:
+; CHECK-AES-NEXT:    rbit x8, x2
+; CHECK-AES-NEXT:    rbit x9, x0
+; CHECK-AES-NEXT:    fmov d0, x3
+; CHECK-AES-NEXT:    fmov d1, x0
+; CHECK-AES-NEXT:    fmov d2, x2
+; CHECK-AES-NEXT:    fmov d3, x8
+; CHECK-AES-NEXT:    fmov d4, x9
+; CHECK-AES-NEXT:    pmull v0.1q, v1.1d, v0.1d
+; CHECK-AES-NEXT:    pmull v3.1q, v4.1d, v3.1d
+; CHECK-AES-NEXT:    fmov d4, x1
+; CHECK-AES-NEXT:    pmull v1.1q, v1.1d, v2.1d
+; CHECK-AES-NEXT:    pmull v4.1q, v4.1d, v2.1d
+; CHECK-AES-NEXT:    fmov x10, d0
+; CHECK-AES-NEXT:    fmov x8, d3
+; CHECK-AES-NEXT:    fmov x0, d1
+; CHECK-AES-NEXT:    fmov x9, d4
+; CHECK-AES-NEXT:    rbit x8, x8
+; CHECK-AES-NEXT:    eor x9, x10, x9
+; CHECK-AES-NEXT:    eor x1, x9, x8, lsr #1
+; CHECK-AES-NEXT:    ret
+  %a = call i96 @llvm.clmul.i96(i96 %x, i96 %y)
+  ret i96 %a
+}
 
 define i128 @clmul_i128(i128 %x, i128 %y) {
 ; CHECK-NEON-LABEL: clmul_i128:

diff  --git a/llvm/test/CodeGen/RISCV/clmul.ll b/llvm/test/CodeGen/RISCV/clmul.ll
index 32216d29eecd2..f62aca0039de4 100644
--- a/llvm/test/CodeGen/RISCV/clmul.ll
+++ b/llvm/test/CodeGen/RISCV/clmul.ll
@@ -5,6 +5,8 @@
 ; RUN: llc -mtriple=riscv64 -mattr=+m -verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,CHECK-M,RV64IM
 ; RUN: llc -mtriple=riscv32 -mattr=+m,+zbs -verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,CHECK-ZBS,RV32IMZBS
 ; RUN: llc -mtriple=riscv64 -mattr=+m,+zbs -verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,CHECK-ZBS,RV64IMZBS
+; RUN: llc -mtriple=riscv32 -mattr=+m,+zbc -verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,CHECK-ZBC,RV32IMZBC
+; RUN: llc -mtriple=riscv64 -mattr=+m,+zbc -verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,CHECK-ZBC,RV64IMZBC
 
 define i4 @clmul_i4(i4 %a, i4 %b) nounwind {
 ; RV32I-LABEL: clmul_i4:
@@ -144,6 +146,11 @@ define i4 @clmul_i4(i4 %a, i4 %b) nounwind {
 ; RV64IMZBS-NEXT:    and a0, a1, a0
 ; RV64IMZBS-NEXT:    xor a0, a2, a0
 ; RV64IMZBS-NEXT:    ret
+;
+; CHECK-ZBC-LABEL: clmul_i4:
+; CHECK-ZBC:       # %bb.0:
+; CHECK-ZBC-NEXT:    clmul a0, a0, a1
+; CHECK-ZBC-NEXT:    ret
   %res = call i4 @llvm.clmul.i4(i4 %a, i4 %b)
   ret i4 %res
 }
@@ -406,6 +413,11 @@ define i8 @clmul_i8(i8 %a, i8 %b) nounwind {
 ; RV64IMZBS-NEXT:    and a0, a1, a0
 ; RV64IMZBS-NEXT:    xor a0, a2, a0
 ; RV64IMZBS-NEXT:    ret
+;
+; CHECK-ZBC-LABEL: clmul_i8:
+; CHECK-ZBC:       # %bb.0:
+; CHECK-ZBC-NEXT:    clmul a0, a0, a1
+; CHECK-ZBC-NEXT:    ret
   %res = call i8 @llvm.clmul.i8(i8 %a, i8 %b)
   ret i8 %res
 }
@@ -900,6 +912,11 @@ define i16 @clmul_i16(i16 %a, i16 %b) nounwind {
 ; RV64IMZBS-NEXT:    xor a0, a4, a0
 ; RV64IMZBS-NEXT:    xor a0, a2, a0
 ; RV64IMZBS-NEXT:    ret
+;
+; CHECK-ZBC-LABEL: clmul_i16:
+; CHECK-ZBC:       # %bb.0:
+; CHECK-ZBC-NEXT:    clmul a0, a0, a1
+; CHECK-ZBC-NEXT:    ret
   %res = call i16 @llvm.clmul.i16(i16 %a, i16 %b)
   ret i16 %res
 }
@@ -1568,2087 +1585,25793 @@ define i32 @clmul_i32(i32 %a, i32 %b) nounwind {
 ; RV64IMZBS-NEXT:    ld s6, 8(sp) # 8-byte Folded Reload
 ; RV64IMZBS-NEXT:    addi sp, sp, 64
 ; RV64IMZBS-NEXT:    ret
+;
+; CHECK-ZBC-LABEL: clmul_i32:
+; CHECK-ZBC:       # %bb.0:
+; CHECK-ZBC-NEXT:    clmul a0, a0, a1
+; CHECK-ZBC-NEXT:    ret
   %res = call i32 @llvm.clmul.i32(i32 %a, i32 %b)
   ret i32 %res
 }
 
-define i64 @clmul_i64(i64 %a, i64 %b) nounwind {
-; RV32I-LABEL: clmul_i64:
+define i48 @clmul_i48(i48 %a, i48 %b) nounwind {
+; RV32I-LABEL: clmul_i48:
 ; RV32I:       # %bb.0:
-; RV32I-NEXT:    addi sp, sp, -240
-; RV32I-NEXT:    sw ra, 236(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s0, 232(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s1, 228(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s2, 224(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s3, 220(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s4, 216(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s5, 212(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s6, 208(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s7, 204(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s8, 200(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s9, 196(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s10, 192(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    sw s11, 188(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    mv s3, a3
-; RV32I-NEXT:    mv a3, a0
-; RV32I-NEXT:    lui a0, 16
-; RV32I-NEXT:    srli a5, a3, 8
-; RV32I-NEXT:    addi a4, a0, -256
-; RV32I-NEXT:    lui s4, 16
-; RV32I-NEXT:    and a5, a5, a4
-; RV32I-NEXT:    srli a6, a3, 24
-; RV32I-NEXT:    and a7, a3, a4
-; RV32I-NEXT:    slli a7, a7, 8
-; RV32I-NEXT:    slli a0, a3, 24
-; RV32I-NEXT:    sw a0, 180(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    or a5, a5, a6
-; RV32I-NEXT:    or a6, a0, a7
-; RV32I-NEXT:    or a6, a6, a5
-; RV32I-NEXT:    lui a5, 61681
-; RV32I-NEXT:    srli a7, a6, 4
-; RV32I-NEXT:    addi a5, a5, -241
-; RV32I-NEXT:    and a7, a7, a5
-; RV32I-NEXT:    and a6, a6, a5
-; RV32I-NEXT:    slli a6, a6, 4
-; RV32I-NEXT:    lui t0, 209715
-; RV32I-NEXT:    or a7, a7, a6
-; RV32I-NEXT:    addi a6, t0, 819
-; RV32I-NEXT:    srli t0, a7, 2
-; RV32I-NEXT:    and a7, a7, a6
-; RV32I-NEXT:    and t0, t0, a6
-; RV32I-NEXT:    slli a7, a7, 2
-; RV32I-NEXT:    or t1, t0, a7
-; RV32I-NEXT:    srli t2, t1, 1
-; RV32I-NEXT:    lui a7, 349525
-; RV32I-NEXT:    addi t0, a7, 1365
-; RV32I-NEXT:    srli t3, a2, 8
-; RV32I-NEXT:    and t2, t2, t0
-; RV32I-NEXT:    and t3, t3, a4
-; RV32I-NEXT:    srli t4, a2, 24
-; RV32I-NEXT:    and t5, a2, a4
-; RV32I-NEXT:    slli t5, t5, 8
-; RV32I-NEXT:    slli t6, a2, 24
-; RV32I-NEXT:    or t3, t3, t4
-; RV32I-NEXT:    or t4, t6, t5
-; RV32I-NEXT:    and t1, t1, t0
+; RV32I-NEXT:    addi sp, sp, -160
+; RV32I-NEXT:    sw ra, 156(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s0, 152(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s1, 148(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s2, 144(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s3, 140(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s4, 136(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s5, 132(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s6, 128(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s7, 124(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s8, 120(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s9, 116(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s10, 112(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s11, 108(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    mv t2, a3
+; RV32I-NEXT:    slli a3, a0, 1
+; RV32I-NEXT:    sw a3, 104(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, a2, 30
+; RV32I-NEXT:    slli a5, a2, 31
+; RV32I-NEXT:    srai ra, a4, 31
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    sw a5, 40(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, ra, a3
+; RV32I-NEXT:    and a5, a5, a0
+; RV32I-NEXT:    xor a5, a5, a4
+; RV32I-NEXT:    slli a6, a0, 2
+; RV32I-NEXT:    sw a6, 88(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s0, a2, 29
+; RV32I-NEXT:    srai s4, s0, 31
+; RV32I-NEXT:    slli a4, a2, 28
+; RV32I-NEXT:    slli a3, a0, 3
+; RV32I-NEXT:    sw a3, 100(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai t4, a4, 31
+; RV32I-NEXT:    and a6, s4, a6
+; RV32I-NEXT:    and a7, t4, a3
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    slli s1, a2, 27
+; RV32I-NEXT:    slli a3, a0, 4
+; RV32I-NEXT:    sw a3, 92(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai s6, s1, 31
+; RV32I-NEXT:    and a7, s6, a3
+; RV32I-NEXT:    slli t0, a2, 26
+; RV32I-NEXT:    slli a3, a0, 5
+; RV32I-NEXT:    sw a3, 96(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai a4, t0, 31
+; RV32I-NEXT:    sw a4, 32(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and t0, a4, a3
+; RV32I-NEXT:    slli t1, a2, 25
+; RV32I-NEXT:    slli a3, a0, 6
+; RV32I-NEXT:    sw a3, 84(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai s3, t1, 31
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    and t0, s3, a3
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    xor a6, a7, t0
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    slli a3, a0, 7
+; RV32I-NEXT:    sw a3, 68(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a2, 24
+; RV32I-NEXT:    srai a7, a6, 31
+; RV32I-NEXT:    sw a7, 28(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a2, 23
+; RV32I-NEXT:    slli a4, a0, 8
+; RV32I-NEXT:    sw a4, 80(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai s11, a6, 31
+; RV32I-NEXT:    and a6, a7, a3
+; RV32I-NEXT:    and a7, s11, a4
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    slli a7, a2, 22
+; RV32I-NEXT:    slli a3, a0, 9
+; RV32I-NEXT:    sw a3, 76(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai a4, a7, 31
+; RV32I-NEXT:    sw a4, 36(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a7, a4, a3
+; RV32I-NEXT:    slli t0, a2, 21
+; RV32I-NEXT:    slli a3, a0, 10
+; RV32I-NEXT:    sw a3, 72(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai a4, t0, 31
+; RV32I-NEXT:    sw a4, 24(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, a4, a3
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    slli a7, a2, 20
+; RV32I-NEXT:    srai t0, a7, 31
+; RV32I-NEXT:    sw t0, 0(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a7, a2, 19
+; RV32I-NEXT:    srai a4, a7, 31
+; RV32I-NEXT:    sw a4, 16(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a7, a2, 18
+; RV32I-NEXT:    srai t1, a7, 31
+; RV32I-NEXT:    sw t1, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, a0, 11
+; RV32I-NEXT:    sw a3, 48(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a7, t0, a3
+; RV32I-NEXT:    slli a3, a0, 12
+; RV32I-NEXT:    sw a3, 52(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and t0, a4, a3
+; RV32I-NEXT:    slli a3, a0, 13
+; RV32I-NEXT:    sw a3, 60(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    and t0, t1, a3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    slli t0, a2, 17
+; RV32I-NEXT:    srai a4, t0, 31
+; RV32I-NEXT:    sw a4, 8(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t0, a2, 16
+; RV32I-NEXT:    srai t1, t0, 31
+; RV32I-NEXT:    sw t1, 4(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, a0, 14
+; RV32I-NEXT:    sw a3, 64(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and t0, a4, a3
+; RV32I-NEXT:    slli a3, a0, 15
+; RV32I-NEXT:    sw a3, 56(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    and t0, t1, a3
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    xor a6, a7, t0
+; RV32I-NEXT:    xor t3, a5, a6
+; RV32I-NEXT:    slli a6, a2, 15
+; RV32I-NEXT:    slli a7, a2, 14
+; RV32I-NEXT:    srai s1, a6, 31
+; RV32I-NEXT:    srai s2, a7, 31
+; RV32I-NEXT:    slli a6, a0, 16
+; RV32I-NEXT:    slli a7, a0, 17
+; RV32I-NEXT:    and a6, s1, a6
+; RV32I-NEXT:    and a7, s2, a7
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    slli a7, a2, 13
+; RV32I-NEXT:    srai s5, a7, 31
+; RV32I-NEXT:    slli a7, a0, 18
+; RV32I-NEXT:    and a7, s5, a7
+; RV32I-NEXT:    slli t0, a2, 12
+; RV32I-NEXT:    srai s0, t0, 31
+; RV32I-NEXT:    slli t1, a0, 19
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, s0, t1
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    slli a7, a2, 11
+; RV32I-NEXT:    srai s7, a7, 31
+; RV32I-NEXT:    slli a7, a0, 20
+; RV32I-NEXT:    and t1, s7, a7
+; RV32I-NEXT:    slli a7, a2, 10
+; RV32I-NEXT:    srai a7, a7, 31
+; RV32I-NEXT:    slli t5, a0, 21
+; RV32I-NEXT:    xor a6, a6, t1
+; RV32I-NEXT:    and t1, a7, t5
+; RV32I-NEXT:    xor a6, a6, t1
+; RV32I-NEXT:    slli t1, a2, 9
+; RV32I-NEXT:    srai s10, t1, 31
+; RV32I-NEXT:    slli t1, a0, 22
+; RV32I-NEXT:    and t1, s10, t1
+; RV32I-NEXT:    slli t5, a2, 8
+; RV32I-NEXT:    srai a3, t5, 31
+; RV32I-NEXT:    sw a3, 20(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t5, a0, 23
+; RV32I-NEXT:    and t6, a3, t5
+; RV32I-NEXT:    slli t5, a2, 7
+; RV32I-NEXT:    srai t5, t5, 31
+; RV32I-NEXT:    slli s8, a0, 24
+; RV32I-NEXT:    xor t1, t1, t6
+; RV32I-NEXT:    and t6, t5, s8
+; RV32I-NEXT:    xor t1, t1, t6
+; RV32I-NEXT:    slli t6, a2, 6
+; RV32I-NEXT:    srai s8, t6, 31
+; RV32I-NEXT:    slli t6, a0, 25
+; RV32I-NEXT:    and s9, s8, t6
+; RV32I-NEXT:    slli t6, a2, 5
+; RV32I-NEXT:    srai t6, t6, 31
+; RV32I-NEXT:    slli a3, a0, 26
+; RV32I-NEXT:    xor t1, t1, s9
+; RV32I-NEXT:    and a3, t6, a3
+; RV32I-NEXT:    xor a5, t1, a3
+; RV32I-NEXT:    slli t1, a2, 4
+; RV32I-NEXT:    srai s9, t1, 31
+; RV32I-NEXT:    slli t1, a0, 27
+; RV32I-NEXT:    and a4, s9, t1
+; RV32I-NEXT:    slli t1, a2, 3
+; RV32I-NEXT:    srai t1, t1, 31
+; RV32I-NEXT:    slli a3, a0, 28
+; RV32I-NEXT:    xor a4, a5, a4
+; RV32I-NEXT:    and a3, t1, a3
+; RV32I-NEXT:    xor a5, t3, a6
+; RV32I-NEXT:    xor a3, a4, a3
+; RV32I-NEXT:    xor a3, a5, a3
+; RV32I-NEXT:    sw a3, 44(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a2, 2
+; RV32I-NEXT:    slli a3, a2, 1
+; RV32I-NEXT:    srai t0, a5, 31
+; RV32I-NEXT:    srai a6, a3, 31
+; RV32I-NEXT:    slli a3, a0, 29
+; RV32I-NEXT:    slli a4, a0, 30
+; RV32I-NEXT:    and a5, t0, a3
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    srli t3, a0, 31
+; RV32I-NEXT:    slli a3, a1, 1
+; RV32I-NEXT:    xor a5, a5, a4
+; RV32I-NEXT:    or a3, a3, t3
+; RV32I-NEXT:    srli a4, a0, 30
+; RV32I-NEXT:    slli t3, a1, 2
+; RV32I-NEXT:    and a3, ra, a3
+; RV32I-NEXT:    or a4, t3, a4
+; RV32I-NEXT:    srli t3, a0, 29
+; RV32I-NEXT:    slli ra, a1, 3
+; RV32I-NEXT:    and a4, s4, a4
+; RV32I-NEXT:    or t3, ra, t3
+; RV32I-NEXT:    and t3, t4, t3
+; RV32I-NEXT:    lw t4, 40(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, t4, a1
+; RV32I-NEXT:    xor a3, t4, a3
+; RV32I-NEXT:    xor a4, a4, t3
+; RV32I-NEXT:    srli t3, a0, 28
+; RV32I-NEXT:    slli t4, a1, 4
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    or a4, t4, t3
+; RV32I-NEXT:    srli t3, a0, 27
+; RV32I-NEXT:    slli t4, a1, 5
+; RV32I-NEXT:    and a4, s6, a4
 ; RV32I-NEXT:    or t3, t4, t3
-; RV32I-NEXT:    srli t4, t3, 4
-; RV32I-NEXT:    and t3, t3, a5
-; RV32I-NEXT:    and t4, t4, a5
-; RV32I-NEXT:    slli t3, t3, 4
-; RV32I-NEXT:    slli t1, t1, 1
+; RV32I-NEXT:    srli t4, a0, 26
+; RV32I-NEXT:    slli s4, a1, 6
+; RV32I-NEXT:    lw s6, 32(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t3, s6, t3
+; RV32I-NEXT:    or t4, s4, t4
+; RV32I-NEXT:    xor t3, a4, t3
+; RV32I-NEXT:    and t4, s3, t4
+; RV32I-NEXT:    srai a4, a2, 31
+; RV32I-NEXT:    slli a2, a0, 31
+; RV32I-NEXT:    xor t3, t3, t4
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    xor a2, a5, a2
+; RV32I-NEXT:    xor a3, a3, t3
+; RV32I-NEXT:    srli a5, a0, 25
+; RV32I-NEXT:    slli t3, a1, 7
+; RV32I-NEXT:    srli t4, a0, 24
+; RV32I-NEXT:    slli s3, a1, 8
+; RV32I-NEXT:    or a5, t3, a5
+; RV32I-NEXT:    or t3, s3, t4
+; RV32I-NEXT:    lw t4, 28(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, t4, a5
+; RV32I-NEXT:    and t3, s11, t3
+; RV32I-NEXT:    srli t4, a0, 23
+; RV32I-NEXT:    slli s3, a1, 9
+; RV32I-NEXT:    xor a5, a5, t3
+; RV32I-NEXT:    or t3, s3, t4
+; RV32I-NEXT:    srli t4, a0, 22
+; RV32I-NEXT:    slli s3, a1, 10
+; RV32I-NEXT:    lw s4, 36(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t3, s4, t3
+; RV32I-NEXT:    or t4, s3, t4
+; RV32I-NEXT:    xor a5, a5, t3
+; RV32I-NEXT:    lw t3, 24(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t3, t3, t4
+; RV32I-NEXT:    srli t4, a0, 21
+; RV32I-NEXT:    slli s3, a1, 11
+; RV32I-NEXT:    xor a5, a5, t3
+; RV32I-NEXT:    or t3, s3, t4
+; RV32I-NEXT:    srli t4, a0, 20
+; RV32I-NEXT:    slli s3, a1, 12
+; RV32I-NEXT:    lw s4, 0(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t3, s4, t3
+; RV32I-NEXT:    or t4, s3, t4
+; RV32I-NEXT:    srli s3, a0, 19
+; RV32I-NEXT:    slli s4, a1, 13
+; RV32I-NEXT:    lw s6, 16(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, s6, t4
+; RV32I-NEXT:    or s3, s4, s3
+; RV32I-NEXT:    xor t3, t3, t4
+; RV32I-NEXT:    lw t4, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, t4, s3
+; RV32I-NEXT:    srli s3, a0, 18
+; RV32I-NEXT:    slli s4, a1, 14
+; RV32I-NEXT:    xor t3, t3, t4
+; RV32I-NEXT:    or t4, s4, s3
+; RV32I-NEXT:    srli s3, a0, 17
+; RV32I-NEXT:    slli s4, a1, 15
+; RV32I-NEXT:    lw s6, 8(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, s6, t4
+; RV32I-NEXT:    or s3, s4, s3
+; RV32I-NEXT:    xor t3, t3, t4
+; RV32I-NEXT:    lw t4, 4(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, t4, s3
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    xor a5, t3, t4
+; RV32I-NEXT:    srli t3, a0, 16
+; RV32I-NEXT:    slli t4, a1, 16
+; RV32I-NEXT:    srli s3, a0, 15
+; RV32I-NEXT:    slli s4, a1, 17
 ; RV32I-NEXT:    or t3, t4, t3
-; RV32I-NEXT:    srli t4, t3, 2
-; RV32I-NEXT:    and t3, t3, a6
-; RV32I-NEXT:    and t4, t4, a6
-; RV32I-NEXT:    slli t3, t3, 2
-; RV32I-NEXT:    or t1, t2, t1
-; RV32I-NEXT:    or t2, t4, t3
-; RV32I-NEXT:    srli t3, t2, 1
-; RV32I-NEXT:    and t2, t2, t0
-; RV32I-NEXT:    and t3, t3, t0
-; RV32I-NEXT:    slli t2, t2, 1
-; RV32I-NEXT:    slli t4, t1, 1
-; RV32I-NEXT:    or t3, t3, t2
-; RV32I-NEXT:    andi t5, t3, 2
-; RV32I-NEXT:    andi t6, t3, 1
-; RV32I-NEXT:    seqz t5, t5
-; RV32I-NEXT:    seqz t6, t6
-; RV32I-NEXT:    addi t5, t5, -1
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    and t4, t5, t4
-; RV32I-NEXT:    and t5, t6, t1
-; RV32I-NEXT:    slli t6, t1, 2
-; RV32I-NEXT:    andi s0, t3, 4
-; RV32I-NEXT:    seqz s0, s0
-; RV32I-NEXT:    andi s1, t3, 8
-; RV32I-NEXT:    addi s0, s0, -1
-; RV32I-NEXT:    seqz s1, s1
-; RV32I-NEXT:    slli s2, t1, 3
-; RV32I-NEXT:    addi s1, s1, -1
-; RV32I-NEXT:    and t6, s0, t6
-; RV32I-NEXT:    and s0, s1, s2
-; RV32I-NEXT:    xor t4, t5, t4
-; RV32I-NEXT:    xor t5, t6, s0
-; RV32I-NEXT:    xor t4, t4, t5
-; RV32I-NEXT:    andi t5, t3, 16
-; RV32I-NEXT:    slli t6, t1, 4
-; RV32I-NEXT:    seqz t5, t5
-; RV32I-NEXT:    addi t5, t5, -1
-; RV32I-NEXT:    andi s0, t3, 32
-; RV32I-NEXT:    and t5, t5, t6
-; RV32I-NEXT:    seqz t6, s0
-; RV32I-NEXT:    slli s0, t1, 5
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    and t6, t6, s0
-; RV32I-NEXT:    andi s0, t3, 64
-; RV32I-NEXT:    xor t5, t5, t6
-; RV32I-NEXT:    seqz t6, s0
-; RV32I-NEXT:    slli s0, t1, 6
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    and t6, t6, s0
-; RV32I-NEXT:    andi s0, t3, 128
-; RV32I-NEXT:    xor t5, t5, t6
-; RV32I-NEXT:    seqz t6, s0
-; RV32I-NEXT:    slli s0, t1, 7
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    and t6, t6, s0
-; RV32I-NEXT:    andi s0, t3, 256
-; RV32I-NEXT:    slli s1, t1, 8
-; RV32I-NEXT:    seqz s0, s0
-; RV32I-NEXT:    addi s0, s0, -1
-; RV32I-NEXT:    andi s2, t3, 512
-; RV32I-NEXT:    and s0, s0, s1
-; RV32I-NEXT:    seqz s1, s2
-; RV32I-NEXT:    slli s2, t1, 9
-; RV32I-NEXT:    addi s1, s1, -1
-; RV32I-NEXT:    xor t6, t6, s0
-; RV32I-NEXT:    and s0, s1, s2
-; RV32I-NEXT:    xor t4, t4, t5
-; RV32I-NEXT:    xor t5, t6, s0
-; RV32I-NEXT:    slli t6, t1, 10
-; RV32I-NEXT:    andi s0, t3, 1024
-; RV32I-NEXT:    seqz s0, s0
-; RV32I-NEXT:    li s1, 1
-; RV32I-NEXT:    addi s0, s0, -1
-; RV32I-NEXT:    slli s5, s1, 11
-; RV32I-NEXT:    and t6, s0, t6
-; RV32I-NEXT:    and s0, t3, s5
-; RV32I-NEXT:    xor t5, t5, t6
-; RV32I-NEXT:    seqz t6, s0
-; RV32I-NEXT:    slli s0, t1, 11
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    and t6, t6, s0
-; RV32I-NEXT:    lui s6, 1
-; RV32I-NEXT:    slli s0, t1, 12
-; RV32I-NEXT:    and s1, t3, s6
-; RV32I-NEXT:    seqz s1, s1
-; RV32I-NEXT:    lui s7, 2
-; RV32I-NEXT:    addi s1, s1, -1
-; RV32I-NEXT:    and s2, t3, s7
-; RV32I-NEXT:    and s0, s1, s0
-; RV32I-NEXT:    seqz s1, s2
-; RV32I-NEXT:    slli s2, t1, 13
-; RV32I-NEXT:    addi s1, s1, -1
-; RV32I-NEXT:    xor t6, t6, s0
-; RV32I-NEXT:    and s0, s1, s2
-; RV32I-NEXT:    xor t6, t6, s0
-; RV32I-NEXT:    lui s8, 4
-; RV32I-NEXT:    slli s0, t1, 14
-; RV32I-NEXT:    and s1, t3, s8
-; RV32I-NEXT:    seqz s1, s1
-; RV32I-NEXT:    lui a0, 8
-; RV32I-NEXT:    addi s1, s1, -1
-; RV32I-NEXT:    and s2, t3, a0
-; RV32I-NEXT:    and s0, s1, s0
-; RV32I-NEXT:    seqz s1, s2
-; RV32I-NEXT:    slli s2, t1, 15
-; RV32I-NEXT:    addi s1, s1, -1
-; RV32I-NEXT:    xor t6, t6, s0
-; RV32I-NEXT:    and s0, s1, s2
-; RV32I-NEXT:    xor t4, t4, t5
-; RV32I-NEXT:    xor t5, t6, s0
-; RV32I-NEXT:    xor t4, t4, t5
-; RV32I-NEXT:    slli t5, t1, 16
-; RV32I-NEXT:    and t6, t3, s4
-; RV32I-NEXT:    lui s2, 32
-; RV32I-NEXT:    seqz t6, t6
-; RV32I-NEXT:    and s0, t3, s2
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    seqz s0, s0
-; RV32I-NEXT:    slli s1, t1, 17
-; RV32I-NEXT:    addi s0, s0, -1
-; RV32I-NEXT:    and t5, t6, t5
-; RV32I-NEXT:    and s0, s0, s1
-; RV32I-NEXT:    xor t5, t5, s0
-; RV32I-NEXT:    lui a0, 64
-; RV32I-NEXT:    slli t6, t1, 18
-; RV32I-NEXT:    and s0, t3, a0
-; RV32I-NEXT:    seqz s0, s0
-; RV32I-NEXT:    lui a0, 128
-; RV32I-NEXT:    addi s0, s0, -1
-; RV32I-NEXT:    and s1, t3, a0
-; RV32I-NEXT:    and t6, s0, t6
-; RV32I-NEXT:    seqz s0, s1
-; RV32I-NEXT:    slli s1, t1, 19
-; RV32I-NEXT:    addi s0, s0, -1
-; RV32I-NEXT:    xor t5, t5, t6
-; RV32I-NEXT:    and s0, s0, s1
-; RV32I-NEXT:    xor t5, t5, s0
-; RV32I-NEXT:    lui a0, 256
-; RV32I-NEXT:    slli t6, t1, 20
-; RV32I-NEXT:    and s0, t3, a0
-; RV32I-NEXT:    seqz s0, s0
-; RV32I-NEXT:    lui s1, 512
-; RV32I-NEXT:    addi s0, s0, -1
-; RV32I-NEXT:    and s1, t3, s1
-; RV32I-NEXT:    and t6, s0, t6
-; RV32I-NEXT:    seqz s0, s1
-; RV32I-NEXT:    addi s0, s0, -1
-; RV32I-NEXT:    slli s1, t1, 21
-; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    or t4, s4, s3
+; RV32I-NEXT:    and t3, s1, t3
+; RV32I-NEXT:    and t4, s2, t4
+; RV32I-NEXT:    srli s1, a0, 14
+; RV32I-NEXT:    slli s2, a1, 18
+; RV32I-NEXT:    xor t3, t3, t4
+; RV32I-NEXT:    or t4, s2, s1
+; RV32I-NEXT:    srli s1, a0, 13
+; RV32I-NEXT:    slli s2, a1, 19
+; RV32I-NEXT:    and t4, s5, t4
+; RV32I-NEXT:    or s1, s2, s1
+; RV32I-NEXT:    xor t3, t3, t4
 ; RV32I-NEXT:    and s0, s0, s1
-; RV32I-NEXT:    xor t5, t5, s0
-; RV32I-NEXT:    lui a0, 1024
-; RV32I-NEXT:    xor t4, t4, t5
-; RV32I-NEXT:    and t5, t3, a0
-; RV32I-NEXT:    seqz t5, t5
-; RV32I-NEXT:    lui a0, 2048
-; RV32I-NEXT:    addi t5, t5, -1
-; RV32I-NEXT:    and t6, t3, a0
-; RV32I-NEXT:    seqz t6, t6
-; RV32I-NEXT:    slli s0, t1, 22
-; RV32I-NEXT:    and t5, t5, s0
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    slli s0, t1, 23
-; RV32I-NEXT:    lui a0, 4096
-; RV32I-NEXT:    and t6, t6, s0
-; RV32I-NEXT:    and s0, t3, a0
-; RV32I-NEXT:    xor t5, t5, t6
-; RV32I-NEXT:    seqz t6, s0
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    lui s1, 8192
-; RV32I-NEXT:    slli s0, t1, 24
-; RV32I-NEXT:    and s1, t3, s1
-; RV32I-NEXT:    lui s9, 8192
-; RV32I-NEXT:    and t6, t6, s0
-; RV32I-NEXT:    seqz s0, s1
-; RV32I-NEXT:    xor t5, t5, t6
-; RV32I-NEXT:    addi s0, s0, -1
-; RV32I-NEXT:    slli t6, t1, 25
-; RV32I-NEXT:    lui s1, 16384
-; RV32I-NEXT:    and t6, s0, t6
-; RV32I-NEXT:    and s0, t3, s1
-; RV32I-NEXT:    lui s10, 16384
-; RV32I-NEXT:    xor t5, t5, t6
-; RV32I-NEXT:    seqz t6, s0
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    lui s1, 32768
-; RV32I-NEXT:    slli s0, t1, 26
-; RV32I-NEXT:    and s1, t3, s1
-; RV32I-NEXT:    lui s11, 32768
-; RV32I-NEXT:    and t6, t6, s0
-; RV32I-NEXT:    seqz s0, s1
-; RV32I-NEXT:    xor t5, t5, t6
-; RV32I-NEXT:    addi s0, s0, -1
-; RV32I-NEXT:    slli t6, t1, 27
-; RV32I-NEXT:    lui s1, 65536
-; RV32I-NEXT:    and t6, s0, t6
-; RV32I-NEXT:    and s0, t3, s1
-; RV32I-NEXT:    lui s4, 65536
-; RV32I-NEXT:    xor t5, t5, t6
-; RV32I-NEXT:    seqz t6, s0
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    slli s0, t1, 28
-; RV32I-NEXT:    and t6, t6, s0
-; RV32I-NEXT:    lui s0, 131072
-; RV32I-NEXT:    xor t5, t5, t6
-; RV32I-NEXT:    and t6, t3, s0
-; RV32I-NEXT:    lui s1, 131072
-; RV32I-NEXT:    seqz t6, t6
-; RV32I-NEXT:    lui s0, 262144
-; RV32I-NEXT:    addi t6, t6, -1
-; RV32I-NEXT:    and t3, t3, s0
-; RV32I-NEXT:    lui ra, 262144
-; RV32I-NEXT:    seqz t3, t3
-; RV32I-NEXT:    slli s0, t1, 29
-; RV32I-NEXT:    and t6, t6, s0
-; RV32I-NEXT:    addi t3, t3, -1
-; RV32I-NEXT:    srli t2, t2, 31
-; RV32I-NEXT:    slli s0, t1, 30
-; RV32I-NEXT:    and t3, t3, s0
-; RV32I-NEXT:    seqz t2, t2
-; RV32I-NEXT:    slli t1, t1, 31
-; RV32I-NEXT:    addi t2, t2, -1
-; RV32I-NEXT:    xor t3, t6, t3
-; RV32I-NEXT:    and t1, t2, t1
-; RV32I-NEXT:    xor t2, t4, t5
-; RV32I-NEXT:    xor t1, t3, t1
-; RV32I-NEXT:    xor t1, t2, t1
-; RV32I-NEXT:    srli t2, t1, 8
-; RV32I-NEXT:    and t2, t2, a4
-; RV32I-NEXT:    and a4, t1, a4
-; RV32I-NEXT:    srli t3, t1, 24
-; RV32I-NEXT:    slli t1, t1, 24
-; RV32I-NEXT:    slli a4, a4, 8
-; RV32I-NEXT:    or t2, t2, t3
-; RV32I-NEXT:    or a4, t1, a4
-; RV32I-NEXT:    or a4, a4, t2
-; RV32I-NEXT:    srli t1, a4, 4
+; RV32I-NEXT:    srli t4, a0, 12
+; RV32I-NEXT:    slli s1, a1, 20
+; RV32I-NEXT:    xor t3, t3, s0
+; RV32I-NEXT:    or t4, s1, t4
+; RV32I-NEXT:    srli s0, a0, 11
+; RV32I-NEXT:    slli s1, a1, 21
+; RV32I-NEXT:    and t4, s7, t4
+; RV32I-NEXT:    or s0, s1, s0
+; RV32I-NEXT:    xor t3, t3, t4
+; RV32I-NEXT:    and a7, a7, s0
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    xor a5, t3, a7
+; RV32I-NEXT:    srli t3, a0, 10
+; RV32I-NEXT:    slli t4, a1, 22
+; RV32I-NEXT:    xor a7, a3, a5
+; RV32I-NEXT:    or a3, t4, t3
+; RV32I-NEXT:    srli a5, a0, 9
+; RV32I-NEXT:    slli t3, a1, 23
+; RV32I-NEXT:    and a3, s10, a3
+; RV32I-NEXT:    or a5, t3, a5
+; RV32I-NEXT:    srli t3, a0, 8
+; RV32I-NEXT:    slli t4, a1, 24
+; RV32I-NEXT:    lw s0, 20(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, s0, a5
+; RV32I-NEXT:    or t3, t4, t3
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    and a5, t5, t3
+; RV32I-NEXT:    srli t3, a0, 7
+; RV32I-NEXT:    slli t4, a1, 25
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    or a5, t4, t3
+; RV32I-NEXT:    srli t3, a0, 6
+; RV32I-NEXT:    slli t4, a1, 26
+; RV32I-NEXT:    and a5, s8, a5
+; RV32I-NEXT:    or t3, t4, t3
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    and a5, t6, t3
+; RV32I-NEXT:    srli t3, a0, 5
+; RV32I-NEXT:    slli t4, a1, 27
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    or a5, t4, t3
+; RV32I-NEXT:    srli t3, a0, 4
+; RV32I-NEXT:    slli t4, a1, 28
+; RV32I-NEXT:    and a5, s9, a5
+; RV32I-NEXT:    or t3, t4, t3
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    and a5, t1, t3
+; RV32I-NEXT:    srli t1, a0, 3
+; RV32I-NEXT:    slli t3, a1, 29
+; RV32I-NEXT:    srli t4, a0, 2
+; RV32I-NEXT:    slli t5, a1, 30
+; RV32I-NEXT:    or t1, t3, t1
+; RV32I-NEXT:    or t3, t5, t4
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    and a6, a6, t3
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    xor a5, t0, a6
+; RV32I-NEXT:    srli a6, a0, 1
+; RV32I-NEXT:    slli a1, a1, 31
+; RV32I-NEXT:    or a1, a1, a6
+; RV32I-NEXT:    slli a6, t2, 31
+; RV32I-NEXT:    and a1, a4, a1
+; RV32I-NEXT:    srai a4, a6, 31
+; RV32I-NEXT:    xor a1, a5, a1
+; RV32I-NEXT:    and a0, a4, a0
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    slli a1, t2, 30
+; RV32I-NEXT:    srai a1, a1, 31
+; RV32I-NEXT:    slli a4, t2, 29
+; RV32I-NEXT:    lw a5, 104(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 88(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a4, a1
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    slli a1, t2, 28
+; RV32I-NEXT:    srai a1, a1, 31
+; RV32I-NEXT:    slli a4, t2, 27
+; RV32I-NEXT:    lw a5, 100(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 92(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a4, a1
+; RV32I-NEXT:    xor a3, a7, a3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a0, a3, a0
+; RV32I-NEXT:    slli a1, t2, 26
+; RV32I-NEXT:    srai a1, a1, 31
+; RV32I-NEXT:    slli a3, t2, 25
+; RV32I-NEXT:    lw a4, 96(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a4
+; RV32I-NEXT:    srai a3, a3, 31
+; RV32I-NEXT:    lw a4, 84(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    slli a4, t2, 24
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    lw a3, 68(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    slli a4, t2, 23
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    lw a3, 80(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    slli a4, t2, 22
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    lw a3, 76(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    slli a4, t2, 21
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    lw a3, 72(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    slli a4, t2, 20
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    lw a3, 48(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    slli a4, t2, 19
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    lw a3, 52(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    slli a4, t2, 18
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    lw a3, 60(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    lui a4, 8
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a3, t2, a4
+; RV32I-NEXT:    slli t2, t2, 17
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    srai a4, t2, 31
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    lw a5, 64(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    and a4, a4, a5
-; RV32I-NEXT:    and a5, t1, a5
+; RV32I-NEXT:    lw a5, 56(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    xor a1, a0, a1
+; RV32I-NEXT:    xor a3, a4, a3
+; RV32I-NEXT:    lw a0, 44(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    xor a0, a0, a2
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lw ra, 156(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s0, 152(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s1, 148(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s2, 144(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 140(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s4, 136(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s5, 132(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 128(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 124(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 120(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s9, 116(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s10, 112(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s11, 108(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    addi sp, sp, 160
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: clmul_i48:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    slli a2, a0, 1
+; RV64I-NEXT:    slli a3, a1, 62
+; RV64I-NEXT:    slli a4, a1, 63
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a2, a3, a2
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    xor a2, a4, a2
+; RV64I-NEXT:    slli a3, a0, 2
+; RV64I-NEXT:    slli a4, a1, 61
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    slli a5, a1, 60
+; RV64I-NEXT:    slli a6, a0, 3
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    slli a4, a1, 59
+; RV64I-NEXT:    slli a5, a0, 4
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 58
+; RV64I-NEXT:    slli a6, a0, 5
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    slli a6, a1, 57
+; RV64I-NEXT:    slli a7, a0, 6
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    slli a3, a0, 7
+; RV64I-NEXT:    slli a4, a1, 56
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    slli a5, a1, 55
+; RV64I-NEXT:    slli a6, a0, 8
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    slli a4, a1, 54
+; RV64I-NEXT:    slli a5, a0, 9
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 53
+; RV64I-NEXT:    slli a6, a0, 10
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    slli a4, a1, 52
+; RV64I-NEXT:    slli a5, a0, 11
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 51
+; RV64I-NEXT:    slli a6, a0, 12
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    slli a6, a1, 50
+; RV64I-NEXT:    slli a7, a0, 13
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 49
+; RV64I-NEXT:    slli a6, a0, 14
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    slli a6, a1, 48
+; RV64I-NEXT:    slli a7, a0, 15
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    slli a3, a0, 16
+; RV64I-NEXT:    slli a5, a1, 47
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    slli a6, a1, 46
+; RV64I-NEXT:    slli a7, a0, 17
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    and a3, a5, a3
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a3, a3, a5
+; RV64I-NEXT:    slli a5, a1, 45
+; RV64I-NEXT:    slli a6, a0, 18
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    slli a6, a1, 44
+; RV64I-NEXT:    slli a7, a0, 19
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    xor a3, a3, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a3, a3, a5
+; RV64I-NEXT:    slli a5, a1, 43
+; RV64I-NEXT:    slli a6, a0, 20
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    slli a6, a1, 42
+; RV64I-NEXT:    slli a7, a0, 21
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    xor a3, a3, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    xor a3, a3, a5
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, a1, 41
+; RV64I-NEXT:    slli a4, a0, 22
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    and a3, a3, a4
+; RV64I-NEXT:    slli a4, a1, 40
+; RV64I-NEXT:    slli a5, a0, 23
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 39
+; RV64I-NEXT:    slli a6, a0, 24
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    slli a4, a1, 38
+; RV64I-NEXT:    slli a5, a0, 25
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 37
+; RV64I-NEXT:    slli a6, a0, 26
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    slli a4, a1, 36
+; RV64I-NEXT:    slli a5, a0, 27
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 35
+; RV64I-NEXT:    slli a6, a0, 28
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    slli a5, a0, 29
+; RV64I-NEXT:    slli a6, a1, 34
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    slli a7, a1, 33
+; RV64I-NEXT:    slli t0, a0, 30
+; RV64I-NEXT:    srai a7, a7, 63
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a7, t0
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    xor a4, a5, a6
+; RV64I-NEXT:    slli a5, a0, 31
+; RV64I-NEXT:    sraiw a6, a1, 31
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    slli a6, a1, 31
+; RV64I-NEXT:    slli a7, a0, 32
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 30
+; RV64I-NEXT:    slli a6, a0, 33
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    slli a6, a1, 29
+; RV64I-NEXT:    slli a7, a0, 34
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 28
+; RV64I-NEXT:    slli a6, a0, 35
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    slli a6, a1, 27
+; RV64I-NEXT:    slli a7, a0, 36
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    slli a3, a1, 26
+; RV64I-NEXT:    slli a4, a0, 37
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    and a3, a3, a4
+; RV64I-NEXT:    slli a4, a1, 25
+; RV64I-NEXT:    slli a5, a0, 38
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 24
+; RV64I-NEXT:    slli a6, a0, 39
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    slli a4, a1, 23
+; RV64I-NEXT:    slli a5, a0, 40
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 22
+; RV64I-NEXT:    slli a6, a0, 41
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    slli a4, a1, 21
+; RV64I-NEXT:    slli a5, a0, 42
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 20
+; RV64I-NEXT:    slli a6, a0, 43
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    slli a4, a1, 19
+; RV64I-NEXT:    slli a5, a0, 44
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a5, a1, 18
+; RV64I-NEXT:    slli a6, a0, 45
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    li a4, 1
+; RV64I-NEXT:    slli a5, a0, 46
+; RV64I-NEXT:    slli a4, a4, 47
+; RV64I-NEXT:    and a4, a1, a4
+; RV64I-NEXT:    slli a1, a1, 17
+; RV64I-NEXT:    srai a1, a1, 63
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a0, a0, 47
+; RV64I-NEXT:    and a1, a1, a5
+; RV64I-NEXT:    and a0, a4, a0
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    xor a0, a1, a0
+; RV64I-NEXT:    xor a0, a2, a0
+; RV64I-NEXT:    ret
+;
+; RV32IM-LABEL: clmul_i48:
+; RV32IM:       # %bb.0:
+; RV32IM-NEXT:    addi sp, sp, -160
+; RV32IM-NEXT:    sw ra, 156(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s0, 152(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s1, 148(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s2, 144(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s3, 140(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s4, 136(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s5, 132(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s6, 128(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s7, 124(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s8, 120(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s9, 116(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s10, 112(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s11, 108(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv t2, a3
+; RV32IM-NEXT:    slli a3, a0, 1
+; RV32IM-NEXT:    sw a3, 104(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, a2, 30
+; RV32IM-NEXT:    slli a5, a2, 31
+; RV32IM-NEXT:    srai ra, a4, 31
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    sw a5, 40(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, ra, a3
+; RV32IM-NEXT:    and a5, a5, a0
+; RV32IM-NEXT:    xor a5, a5, a4
+; RV32IM-NEXT:    slli a6, a0, 2
+; RV32IM-NEXT:    sw a6, 88(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli s0, a2, 29
+; RV32IM-NEXT:    srai s4, s0, 31
+; RV32IM-NEXT:    slli a4, a2, 28
+; RV32IM-NEXT:    slli a3, a0, 3
+; RV32IM-NEXT:    sw a3, 100(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai t4, a4, 31
+; RV32IM-NEXT:    and a6, s4, a6
+; RV32IM-NEXT:    and a7, t4, a3
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    slli s1, a2, 27
+; RV32IM-NEXT:    slli a3, a0, 4
+; RV32IM-NEXT:    sw a3, 92(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai s6, s1, 31
+; RV32IM-NEXT:    and a7, s6, a3
+; RV32IM-NEXT:    slli t0, a2, 26
+; RV32IM-NEXT:    slli a3, a0, 5
+; RV32IM-NEXT:    sw a3, 96(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai a4, t0, 31
+; RV32IM-NEXT:    sw a4, 32(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and t0, a4, a3
+; RV32IM-NEXT:    slli t1, a2, 25
+; RV32IM-NEXT:    slli a3, a0, 6
+; RV32IM-NEXT:    sw a3, 84(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai s3, t1, 31
+; RV32IM-NEXT:    xor a7, a7, t0
+; RV32IM-NEXT:    and t0, s3, a3
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    xor a6, a7, t0
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    slli a3, a0, 7
+; RV32IM-NEXT:    sw a3, 68(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a6, a2, 24
+; RV32IM-NEXT:    srai a7, a6, 31
+; RV32IM-NEXT:    sw a7, 28(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a6, a2, 23
+; RV32IM-NEXT:    slli a4, a0, 8
+; RV32IM-NEXT:    sw a4, 80(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai s11, a6, 31
+; RV32IM-NEXT:    and a6, a7, a3
+; RV32IM-NEXT:    and a7, s11, a4
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    slli a7, a2, 22
+; RV32IM-NEXT:    slli a3, a0, 9
+; RV32IM-NEXT:    sw a3, 76(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai a4, a7, 31
+; RV32IM-NEXT:    sw a4, 36(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a7, a4, a3
+; RV32IM-NEXT:    slli t0, a2, 21
+; RV32IM-NEXT:    slli a3, a0, 10
+; RV32IM-NEXT:    sw a3, 72(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai a4, t0, 31
+; RV32IM-NEXT:    sw a4, 24(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    and a7, a4, a3
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    slli a7, a2, 20
+; RV32IM-NEXT:    srai t0, a7, 31
+; RV32IM-NEXT:    sw t0, 0(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a7, a2, 19
+; RV32IM-NEXT:    srai a4, a7, 31
+; RV32IM-NEXT:    sw a4, 16(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a7, a2, 18
+; RV32IM-NEXT:    srai t1, a7, 31
+; RV32IM-NEXT:    sw t1, 12(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a3, a0, 11
+; RV32IM-NEXT:    sw a3, 48(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a7, t0, a3
+; RV32IM-NEXT:    slli a3, a0, 12
+; RV32IM-NEXT:    sw a3, 52(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and t0, a4, a3
+; RV32IM-NEXT:    slli a3, a0, 13
+; RV32IM-NEXT:    sw a3, 60(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a7, a7, t0
+; RV32IM-NEXT:    and t0, t1, a3
+; RV32IM-NEXT:    xor a7, a7, t0
+; RV32IM-NEXT:    slli t0, a2, 17
+; RV32IM-NEXT:    srai a4, t0, 31
+; RV32IM-NEXT:    sw a4, 8(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli t0, a2, 16
+; RV32IM-NEXT:    srai t1, t0, 31
+; RV32IM-NEXT:    sw t1, 4(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a3, a0, 14
+; RV32IM-NEXT:    sw a3, 64(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and t0, a4, a3
+; RV32IM-NEXT:    slli a3, a0, 15
+; RV32IM-NEXT:    sw a3, 56(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a7, a7, t0
+; RV32IM-NEXT:    and t0, t1, a3
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    xor a6, a7, t0
+; RV32IM-NEXT:    xor t3, a5, a6
+; RV32IM-NEXT:    slli a6, a2, 15
+; RV32IM-NEXT:    slli a7, a2, 14
+; RV32IM-NEXT:    srai s1, a6, 31
+; RV32IM-NEXT:    srai s2, a7, 31
+; RV32IM-NEXT:    slli a6, a0, 16
+; RV32IM-NEXT:    slli a7, a0, 17
+; RV32IM-NEXT:    and a6, s1, a6
+; RV32IM-NEXT:    and a7, s2, a7
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    slli a7, a2, 13
+; RV32IM-NEXT:    srai s5, a7, 31
+; RV32IM-NEXT:    slli a7, a0, 18
+; RV32IM-NEXT:    and a7, s5, a7
+; RV32IM-NEXT:    slli t0, a2, 12
+; RV32IM-NEXT:    srai s0, t0, 31
+; RV32IM-NEXT:    slli t1, a0, 19
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    and a7, s0, t1
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    slli a7, a2, 11
+; RV32IM-NEXT:    srai s7, a7, 31
+; RV32IM-NEXT:    slli a7, a0, 20
+; RV32IM-NEXT:    and t1, s7, a7
+; RV32IM-NEXT:    slli a7, a2, 10
+; RV32IM-NEXT:    srai a7, a7, 31
+; RV32IM-NEXT:    slli t5, a0, 21
+; RV32IM-NEXT:    xor a6, a6, t1
+; RV32IM-NEXT:    and t1, a7, t5
+; RV32IM-NEXT:    xor a6, a6, t1
+; RV32IM-NEXT:    slli t1, a2, 9
+; RV32IM-NEXT:    srai s10, t1, 31
+; RV32IM-NEXT:    slli t1, a0, 22
+; RV32IM-NEXT:    and t1, s10, t1
+; RV32IM-NEXT:    slli t5, a2, 8
+; RV32IM-NEXT:    srai a3, t5, 31
+; RV32IM-NEXT:    sw a3, 20(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli t5, a0, 23
+; RV32IM-NEXT:    and t6, a3, t5
+; RV32IM-NEXT:    slli t5, a2, 7
+; RV32IM-NEXT:    srai t5, t5, 31
+; RV32IM-NEXT:    slli s8, a0, 24
+; RV32IM-NEXT:    xor t1, t1, t6
+; RV32IM-NEXT:    and t6, t5, s8
+; RV32IM-NEXT:    xor t1, t1, t6
+; RV32IM-NEXT:    slli t6, a2, 6
+; RV32IM-NEXT:    srai s8, t6, 31
+; RV32IM-NEXT:    slli t6, a0, 25
+; RV32IM-NEXT:    and s9, s8, t6
+; RV32IM-NEXT:    slli t6, a2, 5
+; RV32IM-NEXT:    srai t6, t6, 31
+; RV32IM-NEXT:    slli a3, a0, 26
+; RV32IM-NEXT:    xor t1, t1, s9
+; RV32IM-NEXT:    and a3, t6, a3
+; RV32IM-NEXT:    xor a5, t1, a3
+; RV32IM-NEXT:    slli t1, a2, 4
+; RV32IM-NEXT:    srai s9, t1, 31
+; RV32IM-NEXT:    slli t1, a0, 27
+; RV32IM-NEXT:    and a4, s9, t1
+; RV32IM-NEXT:    slli t1, a2, 3
+; RV32IM-NEXT:    srai t1, t1, 31
+; RV32IM-NEXT:    slli a3, a0, 28
+; RV32IM-NEXT:    xor a4, a5, a4
+; RV32IM-NEXT:    and a3, t1, a3
+; RV32IM-NEXT:    xor a5, t3, a6
+; RV32IM-NEXT:    xor a3, a4, a3
+; RV32IM-NEXT:    xor a3, a5, a3
+; RV32IM-NEXT:    sw a3, 44(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, a2, 2
+; RV32IM-NEXT:    slli a3, a2, 1
+; RV32IM-NEXT:    srai t0, a5, 31
+; RV32IM-NEXT:    srai a6, a3, 31
+; RV32IM-NEXT:    slli a3, a0, 29
+; RV32IM-NEXT:    slli a4, a0, 30
+; RV32IM-NEXT:    and a5, t0, a3
+; RV32IM-NEXT:    and a4, a6, a4
+; RV32IM-NEXT:    srli t3, a0, 31
+; RV32IM-NEXT:    slli a3, a1, 1
+; RV32IM-NEXT:    xor a5, a5, a4
+; RV32IM-NEXT:    or a3, a3, t3
+; RV32IM-NEXT:    srli a4, a0, 30
+; RV32IM-NEXT:    slli t3, a1, 2
+; RV32IM-NEXT:    and a3, ra, a3
+; RV32IM-NEXT:    or a4, t3, a4
+; RV32IM-NEXT:    srli t3, a0, 29
+; RV32IM-NEXT:    slli ra, a1, 3
+; RV32IM-NEXT:    and a4, s4, a4
+; RV32IM-NEXT:    or t3, ra, t3
+; RV32IM-NEXT:    and t3, t4, t3
+; RV32IM-NEXT:    lw t4, 40(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t4, t4, a1
+; RV32IM-NEXT:    xor a3, t4, a3
+; RV32IM-NEXT:    xor a4, a4, t3
+; RV32IM-NEXT:    srli t3, a0, 28
+; RV32IM-NEXT:    slli t4, a1, 4
+; RV32IM-NEXT:    xor a3, a3, a4
+; RV32IM-NEXT:    or a4, t4, t3
+; RV32IM-NEXT:    srli t3, a0, 27
+; RV32IM-NEXT:    slli t4, a1, 5
+; RV32IM-NEXT:    and a4, s6, a4
+; RV32IM-NEXT:    or t3, t4, t3
+; RV32IM-NEXT:    srli t4, a0, 26
+; RV32IM-NEXT:    slli s4, a1, 6
+; RV32IM-NEXT:    lw s6, 32(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t3, s6, t3
+; RV32IM-NEXT:    or t4, s4, t4
+; RV32IM-NEXT:    xor t3, a4, t3
+; RV32IM-NEXT:    and t4, s3, t4
+; RV32IM-NEXT:    srai a4, a2, 31
+; RV32IM-NEXT:    slli a2, a0, 31
+; RV32IM-NEXT:    xor t3, t3, t4
+; RV32IM-NEXT:    and a2, a4, a2
+; RV32IM-NEXT:    xor a2, a5, a2
+; RV32IM-NEXT:    xor a3, a3, t3
+; RV32IM-NEXT:    srli a5, a0, 25
+; RV32IM-NEXT:    slli t3, a1, 7
+; RV32IM-NEXT:    srli t4, a0, 24
+; RV32IM-NEXT:    slli s3, a1, 8
+; RV32IM-NEXT:    or a5, t3, a5
+; RV32IM-NEXT:    or t3, s3, t4
+; RV32IM-NEXT:    lw t4, 28(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a5, t4, a5
+; RV32IM-NEXT:    and t3, s11, t3
+; RV32IM-NEXT:    srli t4, a0, 23
+; RV32IM-NEXT:    slli s3, a1, 9
+; RV32IM-NEXT:    xor a5, a5, t3
+; RV32IM-NEXT:    or t3, s3, t4
+; RV32IM-NEXT:    srli t4, a0, 22
+; RV32IM-NEXT:    slli s3, a1, 10
+; RV32IM-NEXT:    lw s4, 36(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t3, s4, t3
+; RV32IM-NEXT:    or t4, s3, t4
+; RV32IM-NEXT:    xor a5, a5, t3
+; RV32IM-NEXT:    lw t3, 24(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t3, t3, t4
+; RV32IM-NEXT:    srli t4, a0, 21
+; RV32IM-NEXT:    slli s3, a1, 11
+; RV32IM-NEXT:    xor a5, a5, t3
+; RV32IM-NEXT:    or t3, s3, t4
+; RV32IM-NEXT:    srli t4, a0, 20
+; RV32IM-NEXT:    slli s3, a1, 12
+; RV32IM-NEXT:    lw s4, 0(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t3, s4, t3
+; RV32IM-NEXT:    or t4, s3, t4
+; RV32IM-NEXT:    srli s3, a0, 19
+; RV32IM-NEXT:    slli s4, a1, 13
+; RV32IM-NEXT:    lw s6, 16(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t4, s6, t4
+; RV32IM-NEXT:    or s3, s4, s3
+; RV32IM-NEXT:    xor t3, t3, t4
+; RV32IM-NEXT:    lw t4, 12(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t4, t4, s3
+; RV32IM-NEXT:    srli s3, a0, 18
+; RV32IM-NEXT:    slli s4, a1, 14
+; RV32IM-NEXT:    xor t3, t3, t4
+; RV32IM-NEXT:    or t4, s4, s3
+; RV32IM-NEXT:    srli s3, a0, 17
+; RV32IM-NEXT:    slli s4, a1, 15
+; RV32IM-NEXT:    lw s6, 8(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t4, s6, t4
+; RV32IM-NEXT:    or s3, s4, s3
+; RV32IM-NEXT:    xor t3, t3, t4
+; RV32IM-NEXT:    lw t4, 4(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t4, t4, s3
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    xor a5, t3, t4
+; RV32IM-NEXT:    srli t3, a0, 16
+; RV32IM-NEXT:    slli t4, a1, 16
+; RV32IM-NEXT:    srli s3, a0, 15
+; RV32IM-NEXT:    slli s4, a1, 17
+; RV32IM-NEXT:    or t3, t4, t3
+; RV32IM-NEXT:    or t4, s4, s3
+; RV32IM-NEXT:    and t3, s1, t3
+; RV32IM-NEXT:    and t4, s2, t4
+; RV32IM-NEXT:    srli s1, a0, 14
+; RV32IM-NEXT:    slli s2, a1, 18
+; RV32IM-NEXT:    xor t3, t3, t4
+; RV32IM-NEXT:    or t4, s2, s1
+; RV32IM-NEXT:    srli s1, a0, 13
+; RV32IM-NEXT:    slli s2, a1, 19
+; RV32IM-NEXT:    and t4, s5, t4
+; RV32IM-NEXT:    or s1, s2, s1
+; RV32IM-NEXT:    xor t3, t3, t4
+; RV32IM-NEXT:    and s0, s0, s1
+; RV32IM-NEXT:    srli t4, a0, 12
+; RV32IM-NEXT:    slli s1, a1, 20
+; RV32IM-NEXT:    xor t3, t3, s0
+; RV32IM-NEXT:    or t4, s1, t4
+; RV32IM-NEXT:    srli s0, a0, 11
+; RV32IM-NEXT:    slli s1, a1, 21
+; RV32IM-NEXT:    and t4, s7, t4
+; RV32IM-NEXT:    or s0, s1, s0
+; RV32IM-NEXT:    xor t3, t3, t4
+; RV32IM-NEXT:    and a7, a7, s0
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    xor a5, t3, a7
+; RV32IM-NEXT:    srli t3, a0, 10
+; RV32IM-NEXT:    slli t4, a1, 22
+; RV32IM-NEXT:    xor a7, a3, a5
+; RV32IM-NEXT:    or a3, t4, t3
+; RV32IM-NEXT:    srli a5, a0, 9
+; RV32IM-NEXT:    slli t3, a1, 23
+; RV32IM-NEXT:    and a3, s10, a3
+; RV32IM-NEXT:    or a5, t3, a5
+; RV32IM-NEXT:    srli t3, a0, 8
+; RV32IM-NEXT:    slli t4, a1, 24
+; RV32IM-NEXT:    lw s0, 20(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a5, s0, a5
+; RV32IM-NEXT:    or t3, t4, t3
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    and a5, t5, t3
+; RV32IM-NEXT:    srli t3, a0, 7
+; RV32IM-NEXT:    slli t4, a1, 25
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    or a5, t4, t3
+; RV32IM-NEXT:    srli t3, a0, 6
+; RV32IM-NEXT:    slli t4, a1, 26
+; RV32IM-NEXT:    and a5, s8, a5
+; RV32IM-NEXT:    or t3, t4, t3
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    and a5, t6, t3
+; RV32IM-NEXT:    srli t3, a0, 5
+; RV32IM-NEXT:    slli t4, a1, 27
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    or a5, t4, t3
+; RV32IM-NEXT:    srli t3, a0, 4
+; RV32IM-NEXT:    slli t4, a1, 28
+; RV32IM-NEXT:    and a5, s9, a5
+; RV32IM-NEXT:    or t3, t4, t3
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    and a5, t1, t3
+; RV32IM-NEXT:    srli t1, a0, 3
+; RV32IM-NEXT:    slli t3, a1, 29
+; RV32IM-NEXT:    srli t4, a0, 2
+; RV32IM-NEXT:    slli t5, a1, 30
+; RV32IM-NEXT:    or t1, t3, t1
+; RV32IM-NEXT:    or t3, t5, t4
+; RV32IM-NEXT:    and t0, t0, t1
+; RV32IM-NEXT:    and a6, a6, t3
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    xor a5, t0, a6
+; RV32IM-NEXT:    srli a6, a0, 1
+; RV32IM-NEXT:    slli a1, a1, 31
+; RV32IM-NEXT:    or a1, a1, a6
+; RV32IM-NEXT:    slli a6, t2, 31
+; RV32IM-NEXT:    and a1, a4, a1
+; RV32IM-NEXT:    srai a4, a6, 31
+; RV32IM-NEXT:    xor a1, a5, a1
+; RV32IM-NEXT:    and a0, a4, a0
+; RV32IM-NEXT:    xor a0, a1, a0
+; RV32IM-NEXT:    slli a1, t2, 30
+; RV32IM-NEXT:    srai a1, a1, 31
+; RV32IM-NEXT:    slli a4, t2, 29
+; RV32IM-NEXT:    lw a5, 104(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, a5
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    lw a1, 88(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a4, a1
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    slli a1, t2, 28
+; RV32IM-NEXT:    srai a1, a1, 31
+; RV32IM-NEXT:    slli a4, t2, 27
+; RV32IM-NEXT:    lw a5, 100(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, a5
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    lw a1, 92(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a4, a1
+; RV32IM-NEXT:    xor a3, a7, a3
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    xor a0, a3, a0
+; RV32IM-NEXT:    slli a1, t2, 26
+; RV32IM-NEXT:    srai a1, a1, 31
+; RV32IM-NEXT:    slli a3, t2, 25
+; RV32IM-NEXT:    lw a4, 96(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, a4
+; RV32IM-NEXT:    srai a3, a3, 31
+; RV32IM-NEXT:    lw a4, 84(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a3, a4
+; RV32IM-NEXT:    slli a4, t2, 24
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    lw a3, 68(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a4, a3
+; RV32IM-NEXT:    slli a4, t2, 23
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    lw a3, 80(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a4, a3
+; RV32IM-NEXT:    slli a4, t2, 22
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    lw a3, 76(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a4, a3
+; RV32IM-NEXT:    slli a4, t2, 21
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    lw a3, 72(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a4, a3
+; RV32IM-NEXT:    slli a4, t2, 20
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    lw a3, 48(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a4, a3
+; RV32IM-NEXT:    slli a4, t2, 19
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    lw a3, 52(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a4, a3
+; RV32IM-NEXT:    slli a4, t2, 18
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    lw a3, 60(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a4, a3
+; RV32IM-NEXT:    lui a4, 8
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    and a3, t2, a4
+; RV32IM-NEXT:    slli t2, t2, 17
+; RV32IM-NEXT:    seqz a3, a3
+; RV32IM-NEXT:    srai a4, t2, 31
+; RV32IM-NEXT:    addi a3, a3, -1
+; RV32IM-NEXT:    lw a5, 64(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    lw a5, 56(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a3, a5
+; RV32IM-NEXT:    xor a1, a0, a1
+; RV32IM-NEXT:    xor a3, a4, a3
+; RV32IM-NEXT:    lw a0, 44(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    xor a0, a0, a2
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    lw ra, 156(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s0, 152(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s1, 148(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s2, 144(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s3, 140(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s4, 136(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s5, 132(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s6, 128(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s7, 124(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s8, 120(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s9, 116(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s10, 112(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s11, 108(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    addi sp, sp, 160
+; RV32IM-NEXT:    ret
+;
+; RV64IM-LABEL: clmul_i48:
+; RV64IM:       # %bb.0:
+; RV64IM-NEXT:    addi sp, sp, -64
+; RV64IM-NEXT:    sd s0, 56(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s1, 48(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s2, 40(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s3, 32(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s4, 24(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s5, 16(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s6, 8(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    lui a2, %hi(.LCPI4_0)
+; RV64IM-NEXT:    lui a3, %hi(.LCPI4_1)
+; RV64IM-NEXT:    ld a2, %lo(.LCPI4_0)(a2)
+; RV64IM-NEXT:    ld a3, %lo(.LCPI4_1)(a3)
+; RV64IM-NEXT:    lui a4, %hi(.LCPI4_2)
+; RV64IM-NEXT:    lui a5, %hi(.LCPI4_3)
+; RV64IM-NEXT:    ld a4, %lo(.LCPI4_2)(a4)
+; RV64IM-NEXT:    ld a5, %lo(.LCPI4_3)(a5)
+; RV64IM-NEXT:    and a6, a1, a2
+; RV64IM-NEXT:    and a7, a0, a3
+; RV64IM-NEXT:    and t0, a1, a3
+; RV64IM-NEXT:    and t1, a0, a2
+; RV64IM-NEXT:    mul t2, a7, a6
+; RV64IM-NEXT:    mul t3, t1, t0
+; RV64IM-NEXT:    and t4, a1, a4
+; RV64IM-NEXT:    and t5, a0, a5
+; RV64IM-NEXT:    and a1, a1, a5
+; RV64IM-NEXT:    and a0, a0, a4
+; RV64IM-NEXT:    mul t6, t5, t4
+; RV64IM-NEXT:    mul s0, a0, a1
+; RV64IM-NEXT:    mul s1, a7, t4
+; RV64IM-NEXT:    mul s2, t1, a6
+; RV64IM-NEXT:    mul s3, t5, a1
+; RV64IM-NEXT:    mul s4, a0, t0
+; RV64IM-NEXT:    mul s5, a7, t0
+; RV64IM-NEXT:    mul s6, t1, a1
+; RV64IM-NEXT:    mul a1, a7, a1
+; RV64IM-NEXT:    mul a7, t5, a6
+; RV64IM-NEXT:    mul t1, t1, t4
+; RV64IM-NEXT:    mul t4, a0, t4
+; RV64IM-NEXT:    mul t0, t5, t0
+; RV64IM-NEXT:    mul a0, a0, a6
+; RV64IM-NEXT:    xor a6, t3, t2
+; RV64IM-NEXT:    xor t2, t6, s0
+; RV64IM-NEXT:    xor t3, s2, s1
+; RV64IM-NEXT:    xor t5, s3, s4
+; RV64IM-NEXT:    xor a6, a6, t2
+; RV64IM-NEXT:    xor t2, t3, t5
+; RV64IM-NEXT:    and a3, a6, a3
+; RV64IM-NEXT:    and a2, t2, a2
+; RV64IM-NEXT:    xor a6, s6, s5
+; RV64IM-NEXT:    xor a7, a7, t4
+; RV64IM-NEXT:    xor a1, t1, a1
+; RV64IM-NEXT:    xor a0, t0, a0
+; RV64IM-NEXT:    xor a6, a6, a7
+; RV64IM-NEXT:    xor a0, a1, a0
+; RV64IM-NEXT:    and a1, a6, a5
+; RV64IM-NEXT:    and a0, a0, a4
+; RV64IM-NEXT:    or a2, a2, a3
+; RV64IM-NEXT:    or a0, a1, a0
+; RV64IM-NEXT:    or a0, a2, a0
+; RV64IM-NEXT:    ld s0, 56(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s1, 48(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s2, 40(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s3, 32(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s4, 24(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s5, 16(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s6, 8(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    addi sp, sp, 64
+; RV64IM-NEXT:    ret
+;
+; RV32IMZBS-LABEL: clmul_i48:
+; RV32IMZBS:       # %bb.0:
+; RV32IMZBS-NEXT:    addi sp, sp, -160
+; RV32IMZBS-NEXT:    sw ra, 156(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s0, 152(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s1, 148(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s2, 144(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s3, 140(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s4, 136(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s5, 132(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s6, 128(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s7, 124(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s8, 120(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s9, 116(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s10, 112(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s11, 108(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv a6, a3
+; RV32IMZBS-NEXT:    slli a3, a0, 1
+; RV32IMZBS-NEXT:    sw a3, 104(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, a2, 30
+; RV32IMZBS-NEXT:    slli a5, a2, 31
+; RV32IMZBS-NEXT:    srai t6, a4, 31
+; RV32IMZBS-NEXT:    srai s2, a5, 31
+; RV32IMZBS-NEXT:    and a5, t6, a3
+; RV32IMZBS-NEXT:    and a7, s2, a0
+; RV32IMZBS-NEXT:    xor a5, a7, a5
+; RV32IMZBS-NEXT:    slli a4, a0, 2
+; RV32IMZBS-NEXT:    sw a4, 84(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a7, a2, 29
+; RV32IMZBS-NEXT:    srai t0, a7, 31
+; RV32IMZBS-NEXT:    sw t0, 20(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a7, a2, 28
+; RV32IMZBS-NEXT:    slli a3, a0, 3
+; RV32IMZBS-NEXT:    sw a3, 100(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai s4, a7, 31
+; RV32IMZBS-NEXT:    and a7, t0, a4
+; RV32IMZBS-NEXT:    and t0, s4, a3
+; RV32IMZBS-NEXT:    xor a7, a7, t0
+; RV32IMZBS-NEXT:    slli t0, a2, 27
+; RV32IMZBS-NEXT:    slli a3, a0, 4
+; RV32IMZBS-NEXT:    sw a3, 88(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai s11, t0, 31
+; RV32IMZBS-NEXT:    and t0, s11, a3
+; RV32IMZBS-NEXT:    slli t1, a2, 26
+; RV32IMZBS-NEXT:    slli a3, a0, 5
+; RV32IMZBS-NEXT:    sw a3, 96(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai a4, t1, 31
+; RV32IMZBS-NEXT:    sw a4, 40(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and t1, a4, a3
+; RV32IMZBS-NEXT:    slli t2, a2, 25
+; RV32IMZBS-NEXT:    slli a3, a0, 6
+; RV32IMZBS-NEXT:    sw a3, 92(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai s3, t2, 31
+; RV32IMZBS-NEXT:    xor t0, t0, t1
+; RV32IMZBS-NEXT:    and t1, s3, a3
+; RV32IMZBS-NEXT:    xor a5, a5, a7
+; RV32IMZBS-NEXT:    xor a7, t0, t1
+; RV32IMZBS-NEXT:    xor a5, a5, a7
+; RV32IMZBS-NEXT:    slli a4, a0, 7
+; RV32IMZBS-NEXT:    sw a4, 68(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a7, a2, 24
+; RV32IMZBS-NEXT:    srai a7, a7, 31
+; RV32IMZBS-NEXT:    sw a7, 16(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli s0, a2, 23
+; RV32IMZBS-NEXT:    slli a3, a0, 8
+; RV32IMZBS-NEXT:    sw a3, 80(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai s0, s0, 31
+; RV32IMZBS-NEXT:    and a7, a7, a4
+; RV32IMZBS-NEXT:    and t0, s0, a3
+; RV32IMZBS-NEXT:    xor a7, a7, t0
+; RV32IMZBS-NEXT:    slli t0, a2, 22
+; RV32IMZBS-NEXT:    slli a3, a0, 9
+; RV32IMZBS-NEXT:    sw a3, 72(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai a4, t0, 31
+; RV32IMZBS-NEXT:    sw a4, 36(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and t0, a4, a3
+; RV32IMZBS-NEXT:    slli s1, a2, 21
+; RV32IMZBS-NEXT:    slli a3, a0, 10
+; RV32IMZBS-NEXT:    sw a3, 76(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai s1, s1, 31
+; RV32IMZBS-NEXT:    xor a7, a7, t0
+; RV32IMZBS-NEXT:    and t0, s1, a3
+; RV32IMZBS-NEXT:    xor a7, a7, t0
+; RV32IMZBS-NEXT:    slli t0, a2, 20
+; RV32IMZBS-NEXT:    srai t1, t0, 31
+; RV32IMZBS-NEXT:    sw t1, 0(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t0, a2, 19
+; RV32IMZBS-NEXT:    srai a4, t0, 31
+; RV32IMZBS-NEXT:    sw a4, 12(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t0, a2, 18
+; RV32IMZBS-NEXT:    srai t2, t0, 31
+; RV32IMZBS-NEXT:    sw t2, 8(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a3, a0, 11
+; RV32IMZBS-NEXT:    sw a3, 48(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and t0, t1, a3
+; RV32IMZBS-NEXT:    slli a3, a0, 12
+; RV32IMZBS-NEXT:    sw a3, 60(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and t1, a4, a3
+; RV32IMZBS-NEXT:    slli a3, a0, 13
+; RV32IMZBS-NEXT:    sw a3, 64(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor t0, t0, t1
+; RV32IMZBS-NEXT:    and t1, t2, a3
+; RV32IMZBS-NEXT:    xor t0, t0, t1
+; RV32IMZBS-NEXT:    slli t1, a2, 17
+; RV32IMZBS-NEXT:    srai a4, t1, 31
+; RV32IMZBS-NEXT:    sw a4, 4(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t1, a2, 16
+; RV32IMZBS-NEXT:    srai s9, t1, 31
+; RV32IMZBS-NEXT:    slli a3, a0, 14
+; RV32IMZBS-NEXT:    sw a3, 56(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and t1, a4, a3
+; RV32IMZBS-NEXT:    slli a3, a0, 15
+; RV32IMZBS-NEXT:    sw a3, 52(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor t0, t0, t1
+; RV32IMZBS-NEXT:    and t1, s9, a3
+; RV32IMZBS-NEXT:    xor a5, a5, a7
+; RV32IMZBS-NEXT:    xor a7, t0, t1
+; RV32IMZBS-NEXT:    xor t5, a5, a7
+; RV32IMZBS-NEXT:    slli a7, a2, 15
+; RV32IMZBS-NEXT:    slli t0, a2, 14
+; RV32IMZBS-NEXT:    srai s10, a7, 31
+; RV32IMZBS-NEXT:    srai s6, t0, 31
+; RV32IMZBS-NEXT:    slli a7, a0, 16
+; RV32IMZBS-NEXT:    slli t0, a0, 17
+; RV32IMZBS-NEXT:    and a7, s10, a7
+; RV32IMZBS-NEXT:    and t0, s6, t0
+; RV32IMZBS-NEXT:    xor a7, a7, t0
+; RV32IMZBS-NEXT:    slli t0, a2, 13
+; RV32IMZBS-NEXT:    srai a3, t0, 31
+; RV32IMZBS-NEXT:    sw a3, 32(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t0, a0, 18
+; RV32IMZBS-NEXT:    and t1, a3, t0
+; RV32IMZBS-NEXT:    slli t0, a2, 12
+; RV32IMZBS-NEXT:    srai s5, t0, 31
+; RV32IMZBS-NEXT:    slli t2, a0, 19
+; RV32IMZBS-NEXT:    xor a7, a7, t1
+; RV32IMZBS-NEXT:    and t1, s5, t2
+; RV32IMZBS-NEXT:    xor a7, a7, t1
+; RV32IMZBS-NEXT:    slli t1, a2, 11
+; RV32IMZBS-NEXT:    srai a3, t1, 31
+; RV32IMZBS-NEXT:    sw a3, 28(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t1, a0, 20
+; RV32IMZBS-NEXT:    and t2, a3, t1
+; RV32IMZBS-NEXT:    slli t1, a2, 10
+; RV32IMZBS-NEXT:    srai t1, t1, 31
+; RV32IMZBS-NEXT:    slli t3, a0, 21
+; RV32IMZBS-NEXT:    xor a7, a7, t2
+; RV32IMZBS-NEXT:    and t2, t1, t3
+; RV32IMZBS-NEXT:    xor a7, a7, t2
+; RV32IMZBS-NEXT:    slli t2, a2, 9
+; RV32IMZBS-NEXT:    srai ra, t2, 31
+; RV32IMZBS-NEXT:    slli t2, a0, 22
+; RV32IMZBS-NEXT:    and t3, ra, t2
+; RV32IMZBS-NEXT:    slli t2, a2, 8
+; RV32IMZBS-NEXT:    srai a3, t2, 31
+; RV32IMZBS-NEXT:    sw a3, 24(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t2, a0, 23
+; RV32IMZBS-NEXT:    and t4, a3, t2
+; RV32IMZBS-NEXT:    slli t2, a2, 7
+; RV32IMZBS-NEXT:    srai t2, t2, 31
+; RV32IMZBS-NEXT:    slli s7, a0, 24
+; RV32IMZBS-NEXT:    xor t3, t3, t4
+; RV32IMZBS-NEXT:    and t4, t2, s7
+; RV32IMZBS-NEXT:    xor t4, t3, t4
+; RV32IMZBS-NEXT:    slli t3, a2, 6
+; RV32IMZBS-NEXT:    srai s7, t3, 31
+; RV32IMZBS-NEXT:    slli t3, a0, 25
+; RV32IMZBS-NEXT:    and s8, s7, t3
+; RV32IMZBS-NEXT:    slli t3, a2, 5
+; RV32IMZBS-NEXT:    srai t3, t3, 31
+; RV32IMZBS-NEXT:    slli a3, a0, 26
+; RV32IMZBS-NEXT:    xor t4, t4, s8
+; RV32IMZBS-NEXT:    and a3, t3, a3
+; RV32IMZBS-NEXT:    xor a5, t4, a3
+; RV32IMZBS-NEXT:    slli t4, a2, 4
+; RV32IMZBS-NEXT:    srai s8, t4, 31
+; RV32IMZBS-NEXT:    slli t4, a0, 27
+; RV32IMZBS-NEXT:    and a4, s8, t4
+; RV32IMZBS-NEXT:    slli t4, a2, 3
+; RV32IMZBS-NEXT:    srai t4, t4, 31
+; RV32IMZBS-NEXT:    slli a3, a0, 28
+; RV32IMZBS-NEXT:    xor a4, a5, a4
+; RV32IMZBS-NEXT:    and a3, t4, a3
+; RV32IMZBS-NEXT:    xor a5, t5, a7
+; RV32IMZBS-NEXT:    xor a3, a4, a3
+; RV32IMZBS-NEXT:    srli a4, a0, 31
+; RV32IMZBS-NEXT:    slli a7, a1, 1
+; RV32IMZBS-NEXT:    xor a3, a5, a3
+; RV32IMZBS-NEXT:    sw a3, 44(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a3, a7, a4
+; RV32IMZBS-NEXT:    and a3, t6, a3
+; RV32IMZBS-NEXT:    slli a4, a2, 2
+; RV32IMZBS-NEXT:    srai t0, a4, 31
+; RV32IMZBS-NEXT:    slli a4, a0, 29
+; RV32IMZBS-NEXT:    and a4, t0, a4
+; RV32IMZBS-NEXT:    slli a5, a2, 1
+; RV32IMZBS-NEXT:    srai a7, a5, 31
+; RV32IMZBS-NEXT:    slli t5, a0, 30
+; RV32IMZBS-NEXT:    and t5, a7, t5
+; RV32IMZBS-NEXT:    and t6, s2, a1
+; RV32IMZBS-NEXT:    xor a5, a4, t5
+; RV32IMZBS-NEXT:    xor a4, t6, a3
+; RV32IMZBS-NEXT:    srli t5, a0, 30
+; RV32IMZBS-NEXT:    slli t6, a1, 2
+; RV32IMZBS-NEXT:    srli s2, a0, 29
+; RV32IMZBS-NEXT:    slli a3, a1, 3
+; RV32IMZBS-NEXT:    or t5, t6, t5
+; RV32IMZBS-NEXT:    or a3, a3, s2
+; RV32IMZBS-NEXT:    lw t6, 20(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t5, t6, t5
+; RV32IMZBS-NEXT:    and a3, s4, a3
+; RV32IMZBS-NEXT:    srli t6, a0, 28
+; RV32IMZBS-NEXT:    slli s2, a1, 4
+; RV32IMZBS-NEXT:    xor a3, t5, a3
+; RV32IMZBS-NEXT:    or t5, s2, t6
+; RV32IMZBS-NEXT:    srli t6, a0, 27
+; RV32IMZBS-NEXT:    slli s2, a1, 5
+; RV32IMZBS-NEXT:    and t5, s11, t5
+; RV32IMZBS-NEXT:    or t6, s2, t6
+; RV32IMZBS-NEXT:    srli s2, a0, 26
+; RV32IMZBS-NEXT:    slli s4, a1, 6
+; RV32IMZBS-NEXT:    lw s11, 40(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t6, s11, t6
+; RV32IMZBS-NEXT:    or s2, s4, s2
+; RV32IMZBS-NEXT:    xor t5, t5, t6
+; RV32IMZBS-NEXT:    and t6, s3, s2
+; RV32IMZBS-NEXT:    xor a3, a4, a3
+; RV32IMZBS-NEXT:    xor a4, t5, t6
+; RV32IMZBS-NEXT:    srli t5, a0, 25
+; RV32IMZBS-NEXT:    slli t6, a1, 7
+; RV32IMZBS-NEXT:    srli s2, a0, 24
+; RV32IMZBS-NEXT:    slli s3, a1, 8
+; RV32IMZBS-NEXT:    or t5, t6, t5
+; RV32IMZBS-NEXT:    or t6, s3, s2
+; RV32IMZBS-NEXT:    lw s2, 16(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t5, s2, t5
+; RV32IMZBS-NEXT:    and t6, s0, t6
+; RV32IMZBS-NEXT:    srli s0, a0, 23
+; RV32IMZBS-NEXT:    slli s2, a1, 9
+; RV32IMZBS-NEXT:    xor t5, t5, t6
+; RV32IMZBS-NEXT:    or t6, s2, s0
+; RV32IMZBS-NEXT:    srli s0, a0, 22
+; RV32IMZBS-NEXT:    slli s2, a1, 10
+; RV32IMZBS-NEXT:    lw s3, 36(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t6, s3, t6
+; RV32IMZBS-NEXT:    or s0, s2, s0
+; RV32IMZBS-NEXT:    xor t5, t5, t6
+; RV32IMZBS-NEXT:    and s0, s1, s0
+; RV32IMZBS-NEXT:    xor a3, a3, a4
+; RV32IMZBS-NEXT:    xor a4, t5, s0
+; RV32IMZBS-NEXT:    srli t5, a0, 21
+; RV32IMZBS-NEXT:    slli t6, a1, 11
+; RV32IMZBS-NEXT:    xor a4, a3, a4
+; RV32IMZBS-NEXT:    or a3, t6, t5
+; RV32IMZBS-NEXT:    srli t5, a0, 20
+; RV32IMZBS-NEXT:    slli t6, a1, 12
+; RV32IMZBS-NEXT:    lw s0, 0(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, s0, a3
+; RV32IMZBS-NEXT:    or t5, t6, t5
+; RV32IMZBS-NEXT:    srli t6, a0, 19
+; RV32IMZBS-NEXT:    slli s0, a1, 13
+; RV32IMZBS-NEXT:    lw s1, 12(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t5, s1, t5
+; RV32IMZBS-NEXT:    or t6, s0, t6
+; RV32IMZBS-NEXT:    xor a3, a3, t5
+; RV32IMZBS-NEXT:    lw t5, 8(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t5, t5, t6
+; RV32IMZBS-NEXT:    srli t6, a0, 18
+; RV32IMZBS-NEXT:    slli s0, a1, 14
+; RV32IMZBS-NEXT:    xor a3, a3, t5
+; RV32IMZBS-NEXT:    or t5, s0, t6
+; RV32IMZBS-NEXT:    srli t6, a0, 17
+; RV32IMZBS-NEXT:    slli s0, a1, 15
+; RV32IMZBS-NEXT:    lw s1, 4(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t5, s1, t5
+; RV32IMZBS-NEXT:    or t6, s0, t6
+; RV32IMZBS-NEXT:    xor t5, a3, t5
+; RV32IMZBS-NEXT:    and t6, s9, t6
+; RV32IMZBS-NEXT:    srai a3, a2, 31
+; RV32IMZBS-NEXT:    slli a2, a0, 31
+; RV32IMZBS-NEXT:    xor t5, t5, t6
+; RV32IMZBS-NEXT:    and a2, a3, a2
+; RV32IMZBS-NEXT:    xor a2, a5, a2
+; RV32IMZBS-NEXT:    xor a4, a4, t5
+; RV32IMZBS-NEXT:    srli a5, a0, 16
+; RV32IMZBS-NEXT:    slli t5, a1, 16
+; RV32IMZBS-NEXT:    srli t6, a0, 15
+; RV32IMZBS-NEXT:    slli s0, a1, 17
+; RV32IMZBS-NEXT:    or a5, t5, a5
+; RV32IMZBS-NEXT:    or t5, s0, t6
+; RV32IMZBS-NEXT:    and a5, s10, a5
+; RV32IMZBS-NEXT:    and t5, s6, t5
+; RV32IMZBS-NEXT:    srli t6, a0, 14
+; RV32IMZBS-NEXT:    slli s0, a1, 18
+; RV32IMZBS-NEXT:    xor a5, a5, t5
+; RV32IMZBS-NEXT:    or t5, s0, t6
+; RV32IMZBS-NEXT:    srli t6, a0, 13
+; RV32IMZBS-NEXT:    slli s0, a1, 19
+; RV32IMZBS-NEXT:    lw s1, 32(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t5, s1, t5
+; RV32IMZBS-NEXT:    or t6, s0, t6
+; RV32IMZBS-NEXT:    xor a5, a5, t5
+; RV32IMZBS-NEXT:    and t5, s5, t6
+; RV32IMZBS-NEXT:    srli t6, a0, 12
+; RV32IMZBS-NEXT:    slli s0, a1, 20
+; RV32IMZBS-NEXT:    xor a5, a5, t5
+; RV32IMZBS-NEXT:    or t5, s0, t6
+; RV32IMZBS-NEXT:    srli t6, a0, 11
+; RV32IMZBS-NEXT:    slli s0, a1, 21
+; RV32IMZBS-NEXT:    lw s1, 28(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t5, s1, t5
+; RV32IMZBS-NEXT:    or t6, s0, t6
+; RV32IMZBS-NEXT:    xor a5, a5, t5
+; RV32IMZBS-NEXT:    and t1, t1, t6
+; RV32IMZBS-NEXT:    srli t5, a0, 10
+; RV32IMZBS-NEXT:    slli t6, a1, 22
+; RV32IMZBS-NEXT:    xor a5, a5, t1
+; RV32IMZBS-NEXT:    or t1, t6, t5
+; RV32IMZBS-NEXT:    srli t5, a0, 9
+; RV32IMZBS-NEXT:    slli t6, a1, 23
+; RV32IMZBS-NEXT:    and t1, ra, t1
+; RV32IMZBS-NEXT:    or t5, t6, t5
+; RV32IMZBS-NEXT:    srli t6, a0, 8
+; RV32IMZBS-NEXT:    slli s0, a1, 24
+; RV32IMZBS-NEXT:    lw s1, 24(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t5, s1, t5
+; RV32IMZBS-NEXT:    or t6, s0, t6
+; RV32IMZBS-NEXT:    xor t1, t1, t5
+; RV32IMZBS-NEXT:    and t2, t2, t6
+; RV32IMZBS-NEXT:    srli t5, a0, 7
+; RV32IMZBS-NEXT:    slli t6, a1, 25
+; RV32IMZBS-NEXT:    xor t1, t1, t2
+; RV32IMZBS-NEXT:    or t2, t6, t5
+; RV32IMZBS-NEXT:    srli t5, a0, 6
+; RV32IMZBS-NEXT:    slli t6, a1, 26
+; RV32IMZBS-NEXT:    and t2, s7, t2
+; RV32IMZBS-NEXT:    or t5, t6, t5
+; RV32IMZBS-NEXT:    xor t1, t1, t2
+; RV32IMZBS-NEXT:    and t2, t3, t5
+; RV32IMZBS-NEXT:    srli t3, a0, 5
+; RV32IMZBS-NEXT:    slli t5, a1, 27
+; RV32IMZBS-NEXT:    xor t1, t1, t2
+; RV32IMZBS-NEXT:    or t2, t5, t3
+; RV32IMZBS-NEXT:    srli t3, a0, 4
+; RV32IMZBS-NEXT:    slli t5, a1, 28
+; RV32IMZBS-NEXT:    and t2, s8, t2
+; RV32IMZBS-NEXT:    or t3, t5, t3
+; RV32IMZBS-NEXT:    xor t1, t1, t2
+; RV32IMZBS-NEXT:    and t2, t4, t3
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    xor a5, t1, t2
+; RV32IMZBS-NEXT:    srli t1, a0, 3
+; RV32IMZBS-NEXT:    slli t2, a1, 29
+; RV32IMZBS-NEXT:    srli t3, a0, 2
+; RV32IMZBS-NEXT:    slli t4, a1, 30
+; RV32IMZBS-NEXT:    or t1, t2, t1
+; RV32IMZBS-NEXT:    or t2, t4, t3
+; RV32IMZBS-NEXT:    and t0, t0, t1
+; RV32IMZBS-NEXT:    and a7, a7, t2
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    xor a5, t0, a7
+; RV32IMZBS-NEXT:    srli a7, a0, 1
+; RV32IMZBS-NEXT:    slli a1, a1, 31
+; RV32IMZBS-NEXT:    or a1, a1, a7
+; RV32IMZBS-NEXT:    slli a7, a6, 31
+; RV32IMZBS-NEXT:    and a1, a3, a1
+; RV32IMZBS-NEXT:    srai a3, a7, 31
+; RV32IMZBS-NEXT:    xor a1, a5, a1
+; RV32IMZBS-NEXT:    and a0, a3, a0
+; RV32IMZBS-NEXT:    xor a0, a1, a0
+; RV32IMZBS-NEXT:    slli a1, a6, 30
+; RV32IMZBS-NEXT:    srai a1, a1, 31
+; RV32IMZBS-NEXT:    slli a3, a6, 29
+; RV32IMZBS-NEXT:    lw a5, 104(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, a5
+; RV32IMZBS-NEXT:    srai a3, a3, 31
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    lw a1, 84(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a3, a1
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    slli a1, a6, 28
+; RV32IMZBS-NEXT:    srai a1, a1, 31
+; RV32IMZBS-NEXT:    slli a3, a6, 27
+; RV32IMZBS-NEXT:    lw a5, 100(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, a5
+; RV32IMZBS-NEXT:    srai a3, a3, 31
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    lw a1, 88(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a3, a1
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    slli a1, a6, 26
+; RV32IMZBS-NEXT:    xor a0, a4, a0
+; RV32IMZBS-NEXT:    srai a1, a1, 31
+; RV32IMZBS-NEXT:    lw a3, 96(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, a3
+; RV32IMZBS-NEXT:    slli a3, a6, 25
+; RV32IMZBS-NEXT:    srai a3, a3, 31
+; RV32IMZBS-NEXT:    slli a4, a6, 24
+; RV32IMZBS-NEXT:    lw a5, 92(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a3, a5
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    lw a3, 68(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a4, a3
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    slli a3, a6, 23
+; RV32IMZBS-NEXT:    srai a3, a3, 31
+; RV32IMZBS-NEXT:    slli a4, a6, 22
+; RV32IMZBS-NEXT:    lw a5, 80(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a3, a5
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    lw a3, 72(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a4, a3
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    slli a3, a6, 21
+; RV32IMZBS-NEXT:    srai a3, a3, 31
+; RV32IMZBS-NEXT:    slli a4, a6, 20
+; RV32IMZBS-NEXT:    lw a5, 76(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a3, a5
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    lw a3, 48(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a4, a3
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    slli a3, a6, 19
+; RV32IMZBS-NEXT:    srai a3, a3, 31
+; RV32IMZBS-NEXT:    slli a4, a6, 18
+; RV32IMZBS-NEXT:    lw a5, 60(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a3, a5
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    lw a3, 64(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a4, a3
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    not a3, a6
+; RV32IMZBS-NEXT:    slli a6, a6, 17
+; RV32IMZBS-NEXT:    bexti a3, a3, 15
+; RV32IMZBS-NEXT:    srai a4, a6, 31
+; RV32IMZBS-NEXT:    addi a3, a3, -1
+; RV32IMZBS-NEXT:    lw a5, 56(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    lw a5, 52(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a3, a5
+; RV32IMZBS-NEXT:    xor a1, a0, a1
+; RV32IMZBS-NEXT:    xor a3, a4, a3
+; RV32IMZBS-NEXT:    lw a0, 44(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    xor a0, a0, a2
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    lw ra, 156(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s0, 152(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s1, 148(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s2, 144(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s3, 140(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s4, 136(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s5, 132(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s6, 128(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s7, 124(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s8, 120(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s9, 116(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s10, 112(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s11, 108(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    addi sp, sp, 160
+; RV32IMZBS-NEXT:    ret
+;
+; RV64IMZBS-LABEL: clmul_i48:
+; RV64IMZBS:       # %bb.0:
+; RV64IMZBS-NEXT:    addi sp, sp, -64
+; RV64IMZBS-NEXT:    sd s0, 56(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s1, 48(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s2, 40(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s3, 32(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s4, 24(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s5, 16(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s6, 8(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    lui a2, %hi(.LCPI4_0)
+; RV64IMZBS-NEXT:    lui a3, %hi(.LCPI4_1)
+; RV64IMZBS-NEXT:    ld a2, %lo(.LCPI4_0)(a2)
+; RV64IMZBS-NEXT:    ld a3, %lo(.LCPI4_1)(a3)
+; RV64IMZBS-NEXT:    lui a4, %hi(.LCPI4_2)
+; RV64IMZBS-NEXT:    lui a5, %hi(.LCPI4_3)
+; RV64IMZBS-NEXT:    ld a4, %lo(.LCPI4_2)(a4)
+; RV64IMZBS-NEXT:    ld a5, %lo(.LCPI4_3)(a5)
+; RV64IMZBS-NEXT:    and a6, a1, a2
+; RV64IMZBS-NEXT:    and a7, a0, a3
+; RV64IMZBS-NEXT:    and t0, a1, a3
+; RV64IMZBS-NEXT:    and t1, a0, a2
+; RV64IMZBS-NEXT:    mul t2, a7, a6
+; RV64IMZBS-NEXT:    mul t3, t1, t0
+; RV64IMZBS-NEXT:    and t4, a1, a4
+; RV64IMZBS-NEXT:    and t5, a0, a5
+; RV64IMZBS-NEXT:    and a1, a1, a5
+; RV64IMZBS-NEXT:    and a0, a0, a4
+; RV64IMZBS-NEXT:    mul t6, t5, t4
+; RV64IMZBS-NEXT:    mul s0, a0, a1
+; RV64IMZBS-NEXT:    mul s1, a7, t4
+; RV64IMZBS-NEXT:    mul s2, t1, a6
+; RV64IMZBS-NEXT:    mul s3, t5, a1
+; RV64IMZBS-NEXT:    mul s4, a0, t0
+; RV64IMZBS-NEXT:    mul s5, a7, t0
+; RV64IMZBS-NEXT:    mul s6, t1, a1
+; RV64IMZBS-NEXT:    mul a1, a7, a1
+; RV64IMZBS-NEXT:    mul a7, t5, a6
+; RV64IMZBS-NEXT:    mul t1, t1, t4
+; RV64IMZBS-NEXT:    mul t4, a0, t4
+; RV64IMZBS-NEXT:    mul t0, t5, t0
+; RV64IMZBS-NEXT:    mul a0, a0, a6
+; RV64IMZBS-NEXT:    xor a6, t3, t2
+; RV64IMZBS-NEXT:    xor t2, t6, s0
+; RV64IMZBS-NEXT:    xor t3, s2, s1
+; RV64IMZBS-NEXT:    xor t5, s3, s4
+; RV64IMZBS-NEXT:    xor a6, a6, t2
+; RV64IMZBS-NEXT:    xor t2, t3, t5
+; RV64IMZBS-NEXT:    and a3, a6, a3
+; RV64IMZBS-NEXT:    and a2, t2, a2
+; RV64IMZBS-NEXT:    xor a6, s6, s5
+; RV64IMZBS-NEXT:    xor a7, a7, t4
+; RV64IMZBS-NEXT:    xor a1, t1, a1
+; RV64IMZBS-NEXT:    xor a0, t0, a0
+; RV64IMZBS-NEXT:    xor a6, a6, a7
+; RV64IMZBS-NEXT:    xor a0, a1, a0
+; RV64IMZBS-NEXT:    and a1, a6, a5
+; RV64IMZBS-NEXT:    and a0, a0, a4
+; RV64IMZBS-NEXT:    or a2, a2, a3
+; RV64IMZBS-NEXT:    or a0, a1, a0
+; RV64IMZBS-NEXT:    or a0, a2, a0
+; RV64IMZBS-NEXT:    ld s0, 56(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s1, 48(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s2, 40(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s3, 32(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s4, 24(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s5, 16(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s6, 8(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    addi sp, sp, 64
+; RV64IMZBS-NEXT:    ret
+;
+; RV32IMZBC-LABEL: clmul_i48:
+; RV32IMZBC:       # %bb.0:
+; RV32IMZBC-NEXT:    clmul a1, a1, a2
+; RV32IMZBC-NEXT:    clmul a3, a0, a3
+; RV32IMZBC-NEXT:    clmulh a4, a0, a2
+; RV32IMZBC-NEXT:    xor a1, a3, a1
+; RV32IMZBC-NEXT:    clmul a0, a0, a2
+; RV32IMZBC-NEXT:    xor a1, a4, a1
+; RV32IMZBC-NEXT:    ret
+;
+; RV64IMZBC-LABEL: clmul_i48:
+; RV64IMZBC:       # %bb.0:
+; RV64IMZBC-NEXT:    clmul a0, a0, a1
+; RV64IMZBC-NEXT:    ret
+  %res = call i48 @llvm.clmul.i48(i48 %a, i48 %b)
+  ret i48 %res
+}
+
+define i64 @clmul_i64(i64 %a, i64 %b) nounwind {
+; RV32I-LABEL: clmul_i64:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi sp, sp, -240
+; RV32I-NEXT:    sw ra, 236(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s0, 232(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s1, 228(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s2, 224(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s3, 220(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s4, 216(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s5, 212(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s6, 208(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s7, 204(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s8, 200(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s9, 196(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s10, 192(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s11, 188(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    mv s3, a3
+; RV32I-NEXT:    mv a3, a0
+; RV32I-NEXT:    lui a0, 16
+; RV32I-NEXT:    srli a5, a3, 8
+; RV32I-NEXT:    addi a4, a0, -256
+; RV32I-NEXT:    lui s4, 16
+; RV32I-NEXT:    and a5, a5, a4
+; RV32I-NEXT:    srli a6, a3, 24
+; RV32I-NEXT:    and a7, a3, a4
+; RV32I-NEXT:    slli a7, a7, 8
+; RV32I-NEXT:    slli a0, a3, 24
+; RV32I-NEXT:    sw a0, 180(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    or a5, a5, a6
+; RV32I-NEXT:    or a6, a0, a7
+; RV32I-NEXT:    or a6, a6, a5
+; RV32I-NEXT:    lui a5, 61681
+; RV32I-NEXT:    srli a7, a6, 4
+; RV32I-NEXT:    addi a5, a5, -241
+; RV32I-NEXT:    and a7, a7, a5
+; RV32I-NEXT:    and a6, a6, a5
+; RV32I-NEXT:    slli a6, a6, 4
+; RV32I-NEXT:    lui t0, 209715
+; RV32I-NEXT:    or a7, a7, a6
+; RV32I-NEXT:    addi a6, t0, 819
+; RV32I-NEXT:    srli t0, a7, 2
+; RV32I-NEXT:    and a7, a7, a6
+; RV32I-NEXT:    and t0, t0, a6
+; RV32I-NEXT:    slli a7, a7, 2
+; RV32I-NEXT:    or t1, t0, a7
+; RV32I-NEXT:    srli t2, t1, 1
+; RV32I-NEXT:    lui a7, 349525
+; RV32I-NEXT:    addi t0, a7, 1365
+; RV32I-NEXT:    srli t3, a2, 8
+; RV32I-NEXT:    and t2, t2, t0
+; RV32I-NEXT:    and t3, t3, a4
+; RV32I-NEXT:    srli t4, a2, 24
+; RV32I-NEXT:    and t5, a2, a4
+; RV32I-NEXT:    slli t5, t5, 8
+; RV32I-NEXT:    slli t6, a2, 24
+; RV32I-NEXT:    or t3, t3, t4
+; RV32I-NEXT:    or t4, t6, t5
+; RV32I-NEXT:    and t1, t1, t0
+; RV32I-NEXT:    or t3, t4, t3
+; RV32I-NEXT:    srli t4, t3, 4
+; RV32I-NEXT:    and t3, t3, a5
+; RV32I-NEXT:    and t4, t4, a5
+; RV32I-NEXT:    slli t3, t3, 4
+; RV32I-NEXT:    slli t1, t1, 1
+; RV32I-NEXT:    or t3, t4, t3
+; RV32I-NEXT:    srli t4, t3, 2
+; RV32I-NEXT:    and t3, t3, a6
+; RV32I-NEXT:    and t4, t4, a6
+; RV32I-NEXT:    slli t3, t3, 2
+; RV32I-NEXT:    or t1, t2, t1
+; RV32I-NEXT:    or t2, t4, t3
+; RV32I-NEXT:    srli t3, t2, 1
+; RV32I-NEXT:    and t2, t2, t0
+; RV32I-NEXT:    and t3, t3, t0
+; RV32I-NEXT:    slli t2, t2, 1
+; RV32I-NEXT:    slli t4, t1, 1
+; RV32I-NEXT:    or t3, t3, t2
+; RV32I-NEXT:    andi t5, t3, 2
+; RV32I-NEXT:    andi t6, t3, 1
+; RV32I-NEXT:    seqz t5, t5
+; RV32I-NEXT:    seqz t6, t6
+; RV32I-NEXT:    addi t5, t5, -1
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    and t4, t5, t4
+; RV32I-NEXT:    and t5, t6, t1
+; RV32I-NEXT:    slli t6, t1, 2
+; RV32I-NEXT:    andi s0, t3, 4
+; RV32I-NEXT:    seqz s0, s0
+; RV32I-NEXT:    andi s1, t3, 8
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    seqz s1, s1
+; RV32I-NEXT:    slli s2, t1, 3
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    and t6, s0, t6
+; RV32I-NEXT:    and s0, s1, s2
+; RV32I-NEXT:    xor t4, t5, t4
+; RV32I-NEXT:    xor t5, t6, s0
+; RV32I-NEXT:    xor t4, t4, t5
+; RV32I-NEXT:    andi t5, t3, 16
+; RV32I-NEXT:    slli t6, t1, 4
+; RV32I-NEXT:    seqz t5, t5
+; RV32I-NEXT:    addi t5, t5, -1
+; RV32I-NEXT:    andi s0, t3, 32
+; RV32I-NEXT:    and t5, t5, t6
+; RV32I-NEXT:    seqz t6, s0
+; RV32I-NEXT:    slli s0, t1, 5
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    and t6, t6, s0
+; RV32I-NEXT:    andi s0, t3, 64
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    seqz t6, s0
+; RV32I-NEXT:    slli s0, t1, 6
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    and t6, t6, s0
+; RV32I-NEXT:    andi s0, t3, 128
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    seqz t6, s0
+; RV32I-NEXT:    slli s0, t1, 7
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    and t6, t6, s0
+; RV32I-NEXT:    andi s0, t3, 256
+; RV32I-NEXT:    slli s1, t1, 8
+; RV32I-NEXT:    seqz s0, s0
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    andi s2, t3, 512
+; RV32I-NEXT:    and s0, s0, s1
+; RV32I-NEXT:    seqz s1, s2
+; RV32I-NEXT:    slli s2, t1, 9
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    xor t6, t6, s0
+; RV32I-NEXT:    and s0, s1, s2
+; RV32I-NEXT:    xor t4, t4, t5
+; RV32I-NEXT:    xor t5, t6, s0
+; RV32I-NEXT:    slli t6, t1, 10
+; RV32I-NEXT:    andi s0, t3, 1024
+; RV32I-NEXT:    seqz s0, s0
+; RV32I-NEXT:    li s1, 1
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    slli s5, s1, 11
+; RV32I-NEXT:    and t6, s0, t6
+; RV32I-NEXT:    and s0, t3, s5
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    seqz t6, s0
+; RV32I-NEXT:    slli s0, t1, 11
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    and t6, t6, s0
+; RV32I-NEXT:    lui s6, 1
+; RV32I-NEXT:    slli s0, t1, 12
+; RV32I-NEXT:    and s1, t3, s6
+; RV32I-NEXT:    seqz s1, s1
+; RV32I-NEXT:    lui s7, 2
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    and s2, t3, s7
+; RV32I-NEXT:    and s0, s1, s0
+; RV32I-NEXT:    seqz s1, s2
+; RV32I-NEXT:    slli s2, t1, 13
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    xor t6, t6, s0
+; RV32I-NEXT:    and s0, s1, s2
+; RV32I-NEXT:    xor t6, t6, s0
+; RV32I-NEXT:    lui s8, 4
+; RV32I-NEXT:    slli s0, t1, 14
+; RV32I-NEXT:    and s1, t3, s8
+; RV32I-NEXT:    seqz s1, s1
+; RV32I-NEXT:    lui a0, 8
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    and s2, t3, a0
+; RV32I-NEXT:    and s0, s1, s0
+; RV32I-NEXT:    seqz s1, s2
+; RV32I-NEXT:    slli s2, t1, 15
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    xor t6, t6, s0
+; RV32I-NEXT:    and s0, s1, s2
+; RV32I-NEXT:    xor t4, t4, t5
+; RV32I-NEXT:    xor t5, t6, s0
+; RV32I-NEXT:    xor t4, t4, t5
+; RV32I-NEXT:    slli t5, t1, 16
+; RV32I-NEXT:    and t6, t3, s4
+; RV32I-NEXT:    lui s2, 32
+; RV32I-NEXT:    seqz t6, t6
+; RV32I-NEXT:    and s0, t3, s2
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    seqz s0, s0
+; RV32I-NEXT:    slli s1, t1, 17
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    and t5, t6, t5
+; RV32I-NEXT:    and s0, s0, s1
+; RV32I-NEXT:    xor t5, t5, s0
+; RV32I-NEXT:    lui a0, 64
+; RV32I-NEXT:    slli t6, t1, 18
+; RV32I-NEXT:    and s0, t3, a0
+; RV32I-NEXT:    seqz s0, s0
+; RV32I-NEXT:    lui a0, 128
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    and s1, t3, a0
+; RV32I-NEXT:    and t6, s0, t6
+; RV32I-NEXT:    seqz s0, s1
+; RV32I-NEXT:    slli s1, t1, 19
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    and s0, s0, s1
+; RV32I-NEXT:    xor t5, t5, s0
+; RV32I-NEXT:    lui a0, 256
+; RV32I-NEXT:    slli t6, t1, 20
+; RV32I-NEXT:    and s0, t3, a0
+; RV32I-NEXT:    seqz s0, s0
+; RV32I-NEXT:    lui s1, 512
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    and s1, t3, s1
+; RV32I-NEXT:    and t6, s0, t6
+; RV32I-NEXT:    seqz s0, s1
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    slli s1, t1, 21
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    and s0, s0, s1
+; RV32I-NEXT:    xor t5, t5, s0
+; RV32I-NEXT:    lui a0, 1024
+; RV32I-NEXT:    xor t4, t4, t5
+; RV32I-NEXT:    and t5, t3, a0
+; RV32I-NEXT:    seqz t5, t5
+; RV32I-NEXT:    lui a0, 2048
+; RV32I-NEXT:    addi t5, t5, -1
+; RV32I-NEXT:    and t6, t3, a0
+; RV32I-NEXT:    seqz t6, t6
+; RV32I-NEXT:    slli s0, t1, 22
+; RV32I-NEXT:    and t5, t5, s0
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    slli s0, t1, 23
+; RV32I-NEXT:    lui a0, 4096
+; RV32I-NEXT:    and t6, t6, s0
+; RV32I-NEXT:    and s0, t3, a0
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    seqz t6, s0
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    lui s1, 8192
+; RV32I-NEXT:    slli s0, t1, 24
+; RV32I-NEXT:    and s1, t3, s1
+; RV32I-NEXT:    lui s9, 8192
+; RV32I-NEXT:    and t6, t6, s0
+; RV32I-NEXT:    seqz s0, s1
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    slli t6, t1, 25
+; RV32I-NEXT:    lui s1, 16384
+; RV32I-NEXT:    and t6, s0, t6
+; RV32I-NEXT:    and s0, t3, s1
+; RV32I-NEXT:    lui s10, 16384
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    seqz t6, s0
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    lui s1, 32768
+; RV32I-NEXT:    slli s0, t1, 26
+; RV32I-NEXT:    and s1, t3, s1
+; RV32I-NEXT:    lui s11, 32768
+; RV32I-NEXT:    and t6, t6, s0
+; RV32I-NEXT:    seqz s0, s1
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    slli t6, t1, 27
+; RV32I-NEXT:    lui s1, 65536
+; RV32I-NEXT:    and t6, s0, t6
+; RV32I-NEXT:    and s0, t3, s1
+; RV32I-NEXT:    lui s4, 65536
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    seqz t6, s0
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    slli s0, t1, 28
+; RV32I-NEXT:    and t6, t6, s0
+; RV32I-NEXT:    lui s0, 131072
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    and t6, t3, s0
+; RV32I-NEXT:    lui s1, 131072
+; RV32I-NEXT:    seqz t6, t6
+; RV32I-NEXT:    lui s0, 262144
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    and t3, t3, s0
+; RV32I-NEXT:    lui ra, 262144
+; RV32I-NEXT:    seqz t3, t3
+; RV32I-NEXT:    slli s0, t1, 29
+; RV32I-NEXT:    and t6, t6, s0
+; RV32I-NEXT:    addi t3, t3, -1
+; RV32I-NEXT:    srli t2, t2, 31
+; RV32I-NEXT:    slli s0, t1, 30
+; RV32I-NEXT:    and t3, t3, s0
+; RV32I-NEXT:    seqz t2, t2
+; RV32I-NEXT:    slli t1, t1, 31
+; RV32I-NEXT:    addi t2, t2, -1
+; RV32I-NEXT:    xor t3, t6, t3
+; RV32I-NEXT:    and t1, t2, t1
+; RV32I-NEXT:    xor t2, t4, t5
+; RV32I-NEXT:    xor t1, t3, t1
+; RV32I-NEXT:    xor t1, t2, t1
+; RV32I-NEXT:    srli t2, t1, 8
+; RV32I-NEXT:    and t2, t2, a4
+; RV32I-NEXT:    and a4, t1, a4
+; RV32I-NEXT:    srli t3, t1, 24
+; RV32I-NEXT:    slli t1, t1, 24
+; RV32I-NEXT:    slli a4, a4, 8
+; RV32I-NEXT:    or t2, t2, t3
+; RV32I-NEXT:    or a4, t1, a4
+; RV32I-NEXT:    or a4, a4, t2
+; RV32I-NEXT:    srli t1, a4, 4
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    and a5, t1, a5
+; RV32I-NEXT:    slli a4, a4, 4
+; RV32I-NEXT:    or a4, a5, a4
+; RV32I-NEXT:    srli a5, a4, 2
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    slli a4, a4, 2
+; RV32I-NEXT:    or a4, a5, a4
+; RV32I-NEXT:    and a5, a4, t0
+; RV32I-NEXT:    srli a4, a4, 1
+; RV32I-NEXT:    addi a6, a7, 1364
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    slli a5, a5, 1
+; RV32I-NEXT:    or a4, a4, a5
+; RV32I-NEXT:    sw a4, 184(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a4, a2, 2
+; RV32I-NEXT:    slli a5, a1, 1
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    addi a6, a4, -1
+; RV32I-NEXT:    sw a6, 176(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a4, a2, 1
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    addi a6, a4, -1
+; RV32I-NEXT:    sw a6, 172(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a4, a2, 4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a6, a6, a1
+; RV32I-NEXT:    xor a5, a6, a5
+; RV32I-NEXT:    addi a7, a4, -1
+; RV32I-NEXT:    sw a7, 168(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, a1, 2
+; RV32I-NEXT:    andi a6, a2, 8
+; RV32I-NEXT:    and a4, a7, a4
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 164(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 3
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    andi a7, a2, 16
+; RV32I-NEXT:    xor a4, a4, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    xor a4, a5, a4
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 160(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a1, 4
+; RV32I-NEXT:    andi a6, a2, 32
+; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 156(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 5
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    andi a7, a2, 64
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 152(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 6
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    andi a7, a2, 128
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 148(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a1, 7
+; RV32I-NEXT:    andi a6, a2, 256
+; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 144(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 8
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    andi a7, a2, 512
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 140(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 9
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    andi a7, a2, 1024
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 136(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 10
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    and a7, a2, s5
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 132(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a1, 11
+; RV32I-NEXT:    and a6, a2, s6
+; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 128(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 12
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    and a7, a2, s7
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 124(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 13
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    and a7, a2, s8
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 120(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a1, 14
+; RV32I-NEXT:    lui t0, 8
+; RV32I-NEXT:    and a6, a2, t0
+; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 116(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 15
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    lui t6, 16
+; RV32I-NEXT:    and a7, a2, t6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 112(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 16
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    and a7, a2, s2
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 108(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 17
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    lui t1, 64
+; RV32I-NEXT:    and a7, a2, t1
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 104(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 18
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    lui t2, 128
+; RV32I-NEXT:    and a7, a2, t2
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 100(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a1, 19
+; RV32I-NEXT:    lui t3, 256
+; RV32I-NEXT:    and a6, a2, t3
+; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 96(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 20
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    lui t4, 512
+; RV32I-NEXT:    and a7, a2, t4
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 92(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 21
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    lui t5, 1024
+; RV32I-NEXT:    and a7, a2, t5
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 88(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 22
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    lui s0, 2048
+; RV32I-NEXT:    and a7, a2, s0
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 84(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 23
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    and a7, a2, a0
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a7, a6, -1
+; RV32I-NEXT:    sw a7, 80(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a1, 24
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    and a7, a2, s9
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    sw a4, 48(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    sw a6, 76(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, a1, 25
+; RV32I-NEXT:    and a5, a2, s10
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a6, a5, -1
+; RV32I-NEXT:    sw a6, 72(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a1, 26
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    and a6, a2, s11
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a6, a5, -1
+; RV32I-NEXT:    sw a6, 68(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a1, 27
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    and a6, a2, s4
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a6, a5, -1
+; RV32I-NEXT:    sw a6, 64(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a1, 28
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    and a6, a2, s1
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a6, a5, -1
+; RV32I-NEXT:    sw a6, 60(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a1, 29
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    and a6, a2, ra
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a6, a5, -1
+; RV32I-NEXT:    sw a6, 56(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a1, 30
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    srli a2, a2, 31
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    slli a1, a1, 31
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    sw a2, 52(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a2, a1
+; RV32I-NEXT:    andi a2, s3, 2
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    andi a5, s3, 1
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a6, a3, 1
+; RV32I-NEXT:    sw a6, 44(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a6
+; RV32I-NEXT:    and a5, a5, a3
+; RV32I-NEXT:    xor a1, a4, a1
+; RV32I-NEXT:    sw a1, 40(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a5, a2
+; RV32I-NEXT:    andi a1, s3, 4
+; RV32I-NEXT:    andi a4, s3, 8
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    slli a5, a3, 2
+; RV32I-NEXT:    sw a5, 36(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a3, 3
+; RV32I-NEXT:    sw a6, 32(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    andi a4, s3, 16
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    seqz a2, a4
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    andi a4, s3, 32
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    slli a5, a3, 4
+; RV32I-NEXT:    sw a5, 28(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    slli a6, a3, 5
+; RV32I-NEXT:    sw a6, 24(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a5, s3, 64
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a6, a3, 6
+; RV32I-NEXT:    sw a6, 20(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a5, a6
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    andi a4, s3, 128
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a4
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    andi a4, s3, 256
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    slli a5, a3, 7
+; RV32I-NEXT:    sw a5, 16(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    slli a6, a3, 8
+; RV32I-NEXT:    sw a6, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a5, s3, 512
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a6, a3, 9
+; RV32I-NEXT:    sw a6, 8(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a4, s3, 1024
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    slli a6, a3, 10
+; RV32I-NEXT:    sw a6, 4(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, s3, s5
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a4
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    and a4, s3, s6
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    slli s11, a3, 11
+; RV32I-NEXT:    and a2, a2, s11
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    and a5, s3, s7
+; RV32I-NEXT:    slli s10, a3, 12
+; RV32I-NEXT:    and a4, a4, s10
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli s9, a3, 13
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a5, s9
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, s3, s8
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a4
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    and a4, s3, t0
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    slli s8, a3, 14
+; RV32I-NEXT:    and a2, a2, s8
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    and a5, s3, t6
+; RV32I-NEXT:    slli s7, a3, 15
+; RV32I-NEXT:    and a4, a4, s7
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    and a4, s3, s2
+; RV32I-NEXT:    slli s6, a3, 16
+; RV32I-NEXT:    and a5, a5, s6
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    and a5, s3, t1
+; RV32I-NEXT:    slli s5, a3, 17
+; RV32I-NEXT:    and a4, a4, s5
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli s1, a3, 18
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a5, s1
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, s3, t2
+; RV32I-NEXT:    xor s2, a1, a2
+; RV32I-NEXT:    seqz a1, a4
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    and a2, s3, t3
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    slli s4, a3, 19
+; RV32I-NEXT:    and a1, a1, s4
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    and a4, s3, t4
+; RV32I-NEXT:    slli t6, a3, 20
+; RV32I-NEXT:    and a2, a2, t6
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    and a2, s3, t5
+; RV32I-NEXT:    slli t5, a3, 21
+; RV32I-NEXT:    and a4, a4, t5
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    and a4, s3, s0
+; RV32I-NEXT:    slli t4, a3, 22
+; RV32I-NEXT:    and a2, a2, t4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    and a2, s3, a0
+; RV32I-NEXT:    slli t3, a3, 23
+; RV32I-NEXT:    and a4, a4, t3
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    lw s0, 180(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a2, s0
+; RV32I-NEXT:    lui a0, 8192
+; RV32I-NEXT:    and a2, s3, a0
+; RV32I-NEXT:    xor a7, a1, a5
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    lui a0, 16384
+; RV32I-NEXT:    and a2, s3, a0
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    slli t2, a3, 25
+; RV32I-NEXT:    and a1, a1, t2
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    lui a0, 32768
+; RV32I-NEXT:    and a4, s3, a0
+; RV32I-NEXT:    slli t1, a3, 26
+; RV32I-NEXT:    and a2, a2, t1
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    lui a0, 65536
+; RV32I-NEXT:    and a2, s3, a0
+; RV32I-NEXT:    slli t0, a3, 27
+; RV32I-NEXT:    and a4, a4, t0
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    lui a0, 131072
+; RV32I-NEXT:    and a0, s3, a0
+; RV32I-NEXT:    slli a6, a3, 28
+; RV32I-NEXT:    and a2, a2, a6
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    xor a2, a1, a2
+; RV32I-NEXT:    addi a1, a0, -1
+; RV32I-NEXT:    lui a0, 262144
+; RV32I-NEXT:    and a0, s3, a0
+; RV32I-NEXT:    slli a5, a3, 29
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    xor a2, a2, a1
+; RV32I-NEXT:    addi a1, a0, -1
+; RV32I-NEXT:    srli s3, s3, 31
+; RV32I-NEXT:    slli a4, a3, 30
+; RV32I-NEXT:    and a0, a1, a4
+; RV32I-NEXT:    seqz a1, s3
+; RV32I-NEXT:    addi s3, a1, -1
+; RV32I-NEXT:    slli a1, a3, 31
+; RV32I-NEXT:    xor a0, a2, a0
+; RV32I-NEXT:    and a2, s3, a1
+; RV32I-NEXT:    xor a7, s2, a7
+; RV32I-NEXT:    xor a0, a0, a2
+; RV32I-NEXT:    lw a2, 48(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s2, 40(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    xor s2, a2, s2
+; RV32I-NEXT:    xor a2, a7, a0
+; RV32I-NEXT:    lw a0, 176(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 44(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a0, a7
+; RV32I-NEXT:    lw a7, 172(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a7, a3
+; RV32I-NEXT:    lw a7, 168(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 36(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a7, s3
+; RV32I-NEXT:    lw s3, 164(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw ra, 32(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, ra
+; RV32I-NEXT:    xor a0, a3, a0
+; RV32I-NEXT:    xor a3, a7, s3
+; RV32I-NEXT:    lw a7, 160(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 28(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a7, s3
+; RV32I-NEXT:    lw s3, 156(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw ra, 24(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, ra
+; RV32I-NEXT:    xor a7, a7, s3
+; RV32I-NEXT:    lw s3, 152(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw ra, 20(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, ra
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    xor a3, a7, s3
+; RV32I-NEXT:    lw a7, 148(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 16(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a7, s3
+; RV32I-NEXT:    lw s3, 144(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw ra, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, ra
+; RV32I-NEXT:    xor a7, a7, s3
+; RV32I-NEXT:    lw s3, 140(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw ra, 8(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, ra
+; RV32I-NEXT:    xor a7, a7, s3
+; RV32I-NEXT:    lw s3, 136(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw ra, 4(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, ra
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    xor a3, a7, s3
+; RV32I-NEXT:    lw a7, 132(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a7, s11
+; RV32I-NEXT:    lw s3, 128(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, s10
+; RV32I-NEXT:    xor a7, a7, s3
+; RV32I-NEXT:    lw s3, 124(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, s9
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    xor a3, a7, s3
+; RV32I-NEXT:    lw a7, 120(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a7, s8
+; RV32I-NEXT:    lw s3, 116(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, s7
+; RV32I-NEXT:    xor a7, a7, s3
+; RV32I-NEXT:    lw s3, 112(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, s6
+; RV32I-NEXT:    xor a7, a7, s3
+; RV32I-NEXT:    lw s3, 108(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s3, s3, s5
+; RV32I-NEXT:    xor a7, a7, s3
+; RV32I-NEXT:    lw s3, 104(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s1, s3, s1
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    xor a3, a7, s1
+; RV32I-NEXT:    lw a7, 100(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a7, s4
+; RV32I-NEXT:    lw s1, 96(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t6, s1, t6
+; RV32I-NEXT:    xor a7, a7, t6
+; RV32I-NEXT:    lw t6, 92(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t5, t6, t5
+; RV32I-NEXT:    xor a7, a7, t5
+; RV32I-NEXT:    lw t5, 88(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, t5, t4
+; RV32I-NEXT:    xor a7, a7, t4
+; RV32I-NEXT:    lw t4, 84(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t3, t4, t3
+; RV32I-NEXT:    xor a7, a7, t3
+; RV32I-NEXT:    lw t3, 80(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t3, t3, s0
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    xor a3, a7, t3
+; RV32I-NEXT:    xor a2, a2, s2
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    lw a3, 76(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, t2
+; RV32I-NEXT:    lw a7, 72(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a7, t1
+; RV32I-NEXT:    xor a3, a3, a7
+; RV32I-NEXT:    lw a7, 68(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a7, t0
+; RV32I-NEXT:    xor a3, a3, a7
+; RV32I-NEXT:    lw a7, 64(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a3, a3, a6
+; RV32I-NEXT:    lw a6, 60(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 56(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    lw a4, 52(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a4, a1
+; RV32I-NEXT:    xor a3, a3, a1
+; RV32I-NEXT:    lw a1, 184(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    srli a1, a1, 1
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    lw ra, 236(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s0, 232(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s1, 228(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s2, 224(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 220(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s4, 216(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s5, 212(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 208(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 204(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 200(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s9, 196(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s10, 192(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s11, 188(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    addi sp, sp, 240
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: clmul_i64:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    slli a2, a0, 1
+; RV64I-NEXT:    andi a3, a1, 2
+; RV64I-NEXT:    andi a4, a1, 1
+; RV64I-NEXT:    seqz a3, a3
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a2, a3, a2
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    xor a2, a4, a2
+; RV64I-NEXT:    andi a3, a1, 4
+; RV64I-NEXT:    slli a4, a0, 2
+; RV64I-NEXT:    seqz a3, a3
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    andi a5, a1, 8
+; RV64I-NEXT:    and a3, a3, a4
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    slli a5, a0, 3
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    andi a5, a1, 16
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    slli a5, a0, 4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    andi a5, a1, 32
+; RV64I-NEXT:    slli a6, a0, 5
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    andi a7, a1, 64
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    slli a7, a0, 6
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    xor a4, a2, a4
+; RV64I-NEXT:    andi a2, a1, 128
+; RV64I-NEXT:    slli a3, a0, 7
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    andi a5, a1, 256
+; RV64I-NEXT:    and a2, a2, a3
+; RV64I-NEXT:    seqz a3, a5
+; RV64I-NEXT:    slli a5, a0, 8
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    andi a5, a1, 512
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    seqz a3, a5
+; RV64I-NEXT:    slli a5, a0, 9
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    andi a5, a1, 1024
+; RV64I-NEXT:    xor a3, a2, a3
+; RV64I-NEXT:    seqz a2, a5
+; RV64I-NEXT:    slli a5, a0, 10
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    and a5, a2, a5
+; RV64I-NEXT:    li a2, 1
+; RV64I-NEXT:    xor a3, a3, a5
+; RV64I-NEXT:    slli a5, a2, 11
+; RV64I-NEXT:    xor a3, a4, a3
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    lui a5, 1
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    slli a6, a0, 11
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a0, 12
+; RV64I-NEXT:    lui a7, 2
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a0, 13
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    lui a6, 4
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a1, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 8
+; RV64I-NEXT:    slli a6, a0, 14
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a6, 16
+; RV64I-NEXT:    slli a7, a0, 15
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 16
+; RV64I-NEXT:    lui a7, 32
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a6, 64
+; RV64I-NEXT:    slli a7, a0, 17
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a7, a0, 18
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    lui a5, 128
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    lui a5, 256
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    slli a6, a0, 19
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a0, 20
+; RV64I-NEXT:    lui a7, 512
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a6, 1024
+; RV64I-NEXT:    slli a7, a0, 21
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 22
+; RV64I-NEXT:    lui a7, 2048
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a6, 4096
+; RV64I-NEXT:    slli a7, a0, 23
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a7, a0, 24
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    lui a5, 8192
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    lui a5, 16384
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    slli a6, a0, 25
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a0, 26
+; RV64I-NEXT:    lui a7, 32768
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a6, 65536
+; RV64I-NEXT:    slli a7, a0, 27
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 28
+; RV64I-NEXT:    lui a7, 131072
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a6, 262144
+; RV64I-NEXT:    slli a7, a0, 29
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 30
+; RV64I-NEXT:    sraiw a7, a1, 31
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    slli a7, a0, 31
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    slli a5, a2, 32
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    slli a5, a2, 33
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    slli a6, a0, 32
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a0, 33
+; RV64I-NEXT:    slli a7, a2, 34
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a2, 35
+; RV64I-NEXT:    slli a7, a0, 34
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 35
+; RV64I-NEXT:    slli a7, a2, 36
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a2, 37
+; RV64I-NEXT:    slli a7, a0, 36
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 37
+; RV64I-NEXT:    slli a7, a2, 38
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a0, 38
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    slli a6, a2, 39
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a1, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a5, a2, 40
+; RV64I-NEXT:    slli a6, a0, 39
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a2, 41
+; RV64I-NEXT:    slli a7, a0, 40
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 41
+; RV64I-NEXT:    slli a7, a2, 42
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a2, 43
+; RV64I-NEXT:    slli a7, a0, 42
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 43
+; RV64I-NEXT:    slli a7, a2, 44
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a2, 45
+; RV64I-NEXT:    slli a7, a0, 44
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 45
+; RV64I-NEXT:    slli a7, a2, 46
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a0, 46
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    slli a6, a2, 47
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, a1, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a5, a2, 48
+; RV64I-NEXT:    slli a6, a0, 47
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a2, 49
+; RV64I-NEXT:    slli a7, a0, 48
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 49
+; RV64I-NEXT:    slli a7, a2, 50
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a2, 51
+; RV64I-NEXT:    slli a7, a0, 50
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 51
+; RV64I-NEXT:    slli a7, a2, 52
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a2, 53
+; RV64I-NEXT:    slli a7, a0, 52
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 53
+; RV64I-NEXT:    slli a7, a2, 54
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a2, 55
+; RV64I-NEXT:    slli a7, a0, 54
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a5, a0, 55
+; RV64I-NEXT:    slli a7, a2, 56
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a2, 57
+; RV64I-NEXT:    slli a7, a0, 56
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a7, a2, 58
+; RV64I-NEXT:    slli t0, a0, 57
+; RV64I-NEXT:    and a7, a1, a7
+; RV64I-NEXT:    and a6, a6, t0
+; RV64I-NEXT:    seqz a7, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    addi a7, a7, -1
+; RV64I-NEXT:    slli a6, a0, 58
+; RV64I-NEXT:    slli t0, a2, 59
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    and a7, a1, t0
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a7, a2, 60
+; RV64I-NEXT:    slli t0, a0, 59
+; RV64I-NEXT:    and a7, a1, a7
+; RV64I-NEXT:    and a6, a6, t0
+; RV64I-NEXT:    seqz a7, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    addi a7, a7, -1
+; RV64I-NEXT:    slli a6, a0, 60
+; RV64I-NEXT:    slli t0, a2, 61
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    and a7, a1, t0
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    slli a2, a2, 62
+; RV64I-NEXT:    slli a7, a0, 61
+; RV64I-NEXT:    and a2, a1, a2
+; RV64I-NEXT:    and a6, a6, a7
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    slli a6, a0, 62
+; RV64I-NEXT:    srli a1, a1, 63
+; RV64I-NEXT:    and a2, a2, a6
+; RV64I-NEXT:    seqz a1, a1
+; RV64I-NEXT:    slli a0, a0, 63
+; RV64I-NEXT:    addi a1, a1, -1
+; RV64I-NEXT:    xor a2, a5, a2
+; RV64I-NEXT:    and a0, a1, a0
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    xor a0, a2, a0
+; RV64I-NEXT:    xor a0, a3, a0
+; RV64I-NEXT:    ret
+;
+; RV32IM-LABEL: clmul_i64:
+; RV32IM:       # %bb.0:
+; RV32IM-NEXT:    addi sp, sp, -80
+; RV32IM-NEXT:    sw ra, 76(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s0, 72(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s1, 68(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s2, 64(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s3, 60(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s4, 56(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s5, 52(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s6, 48(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s7, 44(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s8, 40(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s9, 36(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s10, 32(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s11, 28(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw a3, 24(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw a1, 16(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lui a4, 16
+; RV32IM-NEXT:    srli a5, a2, 8
+; RV32IM-NEXT:    addi t0, a4, -256
+; RV32IM-NEXT:    and a4, a5, t0
+; RV32IM-NEXT:    srli a5, a2, 24
+; RV32IM-NEXT:    and a6, a2, t0
+; RV32IM-NEXT:    slli a6, a6, 8
+; RV32IM-NEXT:    slli a7, a2, 24
+; RV32IM-NEXT:    or a4, a4, a5
+; RV32IM-NEXT:    or a5, a7, a6
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    lui a5, 61681
+; RV32IM-NEXT:    srli a6, a4, 4
+; RV32IM-NEXT:    addi t1, a5, -241
+; RV32IM-NEXT:    and a5, a6, t1
+; RV32IM-NEXT:    and a4, a4, t1
+; RV32IM-NEXT:    slli a4, a4, 4
+; RV32IM-NEXT:    lui a6, 209715
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    addi t2, a6, 819
+; RV32IM-NEXT:    srli a5, a4, 2
+; RV32IM-NEXT:    and a4, a4, t2
+; RV32IM-NEXT:    and a5, a5, t2
+; RV32IM-NEXT:    slli a4, a4, 2
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    lui a1, 349525
+; RV32IM-NEXT:    srli a5, a4, 1
+; RV32IM-NEXT:    addi t4, a1, 1365
+; RV32IM-NEXT:    and a5, a5, t4
+; RV32IM-NEXT:    sw a0, 20(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a6, a0, 8
+; RV32IM-NEXT:    and a4, a4, t4
+; RV32IM-NEXT:    and a6, a6, t0
+; RV32IM-NEXT:    srli a7, a0, 24
+; RV32IM-NEXT:    and t5, a0, t0
+; RV32IM-NEXT:    slli t5, t5, 8
+; RV32IM-NEXT:    slli t6, a0, 24
+; RV32IM-NEXT:    or a6, a6, a7
+; RV32IM-NEXT:    or a7, t6, t5
+; RV32IM-NEXT:    slli a4, a4, 1
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    srli a7, a6, 4
+; RV32IM-NEXT:    and a6, a6, t1
+; RV32IM-NEXT:    and a7, a7, t1
+; RV32IM-NEXT:    slli a6, a6, 4
+; RV32IM-NEXT:    or t5, a5, a4
+; RV32IM-NEXT:    or a4, a7, a6
+; RV32IM-NEXT:    srli a5, a4, 2
+; RV32IM-NEXT:    and a4, a4, t2
+; RV32IM-NEXT:    and a5, a5, t2
+; RV32IM-NEXT:    slli a4, a4, 2
+; RV32IM-NEXT:    lui a6, 69905
+; RV32IM-NEXT:    or a5, a5, a4
+; RV32IM-NEXT:    addi a4, a6, 273
+; RV32IM-NEXT:    srli a6, a5, 1
+; RV32IM-NEXT:    and a6, a6, t4
+; RV32IM-NEXT:    and a5, a5, t4
+; RV32IM-NEXT:    slli a5, a5, 1
+; RV32IM-NEXT:    lui a7, 139810
+; RV32IM-NEXT:    or t6, a6, a5
+; RV32IM-NEXT:    addi a5, a7, 546
+; RV32IM-NEXT:    and s0, t5, a4
+; RV32IM-NEXT:    and s1, t6, a5
+; RV32IM-NEXT:    and s2, t5, a5
+; RV32IM-NEXT:    and s3, t6, a4
+; RV32IM-NEXT:    mul s4, s1, s0
+; RV32IM-NEXT:    mul s5, s3, s2
+; RV32IM-NEXT:    lui a6, 559241
+; RV32IM-NEXT:    lui a7, 279620
+; RV32IM-NEXT:    addi s11, a6, -1912
+; RV32IM-NEXT:    addi t3, a7, 1092
+; RV32IM-NEXT:    and s6, t5, s11
+; RV32IM-NEXT:    and s7, t6, t3
+; RV32IM-NEXT:    and t5, t5, t3
+; RV32IM-NEXT:    and t6, t6, s11
+; RV32IM-NEXT:    mul s8, s7, s6
+; RV32IM-NEXT:    mul s9, t6, t5
+; RV32IM-NEXT:    mul s10, s1, s6
+; RV32IM-NEXT:    mul a6, s3, s0
+; RV32IM-NEXT:    mul ra, s7, t5
+; RV32IM-NEXT:    mul a3, t6, s2
+; RV32IM-NEXT:    mul a1, s1, s2
+; RV32IM-NEXT:    mul s1, s1, t5
+; RV32IM-NEXT:    mul t5, s3, t5
+; RV32IM-NEXT:    mul s3, s3, s6
+; RV32IM-NEXT:    mul s6, t6, s6
+; RV32IM-NEXT:    mul a0, s7, s0
+; RV32IM-NEXT:    mul s2, s7, s2
+; RV32IM-NEXT:    mul t6, t6, s0
+; RV32IM-NEXT:    xor s0, s5, s4
+; RV32IM-NEXT:    xor s4, s8, s9
+; RV32IM-NEXT:    xor s5, a6, s10
+; RV32IM-NEXT:    xor a3, ra, a3
+; RV32IM-NEXT:    xor s0, s0, s4
+; RV32IM-NEXT:    xor a3, s5, a3
+; RV32IM-NEXT:    and s0, s0, a5
+; RV32IM-NEXT:    and a3, a3, a4
+; RV32IM-NEXT:    xor a1, t5, a1
+; RV32IM-NEXT:    xor a0, a0, s6
+; RV32IM-NEXT:    xor t5, s3, s1
+; RV32IM-NEXT:    xor t6, s2, t6
+; RV32IM-NEXT:    xor a0, a1, a0
+; RV32IM-NEXT:    xor a1, t5, t6
+; RV32IM-NEXT:    and a0, a0, t3
+; RV32IM-NEXT:    and a1, a1, s11
+; RV32IM-NEXT:    or a3, a3, s0
+; RV32IM-NEXT:    or a0, a0, a1
+; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    srli a1, a0, 8
+; RV32IM-NEXT:    and a1, a1, t0
+; RV32IM-NEXT:    srli a3, a0, 24
+; RV32IM-NEXT:    and t0, a0, t0
+; RV32IM-NEXT:    slli a0, a0, 24
+; RV32IM-NEXT:    slli t0, t0, 8
+; RV32IM-NEXT:    or a1, a1, a3
+; RV32IM-NEXT:    or a0, a0, t0
+; RV32IM-NEXT:    or a0, a0, a1
+; RV32IM-NEXT:    srli a1, a0, 4
+; RV32IM-NEXT:    and a0, a0, t1
+; RV32IM-NEXT:    and a1, a1, t1
+; RV32IM-NEXT:    slli a0, a0, 4
+; RV32IM-NEXT:    or a0, a1, a0
+; RV32IM-NEXT:    srli a1, a0, 2
+; RV32IM-NEXT:    and a0, a0, t2
+; RV32IM-NEXT:    and a1, a1, t2
+; RV32IM-NEXT:    slli a0, a0, 2
+; RV32IM-NEXT:    or a0, a1, a0
+; RV32IM-NEXT:    srli a1, a0, 1
+; RV32IM-NEXT:    lui a3, 349525
+; RV32IM-NEXT:    addi a3, a3, 1364
+; RV32IM-NEXT:    and a0, a0, t4
+; RV32IM-NEXT:    and a1, a1, a3
+; RV32IM-NEXT:    slli a0, a0, 1
+; RV32IM-NEXT:    or a0, a1, a0
+; RV32IM-NEXT:    sw a0, 12(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv a7, a4
+; RV32IM-NEXT:    and t0, a2, a4
+; RV32IM-NEXT:    lw a6, 16(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a0, a6, a5
+; RV32IM-NEXT:    and t1, a2, a5
+; RV32IM-NEXT:    and a1, a6, a4
+; RV32IM-NEXT:    mul a3, a0, t0
+; RV32IM-NEXT:    mul t4, a1, t1
+; RV32IM-NEXT:    and t2, a2, s11
+; RV32IM-NEXT:    and t5, a6, t3
+; RV32IM-NEXT:    and a2, a2, t3
+; RV32IM-NEXT:    and t6, a6, s11
+; RV32IM-NEXT:    mul s0, t5, t2
+; RV32IM-NEXT:    mul s2, t6, a2
+; RV32IM-NEXT:    mul s3, a0, t2
+; RV32IM-NEXT:    mul s4, a1, t0
+; RV32IM-NEXT:    mul s5, t5, a2
+; RV32IM-NEXT:    mul s6, t6, t1
+; RV32IM-NEXT:    mul s7, a0, t1
+; RV32IM-NEXT:    mul s8, a1, a2
+; RV32IM-NEXT:    mul s9, t5, t0
+; RV32IM-NEXT:    mul s10, t6, t2
+; RV32IM-NEXT:    mul a0, a0, a2
+; RV32IM-NEXT:    mul a1, a1, t2
+; RV32IM-NEXT:    mul t5, t5, t1
+; RV32IM-NEXT:    mul t6, t6, t0
+; RV32IM-NEXT:    xor a3, t4, a3
+; RV32IM-NEXT:    xor t4, s0, s2
+; RV32IM-NEXT:    xor s0, s4, s3
+; RV32IM-NEXT:    xor s2, s5, s6
+; RV32IM-NEXT:    xor a3, a3, t4
+; RV32IM-NEXT:    xor t4, s0, s2
+; RV32IM-NEXT:    and a3, a3, a5
+; RV32IM-NEXT:    and t4, t4, a4
+; RV32IM-NEXT:    xor s0, s8, s7
+; RV32IM-NEXT:    xor s2, s9, s10
+; RV32IM-NEXT:    xor a0, a1, a0
+; RV32IM-NEXT:    xor a1, t5, t6
+; RV32IM-NEXT:    xor t5, s0, s2
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    and a1, t5, t3
+; RV32IM-NEXT:    and s0, a0, s11
+; RV32IM-NEXT:    or a0, t4, a3
+; RV32IM-NEXT:    sw a0, 16(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a1, a1, s0
+; RV32IM-NEXT:    sw a1, 8(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw a0, 24(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a0, a4
+; RV32IM-NEXT:    lw a4, 20(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a4, a5
+; RV32IM-NEXT:    and s2, a0, a5
+; RV32IM-NEXT:    mv s0, a5
+; RV32IM-NEXT:    and t4, a4, a7
+; RV32IM-NEXT:    mul s3, a1, a3
+; RV32IM-NEXT:    mul s4, t4, s2
+; RV32IM-NEXT:    mv s1, s11
+; RV32IM-NEXT:    and s5, a0, s11
+; RV32IM-NEXT:    and t5, a4, t3
+; RV32IM-NEXT:    and s6, a0, t3
+; RV32IM-NEXT:    and a0, a4, s11
+; RV32IM-NEXT:    mul t6, t5, s5
+; RV32IM-NEXT:    mul s7, a0, s6
+; RV32IM-NEXT:    mul s8, a1, s5
+; RV32IM-NEXT:    mul s9, t4, a3
+; RV32IM-NEXT:    mul s10, t5, s6
+; RV32IM-NEXT:    mul s11, a0, s2
+; RV32IM-NEXT:    mul ra, t4, s6
+; RV32IM-NEXT:    mul s6, a1, s6
+; RV32IM-NEXT:    mul a6, a0, s5
+; RV32IM-NEXT:    mul s5, t4, s5
+; RV32IM-NEXT:    mul a4, a1, s2
+; RV32IM-NEXT:    mul a5, t5, a3
+; RV32IM-NEXT:    mul s2, t5, s2
+; RV32IM-NEXT:    mul a3, a0, a3
+; RV32IM-NEXT:    xor s3, s4, s3
+; RV32IM-NEXT:    xor t6, t6, s7
+; RV32IM-NEXT:    xor s4, s9, s8
+; RV32IM-NEXT:    xor s7, s10, s11
+; RV32IM-NEXT:    xor t6, s3, t6
+; RV32IM-NEXT:    xor s3, s4, s7
+; RV32IM-NEXT:    mv s8, s0
+; RV32IM-NEXT:    and t6, t6, s0
+; RV32IM-NEXT:    and s3, s3, a7
+; RV32IM-NEXT:    xor a4, ra, a4
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    xor a6, s5, s6
+; RV32IM-NEXT:    xor a3, s2, a3
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    xor a3, a6, a3
+; RV32IM-NEXT:    and a4, a4, t3
+; RV32IM-NEXT:    mv s9, s1
+; RV32IM-NEXT:    and a3, a3, s1
+; RV32IM-NEXT:    or a5, s3, t6
+; RV32IM-NEXT:    or a3, a4, a3
+; RV32IM-NEXT:    lw a4, 16(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s0, 8(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    or s0, a4, s0
+; RV32IM-NEXT:    or a3, a5, a3
+; RV32IM-NEXT:    lw a4, 12(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    srli a4, a4, 1
+; RV32IM-NEXT:    xor a3, a3, s0
+; RV32IM-NEXT:    mul a5, a1, t0
+; RV32IM-NEXT:    mul a6, t4, t1
+; RV32IM-NEXT:    mul t6, t5, t2
+; RV32IM-NEXT:    mul s0, a0, a2
+; RV32IM-NEXT:    mul s1, a1, t2
+; RV32IM-NEXT:    mul s2, t4, t0
+; RV32IM-NEXT:    mul s3, t5, a2
+; RV32IM-NEXT:    mul s4, a0, t1
+; RV32IM-NEXT:    mul s5, a1, t1
+; RV32IM-NEXT:    mul s6, t4, a2
+; RV32IM-NEXT:    mul s7, t5, t0
+; RV32IM-NEXT:    mul a1, a1, a2
+; RV32IM-NEXT:    mul a2, a0, t2
+; RV32IM-NEXT:    mul t2, t4, t2
+; RV32IM-NEXT:    mul t1, t5, t1
+; RV32IM-NEXT:    mul a0, a0, t0
+; RV32IM-NEXT:    xor a5, a6, a5
+; RV32IM-NEXT:    xor a6, t6, s0
+; RV32IM-NEXT:    xor t0, s2, s1
+; RV32IM-NEXT:    xor t4, s3, s4
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    xor a6, t0, t4
+; RV32IM-NEXT:    and a5, a5, s8
+; RV32IM-NEXT:    and a6, a6, a7
+; RV32IM-NEXT:    xor t0, s6, s5
+; RV32IM-NEXT:    xor a2, s7, a2
+; RV32IM-NEXT:    xor a1, t2, a1
+; RV32IM-NEXT:    xor a0, t1, a0
+; RV32IM-NEXT:    xor a2, t0, a2
+; RV32IM-NEXT:    xor a0, a1, a0
+; RV32IM-NEXT:    and a1, a2, t3
+; RV32IM-NEXT:    and a0, a0, s9
+; RV32IM-NEXT:    or a2, a6, a5
+; RV32IM-NEXT:    or a0, a1, a0
+; RV32IM-NEXT:    xor a1, a4, a3
+; RV32IM-NEXT:    or a0, a2, a0
+; RV32IM-NEXT:    lw ra, 76(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s0, 72(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s1, 68(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s2, 64(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s3, 60(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s4, 56(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s5, 52(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s6, 48(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s7, 44(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s8, 40(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s9, 36(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s10, 32(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s11, 28(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    addi sp, sp, 80
+; RV32IM-NEXT:    ret
+;
+; RV64IM-LABEL: clmul_i64:
+; RV64IM:       # %bb.0:
+; RV64IM-NEXT:    addi sp, sp, -32
+; RV64IM-NEXT:    sd s0, 24(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s1, 16(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s2, 8(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s3, 0(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    lui a2, 69905
+; RV64IM-NEXT:    lui a3, 139810
+; RV64IM-NEXT:    addi a2, a2, 273
+; RV64IM-NEXT:    addi a3, a3, 546
+; RV64IM-NEXT:    slli a4, a2, 32
+; RV64IM-NEXT:    slli a5, a3, 32
+; RV64IM-NEXT:    add a2, a2, a4
+; RV64IM-NEXT:    add a3, a3, a5
+; RV64IM-NEXT:    and a4, a1, a2
+; RV64IM-NEXT:    and a5, a0, a3
+; RV64IM-NEXT:    mul a6, a5, a4
+; RV64IM-NEXT:    and a7, a1, a3
+; RV64IM-NEXT:    and t0, a0, a2
+; RV64IM-NEXT:    lui t1, 279620
+; RV64IM-NEXT:    mul t2, t0, a7
+; RV64IM-NEXT:    addi t1, t1, 1092
+; RV64IM-NEXT:    lui t3, %hi(.LCPI5_0)
+; RV64IM-NEXT:    slli t4, t1, 32
+; RV64IM-NEXT:    ld t3, %lo(.LCPI5_0)(t3)
+; RV64IM-NEXT:    add t1, t1, t4
+; RV64IM-NEXT:    and t4, a0, t1
+; RV64IM-NEXT:    and t5, a1, t1
+; RV64IM-NEXT:    mul t6, t0, a4
+; RV64IM-NEXT:    mul s0, t4, t5
+; RV64IM-NEXT:    xor a6, t2, a6
+; RV64IM-NEXT:    and a1, a1, t3
+; RV64IM-NEXT:    mul t2, t4, a1
+; RV64IM-NEXT:    and a0, a0, t3
+; RV64IM-NEXT:    mul s1, a0, t5
+; RV64IM-NEXT:    mul s2, a5, a1
+; RV64IM-NEXT:    xor t6, t6, s0
+; RV64IM-NEXT:    mul s0, a0, a7
+; RV64IM-NEXT:    mul s3, a5, a7
+; RV64IM-NEXT:    mul a5, a5, t5
+; RV64IM-NEXT:    mul t5, t0, t5
+; RV64IM-NEXT:    mul a7, t4, a7
+; RV64IM-NEXT:    mul t4, t4, a4
+; RV64IM-NEXT:    mul t0, t0, a1
+; RV64IM-NEXT:    mul a1, a0, a1
+; RV64IM-NEXT:    mul a0, a0, a4
+; RV64IM-NEXT:    xor a4, a6, t2
+; RV64IM-NEXT:    xor a6, t6, s2
+; RV64IM-NEXT:    xor a4, a4, s1
+; RV64IM-NEXT:    xor a6, a6, s0
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    and a2, a6, a2
+; RV64IM-NEXT:    xor a4, t5, s3
+; RV64IM-NEXT:    xor a5, a5, a7
+; RV64IM-NEXT:    xor a4, a4, t4
+; RV64IM-NEXT:    xor a5, t0, a5
+; RV64IM-NEXT:    xor a1, a4, a1
+; RV64IM-NEXT:    xor a0, a5, a0
+; RV64IM-NEXT:    and a1, a1, t1
+; RV64IM-NEXT:    and a0, a0, t3
+; RV64IM-NEXT:    or a2, a2, a3
+; RV64IM-NEXT:    or a0, a1, a0
+; RV64IM-NEXT:    or a0, a2, a0
+; RV64IM-NEXT:    ld s0, 24(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s1, 16(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s2, 8(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s3, 0(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    addi sp, sp, 32
+; RV64IM-NEXT:    ret
+;
+; RV32IMZBS-LABEL: clmul_i64:
+; RV32IMZBS:       # %bb.0:
+; RV32IMZBS-NEXT:    addi sp, sp, -80
+; RV32IMZBS-NEXT:    sw ra, 76(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s0, 72(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s1, 68(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s2, 64(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s3, 60(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s4, 56(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s5, 52(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s6, 48(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s7, 44(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s8, 40(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s9, 36(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s10, 32(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s11, 28(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw a3, 24(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw a1, 16(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lui a4, 16
+; RV32IMZBS-NEXT:    srli a5, a2, 8
+; RV32IMZBS-NEXT:    addi t0, a4, -256
+; RV32IMZBS-NEXT:    and a4, a5, t0
+; RV32IMZBS-NEXT:    srli a5, a2, 24
+; RV32IMZBS-NEXT:    and a6, a2, t0
+; RV32IMZBS-NEXT:    slli a6, a6, 8
+; RV32IMZBS-NEXT:    slli a7, a2, 24
+; RV32IMZBS-NEXT:    or a4, a4, a5
+; RV32IMZBS-NEXT:    or a5, a7, a6
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    lui a5, 61681
+; RV32IMZBS-NEXT:    srli a6, a4, 4
+; RV32IMZBS-NEXT:    addi t1, a5, -241
+; RV32IMZBS-NEXT:    and a5, a6, t1
+; RV32IMZBS-NEXT:    and a4, a4, t1
+; RV32IMZBS-NEXT:    slli a4, a4, 4
+; RV32IMZBS-NEXT:    lui a6, 209715
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    addi t2, a6, 819
+; RV32IMZBS-NEXT:    srli a5, a4, 2
+; RV32IMZBS-NEXT:    and a4, a4, t2
+; RV32IMZBS-NEXT:    and a5, a5, t2
+; RV32IMZBS-NEXT:    slli a4, a4, 2
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    lui a1, 349525
+; RV32IMZBS-NEXT:    srli a5, a4, 1
+; RV32IMZBS-NEXT:    addi t4, a1, 1365
+; RV32IMZBS-NEXT:    and a5, a5, t4
+; RV32IMZBS-NEXT:    sw a0, 20(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a6, a0, 8
+; RV32IMZBS-NEXT:    and a4, a4, t4
+; RV32IMZBS-NEXT:    and a6, a6, t0
+; RV32IMZBS-NEXT:    srli a7, a0, 24
+; RV32IMZBS-NEXT:    and t5, a0, t0
+; RV32IMZBS-NEXT:    slli t5, t5, 8
+; RV32IMZBS-NEXT:    slli t6, a0, 24
+; RV32IMZBS-NEXT:    or a6, a6, a7
+; RV32IMZBS-NEXT:    or a7, t6, t5
+; RV32IMZBS-NEXT:    slli a4, a4, 1
+; RV32IMZBS-NEXT:    or a6, a7, a6
+; RV32IMZBS-NEXT:    srli a7, a6, 4
+; RV32IMZBS-NEXT:    and a6, a6, t1
+; RV32IMZBS-NEXT:    and a7, a7, t1
+; RV32IMZBS-NEXT:    slli a6, a6, 4
+; RV32IMZBS-NEXT:    or t5, a5, a4
+; RV32IMZBS-NEXT:    or a4, a7, a6
+; RV32IMZBS-NEXT:    srli a5, a4, 2
+; RV32IMZBS-NEXT:    and a4, a4, t2
+; RV32IMZBS-NEXT:    and a5, a5, t2
+; RV32IMZBS-NEXT:    slli a4, a4, 2
+; RV32IMZBS-NEXT:    lui a6, 69905
+; RV32IMZBS-NEXT:    or a5, a5, a4
+; RV32IMZBS-NEXT:    addi a4, a6, 273
+; RV32IMZBS-NEXT:    srli a6, a5, 1
+; RV32IMZBS-NEXT:    and a6, a6, t4
+; RV32IMZBS-NEXT:    and a5, a5, t4
+; RV32IMZBS-NEXT:    slli a5, a5, 1
+; RV32IMZBS-NEXT:    lui a7, 139810
+; RV32IMZBS-NEXT:    or t6, a6, a5
+; RV32IMZBS-NEXT:    addi a5, a7, 546
+; RV32IMZBS-NEXT:    and s0, t5, a4
+; RV32IMZBS-NEXT:    and s1, t6, a5
+; RV32IMZBS-NEXT:    and s2, t5, a5
+; RV32IMZBS-NEXT:    and s3, t6, a4
+; RV32IMZBS-NEXT:    mul s4, s1, s0
+; RV32IMZBS-NEXT:    mul s5, s3, s2
+; RV32IMZBS-NEXT:    lui a6, 559241
+; RV32IMZBS-NEXT:    lui a7, 279620
+; RV32IMZBS-NEXT:    addi s11, a6, -1912
+; RV32IMZBS-NEXT:    addi t3, a7, 1092
+; RV32IMZBS-NEXT:    and s6, t5, s11
+; RV32IMZBS-NEXT:    and s7, t6, t3
+; RV32IMZBS-NEXT:    and t5, t5, t3
+; RV32IMZBS-NEXT:    and t6, t6, s11
+; RV32IMZBS-NEXT:    mul s8, s7, s6
+; RV32IMZBS-NEXT:    mul s9, t6, t5
+; RV32IMZBS-NEXT:    mul s10, s1, s6
+; RV32IMZBS-NEXT:    mul a6, s3, s0
+; RV32IMZBS-NEXT:    mul ra, s7, t5
+; RV32IMZBS-NEXT:    mul a3, t6, s2
+; RV32IMZBS-NEXT:    mul a1, s1, s2
+; RV32IMZBS-NEXT:    mul s1, s1, t5
+; RV32IMZBS-NEXT:    mul t5, s3, t5
+; RV32IMZBS-NEXT:    mul s3, s3, s6
+; RV32IMZBS-NEXT:    mul s6, t6, s6
+; RV32IMZBS-NEXT:    mul a0, s7, s0
+; RV32IMZBS-NEXT:    mul s2, s7, s2
+; RV32IMZBS-NEXT:    mul t6, t6, s0
+; RV32IMZBS-NEXT:    xor s0, s5, s4
+; RV32IMZBS-NEXT:    xor s4, s8, s9
+; RV32IMZBS-NEXT:    xor s5, a6, s10
+; RV32IMZBS-NEXT:    xor a3, ra, a3
+; RV32IMZBS-NEXT:    xor s0, s0, s4
+; RV32IMZBS-NEXT:    xor a3, s5, a3
+; RV32IMZBS-NEXT:    and s0, s0, a5
+; RV32IMZBS-NEXT:    and a3, a3, a4
+; RV32IMZBS-NEXT:    xor a1, t5, a1
+; RV32IMZBS-NEXT:    xor a0, a0, s6
+; RV32IMZBS-NEXT:    xor t5, s3, s1
+; RV32IMZBS-NEXT:    xor t6, s2, t6
+; RV32IMZBS-NEXT:    xor a0, a1, a0
+; RV32IMZBS-NEXT:    xor a1, t5, t6
+; RV32IMZBS-NEXT:    and a0, a0, t3
+; RV32IMZBS-NEXT:    and a1, a1, s11
+; RV32IMZBS-NEXT:    or a3, a3, s0
+; RV32IMZBS-NEXT:    or a0, a0, a1
+; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    srli a1, a0, 8
+; RV32IMZBS-NEXT:    and a1, a1, t0
+; RV32IMZBS-NEXT:    srli a3, a0, 24
+; RV32IMZBS-NEXT:    and t0, a0, t0
+; RV32IMZBS-NEXT:    slli a0, a0, 24
+; RV32IMZBS-NEXT:    slli t0, t0, 8
+; RV32IMZBS-NEXT:    or a1, a1, a3
+; RV32IMZBS-NEXT:    or a0, a0, t0
+; RV32IMZBS-NEXT:    or a0, a0, a1
+; RV32IMZBS-NEXT:    srli a1, a0, 4
+; RV32IMZBS-NEXT:    and a0, a0, t1
+; RV32IMZBS-NEXT:    and a1, a1, t1
+; RV32IMZBS-NEXT:    slli a0, a0, 4
+; RV32IMZBS-NEXT:    or a0, a1, a0
+; RV32IMZBS-NEXT:    srli a1, a0, 2
+; RV32IMZBS-NEXT:    and a0, a0, t2
+; RV32IMZBS-NEXT:    and a1, a1, t2
+; RV32IMZBS-NEXT:    slli a0, a0, 2
+; RV32IMZBS-NEXT:    or a0, a1, a0
+; RV32IMZBS-NEXT:    srli a1, a0, 1
+; RV32IMZBS-NEXT:    lui a3, 349525
+; RV32IMZBS-NEXT:    addi a3, a3, 1364
+; RV32IMZBS-NEXT:    and a0, a0, t4
+; RV32IMZBS-NEXT:    and a1, a1, a3
+; RV32IMZBS-NEXT:    slli a0, a0, 1
+; RV32IMZBS-NEXT:    or a0, a1, a0
+; RV32IMZBS-NEXT:    sw a0, 12(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv a7, a4
+; RV32IMZBS-NEXT:    and t0, a2, a4
+; RV32IMZBS-NEXT:    lw a6, 16(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a0, a6, a5
+; RV32IMZBS-NEXT:    and t1, a2, a5
+; RV32IMZBS-NEXT:    and a1, a6, a4
+; RV32IMZBS-NEXT:    mul a3, a0, t0
+; RV32IMZBS-NEXT:    mul t4, a1, t1
+; RV32IMZBS-NEXT:    and t2, a2, s11
+; RV32IMZBS-NEXT:    and t5, a6, t3
+; RV32IMZBS-NEXT:    and a2, a2, t3
+; RV32IMZBS-NEXT:    and t6, a6, s11
+; RV32IMZBS-NEXT:    mul s0, t5, t2
+; RV32IMZBS-NEXT:    mul s2, t6, a2
+; RV32IMZBS-NEXT:    mul s3, a0, t2
+; RV32IMZBS-NEXT:    mul s4, a1, t0
+; RV32IMZBS-NEXT:    mul s5, t5, a2
+; RV32IMZBS-NEXT:    mul s6, t6, t1
+; RV32IMZBS-NEXT:    mul s7, a0, t1
+; RV32IMZBS-NEXT:    mul s8, a1, a2
+; RV32IMZBS-NEXT:    mul s9, t5, t0
+; RV32IMZBS-NEXT:    mul s10, t6, t2
+; RV32IMZBS-NEXT:    mul a0, a0, a2
+; RV32IMZBS-NEXT:    mul a1, a1, t2
+; RV32IMZBS-NEXT:    mul t5, t5, t1
+; RV32IMZBS-NEXT:    mul t6, t6, t0
+; RV32IMZBS-NEXT:    xor a3, t4, a3
+; RV32IMZBS-NEXT:    xor t4, s0, s2
+; RV32IMZBS-NEXT:    xor s0, s4, s3
+; RV32IMZBS-NEXT:    xor s2, s5, s6
+; RV32IMZBS-NEXT:    xor a3, a3, t4
+; RV32IMZBS-NEXT:    xor t4, s0, s2
+; RV32IMZBS-NEXT:    and a3, a3, a5
+; RV32IMZBS-NEXT:    and t4, t4, a4
+; RV32IMZBS-NEXT:    xor s0, s8, s7
+; RV32IMZBS-NEXT:    xor s2, s9, s10
+; RV32IMZBS-NEXT:    xor a0, a1, a0
+; RV32IMZBS-NEXT:    xor a1, t5, t6
+; RV32IMZBS-NEXT:    xor t5, s0, s2
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    and a1, t5, t3
+; RV32IMZBS-NEXT:    and s0, a0, s11
+; RV32IMZBS-NEXT:    or a0, t4, a3
+; RV32IMZBS-NEXT:    sw a0, 16(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a1, a1, s0
+; RV32IMZBS-NEXT:    sw a1, 8(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw a0, 24(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a0, a4
+; RV32IMZBS-NEXT:    lw a4, 20(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a4, a5
+; RV32IMZBS-NEXT:    and s2, a0, a5
+; RV32IMZBS-NEXT:    mv s0, a5
+; RV32IMZBS-NEXT:    and t4, a4, a7
+; RV32IMZBS-NEXT:    mul s3, a1, a3
+; RV32IMZBS-NEXT:    mul s4, t4, s2
+; RV32IMZBS-NEXT:    mv s1, s11
+; RV32IMZBS-NEXT:    and s5, a0, s11
+; RV32IMZBS-NEXT:    and t5, a4, t3
+; RV32IMZBS-NEXT:    and s6, a0, t3
+; RV32IMZBS-NEXT:    and a0, a4, s11
+; RV32IMZBS-NEXT:    mul t6, t5, s5
+; RV32IMZBS-NEXT:    mul s7, a0, s6
+; RV32IMZBS-NEXT:    mul s8, a1, s5
+; RV32IMZBS-NEXT:    mul s9, t4, a3
+; RV32IMZBS-NEXT:    mul s10, t5, s6
+; RV32IMZBS-NEXT:    mul s11, a0, s2
+; RV32IMZBS-NEXT:    mul ra, t4, s6
+; RV32IMZBS-NEXT:    mul s6, a1, s6
+; RV32IMZBS-NEXT:    mul a6, a0, s5
+; RV32IMZBS-NEXT:    mul s5, t4, s5
+; RV32IMZBS-NEXT:    mul a4, a1, s2
+; RV32IMZBS-NEXT:    mul a5, t5, a3
+; RV32IMZBS-NEXT:    mul s2, t5, s2
+; RV32IMZBS-NEXT:    mul a3, a0, a3
+; RV32IMZBS-NEXT:    xor s3, s4, s3
+; RV32IMZBS-NEXT:    xor t6, t6, s7
+; RV32IMZBS-NEXT:    xor s4, s9, s8
+; RV32IMZBS-NEXT:    xor s7, s10, s11
+; RV32IMZBS-NEXT:    xor t6, s3, t6
+; RV32IMZBS-NEXT:    xor s3, s4, s7
+; RV32IMZBS-NEXT:    mv s8, s0
+; RV32IMZBS-NEXT:    and t6, t6, s0
+; RV32IMZBS-NEXT:    and s3, s3, a7
+; RV32IMZBS-NEXT:    xor a4, ra, a4
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    xor a6, s5, s6
+; RV32IMZBS-NEXT:    xor a3, s2, a3
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    xor a3, a6, a3
+; RV32IMZBS-NEXT:    and a4, a4, t3
+; RV32IMZBS-NEXT:    mv s9, s1
+; RV32IMZBS-NEXT:    and a3, a3, s1
+; RV32IMZBS-NEXT:    or a5, s3, t6
+; RV32IMZBS-NEXT:    or a3, a4, a3
+; RV32IMZBS-NEXT:    lw a4, 16(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s0, 8(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    or s0, a4, s0
+; RV32IMZBS-NEXT:    or a3, a5, a3
+; RV32IMZBS-NEXT:    lw a4, 12(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    srli a4, a4, 1
+; RV32IMZBS-NEXT:    xor a3, a3, s0
+; RV32IMZBS-NEXT:    mul a5, a1, t0
+; RV32IMZBS-NEXT:    mul a6, t4, t1
+; RV32IMZBS-NEXT:    mul t6, t5, t2
+; RV32IMZBS-NEXT:    mul s0, a0, a2
+; RV32IMZBS-NEXT:    mul s1, a1, t2
+; RV32IMZBS-NEXT:    mul s2, t4, t0
+; RV32IMZBS-NEXT:    mul s3, t5, a2
+; RV32IMZBS-NEXT:    mul s4, a0, t1
+; RV32IMZBS-NEXT:    mul s5, a1, t1
+; RV32IMZBS-NEXT:    mul s6, t4, a2
+; RV32IMZBS-NEXT:    mul s7, t5, t0
+; RV32IMZBS-NEXT:    mul a1, a1, a2
+; RV32IMZBS-NEXT:    mul a2, a0, t2
+; RV32IMZBS-NEXT:    mul t2, t4, t2
+; RV32IMZBS-NEXT:    mul t1, t5, t1
+; RV32IMZBS-NEXT:    mul a0, a0, t0
+; RV32IMZBS-NEXT:    xor a5, a6, a5
+; RV32IMZBS-NEXT:    xor a6, t6, s0
+; RV32IMZBS-NEXT:    xor t0, s2, s1
+; RV32IMZBS-NEXT:    xor t4, s3, s4
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    xor a6, t0, t4
+; RV32IMZBS-NEXT:    and a5, a5, s8
+; RV32IMZBS-NEXT:    and a6, a6, a7
+; RV32IMZBS-NEXT:    xor t0, s6, s5
+; RV32IMZBS-NEXT:    xor a2, s7, a2
+; RV32IMZBS-NEXT:    xor a1, t2, a1
+; RV32IMZBS-NEXT:    xor a0, t1, a0
+; RV32IMZBS-NEXT:    xor a2, t0, a2
+; RV32IMZBS-NEXT:    xor a0, a1, a0
+; RV32IMZBS-NEXT:    and a1, a2, t3
+; RV32IMZBS-NEXT:    and a0, a0, s9
+; RV32IMZBS-NEXT:    or a2, a6, a5
+; RV32IMZBS-NEXT:    or a0, a1, a0
+; RV32IMZBS-NEXT:    xor a1, a4, a3
+; RV32IMZBS-NEXT:    or a0, a2, a0
+; RV32IMZBS-NEXT:    lw ra, 76(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s0, 72(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s1, 68(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s2, 64(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s3, 60(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s4, 56(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s5, 52(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s6, 48(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s7, 44(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s8, 40(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s9, 36(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s10, 32(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s11, 28(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    addi sp, sp, 80
+; RV32IMZBS-NEXT:    ret
+;
+; RV64IMZBS-LABEL: clmul_i64:
+; RV64IMZBS:       # %bb.0:
+; RV64IMZBS-NEXT:    addi sp, sp, -32
+; RV64IMZBS-NEXT:    sd s0, 24(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s1, 16(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s2, 8(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s3, 0(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    lui a2, 69905
+; RV64IMZBS-NEXT:    lui a3, 139810
+; RV64IMZBS-NEXT:    addi a2, a2, 273
+; RV64IMZBS-NEXT:    addi a3, a3, 546
+; RV64IMZBS-NEXT:    slli a4, a2, 32
+; RV64IMZBS-NEXT:    slli a5, a3, 32
+; RV64IMZBS-NEXT:    add a2, a2, a4
+; RV64IMZBS-NEXT:    add a3, a3, a5
+; RV64IMZBS-NEXT:    and a4, a1, a2
+; RV64IMZBS-NEXT:    and a5, a0, a3
+; RV64IMZBS-NEXT:    mul a6, a5, a4
+; RV64IMZBS-NEXT:    and a7, a1, a3
+; RV64IMZBS-NEXT:    and t0, a0, a2
+; RV64IMZBS-NEXT:    lui t1, 279620
+; RV64IMZBS-NEXT:    mul t2, t0, a7
+; RV64IMZBS-NEXT:    addi t1, t1, 1092
+; RV64IMZBS-NEXT:    lui t3, %hi(.LCPI5_0)
+; RV64IMZBS-NEXT:    slli t4, t1, 32
+; RV64IMZBS-NEXT:    ld t3, %lo(.LCPI5_0)(t3)
+; RV64IMZBS-NEXT:    add t1, t1, t4
+; RV64IMZBS-NEXT:    and t4, a0, t1
+; RV64IMZBS-NEXT:    and t5, a1, t1
+; RV64IMZBS-NEXT:    mul t6, t0, a4
+; RV64IMZBS-NEXT:    mul s0, t4, t5
+; RV64IMZBS-NEXT:    xor a6, t2, a6
+; RV64IMZBS-NEXT:    and a1, a1, t3
+; RV64IMZBS-NEXT:    mul t2, t4, a1
+; RV64IMZBS-NEXT:    and a0, a0, t3
+; RV64IMZBS-NEXT:    mul s1, a0, t5
+; RV64IMZBS-NEXT:    mul s2, a5, a1
+; RV64IMZBS-NEXT:    xor t6, t6, s0
+; RV64IMZBS-NEXT:    mul s0, a0, a7
+; RV64IMZBS-NEXT:    mul s3, a5, a7
+; RV64IMZBS-NEXT:    mul a5, a5, t5
+; RV64IMZBS-NEXT:    mul t5, t0, t5
+; RV64IMZBS-NEXT:    mul a7, t4, a7
+; RV64IMZBS-NEXT:    mul t4, t4, a4
+; RV64IMZBS-NEXT:    mul t0, t0, a1
+; RV64IMZBS-NEXT:    mul a1, a0, a1
+; RV64IMZBS-NEXT:    mul a0, a0, a4
+; RV64IMZBS-NEXT:    xor a4, a6, t2
+; RV64IMZBS-NEXT:    xor a6, t6, s2
+; RV64IMZBS-NEXT:    xor a4, a4, s1
+; RV64IMZBS-NEXT:    xor a6, a6, s0
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    and a2, a6, a2
+; RV64IMZBS-NEXT:    xor a4, t5, s3
+; RV64IMZBS-NEXT:    xor a5, a5, a7
+; RV64IMZBS-NEXT:    xor a4, a4, t4
+; RV64IMZBS-NEXT:    xor a5, t0, a5
+; RV64IMZBS-NEXT:    xor a1, a4, a1
+; RV64IMZBS-NEXT:    xor a0, a5, a0
+; RV64IMZBS-NEXT:    and a1, a1, t1
+; RV64IMZBS-NEXT:    and a0, a0, t3
+; RV64IMZBS-NEXT:    or a2, a2, a3
+; RV64IMZBS-NEXT:    or a0, a1, a0
+; RV64IMZBS-NEXT:    or a0, a2, a0
+; RV64IMZBS-NEXT:    ld s0, 24(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s1, 16(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s2, 8(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s3, 0(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    addi sp, sp, 32
+; RV64IMZBS-NEXT:    ret
+;
+; RV32IMZBC-LABEL: clmul_i64:
+; RV32IMZBC:       # %bb.0:
+; RV32IMZBC-NEXT:    clmul a1, a1, a2
+; RV32IMZBC-NEXT:    clmul a3, a0, a3
+; RV32IMZBC-NEXT:    clmulh a4, a0, a2
+; RV32IMZBC-NEXT:    xor a1, a3, a1
+; RV32IMZBC-NEXT:    clmul a0, a0, a2
+; RV32IMZBC-NEXT:    xor a1, a4, a1
+; RV32IMZBC-NEXT:    ret
+;
+; RV64IMZBC-LABEL: clmul_i64:
+; RV64IMZBC:       # %bb.0:
+; RV64IMZBC-NEXT:    clmul a0, a0, a1
+; RV64IMZBC-NEXT:    ret
+  %res = call i64 @llvm.clmul.i64(i64 %a, i64 %b)
+  ret i64 %res
+}
+
+define i64 @clmul_i64_zext(i32 %x, i32 %y) {
+; RV32I-LABEL: clmul_i64_zext:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi sp, sp, -64
+; RV32I-NEXT:    .cfi_def_cfa_offset 64
+; RV32I-NEXT:    sw ra, 60(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s0, 56(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s1, 52(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s2, 48(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s3, 44(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s4, 40(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s5, 36(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s6, 32(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s7, 28(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s8, 24(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s9, 20(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s10, 16(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s11, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    .cfi_offset ra, -4
+; RV32I-NEXT:    .cfi_offset s0, -8
+; RV32I-NEXT:    .cfi_offset s1, -12
+; RV32I-NEXT:    .cfi_offset s2, -16
+; RV32I-NEXT:    .cfi_offset s3, -20
+; RV32I-NEXT:    .cfi_offset s4, -24
+; RV32I-NEXT:    .cfi_offset s5, -28
+; RV32I-NEXT:    .cfi_offset s6, -32
+; RV32I-NEXT:    .cfi_offset s7, -36
+; RV32I-NEXT:    .cfi_offset s8, -40
+; RV32I-NEXT:    .cfi_offset s9, -44
+; RV32I-NEXT:    .cfi_offset s10, -48
+; RV32I-NEXT:    .cfi_offset s11, -52
+; RV32I-NEXT:    mv a3, a1
+; RV32I-NEXT:    mv t2, a0
+; RV32I-NEXT:    lui a1, 16
+; RV32I-NEXT:    srli a0, a0, 8
+; RV32I-NEXT:    addi t1, a1, -256
+; RV32I-NEXT:    lui s3, 16
+; RV32I-NEXT:    and a0, a0, t1
+; RV32I-NEXT:    srli a2, t2, 24
+; RV32I-NEXT:    and a4, t2, t1
+; RV32I-NEXT:    slli a4, a4, 8
+; RV32I-NEXT:    slli a1, t2, 24
+; RV32I-NEXT:    sw a1, 8(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    or a0, a0, a2
+; RV32I-NEXT:    or a2, a1, a4
+; RV32I-NEXT:    or a0, a2, a0
+; RV32I-NEXT:    lui a2, 61681
+; RV32I-NEXT:    srli a4, a0, 4
+; RV32I-NEXT:    addi a6, a2, -241
+; RV32I-NEXT:    and a2, a4, a6
+; RV32I-NEXT:    and a0, a0, a6
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    lui a4, 209715
+; RV32I-NEXT:    or a0, a2, a0
+; RV32I-NEXT:    addi a7, a4, 819
+; RV32I-NEXT:    srli a2, a0, 2
+; RV32I-NEXT:    and a0, a0, a7
+; RV32I-NEXT:    and a2, a2, a7
+; RV32I-NEXT:    slli a0, a0, 2
+; RV32I-NEXT:    or a0, a2, a0
+; RV32I-NEXT:    srli a2, a0, 1
+; RV32I-NEXT:    lui t0, 349525
+; RV32I-NEXT:    addi t0, t0, 1365
+; RV32I-NEXT:    srli a5, a3, 8
+; RV32I-NEXT:    and a2, a2, t0
+; RV32I-NEXT:    mv a1, t1
+; RV32I-NEXT:    sw t1, 4(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, t1
+; RV32I-NEXT:    srli t1, a3, 24
+; RV32I-NEXT:    and t3, a3, a1
+; RV32I-NEXT:    slli t3, t3, 8
+; RV32I-NEXT:    slli t4, a3, 24
+; RV32I-NEXT:    or a5, a5, t1
+; RV32I-NEXT:    or t1, t4, t3
+; RV32I-NEXT:    and a0, a0, t0
+; RV32I-NEXT:    or a5, t1, a5
+; RV32I-NEXT:    srli t1, a5, 4
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    and t1, t1, a6
+; RV32I-NEXT:    slli a5, a5, 4
+; RV32I-NEXT:    slli a0, a0, 1
+; RV32I-NEXT:    or a5, t1, a5
+; RV32I-NEXT:    srli t1, a5, 2
+; RV32I-NEXT:    and a5, a5, a7
+; RV32I-NEXT:    and t1, t1, a7
+; RV32I-NEXT:    slli a5, a5, 2
+; RV32I-NEXT:    or s10, a2, a0
+; RV32I-NEXT:    or a0, t1, a5
+; RV32I-NEXT:    srli a2, a0, 1
+; RV32I-NEXT:    and a0, a0, t0
+; RV32I-NEXT:    and a2, a2, t0
+; RV32I-NEXT:    slli a5, a0, 1
+; RV32I-NEXT:    slli t1, s10, 1
+; RV32I-NEXT:    or a0, a2, a5
+; RV32I-NEXT:    andi a2, a0, 2
+; RV32I-NEXT:    andi t3, a0, 1
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    seqz t3, t3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    addi t3, t3, -1
+; RV32I-NEXT:    and a2, a2, t1
+; RV32I-NEXT:    and t1, t3, s10
+; RV32I-NEXT:    slli t3, s10, 2
+; RV32I-NEXT:    andi t4, a0, 4
+; RV32I-NEXT:    seqz t4, t4
+; RV32I-NEXT:    andi t5, a0, 8
+; RV32I-NEXT:    addi t4, t4, -1
+; RV32I-NEXT:    seqz t5, t5
+; RV32I-NEXT:    slli t6, s10, 3
+; RV32I-NEXT:    addi t5, t5, -1
+; RV32I-NEXT:    and t3, t4, t3
+; RV32I-NEXT:    and t4, t5, t6
+; RV32I-NEXT:    xor a2, t1, a2
+; RV32I-NEXT:    xor t1, t3, t4
+; RV32I-NEXT:    xor a2, a2, t1
+; RV32I-NEXT:    andi t1, a0, 16
+; RV32I-NEXT:    slli t3, s10, 4
+; RV32I-NEXT:    seqz t1, t1
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    andi t4, a0, 32
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    seqz t3, t4
+; RV32I-NEXT:    slli t4, s10, 5
+; RV32I-NEXT:    addi t3, t3, -1
+; RV32I-NEXT:    and t3, t3, t4
+; RV32I-NEXT:    andi t4, a0, 64
+; RV32I-NEXT:    xor t1, t1, t3
+; RV32I-NEXT:    seqz t3, t4
+; RV32I-NEXT:    slli t4, s10, 6
+; RV32I-NEXT:    addi t3, t3, -1
+; RV32I-NEXT:    and t3, t3, t4
+; RV32I-NEXT:    andi t4, a0, 128
+; RV32I-NEXT:    xor t1, t1, t3
+; RV32I-NEXT:    seqz t3, t4
+; RV32I-NEXT:    slli t4, s10, 7
+; RV32I-NEXT:    addi t3, t3, -1
+; RV32I-NEXT:    and t3, t3, t4
+; RV32I-NEXT:    andi t4, a0, 256
+; RV32I-NEXT:    slli t5, s10, 8
+; RV32I-NEXT:    seqz t4, t4
+; RV32I-NEXT:    addi t4, t4, -1
+; RV32I-NEXT:    andi t6, a0, 512
+; RV32I-NEXT:    and t4, t4, t5
+; RV32I-NEXT:    seqz t5, t6
+; RV32I-NEXT:    slli t6, s10, 9
+; RV32I-NEXT:    addi t5, t5, -1
+; RV32I-NEXT:    xor t3, t3, t4
+; RV32I-NEXT:    and t4, t5, t6
+; RV32I-NEXT:    xor a2, a2, t1
+; RV32I-NEXT:    xor t3, t3, t4
+; RV32I-NEXT:    slli t4, s10, 10
+; RV32I-NEXT:    andi t1, a0, 1024
+; RV32I-NEXT:    seqz t1, t1
+; RV32I-NEXT:    li t5, 1
+; RV32I-NEXT:    addi t6, t1, -1
+; RV32I-NEXT:    slli t1, t5, 11
+; RV32I-NEXT:    and t4, t6, t4
+; RV32I-NEXT:    and t5, a0, t1
+; RV32I-NEXT:    xor t3, t3, t4
+; RV32I-NEXT:    seqz t4, t5
+; RV32I-NEXT:    slli t5, s10, 11
+; RV32I-NEXT:    addi t4, t4, -1
+; RV32I-NEXT:    and t4, t4, t5
+; RV32I-NEXT:    lui a1, 1
+; RV32I-NEXT:    slli t5, s10, 12
+; RV32I-NEXT:    and t6, a0, a1
+; RV32I-NEXT:    seqz t6, t6
+; RV32I-NEXT:    lui s0, 2
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    and s0, a0, s0
+; RV32I-NEXT:    and t5, t6, t5
+; RV32I-NEXT:    seqz t6, s0
+; RV32I-NEXT:    slli s0, s10, 13
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    xor t4, t4, t5
+; RV32I-NEXT:    and t5, t6, s0
+; RV32I-NEXT:    xor t6, t4, t5
+; RV32I-NEXT:    lui a1, 4
+; RV32I-NEXT:    slli s0, s10, 14
+; RV32I-NEXT:    and t5, a0, a1
+; RV32I-NEXT:    seqz s1, t5
+; RV32I-NEXT:    lui t5, 8
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    and s2, a0, t5
+; RV32I-NEXT:    and s0, s1, s0
+; RV32I-NEXT:    seqz s1, s2
+; RV32I-NEXT:    slli s2, s10, 15
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    xor t6, t6, s0
+; RV32I-NEXT:    and s0, s1, s2
+; RV32I-NEXT:    xor a2, a2, t3
+; RV32I-NEXT:    xor t3, t6, s0
+; RV32I-NEXT:    xor a2, a2, t3
+; RV32I-NEXT:    slli t3, s10, 16
+; RV32I-NEXT:    and s0, a0, s3
+; RV32I-NEXT:    lui t6, 32
+; RV32I-NEXT:    seqz s0, s0
+; RV32I-NEXT:    and s1, a0, t6
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    seqz s1, s1
+; RV32I-NEXT:    slli s2, s10, 17
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    and t3, s0, t3
+; RV32I-NEXT:    and s0, s1, s2
+; RV32I-NEXT:    xor t3, t3, s0
+; RV32I-NEXT:    lui s1, 64
+; RV32I-NEXT:    slli s2, s10, 18
+; RV32I-NEXT:    and s0, a0, s1
+; RV32I-NEXT:    seqz s3, s0
+; RV32I-NEXT:    lui s0, 128
+; RV32I-NEXT:    addi s3, s3, -1
+; RV32I-NEXT:    and s4, a0, s0
+; RV32I-NEXT:    and s2, s3, s2
+; RV32I-NEXT:    seqz s3, s4
+; RV32I-NEXT:    slli s4, s10, 19
+; RV32I-NEXT:    addi s3, s3, -1
+; RV32I-NEXT:    xor t3, t3, s2
+; RV32I-NEXT:    and s2, s3, s4
+; RV32I-NEXT:    xor t3, t3, s2
+; RV32I-NEXT:    lui s2, 256
+; RV32I-NEXT:    slli s4, s10, 20
+; RV32I-NEXT:    and s3, a0, s2
+; RV32I-NEXT:    seqz s5, s3
+; RV32I-NEXT:    lui s3, 512
+; RV32I-NEXT:    addi s5, s5, -1
+; RV32I-NEXT:    and s6, a0, s3
+; RV32I-NEXT:    and s4, s5, s4
+; RV32I-NEXT:    seqz s5, s6
+; RV32I-NEXT:    slli s6, s10, 21
+; RV32I-NEXT:    addi s5, s5, -1
+; RV32I-NEXT:    xor t3, t3, s4
+; RV32I-NEXT:    and s4, s5, s6
+; RV32I-NEXT:    xor t3, t3, s4
+; RV32I-NEXT:    lui s4, 1024
+; RV32I-NEXT:    xor a4, a2, t3
+; RV32I-NEXT:    and t3, a0, s4
+; RV32I-NEXT:    slli s6, s10, 22
+; RV32I-NEXT:    seqz t3, t3
+; RV32I-NEXT:    addi t3, t3, -1
+; RV32I-NEXT:    lui s5, 2048
+; RV32I-NEXT:    and t3, t3, s6
+; RV32I-NEXT:    and s6, a0, s5
+; RV32I-NEXT:    slli s7, s10, 23
+; RV32I-NEXT:    seqz s6, s6
+; RV32I-NEXT:    addi s8, s6, -1
+; RV32I-NEXT:    lui s6, 4096
+; RV32I-NEXT:    and s7, s8, s7
+; RV32I-NEXT:    and s8, a0, s6
+; RV32I-NEXT:    xor t3, t3, s7
+; RV32I-NEXT:    seqz s7, s8
+; RV32I-NEXT:    addi s8, s7, -1
+; RV32I-NEXT:    lui s7, 8192
+; RV32I-NEXT:    slli s9, s10, 24
+; RV32I-NEXT:    and a1, a0, s7
+; RV32I-NEXT:    and s8, s8, s9
+; RV32I-NEXT:    seqz s9, a1
+; RV32I-NEXT:    xor t3, t3, s8
+; RV32I-NEXT:    addi s9, s9, -1
+; RV32I-NEXT:    slli a1, s10, 25
+; RV32I-NEXT:    lui s8, 16384
+; RV32I-NEXT:    and s9, s9, a1
+; RV32I-NEXT:    and a1, a0, s8
+; RV32I-NEXT:    xor t3, t3, s9
+; RV32I-NEXT:    seqz s9, a1
+; RV32I-NEXT:    addi a2, s9, -1
+; RV32I-NEXT:    lui s9, 32768
+; RV32I-NEXT:    slli ra, s10, 26
+; RV32I-NEXT:    and a1, a0, s9
+; RV32I-NEXT:    and a2, a2, ra
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    xor t3, t3, a2
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    slli ra, s10, 27
+; RV32I-NEXT:    lui t4, 65536
+; RV32I-NEXT:    and a1, a1, ra
+; RV32I-NEXT:    and ra, a0, t4
+; RV32I-NEXT:    xor a1, t3, a1
+; RV32I-NEXT:    seqz t3, ra
+; RV32I-NEXT:    addi t3, t3, -1
+; RV32I-NEXT:    slli ra, s10, 28
+; RV32I-NEXT:    and ra, t3, ra
+; RV32I-NEXT:    lui t3, 131072
+; RV32I-NEXT:    xor a2, a1, ra
+; RV32I-NEXT:    and ra, a0, t3
+; RV32I-NEXT:    seqz a1, ra
+; RV32I-NEXT:    lui ra, 262144
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    and a0, a0, ra
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    slli s11, s10, 29
+; RV32I-NEXT:    and a1, a1, s11
+; RV32I-NEXT:    addi a0, a0, -1
+; RV32I-NEXT:    srli a5, a5, 31
+; RV32I-NEXT:    slli s11, s10, 30
+; RV32I-NEXT:    and a0, a0, s11
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    slli s10, s10, 31
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    and a1, a5, s10
+; RV32I-NEXT:    xor a2, a4, a2
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a0, a2, a0
+; RV32I-NEXT:    srli a1, a0, 8
+; RV32I-NEXT:    lw a2, 4(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a2
+; RV32I-NEXT:    and a2, a0, a2
+; RV32I-NEXT:    srli a4, a0, 24
+; RV32I-NEXT:    slli a0, a0, 24
+; RV32I-NEXT:    slli a2, a2, 8
+; RV32I-NEXT:    or a1, a1, a4
+; RV32I-NEXT:    or a0, a0, a2
+; RV32I-NEXT:    or a0, a0, a1
+; RV32I-NEXT:    srli a1, a0, 4
+; RV32I-NEXT:    and a0, a0, a6
+; RV32I-NEXT:    and a1, a1, a6
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    or a0, a1, a0
+; RV32I-NEXT:    srli a1, a0, 2
+; RV32I-NEXT:    and a0, a0, a7
+; RV32I-NEXT:    and a1, a1, a7
+; RV32I-NEXT:    slli a0, a0, 2
+; RV32I-NEXT:    or a1, a1, a0
+; RV32I-NEXT:    srli a0, a1, 1
+; RV32I-NEXT:    lui a2, 349525
+; RV32I-NEXT:    addi a2, a2, 1364
+; RV32I-NEXT:    and a0, a0, a2
+; RV32I-NEXT:    andi a2, a3, 2
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    andi a4, a3, 1
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    slli a5, t2, 1
+; RV32I-NEXT:    and a2, a2, a5
+; RV32I-NEXT:    and a4, a4, t2
+; RV32I-NEXT:    and a1, a1, t0
+; RV32I-NEXT:    xor a2, a4, a2
+; RV32I-NEXT:    andi a4, a3, 4
+; RV32I-NEXT:    andi a5, a3, 8
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a6, t2, 2
+; RV32I-NEXT:    slli a7, t2, 3
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    and a5, a5, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    andi a5, a3, 16
+; RV32I-NEXT:    xor a4, a2, a4
+; RV32I-NEXT:    seqz a2, a5
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    andi a5, a3, 32
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    slli a6, t2, 4
+; RV32I-NEXT:    and a2, a2, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a6, t2, 5
+; RV32I-NEXT:    andi a7, a3, 64
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    slli a7, t2, 6
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    and a5, a6, a7
+; RV32I-NEXT:    slli a1, a1, 1
+; RV32I-NEXT:    xor a5, a2, a5
+; RV32I-NEXT:    or a2, a0, a1
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    slli a0, t2, 7
+; RV32I-NEXT:    andi a1, a3, 128
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    andi a5, a3, 256
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a6, t2, 8
+; RV32I-NEXT:    and a0, a1, a0
+; RV32I-NEXT:    and a1, a5, a6
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    andi a1, a3, 512
+; RV32I-NEXT:    slli a5, t2, 9
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    andi a6, a3, 1024
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a6, t2, 10
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    and a1, a5, a6
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    and a1, a3, t1
+; RV32I-NEXT:    xor a0, a4, a0
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    lui a4, 1
+; RV32I-NEXT:    and a4, a3, a4
+; RV32I-NEXT:    slli a5, t2, 11
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    slli a5, t2, 12
+; RV32I-NEXT:    lui a6, 2
+; RV32I-NEXT:    and a6, a3, a6
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a6, t2, 13
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    and a4, a5, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lui a4, 4
+; RV32I-NEXT:    and a4, a3, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    seqz a1, a4
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    and a4, a3, t5
+; RV32I-NEXT:    slli a5, t2, 14
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    slli a5, t2, 15
+; RV32I-NEXT:    lui a6, 16
+; RV32I-NEXT:    and a6, a3, a6
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a4, t2, 16
+; RV32I-NEXT:    and a6, a3, t6
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a4, t2, 17
+; RV32I-NEXT:    and s1, a3, s1
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    seqz a5, s1
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a6, t2, 18
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    and a4, a5, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    and s0, a3, s0
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    seqz a1, s0
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    and a4, a3, s2
+; RV32I-NEXT:    slli a5, t2, 19
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    slli a5, t2, 20
+; RV32I-NEXT:    and a6, a3, s3
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a4, t2, 21
+; RV32I-NEXT:    and a6, a3, s4
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a4, t2, 22
+; RV32I-NEXT:    and a6, a3, s5
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a4, t2, 23
+; RV32I-NEXT:    and a6, a3, s6
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    lw a4, 8(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    and a5, a3, s7
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    and a5, a3, s8
+; RV32I-NEXT:    slli a6, t2, 25
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    slli a6, t2, 26
+; RV32I-NEXT:    and a7, a3, s9
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    slli a5, t2, 27
+; RV32I-NEXT:    and a7, a3, t4
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    slli a5, t2, 28
+; RV32I-NEXT:    and a7, a3, t3
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    slli a5, t2, 29
+; RV32I-NEXT:    and a7, a3, ra
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    slli a5, t2, 30
+; RV32I-NEXT:    srli a3, a3, 31
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    slli t2, t2, 31
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a3, a3, t2
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a3, a4, a3
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    srli a1, a2, 1
+; RV32I-NEXT:    lw ra, 60(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s0, 56(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s1, 52(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s2, 48(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 44(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s4, 40(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s5, 36(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 32(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 28(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 24(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s9, 20(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s10, 16(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s11, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    .cfi_restore ra
+; RV32I-NEXT:    .cfi_restore s0
+; RV32I-NEXT:    .cfi_restore s1
+; RV32I-NEXT:    .cfi_restore s2
+; RV32I-NEXT:    .cfi_restore s3
+; RV32I-NEXT:    .cfi_restore s4
+; RV32I-NEXT:    .cfi_restore s5
+; RV32I-NEXT:    .cfi_restore s6
+; RV32I-NEXT:    .cfi_restore s7
+; RV32I-NEXT:    .cfi_restore s8
+; RV32I-NEXT:    .cfi_restore s9
+; RV32I-NEXT:    .cfi_restore s10
+; RV32I-NEXT:    .cfi_restore s11
+; RV32I-NEXT:    addi sp, sp, 64
+; RV32I-NEXT:    .cfi_def_cfa_offset 0
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: clmul_i64_zext:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    slli a0, a0, 32
+; RV64I-NEXT:    andi a2, a1, 2
+; RV64I-NEXT:    srli a3, a0, 32
+; RV64I-NEXT:    srli a4, a0, 31
+; RV64I-NEXT:    andi a5, a1, 1
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    and a2, a2, a4
+; RV64I-NEXT:    and a3, a5, a3
+; RV64I-NEXT:    xor a2, a3, a2
+; RV64I-NEXT:    srli a3, a0, 30
+; RV64I-NEXT:    andi a4, a1, 4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    andi a5, a1, 8
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    srli a6, a0, 29
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    andi a4, a1, 16
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    seqz a3, a4
+; RV64I-NEXT:    srli a4, a0, 28
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    and a3, a3, a4
+; RV64I-NEXT:    andi a4, a1, 32
+; RV64I-NEXT:    srli a5, a0, 27
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    andi a6, a1, 64
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    srli a6, a0, 26
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    andi a4, a1, 128
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    seqz a3, a4
+; RV64I-NEXT:    srli a4, a0, 25
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    and a3, a3, a4
+; RV64I-NEXT:    andi a4, a1, 256
+; RV64I-NEXT:    srli a5, a0, 24
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    andi a6, a1, 512
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    srli a6, a0, 23
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    andi a4, a1, 1024
+; RV64I-NEXT:    srli a5, a0, 22
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    li a6, 1
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    slli a6, a6, 11
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a1, a6
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    seqz a3, a4
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    lui a4, 1
+; RV64I-NEXT:    srli a5, a0, 21
+; RV64I-NEXT:    and a4, a1, a4
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 2
+; RV64I-NEXT:    srli a6, a0, 20
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    srli a6, a0, 19
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    lui a4, 4
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    and a4, a1, a4
+; RV64I-NEXT:    seqz a3, a4
+; RV64I-NEXT:    lui a4, 8
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    and a4, a1, a4
+; RV64I-NEXT:    srli a5, a0, 18
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    srli a5, a0, 17
+; RV64I-NEXT:    lui a6, 16
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    and a5, a1, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 32
+; RV64I-NEXT:    srli a6, a0, 16
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    srli a4, a0, 15
+; RV64I-NEXT:    lui a6, 64
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    and a5, a1, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    srli a5, a0, 14
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    lui a5, 128
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    seqz a3, a5
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    lui a4, 256
+; RV64I-NEXT:    srli a5, a0, 13
+; RV64I-NEXT:    and a4, a1, a4
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 512
+; RV64I-NEXT:    srli a6, a0, 12
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    srli a4, a0, 11
+; RV64I-NEXT:    lui a6, 1024
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    and a5, a1, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 2048
+; RV64I-NEXT:    srli a6, a0, 10
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    srli a4, a0, 9
+; RV64I-NEXT:    lui a6, 4096
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    and a5, a1, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    srli a5, a0, 8
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    lui a5, 8192
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    lui a5, 16384
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, a1, a5
+; RV64I-NEXT:    srli a6, a0, 7
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    srli a6, a0, 6
+; RV64I-NEXT:    lui a7, 32768
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a6, 65536
+; RV64I-NEXT:    srli a7, a0, 5
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    srli a5, a0, 4
+; RV64I-NEXT:    lui a7, 131072
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a6, 262144
+; RV64I-NEXT:    srli a7, a0, 3
+; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    srli a5, a0, 2
+; RV64I-NEXT:    sraiw a1, a1, 31
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    seqz a1, a1
+; RV64I-NEXT:    srli a0, a0, 1
+; RV64I-NEXT:    addi a1, a1, -1
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a0, a1, a0
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    xor a0, a4, a0
+; RV64I-NEXT:    xor a0, a2, a0
+; RV64I-NEXT:    ret
+;
+; RV32IM-LABEL: clmul_i64_zext:
+; RV32IM:       # %bb.0:
+; RV32IM-NEXT:    addi sp, sp, -64
+; RV32IM-NEXT:    .cfi_def_cfa_offset 64
+; RV32IM-NEXT:    sw ra, 60(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s0, 56(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s1, 52(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s2, 48(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s3, 44(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s4, 40(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s5, 36(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s6, 32(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s7, 28(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s8, 24(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s9, 20(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s10, 16(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s11, 12(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    .cfi_offset ra, -4
+; RV32IM-NEXT:    .cfi_offset s0, -8
+; RV32IM-NEXT:    .cfi_offset s1, -12
+; RV32IM-NEXT:    .cfi_offset s2, -16
+; RV32IM-NEXT:    .cfi_offset s3, -20
+; RV32IM-NEXT:    .cfi_offset s4, -24
+; RV32IM-NEXT:    .cfi_offset s5, -28
+; RV32IM-NEXT:    .cfi_offset s6, -32
+; RV32IM-NEXT:    .cfi_offset s7, -36
+; RV32IM-NEXT:    .cfi_offset s8, -40
+; RV32IM-NEXT:    .cfi_offset s9, -44
+; RV32IM-NEXT:    .cfi_offset s10, -48
+; RV32IM-NEXT:    .cfi_offset s11, -52
+; RV32IM-NEXT:    mv a4, a0
+; RV32IM-NEXT:    lui a2, 16
+; RV32IM-NEXT:    srli a3, a1, 8
+; RV32IM-NEXT:    addi t1, a2, -256
+; RV32IM-NEXT:    and a3, a3, t1
+; RV32IM-NEXT:    srli a2, a1, 24
+; RV32IM-NEXT:    and a5, a1, t1
+; RV32IM-NEXT:    slli a5, a5, 8
+; RV32IM-NEXT:    slli a6, a1, 24
+; RV32IM-NEXT:    or a2, a3, a2
+; RV32IM-NEXT:    or a3, a6, a5
+; RV32IM-NEXT:    or a2, a3, a2
+; RV32IM-NEXT:    lui a3, 61681
+; RV32IM-NEXT:    srli a5, a2, 4
+; RV32IM-NEXT:    addi a6, a3, -241
+; RV32IM-NEXT:    and a3, a5, a6
+; RV32IM-NEXT:    and a2, a2, a6
+; RV32IM-NEXT:    slli a2, a2, 4
+; RV32IM-NEXT:    lui a5, 209715
+; RV32IM-NEXT:    or a2, a3, a2
+; RV32IM-NEXT:    addi t0, a5, 819
+; RV32IM-NEXT:    srli a3, a2, 2
+; RV32IM-NEXT:    and a2, a2, t0
+; RV32IM-NEXT:    and a3, a3, t0
+; RV32IM-NEXT:    slli a2, a2, 2
+; RV32IM-NEXT:    or a2, a3, a2
+; RV32IM-NEXT:    lui a0, 349525
+; RV32IM-NEXT:    srli a3, a2, 1
+; RV32IM-NEXT:    addi t2, a0, 1365
+; RV32IM-NEXT:    and a3, a3, t2
+; RV32IM-NEXT:    srli a5, a4, 8
+; RV32IM-NEXT:    and a2, a2, t2
+; RV32IM-NEXT:    and a5, a5, t1
+; RV32IM-NEXT:    srli a7, a4, 24
+; RV32IM-NEXT:    and t3, a4, t1
+; RV32IM-NEXT:    slli t3, t3, 8
+; RV32IM-NEXT:    slli t4, a4, 24
+; RV32IM-NEXT:    or a5, a5, a7
+; RV32IM-NEXT:    or a7, t4, t3
+; RV32IM-NEXT:    slli a2, a2, 1
+; RV32IM-NEXT:    or a5, a7, a5
+; RV32IM-NEXT:    srli a7, a5, 4
+; RV32IM-NEXT:    and a5, a5, a6
+; RV32IM-NEXT:    and a7, a7, a6
+; RV32IM-NEXT:    slli a5, a5, 4
+; RV32IM-NEXT:    or t3, a3, a2
+; RV32IM-NEXT:    or a2, a7, a5
+; RV32IM-NEXT:    srli a3, a2, 2
+; RV32IM-NEXT:    and a2, a2, t0
+; RV32IM-NEXT:    and a3, a3, t0
+; RV32IM-NEXT:    slli a2, a2, 2
+; RV32IM-NEXT:    lui a5, 69905
+; RV32IM-NEXT:    or a3, a3, a2
+; RV32IM-NEXT:    addi a2, a5, 273
+; RV32IM-NEXT:    srli a5, a3, 1
+; RV32IM-NEXT:    and a5, a5, t2
+; RV32IM-NEXT:    and a3, a3, t2
+; RV32IM-NEXT:    slli a3, a3, 1
+; RV32IM-NEXT:    lui a7, 139810
+; RV32IM-NEXT:    or t4, a5, a3
+; RV32IM-NEXT:    addi a3, a7, 546
+; RV32IM-NEXT:    and t5, t3, a2
+; RV32IM-NEXT:    and t6, t4, a3
+; RV32IM-NEXT:    and s0, t3, a3
+; RV32IM-NEXT:    and s1, t4, a2
+; RV32IM-NEXT:    mul s2, t6, t5
+; RV32IM-NEXT:    mul s3, s1, s0
+; RV32IM-NEXT:    lui a5, 559241
+; RV32IM-NEXT:    lui a7, 279620
+; RV32IM-NEXT:    addi a5, a5, -1912
+; RV32IM-NEXT:    addi a7, a7, 1092
+; RV32IM-NEXT:    and s4, t3, a5
+; RV32IM-NEXT:    and s5, t4, a7
+; RV32IM-NEXT:    and t3, t3, a7
+; RV32IM-NEXT:    and t4, t4, a5
+; RV32IM-NEXT:    mul s6, s5, s4
+; RV32IM-NEXT:    mul s7, t4, t3
+; RV32IM-NEXT:    mul s8, t6, s4
+; RV32IM-NEXT:    mul s9, s1, t5
+; RV32IM-NEXT:    mul s10, s5, t3
+; RV32IM-NEXT:    mul s11, t4, s0
+; RV32IM-NEXT:    mul ra, t6, s0
+; RV32IM-NEXT:    mul t6, t6, t3
+; RV32IM-NEXT:    mul t3, s1, t3
+; RV32IM-NEXT:    mul s1, s1, s4
+; RV32IM-NEXT:    mul s4, t4, s4
+; RV32IM-NEXT:    mul a0, s5, t5
+; RV32IM-NEXT:    mul s0, s5, s0
+; RV32IM-NEXT:    mul t4, t4, t5
+; RV32IM-NEXT:    xor t5, s3, s2
+; RV32IM-NEXT:    xor s2, s6, s7
+; RV32IM-NEXT:    xor s3, s9, s8
+; RV32IM-NEXT:    xor s5, s10, s11
+; RV32IM-NEXT:    xor t5, t5, s2
+; RV32IM-NEXT:    xor s2, s3, s5
+; RV32IM-NEXT:    and t5, t5, a3
+; RV32IM-NEXT:    and s2, s2, a2
+; RV32IM-NEXT:    xor t3, t3, ra
+; RV32IM-NEXT:    xor a0, a0, s4
+; RV32IM-NEXT:    xor t6, s1, t6
+; RV32IM-NEXT:    xor t4, s0, t4
+; RV32IM-NEXT:    xor a0, t3, a0
+; RV32IM-NEXT:    xor t3, t6, t4
+; RV32IM-NEXT:    and a0, a0, a7
+; RV32IM-NEXT:    and t3, t3, a5
+; RV32IM-NEXT:    or t4, s2, t5
+; RV32IM-NEXT:    or a0, a0, t3
+; RV32IM-NEXT:    or a0, t4, a0
+; RV32IM-NEXT:    srli t3, a0, 8
+; RV32IM-NEXT:    and t3, t3, t1
+; RV32IM-NEXT:    srli t4, a0, 24
+; RV32IM-NEXT:    and t1, a0, t1
+; RV32IM-NEXT:    slli a0, a0, 24
+; RV32IM-NEXT:    slli t1, t1, 8
+; RV32IM-NEXT:    or t3, t3, t4
+; RV32IM-NEXT:    or a0, a0, t1
+; RV32IM-NEXT:    or a0, a0, t3
+; RV32IM-NEXT:    srli t1, a0, 4
+; RV32IM-NEXT:    and a0, a0, a6
+; RV32IM-NEXT:    and a6, t1, a6
+; RV32IM-NEXT:    slli a0, a0, 4
+; RV32IM-NEXT:    or a0, a6, a0
+; RV32IM-NEXT:    srli a6, a0, 2
+; RV32IM-NEXT:    and a0, a0, t0
+; RV32IM-NEXT:    and a6, a6, t0
+; RV32IM-NEXT:    slli a0, a0, 2
+; RV32IM-NEXT:    or a0, a6, a0
+; RV32IM-NEXT:    srli a6, a0, 1
+; RV32IM-NEXT:    lui t0, 349525
+; RV32IM-NEXT:    addi t0, t0, 1364
+; RV32IM-NEXT:    and a0, a0, t2
+; RV32IM-NEXT:    and a6, a6, t0
+; RV32IM-NEXT:    slli a0, a0, 1
+; RV32IM-NEXT:    or a0, a6, a0
+; RV32IM-NEXT:    and a6, a1, a2
+; RV32IM-NEXT:    and t0, a4, a3
+; RV32IM-NEXT:    and t1, a1, a3
+; RV32IM-NEXT:    and t2, a4, a2
+; RV32IM-NEXT:    mul t3, t0, a6
+; RV32IM-NEXT:    mul t4, t2, t1
+; RV32IM-NEXT:    and t5, a1, a5
+; RV32IM-NEXT:    and t6, a4, a7
+; RV32IM-NEXT:    and a1, a1, a7
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    mul s0, t6, t5
+; RV32IM-NEXT:    mul s1, a4, a1
+; RV32IM-NEXT:    mul s2, t0, t5
+; RV32IM-NEXT:    mul s3, t2, a6
+; RV32IM-NEXT:    mul s4, t6, a1
+; RV32IM-NEXT:    mul s5, a4, t1
+; RV32IM-NEXT:    mul s6, t0, t1
+; RV32IM-NEXT:    mul s7, t2, a1
+; RV32IM-NEXT:    mul a1, t0, a1
+; RV32IM-NEXT:    mul t0, t6, a6
+; RV32IM-NEXT:    mul t2, t2, t5
+; RV32IM-NEXT:    mul t5, a4, t5
+; RV32IM-NEXT:    mul t1, t6, t1
+; RV32IM-NEXT:    mul a4, a4, a6
+; RV32IM-NEXT:    xor a6, t4, t3
+; RV32IM-NEXT:    xor s0, s0, s1
+; RV32IM-NEXT:    xor t3, s3, s2
+; RV32IM-NEXT:    xor t4, s4, s5
+; RV32IM-NEXT:    xor a6, a6, s0
+; RV32IM-NEXT:    xor t3, t3, t4
+; RV32IM-NEXT:    and a3, a6, a3
+; RV32IM-NEXT:    and a2, t3, a2
+; RV32IM-NEXT:    xor a6, s7, s6
+; RV32IM-NEXT:    xor t0, t0, t5
+; RV32IM-NEXT:    xor a1, t2, a1
+; RV32IM-NEXT:    xor a4, t1, a4
+; RV32IM-NEXT:    xor a6, a6, t0
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    and a4, a6, a7
+; RV32IM-NEXT:    and a1, a1, a5
+; RV32IM-NEXT:    or a2, a2, a3
+; RV32IM-NEXT:    or a4, a4, a1
+; RV32IM-NEXT:    srli a1, a0, 1
+; RV32IM-NEXT:    or a0, a2, a4
+; RV32IM-NEXT:    lw ra, 60(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s0, 56(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s1, 52(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s2, 48(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s3, 44(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s4, 40(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s5, 36(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s6, 32(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s7, 28(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s8, 24(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s9, 20(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s10, 16(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s11, 12(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    .cfi_restore ra
+; RV32IM-NEXT:    .cfi_restore s0
+; RV32IM-NEXT:    .cfi_restore s1
+; RV32IM-NEXT:    .cfi_restore s2
+; RV32IM-NEXT:    .cfi_restore s3
+; RV32IM-NEXT:    .cfi_restore s4
+; RV32IM-NEXT:    .cfi_restore s5
+; RV32IM-NEXT:    .cfi_restore s6
+; RV32IM-NEXT:    .cfi_restore s7
+; RV32IM-NEXT:    .cfi_restore s8
+; RV32IM-NEXT:    .cfi_restore s9
+; RV32IM-NEXT:    .cfi_restore s10
+; RV32IM-NEXT:    .cfi_restore s11
+; RV32IM-NEXT:    addi sp, sp, 64
+; RV32IM-NEXT:    .cfi_def_cfa_offset 0
+; RV32IM-NEXT:    ret
+;
+; RV64IM-LABEL: clmul_i64_zext:
+; RV64IM:       # %bb.0:
+; RV64IM-NEXT:    addi sp, sp, -48
+; RV64IM-NEXT:    .cfi_def_cfa_offset 48
+; RV64IM-NEXT:    sd s0, 40(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s1, 32(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s2, 24(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s3, 16(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s4, 8(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    .cfi_offset s0, -8
+; RV64IM-NEXT:    .cfi_offset s1, -16
+; RV64IM-NEXT:    .cfi_offset s2, -24
+; RV64IM-NEXT:    .cfi_offset s3, -32
+; RV64IM-NEXT:    .cfi_offset s4, -40
+; RV64IM-NEXT:    lui a2, %hi(.LCPI6_0)
+; RV64IM-NEXT:    ld a2, %lo(.LCPI6_0)(a2)
+; RV64IM-NEXT:    lui a3, 69905
+; RV64IM-NEXT:    lui a4, 139810
+; RV64IM-NEXT:    addi a3, a3, 273
+; RV64IM-NEXT:    addi a6, a4, 546
+; RV64IM-NEXT:    and a5, a1, a3
+; RV64IM-NEXT:    and a7, a0, a6
+; RV64IM-NEXT:    and t0, a0, a2
+; RV64IM-NEXT:    and t1, a1, a2
+; RV64IM-NEXT:    mul t2, a7, a5
+; RV64IM-NEXT:    and t3, a1, a6
+; RV64IM-NEXT:    and t4, a0, a3
+; RV64IM-NEXT:    mul t5, t4, t3
+; RV64IM-NEXT:    lui a4, 279620
+; RV64IM-NEXT:    addi a4, a4, 1092
+; RV64IM-NEXT:    slli t1, t1, 32
+; RV64IM-NEXT:    and a0, a0, a4
+; RV64IM-NEXT:    srli t6, t1, 32
+; RV64IM-NEXT:    mul s0, a0, t6
+; RV64IM-NEXT:    slli t0, t0, 32
+; RV64IM-NEXT:    and a1, a1, a4
+; RV64IM-NEXT:    srli s1, t0, 32
+; RV64IM-NEXT:    mul s2, s1, a1
+; RV64IM-NEXT:    slli s3, a6, 32
+; RV64IM-NEXT:    xor t2, t5, t2
+; RV64IM-NEXT:    mul t5, t4, a5
+; RV64IM-NEXT:    xor t2, t2, s0
+; RV64IM-NEXT:    mul s0, a0, a1
+; RV64IM-NEXT:    xor t2, t2, s2
+; RV64IM-NEXT:    add a6, a6, s3
+; RV64IM-NEXT:    mul s2, a7, t6
+; RV64IM-NEXT:    and a6, t2, a6
+; RV64IM-NEXT:    mul t2, s1, t3
+; RV64IM-NEXT:    mul s3, a7, t3
+; RV64IM-NEXT:    mul s4, t4, a1
+; RV64IM-NEXT:    mulhu t0, t0, t1
+; RV64IM-NEXT:    mul t1, a0, a5
+; RV64IM-NEXT:    mul a1, a7, a1
+; RV64IM-NEXT:    mul a0, a0, t3
+; RV64IM-NEXT:    xor a7, t5, s0
+; RV64IM-NEXT:    mul t3, t4, t6
+; RV64IM-NEXT:    xor a7, a7, s2
+; RV64IM-NEXT:    mul a5, s1, a5
+; RV64IM-NEXT:    xor a7, a7, t2
+; RV64IM-NEXT:    slli t2, a3, 32
+; RV64IM-NEXT:    add a3, a3, t2
+; RV64IM-NEXT:    xor t2, s4, s3
+; RV64IM-NEXT:    and a3, a7, a3
+; RV64IM-NEXT:    xor a7, t2, t1
+; RV64IM-NEXT:    xor a7, a7, t0
+; RV64IM-NEXT:    xor a0, a1, a0
+; RV64IM-NEXT:    slli a1, a4, 32
+; RV64IM-NEXT:    xor a0, t3, a0
+; RV64IM-NEXT:    add a1, a4, a1
+; RV64IM-NEXT:    xor a0, a0, a5
+; RV64IM-NEXT:    and a1, a7, a1
+; RV64IM-NEXT:    and a0, a0, a2
+; RV64IM-NEXT:    or a2, a3, a6
+; RV64IM-NEXT:    or a0, a1, a0
+; RV64IM-NEXT:    or a0, a2, a0
+; RV64IM-NEXT:    ld s0, 40(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s1, 32(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s2, 24(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s3, 16(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s4, 8(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    .cfi_restore s0
+; RV64IM-NEXT:    .cfi_restore s1
+; RV64IM-NEXT:    .cfi_restore s2
+; RV64IM-NEXT:    .cfi_restore s3
+; RV64IM-NEXT:    .cfi_restore s4
+; RV64IM-NEXT:    addi sp, sp, 48
+; RV64IM-NEXT:    .cfi_def_cfa_offset 0
+; RV64IM-NEXT:    ret
+;
+; RV32IMZBS-LABEL: clmul_i64_zext:
+; RV32IMZBS:       # %bb.0:
+; RV32IMZBS-NEXT:    addi sp, sp, -64
+; RV32IMZBS-NEXT:    .cfi_def_cfa_offset 64
+; RV32IMZBS-NEXT:    sw ra, 60(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s0, 56(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s1, 52(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s2, 48(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s3, 44(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s4, 40(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s5, 36(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s6, 32(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s7, 28(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s8, 24(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s9, 20(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s10, 16(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s11, 12(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    .cfi_offset ra, -4
+; RV32IMZBS-NEXT:    .cfi_offset s0, -8
+; RV32IMZBS-NEXT:    .cfi_offset s1, -12
+; RV32IMZBS-NEXT:    .cfi_offset s2, -16
+; RV32IMZBS-NEXT:    .cfi_offset s3, -20
+; RV32IMZBS-NEXT:    .cfi_offset s4, -24
+; RV32IMZBS-NEXT:    .cfi_offset s5, -28
+; RV32IMZBS-NEXT:    .cfi_offset s6, -32
+; RV32IMZBS-NEXT:    .cfi_offset s7, -36
+; RV32IMZBS-NEXT:    .cfi_offset s8, -40
+; RV32IMZBS-NEXT:    .cfi_offset s9, -44
+; RV32IMZBS-NEXT:    .cfi_offset s10, -48
+; RV32IMZBS-NEXT:    .cfi_offset s11, -52
+; RV32IMZBS-NEXT:    mv a4, a0
+; RV32IMZBS-NEXT:    lui a2, 16
+; RV32IMZBS-NEXT:    srli a3, a1, 8
+; RV32IMZBS-NEXT:    addi t1, a2, -256
+; RV32IMZBS-NEXT:    and a3, a3, t1
+; RV32IMZBS-NEXT:    srli a2, a1, 24
+; RV32IMZBS-NEXT:    and a5, a1, t1
+; RV32IMZBS-NEXT:    slli a5, a5, 8
+; RV32IMZBS-NEXT:    slli a6, a1, 24
+; RV32IMZBS-NEXT:    or a2, a3, a2
+; RV32IMZBS-NEXT:    or a3, a6, a5
+; RV32IMZBS-NEXT:    or a2, a3, a2
+; RV32IMZBS-NEXT:    lui a3, 61681
+; RV32IMZBS-NEXT:    srli a5, a2, 4
+; RV32IMZBS-NEXT:    addi a6, a3, -241
+; RV32IMZBS-NEXT:    and a3, a5, a6
+; RV32IMZBS-NEXT:    and a2, a2, a6
+; RV32IMZBS-NEXT:    slli a2, a2, 4
+; RV32IMZBS-NEXT:    lui a5, 209715
+; RV32IMZBS-NEXT:    or a2, a3, a2
+; RV32IMZBS-NEXT:    addi t0, a5, 819
+; RV32IMZBS-NEXT:    srli a3, a2, 2
+; RV32IMZBS-NEXT:    and a2, a2, t0
+; RV32IMZBS-NEXT:    and a3, a3, t0
+; RV32IMZBS-NEXT:    slli a2, a2, 2
+; RV32IMZBS-NEXT:    or a2, a3, a2
+; RV32IMZBS-NEXT:    lui a0, 349525
+; RV32IMZBS-NEXT:    srli a3, a2, 1
+; RV32IMZBS-NEXT:    addi t2, a0, 1365
+; RV32IMZBS-NEXT:    and a3, a3, t2
+; RV32IMZBS-NEXT:    srli a5, a4, 8
+; RV32IMZBS-NEXT:    and a2, a2, t2
+; RV32IMZBS-NEXT:    and a5, a5, t1
+; RV32IMZBS-NEXT:    srli a7, a4, 24
+; RV32IMZBS-NEXT:    and t3, a4, t1
+; RV32IMZBS-NEXT:    slli t3, t3, 8
+; RV32IMZBS-NEXT:    slli t4, a4, 24
+; RV32IMZBS-NEXT:    or a5, a5, a7
+; RV32IMZBS-NEXT:    or a7, t4, t3
+; RV32IMZBS-NEXT:    slli a2, a2, 1
+; RV32IMZBS-NEXT:    or a5, a7, a5
+; RV32IMZBS-NEXT:    srli a7, a5, 4
+; RV32IMZBS-NEXT:    and a5, a5, a6
+; RV32IMZBS-NEXT:    and a7, a7, a6
+; RV32IMZBS-NEXT:    slli a5, a5, 4
+; RV32IMZBS-NEXT:    or t3, a3, a2
+; RV32IMZBS-NEXT:    or a2, a7, a5
+; RV32IMZBS-NEXT:    srli a3, a2, 2
+; RV32IMZBS-NEXT:    and a2, a2, t0
+; RV32IMZBS-NEXT:    and a3, a3, t0
+; RV32IMZBS-NEXT:    slli a2, a2, 2
+; RV32IMZBS-NEXT:    lui a5, 69905
+; RV32IMZBS-NEXT:    or a3, a3, a2
+; RV32IMZBS-NEXT:    addi a2, a5, 273
+; RV32IMZBS-NEXT:    srli a5, a3, 1
+; RV32IMZBS-NEXT:    and a5, a5, t2
+; RV32IMZBS-NEXT:    and a3, a3, t2
+; RV32IMZBS-NEXT:    slli a3, a3, 1
+; RV32IMZBS-NEXT:    lui a7, 139810
+; RV32IMZBS-NEXT:    or t4, a5, a3
+; RV32IMZBS-NEXT:    addi a3, a7, 546
+; RV32IMZBS-NEXT:    and t5, t3, a2
+; RV32IMZBS-NEXT:    and t6, t4, a3
+; RV32IMZBS-NEXT:    and s0, t3, a3
+; RV32IMZBS-NEXT:    and s1, t4, a2
+; RV32IMZBS-NEXT:    mul s2, t6, t5
+; RV32IMZBS-NEXT:    mul s3, s1, s0
+; RV32IMZBS-NEXT:    lui a5, 559241
+; RV32IMZBS-NEXT:    lui a7, 279620
+; RV32IMZBS-NEXT:    addi a5, a5, -1912
+; RV32IMZBS-NEXT:    addi a7, a7, 1092
+; RV32IMZBS-NEXT:    and s4, t3, a5
+; RV32IMZBS-NEXT:    and s5, t4, a7
+; RV32IMZBS-NEXT:    and t3, t3, a7
+; RV32IMZBS-NEXT:    and t4, t4, a5
+; RV32IMZBS-NEXT:    mul s6, s5, s4
+; RV32IMZBS-NEXT:    mul s7, t4, t3
+; RV32IMZBS-NEXT:    mul s8, t6, s4
+; RV32IMZBS-NEXT:    mul s9, s1, t5
+; RV32IMZBS-NEXT:    mul s10, s5, t3
+; RV32IMZBS-NEXT:    mul s11, t4, s0
+; RV32IMZBS-NEXT:    mul ra, t6, s0
+; RV32IMZBS-NEXT:    mul t6, t6, t3
+; RV32IMZBS-NEXT:    mul t3, s1, t3
+; RV32IMZBS-NEXT:    mul s1, s1, s4
+; RV32IMZBS-NEXT:    mul s4, t4, s4
+; RV32IMZBS-NEXT:    mul a0, s5, t5
+; RV32IMZBS-NEXT:    mul s0, s5, s0
+; RV32IMZBS-NEXT:    mul t4, t4, t5
+; RV32IMZBS-NEXT:    xor t5, s3, s2
+; RV32IMZBS-NEXT:    xor s2, s6, s7
+; RV32IMZBS-NEXT:    xor s3, s9, s8
+; RV32IMZBS-NEXT:    xor s5, s10, s11
+; RV32IMZBS-NEXT:    xor t5, t5, s2
+; RV32IMZBS-NEXT:    xor s2, s3, s5
+; RV32IMZBS-NEXT:    and t5, t5, a3
+; RV32IMZBS-NEXT:    and s2, s2, a2
+; RV32IMZBS-NEXT:    xor t3, t3, ra
+; RV32IMZBS-NEXT:    xor a0, a0, s4
+; RV32IMZBS-NEXT:    xor t6, s1, t6
+; RV32IMZBS-NEXT:    xor t4, s0, t4
+; RV32IMZBS-NEXT:    xor a0, t3, a0
+; RV32IMZBS-NEXT:    xor t3, t6, t4
+; RV32IMZBS-NEXT:    and a0, a0, a7
+; RV32IMZBS-NEXT:    and t3, t3, a5
+; RV32IMZBS-NEXT:    or t4, s2, t5
+; RV32IMZBS-NEXT:    or a0, a0, t3
+; RV32IMZBS-NEXT:    or a0, t4, a0
+; RV32IMZBS-NEXT:    srli t3, a0, 8
+; RV32IMZBS-NEXT:    and t3, t3, t1
+; RV32IMZBS-NEXT:    srli t4, a0, 24
+; RV32IMZBS-NEXT:    and t1, a0, t1
+; RV32IMZBS-NEXT:    slli a0, a0, 24
+; RV32IMZBS-NEXT:    slli t1, t1, 8
+; RV32IMZBS-NEXT:    or t3, t3, t4
+; RV32IMZBS-NEXT:    or a0, a0, t1
+; RV32IMZBS-NEXT:    or a0, a0, t3
+; RV32IMZBS-NEXT:    srli t1, a0, 4
+; RV32IMZBS-NEXT:    and a0, a0, a6
+; RV32IMZBS-NEXT:    and a6, t1, a6
+; RV32IMZBS-NEXT:    slli a0, a0, 4
+; RV32IMZBS-NEXT:    or a0, a6, a0
+; RV32IMZBS-NEXT:    srli a6, a0, 2
+; RV32IMZBS-NEXT:    and a0, a0, t0
+; RV32IMZBS-NEXT:    and a6, a6, t0
+; RV32IMZBS-NEXT:    slli a0, a0, 2
+; RV32IMZBS-NEXT:    or a0, a6, a0
+; RV32IMZBS-NEXT:    srli a6, a0, 1
+; RV32IMZBS-NEXT:    lui t0, 349525
+; RV32IMZBS-NEXT:    addi t0, t0, 1364
+; RV32IMZBS-NEXT:    and a0, a0, t2
+; RV32IMZBS-NEXT:    and a6, a6, t0
+; RV32IMZBS-NEXT:    slli a0, a0, 1
+; RV32IMZBS-NEXT:    or a0, a6, a0
+; RV32IMZBS-NEXT:    and a6, a1, a2
+; RV32IMZBS-NEXT:    and t0, a4, a3
+; RV32IMZBS-NEXT:    and t1, a1, a3
+; RV32IMZBS-NEXT:    and t2, a4, a2
+; RV32IMZBS-NEXT:    mul t3, t0, a6
+; RV32IMZBS-NEXT:    mul t4, t2, t1
+; RV32IMZBS-NEXT:    and t5, a1, a5
+; RV32IMZBS-NEXT:    and t6, a4, a7
+; RV32IMZBS-NEXT:    and a1, a1, a7
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    mul s0, t6, t5
+; RV32IMZBS-NEXT:    mul s1, a4, a1
+; RV32IMZBS-NEXT:    mul s2, t0, t5
+; RV32IMZBS-NEXT:    mul s3, t2, a6
+; RV32IMZBS-NEXT:    mul s4, t6, a1
+; RV32IMZBS-NEXT:    mul s5, a4, t1
+; RV32IMZBS-NEXT:    mul s6, t0, t1
+; RV32IMZBS-NEXT:    mul s7, t2, a1
+; RV32IMZBS-NEXT:    mul a1, t0, a1
+; RV32IMZBS-NEXT:    mul t0, t6, a6
+; RV32IMZBS-NEXT:    mul t2, t2, t5
+; RV32IMZBS-NEXT:    mul t5, a4, t5
+; RV32IMZBS-NEXT:    mul t1, t6, t1
+; RV32IMZBS-NEXT:    mul a4, a4, a6
+; RV32IMZBS-NEXT:    xor a6, t4, t3
+; RV32IMZBS-NEXT:    xor s0, s0, s1
+; RV32IMZBS-NEXT:    xor t3, s3, s2
+; RV32IMZBS-NEXT:    xor t4, s4, s5
+; RV32IMZBS-NEXT:    xor a6, a6, s0
+; RV32IMZBS-NEXT:    xor t3, t3, t4
+; RV32IMZBS-NEXT:    and a3, a6, a3
+; RV32IMZBS-NEXT:    and a2, t3, a2
+; RV32IMZBS-NEXT:    xor a6, s7, s6
+; RV32IMZBS-NEXT:    xor t0, t0, t5
+; RV32IMZBS-NEXT:    xor a1, t2, a1
+; RV32IMZBS-NEXT:    xor a4, t1, a4
+; RV32IMZBS-NEXT:    xor a6, a6, t0
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    and a4, a6, a7
+; RV32IMZBS-NEXT:    and a1, a1, a5
+; RV32IMZBS-NEXT:    or a2, a2, a3
+; RV32IMZBS-NEXT:    or a4, a4, a1
+; RV32IMZBS-NEXT:    srli a1, a0, 1
+; RV32IMZBS-NEXT:    or a0, a2, a4
+; RV32IMZBS-NEXT:    lw ra, 60(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s0, 56(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s1, 52(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s2, 48(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s3, 44(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s4, 40(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s5, 36(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s6, 32(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s7, 28(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s8, 24(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s9, 20(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s10, 16(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s11, 12(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    .cfi_restore ra
+; RV32IMZBS-NEXT:    .cfi_restore s0
+; RV32IMZBS-NEXT:    .cfi_restore s1
+; RV32IMZBS-NEXT:    .cfi_restore s2
+; RV32IMZBS-NEXT:    .cfi_restore s3
+; RV32IMZBS-NEXT:    .cfi_restore s4
+; RV32IMZBS-NEXT:    .cfi_restore s5
+; RV32IMZBS-NEXT:    .cfi_restore s6
+; RV32IMZBS-NEXT:    .cfi_restore s7
+; RV32IMZBS-NEXT:    .cfi_restore s8
+; RV32IMZBS-NEXT:    .cfi_restore s9
+; RV32IMZBS-NEXT:    .cfi_restore s10
+; RV32IMZBS-NEXT:    .cfi_restore s11
+; RV32IMZBS-NEXT:    addi sp, sp, 64
+; RV32IMZBS-NEXT:    .cfi_def_cfa_offset 0
+; RV32IMZBS-NEXT:    ret
+;
+; RV64IMZBS-LABEL: clmul_i64_zext:
+; RV64IMZBS:       # %bb.0:
+; RV64IMZBS-NEXT:    addi sp, sp, -48
+; RV64IMZBS-NEXT:    .cfi_def_cfa_offset 48
+; RV64IMZBS-NEXT:    sd s0, 40(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s1, 32(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s2, 24(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s3, 16(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s4, 8(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    .cfi_offset s0, -8
+; RV64IMZBS-NEXT:    .cfi_offset s1, -16
+; RV64IMZBS-NEXT:    .cfi_offset s2, -24
+; RV64IMZBS-NEXT:    .cfi_offset s3, -32
+; RV64IMZBS-NEXT:    .cfi_offset s4, -40
+; RV64IMZBS-NEXT:    lui a2, %hi(.LCPI6_0)
+; RV64IMZBS-NEXT:    ld a2, %lo(.LCPI6_0)(a2)
+; RV64IMZBS-NEXT:    lui a3, 69905
+; RV64IMZBS-NEXT:    lui a4, 139810
+; RV64IMZBS-NEXT:    addi a3, a3, 273
+; RV64IMZBS-NEXT:    addi a6, a4, 546
+; RV64IMZBS-NEXT:    and a5, a1, a3
+; RV64IMZBS-NEXT:    and a7, a0, a6
+; RV64IMZBS-NEXT:    and t0, a0, a2
+; RV64IMZBS-NEXT:    and t1, a1, a2
+; RV64IMZBS-NEXT:    mul t2, a7, a5
+; RV64IMZBS-NEXT:    and t3, a1, a6
+; RV64IMZBS-NEXT:    and t4, a0, a3
+; RV64IMZBS-NEXT:    mul t5, t4, t3
+; RV64IMZBS-NEXT:    lui a4, 279620
+; RV64IMZBS-NEXT:    addi a4, a4, 1092
+; RV64IMZBS-NEXT:    slli t1, t1, 32
+; RV64IMZBS-NEXT:    and a0, a0, a4
+; RV64IMZBS-NEXT:    srli t6, t1, 32
+; RV64IMZBS-NEXT:    mul s0, a0, t6
+; RV64IMZBS-NEXT:    slli t0, t0, 32
+; RV64IMZBS-NEXT:    and a1, a1, a4
+; RV64IMZBS-NEXT:    srli s1, t0, 32
+; RV64IMZBS-NEXT:    mul s2, s1, a1
+; RV64IMZBS-NEXT:    slli s3, a6, 32
+; RV64IMZBS-NEXT:    xor t2, t5, t2
+; RV64IMZBS-NEXT:    mul t5, t4, a5
+; RV64IMZBS-NEXT:    xor t2, t2, s0
+; RV64IMZBS-NEXT:    mul s0, a0, a1
+; RV64IMZBS-NEXT:    xor t2, t2, s2
+; RV64IMZBS-NEXT:    add a6, a6, s3
+; RV64IMZBS-NEXT:    mul s2, a7, t6
+; RV64IMZBS-NEXT:    and a6, t2, a6
+; RV64IMZBS-NEXT:    mul t2, s1, t3
+; RV64IMZBS-NEXT:    mul s3, a7, t3
+; RV64IMZBS-NEXT:    mul s4, t4, a1
+; RV64IMZBS-NEXT:    mulhu t0, t0, t1
+; RV64IMZBS-NEXT:    mul t1, a0, a5
+; RV64IMZBS-NEXT:    mul a1, a7, a1
+; RV64IMZBS-NEXT:    mul a0, a0, t3
+; RV64IMZBS-NEXT:    xor a7, t5, s0
+; RV64IMZBS-NEXT:    mul t3, t4, t6
+; RV64IMZBS-NEXT:    xor a7, a7, s2
+; RV64IMZBS-NEXT:    mul a5, s1, a5
+; RV64IMZBS-NEXT:    xor a7, a7, t2
+; RV64IMZBS-NEXT:    slli t2, a3, 32
+; RV64IMZBS-NEXT:    add a3, a3, t2
+; RV64IMZBS-NEXT:    xor t2, s4, s3
+; RV64IMZBS-NEXT:    and a3, a7, a3
+; RV64IMZBS-NEXT:    xor a7, t2, t1
+; RV64IMZBS-NEXT:    xor a7, a7, t0
+; RV64IMZBS-NEXT:    xor a0, a1, a0
+; RV64IMZBS-NEXT:    slli a1, a4, 32
+; RV64IMZBS-NEXT:    xor a0, t3, a0
+; RV64IMZBS-NEXT:    add a1, a4, a1
+; RV64IMZBS-NEXT:    xor a0, a0, a5
+; RV64IMZBS-NEXT:    and a1, a7, a1
+; RV64IMZBS-NEXT:    and a0, a0, a2
+; RV64IMZBS-NEXT:    or a2, a3, a6
+; RV64IMZBS-NEXT:    or a0, a1, a0
+; RV64IMZBS-NEXT:    or a0, a2, a0
+; RV64IMZBS-NEXT:    ld s0, 40(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s1, 32(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s2, 24(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s3, 16(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s4, 8(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    .cfi_restore s0
+; RV64IMZBS-NEXT:    .cfi_restore s1
+; RV64IMZBS-NEXT:    .cfi_restore s2
+; RV64IMZBS-NEXT:    .cfi_restore s3
+; RV64IMZBS-NEXT:    .cfi_restore s4
+; RV64IMZBS-NEXT:    addi sp, sp, 48
+; RV64IMZBS-NEXT:    .cfi_def_cfa_offset 0
+; RV64IMZBS-NEXT:    ret
+;
+; RV32IMZBC-LABEL: clmul_i64_zext:
+; RV32IMZBC:       # %bb.0:
+; RV32IMZBC-NEXT:    clmul a2, a0, a1
+; RV32IMZBC-NEXT:    clmulh a1, a0, a1
+; RV32IMZBC-NEXT:    mv a0, a2
+; RV32IMZBC-NEXT:    ret
+;
+; RV64IMZBC-LABEL: clmul_i64_zext:
+; RV64IMZBC:       # %bb.0:
+; RV64IMZBC-NEXT:    slli a1, a1, 32
+; RV64IMZBC-NEXT:    slli a0, a0, 32
+; RV64IMZBC-NEXT:    clmulh a0, a0, a1
+; RV64IMZBC-NEXT:    ret
+  %zextx = zext i32 %x to i64
+  %zexty = zext i32 %y to i64
+  %a = call i64 @llvm.clmul.i64(i64 %zextx, i64 %zexty)
+  ret i64 %a
+}
+
+define i96 @clmul_i96(i96 %x, i96 %y) {
+; RV32I-LABEL: clmul_i96:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi sp, sp, -496
+; RV32I-NEXT:    .cfi_def_cfa_offset 496
+; RV32I-NEXT:    sw ra, 492(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s0, 488(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s1, 484(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s2, 480(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s3, 476(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s4, 472(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s5, 468(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s6, 464(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s7, 460(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s8, 456(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s9, 452(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s10, 448(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s11, 444(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    .cfi_offset ra, -4
+; RV32I-NEXT:    .cfi_offset s0, -8
+; RV32I-NEXT:    .cfi_offset s1, -12
+; RV32I-NEXT:    .cfi_offset s2, -16
+; RV32I-NEXT:    .cfi_offset s3, -20
+; RV32I-NEXT:    .cfi_offset s4, -24
+; RV32I-NEXT:    .cfi_offset s5, -28
+; RV32I-NEXT:    .cfi_offset s6, -32
+; RV32I-NEXT:    .cfi_offset s7, -36
+; RV32I-NEXT:    .cfi_offset s8, -40
+; RV32I-NEXT:    .cfi_offset s9, -44
+; RV32I-NEXT:    .cfi_offset s10, -48
+; RV32I-NEXT:    .cfi_offset s11, -52
+; RV32I-NEXT:    sw a0, 300(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw s10, 4(a1)
+; RV32I-NEXT:    lw a0, 8(a1)
+; RV32I-NEXT:    lw a4, 0(a2)
+; RV32I-NEXT:    lw s2, 4(a2)
+; RV32I-NEXT:    lw s11, 8(a2)
+; RV32I-NEXT:    srli a2, s10, 31
+; RV32I-NEXT:    slli a5, a0, 1
+; RV32I-NEXT:    slli a6, a4, 30
+; RV32I-NEXT:    slli a7, a4, 31
+; RV32I-NEXT:    or a2, a5, a2
+; RV32I-NEXT:    srai a5, a6, 31
+; RV32I-NEXT:    sw a5, 368(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai a6, a7, 31
+; RV32I-NEXT:    sw a6, 364(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a5, a2
+; RV32I-NEXT:    and a5, a6, a0
+; RV32I-NEXT:    xor a2, a5, a2
+; RV32I-NEXT:    srli a5, s10, 30
+; RV32I-NEXT:    slli a6, a0, 2
+; RV32I-NEXT:    or a5, a6, a5
+; RV32I-NEXT:    slli a6, a4, 29
+; RV32I-NEXT:    srai t1, a6, 31
+; RV32I-NEXT:    sw t1, 360(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a6, s10, 29
+; RV32I-NEXT:    slli a7, a0, 3
+; RV32I-NEXT:    slli t0, a4, 28
+; RV32I-NEXT:    or a6, a7, a6
+; RV32I-NEXT:    srai a7, t0, 31
+; RV32I-NEXT:    sw a7, 356(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, t1, a5
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    srli a6, s10, 28
+; RV32I-NEXT:    slli a7, a0, 4
+; RV32I-NEXT:    slli t0, a4, 27
+; RV32I-NEXT:    or a6, a7, a6
+; RV32I-NEXT:    srai a7, t0, 31
+; RV32I-NEXT:    sw a7, 352(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    srli a7, s10, 27
+; RV32I-NEXT:    slli t0, a0, 5
+; RV32I-NEXT:    slli t1, a4, 26
+; RV32I-NEXT:    or a7, t0, a7
+; RV32I-NEXT:    srai t0, t1, 31
+; RV32I-NEXT:    sw t0, 348(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a7, t0, a7
+; RV32I-NEXT:    srli t0, s10, 26
+; RV32I-NEXT:    slli t1, a0, 6
+; RV32I-NEXT:    slli t2, a4, 25
+; RV32I-NEXT:    or t0, t1, t0
+; RV32I-NEXT:    srai t1, t2, 31
+; RV32I-NEXT:    sw t1, 344(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, t1, t0
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    xor a5, a6, a7
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    srli a5, s10, 25
+; RV32I-NEXT:    slli a6, a0, 7
+; RV32I-NEXT:    or a5, a6, a5
+; RV32I-NEXT:    slli a6, a4, 24
+; RV32I-NEXT:    srai t1, a6, 31
+; RV32I-NEXT:    sw t1, 340(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a6, s10, 24
+; RV32I-NEXT:    slli a7, a0, 8
+; RV32I-NEXT:    slli t0, a4, 23
+; RV32I-NEXT:    or a6, a7, a6
+; RV32I-NEXT:    srai a7, t0, 31
+; RV32I-NEXT:    sw a7, 336(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, t1, a5
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    srli a6, s10, 23
+; RV32I-NEXT:    slli a7, a0, 9
+; RV32I-NEXT:    slli t0, a4, 22
+; RV32I-NEXT:    or a6, a7, a6
+; RV32I-NEXT:    srai a7, t0, 31
+; RV32I-NEXT:    sw a7, 332(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    srli a7, s10, 22
+; RV32I-NEXT:    slli t0, a0, 10
+; RV32I-NEXT:    slli t1, a4, 21
+; RV32I-NEXT:    or a7, t0, a7
+; RV32I-NEXT:    srai t0, t1, 31
+; RV32I-NEXT:    sw t0, 328(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    and a6, t0, a7
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    srli a6, s10, 21
+; RV32I-NEXT:    slli a7, a0, 11
+; RV32I-NEXT:    slli t0, a4, 20
+; RV32I-NEXT:    or a6, a7, a6
+; RV32I-NEXT:    srai a7, t0, 31
+; RV32I-NEXT:    sw a7, 324(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    srli a7, s10, 20
+; RV32I-NEXT:    slli t0, a0, 12
+; RV32I-NEXT:    slli t1, a4, 19
+; RV32I-NEXT:    or a7, t0, a7
+; RV32I-NEXT:    srai t0, t1, 31
+; RV32I-NEXT:    sw t0, 320(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a7, t0, a7
+; RV32I-NEXT:    srli t0, s10, 19
+; RV32I-NEXT:    slli t1, a0, 13
+; RV32I-NEXT:    slli t2, a4, 18
+; RV32I-NEXT:    or t0, t1, t0
+; RV32I-NEXT:    srai t1, t2, 31
+; RV32I-NEXT:    sw t1, 316(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, t1, t0
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    srli a7, s10, 18
+; RV32I-NEXT:    slli t0, a0, 14
+; RV32I-NEXT:    slli t1, a4, 17
+; RV32I-NEXT:    or a7, t0, a7
+; RV32I-NEXT:    srai t2, t1, 31
+; RV32I-NEXT:    sw t2, 400(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli t0, s10, 17
+; RV32I-NEXT:    slli t1, a0, 15
+; RV32I-NEXT:    or t0, t1, t0
+; RV32I-NEXT:    slli t1, a4, 16
+; RV32I-NEXT:    and a7, t2, a7
+; RV32I-NEXT:    srai t1, t1, 31
+; RV32I-NEXT:    sw t1, 396(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, t1, t0
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    xor a5, a6, a7
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    srli a5, s10, 16
+; RV32I-NEXT:    slli a6, a0, 16
+; RV32I-NEXT:    or a5, a6, a5
+; RV32I-NEXT:    slli a6, a4, 15
+; RV32I-NEXT:    srli a7, s10, 15
+; RV32I-NEXT:    slli t0, a0, 17
+; RV32I-NEXT:    or a7, t0, a7
+; RV32I-NEXT:    slli t0, a4, 14
+; RV32I-NEXT:    srai a6, a6, 31
+; RV32I-NEXT:    sw a6, 392(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai t0, t0, 31
+; RV32I-NEXT:    sw t0, 388(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    and a6, t0, a7
+; RV32I-NEXT:    srli a7, s10, 14
+; RV32I-NEXT:    slli t0, a0, 18
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    or a6, t0, a7
+; RV32I-NEXT:    srli a7, s10, 13
+; RV32I-NEXT:    slli t0, a0, 19
+; RV32I-NEXT:    or a7, t0, a7
+; RV32I-NEXT:    slli t0, a4, 13
+; RV32I-NEXT:    srai t1, t0, 31
+; RV32I-NEXT:    sw t1, 384(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t0, a4, 12
+; RV32I-NEXT:    and a6, t1, a6
+; RV32I-NEXT:    srai t0, t0, 31
+; RV32I-NEXT:    sw t0, 380(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    and a6, t0, a7
+; RV32I-NEXT:    srli a7, s10, 12
+; RV32I-NEXT:    slli t0, a0, 20
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    or a6, t0, a7
+; RV32I-NEXT:    srli a7, s10, 11
+; RV32I-NEXT:    slli t0, a0, 21
+; RV32I-NEXT:    or a7, t0, a7
+; RV32I-NEXT:    slli t0, a4, 11
+; RV32I-NEXT:    srai t1, t0, 31
+; RV32I-NEXT:    sw t1, 376(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t0, a4, 10
+; RV32I-NEXT:    and a6, t1, a6
+; RV32I-NEXT:    srai t0, t0, 31
+; RV32I-NEXT:    sw t0, 372(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    and a6, t0, a7
+; RV32I-NEXT:    srli a7, s10, 10
+; RV32I-NEXT:    slli t0, a0, 22
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    or a6, t0, a7
+; RV32I-NEXT:    srli a7, s10, 9
+; RV32I-NEXT:    slli t0, a0, 23
+; RV32I-NEXT:    or a7, t0, a7
+; RV32I-NEXT:    slli t0, a4, 9
+; RV32I-NEXT:    srli t1, s10, 8
+; RV32I-NEXT:    slli t2, a0, 24
+; RV32I-NEXT:    or t1, t2, t1
+; RV32I-NEXT:    srai t0, t0, 31
+; RV32I-NEXT:    sw t0, 440(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, t0, a6
+; RV32I-NEXT:    slli t0, a4, 8
+; RV32I-NEXT:    srai t2, t0, 31
+; RV32I-NEXT:    sw t2, 436(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t0, a4, 7
+; RV32I-NEXT:    and a7, t2, a7
+; RV32I-NEXT:    srai t0, t0, 31
+; RV32I-NEXT:    sw t0, 432(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, t0, t1
+; RV32I-NEXT:    srli t0, s10, 7
+; RV32I-NEXT:    slli t1, a0, 25
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    or a7, t1, t0
+; RV32I-NEXT:    srli t0, s10, 6
+; RV32I-NEXT:    slli t1, a0, 26
+; RV32I-NEXT:    or t0, t1, t0
+; RV32I-NEXT:    slli t1, a4, 6
+; RV32I-NEXT:    srai t2, t1, 31
+; RV32I-NEXT:    sw t2, 428(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t1, a4, 5
+; RV32I-NEXT:    and a7, t2, a7
+; RV32I-NEXT:    srai t1, t1, 31
+; RV32I-NEXT:    sw t1, 424(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, t1, t0
+; RV32I-NEXT:    srli t0, s10, 5
+; RV32I-NEXT:    slli t1, a0, 27
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    or a7, t1, t0
+; RV32I-NEXT:    srli t0, s10, 4
+; RV32I-NEXT:    slli t1, a0, 28
+; RV32I-NEXT:    or t0, t1, t0
+; RV32I-NEXT:    slli t1, a4, 4
+; RV32I-NEXT:    srai t2, t1, 31
+; RV32I-NEXT:    sw t2, 420(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t1, a4, 3
+; RV32I-NEXT:    and a7, t2, a7
+; RV32I-NEXT:    srai t1, t1, 31
+; RV32I-NEXT:    sw t1, 416(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, t1, t0
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    xor a5, a6, a7
+; RV32I-NEXT:    srli a6, s10, 3
+; RV32I-NEXT:    slli a7, a0, 29
+; RV32I-NEXT:    srli t0, s10, 2
+; RV32I-NEXT:    slli t1, a0, 30
+; RV32I-NEXT:    or a6, a7, a6
+; RV32I-NEXT:    or a7, t1, t0
+; RV32I-NEXT:    slli t0, a4, 2
+; RV32I-NEXT:    slli t1, a4, 1
+; RV32I-NEXT:    srai t0, t0, 31
+; RV32I-NEXT:    sw t0, 412(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai t1, t1, 31
+; RV32I-NEXT:    sw t1, 408(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, t0, a6
+; RV32I-NEXT:    and a7, t1, a7
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    xor a5, a6, a7
+; RV32I-NEXT:    slli a0, a0, 31
+; RV32I-NEXT:    srli a6, s10, 1
+; RV32I-NEXT:    or a6, a0, a6
+; RV32I-NEXT:    lw a0, 0(a1)
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    sw a4, 404(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, s2, 31
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    srai a1, a1, 31
+; RV32I-NEXT:    sw a1, 296(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a5, a4
+; RV32I-NEXT:    and a1, a1, s10
+; RV32I-NEXT:    slli a5, s10, 1
+; RV32I-NEXT:    srli a6, a0, 31
+; RV32I-NEXT:    xor a1, a4, a1
+; RV32I-NEXT:    or t1, a6, a5
+; RV32I-NEXT:    slli a4, s10, 2
+; RV32I-NEXT:    srli a5, a0, 30
+; RV32I-NEXT:    or a7, a5, a4
+; RV32I-NEXT:    sw a7, 152(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 30
+; RV32I-NEXT:    srai a5, a4, 31
+; RV32I-NEXT:    sw a5, 292(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 29
+; RV32I-NEXT:    srai a6, a4, 31
+; RV32I-NEXT:    sw a6, 288(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a5, t1
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    and a4, a6, a7
+; RV32I-NEXT:    slli a5, s10, 3
+; RV32I-NEXT:    srli a6, a0, 29
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    or t0, a6, a5
+; RV32I-NEXT:    sw t0, 168(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a4, a0, 28
+; RV32I-NEXT:    slli a5, s10, 4
+; RV32I-NEXT:    or a7, a5, a4
+; RV32I-NEXT:    sw a7, 172(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 28
+; RV32I-NEXT:    srai a5, a4, 31
+; RV32I-NEXT:    sw a5, 284(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 27
+; RV32I-NEXT:    srai a6, a4, 31
+; RV32I-NEXT:    sw a6, 280(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a5, t0
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    and a4, a6, a7
+; RV32I-NEXT:    slli a5, s10, 5
+; RV32I-NEXT:    srli a6, a0, 27
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    or s3, a6, a5
+; RV32I-NEXT:    slli a4, s10, 6
+; RV32I-NEXT:    srli a5, a0, 26
+; RV32I-NEXT:    or t2, a5, a4
+; RV32I-NEXT:    sw t2, 148(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 26
+; RV32I-NEXT:    slli a5, s10, 7
+; RV32I-NEXT:    srli a6, a0, 25
+; RV32I-NEXT:    or s4, a6, a5
+; RV32I-NEXT:    srai a6, a4, 31
+; RV32I-NEXT:    sw a6, 276(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 25
+; RV32I-NEXT:    slli a5, s2, 24
+; RV32I-NEXT:    srai t0, a4, 31
+; RV32I-NEXT:    sw t0, 268(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai a7, a5, 31
+; RV32I-NEXT:    sw a7, 272(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, s3
+; RV32I-NEXT:    and a5, t0, t2
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a7, s4
+; RV32I-NEXT:    srli a6, a0, 24
+; RV32I-NEXT:    slli a7, s10, 8
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    or a3, a7, a6
+; RV32I-NEXT:    slli a5, s10, 9
+; RV32I-NEXT:    srli a6, a0, 23
+; RV32I-NEXT:    or t0, a6, a5
+; RV32I-NEXT:    sw t0, 164(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s2, 23
+; RV32I-NEXT:    srai a6, a5, 31
+; RV32I-NEXT:    sw a6, 264(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s2, 22
+; RV32I-NEXT:    srai a7, a5, 31
+; RV32I-NEXT:    sw a7, 260(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a6, a3
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a7, t0
+; RV32I-NEXT:    slli a6, s10, 10
+; RV32I-NEXT:    srli a7, a0, 22
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    or t0, a7, a6
+; RV32I-NEXT:    sw t0, 144(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s10, 11
+; RV32I-NEXT:    srli a6, a0, 21
+; RV32I-NEXT:    or t2, a6, a5
+; RV32I-NEXT:    sw t2, 132(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s2, 21
+; RV32I-NEXT:    srai a6, a5, 31
+; RV32I-NEXT:    sw a6, 256(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s2, 20
+; RV32I-NEXT:    srai a7, a5, 31
+; RV32I-NEXT:    sw a7, 252(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a6, t0
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a7, t2
+; RV32I-NEXT:    slli a6, s10, 12
+; RV32I-NEXT:    srli a7, a0, 20
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    or t0, a7, a6
+; RV32I-NEXT:    sw t0, 120(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s10, 13
+; RV32I-NEXT:    srli a6, a0, 19
+; RV32I-NEXT:    or t2, a6, a5
+; RV32I-NEXT:    sw t2, 112(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s2, 19
+; RV32I-NEXT:    srai a6, a5, 31
+; RV32I-NEXT:    sw a6, 248(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s2, 18
+; RV32I-NEXT:    srai a7, a5, 31
+; RV32I-NEXT:    sw a7, 244(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a6, t0
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a7, t2
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    slli a2, s10, 14
+; RV32I-NEXT:    srli a4, a0, 18
+; RV32I-NEXT:    slli a5, s10, 15
+; RV32I-NEXT:    srli a6, a0, 17
+; RV32I-NEXT:    or a7, a4, a2
+; RV32I-NEXT:    sw a7, 104(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    or a5, a6, a5
+; RV32I-NEXT:    sw a5, 140(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, s2, 17
+; RV32I-NEXT:    slli a4, s2, 16
+; RV32I-NEXT:    srai a2, a2, 31
+; RV32I-NEXT:    sw a2, 240(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    sw a4, 236(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a7
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    srli a5, a0, 16
+; RV32I-NEXT:    slli a6, s10, 16
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    or a7, a6, a5
+; RV32I-NEXT:    sw a7, 128(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a4, a0, 15
+; RV32I-NEXT:    slli a5, s10, 17
+; RV32I-NEXT:    or t0, a5, a4
+; RV32I-NEXT:    sw t0, 116(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 15
+; RV32I-NEXT:    srai a5, a4, 31
+; RV32I-NEXT:    sw a5, 232(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 14
+; RV32I-NEXT:    srai a6, a4, 31
+; RV32I-NEXT:    sw a6, 228(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a5, a7
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a6, t0
+; RV32I-NEXT:    srli a5, a0, 14
+; RV32I-NEXT:    slli a6, s10, 18
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    or t0, a6, a5
+; RV32I-NEXT:    sw t0, 108(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a4, a0, 13
+; RV32I-NEXT:    slli a5, s10, 19
+; RV32I-NEXT:    or a7, a5, a4
+; RV32I-NEXT:    sw a7, 160(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 13
+; RV32I-NEXT:    srai a5, a4, 31
+; RV32I-NEXT:    sw a5, 224(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 12
+; RV32I-NEXT:    srai a6, a4, 31
+; RV32I-NEXT:    sw a6, 220(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a5, t0
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a6, a7
+; RV32I-NEXT:    srli a5, a0, 12
+; RV32I-NEXT:    slli a6, s10, 20
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    or t0, a6, a5
+; RV32I-NEXT:    sw t0, 100(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a4, a0, 11
+; RV32I-NEXT:    slli a5, s10, 21
+; RV32I-NEXT:    or a7, a5, a4
+; RV32I-NEXT:    sw a7, 156(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 11
+; RV32I-NEXT:    srai a5, a4, 31
+; RV32I-NEXT:    sw a5, 216(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 10
+; RV32I-NEXT:    srai a6, a4, 31
+; RV32I-NEXT:    sw a6, 212(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a5, t0
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a6, a7
+; RV32I-NEXT:    srli a5, a0, 10
+; RV32I-NEXT:    slli a6, s10, 22
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    or a7, a6, a5
+; RV32I-NEXT:    sw a7, 136(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a4, a0, 9
+; RV32I-NEXT:    slli a5, s10, 23
+; RV32I-NEXT:    or t0, a5, a4
+; RV32I-NEXT:    sw t0, 124(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 9
+; RV32I-NEXT:    srai a5, a4, 31
+; RV32I-NEXT:    sw a5, 200(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 8
+; RV32I-NEXT:    srai a6, a4, 31
+; RV32I-NEXT:    sw a6, 196(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a5, a7
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a6, t0
+; RV32I-NEXT:    slli a5, s10, 24
+; RV32I-NEXT:    srli a6, a0, 8
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    or t2, a6, a5
+; RV32I-NEXT:    sw t2, 96(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a4, a0, 7
+; RV32I-NEXT:    slli a5, s10, 25
+; RV32I-NEXT:    or t3, a5, a4
+; RV32I-NEXT:    slli a4, s2, 7
+; RV32I-NEXT:    srli a5, a0, 6
+; RV32I-NEXT:    slli a6, s10, 26
+; RV32I-NEXT:    or t5, a6, a5
+; RV32I-NEXT:    srai a7, a4, 31
+; RV32I-NEXT:    sw a7, 188(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s2, 6
+; RV32I-NEXT:    slli a5, s2, 5
+; RV32I-NEXT:    srai t0, a4, 31
+; RV32I-NEXT:    sw t0, 184(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srai a6, a5, 31
+; RV32I-NEXT:    sw a6, 204(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a7, t2
+; RV32I-NEXT:    and a5, t0, t3
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a6, t5
+; RV32I-NEXT:    srli a6, a0, 5
+; RV32I-NEXT:    slli a7, s10, 27
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    or t4, a7, a6
+; RV32I-NEXT:    slli a5, s10, 28
+; RV32I-NEXT:    srli a6, a0, 4
+; RV32I-NEXT:    or t2, a6, a5
+; RV32I-NEXT:    slli a5, s2, 4
+; RV32I-NEXT:    srai a7, a5, 31
+; RV32I-NEXT:    sw a7, 176(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s2, 3
+; RV32I-NEXT:    srai a6, a5, 31
+; RV32I-NEXT:    sw a6, 192(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a7, t4
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a6, t2
+; RV32I-NEXT:    srli a6, a0, 3
+; RV32I-NEXT:    slli a7, s10, 29
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    or t6, a7, a6
+; RV32I-NEXT:    srli a5, a0, 2
+; RV32I-NEXT:    slli a6, s10, 30
+; RV32I-NEXT:    or s1, a6, a5
+; RV32I-NEXT:    slli a5, s2, 2
+; RV32I-NEXT:    srai a7, a5, 31
+; RV32I-NEXT:    sw a7, 180(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s2, 1
+; RV32I-NEXT:    srai a6, a5, 31
+; RV32I-NEXT:    sw a6, 208(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a7, t6
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a6, s1
+; RV32I-NEXT:    srli a6, a0, 1
+; RV32I-NEXT:    slli a7, s10, 31
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    or t0, a7, a6
+; RV32I-NEXT:    srai a5, s2, 31
+; RV32I-NEXT:    slli a6, s11, 31
+; RV32I-NEXT:    and a5, a5, t0
+; RV32I-NEXT:    srai a6, a6, 31
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a6, a0
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    slli a5, s11, 30
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli s5, a0, 1
+; RV32I-NEXT:    and a5, a5, s5
+; RV32I-NEXT:    slli a6, s11, 29
+; RV32I-NEXT:    srai a6, a6, 31
+; RV32I-NEXT:    slli s6, a0, 2
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a6, s6
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    slli a2, s11, 28
+; RV32I-NEXT:    slli a4, s11, 27
+; RV32I-NEXT:    srai a2, a2, 31
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    slli a7, a0, 3
+; RV32I-NEXT:    slli s7, a0, 4
+; RV32I-NEXT:    and a2, a2, a7
+; RV32I-NEXT:    and a4, a4, s7
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    slli a4, s11, 26
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    slli s8, a0, 5
+; RV32I-NEXT:    and a4, a4, s8
+; RV32I-NEXT:    slli a5, s11, 25
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli s9, a0, 6
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a5, s9
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    slli a4, s11, 24
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    slli ra, a0, 7
+; RV32I-NEXT:    and a4, a4, ra
+; RV32I-NEXT:    slli a5, s11, 23
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli a6, a0, 8
+; RV32I-NEXT:    sw a6, 92(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a5, a6
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    slli a4, s11, 22
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    slli a5, a0, 9
+; RV32I-NEXT:    sw a5, 88(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    slli a5, s11, 21
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli a6, a0, 10
+; RV32I-NEXT:    sw a6, 84(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a5, a6
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    slli a4, s11, 20
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    slli a5, a0, 11
+; RV32I-NEXT:    sw a5, 80(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    slli a5, s11, 19
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli a6, a0, 12
+; RV32I-NEXT:    sw a6, 76(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a5, a6
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    slli a4, s11, 18
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    slli a5, a0, 13
+; RV32I-NEXT:    sw a5, 72(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    slli a5, s11, 17
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli a6, a0, 14
+; RV32I-NEXT:    sw a6, 64(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    and a4, a5, a6
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    slli a4, s11, 16
+; RV32I-NEXT:    srai a4, a4, 31
+; RV32I-NEXT:    slli a5, a0, 15
+; RV32I-NEXT:    sw a5, 60(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    slli a5, s11, 15
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli a6, a0, 16
+; RV32I-NEXT:    sw a6, 304(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    slli a6, s11, 14
+; RV32I-NEXT:    srai a6, a6, 31
+; RV32I-NEXT:    slli s0, a0, 17
+; RV32I-NEXT:    sw s0, 56(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a6, s0
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    slli a5, s11, 13
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli a6, a0, 18
+; RV32I-NEXT:    sw a6, 48(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    slli a6, s11, 12
+; RV32I-NEXT:    srai a6, a6, 31
+; RV32I-NEXT:    slli s0, a0, 19
+; RV32I-NEXT:    sw s0, 40(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a6, s0
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    slli a5, s11, 11
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli a6, a0, 20
+; RV32I-NEXT:    sw a6, 32(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    slli a6, s11, 10
+; RV32I-NEXT:    srai a6, a6, 31
+; RV32I-NEXT:    slli s0, a0, 21
+; RV32I-NEXT:    sw s0, 16(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a6, s0
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    slli a5, s11, 9
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli a6, a0, 22
+; RV32I-NEXT:    sw a6, 308(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    slli a6, s11, 8
+; RV32I-NEXT:    srai a6, a6, 31
+; RV32I-NEXT:    slli s0, a0, 23
+; RV32I-NEXT:    sw s0, 312(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a6, s0
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    slli a5, s11, 7
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli a6, a0, 24
+; RV32I-NEXT:    sw a6, 36(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    slli a6, s11, 6
+; RV32I-NEXT:    srai a6, a6, 31
+; RV32I-NEXT:    slli s0, a0, 25
+; RV32I-NEXT:    sw s0, 24(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a6, s0
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    slli a5, s11, 5
+; RV32I-NEXT:    srai a5, a5, 31
+; RV32I-NEXT:    slli a6, a0, 26
+; RV32I-NEXT:    sw a6, 20(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    slli a6, s11, 4
+; RV32I-NEXT:    srai s0, a6, 31
+; RV32I-NEXT:    slli a6, a0, 27
+; RV32I-NEXT:    sw a6, 28(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, s0, a6
+; RV32I-NEXT:    xor s0, a1, a2
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    slli a1, s11, 3
+; RV32I-NEXT:    slli a2, s11, 2
+; RV32I-NEXT:    srai a1, a1, 31
+; RV32I-NEXT:    srai a2, a2, 31
+; RV32I-NEXT:    slli a6, a0, 28
+; RV32I-NEXT:    sw a6, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a0, 29
+; RV32I-NEXT:    sw a5, 44(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a1, a6
+; RV32I-NEXT:    and a2, a2, a5
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    slli a2, s11, 1
+; RV32I-NEXT:    srai a5, a2, 31
+; RV32I-NEXT:    slli a2, a0, 30
+; RV32I-NEXT:    sw a2, 52(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a5, a2
+; RV32I-NEXT:    slli a5, a0, 31
+; RV32I-NEXT:    sw a5, 68(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    and a2, s11, a5
+; RV32I-NEXT:    xor a4, s0, a4
+; RV32I-NEXT:    sw a4, 8(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    sw a1, 4(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a1, 368(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, t1
+; RV32I-NEXT:    lw a2, 364(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, s10
+; RV32I-NEXT:    lw a4, 152(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 360(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    lw t1, 168(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 356(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a5, t1
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    xor a2, a4, s10
+; RV32I-NEXT:    lw a4, 172(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 352(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    lw a5, 348(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a5, s3
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw a5, 148(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 344(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a6, a5
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a4, s10
+; RV32I-NEXT:    lw a4, 340(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s4
+; RV32I-NEXT:    lw a5, 336(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a5, a3
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s0, 164(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 332(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a5, s0
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw a5, 144(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 328(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a6, a5
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a4, s10
+; RV32I-NEXT:    lw a4, 132(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 324(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    lw a5, 120(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 320(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a6, a5
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw a5, 112(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 316(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a6, a5
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw a5, 104(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 400(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a6, a5
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw a5, 140(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 396(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a6, a5
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a4, s10
+; RV32I-NEXT:    lw a4, 392(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 128(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    lw a5, 388(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 116(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a5, a6
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw a5, 384(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 108(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a5, a6
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 160(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 380(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a5, s10
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw a5, 376(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 100(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a5, a6
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 156(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 372(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, a5, s10
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a4, s10
+; RV32I-NEXT:    lw a4, 440(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 136(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    lw s10, 436(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 124(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, a5
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 432(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 96(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, a5
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 428(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, t3
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 424(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, t5
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 420(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, t4
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 416(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, t2
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a4, s10
+; RV32I-NEXT:    lw a4, 412(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t6
+; RV32I-NEXT:    lw s10, 408(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, s1
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 404(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, t0
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 296(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, a0
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 292(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    mv a5, s5
+; RV32I-NEXT:    and s10, s10, s5
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 288(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    mv t0, s6
+; RV32I-NEXT:    and s10, s10, s6
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 284(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, a7
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 280(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    mv t2, s7
+; RV32I-NEXT:    and s10, s10, s7
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a4, s10
+; RV32I-NEXT:    lw a4, 276(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    mv a6, s8
+; RV32I-NEXT:    and a4, a4, s8
+; RV32I-NEXT:    lw s10, 268(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    mv t1, s9
+; RV32I-NEXT:    and s10, s10, s9
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 272(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    mv s0, ra
+; RV32I-NEXT:    and s10, s10, ra
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 264(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 92(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, t3
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 260(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t4, 88(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, t4
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 256(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t5, 84(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, t5
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 252(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s11, 80(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, s11
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 248(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s4, 76(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, s4
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 244(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 72(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, s3
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a4, s10
+; RV32I-NEXT:    lw a4, 240(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t6, 64(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t6
+; RV32I-NEXT:    lw s10, 236(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s1, 60(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, s1
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 232(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a3, 304(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, a3
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 228(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s5, 56(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, s5
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 224(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 48(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, s6
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 220(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 40(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, s7
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 216(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 32(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, s8
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 212(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s9, 16(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, s9
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 200(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw ra, 308(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, ra
+; RV32I-NEXT:    xor a4, a4, s10
+; RV32I-NEXT:    lw s10, 196(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw ra, 312(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, s10, ra
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a4, s10
+; RV32I-NEXT:    lw a3, 8(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a4, 4(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    sw a3, 296(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor s10, a1, a2
+; RV32I-NEXT:    lw a1, 368(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    lw a2, 364(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a2, a0
+; RV32I-NEXT:    lw a2, 188(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw ra, 36(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, ra
+; RV32I-NEXT:    lw a4, 184(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a3, 24(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a3
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 360(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, t0
+; RV32I-NEXT:    lw a4, 356(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a7
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 204(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 20(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a7
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 352(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, t2
+; RV32I-NEXT:    lw a4, 348(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 344(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t1
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 176(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 28(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 340(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, s0
+; RV32I-NEXT:    lw a4, 336(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t3
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 332(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t4
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 328(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t5
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 192(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t1, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t1
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 324(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, s11
+; RV32I-NEXT:    lw a4, 320(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s4
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 316(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s3
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 400(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 396(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s1
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 180(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t0, 44(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t0
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 392(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a4, 304(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a4
+; RV32I-NEXT:    lw a4, 388(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s5
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 384(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 380(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s7
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 376(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s8
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 372(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s9
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 208(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 52(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t2
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 440(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a4, 308(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a4
+; RV32I-NEXT:    lw a4, 436(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 312(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 432(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, ra
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 428(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a3
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 424(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a7
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 420(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a4, 416(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t1
+; RV32I-NEXT:    lw a5, 412(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a5, t0
+; RV32I-NEXT:    lw a5, 408(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a5, t2
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    xor a4, a7, a6
+; RV32I-NEXT:    lw a6, 68(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, s2, a6
+; RV32I-NEXT:    lw a5, 404(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    xor a0, a0, a4
+; RV32I-NEXT:    xor a1, s10, a2
+; RV32I-NEXT:    lw a2, 300(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    sw a0, 0(a2)
+; RV32I-NEXT:    sw a1, 4(a2)
+; RV32I-NEXT:    lw a0, 296(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    sw a0, 8(a2)
+; RV32I-NEXT:    lw ra, 492(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s0, 488(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s1, 484(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s2, 480(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 476(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s4, 472(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s5, 468(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 464(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 460(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 456(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s9, 452(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s10, 448(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s11, 444(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    .cfi_restore ra
+; RV32I-NEXT:    .cfi_restore s0
+; RV32I-NEXT:    .cfi_restore s1
+; RV32I-NEXT:    .cfi_restore s2
+; RV32I-NEXT:    .cfi_restore s3
+; RV32I-NEXT:    .cfi_restore s4
+; RV32I-NEXT:    .cfi_restore s5
+; RV32I-NEXT:    .cfi_restore s6
+; RV32I-NEXT:    .cfi_restore s7
+; RV32I-NEXT:    .cfi_restore s8
+; RV32I-NEXT:    .cfi_restore s9
+; RV32I-NEXT:    .cfi_restore s10
+; RV32I-NEXT:    .cfi_restore s11
+; RV32I-NEXT:    addi sp, sp, 496
+; RV32I-NEXT:    .cfi_def_cfa_offset 0
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: clmul_i96:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi sp, sp, -688
+; RV64I-NEXT:    .cfi_def_cfa_offset 688
+; RV64I-NEXT:    sd ra, 680(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s0, 672(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s1, 664(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s2, 656(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s3, 648(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s4, 640(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s5, 632(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s6, 624(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s7, 616(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s8, 608(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s9, 600(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s10, 592(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s11, 584(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    .cfi_offset ra, -8
+; RV64I-NEXT:    .cfi_offset s0, -16
+; RV64I-NEXT:    .cfi_offset s1, -24
+; RV64I-NEXT:    .cfi_offset s2, -32
+; RV64I-NEXT:    .cfi_offset s3, -40
+; RV64I-NEXT:    .cfi_offset s4, -48
+; RV64I-NEXT:    .cfi_offset s5, -56
+; RV64I-NEXT:    .cfi_offset s6, -64
+; RV64I-NEXT:    .cfi_offset s7, -72
+; RV64I-NEXT:    .cfi_offset s8, -80
+; RV64I-NEXT:    .cfi_offset s9, -88
+; RV64I-NEXT:    .cfi_offset s10, -96
+; RV64I-NEXT:    .cfi_offset s11, -104
+; RV64I-NEXT:    mv t0, a3
+; RV64I-NEXT:    slli a3, a0, 1
+; RV64I-NEXT:    sd a3, 568(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a4, a2, 62
+; RV64I-NEXT:    slli a5, a2, 63
+; RV64I-NEXT:    srai t3, a4, 63
+; RV64I-NEXT:    srai s2, a5, 63
+; RV64I-NEXT:    and a4, t3, a3
+; RV64I-NEXT:    and a5, s2, a0
+; RV64I-NEXT:    xor a4, a5, a4
+; RV64I-NEXT:    slli a6, a0, 2
+; RV64I-NEXT:    sd a6, 536(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a2, 61
+; RV64I-NEXT:    srai a7, a5, 63
+; RV64I-NEXT:    sd a7, 152(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a2, 60
+; RV64I-NEXT:    slli a3, a0, 3
+; RV64I-NEXT:    sd a3, 576(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    srai t1, a5, 63
+; RV64I-NEXT:    sd t1, 136(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a7, a6
+; RV64I-NEXT:    and a6, t1, a3
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    slli a6, a2, 59
+; RV64I-NEXT:    slli a3, a0, 4
+; RV64I-NEXT:    sd a3, 552(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    sd a6, 248(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a6, a6, a3
+; RV64I-NEXT:    slli a7, a2, 58
+; RV64I-NEXT:    slli a3, a0, 5
+; RV64I-NEXT:    sd a3, 560(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    srai a7, a7, 63
+; RV64I-NEXT:    sd a7, 304(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a7, a7, a3
+; RV64I-NEXT:    slli t1, a2, 57
+; RV64I-NEXT:    slli a3, a0, 6
+; RV64I-NEXT:    sd a3, 544(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    srai t1, t1, 63
+; RV64I-NEXT:    sd t1, 128(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a6, a6, a7
+; RV64I-NEXT:    and a7, t1, a3
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    xor a5, a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    slli a3, a0, 7
+; RV64I-NEXT:    sd a3, 512(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a2, 56
+; RV64I-NEXT:    srai t1, a5, 63
+; RV64I-NEXT:    sd t1, 64(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a2, 55
+; RV64I-NEXT:    slli a6, a0, 8
+; RV64I-NEXT:    sd a6, 520(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    srai a7, a5, 63
+; RV64I-NEXT:    sd a7, 312(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, t1, a3
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    slli a6, a2, 54
+; RV64I-NEXT:    slli a3, a0, 9
+; RV64I-NEXT:    sd a3, 528(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    srai s3, a6, 63
+; RV64I-NEXT:    and a6, s3, a3
+; RV64I-NEXT:    slli a7, a2, 53
+; RV64I-NEXT:    slli a3, a0, 10
+; RV64I-NEXT:    sd a3, 504(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    srai s4, a7, 63
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    and a6, s4, a3
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    slli a6, a2, 52
+; RV64I-NEXT:    srai a7, a6, 63
+; RV64I-NEXT:    sd a7, 160(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a2, 51
+; RV64I-NEXT:    srai t2, a6, 63
+; RV64I-NEXT:    sd t2, 88(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a2, 50
+; RV64I-NEXT:    srai t1, a6, 63
+; RV64I-NEXT:    sd t1, 320(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 11
+; RV64I-NEXT:    sd a3, 472(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a6, a7, a3
+; RV64I-NEXT:    slli a3, a0, 12
+; RV64I-NEXT:    sd a3, 488(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a7, t2, a3
+; RV64I-NEXT:    slli a3, a0, 13
+; RV64I-NEXT:    sd a3, 496(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a6, a6, a7
+; RV64I-NEXT:    and a7, t1, a3
+; RV64I-NEXT:    xor a6, a6, a7
+; RV64I-NEXT:    slli a7, a2, 49
+; RV64I-NEXT:    srai t2, a7, 63
+; RV64I-NEXT:    sd t2, 56(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a7, a2, 48
+; RV64I-NEXT:    srai t1, a7, 63
+; RV64I-NEXT:    sd t1, 184(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 14
+; RV64I-NEXT:    sd a3, 464(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a7, t2, a3
+; RV64I-NEXT:    slli a3, a0, 15
+; RV64I-NEXT:    sd a3, 480(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a6, a6, a7
+; RV64I-NEXT:    and a7, t1, a3
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    xor a5, a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    slli a5, a2, 47
+; RV64I-NEXT:    slli a6, a2, 46
+; RV64I-NEXT:    srai a7, a5, 63
+; RV64I-NEXT:    sd a7, 264(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    srai a6, a6, 63
+; RV64I-NEXT:    sd a6, 256(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a0, 16
+; RV64I-NEXT:    sd a5, 448(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 17
+; RV64I-NEXT:    sd a3, 456(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    and a6, a6, a3
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    slli a6, a2, 45
+; RV64I-NEXT:    srai a7, a6, 63
+; RV64I-NEXT:    sd a7, 232(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a2, 44
+; RV64I-NEXT:    srai t1, a6, 63
+; RV64I-NEXT:    sd t1, 224(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 18
+; RV64I-NEXT:    sd a3, 432(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a6, a7, a3
+; RV64I-NEXT:    slli a3, a0, 19
+; RV64I-NEXT:    sd a3, 440(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    and a6, t1, a3
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    slli a6, a2, 43
+; RV64I-NEXT:    srai a7, a6, 63
+; RV64I-NEXT:    sd a7, 192(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a2, 42
+; RV64I-NEXT:    srai t1, a6, 63
+; RV64I-NEXT:    sd t1, 176(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 20
+; RV64I-NEXT:    sd a3, 416(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a6, a7, a3
+; RV64I-NEXT:    slli a3, a0, 21
+; RV64I-NEXT:    sd a3, 424(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    and a6, t1, a3
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    slli a6, a2, 41
+; RV64I-NEXT:    srai a7, a6, 63
+; RV64I-NEXT:    sd a7, 0(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a2, 40
+; RV64I-NEXT:    srai t1, a6, 63
+; RV64I-NEXT:    sd t1, 80(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a2, 39
+; RV64I-NEXT:    srai t2, a6, 63
+; RV64I-NEXT:    sd t2, 72(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 22
+; RV64I-NEXT:    sd a3, 384(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a6, a7, a3
+; RV64I-NEXT:    slli a3, a0, 23
+; RV64I-NEXT:    sd a3, 392(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a7, t1, a3
+; RV64I-NEXT:    slli a3, a0, 24
+; RV64I-NEXT:    sd a3, 408(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a6, a6, a7
+; RV64I-NEXT:    and a7, t2, a3
+; RV64I-NEXT:    xor a6, a6, a7
+; RV64I-NEXT:    slli a7, a2, 38
+; RV64I-NEXT:    srai t1, a7, 63
+; RV64I-NEXT:    sd t1, 48(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a7, a2, 37
+; RV64I-NEXT:    srai t2, a7, 63
+; RV64I-NEXT:    sd t2, 40(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 25
+; RV64I-NEXT:    sd a3, 368(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a7, t1, a3
+; RV64I-NEXT:    slli a3, a0, 26
+; RV64I-NEXT:    sd a3, 400(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a6, a6, a7
+; RV64I-NEXT:    and a7, t2, a3
+; RV64I-NEXT:    xor a6, a6, a7
+; RV64I-NEXT:    slli a7, a2, 36
+; RV64I-NEXT:    srai t1, a7, 63
+; RV64I-NEXT:    sd t1, 24(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a7, a2, 35
+; RV64I-NEXT:    srai t2, a7, 63
+; RV64I-NEXT:    sd t2, 16(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 27
+; RV64I-NEXT:    sd a3, 352(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a7, t1, a3
+; RV64I-NEXT:    slli a3, a0, 28
+; RV64I-NEXT:    sd a3, 376(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a6, a6, a7
+; RV64I-NEXT:    and a7, t2, a3
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    xor a5, a6, a7
+; RV64I-NEXT:    xor t1, a4, a5
+; RV64I-NEXT:    slli a4, a2, 34
+; RV64I-NEXT:    srai a5, a4, 63
+; RV64I-NEXT:    sd a5, 8(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a4, a2, 33
+; RV64I-NEXT:    srai a6, a4, 63
+; RV64I-NEXT:    sd a6, 144(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 29
+; RV64I-NEXT:    sd a3, 344(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a5, a3
+; RV64I-NEXT:    slli a3, a0, 30
+; RV64I-NEXT:    sd a3, 360(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a6, a3
+; RV64I-NEXT:    slli a6, a2, 31
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    srai a5, a6, 63
+; RV64I-NEXT:    sd a5, 32(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 31
+; RV64I-NEXT:    sd a3, 336(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sraiw t2, a2, 31
+; RV64I-NEXT:    and a6, t2, a3
+; RV64I-NEXT:    slli a7, a0, 32
+; RV64I-NEXT:    xor a4, a4, a6
+; RV64I-NEXT:    and a6, a5, a7
+; RV64I-NEXT:    xor a4, a4, a6
+; RV64I-NEXT:    slli a6, a2, 30
+; RV64I-NEXT:    srai t4, a6, 63
+; RV64I-NEXT:    slli a6, a0, 33
+; RV64I-NEXT:    and a6, t4, a6
+; RV64I-NEXT:    slli a7, a2, 29
+; RV64I-NEXT:    srai s9, a7, 63
+; RV64I-NEXT:    slli a7, a0, 34
+; RV64I-NEXT:    xor a4, a4, a6
+; RV64I-NEXT:    and a6, s9, a7
+; RV64I-NEXT:    xor a4, a4, a6
+; RV64I-NEXT:    slli a6, a2, 28
+; RV64I-NEXT:    srai s6, a6, 63
+; RV64I-NEXT:    slli a7, a0, 35
+; RV64I-NEXT:    and t5, s6, a7
+; RV64I-NEXT:    slli a7, a2, 27
+; RV64I-NEXT:    srai s7, a7, 63
+; RV64I-NEXT:    slli t6, a0, 36
+; RV64I-NEXT:    xor a4, a4, t5
+; RV64I-NEXT:    and t5, s7, t6
+; RV64I-NEXT:    xor a4, a4, t5
+; RV64I-NEXT:    slli t5, a2, 26
+; RV64I-NEXT:    srai a3, t5, 63
+; RV64I-NEXT:    sd a3, 240(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli t5, a0, 37
+; RV64I-NEXT:    and t6, a3, t5
+; RV64I-NEXT:    slli t5, a2, 25
+; RV64I-NEXT:    srai s8, t5, 63
+; RV64I-NEXT:    slli s0, a0, 38
+; RV64I-NEXT:    and s0, s8, s0
+; RV64I-NEXT:    slli s1, a2, 24
+; RV64I-NEXT:    srai a3, s1, 63
+; RV64I-NEXT:    sd a3, 216(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s1, a0, 39
+; RV64I-NEXT:    xor t6, t6, s0
+; RV64I-NEXT:    and s0, a3, s1
+; RV64I-NEXT:    xor s0, t6, s0
+; RV64I-NEXT:    slli t6, a2, 23
+; RV64I-NEXT:    srai t6, t6, 63
+; RV64I-NEXT:    slli s1, a0, 40
+; RV64I-NEXT:    and s1, t6, s1
+; RV64I-NEXT:    slli s5, a2, 22
+; RV64I-NEXT:    srai a3, s5, 63
+; RV64I-NEXT:    sd a3, 208(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s5, a0, 41
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    and s1, a3, s5
+; RV64I-NEXT:    xor s1, s0, s1
+; RV64I-NEXT:    slli s0, a2, 21
+; RV64I-NEXT:    srai s0, s0, 63
+; RV64I-NEXT:    slli s5, a0, 42
+; RV64I-NEXT:    and s5, s0, s5
+; RV64I-NEXT:    slli s10, a2, 20
+; RV64I-NEXT:    srai a3, s10, 63
+; RV64I-NEXT:    sd a3, 200(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s10, a0, 43
+; RV64I-NEXT:    xor s1, s1, s5
+; RV64I-NEXT:    and s5, a3, s10
+; RV64I-NEXT:    xor s10, s1, s5
+; RV64I-NEXT:    slli s1, a2, 19
+; RV64I-NEXT:    srai s1, s1, 63
+; RV64I-NEXT:    slli s5, a0, 44
+; RV64I-NEXT:    and s11, s1, s5
+; RV64I-NEXT:    slli s5, a2, 18
+; RV64I-NEXT:    srai s5, s5, 63
+; RV64I-NEXT:    slli ra, a0, 45
+; RV64I-NEXT:    xor s10, s10, s11
+; RV64I-NEXT:    and s11, s5, ra
+; RV64I-NEXT:    xor t1, t1, a4
+; RV64I-NEXT:    xor a6, s10, s11
+; RV64I-NEXT:    slli s10, a2, 17
+; RV64I-NEXT:    slli s11, a2, 16
+; RV64I-NEXT:    srai a3, s10, 63
+; RV64I-NEXT:    sd a3, 120(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    srai a4, s11, 63
+; RV64I-NEXT:    sd a4, 112(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s10, a0, 46
+; RV64I-NEXT:    slli s11, a0, 47
+; RV64I-NEXT:    and s10, a3, s10
+; RV64I-NEXT:    and s11, a4, s11
+; RV64I-NEXT:    xor s11, s10, s11
+; RV64I-NEXT:    slli s10, a2, 15
+; RV64I-NEXT:    srai a3, s10, 63
+; RV64I-NEXT:    sd a3, 296(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s10, a0, 48
+; RV64I-NEXT:    and ra, a3, s10
+; RV64I-NEXT:    slli s10, a2, 14
+; RV64I-NEXT:    srai a4, s10, 63
+; RV64I-NEXT:    sd a4, 104(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 49
+; RV64I-NEXT:    xor s11, s11, ra
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a4, s11, a3
+; RV64I-NEXT:    slli s11, a2, 13
+; RV64I-NEXT:    srai a3, s11, 63
+; RV64I-NEXT:    sd a3, 288(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s11, a0, 50
+; RV64I-NEXT:    and ra, a3, s11
+; RV64I-NEXT:    slli s11, a2, 12
+; RV64I-NEXT:    srai s11, s11, 63
+; RV64I-NEXT:    slli a3, a0, 51
+; RV64I-NEXT:    xor a4, a4, ra
+; RV64I-NEXT:    and a3, s11, a3
+; RV64I-NEXT:    xor a5, a4, a3
+; RV64I-NEXT:    slli a4, a2, 11
+; RV64I-NEXT:    srai a3, a4, 63
+; RV64I-NEXT:    sd a3, 280(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a4, a0, 52
+; RV64I-NEXT:    and a4, a3, a4
+; RV64I-NEXT:    slli ra, a2, 10
+; RV64I-NEXT:    srai ra, ra, 63
+; RV64I-NEXT:    slli a3, a0, 53
+; RV64I-NEXT:    xor a4, a5, a4
+; RV64I-NEXT:    and a3, ra, a3
+; RV64I-NEXT:    xor a3, a4, a3
+; RV64I-NEXT:    slli a4, a2, 9
+; RV64I-NEXT:    srai a5, a4, 63
+; RV64I-NEXT:    sd a5, 272(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a4, a0, 54
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    slli a5, a2, 8
+; RV64I-NEXT:    srai a7, a5, 63
+; RV64I-NEXT:    sd a7, 96(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a0, 55
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a7, a5
+; RV64I-NEXT:    xor a5, t1, a6
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    xor a3, a5, a3
+; RV64I-NEXT:    sd a3, 328(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a2, 7
+; RV64I-NEXT:    slli a4, a2, 6
+; RV64I-NEXT:    srai t5, a3, 63
+; RV64I-NEXT:    srai a5, a4, 63
+; RV64I-NEXT:    sd a5, 168(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a3, a0, 56
+; RV64I-NEXT:    slli a4, a0, 57
+; RV64I-NEXT:    and a3, t5, a3
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    srli a5, a0, 63
+; RV64I-NEXT:    slli a6, a1, 1
+; RV64I-NEXT:    xor a4, a3, a4
+; RV64I-NEXT:    or a3, a6, a5
+; RV64I-NEXT:    and a5, t3, a3
+; RV64I-NEXT:    slli a3, a2, 5
+; RV64I-NEXT:    srai t1, a3, 63
+; RV64I-NEXT:    slli a6, a0, 58
+; RV64I-NEXT:    and a6, t1, a6
+; RV64I-NEXT:    and t3, s2, a1
+; RV64I-NEXT:    xor a7, a4, a6
+; RV64I-NEXT:    xor a5, t3, a5
+; RV64I-NEXT:    srli a6, a0, 62
+; RV64I-NEXT:    slli t3, a1, 2
+; RV64I-NEXT:    srli s2, a0, 61
+; RV64I-NEXT:    slli a3, a1, 3
+; RV64I-NEXT:    or a6, t3, a6
+; RV64I-NEXT:    or a3, a3, s2
+; RV64I-NEXT:    ld a4, 152(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a4, a6
+; RV64I-NEXT:    ld a4, 136(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    srli t3, a0, 60
+; RV64I-NEXT:    slli s2, a1, 4
+; RV64I-NEXT:    xor a4, a6, a3
+; RV64I-NEXT:    or a6, s2, t3
+; RV64I-NEXT:    srli t3, a0, 59
+; RV64I-NEXT:    slli s2, a1, 5
+; RV64I-NEXT:    ld a3, 248(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a3, a6
+; RV64I-NEXT:    or t3, s2, t3
+; RV64I-NEXT:    srli s2, a0, 58
+; RV64I-NEXT:    slli a3, a1, 6
+; RV64I-NEXT:    ld s10, 304(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t3, s10, t3
+; RV64I-NEXT:    or a3, a3, s2
+; RV64I-NEXT:    xor a6, a6, t3
+; RV64I-NEXT:    ld t3, 128(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, t3, a3
+; RV64I-NEXT:    xor a4, a5, a4
+; RV64I-NEXT:    xor a3, a6, a3
+; RV64I-NEXT:    srli a5, a0, 57
+; RV64I-NEXT:    slli a6, a1, 7
+; RV64I-NEXT:    xor a3, a4, a3
+; RV64I-NEXT:    or a4, a6, a5
+; RV64I-NEXT:    srli a5, a0, 56
+; RV64I-NEXT:    slli a6, a1, 8
+; RV64I-NEXT:    ld t3, 64(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, t3, a4
+; RV64I-NEXT:    or a5, a6, a5
+; RV64I-NEXT:    srli a6, a0, 55
+; RV64I-NEXT:    slli t3, a1, 9
+; RV64I-NEXT:    ld s2, 312(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, s2, a5
+; RV64I-NEXT:    or a6, t3, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, s3, a6
+; RV64I-NEXT:    srli a6, a0, 54
+; RV64I-NEXT:    slli t3, a1, 10
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    or a5, t3, a6
+; RV64I-NEXT:    and a5, s4, a5
+; RV64I-NEXT:    slli a6, a2, 4
+; RV64I-NEXT:    srai t3, a6, 63
+; RV64I-NEXT:    slli a6, a0, 59
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, t3, a6
+; RV64I-NEXT:    xor a5, a7, a5
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    srli a4, a0, 53
+; RV64I-NEXT:    slli a6, a1, 11
+; RV64I-NEXT:    srli a7, a0, 52
+; RV64I-NEXT:    slli s2, a1, 12
+; RV64I-NEXT:    or a4, a6, a4
+; RV64I-NEXT:    or a6, s2, a7
+; RV64I-NEXT:    ld a7, 160(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a7, a4
+; RV64I-NEXT:    ld a7, 88(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    srli a7, a0, 51
+; RV64I-NEXT:    slli s2, a1, 13
+; RV64I-NEXT:    xor a4, a4, a6
+; RV64I-NEXT:    or a6, s2, a7
+; RV64I-NEXT:    srli a7, a0, 50
+; RV64I-NEXT:    slli s2, a1, 14
+; RV64I-NEXT:    ld s3, 320(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, s3, a6
+; RV64I-NEXT:    or a7, s2, a7
+; RV64I-NEXT:    xor a4, a4, a6
+; RV64I-NEXT:    ld a6, 56(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a6, a7
+; RV64I-NEXT:    srli a7, a0, 49
+; RV64I-NEXT:    slli s2, a1, 15
+; RV64I-NEXT:    xor a4, a4, a6
+; RV64I-NEXT:    or a6, s2, a7
+; RV64I-NEXT:    ld a7, 184(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    slli a7, a2, 3
+; RV64I-NEXT:    srai s2, a7, 63
+; RV64I-NEXT:    slli a7, a0, 60
+; RV64I-NEXT:    xor a4, a4, a6
+; RV64I-NEXT:    and a6, s2, a7
+; RV64I-NEXT:    xor s3, a5, a6
+; RV64I-NEXT:    xor s4, a3, a4
+; RV64I-NEXT:    srli a3, a0, 48
+; RV64I-NEXT:    slli a4, a1, 16
+; RV64I-NEXT:    srli a5, a0, 47
+; RV64I-NEXT:    slli a6, a1, 17
+; RV64I-NEXT:    or a3, a4, a3
+; RV64I-NEXT:    or a4, a6, a5
+; RV64I-NEXT:    ld a5, 264(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a5, a3
+; RV64I-NEXT:    ld a5, 256(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    srli a5, a0, 46
+; RV64I-NEXT:    slli a6, a1, 18
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    or a4, a6, a5
+; RV64I-NEXT:    srli a5, a0, 45
+; RV64I-NEXT:    slli a6, a1, 19
+; RV64I-NEXT:    ld a7, 232(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a7, a4
+; RV64I-NEXT:    or a5, a6, a5
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    ld a4, 224(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    srli a5, a0, 44
+; RV64I-NEXT:    slli a6, a1, 20
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    or a4, a6, a5
+; RV64I-NEXT:    srli a5, a0, 43
+; RV64I-NEXT:    slli a6, a1, 21
+; RV64I-NEXT:    ld a7, 192(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a7, a4
+; RV64I-NEXT:    or a5, a6, a5
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    ld a4, 176(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    srli a5, a0, 42
+; RV64I-NEXT:    slli a6, a1, 22
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    or a4, a6, a5
+; RV64I-NEXT:    srli a5, a0, 41
+; RV64I-NEXT:    slli a6, a1, 23
+; RV64I-NEXT:    ld a7, 0(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a7, a4
+; RV64I-NEXT:    or a5, a6, a5
+; RV64I-NEXT:    srli a6, a0, 40
+; RV64I-NEXT:    slli a7, a1, 24
+; RV64I-NEXT:    ld s10, 80(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, s10, a5
+; RV64I-NEXT:    or a6, a7, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    ld a5, 72(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    srli a6, a0, 39
+; RV64I-NEXT:    slli a7, a1, 25
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    or a5, a7, a6
+; RV64I-NEXT:    srli a6, a0, 38
+; RV64I-NEXT:    slli a7, a1, 26
+; RV64I-NEXT:    ld s10, 48(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, s10, a5
+; RV64I-NEXT:    or a6, a7, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    ld a5, 40(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    srli a6, a0, 37
+; RV64I-NEXT:    slli a7, a1, 27
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    or a5, a7, a6
+; RV64I-NEXT:    srli a6, a0, 36
+; RV64I-NEXT:    slli a7, a1, 28
+; RV64I-NEXT:    ld s10, 24(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, s10, a5
+; RV64I-NEXT:    or a6, a7, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    ld a5, 16(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    xor a3, s4, a3
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    srli a5, a0, 35
+; RV64I-NEXT:    slli a6, a1, 29
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    or a4, a6, a5
+; RV64I-NEXT:    srli a5, a0, 34
+; RV64I-NEXT:    slli a6, a1, 30
+; RV64I-NEXT:    ld a7, 8(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a7, a4
+; RV64I-NEXT:    or a5, a6, a5
+; RV64I-NEXT:    srli a6, a0, 33
+; RV64I-NEXT:    slli a7, a1, 31
+; RV64I-NEXT:    ld s4, 144(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, s4, a5
+; RV64I-NEXT:    or a6, a7, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, t2, a6
+; RV64I-NEXT:    srli a6, a0, 32
+; RV64I-NEXT:    slli a7, a1, 32
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    or a5, a7, a6
+; RV64I-NEXT:    srli a6, a0, 31
+; RV64I-NEXT:    slli a7, a1, 33
+; RV64I-NEXT:    ld t2, 32(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, t2, a5
+; RV64I-NEXT:    or a6, a7, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, t4, a6
+; RV64I-NEXT:    srli a6, a0, 30
+; RV64I-NEXT:    slli a7, a1, 34
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    or a5, a7, a6
+; RV64I-NEXT:    srli a6, a0, 29
+; RV64I-NEXT:    slli a7, a1, 35
+; RV64I-NEXT:    and a5, s9, a5
+; RV64I-NEXT:    or a6, a7, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    and a5, s6, a6
+; RV64I-NEXT:    srli a6, a0, 28
+; RV64I-NEXT:    slli a7, a1, 36
+; RV64I-NEXT:    xor a5, a4, a5
+; RV64I-NEXT:    or a4, a7, a6
+; RV64I-NEXT:    and a6, s7, a4
+; RV64I-NEXT:    slli a4, a2, 2
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    slli a7, a0, 61
+; RV64I-NEXT:    xor a6, a5, a6
+; RV64I-NEXT:    and a5, a4, a7
+; RV64I-NEXT:    xor a5, s3, a5
+; RV64I-NEXT:    xor a6, a3, a6
+; RV64I-NEXT:    srli a3, a0, 27
+; RV64I-NEXT:    slli a7, a1, 37
+; RV64I-NEXT:    srli t2, a0, 26
+; RV64I-NEXT:    slli t4, a1, 38
+; RV64I-NEXT:    or a3, a7, a3
+; RV64I-NEXT:    or a7, t4, t2
+; RV64I-NEXT:    ld t2, 240(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, t2, a3
+; RV64I-NEXT:    and a7, s8, a7
+; RV64I-NEXT:    srli t2, a0, 25
+; RV64I-NEXT:    slli t4, a1, 39
+; RV64I-NEXT:    xor a3, a3, a7
+; RV64I-NEXT:    or a7, t4, t2
+; RV64I-NEXT:    srli t2, a0, 24
+; RV64I-NEXT:    slli t4, a1, 40
+; RV64I-NEXT:    ld s3, 216(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, s3, a7
+; RV64I-NEXT:    or t2, t4, t2
+; RV64I-NEXT:    xor a3, a3, a7
+; RV64I-NEXT:    and a7, t6, t2
+; RV64I-NEXT:    srli t2, a0, 23
+; RV64I-NEXT:    slli t4, a1, 41
+; RV64I-NEXT:    xor a3, a3, a7
+; RV64I-NEXT:    or a7, t4, t2
+; RV64I-NEXT:    srli t2, a0, 22
+; RV64I-NEXT:    slli t4, a1, 42
+; RV64I-NEXT:    ld t6, 208(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, t6, a7
+; RV64I-NEXT:    or t2, t4, t2
+; RV64I-NEXT:    xor a3, a3, a7
+; RV64I-NEXT:    and a7, s0, t2
+; RV64I-NEXT:    srli t2, a0, 21
+; RV64I-NEXT:    slli t4, a1, 43
+; RV64I-NEXT:    xor a3, a3, a7
+; RV64I-NEXT:    or a7, t4, t2
+; RV64I-NEXT:    srli t2, a0, 20
+; RV64I-NEXT:    slli t4, a1, 44
+; RV64I-NEXT:    ld t6, 200(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, t6, a7
+; RV64I-NEXT:    or t2, t4, t2
+; RV64I-NEXT:    xor a3, a3, a7
+; RV64I-NEXT:    and a7, s1, t2
+; RV64I-NEXT:    srli t2, a0, 19
+; RV64I-NEXT:    slli t4, a1, 45
+; RV64I-NEXT:    xor a3, a3, a7
+; RV64I-NEXT:    or a7, t4, t2
+; RV64I-NEXT:    and t2, s5, a7
+; RV64I-NEXT:    slli a7, a2, 1
+; RV64I-NEXT:    srai a7, a7, 63
+; RV64I-NEXT:    slli t4, a0, 62
+; RV64I-NEXT:    xor a3, a3, t2
+; RV64I-NEXT:    and t2, a7, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    xor a6, a6, a3
+; RV64I-NEXT:    srli a3, a0, 18
+; RV64I-NEXT:    slli t2, a1, 46
+; RV64I-NEXT:    srli t4, a0, 17
+; RV64I-NEXT:    slli t6, a1, 47
+; RV64I-NEXT:    or a3, t2, a3
+; RV64I-NEXT:    or t2, t6, t4
+; RV64I-NEXT:    ld t4, 120(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, t4, a3
+; RV64I-NEXT:    ld t4, 112(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t2, t4, t2
+; RV64I-NEXT:    srli t4, a0, 16
+; RV64I-NEXT:    slli t6, a1, 48
+; RV64I-NEXT:    xor a3, a3, t2
+; RV64I-NEXT:    or t2, t6, t4
+; RV64I-NEXT:    srli t4, a0, 15
+; RV64I-NEXT:    slli t6, a1, 49
+; RV64I-NEXT:    ld s0, 296(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t2, s0, t2
+; RV64I-NEXT:    or t4, t6, t4
+; RV64I-NEXT:    xor a3, a3, t2
+; RV64I-NEXT:    ld t2, 104(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t2, t2, t4
+; RV64I-NEXT:    srli t4, a0, 14
+; RV64I-NEXT:    slli t6, a1, 50
+; RV64I-NEXT:    xor a3, a3, t2
+; RV64I-NEXT:    or t2, t6, t4
+; RV64I-NEXT:    srli t4, a0, 13
+; RV64I-NEXT:    slli t6, a1, 51
+; RV64I-NEXT:    ld s0, 288(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t2, s0, t2
+; RV64I-NEXT:    or t4, t6, t4
+; RV64I-NEXT:    xor a3, a3, t2
+; RV64I-NEXT:    and t2, s11, t4
+; RV64I-NEXT:    srli t4, a0, 12
+; RV64I-NEXT:    slli t6, a1, 52
+; RV64I-NEXT:    xor a3, a3, t2
+; RV64I-NEXT:    or t2, t6, t4
+; RV64I-NEXT:    srli t4, a0, 11
+; RV64I-NEXT:    slli t6, a1, 53
+; RV64I-NEXT:    ld s0, 280(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t2, s0, t2
+; RV64I-NEXT:    or t4, t6, t4
+; RV64I-NEXT:    xor a3, a3, t2
+; RV64I-NEXT:    and t2, ra, t4
+; RV64I-NEXT:    srli t4, a0, 10
+; RV64I-NEXT:    slli t6, a1, 54
+; RV64I-NEXT:    xor a3, a3, t2
+; RV64I-NEXT:    or t2, t6, t4
+; RV64I-NEXT:    srli t4, a0, 9
+; RV64I-NEXT:    slli t6, a1, 55
+; RV64I-NEXT:    ld s0, 272(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t2, s0, t2
+; RV64I-NEXT:    or t4, t6, t4
+; RV64I-NEXT:    xor a3, a3, t2
+; RV64I-NEXT:    ld t2, 96(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t2, t2, t4
+; RV64I-NEXT:    srli t4, a0, 8
+; RV64I-NEXT:    slli t6, a1, 56
+; RV64I-NEXT:    xor t2, a3, t2
+; RV64I-NEXT:    or a3, t6, t4
+; RV64I-NEXT:    srli t4, a0, 7
+; RV64I-NEXT:    slli t6, a1, 57
+; RV64I-NEXT:    and a3, t5, a3
+; RV64I-NEXT:    or t4, t6, t4
+; RV64I-NEXT:    srli t5, a0, 6
+; RV64I-NEXT:    slli t6, a1, 58
+; RV64I-NEXT:    ld s0, 168(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t4, s0, t4
+; RV64I-NEXT:    or t5, t6, t5
+; RV64I-NEXT:    xor a3, a3, t4
+; RV64I-NEXT:    and t1, t1, t5
+; RV64I-NEXT:    srli t4, a0, 5
+; RV64I-NEXT:    slli t5, a1, 59
+; RV64I-NEXT:    xor a3, a3, t1
+; RV64I-NEXT:    or t1, t5, t4
+; RV64I-NEXT:    srli t4, a0, 4
+; RV64I-NEXT:    slli t5, a1, 60
+; RV64I-NEXT:    and t1, t3, t1
+; RV64I-NEXT:    or t3, t5, t4
+; RV64I-NEXT:    xor a3, a3, t1
+; RV64I-NEXT:    and t1, s2, t3
+; RV64I-NEXT:    srli t3, a0, 3
+; RV64I-NEXT:    slli t4, a1, 61
+; RV64I-NEXT:    xor a3, a3, t1
+; RV64I-NEXT:    or t1, t4, t3
+; RV64I-NEXT:    srli t3, a0, 2
+; RV64I-NEXT:    slli t4, a1, 62
+; RV64I-NEXT:    and a4, a4, t1
+; RV64I-NEXT:    or t1, t4, t3
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    and a4, a7, t1
+; RV64I-NEXT:    slli a1, a1, 63
+; RV64I-NEXT:    srli a7, a0, 1
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    or a1, a1, a7
+; RV64I-NEXT:    srai a2, a2, 63
+; RV64I-NEXT:    slli a4, a0, 63
+; RV64I-NEXT:    and a4, a2, a4
+; RV64I-NEXT:    slli a7, t0, 63
+; RV64I-NEXT:    and a1, a2, a1
+; RV64I-NEXT:    srai a2, a7, 63
+; RV64I-NEXT:    xor a1, a3, a1
+; RV64I-NEXT:    and a0, a2, a0
+; RV64I-NEXT:    xor a0, a1, a0
+; RV64I-NEXT:    slli a1, t0, 62
+; RV64I-NEXT:    srai a1, a1, 63
+; RV64I-NEXT:    slli a2, t0, 61
+; RV64I-NEXT:    ld a3, 568(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a1, a1, a3
+; RV64I-NEXT:    srai a2, a2, 63
+; RV64I-NEXT:    xor a0, a0, a1
+; RV64I-NEXT:    ld a1, 536(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a1, a2, a1
+; RV64I-NEXT:    xor a2, a6, t2
+; RV64I-NEXT:    xor a1, a0, a1
+; RV64I-NEXT:    xor a0, a5, a4
+; RV64I-NEXT:    xor a1, a2, a1
+; RV64I-NEXT:    slli a2, t0, 60
+; RV64I-NEXT:    slli a3, t0, 59
+; RV64I-NEXT:    srai a2, a2, 63
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    ld a4, 576(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a2, a2, a4
+; RV64I-NEXT:    ld a4, 552(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a4
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 58
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 57
+; RV64I-NEXT:    ld a5, 560(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 544(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 56
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 55
+; RV64I-NEXT:    ld a5, 512(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 520(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 54
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 53
+; RV64I-NEXT:    ld a5, 528(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 504(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 52
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 51
+; RV64I-NEXT:    ld a5, 472(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 488(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 50
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 49
+; RV64I-NEXT:    ld a5, 496(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 464(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 48
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    ld a2, 480(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a2, a3, a2
+; RV64I-NEXT:    slli a3, t0, 47
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 46
+; RV64I-NEXT:    ld a5, 448(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 456(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 45
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 44
+; RV64I-NEXT:    ld a5, 432(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 440(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 43
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 42
+; RV64I-NEXT:    ld a5, 416(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 424(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 41
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 40
+; RV64I-NEXT:    ld a5, 384(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 392(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 39
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 38
+; RV64I-NEXT:    ld a5, 408(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 368(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    slli a3, t0, 37
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    slli a4, t0, 36
+; RV64I-NEXT:    ld a5, 400(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    ld a3, 352(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    slli a4, t0, 35
+; RV64I-NEXT:    slli a5, t0, 34
+; RV64I-NEXT:    srai a4, a4, 63
+; RV64I-NEXT:    srai a5, a5, 63
+; RV64I-NEXT:    ld a6, 376(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    ld a6, 344(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    slli a3, t0, 33
+; RV64I-NEXT:    sraiw a5, t0, 31
+; RV64I-NEXT:    srai a3, a3, 63
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    ld a6, 360(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, a6
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    xor a3, a4, a3
+; RV64I-NEXT:    ld a4, 336(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    xor a3, a3, a4
+; RV64I-NEXT:    ld a2, 328(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    xor a0, a2, a0
+; RV64I-NEXT:    xor a1, a1, a3
+; RV64I-NEXT:    ld ra, 680(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s0, 672(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s1, 664(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s2, 656(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 648(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s4, 640(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 632(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s6, 624(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s7, 616(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s8, 608(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s9, 600(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s10, 592(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s11, 584(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    .cfi_restore ra
+; RV64I-NEXT:    .cfi_restore s0
+; RV64I-NEXT:    .cfi_restore s1
+; RV64I-NEXT:    .cfi_restore s2
+; RV64I-NEXT:    .cfi_restore s3
+; RV64I-NEXT:    .cfi_restore s4
+; RV64I-NEXT:    .cfi_restore s5
+; RV64I-NEXT:    .cfi_restore s6
+; RV64I-NEXT:    .cfi_restore s7
+; RV64I-NEXT:    .cfi_restore s8
+; RV64I-NEXT:    .cfi_restore s9
+; RV64I-NEXT:    .cfi_restore s10
+; RV64I-NEXT:    .cfi_restore s11
+; RV64I-NEXT:    addi sp, sp, 688
+; RV64I-NEXT:    .cfi_def_cfa_offset 0
+; RV64I-NEXT:    ret
+;
+; RV32IM-LABEL: clmul_i96:
+; RV32IM:       # %bb.0:
+; RV32IM-NEXT:    addi sp, sp, -496
+; RV32IM-NEXT:    .cfi_def_cfa_offset 496
+; RV32IM-NEXT:    sw ra, 492(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s0, 488(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s1, 484(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s2, 480(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s3, 476(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s4, 472(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s5, 468(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s6, 464(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s7, 460(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s8, 456(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s9, 452(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s10, 448(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s11, 444(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    .cfi_offset ra, -4
+; RV32IM-NEXT:    .cfi_offset s0, -8
+; RV32IM-NEXT:    .cfi_offset s1, -12
+; RV32IM-NEXT:    .cfi_offset s2, -16
+; RV32IM-NEXT:    .cfi_offset s3, -20
+; RV32IM-NEXT:    .cfi_offset s4, -24
+; RV32IM-NEXT:    .cfi_offset s5, -28
+; RV32IM-NEXT:    .cfi_offset s6, -32
+; RV32IM-NEXT:    .cfi_offset s7, -36
+; RV32IM-NEXT:    .cfi_offset s8, -40
+; RV32IM-NEXT:    .cfi_offset s9, -44
+; RV32IM-NEXT:    .cfi_offset s10, -48
+; RV32IM-NEXT:    .cfi_offset s11, -52
+; RV32IM-NEXT:    sw a0, 300(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw s10, 4(a1)
+; RV32IM-NEXT:    lw a0, 8(a1)
+; RV32IM-NEXT:    lw a4, 0(a2)
+; RV32IM-NEXT:    lw s2, 4(a2)
+; RV32IM-NEXT:    lw s11, 8(a2)
+; RV32IM-NEXT:    srli a2, s10, 31
+; RV32IM-NEXT:    slli a5, a0, 1
+; RV32IM-NEXT:    slli a6, a4, 30
+; RV32IM-NEXT:    slli a7, a4, 31
+; RV32IM-NEXT:    or a2, a5, a2
+; RV32IM-NEXT:    srai a5, a6, 31
+; RV32IM-NEXT:    sw a5, 368(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai a6, a7, 31
+; RV32IM-NEXT:    sw a6, 364(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a2, a5, a2
+; RV32IM-NEXT:    and a5, a6, a0
+; RV32IM-NEXT:    xor a2, a5, a2
+; RV32IM-NEXT:    srli a5, s10, 30
+; RV32IM-NEXT:    slli a6, a0, 2
+; RV32IM-NEXT:    or a5, a6, a5
+; RV32IM-NEXT:    slli a6, a4, 29
+; RV32IM-NEXT:    srai t1, a6, 31
+; RV32IM-NEXT:    sw t1, 360(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a6, s10, 29
+; RV32IM-NEXT:    slli a7, a0, 3
+; RV32IM-NEXT:    slli t0, a4, 28
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    srai a7, t0, 31
+; RV32IM-NEXT:    sw a7, 356(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, t1, a5
+; RV32IM-NEXT:    and a6, a7, a6
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    srli a6, s10, 28
+; RV32IM-NEXT:    slli a7, a0, 4
+; RV32IM-NEXT:    slli t0, a4, 27
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    srai a7, t0, 31
+; RV32IM-NEXT:    sw a7, 352(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a6, a7, a6
+; RV32IM-NEXT:    srli a7, s10, 27
+; RV32IM-NEXT:    slli t0, a0, 5
+; RV32IM-NEXT:    slli t1, a4, 26
+; RV32IM-NEXT:    or a7, t0, a7
+; RV32IM-NEXT:    srai t0, t1, 31
+; RV32IM-NEXT:    sw t0, 348(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a7, t0, a7
+; RV32IM-NEXT:    srli t0, s10, 26
+; RV32IM-NEXT:    slli t1, a0, 6
+; RV32IM-NEXT:    slli t2, a4, 25
+; RV32IM-NEXT:    or t0, t1, t0
+; RV32IM-NEXT:    srai t1, t2, 31
+; RV32IM-NEXT:    sw t1, 344(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    and a7, t1, t0
+; RV32IM-NEXT:    xor a2, a2, a5
+; RV32IM-NEXT:    xor a5, a6, a7
+; RV32IM-NEXT:    xor a2, a2, a5
+; RV32IM-NEXT:    srli a5, s10, 25
+; RV32IM-NEXT:    slli a6, a0, 7
+; RV32IM-NEXT:    or a5, a6, a5
+; RV32IM-NEXT:    slli a6, a4, 24
+; RV32IM-NEXT:    srai t1, a6, 31
+; RV32IM-NEXT:    sw t1, 340(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a6, s10, 24
+; RV32IM-NEXT:    slli a7, a0, 8
+; RV32IM-NEXT:    slli t0, a4, 23
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    srai a7, t0, 31
+; RV32IM-NEXT:    sw a7, 336(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, t1, a5
+; RV32IM-NEXT:    and a6, a7, a6
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    srli a6, s10, 23
+; RV32IM-NEXT:    slli a7, a0, 9
+; RV32IM-NEXT:    slli t0, a4, 22
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    srai a7, t0, 31
+; RV32IM-NEXT:    sw a7, 332(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a6, a7, a6
+; RV32IM-NEXT:    srli a7, s10, 22
+; RV32IM-NEXT:    slli t0, a0, 10
+; RV32IM-NEXT:    slli t1, a4, 21
+; RV32IM-NEXT:    or a7, t0, a7
+; RV32IM-NEXT:    srai t0, t1, 31
+; RV32IM-NEXT:    sw t0, 328(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    and a6, t0, a7
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    srli a6, s10, 21
+; RV32IM-NEXT:    slli a7, a0, 11
+; RV32IM-NEXT:    slli t0, a4, 20
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    srai a7, t0, 31
+; RV32IM-NEXT:    sw a7, 324(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a6, a7, a6
+; RV32IM-NEXT:    srli a7, s10, 20
+; RV32IM-NEXT:    slli t0, a0, 12
+; RV32IM-NEXT:    slli t1, a4, 19
+; RV32IM-NEXT:    or a7, t0, a7
+; RV32IM-NEXT:    srai t0, t1, 31
+; RV32IM-NEXT:    sw t0, 320(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a7, t0, a7
+; RV32IM-NEXT:    srli t0, s10, 19
+; RV32IM-NEXT:    slli t1, a0, 13
+; RV32IM-NEXT:    slli t2, a4, 18
+; RV32IM-NEXT:    or t0, t1, t0
+; RV32IM-NEXT:    srai t1, t2, 31
+; RV32IM-NEXT:    sw t1, 316(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    and a7, t1, t0
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    srli a7, s10, 18
+; RV32IM-NEXT:    slli t0, a0, 14
+; RV32IM-NEXT:    slli t1, a4, 17
+; RV32IM-NEXT:    or a7, t0, a7
+; RV32IM-NEXT:    srai t2, t1, 31
+; RV32IM-NEXT:    sw t2, 400(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli t0, s10, 17
+; RV32IM-NEXT:    slli t1, a0, 15
+; RV32IM-NEXT:    or t0, t1, t0
+; RV32IM-NEXT:    slli t1, a4, 16
+; RV32IM-NEXT:    and a7, t2, a7
+; RV32IM-NEXT:    srai t1, t1, 31
+; RV32IM-NEXT:    sw t1, 396(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    and a7, t1, t0
+; RV32IM-NEXT:    xor a2, a2, a5
+; RV32IM-NEXT:    xor a5, a6, a7
+; RV32IM-NEXT:    xor a2, a2, a5
+; RV32IM-NEXT:    srli a5, s10, 16
+; RV32IM-NEXT:    slli a6, a0, 16
+; RV32IM-NEXT:    or a5, a6, a5
+; RV32IM-NEXT:    slli a6, a4, 15
+; RV32IM-NEXT:    srli a7, s10, 15
+; RV32IM-NEXT:    slli t0, a0, 17
+; RV32IM-NEXT:    or a7, t0, a7
+; RV32IM-NEXT:    slli t0, a4, 14
+; RV32IM-NEXT:    srai a6, a6, 31
+; RV32IM-NEXT:    sw a6, 392(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai t0, t0, 31
+; RV32IM-NEXT:    sw t0, 388(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a6, a5
+; RV32IM-NEXT:    and a6, t0, a7
+; RV32IM-NEXT:    srli a7, s10, 14
+; RV32IM-NEXT:    slli t0, a0, 18
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    or a6, t0, a7
+; RV32IM-NEXT:    srli a7, s10, 13
+; RV32IM-NEXT:    slli t0, a0, 19
+; RV32IM-NEXT:    or a7, t0, a7
+; RV32IM-NEXT:    slli t0, a4, 13
+; RV32IM-NEXT:    srai t1, t0, 31
+; RV32IM-NEXT:    sw t1, 384(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli t0, a4, 12
+; RV32IM-NEXT:    and a6, t1, a6
+; RV32IM-NEXT:    srai t0, t0, 31
+; RV32IM-NEXT:    sw t0, 380(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    and a6, t0, a7
+; RV32IM-NEXT:    srli a7, s10, 12
+; RV32IM-NEXT:    slli t0, a0, 20
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    or a6, t0, a7
+; RV32IM-NEXT:    srli a7, s10, 11
+; RV32IM-NEXT:    slli t0, a0, 21
+; RV32IM-NEXT:    or a7, t0, a7
+; RV32IM-NEXT:    slli t0, a4, 11
+; RV32IM-NEXT:    srai t1, t0, 31
+; RV32IM-NEXT:    sw t1, 376(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli t0, a4, 10
+; RV32IM-NEXT:    and a6, t1, a6
+; RV32IM-NEXT:    srai t0, t0, 31
+; RV32IM-NEXT:    sw t0, 372(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    and a6, t0, a7
+; RV32IM-NEXT:    srli a7, s10, 10
+; RV32IM-NEXT:    slli t0, a0, 22
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    or a6, t0, a7
+; RV32IM-NEXT:    srli a7, s10, 9
+; RV32IM-NEXT:    slli t0, a0, 23
+; RV32IM-NEXT:    or a7, t0, a7
+; RV32IM-NEXT:    slli t0, a4, 9
+; RV32IM-NEXT:    srli t1, s10, 8
+; RV32IM-NEXT:    slli t2, a0, 24
+; RV32IM-NEXT:    or t1, t2, t1
+; RV32IM-NEXT:    srai t0, t0, 31
+; RV32IM-NEXT:    sw t0, 440(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a6, t0, a6
+; RV32IM-NEXT:    slli t0, a4, 8
+; RV32IM-NEXT:    srai t2, t0, 31
+; RV32IM-NEXT:    sw t2, 436(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli t0, a4, 7
+; RV32IM-NEXT:    and a7, t2, a7
+; RV32IM-NEXT:    srai t0, t0, 31
+; RV32IM-NEXT:    sw t0, 432(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    and a7, t0, t1
+; RV32IM-NEXT:    srli t0, s10, 7
+; RV32IM-NEXT:    slli t1, a0, 25
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    or a7, t1, t0
+; RV32IM-NEXT:    srli t0, s10, 6
+; RV32IM-NEXT:    slli t1, a0, 26
+; RV32IM-NEXT:    or t0, t1, t0
+; RV32IM-NEXT:    slli t1, a4, 6
+; RV32IM-NEXT:    srai t2, t1, 31
+; RV32IM-NEXT:    sw t2, 428(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli t1, a4, 5
+; RV32IM-NEXT:    and a7, t2, a7
+; RV32IM-NEXT:    srai t1, t1, 31
+; RV32IM-NEXT:    sw t1, 424(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    and a7, t1, t0
+; RV32IM-NEXT:    srli t0, s10, 5
+; RV32IM-NEXT:    slli t1, a0, 27
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    or a7, t1, t0
+; RV32IM-NEXT:    srli t0, s10, 4
+; RV32IM-NEXT:    slli t1, a0, 28
+; RV32IM-NEXT:    or t0, t1, t0
+; RV32IM-NEXT:    slli t1, a4, 4
+; RV32IM-NEXT:    srai t2, t1, 31
+; RV32IM-NEXT:    sw t2, 420(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli t1, a4, 3
+; RV32IM-NEXT:    and a7, t2, a7
+; RV32IM-NEXT:    srai t1, t1, 31
+; RV32IM-NEXT:    sw t1, 416(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    and a7, t1, t0
+; RV32IM-NEXT:    xor a2, a2, a5
+; RV32IM-NEXT:    xor a5, a6, a7
+; RV32IM-NEXT:    srli a6, s10, 3
+; RV32IM-NEXT:    slli a7, a0, 29
+; RV32IM-NEXT:    srli t0, s10, 2
+; RV32IM-NEXT:    slli t1, a0, 30
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    or a7, t1, t0
+; RV32IM-NEXT:    slli t0, a4, 2
+; RV32IM-NEXT:    slli t1, a4, 1
+; RV32IM-NEXT:    srai t0, t0, 31
+; RV32IM-NEXT:    sw t0, 412(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai t1, t1, 31
+; RV32IM-NEXT:    sw t1, 408(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a6, t0, a6
+; RV32IM-NEXT:    and a7, t1, a7
+; RV32IM-NEXT:    xor a2, a2, a5
+; RV32IM-NEXT:    xor a5, a6, a7
+; RV32IM-NEXT:    slli a0, a0, 31
+; RV32IM-NEXT:    srli a6, s10, 1
+; RV32IM-NEXT:    or a6, a0, a6
+; RV32IM-NEXT:    lw a0, 0(a1)
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    sw a4, 404(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a1, s2, 31
+; RV32IM-NEXT:    and a4, a4, a6
+; RV32IM-NEXT:    srai a1, a1, 31
+; RV32IM-NEXT:    sw a1, 296(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a4, a5, a4
+; RV32IM-NEXT:    and a1, a1, s10
+; RV32IM-NEXT:    slli a5, s10, 1
+; RV32IM-NEXT:    srli a6, a0, 31
+; RV32IM-NEXT:    xor a1, a4, a1
+; RV32IM-NEXT:    or t1, a6, a5
+; RV32IM-NEXT:    slli a4, s10, 2
+; RV32IM-NEXT:    srli a5, a0, 30
+; RV32IM-NEXT:    or a7, a5, a4
+; RV32IM-NEXT:    sw a7, 152(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 30
+; RV32IM-NEXT:    srai a5, a4, 31
+; RV32IM-NEXT:    sw a5, 292(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 29
+; RV32IM-NEXT:    srai a6, a4, 31
+; RV32IM-NEXT:    sw a6, 288(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a5, t1
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    and a4, a6, a7
+; RV32IM-NEXT:    slli a5, s10, 3
+; RV32IM-NEXT:    srli a6, a0, 29
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    or t0, a6, a5
+; RV32IM-NEXT:    sw t0, 168(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a4, a0, 28
+; RV32IM-NEXT:    slli a5, s10, 4
+; RV32IM-NEXT:    or a7, a5, a4
+; RV32IM-NEXT:    sw a7, 172(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 28
+; RV32IM-NEXT:    srai a5, a4, 31
+; RV32IM-NEXT:    sw a5, 284(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 27
+; RV32IM-NEXT:    srai a6, a4, 31
+; RV32IM-NEXT:    sw a6, 280(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a5, t0
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    and a4, a6, a7
+; RV32IM-NEXT:    slli a5, s10, 5
+; RV32IM-NEXT:    srli a6, a0, 27
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    or s3, a6, a5
+; RV32IM-NEXT:    slli a4, s10, 6
+; RV32IM-NEXT:    srli a5, a0, 26
+; RV32IM-NEXT:    or t2, a5, a4
+; RV32IM-NEXT:    sw t2, 148(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 26
+; RV32IM-NEXT:    slli a5, s10, 7
+; RV32IM-NEXT:    srli a6, a0, 25
+; RV32IM-NEXT:    or s4, a6, a5
+; RV32IM-NEXT:    srai a6, a4, 31
+; RV32IM-NEXT:    sw a6, 276(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 25
+; RV32IM-NEXT:    slli a5, s2, 24
+; RV32IM-NEXT:    srai t0, a4, 31
+; RV32IM-NEXT:    sw t0, 268(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai a7, a5, 31
+; RV32IM-NEXT:    sw a7, 272(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a6, s3
+; RV32IM-NEXT:    and a5, t0, t2
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a7, s4
+; RV32IM-NEXT:    srli a6, a0, 24
+; RV32IM-NEXT:    slli a7, s10, 8
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    or a3, a7, a6
+; RV32IM-NEXT:    slli a5, s10, 9
+; RV32IM-NEXT:    srli a6, a0, 23
+; RV32IM-NEXT:    or t0, a6, a5
+; RV32IM-NEXT:    sw t0, 164(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, s2, 23
+; RV32IM-NEXT:    srai a6, a5, 31
+; RV32IM-NEXT:    sw a6, 264(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, s2, 22
+; RV32IM-NEXT:    srai a7, a5, 31
+; RV32IM-NEXT:    sw a7, 260(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a6, a3
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a7, t0
+; RV32IM-NEXT:    slli a6, s10, 10
+; RV32IM-NEXT:    srli a7, a0, 22
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    or t0, a7, a6
+; RV32IM-NEXT:    sw t0, 144(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, s10, 11
+; RV32IM-NEXT:    srli a6, a0, 21
+; RV32IM-NEXT:    or t2, a6, a5
+; RV32IM-NEXT:    sw t2, 132(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, s2, 21
+; RV32IM-NEXT:    srai a6, a5, 31
+; RV32IM-NEXT:    sw a6, 256(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, s2, 20
+; RV32IM-NEXT:    srai a7, a5, 31
+; RV32IM-NEXT:    sw a7, 252(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a6, t0
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a7, t2
+; RV32IM-NEXT:    slli a6, s10, 12
+; RV32IM-NEXT:    srli a7, a0, 20
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    or t0, a7, a6
+; RV32IM-NEXT:    sw t0, 120(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, s10, 13
+; RV32IM-NEXT:    srli a6, a0, 19
+; RV32IM-NEXT:    or t2, a6, a5
+; RV32IM-NEXT:    sw t2, 112(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, s2, 19
+; RV32IM-NEXT:    srai a6, a5, 31
+; RV32IM-NEXT:    sw a6, 248(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, s2, 18
+; RV32IM-NEXT:    srai a7, a5, 31
+; RV32IM-NEXT:    sw a7, 244(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a6, t0
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a7, t2
+; RV32IM-NEXT:    xor a1, a2, a1
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    slli a2, s10, 14
+; RV32IM-NEXT:    srli a4, a0, 18
+; RV32IM-NEXT:    slli a5, s10, 15
+; RV32IM-NEXT:    srli a6, a0, 17
+; RV32IM-NEXT:    or a7, a4, a2
+; RV32IM-NEXT:    sw a7, 104(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a5, a6, a5
+; RV32IM-NEXT:    sw a5, 140(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a2, s2, 17
+; RV32IM-NEXT:    slli a4, s2, 16
+; RV32IM-NEXT:    srai a2, a2, 31
+; RV32IM-NEXT:    sw a2, 240(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    sw a4, 236(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a2, a2, a7
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    srli a5, a0, 16
+; RV32IM-NEXT:    slli a6, s10, 16
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    or a7, a6, a5
+; RV32IM-NEXT:    sw a7, 128(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a4, a0, 15
+; RV32IM-NEXT:    slli a5, s10, 17
+; RV32IM-NEXT:    or t0, a5, a4
+; RV32IM-NEXT:    sw t0, 116(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 15
+; RV32IM-NEXT:    srai a5, a4, 31
+; RV32IM-NEXT:    sw a5, 232(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 14
+; RV32IM-NEXT:    srai a6, a4, 31
+; RV32IM-NEXT:    sw a6, 228(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a5, a7
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    and a4, a6, t0
+; RV32IM-NEXT:    srli a5, a0, 14
+; RV32IM-NEXT:    slli a6, s10, 18
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    or t0, a6, a5
+; RV32IM-NEXT:    sw t0, 108(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a4, a0, 13
+; RV32IM-NEXT:    slli a5, s10, 19
+; RV32IM-NEXT:    or a7, a5, a4
+; RV32IM-NEXT:    sw a7, 160(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 13
+; RV32IM-NEXT:    srai a5, a4, 31
+; RV32IM-NEXT:    sw a5, 224(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 12
+; RV32IM-NEXT:    srai a6, a4, 31
+; RV32IM-NEXT:    sw a6, 220(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a5, t0
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    and a4, a6, a7
+; RV32IM-NEXT:    srli a5, a0, 12
+; RV32IM-NEXT:    slli a6, s10, 20
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    or t0, a6, a5
+; RV32IM-NEXT:    sw t0, 100(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a4, a0, 11
+; RV32IM-NEXT:    slli a5, s10, 21
+; RV32IM-NEXT:    or a7, a5, a4
+; RV32IM-NEXT:    sw a7, 156(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 11
+; RV32IM-NEXT:    srai a5, a4, 31
+; RV32IM-NEXT:    sw a5, 216(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 10
+; RV32IM-NEXT:    srai a6, a4, 31
+; RV32IM-NEXT:    sw a6, 212(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a5, t0
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    and a4, a6, a7
+; RV32IM-NEXT:    srli a5, a0, 10
+; RV32IM-NEXT:    slli a6, s10, 22
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    or a7, a6, a5
+; RV32IM-NEXT:    sw a7, 136(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a4, a0, 9
+; RV32IM-NEXT:    slli a5, s10, 23
+; RV32IM-NEXT:    or t0, a5, a4
+; RV32IM-NEXT:    sw t0, 124(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 9
+; RV32IM-NEXT:    srai a5, a4, 31
+; RV32IM-NEXT:    sw a5, 200(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 8
+; RV32IM-NEXT:    srai a6, a4, 31
+; RV32IM-NEXT:    sw a6, 196(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a5, a7
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    and a4, a6, t0
+; RV32IM-NEXT:    slli a5, s10, 24
+; RV32IM-NEXT:    srli a6, a0, 8
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    or t2, a6, a5
+; RV32IM-NEXT:    sw t2, 96(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a4, a0, 7
+; RV32IM-NEXT:    slli a5, s10, 25
+; RV32IM-NEXT:    or t3, a5, a4
+; RV32IM-NEXT:    slli a4, s2, 7
+; RV32IM-NEXT:    srli a5, a0, 6
+; RV32IM-NEXT:    slli a6, s10, 26
+; RV32IM-NEXT:    or t5, a6, a5
+; RV32IM-NEXT:    srai a7, a4, 31
+; RV32IM-NEXT:    sw a7, 188(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a4, s2, 6
+; RV32IM-NEXT:    slli a5, s2, 5
+; RV32IM-NEXT:    srai t0, a4, 31
+; RV32IM-NEXT:    sw t0, 184(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srai a6, a5, 31
+; RV32IM-NEXT:    sw a6, 204(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a7, t2
+; RV32IM-NEXT:    and a5, t0, t3
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a6, t5
+; RV32IM-NEXT:    srli a6, a0, 5
+; RV32IM-NEXT:    slli a7, s10, 27
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    or t4, a7, a6
+; RV32IM-NEXT:    slli a5, s10, 28
+; RV32IM-NEXT:    srli a6, a0, 4
+; RV32IM-NEXT:    or t2, a6, a5
+; RV32IM-NEXT:    slli a5, s2, 4
+; RV32IM-NEXT:    srai a7, a5, 31
+; RV32IM-NEXT:    sw a7, 176(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, s2, 3
+; RV32IM-NEXT:    srai a6, a5, 31
+; RV32IM-NEXT:    sw a6, 192(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a7, t4
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a6, t2
+; RV32IM-NEXT:    srli a6, a0, 3
+; RV32IM-NEXT:    slli a7, s10, 29
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    or t6, a7, a6
+; RV32IM-NEXT:    srli a5, a0, 2
+; RV32IM-NEXT:    slli a6, s10, 30
+; RV32IM-NEXT:    or s1, a6, a5
+; RV32IM-NEXT:    slli a5, s2, 2
+; RV32IM-NEXT:    srai a7, a5, 31
+; RV32IM-NEXT:    sw a7, 180(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, s2, 1
+; RV32IM-NEXT:    srai a6, a5, 31
+; RV32IM-NEXT:    sw a6, 208(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a7, t6
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a6, s1
+; RV32IM-NEXT:    srli a6, a0, 1
+; RV32IM-NEXT:    slli a7, s10, 31
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    or t0, a7, a6
+; RV32IM-NEXT:    srai a5, s2, 31
+; RV32IM-NEXT:    slli a6, s11, 31
+; RV32IM-NEXT:    and a5, a5, t0
+; RV32IM-NEXT:    srai a6, a6, 31
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a6, a0
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    slli a5, s11, 30
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli s5, a0, 1
+; RV32IM-NEXT:    and a5, a5, s5
+; RV32IM-NEXT:    slli a6, s11, 29
+; RV32IM-NEXT:    srai a6, a6, 31
+; RV32IM-NEXT:    slli s6, a0, 2
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a6, s6
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    slli a2, s11, 28
+; RV32IM-NEXT:    slli a4, s11, 27
+; RV32IM-NEXT:    srai a2, a2, 31
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    slli a7, a0, 3
+; RV32IM-NEXT:    slli s7, a0, 4
+; RV32IM-NEXT:    and a2, a2, a7
+; RV32IM-NEXT:    and a4, a4, s7
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    slli a4, s11, 26
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    slli s8, a0, 5
+; RV32IM-NEXT:    and a4, a4, s8
+; RV32IM-NEXT:    slli a5, s11, 25
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli s9, a0, 6
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    and a4, a5, s9
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    slli a4, s11, 24
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    slli ra, a0, 7
+; RV32IM-NEXT:    and a4, a4, ra
+; RV32IM-NEXT:    slli a5, s11, 23
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli a6, a0, 8
+; RV32IM-NEXT:    sw a6, 92(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    and a4, a5, a6
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    slli a4, s11, 22
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    slli a5, a0, 9
+; RV32IM-NEXT:    sw a5, 88(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    slli a5, s11, 21
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli a6, a0, 10
+; RV32IM-NEXT:    sw a6, 84(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    and a4, a5, a6
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    slli a4, s11, 20
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    slli a5, a0, 11
+; RV32IM-NEXT:    sw a5, 80(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    slli a5, s11, 19
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli a6, a0, 12
+; RV32IM-NEXT:    sw a6, 76(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    and a4, a5, a6
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    slli a4, s11, 18
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    slli a5, a0, 13
+; RV32IM-NEXT:    sw a5, 72(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    slli a5, s11, 17
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli a6, a0, 14
+; RV32IM-NEXT:    sw a6, 64(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    and a4, a5, a6
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    slli a4, s11, 16
+; RV32IM-NEXT:    srai a4, a4, 31
+; RV32IM-NEXT:    slli a5, a0, 15
+; RV32IM-NEXT:    sw a5, 60(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    slli a5, s11, 15
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli a6, a0, 16
+; RV32IM-NEXT:    sw a6, 304(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a5, a6
+; RV32IM-NEXT:    slli a6, s11, 14
+; RV32IM-NEXT:    srai a6, a6, 31
+; RV32IM-NEXT:    slli s0, a0, 17
+; RV32IM-NEXT:    sw s0, 56(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a6, s0
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    slli a5, s11, 13
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli a6, a0, 18
+; RV32IM-NEXT:    sw a6, 48(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a5, a6
+; RV32IM-NEXT:    slli a6, s11, 12
+; RV32IM-NEXT:    srai a6, a6, 31
+; RV32IM-NEXT:    slli s0, a0, 19
+; RV32IM-NEXT:    sw s0, 40(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a6, s0
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    slli a5, s11, 11
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli a6, a0, 20
+; RV32IM-NEXT:    sw a6, 32(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a5, a6
+; RV32IM-NEXT:    slli a6, s11, 10
+; RV32IM-NEXT:    srai a6, a6, 31
+; RV32IM-NEXT:    slli s0, a0, 21
+; RV32IM-NEXT:    sw s0, 16(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a6, s0
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    slli a5, s11, 9
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli a6, a0, 22
+; RV32IM-NEXT:    sw a6, 308(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a5, a6
+; RV32IM-NEXT:    slli a6, s11, 8
+; RV32IM-NEXT:    srai a6, a6, 31
+; RV32IM-NEXT:    slli s0, a0, 23
+; RV32IM-NEXT:    sw s0, 312(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a6, s0
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    slli a5, s11, 7
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli a6, a0, 24
+; RV32IM-NEXT:    sw a6, 36(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a5, a6
+; RV32IM-NEXT:    slli a6, s11, 6
+; RV32IM-NEXT:    srai a6, a6, 31
+; RV32IM-NEXT:    slli s0, a0, 25
+; RV32IM-NEXT:    sw s0, 24(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, a6, s0
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    slli a5, s11, 5
+; RV32IM-NEXT:    srai a5, a5, 31
+; RV32IM-NEXT:    slli a6, a0, 26
+; RV32IM-NEXT:    sw a6, 20(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a5, a6
+; RV32IM-NEXT:    slli a6, s11, 4
+; RV32IM-NEXT:    srai s0, a6, 31
+; RV32IM-NEXT:    slli a6, a0, 27
+; RV32IM-NEXT:    sw a6, 28(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    and a5, s0, a6
+; RV32IM-NEXT:    xor s0, a1, a2
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    slli a1, s11, 3
+; RV32IM-NEXT:    slli a2, s11, 2
+; RV32IM-NEXT:    srai a1, a1, 31
+; RV32IM-NEXT:    srai a2, a2, 31
+; RV32IM-NEXT:    slli a6, a0, 28
+; RV32IM-NEXT:    sw a6, 12(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a5, a0, 29
+; RV32IM-NEXT:    sw a5, 44(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a1, a1, a6
+; RV32IM-NEXT:    and a2, a2, a5
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    slli a2, s11, 1
+; RV32IM-NEXT:    srai a5, a2, 31
+; RV32IM-NEXT:    slli a2, a0, 30
+; RV32IM-NEXT:    sw a2, 52(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a2, a5, a2
+; RV32IM-NEXT:    slli a5, a0, 31
+; RV32IM-NEXT:    sw a5, 68(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    and a2, s11, a5
+; RV32IM-NEXT:    xor a4, s0, a4
+; RV32IM-NEXT:    sw a4, 8(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    sw a1, 4(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw a1, 368(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, t1
+; RV32IM-NEXT:    lw a2, 364(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a2, a2, s10
+; RV32IM-NEXT:    lw a4, 152(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 360(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a5, a4
+; RV32IM-NEXT:    lw t1, 168(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 356(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a5, t1
+; RV32IM-NEXT:    xor a1, a2, a1
+; RV32IM-NEXT:    xor a2, a4, s10
+; RV32IM-NEXT:    lw a4, 172(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 352(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a5, a4
+; RV32IM-NEXT:    lw a5, 348(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a5, s3
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw a5, 148(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a6, 344(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a6, a5
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a2, a4, s10
+; RV32IM-NEXT:    lw a4, 340(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s4
+; RV32IM-NEXT:    lw a5, 336(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a5, a3
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s0, 164(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 332(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a5, s0
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw a5, 144(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a6, 328(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a6, a5
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a2, a4, s10
+; RV32IM-NEXT:    lw a4, 132(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 324(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a5, a4
+; RV32IM-NEXT:    lw a5, 120(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a6, 320(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a6, a5
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw a5, 112(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a6, 316(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a6, a5
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw a5, 104(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a6, 400(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a6, a5
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw a5, 140(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a6, 396(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a6, a5
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a2, a4, s10
+; RV32IM-NEXT:    lw a4, 392(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 128(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    lw a5, 388(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a6, 116(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a5, a6
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw a5, 384(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a6, 108(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a5, a6
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 160(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 380(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a5, s10
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw a5, 376(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a6, 100(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a5, a6
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 156(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 372(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a5, s10
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a2, a4, s10
+; RV32IM-NEXT:    lw a4, 440(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 136(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    lw s10, 436(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 124(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, a5
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 432(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 96(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, a5
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 428(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, t3
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 424(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, t5
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 420(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, t4
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 416(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, t2
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a2, a4, s10
+; RV32IM-NEXT:    lw a4, 412(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t6
+; RV32IM-NEXT:    lw s10, 408(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, s1
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 404(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, t0
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 296(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, a0
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 292(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mv a5, s5
+; RV32IM-NEXT:    and s10, s10, s5
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 288(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mv t0, s6
+; RV32IM-NEXT:    and s10, s10, s6
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 284(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, a7
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 280(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mv t2, s7
+; RV32IM-NEXT:    and s10, s10, s7
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a2, a4, s10
+; RV32IM-NEXT:    lw a4, 276(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mv a6, s8
+; RV32IM-NEXT:    and a4, a4, s8
+; RV32IM-NEXT:    lw s10, 268(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mv t1, s9
+; RV32IM-NEXT:    and s10, s10, s9
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 272(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mv s0, ra
+; RV32IM-NEXT:    and s10, s10, ra
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 264(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw t3, 92(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, t3
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 260(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw t4, 88(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, t4
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 256(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw t5, 84(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, t5
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 252(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s11, 80(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, s11
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 248(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s4, 76(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, s4
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 244(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s3, 72(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, s3
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a2, a4, s10
+; RV32IM-NEXT:    lw a4, 240(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw t6, 64(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t6
+; RV32IM-NEXT:    lw s10, 236(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s1, 60(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, s1
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 232(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a3, 304(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, a3
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 228(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s5, 56(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, s5
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 224(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s6, 48(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, s6
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 220(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s7, 40(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, s7
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 216(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s8, 32(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, s8
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 212(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s9, 16(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, s9
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 200(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw ra, 308(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, ra
+; RV32IM-NEXT:    xor a4, a4, s10
+; RV32IM-NEXT:    lw s10, 196(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw ra, 312(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, s10, ra
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a2, a4, s10
+; RV32IM-NEXT:    lw a3, 8(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a4, 4(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    xor a3, a3, a4
+; RV32IM-NEXT:    sw a3, 296(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor s10, a1, a2
+; RV32IM-NEXT:    lw a1, 368(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, a5
+; RV32IM-NEXT:    lw a2, 364(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a0, a2, a0
+; RV32IM-NEXT:    lw a2, 188(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw ra, 36(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a2, a2, ra
+; RV32IM-NEXT:    lw a4, 184(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a3, 24(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a3
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    lw a1, 360(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, t0
+; RV32IM-NEXT:    lw a4, 356(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a7
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 204(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a7, 20(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a7
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    lw a1, 352(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, t2
+; RV32IM-NEXT:    lw a4, 348(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a6
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 344(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t1
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 176(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a6, 28(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a6
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    lw a1, 340(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, s0
+; RV32IM-NEXT:    lw a4, 336(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t3
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 332(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t4
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 328(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t5
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 192(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw t1, 12(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t1
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    lw a1, 324(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, s11
+; RV32IM-NEXT:    lw a4, 320(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s4
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 316(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s3
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 400(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t6
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 396(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s1
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 180(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw t0, 44(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t0
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    lw a1, 392(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a4, 304(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, a4
+; RV32IM-NEXT:    lw a4, 388(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s5
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 384(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s6
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 380(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s7
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 376(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s8
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 372(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s9
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 208(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw t2, 52(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t2
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    lw a1, 440(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a4, 308(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, a4
+; RV32IM-NEXT:    lw a4, 436(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a5, 312(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 432(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, ra
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 428(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a3
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 424(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a7
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 420(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, a6
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    lw a4, 416(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, t1
+; RV32IM-NEXT:    lw a5, 412(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a7, a5, t0
+; RV32IM-NEXT:    lw a5, 408(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a6, a5, t2
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    xor a4, a7, a6
+; RV32IM-NEXT:    lw a6, 68(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, s2, a6
+; RV32IM-NEXT:    lw a5, 404(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a5, a5, a6
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    xor a2, a2, a3
+; RV32IM-NEXT:    xor a0, a0, a4
+; RV32IM-NEXT:    xor a1, s10, a2
+; RV32IM-NEXT:    lw a2, 300(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    sw a0, 0(a2)
+; RV32IM-NEXT:    sw a1, 4(a2)
+; RV32IM-NEXT:    lw a0, 296(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    sw a0, 8(a2)
+; RV32IM-NEXT:    lw ra, 492(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s0, 488(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s1, 484(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s2, 480(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s3, 476(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s4, 472(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s5, 468(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s6, 464(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s7, 460(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s8, 456(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s9, 452(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s10, 448(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s11, 444(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    .cfi_restore ra
+; RV32IM-NEXT:    .cfi_restore s0
+; RV32IM-NEXT:    .cfi_restore s1
+; RV32IM-NEXT:    .cfi_restore s2
+; RV32IM-NEXT:    .cfi_restore s3
+; RV32IM-NEXT:    .cfi_restore s4
+; RV32IM-NEXT:    .cfi_restore s5
+; RV32IM-NEXT:    .cfi_restore s6
+; RV32IM-NEXT:    .cfi_restore s7
+; RV32IM-NEXT:    .cfi_restore s8
+; RV32IM-NEXT:    .cfi_restore s9
+; RV32IM-NEXT:    .cfi_restore s10
+; RV32IM-NEXT:    .cfi_restore s11
+; RV32IM-NEXT:    addi sp, sp, 496
+; RV32IM-NEXT:    .cfi_def_cfa_offset 0
+; RV32IM-NEXT:    ret
+;
+; RV64IM-LABEL: clmul_i96:
+; RV64IM:       # %bb.0:
+; RV64IM-NEXT:    addi sp, sp, -688
+; RV64IM-NEXT:    .cfi_def_cfa_offset 688
+; RV64IM-NEXT:    sd ra, 680(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s0, 672(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s1, 664(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s2, 656(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s3, 648(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s4, 640(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s5, 632(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s6, 624(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s7, 616(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s8, 608(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s9, 600(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s10, 592(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s11, 584(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    .cfi_offset ra, -8
+; RV64IM-NEXT:    .cfi_offset s0, -16
+; RV64IM-NEXT:    .cfi_offset s1, -24
+; RV64IM-NEXT:    .cfi_offset s2, -32
+; RV64IM-NEXT:    .cfi_offset s3, -40
+; RV64IM-NEXT:    .cfi_offset s4, -48
+; RV64IM-NEXT:    .cfi_offset s5, -56
+; RV64IM-NEXT:    .cfi_offset s6, -64
+; RV64IM-NEXT:    .cfi_offset s7, -72
+; RV64IM-NEXT:    .cfi_offset s8, -80
+; RV64IM-NEXT:    .cfi_offset s9, -88
+; RV64IM-NEXT:    .cfi_offset s10, -96
+; RV64IM-NEXT:    .cfi_offset s11, -104
+; RV64IM-NEXT:    mv t0, a3
+; RV64IM-NEXT:    slli a3, a0, 1
+; RV64IM-NEXT:    sd a3, 568(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a4, a2, 62
+; RV64IM-NEXT:    slli a5, a2, 63
+; RV64IM-NEXT:    srai t3, a4, 63
+; RV64IM-NEXT:    srai s2, a5, 63
+; RV64IM-NEXT:    and a4, t3, a3
+; RV64IM-NEXT:    and a5, s2, a0
+; RV64IM-NEXT:    xor a4, a5, a4
+; RV64IM-NEXT:    slli a6, a0, 2
+; RV64IM-NEXT:    sd a6, 536(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a5, a2, 61
+; RV64IM-NEXT:    srai a7, a5, 63
+; RV64IM-NEXT:    sd a7, 152(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a5, a2, 60
+; RV64IM-NEXT:    slli a3, a0, 3
+; RV64IM-NEXT:    sd a3, 576(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    srai t1, a5, 63
+; RV64IM-NEXT:    sd t1, 136(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a5, a7, a6
+; RV64IM-NEXT:    and a6, t1, a3
+; RV64IM-NEXT:    xor a5, a5, a6
+; RV64IM-NEXT:    slli a6, a2, 59
+; RV64IM-NEXT:    slli a3, a0, 4
+; RV64IM-NEXT:    sd a3, 552(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    srai a6, a6, 63
+; RV64IM-NEXT:    sd a6, 248(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a6, a6, a3
+; RV64IM-NEXT:    slli a7, a2, 58
+; RV64IM-NEXT:    slli a3, a0, 5
+; RV64IM-NEXT:    sd a3, 560(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    srai a7, a7, 63
+; RV64IM-NEXT:    sd a7, 304(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a7, a7, a3
+; RV64IM-NEXT:    slli t1, a2, 57
+; RV64IM-NEXT:    slli a3, a0, 6
+; RV64IM-NEXT:    sd a3, 544(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    srai t1, t1, 63
+; RV64IM-NEXT:    sd t1, 128(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    xor a6, a6, a7
+; RV64IM-NEXT:    and a7, t1, a3
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    xor a5, a6, a7
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    slli a3, a0, 7
+; RV64IM-NEXT:    sd a3, 512(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a5, a2, 56
+; RV64IM-NEXT:    srai t1, a5, 63
+; RV64IM-NEXT:    sd t1, 64(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a5, a2, 55
+; RV64IM-NEXT:    slli a6, a0, 8
+; RV64IM-NEXT:    sd a6, 520(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    srai a7, a5, 63
+; RV64IM-NEXT:    sd a7, 312(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a5, t1, a3
+; RV64IM-NEXT:    and a6, a7, a6
+; RV64IM-NEXT:    xor a5, a5, a6
+; RV64IM-NEXT:    slli a6, a2, 54
+; RV64IM-NEXT:    slli a3, a0, 9
+; RV64IM-NEXT:    sd a3, 528(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    srai s3, a6, 63
+; RV64IM-NEXT:    and a6, s3, a3
+; RV64IM-NEXT:    slli a7, a2, 53
+; RV64IM-NEXT:    slli a3, a0, 10
+; RV64IM-NEXT:    sd a3, 504(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    srai s4, a7, 63
+; RV64IM-NEXT:    xor a5, a5, a6
+; RV64IM-NEXT:    and a6, s4, a3
+; RV64IM-NEXT:    xor a5, a5, a6
+; RV64IM-NEXT:    slli a6, a2, 52
+; RV64IM-NEXT:    srai a7, a6, 63
+; RV64IM-NEXT:    sd a7, 160(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a6, a2, 51
+; RV64IM-NEXT:    srai t2, a6, 63
+; RV64IM-NEXT:    sd t2, 88(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a6, a2, 50
+; RV64IM-NEXT:    srai t1, a6, 63
+; RV64IM-NEXT:    sd t1, 320(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 11
+; RV64IM-NEXT:    sd a3, 472(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a6, a7, a3
+; RV64IM-NEXT:    slli a3, a0, 12
+; RV64IM-NEXT:    sd a3, 488(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a7, t2, a3
+; RV64IM-NEXT:    slli a3, a0, 13
+; RV64IM-NEXT:    sd a3, 496(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    xor a6, a6, a7
+; RV64IM-NEXT:    and a7, t1, a3
+; RV64IM-NEXT:    xor a6, a6, a7
+; RV64IM-NEXT:    slli a7, a2, 49
+; RV64IM-NEXT:    srai t2, a7, 63
+; RV64IM-NEXT:    sd t2, 56(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a7, a2, 48
+; RV64IM-NEXT:    srai t1, a7, 63
+; RV64IM-NEXT:    sd t1, 184(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 14
+; RV64IM-NEXT:    sd a3, 464(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a7, t2, a3
+; RV64IM-NEXT:    slli a3, a0, 15
+; RV64IM-NEXT:    sd a3, 480(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    xor a6, a6, a7
+; RV64IM-NEXT:    and a7, t1, a3
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    xor a5, a6, a7
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    slli a5, a2, 47
+; RV64IM-NEXT:    slli a6, a2, 46
+; RV64IM-NEXT:    srai a7, a5, 63
+; RV64IM-NEXT:    sd a7, 264(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    srai a6, a6, 63
+; RV64IM-NEXT:    sd a6, 256(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a5, a0, 16
+; RV64IM-NEXT:    sd a5, 448(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 17
+; RV64IM-NEXT:    sd a3, 456(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a5, a7, a5
+; RV64IM-NEXT:    and a6, a6, a3
+; RV64IM-NEXT:    xor a5, a5, a6
+; RV64IM-NEXT:    slli a6, a2, 45
+; RV64IM-NEXT:    srai a7, a6, 63
+; RV64IM-NEXT:    sd a7, 232(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a6, a2, 44
+; RV64IM-NEXT:    srai t1, a6, 63
+; RV64IM-NEXT:    sd t1, 224(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 18
+; RV64IM-NEXT:    sd a3, 432(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a6, a7, a3
+; RV64IM-NEXT:    slli a3, a0, 19
+; RV64IM-NEXT:    sd a3, 440(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    xor a5, a5, a6
+; RV64IM-NEXT:    and a6, t1, a3
+; RV64IM-NEXT:    xor a5, a5, a6
+; RV64IM-NEXT:    slli a6, a2, 43
+; RV64IM-NEXT:    srai a7, a6, 63
+; RV64IM-NEXT:    sd a7, 192(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a6, a2, 42
+; RV64IM-NEXT:    srai t1, a6, 63
+; RV64IM-NEXT:    sd t1, 176(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 20
+; RV64IM-NEXT:    sd a3, 416(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a6, a7, a3
+; RV64IM-NEXT:    slli a3, a0, 21
+; RV64IM-NEXT:    sd a3, 424(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    xor a5, a5, a6
+; RV64IM-NEXT:    and a6, t1, a3
+; RV64IM-NEXT:    xor a5, a5, a6
+; RV64IM-NEXT:    slli a6, a2, 41
+; RV64IM-NEXT:    srai a7, a6, 63
+; RV64IM-NEXT:    sd a7, 0(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a6, a2, 40
+; RV64IM-NEXT:    srai t1, a6, 63
+; RV64IM-NEXT:    sd t1, 80(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a6, a2, 39
+; RV64IM-NEXT:    srai t2, a6, 63
+; RV64IM-NEXT:    sd t2, 72(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 22
+; RV64IM-NEXT:    sd a3, 384(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a6, a7, a3
+; RV64IM-NEXT:    slli a3, a0, 23
+; RV64IM-NEXT:    sd a3, 392(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a7, t1, a3
+; RV64IM-NEXT:    slli a3, a0, 24
+; RV64IM-NEXT:    sd a3, 408(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    xor a6, a6, a7
+; RV64IM-NEXT:    and a7, t2, a3
+; RV64IM-NEXT:    xor a6, a6, a7
+; RV64IM-NEXT:    slli a7, a2, 38
+; RV64IM-NEXT:    srai t1, a7, 63
+; RV64IM-NEXT:    sd t1, 48(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a7, a2, 37
+; RV64IM-NEXT:    srai t2, a7, 63
+; RV64IM-NEXT:    sd t2, 40(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 25
+; RV64IM-NEXT:    sd a3, 368(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a7, t1, a3
+; RV64IM-NEXT:    slli a3, a0, 26
+; RV64IM-NEXT:    sd a3, 400(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    xor a6, a6, a7
+; RV64IM-NEXT:    and a7, t2, a3
+; RV64IM-NEXT:    xor a6, a6, a7
+; RV64IM-NEXT:    slli a7, a2, 36
+; RV64IM-NEXT:    srai t1, a7, 63
+; RV64IM-NEXT:    sd t1, 24(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a7, a2, 35
+; RV64IM-NEXT:    srai t2, a7, 63
+; RV64IM-NEXT:    sd t2, 16(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 27
+; RV64IM-NEXT:    sd a3, 352(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a7, t1, a3
+; RV64IM-NEXT:    slli a3, a0, 28
+; RV64IM-NEXT:    sd a3, 376(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    xor a6, a6, a7
+; RV64IM-NEXT:    and a7, t2, a3
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    xor a5, a6, a7
+; RV64IM-NEXT:    xor t1, a4, a5
+; RV64IM-NEXT:    slli a4, a2, 34
+; RV64IM-NEXT:    srai a5, a4, 63
+; RV64IM-NEXT:    sd a5, 8(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a4, a2, 33
+; RV64IM-NEXT:    srai a6, a4, 63
+; RV64IM-NEXT:    sd a6, 144(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 29
+; RV64IM-NEXT:    sd a3, 344(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a4, a5, a3
+; RV64IM-NEXT:    slli a3, a0, 30
+; RV64IM-NEXT:    sd a3, 360(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a5, a6, a3
+; RV64IM-NEXT:    slli a6, a2, 31
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    srai a5, a6, 63
+; RV64IM-NEXT:    sd a5, 32(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 31
+; RV64IM-NEXT:    sd a3, 336(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sraiw t2, a2, 31
+; RV64IM-NEXT:    and a6, t2, a3
+; RV64IM-NEXT:    slli a7, a0, 32
+; RV64IM-NEXT:    xor a4, a4, a6
+; RV64IM-NEXT:    and a6, a5, a7
+; RV64IM-NEXT:    xor a4, a4, a6
+; RV64IM-NEXT:    slli a6, a2, 30
+; RV64IM-NEXT:    srai t4, a6, 63
+; RV64IM-NEXT:    slli a6, a0, 33
+; RV64IM-NEXT:    and a6, t4, a6
+; RV64IM-NEXT:    slli a7, a2, 29
+; RV64IM-NEXT:    srai s9, a7, 63
+; RV64IM-NEXT:    slli a7, a0, 34
+; RV64IM-NEXT:    xor a4, a4, a6
+; RV64IM-NEXT:    and a6, s9, a7
+; RV64IM-NEXT:    xor a4, a4, a6
+; RV64IM-NEXT:    slli a6, a2, 28
+; RV64IM-NEXT:    srai s6, a6, 63
+; RV64IM-NEXT:    slli a7, a0, 35
+; RV64IM-NEXT:    and t5, s6, a7
+; RV64IM-NEXT:    slli a7, a2, 27
+; RV64IM-NEXT:    srai s7, a7, 63
+; RV64IM-NEXT:    slli t6, a0, 36
+; RV64IM-NEXT:    xor a4, a4, t5
+; RV64IM-NEXT:    and t5, s7, t6
+; RV64IM-NEXT:    xor a4, a4, t5
+; RV64IM-NEXT:    slli t5, a2, 26
+; RV64IM-NEXT:    srai a3, t5, 63
+; RV64IM-NEXT:    sd a3, 240(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli t5, a0, 37
+; RV64IM-NEXT:    and t6, a3, t5
+; RV64IM-NEXT:    slli t5, a2, 25
+; RV64IM-NEXT:    srai s8, t5, 63
+; RV64IM-NEXT:    slli s0, a0, 38
+; RV64IM-NEXT:    and s0, s8, s0
+; RV64IM-NEXT:    slli s1, a2, 24
+; RV64IM-NEXT:    srai a3, s1, 63
+; RV64IM-NEXT:    sd a3, 216(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli s1, a0, 39
+; RV64IM-NEXT:    xor t6, t6, s0
+; RV64IM-NEXT:    and s0, a3, s1
+; RV64IM-NEXT:    xor s0, t6, s0
+; RV64IM-NEXT:    slli t6, a2, 23
+; RV64IM-NEXT:    srai t6, t6, 63
+; RV64IM-NEXT:    slli s1, a0, 40
+; RV64IM-NEXT:    and s1, t6, s1
+; RV64IM-NEXT:    slli s5, a2, 22
+; RV64IM-NEXT:    srai a3, s5, 63
+; RV64IM-NEXT:    sd a3, 208(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli s5, a0, 41
+; RV64IM-NEXT:    xor s0, s0, s1
+; RV64IM-NEXT:    and s1, a3, s5
+; RV64IM-NEXT:    xor s1, s0, s1
+; RV64IM-NEXT:    slli s0, a2, 21
+; RV64IM-NEXT:    srai s0, s0, 63
+; RV64IM-NEXT:    slli s5, a0, 42
+; RV64IM-NEXT:    and s5, s0, s5
+; RV64IM-NEXT:    slli s10, a2, 20
+; RV64IM-NEXT:    srai a3, s10, 63
+; RV64IM-NEXT:    sd a3, 200(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli s10, a0, 43
+; RV64IM-NEXT:    xor s1, s1, s5
+; RV64IM-NEXT:    and s5, a3, s10
+; RV64IM-NEXT:    xor s10, s1, s5
+; RV64IM-NEXT:    slli s1, a2, 19
+; RV64IM-NEXT:    srai s1, s1, 63
+; RV64IM-NEXT:    slli s5, a0, 44
+; RV64IM-NEXT:    and s11, s1, s5
+; RV64IM-NEXT:    slli s5, a2, 18
+; RV64IM-NEXT:    srai s5, s5, 63
+; RV64IM-NEXT:    slli ra, a0, 45
+; RV64IM-NEXT:    xor s10, s10, s11
+; RV64IM-NEXT:    and s11, s5, ra
+; RV64IM-NEXT:    xor t1, t1, a4
+; RV64IM-NEXT:    xor a6, s10, s11
+; RV64IM-NEXT:    slli s10, a2, 17
+; RV64IM-NEXT:    slli s11, a2, 16
+; RV64IM-NEXT:    srai a3, s10, 63
+; RV64IM-NEXT:    sd a3, 120(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    srai a4, s11, 63
+; RV64IM-NEXT:    sd a4, 112(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli s10, a0, 46
+; RV64IM-NEXT:    slli s11, a0, 47
+; RV64IM-NEXT:    and s10, a3, s10
+; RV64IM-NEXT:    and s11, a4, s11
+; RV64IM-NEXT:    xor s11, s10, s11
+; RV64IM-NEXT:    slli s10, a2, 15
+; RV64IM-NEXT:    srai a3, s10, 63
+; RV64IM-NEXT:    sd a3, 296(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli s10, a0, 48
+; RV64IM-NEXT:    and ra, a3, s10
+; RV64IM-NEXT:    slli s10, a2, 14
+; RV64IM-NEXT:    srai a4, s10, 63
+; RV64IM-NEXT:    sd a4, 104(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 49
+; RV64IM-NEXT:    xor s11, s11, ra
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a4, s11, a3
+; RV64IM-NEXT:    slli s11, a2, 13
+; RV64IM-NEXT:    srai a3, s11, 63
+; RV64IM-NEXT:    sd a3, 288(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli s11, a0, 50
+; RV64IM-NEXT:    and ra, a3, s11
+; RV64IM-NEXT:    slli s11, a2, 12
+; RV64IM-NEXT:    srai s11, s11, 63
+; RV64IM-NEXT:    slli a3, a0, 51
+; RV64IM-NEXT:    xor a4, a4, ra
+; RV64IM-NEXT:    and a3, s11, a3
+; RV64IM-NEXT:    xor a5, a4, a3
+; RV64IM-NEXT:    slli a4, a2, 11
+; RV64IM-NEXT:    srai a3, a4, 63
+; RV64IM-NEXT:    sd a3, 280(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a4, a0, 52
+; RV64IM-NEXT:    and a4, a3, a4
+; RV64IM-NEXT:    slli ra, a2, 10
+; RV64IM-NEXT:    srai ra, ra, 63
+; RV64IM-NEXT:    slli a3, a0, 53
+; RV64IM-NEXT:    xor a4, a5, a4
+; RV64IM-NEXT:    and a3, ra, a3
+; RV64IM-NEXT:    xor a3, a4, a3
+; RV64IM-NEXT:    slli a4, a2, 9
+; RV64IM-NEXT:    srai a5, a4, 63
+; RV64IM-NEXT:    sd a5, 272(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a4, a0, 54
+; RV64IM-NEXT:    and a4, a5, a4
+; RV64IM-NEXT:    slli a5, a2, 8
+; RV64IM-NEXT:    srai a7, a5, 63
+; RV64IM-NEXT:    sd a7, 96(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a5, a0, 55
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    and a4, a7, a5
+; RV64IM-NEXT:    xor a5, t1, a6
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    xor a3, a5, a3
+; RV64IM-NEXT:    sd a3, 328(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a2, 7
+; RV64IM-NEXT:    slli a4, a2, 6
+; RV64IM-NEXT:    srai t5, a3, 63
+; RV64IM-NEXT:    srai a5, a4, 63
+; RV64IM-NEXT:    sd a5, 168(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    slli a3, a0, 56
+; RV64IM-NEXT:    slli a4, a0, 57
+; RV64IM-NEXT:    and a3, t5, a3
+; RV64IM-NEXT:    and a4, a5, a4
+; RV64IM-NEXT:    srli a5, a0, 63
+; RV64IM-NEXT:    slli a6, a1, 1
+; RV64IM-NEXT:    xor a4, a3, a4
+; RV64IM-NEXT:    or a3, a6, a5
+; RV64IM-NEXT:    and a5, t3, a3
+; RV64IM-NEXT:    slli a3, a2, 5
+; RV64IM-NEXT:    srai t1, a3, 63
+; RV64IM-NEXT:    slli a6, a0, 58
+; RV64IM-NEXT:    and a6, t1, a6
+; RV64IM-NEXT:    and t3, s2, a1
+; RV64IM-NEXT:    xor a7, a4, a6
+; RV64IM-NEXT:    xor a5, t3, a5
+; RV64IM-NEXT:    srli a6, a0, 62
+; RV64IM-NEXT:    slli t3, a1, 2
+; RV64IM-NEXT:    srli s2, a0, 61
+; RV64IM-NEXT:    slli a3, a1, 3
+; RV64IM-NEXT:    or a6, t3, a6
+; RV64IM-NEXT:    or a3, a3, s2
+; RV64IM-NEXT:    ld a4, 152(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a6, a4, a6
+; RV64IM-NEXT:    ld a4, 136(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    srli t3, a0, 60
+; RV64IM-NEXT:    slli s2, a1, 4
+; RV64IM-NEXT:    xor a4, a6, a3
+; RV64IM-NEXT:    or a6, s2, t3
+; RV64IM-NEXT:    srli t3, a0, 59
+; RV64IM-NEXT:    slli s2, a1, 5
+; RV64IM-NEXT:    ld a3, 248(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a6, a3, a6
+; RV64IM-NEXT:    or t3, s2, t3
+; RV64IM-NEXT:    srli s2, a0, 58
+; RV64IM-NEXT:    slli a3, a1, 6
+; RV64IM-NEXT:    ld s10, 304(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and t3, s10, t3
+; RV64IM-NEXT:    or a3, a3, s2
+; RV64IM-NEXT:    xor a6, a6, t3
+; RV64IM-NEXT:    ld t3, 128(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, t3, a3
+; RV64IM-NEXT:    xor a4, a5, a4
+; RV64IM-NEXT:    xor a3, a6, a3
+; RV64IM-NEXT:    srli a5, a0, 57
+; RV64IM-NEXT:    slli a6, a1, 7
+; RV64IM-NEXT:    xor a3, a4, a3
+; RV64IM-NEXT:    or a4, a6, a5
+; RV64IM-NEXT:    srli a5, a0, 56
+; RV64IM-NEXT:    slli a6, a1, 8
+; RV64IM-NEXT:    ld t3, 64(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, t3, a4
+; RV64IM-NEXT:    or a5, a6, a5
+; RV64IM-NEXT:    srli a6, a0, 55
+; RV64IM-NEXT:    slli t3, a1, 9
+; RV64IM-NEXT:    ld s2, 312(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a5, s2, a5
+; RV64IM-NEXT:    or a6, t3, a6
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    and a5, s3, a6
+; RV64IM-NEXT:    srli a6, a0, 54
+; RV64IM-NEXT:    slli t3, a1, 10
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    or a5, t3, a6
+; RV64IM-NEXT:    and a5, s4, a5
+; RV64IM-NEXT:    slli a6, a2, 4
+; RV64IM-NEXT:    srai t3, a6, 63
+; RV64IM-NEXT:    slli a6, a0, 59
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    and a5, t3, a6
+; RV64IM-NEXT:    xor a5, a7, a5
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    srli a4, a0, 53
+; RV64IM-NEXT:    slli a6, a1, 11
+; RV64IM-NEXT:    srli a7, a0, 52
+; RV64IM-NEXT:    slli s2, a1, 12
+; RV64IM-NEXT:    or a4, a6, a4
+; RV64IM-NEXT:    or a6, s2, a7
+; RV64IM-NEXT:    ld a7, 160(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, a7, a4
+; RV64IM-NEXT:    ld a7, 88(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a6, a7, a6
+; RV64IM-NEXT:    srli a7, a0, 51
+; RV64IM-NEXT:    slli s2, a1, 13
+; RV64IM-NEXT:    xor a4, a4, a6
+; RV64IM-NEXT:    or a6, s2, a7
+; RV64IM-NEXT:    srli a7, a0, 50
+; RV64IM-NEXT:    slli s2, a1, 14
+; RV64IM-NEXT:    ld s3, 320(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a6, s3, a6
+; RV64IM-NEXT:    or a7, s2, a7
+; RV64IM-NEXT:    xor a4, a4, a6
+; RV64IM-NEXT:    ld a6, 56(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a6, a6, a7
+; RV64IM-NEXT:    srli a7, a0, 49
+; RV64IM-NEXT:    slli s2, a1, 15
+; RV64IM-NEXT:    xor a4, a4, a6
+; RV64IM-NEXT:    or a6, s2, a7
+; RV64IM-NEXT:    ld a7, 184(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a6, a7, a6
+; RV64IM-NEXT:    slli a7, a2, 3
+; RV64IM-NEXT:    srai s2, a7, 63
+; RV64IM-NEXT:    slli a7, a0, 60
+; RV64IM-NEXT:    xor a4, a4, a6
+; RV64IM-NEXT:    and a6, s2, a7
+; RV64IM-NEXT:    xor s3, a5, a6
+; RV64IM-NEXT:    xor s4, a3, a4
+; RV64IM-NEXT:    srli a3, a0, 48
+; RV64IM-NEXT:    slli a4, a1, 16
+; RV64IM-NEXT:    srli a5, a0, 47
+; RV64IM-NEXT:    slli a6, a1, 17
+; RV64IM-NEXT:    or a3, a4, a3
+; RV64IM-NEXT:    or a4, a6, a5
+; RV64IM-NEXT:    ld a5, 264(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a5, a3
+; RV64IM-NEXT:    ld a5, 256(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, a5, a4
+; RV64IM-NEXT:    srli a5, a0, 46
+; RV64IM-NEXT:    slli a6, a1, 18
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    or a4, a6, a5
+; RV64IM-NEXT:    srli a5, a0, 45
+; RV64IM-NEXT:    slli a6, a1, 19
+; RV64IM-NEXT:    ld a7, 232(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, a7, a4
+; RV64IM-NEXT:    or a5, a6, a5
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    ld a4, 224(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, a4, a5
+; RV64IM-NEXT:    srli a5, a0, 44
+; RV64IM-NEXT:    slli a6, a1, 20
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    or a4, a6, a5
+; RV64IM-NEXT:    srli a5, a0, 43
+; RV64IM-NEXT:    slli a6, a1, 21
+; RV64IM-NEXT:    ld a7, 192(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, a7, a4
+; RV64IM-NEXT:    or a5, a6, a5
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    ld a4, 176(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, a4, a5
+; RV64IM-NEXT:    srli a5, a0, 42
+; RV64IM-NEXT:    slli a6, a1, 22
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    or a4, a6, a5
+; RV64IM-NEXT:    srli a5, a0, 41
+; RV64IM-NEXT:    slli a6, a1, 23
+; RV64IM-NEXT:    ld a7, 0(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, a7, a4
+; RV64IM-NEXT:    or a5, a6, a5
+; RV64IM-NEXT:    srli a6, a0, 40
+; RV64IM-NEXT:    slli a7, a1, 24
+; RV64IM-NEXT:    ld s10, 80(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a5, s10, a5
+; RV64IM-NEXT:    or a6, a7, a6
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    ld a5, 72(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a5, a5, a6
+; RV64IM-NEXT:    srli a6, a0, 39
+; RV64IM-NEXT:    slli a7, a1, 25
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    or a5, a7, a6
+; RV64IM-NEXT:    srli a6, a0, 38
+; RV64IM-NEXT:    slli a7, a1, 26
+; RV64IM-NEXT:    ld s10, 48(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a5, s10, a5
+; RV64IM-NEXT:    or a6, a7, a6
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    ld a5, 40(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a5, a5, a6
+; RV64IM-NEXT:    srli a6, a0, 37
+; RV64IM-NEXT:    slli a7, a1, 27
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    or a5, a7, a6
+; RV64IM-NEXT:    srli a6, a0, 36
+; RV64IM-NEXT:    slli a7, a1, 28
+; RV64IM-NEXT:    ld s10, 24(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a5, s10, a5
+; RV64IM-NEXT:    or a6, a7, a6
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    ld a5, 16(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a5, a5, a6
+; RV64IM-NEXT:    xor a3, s4, a3
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    srli a5, a0, 35
+; RV64IM-NEXT:    slli a6, a1, 29
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    or a4, a6, a5
+; RV64IM-NEXT:    srli a5, a0, 34
+; RV64IM-NEXT:    slli a6, a1, 30
+; RV64IM-NEXT:    ld a7, 8(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, a7, a4
+; RV64IM-NEXT:    or a5, a6, a5
+; RV64IM-NEXT:    srli a6, a0, 33
+; RV64IM-NEXT:    slli a7, a1, 31
+; RV64IM-NEXT:    ld s4, 144(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a5, s4, a5
+; RV64IM-NEXT:    or a6, a7, a6
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    and a5, t2, a6
+; RV64IM-NEXT:    srli a6, a0, 32
+; RV64IM-NEXT:    slli a7, a1, 32
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    or a5, a7, a6
+; RV64IM-NEXT:    srli a6, a0, 31
+; RV64IM-NEXT:    slli a7, a1, 33
+; RV64IM-NEXT:    ld t2, 32(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a5, t2, a5
+; RV64IM-NEXT:    or a6, a7, a6
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    and a5, t4, a6
+; RV64IM-NEXT:    srli a6, a0, 30
+; RV64IM-NEXT:    slli a7, a1, 34
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    or a5, a7, a6
+; RV64IM-NEXT:    srli a6, a0, 29
+; RV64IM-NEXT:    slli a7, a1, 35
+; RV64IM-NEXT:    and a5, s9, a5
+; RV64IM-NEXT:    or a6, a7, a6
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    and a5, s6, a6
+; RV64IM-NEXT:    srli a6, a0, 28
+; RV64IM-NEXT:    slli a7, a1, 36
+; RV64IM-NEXT:    xor a5, a4, a5
+; RV64IM-NEXT:    or a4, a7, a6
+; RV64IM-NEXT:    and a6, s7, a4
+; RV64IM-NEXT:    slli a4, a2, 2
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    slli a7, a0, 61
+; RV64IM-NEXT:    xor a6, a5, a6
+; RV64IM-NEXT:    and a5, a4, a7
+; RV64IM-NEXT:    xor a5, s3, a5
+; RV64IM-NEXT:    xor a6, a3, a6
+; RV64IM-NEXT:    srli a3, a0, 27
+; RV64IM-NEXT:    slli a7, a1, 37
+; RV64IM-NEXT:    srli t2, a0, 26
+; RV64IM-NEXT:    slli t4, a1, 38
+; RV64IM-NEXT:    or a3, a7, a3
+; RV64IM-NEXT:    or a7, t4, t2
+; RV64IM-NEXT:    ld t2, 240(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, t2, a3
+; RV64IM-NEXT:    and a7, s8, a7
+; RV64IM-NEXT:    srli t2, a0, 25
+; RV64IM-NEXT:    slli t4, a1, 39
+; RV64IM-NEXT:    xor a3, a3, a7
+; RV64IM-NEXT:    or a7, t4, t2
+; RV64IM-NEXT:    srli t2, a0, 24
+; RV64IM-NEXT:    slli t4, a1, 40
+; RV64IM-NEXT:    ld s3, 216(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a7, s3, a7
+; RV64IM-NEXT:    or t2, t4, t2
+; RV64IM-NEXT:    xor a3, a3, a7
+; RV64IM-NEXT:    and a7, t6, t2
+; RV64IM-NEXT:    srli t2, a0, 23
+; RV64IM-NEXT:    slli t4, a1, 41
+; RV64IM-NEXT:    xor a3, a3, a7
+; RV64IM-NEXT:    or a7, t4, t2
+; RV64IM-NEXT:    srli t2, a0, 22
+; RV64IM-NEXT:    slli t4, a1, 42
+; RV64IM-NEXT:    ld t6, 208(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a7, t6, a7
+; RV64IM-NEXT:    or t2, t4, t2
+; RV64IM-NEXT:    xor a3, a3, a7
+; RV64IM-NEXT:    and a7, s0, t2
+; RV64IM-NEXT:    srli t2, a0, 21
+; RV64IM-NEXT:    slli t4, a1, 43
+; RV64IM-NEXT:    xor a3, a3, a7
+; RV64IM-NEXT:    or a7, t4, t2
+; RV64IM-NEXT:    srli t2, a0, 20
+; RV64IM-NEXT:    slli t4, a1, 44
+; RV64IM-NEXT:    ld t6, 200(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a7, t6, a7
+; RV64IM-NEXT:    or t2, t4, t2
+; RV64IM-NEXT:    xor a3, a3, a7
+; RV64IM-NEXT:    and a7, s1, t2
+; RV64IM-NEXT:    srli t2, a0, 19
+; RV64IM-NEXT:    slli t4, a1, 45
+; RV64IM-NEXT:    xor a3, a3, a7
+; RV64IM-NEXT:    or a7, t4, t2
+; RV64IM-NEXT:    and t2, s5, a7
+; RV64IM-NEXT:    slli a7, a2, 1
+; RV64IM-NEXT:    srai a7, a7, 63
+; RV64IM-NEXT:    slli t4, a0, 62
+; RV64IM-NEXT:    xor a3, a3, t2
+; RV64IM-NEXT:    and t2, a7, t4
+; RV64IM-NEXT:    xor a5, a5, t2
+; RV64IM-NEXT:    xor a6, a6, a3
+; RV64IM-NEXT:    srli a3, a0, 18
+; RV64IM-NEXT:    slli t2, a1, 46
+; RV64IM-NEXT:    srli t4, a0, 17
+; RV64IM-NEXT:    slli t6, a1, 47
+; RV64IM-NEXT:    or a3, t2, a3
+; RV64IM-NEXT:    or t2, t6, t4
+; RV64IM-NEXT:    ld t4, 120(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, t4, a3
+; RV64IM-NEXT:    ld t4, 112(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and t2, t4, t2
+; RV64IM-NEXT:    srli t4, a0, 16
+; RV64IM-NEXT:    slli t6, a1, 48
+; RV64IM-NEXT:    xor a3, a3, t2
+; RV64IM-NEXT:    or t2, t6, t4
+; RV64IM-NEXT:    srli t4, a0, 15
+; RV64IM-NEXT:    slli t6, a1, 49
+; RV64IM-NEXT:    ld s0, 296(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and t2, s0, t2
+; RV64IM-NEXT:    or t4, t6, t4
+; RV64IM-NEXT:    xor a3, a3, t2
+; RV64IM-NEXT:    ld t2, 104(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and t2, t2, t4
+; RV64IM-NEXT:    srli t4, a0, 14
+; RV64IM-NEXT:    slli t6, a1, 50
+; RV64IM-NEXT:    xor a3, a3, t2
+; RV64IM-NEXT:    or t2, t6, t4
+; RV64IM-NEXT:    srli t4, a0, 13
+; RV64IM-NEXT:    slli t6, a1, 51
+; RV64IM-NEXT:    ld s0, 288(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and t2, s0, t2
+; RV64IM-NEXT:    or t4, t6, t4
+; RV64IM-NEXT:    xor a3, a3, t2
+; RV64IM-NEXT:    and t2, s11, t4
+; RV64IM-NEXT:    srli t4, a0, 12
+; RV64IM-NEXT:    slli t6, a1, 52
+; RV64IM-NEXT:    xor a3, a3, t2
+; RV64IM-NEXT:    or t2, t6, t4
+; RV64IM-NEXT:    srli t4, a0, 11
+; RV64IM-NEXT:    slli t6, a1, 53
+; RV64IM-NEXT:    ld s0, 280(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and t2, s0, t2
+; RV64IM-NEXT:    or t4, t6, t4
+; RV64IM-NEXT:    xor a3, a3, t2
+; RV64IM-NEXT:    and t2, ra, t4
+; RV64IM-NEXT:    srli t4, a0, 10
+; RV64IM-NEXT:    slli t6, a1, 54
+; RV64IM-NEXT:    xor a3, a3, t2
+; RV64IM-NEXT:    or t2, t6, t4
+; RV64IM-NEXT:    srli t4, a0, 9
+; RV64IM-NEXT:    slli t6, a1, 55
+; RV64IM-NEXT:    ld s0, 272(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and t2, s0, t2
+; RV64IM-NEXT:    or t4, t6, t4
+; RV64IM-NEXT:    xor a3, a3, t2
+; RV64IM-NEXT:    ld t2, 96(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and t2, t2, t4
+; RV64IM-NEXT:    srli t4, a0, 8
+; RV64IM-NEXT:    slli t6, a1, 56
+; RV64IM-NEXT:    xor t2, a3, t2
+; RV64IM-NEXT:    or a3, t6, t4
+; RV64IM-NEXT:    srli t4, a0, 7
+; RV64IM-NEXT:    slli t6, a1, 57
+; RV64IM-NEXT:    and a3, t5, a3
+; RV64IM-NEXT:    or t4, t6, t4
+; RV64IM-NEXT:    srli t5, a0, 6
+; RV64IM-NEXT:    slli t6, a1, 58
+; RV64IM-NEXT:    ld s0, 168(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and t4, s0, t4
+; RV64IM-NEXT:    or t5, t6, t5
+; RV64IM-NEXT:    xor a3, a3, t4
+; RV64IM-NEXT:    and t1, t1, t5
+; RV64IM-NEXT:    srli t4, a0, 5
+; RV64IM-NEXT:    slli t5, a1, 59
+; RV64IM-NEXT:    xor a3, a3, t1
+; RV64IM-NEXT:    or t1, t5, t4
+; RV64IM-NEXT:    srli t4, a0, 4
+; RV64IM-NEXT:    slli t5, a1, 60
+; RV64IM-NEXT:    and t1, t3, t1
+; RV64IM-NEXT:    or t3, t5, t4
+; RV64IM-NEXT:    xor a3, a3, t1
+; RV64IM-NEXT:    and t1, s2, t3
+; RV64IM-NEXT:    srli t3, a0, 3
+; RV64IM-NEXT:    slli t4, a1, 61
+; RV64IM-NEXT:    xor a3, a3, t1
+; RV64IM-NEXT:    or t1, t4, t3
+; RV64IM-NEXT:    srli t3, a0, 2
+; RV64IM-NEXT:    slli t4, a1, 62
+; RV64IM-NEXT:    and a4, a4, t1
+; RV64IM-NEXT:    or t1, t4, t3
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    and a4, a7, t1
+; RV64IM-NEXT:    slli a1, a1, 63
+; RV64IM-NEXT:    srli a7, a0, 1
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    or a1, a1, a7
+; RV64IM-NEXT:    srai a2, a2, 63
+; RV64IM-NEXT:    slli a4, a0, 63
+; RV64IM-NEXT:    and a4, a2, a4
+; RV64IM-NEXT:    slli a7, t0, 63
+; RV64IM-NEXT:    and a1, a2, a1
+; RV64IM-NEXT:    srai a2, a7, 63
+; RV64IM-NEXT:    xor a1, a3, a1
+; RV64IM-NEXT:    and a0, a2, a0
+; RV64IM-NEXT:    xor a0, a1, a0
+; RV64IM-NEXT:    slli a1, t0, 62
+; RV64IM-NEXT:    srai a1, a1, 63
+; RV64IM-NEXT:    slli a2, t0, 61
+; RV64IM-NEXT:    ld a3, 568(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a1, a1, a3
+; RV64IM-NEXT:    srai a2, a2, 63
+; RV64IM-NEXT:    xor a0, a0, a1
+; RV64IM-NEXT:    ld a1, 536(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a1, a2, a1
+; RV64IM-NEXT:    xor a2, a6, t2
+; RV64IM-NEXT:    xor a1, a0, a1
+; RV64IM-NEXT:    xor a0, a5, a4
+; RV64IM-NEXT:    xor a1, a2, a1
+; RV64IM-NEXT:    slli a2, t0, 60
+; RV64IM-NEXT:    slli a3, t0, 59
+; RV64IM-NEXT:    srai a2, a2, 63
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    ld a4, 576(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a2, a2, a4
+; RV64IM-NEXT:    ld a4, 552(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a4
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 58
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 57
+; RV64IM-NEXT:    ld a5, 560(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 544(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 56
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 55
+; RV64IM-NEXT:    ld a5, 512(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 520(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 54
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 53
+; RV64IM-NEXT:    ld a5, 528(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 504(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 52
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 51
+; RV64IM-NEXT:    ld a5, 472(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 488(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 50
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 49
+; RV64IM-NEXT:    ld a5, 496(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 464(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 48
+; RV64IM-NEXT:    xor a1, a1, a2
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    ld a2, 480(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a2, a3, a2
+; RV64IM-NEXT:    slli a3, t0, 47
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 46
+; RV64IM-NEXT:    ld a5, 448(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 456(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 45
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 44
+; RV64IM-NEXT:    ld a5, 432(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 440(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 43
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 42
+; RV64IM-NEXT:    ld a5, 416(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 424(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 41
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 40
+; RV64IM-NEXT:    ld a5, 384(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 392(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 39
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 38
+; RV64IM-NEXT:    ld a5, 408(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 368(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    slli a3, t0, 37
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    slli a4, t0, 36
+; RV64IM-NEXT:    ld a5, 400(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    ld a3, 352(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a4, a3
+; RV64IM-NEXT:    slli a4, t0, 35
+; RV64IM-NEXT:    slli a5, t0, 34
+; RV64IM-NEXT:    srai a4, a4, 63
+; RV64IM-NEXT:    srai a5, a5, 63
+; RV64IM-NEXT:    ld a6, 376(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, a4, a6
+; RV64IM-NEXT:    ld a6, 344(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a5, a5, a6
+; RV64IM-NEXT:    xor a2, a2, a3
+; RV64IM-NEXT:    xor a4, a4, a5
+; RV64IM-NEXT:    slli a3, t0, 33
+; RV64IM-NEXT:    sraiw a5, t0, 31
+; RV64IM-NEXT:    srai a3, a3, 63
+; RV64IM-NEXT:    seqz a5, a5
+; RV64IM-NEXT:    ld a6, 360(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a3, a3, a6
+; RV64IM-NEXT:    addi a5, a5, -1
+; RV64IM-NEXT:    xor a3, a4, a3
+; RV64IM-NEXT:    ld a4, 336(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, a5, a4
+; RV64IM-NEXT:    xor a1, a1, a2
+; RV64IM-NEXT:    xor a3, a3, a4
+; RV64IM-NEXT:    ld a2, 328(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    xor a0, a2, a0
+; RV64IM-NEXT:    xor a1, a1, a3
+; RV64IM-NEXT:    ld ra, 680(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s0, 672(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s1, 664(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s2, 656(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s3, 648(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s4, 640(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s5, 632(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s6, 624(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s7, 616(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s8, 608(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s9, 600(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s10, 592(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s11, 584(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    .cfi_restore ra
+; RV64IM-NEXT:    .cfi_restore s0
+; RV64IM-NEXT:    .cfi_restore s1
+; RV64IM-NEXT:    .cfi_restore s2
+; RV64IM-NEXT:    .cfi_restore s3
+; RV64IM-NEXT:    .cfi_restore s4
+; RV64IM-NEXT:    .cfi_restore s5
+; RV64IM-NEXT:    .cfi_restore s6
+; RV64IM-NEXT:    .cfi_restore s7
+; RV64IM-NEXT:    .cfi_restore s8
+; RV64IM-NEXT:    .cfi_restore s9
+; RV64IM-NEXT:    .cfi_restore s10
+; RV64IM-NEXT:    .cfi_restore s11
+; RV64IM-NEXT:    addi sp, sp, 688
+; RV64IM-NEXT:    .cfi_def_cfa_offset 0
+; RV64IM-NEXT:    ret
+;
+; RV32IMZBS-LABEL: clmul_i96:
+; RV32IMZBS:       # %bb.0:
+; RV32IMZBS-NEXT:    addi sp, sp, -496
+; RV32IMZBS-NEXT:    .cfi_def_cfa_offset 496
+; RV32IMZBS-NEXT:    sw ra, 492(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s0, 488(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s1, 484(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s2, 480(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s3, 476(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s4, 472(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s5, 468(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s6, 464(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s7, 460(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s8, 456(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s9, 452(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s10, 448(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s11, 444(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    .cfi_offset ra, -4
+; RV32IMZBS-NEXT:    .cfi_offset s0, -8
+; RV32IMZBS-NEXT:    .cfi_offset s1, -12
+; RV32IMZBS-NEXT:    .cfi_offset s2, -16
+; RV32IMZBS-NEXT:    .cfi_offset s3, -20
+; RV32IMZBS-NEXT:    .cfi_offset s4, -24
+; RV32IMZBS-NEXT:    .cfi_offset s5, -28
+; RV32IMZBS-NEXT:    .cfi_offset s6, -32
+; RV32IMZBS-NEXT:    .cfi_offset s7, -36
+; RV32IMZBS-NEXT:    .cfi_offset s8, -40
+; RV32IMZBS-NEXT:    .cfi_offset s9, -44
+; RV32IMZBS-NEXT:    .cfi_offset s10, -48
+; RV32IMZBS-NEXT:    .cfi_offset s11, -52
+; RV32IMZBS-NEXT:    sw a0, 300(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw s10, 4(a1)
+; RV32IMZBS-NEXT:    lw a0, 8(a1)
+; RV32IMZBS-NEXT:    lw a4, 0(a2)
+; RV32IMZBS-NEXT:    lw s2, 4(a2)
+; RV32IMZBS-NEXT:    lw s11, 8(a2)
+; RV32IMZBS-NEXT:    srli a2, s10, 31
+; RV32IMZBS-NEXT:    slli a5, a0, 1
+; RV32IMZBS-NEXT:    slli a6, a4, 30
+; RV32IMZBS-NEXT:    slli a7, a4, 31
+; RV32IMZBS-NEXT:    or a2, a5, a2
+; RV32IMZBS-NEXT:    srai a5, a6, 31
+; RV32IMZBS-NEXT:    sw a5, 368(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai a6, a7, 31
+; RV32IMZBS-NEXT:    sw a6, 364(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a2, a5, a2
+; RV32IMZBS-NEXT:    and a5, a6, a0
+; RV32IMZBS-NEXT:    xor a2, a5, a2
+; RV32IMZBS-NEXT:    srli a5, s10, 30
+; RV32IMZBS-NEXT:    slli a6, a0, 2
+; RV32IMZBS-NEXT:    or a5, a6, a5
+; RV32IMZBS-NEXT:    slli a6, a4, 29
+; RV32IMZBS-NEXT:    srai t1, a6, 31
+; RV32IMZBS-NEXT:    sw t1, 360(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a6, s10, 29
+; RV32IMZBS-NEXT:    slli a7, a0, 3
+; RV32IMZBS-NEXT:    slli t0, a4, 28
+; RV32IMZBS-NEXT:    or a6, a7, a6
+; RV32IMZBS-NEXT:    srai a7, t0, 31
+; RV32IMZBS-NEXT:    sw a7, 356(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, t1, a5
+; RV32IMZBS-NEXT:    and a6, a7, a6
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    srli a6, s10, 28
+; RV32IMZBS-NEXT:    slli a7, a0, 4
+; RV32IMZBS-NEXT:    slli t0, a4, 27
+; RV32IMZBS-NEXT:    or a6, a7, a6
+; RV32IMZBS-NEXT:    srai a7, t0, 31
+; RV32IMZBS-NEXT:    sw a7, 352(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a6, a7, a6
+; RV32IMZBS-NEXT:    srli a7, s10, 27
+; RV32IMZBS-NEXT:    slli t0, a0, 5
+; RV32IMZBS-NEXT:    slli t1, a4, 26
+; RV32IMZBS-NEXT:    or a7, t0, a7
+; RV32IMZBS-NEXT:    srai t0, t1, 31
+; RV32IMZBS-NEXT:    sw t0, 348(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a7, t0, a7
+; RV32IMZBS-NEXT:    srli t0, s10, 26
+; RV32IMZBS-NEXT:    slli t1, a0, 6
+; RV32IMZBS-NEXT:    slli t2, a4, 25
+; RV32IMZBS-NEXT:    or t0, t1, t0
+; RV32IMZBS-NEXT:    srai t1, t2, 31
+; RV32IMZBS-NEXT:    sw t1, 344(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a6, a6, a7
+; RV32IMZBS-NEXT:    and a7, t1, t0
+; RV32IMZBS-NEXT:    xor a2, a2, a5
+; RV32IMZBS-NEXT:    xor a5, a6, a7
+; RV32IMZBS-NEXT:    xor a2, a2, a5
+; RV32IMZBS-NEXT:    srli a5, s10, 25
+; RV32IMZBS-NEXT:    slli a6, a0, 7
+; RV32IMZBS-NEXT:    or a5, a6, a5
+; RV32IMZBS-NEXT:    slli a6, a4, 24
+; RV32IMZBS-NEXT:    srai t1, a6, 31
+; RV32IMZBS-NEXT:    sw t1, 340(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a6, s10, 24
+; RV32IMZBS-NEXT:    slli a7, a0, 8
+; RV32IMZBS-NEXT:    slli t0, a4, 23
+; RV32IMZBS-NEXT:    or a6, a7, a6
+; RV32IMZBS-NEXT:    srai a7, t0, 31
+; RV32IMZBS-NEXT:    sw a7, 336(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, t1, a5
+; RV32IMZBS-NEXT:    and a6, a7, a6
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    srli a6, s10, 23
+; RV32IMZBS-NEXT:    slli a7, a0, 9
+; RV32IMZBS-NEXT:    slli t0, a4, 22
+; RV32IMZBS-NEXT:    or a6, a7, a6
+; RV32IMZBS-NEXT:    srai a7, t0, 31
+; RV32IMZBS-NEXT:    sw a7, 332(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a6, a7, a6
+; RV32IMZBS-NEXT:    srli a7, s10, 22
+; RV32IMZBS-NEXT:    slli t0, a0, 10
+; RV32IMZBS-NEXT:    slli t1, a4, 21
+; RV32IMZBS-NEXT:    or a7, t0, a7
+; RV32IMZBS-NEXT:    srai t0, t1, 31
+; RV32IMZBS-NEXT:    sw t0, 328(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    and a6, t0, a7
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    srli a6, s10, 21
+; RV32IMZBS-NEXT:    slli a7, a0, 11
+; RV32IMZBS-NEXT:    slli t0, a4, 20
+; RV32IMZBS-NEXT:    or a6, a7, a6
+; RV32IMZBS-NEXT:    srai a7, t0, 31
+; RV32IMZBS-NEXT:    sw a7, 324(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a6, a7, a6
+; RV32IMZBS-NEXT:    srli a7, s10, 20
+; RV32IMZBS-NEXT:    slli t0, a0, 12
+; RV32IMZBS-NEXT:    slli t1, a4, 19
+; RV32IMZBS-NEXT:    or a7, t0, a7
+; RV32IMZBS-NEXT:    srai t0, t1, 31
+; RV32IMZBS-NEXT:    sw t0, 320(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a7, t0, a7
+; RV32IMZBS-NEXT:    srli t0, s10, 19
+; RV32IMZBS-NEXT:    slli t1, a0, 13
+; RV32IMZBS-NEXT:    slli t2, a4, 18
+; RV32IMZBS-NEXT:    or t0, t1, t0
+; RV32IMZBS-NEXT:    srai t1, t2, 31
+; RV32IMZBS-NEXT:    sw t1, 316(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a6, a6, a7
+; RV32IMZBS-NEXT:    and a7, t1, t0
+; RV32IMZBS-NEXT:    xor a6, a6, a7
+; RV32IMZBS-NEXT:    srli a7, s10, 18
+; RV32IMZBS-NEXT:    slli t0, a0, 14
+; RV32IMZBS-NEXT:    slli t1, a4, 17
+; RV32IMZBS-NEXT:    or a7, t0, a7
+; RV32IMZBS-NEXT:    srai t2, t1, 31
+; RV32IMZBS-NEXT:    sw t2, 400(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli t0, s10, 17
+; RV32IMZBS-NEXT:    slli t1, a0, 15
+; RV32IMZBS-NEXT:    or t0, t1, t0
+; RV32IMZBS-NEXT:    slli t1, a4, 16
+; RV32IMZBS-NEXT:    and a7, t2, a7
+; RV32IMZBS-NEXT:    srai t1, t1, 31
+; RV32IMZBS-NEXT:    sw t1, 396(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a6, a6, a7
+; RV32IMZBS-NEXT:    and a7, t1, t0
+; RV32IMZBS-NEXT:    xor a2, a2, a5
+; RV32IMZBS-NEXT:    xor a5, a6, a7
+; RV32IMZBS-NEXT:    xor a2, a2, a5
+; RV32IMZBS-NEXT:    srli a5, s10, 16
+; RV32IMZBS-NEXT:    slli a6, a0, 16
+; RV32IMZBS-NEXT:    or a5, a6, a5
+; RV32IMZBS-NEXT:    slli a6, a4, 15
+; RV32IMZBS-NEXT:    srli a7, s10, 15
+; RV32IMZBS-NEXT:    slli t0, a0, 17
+; RV32IMZBS-NEXT:    or a7, t0, a7
+; RV32IMZBS-NEXT:    slli t0, a4, 14
+; RV32IMZBS-NEXT:    srai a6, a6, 31
+; RV32IMZBS-NEXT:    sw a6, 392(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai t0, t0, 31
+; RV32IMZBS-NEXT:    sw t0, 388(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a6, a5
+; RV32IMZBS-NEXT:    and a6, t0, a7
+; RV32IMZBS-NEXT:    srli a7, s10, 14
+; RV32IMZBS-NEXT:    slli t0, a0, 18
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    or a6, t0, a7
+; RV32IMZBS-NEXT:    srli a7, s10, 13
+; RV32IMZBS-NEXT:    slli t0, a0, 19
+; RV32IMZBS-NEXT:    or a7, t0, a7
+; RV32IMZBS-NEXT:    slli t0, a4, 13
+; RV32IMZBS-NEXT:    srai t1, t0, 31
+; RV32IMZBS-NEXT:    sw t1, 384(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t0, a4, 12
+; RV32IMZBS-NEXT:    and a6, t1, a6
+; RV32IMZBS-NEXT:    srai t0, t0, 31
+; RV32IMZBS-NEXT:    sw t0, 380(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    and a6, t0, a7
+; RV32IMZBS-NEXT:    srli a7, s10, 12
+; RV32IMZBS-NEXT:    slli t0, a0, 20
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    or a6, t0, a7
+; RV32IMZBS-NEXT:    srli a7, s10, 11
+; RV32IMZBS-NEXT:    slli t0, a0, 21
+; RV32IMZBS-NEXT:    or a7, t0, a7
+; RV32IMZBS-NEXT:    slli t0, a4, 11
+; RV32IMZBS-NEXT:    srai t1, t0, 31
+; RV32IMZBS-NEXT:    sw t1, 376(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t0, a4, 10
+; RV32IMZBS-NEXT:    and a6, t1, a6
+; RV32IMZBS-NEXT:    srai t0, t0, 31
+; RV32IMZBS-NEXT:    sw t0, 372(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    and a6, t0, a7
+; RV32IMZBS-NEXT:    srli a7, s10, 10
+; RV32IMZBS-NEXT:    slli t0, a0, 22
+; RV32IMZBS-NEXT:    xor a5, a5, a6
+; RV32IMZBS-NEXT:    or a6, t0, a7
+; RV32IMZBS-NEXT:    srli a7, s10, 9
+; RV32IMZBS-NEXT:    slli t0, a0, 23
+; RV32IMZBS-NEXT:    or a7, t0, a7
+; RV32IMZBS-NEXT:    slli t0, a4, 9
+; RV32IMZBS-NEXT:    srli t1, s10, 8
+; RV32IMZBS-NEXT:    slli t2, a0, 24
+; RV32IMZBS-NEXT:    or t1, t2, t1
+; RV32IMZBS-NEXT:    srai t0, t0, 31
+; RV32IMZBS-NEXT:    sw t0, 440(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a6, t0, a6
+; RV32IMZBS-NEXT:    slli t0, a4, 8
+; RV32IMZBS-NEXT:    srai t2, t0, 31
+; RV32IMZBS-NEXT:    sw t2, 436(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t0, a4, 7
+; RV32IMZBS-NEXT:    and a7, t2, a7
+; RV32IMZBS-NEXT:    srai t0, t0, 31
+; RV32IMZBS-NEXT:    sw t0, 432(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a6, a6, a7
+; RV32IMZBS-NEXT:    and a7, t0, t1
+; RV32IMZBS-NEXT:    srli t0, s10, 7
+; RV32IMZBS-NEXT:    slli t1, a0, 25
+; RV32IMZBS-NEXT:    xor a6, a6, a7
+; RV32IMZBS-NEXT:    or a7, t1, t0
+; RV32IMZBS-NEXT:    srli t0, s10, 6
+; RV32IMZBS-NEXT:    slli t1, a0, 26
+; RV32IMZBS-NEXT:    or t0, t1, t0
+; RV32IMZBS-NEXT:    slli t1, a4, 6
+; RV32IMZBS-NEXT:    srai t2, t1, 31
+; RV32IMZBS-NEXT:    sw t2, 428(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t1, a4, 5
+; RV32IMZBS-NEXT:    and a7, t2, a7
+; RV32IMZBS-NEXT:    srai t1, t1, 31
+; RV32IMZBS-NEXT:    sw t1, 424(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a6, a6, a7
+; RV32IMZBS-NEXT:    and a7, t1, t0
+; RV32IMZBS-NEXT:    srli t0, s10, 5
+; RV32IMZBS-NEXT:    slli t1, a0, 27
+; RV32IMZBS-NEXT:    xor a6, a6, a7
+; RV32IMZBS-NEXT:    or a7, t1, t0
+; RV32IMZBS-NEXT:    srli t0, s10, 4
+; RV32IMZBS-NEXT:    slli t1, a0, 28
+; RV32IMZBS-NEXT:    or t0, t1, t0
+; RV32IMZBS-NEXT:    slli t1, a4, 4
+; RV32IMZBS-NEXT:    srai t2, t1, 31
+; RV32IMZBS-NEXT:    sw t2, 420(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli t1, a4, 3
+; RV32IMZBS-NEXT:    and a7, t2, a7
+; RV32IMZBS-NEXT:    srai t1, t1, 31
+; RV32IMZBS-NEXT:    sw t1, 416(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a6, a6, a7
+; RV32IMZBS-NEXT:    and a7, t1, t0
+; RV32IMZBS-NEXT:    xor a2, a2, a5
+; RV32IMZBS-NEXT:    xor a5, a6, a7
+; RV32IMZBS-NEXT:    srli a6, s10, 3
+; RV32IMZBS-NEXT:    slli a7, a0, 29
+; RV32IMZBS-NEXT:    srli t0, s10, 2
+; RV32IMZBS-NEXT:    slli t1, a0, 30
+; RV32IMZBS-NEXT:    or a6, a7, a6
+; RV32IMZBS-NEXT:    or a7, t1, t0
+; RV32IMZBS-NEXT:    slli t0, a4, 2
+; RV32IMZBS-NEXT:    slli t1, a4, 1
+; RV32IMZBS-NEXT:    srai t0, t0, 31
+; RV32IMZBS-NEXT:    sw t0, 412(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai t1, t1, 31
+; RV32IMZBS-NEXT:    sw t1, 408(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a6, t0, a6
+; RV32IMZBS-NEXT:    and a7, t1, a7
+; RV32IMZBS-NEXT:    xor a2, a2, a5
+; RV32IMZBS-NEXT:    xor a5, a6, a7
+; RV32IMZBS-NEXT:    slli a0, a0, 31
+; RV32IMZBS-NEXT:    srli a6, s10, 1
+; RV32IMZBS-NEXT:    or a6, a0, a6
+; RV32IMZBS-NEXT:    lw a0, 0(a1)
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    sw a4, 404(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a1, s2, 31
+; RV32IMZBS-NEXT:    and a4, a4, a6
+; RV32IMZBS-NEXT:    srai a1, a1, 31
+; RV32IMZBS-NEXT:    sw a1, 296(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a4, a5, a4
+; RV32IMZBS-NEXT:    and a1, a1, s10
+; RV32IMZBS-NEXT:    slli a5, s10, 1
+; RV32IMZBS-NEXT:    srli a6, a0, 31
+; RV32IMZBS-NEXT:    xor a1, a4, a1
+; RV32IMZBS-NEXT:    or t1, a6, a5
+; RV32IMZBS-NEXT:    slli a4, s10, 2
+; RV32IMZBS-NEXT:    srli a5, a0, 30
+; RV32IMZBS-NEXT:    or a7, a5, a4
+; RV32IMZBS-NEXT:    sw a7, 152(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 30
+; RV32IMZBS-NEXT:    srai a5, a4, 31
+; RV32IMZBS-NEXT:    sw a5, 292(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 29
+; RV32IMZBS-NEXT:    srai a6, a4, 31
+; RV32IMZBS-NEXT:    sw a6, 288(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a5, t1
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    and a4, a6, a7
+; RV32IMZBS-NEXT:    slli a5, s10, 3
+; RV32IMZBS-NEXT:    srli a6, a0, 29
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    or t0, a6, a5
+; RV32IMZBS-NEXT:    sw t0, 168(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a4, a0, 28
+; RV32IMZBS-NEXT:    slli a5, s10, 4
+; RV32IMZBS-NEXT:    or a7, a5, a4
+; RV32IMZBS-NEXT:    sw a7, 172(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 28
+; RV32IMZBS-NEXT:    srai a5, a4, 31
+; RV32IMZBS-NEXT:    sw a5, 284(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 27
+; RV32IMZBS-NEXT:    srai a6, a4, 31
+; RV32IMZBS-NEXT:    sw a6, 280(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a5, t0
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    and a4, a6, a7
+; RV32IMZBS-NEXT:    slli a5, s10, 5
+; RV32IMZBS-NEXT:    srli a6, a0, 27
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    or s3, a6, a5
+; RV32IMZBS-NEXT:    slli a4, s10, 6
+; RV32IMZBS-NEXT:    srli a5, a0, 26
+; RV32IMZBS-NEXT:    or t2, a5, a4
+; RV32IMZBS-NEXT:    sw t2, 148(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 26
+; RV32IMZBS-NEXT:    slli a5, s10, 7
+; RV32IMZBS-NEXT:    srli a6, a0, 25
+; RV32IMZBS-NEXT:    or s4, a6, a5
+; RV32IMZBS-NEXT:    srai a6, a4, 31
+; RV32IMZBS-NEXT:    sw a6, 276(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 25
+; RV32IMZBS-NEXT:    slli a5, s2, 24
+; RV32IMZBS-NEXT:    srai t0, a4, 31
+; RV32IMZBS-NEXT:    sw t0, 268(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai a7, a5, 31
+; RV32IMZBS-NEXT:    sw a7, 272(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a6, s3
+; RV32IMZBS-NEXT:    and a5, t0, t2
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a7, s4
+; RV32IMZBS-NEXT:    srli a6, a0, 24
+; RV32IMZBS-NEXT:    slli a7, s10, 8
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    or a3, a7, a6
+; RV32IMZBS-NEXT:    slli a5, s10, 9
+; RV32IMZBS-NEXT:    srli a6, a0, 23
+; RV32IMZBS-NEXT:    or t0, a6, a5
+; RV32IMZBS-NEXT:    sw t0, 164(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, s2, 23
+; RV32IMZBS-NEXT:    srai a6, a5, 31
+; RV32IMZBS-NEXT:    sw a6, 264(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, s2, 22
+; RV32IMZBS-NEXT:    srai a7, a5, 31
+; RV32IMZBS-NEXT:    sw a7, 260(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a6, a3
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a7, t0
+; RV32IMZBS-NEXT:    slli a6, s10, 10
+; RV32IMZBS-NEXT:    srli a7, a0, 22
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    or t0, a7, a6
+; RV32IMZBS-NEXT:    sw t0, 144(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, s10, 11
+; RV32IMZBS-NEXT:    srli a6, a0, 21
+; RV32IMZBS-NEXT:    or t2, a6, a5
+; RV32IMZBS-NEXT:    sw t2, 132(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, s2, 21
+; RV32IMZBS-NEXT:    srai a6, a5, 31
+; RV32IMZBS-NEXT:    sw a6, 256(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, s2, 20
+; RV32IMZBS-NEXT:    srai a7, a5, 31
+; RV32IMZBS-NEXT:    sw a7, 252(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a6, t0
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a7, t2
+; RV32IMZBS-NEXT:    slli a6, s10, 12
+; RV32IMZBS-NEXT:    srli a7, a0, 20
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    or t0, a7, a6
+; RV32IMZBS-NEXT:    sw t0, 120(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, s10, 13
+; RV32IMZBS-NEXT:    srli a6, a0, 19
+; RV32IMZBS-NEXT:    or t2, a6, a5
+; RV32IMZBS-NEXT:    sw t2, 112(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, s2, 19
+; RV32IMZBS-NEXT:    srai a6, a5, 31
+; RV32IMZBS-NEXT:    sw a6, 248(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, s2, 18
+; RV32IMZBS-NEXT:    srai a7, a5, 31
+; RV32IMZBS-NEXT:    sw a7, 244(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a6, t0
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a7, t2
+; RV32IMZBS-NEXT:    xor a1, a2, a1
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    slli a2, s10, 14
+; RV32IMZBS-NEXT:    srli a4, a0, 18
+; RV32IMZBS-NEXT:    slli a5, s10, 15
+; RV32IMZBS-NEXT:    srli a6, a0, 17
+; RV32IMZBS-NEXT:    or a7, a4, a2
+; RV32IMZBS-NEXT:    sw a7, 104(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a5, a6, a5
+; RV32IMZBS-NEXT:    sw a5, 140(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a2, s2, 17
+; RV32IMZBS-NEXT:    slli a4, s2, 16
+; RV32IMZBS-NEXT:    srai a2, a2, 31
+; RV32IMZBS-NEXT:    sw a2, 240(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    sw a4, 236(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a2, a2, a7
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    srli a5, a0, 16
+; RV32IMZBS-NEXT:    slli a6, s10, 16
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    or a7, a6, a5
+; RV32IMZBS-NEXT:    sw a7, 128(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a4, a0, 15
+; RV32IMZBS-NEXT:    slli a5, s10, 17
+; RV32IMZBS-NEXT:    or t0, a5, a4
+; RV32IMZBS-NEXT:    sw t0, 116(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 15
+; RV32IMZBS-NEXT:    srai a5, a4, 31
+; RV32IMZBS-NEXT:    sw a5, 232(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 14
+; RV32IMZBS-NEXT:    srai a6, a4, 31
+; RV32IMZBS-NEXT:    sw a6, 228(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a5, a7
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    and a4, a6, t0
+; RV32IMZBS-NEXT:    srli a5, a0, 14
+; RV32IMZBS-NEXT:    slli a6, s10, 18
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    or t0, a6, a5
+; RV32IMZBS-NEXT:    sw t0, 108(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a4, a0, 13
+; RV32IMZBS-NEXT:    slli a5, s10, 19
+; RV32IMZBS-NEXT:    or a7, a5, a4
+; RV32IMZBS-NEXT:    sw a7, 160(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 13
+; RV32IMZBS-NEXT:    srai a5, a4, 31
+; RV32IMZBS-NEXT:    sw a5, 224(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 12
+; RV32IMZBS-NEXT:    srai a6, a4, 31
+; RV32IMZBS-NEXT:    sw a6, 220(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a5, t0
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    and a4, a6, a7
+; RV32IMZBS-NEXT:    srli a5, a0, 12
+; RV32IMZBS-NEXT:    slli a6, s10, 20
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    or t0, a6, a5
+; RV32IMZBS-NEXT:    sw t0, 100(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a4, a0, 11
+; RV32IMZBS-NEXT:    slli a5, s10, 21
+; RV32IMZBS-NEXT:    or a7, a5, a4
+; RV32IMZBS-NEXT:    sw a7, 156(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 11
+; RV32IMZBS-NEXT:    srai a5, a4, 31
+; RV32IMZBS-NEXT:    sw a5, 216(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 10
+; RV32IMZBS-NEXT:    srai a6, a4, 31
+; RV32IMZBS-NEXT:    sw a6, 212(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a5, t0
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    and a4, a6, a7
+; RV32IMZBS-NEXT:    srli a5, a0, 10
+; RV32IMZBS-NEXT:    slli a6, s10, 22
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    or a7, a6, a5
+; RV32IMZBS-NEXT:    sw a7, 136(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a4, a0, 9
+; RV32IMZBS-NEXT:    slli a5, s10, 23
+; RV32IMZBS-NEXT:    or t0, a5, a4
+; RV32IMZBS-NEXT:    sw t0, 124(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 9
+; RV32IMZBS-NEXT:    srai a5, a4, 31
+; RV32IMZBS-NEXT:    sw a5, 200(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 8
+; RV32IMZBS-NEXT:    srai a6, a4, 31
+; RV32IMZBS-NEXT:    sw a6, 196(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a5, a7
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    and a4, a6, t0
+; RV32IMZBS-NEXT:    slli a5, s10, 24
+; RV32IMZBS-NEXT:    srli a6, a0, 8
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    or t2, a6, a5
+; RV32IMZBS-NEXT:    sw t2, 96(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a4, a0, 7
+; RV32IMZBS-NEXT:    slli a5, s10, 25
+; RV32IMZBS-NEXT:    or t3, a5, a4
+; RV32IMZBS-NEXT:    slli a4, s2, 7
+; RV32IMZBS-NEXT:    srli a5, a0, 6
+; RV32IMZBS-NEXT:    slli a6, s10, 26
+; RV32IMZBS-NEXT:    or t5, a6, a5
+; RV32IMZBS-NEXT:    srai a7, a4, 31
+; RV32IMZBS-NEXT:    sw a7, 188(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a4, s2, 6
+; RV32IMZBS-NEXT:    slli a5, s2, 5
+; RV32IMZBS-NEXT:    srai t0, a4, 31
+; RV32IMZBS-NEXT:    sw t0, 184(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srai a6, a5, 31
+; RV32IMZBS-NEXT:    sw a6, 204(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a7, t2
+; RV32IMZBS-NEXT:    and a5, t0, t3
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a6, t5
+; RV32IMZBS-NEXT:    srli a6, a0, 5
+; RV32IMZBS-NEXT:    slli a7, s10, 27
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    or t4, a7, a6
+; RV32IMZBS-NEXT:    slli a5, s10, 28
+; RV32IMZBS-NEXT:    srli a6, a0, 4
+; RV32IMZBS-NEXT:    or t2, a6, a5
+; RV32IMZBS-NEXT:    slli a5, s2, 4
+; RV32IMZBS-NEXT:    srai a7, a5, 31
+; RV32IMZBS-NEXT:    sw a7, 176(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, s2, 3
+; RV32IMZBS-NEXT:    srai a6, a5, 31
+; RV32IMZBS-NEXT:    sw a6, 192(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a7, t4
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a6, t2
+; RV32IMZBS-NEXT:    srli a6, a0, 3
+; RV32IMZBS-NEXT:    slli a7, s10, 29
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    or t6, a7, a6
+; RV32IMZBS-NEXT:    srli a5, a0, 2
+; RV32IMZBS-NEXT:    slli a6, s10, 30
+; RV32IMZBS-NEXT:    or s1, a6, a5
+; RV32IMZBS-NEXT:    slli a5, s2, 2
+; RV32IMZBS-NEXT:    srai a7, a5, 31
+; RV32IMZBS-NEXT:    sw a7, 180(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, s2, 1
+; RV32IMZBS-NEXT:    srai a6, a5, 31
+; RV32IMZBS-NEXT:    sw a6, 208(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a7, t6
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a6, s1
+; RV32IMZBS-NEXT:    srli a6, a0, 1
+; RV32IMZBS-NEXT:    slli a7, s10, 31
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    or t0, a7, a6
+; RV32IMZBS-NEXT:    srai a5, s2, 31
+; RV32IMZBS-NEXT:    slli a6, s11, 31
+; RV32IMZBS-NEXT:    and a5, a5, t0
+; RV32IMZBS-NEXT:    srai a6, a6, 31
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a6, a0
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    slli a5, s11, 30
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli s5, a0, 1
+; RV32IMZBS-NEXT:    and a5, a5, s5
+; RV32IMZBS-NEXT:    slli a6, s11, 29
+; RV32IMZBS-NEXT:    srai a6, a6, 31
+; RV32IMZBS-NEXT:    slli s6, a0, 2
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a6, s6
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    slli a2, s11, 28
+; RV32IMZBS-NEXT:    slli a4, s11, 27
+; RV32IMZBS-NEXT:    srai a2, a2, 31
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    slli a7, a0, 3
+; RV32IMZBS-NEXT:    slli s7, a0, 4
+; RV32IMZBS-NEXT:    and a2, a2, a7
+; RV32IMZBS-NEXT:    and a4, a4, s7
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    slli a4, s11, 26
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    slli s8, a0, 5
+; RV32IMZBS-NEXT:    and a4, a4, s8
+; RV32IMZBS-NEXT:    slli a5, s11, 25
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli s9, a0, 6
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    and a4, a5, s9
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    slli a4, s11, 24
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    slli ra, a0, 7
+; RV32IMZBS-NEXT:    and a4, a4, ra
+; RV32IMZBS-NEXT:    slli a5, s11, 23
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 8
+; RV32IMZBS-NEXT:    sw a6, 92(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    and a4, a5, a6
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    slli a4, s11, 22
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    slli a5, a0, 9
+; RV32IMZBS-NEXT:    sw a5, 88(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    slli a5, s11, 21
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 10
+; RV32IMZBS-NEXT:    sw a6, 84(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    and a4, a5, a6
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    slli a4, s11, 20
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    slli a5, a0, 11
+; RV32IMZBS-NEXT:    sw a5, 80(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    slli a5, s11, 19
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 12
+; RV32IMZBS-NEXT:    sw a6, 76(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    and a4, a5, a6
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    slli a4, s11, 18
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    slli a5, a0, 13
+; RV32IMZBS-NEXT:    sw a5, 72(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    slli a5, s11, 17
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 14
+; RV32IMZBS-NEXT:    sw a6, 64(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    and a4, a5, a6
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    slli a4, s11, 16
+; RV32IMZBS-NEXT:    srai a4, a4, 31
+; RV32IMZBS-NEXT:    slli a5, a0, 15
+; RV32IMZBS-NEXT:    sw a5, 60(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    slli a5, s11, 15
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 16
+; RV32IMZBS-NEXT:    sw a6, 304(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a5, a6
+; RV32IMZBS-NEXT:    slli a6, s11, 14
+; RV32IMZBS-NEXT:    srai a6, a6, 31
+; RV32IMZBS-NEXT:    slli s0, a0, 17
+; RV32IMZBS-NEXT:    sw s0, 56(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a6, s0
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    slli a5, s11, 13
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 18
+; RV32IMZBS-NEXT:    sw a6, 48(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a5, a6
+; RV32IMZBS-NEXT:    slli a6, s11, 12
+; RV32IMZBS-NEXT:    srai a6, a6, 31
+; RV32IMZBS-NEXT:    slli s0, a0, 19
+; RV32IMZBS-NEXT:    sw s0, 40(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a6, s0
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    slli a5, s11, 11
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 20
+; RV32IMZBS-NEXT:    sw a6, 32(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a5, a6
+; RV32IMZBS-NEXT:    slli a6, s11, 10
+; RV32IMZBS-NEXT:    srai a6, a6, 31
+; RV32IMZBS-NEXT:    slli s0, a0, 21
+; RV32IMZBS-NEXT:    sw s0, 16(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a6, s0
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    slli a5, s11, 9
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 22
+; RV32IMZBS-NEXT:    sw a6, 308(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a5, a6
+; RV32IMZBS-NEXT:    slli a6, s11, 8
+; RV32IMZBS-NEXT:    srai a6, a6, 31
+; RV32IMZBS-NEXT:    slli s0, a0, 23
+; RV32IMZBS-NEXT:    sw s0, 312(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a6, s0
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    slli a5, s11, 7
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 24
+; RV32IMZBS-NEXT:    sw a6, 36(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a5, a6
+; RV32IMZBS-NEXT:    slli a6, s11, 6
+; RV32IMZBS-NEXT:    srai a6, a6, 31
+; RV32IMZBS-NEXT:    slli s0, a0, 25
+; RV32IMZBS-NEXT:    sw s0, 24(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, a6, s0
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    slli a5, s11, 5
+; RV32IMZBS-NEXT:    srai a5, a5, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 26
+; RV32IMZBS-NEXT:    sw a6, 20(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a5, a6
+; RV32IMZBS-NEXT:    slli a6, s11, 4
+; RV32IMZBS-NEXT:    srai s0, a6, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 27
+; RV32IMZBS-NEXT:    sw a6, 28(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    and a5, s0, a6
+; RV32IMZBS-NEXT:    xor s0, a1, a2
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    slli a1, s11, 3
+; RV32IMZBS-NEXT:    slli a2, s11, 2
+; RV32IMZBS-NEXT:    srai a1, a1, 31
+; RV32IMZBS-NEXT:    srai a2, a2, 31
+; RV32IMZBS-NEXT:    slli a6, a0, 28
+; RV32IMZBS-NEXT:    sw a6, 12(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a5, a0, 29
+; RV32IMZBS-NEXT:    sw a5, 44(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a1, a1, a6
+; RV32IMZBS-NEXT:    and a2, a2, a5
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    slli a2, s11, 1
+; RV32IMZBS-NEXT:    srai a5, a2, 31
+; RV32IMZBS-NEXT:    slli a2, a0, 30
+; RV32IMZBS-NEXT:    sw a2, 52(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a2, a5, a2
+; RV32IMZBS-NEXT:    slli a5, a0, 31
+; RV32IMZBS-NEXT:    sw a5, 68(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    and a2, s11, a5
+; RV32IMZBS-NEXT:    xor a4, s0, a4
+; RV32IMZBS-NEXT:    sw a4, 8(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    sw a1, 4(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw a1, 368(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, t1
+; RV32IMZBS-NEXT:    lw a2, 364(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a2, a2, s10
+; RV32IMZBS-NEXT:    lw a4, 152(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 360(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a5, a4
+; RV32IMZBS-NEXT:    lw t1, 168(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 356(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a5, t1
+; RV32IMZBS-NEXT:    xor a1, a2, a1
+; RV32IMZBS-NEXT:    xor a2, a4, s10
+; RV32IMZBS-NEXT:    lw a4, 172(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 352(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a5, a4
+; RV32IMZBS-NEXT:    lw a5, 348(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a5, s3
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw a5, 148(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a6, 344(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a6, a5
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a2, a4, s10
+; RV32IMZBS-NEXT:    lw a4, 340(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s4
+; RV32IMZBS-NEXT:    lw a5, 336(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a5, a3
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s0, 164(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 332(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a5, s0
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw a5, 144(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a6, 328(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a6, a5
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a2, a4, s10
+; RV32IMZBS-NEXT:    lw a4, 132(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 324(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a5, a4
+; RV32IMZBS-NEXT:    lw a5, 120(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a6, 320(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a6, a5
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw a5, 112(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a6, 316(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a6, a5
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw a5, 104(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a6, 400(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a6, a5
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw a5, 140(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a6, 396(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a6, a5
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a2, a4, s10
+; RV32IMZBS-NEXT:    lw a4, 392(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 128(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    lw a5, 388(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a6, 116(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a5, a6
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw a5, 384(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a6, 108(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a5, a6
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 160(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 380(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a5, s10
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw a5, 376(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a6, 100(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a5, a6
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 156(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 372(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a5, s10
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a2, a4, s10
+; RV32IMZBS-NEXT:    lw a4, 440(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 136(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    lw s10, 436(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 124(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, a5
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 432(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 96(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, a5
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 428(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, t3
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 424(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, t5
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 420(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, t4
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 416(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, t2
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a2, a4, s10
+; RV32IMZBS-NEXT:    lw a4, 412(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t6
+; RV32IMZBS-NEXT:    lw s10, 408(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, s1
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 404(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, t0
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 296(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, a0
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 292(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mv a5, s5
+; RV32IMZBS-NEXT:    and s10, s10, s5
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 288(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mv t0, s6
+; RV32IMZBS-NEXT:    and s10, s10, s6
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 284(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, a7
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 280(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mv t2, s7
+; RV32IMZBS-NEXT:    and s10, s10, s7
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a2, a4, s10
+; RV32IMZBS-NEXT:    lw a4, 276(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mv a6, s8
+; RV32IMZBS-NEXT:    and a4, a4, s8
+; RV32IMZBS-NEXT:    lw s10, 268(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mv t1, s9
+; RV32IMZBS-NEXT:    and s10, s10, s9
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 272(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mv s0, ra
+; RV32IMZBS-NEXT:    and s10, s10, ra
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 264(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw t3, 92(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, t3
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 260(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw t4, 88(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, t4
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 256(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw t5, 84(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, t5
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 252(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s11, 80(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, s11
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 248(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s4, 76(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, s4
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 244(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s3, 72(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, s3
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a2, a4, s10
+; RV32IMZBS-NEXT:    lw a4, 240(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw t6, 64(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t6
+; RV32IMZBS-NEXT:    lw s10, 236(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s1, 60(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, s1
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 232(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a3, 304(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, a3
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 228(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s5, 56(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, s5
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 224(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s6, 48(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, s6
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 220(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s7, 40(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, s7
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 216(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s8, 32(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, s8
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 212(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s9, 16(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, s9
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 200(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw ra, 308(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, ra
+; RV32IMZBS-NEXT:    xor a4, a4, s10
+; RV32IMZBS-NEXT:    lw s10, 196(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw ra, 312(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, s10, ra
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a2, a4, s10
+; RV32IMZBS-NEXT:    lw a3, 8(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a4, 4(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    xor a3, a3, a4
+; RV32IMZBS-NEXT:    sw a3, 296(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor s10, a1, a2
+; RV32IMZBS-NEXT:    lw a1, 368(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, a5
+; RV32IMZBS-NEXT:    lw a2, 364(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a0, a2, a0
+; RV32IMZBS-NEXT:    lw a2, 188(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw ra, 36(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a2, a2, ra
+; RV32IMZBS-NEXT:    lw a4, 184(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a3, 24(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a3
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    lw a1, 360(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, t0
+; RV32IMZBS-NEXT:    lw a4, 356(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a7
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 204(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a7, 20(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a7
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    lw a1, 352(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, t2
+; RV32IMZBS-NEXT:    lw a4, 348(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a6
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 344(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t1
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 176(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a6, 28(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a6
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    lw a1, 340(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, s0
+; RV32IMZBS-NEXT:    lw a4, 336(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t3
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 332(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t4
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 328(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t5
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 192(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw t1, 12(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t1
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    lw a1, 324(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, s11
+; RV32IMZBS-NEXT:    lw a4, 320(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s4
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 316(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s3
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 400(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t6
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 396(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s1
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 180(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw t0, 44(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t0
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    lw a1, 392(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a4, 304(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 388(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s5
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 384(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s6
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 380(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s7
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 376(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s8
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 372(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s9
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 208(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw t2, 52(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t2
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    lw a1, 440(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a4, 308(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 436(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a5, 312(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 432(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, ra
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 428(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a3
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 424(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a7
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 420(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, a6
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    lw a4, 416(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, t1
+; RV32IMZBS-NEXT:    lw a5, 412(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a7, a5, t0
+; RV32IMZBS-NEXT:    lw a5, 408(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a6, a5, t2
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    xor a4, a7, a6
+; RV32IMZBS-NEXT:    lw a6, 68(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, s2, a6
+; RV32IMZBS-NEXT:    lw a5, 404(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a5, a5, a6
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    xor a2, a2, a3
+; RV32IMZBS-NEXT:    xor a0, a0, a4
+; RV32IMZBS-NEXT:    xor a1, s10, a2
+; RV32IMZBS-NEXT:    lw a2, 300(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    sw a0, 0(a2)
+; RV32IMZBS-NEXT:    sw a1, 4(a2)
+; RV32IMZBS-NEXT:    lw a0, 296(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    sw a0, 8(a2)
+; RV32IMZBS-NEXT:    lw ra, 492(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s0, 488(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s1, 484(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s2, 480(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s3, 476(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s4, 472(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s5, 468(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s6, 464(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s7, 460(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s8, 456(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s9, 452(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s10, 448(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s11, 444(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    .cfi_restore ra
+; RV32IMZBS-NEXT:    .cfi_restore s0
+; RV32IMZBS-NEXT:    .cfi_restore s1
+; RV32IMZBS-NEXT:    .cfi_restore s2
+; RV32IMZBS-NEXT:    .cfi_restore s3
+; RV32IMZBS-NEXT:    .cfi_restore s4
+; RV32IMZBS-NEXT:    .cfi_restore s5
+; RV32IMZBS-NEXT:    .cfi_restore s6
+; RV32IMZBS-NEXT:    .cfi_restore s7
+; RV32IMZBS-NEXT:    .cfi_restore s8
+; RV32IMZBS-NEXT:    .cfi_restore s9
+; RV32IMZBS-NEXT:    .cfi_restore s10
+; RV32IMZBS-NEXT:    .cfi_restore s11
+; RV32IMZBS-NEXT:    addi sp, sp, 496
+; RV32IMZBS-NEXT:    .cfi_def_cfa_offset 0
+; RV32IMZBS-NEXT:    ret
+;
+; RV64IMZBS-LABEL: clmul_i96:
+; RV64IMZBS:       # %bb.0:
+; RV64IMZBS-NEXT:    addi sp, sp, -688
+; RV64IMZBS-NEXT:    .cfi_def_cfa_offset 688
+; RV64IMZBS-NEXT:    sd ra, 680(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s0, 672(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s1, 664(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s2, 656(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s3, 648(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s4, 640(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s5, 632(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s6, 624(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s7, 616(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s8, 608(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s9, 600(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s10, 592(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s11, 584(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    .cfi_offset ra, -8
+; RV64IMZBS-NEXT:    .cfi_offset s0, -16
+; RV64IMZBS-NEXT:    .cfi_offset s1, -24
+; RV64IMZBS-NEXT:    .cfi_offset s2, -32
+; RV64IMZBS-NEXT:    .cfi_offset s3, -40
+; RV64IMZBS-NEXT:    .cfi_offset s4, -48
+; RV64IMZBS-NEXT:    .cfi_offset s5, -56
+; RV64IMZBS-NEXT:    .cfi_offset s6, -64
+; RV64IMZBS-NEXT:    .cfi_offset s7, -72
+; RV64IMZBS-NEXT:    .cfi_offset s8, -80
+; RV64IMZBS-NEXT:    .cfi_offset s9, -88
+; RV64IMZBS-NEXT:    .cfi_offset s10, -96
+; RV64IMZBS-NEXT:    .cfi_offset s11, -104
+; RV64IMZBS-NEXT:    mv t0, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 1
+; RV64IMZBS-NEXT:    sd a3, 568(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a4, a2, 62
+; RV64IMZBS-NEXT:    slli a5, a2, 63
+; RV64IMZBS-NEXT:    srai t3, a4, 63
+; RV64IMZBS-NEXT:    srai s2, a5, 63
+; RV64IMZBS-NEXT:    and a4, t3, a3
+; RV64IMZBS-NEXT:    and a5, s2, a0
+; RV64IMZBS-NEXT:    xor a4, a5, a4
+; RV64IMZBS-NEXT:    slli a6, a0, 2
+; RV64IMZBS-NEXT:    sd a6, 536(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a5, a2, 61
+; RV64IMZBS-NEXT:    srai a7, a5, 63
+; RV64IMZBS-NEXT:    sd a7, 152(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a5, a2, 60
+; RV64IMZBS-NEXT:    slli a3, a0, 3
+; RV64IMZBS-NEXT:    sd a3, 576(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    srai t1, a5, 63
+; RV64IMZBS-NEXT:    sd t1, 136(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a5, a7, a6
+; RV64IMZBS-NEXT:    and a6, t1, a3
+; RV64IMZBS-NEXT:    xor a5, a5, a6
+; RV64IMZBS-NEXT:    slli a6, a2, 59
+; RV64IMZBS-NEXT:    slli a3, a0, 4
+; RV64IMZBS-NEXT:    sd a3, 552(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    srai a6, a6, 63
+; RV64IMZBS-NEXT:    sd a6, 248(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a6, a6, a3
+; RV64IMZBS-NEXT:    slli a7, a2, 58
+; RV64IMZBS-NEXT:    slli a3, a0, 5
+; RV64IMZBS-NEXT:    sd a3, 560(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    srai a7, a7, 63
+; RV64IMZBS-NEXT:    sd a7, 304(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a7, a7, a3
+; RV64IMZBS-NEXT:    slli t1, a2, 57
+; RV64IMZBS-NEXT:    slli a3, a0, 6
+; RV64IMZBS-NEXT:    sd a3, 544(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    srai t1, t1, 63
+; RV64IMZBS-NEXT:    sd t1, 128(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    xor a6, a6, a7
+; RV64IMZBS-NEXT:    and a7, t1, a3
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    xor a5, a6, a7
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    slli a3, a0, 7
+; RV64IMZBS-NEXT:    sd a3, 512(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a5, a2, 56
+; RV64IMZBS-NEXT:    srai t1, a5, 63
+; RV64IMZBS-NEXT:    sd t1, 64(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a5, a2, 55
+; RV64IMZBS-NEXT:    slli a6, a0, 8
+; RV64IMZBS-NEXT:    sd a6, 520(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    srai a7, a5, 63
+; RV64IMZBS-NEXT:    sd a7, 312(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a5, t1, a3
+; RV64IMZBS-NEXT:    and a6, a7, a6
+; RV64IMZBS-NEXT:    xor a5, a5, a6
+; RV64IMZBS-NEXT:    slli a6, a2, 54
+; RV64IMZBS-NEXT:    slli a3, a0, 9
+; RV64IMZBS-NEXT:    sd a3, 528(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    srai s3, a6, 63
+; RV64IMZBS-NEXT:    and a6, s3, a3
+; RV64IMZBS-NEXT:    slli a7, a2, 53
+; RV64IMZBS-NEXT:    slli a3, a0, 10
+; RV64IMZBS-NEXT:    sd a3, 504(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    srai s4, a7, 63
+; RV64IMZBS-NEXT:    xor a5, a5, a6
+; RV64IMZBS-NEXT:    and a6, s4, a3
+; RV64IMZBS-NEXT:    xor a5, a5, a6
+; RV64IMZBS-NEXT:    slli a6, a2, 52
+; RV64IMZBS-NEXT:    srai a7, a6, 63
+; RV64IMZBS-NEXT:    sd a7, 160(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a6, a2, 51
+; RV64IMZBS-NEXT:    srai t2, a6, 63
+; RV64IMZBS-NEXT:    sd t2, 88(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a6, a2, 50
+; RV64IMZBS-NEXT:    srai t1, a6, 63
+; RV64IMZBS-NEXT:    sd t1, 320(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 11
+; RV64IMZBS-NEXT:    sd a3, 472(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a6, a7, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 12
+; RV64IMZBS-NEXT:    sd a3, 488(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a7, t2, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 13
+; RV64IMZBS-NEXT:    sd a3, 496(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    xor a6, a6, a7
+; RV64IMZBS-NEXT:    and a7, t1, a3
+; RV64IMZBS-NEXT:    xor a6, a6, a7
+; RV64IMZBS-NEXT:    slli a7, a2, 49
+; RV64IMZBS-NEXT:    srai t2, a7, 63
+; RV64IMZBS-NEXT:    sd t2, 56(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a7, a2, 48
+; RV64IMZBS-NEXT:    srai t1, a7, 63
+; RV64IMZBS-NEXT:    sd t1, 184(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 14
+; RV64IMZBS-NEXT:    sd a3, 464(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a7, t2, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 15
+; RV64IMZBS-NEXT:    sd a3, 480(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    xor a6, a6, a7
+; RV64IMZBS-NEXT:    and a7, t1, a3
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    xor a5, a6, a7
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    slli a5, a2, 47
+; RV64IMZBS-NEXT:    slli a6, a2, 46
+; RV64IMZBS-NEXT:    srai a7, a5, 63
+; RV64IMZBS-NEXT:    sd a7, 264(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    srai a6, a6, 63
+; RV64IMZBS-NEXT:    sd a6, 256(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a5, a0, 16
+; RV64IMZBS-NEXT:    sd a5, 448(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 17
+; RV64IMZBS-NEXT:    sd a3, 456(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a5, a7, a5
+; RV64IMZBS-NEXT:    and a6, a6, a3
+; RV64IMZBS-NEXT:    xor a5, a5, a6
+; RV64IMZBS-NEXT:    slli a6, a2, 45
+; RV64IMZBS-NEXT:    srai a7, a6, 63
+; RV64IMZBS-NEXT:    sd a7, 232(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a6, a2, 44
+; RV64IMZBS-NEXT:    srai t1, a6, 63
+; RV64IMZBS-NEXT:    sd t1, 224(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 18
+; RV64IMZBS-NEXT:    sd a3, 432(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a6, a7, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 19
+; RV64IMZBS-NEXT:    sd a3, 440(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    xor a5, a5, a6
+; RV64IMZBS-NEXT:    and a6, t1, a3
+; RV64IMZBS-NEXT:    xor a5, a5, a6
+; RV64IMZBS-NEXT:    slli a6, a2, 43
+; RV64IMZBS-NEXT:    srai a7, a6, 63
+; RV64IMZBS-NEXT:    sd a7, 192(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a6, a2, 42
+; RV64IMZBS-NEXT:    srai t1, a6, 63
+; RV64IMZBS-NEXT:    sd t1, 176(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 20
+; RV64IMZBS-NEXT:    sd a3, 416(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a6, a7, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 21
+; RV64IMZBS-NEXT:    sd a3, 424(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    xor a5, a5, a6
+; RV64IMZBS-NEXT:    and a6, t1, a3
+; RV64IMZBS-NEXT:    xor a5, a5, a6
+; RV64IMZBS-NEXT:    slli a6, a2, 41
+; RV64IMZBS-NEXT:    srai a7, a6, 63
+; RV64IMZBS-NEXT:    sd a7, 0(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a6, a2, 40
+; RV64IMZBS-NEXT:    srai t1, a6, 63
+; RV64IMZBS-NEXT:    sd t1, 80(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a6, a2, 39
+; RV64IMZBS-NEXT:    srai t2, a6, 63
+; RV64IMZBS-NEXT:    sd t2, 72(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 22
+; RV64IMZBS-NEXT:    sd a3, 384(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a6, a7, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 23
+; RV64IMZBS-NEXT:    sd a3, 392(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a7, t1, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 24
+; RV64IMZBS-NEXT:    sd a3, 408(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    xor a6, a6, a7
+; RV64IMZBS-NEXT:    and a7, t2, a3
+; RV64IMZBS-NEXT:    xor a6, a6, a7
+; RV64IMZBS-NEXT:    slli a7, a2, 38
+; RV64IMZBS-NEXT:    srai t1, a7, 63
+; RV64IMZBS-NEXT:    sd t1, 48(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a7, a2, 37
+; RV64IMZBS-NEXT:    srai t2, a7, 63
+; RV64IMZBS-NEXT:    sd t2, 40(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 25
+; RV64IMZBS-NEXT:    sd a3, 368(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a7, t1, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 26
+; RV64IMZBS-NEXT:    sd a3, 400(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    xor a6, a6, a7
+; RV64IMZBS-NEXT:    and a7, t2, a3
+; RV64IMZBS-NEXT:    xor a6, a6, a7
+; RV64IMZBS-NEXT:    slli a7, a2, 36
+; RV64IMZBS-NEXT:    srai t1, a7, 63
+; RV64IMZBS-NEXT:    sd t1, 24(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a7, a2, 35
+; RV64IMZBS-NEXT:    srai t2, a7, 63
+; RV64IMZBS-NEXT:    sd t2, 16(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 27
+; RV64IMZBS-NEXT:    sd a3, 352(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a7, t1, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 28
+; RV64IMZBS-NEXT:    sd a3, 376(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    xor a6, a6, a7
+; RV64IMZBS-NEXT:    and a7, t2, a3
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    xor a5, a6, a7
+; RV64IMZBS-NEXT:    xor t1, a4, a5
+; RV64IMZBS-NEXT:    slli a4, a2, 34
+; RV64IMZBS-NEXT:    srai a5, a4, 63
+; RV64IMZBS-NEXT:    sd a5, 8(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a4, a2, 33
+; RV64IMZBS-NEXT:    srai a6, a4, 63
+; RV64IMZBS-NEXT:    sd a6, 144(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 29
+; RV64IMZBS-NEXT:    sd a3, 344(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a4, a5, a3
+; RV64IMZBS-NEXT:    slli a3, a0, 30
+; RV64IMZBS-NEXT:    sd a3, 360(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a5, a6, a3
+; RV64IMZBS-NEXT:    slli a6, a2, 31
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    srai a5, a6, 63
+; RV64IMZBS-NEXT:    sd a5, 32(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 31
+; RV64IMZBS-NEXT:    sd a3, 336(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sraiw t2, a2, 31
+; RV64IMZBS-NEXT:    and a6, t2, a3
+; RV64IMZBS-NEXT:    slli a7, a0, 32
+; RV64IMZBS-NEXT:    xor a4, a4, a6
+; RV64IMZBS-NEXT:    and a6, a5, a7
+; RV64IMZBS-NEXT:    xor a4, a4, a6
+; RV64IMZBS-NEXT:    slli a6, a2, 30
+; RV64IMZBS-NEXT:    srai t4, a6, 63
+; RV64IMZBS-NEXT:    slli a6, a0, 33
+; RV64IMZBS-NEXT:    and a6, t4, a6
+; RV64IMZBS-NEXT:    slli a7, a2, 29
+; RV64IMZBS-NEXT:    srai s9, a7, 63
+; RV64IMZBS-NEXT:    slli a7, a0, 34
+; RV64IMZBS-NEXT:    xor a4, a4, a6
+; RV64IMZBS-NEXT:    and a6, s9, a7
+; RV64IMZBS-NEXT:    xor a4, a4, a6
+; RV64IMZBS-NEXT:    slli a6, a2, 28
+; RV64IMZBS-NEXT:    srai s6, a6, 63
+; RV64IMZBS-NEXT:    slli a7, a0, 35
+; RV64IMZBS-NEXT:    and t5, s6, a7
+; RV64IMZBS-NEXT:    slli a7, a2, 27
+; RV64IMZBS-NEXT:    srai s7, a7, 63
+; RV64IMZBS-NEXT:    slli t6, a0, 36
+; RV64IMZBS-NEXT:    xor a4, a4, t5
+; RV64IMZBS-NEXT:    and t5, s7, t6
+; RV64IMZBS-NEXT:    xor a4, a4, t5
+; RV64IMZBS-NEXT:    slli t5, a2, 26
+; RV64IMZBS-NEXT:    srai a3, t5, 63
+; RV64IMZBS-NEXT:    sd a3, 240(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli t5, a0, 37
+; RV64IMZBS-NEXT:    and t6, a3, t5
+; RV64IMZBS-NEXT:    slli t5, a2, 25
+; RV64IMZBS-NEXT:    srai s8, t5, 63
+; RV64IMZBS-NEXT:    slli s0, a0, 38
+; RV64IMZBS-NEXT:    and s0, s8, s0
+; RV64IMZBS-NEXT:    slli s1, a2, 24
+; RV64IMZBS-NEXT:    srai a3, s1, 63
+; RV64IMZBS-NEXT:    sd a3, 216(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli s1, a0, 39
+; RV64IMZBS-NEXT:    xor t6, t6, s0
+; RV64IMZBS-NEXT:    and s0, a3, s1
+; RV64IMZBS-NEXT:    xor s0, t6, s0
+; RV64IMZBS-NEXT:    slli t6, a2, 23
+; RV64IMZBS-NEXT:    srai t6, t6, 63
+; RV64IMZBS-NEXT:    slli s1, a0, 40
+; RV64IMZBS-NEXT:    and s1, t6, s1
+; RV64IMZBS-NEXT:    slli s5, a2, 22
+; RV64IMZBS-NEXT:    srai a3, s5, 63
+; RV64IMZBS-NEXT:    sd a3, 208(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli s5, a0, 41
+; RV64IMZBS-NEXT:    xor s0, s0, s1
+; RV64IMZBS-NEXT:    and s1, a3, s5
+; RV64IMZBS-NEXT:    xor s1, s0, s1
+; RV64IMZBS-NEXT:    slli s0, a2, 21
+; RV64IMZBS-NEXT:    srai s0, s0, 63
+; RV64IMZBS-NEXT:    slli s5, a0, 42
+; RV64IMZBS-NEXT:    and s5, s0, s5
+; RV64IMZBS-NEXT:    slli s10, a2, 20
+; RV64IMZBS-NEXT:    srai a3, s10, 63
+; RV64IMZBS-NEXT:    sd a3, 200(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli s10, a0, 43
+; RV64IMZBS-NEXT:    xor s1, s1, s5
+; RV64IMZBS-NEXT:    and s5, a3, s10
+; RV64IMZBS-NEXT:    xor s10, s1, s5
+; RV64IMZBS-NEXT:    slli s1, a2, 19
+; RV64IMZBS-NEXT:    srai s1, s1, 63
+; RV64IMZBS-NEXT:    slli s5, a0, 44
+; RV64IMZBS-NEXT:    and s11, s1, s5
+; RV64IMZBS-NEXT:    slli s5, a2, 18
+; RV64IMZBS-NEXT:    srai s5, s5, 63
+; RV64IMZBS-NEXT:    slli ra, a0, 45
+; RV64IMZBS-NEXT:    xor s10, s10, s11
+; RV64IMZBS-NEXT:    and s11, s5, ra
+; RV64IMZBS-NEXT:    xor t1, t1, a4
+; RV64IMZBS-NEXT:    xor a6, s10, s11
+; RV64IMZBS-NEXT:    slli s10, a2, 17
+; RV64IMZBS-NEXT:    slli s11, a2, 16
+; RV64IMZBS-NEXT:    srai a3, s10, 63
+; RV64IMZBS-NEXT:    sd a3, 120(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    srai a4, s11, 63
+; RV64IMZBS-NEXT:    sd a4, 112(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli s10, a0, 46
+; RV64IMZBS-NEXT:    slli s11, a0, 47
+; RV64IMZBS-NEXT:    and s10, a3, s10
+; RV64IMZBS-NEXT:    and s11, a4, s11
+; RV64IMZBS-NEXT:    xor s11, s10, s11
+; RV64IMZBS-NEXT:    slli s10, a2, 15
+; RV64IMZBS-NEXT:    srai a3, s10, 63
+; RV64IMZBS-NEXT:    sd a3, 296(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli s10, a0, 48
+; RV64IMZBS-NEXT:    and ra, a3, s10
+; RV64IMZBS-NEXT:    slli s10, a2, 14
+; RV64IMZBS-NEXT:    srai a4, s10, 63
+; RV64IMZBS-NEXT:    sd a4, 104(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 49
+; RV64IMZBS-NEXT:    xor s11, s11, ra
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a4, s11, a3
+; RV64IMZBS-NEXT:    slli s11, a2, 13
+; RV64IMZBS-NEXT:    srai a3, s11, 63
+; RV64IMZBS-NEXT:    sd a3, 288(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli s11, a0, 50
+; RV64IMZBS-NEXT:    and ra, a3, s11
+; RV64IMZBS-NEXT:    slli s11, a2, 12
+; RV64IMZBS-NEXT:    srai s11, s11, 63
+; RV64IMZBS-NEXT:    slli a3, a0, 51
+; RV64IMZBS-NEXT:    xor a4, a4, ra
+; RV64IMZBS-NEXT:    and a3, s11, a3
+; RV64IMZBS-NEXT:    xor a5, a4, a3
+; RV64IMZBS-NEXT:    slli a4, a2, 11
+; RV64IMZBS-NEXT:    srai a3, a4, 63
+; RV64IMZBS-NEXT:    sd a3, 280(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a4, a0, 52
+; RV64IMZBS-NEXT:    and a4, a3, a4
+; RV64IMZBS-NEXT:    slli ra, a2, 10
+; RV64IMZBS-NEXT:    srai ra, ra, 63
+; RV64IMZBS-NEXT:    slli a3, a0, 53
+; RV64IMZBS-NEXT:    xor a4, a5, a4
+; RV64IMZBS-NEXT:    and a3, ra, a3
+; RV64IMZBS-NEXT:    xor a3, a4, a3
+; RV64IMZBS-NEXT:    slli a4, a2, 9
+; RV64IMZBS-NEXT:    srai a5, a4, 63
+; RV64IMZBS-NEXT:    sd a5, 272(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a4, a0, 54
+; RV64IMZBS-NEXT:    and a4, a5, a4
+; RV64IMZBS-NEXT:    slli a5, a2, 8
+; RV64IMZBS-NEXT:    srai a7, a5, 63
+; RV64IMZBS-NEXT:    sd a7, 96(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a5, a0, 55
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    and a4, a7, a5
+; RV64IMZBS-NEXT:    xor a5, t1, a6
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    xor a3, a5, a3
+; RV64IMZBS-NEXT:    sd a3, 328(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a2, 7
+; RV64IMZBS-NEXT:    slli a4, a2, 6
+; RV64IMZBS-NEXT:    srai t5, a3, 63
+; RV64IMZBS-NEXT:    srai a5, a4, 63
+; RV64IMZBS-NEXT:    sd a5, 168(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    slli a3, a0, 56
+; RV64IMZBS-NEXT:    slli a4, a0, 57
+; RV64IMZBS-NEXT:    and a3, t5, a3
+; RV64IMZBS-NEXT:    and a4, a5, a4
+; RV64IMZBS-NEXT:    srli a5, a0, 63
+; RV64IMZBS-NEXT:    slli a6, a1, 1
+; RV64IMZBS-NEXT:    xor a4, a3, a4
+; RV64IMZBS-NEXT:    or a3, a6, a5
+; RV64IMZBS-NEXT:    and a5, t3, a3
+; RV64IMZBS-NEXT:    slli a3, a2, 5
+; RV64IMZBS-NEXT:    srai t1, a3, 63
+; RV64IMZBS-NEXT:    slli a6, a0, 58
+; RV64IMZBS-NEXT:    and a6, t1, a6
+; RV64IMZBS-NEXT:    and t3, s2, a1
+; RV64IMZBS-NEXT:    xor a7, a4, a6
+; RV64IMZBS-NEXT:    xor a5, t3, a5
+; RV64IMZBS-NEXT:    srli a6, a0, 62
+; RV64IMZBS-NEXT:    slli t3, a1, 2
+; RV64IMZBS-NEXT:    srli s2, a0, 61
+; RV64IMZBS-NEXT:    slli a3, a1, 3
+; RV64IMZBS-NEXT:    or a6, t3, a6
+; RV64IMZBS-NEXT:    or a3, a3, s2
+; RV64IMZBS-NEXT:    ld a4, 152(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a6, a4, a6
+; RV64IMZBS-NEXT:    ld a4, 136(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    srli t3, a0, 60
+; RV64IMZBS-NEXT:    slli s2, a1, 4
+; RV64IMZBS-NEXT:    xor a4, a6, a3
+; RV64IMZBS-NEXT:    or a6, s2, t3
+; RV64IMZBS-NEXT:    srli t3, a0, 59
+; RV64IMZBS-NEXT:    slli s2, a1, 5
+; RV64IMZBS-NEXT:    ld a3, 248(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a6, a3, a6
+; RV64IMZBS-NEXT:    or t3, s2, t3
+; RV64IMZBS-NEXT:    srli s2, a0, 58
+; RV64IMZBS-NEXT:    slli a3, a1, 6
+; RV64IMZBS-NEXT:    ld s10, 304(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and t3, s10, t3
+; RV64IMZBS-NEXT:    or a3, a3, s2
+; RV64IMZBS-NEXT:    xor a6, a6, t3
+; RV64IMZBS-NEXT:    ld t3, 128(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, t3, a3
+; RV64IMZBS-NEXT:    xor a4, a5, a4
+; RV64IMZBS-NEXT:    xor a3, a6, a3
+; RV64IMZBS-NEXT:    srli a5, a0, 57
+; RV64IMZBS-NEXT:    slli a6, a1, 7
+; RV64IMZBS-NEXT:    xor a3, a4, a3
+; RV64IMZBS-NEXT:    or a4, a6, a5
+; RV64IMZBS-NEXT:    srli a5, a0, 56
+; RV64IMZBS-NEXT:    slli a6, a1, 8
+; RV64IMZBS-NEXT:    ld t3, 64(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, t3, a4
+; RV64IMZBS-NEXT:    or a5, a6, a5
+; RV64IMZBS-NEXT:    srli a6, a0, 55
+; RV64IMZBS-NEXT:    slli t3, a1, 9
+; RV64IMZBS-NEXT:    ld s2, 312(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a5, s2, a5
+; RV64IMZBS-NEXT:    or a6, t3, a6
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    and a5, s3, a6
+; RV64IMZBS-NEXT:    srli a6, a0, 54
+; RV64IMZBS-NEXT:    slli t3, a1, 10
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    or a5, t3, a6
+; RV64IMZBS-NEXT:    and a5, s4, a5
+; RV64IMZBS-NEXT:    slli a6, a2, 4
+; RV64IMZBS-NEXT:    srai t3, a6, 63
+; RV64IMZBS-NEXT:    slli a6, a0, 59
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    and a5, t3, a6
+; RV64IMZBS-NEXT:    xor a5, a7, a5
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    srli a4, a0, 53
+; RV64IMZBS-NEXT:    slli a6, a1, 11
+; RV64IMZBS-NEXT:    srli a7, a0, 52
+; RV64IMZBS-NEXT:    slli s2, a1, 12
+; RV64IMZBS-NEXT:    or a4, a6, a4
+; RV64IMZBS-NEXT:    or a6, s2, a7
+; RV64IMZBS-NEXT:    ld a7, 160(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, a7, a4
+; RV64IMZBS-NEXT:    ld a7, 88(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a6, a7, a6
+; RV64IMZBS-NEXT:    srli a7, a0, 51
+; RV64IMZBS-NEXT:    slli s2, a1, 13
+; RV64IMZBS-NEXT:    xor a4, a4, a6
+; RV64IMZBS-NEXT:    or a6, s2, a7
+; RV64IMZBS-NEXT:    srli a7, a0, 50
+; RV64IMZBS-NEXT:    slli s2, a1, 14
+; RV64IMZBS-NEXT:    ld s3, 320(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a6, s3, a6
+; RV64IMZBS-NEXT:    or a7, s2, a7
+; RV64IMZBS-NEXT:    xor a4, a4, a6
+; RV64IMZBS-NEXT:    ld a6, 56(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a6, a6, a7
+; RV64IMZBS-NEXT:    srli a7, a0, 49
+; RV64IMZBS-NEXT:    slli s2, a1, 15
+; RV64IMZBS-NEXT:    xor a4, a4, a6
+; RV64IMZBS-NEXT:    or a6, s2, a7
+; RV64IMZBS-NEXT:    ld a7, 184(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a6, a7, a6
+; RV64IMZBS-NEXT:    slli a7, a2, 3
+; RV64IMZBS-NEXT:    srai s2, a7, 63
+; RV64IMZBS-NEXT:    slli a7, a0, 60
+; RV64IMZBS-NEXT:    xor a4, a4, a6
+; RV64IMZBS-NEXT:    and a6, s2, a7
+; RV64IMZBS-NEXT:    xor s3, a5, a6
+; RV64IMZBS-NEXT:    xor s4, a3, a4
+; RV64IMZBS-NEXT:    srli a3, a0, 48
+; RV64IMZBS-NEXT:    slli a4, a1, 16
+; RV64IMZBS-NEXT:    srli a5, a0, 47
+; RV64IMZBS-NEXT:    slli a6, a1, 17
+; RV64IMZBS-NEXT:    or a3, a4, a3
+; RV64IMZBS-NEXT:    or a4, a6, a5
+; RV64IMZBS-NEXT:    ld a5, 264(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a5, a3
+; RV64IMZBS-NEXT:    ld a5, 256(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, a5, a4
+; RV64IMZBS-NEXT:    srli a5, a0, 46
+; RV64IMZBS-NEXT:    slli a6, a1, 18
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    or a4, a6, a5
+; RV64IMZBS-NEXT:    srli a5, a0, 45
+; RV64IMZBS-NEXT:    slli a6, a1, 19
+; RV64IMZBS-NEXT:    ld a7, 232(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, a7, a4
+; RV64IMZBS-NEXT:    or a5, a6, a5
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    ld a4, 224(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, a4, a5
+; RV64IMZBS-NEXT:    srli a5, a0, 44
+; RV64IMZBS-NEXT:    slli a6, a1, 20
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    or a4, a6, a5
+; RV64IMZBS-NEXT:    srli a5, a0, 43
+; RV64IMZBS-NEXT:    slli a6, a1, 21
+; RV64IMZBS-NEXT:    ld a7, 192(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, a7, a4
+; RV64IMZBS-NEXT:    or a5, a6, a5
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    ld a4, 176(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, a4, a5
+; RV64IMZBS-NEXT:    srli a5, a0, 42
+; RV64IMZBS-NEXT:    slli a6, a1, 22
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    or a4, a6, a5
+; RV64IMZBS-NEXT:    srli a5, a0, 41
+; RV64IMZBS-NEXT:    slli a6, a1, 23
+; RV64IMZBS-NEXT:    ld a7, 0(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, a7, a4
+; RV64IMZBS-NEXT:    or a5, a6, a5
+; RV64IMZBS-NEXT:    srli a6, a0, 40
+; RV64IMZBS-NEXT:    slli a7, a1, 24
+; RV64IMZBS-NEXT:    ld s10, 80(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a5, s10, a5
+; RV64IMZBS-NEXT:    or a6, a7, a6
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    ld a5, 72(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a5, a5, a6
+; RV64IMZBS-NEXT:    srli a6, a0, 39
+; RV64IMZBS-NEXT:    slli a7, a1, 25
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    or a5, a7, a6
+; RV64IMZBS-NEXT:    srli a6, a0, 38
+; RV64IMZBS-NEXT:    slli a7, a1, 26
+; RV64IMZBS-NEXT:    ld s10, 48(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a5, s10, a5
+; RV64IMZBS-NEXT:    or a6, a7, a6
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    ld a5, 40(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a5, a5, a6
+; RV64IMZBS-NEXT:    srli a6, a0, 37
+; RV64IMZBS-NEXT:    slli a7, a1, 27
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    or a5, a7, a6
+; RV64IMZBS-NEXT:    srli a6, a0, 36
+; RV64IMZBS-NEXT:    slli a7, a1, 28
+; RV64IMZBS-NEXT:    ld s10, 24(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a5, s10, a5
+; RV64IMZBS-NEXT:    or a6, a7, a6
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    ld a5, 16(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a5, a5, a6
+; RV64IMZBS-NEXT:    xor a3, s4, a3
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    srli a5, a0, 35
+; RV64IMZBS-NEXT:    slli a6, a1, 29
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    or a4, a6, a5
+; RV64IMZBS-NEXT:    srli a5, a0, 34
+; RV64IMZBS-NEXT:    slli a6, a1, 30
+; RV64IMZBS-NEXT:    ld a7, 8(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, a7, a4
+; RV64IMZBS-NEXT:    or a5, a6, a5
+; RV64IMZBS-NEXT:    srli a6, a0, 33
+; RV64IMZBS-NEXT:    slli a7, a1, 31
+; RV64IMZBS-NEXT:    ld s4, 144(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a5, s4, a5
+; RV64IMZBS-NEXT:    or a6, a7, a6
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    and a5, t2, a6
+; RV64IMZBS-NEXT:    srli a6, a0, 32
+; RV64IMZBS-NEXT:    slli a7, a1, 32
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    or a5, a7, a6
+; RV64IMZBS-NEXT:    srli a6, a0, 31
+; RV64IMZBS-NEXT:    slli a7, a1, 33
+; RV64IMZBS-NEXT:    ld t2, 32(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a5, t2, a5
+; RV64IMZBS-NEXT:    or a6, a7, a6
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    and a5, t4, a6
+; RV64IMZBS-NEXT:    srli a6, a0, 30
+; RV64IMZBS-NEXT:    slli a7, a1, 34
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    or a5, a7, a6
+; RV64IMZBS-NEXT:    srli a6, a0, 29
+; RV64IMZBS-NEXT:    slli a7, a1, 35
+; RV64IMZBS-NEXT:    and a5, s9, a5
+; RV64IMZBS-NEXT:    or a6, a7, a6
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    and a5, s6, a6
+; RV64IMZBS-NEXT:    srli a6, a0, 28
+; RV64IMZBS-NEXT:    slli a7, a1, 36
+; RV64IMZBS-NEXT:    xor a5, a4, a5
+; RV64IMZBS-NEXT:    or a4, a7, a6
+; RV64IMZBS-NEXT:    and a6, s7, a4
+; RV64IMZBS-NEXT:    slli a4, a2, 2
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    slli a7, a0, 61
+; RV64IMZBS-NEXT:    xor a6, a5, a6
+; RV64IMZBS-NEXT:    and a5, a4, a7
+; RV64IMZBS-NEXT:    xor a5, s3, a5
+; RV64IMZBS-NEXT:    xor a6, a3, a6
+; RV64IMZBS-NEXT:    srli a3, a0, 27
+; RV64IMZBS-NEXT:    slli a7, a1, 37
+; RV64IMZBS-NEXT:    srli t2, a0, 26
+; RV64IMZBS-NEXT:    slli t4, a1, 38
+; RV64IMZBS-NEXT:    or a3, a7, a3
+; RV64IMZBS-NEXT:    or a7, t4, t2
+; RV64IMZBS-NEXT:    ld t2, 240(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, t2, a3
+; RV64IMZBS-NEXT:    and a7, s8, a7
+; RV64IMZBS-NEXT:    srli t2, a0, 25
+; RV64IMZBS-NEXT:    slli t4, a1, 39
+; RV64IMZBS-NEXT:    xor a3, a3, a7
+; RV64IMZBS-NEXT:    or a7, t4, t2
+; RV64IMZBS-NEXT:    srli t2, a0, 24
+; RV64IMZBS-NEXT:    slli t4, a1, 40
+; RV64IMZBS-NEXT:    ld s3, 216(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a7, s3, a7
+; RV64IMZBS-NEXT:    or t2, t4, t2
+; RV64IMZBS-NEXT:    xor a3, a3, a7
+; RV64IMZBS-NEXT:    and a7, t6, t2
+; RV64IMZBS-NEXT:    srli t2, a0, 23
+; RV64IMZBS-NEXT:    slli t4, a1, 41
+; RV64IMZBS-NEXT:    xor a3, a3, a7
+; RV64IMZBS-NEXT:    or a7, t4, t2
+; RV64IMZBS-NEXT:    srli t2, a0, 22
+; RV64IMZBS-NEXT:    slli t4, a1, 42
+; RV64IMZBS-NEXT:    ld t6, 208(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a7, t6, a7
+; RV64IMZBS-NEXT:    or t2, t4, t2
+; RV64IMZBS-NEXT:    xor a3, a3, a7
+; RV64IMZBS-NEXT:    and a7, s0, t2
+; RV64IMZBS-NEXT:    srli t2, a0, 21
+; RV64IMZBS-NEXT:    slli t4, a1, 43
+; RV64IMZBS-NEXT:    xor a3, a3, a7
+; RV64IMZBS-NEXT:    or a7, t4, t2
+; RV64IMZBS-NEXT:    srli t2, a0, 20
+; RV64IMZBS-NEXT:    slli t4, a1, 44
+; RV64IMZBS-NEXT:    ld t6, 200(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a7, t6, a7
+; RV64IMZBS-NEXT:    or t2, t4, t2
+; RV64IMZBS-NEXT:    xor a3, a3, a7
+; RV64IMZBS-NEXT:    and a7, s1, t2
+; RV64IMZBS-NEXT:    srli t2, a0, 19
+; RV64IMZBS-NEXT:    slli t4, a1, 45
+; RV64IMZBS-NEXT:    xor a3, a3, a7
+; RV64IMZBS-NEXT:    or a7, t4, t2
+; RV64IMZBS-NEXT:    and t2, s5, a7
+; RV64IMZBS-NEXT:    slli a7, a2, 1
+; RV64IMZBS-NEXT:    srai a7, a7, 63
+; RV64IMZBS-NEXT:    slli t4, a0, 62
+; RV64IMZBS-NEXT:    xor a3, a3, t2
+; RV64IMZBS-NEXT:    and t2, a7, t4
+; RV64IMZBS-NEXT:    xor a5, a5, t2
+; RV64IMZBS-NEXT:    xor a6, a6, a3
+; RV64IMZBS-NEXT:    srli a3, a0, 18
+; RV64IMZBS-NEXT:    slli t2, a1, 46
+; RV64IMZBS-NEXT:    srli t4, a0, 17
+; RV64IMZBS-NEXT:    slli t6, a1, 47
+; RV64IMZBS-NEXT:    or a3, t2, a3
+; RV64IMZBS-NEXT:    or t2, t6, t4
+; RV64IMZBS-NEXT:    ld t4, 120(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, t4, a3
+; RV64IMZBS-NEXT:    ld t4, 112(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and t2, t4, t2
+; RV64IMZBS-NEXT:    srli t4, a0, 16
+; RV64IMZBS-NEXT:    slli t6, a1, 48
+; RV64IMZBS-NEXT:    xor a3, a3, t2
+; RV64IMZBS-NEXT:    or t2, t6, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 15
+; RV64IMZBS-NEXT:    slli t6, a1, 49
+; RV64IMZBS-NEXT:    ld s0, 296(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and t2, s0, t2
+; RV64IMZBS-NEXT:    or t4, t6, t4
+; RV64IMZBS-NEXT:    xor a3, a3, t2
+; RV64IMZBS-NEXT:    ld t2, 104(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and t2, t2, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 14
+; RV64IMZBS-NEXT:    slli t6, a1, 50
+; RV64IMZBS-NEXT:    xor a3, a3, t2
+; RV64IMZBS-NEXT:    or t2, t6, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 13
+; RV64IMZBS-NEXT:    slli t6, a1, 51
+; RV64IMZBS-NEXT:    ld s0, 288(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and t2, s0, t2
+; RV64IMZBS-NEXT:    or t4, t6, t4
+; RV64IMZBS-NEXT:    xor a3, a3, t2
+; RV64IMZBS-NEXT:    and t2, s11, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 12
+; RV64IMZBS-NEXT:    slli t6, a1, 52
+; RV64IMZBS-NEXT:    xor a3, a3, t2
+; RV64IMZBS-NEXT:    or t2, t6, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 11
+; RV64IMZBS-NEXT:    slli t6, a1, 53
+; RV64IMZBS-NEXT:    ld s0, 280(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and t2, s0, t2
+; RV64IMZBS-NEXT:    or t4, t6, t4
+; RV64IMZBS-NEXT:    xor a3, a3, t2
+; RV64IMZBS-NEXT:    and t2, ra, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 10
+; RV64IMZBS-NEXT:    slli t6, a1, 54
+; RV64IMZBS-NEXT:    xor a3, a3, t2
+; RV64IMZBS-NEXT:    or t2, t6, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 9
+; RV64IMZBS-NEXT:    slli t6, a1, 55
+; RV64IMZBS-NEXT:    ld s0, 272(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and t2, s0, t2
+; RV64IMZBS-NEXT:    or t4, t6, t4
+; RV64IMZBS-NEXT:    xor a3, a3, t2
+; RV64IMZBS-NEXT:    ld t2, 96(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and t2, t2, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 8
+; RV64IMZBS-NEXT:    slli t6, a1, 56
+; RV64IMZBS-NEXT:    xor t2, a3, t2
+; RV64IMZBS-NEXT:    or a3, t6, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 7
+; RV64IMZBS-NEXT:    slli t6, a1, 57
+; RV64IMZBS-NEXT:    and a3, t5, a3
+; RV64IMZBS-NEXT:    or t4, t6, t4
+; RV64IMZBS-NEXT:    srli t5, a0, 6
+; RV64IMZBS-NEXT:    slli t6, a1, 58
+; RV64IMZBS-NEXT:    ld s0, 168(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and t4, s0, t4
+; RV64IMZBS-NEXT:    or t5, t6, t5
+; RV64IMZBS-NEXT:    xor a3, a3, t4
+; RV64IMZBS-NEXT:    and t1, t1, t5
+; RV64IMZBS-NEXT:    srli t4, a0, 5
+; RV64IMZBS-NEXT:    slli t5, a1, 59
+; RV64IMZBS-NEXT:    xor a3, a3, t1
+; RV64IMZBS-NEXT:    or t1, t5, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 4
+; RV64IMZBS-NEXT:    slli t5, a1, 60
+; RV64IMZBS-NEXT:    and t1, t3, t1
+; RV64IMZBS-NEXT:    or t3, t5, t4
+; RV64IMZBS-NEXT:    xor a3, a3, t1
+; RV64IMZBS-NEXT:    and t1, s2, t3
+; RV64IMZBS-NEXT:    srli t3, a0, 3
+; RV64IMZBS-NEXT:    slli t4, a1, 61
+; RV64IMZBS-NEXT:    xor a3, a3, t1
+; RV64IMZBS-NEXT:    or t1, t4, t3
+; RV64IMZBS-NEXT:    srli t3, a0, 2
+; RV64IMZBS-NEXT:    slli t4, a1, 62
+; RV64IMZBS-NEXT:    and a4, a4, t1
+; RV64IMZBS-NEXT:    or t1, t4, t3
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    and a4, a7, t1
+; RV64IMZBS-NEXT:    slli a1, a1, 63
+; RV64IMZBS-NEXT:    srli a7, a0, 1
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    or a1, a1, a7
+; RV64IMZBS-NEXT:    srai a2, a2, 63
+; RV64IMZBS-NEXT:    slli a4, a0, 63
+; RV64IMZBS-NEXT:    and a4, a2, a4
+; RV64IMZBS-NEXT:    slli a7, t0, 63
+; RV64IMZBS-NEXT:    and a1, a2, a1
+; RV64IMZBS-NEXT:    srai a2, a7, 63
+; RV64IMZBS-NEXT:    xor a1, a3, a1
+; RV64IMZBS-NEXT:    and a0, a2, a0
+; RV64IMZBS-NEXT:    xor a0, a1, a0
+; RV64IMZBS-NEXT:    slli a1, t0, 62
+; RV64IMZBS-NEXT:    srai a1, a1, 63
+; RV64IMZBS-NEXT:    slli a2, t0, 61
+; RV64IMZBS-NEXT:    ld a3, 568(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a1, a1, a3
+; RV64IMZBS-NEXT:    srai a2, a2, 63
+; RV64IMZBS-NEXT:    xor a0, a0, a1
+; RV64IMZBS-NEXT:    ld a1, 536(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a1, a2, a1
+; RV64IMZBS-NEXT:    xor a2, a6, t2
+; RV64IMZBS-NEXT:    xor a1, a0, a1
+; RV64IMZBS-NEXT:    xor a0, a5, a4
+; RV64IMZBS-NEXT:    xor a1, a2, a1
+; RV64IMZBS-NEXT:    slli a2, t0, 60
+; RV64IMZBS-NEXT:    slli a3, t0, 59
+; RV64IMZBS-NEXT:    srai a2, a2, 63
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    ld a4, 576(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a2, a2, a4
+; RV64IMZBS-NEXT:    ld a4, 552(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a4
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 58
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 57
+; RV64IMZBS-NEXT:    ld a5, 560(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 544(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 56
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 55
+; RV64IMZBS-NEXT:    ld a5, 512(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 520(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 54
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 53
+; RV64IMZBS-NEXT:    ld a5, 528(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 504(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 52
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 51
+; RV64IMZBS-NEXT:    ld a5, 472(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 488(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 50
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 49
+; RV64IMZBS-NEXT:    ld a5, 496(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 464(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 48
+; RV64IMZBS-NEXT:    xor a1, a1, a2
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    ld a2, 480(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a2, a3, a2
+; RV64IMZBS-NEXT:    slli a3, t0, 47
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 46
+; RV64IMZBS-NEXT:    ld a5, 448(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 456(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 45
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 44
+; RV64IMZBS-NEXT:    ld a5, 432(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 440(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 43
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 42
+; RV64IMZBS-NEXT:    ld a5, 416(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 424(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 41
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 40
+; RV64IMZBS-NEXT:    ld a5, 384(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 392(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 39
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 38
+; RV64IMZBS-NEXT:    ld a5, 408(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 368(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    slli a3, t0, 37
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    slli a4, t0, 36
+; RV64IMZBS-NEXT:    ld a5, 400(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    ld a3, 352(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a4, a3
+; RV64IMZBS-NEXT:    slli a4, t0, 35
+; RV64IMZBS-NEXT:    slli a5, t0, 34
+; RV64IMZBS-NEXT:    srai a4, a4, 63
+; RV64IMZBS-NEXT:    srai a5, a5, 63
+; RV64IMZBS-NEXT:    ld a6, 376(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, a4, a6
+; RV64IMZBS-NEXT:    ld a6, 344(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a5, a5, a6
+; RV64IMZBS-NEXT:    xor a2, a2, a3
+; RV64IMZBS-NEXT:    xor a4, a4, a5
+; RV64IMZBS-NEXT:    slli a3, t0, 33
+; RV64IMZBS-NEXT:    sraiw a5, t0, 31
+; RV64IMZBS-NEXT:    srai a3, a3, 63
+; RV64IMZBS-NEXT:    seqz a5, a5
+; RV64IMZBS-NEXT:    ld a6, 360(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a3, a3, a6
+; RV64IMZBS-NEXT:    addi a5, a5, -1
+; RV64IMZBS-NEXT:    xor a3, a4, a3
+; RV64IMZBS-NEXT:    ld a4, 336(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, a5, a4
+; RV64IMZBS-NEXT:    xor a1, a1, a2
+; RV64IMZBS-NEXT:    xor a3, a3, a4
+; RV64IMZBS-NEXT:    ld a2, 328(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    xor a0, a2, a0
+; RV64IMZBS-NEXT:    xor a1, a1, a3
+; RV64IMZBS-NEXT:    ld ra, 680(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s0, 672(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s1, 664(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s2, 656(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s3, 648(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s4, 640(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s5, 632(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s6, 624(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s7, 616(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s8, 608(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s9, 600(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s10, 592(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s11, 584(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    .cfi_restore ra
+; RV64IMZBS-NEXT:    .cfi_restore s0
+; RV64IMZBS-NEXT:    .cfi_restore s1
+; RV64IMZBS-NEXT:    .cfi_restore s2
+; RV64IMZBS-NEXT:    .cfi_restore s3
+; RV64IMZBS-NEXT:    .cfi_restore s4
+; RV64IMZBS-NEXT:    .cfi_restore s5
+; RV64IMZBS-NEXT:    .cfi_restore s6
+; RV64IMZBS-NEXT:    .cfi_restore s7
+; RV64IMZBS-NEXT:    .cfi_restore s8
+; RV64IMZBS-NEXT:    .cfi_restore s9
+; RV64IMZBS-NEXT:    .cfi_restore s10
+; RV64IMZBS-NEXT:    .cfi_restore s11
+; RV64IMZBS-NEXT:    addi sp, sp, 688
+; RV64IMZBS-NEXT:    .cfi_def_cfa_offset 0
+; RV64IMZBS-NEXT:    ret
+;
+; RV32IMZBC-LABEL: clmul_i96:
+; RV32IMZBC:       # %bb.0:
+; RV32IMZBC-NEXT:    addi sp, sp, -16
+; RV32IMZBC-NEXT:    .cfi_def_cfa_offset 16
+; RV32IMZBC-NEXT:    sw s0, 12(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s1, 8(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s2, 4(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    .cfi_offset s0, -4
+; RV32IMZBC-NEXT:    .cfi_offset s1, -8
+; RV32IMZBC-NEXT:    .cfi_offset s2, -12
+; RV32IMZBC-NEXT:    lw a4, 4(a2)
+; RV32IMZBC-NEXT:    lui a6, 16
+; RV32IMZBC-NEXT:    lw a3, 0(a2)
+; RV32IMZBC-NEXT:    lw a5, 8(a2)
+; RV32IMZBC-NEXT:    lw a2, 0(a1)
+; RV32IMZBC-NEXT:    addi a7, a6, -256
+; RV32IMZBC-NEXT:    srli a6, a4, 8
+; RV32IMZBC-NEXT:    srli t0, a4, 24
+; RV32IMZBC-NEXT:    and t1, a4, a7
+; RV32IMZBC-NEXT:    and a6, a6, a7
+; RV32IMZBC-NEXT:    slli t1, t1, 8
+; RV32IMZBC-NEXT:    slli t2, a4, 24
+; RV32IMZBC-NEXT:    or a6, a6, t0
+; RV32IMZBC-NEXT:    or t0, t2, t1
+; RV32IMZBC-NEXT:    or a6, t0, a6
+; RV32IMZBC-NEXT:    lui t0, 61681
+; RV32IMZBC-NEXT:    srli t1, a6, 4
+; RV32IMZBC-NEXT:    addi t0, t0, -241
+; RV32IMZBC-NEXT:    and t1, t1, t0
+; RV32IMZBC-NEXT:    and a6, a6, t0
+; RV32IMZBC-NEXT:    slli a6, a6, 4
+; RV32IMZBC-NEXT:    lui t2, 209715
+; RV32IMZBC-NEXT:    or a6, t1, a6
+; RV32IMZBC-NEXT:    addi t1, t2, 819
+; RV32IMZBC-NEXT:    srli t2, a6, 2
+; RV32IMZBC-NEXT:    and a6, a6, t1
+; RV32IMZBC-NEXT:    and t2, t2, t1
+; RV32IMZBC-NEXT:    slli a6, a6, 2
+; RV32IMZBC-NEXT:    or t4, t2, a6
+; RV32IMZBC-NEXT:    lw a6, 4(a1)
+; RV32IMZBC-NEXT:    lw a1, 8(a1)
+; RV32IMZBC-NEXT:    srli t5, t4, 1
+; RV32IMZBC-NEXT:    lui t2, 349525
+; RV32IMZBC-NEXT:    srli t6, a2, 8
+; RV32IMZBC-NEXT:    addi t3, t2, 1365
+; RV32IMZBC-NEXT:    and t6, t6, a7
+; RV32IMZBC-NEXT:    srli s0, a2, 24
+; RV32IMZBC-NEXT:    and s1, a2, a7
+; RV32IMZBC-NEXT:    slli s1, s1, 8
+; RV32IMZBC-NEXT:    slli s2, a2, 24
+; RV32IMZBC-NEXT:    or t6, t6, s0
+; RV32IMZBC-NEXT:    or s0, s2, s1
+; RV32IMZBC-NEXT:    and t5, t5, t3
+; RV32IMZBC-NEXT:    or t6, s0, t6
+; RV32IMZBC-NEXT:    srli s0, t6, 4
+; RV32IMZBC-NEXT:    and t6, t6, t0
+; RV32IMZBC-NEXT:    and s0, s0, t0
+; RV32IMZBC-NEXT:    slli t6, t6, 4
+; RV32IMZBC-NEXT:    and t4, t4, t3
+; RV32IMZBC-NEXT:    or t6, s0, t6
+; RV32IMZBC-NEXT:    srli s0, t6, 2
+; RV32IMZBC-NEXT:    and t6, t6, t1
+; RV32IMZBC-NEXT:    and s0, s0, t1
+; RV32IMZBC-NEXT:    slli t6, t6, 2
+; RV32IMZBC-NEXT:    slli t4, t4, 1
+; RV32IMZBC-NEXT:    or t6, s0, t6
+; RV32IMZBC-NEXT:    or t4, t5, t4
+; RV32IMZBC-NEXT:    srli t5, t6, 1
+; RV32IMZBC-NEXT:    and t5, t5, t3
+; RV32IMZBC-NEXT:    srli s0, a3, 8
+; RV32IMZBC-NEXT:    and s0, s0, a7
+; RV32IMZBC-NEXT:    srli s1, a3, 24
+; RV32IMZBC-NEXT:    or s0, s0, s1
+; RV32IMZBC-NEXT:    and s1, a3, a7
+; RV32IMZBC-NEXT:    slli s1, s1, 8
+; RV32IMZBC-NEXT:    slli s2, a3, 24
+; RV32IMZBC-NEXT:    and t6, t6, t3
+; RV32IMZBC-NEXT:    or s1, s2, s1
+; RV32IMZBC-NEXT:    slli t6, t6, 1
+; RV32IMZBC-NEXT:    or s0, s1, s0
+; RV32IMZBC-NEXT:    srli s1, s0, 4
+; RV32IMZBC-NEXT:    and s0, s0, t0
+; RV32IMZBC-NEXT:    and s1, s1, t0
+; RV32IMZBC-NEXT:    slli s0, s0, 4
+; RV32IMZBC-NEXT:    or t5, t5, t6
+; RV32IMZBC-NEXT:    or s0, s1, s0
+; RV32IMZBC-NEXT:    srli t6, s0, 2
+; RV32IMZBC-NEXT:    and s0, s0, t1
+; RV32IMZBC-NEXT:    and t6, t6, t1
+; RV32IMZBC-NEXT:    slli s0, s0, 2
+; RV32IMZBC-NEXT:    or t6, t6, s0
+; RV32IMZBC-NEXT:    srli s0, a6, 8
+; RV32IMZBC-NEXT:    and s0, s0, a7
+; RV32IMZBC-NEXT:    srli s1, a6, 24
+; RV32IMZBC-NEXT:    or s0, s0, s1
+; RV32IMZBC-NEXT:    and s1, a6, a7
+; RV32IMZBC-NEXT:    slli s1, s1, 8
+; RV32IMZBC-NEXT:    slli s2, a6, 24
+; RV32IMZBC-NEXT:    or s1, s2, s1
+; RV32IMZBC-NEXT:    srli s2, t6, 1
+; RV32IMZBC-NEXT:    and s2, s2, t3
+; RV32IMZBC-NEXT:    or s0, s1, s0
+; RV32IMZBC-NEXT:    srli s1, s0, 4
+; RV32IMZBC-NEXT:    and s0, s0, t0
+; RV32IMZBC-NEXT:    and s1, s1, t0
+; RV32IMZBC-NEXT:    slli s0, s0, 4
+; RV32IMZBC-NEXT:    and t6, t6, t3
+; RV32IMZBC-NEXT:    or s0, s1, s0
+; RV32IMZBC-NEXT:    srli s1, s0, 2
+; RV32IMZBC-NEXT:    and s0, s0, t1
+; RV32IMZBC-NEXT:    and s1, s1, t1
+; RV32IMZBC-NEXT:    slli s0, s0, 2
+; RV32IMZBC-NEXT:    slli t6, t6, 1
+; RV32IMZBC-NEXT:    or s0, s1, s0
+; RV32IMZBC-NEXT:    srli s1, s0, 1
+; RV32IMZBC-NEXT:    and s0, s0, t3
+; RV32IMZBC-NEXT:    and s1, s1, t3
+; RV32IMZBC-NEXT:    slli s0, s0, 1
+; RV32IMZBC-NEXT:    or t6, s2, t6
+; RV32IMZBC-NEXT:    or s0, s1, s0
+; RV32IMZBC-NEXT:    clmul t5, t5, t4
+; RV32IMZBC-NEXT:    clmul t6, s0, t6
+; RV32IMZBC-NEXT:    clmulh t4, s0, t4
+; RV32IMZBC-NEXT:    xor t5, t6, t5
+; RV32IMZBC-NEXT:    xor t4, t4, t5
+; RV32IMZBC-NEXT:    srli t5, t4, 8
+; RV32IMZBC-NEXT:    and t5, t5, a7
+; RV32IMZBC-NEXT:    srli t6, t4, 24
+; RV32IMZBC-NEXT:    and a7, t4, a7
+; RV32IMZBC-NEXT:    slli t4, t4, 24
+; RV32IMZBC-NEXT:    slli a7, a7, 8
+; RV32IMZBC-NEXT:    or t5, t5, t6
+; RV32IMZBC-NEXT:    or a7, t4, a7
+; RV32IMZBC-NEXT:    or a7, a7, t5
+; RV32IMZBC-NEXT:    srli t4, a7, 4
+; RV32IMZBC-NEXT:    and a7, a7, t0
+; RV32IMZBC-NEXT:    and t0, t4, t0
+; RV32IMZBC-NEXT:    slli a7, a7, 4
+; RV32IMZBC-NEXT:    or a7, t0, a7
+; RV32IMZBC-NEXT:    srli t0, a7, 2
+; RV32IMZBC-NEXT:    and a7, a7, t1
+; RV32IMZBC-NEXT:    and t0, t0, t1
+; RV32IMZBC-NEXT:    slli a7, a7, 2
+; RV32IMZBC-NEXT:    or a7, t0, a7
+; RV32IMZBC-NEXT:    srli t0, a7, 1
+; RV32IMZBC-NEXT:    addi t1, t2, 1364
+; RV32IMZBC-NEXT:    and a7, a7, t3
+; RV32IMZBC-NEXT:    and t0, t0, t1
+; RV32IMZBC-NEXT:    slli a7, a7, 1
+; RV32IMZBC-NEXT:    or a7, t0, a7
+; RV32IMZBC-NEXT:    clmulr t0, a6, a4
+; RV32IMZBC-NEXT:    clmul a1, a1, a3
+; RV32IMZBC-NEXT:    clmul a5, a2, a5
+; RV32IMZBC-NEXT:    srli a7, a7, 1
+; RV32IMZBC-NEXT:    slli t0, t0, 31
+; RV32IMZBC-NEXT:    or a7, a7, t0
+; RV32IMZBC-NEXT:    xor a1, a5, a1
+; RV32IMZBC-NEXT:    clmul a5, a6, a3
+; RV32IMZBC-NEXT:    clmul a4, a2, a4
+; RV32IMZBC-NEXT:    clmulh a6, a2, a3
+; RV32IMZBC-NEXT:    clmul a2, a2, a3
+; RV32IMZBC-NEXT:    xor a1, a7, a1
+; RV32IMZBC-NEXT:    xor a4, a4, a5
+; RV32IMZBC-NEXT:    xor a3, a6, a4
+; RV32IMZBC-NEXT:    sw a2, 0(a0)
+; RV32IMZBC-NEXT:    sw a3, 4(a0)
+; RV32IMZBC-NEXT:    sw a1, 8(a0)
+; RV32IMZBC-NEXT:    lw s0, 12(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s1, 8(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s2, 4(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    .cfi_restore s0
+; RV32IMZBC-NEXT:    .cfi_restore s1
+; RV32IMZBC-NEXT:    .cfi_restore s2
+; RV32IMZBC-NEXT:    addi sp, sp, 16
+; RV32IMZBC-NEXT:    .cfi_def_cfa_offset 0
+; RV32IMZBC-NEXT:    ret
+;
+; RV64IMZBC-LABEL: clmul_i96:
+; RV64IMZBC:       # %bb.0:
+; RV64IMZBC-NEXT:    clmul a1, a1, a2
+; RV64IMZBC-NEXT:    clmul a3, a0, a3
+; RV64IMZBC-NEXT:    clmulh a4, a0, a2
+; RV64IMZBC-NEXT:    xor a1, a3, a1
+; RV64IMZBC-NEXT:    clmul a0, a0, a2
+; RV64IMZBC-NEXT:    xor a1, a4, a1
+; RV64IMZBC-NEXT:    ret
+  %a = call i96 @llvm.clmul.i96(i96 %x, i96 %y)
+  ret i96 %a
+}
+
+define i128 @clmul_i128(i128 %x, i128 %y) {
+; RV32I-LABEL: clmul_i128:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi sp, sp, -880
+; RV32I-NEXT:    .cfi_def_cfa_offset 880
+; RV32I-NEXT:    sw ra, 876(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s0, 872(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s1, 868(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s2, 864(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s3, 860(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s4, 856(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s5, 852(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s6, 848(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s7, 844(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s8, 840(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s9, 836(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s10, 832(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s11, 828(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    .cfi_offset ra, -4
+; RV32I-NEXT:    .cfi_offset s0, -8
+; RV32I-NEXT:    .cfi_offset s1, -12
+; RV32I-NEXT:    .cfi_offset s2, -16
+; RV32I-NEXT:    .cfi_offset s3, -20
+; RV32I-NEXT:    .cfi_offset s4, -24
+; RV32I-NEXT:    .cfi_offset s5, -28
+; RV32I-NEXT:    .cfi_offset s6, -32
+; RV32I-NEXT:    .cfi_offset s7, -36
+; RV32I-NEXT:    .cfi_offset s8, -40
+; RV32I-NEXT:    .cfi_offset s9, -44
+; RV32I-NEXT:    .cfi_offset s10, -48
+; RV32I-NEXT:    .cfi_offset s11, -52
+; RV32I-NEXT:    mv t5, a2
+; RV32I-NEXT:    sw a0, 284(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 4(a1)
+; RV32I-NEXT:    sw a5, 544(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a0, 16
+; RV32I-NEXT:    lw s4, 8(a1)
+; RV32I-NEXT:    lw a3, 12(a1)
+; RV32I-NEXT:    sw a3, 276(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a2, a0, -256
+; RV32I-NEXT:    lui t4, 16
+; RV32I-NEXT:    srli a0, a5, 8
+; RV32I-NEXT:    srli a3, a5, 24
+; RV32I-NEXT:    and a4, a5, a2
+; RV32I-NEXT:    slli a5, a5, 24
+; RV32I-NEXT:    sw a5, 644(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a0, a0, a2
+; RV32I-NEXT:    slli a4, a4, 8
+; RV32I-NEXT:    or a0, a0, a3
+; RV32I-NEXT:    or a3, a5, a4
+; RV32I-NEXT:    lui a4, 61681
+; RV32I-NEXT:    or a0, a3, a0
+; RV32I-NEXT:    addi s11, a4, -241
+; RV32I-NEXT:    srli a3, a0, 4
+; RV32I-NEXT:    and a0, a0, s11
+; RV32I-NEXT:    and a3, a3, s11
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    or a0, a3, a0
+; RV32I-NEXT:    lui a3, 209715
+; RV32I-NEXT:    srli a4, a0, 2
+; RV32I-NEXT:    addi t6, a3, 819
+; RV32I-NEXT:    and a3, a4, t6
+; RV32I-NEXT:    and a0, a0, t6
+; RV32I-NEXT:    slli a0, a0, 2
+; RV32I-NEXT:    lui s8, 349525
+; RV32I-NEXT:    or a0, a3, a0
+; RV32I-NEXT:    addi s5, s8, 1365
+; RV32I-NEXT:    srli a3, a0, 1
+; RV32I-NEXT:    and a0, a0, s5
+; RV32I-NEXT:    and a3, a3, s5
+; RV32I-NEXT:    slli a0, a0, 1
+; RV32I-NEXT:    or t2, a3, a0
+; RV32I-NEXT:    srli a0, t2, 8
+; RV32I-NEXT:    lw a6, 4(t5)
+; RV32I-NEXT:    and a0, a0, a2
+; RV32I-NEXT:    srli a3, t2, 24
+; RV32I-NEXT:    and a4, t2, a2
+; RV32I-NEXT:    slli a5, t2, 24
+; RV32I-NEXT:    sw a5, 824(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, a4, 8
+; RV32I-NEXT:    or a0, a0, a3
+; RV32I-NEXT:    or a3, a5, a4
+; RV32I-NEXT:    or a0, a3, a0
+; RV32I-NEXT:    lw s6, 8(t5)
+; RV32I-NEXT:    lw a3, 12(t5)
+; RV32I-NEXT:    sw a3, 272(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a3, a0, 4
+; RV32I-NEXT:    and a3, a3, s11
+; RV32I-NEXT:    sw a6, 280(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a4, a6, 8
+; RV32I-NEXT:    and a0, a0, s11
+; RV32I-NEXT:    and a4, a4, a2
+; RV32I-NEXT:    srli a5, a6, 24
+; RV32I-NEXT:    and a7, a6, a2
+; RV32I-NEXT:    slli a7, a7, 8
+; RV32I-NEXT:    slli t0, a6, 24
+; RV32I-NEXT:    or a4, a4, a5
+; RV32I-NEXT:    or a5, t0, a7
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    or a4, a5, a4
+; RV32I-NEXT:    srli a5, a4, 4
+; RV32I-NEXT:    and a4, a4, s11
+; RV32I-NEXT:    and a5, a5, s11
+; RV32I-NEXT:    slli a4, a4, 4
+; RV32I-NEXT:    or a0, a3, a0
+; RV32I-NEXT:    or a4, a5, a4
+; RV32I-NEXT:    srli a3, a4, 2
+; RV32I-NEXT:    and a4, a4, t6
+; RV32I-NEXT:    and a3, a3, t6
+; RV32I-NEXT:    slli a4, a4, 2
+; RV32I-NEXT:    srli a5, a0, 2
+; RV32I-NEXT:    or a3, a3, a4
+; RV32I-NEXT:    srli a4, a3, 1
+; RV32I-NEXT:    and a3, a3, s5
+; RV32I-NEXT:    and a4, a4, s5
+; RV32I-NEXT:    slli s0, a3, 1
+; RV32I-NEXT:    sw s0, 820(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, t6
+; RV32I-NEXT:    or s0, a4, s0
+; RV32I-NEXT:    and a0, a0, t6
+; RV32I-NEXT:    srli a4, s0, 8
+; RV32I-NEXT:    slli a0, a0, 2
+; RV32I-NEXT:    and a4, a4, a2
+; RV32I-NEXT:    srli a5, s0, 24
+; RV32I-NEXT:    and a7, s0, a2
+; RV32I-NEXT:    slli t0, s0, 24
+; RV32I-NEXT:    slli a7, a7, 8
+; RV32I-NEXT:    or a4, a4, a5
+; RV32I-NEXT:    or a5, t0, a7
+; RV32I-NEXT:    or a0, a3, a0
+; RV32I-NEXT:    or a4, a5, a4
+; RV32I-NEXT:    srli a3, a4, 4
+; RV32I-NEXT:    and a4, a4, s11
+; RV32I-NEXT:    and a3, a3, s11
+; RV32I-NEXT:    slli a4, a4, 4
+; RV32I-NEXT:    srli a5, a0, 1
+; RV32I-NEXT:    or a3, a3, a4
+; RV32I-NEXT:    srli a4, a3, 2
+; RV32I-NEXT:    and a3, a3, t6
+; RV32I-NEXT:    and a4, a4, t6
+; RV32I-NEXT:    slli a3, a3, 2
+; RV32I-NEXT:    and a5, a5, s5
+; RV32I-NEXT:    or a3, a4, a3
+; RV32I-NEXT:    srli a4, a3, 1
+; RV32I-NEXT:    and a3, a3, s5
+; RV32I-NEXT:    and a4, a4, s5
+; RV32I-NEXT:    slli a3, a3, 1
+; RV32I-NEXT:    and a7, a0, s5
+; RV32I-NEXT:    or a0, a4, a3
+; RV32I-NEXT:    slli a7, a7, 1
+; RV32I-NEXT:    andi a4, a0, 1
+; RV32I-NEXT:    or a5, a5, a7
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    andi a7, a0, 2
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    slli t0, a5, 1
+; RV32I-NEXT:    and a7, a7, t0
+; RV32I-NEXT:    andi t0, a0, 4
+; RV32I-NEXT:    xor a4, a4, a7
+; RV32I-NEXT:    seqz a7, t0
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    andi t0, a0, 8
+; RV32I-NEXT:    slli t1, a5, 2
+; RV32I-NEXT:    seqz t0, t0
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    slli t3, a5, 3
+; RV32I-NEXT:    and a7, a7, t1
+; RV32I-NEXT:    and t0, t0, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    andi t0, a0, 16
+; RV32I-NEXT:    xor a4, a4, a7
+; RV32I-NEXT:    seqz a7, t0
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    andi t0, a0, 32
+; RV32I-NEXT:    slli t1, a5, 4
+; RV32I-NEXT:    seqz t0, t0
+; RV32I-NEXT:    and a7, a7, t1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    slli t1, a5, 5
+; RV32I-NEXT:    andi t3, a0, 64
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    seqz t1, t3
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli t3, a5, 6
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    and t0, t1, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    andi t0, a0, 128
+; RV32I-NEXT:    xor a4, a4, a7
+; RV32I-NEXT:    seqz a7, t0
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    andi t0, a0, 256
+; RV32I-NEXT:    slli t1, a5, 7
+; RV32I-NEXT:    seqz t0, t0
+; RV32I-NEXT:    and a7, a7, t1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    slli t1, a5, 8
+; RV32I-NEXT:    andi t3, a0, 512
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    seqz t1, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli t0, a5, 9
+; RV32I-NEXT:    andi t3, a0, 1024
+; RV32I-NEXT:    and t0, t1, t0
+; RV32I-NEXT:    seqz t1, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli t0, a5, 10
+; RV32I-NEXT:    li t3, 1
+; RV32I-NEXT:    and t0, t1, t0
+; RV32I-NEXT:    slli s10, t3, 11
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    and t0, a0, s10
+; RV32I-NEXT:    xor a4, a4, a7
+; RV32I-NEXT:    seqz a7, t0
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    lui a6, 1
+; RV32I-NEXT:    slli t0, a5, 11
+; RV32I-NEXT:    and t1, a0, a6
+; RV32I-NEXT:    lui s2, 1
+; RV32I-NEXT:    and a7, a7, t0
+; RV32I-NEXT:    seqz t0, t1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    lui a6, 2
+; RV32I-NEXT:    slli t1, a5, 12
+; RV32I-NEXT:    and t3, a0, a6
+; RV32I-NEXT:    lui s3, 2
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    seqz t1, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli t0, a5, 13
+; RV32I-NEXT:    lui a6, 4
+; RV32I-NEXT:    and t0, t1, t0
+; RV32I-NEXT:    and t1, a0, a6
+; RV32I-NEXT:    lui s9, 4
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    seqz t0, t1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    lui a6, 8
+; RV32I-NEXT:    slli t1, a5, 14
+; RV32I-NEXT:    and t3, a0, a6
+; RV32I-NEXT:    lui s7, 8
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    seqz t1, t3
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli t3, a5, 15
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    and t0, t1, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    and t0, a0, t4
+; RV32I-NEXT:    xor a4, a4, a7
+; RV32I-NEXT:    seqz a7, t0
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    lui a6, 32
+; RV32I-NEXT:    slli t0, a5, 16
+; RV32I-NEXT:    and t1, a0, a6
+; RV32I-NEXT:    and a7, a7, t0
+; RV32I-NEXT:    seqz t0, t1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    lui a6, 64
+; RV32I-NEXT:    slli t1, a5, 17
+; RV32I-NEXT:    and t3, a0, a6
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    seqz t1, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli t0, a5, 18
+; RV32I-NEXT:    lui a6, 128
+; RV32I-NEXT:    and t0, t1, t0
+; RV32I-NEXT:    and t1, a0, a6
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    seqz t0, t1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    lui a6, 256
+; RV32I-NEXT:    slli t1, a5, 19
+; RV32I-NEXT:    and t3, a0, a6
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    seqz t1, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli t0, a5, 20
+; RV32I-NEXT:    lui a6, 512
+; RV32I-NEXT:    and t0, t1, t0
+; RV32I-NEXT:    and t1, a0, a6
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    seqz t0, t1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    slli t1, a5, 21
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    lui a6, 1024
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    and t0, a0, a6
+; RV32I-NEXT:    xor a4, a4, a7
+; RV32I-NEXT:    seqz a7, t0
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    lui a6, 2048
+; RV32I-NEXT:    slli t0, a5, 22
+; RV32I-NEXT:    and t1, a0, a6
+; RV32I-NEXT:    lui s1, 2048
+; RV32I-NEXT:    and a7, a7, t0
+; RV32I-NEXT:    seqz t0, t1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    lui a6, 4096
+; RV32I-NEXT:    slli t1, a5, 23
+; RV32I-NEXT:    and t3, a0, a6
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    seqz t1, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli t0, a5, 24
+; RV32I-NEXT:    lui a6, 8192
+; RV32I-NEXT:    and t0, t1, t0
+; RV32I-NEXT:    and t1, a0, a6
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    seqz t0, t1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    lui a6, 16384
+; RV32I-NEXT:    slli t1, a5, 25
+; RV32I-NEXT:    and t3, a0, a6
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    seqz t1, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli t0, a5, 26
+; RV32I-NEXT:    lui a6, 32768
+; RV32I-NEXT:    and t0, t1, t0
+; RV32I-NEXT:    and t1, a0, a6
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    seqz t0, t1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    lui a6, 65536
+; RV32I-NEXT:    slli t1, a5, 27
+; RV32I-NEXT:    and t3, a0, a6
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    seqz t1, t3
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli t3, a5, 28
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    and t0, t1, t3
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    lui a6, 131072
+; RV32I-NEXT:    xor a4, a4, a7
+; RV32I-NEXT:    and a7, a0, a6
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    lui a6, 262144
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    and a0, a0, a6
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    slli t0, a5, 29
+; RV32I-NEXT:    and a7, a7, t0
+; RV32I-NEXT:    addi a0, a0, -1
+; RV32I-NEXT:    srli a3, a3, 31
+; RV32I-NEXT:    slli t0, a5, 30
+; RV32I-NEXT:    and a0, a0, t0
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    slli a5, a5, 31
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    xor a0, a7, a0
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    lw t4, 0(a1)
+; RV32I-NEXT:    xor a0, a4, a0
+; RV32I-NEXT:    srli a1, a0, 8
+; RV32I-NEXT:    and a1, a1, a2
+; RV32I-NEXT:    srli a3, a0, 24
+; RV32I-NEXT:    or a1, a1, a3
+; RV32I-NEXT:    srli a3, t4, 8
+; RV32I-NEXT:    and a3, a3, a2
+; RV32I-NEXT:    srli a4, t4, 24
+; RV32I-NEXT:    or a3, a3, a4
+; RV32I-NEXT:    slli a4, a0, 24
+; RV32I-NEXT:    and a0, a0, a2
+; RV32I-NEXT:    and a5, t4, a2
+; RV32I-NEXT:    slli a5, a5, 8
+; RV32I-NEXT:    slli a6, t4, 24
+; RV32I-NEXT:    sw a6, 296(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a0, a0, 8
+; RV32I-NEXT:    or a5, a6, a5
+; RV32I-NEXT:    or a0, a4, a0
+; RV32I-NEXT:    or a3, a5, a3
+; RV32I-NEXT:    srli a4, a3, 4
+; RV32I-NEXT:    and a3, a3, s11
+; RV32I-NEXT:    and a4, a4, s11
+; RV32I-NEXT:    slli a3, a3, 4
+; RV32I-NEXT:    or a0, a0, a1
+; RV32I-NEXT:    or a3, a4, a3
+; RV32I-NEXT:    srli a1, a0, 4
+; RV32I-NEXT:    and a0, a0, s11
+; RV32I-NEXT:    and a1, a1, s11
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    srli a4, a3, 2
+; RV32I-NEXT:    and a3, a3, t6
+; RV32I-NEXT:    and a4, a4, t6
+; RV32I-NEXT:    slli a3, a3, 2
+; RV32I-NEXT:    or a0, a1, a0
+; RV32I-NEXT:    or a3, a4, a3
+; RV32I-NEXT:    srli a1, a3, 1
+; RV32I-NEXT:    and a3, a3, s5
+; RV32I-NEXT:    and a1, a1, s5
+; RV32I-NEXT:    slli a3, a3, 1
+; RV32I-NEXT:    or t1, a1, a3
+; RV32I-NEXT:    andi a1, s0, 2
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    andi a3, s0, 1
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 816(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a1, a3
+; RV32I-NEXT:    addi a3, a1, -1
+; RV32I-NEXT:    sw a3, 812(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t1, 1
+; RV32I-NEXT:    sw a1, 640(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a4, a1
+; RV32I-NEXT:    and a3, a3, t1
+; RV32I-NEXT:    xor a1, a3, a1
+; RV32I-NEXT:    andi a3, s0, 4
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    andi a4, s0, 8
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 808(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a6, a3, -1
+; RV32I-NEXT:    sw a6, 804(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t1, 2
+; RV32I-NEXT:    sw a3, 636(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    slli a4, t1, 3
+; RV32I-NEXT:    sw a4, 632(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    andi a5, s0, 16
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a6, a4, -1
+; RV32I-NEXT:    sw a6, 800(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a4, s0, 32
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    andi a5, s0, 64
+; RV32I-NEXT:    addi a7, a4, -1
+; RV32I-NEXT:    sw a7, 796(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi t0, a4, -1
+; RV32I-NEXT:    sw t0, 792(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t1, 4
+; RV32I-NEXT:    sw a4, 628(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    slli a5, t1, 5
+; RV32I-NEXT:    sw a5, 624(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    slli a6, t1, 6
+; RV32I-NEXT:    sw a6, 620(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, t0, a6
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    srli a3, a0, 2
+; RV32I-NEXT:    and a0, a0, t6
+; RV32I-NEXT:    and a3, a3, t6
+; RV32I-NEXT:    slli a0, a0, 2
+; RV32I-NEXT:    or a0, a3, a0
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    andi a3, s0, 128
+; RV32I-NEXT:    andi a4, s0, 256
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 788(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a6, a4, -1
+; RV32I-NEXT:    sw a6, 784(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t1, 7
+; RV32I-NEXT:    sw a3, 616(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t1, 8
+; RV32I-NEXT:    sw a4, 612(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    andi a4, s0, 512
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    andi a5, s0, 1024
+; RV32I-NEXT:    addi a6, a4, -1
+; RV32I-NEXT:    sw a6, 780(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a7, a4, -1
+; RV32I-NEXT:    sw a7, 776(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t1, 9
+; RV32I-NEXT:    sw a4, 608(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    slli a5, t1, 10
+; RV32I-NEXT:    sw a5, 604(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    and a4, a7, a5
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    and a4, s0, s10
+; RV32I-NEXT:    sw s10, 288(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 768(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, s0, s2
+; RV32I-NEXT:    lui ra, 1
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    and a4, s0, s3
+; RV32I-NEXT:    addi a6, a3, -1
+; RV32I-NEXT:    sw a6, 764(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a7, a3, -1
+; RV32I-NEXT:    sw a7, 760(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t1, 11
+; RV32I-NEXT:    sw a3, 600(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    slli a4, t1, 12
+; RV32I-NEXT:    sw a4, 596(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    slli a5, t1, 13
+; RV32I-NEXT:    sw a5, 592(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    and a4, a7, a5
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    and a4, s0, s9
+; RV32I-NEXT:    lui s3, 4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a5, s0, s7
+; RV32I-NEXT:    addi a6, a4, -1
+; RV32I-NEXT:    sw a6, 756(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a7, a4, -1
+; RV32I-NEXT:    sw a7, 752(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t1, 14
+; RV32I-NEXT:    sw a4, 588(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    slli a5, t1, 15
+; RV32I-NEXT:    sw a5, 584(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    and a4, a7, a5
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    addi a5, s8, 1364
+; RV32I-NEXT:    sw a5, 700(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a4, a0, 1
+; RV32I-NEXT:    and a0, a0, s5
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    slli a0, a0, 1
+; RV32I-NEXT:    or a0, a4, a0
+; RV32I-NEXT:    sw a0, 772(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a0, a1, a3
+; RV32I-NEXT:    lui a1, 16
+; RV32I-NEXT:    and a1, s0, a1
+; RV32I-NEXT:    lui s8, 16
+; RV32I-NEXT:    lui a3, 32
+; RV32I-NEXT:    and a3, s0, a3
+; RV32I-NEXT:    lui s9, 32
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 748(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 744(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t1, 16
+; RV32I-NEXT:    sw a1, 580(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t1, 17
+; RV32I-NEXT:    sw a3, 576(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a4, a1
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lui a3, 64
+; RV32I-NEXT:    and a3, s0, a3
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    lui a4, 128
+; RV32I-NEXT:    and a4, s0, a4
+; RV32I-NEXT:    lui s7, 128
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 740(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a6, a3, -1
+; RV32I-NEXT:    sw a6, 736(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t1, 18
+; RV32I-NEXT:    sw a3, 572(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    slli a4, t1, 19
+; RV32I-NEXT:    sw a4, 568(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a3, a6, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lui a3, 256
+; RV32I-NEXT:    and a3, s0, a3
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    lui a4, 1024
+; RV32I-NEXT:    and a4, s0, a4
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 724(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a6, a3, -1
+; RV32I-NEXT:    sw a6, 732(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, s0, s1
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    slli a4, t1, 20
+; RV32I-NEXT:    sw a4, 556(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a7, a3, -1
+; RV32I-NEXT:    sw a7, 728(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, a4
+; RV32I-NEXT:    slli a4, t1, 22
+; RV32I-NEXT:    sw a4, 564(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, t1, 23
+; RV32I-NEXT:    sw a5, 560(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lui s1, 512
+; RV32I-NEXT:    and a3, s0, s1
+; RV32I-NEXT:    lui s2, 4096
+; RV32I-NEXT:    and a5, s0, s2
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a7, a3, -1
+; RV32I-NEXT:    sw a7, 716(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a6, a5, -1
+; RV32I-NEXT:    sw a6, 720(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t1, 21
+; RV32I-NEXT:    sw a3, 548(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, t1, 24
+; RV32I-NEXT:    sw a5, 552(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a7, a3
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a6, 0(t5)
+; RV32I-NEXT:    lui t5, 8192
+; RV32I-NEXT:    and a3, s0, t5
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    lui a5, 16384
+; RV32I-NEXT:    and a5, s0, a5
+; RV32I-NEXT:    addi a7, a3, -1
+; RV32I-NEXT:    sw a7, 712(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a5
+; RV32I-NEXT:    addi t0, a3, -1
+; RV32I-NEXT:    sw t0, 708(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t1, 25
+; RV32I-NEXT:    sw a3, 540(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a7, a3
+; RV32I-NEXT:    slli a5, t1, 26
+; RV32I-NEXT:    sw a5, 536(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a3, a4, a3
+; RV32I-NEXT:    and a4, t0, a5
+; RV32I-NEXT:    xor a1, a0, a1
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    srli a0, a6, 8
+; RV32I-NEXT:    sw a2, 652(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a2
+; RV32I-NEXT:    and a0, a0, a2
+; RV32I-NEXT:    slli a4, a4, 8
+; RV32I-NEXT:    srli a5, a6, 24
+; RV32I-NEXT:    slli a7, a6, 24
+; RV32I-NEXT:    or a0, a0, a5
+; RV32I-NEXT:    or a4, a7, a4
+; RV32I-NEXT:    or a0, a4, a0
+; RV32I-NEXT:    lui t0, 32768
+; RV32I-NEXT:    and a4, s0, t0
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    srli a5, a0, 4
+; RV32I-NEXT:    addi a2, a4, -1
+; RV32I-NEXT:    sw a2, 704(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a5, s11
+; RV32I-NEXT:    and a0, a0, s11
+; RV32I-NEXT:    slli a5, t1, 27
+; RV32I-NEXT:    sw a5, 532(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    and a5, a2, a5
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    or a0, a4, a0
+; RV32I-NEXT:    srli a4, a0, 2
+; RV32I-NEXT:    sw t6, 648(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a0, a0, t6
+; RV32I-NEXT:    and a4, a4, t6
+; RV32I-NEXT:    slli a0, a0, 2
+; RV32I-NEXT:    or a0, a4, a0
+; RV32I-NEXT:    lui a4, 65536
+; RV32I-NEXT:    and a4, s0, a4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    srli a5, a0, 1
+; RV32I-NEXT:    addi a2, a4, -1
+; RV32I-NEXT:    sw a2, 696(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s5, 656(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a5, s5
+; RV32I-NEXT:    and a0, a0, s5
+; RV32I-NEXT:    slli a5, t1, 28
+; RV32I-NEXT:    sw a5, 528(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t3, a0, 1
+; RV32I-NEXT:    and a0, a2, a5
+; RV32I-NEXT:    xor a3, a3, a0
+; RV32I-NEXT:    or a0, a4, t3
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    sw a1, 680(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a1, a0, 2
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    andi a3, a0, 1
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 524(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a1, a3
+; RV32I-NEXT:    addi a3, a1, -1
+; RV32I-NEXT:    sw a3, 520(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t2, 1
+; RV32I-NEXT:    sw a1, 692(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a4, a1
+; RV32I-NEXT:    and a3, a3, t2
+; RV32I-NEXT:    xor a2, a3, a1
+; RV32I-NEXT:    andi a3, a0, 4
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    andi a4, a0, 8
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 516(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a4, a3, -1
+; RV32I-NEXT:    sw a4, 512(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t2, 2
+; RV32I-NEXT:    sw a3, 688(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    slli a1, t2, 3
+; RV32I-NEXT:    sw a1, 684(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a4, a1
+; RV32I-NEXT:    andi a5, a0, 16
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a1, a4, -1
+; RV32I-NEXT:    sw a1, 500(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a4, a0, 32
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    andi a5, a0, 64
+; RV32I-NEXT:    addi a7, a4, -1
+; RV32I-NEXT:    sw a7, 496(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    sw a4, 492(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, t2, 4
+; RV32I-NEXT:    sw a5, 668(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    slli a5, t2, 5
+; RV32I-NEXT:    sw a5, 664(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    slli t6, t2, 6
+; RV32I-NEXT:    sw t6, 660(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a7, a1, a5
+; RV32I-NEXT:    and a5, a4, t6
+; RV32I-NEXT:    xor a1, a2, a3
+; RV32I-NEXT:    xor a4, a7, a5
+; RV32I-NEXT:    andi a3, a0, 128
+; RV32I-NEXT:    andi a5, a0, 256
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    sw a3, 488(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 484(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 7
+; RV32I-NEXT:    sw a2, 380(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a7, t2, 8
+; RV32I-NEXT:    sw a7, 376(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a3, a2
+; RV32I-NEXT:    and a5, a5, a7
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    andi a5, a0, 512
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    andi a7, a0, 1024
+; RV32I-NEXT:    addi a2, a5, -1
+; RV32I-NEXT:    sw a2, 480(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a5, a7
+; RV32I-NEXT:    addi a7, a5, -1
+; RV32I-NEXT:    sw a7, 476(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, t2, 9
+; RV32I-NEXT:    sw a5, 372(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a2, a5
+; RV32I-NEXT:    slli a2, t2, 10
+; RV32I-NEXT:    sw a2, 368(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    and a5, a7, a2
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lui a4, 131072
+; RV32I-NEXT:    and a4, s0, a4
+; RV32I-NEXT:    lui a5, 262144
+; RV32I-NEXT:    and a5, s0, a5
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a2, a4, -1
+; RV32I-NEXT:    sw a2, 676(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a7, a5, -1
+; RV32I-NEXT:    sw a7, 672(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t1, 29
+; RV32I-NEXT:    sw a4, 508(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, t1, 30
+; RV32I-NEXT:    sw t1, 384(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw a5, 504(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a2, a4
+; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a3, a0, s10
+; RV32I-NEXT:    and a5, a0, ra
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    sw a3, 468(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 464(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 11
+; RV32I-NEXT:    sw a2, 364(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a7, t2, 12
+; RV32I-NEXT:    sw a7, 360(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a3, a2
+; RV32I-NEXT:    and a5, a5, a7
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lui a5, 2
+; RV32I-NEXT:    and a5, a0, a5
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    and a7, a0, s3
+; RV32I-NEXT:    addi a2, a5, -1
+; RV32I-NEXT:    sw a2, 460(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a5, a7
+; RV32I-NEXT:    addi a7, a5, -1
+; RV32I-NEXT:    sw a7, 456(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, t2, 13
+; RV32I-NEXT:    sw a5, 356(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a2, a5
+; RV32I-NEXT:    slli a2, t2, 14
+; RV32I-NEXT:    sw a2, 352(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    and a5, a7, a2
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 820(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    srli a5, a5, 31
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    lui a2, 8
+; RV32I-NEXT:    and a7, a0, a2
+; RV32I-NEXT:    addi t6, a5, -1
+; RV32I-NEXT:    sw t6, 820(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a5, a7
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 452(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a7, t2, 15
+; RV32I-NEXT:    sw a7, 348(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t1, 31
+; RV32I-NEXT:    sw a2, 472(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a7
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    and a5, t6, a2
+; RV32I-NEXT:    xor ra, a4, a5
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a3, a0, s8
+; RV32I-NEXT:    and a4, a0, s9
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    sw a3, 448(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    sw a4, 444(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s9, t2, 16
+; RV32I-NEXT:    slli s10, t2, 17
+; RV32I-NEXT:    and a3, a3, s9
+; RV32I-NEXT:    and a4, a4, s10
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    lui a4, 64
+; RV32I-NEXT:    and a4, a0, a4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a5, a0, s7
+; RV32I-NEXT:    addi a2, a4, -1
+; RV32I-NEXT:    sw a2, 440(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 436(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s8, t2, 18
+; RV32I-NEXT:    and a4, a2, s8
+; RV32I-NEXT:    slli s7, t2, 19
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    and a4, a5, s7
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    lui a4, 256
+; RV32I-NEXT:    and a4, a0, a4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a5, a0, s1
+; RV32I-NEXT:    addi a2, a4, -1
+; RV32I-NEXT:    sw a2, 432(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 428(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s3, t2, 20
+; RV32I-NEXT:    and a4, a2, s3
+; RV32I-NEXT:    slli s5, t2, 21
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    and a4, a5, s5
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    lui a4, 1024
+; RV32I-NEXT:    and a4, a0, a4
+; RV32I-NEXT:    xor t1, a1, a3
+; RV32I-NEXT:    seqz a1, a4
+; RV32I-NEXT:    addi a2, a1, -1
+; RV32I-NEXT:    sw a2, 424(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a1, 2048
+; RV32I-NEXT:    and a1, a0, a1
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    and a3, a0, s2
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 420(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a1, a3
+; RV32I-NEXT:    slli s2, t2, 22
+; RV32I-NEXT:    slli s1, t2, 23
+; RV32I-NEXT:    and a3, a2, s2
+; RV32I-NEXT:    and a4, a4, s1
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    sw a1, 416(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw s0, 824(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, s0
+; RV32I-NEXT:    and a4, a0, t5
+; RV32I-NEXT:    xor a1, a3, a1
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    sw a3, 412(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t6, t2, 25
+; RV32I-NEXT:    and a3, a3, t6
+; RV32I-NEXT:    lui a4, 16384
+; RV32I-NEXT:    and a4, a0, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    sw a3, 408(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t5, t2, 26
+; RV32I-NEXT:    and a3, a3, t5
+; RV32I-NEXT:    and a4, a0, t0
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    sw a3, 404(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t0, t2, 27
+; RV32I-NEXT:    and a3, a3, t0
+; RV32I-NEXT:    lui a4, 65536
+; RV32I-NEXT:    and a4, a0, a4
+; RV32I-NEXT:    xor a2, a1, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    sw a3, 400(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a7, t2, 28
+; RV32I-NEXT:    and a3, a3, a7
+; RV32I-NEXT:    lui a1, 131072
+; RV32I-NEXT:    and a1, a0, a1
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    addi a3, a1, -1
+; RV32I-NEXT:    sw a3, 396(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a1, 262144
+; RV32I-NEXT:    and a0, a0, a1
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    srli a1, t3, 31
+; RV32I-NEXT:    addi a4, a0, -1
+; RV32I-NEXT:    sw a4, 388(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a0, a1
+; RV32I-NEXT:    addi a0, a0, -1
+; RV32I-NEXT:    sw a0, 392(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t3, t2, 29
+; RV32I-NEXT:    and a5, a3, t3
+; RV32I-NEXT:    slli a3, t2, 30
+; RV32I-NEXT:    and a1, a4, a3
+; RV32I-NEXT:    slli a4, t2, 31
+; RV32I-NEXT:    xor a5, a5, a1
+; RV32I-NEXT:    and a1, a0, a4
+; RV32I-NEXT:    xor a2, t1, a2
+; RV32I-NEXT:    xor a1, a5, a1
+; RV32I-NEXT:    lw a0, 680(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    xor a0, a0, ra
+; RV32I-NEXT:    xor ra, a2, a1
+; RV32I-NEXT:    lw a1, 816(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a2, 692(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a2
+; RV32I-NEXT:    lw a2, 812(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, t2
+; RV32I-NEXT:    lw a5, 808(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t1, 688(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a5, t1
+; RV32I-NEXT:    lw a5, 804(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 684(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, t2
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    xor a2, t1, t2
+; RV32I-NEXT:    lw a5, 800(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t1, 668(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a5, t1
+; RV32I-NEXT:    lw a5, 796(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 664(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, t2
+; RV32I-NEXT:    xor t1, t1, t2
+; RV32I-NEXT:    lw a5, 792(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 660(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, t2
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, t1, t2
+; RV32I-NEXT:    lw a5, 788(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t1, 380(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a5, t1
+; RV32I-NEXT:    lw a5, 784(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 376(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, t2
+; RV32I-NEXT:    xor t1, t1, t2
+; RV32I-NEXT:    lw a5, 780(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 372(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, t2
+; RV32I-NEXT:    xor t1, t1, t2
+; RV32I-NEXT:    lw a5, 776(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 368(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, t2
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, t1, t2
+; RV32I-NEXT:    lw a5, 768(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t1, 364(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a5, t1
+; RV32I-NEXT:    lw a5, 764(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 360(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, t2
+; RV32I-NEXT:    xor t1, t1, t2
+; RV32I-NEXT:    lw a5, 760(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 356(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, t2
+; RV32I-NEXT:    xor t1, t1, t2
+; RV32I-NEXT:    lw a5, 756(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 352(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, t2
+; RV32I-NEXT:    xor t1, t1, t2
+; RV32I-NEXT:    lw a5, 752(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 348(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, t2
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, t1, t2
+; RV32I-NEXT:    lw a5, 748(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a5, s9
+; RV32I-NEXT:    lw a5, 744(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, s10
+; RV32I-NEXT:    xor t1, t1, t2
+; RV32I-NEXT:    lw a5, 740(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, s8
+; RV32I-NEXT:    xor t1, t1, t2
+; RV32I-NEXT:    lw a5, 736(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, s7
+; RV32I-NEXT:    xor t1, t1, t2
+; RV32I-NEXT:    lw a5, 724(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, s3
+; RV32I-NEXT:    xor t1, t1, t2
+; RV32I-NEXT:    lw a5, 716(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t2, a5, s5
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, t1, t2
+; RV32I-NEXT:    xor a0, ra, a0
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    lw a2, 732(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, s2
+; RV32I-NEXT:    lw a5, 728(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a5, s1
+; RV32I-NEXT:    xor a2, a2, t1
+; RV32I-NEXT:    lw a5, 720(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a5, s0
+; RV32I-NEXT:    xor a2, a2, t1
+; RV32I-NEXT:    lw a5, 712(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a5, t6
+; RV32I-NEXT:    xor a2, a2, t1
+; RV32I-NEXT:    lw a5, 708(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a5, t5
+; RV32I-NEXT:    xor a2, a2, t1
+; RV32I-NEXT:    lw a5, 704(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t0, a5, t0
+; RV32I-NEXT:    xor a2, a2, t0
+; RV32I-NEXT:    lw a5, 696(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a5, a7
+; RV32I-NEXT:    xor a2, a2, a7
+; RV32I-NEXT:    lw a5, 772(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    srli a7, a5, 1
+; RV32I-NEXT:    xor a0, a7, a0
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    lw a2, 676(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, t3
+; RV32I-NEXT:    lw a5, 672(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    lw a3, 820(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    srli a3, a0, 8
+; RV32I-NEXT:    lw a5, 652(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    srli a4, a0, 24
+; RV32I-NEXT:    or a3, a3, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    slli a2, a0, 24
+; RV32I-NEXT:    and a0, a0, a5
+; RV32I-NEXT:    slli a0, a0, 8
+; RV32I-NEXT:    srli a4, a1, 8
+; RV32I-NEXT:    or a0, a2, a0
+; RV32I-NEXT:    and a2, a4, a5
+; RV32I-NEXT:    srli a4, a1, 24
+; RV32I-NEXT:    and a5, a1, a5
+; RV32I-NEXT:    slli a1, a1, 24
+; RV32I-NEXT:    slli a5, a5, 8
+; RV32I-NEXT:    or a2, a2, a4
+; RV32I-NEXT:    or a1, a1, a5
+; RV32I-NEXT:    or a0, a0, a3
+; RV32I-NEXT:    or a1, a1, a2
+; RV32I-NEXT:    srli a2, a0, 4
+; RV32I-NEXT:    mv s5, s11
+; RV32I-NEXT:    and a0, a0, s11
+; RV32I-NEXT:    and a2, a2, s11
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    srli a3, a1, 4
+; RV32I-NEXT:    and a1, a1, s11
+; RV32I-NEXT:    and a3, a3, s11
+; RV32I-NEXT:    slli a1, a1, 4
+; RV32I-NEXT:    or a0, a2, a0
+; RV32I-NEXT:    or a1, a3, a1
+; RV32I-NEXT:    srli a2, a0, 2
+; RV32I-NEXT:    lw a4, 648(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a0, a4
+; RV32I-NEXT:    and a2, a2, a4
+; RV32I-NEXT:    slli a0, a0, 2
+; RV32I-NEXT:    srli a3, a1, 2
+; RV32I-NEXT:    and a1, a1, a4
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    slli a1, a1, 2
+; RV32I-NEXT:    or a0, a2, a0
+; RV32I-NEXT:    or a3, a3, a1
+; RV32I-NEXT:    sw a3, 268(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a1, a0, 1
+; RV32I-NEXT:    lw a2, 656(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a0, a2
+; RV32I-NEXT:    lw a2, 700(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a2
+; RV32I-NEXT:    slli a0, a0, 1
+; RV32I-NEXT:    or a0, a1, a0
+; RV32I-NEXT:    srli a1, a3, 1
+; RV32I-NEXT:    sw a1, 264(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a0, a0, 1
+; RV32I-NEXT:    slli a1, a1, 31
+; RV32I-NEXT:    or a0, a0, a1
+; RV32I-NEXT:    sw a0, 208(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a0, a6, 2
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    andi a1, a6, 1
+; RV32I-NEXT:    addi a2, a0, -1
+; RV32I-NEXT:    sw a2, 760(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a0, a1
+; RV32I-NEXT:    addi a1, a0, -1
+; RV32I-NEXT:    sw a1, 756(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a0, s4, 1
+; RV32I-NEXT:    sw a0, 244(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a0, a2, a0
+; RV32I-NEXT:    and a1, a1, s4
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    andi a1, a6, 4
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    andi a2, a6, 8
+; RV32I-NEXT:    addi a3, a1, -1
+; RV32I-NEXT:    sw a3, 744(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    addi a2, a1, -1
+; RV32I-NEXT:    sw a2, 740(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, s4, 2
+; RV32I-NEXT:    sw a1, 240(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a3, a1
+; RV32I-NEXT:    slli a3, s4, 3
+; RV32I-NEXT:    sw a3, 248(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    andi a3, a6, 16
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a4, a2, -1
+; RV32I-NEXT:    sw a4, 300(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a2, a6, 32
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    andi a3, a6, 64
+; RV32I-NEXT:    addi a5, a2, -1
+; RV32I-NEXT:    sw a5, 728(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a7, a2, -1
+; RV32I-NEXT:    sw a7, 720(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, s4, 4
+; RV32I-NEXT:    sw a2, 216(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    slli a3, s4, 5
+; RV32I-NEXT:    sw a3, 232(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    slli a4, s4, 6
+; RV32I-NEXT:    sw a4, 236(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    and a3, a7, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    xor a0, a0, a2
+; RV32I-NEXT:    andi a1, a6, 128
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    andi a2, a6, 256
+; RV32I-NEXT:    addi a3, a1, -1
+; RV32I-NEXT:    sw a3, 712(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    addi a2, a1, -1
+; RV32I-NEXT:    sw a2, 708(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, s4, 7
+; RV32I-NEXT:    sw a1, 224(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a3, a1
+; RV32I-NEXT:    slli a3, s4, 8
+; RV32I-NEXT:    sw a3, 220(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    andi a3, a6, 512
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    sw a2, 704(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, s4, 9
+; RV32I-NEXT:    sw a3, 228(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    andi a3, a6, 1024
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    sw a2, 824(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, s4, 10
+; RV32I-NEXT:    sw a3, 212(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    lw s11, 288(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a6, s11
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a4, a2, -1
+; RV32I-NEXT:    sw a4, 820(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a2, 1
+; RV32I-NEXT:    and a2, a6, a2
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    lui s8, 2
+; RV32I-NEXT:    and a3, a6, s8
+; RV32I-NEXT:    addi a7, a2, -1
+; RV32I-NEXT:    sw a7, 816(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi t1, a2, -1
+; RV32I-NEXT:    sw t1, 812(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, s4, 11
+; RV32I-NEXT:    sw a2, 184(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    slli a3, s4, 12
+; RV32I-NEXT:    sw a3, 200(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a7, a3
+; RV32I-NEXT:    slli a4, s4, 13
+; RV32I-NEXT:    sw a4, 204(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    and a3, t1, a4
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    lui a3, 4
+; RV32I-NEXT:    and a3, a6, a3
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    lui t6, 8
+; RV32I-NEXT:    and a4, a6, t6
+; RV32I-NEXT:    addi a7, a3, -1
+; RV32I-NEXT:    sw a7, 808(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a4, a3, -1
+; RV32I-NEXT:    sw a4, 804(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, s4, 14
+; RV32I-NEXT:    sw a3, 192(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a7, a3
+; RV32I-NEXT:    slli a5, s4, 15
+; RV32I-NEXT:    sw a5, 196(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    and a3, a4, a5
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    lui s1, 16
+; RV32I-NEXT:    and a1, a6, s1
+; RV32I-NEXT:    lui t3, 32
+; RV32I-NEXT:    and a3, a6, t3
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    sw a1, 800(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    sw a3, 796(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, s4, 16
+; RV32I-NEXT:    sw a5, 176(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, s4, 17
+; RV32I-NEXT:    sw a4, 188(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lui t5, 64
+; RV32I-NEXT:    and a3, a6, t5
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    lui s3, 128
+; RV32I-NEXT:    and a4, a6, s3
+; RV32I-NEXT:    addi t2, a3, -1
+; RV32I-NEXT:    sw t2, 792(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a4, a3, -1
+; RV32I-NEXT:    sw a4, 788(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, s4, 18
+; RV32I-NEXT:    sw a3, 164(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, t2, a3
+; RV32I-NEXT:    slli a5, s4, 19
+; RV32I-NEXT:    sw a5, 180(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a3, a4, a5
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lui s2, 256
+; RV32I-NEXT:    and a3, a6, s2
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    lui t1, 512
+; RV32I-NEXT:    and a4, a6, t1
+; RV32I-NEXT:    addi t2, a3, -1
+; RV32I-NEXT:    sw t2, 784(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a4, a3, -1
+; RV32I-NEXT:    sw a4, 780(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, s4, 20
+; RV32I-NEXT:    sw a3, 152(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, t2, a3
+; RV32I-NEXT:    slli a5, s4, 21
+; RV32I-NEXT:    sw a5, 172(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a3, a4, a5
+; RV32I-NEXT:    xor a0, a0, a2
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    xor a5, a0, a1
+; RV32I-NEXT:    lui t0, 1024
+; RV32I-NEXT:    and a1, a6, t0
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    lui ra, 2048
+; RV32I-NEXT:    and a2, a6, ra
+; RV32I-NEXT:    addi a3, a1, -1
+; RV32I-NEXT:    sw a3, 776(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    addi a2, a1, -1
+; RV32I-NEXT:    sw a2, 772(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, s4, 22
+; RV32I-NEXT:    sw a1, 140(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a3, a1
+; RV32I-NEXT:    slli a0, s4, 23
+; RV32I-NEXT:    sw a0, 168(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a0
+; RV32I-NEXT:    lui a7, 4096
+; RV32I-NEXT:    and a3, a6, a7
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    sw a2, 768(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s9, s4, 24
+; RV32I-NEXT:    and a2, a2, s9
+; RV32I-NEXT:    sw s9, 136(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui t2, 8192
+; RV32I-NEXT:    and a3, a6, t2
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    sw a2, 764(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a0, s4, 25
+; RV32I-NEXT:    sw a0, 160(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a0
+; RV32I-NEXT:    lui s0, 16384
+; RV32I-NEXT:    and a3, a6, s0
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    sw a2, 752(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a0, s4, 26
+; RV32I-NEXT:    sw a0, 148(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a0
+; RV32I-NEXT:    lui a0, 32768
+; RV32I-NEXT:    and a3, a6, a0
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    sw a2, 748(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a0, s4, 27
+; RV32I-NEXT:    sw a0, 156(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a0
+; RV32I-NEXT:    lui s7, 65536
+; RV32I-NEXT:    and a3, a6, s7
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    sw a2, 736(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a0, s4, 28
+; RV32I-NEXT:    sw a0, 144(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a0
+; RV32I-NEXT:    lui a0, 131072
+; RV32I-NEXT:    and a3, a6, a0
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a0, a2, -1
+; RV32I-NEXT:    sw a0, 732(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a2, 262144
+; RV32I-NEXT:    and a2, a6, a2
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    srli a3, a6, 31
+; RV32I-NEXT:    addi a4, a2, -1
+; RV32I-NEXT:    sw a4, 724(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    sw a2, 716(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, s4, 29
+; RV32I-NEXT:    sw a3, 256(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a0, a0, a3
+; RV32I-NEXT:    slli a3, s4, 30
+; RV32I-NEXT:    sw a3, 252(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    slli a6, s4, 31
+; RV32I-NEXT:    sw a6, 260(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a0, a3
+; RV32I-NEXT:    and a3, a2, a6
+; RV32I-NEXT:    xor a0, a5, a1
+; RV32I-NEXT:    xor a2, a4, a3
+; RV32I-NEXT:    xor s10, a0, a2
+; RV32I-NEXT:    andi a0, s6, 2
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    andi a1, s6, 1
+; RV32I-NEXT:    addi a3, a0, -1
+; RV32I-NEXT:    sw a3, 124(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a0, a1
+; RV32I-NEXT:    addi a1, a0, -1
+; RV32I-NEXT:    sw a1, 120(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a0, t4, 1
+; RV32I-NEXT:    sw a0, 380(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a0, a3, a0
+; RV32I-NEXT:    and a1, a1, t4
+; RV32I-NEXT:    xor a2, a1, a0
+; RV32I-NEXT:    andi a1, s6, 4
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    andi a3, s6, 8
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 116(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a1, a3
+; RV32I-NEXT:    addi a5, a1, -1
+; RV32I-NEXT:    sw a5, 112(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t4, 2
+; RV32I-NEXT:    sw a1, 376(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a4, a1
+; RV32I-NEXT:    slli a3, t4, 3
+; RV32I-NEXT:    sw a3, 372(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    andi a4, s6, 16
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 108(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a3, s6, 32
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    andi a4, s6, 64
+; RV32I-NEXT:    addi a0, a3, -1
+; RV32I-NEXT:    sw a0, 96(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a6, a3, -1
+; RV32I-NEXT:    sw a6, 100(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t4, 4
+; RV32I-NEXT:    sw a3, 368(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    slli a4, t4, 5
+; RV32I-NEXT:    sw a4, 364(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a0, a4
+; RV32I-NEXT:    slli a5, t4, 6
+; RV32I-NEXT:    sw a5, 360(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    and a4, a6, a5
+; RV32I-NEXT:    xor a0, a2, a1
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    andi a1, s6, 128
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    andi a3, s6, 256
+; RV32I-NEXT:    addi a2, a1, -1
+; RV32I-NEXT:    sw a2, 92(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a1, a3
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 88(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t4, 7
+; RV32I-NEXT:    sw a1, 356(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a2, a1
+; RV32I-NEXT:    slli a3, t4, 8
+; RV32I-NEXT:    sw a3, 352(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    andi a4, s6, 512
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a2, a3, -1
+; RV32I-NEXT:    sw a2, 84(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t4, 9
+; RV32I-NEXT:    sw a3, 348(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a2, a3
+; RV32I-NEXT:    andi a4, s6, 1024
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a2, a3, -1
+; RV32I-NEXT:    sw a2, 80(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t4, 10
+; RV32I-NEXT:    sw a3, 344(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a2, a3
+; RV32I-NEXT:    and a4, s6, s11
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a2, a3, -1
+; RV32I-NEXT:    sw a2, 76(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a3, 1
+; RV32I-NEXT:    and a3, s6, a3
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    and a4, s6, s8
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 68(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a6, a3, -1
+; RV32I-NEXT:    sw a6, 72(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t4, 11
+; RV32I-NEXT:    sw a3, 340(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a2, a3
+; RV32I-NEXT:    slli a4, t4, 12
+; RV32I-NEXT:    sw a4, 696(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    slli a5, t4, 13
+; RV32I-NEXT:    sw a5, 336(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    and a4, a6, a5
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    lui a2, 4
+; RV32I-NEXT:    and a4, s6, a2
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a5, s6, t6
+; RV32I-NEXT:    addi a6, a4, -1
+; RV32I-NEXT:    sw a6, 60(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a2, a4, -1
+; RV32I-NEXT:    sw a2, 64(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t4, 14
+; RV32I-NEXT:    sw a4, 332(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    slli a5, t4, 15
+; RV32I-NEXT:    sw a5, 328(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    and a4, a2, a5
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    and a1, s6, s1
+; RV32I-NEXT:    seqz s1, a1
+; RV32I-NEXT:    and a1, s6, t3
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    seqz t6, a1
+; RV32I-NEXT:    addi t6, t6, -1
+; RV32I-NEXT:    slli a1, t4, 16
+; RV32I-NEXT:    sw a1, 324(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, s1, a1
+; RV32I-NEXT:    slli a3, t4, 17
+; RV32I-NEXT:    sw a3, 320(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, t6, a3
+; RV32I-NEXT:    and a4, s6, t5
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz t5, a4
+; RV32I-NEXT:    addi t5, t5, -1
+; RV32I-NEXT:    slli a3, t4, 18
+; RV32I-NEXT:    sw a3, 316(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, t5, a3
+; RV32I-NEXT:    and a4, s6, s3
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz s3, a4
+; RV32I-NEXT:    addi s3, s3, -1
+; RV32I-NEXT:    sw s3, 48(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t4, 19
+; RV32I-NEXT:    sw a3, 312(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, s3, a3
+; RV32I-NEXT:    and a4, s6, s2
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz t3, a4
+; RV32I-NEXT:    addi t3, t3, -1
+; RV32I-NEXT:    slli a3, t4, 20
+; RV32I-NEXT:    sw a3, 692(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, t3, a3
+; RV32I-NEXT:    and a4, s6, t1
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz t1, a4
+; RV32I-NEXT:    addi t1, t1, -1
+; RV32I-NEXT:    slli a3, t4, 21
+; RV32I-NEXT:    sw a3, 688(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, t1, a3
+; RV32I-NEXT:    and a4, s6, t0
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    addi a2, a3, -1
+; RV32I-NEXT:    sw a2, 32(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, s6, ra
+; RV32I-NEXT:    and a3, s6, a7
+; RV32I-NEXT:    seqz a7, a1
+; RV32I-NEXT:    seqz t0, a3
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    slli a1, t4, 22
+; RV32I-NEXT:    sw a1, 308(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t4, 23
+; RV32I-NEXT:    sw a4, 684(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a2, a1
+; RV32I-NEXT:    and a3, a7, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lw ra, 296(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, t0, ra
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a3, s6, t2
+; RV32I-NEXT:    seqz a5, a3
+; RV32I-NEXT:    and a3, s6, s0
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 28(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a3
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    sw a4, 24(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t4, 25
+; RV32I-NEXT:    sw a3, 304(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    slli a6, t4, 26
+; RV32I-NEXT:    sw a6, 680(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a3, a4, a6
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lui a2, 32768
+; RV32I-NEXT:    and a3, s6, a2
+; RV32I-NEXT:    seqz s2, a3
+; RV32I-NEXT:    and a3, s6, s7
+; RV32I-NEXT:    addi s2, s2, -1
+; RV32I-NEXT:    sw s2, 20(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz s0, a3
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    sw s0, 16(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, t4, 27
+; RV32I-NEXT:    sw a3, 676(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, s2, a3
+; RV32I-NEXT:    slli a6, t4, 28
+; RV32I-NEXT:    sw a6, 672(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a3, s0, a6
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    sw s4, 132(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a3, s4, 8
+; RV32I-NEXT:    lw a5, 652(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    and a6, s4, a5
+; RV32I-NEXT:    slli a6, a6, 8
+; RV32I-NEXT:    srli t2, s4, 24
+; RV32I-NEXT:    or a3, a3, t2
+; RV32I-NEXT:    or a6, s9, a6
+; RV32I-NEXT:    xor s4, a0, a1
+; RV32I-NEXT:    or a0, a6, a3
+; RV32I-NEXT:    lui a1, 131072
+; RV32I-NEXT:    and a1, s6, a1
+; RV32I-NEXT:    lui a2, 262144
+; RV32I-NEXT:    and a3, s6, a2
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    seqz t2, a3
+; RV32I-NEXT:    addi s3, a1, -1
+; RV32I-NEXT:    addi s11, t2, -1
+; RV32I-NEXT:    srli a1, a0, 4
+; RV32I-NEXT:    and a0, a0, s5
+; RV32I-NEXT:    and a1, a1, s5
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    mv s2, t4
+; RV32I-NEXT:    slli a3, t4, 29
+; RV32I-NEXT:    sw a3, 668(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t2, t4, 30
+; RV32I-NEXT:    sw t2, 664(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, s3, a3
+; RV32I-NEXT:    and s7, s11, t2
+; RV32I-NEXT:    xor a6, a6, s7
+; RV32I-NEXT:    or a0, a1, a0
+; RV32I-NEXT:    srli a1, a0, 2
+; RV32I-NEXT:    lw t2, 648(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a0, t2
+; RV32I-NEXT:    and a1, a1, t2
+; RV32I-NEXT:    slli a0, a0, 2
+; RV32I-NEXT:    or a1, a1, a0
+; RV32I-NEXT:    srli a0, s6, 31
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    srli s7, a1, 1
+; RV32I-NEXT:    addi s0, a0, -1
+; RV32I-NEXT:    lw a4, 656(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a4
+; RV32I-NEXT:    and a1, a1, a4
+; RV32I-NEXT:    slli a0, t4, 31
+; RV32I-NEXT:    sw t4, 292(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw a0, 660(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a1, 1
+; RV32I-NEXT:    and s8, s0, a0
+; RV32I-NEXT:    xor a6, a6, s8
+; RV32I-NEXT:    or a1, s7, a1
+; RV32I-NEXT:    xor a6, s4, a6
+; RV32I-NEXT:    slli s4, a1, 1
+; RV32I-NEXT:    lw a0, 524(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s4
+; RV32I-NEXT:    lw a0, 520(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, a1
+; RV32I-NEXT:    xor a2, a6, s10
+; RV32I-NEXT:    xor a6, s7, s4
+; RV32I-NEXT:    slli s4, a1, 2
+; RV32I-NEXT:    slli s7, a1, 3
+; RV32I-NEXT:    lw a0, 516(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s4
+; RV32I-NEXT:    lw a0, 512(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s7
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    slli s7, a1, 4
+; RV32I-NEXT:    lw a0, 500(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s7
+; RV32I-NEXT:    slli s8, a1, 5
+; RV32I-NEXT:    lw a0, 496(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, a0, s8
+; RV32I-NEXT:    slli s10, a1, 6
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 492(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, a0, s10
+; RV32I-NEXT:    xor a6, a6, s4
+; RV32I-NEXT:    xor s4, s7, s8
+; RV32I-NEXT:    lw a0, 208(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    xor a0, a0, a2
+; RV32I-NEXT:    sw a0, 208(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a6, s4
+; RV32I-NEXT:    slli a6, a1, 7
+; RV32I-NEXT:    slli s4, a1, 8
+; RV32I-NEXT:    lw a0, 488(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a0, a6
+; RV32I-NEXT:    lw a0, 484(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s4
+; RV32I-NEXT:    xor a6, a6, s4
+; RV32I-NEXT:    slli s4, a1, 9
+; RV32I-NEXT:    lw a0, 480(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s4
+; RV32I-NEXT:    slli s7, a1, 10
+; RV32I-NEXT:    xor a6, a6, s4
+; RV32I-NEXT:    lw a0, 476(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s7
+; RV32I-NEXT:    xor a6, a6, s4
+; RV32I-NEXT:    slli s4, a1, 11
+; RV32I-NEXT:    lw a0, 468(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s4
+; RV32I-NEXT:    slli s7, a1, 12
+; RV32I-NEXT:    lw a0, 464(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s7
+; RV32I-NEXT:    slli s8, a1, 13
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a0, 460(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s8
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    slli s7, a1, 14
+; RV32I-NEXT:    lw a0, 456(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s7
+; RV32I-NEXT:    slli s8, a1, 15
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a0, 452(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s8
+; RV32I-NEXT:    xor a2, a2, a6
+; RV32I-NEXT:    xor a6, s4, s7
+; RV32I-NEXT:    slli s4, a1, 16
+; RV32I-NEXT:    slli s7, a1, 17
+; RV32I-NEXT:    lw a0, 448(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s4
+; RV32I-NEXT:    lw a0, 444(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s7
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    slli s7, a1, 18
+; RV32I-NEXT:    lw a0, 440(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s7
+; RV32I-NEXT:    slli s8, a1, 19
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a0, 436(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s8
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    slli s7, a1, 20
+; RV32I-NEXT:    lw a0, 432(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s7
+; RV32I-NEXT:    slli s8, a1, 21
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a0, 428(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s8
+; RV32I-NEXT:    xor a2, a2, a6
+; RV32I-NEXT:    xor a6, s4, s7
+; RV32I-NEXT:    xor a2, a2, a6
+; RV32I-NEXT:    slli a6, a1, 22
+; RV32I-NEXT:    lw a0, 424(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a0, a6
+; RV32I-NEXT:    slli s4, a1, 23
+; RV32I-NEXT:    lw a0, 420(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s4
+; RV32I-NEXT:    slli s7, a1, 24
+; RV32I-NEXT:    xor a6, a6, s4
+; RV32I-NEXT:    lw a0, 416(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s7
+; RV32I-NEXT:    xor a6, a6, s4
+; RV32I-NEXT:    slli s4, a1, 25
+; RV32I-NEXT:    lw a0, 412(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s4
+; RV32I-NEXT:    slli s7, a1, 26
+; RV32I-NEXT:    xor a6, a6, s4
+; RV32I-NEXT:    lw a0, 408(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s7
+; RV32I-NEXT:    xor a6, a6, s4
+; RV32I-NEXT:    slli s4, a1, 27
+; RV32I-NEXT:    lw a0, 404(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s4
+; RV32I-NEXT:    slli s7, a1, 28
+; RV32I-NEXT:    xor a6, a6, s4
+; RV32I-NEXT:    lw a0, 400(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s7
+; RV32I-NEXT:    xor a6, a6, s4
+; RV32I-NEXT:    slli s4, a1, 29
+; RV32I-NEXT:    lw a0, 396(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a0, s4
+; RV32I-NEXT:    slli s7, a1, 30
+; RV32I-NEXT:    lw a0, 388(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, s7
+; RV32I-NEXT:    slli a1, a1, 31
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a0, 392(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a0, a1
+; RV32I-NEXT:    xor a2, a2, a6
+; RV32I-NEXT:    xor a1, s4, a1
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    srli a2, s6, 8
+; RV32I-NEXT:    and a2, a2, a5
+; RV32I-NEXT:    srli a6, s6, 24
+; RV32I-NEXT:    or a2, a2, a6
+; RV32I-NEXT:    and a6, s6, a5
+; RV32I-NEXT:    slli a6, a6, 8
+; RV32I-NEXT:    slli t4, s6, 24
+; RV32I-NEXT:    or a6, t4, a6
+; RV32I-NEXT:    srli t4, a1, 8
+; RV32I-NEXT:    and t4, t4, a5
+; RV32I-NEXT:    srli s4, a1, 24
+; RV32I-NEXT:    or t4, t4, s4
+; RV32I-NEXT:    or a2, a6, a2
+; RV32I-NEXT:    srli a6, a2, 4
+; RV32I-NEXT:    and a2, a2, s5
+; RV32I-NEXT:    and a6, a6, s5
+; RV32I-NEXT:    slli a2, a2, 4
+; RV32I-NEXT:    or a2, a6, a2
+; RV32I-NEXT:    and a6, a1, a5
+; RV32I-NEXT:    slli a1, a1, 24
+; RV32I-NEXT:    slli a6, a6, 8
+; RV32I-NEXT:    srli s4, a2, 2
+; RV32I-NEXT:    and a2, a2, t2
+; RV32I-NEXT:    and s4, s4, t2
+; RV32I-NEXT:    mv s9, t2
+; RV32I-NEXT:    slli a2, a2, 2
+; RV32I-NEXT:    or a1, a1, a6
+; RV32I-NEXT:    or a2, s4, a2
+; RV32I-NEXT:    srli a6, a2, 1
+; RV32I-NEXT:    and a2, a2, a4
+; RV32I-NEXT:    and a6, a6, a4
+; RV32I-NEXT:    slli a2, a2, 1
+; RV32I-NEXT:    or a1, a1, t4
+; RV32I-NEXT:    or s8, a6, a2
+; RV32I-NEXT:    andi a6, s8, 2
+; RV32I-NEXT:    andi t4, s8, 1
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    seqz t4, t4
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    addi t4, t4, -1
+; RV32I-NEXT:    lw s4, 640(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a6, s4
+; RV32I-NEXT:    lw a0, 384(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, t4, a0
+; RV32I-NEXT:    xor a6, t4, a6
+; RV32I-NEXT:    andi t4, s8, 4
+; RV32I-NEXT:    seqz t4, t4
+; RV32I-NEXT:    andi s4, s8, 8
+; RV32I-NEXT:    addi t4, t4, -1
+; RV32I-NEXT:    seqz s4, s4
+; RV32I-NEXT:    lw s7, 636(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, t4, s7
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    lw s7, 632(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s4, s7
+; RV32I-NEXT:    andi s7, s8, 16
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    seqz s4, s7
+; RV32I-NEXT:    xor a6, a6, t4
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    lw t4, 628(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, s4, t4
+; RV32I-NEXT:    andi s4, s8, 32
+; RV32I-NEXT:    seqz s4, s4
+; RV32I-NEXT:    andi s7, s8, 64
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    lw a0, 624(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s4, a0
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    lw a0, 620(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s7, a0
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    andi s4, s8, 128
+; RV32I-NEXT:    xor a6, a6, t4
+; RV32I-NEXT:    seqz t4, s4
+; RV32I-NEXT:    addi t4, t4, -1
+; RV32I-NEXT:    andi s4, s8, 256
+; RV32I-NEXT:    lw a0, 616(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, t4, a0
+; RV32I-NEXT:    seqz s4, s4
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    andi s7, s8, 512
+; RV32I-NEXT:    lw a0, 612(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s4, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    lw a0, 608(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s7, a0
+; RV32I-NEXT:    andi s7, s8, 1024
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    seqz s4, s7
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    lw t2, 288(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s8, t2
+; RV32I-NEXT:    lw a0, 604(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s4, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    lw a0, 600(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s7, a0
+; RV32I-NEXT:    lui a0, 1
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    lui a0, 2
+; RV32I-NEXT:    and s10, s8, a0
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    seqz s10, s10
+; RV32I-NEXT:    lw a0, 596(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    addi s10, s10, -1
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a0, 592(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s10, a0
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lui a0, 4
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    lui a0, 8
+; RV32I-NEXT:    and s10, s8, a0
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    seqz s10, s10
+; RV32I-NEXT:    lw a0, 588(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    addi s10, s10, -1
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a0, 584(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s10, a0
+; RV32I-NEXT:    xor a6, a6, t4
+; RV32I-NEXT:    xor t4, s4, s7
+; RV32I-NEXT:    lui a0, 16
+; RV32I-NEXT:    and s4, s8, a0
+; RV32I-NEXT:    lui a0, 32
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    seqz s4, s4
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    lw a0, 580(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s4, a0
+; RV32I-NEXT:    lw a0, 576(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lui a0, 64
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    lui a0, 128
+; RV32I-NEXT:    and s10, s8, a0
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    seqz s10, s10
+; RV32I-NEXT:    lw a0, 572(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    addi s10, s10, -1
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a0, 568(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s10, a0
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lui a0, 256
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    lui a0, 512
+; RV32I-NEXT:    and s10, s8, a0
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    seqz s10, s10
+; RV32I-NEXT:    lw a0, 556(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    addi s10, s10, -1
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a0, 548(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s10, a0
+; RV32I-NEXT:    xor a6, a6, t4
+; RV32I-NEXT:    xor t4, s4, s7
+; RV32I-NEXT:    xor a6, a6, t4
+; RV32I-NEXT:    lui s6, 1024
+; RV32I-NEXT:    and t4, s8, s6
+; RV32I-NEXT:    seqz t4, t4
+; RV32I-NEXT:    lui a0, 2048
+; RV32I-NEXT:    and s4, s8, a0
+; RV32I-NEXT:    addi t4, t4, -1
+; RV32I-NEXT:    seqz s4, s4
+; RV32I-NEXT:    lw a0, 564(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, t4, a0
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    lw a0, 560(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s4, a0
+; RV32I-NEXT:    lui s10, 4096
+; RV32I-NEXT:    and s7, s8, s10
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    seqz s4, s7
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    lui a0, 8192
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    lw a0, 552(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s4, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    lw a0, 540(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s7, a0
+; RV32I-NEXT:    lui a0, 16384
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    seqz s4, s7
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    lui a0, 32768
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    lw a0, 536(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s4, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    lw a0, 532(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s7, a0
+; RV32I-NEXT:    lui a0, 65536
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    seqz s4, s7
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    lui a0, 131072
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    lw a0, 528(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s4, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    lw a0, 508(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s7, a0
+; RV32I-NEXT:    lui a0, 262144
+; RV32I-NEXT:    and s7, s8, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    srli a2, a2, 31
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    lw a0, 504(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a0, 472(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, a0
+; RV32I-NEXT:    xor a6, a6, t4
+; RV32I-NEXT:    xor a2, s4, a2
+; RV32I-NEXT:    xor a2, a6, a2
+; RV32I-NEXT:    srli a6, a1, 4
+; RV32I-NEXT:    and a6, a6, s5
+; RV32I-NEXT:    and a1, a1, s5
+; RV32I-NEXT:    slli a1, a1, 4
+; RV32I-NEXT:    srli t4, a2, 8
+; RV32I-NEXT:    or a1, a6, a1
+; RV32I-NEXT:    and a6, t4, a5
+; RV32I-NEXT:    srli t4, a2, 24
+; RV32I-NEXT:    and s4, a2, a5
+; RV32I-NEXT:    slli a2, a2, 24
+; RV32I-NEXT:    slli s4, s4, 8
+; RV32I-NEXT:    or a6, a6, t4
+; RV32I-NEXT:    or a2, a2, s4
+; RV32I-NEXT:    srli t4, a1, 2
+; RV32I-NEXT:    and a1, a1, s9
+; RV32I-NEXT:    and t4, t4, s9
+; RV32I-NEXT:    slli a1, a1, 2
+; RV32I-NEXT:    or a1, t4, a1
+; RV32I-NEXT:    or a2, a2, a6
+; RV32I-NEXT:    srli a6, a1, 1
+; RV32I-NEXT:    and a1, a1, a4
+; RV32I-NEXT:    lw a0, 700(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a6, a0
+; RV32I-NEXT:    slli a1, a1, 1
+; RV32I-NEXT:    or a1, a6, a1
+; RV32I-NEXT:    srli a6, a2, 4
+; RV32I-NEXT:    sw s5, 104(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, a6, s5
+; RV32I-NEXT:    and a2, a2, s5
+; RV32I-NEXT:    slli a2, a2, 4
+; RV32I-NEXT:    lw s5, 544(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    slli a3, s5, 1
+; RV32I-NEXT:    sw a3, 128(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 124(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, a5, a3
+; RV32I-NEXT:    lw a3, 120(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a3, s5
+; RV32I-NEXT:    or a2, a6, a2
+; RV32I-NEXT:    xor a6, s4, t4
+; RV32I-NEXT:    slli a3, s5, 2
+; RV32I-NEXT:    sw a3, 124(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s4, s5, 3
+; RV32I-NEXT:    sw s4, 120(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 116(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, a5, a3
+; RV32I-NEXT:    lw a3, 112(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a3, s4
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    slli a3, s5, 4
+; RV32I-NEXT:    sw a3, 116(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 108(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a5, a3
+; RV32I-NEXT:    slli a3, s5, 5
+; RV32I-NEXT:    sw a3, 112(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 96(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a5, a3
+; RV32I-NEXT:    slli a3, s5, 6
+; RV32I-NEXT:    sw a3, 108(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a5, 100(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a5, a3
+; RV32I-NEXT:    xor a6, a6, t4
+; RV32I-NEXT:    xor t4, s4, s7
+; RV32I-NEXT:    srli s4, a2, 2
+; RV32I-NEXT:    and a2, a2, s9
+; RV32I-NEXT:    and s4, s4, s9
+; RV32I-NEXT:    slli a2, a2, 2
+; RV32I-NEXT:    or a2, s4, a2
+; RV32I-NEXT:    xor a6, a6, t4
+; RV32I-NEXT:    slli a3, s5, 7
+; RV32I-NEXT:    sw a3, 100(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s4, s5, 8
+; RV32I-NEXT:    sw s4, 96(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 92(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, a5, a3
+; RV32I-NEXT:    lw a3, 88(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a3, s4
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    slli a3, s5, 9
+; RV32I-NEXT:    sw a3, 92(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 84(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a5, a3
+; RV32I-NEXT:    slli a3, s5, 10
+; RV32I-NEXT:    sw a3, 88(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    lw a5, 80(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a5, a3
+; RV32I-NEXT:    xor t4, t4, s4
+; RV32I-NEXT:    slli a3, s5, 11
+; RV32I-NEXT:    sw a3, 84(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 76(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a5, a3
+; RV32I-NEXT:    slli a3, s5, 12
+; RV32I-NEXT:    sw a3, 80(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 68(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a5, a3
+; RV32I-NEXT:    slli a3, s5, 13
+; RV32I-NEXT:    sw a3, 76(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a5, 72(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a5, a3
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    slli a3, s5, 14
+; RV32I-NEXT:    sw a3, 72(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 60(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a5, a3
+; RV32I-NEXT:    slli a3, s5, 15
+; RV32I-NEXT:    sw a3, 68(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor s4, s4, s7
+; RV32I-NEXT:    lw a5, 64(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a5, a3
+; RV32I-NEXT:    xor a6, a6, t4
+; RV32I-NEXT:    xor t4, s4, s7
+; RV32I-NEXT:    slli a3, s5, 16
+; RV32I-NEXT:    sw a3, 64(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s4, s5, 17
+; RV32I-NEXT:    sw s4, 60(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and s1, s1, a3
+; RV32I-NEXT:    and t6, t6, s4
+; RV32I-NEXT:    xor t6, s1, t6
+; RV32I-NEXT:    slli a3, s5, 18
+; RV32I-NEXT:    sw a3, 56(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and t5, t5, a3
+; RV32I-NEXT:    slli a3, s5, 19
+; RV32I-NEXT:    sw a3, 52(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor t5, t6, t5
+; RV32I-NEXT:    lw a5, 48(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t6, a5, a3
+; RV32I-NEXT:    xor t5, t5, t6
+; RV32I-NEXT:    slli a3, s5, 20
+; RV32I-NEXT:    sw a3, 48(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and t3, t3, a3
+; RV32I-NEXT:    slli a3, s5, 21
+; RV32I-NEXT:    sw a3, 44(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor t3, t5, t3
+; RV32I-NEXT:    and t1, t1, a3
+; RV32I-NEXT:    xor a6, a6, t4
+; RV32I-NEXT:    xor t1, t3, t1
+; RV32I-NEXT:    srli t3, a2, 1
+; RV32I-NEXT:    and a2, a2, a4
+; RV32I-NEXT:    and t3, t3, a0
+; RV32I-NEXT:    slli a2, a2, 1
+; RV32I-NEXT:    or a2, t3, a2
+; RV32I-NEXT:    xor a6, a6, t1
+; RV32I-NEXT:    slli a0, s5, 22
+; RV32I-NEXT:    sw a0, 40(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a3, s5, 23
+; RV32I-NEXT:    sw a3, 36(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a4, 32(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a4, a0
+; RV32I-NEXT:    and a7, a7, a3
+; RV32I-NEXT:    xor a7, t1, a7
+; RV32I-NEXT:    lw t1, 644(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t0, t0, t1
+; RV32I-NEXT:    xor a7, a7, t0
+; RV32I-NEXT:    slli a0, s5, 25
+; RV32I-NEXT:    sw a0, 32(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 28(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a0
+; RV32I-NEXT:    slli a0, s5, 26
+; RV32I-NEXT:    sw a0, 28(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a5, a7, a5
+; RV32I-NEXT:    lw a4, 24(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a0
+; RV32I-NEXT:    xor a4, a5, a4
+; RV32I-NEXT:    slli a0, s5, 27
+; RV32I-NEXT:    sw a0, 24(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a5, 20(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a0
+; RV32I-NEXT:    slli a0, s5, 28
+; RV32I-NEXT:    sw a0, 20(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a5, 16(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a0
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    slli a0, s5, 29
+; RV32I-NEXT:    sw a0, 16(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a3, s3, a0
+; RV32I-NEXT:    slli a0, s5, 30
+; RV32I-NEXT:    sw a0, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, s11, a0
+; RV32I-NEXT:    slli a0, s5, 31
+; RV32I-NEXT:    sw a0, 8(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    and a0, s0, a0
+; RV32I-NEXT:    xor a4, a6, a4
+; RV32I-NEXT:    xor a0, a3, a0
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    sw a1, 4(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a0, a4, a0
+; RV32I-NEXT:    lw a6, 272(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    andi a1, a6, 2
+; RV32I-NEXT:    andi a2, a6, 1
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    lw a3, 380(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a3
+; RV32I-NEXT:    and a2, a2, s2
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    andi a2, a6, 4
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    andi a3, a6, 8
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    lw a4, 376(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    lw a4, 372(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    andi a4, a6, 16
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    lw a2, 368(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a3, a2
+; RV32I-NEXT:    andi a3, a6, 32
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    andi a4, a6, 64
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    lw a5, 364(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    lw a3, 360(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    andi a3, a6, 128
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    andi a3, a6, 256
+; RV32I-NEXT:    lw a4, 356(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, a4
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    andi a4, a6, 512
+; RV32I-NEXT:    lw a5, 352(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    lw a3, 348(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    andi a4, a6, 1024
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    mv t1, t2
+; RV32I-NEXT:    and a4, a6, t2
+; RV32I-NEXT:    lw a5, 344(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    lw a3, 340(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    lui t2, 1
+; RV32I-NEXT:    and a4, a6, t2
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    lui t3, 2
+; RV32I-NEXT:    and a5, a6, t3
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    lw a7, 696(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a7
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    lw a4, 336(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    lui t4, 4
+; RV32I-NEXT:    and a4, a6, t4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    lui t5, 8
+; RV32I-NEXT:    and a5, a6, t5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    lw a7, 332(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a7
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    lw a4, 328(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lui s0, 16
+; RV32I-NEXT:    and a2, a6, s0
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    lui s3, 32
+; RV32I-NEXT:    and a3, a6, s3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    lw a4, 324(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    lw a4, 320(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    lui s5, 64
+; RV32I-NEXT:    and a4, a6, s5
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    lui a4, 128
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    lw a5, 316(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    lw a3, 312(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    lui a4, 256
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    lui s11, 256
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    lui a4, 512
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    lw a5, 692(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    lw a3, 688(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    and a4, a6, s6
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    lw a2, 308(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a3, a2
+; RV32I-NEXT:    lui a3, 2048
+; RV32I-NEXT:    and a3, a6, a3
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    and a4, a6, s10
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    lw a5, 684(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    and a3, a4, ra
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    lui a3, 8192
+; RV32I-NEXT:    and a3, a6, a3
+; RV32I-NEXT:    lui s2, 8192
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    lui a4, 16384
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    lw a5, 304(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    lw a3, 680(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    lui a3, 32768
+; RV32I-NEXT:    and a3, a6, a3
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    lui a4, 65536
+; RV32I-NEXT:    and a4, a6, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    lw a5, 676(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    lw a3, 672(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    lui a3, 131072
+; RV32I-NEXT:    and a3, a6, a3
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    lui a3, 262144
+; RV32I-NEXT:    and a3, a6, a3
+; RV32I-NEXT:    lw a4, 668(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, a4
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    srli a4, a6, 31
+; RV32I-NEXT:    lw a5, 664(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    lw a3, 660(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    lw t0, 276(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    slli a4, t0, 1
+; RV32I-NEXT:    lw a5, 760(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    lw a5, 756(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, t0
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    xor a4, a5, a4
+; RV32I-NEXT:    slli a3, t0, 2
+; RV32I-NEXT:    slli a5, t0, 3
+; RV32I-NEXT:    lw a6, 744(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a6, a3
+; RV32I-NEXT:    lw a6, 740(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    slli a5, t0, 4
+; RV32I-NEXT:    lw s6, 300(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, s6, a5
+; RV32I-NEXT:    slli a6, t0, 5
+; RV32I-NEXT:    lw a7, 728(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    slli a7, t0, 6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 720(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    xor a3, a4, a3
+; RV32I-NEXT:    xor a4, a5, a6
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a3, a3, a4
+; RV32I-NEXT:    slli a2, t0, 7
+; RV32I-NEXT:    slli a4, t0, 8
+; RV32I-NEXT:    lw a5, 712(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a5, a2
+; RV32I-NEXT:    lw a5, 708(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    slli a4, t0, 9
+; RV32I-NEXT:    lw a5, 704(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    slli a5, t0, 10
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    lw a4, 824(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    slli a4, t0, 11
+; RV32I-NEXT:    lw a5, 820(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    slli a5, t0, 12
+; RV32I-NEXT:    lw a6, 816(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    slli a6, t0, 13
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a5, 812(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    slli a5, t0, 14
+; RV32I-NEXT:    lw a6, 808(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    slli a6, t0, 15
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a5, 804(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    xor a2, a3, a2
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    sw a0, 0(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    slli a0, t0, 16
+; RV32I-NEXT:    slli a1, t0, 17
+; RV32I-NEXT:    lw a3, 800(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a3, a0
+; RV32I-NEXT:    lw a3, 796(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a3, a1
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    slli a1, t0, 18
+; RV32I-NEXT:    lw a3, 792(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a3, a1
+; RV32I-NEXT:    slli a3, t0, 19
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 788(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    slli a1, t0, 20
+; RV32I-NEXT:    lw a3, 784(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a3, a1
+; RV32I-NEXT:    slli a3, t0, 21
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 780(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    slli a1, t0, 22
+; RV32I-NEXT:    lw a3, 776(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a3, a1
+; RV32I-NEXT:    slli a3, t0, 23
+; RV32I-NEXT:    lw a4, 772(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    slli a4, t0, 24
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lw a3, 768(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    slli a3, t0, 25
+; RV32I-NEXT:    lw a4, 764(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    slli a4, t0, 26
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lw a3, 752(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    slli a3, t0, 27
+; RV32I-NEXT:    lw a4, 748(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    slli a4, t0, 28
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lw a3, 736(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    xor a0, a2, a0
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    xor s8, a0, a1
+; RV32I-NEXT:    slli a0, t0, 29
+; RV32I-NEXT:    lw a1, 732(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a1, a0
+; RV32I-NEXT:    slli a1, t0, 30
+; RV32I-NEXT:    lw a2, 724(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a2, a1
+; RV32I-NEXT:    slli a2, t0, 31
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 716(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a2
+; RV32I-NEXT:    xor ra, a0, a1
+; RV32I-NEXT:    lw s7, 280(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    andi a0, s7, 2
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    andi a1, s7, 1
+; RV32I-NEXT:    addi a2, a0, -1
+; RV32I-NEXT:    sw a2, 276(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a0, a1
+; RV32I-NEXT:    lw a1, 244(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a2, a1
+; RV32I-NEXT:    addi a2, a0, -1
+; RV32I-NEXT:    sw a2, 272(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a0, 132(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a2, a0
+; RV32I-NEXT:    andi a2, s7, 4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    addi a3, a1, -1
+; RV32I-NEXT:    sw a3, 244(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a1, s7, 8
+; RV32I-NEXT:    lw a2, 240(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a3, a2
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 240(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a1, s7, 16
+; RV32I-NEXT:    lw a3, 248(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    addi a3, a1, -1
+; RV32I-NEXT:    sw a3, 248(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a1, 216(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a3, a1
+; RV32I-NEXT:    andi a3, s7, 32
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    andi a4, s7, 64
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 216(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a3, a4
+; RV32I-NEXT:    lw a4, 232(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    addi a5, a3, -1
+; RV32I-NEXT:    sw a5, 232(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a3, 236(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a5, a3
+; RV32I-NEXT:    xor a0, a0, a2
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    andi a1, s7, 128
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    andi a2, s7, 256
+; RV32I-NEXT:    addi a3, a1, -1
+; RV32I-NEXT:    sw a3, 236(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    lw a2, 224(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a3, a2
+; RV32I-NEXT:    addi a3, a1, -1
+; RV32I-NEXT:    sw a3, 224(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a1, 220(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a3, a1
+; RV32I-NEXT:    andi a3, s7, 512
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a4, a2, -1
+; RV32I-NEXT:    sw a4, 220(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a2, s7, 1024
+; RV32I-NEXT:    lw a3, 228(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    addi a3, a2, -1
+; RV32I-NEXT:    sw a3, 228(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a2, 212(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a3, a2
+; RV32I-NEXT:    and a3, s7, t1
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz s9, a3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    addi s9, s9, -1
+; RV32I-NEXT:    sw s9, 288(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a1, 184(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, s9, a1
+; RV32I-NEXT:    and a2, s7, t2
+; RV32I-NEXT:    seqz s4, a2
+; RV32I-NEXT:    and a2, s7, t3
+; RV32I-NEXT:    addi s4, s4, -1
+; RV32I-NEXT:    sw s4, 212(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz s1, a2
+; RV32I-NEXT:    lw a2, 200(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, s4, a2
+; RV32I-NEXT:    addi s1, s1, -1
+; RV32I-NEXT:    sw s1, 200(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    lw a2, 204(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, s1, a2
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    and a2, s7, t4
+; RV32I-NEXT:    seqz t6, a2
+; RV32I-NEXT:    and a2, s7, t5
+; RV32I-NEXT:    addi s4, t6, -1
+; RV32I-NEXT:    seqz t4, a2
+; RV32I-NEXT:    lw a2, 192(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, s4, a2
+; RV32I-NEXT:    addi s1, t4, -1
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    lw a2, 196(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, s1, a2
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    and a2, s7, s0
+; RV32I-NEXT:    xor t1, a0, a1
+; RV32I-NEXT:    seqz t3, a2
+; RV32I-NEXT:    addi t6, t3, -1
+; RV32I-NEXT:    and a0, s7, s3
+; RV32I-NEXT:    lw a1, 176(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, t6, a1
+; RV32I-NEXT:    seqz a7, a0
+; RV32I-NEXT:    addi t4, a7, -1
+; RV32I-NEXT:    and a0, s7, s5
+; RV32I-NEXT:    lw a2, 188(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, t4, a2
+; RV32I-NEXT:    seqz a6, a0
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    addi t3, a6, -1
+; RV32I-NEXT:    lw a0, 164(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, t3, a0
+; RV32I-NEXT:    lui a2, 128
+; RV32I-NEXT:    and a2, s7, a2
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    seqz a5, a2
+; RV32I-NEXT:    addi a7, a5, -1
+; RV32I-NEXT:    and a1, s7, s11
+; RV32I-NEXT:    lw a2, 180(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a7, a2
+; RV32I-NEXT:    seqz s0, a1
+; RV32I-NEXT:    xor a0, a0, a2
+; RV32I-NEXT:    addi s0, s0, -1
+; RV32I-NEXT:    sw s0, 204(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a1, 152(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, s0, a1
+; RV32I-NEXT:    lui a2, 512
+; RV32I-NEXT:    and a2, s7, a2
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    seqz s3, a2
+; RV32I-NEXT:    addi s3, s3, -1
+; RV32I-NEXT:    sw s3, 192(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a1, 1024
+; RV32I-NEXT:    and a1, s7, a1
+; RV32I-NEXT:    lw a2, 172(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, s3, a2
+; RV32I-NEXT:    seqz a3, a1
+; RV32I-NEXT:    xor t5, a0, a2
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    sw a3, 196(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a0, 140(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a3, a0
+; RV32I-NEXT:    lui a1, 2048
+; RV32I-NEXT:    and a1, s7, a1
+; RV32I-NEXT:    seqz t0, a1
+; RV32I-NEXT:    and a1, s7, s10
+; RV32I-NEXT:    addi t0, t0, -1
+; RV32I-NEXT:    seqz t2, a1
+; RV32I-NEXT:    lw a1, 168(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, t0, a1
+; RV32I-NEXT:    addi t2, t2, -1
+; RV32I-NEXT:    sw t2, 188(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 136(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, t2, a1
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    and a1, s7, s2
+; RV32I-NEXT:    seqz a4, a1
+; RV32I-NEXT:    lui a1, 16384
+; RV32I-NEXT:    and a1, s7, a1
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    lw a2, 160(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a5, a2
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    xor a0, a0, a2
+; RV32I-NEXT:    lw a2, 148(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    xor s2, a0, a2
+; RV32I-NEXT:    lui a0, 32768
+; RV32I-NEXT:    and a0, s7, a0
+; RV32I-NEXT:    seqz a2, a0
+; RV32I-NEXT:    lui a0, 65536
+; RV32I-NEXT:    and a0, s7, a0
+; RV32I-NEXT:    mv a1, s7
+; RV32I-NEXT:    addi a3, a2, -1
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    lw a2, 156(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a3, a2
+; RV32I-NEXT:    addi a2, a0, -1
+; RV32I-NEXT:    xor s2, s2, s7
+; RV32I-NEXT:    lw a0, 144(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a2, a0
+; RV32I-NEXT:    xor t5, t1, t5
+; RV32I-NEXT:    xor s2, s2, s7
+; RV32I-NEXT:    xor t1, s8, ra
+; RV32I-NEXT:    xor t5, t5, s2
+; RV32I-NEXT:    lw a0, 640(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s2, 524(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s2, s2, a0
+; RV32I-NEXT:    lw a0, 520(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 384(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s5, a0, t2
+; RV32I-NEXT:    lw a0, 636(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 516(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    lw a0, 632(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 512(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s2, s5, s2
+; RV32I-NEXT:    xor s5, s7, s8
+; RV32I-NEXT:    lw a0, 628(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 500(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    lw a0, 624(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 496(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 620(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 492(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    xor s5, s7, s8
+; RV32I-NEXT:    lw a0, 616(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 488(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    lw a0, 612(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 484(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 608(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 480(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 604(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 476(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    xor s5, s7, s8
+; RV32I-NEXT:    lw a0, 600(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 468(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    lw a0, 596(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 464(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 592(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 460(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 588(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 456(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 584(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 452(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    xor s5, s7, s8
+; RV32I-NEXT:    lw a0, 580(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 448(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    lw a0, 576(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 444(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 572(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 440(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 568(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 436(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 556(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 432(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 548(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 428(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    xor s5, s7, s8
+; RV32I-NEXT:    lw a0, 564(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 424(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    lw a0, 560(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 420(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 552(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 416(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 540(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 412(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, s8, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 536(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 408(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, t2, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 532(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 404(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, t2, a0
+; RV32I-NEXT:    xor s7, s7, s8
+; RV32I-NEXT:    lw a0, 528(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 400(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s8, t2, a0
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    xor s5, s7, s8
+; RV32I-NEXT:    lui a0, 131072
+; RV32I-NEXT:    and s7, a1, a0
+; RV32I-NEXT:    lui a0, 262144
+; RV32I-NEXT:    and s8, a1, a0
+; RV32I-NEXT:    seqz s7, s7
+; RV32I-NEXT:    seqz ra, s8
+; RV32I-NEXT:    addi s7, s7, -1
+; RV32I-NEXT:    sw s7, 640(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi ra, ra, -1
+; RV32I-NEXT:    sw ra, 636(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a0, 256(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s7, a0
+; RV32I-NEXT:    lw a0, 252(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, ra, a0
+; RV32I-NEXT:    xor s7, s7, s10
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    lw a0, 508(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 396(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s5, t2, a0
+; RV32I-NEXT:    lw a0, 504(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 388(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s10, t2, a0
+; RV32I-NEXT:    xor s5, s5, s10
+; RV32I-NEXT:    srli s10, a1, 31
+; RV32I-NEXT:    lw a0, 472(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 392(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a1, a0
+; RV32I-NEXT:    seqz s10, s10
+; RV32I-NEXT:    xor s5, s5, s11
+; RV32I-NEXT:    addi s10, s10, -1
+; RV32I-NEXT:    sw s10, 632(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    lw a0, 260(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s5, s10, a0
+; RV32I-NEXT:    xor s5, s7, s5
+; RV32I-NEXT:    srli s7, s2, 8
+; RV32I-NEXT:    xor t5, t5, s5
+; RV32I-NEXT:    lw a0, 652(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s5, s7, a0
+; RV32I-NEXT:    and s7, s2, a0
+; RV32I-NEXT:    srli s11, s2, 24
+; RV32I-NEXT:    slli s2, s2, 24
+; RV32I-NEXT:    slli s7, s7, 8
+; RV32I-NEXT:    or s5, s5, s11
+; RV32I-NEXT:    or s2, s2, s7
+; RV32I-NEXT:    xor t1, t5, t1
+; RV32I-NEXT:    or t5, s2, s5
+; RV32I-NEXT:    lw a0, 0(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    xor t1, a0, t1
+; RV32I-NEXT:    srli s2, t5, 4
+; RV32I-NEXT:    lw a0, 104(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s2, s2, a0
+; RV32I-NEXT:    and t5, t5, a0
+; RV32I-NEXT:    lw a0, 700(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 264(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s5, a1, a0
+; RV32I-NEXT:    lw a1, 268(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t2, 656(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a1, t2
+; RV32I-NEXT:    slli s7, s7, 1
+; RV32I-NEXT:    slli t5, t5, 4
+; RV32I-NEXT:    or s5, s5, s7
+; RV32I-NEXT:    or t5, s2, t5
+; RV32I-NEXT:    srli s2, t5, 2
+; RV32I-NEXT:    lw a1, 648(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t5, t5, a1
+; RV32I-NEXT:    and s2, s2, a1
+; RV32I-NEXT:    slli t5, t5, 2
+; RV32I-NEXT:    or t5, s2, t5
+; RV32I-NEXT:    srli s2, s5, 1
+; RV32I-NEXT:    xor t1, s2, t1
+; RV32I-NEXT:    srli s2, t5, 1
+; RV32I-NEXT:    and s2, s2, a0
+; RV32I-NEXT:    and t5, t5, t2
+; RV32I-NEXT:    lw a0, 4(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    srli s5, a0, 1
+; RV32I-NEXT:    slli s7, t5, 1
+; RV32I-NEXT:    xor a0, t1, s5
+; RV32I-NEXT:    sw a0, 700(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    or a0, s2, s7
+; RV32I-NEXT:    sw a0, 656(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a0, 128(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 760(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s2, a1, a0
+; RV32I-NEXT:    lw a0, 544(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 756(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s5, a1, a0
+; RV32I-NEXT:    lw a0, 124(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 744(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a1, a0
+; RV32I-NEXT:    lw a0, 120(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 740(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a1, a0
+; RV32I-NEXT:    xor s2, s5, s2
+; RV32I-NEXT:    xor s5, s7, s11
+; RV32I-NEXT:    lw a0, 116(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, s6, a0
+; RV32I-NEXT:    lw a0, 112(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 728(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a1, a0
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 108(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 720(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a1, a0
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    xor s5, s7, s11
+; RV32I-NEXT:    lw a0, 100(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 712(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a1, a0
+; RV32I-NEXT:    lw a0, 96(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 708(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a1, a0
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 92(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 704(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a1, a0
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 824(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 88(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    xor s5, s7, s11
+; RV32I-NEXT:    lw a0, 820(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 84(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, a1
+; RV32I-NEXT:    lw a0, 816(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 80(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 812(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 76(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 808(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 72(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 804(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 68(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    xor s5, s7, s11
+; RV32I-NEXT:    lw a0, 800(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 64(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, a1
+; RV32I-NEXT:    lw a0, 796(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 60(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 792(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 56(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 788(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 52(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 784(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 48(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 780(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 44(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    xor s5, s7, s11
+; RV32I-NEXT:    lw a0, 776(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 40(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, a1
+; RV32I-NEXT:    lw a0, 772(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 36(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 644(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 768(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a1, a0
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 764(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 32(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 752(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 28(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 748(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 24(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 736(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 20(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s2, s2, s5
+; RV32I-NEXT:    xor s5, s7, s11
+; RV32I-NEXT:    lw a0, 732(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 16(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s7, a0, a1
+; RV32I-NEXT:    lw a0, 724(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor s7, s7, s11
+; RV32I-NEXT:    lw a0, 716(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 8(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a0, a1
+; RV32I-NEXT:    xor a0, s2, s5
+; RV32I-NEXT:    sw a0, 652(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a0, s7, s11
+; RV32I-NEXT:    sw a0, 648(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw t1, 380(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a0, 276(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s5, a0, t1
+; RV32I-NEXT:    lw a0, 292(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 272(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, a1, a0
+; RV32I-NEXT:    lw t2, 376(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a0, 244(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a0, t2
+; RV32I-NEXT:    lw t5, 372(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a0, 240(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a0, t5
+; RV32I-NEXT:    xor s5, s11, s5
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    lw s0, 368(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 248(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, s0
+; RV32I-NEXT:    lw s2, 364(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 216(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, s3, s2
+; RV32I-NEXT:    xor a1, a1, s11
+; RV32I-NEXT:    lw s3, 360(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 232(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, s6, s3
+; RV32I-NEXT:    xor a0, s5, a0
+; RV32I-NEXT:    xor a1, a1, s11
+; RV32I-NEXT:    lw s6, 356(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s5, 236(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s5, s5, s6
+; RV32I-NEXT:    lw s7, 352(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 224(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, s8, s7
+; RV32I-NEXT:    xor s5, s5, s11
+; RV32I-NEXT:    lw s8, 348(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s10, 220(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, s10, s8
+; RV32I-NEXT:    xor s5, s5, s11
+; RV32I-NEXT:    lw s10, 344(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s11, 228(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, s11, s10
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, s5, s11
+; RV32I-NEXT:    lw ra, 340(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s5, 288(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s5, s5, ra
+; RV32I-NEXT:    lw s11, 696(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s9, 212(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s11, s9, s11
+; RV32I-NEXT:    xor s5, s5, s11
+; RV32I-NEXT:    lw s11, 336(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s9, 200(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s9, s9, s11
+; RV32I-NEXT:    xor s5, s5, s9
+; RV32I-NEXT:    lw s9, 332(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, s4, s9
+; RV32I-NEXT:    xor s4, s5, s4
+; RV32I-NEXT:    lw s5, 328(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s1, s1, s5
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, s4, s1
+; RV32I-NEXT:    lw s1, 324(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t6, t6, s1
+; RV32I-NEXT:    lw s4, 320(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t4, t4, s4
+; RV32I-NEXT:    xor t4, t6, t4
+; RV32I-NEXT:    lw t6, 316(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t3, t3, t6
+; RV32I-NEXT:    xor t3, t4, t3
+; RV32I-NEXT:    lw t4, 312(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a7, t4
+; RV32I-NEXT:    xor a7, t3, a7
+; RV32I-NEXT:    lw t3, 692(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 204(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t3, a6, t3
+; RV32I-NEXT:    xor a7, a7, t3
+; RV32I-NEXT:    lw t3, 688(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 192(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t3, a6, t3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, a7, t3
+; RV32I-NEXT:    lw t3, 308(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 196(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a6, t3
+; RV32I-NEXT:    lw a7, 684(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, t0, a7
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    lw t0, 296(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 188(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a7, a7, t0
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    lw a7, 304(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a7
+; RV32I-NEXT:    xor a5, a6, a5
+; RV32I-NEXT:    lw a6, 680(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a6
+; RV32I-NEXT:    xor a4, a5, a4
+; RV32I-NEXT:    lw a5, 676(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a5
+; RV32I-NEXT:    xor a3, a4, a3
+; RV32I-NEXT:    lw a4, 672(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a2, a3, a2
+; RV32I-NEXT:    lw a1, 668(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a3, 640(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a3, a1
+; RV32I-NEXT:    lw a3, 664(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a4, 636(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lw a3, 660(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a4, 632(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    xor a2, a0, a2
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lw a0, 652(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a3, 648(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    xor a0, a0, a3
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    lw a2, 760(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a2, t1
+; RV32I-NEXT:    lw a3, 292(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a4, 756(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, a3
+; RV32I-NEXT:    lw a4, 744(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t2
+; RV32I-NEXT:    lw a5, 740(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, t5
+; RV32I-NEXT:    xor a2, a3, a2
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a3, 300(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, s0
+; RV32I-NEXT:    lw a5, 728(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s2
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 720(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s3
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a4, 712(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s6
+; RV32I-NEXT:    lw a5, 708(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a5, 704(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s8
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a5, 824(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s10
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a3, 820(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, ra
+; RV32I-NEXT:    lw a5, 816(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 696(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 812(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s11
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 808(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s9
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 804(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s5
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a4, 800(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s1
+; RV32I-NEXT:    lw a5, 796(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s4
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a5, 792(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, t6
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a5, 788(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, t4
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a5, 784(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 692(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a5, 780(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 688(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    lw a3, 776(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, t3
+; RV32I-NEXT:    lw a5, 772(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 684(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 768(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, t0
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 764(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a7
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 752(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 680(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 748(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 676(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    lw a5, 736(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 672(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    xor a3, a3, a5
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    xor a2, a2, a3
+; RV32I-NEXT:    lw a1, 732(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a3, 668(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a3
+; RV32I-NEXT:    lw a3, 724(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a4, 664(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lw a3, 716(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a4, 660(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lw a3, 656(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    srli a3, a3, 1
+; RV32I-NEXT:    xor a0, a3, a0
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    lw a2, 284(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    sw a1, 0(a2)
+; RV32I-NEXT:    sw a0, 4(a2)
+; RV32I-NEXT:    lw a0, 208(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    sw a0, 8(a2)
+; RV32I-NEXT:    lw a0, 700(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    sw a0, 12(a2)
+; RV32I-NEXT:    lw ra, 876(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s0, 872(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s1, 868(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s2, 864(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 860(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s4, 856(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s5, 852(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 848(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 844(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 840(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s9, 836(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s10, 832(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s11, 828(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    .cfi_restore ra
+; RV32I-NEXT:    .cfi_restore s0
+; RV32I-NEXT:    .cfi_restore s1
+; RV32I-NEXT:    .cfi_restore s2
+; RV32I-NEXT:    .cfi_restore s3
+; RV32I-NEXT:    .cfi_restore s4
+; RV32I-NEXT:    .cfi_restore s5
+; RV32I-NEXT:    .cfi_restore s6
+; RV32I-NEXT:    .cfi_restore s7
+; RV32I-NEXT:    .cfi_restore s8
+; RV32I-NEXT:    .cfi_restore s9
+; RV32I-NEXT:    .cfi_restore s10
+; RV32I-NEXT:    .cfi_restore s11
+; RV32I-NEXT:    addi sp, sp, 880
+; RV32I-NEXT:    .cfi_def_cfa_offset 0
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: clmul_i128:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi sp, sp, -1152
+; RV64I-NEXT:    .cfi_def_cfa_offset 1152
+; RV64I-NEXT:    sd ra, 1144(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s0, 1136(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s1, 1128(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s2, 1120(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s3, 1112(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s4, 1104(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s5, 1096(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s6, 1088(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s7, 1080(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s8, 1072(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s9, 1064(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s10, 1056(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s11, 1048(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    .cfi_offset ra, -8
+; RV64I-NEXT:    .cfi_offset s0, -16
+; RV64I-NEXT:    .cfi_offset s1, -24
+; RV64I-NEXT:    .cfi_offset s2, -32
+; RV64I-NEXT:    .cfi_offset s3, -40
+; RV64I-NEXT:    .cfi_offset s4, -48
+; RV64I-NEXT:    .cfi_offset s5, -56
+; RV64I-NEXT:    .cfi_offset s6, -64
+; RV64I-NEXT:    .cfi_offset s7, -72
+; RV64I-NEXT:    .cfi_offset s8, -80
+; RV64I-NEXT:    .cfi_offset s9, -88
+; RV64I-NEXT:    .cfi_offset s10, -96
+; RV64I-NEXT:    .cfi_offset s11, -104
+; RV64I-NEXT:    mv s5, a3
+; RV64I-NEXT:    mv a3, a0
+; RV64I-NEXT:    srli a5, a0, 24
+; RV64I-NEXT:    lui a4, 4080
+; RV64I-NEXT:    and a7, a5, a4
+; RV64I-NEXT:    li a6, 255
+; RV64I-NEXT:    srli a5, a0, 8
+; RV64I-NEXT:    slli a6, a6, 24
+; RV64I-NEXT:    and t0, a5, a6
+; RV64I-NEXT:    lui a0, 16
+; RV64I-NEXT:    srli t1, a3, 40
+; RV64I-NEXT:    addi a5, a0, -256
+; RV64I-NEXT:    lui s6, 16
+; RV64I-NEXT:    and t1, t1, a5
+; RV64I-NEXT:    srli t2, a3, 56
+; RV64I-NEXT:    or a7, t0, a7
+; RV64I-NEXT:    or t0, t1, t2
+; RV64I-NEXT:    or a7, a7, t0
+; RV64I-NEXT:    and t0, a3, a4
+; RV64I-NEXT:    srliw t1, a3, 24
+; RV64I-NEXT:    slli t0, t0, 24
+; RV64I-NEXT:    slli t1, t1, 32
+; RV64I-NEXT:    or t0, t0, t1
+; RV64I-NEXT:    and t1, a3, a5
+; RV64I-NEXT:    slli t1, t1, 40
+; RV64I-NEXT:    slli a0, a3, 56
+; RV64I-NEXT:    sd a0, 944(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    or t1, a0, t1
+; RV64I-NEXT:    lui t2, 61681
+; RV64I-NEXT:    or t0, t1, t0
+; RV64I-NEXT:    addi t1, t2, -241
+; RV64I-NEXT:    or t0, t0, a7
+; RV64I-NEXT:    slli a7, t1, 32
+; RV64I-NEXT:    srli t2, t0, 4
+; RV64I-NEXT:    add a7, t1, a7
+; RV64I-NEXT:    and t1, t2, a7
+; RV64I-NEXT:    and t0, t0, a7
+; RV64I-NEXT:    lui t2, 209715
+; RV64I-NEXT:    slli t0, t0, 4
+; RV64I-NEXT:    addi t2, t2, 819
+; RV64I-NEXT:    or t1, t1, t0
+; RV64I-NEXT:    slli t0, t2, 32
+; RV64I-NEXT:    srli t3, t1, 2
+; RV64I-NEXT:    add t0, t2, t0
+; RV64I-NEXT:    and t2, t3, t0
+; RV64I-NEXT:    and t1, t1, t0
+; RV64I-NEXT:    lui t3, 349525
+; RV64I-NEXT:    slli t1, t1, 2
+; RV64I-NEXT:    addi t3, t3, 1365
+; RV64I-NEXT:    or t2, t2, t1
+; RV64I-NEXT:    slli t1, t3, 32
+; RV64I-NEXT:    srli t4, t2, 1
+; RV64I-NEXT:    add t1, t3, t1
+; RV64I-NEXT:    and t3, t4, t1
+; RV64I-NEXT:    srli t4, a2, 24
+; RV64I-NEXT:    srli t5, a2, 8
+; RV64I-NEXT:    and t4, t4, a4
+; RV64I-NEXT:    and t5, t5, a6
+; RV64I-NEXT:    or t4, t5, t4
+; RV64I-NEXT:    srli t5, a2, 40
+; RV64I-NEXT:    and t5, t5, a5
+; RV64I-NEXT:    srli t6, a2, 56
+; RV64I-NEXT:    or t5, t5, t6
+; RV64I-NEXT:    and t6, a2, a4
+; RV64I-NEXT:    slli t6, t6, 24
+; RV64I-NEXT:    srliw s0, a2, 24
+; RV64I-NEXT:    slli s0, s0, 32
+; RV64I-NEXT:    and s1, a2, a5
+; RV64I-NEXT:    slli s1, s1, 40
+; RV64I-NEXT:    slli s2, a2, 56
+; RV64I-NEXT:    or t6, t6, s0
+; RV64I-NEXT:    or s0, s2, s1
+; RV64I-NEXT:    or t4, t4, t5
+; RV64I-NEXT:    or t5, s0, t6
+; RV64I-NEXT:    and t2, t2, t1
+; RV64I-NEXT:    or t4, t5, t4
+; RV64I-NEXT:    srli t5, t4, 4
+; RV64I-NEXT:    and t4, t4, a7
+; RV64I-NEXT:    and t5, t5, a7
+; RV64I-NEXT:    slli t4, t4, 4
+; RV64I-NEXT:    slli t2, t2, 1
+; RV64I-NEXT:    or t4, t5, t4
+; RV64I-NEXT:    srli t5, t4, 2
+; RV64I-NEXT:    and t4, t4, t0
+; RV64I-NEXT:    and t5, t5, t0
+; RV64I-NEXT:    slli t4, t4, 2
+; RV64I-NEXT:    or t2, t3, t2
+; RV64I-NEXT:    or t3, t5, t4
+; RV64I-NEXT:    srli t4, t3, 1
+; RV64I-NEXT:    and t3, t3, t1
+; RV64I-NEXT:    and t4, t4, t1
+; RV64I-NEXT:    slli t3, t3, 1
+; RV64I-NEXT:    slli t5, t2, 1
+; RV64I-NEXT:    or t4, t4, t3
+; RV64I-NEXT:    andi t6, t4, 2
+; RV64I-NEXT:    andi s0, t4, 1
+; RV64I-NEXT:    seqz t6, t6
+; RV64I-NEXT:    seqz s0, s0
+; RV64I-NEXT:    addi t6, t6, -1
+; RV64I-NEXT:    addi s0, s0, -1
+; RV64I-NEXT:    and t5, t6, t5
+; RV64I-NEXT:    and t6, s0, t2
+; RV64I-NEXT:    slli s0, t2, 2
+; RV64I-NEXT:    andi s1, t4, 4
+; RV64I-NEXT:    seqz s1, s1
+; RV64I-NEXT:    andi s2, t4, 8
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    seqz s2, s2
+; RV64I-NEXT:    slli s3, t2, 3
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    and s0, s1, s0
+; RV64I-NEXT:    and s1, s2, s3
+; RV64I-NEXT:    xor t5, t6, t5
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    xor t5, t5, s0
+; RV64I-NEXT:    andi t6, t4, 16
+; RV64I-NEXT:    slli s0, t2, 4
+; RV64I-NEXT:    seqz t6, t6
+; RV64I-NEXT:    addi t6, t6, -1
+; RV64I-NEXT:    andi s1, t4, 32
+; RV64I-NEXT:    and t6, t6, s0
+; RV64I-NEXT:    seqz s0, s1
+; RV64I-NEXT:    slli s1, t2, 5
+; RV64I-NEXT:    addi s0, s0, -1
+; RV64I-NEXT:    and s0, s0, s1
+; RV64I-NEXT:    andi s1, t4, 64
+; RV64I-NEXT:    xor t6, t6, s0
+; RV64I-NEXT:    seqz s0, s1
+; RV64I-NEXT:    slli s1, t2, 6
+; RV64I-NEXT:    addi s0, s0, -1
+; RV64I-NEXT:    and s0, s0, s1
+; RV64I-NEXT:    andi s1, t4, 128
+; RV64I-NEXT:    xor t6, t6, s0
+; RV64I-NEXT:    seqz s0, s1
+; RV64I-NEXT:    slli s1, t2, 7
+; RV64I-NEXT:    addi s0, s0, -1
+; RV64I-NEXT:    and s0, s0, s1
+; RV64I-NEXT:    andi s1, t4, 256
+; RV64I-NEXT:    slli s2, t2, 8
+; RV64I-NEXT:    seqz s1, s1
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    andi s3, t4, 512
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    slli s3, t2, 9
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    and s1, s2, s3
+; RV64I-NEXT:    xor t6, t5, t6
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    slli s1, t2, 10
+; RV64I-NEXT:    andi t5, t4, 1024
+; RV64I-NEXT:    seqz s2, t5
+; RV64I-NEXT:    li t5, 1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli a0, t5, 11
+; RV64I-NEXT:    sd a0, 424(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    slli s2, t2, 11
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    lui a0, 1
+; RV64I-NEXT:    slli s2, t2, 12
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    seqz s3, s3
+; RV64I-NEXT:    lui a0, 2
+; RV64I-NEXT:    addi s3, s3, -1
+; RV64I-NEXT:    and s4, t4, a0
+; RV64I-NEXT:    and s2, s3, s2
+; RV64I-NEXT:    seqz s3, s4
+; RV64I-NEXT:    slli s4, t2, 13
+; RV64I-NEXT:    addi s3, s3, -1
+; RV64I-NEXT:    xor s1, s1, s2
+; RV64I-NEXT:    and s2, s3, s4
+; RV64I-NEXT:    xor s1, s1, s2
+; RV64I-NEXT:    lui a0, 4
+; RV64I-NEXT:    slli s2, t2, 14
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    seqz s3, s3
+; RV64I-NEXT:    lui a0, 8
+; RV64I-NEXT:    addi s3, s3, -1
+; RV64I-NEXT:    and s4, t4, a0
+; RV64I-NEXT:    and s2, s3, s2
+; RV64I-NEXT:    seqz s3, s4
+; RV64I-NEXT:    slli s4, t2, 15
+; RV64I-NEXT:    addi s3, s3, -1
+; RV64I-NEXT:    xor s1, s1, s2
+; RV64I-NEXT:    and s2, s3, s4
+; RV64I-NEXT:    xor t6, t6, s0
+; RV64I-NEXT:    xor s0, s1, s2
+; RV64I-NEXT:    xor t6, t6, s0
+; RV64I-NEXT:    slli s0, t2, 16
+; RV64I-NEXT:    and s1, t4, s6
+; RV64I-NEXT:    seqz s1, s1
+; RV64I-NEXT:    lui a0, 32
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    and s0, s1, s0
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    slli s2, t2, 17
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    lui a0, 64
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    and s1, t4, a0
+; RV64I-NEXT:    slli s2, t2, 18
+; RV64I-NEXT:    seqz s1, s1
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    lui a0, 128
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    lui a0, 256
+; RV64I-NEXT:    slli s2, t2, 19
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 20
+; RV64I-NEXT:    lui a0, 512
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli s2, t2, 21
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    lui a0, 1024
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    and s1, t4, a0
+; RV64I-NEXT:    xor t6, t6, s0
+; RV64I-NEXT:    seqz s0, s1
+; RV64I-NEXT:    addi s0, s0, -1
+; RV64I-NEXT:    lui a0, 2048
+; RV64I-NEXT:    slli s1, t2, 22
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    and s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    lui s3, 4096
+; RV64I-NEXT:    slli s2, t2, 23
+; RV64I-NEXT:    and s3, t4, s3
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 24
+; RV64I-NEXT:    lui a0, 8192
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    lui a0, 16384
+; RV64I-NEXT:    slli s2, t2, 25
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 26
+; RV64I-NEXT:    lui a0, 32768
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    lui s3, 65536
+; RV64I-NEXT:    slli s2, t2, 27
+; RV64I-NEXT:    and s3, t4, s3
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s3, t2, 28
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    and s1, s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    lui a0, 131072
+; RV64I-NEXT:    xor t6, t6, s0
+; RV64I-NEXT:    and s0, t4, a0
+; RV64I-NEXT:    seqz s0, s0
+; RV64I-NEXT:    lui a0, 262144
+; RV64I-NEXT:    addi s0, s0, -1
+; RV64I-NEXT:    and s1, t4, a0
+; RV64I-NEXT:    seqz s1, s1
+; RV64I-NEXT:    slli s2, t2, 29
+; RV64I-NEXT:    and s0, s0, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli s2, t2, 30
+; RV64I-NEXT:    sraiw s3, t4, 31
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 31
+; RV64I-NEXT:    slli s6, t5, 32
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, s6
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli a0, t5, 33
+; RV64I-NEXT:    sd a0, 408(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s2, t2, 32
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 33
+; RV64I-NEXT:    slli s10, t5, 34
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, s10
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli s11, t5, 35
+; RV64I-NEXT:    slli s2, t2, 34
+; RV64I-NEXT:    and s3, t4, s11
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 35
+; RV64I-NEXT:    slli ra, t5, 36
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, ra
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli s2, t2, 36
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    slli a0, t5, 37
+; RV64I-NEXT:    sd a0, 400(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    and s1, t4, a0
+; RV64I-NEXT:    xor t6, t6, s0
+; RV64I-NEXT:    seqz s0, s1
+; RV64I-NEXT:    addi s0, s0, -1
+; RV64I-NEXT:    slli s9, t5, 38
+; RV64I-NEXT:    slli s1, t2, 37
+; RV64I-NEXT:    and s2, t4, s9
+; RV64I-NEXT:    and s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli s8, t5, 39
+; RV64I-NEXT:    slli s2, t2, 38
+; RV64I-NEXT:    and s3, t4, s8
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 39
+; RV64I-NEXT:    slli a0, t5, 40
+; RV64I-NEXT:    sd a0, 392(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli a0, t5, 41
+; RV64I-NEXT:    sd a0, 384(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s2, t2, 40
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 41
+; RV64I-NEXT:    slli a0, t5, 42
+; RV64I-NEXT:    sd a0, 376(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli a0, t5, 43
+; RV64I-NEXT:    sd a0, 368(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s2, t2, 42
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 43
+; RV64I-NEXT:    slli a0, t5, 44
+; RV64I-NEXT:    sd a0, 360(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli a0, t5, 45
+; RV64I-NEXT:    sd a0, 352(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s2, t2, 44
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s3, t2, 45
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    and s1, s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    slli a0, t5, 46
+; RV64I-NEXT:    sd a0, 336(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor t6, t6, s0
+; RV64I-NEXT:    and s0, t4, a0
+; RV64I-NEXT:    seqz s0, s0
+; RV64I-NEXT:    slli a0, t5, 47
+; RV64I-NEXT:    sd a0, 344(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    addi s0, s0, -1
+; RV64I-NEXT:    and s1, t4, a0
+; RV64I-NEXT:    seqz s1, s1
+; RV64I-NEXT:    slli s2, t2, 46
+; RV64I-NEXT:    and s0, s0, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli s2, t2, 47
+; RV64I-NEXT:    slli a0, t5, 48
+; RV64I-NEXT:    sd a0, 328(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli a0, t5, 49
+; RV64I-NEXT:    sd a0, 320(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s2, t2, 48
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 49
+; RV64I-NEXT:    slli a0, t5, 50
+; RV64I-NEXT:    sd a0, 1000(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli a0, t5, 51
+; RV64I-NEXT:    sd a0, 1040(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s2, t2, 50
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 51
+; RV64I-NEXT:    slli a0, t5, 52
+; RV64I-NEXT:    sd a0, 1032(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli a0, t5, 53
+; RV64I-NEXT:    sd a0, 1024(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s2, t2, 52
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 53
+; RV64I-NEXT:    slli a0, t5, 54
+; RV64I-NEXT:    sd a0, 1016(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, a0
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli a0, t5, 55
+; RV64I-NEXT:    sd a0, 1008(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s2, t2, 54
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli s1, t2, 55
+; RV64I-NEXT:    slli s7, t5, 56
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    and s2, t4, s7
+; RV64I-NEXT:    sd s7, 304(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor s0, s0, s1
+; RV64I-NEXT:    seqz s1, s2
+; RV64I-NEXT:    addi s1, s1, -1
+; RV64I-NEXT:    slli a0, t5, 57
+; RV64I-NEXT:    sd a0, 288(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s2, t2, 56
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    and s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli a0, t5, 58
+; RV64I-NEXT:    sd a0, 992(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s3, t2, 57
+; RV64I-NEXT:    and s4, t4, a0
+; RV64I-NEXT:    and s2, s2, s3
+; RV64I-NEXT:    seqz s3, s4
+; RV64I-NEXT:    xor s1, s1, s2
+; RV64I-NEXT:    addi s3, s3, -1
+; RV64I-NEXT:    slli s2, t2, 58
+; RV64I-NEXT:    slli a0, t5, 59
+; RV64I-NEXT:    sd a0, 984(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and s2, s3, s2
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    xor s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli a0, t5, 60
+; RV64I-NEXT:    sd a0, 976(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli s3, t2, 59
+; RV64I-NEXT:    and s4, t4, a0
+; RV64I-NEXT:    and s2, s2, s3
+; RV64I-NEXT:    seqz s3, s4
+; RV64I-NEXT:    xor s1, s1, s2
+; RV64I-NEXT:    addi s3, s3, -1
+; RV64I-NEXT:    slli s2, t2, 60
+; RV64I-NEXT:    slli a0, t5, 61
+; RV64I-NEXT:    sd a0, 960(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and s2, s3, s2
+; RV64I-NEXT:    and s3, t4, a0
+; RV64I-NEXT:    xor s1, s1, s2
+; RV64I-NEXT:    seqz s2, s3
+; RV64I-NEXT:    addi s2, s2, -1
+; RV64I-NEXT:    slli t5, t5, 62
+; RV64I-NEXT:    sd t5, 968(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and t4, t4, t5
+; RV64I-NEXT:    slli t5, t2, 61
+; RV64I-NEXT:    and t5, s2, t5
+; RV64I-NEXT:    seqz t4, t4
+; RV64I-NEXT:    xor t5, s1, t5
+; RV64I-NEXT:    addi t4, t4, -1
+; RV64I-NEXT:    srli t3, t3, 63
+; RV64I-NEXT:    slli s1, t2, 62
+; RV64I-NEXT:    and t4, t4, s1
+; RV64I-NEXT:    seqz t3, t3
+; RV64I-NEXT:    slli t2, t2, 63
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    xor t4, t5, t4
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    xor t3, t6, s0
+; RV64I-NEXT:    xor t2, t4, t2
+; RV64I-NEXT:    xor t2, t3, t2
+; RV64I-NEXT:    srli t3, t2, 40
+; RV64I-NEXT:    and t3, t3, a5
+; RV64I-NEXT:    srli t4, t2, 56
+; RV64I-NEXT:    or t3, t3, t4
+; RV64I-NEXT:    srli t4, t2, 24
+; RV64I-NEXT:    srli t5, t2, 8
+; RV64I-NEXT:    and a6, t5, a6
+; RV64I-NEXT:    and t4, t4, a4
+; RV64I-NEXT:    or a6, a6, t4
+; RV64I-NEXT:    srliw t4, t2, 24
+; RV64I-NEXT:    slli t4, t4, 32
+; RV64I-NEXT:    and a4, t2, a4
+; RV64I-NEXT:    slli a4, a4, 24
+; RV64I-NEXT:    and a5, t2, a5
+; RV64I-NEXT:    slli t2, t2, 56
+; RV64I-NEXT:    slli a5, a5, 40
+; RV64I-NEXT:    or a4, a4, t4
+; RV64I-NEXT:    or a5, t2, a5
+; RV64I-NEXT:    or a6, a6, t3
+; RV64I-NEXT:    or a4, a5, a4
+; RV64I-NEXT:    or a4, a4, a6
+; RV64I-NEXT:    srli a5, a4, 4
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    slli a4, a4, 4
+; RV64I-NEXT:    or a4, a5, a4
+; RV64I-NEXT:    srli a5, a4, 2
+; RV64I-NEXT:    and a4, a4, t0
+; RV64I-NEXT:    and a5, a5, t0
+; RV64I-NEXT:    slli a4, a4, 2
+; RV64I-NEXT:    or a4, a5, a4
+; RV64I-NEXT:    srli a5, a4, 1
+; RV64I-NEXT:    and a5, a5, t1
+; RV64I-NEXT:    and a4, a4, t1
+; RV64I-NEXT:    andi a6, a2, 2
+; RV64I-NEXT:    slli a4, a4, 1
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    or a4, a5, a4
+; RV64I-NEXT:    sd a4, 952(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    sd a6, 936(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a4, a1, 1
+; RV64I-NEXT:    andi a5, a2, 1
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a0, a5, -1
+; RV64I-NEXT:    sd a0, 928(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    andi a5, a2, 4
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    and a6, a0, a1
+; RV64I-NEXT:    xor a4, a6, a4
+; RV64I-NEXT:    addi a0, a5, -1
+; RV64I-NEXT:    sd a0, 920(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 2
+; RV64I-NEXT:    andi a6, a2, 8
+; RV64I-NEXT:    and a5, a0, a5
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a0, a6, -1
+; RV64I-NEXT:    sd a0, 912(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 3
+; RV64I-NEXT:    and a6, a0, a6
+; RV64I-NEXT:    andi a7, a2, 16
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a0, a6, -1
+; RV64I-NEXT:    sd a0, 904(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 4
+; RV64I-NEXT:    andi a6, a2, 32
+; RV64I-NEXT:    and a5, a0, a5
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a0, a6, -1
+; RV64I-NEXT:    sd a0, 896(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 5
+; RV64I-NEXT:    and a6, a0, a6
+; RV64I-NEXT:    andi a7, a2, 64
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a0, a6, -1
+; RV64I-NEXT:    sd a0, 888(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 6
+; RV64I-NEXT:    and a6, a0, a6
+; RV64I-NEXT:    andi a7, a2, 128
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a0, a6, -1
+; RV64I-NEXT:    sd a0, 880(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 7
+; RV64I-NEXT:    andi a6, a2, 256
+; RV64I-NEXT:    and a5, a0, a5
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a0, a6, -1
+; RV64I-NEXT:    sd a0, 872(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 8
+; RV64I-NEXT:    and a6, a0, a6
+; RV64I-NEXT:    andi a7, a2, 512
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a0, a6, -1
+; RV64I-NEXT:    sd a0, 864(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 9
+; RV64I-NEXT:    and a6, a0, a6
+; RV64I-NEXT:    andi a7, a2, 1024
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a0, a6, -1
+; RV64I-NEXT:    sd a0, 856(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 10
+; RV64I-NEXT:    and a6, a0, a6
+; RV64I-NEXT:    ld a0, 424(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, a0
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 848(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 11
+; RV64I-NEXT:    lui a6, 1
+; RV64I-NEXT:    and a6, a2, a6
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 840(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 12
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 2
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 832(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 13
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 4
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 824(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 14
+; RV64I-NEXT:    lui a6, 8
+; RV64I-NEXT:    and a6, a2, a6
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 816(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 15
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 16
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 808(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 16
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 32
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 800(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 17
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 64
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 792(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 18
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 128
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 784(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 19
+; RV64I-NEXT:    lui a6, 256
+; RV64I-NEXT:    and a6, a2, a6
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 776(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 20
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 512
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 768(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 21
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 1024
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 760(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 22
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 2048
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 752(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 23
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 4096
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 744(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 24
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 8192
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 736(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 25
+; RV64I-NEXT:    lui a6, 16384
+; RV64I-NEXT:    and a6, a2, a6
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 728(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 26
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 32768
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 720(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 27
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 65536
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 712(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 28
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 131072
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 704(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 29
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    lui a7, 262144
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 696(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 30
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    sraiw a7, a2, 31
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 688(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 31
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    and a7, a2, s6
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 680(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 32
+; RV64I-NEXT:    ld t0, 408(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a2, t0
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 672(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 33
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    and a7, a2, s10
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 664(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 34
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    and a7, a2, s11
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 656(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 35
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    and a7, a2, ra
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 648(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 36
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld t1, 400(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, t1
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 640(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 37
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    and a7, a2, s9
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 632(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 38
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    and a7, a2, s8
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 624(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 39
+; RV64I-NEXT:    ld t2, 392(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a2, t2
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 616(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 40
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld t3, 384(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, t3
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 608(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 41
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld t4, 376(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, t4
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 600(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 42
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld t5, 368(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, t5
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 592(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 43
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld t6, 360(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, t6
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 584(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 44
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld s0, 352(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, s0
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 576(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 45
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld s1, 336(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, s1
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 568(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 46
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld s3, 344(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, s3
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 560(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 47
+; RV64I-NEXT:    ld s2, 328(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a2, s2
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 552(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 48
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld s4, 320(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, s4
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 544(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 49
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld a7, 1000(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 536(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 50
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld a7, 1040(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 528(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 51
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld a7, 1032(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 520(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 52
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld a7, 1024(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 512(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 53
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld a7, 1016(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 504(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 54
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    ld a7, 1008(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a2, a7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    addi a7, a6, -1
+; RV64I-NEXT:    sd a7, 496(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a1, 55
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    and a7, a2, s7
+; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a6, a7
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    sd a4, 416(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    addi a6, a6, -1
+; RV64I-NEXT:    sd a6, 488(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a4, a1, 56
+; RV64I-NEXT:    ld a7, 288(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, a2, a7
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a6, a5, -1
+; RV64I-NEXT:    sd a6, 480(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 57
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    ld a6, 992(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a2, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a6, a5, -1
+; RV64I-NEXT:    sd a6, 472(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 58
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    ld a6, 984(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a2, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a6, a5, -1
+; RV64I-NEXT:    sd a6, 464(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 59
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    ld a6, 976(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a2, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a6, a5, -1
+; RV64I-NEXT:    sd a6, 456(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 60
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    ld a6, 960(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a2, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a6, a5, -1
+; RV64I-NEXT:    sd a6, 448(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 61
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    ld a6, 968(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a2, a6
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    addi a6, a5, -1
+; RV64I-NEXT:    sd a6, 440(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a5, a1, 62
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    srli a2, a2, 63
+; RV64I-NEXT:    xor a4, a4, a5
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    slli a1, a1, 63
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    sd a2, 432(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a1, a2, a1
+; RV64I-NEXT:    andi a2, s5, 2
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    andi a5, s5, 1
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a3, 1
+; RV64I-NEXT:    sd a6, 312(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a2, a2, a6
+; RV64I-NEXT:    and a5, a5, a3
+; RV64I-NEXT:    xor a1, a4, a1
+; RV64I-NEXT:    sd a1, 296(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a2, a5, a2
+; RV64I-NEXT:    andi a1, s5, 4
+; RV64I-NEXT:    andi a4, s5, 8
+; RV64I-NEXT:    seqz a1, a1
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a1, a1, -1
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a5, a3, 2
+; RV64I-NEXT:    sd a5, 280(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli a6, a3, 3
+; RV64I-NEXT:    sd a6, 272(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a1, a1, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    xor a1, a1, a4
+; RV64I-NEXT:    andi a4, s5, 16
+; RV64I-NEXT:    xor a1, a2, a1
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    andi a4, s5, 32
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, a3, 4
+; RV64I-NEXT:    sd a5, 264(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a6, a3, 5
+; RV64I-NEXT:    sd a6, 256(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    andi a5, s5, 64
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a3, 6
+; RV64I-NEXT:    sd a6, 248(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a6
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    andi a4, s5, 128
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    andi a4, s5, 256
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, a3, 7
+; RV64I-NEXT:    sd a5, 240(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a6, a3, 8
+; RV64I-NEXT:    sd a6, 232(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    andi a5, s5, 512
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a6, a3, 9
+; RV64I-NEXT:    sd a6, 224(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    andi a4, s5, 1024
+; RV64I-NEXT:    and a5, a5, a6
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a6, a3, 10
+; RV64I-NEXT:    sd a6, 216(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, s5, a0
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    lui a0, 1
+; RV64I-NEXT:    and a4, s5, a0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a0, a3, 11
+; RV64I-NEXT:    sd a0, 424(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a2, a2, a0
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 2
+; RV64I-NEXT:    and a5, s5, a5
+; RV64I-NEXT:    slli a0, a3, 12
+; RV64I-NEXT:    sd a0, 208(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a0, a3, 13
+; RV64I-NEXT:    sd a0, 200(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a0
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    lui a0, 4
+; RV64I-NEXT:    and a4, s5, a0
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    lui a4, 8
+; RV64I-NEXT:    and a4, s5, a4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a0, a3, 14
+; RV64I-NEXT:    sd a0, 192(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a2, a2, a0
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 16
+; RV64I-NEXT:    and a5, s5, a5
+; RV64I-NEXT:    slli a0, a3, 15
+; RV64I-NEXT:    sd a0, 184(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a0, 32
+; RV64I-NEXT:    and a4, s5, a0
+; RV64I-NEXT:    slli a0, a3, 16
+; RV64I-NEXT:    sd a0, 176(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a5, a0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 64
+; RV64I-NEXT:    and a5, s5, a5
+; RV64I-NEXT:    slli a0, a3, 17
+; RV64I-NEXT:    sd a0, 168(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a0, a3, 18
+; RV64I-NEXT:    sd a0, 160(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a0
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    lui a4, 128
+; RV64I-NEXT:    and a4, s5, a4
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    lui a4, 256
+; RV64I-NEXT:    and a4, s5, a4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a0, a3, 19
+; RV64I-NEXT:    sd a0, 152(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a2, a2, a0
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 512
+; RV64I-NEXT:    and a5, s5, a5
+; RV64I-NEXT:    slli a0, a3, 20
+; RV64I-NEXT:    sd a0, 144(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a4, 1024
+; RV64I-NEXT:    and a4, s5, a4
+; RV64I-NEXT:    slli a0, a3, 21
+; RV64I-NEXT:    sd a0, 136(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a5, a0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 2048
+; RV64I-NEXT:    and a5, s5, a5
+; RV64I-NEXT:    slli a0, a3, 22
+; RV64I-NEXT:    sd a0, 128(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a4, 4096
+; RV64I-NEXT:    and a4, s5, a4
+; RV64I-NEXT:    slli a0, a3, 23
+; RV64I-NEXT:    sd a0, 120(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a5, a0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a0, a3, 24
+; RV64I-NEXT:    sd a0, 112(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    lui a4, 8192
+; RV64I-NEXT:    and a4, s5, a4
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    lui a4, 16384
+; RV64I-NEXT:    and a4, s5, a4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a0, a3, 25
+; RV64I-NEXT:    sd a0, 104(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a2, a2, a0
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 32768
+; RV64I-NEXT:    and a5, s5, a5
+; RV64I-NEXT:    slli a0, a3, 26
+; RV64I-NEXT:    sd a0, 96(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a4, 65536
+; RV64I-NEXT:    and a4, s5, a4
+; RV64I-NEXT:    slli a0, a3, 27
+; RV64I-NEXT:    sd a0, 88(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a5, a0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a0, 131072
+; RV64I-NEXT:    and a5, s5, a0
+; RV64I-NEXT:    slli a0, a3, 28
+; RV64I-NEXT:    sd a0, 80(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a4, 262144
+; RV64I-NEXT:    and a4, s5, a4
+; RV64I-NEXT:    slli a0, a3, 29
+; RV64I-NEXT:    sd a0, 72(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a5, a0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a0, a3, 30
+; RV64I-NEXT:    sd a0, 64(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sraiw a5, s5, 31
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a0, a3, 31
+; RV64I-NEXT:    sd a0, 56(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a0
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, s5, s6
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    and a4, s5, t0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a0, a3, 32
+; RV64I-NEXT:    sd a0, 408(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a2, a2, a0
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, s5, s10
+; RV64I-NEXT:    slli a0, a3, 33
+; RV64I-NEXT:    sd a0, 48(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    and a4, s5, s11
+; RV64I-NEXT:    slli a0, a3, 34
+; RV64I-NEXT:    sd a0, 40(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a5, a0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, s5, ra
+; RV64I-NEXT:    slli a0, a3, 35
+; RV64I-NEXT:    sd a0, 32(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    and a4, s5, t1
+; RV64I-NEXT:    slli a0, a3, 36
+; RV64I-NEXT:    sd a0, 400(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a5, a0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, s5, s9
+; RV64I-NEXT:    slli a0, a3, 37
+; RV64I-NEXT:    sd a0, 24(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a0, a3, 38
+; RV64I-NEXT:    sd a0, 16(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a0
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, s5, s8
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    and a4, s5, t2
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a0, a3, 39
+; RV64I-NEXT:    sd a0, 392(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a2, a2, a0
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, s5, t3
+; RV64I-NEXT:    slli a0, a3, 40
+; RV64I-NEXT:    sd a0, 384(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a4, a4, a0
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    and a4, s5, t4
+; RV64I-NEXT:    slli a0, a3, 41
+; RV64I-NEXT:    sd a0, 376(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and a5, a5, a0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, s5, t5
+; RV64I-NEXT:    slli ra, a3, 42
+; RV64I-NEXT:    and a4, a4, ra
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    and a4, s5, t6
+; RV64I-NEXT:    slli s11, a3, 43
+; RV64I-NEXT:    and a5, a5, s11
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and a5, s5, s0
+; RV64I-NEXT:    slli s10, a3, 44
+; RV64I-NEXT:    and a4, a4, s10
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    and a4, s5, s1
+; RV64I-NEXT:    slli s9, a3, 45
+; RV64I-NEXT:    and a5, a5, s9
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli s8, a3, 46
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    and a4, a4, s8
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, s5, s3
+; RV64I-NEXT:    xor s3, a1, a2
+; RV64I-NEXT:    seqz a1, a4
+; RV64I-NEXT:    addi a1, a1, -1
+; RV64I-NEXT:    and a2, s5, s2
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    slli s7, a3, 47
+; RV64I-NEXT:    and a1, a1, s7
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    and a4, s5, s4
+; RV64I-NEXT:    slli s4, a3, 48
+; RV64I-NEXT:    and a2, a2, s4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    ld a0, 1000(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a2, s5, a0
+; RV64I-NEXT:    slli s6, a3, 49
+; RV64I-NEXT:    and a4, a4, s6
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    xor a1, a1, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    ld a0, 1040(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, s5, a0
+; RV64I-NEXT:    slli s2, a3, 50
+; RV64I-NEXT:    and a2, a2, s2
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    ld a0, 1032(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a2, s5, a0
+; RV64I-NEXT:    slli s1, a3, 51
+; RV64I-NEXT:    and a4, a4, s1
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    xor a1, a1, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    ld a0, 1024(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, s5, a0
+; RV64I-NEXT:    slli s0, a3, 52
+; RV64I-NEXT:    and a2, a2, s0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    ld a0, 1016(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a2, s5, a0
+; RV64I-NEXT:    slli t5, a3, 53
+; RV64I-NEXT:    and a4, a4, t5
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    xor a1, a1, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    ld a0, 1008(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, s5, a0
+; RV64I-NEXT:    slli t4, a3, 54
+; RV64I-NEXT:    and a2, a2, t4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli t3, a3, 55
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    and a2, a4, t3
+; RV64I-NEXT:    xor t2, a1, a2
+; RV64I-NEXT:    ld a0, 304(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a1, s5, a0
+; RV64I-NEXT:    seqz a1, a1
+; RV64I-NEXT:    and a2, s5, a7
+; RV64I-NEXT:    addi a1, a1, -1
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    ld t6, 944(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a1, a1, t6
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    ld a0, 992(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, s5, a0
+; RV64I-NEXT:    slli t1, a3, 57
+; RV64I-NEXT:    and a2, a2, t1
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    ld a0, 984(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a2, s5, a0
+; RV64I-NEXT:    slli t0, a3, 58
+; RV64I-NEXT:    and a4, a4, t0
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    xor a1, a1, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    ld a0, 976(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, s5, a0
+; RV64I-NEXT:    slli a7, a3, 59
+; RV64I-NEXT:    and a2, a2, a7
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a1, a1, a2
+; RV64I-NEXT:    addi a2, a4, -1
+; RV64I-NEXT:    ld a0, 960(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a0, s5, a0
+; RV64I-NEXT:    slli a6, a3, 60
+; RV64I-NEXT:    and a2, a2, a6
+; RV64I-NEXT:    seqz a0, a0
+; RV64I-NEXT:    xor a2, a1, a2
+; RV64I-NEXT:    addi a1, a0, -1
+; RV64I-NEXT:    ld a0, 968(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a0, s5, a0
+; RV64I-NEXT:    slli a5, a3, 61
+; RV64I-NEXT:    and a1, a1, a5
+; RV64I-NEXT:    seqz a0, a0
+; RV64I-NEXT:    xor a2, a2, a1
+; RV64I-NEXT:    addi a1, a0, -1
+; RV64I-NEXT:    srli s5, s5, 63
+; RV64I-NEXT:    slli a4, a3, 62
+; RV64I-NEXT:    and a0, a1, a4
+; RV64I-NEXT:    seqz a1, s5
+; RV64I-NEXT:    addi s5, a1, -1
+; RV64I-NEXT:    slli a1, a3, 63
+; RV64I-NEXT:    xor a0, a2, a0
+; RV64I-NEXT:    and a2, s5, a1
+; RV64I-NEXT:    xor t2, s3, t2
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    ld a2, 416(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 296(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    xor a2, a2, s3
+; RV64I-NEXT:    sd a2, 1040(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor t2, t2, a0
+; RV64I-NEXT:    ld a0, 936(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld a2, 312(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a0, a0, a2
+; RV64I-NEXT:    ld a2, 928(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a2, a2, a3
+; RV64I-NEXT:    ld a3, 920(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 280(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, s3
+; RV64I-NEXT:    ld s3, 912(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 272(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a0, a2, a0
+; RV64I-NEXT:    xor a2, a3, s5
+; RV64I-NEXT:    ld a3, 904(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 264(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, s3
+; RV64I-NEXT:    ld s3, 896(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 256(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 888(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 248(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    xor a2, a3, s5
+; RV64I-NEXT:    ld a3, 880(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 240(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, s3
+; RV64I-NEXT:    ld s3, 872(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 232(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 864(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 224(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 856(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 216(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    xor a2, a3, s5
+; RV64I-NEXT:    ld a3, 848(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 424(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, s3
+; RV64I-NEXT:    ld s3, 840(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 208(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 832(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 200(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    xor a2, a3, s5
+; RV64I-NEXT:    ld a3, 824(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 192(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, s3
+; RV64I-NEXT:    ld s3, 816(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 184(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 808(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 176(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 800(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 168(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 792(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 160(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    xor a2, a3, s5
+; RV64I-NEXT:    ld a3, 784(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 152(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, s3
+; RV64I-NEXT:    ld s3, 776(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 144(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 768(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 136(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 760(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 128(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 752(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 120(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 744(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 112(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    xor a2, a3, s5
+; RV64I-NEXT:    ld a3, 736(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 104(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, s3
+; RV64I-NEXT:    ld s3, 728(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 96(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 720(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 88(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 712(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 80(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 704(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 72(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 696(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 64(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 688(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 56(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    xor a2, a3, s5
+; RV64I-NEXT:    ld a3, 680(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 408(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, s3
+; RV64I-NEXT:    ld s3, 672(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 48(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 664(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 40(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 656(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 32(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 648(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 400(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 640(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 24(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 632(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 16(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    xor a2, a3, s5
+; RV64I-NEXT:    ld a3, 624(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 392(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, s3
+; RV64I-NEXT:    ld s3, 616(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 384(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 608(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 376(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s5
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 600(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, ra
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 592(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s11
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 584(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s10
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 576(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s9
+; RV64I-NEXT:    xor a3, a3, s5
+; RV64I-NEXT:    ld s3, 568(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s5, s3, s8
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    xor a2, a3, s5
+; RV64I-NEXT:    ld a3, 560(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a3, a3, s7
+; RV64I-NEXT:    ld s3, 552(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s4, s3, s4
+; RV64I-NEXT:    xor a3, a3, s4
+; RV64I-NEXT:    ld s3, 544(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s3, s3, s6
+; RV64I-NEXT:    xor a3, a3, s3
+; RV64I-NEXT:    ld s3, 536(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s2, s3, s2
+; RV64I-NEXT:    xor a3, a3, s2
+; RV64I-NEXT:    ld s2, 528(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s1, s2, s1
+; RV64I-NEXT:    xor a3, a3, s1
+; RV64I-NEXT:    ld s1, 520(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and s0, s1, s0
+; RV64I-NEXT:    xor a3, a3, s0
+; RV64I-NEXT:    ld s0, 512(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t5, s0, t5
+; RV64I-NEXT:    xor a3, a3, t5
+; RV64I-NEXT:    ld t5, 504(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t4, t5, t4
+; RV64I-NEXT:    xor a3, a3, t4
+; RV64I-NEXT:    ld t4, 496(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t3, t4, t3
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    xor a2, a3, t3
+; RV64I-NEXT:    ld a3, 1040(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    xor a3, t2, a3
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    ld a2, 488(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a2, a2, t6
+; RV64I-NEXT:    ld t2, 480(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t1, t2, t1
+; RV64I-NEXT:    xor a2, a2, t1
+; RV64I-NEXT:    ld t1, 472(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and t0, t1, t0
+; RV64I-NEXT:    xor a2, a2, t0
+; RV64I-NEXT:    ld t0, 464(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, t0, a7
+; RV64I-NEXT:    xor a2, a2, a7
+; RV64I-NEXT:    ld a7, 456(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a6, a7, a6
+; RV64I-NEXT:    xor a2, a2, a6
+; RV64I-NEXT:    ld a6, 448(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    ld a5, 440(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    ld a4, 432(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a1, a4, a1
+; RV64I-NEXT:    xor a2, a2, a1
+; RV64I-NEXT:    ld a1, 952(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    srli a1, a1, 1
+; RV64I-NEXT:    xor a1, a1, a3
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    ld ra, 1144(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s0, 1136(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s1, 1128(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s2, 1120(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 1112(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s4, 1104(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 1096(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s6, 1088(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s7, 1080(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s8, 1072(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s9, 1064(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s10, 1056(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s11, 1048(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    .cfi_restore ra
+; RV64I-NEXT:    .cfi_restore s0
+; RV64I-NEXT:    .cfi_restore s1
+; RV64I-NEXT:    .cfi_restore s2
+; RV64I-NEXT:    .cfi_restore s3
+; RV64I-NEXT:    .cfi_restore s4
+; RV64I-NEXT:    .cfi_restore s5
+; RV64I-NEXT:    .cfi_restore s6
+; RV64I-NEXT:    .cfi_restore s7
+; RV64I-NEXT:    .cfi_restore s8
+; RV64I-NEXT:    .cfi_restore s9
+; RV64I-NEXT:    .cfi_restore s10
+; RV64I-NEXT:    .cfi_restore s11
+; RV64I-NEXT:    addi sp, sp, 1152
+; RV64I-NEXT:    .cfi_def_cfa_offset 0
+; RV64I-NEXT:    ret
+;
+; RV32IM-LABEL: clmul_i128:
+; RV32IM:       # %bb.0:
+; RV32IM-NEXT:    addi sp, sp, -240
+; RV32IM-NEXT:    .cfi_def_cfa_offset 240
+; RV32IM-NEXT:    sw ra, 236(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s0, 232(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s1, 228(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s2, 224(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s3, 220(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s4, 216(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s5, 212(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s6, 208(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s7, 204(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s8, 200(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s9, 196(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s10, 192(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s11, 188(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    .cfi_offset ra, -4
+; RV32IM-NEXT:    .cfi_offset s0, -8
+; RV32IM-NEXT:    .cfi_offset s1, -12
+; RV32IM-NEXT:    .cfi_offset s2, -16
+; RV32IM-NEXT:    .cfi_offset s3, -20
+; RV32IM-NEXT:    .cfi_offset s4, -24
+; RV32IM-NEXT:    .cfi_offset s5, -28
+; RV32IM-NEXT:    .cfi_offset s6, -32
+; RV32IM-NEXT:    .cfi_offset s7, -36
+; RV32IM-NEXT:    .cfi_offset s8, -40
+; RV32IM-NEXT:    .cfi_offset s9, -44
+; RV32IM-NEXT:    .cfi_offset s10, -48
+; RV32IM-NEXT:    .cfi_offset s11, -52
+; RV32IM-NEXT:    sw a2, 144(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv a6, a1
+; RV32IM-NEXT:    sw a0, 80(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw a5, 4(a2)
+; RV32IM-NEXT:    sw a5, 76(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lui a0, 16
+; RV32IM-NEXT:    lw a1, 8(a2)
+; RV32IM-NEXT:    sw a1, 148(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw a1, 12(a2)
+; RV32IM-NEXT:    sw a1, 68(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    addi t0, a0, -256
+; RV32IM-NEXT:    srli a0, a5, 8
+; RV32IM-NEXT:    srli a3, a5, 24
+; RV32IM-NEXT:    and a4, a5, t0
+; RV32IM-NEXT:    slli a5, a5, 24
+; RV32IM-NEXT:    and a0, a0, t0
+; RV32IM-NEXT:    slli a4, a4, 8
+; RV32IM-NEXT:    or a0, a0, a3
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    lui a3, 61681
+; RV32IM-NEXT:    or a0, a4, a0
+; RV32IM-NEXT:    addi t5, a3, -241
+; RV32IM-NEXT:    srli a3, a0, 4
+; RV32IM-NEXT:    and a0, a0, t5
+; RV32IM-NEXT:    and a3, a3, t5
+; RV32IM-NEXT:    slli a0, a0, 4
+; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    lui a3, 209715
+; RV32IM-NEXT:    srli a4, a0, 2
+; RV32IM-NEXT:    addi t4, a3, 819
+; RV32IM-NEXT:    and a3, a4, t4
+; RV32IM-NEXT:    and a0, a0, t4
+; RV32IM-NEXT:    slli a0, a0, 2
+; RV32IM-NEXT:    lui a1, 349525
+; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    addi s10, a1, 1365
+; RV32IM-NEXT:    srli a3, a0, 1
+; RV32IM-NEXT:    and a0, a0, s10
+; RV32IM-NEXT:    and a3, a3, s10
+; RV32IM-NEXT:    slli a0, a0, 1
+; RV32IM-NEXT:    or a5, a3, a0
+; RV32IM-NEXT:    sw a5, 132(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a0, a5, 8
+; RV32IM-NEXT:    and a0, a0, t0
+; RV32IM-NEXT:    srli a3, a5, 24
+; RV32IM-NEXT:    and a4, a5, t0
+; RV32IM-NEXT:    slli a5, a5, 24
+; RV32IM-NEXT:    slli a4, a4, 8
+; RV32IM-NEXT:    or a0, a0, a3
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    or a0, a4, a0
+; RV32IM-NEXT:    sw a6, 156(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw a7, 4(a6)
+; RV32IM-NEXT:    srli a3, a0, 4
+; RV32IM-NEXT:    and a3, a3, t5
+; RV32IM-NEXT:    and a0, a0, t5
+; RV32IM-NEXT:    slli a0, a0, 4
+; RV32IM-NEXT:    lw a2, 8(a6)
+; RV32IM-NEXT:    sw a2, 40(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw a2, 12(a6)
+; RV32IM-NEXT:    sw a2, 72(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    srli a3, a0, 2
+; RV32IM-NEXT:    sw a7, 64(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a4, a7, 8
+; RV32IM-NEXT:    and a3, a3, t4
+; RV32IM-NEXT:    and a4, a4, t0
+; RV32IM-NEXT:    srli a5, a7, 24
+; RV32IM-NEXT:    and a6, a7, t0
+; RV32IM-NEXT:    slli a6, a6, 8
+; RV32IM-NEXT:    slli a7, a7, 24
+; RV32IM-NEXT:    or a4, a4, a5
+; RV32IM-NEXT:    or a5, a7, a6
+; RV32IM-NEXT:    and a0, a0, t4
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    srli a5, a4, 4
+; RV32IM-NEXT:    and a4, a4, t5
+; RV32IM-NEXT:    and a5, a5, t5
+; RV32IM-NEXT:    slli a4, a4, 4
+; RV32IM-NEXT:    slli a0, a0, 2
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    srli a5, a4, 2
+; RV32IM-NEXT:    and a4, a4, t4
+; RV32IM-NEXT:    and a5, a5, t4
+; RV32IM-NEXT:    slli a4, a4, 2
+; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    srli a3, a4, 1
+; RV32IM-NEXT:    and a4, a4, s10
+; RV32IM-NEXT:    and a3, a3, s10
+; RV32IM-NEXT:    slli a4, a4, 1
+; RV32IM-NEXT:    srli a5, a0, 1
+; RV32IM-NEXT:    or t1, a3, a4
+; RV32IM-NEXT:    sw t1, 124(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a3, a5, s10
+; RV32IM-NEXT:    srli a4, t1, 8
+; RV32IM-NEXT:    and a0, a0, s10
+; RV32IM-NEXT:    and a4, a4, t0
+; RV32IM-NEXT:    srli a5, t1, 24
+; RV32IM-NEXT:    and a6, t1, t0
+; RV32IM-NEXT:    slli t1, t1, 24
+; RV32IM-NEXT:    slli a6, a6, 8
+; RV32IM-NEXT:    or a4, a4, a5
+; RV32IM-NEXT:    or a5, t1, a6
+; RV32IM-NEXT:    slli a0, a0, 1
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    srli a5, a4, 4
+; RV32IM-NEXT:    and a4, a4, t5
+; RV32IM-NEXT:    and a5, a5, t5
+; RV32IM-NEXT:    slli a4, a4, 4
+; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    srli a3, a4, 2
+; RV32IM-NEXT:    and a4, a4, t4
+; RV32IM-NEXT:    and a3, a3, t4
+; RV32IM-NEXT:    slli a4, a4, 2
+; RV32IM-NEXT:    lui a5, 69905
+; RV32IM-NEXT:    or a4, a3, a4
+; RV32IM-NEXT:    addi a7, a5, 273
+; RV32IM-NEXT:    srli a5, a4, 1
+; RV32IM-NEXT:    and a5, a5, s10
+; RV32IM-NEXT:    and a4, a4, s10
+; RV32IM-NEXT:    slli a4, a4, 1
+; RV32IM-NEXT:    lui a6, 139810
+; RV32IM-NEXT:    or t1, a5, a4
+; RV32IM-NEXT:    addi t2, a6, 546
+; RV32IM-NEXT:    and t6, a0, a7
+; RV32IM-NEXT:    and s0, t1, t2
+; RV32IM-NEXT:    and s2, a0, t2
+; RV32IM-NEXT:    and s3, t1, a7
+; RV32IM-NEXT:    lui a5, 559241
+; RV32IM-NEXT:    lui a6, 279620
+; RV32IM-NEXT:    addi s1, a5, -1912
+; RV32IM-NEXT:    addi t3, a6, 1092
+; RV32IM-NEXT:    and s4, a0, s1
+; RV32IM-NEXT:    and s5, t1, t3
+; RV32IM-NEXT:    and a0, a0, t3
+; RV32IM-NEXT:    and t1, t1, s1
+; RV32IM-NEXT:    mul s6, s0, t6
+; RV32IM-NEXT:    mul s7, s3, s2
+; RV32IM-NEXT:    mul s8, s5, s4
+; RV32IM-NEXT:    mul s9, t1, a0
+; RV32IM-NEXT:    mul s11, s0, s4
+; RV32IM-NEXT:    mul ra, s3, t6
+; RV32IM-NEXT:    mul a6, s5, a0
+; RV32IM-NEXT:    mul a5, t1, s2
+; RV32IM-NEXT:    mul a4, s0, s2
+; RV32IM-NEXT:    mul a3, s3, a0
+; RV32IM-NEXT:    mul a2, s5, t6
+; RV32IM-NEXT:    mul a1, t1, s4
+; RV32IM-NEXT:    mul a0, s0, a0
+; RV32IM-NEXT:    mul s0, s3, s4
+; RV32IM-NEXT:    mul s2, s5, s2
+; RV32IM-NEXT:    mul t1, t1, t6
+; RV32IM-NEXT:    xor t6, s7, s6
+; RV32IM-NEXT:    xor s3, s8, s9
+; RV32IM-NEXT:    xor s4, ra, s11
+; RV32IM-NEXT:    xor a5, a6, a5
+; RV32IM-NEXT:    xor a6, t6, s3
+; RV32IM-NEXT:    xor a5, s4, a5
+; RV32IM-NEXT:    xor a3, a3, a4
+; RV32IM-NEXT:    xor a1, a2, a1
+; RV32IM-NEXT:    and a2, a6, t2
+; RV32IM-NEXT:    and a5, a5, a7
+; RV32IM-NEXT:    xor a0, s0, a0
+; RV32IM-NEXT:    xor a4, s2, t1
+; RV32IM-NEXT:    xor a1, a3, a1
+; RV32IM-NEXT:    xor a0, a0, a4
+; RV32IM-NEXT:    and a1, a1, t3
+; RV32IM-NEXT:    and a0, a0, s1
+; RV32IM-NEXT:    or a2, a5, a2
+; RV32IM-NEXT:    or a0, a1, a0
+; RV32IM-NEXT:    or a0, a2, a0
+; RV32IM-NEXT:    srli a1, a0, 8
+; RV32IM-NEXT:    and a1, a1, t0
+; RV32IM-NEXT:    srli a2, a0, 24
+; RV32IM-NEXT:    and a3, a0, t0
+; RV32IM-NEXT:    slli a0, a0, 24
+; RV32IM-NEXT:    slli a3, a3, 8
+; RV32IM-NEXT:    or a1, a1, a2
+; RV32IM-NEXT:    or a3, a0, a3
+; RV32IM-NEXT:    lw a0, 156(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a0, 0(a0)
+; RV32IM-NEXT:    or a1, a3, a1
+; RV32IM-NEXT:    srli a2, a1, 4
+; RV32IM-NEXT:    and a1, a1, t5
+; RV32IM-NEXT:    and a2, a2, t5
+; RV32IM-NEXT:    slli a1, a1, 4
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    srli a2, a1, 2
+; RV32IM-NEXT:    and a2, a2, t4
+; RV32IM-NEXT:    and a1, a1, t4
+; RV32IM-NEXT:    sw a0, 84(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a3, a0, 8
+; RV32IM-NEXT:    slli a1, a1, 2
+; RV32IM-NEXT:    and a3, a3, t0
+; RV32IM-NEXT:    srli a4, a0, 24
+; RV32IM-NEXT:    and a5, a0, t0
+; RV32IM-NEXT:    slli a5, a5, 8
+; RV32IM-NEXT:    slli a6, a0, 24
+; RV32IM-NEXT:    or a3, a3, a4
+; RV32IM-NEXT:    or a4, a6, a5
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    or a3, a4, a3
+; RV32IM-NEXT:    srli a2, a3, 4
+; RV32IM-NEXT:    and a3, a3, t5
+; RV32IM-NEXT:    and a2, a2, t5
+; RV32IM-NEXT:    slli a3, a3, 4
+; RV32IM-NEXT:    srli a4, a1, 1
+; RV32IM-NEXT:    or a2, a2, a3
+; RV32IM-NEXT:    srli a3, a2, 2
+; RV32IM-NEXT:    and a2, a2, t4
+; RV32IM-NEXT:    and a3, a3, t4
+; RV32IM-NEXT:    slli a2, a2, 2
+; RV32IM-NEXT:    lui s11, 349525
+; RV32IM-NEXT:    addi s11, s11, 1364
+; RV32IM-NEXT:    sw s11, 156(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a2, a3, a2
+; RV32IM-NEXT:    srli a3, a2, 1
+; RV32IM-NEXT:    and a2, a2, s10
+; RV32IM-NEXT:    and a3, a3, s10
+; RV32IM-NEXT:    slli a2, a2, 1
+; RV32IM-NEXT:    and a1, a1, s10
+; RV32IM-NEXT:    mv ra, s10
+; RV32IM-NEXT:    or a2, a3, a2
+; RV32IM-NEXT:    sw a7, 164(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw a0, 132(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t1, a0, a7
+; RV32IM-NEXT:    sw t2, 184(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and s7, a2, t2
+; RV32IM-NEXT:    and t2, a0, t2
+; RV32IM-NEXT:    and a3, a2, a7
+; RV32IM-NEXT:    and s0, a0, s1
+; RV32IM-NEXT:    sw t3, 168(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and t6, a0, t3
+; RV32IM-NEXT:    and t3, a2, t3
+; RV32IM-NEXT:    and s10, a2, s1
+; RV32IM-NEXT:    mv a7, t1
+; RV32IM-NEXT:    mul a2, s7, t1
+; RV32IM-NEXT:    mv t1, a3
+; RV32IM-NEXT:    mul a3, a3, t2
+; RV32IM-NEXT:    mul a5, t3, s0
+; RV32IM-NEXT:    mul a6, s10, t6
+; RV32IM-NEXT:    mul s2, s7, s0
+; RV32IM-NEXT:    mul s3, t1, a7
+; RV32IM-NEXT:    mul s4, t3, t6
+; RV32IM-NEXT:    mul s5, s10, t2
+; RV32IM-NEXT:    mul s6, s7, t2
+; RV32IM-NEXT:    sw t2, 120(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv a0, s7
+; RV32IM-NEXT:    sw s7, 140(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul s7, t1, t6
+; RV32IM-NEXT:    sw t1, 136(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul s8, t3, a7
+; RV32IM-NEXT:    sw a7, 96(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw t3, 132(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul s9, s10, s0
+; RV32IM-NEXT:    sw s10, 128(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a4, s11
+; RV32IM-NEXT:    slli a1, a1, 1
+; RV32IM-NEXT:    or a1, a4, a1
+; RV32IM-NEXT:    sw a1, 116(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a2, a3, a2
+; RV32IM-NEXT:    xor a3, a5, a6
+; RV32IM-NEXT:    lw a1, 144(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a1, 0(a1)
+; RV32IM-NEXT:    xor a4, s3, s2
+; RV32IM-NEXT:    xor a5, s4, s5
+; RV32IM-NEXT:    xor a2, a2, a3
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    xor a3, s7, s6
+; RV32IM-NEXT:    xor a5, s8, s9
+; RV32IM-NEXT:    srli a6, a1, 8
+; RV32IM-NEXT:    sw t0, 176(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and s2, a1, t0
+; RV32IM-NEXT:    and a6, a6, t0
+; RV32IM-NEXT:    slli s2, s2, 8
+; RV32IM-NEXT:    mul s3, a0, t6
+; RV32IM-NEXT:    mul s4, t1, s0
+; RV32IM-NEXT:    sw a1, 144(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli s5, a1, 24
+; RV32IM-NEXT:    slli s6, a1, 24
+; RV32IM-NEXT:    or a6, a6, s5
+; RV32IM-NEXT:    or s2, s6, s2
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    or a5, s2, a6
+; RV32IM-NEXT:    mul a6, t3, t2
+; RV32IM-NEXT:    mul s2, s10, a7
+; RV32IM-NEXT:    srli s5, a5, 4
+; RV32IM-NEXT:    sw t5, 180(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a5, t5
+; RV32IM-NEXT:    and s5, s5, t5
+; RV32IM-NEXT:    slli a5, a5, 4
+; RV32IM-NEXT:    xor s3, s4, s3
+; RV32IM-NEXT:    or a5, s5, a5
+; RV32IM-NEXT:    srli s4, a5, 2
+; RV32IM-NEXT:    sw t4, 152(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a5, t4
+; RV32IM-NEXT:    and s4, s4, t4
+; RV32IM-NEXT:    slli a5, a5, 2
+; RV32IM-NEXT:    xor a6, a6, s2
+; RV32IM-NEXT:    or a5, s4, a5
+; RV32IM-NEXT:    srli s2, a5, 1
+; RV32IM-NEXT:    sw ra, 172(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, a5, ra
+; RV32IM-NEXT:    and s2, s2, ra
+; RV32IM-NEXT:    slli a5, a5, 1
+; RV32IM-NEXT:    xor a6, s3, a6
+; RV32IM-NEXT:    or a5, s2, a5
+; RV32IM-NEXT:    lw t1, 184(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a0, a2, t1
+; RV32IM-NEXT:    sw a0, 92(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw t0, 164(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s10, a4, t0
+; RV32IM-NEXT:    lw t5, 168(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s3, a3, t5
+; RV32IM-NEXT:    mv a1, s1
+; RV32IM-NEXT:    sw s1, 160(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and s1, a6, s1
+; RV32IM-NEXT:    and t4, a5, t0
+; RV32IM-NEXT:    and a3, a5, t1
+; RV32IM-NEXT:    and a6, a5, a1
+; RV32IM-NEXT:    and a0, a5, t5
+; RV32IM-NEXT:    lw a7, 124(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s4, a7, t1
+; RV32IM-NEXT:    and s2, a7, t0
+; RV32IM-NEXT:    and a2, a7, t5
+; RV32IM-NEXT:    and a7, a7, a1
+; RV32IM-NEXT:    mul t2, s4, t4
+; RV32IM-NEXT:    mul s5, s2, a3
+; RV32IM-NEXT:    mul s6, a2, a6
+; RV32IM-NEXT:    sw a0, 112(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul s7, a7, a0
+; RV32IM-NEXT:    mul s8, s4, a6
+; RV32IM-NEXT:    mv a1, a6
+; RV32IM-NEXT:    sw a6, 108(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul s9, s2, t4
+; RV32IM-NEXT:    sw t4, 100(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul s11, a2, a0
+; RV32IM-NEXT:    mul ra, a7, a3
+; RV32IM-NEXT:    mul t3, s4, a3
+; RV32IM-NEXT:    sw a3, 104(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul t0, s2, a0
+; RV32IM-NEXT:    mul t1, a2, t4
+; RV32IM-NEXT:    mul a5, a7, a6
+; RV32IM-NEXT:    mul a6, s4, a0
+; RV32IM-NEXT:    mul a4, s2, a1
+; RV32IM-NEXT:    mul a1, a2, a3
+; RV32IM-NEXT:    mul a0, a7, t4
+; RV32IM-NEXT:    lw a3, 92(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    or s10, s10, a3
+; RV32IM-NEXT:    or s1, s3, s1
+; RV32IM-NEXT:    xor t2, s5, t2
+; RV32IM-NEXT:    xor s3, s6, s7
+; RV32IM-NEXT:    xor s5, s9, s8
+; RV32IM-NEXT:    xor s6, s11, ra
+; RV32IM-NEXT:    xor t2, t2, s3
+; RV32IM-NEXT:    xor s3, s5, s6
+; RV32IM-NEXT:    xor t0, t0, t3
+; RV32IM-NEXT:    xor a5, t1, a5
+; RV32IM-NEXT:    xor a3, a4, a6
+; RV32IM-NEXT:    xor a0, a1, a0
+; RV32IM-NEXT:    xor a1, t0, a5
+; RV32IM-NEXT:    xor a0, a3, a0
+; RV32IM-NEXT:    lw s9, 184(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, t2, s9
+; RV32IM-NEXT:    lw s11, 164(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, s3, s11
+; RV32IM-NEXT:    mv s8, t5
+; RV32IM-NEXT:    and a1, a1, t5
+; RV32IM-NEXT:    lw t5, 160(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a0, a0, t5
+; RV32IM-NEXT:    or a3, a4, a3
+; RV32IM-NEXT:    or a0, a1, a0
+; RV32IM-NEXT:    or a1, s10, s1
+; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    lw a1, 116(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    srli a1, a1, 1
+; RV32IM-NEXT:    xor a0, a1, a0
+; RV32IM-NEXT:    srli a1, a0, 8
+; RV32IM-NEXT:    lw t3, 176(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, t3
+; RV32IM-NEXT:    srli a3, a0, 24
+; RV32IM-NEXT:    and a4, a0, t3
+; RV32IM-NEXT:    slli a5, a0, 24
+; RV32IM-NEXT:    slli a4, a4, 8
+; RV32IM-NEXT:    or a0, a1, a3
+; RV32IM-NEXT:    or s3, a5, a4
+; RV32IM-NEXT:    lw s6, 96(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a1, s4, s6
+; RV32IM-NEXT:    mul a3, s4, s0
+; RV32IM-NEXT:    lw s7, 120(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a4, s4, s7
+; RV32IM-NEXT:    mul a5, s4, t6
+; RV32IM-NEXT:    mul a6, s2, s7
+; RV32IM-NEXT:    mul t0, a2, s0
+; RV32IM-NEXT:    mul t1, a7, t6
+; RV32IM-NEXT:    mul t2, s2, s6
+; RV32IM-NEXT:    mul s1, a2, t6
+; RV32IM-NEXT:    mul s4, a7, s7
+; RV32IM-NEXT:    mul t4, s2, t6
+; RV32IM-NEXT:    mul s2, s2, s0
+; RV32IM-NEXT:    mul s0, a7, s0
+; RV32IM-NEXT:    mul s5, a2, s6
+; RV32IM-NEXT:    mul a2, a2, s7
+; RV32IM-NEXT:    mul a7, a7, s6
+; RV32IM-NEXT:    or a0, s3, a0
+; RV32IM-NEXT:    xor a1, a6, a1
+; RV32IM-NEXT:    xor a6, t0, t1
+; RV32IM-NEXT:    xor a3, t2, a3
+; RV32IM-NEXT:    xor t0, s1, s4
+; RV32IM-NEXT:    xor a1, a1, a6
+; RV32IM-NEXT:    xor a3, a3, t0
+; RV32IM-NEXT:    xor a4, t4, a4
+; RV32IM-NEXT:    xor a6, s5, s0
+; RV32IM-NEXT:    xor a5, s2, a5
+; RV32IM-NEXT:    xor a2, a2, a7
+; RV32IM-NEXT:    xor a4, a4, a6
+; RV32IM-NEXT:    xor a2, a5, a2
+; RV32IM-NEXT:    mv t1, s9
+; RV32IM-NEXT:    and a1, a1, s9
+; RV32IM-NEXT:    and a3, a3, s11
+; RV32IM-NEXT:    mv a6, s8
+; RV32IM-NEXT:    and a4, a4, s8
+; RV32IM-NEXT:    and a2, a2, t5
+; RV32IM-NEXT:    or a1, a3, a1
+; RV32IM-NEXT:    or a2, a4, a2
+; RV32IM-NEXT:    or a1, a1, a2
+; RV32IM-NEXT:    srli a2, a0, 4
+; RV32IM-NEXT:    lw a5, 180(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a2, a2, a5
+; RV32IM-NEXT:    and a0, a0, a5
+; RV32IM-NEXT:    slli a0, a0, 4
+; RV32IM-NEXT:    srli a3, a1, 8
+; RV32IM-NEXT:    or a0, a2, a0
+; RV32IM-NEXT:    and a2, a3, t3
+; RV32IM-NEXT:    srli a3, a1, 24
+; RV32IM-NEXT:    and a4, a1, t3
+; RV32IM-NEXT:    slli a1, a1, 24
+; RV32IM-NEXT:    slli a4, a4, 8
+; RV32IM-NEXT:    or a2, a2, a3
+; RV32IM-NEXT:    or a1, a1, a4
+; RV32IM-NEXT:    srli a3, a0, 2
+; RV32IM-NEXT:    lw a4, 152(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a0, a0, a4
+; RV32IM-NEXT:    and a3, a3, a4
+; RV32IM-NEXT:    slli a0, a0, 2
+; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    or a1, a1, a2
+; RV32IM-NEXT:    srli a2, a0, 1
+; RV32IM-NEXT:    lw a3, 172(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a7, a0, a3
+; RV32IM-NEXT:    lw a0, 156(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a0, a2, a0
+; RV32IM-NEXT:    sw a0, 96(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a2, a7, 1
+; RV32IM-NEXT:    srli a0, a1, 4
+; RV32IM-NEXT:    and a1, a1, a5
+; RV32IM-NEXT:    and a0, a0, a5
+; RV32IM-NEXT:    slli a1, a1, 4
+; RV32IM-NEXT:    lw a3, 144(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s9, a3, s11
+; RV32IM-NEXT:    sw s9, 116(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv a4, t1
+; RV32IM-NEXT:    and s8, a3, t1
+; RV32IM-NEXT:    sw s8, 124(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv a7, t5
+; RV32IM-NEXT:    and t1, a3, t5
+; RV32IM-NEXT:    and s3, a3, a6
+; RV32IM-NEXT:    sw s3, 144(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw ra, 40(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s2, ra, a4
+; RV32IM-NEXT:    sw s2, 44(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and t4, ra, s11
+; RV32IM-NEXT:    sw t4, 60(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a3, s2, s9
+; RV32IM-NEXT:    mul a4, t4, s8
+; RV32IM-NEXT:    and s4, ra, a6
+; RV32IM-NEXT:    mv s10, a6
+; RV32IM-NEXT:    sw s4, 56(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and s6, ra, t5
+; RV32IM-NEXT:    sw s6, 52(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a5, s4, t1
+; RV32IM-NEXT:    mul a6, s6, s3
+; RV32IM-NEXT:    mul t0, s2, t1
+; RV32IM-NEXT:    mv s5, t1
+; RV32IM-NEXT:    sw t1, 120(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul t1, t4, s9
+; RV32IM-NEXT:    mul t2, s4, s3
+; RV32IM-NEXT:    mul t3, s6, s8
+; RV32IM-NEXT:    mul t5, s2, s8
+; RV32IM-NEXT:    mul t6, t4, s3
+; RV32IM-NEXT:    mul s0, s4, s9
+; RV32IM-NEXT:    mul s1, s6, s5
+; RV32IM-NEXT:    mul s2, s2, s3
+; RV32IM-NEXT:    mul s3, t4, s5
+; RV32IM-NEXT:    mul s5, s4, s8
+; RV32IM-NEXT:    mul s8, s6, s9
+; RV32IM-NEXT:    lw t4, 96(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    or a2, t4, a2
+; RV32IM-NEXT:    sw a2, 36(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a0, a0, a1
+; RV32IM-NEXT:    xor a3, a4, a3
+; RV32IM-NEXT:    xor a1, a5, a6
+; RV32IM-NEXT:    xor a2, t1, t0
+; RV32IM-NEXT:    xor a4, t2, t3
+; RV32IM-NEXT:    xor a1, a3, a1
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    xor a3, t6, t5
+; RV32IM-NEXT:    xor s0, s0, s1
+; RV32IM-NEXT:    xor a4, s3, s2
+; RV32IM-NEXT:    xor a5, s5, s8
+; RV32IM-NEXT:    xor a3, a3, s0
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    lw s7, 184(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, s7
+; RV32IM-NEXT:    and a2, a2, s11
+; RV32IM-NEXT:    and a3, a3, s10
+; RV32IM-NEXT:    and a4, a4, a7
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    sw a1, 32(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a5, a3, a4
+; RV32IM-NEXT:    srli a4, a0, 2
+; RV32IM-NEXT:    lw a3, 152(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a6, a0, a3
+; RV32IM-NEXT:    lw a2, 84(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a2, s7
+; RV32IM-NEXT:    and t4, a2, s11
+; RV32IM-NEXT:    and a0, a2, s10
+; RV32IM-NEXT:    and a2, a2, a7
+; RV32IM-NEXT:    lw t3, 148(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s4, t3, s11
+; RV32IM-NEXT:    and s9, t3, s7
+; RV32IM-NEXT:    mul t0, a1, s4
+; RV32IM-NEXT:    mul t1, t4, s9
+; RV32IM-NEXT:    and s3, t3, a7
+; RV32IM-NEXT:    and t5, t3, s10
+; RV32IM-NEXT:    mul t2, a0, s3
+; RV32IM-NEXT:    mul t3, a2, t5
+; RV32IM-NEXT:    mul t6, a1, s3
+; RV32IM-NEXT:    mul s0, t4, s4
+; RV32IM-NEXT:    mul s1, a0, t5
+; RV32IM-NEXT:    mul s5, a2, s9
+; RV32IM-NEXT:    mul s8, a1, s9
+; RV32IM-NEXT:    sw s9, 20(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv a7, a1
+; RV32IM-NEXT:    sw a1, 96(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul s10, t4, t5
+; RV32IM-NEXT:    sw t5, 4(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw t4, 92(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a1, a0, s4
+; RV32IM-NEXT:    sw s4, 16(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv s2, a0
+; RV32IM-NEXT:    sw a0, 88(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv s6, s3
+; RV32IM-NEXT:    sw s3, 24(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a0, a2, s3
+; RV32IM-NEXT:    sw a2, 84(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a4, a3
+; RV32IM-NEXT:    slli a6, a6, 2
+; RV32IM-NEXT:    or s3, a4, a6
+; RV32IM-NEXT:    sw s3, 48(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw a4, 32(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    or a4, a4, a5
+; RV32IM-NEXT:    sw a4, 32(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a4, t1, t0
+; RV32IM-NEXT:    xor a5, t2, t3
+; RV32IM-NEXT:    xor a6, s0, t6
+; RV32IM-NEXT:    xor t0, s1, s5
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    xor a5, a6, t0
+; RV32IM-NEXT:    xor a6, s10, s8
+; RV32IM-NEXT:    xor a0, a1, a0
+; RV32IM-NEXT:    and a4, a4, s7
+; RV32IM-NEXT:    and a5, a5, s11
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    xor a0, a6, a0
+; RV32IM-NEXT:    mul a5, a7, t5
+; RV32IM-NEXT:    mul a6, t4, s6
+; RV32IM-NEXT:    srli t0, ra, 8
+; RV32IM-NEXT:    lw a1, 176(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t1, ra, a1
+; RV32IM-NEXT:    and t0, t0, a1
+; RV32IM-NEXT:    slli t1, t1, 8
+; RV32IM-NEXT:    mul t2, s2, s9
+; RV32IM-NEXT:    mul t3, a2, s4
+; RV32IM-NEXT:    srli t5, ra, 24
+; RV32IM-NEXT:    slli t6, ra, 24
+; RV32IM-NEXT:    or t0, t0, t5
+; RV32IM-NEXT:    or t1, t6, t1
+; RV32IM-NEXT:    xor a5, a6, a5
+; RV32IM-NEXT:    or a6, t1, t0
+; RV32IM-NEXT:    srli t0, a6, 4
+; RV32IM-NEXT:    lw a1, 180(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a6, a6, a1
+; RV32IM-NEXT:    and t0, t0, a1
+; RV32IM-NEXT:    slli a6, a6, 4
+; RV32IM-NEXT:    xor t1, t2, t3
+; RV32IM-NEXT:    or a6, t0, a6
+; RV32IM-NEXT:    srli t0, a6, 2
+; RV32IM-NEXT:    and a6, a6, a3
+; RV32IM-NEXT:    and t0, t0, a3
+; RV32IM-NEXT:    mv s4, a3
+; RV32IM-NEXT:    slli a6, a6, 2
+; RV32IM-NEXT:    xor a5, a5, t1
+; RV32IM-NEXT:    or a6, t0, a6
+; RV32IM-NEXT:    lw a1, 168(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a0, a0, a1
+; RV32IM-NEXT:    lw s6, 160(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a5, a5, s6
+; RV32IM-NEXT:    srli t0, a6, 1
+; RV32IM-NEXT:    lw a3, 172(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a6, a6, a3
+; RV32IM-NEXT:    and t0, t0, a3
+; RV32IM-NEXT:    slli a6, a6, 1
+; RV32IM-NEXT:    or a0, a0, a5
+; RV32IM-NEXT:    or a5, t0, a6
+; RV32IM-NEXT:    or a3, a4, a0
+; RV32IM-NEXT:    and a4, a5, s7
+; RV32IM-NEXT:    mv ra, s7
+; RV32IM-NEXT:    and a6, a5, s11
+; RV32IM-NEXT:    and t6, a5, a1
+; RV32IM-NEXT:    mv s11, a1
+; RV32IM-NEXT:    and s5, a5, s6
+; RV32IM-NEXT:    srli s9, s3, 1
+; RV32IM-NEXT:    sw s9, 40(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw t4, 100(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a5, a4, t4
+; RV32IM-NEXT:    lw s2, 104(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t0, a6, s2
+; RV32IM-NEXT:    lw s7, 108(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t1, t6, s7
+; RV32IM-NEXT:    lw a2, 112(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t2, s5, a2
+; RV32IM-NEXT:    mul t3, a4, s7
+; RV32IM-NEXT:    mul t5, a6, t4
+; RV32IM-NEXT:    mul s0, t6, a2
+; RV32IM-NEXT:    mul s1, s5, s2
+; RV32IM-NEXT:    mul s8, a4, s2
+; RV32IM-NEXT:    mul s10, a6, a2
+; RV32IM-NEXT:    mul a7, t6, t4
+; RV32IM-NEXT:    mul a1, s5, s7
+; RV32IM-NEXT:    lw a0, 36(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    srli a0, a0, 1
+; RV32IM-NEXT:    slli s9, s9, 31
+; RV32IM-NEXT:    or a0, a0, s9
+; RV32IM-NEXT:    sw a0, 12(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw a0, 32(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    xor a0, a3, a0
+; RV32IM-NEXT:    sw a0, 8(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a0, t0, a5
+; RV32IM-NEXT:    xor a3, t1, t2
+; RV32IM-NEXT:    xor a5, t5, t3
+; RV32IM-NEXT:    xor s0, s0, s1
+; RV32IM-NEXT:    xor a3, a0, a3
+; RV32IM-NEXT:    xor a5, a5, s0
+; RV32IM-NEXT:    xor a0, s10, s8
+; RV32IM-NEXT:    xor a1, a7, a1
+; RV32IM-NEXT:    mul a7, a4, a2
+; RV32IM-NEXT:    mul a4, a6, s7
+; RV32IM-NEXT:    lw t2, 148(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    srli a6, t2, 8
+; RV32IM-NEXT:    lw s8, 176(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t0, t2, s8
+; RV32IM-NEXT:    and a6, a6, s8
+; RV32IM-NEXT:    slli t0, t0, 8
+; RV32IM-NEXT:    srli t1, t2, 24
+; RV32IM-NEXT:    slli t2, t2, 24
+; RV32IM-NEXT:    or a6, a6, t1
+; RV32IM-NEXT:    or t0, t2, t0
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    or a1, t0, a6
+; RV32IM-NEXT:    mul a6, t6, s2
+; RV32IM-NEXT:    mul t0, s5, t4
+; RV32IM-NEXT:    srli t1, a1, 4
+; RV32IM-NEXT:    lw s9, 180(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, s9
+; RV32IM-NEXT:    and t1, t1, s9
+; RV32IM-NEXT:    slli a1, a1, 4
+; RV32IM-NEXT:    xor a2, a4, a7
+; RV32IM-NEXT:    or a1, t1, a1
+; RV32IM-NEXT:    srli a4, a1, 2
+; RV32IM-NEXT:    mv t4, s4
+; RV32IM-NEXT:    and a1, a1, s4
+; RV32IM-NEXT:    and a4, a4, s4
+; RV32IM-NEXT:    slli a1, a1, 2
+; RV32IM-NEXT:    xor a6, a6, t0
+; RV32IM-NEXT:    or a1, a4, a1
+; RV32IM-NEXT:    srli a4, a1, 1
+; RV32IM-NEXT:    lw a7, 172(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, a7
+; RV32IM-NEXT:    and a4, a4, a7
+; RV32IM-NEXT:    slli a1, a1, 1
+; RV32IM-NEXT:    xor a2, a2, a6
+; RV32IM-NEXT:    or a1, a4, a1
+; RV32IM-NEXT:    lw s2, 164(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a1, s2
+; RV32IM-NEXT:    and a6, a1, ra
+; RV32IM-NEXT:    mv s0, s6
+; RV32IM-NEXT:    and t0, a1, s6
+; RV32IM-NEXT:    mv s7, s11
+; RV32IM-NEXT:    and a1, a1, s11
+; RV32IM-NEXT:    lw s5, 140(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t1, s5, a4
+; RV32IM-NEXT:    lw s1, 136(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t2, s1, a6
+; RV32IM-NEXT:    lw s4, 132(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t3, s4, t0
+; RV32IM-NEXT:    lw s6, 128(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t5, s6, a1
+; RV32IM-NEXT:    and a3, a3, ra
+; RV32IM-NEXT:    and a5, a5, s2
+; RV32IM-NEXT:    mv s11, s2
+; RV32IM-NEXT:    and a0, a0, s7
+; RV32IM-NEXT:    and a2, a2, s0
+; RV32IM-NEXT:    mv s10, s0
+; RV32IM-NEXT:    or a3, a5, a3
+; RV32IM-NEXT:    or a0, a0, a2
+; RV32IM-NEXT:    mul a2, s5, t0
+; RV32IM-NEXT:    mul a5, s1, a4
+; RV32IM-NEXT:    mul t6, s4, a1
+; RV32IM-NEXT:    mul s0, s6, a6
+; RV32IM-NEXT:    xor t1, t2, t1
+; RV32IM-NEXT:    xor t2, t3, t5
+; RV32IM-NEXT:    mul t3, s1, a1
+; RV32IM-NEXT:    mul a1, s5, a1
+; RV32IM-NEXT:    mul t5, s6, t0
+; RV32IM-NEXT:    mul t0, s1, t0
+; RV32IM-NEXT:    mul s1, s5, a6
+; RV32IM-NEXT:    mul s5, s4, a4
+; RV32IM-NEXT:    mul a6, s4, a6
+; RV32IM-NEXT:    mul a4, s6, a4
+; RV32IM-NEXT:    xor a2, a5, a2
+; RV32IM-NEXT:    xor a5, t6, s0
+; RV32IM-NEXT:    xor t1, t1, t2
+; RV32IM-NEXT:    xor a2, a2, a5
+; RV32IM-NEXT:    and a5, t1, ra
+; RV32IM-NEXT:    and a2, a2, s2
+; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    or a2, a2, a5
+; RV32IM-NEXT:    xor a3, t3, s1
+; RV32IM-NEXT:    xor a5, s5, t5
+; RV32IM-NEXT:    xor a1, t0, a1
+; RV32IM-NEXT:    xor a4, a6, a4
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    and a3, a3, s7
+; RV32IM-NEXT:    and a1, a1, s10
+; RV32IM-NEXT:    or a1, a3, a1
+; RV32IM-NEXT:    srli a3, a0, 8
+; RV32IM-NEXT:    and a3, a3, s8
+; RV32IM-NEXT:    srli a4, a0, 24
+; RV32IM-NEXT:    or a3, a3, a4
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    slli a2, a0, 24
+; RV32IM-NEXT:    and a0, a0, s8
+; RV32IM-NEXT:    slli a0, a0, 8
+; RV32IM-NEXT:    srli a4, a1, 8
+; RV32IM-NEXT:    or a0, a2, a0
+; RV32IM-NEXT:    and a2, a4, s8
+; RV32IM-NEXT:    srli a4, a1, 24
+; RV32IM-NEXT:    and a5, a1, s8
+; RV32IM-NEXT:    slli a1, a1, 24
+; RV32IM-NEXT:    slli a5, a5, 8
+; RV32IM-NEXT:    or a2, a2, a4
+; RV32IM-NEXT:    or a1, a1, a5
+; RV32IM-NEXT:    or a0, a0, a3
+; RV32IM-NEXT:    or a1, a1, a2
+; RV32IM-NEXT:    srli a2, a0, 4
+; RV32IM-NEXT:    and a0, a0, s9
+; RV32IM-NEXT:    and a2, a2, s9
+; RV32IM-NEXT:    slli a0, a0, 4
+; RV32IM-NEXT:    srli a3, a1, 4
+; RV32IM-NEXT:    and a1, a1, s9
+; RV32IM-NEXT:    and a3, a3, s9
+; RV32IM-NEXT:    slli a1, a1, 4
+; RV32IM-NEXT:    or a0, a2, a0
+; RV32IM-NEXT:    or a1, a3, a1
+; RV32IM-NEXT:    srli a2, a0, 2
+; RV32IM-NEXT:    and a0, a0, t4
+; RV32IM-NEXT:    and a2, a2, t4
+; RV32IM-NEXT:    slli a0, a0, 2
+; RV32IM-NEXT:    srli a3, a1, 2
+; RV32IM-NEXT:    and a1, a1, t4
+; RV32IM-NEXT:    and a3, a3, t4
+; RV32IM-NEXT:    slli a1, a1, 2
+; RV32IM-NEXT:    or a0, a2, a0
+; RV32IM-NEXT:    or a1, a3, a1
+; RV32IM-NEXT:    srli a2, a0, 1
+; RV32IM-NEXT:    and a0, a0, a7
+; RV32IM-NEXT:    lw a4, 156(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a2, a2, a4
+; RV32IM-NEXT:    slli a0, a0, 1
+; RV32IM-NEXT:    srli a3, a1, 1
+; RV32IM-NEXT:    and a1, a1, a7
+; RV32IM-NEXT:    and a3, a3, a4
+; RV32IM-NEXT:    slli a1, a1, 1
+; RV32IM-NEXT:    or a0, a2, a0
+; RV32IM-NEXT:    or a1, a3, a1
+; RV32IM-NEXT:    lw a2, 64(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a2, ra
+; RV32IM-NEXT:    and s2, a2, s2
+; RV32IM-NEXT:    and t5, a2, s7
+; RV32IM-NEXT:    and s3, a2, s10
+; RV32IM-NEXT:    lw a6, 4(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a2, s3, a6
+; RV32IM-NEXT:    sw s3, 64(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a3, t5, a6
+; RV32IM-NEXT:    sw s2, 36(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a5, s2, a6
+; RV32IM-NEXT:    sw a4, 32(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a6, a4, a6
+; RV32IM-NEXT:    lw t6, 16(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a7, a4, t6
+; RV32IM-NEXT:    lw s0, 20(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t0, s2, s0
+; RV32IM-NEXT:    lw t4, 24(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t1, t5, t4
+; RV32IM-NEXT:    mul t2, a4, t4
+; RV32IM-NEXT:    mul t3, s2, t6
+; RV32IM-NEXT:    mul s1, s3, s0
+; RV32IM-NEXT:    mul s5, s3, t4
+; RV32IM-NEXT:    mul t4, s2, t4
+; RV32IM-NEXT:    mul s8, a4, s0
+; RV32IM-NEXT:    sw t5, 28(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul s9, t5, t6
+; RV32IM-NEXT:    mul s2, t5, s0
+; RV32IM-NEXT:    mul s3, s3, t6
+; RV32IM-NEXT:    lw a4, 12(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw t5, 8(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    xor a4, a4, t5
+; RV32IM-NEXT:    sw a4, 148(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a0, a1, a0
+; RV32IM-NEXT:    sw a0, 24(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a0, t0, a7
+; RV32IM-NEXT:    xor a1, t1, a2
+; RV32IM-NEXT:    xor a2, t3, t2
+; RV32IM-NEXT:    xor a3, a3, s1
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    xor a2, a2, a3
+; RV32IM-NEXT:    xor a1, a5, s8
+; RV32IM-NEXT:    xor a3, s9, s5
+; RV32IM-NEXT:    xor a5, t4, a6
+; RV32IM-NEXT:    xor a6, s2, s3
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    xor a3, a5, a6
+; RV32IM-NEXT:    mv s6, ra
+; RV32IM-NEXT:    and a4, a0, ra
+; RV32IM-NEXT:    mv s4, s11
+; RV32IM-NEXT:    and a2, a2, s11
+; RV32IM-NEXT:    and a1, a1, s7
+; RV32IM-NEXT:    mv s0, s10
+; RV32IM-NEXT:    and a3, a3, s10
+; RV32IM-NEXT:    lw t1, 68(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a5, t1, s11
+; RV32IM-NEXT:    and a6, t1, ra
+; RV32IM-NEXT:    and a7, t1, s10
+; RV32IM-NEXT:    and t0, t1, s7
+; RV32IM-NEXT:    lw t6, 84(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t1, t6, t0
+; RV32IM-NEXT:    lw t5, 88(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t2, t5, t0
+; RV32IM-NEXT:    lw a0, 92(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t3, a0, t0
+; RV32IM-NEXT:    lw s10, 96(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t0, s10, t0
+; RV32IM-NEXT:    mul t4, s10, a5
+; RV32IM-NEXT:    mul s1, a0, a6
+; RV32IM-NEXT:    mul s2, t5, a7
+; RV32IM-NEXT:    mul s3, s10, a7
+; RV32IM-NEXT:    mul s5, a0, a5
+; RV32IM-NEXT:    mul s8, t6, a6
+; RV32IM-NEXT:    mul s9, t6, a7
+; RV32IM-NEXT:    mul a7, a0, a7
+; RV32IM-NEXT:    mul s10, s10, a6
+; RV32IM-NEXT:    mul a0, t5, a5
+; RV32IM-NEXT:    mul a6, t5, a6
+; RV32IM-NEXT:    mul a5, t6, a5
+; RV32IM-NEXT:    or t5, a2, a4
+; RV32IM-NEXT:    or a2, a1, a3
+; RV32IM-NEXT:    xor a3, s1, t4
+; RV32IM-NEXT:    xor a4, s2, t1
+; RV32IM-NEXT:    xor t1, s5, s3
+; RV32IM-NEXT:    xor t2, t2, s8
+; RV32IM-NEXT:    xor a3, a3, a4
+; RV32IM-NEXT:    xor a4, t1, t2
+; RV32IM-NEXT:    xor t1, t3, s10
+; RV32IM-NEXT:    xor a0, a0, s9
+; RV32IM-NEXT:    xor a7, a7, t0
+; RV32IM-NEXT:    xor a5, a6, a5
+; RV32IM-NEXT:    xor a0, t1, a0
+; RV32IM-NEXT:    xor a5, a7, a5
+; RV32IM-NEXT:    and a3, a3, ra
+; RV32IM-NEXT:    and a4, a4, s11
+; RV32IM-NEXT:    and a0, a0, s7
+; RV32IM-NEXT:    and a5, a5, s0
+; RV32IM-NEXT:    or a1, a4, a3
+; RV32IM-NEXT:    or a3, a0, a5
+; RV32IM-NEXT:    lw a0, 72(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a0, ra
+; RV32IM-NEXT:    and a5, a0, s11
+; RV32IM-NEXT:    and a6, a0, s7
+; RV32IM-NEXT:    and a7, a0, s0
+; RV32IM-NEXT:    lw t6, 116(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t0, a4, t6
+; RV32IM-NEXT:    lw a0, 120(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t1, a4, a0
+; RV32IM-NEXT:    lw s11, 124(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t2, a4, s11
+; RV32IM-NEXT:    lw s5, 144(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a4, a4, s5
+; RV32IM-NEXT:    mul t3, a5, s11
+; RV32IM-NEXT:    mul t4, a6, a0
+; RV32IM-NEXT:    mul s1, a7, s5
+; RV32IM-NEXT:    mul s2, a5, t6
+; RV32IM-NEXT:    mul s3, a6, s5
+; RV32IM-NEXT:    mul s9, a7, s11
+; RV32IM-NEXT:    mul s10, a5, s5
+; RV32IM-NEXT:    mul a5, a5, a0
+; RV32IM-NEXT:    mul s5, a6, t6
+; RV32IM-NEXT:    mul s8, a7, a0
+; RV32IM-NEXT:    mul a0, a6, s11
+; RV32IM-NEXT:    mul a7, a7, t6
+; RV32IM-NEXT:    or ra, t5, a2
+; RV32IM-NEXT:    or a1, a1, a3
+; RV32IM-NEXT:    sw a1, 68(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a1, t3, t0
+; RV32IM-NEXT:    xor a2, t4, s1
+; RV32IM-NEXT:    xor t0, s2, t1
+; RV32IM-NEXT:    xor t1, s3, s9
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a2, t0, t1
+; RV32IM-NEXT:    xor t0, s10, t2
+; RV32IM-NEXT:    xor t1, s5, s8
+; RV32IM-NEXT:    xor a4, a5, a4
+; RV32IM-NEXT:    xor a0, a0, a7
+; RV32IM-NEXT:    xor a5, t0, t1
+; RV32IM-NEXT:    xor a0, a4, a0
+; RV32IM-NEXT:    and a1, a1, s6
+; RV32IM-NEXT:    and a2, a2, s4
+; RV32IM-NEXT:    mv a3, s7
+; RV32IM-NEXT:    and a4, a5, s7
+; RV32IM-NEXT:    and a0, a0, s0
+; RV32IM-NEXT:    or t6, a2, a1
+; RV32IM-NEXT:    or a6, a4, a0
+; RV32IM-NEXT:    lw a0, 76(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a0, s4
+; RV32IM-NEXT:    mv s11, s4
+; RV32IM-NEXT:    and s7, a0, s6
+; RV32IM-NEXT:    and a7, a0, s0
+; RV32IM-NEXT:    and a5, a0, a3
+; RV32IM-NEXT:    mv s6, a3
+; RV32IM-NEXT:    lw a0, 44(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    sw a4, 72(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul t0, a0, a4
+; RV32IM-NEXT:    mul t1, a0, a7
+; RV32IM-NEXT:    mul t2, a0, s7
+; RV32IM-NEXT:    mul t3, a0, a5
+; RV32IM-NEXT:    lw a2, 60(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t4, a2, s7
+; RV32IM-NEXT:    lw a0, 56(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t5, a0, a7
+; RV32IM-NEXT:    lw a3, 52(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul s1, a3, a5
+; RV32IM-NEXT:    mul s2, a2, a4
+; RV32IM-NEXT:    mul s3, a0, a5
+; RV32IM-NEXT:    mul s9, a3, s7
+; RV32IM-NEXT:    mul s10, a2, a5
+; RV32IM-NEXT:    mul s8, a2, a7
+; RV32IM-NEXT:    mul s5, a0, a4
+; RV32IM-NEXT:    mul a2, a3, a7
+; RV32IM-NEXT:    mul a1, a0, s7
+; RV32IM-NEXT:    mul a0, a3, a4
+; RV32IM-NEXT:    lw a3, 68(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    xor a3, a3, ra
+; RV32IM-NEXT:    sw a3, 76(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a3, t6, a6
+; RV32IM-NEXT:    sw a3, 68(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a6, t4, t0
+; RV32IM-NEXT:    xor t0, t5, s1
+; RV32IM-NEXT:    xor t1, s2, t1
+; RV32IM-NEXT:    xor t4, s3, s9
+; RV32IM-NEXT:    xor t6, a6, t0
+; RV32IM-NEXT:    xor t1, t1, t4
+; RV32IM-NEXT:    xor t0, s10, t2
+; RV32IM-NEXT:    xor a2, s5, a2
+; RV32IM-NEXT:    xor t2, s8, t3
+; RV32IM-NEXT:    xor a0, a1, a0
+; RV32IM-NEXT:    xor a1, t0, a2
+; RV32IM-NEXT:    xor a6, t2, a0
+; RV32IM-NEXT:    lw a0, 140(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s9, 100(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul ra, a0, s9
+; RV32IM-NEXT:    lw s10, 108(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t0, a0, s10
+; RV32IM-NEXT:    lw a3, 104(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t2, a0, a3
+; RV32IM-NEXT:    lw s0, 112(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t3, a0, s0
+; RV32IM-NEXT:    lw s3, 128(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t4, s3, s0
+; RV32IM-NEXT:    lw s5, 132(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t5, s5, s0
+; RV32IM-NEXT:    lw a0, 136(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul s0, a0, s0
+; RV32IM-NEXT:    mul s1, a0, a3
+; RV32IM-NEXT:    mul s2, s3, a3
+; RV32IM-NEXT:    mul a4, s5, a3
+; RV32IM-NEXT:    mul s4, s5, s10
+; RV32IM-NEXT:    mul s5, s5, s9
+; RV32IM-NEXT:    mul s8, a0, s9
+; RV32IM-NEXT:    mul a3, a0, s10
+; RV32IM-NEXT:    mul s10, s3, s10
+; RV32IM-NEXT:    mul a0, s3, s9
+; RV32IM-NEXT:    lw a2, 184(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t6, t6, a2
+; RV32IM-NEXT:    and t1, t1, s11
+; RV32IM-NEXT:    mv s3, s6
+; RV32IM-NEXT:    and a1, a1, s6
+; RV32IM-NEXT:    lw s6, 160(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a6, a6, s6
+; RV32IM-NEXT:    or t1, t1, t6
+; RV32IM-NEXT:    or a1, a1, a6
+; RV32IM-NEXT:    xor t6, s1, ra
+; RV32IM-NEXT:    xor a6, s4, t4
+; RV32IM-NEXT:    xor t0, s8, t0
+; RV32IM-NEXT:    xor t4, t5, s2
+; RV32IM-NEXT:    xor t5, t6, a6
+; RV32IM-NEXT:    xor a6, t0, t4
+; RV32IM-NEXT:    xor t0, s0, t2
+; RV32IM-NEXT:    xor t2, s5, s10
+; RV32IM-NEXT:    xor t3, a3, t3
+; RV32IM-NEXT:    xor a0, a4, a0
+; RV32IM-NEXT:    xor t0, t0, t2
+; RV32IM-NEXT:    xor a0, t3, a0
+; RV32IM-NEXT:    and a3, t5, a2
+; RV32IM-NEXT:    and a6, a6, s11
+; RV32IM-NEXT:    and t0, t0, s3
+; RV32IM-NEXT:    and a0, a0, s6
+; RV32IM-NEXT:    mv s10, s6
+; RV32IM-NEXT:    or a2, a6, a3
+; RV32IM-NEXT:    or a0, t0, a0
+; RV32IM-NEXT:    or a1, t1, a1
+; RV32IM-NEXT:    or a0, a2, a0
+; RV32IM-NEXT:    lw a2, 68(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    srli a2, a0, 8
+; RV32IM-NEXT:    lw a3, 76(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    xor a1, a3, a1
+; RV32IM-NEXT:    lw a3, 176(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a2, a2, a3
+; RV32IM-NEXT:    and a6, a0, a3
+; RV32IM-NEXT:    srli t0, a0, 24
+; RV32IM-NEXT:    slli a0, a0, 24
+; RV32IM-NEXT:    slli a6, a6, 8
+; RV32IM-NEXT:    or a2, a2, t0
+; RV32IM-NEXT:    or a0, a0, a6
+; RV32IM-NEXT:    or a0, a0, a2
+; RV32IM-NEXT:    lw a2, 48(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw t0, 172(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a2, a2, t0
+; RV32IM-NEXT:    lw a3, 40(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a4, 156(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a6, a3, a4
+; RV32IM-NEXT:    slli a2, a2, 1
+; RV32IM-NEXT:    or a2, a6, a2
+; RV32IM-NEXT:    srli a6, a0, 4
+; RV32IM-NEXT:    lw a3, 180(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a6, a6, a3
+; RV32IM-NEXT:    and a0, a0, a3
+; RV32IM-NEXT:    srli a2, a2, 1
+; RV32IM-NEXT:    slli a0, a0, 4
+; RV32IM-NEXT:    xor a1, a2, a1
+; RV32IM-NEXT:    or a0, a6, a0
+; RV32IM-NEXT:    srli a2, a0, 2
+; RV32IM-NEXT:    lw a3, 152(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a0, a0, a3
+; RV32IM-NEXT:    and a2, a2, a3
+; RV32IM-NEXT:    slli a0, a0, 2
+; RV32IM-NEXT:    lw a3, 24(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    srli a6, a3, 1
+; RV32IM-NEXT:    or a0, a2, a0
+; RV32IM-NEXT:    srli a2, a0, 1
+; RV32IM-NEXT:    and a0, a0, t0
+; RV32IM-NEXT:    and a2, a2, a4
+; RV32IM-NEXT:    slli t0, a0, 1
+; RV32IM-NEXT:    xor a0, a1, a6
+; RV32IM-NEXT:    sw a0, 180(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a0, a2, t0
+; RV32IM-NEXT:    sw a0, 176(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw s8, 116(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a4, 32(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a1, a4, s8
+; RV32IM-NEXT:    lw a0, 36(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw t1, 124(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a2, a0, t1
+; RV32IM-NEXT:    lw s9, 120(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s5, 28(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a6, s5, s9
+; RV32IM-NEXT:    lw a3, 64(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw ra, 144(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t0, a3, ra
+; RV32IM-NEXT:    mul t2, a4, s9
+; RV32IM-NEXT:    mul t3, a0, s8
+; RV32IM-NEXT:    mul t4, s5, ra
+; RV32IM-NEXT:    mul t5, a3, t1
+; RV32IM-NEXT:    mul t6, a4, t1
+; RV32IM-NEXT:    mul s0, a0, ra
+; RV32IM-NEXT:    mul s1, s5, s8
+; RV32IM-NEXT:    mul s2, a3, s9
+; RV32IM-NEXT:    mul s3, a4, ra
+; RV32IM-NEXT:    mul s4, a0, s9
+; RV32IM-NEXT:    mul s5, s5, t1
+; RV32IM-NEXT:    mul s6, a3, s8
+; RV32IM-NEXT:    xor a1, a2, a1
+; RV32IM-NEXT:    xor a2, a6, t0
+; RV32IM-NEXT:    xor a6, t3, t2
+; RV32IM-NEXT:    xor t0, t4, t5
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a2, a6, t0
+; RV32IM-NEXT:    xor a6, s0, t6
+; RV32IM-NEXT:    xor t0, s1, s2
+; RV32IM-NEXT:    lw s11, 184(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, s11
+; RV32IM-NEXT:    lw t4, 164(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a2, a2, t4
+; RV32IM-NEXT:    xor t2, s4, s3
+; RV32IM-NEXT:    xor t3, s5, s6
+; RV32IM-NEXT:    xor a6, a6, t0
+; RV32IM-NEXT:    xor t0, t2, t3
+; RV32IM-NEXT:    lw a0, 168(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a6, a6, a0
+; RV32IM-NEXT:    and t0, t0, s10
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    sw a1, 172(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a0, a6, t0
+; RV32IM-NEXT:    sw a0, 156(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw a0, 96(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a3, 72(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a6, a0, a3
+; RV32IM-NEXT:    lw s6, 92(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t0, s6, s7
+; RV32IM-NEXT:    lw s10, 88(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t2, s10, a7
+; RV32IM-NEXT:    lw s5, 84(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t3, s5, a5
+; RV32IM-NEXT:    mul a2, a0, a7
+; RV32IM-NEXT:    mul t5, s6, a3
+; RV32IM-NEXT:    mul t6, s10, a5
+; RV32IM-NEXT:    mul s0, s5, s7
+; RV32IM-NEXT:    mul s1, a0, s7
+; RV32IM-NEXT:    mul s2, s6, a5
+; RV32IM-NEXT:    mul s3, s10, a3
+; RV32IM-NEXT:    mul s4, s5, a7
+; RV32IM-NEXT:    mul a1, a0, a5
+; RV32IM-NEXT:    mul a7, s6, a7
+; RV32IM-NEXT:    mul a4, s10, s7
+; RV32IM-NEXT:    mv s7, s10
+; RV32IM-NEXT:    mul a5, s5, a3
+; RV32IM-NEXT:    xor a6, t0, a6
+; RV32IM-NEXT:    xor t0, t2, t3
+; RV32IM-NEXT:    xor t2, t5, a2
+; RV32IM-NEXT:    xor t3, t6, s0
+; RV32IM-NEXT:    xor a6, a6, t0
+; RV32IM-NEXT:    xor t0, t2, t3
+; RV32IM-NEXT:    xor t2, s2, s1
+; RV32IM-NEXT:    xor t3, s3, s4
+; RV32IM-NEXT:    and a6, a6, s11
+; RV32IM-NEXT:    mv s10, t4
+; RV32IM-NEXT:    and t0, t0, t4
+; RV32IM-NEXT:    xor a3, a7, a1
+; RV32IM-NEXT:    xor a4, a4, a5
+; RV32IM-NEXT:    xor a5, t2, t3
+; RV32IM-NEXT:    xor a3, a3, a4
+; RV32IM-NEXT:    mul a4, a0, s8
+; RV32IM-NEXT:    mul a7, s6, t1
+; RV32IM-NEXT:    mul t2, s7, s9
+; RV32IM-NEXT:    mul t3, s5, ra
+; RV32IM-NEXT:    lw s4, 168(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a5, a5, s4
+; RV32IM-NEXT:    lw s3, 160(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a3, s3
+; RV32IM-NEXT:    or a6, t0, a6
+; RV32IM-NEXT:    or a3, a5, a3
+; RV32IM-NEXT:    lw a1, 172(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw a2, 156(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    or a1, a1, a2
+; RV32IM-NEXT:    or a2, a6, a3
+; RV32IM-NEXT:    mul a3, a0, s9
+; RV32IM-NEXT:    mul a5, s6, s8
+; RV32IM-NEXT:    mul a6, s7, ra
+; RV32IM-NEXT:    mul t0, s5, t1
+; RV32IM-NEXT:    xor a4, a7, a4
+; RV32IM-NEXT:    xor a7, t2, t3
+; RV32IM-NEXT:    mul t2, a0, t1
+; RV32IM-NEXT:    mul t3, s6, ra
+; RV32IM-NEXT:    mul t4, a0, ra
+; RV32IM-NEXT:    mul t5, s7, s8
+; RV32IM-NEXT:    mul t6, s6, s9
+; RV32IM-NEXT:    mul s0, s5, s9
+; RV32IM-NEXT:    mul s1, s7, t1
+; RV32IM-NEXT:    mul s2, s5, s8
+; RV32IM-NEXT:    xor a3, a5, a3
+; RV32IM-NEXT:    xor a5, a6, t0
+; RV32IM-NEXT:    xor a4, a4, a7
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    and a4, a4, s11
+; RV32IM-NEXT:    and a3, a3, s10
+; RV32IM-NEXT:    xor a1, a2, a1
+; RV32IM-NEXT:    or a3, a3, a4
+; RV32IM-NEXT:    xor a2, t3, t2
+; RV32IM-NEXT:    xor a4, t5, s0
+; RV32IM-NEXT:    xor a5, t6, t4
+; RV32IM-NEXT:    xor a6, s1, s2
+; RV32IM-NEXT:    xor a2, a2, a4
+; RV32IM-NEXT:    xor a4, a5, a6
+; RV32IM-NEXT:    and a2, a2, s4
+; RV32IM-NEXT:    and a4, a4, s3
+; RV32IM-NEXT:    or a2, a2, a4
+; RV32IM-NEXT:    lw a4, 176(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    srli a4, a4, 1
+; RV32IM-NEXT:    xor a1, a4, a1
+; RV32IM-NEXT:    or a2, a3, a2
+; RV32IM-NEXT:    lw a3, 80(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    sw a2, 0(a3)
+; RV32IM-NEXT:    sw a1, 4(a3)
+; RV32IM-NEXT:    lw a1, 148(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    sw a1, 8(a3)
+; RV32IM-NEXT:    lw a0, 180(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    sw a0, 12(a3)
+; RV32IM-NEXT:    lw ra, 236(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s0, 232(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s1, 228(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s2, 224(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s3, 220(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s4, 216(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s5, 212(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s6, 208(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s7, 204(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s8, 200(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s9, 196(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s10, 192(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s11, 188(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    .cfi_restore ra
+; RV32IM-NEXT:    .cfi_restore s0
+; RV32IM-NEXT:    .cfi_restore s1
+; RV32IM-NEXT:    .cfi_restore s2
+; RV32IM-NEXT:    .cfi_restore s3
+; RV32IM-NEXT:    .cfi_restore s4
+; RV32IM-NEXT:    .cfi_restore s5
+; RV32IM-NEXT:    .cfi_restore s6
+; RV32IM-NEXT:    .cfi_restore s7
+; RV32IM-NEXT:    .cfi_restore s8
+; RV32IM-NEXT:    .cfi_restore s9
+; RV32IM-NEXT:    .cfi_restore s10
+; RV32IM-NEXT:    .cfi_restore s11
+; RV32IM-NEXT:    addi sp, sp, 240
+; RV32IM-NEXT:    .cfi_def_cfa_offset 0
+; RV32IM-NEXT:    ret
+;
+; RV64IM-LABEL: clmul_i128:
+; RV64IM:       # %bb.0:
+; RV64IM-NEXT:    addi sp, sp, -128
+; RV64IM-NEXT:    .cfi_def_cfa_offset 128
+; RV64IM-NEXT:    sd ra, 120(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s0, 112(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s1, 104(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s2, 96(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s3, 88(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s4, 80(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s5, 72(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s6, 64(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s7, 56(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s8, 48(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s9, 40(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s10, 32(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s11, 24(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    .cfi_offset ra, -8
+; RV64IM-NEXT:    .cfi_offset s0, -16
+; RV64IM-NEXT:    .cfi_offset s1, -24
+; RV64IM-NEXT:    .cfi_offset s2, -32
+; RV64IM-NEXT:    .cfi_offset s3, -40
+; RV64IM-NEXT:    .cfi_offset s4, -48
+; RV64IM-NEXT:    .cfi_offset s5, -56
+; RV64IM-NEXT:    .cfi_offset s6, -64
+; RV64IM-NEXT:    .cfi_offset s7, -72
+; RV64IM-NEXT:    .cfi_offset s8, -80
+; RV64IM-NEXT:    .cfi_offset s9, -88
+; RV64IM-NEXT:    .cfi_offset s10, -96
+; RV64IM-NEXT:    .cfi_offset s11, -104
+; RV64IM-NEXT:    sd a3, 16(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd a1, 8(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    mv s3, a0
+; RV64IM-NEXT:    srli a4, a2, 24
+; RV64IM-NEXT:    lui a0, 4080
+; RV64IM-NEXT:    and a4, a4, a0
+; RV64IM-NEXT:    li t3, 255
+; RV64IM-NEXT:    srli a5, a2, 8
+; RV64IM-NEXT:    slli t3, t3, 24
+; RV64IM-NEXT:    and a5, a5, t3
+; RV64IM-NEXT:    lui a6, 16
+; RV64IM-NEXT:    srli a7, a2, 40
+; RV64IM-NEXT:    addi t1, a6, -256
+; RV64IM-NEXT:    and a6, a7, t1
+; RV64IM-NEXT:    srli a7, a2, 56
+; RV64IM-NEXT:    or a4, a5, a4
+; RV64IM-NEXT:    or a5, a6, a7
+; RV64IM-NEXT:    or a4, a4, a5
+; RV64IM-NEXT:    and a5, a2, a0
+; RV64IM-NEXT:    srliw a6, a2, 24
+; RV64IM-NEXT:    slli a5, a5, 24
+; RV64IM-NEXT:    slli a6, a6, 32
+; RV64IM-NEXT:    or a5, a5, a6
+; RV64IM-NEXT:    and a6, a2, t1
+; RV64IM-NEXT:    slli a6, a6, 40
+; RV64IM-NEXT:    slli a7, a2, 56
+; RV64IM-NEXT:    or a6, a7, a6
+; RV64IM-NEXT:    lui a7, 61681
+; RV64IM-NEXT:    or a5, a6, a5
+; RV64IM-NEXT:    addi t4, a7, -241
+; RV64IM-NEXT:    or a4, a5, a4
+; RV64IM-NEXT:    slli a5, t4, 32
+; RV64IM-NEXT:    srli a6, a4, 4
+; RV64IM-NEXT:    add t4, t4, a5
+; RV64IM-NEXT:    and a5, a6, t4
+; RV64IM-NEXT:    and a4, a4, t4
+; RV64IM-NEXT:    lui a6, 209715
+; RV64IM-NEXT:    slli a4, a4, 4
+; RV64IM-NEXT:    addi t5, a6, 819
+; RV64IM-NEXT:    or a4, a5, a4
+; RV64IM-NEXT:    slli a5, t5, 32
+; RV64IM-NEXT:    srli a6, a4, 2
+; RV64IM-NEXT:    add t5, t5, a5
+; RV64IM-NEXT:    and a5, a6, t5
+; RV64IM-NEXT:    lui a6, 349525
+; RV64IM-NEXT:    and a4, a4, t5
+; RV64IM-NEXT:    addi t2, a6, 1365
+; RV64IM-NEXT:    slli a4, a4, 2
+; RV64IM-NEXT:    slli a6, t2, 32
+; RV64IM-NEXT:    or a4, a5, a4
+; RV64IM-NEXT:    add a1, t2, a6
+; RV64IM-NEXT:    srli a5, a4, 1
+; RV64IM-NEXT:    and a4, a4, a1
+; RV64IM-NEXT:    and a5, a5, a1
+; RV64IM-NEXT:    slli a4, a4, 1
+; RV64IM-NEXT:    or t6, a5, a4
+; RV64IM-NEXT:    srli a4, s3, 24
+; RV64IM-NEXT:    srli a5, s3, 8
+; RV64IM-NEXT:    and a4, a4, a0
+; RV64IM-NEXT:    and a5, a5, t3
+; RV64IM-NEXT:    or a4, a5, a4
+; RV64IM-NEXT:    srli a5, s3, 40
+; RV64IM-NEXT:    and a5, a5, t1
+; RV64IM-NEXT:    srli a6, s3, 56
+; RV64IM-NEXT:    or a5, a5, a6
+; RV64IM-NEXT:    and a6, s3, a0
+; RV64IM-NEXT:    slli a6, a6, 24
+; RV64IM-NEXT:    srliw a7, s3, 24
+; RV64IM-NEXT:    slli a7, a7, 32
+; RV64IM-NEXT:    and s0, s3, t1
+; RV64IM-NEXT:    slli s0, s0, 40
+; RV64IM-NEXT:    slli s1, s3, 56
+; RV64IM-NEXT:    or a6, a6, a7
+; RV64IM-NEXT:    or s0, s1, s0
+; RV64IM-NEXT:    or a4, a4, a5
+; RV64IM-NEXT:    or a5, s0, a6
+; RV64IM-NEXT:    lui a6, 69905
+; RV64IM-NEXT:    or a4, a5, a4
+; RV64IM-NEXT:    srli a5, a4, 4
+; RV64IM-NEXT:    and a4, a4, t4
+; RV64IM-NEXT:    and a5, a5, t4
+; RV64IM-NEXT:    slli a4, a4, 4
+; RV64IM-NEXT:    addi a6, a6, 273
+; RV64IM-NEXT:    or a4, a5, a4
+; RV64IM-NEXT:    srli a5, a4, 2
+; RV64IM-NEXT:    and a4, a4, t5
+; RV64IM-NEXT:    and a5, a5, t5
+; RV64IM-NEXT:    slli a4, a4, 2
+; RV64IM-NEXT:    slli a7, a6, 32
+; RV64IM-NEXT:    or a5, a5, a4
+; RV64IM-NEXT:    add t2, a6, a7
+; RV64IM-NEXT:    srli a6, a5, 1
+; RV64IM-NEXT:    sd a1, 0(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    and a6, a6, a1
+; RV64IM-NEXT:    lui a7, 139810
+; RV64IM-NEXT:    and a5, a5, a1
+; RV64IM-NEXT:    addi a7, a7, 546
+; RV64IM-NEXT:    slli a5, a5, 1
+; RV64IM-NEXT:    slli s0, a7, 32
+; RV64IM-NEXT:    or s1, a6, a5
+; RV64IM-NEXT:    add a5, a7, s0
+; RV64IM-NEXT:    and s0, t6, t2
+; RV64IM-NEXT:    and s2, s1, a5
+; RV64IM-NEXT:    mul t0, s2, s0
+; RV64IM-NEXT:    lui a6, %hi(.LCPI8_0)
+; RV64IM-NEXT:    ld a6, %lo(.LCPI8_0)(a6)
+; RV64IM-NEXT:    lui a7, 279620
+; RV64IM-NEXT:    and s4, t6, a5
+; RV64IM-NEXT:    addi a7, a7, 1092
+; RV64IM-NEXT:    and s5, s1, t2
+; RV64IM-NEXT:    slli s6, a7, 32
+; RV64IM-NEXT:    mul s7, s5, s4
+; RV64IM-NEXT:    add a7, a7, s6
+; RV64IM-NEXT:    and s6, t6, a6
+; RV64IM-NEXT:    and s8, s1, a7
+; RV64IM-NEXT:    and t6, t6, a7
+; RV64IM-NEXT:    and s1, s1, a6
+; RV64IM-NEXT:    mul s9, s8, s6
+; RV64IM-NEXT:    mul s10, s1, t6
+; RV64IM-NEXT:    mul s11, s2, s6
+; RV64IM-NEXT:    mul ra, s5, s0
+; RV64IM-NEXT:    mul a4, s8, t6
+; RV64IM-NEXT:    mul a3, s1, s4
+; RV64IM-NEXT:    mul a1, s2, s4
+; RV64IM-NEXT:    mul s2, s2, t6
+; RV64IM-NEXT:    mul t6, s5, t6
+; RV64IM-NEXT:    mul s5, s5, s6
+; RV64IM-NEXT:    mul s6, s1, s6
+; RV64IM-NEXT:    mul a0, s8, s0
+; RV64IM-NEXT:    mul s4, s8, s4
+; RV64IM-NEXT:    mul s0, s1, s0
+; RV64IM-NEXT:    xor t0, s7, t0
+; RV64IM-NEXT:    xor s1, s9, s10
+; RV64IM-NEXT:    xor s7, ra, s11
+; RV64IM-NEXT:    xor a3, a4, a3
+; RV64IM-NEXT:    xor a4, t0, s1
+; RV64IM-NEXT:    xor a3, s7, a3
+; RV64IM-NEXT:    and a4, a4, a5
+; RV64IM-NEXT:    and a3, a3, t2
+; RV64IM-NEXT:    xor a1, t6, a1
+; RV64IM-NEXT:    xor a0, a0, s6
+; RV64IM-NEXT:    xor t0, s5, s2
+; RV64IM-NEXT:    xor t6, s4, s0
+; RV64IM-NEXT:    xor a0, a1, a0
+; RV64IM-NEXT:    xor a1, t0, t6
+; RV64IM-NEXT:    and a0, a0, a7
+; RV64IM-NEXT:    and a1, a1, a6
+; RV64IM-NEXT:    or a3, a3, a4
+; RV64IM-NEXT:    or a0, a0, a1
+; RV64IM-NEXT:    or a0, a3, a0
+; RV64IM-NEXT:    srli a1, a0, 40
+; RV64IM-NEXT:    and a1, a1, t1
+; RV64IM-NEXT:    srli a3, a0, 56
+; RV64IM-NEXT:    or a1, a1, a3
+; RV64IM-NEXT:    srli a3, a0, 24
+; RV64IM-NEXT:    srli a4, a0, 8
+; RV64IM-NEXT:    lui t0, 4080
+; RV64IM-NEXT:    and a3, a3, t0
+; RV64IM-NEXT:    and a4, a4, t3
+; RV64IM-NEXT:    or a3, a4, a3
+; RV64IM-NEXT:    srliw a4, a0, 24
+; RV64IM-NEXT:    slli a4, a4, 32
+; RV64IM-NEXT:    and t0, a0, t0
+; RV64IM-NEXT:    slli t0, t0, 24
+; RV64IM-NEXT:    and t1, a0, t1
+; RV64IM-NEXT:    slli a0, a0, 56
+; RV64IM-NEXT:    slli t1, t1, 40
+; RV64IM-NEXT:    or a4, t0, a4
+; RV64IM-NEXT:    or a0, a0, t1
+; RV64IM-NEXT:    or a1, a3, a1
+; RV64IM-NEXT:    or a0, a0, a4
+; RV64IM-NEXT:    or a0, a0, a1
+; RV64IM-NEXT:    srli a1, a0, 4
+; RV64IM-NEXT:    and a0, a0, t4
+; RV64IM-NEXT:    and a1, a1, t4
+; RV64IM-NEXT:    slli a0, a0, 4
+; RV64IM-NEXT:    or a0, a1, a0
+; RV64IM-NEXT:    srli a1, a0, 2
+; RV64IM-NEXT:    and a0, a0, t5
+; RV64IM-NEXT:    and a1, a1, t5
+; RV64IM-NEXT:    slli a0, a0, 2
+; RV64IM-NEXT:    or a0, a1, a0
+; RV64IM-NEXT:    srli a1, a0, 1
+; RV64IM-NEXT:    and t0, a2, t2
+; RV64IM-NEXT:    ld s4, 8(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a4, s4, a5
+; RV64IM-NEXT:    and t1, a2, a5
+; RV64IM-NEXT:    and t4, s4, t2
+; RV64IM-NEXT:    mul t5, a4, t0
+; RV64IM-NEXT:    mul t6, t4, t1
+; RV64IM-NEXT:    and s0, s4, a7
+; RV64IM-NEXT:    and t3, a2, a7
+; RV64IM-NEXT:    mul s1, t4, t0
+; RV64IM-NEXT:    mul s2, s0, t3
+; RV64IM-NEXT:    ld a3, 0(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and a1, a1, a3
+; RV64IM-NEXT:    and a0, a0, a3
+; RV64IM-NEXT:    slli a0, a0, 1
+; RV64IM-NEXT:    and a3, a2, a6
+; RV64IM-NEXT:    or a0, a1, a0
+; RV64IM-NEXT:    sd a0, 0(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    mul a0, s0, a3
+; RV64IM-NEXT:    and a1, s4, a6
+; RV64IM-NEXT:    mul s4, a4, a3
+; RV64IM-NEXT:    mul s5, a1, t3
+; RV64IM-NEXT:    mul s6, a1, t1
+; RV64IM-NEXT:    xor t5, t6, t5
+; RV64IM-NEXT:    xor t6, s1, s2
+; RV64IM-NEXT:    mul s1, a4, t1
+; RV64IM-NEXT:    mul s2, t4, t3
+; RV64IM-NEXT:    mul a4, a4, t3
+; RV64IM-NEXT:    mul s7, s0, t1
+; RV64IM-NEXT:    mul s0, s0, t0
+; RV64IM-NEXT:    mul t4, t4, a3
+; RV64IM-NEXT:    xor a0, t5, a0
+; RV64IM-NEXT:    xor t5, t6, s4
+; RV64IM-NEXT:    xor a0, a0, s5
+; RV64IM-NEXT:    xor t5, t5, s6
+; RV64IM-NEXT:    and a0, a0, a5
+; RV64IM-NEXT:    and t6, t5, t2
+; RV64IM-NEXT:    mul t5, a1, a3
+; RV64IM-NEXT:    mul s4, a1, t0
+; RV64IM-NEXT:    xor a1, s2, s1
+; RV64IM-NEXT:    xor a4, a4, s7
+; RV64IM-NEXT:    xor s0, a1, s0
+; RV64IM-NEXT:    xor a4, t4, a4
+; RV64IM-NEXT:    ld a2, 16(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and s1, a2, t2
+; RV64IM-NEXT:    and t4, s3, a5
+; RV64IM-NEXT:    and s2, a2, a5
+; RV64IM-NEXT:    and a1, s3, t2
+; RV64IM-NEXT:    mul s5, t4, s1
+; RV64IM-NEXT:    mul s6, a1, s2
+; RV64IM-NEXT:    xor s0, s0, t5
+; RV64IM-NEXT:    xor a4, a4, s4
+; RV64IM-NEXT:    and t5, s3, a7
+; RV64IM-NEXT:    and s4, a2, a7
+; RV64IM-NEXT:    mul s7, a1, s1
+; RV64IM-NEXT:    mul s8, t5, s4
+; RV64IM-NEXT:    and s0, s0, a7
+; RV64IM-NEXT:    and a4, a4, a6
+; RV64IM-NEXT:    or t6, t6, a0
+; RV64IM-NEXT:    or s0, s0, a4
+; RV64IM-NEXT:    xor a4, s6, s5
+; RV64IM-NEXT:    and s5, a2, a6
+; RV64IM-NEXT:    mul s6, t5, s5
+; RV64IM-NEXT:    and a0, s3, a6
+; RV64IM-NEXT:    mul s3, a0, s4
+; RV64IM-NEXT:    mul s9, t4, s5
+; RV64IM-NEXT:    xor s7, s7, s8
+; RV64IM-NEXT:    mul s8, a0, s2
+; RV64IM-NEXT:    mul s10, t4, s2
+; RV64IM-NEXT:    mul s11, a1, s4
+; RV64IM-NEXT:    mul s4, t4, s4
+; RV64IM-NEXT:    mul s2, t5, s2
+; RV64IM-NEXT:    mul ra, t5, s1
+; RV64IM-NEXT:    mul a2, a1, s5
+; RV64IM-NEXT:    mul s5, a0, s5
+; RV64IM-NEXT:    mul s1, a0, s1
+; RV64IM-NEXT:    xor a4, a4, s6
+; RV64IM-NEXT:    xor s6, s7, s9
+; RV64IM-NEXT:    xor a4, a4, s3
+; RV64IM-NEXT:    xor s3, s6, s8
+; RV64IM-NEXT:    and a4, a4, a5
+; RV64IM-NEXT:    and s3, s3, t2
+; RV64IM-NEXT:    xor s6, s11, s10
+; RV64IM-NEXT:    xor s2, s4, s2
+; RV64IM-NEXT:    xor s4, s6, ra
+; RV64IM-NEXT:    xor a2, a2, s2
+; RV64IM-NEXT:    xor s2, s4, s5
+; RV64IM-NEXT:    xor a2, a2, s1
+; RV64IM-NEXT:    mul s1, t4, t0
+; RV64IM-NEXT:    mul s4, a1, t1
+; RV64IM-NEXT:    and s2, s2, a7
+; RV64IM-NEXT:    and a2, a2, a6
+; RV64IM-NEXT:    mul s5, a1, t0
+; RV64IM-NEXT:    mul s6, t5, t3
+; RV64IM-NEXT:    or a4, s3, a4
+; RV64IM-NEXT:    or a2, s2, a2
+; RV64IM-NEXT:    or t6, t6, s0
+; RV64IM-NEXT:    or a2, a4, a2
+; RV64IM-NEXT:    ld a4, 0(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    srli a4, a4, 1
+; RV64IM-NEXT:    xor a2, a2, t6
+; RV64IM-NEXT:    xor t6, s4, s1
+; RV64IM-NEXT:    mul s0, t5, a3
+; RV64IM-NEXT:    mul s1, a0, t3
+; RV64IM-NEXT:    mul s2, t4, a3
+; RV64IM-NEXT:    xor s3, s5, s6
+; RV64IM-NEXT:    mul s4, a0, t1
+; RV64IM-NEXT:    mul s5, t4, t1
+; RV64IM-NEXT:    mul s6, a1, t3
+; RV64IM-NEXT:    mul t3, t4, t3
+; RV64IM-NEXT:    mul t1, t5, t1
+; RV64IM-NEXT:    mul t4, t5, t0
+; RV64IM-NEXT:    mul a1, a1, a3
+; RV64IM-NEXT:    mul a3, a0, a3
+; RV64IM-NEXT:    mul a0, a0, t0
+; RV64IM-NEXT:    xor t0, t6, s0
+; RV64IM-NEXT:    xor t5, s3, s2
+; RV64IM-NEXT:    xor t0, t0, s1
+; RV64IM-NEXT:    xor t5, t5, s4
+; RV64IM-NEXT:    and a5, t0, a5
+; RV64IM-NEXT:    and t0, t5, t2
+; RV64IM-NEXT:    xor t2, s6, s5
+; RV64IM-NEXT:    xor t1, t3, t1
+; RV64IM-NEXT:    xor t2, t2, t4
+; RV64IM-NEXT:    xor a1, a1, t1
+; RV64IM-NEXT:    xor a3, t2, a3
+; RV64IM-NEXT:    xor a0, a1, a0
+; RV64IM-NEXT:    and a1, a3, a7
+; RV64IM-NEXT:    and a0, a0, a6
+; RV64IM-NEXT:    or a3, t0, a5
+; RV64IM-NEXT:    or a0, a1, a0
+; RV64IM-NEXT:    xor a1, a4, a2
+; RV64IM-NEXT:    or a0, a3, a0
+; RV64IM-NEXT:    ld ra, 120(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s0, 112(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s1, 104(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s2, 96(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s3, 88(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s4, 80(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s5, 72(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s6, 64(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s7, 56(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s8, 48(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s9, 40(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s10, 32(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s11, 24(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    .cfi_restore ra
+; RV64IM-NEXT:    .cfi_restore s0
+; RV64IM-NEXT:    .cfi_restore s1
+; RV64IM-NEXT:    .cfi_restore s2
+; RV64IM-NEXT:    .cfi_restore s3
+; RV64IM-NEXT:    .cfi_restore s4
+; RV64IM-NEXT:    .cfi_restore s5
+; RV64IM-NEXT:    .cfi_restore s6
+; RV64IM-NEXT:    .cfi_restore s7
+; RV64IM-NEXT:    .cfi_restore s8
+; RV64IM-NEXT:    .cfi_restore s9
+; RV64IM-NEXT:    .cfi_restore s10
+; RV64IM-NEXT:    .cfi_restore s11
+; RV64IM-NEXT:    addi sp, sp, 128
+; RV64IM-NEXT:    .cfi_def_cfa_offset 0
+; RV64IM-NEXT:    ret
+;
+; RV32IMZBS-LABEL: clmul_i128:
+; RV32IMZBS:       # %bb.0:
+; RV32IMZBS-NEXT:    addi sp, sp, -240
+; RV32IMZBS-NEXT:    .cfi_def_cfa_offset 240
+; RV32IMZBS-NEXT:    sw ra, 236(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s0, 232(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s1, 228(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s2, 224(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s3, 220(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s4, 216(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s5, 212(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s6, 208(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s7, 204(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s8, 200(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s9, 196(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s10, 192(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s11, 188(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    .cfi_offset ra, -4
+; RV32IMZBS-NEXT:    .cfi_offset s0, -8
+; RV32IMZBS-NEXT:    .cfi_offset s1, -12
+; RV32IMZBS-NEXT:    .cfi_offset s2, -16
+; RV32IMZBS-NEXT:    .cfi_offset s3, -20
+; RV32IMZBS-NEXT:    .cfi_offset s4, -24
+; RV32IMZBS-NEXT:    .cfi_offset s5, -28
+; RV32IMZBS-NEXT:    .cfi_offset s6, -32
+; RV32IMZBS-NEXT:    .cfi_offset s7, -36
+; RV32IMZBS-NEXT:    .cfi_offset s8, -40
+; RV32IMZBS-NEXT:    .cfi_offset s9, -44
+; RV32IMZBS-NEXT:    .cfi_offset s10, -48
+; RV32IMZBS-NEXT:    .cfi_offset s11, -52
+; RV32IMZBS-NEXT:    sw a2, 144(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv a6, a1
+; RV32IMZBS-NEXT:    sw a0, 80(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw a5, 4(a2)
+; RV32IMZBS-NEXT:    sw a5, 76(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lui a0, 16
+; RV32IMZBS-NEXT:    lw a1, 8(a2)
+; RV32IMZBS-NEXT:    sw a1, 148(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw a1, 12(a2)
+; RV32IMZBS-NEXT:    sw a1, 68(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    addi t0, a0, -256
+; RV32IMZBS-NEXT:    srli a0, a5, 8
+; RV32IMZBS-NEXT:    srli a3, a5, 24
+; RV32IMZBS-NEXT:    and a4, a5, t0
+; RV32IMZBS-NEXT:    slli a5, a5, 24
+; RV32IMZBS-NEXT:    and a0, a0, t0
+; RV32IMZBS-NEXT:    slli a4, a4, 8
+; RV32IMZBS-NEXT:    or a0, a0, a3
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    lui a3, 61681
+; RV32IMZBS-NEXT:    or a0, a4, a0
+; RV32IMZBS-NEXT:    addi t5, a3, -241
+; RV32IMZBS-NEXT:    srli a3, a0, 4
+; RV32IMZBS-NEXT:    and a0, a0, t5
+; RV32IMZBS-NEXT:    and a3, a3, t5
+; RV32IMZBS-NEXT:    slli a0, a0, 4
+; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    lui a3, 209715
+; RV32IMZBS-NEXT:    srli a4, a0, 2
+; RV32IMZBS-NEXT:    addi t4, a3, 819
+; RV32IMZBS-NEXT:    and a3, a4, t4
+; RV32IMZBS-NEXT:    and a0, a0, t4
+; RV32IMZBS-NEXT:    slli a0, a0, 2
+; RV32IMZBS-NEXT:    lui a1, 349525
+; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    addi s10, a1, 1365
+; RV32IMZBS-NEXT:    srli a3, a0, 1
+; RV32IMZBS-NEXT:    and a0, a0, s10
+; RV32IMZBS-NEXT:    and a3, a3, s10
+; RV32IMZBS-NEXT:    slli a0, a0, 1
+; RV32IMZBS-NEXT:    or a5, a3, a0
+; RV32IMZBS-NEXT:    sw a5, 132(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a0, a5, 8
+; RV32IMZBS-NEXT:    and a0, a0, t0
+; RV32IMZBS-NEXT:    srli a3, a5, 24
+; RV32IMZBS-NEXT:    and a4, a5, t0
+; RV32IMZBS-NEXT:    slli a5, a5, 24
+; RV32IMZBS-NEXT:    slli a4, a4, 8
+; RV32IMZBS-NEXT:    or a0, a0, a3
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    or a0, a4, a0
+; RV32IMZBS-NEXT:    sw a6, 156(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw a7, 4(a6)
+; RV32IMZBS-NEXT:    srli a3, a0, 4
+; RV32IMZBS-NEXT:    and a3, a3, t5
+; RV32IMZBS-NEXT:    and a0, a0, t5
+; RV32IMZBS-NEXT:    slli a0, a0, 4
+; RV32IMZBS-NEXT:    lw a2, 8(a6)
+; RV32IMZBS-NEXT:    sw a2, 40(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw a2, 12(a6)
+; RV32IMZBS-NEXT:    sw a2, 72(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    srli a3, a0, 2
+; RV32IMZBS-NEXT:    sw a7, 64(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a4, a7, 8
+; RV32IMZBS-NEXT:    and a3, a3, t4
+; RV32IMZBS-NEXT:    and a4, a4, t0
+; RV32IMZBS-NEXT:    srli a5, a7, 24
+; RV32IMZBS-NEXT:    and a6, a7, t0
+; RV32IMZBS-NEXT:    slli a6, a6, 8
+; RV32IMZBS-NEXT:    slli a7, a7, 24
+; RV32IMZBS-NEXT:    or a4, a4, a5
+; RV32IMZBS-NEXT:    or a5, a7, a6
+; RV32IMZBS-NEXT:    and a0, a0, t4
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    srli a5, a4, 4
+; RV32IMZBS-NEXT:    and a4, a4, t5
+; RV32IMZBS-NEXT:    and a5, a5, t5
+; RV32IMZBS-NEXT:    slli a4, a4, 4
+; RV32IMZBS-NEXT:    slli a0, a0, 2
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    srli a5, a4, 2
+; RV32IMZBS-NEXT:    and a4, a4, t4
+; RV32IMZBS-NEXT:    and a5, a5, t4
+; RV32IMZBS-NEXT:    slli a4, a4, 2
+; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    srli a3, a4, 1
+; RV32IMZBS-NEXT:    and a4, a4, s10
+; RV32IMZBS-NEXT:    and a3, a3, s10
+; RV32IMZBS-NEXT:    slli a4, a4, 1
+; RV32IMZBS-NEXT:    srli a5, a0, 1
+; RV32IMZBS-NEXT:    or t1, a3, a4
+; RV32IMZBS-NEXT:    sw t1, 124(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a3, a5, s10
+; RV32IMZBS-NEXT:    srli a4, t1, 8
+; RV32IMZBS-NEXT:    and a0, a0, s10
+; RV32IMZBS-NEXT:    and a4, a4, t0
+; RV32IMZBS-NEXT:    srli a5, t1, 24
+; RV32IMZBS-NEXT:    and a6, t1, t0
+; RV32IMZBS-NEXT:    slli t1, t1, 24
+; RV32IMZBS-NEXT:    slli a6, a6, 8
+; RV32IMZBS-NEXT:    or a4, a4, a5
+; RV32IMZBS-NEXT:    or a5, t1, a6
+; RV32IMZBS-NEXT:    slli a0, a0, 1
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    srli a5, a4, 4
+; RV32IMZBS-NEXT:    and a4, a4, t5
+; RV32IMZBS-NEXT:    and a5, a5, t5
+; RV32IMZBS-NEXT:    slli a4, a4, 4
+; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    srli a3, a4, 2
+; RV32IMZBS-NEXT:    and a4, a4, t4
+; RV32IMZBS-NEXT:    and a3, a3, t4
+; RV32IMZBS-NEXT:    slli a4, a4, 2
+; RV32IMZBS-NEXT:    lui a5, 69905
+; RV32IMZBS-NEXT:    or a4, a3, a4
+; RV32IMZBS-NEXT:    addi a7, a5, 273
+; RV32IMZBS-NEXT:    srli a5, a4, 1
+; RV32IMZBS-NEXT:    and a5, a5, s10
+; RV32IMZBS-NEXT:    and a4, a4, s10
+; RV32IMZBS-NEXT:    slli a4, a4, 1
+; RV32IMZBS-NEXT:    lui a6, 139810
+; RV32IMZBS-NEXT:    or t1, a5, a4
+; RV32IMZBS-NEXT:    addi t2, a6, 546
+; RV32IMZBS-NEXT:    and t6, a0, a7
+; RV32IMZBS-NEXT:    and s0, t1, t2
+; RV32IMZBS-NEXT:    and s2, a0, t2
+; RV32IMZBS-NEXT:    and s3, t1, a7
+; RV32IMZBS-NEXT:    lui a5, 559241
+; RV32IMZBS-NEXT:    lui a6, 279620
+; RV32IMZBS-NEXT:    addi s1, a5, -1912
+; RV32IMZBS-NEXT:    addi t3, a6, 1092
+; RV32IMZBS-NEXT:    and s4, a0, s1
+; RV32IMZBS-NEXT:    and s5, t1, t3
+; RV32IMZBS-NEXT:    and a0, a0, t3
+; RV32IMZBS-NEXT:    and t1, t1, s1
+; RV32IMZBS-NEXT:    mul s6, s0, t6
+; RV32IMZBS-NEXT:    mul s7, s3, s2
+; RV32IMZBS-NEXT:    mul s8, s5, s4
+; RV32IMZBS-NEXT:    mul s9, t1, a0
+; RV32IMZBS-NEXT:    mul s11, s0, s4
+; RV32IMZBS-NEXT:    mul ra, s3, t6
+; RV32IMZBS-NEXT:    mul a6, s5, a0
+; RV32IMZBS-NEXT:    mul a5, t1, s2
+; RV32IMZBS-NEXT:    mul a4, s0, s2
+; RV32IMZBS-NEXT:    mul a3, s3, a0
+; RV32IMZBS-NEXT:    mul a2, s5, t6
+; RV32IMZBS-NEXT:    mul a1, t1, s4
+; RV32IMZBS-NEXT:    mul a0, s0, a0
+; RV32IMZBS-NEXT:    mul s0, s3, s4
+; RV32IMZBS-NEXT:    mul s2, s5, s2
+; RV32IMZBS-NEXT:    mul t1, t1, t6
+; RV32IMZBS-NEXT:    xor t6, s7, s6
+; RV32IMZBS-NEXT:    xor s3, s8, s9
+; RV32IMZBS-NEXT:    xor s4, ra, s11
+; RV32IMZBS-NEXT:    xor a5, a6, a5
+; RV32IMZBS-NEXT:    xor a6, t6, s3
+; RV32IMZBS-NEXT:    xor a5, s4, a5
+; RV32IMZBS-NEXT:    xor a3, a3, a4
+; RV32IMZBS-NEXT:    xor a1, a2, a1
+; RV32IMZBS-NEXT:    and a2, a6, t2
+; RV32IMZBS-NEXT:    and a5, a5, a7
+; RV32IMZBS-NEXT:    xor a0, s0, a0
+; RV32IMZBS-NEXT:    xor a4, s2, t1
+; RV32IMZBS-NEXT:    xor a1, a3, a1
+; RV32IMZBS-NEXT:    xor a0, a0, a4
+; RV32IMZBS-NEXT:    and a1, a1, t3
+; RV32IMZBS-NEXT:    and a0, a0, s1
+; RV32IMZBS-NEXT:    or a2, a5, a2
+; RV32IMZBS-NEXT:    or a0, a1, a0
+; RV32IMZBS-NEXT:    or a0, a2, a0
+; RV32IMZBS-NEXT:    srli a1, a0, 8
+; RV32IMZBS-NEXT:    and a1, a1, t0
+; RV32IMZBS-NEXT:    srli a2, a0, 24
+; RV32IMZBS-NEXT:    and a3, a0, t0
+; RV32IMZBS-NEXT:    slli a0, a0, 24
+; RV32IMZBS-NEXT:    slli a3, a3, 8
+; RV32IMZBS-NEXT:    or a1, a1, a2
+; RV32IMZBS-NEXT:    or a3, a0, a3
+; RV32IMZBS-NEXT:    lw a0, 156(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a0, 0(a0)
+; RV32IMZBS-NEXT:    or a1, a3, a1
+; RV32IMZBS-NEXT:    srli a2, a1, 4
+; RV32IMZBS-NEXT:    and a1, a1, t5
+; RV32IMZBS-NEXT:    and a2, a2, t5
+; RV32IMZBS-NEXT:    slli a1, a1, 4
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    srli a2, a1, 2
+; RV32IMZBS-NEXT:    and a2, a2, t4
+; RV32IMZBS-NEXT:    and a1, a1, t4
+; RV32IMZBS-NEXT:    sw a0, 84(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a3, a0, 8
+; RV32IMZBS-NEXT:    slli a1, a1, 2
+; RV32IMZBS-NEXT:    and a3, a3, t0
+; RV32IMZBS-NEXT:    srli a4, a0, 24
+; RV32IMZBS-NEXT:    and a5, a0, t0
+; RV32IMZBS-NEXT:    slli a5, a5, 8
+; RV32IMZBS-NEXT:    slli a6, a0, 24
+; RV32IMZBS-NEXT:    or a3, a3, a4
+; RV32IMZBS-NEXT:    or a4, a6, a5
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    or a3, a4, a3
+; RV32IMZBS-NEXT:    srli a2, a3, 4
+; RV32IMZBS-NEXT:    and a3, a3, t5
+; RV32IMZBS-NEXT:    and a2, a2, t5
+; RV32IMZBS-NEXT:    slli a3, a3, 4
+; RV32IMZBS-NEXT:    srli a4, a1, 1
+; RV32IMZBS-NEXT:    or a2, a2, a3
+; RV32IMZBS-NEXT:    srli a3, a2, 2
+; RV32IMZBS-NEXT:    and a2, a2, t4
+; RV32IMZBS-NEXT:    and a3, a3, t4
+; RV32IMZBS-NEXT:    slli a2, a2, 2
+; RV32IMZBS-NEXT:    lui s11, 349525
+; RV32IMZBS-NEXT:    addi s11, s11, 1364
+; RV32IMZBS-NEXT:    sw s11, 156(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a2, a3, a2
+; RV32IMZBS-NEXT:    srli a3, a2, 1
+; RV32IMZBS-NEXT:    and a2, a2, s10
+; RV32IMZBS-NEXT:    and a3, a3, s10
+; RV32IMZBS-NEXT:    slli a2, a2, 1
+; RV32IMZBS-NEXT:    and a1, a1, s10
+; RV32IMZBS-NEXT:    mv ra, s10
+; RV32IMZBS-NEXT:    or a2, a3, a2
+; RV32IMZBS-NEXT:    sw a7, 164(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw a0, 132(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t1, a0, a7
+; RV32IMZBS-NEXT:    sw t2, 184(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and s7, a2, t2
+; RV32IMZBS-NEXT:    and t2, a0, t2
+; RV32IMZBS-NEXT:    and a3, a2, a7
+; RV32IMZBS-NEXT:    and s0, a0, s1
+; RV32IMZBS-NEXT:    sw t3, 168(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and t6, a0, t3
+; RV32IMZBS-NEXT:    and t3, a2, t3
+; RV32IMZBS-NEXT:    and s10, a2, s1
+; RV32IMZBS-NEXT:    mv a7, t1
+; RV32IMZBS-NEXT:    mul a2, s7, t1
+; RV32IMZBS-NEXT:    mv t1, a3
+; RV32IMZBS-NEXT:    mul a3, a3, t2
+; RV32IMZBS-NEXT:    mul a5, t3, s0
+; RV32IMZBS-NEXT:    mul a6, s10, t6
+; RV32IMZBS-NEXT:    mul s2, s7, s0
+; RV32IMZBS-NEXT:    mul s3, t1, a7
+; RV32IMZBS-NEXT:    mul s4, t3, t6
+; RV32IMZBS-NEXT:    mul s5, s10, t2
+; RV32IMZBS-NEXT:    mul s6, s7, t2
+; RV32IMZBS-NEXT:    sw t2, 120(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv a0, s7
+; RV32IMZBS-NEXT:    sw s7, 140(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul s7, t1, t6
+; RV32IMZBS-NEXT:    sw t1, 136(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul s8, t3, a7
+; RV32IMZBS-NEXT:    sw a7, 96(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw t3, 132(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul s9, s10, s0
+; RV32IMZBS-NEXT:    sw s10, 128(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a4, s11
+; RV32IMZBS-NEXT:    slli a1, a1, 1
+; RV32IMZBS-NEXT:    or a1, a4, a1
+; RV32IMZBS-NEXT:    sw a1, 116(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a2, a3, a2
+; RV32IMZBS-NEXT:    xor a3, a5, a6
+; RV32IMZBS-NEXT:    lw a1, 144(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a1, 0(a1)
+; RV32IMZBS-NEXT:    xor a4, s3, s2
+; RV32IMZBS-NEXT:    xor a5, s4, s5
+; RV32IMZBS-NEXT:    xor a2, a2, a3
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    xor a3, s7, s6
+; RV32IMZBS-NEXT:    xor a5, s8, s9
+; RV32IMZBS-NEXT:    srli a6, a1, 8
+; RV32IMZBS-NEXT:    sw t0, 176(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and s2, a1, t0
+; RV32IMZBS-NEXT:    and a6, a6, t0
+; RV32IMZBS-NEXT:    slli s2, s2, 8
+; RV32IMZBS-NEXT:    mul s3, a0, t6
+; RV32IMZBS-NEXT:    mul s4, t1, s0
+; RV32IMZBS-NEXT:    sw a1, 144(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli s5, a1, 24
+; RV32IMZBS-NEXT:    slli s6, a1, 24
+; RV32IMZBS-NEXT:    or a6, a6, s5
+; RV32IMZBS-NEXT:    or s2, s6, s2
+; RV32IMZBS-NEXT:    xor a3, a3, a5
+; RV32IMZBS-NEXT:    or a5, s2, a6
+; RV32IMZBS-NEXT:    mul a6, t3, t2
+; RV32IMZBS-NEXT:    mul s2, s10, a7
+; RV32IMZBS-NEXT:    srli s5, a5, 4
+; RV32IMZBS-NEXT:    sw t5, 180(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a5, t5
+; RV32IMZBS-NEXT:    and s5, s5, t5
+; RV32IMZBS-NEXT:    slli a5, a5, 4
+; RV32IMZBS-NEXT:    xor s3, s4, s3
+; RV32IMZBS-NEXT:    or a5, s5, a5
+; RV32IMZBS-NEXT:    srli s4, a5, 2
+; RV32IMZBS-NEXT:    sw t4, 152(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a5, t4
+; RV32IMZBS-NEXT:    and s4, s4, t4
+; RV32IMZBS-NEXT:    slli a5, a5, 2
+; RV32IMZBS-NEXT:    xor a6, a6, s2
+; RV32IMZBS-NEXT:    or a5, s4, a5
+; RV32IMZBS-NEXT:    srli s2, a5, 1
+; RV32IMZBS-NEXT:    sw ra, 172(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, a5, ra
+; RV32IMZBS-NEXT:    and s2, s2, ra
+; RV32IMZBS-NEXT:    slli a5, a5, 1
+; RV32IMZBS-NEXT:    xor a6, s3, a6
+; RV32IMZBS-NEXT:    or a5, s2, a5
+; RV32IMZBS-NEXT:    lw t1, 184(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a0, a2, t1
+; RV32IMZBS-NEXT:    sw a0, 92(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw t0, 164(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s10, a4, t0
+; RV32IMZBS-NEXT:    lw t5, 168(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s3, a3, t5
+; RV32IMZBS-NEXT:    mv a1, s1
+; RV32IMZBS-NEXT:    sw s1, 160(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and s1, a6, s1
+; RV32IMZBS-NEXT:    and t4, a5, t0
+; RV32IMZBS-NEXT:    and a3, a5, t1
+; RV32IMZBS-NEXT:    and a6, a5, a1
+; RV32IMZBS-NEXT:    and a0, a5, t5
+; RV32IMZBS-NEXT:    lw a7, 124(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s4, a7, t1
+; RV32IMZBS-NEXT:    and s2, a7, t0
+; RV32IMZBS-NEXT:    and a2, a7, t5
+; RV32IMZBS-NEXT:    and a7, a7, a1
+; RV32IMZBS-NEXT:    mul t2, s4, t4
+; RV32IMZBS-NEXT:    mul s5, s2, a3
+; RV32IMZBS-NEXT:    mul s6, a2, a6
+; RV32IMZBS-NEXT:    sw a0, 112(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul s7, a7, a0
+; RV32IMZBS-NEXT:    mul s8, s4, a6
+; RV32IMZBS-NEXT:    mv a1, a6
+; RV32IMZBS-NEXT:    sw a6, 108(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul s9, s2, t4
+; RV32IMZBS-NEXT:    sw t4, 100(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul s11, a2, a0
+; RV32IMZBS-NEXT:    mul ra, a7, a3
+; RV32IMZBS-NEXT:    mul t3, s4, a3
+; RV32IMZBS-NEXT:    sw a3, 104(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul t0, s2, a0
+; RV32IMZBS-NEXT:    mul t1, a2, t4
+; RV32IMZBS-NEXT:    mul a5, a7, a6
+; RV32IMZBS-NEXT:    mul a6, s4, a0
+; RV32IMZBS-NEXT:    mul a4, s2, a1
+; RV32IMZBS-NEXT:    mul a1, a2, a3
+; RV32IMZBS-NEXT:    mul a0, a7, t4
+; RV32IMZBS-NEXT:    lw a3, 92(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    or s10, s10, a3
+; RV32IMZBS-NEXT:    or s1, s3, s1
+; RV32IMZBS-NEXT:    xor t2, s5, t2
+; RV32IMZBS-NEXT:    xor s3, s6, s7
+; RV32IMZBS-NEXT:    xor s5, s9, s8
+; RV32IMZBS-NEXT:    xor s6, s11, ra
+; RV32IMZBS-NEXT:    xor t2, t2, s3
+; RV32IMZBS-NEXT:    xor s3, s5, s6
+; RV32IMZBS-NEXT:    xor t0, t0, t3
+; RV32IMZBS-NEXT:    xor a5, t1, a5
+; RV32IMZBS-NEXT:    xor a3, a4, a6
+; RV32IMZBS-NEXT:    xor a0, a1, a0
+; RV32IMZBS-NEXT:    xor a1, t0, a5
+; RV32IMZBS-NEXT:    xor a0, a3, a0
+; RV32IMZBS-NEXT:    lw s9, 184(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, t2, s9
+; RV32IMZBS-NEXT:    lw s11, 164(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, s3, s11
+; RV32IMZBS-NEXT:    mv s8, t5
+; RV32IMZBS-NEXT:    and a1, a1, t5
+; RV32IMZBS-NEXT:    lw t5, 160(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a0, a0, t5
+; RV32IMZBS-NEXT:    or a3, a4, a3
+; RV32IMZBS-NEXT:    or a0, a1, a0
+; RV32IMZBS-NEXT:    or a1, s10, s1
+; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    lw a1, 116(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    srli a1, a1, 1
+; RV32IMZBS-NEXT:    xor a0, a1, a0
+; RV32IMZBS-NEXT:    srli a1, a0, 8
+; RV32IMZBS-NEXT:    lw t3, 176(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, t3
+; RV32IMZBS-NEXT:    srli a3, a0, 24
+; RV32IMZBS-NEXT:    and a4, a0, t3
+; RV32IMZBS-NEXT:    slli a5, a0, 24
+; RV32IMZBS-NEXT:    slli a4, a4, 8
+; RV32IMZBS-NEXT:    or a0, a1, a3
+; RV32IMZBS-NEXT:    or s3, a5, a4
+; RV32IMZBS-NEXT:    lw s6, 96(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a1, s4, s6
+; RV32IMZBS-NEXT:    mul a3, s4, s0
+; RV32IMZBS-NEXT:    lw s7, 120(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a4, s4, s7
+; RV32IMZBS-NEXT:    mul a5, s4, t6
+; RV32IMZBS-NEXT:    mul a6, s2, s7
+; RV32IMZBS-NEXT:    mul t0, a2, s0
+; RV32IMZBS-NEXT:    mul t1, a7, t6
+; RV32IMZBS-NEXT:    mul t2, s2, s6
+; RV32IMZBS-NEXT:    mul s1, a2, t6
+; RV32IMZBS-NEXT:    mul s4, a7, s7
+; RV32IMZBS-NEXT:    mul t4, s2, t6
+; RV32IMZBS-NEXT:    mul s2, s2, s0
+; RV32IMZBS-NEXT:    mul s0, a7, s0
+; RV32IMZBS-NEXT:    mul s5, a2, s6
+; RV32IMZBS-NEXT:    mul a2, a2, s7
+; RV32IMZBS-NEXT:    mul a7, a7, s6
+; RV32IMZBS-NEXT:    or a0, s3, a0
+; RV32IMZBS-NEXT:    xor a1, a6, a1
+; RV32IMZBS-NEXT:    xor a6, t0, t1
+; RV32IMZBS-NEXT:    xor a3, t2, a3
+; RV32IMZBS-NEXT:    xor t0, s1, s4
+; RV32IMZBS-NEXT:    xor a1, a1, a6
+; RV32IMZBS-NEXT:    xor a3, a3, t0
+; RV32IMZBS-NEXT:    xor a4, t4, a4
+; RV32IMZBS-NEXT:    xor a6, s5, s0
+; RV32IMZBS-NEXT:    xor a5, s2, a5
+; RV32IMZBS-NEXT:    xor a2, a2, a7
+; RV32IMZBS-NEXT:    xor a4, a4, a6
+; RV32IMZBS-NEXT:    xor a2, a5, a2
+; RV32IMZBS-NEXT:    mv t1, s9
+; RV32IMZBS-NEXT:    and a1, a1, s9
+; RV32IMZBS-NEXT:    and a3, a3, s11
+; RV32IMZBS-NEXT:    mv a6, s8
+; RV32IMZBS-NEXT:    and a4, a4, s8
+; RV32IMZBS-NEXT:    and a2, a2, t5
+; RV32IMZBS-NEXT:    or a1, a3, a1
+; RV32IMZBS-NEXT:    or a2, a4, a2
+; RV32IMZBS-NEXT:    or a1, a1, a2
+; RV32IMZBS-NEXT:    srli a2, a0, 4
+; RV32IMZBS-NEXT:    lw a5, 180(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a2, a2, a5
+; RV32IMZBS-NEXT:    and a0, a0, a5
+; RV32IMZBS-NEXT:    slli a0, a0, 4
+; RV32IMZBS-NEXT:    srli a3, a1, 8
+; RV32IMZBS-NEXT:    or a0, a2, a0
+; RV32IMZBS-NEXT:    and a2, a3, t3
+; RV32IMZBS-NEXT:    srli a3, a1, 24
+; RV32IMZBS-NEXT:    and a4, a1, t3
+; RV32IMZBS-NEXT:    slli a1, a1, 24
+; RV32IMZBS-NEXT:    slli a4, a4, 8
+; RV32IMZBS-NEXT:    or a2, a2, a3
+; RV32IMZBS-NEXT:    or a1, a1, a4
+; RV32IMZBS-NEXT:    srli a3, a0, 2
+; RV32IMZBS-NEXT:    lw a4, 152(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a0, a0, a4
+; RV32IMZBS-NEXT:    and a3, a3, a4
+; RV32IMZBS-NEXT:    slli a0, a0, 2
+; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    or a1, a1, a2
+; RV32IMZBS-NEXT:    srli a2, a0, 1
+; RV32IMZBS-NEXT:    lw a3, 172(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a7, a0, a3
+; RV32IMZBS-NEXT:    lw a0, 156(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a0, a2, a0
+; RV32IMZBS-NEXT:    sw a0, 96(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a2, a7, 1
+; RV32IMZBS-NEXT:    srli a0, a1, 4
+; RV32IMZBS-NEXT:    and a1, a1, a5
+; RV32IMZBS-NEXT:    and a0, a0, a5
+; RV32IMZBS-NEXT:    slli a1, a1, 4
+; RV32IMZBS-NEXT:    lw a3, 144(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s9, a3, s11
+; RV32IMZBS-NEXT:    sw s9, 116(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv a4, t1
+; RV32IMZBS-NEXT:    and s8, a3, t1
+; RV32IMZBS-NEXT:    sw s8, 124(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv a7, t5
+; RV32IMZBS-NEXT:    and t1, a3, t5
+; RV32IMZBS-NEXT:    and s3, a3, a6
+; RV32IMZBS-NEXT:    sw s3, 144(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw ra, 40(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s2, ra, a4
+; RV32IMZBS-NEXT:    sw s2, 44(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and t4, ra, s11
+; RV32IMZBS-NEXT:    sw t4, 60(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a3, s2, s9
+; RV32IMZBS-NEXT:    mul a4, t4, s8
+; RV32IMZBS-NEXT:    and s4, ra, a6
+; RV32IMZBS-NEXT:    mv s10, a6
+; RV32IMZBS-NEXT:    sw s4, 56(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and s6, ra, t5
+; RV32IMZBS-NEXT:    sw s6, 52(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a5, s4, t1
+; RV32IMZBS-NEXT:    mul a6, s6, s3
+; RV32IMZBS-NEXT:    mul t0, s2, t1
+; RV32IMZBS-NEXT:    mv s5, t1
+; RV32IMZBS-NEXT:    sw t1, 120(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul t1, t4, s9
+; RV32IMZBS-NEXT:    mul t2, s4, s3
+; RV32IMZBS-NEXT:    mul t3, s6, s8
+; RV32IMZBS-NEXT:    mul t5, s2, s8
+; RV32IMZBS-NEXT:    mul t6, t4, s3
+; RV32IMZBS-NEXT:    mul s0, s4, s9
+; RV32IMZBS-NEXT:    mul s1, s6, s5
+; RV32IMZBS-NEXT:    mul s2, s2, s3
+; RV32IMZBS-NEXT:    mul s3, t4, s5
+; RV32IMZBS-NEXT:    mul s5, s4, s8
+; RV32IMZBS-NEXT:    mul s8, s6, s9
+; RV32IMZBS-NEXT:    lw t4, 96(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    or a2, t4, a2
+; RV32IMZBS-NEXT:    sw a2, 36(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a0, a0, a1
+; RV32IMZBS-NEXT:    xor a3, a4, a3
+; RV32IMZBS-NEXT:    xor a1, a5, a6
+; RV32IMZBS-NEXT:    xor a2, t1, t0
+; RV32IMZBS-NEXT:    xor a4, t2, t3
+; RV32IMZBS-NEXT:    xor a1, a3, a1
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    xor a3, t6, t5
+; RV32IMZBS-NEXT:    xor s0, s0, s1
+; RV32IMZBS-NEXT:    xor a4, s3, s2
+; RV32IMZBS-NEXT:    xor a5, s5, s8
+; RV32IMZBS-NEXT:    xor a3, a3, s0
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    lw s7, 184(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, s7
+; RV32IMZBS-NEXT:    and a2, a2, s11
+; RV32IMZBS-NEXT:    and a3, a3, s10
+; RV32IMZBS-NEXT:    and a4, a4, a7
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    sw a1, 32(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a5, a3, a4
+; RV32IMZBS-NEXT:    srli a4, a0, 2
+; RV32IMZBS-NEXT:    lw a3, 152(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a6, a0, a3
+; RV32IMZBS-NEXT:    lw a2, 84(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a2, s7
+; RV32IMZBS-NEXT:    and t4, a2, s11
+; RV32IMZBS-NEXT:    and a0, a2, s10
+; RV32IMZBS-NEXT:    and a2, a2, a7
+; RV32IMZBS-NEXT:    lw t3, 148(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s4, t3, s11
+; RV32IMZBS-NEXT:    and s9, t3, s7
+; RV32IMZBS-NEXT:    mul t0, a1, s4
+; RV32IMZBS-NEXT:    mul t1, t4, s9
+; RV32IMZBS-NEXT:    and s3, t3, a7
+; RV32IMZBS-NEXT:    and t5, t3, s10
+; RV32IMZBS-NEXT:    mul t2, a0, s3
+; RV32IMZBS-NEXT:    mul t3, a2, t5
+; RV32IMZBS-NEXT:    mul t6, a1, s3
+; RV32IMZBS-NEXT:    mul s0, t4, s4
+; RV32IMZBS-NEXT:    mul s1, a0, t5
+; RV32IMZBS-NEXT:    mul s5, a2, s9
+; RV32IMZBS-NEXT:    mul s8, a1, s9
+; RV32IMZBS-NEXT:    sw s9, 20(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv a7, a1
+; RV32IMZBS-NEXT:    sw a1, 96(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul s10, t4, t5
+; RV32IMZBS-NEXT:    sw t5, 4(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw t4, 92(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a1, a0, s4
+; RV32IMZBS-NEXT:    sw s4, 16(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv s2, a0
+; RV32IMZBS-NEXT:    sw a0, 88(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv s6, s3
+; RV32IMZBS-NEXT:    sw s3, 24(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a0, a2, s3
+; RV32IMZBS-NEXT:    sw a2, 84(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a4, a3
+; RV32IMZBS-NEXT:    slli a6, a6, 2
+; RV32IMZBS-NEXT:    or s3, a4, a6
+; RV32IMZBS-NEXT:    sw s3, 48(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw a4, 32(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    or a4, a4, a5
+; RV32IMZBS-NEXT:    sw a4, 32(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a4, t1, t0
+; RV32IMZBS-NEXT:    xor a5, t2, t3
+; RV32IMZBS-NEXT:    xor a6, s0, t6
+; RV32IMZBS-NEXT:    xor t0, s1, s5
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    xor a5, a6, t0
+; RV32IMZBS-NEXT:    xor a6, s10, s8
+; RV32IMZBS-NEXT:    xor a0, a1, a0
+; RV32IMZBS-NEXT:    and a4, a4, s7
+; RV32IMZBS-NEXT:    and a5, a5, s11
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    xor a0, a6, a0
+; RV32IMZBS-NEXT:    mul a5, a7, t5
+; RV32IMZBS-NEXT:    mul a6, t4, s6
+; RV32IMZBS-NEXT:    srli t0, ra, 8
+; RV32IMZBS-NEXT:    lw a1, 176(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t1, ra, a1
+; RV32IMZBS-NEXT:    and t0, t0, a1
+; RV32IMZBS-NEXT:    slli t1, t1, 8
+; RV32IMZBS-NEXT:    mul t2, s2, s9
+; RV32IMZBS-NEXT:    mul t3, a2, s4
+; RV32IMZBS-NEXT:    srli t5, ra, 24
+; RV32IMZBS-NEXT:    slli t6, ra, 24
+; RV32IMZBS-NEXT:    or t0, t0, t5
+; RV32IMZBS-NEXT:    or t1, t6, t1
+; RV32IMZBS-NEXT:    xor a5, a6, a5
+; RV32IMZBS-NEXT:    or a6, t1, t0
+; RV32IMZBS-NEXT:    srli t0, a6, 4
+; RV32IMZBS-NEXT:    lw a1, 180(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a6, a6, a1
+; RV32IMZBS-NEXT:    and t0, t0, a1
+; RV32IMZBS-NEXT:    slli a6, a6, 4
+; RV32IMZBS-NEXT:    xor t1, t2, t3
+; RV32IMZBS-NEXT:    or a6, t0, a6
+; RV32IMZBS-NEXT:    srli t0, a6, 2
+; RV32IMZBS-NEXT:    and a6, a6, a3
+; RV32IMZBS-NEXT:    and t0, t0, a3
+; RV32IMZBS-NEXT:    mv s4, a3
+; RV32IMZBS-NEXT:    slli a6, a6, 2
+; RV32IMZBS-NEXT:    xor a5, a5, t1
+; RV32IMZBS-NEXT:    or a6, t0, a6
+; RV32IMZBS-NEXT:    lw a1, 168(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a0, a0, a1
+; RV32IMZBS-NEXT:    lw s6, 160(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a5, a5, s6
+; RV32IMZBS-NEXT:    srli t0, a6, 1
+; RV32IMZBS-NEXT:    lw a3, 172(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a6, a6, a3
+; RV32IMZBS-NEXT:    and t0, t0, a3
+; RV32IMZBS-NEXT:    slli a6, a6, 1
+; RV32IMZBS-NEXT:    or a0, a0, a5
+; RV32IMZBS-NEXT:    or a5, t0, a6
+; RV32IMZBS-NEXT:    or a3, a4, a0
+; RV32IMZBS-NEXT:    and a4, a5, s7
+; RV32IMZBS-NEXT:    mv ra, s7
+; RV32IMZBS-NEXT:    and a6, a5, s11
+; RV32IMZBS-NEXT:    and t6, a5, a1
+; RV32IMZBS-NEXT:    mv s11, a1
+; RV32IMZBS-NEXT:    and s5, a5, s6
+; RV32IMZBS-NEXT:    srli s9, s3, 1
+; RV32IMZBS-NEXT:    sw s9, 40(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw t4, 100(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a5, a4, t4
+; RV32IMZBS-NEXT:    lw s2, 104(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t0, a6, s2
+; RV32IMZBS-NEXT:    lw s7, 108(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t1, t6, s7
+; RV32IMZBS-NEXT:    lw a2, 112(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t2, s5, a2
+; RV32IMZBS-NEXT:    mul t3, a4, s7
+; RV32IMZBS-NEXT:    mul t5, a6, t4
+; RV32IMZBS-NEXT:    mul s0, t6, a2
+; RV32IMZBS-NEXT:    mul s1, s5, s2
+; RV32IMZBS-NEXT:    mul s8, a4, s2
+; RV32IMZBS-NEXT:    mul s10, a6, a2
+; RV32IMZBS-NEXT:    mul a7, t6, t4
+; RV32IMZBS-NEXT:    mul a1, s5, s7
+; RV32IMZBS-NEXT:    lw a0, 36(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    srli a0, a0, 1
+; RV32IMZBS-NEXT:    slli s9, s9, 31
+; RV32IMZBS-NEXT:    or a0, a0, s9
+; RV32IMZBS-NEXT:    sw a0, 12(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw a0, 32(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    xor a0, a3, a0
+; RV32IMZBS-NEXT:    sw a0, 8(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a0, t0, a5
+; RV32IMZBS-NEXT:    xor a3, t1, t2
+; RV32IMZBS-NEXT:    xor a5, t5, t3
+; RV32IMZBS-NEXT:    xor s0, s0, s1
+; RV32IMZBS-NEXT:    xor a3, a0, a3
+; RV32IMZBS-NEXT:    xor a5, a5, s0
+; RV32IMZBS-NEXT:    xor a0, s10, s8
+; RV32IMZBS-NEXT:    xor a1, a7, a1
+; RV32IMZBS-NEXT:    mul a7, a4, a2
+; RV32IMZBS-NEXT:    mul a4, a6, s7
+; RV32IMZBS-NEXT:    lw t2, 148(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    srli a6, t2, 8
+; RV32IMZBS-NEXT:    lw s8, 176(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t0, t2, s8
+; RV32IMZBS-NEXT:    and a6, a6, s8
+; RV32IMZBS-NEXT:    slli t0, t0, 8
+; RV32IMZBS-NEXT:    srli t1, t2, 24
+; RV32IMZBS-NEXT:    slli t2, t2, 24
+; RV32IMZBS-NEXT:    or a6, a6, t1
+; RV32IMZBS-NEXT:    or t0, t2, t0
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    or a1, t0, a6
+; RV32IMZBS-NEXT:    mul a6, t6, s2
+; RV32IMZBS-NEXT:    mul t0, s5, t4
+; RV32IMZBS-NEXT:    srli t1, a1, 4
+; RV32IMZBS-NEXT:    lw s9, 180(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, s9
+; RV32IMZBS-NEXT:    and t1, t1, s9
+; RV32IMZBS-NEXT:    slli a1, a1, 4
+; RV32IMZBS-NEXT:    xor a2, a4, a7
+; RV32IMZBS-NEXT:    or a1, t1, a1
+; RV32IMZBS-NEXT:    srli a4, a1, 2
+; RV32IMZBS-NEXT:    mv t4, s4
+; RV32IMZBS-NEXT:    and a1, a1, s4
+; RV32IMZBS-NEXT:    and a4, a4, s4
+; RV32IMZBS-NEXT:    slli a1, a1, 2
+; RV32IMZBS-NEXT:    xor a6, a6, t0
+; RV32IMZBS-NEXT:    or a1, a4, a1
+; RV32IMZBS-NEXT:    srli a4, a1, 1
+; RV32IMZBS-NEXT:    lw a7, 172(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, a7
+; RV32IMZBS-NEXT:    and a4, a4, a7
+; RV32IMZBS-NEXT:    slli a1, a1, 1
+; RV32IMZBS-NEXT:    xor a2, a2, a6
+; RV32IMZBS-NEXT:    or a1, a4, a1
+; RV32IMZBS-NEXT:    lw s2, 164(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a1, s2
+; RV32IMZBS-NEXT:    and a6, a1, ra
+; RV32IMZBS-NEXT:    mv s0, s6
+; RV32IMZBS-NEXT:    and t0, a1, s6
+; RV32IMZBS-NEXT:    mv s7, s11
+; RV32IMZBS-NEXT:    and a1, a1, s11
+; RV32IMZBS-NEXT:    lw s5, 140(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t1, s5, a4
+; RV32IMZBS-NEXT:    lw s1, 136(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t2, s1, a6
+; RV32IMZBS-NEXT:    lw s4, 132(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t3, s4, t0
+; RV32IMZBS-NEXT:    lw s6, 128(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t5, s6, a1
+; RV32IMZBS-NEXT:    and a3, a3, ra
+; RV32IMZBS-NEXT:    and a5, a5, s2
+; RV32IMZBS-NEXT:    mv s11, s2
+; RV32IMZBS-NEXT:    and a0, a0, s7
+; RV32IMZBS-NEXT:    and a2, a2, s0
+; RV32IMZBS-NEXT:    mv s10, s0
+; RV32IMZBS-NEXT:    or a3, a5, a3
+; RV32IMZBS-NEXT:    or a0, a0, a2
+; RV32IMZBS-NEXT:    mul a2, s5, t0
+; RV32IMZBS-NEXT:    mul a5, s1, a4
+; RV32IMZBS-NEXT:    mul t6, s4, a1
+; RV32IMZBS-NEXT:    mul s0, s6, a6
+; RV32IMZBS-NEXT:    xor t1, t2, t1
+; RV32IMZBS-NEXT:    xor t2, t3, t5
+; RV32IMZBS-NEXT:    mul t3, s1, a1
+; RV32IMZBS-NEXT:    mul a1, s5, a1
+; RV32IMZBS-NEXT:    mul t5, s6, t0
+; RV32IMZBS-NEXT:    mul t0, s1, t0
+; RV32IMZBS-NEXT:    mul s1, s5, a6
+; RV32IMZBS-NEXT:    mul s5, s4, a4
+; RV32IMZBS-NEXT:    mul a6, s4, a6
+; RV32IMZBS-NEXT:    mul a4, s6, a4
+; RV32IMZBS-NEXT:    xor a2, a5, a2
+; RV32IMZBS-NEXT:    xor a5, t6, s0
+; RV32IMZBS-NEXT:    xor t1, t1, t2
+; RV32IMZBS-NEXT:    xor a2, a2, a5
+; RV32IMZBS-NEXT:    and a5, t1, ra
+; RV32IMZBS-NEXT:    and a2, a2, s2
+; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    or a2, a2, a5
+; RV32IMZBS-NEXT:    xor a3, t3, s1
+; RV32IMZBS-NEXT:    xor a5, s5, t5
+; RV32IMZBS-NEXT:    xor a1, t0, a1
+; RV32IMZBS-NEXT:    xor a4, a6, a4
+; RV32IMZBS-NEXT:    xor a3, a3, a5
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    and a3, a3, s7
+; RV32IMZBS-NEXT:    and a1, a1, s10
+; RV32IMZBS-NEXT:    or a1, a3, a1
+; RV32IMZBS-NEXT:    srli a3, a0, 8
+; RV32IMZBS-NEXT:    and a3, a3, s8
+; RV32IMZBS-NEXT:    srli a4, a0, 24
+; RV32IMZBS-NEXT:    or a3, a3, a4
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    slli a2, a0, 24
+; RV32IMZBS-NEXT:    and a0, a0, s8
+; RV32IMZBS-NEXT:    slli a0, a0, 8
+; RV32IMZBS-NEXT:    srli a4, a1, 8
+; RV32IMZBS-NEXT:    or a0, a2, a0
+; RV32IMZBS-NEXT:    and a2, a4, s8
+; RV32IMZBS-NEXT:    srli a4, a1, 24
+; RV32IMZBS-NEXT:    and a5, a1, s8
+; RV32IMZBS-NEXT:    slli a1, a1, 24
+; RV32IMZBS-NEXT:    slli a5, a5, 8
+; RV32IMZBS-NEXT:    or a2, a2, a4
+; RV32IMZBS-NEXT:    or a1, a1, a5
+; RV32IMZBS-NEXT:    or a0, a0, a3
+; RV32IMZBS-NEXT:    or a1, a1, a2
+; RV32IMZBS-NEXT:    srli a2, a0, 4
+; RV32IMZBS-NEXT:    and a0, a0, s9
+; RV32IMZBS-NEXT:    and a2, a2, s9
+; RV32IMZBS-NEXT:    slli a0, a0, 4
+; RV32IMZBS-NEXT:    srli a3, a1, 4
+; RV32IMZBS-NEXT:    and a1, a1, s9
+; RV32IMZBS-NEXT:    and a3, a3, s9
+; RV32IMZBS-NEXT:    slli a1, a1, 4
+; RV32IMZBS-NEXT:    or a0, a2, a0
+; RV32IMZBS-NEXT:    or a1, a3, a1
+; RV32IMZBS-NEXT:    srli a2, a0, 2
+; RV32IMZBS-NEXT:    and a0, a0, t4
+; RV32IMZBS-NEXT:    and a2, a2, t4
+; RV32IMZBS-NEXT:    slli a0, a0, 2
+; RV32IMZBS-NEXT:    srli a3, a1, 2
+; RV32IMZBS-NEXT:    and a1, a1, t4
+; RV32IMZBS-NEXT:    and a3, a3, t4
+; RV32IMZBS-NEXT:    slli a1, a1, 2
+; RV32IMZBS-NEXT:    or a0, a2, a0
+; RV32IMZBS-NEXT:    or a1, a3, a1
+; RV32IMZBS-NEXT:    srli a2, a0, 1
+; RV32IMZBS-NEXT:    and a0, a0, a7
+; RV32IMZBS-NEXT:    lw a4, 156(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a2, a2, a4
+; RV32IMZBS-NEXT:    slli a0, a0, 1
+; RV32IMZBS-NEXT:    srli a3, a1, 1
+; RV32IMZBS-NEXT:    and a1, a1, a7
+; RV32IMZBS-NEXT:    and a3, a3, a4
+; RV32IMZBS-NEXT:    slli a1, a1, 1
+; RV32IMZBS-NEXT:    or a0, a2, a0
+; RV32IMZBS-NEXT:    or a1, a3, a1
+; RV32IMZBS-NEXT:    lw a2, 64(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a2, ra
+; RV32IMZBS-NEXT:    and s2, a2, s2
+; RV32IMZBS-NEXT:    and t5, a2, s7
+; RV32IMZBS-NEXT:    and s3, a2, s10
+; RV32IMZBS-NEXT:    lw a6, 4(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a2, s3, a6
+; RV32IMZBS-NEXT:    sw s3, 64(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a3, t5, a6
+; RV32IMZBS-NEXT:    sw s2, 36(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a5, s2, a6
+; RV32IMZBS-NEXT:    sw a4, 32(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a6, a4, a6
+; RV32IMZBS-NEXT:    lw t6, 16(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a7, a4, t6
+; RV32IMZBS-NEXT:    lw s0, 20(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t0, s2, s0
+; RV32IMZBS-NEXT:    lw t4, 24(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t1, t5, t4
+; RV32IMZBS-NEXT:    mul t2, a4, t4
+; RV32IMZBS-NEXT:    mul t3, s2, t6
+; RV32IMZBS-NEXT:    mul s1, s3, s0
+; RV32IMZBS-NEXT:    mul s5, s3, t4
+; RV32IMZBS-NEXT:    mul t4, s2, t4
+; RV32IMZBS-NEXT:    mul s8, a4, s0
+; RV32IMZBS-NEXT:    sw t5, 28(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul s9, t5, t6
+; RV32IMZBS-NEXT:    mul s2, t5, s0
+; RV32IMZBS-NEXT:    mul s3, s3, t6
+; RV32IMZBS-NEXT:    lw a4, 12(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw t5, 8(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    xor a4, a4, t5
+; RV32IMZBS-NEXT:    sw a4, 148(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a0, a1, a0
+; RV32IMZBS-NEXT:    sw a0, 24(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a0, t0, a7
+; RV32IMZBS-NEXT:    xor a1, t1, a2
+; RV32IMZBS-NEXT:    xor a2, t3, t2
+; RV32IMZBS-NEXT:    xor a3, a3, s1
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    xor a2, a2, a3
+; RV32IMZBS-NEXT:    xor a1, a5, s8
+; RV32IMZBS-NEXT:    xor a3, s9, s5
+; RV32IMZBS-NEXT:    xor a5, t4, a6
+; RV32IMZBS-NEXT:    xor a6, s2, s3
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    xor a3, a5, a6
+; RV32IMZBS-NEXT:    mv s6, ra
+; RV32IMZBS-NEXT:    and a4, a0, ra
+; RV32IMZBS-NEXT:    mv s4, s11
+; RV32IMZBS-NEXT:    and a2, a2, s11
+; RV32IMZBS-NEXT:    and a1, a1, s7
+; RV32IMZBS-NEXT:    mv s0, s10
+; RV32IMZBS-NEXT:    and a3, a3, s10
+; RV32IMZBS-NEXT:    lw t1, 68(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a5, t1, s11
+; RV32IMZBS-NEXT:    and a6, t1, ra
+; RV32IMZBS-NEXT:    and a7, t1, s10
+; RV32IMZBS-NEXT:    and t0, t1, s7
+; RV32IMZBS-NEXT:    lw t6, 84(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t1, t6, t0
+; RV32IMZBS-NEXT:    lw t5, 88(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t2, t5, t0
+; RV32IMZBS-NEXT:    lw a0, 92(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t3, a0, t0
+; RV32IMZBS-NEXT:    lw s10, 96(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t0, s10, t0
+; RV32IMZBS-NEXT:    mul t4, s10, a5
+; RV32IMZBS-NEXT:    mul s1, a0, a6
+; RV32IMZBS-NEXT:    mul s2, t5, a7
+; RV32IMZBS-NEXT:    mul s3, s10, a7
+; RV32IMZBS-NEXT:    mul s5, a0, a5
+; RV32IMZBS-NEXT:    mul s8, t6, a6
+; RV32IMZBS-NEXT:    mul s9, t6, a7
+; RV32IMZBS-NEXT:    mul a7, a0, a7
+; RV32IMZBS-NEXT:    mul s10, s10, a6
+; RV32IMZBS-NEXT:    mul a0, t5, a5
+; RV32IMZBS-NEXT:    mul a6, t5, a6
+; RV32IMZBS-NEXT:    mul a5, t6, a5
+; RV32IMZBS-NEXT:    or t5, a2, a4
+; RV32IMZBS-NEXT:    or a2, a1, a3
+; RV32IMZBS-NEXT:    xor a3, s1, t4
+; RV32IMZBS-NEXT:    xor a4, s2, t1
+; RV32IMZBS-NEXT:    xor t1, s5, s3
+; RV32IMZBS-NEXT:    xor t2, t2, s8
+; RV32IMZBS-NEXT:    xor a3, a3, a4
+; RV32IMZBS-NEXT:    xor a4, t1, t2
+; RV32IMZBS-NEXT:    xor t1, t3, s10
+; RV32IMZBS-NEXT:    xor a0, a0, s9
+; RV32IMZBS-NEXT:    xor a7, a7, t0
+; RV32IMZBS-NEXT:    xor a5, a6, a5
+; RV32IMZBS-NEXT:    xor a0, t1, a0
+; RV32IMZBS-NEXT:    xor a5, a7, a5
+; RV32IMZBS-NEXT:    and a3, a3, ra
+; RV32IMZBS-NEXT:    and a4, a4, s11
+; RV32IMZBS-NEXT:    and a0, a0, s7
+; RV32IMZBS-NEXT:    and a5, a5, s0
+; RV32IMZBS-NEXT:    or a1, a4, a3
+; RV32IMZBS-NEXT:    or a3, a0, a5
+; RV32IMZBS-NEXT:    lw a0, 72(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a0, ra
+; RV32IMZBS-NEXT:    and a5, a0, s11
+; RV32IMZBS-NEXT:    and a6, a0, s7
+; RV32IMZBS-NEXT:    and a7, a0, s0
+; RV32IMZBS-NEXT:    lw t6, 116(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t0, a4, t6
+; RV32IMZBS-NEXT:    lw a0, 120(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t1, a4, a0
+; RV32IMZBS-NEXT:    lw s11, 124(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t2, a4, s11
+; RV32IMZBS-NEXT:    lw s5, 144(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a4, a4, s5
+; RV32IMZBS-NEXT:    mul t3, a5, s11
+; RV32IMZBS-NEXT:    mul t4, a6, a0
+; RV32IMZBS-NEXT:    mul s1, a7, s5
+; RV32IMZBS-NEXT:    mul s2, a5, t6
+; RV32IMZBS-NEXT:    mul s3, a6, s5
+; RV32IMZBS-NEXT:    mul s9, a7, s11
+; RV32IMZBS-NEXT:    mul s10, a5, s5
+; RV32IMZBS-NEXT:    mul a5, a5, a0
+; RV32IMZBS-NEXT:    mul s5, a6, t6
+; RV32IMZBS-NEXT:    mul s8, a7, a0
+; RV32IMZBS-NEXT:    mul a0, a6, s11
+; RV32IMZBS-NEXT:    mul a7, a7, t6
+; RV32IMZBS-NEXT:    or ra, t5, a2
+; RV32IMZBS-NEXT:    or a1, a1, a3
+; RV32IMZBS-NEXT:    sw a1, 68(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a1, t3, t0
+; RV32IMZBS-NEXT:    xor a2, t4, s1
+; RV32IMZBS-NEXT:    xor t0, s2, t1
+; RV32IMZBS-NEXT:    xor t1, s3, s9
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a2, t0, t1
+; RV32IMZBS-NEXT:    xor t0, s10, t2
+; RV32IMZBS-NEXT:    xor t1, s5, s8
+; RV32IMZBS-NEXT:    xor a4, a5, a4
+; RV32IMZBS-NEXT:    xor a0, a0, a7
+; RV32IMZBS-NEXT:    xor a5, t0, t1
+; RV32IMZBS-NEXT:    xor a0, a4, a0
+; RV32IMZBS-NEXT:    and a1, a1, s6
+; RV32IMZBS-NEXT:    and a2, a2, s4
+; RV32IMZBS-NEXT:    mv a3, s7
+; RV32IMZBS-NEXT:    and a4, a5, s7
+; RV32IMZBS-NEXT:    and a0, a0, s0
+; RV32IMZBS-NEXT:    or t6, a2, a1
+; RV32IMZBS-NEXT:    or a6, a4, a0
+; RV32IMZBS-NEXT:    lw a0, 76(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a0, s4
+; RV32IMZBS-NEXT:    mv s11, s4
+; RV32IMZBS-NEXT:    and s7, a0, s6
+; RV32IMZBS-NEXT:    and a7, a0, s0
+; RV32IMZBS-NEXT:    and a5, a0, a3
+; RV32IMZBS-NEXT:    mv s6, a3
+; RV32IMZBS-NEXT:    lw a0, 44(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    sw a4, 72(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul t0, a0, a4
+; RV32IMZBS-NEXT:    mul t1, a0, a7
+; RV32IMZBS-NEXT:    mul t2, a0, s7
+; RV32IMZBS-NEXT:    mul t3, a0, a5
+; RV32IMZBS-NEXT:    lw a2, 60(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t4, a2, s7
+; RV32IMZBS-NEXT:    lw a0, 56(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t5, a0, a7
+; RV32IMZBS-NEXT:    lw a3, 52(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul s1, a3, a5
+; RV32IMZBS-NEXT:    mul s2, a2, a4
+; RV32IMZBS-NEXT:    mul s3, a0, a5
+; RV32IMZBS-NEXT:    mul s9, a3, s7
+; RV32IMZBS-NEXT:    mul s10, a2, a5
+; RV32IMZBS-NEXT:    mul s8, a2, a7
+; RV32IMZBS-NEXT:    mul s5, a0, a4
+; RV32IMZBS-NEXT:    mul a2, a3, a7
+; RV32IMZBS-NEXT:    mul a1, a0, s7
+; RV32IMZBS-NEXT:    mul a0, a3, a4
+; RV32IMZBS-NEXT:    lw a3, 68(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    xor a3, a3, ra
+; RV32IMZBS-NEXT:    sw a3, 76(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a3, t6, a6
+; RV32IMZBS-NEXT:    sw a3, 68(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a6, t4, t0
+; RV32IMZBS-NEXT:    xor t0, t5, s1
+; RV32IMZBS-NEXT:    xor t1, s2, t1
+; RV32IMZBS-NEXT:    xor t4, s3, s9
+; RV32IMZBS-NEXT:    xor t6, a6, t0
+; RV32IMZBS-NEXT:    xor t1, t1, t4
+; RV32IMZBS-NEXT:    xor t0, s10, t2
+; RV32IMZBS-NEXT:    xor a2, s5, a2
+; RV32IMZBS-NEXT:    xor t2, s8, t3
+; RV32IMZBS-NEXT:    xor a0, a1, a0
+; RV32IMZBS-NEXT:    xor a1, t0, a2
+; RV32IMZBS-NEXT:    xor a6, t2, a0
+; RV32IMZBS-NEXT:    lw a0, 140(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s9, 100(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul ra, a0, s9
+; RV32IMZBS-NEXT:    lw s10, 108(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t0, a0, s10
+; RV32IMZBS-NEXT:    lw a3, 104(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t2, a0, a3
+; RV32IMZBS-NEXT:    lw s0, 112(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t3, a0, s0
+; RV32IMZBS-NEXT:    lw s3, 128(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t4, s3, s0
+; RV32IMZBS-NEXT:    lw s5, 132(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t5, s5, s0
+; RV32IMZBS-NEXT:    lw a0, 136(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul s0, a0, s0
+; RV32IMZBS-NEXT:    mul s1, a0, a3
+; RV32IMZBS-NEXT:    mul s2, s3, a3
+; RV32IMZBS-NEXT:    mul a4, s5, a3
+; RV32IMZBS-NEXT:    mul s4, s5, s10
+; RV32IMZBS-NEXT:    mul s5, s5, s9
+; RV32IMZBS-NEXT:    mul s8, a0, s9
+; RV32IMZBS-NEXT:    mul a3, a0, s10
+; RV32IMZBS-NEXT:    mul s10, s3, s10
+; RV32IMZBS-NEXT:    mul a0, s3, s9
+; RV32IMZBS-NEXT:    lw a2, 184(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t6, t6, a2
+; RV32IMZBS-NEXT:    and t1, t1, s11
+; RV32IMZBS-NEXT:    mv s3, s6
+; RV32IMZBS-NEXT:    and a1, a1, s6
+; RV32IMZBS-NEXT:    lw s6, 160(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a6, a6, s6
+; RV32IMZBS-NEXT:    or t1, t1, t6
+; RV32IMZBS-NEXT:    or a1, a1, a6
+; RV32IMZBS-NEXT:    xor t6, s1, ra
+; RV32IMZBS-NEXT:    xor a6, s4, t4
+; RV32IMZBS-NEXT:    xor t0, s8, t0
+; RV32IMZBS-NEXT:    xor t4, t5, s2
+; RV32IMZBS-NEXT:    xor t5, t6, a6
+; RV32IMZBS-NEXT:    xor a6, t0, t4
+; RV32IMZBS-NEXT:    xor t0, s0, t2
+; RV32IMZBS-NEXT:    xor t2, s5, s10
+; RV32IMZBS-NEXT:    xor t3, a3, t3
+; RV32IMZBS-NEXT:    xor a0, a4, a0
+; RV32IMZBS-NEXT:    xor t0, t0, t2
+; RV32IMZBS-NEXT:    xor a0, t3, a0
+; RV32IMZBS-NEXT:    and a3, t5, a2
+; RV32IMZBS-NEXT:    and a6, a6, s11
+; RV32IMZBS-NEXT:    and t0, t0, s3
+; RV32IMZBS-NEXT:    and a0, a0, s6
+; RV32IMZBS-NEXT:    mv s10, s6
+; RV32IMZBS-NEXT:    or a2, a6, a3
+; RV32IMZBS-NEXT:    or a0, t0, a0
+; RV32IMZBS-NEXT:    or a1, t1, a1
+; RV32IMZBS-NEXT:    or a0, a2, a0
+; RV32IMZBS-NEXT:    lw a2, 68(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    srli a2, a0, 8
+; RV32IMZBS-NEXT:    lw a3, 76(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    xor a1, a3, a1
+; RV32IMZBS-NEXT:    lw a3, 176(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a2, a2, a3
+; RV32IMZBS-NEXT:    and a6, a0, a3
+; RV32IMZBS-NEXT:    srli t0, a0, 24
+; RV32IMZBS-NEXT:    slli a0, a0, 24
+; RV32IMZBS-NEXT:    slli a6, a6, 8
+; RV32IMZBS-NEXT:    or a2, a2, t0
+; RV32IMZBS-NEXT:    or a0, a0, a6
+; RV32IMZBS-NEXT:    or a0, a0, a2
+; RV32IMZBS-NEXT:    lw a2, 48(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw t0, 172(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a2, a2, t0
+; RV32IMZBS-NEXT:    lw a3, 40(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a4, 156(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a6, a3, a4
+; RV32IMZBS-NEXT:    slli a2, a2, 1
+; RV32IMZBS-NEXT:    or a2, a6, a2
+; RV32IMZBS-NEXT:    srli a6, a0, 4
+; RV32IMZBS-NEXT:    lw a3, 180(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a6, a6, a3
+; RV32IMZBS-NEXT:    and a0, a0, a3
+; RV32IMZBS-NEXT:    srli a2, a2, 1
+; RV32IMZBS-NEXT:    slli a0, a0, 4
+; RV32IMZBS-NEXT:    xor a1, a2, a1
+; RV32IMZBS-NEXT:    or a0, a6, a0
+; RV32IMZBS-NEXT:    srli a2, a0, 2
+; RV32IMZBS-NEXT:    lw a3, 152(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a0, a0, a3
+; RV32IMZBS-NEXT:    and a2, a2, a3
+; RV32IMZBS-NEXT:    slli a0, a0, 2
+; RV32IMZBS-NEXT:    lw a3, 24(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    srli a6, a3, 1
+; RV32IMZBS-NEXT:    or a0, a2, a0
+; RV32IMZBS-NEXT:    srli a2, a0, 1
+; RV32IMZBS-NEXT:    and a0, a0, t0
+; RV32IMZBS-NEXT:    and a2, a2, a4
+; RV32IMZBS-NEXT:    slli t0, a0, 1
+; RV32IMZBS-NEXT:    xor a0, a1, a6
+; RV32IMZBS-NEXT:    sw a0, 180(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a0, a2, t0
+; RV32IMZBS-NEXT:    sw a0, 176(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw s8, 116(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a4, 32(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a1, a4, s8
+; RV32IMZBS-NEXT:    lw a0, 36(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw t1, 124(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a2, a0, t1
+; RV32IMZBS-NEXT:    lw s9, 120(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s5, 28(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a6, s5, s9
+; RV32IMZBS-NEXT:    lw a3, 64(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw ra, 144(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t0, a3, ra
+; RV32IMZBS-NEXT:    mul t2, a4, s9
+; RV32IMZBS-NEXT:    mul t3, a0, s8
+; RV32IMZBS-NEXT:    mul t4, s5, ra
+; RV32IMZBS-NEXT:    mul t5, a3, t1
+; RV32IMZBS-NEXT:    mul t6, a4, t1
+; RV32IMZBS-NEXT:    mul s0, a0, ra
+; RV32IMZBS-NEXT:    mul s1, s5, s8
+; RV32IMZBS-NEXT:    mul s2, a3, s9
+; RV32IMZBS-NEXT:    mul s3, a4, ra
+; RV32IMZBS-NEXT:    mul s4, a0, s9
+; RV32IMZBS-NEXT:    mul s5, s5, t1
+; RV32IMZBS-NEXT:    mul s6, a3, s8
+; RV32IMZBS-NEXT:    xor a1, a2, a1
+; RV32IMZBS-NEXT:    xor a2, a6, t0
+; RV32IMZBS-NEXT:    xor a6, t3, t2
+; RV32IMZBS-NEXT:    xor t0, t4, t5
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a2, a6, t0
+; RV32IMZBS-NEXT:    xor a6, s0, t6
+; RV32IMZBS-NEXT:    xor t0, s1, s2
+; RV32IMZBS-NEXT:    lw s11, 184(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, s11
+; RV32IMZBS-NEXT:    lw t4, 164(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a2, a2, t4
+; RV32IMZBS-NEXT:    xor t2, s4, s3
+; RV32IMZBS-NEXT:    xor t3, s5, s6
+; RV32IMZBS-NEXT:    xor a6, a6, t0
+; RV32IMZBS-NEXT:    xor t0, t2, t3
+; RV32IMZBS-NEXT:    lw a0, 168(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a6, a6, a0
+; RV32IMZBS-NEXT:    and t0, t0, s10
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    sw a1, 172(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a0, a6, t0
+; RV32IMZBS-NEXT:    sw a0, 156(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw a0, 96(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a3, 72(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a6, a0, a3
+; RV32IMZBS-NEXT:    lw s6, 92(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t0, s6, s7
+; RV32IMZBS-NEXT:    lw s10, 88(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t2, s10, a7
+; RV32IMZBS-NEXT:    lw s5, 84(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t3, s5, a5
+; RV32IMZBS-NEXT:    mul a2, a0, a7
+; RV32IMZBS-NEXT:    mul t5, s6, a3
+; RV32IMZBS-NEXT:    mul t6, s10, a5
+; RV32IMZBS-NEXT:    mul s0, s5, s7
+; RV32IMZBS-NEXT:    mul s1, a0, s7
+; RV32IMZBS-NEXT:    mul s2, s6, a5
+; RV32IMZBS-NEXT:    mul s3, s10, a3
+; RV32IMZBS-NEXT:    mul s4, s5, a7
+; RV32IMZBS-NEXT:    mul a1, a0, a5
+; RV32IMZBS-NEXT:    mul a7, s6, a7
+; RV32IMZBS-NEXT:    mul a4, s10, s7
+; RV32IMZBS-NEXT:    mv s7, s10
+; RV32IMZBS-NEXT:    mul a5, s5, a3
+; RV32IMZBS-NEXT:    xor a6, t0, a6
+; RV32IMZBS-NEXT:    xor t0, t2, t3
+; RV32IMZBS-NEXT:    xor t2, t5, a2
+; RV32IMZBS-NEXT:    xor t3, t6, s0
+; RV32IMZBS-NEXT:    xor a6, a6, t0
+; RV32IMZBS-NEXT:    xor t0, t2, t3
+; RV32IMZBS-NEXT:    xor t2, s2, s1
+; RV32IMZBS-NEXT:    xor t3, s3, s4
+; RV32IMZBS-NEXT:    and a6, a6, s11
+; RV32IMZBS-NEXT:    mv s10, t4
+; RV32IMZBS-NEXT:    and t0, t0, t4
+; RV32IMZBS-NEXT:    xor a3, a7, a1
+; RV32IMZBS-NEXT:    xor a4, a4, a5
+; RV32IMZBS-NEXT:    xor a5, t2, t3
+; RV32IMZBS-NEXT:    xor a3, a3, a4
+; RV32IMZBS-NEXT:    mul a4, a0, s8
+; RV32IMZBS-NEXT:    mul a7, s6, t1
+; RV32IMZBS-NEXT:    mul t2, s7, s9
+; RV32IMZBS-NEXT:    mul t3, s5, ra
+; RV32IMZBS-NEXT:    lw s4, 168(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a5, a5, s4
+; RV32IMZBS-NEXT:    lw s3, 160(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a3, s3
+; RV32IMZBS-NEXT:    or a6, t0, a6
+; RV32IMZBS-NEXT:    or a3, a5, a3
+; RV32IMZBS-NEXT:    lw a1, 172(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw a2, 156(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    or a1, a1, a2
+; RV32IMZBS-NEXT:    or a2, a6, a3
+; RV32IMZBS-NEXT:    mul a3, a0, s9
+; RV32IMZBS-NEXT:    mul a5, s6, s8
+; RV32IMZBS-NEXT:    mul a6, s7, ra
+; RV32IMZBS-NEXT:    mul t0, s5, t1
+; RV32IMZBS-NEXT:    xor a4, a7, a4
+; RV32IMZBS-NEXT:    xor a7, t2, t3
+; RV32IMZBS-NEXT:    mul t2, a0, t1
+; RV32IMZBS-NEXT:    mul t3, s6, ra
+; RV32IMZBS-NEXT:    mul t4, a0, ra
+; RV32IMZBS-NEXT:    mul t5, s7, s8
+; RV32IMZBS-NEXT:    mul t6, s6, s9
+; RV32IMZBS-NEXT:    mul s0, s5, s9
+; RV32IMZBS-NEXT:    mul s1, s7, t1
+; RV32IMZBS-NEXT:    mul s2, s5, s8
+; RV32IMZBS-NEXT:    xor a3, a5, a3
+; RV32IMZBS-NEXT:    xor a5, a6, t0
+; RV32IMZBS-NEXT:    xor a4, a4, a7
+; RV32IMZBS-NEXT:    xor a3, a3, a5
+; RV32IMZBS-NEXT:    and a4, a4, s11
+; RV32IMZBS-NEXT:    and a3, a3, s10
+; RV32IMZBS-NEXT:    xor a1, a2, a1
+; RV32IMZBS-NEXT:    or a3, a3, a4
+; RV32IMZBS-NEXT:    xor a2, t3, t2
+; RV32IMZBS-NEXT:    xor a4, t5, s0
+; RV32IMZBS-NEXT:    xor a5, t6, t4
+; RV32IMZBS-NEXT:    xor a6, s1, s2
+; RV32IMZBS-NEXT:    xor a2, a2, a4
+; RV32IMZBS-NEXT:    xor a4, a5, a6
+; RV32IMZBS-NEXT:    and a2, a2, s4
+; RV32IMZBS-NEXT:    and a4, a4, s3
+; RV32IMZBS-NEXT:    or a2, a2, a4
+; RV32IMZBS-NEXT:    lw a4, 176(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    srli a4, a4, 1
+; RV32IMZBS-NEXT:    xor a1, a4, a1
+; RV32IMZBS-NEXT:    or a2, a3, a2
+; RV32IMZBS-NEXT:    lw a3, 80(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    sw a2, 0(a3)
+; RV32IMZBS-NEXT:    sw a1, 4(a3)
+; RV32IMZBS-NEXT:    lw a1, 148(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    sw a1, 8(a3)
+; RV32IMZBS-NEXT:    lw a0, 180(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    sw a0, 12(a3)
+; RV32IMZBS-NEXT:    lw ra, 236(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s0, 232(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s1, 228(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s2, 224(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s3, 220(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s4, 216(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s5, 212(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s6, 208(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s7, 204(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s8, 200(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s9, 196(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s10, 192(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s11, 188(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    .cfi_restore ra
+; RV32IMZBS-NEXT:    .cfi_restore s0
+; RV32IMZBS-NEXT:    .cfi_restore s1
+; RV32IMZBS-NEXT:    .cfi_restore s2
+; RV32IMZBS-NEXT:    .cfi_restore s3
+; RV32IMZBS-NEXT:    .cfi_restore s4
+; RV32IMZBS-NEXT:    .cfi_restore s5
+; RV32IMZBS-NEXT:    .cfi_restore s6
+; RV32IMZBS-NEXT:    .cfi_restore s7
+; RV32IMZBS-NEXT:    .cfi_restore s8
+; RV32IMZBS-NEXT:    .cfi_restore s9
+; RV32IMZBS-NEXT:    .cfi_restore s10
+; RV32IMZBS-NEXT:    .cfi_restore s11
+; RV32IMZBS-NEXT:    addi sp, sp, 240
+; RV32IMZBS-NEXT:    .cfi_def_cfa_offset 0
+; RV32IMZBS-NEXT:    ret
+;
+; RV64IMZBS-LABEL: clmul_i128:
+; RV64IMZBS:       # %bb.0:
+; RV64IMZBS-NEXT:    addi sp, sp, -128
+; RV64IMZBS-NEXT:    .cfi_def_cfa_offset 128
+; RV64IMZBS-NEXT:    sd ra, 120(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s0, 112(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s1, 104(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s2, 96(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s3, 88(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s4, 80(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s5, 72(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s6, 64(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s7, 56(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s8, 48(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s9, 40(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s10, 32(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s11, 24(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    .cfi_offset ra, -8
+; RV64IMZBS-NEXT:    .cfi_offset s0, -16
+; RV64IMZBS-NEXT:    .cfi_offset s1, -24
+; RV64IMZBS-NEXT:    .cfi_offset s2, -32
+; RV64IMZBS-NEXT:    .cfi_offset s3, -40
+; RV64IMZBS-NEXT:    .cfi_offset s4, -48
+; RV64IMZBS-NEXT:    .cfi_offset s5, -56
+; RV64IMZBS-NEXT:    .cfi_offset s6, -64
+; RV64IMZBS-NEXT:    .cfi_offset s7, -72
+; RV64IMZBS-NEXT:    .cfi_offset s8, -80
+; RV64IMZBS-NEXT:    .cfi_offset s9, -88
+; RV64IMZBS-NEXT:    .cfi_offset s10, -96
+; RV64IMZBS-NEXT:    .cfi_offset s11, -104
+; RV64IMZBS-NEXT:    sd a3, 16(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd a1, 8(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    mv s3, a0
+; RV64IMZBS-NEXT:    srli a4, a2, 24
+; RV64IMZBS-NEXT:    lui a0, 4080
+; RV64IMZBS-NEXT:    and a4, a4, a0
+; RV64IMZBS-NEXT:    li t3, 255
+; RV64IMZBS-NEXT:    srli a5, a2, 8
+; RV64IMZBS-NEXT:    slli t3, t3, 24
+; RV64IMZBS-NEXT:    and a5, a5, t3
+; RV64IMZBS-NEXT:    lui a6, 16
+; RV64IMZBS-NEXT:    srli a7, a2, 40
+; RV64IMZBS-NEXT:    addi t1, a6, -256
+; RV64IMZBS-NEXT:    and a6, a7, t1
+; RV64IMZBS-NEXT:    srli a7, a2, 56
+; RV64IMZBS-NEXT:    or a4, a5, a4
+; RV64IMZBS-NEXT:    or a5, a6, a7
+; RV64IMZBS-NEXT:    or a4, a4, a5
+; RV64IMZBS-NEXT:    and a5, a2, a0
+; RV64IMZBS-NEXT:    srliw a6, a2, 24
+; RV64IMZBS-NEXT:    slli a5, a5, 24
+; RV64IMZBS-NEXT:    slli a6, a6, 32
+; RV64IMZBS-NEXT:    or a5, a5, a6
+; RV64IMZBS-NEXT:    and a6, a2, t1
+; RV64IMZBS-NEXT:    slli a6, a6, 40
+; RV64IMZBS-NEXT:    slli a7, a2, 56
+; RV64IMZBS-NEXT:    or a6, a7, a6
+; RV64IMZBS-NEXT:    lui a7, 61681
+; RV64IMZBS-NEXT:    or a5, a6, a5
+; RV64IMZBS-NEXT:    addi t4, a7, -241
+; RV64IMZBS-NEXT:    or a4, a5, a4
+; RV64IMZBS-NEXT:    slli a5, t4, 32
+; RV64IMZBS-NEXT:    srli a6, a4, 4
+; RV64IMZBS-NEXT:    add t4, t4, a5
+; RV64IMZBS-NEXT:    and a5, a6, t4
+; RV64IMZBS-NEXT:    and a4, a4, t4
+; RV64IMZBS-NEXT:    lui a6, 209715
+; RV64IMZBS-NEXT:    slli a4, a4, 4
+; RV64IMZBS-NEXT:    addi t5, a6, 819
+; RV64IMZBS-NEXT:    or a4, a5, a4
+; RV64IMZBS-NEXT:    slli a5, t5, 32
+; RV64IMZBS-NEXT:    srli a6, a4, 2
+; RV64IMZBS-NEXT:    add t5, t5, a5
+; RV64IMZBS-NEXT:    and a5, a6, t5
+; RV64IMZBS-NEXT:    lui a6, 349525
+; RV64IMZBS-NEXT:    and a4, a4, t5
+; RV64IMZBS-NEXT:    addi t2, a6, 1365
+; RV64IMZBS-NEXT:    slli a4, a4, 2
+; RV64IMZBS-NEXT:    slli a6, t2, 32
+; RV64IMZBS-NEXT:    or a4, a5, a4
+; RV64IMZBS-NEXT:    add a1, t2, a6
+; RV64IMZBS-NEXT:    srli a5, a4, 1
+; RV64IMZBS-NEXT:    and a4, a4, a1
+; RV64IMZBS-NEXT:    and a5, a5, a1
+; RV64IMZBS-NEXT:    slli a4, a4, 1
+; RV64IMZBS-NEXT:    or t6, a5, a4
+; RV64IMZBS-NEXT:    srli a4, s3, 24
+; RV64IMZBS-NEXT:    srli a5, s3, 8
+; RV64IMZBS-NEXT:    and a4, a4, a0
+; RV64IMZBS-NEXT:    and a5, a5, t3
+; RV64IMZBS-NEXT:    or a4, a5, a4
+; RV64IMZBS-NEXT:    srli a5, s3, 40
+; RV64IMZBS-NEXT:    and a5, a5, t1
+; RV64IMZBS-NEXT:    srli a6, s3, 56
+; RV64IMZBS-NEXT:    or a5, a5, a6
+; RV64IMZBS-NEXT:    and a6, s3, a0
+; RV64IMZBS-NEXT:    slli a6, a6, 24
+; RV64IMZBS-NEXT:    srliw a7, s3, 24
+; RV64IMZBS-NEXT:    slli a7, a7, 32
+; RV64IMZBS-NEXT:    and s0, s3, t1
+; RV64IMZBS-NEXT:    slli s0, s0, 40
+; RV64IMZBS-NEXT:    slli s1, s3, 56
+; RV64IMZBS-NEXT:    or a6, a6, a7
+; RV64IMZBS-NEXT:    or s0, s1, s0
+; RV64IMZBS-NEXT:    or a4, a4, a5
+; RV64IMZBS-NEXT:    or a5, s0, a6
+; RV64IMZBS-NEXT:    lui a6, 69905
+; RV64IMZBS-NEXT:    or a4, a5, a4
+; RV64IMZBS-NEXT:    srli a5, a4, 4
+; RV64IMZBS-NEXT:    and a4, a4, t4
+; RV64IMZBS-NEXT:    and a5, a5, t4
+; RV64IMZBS-NEXT:    slli a4, a4, 4
+; RV64IMZBS-NEXT:    addi a6, a6, 273
+; RV64IMZBS-NEXT:    or a4, a5, a4
+; RV64IMZBS-NEXT:    srli a5, a4, 2
+; RV64IMZBS-NEXT:    and a4, a4, t5
+; RV64IMZBS-NEXT:    and a5, a5, t5
+; RV64IMZBS-NEXT:    slli a4, a4, 2
+; RV64IMZBS-NEXT:    slli a7, a6, 32
+; RV64IMZBS-NEXT:    or a5, a5, a4
+; RV64IMZBS-NEXT:    add t2, a6, a7
+; RV64IMZBS-NEXT:    srli a6, a5, 1
+; RV64IMZBS-NEXT:    sd a1, 0(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    and a6, a6, a1
+; RV64IMZBS-NEXT:    lui a7, 139810
+; RV64IMZBS-NEXT:    and a5, a5, a1
+; RV64IMZBS-NEXT:    addi a7, a7, 546
+; RV64IMZBS-NEXT:    slli a5, a5, 1
+; RV64IMZBS-NEXT:    slli s0, a7, 32
+; RV64IMZBS-NEXT:    or s1, a6, a5
+; RV64IMZBS-NEXT:    add a5, a7, s0
+; RV64IMZBS-NEXT:    and s0, t6, t2
+; RV64IMZBS-NEXT:    and s2, s1, a5
+; RV64IMZBS-NEXT:    mul t0, s2, s0
+; RV64IMZBS-NEXT:    lui a6, %hi(.LCPI8_0)
+; RV64IMZBS-NEXT:    ld a6, %lo(.LCPI8_0)(a6)
+; RV64IMZBS-NEXT:    lui a7, 279620
+; RV64IMZBS-NEXT:    and s4, t6, a5
+; RV64IMZBS-NEXT:    addi a7, a7, 1092
+; RV64IMZBS-NEXT:    and s5, s1, t2
+; RV64IMZBS-NEXT:    slli s6, a7, 32
+; RV64IMZBS-NEXT:    mul s7, s5, s4
+; RV64IMZBS-NEXT:    add a7, a7, s6
+; RV64IMZBS-NEXT:    and s6, t6, a6
+; RV64IMZBS-NEXT:    and s8, s1, a7
+; RV64IMZBS-NEXT:    and t6, t6, a7
+; RV64IMZBS-NEXT:    and s1, s1, a6
+; RV64IMZBS-NEXT:    mul s9, s8, s6
+; RV64IMZBS-NEXT:    mul s10, s1, t6
+; RV64IMZBS-NEXT:    mul s11, s2, s6
+; RV64IMZBS-NEXT:    mul ra, s5, s0
+; RV64IMZBS-NEXT:    mul a4, s8, t6
+; RV64IMZBS-NEXT:    mul a3, s1, s4
+; RV64IMZBS-NEXT:    mul a1, s2, s4
+; RV64IMZBS-NEXT:    mul s2, s2, t6
+; RV64IMZBS-NEXT:    mul t6, s5, t6
+; RV64IMZBS-NEXT:    mul s5, s5, s6
+; RV64IMZBS-NEXT:    mul s6, s1, s6
+; RV64IMZBS-NEXT:    mul a0, s8, s0
+; RV64IMZBS-NEXT:    mul s4, s8, s4
+; RV64IMZBS-NEXT:    mul s0, s1, s0
+; RV64IMZBS-NEXT:    xor t0, s7, t0
+; RV64IMZBS-NEXT:    xor s1, s9, s10
+; RV64IMZBS-NEXT:    xor s7, ra, s11
+; RV64IMZBS-NEXT:    xor a3, a4, a3
+; RV64IMZBS-NEXT:    xor a4, t0, s1
+; RV64IMZBS-NEXT:    xor a3, s7, a3
+; RV64IMZBS-NEXT:    and a4, a4, a5
+; RV64IMZBS-NEXT:    and a3, a3, t2
+; RV64IMZBS-NEXT:    xor a1, t6, a1
+; RV64IMZBS-NEXT:    xor a0, a0, s6
+; RV64IMZBS-NEXT:    xor t0, s5, s2
+; RV64IMZBS-NEXT:    xor t6, s4, s0
+; RV64IMZBS-NEXT:    xor a0, a1, a0
+; RV64IMZBS-NEXT:    xor a1, t0, t6
+; RV64IMZBS-NEXT:    and a0, a0, a7
+; RV64IMZBS-NEXT:    and a1, a1, a6
+; RV64IMZBS-NEXT:    or a3, a3, a4
+; RV64IMZBS-NEXT:    or a0, a0, a1
+; RV64IMZBS-NEXT:    or a0, a3, a0
+; RV64IMZBS-NEXT:    srli a1, a0, 40
+; RV64IMZBS-NEXT:    and a1, a1, t1
+; RV64IMZBS-NEXT:    srli a3, a0, 56
+; RV64IMZBS-NEXT:    or a1, a1, a3
+; RV64IMZBS-NEXT:    srli a3, a0, 24
+; RV64IMZBS-NEXT:    srli a4, a0, 8
+; RV64IMZBS-NEXT:    lui t0, 4080
+; RV64IMZBS-NEXT:    and a3, a3, t0
+; RV64IMZBS-NEXT:    and a4, a4, t3
+; RV64IMZBS-NEXT:    or a3, a4, a3
+; RV64IMZBS-NEXT:    srliw a4, a0, 24
+; RV64IMZBS-NEXT:    slli a4, a4, 32
+; RV64IMZBS-NEXT:    and t0, a0, t0
+; RV64IMZBS-NEXT:    slli t0, t0, 24
+; RV64IMZBS-NEXT:    and t1, a0, t1
+; RV64IMZBS-NEXT:    slli a0, a0, 56
+; RV64IMZBS-NEXT:    slli t1, t1, 40
+; RV64IMZBS-NEXT:    or a4, t0, a4
+; RV64IMZBS-NEXT:    or a0, a0, t1
+; RV64IMZBS-NEXT:    or a1, a3, a1
+; RV64IMZBS-NEXT:    or a0, a0, a4
+; RV64IMZBS-NEXT:    or a0, a0, a1
+; RV64IMZBS-NEXT:    srli a1, a0, 4
+; RV64IMZBS-NEXT:    and a0, a0, t4
+; RV64IMZBS-NEXT:    and a1, a1, t4
+; RV64IMZBS-NEXT:    slli a0, a0, 4
+; RV64IMZBS-NEXT:    or a0, a1, a0
+; RV64IMZBS-NEXT:    srli a1, a0, 2
+; RV64IMZBS-NEXT:    and a0, a0, t5
+; RV64IMZBS-NEXT:    and a1, a1, t5
+; RV64IMZBS-NEXT:    slli a0, a0, 2
+; RV64IMZBS-NEXT:    or a0, a1, a0
+; RV64IMZBS-NEXT:    srli a1, a0, 1
+; RV64IMZBS-NEXT:    and t0, a2, t2
+; RV64IMZBS-NEXT:    ld s4, 8(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a4, s4, a5
+; RV64IMZBS-NEXT:    and t1, a2, a5
+; RV64IMZBS-NEXT:    and t4, s4, t2
+; RV64IMZBS-NEXT:    mul t5, a4, t0
+; RV64IMZBS-NEXT:    mul t6, t4, t1
+; RV64IMZBS-NEXT:    and s0, s4, a7
+; RV64IMZBS-NEXT:    and t3, a2, a7
+; RV64IMZBS-NEXT:    mul s1, t4, t0
+; RV64IMZBS-NEXT:    mul s2, s0, t3
+; RV64IMZBS-NEXT:    ld a3, 0(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and a1, a1, a3
+; RV64IMZBS-NEXT:    and a0, a0, a3
+; RV64IMZBS-NEXT:    slli a0, a0, 1
+; RV64IMZBS-NEXT:    and a3, a2, a6
+; RV64IMZBS-NEXT:    or a0, a1, a0
+; RV64IMZBS-NEXT:    sd a0, 0(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    mul a0, s0, a3
+; RV64IMZBS-NEXT:    and a1, s4, a6
+; RV64IMZBS-NEXT:    mul s4, a4, a3
+; RV64IMZBS-NEXT:    mul s5, a1, t3
+; RV64IMZBS-NEXT:    mul s6, a1, t1
+; RV64IMZBS-NEXT:    xor t5, t6, t5
+; RV64IMZBS-NEXT:    xor t6, s1, s2
+; RV64IMZBS-NEXT:    mul s1, a4, t1
+; RV64IMZBS-NEXT:    mul s2, t4, t3
+; RV64IMZBS-NEXT:    mul a4, a4, t3
+; RV64IMZBS-NEXT:    mul s7, s0, t1
+; RV64IMZBS-NEXT:    mul s0, s0, t0
+; RV64IMZBS-NEXT:    mul t4, t4, a3
+; RV64IMZBS-NEXT:    xor a0, t5, a0
+; RV64IMZBS-NEXT:    xor t5, t6, s4
+; RV64IMZBS-NEXT:    xor a0, a0, s5
+; RV64IMZBS-NEXT:    xor t5, t5, s6
+; RV64IMZBS-NEXT:    and a0, a0, a5
+; RV64IMZBS-NEXT:    and t6, t5, t2
+; RV64IMZBS-NEXT:    mul t5, a1, a3
+; RV64IMZBS-NEXT:    mul s4, a1, t0
+; RV64IMZBS-NEXT:    xor a1, s2, s1
+; RV64IMZBS-NEXT:    xor a4, a4, s7
+; RV64IMZBS-NEXT:    xor s0, a1, s0
+; RV64IMZBS-NEXT:    xor a4, t4, a4
+; RV64IMZBS-NEXT:    ld a2, 16(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and s1, a2, t2
+; RV64IMZBS-NEXT:    and t4, s3, a5
+; RV64IMZBS-NEXT:    and s2, a2, a5
+; RV64IMZBS-NEXT:    and a1, s3, t2
+; RV64IMZBS-NEXT:    mul s5, t4, s1
+; RV64IMZBS-NEXT:    mul s6, a1, s2
+; RV64IMZBS-NEXT:    xor s0, s0, t5
+; RV64IMZBS-NEXT:    xor a4, a4, s4
+; RV64IMZBS-NEXT:    and t5, s3, a7
+; RV64IMZBS-NEXT:    and s4, a2, a7
+; RV64IMZBS-NEXT:    mul s7, a1, s1
+; RV64IMZBS-NEXT:    mul s8, t5, s4
+; RV64IMZBS-NEXT:    and s0, s0, a7
+; RV64IMZBS-NEXT:    and a4, a4, a6
+; RV64IMZBS-NEXT:    or t6, t6, a0
+; RV64IMZBS-NEXT:    or s0, s0, a4
+; RV64IMZBS-NEXT:    xor a4, s6, s5
+; RV64IMZBS-NEXT:    and s5, a2, a6
+; RV64IMZBS-NEXT:    mul s6, t5, s5
+; RV64IMZBS-NEXT:    and a0, s3, a6
+; RV64IMZBS-NEXT:    mul s3, a0, s4
+; RV64IMZBS-NEXT:    mul s9, t4, s5
+; RV64IMZBS-NEXT:    xor s7, s7, s8
+; RV64IMZBS-NEXT:    mul s8, a0, s2
+; RV64IMZBS-NEXT:    mul s10, t4, s2
+; RV64IMZBS-NEXT:    mul s11, a1, s4
+; RV64IMZBS-NEXT:    mul s4, t4, s4
+; RV64IMZBS-NEXT:    mul s2, t5, s2
+; RV64IMZBS-NEXT:    mul ra, t5, s1
+; RV64IMZBS-NEXT:    mul a2, a1, s5
+; RV64IMZBS-NEXT:    mul s5, a0, s5
+; RV64IMZBS-NEXT:    mul s1, a0, s1
+; RV64IMZBS-NEXT:    xor a4, a4, s6
+; RV64IMZBS-NEXT:    xor s6, s7, s9
+; RV64IMZBS-NEXT:    xor a4, a4, s3
+; RV64IMZBS-NEXT:    xor s3, s6, s8
+; RV64IMZBS-NEXT:    and a4, a4, a5
+; RV64IMZBS-NEXT:    and s3, s3, t2
+; RV64IMZBS-NEXT:    xor s6, s11, s10
+; RV64IMZBS-NEXT:    xor s2, s4, s2
+; RV64IMZBS-NEXT:    xor s4, s6, ra
+; RV64IMZBS-NEXT:    xor a2, a2, s2
+; RV64IMZBS-NEXT:    xor s2, s4, s5
+; RV64IMZBS-NEXT:    xor a2, a2, s1
+; RV64IMZBS-NEXT:    mul s1, t4, t0
+; RV64IMZBS-NEXT:    mul s4, a1, t1
+; RV64IMZBS-NEXT:    and s2, s2, a7
+; RV64IMZBS-NEXT:    and a2, a2, a6
+; RV64IMZBS-NEXT:    mul s5, a1, t0
+; RV64IMZBS-NEXT:    mul s6, t5, t3
+; RV64IMZBS-NEXT:    or a4, s3, a4
+; RV64IMZBS-NEXT:    or a2, s2, a2
+; RV64IMZBS-NEXT:    or t6, t6, s0
+; RV64IMZBS-NEXT:    or a2, a4, a2
+; RV64IMZBS-NEXT:    ld a4, 0(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    srli a4, a4, 1
+; RV64IMZBS-NEXT:    xor a2, a2, t6
+; RV64IMZBS-NEXT:    xor t6, s4, s1
+; RV64IMZBS-NEXT:    mul s0, t5, a3
+; RV64IMZBS-NEXT:    mul s1, a0, t3
+; RV64IMZBS-NEXT:    mul s2, t4, a3
+; RV64IMZBS-NEXT:    xor s3, s5, s6
+; RV64IMZBS-NEXT:    mul s4, a0, t1
+; RV64IMZBS-NEXT:    mul s5, t4, t1
+; RV64IMZBS-NEXT:    mul s6, a1, t3
+; RV64IMZBS-NEXT:    mul t3, t4, t3
+; RV64IMZBS-NEXT:    mul t1, t5, t1
+; RV64IMZBS-NEXT:    mul t4, t5, t0
+; RV64IMZBS-NEXT:    mul a1, a1, a3
+; RV64IMZBS-NEXT:    mul a3, a0, a3
+; RV64IMZBS-NEXT:    mul a0, a0, t0
+; RV64IMZBS-NEXT:    xor t0, t6, s0
+; RV64IMZBS-NEXT:    xor t5, s3, s2
+; RV64IMZBS-NEXT:    xor t0, t0, s1
+; RV64IMZBS-NEXT:    xor t5, t5, s4
+; RV64IMZBS-NEXT:    and a5, t0, a5
+; RV64IMZBS-NEXT:    and t0, t5, t2
+; RV64IMZBS-NEXT:    xor t2, s6, s5
+; RV64IMZBS-NEXT:    xor t1, t3, t1
+; RV64IMZBS-NEXT:    xor t2, t2, t4
+; RV64IMZBS-NEXT:    xor a1, a1, t1
+; RV64IMZBS-NEXT:    xor a3, t2, a3
+; RV64IMZBS-NEXT:    xor a0, a1, a0
+; RV64IMZBS-NEXT:    and a1, a3, a7
+; RV64IMZBS-NEXT:    and a0, a0, a6
+; RV64IMZBS-NEXT:    or a3, t0, a5
+; RV64IMZBS-NEXT:    or a0, a1, a0
+; RV64IMZBS-NEXT:    xor a1, a4, a2
+; RV64IMZBS-NEXT:    or a0, a3, a0
+; RV64IMZBS-NEXT:    ld ra, 120(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s0, 112(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s1, 104(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s2, 96(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s3, 88(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s4, 80(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s5, 72(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s6, 64(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s7, 56(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s8, 48(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s9, 40(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s10, 32(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s11, 24(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    .cfi_restore ra
+; RV64IMZBS-NEXT:    .cfi_restore s0
+; RV64IMZBS-NEXT:    .cfi_restore s1
+; RV64IMZBS-NEXT:    .cfi_restore s2
+; RV64IMZBS-NEXT:    .cfi_restore s3
+; RV64IMZBS-NEXT:    .cfi_restore s4
+; RV64IMZBS-NEXT:    .cfi_restore s5
+; RV64IMZBS-NEXT:    .cfi_restore s6
+; RV64IMZBS-NEXT:    .cfi_restore s7
+; RV64IMZBS-NEXT:    .cfi_restore s8
+; RV64IMZBS-NEXT:    .cfi_restore s9
+; RV64IMZBS-NEXT:    .cfi_restore s10
+; RV64IMZBS-NEXT:    .cfi_restore s11
+; RV64IMZBS-NEXT:    addi sp, sp, 128
+; RV64IMZBS-NEXT:    .cfi_def_cfa_offset 0
+; RV64IMZBS-NEXT:    ret
+;
+; RV32IMZBC-LABEL: clmul_i128:
+; RV32IMZBC:       # %bb.0:
+; RV32IMZBC-NEXT:    addi sp, sp, -32
+; RV32IMZBC-NEXT:    .cfi_def_cfa_offset 32
+; RV32IMZBC-NEXT:    sw s0, 28(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s1, 24(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s2, 20(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s3, 16(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s4, 12(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    .cfi_offset s0, -4
+; RV32IMZBC-NEXT:    .cfi_offset s1, -8
+; RV32IMZBC-NEXT:    .cfi_offset s2, -12
+; RV32IMZBC-NEXT:    .cfi_offset s3, -16
+; RV32IMZBC-NEXT:    .cfi_offset s4, -20
+; RV32IMZBC-NEXT:    lw a3, 4(a2)
+; RV32IMZBC-NEXT:    lui a7, 16
+; RV32IMZBC-NEXT:    lw a5, 8(a2)
+; RV32IMZBC-NEXT:    lw a6, 12(a2)
+; RV32IMZBC-NEXT:    lw a4, 0(a1)
+; RV32IMZBC-NEXT:    addi t1, a7, -256
+; RV32IMZBC-NEXT:    srli a7, a3, 8
+; RV32IMZBC-NEXT:    srli t0, a3, 24
+; RV32IMZBC-NEXT:    and t2, a3, t1
+; RV32IMZBC-NEXT:    and a7, a7, t1
+; RV32IMZBC-NEXT:    slli t2, t2, 8
+; RV32IMZBC-NEXT:    slli t3, a3, 24
+; RV32IMZBC-NEXT:    or a7, a7, t0
+; RV32IMZBC-NEXT:    or t0, t3, t2
+; RV32IMZBC-NEXT:    or a7, t0, a7
+; RV32IMZBC-NEXT:    lui t0, 61681
+; RV32IMZBC-NEXT:    srli t3, a7, 4
+; RV32IMZBC-NEXT:    addi t2, t0, -241
+; RV32IMZBC-NEXT:    and t0, t3, t2
+; RV32IMZBC-NEXT:    and a7, a7, t2
+; RV32IMZBC-NEXT:    slli a7, a7, 4
+; RV32IMZBC-NEXT:    lui t3, 209715
+; RV32IMZBC-NEXT:    or a7, t0, a7
+; RV32IMZBC-NEXT:    addi t3, t3, 819
+; RV32IMZBC-NEXT:    srli t0, a7, 2
+; RV32IMZBC-NEXT:    and a7, a7, t3
+; RV32IMZBC-NEXT:    and t0, t0, t3
+; RV32IMZBC-NEXT:    slli a7, a7, 2
+; RV32IMZBC-NEXT:    lw a2, 0(a2)
+; RV32IMZBC-NEXT:    or t6, t0, a7
+; RV32IMZBC-NEXT:    lw a7, 4(a1)
+; RV32IMZBC-NEXT:    lw t0, 8(a1)
+; RV32IMZBC-NEXT:    lw a1, 12(a1)
+; RV32IMZBC-NEXT:    srli s0, t6, 1
+; RV32IMZBC-NEXT:    lui t4, 349525
+; RV32IMZBC-NEXT:    srli s1, a4, 8
+; RV32IMZBC-NEXT:    addi t5, t4, 1365
+; RV32IMZBC-NEXT:    and s1, s1, t1
+; RV32IMZBC-NEXT:    srli s2, a4, 24
+; RV32IMZBC-NEXT:    and s3, a4, t1
+; RV32IMZBC-NEXT:    slli s3, s3, 8
+; RV32IMZBC-NEXT:    slli s4, a4, 24
+; RV32IMZBC-NEXT:    or s1, s1, s2
+; RV32IMZBC-NEXT:    or s2, s4, s3
+; RV32IMZBC-NEXT:    and s0, s0, t5
+; RV32IMZBC-NEXT:    or s1, s2, s1
+; RV32IMZBC-NEXT:    srli s2, s1, 4
+; RV32IMZBC-NEXT:    and s1, s1, t2
+; RV32IMZBC-NEXT:    and s2, s2, t2
+; RV32IMZBC-NEXT:    slli s1, s1, 4
+; RV32IMZBC-NEXT:    and t6, t6, t5
+; RV32IMZBC-NEXT:    or s1, s2, s1
+; RV32IMZBC-NEXT:    srli s2, s1, 2
+; RV32IMZBC-NEXT:    and s1, s1, t3
+; RV32IMZBC-NEXT:    and s2, s2, t3
+; RV32IMZBC-NEXT:    slli s1, s1, 2
+; RV32IMZBC-NEXT:    slli t6, t6, 1
+; RV32IMZBC-NEXT:    or s1, s2, s1
+; RV32IMZBC-NEXT:    or t6, s0, t6
+; RV32IMZBC-NEXT:    srli s0, s1, 1
+; RV32IMZBC-NEXT:    and s0, s0, t5
+; RV32IMZBC-NEXT:    srli s2, a2, 8
+; RV32IMZBC-NEXT:    and s2, s2, t1
+; RV32IMZBC-NEXT:    srli s3, a2, 24
+; RV32IMZBC-NEXT:    or s2, s2, s3
+; RV32IMZBC-NEXT:    and s3, a2, t1
+; RV32IMZBC-NEXT:    slli s3, s3, 8
+; RV32IMZBC-NEXT:    slli s4, a2, 24
+; RV32IMZBC-NEXT:    and s1, s1, t5
+; RV32IMZBC-NEXT:    or s3, s4, s3
+; RV32IMZBC-NEXT:    slli s1, s1, 1
+; RV32IMZBC-NEXT:    or s2, s3, s2
+; RV32IMZBC-NEXT:    srli s3, s2, 4
+; RV32IMZBC-NEXT:    and s2, s2, t2
+; RV32IMZBC-NEXT:    and s3, s3, t2
+; RV32IMZBC-NEXT:    slli s2, s2, 4
+; RV32IMZBC-NEXT:    or s0, s0, s1
+; RV32IMZBC-NEXT:    or s1, s3, s2
+; RV32IMZBC-NEXT:    srli s2, s1, 2
+; RV32IMZBC-NEXT:    and s1, s1, t3
+; RV32IMZBC-NEXT:    and s2, s2, t3
+; RV32IMZBC-NEXT:    slli s1, s1, 2
+; RV32IMZBC-NEXT:    or s1, s2, s1
+; RV32IMZBC-NEXT:    srli s2, a7, 8
+; RV32IMZBC-NEXT:    and s2, s2, t1
+; RV32IMZBC-NEXT:    srli s3, a7, 24
+; RV32IMZBC-NEXT:    or s2, s2, s3
+; RV32IMZBC-NEXT:    and s3, a7, t1
+; RV32IMZBC-NEXT:    slli s3, s3, 8
+; RV32IMZBC-NEXT:    slli s4, a7, 24
+; RV32IMZBC-NEXT:    or s3, s4, s3
+; RV32IMZBC-NEXT:    srli s4, s1, 1
+; RV32IMZBC-NEXT:    and s4, s4, t5
+; RV32IMZBC-NEXT:    or s2, s3, s2
+; RV32IMZBC-NEXT:    srli s3, s2, 4
+; RV32IMZBC-NEXT:    and s2, s2, t2
+; RV32IMZBC-NEXT:    and s3, s3, t2
+; RV32IMZBC-NEXT:    slli s2, s2, 4
+; RV32IMZBC-NEXT:    and s1, s1, t5
+; RV32IMZBC-NEXT:    or s2, s3, s2
+; RV32IMZBC-NEXT:    srli s3, s2, 2
+; RV32IMZBC-NEXT:    and s2, s2, t3
+; RV32IMZBC-NEXT:    and s3, s3, t3
+; RV32IMZBC-NEXT:    slli s2, s2, 2
+; RV32IMZBC-NEXT:    slli s1, s1, 1
+; RV32IMZBC-NEXT:    or s2, s3, s2
+; RV32IMZBC-NEXT:    srli s3, s2, 1
+; RV32IMZBC-NEXT:    and s2, s2, t5
+; RV32IMZBC-NEXT:    and s3, s3, t5
+; RV32IMZBC-NEXT:    slli s2, s2, 1
+; RV32IMZBC-NEXT:    or s1, s4, s1
+; RV32IMZBC-NEXT:    or s2, s3, s2
+; RV32IMZBC-NEXT:    clmul s0, s0, t6
+; RV32IMZBC-NEXT:    clmul s1, s2, s1
+; RV32IMZBC-NEXT:    clmulh t6, s2, t6
+; RV32IMZBC-NEXT:    xor s0, s1, s0
+; RV32IMZBC-NEXT:    xor t6, t6, s0
+; RV32IMZBC-NEXT:    srli s0, t6, 8
+; RV32IMZBC-NEXT:    and s0, s0, t1
+; RV32IMZBC-NEXT:    srli s1, t6, 24
+; RV32IMZBC-NEXT:    and t1, t6, t1
+; RV32IMZBC-NEXT:    slli t6, t6, 24
+; RV32IMZBC-NEXT:    slli t1, t1, 8
+; RV32IMZBC-NEXT:    or s0, s0, s1
+; RV32IMZBC-NEXT:    or t1, t6, t1
+; RV32IMZBC-NEXT:    or t1, t1, s0
+; RV32IMZBC-NEXT:    srli t6, t1, 4
+; RV32IMZBC-NEXT:    and t1, t1, t2
+; RV32IMZBC-NEXT:    and t2, t6, t2
+; RV32IMZBC-NEXT:    slli t1, t1, 4
+; RV32IMZBC-NEXT:    or t1, t2, t1
+; RV32IMZBC-NEXT:    srli t2, t1, 2
+; RV32IMZBC-NEXT:    and t1, t1, t3
+; RV32IMZBC-NEXT:    and t2, t2, t3
+; RV32IMZBC-NEXT:    slli t1, t1, 2
+; RV32IMZBC-NEXT:    or t1, t2, t1
+; RV32IMZBC-NEXT:    srli t2, t1, 1
+; RV32IMZBC-NEXT:    addi t3, t4, 1364
+; RV32IMZBC-NEXT:    and t1, t1, t5
+; RV32IMZBC-NEXT:    and t2, t2, t3
+; RV32IMZBC-NEXT:    slli t1, t1, 1
+; RV32IMZBC-NEXT:    or t1, t2, t1
+; RV32IMZBC-NEXT:    clmulr t2, a7, a3
+; RV32IMZBC-NEXT:    clmul t3, t0, a2
+; RV32IMZBC-NEXT:    clmul t4, a4, a5
+; RV32IMZBC-NEXT:    srli t1, t1, 1
+; RV32IMZBC-NEXT:    slli t2, t2, 31
+; RV32IMZBC-NEXT:    or t1, t1, t2
+; RV32IMZBC-NEXT:    xor t2, t4, t3
+; RV32IMZBC-NEXT:    clmul a1, a1, a2
+; RV32IMZBC-NEXT:    clmul t3, t0, a3
+; RV32IMZBC-NEXT:    clmul t4, a7, a5
+; RV32IMZBC-NEXT:    clmul a6, a4, a6
+; RV32IMZBC-NEXT:    clmulh t0, t0, a2
+; RV32IMZBC-NEXT:    clmulh a5, a4, a5
+; RV32IMZBC-NEXT:    xor a1, t3, a1
+; RV32IMZBC-NEXT:    xor a6, a6, t4
+; RV32IMZBC-NEXT:    xor a1, t0, a1
+; RV32IMZBC-NEXT:    xor a5, a5, a6
+; RV32IMZBC-NEXT:    xor a6, t1, t2
+; RV32IMZBC-NEXT:    xor a1, a5, a1
+; RV32IMZBC-NEXT:    clmul a5, a7, a2
+; RV32IMZBC-NEXT:    clmul t0, a4, a3
+; RV32IMZBC-NEXT:    clmulh a3, a7, a3
+; RV32IMZBC-NEXT:    clmulh a7, a4, a2
+; RV32IMZBC-NEXT:    xor a5, t0, a5
+; RV32IMZBC-NEXT:    clmul a2, a4, a2
+; RV32IMZBC-NEXT:    xor a1, a3, a1
+; RV32IMZBC-NEXT:    xor a3, a7, a5
+; RV32IMZBC-NEXT:    sw a2, 0(a0)
+; RV32IMZBC-NEXT:    sw a3, 4(a0)
+; RV32IMZBC-NEXT:    sw a6, 8(a0)
+; RV32IMZBC-NEXT:    sw a1, 12(a0)
+; RV32IMZBC-NEXT:    lw s0, 28(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s1, 24(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s2, 20(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s3, 16(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s4, 12(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    .cfi_restore s0
+; RV32IMZBC-NEXT:    .cfi_restore s1
+; RV32IMZBC-NEXT:    .cfi_restore s2
+; RV32IMZBC-NEXT:    .cfi_restore s3
+; RV32IMZBC-NEXT:    .cfi_restore s4
+; RV32IMZBC-NEXT:    addi sp, sp, 32
+; RV32IMZBC-NEXT:    .cfi_def_cfa_offset 0
+; RV32IMZBC-NEXT:    ret
+;
+; RV64IMZBC-LABEL: clmul_i128:
+; RV64IMZBC:       # %bb.0:
+; RV64IMZBC-NEXT:    clmul a1, a1, a2
+; RV64IMZBC-NEXT:    clmul a3, a0, a3
+; RV64IMZBC-NEXT:    clmulh a4, a0, a2
+; RV64IMZBC-NEXT:    xor a1, a3, a1
+; RV64IMZBC-NEXT:    clmul a0, a0, a2
+; RV64IMZBC-NEXT:    xor a1, a4, a1
+; RV64IMZBC-NEXT:    ret
+  %a = call i128 @llvm.clmul.i128(i128 %x, i128 %y)
+  ret i128 %a
+}
+
+define i128 @clmul_i128_zext(i64 %x, i64 %y) {
+; RV32I-LABEL: clmul_i128_zext:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi sp, sp, -544
+; RV32I-NEXT:    .cfi_def_cfa_offset 544
+; RV32I-NEXT:    sw ra, 540(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s0, 536(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s1, 532(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s2, 528(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s3, 524(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s4, 520(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s5, 516(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s6, 512(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s7, 508(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s8, 504(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s9, 500(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s10, 496(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s11, 492(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    .cfi_offset ra, -4
+; RV32I-NEXT:    .cfi_offset s0, -8
+; RV32I-NEXT:    .cfi_offset s1, -12
+; RV32I-NEXT:    .cfi_offset s2, -16
+; RV32I-NEXT:    .cfi_offset s3, -20
+; RV32I-NEXT:    .cfi_offset s4, -24
+; RV32I-NEXT:    .cfi_offset s5, -28
+; RV32I-NEXT:    .cfi_offset s6, -32
+; RV32I-NEXT:    .cfi_offset s7, -36
+; RV32I-NEXT:    .cfi_offset s8, -40
+; RV32I-NEXT:    .cfi_offset s9, -44
+; RV32I-NEXT:    .cfi_offset s10, -48
+; RV32I-NEXT:    .cfi_offset s11, -52
+; RV32I-NEXT:    mv t6, a4
+; RV32I-NEXT:    mv t2, a2
+; RV32I-NEXT:    mv s6, a1
+; RV32I-NEXT:    sw a0, 480(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a1, 16
+; RV32I-NEXT:    srli a0, a2, 8
+; RV32I-NEXT:    addi s10, a1, -256
+; RV32I-NEXT:    lui s1, 16
+; RV32I-NEXT:    and a0, a0, s10
+; RV32I-NEXT:    srli a1, a2, 24
+; RV32I-NEXT:    or a0, a0, a1
+; RV32I-NEXT:    and a1, a2, s10
+; RV32I-NEXT:    slli a1, a1, 8
+; RV32I-NEXT:    slli a2, a2, 24
+; RV32I-NEXT:    sw a2, 476(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    or a1, a2, a1
+; RV32I-NEXT:    lui a4, 61681
+; RV32I-NEXT:    or a0, a1, a0
+; RV32I-NEXT:    addi s5, a4, -241
+; RV32I-NEXT:    srli a1, a0, 4
+; RV32I-NEXT:    and a0, a0, s5
+; RV32I-NEXT:    and a1, a1, s5
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    or a0, a1, a0
+; RV32I-NEXT:    lui a1, 209715
+; RV32I-NEXT:    srli a4, a0, 2
+; RV32I-NEXT:    addi s7, a1, 819
+; RV32I-NEXT:    and a1, a4, s7
+; RV32I-NEXT:    and a0, a0, s7
+; RV32I-NEXT:    slli a4, a0, 2
+; RV32I-NEXT:    lui a0, 349525
+; RV32I-NEXT:    or a1, a1, a4
+; RV32I-NEXT:    addi s0, a0, 1365
+; RV32I-NEXT:    srli a4, a1, 1
+; RV32I-NEXT:    and a1, a1, s0
+; RV32I-NEXT:    and a4, a4, s0
+; RV32I-NEXT:    slli a1, a1, 1
+; RV32I-NEXT:    or t1, a4, a1
+; RV32I-NEXT:    srli a1, t1, 8
+; RV32I-NEXT:    and a1, a1, s10
+; RV32I-NEXT:    srli a4, t1, 24
+; RV32I-NEXT:    and a5, t1, s10
+; RV32I-NEXT:    slli a6, t1, 24
+; RV32I-NEXT:    sw a6, 468(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a5, 8
+; RV32I-NEXT:    or a1, a1, a4
+; RV32I-NEXT:    or a4, a6, a5
+; RV32I-NEXT:    or a1, a4, a1
+; RV32I-NEXT:    srli a4, a1, 4
+; RV32I-NEXT:    and a1, a1, s5
+; RV32I-NEXT:    and a4, a4, s5
+; RV32I-NEXT:    slli a1, a1, 4
+; RV32I-NEXT:    or a1, a4, a1
+; RV32I-NEXT:    srli a4, t6, 8
+; RV32I-NEXT:    srli a5, a1, 2
+; RV32I-NEXT:    and a4, a4, s10
+; RV32I-NEXT:    srli a6, t6, 24
+; RV32I-NEXT:    and a7, t6, s10
+; RV32I-NEXT:    slli a7, a7, 8
+; RV32I-NEXT:    slli t0, t6, 24
+; RV32I-NEXT:    or a4, a4, a6
+; RV32I-NEXT:    or a6, t0, a7
+; RV32I-NEXT:    and a5, a5, s7
+; RV32I-NEXT:    or a4, a6, a4
+; RV32I-NEXT:    srli a6, a4, 4
+; RV32I-NEXT:    and a4, a4, s5
+; RV32I-NEXT:    and a6, a6, s5
 ; RV32I-NEXT:    slli a4, a4, 4
-; RV32I-NEXT:    or a4, a5, a4
-; RV32I-NEXT:    srli a5, a4, 2
-; RV32I-NEXT:    and a4, a4, a6
-; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    and a1, a1, s7
+; RV32I-NEXT:    or a4, a6, a4
+; RV32I-NEXT:    srli a6, a4, 2
+; RV32I-NEXT:    and a4, a4, s7
+; RV32I-NEXT:    and a6, a6, s7
 ; RV32I-NEXT:    slli a4, a4, 2
-; RV32I-NEXT:    or a4, a5, a4
-; RV32I-NEXT:    and a5, a4, t0
-; RV32I-NEXT:    srli a4, a4, 1
-; RV32I-NEXT:    addi a6, a7, 1364
-; RV32I-NEXT:    and a4, a4, a6
-; RV32I-NEXT:    slli a5, a5, 1
-; RV32I-NEXT:    or a4, a4, a5
-; RV32I-NEXT:    sw a4, 184(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    andi a4, a2, 2
-; RV32I-NEXT:    slli a5, a1, 1
-; RV32I-NEXT:    seqz a4, a4
-; RV32I-NEXT:    addi a6, a4, -1
-; RV32I-NEXT:    sw a6, 176(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    andi a4, a2, 1
-; RV32I-NEXT:    and a5, a6, a5
-; RV32I-NEXT:    seqz a4, a4
-; RV32I-NEXT:    addi a6, a4, -1
-; RV32I-NEXT:    sw a6, 172(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    andi a4, a2, 4
+; RV32I-NEXT:    slli a1, a1, 2
+; RV32I-NEXT:    or a4, a6, a4
+; RV32I-NEXT:    srli a6, a4, 1
+; RV32I-NEXT:    and a4, a4, s0
+; RV32I-NEXT:    and a6, a6, s0
+; RV32I-NEXT:    slli a4, a4, 1
+; RV32I-NEXT:    sw a4, 464(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    or a1, a5, a1
+; RV32I-NEXT:    or s2, a6, a4
+; RV32I-NEXT:    srli a4, a1, 1
+; RV32I-NEXT:    srli a5, s2, 8
+; RV32I-NEXT:    and a4, a4, s0
+; RV32I-NEXT:    and a5, a5, s10
+; RV32I-NEXT:    srli a6, s2, 24
+; RV32I-NEXT:    and a7, s2, s10
+; RV32I-NEXT:    slli t0, s2, 24
+; RV32I-NEXT:    slli a7, a7, 8
+; RV32I-NEXT:    or a5, a5, a6
+; RV32I-NEXT:    or a6, t0, a7
+; RV32I-NEXT:    and a1, a1, s0
+; RV32I-NEXT:    or a5, a6, a5
+; RV32I-NEXT:    srli a6, a5, 4
+; RV32I-NEXT:    and a5, a5, s5
+; RV32I-NEXT:    and a6, a6, s5
+; RV32I-NEXT:    slli a5, a5, 4
+; RV32I-NEXT:    slli a1, a1, 1
+; RV32I-NEXT:    or a5, a6, a5
+; RV32I-NEXT:    srli a6, a5, 2
+; RV32I-NEXT:    and a5, a5, s7
+; RV32I-NEXT:    and a6, a6, s7
+; RV32I-NEXT:    slli a5, a5, 2
+; RV32I-NEXT:    or t0, a4, a1
+; RV32I-NEXT:    or a1, a6, a5
+; RV32I-NEXT:    srli a4, a1, 1
+; RV32I-NEXT:    and a1, a1, s0
+; RV32I-NEXT:    and a4, a4, s0
+; RV32I-NEXT:    slli a1, a1, 1
+; RV32I-NEXT:    slli a5, t0, 1
+; RV32I-NEXT:    or t3, a4, a1
+; RV32I-NEXT:    andi a4, t3, 2
+; RV32I-NEXT:    andi a6, t3, 1
 ; RV32I-NEXT:    seqz a4, a4
-; RV32I-NEXT:    and a6, a6, a1
-; RV32I-NEXT:    xor a5, a6, a5
-; RV32I-NEXT:    addi a7, a4, -1
-; RV32I-NEXT:    sw a7, 168(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a4, a1, 2
-; RV32I-NEXT:    andi a6, a2, 8
-; RV32I-NEXT:    and a4, a7, a4
 ; RV32I-NEXT:    seqz a6, a6
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 164(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 3
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    andi a7, a2, 16
-; RV32I-NEXT:    xor a4, a4, a6
-; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    and a4, a4, a5
+; RV32I-NEXT:    and a5, a6, t0
 ; RV32I-NEXT:    xor a4, a5, a4
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 160(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a5, a1, 4
-; RV32I-NEXT:    andi a6, a2, 32
-; RV32I-NEXT:    and a5, a7, a5
-; RV32I-NEXT:    seqz a6, a6
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 156(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 5
+; RV32I-NEXT:    andi a5, t3, 4
+; RV32I-NEXT:    slli a6, t0, 2
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    andi a7, t3, 8
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    slli a7, t0, 3
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    andi a7, t3, 16
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    slli a7, t0, 4
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    andi a7, t3, 32
+; RV32I-NEXT:    slli t4, t0, 5
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    andi t5, t3, 64
+; RV32I-NEXT:    and a7, a7, t4
+; RV32I-NEXT:    seqz t4, t5
+; RV32I-NEXT:    slli t5, t0, 6
+; RV32I-NEXT:    addi t4, t4, -1
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, t4, t5
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    xor a5, a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    andi a5, t3, 128
+; RV32I-NEXT:    slli a6, t0, 7
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    andi a7, t3, 256
+; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    slli a7, t0, 8
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    andi a7, t3, 512
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    slli a7, t0, 9
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    andi a7, t3, 1024
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    slli a7, t0, 10
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    li a7, 1
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    slli s3, a7, 11
+; RV32I-NEXT:    slli a6, t0, 11
+; RV32I-NEXT:    and a7, t3, s3
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    lui a2, 1
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    and t4, t3, a2
 ; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    andi a7, a2, 64
+; RV32I-NEXT:    seqz a7, t4
+; RV32I-NEXT:    slli t4, t0, 12
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    and a7, a7, t4
+; RV32I-NEXT:    lui a2, 2
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, t3, a2
+; RV32I-NEXT:    slli t4, t0, 13
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    lui a2, 4
+; RV32I-NEXT:    and a7, a7, t4
+; RV32I-NEXT:    and t4, t3, a2
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    seqz a7, t4
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    lui a2, 8
+; RV32I-NEXT:    slli t4, t0, 14
+; RV32I-NEXT:    and t5, t3, a2
+; RV32I-NEXT:    and a7, a7, t4
+; RV32I-NEXT:    seqz t4, t5
+; RV32I-NEXT:    addi t4, t4, -1
+; RV32I-NEXT:    slli t5, t0, 15
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, t4, t5
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    xor a5, a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, t3, s1
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    lui a2, 32
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    and a6, t3, a2
+; RV32I-NEXT:    lui s11, 32
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    slli a7, t0, 16
+; RV32I-NEXT:    and a5, a5, a7
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    slli a7, t0, 17
+; RV32I-NEXT:    lui a2, 64
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    and a7, t3, a2
+; RV32I-NEXT:    lui s9, 64
 ; RV32I-NEXT:    xor a5, a5, a6
 ; RV32I-NEXT:    seqz a6, a7
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 152(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 6
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    lui a2, 128
+; RV32I-NEXT:    slli a7, t0, 18
+; RV32I-NEXT:    and t4, t3, a2
+; RV32I-NEXT:    lui t5, 128
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    seqz a7, t4
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    slli a6, t0, 19
+; RV32I-NEXT:    lui a2, 256
 ; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    andi a7, a2, 128
+; RV32I-NEXT:    and a7, t3, a2
 ; RV32I-NEXT:    xor a5, a5, a6
 ; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    lui a2, 512
+; RV32I-NEXT:    slli a7, t0, 20
+; RV32I-NEXT:    and t4, t3, a2
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    seqz a7, t4
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    slli t4, t0, 21
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    and a6, a7, t4
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lui a2, 1024
 ; RV32I-NEXT:    xor a4, a4, a5
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 148(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a5, a1, 7
-; RV32I-NEXT:    andi a6, a2, 256
-; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    and a5, t3, a2
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    lui a2, 2048
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    and a6, t3, a2
 ; RV32I-NEXT:    seqz a6, a6
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 144(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 8
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    andi a7, a2, 512
+; RV32I-NEXT:    slli a7, t0, 22
+; RV32I-NEXT:    and a5, a5, a7
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    slli a7, t0, 23
+; RV32I-NEXT:    lui a2, 4096
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    and a7, t3, a2
 ; RV32I-NEXT:    xor a5, a5, a6
 ; RV32I-NEXT:    seqz a6, a7
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 140(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 9
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    lui a2, 8192
+; RV32I-NEXT:    slli a7, t0, 24
+; RV32I-NEXT:    and t4, t3, a2
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    seqz a7, t4
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    slli a6, t0, 25
+; RV32I-NEXT:    lui a2, 16384
 ; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    andi a7, a2, 1024
+; RV32I-NEXT:    and a7, t3, a2
+; RV32I-NEXT:    lui ra, 16384
 ; RV32I-NEXT:    xor a5, a5, a6
 ; RV32I-NEXT:    seqz a6, a7
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 136(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 10
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    lui a2, 32768
+; RV32I-NEXT:    slli a7, t0, 26
+; RV32I-NEXT:    and t4, t3, a2
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    seqz a7, t4
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    slli a6, t0, 27
+; RV32I-NEXT:    lui a2, 65536
 ; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    and a7, a2, s5
+; RV32I-NEXT:    and a7, t3, a2
 ; RV32I-NEXT:    xor a5, a5, a6
 ; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    slli a7, t0, 28
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    lui a2, 131072
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    and a6, t3, a2
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    lui a2, 262144
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    and a7, t3, a2
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    slli t3, t0, 29
+; RV32I-NEXT:    and a6, a6, t3
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    srli a1, a1, 31
+; RV32I-NEXT:    slli t3, t0, 30
+; RV32I-NEXT:    and a7, a7, t3
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    slli t0, t0, 31
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a1, a1, t0
 ; RV32I-NEXT:    xor a4, a4, a5
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 132(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a5, a1, 11
-; RV32I-NEXT:    and a6, a2, s6
-; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    xor a1, a6, a1
+; RV32I-NEXT:    xor a1, a4, a1
+; RV32I-NEXT:    srli a4, a1, 8
+; RV32I-NEXT:    and a4, a4, s10
+; RV32I-NEXT:    srli a5, a1, 24
+; RV32I-NEXT:    and a6, a1, s10
+; RV32I-NEXT:    slli a1, a1, 24
+; RV32I-NEXT:    slli a6, a6, 8
+; RV32I-NEXT:    or a4, a4, a5
+; RV32I-NEXT:    or a1, a1, a6
+; RV32I-NEXT:    or a1, a1, a4
+; RV32I-NEXT:    srli a4, a1, 4
+; RV32I-NEXT:    and a4, a4, s5
+; RV32I-NEXT:    and a1, a1, s5
+; RV32I-NEXT:    slli a1, a1, 4
+; RV32I-NEXT:    srli a5, s6, 8
+; RV32I-NEXT:    or a1, a4, a1
+; RV32I-NEXT:    and a4, a5, s10
+; RV32I-NEXT:    srli a5, a1, 2
+; RV32I-NEXT:    srli a6, s6, 24
+; RV32I-NEXT:    or a4, a4, a6
+; RV32I-NEXT:    and a5, a5, s7
+; RV32I-NEXT:    and a1, a1, s7
+; RV32I-NEXT:    and a6, s6, s10
+; RV32I-NEXT:    slli a6, a6, 8
+; RV32I-NEXT:    slli a2, s6, 24
+; RV32I-NEXT:    sw a2, 472(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a1, 2
+; RV32I-NEXT:    or a6, a2, a6
+; RV32I-NEXT:    or a1, a5, a1
+; RV32I-NEXT:    or a4, a6, a4
+; RV32I-NEXT:    srli a5, a4, 4
+; RV32I-NEXT:    and a4, a4, s5
+; RV32I-NEXT:    and a5, a5, s5
+; RV32I-NEXT:    slli a4, a4, 4
+; RV32I-NEXT:    or a4, a5, a4
+; RV32I-NEXT:    addi a2, a0, 1364
+; RV32I-NEXT:    sw a2, 484(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a0, a1, 1
+; RV32I-NEXT:    and a1, a1, s0
+; RV32I-NEXT:    and a0, a0, a2
+; RV32I-NEXT:    slli a1, a1, 1
+; RV32I-NEXT:    srli a5, a4, 2
+; RV32I-NEXT:    and a4, a4, s7
+; RV32I-NEXT:    and a5, a5, s7
+; RV32I-NEXT:    slli a4, a4, 2
+; RV32I-NEXT:    or a0, a0, a1
+; RV32I-NEXT:    sw a0, 372(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    or a4, a5, a4
+; RV32I-NEXT:    srli a0, a4, 1
+; RV32I-NEXT:    and a1, a4, s0
+; RV32I-NEXT:    and a0, a0, s0
+; RV32I-NEXT:    slli a1, a1, 1
+; RV32I-NEXT:    or a2, a0, a1
+; RV32I-NEXT:    andi a0, s2, 2
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    andi a4, s2, 1
+; RV32I-NEXT:    addi a1, a0, -1
+; RV32I-NEXT:    sw a1, 400(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a0, a4
+; RV32I-NEXT:    addi a4, a0, -1
+; RV32I-NEXT:    sw a4, 396(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a0, a2, 1
+; RV32I-NEXT:    sw a0, 460(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a0, a1, a0
+; RV32I-NEXT:    and a4, a4, a2
+; RV32I-NEXT:    xor a0, a4, a0
+; RV32I-NEXT:    andi a4, s2, 4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    andi a5, s2, 8
+; RV32I-NEXT:    addi a6, a4, -1
+; RV32I-NEXT:    sw a6, 384(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 380(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 2
+; RV32I-NEXT:    sw a1, 456(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a1
+; RV32I-NEXT:    slli a1, a2, 3
+; RV32I-NEXT:    sw a1, 452(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a1
+; RV32I-NEXT:    andi a6, s2, 16
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a7, a5, -1
+; RV32I-NEXT:    sw a7, 360(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a5, s2, 32
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    andi a6, s2, 64
+; RV32I-NEXT:    addi t0, a5, -1
+; RV32I-NEXT:    sw t0, 356(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi t3, a5, -1
+; RV32I-NEXT:    sw t3, 352(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 4
+; RV32I-NEXT:    sw a1, 448(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a7, a1
+; RV32I-NEXT:    slli a1, a2, 5
+; RV32I-NEXT:    sw a1, 444(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, t0, a1
+; RV32I-NEXT:    slli a1, a2, 6
+; RV32I-NEXT:    sw a1, 440(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    and a6, t3, a1
+; RV32I-NEXT:    xor a0, a0, a4
+; RV32I-NEXT:    xor a4, a5, a6
+; RV32I-NEXT:    xor a0, a0, a4
+; RV32I-NEXT:    andi a4, s2, 128
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    andi a5, s2, 256
+; RV32I-NEXT:    addi a6, a4, -1
+; RV32I-NEXT:    sw a6, 344(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 340(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 7
+; RV32I-NEXT:    sw a1, 436(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a1
+; RV32I-NEXT:    slli a1, a2, 8
+; RV32I-NEXT:    sw a1, 432(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a1
+; RV32I-NEXT:    andi a6, s2, 512
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 328(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 9
+; RV32I-NEXT:    sw a1, 428(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a1
+; RV32I-NEXT:    andi a6, s2, 1024
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 320(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 10
+; RV32I-NEXT:    sw a1, 424(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a1
+; RV32I-NEXT:    sw s3, 288(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, s2, s3
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a7, a5, -1
+; RV32I-NEXT:    sw a7, 308(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui s1, 1
+; RV32I-NEXT:    and a5, s2, s1
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    lui s8, 2
+; RV32I-NEXT:    and a6, s2, s8
+; RV32I-NEXT:    addi t0, a5, -1
+; RV32I-NEXT:    sw t0, 304(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi t4, a5, -1
+; RV32I-NEXT:    sw t4, 300(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 11
+; RV32I-NEXT:    sw a1, 420(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a7, a1
+; RV32I-NEXT:    slli a1, a2, 12
+; RV32I-NEXT:    sw a1, 416(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, t0, a1
+; RV32I-NEXT:    slli a1, a2, 13
+; RV32I-NEXT:    sw a1, 412(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    and a6, t4, a1
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lui t3, 4
+; RV32I-NEXT:    and a6, s2, t3
 ; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    lui a1, 8
+; RV32I-NEXT:    and a7, s2, a1
+; RV32I-NEXT:    addi t0, a6, -1
+; RV32I-NEXT:    sw t0, 296(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a6, a7
 ; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 128(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 12
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    and a7, a2, s7
+; RV32I-NEXT:    sw a7, 292(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 14
+; RV32I-NEXT:    sw a1, 408(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, t0, a1
+; RV32I-NEXT:    slli a1, a2, 15
+; RV32I-NEXT:    sw a1, 404(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    and a6, a7, a1
+; RV32I-NEXT:    xor a0, a0, a4
+; RV32I-NEXT:    xor a4, a5, a6
+; RV32I-NEXT:    xor a0, a0, a4
+; RV32I-NEXT:    lui s4, 16
+; RV32I-NEXT:    and a4, s2, s4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a5, s2, s11
+; RV32I-NEXT:    addi a6, a4, -1
+; RV32I-NEXT:    sw a6, 284(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 280(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 16
+; RV32I-NEXT:    sw a1, 392(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a6, a1
+; RV32I-NEXT:    slli a1, a2, 17
+; RV32I-NEXT:    sw a1, 388(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a1
+; RV32I-NEXT:    and a6, s2, s9
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 276(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 18
+; RV32I-NEXT:    sw a1, 376(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a1
+; RV32I-NEXT:    and a6, s2, t5
+; RV32I-NEXT:    lui s9, 128
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 272(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 19
+; RV32I-NEXT:    sw a1, 368(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a1
+; RV32I-NEXT:    lui a1, 256
+; RV32I-NEXT:    and a6, s2, a1
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 264(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 20
+; RV32I-NEXT:    sw a1, 364(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a1
+; RV32I-NEXT:    lui a1, 512
+; RV32I-NEXT:    and a6, s2, a1
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 260(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 21
+; RV32I-NEXT:    sw a1, 348(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a1
+; RV32I-NEXT:    lui a1, 1024
+; RV32I-NEXT:    and a6, s2, a1
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a7, a5, -1
+; RV32I-NEXT:    sw a7, 252(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a1, 2048
+; RV32I-NEXT:    and a5, s2, a1
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    lui a1, 4096
+; RV32I-NEXT:    and a6, s2, a1
+; RV32I-NEXT:    addi t0, a5, -1
+; RV32I-NEXT:    sw t0, 248(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi t4, a5, -1
+; RV32I-NEXT:    sw t4, 244(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 22
+; RV32I-NEXT:    sw a1, 336(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a7, a1
+; RV32I-NEXT:    slli a1, a2, 23
+; RV32I-NEXT:    sw a1, 332(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, t0, a1
+; RV32I-NEXT:    slli a1, a2, 24
+; RV32I-NEXT:    sw a1, 324(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    xor a5, a5, a6
-; RV32I-NEXT:    seqz a6, a7
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 124(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 13
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    and a7, a2, s8
+; RV32I-NEXT:    and a6, t4, a1
 ; RV32I-NEXT:    xor a5, a5, a6
-; RV32I-NEXT:    seqz a6, a7
-; RV32I-NEXT:    xor a4, a4, a5
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 120(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a5, a1, 14
-; RV32I-NEXT:    lui t0, 8
-; RV32I-NEXT:    and a6, a2, t0
-; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    lui a1, 8192
+; RV32I-NEXT:    and a6, s2, a1
 ; RV32I-NEXT:    seqz a6, a6
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 116(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 15
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    lui t6, 16
-; RV32I-NEXT:    and a7, a2, t6
-; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    and a7, s2, ra
+; RV32I-NEXT:    addi t0, a6, -1
+; RV32I-NEXT:    sw t0, 240(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    seqz a6, a7
 ; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 112(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 16
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    and a7, a2, s2
+; RV32I-NEXT:    sw a7, 236(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, a2, 25
+; RV32I-NEXT:    sw a1, 316(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, t0, a1
+; RV32I-NEXT:    slli a1, a2, 26
+; RV32I-NEXT:    sw a1, 312(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    xor a5, a5, a6
-; RV32I-NEXT:    seqz a6, a7
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 108(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 17
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    lui t1, 64
-; RV32I-NEXT:    and a7, a2, t1
+; RV32I-NEXT:    and a6, a7, a1
+; RV32I-NEXT:    xor a4, a0, a4
+; RV32I-NEXT:    xor a0, a5, a6
+; RV32I-NEXT:    srli a5, a3, 8
+; RV32I-NEXT:    sw s10, 168(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, a3, s10
+; RV32I-NEXT:    and a5, a5, s10
+; RV32I-NEXT:    slli a6, a6, 8
+; RV32I-NEXT:    srli a7, a3, 24
+; RV32I-NEXT:    slli t0, a3, 24
+; RV32I-NEXT:    or a5, a5, a7
+; RV32I-NEXT:    or a6, t0, a6
+; RV32I-NEXT:    or a5, a6, a5
+; RV32I-NEXT:    lui t5, 32768
+; RV32I-NEXT:    and a6, s2, t5
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    srli a7, a5, 4
+; RV32I-NEXT:    addi t0, a6, -1
+; RV32I-NEXT:    sw t0, 224(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, a7, s5
+; RV32I-NEXT:    and a5, a5, s5
+; RV32I-NEXT:    slli a1, a2, 27
+; RV32I-NEXT:    sw a1, 268(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, a5, 4
+; RV32I-NEXT:    and a7, t0, a1
+; RV32I-NEXT:    xor a0, a0, a7
+; RV32I-NEXT:    or a5, a6, a5
+; RV32I-NEXT:    srli a6, a5, 2
+; RV32I-NEXT:    and a5, a5, s7
+; RV32I-NEXT:    and a6, a6, s7
+; RV32I-NEXT:    slli a5, a5, 2
+; RV32I-NEXT:    or a5, a6, a5
+; RV32I-NEXT:    lui s10, 65536
+; RV32I-NEXT:    and a6, s2, s10
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    srli a7, a5, 1
+; RV32I-NEXT:    addi t0, a6, -1
+; RV32I-NEXT:    sw t0, 212(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s0, 488(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, a7, s0
+; RV32I-NEXT:    and a5, a5, s0
+; RV32I-NEXT:    slli a1, a2, 28
+; RV32I-NEXT:    sw a1, 256(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t4, a5, 1
+; RV32I-NEXT:    and a5, t0, a1
+; RV32I-NEXT:    xor a5, a0, a5
+; RV32I-NEXT:    or a0, a6, t4
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    sw a4, 132(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a4, a0, 2
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    andi a5, a0, 1
+; RV32I-NEXT:    addi a1, a4, -1
+; RV32I-NEXT:    sw a1, 232(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 228(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t1, 1
+; RV32I-NEXT:    sw a4, 156(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a4, a1, a4
+; RV32I-NEXT:    and a5, a5, t1
+; RV32I-NEXT:    xor a4, a5, a4
+; RV32I-NEXT:    andi a5, a0, 4
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    andi a6, a0, 8
+; RV32I-NEXT:    addi a1, a5, -1
+; RV32I-NEXT:    sw a1, 220(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a6, a5, -1
+; RV32I-NEXT:    sw a6, 216(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a5, t1, 2
+; RV32I-NEXT:    sw a5, 144(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a1, a5
+; RV32I-NEXT:    slli a1, t1, 3
+; RV32I-NEXT:    sw a1, 136(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, a6, a1
+; RV32I-NEXT:    andi a7, a0, 16
 ; RV32I-NEXT:    xor a5, a5, a6
 ; RV32I-NEXT:    seqz a6, a7
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 104(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 18
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    lui t2, 128
-; RV32I-NEXT:    and a7, a2, t2
-; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    addi a1, a6, -1
+; RV32I-NEXT:    sw a1, 200(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a6, a0, 32
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    andi a7, a0, 64
+; RV32I-NEXT:    addi t0, a6, -1
+; RV32I-NEXT:    sw t0, 196(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    seqz a6, a7
+; RV32I-NEXT:    addi s0, a6, -1
+; RV32I-NEXT:    sw s0, 192(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, t1, 4
+; RV32I-NEXT:    sw a6, 112(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, a1, a6
+; RV32I-NEXT:    slli a1, t1, 5
+; RV32I-NEXT:    sw a1, 108(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a7, t0, a1
+; RV32I-NEXT:    slli a1, t1, 6
+; RV32I-NEXT:    sw a1, 104(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, s0, a1
 ; RV32I-NEXT:    xor a4, a4, a5
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 100(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a5, a1, 19
-; RV32I-NEXT:    lui t3, 256
-; RV32I-NEXT:    and a6, a2, t3
-; RV32I-NEXT:    and a5, a7, a5
+; RV32I-NEXT:    xor a5, a6, a7
+; RV32I-NEXT:    andi a6, a0, 128
+; RV32I-NEXT:    andi a7, a0, 256
 ; RV32I-NEXT:    seqz a6, a6
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 96(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 20
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    lui t4, 512
-; RV32I-NEXT:    and a7, a2, t4
-; RV32I-NEXT:    xor a5, a5, a6
-; RV32I-NEXT:    seqz a6, a7
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 92(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 21
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    lui t5, 1024
-; RV32I-NEXT:    and a7, a2, t5
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    sw a6, 188(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    sw a7, 184(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t1, 7
+; RV32I-NEXT:    sw a1, 92(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t0, t1, 8
+; RV32I-NEXT:    sw t0, 88(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, a6, a1
+; RV32I-NEXT:    and a7, a7, t0
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    andi a7, a0, 512
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    andi t0, a0, 1024
+; RV32I-NEXT:    addi a1, a7, -1
+; RV32I-NEXT:    sw a1, 180(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a7, t0
+; RV32I-NEXT:    addi t0, a7, -1
+; RV32I-NEXT:    sw t0, 176(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a7, t1, 9
+; RV32I-NEXT:    sw a7, 76(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a7, a1, a7
+; RV32I-NEXT:    slli a1, t1, 10
+; RV32I-NEXT:    sw a1, 72(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    and a7, t0, a1
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    xor a5, a6, a7
+; RV32I-NEXT:    lui a1, 131072
+; RV32I-NEXT:    and a6, s2, a1
+; RV32I-NEXT:    lui ra, 262144
+; RV32I-NEXT:    and a7, s2, ra
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    addi t0, a6, -1
+; RV32I-NEXT:    sw t0, 128(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi s0, a7, -1
+; RV32I-NEXT:    sw s0, 124(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a6, a2, 29
+; RV32I-NEXT:    sw a6, 208(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a7, a2, 30
+; RV32I-NEXT:    sw a7, 204(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a6, t0, a6
+; RV32I-NEXT:    and a7, s0, a7
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a0, s3
+; RV32I-NEXT:    and a7, a0, s1
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 164(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    sw a7, 160(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t0, t1, 11
+; RV32I-NEXT:    sw t0, 44(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s0, t1, 12
+; RV32I-NEXT:    sw s0, 40(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, t0
+; RV32I-NEXT:    and a7, a7, s0
+; RV32I-NEXT:    xor a5, a5, a7
+; RV32I-NEXT:    and a7, a0, s8
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    and t0, a0, t3
+; RV32I-NEXT:    addi t3, a7, -1
+; RV32I-NEXT:    sw t3, 152(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a7, t0
+; RV32I-NEXT:    addi t0, a7, -1
+; RV32I-NEXT:    sw t0, 148(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a7, t1, 13
+; RV32I-NEXT:    sw a7, 28(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a7, t3, a7
+; RV32I-NEXT:    slli t3, t1, 14
+; RV32I-NEXT:    sw t3, 24(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a5, a5, a7
+; RV32I-NEXT:    and a7, t0, t3
+; RV32I-NEXT:    xor a5, a5, a7
+; RV32I-NEXT:    lw a7, 464(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    srli a7, a7, 31
+; RV32I-NEXT:    seqz a7, a7
+; RV32I-NEXT:    lui t0, 8
+; RV32I-NEXT:    and t0, a0, t0
+; RV32I-NEXT:    addi t3, a7, -1
+; RV32I-NEXT:    sw t3, 56(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a7, t0
+; RV32I-NEXT:    addi a7, a7, -1
+; RV32I-NEXT:    sw a7, 140(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s0, t1, 15
+; RV32I-NEXT:    sw s0, 20(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t0, a2, 31
+; RV32I-NEXT:    sw t0, 172(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a7, a7, s0
+; RV32I-NEXT:    xor a5, a5, a7
+; RV32I-NEXT:    and a7, t3, t0
+; RV32I-NEXT:    xor a6, a6, a7
+; RV32I-NEXT:    sw a6, 464(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    and a5, a0, s4
+; RV32I-NEXT:    and a6, a0, s11
+; RV32I-NEXT:    seqz a5, a5
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 120(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a6, a6, -1
+; RV32I-NEXT:    sw a6, 116(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a7, t1, 16
+; RV32I-NEXT:    sw a7, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t0, t1, 17
+; RV32I-NEXT:    sw t0, 8(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a5, a5, a7
+; RV32I-NEXT:    and a6, a6, t0
 ; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lui a6, 64
+; RV32I-NEXT:    and a6, a0, a6
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    and a7, a0, s9
+; RV32I-NEXT:    addi t0, a6, -1
+; RV32I-NEXT:    sw t0, 100(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    seqz a6, a7
 ; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 88(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 22
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    lui s0, 2048
-; RV32I-NEXT:    and a7, a2, s0
+; RV32I-NEXT:    sw a7, 96(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s11, t1, 18
+; RV32I-NEXT:    and a6, t0, s11
+; RV32I-NEXT:    slli t0, t1, 19
+; RV32I-NEXT:    sw t0, 4(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    xor a5, a5, a6
-; RV32I-NEXT:    seqz a6, a7
-; RV32I-NEXT:    addi a7, a6, -1
-; RV32I-NEXT:    sw a7, 84(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 23
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    and a7, a2, a0
+; RV32I-NEXT:    and a6, a7, t0
 ; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lui a6, 256
+; RV32I-NEXT:    and a6, a0, a6
+; RV32I-NEXT:    seqz a6, a6
+; RV32I-NEXT:    lui a7, 512
+; RV32I-NEXT:    and a7, a0, a7
+; RV32I-NEXT:    addi t0, a6, -1
+; RV32I-NEXT:    sw t0, 84(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    seqz a6, a7
 ; RV32I-NEXT:    addi a7, a6, -1
 ; RV32I-NEXT:    sw a7, 80(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a1, 24
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    and a7, a2, s9
+; RV32I-NEXT:    slli s9, t1, 20
+; RV32I-NEXT:    and a6, t0, s9
+; RV32I-NEXT:    slli s8, t1, 21
 ; RV32I-NEXT:    xor a5, a5, a6
-; RV32I-NEXT:    seqz a6, a7
-; RV32I-NEXT:    xor a4, a4, a5
-; RV32I-NEXT:    sw a4, 48(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    addi a6, a6, -1
-; RV32I-NEXT:    sw a6, 76(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a4, a1, 25
-; RV32I-NEXT:    and a5, a2, s10
-; RV32I-NEXT:    and a4, a6, a4
-; RV32I-NEXT:    seqz a5, a5
-; RV32I-NEXT:    addi a6, a5, -1
-; RV32I-NEXT:    sw a6, 72(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a5, a1, 26
-; RV32I-NEXT:    and a5, a6, a5
-; RV32I-NEXT:    and a6, a2, s11
-; RV32I-NEXT:    xor a4, a4, a5
-; RV32I-NEXT:    seqz a5, a6
-; RV32I-NEXT:    addi a6, a5, -1
+; RV32I-NEXT:    and a6, a7, s8
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lui a6, 1024
+; RV32I-NEXT:    and a6, a0, a6
+; RV32I-NEXT:    xor t3, a4, a5
+; RV32I-NEXT:    seqz a4, a6
+; RV32I-NEXT:    addi a6, a4, -1
 ; RV32I-NEXT:    sw a6, 68(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a5, a1, 27
-; RV32I-NEXT:    and a5, a6, a5
-; RV32I-NEXT:    and a6, a2, s4
+; RV32I-NEXT:    lui a4, 2048
+; RV32I-NEXT:    and a4, a0, a4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    lui a5, 4096
+; RV32I-NEXT:    and a5, a0, a5
+; RV32I-NEXT:    addi a7, a4, -1
+; RV32I-NEXT:    sw a7, 64(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a4, a5
+; RV32I-NEXT:    slli s3, t1, 22
+; RV32I-NEXT:    slli s4, t1, 23
+; RV32I-NEXT:    and a5, a6, s3
+; RV32I-NEXT:    and a6, a7, s4
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    sw a4, 60(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw s2, 468(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s2
+; RV32I-NEXT:    lui a6, 8192
+; RV32I-NEXT:    and a6, a0, a6
+; RV32I-NEXT:    xor a4, a5, a4
+; RV32I-NEXT:    seqz a5, a6
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 52(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s1, t1, 25
+; RV32I-NEXT:    and a5, a5, s1
+; RV32I-NEXT:    lui a6, 16384
+; RV32I-NEXT:    and a6, a0, a6
 ; RV32I-NEXT:    xor a4, a4, a5
 ; RV32I-NEXT:    seqz a5, a6
-; RV32I-NEXT:    addi a6, a5, -1
-; RV32I-NEXT:    sw a6, 64(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a5, a1, 28
-; RV32I-NEXT:    and a5, a6, a5
-; RV32I-NEXT:    and a6, a2, s1
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 48(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli s0, t1, 26
+; RV32I-NEXT:    and a5, a5, s0
+; RV32I-NEXT:    and a6, a0, t5
 ; RV32I-NEXT:    xor a4, a4, a5
 ; RV32I-NEXT:    seqz a5, a6
-; RV32I-NEXT:    addi a6, a5, -1
-; RV32I-NEXT:    sw a6, 60(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a5, a1, 29
-; RV32I-NEXT:    and a5, a6, a5
-; RV32I-NEXT:    and a6, a2, ra
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 36(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t5, t1, 27
+; RV32I-NEXT:    and a5, a5, t5
+; RV32I-NEXT:    and a6, a0, s10
 ; RV32I-NEXT:    xor a4, a4, a5
 ; RV32I-NEXT:    seqz a5, a6
-; RV32I-NEXT:    addi a6, a5, -1
-; RV32I-NEXT:    sw a6, 56(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a5, a1, 30
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 32(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli t0, t1, 28
+; RV32I-NEXT:    and a5, a5, t0
+; RV32I-NEXT:    and s10, a0, a1
+; RV32I-NEXT:    xor a4, a4, a5
+; RV32I-NEXT:    seqz a5, s10
+; RV32I-NEXT:    addi a5, a5, -1
+; RV32I-NEXT:    sw a5, 16(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a0, a0, ra
+; RV32I-NEXT:    seqz s10, a0
+; RV32I-NEXT:    srli a0, t4, 31
+; RV32I-NEXT:    addi s10, s10, -1
+; RV32I-NEXT:    seqz ra, a0
+; RV32I-NEXT:    addi ra, ra, -1
+; RV32I-NEXT:    slli a7, t1, 29
+; RV32I-NEXT:    and a1, a5, a7
+; RV32I-NEXT:    slli a6, t1, 30
+; RV32I-NEXT:    and a0, s10, a6
+; RV32I-NEXT:    slli a5, t1, 31
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    and a1, ra, a5
+; RV32I-NEXT:    xor a4, t3, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 132(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 464(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    xor a1, a1, t3
+; RV32I-NEXT:    xor a0, a4, a0
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    lw a1, 372(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    srli a1, a1, 1
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    srli a1, a0, 8
+; RV32I-NEXT:    lw t4, 168(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, t4
+; RV32I-NEXT:    srli a4, a0, 24
+; RV32I-NEXT:    and t3, a0, t4
+; RV32I-NEXT:    slli a0, a0, 24
+; RV32I-NEXT:    slli t3, t3, 8
+; RV32I-NEXT:    or a1, a1, a4
+; RV32I-NEXT:    or a0, a0, t3
+; RV32I-NEXT:    or a0, a0, a1
+; RV32I-NEXT:    srli a1, a0, 4
+; RV32I-NEXT:    and a0, a0, s5
+; RV32I-NEXT:    and a1, a1, s5
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    or a0, a1, a0
+; RV32I-NEXT:    srli a1, a0, 2
+; RV32I-NEXT:    and a0, a0, s7
+; RV32I-NEXT:    and a1, a1, s7
+; RV32I-NEXT:    slli a0, a0, 2
+; RV32I-NEXT:    or a0, a1, a0
+; RV32I-NEXT:    srli a1, a0, 1
+; RV32I-NEXT:    lw a4, 488(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a0, a4
+; RV32I-NEXT:    lw a4, 484(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a4
+; RV32I-NEXT:    slli a0, a0, 1
+; RV32I-NEXT:    or a0, a1, a0
+; RV32I-NEXT:    sw a0, 464(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a0, 400(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 156(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a0, a1
+; RV32I-NEXT:    lw a1, 396(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, t1
+; RV32I-NEXT:    lw a4, 384(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t1, 144(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t1
+; RV32I-NEXT:    lw t1, 380(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 136(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    xor a1, a4, t1
+; RV32I-NEXT:    lw a4, 360(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t1, 112(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t1
+; RV32I-NEXT:    lw t1, 356(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 108(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 352(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 104(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, a4, t1
+; RV32I-NEXT:    lw a4, 344(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t1, 92(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t1
+; RV32I-NEXT:    lw t1, 340(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 88(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 328(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 76(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 320(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 72(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, a4, t1
+; RV32I-NEXT:    lw a4, 308(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t1, 44(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t1
+; RV32I-NEXT:    lw t1, 304(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 40(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 300(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 28(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 296(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 24(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 292(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 20(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, a4, t1
+; RV32I-NEXT:    lw a4, 284(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t1, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, t1
+; RV32I-NEXT:    lw t1, 280(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 8(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 276(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, s11
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 272(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t3, 4(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t3
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 264(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, s9
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 260(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, s8
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, a4, t1
+; RV32I-NEXT:    lw a4, 252(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, s3
+; RV32I-NEXT:    lw t1, 248(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, s4
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 244(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, s2
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 240(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, s1
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 236(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, s0
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 224(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t5
+; RV32I-NEXT:    xor a4, a4, t1
+; RV32I-NEXT:    lw t1, 212(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t0, t1, t0
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, a4, t0
+; RV32I-NEXT:    lw a4, 128(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a4, a7
+; RV32I-NEXT:    lw a7, 124(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a4, a4, a6
+; RV32I-NEXT:    lw a6, 56(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    and a5, a6, a5
-; RV32I-NEXT:    srli a2, a2, 31
+; RV32I-NEXT:    xor a0, a0, a1
 ; RV32I-NEXT:    xor a4, a4, a5
-; RV32I-NEXT:    seqz a2, a2
-; RV32I-NEXT:    slli a1, a1, 31
-; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    sw a2, 52(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a1, 460(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 232(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a5, a1
+; RV32I-NEXT:    lw a5, 228(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a5, a2
+; RV32I-NEXT:    lw a5, 456(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 220(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    lw a6, 452(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 216(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    xor a2, a5, a6
+; RV32I-NEXT:    lw a5, 448(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 200(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    lw a6, 444(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 196(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 440(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 192(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a5, a6
+; RV32I-NEXT:    lw a5, 436(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 188(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    lw a6, 432(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 184(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 428(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 180(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 424(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 176(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a5, a6
+; RV32I-NEXT:    lw a5, 420(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 164(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    lw a6, 416(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 160(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 412(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 152(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 408(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 148(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 404(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 140(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a5, a6
+; RV32I-NEXT:    lw a5, 392(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a6, 120(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a6, a5
+; RV32I-NEXT:    lw a6, 388(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 116(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 376(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 100(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 368(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 96(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 364(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 84(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 348(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a7, 80(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a7, a6
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    xor a2, a5, a6
+; RV32I-NEXT:    xor a0, a0, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    lw a2, 336(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a4, 68(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    lw a4, 332(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 64(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    lw a4, 324(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 60(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    lw a4, 316(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 52(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    lw a4, 312(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 48(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    lw a4, 268(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 36(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    lw a4, 256(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 32(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a4, a5, a4
+; RV32I-NEXT:    xor a2, a2, a4
+; RV32I-NEXT:    srli a4, a0, 8
+; RV32I-NEXT:    and a4, a4, t4
+; RV32I-NEXT:    srli a5, a0, 24
+; RV32I-NEXT:    or a4, a4, a5
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    lw a2, 208(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 16(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a5, a2
+; RV32I-NEXT:    lw a5, 204(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, s10, a5
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    lw a5, 172(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, ra, a5
+; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    and a5, a0, t4
+; RV32I-NEXT:    slli a0, a0, 24
+; RV32I-NEXT:    slli a5, a5, 8
+; RV32I-NEXT:    or a0, a0, a5
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    or a0, a0, a4
+; RV32I-NEXT:    srli a2, a1, 8
+; RV32I-NEXT:    and a2, a2, t4
+; RV32I-NEXT:    srli a4, a1, 24
+; RV32I-NEXT:    or a2, a2, a4
+; RV32I-NEXT:    and a4, a1, t4
+; RV32I-NEXT:    slli a1, a1, 24
+; RV32I-NEXT:    slli a4, a4, 8
+; RV32I-NEXT:    or a1, a1, a4
+; RV32I-NEXT:    srli a4, a0, 4
+; RV32I-NEXT:    or a1, a1, a2
+; RV32I-NEXT:    and a2, a4, s5
+; RV32I-NEXT:    and a0, a0, s5
+; RV32I-NEXT:    srli a4, a1, 4
+; RV32I-NEXT:    and a4, a4, s5
+; RV32I-NEXT:    and a1, a1, s5
+; RV32I-NEXT:    slli a0, a0, 4
+; RV32I-NEXT:    slli a1, a1, 4
+; RV32I-NEXT:    or a0, a2, a0
+; RV32I-NEXT:    or a1, a4, a1
+; RV32I-NEXT:    srli a2, a0, 2
+; RV32I-NEXT:    and a0, a0, s7
+; RV32I-NEXT:    and a2, a2, s7
+; RV32I-NEXT:    slli a0, a0, 2
+; RV32I-NEXT:    or a0, a2, a0
+; RV32I-NEXT:    sw a0, 468(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a0, a1, 2
+; RV32I-NEXT:    and a0, a0, s7
+; RV32I-NEXT:    and a1, a1, s7
+; RV32I-NEXT:    slli a1, a1, 2
+; RV32I-NEXT:    andi a2, a3, 2
+; RV32I-NEXT:    or a4, a0, a1
+; RV32I-NEXT:    sw a4, 444(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    seqz a0, a2
+; RV32I-NEXT:    addi a2, a0, -1
+; RV32I-NEXT:    sw a2, 460(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a0, a3, 1
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    slli a1, t2, 1
 ; RV32I-NEXT:    and a1, a2, a1
-; RV32I-NEXT:    andi a2, s3, 2
+; RV32I-NEXT:    addi a2, a0, -1
+; RV32I-NEXT:    sw a2, 456(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    srli a0, a4, 1
+; RV32I-NEXT:    and a2, a2, t2
+; RV32I-NEXT:    lw a4, 484(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a0, a4
+; RV32I-NEXT:    sw a0, 448(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    andi a0, a3, 4
+; RV32I-NEXT:    andi a2, a3, 8
+; RV32I-NEXT:    seqz a0, a0
 ; RV32I-NEXT:    seqz a2, a2
-; RV32I-NEXT:    andi a5, s3, 1
-; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    seqz a5, a5
-; RV32I-NEXT:    addi a5, a5, -1
-; RV32I-NEXT:    slli a6, a3, 1
-; RV32I-NEXT:    sw a6, 44(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    and a2, a2, a6
-; RV32I-NEXT:    and a5, a5, a3
-; RV32I-NEXT:    xor a1, a4, a1
-; RV32I-NEXT:    sw a1, 40(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    xor a2, a5, a2
-; RV32I-NEXT:    andi a1, s3, 4
-; RV32I-NEXT:    andi a4, s3, 8
+; RV32I-NEXT:    addi a4, a0, -1
+; RV32I-NEXT:    sw a4, 484(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a5, a2, -1
+; RV32I-NEXT:    sw a5, 452(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a0, t2, 2
+; RV32I-NEXT:    slli a2, t2, 3
+; RV32I-NEXT:    and a0, a4, a0
+; RV32I-NEXT:    and a2, a5, a2
+; RV32I-NEXT:    xor a0, a0, a2
+; RV32I-NEXT:    andi a2, a3, 16
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 440(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a1, a3, 32
 ; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    slli a2, t2, 4
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    addi a5, a1, -1
+; RV32I-NEXT:    sw a5, 436(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t2, 5
+; RV32I-NEXT:    andi a4, a3, 64
+; RV32I-NEXT:    and a1, a5, a1
 ; RV32I-NEXT:    seqz a4, a4
-; RV32I-NEXT:    addi a1, a1, -1
-; RV32I-NEXT:    addi a4, a4, -1
-; RV32I-NEXT:    slli a5, a3, 2
-; RV32I-NEXT:    sw a5, 36(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    slli a6, a3, 3
-; RV32I-NEXT:    sw a6, 32(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    and a1, a1, a5
-; RV32I-NEXT:    and a4, a4, a6
-; RV32I-NEXT:    xor a1, a1, a4
-; RV32I-NEXT:    andi a4, s3, 16
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 432(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t2, 6
 ; RV32I-NEXT:    xor a1, a2, a1
-; RV32I-NEXT:    seqz a2, a4
-; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    andi a4, s3, 32
+; RV32I-NEXT:    and a2, a5, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    andi a2, a3, 128
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 428(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a1, a3, 256
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    slli a2, t2, 7
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    addi a5, a1, -1
+; RV32I-NEXT:    sw a5, 424(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t2, 8
+; RV32I-NEXT:    andi a4, a3, 512
+; RV32I-NEXT:    and a1, a5, a1
 ; RV32I-NEXT:    seqz a4, a4
-; RV32I-NEXT:    slli a5, a3, 4
-; RV32I-NEXT:    sw a5, 28(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    and a2, a2, a5
-; RV32I-NEXT:    addi a4, a4, -1
-; RV32I-NEXT:    slli a6, a3, 5
-; RV32I-NEXT:    sw a6, 24(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    andi a5, s3, 64
-; RV32I-NEXT:    and a4, a4, a6
-; RV32I-NEXT:    seqz a5, a5
-; RV32I-NEXT:    addi a5, a5, -1
-; RV32I-NEXT:    slli a6, a3, 6
-; RV32I-NEXT:    sw a6, 20(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    xor a2, a2, a4
-; RV32I-NEXT:    and a4, a5, a6
-; RV32I-NEXT:    xor a2, a2, a4
-; RV32I-NEXT:    andi a4, s3, 128
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 420(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 9
+; RV32I-NEXT:    andi a4, a3, 1024
+; RV32I-NEXT:    and a2, a5, a2
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 416(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t2, 10
 ; RV32I-NEXT:    xor a1, a1, a2
-; RV32I-NEXT:    seqz a2, a4
-; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    andi a4, s3, 256
+; RV32I-NEXT:    and a2, a5, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    lw s5, 288(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a3, s5
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 412(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui s2, 1
+; RV32I-NEXT:    and a1, a3, s2
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    slli a2, t2, 11
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    addi a5, a1, -1
+; RV32I-NEXT:    sw a5, 408(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t2, 12
+; RV32I-NEXT:    lui s7, 2
+; RV32I-NEXT:    and a4, a3, s7
+; RV32I-NEXT:    and a1, a5, a1
 ; RV32I-NEXT:    seqz a4, a4
-; RV32I-NEXT:    slli a5, a3, 7
-; RV32I-NEXT:    sw a5, 16(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    and a2, a2, a5
-; RV32I-NEXT:    addi a4, a4, -1
-; RV32I-NEXT:    slli a6, a3, 8
-; RV32I-NEXT:    sw a6, 12(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    andi a5, s3, 512
-; RV32I-NEXT:    and a4, a4, a6
-; RV32I-NEXT:    seqz a5, a5
-; RV32I-NEXT:    xor a2, a2, a4
-; RV32I-NEXT:    addi a5, a5, -1
-; RV32I-NEXT:    slli a6, a3, 9
-; RV32I-NEXT:    sw a6, 8(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    andi a4, s3, 1024
-; RV32I-NEXT:    and a5, a5, a6
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 404(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t2, 13
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    and a2, a5, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    lui s4, 4
+; RV32I-NEXT:    and a2, a3, s4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 400(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui s9, 8
+; RV32I-NEXT:    and a1, a3, s9
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    slli a2, t2, 14
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    addi a5, a1, -1
+; RV32I-NEXT:    sw a5, 396(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t2, 15
+; RV32I-NEXT:    lui s8, 16
+; RV32I-NEXT:    and a4, a3, s8
+; RV32I-NEXT:    and a1, a5, a1
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 392(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 16
+; RV32I-NEXT:    lui s3, 32
+; RV32I-NEXT:    and a4, a3, s3
+; RV32I-NEXT:    and a2, a5, a2
 ; RV32I-NEXT:    seqz a4, a4
-; RV32I-NEXT:    addi a4, a4, -1
-; RV32I-NEXT:    slli a6, a3, 10
-; RV32I-NEXT:    sw a6, 4(sp) # 4-byte Folded Spill
-; RV32I-NEXT:    xor a2, a2, a5
-; RV32I-NEXT:    and a4, a4, a6
-; RV32I-NEXT:    xor a2, a2, a4
-; RV32I-NEXT:    and a4, s3, s5
 ; RV32I-NEXT:    xor a1, a1, a2
-; RV32I-NEXT:    seqz a2, a4
-; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    and a4, s3, s6
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 388(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 17
+; RV32I-NEXT:    lui a7, 64
+; RV32I-NEXT:    and a4, a3, a7
+; RV32I-NEXT:    and a2, a5, a2
 ; RV32I-NEXT:    seqz a4, a4
-; RV32I-NEXT:    slli s11, a3, 11
-; RV32I-NEXT:    and a2, a2, s11
-; RV32I-NEXT:    addi a4, a4, -1
-; RV32I-NEXT:    and a5, s3, s7
-; RV32I-NEXT:    slli s10, a3, 12
-; RV32I-NEXT:    and a4, a4, s10
-; RV32I-NEXT:    seqz a5, a5
-; RV32I-NEXT:    addi a5, a5, -1
-; RV32I-NEXT:    slli s9, a3, 13
-; RV32I-NEXT:    xor a2, a2, a4
-; RV32I-NEXT:    and a4, a5, s9
-; RV32I-NEXT:    xor a2, a2, a4
-; RV32I-NEXT:    and a4, s3, s8
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 384(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a4, t2, 18
 ; RV32I-NEXT:    xor a1, a1, a2
-; RV32I-NEXT:    seqz a2, a4
-; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    and a4, s3, t0
+; RV32I-NEXT:    and a2, a5, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    lui t0, 128
+; RV32I-NEXT:    and a2, a3, t0
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    addi a4, a1, -1
+; RV32I-NEXT:    sw a4, 380(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui t1, 256
+; RV32I-NEXT:    and a1, a3, t1
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    slli a2, t2, 19
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    addi a5, a1, -1
+; RV32I-NEXT:    sw a5, 376(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t2, 20
+; RV32I-NEXT:    lui t3, 512
+; RV32I-NEXT:    and a4, a3, t3
+; RV32I-NEXT:    and a1, a5, a1
 ; RV32I-NEXT:    seqz a4, a4
-; RV32I-NEXT:    slli s8, a3, 14
-; RV32I-NEXT:    and a2, a2, s8
-; RV32I-NEXT:    addi a4, a4, -1
-; RV32I-NEXT:    and a5, s3, t6
-; RV32I-NEXT:    slli s7, a3, 15
-; RV32I-NEXT:    and a4, a4, s7
-; RV32I-NEXT:    seqz a5, a5
-; RV32I-NEXT:    xor a2, a2, a4
-; RV32I-NEXT:    addi a5, a5, -1
-; RV32I-NEXT:    and a4, s3, s2
-; RV32I-NEXT:    slli s6, a3, 16
-; RV32I-NEXT:    and a5, a5, s6
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 372(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 21
+; RV32I-NEXT:    lui t4, 1024
+; RV32I-NEXT:    and a4, a3, t4
+; RV32I-NEXT:    and a2, a5, a2
 ; RV32I-NEXT:    seqz a4, a4
-; RV32I-NEXT:    xor a2, a2, a5
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 368(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 22
+; RV32I-NEXT:    lui t5, 2048
+; RV32I-NEXT:    and a4, a3, t5
+; RV32I-NEXT:    and a2, a5, a2
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    addi a5, a4, -1
+; RV32I-NEXT:    sw a5, 364(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 23
+; RV32I-NEXT:    lui s0, 4096
+; RV32I-NEXT:    and a4, a3, s0
+; RV32I-NEXT:    and a2, a5, a2
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a1, a1, a2
 ; RV32I-NEXT:    addi a4, a4, -1
-; RV32I-NEXT:    and a5, s3, t1
-; RV32I-NEXT:    slli s5, a3, 17
-; RV32I-NEXT:    and a4, a4, s5
-; RV32I-NEXT:    seqz a5, a5
-; RV32I-NEXT:    addi a5, a5, -1
-; RV32I-NEXT:    slli s1, a3, 18
-; RV32I-NEXT:    xor a2, a2, a4
-; RV32I-NEXT:    and a4, a5, s1
-; RV32I-NEXT:    xor a2, a2, a4
-; RV32I-NEXT:    and a4, s3, t2
-; RV32I-NEXT:    xor s2, a1, a2
-; RV32I-NEXT:    seqz a1, a4
+; RV32I-NEXT:    sw a4, 360(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lw a2, 476(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    lui s1, 8192
+; RV32I-NEXT:    and a4, a3, s1
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    addi a4, a2, -1
+; RV32I-NEXT:    sw a4, 476(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a1, t2, 25
+; RV32I-NEXT:    lui a5, 16384
+; RV32I-NEXT:    and a2, a3, a5
+; RV32I-NEXT:    and a1, a4, a1
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    addi a4, a2, -1
+; RV32I-NEXT:    sw a4, 356(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 26
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    lui a6, 32768
+; RV32I-NEXT:    and a4, a3, a6
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a4
+; RV32I-NEXT:    addi a4, a2, -1
+; RV32I-NEXT:    sw a4, 352(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 27
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    lui a4, 65536
+; RV32I-NEXT:    and a4, a3, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a4
+; RV32I-NEXT:    addi a4, a2, -1
+; RV32I-NEXT:    sw a4, 348(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 28
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    lui a4, 131072
+; RV32I-NEXT:    and a4, a3, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a4
+; RV32I-NEXT:    addi a4, a2, -1
+; RV32I-NEXT:    sw a4, 344(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 29
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    lui a4, 262144
+; RV32I-NEXT:    and a4, a3, a4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a4
+; RV32I-NEXT:    addi a4, a2, -1
+; RV32I-NEXT:    sw a4, 340(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    slli a2, t2, 30
+; RV32I-NEXT:    and a2, a4, a2
+; RV32I-NEXT:    srli a3, a3, 31
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    slli t2, t2, 31
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    sw a2, 336(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, t2
+; RV32I-NEXT:    andi a3, t6, 2
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    sw a0, 328(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    slli a1, s6, 1
+; RV32I-NEXT:    sw a1, 332(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a0, t6, 1
+; RV32I-NEXT:    and a1, a2, a1
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    addi a0, a0, -1
+; RV32I-NEXT:    andi a2, t6, 4
+; RV32I-NEXT:    and a0, a0, s6
+; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    slli a3, s6, 2
+; RV32I-NEXT:    sw a3, 324(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a1, t6, 8
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    slli a3, s6, 3
+; RV32I-NEXT:    sw a3, 320(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a1, a3
+; RV32I-NEXT:    andi a3, t6, 16
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    slli a3, s6, 4
+; RV32I-NEXT:    sw a3, 316(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a1, t6, 32
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    slli a3, s6, 5
+; RV32I-NEXT:    sw a3, 312(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a1, a3
+; RV32I-NEXT:    andi a3, t6, 64
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    slli a3, s6, 6
+; RV32I-NEXT:    sw a3, 308(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    andi a3, t6, 128
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    slli a3, s6, 7
+; RV32I-NEXT:    sw a3, 304(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    andi a1, t6, 256
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    slli a3, s6, 8
+; RV32I-NEXT:    sw a3, 300(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a1, a3
+; RV32I-NEXT:    andi a3, t6, 512
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    slli a3, s6, 9
+; RV32I-NEXT:    sw a3, 296(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    andi a3, t6, 1024
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    slli a3, s6, 10
+; RV32I-NEXT:    sw a3, 292(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    and a3, t6, s5
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    and a1, t6, s2
+; RV32I-NEXT:    slli a3, s6, 11
+; RV32I-NEXT:    sw a3, 288(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a2, a2, a3
+; RV32I-NEXT:    seqz a1, a1
 ; RV32I-NEXT:    addi a1, a1, -1
-; RV32I-NEXT:    and a2, s3, t3
-; RV32I-NEXT:    seqz a2, a2
-; RV32I-NEXT:    slli s4, a3, 19
-; RV32I-NEXT:    and a1, a1, s4
+; RV32I-NEXT:    slli a3, s6, 12
+; RV32I-NEXT:    sw a3, 284(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    and a1, a1, a3
+; RV32I-NEXT:    and a3, t6, s7
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    seqz a2, a3
 ; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    and a4, s3, t4
-; RV32I-NEXT:    slli t6, a3, 20
-; RV32I-NEXT:    and a2, a2, t6
-; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    slli ra, s6, 13
+; RV32I-NEXT:    and a2, a2, ra
+; RV32I-NEXT:    and a3, t6, s4
 ; RV32I-NEXT:    xor a1, a1, a2
-; RV32I-NEXT:    addi a4, a4, -1
-; RV32I-NEXT:    and a2, s3, t5
-; RV32I-NEXT:    slli t5, a3, 21
-; RV32I-NEXT:    and a4, a4, t5
-; RV32I-NEXT:    seqz a2, a2
-; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    xor a0, a0, a1
 ; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    and a4, s3, s0
-; RV32I-NEXT:    slli t4, a3, 22
-; RV32I-NEXT:    and a2, a2, t4
-; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a1, t6, s9
+; RV32I-NEXT:    slli s11, s6, 14
+; RV32I-NEXT:    and a2, a2, s11
+; RV32I-NEXT:    seqz a1, a1
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    slli s10, s6, 15
+; RV32I-NEXT:    and a1, a1, s10
+; RV32I-NEXT:    and a3, t6, s8
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    slli s9, s6, 16
+; RV32I-NEXT:    and a2, a2, s9
+; RV32I-NEXT:    and a3, t6, s3
 ; RV32I-NEXT:    xor a1, a1, a2
-; RV32I-NEXT:    addi a4, a4, -1
-; RV32I-NEXT:    and a2, s3, a0
-; RV32I-NEXT:    slli t3, a3, 23
-; RV32I-NEXT:    and a4, a4, t3
-; RV32I-NEXT:    seqz a2, a2
-; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    seqz a2, a3
 ; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    lw s0, 180(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a5, a2, s0
-; RV32I-NEXT:    lui a0, 8192
-; RV32I-NEXT:    and a2, s3, a0
-; RV32I-NEXT:    xor a7, a1, a5
-; RV32I-NEXT:    seqz a1, a2
+; RV32I-NEXT:    slli s5, s6, 17
+; RV32I-NEXT:    and a2, a2, s5
+; RV32I-NEXT:    and a3, t6, a7
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    slli s4, s6, 18
+; RV32I-NEXT:    and a2, a2, s4
+; RV32I-NEXT:    and a3, t6, t0
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    and a1, t6, t1
+; RV32I-NEXT:    slli s8, s6, 19
+; RV32I-NEXT:    and a2, a2, s8
+; RV32I-NEXT:    seqz a1, a1
 ; RV32I-NEXT:    addi a1, a1, -1
-; RV32I-NEXT:    lui a0, 16384
-; RV32I-NEXT:    and a2, s3, a0
-; RV32I-NEXT:    seqz a2, a2
-; RV32I-NEXT:    slli t2, a3, 25
-; RV32I-NEXT:    and a1, a1, t2
+; RV32I-NEXT:    slli s2, s6, 20
+; RV32I-NEXT:    and a1, a1, s2
+; RV32I-NEXT:    and a3, t6, t3
+; RV32I-NEXT:    xor a1, a2, a1
+; RV32I-NEXT:    seqz a2, a3
 ; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    lui a0, 32768
-; RV32I-NEXT:    and a4, s3, a0
-; RV32I-NEXT:    slli t1, a3, 26
+; RV32I-NEXT:    slli t3, s6, 21
+; RV32I-NEXT:    and a2, a2, t3
+; RV32I-NEXT:    and a3, t6, t4
+; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a2, a2, -1
+; RV32I-NEXT:    slli t1, s6, 22
 ; RV32I-NEXT:    and a2, a2, t1
-; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    and a3, t6, t5
 ; RV32I-NEXT:    xor a1, a1, a2
+; RV32I-NEXT:    seqz a2, a3
+; RV32I-NEXT:    addi a3, a2, -1
+; RV32I-NEXT:    and a2, t6, s0
+; RV32I-NEXT:    seqz a4, a2
+; RV32I-NEXT:    slli s3, s6, 23
+; RV32I-NEXT:    and a3, a3, s3
 ; RV32I-NEXT:    addi a4, a4, -1
-; RV32I-NEXT:    lui a0, 65536
-; RV32I-NEXT:    and a2, s3, a0
-; RV32I-NEXT:    slli t0, a3, 27
-; RV32I-NEXT:    and a4, a4, t0
-; RV32I-NEXT:    seqz a2, a2
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    lw t0, 472(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a4, t0
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a3, t6, s1
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    seqz a1, a3
+; RV32I-NEXT:    addi a1, a1, -1
+; RV32I-NEXT:    and a3, t6, a5
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    slli s1, s6, 25
+; RV32I-NEXT:    and a1, a1, s1
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    and a4, t6, a6
+; RV32I-NEXT:    slli a6, s6, 26
+; RV32I-NEXT:    and a3, a3, a6
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    lui a2, 65536
+; RV32I-NEXT:    and a3, t6, a2
+; RV32I-NEXT:    slli a7, s6, 27
+; RV32I-NEXT:    and a4, a4, a7
+; RV32I-NEXT:    seqz a3, a3
 ; RV32I-NEXT:    xor a1, a1, a4
-; RV32I-NEXT:    addi a2, a2, -1
-; RV32I-NEXT:    lui a0, 131072
-; RV32I-NEXT:    and a0, s3, a0
-; RV32I-NEXT:    slli a6, a3, 28
-; RV32I-NEXT:    and a2, a2, a6
-; RV32I-NEXT:    seqz a0, a0
-; RV32I-NEXT:    xor a2, a1, a2
-; RV32I-NEXT:    addi a1, a0, -1
-; RV32I-NEXT:    lui a0, 262144
-; RV32I-NEXT:    and a0, s3, a0
-; RV32I-NEXT:    slli a5, a3, 29
-; RV32I-NEXT:    and a1, a1, a5
-; RV32I-NEXT:    seqz a0, a0
-; RV32I-NEXT:    xor a2, a2, a1
-; RV32I-NEXT:    addi a1, a0, -1
-; RV32I-NEXT:    srli s3, s3, 31
-; RV32I-NEXT:    slli a4, a3, 30
-; RV32I-NEXT:    and a0, a1, a4
-; RV32I-NEXT:    seqz a1, s3
-; RV32I-NEXT:    addi s3, a1, -1
-; RV32I-NEXT:    slli a1, a3, 31
-; RV32I-NEXT:    xor a0, a2, a0
-; RV32I-NEXT:    and a2, s3, a1
-; RV32I-NEXT:    xor a7, s2, a7
-; RV32I-NEXT:    xor a0, a0, a2
-; RV32I-NEXT:    lw a2, 48(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s2, 40(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    xor s2, a2, s2
-; RV32I-NEXT:    xor a2, a7, a0
-; RV32I-NEXT:    lw a0, 176(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw a7, 44(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a0, a0, a7
-; RV32I-NEXT:    lw a7, 172(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a3, a7, a3
-; RV32I-NEXT:    lw a7, 168(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s3, 36(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a7, a7, s3
-; RV32I-NEXT:    lw s3, 164(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw ra, 32(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, ra
-; RV32I-NEXT:    xor a0, a3, a0
-; RV32I-NEXT:    xor a3, a7, s3
-; RV32I-NEXT:    lw a7, 160(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s3, 28(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a7, a7, s3
-; RV32I-NEXT:    lw s3, 156(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw ra, 24(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, ra
-; RV32I-NEXT:    xor a7, a7, s3
-; RV32I-NEXT:    lw s3, 152(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw ra, 20(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, ra
-; RV32I-NEXT:    xor a0, a0, a3
-; RV32I-NEXT:    xor a3, a7, s3
-; RV32I-NEXT:    lw a7, 148(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s3, 16(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a7, a7, s3
-; RV32I-NEXT:    lw s3, 144(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw ra, 12(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, ra
-; RV32I-NEXT:    xor a7, a7, s3
-; RV32I-NEXT:    lw s3, 140(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw ra, 8(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, ra
-; RV32I-NEXT:    xor a7, a7, s3
-; RV32I-NEXT:    lw s3, 136(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw ra, 4(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, ra
-; RV32I-NEXT:    xor a0, a0, a3
-; RV32I-NEXT:    xor a3, a7, s3
-; RV32I-NEXT:    lw a7, 132(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a7, a7, s11
-; RV32I-NEXT:    lw s3, 128(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, s10
-; RV32I-NEXT:    xor a7, a7, s3
-; RV32I-NEXT:    lw s3, 124(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, s9
-; RV32I-NEXT:    xor a0, a0, a3
-; RV32I-NEXT:    xor a3, a7, s3
-; RV32I-NEXT:    lw a7, 120(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a7, a7, s8
-; RV32I-NEXT:    lw s3, 116(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, s7
-; RV32I-NEXT:    xor a7, a7, s3
-; RV32I-NEXT:    lw s3, 112(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, s6
-; RV32I-NEXT:    xor a7, a7, s3
-; RV32I-NEXT:    lw s3, 108(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s3, s3, s5
-; RV32I-NEXT:    xor a7, a7, s3
-; RV32I-NEXT:    lw s3, 104(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and s1, s3, s1
-; RV32I-NEXT:    xor a0, a0, a3
-; RV32I-NEXT:    xor a3, a7, s1
-; RV32I-NEXT:    lw a7, 100(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a7, a7, s4
-; RV32I-NEXT:    lw s1, 96(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and t6, s1, t6
-; RV32I-NEXT:    xor a7, a7, t6
-; RV32I-NEXT:    lw t6, 92(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and t5, t6, t5
-; RV32I-NEXT:    xor a7, a7, t5
-; RV32I-NEXT:    lw t5, 88(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and t4, t5, t4
-; RV32I-NEXT:    xor a7, a7, t4
-; RV32I-NEXT:    lw t4, 84(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and t3, t4, t3
-; RV32I-NEXT:    xor a7, a7, t3
-; RV32I-NEXT:    lw t3, 80(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and t3, t3, s0
-; RV32I-NEXT:    xor a0, a0, a3
-; RV32I-NEXT:    xor a3, a7, t3
-; RV32I-NEXT:    xor a2, a2, s2
-; RV32I-NEXT:    xor a0, a0, a3
-; RV32I-NEXT:    lw a3, 76(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    lui a2, 131072
+; RV32I-NEXT:    and a4, t6, a2
+; RV32I-NEXT:    slli t4, s6, 28
+; RV32I-NEXT:    and a3, a3, t4
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    lui a2, 262144
+; RV32I-NEXT:    and a3, t6, a2
+; RV32I-NEXT:    slli t5, s6, 29
+; RV32I-NEXT:    and a4, a4, t5
+; RV32I-NEXT:    seqz a3, a3
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    addi a3, a3, -1
+; RV32I-NEXT:    srli a4, t6, 31
+; RV32I-NEXT:    slli t2, s6, 30
 ; RV32I-NEXT:    and a3, a3, t2
-; RV32I-NEXT:    lw a7, 72(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a7, a7, t1
-; RV32I-NEXT:    xor a3, a3, a7
-; RV32I-NEXT:    lw a7, 68(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a7, a7, t0
-; RV32I-NEXT:    xor a3, a3, a7
-; RV32I-NEXT:    lw a7, 64(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a6, a7, a6
-; RV32I-NEXT:    xor a3, a3, a6
-; RV32I-NEXT:    lw a6, 60(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a5, a6, a5
-; RV32I-NEXT:    xor a3, a3, a5
-; RV32I-NEXT:    lw a5, 56(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a4, a5, a4
-; RV32I-NEXT:    xor a3, a3, a4
-; RV32I-NEXT:    lw a4, 52(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    and a1, a4, a1
-; RV32I-NEXT:    xor a3, a3, a1
-; RV32I-NEXT:    lw a1, 184(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    seqz a4, a4
+; RV32I-NEXT:    addi a4, a4, -1
+; RV32I-NEXT:    slli s0, s6, 31
+; RV32I-NEXT:    xor a1, a1, a3
+; RV32I-NEXT:    and a4, a4, s0
+; RV32I-NEXT:    xor a1, a1, a4
+; RV32I-NEXT:    lw a2, 488(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a3, 444(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a3, a3, a2
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    slli a3, a3, 1
+; RV32I-NEXT:    lw a1, 448(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    or a1, a1, a3
+; RV32I-NEXT:    lw a5, 328(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    xor a5, a0, a5
+; RV32I-NEXT:    lw a3, 464(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    srli a3, a3, 1
+; RV32I-NEXT:    lw a0, 468(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    srli t6, a0, 1
 ; RV32I-NEXT:    srli a1, a1, 1
-; RV32I-NEXT:    xor a1, a1, a2
-; RV32I-NEXT:    xor a0, a0, a3
-; RV32I-NEXT:    lw ra, 236(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s0, 232(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s1, 228(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s2, 224(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s3, 220(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s4, 216(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s5, 212(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s6, 208(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s7, 204(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s8, 200(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s9, 196(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s10, 192(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    lw s11, 188(sp) # 4-byte Folded Reload
-; RV32I-NEXT:    addi sp, sp, 240
+; RV32I-NEXT:    slli a4, t6, 31
+; RV32I-NEXT:    or s7, a3, a4
+; RV32I-NEXT:    xor a4, a1, a5
+; RV32I-NEXT:    and a3, t6, a2
+; RV32I-NEXT:    and a2, a0, a2
+; RV32I-NEXT:    lw a0, 460(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 332(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t6, a0, a1
+; RV32I-NEXT:    lw a0, 456(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s6, a0, s6
+; RV32I-NEXT:    lw a0, 484(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a1, 324(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a0, a1
+; RV32I-NEXT:    lw a0, 452(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 320(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a0, a0, a5
+; RV32I-NEXT:    xor t6, s6, t6
+; RV32I-NEXT:    xor a0, a1, a0
+; RV32I-NEXT:    lw a1, 440(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw a5, 316(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a1, a1, a5
+; RV32I-NEXT:    lw a5, 436(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 312(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s6, a5, s6
+; RV32I-NEXT:    xor a1, a1, s6
+; RV32I-NEXT:    lw a5, 432(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 308(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s6, a5, s6
+; RV32I-NEXT:    xor a0, t6, a0
+; RV32I-NEXT:    xor a1, a1, s6
+; RV32I-NEXT:    lw a5, 428(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t6, 304(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t6, a5, t6
+; RV32I-NEXT:    lw a5, 424(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 300(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s6, a5, s6
+; RV32I-NEXT:    xor t6, t6, s6
+; RV32I-NEXT:    lw a5, 420(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 296(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s6, a5, s6
+; RV32I-NEXT:    xor t6, t6, s6
+; RV32I-NEXT:    lw a5, 416(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 292(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s6, a5, s6
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, t6, s6
+; RV32I-NEXT:    lw a5, 412(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw t6, 288(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t6, a5, t6
+; RV32I-NEXT:    lw a5, 408(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 284(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s6, a5, s6
+; RV32I-NEXT:    xor t6, t6, s6
+; RV32I-NEXT:    lw a5, 404(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s6, a5, ra
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, t6, s6
+; RV32I-NEXT:    lw a5, 400(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t6, a5, s11
+; RV32I-NEXT:    lw a5, 396(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s6, a5, s10
+; RV32I-NEXT:    xor t6, t6, s6
+; RV32I-NEXT:    lw a5, 392(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s6, a5, s9
+; RV32I-NEXT:    xor t6, t6, s6
+; RV32I-NEXT:    lw a5, 388(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s5, a5, s5
+; RV32I-NEXT:    xor t6, t6, s5
+; RV32I-NEXT:    lw a5, 384(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s4, a5, s4
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, t6, s4
+; RV32I-NEXT:    lw a5, 380(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t6, a5, s8
+; RV32I-NEXT:    lw a5, 376(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and s2, a5, s2
+; RV32I-NEXT:    xor t6, t6, s2
+; RV32I-NEXT:    lw a5, 372(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t3, a5, t3
+; RV32I-NEXT:    xor t3, t6, t3
+; RV32I-NEXT:    lw a5, 368(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, a5, t1
+; RV32I-NEXT:    xor t1, t3, t1
+; RV32I-NEXT:    lw a5, 364(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s3
+; RV32I-NEXT:    xor a5, t1, a5
+; RV32I-NEXT:    lw t1, 360(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and t1, t1, t0
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, a5, t1
+; RV32I-NEXT:    lw a5, 476(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a5, a5, s1
+; RV32I-NEXT:    lw t0, 356(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, t0, a6
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 352(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a6, a7
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 348(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a6, t4
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 344(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a6, t5
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 340(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a6, t2
+; RV32I-NEXT:    xor a5, a5, a6
+; RV32I-NEXT:    lw a6, 336(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    and a6, a6, s0
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    xor a1, a5, a6
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    slli a2, a2, 1
+; RV32I-NEXT:    or a2, a3, a2
+; RV32I-NEXT:    lw a1, 480(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    sw a0, 0(a1)
+; RV32I-NEXT:    sw a4, 4(a1)
+; RV32I-NEXT:    srli a2, a2, 1
+; RV32I-NEXT:    sw s7, 8(a1)
+; RV32I-NEXT:    sw a2, 12(a1)
+; RV32I-NEXT:    lw ra, 540(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s0, 536(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s1, 532(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s2, 528(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s3, 524(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s4, 520(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s5, 516(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s6, 512(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s7, 508(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s8, 504(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s9, 500(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s10, 496(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s11, 492(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    .cfi_restore ra
+; RV32I-NEXT:    .cfi_restore s0
+; RV32I-NEXT:    .cfi_restore s1
+; RV32I-NEXT:    .cfi_restore s2
+; RV32I-NEXT:    .cfi_restore s3
+; RV32I-NEXT:    .cfi_restore s4
+; RV32I-NEXT:    .cfi_restore s5
+; RV32I-NEXT:    .cfi_restore s6
+; RV32I-NEXT:    .cfi_restore s7
+; RV32I-NEXT:    .cfi_restore s8
+; RV32I-NEXT:    .cfi_restore s9
+; RV32I-NEXT:    .cfi_restore s10
+; RV32I-NEXT:    .cfi_restore s11
+; RV32I-NEXT:    addi sp, sp, 544
+; RV32I-NEXT:    .cfi_def_cfa_offset 0
 ; RV32I-NEXT:    ret
 ;
-; RV64I-LABEL: clmul_i64:
+; RV64I-LABEL: clmul_i128_zext:
 ; RV64I:       # %bb.0:
-; RV64I-NEXT:    slli a2, a0, 1
-; RV64I-NEXT:    andi a3, a1, 2
-; RV64I-NEXT:    andi a4, a1, 1
-; RV64I-NEXT:    seqz a3, a3
+; RV64I-NEXT:    addi sp, sp, -272
+; RV64I-NEXT:    .cfi_def_cfa_offset 272
+; RV64I-NEXT:    sd ra, 264(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s0, 256(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s1, 248(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s2, 240(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s3, 232(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s4, 224(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s5, 216(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s6, 208(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s7, 200(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s8, 192(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s9, 184(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s10, 176(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s11, 168(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    .cfi_offset ra, -8
+; RV64I-NEXT:    .cfi_offset s0, -16
+; RV64I-NEXT:    .cfi_offset s1, -24
+; RV64I-NEXT:    .cfi_offset s2, -32
+; RV64I-NEXT:    .cfi_offset s3, -40
+; RV64I-NEXT:    .cfi_offset s4, -48
+; RV64I-NEXT:    .cfi_offset s5, -56
+; RV64I-NEXT:    .cfi_offset s6, -64
+; RV64I-NEXT:    .cfi_offset s7, -72
+; RV64I-NEXT:    .cfi_offset s8, -80
+; RV64I-NEXT:    .cfi_offset s9, -88
+; RV64I-NEXT:    .cfi_offset s10, -96
+; RV64I-NEXT:    .cfi_offset s11, -104
+; RV64I-NEXT:    mv a6, a1
+; RV64I-NEXT:    mv s10, a0
+; RV64I-NEXT:    lui a5, 4080
+; RV64I-NEXT:    srli a0, a0, 24
+; RV64I-NEXT:    li t6, 255
+; RV64I-NEXT:    and a0, a0, a5
+; RV64I-NEXT:    srli a1, s10, 8
+; RV64I-NEXT:    slli s2, t6, 24
+; RV64I-NEXT:    and a1, a1, s2
+; RV64I-NEXT:    lui a2, 16
+; RV64I-NEXT:    srli a3, s10, 40
+; RV64I-NEXT:    addi s1, a2, -256
+; RV64I-NEXT:    lui s3, 16
+; RV64I-NEXT:    and a3, a3, s1
+; RV64I-NEXT:    srli a4, s10, 56
+; RV64I-NEXT:    or a0, a1, a0
+; RV64I-NEXT:    or a3, a3, a4
+; RV64I-NEXT:    or a0, a0, a3
+; RV64I-NEXT:    and a1, s10, a5
+; RV64I-NEXT:    srliw a3, s10, 24
+; RV64I-NEXT:    slli a1, a1, 24
+; RV64I-NEXT:    slli a3, a3, 32
+; RV64I-NEXT:    or a1, a1, a3
+; RV64I-NEXT:    and a3, s10, s1
+; RV64I-NEXT:    slli a3, a3, 40
+; RV64I-NEXT:    slli a2, s10, 56
+; RV64I-NEXT:    sd a2, 160(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    or a3, a2, a3
+; RV64I-NEXT:    lui a4, 61681
+; RV64I-NEXT:    or a1, a3, a1
+; RV64I-NEXT:    addi a7, a4, -241
+; RV64I-NEXT:    or a0, a1, a0
+; RV64I-NEXT:    slli a1, a7, 32
+; RV64I-NEXT:    srli a3, a0, 4
+; RV64I-NEXT:    add t6, a7, a1
+; RV64I-NEXT:    and a1, a3, t6
+; RV64I-NEXT:    and a0, a0, t6
+; RV64I-NEXT:    lui a3, 209715
+; RV64I-NEXT:    slli a0, a0, 4
+; RV64I-NEXT:    addi t0, a3, 819
+; RV64I-NEXT:    or a0, a1, a0
+; RV64I-NEXT:    slli a1, t0, 32
+; RV64I-NEXT:    srli a3, a0, 2
+; RV64I-NEXT:    add t0, t0, a1
+; RV64I-NEXT:    and a1, a3, t0
+; RV64I-NEXT:    and a0, a0, t0
+; RV64I-NEXT:    lui a3, 349525
+; RV64I-NEXT:    slli a0, a0, 2
+; RV64I-NEXT:    addi t1, a3, 1365
+; RV64I-NEXT:    or a0, a1, a0
+; RV64I-NEXT:    slli a1, t1, 32
+; RV64I-NEXT:    srli a3, a0, 1
+; RV64I-NEXT:    add t1, t1, a1
+; RV64I-NEXT:    and a1, a3, t1
+; RV64I-NEXT:    srli a3, a6, 24
+; RV64I-NEXT:    srli a4, a6, 8
+; RV64I-NEXT:    lui a2, 4080
+; RV64I-NEXT:    and a3, a3, a2
+; RV64I-NEXT:    and a4, a4, s2
+; RV64I-NEXT:    or a3, a4, a3
+; RV64I-NEXT:    srli a4, a6, 40
+; RV64I-NEXT:    and a4, a4, s1
+; RV64I-NEXT:    srli a5, a6, 56
+; RV64I-NEXT:    or a4, a4, a5
+; RV64I-NEXT:    and a5, a6, a2
+; RV64I-NEXT:    slli a5, a5, 24
+; RV64I-NEXT:    srliw t2, a6, 24
+; RV64I-NEXT:    slli t2, t2, 32
+; RV64I-NEXT:    and t3, a6, s1
+; RV64I-NEXT:    slli t3, t3, 40
+; RV64I-NEXT:    slli t4, a6, 56
+; RV64I-NEXT:    or a5, a5, t2
+; RV64I-NEXT:    or t2, t4, t3
+; RV64I-NEXT:    or a3, a3, a4
+; RV64I-NEXT:    or a4, t2, a5
+; RV64I-NEXT:    and a0, a0, t1
+; RV64I-NEXT:    or a3, a4, a3
+; RV64I-NEXT:    srli a4, a3, 4
+; RV64I-NEXT:    and a3, a3, t6
+; RV64I-NEXT:    and a4, a4, t6
+; RV64I-NEXT:    slli a3, a3, 4
+; RV64I-NEXT:    slli a0, a0, 1
+; RV64I-NEXT:    or a3, a4, a3
+; RV64I-NEXT:    srli a4, a3, 2
+; RV64I-NEXT:    and a3, a3, t0
+; RV64I-NEXT:    and a4, a4, t0
+; RV64I-NEXT:    slli a3, a3, 2
+; RV64I-NEXT:    or a1, a1, a0
+; RV64I-NEXT:    or a3, a4, a3
+; RV64I-NEXT:    srli a0, a3, 1
+; RV64I-NEXT:    and s5, a3, t1
+; RV64I-NEXT:    and a0, a0, t1
+; RV64I-NEXT:    slli s9, s5, 1
+; RV64I-NEXT:    slli a3, a1, 1
+; RV64I-NEXT:    or a0, a0, s9
+; RV64I-NEXT:    andi a4, a0, 2
+; RV64I-NEXT:    andi a5, a0, 1
 ; RV64I-NEXT:    seqz a4, a4
-; RV64I-NEXT:    addi a3, a3, -1
-; RV64I-NEXT:    addi a4, a4, -1
-; RV64I-NEXT:    and a2, a3, a2
-; RV64I-NEXT:    and a4, a4, a0
-; RV64I-NEXT:    xor a2, a4, a2
-; RV64I-NEXT:    andi a3, a1, 4
-; RV64I-NEXT:    slli a4, a0, 2
-; RV64I-NEXT:    seqz a3, a3
-; RV64I-NEXT:    addi a3, a3, -1
-; RV64I-NEXT:    andi a5, a1, 8
-; RV64I-NEXT:    and a3, a3, a4
-; RV64I-NEXT:    seqz a4, a5
-; RV64I-NEXT:    slli a5, a0, 3
+; RV64I-NEXT:    seqz a5, a5
 ; RV64I-NEXT:    addi a4, a4, -1
-; RV64I-NEXT:    and a4, a4, a5
-; RV64I-NEXT:    andi a5, a1, 16
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    and a3, a4, a3
+; RV64I-NEXT:    and a5, a5, a1
+; RV64I-NEXT:    slli a4, a1, 2
+; RV64I-NEXT:    andi t2, a0, 4
+; RV64I-NEXT:    seqz t2, t2
+; RV64I-NEXT:    andi t3, a0, 8
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    seqz t3, t3
+; RV64I-NEXT:    slli t4, a1, 3
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    and a4, t2, a4
+; RV64I-NEXT:    and t2, t3, t4
+; RV64I-NEXT:    xor a3, a5, a3
+; RV64I-NEXT:    xor a4, a4, t2
 ; RV64I-NEXT:    xor a3, a3, a4
-; RV64I-NEXT:    seqz a4, a5
-; RV64I-NEXT:    slli a5, a0, 4
+; RV64I-NEXT:    andi a4, a0, 16
+; RV64I-NEXT:    slli a5, a1, 4
+; RV64I-NEXT:    seqz a4, a4
 ; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    andi t2, a0, 32
 ; RV64I-NEXT:    and a4, a4, a5
-; RV64I-NEXT:    andi a5, a1, 32
-; RV64I-NEXT:    slli a6, a0, 5
-; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    seqz a5, t2
+; RV64I-NEXT:    slli t2, a1, 5
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    andi a7, a1, 64
-; RV64I-NEXT:    and a5, a5, a6
-; RV64I-NEXT:    seqz a6, a7
-; RV64I-NEXT:    slli a7, a0, 6
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    and a5, a6, a7
-; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    and a5, a5, t2
+; RV64I-NEXT:    andi t2, a0, 64
 ; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    xor a4, a2, a4
-; RV64I-NEXT:    andi a2, a1, 128
-; RV64I-NEXT:    slli a3, a0, 7
-; RV64I-NEXT:    seqz a2, a2
-; RV64I-NEXT:    addi a2, a2, -1
-; RV64I-NEXT:    andi a5, a1, 256
-; RV64I-NEXT:    and a2, a2, a3
-; RV64I-NEXT:    seqz a3, a5
-; RV64I-NEXT:    slli a5, a0, 8
-; RV64I-NEXT:    addi a3, a3, -1
-; RV64I-NEXT:    and a3, a3, a5
-; RV64I-NEXT:    andi a5, a1, 512
-; RV64I-NEXT:    xor a2, a2, a3
-; RV64I-NEXT:    seqz a3, a5
-; RV64I-NEXT:    slli a5, a0, 9
-; RV64I-NEXT:    addi a3, a3, -1
-; RV64I-NEXT:    and a3, a3, a5
-; RV64I-NEXT:    andi a5, a1, 1024
-; RV64I-NEXT:    xor a3, a2, a3
-; RV64I-NEXT:    seqz a2, a5
-; RV64I-NEXT:    slli a5, a0, 10
-; RV64I-NEXT:    addi a2, a2, -1
-; RV64I-NEXT:    and a5, a2, a5
-; RV64I-NEXT:    li a2, 1
-; RV64I-NEXT:    xor a3, a3, a5
-; RV64I-NEXT:    slli a5, a2, 11
-; RV64I-NEXT:    xor a3, a4, a3
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    seqz a4, a5
-; RV64I-NEXT:    lui a5, 1
-; RV64I-NEXT:    addi a4, a4, -1
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    slli a6, a0, 11
-; RV64I-NEXT:    seqz a5, a5
-; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    seqz a5, t2
+; RV64I-NEXT:    slli t2, a1, 6
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a0, 12
-; RV64I-NEXT:    lui a7, 2
-; RV64I-NEXT:    and a5, a5, a6
-; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    and a5, a5, t2
+; RV64I-NEXT:    andi t2, a0, 128
 ; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    seqz a5, t2
+; RV64I-NEXT:    slli t2, a1, 7
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a0, 13
-; RV64I-NEXT:    and a5, a5, a6
-; RV64I-NEXT:    lui a6, 4
+; RV64I-NEXT:    and a5, a5, t2
+; RV64I-NEXT:    andi t2, a0, 256
+; RV64I-NEXT:    slli t3, a1, 8
+; RV64I-NEXT:    seqz t2, t2
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    andi t4, a0, 512
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    slli t4, a1, 9
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    and t2, t3, t4
+; RV64I-NEXT:    xor a4, a3, a4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    slli t2, a1, 10
+; RV64I-NEXT:    andi a3, a0, 1024
+; RV64I-NEXT:    seqz t3, a3
+; RV64I-NEXT:    li s0, 1
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli a2, s0, 11
+; RV64I-NEXT:    sd a2, 152(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    slli t3, a1, 11
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    lui a2, 1
+; RV64I-NEXT:    slli t3, a1, 12
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    seqz t4, t4
+; RV64I-NEXT:    lui a2, 2
+; RV64I-NEXT:    addi t4, t4, -1
+; RV64I-NEXT:    and t5, a0, a2
+; RV64I-NEXT:    and t3, t4, t3
+; RV64I-NEXT:    seqz t4, t5
+; RV64I-NEXT:    slli t5, a1, 13
+; RV64I-NEXT:    addi t4, t4, -1
+; RV64I-NEXT:    xor t2, t2, t3
+; RV64I-NEXT:    and t3, t4, t5
+; RV64I-NEXT:    xor t2, t2, t3
+; RV64I-NEXT:    lui a2, 4
+; RV64I-NEXT:    slli t3, a1, 14
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    seqz t4, t4
+; RV64I-NEXT:    lui a2, 8
+; RV64I-NEXT:    addi t4, t4, -1
+; RV64I-NEXT:    and t5, a0, a2
+; RV64I-NEXT:    and t3, t4, t3
+; RV64I-NEXT:    seqz t4, t5
+; RV64I-NEXT:    slli t5, a1, 15
+; RV64I-NEXT:    addi t4, t4, -1
+; RV64I-NEXT:    xor t2, t2, t3
+; RV64I-NEXT:    and t3, t4, t5
 ; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    and a5, a1, a6
-; RV64I-NEXT:    xor a3, a3, a4
-; RV64I-NEXT:    seqz a4, a5
-; RV64I-NEXT:    addi a4, a4, -1
-; RV64I-NEXT:    lui a5, 8
-; RV64I-NEXT:    slli a6, a0, 14
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    and a4, a4, a6
-; RV64I-NEXT:    seqz a5, a5
-; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    lui a6, 16
-; RV64I-NEXT:    slli a7, a0, 15
-; RV64I-NEXT:    and a6, a1, a6
-; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
+; RV64I-NEXT:    xor a5, t2, t3
 ; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 16
-; RV64I-NEXT:    lui a7, 32
-; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    slli a5, a1, 16
+; RV64I-NEXT:    and t2, a0, s3
+; RV64I-NEXT:    seqz t2, t2
+; RV64I-NEXT:    lui a2, 32
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    and a5, t2, a5
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    slli t3, a1, 17
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    lui a2, 64
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    and t2, a0, a2
+; RV64I-NEXT:    slli t3, a1, 18
+; RV64I-NEXT:    seqz t2, t2
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    lui a2, 128
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    slli t3, a1, 19
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    lui a2, 256
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    and t2, a0, a2
+; RV64I-NEXT:    slli t3, a1, 20
+; RV64I-NEXT:    seqz t2, t2
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    lui a2, 512
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    slli t3, a1, 21
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    lui a2, 1024
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    and t2, a0, a2
 ; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    seqz a5, t2
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    lui a6, 64
-; RV64I-NEXT:    slli a7, a0, 17
-; RV64I-NEXT:    and a6, a1, a6
-; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a7, a0, 18
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    and a5, a6, a7
+; RV64I-NEXT:    lui a2, 2048
+; RV64I-NEXT:    slli t2, a1, 22
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    and a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    lui a2, 4096
+; RV64I-NEXT:    slli t3, a1, 23
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 24
+; RV64I-NEXT:    lui a2, 8192
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    lui a2, 16384
+; RV64I-NEXT:    slli t3, a1, 25
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 26
+; RV64I-NEXT:    lui a2, 32768
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    lui a2, 65536
+; RV64I-NEXT:    slli t3, a1, 27
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t4, a1, 28
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    and t2, t3, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    lui a2, 131072
 ; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    lui a5, 128
-; RV64I-NEXT:    xor a3, a3, a4
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    seqz a4, a5
-; RV64I-NEXT:    lui a5, 256
-; RV64I-NEXT:    addi a4, a4, -1
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    slli a6, a0, 19
+; RV64I-NEXT:    and a5, a0, a2
 ; RV64I-NEXT:    seqz a5, a5
-; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    lui a2, 262144
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a0, 20
-; RV64I-NEXT:    lui a7, 512
-; RV64I-NEXT:    and a5, a5, a6
-; RV64I-NEXT:    and a6, a1, a7
+; RV64I-NEXT:    and t2, a0, a2
+; RV64I-NEXT:    seqz t2, t2
+; RV64I-NEXT:    slli t3, a1, 29
+; RV64I-NEXT:    and a5, a5, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli t3, a1, 30
+; RV64I-NEXT:    sraiw t4, a0, 31
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 31
+; RV64I-NEXT:    slli a2, s0, 32
+; RV64I-NEXT:    sd a2, 144(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli a2, s0, 33
+; RV64I-NEXT:    sd a2, 136(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli t3, a1, 32
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 33
+; RV64I-NEXT:    slli a2, s0, 34
+; RV64I-NEXT:    sd a2, 128(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli a2, s0, 35
+; RV64I-NEXT:    sd a2, 120(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli t3, a1, 34
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 35
+; RV64I-NEXT:    slli a2, s0, 36
+; RV64I-NEXT:    sd a2, 112(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli t3, a1, 36
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    slli a2, s0, 37
+; RV64I-NEXT:    sd a2, 88(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    and t2, a0, a2
 ; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    seqz a5, t2
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    lui a6, 1024
-; RV64I-NEXT:    slli a7, a0, 21
-; RV64I-NEXT:    and a6, a1, a6
-; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 22
-; RV64I-NEXT:    lui a7, 2048
-; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a2, s0, 38
+; RV64I-NEXT:    sd a2, 96(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli t2, a1, 37
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    and a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli a2, s0, 39
+; RV64I-NEXT:    sd a2, 104(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli t3, a1, 38
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 39
+; RV64I-NEXT:    slli a2, s0, 40
+; RV64I-NEXT:    sd a2, 80(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli a2, s0, 41
+; RV64I-NEXT:    sd a2, 72(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli t3, a1, 40
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 41
+; RV64I-NEXT:    slli a2, s0, 42
+; RV64I-NEXT:    sd a2, 64(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli a2, s0, 43
+; RV64I-NEXT:    sd a2, 56(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli t3, a1, 42
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 43
+; RV64I-NEXT:    slli a2, s0, 44
+; RV64I-NEXT:    sd a2, 48(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli a2, s0, 45
+; RV64I-NEXT:    sd a2, 40(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli t3, a1, 44
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t4, a1, 45
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    and t2, t3, t4
+; RV64I-NEXT:    xor a5, a5, t2
+; RV64I-NEXT:    slli a2, s0, 46
+; RV64I-NEXT:    sd a2, 24(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    xor a7, a4, a5
+; RV64I-NEXT:    and a4, a0, a2
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a2, s0, 47
+; RV64I-NEXT:    sd a2, 32(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    and t2, a0, a2
+; RV64I-NEXT:    seqz t2, t2
+; RV64I-NEXT:    slli t3, a1, 46
+; RV64I-NEXT:    and a4, a4, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli t3, a1, 47
+; RV64I-NEXT:    slli a2, s0, 48
+; RV64I-NEXT:    sd a2, 16(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    and t3, a0, a2
+; RV64I-NEXT:    xor a4, a4, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli a2, s0, 49
+; RV64I-NEXT:    sd a2, 8(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    slli t3, a1, 48
+; RV64I-NEXT:    and t4, a0, a2
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a4, a4, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 49
+; RV64I-NEXT:    slli ra, s0, 50
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, ra
+; RV64I-NEXT:    xor a4, a4, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli s11, s0, 51
+; RV64I-NEXT:    slli t3, a1, 50
+; RV64I-NEXT:    and t4, a0, s11
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a4, a4, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 51
+; RV64I-NEXT:    slli s8, s0, 52
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, s8
+; RV64I-NEXT:    xor a4, a4, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli s7, s0, 53
+; RV64I-NEXT:    slli t3, a1, 52
+; RV64I-NEXT:    and t4, a0, s7
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a4, a4, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 53
+; RV64I-NEXT:    slli s6, s0, 54
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, s6
+; RV64I-NEXT:    xor a4, a4, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli s5, s0, 55
+; RV64I-NEXT:    slli t3, a1, 54
+; RV64I-NEXT:    and t4, a0, s5
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    xor a4, a4, t2
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t2, a1, 55
+; RV64I-NEXT:    slli s4, s0, 56
+; RV64I-NEXT:    and t2, t3, t2
+; RV64I-NEXT:    and t3, a0, s4
+; RV64I-NEXT:    xor a5, a4, t2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli s3, s0, 57
+; RV64I-NEXT:    slli t3, a1, 56
+; RV64I-NEXT:    and t4, a0, s3
+; RV64I-NEXT:    and t2, t2, t3
+; RV64I-NEXT:    seqz t3, t4
+; RV64I-NEXT:    addi t3, t3, -1
+; RV64I-NEXT:    slli t5, s0, 58
+; RV64I-NEXT:    slli t4, a1, 57
+; RV64I-NEXT:    and a2, a0, t5
+; RV64I-NEXT:    and t3, t3, t4
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    xor t2, t2, t3
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    slli t3, a1, 58
+; RV64I-NEXT:    slli t4, s0, 59
+; RV64I-NEXT:    and a2, a2, t3
+; RV64I-NEXT:    and t3, a0, t4
+; RV64I-NEXT:    xor a4, t2, a2
+; RV64I-NEXT:    seqz t2, t3
+; RV64I-NEXT:    addi t2, t2, -1
+; RV64I-NEXT:    slli t3, s0, 60
+; RV64I-NEXT:    slli a2, a1, 59
+; RV64I-NEXT:    and a3, a0, t3
+; RV64I-NEXT:    and a2, t2, a2
+; RV64I-NEXT:    seqz a3, a3
+; RV64I-NEXT:    xor a2, a4, a2
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    slli a4, a1, 60
+; RV64I-NEXT:    slli t2, s0, 61
+; RV64I-NEXT:    and a3, a3, a4
+; RV64I-NEXT:    and a4, a0, t2
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    seqz a3, a4
+; RV64I-NEXT:    addi a4, a3, -1
+; RV64I-NEXT:    slli a3, s0, 62
+; RV64I-NEXT:    and a0, a0, a3
+; RV64I-NEXT:    slli s0, a1, 61
+; RV64I-NEXT:    and a4, a4, s0
+; RV64I-NEXT:    seqz a0, a0
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a0, a0, -1
+; RV64I-NEXT:    srli a4, s9, 63
+; RV64I-NEXT:    slli s0, a1, 62
+; RV64I-NEXT:    and a0, a0, s0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a1, a1, 63
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    xor a0, a2, a0
+; RV64I-NEXT:    and a1, a4, a1
+; RV64I-NEXT:    xor a2, a7, a5
+; RV64I-NEXT:    xor a0, a0, a1
+; RV64I-NEXT:    xor a0, a2, a0
+; RV64I-NEXT:    srli a1, a0, 40
+; RV64I-NEXT:    and a1, a1, s1
+; RV64I-NEXT:    srli a2, a0, 56
+; RV64I-NEXT:    or a1, a1, a2
+; RV64I-NEXT:    srli a2, a0, 24
+; RV64I-NEXT:    srli a4, a0, 8
+; RV64I-NEXT:    and a4, a4, s2
+; RV64I-NEXT:    lui a5, 4080
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    or a2, a4, a2
+; RV64I-NEXT:    srliw a4, a0, 24
+; RV64I-NEXT:    slli a4, a4, 32
+; RV64I-NEXT:    and a5, a0, a5
+; RV64I-NEXT:    slli a5, a5, 24
+; RV64I-NEXT:    and a7, a0, s1
+; RV64I-NEXT:    slli a0, a0, 56
+; RV64I-NEXT:    slli a7, a7, 40
+; RV64I-NEXT:    or a4, a5, a4
+; RV64I-NEXT:    or a0, a0, a7
+; RV64I-NEXT:    or a1, a2, a1
+; RV64I-NEXT:    or a0, a0, a4
+; RV64I-NEXT:    or a0, a0, a1
+; RV64I-NEXT:    srli a1, a0, 4
+; RV64I-NEXT:    and a0, a0, t6
+; RV64I-NEXT:    and a1, a1, t6
+; RV64I-NEXT:    slli a0, a0, 4
+; RV64I-NEXT:    or a0, a1, a0
+; RV64I-NEXT:    srli a1, a0, 2
+; RV64I-NEXT:    and a0, a0, t0
+; RV64I-NEXT:    and a1, a1, t0
+; RV64I-NEXT:    slli a0, a0, 2
+; RV64I-NEXT:    or a0, a1, a0
+; RV64I-NEXT:    srli a1, a0, 1
+; RV64I-NEXT:    and a1, a1, t1
+; RV64I-NEXT:    andi a2, a6, 2
+; RV64I-NEXT:    and a0, a0, t1
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    andi a4, a6, 1
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, s10, 1
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a0, a0, 1
+; RV64I-NEXT:    and a4, a4, s10
+; RV64I-NEXT:    or a1, a1, a0
+; RV64I-NEXT:    xor a2, a4, a2
+; RV64I-NEXT:    andi a0, a6, 4
+; RV64I-NEXT:    andi a4, a6, 8
+; RV64I-NEXT:    seqz a0, a0
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a0, a0, -1
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a5, s10, 2
+; RV64I-NEXT:    slli a7, s10, 3
+; RV64I-NEXT:    and a0, a0, a5
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    xor a0, a0, a4
+; RV64I-NEXT:    andi a4, a6, 16
+; RV64I-NEXT:    xor a0, a2, a0
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    andi a4, a6, 32
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, s10, 4
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a5, s10, 5
+; RV64I-NEXT:    andi a7, a6, 64
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    seqz a5, a7
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    lui a6, 4096
-; RV64I-NEXT:    slli a7, a0, 23
-; RV64I-NEXT:    and a6, a1, a6
-; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a7, a0, 24
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    and a5, a6, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    lui a5, 8192
-; RV64I-NEXT:    xor a3, a3, a4
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    seqz a4, a5
-; RV64I-NEXT:    lui a5, 16384
+; RV64I-NEXT:    slli a7, s10, 6
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    andi a4, a6, 128
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    andi a4, a6, 256
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, s10, 7
+; RV64I-NEXT:    and a2, a2, a5
 ; RV64I-NEXT:    addi a4, a4, -1
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    slli a6, a0, 25
-; RV64I-NEXT:    seqz a5, a5
-; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    slli a5, s10, 8
+; RV64I-NEXT:    andi a7, a6, 512
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a0, 26
-; RV64I-NEXT:    lui a7, 32768
-; RV64I-NEXT:    and a5, a5, a6
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a4, s10, 9
+; RV64I-NEXT:    andi a7, a6, 1024
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    seqz a5, a7
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    lui a6, 65536
-; RV64I-NEXT:    slli a7, a0, 27
-; RV64I-NEXT:    and a6, a1, a6
-; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 28
-; RV64I-NEXT:    lui a7, 131072
+; RV64I-NEXT:    slli a7, s10, 10
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    ld a4, 152(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    lui a4, 1
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, s10, 11
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 2
 ; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a7, s10, 12
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    seqz a5, a5
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    lui a6, 262144
-; RV64I-NEXT:    slli a7, a0, 29
-; RV64I-NEXT:    and a6, a1, a6
-; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 30
-; RV64I-NEXT:    sraiw a7, a1, 31
+; RV64I-NEXT:    slli a7, s10, 13
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    lui a4, 4
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    lui a4, 8
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, s10, 14
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 16
 ; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    seqz a6, a7
-; RV64I-NEXT:    slli a7, a0, 31
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    and a5, a6, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    slli a5, a2, 32
-; RV64I-NEXT:    xor a3, a3, a4
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    seqz a4, a5
-; RV64I-NEXT:    slli a5, a2, 33
+; RV64I-NEXT:    slli a7, s10, 15
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    lui a4, 32
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    slli a7, s10, 16
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
 ; RV64I-NEXT:    addi a4, a4, -1
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    slli a6, a0, 32
+; RV64I-NEXT:    lui a5, 64
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    slli a7, s10, 17
+; RV64I-NEXT:    and a4, a4, a7
 ; RV64I-NEXT:    seqz a5, a5
-; RV64I-NEXT:    and a4, a4, a6
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a0, 33
-; RV64I-NEXT:    slli a7, a2, 34
-; RV64I-NEXT:    and a5, a5, a6
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a7, s10, 18
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    lui a4, 128
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    lui a4, 256
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, s10, 19
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 512
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    slli a7, s10, 20
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a2, 35
-; RV64I-NEXT:    slli a7, a0, 34
-; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    lui a4, 1024
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    slli a7, s10, 21
 ; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 35
-; RV64I-NEXT:    slli a7, a2, 36
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 2048
 ; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a7, s10, 22
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a2, 37
-; RV64I-NEXT:    slli a7, a0, 36
-; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    lui a4, 4096
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    slli a7, s10, 23
 ; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 37
-; RV64I-NEXT:    slli a7, a2, 38
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a7, s10, 24
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    lui a4, 8192
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    lui a4, 16384
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, s10, 25
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    lui a5, 32768
 ; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a7, s10, 26
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a0, 38
-; RV64I-NEXT:    and a5, a5, a6
-; RV64I-NEXT:    slli a6, a2, 39
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    and a5, a1, a6
-; RV64I-NEXT:    xor a3, a3, a4
-; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    lui a4, 65536
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    slli a7, s10, 27
+; RV64I-NEXT:    and a5, a5, a7
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
 ; RV64I-NEXT:    addi a4, a4, -1
-; RV64I-NEXT:    slli a5, a2, 40
-; RV64I-NEXT:    slli a6, a0, 39
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    lui a5, 131072
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    slli a7, s10, 28
+; RV64I-NEXT:    and a4, a4, a7
 ; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a2, 41
-; RV64I-NEXT:    slli a7, a0, 40
-; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    lui a4, 262144
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    slli a7, s10, 29
 ; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 41
-; RV64I-NEXT:    slli a7, a2, 42
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a5, s10, 30
+; RV64I-NEXT:    sraiw a7, a6, 31
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a7, s10, 31
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    ld a4, 144(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    ld a4, 136(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, s10, 32
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    ld a5, 128(sp) # 8-byte Folded Reload
 ; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a7, s10, 33
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a2, 43
-; RV64I-NEXT:    slli a7, a0, 42
-; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    ld a4, 120(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    slli a7, s10, 34
 ; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 43
-; RV64I-NEXT:    slli a7, a2, 44
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    ld a5, 112(sp) # 8-byte Folded Reload
 ; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a7, s10, 35
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a2, 45
-; RV64I-NEXT:    slli a7, a0, 44
-; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    ld a4, 88(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    slli a7, s10, 36
 ; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 45
-; RV64I-NEXT:    slli a7, a2, 46
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    ld a5, 96(sp) # 8-byte Folded Reload
 ; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a7, s10, 37
+; RV64I-NEXT:    and a4, a4, a7
+; RV64I-NEXT:    seqz a5, a5
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a0, 46
-; RV64I-NEXT:    and a5, a5, a6
-; RV64I-NEXT:    slli a6, a2, 47
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    and a5, a1, a6
-; RV64I-NEXT:    xor a3, a3, a4
-; RV64I-NEXT:    seqz a4, a5
+; RV64I-NEXT:    slli a7, s10, 38
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    ld a4, 104(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    ld a4, 80(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    slli a5, s10, 39
+; RV64I-NEXT:    and a2, a2, a5
 ; RV64I-NEXT:    addi a4, a4, -1
-; RV64I-NEXT:    slli a5, a2, 48
-; RV64I-NEXT:    slli a6, a0, 47
-; RV64I-NEXT:    and a5, a1, a5
-; RV64I-NEXT:    and a4, a4, a6
+; RV64I-NEXT:    ld a5, 72(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a5, a6, a5
+; RV64I-NEXT:    slli a7, s10, 40
+; RV64I-NEXT:    and a4, a4, a7
 ; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    xor a2, a2, a4
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a2, 49
-; RV64I-NEXT:    slli a7, a0, 48
-; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    ld a4, 64(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    slli a7, s10, 41
 ; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 49
-; RV64I-NEXT:    slli a7, a2, 50
-; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a5, s10, 42
+; RV64I-NEXT:    ld a7, 56(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a6, a7
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a4, s10, 43
+; RV64I-NEXT:    ld a7, 48(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a6, a7
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a4, s10, 44
+; RV64I-NEXT:    ld a7, 40(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a6, a7
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a4, s10, 45
+; RV64I-NEXT:    ld a7, 24(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a6, a7
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a7, s10, 46
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    ld a4, 32(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    seqz a2, a4
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    ld a4, 16(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a4, a6, a4
+; RV64I-NEXT:    slli a5, s10, 47
+; RV64I-NEXT:    seqz a4, a4
+; RV64I-NEXT:    and a2, a2, a5
+; RV64I-NEXT:    addi a4, a4, -1
+; RV64I-NEXT:    slli a5, s10, 48
+; RV64I-NEXT:    ld a7, 8(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a7, a6, a7
+; RV64I-NEXT:    and a4, a4, a5
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a4, s10, 49
+; RV64I-NEXT:    and a7, a6, ra
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a4, s10, 50
+; RV64I-NEXT:    and a7, a6, s11
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a4, s10, 51
+; RV64I-NEXT:    and a7, a6, s8
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    slli a4, s10, 52
+; RV64I-NEXT:    and a7, a6, s7
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a2, 51
-; RV64I-NEXT:    slli a7, a0, 50
-; RV64I-NEXT:    and a6, a1, a6
-; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 51
-; RV64I-NEXT:    slli a7, a2, 52
-; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a4, s10, 53
+; RV64I-NEXT:    and a7, a6, s6
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    seqz a5, a7
+; RV64I-NEXT:    xor a2, a2, a4
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a2, 53
-; RV64I-NEXT:    slli a7, a0, 52
-; RV64I-NEXT:    and a6, a1, a6
-; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 53
-; RV64I-NEXT:    slli a7, a2, 54
-; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a4, s10, 54
+; RV64I-NEXT:    and a7, a6, s5
+; RV64I-NEXT:    and a4, a5, a4
+; RV64I-NEXT:    seqz a5, a7
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a2, 55
-; RV64I-NEXT:    slli a7, a0, 54
-; RV64I-NEXT:    and a6, a1, a6
-; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a5, a0, 55
-; RV64I-NEXT:    slli a7, a2, 56
-; RV64I-NEXT:    and a5, a6, a5
-; RV64I-NEXT:    and a6, a1, a7
-; RV64I-NEXT:    xor a4, a4, a5
-; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli a7, s10, 55
+; RV64I-NEXT:    xor a2, a2, a4
+; RV64I-NEXT:    and a4, a5, a7
+; RV64I-NEXT:    xor a4, a2, a4
+; RV64I-NEXT:    and a2, a6, s4
+; RV64I-NEXT:    seqz a2, a2
+; RV64I-NEXT:    and a5, a6, s3
+; RV64I-NEXT:    addi a2, a2, -1
+; RV64I-NEXT:    seqz a5, a5
+; RV64I-NEXT:    ld a7, 160(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    and a2, a2, a7
 ; RV64I-NEXT:    addi a5, a5, -1
-; RV64I-NEXT:    slli a6, a2, 57
-; RV64I-NEXT:    slli a7, a0, 56
-; RV64I-NEXT:    and a6, a1, a6
+; RV64I-NEXT:    slli a7, s10, 57
+; RV64I-NEXT:    and t0, a6, t5
 ; RV64I-NEXT:    and a5, a5, a7
-; RV64I-NEXT:    seqz a6, a6
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a7, a2, 58
-; RV64I-NEXT:    slli t0, a0, 57
-; RV64I-NEXT:    and a7, a1, a7
-; RV64I-NEXT:    and a6, a6, t0
-; RV64I-NEXT:    seqz a7, a7
-; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    seqz a7, t0
+; RV64I-NEXT:    xor a2, a2, a5
 ; RV64I-NEXT:    addi a7, a7, -1
-; RV64I-NEXT:    slli a6, a0, 58
-; RV64I-NEXT:    slli t0, a2, 59
-; RV64I-NEXT:    and a6, a7, a6
-; RV64I-NEXT:    and a7, a1, t0
-; RV64I-NEXT:    xor a5, a5, a6
-; RV64I-NEXT:    seqz a6, a7
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a7, a2, 60
-; RV64I-NEXT:    slli t0, a0, 59
-; RV64I-NEXT:    and a7, a1, a7
-; RV64I-NEXT:    and a6, a6, t0
-; RV64I-NEXT:    seqz a7, a7
-; RV64I-NEXT:    xor a5, a5, a6
+; RV64I-NEXT:    slli a5, s10, 58
+; RV64I-NEXT:    and t0, a6, t4
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a7, t0
+; RV64I-NEXT:    xor a2, a2, a5
 ; RV64I-NEXT:    addi a7, a7, -1
-; RV64I-NEXT:    slli a6, a0, 60
-; RV64I-NEXT:    slli t0, a2, 61
-; RV64I-NEXT:    and a6, a7, a6
-; RV64I-NEXT:    and a7, a1, t0
-; RV64I-NEXT:    xor a5, a5, a6
-; RV64I-NEXT:    seqz a6, a7
-; RV64I-NEXT:    addi a6, a6, -1
-; RV64I-NEXT:    slli a2, a2, 62
-; RV64I-NEXT:    slli a7, a0, 61
-; RV64I-NEXT:    and a2, a1, a2
-; RV64I-NEXT:    and a6, a6, a7
-; RV64I-NEXT:    seqz a2, a2
-; RV64I-NEXT:    xor a5, a5, a6
-; RV64I-NEXT:    addi a2, a2, -1
-; RV64I-NEXT:    slli a6, a0, 62
-; RV64I-NEXT:    srli a1, a1, 63
-; RV64I-NEXT:    and a2, a2, a6
-; RV64I-NEXT:    seqz a1, a1
-; RV64I-NEXT:    slli a0, a0, 63
-; RV64I-NEXT:    addi a1, a1, -1
-; RV64I-NEXT:    xor a2, a5, a2
-; RV64I-NEXT:    and a0, a1, a0
-; RV64I-NEXT:    xor a3, a3, a4
-; RV64I-NEXT:    xor a0, a2, a0
-; RV64I-NEXT:    xor a0, a3, a0
+; RV64I-NEXT:    slli a5, s10, 59
+; RV64I-NEXT:    and t0, a6, t3
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a7, t0
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a7, a7, -1
+; RV64I-NEXT:    slli a5, s10, 60
+; RV64I-NEXT:    and t0, a6, t2
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a7, t0
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a7, a7, -1
+; RV64I-NEXT:    slli a5, s10, 61
+; RV64I-NEXT:    and a3, a6, a3
+; RV64I-NEXT:    and a5, a7, a5
+; RV64I-NEXT:    seqz a3, a3
+; RV64I-NEXT:    xor a2, a2, a5
+; RV64I-NEXT:    addi a3, a3, -1
+; RV64I-NEXT:    slli a5, s10, 62
+; RV64I-NEXT:    srli a6, a6, 63
+; RV64I-NEXT:    and a3, a3, a5
+; RV64I-NEXT:    seqz a5, a6
+; RV64I-NEXT:    slli s10, s10, 63
+; RV64I-NEXT:    addi a5, a5, -1
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    and a3, a5, s10
+; RV64I-NEXT:    xor a0, a0, a4
+; RV64I-NEXT:    xor a2, a2, a3
+; RV64I-NEXT:    xor a0, a0, a2
+; RV64I-NEXT:    srli a1, a1, 1
+; RV64I-NEXT:    ld ra, 264(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s0, 256(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s1, 248(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s2, 240(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s3, 232(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s4, 224(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s5, 216(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s6, 208(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s7, 200(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s8, 192(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s9, 184(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s10, 176(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s11, 168(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    .cfi_restore ra
+; RV64I-NEXT:    .cfi_restore s0
+; RV64I-NEXT:    .cfi_restore s1
+; RV64I-NEXT:    .cfi_restore s2
+; RV64I-NEXT:    .cfi_restore s3
+; RV64I-NEXT:    .cfi_restore s4
+; RV64I-NEXT:    .cfi_restore s5
+; RV64I-NEXT:    .cfi_restore s6
+; RV64I-NEXT:    .cfi_restore s7
+; RV64I-NEXT:    .cfi_restore s8
+; RV64I-NEXT:    .cfi_restore s9
+; RV64I-NEXT:    .cfi_restore s10
+; RV64I-NEXT:    .cfi_restore s11
+; RV64I-NEXT:    addi sp, sp, 272
+; RV64I-NEXT:    .cfi_def_cfa_offset 0
 ; RV64I-NEXT:    ret
 ;
-; RV32IM-LABEL: clmul_i64:
+; RV32IM-LABEL: clmul_i128_zext:
 ; RV32IM:       # %bb.0:
-; RV32IM-NEXT:    addi sp, sp, -80
-; RV32IM-NEXT:    sw ra, 76(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s0, 72(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s1, 68(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s2, 64(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s3, 60(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s4, 56(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s5, 52(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s6, 48(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s7, 44(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s8, 40(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s9, 36(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s10, 32(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw s11, 28(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw a3, 24(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    sw a1, 16(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    lui a4, 16
-; RV32IM-NEXT:    srli a5, a2, 8
-; RV32IM-NEXT:    addi t0, a4, -256
-; RV32IM-NEXT:    and a4, a5, t0
-; RV32IM-NEXT:    srli a5, a2, 24
-; RV32IM-NEXT:    and a6, a2, t0
-; RV32IM-NEXT:    slli a6, a6, 8
-; RV32IM-NEXT:    slli a7, a2, 24
-; RV32IM-NEXT:    or a4, a4, a5
-; RV32IM-NEXT:    or a5, a7, a6
-; RV32IM-NEXT:    or a4, a5, a4
-; RV32IM-NEXT:    lui a5, 61681
-; RV32IM-NEXT:    srli a6, a4, 4
-; RV32IM-NEXT:    addi t1, a5, -241
-; RV32IM-NEXT:    and a5, a6, t1
-; RV32IM-NEXT:    and a4, a4, t1
-; RV32IM-NEXT:    slli a4, a4, 4
-; RV32IM-NEXT:    lui a6, 209715
-; RV32IM-NEXT:    or a4, a5, a4
-; RV32IM-NEXT:    addi t2, a6, 819
-; RV32IM-NEXT:    srli a5, a4, 2
-; RV32IM-NEXT:    and a4, a4, t2
-; RV32IM-NEXT:    and a5, a5, t2
-; RV32IM-NEXT:    slli a4, a4, 2
-; RV32IM-NEXT:    or a4, a5, a4
-; RV32IM-NEXT:    lui a1, 349525
-; RV32IM-NEXT:    srli a5, a4, 1
-; RV32IM-NEXT:    addi t4, a1, 1365
-; RV32IM-NEXT:    and a5, a5, t4
-; RV32IM-NEXT:    sw a0, 20(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    srli a6, a0, 8
-; RV32IM-NEXT:    and a4, a4, t4
-; RV32IM-NEXT:    and a6, a6, t0
-; RV32IM-NEXT:    srli a7, a0, 24
-; RV32IM-NEXT:    and t5, a0, t0
-; RV32IM-NEXT:    slli t5, t5, 8
-; RV32IM-NEXT:    slli t6, a0, 24
-; RV32IM-NEXT:    or a6, a6, a7
-; RV32IM-NEXT:    or a7, t6, t5
-; RV32IM-NEXT:    slli a4, a4, 1
-; RV32IM-NEXT:    or a6, a7, a6
-; RV32IM-NEXT:    srli a7, a6, 4
-; RV32IM-NEXT:    and a6, a6, t1
-; RV32IM-NEXT:    and a7, a7, t1
-; RV32IM-NEXT:    slli a6, a6, 4
-; RV32IM-NEXT:    or t5, a5, a4
-; RV32IM-NEXT:    or a4, a7, a6
-; RV32IM-NEXT:    srli a5, a4, 2
-; RV32IM-NEXT:    and a4, a4, t2
-; RV32IM-NEXT:    and a5, a5, t2
-; RV32IM-NEXT:    slli a4, a4, 2
-; RV32IM-NEXT:    lui a6, 69905
-; RV32IM-NEXT:    or a5, a5, a4
-; RV32IM-NEXT:    addi a4, a6, 273
-; RV32IM-NEXT:    srli a6, a5, 1
-; RV32IM-NEXT:    and a6, a6, t4
-; RV32IM-NEXT:    and a5, a5, t4
-; RV32IM-NEXT:    slli a5, a5, 1
-; RV32IM-NEXT:    lui a7, 139810
-; RV32IM-NEXT:    or t6, a6, a5
-; RV32IM-NEXT:    addi a5, a7, 546
-; RV32IM-NEXT:    and s0, t5, a4
-; RV32IM-NEXT:    and s1, t6, a5
-; RV32IM-NEXT:    and s2, t5, a5
-; RV32IM-NEXT:    and s3, t6, a4
-; RV32IM-NEXT:    mul s4, s1, s0
-; RV32IM-NEXT:    mul s5, s3, s2
-; RV32IM-NEXT:    lui a6, 559241
-; RV32IM-NEXT:    lui a7, 279620
-; RV32IM-NEXT:    addi s11, a6, -1912
-; RV32IM-NEXT:    addi t3, a7, 1092
-; RV32IM-NEXT:    and s6, t5, s11
-; RV32IM-NEXT:    and s7, t6, t3
-; RV32IM-NEXT:    and t5, t5, t3
-; RV32IM-NEXT:    and t6, t6, s11
-; RV32IM-NEXT:    mul s8, s7, s6
-; RV32IM-NEXT:    mul s9, t6, t5
-; RV32IM-NEXT:    mul s10, s1, s6
-; RV32IM-NEXT:    mul a6, s3, s0
-; RV32IM-NEXT:    mul ra, s7, t5
-; RV32IM-NEXT:    mul a3, t6, s2
-; RV32IM-NEXT:    mul a1, s1, s2
-; RV32IM-NEXT:    mul s1, s1, t5
-; RV32IM-NEXT:    mul t5, s3, t5
-; RV32IM-NEXT:    mul s3, s3, s6
-; RV32IM-NEXT:    mul s6, t6, s6
-; RV32IM-NEXT:    mul a0, s7, s0
-; RV32IM-NEXT:    mul s2, s7, s2
-; RV32IM-NEXT:    mul t6, t6, s0
-; RV32IM-NEXT:    xor s0, s5, s4
-; RV32IM-NEXT:    xor s4, s8, s9
-; RV32IM-NEXT:    xor s5, a6, s10
-; RV32IM-NEXT:    xor a3, ra, a3
-; RV32IM-NEXT:    xor s0, s0, s4
-; RV32IM-NEXT:    xor a3, s5, a3
-; RV32IM-NEXT:    and s0, s0, a5
-; RV32IM-NEXT:    and a3, a3, a4
-; RV32IM-NEXT:    xor a1, t5, a1
-; RV32IM-NEXT:    xor a0, a0, s6
-; RV32IM-NEXT:    xor t5, s3, s1
-; RV32IM-NEXT:    xor t6, s2, t6
+; RV32IM-NEXT:    addi sp, sp, -176
+; RV32IM-NEXT:    .cfi_def_cfa_offset 176
+; RV32IM-NEXT:    sw ra, 172(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s0, 168(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s1, 164(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s2, 160(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s3, 156(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s4, 152(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s5, 148(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s6, 144(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s7, 140(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s8, 136(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s9, 132(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s10, 128(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s11, 124(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    .cfi_offset ra, -4
+; RV32IM-NEXT:    .cfi_offset s0, -8
+; RV32IM-NEXT:    .cfi_offset s1, -12
+; RV32IM-NEXT:    .cfi_offset s2, -16
+; RV32IM-NEXT:    .cfi_offset s3, -20
+; RV32IM-NEXT:    .cfi_offset s4, -24
+; RV32IM-NEXT:    .cfi_offset s5, -28
+; RV32IM-NEXT:    .cfi_offset s6, -32
+; RV32IM-NEXT:    .cfi_offset s7, -36
+; RV32IM-NEXT:    .cfi_offset s8, -40
+; RV32IM-NEXT:    .cfi_offset s9, -44
+; RV32IM-NEXT:    .cfi_offset s10, -48
+; RV32IM-NEXT:    .cfi_offset s11, -52
+; RV32IM-NEXT:    sw a3, 112(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv a3, a2
+; RV32IM-NEXT:    mv t2, a1
+; RV32IM-NEXT:    sw a0, 96(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lui a0, 16
+; RV32IM-NEXT:    sw a4, 92(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a1, a4, 8
+; RV32IM-NEXT:    addi t3, a0, -256
+; RV32IM-NEXT:    and a0, a1, t3
+; RV32IM-NEXT:    srli a1, a4, 24
+; RV32IM-NEXT:    or a0, a0, a1
+; RV32IM-NEXT:    and a1, a4, t3
+; RV32IM-NEXT:    slli a1, a1, 8
+; RV32IM-NEXT:    slli a2, a4, 24
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    lui a2, 61681
+; RV32IM-NEXT:    or a0, a1, a0
+; RV32IM-NEXT:    addi s3, a2, -241
+; RV32IM-NEXT:    srli a1, a0, 4
+; RV32IM-NEXT:    and a0, a0, s3
+; RV32IM-NEXT:    and a1, a1, s3
+; RV32IM-NEXT:    slli a0, a0, 4
+; RV32IM-NEXT:    or a0, a1, a0
+; RV32IM-NEXT:    lui a1, 209715
+; RV32IM-NEXT:    srli a2, a0, 2
+; RV32IM-NEXT:    addi s2, a1, 819
+; RV32IM-NEXT:    and a1, a2, s2
+; RV32IM-NEXT:    and a0, a0, s2
+; RV32IM-NEXT:    slli a2, a0, 2
+; RV32IM-NEXT:    lui a0, 349525
+; RV32IM-NEXT:    or a1, a1, a2
+; RV32IM-NEXT:    addi t5, a0, 1365
+; RV32IM-NEXT:    srli a2, a1, 1
+; RV32IM-NEXT:    and a1, a1, t5
+; RV32IM-NEXT:    and a2, a2, t5
+; RV32IM-NEXT:    slli a1, a1, 1
+; RV32IM-NEXT:    or a7, a2, a1
+; RV32IM-NEXT:    sw a7, 72(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a1, a7, 8
+; RV32IM-NEXT:    and a1, a1, t3
+; RV32IM-NEXT:    srli a2, a7, 24
+; RV32IM-NEXT:    and a6, a7, t3
+; RV32IM-NEXT:    slli a7, a7, 24
+; RV32IM-NEXT:    slli a6, a6, 8
+; RV32IM-NEXT:    or a1, a1, a2
+; RV32IM-NEXT:    or a2, a7, a6
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    srli a2, a1, 4
+; RV32IM-NEXT:    and a1, a1, s3
+; RV32IM-NEXT:    and a2, a2, s3
+; RV32IM-NEXT:    slli a1, a1, 4
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    srli a2, a1, 2
+; RV32IM-NEXT:    sw a3, 88(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a6, a3, 8
+; RV32IM-NEXT:    and a2, a2, s2
+; RV32IM-NEXT:    and a6, a6, t3
+; RV32IM-NEXT:    srli a7, a3, 24
+; RV32IM-NEXT:    and t0, a3, t3
+; RV32IM-NEXT:    slli t0, t0, 8
+; RV32IM-NEXT:    slli t1, a3, 24
+; RV32IM-NEXT:    or a6, a6, a7
+; RV32IM-NEXT:    or a7, t1, t0
+; RV32IM-NEXT:    and a1, a1, s2
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    srli a7, a6, 4
+; RV32IM-NEXT:    and a6, a6, s3
+; RV32IM-NEXT:    and a7, a7, s3
+; RV32IM-NEXT:    slli a6, a6, 4
+; RV32IM-NEXT:    slli a1, a1, 2
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    srli a7, a6, 2
+; RV32IM-NEXT:    and a6, a6, s2
+; RV32IM-NEXT:    and a7, a7, s2
+; RV32IM-NEXT:    slli a6, a6, 2
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    or a2, a7, a6
+; RV32IM-NEXT:    srli a6, a2, 1
+; RV32IM-NEXT:    and a2, a2, t5
+; RV32IM-NEXT:    and a6, a6, t5
+; RV32IM-NEXT:    slli a2, a2, 1
+; RV32IM-NEXT:    srli a7, a1, 1
+; RV32IM-NEXT:    or t1, a6, a2
+; RV32IM-NEXT:    sw t1, 80(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a2, a7, t5
+; RV32IM-NEXT:    srli a6, t1, 8
+; RV32IM-NEXT:    and a1, a1, t5
+; RV32IM-NEXT:    and a6, a6, t3
+; RV32IM-NEXT:    srli a7, t1, 24
+; RV32IM-NEXT:    and t0, t1, t3
+; RV32IM-NEXT:    slli t1, t1, 24
+; RV32IM-NEXT:    slli t0, t0, 8
+; RV32IM-NEXT:    or a6, a6, a7
+; RV32IM-NEXT:    or a7, t1, t0
+; RV32IM-NEXT:    slli a1, a1, 1
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    srli a7, a6, 4
+; RV32IM-NEXT:    and a6, a6, s3
+; RV32IM-NEXT:    and a7, a7, s3
+; RV32IM-NEXT:    slli a6, a6, 4
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    or a2, a7, a6
+; RV32IM-NEXT:    srli a6, a2, 2
+; RV32IM-NEXT:    and a2, a2, s2
+; RV32IM-NEXT:    and a6, a6, s2
+; RV32IM-NEXT:    slli a2, a2, 2
+; RV32IM-NEXT:    lui a7, 69905
+; RV32IM-NEXT:    or a2, a6, a2
+; RV32IM-NEXT:    addi a0, a7, 273
+; RV32IM-NEXT:    srli a7, a2, 1
+; RV32IM-NEXT:    and a7, a7, t5
+; RV32IM-NEXT:    and a2, a2, t5
+; RV32IM-NEXT:    slli a2, a2, 1
+; RV32IM-NEXT:    lui t0, 139810
+; RV32IM-NEXT:    or a2, a7, a2
+; RV32IM-NEXT:    addi a6, t0, 546
+; RV32IM-NEXT:    and t6, a1, a0
+; RV32IM-NEXT:    and s0, a2, a6
+; RV32IM-NEXT:    and s1, a1, a6
+; RV32IM-NEXT:    and s4, a2, a0
+; RV32IM-NEXT:    mv a7, a0
+; RV32IM-NEXT:    mul s5, s0, t6
+; RV32IM-NEXT:    mul s6, s4, s1
+; RV32IM-NEXT:    lui t0, 559241
+; RV32IM-NEXT:    lui t1, 279620
+; RV32IM-NEXT:    addi t4, t0, -1912
+; RV32IM-NEXT:    addi t1, t1, 1092
+; RV32IM-NEXT:    and s7, a1, t4
+; RV32IM-NEXT:    and s8, a2, t1
+; RV32IM-NEXT:    and a1, a1, t1
+; RV32IM-NEXT:    and a2, a2, t4
+; RV32IM-NEXT:    mul s9, s8, s7
+; RV32IM-NEXT:    mul s10, a2, a1
+; RV32IM-NEXT:    mul s11, s0, s7
+; RV32IM-NEXT:    mul ra, s4, t6
+; RV32IM-NEXT:    mul a5, s8, a1
+; RV32IM-NEXT:    mul a4, a2, s1
+; RV32IM-NEXT:    mul a3, s0, s1
+; RV32IM-NEXT:    mul s0, s0, a1
+; RV32IM-NEXT:    mul a1, s4, a1
+; RV32IM-NEXT:    mul s4, s4, s7
+; RV32IM-NEXT:    mul s7, a2, s7
+; RV32IM-NEXT:    mul a0, s8, t6
+; RV32IM-NEXT:    mul s1, s8, s1
+; RV32IM-NEXT:    mul a2, a2, t6
+; RV32IM-NEXT:    xor t6, s6, s5
+; RV32IM-NEXT:    xor s5, s9, s10
+; RV32IM-NEXT:    xor s6, ra, s11
+; RV32IM-NEXT:    xor a4, a5, a4
+; RV32IM-NEXT:    xor a5, t6, s5
+; RV32IM-NEXT:    xor a4, s6, a4
+; RV32IM-NEXT:    and a5, a5, a6
+; RV32IM-NEXT:    and a4, a4, a7
+; RV32IM-NEXT:    mv s5, a7
+; RV32IM-NEXT:    xor a1, a1, a3
+; RV32IM-NEXT:    xor a0, a0, s7
+; RV32IM-NEXT:    xor a3, s4, s0
+; RV32IM-NEXT:    xor a2, s1, a2
 ; RV32IM-NEXT:    xor a0, a1, a0
-; RV32IM-NEXT:    xor a1, t5, t6
-; RV32IM-NEXT:    and a0, a0, t3
-; RV32IM-NEXT:    and a1, a1, s11
-; RV32IM-NEXT:    or a3, a3, s0
+; RV32IM-NEXT:    xor a2, a3, a2
+; RV32IM-NEXT:    and a0, a0, t1
+; RV32IM-NEXT:    and a1, a2, t4
+; RV32IM-NEXT:    or a4, a4, a5
 ; RV32IM-NEXT:    or a0, a0, a1
-; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    or a0, a4, a0
 ; RV32IM-NEXT:    srli a1, a0, 8
-; RV32IM-NEXT:    and a1, a1, t0
-; RV32IM-NEXT:    srli a3, a0, 24
-; RV32IM-NEXT:    and t0, a0, t0
+; RV32IM-NEXT:    and a1, a1, t3
+; RV32IM-NEXT:    srli a2, a0, 24
+; RV32IM-NEXT:    and a3, a0, t3
 ; RV32IM-NEXT:    slli a0, a0, 24
-; RV32IM-NEXT:    slli t0, t0, 8
-; RV32IM-NEXT:    or a1, a1, a3
-; RV32IM-NEXT:    or a0, a0, t0
+; RV32IM-NEXT:    slli a3, a3, 8
+; RV32IM-NEXT:    or a1, a1, a2
+; RV32IM-NEXT:    or a0, a0, a3
 ; RV32IM-NEXT:    or a0, a0, a1
 ; RV32IM-NEXT:    srli a1, a0, 4
-; RV32IM-NEXT:    and a0, a0, t1
-; RV32IM-NEXT:    and a1, a1, t1
+; RV32IM-NEXT:    and a0, a0, s3
+; RV32IM-NEXT:    and a1, a1, s3
 ; RV32IM-NEXT:    slli a0, a0, 4
 ; RV32IM-NEXT:    or a0, a1, a0
 ; RV32IM-NEXT:    srli a1, a0, 2
-; RV32IM-NEXT:    and a0, a0, t2
-; RV32IM-NEXT:    and a1, a1, t2
+; RV32IM-NEXT:    and a1, a1, s2
+; RV32IM-NEXT:    and a0, a0, s2
+; RV32IM-NEXT:    sw t2, 84(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    srli a2, t2, 8
 ; RV32IM-NEXT:    slli a0, a0, 2
-; RV32IM-NEXT:    or a0, a1, a0
-; RV32IM-NEXT:    srli a1, a0, 1
-; RV32IM-NEXT:    lui a3, 349525
-; RV32IM-NEXT:    addi a3, a3, 1364
-; RV32IM-NEXT:    and a0, a0, t4
-; RV32IM-NEXT:    and a1, a1, a3
-; RV32IM-NEXT:    slli a0, a0, 1
-; RV32IM-NEXT:    or a0, a1, a0
-; RV32IM-NEXT:    sw a0, 12(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    mv a7, a4
-; RV32IM-NEXT:    and t0, a2, a4
-; RV32IM-NEXT:    lw a6, 16(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    and a0, a6, a5
-; RV32IM-NEXT:    and t1, a2, a5
-; RV32IM-NEXT:    and a1, a6, a4
-; RV32IM-NEXT:    mul a3, a0, t0
-; RV32IM-NEXT:    mul t4, a1, t1
-; RV32IM-NEXT:    and t2, a2, s11
-; RV32IM-NEXT:    and t5, a6, t3
 ; RV32IM-NEXT:    and a2, a2, t3
-; RV32IM-NEXT:    and t6, a6, s11
-; RV32IM-NEXT:    mul s0, t5, t2
-; RV32IM-NEXT:    mul s2, t6, a2
-; RV32IM-NEXT:    mul s3, a0, t2
-; RV32IM-NEXT:    mul s4, a1, t0
-; RV32IM-NEXT:    mul s5, t5, a2
-; RV32IM-NEXT:    mul s6, t6, t1
-; RV32IM-NEXT:    mul s7, a0, t1
-; RV32IM-NEXT:    mul s8, a1, a2
-; RV32IM-NEXT:    mul s9, t5, t0
-; RV32IM-NEXT:    mul s10, t6, t2
-; RV32IM-NEXT:    mul a0, a0, a2
-; RV32IM-NEXT:    mul a1, a1, t2
-; RV32IM-NEXT:    mul t5, t5, t1
-; RV32IM-NEXT:    mul t6, t6, t0
-; RV32IM-NEXT:    xor a3, t4, a3
-; RV32IM-NEXT:    xor t4, s0, s2
-; RV32IM-NEXT:    xor s0, s4, s3
-; RV32IM-NEXT:    xor s2, s5, s6
-; RV32IM-NEXT:    xor a3, a3, t4
-; RV32IM-NEXT:    xor t4, s0, s2
-; RV32IM-NEXT:    and a3, a3, a5
-; RV32IM-NEXT:    and t4, t4, a4
-; RV32IM-NEXT:    xor s0, s8, s7
-; RV32IM-NEXT:    xor s2, s9, s10
+; RV32IM-NEXT:    srli a3, t2, 24
+; RV32IM-NEXT:    and a4, t2, t3
+; RV32IM-NEXT:    slli a4, a4, 8
+; RV32IM-NEXT:    slli a5, t2, 24
+; RV32IM-NEXT:    or a2, a2, a3
+; RV32IM-NEXT:    or a4, a5, a4
+; RV32IM-NEXT:    or a0, a1, a0
+; RV32IM-NEXT:    or a2, a4, a2
+; RV32IM-NEXT:    srli a1, a2, 4
+; RV32IM-NEXT:    and a2, a2, s3
+; RV32IM-NEXT:    and a1, a1, s3
+; RV32IM-NEXT:    slli a2, a2, 4
+; RV32IM-NEXT:    srli t0, a0, 1
+; RV32IM-NEXT:    or a1, a1, a2
+; RV32IM-NEXT:    srli a2, a1, 2
+; RV32IM-NEXT:    and a1, a1, s2
+; RV32IM-NEXT:    and a2, a2, s2
+; RV32IM-NEXT:    slli a1, a1, 2
+; RV32IM-NEXT:    lui t2, 349525
+; RV32IM-NEXT:    addi a7, t2, 1364
+; RV32IM-NEXT:    sw a7, 76(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    srli a2, a1, 1
+; RV32IM-NEXT:    and a1, a1, t5
+; RV32IM-NEXT:    and a2, a2, t5
+; RV32IM-NEXT:    slli a1, a1, 1
+; RV32IM-NEXT:    and t6, a0, t5
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    lw a3, 72(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mv a2, s5
+; RV32IM-NEXT:    sw s5, 100(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and s4, a3, s5
+; RV32IM-NEXT:    sw a6, 120(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a0, a1, a6
+; RV32IM-NEXT:    and s5, a3, a6
+; RV32IM-NEXT:    and a2, a1, a2
+; RV32IM-NEXT:    mul a6, a0, s4
+; RV32IM-NEXT:    mul a4, a2, s5
+; RV32IM-NEXT:    and a5, a3, t4
+; RV32IM-NEXT:    sw t1, 116(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and s0, a1, t1
+; RV32IM-NEXT:    and s9, a3, t1
+; RV32IM-NEXT:    and s1, a1, t4
+; RV32IM-NEXT:    mv a1, a5
+; RV32IM-NEXT:    mul a3, s0, a5
+; RV32IM-NEXT:    mul a5, s1, s9
+; RV32IM-NEXT:    mul s6, a0, a1
+; RV32IM-NEXT:    mv s7, a1
+; RV32IM-NEXT:    mv t1, a2
+; RV32IM-NEXT:    mul s8, a2, s4
+; RV32IM-NEXT:    mul s10, s0, s9
+; RV32IM-NEXT:    mul s11, s1, s5
+; RV32IM-NEXT:    mul ra, a0, s5
+; RV32IM-NEXT:    sw s5, 44(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv t2, a0
+; RV32IM-NEXT:    sw a0, 68(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a2, a2, s9
+; RV32IM-NEXT:    sw t1, 64(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a1, s0, s4
+; RV32IM-NEXT:    sw s4, 40(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s0, 60(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a0, s1, s7
+; RV32IM-NEXT:    sw s7, 8(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    sw s1, 72(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and t0, t0, a7
+; RV32IM-NEXT:    slli a7, t6, 1
+; RV32IM-NEXT:    or a7, t0, a7
+; RV32IM-NEXT:    sw a7, 52(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a4, a4, a6
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    xor a5, s8, s6
+; RV32IM-NEXT:    xor a6, s10, s11
+; RV32IM-NEXT:    xor a3, a4, a3
+; RV32IM-NEXT:    xor a4, a5, a6
+; RV32IM-NEXT:    xor a2, a2, ra
 ; RV32IM-NEXT:    xor a0, a1, a0
-; RV32IM-NEXT:    xor a1, t5, t6
-; RV32IM-NEXT:    xor t5, s0, s2
-; RV32IM-NEXT:    xor a0, a0, a1
-; RV32IM-NEXT:    and a1, t5, t3
-; RV32IM-NEXT:    and s0, a0, s11
-; RV32IM-NEXT:    or a0, t4, a3
-; RV32IM-NEXT:    sw a0, 16(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    or a1, a1, s0
-; RV32IM-NEXT:    sw a1, 8(sp) # 4-byte Folded Spill
-; RV32IM-NEXT:    lw a0, 24(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    and a3, a0, a4
-; RV32IM-NEXT:    lw a4, 20(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    and a1, a4, a5
-; RV32IM-NEXT:    and s2, a0, a5
-; RV32IM-NEXT:    mv s0, a5
-; RV32IM-NEXT:    and t4, a4, a7
-; RV32IM-NEXT:    mul s3, a1, a3
-; RV32IM-NEXT:    mul s4, t4, s2
-; RV32IM-NEXT:    mv s1, s11
-; RV32IM-NEXT:    and s5, a0, s11
-; RV32IM-NEXT:    and t5, a4, t3
-; RV32IM-NEXT:    and s6, a0, t3
-; RV32IM-NEXT:    and a0, a4, s11
-; RV32IM-NEXT:    mul t6, t5, s5
-; RV32IM-NEXT:    mul s7, a0, s6
-; RV32IM-NEXT:    mul s8, a1, s5
-; RV32IM-NEXT:    mul s9, t4, a3
-; RV32IM-NEXT:    mul s10, t5, s6
-; RV32IM-NEXT:    mul s11, a0, s2
-; RV32IM-NEXT:    mul ra, t4, s6
-; RV32IM-NEXT:    mul s6, a1, s6
-; RV32IM-NEXT:    mul a6, a0, s5
-; RV32IM-NEXT:    mul s5, t4, s5
-; RV32IM-NEXT:    mul a4, a1, s2
-; RV32IM-NEXT:    mul a5, t5, a3
-; RV32IM-NEXT:    mul s2, t5, s2
-; RV32IM-NEXT:    mul a3, a0, a3
-; RV32IM-NEXT:    xor s3, s4, s3
-; RV32IM-NEXT:    xor t6, t6, s7
-; RV32IM-NEXT:    xor s4, s9, s8
-; RV32IM-NEXT:    xor s7, s10, s11
-; RV32IM-NEXT:    xor t6, s3, t6
-; RV32IM-NEXT:    xor s3, s4, s7
-; RV32IM-NEXT:    mv s8, s0
-; RV32IM-NEXT:    and t6, t6, s0
-; RV32IM-NEXT:    and s3, s3, a7
-; RV32IM-NEXT:    xor a4, ra, a4
+; RV32IM-NEXT:    lw s8, 112(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    srli a1, s8, 8
+; RV32IM-NEXT:    sw t3, 36(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a5, s8, t3
+; RV32IM-NEXT:    and a1, a1, t3
+; RV32IM-NEXT:    slli a5, a5, 8
+; RV32IM-NEXT:    mul a6, t2, s9
+; RV32IM-NEXT:    mul t0, t1, s7
+; RV32IM-NEXT:    srli s6, s8, 24
+; RV32IM-NEXT:    slli s8, s8, 24
+; RV32IM-NEXT:    or a1, a1, s6
+; RV32IM-NEXT:    or a5, s8, a5
+; RV32IM-NEXT:    xor a0, a2, a0
+; RV32IM-NEXT:    or a1, a5, a1
+; RV32IM-NEXT:    mul a2, s0, s5
+; RV32IM-NEXT:    mul a5, s1, s4
+; RV32IM-NEXT:    srli s6, a1, 4
+; RV32IM-NEXT:    sw s3, 104(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a1, a1, s3
+; RV32IM-NEXT:    and s6, s6, s3
+; RV32IM-NEXT:    slli a1, a1, 4
+; RV32IM-NEXT:    xor a6, t0, a6
+; RV32IM-NEXT:    or a1, s6, a1
+; RV32IM-NEXT:    srli t0, a1, 2
+; RV32IM-NEXT:    sw s2, 56(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a1, a1, s2
+; RV32IM-NEXT:    and t0, t0, s2
+; RV32IM-NEXT:    slli a1, a1, 2
+; RV32IM-NEXT:    xor a2, a2, a5
+; RV32IM-NEXT:    or a1, t0, a1
+; RV32IM-NEXT:    srli a5, a1, 1
+; RV32IM-NEXT:    sw t5, 108(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a1, a1, t5
+; RV32IM-NEXT:    and a5, a5, t5
+; RV32IM-NEXT:    slli a1, a1, 1
+; RV32IM-NEXT:    xor a2, a6, a2
+; RV32IM-NEXT:    or a1, a5, a1
+; RV32IM-NEXT:    lw s4, 120(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t2, a3, s4
+; RV32IM-NEXT:    lw s5, 100(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a4, s5
+; RV32IM-NEXT:    sw a3, 32(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw t1, 116(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a0, a0, t1
+; RV32IM-NEXT:    sw a0, 28(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv s1, t4
+; RV32IM-NEXT:    and a0, a2, t4
+; RV32IM-NEXT:    sw a0, 24(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    and a4, a1, s5
+; RV32IM-NEXT:    and s8, a1, s4
+; RV32IM-NEXT:    and s10, a1, t4
+; RV32IM-NEXT:    and s11, a1, t1
+; RV32IM-NEXT:    lw a3, 80(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a0, a3, s4
+; RV32IM-NEXT:    and a1, a3, s5
+; RV32IM-NEXT:    and a2, a3, t1
+; RV32IM-NEXT:    and s3, a3, t4
+; RV32IM-NEXT:    mul a3, a0, a4
+; RV32IM-NEXT:    sw a3, 80(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a3, a1, s8
+; RV32IM-NEXT:    sw a3, 20(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a3, a2, s10
+; RV32IM-NEXT:    sw a3, 16(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul a3, s3, s11
+; RV32IM-NEXT:    sw a3, 12(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul s6, a0, s10
+; RV32IM-NEXT:    mul s0, a1, a4
+; RV32IM-NEXT:    mv a3, a4
+; RV32IM-NEXT:    sw a4, 48(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul t6, a2, s11
+; RV32IM-NEXT:    mul t5, s3, s8
+; RV32IM-NEXT:    mul t4, a0, s8
+; RV32IM-NEXT:    mul t3, a1, s11
+; RV32IM-NEXT:    mul t0, a2, a4
+; RV32IM-NEXT:    mul a7, s3, s10
+; RV32IM-NEXT:    mul a6, a0, s11
+; RV32IM-NEXT:    mul a5, a1, s10
+; RV32IM-NEXT:    mul a4, a2, s8
+; RV32IM-NEXT:    mul a3, s3, a3
+; RV32IM-NEXT:    lw t1, 32(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    or t2, t1, t2
+; RV32IM-NEXT:    lw t1, 28(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw ra, 24(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    or t1, t1, ra
+; RV32IM-NEXT:    lw ra, 80(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s2, 20(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    xor ra, s2, ra
+; RV32IM-NEXT:    lw s2, 16(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s7, 12(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    xor s2, s2, s7
+; RV32IM-NEXT:    xor s0, s0, s6
+; RV32IM-NEXT:    xor t5, t6, t5
+; RV32IM-NEXT:    xor t6, ra, s2
+; RV32IM-NEXT:    xor t5, s0, t5
+; RV32IM-NEXT:    xor t3, t3, t4
+; RV32IM-NEXT:    xor a7, t0, a7
 ; RV32IM-NEXT:    xor a5, a5, a6
-; RV32IM-NEXT:    xor a6, s5, s6
-; RV32IM-NEXT:    xor a3, s2, a3
-; RV32IM-NEXT:    xor a4, a4, a5
-; RV32IM-NEXT:    xor a3, a6, a3
-; RV32IM-NEXT:    and a4, a4, t3
-; RV32IM-NEXT:    mv s9, s1
+; RV32IM-NEXT:    xor a3, a4, a3
+; RV32IM-NEXT:    xor a4, t3, a7
+; RV32IM-NEXT:    xor a3, a5, a3
+; RV32IM-NEXT:    and a5, t6, s4
+; RV32IM-NEXT:    and a6, t5, s5
+; RV32IM-NEXT:    lw s4, 116(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s4
 ; RV32IM-NEXT:    and a3, a3, s1
-; RV32IM-NEXT:    or a5, s3, t6
+; RV32IM-NEXT:    mv s6, s1
+; RV32IM-NEXT:    or a5, a6, a5
 ; RV32IM-NEXT:    or a3, a4, a3
-; RV32IM-NEXT:    lw a4, 16(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s0, 8(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    or s0, a4, s0
+; RV32IM-NEXT:    or a4, t2, t1
 ; RV32IM-NEXT:    or a3, a5, a3
-; RV32IM-NEXT:    lw a4, 12(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    xor a3, a3, a4
+; RV32IM-NEXT:    lw a4, 52(sp) # 4-byte Folded Reload
 ; RV32IM-NEXT:    srli a4, a4, 1
-; RV32IM-NEXT:    xor a3, a3, s0
-; RV32IM-NEXT:    mul a5, a1, t0
-; RV32IM-NEXT:    mul a6, t4, t1
-; RV32IM-NEXT:    mul t6, t5, t2
-; RV32IM-NEXT:    mul s0, a0, a2
-; RV32IM-NEXT:    mul s1, a1, t2
-; RV32IM-NEXT:    mul s2, t4, t0
-; RV32IM-NEXT:    mul s3, t5, a2
-; RV32IM-NEXT:    mul s4, a0, t1
-; RV32IM-NEXT:    mul s5, a1, t1
-; RV32IM-NEXT:    mul s6, t4, a2
-; RV32IM-NEXT:    mul s7, t5, t0
-; RV32IM-NEXT:    mul a1, a1, a2
-; RV32IM-NEXT:    mul a2, a0, t2
-; RV32IM-NEXT:    mul t2, t4, t2
-; RV32IM-NEXT:    mul t1, t5, t1
-; RV32IM-NEXT:    mul a0, a0, t0
-; RV32IM-NEXT:    xor a5, a6, a5
-; RV32IM-NEXT:    xor a6, t6, s0
-; RV32IM-NEXT:    xor t0, s2, s1
-; RV32IM-NEXT:    xor t4, s3, s4
-; RV32IM-NEXT:    xor a5, a5, a6
-; RV32IM-NEXT:    xor a6, t0, t4
-; RV32IM-NEXT:    and a5, a5, s8
-; RV32IM-NEXT:    and a6, a6, a7
-; RV32IM-NEXT:    xor t0, s6, s5
-; RV32IM-NEXT:    xor a2, s7, a2
-; RV32IM-NEXT:    xor a1, t2, a1
-; RV32IM-NEXT:    xor a0, t1, a0
-; RV32IM-NEXT:    xor a2, t0, a2
-; RV32IM-NEXT:    xor a0, a1, a0
-; RV32IM-NEXT:    and a1, a2, t3
-; RV32IM-NEXT:    and a0, a0, s9
-; RV32IM-NEXT:    or a2, a6, a5
-; RV32IM-NEXT:    or a0, a1, a0
-; RV32IM-NEXT:    xor a1, a4, a3
+; RV32IM-NEXT:    xor a3, a4, a3
+; RV32IM-NEXT:    srli a4, a3, 8
+; RV32IM-NEXT:    lw s2, 36(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s2
+; RV32IM-NEXT:    srli a5, a3, 24
+; RV32IM-NEXT:    and a6, a3, s2
+; RV32IM-NEXT:    slli a3, a3, 24
+; RV32IM-NEXT:    slli a6, a6, 8
+; RV32IM-NEXT:    or a4, a4, a5
+; RV32IM-NEXT:    or a3, a3, a6
+; RV32IM-NEXT:    or a3, a3, a4
+; RV32IM-NEXT:    srli a4, a3, 4
+; RV32IM-NEXT:    lw a5, 104(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a3, a5
+; RV32IM-NEXT:    and a4, a4, a5
+; RV32IM-NEXT:    slli a3, a3, 4
+; RV32IM-NEXT:    or ra, a4, a3
+; RV32IM-NEXT:    lw s0, 40(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a3, a0, s0
+; RV32IM-NEXT:    lw s1, 44(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a4, a1, s1
+; RV32IM-NEXT:    lw s7, 8(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a5, a2, s7
+; RV32IM-NEXT:    mul a6, s3, s9
+; RV32IM-NEXT:    mul a7, a0, s7
+; RV32IM-NEXT:    mul t0, a1, s0
+; RV32IM-NEXT:    mul t1, a2, s9
+; RV32IM-NEXT:    mul t2, s3, s1
+; RV32IM-NEXT:    mul t3, a0, s1
+; RV32IM-NEXT:    mul t4, a1, s9
+; RV32IM-NEXT:    mul t5, a2, s0
+; RV32IM-NEXT:    mul t6, s3, s7
+; RV32IM-NEXT:    mul a0, a0, s9
+; RV32IM-NEXT:    mul a1, a1, s7
+; RV32IM-NEXT:    mul a2, a2, s1
+; RV32IM-NEXT:    mul s0, s3, s0
+; RV32IM-NEXT:    xor a3, a4, a3
+; RV32IM-NEXT:    xor a4, a5, a6
+; RV32IM-NEXT:    xor a5, t0, a7
+; RV32IM-NEXT:    xor a6, t1, t2
+; RV32IM-NEXT:    xor a3, a3, a4
+; RV32IM-NEXT:    xor a4, a5, a6
+; RV32IM-NEXT:    xor a5, t4, t3
+; RV32IM-NEXT:    xor a6, t5, t6
+; RV32IM-NEXT:    xor a5, a5, a6
+; RV32IM-NEXT:    xor a0, a1, a0
+; RV32IM-NEXT:    xor a2, a2, s0
+; RV32IM-NEXT:    srli a1, ra, 2
+; RV32IM-NEXT:    xor a0, a0, a2
+; RV32IM-NEXT:    lw s0, 56(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, s0
+; RV32IM-NEXT:    lw s7, 120(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a2, a3, s7
+; RV32IM-NEXT:    mv s1, s5
+; RV32IM-NEXT:    and a3, a4, s5
+; RV32IM-NEXT:    and a4, a5, s4
+; RV32IM-NEXT:    mv s5, s4
+; RV32IM-NEXT:    mv s9, s6
+; RV32IM-NEXT:    and a0, a0, s6
+; RV32IM-NEXT:    or a2, a3, a2
+; RV32IM-NEXT:    or a0, a4, a0
+; RV32IM-NEXT:    and a3, ra, s0
 ; RV32IM-NEXT:    or a0, a2, a0
+; RV32IM-NEXT:    slli a3, a3, 2
+; RV32IM-NEXT:    srli a2, a0, 8
+; RV32IM-NEXT:    or a1, a1, a3
+; RV32IM-NEXT:    and a2, a2, s2
+; RV32IM-NEXT:    srli a3, a0, 24
+; RV32IM-NEXT:    and a4, a0, s2
+; RV32IM-NEXT:    lw t3, 68(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s4, 48(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a5, t3, s4
+; RV32IM-NEXT:    lw t4, 64(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a6, t4, s8
+; RV32IM-NEXT:    lw t5, 60(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a7, t5, s10
+; RV32IM-NEXT:    lw t6, 72(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul t0, t6, s11
+; RV32IM-NEXT:    slli a0, a0, 24
+; RV32IM-NEXT:    slli a4, a4, 8
+; RV32IM-NEXT:    or a2, a2, a3
+; RV32IM-NEXT:    or a0, a0, a4
+; RV32IM-NEXT:    srli a3, a1, 1
+; RV32IM-NEXT:    lw s3, 108(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a1, a1, s3
 ; RV32IM-NEXT:    lw ra, 76(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s0, 72(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s1, 68(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s2, 64(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s3, 60(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s4, 56(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s5, 52(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s6, 48(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s7, 44(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s8, 40(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s9, 36(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s10, 32(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    lw s11, 28(sp) # 4-byte Folded Reload
-; RV32IM-NEXT:    addi sp, sp, 80
+; RV32IM-NEXT:    and a3, a3, ra
+; RV32IM-NEXT:    slli a1, a1, 1
+; RV32IM-NEXT:    or a1, a3, a1
+; RV32IM-NEXT:    sw a1, 80(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a0, a0, a2
+; RV32IM-NEXT:    xor a1, a6, a5
+; RV32IM-NEXT:    xor a2, a7, t0
+; RV32IM-NEXT:    mul a3, t3, s10
+; RV32IM-NEXT:    mul a4, t4, s4
+; RV32IM-NEXT:    mul a5, t5, s11
+; RV32IM-NEXT:    mul a6, t6, s8
+; RV32IM-NEXT:    mul a7, t3, s8
+; RV32IM-NEXT:    mul t0, t3, s11
+; RV32IM-NEXT:    mul t1, t4, s11
+; RV32IM-NEXT:    mul t2, t4, s10
+; RV32IM-NEXT:    mul t3, t6, s10
+; RV32IM-NEXT:    mul t4, t5, s4
+; RV32IM-NEXT:    mul t5, t5, s8
+; RV32IM-NEXT:    mul t6, t6, s4
+; RV32IM-NEXT:    xor a1, a1, a2
+; RV32IM-NEXT:    xor a3, a4, a3
+; RV32IM-NEXT:    xor a2, a5, a6
+; RV32IM-NEXT:    srli a4, a0, 4
+; RV32IM-NEXT:    lw s6, 104(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a4, a4, s6
+; RV32IM-NEXT:    xor a2, a3, a2
+; RV32IM-NEXT:    and a1, a1, s7
+; RV32IM-NEXT:    and a2, a2, s1
+; RV32IM-NEXT:    xor a3, t1, a7
+; RV32IM-NEXT:    xor a5, t4, t3
+; RV32IM-NEXT:    xor a6, t2, t0
+; RV32IM-NEXT:    xor a7, t5, t6
+; RV32IM-NEXT:    xor a3, a3, a5
+; RV32IM-NEXT:    xor a5, a6, a7
+; RV32IM-NEXT:    mv s11, s5
+; RV32IM-NEXT:    and a3, a3, s5
+; RV32IM-NEXT:    mv t4, s9
+; RV32IM-NEXT:    and a5, a5, s9
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    or a3, a3, a5
+; RV32IM-NEXT:    and a0, a0, s6
+; RV32IM-NEXT:    or a1, a1, a3
+; RV32IM-NEXT:    slli a0, a0, 4
+; RV32IM-NEXT:    srli a2, a1, 8
+; RV32IM-NEXT:    or a0, a4, a0
+; RV32IM-NEXT:    and a2, a2, s2
+; RV32IM-NEXT:    srli a3, a1, 24
+; RV32IM-NEXT:    and a4, a1, s2
+; RV32IM-NEXT:    slli a1, a1, 24
+; RV32IM-NEXT:    slli a4, a4, 8
+; RV32IM-NEXT:    or a2, a2, a3
+; RV32IM-NEXT:    or a1, a1, a4
+; RV32IM-NEXT:    srli a3, a0, 2
+; RV32IM-NEXT:    or a1, a1, a2
+; RV32IM-NEXT:    srli a2, a1, 4
+; RV32IM-NEXT:    and a1, a1, s6
+; RV32IM-NEXT:    and a2, a2, s6
+; RV32IM-NEXT:    slli a1, a1, 4
+; RV32IM-NEXT:    and a4, a3, s0
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    and a0, a0, s0
+; RV32IM-NEXT:    srli a2, a1, 2
+; RV32IM-NEXT:    and a2, a2, s0
+; RV32IM-NEXT:    and a1, a1, s0
+; RV32IM-NEXT:    lw a5, 112(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t0, a5, s1
+; RV32IM-NEXT:    lw a3, 88(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a6, a3, s7
+; RV32IM-NEXT:    and t3, a5, s7
+; RV32IM-NEXT:    and a7, a3, s1
+; RV32IM-NEXT:    mv s8, s1
+; RV32IM-NEXT:    mul s9, a6, t0
+; RV32IM-NEXT:    mul t1, a7, t3
+; RV32IM-NEXT:    mv s0, t4
+; RV32IM-NEXT:    and t4, a5, t4
+; RV32IM-NEXT:    and t5, a3, s5
+; RV32IM-NEXT:    and t2, a5, s5
+; RV32IM-NEXT:    and t6, a3, s0
+; RV32IM-NEXT:    mv a3, s0
+; RV32IM-NEXT:    mul s0, t5, t4
+; RV32IM-NEXT:    mul s1, t6, t2
+; RV32IM-NEXT:    mul s2, a6, t4
+; RV32IM-NEXT:    mul s4, a7, t0
+; RV32IM-NEXT:    mul s5, t5, t2
+; RV32IM-NEXT:    mul s6, t6, t3
+; RV32IM-NEXT:    slli a0, a0, 2
+; RV32IM-NEXT:    slli a1, a1, 2
+; RV32IM-NEXT:    or a0, a4, a0
+; RV32IM-NEXT:    sw a0, 112(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    or a1, a2, a1
+; RV32IM-NEXT:    srli a0, a1, 1
+; RV32IM-NEXT:    and a1, a1, s3
+; RV32IM-NEXT:    and a0, a0, ra
+; RV32IM-NEXT:    sw a0, 76(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    slli a1, a1, 1
+; RV32IM-NEXT:    sw a1, 88(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor a0, t1, s9
+; RV32IM-NEXT:    xor s0, s0, s1
+; RV32IM-NEXT:    xor a1, s4, s2
+; RV32IM-NEXT:    xor a4, s5, s6
+; RV32IM-NEXT:    xor a0, a0, s0
+; RV32IM-NEXT:    xor a1, a1, a4
+; RV32IM-NEXT:    and a0, a0, s7
+; RV32IM-NEXT:    sw a0, 72(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mv a2, s8
+; RV32IM-NEXT:    and t1, a1, s8
+; RV32IM-NEXT:    mul s0, a6, t3
+; RV32IM-NEXT:    mul s1, a7, t2
+; RV32IM-NEXT:    sw t0, 104(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    mul s2, t5, t0
+; RV32IM-NEXT:    mul s4, t6, t4
+; RV32IM-NEXT:    mul a6, a6, t2
+; RV32IM-NEXT:    mul a7, a7, t4
+; RV32IM-NEXT:    mul s5, t5, t3
+; RV32IM-NEXT:    mul s6, t6, t0
+; RV32IM-NEXT:    lw a0, 92(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s8, a0, s8
+; RV32IM-NEXT:    mv a5, a2
+; RV32IM-NEXT:    and s9, a0, s7
+; RV32IM-NEXT:    and s10, a0, a3
+; RV32IM-NEXT:    mv a1, s11
+; RV32IM-NEXT:    and s11, a0, s11
+; RV32IM-NEXT:    lw a2, 84(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and t5, a2, s7
+; RV32IM-NEXT:    and a4, a2, a5
+; RV32IM-NEXT:    and a0, a2, a1
+; RV32IM-NEXT:    and a5, a2, a3
+; RV32IM-NEXT:    mul ra, t5, s8
+; RV32IM-NEXT:    mul t0, a4, s9
+; RV32IM-NEXT:    mul t6, a0, s10
+; RV32IM-NEXT:    mul a2, a5, s11
+; RV32IM-NEXT:    lw s3, 88(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s7, 76(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    or s3, s7, s3
+; RV32IM-NEXT:    sw s3, 92(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    lw s3, 72(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    or t1, t1, s3
+; RV32IM-NEXT:    sw t1, 88(sp) # 4-byte Folded Spill
+; RV32IM-NEXT:    xor s0, s1, s0
+; RV32IM-NEXT:    xor t1, s2, s4
+; RV32IM-NEXT:    xor a6, a7, a6
+; RV32IM-NEXT:    xor a7, s5, s6
+; RV32IM-NEXT:    xor t1, s0, t1
+; RV32IM-NEXT:    xor a6, a6, a7
+; RV32IM-NEXT:    and a7, t1, a1
+; RV32IM-NEXT:    mv a1, a3
+; RV32IM-NEXT:    and a6, a6, a3
+; RV32IM-NEXT:    xor a3, t0, ra
+; RV32IM-NEXT:    xor a2, t6, a2
+; RV32IM-NEXT:    mul t1, t5, s10
+; RV32IM-NEXT:    mul t6, a4, s8
+; RV32IM-NEXT:    mul s0, a0, s11
+; RV32IM-NEXT:    mul s1, a5, s9
+; RV32IM-NEXT:    mul s2, t5, s9
+; RV32IM-NEXT:    mul s3, a4, s11
+; RV32IM-NEXT:    mul s4, a0, s8
+; RV32IM-NEXT:    mul s5, a5, s10
+; RV32IM-NEXT:    mul s6, t5, s11
+; RV32IM-NEXT:    mul s10, a4, s10
+; RV32IM-NEXT:    mul s9, a0, s9
+; RV32IM-NEXT:    mul s8, a5, s8
+; RV32IM-NEXT:    or a6, a7, a6
+; RV32IM-NEXT:    xor a2, a3, a2
+; RV32IM-NEXT:    xor a3, t6, t1
+; RV32IM-NEXT:    xor s0, s0, s1
+; RV32IM-NEXT:    xor a3, a3, s0
+; RV32IM-NEXT:    xor a7, s3, s2
+; RV32IM-NEXT:    xor t1, s4, s5
+; RV32IM-NEXT:    lw t6, 80(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    srli t6, t6, 1
+; RV32IM-NEXT:    xor s0, s10, s6
+; RV32IM-NEXT:    lw t0, 112(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    srli s1, t0, 1
+; RV32IM-NEXT:    xor s2, s9, s8
+; RV32IM-NEXT:    slli s3, s1, 31
+; RV32IM-NEXT:    lw s7, 120(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a2, a2, s7
+; RV32IM-NEXT:    lw s11, 100(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a3, a3, s11
+; RV32IM-NEXT:    xor a7, a7, t1
+; RV32IM-NEXT:    xor t1, s0, s2
+; RV32IM-NEXT:    lw ra, 116(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and a7, a7, ra
+; RV32IM-NEXT:    and t1, t1, a1
+; RV32IM-NEXT:    mv s10, a1
+; RV32IM-NEXT:    or a2, a3, a2
+; RV32IM-NEXT:    or a3, a7, t1
+; RV32IM-NEXT:    lw a1, 88(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    or a6, a1, a6
+; RV32IM-NEXT:    or a2, a2, a3
+; RV32IM-NEXT:    lw a3, 92(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    srli a3, a3, 1
+; RV32IM-NEXT:    xor s0, a2, a6
+; RV32IM-NEXT:    or t6, t6, s3
+; RV32IM-NEXT:    xor s0, a3, s0
+; RV32IM-NEXT:    lw a2, 108(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    and s1, s1, a2
+; RV32IM-NEXT:    and a2, t0, a2
+; RV32IM-NEXT:    lw a1, 104(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    mul a3, t5, a1
+; RV32IM-NEXT:    mul a6, a4, t3
+; RV32IM-NEXT:    mul a7, a0, t4
+; RV32IM-NEXT:    mul s8, a5, t2
+; RV32IM-NEXT:    mul t1, t5, t4
+; RV32IM-NEXT:    mul s2, a4, a1
+; RV32IM-NEXT:    mul s3, a0, t2
+; RV32IM-NEXT:    mul s4, a5, t3
+; RV32IM-NEXT:    mul s5, t5, t3
+; RV32IM-NEXT:    mul s6, a4, t2
+; RV32IM-NEXT:    mul s9, t5, t2
+; RV32IM-NEXT:    mul t5, a0, a1
+; RV32IM-NEXT:    mul a4, a4, t4
+; RV32IM-NEXT:    mul t4, a5, t4
+; RV32IM-NEXT:    mul a0, a0, t3
+; RV32IM-NEXT:    mul a1, a5, a1
+; RV32IM-NEXT:    xor a3, a6, a3
+; RV32IM-NEXT:    xor a6, a7, s8
+; RV32IM-NEXT:    xor a7, s2, t1
+; RV32IM-NEXT:    xor t0, s3, s4
+; RV32IM-NEXT:    xor a3, a3, a6
+; RV32IM-NEXT:    xor a6, a7, t0
+; RV32IM-NEXT:    and a3, a3, s7
+; RV32IM-NEXT:    and a6, a6, s11
+; RV32IM-NEXT:    xor a7, s6, s5
+; RV32IM-NEXT:    xor t0, t5, t4
+; RV32IM-NEXT:    xor a4, a4, s9
+; RV32IM-NEXT:    xor a0, a0, a1
+; RV32IM-NEXT:    xor a1, a7, t0
+; RV32IM-NEXT:    xor a0, a4, a0
+; RV32IM-NEXT:    and a1, a1, ra
+; RV32IM-NEXT:    and a0, a0, s10
+; RV32IM-NEXT:    or a3, a6, a3
+; RV32IM-NEXT:    or a0, a1, a0
+; RV32IM-NEXT:    or a0, a3, a0
+; RV32IM-NEXT:    slli a2, a2, 1
+; RV32IM-NEXT:    or a2, s1, a2
+; RV32IM-NEXT:    lw a1, 96(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    sw a0, 0(a1)
+; RV32IM-NEXT:    sw s0, 4(a1)
+; RV32IM-NEXT:    srli a2, a2, 1
+; RV32IM-NEXT:    sw t6, 8(a1)
+; RV32IM-NEXT:    sw a2, 12(a1)
+; RV32IM-NEXT:    lw ra, 172(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s0, 168(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s1, 164(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s2, 160(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s3, 156(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s4, 152(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s5, 148(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s6, 144(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s7, 140(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s8, 136(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s9, 132(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s10, 128(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    lw s11, 124(sp) # 4-byte Folded Reload
+; RV32IM-NEXT:    .cfi_restore ra
+; RV32IM-NEXT:    .cfi_restore s0
+; RV32IM-NEXT:    .cfi_restore s1
+; RV32IM-NEXT:    .cfi_restore s2
+; RV32IM-NEXT:    .cfi_restore s3
+; RV32IM-NEXT:    .cfi_restore s4
+; RV32IM-NEXT:    .cfi_restore s5
+; RV32IM-NEXT:    .cfi_restore s6
+; RV32IM-NEXT:    .cfi_restore s7
+; RV32IM-NEXT:    .cfi_restore s8
+; RV32IM-NEXT:    .cfi_restore s9
+; RV32IM-NEXT:    .cfi_restore s10
+; RV32IM-NEXT:    .cfi_restore s11
+; RV32IM-NEXT:    addi sp, sp, 176
+; RV32IM-NEXT:    .cfi_def_cfa_offset 0
 ; RV32IM-NEXT:    ret
 ;
-; RV64IM-LABEL: clmul_i64:
+; RV64IM-LABEL: clmul_i128_zext:
 ; RV64IM:       # %bb.0:
-; RV64IM-NEXT:    addi sp, sp, -32
-; RV64IM-NEXT:    sd s0, 24(sp) # 8-byte Folded Spill
-; RV64IM-NEXT:    sd s1, 16(sp) # 8-byte Folded Spill
-; RV64IM-NEXT:    sd s2, 8(sp) # 8-byte Folded Spill
-; RV64IM-NEXT:    sd s3, 0(sp) # 8-byte Folded Spill
-; RV64IM-NEXT:    lui a2, 69905
-; RV64IM-NEXT:    lui a3, 139810
-; RV64IM-NEXT:    addi a2, a2, 273
-; RV64IM-NEXT:    addi a3, a3, 546
-; RV64IM-NEXT:    slli a4, a2, 32
-; RV64IM-NEXT:    slli a5, a3, 32
-; RV64IM-NEXT:    add a2, a2, a4
-; RV64IM-NEXT:    add a3, a3, a5
-; RV64IM-NEXT:    and a4, a1, a2
-; RV64IM-NEXT:    and a5, a0, a3
-; RV64IM-NEXT:    mul a6, a5, a4
-; RV64IM-NEXT:    and a7, a1, a3
-; RV64IM-NEXT:    and t0, a0, a2
-; RV64IM-NEXT:    lui t1, 279620
-; RV64IM-NEXT:    mul t2, t0, a7
-; RV64IM-NEXT:    addi t1, t1, 1092
-; RV64IM-NEXT:    lui t3, %hi(.LCPI4_0)
-; RV64IM-NEXT:    slli t4, t1, 32
-; RV64IM-NEXT:    ld t3, %lo(.LCPI4_0)(t3)
-; RV64IM-NEXT:    add t1, t1, t4
-; RV64IM-NEXT:    and t4, a0, t1
-; RV64IM-NEXT:    and t5, a1, t1
-; RV64IM-NEXT:    mul t6, t0, a4
-; RV64IM-NEXT:    mul s0, t4, t5
-; RV64IM-NEXT:    xor a6, t2, a6
-; RV64IM-NEXT:    and a1, a1, t3
-; RV64IM-NEXT:    mul t2, t4, a1
+; RV64IM-NEXT:    addi sp, sp, -112
+; RV64IM-NEXT:    .cfi_def_cfa_offset 112
+; RV64IM-NEXT:    sd ra, 104(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s0, 96(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s1, 88(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s2, 80(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s3, 72(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s4, 64(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s5, 56(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s6, 48(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s7, 40(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s8, 32(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s9, 24(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s10, 16(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    sd s11, 8(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    .cfi_offset ra, -8
+; RV64IM-NEXT:    .cfi_offset s0, -16
+; RV64IM-NEXT:    .cfi_offset s1, -24
+; RV64IM-NEXT:    .cfi_offset s2, -32
+; RV64IM-NEXT:    .cfi_offset s3, -40
+; RV64IM-NEXT:    .cfi_offset s4, -48
+; RV64IM-NEXT:    .cfi_offset s5, -56
+; RV64IM-NEXT:    .cfi_offset s6, -64
+; RV64IM-NEXT:    .cfi_offset s7, -72
+; RV64IM-NEXT:    .cfi_offset s8, -80
+; RV64IM-NEXT:    .cfi_offset s9, -88
+; RV64IM-NEXT:    .cfi_offset s10, -96
+; RV64IM-NEXT:    .cfi_offset s11, -104
+; RV64IM-NEXT:    mv a2, a0
+; RV64IM-NEXT:    lui a4, 4080
+; RV64IM-NEXT:    srli a3, a1, 24
+; RV64IM-NEXT:    li a5, 255
+; RV64IM-NEXT:    and a4, a3, a4
+; RV64IM-NEXT:    lui a0, 4080
+; RV64IM-NEXT:    srli a3, a1, 8
+; RV64IM-NEXT:    slli t3, a5, 24
+; RV64IM-NEXT:    and a6, a3, t3
+; RV64IM-NEXT:    sd t3, 0(sp) # 8-byte Folded Spill
+; RV64IM-NEXT:    lui a3, 16
+; RV64IM-NEXT:    srli a7, a1, 40
+; RV64IM-NEXT:    addi a5, a3, -256
+; RV64IM-NEXT:    and a7, a7, a5
+; RV64IM-NEXT:    srli t0, a1, 56
+; RV64IM-NEXT:    or a4, a6, a4
+; RV64IM-NEXT:    or a6, a7, t0
+; RV64IM-NEXT:    or a4, a4, a6
+; RV64IM-NEXT:    and a6, a1, a0
+; RV64IM-NEXT:    srliw a7, a1, 24
+; RV64IM-NEXT:    slli a6, a6, 24
+; RV64IM-NEXT:    slli a7, a7, 32
+; RV64IM-NEXT:    or a6, a6, a7
+; RV64IM-NEXT:    and a7, a1, a5
+; RV64IM-NEXT:    slli a7, a7, 40
+; RV64IM-NEXT:    slli t0, a1, 56
+; RV64IM-NEXT:    or a7, t0, a7
+; RV64IM-NEXT:    lui t0, 61681
+; RV64IM-NEXT:    or a6, a7, a6
+; RV64IM-NEXT:    addi a7, t0, -241
+; RV64IM-NEXT:    or a4, a6, a4
+; RV64IM-NEXT:    slli a6, a7, 32
+; RV64IM-NEXT:    srli t0, a4, 4
+; RV64IM-NEXT:    add a6, a7, a6
+; RV64IM-NEXT:    and a7, t0, a6
+; RV64IM-NEXT:    and a4, a4, a6
+; RV64IM-NEXT:    lui t0, 209715
+; RV64IM-NEXT:    slli a4, a4, 4
+; RV64IM-NEXT:    addi t0, t0, 819
+; RV64IM-NEXT:    or a4, a7, a4
+; RV64IM-NEXT:    slli a7, t0, 32
+; RV64IM-NEXT:    srli t1, a4, 2
+; RV64IM-NEXT:    add t0, t0, a7
+; RV64IM-NEXT:    and a7, t1, t0
+; RV64IM-NEXT:    lui t1, 349525
+; RV64IM-NEXT:    and a4, a4, t0
+; RV64IM-NEXT:    addi t1, t1, 1365
+; RV64IM-NEXT:    slli a4, a4, 2
+; RV64IM-NEXT:    slli t2, t1, 32
+; RV64IM-NEXT:    or a7, a7, a4
+; RV64IM-NEXT:    add a4, t1, t2
+; RV64IM-NEXT:    srli t1, a7, 1
+; RV64IM-NEXT:    and a7, a7, a4
+; RV64IM-NEXT:    and t1, t1, a4
+; RV64IM-NEXT:    slli a7, a7, 1
+; RV64IM-NEXT:    or t4, t1, a7
+; RV64IM-NEXT:    srli a7, a2, 24
+; RV64IM-NEXT:    srli t1, a2, 8
+; RV64IM-NEXT:    and a7, a7, a0
+; RV64IM-NEXT:    and t1, t1, t3
+; RV64IM-NEXT:    or a7, t1, a7
+; RV64IM-NEXT:    srli t1, a2, 40
+; RV64IM-NEXT:    and t1, t1, a5
+; RV64IM-NEXT:    srli t2, a2, 56
+; RV64IM-NEXT:    or t1, t1, t2
+; RV64IM-NEXT:    and t2, a2, a0
+; RV64IM-NEXT:    slli t2, t2, 24
+; RV64IM-NEXT:    srliw t3, a2, 24
+; RV64IM-NEXT:    slli t3, t3, 32
+; RV64IM-NEXT:    and t5, a2, a5
+; RV64IM-NEXT:    slli t5, t5, 40
+; RV64IM-NEXT:    slli t6, a2, 56
+; RV64IM-NEXT:    or t2, t2, t3
+; RV64IM-NEXT:    or t3, t6, t5
+; RV64IM-NEXT:    or a7, a7, t1
+; RV64IM-NEXT:    or t1, t3, t2
+; RV64IM-NEXT:    lui t2, 69905
+; RV64IM-NEXT:    or a7, t1, a7
+; RV64IM-NEXT:    srli t1, a7, 4
+; RV64IM-NEXT:    and a7, a7, a6
+; RV64IM-NEXT:    and t1, t1, a6
+; RV64IM-NEXT:    slli a7, a7, 4
+; RV64IM-NEXT:    addi t2, t2, 273
+; RV64IM-NEXT:    or a7, t1, a7
+; RV64IM-NEXT:    srli t1, a7, 2
+; RV64IM-NEXT:    and a7, a7, t0
+; RV64IM-NEXT:    and t1, t1, t0
+; RV64IM-NEXT:    slli a7, a7, 2
+; RV64IM-NEXT:    slli t3, t2, 32
+; RV64IM-NEXT:    or t1, t1, a7
+; RV64IM-NEXT:    add a7, t2, t3
+; RV64IM-NEXT:    srli t2, t1, 1
+; RV64IM-NEXT:    and t2, t2, a4
+; RV64IM-NEXT:    lui t3, 139810
+; RV64IM-NEXT:    and t1, t1, a4
+; RV64IM-NEXT:    addi t3, t3, 546
+; RV64IM-NEXT:    slli t1, t1, 1
+; RV64IM-NEXT:    slli t5, t3, 32
+; RV64IM-NEXT:    or t6, t2, t1
+; RV64IM-NEXT:    add t1, t3, t5
+; RV64IM-NEXT:    and t5, t4, a7
+; RV64IM-NEXT:    and s0, t6, t1
+; RV64IM-NEXT:    mul s1, s0, t5
+; RV64IM-NEXT:    lui t2, %hi(.LCPI9_0)
+; RV64IM-NEXT:    ld t2, %lo(.LCPI9_0)(t2)
+; RV64IM-NEXT:    lui t3, 279620
+; RV64IM-NEXT:    and s2, t4, t1
+; RV64IM-NEXT:    addi t3, t3, 1092
+; RV64IM-NEXT:    and s3, t6, a7
+; RV64IM-NEXT:    slli s4, t3, 32
+; RV64IM-NEXT:    mul s5, s3, s2
+; RV64IM-NEXT:    add t3, t3, s4
+; RV64IM-NEXT:    and s4, t4, t2
+; RV64IM-NEXT:    and s6, t6, t3
+; RV64IM-NEXT:    and t4, t4, t3
+; RV64IM-NEXT:    and t6, t6, t2
+; RV64IM-NEXT:    mul s7, s6, s4
+; RV64IM-NEXT:    mul s8, t6, t4
+; RV64IM-NEXT:    mul s9, s0, s4
+; RV64IM-NEXT:    mul s10, s3, t5
+; RV64IM-NEXT:    mul s11, s6, t4
+; RV64IM-NEXT:    mul ra, t6, s2
+; RV64IM-NEXT:    mul a3, s0, s2
+; RV64IM-NEXT:    mul s0, s0, t4
+; RV64IM-NEXT:    mul t4, s3, t4
+; RV64IM-NEXT:    mul s3, s3, s4
+; RV64IM-NEXT:    mul s4, t6, s4
+; RV64IM-NEXT:    mul a0, s6, t5
+; RV64IM-NEXT:    mul s2, s6, s2
+; RV64IM-NEXT:    mul t5, t6, t5
+; RV64IM-NEXT:    xor t6, s5, s1
+; RV64IM-NEXT:    xor s1, s7, s8
+; RV64IM-NEXT:    xor s5, s10, s9
+; RV64IM-NEXT:    xor s6, s11, ra
+; RV64IM-NEXT:    xor t6, t6, s1
+; RV64IM-NEXT:    xor s1, s5, s6
+; RV64IM-NEXT:    and t6, t6, t1
+; RV64IM-NEXT:    and s1, s1, a7
+; RV64IM-NEXT:    xor a3, t4, a3
+; RV64IM-NEXT:    xor a0, a0, s4
+; RV64IM-NEXT:    xor t4, s3, s0
+; RV64IM-NEXT:    xor t5, s2, t5
+; RV64IM-NEXT:    xor a0, a3, a0
+; RV64IM-NEXT:    xor a3, t4, t5
 ; RV64IM-NEXT:    and a0, a0, t3
-; RV64IM-NEXT:    mul s1, a0, t5
-; RV64IM-NEXT:    mul s2, a5, a1
-; RV64IM-NEXT:    xor t6, t6, s0
-; RV64IM-NEXT:    mul s0, a0, a7
-; RV64IM-NEXT:    mul s3, a5, a7
-; RV64IM-NEXT:    mul a5, a5, t5
-; RV64IM-NEXT:    mul t5, t0, t5
-; RV64IM-NEXT:    mul a7, t4, a7
-; RV64IM-NEXT:    mul t4, t4, a4
+; RV64IM-NEXT:    and a3, a3, t2
+; RV64IM-NEXT:    or t4, s1, t6
+; RV64IM-NEXT:    or a0, a0, a3
+; RV64IM-NEXT:    or a0, t4, a0
+; RV64IM-NEXT:    srli a3, a0, 40
+; RV64IM-NEXT:    and a3, a3, a5
+; RV64IM-NEXT:    srli t4, a0, 56
+; RV64IM-NEXT:    or a3, a3, t4
+; RV64IM-NEXT:    srli t4, a0, 24
+; RV64IM-NEXT:    srli t5, a0, 8
+; RV64IM-NEXT:    lui t6, 4080
+; RV64IM-NEXT:    and t4, t4, t6
+; RV64IM-NEXT:    ld s0, 0(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    and t5, t5, s0
+; RV64IM-NEXT:    or t4, t5, t4
+; RV64IM-NEXT:    srliw t5, a0, 24
+; RV64IM-NEXT:    slli t5, t5, 32
+; RV64IM-NEXT:    and t6, a0, t6
+; RV64IM-NEXT:    slli t6, t6, 24
+; RV64IM-NEXT:    and a5, a0, a5
+; RV64IM-NEXT:    slli a0, a0, 56
+; RV64IM-NEXT:    slli a5, a5, 40
+; RV64IM-NEXT:    or t5, t6, t5
+; RV64IM-NEXT:    or a0, a0, a5
+; RV64IM-NEXT:    or a3, t4, a3
+; RV64IM-NEXT:    or a0, a0, t5
+; RV64IM-NEXT:    or a0, a0, a3
+; RV64IM-NEXT:    srli a3, a0, 4
+; RV64IM-NEXT:    and a0, a0, a6
+; RV64IM-NEXT:    and a3, a3, a6
+; RV64IM-NEXT:    slli a0, a0, 4
+; RV64IM-NEXT:    or a0, a3, a0
+; RV64IM-NEXT:    srli a3, a0, 2
+; RV64IM-NEXT:    and a0, a0, t0
+; RV64IM-NEXT:    and a3, a3, t0
+; RV64IM-NEXT:    slli a0, a0, 2
+; RV64IM-NEXT:    or a0, a3, a0
+; RV64IM-NEXT:    and a3, a1, a7
+; RV64IM-NEXT:    and a5, a2, t1
+; RV64IM-NEXT:    and a6, a1, t1
+; RV64IM-NEXT:    and t0, a2, a7
+; RV64IM-NEXT:    mul t4, a5, a3
+; RV64IM-NEXT:    mul t5, t0, a6
+; RV64IM-NEXT:    srli t6, a0, 1
+; RV64IM-NEXT:    and a0, a0, a4
+; RV64IM-NEXT:    and s0, a2, t3
+; RV64IM-NEXT:    and s1, a1, t3
+; RV64IM-NEXT:    mul s2, t0, a3
+; RV64IM-NEXT:    mul s3, s0, s1
+; RV64IM-NEXT:    and a4, t6, a4
+; RV64IM-NEXT:    slli a0, a0, 1
+; RV64IM-NEXT:    or a0, a4, a0
+; RV64IM-NEXT:    xor a4, t5, t4
+; RV64IM-NEXT:    and a1, a1, t2
+; RV64IM-NEXT:    mul t4, s0, a1
+; RV64IM-NEXT:    and a2, a2, t2
+; RV64IM-NEXT:    mul t5, a2, s1
+; RV64IM-NEXT:    mul t6, a5, a1
+; RV64IM-NEXT:    xor s2, s2, s3
+; RV64IM-NEXT:    mul s3, a2, a6
+; RV64IM-NEXT:    mul s4, a5, a6
+; RV64IM-NEXT:    mul s5, t0, s1
+; RV64IM-NEXT:    mul a5, a5, s1
+; RV64IM-NEXT:    mul a6, s0, a6
+; RV64IM-NEXT:    mul s0, s0, a3
 ; RV64IM-NEXT:    mul t0, t0, a1
-; RV64IM-NEXT:    mul a1, a0, a1
-; RV64IM-NEXT:    mul a0, a0, a4
-; RV64IM-NEXT:    xor a4, a6, t2
-; RV64IM-NEXT:    xor a6, t6, s2
-; RV64IM-NEXT:    xor a4, a4, s1
-; RV64IM-NEXT:    xor a6, a6, s0
-; RV64IM-NEXT:    and a3, a4, a3
-; RV64IM-NEXT:    and a2, a6, a2
-; RV64IM-NEXT:    xor a4, t5, s3
-; RV64IM-NEXT:    xor a5, a5, a7
-; RV64IM-NEXT:    xor a4, a4, t4
+; RV64IM-NEXT:    mul a1, a2, a1
+; RV64IM-NEXT:    mul a2, a2, a3
+; RV64IM-NEXT:    xor a3, a4, t4
+; RV64IM-NEXT:    xor a4, s2, t6
+; RV64IM-NEXT:    xor a3, a3, t5
+; RV64IM-NEXT:    xor a4, a4, s3
+; RV64IM-NEXT:    and a3, a3, t1
+; RV64IM-NEXT:    and a4, a4, a7
+; RV64IM-NEXT:    xor a7, s5, s4
+; RV64IM-NEXT:    xor a5, a5, a6
+; RV64IM-NEXT:    xor a6, a7, s0
 ; RV64IM-NEXT:    xor a5, t0, a5
-; RV64IM-NEXT:    xor a1, a4, a1
-; RV64IM-NEXT:    xor a0, a5, a0
-; RV64IM-NEXT:    and a1, a1, t1
-; RV64IM-NEXT:    and a0, a0, t3
-; RV64IM-NEXT:    or a2, a2, a3
-; RV64IM-NEXT:    or a0, a1, a0
-; RV64IM-NEXT:    or a0, a2, a0
-; RV64IM-NEXT:    ld s0, 24(sp) # 8-byte Folded Reload
-; RV64IM-NEXT:    ld s1, 16(sp) # 8-byte Folded Reload
-; RV64IM-NEXT:    ld s2, 8(sp) # 8-byte Folded Reload
-; RV64IM-NEXT:    ld s3, 0(sp) # 8-byte Folded Reload
-; RV64IM-NEXT:    addi sp, sp, 32
+; RV64IM-NEXT:    xor a1, a6, a1
+; RV64IM-NEXT:    xor a2, a5, a2
+; RV64IM-NEXT:    and a1, a1, t3
+; RV64IM-NEXT:    and a2, a2, t2
+; RV64IM-NEXT:    or a3, a4, a3
+; RV64IM-NEXT:    or a2, a1, a2
+; RV64IM-NEXT:    srli a1, a0, 1
+; RV64IM-NEXT:    or a0, a3, a2
+; RV64IM-NEXT:    ld ra, 104(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s0, 96(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s1, 88(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s2, 80(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s3, 72(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s4, 64(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s5, 56(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s6, 48(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s7, 40(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s8, 32(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s9, 24(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s10, 16(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    ld s11, 8(sp) # 8-byte Folded Reload
+; RV64IM-NEXT:    .cfi_restore ra
+; RV64IM-NEXT:    .cfi_restore s0
+; RV64IM-NEXT:    .cfi_restore s1
+; RV64IM-NEXT:    .cfi_restore s2
+; RV64IM-NEXT:    .cfi_restore s3
+; RV64IM-NEXT:    .cfi_restore s4
+; RV64IM-NEXT:    .cfi_restore s5
+; RV64IM-NEXT:    .cfi_restore s6
+; RV64IM-NEXT:    .cfi_restore s7
+; RV64IM-NEXT:    .cfi_restore s8
+; RV64IM-NEXT:    .cfi_restore s9
+; RV64IM-NEXT:    .cfi_restore s10
+; RV64IM-NEXT:    .cfi_restore s11
+; RV64IM-NEXT:    addi sp, sp, 112
+; RV64IM-NEXT:    .cfi_def_cfa_offset 0
 ; RV64IM-NEXT:    ret
 ;
-; RV32IMZBS-LABEL: clmul_i64:
+; RV32IMZBS-LABEL: clmul_i128_zext:
 ; RV32IMZBS:       # %bb.0:
-; RV32IMZBS-NEXT:    addi sp, sp, -80
-; RV32IMZBS-NEXT:    sw ra, 76(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s0, 72(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s1, 68(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s2, 64(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s3, 60(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s4, 56(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s5, 52(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s6, 48(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s7, 44(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s8, 40(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s9, 36(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s10, 32(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw s11, 28(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw a3, 24(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    sw a1, 16(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    lui a4, 16
-; RV32IMZBS-NEXT:    srli a5, a2, 8
-; RV32IMZBS-NEXT:    addi t0, a4, -256
-; RV32IMZBS-NEXT:    and a4, a5, t0
-; RV32IMZBS-NEXT:    srli a5, a2, 24
-; RV32IMZBS-NEXT:    and a6, a2, t0
-; RV32IMZBS-NEXT:    slli a6, a6, 8
-; RV32IMZBS-NEXT:    slli a7, a2, 24
-; RV32IMZBS-NEXT:    or a4, a4, a5
-; RV32IMZBS-NEXT:    or a5, a7, a6
-; RV32IMZBS-NEXT:    or a4, a5, a4
-; RV32IMZBS-NEXT:    lui a5, 61681
-; RV32IMZBS-NEXT:    srli a6, a4, 4
-; RV32IMZBS-NEXT:    addi t1, a5, -241
-; RV32IMZBS-NEXT:    and a5, a6, t1
-; RV32IMZBS-NEXT:    and a4, a4, t1
-; RV32IMZBS-NEXT:    slli a4, a4, 4
-; RV32IMZBS-NEXT:    lui a6, 209715
-; RV32IMZBS-NEXT:    or a4, a5, a4
-; RV32IMZBS-NEXT:    addi t2, a6, 819
-; RV32IMZBS-NEXT:    srli a5, a4, 2
-; RV32IMZBS-NEXT:    and a4, a4, t2
-; RV32IMZBS-NEXT:    and a5, a5, t2
-; RV32IMZBS-NEXT:    slli a4, a4, 2
-; RV32IMZBS-NEXT:    or a4, a5, a4
-; RV32IMZBS-NEXT:    lui a1, 349525
-; RV32IMZBS-NEXT:    srli a5, a4, 1
-; RV32IMZBS-NEXT:    addi t4, a1, 1365
-; RV32IMZBS-NEXT:    and a5, a5, t4
-; RV32IMZBS-NEXT:    sw a0, 20(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    srli a6, a0, 8
-; RV32IMZBS-NEXT:    and a4, a4, t4
-; RV32IMZBS-NEXT:    and a6, a6, t0
-; RV32IMZBS-NEXT:    srli a7, a0, 24
-; RV32IMZBS-NEXT:    and t5, a0, t0
-; RV32IMZBS-NEXT:    slli t5, t5, 8
-; RV32IMZBS-NEXT:    slli t6, a0, 24
+; RV32IMZBS-NEXT:    addi sp, sp, -176
+; RV32IMZBS-NEXT:    .cfi_def_cfa_offset 176
+; RV32IMZBS-NEXT:    sw ra, 172(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s0, 168(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s1, 164(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s2, 160(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s3, 156(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s4, 152(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s5, 148(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s6, 144(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s7, 140(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s8, 136(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s9, 132(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s10, 128(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s11, 124(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    .cfi_offset ra, -4
+; RV32IMZBS-NEXT:    .cfi_offset s0, -8
+; RV32IMZBS-NEXT:    .cfi_offset s1, -12
+; RV32IMZBS-NEXT:    .cfi_offset s2, -16
+; RV32IMZBS-NEXT:    .cfi_offset s3, -20
+; RV32IMZBS-NEXT:    .cfi_offset s4, -24
+; RV32IMZBS-NEXT:    .cfi_offset s5, -28
+; RV32IMZBS-NEXT:    .cfi_offset s6, -32
+; RV32IMZBS-NEXT:    .cfi_offset s7, -36
+; RV32IMZBS-NEXT:    .cfi_offset s8, -40
+; RV32IMZBS-NEXT:    .cfi_offset s9, -44
+; RV32IMZBS-NEXT:    .cfi_offset s10, -48
+; RV32IMZBS-NEXT:    .cfi_offset s11, -52
+; RV32IMZBS-NEXT:    sw a3, 112(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv a3, a2
+; RV32IMZBS-NEXT:    mv t2, a1
+; RV32IMZBS-NEXT:    sw a0, 96(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lui a0, 16
+; RV32IMZBS-NEXT:    sw a4, 92(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a1, a4, 8
+; RV32IMZBS-NEXT:    addi t3, a0, -256
+; RV32IMZBS-NEXT:    and a0, a1, t3
+; RV32IMZBS-NEXT:    srli a1, a4, 24
+; RV32IMZBS-NEXT:    or a0, a0, a1
+; RV32IMZBS-NEXT:    and a1, a4, t3
+; RV32IMZBS-NEXT:    slli a1, a1, 8
+; RV32IMZBS-NEXT:    slli a2, a4, 24
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    lui a2, 61681
+; RV32IMZBS-NEXT:    or a0, a1, a0
+; RV32IMZBS-NEXT:    addi s3, a2, -241
+; RV32IMZBS-NEXT:    srli a1, a0, 4
+; RV32IMZBS-NEXT:    and a0, a0, s3
+; RV32IMZBS-NEXT:    and a1, a1, s3
+; RV32IMZBS-NEXT:    slli a0, a0, 4
+; RV32IMZBS-NEXT:    or a0, a1, a0
+; RV32IMZBS-NEXT:    lui a1, 209715
+; RV32IMZBS-NEXT:    srli a2, a0, 2
+; RV32IMZBS-NEXT:    addi s2, a1, 819
+; RV32IMZBS-NEXT:    and a1, a2, s2
+; RV32IMZBS-NEXT:    and a0, a0, s2
+; RV32IMZBS-NEXT:    slli a2, a0, 2
+; RV32IMZBS-NEXT:    lui a0, 349525
+; RV32IMZBS-NEXT:    or a1, a1, a2
+; RV32IMZBS-NEXT:    addi t5, a0, 1365
+; RV32IMZBS-NEXT:    srli a2, a1, 1
+; RV32IMZBS-NEXT:    and a1, a1, t5
+; RV32IMZBS-NEXT:    and a2, a2, t5
+; RV32IMZBS-NEXT:    slli a1, a1, 1
+; RV32IMZBS-NEXT:    or a7, a2, a1
+; RV32IMZBS-NEXT:    sw a7, 72(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a1, a7, 8
+; RV32IMZBS-NEXT:    and a1, a1, t3
+; RV32IMZBS-NEXT:    srli a2, a7, 24
+; RV32IMZBS-NEXT:    and a6, a7, t3
+; RV32IMZBS-NEXT:    slli a7, a7, 24
+; RV32IMZBS-NEXT:    slli a6, a6, 8
+; RV32IMZBS-NEXT:    or a1, a1, a2
+; RV32IMZBS-NEXT:    or a2, a7, a6
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    srli a2, a1, 4
+; RV32IMZBS-NEXT:    and a1, a1, s3
+; RV32IMZBS-NEXT:    and a2, a2, s3
+; RV32IMZBS-NEXT:    slli a1, a1, 4
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    srli a2, a1, 2
+; RV32IMZBS-NEXT:    sw a3, 88(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a6, a3, 8
+; RV32IMZBS-NEXT:    and a2, a2, s2
+; RV32IMZBS-NEXT:    and a6, a6, t3
+; RV32IMZBS-NEXT:    srli a7, a3, 24
+; RV32IMZBS-NEXT:    and t0, a3, t3
+; RV32IMZBS-NEXT:    slli t0, t0, 8
+; RV32IMZBS-NEXT:    slli t1, a3, 24
 ; RV32IMZBS-NEXT:    or a6, a6, a7
-; RV32IMZBS-NEXT:    or a7, t6, t5
-; RV32IMZBS-NEXT:    slli a4, a4, 1
+; RV32IMZBS-NEXT:    or a7, t1, t0
+; RV32IMZBS-NEXT:    and a1, a1, s2
 ; RV32IMZBS-NEXT:    or a6, a7, a6
 ; RV32IMZBS-NEXT:    srli a7, a6, 4
-; RV32IMZBS-NEXT:    and a6, a6, t1
-; RV32IMZBS-NEXT:    and a7, a7, t1
+; RV32IMZBS-NEXT:    and a6, a6, s3
+; RV32IMZBS-NEXT:    and a7, a7, s3
 ; RV32IMZBS-NEXT:    slli a6, a6, 4
-; RV32IMZBS-NEXT:    or t5, a5, a4
-; RV32IMZBS-NEXT:    or a4, a7, a6
-; RV32IMZBS-NEXT:    srli a5, a4, 2
-; RV32IMZBS-NEXT:    and a4, a4, t2
-; RV32IMZBS-NEXT:    and a5, a5, t2
-; RV32IMZBS-NEXT:    slli a4, a4, 2
-; RV32IMZBS-NEXT:    lui a6, 69905
-; RV32IMZBS-NEXT:    or a5, a5, a4
-; RV32IMZBS-NEXT:    addi a4, a6, 273
-; RV32IMZBS-NEXT:    srli a6, a5, 1
-; RV32IMZBS-NEXT:    and a6, a6, t4
-; RV32IMZBS-NEXT:    and a5, a5, t4
-; RV32IMZBS-NEXT:    slli a5, a5, 1
-; RV32IMZBS-NEXT:    lui a7, 139810
-; RV32IMZBS-NEXT:    or t6, a6, a5
-; RV32IMZBS-NEXT:    addi a5, a7, 546
-; RV32IMZBS-NEXT:    and s0, t5, a4
-; RV32IMZBS-NEXT:    and s1, t6, a5
-; RV32IMZBS-NEXT:    and s2, t5, a5
-; RV32IMZBS-NEXT:    and s3, t6, a4
-; RV32IMZBS-NEXT:    mul s4, s1, s0
-; RV32IMZBS-NEXT:    mul s5, s3, s2
-; RV32IMZBS-NEXT:    lui a6, 559241
-; RV32IMZBS-NEXT:    lui a7, 279620
-; RV32IMZBS-NEXT:    addi s11, a6, -1912
-; RV32IMZBS-NEXT:    addi t3, a7, 1092
-; RV32IMZBS-NEXT:    and s6, t5, s11
-; RV32IMZBS-NEXT:    and s7, t6, t3
-; RV32IMZBS-NEXT:    and t5, t5, t3
-; RV32IMZBS-NEXT:    and t6, t6, s11
-; RV32IMZBS-NEXT:    mul s8, s7, s6
-; RV32IMZBS-NEXT:    mul s9, t6, t5
-; RV32IMZBS-NEXT:    mul s10, s1, s6
-; RV32IMZBS-NEXT:    mul a6, s3, s0
-; RV32IMZBS-NEXT:    mul ra, s7, t5
-; RV32IMZBS-NEXT:    mul a3, t6, s2
-; RV32IMZBS-NEXT:    mul a1, s1, s2
-; RV32IMZBS-NEXT:    mul s1, s1, t5
-; RV32IMZBS-NEXT:    mul t5, s3, t5
-; RV32IMZBS-NEXT:    mul s3, s3, s6
-; RV32IMZBS-NEXT:    mul s6, t6, s6
-; RV32IMZBS-NEXT:    mul a0, s7, s0
-; RV32IMZBS-NEXT:    mul s2, s7, s2
-; RV32IMZBS-NEXT:    mul t6, t6, s0
-; RV32IMZBS-NEXT:    xor s0, s5, s4
-; RV32IMZBS-NEXT:    xor s4, s8, s9
-; RV32IMZBS-NEXT:    xor s5, a6, s10
-; RV32IMZBS-NEXT:    xor a3, ra, a3
-; RV32IMZBS-NEXT:    xor s0, s0, s4
-; RV32IMZBS-NEXT:    xor a3, s5, a3
-; RV32IMZBS-NEXT:    and s0, s0, a5
-; RV32IMZBS-NEXT:    and a3, a3, a4
-; RV32IMZBS-NEXT:    xor a1, t5, a1
-; RV32IMZBS-NEXT:    xor a0, a0, s6
-; RV32IMZBS-NEXT:    xor t5, s3, s1
-; RV32IMZBS-NEXT:    xor t6, s2, t6
+; RV32IMZBS-NEXT:    slli a1, a1, 2
+; RV32IMZBS-NEXT:    or a6, a7, a6
+; RV32IMZBS-NEXT:    srli a7, a6, 2
+; RV32IMZBS-NEXT:    and a6, a6, s2
+; RV32IMZBS-NEXT:    and a7, a7, s2
+; RV32IMZBS-NEXT:    slli a6, a6, 2
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    or a2, a7, a6
+; RV32IMZBS-NEXT:    srli a6, a2, 1
+; RV32IMZBS-NEXT:    and a2, a2, t5
+; RV32IMZBS-NEXT:    and a6, a6, t5
+; RV32IMZBS-NEXT:    slli a2, a2, 1
+; RV32IMZBS-NEXT:    srli a7, a1, 1
+; RV32IMZBS-NEXT:    or t1, a6, a2
+; RV32IMZBS-NEXT:    sw t1, 80(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a2, a7, t5
+; RV32IMZBS-NEXT:    srli a6, t1, 8
+; RV32IMZBS-NEXT:    and a1, a1, t5
+; RV32IMZBS-NEXT:    and a6, a6, t3
+; RV32IMZBS-NEXT:    srli a7, t1, 24
+; RV32IMZBS-NEXT:    and t0, t1, t3
+; RV32IMZBS-NEXT:    slli t1, t1, 24
+; RV32IMZBS-NEXT:    slli t0, t0, 8
+; RV32IMZBS-NEXT:    or a6, a6, a7
+; RV32IMZBS-NEXT:    or a7, t1, t0
+; RV32IMZBS-NEXT:    slli a1, a1, 1
+; RV32IMZBS-NEXT:    or a6, a7, a6
+; RV32IMZBS-NEXT:    srli a7, a6, 4
+; RV32IMZBS-NEXT:    and a6, a6, s3
+; RV32IMZBS-NEXT:    and a7, a7, s3
+; RV32IMZBS-NEXT:    slli a6, a6, 4
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    or a2, a7, a6
+; RV32IMZBS-NEXT:    srli a6, a2, 2
+; RV32IMZBS-NEXT:    and a2, a2, s2
+; RV32IMZBS-NEXT:    and a6, a6, s2
+; RV32IMZBS-NEXT:    slli a2, a2, 2
+; RV32IMZBS-NEXT:    lui a7, 69905
+; RV32IMZBS-NEXT:    or a2, a6, a2
+; RV32IMZBS-NEXT:    addi a0, a7, 273
+; RV32IMZBS-NEXT:    srli a7, a2, 1
+; RV32IMZBS-NEXT:    and a7, a7, t5
+; RV32IMZBS-NEXT:    and a2, a2, t5
+; RV32IMZBS-NEXT:    slli a2, a2, 1
+; RV32IMZBS-NEXT:    lui t0, 139810
+; RV32IMZBS-NEXT:    or a2, a7, a2
+; RV32IMZBS-NEXT:    addi a6, t0, 546
+; RV32IMZBS-NEXT:    and t6, a1, a0
+; RV32IMZBS-NEXT:    and s0, a2, a6
+; RV32IMZBS-NEXT:    and s1, a1, a6
+; RV32IMZBS-NEXT:    and s4, a2, a0
+; RV32IMZBS-NEXT:    mv a7, a0
+; RV32IMZBS-NEXT:    mul s5, s0, t6
+; RV32IMZBS-NEXT:    mul s6, s4, s1
+; RV32IMZBS-NEXT:    lui t0, 559241
+; RV32IMZBS-NEXT:    lui t1, 279620
+; RV32IMZBS-NEXT:    addi t4, t0, -1912
+; RV32IMZBS-NEXT:    addi t1, t1, 1092
+; RV32IMZBS-NEXT:    and s7, a1, t4
+; RV32IMZBS-NEXT:    and s8, a2, t1
+; RV32IMZBS-NEXT:    and a1, a1, t1
+; RV32IMZBS-NEXT:    and a2, a2, t4
+; RV32IMZBS-NEXT:    mul s9, s8, s7
+; RV32IMZBS-NEXT:    mul s10, a2, a1
+; RV32IMZBS-NEXT:    mul s11, s0, s7
+; RV32IMZBS-NEXT:    mul ra, s4, t6
+; RV32IMZBS-NEXT:    mul a5, s8, a1
+; RV32IMZBS-NEXT:    mul a4, a2, s1
+; RV32IMZBS-NEXT:    mul a3, s0, s1
+; RV32IMZBS-NEXT:    mul s0, s0, a1
+; RV32IMZBS-NEXT:    mul a1, s4, a1
+; RV32IMZBS-NEXT:    mul s4, s4, s7
+; RV32IMZBS-NEXT:    mul s7, a2, s7
+; RV32IMZBS-NEXT:    mul a0, s8, t6
+; RV32IMZBS-NEXT:    mul s1, s8, s1
+; RV32IMZBS-NEXT:    mul a2, a2, t6
+; RV32IMZBS-NEXT:    xor t6, s6, s5
+; RV32IMZBS-NEXT:    xor s5, s9, s10
+; RV32IMZBS-NEXT:    xor s6, ra, s11
+; RV32IMZBS-NEXT:    xor a4, a5, a4
+; RV32IMZBS-NEXT:    xor a5, t6, s5
+; RV32IMZBS-NEXT:    xor a4, s6, a4
+; RV32IMZBS-NEXT:    and a5, a5, a6
+; RV32IMZBS-NEXT:    and a4, a4, a7
+; RV32IMZBS-NEXT:    mv s5, a7
+; RV32IMZBS-NEXT:    xor a1, a1, a3
+; RV32IMZBS-NEXT:    xor a0, a0, s7
+; RV32IMZBS-NEXT:    xor a3, s4, s0
+; RV32IMZBS-NEXT:    xor a2, s1, a2
 ; RV32IMZBS-NEXT:    xor a0, a1, a0
-; RV32IMZBS-NEXT:    xor a1, t5, t6
-; RV32IMZBS-NEXT:    and a0, a0, t3
-; RV32IMZBS-NEXT:    and a1, a1, s11
-; RV32IMZBS-NEXT:    or a3, a3, s0
+; RV32IMZBS-NEXT:    xor a2, a3, a2
+; RV32IMZBS-NEXT:    and a0, a0, t1
+; RV32IMZBS-NEXT:    and a1, a2, t4
+; RV32IMZBS-NEXT:    or a4, a4, a5
 ; RV32IMZBS-NEXT:    or a0, a0, a1
-; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    or a0, a4, a0
 ; RV32IMZBS-NEXT:    srli a1, a0, 8
-; RV32IMZBS-NEXT:    and a1, a1, t0
-; RV32IMZBS-NEXT:    srli a3, a0, 24
-; RV32IMZBS-NEXT:    and t0, a0, t0
+; RV32IMZBS-NEXT:    and a1, a1, t3
+; RV32IMZBS-NEXT:    srli a2, a0, 24
+; RV32IMZBS-NEXT:    and a3, a0, t3
 ; RV32IMZBS-NEXT:    slli a0, a0, 24
-; RV32IMZBS-NEXT:    slli t0, t0, 8
-; RV32IMZBS-NEXT:    or a1, a1, a3
-; RV32IMZBS-NEXT:    or a0, a0, t0
+; RV32IMZBS-NEXT:    slli a3, a3, 8
+; RV32IMZBS-NEXT:    or a1, a1, a2
+; RV32IMZBS-NEXT:    or a0, a0, a3
 ; RV32IMZBS-NEXT:    or a0, a0, a1
 ; RV32IMZBS-NEXT:    srli a1, a0, 4
-; RV32IMZBS-NEXT:    and a0, a0, t1
-; RV32IMZBS-NEXT:    and a1, a1, t1
+; RV32IMZBS-NEXT:    and a0, a0, s3
+; RV32IMZBS-NEXT:    and a1, a1, s3
 ; RV32IMZBS-NEXT:    slli a0, a0, 4
 ; RV32IMZBS-NEXT:    or a0, a1, a0
 ; RV32IMZBS-NEXT:    srli a1, a0, 2
-; RV32IMZBS-NEXT:    and a0, a0, t2
-; RV32IMZBS-NEXT:    and a1, a1, t2
+; RV32IMZBS-NEXT:    and a1, a1, s2
+; RV32IMZBS-NEXT:    and a0, a0, s2
+; RV32IMZBS-NEXT:    sw t2, 84(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    srli a2, t2, 8
 ; RV32IMZBS-NEXT:    slli a0, a0, 2
-; RV32IMZBS-NEXT:    or a0, a1, a0
-; RV32IMZBS-NEXT:    srli a1, a0, 1
-; RV32IMZBS-NEXT:    lui a3, 349525
-; RV32IMZBS-NEXT:    addi a3, a3, 1364
-; RV32IMZBS-NEXT:    and a0, a0, t4
-; RV32IMZBS-NEXT:    and a1, a1, a3
-; RV32IMZBS-NEXT:    slli a0, a0, 1
-; RV32IMZBS-NEXT:    or a0, a1, a0
-; RV32IMZBS-NEXT:    sw a0, 12(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    mv a7, a4
-; RV32IMZBS-NEXT:    and t0, a2, a4
-; RV32IMZBS-NEXT:    lw a6, 16(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    and a0, a6, a5
-; RV32IMZBS-NEXT:    and t1, a2, a5
-; RV32IMZBS-NEXT:    and a1, a6, a4
-; RV32IMZBS-NEXT:    mul a3, a0, t0
-; RV32IMZBS-NEXT:    mul t4, a1, t1
-; RV32IMZBS-NEXT:    and t2, a2, s11
-; RV32IMZBS-NEXT:    and t5, a6, t3
 ; RV32IMZBS-NEXT:    and a2, a2, t3
-; RV32IMZBS-NEXT:    and t6, a6, s11
-; RV32IMZBS-NEXT:    mul s0, t5, t2
-; RV32IMZBS-NEXT:    mul s2, t6, a2
-; RV32IMZBS-NEXT:    mul s3, a0, t2
-; RV32IMZBS-NEXT:    mul s4, a1, t0
-; RV32IMZBS-NEXT:    mul s5, t5, a2
-; RV32IMZBS-NEXT:    mul s6, t6, t1
-; RV32IMZBS-NEXT:    mul s7, a0, t1
-; RV32IMZBS-NEXT:    mul s8, a1, a2
-; RV32IMZBS-NEXT:    mul s9, t5, t0
-; RV32IMZBS-NEXT:    mul s10, t6, t2
-; RV32IMZBS-NEXT:    mul a0, a0, a2
-; RV32IMZBS-NEXT:    mul a1, a1, t2
-; RV32IMZBS-NEXT:    mul t5, t5, t1
-; RV32IMZBS-NEXT:    mul t6, t6, t0
-; RV32IMZBS-NEXT:    xor a3, t4, a3
-; RV32IMZBS-NEXT:    xor t4, s0, s2
-; RV32IMZBS-NEXT:    xor s0, s4, s3
-; RV32IMZBS-NEXT:    xor s2, s5, s6
-; RV32IMZBS-NEXT:    xor a3, a3, t4
-; RV32IMZBS-NEXT:    xor t4, s0, s2
-; RV32IMZBS-NEXT:    and a3, a3, a5
-; RV32IMZBS-NEXT:    and t4, t4, a4
-; RV32IMZBS-NEXT:    xor s0, s8, s7
-; RV32IMZBS-NEXT:    xor s2, s9, s10
+; RV32IMZBS-NEXT:    srli a3, t2, 24
+; RV32IMZBS-NEXT:    and a4, t2, t3
+; RV32IMZBS-NEXT:    slli a4, a4, 8
+; RV32IMZBS-NEXT:    slli a5, t2, 24
+; RV32IMZBS-NEXT:    or a2, a2, a3
+; RV32IMZBS-NEXT:    or a4, a5, a4
+; RV32IMZBS-NEXT:    or a0, a1, a0
+; RV32IMZBS-NEXT:    or a2, a4, a2
+; RV32IMZBS-NEXT:    srli a1, a2, 4
+; RV32IMZBS-NEXT:    and a2, a2, s3
+; RV32IMZBS-NEXT:    and a1, a1, s3
+; RV32IMZBS-NEXT:    slli a2, a2, 4
+; RV32IMZBS-NEXT:    srli t0, a0, 1
+; RV32IMZBS-NEXT:    or a1, a1, a2
+; RV32IMZBS-NEXT:    srli a2, a1, 2
+; RV32IMZBS-NEXT:    and a1, a1, s2
+; RV32IMZBS-NEXT:    and a2, a2, s2
+; RV32IMZBS-NEXT:    slli a1, a1, 2
+; RV32IMZBS-NEXT:    lui t2, 349525
+; RV32IMZBS-NEXT:    addi a7, t2, 1364
+; RV32IMZBS-NEXT:    sw a7, 76(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    srli a2, a1, 1
+; RV32IMZBS-NEXT:    and a1, a1, t5
+; RV32IMZBS-NEXT:    and a2, a2, t5
+; RV32IMZBS-NEXT:    slli a1, a1, 1
+; RV32IMZBS-NEXT:    and t6, a0, t5
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    lw a3, 72(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mv a2, s5
+; RV32IMZBS-NEXT:    sw s5, 100(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and s4, a3, s5
+; RV32IMZBS-NEXT:    sw a6, 120(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a0, a1, a6
+; RV32IMZBS-NEXT:    and s5, a3, a6
+; RV32IMZBS-NEXT:    and a2, a1, a2
+; RV32IMZBS-NEXT:    mul a6, a0, s4
+; RV32IMZBS-NEXT:    mul a4, a2, s5
+; RV32IMZBS-NEXT:    and a5, a3, t4
+; RV32IMZBS-NEXT:    sw t1, 116(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and s0, a1, t1
+; RV32IMZBS-NEXT:    and s9, a3, t1
+; RV32IMZBS-NEXT:    and s1, a1, t4
+; RV32IMZBS-NEXT:    mv a1, a5
+; RV32IMZBS-NEXT:    mul a3, s0, a5
+; RV32IMZBS-NEXT:    mul a5, s1, s9
+; RV32IMZBS-NEXT:    mul s6, a0, a1
+; RV32IMZBS-NEXT:    mv s7, a1
+; RV32IMZBS-NEXT:    mv t1, a2
+; RV32IMZBS-NEXT:    mul s8, a2, s4
+; RV32IMZBS-NEXT:    mul s10, s0, s9
+; RV32IMZBS-NEXT:    mul s11, s1, s5
+; RV32IMZBS-NEXT:    mul ra, a0, s5
+; RV32IMZBS-NEXT:    sw s5, 44(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv t2, a0
+; RV32IMZBS-NEXT:    sw a0, 68(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a2, a2, s9
+; RV32IMZBS-NEXT:    sw t1, 64(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a1, s0, s4
+; RV32IMZBS-NEXT:    sw s4, 40(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s0, 60(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a0, s1, s7
+; RV32IMZBS-NEXT:    sw s7, 8(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    sw s1, 72(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and t0, t0, a7
+; RV32IMZBS-NEXT:    slli a7, t6, 1
+; RV32IMZBS-NEXT:    or a7, t0, a7
+; RV32IMZBS-NEXT:    sw a7, 52(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a4, a4, a6
+; RV32IMZBS-NEXT:    xor a3, a3, a5
+; RV32IMZBS-NEXT:    xor a5, s8, s6
+; RV32IMZBS-NEXT:    xor a6, s10, s11
+; RV32IMZBS-NEXT:    xor a3, a4, a3
+; RV32IMZBS-NEXT:    xor a4, a5, a6
+; RV32IMZBS-NEXT:    xor a2, a2, ra
 ; RV32IMZBS-NEXT:    xor a0, a1, a0
-; RV32IMZBS-NEXT:    xor a1, t5, t6
-; RV32IMZBS-NEXT:    xor t5, s0, s2
-; RV32IMZBS-NEXT:    xor a0, a0, a1
-; RV32IMZBS-NEXT:    and a1, t5, t3
-; RV32IMZBS-NEXT:    and s0, a0, s11
-; RV32IMZBS-NEXT:    or a0, t4, a3
-; RV32IMZBS-NEXT:    sw a0, 16(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    or a1, a1, s0
-; RV32IMZBS-NEXT:    sw a1, 8(sp) # 4-byte Folded Spill
-; RV32IMZBS-NEXT:    lw a0, 24(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    and a3, a0, a4
-; RV32IMZBS-NEXT:    lw a4, 20(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    and a1, a4, a5
-; RV32IMZBS-NEXT:    and s2, a0, a5
-; RV32IMZBS-NEXT:    mv s0, a5
-; RV32IMZBS-NEXT:    and t4, a4, a7
-; RV32IMZBS-NEXT:    mul s3, a1, a3
-; RV32IMZBS-NEXT:    mul s4, t4, s2
-; RV32IMZBS-NEXT:    mv s1, s11
-; RV32IMZBS-NEXT:    and s5, a0, s11
-; RV32IMZBS-NEXT:    and t5, a4, t3
-; RV32IMZBS-NEXT:    and s6, a0, t3
-; RV32IMZBS-NEXT:    and a0, a4, s11
-; RV32IMZBS-NEXT:    mul t6, t5, s5
-; RV32IMZBS-NEXT:    mul s7, a0, s6
-; RV32IMZBS-NEXT:    mul s8, a1, s5
-; RV32IMZBS-NEXT:    mul s9, t4, a3
-; RV32IMZBS-NEXT:    mul s10, t5, s6
-; RV32IMZBS-NEXT:    mul s11, a0, s2
-; RV32IMZBS-NEXT:    mul ra, t4, s6
-; RV32IMZBS-NEXT:    mul s6, a1, s6
-; RV32IMZBS-NEXT:    mul a6, a0, s5
-; RV32IMZBS-NEXT:    mul s5, t4, s5
-; RV32IMZBS-NEXT:    mul a4, a1, s2
-; RV32IMZBS-NEXT:    mul a5, t5, a3
-; RV32IMZBS-NEXT:    mul s2, t5, s2
-; RV32IMZBS-NEXT:    mul a3, a0, a3
-; RV32IMZBS-NEXT:    xor s3, s4, s3
-; RV32IMZBS-NEXT:    xor t6, t6, s7
-; RV32IMZBS-NEXT:    xor s4, s9, s8
-; RV32IMZBS-NEXT:    xor s7, s10, s11
-; RV32IMZBS-NEXT:    xor t6, s3, t6
-; RV32IMZBS-NEXT:    xor s3, s4, s7
-; RV32IMZBS-NEXT:    mv s8, s0
-; RV32IMZBS-NEXT:    and t6, t6, s0
-; RV32IMZBS-NEXT:    and s3, s3, a7
-; RV32IMZBS-NEXT:    xor a4, ra, a4
+; RV32IMZBS-NEXT:    lw s8, 112(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    srli a1, s8, 8
+; RV32IMZBS-NEXT:    sw t3, 36(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a5, s8, t3
+; RV32IMZBS-NEXT:    and a1, a1, t3
+; RV32IMZBS-NEXT:    slli a5, a5, 8
+; RV32IMZBS-NEXT:    mul a6, t2, s9
+; RV32IMZBS-NEXT:    mul t0, t1, s7
+; RV32IMZBS-NEXT:    srli s6, s8, 24
+; RV32IMZBS-NEXT:    slli s8, s8, 24
+; RV32IMZBS-NEXT:    or a1, a1, s6
+; RV32IMZBS-NEXT:    or a5, s8, a5
+; RV32IMZBS-NEXT:    xor a0, a2, a0
+; RV32IMZBS-NEXT:    or a1, a5, a1
+; RV32IMZBS-NEXT:    mul a2, s0, s5
+; RV32IMZBS-NEXT:    mul a5, s1, s4
+; RV32IMZBS-NEXT:    srli s6, a1, 4
+; RV32IMZBS-NEXT:    sw s3, 104(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a1, a1, s3
+; RV32IMZBS-NEXT:    and s6, s6, s3
+; RV32IMZBS-NEXT:    slli a1, a1, 4
+; RV32IMZBS-NEXT:    xor a6, t0, a6
+; RV32IMZBS-NEXT:    or a1, s6, a1
+; RV32IMZBS-NEXT:    srli t0, a1, 2
+; RV32IMZBS-NEXT:    sw s2, 56(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a1, a1, s2
+; RV32IMZBS-NEXT:    and t0, t0, s2
+; RV32IMZBS-NEXT:    slli a1, a1, 2
+; RV32IMZBS-NEXT:    xor a2, a2, a5
+; RV32IMZBS-NEXT:    or a1, t0, a1
+; RV32IMZBS-NEXT:    srli a5, a1, 1
+; RV32IMZBS-NEXT:    sw t5, 108(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a1, a1, t5
+; RV32IMZBS-NEXT:    and a5, a5, t5
+; RV32IMZBS-NEXT:    slli a1, a1, 1
+; RV32IMZBS-NEXT:    xor a2, a6, a2
+; RV32IMZBS-NEXT:    or a1, a5, a1
+; RV32IMZBS-NEXT:    lw s4, 120(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t2, a3, s4
+; RV32IMZBS-NEXT:    lw s5, 100(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a4, s5
+; RV32IMZBS-NEXT:    sw a3, 32(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw t1, 116(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a0, a0, t1
+; RV32IMZBS-NEXT:    sw a0, 28(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv s1, t4
+; RV32IMZBS-NEXT:    and a0, a2, t4
+; RV32IMZBS-NEXT:    sw a0, 24(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    and a4, a1, s5
+; RV32IMZBS-NEXT:    and s8, a1, s4
+; RV32IMZBS-NEXT:    and s10, a1, t4
+; RV32IMZBS-NEXT:    and s11, a1, t1
+; RV32IMZBS-NEXT:    lw a3, 80(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a0, a3, s4
+; RV32IMZBS-NEXT:    and a1, a3, s5
+; RV32IMZBS-NEXT:    and a2, a3, t1
+; RV32IMZBS-NEXT:    and s3, a3, t4
+; RV32IMZBS-NEXT:    mul a3, a0, a4
+; RV32IMZBS-NEXT:    sw a3, 80(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a3, a1, s8
+; RV32IMZBS-NEXT:    sw a3, 20(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a3, a2, s10
+; RV32IMZBS-NEXT:    sw a3, 16(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul a3, s3, s11
+; RV32IMZBS-NEXT:    sw a3, 12(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul s6, a0, s10
+; RV32IMZBS-NEXT:    mul s0, a1, a4
+; RV32IMZBS-NEXT:    mv a3, a4
+; RV32IMZBS-NEXT:    sw a4, 48(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul t6, a2, s11
+; RV32IMZBS-NEXT:    mul t5, s3, s8
+; RV32IMZBS-NEXT:    mul t4, a0, s8
+; RV32IMZBS-NEXT:    mul t3, a1, s11
+; RV32IMZBS-NEXT:    mul t0, a2, a4
+; RV32IMZBS-NEXT:    mul a7, s3, s10
+; RV32IMZBS-NEXT:    mul a6, a0, s11
+; RV32IMZBS-NEXT:    mul a5, a1, s10
+; RV32IMZBS-NEXT:    mul a4, a2, s8
+; RV32IMZBS-NEXT:    mul a3, s3, a3
+; RV32IMZBS-NEXT:    lw t1, 32(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    or t2, t1, t2
+; RV32IMZBS-NEXT:    lw t1, 28(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw ra, 24(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    or t1, t1, ra
+; RV32IMZBS-NEXT:    lw ra, 80(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s2, 20(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    xor ra, s2, ra
+; RV32IMZBS-NEXT:    lw s2, 16(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s7, 12(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    xor s2, s2, s7
+; RV32IMZBS-NEXT:    xor s0, s0, s6
+; RV32IMZBS-NEXT:    xor t5, t6, t5
+; RV32IMZBS-NEXT:    xor t6, ra, s2
+; RV32IMZBS-NEXT:    xor t5, s0, t5
+; RV32IMZBS-NEXT:    xor t3, t3, t4
+; RV32IMZBS-NEXT:    xor a7, t0, a7
 ; RV32IMZBS-NEXT:    xor a5, a5, a6
-; RV32IMZBS-NEXT:    xor a6, s5, s6
-; RV32IMZBS-NEXT:    xor a3, s2, a3
-; RV32IMZBS-NEXT:    xor a4, a4, a5
-; RV32IMZBS-NEXT:    xor a3, a6, a3
-; RV32IMZBS-NEXT:    and a4, a4, t3
-; RV32IMZBS-NEXT:    mv s9, s1
+; RV32IMZBS-NEXT:    xor a3, a4, a3
+; RV32IMZBS-NEXT:    xor a4, t3, a7
+; RV32IMZBS-NEXT:    xor a3, a5, a3
+; RV32IMZBS-NEXT:    and a5, t6, s4
+; RV32IMZBS-NEXT:    and a6, t5, s5
+; RV32IMZBS-NEXT:    lw s4, 116(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s4
 ; RV32IMZBS-NEXT:    and a3, a3, s1
-; RV32IMZBS-NEXT:    or a5, s3, t6
+; RV32IMZBS-NEXT:    mv s6, s1
+; RV32IMZBS-NEXT:    or a5, a6, a5
 ; RV32IMZBS-NEXT:    or a3, a4, a3
-; RV32IMZBS-NEXT:    lw a4, 16(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s0, 8(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    or s0, a4, s0
+; RV32IMZBS-NEXT:    or a4, t2, t1
 ; RV32IMZBS-NEXT:    or a3, a5, a3
-; RV32IMZBS-NEXT:    lw a4, 12(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    xor a3, a3, a4
+; RV32IMZBS-NEXT:    lw a4, 52(sp) # 4-byte Folded Reload
 ; RV32IMZBS-NEXT:    srli a4, a4, 1
-; RV32IMZBS-NEXT:    xor a3, a3, s0
-; RV32IMZBS-NEXT:    mul a5, a1, t0
-; RV32IMZBS-NEXT:    mul a6, t4, t1
-; RV32IMZBS-NEXT:    mul t6, t5, t2
-; RV32IMZBS-NEXT:    mul s0, a0, a2
-; RV32IMZBS-NEXT:    mul s1, a1, t2
-; RV32IMZBS-NEXT:    mul s2, t4, t0
-; RV32IMZBS-NEXT:    mul s3, t5, a2
-; RV32IMZBS-NEXT:    mul s4, a0, t1
-; RV32IMZBS-NEXT:    mul s5, a1, t1
-; RV32IMZBS-NEXT:    mul s6, t4, a2
-; RV32IMZBS-NEXT:    mul s7, t5, t0
-; RV32IMZBS-NEXT:    mul a1, a1, a2
-; RV32IMZBS-NEXT:    mul a2, a0, t2
-; RV32IMZBS-NEXT:    mul t2, t4, t2
-; RV32IMZBS-NEXT:    mul t1, t5, t1
-; RV32IMZBS-NEXT:    mul a0, a0, t0
-; RV32IMZBS-NEXT:    xor a5, a6, a5
-; RV32IMZBS-NEXT:    xor a6, t6, s0
-; RV32IMZBS-NEXT:    xor t0, s2, s1
-; RV32IMZBS-NEXT:    xor t4, s3, s4
+; RV32IMZBS-NEXT:    xor a3, a4, a3
+; RV32IMZBS-NEXT:    srli a4, a3, 8
+; RV32IMZBS-NEXT:    lw s2, 36(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s2
+; RV32IMZBS-NEXT:    srli a5, a3, 24
+; RV32IMZBS-NEXT:    and a6, a3, s2
+; RV32IMZBS-NEXT:    slli a3, a3, 24
+; RV32IMZBS-NEXT:    slli a6, a6, 8
+; RV32IMZBS-NEXT:    or a4, a4, a5
+; RV32IMZBS-NEXT:    or a3, a3, a6
+; RV32IMZBS-NEXT:    or a3, a3, a4
+; RV32IMZBS-NEXT:    srli a4, a3, 4
+; RV32IMZBS-NEXT:    lw a5, 104(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a3, a5
+; RV32IMZBS-NEXT:    and a4, a4, a5
+; RV32IMZBS-NEXT:    slli a3, a3, 4
+; RV32IMZBS-NEXT:    or ra, a4, a3
+; RV32IMZBS-NEXT:    lw s0, 40(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a3, a0, s0
+; RV32IMZBS-NEXT:    lw s1, 44(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a4, a1, s1
+; RV32IMZBS-NEXT:    lw s7, 8(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a5, a2, s7
+; RV32IMZBS-NEXT:    mul a6, s3, s9
+; RV32IMZBS-NEXT:    mul a7, a0, s7
+; RV32IMZBS-NEXT:    mul t0, a1, s0
+; RV32IMZBS-NEXT:    mul t1, a2, s9
+; RV32IMZBS-NEXT:    mul t2, s3, s1
+; RV32IMZBS-NEXT:    mul t3, a0, s1
+; RV32IMZBS-NEXT:    mul t4, a1, s9
+; RV32IMZBS-NEXT:    mul t5, a2, s0
+; RV32IMZBS-NEXT:    mul t6, s3, s7
+; RV32IMZBS-NEXT:    mul a0, a0, s9
+; RV32IMZBS-NEXT:    mul a1, a1, s7
+; RV32IMZBS-NEXT:    mul a2, a2, s1
+; RV32IMZBS-NEXT:    mul s0, s3, s0
+; RV32IMZBS-NEXT:    xor a3, a4, a3
+; RV32IMZBS-NEXT:    xor a4, a5, a6
+; RV32IMZBS-NEXT:    xor a5, t0, a7
+; RV32IMZBS-NEXT:    xor a6, t1, t2
+; RV32IMZBS-NEXT:    xor a3, a3, a4
+; RV32IMZBS-NEXT:    xor a4, a5, a6
+; RV32IMZBS-NEXT:    xor a5, t4, t3
+; RV32IMZBS-NEXT:    xor a6, t5, t6
 ; RV32IMZBS-NEXT:    xor a5, a5, a6
-; RV32IMZBS-NEXT:    xor a6, t0, t4
-; RV32IMZBS-NEXT:    and a5, a5, s8
-; RV32IMZBS-NEXT:    and a6, a6, a7
-; RV32IMZBS-NEXT:    xor t0, s6, s5
-; RV32IMZBS-NEXT:    xor a2, s7, a2
-; RV32IMZBS-NEXT:    xor a1, t2, a1
-; RV32IMZBS-NEXT:    xor a0, t1, a0
-; RV32IMZBS-NEXT:    xor a2, t0, a2
 ; RV32IMZBS-NEXT:    xor a0, a1, a0
-; RV32IMZBS-NEXT:    and a1, a2, t3
-; RV32IMZBS-NEXT:    and a0, a0, s9
-; RV32IMZBS-NEXT:    or a2, a6, a5
-; RV32IMZBS-NEXT:    or a0, a1, a0
-; RV32IMZBS-NEXT:    xor a1, a4, a3
+; RV32IMZBS-NEXT:    xor a2, a2, s0
+; RV32IMZBS-NEXT:    srli a1, ra, 2
+; RV32IMZBS-NEXT:    xor a0, a0, a2
+; RV32IMZBS-NEXT:    lw s0, 56(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, s0
+; RV32IMZBS-NEXT:    lw s7, 120(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a2, a3, s7
+; RV32IMZBS-NEXT:    mv s1, s5
+; RV32IMZBS-NEXT:    and a3, a4, s5
+; RV32IMZBS-NEXT:    and a4, a5, s4
+; RV32IMZBS-NEXT:    mv s5, s4
+; RV32IMZBS-NEXT:    mv s9, s6
+; RV32IMZBS-NEXT:    and a0, a0, s6
+; RV32IMZBS-NEXT:    or a2, a3, a2
+; RV32IMZBS-NEXT:    or a0, a4, a0
+; RV32IMZBS-NEXT:    and a3, ra, s0
 ; RV32IMZBS-NEXT:    or a0, a2, a0
+; RV32IMZBS-NEXT:    slli a3, a3, 2
+; RV32IMZBS-NEXT:    srli a2, a0, 8
+; RV32IMZBS-NEXT:    or a1, a1, a3
+; RV32IMZBS-NEXT:    and a2, a2, s2
+; RV32IMZBS-NEXT:    srli a3, a0, 24
+; RV32IMZBS-NEXT:    and a4, a0, s2
+; RV32IMZBS-NEXT:    lw t3, 68(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s4, 48(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a5, t3, s4
+; RV32IMZBS-NEXT:    lw t4, 64(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a6, t4, s8
+; RV32IMZBS-NEXT:    lw t5, 60(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a7, t5, s10
+; RV32IMZBS-NEXT:    lw t6, 72(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul t0, t6, s11
+; RV32IMZBS-NEXT:    slli a0, a0, 24
+; RV32IMZBS-NEXT:    slli a4, a4, 8
+; RV32IMZBS-NEXT:    or a2, a2, a3
+; RV32IMZBS-NEXT:    or a0, a0, a4
+; RV32IMZBS-NEXT:    srli a3, a1, 1
+; RV32IMZBS-NEXT:    lw s3, 108(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a1, a1, s3
 ; RV32IMZBS-NEXT:    lw ra, 76(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s0, 72(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s1, 68(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s2, 64(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s3, 60(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s4, 56(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s5, 52(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s6, 48(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s7, 44(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s8, 40(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s9, 36(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s10, 32(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    lw s11, 28(sp) # 4-byte Folded Reload
-; RV32IMZBS-NEXT:    addi sp, sp, 80
+; RV32IMZBS-NEXT:    and a3, a3, ra
+; RV32IMZBS-NEXT:    slli a1, a1, 1
+; RV32IMZBS-NEXT:    or a1, a3, a1
+; RV32IMZBS-NEXT:    sw a1, 80(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a0, a0, a2
+; RV32IMZBS-NEXT:    xor a1, a6, a5
+; RV32IMZBS-NEXT:    xor a2, a7, t0
+; RV32IMZBS-NEXT:    mul a3, t3, s10
+; RV32IMZBS-NEXT:    mul a4, t4, s4
+; RV32IMZBS-NEXT:    mul a5, t5, s11
+; RV32IMZBS-NEXT:    mul a6, t6, s8
+; RV32IMZBS-NEXT:    mul a7, t3, s8
+; RV32IMZBS-NEXT:    mul t0, t3, s11
+; RV32IMZBS-NEXT:    mul t1, t4, s11
+; RV32IMZBS-NEXT:    mul t2, t4, s10
+; RV32IMZBS-NEXT:    mul t3, t6, s10
+; RV32IMZBS-NEXT:    mul t4, t5, s4
+; RV32IMZBS-NEXT:    mul t5, t5, s8
+; RV32IMZBS-NEXT:    mul t6, t6, s4
+; RV32IMZBS-NEXT:    xor a1, a1, a2
+; RV32IMZBS-NEXT:    xor a3, a4, a3
+; RV32IMZBS-NEXT:    xor a2, a5, a6
+; RV32IMZBS-NEXT:    srli a4, a0, 4
+; RV32IMZBS-NEXT:    lw s6, 104(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a4, a4, s6
+; RV32IMZBS-NEXT:    xor a2, a3, a2
+; RV32IMZBS-NEXT:    and a1, a1, s7
+; RV32IMZBS-NEXT:    and a2, a2, s1
+; RV32IMZBS-NEXT:    xor a3, t1, a7
+; RV32IMZBS-NEXT:    xor a5, t4, t3
+; RV32IMZBS-NEXT:    xor a6, t2, t0
+; RV32IMZBS-NEXT:    xor a7, t5, t6
+; RV32IMZBS-NEXT:    xor a3, a3, a5
+; RV32IMZBS-NEXT:    xor a5, a6, a7
+; RV32IMZBS-NEXT:    mv s11, s5
+; RV32IMZBS-NEXT:    and a3, a3, s5
+; RV32IMZBS-NEXT:    mv t4, s9
+; RV32IMZBS-NEXT:    and a5, a5, s9
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    or a3, a3, a5
+; RV32IMZBS-NEXT:    and a0, a0, s6
+; RV32IMZBS-NEXT:    or a1, a1, a3
+; RV32IMZBS-NEXT:    slli a0, a0, 4
+; RV32IMZBS-NEXT:    srli a2, a1, 8
+; RV32IMZBS-NEXT:    or a0, a4, a0
+; RV32IMZBS-NEXT:    and a2, a2, s2
+; RV32IMZBS-NEXT:    srli a3, a1, 24
+; RV32IMZBS-NEXT:    and a4, a1, s2
+; RV32IMZBS-NEXT:    slli a1, a1, 24
+; RV32IMZBS-NEXT:    slli a4, a4, 8
+; RV32IMZBS-NEXT:    or a2, a2, a3
+; RV32IMZBS-NEXT:    or a1, a1, a4
+; RV32IMZBS-NEXT:    srli a3, a0, 2
+; RV32IMZBS-NEXT:    or a1, a1, a2
+; RV32IMZBS-NEXT:    srli a2, a1, 4
+; RV32IMZBS-NEXT:    and a1, a1, s6
+; RV32IMZBS-NEXT:    and a2, a2, s6
+; RV32IMZBS-NEXT:    slli a1, a1, 4
+; RV32IMZBS-NEXT:    and a4, a3, s0
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    and a0, a0, s0
+; RV32IMZBS-NEXT:    srli a2, a1, 2
+; RV32IMZBS-NEXT:    and a2, a2, s0
+; RV32IMZBS-NEXT:    and a1, a1, s0
+; RV32IMZBS-NEXT:    lw a5, 112(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t0, a5, s1
+; RV32IMZBS-NEXT:    lw a3, 88(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a6, a3, s7
+; RV32IMZBS-NEXT:    and t3, a5, s7
+; RV32IMZBS-NEXT:    and a7, a3, s1
+; RV32IMZBS-NEXT:    mv s8, s1
+; RV32IMZBS-NEXT:    mul s9, a6, t0
+; RV32IMZBS-NEXT:    mul t1, a7, t3
+; RV32IMZBS-NEXT:    mv s0, t4
+; RV32IMZBS-NEXT:    and t4, a5, t4
+; RV32IMZBS-NEXT:    and t5, a3, s5
+; RV32IMZBS-NEXT:    and t2, a5, s5
+; RV32IMZBS-NEXT:    and t6, a3, s0
+; RV32IMZBS-NEXT:    mv a3, s0
+; RV32IMZBS-NEXT:    mul s0, t5, t4
+; RV32IMZBS-NEXT:    mul s1, t6, t2
+; RV32IMZBS-NEXT:    mul s2, a6, t4
+; RV32IMZBS-NEXT:    mul s4, a7, t0
+; RV32IMZBS-NEXT:    mul s5, t5, t2
+; RV32IMZBS-NEXT:    mul s6, t6, t3
+; RV32IMZBS-NEXT:    slli a0, a0, 2
+; RV32IMZBS-NEXT:    slli a1, a1, 2
+; RV32IMZBS-NEXT:    or a0, a4, a0
+; RV32IMZBS-NEXT:    sw a0, 112(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    or a1, a2, a1
+; RV32IMZBS-NEXT:    srli a0, a1, 1
+; RV32IMZBS-NEXT:    and a1, a1, s3
+; RV32IMZBS-NEXT:    and a0, a0, ra
+; RV32IMZBS-NEXT:    sw a0, 76(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    slli a1, a1, 1
+; RV32IMZBS-NEXT:    sw a1, 88(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor a0, t1, s9
+; RV32IMZBS-NEXT:    xor s0, s0, s1
+; RV32IMZBS-NEXT:    xor a1, s4, s2
+; RV32IMZBS-NEXT:    xor a4, s5, s6
+; RV32IMZBS-NEXT:    xor a0, a0, s0
+; RV32IMZBS-NEXT:    xor a1, a1, a4
+; RV32IMZBS-NEXT:    and a0, a0, s7
+; RV32IMZBS-NEXT:    sw a0, 72(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mv a2, s8
+; RV32IMZBS-NEXT:    and t1, a1, s8
+; RV32IMZBS-NEXT:    mul s0, a6, t3
+; RV32IMZBS-NEXT:    mul s1, a7, t2
+; RV32IMZBS-NEXT:    sw t0, 104(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    mul s2, t5, t0
+; RV32IMZBS-NEXT:    mul s4, t6, t4
+; RV32IMZBS-NEXT:    mul a6, a6, t2
+; RV32IMZBS-NEXT:    mul a7, a7, t4
+; RV32IMZBS-NEXT:    mul s5, t5, t3
+; RV32IMZBS-NEXT:    mul s6, t6, t0
+; RV32IMZBS-NEXT:    lw a0, 92(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s8, a0, s8
+; RV32IMZBS-NEXT:    mv a5, a2
+; RV32IMZBS-NEXT:    and s9, a0, s7
+; RV32IMZBS-NEXT:    and s10, a0, a3
+; RV32IMZBS-NEXT:    mv a1, s11
+; RV32IMZBS-NEXT:    and s11, a0, s11
+; RV32IMZBS-NEXT:    lw a2, 84(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and t5, a2, s7
+; RV32IMZBS-NEXT:    and a4, a2, a5
+; RV32IMZBS-NEXT:    and a0, a2, a1
+; RV32IMZBS-NEXT:    and a5, a2, a3
+; RV32IMZBS-NEXT:    mul ra, t5, s8
+; RV32IMZBS-NEXT:    mul t0, a4, s9
+; RV32IMZBS-NEXT:    mul t6, a0, s10
+; RV32IMZBS-NEXT:    mul a2, a5, s11
+; RV32IMZBS-NEXT:    lw s3, 88(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s7, 76(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    or s3, s7, s3
+; RV32IMZBS-NEXT:    sw s3, 92(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    lw s3, 72(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    or t1, t1, s3
+; RV32IMZBS-NEXT:    sw t1, 88(sp) # 4-byte Folded Spill
+; RV32IMZBS-NEXT:    xor s0, s1, s0
+; RV32IMZBS-NEXT:    xor t1, s2, s4
+; RV32IMZBS-NEXT:    xor a6, a7, a6
+; RV32IMZBS-NEXT:    xor a7, s5, s6
+; RV32IMZBS-NEXT:    xor t1, s0, t1
+; RV32IMZBS-NEXT:    xor a6, a6, a7
+; RV32IMZBS-NEXT:    and a7, t1, a1
+; RV32IMZBS-NEXT:    mv a1, a3
+; RV32IMZBS-NEXT:    and a6, a6, a3
+; RV32IMZBS-NEXT:    xor a3, t0, ra
+; RV32IMZBS-NEXT:    xor a2, t6, a2
+; RV32IMZBS-NEXT:    mul t1, t5, s10
+; RV32IMZBS-NEXT:    mul t6, a4, s8
+; RV32IMZBS-NEXT:    mul s0, a0, s11
+; RV32IMZBS-NEXT:    mul s1, a5, s9
+; RV32IMZBS-NEXT:    mul s2, t5, s9
+; RV32IMZBS-NEXT:    mul s3, a4, s11
+; RV32IMZBS-NEXT:    mul s4, a0, s8
+; RV32IMZBS-NEXT:    mul s5, a5, s10
+; RV32IMZBS-NEXT:    mul s6, t5, s11
+; RV32IMZBS-NEXT:    mul s10, a4, s10
+; RV32IMZBS-NEXT:    mul s9, a0, s9
+; RV32IMZBS-NEXT:    mul s8, a5, s8
+; RV32IMZBS-NEXT:    or a6, a7, a6
+; RV32IMZBS-NEXT:    xor a2, a3, a2
+; RV32IMZBS-NEXT:    xor a3, t6, t1
+; RV32IMZBS-NEXT:    xor s0, s0, s1
+; RV32IMZBS-NEXT:    xor a3, a3, s0
+; RV32IMZBS-NEXT:    xor a7, s3, s2
+; RV32IMZBS-NEXT:    xor t1, s4, s5
+; RV32IMZBS-NEXT:    lw t6, 80(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    srli t6, t6, 1
+; RV32IMZBS-NEXT:    xor s0, s10, s6
+; RV32IMZBS-NEXT:    lw t0, 112(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    srli s1, t0, 1
+; RV32IMZBS-NEXT:    xor s2, s9, s8
+; RV32IMZBS-NEXT:    slli s3, s1, 31
+; RV32IMZBS-NEXT:    lw s7, 120(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a2, a2, s7
+; RV32IMZBS-NEXT:    lw s11, 100(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a3, a3, s11
+; RV32IMZBS-NEXT:    xor a7, a7, t1
+; RV32IMZBS-NEXT:    xor t1, s0, s2
+; RV32IMZBS-NEXT:    lw ra, 116(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and a7, a7, ra
+; RV32IMZBS-NEXT:    and t1, t1, a1
+; RV32IMZBS-NEXT:    mv s10, a1
+; RV32IMZBS-NEXT:    or a2, a3, a2
+; RV32IMZBS-NEXT:    or a3, a7, t1
+; RV32IMZBS-NEXT:    lw a1, 88(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    or a6, a1, a6
+; RV32IMZBS-NEXT:    or a2, a2, a3
+; RV32IMZBS-NEXT:    lw a3, 92(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    srli a3, a3, 1
+; RV32IMZBS-NEXT:    xor s0, a2, a6
+; RV32IMZBS-NEXT:    or t6, t6, s3
+; RV32IMZBS-NEXT:    xor s0, a3, s0
+; RV32IMZBS-NEXT:    lw a2, 108(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    and s1, s1, a2
+; RV32IMZBS-NEXT:    and a2, t0, a2
+; RV32IMZBS-NEXT:    lw a1, 104(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    mul a3, t5, a1
+; RV32IMZBS-NEXT:    mul a6, a4, t3
+; RV32IMZBS-NEXT:    mul a7, a0, t4
+; RV32IMZBS-NEXT:    mul s8, a5, t2
+; RV32IMZBS-NEXT:    mul t1, t5, t4
+; RV32IMZBS-NEXT:    mul s2, a4, a1
+; RV32IMZBS-NEXT:    mul s3, a0, t2
+; RV32IMZBS-NEXT:    mul s4, a5, t3
+; RV32IMZBS-NEXT:    mul s5, t5, t3
+; RV32IMZBS-NEXT:    mul s6, a4, t2
+; RV32IMZBS-NEXT:    mul s9, t5, t2
+; RV32IMZBS-NEXT:    mul t5, a0, a1
+; RV32IMZBS-NEXT:    mul a4, a4, t4
+; RV32IMZBS-NEXT:    mul t4, a5, t4
+; RV32IMZBS-NEXT:    mul a0, a0, t3
+; RV32IMZBS-NEXT:    mul a1, a5, a1
+; RV32IMZBS-NEXT:    xor a3, a6, a3
+; RV32IMZBS-NEXT:    xor a6, a7, s8
+; RV32IMZBS-NEXT:    xor a7, s2, t1
+; RV32IMZBS-NEXT:    xor t0, s3, s4
+; RV32IMZBS-NEXT:    xor a3, a3, a6
+; RV32IMZBS-NEXT:    xor a6, a7, t0
+; RV32IMZBS-NEXT:    and a3, a3, s7
+; RV32IMZBS-NEXT:    and a6, a6, s11
+; RV32IMZBS-NEXT:    xor a7, s6, s5
+; RV32IMZBS-NEXT:    xor t0, t5, t4
+; RV32IMZBS-NEXT:    xor a4, a4, s9
+; RV32IMZBS-NEXT:    xor a0, a0, a1
+; RV32IMZBS-NEXT:    xor a1, a7, t0
+; RV32IMZBS-NEXT:    xor a0, a4, a0
+; RV32IMZBS-NEXT:    and a1, a1, ra
+; RV32IMZBS-NEXT:    and a0, a0, s10
+; RV32IMZBS-NEXT:    or a3, a6, a3
+; RV32IMZBS-NEXT:    or a0, a1, a0
+; RV32IMZBS-NEXT:    or a0, a3, a0
+; RV32IMZBS-NEXT:    slli a2, a2, 1
+; RV32IMZBS-NEXT:    or a2, s1, a2
+; RV32IMZBS-NEXT:    lw a1, 96(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    sw a0, 0(a1)
+; RV32IMZBS-NEXT:    sw s0, 4(a1)
+; RV32IMZBS-NEXT:    srli a2, a2, 1
+; RV32IMZBS-NEXT:    sw t6, 8(a1)
+; RV32IMZBS-NEXT:    sw a2, 12(a1)
+; RV32IMZBS-NEXT:    lw ra, 172(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s0, 168(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s1, 164(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s2, 160(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s3, 156(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s4, 152(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s5, 148(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s6, 144(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s7, 140(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s8, 136(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s9, 132(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s10, 128(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    lw s11, 124(sp) # 4-byte Folded Reload
+; RV32IMZBS-NEXT:    .cfi_restore ra
+; RV32IMZBS-NEXT:    .cfi_restore s0
+; RV32IMZBS-NEXT:    .cfi_restore s1
+; RV32IMZBS-NEXT:    .cfi_restore s2
+; RV32IMZBS-NEXT:    .cfi_restore s3
+; RV32IMZBS-NEXT:    .cfi_restore s4
+; RV32IMZBS-NEXT:    .cfi_restore s5
+; RV32IMZBS-NEXT:    .cfi_restore s6
+; RV32IMZBS-NEXT:    .cfi_restore s7
+; RV32IMZBS-NEXT:    .cfi_restore s8
+; RV32IMZBS-NEXT:    .cfi_restore s9
+; RV32IMZBS-NEXT:    .cfi_restore s10
+; RV32IMZBS-NEXT:    .cfi_restore s11
+; RV32IMZBS-NEXT:    addi sp, sp, 176
+; RV32IMZBS-NEXT:    .cfi_def_cfa_offset 0
 ; RV32IMZBS-NEXT:    ret
 ;
-; RV64IMZBS-LABEL: clmul_i64:
+; RV64IMZBS-LABEL: clmul_i128_zext:
 ; RV64IMZBS:       # %bb.0:
-; RV64IMZBS-NEXT:    addi sp, sp, -32
-; RV64IMZBS-NEXT:    sd s0, 24(sp) # 8-byte Folded Spill
-; RV64IMZBS-NEXT:    sd s1, 16(sp) # 8-byte Folded Spill
-; RV64IMZBS-NEXT:    sd s2, 8(sp) # 8-byte Folded Spill
-; RV64IMZBS-NEXT:    sd s3, 0(sp) # 8-byte Folded Spill
-; RV64IMZBS-NEXT:    lui a2, 69905
-; RV64IMZBS-NEXT:    lui a3, 139810
-; RV64IMZBS-NEXT:    addi a2, a2, 273
-; RV64IMZBS-NEXT:    addi a3, a3, 546
-; RV64IMZBS-NEXT:    slli a4, a2, 32
-; RV64IMZBS-NEXT:    slli a5, a3, 32
-; RV64IMZBS-NEXT:    add a2, a2, a4
-; RV64IMZBS-NEXT:    add a3, a3, a5
-; RV64IMZBS-NEXT:    and a4, a1, a2
-; RV64IMZBS-NEXT:    and a5, a0, a3
-; RV64IMZBS-NEXT:    mul a6, a5, a4
-; RV64IMZBS-NEXT:    and a7, a1, a3
-; RV64IMZBS-NEXT:    and t0, a0, a2
-; RV64IMZBS-NEXT:    lui t1, 279620
-; RV64IMZBS-NEXT:    mul t2, t0, a7
-; RV64IMZBS-NEXT:    addi t1, t1, 1092
-; RV64IMZBS-NEXT:    lui t3, %hi(.LCPI4_0)
-; RV64IMZBS-NEXT:    slli t4, t1, 32
-; RV64IMZBS-NEXT:    ld t3, %lo(.LCPI4_0)(t3)
-; RV64IMZBS-NEXT:    add t1, t1, t4
-; RV64IMZBS-NEXT:    and t4, a0, t1
-; RV64IMZBS-NEXT:    and t5, a1, t1
-; RV64IMZBS-NEXT:    mul t6, t0, a4
-; RV64IMZBS-NEXT:    mul s0, t4, t5
-; RV64IMZBS-NEXT:    xor a6, t2, a6
-; RV64IMZBS-NEXT:    and a1, a1, t3
-; RV64IMZBS-NEXT:    mul t2, t4, a1
+; RV64IMZBS-NEXT:    addi sp, sp, -112
+; RV64IMZBS-NEXT:    .cfi_def_cfa_offset 112
+; RV64IMZBS-NEXT:    sd ra, 104(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s0, 96(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s1, 88(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s2, 80(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s3, 72(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s4, 64(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s5, 56(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s6, 48(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s7, 40(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s8, 32(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s9, 24(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s10, 16(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    sd s11, 8(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    .cfi_offset ra, -8
+; RV64IMZBS-NEXT:    .cfi_offset s0, -16
+; RV64IMZBS-NEXT:    .cfi_offset s1, -24
+; RV64IMZBS-NEXT:    .cfi_offset s2, -32
+; RV64IMZBS-NEXT:    .cfi_offset s3, -40
+; RV64IMZBS-NEXT:    .cfi_offset s4, -48
+; RV64IMZBS-NEXT:    .cfi_offset s5, -56
+; RV64IMZBS-NEXT:    .cfi_offset s6, -64
+; RV64IMZBS-NEXT:    .cfi_offset s7, -72
+; RV64IMZBS-NEXT:    .cfi_offset s8, -80
+; RV64IMZBS-NEXT:    .cfi_offset s9, -88
+; RV64IMZBS-NEXT:    .cfi_offset s10, -96
+; RV64IMZBS-NEXT:    .cfi_offset s11, -104
+; RV64IMZBS-NEXT:    mv a2, a0
+; RV64IMZBS-NEXT:    lui a4, 4080
+; RV64IMZBS-NEXT:    srli a3, a1, 24
+; RV64IMZBS-NEXT:    li a5, 255
+; RV64IMZBS-NEXT:    and a4, a3, a4
+; RV64IMZBS-NEXT:    lui a0, 4080
+; RV64IMZBS-NEXT:    srli a3, a1, 8
+; RV64IMZBS-NEXT:    slli t3, a5, 24
+; RV64IMZBS-NEXT:    and a6, a3, t3
+; RV64IMZBS-NEXT:    sd t3, 0(sp) # 8-byte Folded Spill
+; RV64IMZBS-NEXT:    lui a3, 16
+; RV64IMZBS-NEXT:    srli a7, a1, 40
+; RV64IMZBS-NEXT:    addi a5, a3, -256
+; RV64IMZBS-NEXT:    and a7, a7, a5
+; RV64IMZBS-NEXT:    srli t0, a1, 56
+; RV64IMZBS-NEXT:    or a4, a6, a4
+; RV64IMZBS-NEXT:    or a6, a7, t0
+; RV64IMZBS-NEXT:    or a4, a4, a6
+; RV64IMZBS-NEXT:    and a6, a1, a0
+; RV64IMZBS-NEXT:    srliw a7, a1, 24
+; RV64IMZBS-NEXT:    slli a6, a6, 24
+; RV64IMZBS-NEXT:    slli a7, a7, 32
+; RV64IMZBS-NEXT:    or a6, a6, a7
+; RV64IMZBS-NEXT:    and a7, a1, a5
+; RV64IMZBS-NEXT:    slli a7, a7, 40
+; RV64IMZBS-NEXT:    slli t0, a1, 56
+; RV64IMZBS-NEXT:    or a7, t0, a7
+; RV64IMZBS-NEXT:    lui t0, 61681
+; RV64IMZBS-NEXT:    or a6, a7, a6
+; RV64IMZBS-NEXT:    addi a7, t0, -241
+; RV64IMZBS-NEXT:    or a4, a6, a4
+; RV64IMZBS-NEXT:    slli a6, a7, 32
+; RV64IMZBS-NEXT:    srli t0, a4, 4
+; RV64IMZBS-NEXT:    add a6, a7, a6
+; RV64IMZBS-NEXT:    and a7, t0, a6
+; RV64IMZBS-NEXT:    and a4, a4, a6
+; RV64IMZBS-NEXT:    lui t0, 209715
+; RV64IMZBS-NEXT:    slli a4, a4, 4
+; RV64IMZBS-NEXT:    addi t0, t0, 819
+; RV64IMZBS-NEXT:    or a4, a7, a4
+; RV64IMZBS-NEXT:    slli a7, t0, 32
+; RV64IMZBS-NEXT:    srli t1, a4, 2
+; RV64IMZBS-NEXT:    add t0, t0, a7
+; RV64IMZBS-NEXT:    and a7, t1, t0
+; RV64IMZBS-NEXT:    lui t1, 349525
+; RV64IMZBS-NEXT:    and a4, a4, t0
+; RV64IMZBS-NEXT:    addi t1, t1, 1365
+; RV64IMZBS-NEXT:    slli a4, a4, 2
+; RV64IMZBS-NEXT:    slli t2, t1, 32
+; RV64IMZBS-NEXT:    or a7, a7, a4
+; RV64IMZBS-NEXT:    add a4, t1, t2
+; RV64IMZBS-NEXT:    srli t1, a7, 1
+; RV64IMZBS-NEXT:    and a7, a7, a4
+; RV64IMZBS-NEXT:    and t1, t1, a4
+; RV64IMZBS-NEXT:    slli a7, a7, 1
+; RV64IMZBS-NEXT:    or t4, t1, a7
+; RV64IMZBS-NEXT:    srli a7, a2, 24
+; RV64IMZBS-NEXT:    srli t1, a2, 8
+; RV64IMZBS-NEXT:    and a7, a7, a0
+; RV64IMZBS-NEXT:    and t1, t1, t3
+; RV64IMZBS-NEXT:    or a7, t1, a7
+; RV64IMZBS-NEXT:    srli t1, a2, 40
+; RV64IMZBS-NEXT:    and t1, t1, a5
+; RV64IMZBS-NEXT:    srli t2, a2, 56
+; RV64IMZBS-NEXT:    or t1, t1, t2
+; RV64IMZBS-NEXT:    and t2, a2, a0
+; RV64IMZBS-NEXT:    slli t2, t2, 24
+; RV64IMZBS-NEXT:    srliw t3, a2, 24
+; RV64IMZBS-NEXT:    slli t3, t3, 32
+; RV64IMZBS-NEXT:    and t5, a2, a5
+; RV64IMZBS-NEXT:    slli t5, t5, 40
+; RV64IMZBS-NEXT:    slli t6, a2, 56
+; RV64IMZBS-NEXT:    or t2, t2, t3
+; RV64IMZBS-NEXT:    or t3, t6, t5
+; RV64IMZBS-NEXT:    or a7, a7, t1
+; RV64IMZBS-NEXT:    or t1, t3, t2
+; RV64IMZBS-NEXT:    lui t2, 69905
+; RV64IMZBS-NEXT:    or a7, t1, a7
+; RV64IMZBS-NEXT:    srli t1, a7, 4
+; RV64IMZBS-NEXT:    and a7, a7, a6
+; RV64IMZBS-NEXT:    and t1, t1, a6
+; RV64IMZBS-NEXT:    slli a7, a7, 4
+; RV64IMZBS-NEXT:    addi t2, t2, 273
+; RV64IMZBS-NEXT:    or a7, t1, a7
+; RV64IMZBS-NEXT:    srli t1, a7, 2
+; RV64IMZBS-NEXT:    and a7, a7, t0
+; RV64IMZBS-NEXT:    and t1, t1, t0
+; RV64IMZBS-NEXT:    slli a7, a7, 2
+; RV64IMZBS-NEXT:    slli t3, t2, 32
+; RV64IMZBS-NEXT:    or t1, t1, a7
+; RV64IMZBS-NEXT:    add a7, t2, t3
+; RV64IMZBS-NEXT:    srli t2, t1, 1
+; RV64IMZBS-NEXT:    and t2, t2, a4
+; RV64IMZBS-NEXT:    lui t3, 139810
+; RV64IMZBS-NEXT:    and t1, t1, a4
+; RV64IMZBS-NEXT:    addi t3, t3, 546
+; RV64IMZBS-NEXT:    slli t1, t1, 1
+; RV64IMZBS-NEXT:    slli t5, t3, 32
+; RV64IMZBS-NEXT:    or t6, t2, t1
+; RV64IMZBS-NEXT:    add t1, t3, t5
+; RV64IMZBS-NEXT:    and t5, t4, a7
+; RV64IMZBS-NEXT:    and s0, t6, t1
+; RV64IMZBS-NEXT:    mul s1, s0, t5
+; RV64IMZBS-NEXT:    lui t2, %hi(.LCPI9_0)
+; RV64IMZBS-NEXT:    ld t2, %lo(.LCPI9_0)(t2)
+; RV64IMZBS-NEXT:    lui t3, 279620
+; RV64IMZBS-NEXT:    and s2, t4, t1
+; RV64IMZBS-NEXT:    addi t3, t3, 1092
+; RV64IMZBS-NEXT:    and s3, t6, a7
+; RV64IMZBS-NEXT:    slli s4, t3, 32
+; RV64IMZBS-NEXT:    mul s5, s3, s2
+; RV64IMZBS-NEXT:    add t3, t3, s4
+; RV64IMZBS-NEXT:    and s4, t4, t2
+; RV64IMZBS-NEXT:    and s6, t6, t3
+; RV64IMZBS-NEXT:    and t4, t4, t3
+; RV64IMZBS-NEXT:    and t6, t6, t2
+; RV64IMZBS-NEXT:    mul s7, s6, s4
+; RV64IMZBS-NEXT:    mul s8, t6, t4
+; RV64IMZBS-NEXT:    mul s9, s0, s4
+; RV64IMZBS-NEXT:    mul s10, s3, t5
+; RV64IMZBS-NEXT:    mul s11, s6, t4
+; RV64IMZBS-NEXT:    mul ra, t6, s2
+; RV64IMZBS-NEXT:    mul a3, s0, s2
+; RV64IMZBS-NEXT:    mul s0, s0, t4
+; RV64IMZBS-NEXT:    mul t4, s3, t4
+; RV64IMZBS-NEXT:    mul s3, s3, s4
+; RV64IMZBS-NEXT:    mul s4, t6, s4
+; RV64IMZBS-NEXT:    mul a0, s6, t5
+; RV64IMZBS-NEXT:    mul s2, s6, s2
+; RV64IMZBS-NEXT:    mul t5, t6, t5
+; RV64IMZBS-NEXT:    xor t6, s5, s1
+; RV64IMZBS-NEXT:    xor s1, s7, s8
+; RV64IMZBS-NEXT:    xor s5, s10, s9
+; RV64IMZBS-NEXT:    xor s6, s11, ra
+; RV64IMZBS-NEXT:    xor t6, t6, s1
+; RV64IMZBS-NEXT:    xor s1, s5, s6
+; RV64IMZBS-NEXT:    and t6, t6, t1
+; RV64IMZBS-NEXT:    and s1, s1, a7
+; RV64IMZBS-NEXT:    xor a3, t4, a3
+; RV64IMZBS-NEXT:    xor a0, a0, s4
+; RV64IMZBS-NEXT:    xor t4, s3, s0
+; RV64IMZBS-NEXT:    xor t5, s2, t5
+; RV64IMZBS-NEXT:    xor a0, a3, a0
+; RV64IMZBS-NEXT:    xor a3, t4, t5
 ; RV64IMZBS-NEXT:    and a0, a0, t3
-; RV64IMZBS-NEXT:    mul s1, a0, t5
-; RV64IMZBS-NEXT:    mul s2, a5, a1
-; RV64IMZBS-NEXT:    xor t6, t6, s0
-; RV64IMZBS-NEXT:    mul s0, a0, a7
-; RV64IMZBS-NEXT:    mul s3, a5, a7
-; RV64IMZBS-NEXT:    mul a5, a5, t5
-; RV64IMZBS-NEXT:    mul t5, t0, t5
-; RV64IMZBS-NEXT:    mul a7, t4, a7
-; RV64IMZBS-NEXT:    mul t4, t4, a4
+; RV64IMZBS-NEXT:    and a3, a3, t2
+; RV64IMZBS-NEXT:    or t4, s1, t6
+; RV64IMZBS-NEXT:    or a0, a0, a3
+; RV64IMZBS-NEXT:    or a0, t4, a0
+; RV64IMZBS-NEXT:    srli a3, a0, 40
+; RV64IMZBS-NEXT:    and a3, a3, a5
+; RV64IMZBS-NEXT:    srli t4, a0, 56
+; RV64IMZBS-NEXT:    or a3, a3, t4
+; RV64IMZBS-NEXT:    srli t4, a0, 24
+; RV64IMZBS-NEXT:    srli t5, a0, 8
+; RV64IMZBS-NEXT:    lui t6, 4080
+; RV64IMZBS-NEXT:    and t4, t4, t6
+; RV64IMZBS-NEXT:    ld s0, 0(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    and t5, t5, s0
+; RV64IMZBS-NEXT:    or t4, t5, t4
+; RV64IMZBS-NEXT:    srliw t5, a0, 24
+; RV64IMZBS-NEXT:    slli t5, t5, 32
+; RV64IMZBS-NEXT:    and t6, a0, t6
+; RV64IMZBS-NEXT:    slli t6, t6, 24
+; RV64IMZBS-NEXT:    and a5, a0, a5
+; RV64IMZBS-NEXT:    slli a0, a0, 56
+; RV64IMZBS-NEXT:    slli a5, a5, 40
+; RV64IMZBS-NEXT:    or t5, t6, t5
+; RV64IMZBS-NEXT:    or a0, a0, a5
+; RV64IMZBS-NEXT:    or a3, t4, a3
+; RV64IMZBS-NEXT:    or a0, a0, t5
+; RV64IMZBS-NEXT:    or a0, a0, a3
+; RV64IMZBS-NEXT:    srli a3, a0, 4
+; RV64IMZBS-NEXT:    and a0, a0, a6
+; RV64IMZBS-NEXT:    and a3, a3, a6
+; RV64IMZBS-NEXT:    slli a0, a0, 4
+; RV64IMZBS-NEXT:    or a0, a3, a0
+; RV64IMZBS-NEXT:    srli a3, a0, 2
+; RV64IMZBS-NEXT:    and a0, a0, t0
+; RV64IMZBS-NEXT:    and a3, a3, t0
+; RV64IMZBS-NEXT:    slli a0, a0, 2
+; RV64IMZBS-NEXT:    or a0, a3, a0
+; RV64IMZBS-NEXT:    and a3, a1, a7
+; RV64IMZBS-NEXT:    and a5, a2, t1
+; RV64IMZBS-NEXT:    and a6, a1, t1
+; RV64IMZBS-NEXT:    and t0, a2, a7
+; RV64IMZBS-NEXT:    mul t4, a5, a3
+; RV64IMZBS-NEXT:    mul t5, t0, a6
+; RV64IMZBS-NEXT:    srli t6, a0, 1
+; RV64IMZBS-NEXT:    and a0, a0, a4
+; RV64IMZBS-NEXT:    and s0, a2, t3
+; RV64IMZBS-NEXT:    and s1, a1, t3
+; RV64IMZBS-NEXT:    mul s2, t0, a3
+; RV64IMZBS-NEXT:    mul s3, s0, s1
+; RV64IMZBS-NEXT:    and a4, t6, a4
+; RV64IMZBS-NEXT:    slli a0, a0, 1
+; RV64IMZBS-NEXT:    or a0, a4, a0
+; RV64IMZBS-NEXT:    xor a4, t5, t4
+; RV64IMZBS-NEXT:    and a1, a1, t2
+; RV64IMZBS-NEXT:    mul t4, s0, a1
+; RV64IMZBS-NEXT:    and a2, a2, t2
+; RV64IMZBS-NEXT:    mul t5, a2, s1
+; RV64IMZBS-NEXT:    mul t6, a5, a1
+; RV64IMZBS-NEXT:    xor s2, s2, s3
+; RV64IMZBS-NEXT:    mul s3, a2, a6
+; RV64IMZBS-NEXT:    mul s4, a5, a6
+; RV64IMZBS-NEXT:    mul s5, t0, s1
+; RV64IMZBS-NEXT:    mul a5, a5, s1
+; RV64IMZBS-NEXT:    mul a6, s0, a6
+; RV64IMZBS-NEXT:    mul s0, s0, a3
 ; RV64IMZBS-NEXT:    mul t0, t0, a1
-; RV64IMZBS-NEXT:    mul a1, a0, a1
-; RV64IMZBS-NEXT:    mul a0, a0, a4
-; RV64IMZBS-NEXT:    xor a4, a6, t2
-; RV64IMZBS-NEXT:    xor a6, t6, s2
-; RV64IMZBS-NEXT:    xor a4, a4, s1
-; RV64IMZBS-NEXT:    xor a6, a6, s0
-; RV64IMZBS-NEXT:    and a3, a4, a3
-; RV64IMZBS-NEXT:    and a2, a6, a2
-; RV64IMZBS-NEXT:    xor a4, t5, s3
-; RV64IMZBS-NEXT:    xor a5, a5, a7
-; RV64IMZBS-NEXT:    xor a4, a4, t4
+; RV64IMZBS-NEXT:    mul a1, a2, a1
+; RV64IMZBS-NEXT:    mul a2, a2, a3
+; RV64IMZBS-NEXT:    xor a3, a4, t4
+; RV64IMZBS-NEXT:    xor a4, s2, t6
+; RV64IMZBS-NEXT:    xor a3, a3, t5
+; RV64IMZBS-NEXT:    xor a4, a4, s3
+; RV64IMZBS-NEXT:    and a3, a3, t1
+; RV64IMZBS-NEXT:    and a4, a4, a7
+; RV64IMZBS-NEXT:    xor a7, s5, s4
+; RV64IMZBS-NEXT:    xor a5, a5, a6
+; RV64IMZBS-NEXT:    xor a6, a7, s0
 ; RV64IMZBS-NEXT:    xor a5, t0, a5
-; RV64IMZBS-NEXT:    xor a1, a4, a1
-; RV64IMZBS-NEXT:    xor a0, a5, a0
-; RV64IMZBS-NEXT:    and a1, a1, t1
-; RV64IMZBS-NEXT:    and a0, a0, t3
-; RV64IMZBS-NEXT:    or a2, a2, a3
-; RV64IMZBS-NEXT:    or a0, a1, a0
-; RV64IMZBS-NEXT:    or a0, a2, a0
-; RV64IMZBS-NEXT:    ld s0, 24(sp) # 8-byte Folded Reload
-; RV64IMZBS-NEXT:    ld s1, 16(sp) # 8-byte Folded Reload
-; RV64IMZBS-NEXT:    ld s2, 8(sp) # 8-byte Folded Reload
-; RV64IMZBS-NEXT:    ld s3, 0(sp) # 8-byte Folded Reload
-; RV64IMZBS-NEXT:    addi sp, sp, 32
+; RV64IMZBS-NEXT:    xor a1, a6, a1
+; RV64IMZBS-NEXT:    xor a2, a5, a2
+; RV64IMZBS-NEXT:    and a1, a1, t3
+; RV64IMZBS-NEXT:    and a2, a2, t2
+; RV64IMZBS-NEXT:    or a3, a4, a3
+; RV64IMZBS-NEXT:    or a2, a1, a2
+; RV64IMZBS-NEXT:    srli a1, a0, 1
+; RV64IMZBS-NEXT:    or a0, a3, a2
+; RV64IMZBS-NEXT:    ld ra, 104(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s0, 96(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s1, 88(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s2, 80(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s3, 72(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s4, 64(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s5, 56(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s6, 48(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s7, 40(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s8, 32(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s9, 24(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s10, 16(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    ld s11, 8(sp) # 8-byte Folded Reload
+; RV64IMZBS-NEXT:    .cfi_restore ra
+; RV64IMZBS-NEXT:    .cfi_restore s0
+; RV64IMZBS-NEXT:    .cfi_restore s1
+; RV64IMZBS-NEXT:    .cfi_restore s2
+; RV64IMZBS-NEXT:    .cfi_restore s3
+; RV64IMZBS-NEXT:    .cfi_restore s4
+; RV64IMZBS-NEXT:    .cfi_restore s5
+; RV64IMZBS-NEXT:    .cfi_restore s6
+; RV64IMZBS-NEXT:    .cfi_restore s7
+; RV64IMZBS-NEXT:    .cfi_restore s8
+; RV64IMZBS-NEXT:    .cfi_restore s9
+; RV64IMZBS-NEXT:    .cfi_restore s10
+; RV64IMZBS-NEXT:    .cfi_restore s11
+; RV64IMZBS-NEXT:    addi sp, sp, 112
+; RV64IMZBS-NEXT:    .cfi_def_cfa_offset 0
 ; RV64IMZBS-NEXT:    ret
-  %res = call i64 @llvm.clmul.i64(i64 %a, i64 %b)
-  ret i64 %res
+;
+; RV32IMZBC-LABEL: clmul_i128_zext:
+; RV32IMZBC:       # %bb.0:
+; RV32IMZBC-NEXT:    addi sp, sp, -16
+; RV32IMZBC-NEXT:    .cfi_def_cfa_offset 16
+; RV32IMZBC-NEXT:    sw s0, 12(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    .cfi_offset s0, -4
+; RV32IMZBC-NEXT:    lui a5, 16
+; RV32IMZBC-NEXT:    srli a6, a4, 8
+; RV32IMZBC-NEXT:    addi a5, a5, -256
+; RV32IMZBC-NEXT:    and a6, a6, a5
+; RV32IMZBC-NEXT:    srli a7, a4, 24
+; RV32IMZBC-NEXT:    and t0, a4, a5
+; RV32IMZBC-NEXT:    slli t0, t0, 8
+; RV32IMZBC-NEXT:    slli t1, a4, 24
+; RV32IMZBC-NEXT:    or a6, a6, a7
+; RV32IMZBC-NEXT:    or a7, t1, t0
+; RV32IMZBC-NEXT:    or a7, a7, a6
+; RV32IMZBC-NEXT:    lui a6, 61681
+; RV32IMZBC-NEXT:    srli t0, a7, 4
+; RV32IMZBC-NEXT:    addi a6, a6, -241
+; RV32IMZBC-NEXT:    and t0, t0, a6
+; RV32IMZBC-NEXT:    and a7, a7, a6
+; RV32IMZBC-NEXT:    slli a7, a7, 4
+; RV32IMZBC-NEXT:    lui t1, 209715
+; RV32IMZBC-NEXT:    or t0, t0, a7
+; RV32IMZBC-NEXT:    addi a7, t1, 819
+; RV32IMZBC-NEXT:    srli t1, t0, 2
+; RV32IMZBC-NEXT:    and t0, t0, a7
+; RV32IMZBC-NEXT:    and t1, t1, a7
+; RV32IMZBC-NEXT:    slli t0, t0, 2
+; RV32IMZBC-NEXT:    or t2, t1, t0
+; RV32IMZBC-NEXT:    srli t3, t2, 1
+; RV32IMZBC-NEXT:    lui t0, 349525
+; RV32IMZBC-NEXT:    srli t4, a1, 8
+; RV32IMZBC-NEXT:    addi t1, t0, 1365
+; RV32IMZBC-NEXT:    and t4, t4, a5
+; RV32IMZBC-NEXT:    srli t5, a1, 24
+; RV32IMZBC-NEXT:    and t6, a1, a5
+; RV32IMZBC-NEXT:    slli t6, t6, 8
+; RV32IMZBC-NEXT:    slli s0, a1, 24
+; RV32IMZBC-NEXT:    or t4, t4, t5
+; RV32IMZBC-NEXT:    or t5, s0, t6
+; RV32IMZBC-NEXT:    and t3, t3, t1
+; RV32IMZBC-NEXT:    or t4, t5, t4
+; RV32IMZBC-NEXT:    srli t5, t4, 4
+; RV32IMZBC-NEXT:    and t4, t4, a6
+; RV32IMZBC-NEXT:    and t5, t5, a6
+; RV32IMZBC-NEXT:    slli t4, t4, 4
+; RV32IMZBC-NEXT:    and t2, t2, t1
+; RV32IMZBC-NEXT:    or t4, t5, t4
+; RV32IMZBC-NEXT:    srli t5, t4, 2
+; RV32IMZBC-NEXT:    and t4, t4, a7
+; RV32IMZBC-NEXT:    and t5, t5, a7
+; RV32IMZBC-NEXT:    slli t4, t4, 2
+; RV32IMZBC-NEXT:    slli t2, t2, 1
+; RV32IMZBC-NEXT:    or t4, t5, t4
+; RV32IMZBC-NEXT:    or t2, t3, t2
+; RV32IMZBC-NEXT:    srli t3, t4, 1
+; RV32IMZBC-NEXT:    and t3, t3, t1
+; RV32IMZBC-NEXT:    srli t5, a3, 8
+; RV32IMZBC-NEXT:    and t5, t5, a5
+; RV32IMZBC-NEXT:    srli t6, a3, 24
+; RV32IMZBC-NEXT:    or t5, t5, t6
+; RV32IMZBC-NEXT:    and t6, a3, a5
+; RV32IMZBC-NEXT:    slli t6, t6, 8
+; RV32IMZBC-NEXT:    slli s0, a3, 24
+; RV32IMZBC-NEXT:    and t4, t4, t1
+; RV32IMZBC-NEXT:    or t6, s0, t6
+; RV32IMZBC-NEXT:    slli t4, t4, 1
+; RV32IMZBC-NEXT:    or t5, t6, t5
+; RV32IMZBC-NEXT:    srli t6, t5, 4
+; RV32IMZBC-NEXT:    and t5, t5, a6
+; RV32IMZBC-NEXT:    and t6, t6, a6
+; RV32IMZBC-NEXT:    slli t5, t5, 4
+; RV32IMZBC-NEXT:    or t3, t3, t4
+; RV32IMZBC-NEXT:    or t4, t6, t5
+; RV32IMZBC-NEXT:    srli t5, t4, 2
+; RV32IMZBC-NEXT:    and t4, t4, a7
+; RV32IMZBC-NEXT:    and t5, t5, a7
+; RV32IMZBC-NEXT:    slli t4, t4, 2
+; RV32IMZBC-NEXT:    or t4, t5, t4
+; RV32IMZBC-NEXT:    srli t5, a2, 8
+; RV32IMZBC-NEXT:    and t5, t5, a5
+; RV32IMZBC-NEXT:    srli t6, a2, 24
+; RV32IMZBC-NEXT:    or t5, t5, t6
+; RV32IMZBC-NEXT:    and t6, a2, a5
+; RV32IMZBC-NEXT:    slli t6, t6, 8
+; RV32IMZBC-NEXT:    slli s0, a2, 24
+; RV32IMZBC-NEXT:    or t6, s0, t6
+; RV32IMZBC-NEXT:    srli s0, t4, 1
+; RV32IMZBC-NEXT:    and s0, s0, t1
+; RV32IMZBC-NEXT:    or t5, t6, t5
+; RV32IMZBC-NEXT:    srli t6, t5, 4
+; RV32IMZBC-NEXT:    and t5, t5, a6
+; RV32IMZBC-NEXT:    and t6, t6, a6
+; RV32IMZBC-NEXT:    slli t5, t5, 4
+; RV32IMZBC-NEXT:    and t4, t4, t1
+; RV32IMZBC-NEXT:    or t5, t6, t5
+; RV32IMZBC-NEXT:    srli t6, t5, 2
+; RV32IMZBC-NEXT:    and t5, t5, a7
+; RV32IMZBC-NEXT:    and t6, t6, a7
+; RV32IMZBC-NEXT:    slli t5, t5, 2
+; RV32IMZBC-NEXT:    slli t4, t4, 1
+; RV32IMZBC-NEXT:    or t5, t6, t5
+; RV32IMZBC-NEXT:    srli t6, t5, 1
+; RV32IMZBC-NEXT:    and t5, t5, t1
+; RV32IMZBC-NEXT:    and t6, t6, t1
+; RV32IMZBC-NEXT:    slli t5, t5, 1
+; RV32IMZBC-NEXT:    or t4, s0, t4
+; RV32IMZBC-NEXT:    or t5, t6, t5
+; RV32IMZBC-NEXT:    clmul t3, t3, t2
+; RV32IMZBC-NEXT:    clmul t4, t5, t4
+; RV32IMZBC-NEXT:    clmulh t2, t5, t2
+; RV32IMZBC-NEXT:    xor t3, t4, t3
+; RV32IMZBC-NEXT:    xor t2, t2, t3
+; RV32IMZBC-NEXT:    srli t3, t2, 8
+; RV32IMZBC-NEXT:    and t3, t3, a5
+; RV32IMZBC-NEXT:    srli t4, t2, 24
+; RV32IMZBC-NEXT:    and a5, t2, a5
+; RV32IMZBC-NEXT:    slli t2, t2, 24
+; RV32IMZBC-NEXT:    slli a5, a5, 8
+; RV32IMZBC-NEXT:    or t3, t3, t4
+; RV32IMZBC-NEXT:    or a5, t2, a5
+; RV32IMZBC-NEXT:    or a5, a5, t3
+; RV32IMZBC-NEXT:    srli t2, a5, 4
+; RV32IMZBC-NEXT:    and a5, a5, a6
+; RV32IMZBC-NEXT:    and a6, t2, a6
+; RV32IMZBC-NEXT:    slli a5, a5, 4
+; RV32IMZBC-NEXT:    or a5, a6, a5
+; RV32IMZBC-NEXT:    srli a6, a5, 2
+; RV32IMZBC-NEXT:    and a5, a5, a7
+; RV32IMZBC-NEXT:    and a6, a6, a7
+; RV32IMZBC-NEXT:    slli a5, a5, 2
+; RV32IMZBC-NEXT:    or a5, a6, a5
+; RV32IMZBC-NEXT:    srli a6, a5, 1
+; RV32IMZBC-NEXT:    addi a7, t0, 1364
+; RV32IMZBC-NEXT:    and a6, a6, a7
+; RV32IMZBC-NEXT:    and a5, a5, t1
+; RV32IMZBC-NEXT:    slli a5, a5, 1
+; RV32IMZBC-NEXT:    clmulr a7, a2, a4
+; RV32IMZBC-NEXT:    or a5, a6, a5
+; RV32IMZBC-NEXT:    srli a5, a5, 1
+; RV32IMZBC-NEXT:    slli a7, a7, 31
+; RV32IMZBC-NEXT:    clmul a6, a2, a3
+; RV32IMZBC-NEXT:    clmul t0, a1, a4
+; RV32IMZBC-NEXT:    or a5, a5, a7
+; RV32IMZBC-NEXT:    clmulh a7, a1, a3
+; RV32IMZBC-NEXT:    xor a6, t0, a6
+; RV32IMZBC-NEXT:    clmul a1, a1, a3
+; RV32IMZBC-NEXT:    xor a3, a7, a6
+; RV32IMZBC-NEXT:    clmulh a2, a2, a4
+; RV32IMZBC-NEXT:    sw a1, 0(a0)
+; RV32IMZBC-NEXT:    sw a3, 4(a0)
+; RV32IMZBC-NEXT:    sw a5, 8(a0)
+; RV32IMZBC-NEXT:    sw a2, 12(a0)
+; RV32IMZBC-NEXT:    lw s0, 12(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    .cfi_restore s0
+; RV32IMZBC-NEXT:    addi sp, sp, 16
+; RV32IMZBC-NEXT:    .cfi_def_cfa_offset 0
+; RV32IMZBC-NEXT:    ret
+;
+; RV64IMZBC-LABEL: clmul_i128_zext:
+; RV64IMZBC:       # %bb.0:
+; RV64IMZBC-NEXT:    clmul a2, a0, a1
+; RV64IMZBC-NEXT:    clmulh a1, a0, a1
+; RV64IMZBC-NEXT:    mv a0, a2
+; RV64IMZBC-NEXT:    ret
+  %zextx = zext i64 %x to i128
+  %zexty = zext i64 %y to i128
+  %a = call i128 @llvm.clmul.i128(i128 %zextx, i128 %zexty)
+  ret i128 %a
 }
 
 define i4 @clmul_constfold_i4() nounwind {
@@ -3942,6 +27665,13 @@ define void @commutative_clmul_i8(i8 %x, i8 %y, ptr %p0, ptr %p1) nounwind {
 ; RV64IMZBS-NEXT:    sb a0, 0(a2)
 ; RV64IMZBS-NEXT:    sb a0, 0(a3)
 ; RV64IMZBS-NEXT:    ret
+;
+; CHECK-ZBC-LABEL: commutative_clmul_i8:
+; CHECK-ZBC:       # %bb.0:
+; CHECK-ZBC-NEXT:    clmul a0, a0, a1
+; CHECK-ZBC-NEXT:    sb a0, 0(a2)
+; CHECK-ZBC-NEXT:    sb a0, 0(a3)
+; CHECK-ZBC-NEXT:    ret
   %xy = call i8 @llvm.clmul.i8(i8 %x, i8 %y)
   %yx = call i8 @llvm.clmul.i8(i8 %y, i8 %x)
   store i8 %xy, ptr %p0
@@ -4285,6 +28015,42 @@ define void @mul_use_commutative_clmul_i8(i8 %x, i8 %y, ptr %p0, ptr %p1) nounwi
 ; RV64IMZBS-NEXT:    ld s1, 8(sp) # 8-byte Folded Reload
 ; RV64IMZBS-NEXT:    addi sp, sp, 32
 ; RV64IMZBS-NEXT:    ret
+;
+; RV32IMZBC-LABEL: mul_use_commutative_clmul_i8:
+; RV32IMZBC:       # %bb.0:
+; RV32IMZBC-NEXT:    addi sp, sp, -16
+; RV32IMZBC-NEXT:    sw ra, 12(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s0, 8(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s1, 4(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    mv s0, a3
+; RV32IMZBC-NEXT:    clmul s1, a0, a1
+; RV32IMZBC-NEXT:    sb s1, 0(a2)
+; RV32IMZBC-NEXT:    mv a0, s1
+; RV32IMZBC-NEXT:    call use
+; RV32IMZBC-NEXT:    sb s1, 0(s0)
+; RV32IMZBC-NEXT:    lw ra, 12(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s0, 8(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s1, 4(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    addi sp, sp, 16
+; RV32IMZBC-NEXT:    ret
+;
+; RV64IMZBC-LABEL: mul_use_commutative_clmul_i8:
+; RV64IMZBC:       # %bb.0:
+; RV64IMZBC-NEXT:    addi sp, sp, -32
+; RV64IMZBC-NEXT:    sd ra, 24(sp) # 8-byte Folded Spill
+; RV64IMZBC-NEXT:    sd s0, 16(sp) # 8-byte Folded Spill
+; RV64IMZBC-NEXT:    sd s1, 8(sp) # 8-byte Folded Spill
+; RV64IMZBC-NEXT:    mv s0, a3
+; RV64IMZBC-NEXT:    clmul s1, a0, a1
+; RV64IMZBC-NEXT:    sb s1, 0(a2)
+; RV64IMZBC-NEXT:    mv a0, s1
+; RV64IMZBC-NEXT:    call use
+; RV64IMZBC-NEXT:    sb s1, 0(s0)
+; RV64IMZBC-NEXT:    ld ra, 24(sp) # 8-byte Folded Reload
+; RV64IMZBC-NEXT:    ld s0, 16(sp) # 8-byte Folded Reload
+; RV64IMZBC-NEXT:    ld s1, 8(sp) # 8-byte Folded Reload
+; RV64IMZBC-NEXT:    addi sp, sp, 32
+; RV64IMZBC-NEXT:    ret
   %xy = call i8 @llvm.clmul.i8(i8 %x, i8 %y)
   %yx = call i8 @llvm.clmul.i8(i8 %y, i8 %x)
   store i8 %xy, ptr %p0
@@ -4563,6 +28329,13 @@ define void @neg_commutative_clmul_i8(i8 %x, i8 %y, ptr %p0, ptr %p1) nounwind {
 ; RV64IMZBS-NEXT:    sb a0, 0(a2)
 ; RV64IMZBS-NEXT:    sb a0, 0(a3)
 ; RV64IMZBS-NEXT:    ret
+;
+; CHECK-ZBC-LABEL: neg_commutative_clmul_i8:
+; CHECK-ZBC:       # %bb.0:
+; CHECK-ZBC-NEXT:    clmul a0, a0, a1
+; CHECK-ZBC-NEXT:    sb a0, 0(a2)
+; CHECK-ZBC-NEXT:    sb a0, 0(a3)
+; CHECK-ZBC-NEXT:    ret
   %xy = call i8 @llvm.clmul.i8(i8 %x, i8 %y)
   store i8 %xy, ptr %p0
   store i8 %xy, ptr %p1
@@ -7932,9 +31705,9 @@ define void @commutative_clmul_v2i64(<2 x i64> %x, <2 x i64> %y, ptr %p0, ptr %p
 ; RV64IM-NEXT:    lui t0, 279620
 ; RV64IM-NEXT:    mul s0, t6, t5
 ; RV64IM-NEXT:    addi t1, t0, 1092
-; RV64IM-NEXT:    lui t0, %hi(.LCPI10_0)
+; RV64IM-NEXT:    lui t0, %hi(.LCPI15_0)
 ; RV64IM-NEXT:    slli s1, t1, 32
-; RV64IM-NEXT:    ld t0, %lo(.LCPI10_0)(t0)
+; RV64IM-NEXT:    ld t0, %lo(.LCPI15_0)(t0)
 ; RV64IM-NEXT:    add t1, t1, s1
 ; RV64IM-NEXT:    and s1, a0, t1
 ; RV64IM-NEXT:    and s2, a2, t1
@@ -8701,9 +32474,9 @@ define void @commutative_clmul_v2i64(<2 x i64> %x, <2 x i64> %y, ptr %p0, ptr %p
 ; RV64IMZBS-NEXT:    lui t0, 279620
 ; RV64IMZBS-NEXT:    mul s0, t6, t5
 ; RV64IMZBS-NEXT:    addi t1, t0, 1092
-; RV64IMZBS-NEXT:    lui t0, %hi(.LCPI10_0)
+; RV64IMZBS-NEXT:    lui t0, %hi(.LCPI15_0)
 ; RV64IMZBS-NEXT:    slli s1, t1, 32
-; RV64IMZBS-NEXT:    ld t0, %lo(.LCPI10_0)(t0)
+; RV64IMZBS-NEXT:    ld t0, %lo(.LCPI15_0)(t0)
 ; RV64IMZBS-NEXT:    add t1, t1, s1
 ; RV64IMZBS-NEXT:    and s1, a0, t1
 ; RV64IMZBS-NEXT:    and s2, a2, t1
@@ -8800,6 +32573,48 @@ define void @commutative_clmul_v2i64(<2 x i64> %x, <2 x i64> %y, ptr %p0, ptr %p
 ; RV64IMZBS-NEXT:    ld s8, 8(sp) # 8-byte Folded Reload
 ; RV64IMZBS-NEXT:    addi sp, sp, 80
 ; RV64IMZBS-NEXT:    ret
+;
+; RV32IMZBC-LABEL: commutative_clmul_v2i64:
+; RV32IMZBC:       # %bb.0:
+; RV32IMZBC-NEXT:    lw a4, 0(a0)
+; RV32IMZBC-NEXT:    lw a5, 4(a0)
+; RV32IMZBC-NEXT:    lw a6, 0(a1)
+; RV32IMZBC-NEXT:    lw a7, 4(a1)
+; RV32IMZBC-NEXT:    lw t0, 8(a1)
+; RV32IMZBC-NEXT:    lw a1, 12(a1)
+; RV32IMZBC-NEXT:    lw t1, 8(a0)
+; RV32IMZBC-NEXT:    lw a0, 12(a0)
+; RV32IMZBC-NEXT:    clmul a5, a5, a6
+; RV32IMZBC-NEXT:    clmul a7, a4, a7
+; RV32IMZBC-NEXT:    xor a5, a7, a5
+; RV32IMZBC-NEXT:    clmulh a7, a4, a6
+; RV32IMZBC-NEXT:    clmul a0, a0, t0
+; RV32IMZBC-NEXT:    clmul a1, t1, a1
+; RV32IMZBC-NEXT:    xor a5, a7, a5
+; RV32IMZBC-NEXT:    clmulh a7, t1, t0
+; RV32IMZBC-NEXT:    xor a0, a1, a0
+; RV32IMZBC-NEXT:    clmul a1, a4, a6
+; RV32IMZBC-NEXT:    xor a0, a7, a0
+; RV32IMZBC-NEXT:    clmul a4, t1, t0
+; RV32IMZBC-NEXT:    sw a1, 0(a2)
+; RV32IMZBC-NEXT:    sw a5, 4(a2)
+; RV32IMZBC-NEXT:    sw a4, 8(a2)
+; RV32IMZBC-NEXT:    sw a0, 12(a2)
+; RV32IMZBC-NEXT:    sw a1, 0(a3)
+; RV32IMZBC-NEXT:    sw a5, 4(a3)
+; RV32IMZBC-NEXT:    sw a4, 8(a3)
+; RV32IMZBC-NEXT:    sw a0, 12(a3)
+; RV32IMZBC-NEXT:    ret
+;
+; RV64IMZBC-LABEL: commutative_clmul_v2i64:
+; RV64IMZBC:       # %bb.0:
+; RV64IMZBC-NEXT:    clmul a0, a0, a2
+; RV64IMZBC-NEXT:    clmul a1, a1, a3
+; RV64IMZBC-NEXT:    sd a0, 0(a4)
+; RV64IMZBC-NEXT:    sd a1, 8(a4)
+; RV64IMZBC-NEXT:    sd a0, 0(a5)
+; RV64IMZBC-NEXT:    sd a1, 8(a5)
+; RV64IMZBC-NEXT:    ret
   %xy = call <2 x i64> @llvm.clmul.v2i64(<2 x i64> %x, <2 x i64> %y)
   %yx = call <2 x i64> @llvm.clmul.v2i64(<2 x i64> %y, <2 x i64> %x)
   store <2 x i64> %xy, ptr %p0
@@ -12186,9 +36001,9 @@ define void @mul_use_commutative_clmul_v2i64(<2 x i64> %x, <2 x i64> %y, ptr %p0
 ; RV64IM-NEXT:    lui a7, 279620
 ; RV64IM-NEXT:    mul t6, t5, t4
 ; RV64IM-NEXT:    addi t0, a7, 1092
-; RV64IM-NEXT:    lui a7, %hi(.LCPI11_0)
+; RV64IM-NEXT:    lui a7, %hi(.LCPI16_0)
 ; RV64IM-NEXT:    slli s1, t0, 32
-; RV64IM-NEXT:    ld a7, %lo(.LCPI11_0)(a7)
+; RV64IM-NEXT:    ld a7, %lo(.LCPI16_0)(a7)
 ; RV64IM-NEXT:    add t0, t0, s1
 ; RV64IM-NEXT:    and s1, a0, t0
 ; RV64IM-NEXT:    and s2, a2, t0
@@ -12965,9 +36780,9 @@ define void @mul_use_commutative_clmul_v2i64(<2 x i64> %x, <2 x i64> %y, ptr %p0
 ; RV64IMZBS-NEXT:    lui a7, 279620
 ; RV64IMZBS-NEXT:    mul t6, t5, t4
 ; RV64IMZBS-NEXT:    addi t0, a7, 1092
-; RV64IMZBS-NEXT:    lui a7, %hi(.LCPI11_0)
+; RV64IMZBS-NEXT:    lui a7, %hi(.LCPI16_0)
 ; RV64IMZBS-NEXT:    slli s1, t0, 32
-; RV64IMZBS-NEXT:    ld a7, %lo(.LCPI11_0)(a7)
+; RV64IMZBS-NEXT:    ld a7, %lo(.LCPI16_0)(a7)
 ; RV64IMZBS-NEXT:    add t0, t0, s1
 ; RV64IMZBS-NEXT:    and s1, a0, t0
 ; RV64IMZBS-NEXT:    and s2, a2, t0
@@ -13068,6 +36883,83 @@ define void @mul_use_commutative_clmul_v2i64(<2 x i64> %x, <2 x i64> %y, ptr %p0
 ; RV64IMZBS-NEXT:    ld s8, 0(sp) # 8-byte Folded Reload
 ; RV64IMZBS-NEXT:    addi sp, sp, 80
 ; RV64IMZBS-NEXT:    ret
+;
+; RV32IMZBC-LABEL: mul_use_commutative_clmul_v2i64:
+; RV32IMZBC:       # %bb.0:
+; RV32IMZBC-NEXT:    addi sp, sp, -48
+; RV32IMZBC-NEXT:    sw ra, 44(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s0, 40(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s1, 36(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s2, 32(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s3, 28(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    sw s4, 24(sp) # 4-byte Folded Spill
+; RV32IMZBC-NEXT:    mv s0, a3
+; RV32IMZBC-NEXT:    lw a3, 0(a0)
+; RV32IMZBC-NEXT:    lw a4, 4(a0)
+; RV32IMZBC-NEXT:    lw a5, 0(a1)
+; RV32IMZBC-NEXT:    lw a6, 4(a1)
+; RV32IMZBC-NEXT:    lw a7, 8(a1)
+; RV32IMZBC-NEXT:    lw a1, 12(a1)
+; RV32IMZBC-NEXT:    lw t0, 8(a0)
+; RV32IMZBC-NEXT:    lw a0, 12(a0)
+; RV32IMZBC-NEXT:    clmul a4, a4, a5
+; RV32IMZBC-NEXT:    clmul a6, a3, a6
+; RV32IMZBC-NEXT:    xor a4, a6, a4
+; RV32IMZBC-NEXT:    clmulh a6, a3, a5
+; RV32IMZBC-NEXT:    clmul a0, a0, a7
+; RV32IMZBC-NEXT:    clmul a1, t0, a1
+; RV32IMZBC-NEXT:    xor s1, a6, a4
+; RV32IMZBC-NEXT:    clmulh a4, t0, a7
+; RV32IMZBC-NEXT:    xor a0, a1, a0
+; RV32IMZBC-NEXT:    clmul s2, a3, a5
+; RV32IMZBC-NEXT:    xor s3, a4, a0
+; RV32IMZBC-NEXT:    clmul s4, t0, a7
+; RV32IMZBC-NEXT:    sw s2, 0(a2)
+; RV32IMZBC-NEXT:    sw s1, 4(a2)
+; RV32IMZBC-NEXT:    sw s4, 8(a2)
+; RV32IMZBC-NEXT:    sw s3, 12(a2)
+; RV32IMZBC-NEXT:    mv a0, sp
+; RV32IMZBC-NEXT:    sw s2, 0(sp)
+; RV32IMZBC-NEXT:    sw s1, 4(sp)
+; RV32IMZBC-NEXT:    sw s4, 8(sp)
+; RV32IMZBC-NEXT:    sw s3, 12(sp)
+; RV32IMZBC-NEXT:    call vector_use
+; RV32IMZBC-NEXT:    sw s2, 0(s0)
+; RV32IMZBC-NEXT:    sw s1, 4(s0)
+; RV32IMZBC-NEXT:    sw s4, 8(s0)
+; RV32IMZBC-NEXT:    sw s3, 12(s0)
+; RV32IMZBC-NEXT:    lw ra, 44(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s0, 40(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s1, 36(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s2, 32(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s3, 28(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    lw s4, 24(sp) # 4-byte Folded Reload
+; RV32IMZBC-NEXT:    addi sp, sp, 48
+; RV32IMZBC-NEXT:    ret
+;
+; RV64IMZBC-LABEL: mul_use_commutative_clmul_v2i64:
+; RV64IMZBC:       # %bb.0:
+; RV64IMZBC-NEXT:    addi sp, sp, -32
+; RV64IMZBC-NEXT:    sd ra, 24(sp) # 8-byte Folded Spill
+; RV64IMZBC-NEXT:    sd s0, 16(sp) # 8-byte Folded Spill
+; RV64IMZBC-NEXT:    sd s1, 8(sp) # 8-byte Folded Spill
+; RV64IMZBC-NEXT:    sd s2, 0(sp) # 8-byte Folded Spill
+; RV64IMZBC-NEXT:    mv s0, a5
+; RV64IMZBC-NEXT:    clmul s1, a0, a2
+; RV64IMZBC-NEXT:    clmul s2, a1, a3
+; RV64IMZBC-NEXT:    sd s1, 0(a4)
+; RV64IMZBC-NEXT:    sd s2, 8(a4)
+; RV64IMZBC-NEXT:    mv a0, s1
+; RV64IMZBC-NEXT:    mv a1, s2
+; RV64IMZBC-NEXT:    call vector_use
+; RV64IMZBC-NEXT:    sd s1, 0(s0)
+; RV64IMZBC-NEXT:    sd s2, 8(s0)
+; RV64IMZBC-NEXT:    ld ra, 24(sp) # 8-byte Folded Reload
+; RV64IMZBC-NEXT:    ld s0, 16(sp) # 8-byte Folded Reload
+; RV64IMZBC-NEXT:    ld s1, 8(sp) # 8-byte Folded Reload
+; RV64IMZBC-NEXT:    ld s2, 0(sp) # 8-byte Folded Reload
+; RV64IMZBC-NEXT:    addi sp, sp, 32
+; RV64IMZBC-NEXT:    ret
   %xy = call <2 x i64> @llvm.clmul.v2i64(<2 x i64> %x, <2 x i64> %y)
   %yx = call <2 x i64> @llvm.clmul.v2i64(<2 x i64> %y, <2 x i64> %x)
   store <2 x i64> %xy, ptr %p0

diff  --git a/llvm/test/CodeGen/X86/clmul.ll b/llvm/test/CodeGen/X86/clmul.ll
index 3394d7081a677..ad22eed4508b2 100644
--- a/llvm/test/CodeGen/X86/clmul.ll
+++ b/llvm/test/CodeGen/X86/clmul.ll
@@ -388,6 +388,1732 @@ define i64 @clmul_i64(i64 %a, i64 %b) nounwind {
   ret i64 %res
 }
 
+define i96 @clmul_i96(i96 %x, i96 %y) {
+; SCALAR-LABEL: clmul_i96:
+; SCALAR:       # %bb.0:
+; SCALAR-NEXT:    pushq %rbp
+; SCALAR-NEXT:    .cfi_def_cfa_offset 16
+; SCALAR-NEXT:    pushq %r15
+; SCALAR-NEXT:    .cfi_def_cfa_offset 24
+; SCALAR-NEXT:    pushq %r14
+; SCALAR-NEXT:    .cfi_def_cfa_offset 32
+; SCALAR-NEXT:    pushq %r13
+; SCALAR-NEXT:    .cfi_def_cfa_offset 40
+; SCALAR-NEXT:    pushq %r12
+; SCALAR-NEXT:    .cfi_def_cfa_offset 48
+; SCALAR-NEXT:    pushq %rbx
+; SCALAR-NEXT:    .cfi_def_cfa_offset 56
+; SCALAR-NEXT:    subq $544, %rsp # imm = 0x220
+; SCALAR-NEXT:    .cfi_def_cfa_offset 600
+; SCALAR-NEXT:    .cfi_offset %rbx, -56
+; SCALAR-NEXT:    .cfi_offset %r12, -48
+; SCALAR-NEXT:    .cfi_offset %r13, -40
+; SCALAR-NEXT:    .cfi_offset %r14, -32
+; SCALAR-NEXT:    .cfi_offset %r15, -24
+; SCALAR-NEXT:    .cfi_offset %rbp, -16
+; SCALAR-NEXT:    movq %rcx, %r11
+; SCALAR-NEXT:    movq %rdx, %r10
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $31, %rax
+; SCALAR-NEXT:    movq %rax, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    xorl %ecx, %ecx
+; SCALAR-NEXT:    testl %r11d, %r11d
+; SCALAR-NEXT:    cmovsq %rax, %rcx
+; SCALAR-NEXT:    movq %rcx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $1, %rdi, %rax
+; SCALAR-NEXT:    movq %rdx, %rcx
+; SCALAR-NEXT:    shlq $62, %rcx
+; SCALAR-NEXT:    sarq $63, %rcx
+; SCALAR-NEXT:    movq %rcx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rcx, %rax
+; SCALAR-NEXT:    movl %r10d, %ecx
+; SCALAR-NEXT:    andl $1, %ecx
+; SCALAR-NEXT:    negq %rcx
+; SCALAR-NEXT:    movq %rcx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rsi, %rcx
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $2, %rdi, %r8
+; SCALAR-NEXT:    movq %rdx, %rax
+; SCALAR-NEXT:    shlq $61, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    movq %rax, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $3, %rdi, %rax
+; SCALAR-NEXT:    shlq $60, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r8, %rax
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $4, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $59, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $5, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $58, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $6, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $57, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $7, %rdi, %rax
+; SCALAR-NEXT:    movsbq %r10b, %rdx
+; SCALAR-NEXT:    sarq $7, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $8, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $55, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %r9
+; SCALAR-NEXT:    shldq $9, %rdi, %r9
+; SCALAR-NEXT:    movq %r10, %rax
+; SCALAR-NEXT:    shlq $54, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    movq %rax, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rax, %r9
+; SCALAR-NEXT:    xorq %r8, %r9
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $10, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $53, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r9, %rax
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $11, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $52, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $12, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $51, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $13, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $50, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $14, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $49, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $15, %rdi, %rcx
+; SCALAR-NEXT:    movswq %r10w, %rdx
+; SCALAR-NEXT:    sarq $15, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $16, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $47, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $17, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $46, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $18, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $45, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r8, %rax
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $19, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $44, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %r9
+; SCALAR-NEXT:    shldq $20, %rdi, %r9
+; SCALAR-NEXT:    movq %r10, %rax
+; SCALAR-NEXT:    shlq $43, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    movq %rax, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rax, %r9
+; SCALAR-NEXT:    xorq %r8, %r9
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $21, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $42, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r9, %rax
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $22, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $41, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $23, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $40, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $24, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $39, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $25, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $38, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $26, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $37, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $27, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $36, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $28, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $35, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $29, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $34, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $30, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $33, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $31, %rdi, %rax
+; SCALAR-NEXT:    movslq %r10d, %rdx
+; SCALAR-NEXT:    sarq $31, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r8, %rax
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $32, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $31, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $33, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $30, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r8, %rax
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $34, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $29, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %r9
+; SCALAR-NEXT:    shldq $35, %rdi, %r9
+; SCALAR-NEXT:    movq %r10, %rax
+; SCALAR-NEXT:    shlq $28, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    movq %rax, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rax, %r9
+; SCALAR-NEXT:    xorq %r8, %r9
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $36, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $27, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r9, %rax
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $37, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $26, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $38, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $25, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $39, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $24, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $40, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $23, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $41, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $22, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $42, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $21, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $43, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $20, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $44, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $19, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $45, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $18, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $46, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $17, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $47, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $16, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $48, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $15, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r8, %rax
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $49, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $14, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $50, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $13, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r8, %rax
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $51, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $12, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $52, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $11, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r8, %rax
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $53, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $10, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rsi, %r9
+; SCALAR-NEXT:    shldq $54, %rdi, %r9
+; SCALAR-NEXT:    movq %r10, %rax
+; SCALAR-NEXT:    shlq $9, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    movq %rax, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rax, %r9
+; SCALAR-NEXT:    xorq %r8, %r9
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    shldq $55, %rdi, %rax
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $8, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r9, %rax
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $56, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $7, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $57, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $6, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $58, %rdi, %rcx
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $5, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $59, %rdi, %r8
+; SCALAR-NEXT:    movq %r10, %rdx
+; SCALAR-NEXT:    shlq $4, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $60, %rdi, %rcx
+; SCALAR-NEXT:    leaq (,%r10,8), %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    movq %rsi, %r8
+; SCALAR-NEXT:    shldq $61, %rdi, %r8
+; SCALAR-NEXT:    leaq (,%r10,4), %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    xorq %rcx, %r8
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    shldq $62, %rdi, %rcx
+; SCALAR-NEXT:    leaq (%r10,%r10), %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    shldq $63, %rdi, %rsi
+; SCALAR-NEXT:    sarq $63, %r10
+; SCALAR-NEXT:    andq %r10, %rsi
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    movl %r11d, %esi
+; SCALAR-NEXT:    andl $1, %esi
+; SCALAR-NEXT:    negq %rsi
+; SCALAR-NEXT:    andq %rdi, %rsi
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    movq %r11, %r8
+; SCALAR-NEXT:    shlq $62, %r8
+; SCALAR-NEXT:    sarq $63, %r8
+; SCALAR-NEXT:    leaq (%rdi,%rdi), %rcx
+; SCALAR-NEXT:    andq %rcx, %r8
+; SCALAR-NEXT:    xorq %rsi, %r8
+; SCALAR-NEXT:    movq %r11, %rcx
+; SCALAR-NEXT:    shlq $61, %rcx
+; SCALAR-NEXT:    sarq $63, %rcx
+; SCALAR-NEXT:    leaq (,%rdi,4), %rdx
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %r11, %rax
+; SCALAR-NEXT:    shlq $60, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    leaq (,%rdi,8), %rdx
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $4, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    shlq $59, %rsi
+; SCALAR-NEXT:    sarq $63, %rsi
+; SCALAR-NEXT:    andq %rdx, %rsi
+; SCALAR-NEXT:    xorq %rax, %rsi
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $5, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rax
+; SCALAR-NEXT:    shlq $58, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %rsi, %rax
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $6, %rdx
+; SCALAR-NEXT:    movq %rdx, (%rsp) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    shlq $57, %rsi
+; SCALAR-NEXT:    sarq $63, %rsi
+; SCALAR-NEXT:    andq %rdx, %rsi
+; SCALAR-NEXT:    xorq %rax, %rsi
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $7, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movsbq %r11b, %rax
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %rsi, %rax
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $8, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    shlq $55, %rsi
+; SCALAR-NEXT:    sarq $63, %rsi
+; SCALAR-NEXT:    andq %rdx, %rsi
+; SCALAR-NEXT:    xorq %rax, %rsi
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $9, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rax
+; SCALAR-NEXT:    shlq $54, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %rsi, %rax
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $10, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    shlq $53, %rsi
+; SCALAR-NEXT:    sarq $63, %rsi
+; SCALAR-NEXT:    andq %rdx, %rsi
+; SCALAR-NEXT:    xorq %rax, %rsi
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $11, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rax
+; SCALAR-NEXT:    shlq $52, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %rsi, %rax
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $12, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    shlq $51, %rsi
+; SCALAR-NEXT:    sarq $63, %rsi
+; SCALAR-NEXT:    andq %rdx, %rsi
+; SCALAR-NEXT:    xorq %rax, %rsi
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $13, %rax
+; SCALAR-NEXT:    movq %rax, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %r8
+; SCALAR-NEXT:    shlq $50, %r8
+; SCALAR-NEXT:    sarq $63, %r8
+; SCALAR-NEXT:    andq %rax, %r8
+; SCALAR-NEXT:    xorq %rsi, %r8
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $14, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rax
+; SCALAR-NEXT:    shlq $49, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    andq %rdx, %rax
+; SCALAR-NEXT:    xorq %r8, %rax
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    shlq $15, %rdx
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movswq %r11w, %rcx
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    movq %rdi, %r8
+; SCALAR-NEXT:    shlq $16, %r8
+; SCALAR-NEXT:    movq %r8, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    movq %r11, %rdx
+; SCALAR-NEXT:    shlq $47, %rsi
+; SCALAR-NEXT:    sarq $63, %rsi
+; SCALAR-NEXT:    andq %r8, %rsi
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    movq %rdi, %r8
+; SCALAR-NEXT:    shlq $17, %r8
+; SCALAR-NEXT:    movq %r8, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rcx
+; SCALAR-NEXT:    shlq $46, %rcx
+; SCALAR-NEXT:    sarq $63, %rcx
+; SCALAR-NEXT:    andq %r8, %rcx
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    movq %rdi, %r8
+; SCALAR-NEXT:    shlq $18, %r8
+; SCALAR-NEXT:    movq %r8, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    shlq $45, %rsi
+; SCALAR-NEXT:    sarq $63, %rsi
+; SCALAR-NEXT:    andq %r8, %rsi
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    movq %rdi, %r8
+; SCALAR-NEXT:    shlq $19, %r8
+; SCALAR-NEXT:    movq %r8, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rcx
+; SCALAR-NEXT:    shlq $44, %rcx
+; SCALAR-NEXT:    sarq $63, %rcx
+; SCALAR-NEXT:    andq %r8, %rcx
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    movq %rdi, %r8
+; SCALAR-NEXT:    shlq $20, %r8
+; SCALAR-NEXT:    movq %r8, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    shlq $43, %rsi
+; SCALAR-NEXT:    sarq $63, %rsi
+; SCALAR-NEXT:    andq %r8, %rsi
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    movq %rdi, %r15
+; SCALAR-NEXT:    shlq $21, %r15
+; SCALAR-NEXT:    movq %r11, %rcx
+; SCALAR-NEXT:    shlq $42, %rcx
+; SCALAR-NEXT:    sarq $63, %rcx
+; SCALAR-NEXT:    andq %r15, %rcx
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    movq %rdi, %r8
+; SCALAR-NEXT:    shlq $22, %r8
+; SCALAR-NEXT:    movq %r8, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    shlq $41, %rsi
+; SCALAR-NEXT:    sarq $63, %rsi
+; SCALAR-NEXT:    andq %r8, %rsi
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    movq %rdi, %r12
+; SCALAR-NEXT:    shlq $23, %r12
+; SCALAR-NEXT:    movq %r11, %rcx
+; SCALAR-NEXT:    shlq $40, %rcx
+; SCALAR-NEXT:    sarq $63, %rcx
+; SCALAR-NEXT:    andq %r12, %rcx
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    movq %rdi, %rbx
+; SCALAR-NEXT:    shlq $24, %rbx
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    shlq $39, %rsi
+; SCALAR-NEXT:    sarq $63, %rsi
+; SCALAR-NEXT:    andq %rbx, %rsi
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    movq %rdi, %r11
+; SCALAR-NEXT:    shlq $25, %r11
+; SCALAR-NEXT:    movq %rdx, %rcx
+; SCALAR-NEXT:    shlq $38, %rcx
+; SCALAR-NEXT:    sarq $63, %rcx
+; SCALAR-NEXT:    andq %r11, %rcx
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    movq %rdi, %r8
+; SCALAR-NEXT:    shlq $26, %r8
+; SCALAR-NEXT:    movq %rdx, %r14
+; SCALAR-NEXT:    shlq $37, %r14
+; SCALAR-NEXT:    sarq $63, %r14
+; SCALAR-NEXT:    andq %r8, %r14
+; SCALAR-NEXT:    xorq %rcx, %r14
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    shlq $27, %rsi
+; SCALAR-NEXT:    movq %rdx, %rbp
+; SCALAR-NEXT:    shlq $36, %rbp
+; SCALAR-NEXT:    sarq $63, %rbp
+; SCALAR-NEXT:    andq %rsi, %rbp
+; SCALAR-NEXT:    xorq %r14, %rbp
+; SCALAR-NEXT:    xorq %rax, %rbp
+; SCALAR-NEXT:    movq %rdi, %r9
+; SCALAR-NEXT:    shlq $28, %r9
+; SCALAR-NEXT:    movq %rdx, %rax
+; SCALAR-NEXT:    shlq $35, %rax
+; SCALAR-NEXT:    sarq $63, %rax
+; SCALAR-NEXT:    andq %r9, %rax
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $29, %rcx
+; SCALAR-NEXT:    movq %rcx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %rdx, %r14
+; SCALAR-NEXT:    shlq $34, %r14
+; SCALAR-NEXT:    sarq $63, %r14
+; SCALAR-NEXT:    andq %rcx, %r14
+; SCALAR-NEXT:    xorq %rax, %r14
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $30, %rcx
+; SCALAR-NEXT:    shlq $33, %rdx
+; SCALAR-NEXT:    sarq $63, %rdx
+; SCALAR-NEXT:    andq %rcx, %rdx
+; SCALAR-NEXT:    xorq %rdx, %r14
+; SCALAR-NEXT:    xorq {{[-0-9]+}}(%r{{[sb]}}p), %r14 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rbp, %r14
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rbp # 8-byte Reload
+; SCALAR-NEXT:    leaq (%rdi,%rdi), %rax
+; SCALAR-NEXT:    andq %rax, %rbp
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Reload
+; SCALAR-NEXT:    andq %rdi, %rdx
+; SCALAR-NEXT:    xorq %rbp, %rdx
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; SCALAR-NEXT:    leaq (,%rdi,4), %rbp
+; SCALAR-NEXT:    andq %rbp, %rax
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rbp # 8-byte Reload
+; SCALAR-NEXT:    leaq (,%rdi,8), %r13
+; SCALAR-NEXT:    andq %r13, %rbp
+; SCALAR-NEXT:    xorq %rax, %rbp
+; SCALAR-NEXT:    xorq %rdx, %rbp
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Folded Reload
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rdx, %rax
+; SCALAR-NEXT:    movq %rax, %rdx
+; SCALAR-NEXT:    movq (%rsp), %rax # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rdx, %rax
+; SCALAR-NEXT:    xorq %rbp, %rax
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Folded Reload
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r13, %rdx
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rdx, %r13
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r13, %rdx
+; SCALAR-NEXT:    xorq %rax, %rdx
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Folded Reload
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r13, %rax
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %r13
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r13, %rax
+; SCALAR-NEXT:    movq %rax, %r13
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r13, %rax
+; SCALAR-NEXT:    xorq %rdx, %rax
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Folded Reload
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r13, %rdx
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rdx, %r13
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rdx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r13, %rdx
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r13 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rdx, %r13
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r15 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r13, %r15
+; SCALAR-NEXT:    xorq %rax, %r15
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r12 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %r12
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rbx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r12, %rbx
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r11 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rbx, %r11
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r8 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r11, %r8
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rsi # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r8, %rsi
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r9 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rsi, %r9
+; SCALAR-NEXT:    xorq %r15, %r9
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rcx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $32, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    shlq $33, %rsi
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rsi # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %rsi
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $34, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rsi, %rax
+; SCALAR-NEXT:    movq %rdi, %r8
+; SCALAR-NEXT:    shlq $35, %r8
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r8 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %r8
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    shlq $36, %rsi
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rsi # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r8, %rsi
+; SCALAR-NEXT:    xorq %r9, %rsi
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $37, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $38, %rcx
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rcx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $39, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $40, %rcx
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rcx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $41, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $42, %rcx
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rcx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $43, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $44, %rcx
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rcx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $45, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    xorq %rsi, %rax
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $46, %rcx
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rcx # 8-byte Folded Reload
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    shlq $47, %rsi
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rsi # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $48, %rcx
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rcx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    shlq $49, %rsi
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rsi # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $50, %rcx
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rcx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    shlq $51, %rsi
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rsi # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $52, %rcx
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rcx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    shlq $53, %rsi
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rsi # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    movq %rdi, %r8
+; SCALAR-NEXT:    shlq $54, %r8
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %r8 # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rsi, %r8
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    shlq $55, %rcx
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rcx # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $56, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    shlq $57, %rsi
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rsi # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %rsi
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $58, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rsi, %rax
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    shlq $59, %rsi
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rsi # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %rsi
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $60, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rsi, %rax
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    shlq $61, %rsi
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rsi # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rax, %rsi
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    shlq $62, %rax
+; SCALAR-NEXT:    andq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Folded Reload
+; SCALAR-NEXT:    xorq %rsi, %rax
+; SCALAR-NEXT:    shlq $63, %rdi
+; SCALAR-NEXT:    andq %r10, %rdi
+; SCALAR-NEXT:    xorq %rdi, %rax
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %r14, %rdx
+; SCALAR-NEXT:    addq $544, %rsp # imm = 0x220
+; SCALAR-NEXT:    .cfi_def_cfa_offset 56
+; SCALAR-NEXT:    popq %rbx
+; SCALAR-NEXT:    .cfi_def_cfa_offset 48
+; SCALAR-NEXT:    popq %r12
+; SCALAR-NEXT:    .cfi_def_cfa_offset 40
+; SCALAR-NEXT:    popq %r13
+; SCALAR-NEXT:    .cfi_def_cfa_offset 32
+; SCALAR-NEXT:    popq %r14
+; SCALAR-NEXT:    .cfi_def_cfa_offset 24
+; SCALAR-NEXT:    popq %r15
+; SCALAR-NEXT:    .cfi_def_cfa_offset 16
+; SCALAR-NEXT:    popq %rbp
+; SCALAR-NEXT:    .cfi_def_cfa_offset 8
+; SCALAR-NEXT:    retq
+;
+; SSE2-PCLMUL-LABEL: clmul_i96:
+; SSE2-PCLMUL:       # %bb.0:
+; SSE2-PCLMUL-NEXT:    movq %rdx, %xmm0
+; SSE2-PCLMUL-NEXT:    movq %rsi, %xmm1
+; SSE2-PCLMUL-NEXT:    pclmulqdq $0, %xmm0, %xmm1
+; SSE2-PCLMUL-NEXT:    movq %xmm1, %rax
+; SSE2-PCLMUL-NEXT:    movq %rcx, %xmm1
+; SSE2-PCLMUL-NEXT:    movq %rdi, %xmm2
+; SSE2-PCLMUL-NEXT:    pclmulqdq $0, %xmm2, %xmm1
+; SSE2-PCLMUL-NEXT:    movq %xmm1, %rcx
+; SSE2-PCLMUL-NEXT:    xorq %rax, %rcx
+; SSE2-PCLMUL-NEXT:    pclmulqdq $0, %xmm0, %xmm2
+; SSE2-PCLMUL-NEXT:    pshufd {{.*#+}} xmm0 = xmm2[2,3,2,3]
+; SSE2-PCLMUL-NEXT:    movq %xmm0, %rdx
+; SSE2-PCLMUL-NEXT:    xorq %rcx, %rdx
+; SSE2-PCLMUL-NEXT:    movq %xmm2, %rax
+; SSE2-PCLMUL-NEXT:    retq
+;
+; SSE42-PCLMUL-LABEL: clmul_i96:
+; SSE42-PCLMUL:       # %bb.0:
+; SSE42-PCLMUL-NEXT:    movq %rdx, %xmm0
+; SSE42-PCLMUL-NEXT:    movq %rsi, %xmm1
+; SSE42-PCLMUL-NEXT:    pclmulqdq $0, %xmm0, %xmm1
+; SSE42-PCLMUL-NEXT:    movq %xmm1, %rax
+; SSE42-PCLMUL-NEXT:    movq %rcx, %xmm1
+; SSE42-PCLMUL-NEXT:    movq %rdi, %xmm2
+; SSE42-PCLMUL-NEXT:    pclmulqdq $0, %xmm2, %xmm1
+; SSE42-PCLMUL-NEXT:    movq %xmm1, %rcx
+; SSE42-PCLMUL-NEXT:    xorq %rax, %rcx
+; SSE42-PCLMUL-NEXT:    pclmulqdq $0, %xmm0, %xmm2
+; SSE42-PCLMUL-NEXT:    pextrq $1, %xmm2, %rdx
+; SSE42-PCLMUL-NEXT:    xorq %rcx, %rdx
+; SSE42-PCLMUL-NEXT:    movq %xmm2, %rax
+; SSE42-PCLMUL-NEXT:    retq
+;
+; AVX-LABEL: clmul_i96:
+; AVX:       # %bb.0:
+; AVX-NEXT:    vmovq %rdx, %xmm0
+; AVX-NEXT:    vmovq %rsi, %xmm1
+; AVX-NEXT:    vpclmulqdq $0, %xmm0, %xmm1, %xmm1
+; AVX-NEXT:    vmovq %xmm1, %rax
+; AVX-NEXT:    vmovq %rcx, %xmm1
+; AVX-NEXT:    vmovq %rdi, %xmm2
+; AVX-NEXT:    vpclmulqdq $0, %xmm1, %xmm2, %xmm1
+; AVX-NEXT:    vmovq %xmm1, %rcx
+; AVX-NEXT:    xorq %rax, %rcx
+; AVX-NEXT:    vpclmulqdq $0, %xmm0, %xmm2, %xmm0
+; AVX-NEXT:    vpextrq $1, %xmm0, %rdx
+; AVX-NEXT:    xorq %rcx, %rdx
+; AVX-NEXT:    vmovq %xmm0, %rax
+; AVX-NEXT:    retq
+  %a = call i96 @llvm.clmul.i96(i96 %x, i96 %y)
+  ret i96 %a
+}
+
+define i128 @clmul_i128(i128 %x, i128 %y) {
+; SCALAR-LABEL: clmul_i128:
+; SCALAR:       # %bb.0:
+; SCALAR-NEXT:    pushq %rbp
+; SCALAR-NEXT:    .cfi_def_cfa_offset 16
+; SCALAR-NEXT:    pushq %r15
+; SCALAR-NEXT:    .cfi_def_cfa_offset 24
+; SCALAR-NEXT:    pushq %r14
+; SCALAR-NEXT:    .cfi_def_cfa_offset 32
+; SCALAR-NEXT:    pushq %r13
+; SCALAR-NEXT:    .cfi_def_cfa_offset 40
+; SCALAR-NEXT:    pushq %r12
+; SCALAR-NEXT:    .cfi_def_cfa_offset 48
+; SCALAR-NEXT:    pushq %rbx
+; SCALAR-NEXT:    .cfi_def_cfa_offset 56
+; SCALAR-NEXT:    .cfi_offset %rbx, -56
+; SCALAR-NEXT:    .cfi_offset %r12, -48
+; SCALAR-NEXT:    .cfi_offset %r13, -40
+; SCALAR-NEXT:    .cfi_offset %r14, -32
+; SCALAR-NEXT:    .cfi_offset %r15, -24
+; SCALAR-NEXT:    .cfi_offset %rbp, -16
+; SCALAR-NEXT:    movq %rcx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %rdx, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %rsi, %rbx
+; SCALAR-NEXT:    movq %rdi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %rdx, %rax
+; SCALAR-NEXT:    bswapq %rax
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    shrq $4, %rcx
+; SCALAR-NEXT:    movabsq $1085102592571150095, %r8 # imm = 0xF0F0F0F0F0F0F0F
+; SCALAR-NEXT:    andq %r8, %rcx
+; SCALAR-NEXT:    andq %r8, %rax
+; SCALAR-NEXT:    shlq $4, %rax
+; SCALAR-NEXT:    orq %rcx, %rax
+; SCALAR-NEXT:    movabsq $3689348814741910323, %rsi # imm = 0x3333333333333333
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    andq %rsi, %rcx
+; SCALAR-NEXT:    shrq $2, %rax
+; SCALAR-NEXT:    andq %rsi, %rax
+; SCALAR-NEXT:    leaq (%rax,%rcx,4), %rax
+; SCALAR-NEXT:    movabsq $6148914691236517205, %r11 # imm = 0x5555555555555555
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    andq %r11, %rcx
+; SCALAR-NEXT:    shrq %rax
+; SCALAR-NEXT:    andq %r11, %rax
+; SCALAR-NEXT:    leaq (%rax,%rcx,2), %r12
+; SCALAR-NEXT:    movabsq $1229782938247303441, %r14 # imm = 0x1111111111111111
+; SCALAR-NEXT:    movq %r12, %r10
+; SCALAR-NEXT:    andq %r14, %r10
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    bswapq %rax
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    shrq $4, %rcx
+; SCALAR-NEXT:    andq %r8, %rcx
+; SCALAR-NEXT:    andq %r8, %rax
+; SCALAR-NEXT:    shlq $4, %rax
+; SCALAR-NEXT:    orq %rcx, %rax
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    andq %rsi, %rcx
+; SCALAR-NEXT:    shrq $2, %rax
+; SCALAR-NEXT:    andq %rsi, %rax
+; SCALAR-NEXT:    leaq (%rax,%rcx,4), %rax
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    andq %r11, %rcx
+; SCALAR-NEXT:    shrq %rax
+; SCALAR-NEXT:    andq %r11, %rax
+; SCALAR-NEXT:    leaq (%rax,%rcx,2), %r15
+; SCALAR-NEXT:    movabsq $2459565876494606882, %rcx # imm = 0x2222222222222222
+; SCALAR-NEXT:    movq %r15, %r9
+; SCALAR-NEXT:    andq %rcx, %r9
+; SCALAR-NEXT:    movq %r9, %rax
+; SCALAR-NEXT:    imulq %r10, %rax
+; SCALAR-NEXT:    movq %r12, %r13
+; SCALAR-NEXT:    andq %rcx, %r13
+; SCALAR-NEXT:    movq %r15, %r8
+; SCALAR-NEXT:    andq %r14, %r8
+; SCALAR-NEXT:    movq %r8, %rcx
+; SCALAR-NEXT:    imulq %r13, %rcx
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movabsq $-8608480567731124088, %rax # imm = 0x8888888888888888
+; SCALAR-NEXT:    movq %r12, %rbp
+; SCALAR-NEXT:    andq %rax, %rbp
+; SCALAR-NEXT:    movabsq $4919131752989213764, %rsi # imm = 0x4444444444444444
+; SCALAR-NEXT:    movq %r15, %rdi
+; SCALAR-NEXT:    andq %rsi, %rdi
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    imulq %rbp, %rdx
+; SCALAR-NEXT:    andq %rsi, %r12
+; SCALAR-NEXT:    andq %rax, %r15
+; SCALAR-NEXT:    movq %r15, %rax
+; SCALAR-NEXT:    imulq %r12, %rax
+; SCALAR-NEXT:    xorq %rdx, %rax
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %r9, %rcx
+; SCALAR-NEXT:    imulq %rbp, %rcx
+; SCALAR-NEXT:    movq %r8, %rdx
+; SCALAR-NEXT:    imulq %r10, %rdx
+; SCALAR-NEXT:    xorq %rcx, %rdx
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    imulq %r12, %rsi
+; SCALAR-NEXT:    movq %r15, %rcx
+; SCALAR-NEXT:    imulq %r13, %rcx
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    xorq %rdx, %rcx
+; SCALAR-NEXT:    movabsq $2459565876494606882, %rsi # imm = 0x2222222222222222
+; SCALAR-NEXT:    andq %rsi, %rax
+; SCALAR-NEXT:    andq %r14, %rcx
+; SCALAR-NEXT:    orq %rax, %rcx
+; SCALAR-NEXT:    movq %r15, %rax
+; SCALAR-NEXT:    imulq %rbp, %rax
+; SCALAR-NEXT:    imulq %r8, %rbp
+; SCALAR-NEXT:    imulq %r12, %r8
+; SCALAR-NEXT:    imulq %r9, %r12
+; SCALAR-NEXT:    imulq %r13, %r9
+; SCALAR-NEXT:    xorq %r9, %r8
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    imulq %r10, %rdx
+; SCALAR-NEXT:    xorq %rdx, %rax
+; SCALAR-NEXT:    xorq %r8, %rax
+; SCALAR-NEXT:    movabsq $4919131752989213764, %r9 # imm = 0x4444444444444444
+; SCALAR-NEXT:    andq %r9, %rax
+; SCALAR-NEXT:    orq %rcx, %rax
+; SCALAR-NEXT:    xorq %r12, %rbp
+; SCALAR-NEXT:    imulq %r13, %rdi
+; SCALAR-NEXT:    imulq %r10, %r15
+; SCALAR-NEXT:    xorq %rdi, %r15
+; SCALAR-NEXT:    xorq %rbp, %r15
+; SCALAR-NEXT:    movabsq $-8608480567731124088, %r13 # imm = 0x8888888888888888
+; SCALAR-NEXT:    andq %r13, %r15
+; SCALAR-NEXT:    orq %rax, %r15
+; SCALAR-NEXT:    bswapq %r15
+; SCALAR-NEXT:    movq %r15, %rax
+; SCALAR-NEXT:    shrq $4, %rax
+; SCALAR-NEXT:    movabsq $1085102592571150095, %rcx # imm = 0xF0F0F0F0F0F0F0F
+; SCALAR-NEXT:    andq %rcx, %rax
+; SCALAR-NEXT:    andq %rcx, %r15
+; SCALAR-NEXT:    shlq $4, %r15
+; SCALAR-NEXT:    orq %rax, %r15
+; SCALAR-NEXT:    movq %r15, %rax
+; SCALAR-NEXT:    movabsq $3689348814741910323, %rcx # imm = 0x3333333333333333
+; SCALAR-NEXT:    andq %rcx, %rax
+; SCALAR-NEXT:    shrq $2, %r15
+; SCALAR-NEXT:    andq %rcx, %r15
+; SCALAR-NEXT:    leaq (%r15,%rax,4), %rax
+; SCALAR-NEXT:    andq %rax, %r11
+; SCALAR-NEXT:    shrq %rax
+; SCALAR-NEXT:    movabsq $6148914691236517204, %rcx # imm = 0x5555555555555554
+; SCALAR-NEXT:    andq %rax, %rcx
+; SCALAR-NEXT:    leaq (%rcx,%r11,2), %rax
+; SCALAR-NEXT:    movq %rax, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r10 # 8-byte Reload
+; SCALAR-NEXT:    movq %r10, %r14
+; SCALAR-NEXT:    movabsq $1229782938247303441, %r11 # imm = 0x1111111111111111
+; SCALAR-NEXT:    andq %r11, %r14
+; SCALAR-NEXT:    movq %rbx, %rdx
+; SCALAR-NEXT:    movq %rsi, %rbp
+; SCALAR-NEXT:    andq %rsi, %rdx
+; SCALAR-NEXT:    movq %rdx, %rsi
+; SCALAR-NEXT:    imulq %r14, %rsi
+; SCALAR-NEXT:    movq %r10, %r8
+; SCALAR-NEXT:    andq %rbp, %r8
+; SCALAR-NEXT:    movq %rbx, %rcx
+; SCALAR-NEXT:    andq %r11, %rcx
+; SCALAR-NEXT:    movq %rcx, %rdi
+; SCALAR-NEXT:    imulq %r8, %rdi
+; SCALAR-NEXT:    movq %r8, %rax
+; SCALAR-NEXT:    xorq %rsi, %rdi
+; SCALAR-NEXT:    movq %r10, %r15
+; SCALAR-NEXT:    andq %r13, %r15
+; SCALAR-NEXT:    movq %rbx, %r12
+; SCALAR-NEXT:    andq %r9, %r12
+; SCALAR-NEXT:    movq %r12, %rsi
+; SCALAR-NEXT:    imulq %r15, %rsi
+; SCALAR-NEXT:    andq %r9, %r10
+; SCALAR-NEXT:    andq %r13, %rbx
+; SCALAR-NEXT:    movq %rbx, %r9
+; SCALAR-NEXT:    imulq %r10, %r9
+; SCALAR-NEXT:    movq %r10, %r8
+; SCALAR-NEXT:    xorq %rsi, %r9
+; SCALAR-NEXT:    xorq %rdi, %r9
+; SCALAR-NEXT:    movq %rdx, %rsi
+; SCALAR-NEXT:    imulq %r15, %rsi
+; SCALAR-NEXT:    movq %rcx, %rdi
+; SCALAR-NEXT:    imulq %r14, %rdi
+; SCALAR-NEXT:    xorq %rsi, %rdi
+; SCALAR-NEXT:    movq %r12, %rsi
+; SCALAR-NEXT:    imulq %r10, %rsi
+; SCALAR-NEXT:    movq %rbx, %r10
+; SCALAR-NEXT:    imulq %rax, %r10
+; SCALAR-NEXT:    xorq %rsi, %r10
+; SCALAR-NEXT:    xorq %rdi, %r10
+; SCALAR-NEXT:    andq %rbp, %r9
+; SCALAR-NEXT:    andq %r11, %r10
+; SCALAR-NEXT:    orq %r9, %r10
+; SCALAR-NEXT:    movq %rdx, %rsi
+; SCALAR-NEXT:    imulq %rax, %rsi
+; SCALAR-NEXT:    movq %rax, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %rcx, %rdi
+; SCALAR-NEXT:    imulq %r8, %rdi
+; SCALAR-NEXT:    movq %r8, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    xorq %rsi, %rdi
+; SCALAR-NEXT:    movq %r12, %rsi
+; SCALAR-NEXT:    imulq %r14, %rsi
+; SCALAR-NEXT:    movq %rbx, %r9
+; SCALAR-NEXT:    imulq %r15, %r9
+; SCALAR-NEXT:    xorq %rsi, %r9
+; SCALAR-NEXT:    xorq %rdi, %r9
+; SCALAR-NEXT:    movabsq $4919131752989213764, %rsi # imm = 0x4444444444444444
+; SCALAR-NEXT:    andq %rsi, %r9
+; SCALAR-NEXT:    orq %r10, %r9
+; SCALAR-NEXT:    imulq %r8, %rdx
+; SCALAR-NEXT:    imulq %r15, %rcx
+; SCALAR-NEXT:    xorq %rdx, %rcx
+; SCALAR-NEXT:    imulq %rax, %r12
+; SCALAR-NEXT:    imulq %r14, %rbx
+; SCALAR-NEXT:    xorq %rbx, %r12
+; SCALAR-NEXT:    xorq %rcx, %r12
+; SCALAR-NEXT:    movq %r13, %rdi
+; SCALAR-NEXT:    andq %r13, %r12
+; SCALAR-NEXT:    orq %r9, %r12
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rbx # 8-byte Reload
+; SCALAR-NEXT:    movq %rbx, %rsi
+; SCALAR-NEXT:    andq %r11, %rsi
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r8 # 8-byte Reload
+; SCALAR-NEXT:    movq %r8, %r13
+; SCALAR-NEXT:    andq %rbp, %r13
+; SCALAR-NEXT:    movq %r13, %rdx
+; SCALAR-NEXT:    imulq %rsi, %rdx
+; SCALAR-NEXT:    movq %rbx, %rcx
+; SCALAR-NEXT:    andq %rbp, %rcx
+; SCALAR-NEXT:    movq %r8, %rbp
+; SCALAR-NEXT:    andq %r11, %rbp
+; SCALAR-NEXT:    movq %rbp, %r9
+; SCALAR-NEXT:    imulq %rcx, %r9
+; SCALAR-NEXT:    xorq %rdx, %r9
+; SCALAR-NEXT:    movq %rbx, %rdx
+; SCALAR-NEXT:    andq %rdi, %rdx
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    movq %r8, %rdi
+; SCALAR-NEXT:    movabsq $4919131752989213764, %r11 # imm = 0x4444444444444444
+; SCALAR-NEXT:    andq %r11, %rdi
+; SCALAR-NEXT:    movq %rdi, %r10
+; SCALAR-NEXT:    imulq %rdx, %r10
+; SCALAR-NEXT:    andq %r11, %rbx
+; SCALAR-NEXT:    andq %rax, %r8
+; SCALAR-NEXT:    movq %r8, %r11
+; SCALAR-NEXT:    imulq %rbx, %r11
+; SCALAR-NEXT:    movq %rbx, %rax
+; SCALAR-NEXT:    xorq %r10, %r11
+; SCALAR-NEXT:    xorq %r9, %r11
+; SCALAR-NEXT:    movq %r13, %r9
+; SCALAR-NEXT:    imulq %rdx, %r9
+; SCALAR-NEXT:    movq %rbp, %r10
+; SCALAR-NEXT:    imulq %rsi, %r10
+; SCALAR-NEXT:    xorq %r9, %r10
+; SCALAR-NEXT:    movq %rdi, %r9
+; SCALAR-NEXT:    imulq %rbx, %r9
+; SCALAR-NEXT:    movq %r8, %rbx
+; SCALAR-NEXT:    imulq %rcx, %rbx
+; SCALAR-NEXT:    xorq %r9, %rbx
+; SCALAR-NEXT:    xorq %r10, %rbx
+; SCALAR-NEXT:    movabsq $2459565876494606882, %r9 # imm = 0x2222222222222222
+; SCALAR-NEXT:    andq %r9, %r11
+; SCALAR-NEXT:    movabsq $1229782938247303441, %r9 # imm = 0x1111111111111111
+; SCALAR-NEXT:    andq %r9, %rbx
+; SCALAR-NEXT:    orq %r11, %rbx
+; SCALAR-NEXT:    movq %r13, %r9
+; SCALAR-NEXT:    imulq %rcx, %r9
+; SCALAR-NEXT:    movq %rbp, %r10
+; SCALAR-NEXT:    imulq %rax, %r10
+; SCALAR-NEXT:    xorq %r9, %r10
+; SCALAR-NEXT:    movq %rdi, %r9
+; SCALAR-NEXT:    imulq %rsi, %r9
+; SCALAR-NEXT:    movq %r8, %r11
+; SCALAR-NEXT:    imulq %rdx, %r11
+; SCALAR-NEXT:    xorq %r9, %r11
+; SCALAR-NEXT:    xorq %r10, %r11
+; SCALAR-NEXT:    movabsq $4919131752989213764, %r9 # imm = 0x4444444444444444
+; SCALAR-NEXT:    andq %r9, %r11
+; SCALAR-NEXT:    orq %rbx, %r11
+; SCALAR-NEXT:    imulq %r13, %rax
+; SCALAR-NEXT:    imulq %rbp, %rdx
+; SCALAR-NEXT:    xorq %rax, %rdx
+; SCALAR-NEXT:    imulq %r8, %rsi
+; SCALAR-NEXT:    imulq %rdi, %rcx
+; SCALAR-NEXT:    xorq %rcx, %rsi
+; SCALAR-NEXT:    xorq %rdx, %rsi
+; SCALAR-NEXT:    movabsq $-8608480567731124088, %rbx # imm = 0x8888888888888888
+; SCALAR-NEXT:    andq %rbx, %rsi
+; SCALAR-NEXT:    orq %r11, %rsi
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; SCALAR-NEXT:    shrq %rax
+; SCALAR-NEXT:    xorq %r12, %rsi
+; SCALAR-NEXT:    xorq %rax, %rsi
+; SCALAR-NEXT:    movq %r13, %rcx
+; SCALAR-NEXT:    imulq %r14, %rcx
+; SCALAR-NEXT:    movq %rbp, %rdx
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r11 # 8-byte Reload
+; SCALAR-NEXT:    imulq %r11, %rdx
+; SCALAR-NEXT:    xorq %rcx, %rdx
+; SCALAR-NEXT:    movq %rdi, %rcx
+; SCALAR-NEXT:    imulq %r15, %rcx
+; SCALAR-NEXT:    movq %r8, %r9
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; SCALAR-NEXT:    imulq %rax, %r9
+; SCALAR-NEXT:    xorq %rcx, %r9
+; SCALAR-NEXT:    xorq %rdx, %r9
+; SCALAR-NEXT:    movabsq $2459565876494606882, %rcx # imm = 0x2222222222222222
+; SCALAR-NEXT:    andq %rcx, %r9
+; SCALAR-NEXT:    movq %r13, %rcx
+; SCALAR-NEXT:    imulq %r15, %rcx
+; SCALAR-NEXT:    movq %rbp, %rdx
+; SCALAR-NEXT:    imulq %r14, %rdx
+; SCALAR-NEXT:    xorq %rcx, %rdx
+; SCALAR-NEXT:    movq %rdi, %r10
+; SCALAR-NEXT:    imulq %rax, %r10
+; SCALAR-NEXT:    movq %r8, %rcx
+; SCALAR-NEXT:    imulq %r11, %rcx
+; SCALAR-NEXT:    xorq %r10, %rcx
+; SCALAR-NEXT:    xorq %rdx, %rcx
+; SCALAR-NEXT:    movabsq $1229782938247303441, %rdx # imm = 0x1111111111111111
+; SCALAR-NEXT:    andq %rdx, %rcx
+; SCALAR-NEXT:    orq %r9, %rcx
+; SCALAR-NEXT:    movq %r13, %rdx
+; SCALAR-NEXT:    imulq %r11, %rdx
+; SCALAR-NEXT:    movq %rbp, %r9
+; SCALAR-NEXT:    imulq %rax, %r9
+; SCALAR-NEXT:    xorq %rdx, %r9
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    imulq %r14, %rdx
+; SCALAR-NEXT:    imulq %r8, %r14
+; SCALAR-NEXT:    imulq %r15, %r8
+; SCALAR-NEXT:    xorq %rdx, %r8
+; SCALAR-NEXT:    xorq %r9, %r8
+; SCALAR-NEXT:    movabsq $4919131752989213764, %rdx # imm = 0x4444444444444444
+; SCALAR-NEXT:    andq %rdx, %r8
+; SCALAR-NEXT:    orq %rcx, %r8
+; SCALAR-NEXT:    imulq %rax, %r13
+; SCALAR-NEXT:    imulq %r15, %rbp
+; SCALAR-NEXT:    xorq %r13, %rbp
+; SCALAR-NEXT:    imulq %r11, %rdi
+; SCALAR-NEXT:    xorq %rdi, %r14
+; SCALAR-NEXT:    xorq %rbp, %r14
+; SCALAR-NEXT:    andq %rbx, %r14
+; SCALAR-NEXT:    orq %r8, %r14
+; SCALAR-NEXT:    movq %r14, %rax
+; SCALAR-NEXT:    movq %rsi, %rdx
+; SCALAR-NEXT:    popq %rbx
+; SCALAR-NEXT:    .cfi_def_cfa_offset 48
+; SCALAR-NEXT:    popq %r12
+; SCALAR-NEXT:    .cfi_def_cfa_offset 40
+; SCALAR-NEXT:    popq %r13
+; SCALAR-NEXT:    .cfi_def_cfa_offset 32
+; SCALAR-NEXT:    popq %r14
+; SCALAR-NEXT:    .cfi_def_cfa_offset 24
+; SCALAR-NEXT:    popq %r15
+; SCALAR-NEXT:    .cfi_def_cfa_offset 16
+; SCALAR-NEXT:    popq %rbp
+; SCALAR-NEXT:    .cfi_def_cfa_offset 8
+; SCALAR-NEXT:    retq
+;
+; SSE2-PCLMUL-LABEL: clmul_i128:
+; SSE2-PCLMUL:       # %bb.0:
+; SSE2-PCLMUL-NEXT:    movq %rdx, %xmm0
+; SSE2-PCLMUL-NEXT:    movq %rsi, %xmm1
+; SSE2-PCLMUL-NEXT:    pclmulqdq $0, %xmm0, %xmm1
+; SSE2-PCLMUL-NEXT:    movq %xmm1, %rax
+; SSE2-PCLMUL-NEXT:    movq %rcx, %xmm1
+; SSE2-PCLMUL-NEXT:    movq %rdi, %xmm2
+; SSE2-PCLMUL-NEXT:    pclmulqdq $0, %xmm2, %xmm1
+; SSE2-PCLMUL-NEXT:    movq %xmm1, %rcx
+; SSE2-PCLMUL-NEXT:    xorq %rax, %rcx
+; SSE2-PCLMUL-NEXT:    pclmulqdq $0, %xmm0, %xmm2
+; SSE2-PCLMUL-NEXT:    pshufd {{.*#+}} xmm0 = xmm2[2,3,2,3]
+; SSE2-PCLMUL-NEXT:    movq %xmm0, %rdx
+; SSE2-PCLMUL-NEXT:    xorq %rcx, %rdx
+; SSE2-PCLMUL-NEXT:    movq %xmm2, %rax
+; SSE2-PCLMUL-NEXT:    retq
+;
+; SSE42-PCLMUL-LABEL: clmul_i128:
+; SSE42-PCLMUL:       # %bb.0:
+; SSE42-PCLMUL-NEXT:    movq %rdx, %xmm0
+; SSE42-PCLMUL-NEXT:    movq %rsi, %xmm1
+; SSE42-PCLMUL-NEXT:    pclmulqdq $0, %xmm0, %xmm1
+; SSE42-PCLMUL-NEXT:    movq %xmm1, %rax
+; SSE42-PCLMUL-NEXT:    movq %rcx, %xmm1
+; SSE42-PCLMUL-NEXT:    movq %rdi, %xmm2
+; SSE42-PCLMUL-NEXT:    pclmulqdq $0, %xmm2, %xmm1
+; SSE42-PCLMUL-NEXT:    movq %xmm1, %rcx
+; SSE42-PCLMUL-NEXT:    xorq %rax, %rcx
+; SSE42-PCLMUL-NEXT:    pclmulqdq $0, %xmm0, %xmm2
+; SSE42-PCLMUL-NEXT:    pextrq $1, %xmm2, %rdx
+; SSE42-PCLMUL-NEXT:    xorq %rcx, %rdx
+; SSE42-PCLMUL-NEXT:    movq %xmm2, %rax
+; SSE42-PCLMUL-NEXT:    retq
+;
+; AVX-LABEL: clmul_i128:
+; AVX:       # %bb.0:
+; AVX-NEXT:    vmovq %rdx, %xmm0
+; AVX-NEXT:    vmovq %rsi, %xmm1
+; AVX-NEXT:    vpclmulqdq $0, %xmm0, %xmm1, %xmm1
+; AVX-NEXT:    vmovq %xmm1, %rax
+; AVX-NEXT:    vmovq %rcx, %xmm1
+; AVX-NEXT:    vmovq %rdi, %xmm2
+; AVX-NEXT:    vpclmulqdq $0, %xmm1, %xmm2, %xmm1
+; AVX-NEXT:    vmovq %xmm1, %rcx
+; AVX-NEXT:    xorq %rax, %rcx
+; AVX-NEXT:    vpclmulqdq $0, %xmm0, %xmm2, %xmm0
+; AVX-NEXT:    vpextrq $1, %xmm0, %rdx
+; AVX-NEXT:    xorq %rcx, %rdx
+; AVX-NEXT:    vmovq %xmm0, %rax
+; AVX-NEXT:    retq
+  %a = call i128 @llvm.clmul.i128(i128 %x, i128 %y)
+  ret i128 %a
+}
+
+define i128 @clmul_i128_zext(i64 %x, i64 %y) {
+; SCALAR-LABEL: clmul_i128_zext:
+; SCALAR:       # %bb.0:
+; SCALAR-NEXT:    pushq %rbp
+; SCALAR-NEXT:    .cfi_def_cfa_offset 16
+; SCALAR-NEXT:    pushq %r15
+; SCALAR-NEXT:    .cfi_def_cfa_offset 24
+; SCALAR-NEXT:    pushq %r14
+; SCALAR-NEXT:    .cfi_def_cfa_offset 32
+; SCALAR-NEXT:    pushq %r13
+; SCALAR-NEXT:    .cfi_def_cfa_offset 40
+; SCALAR-NEXT:    pushq %r12
+; SCALAR-NEXT:    .cfi_def_cfa_offset 48
+; SCALAR-NEXT:    pushq %rbx
+; SCALAR-NEXT:    .cfi_def_cfa_offset 56
+; SCALAR-NEXT:    .cfi_offset %rbx, -56
+; SCALAR-NEXT:    .cfi_offset %r12, -48
+; SCALAR-NEXT:    .cfi_offset %r13, -40
+; SCALAR-NEXT:    .cfi_offset %r14, -32
+; SCALAR-NEXT:    .cfi_offset %r15, -24
+; SCALAR-NEXT:    .cfi_offset %rbp, -16
+; SCALAR-NEXT:    movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %rdi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; SCALAR-NEXT:    movq %rsi, %rax
+; SCALAR-NEXT:    bswapq %rax
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    shrq $4, %rcx
+; SCALAR-NEXT:    movabsq $1085102592571150095, %r11 # imm = 0xF0F0F0F0F0F0F0F
+; SCALAR-NEXT:    andq %r11, %rcx
+; SCALAR-NEXT:    andq %r11, %rax
+; SCALAR-NEXT:    shlq $4, %rax
+; SCALAR-NEXT:    orq %rcx, %rax
+; SCALAR-NEXT:    movabsq $3689348814741910323, %rsi # imm = 0x3333333333333333
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    andq %rsi, %rcx
+; SCALAR-NEXT:    shrq $2, %rax
+; SCALAR-NEXT:    andq %rsi, %rax
+; SCALAR-NEXT:    leaq (%rax,%rcx,4), %rax
+; SCALAR-NEXT:    movabsq $6148914691236517205, %r9 # imm = 0x5555555555555555
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    andq %r9, %rcx
+; SCALAR-NEXT:    shrq %rax
+; SCALAR-NEXT:    andq %r9, %rax
+; SCALAR-NEXT:    leaq (%rax,%rcx,2), %r15
+; SCALAR-NEXT:    movabsq $1229782938247303441, %r10 # imm = 0x1111111111111111
+; SCALAR-NEXT:    movq %r15, %rbx
+; SCALAR-NEXT:    andq %r10, %rbx
+; SCALAR-NEXT:    movq %rdi, %rax
+; SCALAR-NEXT:    bswapq %rax
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    shrq $4, %rcx
+; SCALAR-NEXT:    andq %r11, %rcx
+; SCALAR-NEXT:    andq %r11, %rax
+; SCALAR-NEXT:    shlq $4, %rax
+; SCALAR-NEXT:    orq %rcx, %rax
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    andq %rsi, %rcx
+; SCALAR-NEXT:    shrq $2, %rax
+; SCALAR-NEXT:    andq %rsi, %rax
+; SCALAR-NEXT:    leaq (%rax,%rcx,4), %rax
+; SCALAR-NEXT:    movq %rax, %rcx
+; SCALAR-NEXT:    andq %r9, %rcx
+; SCALAR-NEXT:    shrq %rax
+; SCALAR-NEXT:    andq %r9, %rax
+; SCALAR-NEXT:    leaq (%rax,%rcx,2), %r14
+; SCALAR-NEXT:    movabsq $2459565876494606882, %rcx # imm = 0x2222222222222222
+; SCALAR-NEXT:    movq %r14, %rbp
+; SCALAR-NEXT:    andq %rcx, %rbp
+; SCALAR-NEXT:    movq %rbp, %rax
+; SCALAR-NEXT:    imulq %rbx, %rax
+; SCALAR-NEXT:    movq %r15, %r12
+; SCALAR-NEXT:    andq %rcx, %r12
+; SCALAR-NEXT:    movq %r14, %r8
+; SCALAR-NEXT:    andq %r10, %r8
+; SCALAR-NEXT:    movq %r8, %rcx
+; SCALAR-NEXT:    imulq %r12, %rcx
+; SCALAR-NEXT:    xorq %rax, %rcx
+; SCALAR-NEXT:    movabsq $-8608480567731124088, %rax # imm = 0x8888888888888888
+; SCALAR-NEXT:    movq %r15, %r13
+; SCALAR-NEXT:    andq %rax, %r13
+; SCALAR-NEXT:    movabsq $4919131752989213764, %rsi # imm = 0x4444444444444444
+; SCALAR-NEXT:    movq %r14, %rdi
+; SCALAR-NEXT:    andq %rsi, %rdi
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    imulq %r13, %rdx
+; SCALAR-NEXT:    andq %rsi, %r15
+; SCALAR-NEXT:    andq %rax, %r14
+; SCALAR-NEXT:    movq %r14, %rax
+; SCALAR-NEXT:    imulq %r15, %rax
+; SCALAR-NEXT:    xorq %rdx, %rax
+; SCALAR-NEXT:    xorq %rcx, %rax
+; SCALAR-NEXT:    movq %rbp, %rcx
+; SCALAR-NEXT:    imulq %r13, %rcx
+; SCALAR-NEXT:    movq %r8, %rdx
+; SCALAR-NEXT:    imulq %rbx, %rdx
+; SCALAR-NEXT:    xorq %rcx, %rdx
+; SCALAR-NEXT:    movq %rdi, %rsi
+; SCALAR-NEXT:    imulq %r15, %rsi
+; SCALAR-NEXT:    movq %r14, %rcx
+; SCALAR-NEXT:    imulq %r12, %rcx
+; SCALAR-NEXT:    xorq %rsi, %rcx
+; SCALAR-NEXT:    xorq %rdx, %rcx
+; SCALAR-NEXT:    movabsq $2459565876494606882, %rsi # imm = 0x2222222222222222
+; SCALAR-NEXT:    andq %rsi, %rax
+; SCALAR-NEXT:    andq %r10, %rcx
+; SCALAR-NEXT:    orq %rax, %rcx
+; SCALAR-NEXT:    movq %r14, %rax
+; SCALAR-NEXT:    imulq %r13, %rax
+; SCALAR-NEXT:    imulq %r8, %r13
+; SCALAR-NEXT:    imulq %r15, %r8
+; SCALAR-NEXT:    imulq %rbp, %r15
+; SCALAR-NEXT:    imulq %r12, %rbp
+; SCALAR-NEXT:    xorq %rbp, %r8
+; SCALAR-NEXT:    movq %rdi, %rdx
+; SCALAR-NEXT:    imulq %rbx, %rdx
+; SCALAR-NEXT:    xorq %rdx, %rax
+; SCALAR-NEXT:    xorq %r8, %rax
+; SCALAR-NEXT:    movabsq $4919131752989213764, %r8 # imm = 0x4444444444444444
+; SCALAR-NEXT:    andq %r8, %rax
+; SCALAR-NEXT:    orq %rcx, %rax
+; SCALAR-NEXT:    xorq %r15, %r13
+; SCALAR-NEXT:    imulq %r12, %rdi
+; SCALAR-NEXT:    imulq %rbx, %r14
+; SCALAR-NEXT:    xorq %rdi, %r14
+; SCALAR-NEXT:    xorq %r13, %r14
+; SCALAR-NEXT:    movabsq $-8608480567731124088, %rbp # imm = 0x8888888888888888
+; SCALAR-NEXT:    andq %rbp, %r14
+; SCALAR-NEXT:    orq %rax, %r14
+; SCALAR-NEXT:    bswapq %r14
+; SCALAR-NEXT:    movq %r14, %rax
+; SCALAR-NEXT:    shrq $4, %rax
+; SCALAR-NEXT:    andq %r11, %rax
+; SCALAR-NEXT:    andq %r11, %r14
+; SCALAR-NEXT:    shlq $4, %r14
+; SCALAR-NEXT:    orq %rax, %r14
+; SCALAR-NEXT:    movq %r14, %rax
+; SCALAR-NEXT:    movabsq $3689348814741910323, %rcx # imm = 0x3333333333333333
+; SCALAR-NEXT:    andq %rcx, %rax
+; SCALAR-NEXT:    shrq $2, %r14
+; SCALAR-NEXT:    andq %rcx, %r14
+; SCALAR-NEXT:    leaq (%r14,%rax,4), %rax
+; SCALAR-NEXT:    andq %rax, %r9
+; SCALAR-NEXT:    shrq %rax
+; SCALAR-NEXT:    movabsq $6148914691236517204, %rcx # imm = 0x5555555555555554
+; SCALAR-NEXT:    andq %rax, %rcx
+; SCALAR-NEXT:    leaq (%rcx,%r9,2), %rdx
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r9 # 8-byte Reload
+; SCALAR-NEXT:    movq %r9, %rax
+; SCALAR-NEXT:    movq %r10, %r13
+; SCALAR-NEXT:    andq %r10, %rax
+; SCALAR-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %r12 # 8-byte Reload
+; SCALAR-NEXT:    movq %r12, %r11
+; SCALAR-NEXT:    movq %rsi, %rcx
+; SCALAR-NEXT:    andq %rsi, %r11
+; SCALAR-NEXT:    movq %r11, %rsi
+; SCALAR-NEXT:    imulq %rax, %rsi
+; SCALAR-NEXT:    movq %r9, %rbx
+; SCALAR-NEXT:    andq %rcx, %rbx
+; SCALAR-NEXT:    movq %rcx, %r10
+; SCALAR-NEXT:    movq %r12, %rcx
+; SCALAR-NEXT:    andq %r13, %rcx
+; SCALAR-NEXT:    movq %rcx, %rdi
+; SCALAR-NEXT:    imulq %rbx, %rdi
+; SCALAR-NEXT:    xorq %rsi, %rdi
+; SCALAR-NEXT:    movq %r9, %r14
+; SCALAR-NEXT:    andq %rbp, %r14
+; SCALAR-NEXT:    movq %r12, %r15
+; SCALAR-NEXT:    andq %r8, %r15
+; SCALAR-NEXT:    movq %r15, %rsi
+; SCALAR-NEXT:    imulq %r14, %rsi
+; SCALAR-NEXT:    andq %r8, %r9
+; SCALAR-NEXT:    andq %rbp, %r12
+; SCALAR-NEXT:    movq %r12, %r8
+; SCALAR-NEXT:    imulq %r9, %r8
+; SCALAR-NEXT:    xorq %rsi, %r8
+; SCALAR-NEXT:    xorq %rdi, %r8
+; SCALAR-NEXT:    movq %rcx, %rsi
+; SCALAR-NEXT:    andq %r10, %r8
+; SCALAR-NEXT:    movq %r11, %rdi
+; SCALAR-NEXT:    imulq %r14, %rdi
+; SCALAR-NEXT:    imulq %rax, %rsi
+; SCALAR-NEXT:    xorq %rdi, %rsi
+; SCALAR-NEXT:    movq %r15, %rdi
+; SCALAR-NEXT:    imulq %r9, %rdi
+; SCALAR-NEXT:    movq %r12, %r10
+; SCALAR-NEXT:    imulq %rbx, %r10
+; SCALAR-NEXT:    xorq %rdi, %r10
+; SCALAR-NEXT:    xorq %rsi, %r10
+; SCALAR-NEXT:    andq %r13, %r10
+; SCALAR-NEXT:    movq %r15, %rsi
+; SCALAR-NEXT:    imulq %rax, %rsi
+; SCALAR-NEXT:    imulq %r12, %rax
+; SCALAR-NEXT:    movq %r12, %rdi
+; SCALAR-NEXT:    imulq %r14, %rdi
+; SCALAR-NEXT:    imulq %rcx, %r14
+; SCALAR-NEXT:    orq %r8, %r10
+; SCALAR-NEXT:    movq %r11, %r8
+; SCALAR-NEXT:    imulq %rbx, %r8
+; SCALAR-NEXT:    imulq %r9, %rcx
+; SCALAR-NEXT:    xorq %r8, %rcx
+; SCALAR-NEXT:    xorq %rsi, %rdi
+; SCALAR-NEXT:    xorq %rcx, %rdi
+; SCALAR-NEXT:    movabsq $4919131752989213764, %rcx # imm = 0x4444444444444444
+; SCALAR-NEXT:    andq %rcx, %rdi
+; SCALAR-NEXT:    orq %r10, %rdi
+; SCALAR-NEXT:    imulq %r9, %r11
+; SCALAR-NEXT:    xorq %r11, %r14
+; SCALAR-NEXT:    imulq %rbx, %r15
+; SCALAR-NEXT:    xorq %r15, %rax
+; SCALAR-NEXT:    xorq %r14, %rax
+; SCALAR-NEXT:    andq %rbp, %rax
+; SCALAR-NEXT:    orq %rdi, %rax
+; SCALAR-NEXT:    shrq %rdx
+; SCALAR-NEXT:    popq %rbx
+; SCALAR-NEXT:    .cfi_def_cfa_offset 48
+; SCALAR-NEXT:    popq %r12
+; SCALAR-NEXT:    .cfi_def_cfa_offset 40
+; SCALAR-NEXT:    popq %r13
+; SCALAR-NEXT:    .cfi_def_cfa_offset 32
+; SCALAR-NEXT:    popq %r14
+; SCALAR-NEXT:    .cfi_def_cfa_offset 24
+; SCALAR-NEXT:    popq %r15
+; SCALAR-NEXT:    .cfi_def_cfa_offset 16
+; SCALAR-NEXT:    popq %rbp
+; SCALAR-NEXT:    .cfi_def_cfa_offset 8
+; SCALAR-NEXT:    retq
+;
+; SSE2-PCLMUL-LABEL: clmul_i128_zext:
+; SSE2-PCLMUL:       # %bb.0:
+; SSE2-PCLMUL-NEXT:    movq %rsi, %xmm0
+; SSE2-PCLMUL-NEXT:    movq %rdi, %xmm1
+; SSE2-PCLMUL-NEXT:    pclmulqdq $0, %xmm0, %xmm1
+; SSE2-PCLMUL-NEXT:    movq %xmm1, %rax
+; SSE2-PCLMUL-NEXT:    pshufd {{.*#+}} xmm0 = xmm1[2,3,2,3]
+; SSE2-PCLMUL-NEXT:    movq %xmm0, %rdx
+; SSE2-PCLMUL-NEXT:    retq
+;
+; SSE42-PCLMUL-LABEL: clmul_i128_zext:
+; SSE42-PCLMUL:       # %bb.0:
+; SSE42-PCLMUL-NEXT:    movq %rsi, %xmm0
+; SSE42-PCLMUL-NEXT:    movq %rdi, %xmm1
+; SSE42-PCLMUL-NEXT:    pclmulqdq $0, %xmm0, %xmm1
+; SSE42-PCLMUL-NEXT:    movq %xmm1, %rax
+; SSE42-PCLMUL-NEXT:    pextrq $1, %xmm1, %rdx
+; SSE42-PCLMUL-NEXT:    retq
+;
+; AVX-LABEL: clmul_i128_zext:
+; AVX:       # %bb.0:
+; AVX-NEXT:    vmovq %rsi, %xmm0
+; AVX-NEXT:    vmovq %rdi, %xmm1
+; AVX-NEXT:    vpclmulqdq $0, %xmm0, %xmm1, %xmm0
+; AVX-NEXT:    vmovq %xmm0, %rax
+; AVX-NEXT:    vpextrq $1, %xmm0, %rdx
+; AVX-NEXT:    retq
+  %zextx = zext i64 %x to i128
+  %zexty = zext i64 %y to i128
+  %a = call i128 @llvm.clmul.i128(i128 %zextx, i128 %zexty)
+  ret i128 %a
+}
+
 define i8 @clmulr_i8(i8 %a, i8 %b) nounwind {
 ; SCALAR-LABEL: clmulr_i8:
 ; SCALAR:       # %bb.0:


        


More information about the llvm-commits mailing list