[llvm] [AArch64] SVE Shuffleopt: merge reduction reverse into tbl (PR #206047)

Graham Hunter via llvm-commits llvm-commits at lists.llvm.org
Fri Jul 31 06:38:54 PDT 2026


https://github.com/huntergr-arm updated https://github.com/llvm/llvm-project/pull/206047

>From 4bacf904f7ea1de2ee6fe5ae1812191b2519eee9 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Thu, 25 Jun 2026 14:52:48 +0000
Subject: [PATCH 1/4] Add tests with a reverse on the other operand of the
 extend user

---
 .../CodeGen/AArch64/sve-tbl-folding-new-pm.ll | 131 ++++++
 .../CodeGen/AArch64/sve-tbl-folding-opts.ll   | 440 ++++++++++++++++++
 2 files changed, 571 insertions(+)

diff --git a/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll b/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
index 6a533a2419255..d739dd36ef804 100644
--- a/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
+++ b/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
@@ -207,4 +207,135 @@ exit:
   ret void
 }
 
+define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(
+; CHECK-SAME: ptr [[SRC:%.*]], ptr [[SRC2:%.*]], ptr [[DST:%.*]], <vscale x 8 x i1> [[MASK:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[VSCALE:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[STRIDE:%.*]] = shl nuw nsw i64 [[VSCALE]], 2
+; CHECK-NEXT:    [[COMMON_BASE:%.*]] = getelementptr double, ptr [[SRC2]], i64 1
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[ACC_B_F64:%.*]] = phi <vscale x 2 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[FADD_B_F64:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[ACC_G_F64:%.*]] = phi <vscale x 2 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[FADD_G_F64:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[ACC_R_F64:%.*]] = phi <vscale x 2 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[FADD_R_F64:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[ACC_A_F64:%.*]] = phi <vscale x 2 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[FADD_A_F64:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[NEGATED:%.*]] = mul i64 [[IV]], -1
+; CHECK-NEXT:    [[COMMON_TERM_PTR:%.*]] = getelementptr inbounds nuw double, ptr [[COMMON_BASE]], i64 [[NEGATED]]
+; CHECK-NEXT:    [[COMMON_TERM:%.*]] = load <vscale x 2 x double>, ptr [[COMMON_TERM_PTR]], align 8
+; CHECK-NEXT:    [[REVERSED:%.*]] = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> [[COMMON_TERM]])
+; CHECK-NEXT:    [[SRC_GEP:%.*]] = getelementptr inbounds nuw [4 x i16], ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT:    [[BGRA:%.*]] = load <vscale x 8 x i16>, ptr [[SRC_GEP]], align 16
+; CHECK-NEXT:    [[TMP0:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[TMP28:%.*]] = mul nuw <vscale x 2 x i64> [[TMP0]], splat (i64 4)
+; CHECK-NEXT:    [[TMP34:%.*]] = add nuw <vscale x 2 x i64> [[TMP28]], splat (i64 -65536)
+; CHECK-NEXT:    [[TMP3:%.*]] = bitcast <vscale x 2 x i64> [[TMP34]] to <vscale x 8 x i16>
+; CHECK-NEXT:    [[TMP4:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP3]])
+; CHECK-NEXT:    [[TMP5:%.*]] = bitcast <vscale x 8 x i16> [[TMP4]] to <vscale x 2 x i64>
+; CHECK-NEXT:    [[TMP6:%.*]] = uitofp <vscale x 2 x i64> [[TMP5]] to <vscale x 2 x double>
+; CHECK-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[TMP15:%.*]] = mul nuw <vscale x 2 x i64> [[TMP7]], splat (i64 4)
+; CHECK-NEXT:    [[TMP41:%.*]] = add nuw <vscale x 2 x i64> [[TMP15]], splat (i64 -65535)
+; CHECK-NEXT:    [[TMP10:%.*]] = bitcast <vscale x 2 x i64> [[TMP41]] to <vscale x 8 x i16>
+; CHECK-NEXT:    [[TMP11:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP10]])
+; CHECK-NEXT:    [[TMP12:%.*]] = bitcast <vscale x 8 x i16> [[TMP11]] to <vscale x 2 x i64>
+; CHECK-NEXT:    [[TMP13:%.*]] = uitofp <vscale x 2 x i64> [[TMP12]] to <vscale x 2 x double>
+; CHECK-NEXT:    [[TMP14:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[TMP52:%.*]] = mul nuw <vscale x 2 x i64> [[TMP14]], splat (i64 4)
+; CHECK-NEXT:    [[TMP53:%.*]] = add nuw <vscale x 2 x i64> [[TMP52]], splat (i64 -65534)
+; CHECK-NEXT:    [[TMP17:%.*]] = bitcast <vscale x 2 x i64> [[TMP53]] to <vscale x 8 x i16>
+; CHECK-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP17]])
+; CHECK-NEXT:    [[TMP19:%.*]] = bitcast <vscale x 8 x i16> [[TMP18]] to <vscale x 2 x i64>
+; CHECK-NEXT:    [[TMP20:%.*]] = uitofp <vscale x 2 x i64> [[TMP19]] to <vscale x 2 x double>
+; CHECK-NEXT:    [[TMP21:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[TMP43:%.*]] = mul nuw <vscale x 2 x i64> [[TMP21]], splat (i64 4)
+; CHECK-NEXT:    [[TMP44:%.*]] = add nuw <vscale x 2 x i64> [[TMP43]], splat (i64 -65533)
+; CHECK-NEXT:    [[TMP24:%.*]] = bitcast <vscale x 2 x i64> [[TMP44]] to <vscale x 8 x i16>
+; CHECK-NEXT:    [[TMP25:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP24]])
+; CHECK-NEXT:    [[TMP26:%.*]] = bitcast <vscale x 8 x i16> [[TMP25]] to <vscale x 2 x i64>
+; CHECK-NEXT:    [[TMP27:%.*]] = uitofp <vscale x 2 x i64> [[TMP26]] to <vscale x 2 x double>
+; CHECK-NEXT:    [[B_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP6]], [[REVERSED]]
+; CHECK-NEXT:    [[G_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP13]], [[REVERSED]]
+; CHECK-NEXT:    [[R_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP20]], [[REVERSED]]
+; CHECK-NEXT:    [[A_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP27]], [[REVERSED]]
+; CHECK-NEXT:    [[FADD_B_F64]] = fadd <vscale x 2 x double> [[ACC_B_F64]], [[B_MUL_F64]]
+; CHECK-NEXT:    [[FADD_G_F64]] = fadd <vscale x 2 x double> [[ACC_G_F64]], [[G_MUL_F64]]
+; CHECK-NEXT:    [[FADD_R_F64]] = fadd <vscale x 2 x double> [[ACC_R_F64]], [[R_MUL_F64]]
+; CHECK-NEXT:    [[FADD_A_F64]] = fadd <vscale x 2 x double> [[ACC_A_F64]], [[A_MUL_F64]]
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw i64 [[IV]], [[STRIDE]]
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 2048
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[FADD_B_F64_LCSSA:%.*]] = phi <vscale x 2 x double> [ [[FADD_B_F64]], %[[LOOP]] ]
+; CHECK-NEXT:    [[FADD_G_F64_LCSSA:%.*]] = phi <vscale x 2 x double> [ [[FADD_G_F64]], %[[LOOP]] ]
+; CHECK-NEXT:    [[FADD_R_F64_LCSSA:%.*]] = phi <vscale x 2 x double> [ [[FADD_R_F64]], %[[LOOP]] ]
+; CHECK-NEXT:    [[FADD_A_F64_LCSSA:%.*]] = phi <vscale x 2 x double> [ [[FADD_A_F64]], %[[LOOP]] ]
+; CHECK-NEXT:    [[B_ACC:%.*]] = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> [[FADD_B_F64_LCSSA]])
+; CHECK-NEXT:    store double [[B_ACC]], ptr [[DST]], align 8
+; CHECK-NEXT:    [[G_ACC:%.*]] = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> [[FADD_G_F64_LCSSA]])
+; CHECK-NEXT:    [[G_F64_GEP:%.*]] = getelementptr double, ptr [[DST]], i64 1
+; CHECK-NEXT:    store double [[G_ACC]], ptr [[G_F64_GEP]], align 8
+; CHECK-NEXT:    [[R_ACC:%.*]] = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> [[FADD_R_F64_LCSSA]])
+; CHECK-NEXT:    [[R_F64_GEP:%.*]] = getelementptr double, ptr [[DST]], i64 2
+; CHECK-NEXT:    store double [[R_ACC]], ptr [[R_F64_GEP]], align 8
+; CHECK-NEXT:    [[A_ACC:%.*]] = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> [[FADD_A_F64_LCSSA]])
+; CHECK-NEXT:    [[A_F64_GEP:%.*]] = getelementptr double, ptr [[DST]], i64 3
+; CHECK-NEXT:    store double [[A_ACC]], ptr [[A_F64_GEP]], align 8
+; CHECK-NEXT:    ret void
+;
+entry:
+  %vscale = tail call i64 @llvm.vscale.i64()
+  %stride = shl nuw nsw i64 %vscale, 2
+  %common.base = getelementptr double, ptr %src2, i64 1
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.b.f64, %loop ]
+  %acc.g.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.g.f64, %loop ]
+  %acc.r.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.r.f64, %loop ]
+  %acc.a.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.a.f64, %loop ]
+  %negated = mul i64 %iv, -1
+  %common.term.ptr = getelementptr inbounds nuw double, ptr %common.base, i64 %negated
+  %common.term = load <vscale x 2 x double>, ptr %common.term.ptr, align 8
+  %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> %common.term)
+  %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+  %bgra = load <vscale x 8 x i16>, ptr %src.gep, align 16
+  %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+  %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+  %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+  %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+  %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+  %b.f64 = uitofp <vscale x 2 x i16> %b.i16 to <vscale x 2 x double>
+  %g.f64 = uitofp <vscale x 2 x i16> %g.i16 to <vscale x 2 x double>
+  %r.f64 = uitofp <vscale x 2 x i16> %r.i16 to <vscale x 2 x double>
+  %a.f64 = uitofp <vscale x 2 x i16> %a.i16 to <vscale x 2 x double>
+  %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+  %g.mul.f64 = fmul <vscale x 2 x double> %g.f64, %reversed
+  %r.mul.f64 = fmul <vscale x 2 x double> %r.f64, %reversed
+  %a.mul.f64 = fmul <vscale x 2 x double> %a.f64, %reversed
+  %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+  %fadd.g.f64 = fadd <vscale x 2 x double> %acc.g.f64, %g.mul.f64
+  %fadd.r.f64 = fadd <vscale x 2 x double> %acc.r.f64, %r.mul.f64
+  %fadd.a.f64 = fadd <vscale x 2 x double> %acc.a.f64, %a.mul.f64
+  %iv.next = add nuw i64 %iv, %stride
+  %ec = icmp eq i64 %iv.next, 2048
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  %b.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.b.f64)
+  store double %b.acc, ptr %dst
+  %g.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.g.f64)
+  %g.f64.gep = getelementptr double, ptr %dst, i64 1
+  store double %g.acc, ptr %g.f64.gep
+  %r.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.r.f64)
+  %r.f64.gep = getelementptr double, ptr %dst, i64 2
+  store double %r.acc, ptr %r.f64.gep
+  %a.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.a.f64)
+  %a.f64.gep = getelementptr double, ptr %dst, i64 3
+  store double %a.acc, ptr %a.f64.gep
+  ret void
+}
+
 attributes #0 = { "target-features"="+sve" }
diff --git a/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll b/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
index e101489c564c8..fa3df9f39a4ad 100644
--- a/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
+++ b/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
@@ -638,5 +638,445 @@ exit:
   ret void
 }
 
+define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    index z7.d, #0, #4
+; CHECK-NEXT:    mov z2.d, #0xffffffffffff0000
+; CHECK-NEXT:    mov z5.d, #0xffffffffffff0001
+; CHECK-NEXT:    mov x9, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT:    mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT:    movi v0.2d, #0000000000000000
+; CHECK-NEXT:    mov z6.d, x9
+; CHECK-NEXT:    movi v3.2d, #0000000000000000
+; CHECK-NEXT:    mov x8, xzr
+; CHECK-NEXT:    movi v1.2d, #0000000000000000
+; CHECK-NEXT:    ptrue p1.d
+; CHECK-NEXT:    add x9, x1, #8
+; CHECK-NEXT:    add z4.d, z7.d, z2.d
+; CHECK-NEXT:    movi v2.2d, #0000000000000000
+; CHECK-NEXT:    cntw x10
+; CHECK-NEXT:    add z5.d, z7.d, z5.d
+; CHECK-NEXT:    add z6.d, z7.d, z6.d
+; CHECK-NEXT:    rdvl x11, #2
+; CHECK-NEXT:    add z7.d, z7.d, z24.d
+; CHECK-NEXT:  .LBB8_1: // %loop
+; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    ld1h { z24.h }, p0/z, [x0]
+; CHECK-NEXT:    ld1d { z28.d }, p1/z, [x9, x8, lsl #3]
+; CHECK-NEXT:    sub x8, x8, x10
+; CHECK-NEXT:    cmn x8, #2048
+; CHECK-NEXT:    add x0, x0, x11
+; CHECK-NEXT:    tbl z25.h, { z24.h }, z4.h
+; CHECK-NEXT:    tbl z26.h, { z24.h }, z5.h
+; CHECK-NEXT:    tbl z27.h, { z24.h }, z6.h
+; CHECK-NEXT:    tbl z24.h, { z24.h }, z7.h
+; CHECK-NEXT:    rev z28.d, z28.d
+; CHECK-NEXT:    ucvtf z25.d, p1/m, z25.d
+; CHECK-NEXT:    ucvtf z26.d, p1/m, z26.d
+; CHECK-NEXT:    ucvtf z27.d, p1/m, z27.d
+; CHECK-NEXT:    ucvtf z24.d, p1/m, z24.d
+; CHECK-NEXT:    fmul z25.d, z25.d, z28.d
+; CHECK-NEXT:    fmul z26.d, z26.d, z28.d
+; CHECK-NEXT:    fmul z27.d, z27.d, z28.d
+; CHECK-NEXT:    fmul z24.d, z24.d, z28.d
+; CHECK-NEXT:    fadd z0.d, z0.d, z25.d
+; CHECK-NEXT:    fadd z3.d, z3.d, z26.d
+; CHECK-NEXT:    fadd z1.d, z1.d, z27.d
+; CHECK-NEXT:    fadd z2.d, z2.d, z24.d
+; CHECK-NEXT:    b.ne .LBB8_1
+; CHECK-NEXT:  // %bb.2: // %exit
+; CHECK-NEXT:    faddv d0, p1, z0.d
+; CHECK-NEXT:    faddv d3, p1, z3.d
+; CHECK-NEXT:    faddv d1, p1, z1.d
+; CHECK-NEXT:    faddv d2, p1, z2.d
+; CHECK-NEXT:    mov v0.d[1], v3.d[0]
+; CHECK-NEXT:    mov v1.d[1], v2.d[0]
+; CHECK-NEXT:    stp q0, q1, [x2]
+; CHECK-NEXT:    ret
+entry:
+  %vscale = tail call i64 @llvm.vscale.i64()
+  %stride = shl nuw nsw i64 %vscale, 2
+  %common.base = getelementptr double, ptr %src2, i64 1
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.b.f64, %loop ]
+  %acc.g.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.g.f64, %loop ]
+  %acc.r.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.r.f64, %loop ]
+  %acc.a.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.a.f64, %loop ]
+  %negated = mul i64 %iv, -1
+  %common.term.ptr = getelementptr inbounds nuw double, ptr %common.base, i64 %negated
+  %common.term = load <vscale x 2 x double>, ptr %common.term.ptr, align 8
+  %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> %common.term)
+  %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+  %bgra = call <vscale x 8 x i16> @llvm.masked.load(ptr %src.gep, <vscale x 8 x i1> %mask, <vscale x 8 x i16> zeroinitializer)
+  %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+  %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+  %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+  %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+  %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+  %b.f64 = uitofp <vscale x 2 x i16> %b.i16 to <vscale x 2 x double>
+  %g.f64 = uitofp <vscale x 2 x i16> %g.i16 to <vscale x 2 x double>
+  %r.f64 = uitofp <vscale x 2 x i16> %r.i16 to <vscale x 2 x double>
+  %a.f64 = uitofp <vscale x 2 x i16> %a.i16 to <vscale x 2 x double>
+  %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+  %g.mul.f64 = fmul <vscale x 2 x double> %g.f64, %reversed
+  %r.mul.f64 = fmul <vscale x 2 x double> %r.f64, %reversed
+  %a.mul.f64 = fmul <vscale x 2 x double> %a.f64, %reversed
+  %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+  %fadd.g.f64 = fadd <vscale x 2 x double> %acc.g.f64, %g.mul.f64
+  %fadd.r.f64 = fadd <vscale x 2 x double> %acc.r.f64, %r.mul.f64
+  %fadd.a.f64 = fadd <vscale x 2 x double> %acc.a.f64, %a.mul.f64
+  %iv.next = add nuw i64 %iv, %stride
+  %ec = icmp eq i64 %iv.next, 2048
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  %b.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.b.f64)
+  store double %b.acc, ptr %dst
+  %g.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.g.f64)
+  %g.f64.gep = getelementptr double, ptr %dst, i64 1
+  store double %g.acc, ptr %g.f64.gep
+  %r.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.r.f64)
+  %r.f64.gep = getelementptr double, ptr %dst, i64 2
+  store double %r.acc, ptr %r.f64.gep
+  %a.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.a.f64)
+  %a.f64.gep = getelementptr double, ptr %dst, i64 3
+  store double %a.acc, ptr %a.f64.gep
+  ret void
+}
+
+define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_partial_coverage(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_partial_coverage:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    index z7.d, #0, #4
+; CHECK-NEXT:    mov z2.d, #0xffffffffffff0000
+; CHECK-NEXT:    mov z5.d, #0xffffffffffff0001
+; CHECK-NEXT:    mov x9, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT:    mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT:    movi v0.2d, #0000000000000000
+; CHECK-NEXT:    mov z6.d, x9
+; CHECK-NEXT:    movi v3.2d, #0000000000000000
+; CHECK-NEXT:    mov x8, xzr
+; CHECK-NEXT:    movi v1.2d, #0000000000000000
+; CHECK-NEXT:    ptrue p1.d
+; CHECK-NEXT:    add x9, x1, #8
+; CHECK-NEXT:    add z4.d, z7.d, z2.d
+; CHECK-NEXT:    movi v2.2d, #0000000000000000
+; CHECK-NEXT:    cntw x10
+; CHECK-NEXT:    add z5.d, z7.d, z5.d
+; CHECK-NEXT:    add z6.d, z7.d, z6.d
+; CHECK-NEXT:    rdvl x11, #2
+; CHECK-NEXT:    add z7.d, z7.d, z24.d
+; CHECK-NEXT:  .LBB9_1: // %loop
+; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    ld1h { z24.h }, p0/z, [x0]
+; CHECK-NEXT:    ld1d { z28.d }, p1/z, [x9, x8, lsl #3]
+; CHECK-NEXT:    sub x8, x8, x10
+; CHECK-NEXT:    cmn x8, #2048
+; CHECK-NEXT:    add x0, x0, x11
+; CHECK-NEXT:    tbl z25.h, { z24.h }, z4.h
+; CHECK-NEXT:    tbl z26.h, { z24.h }, z5.h
+; CHECK-NEXT:    tbl z27.h, { z24.h }, z6.h
+; CHECK-NEXT:    tbl z24.h, { z24.h }, z7.h
+; CHECK-NEXT:    rev z29.d, z28.d
+; CHECK-NEXT:    ucvtf z25.d, p1/m, z25.d
+; CHECK-NEXT:    ucvtf z26.d, p1/m, z26.d
+; CHECK-NEXT:    ucvtf z27.d, p1/m, z27.d
+; CHECK-NEXT:    ucvtf z24.d, p1/m, z24.d
+; CHECK-NEXT:    fmul z25.d, z25.d, z29.d
+; CHECK-NEXT:    fmul z26.d, z26.d, z28.d
+; CHECK-NEXT:    fmul z27.d, z27.d, z29.d
+; CHECK-NEXT:    fmul z24.d, z24.d, z28.d
+; CHECK-NEXT:    fadd z0.d, z0.d, z25.d
+; CHECK-NEXT:    fadd z3.d, z3.d, z26.d
+; CHECK-NEXT:    fadd z1.d, z1.d, z27.d
+; CHECK-NEXT:    fadd z2.d, z2.d, z24.d
+; CHECK-NEXT:    b.ne .LBB9_1
+; CHECK-NEXT:  // %bb.2: // %exit
+; CHECK-NEXT:    faddv d0, p1, z0.d
+; CHECK-NEXT:    faddv d3, p1, z3.d
+; CHECK-NEXT:    faddv d1, p1, z1.d
+; CHECK-NEXT:    faddv d2, p1, z2.d
+; CHECK-NEXT:    mov v0.d[1], v3.d[0]
+; CHECK-NEXT:    mov v1.d[1], v2.d[0]
+; CHECK-NEXT:    stp q0, q1, [x2]
+; CHECK-NEXT:    ret
+entry:
+  %vscale = tail call i64 @llvm.vscale.i64()
+  %stride = shl nuw nsw i64 %vscale, 2
+  %common.base = getelementptr double, ptr %src2, i64 1
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.b.f64, %loop ]
+  %acc.g.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.g.f64, %loop ]
+  %acc.r.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.r.f64, %loop ]
+  %acc.a.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.a.f64, %loop ]
+  %negated = mul i64 %iv, -1
+  %common.term.ptr = getelementptr inbounds nuw double, ptr %common.base, i64 %negated
+  %common.term = load <vscale x 2 x double>, ptr %common.term.ptr, align 8
+  %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> %common.term)
+  %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+  %bgra = call <vscale x 8 x i16> @llvm.masked.load(ptr %src.gep, <vscale x 8 x i1> %mask, <vscale x 8 x i16> zeroinitializer)
+  %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+  %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+  %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+  %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+  %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+  %b.f64 = uitofp <vscale x 2 x i16> %b.i16 to <vscale x 2 x double>
+  %g.f64 = uitofp <vscale x 2 x i16> %g.i16 to <vscale x 2 x double>
+  %r.f64 = uitofp <vscale x 2 x i16> %r.i16 to <vscale x 2 x double>
+  %a.f64 = uitofp <vscale x 2 x i16> %a.i16 to <vscale x 2 x double>
+  %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+  %g.mul.f64 = fmul <vscale x 2 x double> %g.f64, %common.term
+  %r.mul.f64 = fmul <vscale x 2 x double> %r.f64, %reversed
+  %a.mul.f64 = fmul <vscale x 2 x double> %a.f64, %common.term
+  %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+  %fadd.g.f64 = fadd <vscale x 2 x double> %acc.g.f64, %g.mul.f64
+  %fadd.r.f64 = fadd <vscale x 2 x double> %acc.r.f64, %r.mul.f64
+  %fadd.a.f64 = fadd <vscale x 2 x double> %acc.a.f64, %a.mul.f64
+  %iv.next = add nuw i64 %iv, %stride
+  %ec = icmp eq i64 %iv.next, 2048
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  %b.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.b.f64)
+  store double %b.acc, ptr %dst
+  %g.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.g.f64)
+  %g.f64.gep = getelementptr double, ptr %dst, i64 1
+  store double %g.acc, ptr %g.f64.gep
+  %r.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.r.f64)
+  %r.f64.gep = getelementptr double, ptr %dst, i64 2
+  store double %r.acc, ptr %r.f64.gep
+  %a.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.a.f64)
+  %a.f64.gep = getelementptr double, ptr %dst, i64 3
+  store double %a.acc, ptr %a.f64.gep
+  ret void
+}
+
+;; No reduction, so we cannot safely reorder the lanes.
+define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double_no_rdx(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double_no_rdx:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    index z7.d, #0, #4
+; CHECK-NEXT:    mov z3.d, #0xffffffffffff0000
+; CHECK-NEXT:    mov z5.d, #0xffffffffffff0001
+; CHECK-NEXT:    mov x9, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT:    mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT:    movi v0.2d, #0000000000000000
+; CHECK-NEXT:    mov z6.d, x9
+; CHECK-NEXT:    movi v1.2d, #0000000000000000
+; CHECK-NEXT:    mov x8, xzr
+; CHECK-NEXT:    movi v2.2d, #0000000000000000
+; CHECK-NEXT:    ptrue p1.d
+; CHECK-NEXT:    add x9, x1, #8
+; CHECK-NEXT:    add z4.d, z7.d, z3.d
+; CHECK-NEXT:    movi v3.2d, #0000000000000000
+; CHECK-NEXT:    cntw x10
+; CHECK-NEXT:    add z5.d, z7.d, z5.d
+; CHECK-NEXT:    add z6.d, z7.d, z6.d
+; CHECK-NEXT:    rdvl x11, #2
+; CHECK-NEXT:    add z7.d, z7.d, z24.d
+; CHECK-NEXT:  .LBB10_1: // %loop
+; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    ld1h { z24.h }, p0/z, [x0]
+; CHECK-NEXT:    ld1d { z28.d }, p1/z, [x9, x8, lsl #3]
+; CHECK-NEXT:    sub x8, x8, x10
+; CHECK-NEXT:    cmn x8, #2048
+; CHECK-NEXT:    add x0, x0, x11
+; CHECK-NEXT:    tbl z25.h, { z24.h }, z4.h
+; CHECK-NEXT:    tbl z26.h, { z24.h }, z5.h
+; CHECK-NEXT:    tbl z27.h, { z24.h }, z6.h
+; CHECK-NEXT:    tbl z24.h, { z24.h }, z7.h
+; CHECK-NEXT:    rev z28.d, z28.d
+; CHECK-NEXT:    ucvtf z25.d, p1/m, z25.d
+; CHECK-NEXT:    ucvtf z26.d, p1/m, z26.d
+; CHECK-NEXT:    ucvtf z27.d, p1/m, z27.d
+; CHECK-NEXT:    ucvtf z24.d, p1/m, z24.d
+; CHECK-NEXT:    fmul z25.d, z25.d, z28.d
+; CHECK-NEXT:    fmul z26.d, z26.d, z28.d
+; CHECK-NEXT:    fmul z27.d, z27.d, z28.d
+; CHECK-NEXT:    fmul z24.d, z24.d, z28.d
+; CHECK-NEXT:    fadd z0.d, z0.d, z25.d
+; CHECK-NEXT:    fadd z1.d, z1.d, z26.d
+; CHECK-NEXT:    fadd z2.d, z2.d, z27.d
+; CHECK-NEXT:    fadd z3.d, z3.d, z24.d
+; CHECK-NEXT:    b.ne .LBB10_1
+; CHECK-NEXT:  // %bb.2: // %exit
+; CHECK-NEXT:    str z0, [x2]
+; CHECK-NEXT:    str z1, [x2, #1, mul vl]
+; CHECK-NEXT:    str z2, [x2, #2, mul vl]
+; CHECK-NEXT:    str z3, [x2, #3, mul vl]
+; CHECK-NEXT:    ret
+entry:
+  %vscale = tail call i64 @llvm.vscale.i64()
+  %stride = shl nuw nsw i64 %vscale, 2
+  %common.base = getelementptr double, ptr %src2, i64 1
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.b.f64, %loop ]
+  %acc.g.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.g.f64, %loop ]
+  %acc.r.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.r.f64, %loop ]
+  %acc.a.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.a.f64, %loop ]
+  %negated = mul i64 %iv, -1
+  %common.term.ptr = getelementptr inbounds nuw double, ptr %common.base, i64 %negated
+  %common.term = load <vscale x 2 x double>, ptr %common.term.ptr, align 8
+  %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> %common.term)
+  %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+  %bgra = call <vscale x 8 x i16> @llvm.masked.load(ptr %src.gep, <vscale x 8 x i1> %mask, <vscale x 8 x i16> zeroinitializer)
+  %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+  %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+  %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+  %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+  %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+  %b.f64 = uitofp <vscale x 2 x i16> %b.i16 to <vscale x 2 x double>
+  %g.f64 = uitofp <vscale x 2 x i16> %g.i16 to <vscale x 2 x double>
+  %r.f64 = uitofp <vscale x 2 x i16> %r.i16 to <vscale x 2 x double>
+  %a.f64 = uitofp <vscale x 2 x i16> %a.i16 to <vscale x 2 x double>
+  %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+  %g.mul.f64 = fmul <vscale x 2 x double> %g.f64, %reversed
+  %r.mul.f64 = fmul <vscale x 2 x double> %r.f64, %reversed
+  %a.mul.f64 = fmul <vscale x 2 x double> %a.f64, %reversed
+  %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+  %fadd.g.f64 = fadd <vscale x 2 x double> %acc.g.f64, %g.mul.f64
+  %fadd.r.f64 = fadd <vscale x 2 x double> %acc.r.f64, %r.mul.f64
+  %fadd.a.f64 = fadd <vscale x 2 x double> %acc.a.f64, %a.mul.f64
+  %iv.next = add nuw i64 %iv, %stride
+  %ec = icmp eq i64 %iv.next, 2048
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  store <vscale x 2 x double> %fadd.b.f64, ptr %dst
+  %g.f64.gep = getelementptr <vscale x 2 x double>, ptr %dst, i64 1
+  store <vscale x 2 x double> %fadd.g.f64, ptr %g.f64.gep
+  %r.f64.gep = getelementptr <vscale x 2 x double>, ptr %dst, i64 2
+  store <vscale x 2 x double> %fadd.r.f64, ptr %r.f64.gep
+  %a.f64.gep = getelementptr <vscale x 2 x double>, ptr %dst, i64 3
+  store <vscale x 2 x double> %fadd.a.f64, ptr %a.f64.gep
+  ret void
+}
+
+;; Skip folding the rev into the tbl because it has an extra use.
+define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double_extra_use(ptr %src, ptr %src2, ptr %dst, ptr %dst2, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double_extra_use:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    index z7.d, #0, #4
+; CHECK-NEXT:    mov z1.d, #0xffffffffffff0000
+; CHECK-NEXT:    mov z5.d, #0xffffffffffff0001
+; CHECK-NEXT:    mov x10, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT:    mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT:    movi v0.2d, #0000000000000000
+; CHECK-NEXT:    mov z6.d, x10
+; CHECK-NEXT:    movi v4.2d, #0000000000000000
+; CHECK-NEXT:    mov x8, xzr
+; CHECK-NEXT:    movi v3.2d, #0000000000000000
+; CHECK-NEXT:    ptrue p1.d
+; CHECK-NEXT:    mov x9, xzr
+; CHECK-NEXT:    add z2.d, z7.d, z1.d
+; CHECK-NEXT:    movi v1.2d, #0000000000000000
+; CHECK-NEXT:    add x10, x1, #8
+; CHECK-NEXT:    add z5.d, z7.d, z5.d
+; CHECK-NEXT:    add z6.d, z7.d, z6.d
+; CHECK-NEXT:    cntw x11
+; CHECK-NEXT:    add z7.d, z7.d, z24.d
+; CHECK-NEXT:    rdvl x12, #2
+; CHECK-NEXT:  .LBB11_1: // %loop
+; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    ld1d { z24.d }, p1/z, [x10, x8, lsl #3]
+; CHECK-NEXT:    sub x8, x8, x11
+; CHECK-NEXT:    add x9, x9, x11
+; CHECK-NEXT:    cmn x8, #2048
+; CHECK-NEXT:    rev z24.d, z24.d
+; CHECK-NEXT:    str z24, [x3]
+; CHECK-NEXT:    ld1h { z25.h }, p0/z, [x0]
+; CHECK-NEXT:    add x0, x0, x12
+; CHECK-NEXT:    tbl z26.h, { z25.h }, z2.h
+; CHECK-NEXT:    tbl z27.h, { z25.h }, z5.h
+; CHECK-NEXT:    tbl z28.h, { z25.h }, z6.h
+; CHECK-NEXT:    tbl z25.h, { z25.h }, z7.h
+; CHECK-NEXT:    ucvtf z26.d, p1/m, z26.d
+; CHECK-NEXT:    ucvtf z27.d, p1/m, z27.d
+; CHECK-NEXT:    ucvtf z28.d, p1/m, z28.d
+; CHECK-NEXT:    ucvtf z25.d, p1/m, z25.d
+; CHECK-NEXT:    fmul z26.d, z26.d, z24.d
+; CHECK-NEXT:    fmul z27.d, z27.d, z24.d
+; CHECK-NEXT:    fmul z28.d, z28.d, z24.d
+; CHECK-NEXT:    fmul z24.d, z25.d, z24.d
+; CHECK-NEXT:    fadd z0.d, z0.d, z26.d
+; CHECK-NEXT:    fadd z4.d, z4.d, z27.d
+; CHECK-NEXT:    fadd z3.d, z3.d, z28.d
+; CHECK-NEXT:    fadd z1.d, z1.d, z24.d
+; CHECK-NEXT:    b.ne .LBB11_1
+; CHECK-NEXT:  // %bb.2: // %exit
+; CHECK-NEXT:    faddv d0, p1, z0.d
+; CHECK-NEXT:    faddv d2, p1, z4.d
+; CHECK-NEXT:    faddv d3, p1, z3.d
+; CHECK-NEXT:    faddv d1, p1, z1.d
+; CHECK-NEXT:    mov v0.d[1], v2.d[0]
+; CHECK-NEXT:    mov v3.d[1], v1.d[0]
+; CHECK-NEXT:    stp q0, q3, [x2]
+; CHECK-NEXT:    ret
+entry:
+  %vscale = tail call i64 @llvm.vscale.i64()
+  %stride = shl nuw nsw i64 %vscale, 2
+  %common.base = getelementptr double, ptr %src2, i64 1
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.b.f64, %loop ]
+  %acc.g.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.g.f64, %loop ]
+  %acc.r.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.r.f64, %loop ]
+  %acc.a.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.a.f64, %loop ]
+  %negated = mul i64 %iv, -1
+  %common.term.ptr = getelementptr inbounds nuw double, ptr %common.base, i64 %negated
+  %common.term = load <vscale x 2 x double>, ptr %common.term.ptr, align 8
+  %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> %common.term)
+  %dst2.ptr = getelementptr inbounds nuw double, ptr %dst2, i64 %iv
+  store <vscale x 2 x double> %reversed, ptr %dst2, align 8
+  %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+  %bgra = call <vscale x 8 x i16> @llvm.masked.load(ptr %src.gep, <vscale x 8 x i1> %mask, <vscale x 8 x i16> zeroinitializer)
+  %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+  %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+  %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+  %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+  %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+  %b.f64 = uitofp <vscale x 2 x i16> %b.i16 to <vscale x 2 x double>
+  %g.f64 = uitofp <vscale x 2 x i16> %g.i16 to <vscale x 2 x double>
+  %r.f64 = uitofp <vscale x 2 x i16> %r.i16 to <vscale x 2 x double>
+  %a.f64 = uitofp <vscale x 2 x i16> %a.i16 to <vscale x 2 x double>
+  %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+  %g.mul.f64 = fmul <vscale x 2 x double> %g.f64, %reversed
+  %r.mul.f64 = fmul <vscale x 2 x double> %r.f64, %reversed
+  %a.mul.f64 = fmul <vscale x 2 x double> %a.f64, %reversed
+  %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+  %fadd.g.f64 = fadd <vscale x 2 x double> %acc.g.f64, %g.mul.f64
+  %fadd.r.f64 = fadd <vscale x 2 x double> %acc.r.f64, %r.mul.f64
+  %fadd.a.f64 = fadd <vscale x 2 x double> %acc.a.f64, %a.mul.f64
+  %iv.next = add nuw i64 %iv, %stride
+  %ec = icmp eq i64 %iv.next, 2048
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  %b.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.b.f64)
+  store double %b.acc, ptr %dst
+  %g.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.g.f64)
+  %g.f64.gep = getelementptr double, ptr %dst, i64 1
+  store double %g.acc, ptr %g.f64.gep
+  %r.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.r.f64)
+  %r.f64.gep = getelementptr double, ptr %dst, i64 2
+  store double %r.acc, ptr %r.f64.gep
+  %a.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.a.f64)
+  %a.f64.gep = getelementptr double, ptr %dst, i64 3
+  store double %a.acc, ptr %a.f64.gep
+  ret void
+}
+
 attributes #0 = { "target-features"="+sve" }
 attributes #1 = { "target-features"="+sve" vscale_range(1, 8) }

>From 23ba104ea8864ddad1dc68116f7bda0ea1c8209a Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 31 Jul 2026 10:45:12 +0000
Subject: [PATCH 2/4] Move loop into tbl optimization function

---
 llvm/lib/Target/AArch64/SVEShuffleOpts.cpp | 26 +++++++++++++---------
 1 file changed, 15 insertions(+), 11 deletions(-)

diff --git a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
index 9e87e4e41395b..1ce3e497940ff 100644
--- a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
+++ b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
@@ -157,7 +157,15 @@ static void evaluateDeinterleave(IntrinsicInst *I, DeinterleaveMap &Candidates,
 
 /// Given a map of deinterleaves to zext or uitofp casts, remove the operations
 /// and replace them with tbl shuffles.
-static void optimizeSVEDeinterleavedExtends(DeinterleaveMap Deinterleaves) {
+static bool optimizeSVEDeinterleavedExtends(Loop &L,
+                                            const AArch64TargetLowering &TL,
+                                            const DataLayout DL) {
+  DeinterleaveMap Deinterleaves;
+  for (auto *BB : L.blocks())
+    for (auto &I : *BB)
+      if (match(&I, m_Intrinsic<Intrinsic::vector_deinterleave4>(m_Value())))
+        evaluateDeinterleave(cast<IntrinsicInst>(&I), Deinterleaves, L, TL, DL);
+
   for (auto &[Deinterleave, Extends] : Deinterleaves) {
     VectorType *DestTy = cast<VectorType>(Extends[0]->getDestTy());
     VectorType *SrcTy = cast<VectorType>(Extends[0]->getSrcTy());
@@ -211,6 +219,9 @@ static void optimizeSVEDeinterleavedExtends(DeinterleaveMap Deinterleaves) {
       cast<Instruction>(U)->eraseFromParent();
     Deinterleave->eraseFromParent();
   }
+
+  return !Deinterleaves.empty();
+}
 }
 
 static bool processLoop(Loop &L, const AArch64Subtarget &ST, DataLayout DL) {
@@ -226,17 +237,10 @@ static bool processLoop(Loop &L, const AArch64Subtarget &ST, DataLayout DL) {
   const AArch64TargetLowering &TL = *ST.getTargetLowering();
   assert(DL.isLittleEndian() &&
          "Shuffle optimizations unsupported for big endian targets.");
-  DeinterleaveMap Candidates;
-  for (auto *BB : L.blocks())
-    for (auto &I : *BB)
-      if (match(&I, m_Intrinsic<Intrinsic::vector_deinterleave4>(m_Value())))
-        evaluateDeinterleave(cast<IntrinsicInst>(&I), Candidates, L, TL, DL);
-
-  if (Candidates.empty())
-    return false;
 
-  optimizeSVEDeinterleavedExtends(Candidates);
-  return true;
+  bool Changed = false;
+  Changed |= optimizeSVEDeinterleavedExtends(L, TL, DL);
+  return Changed;
 }
 
 namespace {

>From 36695ac1e5df15cae069b34559fc15e895473217 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 31 Jul 2026 10:46:46 +0000
Subject: [PATCH 3/4] Add match helpers for deinterleaving-extending tbls

---
 llvm/lib/Target/AArch64/SVEShuffleOpts.cpp | 58 ++++++++++++++++++++++
 1 file changed, 58 insertions(+)

diff --git a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
index 1ce3e497940ff..31d500a3b0427 100644
--- a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
+++ b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
@@ -222,6 +222,64 @@ static bool optimizeSVEDeinterleavedExtends(Loop &L,
 
   return !Deinterleaves.empty();
 }
+
+// Match a bitcasted tbl intrinsic, and bind the tbl along with the mask.
+template <typename TblT, typename MaskT>
+static auto m_Tbl(TblT &&Tbl, MaskT &&Mask) {
+  return m_OneUse(
+      m_BitCast(m_OneUse(m_Value(Tbl, m_Intrinsic<Intrinsic::aarch64_sve_tbl>(
+                                          m_Value(), m_Value(Mask))))));
+}
+
+// Match a tbl intrinsic whose result is converted to a floating point value.
+template <typename TblT, typename MaskT>
+static auto m_UIToFPTbl(TblT &&Tbl, MaskT &&Mask) {
+  return m_OneUse(m_UIToFP(m_Tbl(Tbl, Mask)));
+}
+
+// Match either of the above tbls, and recalculate the index of the
+// deinterleaved subvector. Bind the tbl and the index.
+template <typename T> struct deinterleaving_tbl_match {
+  T *&Tbl;
+  unsigned &Idx;
+
+  deinterleaving_tbl_match(T *&Tbl, unsigned &Idx) : Tbl(Tbl), Idx(Idx) {}
+
+  template <typename ITy> bool match(ITy *V) const {
+    // Match the tbl.
+    Value *Mask;
+    if (!PatternMatch::match(
+            V, m_CombineOr(m_Tbl(Tbl, Mask), m_UIToFPTbl(Tbl, Mask))))
+      return false;
+
+    // For a deinterleaving+extending tbl, we will have a known constant values
+    // for the starting index and the step.
+    const APInt *Start;
+    const APInt *Step;
+    if (!PatternMatch::match(
+            Mask, m_BitCast(m_Add(m_Mul(m_Intrinsic<Intrinsic::stepvector>(),
+                                        m_APInt(Step)),
+                                  m_APInt(Start)))))
+      return false;
+
+    unsigned SrcSize = Tbl->getType()->getScalarType()->getScalarSizeInBits();
+    unsigned ResSize = V->getType()->getScalarType()->getScalarSizeInBits();
+    // If the top bits are all ones, we know we're forcing an out-of-range
+    // index. With a deinterleave of 4, we should have 3 invalid indices for
+    // every valid one.
+    if (Start->countLeadingOnes() != ResSize - SrcSize)
+      return false;
+
+    // The start of the valid indices must be between 0 and 3, for the 4
+    // subvectors we're extracting.
+    Idx = Start->getZExtValue() & SrcSize - 1;
+    return Idx >= 0 && Idx < 4;
+  }
+};
+
+template <typename T> static auto m_DeinterleavingTbl(T *&Tbl, unsigned &Idx) {
+  return deinterleaving_tbl_match<T>(Tbl, Idx);
+}
 }
 
 static bool processLoop(Loop &L, const AArch64Subtarget &ST, DataLayout DL) {

>From eb3ff8d6a3d22708ad0892595c36cfa54f7cff8e Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 31 Jul 2026 10:48:24 +0000
Subject: [PATCH 4/4] Add optimization to fold in reverse

---
 llvm/lib/Target/AArch64/SVEShuffleOpts.cpp    | 171 ++++++++++++++++++
 .../CodeGen/AArch64/sve-tbl-folding-new-pm.ll |  53 +++++-
 .../CodeGen/AArch64/sve-tbl-folding-opts.ll   |  95 +++++-----
 3 files changed, 268 insertions(+), 51 deletions(-)

diff --git a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
index 31d500a3b0427..a2367cdad170f 100644
--- a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
+++ b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
@@ -66,6 +66,7 @@
 #include "AArch64.h"
 #include "AArch64Subtarget.h"
 #include "AArch64TargetMachine.h"
+#include "Utils/AArch64BaseInfo.h"
 #include "llvm/Analysis/AssumptionCache.h"
 #include "llvm/Analysis/LoopInfo.h"
 #include "llvm/Analysis/LoopPass.h"
@@ -280,6 +281,175 @@ template <typename T> struct deinterleaving_tbl_match {
 template <typename T> static auto m_DeinterleavingTbl(T *&Tbl, unsigned &Idx) {
   return deinterleaving_tbl_match<T>(Tbl, Idx);
 }
+
+/// We want to find a reverse used only by BinOps where the other term comes
+/// from one of the deinterleave-and-extend tbls we created before. If this
+/// BinOp is only used in a reduction operation in the loop (so the order of
+/// elements within it do not matter), then we can potentially fold the reverse
+/// into the tbl and remove the separate reverse operation.
+/// Something like the following (possibly repeated multiple times):
+///
+/// %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00),
+///                                          %entry ], [ %fadd.b.f64, %loop ]
+/// ...
+/// %rev.load = load <vscale x 2 x double>, ptr %rev.ptr
+/// %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(
+///                                            <vscale x 2 x double> %rev.load)
+/// %bgra = load <vscale x 8 x i16>, ptr %src.gep
+/// %stepvec = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+/// %stride = mul nuw <vscale x 2 x i64> %stepvec, splat (i64 4)
+/// %start = add nuw <vscale x 2 x i64> %stride, splat (i64 -65536)
+/// %bc.to = bitcast <vscale x 2 x i64> %start to <vscale x 8 x i16>
+/// %tbl = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(
+//                         <vscale x 8 x i16> %bgra, <vscale x 8 x i16> %bc.to)
+/// %bc.from = bitcast <vscale x 8 x i16> %tbl to <vscale x 2 x i64>
+/// %b.f64 = uitofp <vscale x 2 x i64> %bc.from to <vscale x 2 x double>
+/// %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+/// %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+/// ...
+/// %b.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double
+///                            0.000000e+00, <vscale x 2 x double> %fadd.b.f64)
+static bool foldAdjacentReversesIntoTbls(Loop &L,
+                                         const AArch64TargetLowering &TL,
+                                         const DataLayout DL) {
+  struct RevTblData {
+    Instruction *Tbl;
+    Use *RevUse;
+    unsigned Idx;
+  };
+
+  // Look for reverse intrinsics used with the results of tbl instructions.
+  SmallVector<RevTblData, 4> Tbls;
+  for (auto *BB : L.blocks())
+    for (auto &I : *BB) {
+      if (!match(&I, m_Intrinsic<Intrinsic::vector_reverse>(m_Value())))
+        continue;
+
+      // Check all uses of the reverse; if there's anything which doesn't
+      // match our expected patterns, then give up on it. The intention is
+      // to remove the reverse, so any remaining users would prevent that.
+      SmallVector<RevTblData, 4> RevUses;
+      for (Use &U : I.uses()) {
+        Instruction *UI = cast<Instruction>(U.getUser());
+
+        // Look for a tbl used to deinterleave and zero extend (and optionally
+        // convert to FP), the result of which is then used in a BinOp with
+        // the reverse.
+        Value *Tbl;
+        unsigned Idx;
+        if (!match(UI, m_OneUse(m_c_BinOp(m_DeinterleavingTbl(Tbl, Idx),
+                                          m_Specific(&I))))) {
+          RevUses.clear();
+          break;
+        }
+
+        // Check that the only use of the BinOp is part of a reduction such that
+        // the order of elements in the vector doesn't matter.
+        // TODO: Allow more operations in the chain.
+        // Look for 2 users -- a phi in the loop, and an outside
+        // reduction intrinsic.
+        Instruction *RdxUpdate = cast<Instruction>(UI->user_back());
+        SmallVector<User *, 2> Users(RdxUpdate->users());
+        if (!match(RdxUpdate, m_BinOp()) || Users.size() != 2) {
+          RevUses.clear();
+          break;
+        }
+
+        Instruction *Phi = cast<Instruction>(Users[0]);
+        Instruction *Reduce = cast<Instruction>(Users[1]);
+
+        // We're looking for an in-loop phi to confirm a reduction.
+        if (!L.contains(Phi))
+          std::swap(Phi, Reduce);
+
+        // If the in-loop user is not a header Phi, or isn't one of the
+        // operands for the RdxUpdate operation, or the reduction op isn't
+        // outside of the loop, abandon this reverse.
+        if (!isa<PHINode>(Phi) || Phi->getParent() != L.getHeader() ||
+            !is_contained(RdxUpdate->operands(), Phi) || L.contains(Reduce)) {
+          RevUses.clear();
+          break;
+        }
+
+        // The out-of-loop user may be an LCSSA phi; look through that.
+        if (isa<PHINode>(Reduce) && Reduce->getNumOperands() == 1 &&
+            Reduce->hasOneUser())
+          Reduce = Reduce->user_back();
+
+        // Make sure the outside user is a supported reduction operation, and if
+        // FP that reassociation is allowed.
+        if (!match(Reduce,
+                   m_CombineOr(
+                       m_Intrinsic<Intrinsic::vector_reduce_add>(),
+                       m_AllowReassoc(
+                           m_Intrinsic<Intrinsic::vector_reduce_fadd>())))) {
+          RevUses.clear();
+          break;
+        }
+        RevUses.push_back({cast<Instruction>(Tbl), &U, Idx});
+      }
+      append_range(Tbls, RevUses);
+    }
+
+  // Convert each candidate we found to perform the reverse in the tbl instead,
+  // then remove the reverse.
+  for (auto [Tbl, RevUse, Idx] : Tbls) {
+    Instruction *Rev = cast<Instruction>(RevUse->get());
+    VectorType *SrcTy = cast<VectorType>(Tbl->getType());
+    VectorType *DstTy = cast<VectorType>(Rev->getType());
+    unsigned SrcBits = SrcTy->getScalarSizeInBits();
+    unsigned DstBits = DstTy->getScalarSizeInBits();
+    VectorType *StepVecTy = VectorType::getInteger(DstTy);
+
+    // We need to create a new mask for the tbl, so that we effectively reverse
+    // the elements with the tbl. This means the resulting vector will be
+    // backwards compared to the original, which is why we check for reductions
+    // where we don't care about the order.
+    //
+    // Similar to the normal deinterleaving+extending mask, we will use out-of
+    // range indices to perform the zero extension. However, we need a negative
+    // stride starting from the highest group of elements. Since we're grouping
+    // by 4 (for now), we need to subtract 4 from the total source element count
+    // for the vector to get the start, then add the index of the extraction.
+    //
+    // The result should be the following:
+    // <EltCnt - 4 + Idx, 0xFFFF, 0xFFFF, 0xFFFF, EltCnt - 8 + Idx, 0xFFFF...>.
+    IRBuilder<> Builder(Tbl);
+
+    // Create and splat the starting value.
+    APInt Invalid = APInt::getAllOnes(DstBits);
+    APInt StartIdx = Invalid << SrcBits;
+    StartIdx -= (4 - Idx);
+    Value *EltCnt = Builder.CreateVScale(StepVecTy->getScalarType());
+    EltCnt = Builder.CreateNUWMul(
+        EltCnt, ConstantInt::get(EltCnt->getType(),
+                                 AArch64::SVEBitsPerBlock / SrcBits));
+    Value *StartVal = Builder.CreateNUWAdd(
+        EltCnt, ConstantInt::get(EltCnt->getType(), StartIdx));
+    StartVal =
+        Builder.CreateVectorSplat(StepVecTy->getElementCount(), StartVal);
+
+    // Create the negative stride.
+    Value *StepVector = Builder.CreateStepVector(StepVecTy);
+    Value *ScaledSteps =
+        Builder.CreateMul(StepVector, ConstantInt::get(StepVecTy, -4));
+
+    // Add the start to the stride, replace the old mask.
+    ScaledSteps = Builder.CreateNUWAdd(ScaledSteps, StartVal);
+    Value *RevExtMask = Builder.CreateBitCast(ScaledSteps, SrcTy);
+    Tbl->setOperand(1, RevExtMask);
+
+    // Skip the original reverse now that we've migrated it to the tbl.
+    RevUse->set(Rev->getOperand(0));
+
+    // And erase the reverse if that was the last use.
+    if (Rev->uses().empty()) {
+      LLVM_DEBUG(dbgs() << "SVETBLOPT: Erasing " << *Rev << "\n");
+      Rev->eraseFromParent();
+    }
+  }
+
+  return !Tbls.empty();
 }
 
 static bool processLoop(Loop &L, const AArch64Subtarget &ST, DataLayout DL) {
@@ -298,6 +468,7 @@ static bool processLoop(Loop &L, const AArch64Subtarget &ST, DataLayout DL) {
 
   bool Changed = false;
   Changed |= optimizeSVEDeinterleavedExtends(L, TL, DL);
+  Changed |= foldAdjacentReversesIntoTbls(L, TL, DL);
   return Changed;
 }
 
diff --git a/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll b/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
index d739dd36ef804..0c377257f747b 100644
--- a/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
+++ b/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
@@ -224,41 +224,76 @@ define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr
 ; CHECK-NEXT:    [[NEGATED:%.*]] = mul i64 [[IV]], -1
 ; CHECK-NEXT:    [[COMMON_TERM_PTR:%.*]] = getelementptr inbounds nuw double, ptr [[COMMON_BASE]], i64 [[NEGATED]]
 ; CHECK-NEXT:    [[COMMON_TERM:%.*]] = load <vscale x 2 x double>, ptr [[COMMON_TERM_PTR]], align 8
-; CHECK-NEXT:    [[REVERSED:%.*]] = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> [[COMMON_TERM]])
 ; CHECK-NEXT:    [[SRC_GEP:%.*]] = getelementptr inbounds nuw [4 x i16], ptr [[SRC]], i64 [[IV]]
 ; CHECK-NEXT:    [[BGRA:%.*]] = load <vscale x 8 x i16>, ptr [[SRC_GEP]], align 16
 ; CHECK-NEXT:    [[TMP0:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
 ; CHECK-NEXT:    [[TMP28:%.*]] = mul nuw <vscale x 2 x i64> [[TMP0]], splat (i64 4)
 ; CHECK-NEXT:    [[TMP34:%.*]] = add nuw <vscale x 2 x i64> [[TMP28]], splat (i64 -65536)
 ; CHECK-NEXT:    [[TMP3:%.*]] = bitcast <vscale x 2 x i64> [[TMP34]] to <vscale x 8 x i16>
-; CHECK-NEXT:    [[TMP4:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP3]])
+; CHECK-NEXT:    [[TMP16:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP29:%.*]] = mul nuw i64 [[TMP16]], 8
+; CHECK-NEXT:    [[TMP30:%.*]] = add nuw i64 [[TMP29]], -65540
+; CHECK-NEXT:    [[DOTSPLATINSERT5:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP30]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT6:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT5]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP31:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[TMP8:%.*]] = mul <vscale x 2 x i64> [[TMP31]], splat (i64 -4)
+; CHECK-NEXT:    [[TMP9:%.*]] = add nuw <vscale x 2 x i64> [[TMP8]], [[DOTSPLAT6]]
+; CHECK-NEXT:    [[TMP39:%.*]] = bitcast <vscale x 2 x i64> [[TMP9]] to <vscale x 8 x i16>
+; CHECK-NEXT:    [[TMP4:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP39]])
 ; CHECK-NEXT:    [[TMP5:%.*]] = bitcast <vscale x 8 x i16> [[TMP4]] to <vscale x 2 x i64>
 ; CHECK-NEXT:    [[TMP6:%.*]] = uitofp <vscale x 2 x i64> [[TMP5]] to <vscale x 2 x double>
 ; CHECK-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
 ; CHECK-NEXT:    [[TMP15:%.*]] = mul nuw <vscale x 2 x i64> [[TMP7]], splat (i64 4)
 ; CHECK-NEXT:    [[TMP41:%.*]] = add nuw <vscale x 2 x i64> [[TMP15]], splat (i64 -65535)
 ; CHECK-NEXT:    [[TMP10:%.*]] = bitcast <vscale x 2 x i64> [[TMP41]] to <vscale x 8 x i16>
-; CHECK-NEXT:    [[TMP11:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP10]])
+; CHECK-NEXT:    [[TMP40:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP42:%.*]] = mul nuw i64 [[TMP40]], 8
+; CHECK-NEXT:    [[TMP45:%.*]] = add nuw i64 [[TMP42]], -65539
+; CHECK-NEXT:    [[DOTSPLATINSERT3:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP45]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT4:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT3]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP54:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[TMP22:%.*]] = mul <vscale x 2 x i64> [[TMP54]], splat (i64 -4)
+; CHECK-NEXT:    [[TMP23:%.*]] = add nuw <vscale x 2 x i64> [[TMP22]], [[DOTSPLAT4]]
+; CHECK-NEXT:    [[TMP55:%.*]] = bitcast <vscale x 2 x i64> [[TMP23]] to <vscale x 8 x i16>
+; CHECK-NEXT:    [[TMP11:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP55]])
 ; CHECK-NEXT:    [[TMP12:%.*]] = bitcast <vscale x 8 x i16> [[TMP11]] to <vscale x 2 x i64>
 ; CHECK-NEXT:    [[TMP13:%.*]] = uitofp <vscale x 2 x i64> [[TMP12]] to <vscale x 2 x double>
 ; CHECK-NEXT:    [[TMP14:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
 ; CHECK-NEXT:    [[TMP52:%.*]] = mul nuw <vscale x 2 x i64> [[TMP14]], splat (i64 4)
 ; CHECK-NEXT:    [[TMP53:%.*]] = add nuw <vscale x 2 x i64> [[TMP52]], splat (i64 -65534)
 ; CHECK-NEXT:    [[TMP17:%.*]] = bitcast <vscale x 2 x i64> [[TMP53]] to <vscale x 8 x i16>
-; CHECK-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP17]])
+; CHECK-NEXT:    [[TMP32:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP33:%.*]] = mul nuw i64 [[TMP32]], 8
+; CHECK-NEXT:    [[TMP56:%.*]] = add nuw i64 [[TMP33]], -65538
+; CHECK-NEXT:    [[DOTSPLATINSERT1:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP56]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT2:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT1]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP35:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[TMP36:%.*]] = mul <vscale x 2 x i64> [[TMP35]], splat (i64 -4)
+; CHECK-NEXT:    [[TMP37:%.*]] = add nuw <vscale x 2 x i64> [[TMP36]], [[DOTSPLAT2]]
+; CHECK-NEXT:    [[TMP38:%.*]] = bitcast <vscale x 2 x i64> [[TMP37]] to <vscale x 8 x i16>
+; CHECK-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP38]])
 ; CHECK-NEXT:    [[TMP19:%.*]] = bitcast <vscale x 8 x i16> [[TMP18]] to <vscale x 2 x i64>
 ; CHECK-NEXT:    [[TMP20:%.*]] = uitofp <vscale x 2 x i64> [[TMP19]] to <vscale x 2 x double>
 ; CHECK-NEXT:    [[TMP21:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
 ; CHECK-NEXT:    [[TMP43:%.*]] = mul nuw <vscale x 2 x i64> [[TMP21]], splat (i64 4)
 ; CHECK-NEXT:    [[TMP44:%.*]] = add nuw <vscale x 2 x i64> [[TMP43]], splat (i64 -65533)
 ; CHECK-NEXT:    [[TMP24:%.*]] = bitcast <vscale x 2 x i64> [[TMP44]] to <vscale x 8 x i16>
-; CHECK-NEXT:    [[TMP25:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP24]])
+; CHECK-NEXT:    [[TMP46:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP47:%.*]] = mul nuw i64 [[TMP46]], 8
+; CHECK-NEXT:    [[TMP48:%.*]] = add nuw i64 [[TMP47]], -65537
+; CHECK-NEXT:    [[DOTSPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP48]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP49:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[TMP50:%.*]] = mul <vscale x 2 x i64> [[TMP49]], splat (i64 -4)
+; CHECK-NEXT:    [[TMP51:%.*]] = add nuw <vscale x 2 x i64> [[TMP50]], [[DOTSPLAT]]
+; CHECK-NEXT:    [[TMP57:%.*]] = bitcast <vscale x 2 x i64> [[TMP51]] to <vscale x 8 x i16>
+; CHECK-NEXT:    [[TMP25:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP57]])
 ; CHECK-NEXT:    [[TMP26:%.*]] = bitcast <vscale x 8 x i16> [[TMP25]] to <vscale x 2 x i64>
 ; CHECK-NEXT:    [[TMP27:%.*]] = uitofp <vscale x 2 x i64> [[TMP26]] to <vscale x 2 x double>
-; CHECK-NEXT:    [[B_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP6]], [[REVERSED]]
-; CHECK-NEXT:    [[G_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP13]], [[REVERSED]]
-; CHECK-NEXT:    [[R_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP20]], [[REVERSED]]
-; CHECK-NEXT:    [[A_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP27]], [[REVERSED]]
+; CHECK-NEXT:    [[B_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP6]], [[COMMON_TERM]]
+; CHECK-NEXT:    [[G_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP13]], [[COMMON_TERM]]
+; CHECK-NEXT:    [[R_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP20]], [[COMMON_TERM]]
+; CHECK-NEXT:    [[A_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP27]], [[COMMON_TERM]]
 ; CHECK-NEXT:    [[FADD_B_F64]] = fadd <vscale x 2 x double> [[ACC_B_F64]], [[B_MUL_F64]]
 ; CHECK-NEXT:    [[FADD_G_F64]] = fadd <vscale x 2 x double> [[ACC_G_F64]], [[G_MUL_F64]]
 ; CHECK-NEXT:    [[FADD_R_F64]] = fadd <vscale x 2 x double> [[ACC_R_F64]], [[R_MUL_F64]]
diff --git a/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll b/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
index fa3df9f39a4ad..b7970a9f38369 100644
--- a/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
+++ b/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
@@ -641,25 +641,33 @@ exit:
 define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
 ; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double:
 ; CHECK:       // %bb.0: // %entry
-; CHECK-NEXT:    index z7.d, #0, #4
-; CHECK-NEXT:    mov z2.d, #0xffffffffffff0000
-; CHECK-NEXT:    mov z5.d, #0xffffffffffff0001
-; CHECK-NEXT:    mov x9, #-65534 // =0xffffffffffff0002
-; CHECK-NEXT:    mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT:    cnth x9
+; CHECK-NEXT:    index z7.d, #0, #-4
 ; CHECK-NEXT:    movi v0.2d, #0000000000000000
-; CHECK-NEXT:    mov z6.d, x9
-; CHECK-NEXT:    movi v3.2d, #0000000000000000
-; CHECK-NEXT:    mov x8, xzr
-; CHECK-NEXT:    movi v1.2d, #0000000000000000
+; CHECK-NEXT:    sub x10, x9, #16, lsl #12 // =65536
+; CHECK-NEXT:    sub x11, x9, #16, lsl #12 // =65536
+; CHECK-NEXT:    movi v4.2d, #0000000000000000
+; CHECK-NEXT:    sub x10, x10, #4
+; CHECK-NEXT:    sub x11, x11, #3
+; CHECK-NEXT:    movi v2.2d, #0000000000000000
+; CHECK-NEXT:    mov z1.d, x10
+; CHECK-NEXT:    sub x10, x9, #16, lsl #12 // =65536
+; CHECK-NEXT:    sub x9, x9, #16, lsl #12 // =65536
+; CHECK-NEXT:    sub x10, x10, #2
+; CHECK-NEXT:    sub x9, x9, #1
+; CHECK-NEXT:    mov z5.d, x11
+; CHECK-NEXT:    mov z6.d, x10
+; CHECK-NEXT:    mov z24.d, x9
 ; CHECK-NEXT:    ptrue p1.d
+; CHECK-NEXT:    add z3.d, z7.d, z1.d
+; CHECK-NEXT:    movi v1.2d, #0000000000000000
+; CHECK-NEXT:    mov x8, xzr
+; CHECK-NEXT:    add z5.d, z7.d, z5.d
 ; CHECK-NEXT:    add x9, x1, #8
-; CHECK-NEXT:    add z4.d, z7.d, z2.d
-; CHECK-NEXT:    movi v2.2d, #0000000000000000
 ; CHECK-NEXT:    cntw x10
-; CHECK-NEXT:    add z5.d, z7.d, z5.d
 ; CHECK-NEXT:    add z6.d, z7.d, z6.d
-; CHECK-NEXT:    rdvl x11, #2
 ; CHECK-NEXT:    add z7.d, z7.d, z24.d
+; CHECK-NEXT:    rdvl x11, #2
 ; CHECK-NEXT:  .LBB8_1: // %loop
 ; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
 ; CHECK-NEXT:    ld1h { z24.h }, p0/z, [x0]
@@ -667,11 +675,10 @@ define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr
 ; CHECK-NEXT:    sub x8, x8, x10
 ; CHECK-NEXT:    cmn x8, #2048
 ; CHECK-NEXT:    add x0, x0, x11
-; CHECK-NEXT:    tbl z25.h, { z24.h }, z4.h
+; CHECK-NEXT:    tbl z25.h, { z24.h }, z3.h
 ; CHECK-NEXT:    tbl z26.h, { z24.h }, z5.h
 ; CHECK-NEXT:    tbl z27.h, { z24.h }, z6.h
 ; CHECK-NEXT:    tbl z24.h, { z24.h }, z7.h
-; CHECK-NEXT:    rev z28.d, z28.d
 ; CHECK-NEXT:    ucvtf z25.d, p1/m, z25.d
 ; CHECK-NEXT:    ucvtf z26.d, p1/m, z26.d
 ; CHECK-NEXT:    ucvtf z27.d, p1/m, z27.d
@@ -681,18 +688,18 @@ define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr
 ; CHECK-NEXT:    fmul z27.d, z27.d, z28.d
 ; CHECK-NEXT:    fmul z24.d, z24.d, z28.d
 ; CHECK-NEXT:    fadd z0.d, z0.d, z25.d
-; CHECK-NEXT:    fadd z3.d, z3.d, z26.d
-; CHECK-NEXT:    fadd z1.d, z1.d, z27.d
-; CHECK-NEXT:    fadd z2.d, z2.d, z24.d
+; CHECK-NEXT:    fadd z4.d, z4.d, z26.d
+; CHECK-NEXT:    fadd z2.d, z2.d, z27.d
+; CHECK-NEXT:    fadd z1.d, z1.d, z24.d
 ; CHECK-NEXT:    b.ne .LBB8_1
 ; CHECK-NEXT:  // %bb.2: // %exit
 ; CHECK-NEXT:    faddv d0, p1, z0.d
-; CHECK-NEXT:    faddv d3, p1, z3.d
-; CHECK-NEXT:    faddv d1, p1, z1.d
+; CHECK-NEXT:    faddv d3, p1, z4.d
 ; CHECK-NEXT:    faddv d2, p1, z2.d
+; CHECK-NEXT:    faddv d1, p1, z1.d
 ; CHECK-NEXT:    mov v0.d[1], v3.d[0]
-; CHECK-NEXT:    mov v1.d[1], v2.d[0]
-; CHECK-NEXT:    stp q0, q1, [x2]
+; CHECK-NEXT:    mov v2.d[1], v1.d[0]
+; CHECK-NEXT:    stp q0, q2, [x2]
 ; CHECK-NEXT:    ret
 entry:
   %vscale = tail call i64 @llvm.vscale.i64()
@@ -751,25 +758,30 @@ exit:
 define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_partial_coverage(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
 ; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_partial_coverage:
 ; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    cnth x9
 ; CHECK-NEXT:    index z7.d, #0, #4
-; CHECK-NEXT:    mov z2.d, #0xffffffffffff0000
-; CHECK-NEXT:    mov z5.d, #0xffffffffffff0001
-; CHECK-NEXT:    mov x9, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT:    index z5.d, #0, #-4
+; CHECK-NEXT:    sub x10, x9, #16, lsl #12 // =65536
+; CHECK-NEXT:    sub x9, x9, #16, lsl #12 // =65536
+; CHECK-NEXT:    mov z6.d, #0xffffffffffff0001
+; CHECK-NEXT:    sub x10, x10, #4
+; CHECK-NEXT:    sub x9, x9, #2
 ; CHECK-NEXT:    mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT:    mov z4.d, x10
+; CHECK-NEXT:    mov z25.d, x9
 ; CHECK-NEXT:    movi v0.2d, #0000000000000000
-; CHECK-NEXT:    mov z6.d, x9
 ; CHECK-NEXT:    movi v3.2d, #0000000000000000
-; CHECK-NEXT:    mov x8, xzr
+; CHECK-NEXT:    movi v2.2d, #0000000000000000
 ; CHECK-NEXT:    movi v1.2d, #0000000000000000
+; CHECK-NEXT:    add z6.d, z7.d, z6.d
 ; CHECK-NEXT:    ptrue p1.d
+; CHECK-NEXT:    mov x8, xzr
+; CHECK-NEXT:    add z4.d, z5.d, z4.d
+; CHECK-NEXT:    add z5.d, z5.d, z25.d
+; CHECK-NEXT:    add z7.d, z7.d, z24.d
 ; CHECK-NEXT:    add x9, x1, #8
-; CHECK-NEXT:    add z4.d, z7.d, z2.d
-; CHECK-NEXT:    movi v2.2d, #0000000000000000
 ; CHECK-NEXT:    cntw x10
-; CHECK-NEXT:    add z5.d, z7.d, z5.d
-; CHECK-NEXT:    add z6.d, z7.d, z6.d
 ; CHECK-NEXT:    rdvl x11, #2
-; CHECK-NEXT:    add z7.d, z7.d, z24.d
 ; CHECK-NEXT:  .LBB9_1: // %loop
 ; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
 ; CHECK-NEXT:    ld1h { z24.h }, p0/z, [x0]
@@ -778,31 +790,30 @@ define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_partial_cov
 ; CHECK-NEXT:    cmn x8, #2048
 ; CHECK-NEXT:    add x0, x0, x11
 ; CHECK-NEXT:    tbl z25.h, { z24.h }, z4.h
-; CHECK-NEXT:    tbl z26.h, { z24.h }, z5.h
-; CHECK-NEXT:    tbl z27.h, { z24.h }, z6.h
+; CHECK-NEXT:    tbl z26.h, { z24.h }, z6.h
+; CHECK-NEXT:    tbl z27.h, { z24.h }, z5.h
 ; CHECK-NEXT:    tbl z24.h, { z24.h }, z7.h
-; CHECK-NEXT:    rev z29.d, z28.d
 ; CHECK-NEXT:    ucvtf z25.d, p1/m, z25.d
 ; CHECK-NEXT:    ucvtf z26.d, p1/m, z26.d
 ; CHECK-NEXT:    ucvtf z27.d, p1/m, z27.d
 ; CHECK-NEXT:    ucvtf z24.d, p1/m, z24.d
-; CHECK-NEXT:    fmul z25.d, z25.d, z29.d
+; CHECK-NEXT:    fmul z25.d, z25.d, z28.d
 ; CHECK-NEXT:    fmul z26.d, z26.d, z28.d
-; CHECK-NEXT:    fmul z27.d, z27.d, z29.d
+; CHECK-NEXT:    fmul z27.d, z27.d, z28.d
 ; CHECK-NEXT:    fmul z24.d, z24.d, z28.d
 ; CHECK-NEXT:    fadd z0.d, z0.d, z25.d
 ; CHECK-NEXT:    fadd z3.d, z3.d, z26.d
-; CHECK-NEXT:    fadd z1.d, z1.d, z27.d
-; CHECK-NEXT:    fadd z2.d, z2.d, z24.d
+; CHECK-NEXT:    fadd z2.d, z2.d, z27.d
+; CHECK-NEXT:    fadd z1.d, z1.d, z24.d
 ; CHECK-NEXT:    b.ne .LBB9_1
 ; CHECK-NEXT:  // %bb.2: // %exit
 ; CHECK-NEXT:    faddv d0, p1, z0.d
 ; CHECK-NEXT:    faddv d3, p1, z3.d
-; CHECK-NEXT:    faddv d1, p1, z1.d
 ; CHECK-NEXT:    faddv d2, p1, z2.d
+; CHECK-NEXT:    faddv d1, p1, z1.d
 ; CHECK-NEXT:    mov v0.d[1], v3.d[0]
-; CHECK-NEXT:    mov v1.d[1], v2.d[0]
-; CHECK-NEXT:    stp q0, q1, [x2]
+; CHECK-NEXT:    mov v2.d[1], v1.d[0]
+; CHECK-NEXT:    stp q0, q2, [x2]
 ; CHECK-NEXT:    ret
 entry:
   %vscale = tail call i64 @llvm.vscale.i64()



More information about the llvm-commits mailing list