[llvm] [AArch64] SVE Shuffleopt: merge reduction reverse into tbl (PR #206047)
Graham Hunter via llvm-commits
llvm-commits at lists.llvm.org
Fri Jul 31 06:38:54 PDT 2026
https://github.com/huntergr-arm updated https://github.com/llvm/llvm-project/pull/206047
>From 4bacf904f7ea1de2ee6fe5ae1812191b2519eee9 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Thu, 25 Jun 2026 14:52:48 +0000
Subject: [PATCH 1/4] Add tests with a reverse on the other operand of the
extend user
---
.../CodeGen/AArch64/sve-tbl-folding-new-pm.ll | 131 ++++++
.../CodeGen/AArch64/sve-tbl-folding-opts.ll | 440 ++++++++++++++++++
2 files changed, 571 insertions(+)
diff --git a/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll b/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
index 6a533a2419255..d739dd36ef804 100644
--- a/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
+++ b/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
@@ -207,4 +207,135 @@ exit:
ret void
}
+define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(
+; CHECK-SAME: ptr [[SRC:%.*]], ptr [[SRC2:%.*]], ptr [[DST:%.*]], <vscale x 8 x i1> [[MASK:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[VSCALE:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[STRIDE:%.*]] = shl nuw nsw i64 [[VSCALE]], 2
+; CHECK-NEXT: [[COMMON_BASE:%.*]] = getelementptr double, ptr [[SRC2]], i64 1
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[ACC_B_F64:%.*]] = phi <vscale x 2 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[FADD_B_F64:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[ACC_G_F64:%.*]] = phi <vscale x 2 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[FADD_G_F64:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[ACC_R_F64:%.*]] = phi <vscale x 2 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[FADD_R_F64:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[ACC_A_F64:%.*]] = phi <vscale x 2 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[FADD_A_F64:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[NEGATED:%.*]] = mul i64 [[IV]], -1
+; CHECK-NEXT: [[COMMON_TERM_PTR:%.*]] = getelementptr inbounds nuw double, ptr [[COMMON_BASE]], i64 [[NEGATED]]
+; CHECK-NEXT: [[COMMON_TERM:%.*]] = load <vscale x 2 x double>, ptr [[COMMON_TERM_PTR]], align 8
+; CHECK-NEXT: [[REVERSED:%.*]] = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> [[COMMON_TERM]])
+; CHECK-NEXT: [[SRC_GEP:%.*]] = getelementptr inbounds nuw [4 x i16], ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT: [[BGRA:%.*]] = load <vscale x 8 x i16>, ptr [[SRC_GEP]], align 16
+; CHECK-NEXT: [[TMP0:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[TMP28:%.*]] = mul nuw <vscale x 2 x i64> [[TMP0]], splat (i64 4)
+; CHECK-NEXT: [[TMP34:%.*]] = add nuw <vscale x 2 x i64> [[TMP28]], splat (i64 -65536)
+; CHECK-NEXT: [[TMP3:%.*]] = bitcast <vscale x 2 x i64> [[TMP34]] to <vscale x 8 x i16>
+; CHECK-NEXT: [[TMP4:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP3]])
+; CHECK-NEXT: [[TMP5:%.*]] = bitcast <vscale x 8 x i16> [[TMP4]] to <vscale x 2 x i64>
+; CHECK-NEXT: [[TMP6:%.*]] = uitofp <vscale x 2 x i64> [[TMP5]] to <vscale x 2 x double>
+; CHECK-NEXT: [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[TMP15:%.*]] = mul nuw <vscale x 2 x i64> [[TMP7]], splat (i64 4)
+; CHECK-NEXT: [[TMP41:%.*]] = add nuw <vscale x 2 x i64> [[TMP15]], splat (i64 -65535)
+; CHECK-NEXT: [[TMP10:%.*]] = bitcast <vscale x 2 x i64> [[TMP41]] to <vscale x 8 x i16>
+; CHECK-NEXT: [[TMP11:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP10]])
+; CHECK-NEXT: [[TMP12:%.*]] = bitcast <vscale x 8 x i16> [[TMP11]] to <vscale x 2 x i64>
+; CHECK-NEXT: [[TMP13:%.*]] = uitofp <vscale x 2 x i64> [[TMP12]] to <vscale x 2 x double>
+; CHECK-NEXT: [[TMP14:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[TMP52:%.*]] = mul nuw <vscale x 2 x i64> [[TMP14]], splat (i64 4)
+; CHECK-NEXT: [[TMP53:%.*]] = add nuw <vscale x 2 x i64> [[TMP52]], splat (i64 -65534)
+; CHECK-NEXT: [[TMP17:%.*]] = bitcast <vscale x 2 x i64> [[TMP53]] to <vscale x 8 x i16>
+; CHECK-NEXT: [[TMP18:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP17]])
+; CHECK-NEXT: [[TMP19:%.*]] = bitcast <vscale x 8 x i16> [[TMP18]] to <vscale x 2 x i64>
+; CHECK-NEXT: [[TMP20:%.*]] = uitofp <vscale x 2 x i64> [[TMP19]] to <vscale x 2 x double>
+; CHECK-NEXT: [[TMP21:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[TMP43:%.*]] = mul nuw <vscale x 2 x i64> [[TMP21]], splat (i64 4)
+; CHECK-NEXT: [[TMP44:%.*]] = add nuw <vscale x 2 x i64> [[TMP43]], splat (i64 -65533)
+; CHECK-NEXT: [[TMP24:%.*]] = bitcast <vscale x 2 x i64> [[TMP44]] to <vscale x 8 x i16>
+; CHECK-NEXT: [[TMP25:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP24]])
+; CHECK-NEXT: [[TMP26:%.*]] = bitcast <vscale x 8 x i16> [[TMP25]] to <vscale x 2 x i64>
+; CHECK-NEXT: [[TMP27:%.*]] = uitofp <vscale x 2 x i64> [[TMP26]] to <vscale x 2 x double>
+; CHECK-NEXT: [[B_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP6]], [[REVERSED]]
+; CHECK-NEXT: [[G_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP13]], [[REVERSED]]
+; CHECK-NEXT: [[R_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP20]], [[REVERSED]]
+; CHECK-NEXT: [[A_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP27]], [[REVERSED]]
+; CHECK-NEXT: [[FADD_B_F64]] = fadd <vscale x 2 x double> [[ACC_B_F64]], [[B_MUL_F64]]
+; CHECK-NEXT: [[FADD_G_F64]] = fadd <vscale x 2 x double> [[ACC_G_F64]], [[G_MUL_F64]]
+; CHECK-NEXT: [[FADD_R_F64]] = fadd <vscale x 2 x double> [[ACC_R_F64]], [[R_MUL_F64]]
+; CHECK-NEXT: [[FADD_A_F64]] = fadd <vscale x 2 x double> [[ACC_A_F64]], [[A_MUL_F64]]
+; CHECK-NEXT: [[IV_NEXT]] = add nuw i64 [[IV]], [[STRIDE]]
+; CHECK-NEXT: [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 2048
+; CHECK-NEXT: br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[FADD_B_F64_LCSSA:%.*]] = phi <vscale x 2 x double> [ [[FADD_B_F64]], %[[LOOP]] ]
+; CHECK-NEXT: [[FADD_G_F64_LCSSA:%.*]] = phi <vscale x 2 x double> [ [[FADD_G_F64]], %[[LOOP]] ]
+; CHECK-NEXT: [[FADD_R_F64_LCSSA:%.*]] = phi <vscale x 2 x double> [ [[FADD_R_F64]], %[[LOOP]] ]
+; CHECK-NEXT: [[FADD_A_F64_LCSSA:%.*]] = phi <vscale x 2 x double> [ [[FADD_A_F64]], %[[LOOP]] ]
+; CHECK-NEXT: [[B_ACC:%.*]] = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> [[FADD_B_F64_LCSSA]])
+; CHECK-NEXT: store double [[B_ACC]], ptr [[DST]], align 8
+; CHECK-NEXT: [[G_ACC:%.*]] = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> [[FADD_G_F64_LCSSA]])
+; CHECK-NEXT: [[G_F64_GEP:%.*]] = getelementptr double, ptr [[DST]], i64 1
+; CHECK-NEXT: store double [[G_ACC]], ptr [[G_F64_GEP]], align 8
+; CHECK-NEXT: [[R_ACC:%.*]] = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> [[FADD_R_F64_LCSSA]])
+; CHECK-NEXT: [[R_F64_GEP:%.*]] = getelementptr double, ptr [[DST]], i64 2
+; CHECK-NEXT: store double [[R_ACC]], ptr [[R_F64_GEP]], align 8
+; CHECK-NEXT: [[A_ACC:%.*]] = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> [[FADD_A_F64_LCSSA]])
+; CHECK-NEXT: [[A_F64_GEP:%.*]] = getelementptr double, ptr [[DST]], i64 3
+; CHECK-NEXT: store double [[A_ACC]], ptr [[A_F64_GEP]], align 8
+; CHECK-NEXT: ret void
+;
+entry:
+ %vscale = tail call i64 @llvm.vscale.i64()
+ %stride = shl nuw nsw i64 %vscale, 2
+ %common.base = getelementptr double, ptr %src2, i64 1
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.b.f64, %loop ]
+ %acc.g.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.g.f64, %loop ]
+ %acc.r.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.r.f64, %loop ]
+ %acc.a.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.a.f64, %loop ]
+ %negated = mul i64 %iv, -1
+ %common.term.ptr = getelementptr inbounds nuw double, ptr %common.base, i64 %negated
+ %common.term = load <vscale x 2 x double>, ptr %common.term.ptr, align 8
+ %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> %common.term)
+ %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+ %bgra = load <vscale x 8 x i16>, ptr %src.gep, align 16
+ %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+ %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+ %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+ %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+ %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+ %b.f64 = uitofp <vscale x 2 x i16> %b.i16 to <vscale x 2 x double>
+ %g.f64 = uitofp <vscale x 2 x i16> %g.i16 to <vscale x 2 x double>
+ %r.f64 = uitofp <vscale x 2 x i16> %r.i16 to <vscale x 2 x double>
+ %a.f64 = uitofp <vscale x 2 x i16> %a.i16 to <vscale x 2 x double>
+ %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+ %g.mul.f64 = fmul <vscale x 2 x double> %g.f64, %reversed
+ %r.mul.f64 = fmul <vscale x 2 x double> %r.f64, %reversed
+ %a.mul.f64 = fmul <vscale x 2 x double> %a.f64, %reversed
+ %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+ %fadd.g.f64 = fadd <vscale x 2 x double> %acc.g.f64, %g.mul.f64
+ %fadd.r.f64 = fadd <vscale x 2 x double> %acc.r.f64, %r.mul.f64
+ %fadd.a.f64 = fadd <vscale x 2 x double> %acc.a.f64, %a.mul.f64
+ %iv.next = add nuw i64 %iv, %stride
+ %ec = icmp eq i64 %iv.next, 2048
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ %b.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.b.f64)
+ store double %b.acc, ptr %dst
+ %g.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.g.f64)
+ %g.f64.gep = getelementptr double, ptr %dst, i64 1
+ store double %g.acc, ptr %g.f64.gep
+ %r.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.r.f64)
+ %r.f64.gep = getelementptr double, ptr %dst, i64 2
+ store double %r.acc, ptr %r.f64.gep
+ %a.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.a.f64)
+ %a.f64.gep = getelementptr double, ptr %dst, i64 3
+ store double %a.acc, ptr %a.f64.gep
+ ret void
+}
+
attributes #0 = { "target-features"="+sve" }
diff --git a/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll b/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
index e101489c564c8..fa3df9f39a4ad 100644
--- a/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
+++ b/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
@@ -638,5 +638,445 @@ exit:
ret void
}
+define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: index z7.d, #0, #4
+; CHECK-NEXT: mov z2.d, #0xffffffffffff0000
+; CHECK-NEXT: mov z5.d, #0xffffffffffff0001
+; CHECK-NEXT: mov x9, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT: mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT: movi v0.2d, #0000000000000000
+; CHECK-NEXT: mov z6.d, x9
+; CHECK-NEXT: movi v3.2d, #0000000000000000
+; CHECK-NEXT: mov x8, xzr
+; CHECK-NEXT: movi v1.2d, #0000000000000000
+; CHECK-NEXT: ptrue p1.d
+; CHECK-NEXT: add x9, x1, #8
+; CHECK-NEXT: add z4.d, z7.d, z2.d
+; CHECK-NEXT: movi v2.2d, #0000000000000000
+; CHECK-NEXT: cntw x10
+; CHECK-NEXT: add z5.d, z7.d, z5.d
+; CHECK-NEXT: add z6.d, z7.d, z6.d
+; CHECK-NEXT: rdvl x11, #2
+; CHECK-NEXT: add z7.d, z7.d, z24.d
+; CHECK-NEXT: .LBB8_1: // %loop
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: ld1h { z24.h }, p0/z, [x0]
+; CHECK-NEXT: ld1d { z28.d }, p1/z, [x9, x8, lsl #3]
+; CHECK-NEXT: sub x8, x8, x10
+; CHECK-NEXT: cmn x8, #2048
+; CHECK-NEXT: add x0, x0, x11
+; CHECK-NEXT: tbl z25.h, { z24.h }, z4.h
+; CHECK-NEXT: tbl z26.h, { z24.h }, z5.h
+; CHECK-NEXT: tbl z27.h, { z24.h }, z6.h
+; CHECK-NEXT: tbl z24.h, { z24.h }, z7.h
+; CHECK-NEXT: rev z28.d, z28.d
+; CHECK-NEXT: ucvtf z25.d, p1/m, z25.d
+; CHECK-NEXT: ucvtf z26.d, p1/m, z26.d
+; CHECK-NEXT: ucvtf z27.d, p1/m, z27.d
+; CHECK-NEXT: ucvtf z24.d, p1/m, z24.d
+; CHECK-NEXT: fmul z25.d, z25.d, z28.d
+; CHECK-NEXT: fmul z26.d, z26.d, z28.d
+; CHECK-NEXT: fmul z27.d, z27.d, z28.d
+; CHECK-NEXT: fmul z24.d, z24.d, z28.d
+; CHECK-NEXT: fadd z0.d, z0.d, z25.d
+; CHECK-NEXT: fadd z3.d, z3.d, z26.d
+; CHECK-NEXT: fadd z1.d, z1.d, z27.d
+; CHECK-NEXT: fadd z2.d, z2.d, z24.d
+; CHECK-NEXT: b.ne .LBB8_1
+; CHECK-NEXT: // %bb.2: // %exit
+; CHECK-NEXT: faddv d0, p1, z0.d
+; CHECK-NEXT: faddv d3, p1, z3.d
+; CHECK-NEXT: faddv d1, p1, z1.d
+; CHECK-NEXT: faddv d2, p1, z2.d
+; CHECK-NEXT: mov v0.d[1], v3.d[0]
+; CHECK-NEXT: mov v1.d[1], v2.d[0]
+; CHECK-NEXT: stp q0, q1, [x2]
+; CHECK-NEXT: ret
+entry:
+ %vscale = tail call i64 @llvm.vscale.i64()
+ %stride = shl nuw nsw i64 %vscale, 2
+ %common.base = getelementptr double, ptr %src2, i64 1
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.b.f64, %loop ]
+ %acc.g.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.g.f64, %loop ]
+ %acc.r.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.r.f64, %loop ]
+ %acc.a.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.a.f64, %loop ]
+ %negated = mul i64 %iv, -1
+ %common.term.ptr = getelementptr inbounds nuw double, ptr %common.base, i64 %negated
+ %common.term = load <vscale x 2 x double>, ptr %common.term.ptr, align 8
+ %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> %common.term)
+ %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+ %bgra = call <vscale x 8 x i16> @llvm.masked.load(ptr %src.gep, <vscale x 8 x i1> %mask, <vscale x 8 x i16> zeroinitializer)
+ %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+ %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+ %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+ %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+ %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+ %b.f64 = uitofp <vscale x 2 x i16> %b.i16 to <vscale x 2 x double>
+ %g.f64 = uitofp <vscale x 2 x i16> %g.i16 to <vscale x 2 x double>
+ %r.f64 = uitofp <vscale x 2 x i16> %r.i16 to <vscale x 2 x double>
+ %a.f64 = uitofp <vscale x 2 x i16> %a.i16 to <vscale x 2 x double>
+ %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+ %g.mul.f64 = fmul <vscale x 2 x double> %g.f64, %reversed
+ %r.mul.f64 = fmul <vscale x 2 x double> %r.f64, %reversed
+ %a.mul.f64 = fmul <vscale x 2 x double> %a.f64, %reversed
+ %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+ %fadd.g.f64 = fadd <vscale x 2 x double> %acc.g.f64, %g.mul.f64
+ %fadd.r.f64 = fadd <vscale x 2 x double> %acc.r.f64, %r.mul.f64
+ %fadd.a.f64 = fadd <vscale x 2 x double> %acc.a.f64, %a.mul.f64
+ %iv.next = add nuw i64 %iv, %stride
+ %ec = icmp eq i64 %iv.next, 2048
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ %b.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.b.f64)
+ store double %b.acc, ptr %dst
+ %g.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.g.f64)
+ %g.f64.gep = getelementptr double, ptr %dst, i64 1
+ store double %g.acc, ptr %g.f64.gep
+ %r.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.r.f64)
+ %r.f64.gep = getelementptr double, ptr %dst, i64 2
+ store double %r.acc, ptr %r.f64.gep
+ %a.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.a.f64)
+ %a.f64.gep = getelementptr double, ptr %dst, i64 3
+ store double %a.acc, ptr %a.f64.gep
+ ret void
+}
+
+define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_partial_coverage(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_partial_coverage:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: index z7.d, #0, #4
+; CHECK-NEXT: mov z2.d, #0xffffffffffff0000
+; CHECK-NEXT: mov z5.d, #0xffffffffffff0001
+; CHECK-NEXT: mov x9, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT: mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT: movi v0.2d, #0000000000000000
+; CHECK-NEXT: mov z6.d, x9
+; CHECK-NEXT: movi v3.2d, #0000000000000000
+; CHECK-NEXT: mov x8, xzr
+; CHECK-NEXT: movi v1.2d, #0000000000000000
+; CHECK-NEXT: ptrue p1.d
+; CHECK-NEXT: add x9, x1, #8
+; CHECK-NEXT: add z4.d, z7.d, z2.d
+; CHECK-NEXT: movi v2.2d, #0000000000000000
+; CHECK-NEXT: cntw x10
+; CHECK-NEXT: add z5.d, z7.d, z5.d
+; CHECK-NEXT: add z6.d, z7.d, z6.d
+; CHECK-NEXT: rdvl x11, #2
+; CHECK-NEXT: add z7.d, z7.d, z24.d
+; CHECK-NEXT: .LBB9_1: // %loop
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: ld1h { z24.h }, p0/z, [x0]
+; CHECK-NEXT: ld1d { z28.d }, p1/z, [x9, x8, lsl #3]
+; CHECK-NEXT: sub x8, x8, x10
+; CHECK-NEXT: cmn x8, #2048
+; CHECK-NEXT: add x0, x0, x11
+; CHECK-NEXT: tbl z25.h, { z24.h }, z4.h
+; CHECK-NEXT: tbl z26.h, { z24.h }, z5.h
+; CHECK-NEXT: tbl z27.h, { z24.h }, z6.h
+; CHECK-NEXT: tbl z24.h, { z24.h }, z7.h
+; CHECK-NEXT: rev z29.d, z28.d
+; CHECK-NEXT: ucvtf z25.d, p1/m, z25.d
+; CHECK-NEXT: ucvtf z26.d, p1/m, z26.d
+; CHECK-NEXT: ucvtf z27.d, p1/m, z27.d
+; CHECK-NEXT: ucvtf z24.d, p1/m, z24.d
+; CHECK-NEXT: fmul z25.d, z25.d, z29.d
+; CHECK-NEXT: fmul z26.d, z26.d, z28.d
+; CHECK-NEXT: fmul z27.d, z27.d, z29.d
+; CHECK-NEXT: fmul z24.d, z24.d, z28.d
+; CHECK-NEXT: fadd z0.d, z0.d, z25.d
+; CHECK-NEXT: fadd z3.d, z3.d, z26.d
+; CHECK-NEXT: fadd z1.d, z1.d, z27.d
+; CHECK-NEXT: fadd z2.d, z2.d, z24.d
+; CHECK-NEXT: b.ne .LBB9_1
+; CHECK-NEXT: // %bb.2: // %exit
+; CHECK-NEXT: faddv d0, p1, z0.d
+; CHECK-NEXT: faddv d3, p1, z3.d
+; CHECK-NEXT: faddv d1, p1, z1.d
+; CHECK-NEXT: faddv d2, p1, z2.d
+; CHECK-NEXT: mov v0.d[1], v3.d[0]
+; CHECK-NEXT: mov v1.d[1], v2.d[0]
+; CHECK-NEXT: stp q0, q1, [x2]
+; CHECK-NEXT: ret
+entry:
+ %vscale = tail call i64 @llvm.vscale.i64()
+ %stride = shl nuw nsw i64 %vscale, 2
+ %common.base = getelementptr double, ptr %src2, i64 1
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.b.f64, %loop ]
+ %acc.g.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.g.f64, %loop ]
+ %acc.r.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.r.f64, %loop ]
+ %acc.a.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.a.f64, %loop ]
+ %negated = mul i64 %iv, -1
+ %common.term.ptr = getelementptr inbounds nuw double, ptr %common.base, i64 %negated
+ %common.term = load <vscale x 2 x double>, ptr %common.term.ptr, align 8
+ %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> %common.term)
+ %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+ %bgra = call <vscale x 8 x i16> @llvm.masked.load(ptr %src.gep, <vscale x 8 x i1> %mask, <vscale x 8 x i16> zeroinitializer)
+ %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+ %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+ %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+ %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+ %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+ %b.f64 = uitofp <vscale x 2 x i16> %b.i16 to <vscale x 2 x double>
+ %g.f64 = uitofp <vscale x 2 x i16> %g.i16 to <vscale x 2 x double>
+ %r.f64 = uitofp <vscale x 2 x i16> %r.i16 to <vscale x 2 x double>
+ %a.f64 = uitofp <vscale x 2 x i16> %a.i16 to <vscale x 2 x double>
+ %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+ %g.mul.f64 = fmul <vscale x 2 x double> %g.f64, %common.term
+ %r.mul.f64 = fmul <vscale x 2 x double> %r.f64, %reversed
+ %a.mul.f64 = fmul <vscale x 2 x double> %a.f64, %common.term
+ %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+ %fadd.g.f64 = fadd <vscale x 2 x double> %acc.g.f64, %g.mul.f64
+ %fadd.r.f64 = fadd <vscale x 2 x double> %acc.r.f64, %r.mul.f64
+ %fadd.a.f64 = fadd <vscale x 2 x double> %acc.a.f64, %a.mul.f64
+ %iv.next = add nuw i64 %iv, %stride
+ %ec = icmp eq i64 %iv.next, 2048
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ %b.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.b.f64)
+ store double %b.acc, ptr %dst
+ %g.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.g.f64)
+ %g.f64.gep = getelementptr double, ptr %dst, i64 1
+ store double %g.acc, ptr %g.f64.gep
+ %r.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.r.f64)
+ %r.f64.gep = getelementptr double, ptr %dst, i64 2
+ store double %r.acc, ptr %r.f64.gep
+ %a.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.a.f64)
+ %a.f64.gep = getelementptr double, ptr %dst, i64 3
+ store double %a.acc, ptr %a.f64.gep
+ ret void
+}
+
+;; No reduction, so we cannot safely reorder the lanes.
+define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double_no_rdx(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double_no_rdx:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: index z7.d, #0, #4
+; CHECK-NEXT: mov z3.d, #0xffffffffffff0000
+; CHECK-NEXT: mov z5.d, #0xffffffffffff0001
+; CHECK-NEXT: mov x9, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT: mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT: movi v0.2d, #0000000000000000
+; CHECK-NEXT: mov z6.d, x9
+; CHECK-NEXT: movi v1.2d, #0000000000000000
+; CHECK-NEXT: mov x8, xzr
+; CHECK-NEXT: movi v2.2d, #0000000000000000
+; CHECK-NEXT: ptrue p1.d
+; CHECK-NEXT: add x9, x1, #8
+; CHECK-NEXT: add z4.d, z7.d, z3.d
+; CHECK-NEXT: movi v3.2d, #0000000000000000
+; CHECK-NEXT: cntw x10
+; CHECK-NEXT: add z5.d, z7.d, z5.d
+; CHECK-NEXT: add z6.d, z7.d, z6.d
+; CHECK-NEXT: rdvl x11, #2
+; CHECK-NEXT: add z7.d, z7.d, z24.d
+; CHECK-NEXT: .LBB10_1: // %loop
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: ld1h { z24.h }, p0/z, [x0]
+; CHECK-NEXT: ld1d { z28.d }, p1/z, [x9, x8, lsl #3]
+; CHECK-NEXT: sub x8, x8, x10
+; CHECK-NEXT: cmn x8, #2048
+; CHECK-NEXT: add x0, x0, x11
+; CHECK-NEXT: tbl z25.h, { z24.h }, z4.h
+; CHECK-NEXT: tbl z26.h, { z24.h }, z5.h
+; CHECK-NEXT: tbl z27.h, { z24.h }, z6.h
+; CHECK-NEXT: tbl z24.h, { z24.h }, z7.h
+; CHECK-NEXT: rev z28.d, z28.d
+; CHECK-NEXT: ucvtf z25.d, p1/m, z25.d
+; CHECK-NEXT: ucvtf z26.d, p1/m, z26.d
+; CHECK-NEXT: ucvtf z27.d, p1/m, z27.d
+; CHECK-NEXT: ucvtf z24.d, p1/m, z24.d
+; CHECK-NEXT: fmul z25.d, z25.d, z28.d
+; CHECK-NEXT: fmul z26.d, z26.d, z28.d
+; CHECK-NEXT: fmul z27.d, z27.d, z28.d
+; CHECK-NEXT: fmul z24.d, z24.d, z28.d
+; CHECK-NEXT: fadd z0.d, z0.d, z25.d
+; CHECK-NEXT: fadd z1.d, z1.d, z26.d
+; CHECK-NEXT: fadd z2.d, z2.d, z27.d
+; CHECK-NEXT: fadd z3.d, z3.d, z24.d
+; CHECK-NEXT: b.ne .LBB10_1
+; CHECK-NEXT: // %bb.2: // %exit
+; CHECK-NEXT: str z0, [x2]
+; CHECK-NEXT: str z1, [x2, #1, mul vl]
+; CHECK-NEXT: str z2, [x2, #2, mul vl]
+; CHECK-NEXT: str z3, [x2, #3, mul vl]
+; CHECK-NEXT: ret
+entry:
+ %vscale = tail call i64 @llvm.vscale.i64()
+ %stride = shl nuw nsw i64 %vscale, 2
+ %common.base = getelementptr double, ptr %src2, i64 1
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.b.f64, %loop ]
+ %acc.g.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.g.f64, %loop ]
+ %acc.r.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.r.f64, %loop ]
+ %acc.a.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.a.f64, %loop ]
+ %negated = mul i64 %iv, -1
+ %common.term.ptr = getelementptr inbounds nuw double, ptr %common.base, i64 %negated
+ %common.term = load <vscale x 2 x double>, ptr %common.term.ptr, align 8
+ %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> %common.term)
+ %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+ %bgra = call <vscale x 8 x i16> @llvm.masked.load(ptr %src.gep, <vscale x 8 x i1> %mask, <vscale x 8 x i16> zeroinitializer)
+ %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+ %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+ %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+ %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+ %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+ %b.f64 = uitofp <vscale x 2 x i16> %b.i16 to <vscale x 2 x double>
+ %g.f64 = uitofp <vscale x 2 x i16> %g.i16 to <vscale x 2 x double>
+ %r.f64 = uitofp <vscale x 2 x i16> %r.i16 to <vscale x 2 x double>
+ %a.f64 = uitofp <vscale x 2 x i16> %a.i16 to <vscale x 2 x double>
+ %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+ %g.mul.f64 = fmul <vscale x 2 x double> %g.f64, %reversed
+ %r.mul.f64 = fmul <vscale x 2 x double> %r.f64, %reversed
+ %a.mul.f64 = fmul <vscale x 2 x double> %a.f64, %reversed
+ %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+ %fadd.g.f64 = fadd <vscale x 2 x double> %acc.g.f64, %g.mul.f64
+ %fadd.r.f64 = fadd <vscale x 2 x double> %acc.r.f64, %r.mul.f64
+ %fadd.a.f64 = fadd <vscale x 2 x double> %acc.a.f64, %a.mul.f64
+ %iv.next = add nuw i64 %iv, %stride
+ %ec = icmp eq i64 %iv.next, 2048
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ store <vscale x 2 x double> %fadd.b.f64, ptr %dst
+ %g.f64.gep = getelementptr <vscale x 2 x double>, ptr %dst, i64 1
+ store <vscale x 2 x double> %fadd.g.f64, ptr %g.f64.gep
+ %r.f64.gep = getelementptr <vscale x 2 x double>, ptr %dst, i64 2
+ store <vscale x 2 x double> %fadd.r.f64, ptr %r.f64.gep
+ %a.f64.gep = getelementptr <vscale x 2 x double>, ptr %dst, i64 3
+ store <vscale x 2 x double> %fadd.a.f64, ptr %a.f64.gep
+ ret void
+}
+
+;; Skip folding the rev into the tbl because it has an extra use.
+define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double_extra_use(ptr %src, ptr %src2, ptr %dst, ptr %dst2, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double_extra_use:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: index z7.d, #0, #4
+; CHECK-NEXT: mov z1.d, #0xffffffffffff0000
+; CHECK-NEXT: mov z5.d, #0xffffffffffff0001
+; CHECK-NEXT: mov x10, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT: mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT: movi v0.2d, #0000000000000000
+; CHECK-NEXT: mov z6.d, x10
+; CHECK-NEXT: movi v4.2d, #0000000000000000
+; CHECK-NEXT: mov x8, xzr
+; CHECK-NEXT: movi v3.2d, #0000000000000000
+; CHECK-NEXT: ptrue p1.d
+; CHECK-NEXT: mov x9, xzr
+; CHECK-NEXT: add z2.d, z7.d, z1.d
+; CHECK-NEXT: movi v1.2d, #0000000000000000
+; CHECK-NEXT: add x10, x1, #8
+; CHECK-NEXT: add z5.d, z7.d, z5.d
+; CHECK-NEXT: add z6.d, z7.d, z6.d
+; CHECK-NEXT: cntw x11
+; CHECK-NEXT: add z7.d, z7.d, z24.d
+; CHECK-NEXT: rdvl x12, #2
+; CHECK-NEXT: .LBB11_1: // %loop
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: ld1d { z24.d }, p1/z, [x10, x8, lsl #3]
+; CHECK-NEXT: sub x8, x8, x11
+; CHECK-NEXT: add x9, x9, x11
+; CHECK-NEXT: cmn x8, #2048
+; CHECK-NEXT: rev z24.d, z24.d
+; CHECK-NEXT: str z24, [x3]
+; CHECK-NEXT: ld1h { z25.h }, p0/z, [x0]
+; CHECK-NEXT: add x0, x0, x12
+; CHECK-NEXT: tbl z26.h, { z25.h }, z2.h
+; CHECK-NEXT: tbl z27.h, { z25.h }, z5.h
+; CHECK-NEXT: tbl z28.h, { z25.h }, z6.h
+; CHECK-NEXT: tbl z25.h, { z25.h }, z7.h
+; CHECK-NEXT: ucvtf z26.d, p1/m, z26.d
+; CHECK-NEXT: ucvtf z27.d, p1/m, z27.d
+; CHECK-NEXT: ucvtf z28.d, p1/m, z28.d
+; CHECK-NEXT: ucvtf z25.d, p1/m, z25.d
+; CHECK-NEXT: fmul z26.d, z26.d, z24.d
+; CHECK-NEXT: fmul z27.d, z27.d, z24.d
+; CHECK-NEXT: fmul z28.d, z28.d, z24.d
+; CHECK-NEXT: fmul z24.d, z25.d, z24.d
+; CHECK-NEXT: fadd z0.d, z0.d, z26.d
+; CHECK-NEXT: fadd z4.d, z4.d, z27.d
+; CHECK-NEXT: fadd z3.d, z3.d, z28.d
+; CHECK-NEXT: fadd z1.d, z1.d, z24.d
+; CHECK-NEXT: b.ne .LBB11_1
+; CHECK-NEXT: // %bb.2: // %exit
+; CHECK-NEXT: faddv d0, p1, z0.d
+; CHECK-NEXT: faddv d2, p1, z4.d
+; CHECK-NEXT: faddv d3, p1, z3.d
+; CHECK-NEXT: faddv d1, p1, z1.d
+; CHECK-NEXT: mov v0.d[1], v2.d[0]
+; CHECK-NEXT: mov v3.d[1], v1.d[0]
+; CHECK-NEXT: stp q0, q3, [x2]
+; CHECK-NEXT: ret
+entry:
+ %vscale = tail call i64 @llvm.vscale.i64()
+ %stride = shl nuw nsw i64 %vscale, 2
+ %common.base = getelementptr double, ptr %src2, i64 1
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.b.f64, %loop ]
+ %acc.g.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.g.f64, %loop ]
+ %acc.r.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.r.f64, %loop ]
+ %acc.a.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00), %entry ], [ %fadd.a.f64, %loop ]
+ %negated = mul i64 %iv, -1
+ %common.term.ptr = getelementptr inbounds nuw double, ptr %common.base, i64 %negated
+ %common.term = load <vscale x 2 x double>, ptr %common.term.ptr, align 8
+ %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> %common.term)
+ %dst2.ptr = getelementptr inbounds nuw double, ptr %dst2, i64 %iv
+ store <vscale x 2 x double> %reversed, ptr %dst2, align 8
+ %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+ %bgra = call <vscale x 8 x i16> @llvm.masked.load(ptr %src.gep, <vscale x 8 x i1> %mask, <vscale x 8 x i16> zeroinitializer)
+ %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+ %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+ %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+ %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+ %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+ %b.f64 = uitofp <vscale x 2 x i16> %b.i16 to <vscale x 2 x double>
+ %g.f64 = uitofp <vscale x 2 x i16> %g.i16 to <vscale x 2 x double>
+ %r.f64 = uitofp <vscale x 2 x i16> %r.i16 to <vscale x 2 x double>
+ %a.f64 = uitofp <vscale x 2 x i16> %a.i16 to <vscale x 2 x double>
+ %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+ %g.mul.f64 = fmul <vscale x 2 x double> %g.f64, %reversed
+ %r.mul.f64 = fmul <vscale x 2 x double> %r.f64, %reversed
+ %a.mul.f64 = fmul <vscale x 2 x double> %a.f64, %reversed
+ %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+ %fadd.g.f64 = fadd <vscale x 2 x double> %acc.g.f64, %g.mul.f64
+ %fadd.r.f64 = fadd <vscale x 2 x double> %acc.r.f64, %r.mul.f64
+ %fadd.a.f64 = fadd <vscale x 2 x double> %acc.a.f64, %a.mul.f64
+ %iv.next = add nuw i64 %iv, %stride
+ %ec = icmp eq i64 %iv.next, 2048
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ %b.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.b.f64)
+ store double %b.acc, ptr %dst
+ %g.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.g.f64)
+ %g.f64.gep = getelementptr double, ptr %dst, i64 1
+ store double %g.acc, ptr %g.f64.gep
+ %r.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.r.f64)
+ %r.f64.gep = getelementptr double, ptr %dst, i64 2
+ store double %r.acc, ptr %r.f64.gep
+ %a.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double 0.000000e+00, <vscale x 2 x double> %fadd.a.f64)
+ %a.f64.gep = getelementptr double, ptr %dst, i64 3
+ store double %a.acc, ptr %a.f64.gep
+ ret void
+}
+
attributes #0 = { "target-features"="+sve" }
attributes #1 = { "target-features"="+sve" vscale_range(1, 8) }
>From 23ba104ea8864ddad1dc68116f7bda0ea1c8209a Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 31 Jul 2026 10:45:12 +0000
Subject: [PATCH 2/4] Move loop into tbl optimization function
---
llvm/lib/Target/AArch64/SVEShuffleOpts.cpp | 26 +++++++++++++---------
1 file changed, 15 insertions(+), 11 deletions(-)
diff --git a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
index 9e87e4e41395b..1ce3e497940ff 100644
--- a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
+++ b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
@@ -157,7 +157,15 @@ static void evaluateDeinterleave(IntrinsicInst *I, DeinterleaveMap &Candidates,
/// Given a map of deinterleaves to zext or uitofp casts, remove the operations
/// and replace them with tbl shuffles.
-static void optimizeSVEDeinterleavedExtends(DeinterleaveMap Deinterleaves) {
+static bool optimizeSVEDeinterleavedExtends(Loop &L,
+ const AArch64TargetLowering &TL,
+ const DataLayout DL) {
+ DeinterleaveMap Deinterleaves;
+ for (auto *BB : L.blocks())
+ for (auto &I : *BB)
+ if (match(&I, m_Intrinsic<Intrinsic::vector_deinterleave4>(m_Value())))
+ evaluateDeinterleave(cast<IntrinsicInst>(&I), Deinterleaves, L, TL, DL);
+
for (auto &[Deinterleave, Extends] : Deinterleaves) {
VectorType *DestTy = cast<VectorType>(Extends[0]->getDestTy());
VectorType *SrcTy = cast<VectorType>(Extends[0]->getSrcTy());
@@ -211,6 +219,9 @@ static void optimizeSVEDeinterleavedExtends(DeinterleaveMap Deinterleaves) {
cast<Instruction>(U)->eraseFromParent();
Deinterleave->eraseFromParent();
}
+
+ return !Deinterleaves.empty();
+}
}
static bool processLoop(Loop &L, const AArch64Subtarget &ST, DataLayout DL) {
@@ -226,17 +237,10 @@ static bool processLoop(Loop &L, const AArch64Subtarget &ST, DataLayout DL) {
const AArch64TargetLowering &TL = *ST.getTargetLowering();
assert(DL.isLittleEndian() &&
"Shuffle optimizations unsupported for big endian targets.");
- DeinterleaveMap Candidates;
- for (auto *BB : L.blocks())
- for (auto &I : *BB)
- if (match(&I, m_Intrinsic<Intrinsic::vector_deinterleave4>(m_Value())))
- evaluateDeinterleave(cast<IntrinsicInst>(&I), Candidates, L, TL, DL);
-
- if (Candidates.empty())
- return false;
- optimizeSVEDeinterleavedExtends(Candidates);
- return true;
+ bool Changed = false;
+ Changed |= optimizeSVEDeinterleavedExtends(L, TL, DL);
+ return Changed;
}
namespace {
>From 36695ac1e5df15cae069b34559fc15e895473217 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 31 Jul 2026 10:46:46 +0000
Subject: [PATCH 3/4] Add match helpers for deinterleaving-extending tbls
---
llvm/lib/Target/AArch64/SVEShuffleOpts.cpp | 58 ++++++++++++++++++++++
1 file changed, 58 insertions(+)
diff --git a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
index 1ce3e497940ff..31d500a3b0427 100644
--- a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
+++ b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
@@ -222,6 +222,64 @@ static bool optimizeSVEDeinterleavedExtends(Loop &L,
return !Deinterleaves.empty();
}
+
+// Match a bitcasted tbl intrinsic, and bind the tbl along with the mask.
+template <typename TblT, typename MaskT>
+static auto m_Tbl(TblT &&Tbl, MaskT &&Mask) {
+ return m_OneUse(
+ m_BitCast(m_OneUse(m_Value(Tbl, m_Intrinsic<Intrinsic::aarch64_sve_tbl>(
+ m_Value(), m_Value(Mask))))));
+}
+
+// Match a tbl intrinsic whose result is converted to a floating point value.
+template <typename TblT, typename MaskT>
+static auto m_UIToFPTbl(TblT &&Tbl, MaskT &&Mask) {
+ return m_OneUse(m_UIToFP(m_Tbl(Tbl, Mask)));
+}
+
+// Match either of the above tbls, and recalculate the index of the
+// deinterleaved subvector. Bind the tbl and the index.
+template <typename T> struct deinterleaving_tbl_match {
+ T *&Tbl;
+ unsigned &Idx;
+
+ deinterleaving_tbl_match(T *&Tbl, unsigned &Idx) : Tbl(Tbl), Idx(Idx) {}
+
+ template <typename ITy> bool match(ITy *V) const {
+ // Match the tbl.
+ Value *Mask;
+ if (!PatternMatch::match(
+ V, m_CombineOr(m_Tbl(Tbl, Mask), m_UIToFPTbl(Tbl, Mask))))
+ return false;
+
+ // For a deinterleaving+extending tbl, we will have a known constant values
+ // for the starting index and the step.
+ const APInt *Start;
+ const APInt *Step;
+ if (!PatternMatch::match(
+ Mask, m_BitCast(m_Add(m_Mul(m_Intrinsic<Intrinsic::stepvector>(),
+ m_APInt(Step)),
+ m_APInt(Start)))))
+ return false;
+
+ unsigned SrcSize = Tbl->getType()->getScalarType()->getScalarSizeInBits();
+ unsigned ResSize = V->getType()->getScalarType()->getScalarSizeInBits();
+ // If the top bits are all ones, we know we're forcing an out-of-range
+ // index. With a deinterleave of 4, we should have 3 invalid indices for
+ // every valid one.
+ if (Start->countLeadingOnes() != ResSize - SrcSize)
+ return false;
+
+ // The start of the valid indices must be between 0 and 3, for the 4
+ // subvectors we're extracting.
+ Idx = Start->getZExtValue() & SrcSize - 1;
+ return Idx >= 0 && Idx < 4;
+ }
+};
+
+template <typename T> static auto m_DeinterleavingTbl(T *&Tbl, unsigned &Idx) {
+ return deinterleaving_tbl_match<T>(Tbl, Idx);
+}
}
static bool processLoop(Loop &L, const AArch64Subtarget &ST, DataLayout DL) {
>From eb3ff8d6a3d22708ad0892595c36cfa54f7cff8e Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 31 Jul 2026 10:48:24 +0000
Subject: [PATCH 4/4] Add optimization to fold in reverse
---
llvm/lib/Target/AArch64/SVEShuffleOpts.cpp | 171 ++++++++++++++++++
.../CodeGen/AArch64/sve-tbl-folding-new-pm.ll | 53 +++++-
.../CodeGen/AArch64/sve-tbl-folding-opts.ll | 95 +++++-----
3 files changed, 268 insertions(+), 51 deletions(-)
diff --git a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
index 31d500a3b0427..a2367cdad170f 100644
--- a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
+++ b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
@@ -66,6 +66,7 @@
#include "AArch64.h"
#include "AArch64Subtarget.h"
#include "AArch64TargetMachine.h"
+#include "Utils/AArch64BaseInfo.h"
#include "llvm/Analysis/AssumptionCache.h"
#include "llvm/Analysis/LoopInfo.h"
#include "llvm/Analysis/LoopPass.h"
@@ -280,6 +281,175 @@ template <typename T> struct deinterleaving_tbl_match {
template <typename T> static auto m_DeinterleavingTbl(T *&Tbl, unsigned &Idx) {
return deinterleaving_tbl_match<T>(Tbl, Idx);
}
+
+/// We want to find a reverse used only by BinOps where the other term comes
+/// from one of the deinterleave-and-extend tbls we created before. If this
+/// BinOp is only used in a reduction operation in the loop (so the order of
+/// elements within it do not matter), then we can potentially fold the reverse
+/// into the tbl and remove the separate reverse operation.
+/// Something like the following (possibly repeated multiple times):
+///
+/// %acc.b.f64 = phi <vscale x 2 x double> [ splat(double 0.000000e+00),
+/// %entry ], [ %fadd.b.f64, %loop ]
+/// ...
+/// %rev.load = load <vscale x 2 x double>, ptr %rev.ptr
+/// %reversed = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(
+/// <vscale x 2 x double> %rev.load)
+/// %bgra = load <vscale x 8 x i16>, ptr %src.gep
+/// %stepvec = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+/// %stride = mul nuw <vscale x 2 x i64> %stepvec, splat (i64 4)
+/// %start = add nuw <vscale x 2 x i64> %stride, splat (i64 -65536)
+/// %bc.to = bitcast <vscale x 2 x i64> %start to <vscale x 8 x i16>
+/// %tbl = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(
+// <vscale x 8 x i16> %bgra, <vscale x 8 x i16> %bc.to)
+/// %bc.from = bitcast <vscale x 8 x i16> %tbl to <vscale x 2 x i64>
+/// %b.f64 = uitofp <vscale x 2 x i64> %bc.from to <vscale x 2 x double>
+/// %b.mul.f64 = fmul <vscale x 2 x double> %b.f64, %reversed
+/// %fadd.b.f64 = fadd <vscale x 2 x double> %acc.b.f64, %b.mul.f64
+/// ...
+/// %b.acc = call fast double @llvm.vector.reduce.fadd.nxv2f64(double
+/// 0.000000e+00, <vscale x 2 x double> %fadd.b.f64)
+static bool foldAdjacentReversesIntoTbls(Loop &L,
+ const AArch64TargetLowering &TL,
+ const DataLayout DL) {
+ struct RevTblData {
+ Instruction *Tbl;
+ Use *RevUse;
+ unsigned Idx;
+ };
+
+ // Look for reverse intrinsics used with the results of tbl instructions.
+ SmallVector<RevTblData, 4> Tbls;
+ for (auto *BB : L.blocks())
+ for (auto &I : *BB) {
+ if (!match(&I, m_Intrinsic<Intrinsic::vector_reverse>(m_Value())))
+ continue;
+
+ // Check all uses of the reverse; if there's anything which doesn't
+ // match our expected patterns, then give up on it. The intention is
+ // to remove the reverse, so any remaining users would prevent that.
+ SmallVector<RevTblData, 4> RevUses;
+ for (Use &U : I.uses()) {
+ Instruction *UI = cast<Instruction>(U.getUser());
+
+ // Look for a tbl used to deinterleave and zero extend (and optionally
+ // convert to FP), the result of which is then used in a BinOp with
+ // the reverse.
+ Value *Tbl;
+ unsigned Idx;
+ if (!match(UI, m_OneUse(m_c_BinOp(m_DeinterleavingTbl(Tbl, Idx),
+ m_Specific(&I))))) {
+ RevUses.clear();
+ break;
+ }
+
+ // Check that the only use of the BinOp is part of a reduction such that
+ // the order of elements in the vector doesn't matter.
+ // TODO: Allow more operations in the chain.
+ // Look for 2 users -- a phi in the loop, and an outside
+ // reduction intrinsic.
+ Instruction *RdxUpdate = cast<Instruction>(UI->user_back());
+ SmallVector<User *, 2> Users(RdxUpdate->users());
+ if (!match(RdxUpdate, m_BinOp()) || Users.size() != 2) {
+ RevUses.clear();
+ break;
+ }
+
+ Instruction *Phi = cast<Instruction>(Users[0]);
+ Instruction *Reduce = cast<Instruction>(Users[1]);
+
+ // We're looking for an in-loop phi to confirm a reduction.
+ if (!L.contains(Phi))
+ std::swap(Phi, Reduce);
+
+ // If the in-loop user is not a header Phi, or isn't one of the
+ // operands for the RdxUpdate operation, or the reduction op isn't
+ // outside of the loop, abandon this reverse.
+ if (!isa<PHINode>(Phi) || Phi->getParent() != L.getHeader() ||
+ !is_contained(RdxUpdate->operands(), Phi) || L.contains(Reduce)) {
+ RevUses.clear();
+ break;
+ }
+
+ // The out-of-loop user may be an LCSSA phi; look through that.
+ if (isa<PHINode>(Reduce) && Reduce->getNumOperands() == 1 &&
+ Reduce->hasOneUser())
+ Reduce = Reduce->user_back();
+
+ // Make sure the outside user is a supported reduction operation, and if
+ // FP that reassociation is allowed.
+ if (!match(Reduce,
+ m_CombineOr(
+ m_Intrinsic<Intrinsic::vector_reduce_add>(),
+ m_AllowReassoc(
+ m_Intrinsic<Intrinsic::vector_reduce_fadd>())))) {
+ RevUses.clear();
+ break;
+ }
+ RevUses.push_back({cast<Instruction>(Tbl), &U, Idx});
+ }
+ append_range(Tbls, RevUses);
+ }
+
+ // Convert each candidate we found to perform the reverse in the tbl instead,
+ // then remove the reverse.
+ for (auto [Tbl, RevUse, Idx] : Tbls) {
+ Instruction *Rev = cast<Instruction>(RevUse->get());
+ VectorType *SrcTy = cast<VectorType>(Tbl->getType());
+ VectorType *DstTy = cast<VectorType>(Rev->getType());
+ unsigned SrcBits = SrcTy->getScalarSizeInBits();
+ unsigned DstBits = DstTy->getScalarSizeInBits();
+ VectorType *StepVecTy = VectorType::getInteger(DstTy);
+
+ // We need to create a new mask for the tbl, so that we effectively reverse
+ // the elements with the tbl. This means the resulting vector will be
+ // backwards compared to the original, which is why we check for reductions
+ // where we don't care about the order.
+ //
+ // Similar to the normal deinterleaving+extending mask, we will use out-of
+ // range indices to perform the zero extension. However, we need a negative
+ // stride starting from the highest group of elements. Since we're grouping
+ // by 4 (for now), we need to subtract 4 from the total source element count
+ // for the vector to get the start, then add the index of the extraction.
+ //
+ // The result should be the following:
+ // <EltCnt - 4 + Idx, 0xFFFF, 0xFFFF, 0xFFFF, EltCnt - 8 + Idx, 0xFFFF...>.
+ IRBuilder<> Builder(Tbl);
+
+ // Create and splat the starting value.
+ APInt Invalid = APInt::getAllOnes(DstBits);
+ APInt StartIdx = Invalid << SrcBits;
+ StartIdx -= (4 - Idx);
+ Value *EltCnt = Builder.CreateVScale(StepVecTy->getScalarType());
+ EltCnt = Builder.CreateNUWMul(
+ EltCnt, ConstantInt::get(EltCnt->getType(),
+ AArch64::SVEBitsPerBlock / SrcBits));
+ Value *StartVal = Builder.CreateNUWAdd(
+ EltCnt, ConstantInt::get(EltCnt->getType(), StartIdx));
+ StartVal =
+ Builder.CreateVectorSplat(StepVecTy->getElementCount(), StartVal);
+
+ // Create the negative stride.
+ Value *StepVector = Builder.CreateStepVector(StepVecTy);
+ Value *ScaledSteps =
+ Builder.CreateMul(StepVector, ConstantInt::get(StepVecTy, -4));
+
+ // Add the start to the stride, replace the old mask.
+ ScaledSteps = Builder.CreateNUWAdd(ScaledSteps, StartVal);
+ Value *RevExtMask = Builder.CreateBitCast(ScaledSteps, SrcTy);
+ Tbl->setOperand(1, RevExtMask);
+
+ // Skip the original reverse now that we've migrated it to the tbl.
+ RevUse->set(Rev->getOperand(0));
+
+ // And erase the reverse if that was the last use.
+ if (Rev->uses().empty()) {
+ LLVM_DEBUG(dbgs() << "SVETBLOPT: Erasing " << *Rev << "\n");
+ Rev->eraseFromParent();
+ }
+ }
+
+ return !Tbls.empty();
}
static bool processLoop(Loop &L, const AArch64Subtarget &ST, DataLayout DL) {
@@ -298,6 +468,7 @@ static bool processLoop(Loop &L, const AArch64Subtarget &ST, DataLayout DL) {
bool Changed = false;
Changed |= optimizeSVEDeinterleavedExtends(L, TL, DL);
+ Changed |= foldAdjacentReversesIntoTbls(L, TL, DL);
return Changed;
}
diff --git a/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll b/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
index d739dd36ef804..0c377257f747b 100644
--- a/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
+++ b/llvm/test/CodeGen/AArch64/sve-tbl-folding-new-pm.ll
@@ -224,41 +224,76 @@ define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr
; CHECK-NEXT: [[NEGATED:%.*]] = mul i64 [[IV]], -1
; CHECK-NEXT: [[COMMON_TERM_PTR:%.*]] = getelementptr inbounds nuw double, ptr [[COMMON_BASE]], i64 [[NEGATED]]
; CHECK-NEXT: [[COMMON_TERM:%.*]] = load <vscale x 2 x double>, ptr [[COMMON_TERM_PTR]], align 8
-; CHECK-NEXT: [[REVERSED:%.*]] = call <vscale x 2 x double> @llvm.vector.reverse.nxv2f64(<vscale x 2 x double> [[COMMON_TERM]])
; CHECK-NEXT: [[SRC_GEP:%.*]] = getelementptr inbounds nuw [4 x i16], ptr [[SRC]], i64 [[IV]]
; CHECK-NEXT: [[BGRA:%.*]] = load <vscale x 8 x i16>, ptr [[SRC_GEP]], align 16
; CHECK-NEXT: [[TMP0:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
; CHECK-NEXT: [[TMP28:%.*]] = mul nuw <vscale x 2 x i64> [[TMP0]], splat (i64 4)
; CHECK-NEXT: [[TMP34:%.*]] = add nuw <vscale x 2 x i64> [[TMP28]], splat (i64 -65536)
; CHECK-NEXT: [[TMP3:%.*]] = bitcast <vscale x 2 x i64> [[TMP34]] to <vscale x 8 x i16>
-; CHECK-NEXT: [[TMP4:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP3]])
+; CHECK-NEXT: [[TMP16:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP29:%.*]] = mul nuw i64 [[TMP16]], 8
+; CHECK-NEXT: [[TMP30:%.*]] = add nuw i64 [[TMP29]], -65540
+; CHECK-NEXT: [[DOTSPLATINSERT5:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP30]], i64 0
+; CHECK-NEXT: [[DOTSPLAT6:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT5]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP31:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[TMP8:%.*]] = mul <vscale x 2 x i64> [[TMP31]], splat (i64 -4)
+; CHECK-NEXT: [[TMP9:%.*]] = add nuw <vscale x 2 x i64> [[TMP8]], [[DOTSPLAT6]]
+; CHECK-NEXT: [[TMP39:%.*]] = bitcast <vscale x 2 x i64> [[TMP9]] to <vscale x 8 x i16>
+; CHECK-NEXT: [[TMP4:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP39]])
; CHECK-NEXT: [[TMP5:%.*]] = bitcast <vscale x 8 x i16> [[TMP4]] to <vscale x 2 x i64>
; CHECK-NEXT: [[TMP6:%.*]] = uitofp <vscale x 2 x i64> [[TMP5]] to <vscale x 2 x double>
; CHECK-NEXT: [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
; CHECK-NEXT: [[TMP15:%.*]] = mul nuw <vscale x 2 x i64> [[TMP7]], splat (i64 4)
; CHECK-NEXT: [[TMP41:%.*]] = add nuw <vscale x 2 x i64> [[TMP15]], splat (i64 -65535)
; CHECK-NEXT: [[TMP10:%.*]] = bitcast <vscale x 2 x i64> [[TMP41]] to <vscale x 8 x i16>
-; CHECK-NEXT: [[TMP11:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP10]])
+; CHECK-NEXT: [[TMP40:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP42:%.*]] = mul nuw i64 [[TMP40]], 8
+; CHECK-NEXT: [[TMP45:%.*]] = add nuw i64 [[TMP42]], -65539
+; CHECK-NEXT: [[DOTSPLATINSERT3:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP45]], i64 0
+; CHECK-NEXT: [[DOTSPLAT4:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT3]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP54:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[TMP22:%.*]] = mul <vscale x 2 x i64> [[TMP54]], splat (i64 -4)
+; CHECK-NEXT: [[TMP23:%.*]] = add nuw <vscale x 2 x i64> [[TMP22]], [[DOTSPLAT4]]
+; CHECK-NEXT: [[TMP55:%.*]] = bitcast <vscale x 2 x i64> [[TMP23]] to <vscale x 8 x i16>
+; CHECK-NEXT: [[TMP11:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP55]])
; CHECK-NEXT: [[TMP12:%.*]] = bitcast <vscale x 8 x i16> [[TMP11]] to <vscale x 2 x i64>
; CHECK-NEXT: [[TMP13:%.*]] = uitofp <vscale x 2 x i64> [[TMP12]] to <vscale x 2 x double>
; CHECK-NEXT: [[TMP14:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
; CHECK-NEXT: [[TMP52:%.*]] = mul nuw <vscale x 2 x i64> [[TMP14]], splat (i64 4)
; CHECK-NEXT: [[TMP53:%.*]] = add nuw <vscale x 2 x i64> [[TMP52]], splat (i64 -65534)
; CHECK-NEXT: [[TMP17:%.*]] = bitcast <vscale x 2 x i64> [[TMP53]] to <vscale x 8 x i16>
-; CHECK-NEXT: [[TMP18:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP17]])
+; CHECK-NEXT: [[TMP32:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP33:%.*]] = mul nuw i64 [[TMP32]], 8
+; CHECK-NEXT: [[TMP56:%.*]] = add nuw i64 [[TMP33]], -65538
+; CHECK-NEXT: [[DOTSPLATINSERT1:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP56]], i64 0
+; CHECK-NEXT: [[DOTSPLAT2:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT1]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP35:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[TMP36:%.*]] = mul <vscale x 2 x i64> [[TMP35]], splat (i64 -4)
+; CHECK-NEXT: [[TMP37:%.*]] = add nuw <vscale x 2 x i64> [[TMP36]], [[DOTSPLAT2]]
+; CHECK-NEXT: [[TMP38:%.*]] = bitcast <vscale x 2 x i64> [[TMP37]] to <vscale x 8 x i16>
+; CHECK-NEXT: [[TMP18:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP38]])
; CHECK-NEXT: [[TMP19:%.*]] = bitcast <vscale x 8 x i16> [[TMP18]] to <vscale x 2 x i64>
; CHECK-NEXT: [[TMP20:%.*]] = uitofp <vscale x 2 x i64> [[TMP19]] to <vscale x 2 x double>
; CHECK-NEXT: [[TMP21:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
; CHECK-NEXT: [[TMP43:%.*]] = mul nuw <vscale x 2 x i64> [[TMP21]], splat (i64 4)
; CHECK-NEXT: [[TMP44:%.*]] = add nuw <vscale x 2 x i64> [[TMP43]], splat (i64 -65533)
; CHECK-NEXT: [[TMP24:%.*]] = bitcast <vscale x 2 x i64> [[TMP44]] to <vscale x 8 x i16>
-; CHECK-NEXT: [[TMP25:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP24]])
+; CHECK-NEXT: [[TMP46:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP47:%.*]] = mul nuw i64 [[TMP46]], 8
+; CHECK-NEXT: [[TMP48:%.*]] = add nuw i64 [[TMP47]], -65537
+; CHECK-NEXT: [[DOTSPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP48]], i64 0
+; CHECK-NEXT: [[DOTSPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP49:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[TMP50:%.*]] = mul <vscale x 2 x i64> [[TMP49]], splat (i64 -4)
+; CHECK-NEXT: [[TMP51:%.*]] = add nuw <vscale x 2 x i64> [[TMP50]], [[DOTSPLAT]]
+; CHECK-NEXT: [[TMP57:%.*]] = bitcast <vscale x 2 x i64> [[TMP51]] to <vscale x 8 x i16>
+; CHECK-NEXT: [[TMP25:%.*]] = call <vscale x 8 x i16> @llvm.aarch64.sve.tbl.nxv8i16(<vscale x 8 x i16> [[BGRA]], <vscale x 8 x i16> [[TMP57]])
; CHECK-NEXT: [[TMP26:%.*]] = bitcast <vscale x 8 x i16> [[TMP25]] to <vscale x 2 x i64>
; CHECK-NEXT: [[TMP27:%.*]] = uitofp <vscale x 2 x i64> [[TMP26]] to <vscale x 2 x double>
-; CHECK-NEXT: [[B_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP6]], [[REVERSED]]
-; CHECK-NEXT: [[G_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP13]], [[REVERSED]]
-; CHECK-NEXT: [[R_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP20]], [[REVERSED]]
-; CHECK-NEXT: [[A_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP27]], [[REVERSED]]
+; CHECK-NEXT: [[B_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP6]], [[COMMON_TERM]]
+; CHECK-NEXT: [[G_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP13]], [[COMMON_TERM]]
+; CHECK-NEXT: [[R_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP20]], [[COMMON_TERM]]
+; CHECK-NEXT: [[A_MUL_F64:%.*]] = fmul <vscale x 2 x double> [[TMP27]], [[COMMON_TERM]]
; CHECK-NEXT: [[FADD_B_F64]] = fadd <vscale x 2 x double> [[ACC_B_F64]], [[B_MUL_F64]]
; CHECK-NEXT: [[FADD_G_F64]] = fadd <vscale x 2 x double> [[ACC_G_F64]], [[G_MUL_F64]]
; CHECK-NEXT: [[FADD_R_F64]] = fadd <vscale x 2 x double> [[ACC_R_F64]], [[R_MUL_F64]]
diff --git a/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll b/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
index fa3df9f39a4ad..b7970a9f38369 100644
--- a/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
+++ b/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
@@ -641,25 +641,33 @@ exit:
define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double:
; CHECK: // %bb.0: // %entry
-; CHECK-NEXT: index z7.d, #0, #4
-; CHECK-NEXT: mov z2.d, #0xffffffffffff0000
-; CHECK-NEXT: mov z5.d, #0xffffffffffff0001
-; CHECK-NEXT: mov x9, #-65534 // =0xffffffffffff0002
-; CHECK-NEXT: mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT: cnth x9
+; CHECK-NEXT: index z7.d, #0, #-4
; CHECK-NEXT: movi v0.2d, #0000000000000000
-; CHECK-NEXT: mov z6.d, x9
-; CHECK-NEXT: movi v3.2d, #0000000000000000
-; CHECK-NEXT: mov x8, xzr
-; CHECK-NEXT: movi v1.2d, #0000000000000000
+; CHECK-NEXT: sub x10, x9, #16, lsl #12 // =65536
+; CHECK-NEXT: sub x11, x9, #16, lsl #12 // =65536
+; CHECK-NEXT: movi v4.2d, #0000000000000000
+; CHECK-NEXT: sub x10, x10, #4
+; CHECK-NEXT: sub x11, x11, #3
+; CHECK-NEXT: movi v2.2d, #0000000000000000
+; CHECK-NEXT: mov z1.d, x10
+; CHECK-NEXT: sub x10, x9, #16, lsl #12 // =65536
+; CHECK-NEXT: sub x9, x9, #16, lsl #12 // =65536
+; CHECK-NEXT: sub x10, x10, #2
+; CHECK-NEXT: sub x9, x9, #1
+; CHECK-NEXT: mov z5.d, x11
+; CHECK-NEXT: mov z6.d, x10
+; CHECK-NEXT: mov z24.d, x9
; CHECK-NEXT: ptrue p1.d
+; CHECK-NEXT: add z3.d, z7.d, z1.d
+; CHECK-NEXT: movi v1.2d, #0000000000000000
+; CHECK-NEXT: mov x8, xzr
+; CHECK-NEXT: add z5.d, z7.d, z5.d
; CHECK-NEXT: add x9, x1, #8
-; CHECK-NEXT: add z4.d, z7.d, z2.d
-; CHECK-NEXT: movi v2.2d, #0000000000000000
; CHECK-NEXT: cntw x10
-; CHECK-NEXT: add z5.d, z7.d, z5.d
; CHECK-NEXT: add z6.d, z7.d, z6.d
-; CHECK-NEXT: rdvl x11, #2
; CHECK-NEXT: add z7.d, z7.d, z24.d
+; CHECK-NEXT: rdvl x11, #2
; CHECK-NEXT: .LBB8_1: // %loop
; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
; CHECK-NEXT: ld1h { z24.h }, p0/z, [x0]
@@ -667,11 +675,10 @@ define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr
; CHECK-NEXT: sub x8, x8, x10
; CHECK-NEXT: cmn x8, #2048
; CHECK-NEXT: add x0, x0, x11
-; CHECK-NEXT: tbl z25.h, { z24.h }, z4.h
+; CHECK-NEXT: tbl z25.h, { z24.h }, z3.h
; CHECK-NEXT: tbl z26.h, { z24.h }, z5.h
; CHECK-NEXT: tbl z27.h, { z24.h }, z6.h
; CHECK-NEXT: tbl z24.h, { z24.h }, z7.h
-; CHECK-NEXT: rev z28.d, z28.d
; CHECK-NEXT: ucvtf z25.d, p1/m, z25.d
; CHECK-NEXT: ucvtf z26.d, p1/m, z26.d
; CHECK-NEXT: ucvtf z27.d, p1/m, z27.d
@@ -681,18 +688,18 @@ define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_double(ptr
; CHECK-NEXT: fmul z27.d, z27.d, z28.d
; CHECK-NEXT: fmul z24.d, z24.d, z28.d
; CHECK-NEXT: fadd z0.d, z0.d, z25.d
-; CHECK-NEXT: fadd z3.d, z3.d, z26.d
-; CHECK-NEXT: fadd z1.d, z1.d, z27.d
-; CHECK-NEXT: fadd z2.d, z2.d, z24.d
+; CHECK-NEXT: fadd z4.d, z4.d, z26.d
+; CHECK-NEXT: fadd z2.d, z2.d, z27.d
+; CHECK-NEXT: fadd z1.d, z1.d, z24.d
; CHECK-NEXT: b.ne .LBB8_1
; CHECK-NEXT: // %bb.2: // %exit
; CHECK-NEXT: faddv d0, p1, z0.d
-; CHECK-NEXT: faddv d3, p1, z3.d
-; CHECK-NEXT: faddv d1, p1, z1.d
+; CHECK-NEXT: faddv d3, p1, z4.d
; CHECK-NEXT: faddv d2, p1, z2.d
+; CHECK-NEXT: faddv d1, p1, z1.d
; CHECK-NEXT: mov v0.d[1], v3.d[0]
-; CHECK-NEXT: mov v1.d[1], v2.d[0]
-; CHECK-NEXT: stp q0, q1, [x2]
+; CHECK-NEXT: mov v2.d[1], v1.d[0]
+; CHECK-NEXT: stp q0, q2, [x2]
; CHECK-NEXT: ret
entry:
%vscale = tail call i64 @llvm.vscale.i64()
@@ -751,25 +758,30 @@ exit:
define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_partial_coverage(ptr %src, ptr %src2, ptr %dst, <vscale x 8 x i1> %mask) #0 {
; CHECK-LABEL: uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_partial_coverage:
; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cnth x9
; CHECK-NEXT: index z7.d, #0, #4
-; CHECK-NEXT: mov z2.d, #0xffffffffffff0000
-; CHECK-NEXT: mov z5.d, #0xffffffffffff0001
-; CHECK-NEXT: mov x9, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT: index z5.d, #0, #-4
+; CHECK-NEXT: sub x10, x9, #16, lsl #12 // =65536
+; CHECK-NEXT: sub x9, x9, #16, lsl #12 // =65536
+; CHECK-NEXT: mov z6.d, #0xffffffffffff0001
+; CHECK-NEXT: sub x10, x10, #4
+; CHECK-NEXT: sub x9, x9, #2
; CHECK-NEXT: mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT: mov z4.d, x10
+; CHECK-NEXT: mov z25.d, x9
; CHECK-NEXT: movi v0.2d, #0000000000000000
-; CHECK-NEXT: mov z6.d, x9
; CHECK-NEXT: movi v3.2d, #0000000000000000
-; CHECK-NEXT: mov x8, xzr
+; CHECK-NEXT: movi v2.2d, #0000000000000000
; CHECK-NEXT: movi v1.2d, #0000000000000000
+; CHECK-NEXT: add z6.d, z7.d, z6.d
; CHECK-NEXT: ptrue p1.d
+; CHECK-NEXT: mov x8, xzr
+; CHECK-NEXT: add z4.d, z5.d, z4.d
+; CHECK-NEXT: add z5.d, z5.d, z25.d
+; CHECK-NEXT: add z7.d, z7.d, z24.d
; CHECK-NEXT: add x9, x1, #8
-; CHECK-NEXT: add z4.d, z7.d, z2.d
-; CHECK-NEXT: movi v2.2d, #0000000000000000
; CHECK-NEXT: cntw x10
-; CHECK-NEXT: add z5.d, z7.d, z5.d
-; CHECK-NEXT: add z6.d, z7.d, z6.d
; CHECK-NEXT: rdvl x11, #2
-; CHECK-NEXT: add z7.d, z7.d, z24.d
; CHECK-NEXT: .LBB9_1: // %loop
; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
; CHECK-NEXT: ld1h { z24.h }, p0/z, [x0]
@@ -778,31 +790,30 @@ define void @uitofp_nxv8i16_to_nxv8f64_deinterleave_fma_with_reverse_partial_cov
; CHECK-NEXT: cmn x8, #2048
; CHECK-NEXT: add x0, x0, x11
; CHECK-NEXT: tbl z25.h, { z24.h }, z4.h
-; CHECK-NEXT: tbl z26.h, { z24.h }, z5.h
-; CHECK-NEXT: tbl z27.h, { z24.h }, z6.h
+; CHECK-NEXT: tbl z26.h, { z24.h }, z6.h
+; CHECK-NEXT: tbl z27.h, { z24.h }, z5.h
; CHECK-NEXT: tbl z24.h, { z24.h }, z7.h
-; CHECK-NEXT: rev z29.d, z28.d
; CHECK-NEXT: ucvtf z25.d, p1/m, z25.d
; CHECK-NEXT: ucvtf z26.d, p1/m, z26.d
; CHECK-NEXT: ucvtf z27.d, p1/m, z27.d
; CHECK-NEXT: ucvtf z24.d, p1/m, z24.d
-; CHECK-NEXT: fmul z25.d, z25.d, z29.d
+; CHECK-NEXT: fmul z25.d, z25.d, z28.d
; CHECK-NEXT: fmul z26.d, z26.d, z28.d
-; CHECK-NEXT: fmul z27.d, z27.d, z29.d
+; CHECK-NEXT: fmul z27.d, z27.d, z28.d
; CHECK-NEXT: fmul z24.d, z24.d, z28.d
; CHECK-NEXT: fadd z0.d, z0.d, z25.d
; CHECK-NEXT: fadd z3.d, z3.d, z26.d
-; CHECK-NEXT: fadd z1.d, z1.d, z27.d
-; CHECK-NEXT: fadd z2.d, z2.d, z24.d
+; CHECK-NEXT: fadd z2.d, z2.d, z27.d
+; CHECK-NEXT: fadd z1.d, z1.d, z24.d
; CHECK-NEXT: b.ne .LBB9_1
; CHECK-NEXT: // %bb.2: // %exit
; CHECK-NEXT: faddv d0, p1, z0.d
; CHECK-NEXT: faddv d3, p1, z3.d
-; CHECK-NEXT: faddv d1, p1, z1.d
; CHECK-NEXT: faddv d2, p1, z2.d
+; CHECK-NEXT: faddv d1, p1, z1.d
; CHECK-NEXT: mov v0.d[1], v3.d[0]
-; CHECK-NEXT: mov v1.d[1], v2.d[0]
-; CHECK-NEXT: stp q0, q1, [x2]
+; CHECK-NEXT: mov v2.d[1], v1.d[0]
+; CHECK-NEXT: stp q0, q2, [x2]
; CHECK-NEXT: ret
entry:
%vscale = tail call i64 @llvm.vscale.i64()
More information about the llvm-commits
mailing list