[llvm] [AArch64] Initial vector lsr tests (PR #226188)
Graham Hunter via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 25 05:10:55 PDT 2026
https://github.com/huntergr-arm updated https://github.com/llvm/llvm-project/pull/226188
>From 23273f04d3c72b28e76a74ee9288c92249542ab7 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Wed, 23 Sep 2026 14:24:44 +0000
Subject: [PATCH 1/2] [AArch64] Initial vector lsr tests
---
.../AArch64/strength-reduce-vector-as-data.ll | 321 ++++++++++++++++++
1 file changed, 321 insertions(+)
create mode 100644 llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
diff --git a/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
new file mode 100644
index 0000000000000..853c94e1307c8
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
@@ -0,0 +1,321 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s | FileCheck %s
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
+; CHECK-LABEL: init_array_of_ptrs_to_structs:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cntd x8
+; CHECK-NEXT: index z0.d, #0, #1
+; CHECK-NEXT: mov z2.d, x1
+; CHECK-NEXT: mov z1.d, x8
+; CHECK-NEXT: mov z3.d, #72 // =0x48
+; CHECK-NEXT: neg x8, x8
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: and x8, x8, x2
+; CHECK-NEXT: .LBB0_1: // %vector.body
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: movprfx z4, z2
+; CHECK-NEXT: mla z4.d, p0/m, z0.d, z3.d
+; CHECK-NEXT: decd x8
+; CHECK-NEXT: add z0.d, z0.d, z1.d
+; CHECK-NEXT: str z4, [x0]
+; CHECK-NEXT: incb x0
+; CHECK-NEXT: cbnz x8, .LBB0_1
+; CHECK-NEXT: // %bb.2: // %middle.block
+; CHECK-NEXT: ret
+entry:
+ %vscale = tail call i64 @llvm.vscale.i64()
+ %step = shl nuw nsw i64 %vscale, 1
+ %min.iters.check = icmp samesign ugt i64 %step, %num_arcs
+ %.not = sub nsw i64 0, %step
+ %n.vec = and i64 %.not, %num_arcs
+ %2 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %step, i64 0
+ %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ br label %vector.body
+
+vector.body:
+ %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+ %vec.ind = phi <vscale x 2 x i64> [ %2, %entry ], [ %vec.ind.next, %vector.body ]
+ %wide.gep = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %vec.ind
+ %3 = getelementptr inbounds nuw [8 x i8], ptr %arc_ptrs, i64 %index
+ store <vscale x 2 x ptr> %wide.gep, ptr %3, align 8
+ %index.next = add nuw i64 %index, %step
+ %vec.ind.next = add nuw nsw <vscale x 2 x i64> %vec.ind, %broadcast.splat
+ %4 = icmp eq i64 %index.next, %n.vec
+ br i1 %4, label %middle.block, label %vector.body
+
+middle.block:
+ ret void
+}
+
+define void @init_array_of_ptrs_to_structs_interleave4(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
+; CHECK-LABEL: init_array_of_ptrs_to_structs_interleave4:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cntd x8
+; CHECK-NEXT: mov w9, #2147483647 // =0x7fffffff
+; CHECK-NEXT: cnth x10
+; CHECK-NEXT: mov z0.d, x8
+; CHECK-NEXT: cntw x8
+; CHECK-NEXT: inch x9
+; CHECK-NEXT: mov z1.d, x8
+; CHECK-NEXT: cntd x8, all, mul #3
+; CHECK-NEXT: index z2.d, #0, #1
+; CHECK-NEXT: mov z3.d, x8
+; CHECK-NEXT: mov z4.d, x10
+; CHECK-NEXT: mov z5.d, x1
+; CHECK-NEXT: mov z6.d, #72 // =0x48
+; CHECK-NEXT: and x8, x9, x2
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: sub x8, x8, x2
+; CHECK-NEXT: .LBB1_1: // %vector.body
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: add z7.d, z2.d, z0.d
+; CHECK-NEXT: add z16.d, z2.d, z1.d
+; CHECK-NEXT: movprfx z18, z5
+; CHECK-NEXT: mla z18.d, p0/m, z2.d, z6.d
+; CHECK-NEXT: add z17.d, z2.d, z3.d
+; CHECK-NEXT: inch x8
+; CHECK-NEXT: add z2.d, z2.d, z4.d
+; CHECK-NEXT: mad z7.d, p0/m, z6.d, z5.d
+; CHECK-NEXT: mad z16.d, p0/m, z6.d, z5.d
+; CHECK-NEXT: mad z17.d, p0/m, z6.d, z5.d
+; CHECK-NEXT: str z18, [x0]
+; CHECK-NEXT: str z7, [x0, #1, mul vl]
+; CHECK-NEXT: str z16, [x0, #2, mul vl]
+; CHECK-NEXT: str z17, [x0, #3, mul vl]
+; CHECK-NEXT: incb x0, all, mul #4
+; CHECK-NEXT: cbnz x8, .LBB1_1
+; CHECK-NEXT: // %bb.2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %0 = tail call i64 @llvm.vscale.i64()
+ %1 = shl nuw nsw i64 %0, 1
+ %2 = shl nuw nsw i64 %0, 1
+ %3 = shl nuw nsw i64 %0, 3
+ %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %1, i64 0
+ %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %4 = add nuw nsw i64 %3, 2147483647
+ %n.mod.vf = and i64 %4, %num_arcs
+ %n.vec = sub nsw i64 %num_arcs, %n.mod.vf
+ %5 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %invariant.op = add nuw <vscale x 2 x i64> %broadcast.splat, %broadcast.splat
+ %invariant.op26 = add nuw <vscale x 2 x i64> %invariant.op, %broadcast.splat
+ %invariant.op27 = add nuw <vscale x 2 x i64> %invariant.op26, %broadcast.splat
+ %6 = shl nuw nsw i64 %0, 6
+ %7 = mul i64 %0, 48
+ %8 = shl i64 %0, 5
+ %9 = shl i64 %0, 4
+ br label %vector.body
+
+vector.body:
+ %lsr.iv36 = phi ptr [ %scevgep37, %vector.body ], [ %arc_ptrs, %entry ]
+ %lsr.iv34 = phi i64 [ %lsr.iv.next35, %vector.body ], [ %n.vec, %entry ]
+ %vec.ind = phi <vscale x 2 x i64> [ %5, %entry ], [ %vec.ind.next, %vector.body ]
+ %step.add = add nuw <vscale x 2 x i64> %vec.ind, %broadcast.splat
+ %step.add.2.reass = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op
+ %step.add.3.reass = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op26
+ %wide.gep = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %vec.ind
+ %wide.gep10 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add
+ %wide.gep11 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add.2.reass
+ %wide.gep12 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add.3.reass
+ %scevgep40 = getelementptr i8, ptr %lsr.iv36, i64 %9
+ %scevgep39 = getelementptr i8, ptr %lsr.iv36, i64 %8
+ %scevgep38 = getelementptr i8, ptr %lsr.iv36, i64 %7
+ store <vscale x 2 x ptr> %wide.gep, ptr %lsr.iv36, align 8
+ store <vscale x 2 x ptr> %wide.gep10, ptr %scevgep40, align 8
+ store <vscale x 2 x ptr> %wide.gep11, ptr %scevgep39, align 8
+ store <vscale x 2 x ptr> %wide.gep12, ptr %scevgep38, align 8
+ %vec.ind.next = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op27
+ %lsr.iv.next35 = sub i64 %lsr.iv34, %3
+ %scevgep37 = getelementptr i8, ptr %lsr.iv36, i64 %6
+ %10 = icmp eq i64 %lsr.iv.next35, 0
+ br i1 %10, label %exit, label %vector.body
+
+exit:
+ ret void
+}
+
+define void @init_offset_array(i64 %0, ptr %1, ptr nofree writeonly captures(none) %2) #0 {
+; CHECK-LABEL: init_offset_array:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cntd x8
+; CHECK-NEXT: index z0.d, x0, #-1
+; CHECK-NEXT: decb x2
+; CHECK-NEXT: neg x8, x8
+; CHECK-NEXT: cmp x0, #1
+; CHECK-NEXT: mov z2.d, x1
+; CHECK-NEXT: mov z1.d, x8
+; CHECK-NEXT: mov z3.d, #56 // =0x38
+; CHECK-NEXT: csinc x9, x0, xzr, hi
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: and x8, x9, x8
+; CHECK-NEXT: add x9, x2, #8
+; CHECK-NEXT: .LBB2_1: // %vector.body
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: movprfx z4, z0
+; CHECK-NEXT: sub z4.d, z4.d, #1 // =0x1
+; CHECK-NEXT: decd x8
+; CHECK-NEXT: add z0.d, z0.d, z1.d
+; CHECK-NEXT: movprfx z5, z2
+; CHECK-NEXT: mla z5.d, p0/m, z4.d, z3.d
+; CHECK-NEXT: fmov x10, d4
+; CHECK-NEXT: rev z5.d, z5.d
+; CHECK-NEXT: st1d { z5.d }, p0, [x9, x10, lsl #3]
+; CHECK-NEXT: cbnz x8, .LBB2_1
+; CHECK-NEXT: // %bb.2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %3 = tail call i64 @llvm.umax.i64(i64 %0, i64 1)
+ %4 = tail call i64 @llvm.vscale.i64()
+ %5 = shl nuw i64 %4, 1
+ %n.mod.vf = urem i64 %3, %5
+ %n.vec = sub i64 %3, %n.mod.vf
+ %6 = sub i64 %0, %n.vec
+ %7 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %0, i64 0
+ %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %8 = sub nsw <vscale x 2 x i64> %broadcast.splat, %7
+ %9 = sub nsw i64 0, %5
+ %broadcast.splatinsert1 = insertelement <vscale x 2 x i64> poison, i64 %9, i64 0
+ %broadcast.splat2 = shufflevector <vscale x 2 x i64> %broadcast.splatinsert1, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %10 = sub i64 1, %5
+ %invariant.gep = getelementptr [8 x i8], ptr %2, i64 %10
+ br label %vector.body
+
+vector.body:
+ %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+ %vec.ind = phi <vscale x 2 x i64> [ %8, %entry ], [ %vec.ind.next, %vector.body ]
+ %11 = add nsw <vscale x 2 x i64> %vec.ind, splat (i64 -1)
+ %12 = extractelement <vscale x 2 x i64> %11, i64 0
+ %wide.gep = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %11
+ %gep = getelementptr [8 x i8], ptr %invariant.gep, i64 %12
+ %reverse = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep)
+ store <vscale x 2 x ptr> %reverse, ptr %gep, align 8
+ %index.next = add nuw i64 %index, %5
+ %vec.ind.next = add nsw <vscale x 2 x i64> %vec.ind, %broadcast.splat2
+ %13 = icmp eq i64 %index.next, %n.vec
+ br i1 %13, label %exit, label %vector.body
+
+exit:
+ ret void
+}
+
+define void @init_offset_array_interleave4(i64 %0, ptr %1, ptr nofree writeonly captures(none) %2) #0 {
+; CHECK-LABEL: init_offset_array_interleave4:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cntd x8
+; CHECK-NEXT: mov x12, #-1 // =0xffffffffffffffff
+; CHECK-NEXT: cnth x9
+; CHECK-NEXT: mov z0.d, x8
+; CHECK-NEXT: neg x8, x8
+; CHECK-NEXT: index z3.d, x0, #-1
+; CHECK-NEXT: incd x12
+; CHECK-NEXT: add x8, x8, #1
+; CHECK-NEXT: cmp x0, #1
+; CHECK-NEXT: neg x10, x9
+; CHECK-NEXT: mov x9, x8
+; CHECK-NEXT: decd x8
+; CHECK-NEXT: cntw x13
+; CHECK-NEXT: mov z1.d, x1
+; CHECK-NEXT: cntd x14, all, mul #3
+; CHECK-NEXT: mov z2.d, #56 // =0x38
+; CHECK-NEXT: csinc x11, x0, xzr, hi
+; CHECK-NEXT: neg x13, x13
+; CHECK-NEXT: neg x14, x14
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: and x10, x11, x10
+; CHECK-NEXT: sub x11, x13, x12
+; CHECK-NEXT: sub x12, x14, x12
+; CHECK-NEXT: .LBB3_1: // %vector.body
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: sub z4.d, z3.d, z0.d
+; CHECK-NEXT: sub z3.d, z3.d, #1 // =0x1
+; CHECK-NEXT: dech x10
+; CHECK-NEXT: sub z5.d, z4.d, z0.d
+; CHECK-NEXT: mad z4.d, p0/m, z2.d, z1.d
+; CHECK-NEXT: movprfx z7, z1
+; CHECK-NEXT: mla z7.d, p0/m, z3.d, z2.d
+; CHECK-NEXT: fmov x13, d3
+; CHECK-NEXT: sub z6.d, z5.d, z0.d
+; CHECK-NEXT: mad z5.d, p0/m, z2.d, z1.d
+; CHECK-NEXT: sub z4.d, z4.d, #56 // =0x38
+; CHECK-NEXT: rev z3.d, z7.d
+; CHECK-NEXT: add x13, x2, x13, lsl #3
+; CHECK-NEXT: movprfx z16, z1
+; CHECK-NEXT: mla z16.d, p0/m, z6.d, z2.d
+; CHECK-NEXT: sub z5.d, z5.d, #56 // =0x38
+; CHECK-NEXT: rev z4.d, z4.d
+; CHECK-NEXT: st1d { z3.d }, p0, [x13, x9, lsl #3]
+; CHECK-NEXT: sub z3.d, z6.d, z0.d
+; CHECK-NEXT: sub z16.d, z16.d, #56 // =0x38
+; CHECK-NEXT: rev z5.d, z5.d
+; CHECK-NEXT: st1d { z4.d }, p0, [x13, x8, lsl #3]
+; CHECK-NEXT: rev z7.d, z16.d
+; CHECK-NEXT: st1d { z5.d }, p0, [x13, x11, lsl #3]
+; CHECK-NEXT: st1d { z7.d }, p0, [x13, x12, lsl #3]
+; CHECK-NEXT: cbnz x10, .LBB3_1
+; CHECK-NEXT: // %bb.2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %3 = tail call i64 @llvm.umax.i64(i64 %0, i64 1)
+ %4 = tail call i64 @llvm.vscale.i64()
+ %5 = shl nuw i64 %4, 1
+ %6 = shl nuw i64 %4, 3
+ %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %5, i64 0
+ %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %n.mod.vf = urem i64 %3, %6
+ %n.vec = sub i64 %3, %n.mod.vf
+ %7 = sub i64 %0, %n.vec
+ %8 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %broadcast.splatinsert2 = insertelement <vscale x 2 x i64> poison, i64 %0, i64 0
+ %broadcast.splat3 = shufflevector <vscale x 2 x i64> %broadcast.splatinsert2, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %9 = sub nsw <vscale x 2 x i64> %broadcast.splat3, %8
+ %10 = add nsw i64 %5, -1
+ %11 = sub i64 1, %5
+ %12 = sub i64 %11, %5
+ %13 = mul i64 %4, -4
+ %14 = sub i64 %13, %10
+ %15 = mul i64 %4, -6
+ %16 = sub i64 %15, %10
+ br label %vector.body
+
+vector.body:
+ %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+ %vec.ind = phi <vscale x 2 x i64> [ %9, %entry ], [ %30, %vector.body ]
+ %17 = sub <vscale x 2 x i64> %vec.ind, %broadcast.splat
+ %18 = sub <vscale x 2 x i64> %17, %broadcast.splat
+ %19 = sub <vscale x 2 x i64> %18, %broadcast.splat
+ %20 = add nsw <vscale x 2 x i64> %vec.ind, splat (i64 -1)
+ %21 = extractelement <vscale x 2 x i64> %20, i64 0
+ %22 = add nsw <vscale x 2 x i64> %17, splat (i64 -1)
+ %23 = add nsw <vscale x 2 x i64> %18, splat (i64 -1)
+ %24 = add nsw <vscale x 2 x i64> %19, splat (i64 -1)
+ %wide.gep = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %20
+ %wide.gep4 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %22
+ %wide.gep5 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %23
+ %wide.gep6 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %24
+ %25 = getelementptr inbounds nuw [8 x i8], ptr %2, i64 %21
+ %26 = getelementptr inbounds [8 x i8], ptr %25, i64 %11
+ %27 = getelementptr inbounds [8 x i8], ptr %25, i64 %12
+ %28 = getelementptr inbounds [8 x i8], ptr %25, i64 %14
+ %29 = getelementptr inbounds [8 x i8], ptr %25, i64 %16
+ %reverse = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep)
+ %reverse7 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep4)
+ %reverse8 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep5)
+ %reverse9 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep6)
+ store <vscale x 2 x ptr> %reverse, ptr %26, align 8
+ store <vscale x 2 x ptr> %reverse7, ptr %27, align 8
+ store <vscale x 2 x ptr> %reverse8, ptr %28, align 8
+ store <vscale x 2 x ptr> %reverse9, ptr %29, align 8
+ %index.next = add nuw i64 %index, %6
+ %30 = sub <vscale x 2 x i64> %19, %broadcast.splat
+ %31 = icmp eq i64 %index.next, %n.vec
+ br i1 %31, label %exit, label %vector.body
+
+exit:
+ ret void
+}
+
+attributes #0 = { nounwind vscale_range(1,16) "target-features"="+sve2" }
>From 8493b9cac6557cdbcf3c12102464b1191b6150cf Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 25 Sep 2026 12:09:47 +0000
Subject: [PATCH 2/2] Copy tests to CGP IR test directory
---
.../AArch64/strength-reduce-vector-as-data.ll | 356 ++++++++++++++++++
1 file changed, 356 insertions(+)
create mode 100644 llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
diff --git a/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
new file mode 100644
index 0000000000000..11cb04cde3c96
--- /dev/null
+++ b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
@@ -0,0 +1,356 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes='require<profile-summary>,function(codegenprepare)' -S %s | FileCheck %s
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
+; CHECK-LABEL: define void @init_array_of_ptrs_to_structs(
+; CHECK-SAME: ptr noalias [[ARC_PTRS:%.*]], ptr [[ARC_NEW:%.*]], i64 [[NUM_ARCS:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[VSCALE:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[STEP:%.*]] = shl nuw nsw i64 [[VSCALE]], 1
+; CHECK-NEXT: [[DOTNOT:%.*]] = sub nsw i64 0, [[STEP]]
+; CHECK-NEXT: [[N_VEC:%.*]] = and i64 [[DOTNOT]], [[NUM_ARCS]]
+; CHECK-NEXT: [[TMP0:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[STEP]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP0]], %[[ENTRY]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[VEC_IND]]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARC_PTRS]], i64 [[INDEX]]
+; CHECK-NEXT: store <vscale x 2 x ptr> [[WIDE_GEP]], ptr [[TMP1]], align 8
+; CHECK-NEXT: [[TMP2:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw nsw i64 [[TMP2]], 1
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %vscale = tail call i64 @llvm.vscale.i64()
+ %step = shl nuw nsw i64 %vscale, 1
+ %min.iters.check = icmp samesign ugt i64 %step, %num_arcs
+ %.not = sub nsw i64 0, %step
+ %n.vec = and i64 %.not, %num_arcs
+ %2 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %step, i64 0
+ %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ br label %vector.body
+
+vector.body:
+ %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+ %vec.ind = phi <vscale x 2 x i64> [ %2, %entry ], [ %vec.ind.next, %vector.body ]
+ %wide.gep = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %vec.ind
+ %3 = getelementptr inbounds nuw [8 x i8], ptr %arc_ptrs, i64 %index
+ store <vscale x 2 x ptr> %wide.gep, ptr %3, align 8
+ %index.next = add nuw i64 %index, %step
+ %vec.ind.next = add nuw nsw <vscale x 2 x i64> %vec.ind, %broadcast.splat
+ %4 = icmp eq i64 %index.next, %n.vec
+ br i1 %4, label %middle.block, label %vector.body
+
+middle.block:
+ ret void
+}
+
+define void @init_array_of_ptrs_to_structs_interleave4(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
+; CHECK-LABEL: define void @init_array_of_ptrs_to_structs_interleave4(
+; CHECK-SAME: ptr noalias [[ARC_PTRS:%.*]], ptr [[ARC_NEW:%.*]], i64 [[NUM_ARCS:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw nsw i64 [[TMP0]], 1
+; CHECK-NEXT: [[TMP2:%.*]] = shl nuw nsw i64 [[TMP0]], 1
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw nsw i64 [[TMP0]], 3
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP1]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP4:%.*]] = add nuw nsw i64 [[TMP3]], 2147483647
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i64 [[TMP4]], [[NUM_ARCS]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub nsw i64 [[NUM_ARCS]], [[N_MOD_VF]]
+; CHECK-NEXT: [[TMP5:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[INVARIANT_OP:%.*]] = add nuw <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[INVARIANT_OP26:%.*]] = add nuw <vscale x 2 x i64> [[INVARIANT_OP]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[INVARIANT_OP27:%.*]] = add nuw <vscale x 2 x i64> [[INVARIANT_OP26]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[LSR_IV36:%.*]] = phi ptr [ [[SCEVGEP37:%.*]], %[[VECTOR_BODY]] ], [ [[ARC_PTRS]], %[[ENTRY]] ]
+; CHECK-NEXT: [[LSR_IV34:%.*]] = phi i64 [ [[LSR_IV_NEXT35:%.*]], %[[VECTOR_BODY]] ], [ [[N_VEC]], %[[ENTRY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP5]], %[[ENTRY]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[STEP_ADD:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[STEP_ADD_2_REASS:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[INVARIANT_OP]]
+; CHECK-NEXT: [[STEP_ADD_3_REASS:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[INVARIANT_OP26]]
+; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[VEC_IND]]
+; CHECK-NEXT: [[WIDE_GEP10:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[STEP_ADD]]
+; CHECK-NEXT: [[WIDE_GEP11:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[STEP_ADD_2_REASS]]
+; CHECK-NEXT: [[WIDE_GEP12:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[STEP_ADD_3_REASS]]
+; CHECK-NEXT: [[TMP6:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP7:%.*]] = shl i64 [[TMP6]], 4
+; CHECK-NEXT: [[SCEVGEP40:%.*]] = getelementptr i8, ptr [[LSR_IV36]], i64 [[TMP7]]
+; CHECK-NEXT: [[TMP8:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP9:%.*]] = shl i64 [[TMP8]], 5
+; CHECK-NEXT: [[SCEVGEP39:%.*]] = getelementptr i8, ptr [[LSR_IV36]], i64 [[TMP9]]
+; CHECK-NEXT: [[TMP10:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP11:%.*]] = mul i64 [[TMP10]], 48
+; CHECK-NEXT: [[SCEVGEP38:%.*]] = getelementptr i8, ptr [[LSR_IV36]], i64 [[TMP11]]
+; CHECK-NEXT: store <vscale x 2 x ptr> [[WIDE_GEP]], ptr [[LSR_IV36]], align 8
+; CHECK-NEXT: store <vscale x 2 x ptr> [[WIDE_GEP10]], ptr [[SCEVGEP40]], align 8
+; CHECK-NEXT: store <vscale x 2 x ptr> [[WIDE_GEP11]], ptr [[SCEVGEP39]], align 8
+; CHECK-NEXT: store <vscale x 2 x ptr> [[WIDE_GEP12]], ptr [[SCEVGEP38]], align 8
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[INVARIANT_OP27]]
+; CHECK-NEXT: [[TMP12:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP13:%.*]] = shl nuw nsw i64 [[TMP12]], 3
+; CHECK-NEXT: [[LSR_IV_NEXT35]] = sub i64 [[LSR_IV34]], [[TMP13]]
+; CHECK-NEXT: [[TMP14:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP15:%.*]] = shl nuw nsw i64 [[TMP14]], 6
+; CHECK-NEXT: [[SCEVGEP37]] = getelementptr i8, ptr [[LSR_IV36]], i64 [[TMP15]]
+; CHECK-NEXT: [[TMP16:%.*]] = icmp eq i64 [[LSR_IV_NEXT35]], 0
+; CHECK-NEXT: br i1 [[TMP16]], label %[[EXIT:.*]], label %[[VECTOR_BODY]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %0 = tail call i64 @llvm.vscale.i64()
+ %1 = shl nuw nsw i64 %0, 1
+ %2 = shl nuw nsw i64 %0, 1
+ %3 = shl nuw nsw i64 %0, 3
+ %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %1, i64 0
+ %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %4 = add nuw nsw i64 %3, 2147483647
+ %n.mod.vf = and i64 %4, %num_arcs
+ %n.vec = sub nsw i64 %num_arcs, %n.mod.vf
+ %5 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %invariant.op = add nuw <vscale x 2 x i64> %broadcast.splat, %broadcast.splat
+ %invariant.op26 = add nuw <vscale x 2 x i64> %invariant.op, %broadcast.splat
+ %invariant.op27 = add nuw <vscale x 2 x i64> %invariant.op26, %broadcast.splat
+ %6 = shl nuw nsw i64 %0, 6
+ %7 = mul i64 %0, 48
+ %8 = shl i64 %0, 5
+ %9 = shl i64 %0, 4
+ br label %vector.body
+
+vector.body:
+ %lsr.iv36 = phi ptr [ %scevgep37, %vector.body ], [ %arc_ptrs, %entry ]
+ %lsr.iv34 = phi i64 [ %lsr.iv.next35, %vector.body ], [ %n.vec, %entry ]
+ %vec.ind = phi <vscale x 2 x i64> [ %5, %entry ], [ %vec.ind.next, %vector.body ]
+ %step.add = add nuw <vscale x 2 x i64> %vec.ind, %broadcast.splat
+ %step.add.2.reass = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op
+ %step.add.3.reass = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op26
+ %wide.gep = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %vec.ind
+ %wide.gep10 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add
+ %wide.gep11 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add.2.reass
+ %wide.gep12 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add.3.reass
+ %scevgep40 = getelementptr i8, ptr %lsr.iv36, i64 %9
+ %scevgep39 = getelementptr i8, ptr %lsr.iv36, i64 %8
+ %scevgep38 = getelementptr i8, ptr %lsr.iv36, i64 %7
+ store <vscale x 2 x ptr> %wide.gep, ptr %lsr.iv36, align 8
+ store <vscale x 2 x ptr> %wide.gep10, ptr %scevgep40, align 8
+ store <vscale x 2 x ptr> %wide.gep11, ptr %scevgep39, align 8
+ store <vscale x 2 x ptr> %wide.gep12, ptr %scevgep38, align 8
+ %vec.ind.next = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op27
+ %lsr.iv.next35 = sub i64 %lsr.iv34, %3
+ %scevgep37 = getelementptr i8, ptr %lsr.iv36, i64 %6
+ %10 = icmp eq i64 %lsr.iv.next35, 0
+ br i1 %10, label %exit, label %vector.body
+
+exit:
+ ret void
+}
+
+define void @init_offset_array(i64 %0, ptr %1, ptr nofree writeonly captures(none) %2) #0 {
+; CHECK-LABEL: define void @init_offset_array(
+; CHECK-SAME: i64 [[TMP0:%.*]], ptr [[TMP1:%.*]], ptr nofree writeonly captures(none) [[TMP2:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP3:%.*]] = tail call i64 @llvm.umax.i64(i64 [[TMP0]], i64 1)
+; CHECK-NEXT: [[TMP4:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[TMP3]], [[TMP5]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP3]], [[N_MOD_VF]]
+; CHECK-NEXT: [[TMP6:%.*]] = sub i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: [[TMP7:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP0]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP8:%.*]] = sub nsw <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP7]]
+; CHECK-NEXT: [[TMP9:%.*]] = sub nsw i64 0, [[TMP5]]
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP9]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT1]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP10:%.*]] = sub i64 1, [[TMP5]]
+; CHECK-NEXT: [[INVARIANT_GEP:%.*]] = getelementptr [8 x i8], ptr [[TMP2]], i64 [[TMP10]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP8]], %[[ENTRY]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP11:%.*]] = add nsw <vscale x 2 x i64> [[VEC_IND]], splat (i64 -1)
+; CHECK-NEXT: [[TMP12:%.*]] = extractelement <vscale x 2 x i64> [[TMP11]], i64 0
+; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr [56 x i8], ptr [[TMP1]], <vscale x 2 x i64> [[TMP11]]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr [8 x i8], ptr [[INVARIANT_GEP]], i64 [[TMP12]]
+; CHECK-NEXT: [[REVERSE:%.*]] = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> [[WIDE_GEP]])
+; CHECK-NEXT: store <vscale x 2 x ptr> [[REVERSE]], ptr [[GEP]], align 8
+; CHECK-NEXT: [[TMP13:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP14:%.*]] = shl nuw i64 [[TMP13]], 1
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP14]]
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add nsw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT2]]
+; CHECK-NEXT: [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP15]], label %[[EXIT:.*]], label %[[VECTOR_BODY]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %3 = tail call i64 @llvm.umax.i64(i64 %0, i64 1)
+ %4 = tail call i64 @llvm.vscale.i64()
+ %5 = shl nuw i64 %4, 1
+ %n.mod.vf = urem i64 %3, %5
+ %n.vec = sub i64 %3, %n.mod.vf
+ %6 = sub i64 %0, %n.vec
+ %7 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %0, i64 0
+ %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %8 = sub nsw <vscale x 2 x i64> %broadcast.splat, %7
+ %9 = sub nsw i64 0, %5
+ %broadcast.splatinsert1 = insertelement <vscale x 2 x i64> poison, i64 %9, i64 0
+ %broadcast.splat2 = shufflevector <vscale x 2 x i64> %broadcast.splatinsert1, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %10 = sub i64 1, %5
+ %invariant.gep = getelementptr [8 x i8], ptr %2, i64 %10
+ br label %vector.body
+
+vector.body:
+ %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+ %vec.ind = phi <vscale x 2 x i64> [ %8, %entry ], [ %vec.ind.next, %vector.body ]
+ %11 = add nsw <vscale x 2 x i64> %vec.ind, splat (i64 -1)
+ %12 = extractelement <vscale x 2 x i64> %11, i64 0
+ %wide.gep = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %11
+ %gep = getelementptr [8 x i8], ptr %invariant.gep, i64 %12
+ %reverse = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep)
+ store <vscale x 2 x ptr> %reverse, ptr %gep, align 8
+ %index.next = add nuw i64 %index, %5
+ %vec.ind.next = add nsw <vscale x 2 x i64> %vec.ind, %broadcast.splat2
+ %13 = icmp eq i64 %index.next, %n.vec
+ br i1 %13, label %exit, label %vector.body
+
+exit:
+ ret void
+}
+
+define void @init_offset_array_interleave4(i64 %0, ptr %1, ptr nofree writeonly captures(none) %2) #0 {
+; CHECK-LABEL: define void @init_offset_array_interleave4(
+; CHECK-SAME: i64 [[TMP0:%.*]], ptr [[TMP1:%.*]], ptr nofree writeonly captures(none) [[TMP2:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP3:%.*]] = tail call i64 @llvm.umax.i64(i64 [[TMP0]], i64 1)
+; CHECK-NEXT: [[TMP4:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-NEXT: [[TMP6:%.*]] = shl nuw i64 [[TMP4]], 3
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP5]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[TMP3]], [[TMP6]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP3]], [[N_MOD_VF]]
+; CHECK-NEXT: [[TMP7:%.*]] = sub i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: [[TMP8:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP0]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT2]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP9:%.*]] = sub nsw <vscale x 2 x i64> [[BROADCAST_SPLAT3]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = add nsw i64 [[TMP5]], -1
+; CHECK-NEXT: [[TMP11:%.*]] = sub i64 1, [[TMP5]]
+; CHECK-NEXT: [[TMP12:%.*]] = sub i64 [[TMP11]], [[TMP5]]
+; CHECK-NEXT: [[TMP13:%.*]] = mul i64 [[TMP4]], -4
+; CHECK-NEXT: [[TMP14:%.*]] = sub i64 [[TMP13]], [[TMP10]]
+; CHECK-NEXT: [[TMP15:%.*]] = mul i64 [[TMP4]], -6
+; CHECK-NEXT: [[TMP16:%.*]] = sub i64 [[TMP15]], [[TMP10]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP9]], %[[ENTRY]] ], [ [[TMP32:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP17:%.*]] = sub <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP18:%.*]] = sub <vscale x 2 x i64> [[TMP17]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP19:%.*]] = sub <vscale x 2 x i64> [[TMP18]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP20:%.*]] = add nsw <vscale x 2 x i64> [[VEC_IND]], splat (i64 -1)
+; CHECK-NEXT: [[TMP21:%.*]] = extractelement <vscale x 2 x i64> [[TMP20]], i64 0
+; CHECK-NEXT: [[TMP22:%.*]] = add nsw <vscale x 2 x i64> [[TMP17]], splat (i64 -1)
+; CHECK-NEXT: [[TMP23:%.*]] = add nsw <vscale x 2 x i64> [[TMP18]], splat (i64 -1)
+; CHECK-NEXT: [[TMP24:%.*]] = add nsw <vscale x 2 x i64> [[TMP19]], splat (i64 -1)
+; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr [56 x i8], ptr [[TMP1]], <vscale x 2 x i64> [[TMP20]]
+; CHECK-NEXT: [[WIDE_GEP4:%.*]] = getelementptr [56 x i8], ptr [[TMP1]], <vscale x 2 x i64> [[TMP22]]
+; CHECK-NEXT: [[WIDE_GEP5:%.*]] = getelementptr [56 x i8], ptr [[TMP1]], <vscale x 2 x i64> [[TMP23]]
+; CHECK-NEXT: [[WIDE_GEP6:%.*]] = getelementptr [56 x i8], ptr [[TMP1]], <vscale x 2 x i64> [[TMP24]]
+; CHECK-NEXT: [[TMP25:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[TMP2]], i64 [[TMP21]]
+; CHECK-NEXT: [[TMP26:%.*]] = getelementptr inbounds [8 x i8], ptr [[TMP25]], i64 [[TMP11]]
+; CHECK-NEXT: [[TMP27:%.*]] = getelementptr inbounds [8 x i8], ptr [[TMP25]], i64 [[TMP12]]
+; CHECK-NEXT: [[TMP28:%.*]] = getelementptr inbounds [8 x i8], ptr [[TMP25]], i64 [[TMP14]]
+; CHECK-NEXT: [[TMP29:%.*]] = getelementptr inbounds [8 x i8], ptr [[TMP25]], i64 [[TMP16]]
+; CHECK-NEXT: [[REVERSE:%.*]] = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> [[WIDE_GEP]])
+; CHECK-NEXT: [[REVERSE7:%.*]] = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> [[WIDE_GEP4]])
+; CHECK-NEXT: [[REVERSE8:%.*]] = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> [[WIDE_GEP5]])
+; CHECK-NEXT: [[REVERSE9:%.*]] = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> [[WIDE_GEP6]])
+; CHECK-NEXT: store <vscale x 2 x ptr> [[REVERSE]], ptr [[TMP26]], align 8
+; CHECK-NEXT: store <vscale x 2 x ptr> [[REVERSE7]], ptr [[TMP27]], align 8
+; CHECK-NEXT: store <vscale x 2 x ptr> [[REVERSE8]], ptr [[TMP28]], align 8
+; CHECK-NEXT: store <vscale x 2 x ptr> [[REVERSE9]], ptr [[TMP29]], align 8
+; CHECK-NEXT: [[TMP30:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP31:%.*]] = shl nuw i64 [[TMP30]], 3
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP31]]
+; CHECK-NEXT: [[TMP32]] = sub <vscale x 2 x i64> [[TMP19]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP33]], label %[[EXIT:.*]], label %[[VECTOR_BODY]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %3 = tail call i64 @llvm.umax.i64(i64 %0, i64 1)
+ %4 = tail call i64 @llvm.vscale.i64()
+ %5 = shl nuw i64 %4, 1
+ %6 = shl nuw i64 %4, 3
+ %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %5, i64 0
+ %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %n.mod.vf = urem i64 %3, %6
+ %n.vec = sub i64 %3, %n.mod.vf
+ %7 = sub i64 %0, %n.vec
+ %8 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %broadcast.splatinsert2 = insertelement <vscale x 2 x i64> poison, i64 %0, i64 0
+ %broadcast.splat3 = shufflevector <vscale x 2 x i64> %broadcast.splatinsert2, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %9 = sub nsw <vscale x 2 x i64> %broadcast.splat3, %8
+ %10 = add nsw i64 %5, -1
+ %11 = sub i64 1, %5
+ %12 = sub i64 %11, %5
+ %13 = mul i64 %4, -4
+ %14 = sub i64 %13, %10
+ %15 = mul i64 %4, -6
+ %16 = sub i64 %15, %10
+ br label %vector.body
+
+vector.body:
+ %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+ %vec.ind = phi <vscale x 2 x i64> [ %9, %entry ], [ %30, %vector.body ]
+ %17 = sub <vscale x 2 x i64> %vec.ind, %broadcast.splat
+ %18 = sub <vscale x 2 x i64> %17, %broadcast.splat
+ %19 = sub <vscale x 2 x i64> %18, %broadcast.splat
+ %20 = add nsw <vscale x 2 x i64> %vec.ind, splat (i64 -1)
+ %21 = extractelement <vscale x 2 x i64> %20, i64 0
+ %22 = add nsw <vscale x 2 x i64> %17, splat (i64 -1)
+ %23 = add nsw <vscale x 2 x i64> %18, splat (i64 -1)
+ %24 = add nsw <vscale x 2 x i64> %19, splat (i64 -1)
+ %wide.gep = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %20
+ %wide.gep4 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %22
+ %wide.gep5 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %23
+ %wide.gep6 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %24
+ %25 = getelementptr inbounds nuw [8 x i8], ptr %2, i64 %21
+ %26 = getelementptr inbounds [8 x i8], ptr %25, i64 %11
+ %27 = getelementptr inbounds [8 x i8], ptr %25, i64 %12
+ %28 = getelementptr inbounds [8 x i8], ptr %25, i64 %14
+ %29 = getelementptr inbounds [8 x i8], ptr %25, i64 %16
+ %reverse = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep)
+ %reverse7 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep4)
+ %reverse8 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep5)
+ %reverse9 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep6)
+ store <vscale x 2 x ptr> %reverse, ptr %26, align 8
+ store <vscale x 2 x ptr> %reverse7, ptr %27, align 8
+ store <vscale x 2 x ptr> %reverse8, ptr %28, align 8
+ store <vscale x 2 x ptr> %reverse9, ptr %29, align 8
+ %index.next = add nuw i64 %index, %6
+ %30 = sub <vscale x 2 x i64> %19, %broadcast.splat
+ %31 = icmp eq i64 %index.next, %n.vec
+ br i1 %31, label %exit, label %vector.body
+
+exit:
+ ret void
+}
+
+attributes #0 = { nounwind vscale_range(1,16) "target-features"="+sve2" }
More information about the llvm-commits
mailing list