[llvm] [AArch64] Initial vector lsr tests (PR #226188)

Graham Hunter via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 05:10:55 PDT 2026


https://github.com/huntergr-arm updated https://github.com/llvm/llvm-project/pull/226188

>From 23273f04d3c72b28e76a74ee9288c92249542ab7 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Wed, 23 Sep 2026 14:24:44 +0000
Subject: [PATCH 1/2] [AArch64] Initial vector lsr tests

---
 .../AArch64/strength-reduce-vector-as-data.ll | 321 ++++++++++++++++++
 1 file changed, 321 insertions(+)
 create mode 100644 llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll

diff --git a/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
new file mode 100644
index 0000000000000..853c94e1307c8
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
@@ -0,0 +1,321 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s | FileCheck %s
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
+; CHECK-LABEL: init_array_of_ptrs_to_structs:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    cntd x8
+; CHECK-NEXT:    index z0.d, #0, #1
+; CHECK-NEXT:    mov z2.d, x1
+; CHECK-NEXT:    mov z1.d, x8
+; CHECK-NEXT:    mov z3.d, #72 // =0x48
+; CHECK-NEXT:    neg x8, x8
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    and x8, x8, x2
+; CHECK-NEXT:  .LBB0_1: // %vector.body
+; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    movprfx z4, z2
+; CHECK-NEXT:    mla z4.d, p0/m, z0.d, z3.d
+; CHECK-NEXT:    decd x8
+; CHECK-NEXT:    add z0.d, z0.d, z1.d
+; CHECK-NEXT:    str z4, [x0]
+; CHECK-NEXT:    incb x0
+; CHECK-NEXT:    cbnz x8, .LBB0_1
+; CHECK-NEXT:  // %bb.2: // %middle.block
+; CHECK-NEXT:    ret
+entry:
+  %vscale = tail call i64 @llvm.vscale.i64()
+  %step = shl nuw nsw i64 %vscale, 1
+  %min.iters.check = icmp samesign ugt i64 %step, %num_arcs
+  %.not = sub nsw i64 0, %step
+  %n.vec = and i64 %.not, %num_arcs
+  %2 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %step, i64 0
+  %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  br label %vector.body
+
+vector.body:
+  %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+  %vec.ind = phi <vscale x 2 x i64> [ %2, %entry ], [ %vec.ind.next, %vector.body ]
+  %wide.gep = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %vec.ind
+  %3 = getelementptr inbounds nuw [8 x i8], ptr %arc_ptrs, i64 %index
+  store <vscale x 2 x ptr> %wide.gep, ptr %3, align 8
+  %index.next = add nuw i64 %index, %step
+  %vec.ind.next = add nuw nsw <vscale x 2 x i64> %vec.ind, %broadcast.splat
+  %4 = icmp eq i64 %index.next, %n.vec
+  br i1 %4, label %middle.block, label %vector.body
+
+middle.block:
+  ret void
+}
+
+define void @init_array_of_ptrs_to_structs_interleave4(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
+; CHECK-LABEL: init_array_of_ptrs_to_structs_interleave4:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    cntd x8
+; CHECK-NEXT:    mov w9, #2147483647 // =0x7fffffff
+; CHECK-NEXT:    cnth x10
+; CHECK-NEXT:    mov z0.d, x8
+; CHECK-NEXT:    cntw x8
+; CHECK-NEXT:    inch x9
+; CHECK-NEXT:    mov z1.d, x8
+; CHECK-NEXT:    cntd x8, all, mul #3
+; CHECK-NEXT:    index z2.d, #0, #1
+; CHECK-NEXT:    mov z3.d, x8
+; CHECK-NEXT:    mov z4.d, x10
+; CHECK-NEXT:    mov z5.d, x1
+; CHECK-NEXT:    mov z6.d, #72 // =0x48
+; CHECK-NEXT:    and x8, x9, x2
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    sub x8, x8, x2
+; CHECK-NEXT:  .LBB1_1: // %vector.body
+; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    add z7.d, z2.d, z0.d
+; CHECK-NEXT:    add z16.d, z2.d, z1.d
+; CHECK-NEXT:    movprfx z18, z5
+; CHECK-NEXT:    mla z18.d, p0/m, z2.d, z6.d
+; CHECK-NEXT:    add z17.d, z2.d, z3.d
+; CHECK-NEXT:    inch x8
+; CHECK-NEXT:    add z2.d, z2.d, z4.d
+; CHECK-NEXT:    mad z7.d, p0/m, z6.d, z5.d
+; CHECK-NEXT:    mad z16.d, p0/m, z6.d, z5.d
+; CHECK-NEXT:    mad z17.d, p0/m, z6.d, z5.d
+; CHECK-NEXT:    str z18, [x0]
+; CHECK-NEXT:    str z7, [x0, #1, mul vl]
+; CHECK-NEXT:    str z16, [x0, #2, mul vl]
+; CHECK-NEXT:    str z17, [x0, #3, mul vl]
+; CHECK-NEXT:    incb x0, all, mul #4
+; CHECK-NEXT:    cbnz x8, .LBB1_1
+; CHECK-NEXT:  // %bb.2: // %exit
+; CHECK-NEXT:    ret
+entry:
+  %0 = tail call i64 @llvm.vscale.i64()
+  %1 = shl nuw nsw i64 %0, 1
+  %2 = shl nuw nsw i64 %0, 1
+  %3 = shl nuw nsw i64 %0, 3
+  %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %1, i64 0
+  %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %4 = add nuw nsw i64 %3, 2147483647
+  %n.mod.vf = and i64 %4, %num_arcs
+  %n.vec = sub nsw i64 %num_arcs, %n.mod.vf
+  %5 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %invariant.op = add nuw <vscale x 2 x i64> %broadcast.splat, %broadcast.splat
+  %invariant.op26 = add nuw <vscale x 2 x i64> %invariant.op, %broadcast.splat
+  %invariant.op27 = add nuw <vscale x 2 x i64> %invariant.op26, %broadcast.splat
+  %6 = shl nuw nsw i64 %0, 6
+  %7 = mul i64 %0, 48
+  %8 = shl i64 %0, 5
+  %9 = shl i64 %0, 4
+  br label %vector.body
+
+vector.body:
+  %lsr.iv36 = phi ptr [ %scevgep37, %vector.body ], [ %arc_ptrs, %entry ]
+  %lsr.iv34 = phi i64 [ %lsr.iv.next35, %vector.body ], [ %n.vec, %entry ]
+  %vec.ind = phi <vscale x 2 x i64> [ %5, %entry ], [ %vec.ind.next, %vector.body ]
+  %step.add = add nuw <vscale x 2 x i64> %vec.ind, %broadcast.splat
+  %step.add.2.reass = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op
+  %step.add.3.reass = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op26
+  %wide.gep = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %vec.ind
+  %wide.gep10 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add
+  %wide.gep11 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add.2.reass
+  %wide.gep12 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add.3.reass
+  %scevgep40 = getelementptr i8, ptr %lsr.iv36, i64 %9
+  %scevgep39 = getelementptr i8, ptr %lsr.iv36, i64 %8
+  %scevgep38 = getelementptr i8, ptr %lsr.iv36, i64 %7
+  store <vscale x 2 x ptr> %wide.gep, ptr %lsr.iv36, align 8
+  store <vscale x 2 x ptr> %wide.gep10, ptr %scevgep40, align 8
+  store <vscale x 2 x ptr> %wide.gep11, ptr %scevgep39, align 8
+  store <vscale x 2 x ptr> %wide.gep12, ptr %scevgep38, align 8
+  %vec.ind.next = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op27
+  %lsr.iv.next35 = sub i64 %lsr.iv34, %3
+  %scevgep37 = getelementptr i8, ptr %lsr.iv36, i64 %6
+  %10 = icmp eq i64 %lsr.iv.next35, 0
+  br i1 %10, label %exit, label %vector.body
+
+exit:
+  ret void
+}
+
+define void @init_offset_array(i64 %0, ptr %1, ptr nofree writeonly captures(none) %2) #0 {
+; CHECK-LABEL: init_offset_array:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    cntd x8
+; CHECK-NEXT:    index z0.d, x0, #-1
+; CHECK-NEXT:    decb x2
+; CHECK-NEXT:    neg x8, x8
+; CHECK-NEXT:    cmp x0, #1
+; CHECK-NEXT:    mov z2.d, x1
+; CHECK-NEXT:    mov z1.d, x8
+; CHECK-NEXT:    mov z3.d, #56 // =0x38
+; CHECK-NEXT:    csinc x9, x0, xzr, hi
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    and x8, x9, x8
+; CHECK-NEXT:    add x9, x2, #8
+; CHECK-NEXT:  .LBB2_1: // %vector.body
+; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    movprfx z4, z0
+; CHECK-NEXT:    sub z4.d, z4.d, #1 // =0x1
+; CHECK-NEXT:    decd x8
+; CHECK-NEXT:    add z0.d, z0.d, z1.d
+; CHECK-NEXT:    movprfx z5, z2
+; CHECK-NEXT:    mla z5.d, p0/m, z4.d, z3.d
+; CHECK-NEXT:    fmov x10, d4
+; CHECK-NEXT:    rev z5.d, z5.d
+; CHECK-NEXT:    st1d { z5.d }, p0, [x9, x10, lsl #3]
+; CHECK-NEXT:    cbnz x8, .LBB2_1
+; CHECK-NEXT:  // %bb.2: // %exit
+; CHECK-NEXT:    ret
+entry:
+  %3 = tail call i64 @llvm.umax.i64(i64 %0, i64 1)
+  %4 = tail call i64 @llvm.vscale.i64()
+  %5 = shl nuw i64 %4, 1
+  %n.mod.vf = urem i64 %3, %5
+  %n.vec = sub i64 %3, %n.mod.vf
+  %6 = sub i64 %0, %n.vec
+  %7 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %0, i64 0
+  %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %8 = sub nsw <vscale x 2 x i64> %broadcast.splat, %7
+  %9 = sub nsw i64 0, %5
+  %broadcast.splatinsert1 = insertelement <vscale x 2 x i64> poison, i64 %9, i64 0
+  %broadcast.splat2 = shufflevector <vscale x 2 x i64> %broadcast.splatinsert1, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %10 = sub i64 1, %5
+  %invariant.gep = getelementptr [8 x i8], ptr %2, i64 %10
+  br label %vector.body
+
+vector.body:
+  %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+  %vec.ind = phi <vscale x 2 x i64> [ %8, %entry ], [ %vec.ind.next, %vector.body ]
+  %11 = add nsw <vscale x 2 x i64> %vec.ind, splat (i64 -1)
+  %12 = extractelement <vscale x 2 x i64> %11, i64 0
+  %wide.gep = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %11
+  %gep = getelementptr [8 x i8], ptr %invariant.gep, i64 %12
+  %reverse = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep)
+  store <vscale x 2 x ptr> %reverse, ptr %gep, align 8
+  %index.next = add nuw i64 %index, %5
+  %vec.ind.next = add nsw <vscale x 2 x i64> %vec.ind, %broadcast.splat2
+  %13 = icmp eq i64 %index.next, %n.vec
+  br i1 %13, label %exit, label %vector.body
+
+exit:
+  ret void
+}
+
+define void @init_offset_array_interleave4(i64 %0, ptr %1, ptr nofree writeonly captures(none) %2) #0 {
+; CHECK-LABEL: init_offset_array_interleave4:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    cntd x8
+; CHECK-NEXT:    mov x12, #-1 // =0xffffffffffffffff
+; CHECK-NEXT:    cnth x9
+; CHECK-NEXT:    mov z0.d, x8
+; CHECK-NEXT:    neg x8, x8
+; CHECK-NEXT:    index z3.d, x0, #-1
+; CHECK-NEXT:    incd x12
+; CHECK-NEXT:    add x8, x8, #1
+; CHECK-NEXT:    cmp x0, #1
+; CHECK-NEXT:    neg x10, x9
+; CHECK-NEXT:    mov x9, x8
+; CHECK-NEXT:    decd x8
+; CHECK-NEXT:    cntw x13
+; CHECK-NEXT:    mov z1.d, x1
+; CHECK-NEXT:    cntd x14, all, mul #3
+; CHECK-NEXT:    mov z2.d, #56 // =0x38
+; CHECK-NEXT:    csinc x11, x0, xzr, hi
+; CHECK-NEXT:    neg x13, x13
+; CHECK-NEXT:    neg x14, x14
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    and x10, x11, x10
+; CHECK-NEXT:    sub x11, x13, x12
+; CHECK-NEXT:    sub x12, x14, x12
+; CHECK-NEXT:  .LBB3_1: // %vector.body
+; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    sub z4.d, z3.d, z0.d
+; CHECK-NEXT:    sub z3.d, z3.d, #1 // =0x1
+; CHECK-NEXT:    dech x10
+; CHECK-NEXT:    sub z5.d, z4.d, z0.d
+; CHECK-NEXT:    mad z4.d, p0/m, z2.d, z1.d
+; CHECK-NEXT:    movprfx z7, z1
+; CHECK-NEXT:    mla z7.d, p0/m, z3.d, z2.d
+; CHECK-NEXT:    fmov x13, d3
+; CHECK-NEXT:    sub z6.d, z5.d, z0.d
+; CHECK-NEXT:    mad z5.d, p0/m, z2.d, z1.d
+; CHECK-NEXT:    sub z4.d, z4.d, #56 // =0x38
+; CHECK-NEXT:    rev z3.d, z7.d
+; CHECK-NEXT:    add x13, x2, x13, lsl #3
+; CHECK-NEXT:    movprfx z16, z1
+; CHECK-NEXT:    mla z16.d, p0/m, z6.d, z2.d
+; CHECK-NEXT:    sub z5.d, z5.d, #56 // =0x38
+; CHECK-NEXT:    rev z4.d, z4.d
+; CHECK-NEXT:    st1d { z3.d }, p0, [x13, x9, lsl #3]
+; CHECK-NEXT:    sub z3.d, z6.d, z0.d
+; CHECK-NEXT:    sub z16.d, z16.d, #56 // =0x38
+; CHECK-NEXT:    rev z5.d, z5.d
+; CHECK-NEXT:    st1d { z4.d }, p0, [x13, x8, lsl #3]
+; CHECK-NEXT:    rev z7.d, z16.d
+; CHECK-NEXT:    st1d { z5.d }, p0, [x13, x11, lsl #3]
+; CHECK-NEXT:    st1d { z7.d }, p0, [x13, x12, lsl #3]
+; CHECK-NEXT:    cbnz x10, .LBB3_1
+; CHECK-NEXT:  // %bb.2: // %exit
+; CHECK-NEXT:    ret
+entry:
+  %3 = tail call i64 @llvm.umax.i64(i64 %0, i64 1)
+  %4 = tail call i64 @llvm.vscale.i64()
+  %5 = shl nuw i64 %4, 1
+  %6 = shl nuw i64 %4, 3
+  %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %5, i64 0
+  %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %n.mod.vf = urem i64 %3, %6
+  %n.vec = sub i64 %3, %n.mod.vf
+  %7 = sub i64 %0, %n.vec
+  %8 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %broadcast.splatinsert2 = insertelement <vscale x 2 x i64> poison, i64 %0, i64 0
+  %broadcast.splat3 = shufflevector <vscale x 2 x i64> %broadcast.splatinsert2, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %9 = sub nsw <vscale x 2 x i64> %broadcast.splat3, %8
+  %10 = add nsw i64 %5, -1
+  %11 = sub i64 1, %5
+  %12 = sub i64 %11, %5
+  %13 = mul i64 %4, -4
+  %14 = sub i64 %13, %10
+  %15 = mul i64 %4, -6
+  %16 = sub i64 %15, %10
+  br label %vector.body
+
+vector.body:
+  %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+  %vec.ind = phi <vscale x 2 x i64> [ %9, %entry ], [ %30, %vector.body ]
+  %17 = sub <vscale x 2 x i64> %vec.ind, %broadcast.splat
+  %18 = sub <vscale x 2 x i64> %17, %broadcast.splat
+  %19 = sub <vscale x 2 x i64> %18, %broadcast.splat
+  %20 = add nsw <vscale x 2 x i64> %vec.ind, splat (i64 -1)
+  %21 = extractelement <vscale x 2 x i64> %20, i64 0
+  %22 = add nsw <vscale x 2 x i64> %17, splat (i64 -1)
+  %23 = add nsw <vscale x 2 x i64> %18, splat (i64 -1)
+  %24 = add nsw <vscale x 2 x i64> %19, splat (i64 -1)
+  %wide.gep = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %20
+  %wide.gep4 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %22
+  %wide.gep5 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %23
+  %wide.gep6 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %24
+  %25 = getelementptr inbounds nuw [8 x i8], ptr %2, i64 %21
+  %26 = getelementptr inbounds [8 x i8], ptr %25, i64 %11
+  %27 = getelementptr inbounds [8 x i8], ptr %25, i64 %12
+  %28 = getelementptr inbounds [8 x i8], ptr %25, i64 %14
+  %29 = getelementptr inbounds [8 x i8], ptr %25, i64 %16
+  %reverse = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep)
+  %reverse7 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep4)
+  %reverse8 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep5)
+  %reverse9 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep6)
+  store <vscale x 2 x ptr> %reverse, ptr %26, align 8
+  store <vscale x 2 x ptr> %reverse7, ptr %27, align 8
+  store <vscale x 2 x ptr> %reverse8, ptr %28, align 8
+  store <vscale x 2 x ptr> %reverse9, ptr %29, align 8
+  %index.next = add nuw i64 %index, %6
+  %30 = sub <vscale x 2 x i64> %19, %broadcast.splat
+  %31 = icmp eq i64 %index.next, %n.vec
+  br i1 %31, label %exit, label %vector.body
+
+exit:
+  ret void
+}
+
+attributes #0 = { nounwind vscale_range(1,16) "target-features"="+sve2" }

>From 8493b9cac6557cdbcf3c12102464b1191b6150cf Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 25 Sep 2026 12:09:47 +0000
Subject: [PATCH 2/2] Copy tests to CGP IR test directory

---
 .../AArch64/strength-reduce-vector-as-data.ll | 356 ++++++++++++++++++
 1 file changed, 356 insertions(+)
 create mode 100644 llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll

diff --git a/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
new file mode 100644
index 0000000000000..11cb04cde3c96
--- /dev/null
+++ b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
@@ -0,0 +1,356 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes='require<profile-summary>,function(codegenprepare)' -S %s | FileCheck %s
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
+; CHECK-LABEL: define void @init_array_of_ptrs_to_structs(
+; CHECK-SAME: ptr noalias [[ARC_PTRS:%.*]], ptr [[ARC_NEW:%.*]], i64 [[NUM_ARCS:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[VSCALE:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[STEP:%.*]] = shl nuw nsw i64 [[VSCALE]], 1
+; CHECK-NEXT:    [[DOTNOT:%.*]] = sub nsw i64 0, [[STEP]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = and i64 [[DOTNOT]], [[NUM_ARCS]]
+; CHECK-NEXT:    [[TMP0:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[STEP]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP0]], %[[ENTRY]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[VEC_IND]]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARC_PTRS]], i64 [[INDEX]]
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[WIDE_GEP]], ptr [[TMP1]], align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw nsw i64 [[TMP2]], 1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %vscale = tail call i64 @llvm.vscale.i64()
+  %step = shl nuw nsw i64 %vscale, 1
+  %min.iters.check = icmp samesign ugt i64 %step, %num_arcs
+  %.not = sub nsw i64 0, %step
+  %n.vec = and i64 %.not, %num_arcs
+  %2 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %step, i64 0
+  %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  br label %vector.body
+
+vector.body:
+  %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+  %vec.ind = phi <vscale x 2 x i64> [ %2, %entry ], [ %vec.ind.next, %vector.body ]
+  %wide.gep = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %vec.ind
+  %3 = getelementptr inbounds nuw [8 x i8], ptr %arc_ptrs, i64 %index
+  store <vscale x 2 x ptr> %wide.gep, ptr %3, align 8
+  %index.next = add nuw i64 %index, %step
+  %vec.ind.next = add nuw nsw <vscale x 2 x i64> %vec.ind, %broadcast.splat
+  %4 = icmp eq i64 %index.next, %n.vec
+  br i1 %4, label %middle.block, label %vector.body
+
+middle.block:
+  ret void
+}
+
+define void @init_array_of_ptrs_to_structs_interleave4(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
+; CHECK-LABEL: define void @init_array_of_ptrs_to_structs_interleave4(
+; CHECK-SAME: ptr noalias [[ARC_PTRS:%.*]], ptr [[ARC_NEW:%.*]], i64 [[NUM_ARCS:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP1:%.*]] = shl nuw nsw i64 [[TMP0]], 1
+; CHECK-NEXT:    [[TMP2:%.*]] = shl nuw nsw i64 [[TMP0]], 1
+; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw nsw i64 [[TMP0]], 3
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP1]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[TMP3]], 2147483647
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = and i64 [[TMP4]], [[NUM_ARCS]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub nsw i64 [[NUM_ARCS]], [[N_MOD_VF]]
+; CHECK-NEXT:    [[TMP5:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[INVARIANT_OP:%.*]] = add nuw <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[INVARIANT_OP26:%.*]] = add nuw <vscale x 2 x i64> [[INVARIANT_OP]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[INVARIANT_OP27:%.*]] = add nuw <vscale x 2 x i64> [[INVARIANT_OP26]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[LSR_IV36:%.*]] = phi ptr [ [[SCEVGEP37:%.*]], %[[VECTOR_BODY]] ], [ [[ARC_PTRS]], %[[ENTRY]] ]
+; CHECK-NEXT:    [[LSR_IV34:%.*]] = phi i64 [ [[LSR_IV_NEXT35:%.*]], %[[VECTOR_BODY]] ], [ [[N_VEC]], %[[ENTRY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP5]], %[[ENTRY]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[STEP_ADD:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[STEP_ADD_2_REASS:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[INVARIANT_OP]]
+; CHECK-NEXT:    [[STEP_ADD_3_REASS:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[INVARIANT_OP26]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[VEC_IND]]
+; CHECK-NEXT:    [[WIDE_GEP10:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[STEP_ADD]]
+; CHECK-NEXT:    [[WIDE_GEP11:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[STEP_ADD_2_REASS]]
+; CHECK-NEXT:    [[WIDE_GEP12:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[STEP_ADD_3_REASS]]
+; CHECK-NEXT:    [[TMP6:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP7:%.*]] = shl i64 [[TMP6]], 4
+; CHECK-NEXT:    [[SCEVGEP40:%.*]] = getelementptr i8, ptr [[LSR_IV36]], i64 [[TMP7]]
+; CHECK-NEXT:    [[TMP8:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP9:%.*]] = shl i64 [[TMP8]], 5
+; CHECK-NEXT:    [[SCEVGEP39:%.*]] = getelementptr i8, ptr [[LSR_IV36]], i64 [[TMP9]]
+; CHECK-NEXT:    [[TMP10:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP11:%.*]] = mul i64 [[TMP10]], 48
+; CHECK-NEXT:    [[SCEVGEP38:%.*]] = getelementptr i8, ptr [[LSR_IV36]], i64 [[TMP11]]
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[WIDE_GEP]], ptr [[LSR_IV36]], align 8
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[WIDE_GEP10]], ptr [[SCEVGEP40]], align 8
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[WIDE_GEP11]], ptr [[SCEVGEP39]], align 8
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[WIDE_GEP12]], ptr [[SCEVGEP38]], align 8
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[INVARIANT_OP27]]
+; CHECK-NEXT:    [[TMP12:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP13:%.*]] = shl nuw nsw i64 [[TMP12]], 3
+; CHECK-NEXT:    [[LSR_IV_NEXT35]] = sub i64 [[LSR_IV34]], [[TMP13]]
+; CHECK-NEXT:    [[TMP14:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP15:%.*]] = shl nuw nsw i64 [[TMP14]], 6
+; CHECK-NEXT:    [[SCEVGEP37]] = getelementptr i8, ptr [[LSR_IV36]], i64 [[TMP15]]
+; CHECK-NEXT:    [[TMP16:%.*]] = icmp eq i64 [[LSR_IV_NEXT35]], 0
+; CHECK-NEXT:    br i1 [[TMP16]], label %[[EXIT:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %0 = tail call i64 @llvm.vscale.i64()
+  %1 = shl nuw nsw i64 %0, 1
+  %2 = shl nuw nsw i64 %0, 1
+  %3 = shl nuw nsw i64 %0, 3
+  %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %1, i64 0
+  %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %4 = add nuw nsw i64 %3, 2147483647
+  %n.mod.vf = and i64 %4, %num_arcs
+  %n.vec = sub nsw i64 %num_arcs, %n.mod.vf
+  %5 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %invariant.op = add nuw <vscale x 2 x i64> %broadcast.splat, %broadcast.splat
+  %invariant.op26 = add nuw <vscale x 2 x i64> %invariant.op, %broadcast.splat
+  %invariant.op27 = add nuw <vscale x 2 x i64> %invariant.op26, %broadcast.splat
+  %6 = shl nuw nsw i64 %0, 6
+  %7 = mul i64 %0, 48
+  %8 = shl i64 %0, 5
+  %9 = shl i64 %0, 4
+  br label %vector.body
+
+vector.body:
+  %lsr.iv36 = phi ptr [ %scevgep37, %vector.body ], [ %arc_ptrs, %entry ]
+  %lsr.iv34 = phi i64 [ %lsr.iv.next35, %vector.body ], [ %n.vec, %entry ]
+  %vec.ind = phi <vscale x 2 x i64> [ %5, %entry ], [ %vec.ind.next, %vector.body ]
+  %step.add = add nuw <vscale x 2 x i64> %vec.ind, %broadcast.splat
+  %step.add.2.reass = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op
+  %step.add.3.reass = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op26
+  %wide.gep = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %vec.ind
+  %wide.gep10 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add
+  %wide.gep11 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add.2.reass
+  %wide.gep12 = getelementptr inbounds nuw [72 x i8], ptr %arc_new, <vscale x 2 x i64> %step.add.3.reass
+  %scevgep40 = getelementptr i8, ptr %lsr.iv36, i64 %9
+  %scevgep39 = getelementptr i8, ptr %lsr.iv36, i64 %8
+  %scevgep38 = getelementptr i8, ptr %lsr.iv36, i64 %7
+  store <vscale x 2 x ptr> %wide.gep, ptr %lsr.iv36, align 8
+  store <vscale x 2 x ptr> %wide.gep10, ptr %scevgep40, align 8
+  store <vscale x 2 x ptr> %wide.gep11, ptr %scevgep39, align 8
+  store <vscale x 2 x ptr> %wide.gep12, ptr %scevgep38, align 8
+  %vec.ind.next = add nuw <vscale x 2 x i64> %vec.ind, %invariant.op27
+  %lsr.iv.next35 = sub i64 %lsr.iv34, %3
+  %scevgep37 = getelementptr i8, ptr %lsr.iv36, i64 %6
+  %10 = icmp eq i64 %lsr.iv.next35, 0
+  br i1 %10, label %exit, label %vector.body
+
+exit:
+  ret void
+}
+
+define void @init_offset_array(i64 %0, ptr %1, ptr nofree writeonly captures(none) %2) #0 {
+; CHECK-LABEL: define void @init_offset_array(
+; CHECK-SAME: i64 [[TMP0:%.*]], ptr [[TMP1:%.*]], ptr nofree writeonly captures(none) [[TMP2:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP3:%.*]] = tail call i64 @llvm.umax.i64(i64 [[TMP0]], i64 1)
+; CHECK-NEXT:    [[TMP4:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP3]], [[TMP5]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP3]], [[N_MOD_VF]]
+; CHECK-NEXT:    [[TMP6:%.*]] = sub i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    [[TMP7:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP0]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP8:%.*]] = sub nsw <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP7]]
+; CHECK-NEXT:    [[TMP9:%.*]] = sub nsw i64 0, [[TMP5]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP9]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT2:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT1]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP10:%.*]] = sub i64 1, [[TMP5]]
+; CHECK-NEXT:    [[INVARIANT_GEP:%.*]] = getelementptr [8 x i8], ptr [[TMP2]], i64 [[TMP10]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP8]], %[[ENTRY]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP11:%.*]] = add nsw <vscale x 2 x i64> [[VEC_IND]], splat (i64 -1)
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <vscale x 2 x i64> [[TMP11]], i64 0
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr [56 x i8], ptr [[TMP1]], <vscale x 2 x i64> [[TMP11]]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr [8 x i8], ptr [[INVARIANT_GEP]], i64 [[TMP12]]
+; CHECK-NEXT:    [[REVERSE:%.*]] = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> [[WIDE_GEP]])
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[REVERSE]], ptr [[GEP]], align 8
+; CHECK-NEXT:    [[TMP13:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP14:%.*]] = shl nuw i64 [[TMP13]], 1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP14]]
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nsw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT2]]
+; CHECK-NEXT:    [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP15]], label %[[EXIT:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %3 = tail call i64 @llvm.umax.i64(i64 %0, i64 1)
+  %4 = tail call i64 @llvm.vscale.i64()
+  %5 = shl nuw i64 %4, 1
+  %n.mod.vf = urem i64 %3, %5
+  %n.vec = sub i64 %3, %n.mod.vf
+  %6 = sub i64 %0, %n.vec
+  %7 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %0, i64 0
+  %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %8 = sub nsw <vscale x 2 x i64> %broadcast.splat, %7
+  %9 = sub nsw i64 0, %5
+  %broadcast.splatinsert1 = insertelement <vscale x 2 x i64> poison, i64 %9, i64 0
+  %broadcast.splat2 = shufflevector <vscale x 2 x i64> %broadcast.splatinsert1, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %10 = sub i64 1, %5
+  %invariant.gep = getelementptr [8 x i8], ptr %2, i64 %10
+  br label %vector.body
+
+vector.body:
+  %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+  %vec.ind = phi <vscale x 2 x i64> [ %8, %entry ], [ %vec.ind.next, %vector.body ]
+  %11 = add nsw <vscale x 2 x i64> %vec.ind, splat (i64 -1)
+  %12 = extractelement <vscale x 2 x i64> %11, i64 0
+  %wide.gep = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %11
+  %gep = getelementptr [8 x i8], ptr %invariant.gep, i64 %12
+  %reverse = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep)
+  store <vscale x 2 x ptr> %reverse, ptr %gep, align 8
+  %index.next = add nuw i64 %index, %5
+  %vec.ind.next = add nsw <vscale x 2 x i64> %vec.ind, %broadcast.splat2
+  %13 = icmp eq i64 %index.next, %n.vec
+  br i1 %13, label %exit, label %vector.body
+
+exit:
+  ret void
+}
+
+define void @init_offset_array_interleave4(i64 %0, ptr %1, ptr nofree writeonly captures(none) %2) #0 {
+; CHECK-LABEL: define void @init_offset_array_interleave4(
+; CHECK-SAME: i64 [[TMP0:%.*]], ptr [[TMP1:%.*]], ptr nofree writeonly captures(none) [[TMP2:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP3:%.*]] = tail call i64 @llvm.umax.i64(i64 [[TMP0]], i64 1)
+; CHECK-NEXT:    [[TMP4:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-NEXT:    [[TMP6:%.*]] = shl nuw i64 [[TMP4]], 3
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP5]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP3]], [[TMP6]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP3]], [[N_MOD_VF]]
+; CHECK-NEXT:    [[TMP7:%.*]] = sub i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    [[TMP8:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP0]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT2]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP9:%.*]] = sub nsw <vscale x 2 x i64> [[BROADCAST_SPLAT3]], [[TMP8]]
+; CHECK-NEXT:    [[TMP10:%.*]] = add nsw i64 [[TMP5]], -1
+; CHECK-NEXT:    [[TMP11:%.*]] = sub i64 1, [[TMP5]]
+; CHECK-NEXT:    [[TMP12:%.*]] = sub i64 [[TMP11]], [[TMP5]]
+; CHECK-NEXT:    [[TMP13:%.*]] = mul i64 [[TMP4]], -4
+; CHECK-NEXT:    [[TMP14:%.*]] = sub i64 [[TMP13]], [[TMP10]]
+; CHECK-NEXT:    [[TMP15:%.*]] = mul i64 [[TMP4]], -6
+; CHECK-NEXT:    [[TMP16:%.*]] = sub i64 [[TMP15]], [[TMP10]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP9]], %[[ENTRY]] ], [ [[TMP32:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP17:%.*]] = sub <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP18:%.*]] = sub <vscale x 2 x i64> [[TMP17]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP19:%.*]] = sub <vscale x 2 x i64> [[TMP18]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP20:%.*]] = add nsw <vscale x 2 x i64> [[VEC_IND]], splat (i64 -1)
+; CHECK-NEXT:    [[TMP21:%.*]] = extractelement <vscale x 2 x i64> [[TMP20]], i64 0
+; CHECK-NEXT:    [[TMP22:%.*]] = add nsw <vscale x 2 x i64> [[TMP17]], splat (i64 -1)
+; CHECK-NEXT:    [[TMP23:%.*]] = add nsw <vscale x 2 x i64> [[TMP18]], splat (i64 -1)
+; CHECK-NEXT:    [[TMP24:%.*]] = add nsw <vscale x 2 x i64> [[TMP19]], splat (i64 -1)
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr [56 x i8], ptr [[TMP1]], <vscale x 2 x i64> [[TMP20]]
+; CHECK-NEXT:    [[WIDE_GEP4:%.*]] = getelementptr [56 x i8], ptr [[TMP1]], <vscale x 2 x i64> [[TMP22]]
+; CHECK-NEXT:    [[WIDE_GEP5:%.*]] = getelementptr [56 x i8], ptr [[TMP1]], <vscale x 2 x i64> [[TMP23]]
+; CHECK-NEXT:    [[WIDE_GEP6:%.*]] = getelementptr [56 x i8], ptr [[TMP1]], <vscale x 2 x i64> [[TMP24]]
+; CHECK-NEXT:    [[TMP25:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[TMP2]], i64 [[TMP21]]
+; CHECK-NEXT:    [[TMP26:%.*]] = getelementptr inbounds [8 x i8], ptr [[TMP25]], i64 [[TMP11]]
+; CHECK-NEXT:    [[TMP27:%.*]] = getelementptr inbounds [8 x i8], ptr [[TMP25]], i64 [[TMP12]]
+; CHECK-NEXT:    [[TMP28:%.*]] = getelementptr inbounds [8 x i8], ptr [[TMP25]], i64 [[TMP14]]
+; CHECK-NEXT:    [[TMP29:%.*]] = getelementptr inbounds [8 x i8], ptr [[TMP25]], i64 [[TMP16]]
+; CHECK-NEXT:    [[REVERSE:%.*]] = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> [[WIDE_GEP]])
+; CHECK-NEXT:    [[REVERSE7:%.*]] = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> [[WIDE_GEP4]])
+; CHECK-NEXT:    [[REVERSE8:%.*]] = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> [[WIDE_GEP5]])
+; CHECK-NEXT:    [[REVERSE9:%.*]] = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> [[WIDE_GEP6]])
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[REVERSE]], ptr [[TMP26]], align 8
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[REVERSE7]], ptr [[TMP27]], align 8
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[REVERSE8]], ptr [[TMP28]], align 8
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[REVERSE9]], ptr [[TMP29]], align 8
+; CHECK-NEXT:    [[TMP30:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP31:%.*]] = shl nuw i64 [[TMP30]], 3
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP31]]
+; CHECK-NEXT:    [[TMP32]] = sub <vscale x 2 x i64> [[TMP19]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP33]], label %[[EXIT:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %3 = tail call i64 @llvm.umax.i64(i64 %0, i64 1)
+  %4 = tail call i64 @llvm.vscale.i64()
+  %5 = shl nuw i64 %4, 1
+  %6 = shl nuw i64 %4, 3
+  %broadcast.splatinsert = insertelement <vscale x 2 x i64> poison, i64 %5, i64 0
+  %broadcast.splat = shufflevector <vscale x 2 x i64> %broadcast.splatinsert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %n.mod.vf = urem i64 %3, %6
+  %n.vec = sub i64 %3, %n.mod.vf
+  %7 = sub i64 %0, %n.vec
+  %8 = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %broadcast.splatinsert2 = insertelement <vscale x 2 x i64> poison, i64 %0, i64 0
+  %broadcast.splat3 = shufflevector <vscale x 2 x i64> %broadcast.splatinsert2, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %9 = sub nsw <vscale x 2 x i64> %broadcast.splat3, %8
+  %10 = add nsw i64 %5, -1
+  %11 = sub i64 1, %5
+  %12 = sub i64 %11, %5
+  %13 = mul i64 %4, -4
+  %14 = sub i64 %13, %10
+  %15 = mul i64 %4, -6
+  %16 = sub i64 %15, %10
+  br label %vector.body
+
+vector.body:
+  %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ]
+  %vec.ind = phi <vscale x 2 x i64> [ %9, %entry ], [ %30, %vector.body ]
+  %17 = sub <vscale x 2 x i64> %vec.ind, %broadcast.splat
+  %18 = sub <vscale x 2 x i64> %17, %broadcast.splat
+  %19 = sub <vscale x 2 x i64> %18, %broadcast.splat
+  %20 = add nsw <vscale x 2 x i64> %vec.ind, splat (i64 -1)
+  %21 = extractelement <vscale x 2 x i64> %20, i64 0
+  %22 = add nsw <vscale x 2 x i64> %17, splat (i64 -1)
+  %23 = add nsw <vscale x 2 x i64> %18, splat (i64 -1)
+  %24 = add nsw <vscale x 2 x i64> %19, splat (i64 -1)
+  %wide.gep = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %20
+  %wide.gep4 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %22
+  %wide.gep5 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %23
+  %wide.gep6 = getelementptr [56 x i8], ptr %1, <vscale x 2 x i64> %24
+  %25 = getelementptr inbounds nuw [8 x i8], ptr %2, i64 %21
+  %26 = getelementptr inbounds [8 x i8], ptr %25, i64 %11
+  %27 = getelementptr inbounds [8 x i8], ptr %25, i64 %12
+  %28 = getelementptr inbounds [8 x i8], ptr %25, i64 %14
+  %29 = getelementptr inbounds [8 x i8], ptr %25, i64 %16
+  %reverse = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep)
+  %reverse7 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep4)
+  %reverse8 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep5)
+  %reverse9 = tail call <vscale x 2 x ptr> @llvm.vector.reverse.nxv2p0(<vscale x 2 x ptr> %wide.gep6)
+  store <vscale x 2 x ptr> %reverse, ptr %26, align 8
+  store <vscale x 2 x ptr> %reverse7, ptr %27, align 8
+  store <vscale x 2 x ptr> %reverse8, ptr %28, align 8
+  store <vscale x 2 x ptr> %reverse9, ptr %29, align 8
+  %index.next = add nuw i64 %index, %6
+  %30 = sub <vscale x 2 x i64> %19, %broadcast.splat
+  %31 = icmp eq i64 %index.next, %n.vec
+  br i1 %31, label %exit, label %vector.body
+
+exit:
+  ret void
+}
+
+attributes #0 = { nounwind vscale_range(1,16) "target-features"="+sve2" }



More information about the llvm-commits mailing list