[llvm] [VPlan] Add transform to convert masked EE plan to branch + unmasked ops. (PR #216848)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 17 14:48:16 PDT 2026
https://github.com/fhahn created https://github.com/llvm/llvm-project/pull/216848
Treat the masked VPlan for early-exit with stores as canonical form, and
convert to a branch and unmasked memory ops.
>From 5cbc781043de8328cdca916323468c2cf5dfd3eb Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sun, 19 Jul 2026 15:30:41 +0100
Subject: [PATCH 1/2] [LV] Precommit tests for early-exit with stores.
---
.../X86/early-exit-with-stores-cost.ll | 74 ++
...y-exit-with-stores-masked-ops-not-legal.ll | 725 ++++++++++++++++++
2 files changed, 799 insertions(+)
create mode 100644 llvm/test/Transforms/LoopVectorize/X86/early-exit-with-stores-cost.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/early-exit-with-stores-masked-ops-not-legal.ll
diff --git a/llvm/test/Transforms/LoopVectorize/X86/early-exit-with-stores-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/early-exit-with-stores-cost.ll
new file mode 100644
index 0000000000000..6cd3810c1fec2
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/early-exit-with-stores-cost.ll
@@ -0,0 +1,74 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -p loop-vectorize -force-vector-width=4 -enable-early-exit-vectorization-with-side-effects -mtriple=x86_64-- -mattr=+avx512f,+avx512bw,+avx512vl -S %s | FileCheck %s
+
+; AVX512 has native masked <4 x i16> load/store so the masked form should be used.
+
+define void @single_store(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define void @single_store(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP0]], align 2
+; CHECK-NEXT: [[TMP1:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP1]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP2]])
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i16> @llvm.masked.load.v4i16.p0(ptr align 2 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i16> poison)
+; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> [[TMP4]], ptr align 2 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP5:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[TMP9]], 20
+; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP9]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %loop.header
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/early-exit-with-stores-masked-ops-not-legal.ll b/llvm/test/Transforms/LoopVectorize/early-exit-with-stores-masked-ops-not-legal.ll
new file mode 100644
index 0000000000000..2fa0e3d1c9ecd
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/early-exit-with-stores-masked-ops-not-legal.ll
@@ -0,0 +1,725 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -p loop-vectorize -force-vector-width=4 -enable-early-exit-vectorization-with-side-effects -S %s | FileCheck %s
+
+define void @single_store(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define void @single_store(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH2:.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP0]], align 2
+; CHECK-NEXT: [[TMP1:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; CHECK-NEXT: br i1 [[TMP3]], label %[[LOOP_LATCH2]], label %[[VECTOR_BODY_NONBAILING:.*]]
+; CHECK: [[VECTOR_BODY_NONBAILING]]:
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i16>, ptr [[TMP4]], align 2
+; CHECK-NEXT: [[TMP5:%.*]] = add nsw <4 x i16> [[WIDE_LOAD1]], splat (i16 1)
+; CHECK-NEXT: store <4 x i16> [[TMP5]], ptr [[TMP4]], align 2
+; CHECK-NEXT: br label %[[LOOP_LATCH2]]
+; CHECK: [[LOOP_LATCH2]]:
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP7:%.*]] = or i1 [[TMP3]], [[TMP6]]
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP8:%.*]] = select i1 [[TMP3]], i64 0, i64 4
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[TMP9]], 20
+; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP9]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %loop.header
+
+exit:
+ ret void
+}
+
+define void @two_stores(ptr dereferenceable(40) noalias %a, ptr dereferenceable(40) noalias %b, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define void @two_stores(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[A:%.*]], ptr noalias dereferenceable(40) [[B:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH2:.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP0]], align 2
+; CHECK-NEXT: [[TMP1:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; CHECK-NEXT: br i1 [[TMP3]], label %[[LOOP_LATCH2]], label %[[VECTOR_BODY_NONBAILING:.*]]
+; CHECK: [[VECTOR_BODY_NONBAILING]]:
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i16>, ptr [[TMP4]], align 2
+; CHECK-NEXT: store <4 x i16> [[WIDE_LOAD1]], ptr [[TMP4]], align 2
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i16> [[WIDE_LOAD1]], ptr [[TMP5]], align 2
+; CHECK-NEXT: br label %[[LOOP_LATCH2]]
+; CHECK: [[LOOP_LATCH2]]:
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP7:%.*]] = or i1 [[TMP3]], [[TMP6]]
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP8:%.*]] = select i1 [[TMP3]], i64 0, i64 4
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[TMP9]], 20
+; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP9]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[A_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[AV:%.*]] = load i16, ptr [[A_ADDR]], align 2
+; CHECK-NEXT: store i16 [[AV]], ptr [[A_ADDR]], align 2
+; CHECK-NEXT: [[B_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: store i16 [[AV]], ptr [[B_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %a.addr = getelementptr inbounds nuw i16, ptr %a, i64 %iv
+ %av = load i16, ptr %a.addr, align 2
+ store i16 %av, ptr %a.addr, align 2
+ %b.addr = getelementptr inbounds nuw i16, ptr %b, i64 %iv
+ store i16 %av, ptr %b.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %loop.header
+
+exit:
+ ret void
+}
+
+; The store is in the latch block, i.e. after the early-exit check.
+define void @store_in_latch(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define void @store_in_latch(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE12:.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT: [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP3]], align 2
+; CHECK-NEXT: [[TMP4:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP5:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP4]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP5]])
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP0]]
+; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP1]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP2]]
+; CHECK-NEXT: [[TMP10:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 0
+; CHECK-NEXT: br i1 [[TMP10]], label %[[PRED_LOAD_IF:.*]], label %[[PRED_LOAD_CONTINUE:.*]]
+; CHECK: [[PRED_LOAD_IF]]:
+; CHECK-NEXT: [[TMP11:%.*]] = load i16, ptr [[TMP6]], align 2
+; CHECK-NEXT: [[TMP12:%.*]] = insertelement <4 x i16> poison, i16 [[TMP11]], i64 0
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE]]
+; CHECK: [[PRED_LOAD_CONTINUE]]:
+; CHECK-NEXT: [[TMP13:%.*]] = phi <4 x i16> [ poison, %[[VECTOR_BODY]] ], [ [[TMP12]], %[[PRED_LOAD_IF]] ]
+; CHECK-NEXT: [[TMP14:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 1
+; CHECK-NEXT: br i1 [[TMP14]], label %[[PRED_LOAD_IF1:.*]], label %[[PRED_LOAD_CONTINUE2:.*]]
+; CHECK: [[PRED_LOAD_IF1]]:
+; CHECK-NEXT: [[TMP15:%.*]] = load i16, ptr [[TMP7]], align 2
+; CHECK-NEXT: [[TMP16:%.*]] = insertelement <4 x i16> [[TMP13]], i16 [[TMP15]], i64 1
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE2]]
+; CHECK: [[PRED_LOAD_CONTINUE2]]:
+; CHECK-NEXT: [[TMP17:%.*]] = phi <4 x i16> [ [[TMP13]], %[[PRED_LOAD_CONTINUE]] ], [ [[TMP16]], %[[PRED_LOAD_IF1]] ]
+; CHECK-NEXT: [[TMP18:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 2
+; CHECK-NEXT: br i1 [[TMP18]], label %[[PRED_LOAD_IF3:.*]], label %[[PRED_LOAD_CONTINUE4:.*]]
+; CHECK: [[PRED_LOAD_IF3]]:
+; CHECK-NEXT: [[TMP19:%.*]] = load i16, ptr [[TMP8]], align 2
+; CHECK-NEXT: [[TMP20:%.*]] = insertelement <4 x i16> [[TMP17]], i16 [[TMP19]], i64 2
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE4]]
+; CHECK: [[PRED_LOAD_CONTINUE4]]:
+; CHECK-NEXT: [[TMP21:%.*]] = phi <4 x i16> [ [[TMP17]], %[[PRED_LOAD_CONTINUE2]] ], [ [[TMP20]], %[[PRED_LOAD_IF3]] ]
+; CHECK-NEXT: [[TMP22:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 3
+; CHECK-NEXT: br i1 [[TMP22]], label %[[PRED_LOAD_IF5:.*]], label %[[PRED_LOAD_CONTINUE6:.*]]
+; CHECK: [[PRED_LOAD_IF5]]:
+; CHECK-NEXT: [[TMP23:%.*]] = load i16, ptr [[TMP9]], align 2
+; CHECK-NEXT: [[TMP24:%.*]] = insertelement <4 x i16> [[TMP21]], i16 [[TMP23]], i64 3
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE6]]
+; CHECK: [[PRED_LOAD_CONTINUE6]]:
+; CHECK-NEXT: [[TMP25:%.*]] = phi <4 x i16> [ [[TMP21]], %[[PRED_LOAD_CONTINUE4]] ], [ [[TMP24]], %[[PRED_LOAD_IF5]] ]
+; CHECK-NEXT: [[TMP26:%.*]] = add nsw <4 x i16> [[TMP25]], splat (i16 1)
+; CHECK-NEXT: br i1 [[TMP10]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; CHECK: [[PRED_STORE_IF]]:
+; CHECK-NEXT: [[TMP27:%.*]] = extractelement <4 x i16> [[TMP26]], i64 0
+; CHECK-NEXT: store i16 [[TMP27]], ptr [[TMP6]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE]]
+; CHECK: [[PRED_STORE_CONTINUE]]:
+; CHECK-NEXT: br i1 [[TMP14]], label %[[PRED_STORE_IF7:.*]], label %[[PRED_STORE_CONTINUE8:.*]]
+; CHECK: [[PRED_STORE_IF7]]:
+; CHECK-NEXT: [[TMP28:%.*]] = extractelement <4 x i16> [[TMP26]], i64 1
+; CHECK-NEXT: store i16 [[TMP28]], ptr [[TMP7]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE8]]
+; CHECK: [[PRED_STORE_CONTINUE8]]:
+; CHECK-NEXT: br i1 [[TMP18]], label %[[PRED_STORE_IF9:.*]], label %[[PRED_STORE_CONTINUE10:.*]]
+; CHECK: [[PRED_STORE_IF9]]:
+; CHECK-NEXT: [[TMP29:%.*]] = extractelement <4 x i16> [[TMP26]], i64 2
+; CHECK-NEXT: store i16 [[TMP29]], ptr [[TMP8]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE10]]
+; CHECK: [[PRED_STORE_CONTINUE10]]:
+; CHECK-NEXT: br i1 [[TMP22]], label %[[PRED_STORE_IF11:.*]], label %[[PRED_STORE_CONTINUE12]]
+; CHECK: [[PRED_STORE_IF11]]:
+; CHECK-NEXT: [[TMP30:%.*]] = extractelement <4 x i16> [[TMP26]], i64 3
+; CHECK-NEXT: store i16 [[TMP30]], ptr [[TMP9]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE12]]
+; CHECK: [[PRED_STORE_CONTINUE12]]:
+; CHECK-NEXT: [[TMP31:%.*]] = freeze <4 x i1> [[TMP4]]
+; CHECK-NEXT: [[TMP32:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP31]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP34:%.*]] = or i1 [[TMP32]], [[TMP33]]
+; CHECK-NEXT: br i1 [[TMP34]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP35:%.*]] = add i64 [[INDEX]], [[TMP5]]
+; CHECK-NEXT: [[TMP36:%.*]] = icmp eq i64 [[TMP35]], 20
+; CHECK-NEXT: br i1 [[TMP36]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP35]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %loop.latch
+
+loop.latch:
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %loop.header
+
+exit:
+ ret void
+}
+
+; A value defined in the body (%data) is live out of the loop.
+define i16 @body_live_out(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define i16 @body_live_out(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE12:.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT: [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP3]], align 2
+; CHECK-NEXT: [[TMP4:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP5:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP4]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP5]])
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP0]]
+; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP1]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP2]]
+; CHECK-NEXT: [[TMP10:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 0
+; CHECK-NEXT: br i1 [[TMP10]], label %[[PRED_LOAD_IF:.*]], label %[[PRED_LOAD_CONTINUE:.*]]
+; CHECK: [[PRED_LOAD_IF]]:
+; CHECK-NEXT: [[TMP11:%.*]] = load i16, ptr [[TMP6]], align 2
+; CHECK-NEXT: [[TMP12:%.*]] = insertelement <4 x i16> poison, i16 [[TMP11]], i64 0
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE]]
+; CHECK: [[PRED_LOAD_CONTINUE]]:
+; CHECK-NEXT: [[TMP13:%.*]] = phi <4 x i16> [ poison, %[[VECTOR_BODY]] ], [ [[TMP12]], %[[PRED_LOAD_IF]] ]
+; CHECK-NEXT: [[TMP14:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 1
+; CHECK-NEXT: br i1 [[TMP14]], label %[[PRED_LOAD_IF1:.*]], label %[[PRED_LOAD_CONTINUE2:.*]]
+; CHECK: [[PRED_LOAD_IF1]]:
+; CHECK-NEXT: [[TMP15:%.*]] = load i16, ptr [[TMP7]], align 2
+; CHECK-NEXT: [[TMP16:%.*]] = insertelement <4 x i16> [[TMP13]], i16 [[TMP15]], i64 1
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE2]]
+; CHECK: [[PRED_LOAD_CONTINUE2]]:
+; CHECK-NEXT: [[TMP17:%.*]] = phi <4 x i16> [ [[TMP13]], %[[PRED_LOAD_CONTINUE]] ], [ [[TMP16]], %[[PRED_LOAD_IF1]] ]
+; CHECK-NEXT: [[TMP18:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 2
+; CHECK-NEXT: br i1 [[TMP18]], label %[[PRED_LOAD_IF3:.*]], label %[[PRED_LOAD_CONTINUE4:.*]]
+; CHECK: [[PRED_LOAD_IF3]]:
+; CHECK-NEXT: [[TMP19:%.*]] = load i16, ptr [[TMP8]], align 2
+; CHECK-NEXT: [[TMP20:%.*]] = insertelement <4 x i16> [[TMP17]], i16 [[TMP19]], i64 2
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE4]]
+; CHECK: [[PRED_LOAD_CONTINUE4]]:
+; CHECK-NEXT: [[TMP21:%.*]] = phi <4 x i16> [ [[TMP17]], %[[PRED_LOAD_CONTINUE2]] ], [ [[TMP20]], %[[PRED_LOAD_IF3]] ]
+; CHECK-NEXT: [[TMP22:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 3
+; CHECK-NEXT: br i1 [[TMP22]], label %[[PRED_LOAD_IF5:.*]], label %[[PRED_LOAD_CONTINUE6:.*]]
+; CHECK: [[PRED_LOAD_IF5]]:
+; CHECK-NEXT: [[TMP23:%.*]] = load i16, ptr [[TMP9]], align 2
+; CHECK-NEXT: [[TMP24:%.*]] = insertelement <4 x i16> [[TMP21]], i16 [[TMP23]], i64 3
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE6]]
+; CHECK: [[PRED_LOAD_CONTINUE6]]:
+; CHECK-NEXT: [[TMP25:%.*]] = phi <4 x i16> [ [[TMP21]], %[[PRED_LOAD_CONTINUE4]] ], [ [[TMP24]], %[[PRED_LOAD_IF5]] ]
+; CHECK-NEXT: [[TMP26:%.*]] = add nsw <4 x i16> [[TMP25]], splat (i16 1)
+; CHECK-NEXT: br i1 [[TMP10]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; CHECK: [[PRED_STORE_IF]]:
+; CHECK-NEXT: [[TMP27:%.*]] = extractelement <4 x i16> [[TMP26]], i64 0
+; CHECK-NEXT: store i16 [[TMP27]], ptr [[TMP6]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE]]
+; CHECK: [[PRED_STORE_CONTINUE]]:
+; CHECK-NEXT: br i1 [[TMP14]], label %[[PRED_STORE_IF7:.*]], label %[[PRED_STORE_CONTINUE8:.*]]
+; CHECK: [[PRED_STORE_IF7]]:
+; CHECK-NEXT: [[TMP28:%.*]] = extractelement <4 x i16> [[TMP26]], i64 1
+; CHECK-NEXT: store i16 [[TMP28]], ptr [[TMP7]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE8]]
+; CHECK: [[PRED_STORE_CONTINUE8]]:
+; CHECK-NEXT: br i1 [[TMP18]], label %[[PRED_STORE_IF9:.*]], label %[[PRED_STORE_CONTINUE10:.*]]
+; CHECK: [[PRED_STORE_IF9]]:
+; CHECK-NEXT: [[TMP29:%.*]] = extractelement <4 x i16> [[TMP26]], i64 2
+; CHECK-NEXT: store i16 [[TMP29]], ptr [[TMP8]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE10]]
+; CHECK: [[PRED_STORE_CONTINUE10]]:
+; CHECK-NEXT: br i1 [[TMP22]], label %[[PRED_STORE_IF11:.*]], label %[[PRED_STORE_CONTINUE12]]
+; CHECK: [[PRED_STORE_IF11]]:
+; CHECK-NEXT: [[TMP30:%.*]] = extractelement <4 x i16> [[TMP26]], i64 3
+; CHECK-NEXT: store i16 [[TMP30]], ptr [[TMP9]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE12]]
+; CHECK: [[PRED_STORE_CONTINUE12]]:
+; CHECK-NEXT: [[TMP31:%.*]] = freeze <4 x i1> [[TMP4]]
+; CHECK-NEXT: [[TMP32:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP31]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP34:%.*]] = or i1 [[TMP32]], [[TMP33]]
+; CHECK-NEXT: br i1 [[TMP34]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP35:%.*]] = extractelement <4 x i16> [[TMP25]], i64 3
+; CHECK-NEXT: [[TMP36:%.*]] = add i64 [[INDEX]], [[TMP5]]
+; CHECK-NEXT: [[TMP37:%.*]] = icmp eq i64 [[TMP36]], 20
+; CHECK-NEXT: br i1 [[TMP37]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP36]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[DATA_LCSSA:%.*]] = phi i16 [ [[DATA]], %[[LOOP_LATCH]] ], [ [[DATA]], %[[LOOP_HEADER]] ], [ [[TMP35]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i16 [[DATA_LCSSA]]
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %loop.header
+
+exit:
+ %data.lcssa = phi i16 [ %data, %loop.latch ], [ %data, %loop.header ]
+ ret i16 %data.lcssa
+}
+
+; Trip count (21) is not a multiple of the vector width (4): the vector loop runs
+; for the first 20 iterations (vector trip count) and the middle block compares
+; the resume IV against the full trip count, so the scalar loop runs the tail.
+define void @nondivisible_trip_count(ptr dereferenceable(42) noalias %array, ptr align 2 dereferenceable(42) readonly %pred) {
+; CHECK-LABEL: define void @nondivisible_trip_count(
+; CHECK-SAME: ptr noalias dereferenceable(42) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(42) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH2:.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP0]], align 2
+; CHECK-NEXT: [[TMP1:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; CHECK-NEXT: br i1 [[TMP3]], label %[[LOOP_LATCH2]], label %[[VECTOR_BODY_NONBAILING:.*]]
+; CHECK: [[VECTOR_BODY_NONBAILING]]:
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i16>, ptr [[TMP4]], align 2
+; CHECK-NEXT: [[TMP5:%.*]] = add nsw <4 x i16> [[WIDE_LOAD1]], splat (i16 1)
+; CHECK-NEXT: store <4 x i16> [[TMP5]], ptr [[TMP4]], align 2
+; CHECK-NEXT: br label %[[LOOP_LATCH2]]
+; CHECK: [[LOOP_LATCH2]]:
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP7:%.*]] = or i1 [[TMP3]], [[TMP6]]
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP8:%.*]] = select i1 [[TMP3]], i64 0, i64 4
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[TMP9]], 21
+; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP9]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 21
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 21
+ br i1 %counted.cond, label %exit, label %loop.header
+
+exit:
+ ret void
+}
+
+; The store's address (%array[%iv + 16]) is offset from the induction variable by
+; a constant.
+define void @offset_store_address(ptr dereferenceable(80) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define void @offset_store_address(
+; CHECK-SAME: ptr noalias dereferenceable(80) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH2:.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP0]], align 2
+; CHECK-NEXT: [[TMP1:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; CHECK-NEXT: br i1 [[TMP3]], label %[[LOOP_LATCH2]], label %[[VECTOR_BODY_NONBAILING:.*]]
+; CHECK: [[VECTOR_BODY_NONBAILING]]:
+; CHECK-NEXT: [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP4]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i16>, ptr [[TMP5]], align 2
+; CHECK-NEXT: [[TMP6:%.*]] = add nsw <4 x i16> [[WIDE_LOAD1]], splat (i16 3)
+; CHECK-NEXT: store <4 x i16> [[TMP6]], ptr [[TMP5]], align 2
+; CHECK-NEXT: br label %[[LOOP_LATCH2]]
+; CHECK: [[LOOP_LATCH2]]:
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP8:%.*]] = or i1 [[TMP3]], [[TMP7]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP9:%.*]] = select i1 [[TMP3]], i64 0, i64 4
+; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[INDEX]], [[TMP9]]
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[TMP10]], 20
+; CHECK-NEXT: br i1 [[TMP11]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP10]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[OFF:%.*]] = add nuw nsw i64 [[IV]], 16
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[OFF]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 3
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %off = add nuw nsw i64 %iv, 16
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %off
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 3
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %loop.header
+
+exit:
+ ret void
+}
+
+; A load from a loop-invariant address (%cfg) in the body.
+define void @single_store_with_invariant_load(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred, ptr noalias readonly %cfg) {
+; CHECK-LABEL: define void @single_store_with_invariant_load(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]], ptr noalias readonly [[CFG:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH2:.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP0]], align 2
+; CHECK-NEXT: [[TMP1:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; CHECK-NEXT: br i1 [[TMP3]], label %[[LOOP_LATCH2]], label %[[VECTOR_BODY_NONBAILING:.*]]
+; CHECK: [[VECTOR_BODY_NONBAILING]]:
+; CHECK-NEXT: [[TMP4:%.*]] = load i16, ptr [[CFG]], align 2
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i16> poison, i16 [[TMP4]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i16> [[BROADCAST_SPLATINSERT]], <4 x i16> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i16>, ptr [[TMP5]], align 2
+; CHECK-NEXT: [[TMP6:%.*]] = add nsw <4 x i16> [[WIDE_LOAD1]], splat (i16 1)
+; CHECK-NEXT: [[TMP7:%.*]] = add nsw <4 x i16> [[TMP6]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: store <4 x i16> [[TMP7]], ptr [[TMP5]], align 2
+; CHECK-NEXT: br label %[[LOOP_LATCH2]]
+; CHECK: [[LOOP_LATCH2]]:
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP9:%.*]] = or i1 [[TMP3]], [[TMP8]]
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP10:%.*]] = select i1 [[TMP3]], i64 0, i64 4
+; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[INDEX]], [[TMP10]]
+; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[TMP11]], 20
+; CHECK-NEXT: br i1 [[TMP12]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP11]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[CFGVAL:%.*]] = load i16, ptr [[CFG]], align 2
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC0:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[INC0]], [[CFGVAL]]
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP16:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %cfgval = load i16, ptr %cfg, align 2
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc0 = add nsw i16 %data, 1
+ %inc = add nsw i16 %inc0, %cfgval
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %loop.header
+
+exit:
+ ret void
+}
+
+define void @i32_iv(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define void @i32_iv(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH2:.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i32 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP0]], align 2
+; CHECK-NEXT: [[TMP1:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; CHECK-NEXT: br i1 [[TMP3]], label %[[LOOP_LATCH2]], label %[[VECTOR_BODY_NONBAILING:.*]]
+; CHECK: [[VECTOR_BODY_NONBAILING]]:
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i32 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i16>, ptr [[TMP4]], align 2
+; CHECK-NEXT: [[TMP5:%.*]] = add nsw <4 x i16> [[WIDE_LOAD1]], splat (i16 1)
+; CHECK-NEXT: store <4 x i16> [[TMP5]], ptr [[TMP4]], align 2
+; CHECK-NEXT: br label %[[LOOP_LATCH2]]
+; CHECK: [[LOOP_LATCH2]]:
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i32 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP7:%.*]] = or i1 [[TMP3]], [[TMP6]]
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP8:%.*]] = select i1 [[TMP3]], i32 0, i32 4
+; CHECK-NEXT: [[TMP9:%.*]] = add i32 [[INDEX]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i32 [[TMP9]], 20
+; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[TMP9]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i32 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i32 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i32 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i32 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP18:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i32 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i32 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i32 %iv, 1
+ %counted.cond = icmp eq i32 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %loop.header
+
+exit:
+ ret void
+}
>From ed222497387f049a05136e9347d3ce6f4433e20a Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Mon, 17 Aug 2026 22:45:30 +0100
Subject: [PATCH 2/2] [VPlan] Add transform to convert masked EE plan to branch
+ unmasked ops.
Treat the masked VPlan for early-exit with stores as canonical form, and
convert to a branch and unmasked memory ops.
---
.../Vectorize/LoopVectorizationPlanner.cpp | 2 +-
.../Transforms/Vectorize/LoopVectorize.cpp | 4 +
llvm/lib/Transforms/Vectorize/VPlan.h | 6 +
.../Transforms/Vectorize/VPlanTransforms.cpp | 141 +++++++++++++++++-
.../Transforms/Vectorize/VPlanTransforms.h | 10 ++
llvm/lib/Transforms/Vectorize/VPlanValue.h | 4 +-
...rized_conditional_ops_uncountable_exits.ll | 73 +--------
7 files changed, 172 insertions(+), 68 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index ff9b9171d8c8c..c431ed4164b6a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -57,7 +57,7 @@ cl::opt<bool> llvm::PreferInLoopReductions(
/// Note: This currently only applies to `llvm.masked.load` and
/// `llvm.masked.store`. TODO: Extend this to cover other operations as needed.
-static cl::opt<bool> ForceTargetSupportsMaskedMemoryOps(
+cl::opt<bool> ForceTargetSupportsMaskedMemoryOps(
"force-target-supports-masked-memory-ops", cl::init(false), cl::Hidden,
cl::desc("Assume the target supports masked memory operations (used for "
"testing)."));
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index c12a4f562f600..0034bb0f140db 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5491,6 +5491,10 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
if (!VPlan1)
return;
+ if (Legal->hasUncountableExitWithSideEffects())
+ RUN_VPLAN_PASS(VPlanTransforms::convertMaskedEarlyExitToBailToScalar,
+ *VPlan1, TTI);
+
if (!OrigLoop->isInnermost()) {
// For outer loops, computeMaxVF returns a single non-scalar VF; build a
// plan for that VF only.
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 4bb18e46c64ad..28fa0e7b1336e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1501,6 +1501,12 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
return isMasked() ? getOperand(getNumOperands() - 1) : nullptr;
}
+ /// Remove the mask from a masked VPInstruction, so it is unconditional.
+ void dropMask() {
+ assert(isMasked() && "recipe is not masked");
+ VPUser::removeOperand(getNumOperands() - 1);
+ }
+
/// Returns an iterator range over the operands excluding the mask operand
/// if present.
iterator_range<operand_iterator> operandsWithoutMask() {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index e75e9cf4e9103..9ce3a13d0e40d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -43,6 +43,8 @@ using namespace llvm;
using namespace VPlanPatternMatch;
using namespace SCEVPatternMatch;
+extern cl::opt<bool> ForceTargetSupportsMaskedMemoryOps;
+
/// If the pointer operand \p Addr of a memory access is an affine AddRec
/// w.r.t. \p L with a constant stride, return the stride in units of
/// \p AccessTy. Otherwise return std::nullopt.
@@ -3417,6 +3419,142 @@ bool VPlanTransforms::handleUncountableEarlyExits(
return true;
}
+/// Returns true if masking any of the loads or stores \p MaskedMemOps would be
+/// scalarized on \p TTI. Unlike VFSelectionContext::isLegalMaskedLoadOrStore,
+/// this ignores -force-target-supports-masked-memory-ops: that flag forces the
+/// masked form to be built, but the bail decision must reflect the real target.
+static bool maskingWouldScalarize(ArrayRef<VPInstruction *> MaskedMemOps,
+ const TargetTransformInfo &TTI) {
+ if (ForceTargetSupportsMaskedMemoryOps)
+ return false;
+ return !all_of(MaskedMemOps, [&TTI](VPInstruction *VPI) {
+ bool IsLoad = VPI->getOpcode() == Instruction::Load;
+ Type *Ty = (IsLoad ? VPI : VPI->getOperand(0))->getScalarType();
+ unsigned AS = cast<PointerType>(VPI->getOperand(!IsLoad)->getScalarType())
+ ->getAddressSpace();
+ // TODO: Model the alignment on the Load/Store VPInstruction, so it does not
+ // have to be retrieved from the underlying instruction.
+ Align Alignment = getLoadStoreAlignment(VPI->getUnderlyingInstr());
+ return IsLoad ? TTI.isLegalMaskedLoad(Ty, Alignment, AS)
+ : TTI.isLegalMaskedStore(Ty, Alignment, AS);
+ });
+}
+
+void VPlanTransforms::convertMaskedEarlyExitToBailToScalar(
+ VPlan &Plan, const TargetTransformInfo &TTI) {
+ // Convert the masked (partial-commit) form built by
+ // handleUncountableExitsWithSideEffects, which masks the memory operations
+ // with active-lane-mask(0, first-active-lane(vp<%cond>), 1) and resumes the
+ // scalar loop at IV + first-active-lane, to
+ //
+ // vector.body: (header)
+ // <IV + condition recipes>
+ // EMIT vp<%any> = any-of vp<%cond>
+ // EMIT branch-on-cond vp<%any> -> latch (skip) / body
+ // vector.body.nonbailing: (body)
+ // <unmasked memory ops>
+ // latch:
+ // EMIT branch-on-two-conds vp<%any>, <counted-exit-cond>
+ //
+ // resuming the scalar loop at IV + select(vp<%any>, 0, VF).
+ VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
+ if (!LoopRegion)
+ return;
+ VPBasicBlock *HeaderVPBB = LoopRegion->getEntryBasicBlock();
+ VPBasicBlock *LatchVPBB = LoopRegion->getExitingBasicBlock();
+ // The masked form has a separate header and latch, chained header -> latch.
+ if (HeaderVPBB->getSingleSuccessor() != LatchVPBB)
+ return;
+
+ // Match the any-of of the uncountable exit condition in the latch.
+ VPInstruction *AnyOfR;
+ VPValue *Cond;
+ if (!match(LatchVPBB->getTerminator(),
+ m_BranchOnTwoConds(m_CombineAnd(m_VPInstruction(AnyOfR),
+ m_AnyOf(m_VPValue(Cond))),
+ m_VPValue())) ||
+ AnyOfR->getParent() != LatchVPBB)
+ return;
+
+ // Match the mask on the memory operations to skip on a bail; the
+ // first-active-lane may be cast to the IV's type.
+ auto *MaskR = find_singleton<VPInstruction>(
+ *HeaderVPBB, [Cond](VPRecipeBase &R, bool) -> VPInstruction * {
+ if (match(&R, m_VPInstruction<VPInstruction::ActiveLaneMask>(
+ m_ZeroInt(), m_ZExtOrTruncOrSelf(m_FirstActiveLane(
+ m_Specific(Cond))))))
+ return cast<VPInstruction>(&R);
+ return nullptr;
+ });
+ if (!MaskR)
+ return;
+
+ // The recipes after the mask become the non-bailing body.
+ auto Body = make_range(std::next(MaskR->getIterator()), HeaderVPBB->end());
+ SmallPtrSet<VPRecipeBase *, 16> BodyRecipes;
+ for (VPRecipeBase &R : Body)
+ BodyRecipes.insert(&R);
+
+ // All users of the mask must be loads or stores in the body. The producer
+ // masks every side-effecting memory operation with this single mask, so this
+ // also guarantees none is left on the bail path.
+ SmallVector<VPInstruction *> MaskedMemOps;
+ for (VPUser *U : MaskR->users()) {
+ auto *VPI = dyn_cast<VPInstruction>(U);
+ if (!VPI ||
+ !is_contained({Instruction::Load, Instruction::Store},
+ VPI->getOpcode()) ||
+ VPI->getMask() != MaskR || !BodyRecipes.contains(VPI))
+ return;
+ MaskedMemOps.push_back(VPI);
+ }
+
+ // The body is skipped on a bail, so must not define values used outside it.
+ // TODO: Support body live-outs by recomputing them in the scalar loop.
+ for (VPRecipeBase &R : Body)
+ for (VPValue *Def : R.definedValues())
+ if (any_of(Def->users(), [&BodyRecipes](VPUser *U) {
+ return !BodyRecipes.contains(cast<VPRecipeBase>(U));
+ }))
+ return;
+
+ // Bailing trades extra control flow and re-executed iterations for unmasked
+ // memory operations; only worthwhile if masking them would be scalarized.
+ if (!maskingWouldScalarize(MaskedMemOps, TTI))
+ return;
+
+ // Split the header after the mask and unmask the memory operations, which
+ // leaves the mask itself dead.
+ VPBasicBlock *BodyVPBB = HeaderVPBB->splitAt(std::next(MaskR->getIterator()));
+ BodyVPBB->setName("vector.body.nonbailing");
+ for (VPInstruction *VPI : MaskedMemOps)
+ VPI->dropMask();
+ VPValue *FirstActive = MaskR->getOperand(1);
+ MaskR->eraseFromParent();
+
+ // Move the any-of to the header to feed the skip branch; the latch's
+ // BranchOnTwoConds keeps using it.
+ AnyOfR->moveBefore(*HeaderVPBB, HeaderVPBB->end());
+
+ // Add the skip edge header -> latch and make it the true successor, so the
+ // branch skips the body.
+ VPBlockUtils::connectBlocks(HeaderVPBB, LatchVPBB);
+ HeaderVPBB->swapSuccessors();
+ VPBuilder(HeaderVPBB).createNaryOp(VPInstruction::BranchOnCond, AnyOfR);
+
+ // Resume the scalar loop at IV + select(AnyOf, 0, VF).
+ VPBasicBlock *MiddleVPBB = Plan.getMiddleBlock();
+ Type *IVScalarTy = FirstActive->getScalarType();
+ VPBuilder MiddleBuilder(MiddleVPBB, MiddleVPBB->getFirstNonPhi());
+ VPValue *VFAsIVTy = MiddleBuilder.createScalarZExtOrTrunc(
+ &Plan.getVF(), IVScalarTy, DebugLoc());
+ VPValue *CommittedLanes =
+ MiddleBuilder.createSelect(AnyOfR, Plan.getZero(IVScalarTy), VFAsIVTy);
+ FirstActive->replaceAllUsesWith(CommittedLanes);
+ vputils::recursivelyDeleteDeadRecipes(FirstActive);
+ Plan.setUF(1);
+}
+
/// This function tries convert extended in-loop reductions to
/// VPExpressionRecipe and clamp the \p Range if it is beneficial and
/// valid. The created recipe must be decomposed to its constituent
@@ -5441,7 +5579,8 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
// A predicated access can only be widened (rather than scalarized) if
// the target supports a masked load/store for it.
// TODO: Determine if a load/store needs predication directly in VPlan.
- bool IsPredicated = RecipeBuilder.isPredicatedInst(I);
+ bool IsPredicated =
+ RecipeBuilder.isPredicatedInst(I) && VPI->isMasked();
if (IsPredicated && !CostCtx.Config.isLegalMaskedLoadOrStore(
IsLoad, ScalarTy, getLoadStoreAlignment(I),
getLoadStoreAddressSpace(I)))
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index e1832cd2de27a..48c6710184e65 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -547,6 +547,16 @@ struct VPlanTransforms {
/// \p Plan.
static void introduceMasksAndLinearize(VPlan &Plan);
+ /// Convert the masked (partial-commit) form of an uncountable early-exit loop
+ /// with side effects to the bail-to-scalar form: branch around the loop body
+ /// if any lane takes the exit, leaving the memory operations unmasked, and
+ /// re-execute the iteration in the scalar loop. Must run after
+ /// introduceMasksAndLinearize. Does nothing if \p Plan is not in the expected
+ /// masked form, or if masking would not be scalarized on \p TTI.
+ static void
+ convertMaskedEarlyExitToBailToScalar(VPlan &Plan,
+ const TargetTransformInfo &TTI);
+
/// Replace a VPWidenCanonicalIVRecipe if it is present in \p Plan, with a
/// VPWidenIntOrFpInductionRecipe, provided it would not cause additional
/// spills for \p VF at unroll factor \p UF.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanValue.h b/llvm/lib/Transforms/Vectorize/VPlanValue.h
index b318c38c796a3..80c68013b592f 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanValue.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanValue.h
@@ -399,10 +399,12 @@ class VPMultiDefValue : public VPRecipeValue {
/// This class augments VPValue with operands which provide the inverse def-use
/// edges from VPValue's users to their defs.
class LLVM_ABI_FOR_TEST VPUser {
- /// Grant access to removeOperand for VPPhiAccessors, the only supported user.
+ /// Grant access to removeOperand for VPPhiAccessors.
friend class VPPhiAccessors;
/// Grant access to addOperand for VPWidenMemoryRecipe.
friend class VPWidenMemoryRecipe;
+ /// Grant access to removeOperand for VPInstruction::dropMask.
+ friend class VPInstruction;
SmallVector<VPValue *, 2> Operands;
diff --git a/llvm/test/Transforms/LoopVectorize/scalarized_conditional_ops_uncountable_exits.ll b/llvm/test/Transforms/LoopVectorize/scalarized_conditional_ops_uncountable_exits.ll
index c572f15086082..2b7a147790511 100644
--- a/llvm/test/Transforms/LoopVectorize/scalarized_conditional_ops_uncountable_exits.ll
+++ b/llvm/test/Transforms/LoopVectorize/scalarized_conditional_ops_uncountable_exits.ll
@@ -10,82 +10,25 @@ define void @loop_contains_store_condition_load_has_single_user(ptr dereferencea
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE12:.*]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
-; CHECK-NEXT: [[TMP25:%.*]] = add i64 [[INDEX]], 2
-; CHECK-NEXT: [[TMP26:%.*]] = add i64 [[INDEX]], 3
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP1]], align 2
; CHECK-NEXT: [[TMP2:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
-; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP2]], i1 false)
-; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP3]])
+; CHECK-NEXT: [[TMP5:%.*]] = freeze <4 x i1> [[TMP2]]
+; CHECK-NEXT: [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT: br i1 [[TMP6]], label %[[PRED_STORE_CONTINUE12]], label %[[VECTOR_BODY_NONBAILING:.*]]
+; CHECK: [[VECTOR_BODY_NONBAILING]]:
; CHECK-NEXT: [[TMP31:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP32:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP0]]
-; CHECK-NEXT: [[TMP33:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP25]]
-; CHECK-NEXT: [[TMP34:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP26]]
-; CHECK-NEXT: [[TMP35:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 0
-; CHECK-NEXT: br i1 [[TMP35]], label %[[PRED_LOAD_IF:.*]], label %[[PRED_LOAD_CONTINUE:.*]]
-; CHECK: [[PRED_LOAD_IF]]:
-; CHECK-NEXT: [[TMP11:%.*]] = load i16, ptr [[TMP31]], align 2
-; CHECK-NEXT: [[TMP12:%.*]] = insertelement <4 x i16> poison, i16 [[TMP11]], i64 0
-; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE]]
-; CHECK: [[PRED_LOAD_CONTINUE]]:
-; CHECK-NEXT: [[TMP13:%.*]] = phi <4 x i16> [ poison, %[[VECTOR_BODY]] ], [ [[TMP12]], %[[PRED_LOAD_IF]] ]
-; CHECK-NEXT: [[TMP14:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 1
-; CHECK-NEXT: br i1 [[TMP14]], label %[[PRED_LOAD_IF1:.*]], label %[[PRED_LOAD_CONTINUE2:.*]]
-; CHECK: [[PRED_LOAD_IF1]]:
-; CHECK-NEXT: [[TMP15:%.*]] = load i16, ptr [[TMP32]], align 2
-; CHECK-NEXT: [[TMP16:%.*]] = insertelement <4 x i16> [[TMP13]], i16 [[TMP15]], i64 1
-; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE2]]
-; CHECK: [[PRED_LOAD_CONTINUE2]]:
-; CHECK-NEXT: [[TMP17:%.*]] = phi <4 x i16> [ [[TMP13]], %[[PRED_LOAD_CONTINUE]] ], [ [[TMP16]], %[[PRED_LOAD_IF1]] ]
-; CHECK-NEXT: [[TMP18:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 2
-; CHECK-NEXT: br i1 [[TMP18]], label %[[PRED_LOAD_IF3:.*]], label %[[PRED_LOAD_CONTINUE4:.*]]
-; CHECK: [[PRED_LOAD_IF3]]:
-; CHECK-NEXT: [[TMP19:%.*]] = load i16, ptr [[TMP33]], align 2
-; CHECK-NEXT: [[TMP20:%.*]] = insertelement <4 x i16> [[TMP17]], i16 [[TMP19]], i64 2
-; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE4]]
-; CHECK: [[PRED_LOAD_CONTINUE4]]:
-; CHECK-NEXT: [[TMP21:%.*]] = phi <4 x i16> [ [[TMP17]], %[[PRED_LOAD_CONTINUE2]] ], [ [[TMP20]], %[[PRED_LOAD_IF3]] ]
-; CHECK-NEXT: [[TMP22:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 3
-; CHECK-NEXT: br i1 [[TMP22]], label %[[PRED_LOAD_IF5:.*]], label %[[PRED_LOAD_CONTINUE6:.*]]
-; CHECK: [[PRED_LOAD_IF5]]:
-; CHECK-NEXT: [[TMP23:%.*]] = load i16, ptr [[TMP34]], align 2
-; CHECK-NEXT: [[TMP24:%.*]] = insertelement <4 x i16> [[TMP21]], i16 [[TMP23]], i64 3
-; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE6]]
-; CHECK: [[PRED_LOAD_CONTINUE6]]:
-; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = phi <4 x i16> [ [[TMP21]], %[[PRED_LOAD_CONTINUE4]] ], [ [[TMP24]], %[[PRED_LOAD_IF5]] ]
-; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
-; CHECK-NEXT: br i1 [[TMP35]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
-; CHECK: [[PRED_STORE_IF]]:
-; CHECK-NEXT: [[TMP27:%.*]] = extractelement <4 x i16> [[TMP4]], i64 0
-; CHECK-NEXT: store i16 [[TMP27]], ptr [[TMP31]], align 2
-; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE]]
-; CHECK: [[PRED_STORE_CONTINUE]]:
-; CHECK-NEXT: br i1 [[TMP14]], label %[[PRED_STORE_IF7:.*]], label %[[PRED_STORE_CONTINUE8:.*]]
-; CHECK: [[PRED_STORE_IF7]]:
-; CHECK-NEXT: [[TMP28:%.*]] = extractelement <4 x i16> [[TMP4]], i64 1
-; CHECK-NEXT: store i16 [[TMP28]], ptr [[TMP32]], align 2
-; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE8]]
-; CHECK: [[PRED_STORE_CONTINUE8]]:
-; CHECK-NEXT: br i1 [[TMP18]], label %[[PRED_STORE_IF9:.*]], label %[[PRED_STORE_CONTINUE10:.*]]
-; CHECK: [[PRED_STORE_IF9]]:
-; CHECK-NEXT: [[TMP29:%.*]] = extractelement <4 x i16> [[TMP4]], i64 2
-; CHECK-NEXT: store i16 [[TMP29]], ptr [[TMP33]], align 2
-; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE10]]
-; CHECK: [[PRED_STORE_CONTINUE10]]:
-; CHECK-NEXT: br i1 [[TMP22]], label %[[PRED_STORE_IF11:.*]], label %[[PRED_STORE_CONTINUE12]]
-; CHECK: [[PRED_STORE_IF11]]:
-; CHECK-NEXT: [[TMP30:%.*]] = extractelement <4 x i16> [[TMP4]], i64 3
-; CHECK-NEXT: store i16 [[TMP30]], ptr [[TMP34]], align 2
+; CHECK-NEXT: [[TMP24:%.*]] = load <4 x i16>, ptr [[TMP31]], align 2
+; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i16> [[TMP24]], splat (i16 1)
+; CHECK-NEXT: store <4 x i16> [[TMP4]], ptr [[TMP31]], align 2
; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE12]]
; CHECK: [[PRED_STORE_CONTINUE12]]:
-; CHECK-NEXT: [[TMP5:%.*]] = freeze <4 x i1> [[TMP2]]
-; CHECK-NEXT: [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
; CHECK-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP3:%.*]] = select i1 [[TMP6]], i64 0, i64 4
; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], [[TMP3]]
; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[TMP9]], 20
; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
More information about the llvm-commits
mailing list