[llvm] [LoopVectorize] Fix double-application of FindIV reduction expression in epilogue (PR #219362)

via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 21:42:15 PDT 2026


https://github.com/im-lunex created https://github.com/llvm/llvm-project/pull/219362

added a flag that tracks when the math gets moved... when it does we flip the flag on... the safety check reads the flag and if its on the extra loop is disabled.. Before that we were guessing from instruction types which broke when a later pass changed the shape of things. Now we just ask the flag.

added test..

reason - i choose this approach because of this one looks the cleanest.. among other approaches.. while looking into it i also got more choices ..

- stopping the math simplification (Disable the Fold)
-> this will slowdown the optimization and not a good idea..

- change the Safety guards Check (Check Semantics)
-> it would be too complex to implement and the pattern will be also a headache to implement

- Reorder the Compiler Steps
-> too risky to do might break or effect the performance.. 

fixes: #219211

>From 3db3b2e179b85fdaba40e6682848f2a717f0e80c Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Tue, 25 Aug 2026 23:50:10 +0600
Subject: [PATCH 1/4] cgp: fix sdiv -1 unobserved-lane miscompile

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp           | 14 ++++
 .../X86/store-extract-division-ub.ll          | 77 +++++++++++++++++++
 2 files changed, 91 insertions(+)
 create mode 100644 llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 53b7a81537975..09a7935b7aafd 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8349,6 +8349,18 @@ class VectorPromoteHelper {
     llvm_unreachable(nullptr);
   }
 
+  static bool
+  canCauseUndefinedBehaviorOnUnobservedLanes(const Instruction *Use) {
+    switch (Use->getOpcode()) {
+    default:
+      return false;
+    case Instruction::SDiv:
+    case Instruction::SRem:
+      return isa<ConstantInt>(Use->getOperand(1)) &&
+             cast<ConstantInt>(Use->getOperand(1))->isMinusOne();
+    }
+  }
+
 public:
   VectorPromoteHelper(const DataLayout &DL, const TargetLowering &TLI,
                       const TargetTransformInfo &TTI, Instruction *Transition,
@@ -8367,6 +8379,8 @@ class VectorPromoteHelper {
   /// Check if it is profitable to promote \p ToBePromoted
   /// by moving downward the transition through.
   bool shouldPromote(const Instruction *ToBePromoted) const {
+    if (canCauseUndefinedBehaviorOnUnobservedLanes(ToBePromoted))
+      return false;
     // Promote only if all the operands can be statically expanded.
     // Indeed, we do not want to introduce any new kind of transitions.
     for (const Use &U : ToBePromoted->operands()) {
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll b/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll
new file mode 100644
index 0000000000000..35978e59a6913
--- /dev/null
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll
@@ -0,0 +1,77 @@
+; test for issue (https://github.com/llvm/llvm-project/issues/218570)
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes='require<profile-summary>,function(codegenprepare)' -stress-cgp-store-extract < %s | FileCheck %s
+
+target triple = "x86_64-unknown-linux-gnu"
+
+define void @no_promote_sdiv_minus_one(ptr %src, ptr %dst) {
+; CHECK-LABEL: define void @no_promote_sdiv_minus_one(
+; CHECK-SAME: ptr [[SRC:%.*]], ptr [[DST:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[V:%.*]] = load <2 x i8>, ptr [[SRC]], align 1
+; CHECK-NEXT:    [[E:%.*]] = extractelement <2 x i8> [[V]], i32 0
+; CHECK-NEXT:    [[R:%.*]] = sdiv i8 [[E]], -1
+; CHECK-NEXT:    store i8 [[R]], ptr [[DST]], align 1
+; CHECK-NEXT:    ret void
+;
+entry:
+  %v = load <2 x i8>, ptr %src, align 1
+  %e = extractelement <2 x i8> %v, i32 0
+  %r = sdiv i8 %e, -1
+  store i8 %r, ptr %dst, align 1
+  ret void
+}
+
+define void @no_promote_srem_minus_one(ptr %src, ptr %dst) {
+; CHECK-LABEL: define void @no_promote_srem_minus_one(
+; CHECK-SAME: ptr [[SRC:%.*]], ptr [[DST:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[V:%.*]] = load <2 x i8>, ptr [[SRC]], align 1
+; CHECK-NEXT:    [[E:%.*]] = extractelement <2 x i8> [[V]], i32 0
+; CHECK-NEXT:    [[R:%.*]] = srem i8 [[E]], -1
+; CHECK-NEXT:    store i8 [[R]], ptr [[DST]], align 1
+; CHECK-NEXT:    ret void
+;
+entry:
+  %v = load <2 x i8>, ptr %src, align 1
+  %e = extractelement <2 x i8> %v, i32 0
+  %r = srem i8 %e, -1
+  store i8 %r, ptr %dst, align 1
+  ret void
+}
+
+define void @promote_sdiv_two(ptr %src, ptr %dst) {
+; CHECK-LABEL: define void @promote_sdiv_two(
+; CHECK-SAME: ptr [[SRC:%.*]], ptr [[DST:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[V:%.*]] = load <2 x i8>, ptr [[SRC]], align 1
+; CHECK-NEXT:    [[R:%.*]] = sdiv <2 x i8> [[V]], splat (i8 2)
+; CHECK-NEXT:    [[E:%.*]] = extractelement <2 x i8> [[R]], i32 0
+; CHECK-NEXT:    store i8 [[E]], ptr [[DST]], align 1
+; CHECK-NEXT:    ret void
+;
+entry:
+  %v = load <2 x i8>, ptr %src, align 1
+  %e = extractelement <2 x i8> %v, i32 0
+  %r = sdiv i8 %e, 2
+  store i8 %r, ptr %dst, align 1
+  ret void
+}
+
+define void @promote_udiv_seven(ptr %src, ptr %dst) {
+; CHECK-LABEL: define void @promote_udiv_seven(
+; CHECK-SAME: ptr [[SRC:%.*]], ptr [[DST:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[V:%.*]] = load <2 x i8>, ptr [[SRC]], align 1
+; CHECK-NEXT:    [[R:%.*]] = udiv <2 x i8> [[V]], splat (i8 7)
+; CHECK-NEXT:    [[E:%.*]] = extractelement <2 x i8> [[R]], i32 0
+; CHECK-NEXT:    store i8 [[E]], ptr [[DST]], align 1
+; CHECK-NEXT:    ret void
+;
+entry:
+  %v = load <2 x i8>, ptr %src, align 1
+  %e = extractelement <2 x i8> %v, i32 0
+  %r = udiv i8 %e, 7
+  store i8 %r, ptr %dst, align 1
+  ret void
+}

>From 2526b4bee83184a578f1c2ee6c6f0285916a53b5 Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Wed, 26 Aug 2026 09:52:05 +0600
Subject: [PATCH 2/4] try new approch with llvms [speculative-execution] safety
 check

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp           | 14 +------------
 .../X86/store-extract-division-ub.ll          | 20 ++++++++++++++++++-
 2 files changed, 20 insertions(+), 14 deletions(-)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 09a7935b7aafd..fdf5fdf20e66d 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8349,18 +8349,6 @@ class VectorPromoteHelper {
     llvm_unreachable(nullptr);
   }
 
-  static bool
-  canCauseUndefinedBehaviorOnUnobservedLanes(const Instruction *Use) {
-    switch (Use->getOpcode()) {
-    default:
-      return false;
-    case Instruction::SDiv:
-    case Instruction::SRem:
-      return isa<ConstantInt>(Use->getOperand(1)) &&
-             cast<ConstantInt>(Use->getOperand(1))->isMinusOne();
-    }
-  }
-
 public:
   VectorPromoteHelper(const DataLayout &DL, const TargetLowering &TLI,
                       const TargetTransformInfo &TTI, Instruction *Transition,
@@ -8379,7 +8367,7 @@ class VectorPromoteHelper {
   /// Check if it is profitable to promote \p ToBePromoted
   /// by moving downward the transition through.
   bool shouldPromote(const Instruction *ToBePromoted) const {
-    if (canCauseUndefinedBehaviorOnUnobservedLanes(ToBePromoted))
+    if (!isSafeToSpeculativelyExecute(ToBePromoted))
       return false;
     // Promote only if all the operands can be statically expanded.
     // Indeed, we do not want to introduce any new kind of transitions.
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll b/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll
index 35978e59a6913..07e65da5b46db 100644
--- a/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll
@@ -1,5 +1,5 @@
-; test for issue (https://github.com/llvm/llvm-project/issues/218570)
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; test for issue (https://github.com/llvm/llvm-project/issues/218570)
 ; RUN: opt -S -passes='require<profile-summary>,function(codegenprepare)' -stress-cgp-store-extract < %s | FileCheck %s
 
 target triple = "x86_64-unknown-linux-gnu"
@@ -40,6 +40,24 @@ entry:
   ret void
 }
 
+define void @no_promote_sdiv_poison(ptr %src, ptr %dst) {
+; CHECK-LABEL: define void @no_promote_sdiv_poison(
+; CHECK-SAME: ptr [[SRC:%.*]], ptr [[DST:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[V:%.*]] = load <2 x i8>, ptr [[SRC]], align 1
+; CHECK-NEXT:    [[E:%.*]] = extractelement <2 x i8> [[V]], i32 0
+; CHECK-NEXT:    [[R:%.*]] = sdiv i8 [[E]], poison
+; CHECK-NEXT:    store i8 [[R]], ptr [[DST]], align 1
+; CHECK-NEXT:    ret void
+;
+entry:
+  %v = load <2 x i8>, ptr %src, align 1
+  %e = extractelement <2 x i8> %v, i32 0
+  %r = sdiv i8 %e, poison
+  store i8 %r, ptr %dst, align 1
+  ret void
+}
+
 define void @promote_sdiv_two(ptr %src, ptr %dst) {
 ; CHECK-LABEL: define void @promote_sdiv_two(
 ; CHECK-SAME: ptr [[SRC:%.*]], ptr [[DST:%.*]]) {

>From f89ac6913f838eec5e88341e50e88ad45c778c8f Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Thu, 27 Aug 2026 11:05:57 +0600
Subject: [PATCH 3/4] safety and test enhancement

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp           |  5 ++++-
 .../X86/store-extract-division-ub.ll          | 19 +++++++++++++++++++
 2 files changed, 23 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index fdf5fdf20e66d..2ca29d7b2f243 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8367,7 +8367,10 @@ class VectorPromoteHelper {
   /// Check if it is profitable to promote \p ToBePromoted
   /// by moving downward the transition through.
   bool shouldPromote(const Instruction *ToBePromoted) const {
-    if (!isSafeToSpeculativelyExecute(ToBePromoted))
+    if (!isSafeToSpeculativelyExecute(
+            ToBePromoted, /*CtxI=*/nullptr, /*AC=*/nullptr, /*DT=*/nullptr,
+            /*TLI=*/nullptr, /*UseVariableInfo=*/false,
+            /*IgnoreUBImplyingAttrs=*/false))
       return false;
     // Promote only if all the operands can be statically expanded.
     // Indeed, we do not want to introduce any new kind of transitions.
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll b/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll
index 07e65da5b46db..872a527d5de6d 100644
--- a/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/store-extract-division-ub.ll
@@ -93,3 +93,22 @@ entry:
   store i8 %r, ptr %dst, align 1
   ret void
 }
+
+define void @no_promote_sdiv_intmin(ptr %src, ptr %dst) {
+; CHECK-LABEL: define void @no_promote_sdiv_intmin(
+; CHECK-SAME: ptr [[SRC:%.*]], ptr [[DST:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[V:%.*]] = load <2 x i8>, ptr [[SRC]], align 1
+; CHECK-NEXT:    [[E:%.*]] = extractelement <2 x i8> [[V]], i32 0
+; CHECK-NEXT:    [[R:%.*]] = sdiv i8 -128, [[E]]
+; CHECK-NEXT:    store i8 [[R]], ptr [[DST]], align 1
+; CHECK-NEXT:    ret void
+;
+entry:
+  %v = load <2 x i8>, ptr %src, align 1
+  %e = extractelement <2 x i8> %v, i32 0
+  %r = sdiv i8 -128, %e
+  store i8 %r, ptr %dst, align 1
+  ret void
+}
+

>From 62e6d09311a37f766125301f1553b34eeec22854 Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Fri, 28 Aug 2026 10:25:04 +0600
Subject: [PATCH 4/4] [LoopVectorize] Fix double-application in epilogue

---
 .../Vectorize/LoopVectorizationPlanner.cpp    | 14 +----
 llvm/lib/Transforms/Vectorize/VPlan.h         | 10 ++-
 .../Transforms/Vectorize/VPlanTransforms.cpp  |  7 ++-
 .../X86/find-iv-sunk-expr-epilogue.ll         | 62 +++++++++++++++++++
 4 files changed, 79 insertions(+), 14 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index c464c7894ae5c..82634ffb47b9d 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -879,18 +879,8 @@ static bool hasUnsupportedHeaderPhiRecipe(VPlan &Plan) {
               RecurrenceDescriptor::isFindLastRecurrenceKind(Kind) ||
               !RedPhi->getUnderlyingValue())
             return true;
-          // TODO: Add support for FindIV reductions with sunk expressions: the
-          // resume value from the main loop is in expression domain (e.g.,
-          // mul(ReducedIV, 3)), but the epilogue tracks raw IV values. A sunk
-          // expression is identified by a non-VPInstruction user of
-          // ComputeReductionResult.
-          if (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind)) {
-            auto *RdxResult = vputils::findComputeReductionResult(RedPhi);
-            assert(RdxResult &&
-                   "FindIV reduction must have ComputeReductionResult");
-            return any_of(RdxResult->users(),
-                          std::not_fn(IsaPred<VPInstruction>));
-          }
+          if (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind))
+            return RedPhi->isExpressionSunk();
           return false;
         }
         default:
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 7d2c2fa1bdd23..aaa456ee2e435 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2879,6 +2879,8 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
   /// compare has multiple uses.
   bool HasUsesOutsideReductionChain;
 
+  bool ExpressionSunk = false;
+
 public:
   /// Create a new VPReductionPHIRecipe for the reduction \p Phi.
   VPReductionPHIRecipe(PHINode *Phi, RecurKind Kind, VPValue &Start,
@@ -2895,9 +2897,11 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
 
   VPReductionPHIRecipe *cloneWithOperands(VPValue *Start,
                                           VPValue *BackedgeValue) {
-    return new VPReductionPHIRecipe(
+    auto *Clone = new VPReductionPHIRecipe(
         dyn_cast_or_null<PHINode>(getUnderlyingValue()), getRecurrenceKind(),
         *Start, *BackedgeValue, Style, *this, HasUsesOutsideReductionChain);
+    Clone->ExpressionSunk = ExpressionSunk;
+    return Clone;
   }
 
   VPReductionPHIRecipe *clone() override {
@@ -2943,6 +2947,10 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
     return HasUsesOutsideReductionChain;
   }
 
+  void setExpressionSunk(bool V = true) { ExpressionSunk = V; }
+
+  bool isExpressionSunk() const { return ExpressionSunk; }
+
   /// Returns true if the recipe only uses the first lane of operand \p Op.
   bool usesFirstLaneOnly(const VPValue *Op) const override {
     assert(is_contained(operands(), Op) &&
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index dc1a2697110b0..aae01188900ad 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4609,10 +4609,13 @@ void VPlanTransforms::optimizeFindIVReductions(VPlan &Plan,
 
     // If IVOfExpressionToSink is an expression to sink, sink it now.
     VPValue *VectorRegionExitingVal = ReducedIV;
-    if (IVOfExpressionToSink)
+    bool SunkExpression = false;
+    if (IVOfExpressionToSink) {
       VectorRegionExitingVal =
           cloneBinOpForScalarIV(cast<VPWidenRecipe>(FindLastExpression),
                                 ReducedIV, IVOfExpressionToSink);
+      SunkExpression = true;
+    }
 
     VPValue *NewRdxResult;
     VPValue *StartVPV = PhiR->getStartValue();
@@ -4653,6 +4656,8 @@ void VPlanTransforms::optimizeFindIVReductions(VPlan &Plan,
         cast<PHINode>(PhiR->getUnderlyingInstr()), RecurKind::FindIV, *StartVPV,
         *NewFindLastSelect, RdxUnordered{1}, {},
         PhiR->hasUsesOutsideReductionChain());
+    if (SunkExpression)
+      NewPhiR->setExpressionSunk();
     NewPhiR->insertBefore(PhiR);
     PhiR->replaceAllUsesWith(NewPhiR);
     PhiR->eraseFromParent();
diff --git a/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll b/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
new file mode 100644
index 0000000000000..e57aa5343d70e
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
@@ -0,0 +1,62 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=4 -enable-epilogue-vectorization -epilogue-vectorization-force-VF=4 -S %s | FileCheck %s
+
+; Regression test for https://github.com/llvm/llvm-project/issues/219211
+;
+; FindIV reductions with a sunk expression using a power-of-two multiplier
+; (mul %iv, 4) must NOT produce a double application of the expression.
+; The main loop tracks raw IV, the middle block applies shl 2 once, and
+; epilogue vectorization must be disabled.
+
+; CHECK-LABEL: define i32 @findiv_mul_pow2_sunk(
+; The main vector loop reduces raw IV (not mul):
+; CHECK:       vector.body:
+; CHECK:       middle.block:
+; One shl 2 in the middle block:
+; CHECK:       shl i32 {{.*}}, 2
+; No epilogue:
+; CHECK-NOT:   vec.epilog
+; CHECK:       ret i32
+define i32 @findiv_mul_pow2_sunk(ptr %a, i32 %n) #0 {
+entry:
+  br label %loop
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+  %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+  %l = load i32, ptr %gep, align 4
+  %c = icmp eq i32 %l, 42
+  %expr = mul i32 %iv, 4
+  %sel = select i1 %c, i32 %expr, i32 %rdx
+  %iv.next = add nuw nsw i32 %iv, 1
+  %ec = icmp eq i32 %iv.next, %n
+  br i1 %ec, label %done, label %loop
+done:
+  ret i32 %sel
+}
+
+; CHECK-LABEL: define i32 @findiv_no_sunk_raw_iv(
+; Plain FindIV without expression sinking (no mul).
+; CHECK:       vector.body:
+; CHECK:       middle.block:
+; No shl — the IV is already the result:
+; CHECK-NOT:   shl i32
+; CHECK:       ret i32
+define i32 @findiv_no_sunk_raw_iv(ptr %a, i32 %n) #0 {
+entry:
+  br label %loop
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+  %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+  %l = load i32, ptr %gep, align 4
+  %c = icmp eq i32 %l, 42
+  %sel = select i1 %c, i32 %iv, i32 %rdx
+  %iv.next = add nuw nsw i32 %iv, 1
+  %ec = icmp eq i32 %iv.next, %n
+  br i1 %ec, label %done, label %loop
+done:
+  ret i32 %sel
+}
+
+attributes #0 = { "target-features"="+avx512f" }



More information about the llvm-commits mailing list