[llvm] [SLP]Cost-based choice of the i1 reduction form (PR #223163)

via llvm-commits llvm-commits at lists.llvm.org
Sat Sep 12 11:01:26 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-risc-v

@llvm/pr-subscribers-backend-amdgpu

Author: Alexey Bataev (alexey-bataev)

<details>
<summary>Changes</summary>

i1 reductions (and/or of <n x i1>, add of zexted <n x i1>) can be
emitted as the plain target reduction or in the bitcast-based form
(bitcast to the scalar int type plus a compare for and/or, plus ctpop
for add). Choose the cheaper form by cost, estimating the bitcast-based
form from its emitted components; the ctpop form is estimated as the
cheaper of the scalar bitcast+ctpop and the vector ctpop on the mask
type.

Fixes #<!-- -->40657


---

Patch is 100.48 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/223163.diff


15 Files Affected:

- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+95-28) 
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp (+61) 
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h (+11) 
- (modified) llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll (+5-6) 
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll (+117-115) 
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll (+8-16) 
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll (+7-38) 
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll (+3-4) 
- (modified) llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll (+1-3) 
- (modified) llvm/test/Transforms/SLPVectorizer/X86/entries-different-vf.ll (+3-2) 
- (modified) llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll (+1-3) 
- (modified) llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll (+269-128) 
- (modified) llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll (+4-11) 
- (modified) llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll (+2-1) 
- (modified) llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll (+35-18) 


``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 488bfb36b6ccc..cbcce81d94e52 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -2399,6 +2399,10 @@ class slpvectorizer::BoUpSLP {
                          SmallVectorImpl<Value *> &Op2,
                          OrdersType &ReorderIndices) const;
 
+  /// \returns Cast context for the given graph node.
+  TargetTransformInfo::CastContextHint
+  getCastContextHint(const TreeEntry &TE) const;
+
   ~BoUpSLP();
 
 private:
@@ -2461,10 +2465,6 @@ class slpvectorizer::BoUpSLP {
   /// one.
   Instruction *getRootEntryInstruction(const TreeEntry &Entry) const;
 
-  /// \returns Cast context for the given graph node.
-  TargetTransformInfo::CastContextHint
-  getCastContextHint(const TreeEntry &TE) const;
-
   /// \returns the scale of the given tree entry to the loop iteration.
   /// \p Scalar is the scalar value from the entry, if using the parent for the
   /// external use.
@@ -31155,7 +31155,8 @@ class HorizontalReduction {
     if (!VectorValuesAndScales.empty()) {
       Builder.setFastMathFlags(GroupRdxFMF);
       auto [Res, ResNegated] =
-          emitReduction(Builder, *TTI, ReductionRoot->getType());
+          emitReduction(Builder, *TTI, V.getCastContextHint(V.getRootNode()),
+                        ReductionRoot->getType());
       Builder.setFastMathFlags(RdxFMF);
       // The reduction result of the all-negated parts is subtracted in the
       // final combine.
@@ -31612,9 +31613,11 @@ class HorizontalReduction {
              "Expected floating point types for ordered reduction");
       Builder.SetCurrentDebugLocation(
           cast<Instruction>(ReductionRoot)->getDebugLoc());
-      VectorizedTree = createSingleOp(Builder, *TTI, SuccessRoot, /*Scale=*/1,
-                                      /*IsSigned=*/false, DestTy,
-                                      /*ReducedInTree=*/false, VectorizedTree);
+      VectorizedTree =
+          createSingleOp(Builder, *TTI, V.getCastContextHint(V.getRootNode()),
+                         SuccessRoot, /*Scale=*/1,
+                         /*IsSigned=*/false, DestTy,
+                         /*ReducedInTree=*/false, VectorizedTree);
 
       // Fold trailing scalars [SuccessStart+SuccessWidth, N).
       for (Value *RdxVal :
@@ -31658,8 +31661,9 @@ class HorizontalReduction {
   /// Creates the reduction from the given \p Vec vector value with the given
   /// scale \p Scale and signedness \p IsSigned.
   Value *createSingleOp(IRBuilderBase &Builder, const TargetTransformInfo &TTI,
-                        Value *Vec, unsigned Scale, bool IsSigned, Type *DestTy,
-                        bool ReducedInTree, Value *Start = nullptr) {
+                        TTI::CastContextHint Ctx, Value *Vec, unsigned Scale,
+                        bool IsSigned, Type *DestTy, bool ReducedInTree,
+                        Value *Start = nullptr) {
     Value *Rdx;
     if (ReducedInTree) {
       Rdx = Vec;
@@ -31698,7 +31702,7 @@ class HorizontalReduction {
           Rdx = createOp(Builder, RdxKind, Rdx, SubVec, "rdx.op", ReductionOps);
       }
     } else {
-      Rdx = emitReduction(Vec, Builder, &TTI, DestTy, Start);
+      Rdx = emitReduction(Vec, Builder, &TTI, Ctx, DestTy, Start);
     }
     if (Rdx->getType() != DestTy)
       Rdx = Builder.CreateIntCast(Rdx, DestTy, IsSigned);
@@ -31867,8 +31871,17 @@ class HorizontalReduction {
             auto [RType, IsSigned] = R.getRootNodeTypeWithNoCast().value_or(
                 std::make_pair(RedTy, true));
             if (RType == RedTy) {
-              VectorCost = TTI->getArithmeticReductionCost(RdxOpcode, VectorTy,
-                                                           FMF, CostKind);
+              if (VectorTy->getElementType()->isIntegerTy(1) &&
+                  (RdxKind == RecurKind::And || RdxKind == RecurKind::Or ||
+                   (RdxKind == RecurKind::Add && !ScalarTy->isIntegerTy(1))))
+                VectorCost =
+                    getI1ReductionCost(RdxKind, *TTI, VectorTy, ScalarTy,
+                                       R.getCastContextHint(R.getRootNode()),
+                                       CostKind)
+                        .first;
+              else
+                VectorCost = TTI->getArithmeticReductionCost(
+                    RdxOpcode, VectorTy, FMF, CostKind);
             } else {
               VectorCost = TTI->getExtendedReductionCost(
                   RdxOpcode, !IsSigned, RedTy,
@@ -32011,6 +32024,7 @@ class HorizontalReduction {
   /// combine.
   std::pair<Value *, bool> emitReduction(IRBuilderBase &Builder,
                                          const TargetTransformInfo &TTI,
+                                         TTI::CastContextHint Ctx,
                                          Type *DestTy) {
     Value *ReducedSubTree = nullptr;
     bool ResNegated = false;
@@ -32038,8 +32052,8 @@ class HorizontalReduction {
     // the signs of the operands.
     auto CreateSingleOp = [&](Value *Vec, unsigned Scale, bool IsSigned,
                               bool ReducedInTree, bool Negated) {
-      Value *Rdx = createSingleOp(Builder, TTI, Vec, Scale, IsSigned, DestTy,
-                                  ReducedInTree);
+      Value *Rdx = createSingleOp(Builder, TTI, Ctx, Vec, Scale, IsSigned,
+                                  DestTy, ReducedInTree);
       if (!ReducedSubTree) {
         ReducedSubTree = Rdx;
         ResNegated = Negated;
@@ -32220,27 +32234,66 @@ class HorizontalReduction {
     return {ReducedSubTree, ResNegated};
   }
 
+  /// Emits an i1 reduction in the cheaper of the plain and the bitcast-based
+  /// form. Returns nullptr if not an i1 reduction handled here.
+  Value *emitI1Reduction(Value *VectorizedValue, IRBuilderBase &Builder,
+                         const TargetTransformInfo &TTI,
+                         TTI::CastContextHint Ctx, Type *DestTy, Value *Start) {
+    auto *FTy = cast<FixedVectorType>(VectorizedValue->getType());
+    bool IsAdd = RdxKind == RecurKind::Add &&
+                 DestTy->getScalarType() != FTy->getScalarType();
+    bool IsBoolLogic =
+        (RdxKind == RecurKind::And || RdxKind == RecurKind::Or) &&
+        DestTy->getScalarType() == FTy->getScalarType();
+    if (Start || FTy->getScalarType() != Builder.getInt1Ty() ||
+        (!IsAdd && !IsBoolLogic))
+      return nullptr;
+    if (getI1ReductionCost(
+            RdxKind, TTI, FTy, DestTy->getScalarType(), Ctx,
+            getSLPCostKind(Builder.GetInsertBlock()->getParent()))
+            .second) {
+      // Convert vector_reduce_and/or(<n x i1>) to icmp eq/ne(bitcast <n x i1>
+      // to in) and vector_reduce_add(ZExt(<n x i1>)) to
+      // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
+      Value *V = Builder.CreateBitCast(
+          VectorizedValue, Builder.getIntNTy(FTy->getNumElements()));
+      ++NumVectorInstructions;
+      switch (RdxKind) {
+      case RecurKind::And:
+        return Builder.CreateICmpEQ(V,
+                                    ConstantInt::getAllOnesValue(V->getType()));
+      case RecurKind::Or:
+        return Builder.CreateIsNotNull(V);
+      case RecurKind::Add:
+        return Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
+      default:
+        llvm_unreachable("Unexpected reduction kind for the i1 bitcast form");
+      }
+    }
+    if (IsAdd) {
+      // Keep the extended vector_reduce_add form.
+      VectorizedValue = Builder.CreateZExt(
+          VectorizedValue,
+          getWidenedType(DestTy->getScalarType(), FTy->getNumElements()));
+      ++NumVectorInstructions;
+    }
+    ++NumVectorInstructions;
+    return createSimpleReduction(Builder, VectorizedValue, RdxKind);
+  }
+
   /// Emit a horizontal reduction of the vectorized value.
   /// If \p Start is non-null, emit an ordered reduction intrinsic that
   /// sequentially accumulates into \p Start (only valid for FAdd/FMulAdd).
   Value *emitReduction(Value *VectorizedValue, IRBuilderBase &Builder,
-                       const TargetTransformInfo *TTI, Type *DestTy,
-                       Value *Start = nullptr) {
+                       const TargetTransformInfo *TTI, TTI::CastContextHint Ctx,
+                       Type *DestTy, Value *Start = nullptr) {
     assert(VectorizedValue && "Need to have a vectorized tree node");
     assert(RdxKind != RecurKind::FMulAdd &&
            "A call to the llvm.fmuladd intrinsic is not handled yet");
 
-    auto *FTy = cast<FixedVectorType>(VectorizedValue->getType());
-    if (!Start && FTy->getScalarType() == Builder.getInt1Ty() &&
-        RdxKind == RecurKind::Add &&
-        DestTy->getScalarType() != FTy->getScalarType()) {
-      // Convert vector_reduce_add(ZExt(<n x i1>)) to
-      // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
-      Value *V = Builder.CreateBitCast(
-          VectorizedValue, Builder.getIntNTy(FTy->getNumElements()));
-      ++NumVectorInstructions;
-      return Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
-    }
+    if (Value *Rdx =
+            emitI1Reduction(VectorizedValue, Builder, *TTI, Ctx, DestTy, Start))
+      return Rdx;
     ++NumVectorInstructions;
     if (Start)
       return createOrderedReduction(Builder, RdxKind, VectorizedValue, Start);
@@ -32815,6 +32868,20 @@ bool SLPVectorizerPass::tryToVectorize(
             VecTy, APInt::getAllOnes(getNumElements(VecTy)), /*Insert=*/false,
             /*Extract=*/true, CostKind) +
         TTI.getInstructionCost(Inst, CostKind);
+    // The reduction-only cost cannot price the compared-values tree of i1
+    // and/or reductions, so ties are left to the full analysis.
+    if (Ty->isIntegerTy(1) &&
+        (Kind == RecurKind::And || Kind == RecurKind::Or)) {
+      TTI::CastContextHint Ctx =
+          all_of(Ops, [](Value *Op) { return isa<LoadInst>(Op); })
+              ? TTI::CastContextHint::Normal
+              : TTI::CastContextHint::None;
+      if (getI1ReductionCost(Kind, TTI, cast<FixedVectorType>(VecTy), Ty, Ctx,
+                             CostKind)
+              .first > ScalarCost)
+        return false;
+      return HorRdx.tryToReduce(R, *DL, &TTI, *TLI, AC, *DT) != nullptr;
+    }
     InstructionCost RedCost;
     switch (::getRdxKind(Inst)) {
     case RecurKind::Add:
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
index c5679c52c2a49..c18da825e7e2a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
@@ -14,8 +14,11 @@
 #include "llvm/ADT/STLExtras.h"
 #include "llvm/ADT/Sequence.h"
 #include "llvm/ADT/SmallVector.h"
+#include "llvm/Analysis/IVDescriptors.h"
+#include "llvm/IR/Constants.h"
 #include "llvm/IR/DerivedTypes.h"
 #include "llvm/IR/Instructions.h"
+#include "llvm/IR/Intrinsics.h"
 #include "llvm/IR/Operator.h"
 #include "llvm/IR/Type.h"
 #include "llvm/IR/Value.h"
@@ -240,4 +243,62 @@ InstructionCost getExtractWithExtendCost(const TargetTransformInfo &TTI,
   return TTI.getExtractWithExtendCost(Opcode, Dst, VecTy, Index, CostKind);
 }
 
+static InstructionCost
+getBoolLogicRdxBitcastCost(RecurKind Kind, const TargetTransformInfo &TTI,
+                           FixedVectorType *VectorTy, TTI::CastContextHint Ctx,
+                           TTI::TargetCostKind CostKind) {
+  assert((Kind == RecurKind::And || Kind == RecurKind::Or) &&
+         VectorTy->getElementType()->isIntegerTy(1) &&
+         "Expected and/or reduction of i1");
+  auto *IntTy =
+      IntegerType::get(VectorTy->getContext(), getNumElements(VectorTy));
+  CmpInst::Predicate Pred =
+      Kind == RecurKind::And ? CmpInst::ICMP_EQ : CmpInst::ICMP_NE;
+  // The compare is against the all-ones (and) or zero (or) constant.
+  Constant *CmpConst = Kind == RecurKind::And
+                           ? ConstantInt::getAllOnesValue(IntTy)
+                           : ConstantInt::getNullValue(IntTy);
+  return TTI.getCastInstrCost(Instruction::BitCast, IntTy, VectorTy, Ctx,
+                              CostKind) +
+         TTI.getCmpSelInstrCost(
+             Instruction::ICmp, IntTy, CmpInst::makeCmpResultType(IntTy), Pred,
+             CostKind, /*Op1Info=*/{}, TTI::getOperandInfo(CmpConst));
+}
+
+std::pair<InstructionCost, bool>
+getI1ReductionCost(RecurKind Kind, const TargetTransformInfo &TTI,
+                   FixedVectorType *VectorTy, Type *ScalarTy,
+                   TTI::CastContextHint Ctx, TTI::TargetCostKind CostKind) {
+  unsigned RdxOpcode = RecurrenceDescriptor::getOpcode(Kind);
+  if (Kind == RecurKind::And || Kind == RecurKind::Or) {
+    InstructionCost RdxCost = TTI.getArithmeticReductionCost(
+        RdxOpcode, VectorTy, std::nullopt, CostKind);
+    InstructionCost BitcastCost =
+        getBoolLogicRdxBitcastCost(Kind, TTI, VectorTy, Ctx, CostKind);
+    return {std::min(RdxCost, BitcastCost), BitcastCost < RdxCost};
+  }
+  assert(Kind == RecurKind::Add && !ScalarTy->isIntegerTy(1) &&
+         "Expected add reduction of zexted i1 values");
+  // The extended reduction form.
+  InstructionCost ExtRdxCost =
+      TTI.getExtendedReductionCost(RdxOpcode, /*IsUnsigned=*/true, ScalarTy,
+                                   VectorTy, std::nullopt, CostKind);
+  // The ctpop form is estimated as the cheaper of the scalar bitcast+ctpop
+  // and the vector ctpop on the mask type; it is emitted as bitcast+ctpop.
+  // The zext/trunc of the ctpop result to the destination type is absorbed by
+  // the legalization of the ctpop itself.
+  auto *IntTy =
+      IntegerType::get(VectorTy->getContext(), getNumElements(VectorTy));
+  InstructionCost CtpopCost = std::min(
+      TTI.getCastInstrCost(Instruction::BitCast, IntTy, VectorTy, Ctx,
+                           CostKind) +
+          TTI.getIntrinsicInstrCost(
+              IntrinsicCostAttributes(Intrinsic::ctpop, IntTy, {IntTy}),
+              CostKind),
+      TTI.getIntrinsicInstrCost(
+          IntrinsicCostAttributes(Intrinsic::ctpop, VectorTy, {VectorTy}),
+          CostKind));
+  return {std::min(ExtRdxCost, CtpopCost), CtpopCost <= ExtRdxCost};
+}
+
 } // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
index 75ad9fe88db80..c29782bd4de5b 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
@@ -30,6 +30,7 @@ class Type;
 class User;
 class Value;
 class VectorType;
+enum class RecurKind;
 } // namespace llvm
 
 namespace llvm::slpvectorizer {
@@ -99,6 +100,16 @@ getExtractWithExtendCost(const TargetTransformInfo &TTI, bool ReVec,
                          unsigned Index,
                          const TargetTransformInfo::TargetCostKind CostKind);
 
+/// i1 reductions can be emitted as the plain target reduction or in the
+/// bitcast-based form (bitcast to a scalar integer type plus a compare for
+/// and/or, plus ctpop for add). Returns the cost of the cheaper form and
+/// whether it is the bitcast-based one.
+std::pair<InstructionCost, bool>
+getI1ReductionCost(RecurKind Kind, const TargetTransformInfo &TTI,
+                   FixedVectorType *VectorTy, Type *ScalarTy,
+                   TargetTransformInfo::CastContextHint Ctx,
+                   TargetTransformInfo::TargetCostKind CostKind);
+
 } // namespace llvm::slpvectorizer
 
 #endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll b/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
index 78f68c420f6ab..fe21d4039f240 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
@@ -277,13 +277,12 @@ define i1 @cmp_lt_gt(double %a, double %b, double %c) {
 ; CHECK-NEXT:    [[TMP6:%.*]] = shufflevector <2 x double> [[TMP5]], <2 x double> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP7:%.*]] = fdiv <2 x double> [[TMP3]], [[TMP6]]
 ; CHECK-NEXT:    [[TMP8:%.*]] = fcmp uge <2 x double> [[TMP7]], splat (double f0x3EB0C6F7A0B5ED8D)
-; CHECK-NEXT:    [[SHIFT:%.*]] = shufflevector <2 x i1> [[TMP8]], <2 x i1> poison, <2 x i32> <i32 1, i32 poison>
-; CHECK-NEXT:    [[FOLDEXTEXTBINOP:%.*]] = or <2 x i1> [[TMP8]], [[SHIFT]]
+; CHECK-NEXT:    [[TMP11:%.*]] = bitcast <2 x i1> [[TMP8]] to i2
+; CHECK-NEXT:    [[TMP12:%.*]] = icmp ne i2 [[TMP11]], 0
 ; CHECK-NEXT:    [[TMP9:%.*]] = fcmp ule <2 x double> [[TMP7]], splat (double 1.000000e+00)
-; CHECK-NEXT:    [[SHIFT3:%.*]] = shufflevector <2 x i1> [[TMP9]], <2 x i1> poison, <2 x i32> <i32 1, i32 poison>
-; CHECK-NEXT:    [[TMP10:%.*]] = or <2 x i1> [[TMP9]], [[SHIFT3]]
-; CHECK-NEXT:    [[FOLDEXTEXTBINOP4:%.*]] = and <2 x i1> [[FOLDEXTEXTBINOP]], [[TMP10]]
-; CHECK-NEXT:    [[RETVAL_0:%.*]] = extractelement <2 x i1> [[FOLDEXTEXTBINOP4]], i64 0
+; CHECK-NEXT:    [[TMP13:%.*]] = bitcast <2 x i1> [[TMP9]] to i2
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp ne i2 [[TMP13]], 0
+; CHECK-NEXT:    [[RETVAL_0:%.*]] = and i1 [[TMP12]], [[TMP10]]
 ; CHECK-NEXT:    ret i1 [[RETVAL_0]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
index c98a4d1f2c182..35d804d927ff0 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
@@ -4,141 +4,143 @@
 define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i64 %.sroa.3308.0.copyload, i64 %.neg1, i64 %indvar3788, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8) {
 ; CHECK-LABEL: define void @test(
 ; CHECK-SAME: i64 [[TMP0:%.*]], i64 [[TMP1:%.*]], i64 [[TMP2:%.*]], i64 [[TMP3:%.*]], i64 [[DOTSROA_3341_0_COPYLOAD:%.*]], i64 [[DOTSROA_3308_0_COPYLOAD:%.*]], i64 [[DOTNEG1:%.*]], i64 [[INDVAR3788:%.*]], i64 [[TMP4:%.*]], i64 [[TMP5:%.*]], i64 [[TMP6:%.*]], i64 [[TMP7:%.*]], i64 [[TMP8:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT:  [[_LR_PH_PREHEADER:.*:]]
+; CHECK-NEXT:  [[_LR_PH_PREHEADER:.*]]:
 ; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[TMP1]], i64 0
 ; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[TMP0]], i64 1
-; CHECK-NEXT:    [[TMP13:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 24>
+; CHECK-NEXT:    [[TMP12:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 24>
 ; CHECK-NEXT:    [[TMP17:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 40>
 ; CHECK-NEXT:    [[TMP11:%.*]] = mul i64 [[TMP0]], 48
 ; CHECK-NEXT:    [[TMP14:%.*]] = insertelement <2 x i64> <i64 poison, i64 1>, i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP15:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> <i64 -1, i64 poison>, <2 x i32> <i32 2, i32 0>
-; CHECK-NEXT:    [[TMP16:%.*]] = sub <2 x i64> [[TMP14]], [[TMP15]]
-; CHECK-NEXT:    [[TMP20:%.*]] = shl i64 [[TMP0]], 11
+; CHECK-NEXT:    [[TMP22:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> <i64 -1, i64 poison>, <2 x i32> <i32 2, i32 0>
+; CHECK-NEXT:    [[TMP16:%.*]] = sub <2 x i64> [[TMP14]], [[TMP22]]
+; CHECK-NEXT:    [[TMP25:%.*]] = shl i64 [[TMP0]], 11
 ; CHECK-NEXT:    [[TMP35:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <2 x i64> [[TMP35]], <2 x i64> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP25:%.*]] = shl <2 x i64> [[TMP12]], <i64 11, i64 0>
-; CHECK-NEXT:    [[TMP21:%.*]] = sub i64 1, [[TMP20]]
-; CHECK-NEXT:    [[TMP22:%.*]] = add <2 x i64> [[TMP25]], <i64 8, i64 1>
-; CHECK-NEXT:    [[TMP23:%.*]] = or <2 x i64> [[TMP25]], <i64 8, i64 1>
-; CHECK-NEXT:    [[TMP33:%.*]] = shufflevector <2 x i64> [[TMP22]], <2 x i64> [[TMP23]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP33:%.*]] = shufflevector <2 x i64> [[TMP35]], <2 x i64> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP20:%.*]] = shl <2 x i64> [[TMP33]], <i64 11, i64 0>
+; CHECK-NEXT:    [[TMP21:%.*]] = sub i64 1, [[TMP25]]
+; CHECK-NEXT:    [[TMP41:%.*]] = add <2 x i64> [[TMP20]], <i64 8, i64 1>
+; CHECK-NEXT:    [[TMP23:%.*]] = or <2 x i64> [[TMP20]], <i64 8, i64 1>
+; CHECK-NEXT:    [[TMP44:%.*]] = shufflevector <2 x i64> [[TMP41]], <...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/223163


More information about the llvm-commits mailing list