[llvm] [SLP]Cost-based choice of the i1 reduction form (PR #223163)
via llvm-commits
llvm-commits at lists.llvm.org
Sat Sep 12 11:01:26 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-risc-v
@llvm/pr-subscribers-backend-amdgpu
Author: Alexey Bataev (alexey-bataev)
<details>
<summary>Changes</summary>
i1 reductions (and/or of <n x i1>, add of zexted <n x i1>) can be
emitted as the plain target reduction or in the bitcast-based form
(bitcast to the scalar int type plus a compare for and/or, plus ctpop
for add). Choose the cheaper form by cost, estimating the bitcast-based
form from its emitted components; the ctpop form is estimated as the
cheaper of the scalar bitcast+ctpop and the vector ctpop on the mask
type.
Fixes #<!-- -->40657
---
Patch is 100.48 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/223163.diff
15 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+95-28)
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp (+61)
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h (+11)
- (modified) llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll (+5-6)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll (+117-115)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll (+8-16)
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll (+7-38)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll (+3-4)
- (modified) llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll (+1-3)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/entries-different-vf.ll (+3-2)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll (+1-3)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll (+269-128)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll (+4-11)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll (+2-1)
- (modified) llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll (+35-18)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 488bfb36b6ccc..cbcce81d94e52 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -2399,6 +2399,10 @@ class slpvectorizer::BoUpSLP {
SmallVectorImpl<Value *> &Op2,
OrdersType &ReorderIndices) const;
+ /// \returns Cast context for the given graph node.
+ TargetTransformInfo::CastContextHint
+ getCastContextHint(const TreeEntry &TE) const;
+
~BoUpSLP();
private:
@@ -2461,10 +2465,6 @@ class slpvectorizer::BoUpSLP {
/// one.
Instruction *getRootEntryInstruction(const TreeEntry &Entry) const;
- /// \returns Cast context for the given graph node.
- TargetTransformInfo::CastContextHint
- getCastContextHint(const TreeEntry &TE) const;
-
/// \returns the scale of the given tree entry to the loop iteration.
/// \p Scalar is the scalar value from the entry, if using the parent for the
/// external use.
@@ -31155,7 +31155,8 @@ class HorizontalReduction {
if (!VectorValuesAndScales.empty()) {
Builder.setFastMathFlags(GroupRdxFMF);
auto [Res, ResNegated] =
- emitReduction(Builder, *TTI, ReductionRoot->getType());
+ emitReduction(Builder, *TTI, V.getCastContextHint(V.getRootNode()),
+ ReductionRoot->getType());
Builder.setFastMathFlags(RdxFMF);
// The reduction result of the all-negated parts is subtracted in the
// final combine.
@@ -31612,9 +31613,11 @@ class HorizontalReduction {
"Expected floating point types for ordered reduction");
Builder.SetCurrentDebugLocation(
cast<Instruction>(ReductionRoot)->getDebugLoc());
- VectorizedTree = createSingleOp(Builder, *TTI, SuccessRoot, /*Scale=*/1,
- /*IsSigned=*/false, DestTy,
- /*ReducedInTree=*/false, VectorizedTree);
+ VectorizedTree =
+ createSingleOp(Builder, *TTI, V.getCastContextHint(V.getRootNode()),
+ SuccessRoot, /*Scale=*/1,
+ /*IsSigned=*/false, DestTy,
+ /*ReducedInTree=*/false, VectorizedTree);
// Fold trailing scalars [SuccessStart+SuccessWidth, N).
for (Value *RdxVal :
@@ -31658,8 +31661,9 @@ class HorizontalReduction {
/// Creates the reduction from the given \p Vec vector value with the given
/// scale \p Scale and signedness \p IsSigned.
Value *createSingleOp(IRBuilderBase &Builder, const TargetTransformInfo &TTI,
- Value *Vec, unsigned Scale, bool IsSigned, Type *DestTy,
- bool ReducedInTree, Value *Start = nullptr) {
+ TTI::CastContextHint Ctx, Value *Vec, unsigned Scale,
+ bool IsSigned, Type *DestTy, bool ReducedInTree,
+ Value *Start = nullptr) {
Value *Rdx;
if (ReducedInTree) {
Rdx = Vec;
@@ -31698,7 +31702,7 @@ class HorizontalReduction {
Rdx = createOp(Builder, RdxKind, Rdx, SubVec, "rdx.op", ReductionOps);
}
} else {
- Rdx = emitReduction(Vec, Builder, &TTI, DestTy, Start);
+ Rdx = emitReduction(Vec, Builder, &TTI, Ctx, DestTy, Start);
}
if (Rdx->getType() != DestTy)
Rdx = Builder.CreateIntCast(Rdx, DestTy, IsSigned);
@@ -31867,8 +31871,17 @@ class HorizontalReduction {
auto [RType, IsSigned] = R.getRootNodeTypeWithNoCast().value_or(
std::make_pair(RedTy, true));
if (RType == RedTy) {
- VectorCost = TTI->getArithmeticReductionCost(RdxOpcode, VectorTy,
- FMF, CostKind);
+ if (VectorTy->getElementType()->isIntegerTy(1) &&
+ (RdxKind == RecurKind::And || RdxKind == RecurKind::Or ||
+ (RdxKind == RecurKind::Add && !ScalarTy->isIntegerTy(1))))
+ VectorCost =
+ getI1ReductionCost(RdxKind, *TTI, VectorTy, ScalarTy,
+ R.getCastContextHint(R.getRootNode()),
+ CostKind)
+ .first;
+ else
+ VectorCost = TTI->getArithmeticReductionCost(
+ RdxOpcode, VectorTy, FMF, CostKind);
} else {
VectorCost = TTI->getExtendedReductionCost(
RdxOpcode, !IsSigned, RedTy,
@@ -32011,6 +32024,7 @@ class HorizontalReduction {
/// combine.
std::pair<Value *, bool> emitReduction(IRBuilderBase &Builder,
const TargetTransformInfo &TTI,
+ TTI::CastContextHint Ctx,
Type *DestTy) {
Value *ReducedSubTree = nullptr;
bool ResNegated = false;
@@ -32038,8 +32052,8 @@ class HorizontalReduction {
// the signs of the operands.
auto CreateSingleOp = [&](Value *Vec, unsigned Scale, bool IsSigned,
bool ReducedInTree, bool Negated) {
- Value *Rdx = createSingleOp(Builder, TTI, Vec, Scale, IsSigned, DestTy,
- ReducedInTree);
+ Value *Rdx = createSingleOp(Builder, TTI, Ctx, Vec, Scale, IsSigned,
+ DestTy, ReducedInTree);
if (!ReducedSubTree) {
ReducedSubTree = Rdx;
ResNegated = Negated;
@@ -32220,27 +32234,66 @@ class HorizontalReduction {
return {ReducedSubTree, ResNegated};
}
+ /// Emits an i1 reduction in the cheaper of the plain and the bitcast-based
+ /// form. Returns nullptr if not an i1 reduction handled here.
+ Value *emitI1Reduction(Value *VectorizedValue, IRBuilderBase &Builder,
+ const TargetTransformInfo &TTI,
+ TTI::CastContextHint Ctx, Type *DestTy, Value *Start) {
+ auto *FTy = cast<FixedVectorType>(VectorizedValue->getType());
+ bool IsAdd = RdxKind == RecurKind::Add &&
+ DestTy->getScalarType() != FTy->getScalarType();
+ bool IsBoolLogic =
+ (RdxKind == RecurKind::And || RdxKind == RecurKind::Or) &&
+ DestTy->getScalarType() == FTy->getScalarType();
+ if (Start || FTy->getScalarType() != Builder.getInt1Ty() ||
+ (!IsAdd && !IsBoolLogic))
+ return nullptr;
+ if (getI1ReductionCost(
+ RdxKind, TTI, FTy, DestTy->getScalarType(), Ctx,
+ getSLPCostKind(Builder.GetInsertBlock()->getParent()))
+ .second) {
+ // Convert vector_reduce_and/or(<n x i1>) to icmp eq/ne(bitcast <n x i1>
+ // to in) and vector_reduce_add(ZExt(<n x i1>)) to
+ // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
+ Value *V = Builder.CreateBitCast(
+ VectorizedValue, Builder.getIntNTy(FTy->getNumElements()));
+ ++NumVectorInstructions;
+ switch (RdxKind) {
+ case RecurKind::And:
+ return Builder.CreateICmpEQ(V,
+ ConstantInt::getAllOnesValue(V->getType()));
+ case RecurKind::Or:
+ return Builder.CreateIsNotNull(V);
+ case RecurKind::Add:
+ return Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
+ default:
+ llvm_unreachable("Unexpected reduction kind for the i1 bitcast form");
+ }
+ }
+ if (IsAdd) {
+ // Keep the extended vector_reduce_add form.
+ VectorizedValue = Builder.CreateZExt(
+ VectorizedValue,
+ getWidenedType(DestTy->getScalarType(), FTy->getNumElements()));
+ ++NumVectorInstructions;
+ }
+ ++NumVectorInstructions;
+ return createSimpleReduction(Builder, VectorizedValue, RdxKind);
+ }
+
/// Emit a horizontal reduction of the vectorized value.
/// If \p Start is non-null, emit an ordered reduction intrinsic that
/// sequentially accumulates into \p Start (only valid for FAdd/FMulAdd).
Value *emitReduction(Value *VectorizedValue, IRBuilderBase &Builder,
- const TargetTransformInfo *TTI, Type *DestTy,
- Value *Start = nullptr) {
+ const TargetTransformInfo *TTI, TTI::CastContextHint Ctx,
+ Type *DestTy, Value *Start = nullptr) {
assert(VectorizedValue && "Need to have a vectorized tree node");
assert(RdxKind != RecurKind::FMulAdd &&
"A call to the llvm.fmuladd intrinsic is not handled yet");
- auto *FTy = cast<FixedVectorType>(VectorizedValue->getType());
- if (!Start && FTy->getScalarType() == Builder.getInt1Ty() &&
- RdxKind == RecurKind::Add &&
- DestTy->getScalarType() != FTy->getScalarType()) {
- // Convert vector_reduce_add(ZExt(<n x i1>)) to
- // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
- Value *V = Builder.CreateBitCast(
- VectorizedValue, Builder.getIntNTy(FTy->getNumElements()));
- ++NumVectorInstructions;
- return Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
- }
+ if (Value *Rdx =
+ emitI1Reduction(VectorizedValue, Builder, *TTI, Ctx, DestTy, Start))
+ return Rdx;
++NumVectorInstructions;
if (Start)
return createOrderedReduction(Builder, RdxKind, VectorizedValue, Start);
@@ -32815,6 +32868,20 @@ bool SLPVectorizerPass::tryToVectorize(
VecTy, APInt::getAllOnes(getNumElements(VecTy)), /*Insert=*/false,
/*Extract=*/true, CostKind) +
TTI.getInstructionCost(Inst, CostKind);
+ // The reduction-only cost cannot price the compared-values tree of i1
+ // and/or reductions, so ties are left to the full analysis.
+ if (Ty->isIntegerTy(1) &&
+ (Kind == RecurKind::And || Kind == RecurKind::Or)) {
+ TTI::CastContextHint Ctx =
+ all_of(Ops, [](Value *Op) { return isa<LoadInst>(Op); })
+ ? TTI::CastContextHint::Normal
+ : TTI::CastContextHint::None;
+ if (getI1ReductionCost(Kind, TTI, cast<FixedVectorType>(VecTy), Ty, Ctx,
+ CostKind)
+ .first > ScalarCost)
+ return false;
+ return HorRdx.tryToReduce(R, *DL, &TTI, *TLI, AC, *DT) != nullptr;
+ }
InstructionCost RedCost;
switch (::getRdxKind(Inst)) {
case RecurKind::Add:
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
index c5679c52c2a49..c18da825e7e2a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
@@ -14,8 +14,11 @@
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/Sequence.h"
#include "llvm/ADT/SmallVector.h"
+#include "llvm/Analysis/IVDescriptors.h"
+#include "llvm/IR/Constants.h"
#include "llvm/IR/DerivedTypes.h"
#include "llvm/IR/Instructions.h"
+#include "llvm/IR/Intrinsics.h"
#include "llvm/IR/Operator.h"
#include "llvm/IR/Type.h"
#include "llvm/IR/Value.h"
@@ -240,4 +243,62 @@ InstructionCost getExtractWithExtendCost(const TargetTransformInfo &TTI,
return TTI.getExtractWithExtendCost(Opcode, Dst, VecTy, Index, CostKind);
}
+static InstructionCost
+getBoolLogicRdxBitcastCost(RecurKind Kind, const TargetTransformInfo &TTI,
+ FixedVectorType *VectorTy, TTI::CastContextHint Ctx,
+ TTI::TargetCostKind CostKind) {
+ assert((Kind == RecurKind::And || Kind == RecurKind::Or) &&
+ VectorTy->getElementType()->isIntegerTy(1) &&
+ "Expected and/or reduction of i1");
+ auto *IntTy =
+ IntegerType::get(VectorTy->getContext(), getNumElements(VectorTy));
+ CmpInst::Predicate Pred =
+ Kind == RecurKind::And ? CmpInst::ICMP_EQ : CmpInst::ICMP_NE;
+ // The compare is against the all-ones (and) or zero (or) constant.
+ Constant *CmpConst = Kind == RecurKind::And
+ ? ConstantInt::getAllOnesValue(IntTy)
+ : ConstantInt::getNullValue(IntTy);
+ return TTI.getCastInstrCost(Instruction::BitCast, IntTy, VectorTy, Ctx,
+ CostKind) +
+ TTI.getCmpSelInstrCost(
+ Instruction::ICmp, IntTy, CmpInst::makeCmpResultType(IntTy), Pred,
+ CostKind, /*Op1Info=*/{}, TTI::getOperandInfo(CmpConst));
+}
+
+std::pair<InstructionCost, bool>
+getI1ReductionCost(RecurKind Kind, const TargetTransformInfo &TTI,
+ FixedVectorType *VectorTy, Type *ScalarTy,
+ TTI::CastContextHint Ctx, TTI::TargetCostKind CostKind) {
+ unsigned RdxOpcode = RecurrenceDescriptor::getOpcode(Kind);
+ if (Kind == RecurKind::And || Kind == RecurKind::Or) {
+ InstructionCost RdxCost = TTI.getArithmeticReductionCost(
+ RdxOpcode, VectorTy, std::nullopt, CostKind);
+ InstructionCost BitcastCost =
+ getBoolLogicRdxBitcastCost(Kind, TTI, VectorTy, Ctx, CostKind);
+ return {std::min(RdxCost, BitcastCost), BitcastCost < RdxCost};
+ }
+ assert(Kind == RecurKind::Add && !ScalarTy->isIntegerTy(1) &&
+ "Expected add reduction of zexted i1 values");
+ // The extended reduction form.
+ InstructionCost ExtRdxCost =
+ TTI.getExtendedReductionCost(RdxOpcode, /*IsUnsigned=*/true, ScalarTy,
+ VectorTy, std::nullopt, CostKind);
+ // The ctpop form is estimated as the cheaper of the scalar bitcast+ctpop
+ // and the vector ctpop on the mask type; it is emitted as bitcast+ctpop.
+ // The zext/trunc of the ctpop result to the destination type is absorbed by
+ // the legalization of the ctpop itself.
+ auto *IntTy =
+ IntegerType::get(VectorTy->getContext(), getNumElements(VectorTy));
+ InstructionCost CtpopCost = std::min(
+ TTI.getCastInstrCost(Instruction::BitCast, IntTy, VectorTy, Ctx,
+ CostKind) +
+ TTI.getIntrinsicInstrCost(
+ IntrinsicCostAttributes(Intrinsic::ctpop, IntTy, {IntTy}),
+ CostKind),
+ TTI.getIntrinsicInstrCost(
+ IntrinsicCostAttributes(Intrinsic::ctpop, VectorTy, {VectorTy}),
+ CostKind));
+ return {std::min(ExtRdxCost, CtpopCost), CtpopCost <= ExtRdxCost};
+}
+
} // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
index 75ad9fe88db80..c29782bd4de5b 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
@@ -30,6 +30,7 @@ class Type;
class User;
class Value;
class VectorType;
+enum class RecurKind;
} // namespace llvm
namespace llvm::slpvectorizer {
@@ -99,6 +100,16 @@ getExtractWithExtendCost(const TargetTransformInfo &TTI, bool ReVec,
unsigned Index,
const TargetTransformInfo::TargetCostKind CostKind);
+/// i1 reductions can be emitted as the plain target reduction or in the
+/// bitcast-based form (bitcast to a scalar integer type plus a compare for
+/// and/or, plus ctpop for add). Returns the cost of the cheaper form and
+/// whether it is the bitcast-based one.
+std::pair<InstructionCost, bool>
+getI1ReductionCost(RecurKind Kind, const TargetTransformInfo &TTI,
+ FixedVectorType *VectorTy, Type *ScalarTy,
+ TargetTransformInfo::CastContextHint Ctx,
+ TargetTransformInfo::TargetCostKind CostKind);
+
} // namespace llvm::slpvectorizer
#endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll b/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
index 78f68c420f6ab..fe21d4039f240 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
@@ -277,13 +277,12 @@ define i1 @cmp_lt_gt(double %a, double %b, double %c) {
; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <2 x double> [[TMP5]], <2 x double> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: [[TMP7:%.*]] = fdiv <2 x double> [[TMP3]], [[TMP6]]
; CHECK-NEXT: [[TMP8:%.*]] = fcmp uge <2 x double> [[TMP7]], splat (double f0x3EB0C6F7A0B5ED8D)
-; CHECK-NEXT: [[SHIFT:%.*]] = shufflevector <2 x i1> [[TMP8]], <2 x i1> poison, <2 x i32> <i32 1, i32 poison>
-; CHECK-NEXT: [[FOLDEXTEXTBINOP:%.*]] = or <2 x i1> [[TMP8]], [[SHIFT]]
+; CHECK-NEXT: [[TMP11:%.*]] = bitcast <2 x i1> [[TMP8]] to i2
+; CHECK-NEXT: [[TMP12:%.*]] = icmp ne i2 [[TMP11]], 0
; CHECK-NEXT: [[TMP9:%.*]] = fcmp ule <2 x double> [[TMP7]], splat (double 1.000000e+00)
-; CHECK-NEXT: [[SHIFT3:%.*]] = shufflevector <2 x i1> [[TMP9]], <2 x i1> poison, <2 x i32> <i32 1, i32 poison>
-; CHECK-NEXT: [[TMP10:%.*]] = or <2 x i1> [[TMP9]], [[SHIFT3]]
-; CHECK-NEXT: [[FOLDEXTEXTBINOP4:%.*]] = and <2 x i1> [[FOLDEXTEXTBINOP]], [[TMP10]]
-; CHECK-NEXT: [[RETVAL_0:%.*]] = extractelement <2 x i1> [[FOLDEXTEXTBINOP4]], i64 0
+; CHECK-NEXT: [[TMP13:%.*]] = bitcast <2 x i1> [[TMP9]] to i2
+; CHECK-NEXT: [[TMP10:%.*]] = icmp ne i2 [[TMP13]], 0
+; CHECK-NEXT: [[RETVAL_0:%.*]] = and i1 [[TMP12]], [[TMP10]]
; CHECK-NEXT: ret i1 [[RETVAL_0]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
index c98a4d1f2c182..35d804d927ff0 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
@@ -4,141 +4,143 @@
define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i64 %.sroa.3308.0.copyload, i64 %.neg1, i64 %indvar3788, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8) {
; CHECK-LABEL: define void @test(
; CHECK-SAME: i64 [[TMP0:%.*]], i64 [[TMP1:%.*]], i64 [[TMP2:%.*]], i64 [[TMP3:%.*]], i64 [[DOTSROA_3341_0_COPYLOAD:%.*]], i64 [[DOTSROA_3308_0_COPYLOAD:%.*]], i64 [[DOTNEG1:%.*]], i64 [[INDVAR3788:%.*]], i64 [[TMP4:%.*]], i64 [[TMP5:%.*]], i64 [[TMP6:%.*]], i64 [[TMP7:%.*]], i64 [[TMP8:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[_LR_PH_PREHEADER:.*:]]
+; CHECK-NEXT: [[_LR_PH_PREHEADER:.*]]:
; CHECK-NEXT: [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[TMP1]], i64 0
; CHECK-NEXT: [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[TMP0]], i64 1
-; CHECK-NEXT: [[TMP13:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 24>
+; CHECK-NEXT: [[TMP12:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 24>
; CHECK-NEXT: [[TMP17:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 40>
; CHECK-NEXT: [[TMP11:%.*]] = mul i64 [[TMP0]], 48
; CHECK-NEXT: [[TMP14:%.*]] = insertelement <2 x i64> <i64 poison, i64 1>, i64 [[TMP0]], i64 0
-; CHECK-NEXT: [[TMP15:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> <i64 -1, i64 poison>, <2 x i32> <i32 2, i32 0>
-; CHECK-NEXT: [[TMP16:%.*]] = sub <2 x i64> [[TMP14]], [[TMP15]]
-; CHECK-NEXT: [[TMP20:%.*]] = shl i64 [[TMP0]], 11
+; CHECK-NEXT: [[TMP22:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> <i64 -1, i64 poison>, <2 x i32> <i32 2, i32 0>
+; CHECK-NEXT: [[TMP16:%.*]] = sub <2 x i64> [[TMP14]], [[TMP22]]
+; CHECK-NEXT: [[TMP25:%.*]] = shl i64 [[TMP0]], 11
; CHECK-NEXT: [[TMP35:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 0
-; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <2 x i64> [[TMP35]], <2 x i64> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP25:%.*]] = shl <2 x i64> [[TMP12]], <i64 11, i64 0>
-; CHECK-NEXT: [[TMP21:%.*]] = sub i64 1, [[TMP20]]
-; CHECK-NEXT: [[TMP22:%.*]] = add <2 x i64> [[TMP25]], <i64 8, i64 1>
-; CHECK-NEXT: [[TMP23:%.*]] = or <2 x i64> [[TMP25]], <i64 8, i64 1>
-; CHECK-NEXT: [[TMP33:%.*]] = shufflevector <2 x i64> [[TMP22]], <2 x i64> [[TMP23]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: [[TMP33:%.*]] = shufflevector <2 x i64> [[TMP35]], <2 x i64> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP20:%.*]] = shl <2 x i64> [[TMP33]], <i64 11, i64 0>
+; CHECK-NEXT: [[TMP21:%.*]] = sub i64 1, [[TMP25]]
+; CHECK-NEXT: [[TMP41:%.*]] = add <2 x i64> [[TMP20]], <i64 8, i64 1>
+; CHECK-NEXT: [[TMP23:%.*]] = or <2 x i64> [[TMP20]], <i64 8, i64 1>
+; CHECK-NEXT: [[TMP44:%.*]] = shufflevector <2 x i64> [[TMP41]], <...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/223163
More information about the llvm-commits
mailing list