[llvm] [SLP]Vectorize bool-to-bitmask reductions as a bitcast of zero tests (PR #223434)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 06:22:40 PDT 2026
https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/223434
>From d70649849a9c6dfc6477de8a38bc3d016b937fa5 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Mon, 14 Sep 2026 08:19:51 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 118 ++++++++++++++++--
.../SLPVectorizer/SLPCostAnalysis.cpp | 29 +++++
.../Vectorize/SLPVectorizer/SLPCostAnalysis.h | 10 ++
.../SLPVectorizer/SLPReductionUtils.cpp | 28 +++++
.../SLPVectorizer/SLPReductionUtils.h | 20 +++
.../Transforms/SLPVectorizer/X86/bool-mask.ll | 101 +++++++++++++++
6 files changed, 297 insertions(+), 9 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 8f34e030ead34..11a8998fd6838 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -31398,14 +31398,20 @@ class HorizontalReduction {
AnyMask = true;
}
}
- if (AnyMask)
- VectorizedRoot = Builder.CreateAnd(VectorizedRoot,
- ConstantVector::get(MaskConsts));
- VectorizedRoot =
- Builder.CreateZExt(VectorizedRoot, getWidenedType(WideTy, VF));
- if (AnyShift)
- VectorizedRoot = Builder.CreateShl(
- VectorizedRoot, ConstantVector::get(ShiftConsts));
+ if (Value *Bitmask = emitBoolBitmaskRdx(
+ Builder, V, *TTI, DL, VectorizedRoot, VL, TrackedToOrig, Pos,
+ MaskConsts, GroupRdxFMF))
+ VectorizedRoot = Bitmask;
+ else {
+ if (AnyMask)
+ VectorizedRoot = Builder.CreateAnd(
+ VectorizedRoot, ConstantVector::get(MaskConsts));
+ VectorizedRoot =
+ Builder.CreateZExt(VectorizedRoot, getWidenedType(WideTy, VF));
+ if (AnyShift)
+ VectorizedRoot = Builder.CreateShl(
+ VectorizedRoot, ConstantVector::get(ShiftConsts));
+ }
}
Type *ScalarTy = VL.front()->getType();
@@ -31421,7 +31427,8 @@ class HorizontalReduction {
RedScalarTy != ScalarTy->getScalarType()
? NarrowedLeafShifts.empty() && V.isSignedMinBitwidthRootNode()
: true,
- V.isReducedBitcastRoot() || V.isReducedCmpBitcastRoot(),
+ V.isReducedBitcastRoot() || V.isReducedCmpBitcastRoot() ||
+ !VectorizedRoot->getType()->isVectorTy(),
GroupNegated});
// Count vectorized reduced values to exclude them from final reduction.
@@ -32047,6 +32054,85 @@ class HorizontalReduction {
return Rdx;
}
+ /// Emits the boolean bitmask form of the narrowed-leaf or-reduction (a
+ /// bitcast of the per-lane zero tests to an integer) if it covers the whole
+ /// reduction in a single vector part and is cheaper than the shift and
+ /// reduction sequence. Returns nullptr otherwise.
+ Value *emitBoolBitmaskRdx(IRBuilderBase &Builder, const BoUpSLP &R,
+ const TargetTransformInfo &TTI,
+ const DataLayout &DL, Value *VectorizedRoot,
+ ArrayRef<Value *> VL,
+ ArrayRef<Value *> TrackedToOrig, unsigned Pos,
+ ArrayRef<Constant *> MaskConsts,
+ FastMathFlags FMF) const {
+ Type *WideTy = ReductionRoot->getType();
+ auto *NarrowVecTy = dyn_cast<VectorType>(VectorizedRoot->getType());
+ if (!NarrowVecTy)
+ return nullptr;
+ Type *NarrowTy = NarrowVecTy->getScalarType();
+ unsigned VF = getNumElements(NarrowVecTy);
+ // Applies only when the whole reduction is a single vector part.
+ if (NarrowedLeafShifts.size() != VL.size() || VF != VL.size() ||
+ !VectorValuesAndScales.empty() ||
+ !R.getRootNode().ReuseShuffleIndices.empty())
+ return nullptr;
+ BoolBitmask Match = isBoolBitmaskRdx(RdxKind, NarrowedLeafShifts, DL);
+ if (Match == BoolBitmask::None)
+ return nullptr;
+ // Emit the bitmask form only if it is cheaper than the shift and
+ // reduction sequence.
+ const TTI::TargetCostKind CostKind = R.getCostKind();
+ auto *WideVecTy = dyn_cast<VectorType>(getWidenedType(WideTy, VF));
+ if (!WideVecTy)
+ return nullptr;
+ const auto *CxtI = cast<Instruction>(ReductionRoot);
+ InstructionCost GenericCost =
+ TTI.getExtendedReductionCost(Instruction::Or, /*IsUnsigned=*/true,
+ WideTy, NarrowVecTy, FMF, CostKind);
+ if (any_of(NarrowedLeafShifts,
+ [](const auto &P) { return P.second.Shift != 0; }))
+ GenericCost += TTI.getArithmeticInstrCost(
+ Instruction::Shl, WideVecTy, CostKind,
+ {TTI::OK_AnyValue, TTI::OP_None},
+ {TTI::OK_NonUniformConstantValue, TTI::OP_None}, {}, CxtI);
+ if (any_of(NarrowedLeafShifts,
+ [](const auto &P) { return !P.second.Mask.isAllOnes(); }))
+ GenericCost += TTI.getArithmeticInstrCost(
+ Instruction::And, NarrowVecTy, CostKind,
+ {TTI::OK_AnyValue, TTI::OP_None},
+ {TTI::OK_NonUniformConstantValue, TTI::OP_None}, {}, CxtI);
+ if (getBoolBitmaskCost(TTI, Match == BoolBitmask::NeedMask, NarrowTy,
+ WideTy, VF, ReductionRoot, CostKind) >= GenericCost)
+ return nullptr;
+ Value *Vec = VectorizedRoot;
+ switch (Match) {
+ case BoolBitmask::NeedMask:
+ Vec = Builder.CreateAnd(Vec, ConstantVector::get(MaskConsts));
+ break;
+ case BoolBitmask::NoMask:
+ break;
+ case BoolBitmask::None:
+ llvm_unreachable("unexpected bool bitmask state");
+ }
+ // Reorder the lanes by their bit position and pack the zero tests into an
+ // integer.
+ SmallVector<int> PermMask(VF, PoisonMaskElem);
+ for (auto [Idx, Val] : enumerate(VL))
+ PermMask[NarrowedLeafShifts.at(TrackedToOrig[Pos + Idx]).Shift] =
+ R.findRootLaneForValue(Val);
+ if (!ShuffleVectorInst::isIdentityMask(PermMask, VF))
+ Vec = Builder.CreateShuffleVector(Vec, PermMask);
+ if (!NarrowTy->isIntegerTy(1)) {
+ Vec = Builder.CreateICmpNE(Vec, Constant::getNullValue(Vec->getType()));
+ ++NumVectorInstructions;
+ }
+ Vec = Builder.CreateBitCast(Vec, Builder.getIntNTy(VF));
+ ++NumVectorInstructions;
+ Vec = Builder.CreateIntCast(Vec, WideTy, /*isSigned=*/false);
+ ++NumVectorInstructions;
+ return Vec;
+ }
+
/// Calculate the cost of a reduction.
InstructionCost getReductionCost(
TargetTransformInfo *TTI, ArrayRef<Value *> ReducedVals,
@@ -32182,6 +32268,20 @@ class HorizontalReduction {
[](const auto &P) { return !P.second.Mask.isAllOnes(); }))
VectorCost += TTI->getArithmeticInstrCost(Instruction::And,
NarrowVecTy, CostKind);
+ // The boolean bitmask emission may be cheaper.
+ BoolBitmask Match =
+ NarrowedLeafShifts.size() == ReduxWidth &&
+ R.getRootNode().ReuseShuffleIndices.empty()
+ ? isBoolBitmaskRdx(RdxKind, NarrowedLeafShifts, DL)
+ : BoolBitmask::None;
+ if (Match != BoolBitmask::None) {
+ InstructionCost BitmaskCost = getBoolBitmaskCost(
+ *TTI, Match == BoolBitmask::NeedMask, ScalarTy, WideTy,
+ ReduxWidth, ReductionRoot, CostKind);
+ for (Instruction *I : NarrowedChainInsts)
+ BitmaskCost -= TTI->getInstructionCost(I, CostKind);
+ VectorCost = std::min(VectorCost, BitmaskCost);
+ }
} else if (DoesRequireReductionOp) {
if (auto *VecTy = dyn_cast<FixedVectorType>(ScalarTy)) {
assert(SLPReVec && "FixedVectorType is not expected.");
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
index 72f65a3386be5..b664c9314a6c9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
@@ -305,4 +305,33 @@ InstructionCost getBoolReduxBitcastCmpCost(const TargetTransformInfo &TTI,
TTI.getOperandInfo(CmpRHS), CmpI);
}
+InstructionCost getBoolBitmaskCost(const TargetTransformInfo &TTI,
+ bool NeedMask, Type *NarrowScalarTy,
+ Type *WideTy, unsigned VF, const Value *Root,
+ const TTI::TargetCostKind CostKind) {
+ Type *NarrowVecTy = getWidenedType(NarrowScalarTy, VF);
+ Type *CmpTy = CmpInst::makeCmpResultType(NarrowVecTy);
+ auto *MaskTy = IntegerType::get(WideTy->getContext(), VF);
+ // The result cast inherits the uses of the reduction root.
+ TTI::CastContextHint CCH = getBoolReduxResultCCH(Root);
+ const auto *CxtI = cast<Instruction>(Root);
+ InstructionCost Cost = 0;
+ if (NeedMask)
+ Cost += TTI.getArithmeticInstrCost(
+ Instruction::And, NarrowVecTy, CostKind,
+ {TTI::OK_AnyValue, TTI::OP_None},
+ {TTI::OK_NonUniformConstantValue, TTI::OP_None}, {}, CxtI);
+ if (!NarrowScalarTy->isIntegerTy(1))
+ Cost += TTI.getCmpSelInstrCost(
+ Instruction::ICmp, NarrowVecTy, CmpTy, CmpInst::ICMP_NE, CostKind,
+ {TTI::OK_AnyValue, TTI::OP_None},
+ {TTI::OK_UniformConstantValue, TTI::OP_None});
+ Cost +=
+ TTI.getCastInstrCost(Instruction::BitCast, MaskTy, CmpTy, CCH, CostKind);
+ if (MaskTy != WideTy)
+ Cost +=
+ TTI.getCastInstrCost(Instruction::ZExt, WideTy, MaskTy, CCH, CostKind);
+ return Cost;
+}
+
} // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
index fbeb01cca5656..fa4fbaad47530 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
@@ -91,6 +91,16 @@ getBoolReduxBitcastCmpCost(const TargetTransformInfo &TTI, RecurKind RdxKind,
ArrayRef<Instruction *> ChainInsts,
TargetTransformInfo::TargetCostKind CostKind);
+/// Returns the cost of the boolean bitmask reduction of a vector of boolean
+/// leaves of type \p NarrowScalarTy, emitted as [and] + zero test + bitcast
+/// [+ zext] to \p WideTy. \p Root is the reduction root, used as the context
+/// of the emitted instructions.
+InstructionCost
+getBoolBitmaskCost(const TargetTransformInfo &TTI, bool NeedMask,
+ Type *NarrowScalarTy, Type *WideTy, unsigned VF,
+ const Value *Root,
+ TargetTransformInfo::TargetCostKind CostKind);
+
/// This is similar to TargetTransformInfo::getScalarizationOverhead, but if
/// ScalarTy is a FixedVectorType, a vector will be inserted or extracted
/// instead of a scalar.
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.cpp
index f8004c5c3b4f5..4c6bcce49fdc1 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.cpp
@@ -9,9 +9,13 @@
#include "SLPReductionUtils.h"
#include "SLPCostAnalysis.h"
+#include "SLPUtils.h"
+#include "llvm/ADT/SmallBitVector.h"
#include "llvm/Analysis/IVDescriptors.h"
+#include "llvm/Analysis/ValueTracking.h"
#include "llvm/IR/Constants.h"
+#include "llvm/IR/DataLayout.h"
#include "llvm/IR/IRBuilder.h"
#include "llvm/IR/Instructions.h"
#include "llvm/IR/Intrinsics.h"
@@ -68,6 +72,30 @@ Type *getBoolReduxWideTy(RecurKind RdxKind, Type *RootTy, Type *LeafTy) {
return nullptr;
}
+BoolBitmask isBoolBitmaskRdx(
+ RecurKind RdxKind,
+ const SmallDenseMap<Value *, NarrowedLeafInfo> &NarrowedLeafShifts,
+ const DataLayout &DL) {
+ if (RdxKind != RecurKind::Or || DL.isBigEndian() ||
+ NarrowedLeafShifts.empty())
+ return BoolBitmask::None;
+ unsigned NumLeaves = NarrowedLeafShifts.size();
+ SmallBitVector Seen(NumLeaves);
+ bool NeedMask = false;
+ for (const auto &[V, L] : NarrowedLeafShifts) {
+ if (L.Shift >= NumLeaves || Seen.test(L.Shift))
+ return BoolBitmask::None;
+ Seen.set(L.Shift);
+ KnownBits Known = computeKnownBits(V, DL);
+ // The masked leaf must be known to be 0 or 1.
+ if ((L.Mask & ~Known.Zero).ugt(1))
+ return BoolBitmask::None;
+ // The mask is redundant if it keeps all not-known-zero bits.
+ NeedMask |= !(Known.Zero | L.Mask).isAllOnes();
+ }
+ return NeedMask ? BoolBitmask::NeedMask : BoolBitmask::NoMask;
+}
+
Value *tryEmitBoolReduxBitcastCmp(IRBuilderBase &Builder,
const TargetTransformInfo &TTI,
RecurKind RdxKind, Value *Vec,
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.h
index b1af337a0f263..8ffd6a3b2428f 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.h
@@ -15,9 +15,11 @@
#ifndef LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPREDUCTIONUTILS_H
#define LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPREDUCTIONUTILS_H
+#include "llvm/ADT/DenseMap.h"
#include "llvm/Analysis/TargetTransformInfo.h"
namespace llvm {
+class DataLayout;
class FastMathFlags;
class IRBuilderBase;
class Instruction;
@@ -29,12 +31,30 @@ enum class RecurKind;
namespace llvm::slpvectorizer {
+struct NarrowedLeafInfo;
+
+/// The result of matching a boolean bitmask reduction over narrowed leaves.
+enum class BoolBitmask {
+ None, // not a boolean bitmask reduction
+ NoMask, // bitmask; the absorbed masks are redundant
+ NeedMask, // bitmask; the absorbed masks must be applied before the zero test
+};
+
/// \returns the wide leaf type if the logical and/or reduction \p RdxKind
/// with the i1 root type \p RootTy and the leaf type \p LeafTy is a
/// booleanized reduction (performed in the wide leaf type, bit 0 of the
/// result is the final value), nullptr otherwise.
Type *getBoolReduxWideTy(RecurKind RdxKind, Type *RootTy, Type *LeafTy);
+/// \returns the BoolBitmask match if the or-reduction of the narrowed leaves
+/// packs each boolean (0 or 1 after masking) leaf into its own bit position
+/// 0..N-1, i.e. it is a bitcast of the per-lane zero tests to an iN integer;
+/// BoolBitmask::None otherwise.
+BoolBitmask isBoolBitmaskRdx(
+ RecurKind RdxKind,
+ const SmallDenseMap<Value *, NarrowedLeafInfo> &NarrowedLeafShifts,
+ const DataLayout &DL);
+
/// \returns the first operand of \p I that does not match \p Phi. If
/// the operand is not an instruction, returns nullptr.
Instruction *getNonPhiOperand(Instruction *I, PHINode *Phi);
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll b/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll
index 7112b055647af..ca0a1d0d9c1bc 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll
@@ -439,3 +439,104 @@ entry:
ret i64 %mask.1.11
}
+define i64 @bitmask_16xi8_shl(ptr %src) {
+; SSE-LABEL: @bitmask_16xi8_shl(
+; SSE-NEXT: entry:
+; SSE-NEXT: [[TMP0:%.*]] = load <16 x i8>, ptr [[SRC:%.*]], align 1
+; SSE-NEXT: [[TMP1:%.*]] = icmp ne <16 x i8> [[TMP0]], zeroinitializer
+; SSE-NEXT: [[TMP2:%.*]] = bitcast <16 x i1> [[TMP1]] to i16
+; SSE-NEXT: [[TMP3:%.*]] = zext i16 [[TMP2]] to i64
+; SSE-NEXT: ret i64 [[TMP3]]
+;
+; AVX-LABEL: @bitmask_16xi8_shl(
+; AVX-NEXT: entry:
+; AVX-NEXT: [[TMP0:%.*]] = load <16 x i8>, ptr [[SRC:%.*]], align 1
+; AVX-NEXT: [[TMP1:%.*]] = icmp ne <16 x i8> [[TMP0]], zeroinitializer
+; AVX-NEXT: [[TMP2:%.*]] = bitcast <16 x i1> [[TMP1]] to i16
+; AVX-NEXT: [[TMP3:%.*]] = zext i16 [[TMP2]] to i64
+; AVX-NEXT: ret i64 [[TMP3]]
+;
+; AVX512-LABEL: @bitmask_16xi8_shl(
+; AVX512-NEXT: entry:
+; AVX512-NEXT: [[TMP0:%.*]] = load <16 x i8>, ptr [[SRC:%.*]], align 1
+; AVX512-NEXT: [[TMP1:%.*]] = icmp ne <16 x i8> [[TMP0]], zeroinitializer
+; AVX512-NEXT: [[TMP2:%.*]] = bitcast <16 x i1> [[TMP1]] to i16
+; AVX512-NEXT: [[TMP3:%.*]] = zext i16 [[TMP2]] to i64
+; AVX512-NEXT: ret i64 [[TMP3]]
+;
+entry:
+ %0 = load i8, ptr %src, align 1, !range !0, !noundef !1
+ %arrayidx.1 = getelementptr inbounds nuw i8, ptr %src, i64 1
+ %1 = load i8, ptr %arrayidx.1, align 1, !range !0, !noundef !1
+ %2 = shl nuw nsw i8 %1, 1
+ %mask.1.18 = or disjoint i8 %2, %0
+ %arrayidx.2 = getelementptr inbounds nuw i8, ptr %src, i64 2
+ %3 = load i8, ptr %arrayidx.2, align 1, !range !0, !noundef !1
+ %4 = shl nuw nsw i8 %3, 2
+ %mask.1.29 = or disjoint i8 %4, %mask.1.18
+ %arrayidx.3 = getelementptr inbounds nuw i8, ptr %src, i64 3
+ %5 = load i8, ptr %arrayidx.3, align 1, !range !0, !noundef !1
+ %6 = shl nuw nsw i8 %5, 3
+ %mask.1.310 = or disjoint i8 %6, %mask.1.29
+ %arrayidx.4 = getelementptr inbounds nuw i8, ptr %src, i64 4
+ %7 = load i8, ptr %arrayidx.4, align 1, !range !0, !noundef !1
+ %8 = shl nuw nsw i8 %7, 4
+ %mask.1.411 = or disjoint i8 %8, %mask.1.310
+ %arrayidx.5 = getelementptr inbounds nuw i8, ptr %src, i64 5
+ %9 = load i8, ptr %arrayidx.5, align 1, !range !0, !noundef !1
+ %10 = shl nuw nsw i8 %9, 5
+ %mask.1.512 = or i8 %10, %mask.1.411
+ %arrayidx.6 = getelementptr inbounds nuw i8, ptr %src, i64 6
+ %11 = load i8, ptr %arrayidx.6, align 1, !range !0, !noundef !1
+ %12 = shl nuw nsw i8 %11, 6
+ %mask.1.613 = or i8 %12, %mask.1.512
+ %arrayidx.7 = getelementptr inbounds nuw i8, ptr %src, i64 7
+ %13 = load i8, ptr %arrayidx.7, align 1, !range !0, !noundef !1
+ %14 = shl nuw i8 %13, 7
+ %mask.1.714 = or i8 %14, %mask.1.613
+ %mask.1.7 = zext i8 %mask.1.714 to i64
+ %arrayidx.8 = getelementptr inbounds nuw i8, ptr %src, i64 8
+ %15 = load i8, ptr %arrayidx.8, align 1, !range !0, !noundef !1
+ %16 = zext nneg i8 %15 to i64
+ %or.8 = shl nuw nsw i64 %16, 8
+ %mask.1.8 = or disjoint i64 %or.8, %mask.1.7
+ %arrayidx.9 = getelementptr inbounds nuw i8, ptr %src, i64 9
+ %17 = load i8, ptr %arrayidx.9, align 1, !range !0, !noundef !1
+ %18 = zext nneg i8 %17 to i64
+ %or.9 = shl nuw nsw i64 %18, 9
+ %mask.1.9 = or disjoint i64 %or.9, %mask.1.8
+ %arrayidx.10 = getelementptr inbounds nuw i8, ptr %src, i64 10
+ %19 = load i8, ptr %arrayidx.10, align 1, !range !0, !noundef !1
+ %20 = zext nneg i8 %19 to i64
+ %or.10 = shl nuw nsw i64 %20, 10
+ %mask.1.10 = or disjoint i64 %or.10, %mask.1.9
+ %arrayidx.11 = getelementptr inbounds nuw i8, ptr %src, i64 11
+ %21 = load i8, ptr %arrayidx.11, align 1, !range !0, !noundef !1
+ %22 = zext nneg i8 %21 to i64
+ %or.11 = shl nuw nsw i64 %22, 11
+ %mask.1.11 = or i64 %or.11, %mask.1.10
+ %arrayidx.12 = getelementptr inbounds nuw i8, ptr %src, i64 12
+ %23 = load i8, ptr %arrayidx.12, align 1, !range !0, !noundef !1
+ %24 = zext nneg i8 %23 to i64
+ %or.12 = shl nuw nsw i64 %24, 12
+ %mask.1.12 = or i64 %or.12, %mask.1.11
+ %arrayidx.13 = getelementptr inbounds nuw i8, ptr %src, i64 13
+ %25 = load i8, ptr %arrayidx.13, align 1, !range !0, !noundef !1
+ %26 = zext nneg i8 %25 to i64
+ %or.13 = shl nuw nsw i64 %26, 13
+ %mask.1.13 = or i64 %or.13, %mask.1.12
+ %arrayidx.14 = getelementptr inbounds nuw i8, ptr %src, i64 14
+ %27 = load i8, ptr %arrayidx.14, align 1, !range !0, !noundef !1
+ %28 = zext nneg i8 %27 to i64
+ %or.14 = shl nuw nsw i64 %28, 14
+ %mask.1.14 = or i64 %or.14, %mask.1.13
+ %arrayidx.15 = getelementptr inbounds nuw i8, ptr %src, i64 15
+ %29 = load i8, ptr %arrayidx.15, align 1, !range !0, !noundef !1
+ %30 = zext nneg i8 %29 to i64
+ %or.15 = shl nuw nsw i64 %30, 15
+ %mask.1.15 = or i64 %or.15, %mask.1.14
+ ret i64 %mask.1.15
+}
+
+!0 = !{i8 0, i8 2}
+!1 = !{}
More information about the llvm-commits
mailing list