[llvm] [SLP]Vectorize bool-to-bitmask reductions as a bitcast of zero tests (PR #223434)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 06:22:40 PDT 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/223434

>From d70649849a9c6dfc6477de8a38bc3d016b937fa5 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Mon, 14 Sep 2026 08:19:51 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 118 ++++++++++++++++--
 .../SLPVectorizer/SLPCostAnalysis.cpp         |  29 +++++
 .../Vectorize/SLPVectorizer/SLPCostAnalysis.h |  10 ++
 .../SLPVectorizer/SLPReductionUtils.cpp       |  28 +++++
 .../SLPVectorizer/SLPReductionUtils.h         |  20 +++
 .../Transforms/SLPVectorizer/X86/bool-mask.ll | 101 +++++++++++++++
 6 files changed, 297 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 8f34e030ead34..11a8998fd6838 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -31398,14 +31398,20 @@ class HorizontalReduction {
               AnyMask = true;
             }
           }
-          if (AnyMask)
-            VectorizedRoot = Builder.CreateAnd(VectorizedRoot,
-                                               ConstantVector::get(MaskConsts));
-          VectorizedRoot =
-              Builder.CreateZExt(VectorizedRoot, getWidenedType(WideTy, VF));
-          if (AnyShift)
-            VectorizedRoot = Builder.CreateShl(
-                VectorizedRoot, ConstantVector::get(ShiftConsts));
+          if (Value *Bitmask = emitBoolBitmaskRdx(
+                  Builder, V, *TTI, DL, VectorizedRoot, VL, TrackedToOrig, Pos,
+                  MaskConsts, GroupRdxFMF))
+            VectorizedRoot = Bitmask;
+          else {
+            if (AnyMask)
+              VectorizedRoot = Builder.CreateAnd(
+                  VectorizedRoot, ConstantVector::get(MaskConsts));
+            VectorizedRoot =
+                Builder.CreateZExt(VectorizedRoot, getWidenedType(WideTy, VF));
+            if (AnyShift)
+              VectorizedRoot = Builder.CreateShl(
+                  VectorizedRoot, ConstantVector::get(ShiftConsts));
+          }
         }
 
         Type *ScalarTy = VL.front()->getType();
@@ -31421,7 +31427,8 @@ class HorizontalReduction {
              RedScalarTy != ScalarTy->getScalarType()
                  ? NarrowedLeafShifts.empty() && V.isSignedMinBitwidthRootNode()
                  : true,
-             V.isReducedBitcastRoot() || V.isReducedCmpBitcastRoot(),
+             V.isReducedBitcastRoot() || V.isReducedCmpBitcastRoot() ||
+                 !VectorizedRoot->getType()->isVectorTy(),
              GroupNegated});
 
         // Count vectorized reduced values to exclude them from final reduction.
@@ -32047,6 +32054,85 @@ class HorizontalReduction {
     return Rdx;
   }
 
+  /// Emits the boolean bitmask form of the narrowed-leaf or-reduction (a
+  /// bitcast of the per-lane zero tests to an integer) if it covers the whole
+  /// reduction in a single vector part and is cheaper than the shift and
+  /// reduction sequence. Returns nullptr otherwise.
+  Value *emitBoolBitmaskRdx(IRBuilderBase &Builder, const BoUpSLP &R,
+                            const TargetTransformInfo &TTI,
+                            const DataLayout &DL, Value *VectorizedRoot,
+                            ArrayRef<Value *> VL,
+                            ArrayRef<Value *> TrackedToOrig, unsigned Pos,
+                            ArrayRef<Constant *> MaskConsts,
+                            FastMathFlags FMF) const {
+    Type *WideTy = ReductionRoot->getType();
+    auto *NarrowVecTy = dyn_cast<VectorType>(VectorizedRoot->getType());
+    if (!NarrowVecTy)
+      return nullptr;
+    Type *NarrowTy = NarrowVecTy->getScalarType();
+    unsigned VF = getNumElements(NarrowVecTy);
+    // Applies only when the whole reduction is a single vector part.
+    if (NarrowedLeafShifts.size() != VL.size() || VF != VL.size() ||
+        !VectorValuesAndScales.empty() ||
+        !R.getRootNode().ReuseShuffleIndices.empty())
+      return nullptr;
+    BoolBitmask Match = isBoolBitmaskRdx(RdxKind, NarrowedLeafShifts, DL);
+    if (Match == BoolBitmask::None)
+      return nullptr;
+    // Emit the bitmask form only if it is cheaper than the shift and
+    // reduction sequence.
+    const TTI::TargetCostKind CostKind = R.getCostKind();
+    auto *WideVecTy = dyn_cast<VectorType>(getWidenedType(WideTy, VF));
+    if (!WideVecTy)
+      return nullptr;
+    const auto *CxtI = cast<Instruction>(ReductionRoot);
+    InstructionCost GenericCost =
+        TTI.getExtendedReductionCost(Instruction::Or, /*IsUnsigned=*/true,
+                                     WideTy, NarrowVecTy, FMF, CostKind);
+    if (any_of(NarrowedLeafShifts,
+               [](const auto &P) { return P.second.Shift != 0; }))
+      GenericCost += TTI.getArithmeticInstrCost(
+          Instruction::Shl, WideVecTy, CostKind,
+          {TTI::OK_AnyValue, TTI::OP_None},
+          {TTI::OK_NonUniformConstantValue, TTI::OP_None}, {}, CxtI);
+    if (any_of(NarrowedLeafShifts,
+               [](const auto &P) { return !P.second.Mask.isAllOnes(); }))
+      GenericCost += TTI.getArithmeticInstrCost(
+          Instruction::And, NarrowVecTy, CostKind,
+          {TTI::OK_AnyValue, TTI::OP_None},
+          {TTI::OK_NonUniformConstantValue, TTI::OP_None}, {}, CxtI);
+    if (getBoolBitmaskCost(TTI, Match == BoolBitmask::NeedMask, NarrowTy,
+                           WideTy, VF, ReductionRoot, CostKind) >= GenericCost)
+      return nullptr;
+    Value *Vec = VectorizedRoot;
+    switch (Match) {
+    case BoolBitmask::NeedMask:
+      Vec = Builder.CreateAnd(Vec, ConstantVector::get(MaskConsts));
+      break;
+    case BoolBitmask::NoMask:
+      break;
+    case BoolBitmask::None:
+      llvm_unreachable("unexpected bool bitmask state");
+    }
+    // Reorder the lanes by their bit position and pack the zero tests into an
+    // integer.
+    SmallVector<int> PermMask(VF, PoisonMaskElem);
+    for (auto [Idx, Val] : enumerate(VL))
+      PermMask[NarrowedLeafShifts.at(TrackedToOrig[Pos + Idx]).Shift] =
+          R.findRootLaneForValue(Val);
+    if (!ShuffleVectorInst::isIdentityMask(PermMask, VF))
+      Vec = Builder.CreateShuffleVector(Vec, PermMask);
+    if (!NarrowTy->isIntegerTy(1)) {
+      Vec = Builder.CreateICmpNE(Vec, Constant::getNullValue(Vec->getType()));
+      ++NumVectorInstructions;
+    }
+    Vec = Builder.CreateBitCast(Vec, Builder.getIntNTy(VF));
+    ++NumVectorInstructions;
+    Vec = Builder.CreateIntCast(Vec, WideTy, /*isSigned=*/false);
+    ++NumVectorInstructions;
+    return Vec;
+  }
+
   /// Calculate the cost of a reduction.
   InstructionCost getReductionCost(
       TargetTransformInfo *TTI, ArrayRef<Value *> ReducedVals,
@@ -32182,6 +32268,20 @@ class HorizontalReduction {
                      [](const auto &P) { return !P.second.Mask.isAllOnes(); }))
             VectorCost += TTI->getArithmeticInstrCost(Instruction::And,
                                                       NarrowVecTy, CostKind);
+          // The boolean bitmask emission may be cheaper.
+          BoolBitmask Match =
+              NarrowedLeafShifts.size() == ReduxWidth &&
+                      R.getRootNode().ReuseShuffleIndices.empty()
+                  ? isBoolBitmaskRdx(RdxKind, NarrowedLeafShifts, DL)
+                  : BoolBitmask::None;
+          if (Match != BoolBitmask::None) {
+            InstructionCost BitmaskCost = getBoolBitmaskCost(
+                *TTI, Match == BoolBitmask::NeedMask, ScalarTy, WideTy,
+                ReduxWidth, ReductionRoot, CostKind);
+            for (Instruction *I : NarrowedChainInsts)
+              BitmaskCost -= TTI->getInstructionCost(I, CostKind);
+            VectorCost = std::min(VectorCost, BitmaskCost);
+          }
         } else if (DoesRequireReductionOp) {
           if (auto *VecTy = dyn_cast<FixedVectorType>(ScalarTy)) {
             assert(SLPReVec && "FixedVectorType is not expected.");
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
index 72f65a3386be5..b664c9314a6c9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
@@ -305,4 +305,33 @@ InstructionCost getBoolReduxBitcastCmpCost(const TargetTransformInfo &TTI,
                                 TTI.getOperandInfo(CmpRHS), CmpI);
 }
 
+InstructionCost getBoolBitmaskCost(const TargetTransformInfo &TTI,
+                                   bool NeedMask, Type *NarrowScalarTy,
+                                   Type *WideTy, unsigned VF, const Value *Root,
+                                   const TTI::TargetCostKind CostKind) {
+  Type *NarrowVecTy = getWidenedType(NarrowScalarTy, VF);
+  Type *CmpTy = CmpInst::makeCmpResultType(NarrowVecTy);
+  auto *MaskTy = IntegerType::get(WideTy->getContext(), VF);
+  // The result cast inherits the uses of the reduction root.
+  TTI::CastContextHint CCH = getBoolReduxResultCCH(Root);
+  const auto *CxtI = cast<Instruction>(Root);
+  InstructionCost Cost = 0;
+  if (NeedMask)
+    Cost += TTI.getArithmeticInstrCost(
+        Instruction::And, NarrowVecTy, CostKind,
+        {TTI::OK_AnyValue, TTI::OP_None},
+        {TTI::OK_NonUniformConstantValue, TTI::OP_None}, {}, CxtI);
+  if (!NarrowScalarTy->isIntegerTy(1))
+    Cost += TTI.getCmpSelInstrCost(
+        Instruction::ICmp, NarrowVecTy, CmpTy, CmpInst::ICMP_NE, CostKind,
+        {TTI::OK_AnyValue, TTI::OP_None},
+        {TTI::OK_UniformConstantValue, TTI::OP_None});
+  Cost +=
+      TTI.getCastInstrCost(Instruction::BitCast, MaskTy, CmpTy, CCH, CostKind);
+  if (MaskTy != WideTy)
+    Cost +=
+        TTI.getCastInstrCost(Instruction::ZExt, WideTy, MaskTy, CCH, CostKind);
+  return Cost;
+}
+
 } // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
index fbeb01cca5656..fa4fbaad47530 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
@@ -91,6 +91,16 @@ getBoolReduxBitcastCmpCost(const TargetTransformInfo &TTI, RecurKind RdxKind,
                            ArrayRef<Instruction *> ChainInsts,
                            TargetTransformInfo::TargetCostKind CostKind);
 
+/// Returns the cost of the boolean bitmask reduction of a vector of boolean
+/// leaves of type \p NarrowScalarTy, emitted as [and] + zero test + bitcast
+/// [+ zext] to \p WideTy. \p Root is the reduction root, used as the context
+/// of the emitted instructions.
+InstructionCost
+getBoolBitmaskCost(const TargetTransformInfo &TTI, bool NeedMask,
+                   Type *NarrowScalarTy, Type *WideTy, unsigned VF,
+                   const Value *Root,
+                   TargetTransformInfo::TargetCostKind CostKind);
+
 /// This is similar to TargetTransformInfo::getScalarizationOverhead, but if
 /// ScalarTy is a FixedVectorType, a vector will be inserted or extracted
 /// instead of a scalar.
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.cpp
index f8004c5c3b4f5..4c6bcce49fdc1 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.cpp
@@ -9,9 +9,13 @@
 #include "SLPReductionUtils.h"
 
 #include "SLPCostAnalysis.h"
+#include "SLPUtils.h"
 
+#include "llvm/ADT/SmallBitVector.h"
 #include "llvm/Analysis/IVDescriptors.h"
+#include "llvm/Analysis/ValueTracking.h"
 #include "llvm/IR/Constants.h"
+#include "llvm/IR/DataLayout.h"
 #include "llvm/IR/IRBuilder.h"
 #include "llvm/IR/Instructions.h"
 #include "llvm/IR/Intrinsics.h"
@@ -68,6 +72,30 @@ Type *getBoolReduxWideTy(RecurKind RdxKind, Type *RootTy, Type *LeafTy) {
   return nullptr;
 }
 
+BoolBitmask isBoolBitmaskRdx(
+    RecurKind RdxKind,
+    const SmallDenseMap<Value *, NarrowedLeafInfo> &NarrowedLeafShifts,
+    const DataLayout &DL) {
+  if (RdxKind != RecurKind::Or || DL.isBigEndian() ||
+      NarrowedLeafShifts.empty())
+    return BoolBitmask::None;
+  unsigned NumLeaves = NarrowedLeafShifts.size();
+  SmallBitVector Seen(NumLeaves);
+  bool NeedMask = false;
+  for (const auto &[V, L] : NarrowedLeafShifts) {
+    if (L.Shift >= NumLeaves || Seen.test(L.Shift))
+      return BoolBitmask::None;
+    Seen.set(L.Shift);
+    KnownBits Known = computeKnownBits(V, DL);
+    // The masked leaf must be known to be 0 or 1.
+    if ((L.Mask & ~Known.Zero).ugt(1))
+      return BoolBitmask::None;
+    // The mask is redundant if it keeps all not-known-zero bits.
+    NeedMask |= !(Known.Zero | L.Mask).isAllOnes();
+  }
+  return NeedMask ? BoolBitmask::NeedMask : BoolBitmask::NoMask;
+}
+
 Value *tryEmitBoolReduxBitcastCmp(IRBuilderBase &Builder,
                                   const TargetTransformInfo &TTI,
                                   RecurKind RdxKind, Value *Vec,
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.h
index b1af337a0f263..8ffd6a3b2428f 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPReductionUtils.h
@@ -15,9 +15,11 @@
 #ifndef LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPREDUCTIONUTILS_H
 #define LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPREDUCTIONUTILS_H
 
+#include "llvm/ADT/DenseMap.h"
 #include "llvm/Analysis/TargetTransformInfo.h"
 
 namespace llvm {
+class DataLayout;
 class FastMathFlags;
 class IRBuilderBase;
 class Instruction;
@@ -29,12 +31,30 @@ enum class RecurKind;
 
 namespace llvm::slpvectorizer {
 
+struct NarrowedLeafInfo;
+
+/// The result of matching a boolean bitmask reduction over narrowed leaves.
+enum class BoolBitmask {
+  None,     // not a boolean bitmask reduction
+  NoMask,   // bitmask; the absorbed masks are redundant
+  NeedMask, // bitmask; the absorbed masks must be applied before the zero test
+};
+
 /// \returns the wide leaf type if the logical and/or reduction \p RdxKind
 /// with the i1 root type \p RootTy and the leaf type \p LeafTy is a
 /// booleanized reduction (performed in the wide leaf type, bit 0 of the
 /// result is the final value), nullptr otherwise.
 Type *getBoolReduxWideTy(RecurKind RdxKind, Type *RootTy, Type *LeafTy);
 
+/// \returns the BoolBitmask match if the or-reduction of the narrowed leaves
+/// packs each boolean (0 or 1 after masking) leaf into its own bit position
+/// 0..N-1, i.e. it is a bitcast of the per-lane zero tests to an iN integer;
+/// BoolBitmask::None otherwise.
+BoolBitmask isBoolBitmaskRdx(
+    RecurKind RdxKind,
+    const SmallDenseMap<Value *, NarrowedLeafInfo> &NarrowedLeafShifts,
+    const DataLayout &DL);
+
 /// \returns the first operand of \p I that does not match \p Phi. If
 /// the operand is not an instruction, returns nullptr.
 Instruction *getNonPhiOperand(Instruction *I, PHINode *Phi);
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll b/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll
index 7112b055647af..ca0a1d0d9c1bc 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll
@@ -439,3 +439,104 @@ entry:
   ret i64 %mask.1.11
 }
 
+define i64 @bitmask_16xi8_shl(ptr %src) {
+; SSE-LABEL: @bitmask_16xi8_shl(
+; SSE-NEXT:  entry:
+; SSE-NEXT:    [[TMP0:%.*]] = load <16 x i8>, ptr [[SRC:%.*]], align 1
+; SSE-NEXT:    [[TMP1:%.*]] = icmp ne <16 x i8> [[TMP0]], zeroinitializer
+; SSE-NEXT:    [[TMP2:%.*]] = bitcast <16 x i1> [[TMP1]] to i16
+; SSE-NEXT:    [[TMP3:%.*]] = zext i16 [[TMP2]] to i64
+; SSE-NEXT:    ret i64 [[TMP3]]
+;
+; AVX-LABEL: @bitmask_16xi8_shl(
+; AVX-NEXT:  entry:
+; AVX-NEXT:    [[TMP0:%.*]] = load <16 x i8>, ptr [[SRC:%.*]], align 1
+; AVX-NEXT:    [[TMP1:%.*]] = icmp ne <16 x i8> [[TMP0]], zeroinitializer
+; AVX-NEXT:    [[TMP2:%.*]] = bitcast <16 x i1> [[TMP1]] to i16
+; AVX-NEXT:    [[TMP3:%.*]] = zext i16 [[TMP2]] to i64
+; AVX-NEXT:    ret i64 [[TMP3]]
+;
+; AVX512-LABEL: @bitmask_16xi8_shl(
+; AVX512-NEXT:  entry:
+; AVX512-NEXT:    [[TMP0:%.*]] = load <16 x i8>, ptr [[SRC:%.*]], align 1
+; AVX512-NEXT:    [[TMP1:%.*]] = icmp ne <16 x i8> [[TMP0]], zeroinitializer
+; AVX512-NEXT:    [[TMP2:%.*]] = bitcast <16 x i1> [[TMP1]] to i16
+; AVX512-NEXT:    [[TMP3:%.*]] = zext i16 [[TMP2]] to i64
+; AVX512-NEXT:    ret i64 [[TMP3]]
+;
+entry:
+  %0 = load i8, ptr %src, align 1, !range !0, !noundef !1
+  %arrayidx.1 = getelementptr inbounds nuw i8, ptr %src, i64 1
+  %1 = load i8, ptr %arrayidx.1, align 1, !range !0, !noundef !1
+  %2 = shl nuw nsw i8 %1, 1
+  %mask.1.18 = or disjoint i8 %2, %0
+  %arrayidx.2 = getelementptr inbounds nuw i8, ptr %src, i64 2
+  %3 = load i8, ptr %arrayidx.2, align 1, !range !0, !noundef !1
+  %4 = shl nuw nsw i8 %3, 2
+  %mask.1.29 = or disjoint i8 %4, %mask.1.18
+  %arrayidx.3 = getelementptr inbounds nuw i8, ptr %src, i64 3
+  %5 = load i8, ptr %arrayidx.3, align 1, !range !0, !noundef !1
+  %6 = shl nuw nsw i8 %5, 3
+  %mask.1.310 = or disjoint i8 %6, %mask.1.29
+  %arrayidx.4 = getelementptr inbounds nuw i8, ptr %src, i64 4
+  %7 = load i8, ptr %arrayidx.4, align 1, !range !0, !noundef !1
+  %8 = shl nuw nsw i8 %7, 4
+  %mask.1.411 = or disjoint i8 %8, %mask.1.310
+  %arrayidx.5 = getelementptr inbounds nuw i8, ptr %src, i64 5
+  %9 = load i8, ptr %arrayidx.5, align 1, !range !0, !noundef !1
+  %10 = shl nuw nsw i8 %9, 5
+  %mask.1.512 = or i8 %10, %mask.1.411
+  %arrayidx.6 = getelementptr inbounds nuw i8, ptr %src, i64 6
+  %11 = load i8, ptr %arrayidx.6, align 1, !range !0, !noundef !1
+  %12 = shl nuw nsw i8 %11, 6
+  %mask.1.613 = or i8 %12, %mask.1.512
+  %arrayidx.7 = getelementptr inbounds nuw i8, ptr %src, i64 7
+  %13 = load i8, ptr %arrayidx.7, align 1, !range !0, !noundef !1
+  %14 = shl nuw i8 %13, 7
+  %mask.1.714 = or i8 %14, %mask.1.613
+  %mask.1.7 = zext i8 %mask.1.714 to i64
+  %arrayidx.8 = getelementptr inbounds nuw i8, ptr %src, i64 8
+  %15 = load i8, ptr %arrayidx.8, align 1, !range !0, !noundef !1
+  %16 = zext nneg i8 %15 to i64
+  %or.8 = shl nuw nsw i64 %16, 8
+  %mask.1.8 = or disjoint i64 %or.8, %mask.1.7
+  %arrayidx.9 = getelementptr inbounds nuw i8, ptr %src, i64 9
+  %17 = load i8, ptr %arrayidx.9, align 1, !range !0, !noundef !1
+  %18 = zext nneg i8 %17 to i64
+  %or.9 = shl nuw nsw i64 %18, 9
+  %mask.1.9 = or disjoint i64 %or.9, %mask.1.8
+  %arrayidx.10 = getelementptr inbounds nuw i8, ptr %src, i64 10
+  %19 = load i8, ptr %arrayidx.10, align 1, !range !0, !noundef !1
+  %20 = zext nneg i8 %19 to i64
+  %or.10 = shl nuw nsw i64 %20, 10
+  %mask.1.10 = or disjoint i64 %or.10, %mask.1.9
+  %arrayidx.11 = getelementptr inbounds nuw i8, ptr %src, i64 11
+  %21 = load i8, ptr %arrayidx.11, align 1, !range !0, !noundef !1
+  %22 = zext nneg i8 %21 to i64
+  %or.11 = shl nuw nsw i64 %22, 11
+  %mask.1.11 = or i64 %or.11, %mask.1.10
+  %arrayidx.12 = getelementptr inbounds nuw i8, ptr %src, i64 12
+  %23 = load i8, ptr %arrayidx.12, align 1, !range !0, !noundef !1
+  %24 = zext nneg i8 %23 to i64
+  %or.12 = shl nuw nsw i64 %24, 12
+  %mask.1.12 = or i64 %or.12, %mask.1.11
+  %arrayidx.13 = getelementptr inbounds nuw i8, ptr %src, i64 13
+  %25 = load i8, ptr %arrayidx.13, align 1, !range !0, !noundef !1
+  %26 = zext nneg i8 %25 to i64
+  %or.13 = shl nuw nsw i64 %26, 13
+  %mask.1.13 = or i64 %or.13, %mask.1.12
+  %arrayidx.14 = getelementptr inbounds nuw i8, ptr %src, i64 14
+  %27 = load i8, ptr %arrayidx.14, align 1, !range !0, !noundef !1
+  %28 = zext nneg i8 %27 to i64
+  %or.14 = shl nuw nsw i64 %28, 14
+  %mask.1.14 = or i64 %or.14, %mask.1.13
+  %arrayidx.15 = getelementptr inbounds nuw i8, ptr %src, i64 15
+  %29 = load i8, ptr %arrayidx.15, align 1, !range !0, !noundef !1
+  %30 = zext nneg i8 %29 to i64
+  %or.15 = shl nuw nsw i64 %30, 15
+  %mask.1.15 = or i64 %or.15, %mask.1.14
+  ret i64 %mask.1.15
+}
+
+!0 = !{i8 0, i8 2}
+!1 = !{}



More information about the llvm-commits mailing list