[llvm] [SLP]Model gathers of extracted integer sub-fields as bitcast+permute+ext (PR #224919)

via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 20 04:54:38 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Alexey Bataev (alexey-bataev)

<details>
<summary>Changes</summary>

Emit gathered zero-extended sub-fields of one wider integer scalar as a
bitcast to the field vector plus a permutation and an extension, instead
of an insertelement chain or a splat-shift+trunc tree.

Fixes #<!-- -->55693

Assisted-by: Cursor


---

Patch is 91.70 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/224919.diff


7 Files Affected:

- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+128-2) 
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp (+133) 
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h (+9) 
- (modified) llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll (+2-5) 
- (modified) llvm/test/Transforms/PhaseOrdering/X86/avg.ll (+123-599) 
- (modified) llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll (+20-136) 
- (modified) llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll (+1-5) 


``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 205311d10e76c..dfc2fe49b8cee 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -2552,6 +2552,15 @@ class slpvectorizer::BoUpSLP {
   template <typename BVTy, typename ResTy, typename... Args>
   ResTy processBuildVector(const TreeEntry *E, Type *ScalarTy, Args &...Params);
 
+  /// Handles the gather of zero-extended sub-fields of the same wider integer
+  /// scalar, emitted as a bitcast to the field vector plus a permutation and
+  /// an extension to the lane type, if the gathered values match.
+  template <typename ResTy, typename BVTy>
+  std::optional<ResTy> processExtractedFieldsGather(
+      BVTy &ShuffleBuilder, const TreeEntry *E, Type *ScalarTy,
+      ArrayRef<Value *> VL,
+      ArrayRef<std::pair<const TreeEntry *, unsigned>> SubVectors);
+
   /// Create a new vector from a list of scalar values.  Produces a sequence
   /// which exploits values reused across lanes, and arranges the inserts
   /// for ease of later optimization.
@@ -12081,7 +12090,12 @@ void BoUpSLP::buildTreeRec(ArrayRef<Value *> VLRef, unsigned Depth,
   TreeEntry::EntryState State = getScalarsVectorizationState(
       S, VL, IsScatterVectorizeUserTE, CurrentOrder, PointerOps, SPtrInfo,
       ExpandShuffleMask);
-  if (State == TreeEntry::NeedToGather) {
+  // The zero-extended sub-fields of the same wider integer scalar are
+  // gathered and emitted as a bitcast to the field vector plus a permutation.
+  // For 2 lanes the emission is not cheaper than the insertelement chain.
+  if (State == TreeEntry::NeedToGather ||
+      (VL.size() > 2 && ReuseShuffleIndices.empty() &&
+       matchGatheredExtractedFields(VL, *DL))) {
     newGatherTreeEntry(VL, S, UserTreeIdx, ReuseShuffleIndices);
     return;
   }
@@ -14145,7 +14159,9 @@ void BoUpSLP::transformNodes() {
             // We use allSameOpcode instead of isAltShuffle because we don't
             // want to use interchangeable instruction here.
             !allSameOpcode(VL) || !allSameBlock(VL)) ||
-          allConstant(VL) || isSplat(VL))
+          allConstant(VL) || isSplat(VL) ||
+          (E.ReuseShuffleIndices.empty() &&
+           matchGatheredExtractedFields(VL, *DL)))
         continue;
       if (ForceLoadGather && E.hasState() && E.getOpcode() == Instruction::Load)
         continue;
@@ -15527,6 +15543,43 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
             cast<FixedVectorType>(Root->getType())->getNumElements()),
         getAllOnesValue(*R.DL, ScalarTy->getScalarType()));
   }
+  /// Estimates the cost of the gather of zero-extended sub-fields of width
+  /// \p FieldWidth of the same wider integer scalar \p Src, permuted by
+  /// \p Mask, as a bitcast to the field vector plus a permutation and an
+  /// extension to the lane type.
+  InstructionCost createExtractedFieldsVector(Value *Src, unsigned FieldWidth,
+                                              ArrayRef<int> Mask,
+                                              const TreeEntry &E) {
+    auto *FieldTy = IntegerType::get(Src->getContext(), FieldWidth);
+    unsigned SrcWidth = Src->getType()->getIntegerBitWidth();
+    assert(SrcWidth % FieldWidth == 0 &&
+           "Expected the field width to divide the source width.");
+    unsigned NumFields = SrcWidth / FieldWidth;
+    auto *FieldVecTy = cast<VectorType>(getWidenedType(FieldTy, NumFields));
+    TTI::CastContextHint CCH = R.getCastContextHint(E);
+    InstructionCost Cost = TTI.getCastInstrCost(
+        Instruction::BitCast, FieldVecTy, Src->getType(), CCH, CostKind);
+    // The permutation is elided only for the full-length identity mask.
+    if (Mask.size() != NumFields ||
+        !ShuffleVectorInst::isIdentityMask(Mask, NumFields))
+      Cost += getShuffleCost(TTI, TTI::SK_PermuteSingleSrc, FieldVecTy,
+                             CostKind, Mask);
+    if (!ScalarTy->isIntegerTy(FieldWidth))
+      Cost += TTI.getCastInstrCost(
+          Instruction::ZExt, getWidenedType(ScalarTy, Mask.size()),
+          getWidenedType(FieldTy, Mask.size()), CCH, CostKind);
+    // The extraction instructions die once the fields are emitted from the
+    // source scalar; their scalar cost is credited back for the gathers
+    // directly feeding the reduction. Instructions vectorized elsewhere are
+    // skipped: their scalar cost is already a part of the scalar baseline.
+    if (R.UserIgnoreList &&
+        (!E.UserTreeIndex || E.UserTreeIndex.UserTE->Idx == 0))
+      for (Instruction *I : make_isa_range<Instruction>(E.Scalars))
+        if (CheckedExtracts.insert(I).second && !R.isVectorized(I) &&
+            R.areAllUsersVectorized(I, &VectorizedVals))
+          Cost -= TTI.getInstructionCost(I, CostKind);
+    return Cost;
+  }
   InstructionCost createFreeze(InstructionCost Cost) { return Cost; }
   /// Finalize emission of the shuffles.
   InstructionCost finalize(
@@ -15664,6 +15717,17 @@ TTI::CastContextHint BoUpSLP::getCastContextHint(const TreeEntry &TE) const {
     if (ShuffleVectorInst::isReverseMask(Mask, Mask.size()))
       return TTI::CastContextHint::Reversed;
   }
+  // A gather of extracted sub-fields inherits the context of the entry
+  // vectorizing the common source scalar, or of the source scalar load.
+  if (TE.isGather())
+    if (std::optional<std::tuple<Value *, unsigned, SmallVector<int>>> Fields =
+            matchGatheredExtractedFields(TE.Scalars, *DL)) {
+      Value *Src = std::get<0>(*Fields);
+      if (ArrayRef<TreeEntry *> TEs = getTreeEntries(Src); !TEs.empty())
+        return getCastContextHint(*TEs.front());
+      if (isa<LoadInst>(Src))
+        return TTI::CastContextHint::Normal;
+    }
   return TTI::CastContextHint::None;
 }
 
@@ -17458,6 +17522,7 @@ bool BoUpSLP::isFullyVectorizableTinyTree(bool ForReduction) const {
                    [this](Value *V) { return EphValues.contains(V); }) &&
            (allConstant(TE->Scalars) || isSplat(TE->Scalars) ||
             TE->Scalars.size() < Limit ||
+            matchGatheredExtractedFields(TE->Scalars, *DL) ||
             // Nodes with copyable lanes may mix in non-extract lanes, which
             // are not representable as a shuffle of the source vector.
             (((TE->hasState() &&
@@ -22275,6 +22340,26 @@ class BoUpSLP::ShuffleInstructionBuilder final : public BaseShuffleAnalysis {
                       return createShuffle(V1, V2, Mask);
                     });
   }
+  /// Emits the gather of zero-extended sub-fields of width \p FieldWidth of
+  /// the same wider integer scalar \p Src, permuted by \p Mask, as a bitcast
+  /// to the field vector plus a permutation and an extension to the lane type.
+  Value *createExtractedFieldsVector(Value *Src, unsigned FieldWidth,
+                                     ArrayRef<int> Mask, const TreeEntry &) {
+    auto *FieldTy = IntegerType::get(Src->getContext(), FieldWidth);
+    unsigned SrcWidth = Src->getType()->getIntegerBitWidth();
+    assert(SrcWidth % FieldWidth == 0 &&
+           "Expected the field width to divide the source width.");
+    Value *Vec = Builder.CreateBitCast(
+        Src, getWidenedType(FieldTy, SrcWidth / FieldWidth));
+    if (auto *I = dyn_cast<Instruction>(Vec)) {
+      R.GatherShuffleExtractSeq.insert(I);
+      R.CSEBlocks.insert(I->getParent());
+    }
+    Vec = createShuffle(Vec, /*V2=*/nullptr, Mask);
+    // The extracted fields are unsigned.
+    Vec = castToScalarTyElem(Vec, /*IsSigned=*/false);
+    return Vec;
+  }
   Value *createFreeze(Value *V) { return Builder.CreateFreeze(V); }
   /// Finalize emission of the shuffles.
   /// \param Action the action (if any) to be performed before final applying of
@@ -22394,6 +22479,44 @@ Value *BoUpSLP::vectorizeOperand(TreeEntry *E, unsigned NodeIdx) {
   return vectorizeTree(getOperandEntry(E, NodeIdx));
 }
 
+template <typename ResTy, typename BVTy>
+std::optional<ResTy> BoUpSLP::processExtractedFieldsGather(
+    BVTy &ShuffleBuilder, const TreeEntry *E, Type *ScalarTy,
+    ArrayRef<Value *> VL,
+    ArrayRef<std::pair<const TreeEntry *, unsigned>> SubVectors) {
+  if (!SubVectors.empty() || !E->ReuseShuffleIndices.empty() ||
+      !ScalarTy->isIntegerTy())
+    return std::nullopt;
+  std::optional<std::tuple<Value *, unsigned, SmallVector<int>>>
+      ExtractedFields = matchGatheredExtractedFields(VL, *DL);
+  if (!ExtractedFields ||
+      ScalarTy->getIntegerBitWidth() < std::get<1>(*ExtractedFields))
+    return std::nullopt;
+  Value *Src = std::get<0>(*ExtractedFields);
+  unsigned FieldWidth = std::get<1>(*ExtractedFields);
+  SmallVector<int> &Mask = std::get<2>(*ExtractedFields);
+  unsigned NumFields = Src->getType()->getIntegerBitWidth() / FieldWidth;
+  // The lane order of the reduction root is unobservable when every gathered
+  // scalar is used only by the reduction operations and each lane holds a
+  // distinct field: emit the fields in the natural order and skip the
+  // permutation.
+  if (!E->UserTreeIndex && UserIgnoreList && Mask.size() == NumFields &&
+      all_of(make_isa_range<Instruction>(VL), [&](Instruction *I) {
+        return !I->hasNUsesOrMore(UsesLimit) &&
+               all_of(I->users(),
+                      [&](User *U) { return UserIgnoreList->contains(U); });
+      })) {
+    SmallVector<int> SortedMask(Mask);
+    sort(SortedMask);
+    // Each field is held at most once; poison lanes are ignored.
+    if (adjacent_find(SortedMask, [](int A, int B) {
+          return A != PoisonMaskElem && A == B;
+        }) == SortedMask.end())
+      std::iota(Mask.begin(), Mask.end(), 0);
+  }
+  return ShuffleBuilder.createExtractedFieldsVector(Src, FieldWidth, Mask, *E);
+}
+
 template <typename BVTy, typename ResTy, typename... Args>
 ResTy BoUpSLP::processBuildVector(const TreeEntry *E, Type *ScalarTy,
                                   Args &...Params) {
@@ -22505,6 +22628,9 @@ ResTy BoUpSLP::processBuildVector(const TreeEntry *E, Type *ScalarTy,
   Type *OrigScalarTy = GatheredScalars.front()->getType();
   auto *VecTy = getWidenedType(ScalarTy, GatheredScalars.size());
   unsigned NumParts = getNumberOfParts(VecTy, ScalarTy, GatheredScalars.size());
+  if (std::optional<ResTy> Res = processExtractedFieldsGather<ResTy>(
+          ShuffleBuilder, E, ScalarTy, GatheredScalars, SubVectors))
+    return *Res;
   if (!all_of(GatheredScalars, IsaPred<UndefValue>)) {
     // Check for gathered extracts.
     bool Resized = false;
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index cd6319fd6cce5..85ee4a6c0c4ab 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -1139,6 +1139,139 @@ TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F) {
   return F->hasOptSize() ? TTI::TCK_CodeSize : TTI::TCK_RecipThroughput;
 }
 
+/// Checks if \p V is a zero-extended sub-field of a wider integer scalar.
+/// Returns the source scalar, the field width and the field offset.
+static std::optional<std::tuple<Value *, unsigned, unsigned>>
+matchExtractedField(Value *V) {
+  if (!V->getType()->isIntegerTy())
+    return std::nullopt;
+  // Field offset for the field-aligned shift amount, if the shifted value of
+  // the given bit width keeps at least one full field.
+  auto GetFieldOffset = [](const APInt *Amt, unsigned BitWidth,
+                           unsigned FieldWidth) -> std::optional<unsigned> {
+    uint64_t ShAmt = Amt->getLimitedValue(BitWidth);
+    if (ShAmt % FieldWidth != 0 || ShAmt + FieldWidth > BitWidth)
+      return std::nullopt;
+    return ShAmt / FieldWidth;
+  };
+  // Checks if the low bits of Val are a sub-field of the given width of a
+  // wider integer scalar. Val is a scalar integer, since V is one, and so is
+  // the matched source.
+  auto MatchLowField =
+      [&](Value *Val,
+          unsigned FieldWidth) -> std::optional<std::pair<Value *, unsigned>> {
+    Value *Src;
+    const APInt *Amt;
+    // Only the low bits of Val are observed, so lshr and ashr are equivalent.
+    if (match(Val, m_Trunc(m_Shr(m_Value(Src), m_APInt(Amt)))) ||
+        match(Val, m_Shr(m_Value(Src), m_APInt(Amt)))) {
+      if (std::optional<unsigned> Offset = GetFieldOffset(
+              Amt, Src->getType()->getIntegerBitWidth(), FieldWidth))
+        return std::make_pair(Src, *Offset);
+      return std::nullopt;
+    }
+    if (match(Val, m_Shr(m_Trunc(m_Value(Src)), m_APInt(Amt)))) {
+      if (std::optional<unsigned> Offset = GetFieldOffset(
+              Amt, Val->getType()->getIntegerBitWidth(), FieldWidth))
+        return std::make_pair(Src, *Offset);
+      return std::nullopt;
+    }
+    if (match(Val, m_Trunc(m_Value(Src))) &&
+        Src->getType()->getIntegerBitWidth() >= FieldWidth)
+      return std::make_pair(Src, 0u);
+    // Val itself is the source of its low field.
+    if (Val->getType()->getIntegerBitWidth() > FieldWidth)
+      return std::make_pair(Val, 0u);
+    return std::nullopt;
+  };
+  Value *Val;
+  const APInt *Mask;
+  // and Val, (1 << FieldWidth) - 1 or zext i<FieldWidth> Val - the low bits of
+  // Val.
+  unsigned FieldWidth = 0;
+  if (match(V, m_c_And(m_Value(Val), m_APInt(Mask))) && Mask->isMask())
+    FieldWidth = Mask->popcount();
+  else if (match(V, m_ZExt(m_Value(Val))))
+    FieldWidth = Val->getType()->getIntegerBitWidth();
+  if (FieldWidth != 0) {
+    if (std::optional<std::pair<Value *, unsigned>> Field =
+            MatchLowField(Val, FieldWidth))
+      return std::make_tuple(Field->first, FieldWidth, Field->second);
+    return std::nullopt;
+  }
+  unsigned LaneWidth = V->getType()->getIntegerBitWidth();
+  Value *Src;
+  const APInt *Amt;
+  if (match(V, m_Trunc(m_LShr(m_Value(Src), m_APInt(Amt))))) {
+    unsigned SrcWidth = Src->getType()->getIntegerBitWidth();
+    uint64_t ShAmt = Amt->getLimitedValue(SrcWidth);
+    // The field itself, if the lane width is the field width.
+    if (std::optional<unsigned> Offset =
+            GetFieldOffset(Amt, SrcWidth, LaneWidth))
+      return std::make_tuple(Src, LaneWidth, *Offset);
+    // The zero-extended top field of the source.
+    unsigned FieldWidth = SrcWidth - ShAmt;
+    if (FieldWidth > 0 && FieldWidth < LaneWidth && ShAmt % FieldWidth == 0)
+      return std::make_tuple(Src, FieldWidth, ShAmt / FieldWidth);
+    return std::nullopt;
+  }
+  if (match(V, m_LShr(m_Value(Src), m_APInt(Amt)))) {
+    // The zero-extended top field of the source, if the result keeps exactly
+    // one field. Look through a truncation of the shifted value.
+    unsigned ShfWidth = Src->getType()->getIntegerBitWidth();
+    uint64_t ShAmt = Amt->getLimitedValue(ShfWidth);
+    unsigned FieldWidth = ShfWidth - ShAmt;
+    if (FieldWidth > 0 && ShAmt % FieldWidth == 0) {
+      match(Src, m_Trunc(m_Value(Src)));
+      return std::make_tuple(Src, FieldWidth, ShAmt / FieldWidth);
+    }
+    return std::nullopt;
+  }
+  if (match(V, m_Trunc(m_Value(Src))))
+    return std::make_tuple(Src, LaneWidth, 0u);
+  return std::nullopt;
+}
+
+std::optional<std::tuple<Value *, unsigned, SmallVector<int>>>
+matchGatheredExtractedFields(ArrayRef<Value *> VL, const DataLayout &DL) {
+  // Splats are emitted as broadcasts, sub-fields of a constant are folded.
+  // The bitcast to the field vector maps lane 0 to the least significant
+  // field on little-endian targets only.
+  if (VL.size() < 2 || !VL.front()->getType()->isIntegerTy() || isSplat(VL) ||
+      DL.isBigEndian())
+    return std::nullopt;
+  Value *Src = nullptr;
+  unsigned FieldWidth = 0;
+  SmallVector<int> Mask(VL.size(), PoisonMaskElem);
+  for (auto [Idx, V] : enumerate(VL)) {
+    if (isa<UndefValue>(V))
+      continue;
+    if (V->getType() != VL.front()->getType())
+      return std::nullopt;
+    std::optional<std::tuple<Value *, unsigned, unsigned>> Field =
+        matchExtractedField(V);
+    if (!Field || (Src && (Src != std::get<0>(*Field) ||
+                           FieldWidth != std::get<1>(*Field))))
+      return std::nullopt;
+    Src = std::get<0>(*Field);
+    FieldWidth = std::get<1>(*Field);
+    Mask[Idx] = std::get<2>(*Field);
+  }
+  // The field width is a whole number of bytes and divides the source
+  // exactly, same as for the packing layout, so the source bitcasts to the
+  // field vector.
+  if (!Src || isa<Constant>(Src) || FieldWidth % 8 != 0 ||
+      Src->getType()->getIntegerBitWidth() % FieldWidth != 0)
+    return std::nullopt;
+  // The same field in every lane is a splat, emitted as a broadcast.
+  if (all_of(Mask, [First = *find_if(Mask, not_equal_to(PoisonMaskElem))](
+                       int MaskElt) {
+        return MaskElt == PoisonMaskElem || MaskElt == First;
+      }))
+    return std::nullopt;
+  return std::make_tuple(Src, FieldWidth, std::move(Mask));
+}
+
 /// Deeper than the standard analysis recursion depth to keep the numeric
 /// bound precise through arithmetic carry chains.
 constexpr unsigned MaxBitPackAnalysisDepth = MaxAnalysisRecursionDepth + 2;
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index bffe539822341..4ac6880d0f17d 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -30,6 +30,7 @@
 #include <limits>
 #include <optional>
 #include <string>
+#include <tuple>
 
 namespace llvm {
 class AssumptionCache;
@@ -419,6 +420,14 @@ TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F);
 /// loses it.
 APInt getScalarMaxValue(const Value *V, unsigned Depth = 0);
 
+/// Checks if the values in \p VL are zero-extended sub-fields of the same
+/// wider integer scalar. Returns the source scalar, the field width and the
+/// field permutation mask. The extraction dual of the lane-packing layout.
+/// The field-to-lane mapping of the bitcast to the field vector is defined
+/// for little-endian targets only.
+std::optional<std::tuple<Value *, unsigned, SmallVector<int>>>
+matchGatheredExtractedFields(ArrayRef<Value *> VL, const DataLayout &DL);
+
 /// Description of a bitfield packing of vector lanes into a scalar value:
 /// every lane contributes a disjoint contiguous byte field of the result.
 struct BitPackInfo {
diff --git a/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll b/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll
index 8af7ae1b0ac77..ba0974ad8d80c 100644
--- a/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll
+++ b/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll
@@ -5,11 +5,8 @@ define noundef i32 @load_ext_extract(ptr %src) {
 ; CHECK-LABEL: define noundef range(i32 0, 1021) i32 @load_ext_extract(
 ; CHECK-SAME: ptr nofree readonly captures(none) [[SRC:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP14:%.*]] = load i32, ptr [[SRC]], align 4
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> poison, i32 [[TMP14]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT:    [[TMP5:%.*]] = and <4 x i32> [[TMP2]], <i32 255, i32 255, i32 255, i32 -1>
+; CHECK-NEXT:    [[X12:%.*]] = load <4 x i8>, ptr [[SRC]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = zext <4 x i8> [[X12]] to <4 x i32>
 ; CHECK-NEXT:    [[ADD3:%.*]] = tail call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP5]])
 ; CHECK-NEXT:    ret i32 [[ADD3]]
 ;
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
index 8f8899d285ad3..be56447572ab3 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
@@ -1,12 +1,12 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
 ; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64    | FileCheck %s --check-prefixes=CHECK,SSE2
 ; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s --check-prefixes=CHECK,SSE4
-; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefi...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/224919


More information about the llvm-commits mailing list