[llvm] [VectorCombine] Fold compare chains to experimental.vector.match (PR #212456)

via llvm-commits llvm-commits at lists.llvm.org
Tue Jul 28 04:58:40 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Benjamin Maxwell (MacDue)

<details>
<summary>Changes</summary>

This adds an initial combine that can fold trees of compares against splats to the experimental.vector.match intrininc.

For example,

```
%c0 = icmp eq <3 x i16> %vec, splat (i16 17)
%c1 = icmp eq <3 x i16> %vec, splat (i16 -6)
%c2 = icmp eq <3 x i16> %vec, splat (i16 42)
%or0 = or <3 x i1> %c0, %c1
%or1 = or <3 x i1> %or0, %c2
```

Could fold to:
```
call <3 x i1> @<!-- -->llvm.experimental.vector.match.v3i16.v3i16(
  <3 x i16> %vec, <3 x i16> <i16 42, i16 -6, i16 17>,
  <3 x i1> splat (i1 true))
```

This is driven by the target cost model.

For SVE2, this allows some simple loops, such as:

```
void foo(int n, int16_t* src, int16_t* __restrict__ dst)  {
    for (int i = 0; i < n; i++) {
        int c = src[i];
        if (c == ' ' || c == '\n' || c == '\r')
            dst[i] = src[i];
    }
}
```

To be vectorized with the `match` instruction.

Assisted-by: Codex (helped with test cases)

---

Patch is 27.13 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/212456.diff


3 Files Affected:

- (modified) llvm/lib/Transforms/Vectorize/VectorCombine.cpp (+122) 
- (added) llvm/test/Transforms/VectorCombine/AArch64/fold-compares-to-match.ll (+248) 
- (added) llvm/test/Transforms/VectorCombine/fold-compares-to-match.ll (+120) 


``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 726f564b1aad9..ace3965444366 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -159,6 +159,7 @@ class VectorCombine {
   bool foldDeinterleaveIntrinsics(Instruction &I);
   bool foldBitcastOfVPLoad(Instruction &I);
   bool foldBitOrderReverseAndSwap(Instruction &I);
+  bool foldDisjunctionToConstantMatch(Instruction &I);
   bool shrinkType(Instruction &I);
   bool shrinkLoadForShuffles(Instruction &I);
   bool shrinkPhiOfShuffles(Instruction &I);
@@ -578,6 +579,123 @@ bool VectorCombine::isExtractExtractCheap(ExtractElementInst *Ext0,
   return OldCost < NewCost;
 }
 
+/// Attempts to fold chains of `eq` comparisons to `vector.match`. For example:
+///
+/// ```
+/// %c0 = icmp eq <vscale x 4 x i16> %vec, splat (i16 100)
+/// %c1 = icmp eq <vscale x 4 x i16> %vec, splat (i16 32)
+/// %c2 = icmp eq <vscale x 4 x i16> %vec, splat (i16 10)
+/// %or = or <vscale x 4 x i1> %c0, %c1
+/// %r = or <vscale x 4 x i1> %or, %c2
+/// ```
+/// folds to:
+/// ```
+/// %r = call <vscale x 4 x i1> @llvm.experimental.vector.match.nxv8i16.v8i16(
+///        <vscale x 4 x i16> %vec, <i16 100, i16 32, i16 10, i16 100>,
+///        splat (i1 true))
+/// ```
+bool VectorCombine::foldDisjunctionToConstantMatch(Instruction &I) {
+  auto *BinOp = dyn_cast<BinaryOperator>(&I);
+  if (!BinOp || BinOp->getOpcode() != Instruction::Or)
+    return false;
+
+  // The vector being checked (%vec in the example).
+  Value *CompareSource = nullptr;
+  // The values being searched for (e.g, 100, 32, 10).
+  SmallSetVector<Constant *, 16> SearchValues;
+
+  SmallPtrSet<Value *, 4> Visited;
+  SmallVector<Value *> Worklist = {BinOp->getOperand(0), BinOp->getOperand(1)};
+  unsigned NumVisited = 0;
+  while (!Worklist.empty()) {
+    auto *Op = dyn_cast<Instruction>(Worklist.pop_back_val());
+    if (++NumVisited >= MaxInstrsToScan || !Op || !Op->hasOneUse())
+      return false;
+
+    if (!Visited.insert(Op).second)
+      continue;
+
+    Constant *C;
+    Value *Source = nullptr;
+    // TODO: Extend to "not equals" compares.
+    if (match(Op, m_c_SpecificICmp(CmpInst::ICMP_EQ, m_Value(Source),
+                                   m_ConstantSplat(m_Constant(C)))) &&
+        !C->getType()->isPointerTy()) {
+      SearchValues.insert(C);
+    } else if (match(Op, m_Intrinsic<Intrinsic::experimental_vector_match>(
+                             m_Value(Source), m_Constant(C),
+                             m_ConstantSplat(m_SpecificInt(1))))) {
+      // Merge with any previous vector match. This ensures the combine always
+      // creates the largest possible match (even if the combine applies to an
+      // `or` earlier in the chain).
+      unsigned NumElts = cast<FixedVectorType>(C->getType())->getNumElements();
+      for (unsigned I = 0; I < NumElts; ++I)
+        if (Constant *Elt = C->getAggregateElement(I))
+          SearchValues.insert(Elt);
+    } else if (Op->getOpcode() == Instruction::Or) {
+      Worklist.append({Op->getOperand(0), Op->getOperand(1)});
+    } else {
+      // InstCombine can fold compares with nearby values to check ranges, which
+      // means this combine does not match. TODO: Extend this fold to cover
+      // these cases too?
+      return false;
+    }
+
+    // Ensure all comparisons are against the same vector.
+    if (!CompareSource)
+      CompareSource = Source;
+    else if (Source && CompareSource != Source)
+      return false;
+  }
+
+  if (SearchValues.size() <= 1)
+    return false;
+
+  auto *ResultTy = cast<VectorType>(BinOp->getType());
+  auto *SrcType = cast<VectorType>(CompareSource->getType());
+
+  // Cost the old code as-if it were all a chain of compares. It could include
+  // a match, but that will always be replaced with another match (with a
+  // needle size >= to the current match).
+  InstructionCost OldCost =
+      TTI.getCmpSelInstrCost(Instruction::ICmp, SrcType, ResultTy,
+                             CmpInst::ICMP_EQ, CostKind) *
+      SearchValues.size();
+  OldCost += TTI.getArithmeticInstrCost(Instruction::Or, ResultTy, CostKind) *
+             (SearchValues.size() - 1);
+
+  ElementCount SrcElts = SrcType->getElementCount();
+
+  InstructionCost NewCost;
+  Constant *NeedleVector;
+  SmallVector<Constant *> MatchValues(SearchValues.getArrayRef());
+
+  // Look for a match needle size (up to the size of SrcType) that's profitable.
+  do {
+    NeedleVector = ConstantVector::get(MatchValues);
+    IntrinsicCostAttributes ICA(Intrinsic::experimental_vector_match, ResultTy,
+                                {SrcType, NeedleVector->getType(), ResultTy});
+    NewCost = TTI.getIntrinsicInstrCost(ICA, CostKind);
+    if (NewCost < OldCost)
+      break;
+
+    // Pad the needle by duplicating the first element.
+    unsigned PadValues = NextPowerOf2(MatchValues.size()) - MatchValues.size();
+    MatchValues.append(SmallVector<Constant *>(PadValues, MatchValues[0]));
+  } while (MatchValues.size() <= SrcElts.getKnownMinValue());
+
+  if (NewCost >= OldCost)
+    return false;
+
+  Value *Match = Builder.CreateIntrinsic(
+      Intrinsic::experimental_vector_match, {SrcType, NeedleVector->getType()},
+      {CompareSource, NeedleVector,
+       ConstantVector::getSplat(ResultTy->getElementCount(),
+                                Builder.getTrue())});
+  replaceValue(I, *Match);
+  return true;
+}
+
 /// Create a shuffle that translates (shifts) 1 element from the input vector
 /// to a new element location.
 static Value *createShiftShuffle(Value *Vec, unsigned OldIndex,
@@ -6543,6 +6661,10 @@ bool VectorCombine::run() {
       if (foldBitOrderReverseAndSwap(I))
         return true;
 
+    if (IsVectorType && Opcode == Instruction::Or)
+      if (foldDisjunctionToConstantMatch(I))
+        return true;
+
     // Otherwise, try folds that improve codegen but may interfere with
     // early IR canonicalizations.
     // The type checking is for run-time efficiency. We can avoid wasting time
diff --git a/llvm/test/Transforms/VectorCombine/AArch64/fold-compares-to-match.ll b/llvm/test/Transforms/VectorCombine/AArch64/fold-compares-to-match.ll
new file mode 100644
index 0000000000000..7ec56a1e3e7d1
--- /dev/null
+++ b/llvm/test/Transforms/VectorCombine/AArch64/fold-compares-to-match.ll
@@ -0,0 +1,248 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -p vector-combine -mtriple=arm64 -mattr=+sve2 -S %s | FileCheck %s --check-prefixes=CHECK,SVE2
+; RUN: opt -p vector-combine -mtriple=arm64 -mattr=+sve -S %s | FileCheck %s --check-prefixes=CHECK,SVE
+
+define <vscale x 8 x i1> @match_values_nxv8i16_basic(<vscale x 8 x i16> %vec) {
+; SVE2-LABEL: define <vscale x 8 x i1> @match_values_nxv8i16_basic(
+; SVE2-SAME: <vscale x 8 x i16> [[VEC:%.*]]) #[[ATTR0:[0-9]+]] {
+; SVE2-NEXT:    [[OR1:%.*]] = call <vscale x 8 x i1> @llvm.experimental.vector.match.nxv8i16.v8i16(<vscale x 8 x i16> [[VEC]], <8 x i16> <i16 42, i16 -6, i16 17, i16 42, i16 42, i16 42, i16 42, i16 42>, <vscale x 8 x i1> splat (i1 true))
+; SVE2-NEXT:    ret <vscale x 8 x i1> [[OR1]]
+;
+; SVE-LABEL: define <vscale x 8 x i1> @match_values_nxv8i16_basic(
+; SVE-SAME: <vscale x 8 x i16> [[VEC:%.*]]) #[[ATTR0:[0-9]+]] {
+; SVE-NEXT:    [[C0:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 17)
+; SVE-NEXT:    [[C1:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 -6)
+; SVE-NEXT:    [[C2:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 42)
+; SVE-NEXT:    [[OR0:%.*]] = or <vscale x 8 x i1> [[C0]], [[C1]]
+; SVE-NEXT:    [[OR1:%.*]] = or <vscale x 8 x i1> [[OR0]], [[C2]]
+; SVE-NEXT:    ret <vscale x 8 x i1> [[OR1]]
+;
+  %c0 = icmp eq <vscale x 8 x i16> %vec, splat (i16 17)
+  %c1 = icmp eq <vscale x 8 x i16> %vec, splat (i16 -6)
+  %c2 = icmp eq <vscale x 8 x i16> %vec, splat (i16 42)
+  %or0 = or <vscale x 8 x i1> %c0, %c1
+  %or1 = or <vscale x 8 x i1> %or0, %c2
+  ret <vscale x 8 x i1> %or1
+}
+
+define <vscale x 16 x i1> @match_values_nxv16i8_balanced(<vscale x 16 x i8> %vec) {
+; SVE2-LABEL: define <vscale x 16 x i1> @match_values_nxv16i8_balanced(
+; SVE2-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; SVE2-NEXT:    [[OR2:%.*]] = call <vscale x 16 x i1> @llvm.experimental.vector.match.nxv16i8.v8i8(<vscale x 16 x i8> [[VEC]], <8 x i8> <i8 12, i8 -4, i8 29, i8 7, i8 12, i8 12, i8 12, i8 12>, <vscale x 16 x i1> splat (i1 true))
+; SVE2-NEXT:    ret <vscale x 16 x i1> [[OR2]]
+;
+; SVE-LABEL: define <vscale x 16 x i1> @match_values_nxv16i8_balanced(
+; SVE-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; SVE-NEXT:    [[C0:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 7)
+; SVE-NEXT:    [[C1:%.*]] = icmp eq <vscale x 16 x i8> splat (i8 29), [[VEC]]
+; SVE-NEXT:    [[C2:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 -4)
+; SVE-NEXT:    [[C3:%.*]] = icmp eq <vscale x 16 x i8> splat (i8 12), [[VEC]]
+; SVE-NEXT:    [[OR0:%.*]] = or <vscale x 16 x i1> [[C0]], [[C1]]
+; SVE-NEXT:    [[OR1:%.*]] = or <vscale x 16 x i1> [[C2]], [[C3]]
+; SVE-NEXT:    [[OR2:%.*]] = or <vscale x 16 x i1> [[OR0]], [[OR1]]
+; SVE-NEXT:    ret <vscale x 16 x i1> [[OR2]]
+;
+  %c0 = icmp eq <vscale x 16 x i8> %vec, splat (i8 7)
+  %c1 = icmp eq <vscale x 16 x i8> splat (i8 29), %vec
+  %c2 = icmp eq <vscale x 16 x i8> %vec, splat (i8 -4)
+  %c3 = icmp eq <vscale x 16 x i8> splat (i8 12), %vec
+  %or0 = or <vscale x 16 x i1> %c0, %c1
+  %or1 = or <vscale x 16 x i1> %c2, %c3
+  %or2 = or <vscale x 16 x i1> %or0, %or1
+  ret <vscale x 16 x i1> %or2
+}
+
+define <vscale x 16 x i1> @match_values_nxv16i8_two_matches(<vscale x 16 x i8> %vec) {
+; SVE2-LABEL: define <vscale x 16 x i1> @match_values_nxv16i8_two_matches(
+; SVE2-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; SVE2-NEXT:    [[OR16:%.*]] = call <vscale x 16 x i1> @llvm.experimental.vector.match.nxv16i8.v16i8(<vscale x 16 x i8> [[VEC]], <16 x i8> <i8 41, i8 37, i8 31, i8 29, i8 23, i8 17, i8 11, i8 7, i8 3, i8 41, i8 41, i8 41, i8 41, i8 41, i8 41, i8 41>, <vscale x 16 x i1> splat (i1 true))
+; SVE2-NEXT:    [[OR17:%.*]] = call <vscale x 16 x i1> @llvm.experimental.vector.match.nxv16i8.v16i8(<vscale x 16 x i8> [[VEC]], <16 x i8> <i8 79, i8 73, i8 71, i8 67, i8 61, i8 59, i8 53, i8 47, i8 43, i8 79, i8 79, i8 79, i8 79, i8 79, i8 79, i8 79>, <vscale x 16 x i1> splat (i1 true))
+; SVE2-NEXT:    [[OR18:%.*]] = or <vscale x 16 x i1> [[OR16]], [[OR17]]
+; SVE2-NEXT:    ret <vscale x 16 x i1> [[OR18]]
+;
+; SVE-LABEL: define <vscale x 16 x i1> @match_values_nxv16i8_two_matches(
+; SVE-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; SVE-NEXT:    [[C0:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 3)
+; SVE-NEXT:    [[C1:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 7)
+; SVE-NEXT:    [[C2:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 11)
+; SVE-NEXT:    [[C3:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 17)
+; SVE-NEXT:    [[C4:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 23)
+; SVE-NEXT:    [[C5:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 29)
+; SVE-NEXT:    [[C6:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 31)
+; SVE-NEXT:    [[C7:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 37)
+; SVE-NEXT:    [[C8:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 41)
+; SVE-NEXT:    [[C9:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 43)
+; SVE-NEXT:    [[C10:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 47)
+; SVE-NEXT:    [[C11:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 53)
+; SVE-NEXT:    [[C12:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 59)
+; SVE-NEXT:    [[C13:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 61)
+; SVE-NEXT:    [[C14:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 67)
+; SVE-NEXT:    [[C15:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 71)
+; SVE-NEXT:    [[C16:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 73)
+; SVE-NEXT:    [[C17:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 79)
+; SVE-NEXT:    [[OR0:%.*]] = or <vscale x 16 x i1> [[C0]], [[C1]]
+; SVE-NEXT:    [[OR1:%.*]] = or <vscale x 16 x i1> [[OR0]], [[C2]]
+; SVE-NEXT:    [[OR2:%.*]] = or <vscale x 16 x i1> [[OR1]], [[C3]]
+; SVE-NEXT:    [[OR3:%.*]] = or <vscale x 16 x i1> [[OR2]], [[C4]]
+; SVE-NEXT:    [[OR4:%.*]] = or <vscale x 16 x i1> [[OR3]], [[C5]]
+; SVE-NEXT:    [[OR5:%.*]] = or <vscale x 16 x i1> [[OR4]], [[C6]]
+; SVE-NEXT:    [[OR6:%.*]] = or <vscale x 16 x i1> [[OR5]], [[C7]]
+; SVE-NEXT:    [[OR7:%.*]] = or <vscale x 16 x i1> [[OR6]], [[C8]]
+; SVE-NEXT:    [[OR8:%.*]] = or <vscale x 16 x i1> [[C9]], [[C10]]
+; SVE-NEXT:    [[OR9:%.*]] = or <vscale x 16 x i1> [[OR8]], [[C11]]
+; SVE-NEXT:    [[OR10:%.*]] = or <vscale x 16 x i1> [[OR9]], [[C12]]
+; SVE-NEXT:    [[OR11:%.*]] = or <vscale x 16 x i1> [[OR10]], [[C13]]
+; SVE-NEXT:    [[OR12:%.*]] = or <vscale x 16 x i1> [[OR11]], [[C14]]
+; SVE-NEXT:    [[OR13:%.*]] = or <vscale x 16 x i1> [[OR12]], [[C15]]
+; SVE-NEXT:    [[OR14:%.*]] = or <vscale x 16 x i1> [[OR13]], [[C16]]
+; SVE-NEXT:    [[OR15:%.*]] = or <vscale x 16 x i1> [[OR14]], [[C17]]
+; SVE-NEXT:    [[OR16:%.*]] = or <vscale x 16 x i1> [[OR7]], [[OR15]]
+; SVE-NEXT:    ret <vscale x 16 x i1> [[OR16]]
+;
+  %c0 = icmp eq <vscale x 16 x i8> %vec, splat (i8 3)
+  %c1 = icmp eq <vscale x 16 x i8> %vec, splat (i8 7)
+  %c2 = icmp eq <vscale x 16 x i8> %vec, splat (i8 11)
+  %c3 = icmp eq <vscale x 16 x i8> %vec, splat (i8 17)
+  %c4 = icmp eq <vscale x 16 x i8> %vec, splat (i8 23)
+  %c5 = icmp eq <vscale x 16 x i8> %vec, splat (i8 29)
+  %c6 = icmp eq <vscale x 16 x i8> %vec, splat (i8 31)
+  %c7 = icmp eq <vscale x 16 x i8> %vec, splat (i8 37)
+  %c8 = icmp eq <vscale x 16 x i8> %vec, splat (i8 41)
+  %c9 = icmp eq <vscale x 16 x i8> %vec, splat (i8 43)
+  %c10 = icmp eq <vscale x 16 x i8> %vec, splat (i8 47)
+  %c11 = icmp eq <vscale x 16 x i8> %vec, splat (i8 53)
+  %c12 = icmp eq <vscale x 16 x i8> %vec, splat (i8 59)
+  %c13 = icmp eq <vscale x 16 x i8> %vec, splat (i8 61)
+  %c14 = icmp eq <vscale x 16 x i8> %vec, splat (i8 67)
+  %c15 = icmp eq <vscale x 16 x i8> %vec, splat (i8 71)
+  %c16 = icmp eq <vscale x 16 x i8> %vec, splat (i8 73)
+  %c17 = icmp eq <vscale x 16 x i8> %vec, splat (i8 79)
+  %or0 = or <vscale x 16 x i1> %c0, %c1
+  %or1 = or <vscale x 16 x i1> %or0, %c2
+  %or2 = or <vscale x 16 x i1> %or1, %c3
+  %or3 = or <vscale x 16 x i1> %or2, %c4
+  %or4 = or <vscale x 16 x i1> %or3, %c5
+  %or5 = or <vscale x 16 x i1> %or4, %c6
+  %or6 = or <vscale x 16 x i1> %or5, %c7
+  %or7 = or <vscale x 16 x i1> %or6, %c8
+  %or8 = or <vscale x 16 x i1> %c9, %c10
+  %or9 = or <vscale x 16 x i1> %or8, %c11
+  %or10 = or <vscale x 16 x i1> %or9, %c12
+  %or11 = or <vscale x 16 x i1> %or10, %c13
+  %or12 = or <vscale x 16 x i1> %or11, %c14
+  %or13 = or <vscale x 16 x i1> %or12, %c15
+  %or14 = or <vscale x 16 x i1> %or13, %c16
+  %or15 = or <vscale x 16 x i1> %or14, %c17
+  %or16 = or <vscale x 16 x i1> %or7, %or15
+  ret <vscale x 16 x i1> %or16
+}
+
+define <vscale x 8 x i1> @match_values_nxv8i16_two_matches(<vscale x 8 x i16> %vec) {
+; SVE2-LABEL: define <vscale x 8 x i1> @match_values_nxv8i16_two_matches(
+; SVE2-SAME: <vscale x 8 x i16> [[VEC:%.*]]) #[[ATTR0]] {
+; SVE2-NEXT:    [[OR4:%.*]] = call <vscale x 8 x i1> @llvm.experimental.vector.match.nxv8i16.v8i16(<vscale x 8 x i16> [[VEC]], <8 x i16> <i16 44, i16 31, i16 23, i16 19, i16 12, i16 5, i16 44, i16 44>, <vscale x 8 x i1> splat (i1 true))
+; SVE2-NEXT:    [[OR9:%.*]] = call <vscale x 8 x i1> @llvm.experimental.vector.match.nxv8i16.v8i16(<vscale x 8 x i16> [[VEC]], <8 x i16> <i16 -29, i16 48, i16 36, i16 27, i16 16, i16 -7, i16 -29, i16 -29>, <vscale x 8 x i1> splat (i1 true))
+; SVE2-NEXT:    [[OR10:%.*]] = or <vscale x 8 x i1> [[OR4]], [[OR9]]
+; SVE2-NEXT:    ret <vscale x 8 x i1> [[OR10]]
+;
+; SVE-LABEL: define <vscale x 8 x i1> @match_values_nxv8i16_two_matches(
+; SVE-SAME: <vscale x 8 x i16> [[VEC:%.*]]) #[[ATTR0]] {
+; SVE-NEXT:    [[C0:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 5)
+; SVE-NEXT:    [[C1:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 12)
+; SVE-NEXT:    [[C2:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 19)
+; SVE-NEXT:    [[C3:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 23)
+; SVE-NEXT:    [[C4:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 31)
+; SVE-NEXT:    [[C5:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 44)
+; SVE-NEXT:    [[C6:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 -7)
+; SVE-NEXT:    [[C7:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 16)
+; SVE-NEXT:    [[C8:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 27)
+; SVE-NEXT:    [[C9:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 36)
+; SVE-NEXT:    [[C10:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 48)
+; SVE-NEXT:    [[C11:%.*]] = icmp eq <vscale x 8 x i16> [[VEC]], splat (i16 -29)
+; SVE-NEXT:    [[OR0:%.*]] = or <vscale x 8 x i1> [[C0]], [[C1]]
+; SVE-NEXT:    [[OR1:%.*]] = or <vscale x 8 x i1> [[OR0]], [[C2]]
+; SVE-NEXT:    [[OR2:%.*]] = or <vscale x 8 x i1> [[OR1]], [[C3]]
+; SVE-NEXT:    [[OR3:%.*]] = or <vscale x 8 x i1> [[OR2]], [[C4]]
+; SVE-NEXT:    [[OR4:%.*]] = or <vscale x 8 x i1> [[OR3]], [[C5]]
+; SVE-NEXT:    [[OR5:%.*]] = or <vscale x 8 x i1> [[C6]], [[C7]]
+; SVE-NEXT:    [[OR6:%.*]] = or <vscale x 8 x i1> [[OR5]], [[C8]]
+; SVE-NEXT:    [[OR7:%.*]] = or <vscale x 8 x i1> [[OR6]], [[C9]]
+; SVE-NEXT:    [[OR8:%.*]] = or <vscale x 8 x i1> [[OR7]], [[C10]]
+; SVE-NEXT:    [[OR9:%.*]] = or <vscale x 8 x i1> [[OR8]], [[C11]]
+; SVE-NEXT:    [[OR10:%.*]] = or <vscale x 8 x i1> [[OR4]], [[OR9]]
+; SVE-NEXT:    ret <vscale x 8 x i1> [[OR10]]
+;
+  %c0 = icmp eq <vscale x 8 x i16> %vec, splat (i16 5)
+  %c1 = icmp eq <vscale x 8 x i16> %vec, splat (i16 12)
+  %c2 = icmp eq <vscale x 8 x i16> %vec, splat (i16 19)
+  %c3 = icmp eq <vscale x 8 x i16> %vec, splat (i16 23)
+  %c4 = icmp eq <vscale x 8 x i16> %vec, splat (i16 31)
+  %c5 = icmp eq <vscale x 8 x i16> %vec, splat (i16 44)
+  %c6 = icmp eq <vscale x 8 x i16> %vec, splat (i16 -7)
+  %c7 = icmp eq <vscale x 8 x i16> %vec, splat (i16 16)
+  %c8 = icmp eq <vscale x 8 x i16> %vec, splat (i16 27)
+  %c9 = icmp eq <vscale x 8 x i16> %vec, splat (i16 36)
+  %c10 = icmp eq <vscale x 8 x i16> %vec, splat (i16 48)
+  %c11 = icmp eq <vscale x 8 x i16> %vec, splat (i16 -29)
+  %or0 = or <vscale x 8 x i1> %c0, %c1
+  %or1 = or <vscale x 8 x i1> %or0, %c2
+  %or2 = or <vscale x 8 x i1> %or1, %c3
+  %or3 = or <vscale x 8 x i1> %or2, %c4
+  %or4 = or <vscale x 8 x i1> %or3, %c5
+  %or5 = or <vscale x 8 x i1> %c6, %c7
+  %or6 = or <vscale x 8 x i1> %or5, %c8
+  %or7 = or <vscale x 8 x i1> %or6, %c9
+  %or8 = or <vscale x 8 x i1> %or7, %c10
+  %or9 = or <vscale x 8 x i1> %or8, %c11
+  %or10 = or <vscale x 8 x i1> %or4, %or9
+  ret <vscale x 8 x i1> %or10
+}
+
+; Unprofitable test: nxv16i8 matches the shape, but a 2-value vector.match should remain too expensive.
+define <vscale x 16 x i1> @match_values_nxv16i8_unprofitable_two_values(<vscale x 16 x i8> %vec) {
+; CHECK-LABEL: define <vscale x 16 x i1> @match_values_nxv16i8_unprofitable_two_values(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    [[C0:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 9)
+; CHECK-NEXT:    [[C1:%.*]] = icmp eq <vscale x 16 x i8> [[VEC]], splat (i8 -7)
+; CHECK-NEXT:    [[OR0:%.*]] = or <vscale x 16 x i1> [[C0]], [[C1]]
+; CHECK-NEXT:    ret <vscale x 16 x i1> [[OR0]]
+;
+  %c0 = icmp eq <vscale x 16 x i8> %vec, splat (i8 9)
+  %c1 = icmp eq <vscale x 16 x i8> %vec, splat (i8 -7)
+  %or0 = or <vscale x 16 x i1> %c0, %c1
+  ret <vscale x 16 x i1> %or0
+}
+
+; Unprofitable test: nxv4i32 matches the shape, but vector.match should remain too expensive.
+define <vscale x 4 x i1> @match_values_nxv4i32_unprofitable(<vscale x 4 x i32> %vec) {
+; CHECK-LABEL: define <vscale x 4 x i1> @match_values_nxv4i32_unprofitable(
+; CHECK-SAME: <vscale x 4 x i32> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[C0:%.*]] = icmp eq <vscale x 4 x i32> [[VEC]], splat (i32 17)
+; CHECK-NEXT:    [[C1:%.*]] = icmp eq <vscale x 4 x i32> [[VEC]], splat (i32 -8)
+; CHECK-NEXT:    [[C2:%.*]] = icmp eq <vscale x 4 x i32> [[VEC]], splat (i32 33)
+; CHECK-NEXT:    [[OR0:%.*]] ...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/212456


More information about the llvm-commits mailing list