[llvm] [CostModel][X86] reduce_add(vXi1) will lower as a scalar ctpop (PR #178400)

via llvm-commits llvm-commits at lists.llvm.org
Wed Jan 28 03:12:43 PST 2026


llvmbot wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Simon Pilgrim (RKSimon)

<details>
<summary>Changes</summary>

Fixes #<!-- -->176906

---
Full diff: https://github.com/llvm/llvm-project/pull/178400.diff


2 Files Affected:

- (modified) llvm/lib/Target/X86/X86TargetTransformInfo.cpp (+11) 
- (modified) llvm/test/Transforms/SLPVectorizer/X86/pr176906.ll (+13-118) 


``````````diff
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index f49b33073dc42..e4ef221e53b12 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -5700,6 +5700,17 @@ X86TTIImpl::getArithmeticReductionCost(unsigned Opcode, VectorType *ValTy,
 
   // Handle bool allof/anyof patterns.
   if (ValVTy->getElementType()->isIntegerTy(1)) {
+    if (ISD == ISD::ADD) {
+      // vXi1 addition reduction will bitcast to scalar and perform a popcount.
+      auto *IntTy = IntegerType::getIntNTy(ValVTy->getContext(),
+                                           ValVTy->getNumElements());
+      IntrinsicCostAttributes ICA(Intrinsic::ctpop, IntTy, {IntTy});
+      return getCastInstrCost(Instruction::BitCast, IntTy, ValVTy,
+                              TargetTransformInfo::CastContextHint::None,
+                              CostKind) +
+             getIntrinsicInstrCost(ICA, CostKind);
+    }
+
     InstructionCost ArithmeticCost = 0;
     if (LT.first != 1 && MTy.isVector() &&
         MTy.getVectorNumElements() < ValVTy->getNumElements()) {
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/pr176906.ll b/llvm/test/Transforms/SLPVectorizer/X86/pr176906.ll
index 6f0f6296f45a8..8ecfe1149aed2 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/pr176906.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/pr176906.ll
@@ -1,8 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64    | FileCheck %s --check-prefixes=CHECK,SSE
-; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s --check-prefixes=CHECK,SSE
-; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
-; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=CHECK,AVX,AVX512
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64    | FileCheck %s --check-prefixes=SSE
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s --check-prefixes=SSE
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=AVX
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=AVX
 
 define i1 @PR176906(ptr %p) {
 ; SSE-LABEL: define i1 @PR176906(
@@ -14,117 +14,15 @@ define i1 @PR176906(ptr %p) {
 ; SSE-NEXT:    [[OK:%.*]] = icmp eq i8 [[TMP3]], 32
 ; SSE-NEXT:    ret i1 [[OK]]
 ;
-; AVX2-LABEL: define i1 @PR176906(
-; AVX2-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
-; AVX2-NEXT:    [[V:%.*]] = load <32 x i8>, ptr [[P]], align 1
-; AVX2-NEXT:    [[TMP1:%.*]] = icmp sgt <32 x i8> [[V]], splat (i8 -1)
-; AVX2-NEXT:    [[TMP2:%.*]] = bitcast <32 x i1> [[TMP1]] to i32
-; AVX2-NEXT:    [[TMP3:%.*]] = call i32 @llvm.ctpop.i32(i32 [[TMP2]])
-; AVX2-NEXT:    [[TMP4:%.*]] = trunc i32 [[TMP3]] to i8
-; AVX2-NEXT:    [[OK:%.*]] = icmp eq i8 [[TMP4]], 32
-; AVX2-NEXT:    ret i1 [[OK]]
-;
-; AVX512-LABEL: define i1 @PR176906(
-; AVX512-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
-; AVX512-NEXT:    [[V:%.*]] = load <32 x i8>, ptr [[P]], align 1
-; AVX512-NEXT:    [[TMP1:%.*]] = icmp sgt <32 x i8> [[V]], splat (i8 -1)
-; AVX512-NEXT:    [[TMP2:%.*]] = extractelement <32 x i1> [[TMP1]], i32 0
-; AVX512-NEXT:    [[Z:%.*]] = zext i1 [[TMP2]] to i8
-; AVX512-NEXT:    [[TMP3:%.*]] = extractelement <32 x i1> [[TMP1]], i32 1
-; AVX512-NEXT:    [[Z_1:%.*]] = zext i1 [[TMP3]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_1:%.*]] = add nuw nsw i8 [[Z]], [[Z_1]]
-; AVX512-NEXT:    [[TMP4:%.*]] = extractelement <32 x i1> [[TMP1]], i32 2
-; AVX512-NEXT:    [[Z_2:%.*]] = zext i1 [[TMP4]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_2:%.*]] = add nuw nsw i8 [[ACC_NEXT_1]], [[Z_2]]
-; AVX512-NEXT:    [[TMP5:%.*]] = extractelement <32 x i1> [[TMP1]], i32 3
-; AVX512-NEXT:    [[Z_3:%.*]] = zext i1 [[TMP5]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_3:%.*]] = add nuw nsw i8 [[ACC_NEXT_2]], [[Z_3]]
-; AVX512-NEXT:    [[TMP6:%.*]] = extractelement <32 x i1> [[TMP1]], i32 4
-; AVX512-NEXT:    [[Z_4:%.*]] = zext i1 [[TMP6]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_4:%.*]] = add nuw nsw i8 [[ACC_NEXT_3]], [[Z_4]]
-; AVX512-NEXT:    [[TMP7:%.*]] = extractelement <32 x i1> [[TMP1]], i32 5
-; AVX512-NEXT:    [[Z_5:%.*]] = zext i1 [[TMP7]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_5:%.*]] = add nuw nsw i8 [[ACC_NEXT_4]], [[Z_5]]
-; AVX512-NEXT:    [[TMP8:%.*]] = extractelement <32 x i1> [[TMP1]], i32 6
-; AVX512-NEXT:    [[Z_6:%.*]] = zext i1 [[TMP8]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_6:%.*]] = add nuw nsw i8 [[ACC_NEXT_5]], [[Z_6]]
-; AVX512-NEXT:    [[TMP9:%.*]] = extractelement <32 x i1> [[TMP1]], i32 7
-; AVX512-NEXT:    [[Z_7:%.*]] = zext i1 [[TMP9]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_7:%.*]] = add nuw nsw i8 [[ACC_NEXT_6]], [[Z_7]]
-; AVX512-NEXT:    [[TMP10:%.*]] = extractelement <32 x i1> [[TMP1]], i32 8
-; AVX512-NEXT:    [[Z_8:%.*]] = zext i1 [[TMP10]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_8:%.*]] = add nuw nsw i8 [[ACC_NEXT_7]], [[Z_8]]
-; AVX512-NEXT:    [[TMP11:%.*]] = extractelement <32 x i1> [[TMP1]], i32 9
-; AVX512-NEXT:    [[Z_9:%.*]] = zext i1 [[TMP11]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_9:%.*]] = add nuw nsw i8 [[ACC_NEXT_8]], [[Z_9]]
-; AVX512-NEXT:    [[TMP12:%.*]] = extractelement <32 x i1> [[TMP1]], i32 10
-; AVX512-NEXT:    [[Z_10:%.*]] = zext i1 [[TMP12]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_10:%.*]] = add nuw nsw i8 [[ACC_NEXT_9]], [[Z_10]]
-; AVX512-NEXT:    [[TMP13:%.*]] = extractelement <32 x i1> [[TMP1]], i32 11
-; AVX512-NEXT:    [[Z_11:%.*]] = zext i1 [[TMP13]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_11:%.*]] = add nuw nsw i8 [[ACC_NEXT_10]], [[Z_11]]
-; AVX512-NEXT:    [[TMP14:%.*]] = extractelement <32 x i1> [[TMP1]], i32 12
-; AVX512-NEXT:    [[Z_12:%.*]] = zext i1 [[TMP14]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_12:%.*]] = add nuw nsw i8 [[ACC_NEXT_11]], [[Z_12]]
-; AVX512-NEXT:    [[TMP15:%.*]] = extractelement <32 x i1> [[TMP1]], i32 13
-; AVX512-NEXT:    [[Z_13:%.*]] = zext i1 [[TMP15]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_13:%.*]] = add nuw nsw i8 [[ACC_NEXT_12]], [[Z_13]]
-; AVX512-NEXT:    [[TMP16:%.*]] = extractelement <32 x i1> [[TMP1]], i32 14
-; AVX512-NEXT:    [[Z_14:%.*]] = zext i1 [[TMP16]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_14:%.*]] = add nuw nsw i8 [[ACC_NEXT_13]], [[Z_14]]
-; AVX512-NEXT:    [[TMP17:%.*]] = extractelement <32 x i1> [[TMP1]], i32 15
-; AVX512-NEXT:    [[Z_15:%.*]] = zext i1 [[TMP17]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_15:%.*]] = add nuw nsw i8 [[ACC_NEXT_14]], [[Z_15]]
-; AVX512-NEXT:    [[TMP18:%.*]] = extractelement <32 x i1> [[TMP1]], i32 16
-; AVX512-NEXT:    [[Z_16:%.*]] = zext i1 [[TMP18]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_16:%.*]] = add nuw nsw i8 [[ACC_NEXT_15]], [[Z_16]]
-; AVX512-NEXT:    [[TMP19:%.*]] = extractelement <32 x i1> [[TMP1]], i32 17
-; AVX512-NEXT:    [[Z_17:%.*]] = zext i1 [[TMP19]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_17:%.*]] = add nuw nsw i8 [[ACC_NEXT_16]], [[Z_17]]
-; AVX512-NEXT:    [[TMP20:%.*]] = extractelement <32 x i1> [[TMP1]], i32 18
-; AVX512-NEXT:    [[Z_18:%.*]] = zext i1 [[TMP20]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_18:%.*]] = add nuw nsw i8 [[ACC_NEXT_17]], [[Z_18]]
-; AVX512-NEXT:    [[TMP21:%.*]] = extractelement <32 x i1> [[TMP1]], i32 19
-; AVX512-NEXT:    [[Z_19:%.*]] = zext i1 [[TMP21]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_19:%.*]] = add nuw nsw i8 [[ACC_NEXT_18]], [[Z_19]]
-; AVX512-NEXT:    [[TMP22:%.*]] = extractelement <32 x i1> [[TMP1]], i32 20
-; AVX512-NEXT:    [[Z_20:%.*]] = zext i1 [[TMP22]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_20:%.*]] = add nuw nsw i8 [[ACC_NEXT_19]], [[Z_20]]
-; AVX512-NEXT:    [[TMP23:%.*]] = extractelement <32 x i1> [[TMP1]], i32 21
-; AVX512-NEXT:    [[Z_21:%.*]] = zext i1 [[TMP23]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_21:%.*]] = add nuw nsw i8 [[ACC_NEXT_20]], [[Z_21]]
-; AVX512-NEXT:    [[TMP24:%.*]] = extractelement <32 x i1> [[TMP1]], i32 22
-; AVX512-NEXT:    [[Z_22:%.*]] = zext i1 [[TMP24]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_22:%.*]] = add nuw nsw i8 [[ACC_NEXT_21]], [[Z_22]]
-; AVX512-NEXT:    [[TMP25:%.*]] = extractelement <32 x i1> [[TMP1]], i32 23
-; AVX512-NEXT:    [[Z_23:%.*]] = zext i1 [[TMP25]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_23:%.*]] = add nuw nsw i8 [[ACC_NEXT_22]], [[Z_23]]
-; AVX512-NEXT:    [[TMP26:%.*]] = extractelement <32 x i1> [[TMP1]], i32 24
-; AVX512-NEXT:    [[Z_24:%.*]] = zext i1 [[TMP26]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_24:%.*]] = add nuw nsw i8 [[ACC_NEXT_23]], [[Z_24]]
-; AVX512-NEXT:    [[TMP27:%.*]] = extractelement <32 x i1> [[TMP1]], i32 25
-; AVX512-NEXT:    [[Z_25:%.*]] = zext i1 [[TMP27]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_25:%.*]] = add nuw nsw i8 [[ACC_NEXT_24]], [[Z_25]]
-; AVX512-NEXT:    [[TMP28:%.*]] = extractelement <32 x i1> [[TMP1]], i32 26
-; AVX512-NEXT:    [[Z_26:%.*]] = zext i1 [[TMP28]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_26:%.*]] = add nuw nsw i8 [[ACC_NEXT_25]], [[Z_26]]
-; AVX512-NEXT:    [[TMP29:%.*]] = extractelement <32 x i1> [[TMP1]], i32 27
-; AVX512-NEXT:    [[Z_27:%.*]] = zext i1 [[TMP29]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_27:%.*]] = add nuw nsw i8 [[ACC_NEXT_26]], [[Z_27]]
-; AVX512-NEXT:    [[TMP30:%.*]] = extractelement <32 x i1> [[TMP1]], i32 28
-; AVX512-NEXT:    [[Z_28:%.*]] = zext i1 [[TMP30]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_28:%.*]] = add nuw nsw i8 [[ACC_NEXT_27]], [[Z_28]]
-; AVX512-NEXT:    [[TMP31:%.*]] = extractelement <32 x i1> [[TMP1]], i32 29
-; AVX512-NEXT:    [[Z_29:%.*]] = zext i1 [[TMP31]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_29:%.*]] = add nuw nsw i8 [[ACC_NEXT_28]], [[Z_29]]
-; AVX512-NEXT:    [[TMP32:%.*]] = extractelement <32 x i1> [[TMP1]], i32 30
-; AVX512-NEXT:    [[Z_30:%.*]] = zext i1 [[TMP32]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_30:%.*]] = add nuw nsw i8 [[ACC_NEXT_29]], [[Z_30]]
-; AVX512-NEXT:    [[TMP33:%.*]] = extractelement <32 x i1> [[TMP1]], i32 31
-; AVX512-NEXT:    [[Z_31:%.*]] = zext i1 [[TMP33]] to i8
-; AVX512-NEXT:    [[ACC_NEXT_31:%.*]] = add nuw nsw i8 [[ACC_NEXT_30]], [[Z_31]]
-; AVX512-NEXT:    [[OK:%.*]] = icmp eq i8 [[ACC_NEXT_31]], 32
-; AVX512-NEXT:    ret i1 [[OK]]
+; AVX-LABEL: define i1 @PR176906(
+; AVX-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
+; AVX-NEXT:    [[V:%.*]] = load <32 x i8>, ptr [[P]], align 1
+; AVX-NEXT:    [[TMP1:%.*]] = icmp sgt <32 x i8> [[V]], splat (i8 -1)
+; AVX-NEXT:    [[TMP2:%.*]] = bitcast <32 x i1> [[TMP1]] to i32
+; AVX-NEXT:    [[TMP3:%.*]] = call i32 @llvm.ctpop.i32(i32 [[TMP2]])
+; AVX-NEXT:    [[TMP4:%.*]] = trunc i32 [[TMP3]] to i8
+; AVX-NEXT:    [[OK:%.*]] = icmp eq i8 [[TMP4]], 32
+; AVX-NEXT:    ret i1 [[OK]]
 ;
   %v = load <32 x i8>, ptr %p, align 1
   %i = extractelement <32 x i8> %v, i64 0
@@ -257,6 +155,3 @@ define i1 @PR176906(ptr %p) {
   %ok = icmp eq i8 %acc.next.31, 32
   ret i1 %ok
 }
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; AVX: {{.*}}
-; CHECK: {{.*}}

``````````

</details>


https://github.com/llvm/llvm-project/pull/178400


More information about the llvm-commits mailing list