[llvm] [CostModel][X86] reduce_add(vXi1) will lower as a scalar ctpop (PR #178400)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Jan 28 03:12:43 PST 2026
llvmbot wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Simon Pilgrim (RKSimon)
<details>
<summary>Changes</summary>
Fixes #<!-- -->176906
---
Full diff: https://github.com/llvm/llvm-project/pull/178400.diff
2 Files Affected:
- (modified) llvm/lib/Target/X86/X86TargetTransformInfo.cpp (+11)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/pr176906.ll (+13-118)
``````````diff
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index f49b33073dc42..e4ef221e53b12 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -5700,6 +5700,17 @@ X86TTIImpl::getArithmeticReductionCost(unsigned Opcode, VectorType *ValTy,
// Handle bool allof/anyof patterns.
if (ValVTy->getElementType()->isIntegerTy(1)) {
+ if (ISD == ISD::ADD) {
+ // vXi1 addition reduction will bitcast to scalar and perform a popcount.
+ auto *IntTy = IntegerType::getIntNTy(ValVTy->getContext(),
+ ValVTy->getNumElements());
+ IntrinsicCostAttributes ICA(Intrinsic::ctpop, IntTy, {IntTy});
+ return getCastInstrCost(Instruction::BitCast, IntTy, ValVTy,
+ TargetTransformInfo::CastContextHint::None,
+ CostKind) +
+ getIntrinsicInstrCost(ICA, CostKind);
+ }
+
InstructionCost ArithmeticCost = 0;
if (LT.first != 1 && MTy.isVector() &&
MTy.getVectorNumElements() < ValVTy->getNumElements()) {
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/pr176906.ll b/llvm/test/Transforms/SLPVectorizer/X86/pr176906.ll
index 6f0f6296f45a8..8ecfe1149aed2 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/pr176906.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/pr176906.ll
@@ -1,8 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64 | FileCheck %s --check-prefixes=CHECK,SSE
-; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s --check-prefixes=CHECK,SSE
-; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
-; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=CHECK,AVX,AVX512
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64 | FileCheck %s --check-prefixes=SSE
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s --check-prefixes=SSE
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=AVX
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=AVX
define i1 @PR176906(ptr %p) {
; SSE-LABEL: define i1 @PR176906(
@@ -14,117 +14,15 @@ define i1 @PR176906(ptr %p) {
; SSE-NEXT: [[OK:%.*]] = icmp eq i8 [[TMP3]], 32
; SSE-NEXT: ret i1 [[OK]]
;
-; AVX2-LABEL: define i1 @PR176906(
-; AVX2-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
-; AVX2-NEXT: [[V:%.*]] = load <32 x i8>, ptr [[P]], align 1
-; AVX2-NEXT: [[TMP1:%.*]] = icmp sgt <32 x i8> [[V]], splat (i8 -1)
-; AVX2-NEXT: [[TMP2:%.*]] = bitcast <32 x i1> [[TMP1]] to i32
-; AVX2-NEXT: [[TMP3:%.*]] = call i32 @llvm.ctpop.i32(i32 [[TMP2]])
-; AVX2-NEXT: [[TMP4:%.*]] = trunc i32 [[TMP3]] to i8
-; AVX2-NEXT: [[OK:%.*]] = icmp eq i8 [[TMP4]], 32
-; AVX2-NEXT: ret i1 [[OK]]
-;
-; AVX512-LABEL: define i1 @PR176906(
-; AVX512-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
-; AVX512-NEXT: [[V:%.*]] = load <32 x i8>, ptr [[P]], align 1
-; AVX512-NEXT: [[TMP1:%.*]] = icmp sgt <32 x i8> [[V]], splat (i8 -1)
-; AVX512-NEXT: [[TMP2:%.*]] = extractelement <32 x i1> [[TMP1]], i32 0
-; AVX512-NEXT: [[Z:%.*]] = zext i1 [[TMP2]] to i8
-; AVX512-NEXT: [[TMP3:%.*]] = extractelement <32 x i1> [[TMP1]], i32 1
-; AVX512-NEXT: [[Z_1:%.*]] = zext i1 [[TMP3]] to i8
-; AVX512-NEXT: [[ACC_NEXT_1:%.*]] = add nuw nsw i8 [[Z]], [[Z_1]]
-; AVX512-NEXT: [[TMP4:%.*]] = extractelement <32 x i1> [[TMP1]], i32 2
-; AVX512-NEXT: [[Z_2:%.*]] = zext i1 [[TMP4]] to i8
-; AVX512-NEXT: [[ACC_NEXT_2:%.*]] = add nuw nsw i8 [[ACC_NEXT_1]], [[Z_2]]
-; AVX512-NEXT: [[TMP5:%.*]] = extractelement <32 x i1> [[TMP1]], i32 3
-; AVX512-NEXT: [[Z_3:%.*]] = zext i1 [[TMP5]] to i8
-; AVX512-NEXT: [[ACC_NEXT_3:%.*]] = add nuw nsw i8 [[ACC_NEXT_2]], [[Z_3]]
-; AVX512-NEXT: [[TMP6:%.*]] = extractelement <32 x i1> [[TMP1]], i32 4
-; AVX512-NEXT: [[Z_4:%.*]] = zext i1 [[TMP6]] to i8
-; AVX512-NEXT: [[ACC_NEXT_4:%.*]] = add nuw nsw i8 [[ACC_NEXT_3]], [[Z_4]]
-; AVX512-NEXT: [[TMP7:%.*]] = extractelement <32 x i1> [[TMP1]], i32 5
-; AVX512-NEXT: [[Z_5:%.*]] = zext i1 [[TMP7]] to i8
-; AVX512-NEXT: [[ACC_NEXT_5:%.*]] = add nuw nsw i8 [[ACC_NEXT_4]], [[Z_5]]
-; AVX512-NEXT: [[TMP8:%.*]] = extractelement <32 x i1> [[TMP1]], i32 6
-; AVX512-NEXT: [[Z_6:%.*]] = zext i1 [[TMP8]] to i8
-; AVX512-NEXT: [[ACC_NEXT_6:%.*]] = add nuw nsw i8 [[ACC_NEXT_5]], [[Z_6]]
-; AVX512-NEXT: [[TMP9:%.*]] = extractelement <32 x i1> [[TMP1]], i32 7
-; AVX512-NEXT: [[Z_7:%.*]] = zext i1 [[TMP9]] to i8
-; AVX512-NEXT: [[ACC_NEXT_7:%.*]] = add nuw nsw i8 [[ACC_NEXT_6]], [[Z_7]]
-; AVX512-NEXT: [[TMP10:%.*]] = extractelement <32 x i1> [[TMP1]], i32 8
-; AVX512-NEXT: [[Z_8:%.*]] = zext i1 [[TMP10]] to i8
-; AVX512-NEXT: [[ACC_NEXT_8:%.*]] = add nuw nsw i8 [[ACC_NEXT_7]], [[Z_8]]
-; AVX512-NEXT: [[TMP11:%.*]] = extractelement <32 x i1> [[TMP1]], i32 9
-; AVX512-NEXT: [[Z_9:%.*]] = zext i1 [[TMP11]] to i8
-; AVX512-NEXT: [[ACC_NEXT_9:%.*]] = add nuw nsw i8 [[ACC_NEXT_8]], [[Z_9]]
-; AVX512-NEXT: [[TMP12:%.*]] = extractelement <32 x i1> [[TMP1]], i32 10
-; AVX512-NEXT: [[Z_10:%.*]] = zext i1 [[TMP12]] to i8
-; AVX512-NEXT: [[ACC_NEXT_10:%.*]] = add nuw nsw i8 [[ACC_NEXT_9]], [[Z_10]]
-; AVX512-NEXT: [[TMP13:%.*]] = extractelement <32 x i1> [[TMP1]], i32 11
-; AVX512-NEXT: [[Z_11:%.*]] = zext i1 [[TMP13]] to i8
-; AVX512-NEXT: [[ACC_NEXT_11:%.*]] = add nuw nsw i8 [[ACC_NEXT_10]], [[Z_11]]
-; AVX512-NEXT: [[TMP14:%.*]] = extractelement <32 x i1> [[TMP1]], i32 12
-; AVX512-NEXT: [[Z_12:%.*]] = zext i1 [[TMP14]] to i8
-; AVX512-NEXT: [[ACC_NEXT_12:%.*]] = add nuw nsw i8 [[ACC_NEXT_11]], [[Z_12]]
-; AVX512-NEXT: [[TMP15:%.*]] = extractelement <32 x i1> [[TMP1]], i32 13
-; AVX512-NEXT: [[Z_13:%.*]] = zext i1 [[TMP15]] to i8
-; AVX512-NEXT: [[ACC_NEXT_13:%.*]] = add nuw nsw i8 [[ACC_NEXT_12]], [[Z_13]]
-; AVX512-NEXT: [[TMP16:%.*]] = extractelement <32 x i1> [[TMP1]], i32 14
-; AVX512-NEXT: [[Z_14:%.*]] = zext i1 [[TMP16]] to i8
-; AVX512-NEXT: [[ACC_NEXT_14:%.*]] = add nuw nsw i8 [[ACC_NEXT_13]], [[Z_14]]
-; AVX512-NEXT: [[TMP17:%.*]] = extractelement <32 x i1> [[TMP1]], i32 15
-; AVX512-NEXT: [[Z_15:%.*]] = zext i1 [[TMP17]] to i8
-; AVX512-NEXT: [[ACC_NEXT_15:%.*]] = add nuw nsw i8 [[ACC_NEXT_14]], [[Z_15]]
-; AVX512-NEXT: [[TMP18:%.*]] = extractelement <32 x i1> [[TMP1]], i32 16
-; AVX512-NEXT: [[Z_16:%.*]] = zext i1 [[TMP18]] to i8
-; AVX512-NEXT: [[ACC_NEXT_16:%.*]] = add nuw nsw i8 [[ACC_NEXT_15]], [[Z_16]]
-; AVX512-NEXT: [[TMP19:%.*]] = extractelement <32 x i1> [[TMP1]], i32 17
-; AVX512-NEXT: [[Z_17:%.*]] = zext i1 [[TMP19]] to i8
-; AVX512-NEXT: [[ACC_NEXT_17:%.*]] = add nuw nsw i8 [[ACC_NEXT_16]], [[Z_17]]
-; AVX512-NEXT: [[TMP20:%.*]] = extractelement <32 x i1> [[TMP1]], i32 18
-; AVX512-NEXT: [[Z_18:%.*]] = zext i1 [[TMP20]] to i8
-; AVX512-NEXT: [[ACC_NEXT_18:%.*]] = add nuw nsw i8 [[ACC_NEXT_17]], [[Z_18]]
-; AVX512-NEXT: [[TMP21:%.*]] = extractelement <32 x i1> [[TMP1]], i32 19
-; AVX512-NEXT: [[Z_19:%.*]] = zext i1 [[TMP21]] to i8
-; AVX512-NEXT: [[ACC_NEXT_19:%.*]] = add nuw nsw i8 [[ACC_NEXT_18]], [[Z_19]]
-; AVX512-NEXT: [[TMP22:%.*]] = extractelement <32 x i1> [[TMP1]], i32 20
-; AVX512-NEXT: [[Z_20:%.*]] = zext i1 [[TMP22]] to i8
-; AVX512-NEXT: [[ACC_NEXT_20:%.*]] = add nuw nsw i8 [[ACC_NEXT_19]], [[Z_20]]
-; AVX512-NEXT: [[TMP23:%.*]] = extractelement <32 x i1> [[TMP1]], i32 21
-; AVX512-NEXT: [[Z_21:%.*]] = zext i1 [[TMP23]] to i8
-; AVX512-NEXT: [[ACC_NEXT_21:%.*]] = add nuw nsw i8 [[ACC_NEXT_20]], [[Z_21]]
-; AVX512-NEXT: [[TMP24:%.*]] = extractelement <32 x i1> [[TMP1]], i32 22
-; AVX512-NEXT: [[Z_22:%.*]] = zext i1 [[TMP24]] to i8
-; AVX512-NEXT: [[ACC_NEXT_22:%.*]] = add nuw nsw i8 [[ACC_NEXT_21]], [[Z_22]]
-; AVX512-NEXT: [[TMP25:%.*]] = extractelement <32 x i1> [[TMP1]], i32 23
-; AVX512-NEXT: [[Z_23:%.*]] = zext i1 [[TMP25]] to i8
-; AVX512-NEXT: [[ACC_NEXT_23:%.*]] = add nuw nsw i8 [[ACC_NEXT_22]], [[Z_23]]
-; AVX512-NEXT: [[TMP26:%.*]] = extractelement <32 x i1> [[TMP1]], i32 24
-; AVX512-NEXT: [[Z_24:%.*]] = zext i1 [[TMP26]] to i8
-; AVX512-NEXT: [[ACC_NEXT_24:%.*]] = add nuw nsw i8 [[ACC_NEXT_23]], [[Z_24]]
-; AVX512-NEXT: [[TMP27:%.*]] = extractelement <32 x i1> [[TMP1]], i32 25
-; AVX512-NEXT: [[Z_25:%.*]] = zext i1 [[TMP27]] to i8
-; AVX512-NEXT: [[ACC_NEXT_25:%.*]] = add nuw nsw i8 [[ACC_NEXT_24]], [[Z_25]]
-; AVX512-NEXT: [[TMP28:%.*]] = extractelement <32 x i1> [[TMP1]], i32 26
-; AVX512-NEXT: [[Z_26:%.*]] = zext i1 [[TMP28]] to i8
-; AVX512-NEXT: [[ACC_NEXT_26:%.*]] = add nuw nsw i8 [[ACC_NEXT_25]], [[Z_26]]
-; AVX512-NEXT: [[TMP29:%.*]] = extractelement <32 x i1> [[TMP1]], i32 27
-; AVX512-NEXT: [[Z_27:%.*]] = zext i1 [[TMP29]] to i8
-; AVX512-NEXT: [[ACC_NEXT_27:%.*]] = add nuw nsw i8 [[ACC_NEXT_26]], [[Z_27]]
-; AVX512-NEXT: [[TMP30:%.*]] = extractelement <32 x i1> [[TMP1]], i32 28
-; AVX512-NEXT: [[Z_28:%.*]] = zext i1 [[TMP30]] to i8
-; AVX512-NEXT: [[ACC_NEXT_28:%.*]] = add nuw nsw i8 [[ACC_NEXT_27]], [[Z_28]]
-; AVX512-NEXT: [[TMP31:%.*]] = extractelement <32 x i1> [[TMP1]], i32 29
-; AVX512-NEXT: [[Z_29:%.*]] = zext i1 [[TMP31]] to i8
-; AVX512-NEXT: [[ACC_NEXT_29:%.*]] = add nuw nsw i8 [[ACC_NEXT_28]], [[Z_29]]
-; AVX512-NEXT: [[TMP32:%.*]] = extractelement <32 x i1> [[TMP1]], i32 30
-; AVX512-NEXT: [[Z_30:%.*]] = zext i1 [[TMP32]] to i8
-; AVX512-NEXT: [[ACC_NEXT_30:%.*]] = add nuw nsw i8 [[ACC_NEXT_29]], [[Z_30]]
-; AVX512-NEXT: [[TMP33:%.*]] = extractelement <32 x i1> [[TMP1]], i32 31
-; AVX512-NEXT: [[Z_31:%.*]] = zext i1 [[TMP33]] to i8
-; AVX512-NEXT: [[ACC_NEXT_31:%.*]] = add nuw nsw i8 [[ACC_NEXT_30]], [[Z_31]]
-; AVX512-NEXT: [[OK:%.*]] = icmp eq i8 [[ACC_NEXT_31]], 32
-; AVX512-NEXT: ret i1 [[OK]]
+; AVX-LABEL: define i1 @PR176906(
+; AVX-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
+; AVX-NEXT: [[V:%.*]] = load <32 x i8>, ptr [[P]], align 1
+; AVX-NEXT: [[TMP1:%.*]] = icmp sgt <32 x i8> [[V]], splat (i8 -1)
+; AVX-NEXT: [[TMP2:%.*]] = bitcast <32 x i1> [[TMP1]] to i32
+; AVX-NEXT: [[TMP3:%.*]] = call i32 @llvm.ctpop.i32(i32 [[TMP2]])
+; AVX-NEXT: [[TMP4:%.*]] = trunc i32 [[TMP3]] to i8
+; AVX-NEXT: [[OK:%.*]] = icmp eq i8 [[TMP4]], 32
+; AVX-NEXT: ret i1 [[OK]]
;
%v = load <32 x i8>, ptr %p, align 1
%i = extractelement <32 x i8> %v, i64 0
@@ -257,6 +155,3 @@ define i1 @PR176906(ptr %p) {
%ok = icmp eq i8 %acc.next.31, 32
ret i1 %ok
}
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; AVX: {{.*}}
-; CHECK: {{.*}}
``````````
</details>
https://github.com/llvm/llvm-project/pull/178400
More information about the llvm-commits
mailing list