[llvm] [VectorCombine] foldShuffleChainsToReduce - add fmaximum/fminimum handling (PR #200818)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Jul 29 06:18:06 PDT 2026
https://github.com/lijinpei-amd updated https://github.com/llvm/llvm-project/pull/200818
>From 0aa0a61ad2dc1f82d4758a42bb7167cd413fb8e9 Mon Sep 17 00:00:00 2001
From: Li Jinpei <jinpli at amd.com>
Date: Mon, 1 Jun 2026 20:15:36 +0800
Subject: [PATCH] [VectorCombine] foldShuffleChainsToReduce - add
fmaximum/fminimum handling
Extend foldShuffleChainsToReduce to fold shuffle-reduction chains of the
floating-point maximum/minimum intrinsics into the corresponding vector
reductions:
llvm.maximum -> llvm.vector.reduce.fmaximum
llvm.minimum -> llvm.vector.reduce.fminimum
Addresses #199038
---
.../include/llvm/Transforms/Utils/LoopUtils.h | 3 +-
llvm/lib/Transforms/Utils/LoopUtils.cpp | 7 +-
.../Transforms/Vectorize/VectorCombine.cpp | 17 ++-
.../fold-shuffle-chains-to-reduce.ll | 123 ++++++++++++++++++
4 files changed, 143 insertions(+), 7 deletions(-)
diff --git a/llvm/include/llvm/Transforms/Utils/LoopUtils.h b/llvm/include/llvm/Transforms/Utils/LoopUtils.h
index 9fa675856ba3f..1c33523c3a6c0 100644
--- a/llvm/include/llvm/Transforms/Utils/LoopUtils.h
+++ b/llvm/include/llvm/Transforms/Utils/LoopUtils.h
@@ -473,7 +473,8 @@ LLVM_ABI bool canHoistLoad(LoadInst &LI, AAResults *AA, DominatorTree *DT,
/// kind.
LLVM_ABI constexpr Intrinsic::ID getReductionIntrinsicID(RecurKind RK);
/// Returns the llvm.vector.reduce min/max intrinsic that corresponds to the
-/// intrinsic op.
+/// intrinsic op, or Intrinsic::not_intrinsic if \p IID is not a min/max op with
+/// an equivalent reduction.
LLVM_ABI Intrinsic::ID getMinMaxReductionIntrinsicID(Intrinsic::ID IID);
/// Returns the arithmetic instruction opcode used when expanding a reduction.
diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index 51848a8e1391e..bbb3bed7d3db8 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -1147,7 +1147,7 @@ constexpr Intrinsic::ID llvm::getReductionIntrinsicID(RecurKind RK) {
Intrinsic::ID llvm::getMinMaxReductionIntrinsicID(Intrinsic::ID IID) {
switch (IID) {
default:
- llvm_unreachable("Unexpected intrinsic id");
+ break;
case Intrinsic::umin:
return Intrinsic::vector_reduce_umin;
case Intrinsic::umax:
@@ -1156,7 +1156,12 @@ Intrinsic::ID llvm::getMinMaxReductionIntrinsicID(Intrinsic::ID IID) {
return Intrinsic::vector_reduce_smin;
case Intrinsic::smax:
return Intrinsic::vector_reduce_smax;
+ case Intrinsic::minimum:
+ return Intrinsic::vector_reduce_fminimum;
+ case Intrinsic::maximum:
+ return Intrinsic::vector_reduce_fmaximum;
}
+ return Intrinsic::not_intrinsic;
}
// This is the inverse to getReductionForBinop
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 726f564b1aad9..ad6e58410a906 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -4078,6 +4078,13 @@ bool VectorCombine::foldShuffleFromReductions(Instruction &I) {
return MadeChanges;
}
+static Intrinsic::ID getMinMaxReductionOp(Value *V) {
+ auto *II = dyn_cast<IntrinsicInst>(V);
+ if (!II || !getMinMaxReductionIntrinsicID(II->getIntrinsicID()))
+ return Intrinsic::not_intrinsic;
+ return II->getIntrinsicID();
+}
+
/// Try to fold a chain of shuffles and ops feeding extractelement(..., 0)
/// into llvm.vector.reduce.*, by tracking which lanes contribute to the
/// extracted lane and reducing the widest vector whose lanes each contribute
@@ -4114,8 +4121,8 @@ bool VectorCombine::foldShuffleChainsToReduce(Instruction &I) {
if (!getReductionForBinop(BO->getOpcode()))
return false;
CommonBinOp = BO->getOpcode();
- } else if (auto *MMI = dyn_cast<MinMaxIntrinsic>(VecOpEE)) {
- CommonCallOp = MMI->getIntrinsicID();
+ } else if (Intrinsic::ID MinMaxID = getMinMaxReductionOp(VecOpEE)) {
+ CommonCallOp = MinMaxID;
} else {
return false;
}
@@ -4129,8 +4136,8 @@ bool VectorCombine::foldShuffleChainsToReduce(Instruction &I) {
auto IsChainNode = [&](Value *V) {
if (auto *BO = dyn_cast<BinaryOperator>(V))
return CommonBinOp && BO->getOpcode() == *CommonBinOp;
- if (auto *MMI = dyn_cast<MinMaxIntrinsic>(V))
- return CommonCallOp && MMI->getIntrinsicID() == *CommonCallOp;
+ if (Intrinsic::ID MinMaxID = getMinMaxReductionOp(V))
+ return CommonCallOp && MinMaxID == *CommonCallOp;
if (auto *SVI = dyn_cast<ShuffleVectorInst>(V))
return isa<PoisonValue>(SVI->getOperand(1));
return false;
@@ -4275,7 +4282,7 @@ bool VectorCombine::foldShuffleChainsToReduce(Instruction &I) {
}
if (!Cut) {
for (Value *V : Nodes) {
- if (!isa<BinaryOperator>(V) && !isa<MinMaxIntrinsic>(V))
+ if (!isa<BinaryOperator>(V) && !getMinMaxReductionOp(V))
continue;
auto It = Demands.find(V);
if (It == Demands.end() || !It->second.Lanes.isAllOnes())
diff --git a/llvm/test/Transforms/VectorCombine/fold-shuffle-chains-to-reduce.ll b/llvm/test/Transforms/VectorCombine/fold-shuffle-chains-to-reduce.ll
index f00ce63159d17..814eb473084fa 100644
--- a/llvm/test/Transforms/VectorCombine/fold-shuffle-chains-to-reduce.ll
+++ b/llvm/test/Transforms/VectorCombine/fold-shuffle-chains-to-reduce.ll
@@ -210,6 +210,129 @@ define i16 @test_no_reduce_v6i16_xor_poison(<6 x i16> %a0) {
ret i16 %7
}
+define float @test_reduce_v8f32_fmaximum(<8 x float> %a0) {
+; CHECK-LABEL: define float @test_reduce_v8f32_fmaximum(
+; CHECK-SAME: <8 x float> [[A0:%.*]]) {
+; CHECK-NEXT: [[TMP1:%.*]] = call float @llvm.vector.reduce.fmaximum.v8f32(<8 x float> [[A0]])
+; CHECK-NEXT: ret float [[TMP1]]
+;
+ %1 = shufflevector <8 x float> %a0, <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison>
+ %2 = tail call <8 x float> @llvm.maximum.v8f32(<8 x float> %a0, <8 x float> %1)
+ %3 = shufflevector <8 x float> %2, <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %4 = tail call <8 x float> @llvm.maximum.v8f32(<8 x float> %2, <8 x float> %3)
+ %5 = shufflevector <8 x float> %4, <8 x float> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %6 = tail call <8 x float> @llvm.maximum.v8f32(<8 x float> %4, <8 x float> %5)
+ %7 = extractelement <8 x float> %6, i64 0
+ ret float %7
+}
+
+define float @test_reduce_v8f32_fminimum(<8 x float> %a0) {
+; CHECK-LABEL: define float @test_reduce_v8f32_fminimum(
+; CHECK-SAME: <8 x float> [[A0:%.*]]) {
+; CHECK-NEXT: [[TMP1:%.*]] = call float @llvm.vector.reduce.fminimum.v8f32(<8 x float> [[A0]])
+; CHECK-NEXT: ret float [[TMP1]]
+;
+ %1 = shufflevector <8 x float> %a0, <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison>
+ %2 = tail call <8 x float> @llvm.minimum.v8f32(<8 x float> %a0, <8 x float> %1)
+ %3 = shufflevector <8 x float> %2, <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %4 = tail call <8 x float> @llvm.minimum.v8f32(<8 x float> %2, <8 x float> %3)
+ %5 = shufflevector <8 x float> %4, <8 x float> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %6 = tail call <8 x float> @llvm.minimum.v8f32(<8 x float> %4, <8 x float> %5)
+ %7 = extractelement <8 x float> %6, i64 0
+ ret float %7
+}
+
+; Negative tests: maxnum/minnum/maximumnum/minimumnum reductions map to
+; llvm.vector.reduce.fmax/fmin, whose comparison order is non-deterministic for
+; signaling-NaN inputs, so these chains are intentionally not folded.
+define float @test_reduce_v8f32_maxnum_neg(<8 x float> %a0) {
+; CHECK-LABEL: define float @test_reduce_v8f32_maxnum_neg(
+; CHECK-SAME: <8 x float> [[A0:%.*]]) {
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <8 x float> [[A0]], <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP2:%.*]] = tail call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[A0]], <8 x float> [[TMP1]])
+; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <8 x float> [[TMP2]], <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP4:%.*]] = tail call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP2]], <8 x float> [[TMP3]])
+; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <8 x float> [[TMP4]], <8 x float> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP6:%.*]] = tail call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; CHECK-NEXT: [[TMP7:%.*]] = extractelement <8 x float> [[TMP6]], i64 0
+; CHECK-NEXT: ret float [[TMP7]]
+;
+ %1 = shufflevector <8 x float> %a0, <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison>
+ %2 = tail call <8 x float> @llvm.maxnum.v8f32(<8 x float> %a0, <8 x float> %1)
+ %3 = shufflevector <8 x float> %2, <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %4 = tail call <8 x float> @llvm.maxnum.v8f32(<8 x float> %2, <8 x float> %3)
+ %5 = shufflevector <8 x float> %4, <8 x float> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %6 = tail call <8 x float> @llvm.maxnum.v8f32(<8 x float> %4, <8 x float> %5)
+ %7 = extractelement <8 x float> %6, i64 0
+ ret float %7
+}
+
+define float @test_reduce_v8f32_minnum_neg(<8 x float> %a0) {
+; CHECK-LABEL: define float @test_reduce_v8f32_minnum_neg(
+; CHECK-SAME: <8 x float> [[A0:%.*]]) {
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <8 x float> [[A0]], <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP2:%.*]] = tail call <8 x float> @llvm.minnum.v8f32(<8 x float> [[A0]], <8 x float> [[TMP1]])
+; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <8 x float> [[TMP2]], <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP4:%.*]] = tail call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP2]], <8 x float> [[TMP3]])
+; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <8 x float> [[TMP4]], <8 x float> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP6:%.*]] = tail call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; CHECK-NEXT: [[TMP7:%.*]] = extractelement <8 x float> [[TMP6]], i64 0
+; CHECK-NEXT: ret float [[TMP7]]
+;
+ %1 = shufflevector <8 x float> %a0, <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison>
+ %2 = tail call <8 x float> @llvm.minnum.v8f32(<8 x float> %a0, <8 x float> %1)
+ %3 = shufflevector <8 x float> %2, <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %4 = tail call <8 x float> @llvm.minnum.v8f32(<8 x float> %2, <8 x float> %3)
+ %5 = shufflevector <8 x float> %4, <8 x float> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %6 = tail call <8 x float> @llvm.minnum.v8f32(<8 x float> %4, <8 x float> %5)
+ %7 = extractelement <8 x float> %6, i64 0
+ ret float %7
+}
+
+define float @test_reduce_v8f32_maximumnum_neg(<8 x float> %a0) {
+; CHECK-LABEL: define float @test_reduce_v8f32_maximumnum_neg(
+; CHECK-SAME: <8 x float> [[A0:%.*]]) {
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <8 x float> [[A0]], <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP2:%.*]] = tail call <8 x float> @llvm.maximumnum.v8f32(<8 x float> [[A0]], <8 x float> [[TMP1]])
+; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <8 x float> [[TMP2]], <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP4:%.*]] = tail call <8 x float> @llvm.maximumnum.v8f32(<8 x float> [[TMP2]], <8 x float> [[TMP3]])
+; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <8 x float> [[TMP4]], <8 x float> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP6:%.*]] = tail call <8 x float> @llvm.maximumnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; CHECK-NEXT: [[TMP7:%.*]] = extractelement <8 x float> [[TMP6]], i64 0
+; CHECK-NEXT: ret float [[TMP7]]
+;
+ %1 = shufflevector <8 x float> %a0, <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison>
+ %2 = tail call <8 x float> @llvm.maximumnum.v8f32(<8 x float> %a0, <8 x float> %1)
+ %3 = shufflevector <8 x float> %2, <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %4 = tail call <8 x float> @llvm.maximumnum.v8f32(<8 x float> %2, <8 x float> %3)
+ %5 = shufflevector <8 x float> %4, <8 x float> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %6 = tail call <8 x float> @llvm.maximumnum.v8f32(<8 x float> %4, <8 x float> %5)
+ %7 = extractelement <8 x float> %6, i64 0
+ ret float %7
+}
+
+define float @test_reduce_v8f32_minimumnum_neg(<8 x float> %a0) {
+; CHECK-LABEL: define float @test_reduce_v8f32_minimumnum_neg(
+; CHECK-SAME: <8 x float> [[A0:%.*]]) {
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <8 x float> [[A0]], <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP2:%.*]] = tail call <8 x float> @llvm.minimumnum.v8f32(<8 x float> [[A0]], <8 x float> [[TMP1]])
+; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <8 x float> [[TMP2]], <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP4:%.*]] = tail call <8 x float> @llvm.minimumnum.v8f32(<8 x float> [[TMP2]], <8 x float> [[TMP3]])
+; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <8 x float> [[TMP4]], <8 x float> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP6:%.*]] = tail call <8 x float> @llvm.minimumnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; CHECK-NEXT: [[TMP7:%.*]] = extractelement <8 x float> [[TMP6]], i64 0
+; CHECK-NEXT: ret float [[TMP7]]
+;
+ %1 = shufflevector <8 x float> %a0, <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison>
+ %2 = tail call <8 x float> @llvm.minimumnum.v8f32(<8 x float> %a0, <8 x float> %1)
+ %3 = shufflevector <8 x float> %2, <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %4 = tail call <8 x float> @llvm.minimumnum.v8f32(<8 x float> %2, <8 x float> %3)
+ %5 = shufflevector <8 x float> %4, <8 x float> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+ %6 = tail call <8 x float> @llvm.minimumnum.v8f32(<8 x float> %4, <8 x float> %5)
+ %7 = extractelement <8 x float> %6, i64 0
+ ret float %7
+}
+
; Partial reduction: reduce lower 8 elements of a 16-element vector using smax.
define i16 @test_partial_reduce_v16i16_v8i16_smax(<16 x i16> %a0) {
; CHECK-LABEL: define i16 @test_partial_reduce_v16i16_v8i16_smax(
More information about the llvm-commits
mailing list