[llvm] [SLP] Inefficient cost-modelling and codegen for reductions with slp-… (PR #197875)
Sushant Gokhale via llvm-commits
llvm-commits at lists.llvm.org
Thu Jun 4 01:37:33 PDT 2026
================
@@ -29189,34 +29190,48 @@ class HorizontalReduction {
private:
/// Creates the reduction from the given \p Vec vector value with the given
/// scale \p Scale and signedness \p IsSigned.
- Value *createSingleOp(IRBuilderBase &Builder, const TargetTransformInfo &TTI,
- Value *Vec, unsigned Scale, bool IsSigned, Type *DestTy,
+ Value *createSingleOp(BoUpSLP &V, IRBuilderBase &Builder,
+ const TargetTransformInfo &TTI, Value *Vec,
+ unsigned Scale, bool IsSigned, Type *DestTy,
bool ReducedInTree) {
Value *Rdx;
if (ReducedInTree) {
Rdx = Vec;
} else if (auto *VecTy = dyn_cast<FixedVectorType>(DestTy)) {
unsigned DestTyNumElements = getNumElements(VecTy);
unsigned VF = getNumElements(Vec->getType()) / DestTyNumElements;
- Rdx = PoisonValue::get(
- getWidenedType(Vec->getType()->getScalarType(), DestTyNumElements));
- for (unsigned I : seq<unsigned>(DestTyNumElements)) {
- // Do reduction for each lane.
- // e.g., do reduce add for
- // VL[0] = <4 x Ty> <a, b, c, d>
- // VL[1] = <4 x Ty> <e, f, g, h>
- // Lane[0] = <2 x Ty> <a, e>
- // Lane[1] = <2 x Ty> <b, f>
- // Lane[2] = <2 x Ty> <c, g>
- // Lane[3] = <2 x Ty> <d, h>
- // result[0] = reduce add Lane[0]
- // result[1] = reduce add Lane[1]
- // result[2] = reduce add Lane[2]
- // result[3] = reduce add Lane[3]
- SmallVector<int, 16> Mask = createStrideMask(I, DestTyNumElements, VF);
- Value *Lane = Builder.CreateShuffleVector(Vec, Mask);
- Rdx = Builder.CreateInsertElement(
- Rdx, emitReduction(Lane, Builder, &TTI, DestTy), I);
+ unsigned RdxOpcode = RecurrenceDescriptor::getOpcode(RdxKind);
+ Rdx = nullptr;
+ /*
+ e.g. Consider vector reduce add.
+
+ Initial reduction is
+ %add0 = add <4 x i32> zeroinitializer, %A
+ %add1 = add <4 x i32> %add0, %B
+
+ where,
+ %A = <a, b, c, d>
+ %B = <e, f, g, h>
+ %add1 = A + B = <a + e, b + f, c + g, d + h>
+
+ After revectorization with VF=2 and Vec = <a, b, c, d, e, f, g, h>,
+ the reduction can be expressed as:
+ %A = shufflevector <8 x i32> %Vec, <8 x i32> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+ %B = shufflevector <8 x i32> %Vec, <8 x i32> poison, <4 x i32> <i32 4, i32 5, i32 6, i32 7>
+ %add0 = add <4 x i32> zeroinitializer, %A
+ %add1 = add <4 x i32> %add0, %B
+ */
----------------
sushgokh wrote:
done
https://github.com/llvm/llvm-project/pull/197875
More information about the llvm-commits
mailing list