[llvm] [AMDGPU] Price only the fmul that fuses into an fadd/fsub as free (PR #226009)

Gheorghe-Teodor Bercea via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 06:14:20 PDT 2026


https://github.com/doru1004 updated https://github.com/llvm/llvm-project/pull/226009

>From 537155fdbb8029c40f3b040ef9f0146d5d28d0f2 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 23 Sep 2026 23:17:33 -0400
Subject: [PATCH 1/3] Price only the fmul that fuses into an fadd/fsub as free

---
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      | 28 ++++---
 .../Analysis/CostModel/AMDGPU/fused_costs.ll  | 84 +++++++++++++++++++
 .../SLPVectorizer/AMDGPU/complex-mul-fma.ll   | 36 ++++++++
 3 files changed, 136 insertions(+), 12 deletions(-)
 create mode 100644 llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index ccb0c7314dcef4..4a7f84f0baed5c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -546,6 +546,18 @@ static bool canFuseFMulWithFAddSub(const SITargetLowering &TLI, Type *Ty,
   return HasFMAD || (FAddSub->hasAllowContract() && FMul->hasAllowContract());
 }
 
+/// An fma holds one multiply, so only one fmul operand fuses with \p FAddSub.
+static const Instruction *getFusedFMul(const SITargetLowering &TLI, Type *Ty,
+                                       const Instruction *FAddSub) {
+  for (const Value *Op : FAddSub->operands()) {
+    const auto *FMul = dyn_cast<Instruction>(Op);
+    if (FMul && FMul->getOpcode() == Instruction::FMul && FMul->hasOneUse() &&
+        canFuseFMulWithFAddSub(TLI, Ty, FMul, FAddSub))
+      return FMul;
+  }
+  return nullptr;
+}
+
 InstructionCost GCNTTIImpl::getArithmeticInstrCost(
     unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
     TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
@@ -613,7 +625,7 @@ InstructionCost GCNTTIImpl::getArithmeticInstrCost(
       if (FAddSub &&
           (FAddSub->getOpcode() == Instruction::FAdd ||
            FAddSub->getOpcode() == Instruction::FSub) &&
-          canFuseFMulWithFAddSub(*TLI, Ty, CxtI, FAddSub))
+          getFusedFMul(*TLI, Ty, FAddSub) == CxtI)
         return TargetTransformInfo::TCC_Free;
     }
     [[fallthrough]];
@@ -1533,17 +1545,9 @@ bool GCNTTIImpl::isProfitableToSinkOperands(Instruction *I,
   // so this stays a move.
   if (I->getOpcode() == Instruction::FAdd ||
       I->getOpcode() == Instruction::FSub) {
-    for (Use &Op : I->operands()) {
-      auto *FMul = dyn_cast<Instruction>(Op.get());
-      if (!FMul || FMul->getOpcode() != Instruction::FMul ||
-          !FMul->hasOneUse() ||
-          !canFuseFMulWithFAddSub(*TLI, I->getType(), FMul, I))
-        continue;
-      // The fused operand. Sink it when it sits in another block, then stop.
-      if (FMul->getParent() != I->getParent())
-        Ops.push_back(&Op);
-      break;
-    }
+    const Instruction *FMul = getFusedFMul(*TLI, I->getType(), I);
+    if (FMul && FMul->getParent() != I->getParent())
+      Ops.push_back(&I->getOperandUse(I->getOperand(0) == FMul ? 0 : 1));
   }
 
   for (auto &Op : I->operands()) {
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
index aeec6fd9165142..1ca2d1e785bf53 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
@@ -251,6 +251,90 @@ define void @fmul_fadd_f64(double %a, double %b, double %c, <2 x double> %va, <2
   ret void
 }
 
+define void @fmul_fmul_fadd_f32(float %a, float %b, float %c, float %d) #0 {
+; SLOWF32-LABEL: 'fmul_fmul_fadd_f32'
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %nc.m0 = fmul float %a, %b
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; FASTF32-LABEL: 'fmul_fmul_fadd_f32'
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc.m0 = fmul float %a, %b
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %nc.m1 = fmul contract float %c, %d
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; SLOWF32-SIZE-LABEL: 'fmul_fmul_fadd_f32'
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %nc.m0 = fmul float %a, %b
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; FASTF32-SIZE-LABEL: 'fmul_fmul_fadd_f32'
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc.m0 = fmul float %a, %b
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %nc.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+  %add.m0 = fmul contract float %a, %b
+  %add.m1 = fmul contract float %c, %d
+  %add = fadd contract float %add.m0, %add.m1
+
+  %sub.m0 = fmul contract float %a, %b
+  %sub.m1 = fmul contract float %c, %d
+  %sub = fsub contract float %sub.m0, %sub.m1
+
+  %multi.m0 = fmul contract float %a, %b
+  %multi.m1 = fmul contract float %c, %d
+  %multi = fadd contract float %multi.m0, %multi.m1
+  %multi.use = fadd contract float %multi.m0, %c
+
+  %nc.m0 = fmul float %a, %b
+  %nc.m1 = fmul contract float %c, %d
+  %nc = fadd contract float %nc.m0, %nc.m1
+  ret void
+}
+
 attributes #0 = { nounwind }
 
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
new file mode 100644
index 00000000000000..26e3b0e71d76e0
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
@@ -0,0 +1,36 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -slp-threshold=1 < %s | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -slp-threshold=1 < %s | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -slp-threshold=1 < %s | FileCheck %s
+
+; The second fmul of each fsub/fadd does not fuse, so packing those pays off.
+
+define void @cmul_store(ptr addrspace(1) %out, float %xr, float %xi, float %wr, float %wi) {
+; CHECK-LABEL: define void @cmul_store(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], float [[XR:%.*]], float [[XI:%.*]], float [[WR:%.*]], float [[WI:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x float> poison, float [[XR]], i64 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x float> [[TMP1]], float [[XI]], i64 1
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <2 x float> poison, float [[WR]], i64 0
+; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP5:%.*]] = fmul contract <2 x float> [[TMP2]], [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <2 x float> poison, float [[WI]], i64 0
+; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <2 x float> [[TMP6]], <2 x float> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP8:%.*]] = fmul contract <2 x float> [[TMP2]], [[TMP7]]
+; CHECK-NEXT:    [[TMP9:%.*]] = shufflevector <2 x float> [[TMP8]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
+; CHECK-NEXT:    [[TMP10:%.*]] = fsub contract <2 x float> [[TMP5]], [[TMP9]]
+; CHECK-NEXT:    [[TMP11:%.*]] = fadd contract <2 x float> [[TMP5]], [[TMP9]]
+; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <2 x float> [[TMP10]], <2 x float> [[TMP11]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    store <2 x float> [[TMP12]], ptr addrspace(1) [[OUT]], align 8
+; CHECK-NEXT:    ret void
+;
+  %xr.wr = fmul contract float %xr, %wr
+  %xi.wi = fmul contract float %xi, %wi
+  %re = fsub contract float %xr.wr, %xi.wi
+  %xi.wr = fmul contract float %xi, %wr
+  %xr.wi = fmul contract float %xr, %wi
+  %im = fadd contract float %xi.wr, %xr.wi
+  store float %re, ptr addrspace(1) %out, align 8
+  %out.im = getelementptr inbounds float, ptr addrspace(1) %out, i64 1
+  store float %im, ptr addrspace(1) %out.im, align 4
+  ret void
+}

>From e52a7fae4f25558ae2bd54c9c0997fecf376591b Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Thu, 24 Sep 2026 08:47:18 -0400
Subject: [PATCH 2/3] Fix the cost and the test.

---
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      | 18 ++++++++++++-
 .../Analysis/CostModel/AMDGPU/fused_costs.ll  | 22 ++++++++++++++++
 .../SLPVectorizer/AMDGPU/complex-mul-fma.ll   | 25 +++----------------
 3 files changed, 42 insertions(+), 23 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 4a7f84f0baed5c..f5119764bbb9db 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -558,6 +558,22 @@ static const Instruction *getFusedFMul(const SITargetLowering &TLI, Type *Ty,
   return nullptr;
 }
 
+static bool isFusedFMul(const SITargetLowering &TLI, Type *Ty,
+                        const Instruction *FMul, const Instruction *FAddSub) {
+  const Instruction *Fused = getFusedFMul(TLI, Ty, FAddSub);
+  if (Fused == FMul)
+    return true;
+  // (a * b + c * d) + e becomes fma(a, b, fma(c, d, e)) if the outer fadd has
+  // reassoc.
+  if (!Fused || FAddSub->getOpcode() != Instruction::FAdd ||
+      !FAddSub->hasOneUse())
+    return false;
+  const auto *Outer = dyn_cast<BinaryOperator>(*FAddSub->user_begin());
+  return Outer && Outer->getOpcode() == Instruction::FAdd &&
+         Outer->hasAllowReassoc() &&
+         canFuseFMulWithFAddSub(TLI, Ty, FMul, Outer);
+}
+
 InstructionCost GCNTTIImpl::getArithmeticInstrCost(
     unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
     TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
@@ -625,7 +641,7 @@ InstructionCost GCNTTIImpl::getArithmeticInstrCost(
       if (FAddSub &&
           (FAddSub->getOpcode() == Instruction::FAdd ||
            FAddSub->getOpcode() == Instruction::FSub) &&
-          getFusedFMul(*TLI, Ty, FAddSub) == CxtI)
+          isFusedFMul(*TLI, Ty, CxtI, FAddSub))
         return TargetTransformInfo::TCC_Free;
     }
     [[fallthrough]];
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
index 1ca2d1e785bf53..3060d103bb1df6 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
@@ -335,6 +335,28 @@ define void @fmul_fmul_fadd_f32(float %a, float %b, float %c, float %d) #0 {
   ret void
 }
 
+define float @fmul_fmul_fadd_reassoc_f32(float %a, float %b, float %c, float %d, float %val) #0 {
+; SLOWF64-LABEL: 'fmul_fmul_fadd_reassoc_f32'
+; SLOWF64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %m0 = fmul contract float %a, %b
+; SLOWF64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %m1 = fmul contract float %c, %d
+; SLOWF64-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sum = fadd contract float %m0, %m1
+; SLOWF64-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %ret = fadd reassoc contract float %sum, %val
+; SLOWF64-NEXT:  Cost Model: Found an estimated cost of 10 for instruction: ret float %ret
+;
+; SLOWF64-SIZE-LABEL: 'fmul_fmul_fadd_reassoc_f32'
+; SLOWF64-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %m0 = fmul contract float %a, %b
+; SLOWF64-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %m1 = fmul contract float %c, %d
+; SLOWF64-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sum = fadd contract float %m0, %m1
+; SLOWF64-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %ret = fadd reassoc contract float %sum, %val
+; SLOWF64-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: ret float %ret
+;
+  %m0 = fmul contract float %a, %b
+  %m1 = fmul contract float %c, %d
+  %sum = fadd contract float %m0, %m1
+  %ret = fadd reassoc contract float %sum, %val
+  ret float %ret
+}
+
 attributes #0 = { nounwind }
 
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
index 26e3b0e71d76e0..b4c0de7fcf65cf 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
@@ -1,28 +1,9 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -slp-threshold=1 < %s | FileCheck %s
-; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -slp-threshold=1 < %s | FileCheck %s
-; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -slp-threshold=1 < %s | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
 
 ; The second fmul of each fsub/fadd does not fuse, so packing those pays off.
-
+; CHECK: Stores SLP vectorized with cost -2
 define void @cmul_store(ptr addrspace(1) %out, float %xr, float %xi, float %wr, float %wi) {
-; CHECK-LABEL: define void @cmul_store(
-; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], float [[XR:%.*]], float [[XI:%.*]], float [[WR:%.*]], float [[WI:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x float> poison, float [[XR]], i64 0
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x float> [[TMP1]], float [[XI]], i64 1
-; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <2 x float> poison, float [[WR]], i64 0
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP5:%.*]] = fmul contract <2 x float> [[TMP2]], [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <2 x float> poison, float [[WI]], i64 0
-; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <2 x float> [[TMP6]], <2 x float> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP8:%.*]] = fmul contract <2 x float> [[TMP2]], [[TMP7]]
-; CHECK-NEXT:    [[TMP9:%.*]] = shufflevector <2 x float> [[TMP8]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
-; CHECK-NEXT:    [[TMP10:%.*]] = fsub contract <2 x float> [[TMP5]], [[TMP9]]
-; CHECK-NEXT:    [[TMP11:%.*]] = fadd contract <2 x float> [[TMP5]], [[TMP9]]
-; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <2 x float> [[TMP10]], <2 x float> [[TMP11]], <2 x i32> <i32 0, i32 3>
-; CHECK-NEXT:    store <2 x float> [[TMP12]], ptr addrspace(1) [[OUT]], align 8
-; CHECK-NEXT:    ret void
-;
   %xr.wr = fmul contract float %xr, %wr
   %xi.wi = fmul contract float %xi, %wi
   %re = fsub contract float %xr.wr, %xi.wi

>From ed38f91e967af1e0265be73ed532b2bcfef2e841 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Thu, 24 Sep 2026 09:07:22 -0400
Subject: [PATCH 3/3] Use new triple

---
 llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll | 3 +--
 1 file changed, 1 insertion(+), 2 deletions(-)

diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
index b4c0de7fcf65cf..3d3b2c151ecf0a 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
@@ -1,5 +1,4 @@
-; RUN: opt -passes=slp-vectorizer -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
-; RUN: opt -passes=slp-vectorizer -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -mtriple=amdgpu9.42-amd-amdhsa -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
 
 ; The second fmul of each fsub/fadd does not fuse, so packing those pays off.
 ; CHECK: Stores SLP vectorized with cost -2



More information about the llvm-commits mailing list