[llvm] 0513c32 - [GlobalISel] When folding to fma correctly find fpext(fmul) with fewer uses (#218013)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 7 20:04:15 PDT 2026


Author: LU-JOHN
Date: 2026-09-07T22:04:10-05:00
New Revision: 0513c32abe1eff329fd22b578f4995e59d6140a2

URL: https://github.com/llvm/llvm-project/commit/0513c32abe1eff329fd22b578f4995e59d6140a2
DIFF: https://github.com/llvm/llvm-project/commit/0513c32abe1eff329fd22b578f4995e59d6140a2.diff

LOG: [GlobalISel] When folding to fma correctly find fpext(fmul) with fewer uses (#218013)

When folding: `(fadd (fpext (fmul x, y)), z) -> (fma (fpext x), (fpext
y), z)` and both addends are candidates, correctly pick the side with
fewer uses. Previously the code looked for a `fmul` with the fewest uses
rather than a `fpext(fmul)`.

---------

Signed-off-by: John Lu <John.Lu at amd.com>

Added: 
    

Modified: 
    llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
    llvm/test/CodeGen/AMDGPU/GlobalISel/combine-fma-add-ext-mul.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp b/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
index 98e76aea515b6..0ee8a9a204c09 100644
--- a/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
+++ b/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
@@ -6551,46 +6551,39 @@ bool CombinerHelper::matchCombineFAddFpExtFMulToFMadOrFMA(
   unsigned PreferredFusedOpcode =
       HasFMAD ? TargetOpcode::G_FMAD : TargetOpcode::G_FMA;
 
-  // If we have two choices trying to fold (fadd (fmul u, v), (fmul x, y)),
-  // prefer to fold the multiply with fewer uses.
-  if (Aggressive && isContractableFMul(*LHS.MI, AllowFusionGlobally) &&
-      isContractableFMul(*RHS.MI, AllowFusionGlobally)) {
-    if (hasMoreUses(*LHS.MI, *RHS.MI, MRI))
+  MachineInstr *LHSFpExtSrc;
+  bool LHSContractable =
+      mi_match(LHS.Reg, MRI, m_GFPExt(m_MInstr(LHSFpExtSrc))) &&
+      isContractableFMul(*LHSFpExtSrc, AllowFusionGlobally) &&
+      TLI.isFPExtFoldable(MI, PreferredFusedOpcode, DstType,
+                          MRI.getType(LHSFpExtSrc->getOperand(1).getReg()));
+  MachineInstr *RHSFpExtSrc;
+  bool RHSContractable =
+      mi_match(RHS.Reg, MRI, m_GFPExt(m_MInstr(RHSFpExtSrc))) &&
+      isContractableFMul(*RHSFpExtSrc, AllowFusionGlobally) &&
+      TLI.isFPExtFoldable(MI, PreferredFusedOpcode, DstType,
+                          MRI.getType(RHSFpExtSrc->getOperand(1).getReg()));
+
+  //  fold (fadd (fpext (fmul x, y)), z) -> (fma (fpext x), (fpext y), z)
+  if (LHSContractable || RHSContractable) {
+    // Ensure that the contractable fmul with the fewest uses (if both are
+    // contractable) is the LHS operand.
+    if (!LHSContractable ||
+        (RHSContractable && hasMoreUses(*LHSFpExtSrc, *RHSFpExtSrc, MRI))) {
       std::swap(LHS, RHS);
-  }
+      LHSFpExtSrc = RHSFpExtSrc;
+    }
 
-  // fold (fadd (fpext (fmul x, y)), z) -> (fma (fpext x), (fpext y), z)
-  MachineInstr *FpExtSrc;
-  if (mi_match(LHS.Reg, MRI, m_GFPExt(m_MInstr(FpExtSrc))) &&
-      isContractableFMul(*FpExtSrc, AllowFusionGlobally) &&
-      TLI.isFPExtFoldable(MI, PreferredFusedOpcode, DstType,
-                          MRI.getType(FpExtSrc->getOperand(1).getReg()))) {
-    unsigned Flags = MI.getFlags() & FpExtSrc->getFlags();
+    unsigned Flags = MI.getFlags() & LHSFpExtSrc->getFlags();
     MatchInfo = [=, &MI](MachineIRBuilder &B) {
-      auto FpExtX = B.buildFPExt(DstType, FpExtSrc->getOperand(1).getReg());
-      auto FpExtY = B.buildFPExt(DstType, FpExtSrc->getOperand(2).getReg());
+      auto FpExtX = B.buildFPExt(DstType, LHSFpExtSrc->getOperand(1).getReg());
+      auto FpExtY = B.buildFPExt(DstType, LHSFpExtSrc->getOperand(2).getReg());
       B.buildInstr(PreferredFusedOpcode, {MI.getOperand(0).getReg()},
                    {FpExtX.getReg(0), FpExtY.getReg(0), RHS.Reg}, Flags);
     };
     return true;
   }
 
-  // fold (fadd z, (fpext (fmul x, y))) -> (fma (fpext x), (fpext y), z)
-  // Note: Commutes FADD operands.
-  if (mi_match(RHS.Reg, MRI, m_GFPExt(m_MInstr(FpExtSrc))) &&
-      isContractableFMul(*FpExtSrc, AllowFusionGlobally) &&
-      TLI.isFPExtFoldable(MI, PreferredFusedOpcode, DstType,
-                          MRI.getType(FpExtSrc->getOperand(1).getReg()))) {
-    unsigned Flags = MI.getFlags() & FpExtSrc->getFlags();
-    MatchInfo = [=, &MI](MachineIRBuilder &B) {
-      auto FpExtX = B.buildFPExt(DstType, FpExtSrc->getOperand(1).getReg());
-      auto FpExtY = B.buildFPExt(DstType, FpExtSrc->getOperand(2).getReg());
-      B.buildInstr(PreferredFusedOpcode, {MI.getOperand(0).getReg()},
-                   {FpExtX.getReg(0), FpExtY.getReg(0), LHS.Reg}, Flags);
-    };
-    return true;
-  }
-
   return false;
 }
 

diff  --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-fma-add-ext-mul.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-fma-add-ext-mul.ll
index 375d46459c637..87d72e6e15acc 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-fma-add-ext-mul.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-fma-add-ext-mul.ll
@@ -151,3 +151,45 @@ define amdgpu_vs <6 x float> @test_6xf16_6xf32_add_ext_mul_rhs(<6 x half> inreg
     %c = fadd fast <6 x float> %z, %b
     ret <6 x float> %c
 }
+
+; This test has two fpext(fmul) operands where the first fmul has multiple uses
+; and the second has only one use. The one with fewer uses (i.e. the second one)
+; should be folded.
+define amdgpu_vs <2 x float> @test_fpext_fmul_multiple_uses(half inreg %x1, half inreg %y1, half inreg %x2, half inreg %y2, half inreg %z) {
+; GFX9-FAST-DENORM-LABEL: test_fpext_fmul_multiple_uses:
+; GFX9-FAST-DENORM:       ; %bb.0: ; %.entry
+; GFX9-FAST-DENORM-NEXT:    v_mov_b32_e32 v0, s1
+; GFX9-FAST-DENORM-NEXT:    v_mov_b32_e32 v1, s4
+; GFX9-FAST-DENORM-NEXT:    v_mac_f16_e32 v1, s0, v0
+; GFX9-FAST-DENORM-NEXT:    v_cvt_f32_f16_e32 v1, v1
+; GFX9-FAST-DENORM-NEXT:    v_mov_b32_e32 v2, s3
+; GFX9-FAST-DENORM-NEXT:    v_mul_f16_e32 v2, s2, v2
+; GFX9-FAST-DENORM-NEXT:    v_mad_mix_f32 v0, s0, v0, v2 op_sel_hi:[1,1,1]
+; GFX9-FAST-DENORM-NEXT:    ; return to shader part epilog
+;
+; GFX10-FAST-DENORM-LABEL: test_fpext_fmul_multiple_uses:
+; GFX10-FAST-DENORM:       ; %bb.0: ; %.entry
+; GFX10-FAST-DENORM-NEXT:    v_mul_f16_e64 v0, s0, s1
+; GFX10-FAST-DENORM-NEXT:    v_add_f16_e32 v1, s4, v0
+; GFX10-FAST-DENORM-NEXT:    v_fma_mix_f32 v0, s2, s3, v0 op_sel_hi:[1,1,1]
+; GFX10-FAST-DENORM-NEXT:    v_cvt_f32_f16_e32 v1, v1
+; GFX10-FAST-DENORM-NEXT:    ; return to shader part epilog
+.entry:
+    ; First fpext(fmul) operand: %mul1 has multiple uses
+    %mul1 = fmul fast half %x1, %y1
+    %ext1 = fpext half %mul1 to float
+    %add_f16 = fadd fast half %mul1, %z  ; Second use of %mul1 at f16 level
+
+    ; Second fpext(fmul) operand: %mul2 has only one use
+    %mul2 = fmul fast half %x2, %y2
+    %ext2 = fpext half %mul2 to float
+
+    ; This fadd has two fpext(fmul) operands
+    %add = fadd fast float %ext1, %ext2
+
+    %ext_add_f16 = fpext half %add_f16 to float
+    %ret0 = insertelement <2 x float> poison, float %add, i32 0
+    %ret1 = insertelement <2 x float> %ret0, float %ext_add_f16, i32 1
+
+    ret <2 x float> %ret1
+}


        


More information about the llvm-commits mailing list