[llvm-branch-commits] [llvm] [SLP] Cancel the phantom load saving on fadd reductions that lose an fma (PR #228903)

via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Sun Oct 4 08:13:31 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-analysis

Author: Dmitry Sidorov (MrSidims)

<details>
<summary>Changes</summary>

Add TargetTransformInfo::consecutiveLoadsCoalesce, implemented by
AMDGPU, whose consecutive scalar loads already coalesce into one wide
access. On such a target cancel the load saving of a contract fadd
reduction over contract fmuls, which otherwise trades every scalar fma
for a saving that never materializes. A reassociable reduction keeps it,
as its vector fmuls fuse into the reduction.

---

<sub>Stack created with <a href="https://github.com/github/gh-stack">GitHub Stacks CLI</a> • <a href="https://gh.io/stacks-feedback">Give Feedback 💬</a></sub>

---

Patch is 109.05 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/228903.diff


10 Files Affected:

- (modified) llvm/include/llvm/Analysis/TargetTransformInfo.h (+7) 
- (modified) llvm/include/llvm/Analysis/TargetTransformInfoImpl.h (+6) 
- (modified) llvm/lib/Analysis/TargetTransformInfo.cpp (+7) 
- (modified) llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp (+22) 
- (modified) llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h (+2) 
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+95-11) 
- (modified) llvm/test/CodeGen/AMDGPU/slp-coalesced-loads.ll (+102-131) 
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-coalesced-loads.ll (+149-31) 
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-coalesced-loads.ll (+366-142) 
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-fma-fusion.ll (+122-72) 


``````````diff
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index a33e6f62e941ed7..9613a1679b7f965 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1946,6 +1946,13 @@ class TargetTransformInfo {
   /// load/store in the given address space.
   LLVM_ABI unsigned getLoadStoreVecRegBitWidth(unsigned AddrSpace) const;
 
+  /// \returns True if the backend coalesces \p NumElts consecutive scalar
+  /// loads of \p ElemTy in the given address space into one wider access,
+  /// \p Alignment being the best alignment known among the loads.
+  LLVM_ABI bool consecutiveLoadsCoalesce(Type *ElemTy, unsigned NumElts,
+                                         Align Alignment,
+                                         unsigned AddrSpace) const;
+
   /// \returns True if the load instruction is legal to vectorize.
   LLVM_ABI bool isLegalToVectorizeLoad(LoadInst *LI) const;
 
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index a19c122c16f20c5..418c806cc372893 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -1129,6 +1129,12 @@ class LLVM_ABI TargetTransformInfoImplBase {
     return 128;
   }
 
+  virtual bool consecutiveLoadsCoalesce(Type *ElemTy, unsigned NumElts,
+                                        Align Alignment,
+                                        unsigned AddrSpace) const {
+    return false;
+  }
+
   virtual bool isLegalToVectorizeLoad(LoadInst *LI) const { return true; }
 
   virtual bool isLegalToVectorizeStore(StoreInst *SI) const { return true; }
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index af73a615f25fa61..99ac456ee2eb053 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -1443,6 +1443,13 @@ unsigned TargetTransformInfo::getLoadStoreVecRegBitWidth(unsigned AS) const {
   return TTIImpl->getLoadStoreVecRegBitWidth(AS);
 }
 
+bool TargetTransformInfo::consecutiveLoadsCoalesce(Type *ElemTy,
+                                                   unsigned NumElts,
+                                                   Align Alignment,
+                                                   unsigned AS) const {
+  return TTIImpl->consecutiveLoadsCoalesce(ElemTy, NumElts, Alignment, AS);
+}
+
 bool TargetTransformInfo::isLegalToVectorizeLoad(LoadInst *LI) const {
   return TTIImpl->isLegalToVectorizeLoad(LI);
 }
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 9662be4530f7856..70ea18254700d1f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -390,6 +390,28 @@ unsigned GCNTTIImpl::getLoadStoreVecRegBitWidth(unsigned AddrSpace) const {
   return 128;
 }
 
+bool GCNTTIImpl::consecutiveLoadsCoalesce(Type *ElemTy, unsigned NumElts,
+                                          Align Alignment,
+                                          unsigned AddrSpace) const {
+  unsigned MaxBits = getLoadStoreVecRegBitWidth(AddrSpace);
+  unsigned ElemBits = DL.getTypeSizeInBits(ElemTy);
+  if (MaxBits < 64 || ElemBits % 8 != 0)
+    return false;
+  unsigned Bits = std::min(ElemBits * NumElts, MaxBits);
+  if (!isLegalToVectorizeLoadChain(Bits / 8, Alignment, AddrSpace))
+    return false;
+  if (Alignment.value() % (Bits / 8) == 0)
+    return true;
+  LLVMContext &Ctx = ElemTy->getContext();
+  unsigned VecSpeed = 0, ElemSpeed = 0;
+  if (!allowsMisalignedMemoryAccesses(Ctx, Bits, AddrSpace, Alignment,
+                                      &VecSpeed))
+    return false;
+  allowsMisalignedMemoryAccesses(Ctx, ElemBits, AddrSpace, Alignment,
+                                 &ElemSpeed);
+  return VecSpeed >= ElemSpeed;
+}
+
 bool GCNTTIImpl::isLegalToVectorizeMemChain(unsigned ChainSizeInBytes,
                                             Align Alignment,
                                             unsigned AddrSpace) const {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
index 593aa5ebc43a26d..8bdd6d45683836e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
@@ -140,6 +140,8 @@ class GCNTTIImpl final : public BasicTTIImplBase<GCNTTIImpl> {
                                 unsigned ChainSizeInBytes,
                                 VectorType *VecTy) const override;
   unsigned getLoadStoreVecRegBitWidth(unsigned AddrSpace) const override;
+  bool consecutiveLoadsCoalesce(Type *ElemTy, unsigned NumElts, Align Alignment,
+                                unsigned AddrSpace) const override;
 
   bool isLegalToVectorizeMemChain(unsigned ChainSizeInBytes, Align Alignment,
                                   unsigned AddrSpace) const;
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 25953e843be4056..26a61193083608a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -473,11 +473,23 @@ class slpvectorizer::BoUpSLP {
 
   TargetTransformInfo::TargetCostKind getCostKind() const { return CostKind; }
 
+  /// Which vectorized load bundles lose the load saving that never
+  /// materializes on targets whose consecutive scalar loads coalesce into the
+  /// same wide access.
+  enum class CoalescedLoadSavings {
+    /// Every bundle keeps its saving.
+    Keep,
+    /// Every clean bundle loses its saving.
+    Cancel,
+  };
+
   /// Calculates the cost of the subtrees, trims non-profitable ones and returns
-  /// final cost.
+  /// final cost. \p LoadSavings selects the load bundles whose saving is
+  /// cancelled.
   InstructionCost
-  calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals = {},
-                                        Instruction *RdxRoot = nullptr);
+  calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
+                                        Instruction *RdxRoot,
+                                        CoalescedLoadSavings LoadSavings);
 
   /// Finds i1 and/or nodes whose operand nodes are booleanized wide leaves
   /// (one-use truncs of wide values or zero-tests of values from [0, 1]) and
@@ -2479,6 +2491,15 @@ class slpvectorizer::BoUpSLP {
       const SmallDenseSet<unsigned, 8> &NodesToKeepBWs, unsigned &MaxDepthLevel,
       bool &IsProfitableToDemote, bool IsTruncRoot) const;
 
+  /// \returns the load saving credited to the vectorized load bundle \p TE
+  /// that never materializes on a target whose consecutive scalar loads
+  /// coalesce into the same wide access, or zero if \p LoadSavings keeps it.
+  /// Only clean bundles with no reordering, reuse or bit-width reduction lose
+  /// the saving.
+  InstructionCost
+  getCoalescedLoadPhantomSaving(const TreeEntry &TE,
+                                CoalescedLoadSavings LoadSavings) const;
+
   /// Builds the list of reorderable operands on the edges \p Edges of the \p
   /// UserTE, which allow reordering (i.e. the operands can be reordered because
   /// they have only one user and reordarable).
@@ -13408,6 +13429,23 @@ InstructionCost BoUpSLP::getUnfusedFMulsPenalty(const TreeEntry &TE) const {
   return Penalty;
 }
 
+/// \returns which load bundles of a reduction over \p VL lose the saving they
+/// are credited for although their scalar loads coalesce anyway. A reduction
+/// that would lose an fma has a stake in every load bundle, and any other
+/// reduction keeps every saving. A reassociable reduction loses no fma because
+/// its vector fmuls fuse into the reduction.
+static BoUpSLP::CoalescedLoadSavings
+getReductionCoalescedLoadSavings(RecurKind RdxKind, FastMathFlags RdxFMF,
+                                 ArrayRef<Value *> VL) {
+  if (RdxKind != RecurKind::FAdd || !RdxFMF.allowContract() ||
+      RdxFMF.allowReassoc() || none_of(VL, [](Value *V) {
+        return match(V,
+                     m_OneUse(m_AllowContract(m_FMul(m_Value(), m_Value()))));
+      }))
+    return BoUpSLP::CoalescedLoadSavings::Keep;
+  return BoUpSLP::CoalescedLoadSavings::Cancel;
+}
+
 // A poor-throughput entry's real vector-vs-scalar savings (fdiv/frem/fsqrt)
 // are already folded into TreeCost like any other entry, including all
 // shuffle/insert/extract overhead elsewhere in the tree. So bypassing the
@@ -18556,6 +18594,43 @@ bool BoUpSLP::isTreeNotExtendable() const {
   return Res;
 }
 
+InstructionCost
+BoUpSLP::getCoalescedLoadPhantomSaving(const TreeEntry &TE,
+                                       CoalescedLoadSavings LoadSavings) const {
+  if (LoadSavings == CoalescedLoadSavings::Keep)
+    return 0;
+  if (!TE.hasState() || TE.isGather() || TE.getOpcode() != Instruction::Load ||
+      TE.State != TreeEntry::Vectorize || TE.getInterleaveFactor())
+    return 0;
+  if (DeletedNodes.contains(&TE) || TransformedToGatherNodes.contains(&TE))
+    return 0;
+  if (!TE.ReuseShuffleIndices.empty() || !TE.ReorderIndices.empty() ||
+      MinBWs.contains(&TE))
+    return 0;
+  if (!all_of(TE.Scalars, IsaPred<LoadInst>))
+    return 0;
+  auto *LI0 = cast<LoadInst>(TE.getMainOp());
+  Align BestAlign = LI0->getAlign();
+  for (Value *V : TE.Scalars)
+    BestAlign = std::max(BestAlign, cast<LoadInst>(V)->getAlign());
+  if (!TTI->consecutiveLoadsCoalesce(LI0->getType(), TE.Scalars.size(),
+                                     BestAlign, LI0->getPointerAddressSpace()))
+    return 0;
+  InstructionCost ScalarLdCost = 0;
+  for (Value *V : TE.Scalars) {
+    auto *LI = cast<LoadInst>(V);
+    ScalarLdCost +=
+        TTI->getMemoryOpCost(Instruction::Load, LI->getType(), LI->getAlign(),
+                             LI->getPointerAddressSpace(), CostKind,
+                             TTI::getOperandInfo(LI->getPointerOperand()), LI);
+  }
+  Type *VecTy = getWidenedType(LI0->getType(), TE.Scalars.size());
+  InstructionCost VecLdCost = TTI->getMemoryOpCost(
+      Instruction::Load, VecTy, LI0->getAlign(), LI0->getPointerAddressSpace(),
+      CostKind, TTI::getOperandInfo(LI0->getPointerOperand()));
+  return ScalarLdCost - VecLdCost;
+}
+
 InstructionCost BoUpSLP::getSpillCost() {
   // Walk the vectorizable tree from the root towards its leaves, tracking
   // which vectorized operand values would be live across each tree edge
@@ -19322,9 +19397,9 @@ void BoUpSLP::detectBooleanizedNodes() {
   }
 }
 
-InstructionCost
-BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
-                                               Instruction *RdxRoot) {
+InstructionCost BoUpSLP::calculateTreeCostAndTrimNonProfitable(
+    ArrayRef<Value *> VectorizedVals, Instruction *RdxRoot,
+    CoalescedLoadSavings LoadSavings) {
   // FIXME: support buildvector of the gather nodes with struct types.
   if (any_of(VectorizableTree, [&](const std::unique_ptr<TreeEntry> &TE) {
         return TE->isGather() &&
@@ -19448,6 +19523,7 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
            "Expected gather nodes with users only.");
 
     InstructionCost C = getEntryCost(&TE, VectorizedVals, CheckedExtracts);
+    C += getCoalescedLoadPhantomSaving(TE, LoadSavings);
     uint64_t Scale = 0;
     bool CostIsFree = C == 0;
     // For gather/buildvector (and split-vectorize) entries, prefer the
@@ -19650,6 +19726,7 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
   };
   auto RecostEntry = [&](const TreeEntry *TE) {
     InstructionCost C = getEntryCost(TE, VectorizedVals, CheckedExtracts);
+    C += getCoalescedLoadPhantomSaving(*TE, LoadSavings);
     if (!C.isValid() || C == 0)
       return C;
     uint64_t Scale = EntryToScale.lookup(TE);
@@ -29640,7 +29717,9 @@ SLPVectorizerPass::vectorizeStoreChainImpl(ArrayRef<Value *> Chain, BoUpSLP &R,
   R.transformNodes();
   R.computeMinimumValueSizes();
 
-  InstructionCost TreeCost = R.calculateTreeCostAndTrimNonProfitable();
+  InstructionCost TreeCost = R.calculateTreeCostAndTrimNonProfitable(
+      /*VectorizedVals=*/{}, /*RdxRoot=*/nullptr,
+      BoUpSLP::CoalescedLoadSavings::Keep);
   R.buildExternalUses();
 
   Size = R.getCanonicalGraphSize() - R.getNumSplatSubtreeEntries();
@@ -30722,7 +30801,9 @@ bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
       }
       R.transformNodes();
       R.computeMinimumValueSizes();
-      InstructionCost TreeCost = R.calculateTreeCostAndTrimNonProfitable();
+      InstructionCost TreeCost = R.calculateTreeCostAndTrimNonProfitable(
+          /*VectorizedVals=*/{}, /*RdxRoot=*/nullptr,
+          BoUpSLP::CoalescedLoadSavings::Keep);
       R.buildExternalUses();
 
       InstructionCost Cost = R.getTreeCost(TreeCost);
@@ -33025,7 +33106,9 @@ class HorizontalReduction {
         }
         V.transformNodes();
         V.computeMinimumValueSizes();
-        InstructionCost TreeCost = V.calculateTreeCostAndTrimNonProfitable(VL);
+        InstructionCost TreeCost = V.calculateTreeCostAndTrimNonProfitable(
+            VL, /*RdxRoot=*/nullptr,
+            getReductionCoalescedLoadSavings(RdxKind, RdxFMF, VL));
         // A negated slice is subtracted in the final combine, it cannot be
         // accumulated lane-wise.
         const bool LoopAccCandidate =
@@ -33679,8 +33762,9 @@ class HorizontalReduction {
 
       V.transformNodes();
       V.computeMinimumValueSizes();
-      InstructionCost TreeCost =
-          V.calculateTreeCostAndTrimNonProfitable(VL, RdxRootInst);
+      InstructionCost TreeCost = V.calculateTreeCostAndTrimNonProfitable(
+          VL, RdxRootInst,
+          getReductionCoalescedLoadSavings(RdxKind, RdxFMF, VL));
       V.buildExternalUses(LocalExternallyUsedValues);
 
       InstructionCost VectorCost, RdxOpCost;
diff --git a/llvm/test/CodeGen/AMDGPU/slp-coalesced-loads.ll b/llvm/test/CodeGen/AMDGPU/slp-coalesced-loads.ll
index 4db68adac00aad9..281221e8181fa6a 100644
--- a/llvm/test/CodeGen/AMDGPU/slp-coalesced-loads.ll
+++ b/llvm/test/CodeGen/AMDGPU/slp-coalesced-loads.ll
@@ -7,24 +7,20 @@ define float @dot8_ordered(ptr addrspace(1) noalias readonly align 4 %a, ptr add
 ; GFX90A-LABEL: dot8_ordered:
 ; GFX90A:       ; %bb.0: ; %entry
 ; GFX90A-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX90A-NEXT:    global_load_dwordx4 v[6:9], v[0:1], off
-; GFX90A-NEXT:    global_load_dwordx4 v[10:13], v[0:1], off offset:16
-; GFX90A-NEXT:    global_load_dwordx4 v[14:17], v[2:3], off offset:16
-; GFX90A-NEXT:    global_load_dwordx4 v[18:21], v[2:3], off
-; GFX90A-NEXT:    s_waitcnt vmcnt(1)
-; GFX90A-NEXT:    v_pk_mul_f32 v[0:1], v[12:13], v[16:17]
+; GFX90A-NEXT:    global_load_dwordx4 v[6:9], v[2:3], off
+; GFX90A-NEXT:    global_load_dwordx4 v[10:13], v[0:1], off
+; GFX90A-NEXT:    global_load_dwordx4 v[14:17], v[0:1], off offset:16
+; GFX90A-NEXT:    global_load_dwordx4 v[18:21], v[2:3], off offset:16
+; GFX90A-NEXT:    s_waitcnt vmcnt(2)
+; GFX90A-NEXT:    v_fma_f32 v0, v10, v6, v4
+; GFX90A-NEXT:    v_fmac_f32_e32 v0, v11, v7
+; GFX90A-NEXT:    v_fmac_f32_e32 v0, v12, v8
+; GFX90A-NEXT:    v_fmac_f32_e32 v0, v13, v9
 ; GFX90A-NEXT:    s_waitcnt vmcnt(0)
-; GFX90A-NEXT:    v_pk_mul_f32 v[6:7], v[6:7], v[18:19]
-; GFX90A-NEXT:    v_add_f32_e32 v4, v4, v6
-; GFX90A-NEXT:    v_pk_mul_f32 v[2:3], v[8:9], v[20:21]
-; GFX90A-NEXT:    v_add_f32_e32 v4, v4, v7
-; GFX90A-NEXT:    v_add_f32_e32 v2, v4, v2
-; GFX90A-NEXT:    v_pk_mul_f32 v[8:9], v[10:11], v[14:15]
-; GFX90A-NEXT:    v_add_f32_e32 v2, v2, v3
-; GFX90A-NEXT:    v_add_f32_e32 v2, v2, v8
-; GFX90A-NEXT:    v_add_f32_e32 v2, v2, v9
-; GFX90A-NEXT:    v_add_f32_e32 v0, v2, v0
-; GFX90A-NEXT:    v_add_f32_e32 v0, v0, v1
+; GFX90A-NEXT:    v_fmac_f32_e32 v0, v14, v18
+; GFX90A-NEXT:    v_fmac_f32_e32 v0, v15, v19
+; GFX90A-NEXT:    v_fmac_f32_e32 v0, v16, v20
+; GFX90A-NEXT:    v_fmac_f32_e32 v0, v17, v21
 ; GFX90A-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX1030-LABEL: dot8_ordered:
@@ -58,24 +54,19 @@ define float @dot8_ordered(ptr addrspace(1) noalias readonly align 4 %a, ptr add
 ; GFX1250-NEXT:    global_load_b128 v[18:21], v[2:3], off offset:16
 ; GFX1250-NEXT:    s_wait_loadcnt 0x2
 ; GFX1250-NEXT:    s_wait_xcnt 0x1
-; GFX1250-NEXT:    v_pk_mul_f32 v[0:1], v[10:11], v[6:7]
-; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_pk_mul_f32 v[2:3], v[12:13], v[8:9]
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-NEXT:    v_add_f32_e32 v0, v4, v0
-; GFX1250-NEXT:    v_add_f32_e32 v0, v0, v1
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
-; GFX1250-NEXT:    v_add_f32_e32 v2, v0, v2
+; GFX1250-NEXT:    v_fma_f32 v0, v10, v6, v4
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_fmac_f32_e32 v0, v11, v7
+; GFX1250-NEXT:    v_fmac_f32_e32 v0, v12, v8
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_fmac_f32_e32 v0, v13, v9
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
-; GFX1250-NEXT:    v_pk_mul_f32 v[0:1], v[14:15], v[18:19]
-; GFX1250-NEXT:    v_add_f32_e32 v2, v2, v3
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
-; GFX1250-NEXT:    v_add_f32_e32 v0, v2, v0
-; GFX1250-NEXT:    v_pk_mul_f32 v[2:3], v[16:17], v[20:21]
-; GFX1250-NEXT:    v_add_f32_e32 v0, v0, v1
+; GFX1250-NEXT:    v_fmac_f32_e32 v0, v14, v18
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-NEXT:    v_add_f32_e32 v0, v0, v2
-; GFX1250-NEXT:    v_add_f32_e32 v0, v0, v3
+; GFX1250-NEXT:    v_fmac_f32_e32 v0, v15, v19
+; GFX1250-NEXT:    v_fmac_f32_e32 v0, v16, v20
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    v_fmac_f32_e32 v0, v17, v21
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
 entry:
   %pa0 = getelementptr inbounds float, ptr addrspace(1) %a, i64 0
@@ -246,17 +237,15 @@ define float @dot_i16_window(ptr addrspace(1) %a, ptr addrspace(1) %b, float %ac
 ; GFX90A-NEXT:    global_load_dwordx2 v[10:11], v[0:1], off
 ; GFX90A-NEXT:    global_load_dwordx4 v[6:9], v[2:3], off
 ; GFX90A-NEXT:    s_waitcnt vmcnt(1)
-; GFX90A-NEXT:    v_cvt_f32_i32_sdwa v1, sext(v10) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1
 ; GFX90A-NEXT:    v_cvt_f32_i32_sdwa v0, sext(v10) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_0
-; GFX90A-NEXT:    v_cvt_f32_i32_sdwa v3, sext(v11) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1
+; GFX90A-NEXT:    v_cvt_f32_i32_sdwa v1, sext(v10) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1
 ; GFX90A-NEXT:    v_cvt_f32_i32_sdwa v2, sext(v11) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_0
+; GFX90A-NEXT:    v_cvt_f32_i32_sdwa v3, sext(v11) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1
 ; GFX90A-NEXT:    s_waitcnt vmcnt(0)
-; GFX90A-NEXT:    v_pk_mul_f32 v[0:1], v[0:1], v[6:7]
-; GFX90A-NEXT:    v_add_f32_e32 v0, v4, v0
-; GFX90A-NEXT:    v_pk_mul_f32 v[2:3], v[2:3], v[8:9]
-; GFX90A-NEXT:    v_add_f32_e32 v0, v0, v1
-; GFX90A-NEXT:    v_add_f32_e32 v0, v0, v2
-; GFX90A-NEXT:    v_add_f32_e32 v0, v0, v3
+; GFX90A-NEXT:    v_fma_f32 v0, v0, v6, v4
+; GFX90A-NEXT:    v_fmac_f32_e32 v0, v1, v7
+; GFX90A-NEXT:    v_fmac_f32_e32 v0, v2, v8
+; GFX90A-NEXT:    v_fmac_f32_e32 v0, v3, v9
 ; GFX90A-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX1030-LABEL: dot_i16_window:
@@ -1099,24 +1088,20 @@ define float @dot8_ordered_lds_a16(ptr addrspace(3) %a, ptr addrspace(3) %b, flo
 ; GFX90A-LABEL: dot8_ordered_lds_a16:
 ; GFX90A:       ; %bb.0: ; %entry
 ; GFX90A-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX90A-NEXT:    ds_read_b128 v[4:7], v0
-; GFX90A-NEXT:    ds_read_b128 v[8:11], v0 offset:16
-; GFX90A-NEXT:    ds_read_b128 v[12:15], v1 offset:16
-; GFX90A-NEXT:    ds_read_b128 v[16:19], v1
-; GFX90A-NEXT:    s_waitcnt lgkmcnt(1)
-; GFX90A-NEXT:    v_pk_mul_f32 v[8:9], v[8:9], v[12:13]
+; GFX90A-NEXT:    ds_read_b128 v[4:7], v1
+; GFX90A-NEXT:    ds_read_b128 v[8:11], v0
+; GFX90A-NEXT:    ds_read_b128 v[12:15], v0 offset:16
+; GFX90A-NEXT:    ds_read_b128 v[16:19], v1 offset:16
+; GFX90A-NEXT:  ...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/228903


More information about the llvm-branch-commits mailing list