[llvm-branch-commits] [llvm] [SLP] Cancel the phantom load saving on fadd reductions that lose an fma (PR #228903)
via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Sun Oct 4 08:13:31 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-analysis
Author: Dmitry Sidorov (MrSidims)
<details>
<summary>Changes</summary>
Add TargetTransformInfo::consecutiveLoadsCoalesce, implemented by
AMDGPU, whose consecutive scalar loads already coalesce into one wide
access. On such a target cancel the load saving of a contract fadd
reduction over contract fmuls, which otherwise trades every scalar fma
for a saving that never materializes. A reassociable reduction keeps it,
as its vector fmuls fuse into the reduction.
---
<sub>Stack created with <a href="https://github.com/github/gh-stack">GitHub Stacks CLI</a> • <a href="https://gh.io/stacks-feedback">Give Feedback 💬</a></sub>
---
Patch is 109.05 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/228903.diff
10 Files Affected:
- (modified) llvm/include/llvm/Analysis/TargetTransformInfo.h (+7)
- (modified) llvm/include/llvm/Analysis/TargetTransformInfoImpl.h (+6)
- (modified) llvm/lib/Analysis/TargetTransformInfo.cpp (+7)
- (modified) llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp (+22)
- (modified) llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h (+2)
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+95-11)
- (modified) llvm/test/CodeGen/AMDGPU/slp-coalesced-loads.ll (+102-131)
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-coalesced-loads.ll (+149-31)
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-coalesced-loads.ll (+366-142)
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-fma-fusion.ll (+122-72)
``````````diff
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index a33e6f62e941ed7..9613a1679b7f965 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1946,6 +1946,13 @@ class TargetTransformInfo {
/// load/store in the given address space.
LLVM_ABI unsigned getLoadStoreVecRegBitWidth(unsigned AddrSpace) const;
+ /// \returns True if the backend coalesces \p NumElts consecutive scalar
+ /// loads of \p ElemTy in the given address space into one wider access,
+ /// \p Alignment being the best alignment known among the loads.
+ LLVM_ABI bool consecutiveLoadsCoalesce(Type *ElemTy, unsigned NumElts,
+ Align Alignment,
+ unsigned AddrSpace) const;
+
/// \returns True if the load instruction is legal to vectorize.
LLVM_ABI bool isLegalToVectorizeLoad(LoadInst *LI) const;
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index a19c122c16f20c5..418c806cc372893 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -1129,6 +1129,12 @@ class LLVM_ABI TargetTransformInfoImplBase {
return 128;
}
+ virtual bool consecutiveLoadsCoalesce(Type *ElemTy, unsigned NumElts,
+ Align Alignment,
+ unsigned AddrSpace) const {
+ return false;
+ }
+
virtual bool isLegalToVectorizeLoad(LoadInst *LI) const { return true; }
virtual bool isLegalToVectorizeStore(StoreInst *SI) const { return true; }
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index af73a615f25fa61..99ac456ee2eb053 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -1443,6 +1443,13 @@ unsigned TargetTransformInfo::getLoadStoreVecRegBitWidth(unsigned AS) const {
return TTIImpl->getLoadStoreVecRegBitWidth(AS);
}
+bool TargetTransformInfo::consecutiveLoadsCoalesce(Type *ElemTy,
+ unsigned NumElts,
+ Align Alignment,
+ unsigned AS) const {
+ return TTIImpl->consecutiveLoadsCoalesce(ElemTy, NumElts, Alignment, AS);
+}
+
bool TargetTransformInfo::isLegalToVectorizeLoad(LoadInst *LI) const {
return TTIImpl->isLegalToVectorizeLoad(LI);
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 9662be4530f7856..70ea18254700d1f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -390,6 +390,28 @@ unsigned GCNTTIImpl::getLoadStoreVecRegBitWidth(unsigned AddrSpace) const {
return 128;
}
+bool GCNTTIImpl::consecutiveLoadsCoalesce(Type *ElemTy, unsigned NumElts,
+ Align Alignment,
+ unsigned AddrSpace) const {
+ unsigned MaxBits = getLoadStoreVecRegBitWidth(AddrSpace);
+ unsigned ElemBits = DL.getTypeSizeInBits(ElemTy);
+ if (MaxBits < 64 || ElemBits % 8 != 0)
+ return false;
+ unsigned Bits = std::min(ElemBits * NumElts, MaxBits);
+ if (!isLegalToVectorizeLoadChain(Bits / 8, Alignment, AddrSpace))
+ return false;
+ if (Alignment.value() % (Bits / 8) == 0)
+ return true;
+ LLVMContext &Ctx = ElemTy->getContext();
+ unsigned VecSpeed = 0, ElemSpeed = 0;
+ if (!allowsMisalignedMemoryAccesses(Ctx, Bits, AddrSpace, Alignment,
+ &VecSpeed))
+ return false;
+ allowsMisalignedMemoryAccesses(Ctx, ElemBits, AddrSpace, Alignment,
+ &ElemSpeed);
+ return VecSpeed >= ElemSpeed;
+}
+
bool GCNTTIImpl::isLegalToVectorizeMemChain(unsigned ChainSizeInBytes,
Align Alignment,
unsigned AddrSpace) const {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
index 593aa5ebc43a26d..8bdd6d45683836e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
@@ -140,6 +140,8 @@ class GCNTTIImpl final : public BasicTTIImplBase<GCNTTIImpl> {
unsigned ChainSizeInBytes,
VectorType *VecTy) const override;
unsigned getLoadStoreVecRegBitWidth(unsigned AddrSpace) const override;
+ bool consecutiveLoadsCoalesce(Type *ElemTy, unsigned NumElts, Align Alignment,
+ unsigned AddrSpace) const override;
bool isLegalToVectorizeMemChain(unsigned ChainSizeInBytes, Align Alignment,
unsigned AddrSpace) const;
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 25953e843be4056..26a61193083608a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -473,11 +473,23 @@ class slpvectorizer::BoUpSLP {
TargetTransformInfo::TargetCostKind getCostKind() const { return CostKind; }
+ /// Which vectorized load bundles lose the load saving that never
+ /// materializes on targets whose consecutive scalar loads coalesce into the
+ /// same wide access.
+ enum class CoalescedLoadSavings {
+ /// Every bundle keeps its saving.
+ Keep,
+ /// Every clean bundle loses its saving.
+ Cancel,
+ };
+
/// Calculates the cost of the subtrees, trims non-profitable ones and returns
- /// final cost.
+ /// final cost. \p LoadSavings selects the load bundles whose saving is
+ /// cancelled.
InstructionCost
- calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals = {},
- Instruction *RdxRoot = nullptr);
+ calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
+ Instruction *RdxRoot,
+ CoalescedLoadSavings LoadSavings);
/// Finds i1 and/or nodes whose operand nodes are booleanized wide leaves
/// (one-use truncs of wide values or zero-tests of values from [0, 1]) and
@@ -2479,6 +2491,15 @@ class slpvectorizer::BoUpSLP {
const SmallDenseSet<unsigned, 8> &NodesToKeepBWs, unsigned &MaxDepthLevel,
bool &IsProfitableToDemote, bool IsTruncRoot) const;
+ /// \returns the load saving credited to the vectorized load bundle \p TE
+ /// that never materializes on a target whose consecutive scalar loads
+ /// coalesce into the same wide access, or zero if \p LoadSavings keeps it.
+ /// Only clean bundles with no reordering, reuse or bit-width reduction lose
+ /// the saving.
+ InstructionCost
+ getCoalescedLoadPhantomSaving(const TreeEntry &TE,
+ CoalescedLoadSavings LoadSavings) const;
+
/// Builds the list of reorderable operands on the edges \p Edges of the \p
/// UserTE, which allow reordering (i.e. the operands can be reordered because
/// they have only one user and reordarable).
@@ -13408,6 +13429,23 @@ InstructionCost BoUpSLP::getUnfusedFMulsPenalty(const TreeEntry &TE) const {
return Penalty;
}
+/// \returns which load bundles of a reduction over \p VL lose the saving they
+/// are credited for although their scalar loads coalesce anyway. A reduction
+/// that would lose an fma has a stake in every load bundle, and any other
+/// reduction keeps every saving. A reassociable reduction loses no fma because
+/// its vector fmuls fuse into the reduction.
+static BoUpSLP::CoalescedLoadSavings
+getReductionCoalescedLoadSavings(RecurKind RdxKind, FastMathFlags RdxFMF,
+ ArrayRef<Value *> VL) {
+ if (RdxKind != RecurKind::FAdd || !RdxFMF.allowContract() ||
+ RdxFMF.allowReassoc() || none_of(VL, [](Value *V) {
+ return match(V,
+ m_OneUse(m_AllowContract(m_FMul(m_Value(), m_Value()))));
+ }))
+ return BoUpSLP::CoalescedLoadSavings::Keep;
+ return BoUpSLP::CoalescedLoadSavings::Cancel;
+}
+
// A poor-throughput entry's real vector-vs-scalar savings (fdiv/frem/fsqrt)
// are already folded into TreeCost like any other entry, including all
// shuffle/insert/extract overhead elsewhere in the tree. So bypassing the
@@ -18556,6 +18594,43 @@ bool BoUpSLP::isTreeNotExtendable() const {
return Res;
}
+InstructionCost
+BoUpSLP::getCoalescedLoadPhantomSaving(const TreeEntry &TE,
+ CoalescedLoadSavings LoadSavings) const {
+ if (LoadSavings == CoalescedLoadSavings::Keep)
+ return 0;
+ if (!TE.hasState() || TE.isGather() || TE.getOpcode() != Instruction::Load ||
+ TE.State != TreeEntry::Vectorize || TE.getInterleaveFactor())
+ return 0;
+ if (DeletedNodes.contains(&TE) || TransformedToGatherNodes.contains(&TE))
+ return 0;
+ if (!TE.ReuseShuffleIndices.empty() || !TE.ReorderIndices.empty() ||
+ MinBWs.contains(&TE))
+ return 0;
+ if (!all_of(TE.Scalars, IsaPred<LoadInst>))
+ return 0;
+ auto *LI0 = cast<LoadInst>(TE.getMainOp());
+ Align BestAlign = LI0->getAlign();
+ for (Value *V : TE.Scalars)
+ BestAlign = std::max(BestAlign, cast<LoadInst>(V)->getAlign());
+ if (!TTI->consecutiveLoadsCoalesce(LI0->getType(), TE.Scalars.size(),
+ BestAlign, LI0->getPointerAddressSpace()))
+ return 0;
+ InstructionCost ScalarLdCost = 0;
+ for (Value *V : TE.Scalars) {
+ auto *LI = cast<LoadInst>(V);
+ ScalarLdCost +=
+ TTI->getMemoryOpCost(Instruction::Load, LI->getType(), LI->getAlign(),
+ LI->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(LI->getPointerOperand()), LI);
+ }
+ Type *VecTy = getWidenedType(LI0->getType(), TE.Scalars.size());
+ InstructionCost VecLdCost = TTI->getMemoryOpCost(
+ Instruction::Load, VecTy, LI0->getAlign(), LI0->getPointerAddressSpace(),
+ CostKind, TTI::getOperandInfo(LI0->getPointerOperand()));
+ return ScalarLdCost - VecLdCost;
+}
+
InstructionCost BoUpSLP::getSpillCost() {
// Walk the vectorizable tree from the root towards its leaves, tracking
// which vectorized operand values would be live across each tree edge
@@ -19322,9 +19397,9 @@ void BoUpSLP::detectBooleanizedNodes() {
}
}
-InstructionCost
-BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
- Instruction *RdxRoot) {
+InstructionCost BoUpSLP::calculateTreeCostAndTrimNonProfitable(
+ ArrayRef<Value *> VectorizedVals, Instruction *RdxRoot,
+ CoalescedLoadSavings LoadSavings) {
// FIXME: support buildvector of the gather nodes with struct types.
if (any_of(VectorizableTree, [&](const std::unique_ptr<TreeEntry> &TE) {
return TE->isGather() &&
@@ -19448,6 +19523,7 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
"Expected gather nodes with users only.");
InstructionCost C = getEntryCost(&TE, VectorizedVals, CheckedExtracts);
+ C += getCoalescedLoadPhantomSaving(TE, LoadSavings);
uint64_t Scale = 0;
bool CostIsFree = C == 0;
// For gather/buildvector (and split-vectorize) entries, prefer the
@@ -19650,6 +19726,7 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
};
auto RecostEntry = [&](const TreeEntry *TE) {
InstructionCost C = getEntryCost(TE, VectorizedVals, CheckedExtracts);
+ C += getCoalescedLoadPhantomSaving(*TE, LoadSavings);
if (!C.isValid() || C == 0)
return C;
uint64_t Scale = EntryToScale.lookup(TE);
@@ -29640,7 +29717,9 @@ SLPVectorizerPass::vectorizeStoreChainImpl(ArrayRef<Value *> Chain, BoUpSLP &R,
R.transformNodes();
R.computeMinimumValueSizes();
- InstructionCost TreeCost = R.calculateTreeCostAndTrimNonProfitable();
+ InstructionCost TreeCost = R.calculateTreeCostAndTrimNonProfitable(
+ /*VectorizedVals=*/{}, /*RdxRoot=*/nullptr,
+ BoUpSLP::CoalescedLoadSavings::Keep);
R.buildExternalUses();
Size = R.getCanonicalGraphSize() - R.getNumSplatSubtreeEntries();
@@ -30722,7 +30801,9 @@ bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
}
R.transformNodes();
R.computeMinimumValueSizes();
- InstructionCost TreeCost = R.calculateTreeCostAndTrimNonProfitable();
+ InstructionCost TreeCost = R.calculateTreeCostAndTrimNonProfitable(
+ /*VectorizedVals=*/{}, /*RdxRoot=*/nullptr,
+ BoUpSLP::CoalescedLoadSavings::Keep);
R.buildExternalUses();
InstructionCost Cost = R.getTreeCost(TreeCost);
@@ -33025,7 +33106,9 @@ class HorizontalReduction {
}
V.transformNodes();
V.computeMinimumValueSizes();
- InstructionCost TreeCost = V.calculateTreeCostAndTrimNonProfitable(VL);
+ InstructionCost TreeCost = V.calculateTreeCostAndTrimNonProfitable(
+ VL, /*RdxRoot=*/nullptr,
+ getReductionCoalescedLoadSavings(RdxKind, RdxFMF, VL));
// A negated slice is subtracted in the final combine, it cannot be
// accumulated lane-wise.
const bool LoopAccCandidate =
@@ -33679,8 +33762,9 @@ class HorizontalReduction {
V.transformNodes();
V.computeMinimumValueSizes();
- InstructionCost TreeCost =
- V.calculateTreeCostAndTrimNonProfitable(VL, RdxRootInst);
+ InstructionCost TreeCost = V.calculateTreeCostAndTrimNonProfitable(
+ VL, RdxRootInst,
+ getReductionCoalescedLoadSavings(RdxKind, RdxFMF, VL));
V.buildExternalUses(LocalExternallyUsedValues);
InstructionCost VectorCost, RdxOpCost;
diff --git a/llvm/test/CodeGen/AMDGPU/slp-coalesced-loads.ll b/llvm/test/CodeGen/AMDGPU/slp-coalesced-loads.ll
index 4db68adac00aad9..281221e8181fa6a 100644
--- a/llvm/test/CodeGen/AMDGPU/slp-coalesced-loads.ll
+++ b/llvm/test/CodeGen/AMDGPU/slp-coalesced-loads.ll
@@ -7,24 +7,20 @@ define float @dot8_ordered(ptr addrspace(1) noalias readonly align 4 %a, ptr add
; GFX90A-LABEL: dot8_ordered:
; GFX90A: ; %bb.0: ; %entry
; GFX90A-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX90A-NEXT: global_load_dwordx4 v[6:9], v[0:1], off
-; GFX90A-NEXT: global_load_dwordx4 v[10:13], v[0:1], off offset:16
-; GFX90A-NEXT: global_load_dwordx4 v[14:17], v[2:3], off offset:16
-; GFX90A-NEXT: global_load_dwordx4 v[18:21], v[2:3], off
-; GFX90A-NEXT: s_waitcnt vmcnt(1)
-; GFX90A-NEXT: v_pk_mul_f32 v[0:1], v[12:13], v[16:17]
+; GFX90A-NEXT: global_load_dwordx4 v[6:9], v[2:3], off
+; GFX90A-NEXT: global_load_dwordx4 v[10:13], v[0:1], off
+; GFX90A-NEXT: global_load_dwordx4 v[14:17], v[0:1], off offset:16
+; GFX90A-NEXT: global_load_dwordx4 v[18:21], v[2:3], off offset:16
+; GFX90A-NEXT: s_waitcnt vmcnt(2)
+; GFX90A-NEXT: v_fma_f32 v0, v10, v6, v4
+; GFX90A-NEXT: v_fmac_f32_e32 v0, v11, v7
+; GFX90A-NEXT: v_fmac_f32_e32 v0, v12, v8
+; GFX90A-NEXT: v_fmac_f32_e32 v0, v13, v9
; GFX90A-NEXT: s_waitcnt vmcnt(0)
-; GFX90A-NEXT: v_pk_mul_f32 v[6:7], v[6:7], v[18:19]
-; GFX90A-NEXT: v_add_f32_e32 v4, v4, v6
-; GFX90A-NEXT: v_pk_mul_f32 v[2:3], v[8:9], v[20:21]
-; GFX90A-NEXT: v_add_f32_e32 v4, v4, v7
-; GFX90A-NEXT: v_add_f32_e32 v2, v4, v2
-; GFX90A-NEXT: v_pk_mul_f32 v[8:9], v[10:11], v[14:15]
-; GFX90A-NEXT: v_add_f32_e32 v2, v2, v3
-; GFX90A-NEXT: v_add_f32_e32 v2, v2, v8
-; GFX90A-NEXT: v_add_f32_e32 v2, v2, v9
-; GFX90A-NEXT: v_add_f32_e32 v0, v2, v0
-; GFX90A-NEXT: v_add_f32_e32 v0, v0, v1
+; GFX90A-NEXT: v_fmac_f32_e32 v0, v14, v18
+; GFX90A-NEXT: v_fmac_f32_e32 v0, v15, v19
+; GFX90A-NEXT: v_fmac_f32_e32 v0, v16, v20
+; GFX90A-NEXT: v_fmac_f32_e32 v0, v17, v21
; GFX90A-NEXT: s_setpc_b64 s[30:31]
;
; GFX1030-LABEL: dot8_ordered:
@@ -58,24 +54,19 @@ define float @dot8_ordered(ptr addrspace(1) noalias readonly align 4 %a, ptr add
; GFX1250-NEXT: global_load_b128 v[18:21], v[2:3], off offset:16
; GFX1250-NEXT: s_wait_loadcnt 0x2
; GFX1250-NEXT: s_wait_xcnt 0x1
-; GFX1250-NEXT: v_pk_mul_f32 v[0:1], v[10:11], v[6:7]
-; GFX1250-NEXT: s_wait_xcnt 0x0
-; GFX1250-NEXT: v_pk_mul_f32 v[2:3], v[12:13], v[8:9]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-NEXT: v_add_f32_e32 v0, v4, v0
-; GFX1250-NEXT: v_add_f32_e32 v0, v0, v1
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
-; GFX1250-NEXT: v_add_f32_e32 v2, v0, v2
+; GFX1250-NEXT: v_fma_f32 v0, v10, v6, v4
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT: v_fmac_f32_e32 v0, v11, v7
+; GFX1250-NEXT: v_fmac_f32_e32 v0, v12, v8
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1250-NEXT: v_fmac_f32_e32 v0, v13, v9
; GFX1250-NEXT: s_wait_loadcnt 0x0
-; GFX1250-NEXT: v_pk_mul_f32 v[0:1], v[14:15], v[18:19]
-; GFX1250-NEXT: v_add_f32_e32 v2, v2, v3
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
-; GFX1250-NEXT: v_add_f32_e32 v0, v2, v0
-; GFX1250-NEXT: v_pk_mul_f32 v[2:3], v[16:17], v[20:21]
-; GFX1250-NEXT: v_add_f32_e32 v0, v0, v1
+; GFX1250-NEXT: v_fmac_f32_e32 v0, v14, v18
; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-NEXT: v_add_f32_e32 v0, v0, v2
-; GFX1250-NEXT: v_add_f32_e32 v0, v0, v3
+; GFX1250-NEXT: v_fmac_f32_e32 v0, v15, v19
+; GFX1250-NEXT: v_fmac_f32_e32 v0, v16, v20
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT: v_fmac_f32_e32 v0, v17, v21
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
entry:
%pa0 = getelementptr inbounds float, ptr addrspace(1) %a, i64 0
@@ -246,17 +237,15 @@ define float @dot_i16_window(ptr addrspace(1) %a, ptr addrspace(1) %b, float %ac
; GFX90A-NEXT: global_load_dwordx2 v[10:11], v[0:1], off
; GFX90A-NEXT: global_load_dwordx4 v[6:9], v[2:3], off
; GFX90A-NEXT: s_waitcnt vmcnt(1)
-; GFX90A-NEXT: v_cvt_f32_i32_sdwa v1, sext(v10) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1
; GFX90A-NEXT: v_cvt_f32_i32_sdwa v0, sext(v10) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_0
-; GFX90A-NEXT: v_cvt_f32_i32_sdwa v3, sext(v11) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1
+; GFX90A-NEXT: v_cvt_f32_i32_sdwa v1, sext(v10) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1
; GFX90A-NEXT: v_cvt_f32_i32_sdwa v2, sext(v11) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_0
+; GFX90A-NEXT: v_cvt_f32_i32_sdwa v3, sext(v11) dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1
; GFX90A-NEXT: s_waitcnt vmcnt(0)
-; GFX90A-NEXT: v_pk_mul_f32 v[0:1], v[0:1], v[6:7]
-; GFX90A-NEXT: v_add_f32_e32 v0, v4, v0
-; GFX90A-NEXT: v_pk_mul_f32 v[2:3], v[2:3], v[8:9]
-; GFX90A-NEXT: v_add_f32_e32 v0, v0, v1
-; GFX90A-NEXT: v_add_f32_e32 v0, v0, v2
-; GFX90A-NEXT: v_add_f32_e32 v0, v0, v3
+; GFX90A-NEXT: v_fma_f32 v0, v0, v6, v4
+; GFX90A-NEXT: v_fmac_f32_e32 v0, v1, v7
+; GFX90A-NEXT: v_fmac_f32_e32 v0, v2, v8
+; GFX90A-NEXT: v_fmac_f32_e32 v0, v3, v9
; GFX90A-NEXT: s_setpc_b64 s[30:31]
;
; GFX1030-LABEL: dot_i16_window:
@@ -1099,24 +1088,20 @@ define float @dot8_ordered_lds_a16(ptr addrspace(3) %a, ptr addrspace(3) %b, flo
; GFX90A-LABEL: dot8_ordered_lds_a16:
; GFX90A: ; %bb.0: ; %entry
; GFX90A-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX90A-NEXT: ds_read_b128 v[4:7], v0
-; GFX90A-NEXT: ds_read_b128 v[8:11], v0 offset:16
-; GFX90A-NEXT: ds_read_b128 v[12:15], v1 offset:16
-; GFX90A-NEXT: ds_read_b128 v[16:19], v1
-; GFX90A-NEXT: s_waitcnt lgkmcnt(1)
-; GFX90A-NEXT: v_pk_mul_f32 v[8:9], v[8:9], v[12:13]
+; GFX90A-NEXT: ds_read_b128 v[4:7], v1
+; GFX90A-NEXT: ds_read_b128 v[8:11], v0
+; GFX90A-NEXT: ds_read_b128 v[12:15], v0 offset:16
+; GFX90A-NEXT: ds_read_b128 v[16:19], v1 offset:16
+; GFX90A-NEXT: ...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/228903
More information about the llvm-branch-commits
mailing list