[llvm] [AMDGPU] Use s_fmamk_f32 when the folded addend is inlinable (PR #218955)
Barbara Mitic via llvm-commits
llvm-commits at lists.llvm.org
Wed Aug 26 08:24:00 PDT 2026
https://github.com/barbara-amd created https://github.com/llvm/llvm-project/pull/218955
When folding a constant into src2 of s_fmac_f32, the AK form (s_fmaak_f32) was always selected, which puts the folded addend in the literal slot. If that addend is an inline constant while one of the multiplicands is not, this wastes the literal slot and still requires an extra s_mov_b32 for the multiplicand.
Choose the instruction variant based on which operand actually needs the literal instead. If the folded addend is inlinable and a multiplicand is not, use the MK form (s_fmamk_f32), placing the non-inlinable multiplicand in the literal slot and encoding the addend inline.
>From 432a8a44e2bc59ad5258b56c09ecc8eb7874f376 Mon Sep 17 00:00:00 2001
From: Barbara Mitic <Barbara.Mitic at amd.com>
Date: Tue, 25 Aug 2026 16:46:25 +0200
Subject: [PATCH] [AMDGPU] SIFoldOperands: Use s_fmamk_f32 when the folded
addend is inlinable
When folding a constant into src2 of s_fmac_f32, the AK form
(s_fmaak_f32) was always selected, which puts the folded addend in the
literal slot. If that addend is an inline constant while one of the
multiplicands is not, this wastes the literal slot and still requires an
extra s_mov_b32 for the multiplicand.
Choose the instruction variant based on which operand actually needs the
literal instead. If the folded addend is inlinable and a multiplicand is
not, use the MK form (s_fmamk_f32), placing the non-inlinable
multiplicand in the literal slot and encoding the addend inline.
The instruction size is unchanged and the redundant s_mov_b32 is
eliminated.
---
llvm/lib/Target/AMDGPU/SIFoldOperands.cpp | 52 +++++++++++++++++--
.../AMDGPU/fold-operands-scalar-fmac.mir | 40 +++++++++++---
2 files changed, 81 insertions(+), 11 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp b/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
index 9d1fc8deb9356..059e98c51e306 100644
--- a/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
@@ -859,13 +859,50 @@ bool SIFoldOperandsImpl::tryAddToFoldList(
if (!OpToFold.isImm())
return false;
- const bool TryAK = OpNo == 3;
+ bool TryAK = OpNo == 3;
+
+ const MCInstrDesc &MKDesc = TII->get(AMDGPU::S_FMAMK_F32);
+ std::optional<int64_t> ImmToFold = OpToFold.getEffectiveImmVal();
+ unsigned KFromOpNo = 0;
+ int64_t KImm = 0;
+ if (TryAK && ImmToFold &&
+ TII->isInlineConstant(*ImmToFold, MKDesc.operands()[3].OperandType)) {
+ // Prefer src1: it already sits in the FMAMK K slot, so no commute.
+ for (unsigned FactorOpNo : {2, 1}) {
+ MachineOperand &Factor = MI->getOperand(FactorOpNo);
+ std::optional<int64_t> FactorImm = TII->getImmOrMaterializedImm(Factor);
+ if (!FactorImm || TII->isInlineConstant(*MI, FactorOpNo, *FactorImm))
+ continue;
+ const unsigned KeptOpNo = (FactorOpNo == 1) ? 2 : 1;
+ const MachineOperand &Kept = MI->getOperand(KeptOpNo);
+ if (Kept.isImm() ? !TII->isInlineConstant(*MI, KeptOpNo)
+ : !Kept.isReg())
+ continue;
+ // s_fmamk_f32 keeps the literal in operand 2, so commute if it sits in
+ // operand 1.
+ if (FactorOpNo == 1 &&
+ !TII->commuteInstruction(*MI, /*NewMI=*/false, 1, 2))
+ continue;
+ KFromOpNo = FactorOpNo;
+ KImm = *FactorImm;
+ TryAK = false;
+ break;
+ }
+ }
+
const unsigned NewOpc = TryAK ? AMDGPU::S_FMAAK_F32 : AMDGPU::S_FMAMK_F32;
+ // OpToFold lands in the K slot, unless a multiplicand has claimed it.
+ const unsigned FoldOpNo = (TryAK || KFromOpNo) ? 3 : 2;
+
+ // Replace the register the constant was materialized into by the literal.
+ const MachineOperand OrigSrc1 = MI->getOperand(2);
+ const bool KWasReg = KFromOpNo && OrigSrc1.isReg();
MI->setDesc(TII->get(NewOpc));
+ if (KWasReg)
+ MI->getOperand(2).ChangeToImmediate(KImm);
// We have to fold into operand which would be Imm not into OpNo.
- bool FoldAsFMAAKorMK =
- tryAddToFoldList(FoldList, MI, TryAK ? 3 : 2, OpToFold);
+ bool FoldAsFMAAKorMK = tryAddToFoldList(FoldList, MI, FoldOpNo, OpToFold);
if (FoldAsFMAAKorMK) {
// Untie Src2 of fmac.
MI->untieRegOperand(3);
@@ -886,6 +923,15 @@ bool SIFoldOperandsImpl::tryAddToFoldList(
return true;
}
MI->setDesc(TII->get(Opc));
+ if (KWasReg) {
+ MachineOperand &Src1 = MI->getOperand(2);
+ Src1.ChangeToRegister(OrigSrc1.getReg(), /*isDef=*/false, /*isImp=*/false,
+ OrigSrc1.isKill(), /*isDead=*/false,
+ OrigSrc1.isUndef());
+ Src1.setSubReg(OrigSrc1.getSubReg());
+ }
+ if (KFromOpNo == 1)
+ TII->commuteInstruction(*MI, /*NewMI=*/false, 1, 2);
return false;
};
diff --git a/llvm/test/CodeGen/AMDGPU/fold-operands-scalar-fmac.mir b/llvm/test/CodeGen/AMDGPU/fold-operands-scalar-fmac.mir
index a245ae7e243cb..0beeba8eaf575 100644
--- a/llvm/test/CodeGen/AMDGPU/fold-operands-scalar-fmac.mir
+++ b/llvm/test/CodeGen/AMDGPU/fold-operands-scalar-fmac.mir
@@ -1,5 +1,6 @@
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py
# RUN: llc -mtriple=amdgpu11.50 -verify-machineinstrs -run-pass=si-fold-operands %s -o - | FileCheck %s
+# RUN: llc -mtriple=amdgpu12.01 -verify-machineinstrs -run-pass=si-fold-operands %s -o - | FileCheck %s
---
name: fmac_fold_inlinable_src0_to_fmamk
@@ -123,18 +124,17 @@ body: |
...
---
-name: fmac_fold_inlinable_src2_to_fmaak_noninlinable_src0_to_fmamk
+name: fmac_fold_inlinable_src2_noninlinable_src0_to_fmamk
tracksRegLiveness: true
body: |
bb.0:
liveins: $sgpr0
- ; CHECK-LABEL: name: fmac_fold_inlinable_src2_to_fmaak_noninlinable_src0_to_fmamk
+ ; CHECK-LABEL: name: fmac_fold_inlinable_src2_noninlinable_src0_to_fmamk
; CHECK: liveins: $sgpr0
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
- ; CHECK-NEXT: %noninlinable:sreg_32 = S_MOV_B32 1234567890
- ; CHECK-NEXT: %fma:sreg_32 = nofpexcept S_FMAAK_F32 %noninlinable, [[COPY]], 1056964608, implicit $mode
+ ; CHECK-NEXT: %fma:sreg_32 = nofpexcept S_FMAMK_F32 [[COPY]], 1234567890, 1056964608, implicit $mode
; CHECK-NEXT: $sgpr0 = COPY %fma
%0:sreg_32 = COPY $sgpr0
%inlinable:sreg_32 = S_MOV_B32 1056964608
@@ -143,18 +143,17 @@ body: |
$sgpr0 = COPY %fma
...
---
-name: fmac_fold_inlinable_src2_to_fmaak_noninlinable_src1_to_fmamk
+name: fmac_fold_inlinable_src2_noninlinable_src1_to_fmamk
tracksRegLiveness: true
body: |
bb.0:
liveins: $sgpr0
- ; CHECK-LABEL: name: fmac_fold_inlinable_src2_to_fmaak_noninlinable_src1_to_fmamk
+ ; CHECK-LABEL: name: fmac_fold_inlinable_src2_noninlinable_src1_to_fmamk
; CHECK: liveins: $sgpr0
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
- ; CHECK-NEXT: %noninlinable:sreg_32 = S_MOV_B32 1234567890
- ; CHECK-NEXT: %fma:sreg_32 = nofpexcept S_FMAAK_F32 [[COPY]], %noninlinable, 1056964608, implicit $mode
+ ; CHECK-NEXT: %fma:sreg_32 = nofpexcept S_FMAMK_F32 [[COPY]], 1234567890, 1056964608, implicit $mode
; CHECK-NEXT: $sgpr0 = COPY %fma
%0:sreg_32 = COPY $sgpr0
%inlinable:sreg_32 = S_MOV_B32 1056964608
@@ -238,3 +237,28 @@ body: |
%fma:sreg_32 = nofpexcept S_FMAC_F32 %noninlinable, %noninlinable, %0, implicit $mode
$sgpr0 = COPY %fma
...
+---
+name: fmac_fold_shared_noninlinable_src1_to_fmamk
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $sgpr0, $sgpr1
+
+ ; CHECK-LABEL: name: fmac_fold_shared_noninlinable_src1_to_fmamk
+ ; CHECK: liveins: $sgpr0, $sgpr1
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+ ; CHECK-NEXT: [[COPY1:%[0-9]+]]:sreg_32 = COPY $sgpr1
+ ; CHECK-NEXT: %fma0:sreg_32 = nofpexcept S_FMAMK_F32 [[COPY]], 1090519040, 1056964608, implicit $mode
+ ; CHECK-NEXT: %fma1:sreg_32 = nofpexcept S_FMAMK_F32 [[COPY1]], 1090519040, 1056964608, implicit $mode
+ ; CHECK-NEXT: $sgpr0 = COPY %fma0
+ ; CHECK-NEXT: $sgpr1 = COPY %fma1
+ %0:sreg_32 = COPY $sgpr0
+ %1:sreg_32 = COPY $sgpr1
+ %inlinable:sreg_32 = S_MOV_B32 1056964608
+ %noninlinable:sreg_32 = S_MOV_B32 1090519040
+ %fma0:sreg_32 = nofpexcept S_FMAC_F32 %0, %noninlinable, %inlinable, implicit $mode
+ %fma1:sreg_32 = nofpexcept S_FMAC_F32 %1, %noninlinable, %inlinable, implicit $mode
+ $sgpr0 = COPY %fma0
+ $sgpr1 = COPY %fma1
+...
More information about the llvm-commits
mailing list