[llvm] 66ae28f - [AMDGPU] Use s_fmamk_f32 when the folded addend is inlinable (#218955)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 17 03:50:00 PDT 2026
Author: Barbara Mitic
Date: 2026-09-17T12:49:55+02:00
New Revision: 66ae28fd8f686b6b558796f70ca388523158a20c
URL: https://github.com/llvm/llvm-project/commit/66ae28fd8f686b6b558796f70ca388523158a20c
DIFF: https://github.com/llvm/llvm-project/commit/66ae28fd8f686b6b558796f70ca388523158a20c.diff
LOG: [AMDGPU] Use s_fmamk_f32 when the folded addend is inlinable (#218955)
When folding a constant into src2 of s_fmac_f32, the AK form
(s_fmaak_f32) was always selected, which puts the folded addend in the
literal slot. If that addend is an inline constant while one of the
multiplicands is not, this wastes the literal slot and still requires an
extra s_mov_b32 for the multiplicand.
Choose the instruction variant based on which operand actually needs the
literal instead. If the folded addend is inlinable and a multiplicand is
not, use the MK form (s_fmamk_f32), placing the non-inlinable
multiplicand in the literal slot and encoding the addend inline.
Added:
Modified:
llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
llvm/test/CodeGen/AMDGPU/fold-operands-scalar-fmac.mir
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp b/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
index e5490b390db08..f92b2643cbdf9 100644
--- a/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
@@ -921,6 +921,20 @@ bool SIFoldOperandsImpl::tryAddToFoldList(
return true;
}
+ // Inlineable constant might have been folded into Imm operand of fmaak or
+ // fmamk and we are trying to fold a non-inlinable constant.
+ if ((Opc == AMDGPU::S_FMAAK_F32 || Opc == AMDGPU::S_FMAMK_F32) &&
+ OpToFold.isImm()) {
+ std::optional<int64_t> ImmVal = OpToFold.getEffectiveImmVal();
+ if (ImmVal && !TII->isInlineConstant(*MI, OpNo, *ImmVal)) {
+ unsigned ImmIdx = Opc == AMDGPU::S_FMAAK_F32 ? 3 : 2;
+ MachineOperand &OpImm = MI->getOperand(ImmIdx);
+ if (!OpImm.isReg() &&
+ TII->isInlineConstant(*MI, MI->getOperand(OpNo), OpImm))
+ return tryToFoldAsFMAAKorMK();
+ }
+ }
+
// Special case for s_setreg_b32
if (OpToFold.isImm()) {
unsigned ImmOpc = 0;
diff --git a/llvm/test/CodeGen/AMDGPU/fold-operands-scalar-fmac.mir b/llvm/test/CodeGen/AMDGPU/fold-operands-scalar-fmac.mir
index a245ae7e243cb..51e929bd06714 100644
--- a/llvm/test/CodeGen/AMDGPU/fold-operands-scalar-fmac.mir
+++ b/llvm/test/CodeGen/AMDGPU/fold-operands-scalar-fmac.mir
@@ -1,5 +1,6 @@
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py
# RUN: llc -mtriple=amdgpu11.50 -verify-machineinstrs -run-pass=si-fold-operands %s -o - | FileCheck %s
+# RUN: llc -mtriple=amdgpu12.01 -verify-machineinstrs -run-pass=si-fold-operands %s -o - | FileCheck %s
---
name: fmac_fold_inlinable_src0_to_fmamk
@@ -123,18 +124,17 @@ body: |
...
---
-name: fmac_fold_inlinable_src2_to_fmaak_noninlinable_src0_to_fmamk
+name: fmac_fold_inlinable_src2_noninlinable_src0_to_fmamk
tracksRegLiveness: true
body: |
bb.0:
liveins: $sgpr0
- ; CHECK-LABEL: name: fmac_fold_inlinable_src2_to_fmaak_noninlinable_src0_to_fmamk
+ ; CHECK-LABEL: name: fmac_fold_inlinable_src2_noninlinable_src0_to_fmamk
; CHECK: liveins: $sgpr0
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
- ; CHECK-NEXT: %noninlinable:sreg_32 = S_MOV_B32 1234567890
- ; CHECK-NEXT: %fma:sreg_32 = nofpexcept S_FMAAK_F32 %noninlinable, [[COPY]], 1056964608, implicit $mode
+ ; CHECK-NEXT: %fma:sreg_32 = nofpexcept S_FMAMK_F32 [[COPY]], 1234567890, 1056964608, implicit $mode
; CHECK-NEXT: $sgpr0 = COPY %fma
%0:sreg_32 = COPY $sgpr0
%inlinable:sreg_32 = S_MOV_B32 1056964608
@@ -143,18 +143,17 @@ body: |
$sgpr0 = COPY %fma
...
---
-name: fmac_fold_inlinable_src2_to_fmaak_noninlinable_src1_to_fmamk
+name: fmac_fold_inlinable_src2_noninlinable_src1_to_fmamk
tracksRegLiveness: true
body: |
bb.0:
liveins: $sgpr0
- ; CHECK-LABEL: name: fmac_fold_inlinable_src2_to_fmaak_noninlinable_src1_to_fmamk
+ ; CHECK-LABEL: name: fmac_fold_inlinable_src2_noninlinable_src1_to_fmamk
; CHECK: liveins: $sgpr0
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
- ; CHECK-NEXT: %noninlinable:sreg_32 = S_MOV_B32 1234567890
- ; CHECK-NEXT: %fma:sreg_32 = nofpexcept S_FMAAK_F32 [[COPY]], %noninlinable, 1056964608, implicit $mode
+ ; CHECK-NEXT: %fma:sreg_32 = nofpexcept S_FMAMK_F32 [[COPY]], 1234567890, 1056964608, implicit $mode
; CHECK-NEXT: $sgpr0 = COPY %fma
%0:sreg_32 = COPY $sgpr0
%inlinable:sreg_32 = S_MOV_B32 1056964608
@@ -238,3 +237,29 @@ body: |
%fma:sreg_32 = nofpexcept S_FMAC_F32 %noninlinable, %noninlinable, %0, implicit $mode
$sgpr0 = COPY %fma
...
+---
+name: fmac_fold_shared_noninlinable_src1_to_fmamk
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $sgpr0, $sgpr1
+
+ ; CHECK-LABEL: name: fmac_fold_shared_noninlinable_src1_to_fmamk
+ ; CHECK: liveins: $sgpr0, $sgpr1
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+ ; CHECK-NEXT: [[COPY1:%[0-9]+]]:sreg_32 = COPY $sgpr1
+ ; CHECK-NEXT: %fma0:sreg_32 = nofpexcept S_FMAMK_F32 [[COPY]], 1090519040, 1056964608, implicit $mode
+ ; CHECK-NEXT: %fma1:sreg_32 = nofpexcept S_FMAMK_F32 [[COPY1]], 1090519040, 1056964608, implicit $mode
+ ; CHECK-NEXT: $sgpr0 = COPY %fma0
+ ; CHECK-NEXT: $sgpr1 = COPY %fma1
+ %0:sreg_32 = COPY $sgpr0
+ %1:sreg_32 = COPY $sgpr1
+ %inlinable:sreg_32 = S_MOV_B32 1056964608
+ %noninlinable:sreg_32 = S_MOV_B32 1090519040
+ %fma0:sreg_32 = nofpexcept S_FMAC_F32 %0, %noninlinable, %inlinable, implicit $mode
+ %fma1:sreg_32 = nofpexcept S_FMAC_F32 %1, %noninlinable, %inlinable, implicit $mode
+ $sgpr0 = COPY %fma0
+ $sgpr1 = COPY %fma1
+...
+
More information about the llvm-commits
mailing list