[llvm] 2592df6 - [NFCI][AMDGPU] Change foldOperand to return changed (#208352)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Jul 8 19:11:35 PDT 2026
Author: Shilei Tian
Date: 2026-07-08T22:11:30-04:00
New Revision: 2592df6cb04e72f548685a152cf93b46b30cee21
URL: https://github.com/llvm/llvm-project/commit/2592df6cb04e72f548685a152cf93b46b30cee21
DIFF: https://github.com/llvm/llvm-project/commit/2592df6cb04e72f548685a152cf93b46b30cee21.diff
LOG: [NFCI][AMDGPU] Change foldOperand to return changed (#208352)
Added:
Modified:
llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp b/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
index ee20f138f245f..3fdbb74eb3342 100644
--- a/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
@@ -238,7 +238,7 @@ class SIFoldOperandsImpl {
bool tryToFoldACImm(const FoldableDef &OpToFold, MachineInstr *UseMI,
unsigned UseOpIdx,
SmallVectorImpl<FoldCandidate> &FoldList) const;
- void foldOperand(FoldableDef OpToFold, MachineInstr *UseMI, int UseOpIdx,
+ bool foldOperand(FoldableDef OpToFold, MachineInstr *UseMI, int UseOpIdx,
SmallVectorImpl<FoldCandidate> &FoldList,
SmallVectorImpl<MachineInstr *> &CopiesToReplace) const;
@@ -1190,24 +1190,25 @@ bool SIFoldOperandsImpl::tryToFoldACImm(
return false;
}
-void SIFoldOperandsImpl::foldOperand(
+bool SIFoldOperandsImpl::foldOperand(
FoldableDef OpToFold, MachineInstr *UseMI, int UseOpIdx,
SmallVectorImpl<FoldCandidate> &FoldList,
SmallVectorImpl<MachineInstr *> &CopiesToReplace) const {
+ bool Changed = false;
const MachineOperand *UseOp = &UseMI->getOperand(UseOpIdx);
if (!isUseSafeToFold(*UseMI, *UseOp))
- return;
+ return Changed;
// FIXME: Fold operands with subregs.
if (UseOp->isReg() && OpToFold.isReg()) {
if (UseOp->isImplicit())
- return;
+ return Changed;
// Allow folding from SGPRs to 16-bit VGPRs.
if (UseOp->getSubReg() != AMDGPU::NoSubRegister &&
(UseOp->getSubReg() != AMDGPU::lo16 ||
!TRI->isSGPRReg(*MRI, OpToFold.getReg())))
- return;
+ return Changed;
}
// Special case for REG_SEQUENCE: We can't fold literals into
@@ -1239,6 +1240,7 @@ void SIFoldOperandsImpl::foldOperand(
if (tryFoldRegSeqSplat(RSUseMI, OpNo, SplatVal, SplatRC)) {
FoldableDef SplatDef(SplatVal, SplatRC);
appendFoldCandidate(FoldList, RSUseMI, OpNo, SplatDef);
+ Changed = true;
continue;
}
}
@@ -1249,15 +1251,15 @@ void SIFoldOperandsImpl::foldOperand(
// FIXME: We should avoid recursing here. There should be a cleaner split
// between the in-place mutations and adding to the fold list.
- foldOperand(OpToFold, RSUseMI, RSUseMI->getOperandNo(RSUse), FoldList,
- CopiesToReplace);
+ Changed |= foldOperand(OpToFold, RSUseMI, RSUseMI->getOperandNo(RSUse),
+ FoldList, CopiesToReplace);
}
- return;
+ return Changed;
}
if (tryToFoldACImm(OpToFold, UseMI, UseOpIdx, FoldList))
- return;
+ return true;
if (frameIndexMayFold(*UseMI, UseOpIdx, OpToFold)) {
// Verify that this is a stack access.
@@ -1266,14 +1268,14 @@ void SIFoldOperandsImpl::foldOperand(
if (TII->isMUBUF(*UseMI)) {
if (TII->getNamedOperand(*UseMI, AMDGPU::OpName::srsrc)->getReg() !=
MFI->getScratchRSrcReg())
- return;
+ return Changed;
// Ensure this is either relative to the current frame or the current
// wave.
MachineOperand &SOff =
*TII->getNamedOperand(*UseMI, AMDGPU::OpName::soffset);
if (!SOff.isImm() || SOff.getImm() != 0)
- return;
+ return Changed;
}
const unsigned Opc = UseMI->getOpcode();
@@ -1285,7 +1287,7 @@ void SIFoldOperandsImpl::foldOperand(
TII->getNamedOperand(*UseMI, AMDGPU::OpName::cpol)->getImm();
if ((CPol & AMDGPU::CPol::SCAL) &&
!AMDGPU::supportsScaleOffset(*TII, NewOpc))
- return;
+ return Changed;
UseMI->setDesc(TII->get(NewOpc));
}
@@ -1294,7 +1296,7 @@ void SIFoldOperandsImpl::foldOperand(
// safe to fold the addressing mode, even pre-GFX9.
UseMI->getOperand(UseOpIdx).ChangeToFrameIndex(OpToFold.getFI());
- return;
+ return true;
}
bool FoldingImmLike =
@@ -1312,7 +1314,7 @@ void SIFoldOperandsImpl::foldOperand(
// so would interfere with the register coalescer's logic which would avoid
// redundant initializations.
if (DestReg.isPhysical() && SrcRC->contains(DestReg))
- return;
+ return Changed;
const TargetRegisterClass *DestRC = TRI->getRegClassForReg(*MRI, DestReg);
// In order to fold immediates into copies, we need to change the copy to a
@@ -1384,12 +1386,13 @@ void SIFoldOperandsImpl::foldOperand(
UseOp = &UseMI->getOperand(UseOpIdx);
}
CopiesToReplace.push_back(UseMI);
+ Changed = true;
break;
}
// We failed to replace the copy, so give up.
if (UseMI->getOpcode() == AMDGPU::COPY)
- return;
+ return Changed;
} else {
if (UseMI->isCopy() && OpToFold.isReg() &&
@@ -1438,11 +1441,12 @@ void SIFoldOperandsImpl::foldOperand(
UseMI->getOperand(1).setIsKill(false);
CopiesToReplace.push_back(UseMI);
OpToFold.OpToFold->setIsKill(false);
+ Changed = true;
// Remove kill flags as kills may now be out of order with uses.
MRI->clearKillFlags(UseReg);
if (foldCopyToAGPRRegSequence(UseMI))
- return;
+ return true;
}
unsigned UseOpc = UseMI->getOpcode();
@@ -1458,7 +1462,7 @@ void SIFoldOperandsImpl::foldOperand(
if (execMayBeModifiedBeforeUse(*MRI,
UseMI->getOperand(UseOpIdx).getReg(),
*OpToFold.DefMI, *UseMI))
- return;
+ return Changed;
UseMI->setDesc(TII->get(AMDGPU::S_MOV_B32));
UseMI->clearFlag(MachineInstr::NoConvergent);
@@ -1475,14 +1479,14 @@ void SIFoldOperandsImpl::foldOperand(
OpToFold.OpToFold->getTargetFlags());
}
UseMI->removeOperand(2); // Remove exec read (or src1 for readlane)
- return;
+ return true;
}
if (OpToFold.isReg() && TRI->isSGPRReg(*MRI, OpToFold.getReg())) {
if (execMayBeModifiedBeforeUse(*MRI,
UseMI->getOperand(UseOpIdx).getReg(),
*OpToFold.DefMI, *UseMI))
- return;
+ return Changed;
// %vgpr = COPY %sgpr0
// %sgpr1 = V_READFIRSTLANE_B32 %vgpr
@@ -1494,7 +1498,7 @@ void SIFoldOperandsImpl::foldOperand(
UseMI->getOperand(1).setIsKill(false);
UseMI->removeOperand(2); // Remove exec read (or src1 for readlane)
UseMI->clearFlag(MachineInstr::NoConvergent);
- return;
+ return true;
}
}
@@ -1504,14 +1508,15 @@ void SIFoldOperandsImpl::foldOperand(
// don't have defined register classes.
if (UseDesc.isVariadic() || UseOp->isImplicit() ||
UseDesc.operands()[UseOpIdx].RegClass == -1)
- return;
+ return Changed;
}
// FIXME: We could try to change the instruction from 64-bit to 32-bit
// to enable more folding opportunities. The shrink operands pass
// already does this.
- tryAddToFoldList(FoldList, UseMI, UseOpIdx, OpToFold);
+ Changed |= tryAddToFoldList(FoldList, UseMI, UseOpIdx, OpToFold);
+ return Changed;
}
static bool evalBinaryInstruction(unsigned Opcode, int32_t &Result,
@@ -1825,8 +1830,8 @@ bool SIFoldOperandsImpl::foldInstOperand(MachineInstr &MI,
MachineInstr *UseMI = U->getParent();
FoldableDef SubOpToFold = OpToFold.getWithSubReg(*TRI, U->getSubReg());
- foldOperand(SubOpToFold, UseMI, UseMI->getOperandNo(U), FoldList,
- CopiesToReplace);
+ Changed |= foldOperand(SubOpToFold, UseMI, UseMI->getOperandNo(U), FoldList,
+ CopiesToReplace);
}
if (CopiesToReplace.empty() && FoldList.empty())
More information about the llvm-commits
mailing list