[llvm] [AMDGPU][True16] Fix MadMix selection (PR #205431)
Petar Avramovic via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 3 03:42:49 PDT 2026
================
@@ -7299,21 +7311,134 @@ AMDGPUInstructionSelector::selectVOP3PMadMixMods(MachineOperand &Root) const {
Register Src;
unsigned Mods;
bool Matched;
- bool NeedsWiden;
- std::tie(Src, Mods) = selectVOP3PMadMixModsImpl(Root, Matched, NeedsWiden);
+ std::tie(Src, Mods) = selectVOP3PMadMixModsImpl(Root, Matched);
+
+ if (madMixSrcNeedsWiden(Src))
+ std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg());
- MachineRegisterInfo *RegInfo = MRI;
return {{
- [=](MachineInstrBuilder &MIB) {
- Register Reg = Src;
- if (NeedsWiden)
- Reg = createVOP3PSrc32FromLo16(Src, MIB.getInstr(), *RegInfo);
- MIB.addReg(Reg);
- },
+ [=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
}};
}
+// With real true16 instructions a 16-bit value lives in a VGPR_16, which cannot
+// be used as a source of the 32-bit mix instructions. Widening it needs a
+// REG_SEQUENCE
+bool AMDGPUInstructionSelector::selectVOP3PMadMixF32(MachineInstr &I) const {
+ if (!Subtarget->useRealTrue16Insts() || !Subtarget->hasFmaMixInsts())
+ return false;
+
+ Register Dst = I.getOperand(0).getReg();
+ if (MRI->getType(Dst) != LLT::scalar(32) ||
+ RBI.getRegBank(Dst, *MRI, TRI)->getID() != AMDGPU::VGPRRegBankID)
+ return false;
+
+ struct MixSrc {
+ Register Reg;
+ int64_t Imm = 0;
+ unsigned Mods = SISrcMods::NONE;
+ bool IsImm = false;
+ bool NeedsWiden = false;
+ } Srcs[3];
+
+ const auto MatchReg = [&](MixSrc &S, MachineOperand &Op) {
----------------
petar-avramovic wrote:
iirc there is a preference to avoid use of lambdas, but this looks mostly fine. Although the NeedsWiden, could be checked in loop and build the reg sequence directly.
https://github.com/llvm/llvm-project/pull/205431
More information about the llvm-commits
mailing list