[llvm] [AMDGPU] merge 16bit mov pairs in post-RA peephole (PR #208625)
Guo Chen via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 14:58:24 PDT 2026
================
@@ -0,0 +1,359 @@
+//===-- SIPostRA16BitMovFolding.cpp ------------------------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+/// \file
+/// This pass performs the post RA 16bit Mov folding
+///
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPU.h"
+#include "GCNSubtarget.h"
+#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
+#include "llvm/ADT/SetVector.h"
+#include "llvm/CodeGen/MachineDominators.h"
+#include "llvm/CodeGen/MachineFunctionPass.h"
+#include "llvm/CodeGen/MachineLoopInfo.h"
+#include "llvm/CodeGen/MachinePostDominators.h"
+#include "llvm/CodeGen/TargetSchedule.h"
+#include "llvm/Support/BranchProbability.h"
+using namespace llvm;
+
+#define DEBUG_TYPE "si-post-ra-16bit-mov-folding"
+
+namespace {
+
+class SIPostRA16BitMovFolding {
+private:
+ const SIInstrInfo *TII = nullptr;
+ const SIRegisterInfo *TRI = nullptr;
+
+ void getMovB16Info(const MachineInstr &MI, const SIRegisterInfo *TRI,
+ MCRegister &SrcReg16, bool &SrcIsVGPR,
+ MCRegister &SrcReg32, bool &SrcIsHi, bool &SrcIsImm,
+ int64_t &ImmVal) const;
+
+ bool mergeSingleMovB16Pair(MachineInstr &Lo, MachineInstr &Hi,
+ bool IsHiFirst) const;
+ bool mergeMovB16Pairs(MachineFunction &MF) const;
+
+public:
+ bool run(MachineFunction &MF);
+};
+
+class SIPostRA16BitMovFoldingLegacy : public MachineFunctionPass {
+public:
+ static char ID;
+
+ SIPostRA16BitMovFoldingLegacy() : MachineFunctionPass(ID) {}
+
+ StringRef getPassName() const override {
+ return "SI post-RA 16bit Mov Folding";
+ }
+
+ void getAnalysisUsage(AnalysisUsage &AU) const override {
+ AU.setPreservesAll();
+ MachineFunctionPass::getAnalysisUsage(AU);
+ }
+
+ bool runOnMachineFunction(MachineFunction &MF) override {
+ return SIPostRA16BitMovFolding().run(MF);
+ }
+};
+
+} // End anonymous namespace.
+
+INITIALIZE_PASS(SIPostRA16BitMovFoldingLegacy, DEBUG_TYPE,
+ "SI Post RA 16bit Mov Folding", false, false)
+
+char SIPostRA16BitMovFoldingLegacy::ID = 0;
+
+char &llvm::SIPostRA16BitMovFoldingLegacyID = SIPostRA16BitMovFoldingLegacy::ID;
+
+// Helper: extract the src operand and whether it is from the hi16 half.
+// Post-RA, both V_MOV_B16_t16_e32 and V_MOV_B16_t16_e64 use VGPR_16 dst
+// physical registers whose encoding already encodes hi/lo (IS_HI16 bit).
+void SIPostRA16BitMovFolding::getMovB16Info(
+ const MachineInstr &MI, const SIRegisterInfo *TRI, MCRegister &SrcReg16,
+ bool &SrcIsVGPR, MCRegister &SrcReg32, bool &SrcIsHi, bool &SrcIsImm,
+ int64_t &ImmVal) const {
+ SrcIsImm = false;
+ SrcIsHi = false;
+ SrcIsVGPR = false;
+ SrcReg16 = MCRegister();
+ SrcReg32 = MCRegister();
+
+ const MachineOperand *SrcOp = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
+
+ if (SrcOp->isImm()) {
+ SrcIsImm = true;
+ ImmVal = SrcOp->getImm();
+ return;
+ }
+
+ SrcReg16 = SrcOp->getReg().asMCReg();
+ SrcIsVGPR = AMDGPU::VGPR_16RegClass.contains(SrcReg16);
+ if (SrcIsVGPR) {
+ SrcIsHi = AMDGPU::isHi16Reg(SrcReg16, *TRI);
+ SrcReg32 = TRI->get32BitRegister(SrcReg16);
+ } else {
+ SrcIsHi = false;
+ SrcReg32 = SrcReg16;
+ }
+}
+
+// clang-format off
+// Try to merge a pair of v_mov_b16 instructions targeting the lo16 and hi16
+// halves of the same VGPR into a single 32-bit instruction.
+//
+// Caller guarantee the pair to be two v_mov_b16 and targets the same dst32
+//
+// Patterns:
+// v_mov_b16 v0.h, 0 v_mov_b16 v0.l, v2.l/s2 => v_and_b32 v0,0xffff,v2/s2
+// v_mov_b16 v0.h, 0 v_mov_b16 v0.l, v2.h => v_lshrrev_b32 v0,16,v2
+// v_mov_b16 v0.l, 0 v_mov_b16 v0.h, v2.l/s2 => v_lshlrev_b32 v0,16,v2/s2
+// v_mov_b16 v0.l, 0 v_mov_b16 v0.h, v2.h => v_and_b32 v0,0xffff0000,v2
+// v_mov_b16 v0.l, v.x/s v_mov_b16 v0.h, v.y/s => v_perm_b32_e64 v0, v.x/s, v.y/s, mask
+// clang-format on
+bool SIPostRA16BitMovFolding::mergeSingleMovB16Pair(MachineInstr &Lo,
+ MachineInstr &Hi,
+ bool IsHiFirst) const {
+ // Lo and Hi share the same Dst32
+ MCRegister LoDst = Lo.getOperand(0).getReg().asMCReg();
+ MCRegister HiDst = Hi.getOperand(0).getReg().asMCReg();
+ MCRegister Dst32 = TRI->get32BitRegister(LoDst);
+
+ // Extract source info for Lo and Hi.
+ MCRegister LoSrc16, LoSrc32, HiSrc16, HiSrc32;
+ bool LoSrcIsHi, HiSrcIsHi, LoSrcIsImm, HiSrcIsImm, LoSrcIsVGPR, HiSrcIsVGPR;
+ int64_t LoImm = 0, HiImm = 0;
+
+ getMovB16Info(Lo, TRI, LoSrc16, LoSrcIsVGPR, LoSrc32, LoSrcIsHi, LoSrcIsImm,
+ LoImm);
+ getMovB16Info(Hi, TRI, HiSrc16, HiSrcIsVGPR, HiSrc32, HiSrcIsHi, HiSrcIsImm,
+ HiImm);
+
+ MachineInstr &FirstMI = IsHiFirst ? Hi : Lo;
+ MachineInstr &SecondMI = IsHiFirst ? Lo : Hi;
+
+ // Data Conflict counter
+ MachineBasicBlock::iterator UpperBound = SecondMI.getIterator();
+ MachineBasicBlock::iterator LowerBound = FirstMI.getIterator();
+ unsigned LoopCnt = 0, UpperBoundCnt = UINT_MAX, LowerBoundCnt = 0;
+
+ MachineBasicBlock &MBB = *Lo.getParent();
+
+ // Check that between Lo and Hi, there are no instructions that:
+ // - modify Dst32
+ // - modify LoSrc16 or HiSrc16 depending on order (data dependency)
+ // We scan from the instruction after the first mov up to (but not including)
+ // the second mov.
+ MCRegister FirstSrc16 = IsHiFirst ? HiSrc16 : LoSrc16;
+ MCRegister FirstDst16 = IsHiFirst ? HiDst : LoDst;
+ MCRegister SecondSrc16 = IsHiFirst ? LoSrc16 : HiSrc16;
+ MCRegister SecondDst16 = IsHiFirst ? LoDst : HiDst;
+ for (MachineInstr &Scan :
+ drop_begin(make_range(FirstMI.getIterator(), SecondMI.getIterator()))) {
----------------
broxigarchen wrote:
Right. Added the check
https://github.com/llvm/llvm-project/pull/208625
More information about the llvm-commits
mailing list