[llvm] [AMDGPU] merge 16bit mov pairs in post-RA peephole (PR #208625)

Guo Chen via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 14:58:24 PDT 2026


================
@@ -0,0 +1,359 @@
+//===-- SIPostRA16BitMovFolding.cpp ------------------------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+/// \file
+/// This pass performs the post RA 16bit Mov folding
+///
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPU.h"
+#include "GCNSubtarget.h"
+#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
+#include "llvm/ADT/SetVector.h"
+#include "llvm/CodeGen/MachineDominators.h"
+#include "llvm/CodeGen/MachineFunctionPass.h"
+#include "llvm/CodeGen/MachineLoopInfo.h"
+#include "llvm/CodeGen/MachinePostDominators.h"
+#include "llvm/CodeGen/TargetSchedule.h"
+#include "llvm/Support/BranchProbability.h"
+using namespace llvm;
+
+#define DEBUG_TYPE "si-post-ra-16bit-mov-folding"
+
+namespace {
+
+class SIPostRA16BitMovFolding {
+private:
+  const SIInstrInfo *TII = nullptr;
+  const SIRegisterInfo *TRI = nullptr;
+
+  void getMovB16Info(const MachineInstr &MI, const SIRegisterInfo *TRI,
+                     MCRegister &SrcReg16, bool &SrcIsVGPR,
+                     MCRegister &SrcReg32, bool &SrcIsHi, bool &SrcIsImm,
+                     int64_t &ImmVal) const;
+
+  bool mergeSingleMovB16Pair(MachineInstr &Lo, MachineInstr &Hi,
+                             bool IsHiFirst) const;
+  bool mergeMovB16Pairs(MachineFunction &MF) const;
+
+public:
+  bool run(MachineFunction &MF);
+};
+
+class SIPostRA16BitMovFoldingLegacy : public MachineFunctionPass {
+public:
+  static char ID;
+
+  SIPostRA16BitMovFoldingLegacy() : MachineFunctionPass(ID) {}
+
+  StringRef getPassName() const override {
+    return "SI post-RA 16bit Mov Folding";
+  }
+
+  void getAnalysisUsage(AnalysisUsage &AU) const override {
+    AU.setPreservesAll();
+    MachineFunctionPass::getAnalysisUsage(AU);
+  }
+
+  bool runOnMachineFunction(MachineFunction &MF) override {
+    return SIPostRA16BitMovFolding().run(MF);
+  }
+};
+
+} // End anonymous namespace.
+
+INITIALIZE_PASS(SIPostRA16BitMovFoldingLegacy, DEBUG_TYPE,
+                "SI Post RA 16bit Mov Folding", false, false)
+
+char SIPostRA16BitMovFoldingLegacy::ID = 0;
+
+char &llvm::SIPostRA16BitMovFoldingLegacyID = SIPostRA16BitMovFoldingLegacy::ID;
+
+// Helper: extract the src operand and whether it is from the hi16 half.
+// Post-RA, both V_MOV_B16_t16_e32 and V_MOV_B16_t16_e64 use VGPR_16 dst
+// physical registers whose encoding already encodes hi/lo (IS_HI16 bit).
+void SIPostRA16BitMovFolding::getMovB16Info(
+    const MachineInstr &MI, const SIRegisterInfo *TRI, MCRegister &SrcReg16,
+    bool &SrcIsVGPR, MCRegister &SrcReg32, bool &SrcIsHi, bool &SrcIsImm,
+    int64_t &ImmVal) const {
+  SrcIsImm = false;
+  SrcIsHi = false;
+  SrcIsVGPR = false;
+  SrcReg16 = MCRegister();
+  SrcReg32 = MCRegister();
+
+  const MachineOperand *SrcOp = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
+
+  if (SrcOp->isImm()) {
+    SrcIsImm = true;
+    ImmVal = SrcOp->getImm();
+    return;
+  }
+
+  SrcReg16 = SrcOp->getReg().asMCReg();
+  SrcIsVGPR = AMDGPU::VGPR_16RegClass.contains(SrcReg16);
+  if (SrcIsVGPR) {
+    SrcIsHi = AMDGPU::isHi16Reg(SrcReg16, *TRI);
+    SrcReg32 = TRI->get32BitRegister(SrcReg16);
+  } else {
+    SrcIsHi = false;
+    SrcReg32 = SrcReg16;
+  }
+}
+
+// clang-format off
+// Try to merge a pair of v_mov_b16 instructions targeting the lo16 and hi16
+// halves of the same VGPR into a single 32-bit instruction.
+//
+// Caller guarantee the pair to be two v_mov_b16 and targets the same dst32
+//
+// Patterns:
+//   v_mov_b16 v0.h, 0        v_mov_b16 v0.l, v2.l/s2  => v_and_b32  v0,0xffff,v2/s2
+//   v_mov_b16 v0.h, 0        v_mov_b16 v0.l, v2.h     => v_lshrrev_b32 v0,16,v2
+//   v_mov_b16 v0.l, 0        v_mov_b16 v0.h, v2.l/s2  => v_lshlrev_b32 v0,16,v2/s2
+//   v_mov_b16 v0.l, 0        v_mov_b16 v0.h, v2.h     => v_and_b32  v0,0xffff0000,v2
+//   v_mov_b16 v0.l, v.x/s    v_mov_b16 v0.h, v.y/s    => v_perm_b32_e64 v0, v.x/s, v.y/s, mask
+// clang-format on
+bool SIPostRA16BitMovFolding::mergeSingleMovB16Pair(MachineInstr &Lo,
+                                                    MachineInstr &Hi,
+                                                    bool IsHiFirst) const {
+  // Lo and Hi share the same Dst32
+  MCRegister LoDst = Lo.getOperand(0).getReg().asMCReg();
+  MCRegister HiDst = Hi.getOperand(0).getReg().asMCReg();
+  MCRegister Dst32 = TRI->get32BitRegister(LoDst);
+
+  // Extract source info for Lo and Hi.
+  MCRegister LoSrc16, LoSrc32, HiSrc16, HiSrc32;
+  bool LoSrcIsHi, HiSrcIsHi, LoSrcIsImm, HiSrcIsImm, LoSrcIsVGPR, HiSrcIsVGPR;
+  int64_t LoImm = 0, HiImm = 0;
+
+  getMovB16Info(Lo, TRI, LoSrc16, LoSrcIsVGPR, LoSrc32, LoSrcIsHi, LoSrcIsImm,
+                LoImm);
+  getMovB16Info(Hi, TRI, HiSrc16, HiSrcIsVGPR, HiSrc32, HiSrcIsHi, HiSrcIsImm,
+                HiImm);
+
+  MachineInstr &FirstMI = IsHiFirst ? Hi : Lo;
+  MachineInstr &SecondMI = IsHiFirst ? Lo : Hi;
+
+  // Data Conflict counter
+  MachineBasicBlock::iterator UpperBound = SecondMI.getIterator();
+  MachineBasicBlock::iterator LowerBound = FirstMI.getIterator();
+  unsigned LoopCnt = 0, UpperBoundCnt = UINT_MAX, LowerBoundCnt = 0;
+
+  MachineBasicBlock &MBB = *Lo.getParent();
+
+  // Check that between Lo and Hi, there are no instructions that:
+  // - modify Dst32
+  // - modify LoSrc16 or HiSrc16 depending on order (data dependency)
+  // We scan from the instruction after the first mov up to (but not including)
+  // the second mov.
+  MCRegister FirstSrc16 = IsHiFirst ? HiSrc16 : LoSrc16;
+  MCRegister FirstDst16 = IsHiFirst ? HiDst : LoDst;
+  MCRegister SecondSrc16 = IsHiFirst ? LoSrc16 : HiSrc16;
+  MCRegister SecondDst16 = IsHiFirst ? LoDst : HiDst;
+  for (MachineInstr &Scan :
+       drop_begin(make_range(FirstMI.getIterator(), SecondMI.getIterator()))) {
----------------
broxigarchen wrote:

Right. Added the check

https://github.com/llvm/llvm-project/pull/208625


More information about the llvm-commits mailing list