[llvm] [AMDGPU][True16] relax d16-write-vgpr32 condition (PR #194477)
Matt Arsenault via llvm-commits
llvm-commits at lists.llvm.org
Mon May 4 09:14:13 PDT 2026
================
@@ -1629,13 +1629,52 @@ AMDGPU::Waitcnt WaitcntBrackets::determineAsyncWait(unsigned N) {
return Wait;
}
+MCPhysReg WaitcntBrackets::determineVGPR16Dependency(const MachineInstr &MI,
+ AMDGPU::InstCounterType T,
+ MCPhysReg Reg) const {
+ const TargetRegisterClass *RC = Context->TRI.getPhysRegBaseClass(Reg);
+ unsigned Size = Context->TRI.getRegSizeInBits(*RC);
+
+ if (!(Size == 16) || !Context->ST.hasD16Writes32BitVgpr())
+ return Reg;
+
+ // With D16Writes32BitVgpr, D16 Inst might clobber the whole vgpr32
+ // check dependency on the other half
+ Register Reg32 = Context->TRI.get32BitRegister(Reg);
+ Register OtherHalf = Context->TRI.getSubReg(
+ Reg32,
+ AMDGPU::isHi16Reg(Reg, Context->TRI) ? AMDGPU::lo16 : AMDGPU::hi16);
+
+ AMDGPU::Waitcnt Wait;
+ for (MCRegUnit RU : regunits(OtherHalf))
+ determineWaitForScore(T, getVMemScore(toVMEMID(RU), T), Wait);
+
+ // no wait on otherhalf
----------------
arsenm wrote:
```suggestion
// No wait on other half.
```
https://github.com/llvm/llvm-project/pull/194477
More information about the llvm-commits
mailing list