[llvm] [VirtRegMap] Coarsen lane mask when `getCoveringSubRegIndexes` fails (PR #201550)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Jun 4 04:23:42 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Igor Wodiany (IgWod)
<details>
<summary>Changes</summary>
`getCoveringSubRegIndexes` may fail when `LiveOutUndefLanes` contains bits at a finer granularity than available sub-registers. Instead of giving up, try to cover lanes with coarser sub-registers, e.g., `sub1` instead of `sub1_hi16`. This is required to prevent failures when #<!-- -->197467 gets reapplied and `hi16` SGPRs are removed from the AMDGPU backend. A minimum reproducer based on the buildbot failure is also added.
Assisted-by: Claude Code
---
Full diff: https://github.com/llvm/llvm-project/pull/201550.diff
2 Files Affected:
- (modified) llvm/lib/CodeGen/VirtRegMap.cpp (+26-5)
- (added) llvm/test/CodeGen/AMDGPU/issue197467-virt-reg-rewrite-failure.ll (+48)
``````````diff
diff --git a/llvm/lib/CodeGen/VirtRegMap.cpp b/llvm/lib/CodeGen/VirtRegMap.cpp
index 972bd8f550e8b..8d130af7de575 100644
--- a/llvm/lib/CodeGen/VirtRegMap.cpp
+++ b/llvm/lib/CodeGen/VirtRegMap.cpp
@@ -698,14 +698,35 @@ void VirtRegRewriter::rewrite() {
PhysReg, MI);
if (LiveOutUndefLanes.any()) {
SmallVector<unsigned, 16> CoveringIndexes;
+ const TargetRegisterClass *RC = MRI->getRegClass(VirtReg);
// TODO: Just use one super register def if none of the lanes
// are needed?
- if (!TRI->getCoveringSubRegIndexes(MRI->getRegClass(VirtReg),
- LiveOutUndefLanes,
- CoveringIndexes))
- llvm_unreachable(
- "cannot represent required subregister defs");
+ if (!TRI->getCoveringSubRegIndexes(RC, LiveOutUndefLanes,
+ CoveringIndexes)) {
+ // LiveOutUndefLanes may contain bits at finer granularity
+ // than any valid subreg index covers, for example a
+ // sub1_hi16 bit when SGPRs have no hi16 subreg. Coarsen
+ // by promoting every partially intersecting index full
+ // mask, then retry. For example promote sub1_hi16 to sub1.
+ // This guarantees a coarsening but not a minimal one as a
+ // wider index may be promoted even if a narrower one would
+ // have sufficed.
+ LaneBitmask CoarsenedLanes;
+ for (unsigned Idx = 1, E = TRI->getNumSubRegIndices();
+ Idx < E; ++Idx) {
+ if (!TRI->isSubRegValidForRegClass(RC, Idx))
+ continue;
+ LaneBitmask M = TRI->getSubRegIndexLaneMask(Idx);
+ if ((M & LiveOutUndefLanes).any())
+ CoarsenedLanes |= M;
+ }
+ if (CoarsenedLanes.any() &&
+ !TRI->getCoveringSubRegIndexes(RC, CoarsenedLanes,
+ CoveringIndexes))
+ llvm_unreachable(
+ "cannot represent required subregister defs");
+ }
// Try to represent the minimum needed live out def as a
// sequence of subregister defs.
diff --git a/llvm/test/CodeGen/AMDGPU/issue197467-virt-reg-rewrite-failure.ll b/llvm/test/CodeGen/AMDGPU/issue197467-virt-reg-rewrite-failure.ll
new file mode 100644
index 0000000000000..f126f7549db63
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/issue197467-virt-reg-rewrite-failure.ll
@@ -0,0 +1,48 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mcpu=gfx90a -O3 < %s | FileCheck %s
+
+target datalayout = "e-m:e-p:64:64-p1:64:64-p2:32:32-p3:32:32-p4:64:64-p5:32:32-p6:32:32-p7:160:256:256:32-p8:128:128:128:48-p9:192:256:256:32-i64:64-v16:16-v24:32-v32:32-v48:64-v96:128-v192:256-v256:256-v512:512-v1024:1024-v2048:2048-n32:64-S32-A5-G1-ni:7:8:9"
+target triple = "amdgcn-amd-amdhsa"
+
+define void @_ZN43LlvmLibcStrftimeTest_TimeFormatFullDateTime3RunEv() {
+; CHECK-LABEL: _ZN43LlvmLibcStrftimeTest_TimeFormatFullDateTime3RunEv:
+; CHECK: ; %bb.0: ; %entry
+; CHECK-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; CHECK-NEXT: s_mov_b64 s[4:5], 0x76c
+; CHECK-NEXT: s_mov_b32 s6, 0x41200000
+; CHECK-NEXT: s_and_b64 vcc, exec, -1
+; CHECK-NEXT: .LBB0_1: ; %while.body.i.i.i.i384.i.i.i
+; CHECK-NEXT: ; =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: v_cvt_f32_u32_e32 v0, s4
+; CHECK-NEXT: s_mul_i32 s5, s4, 0xf6
+; CHECK-NEXT: v_mul_f32_e32 v1, 0x3dcccccd, v0
+; CHECK-NEXT: v_trunc_f32_e32 v1, v1
+; CHECK-NEXT: v_cvt_u32_f32_e32 v2, v1
+; CHECK-NEXT: v_mad_f32 v0, -v1, s6, v0
+; CHECK-NEXT: v_cmp_ge_f32_e64 s[8:9], |v0|, s6
+; CHECK-NEXT: s_cmp_lg_u64 s[8:9], 0
+; CHECK-NEXT: v_readfirstlane_b32 s7, v2
+; CHECK-NEXT: s_addc_u32 s7, s7, 0
+; CHECK-NEXT: s_or_b32 s5, s5, s4
+; CHECK-NEXT: s_or_b32 s5, s5, 1
+; CHECK-NEXT: s_and_b32 s4, s7, 0x7ff
+; CHECK-NEXT: v_mov_b32_e32 v0, s5
+; CHECK-NEXT: buffer_store_byte v0, off, s[0:3], 0
+; CHECK-NEXT: s_mov_b64 vcc, vcc
+; CHECK-NEXT: s_cbranch_vccnz .LBB0_1
+; CHECK-NEXT: ; %bb.2: ; %DummyReturnBlock
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: s_setpc_b64 s[30:31]
+entry:
+ br label %while.body.i.i.i.i384.i.i.i
+
+while.body.i.i.i.i384.i.i.i: ; preds = %while.body.i.i.i.i384.i.i.i, %entry
+ %value.addr.09.i.i.i.i386.i.i.i = phi i64 [ %div.i.i.i.i.i387.i.i.i, %while.body.i.i.i.i384.i.i.i ], [ 1900, %entry ]
+ %div.i.i.i.i.i387.i.i.i = udiv i64 %value.addr.09.i.i.i.i386.i.i.i, 10
+ %.neg165.i = mul i64 %value.addr.09.i.i.i.i386.i.i.i, 246
+ %rem.i.i.i.i.i392.i.i.decomposed.i = or i64 %.neg165.i, %value.addr.09.i.i.i.i386.i.i.i
+ %conv.i.i.i.i.i393.i.i.i = trunc i64 %rem.i.i.i.i.i392.i.i.decomposed.i to i8
+ %switch.offset.i.i.i.i394.i.i.i = or i8 %conv.i.i.i.i.i393.i.i.i, 1
+ store i8 %switch.offset.i.i.i.i394.i.i.i, ptr addrspace(5) null, align 1
+ br label %while.body.i.i.i.i384.i.i.i
+}
``````````
</details>
https://github.com/llvm/llvm-project/pull/201550
More information about the llvm-commits
mailing list