[llvm] [VirtRegMap] Coarsen lane mask when `getCoveringSubRegIndexes` fails (PR #201550)

via llvm-commits llvm-commits at lists.llvm.org
Thu Jun 4 04:23:42 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-amdgpu

Author: Igor Wodiany (IgWod)

<details>
<summary>Changes</summary>

`getCoveringSubRegIndexes` may fail when `LiveOutUndefLanes` contains bits at a finer granularity than available sub-registers. Instead of giving up, try to cover lanes with coarser sub-registers, e.g., `sub1` instead of `sub1_hi16`. This is required to prevent failures when #<!-- -->197467 gets reapplied and `hi16` SGPRs are removed from the AMDGPU backend. A minimum reproducer based on the buildbot failure is also added.

Assisted-by: Claude Code

---
Full diff: https://github.com/llvm/llvm-project/pull/201550.diff


2 Files Affected:

- (modified) llvm/lib/CodeGen/VirtRegMap.cpp (+26-5) 
- (added) llvm/test/CodeGen/AMDGPU/issue197467-virt-reg-rewrite-failure.ll (+48) 


``````````diff
diff --git a/llvm/lib/CodeGen/VirtRegMap.cpp b/llvm/lib/CodeGen/VirtRegMap.cpp
index 972bd8f550e8b..8d130af7de575 100644
--- a/llvm/lib/CodeGen/VirtRegMap.cpp
+++ b/llvm/lib/CodeGen/VirtRegMap.cpp
@@ -698,14 +698,35 @@ void VirtRegRewriter::rewrite() {
                                                           PhysReg, MI);
                 if (LiveOutUndefLanes.any()) {
                   SmallVector<unsigned, 16> CoveringIndexes;
+                  const TargetRegisterClass *RC = MRI->getRegClass(VirtReg);
 
                   // TODO: Just use one super register def if none of the lanes
                   // are needed?
-                  if (!TRI->getCoveringSubRegIndexes(MRI->getRegClass(VirtReg),
-                                                     LiveOutUndefLanes,
-                                                     CoveringIndexes))
-                    llvm_unreachable(
-                        "cannot represent required subregister defs");
+                  if (!TRI->getCoveringSubRegIndexes(RC, LiveOutUndefLanes,
+                                                     CoveringIndexes)) {
+                    // LiveOutUndefLanes may contain bits at finer granularity
+                    // than any valid subreg index covers, for example a
+                    // sub1_hi16 bit when SGPRs have no hi16 subreg. Coarsen
+                    // by promoting every partially intersecting index full
+                    // mask, then retry. For example promote sub1_hi16 to sub1.
+                    // This guarantees a coarsening but not a minimal one as a
+                    // wider index may be promoted even if a narrower one would
+                    // have sufficed.
+                    LaneBitmask CoarsenedLanes;
+                    for (unsigned Idx = 1, E = TRI->getNumSubRegIndices();
+                         Idx < E; ++Idx) {
+                      if (!TRI->isSubRegValidForRegClass(RC, Idx))
+                        continue;
+                      LaneBitmask M = TRI->getSubRegIndexLaneMask(Idx);
+                      if ((M & LiveOutUndefLanes).any())
+                        CoarsenedLanes |= M;
+                    }
+                    if (CoarsenedLanes.any() &&
+                        !TRI->getCoveringSubRegIndexes(RC, CoarsenedLanes,
+                                                       CoveringIndexes))
+                      llvm_unreachable(
+                          "cannot represent required subregister defs");
+                  }
 
                   // Try to represent the minimum needed live out def as a
                   // sequence of subregister defs.
diff --git a/llvm/test/CodeGen/AMDGPU/issue197467-virt-reg-rewrite-failure.ll b/llvm/test/CodeGen/AMDGPU/issue197467-virt-reg-rewrite-failure.ll
new file mode 100644
index 0000000000000..f126f7549db63
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/issue197467-virt-reg-rewrite-failure.ll
@@ -0,0 +1,48 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mcpu=gfx90a -O3 < %s | FileCheck %s
+
+target datalayout = "e-m:e-p:64:64-p1:64:64-p2:32:32-p3:32:32-p4:64:64-p5:32:32-p6:32:32-p7:160:256:256:32-p8:128:128:128:48-p9:192:256:256:32-i64:64-v16:16-v24:32-v32:32-v48:64-v96:128-v192:256-v256:256-v512:512-v1024:1024-v2048:2048-n32:64-S32-A5-G1-ni:7:8:9"
+target triple = "amdgcn-amd-amdhsa"
+
+define void @_ZN43LlvmLibcStrftimeTest_TimeFormatFullDateTime3RunEv() {
+; CHECK-LABEL: _ZN43LlvmLibcStrftimeTest_TimeFormatFullDateTime3RunEv:
+; CHECK:       ; %bb.0: ; %entry
+; CHECK-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; CHECK-NEXT:    s_mov_b64 s[4:5], 0x76c
+; CHECK-NEXT:    s_mov_b32 s6, 0x41200000
+; CHECK-NEXT:    s_and_b64 vcc, exec, -1
+; CHECK-NEXT:  .LBB0_1: ; %while.body.i.i.i.i384.i.i.i
+; CHECK-NEXT:    ; =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    v_cvt_f32_u32_e32 v0, s4
+; CHECK-NEXT:    s_mul_i32 s5, s4, 0xf6
+; CHECK-NEXT:    v_mul_f32_e32 v1, 0x3dcccccd, v0
+; CHECK-NEXT:    v_trunc_f32_e32 v1, v1
+; CHECK-NEXT:    v_cvt_u32_f32_e32 v2, v1
+; CHECK-NEXT:    v_mad_f32 v0, -v1, s6, v0
+; CHECK-NEXT:    v_cmp_ge_f32_e64 s[8:9], |v0|, s6
+; CHECK-NEXT:    s_cmp_lg_u64 s[8:9], 0
+; CHECK-NEXT:    v_readfirstlane_b32 s7, v2
+; CHECK-NEXT:    s_addc_u32 s7, s7, 0
+; CHECK-NEXT:    s_or_b32 s5, s5, s4
+; CHECK-NEXT:    s_or_b32 s5, s5, 1
+; CHECK-NEXT:    s_and_b32 s4, s7, 0x7ff
+; CHECK-NEXT:    v_mov_b32_e32 v0, s5
+; CHECK-NEXT:    buffer_store_byte v0, off, s[0:3], 0
+; CHECK-NEXT:    s_mov_b64 vcc, vcc
+; CHECK-NEXT:    s_cbranch_vccnz .LBB0_1
+; CHECK-NEXT:  ; %bb.2: ; %DummyReturnBlock
+; CHECK-NEXT:    s_waitcnt vmcnt(0)
+; CHECK-NEXT:    s_setpc_b64 s[30:31]
+entry:
+  br label %while.body.i.i.i.i384.i.i.i
+
+while.body.i.i.i.i384.i.i.i:                      ; preds = %while.body.i.i.i.i384.i.i.i, %entry
+  %value.addr.09.i.i.i.i386.i.i.i = phi i64 [ %div.i.i.i.i.i387.i.i.i, %while.body.i.i.i.i384.i.i.i ], [ 1900, %entry ]
+  %div.i.i.i.i.i387.i.i.i = udiv i64 %value.addr.09.i.i.i.i386.i.i.i, 10
+  %.neg165.i = mul i64 %value.addr.09.i.i.i.i386.i.i.i, 246
+  %rem.i.i.i.i.i392.i.i.decomposed.i = or i64 %.neg165.i, %value.addr.09.i.i.i.i386.i.i.i
+  %conv.i.i.i.i.i393.i.i.i = trunc i64 %rem.i.i.i.i.i392.i.i.decomposed.i to i8
+  %switch.offset.i.i.i.i394.i.i.i = or i8 %conv.i.i.i.i.i393.i.i.i, 1
+  store i8 %switch.offset.i.i.i.i394.i.i.i, ptr addrspace(5) null, align 1
+  br label %while.body.i.i.i.i384.i.i.i
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/201550


More information about the llvm-commits mailing list