[llvm] [AMDGPU] Fix bit-packing condition in LiveRegOptimizer (PR #201520)

Steffen Larsen via llvm-commits llvm-commits at lists.llvm.org
Thu Jun 4 00:06:24 PDT 2026


https://github.com/steffenlarsen created https://github.com/llvm/llvm-project/pull/201520

This commit changes the condition for determining the eligibility for bit-packing in LiveRegOptimizer from requiring that the scalar target type is larger than the source type to instead require that the target is a multiple of it.

Fixes https://github.com/llvm/llvm-project/issues/196582.

>From 8b75f07b857b8405a19d092a1bec5f5eb65495e4 Mon Sep 17 00:00:00 2001
From: Steffen Holst Larsen <sholstla at amd.com>
Date: Thu, 4 Jun 2026 01:55:08 -0500
Subject: [PATCH] [AMDGPU] Fix bit-packing condition in LiveRegOptimizer

This commit changes the condition for determining the eligibility for
bit-packing in LiveRegOptimizer from requiring that the scalar target
type is larger than the source type to instead require that the target
is a multiple of it.

Fixes https://github.com/llvm/llvm-project/issues/196582.

Signed-off-by: Steffen Holst Larsen <sholstla at amd.com>
---
 .../AMDGPU/AMDGPULateCodeGenPrepare.cpp       |  6 ++---
 ...mdgpu-late-codegenprepare-crash-non-po2.ll | 25 +++++++++++++++++++
 2 files changed, 28 insertions(+), 3 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare-crash-non-po2.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
index b5c2b366c8e7b..54bc95653b314 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
@@ -115,10 +115,10 @@ class LiveRegOptimizer {
     const auto *TLI = ST.getTargetLowering();
 
     Type *EltTy = VTy->getElementType();
-    // If the element size is not less than the convert to scalar size, then we
-    // can't do any bit packing
+    // If the element size is not is not a multiple scalar size, then we can't
+    // do any bit packing
     if (!EltTy->isIntegerTy() ||
-        EltTy->getScalarSizeInBits() > ConvertToScalar->getScalarSizeInBits())
+        ConvertToScalar->getScalarSizeInBits() % EltTy->getScalarSizeInBits())
       return false;
 
     // Only coerce illegal types
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare-crash-non-po2.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare-crash-non-po2.ll
new file mode 100644
index 0000000000000..8735e32739d16
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare-crash-non-po2.ll
@@ -0,0 +1,25 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgcn-amd-amdhsa -passes=amdgpu-late-codegenprepare %s | FileCheck %s
+
+; Make sure we don't crash on vectors with non-power-of-2 element types.
+; The LiveRegOptimizer's convertToOptType cannot bitcast-pack these into
+; i32 registers because the element size doesn't evenly divide the target size.
+
+define void @non_po2_vector_element(ptr %p, <3 x i1> %mask) {
+; CHECK-LABEL: define void @non_po2_vector_element(
+; CHECK-SAME: ptr [[P:%.*]], <3 x i1> [[MASK:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[X:%.*]] = call <3 x i31> @llvm.masked.load.v3i31.p0(ptr align 1 [[P]], <3 x i1> [[MASK]], <3 x i31> zeroinitializer)
+; CHECK-NEXT:    br label %[[STORE:.*]]
+; CHECK:       [[STORE]]:
+; CHECK-NEXT:    call void @llvm.masked.store.v3i31.p0(<3 x i31> [[X]], ptr align 1 null, <3 x i1> [[MASK]])
+; CHECK-NEXT:    ret void
+;
+entry:
+  %x = call <3 x i31> @llvm.masked.load.v3i31.p0(ptr align 1 %p, <3 x i1> %mask, <3 x i31> zeroinitializer)
+  br label %store
+
+store:
+  call void @llvm.masked.store.v3i31.p0(<3 x i31> %x, ptr align 1 null, <3 x i1> %mask)
+  ret void
+}



More information about the llvm-commits mailing list