[llvm] [AMDGPU] Drop !noundef when widening sub-DWORD constant loads (PR #201085)

Arseniy Obolenskiy via llvm-commits llvm-commits at lists.llvm.org
Tue Jun 2 03:17:48 PDT 2026


https://github.com/aobolensk created https://github.com/llvm/llvm-project/pull/201085

The widened i32 load reads bytes outside the original sub-DWORD load, so new op cannot claim !noundef

>From bea71e0a8fca178c45353249cb6d4139d0390bcb Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Tue, 2 Jun 2026 12:15:48 +0200
Subject: [PATCH] [AMDGPU] Drop !noundef when widening sub-DWORD constant loads

The widened i32 load reads bytes outside the original sub-DWORD load, so new op cannot claim !noundef
---
 .../Target/AMDGPU/AMDGPUCodeGenPrepare.cpp    |  1 +
 .../AMDGPU/AMDGPULateCodeGenPrepare.cpp       |  1 +
 .../AMDGPU/amdgpu-late-codegenprepare.ll      | 28 +++++++++++++++++++
 .../AMDGPU/widen_extending_scalar_loads.ll    | 16 +++++++++++
 4 files changed, 46 insertions(+)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp b/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp
index 1bcfc1da3b84e..f4f2ebd1461f5 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp
@@ -1562,6 +1562,7 @@ bool AMDGPUCodeGenPrepareImpl::visitLoadInst(LoadInst &I) {
     Type *I32Ty = Builder.getInt32Ty();
     LoadInst *WidenLoad = Builder.CreateLoad(I32Ty, I.getPointerOperand());
     WidenLoad->copyMetadata(I);
+    WidenLoad->setMetadata(LLVMContext::MD_noundef, nullptr);
 
     // If we have range metadata, we need to convert the type, and not make
     // assumptions about the high bits.
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
index 3844e68be8e8e..01bfbcaa5ba14 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
@@ -538,6 +538,7 @@ bool AMDGPULateCodeGenPrepare::visitLoadInst(LoadInst &LI) {
   LoadInst *NewLd = IRB.CreateAlignedLoad(IRB.getInt32Ty(), NewPtr, Align(4));
   NewLd->copyMetadata(LI);
   NewLd->setMetadata(LLVMContext::MD_range, nullptr);
+  NewLd->setMetadata(LLVMContext::MD_noundef, nullptr);
 
   unsigned ShAmt = Adjust * 8;
   Value *NewVal = IRB.CreateBitCast(
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll
index 3e232bb1914f8..6300eac9f5bca 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll
@@ -149,3 +149,31 @@ bb7:
   %i8 = phi <4 x i8> [ zeroinitializer, %bb5 ], [ zeroinitializer, %bb3 ]
   br label %bb1
 }
+
+; The widened i32 load reads bytes outside the original i8 load, so !noundef
+; must not be carried over for GFX9.
+define amdgpu_kernel void @no_widen_noundef(ptr addrspace(4) align 4 %p, ptr addrspace(1) %out) {
+; GFX9-LABEL: @no_widen_noundef(
+; GFX9-NEXT:    [[TMP1:%.*]] = getelementptr i8, ptr addrspace(4) [[P:%.*]], i64 0
+; GFX9-NEXT:    [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4
+; GFX9-NEXT:    [[TMP3:%.*]] = lshr i32 [[TMP2]], 8
+; GFX9-NEXT:    [[TMP4:%.*]] = trunc i32 [[TMP3]] to i8
+; GFX9-NEXT:    [[VZ:%.*]] = zext i8 [[TMP4]] to i32
+; GFX9-NEXT:    store i32 [[VZ]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX9-NEXT:    ret void
+;
+; GFX12-LABEL: @no_widen_noundef(
+; GFX12-NEXT:    [[P1:%.*]] = getelementptr inbounds i8, ptr addrspace(4) [[P:%.*]], i64 1
+; GFX12-NEXT:    [[V:%.*]] = load i8, ptr addrspace(4) [[P1]], align 1, !noundef [[META0:![0-9]+]]
+; GFX12-NEXT:    [[VZ:%.*]] = zext i8 [[V]] to i32
+; GFX12-NEXT:    store i32 [[VZ]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX12-NEXT:    ret void
+;
+  %p1 = getelementptr inbounds i8, ptr addrspace(4) %p, i64 1
+  %v = load i8, ptr addrspace(4) %p1, align 1, !noundef !0
+  %vz = zext i8 %v to i32
+  store i32 %vz, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+!0 = !{}
diff --git a/llvm/test/CodeGen/AMDGPU/widen_extending_scalar_loads.ll b/llvm/test/CodeGen/AMDGPU/widen_extending_scalar_loads.ll
index 24c1875159f67..95b6a1a31871d 100644
--- a/llvm/test/CodeGen/AMDGPU/widen_extending_scalar_loads.ll
+++ b/llvm/test/CodeGen/AMDGPU/widen_extending_scalar_loads.ll
@@ -317,6 +317,22 @@ define amdgpu_kernel void @constant_load_i16_align4_invariant(ptr addrspace(1) %
   ret void
 }
 
+; The widened i32 load reads bytes outside the original i16 load, so !noundef
+; must not be carried over.
+define amdgpu_kernel void @constant_load_i16_align4_noundef(ptr addrspace(1) %out, ptr addrspace(4) %in) #0 {
+; OPT-LABEL: @constant_load_i16_align4_noundef(
+; OPT-NEXT:    [[TMP1:%.*]] = load i32, ptr addrspace(4) [[IN:%.*]], align 4
+; OPT-NEXT:    [[TMP2:%.*]] = trunc i32 [[TMP1]] to i16
+; OPT-NEXT:    [[EXT:%.*]] = sext i16 [[TMP2]] to i32
+; OPT-NEXT:    store i32 [[EXT]], ptr addrspace(1) [[OUT:%.*]], align 4
+; OPT-NEXT:    ret void
+;
+  %ld = load i16, ptr addrspace(4) %in, align 4, !noundef !6
+  %ext = sext i16 %ld to i32
+  store i32 %ext, ptr addrspace(1) %out
+  ret void
+}
+
 attributes #0 = { nounwind }
 
 ; OPT: !0 = !{i32 5, i32 0}



More information about the llvm-commits mailing list