[llvm] [AMDGPU] Drop !noundef when widening sub-DWORD constant loads (PR #201085)
Arseniy Obolenskiy via llvm-commits
llvm-commits at lists.llvm.org
Tue Jun 2 03:17:48 PDT 2026
https://github.com/aobolensk created https://github.com/llvm/llvm-project/pull/201085
The widened i32 load reads bytes outside the original sub-DWORD load, so new op cannot claim !noundef
>From bea71e0a8fca178c45353249cb6d4139d0390bcb Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Tue, 2 Jun 2026 12:15:48 +0200
Subject: [PATCH] [AMDGPU] Drop !noundef when widening sub-DWORD constant loads
The widened i32 load reads bytes outside the original sub-DWORD load, so new op cannot claim !noundef
---
.../Target/AMDGPU/AMDGPUCodeGenPrepare.cpp | 1 +
.../AMDGPU/AMDGPULateCodeGenPrepare.cpp | 1 +
.../AMDGPU/amdgpu-late-codegenprepare.ll | 28 +++++++++++++++++++
.../AMDGPU/widen_extending_scalar_loads.ll | 16 +++++++++++
4 files changed, 46 insertions(+)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp b/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp
index 1bcfc1da3b84e..f4f2ebd1461f5 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp
@@ -1562,6 +1562,7 @@ bool AMDGPUCodeGenPrepareImpl::visitLoadInst(LoadInst &I) {
Type *I32Ty = Builder.getInt32Ty();
LoadInst *WidenLoad = Builder.CreateLoad(I32Ty, I.getPointerOperand());
WidenLoad->copyMetadata(I);
+ WidenLoad->setMetadata(LLVMContext::MD_noundef, nullptr);
// If we have range metadata, we need to convert the type, and not make
// assumptions about the high bits.
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
index 3844e68be8e8e..01bfbcaa5ba14 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
@@ -538,6 +538,7 @@ bool AMDGPULateCodeGenPrepare::visitLoadInst(LoadInst &LI) {
LoadInst *NewLd = IRB.CreateAlignedLoad(IRB.getInt32Ty(), NewPtr, Align(4));
NewLd->copyMetadata(LI);
NewLd->setMetadata(LLVMContext::MD_range, nullptr);
+ NewLd->setMetadata(LLVMContext::MD_noundef, nullptr);
unsigned ShAmt = Adjust * 8;
Value *NewVal = IRB.CreateBitCast(
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll
index 3e232bb1914f8..6300eac9f5bca 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll
@@ -149,3 +149,31 @@ bb7:
%i8 = phi <4 x i8> [ zeroinitializer, %bb5 ], [ zeroinitializer, %bb3 ]
br label %bb1
}
+
+; The widened i32 load reads bytes outside the original i8 load, so !noundef
+; must not be carried over for GFX9.
+define amdgpu_kernel void @no_widen_noundef(ptr addrspace(4) align 4 %p, ptr addrspace(1) %out) {
+; GFX9-LABEL: @no_widen_noundef(
+; GFX9-NEXT: [[TMP1:%.*]] = getelementptr i8, ptr addrspace(4) [[P:%.*]], i64 0
+; GFX9-NEXT: [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4
+; GFX9-NEXT: [[TMP3:%.*]] = lshr i32 [[TMP2]], 8
+; GFX9-NEXT: [[TMP4:%.*]] = trunc i32 [[TMP3]] to i8
+; GFX9-NEXT: [[VZ:%.*]] = zext i8 [[TMP4]] to i32
+; GFX9-NEXT: store i32 [[VZ]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX9-NEXT: ret void
+;
+; GFX12-LABEL: @no_widen_noundef(
+; GFX12-NEXT: [[P1:%.*]] = getelementptr inbounds i8, ptr addrspace(4) [[P:%.*]], i64 1
+; GFX12-NEXT: [[V:%.*]] = load i8, ptr addrspace(4) [[P1]], align 1, !noundef [[META0:![0-9]+]]
+; GFX12-NEXT: [[VZ:%.*]] = zext i8 [[V]] to i32
+; GFX12-NEXT: store i32 [[VZ]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX12-NEXT: ret void
+;
+ %p1 = getelementptr inbounds i8, ptr addrspace(4) %p, i64 1
+ %v = load i8, ptr addrspace(4) %p1, align 1, !noundef !0
+ %vz = zext i8 %v to i32
+ store i32 %vz, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+!0 = !{}
diff --git a/llvm/test/CodeGen/AMDGPU/widen_extending_scalar_loads.ll b/llvm/test/CodeGen/AMDGPU/widen_extending_scalar_loads.ll
index 24c1875159f67..95b6a1a31871d 100644
--- a/llvm/test/CodeGen/AMDGPU/widen_extending_scalar_loads.ll
+++ b/llvm/test/CodeGen/AMDGPU/widen_extending_scalar_loads.ll
@@ -317,6 +317,22 @@ define amdgpu_kernel void @constant_load_i16_align4_invariant(ptr addrspace(1) %
ret void
}
+; The widened i32 load reads bytes outside the original i16 load, so !noundef
+; must not be carried over.
+define amdgpu_kernel void @constant_load_i16_align4_noundef(ptr addrspace(1) %out, ptr addrspace(4) %in) #0 {
+; OPT-LABEL: @constant_load_i16_align4_noundef(
+; OPT-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(4) [[IN:%.*]], align 4
+; OPT-NEXT: [[TMP2:%.*]] = trunc i32 [[TMP1]] to i16
+; OPT-NEXT: [[EXT:%.*]] = sext i16 [[TMP2]] to i32
+; OPT-NEXT: store i32 [[EXT]], ptr addrspace(1) [[OUT:%.*]], align 4
+; OPT-NEXT: ret void
+;
+ %ld = load i16, ptr addrspace(4) %in, align 4, !noundef !6
+ %ext = sext i16 %ld to i32
+ store i32 %ext, ptr addrspace(1) %out
+ ret void
+}
+
attributes #0 = { nounwind }
; OPT: !0 = !{i32 5, i32 0}
More information about the llvm-commits
mailing list