[llvm-branch-commits] [llvm] Reland "[AMDGPU] PromoteAlloca: flatten homogeneous structs to vectors" (#221058) (PR #223790)
Domenic Nutile via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Tue Sep 22 09:50:14 PDT 2026
https://github.com/saxlungs updated https://github.com/llvm/llvm-project/pull/223790
>From f5d78dfd04469a96a2d77624864de6605150ba89 Mon Sep 17 00:00:00 2001
From: Domenic Nutile <domenic.nutile at gmail.com>
Date: Tue, 15 Sep 2026 15:16:42 -0400
Subject: [PATCH] Reland "[AMDGPU] PromoteAlloca: flatten homogeneous structs
to vectors" (#221058)
This relands #217055
The original commit revealed a latent issue in eliminateFrameIndex in
SIRegisterInfo where SCC can be clobbered before reading it on
gfx900/gfx90a. This change itself has no known issues.
---
.../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 29 +++-
.../AMDGPU/eliminate-frame-index-select.ll | 7 +-
.../promote-alloca-homogeneous-struct.ll | 152 ++++++++++++++++++
3 files changed, 181 insertions(+), 7 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 4f54ca98bd5fcb..a1936dfba11312 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -912,6 +912,27 @@ static BasicBlock::iterator skipToNonAllocaInsertPt(BasicBlock &BB,
return I;
}
+/// Peel nested aggregates down to a single uniform element type, multiplying
+/// NumElems by the element count of each layer peeled.
+static Type *peelAggregateToElementType(Type *Ty, uint64_t &NumElems) {
+ while (true) {
+ if (auto *ArrayTy = dyn_cast<ArrayType>(Ty)) {
+ NumElems *= ArrayTy->getNumElements();
+ Ty = ArrayTy->getElementType();
+ continue;
+ }
+
+ auto *StructTy = dyn_cast<StructType>(Ty);
+ if (!StructTy || !StructTy->containsHomogeneousTypes())
+ break;
+
+ NumElems *= StructTy->getNumElements();
+ Ty = StructTy->getElementType(0);
+ }
+
+ return Ty;
+}
+
FixedVectorType *
AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
if (DisablePromoteAllocaToVector) {
@@ -920,13 +941,9 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
}
auto *VectorTy = dyn_cast<FixedVectorType>(AllocaTy);
- if (auto *ArrayTy = dyn_cast<ArrayType>(AllocaTy)) {
+ if (AllocaTy->isAggregateType()) {
uint64_t NumElems = 1;
- Type *ElemTy;
- do {
- NumElems *= ArrayTy->getNumElements();
- ElemTy = ArrayTy->getElementType();
- } while ((ArrayTy = dyn_cast<ArrayType>(ElemTy)));
+ Type *ElemTy = peelAggregateToElementType(AllocaTy, NumElems);
// Check for array of vectors
auto *InnerVectorTy = dyn_cast<FixedVectorType>(ElemTy);
diff --git a/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll b/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
index d7caf2293d5600..5e1514ae7de9e8 100644
--- a/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
@@ -1,5 +1,10 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgpu10.30 < %s | FileCheck %s
+; RUN: llc -mtriple=amdgpu10.30 -disable-promote-alloca-to-vector < %s | FileCheck %s
+
+; %struct.wobble is homogeneous and flattens to <3 x float>, so the alloca is
+; promoted to a vector and no frame index is left to eliminate.
+; -disable-promote-alloca-to-vector keeps it on the stack so this still covers
+; the case it was written for.
%struct.wobble = type { %struct.quux }
%struct.quux = type { float, float, float }
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
new file mode 100644
index 00000000000000..152c114e291f29
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
@@ -0,0 +1,152 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu7.00-amd-amdhsa -passes=sroa,amdgpu-promote-alloca < %s | FileCheck %s
+
+%wrapper = type { [1 x i32] }
+%pair = type { i32, i32 }
+%nested = type { %pair, %pair }
+%simple_struct = type { i32, i32, i32, i32 }
+%wobble = type { { float, float, float } }
+
+; A single-field wrapper struct: [4 x { [1 x i32] }] is really <4 x i32>.
+define amdgpu_kernel void @wrapper_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @wrapper_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 11, i32 0
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x i32> [[TMP1]], i32 22, i32 1
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 33, i32 2
+; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 44, i32 3
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <4 x i32> [[TMP4]], i32 [[IDX]]
+; CHECK-NEXT: store i32 [[TMP5]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+ %alloca = alloca [4 x %wrapper], align 16, addrspace(5)
+ %p0 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 0
+ store i32 11, ptr addrspace(5) %p0, align 4
+ %p1 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 1
+ store i32 22, ptr addrspace(5) %p1, align 4
+ %p2 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 2
+ store i32 33, ptr addrspace(5) %p2, align 4
+ %p3 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 3
+ store i32 44, ptr addrspace(5) %p3, align 4
+ %gep = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 %idx
+ %load = load i32, ptr addrspace(5) %gep, align 4
+ store i32 %load, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; A homogeneous multi-field struct nest: [1 x { {i32,i32}, {i32,i32} }] is <4 x i32>.
+define amdgpu_kernel void @homogeneous_nested_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @homogeneous_nested_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 11, i32 0
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x i32> [[TMP1]], i32 22, i32 1
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 33, i32 2
+; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 44, i32 3
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <4 x i32> [[TMP4]], i32 [[IDX]]
+; CHECK-NEXT: store i32 [[TMP5]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+ %alloca = alloca [1 x %nested], align 16, addrspace(5)
+ %p0 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 0, i32 0
+ store i32 11, ptr addrspace(5) %p0, align 4
+ %p1 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 0, i32 1
+ store i32 22, ptr addrspace(5) %p1, align 4
+ %p2 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 1, i32 0
+ store i32 33, ptr addrspace(5) %p2, align 4
+ %p3 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 1, i32 1
+ store i32 44, ptr addrspace(5) %p3, align 4
+ %gep = getelementptr i32, ptr addrspace(5) %alloca, i32 %idx
+ %load = load i32, ptr addrspace(5) %gep, align 4
+ store i32 %load, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; The alloca does not have to be an array: a homogeneous struct on its own is
+; flattened the same way. { i32, i32, i32, i32 } is <4 x i32>.
+define amdgpu_kernel void @direct_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @direct_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 11, i32 0
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x i32> [[TMP1]], i32 22, i32 1
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 33, i32 2
+; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 44, i32 3
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <4 x i32> [[TMP4]], i32 [[IDX]]
+; CHECK-NEXT: store i32 [[TMP5]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+ %alloca = alloca %simple_struct, align 16, addrspace(5)
+ %p0 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 0
+ store i32 11, ptr addrspace(5) %p0, align 4
+ %p1 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 1
+ store i32 22, ptr addrspace(5) %p1, align 4
+ %p2 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 2
+ store i32 33, ptr addrspace(5) %p2, align 4
+ %p3 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 3
+ store i32 44, ptr addrspace(5) %p3, align 4
+ %gep = getelementptr i32, ptr addrspace(5) %alloca, i32 %idx
+ %load = load i32, ptr addrspace(5) %gep, align 4
+ store i32 %load, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; A direct struct wrapping a homogeneous struct of scalars: { {float,float,float} }
+; is <3 x float>
+define amdgpu_kernel void @direct_wrapper_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @direct_wrapper_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <3 x float> poison
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <3 x float> [[ALLOCA]], float 1.000000e+00, i32 0
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <3 x float> [[TMP1]], float 2.000000e+00, i32 1
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <3 x float> [[TMP2]], float 3.000000e+00, i32 2
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <3 x float> [[TMP3]], i32 [[IDX]]
+; CHECK-NEXT: store float [[TMP4]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+ %alloca = alloca %wobble, align 16, addrspace(5)
+ %p0 = getelementptr %wobble, ptr addrspace(5) %alloca, i32 0, i32 0, i32 0
+ store float 1.0, ptr addrspace(5) %p0, align 4
+ %p1 = getelementptr %wobble, ptr addrspace(5) %alloca, i32 0, i32 0, i32 1
+ store float 2.0, ptr addrspace(5) %p1, align 4
+ %p2 = getelementptr %wobble, ptr addrspace(5) %alloca, i32 0, i32 0, i32 2
+ store float 3.0, ptr addrspace(5) %p2, align 4
+ %gep = getelementptr float, ptr addrspace(5) %alloca, i32 %idx
+ %load = load float, ptr addrspace(5) %gep, align 4
+ store float %load, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; Negative: fields of differing types must not be flattened.
+define amdgpu_kernel void @heterogeneous_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @heterogeneous_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT: [[ALLOCA:%.*]] = alloca [4 x { i32, i8 }], align 16, addrspace(5)
+; CHECK-NEXT: [[P0:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 0
+; CHECK-NEXT: store i32 11, ptr addrspace(5) [[P0]], align 4
+; CHECK-NEXT: [[P1:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 1
+; CHECK-NEXT: store i32 22, ptr addrspace(5) [[P1]], align 4
+; CHECK-NEXT: [[P2:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 2
+; CHECK-NEXT: store i32 33, ptr addrspace(5) [[P2]], align 4
+; CHECK-NEXT: [[P3:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 3
+; CHECK-NEXT: store i32 44, ptr addrspace(5) [[P3]], align 4
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 [[IDX]]
+; CHECK-NEXT: [[LOAD:%.*]] = load i32, ptr addrspace(5) [[GEP]], align 4
+; CHECK-NEXT: store i32 [[LOAD]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+ %alloca = alloca [4 x { i32, i8 }], align 16, addrspace(5)
+ %p0 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 0
+ store i32 11, ptr addrspace(5) %p0, align 4
+ %p1 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 1
+ store i32 22, ptr addrspace(5) %p1, align 4
+ %p2 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 2
+ store i32 33, ptr addrspace(5) %p2, align 4
+ %p3 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 3
+ store i32 44, ptr addrspace(5) %p3, align 4
+ %gep = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 %idx
+ %load = load i32, ptr addrspace(5) %gep, align 4
+ store i32 %load, ptr addrspace(1) %out, align 4
+ ret void
+}
More information about the llvm-branch-commits
mailing list