[llvm-branch-commits] [llvm] Reland "[AMDGPU] PromoteAlloca: flatten homogeneous structs to vectors" (#221058) (PR #223790)

Domenic Nutile via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Tue Sep 22 07:12:32 PDT 2026


https://github.com/saxlungs updated https://github.com/llvm/llvm-project/pull/223790

>From 654f8bdc148742b162873102d42c448047addcca Mon Sep 17 00:00:00 2001
From: Domenic Nutile <domenic.nutile at gmail.com>
Date: Tue, 15 Sep 2026 15:16:42 -0400
Subject: [PATCH] Reland "[AMDGPU] PromoteAlloca: flatten homogeneous structs
 to vectors" (#221058)

This relands #217055

The original commit revealed a latent issue in eliminateFrameIndex in
SIRegisterInfo where SCC can be clobbered before reading it on
gfx900/gfx90a. This change itself has no known issues.
---
 .../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp |  29 +++-
 .../AMDGPU/eliminate-frame-index-select.ll    |   7 +-
 .../promote-alloca-homogeneous-struct.ll      | 152 ++++++++++++++++++
 3 files changed, 181 insertions(+), 7 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 4f54ca98bd5fc..a1936dfba1131 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -912,6 +912,27 @@ static BasicBlock::iterator skipToNonAllocaInsertPt(BasicBlock &BB,
   return I;
 }
 
+/// Peel nested aggregates down to a single uniform element type, multiplying
+/// NumElems by the element count of each layer peeled.
+static Type *peelAggregateToElementType(Type *Ty, uint64_t &NumElems) {
+  while (true) {
+    if (auto *ArrayTy = dyn_cast<ArrayType>(Ty)) {
+      NumElems *= ArrayTy->getNumElements();
+      Ty = ArrayTy->getElementType();
+      continue;
+    }
+
+    auto *StructTy = dyn_cast<StructType>(Ty);
+    if (!StructTy || !StructTy->containsHomogeneousTypes())
+      break;
+
+    NumElems *= StructTy->getNumElements();
+    Ty = StructTy->getElementType(0);
+  }
+
+  return Ty;
+}
+
 FixedVectorType *
 AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
   if (DisablePromoteAllocaToVector) {
@@ -920,13 +941,9 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
   }
 
   auto *VectorTy = dyn_cast<FixedVectorType>(AllocaTy);
-  if (auto *ArrayTy = dyn_cast<ArrayType>(AllocaTy)) {
+  if (AllocaTy->isAggregateType()) {
     uint64_t NumElems = 1;
-    Type *ElemTy;
-    do {
-      NumElems *= ArrayTy->getNumElements();
-      ElemTy = ArrayTy->getElementType();
-    } while ((ArrayTy = dyn_cast<ArrayType>(ElemTy)));
+    Type *ElemTy = peelAggregateToElementType(AllocaTy, NumElems);
 
     // Check for array of vectors
     auto *InnerVectorTy = dyn_cast<FixedVectorType>(ElemTy);
diff --git a/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll b/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
index d7caf2293d560..5e1514ae7de9e 100644
--- a/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
@@ -1,5 +1,10 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgpu10.30 < %s | FileCheck %s
+; RUN: llc -mtriple=amdgpu10.30 -disable-promote-alloca-to-vector < %s | FileCheck %s
+
+; %struct.wobble is homogeneous and flattens to <3 x float>, so the alloca is
+; promoted to a vector and no frame index is left to eliminate.
+; -disable-promote-alloca-to-vector keeps it on the stack so this still covers
+; the case it was written for.
 
 %struct.wobble = type { %struct.quux }
 %struct.quux = type { float, float, float }
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
new file mode 100644
index 0000000000000..152c114e291f2
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
@@ -0,0 +1,152 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu7.00-amd-amdhsa -passes=sroa,amdgpu-promote-alloca < %s | FileCheck %s
+
+%wrapper = type { [1 x i32] }
+%pair = type { i32, i32 }
+%nested = type { %pair, %pair }
+%simple_struct = type { i32, i32, i32, i32 }
+%wobble = type { { float, float, float } }
+
+; A single-field wrapper struct: [4 x { [1 x i32] }] is really <4 x i32>.
+define amdgpu_kernel void @wrapper_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @wrapper_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 11, i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i32> [[TMP1]], i32 22, i32 1
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 33, i32 2
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 44, i32 3
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i32> [[TMP4]], i32 [[IDX]]
+; CHECK-NEXT:    store i32 [[TMP5]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+  %alloca = alloca [4 x %wrapper], align 16, addrspace(5)
+  %p0 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 0
+  store i32 11, ptr addrspace(5) %p0, align 4
+  %p1 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 1
+  store i32 22, ptr addrspace(5) %p1, align 4
+  %p2 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 2
+  store i32 33, ptr addrspace(5) %p2, align 4
+  %p3 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 3
+  store i32 44, ptr addrspace(5) %p3, align 4
+  %gep = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 %idx
+  %load = load i32, ptr addrspace(5) %gep, align 4
+  store i32 %load, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; A homogeneous multi-field struct nest: [1 x { {i32,i32}, {i32,i32} }] is <4 x i32>.
+define amdgpu_kernel void @homogeneous_nested_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @homogeneous_nested_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 11, i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i32> [[TMP1]], i32 22, i32 1
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 33, i32 2
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 44, i32 3
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i32> [[TMP4]], i32 [[IDX]]
+; CHECK-NEXT:    store i32 [[TMP5]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+  %alloca = alloca [1 x %nested], align 16, addrspace(5)
+  %p0 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 0, i32 0
+  store i32 11, ptr addrspace(5) %p0, align 4
+  %p1 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 0, i32 1
+  store i32 22, ptr addrspace(5) %p1, align 4
+  %p2 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 1, i32 0
+  store i32 33, ptr addrspace(5) %p2, align 4
+  %p3 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 1, i32 1
+  store i32 44, ptr addrspace(5) %p3, align 4
+  %gep = getelementptr i32, ptr addrspace(5) %alloca, i32 %idx
+  %load = load i32, ptr addrspace(5) %gep, align 4
+  store i32 %load, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; The alloca does not have to be an array: a homogeneous struct on its own is
+; flattened the same way. { i32, i32, i32, i32 } is <4 x i32>.
+define amdgpu_kernel void @direct_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @direct_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 11, i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i32> [[TMP1]], i32 22, i32 1
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 33, i32 2
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 44, i32 3
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i32> [[TMP4]], i32 [[IDX]]
+; CHECK-NEXT:    store i32 [[TMP5]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+  %alloca = alloca %simple_struct, align 16, addrspace(5)
+  %p0 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 0
+  store i32 11, ptr addrspace(5) %p0, align 4
+  %p1 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 1
+  store i32 22, ptr addrspace(5) %p1, align 4
+  %p2 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 2
+  store i32 33, ptr addrspace(5) %p2, align 4
+  %p3 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 3
+  store i32 44, ptr addrspace(5) %p3, align 4
+  %gep = getelementptr i32, ptr addrspace(5) %alloca, i32 %idx
+  %load = load i32, ptr addrspace(5) %gep, align 4
+  store i32 %load, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; A direct struct wrapping a homogeneous struct of scalars: { {float,float,float} }
+; is <3 x float>
+define amdgpu_kernel void @direct_wrapper_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @direct_wrapper_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <3 x float> poison
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <3 x float> [[ALLOCA]], float 1.000000e+00, i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <3 x float> [[TMP1]], float 2.000000e+00, i32 1
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <3 x float> [[TMP2]], float 3.000000e+00, i32 2
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <3 x float> [[TMP3]], i32 [[IDX]]
+; CHECK-NEXT:    store float [[TMP4]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+  %alloca = alloca %wobble, align 16, addrspace(5)
+  %p0 = getelementptr %wobble, ptr addrspace(5) %alloca, i32 0, i32 0, i32 0
+  store float 1.0, ptr addrspace(5) %p0, align 4
+  %p1 = getelementptr %wobble, ptr addrspace(5) %alloca, i32 0, i32 0, i32 1
+  store float 2.0, ptr addrspace(5) %p1, align 4
+  %p2 = getelementptr %wobble, ptr addrspace(5) %alloca, i32 0, i32 0, i32 2
+  store float 3.0, ptr addrspace(5) %p2, align 4
+  %gep = getelementptr float, ptr addrspace(5) %alloca, i32 %idx
+  %load = load float, ptr addrspace(5) %gep, align 4
+  store float %load, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; Negative: fields of differing types must not be flattened.
+define amdgpu_kernel void @heterogeneous_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @heterogeneous_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT:    [[ALLOCA:%.*]] = alloca [4 x { i32, i8 }], align 16, addrspace(5)
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 0
+; CHECK-NEXT:    store i32 11, ptr addrspace(5) [[P0]], align 4
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 1
+; CHECK-NEXT:    store i32 22, ptr addrspace(5) [[P1]], align 4
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 2
+; CHECK-NEXT:    store i32 33, ptr addrspace(5) [[P2]], align 4
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 3
+; CHECK-NEXT:    store i32 44, ptr addrspace(5) [[P3]], align 4
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 [[IDX]]
+; CHECK-NEXT:    [[LOAD:%.*]] = load i32, ptr addrspace(5) [[GEP]], align 4
+; CHECK-NEXT:    store i32 [[LOAD]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+  %alloca = alloca [4 x { i32, i8 }], align 16, addrspace(5)
+  %p0 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 0
+  store i32 11, ptr addrspace(5) %p0, align 4
+  %p1 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 1
+  store i32 22, ptr addrspace(5) %p1, align 4
+  %p2 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 2
+  store i32 33, ptr addrspace(5) %p2, align 4
+  %p3 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 3
+  store i32 44, ptr addrspace(5) %p3, align 4
+  %gep = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 %idx
+  %load = load i32, ptr addrspace(5) %gep, align 4
+  store i32 %load, ptr addrspace(1) %out, align 4
+  ret void
+}



More information about the llvm-branch-commits mailing list