[llvm] [AMDGPU] PromoteAlloca: flatten homogeneous structs to vectors (PR #217055)

Domenic Nutile via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 07:53:38 PDT 2026


https://github.com/saxlungs updated https://github.com/llvm/llvm-project/pull/217055

>From ac2d1cfb577191b381157bb5f6b394c8575b0c1f Mon Sep 17 00:00:00 2001
From: Domenic Nutile <domenic.nutile at gmail.com>
Date: Tue, 18 Aug 2026 11:27:20 -0400
Subject: [PATCH 1/3] [AMDGPU] PromoteAlloca: flatten homogeneous structs to
 vectors

getVectorTypeForAlloca() peeled nested ArrayType and one inner
FixedVectorType, but stopped at any StructType. An alloca of an array of
structs was therefore rejected with "Cannot convert type to vector" and
fell back to scratch, even when the struct was a trivial wrapper around a
scalar.

Peel structs too, but only when every field has the same type and the
struct has no padding, so flattened elements keep the byte offsets the
surrounding index arithmetic assumes. Structs with differing field types
or with padding are left alone.
---
 .../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 41 +++++++++++--
 .../AMDGPU/eliminate-frame-index-select.ll    |  7 ++-
 .../promote-alloca-homogeneous-struct.ll      | 59 +++++++++++++++++++
 3 files changed, 100 insertions(+), 7 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 3c1730397ab96..b582964c9575d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -892,6 +892,39 @@ static BasicBlock::iterator skipToNonAllocaInsertPt(BasicBlock &BB,
   return I;
 }
 
+/// Peel nested aggregates down to a single uniform element type, multiplying
+/// NumElems by the element count of each layer peeled.
+static Type *peelAggregateToElementType(Type *Ty, const DataLayout &DL,
+                                        uint64_t &NumElems) {
+  while (true) {
+    if (auto *ArrayTy = dyn_cast<ArrayType>(Ty)) {
+      NumElems *= ArrayTy->getNumElements();
+      Ty = ArrayTy->getElementType();
+      continue;
+    }
+
+    auto *StructTy = dyn_cast<StructType>(Ty);
+    if (!StructTy || StructTy->getNumElements() == 0)
+      break;
+
+    Type *FieldTy = StructTy->getElementType(0);
+    if (!all_of(StructTy->elements(),
+                [FieldTy](Type *T) { return T == FieldTy; }))
+      break;
+
+    // Reject any struct whose fields are not laid out back-to-back, since
+    // flattening would move elements relative to the alloca's byte offsets.
+    if (DL.getTypeAllocSize(StructTy) !=
+        StructTy->getNumElements() * DL.getTypeAllocSize(FieldTy))
+      break;
+
+    NumElems *= StructTy->getNumElements();
+    Ty = FieldTy;
+  }
+
+  return Ty;
+}
+
 FixedVectorType *
 AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
   if (DisablePromoteAllocaToVector) {
@@ -900,13 +933,9 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
   }
 
   auto *VectorTy = dyn_cast<FixedVectorType>(AllocaTy);
-  if (auto *ArrayTy = dyn_cast<ArrayType>(AllocaTy)) {
+  if (AllocaTy->isAggregateType()) {
     uint64_t NumElems = 1;
-    Type *ElemTy;
-    do {
-      NumElems *= ArrayTy->getNumElements();
-      ElemTy = ArrayTy->getElementType();
-    } while ((ArrayTy = dyn_cast<ArrayType>(ElemTy)));
+    Type *ElemTy = peelAggregateToElementType(AllocaTy, DL, NumElems);
 
     // Check for array of vectors
     auto *InnerVectorTy = dyn_cast<FixedVectorType>(ElemTy);
diff --git a/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll b/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
index d7caf2293d560..938e8bcc4bfb1 100644
--- a/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
@@ -1,5 +1,9 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgpu10.30 < %s | FileCheck %s
+; RUN: llc -mtriple=amdgpu10.30 -disable-promote-alloca-to-vector < %s | FileCheck %s
+
+; %struct.wobble flattens to <3 x float>, so the alloca is promoted and there is
+; no frame index left to eliminate. -disable-promote-alloca-to-vector will Keep it
+; on the stack so this still covers the case it was written for.
 
 %struct.wobble = type { %struct.quux }
 %struct.quux = type { float, float, float }
@@ -8,6 +12,7 @@ declare hidden %struct.wobble @foo(%struct.quux)
 
 ; s_cselect_b32 does not allow vreg & should use the sreg frameindex generated
 ; by v_readfirstlane_b32 in eliminateFrameIndex
+;
 define void @wobble() #0 {
 ; CHECK-LABEL: wobble:
 ; CHECK:       ; %bb.0: ; %bb
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
new file mode 100644
index 0000000000000..3deb4ac5208c8
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
@@ -0,0 +1,59 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -S -mtriple=amdgcn-amd-amdhsa -passes=sroa,amdgpu-promote-alloca < %s | FileCheck %s
+
+%wrapper = type { [1 x i32] }
+%pair = type { i32, i32 }
+%nested = type { %pair, %pair }
+
+; A single-field wrapper struct: [4 x { [1 x i32] }] is really <4 x i32>.
+define amdgpu_kernel void @wrapper_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @wrapper_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 1, i32 [[IDX]]
+; CHECK-NEXT:    store i32 1, ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+  %alloca = alloca [4 x %wrapper], align 16, addrspace(5)
+  %gep = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 %idx
+  store i32 1, ptr addrspace(5) %gep, align 4
+  %load = load i32, ptr addrspace(5) %gep, align 4
+  store i32 %load, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; A homogeneous multi-field struct nest: [1 x { {i32,i32}, {i32,i32} }] is <4 x i32>.
+define amdgpu_kernel void @homogeneous_nested_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @homogeneous_nested_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 1, i32 [[IDX]]
+; CHECK-NEXT:    store i32 1, ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+  %alloca = alloca [1 x %nested], align 16, addrspace(5)
+  %gep = getelementptr i32, ptr addrspace(5) %alloca, i32 %idx
+  store i32 1, ptr addrspace(5) %gep, align 4
+  %load = load i32, ptr addrspace(5) %gep, align 4
+  store i32 %load, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; Negative: fields of differing types must not be flattened.
+define amdgpu_kernel void @heterogeneous_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @heterogeneous_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT:    [[ALLOCA:%.*]] = alloca [4 x { i32, i8 }], align 16, addrspace(5)
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 [[IDX]]
+; CHECK-NEXT:    store i32 1, ptr addrspace(5) [[GEP]], align 4
+; CHECK-NEXT:    [[LOAD:%.*]] = load i32, ptr addrspace(5) [[GEP]], align 4
+; CHECK-NEXT:    store i32 [[LOAD]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+  %alloca = alloca [4 x { i32, i8 }], align 16, addrspace(5)
+  %gep = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 %idx
+  store i32 1, ptr addrspace(5) %gep, align 4
+  %load = load i32, ptr addrspace(5) %gep, align 4
+  store i32 %load, ptr addrspace(1) %out, align 4
+  ret void
+}

>From 11e4850a4c588d53d7e6a6d2fee59b788a0eecb6 Mon Sep 17 00:00:00 2001
From: Domenic Nutile <domenic.nutile at gmail.com>
Date: Wed, 19 Aug 2026 10:38:03 -0400
Subject: [PATCH 2/3] Update script version and triple format per PR feedback

---
 llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
index 3deb4ac5208c8..a1f34cf34afea 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
@@ -1,5 +1,5 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
-; RUN: opt -S -mtriple=amdgcn-amd-amdhsa -passes=sroa,amdgpu-promote-alloca < %s | FileCheck %s
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu7.00-amd-amdhsa -passes=sroa,amdgpu-promote-alloca < %s | FileCheck %s
 
 %wrapper = type { [1 x i32] }
 %pair = type { i32, i32 }

>From 79a6d767f7dbc503ad6a407b178dba97844d8366 Mon Sep 17 00:00:00 2001
From: Domenic Nutile <domenic.nutile at gmail.com>
Date: Wed, 19 Aug 2026 13:22:52 -0400
Subject: [PATCH 3/3] Simplify logic via suggestions from PR feedback. Add
 additional test cases, make tests a bit more complex so they don't fold into
 simple store of constant

---
 .../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp |  20 +---
 .../AMDGPU/eliminate-frame-index-select.ll    |   8 +-
 .../promote-alloca-homogeneous-struct.ll      | 109 ++++++++++++++++--
 3 files changed, 109 insertions(+), 28 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index b582964c9575d..b2bd81c9a9ee0 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -894,8 +894,7 @@ static BasicBlock::iterator skipToNonAllocaInsertPt(BasicBlock &BB,
 
 /// Peel nested aggregates down to a single uniform element type, multiplying
 /// NumElems by the element count of each layer peeled.
-static Type *peelAggregateToElementType(Type *Ty, const DataLayout &DL,
-                                        uint64_t &NumElems) {
+static Type *peelAggregateToElementType(Type *Ty, uint64_t &NumElems) {
   while (true) {
     if (auto *ArrayTy = dyn_cast<ArrayType>(Ty)) {
       NumElems *= ArrayTy->getNumElements();
@@ -904,22 +903,11 @@ static Type *peelAggregateToElementType(Type *Ty, const DataLayout &DL,
     }
 
     auto *StructTy = dyn_cast<StructType>(Ty);
-    if (!StructTy || StructTy->getNumElements() == 0)
-      break;
-
-    Type *FieldTy = StructTy->getElementType(0);
-    if (!all_of(StructTy->elements(),
-                [FieldTy](Type *T) { return T == FieldTy; }))
-      break;
-
-    // Reject any struct whose fields are not laid out back-to-back, since
-    // flattening would move elements relative to the alloca's byte offsets.
-    if (DL.getTypeAllocSize(StructTy) !=
-        StructTy->getNumElements() * DL.getTypeAllocSize(FieldTy))
+    if (!StructTy || !StructTy->containsHomogeneousTypes())
       break;
 
     NumElems *= StructTy->getNumElements();
-    Ty = FieldTy;
+    Ty = StructTy->getElementType(0);
   }
 
   return Ty;
@@ -935,7 +923,7 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
   auto *VectorTy = dyn_cast<FixedVectorType>(AllocaTy);
   if (AllocaTy->isAggregateType()) {
     uint64_t NumElems = 1;
-    Type *ElemTy = peelAggregateToElementType(AllocaTy, DL, NumElems);
+    Type *ElemTy = peelAggregateToElementType(AllocaTy, NumElems);
 
     // Check for array of vectors
     auto *InnerVectorTy = dyn_cast<FixedVectorType>(ElemTy);
diff --git a/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll b/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
index 938e8bcc4bfb1..5e1514ae7de9e 100644
--- a/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/eliminate-frame-index-select.ll
@@ -1,9 +1,10 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
 ; RUN: llc -mtriple=amdgpu10.30 -disable-promote-alloca-to-vector < %s | FileCheck %s
 
-; %struct.wobble flattens to <3 x float>, so the alloca is promoted and there is
-; no frame index left to eliminate. -disable-promote-alloca-to-vector will Keep it
-; on the stack so this still covers the case it was written for.
+; %struct.wobble is homogeneous and flattens to <3 x float>, so the alloca is
+; promoted to a vector and no frame index is left to eliminate.
+; -disable-promote-alloca-to-vector keeps it on the stack so this still covers
+; the case it was written for.
 
 %struct.wobble = type { %struct.quux }
 %struct.quux = type { float, float, float }
@@ -12,7 +13,6 @@ declare hidden %struct.wobble @foo(%struct.quux)
 
 ; s_cselect_b32 does not allow vreg & should use the sreg frameindex generated
 ; by v_readfirstlane_b32 in eliminateFrameIndex
-;
 define void @wobble() #0 {
 ; CHECK-LABEL: wobble:
 ; CHECK:       ; %bb.0: ; %bb
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
index a1f34cf34afea..152c114e291f2 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-homogeneous-struct.ll
@@ -4,19 +4,32 @@
 %wrapper = type { [1 x i32] }
 %pair = type { i32, i32 }
 %nested = type { %pair, %pair }
+%simple_struct = type { i32, i32, i32, i32 }
+%wobble = type { { float, float, float } }
 
 ; A single-field wrapper struct: [4 x { [1 x i32] }] is really <4 x i32>.
 define amdgpu_kernel void @wrapper_struct(ptr addrspace(1) %out, i32 %idx) {
 ; CHECK-LABEL: define amdgpu_kernel void @wrapper_struct(
 ; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
 ; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 1, i32 [[IDX]]
-; CHECK-NEXT:    store i32 1, ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 11, i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i32> [[TMP1]], i32 22, i32 1
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 33, i32 2
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 44, i32 3
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i32> [[TMP4]], i32 [[IDX]]
+; CHECK-NEXT:    store i32 [[TMP5]], ptr addrspace(1) [[OUT]], align 4
 ; CHECK-NEXT:    ret void
 ;
   %alloca = alloca [4 x %wrapper], align 16, addrspace(5)
+  %p0 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 0
+  store i32 11, ptr addrspace(5) %p0, align 4
+  %p1 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 1
+  store i32 22, ptr addrspace(5) %p1, align 4
+  %p2 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 2
+  store i32 33, ptr addrspace(5) %p2, align 4
+  %p3 = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 3
+  store i32 44, ptr addrspace(5) %p3, align 4
   %gep = getelementptr [4 x %wrapper], ptr addrspace(5) %alloca, i32 0, i32 %idx
-  store i32 1, ptr addrspace(5) %gep, align 4
   %load = load i32, ptr addrspace(5) %gep, align 4
   store i32 %load, ptr addrspace(1) %out, align 4
   ret void
@@ -27,32 +40,112 @@ define amdgpu_kernel void @homogeneous_nested_struct(ptr addrspace(1) %out, i32
 ; CHECK-LABEL: define amdgpu_kernel void @homogeneous_nested_struct(
 ; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
 ; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 1, i32 [[IDX]]
-; CHECK-NEXT:    store i32 1, ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 11, i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i32> [[TMP1]], i32 22, i32 1
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 33, i32 2
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 44, i32 3
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i32> [[TMP4]], i32 [[IDX]]
+; CHECK-NEXT:    store i32 [[TMP5]], ptr addrspace(1) [[OUT]], align 4
 ; CHECK-NEXT:    ret void
 ;
   %alloca = alloca [1 x %nested], align 16, addrspace(5)
+  %p0 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 0, i32 0
+  store i32 11, ptr addrspace(5) %p0, align 4
+  %p1 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 0, i32 1
+  store i32 22, ptr addrspace(5) %p1, align 4
+  %p2 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 1, i32 0
+  store i32 33, ptr addrspace(5) %p2, align 4
+  %p3 = getelementptr [1 x %nested], ptr addrspace(5) %alloca, i32 0, i32 0, i32 1, i32 1
+  store i32 44, ptr addrspace(5) %p3, align 4
   %gep = getelementptr i32, ptr addrspace(5) %alloca, i32 %idx
-  store i32 1, ptr addrspace(5) %gep, align 4
   %load = load i32, ptr addrspace(5) %gep, align 4
   store i32 %load, ptr addrspace(1) %out, align 4
   ret void
 }
 
+; The alloca does not have to be an array: a homogeneous struct on its own is
+; flattened the same way. { i32, i32, i32, i32 } is <4 x i32>.
+define amdgpu_kernel void @direct_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @direct_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 11, i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i32> [[TMP1]], i32 22, i32 1
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 33, i32 2
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 44, i32 3
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i32> [[TMP4]], i32 [[IDX]]
+; CHECK-NEXT:    store i32 [[TMP5]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+  %alloca = alloca %simple_struct, align 16, addrspace(5)
+  %p0 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 0
+  store i32 11, ptr addrspace(5) %p0, align 4
+  %p1 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 1
+  store i32 22, ptr addrspace(5) %p1, align 4
+  %p2 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 2
+  store i32 33, ptr addrspace(5) %p2, align 4
+  %p3 = getelementptr %simple_struct, ptr addrspace(5) %alloca, i32 0, i32 3
+  store i32 44, ptr addrspace(5) %p3, align 4
+  %gep = getelementptr i32, ptr addrspace(5) %alloca, i32 %idx
+  %load = load i32, ptr addrspace(5) %gep, align 4
+  store i32 %load, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; A direct struct wrapping a homogeneous struct of scalars: { {float,float,float} }
+; is <3 x float>
+define amdgpu_kernel void @direct_wrapper_struct(ptr addrspace(1) %out, i32 %idx) {
+; CHECK-LABEL: define amdgpu_kernel void @direct_wrapper_struct(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <3 x float> poison
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <3 x float> [[ALLOCA]], float 1.000000e+00, i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <3 x float> [[TMP1]], float 2.000000e+00, i32 1
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <3 x float> [[TMP2]], float 3.000000e+00, i32 2
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <3 x float> [[TMP3]], i32 [[IDX]]
+; CHECK-NEXT:    store float [[TMP4]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+  %alloca = alloca %wobble, align 16, addrspace(5)
+  %p0 = getelementptr %wobble, ptr addrspace(5) %alloca, i32 0, i32 0, i32 0
+  store float 1.0, ptr addrspace(5) %p0, align 4
+  %p1 = getelementptr %wobble, ptr addrspace(5) %alloca, i32 0, i32 0, i32 1
+  store float 2.0, ptr addrspace(5) %p1, align 4
+  %p2 = getelementptr %wobble, ptr addrspace(5) %alloca, i32 0, i32 0, i32 2
+  store float 3.0, ptr addrspace(5) %p2, align 4
+  %gep = getelementptr float, ptr addrspace(5) %alloca, i32 %idx
+  %load = load float, ptr addrspace(5) %gep, align 4
+  store float %load, ptr addrspace(1) %out, align 4
+  ret void
+}
+
 ; Negative: fields of differing types must not be flattened.
 define amdgpu_kernel void @heterogeneous_struct(ptr addrspace(1) %out, i32 %idx) {
 ; CHECK-LABEL: define amdgpu_kernel void @heterogeneous_struct(
 ; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[IDX:%.*]]) {
 ; CHECK-NEXT:    [[ALLOCA:%.*]] = alloca [4 x { i32, i8 }], align 16, addrspace(5)
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 0
+; CHECK-NEXT:    store i32 11, ptr addrspace(5) [[P0]], align 4
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 1
+; CHECK-NEXT:    store i32 22, ptr addrspace(5) [[P1]], align 4
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 2
+; CHECK-NEXT:    store i32 33, ptr addrspace(5) [[P2]], align 4
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 3
+; CHECK-NEXT:    store i32 44, ptr addrspace(5) [[P3]], align 4
 ; CHECK-NEXT:    [[GEP:%.*]] = getelementptr [4 x { i32, i8 }], ptr addrspace(5) [[ALLOCA]], i32 0, i32 [[IDX]]
-; CHECK-NEXT:    store i32 1, ptr addrspace(5) [[GEP]], align 4
 ; CHECK-NEXT:    [[LOAD:%.*]] = load i32, ptr addrspace(5) [[GEP]], align 4
 ; CHECK-NEXT:    store i32 [[LOAD]], ptr addrspace(1) [[OUT]], align 4
 ; CHECK-NEXT:    ret void
 ;
   %alloca = alloca [4 x { i32, i8 }], align 16, addrspace(5)
+  %p0 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 0
+  store i32 11, ptr addrspace(5) %p0, align 4
+  %p1 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 1
+  store i32 22, ptr addrspace(5) %p1, align 4
+  %p2 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 2
+  store i32 33, ptr addrspace(5) %p2, align 4
+  %p3 = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 3
+  store i32 44, ptr addrspace(5) %p3, align 4
   %gep = getelementptr [4 x { i32, i8 }], ptr addrspace(5) %alloca, i32 0, i32 %idx
-  store i32 1, ptr addrspace(5) %gep, align 4
   %load = load i32, ptr addrspace(5) %gep, align 4
   store i32 %load, ptr addrspace(1) %out, align 4
   ret void



More information about the llvm-commits mailing list