[llvm] [AMDGPU] Add missing conversion link in alloca promotion pass (PR #177945)

Steffen Larsen via llvm-commits llvm-commits at lists.llvm.org
Mon Jan 26 04:50:36 PST 2026


https://github.com/steffenlarsen created https://github.com/llvm/llvm-project/pull/177945

The AMDGPU promote alloca pass is missing a conversion link when casting between vectors of pointers and pointers or vectors of pointers with different number of elements. This causes codegen to crash due to invalid casts being generated. To address this, this commit adds the missing conversion link.

In addition to this, the commit moves the common load/store cast logic into a new function `createLoadStoreCastChain`.

>From 62cc0c88b552f2ee09e7137478339f82221f87ab Mon Sep 17 00:00:00 2001
From: Steffen Holst Larsen <HolstLarsen.Steffen at amd.com>
Date: Sat, 24 Jan 2026 06:51:23 -0600
Subject: [PATCH] [AMDGPU] Add missing conversion link in alloca promotion pass

The AMDGPU promote alloca pass is missing a conversion link when casting
between vectors of pointers and pointers or vectors of pointers with
different number of elements. This causes codegen to crash due to
invalid casts being generated. To address this, this commit adds the
missing conversion link.

In addition to this, the commit moves the common load/store cast logic
into a new function `createLoadStoreCastChain`.

Signed-off-by: Steffen Holst Larsen <HolstLarsen.Steffen at amd.com>
---
 .../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 112 +++++++++++-------
 .../AMDGPU/promote-alloca-loadstores.ll       |  53 ++++++++-
 2 files changed, 115 insertions(+), 50 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 3355c277e50d2..bdf74810eab9c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -582,6 +582,66 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
   return Result;
 }
 
+/// Creates a cast chain to convert Val to DstTy for load/store promotions.
+///
+/// \param DL      Module Data Layout.
+/// \param Builder IRBuilder to insert instructions with.
+/// \param Val     Value to be casted.
+/// \param DstTy   Destination type for the cast.
+/// \return the final value in the created cast chain.
+template <typename FolderTy, typename InserterTy>
+static Value *createLoadStoreCastChain(const DataLayout &DL,
+                                       IRBuilder<FolderTy, InserterTy> &Builder,
+                                       Value *Val, Type *DstTy) {
+  const auto CreateTempPtrIntCast = [&Builder, DL](Value *Val,
+                                                   Type *PtrTy) -> Value * {
+    assert(DL.getTypeStoreSize(Val->getType()) == DL.getTypeStoreSize(PtrTy));
+    const unsigned Size = DL.getTypeStoreSizeInBits(PtrTy);
+    if (!PtrTy->isVectorTy())
+      return Builder.CreateBitOrPointerCast(Val, Builder.getIntNTy(Size));
+    const unsigned NumPtrElts = cast<FixedVectorType>(PtrTy)->getNumElements();
+    // If we want to cast to cast, e.g. a <2 x ptr> into a <4 x i32>, we need
+    // to first cast the ptr vector to <2 x i64>.
+    assert((Size % NumPtrElts == 0) && "Vector size not divisble");
+    Type *EltTy = Builder.getIntNTy(Size / NumPtrElts);
+    return Builder.CreateBitOrPointerCast(
+        Val, FixedVectorType::get(EltTy, NumPtrElts));
+  };
+
+  {
+    // If we are casting between a vector of pointers and either a pointer or
+    // a vector of pointers with a different number of elements, we need an
+    // additional intermediate cast. Examples:
+    //   <2 x ptr addrspace(5)> -> ptr
+    //     => <2 x ptr addrspace(5)> -> <2 x i32> -> i64 -> ptr
+    //   ptr -> <2 x ptr addrspace(5)>
+    //     => ptr -> i64 -> <2 x i32> -> <2 x ptr addrspace(5)>
+    //   <4 x ptr addrspace(5)> -> <2 x ptr>
+    //     => <4 x ptr addrspace(5)> -> <4 x i32> -> <2 x i64> -> <2 x ptr>
+    //   <2 x ptr> -> <4 x ptr addrspace(5)>
+    //     => <2 x ptr> -> <2 x i64> -> <4 x i32> -> <4 x ptr addrspace(5)>
+    // Where the first conversion in each chain is the conversion done here
+    // and the rest are done after this block.
+    Type *ValTy = Val->getType();
+    bool ValIsPtrVec =
+        ValTy->isVectorTy() && ValTy->getScalarType()->isPointerTy();
+    bool DstIsPtrVec =
+        DstTy->isVectorTy() && DstTy->getScalarType()->isPointerTy();
+    if (((ValIsPtrVec || DstIsPtrVec) &&
+         (DstTy->isPointerTy() || ValTy->isPointerTy())) ||
+        (ValIsPtrVec && DstIsPtrVec &&
+         cast<VectorType>(DstTy)->getElementCount() !=
+             cast<VectorType>(ValTy)->getElementCount()))
+      Val = CreateTempPtrIntCast(Val, ValTy);
+  }
+
+  if (DstTy->isPtrOrPtrVectorTy())
+    Val = CreateTempPtrIntCast(Val, DstTy);
+  else if (Val->getType()->isPtrOrPtrVectorTy())
+    Val = CreateTempPtrIntCast(Val, Val->getType());
+  return Builder.CreateBitOrPointerCast(Val, DstTy);
+}
+
 /// Promotes a single user of the alloca to a vector form.
 ///
 /// \param Inst           Instruction to be promoted.
@@ -606,21 +666,6 @@ static Value *promoteAllocaUserToVector(Instruction *Inst, const DataLayout &DL,
                                         InstSimplifyFolder(DL));
   Builder.SetInsertPoint(Inst);
 
-  const auto CreateTempPtrIntCast = [&Builder, DL](Value *Val,
-                                                   Type *PtrTy) -> Value * {
-    assert(DL.getTypeStoreSize(Val->getType()) == DL.getTypeStoreSize(PtrTy));
-    const unsigned Size = DL.getTypeStoreSizeInBits(PtrTy);
-    if (!PtrTy->isVectorTy())
-      return Builder.CreateBitOrPointerCast(Val, Builder.getIntNTy(Size));
-    const unsigned NumPtrElts = cast<FixedVectorType>(PtrTy)->getNumElements();
-    // If we want to cast to cast, e.g. a <2 x ptr> into a <4 x i32>, we need to
-    // first cast the ptr vector to <2 x i64>.
-    assert((Size % NumPtrElts == 0) && "Vector size not divisble");
-    Type *EltTy = Builder.getIntNTy(Size / NumPtrElts);
-    return Builder.CreateBitOrPointerCast(
-        Val, FixedVectorType::get(EltTy, NumPtrElts));
-  };
-
   Type *VecEltTy = AA.Vector.Ty->getElementType();
 
   switch (Inst->getOpcode()) {
@@ -634,12 +679,8 @@ static Value *promoteAllocaUserToVector(Instruction *Inst, const DataLayout &DL,
     TypeSize AccessSize = DL.getTypeStoreSize(AccessTy);
     if (Constant *CI = dyn_cast<Constant>(Index)) {
       if (CI->isZeroValue() && AccessSize == VecStoreSize) {
-        if (AccessTy->isPtrOrPtrVectorTy())
-          CurVal = CreateTempPtrIntCast(CurVal, AccessTy);
-        else if (CurVal->getType()->isPtrOrPtrVectorTy())
-          CurVal = CreateTempPtrIntCast(CurVal, CurVal->getType());
-        Value *NewVal = Builder.CreateBitOrPointerCast(CurVal, AccessTy);
-        Inst->replaceAllUsesWith(NewVal);
+        Inst->replaceAllUsesWith(
+            createLoadStoreCastChain(DL, Builder, CurVal, AccessTy));
         return nullptr;
       }
     }
@@ -689,13 +730,8 @@ static Value *promoteAllocaUserToVector(Instruction *Inst, const DataLayout &DL,
             SubVec, Builder.CreateExtractElement(CurVal, CurIdx), K);
       }
 
-      if (AccessTy->isPtrOrPtrVectorTy())
-        SubVec = CreateTempPtrIntCast(SubVec, AccessTy);
-      else if (SubVecTy->isPtrOrPtrVectorTy())
-        SubVec = CreateTempPtrIntCast(SubVec, SubVecTy);
-
-      SubVec = Builder.CreateBitOrPointerCast(SubVec, AccessTy);
-      Inst->replaceAllUsesWith(SubVec);
+      Inst->replaceAllUsesWith(
+          createLoadStoreCastChain(DL, Builder, SubVec, AccessTy));
       return nullptr;
     }
 
@@ -719,15 +755,9 @@ static Value *promoteAllocaUserToVector(Instruction *Inst, const DataLayout &DL,
     // We're storing the full vector, we can handle this without knowing CurVal.
     Type *AccessTy = Val->getType();
     TypeSize AccessSize = DL.getTypeStoreSize(AccessTy);
-    if (Constant *CI = dyn_cast<Constant>(Index)) {
-      if (CI->isZeroValue() && AccessSize == VecStoreSize) {
-        if (AccessTy->isPtrOrPtrVectorTy())
-          Val = CreateTempPtrIntCast(Val, AccessTy);
-        else if (AA.Vector.Ty->isPtrOrPtrVectorTy())
-          Val = CreateTempPtrIntCast(Val, AA.Vector.Ty);
-        return Builder.CreateBitOrPointerCast(Val, AA.Vector.Ty);
-      }
-    }
+    if (Constant *CI = dyn_cast<Constant>(Index))
+      if (CI->isZeroValue() && AccessSize == VecStoreSize)
+        return createLoadStoreCastChain(DL, Builder, Val, AA.Vector.Ty);
 
     // Storing a subvector.
     if (isa<FixedVectorType>(AccessTy)) {
@@ -738,13 +768,7 @@ static Value *promoteAllocaUserToVector(Instruction *Inst, const DataLayout &DL,
       auto *SubVecTy = FixedVectorType::get(VecEltTy, NumWrittenElts);
       assert(DL.getTypeStoreSize(SubVecTy) == DL.getTypeStoreSize(AccessTy));
 
-      if (SubVecTy->isPtrOrPtrVectorTy())
-        Val = CreateTempPtrIntCast(Val, SubVecTy);
-      else if (AccessTy->isPtrOrPtrVectorTy())
-        Val = CreateTempPtrIntCast(Val, AccessTy);
-
-      Val = Builder.CreateBitOrPointerCast(Val, SubVecTy);
-
+      Val = createLoadStoreCastChain(DL, Builder, Val, SubVecTy);
       Value *CurVec = GetCurVal();
       for (unsigned K = 0, NumElts = std::min(NumWrittenElts, NumVecElts);
            K < NumElts; ++K) {
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-loadstores.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-loadstores.ll
index 015ce256a80c2..b2d49d8908ad9 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-loadstores.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-loadstores.ll
@@ -9,9 +9,9 @@ define amdgpu_kernel void @test_overwrite(i64 %val, i1 %cond) {
 ; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <3 x i64> [[STACK]], i64 43, i32 0
 ; CHECK-NEXT:    br i1 [[COND]], label [[LOOP:%.*]], label [[END:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[PROMOTEALLOCA1:%.*]] = phi <3 x i64> [ [[TMP3:%.*]], [[LOOP]] ], [ [[TMP0]], [[ENTRY:%.*]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <3 x i64> [[PROMOTEALLOCA1]], i32 0
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <3 x i64> [[PROMOTEALLOCA1]], i64 68, i32 0
+; CHECK-NEXT:    [[PROMOTEALLOCA2:%.*]] = phi <3 x i64> [ [[TMP3:%.*]], [[LOOP]] ], [ [[TMP0]], [[ENTRY:%.*]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <3 x i64> [[PROMOTEALLOCA2]], i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <3 x i64> [[PROMOTEALLOCA2]], i64 68, i32 0
 ; CHECK-NEXT:    [[TMP3]] = insertelement <3 x i64> [[TMP2]], i64 32, i32 0
 ; CHECK-NEXT:    [[LOOP_CC:%.*]] = icmp ne i64 [[TMP1]], 68
 ; CHECK-NEXT:    br i1 [[LOOP_CC]], label [[LOOP]], label [[END]]
@@ -67,9 +67,9 @@ define amdgpu_kernel void @test_no_overwrite(i64 %val, i1 %cond) {
 ; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <3 x i64> [[STACK]], i64 43, i32 0
 ; CHECK-NEXT:    br i1 [[COND]], label [[LOOP:%.*]], label [[END:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[PROMOTEALLOCA1:%.*]] = phi <3 x i64> [ [[TMP2:%.*]], [[LOOP]] ], [ [[TMP0]], [[ENTRY:%.*]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <3 x i64> [[PROMOTEALLOCA1]], i32 0
-; CHECK-NEXT:    [[TMP2]] = insertelement <3 x i64> [[PROMOTEALLOCA1]], i64 32, i32 1
+; CHECK-NEXT:    [[PROMOTEALLOCA2:%.*]] = phi <3 x i64> [ [[TMP2:%.*]], [[LOOP]] ], [ [[TMP0]], [[ENTRY:%.*]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <3 x i64> [[PROMOTEALLOCA2]], i32 0
+; CHECK-NEXT:    [[TMP2]] = insertelement <3 x i64> [[PROMOTEALLOCA2]], i64 32, i32 1
 ; CHECK-NEXT:    [[LOOP_CC:%.*]] = icmp ne i64 [[TMP1]], 32
 ; CHECK-NEXT:    br i1 [[LOOP_CC]], label [[LOOP]], label [[END]]
 ; CHECK:       end:
@@ -192,6 +192,47 @@ entry:
   ret void
 }
 
+define void @alloca_load_store_ptr_ptrvec(ptr %arg) {
+; CHECK-LABEL: define void @alloca_load_store_ptr_ptrvec
+; CHECK-SAME: (ptr [[ARG:%.*]]) {
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <2 x ptr addrspace(3)> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = ptrtoint ptr [[ARG]] to i64
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[TMP0]] to <2 x i32>
+; CHECK-NEXT:    [[TMP2:%.*]] = inttoptr <2 x i32> [[TMP1]] to <2 x ptr addrspace(3)>
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca <2 x ptr addrspace(3)>, align 8, addrspace(5)
+  store ptr %arg, ptr addrspace(5) %alloca, align 8
+  %tmp = load ptr, ptr addrspace(5) %alloca, align 8
+  ret void
+}
+
+define void @alloca_load_store_diff_size_ptrvecs(<2 x ptr> %arg1, <4 x ptr addrspace(3)> %arg2) {
+; CHECK-LABEL: define void @alloca_load_store_diff_size_ptrvecs
+; CHECK-SAME: (<2 x ptr> [[ARG1:%.*]], <4 x ptr addrspace(3)> [[ARG2:%.*]]) {
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[ALLOCA2:%.*]] = freeze <2 x ptr> poison
+; CHECK-NEXT:    [[ALLOCA1:%.*]] = freeze <4 x ptr addrspace(3)> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = ptrtoint <2 x ptr> [[ARG1]] to <2 x i64>
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast <2 x i64> [[TMP0]] to <4 x i32>
+; CHECK-NEXT:    [[TMP2:%.*]] = inttoptr <4 x i32> [[TMP1]] to <4 x ptr addrspace(3)>
+; CHECK-NEXT:    [[TMP3:%.*]] = ptrtoint <4 x ptr addrspace(3)> [[ARG2]] to <4 x i32>
+; CHECK-NEXT:    [[TMP4:%.*]] = bitcast <4 x i32> [[TMP3]] to <2 x i64>
+; CHECK-NEXT:    [[TMP5:%.*]] = inttoptr <2 x i64> [[TMP4]] to <2 x ptr>
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca1 = alloca <4 x ptr addrspace(3)>, align 8, addrspace(5)
+  %alloca2 = alloca <2 x ptr>, align 8, addrspace(5)
+  store <2 x ptr> %arg1, ptr addrspace(5) %alloca1, align 8
+  store <4 x ptr addrspace(3)> %arg2, ptr addrspace(5) %alloca2, align 8
+  %tmp1 = load <2 x ptr>, ptr addrspace(5) %alloca1, align 8
+  %tmp2 = load <4 x ptr addrspace(3)>, ptr addrspace(5) %alloca2, align 8
+  ret void
+}
+
 ; Will not vectorize because we're accessing a 64 bit vector with a 32 bits pointer.
 define ptr addrspace(3) @alloca_load_store_ptr_mixed_full_ivec(ptr addrspace(3) %arg) {
 ; CHECK-LABEL: define ptr addrspace(3) @alloca_load_store_ptr_mixed_full_ivec



More information about the llvm-commits mailing list