[llvm] [AMDGPU] Promote allocas used by phi/select (PR #221870)
Matt Arsenault via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 13 07:37:57 PDT 2026
================
@@ -977,15 +1112,57 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
}
void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
- if (AA.HaveSelectOrPHI) {
- LLVM_DEBUG(dbgs() << " Cannot convert to vector due to select or phi\n");
+ // A phi/select that merges with null or a non-alloca object can't be given a
+ // vector lane, so the whole group is unpromotable to vector.
+ if (AA.HaveUnpromotableMerge) {
+ LLVM_DEBUG(dbgs() << " Cannot convert to vector due to merge with "
+ "null/unknown object\n");
return;
}
- Type *AllocaTy = AA.Alloca->getAllocatedType();
- AA.Vector.Ty = getVectorTypeForAlloca(AllocaTy);
- if (!AA.Vector.Ty)
+ if (DisablePromoteAllocaToVector) {
+ LLVM_DEBUG(dbgs() << " Promote alloca to vectors is disabled\n");
return;
+ }
+
+ // Find a valid common element type and assign each member a contiguous lane
+ // range in the combined vector.
+ Type *CombinedEltTy = nullptr;
+ unsigned TotalElems = 0;
+ for (AllocaInst *Member : AA.Members) {
+ Type *MemberTy = Member->getAllocatedType();
+ unsigned ElemCnt;
+ if (MemberTy->isSingleValueType() && !MemberTy->isVectorTy()) {
+ if (CombinedEltTy && CombinedEltTy != MemberTy)
+ return;
+
+ CombinedEltTy = MemberTy;
+ ElemCnt = 1;
+ } else {
+ MemberTy = getVectorTypeForAlloca(MemberTy);
+ // Failed to find a proper vector type or found a different element type,
+ // abort promotion.
+ if (!MemberTy ||
+ (CombinedEltTy && CombinedEltTy != MemberTy->getScalarType()))
+ return;
+ CombinedEltTy = MemberTy->getScalarType();
+ ElemCnt = cast<FixedVectorType>(MemberTy)->getNumElements();
+ }
+
+ AA.Vector.BaseLane[Member] = TotalElems;
+ TotalElems += ElemCnt;
+ }
+
+ AA.Vector.Ty = FixedVectorType::get(CombinedEltTy, TotalElems);
+ // Re-check the combined vector against the register-size limit.
+ const unsigned MaxElements =
+ (MaxVectorRegs * 32) / DL.getTypeSizeInBits(CombinedEltTy);
+ if (TotalElems > MaxElements) {
----------------
arsenm wrote:
This isn't an accurate register size count, but that's mostly a preexisting issue
https://github.com/llvm/llvm-project/pull/221870
More information about the llvm-commits
mailing list