[llvm] [VectorCombine] Fold contiguous loads into a single vector load (PR #185736)
Matt Arsenault via llvm-commits
llvm-commits at lists.llvm.org
Wed Jun 3 06:56:17 PDT 2026
================
@@ -6120,6 +6055,167 @@ bool VectorCombine::shrinkPhiOfShuffles(Instruction &I) {
return true;
}
+/// Try to fold lanes assembled from contiguous vector-load elements into one
+/// load of the result type.
+///
+/// 1. Trace lanes:
+/// result lane 0 result lane 1 ... result lane N
+/// | | |
+/// +------- look through shuffles ---------+
+/// |
+/// source load + source lane
+///
+/// 2. Check layout:
+/// same base pointer and contiguous offsets?
+///
+/// 3. Model old cost:
+/// current op + unique loads + original GEPs
+///
+/// 4. Model new cost:
+/// ptradd(base, start byte offset) + one vector load
+///
+/// 5. Replace:
+/// if NewCost is cheaper
+///
+/// For example:
+///
+/// %p = getelementptr float, ptr %base, i64 4
+/// %v = load <4 x float>, ptr %p
+/// base+16 base+20 base+24 base+28
+/// lane 0 lane 1 lane 2 lane 3
+/// | |
+/// +---------+ contiguous
+/// |
+/// %r = shufflevector %v, poison, <2, 3>
+/// |
+/// v
+/// %q = getelementptr i8, ptr %base, i64 24
+/// %r = load <2 x float>, ptr %q
+bool VectorCombine::foldContiguousLoads(Instruction &I) {
+ auto *VT = dyn_cast<FixedVectorType>(I.getType());
+ if (!VT || I.use_empty())
+ return false;
+
+ Type *EltTy = VT->getElementType();
+ if (!DL->typeSizeEqualsStoreSize(EltTy))
+ return false;
+
+ uint64_t MaxInt64 =
+ static_cast<uint64_t>(std::numeric_limits<int64_t>::max());
+ uint64_t ElementSizeBits = DL->getTypeStoreSizeInBits(EltTy);
+ assert((ElementSizeBits <= MaxInt64) && "element size far too large?");
+ int64_t ElementSizeBitsI64 = static_cast<int64_t>(ElementSizeBits);
+ unsigned NumElts = VT->getNumElements();
+ Value *CommonBase = nullptr;
+ int64_t StartBitOffset = 0, FirstLoadByteOffset = 0;
+ LoadInst *FirstLI = nullptr;
+ SmallPtrSet<LoadInst *, 4> Loads;
+ for (unsigned Lane = 0; Lane < NumElts; ++Lane) {
+ // Step 1: Trace this result lane through shuffle users to find the source
+ // instruction and the lane selected from it.
+ InstLane IL = lookThroughShuffles(&I, Lane);
+ if (!IL.first)
+ return false;
+
+ auto *LI = dyn_cast<LoadInst>(IL.first);
+ if (!LI)
+ return false;
+
+ if (!LI->isSimple() || !LI->hasOneUse())
+ return false;
+
+ auto *LIVTy = dyn_cast<FixedVectorType>(LI->getType());
+ if (!LIVTy || LIVTy->getElementType() != EltTy)
+ return false;
+
+ if (LI->getParent() != I.getParent())
+ return false;
+
+ if (isMemModifiedBetween(std::next(LI->getIterator()), I.getIterator(),
+ MemoryLocation::get(LI), AA))
+ return false;
+
+ // Step 2: Convert the load pointer and selected source lane into an
+ // absolute bit offset: byte offset of the load pointer plus lane offset
+ // within the load.
+ int64_t LoadByteOffset = 0;
+ Value *Base = GetPointerBaseWithConstantOffset(LI->getPointerOperand(),
+ LoadByteOffset, *DL);
+ int64_t SourceLaneBitOffset =
+ (LoadByteOffset * 8) + (IL.second * ElementSizeBitsI64);
+ if (Lane == 0) {
+ if (SourceLaneBitOffset % 8 != 0)
+ return false;
+
+ CommonBase = Base;
+ StartBitOffset = SourceLaneBitOffset;
+ FirstLI = LI;
+ FirstLoadByteOffset = LoadByteOffset;
+ } else {
+ // Step 2: All later result lanes must use the same underlying base
+ // pointer and appear at the element-stride offset expected from the first
+ // result lane.
+ if (Base != CommonBase)
+ return false;
+
+ assert(Lane <= MaxInt64 / static_cast<uint64_t>(ElementSizeBitsI64) &&
+ "lane offset far too large?");
+
+ int64_t ExpectedBitOffset =
+ StartBitOffset + static_cast<int64_t>(Lane) * ElementSizeBitsI64;
+ if (SourceLaneBitOffset != ExpectedBitOffset)
+ return false;
+ }
+
+ Loads.insert(LI);
+ }
+
+ // Step 3: Model the current form: the shuffle-like instruction, each unique
+ // source load, and any GEP used to compute those load addresses.
+ InstructionCost OldCost = TTI.getInstructionCost(&I, CostKind);
+ for (LoadInst *LI : Loads) {
+ OldCost += TTI.getInstructionCost(LI, CostKind);
+ if (auto *GEP = dyn_cast<GEPOperator>(LI->getPointerOperand())) {
+ SmallVector<const Value *> Indices(GEP->indices());
+ OldCost +=
+ TTI.getGEPCost(GEP->getSourceElementType(), GEP->getPointerOperand(),
+ Indices, LI->getType(), CostKind);
+ }
+ }
+
+ int64_t StartByteOffset = StartBitOffset / 8;
+ Type *IndexTy = DL->getIndexType(CommonBase->getType());
+ auto *StartByteOffsetValue = ConstantInt::get(IndexTy, StartByteOffset,
+ /*isSigned=*/true);
+ int64_t ByteOffsetFromFirstLoad = StartByteOffset - FirstLoadByteOffset;
+ Align NewAlign =
+ commonAlignment(FirstLI->getAlign(), ByteOffsetFromFirstLoad);
+ // Step 4: Model the replacement: one vector load from the adjusted alignment
+ // and the byte-offset GEP that CreatePtrAdd will emit.
+ InstructionCost NewCost =
+ TTI.getMemoryOpCost(Instruction::Load, VT, NewAlign,
+ FirstLI->getPointerAddressSpace(), CostKind);
+ SmallVector<const Value *> NewIndices = {StartByteOffsetValue};
+ NewCost +=
+ TTI.getGEPCost(Builder.getInt8Ty(), CommonBase, NewIndices, VT, CostKind);
+
+ LLVM_DEBUG(dbgs() << "Found contiguous loads to fold: " << I
+ << "\n OldCost: " << OldCost << " vs NewCost: " << NewCost
+ << "\n");
+
+ if (OldCost <= NewCost)
+ return false;
+
+ // Step 5: Emit the same byte-offset GEP modeled above, then load the
+ // contiguous result vector from it.
+ Value *NewBasePtr = Builder.CreatePtrAdd(CommonBase, StartByteOffsetValue);
+ LoadInst *NewLoad = Builder.CreateAlignedLoad(VT, NewBasePtr, NewAlign);
+ copyMetadataForLoad(*NewLoad, *FirstLI);
----------------
arsenm wrote:
Missing test for metadata preservation
https://github.com/llvm/llvm-project/pull/185736
More information about the llvm-commits
mailing list