[llvm] [VectorCombine] Fold contiguous loads into a single vector load (PR #185736)

Matt Arsenault via llvm-commits llvm-commits at lists.llvm.org
Wed Jun 3 06:56:17 PDT 2026


================
@@ -6120,6 +6055,167 @@ bool VectorCombine::shrinkPhiOfShuffles(Instruction &I) {
   return true;
 }
 
+/// Try to fold lanes assembled from contiguous vector-load elements into one
+/// load of the result type.
+///
+///   1. Trace lanes:
+///      result lane 0   result lane 1   ...   result lane N
+///           |               |                       |
+///           +------- look through shuffles ---------+
+///                           |
+///                 source load + source lane
+///
+///   2. Check layout:
+///                 same base pointer and contiguous offsets?
+///
+///   3. Model old cost:
+///                 current op + unique loads + original GEPs
+///
+///   4. Model new cost:
+///                 ptradd(base, start byte offset) + one vector load
+///
+///   5. Replace:
+///                 if NewCost is cheaper
+///
+/// For example:
+///
+///   %p = getelementptr float, ptr %base, i64 4
+///   %v = load <4 x float>, ptr %p
+///          base+16   base+20   base+24   base+28
+///          lane 0    lane 1    lane 2    lane 3
+///                              |         |
+///                              +---------+  contiguous
+///                                  |
+///   %r = shufflevector %v, poison, <2, 3>
+///                                  |
+///                                  v
+///   %q = getelementptr i8, ptr %base, i64 24
+///   %r = load <2 x float>, ptr %q
+bool VectorCombine::foldContiguousLoads(Instruction &I) {
+  auto *VT = dyn_cast<FixedVectorType>(I.getType());
+  if (!VT || I.use_empty())
+    return false;
+
+  Type *EltTy = VT->getElementType();
+  if (!DL->typeSizeEqualsStoreSize(EltTy))
+    return false;
+
+  uint64_t MaxInt64 =
+      static_cast<uint64_t>(std::numeric_limits<int64_t>::max());
+  uint64_t ElementSizeBits = DL->getTypeStoreSizeInBits(EltTy);
+  assert((ElementSizeBits <= MaxInt64) && "element size far too large?");
+  int64_t ElementSizeBitsI64 = static_cast<int64_t>(ElementSizeBits);
+  unsigned NumElts = VT->getNumElements();
+  Value *CommonBase = nullptr;
+  int64_t StartBitOffset = 0, FirstLoadByteOffset = 0;
+  LoadInst *FirstLI = nullptr;
+  SmallPtrSet<LoadInst *, 4> Loads;
+  for (unsigned Lane = 0; Lane < NumElts; ++Lane) {
+    // Step 1: Trace this result lane through shuffle users to find the source
+    // instruction and the lane selected from it.
+    InstLane IL = lookThroughShuffles(&I, Lane);
+    if (!IL.first)
+      return false;
+
+    auto *LI = dyn_cast<LoadInst>(IL.first);
+    if (!LI)
+      return false;
+
+    if (!LI->isSimple() || !LI->hasOneUse())
+      return false;
+
+    auto *LIVTy = dyn_cast<FixedVectorType>(LI->getType());
+    if (!LIVTy || LIVTy->getElementType() != EltTy)
+      return false;
+
+    if (LI->getParent() != I.getParent())
+      return false;
+
+    if (isMemModifiedBetween(std::next(LI->getIterator()), I.getIterator(),
+                             MemoryLocation::get(LI), AA))
+      return false;
+
+    // Step 2: Convert the load pointer and selected source lane into an
+    // absolute bit offset: byte offset of the load pointer plus lane offset
+    // within the load.
+    int64_t LoadByteOffset = 0;
+    Value *Base = GetPointerBaseWithConstantOffset(LI->getPointerOperand(),
+                                                   LoadByteOffset, *DL);
+    int64_t SourceLaneBitOffset =
+        (LoadByteOffset * 8) + (IL.second * ElementSizeBitsI64);
+    if (Lane == 0) {
+      if (SourceLaneBitOffset % 8 != 0)
+        return false;
+
+      CommonBase = Base;
+      StartBitOffset = SourceLaneBitOffset;
+      FirstLI = LI;
+      FirstLoadByteOffset = LoadByteOffset;
+    } else {
+      // Step 2: All later result lanes must use the same underlying base
+      // pointer and appear at the element-stride offset expected from the first
+      // result lane.
+      if (Base != CommonBase)
+        return false;
+
+      assert(Lane <= MaxInt64 / static_cast<uint64_t>(ElementSizeBitsI64) &&
+             "lane offset far too large?");
+
+      int64_t ExpectedBitOffset =
+          StartBitOffset + static_cast<int64_t>(Lane) * ElementSizeBitsI64;
+      if (SourceLaneBitOffset != ExpectedBitOffset)
+        return false;
+    }
+
+    Loads.insert(LI);
+  }
+
+  // Step 3: Model the current form: the shuffle-like instruction, each unique
+  // source load, and any GEP used to compute those load addresses.
+  InstructionCost OldCost = TTI.getInstructionCost(&I, CostKind);
+  for (LoadInst *LI : Loads) {
+    OldCost += TTI.getInstructionCost(LI, CostKind);
+    if (auto *GEP = dyn_cast<GEPOperator>(LI->getPointerOperand())) {
+      SmallVector<const Value *> Indices(GEP->indices());
+      OldCost +=
+          TTI.getGEPCost(GEP->getSourceElementType(), GEP->getPointerOperand(),
+                         Indices, LI->getType(), CostKind);
+    }
+  }
+
+  int64_t StartByteOffset = StartBitOffset / 8;
+  Type *IndexTy = DL->getIndexType(CommonBase->getType());
+  auto *StartByteOffsetValue = ConstantInt::get(IndexTy, StartByteOffset,
+                                                /*isSigned=*/true);
+  int64_t ByteOffsetFromFirstLoad = StartByteOffset - FirstLoadByteOffset;
+  Align NewAlign =
+      commonAlignment(FirstLI->getAlign(), ByteOffsetFromFirstLoad);
+  // Step 4: Model the replacement: one vector load from the adjusted alignment
+  // and the byte-offset GEP that CreatePtrAdd will emit.
+  InstructionCost NewCost =
+      TTI.getMemoryOpCost(Instruction::Load, VT, NewAlign,
+                          FirstLI->getPointerAddressSpace(), CostKind);
+  SmallVector<const Value *> NewIndices = {StartByteOffsetValue};
+  NewCost +=
+      TTI.getGEPCost(Builder.getInt8Ty(), CommonBase, NewIndices, VT, CostKind);
+
+  LLVM_DEBUG(dbgs() << "Found contiguous loads to fold: " << I
+                    << "\n  OldCost: " << OldCost << " vs NewCost: " << NewCost
+                    << "\n");
+
+  if (OldCost <= NewCost)
+    return false;
+
+  // Step 5: Emit the same byte-offset GEP modeled above, then load the
+  // contiguous result vector from it.
+  Value *NewBasePtr = Builder.CreatePtrAdd(CommonBase, StartByteOffsetValue);
+  LoadInst *NewLoad = Builder.CreateAlignedLoad(VT, NewBasePtr, NewAlign);
+  copyMetadataForLoad(*NewLoad, *FirstLI);
----------------
arsenm wrote:

Missing test for metadata preservation 

https://github.com/llvm/llvm-project/pull/185736


More information about the llvm-commits mailing list