[llvm] [VectorCombine] Fold contiguous loads into a single vector load (PR #185736)

Matt Arsenault via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 07:11:03 PDT 2026


================
@@ -6864,6 +6866,181 @@ bool VectorCombine::shrinkPhiOfShuffles(Instruction &I) {
   return true;
 }
 
+/// Try to fold lanes assembled from contiguous vector-load elements into one
+/// load of the result type.
+///
+///   1. Trace lanes:
+///      result lane 0   result lane 1   ...   result lane N
+///           |               |                       |
+///           +------- look through shuffles ---------+
+///                           |
+///                 source load + source lane
+///
+///   2. Check layout:
+///                 same base pointer and contiguous offsets?
+///
+///   3. Model old cost:
+///                 current op + unique loads + original GEPs
+///
+///   4. Model new cost:
+///                 ptradd(base, start byte offset) + one vector load
+///
+///   5. Replace:
+///                 if NewCost is cheaper
+///
+/// For example:
+///
+///   %p = getelementptr float, ptr %base, i64 4
+///   %v = load <4 x float>, ptr %p
+///          base+16   base+20   base+24   base+28
+///          lane 0    lane 1    lane 2    lane 3
+///                              |         |
+///                              +---------+  contiguous
+///                                  |
+///   %r = shufflevector %v, poison, <2, 3>
+///                                  |
+///                                  v
+///   %q = getelementptr i8, ptr %base, i64 24
+///   %r = load <2 x float>, ptr %q
+bool VectorCombine::foldContiguousLoads(Instruction &I) {
+  auto *VT = dyn_cast<FixedVectorType>(I.getType());
+  if (!VT || I.use_empty())
+    return false;
+
+  Type *EltTy = VT->getElementType();
+  if (!DL->typeSizeEqualsStoreSize(EltTy))
+    return false;
+
+  uint64_t ElementSizeBytes = DL->getTypeStoreSize(EltTy);
+  unsigned NumElts = VT->getNumElements();
+  Value *CommonBase = nullptr;
+  APInt StartByteOffset(1, 0), FirstLoadByteOffset(1, 0);
+  LoadInst *FirstLI = nullptr;
+  SmallPtrSet<LoadInst *, 4> Loads;
+  for (unsigned Lane = 0; Lane < NumElts; ++Lane) {
+    // Step 1: Trace this result lane through shuffle users to find the source
+    // instruction and the lane selected from it.
+    InstLane IL = lookThroughShuffles(&I, Lane);
+    if (!IL.first)
+      return false;
+
+    auto *LI = dyn_cast<LoadInst>(IL.first);
+    if (!LI)
+      return false;
+
+    if (!LI->isSimple() || !LI->hasOneUse())
+      return false;
+
+    auto *LIVTy = dyn_cast<FixedVectorType>(LI->getType());
+    if (!LIVTy || LIVTy->getElementType() != EltTy)
+      return false;
+
+    if (LI->getParent() != I.getParent())
+      return false;
+
+    if (isMemModifiedBetween(std::next(LI->getIterator()), I.getIterator(),
+                             MemoryLocation::get(LI), AA))
+      return false;
+
+    // Step 2: Convert the load pointer and selected source lane into a byte
+    // offset in the pointer index type.
+    int64_t LoadByteOffset = 0;
+    Value *Base = GetPointerBaseWithConstantOffset(LI->getPointerOperand(),
+                                                   LoadByteOffset, *DL);
+    unsigned IndexBits = DL->getIndexTypeSizeInBits(Base->getType());
+    APInt LoadByteOffsetAP(IndexBits, LoadByteOffset, /*isSigned=*/true);
+    APInt SourceByteOffset =
+        LoadByteOffsetAP + static_cast<uint64_t>(IL.second) * ElementSizeBytes;
+    if (Lane == 0) {
+      CommonBase = Base;
+      StartByteOffset = SourceByteOffset;
+      FirstLI = LI;
+      FirstLoadByteOffset = LoadByteOffsetAP;
+    } else {
+      // Step 2: All later result lanes must use the same underlying base
+      // pointer and appear at the element-stride offset expected from the first
+      // result lane.
+      if (Base != CommonBase)
+        return false;
+      if (IndexBits != StartByteOffset.getBitWidth())
+        return false;
+
+      APInt ExpectedByteOffset = StartByteOffset + Lane * ElementSizeBytes;
+      if (SourceByteOffset != ExpectedByteOffset)
+        return false;
+    }
+
+    Loads.insert(LI);
+  }
+
+  // Step 3: Model the current form: the shuffle-like instruction, each unique
+  // source load, and any GEP used to compute those load addresses.
+  InstructionCost OldCost = TTI.getInstructionCost(&I, CostKind);
+  for (LoadInst *LI : Loads) {
+    OldCost += TTI.getInstructionCost(LI, CostKind);
+    if (auto *GEP = dyn_cast<GetElementPtrInst>(LI->getPointerOperand());
+        GEP && GEP->hasOneUse()) {
+      SmallVector<const Value *> Indices(GEP->indices());
+      OldCost +=
+          TTI.getGEPCost(GEP->getSourceElementType(), GEP->getPointerOperand(),
+                         Indices, LI->getType(), CostKind);
+    }
+  }
+
+  Type *IndexTy = DL->getIndexType(CommonBase->getType());
+  auto *StartByteOffsetValue = ConstantInt::get(IndexTy, StartByteOffset);
+  APInt ByteOffsetFromFirstLoad = StartByteOffset - FirstLoadByteOffset;
+  unsigned AlignOffsetBits =
+      std::min<unsigned>(ByteOffsetFromFirstLoad.getBitWidth(), 64);
+  uint64_t ByteOffsetFromFirstLoadForAlign =
+      ByteOffsetFromFirstLoad.getLoBits(AlignOffsetBits).getZExtValue();
----------------
arsenm wrote:

Avoid this special case? Can't you keep everything in APInt? 

https://github.com/llvm/llvm-project/pull/185736


More information about the llvm-commits mailing list