[llvm] 3973230 - [InterleavedLoadCombine] Do not combine loads across basic blocks (#223915)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 16 01:05:12 PDT 2026


Author: Madhur Amilkanthwar
Date: 2026-09-16T13:35:07+05:30
New Revision: 39732305282242041dad8c8edad6a6b5804f1d3a

URL: https://github.com/llvm/llvm-project/commit/39732305282242041dad8c8edad6a6b5804f1d3a
DIFF: https://github.com/llvm/llvm-project/commit/39732305282242041dad8c8edad6a6b5804f1d3a.diff

LOG: [InterleavedLoadCombine] Do not combine loads across basic blocks (#223915)

This pass turns a group of interleaved loads into a single wide load
inserted at the first load. That is only valid when all of the loads are
in the same basic block. Otherwise the wide load can read memory that
the original program only accessed on a conditional path.

The offset index keyed candidates on base pointer, type and offset only,
so it could pair loads from different blocks. Restrict each block's
matching to candidates whose loads are in that block. This also keeps
the candidate list and index per block, so they stay small.

Fixes a miscompile introduced by #213053.

Added: 
    llvm/test/CodeGen/AArch64/interleaved-load-combine-crossblock.ll

Modified: 
    llvm/lib/CodeGen/InterleavedLoadCombinePass.cpp

Removed: 
    


################################################################################
diff  --git a/llvm/lib/CodeGen/InterleavedLoadCombinePass.cpp b/llvm/lib/CodeGen/InterleavedLoadCombinePass.cpp
index e9cd1e58f174d..3c93e293ca9f7 100644
--- a/llvm/lib/CodeGen/InterleavedLoadCombinePass.cpp
+++ b/llvm/lib/CodeGen/InterleavedLoadCombinePass.cpp
@@ -639,11 +639,13 @@ static raw_ostream &operator<<(raw_ostream &OS, const Polynomial &S) {
 #endif
 
 /// Address key of a candidate's first vector element: the common base pointer,
-/// the vector type and the offset polynomial. Candidates are matched one basic
-/// block at a time, so the block is implicit in the index. Two candidates
-/// belong to the same interleaved group iff their keys agree on everything but
-/// the constant offset, so consecutive elements are located by building the
-/// neighbouring keys and looking them up.
+/// the vector type and the offset polynomial. Candidates are collected and
+/// matched one basic block at a time and only candidates whose loads live in
+/// that block take part (see run()), so the block is common to a whole index
+/// and need not be part of the key. Two candidates belong to the same
+/// interleaved group iff their keys agree on everything but the constant
+/// offset, so consecutive elements are located by building the neighbouring
+/// keys and looking them up.
 struct OffsetKey {
   Value *PV;
   FixedVectorType *VTy;
@@ -1246,8 +1248,9 @@ bool InterleavedLoadCombineImpl::run() {
 
   // Start with the highest factor to avoid combining and recombining.
   for (unsigned Factor = MaxFactor; Factor >= 2; Factor--) {
-    // Matching only ever pairs candidates from the same block, so process one
-    // block at a time and keep the candidate list and offset index small.
+    // Process one block at a time. A group can only be combined when all of its
+    // loads are in a single block, so keeping the candidate list and the offset
+    // index per block keeps both small.
     for (BasicBlock &BB : F) {
       std::list<VectorInfo> Candidates;
       for (Instruction &I : BB) {
@@ -1260,13 +1263,18 @@ bool InterleavedLoadCombineImpl::run() {
           continue;
 
         Candidates.emplace_back(cast<FixedVectorType>(SVI->getType()));
+        VectorInfo &C = Candidates.back();
 
-        if (!VectorInfo::computeFromSVI(SVI, Candidates.back(), DL)) {
+        if (!VectorInfo::computeFromSVI(SVI, C, DL) ||
+            !C.isInterleaved(Factor, DL)) {
           Candidates.pop_back();
           continue;
         }
 
-        if (!Candidates.back().isInterleaved(Factor, DL))
+        // Only combine loads that live in the block being processed. Widening
+        // over loads from another block could read memory that is only
+        // conditionally accessed.
+        if (C.BB != &BB)
           Candidates.pop_back();
       }
 

diff  --git a/llvm/test/CodeGen/AArch64/interleaved-load-combine-crossblock.ll b/llvm/test/CodeGen/AArch64/interleaved-load-combine-crossblock.ll
new file mode 100644
index 0000000000000..88b0481447403
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/interleaved-load-combine-crossblock.ll
@@ -0,0 +1,50 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=interleaved-load-combine < %s | FileCheck %s
+
+target triple = "arm64--linux-gnu"
+
+; The shuffles share a block but their loads do not. Combining would hoist a
+; wide load into %entry over memory only accessed when %c is true, so the loads
+; must be left untouched.
+
+define void @no_combine_across_blocks(ptr %ptr, i1 %c, ptr %o1, ptr %o2) {
+; CHECK-LABEL: define void @no_combine_across_blocks(
+; CHECK-SAME: ptr [[PTR:%.*]], i1 [[C:%.*]], ptr [[O1:%.*]], ptr [[O2:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr inbounds float, ptr [[PTR]], i64 0
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr inbounds float, ptr [[PTR]], i64 3
+; CHECK-NEXT:    [[LD1:%.*]] = load <4 x float>, ptr [[P0]], align 4
+; CHECK-NEXT:    [[LD2:%.*]] = load <4 x float>, ptr [[P3]], align 4
+; CHECK-NEXT:    br i1 [[C]], label %[[THEN:.*]], label %[[EXIT:.*]]
+; CHECK:       [[THEN]]:
+; CHECK-NEXT:    [[P4:%.*]] = getelementptr inbounds float, ptr [[PTR]], i64 4
+; CHECK-NEXT:    [[LD3:%.*]] = load <4 x float>, ptr [[P0]], align 4
+; CHECK-NEXT:    [[LD4:%.*]] = load <4 x float>, ptr [[P4]], align 4
+; CHECK-NEXT:    [[C0:%.*]] = shufflevector <4 x float> [[LD1]], <4 x float> [[LD2]], <4 x i32> <i32 0, i32 2, i32 5, i32 7>
+; CHECK-NEXT:    [[C1:%.*]] = shufflevector <4 x float> [[LD3]], <4 x float> [[LD4]], <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-NEXT:    store <4 x float> [[C0]], ptr [[O1]], align 4
+; CHECK-NEXT:    store <4 x float> [[C1]], ptr [[O2]], align 4
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %p0 = getelementptr inbounds float, ptr %ptr, i64 0
+  %p3 = getelementptr inbounds float, ptr %ptr, i64 3
+  %ld1 = load <4 x float>, ptr %p0, align 4
+  %ld2 = load <4 x float>, ptr %p3, align 4
+  br i1 %c, label %then, label %exit
+
+then:
+  %p4 = getelementptr inbounds float, ptr %ptr, i64 4
+  %ld3 = load <4 x float>, ptr %p0, align 4
+  %ld4 = load <4 x float>, ptr %p4, align 4
+  %c0 = shufflevector <4 x float> %ld1, <4 x float> %ld2, <4 x i32> <i32 0, i32 2, i32 5, i32 7>
+  %c1 = shufflevector <4 x float> %ld3, <4 x float> %ld4, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+  store <4 x float> %c0, ptr %o1, align 4
+  store <4 x float> %c1, ptr %o2, align 4
+  br label %exit
+
+exit:
+  ret void
+}


        


More information about the llvm-commits mailing list