[llvm] eb9091d - [InterleavedLoadCombine] Do not widen loads past a may-not-return instr (#223954)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 16 03:07:37 PDT 2026
Author: Madhur Amilkanthwar
Date: 2026-09-16T15:37:31+05:30
New Revision: eb9091da5c418a5c383a5da7fb39a1e22ba7c60d
URL: https://github.com/llvm/llvm-project/commit/eb9091da5c418a5c383a5da7fb39a1e22ba7c60d
DIFF: https://github.com/llvm/llvm-project/commit/eb9091da5c418a5c383a5da7fb39a1e22ba7c60d.diff
LOG: [InterleavedLoadCombine] Do not widen loads past a may-not-return instr (#223954)
The combined load reads the whole span at once and is inserted at the
first load, so it effectively hoists the later loads up to that point.
If an instruction between them may not transfer control to its successor
-- a call that might not return, or might throw -- then a load the
original program reached only conditionally would run unconditionally.
All combined loads are in one block, so bail unless the span from the
first to the last load is guaranteed to transfer execution to its
successor.
This gap predates the offset-index change; the old findPattern() did not
check for it either.
Assisted by AI tools.
Added:
llvm/test/CodeGen/AArch64/interleaved-load-combine-may-not-return.ll
Modified:
llvm/lib/CodeGen/InterleavedLoadCombinePass.cpp
Removed:
################################################################################
diff --git a/llvm/lib/CodeGen/InterleavedLoadCombinePass.cpp b/llvm/lib/CodeGen/InterleavedLoadCombinePass.cpp
index 3c93e293ca9f7..c5f093bf34a91 100644
--- a/llvm/lib/CodeGen/InterleavedLoadCombinePass.cpp
+++ b/llvm/lib/CodeGen/InterleavedLoadCombinePass.cpp
@@ -27,6 +27,7 @@
#include "llvm/Analysis/MemorySSAUpdater.h"
#include "llvm/Analysis/OptimizationRemarkEmitter.h"
#include "llvm/Analysis/TargetTransformInfo.h"
+#include "llvm/Analysis/ValueTracking.h"
#include "llvm/CodeGen/InterleavedLoadCombine.h"
#include "llvm/CodeGen/Passes.h"
#include "llvm/CodeGen/TargetLowering.h"
@@ -1183,6 +1184,20 @@ bool InterleavedLoadCombineImpl::combine(ArrayRef<VectorInfo *> InterleavedLoad,
}
assert(!LIs.empty() && "There are no LoadInst to combine");
+ // The wide load reads the whole span at once and is inserted at the first
+ // load, so widening must not pull a later load across an instruction that may
+ // not transfer control to its successor (e.g. a call that might not return or
+ // might throw). Otherwise a load the original program reached only
+ // conditionally would run unconditionally. All combined loads are in one
+ // block, so check the span from the first to the last is barrier-free.
+ LoadInst *Last = First;
+ for (auto *LI : LIs)
+ if (Last->comesBefore(LI))
+ Last = LI;
+ if (!isGuaranteedToTransferExecutionToSuccessor(First->getIterator(),
+ Last->getIterator()))
+ return false;
+
// It is necessary that insertion point dominates all final ShuffleVectorInst.
for (const VectorInfo *VI : InterleavedLoad) {
if (!DT.dominates(InsertionPoint, VI->SVI))
diff --git a/llvm/test/CodeGen/AArch64/interleaved-load-combine-may-not-return.ll b/llvm/test/CodeGen/AArch64/interleaved-load-combine-may-not-return.ll
new file mode 100644
index 0000000000000..f55322f23d57d
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/interleaved-load-combine-may-not-return.ll
@@ -0,0 +1,68 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=interleaved-load-combine < %s | FileCheck %s
+
+target triple = "arm64--linux-gnu"
+
+; A call with no memory effects but without willreturn may not return. It sits
+; between the two loads, so widening them into one load at the first load would
+; execute the second load's access even when the call never returns. The pass
+; must leave the loads alone.
+declare void @may_not_return() memory(none)
+
+; A call that is guaranteed to return does not block the transform.
+declare void @always_returns() memory(none) willreturn nounwind
+
+define void @no_combine_before_may_not_return(ptr %ptr, ptr %o1, ptr %o2) {
+; CHECK-LABEL: define void @no_combine_before_may_not_return(
+; CHECK-SAME: ptr [[PTR:%.*]], ptr [[O1:%.*]], ptr [[O2:%.*]]) {
+; CHECK-NEXT: [[P0:%.*]] = getelementptr inbounds <4 x float>, ptr [[PTR]], i64 0
+; CHECK-NEXT: [[P1:%.*]] = getelementptr inbounds <4 x float>, ptr [[PTR]], i64 1
+; CHECK-NEXT: [[A0:%.*]] = load <4 x float>, ptr [[P0]], align 16
+; CHECK-NEXT: call void @may_not_return()
+; CHECK-NEXT: [[A1:%.*]] = load <4 x float>, ptr [[P1]], align 16
+; CHECK-NEXT: [[AE:%.*]] = shufflevector <4 x float> [[A0]], <4 x float> [[A1]], <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-NEXT: [[AO:%.*]] = shufflevector <4 x float> [[A0]], <4 x float> [[A1]], <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-NEXT: store <4 x float> [[AE]], ptr [[O1]], align 16
+; CHECK-NEXT: store <4 x float> [[AO]], ptr [[O2]], align 16
+; CHECK-NEXT: ret void
+;
+ %p0 = getelementptr inbounds <4 x float>, ptr %ptr, i64 0
+ %p1 = getelementptr inbounds <4 x float>, ptr %ptr, i64 1
+ %a0 = load <4 x float>, ptr %p0, align 16
+ call void @may_not_return()
+ %a1 = load <4 x float>, ptr %p1, align 16
+ %ae = shufflevector <4 x float> %a0, <4 x float> %a1, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+ %ao = shufflevector <4 x float> %a0, <4 x float> %a1, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+ store <4 x float> %ae, ptr %o1, align 16
+ store <4 x float> %ao, ptr %o2, align 16
+ ret void
+}
+
+define void @combine_before_returning_call(ptr %ptr, ptr %o1, ptr %o2) {
+; CHECK-LABEL: define void @combine_before_returning_call(
+; CHECK-SAME: ptr [[PTR:%.*]], ptr [[O1:%.*]], ptr [[O2:%.*]]) {
+; CHECK-NEXT: [[P0:%.*]] = getelementptr inbounds <4 x float>, ptr [[PTR]], i64 0
+; CHECK-NEXT: [[P1:%.*]] = getelementptr inbounds <4 x float>, ptr [[PTR]], i64 1
+; CHECK-NEXT: [[INTERLEAVED_WIDE_LOAD:%.*]] = load <8 x float>, ptr [[P0]], align 16
+; CHECK-NEXT: [[A0:%.*]] = load <4 x float>, ptr [[P0]], align 16
+; CHECK-NEXT: call void @always_returns()
+; CHECK-NEXT: [[A1:%.*]] = load <4 x float>, ptr [[P1]], align 16
+; CHECK-NEXT: [[INTERLEAVED_SHUFFLE:%.*]] = shufflevector <8 x float> [[INTERLEAVED_WIDE_LOAD]], <8 x float> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-NEXT: [[AE:%.*]] = shufflevector <4 x float> [[A0]], <4 x float> [[A1]], <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-NEXT: [[INTERLEAVED_SHUFFLE1:%.*]] = shufflevector <8 x float> [[INTERLEAVED_WIDE_LOAD]], <8 x float> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-NEXT: [[AO:%.*]] = shufflevector <4 x float> [[A0]], <4 x float> [[A1]], <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-NEXT: store <4 x float> [[INTERLEAVED_SHUFFLE]], ptr [[O1]], align 16
+; CHECK-NEXT: store <4 x float> [[INTERLEAVED_SHUFFLE1]], ptr [[O2]], align 16
+; CHECK-NEXT: ret void
+;
+ %p0 = getelementptr inbounds <4 x float>, ptr %ptr, i64 0
+ %p1 = getelementptr inbounds <4 x float>, ptr %ptr, i64 1
+ %a0 = load <4 x float>, ptr %p0, align 16
+ call void @always_returns()
+ %a1 = load <4 x float>, ptr %p1, align 16
+ %ae = shufflevector <4 x float> %a0, <4 x float> %a1, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+ %ao = shufflevector <4 x float> %a0, <4 x float> %a1, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+ store <4 x float> %ae, ptr %o1, align 16
+ store <4 x float> %ao, ptr %o2, align 16
+ ret void
+}
More information about the llvm-commits
mailing list