[llvm] [LV] Drop prefetches from vector loops (PR #226895)
Dávid Bolvanský via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 27 23:43:28 PDT 2026
https://github.com/davidbolvansky created https://github.com/llvm/llvm-project/pull/226895
`llvm.prefetch` can currently prevent an otherwise vectorizable loop from being vectorized. For example:
```c
void add(int *restrict a, int *restrict b, int n) {
for (int i = 0; i < n; ++i) {
a[i] += b[i];
__builtin_prefetch(&b[i + 8]);
}
}
```
Permit LoopVectorize to omit prefetches from the vector loop. Ignore them during memory dependence and legality checks, and remove their VPlan recipes before call-widening decisions. VPlan DCE then removes address calculations used only by the hint. The scalar remainder retains its original prefetch.
This is sound because `llvm.prefetch` has no observable behavior and cannot trap, so removing it does not change program semantics. Conditional prefetches can be handled in the same way.
>From 3013176507fdf314485ab5237f4422a7ca4d640c Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?D=C3=A1vid=20Bolvansk=C3=BD?= <david.bolvansky at gmail.com>
Date: Mon, 28 Sep 2026 08:42:54 +0200
Subject: [PATCH] [LV] Drop prefetches from vector loops
---
llvm/lib/Analysis/LoopAccessAnalysis.cpp | 7 +
.../Vectorize/LoopVectorizationLegality.cpp | 9 +
.../Transforms/Vectorize/VPlanTransforms.cpp | 15 ++
.../test/Transforms/LoopVectorize/prefetch.ll | 161 ++++++++++++++++++
4 files changed, 192 insertions(+)
create mode 100644 llvm/test/Transforms/LoopVectorize/prefetch.ll
diff --git a/llvm/lib/Analysis/LoopAccessAnalysis.cpp b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
index 804ffe5bb2edd..ae63ee3138afc 100644
--- a/llvm/lib/Analysis/LoopAccessAnalysis.cpp
+++ b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
@@ -50,6 +50,7 @@
#include "llvm/IR/Instructions.h"
#include "llvm/IR/IntrinsicInst.h"
#include "llvm/IR/PassManager.h"
+#include "llvm/IR/PatternMatch.h"
#include "llvm/IR/Type.h"
#include "llvm/IR/Value.h"
#include "llvm/IR/ValueHandle.h"
@@ -2732,6 +2733,12 @@ bool LoopAccessInfo::analyzeLoop(AAResults *AA, const LoopInfo *LI,
// Scan the BB and collect legal loads and stores. Also detect any
// convergent instructions.
for (Instruction &I : *BB) {
+ // Prefetches are optional hints and are dropped by the loop vectorizer.
+ // Do not let their pointer operands affect memory dependence analysis.
+ if (PatternMatch::match(&I,
+ PatternMatch::m_Intrinsic<Intrinsic::prefetch>()))
+ continue;
+
if (auto *Call = dyn_cast<CallBase>(&I)) {
if (Call->isConvergent())
HasConvergentOp = true;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index 862470ce0de6d..cd8f2bcbab172 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -833,6 +833,11 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
BasicBlock *BB = I.getParent();
BasicBlock *Header = TheLoop->getHeader();
+ // llvm.prefetch is an optional hint with no semantic effect. It is omitted
+ // from the vector loop when VPlan recipes are created.
+ if (match(&I, m_Intrinsic<Intrinsic::prefetch>()))
+ return true;
+
if (auto *Phi = dyn_cast<PHINode>(&I)) {
Type *PhiTy = Phi->getType();
// Check that this PHI type is allowed.
@@ -1381,6 +1386,10 @@ bool LoopVectorizationLegality::blockCanBePredicated(
BasicBlock *BB, SmallPtrSetImpl<Value *> &SafePtrs,
SmallPtrSetImpl<const Instruction *> &MaskedOp) const {
for (Instruction &I : *BB) {
+ // Prefetches are omitted from the vector loop.
+ if (match(&I, m_Intrinsic<Intrinsic::prefetch>()))
+ continue;
+
// We can predicate blocks with calls to assume, as long as we drop them in
// case we flatten the CFG via predication.
if (match(&I, m_Intrinsic<Intrinsic::assume>())) {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 13abd173502e9..f3a03f3fb1baf 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -35,6 +35,7 @@
#include "llvm/Analysis/VectorUtils.h"
#include "llvm/IR/Intrinsics.h"
#include "llvm/IR/Metadata.h"
+#include "llvm/IR/PatternMatch.h"
#include "llvm/Support/Casting.h"
#include "llvm/Support/CommandLine.h"
#include "llvm/Support/TypeSize.h"
@@ -89,6 +90,14 @@ bool VPlanTransforms::tryToConvertVPInstructionsToVPRecipes(
Instruction *Inst = cast<Instruction>(VPV->getUnderlyingValue());
+ // llvm.prefetch is an optional hint. Drop it from the vector loop;
+ // subsequent VPlan DCE removes address computations used only by it.
+ if (PatternMatch::match(
+ Inst, PatternMatch::m_Intrinsic<Intrinsic::prefetch>())) {
+ Ingredient.eraseFromParent();
+ continue;
+ }
+
// Atomic accesses and fences have ordering/atomicity semantics that
// cannot be preserved by lane-wise widening.
if (isa<AtomicRMWInst, AtomicCmpXchgInst, FenceInst>(Inst))
@@ -5900,6 +5909,12 @@ void VPlanTransforms::makeCallWideningDecisions(VPlan &Plan, VFRange &Range,
continue;
auto *CI = cast<CallInst>(VPI.getUnderlyingInstr());
+ if (PatternMatch::match(
+ CI, PatternMatch::m_Intrinsic<Intrinsic::prefetch>())) {
+ VPI.eraseFromParent();
+ continue;
+ }
+
SmallVector<VPValue *, 4> Ops(VPI.op_begin(),
VPI.op_begin() + CI->arg_size());
diff --git a/llvm/test/Transforms/LoopVectorize/prefetch.ll b/llvm/test/Transforms/LoopVectorize/prefetch.ll
new file mode 100644
index 0000000000000..95478f9ab4ad9
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/prefetch.ll
@@ -0,0 +1,161 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt < %s -passes=loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S | FileCheck %s
+
+declare void @llvm.prefetch.p0(ptr readonly, i32 immarg, i32 immarg, i32 immarg)
+
+define void @prefetch(ptr noalias %a, ptr noalias %b, i64 %n) {
+; CHECK-LABEL: define void @prefetch(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: store <4 x i32> [[TMP3]], ptr [[TMP1]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT: [[A_PTR:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[B_PTR:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[A_VAL:%.*]] = load i32, ptr [[A_PTR]], align 4
+; CHECK-NEXT: [[B_VAL:%.*]] = load i32, ptr [[B_PTR]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[A_VAL]], [[B_VAL]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[A_PTR]], align 4
+; CHECK-NEXT: [[LOOKAHEAD:%.*]] = add nuw i64 [[IV]], 8
+; CHECK-NEXT: [[PREFETCH_PTR:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[LOOKAHEAD]]
+; CHECK-NEXT: call void @llvm.prefetch.p0(ptr [[PREFETCH_PTR]], i32 0, i32 3, i32 1)
+; CHECK-NEXT: br label %[[LATCH]]
+; CHECK: [[LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %a.ptr = getelementptr inbounds i32, ptr %a, i64 %iv
+ %b.ptr = getelementptr inbounds i32, ptr %b, i64 %iv
+ %a.val = load i32, ptr %a.ptr, align 4
+ %b.val = load i32, ptr %b.ptr, align 4
+ %sum = add i32 %a.val, %b.val
+ store i32 %sum, ptr %a.ptr, align 4
+ %lookahead = add nuw i64 %iv, 8
+ %prefetch.ptr = getelementptr inbounds i32, ptr %b, i64 %lookahead
+ call void @llvm.prefetch.p0(ptr %prefetch.ptr, i32 0, i32 3, i32 1)
+ br label %latch
+
+latch:
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @conditional_prefetch(ptr noalias %a, ptr noalias %b, i64 %n) {
+; CHECK-LABEL: define void @conditional_prefetch(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: store <4 x i32> [[TMP3]], ptr [[TMP1]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT: [[A_PTR:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[B_PTR:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[A_VAL:%.*]] = load i32, ptr [[A_PTR]], align 4
+; CHECK-NEXT: [[B_VAL:%.*]] = load i32, ptr [[B_PTR]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[A_VAL]], [[B_VAL]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[A_PTR]], align 4
+; CHECK-NEXT: [[POSITIVE:%.*]] = icmp sgt i32 [[SUM]], 0
+; CHECK-NEXT: br i1 [[POSITIVE]], label %[[PREFETCH:.*]], label %[[LATCH]]
+; CHECK: [[PREFETCH]]:
+; CHECK-NEXT: [[LOOKAHEAD:%.*]] = add nuw i64 [[IV]], 8
+; CHECK-NEXT: [[PREFETCH_PTR:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[LOOKAHEAD]]
+; CHECK-NEXT: call void @llvm.prefetch.p0(ptr [[PREFETCH_PTR]], i32 0, i32 3, i32 1)
+; CHECK-NEXT: br label %[[LATCH]]
+; CHECK: [[LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %a.ptr = getelementptr inbounds i32, ptr %a, i64 %iv
+ %b.ptr = getelementptr inbounds i32, ptr %b, i64 %iv
+ %a.val = load i32, ptr %a.ptr, align 4
+ %b.val = load i32, ptr %b.ptr, align 4
+ %sum = add i32 %a.val, %b.val
+ store i32 %sum, ptr %a.ptr, align 4
+ %positive = icmp sgt i32 %sum, 0
+ br i1 %positive, label %prefetch, label %latch
+
+prefetch:
+ %lookahead = add nuw i64 %iv, 8
+ %prefetch.ptr = getelementptr inbounds i32, ptr %b, i64 %lookahead
+ call void @llvm.prefetch.p0(ptr %prefetch.ptr, i32 0, i32 3, i32 1)
+ br label %latch
+
+latch:
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
+; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
+;.
More information about the llvm-commits
mailing list