[llvm] [LoopIdiom] Form memset on runtime-trip multi-store loops. (PR #206354)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Sun Jun 28 10:50:46 PDT 2026
https://github.com/fhahn created https://github.com/llvm/llvm-project/pull/206354
For runtime trip counts, mayLoopAccessLocation cannot bound the size of the access, which prevents forming memsets for loops with multiple stores of the same value.
If all may-aliasing stores write the same value, we can still form potentially overlapping memsets, as the order of the memsets or writing the same location multiple times should not matter.
On a large C/C++ based corpus (32k modules), we form ~2% more memsets.
base patch
memsets formed 90,063 91,853 +1.99%
>From 0049c853b39049977312598a1f65872320d60045 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Mon, 8 Jun 2026 09:17:42 +0100
Subject: [PATCH] [LoopIdiom] Form memset on runtime-trip multi-store loops.
For runtime trip counts, mayLoopAccessLocation cannot bound the size of
the access, which prevents forming memsets for loops with multiple
stores of the same value.
If all may-aliasing stores write the same value, we can still form
potentially overlapping memsets, as the order of the memsets or writing
the same location multiple times should not matter.
On a large C/C++ based corpus (32k modules), we form ~2% more memsets.
base patch
memsets formed 90,063 91,853 +1.99%
---
.../Transforms/Scalar/LoopIdiomRecognize.cpp | 38 ++++++++++++----
.../LoopIdiom/memset-multiple-accesses.ll | 43 ++++++++++++-------
2 files changed, 58 insertions(+), 23 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
index f99dad18aa5c9..ef9c1336e92e6 100644
--- a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
@@ -979,14 +979,29 @@ bool LoopIdiomRecognize::processLoopMemSet(MemSetInst *MSI,
/*IsLoopMemset=*/true);
}
+/// Return true if \p I is a (simple, loop-invariant-valued) store of the same
+/// bytewise value \p SplatByte.
+static bool isSameByteValueStore(Instruction &I, Value *SplatByte, Loop *L,
+ const DataLayout &DL) {
+ assert(SplatByte && "expected a bytewise splat value to match against");
+ auto *SI = dyn_cast<StoreInst>(&I);
+ if (!SI || !SI->isSimple() || !L->isLoopInvariant(SI->getValueOperand()))
+ return false;
+ return isBytewiseValue(SI->getValueOperand(), DL) == SplatByte;
+}
+
/// mayLoopAccessLocation - Return true if the specified loop might access the
/// specified pointer location, which is a loop-strided access. The 'Access'
/// argument specifies what the verboten forms of access are (read or write).
-static bool
-mayLoopAccessLocation(Value *Ptr, ModRefInfo Access, Loop *L,
- const SCEV *BECount, const SCEV *StoreSizeSCEV,
- AliasAnalysis &AA,
- SmallPtrSetImpl<Instruction *> &IgnoredInsts) {
+///
+/// When the access size cannot be bounded, fall back to allow stores writing
+/// the same byte value \p SplatByte.
+static bool mayLoopAccessLocation(Value *Ptr, ModRefInfo Access, Loop *L,
+ const SCEV *BECount,
+ const SCEV *StoreSizeSCEV, AliasAnalysis &AA,
+ SmallPtrSetImpl<Instruction *> &IgnoredInsts,
+ Value *SplatByte = nullptr,
+ const DataLayout *DL = nullptr) {
// Get the location that may be stored across the loop. Since the access is
// strided positively through memory, we say that the modified location starts
// at the pointer and has infinite size.
@@ -1010,11 +1025,18 @@ mayLoopAccessLocation(Value *Ptr, ModRefInfo Access, Loop *L,
// which will then no-alias a store to &A[100].
MemoryLocation StoreLoc(Ptr, AccessSize);
+ // Only consult the same-byte-value fallback when the access size stayed
+ // infinite (non-constant trip count); with a precise size AA is accurate.
+ bool TrySameByteValue = !AccessSize.isPrecise() && SplatByte && DL;
+
for (BasicBlock *B : L->blocks())
for (Instruction &I : *B)
if (!IgnoredInsts.contains(&I) &&
- isModOrRefSet(AA.getModRefInfo(&I, StoreLoc) & Access))
+ isModOrRefSet(AA.getModRefInfo(&I, StoreLoc) & Access)) {
+ if (TrySameByteValue && isSameByteValueStore(I, SplatByte, L, *DL))
+ continue;
return true;
+ }
return false;
}
@@ -1098,15 +1120,15 @@ bool LoopIdiomRecognize::processLoopStridedStore(
// the return value will read this comment, and leave them alone.
Changed = true;
+ Value *SplatValue = isBytewiseValue(StoredVal, *DL);
if (mayLoopAccessLocation(BasePtr, ModRefInfo::ModRef, CurLoop, BECount,
- StoreSizeSCEV, *AA, Stores))
+ StoreSizeSCEV, *AA, Stores, SplatValue, DL))
return Changed;
if (avoidLIRForMultiBlockLoop(/*IsMemset=*/true, IsLoopMemset))
return Changed;
// Okay, everything looks good, insert the memset.
- Value *SplatValue = isBytewiseValue(StoredVal, *DL);
Constant *PatternValue = nullptr;
if (!SplatValue)
PatternValue = getMemSetPatternValue(StoredVal, DL);
diff --git a/llvm/test/Transforms/LoopIdiom/memset-multiple-accesses.ll b/llvm/test/Transforms/LoopIdiom/memset-multiple-accesses.ll
index c5a14278b773c..434068512f01e 100644
--- a/llvm/test/Transforms/LoopIdiom/memset-multiple-accesses.ll
+++ b/llvm/test/Transforms/LoopIdiom/memset-multiple-accesses.ll
@@ -5,15 +5,18 @@ define void @zero_two_disjoint_fields(ptr %p, i64 %n) {
; CHECK-LABEL: define void @zero_two_disjoint_fields(
; CHECK-SAME: ptr [[P:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P]], i64 4
+; CHECK-NEXT: [[TMP0:%.*]] = shl nuw i64 [[N]], 2
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[SCEVGEP]], i8 0, i64 [[TMP0]], i1 false)
+; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr nuw i8, ptr [[P]], i64 868
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[SCEVGEP1]], i8 0, i64 [[TMP0]], i1 false)
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[FIELD_A:%.*]] = getelementptr inbounds nuw i8, ptr [[P]], i64 4
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds nuw float, ptr [[FIELD_A]], i64 [[IV]]
-; CHECK-NEXT: store float 0.000000e+00, ptr [[GEP_A]], align 4
; CHECK-NEXT: [[FIELD_B:%.*]] = getelementptr inbounds nuw i8, ptr [[P]], i64 868
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds nuw float, ptr [[FIELD_B]], i64 [[IV]]
-; CHECK-NEXT: store float 0.000000e+00, ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
@@ -43,13 +46,14 @@ define void @zero_two_unknown_pointers(ptr %a, ptr %b, i64 %n) {
; CHECK-LABEL: define void @zero_two_unknown_pointers(
; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = shl nuw i64 [[N]], 2
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[A]], i8 0, i64 [[TMP0]], i1 false)
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[B]], i8 0, i64 [[TMP0]], i1 false)
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds nuw float, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT: store float 0.000000e+00, ptr [[GEP_A]], align 4
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds nuw float, ptr [[B]], i64 [[IV]]
-; CHECK-NEXT: store float 0.000000e+00, ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
@@ -77,13 +81,14 @@ define void @diff_size(ptr %a, ptr %b, i64 %n) {
; CHECK-LABEL: define void @diff_size(
; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = shl nuw i64 [[N]], 2
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[A]], i8 0, i64 [[TMP0]], i1 false)
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 1 [[B]], i8 0, i64 [[N]], i1 false)
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds nuw i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT: store i32 0, ptr [[GEP_A]], align 4
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IV]]
-; CHECK-NEXT: store i8 0, ptr [[GEP_B]], align 1
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
@@ -179,15 +184,16 @@ define void @zero_three_unknown_pointers(ptr %a, ptr %b, ptr %c, i64 %n) {
; CHECK-LABEL: define void @zero_three_unknown_pointers(
; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = shl nuw i64 [[N]], 2
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[A]], i8 0, i64 [[TMP0]], i1 false)
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[B]], i8 0, i64 [[TMP0]], i1 false)
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[C]], i8 0, i64 [[TMP0]], i1 false)
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds nuw float, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT: store float 0.000000e+00, ptr [[GEP_A]], align 4
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds nuw float, ptr [[B]], i64 [[IV]]
-; CHECK-NEXT: store float 0.000000e+00, ptr [[GEP_B]], align 4
; CHECK-NEXT: [[GEP_C:%.*]] = getelementptr inbounds nuw float, ptr [[C]], i64 [[IV]]
-; CHECK-NEXT: store float 0.000000e+00, ptr [[GEP_C]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
@@ -217,15 +223,19 @@ define void @zero_three_negstride(ptr %a, ptr %b, ptr %c, i64 %n) {
; CHECK-LABEL: define void @zero_three_negstride(
; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 4
+; CHECK-NEXT: [[TMP0:%.*]] = shl nuw i64 [[N]], 2
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[SCEVGEP]], i8 0, i64 [[TMP0]], i1 false)
+; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B]], i64 4
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[SCEVGEP1]], i8 0, i64 [[TMP0]], i1 false)
+; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[C]], i64 4
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[SCEVGEP2]], i8 0, i64 [[TMP0]], i1 false)
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[N]], %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT: store i32 0, ptr [[GEP_A]], align 4
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
-; CHECK-NEXT: store i32 0, ptr [[GEP_B]], align 4
; CHECK-NEXT: [[GEP_C:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
-; CHECK-NEXT: store i32 0, ptr [[GEP_C]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nsw i64 [[IV]], -1
; CHECK-NEXT: [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], 0
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
@@ -255,17 +265,20 @@ define void @zero_three_partial_overlap(ptr %a, i64 %n) {
; CHECK-LABEL: define void @zero_three_partial_overlap(
; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = shl nuw i64 [[N]], 2
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[A]], i8 0, i64 [[TMP0]], i1 false)
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[A]], i64 2
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[SCEVGEP]], i8 0, i64 [[TMP0]], i1 false)
+; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr nuw i8, ptr [[A]], i64 4
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 [[SCEVGEP1]], i8 0, i64 [[TMP0]], i1 false)
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_0:%.*]] = getelementptr inbounds nuw i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT: store i32 0, ptr [[GEP_0]], align 4
; CHECK-NEXT: [[A_2:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 2
; CHECK-NEXT: [[GEP_2:%.*]] = getelementptr inbounds nuw i32, ptr [[A_2]], i64 [[IV]]
-; CHECK-NEXT: store i32 0, ptr [[GEP_2]], align 4
; CHECK-NEXT: [[A_4:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 4
; CHECK-NEXT: [[GEP_4:%.*]] = getelementptr inbounds nuw i32, ptr [[A_4]], i64 [[IV]]
-; CHECK-NEXT: store i32 0, ptr [[GEP_4]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
More information about the llvm-commits
mailing list