[llvm] [LoopIdiom] Add a range attribute to formed memset/memcpy/memmove (PR #226801)
Kuba Mracek via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 27 08:49:37 PDT 2026
https://github.com/kubamracek created https://github.com/llvm/llvm-project/pull/226801
Part 1 of optimizing a recognized memset in a loop like the following: Adding a range attribute to the memset call to that the upper bound is carried in the IR. Part 2 will prevent lowering a (non-constant) llvm.memset with a small upper bound into a memset libcall (expensive) and instead emit a branch or per-size branches instead (cheap).
This PR:
> Attach a range attribute to the length (or, for memset.pattern, the count) operand. The upper bound comes from the loop's constant maximum backedge-taken count and conditions guarding the loop, which also hold at the new call in the preheader. For a loop like this...
>
> for (unsigned i = n; i < 4; ++i)
> dst[i] = -1;
>
> ...the memset gets `i64 range(i64 0, 17)`.
>
> LoopIdiomRecognize applies this to memset, memset.pattern, memcpy and memmove calls.
Assisted-by: Claude
>From d05e7ec6bbff2048ba00daef0f340266469b3669 Mon Sep 17 00:00:00 2001
From: Kuba Mracek <mracek at apple.com>
Date: Sun, 27 Sep 2026 16:33:11 +0100
Subject: [PATCH] [LoopIdiom] Add a range attribute to formed
memset/memcpy/memmove
Attach a range attribute to the length (or, for memset.pattern, the count)
operand. The upper bound comes from the loop's constant maximum backedge-taken
count and conditions guarding the loop, which also hold at the new call in the
preheader. For a loop like this...
for (unsigned i = n; i < 4; ++i)
dst[i] = -1;
...the memset gets `i64 range(i64 0, 17)`.
LoopIdiomRecognize applies this to memset, memset.pattern, memcpy and memmove
calls.
Assisted-by: Claude
---
.../Transforms/Scalar/LoopIdiomRecognize.cpp | 51 ++++++++++++-
.../LoopIdiom/mem-intrinsic-length-range.ll | 74 +++++++++++++++++++
2 files changed, 124 insertions(+), 1 deletion(-)
create mode 100644 llvm/test/Transforms/LoopIdiom/mem-intrinsic-length-range.ll
diff --git a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
index 8c098e0d68d72..0529ca6867617 100644
--- a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
@@ -1089,6 +1089,44 @@ static const SCEV *getNumBytes(const SCEV *BECount, Type *IntPtr,
SCEV::FlagNUW);
}
+/// Add a range to newly formed memset/memmove/memcpy intrinsic with an upper
+/// bound derived from the loop's constant max trip count.
+static void addRangeAttrFromTripCount(CallInst *NewCall, unsigned ArgNo,
+ uint64_t ElemsPerIter,
+ const SCEV *BECount, Loop *L,
+ ScalarEvolution *SE) {
+ Value *Len = NewCall->getArgOperand(ArgNo);
+ if (isa<Constant>(Len))
+ return;
+
+ // Two upper bounds:
+ // (1) constant max backedge-taken count
+ const APInt *Max1;
+ if (!match(SE->getConstantMaxBackedgeTakenCount(L), m_scev_APInt(Max1)))
+ return;
+
+ // (2) loop guards (new call is in preheader, so guards dominate the call)
+ APInt Max2 = SE->getUnsignedRangeMax(SE->applyLoopGuards(BECount, L));
+
+ // Use the tighter bound
+ unsigned MaxWidth = std::max(Max1->getBitWidth(), Max2.getBitWidth());
+ APInt MaxBTC = APIntOps::umin(Max1->zext(MaxWidth), Max2.zext(MaxWidth));
+
+ // Bail if bound beyond bitwidth, overflows, or is MaxValue
+ unsigned BW = Len->getType()->getIntegerBitWidth();
+ if (MaxBTC.getActiveBits() >= BW)
+ return;
+ APInt MaxTripCount = MaxBTC.zext(BW) + 1;
+ bool Overflow;
+ APInt MaxLen = MaxTripCount.umul_ov(APInt(BW, ElemsPerIter), Overflow);
+ if (Overflow || MaxLen.isMaxValue())
+ return;
+
+ NewCall->addParamAttr(
+ ArgNo, Attribute::get(NewCall->getContext(), Attribute::Range,
+ ConstantRange(APInt::getZero(BW), MaxLen + 1)));
+}
+
/// processLoopStridedStore - We see a strided store of some value. If we can
/// transform this into a memset or memset_pattern in the loop preheader, do so.
bool LoopIdiomRecognize::processLoopStridedStore(
@@ -1155,6 +1193,7 @@ bool LoopIdiomRecognize::processLoopStridedStore(
// of pattern repetitions if the memset.pattern intrinsic is being used.
Value *MemsetArg;
std::optional<int64_t> BytesWritten;
+ uint64_t PatternRepsPerTrip = 0;
if (PatternValue && (HasMemsetPattern || ForceMemsetPatternIntrinsic)) {
const SCEV *TripCountS =
@@ -1166,7 +1205,7 @@ bool LoopIdiomRecognize::processLoopStridedStore(
return Changed;
Value *TripCount = Expander.expandCodeFor(TripCountS, IntIdxTy,
Preheader->getTerminator());
- uint64_t PatternRepsPerTrip =
+ PatternRepsPerTrip =
(ConstStoreSize->getValue()->getZExtValue() * 8) /
DL->getTypeSizeInBits(PatternValue->getType());
// If ConstStoreSize is not equal to the width of PatternValue, then
@@ -1210,6 +1249,11 @@ bool LoopIdiomRecognize::processLoopStridedStore(
NewCall = Builder.CreateMemSet(BasePtr, SplatValue, MemsetArg,
MaybeAlign(StoreAlignment),
/*isVolatile=*/false, AATags);
+ if (auto *ConstStoreSize = dyn_cast<SCEVConstant>(StoreSizeSCEV)) {
+ addLengthRangeFromTripCount(NewCall, /*ArgNo*/ 2,
+ ConstStoreSize->getValue()->getZExtValue(),
+ BECount, CurLoop, SE);
+ }
} else if (ForceMemsetPatternIntrinsic ||
isLibFuncEmittable(M, TLI, LibFunc_memset_pattern16)) {
assert(isa<SCEVConstant>(StoreSizeSCEV) && "Expected constant store size");
@@ -1222,6 +1266,9 @@ bool LoopIdiomRecognize::processLoopStridedStore(
if (StoreAlignment)
cast<MemSetPatternInst>(NewCall)->setDestAlignment(*StoreAlignment);
NewCall->setAAMetadata(AATags);
+ assert(PatternRepsPerTrip != 0 && "PatternRepsPerTrip must be set");
+ addLengthRangeFromTripCount(NewCall, /*ArgNo*/ 2, PatternRepsPerTrip,
+ BECount, CurLoop, SE);
} else {
// Neither a memset, nor memset_pattern16
return Changed;
@@ -1534,6 +1581,8 @@ bool LoopIdiomRecognize::processLoopStoreOfLoopLoad(
StoreBasePtr, *StoreAlign, LoadBasePtr, *LoadAlign, NumBytes, StoreSize,
AATags);
}
+ addRangeAttrFromTripCount(NewCall, /*ArgNo*/ 2, StoreSize, BECount, CurLoop,
+ SE);
NewCall->setDebugLoc(TheStore->getDebugLoc());
if (MSSAU) {
diff --git a/llvm/test/Transforms/LoopIdiom/mem-intrinsic-length-range.ll b/llvm/test/Transforms/LoopIdiom/mem-intrinsic-length-range.ll
new file mode 100644
index 0000000000000..dbdcdf6581e30
--- /dev/null
+++ b/llvm/test/Transforms/LoopIdiom/mem-intrinsic-length-range.ll
@@ -0,0 +1,74 @@
+; RUN: opt -passes=loop-idiom -S < %s | FileCheck %s
+
+target datalayout = "e-m:o-i64:64-i128:128-n32:64-S128"
+target triple = "arm64-apple-macosx14.0.0"
+
+; void bounded_by_exit_condition(int *p, unsigned long start) {
+; for (unsigned long i = start; i < 4; ++i)
+; p[i] = -1;
+; }
+
+define void @bounded_by_exit_condition(ptr %p, i64 %start) {
+; CHECK-LABEL: define void @bounded_by_exit_condition(
+; CHECK: call void @llvm.memset.p0.i64(ptr align 4 {{%.*}}, i8 -1, i64 range(i64 0, 17) {{%.*}}, i1 false)
+entry:
+ %guard = icmp ult i64 %start, 4
+ br i1 %guard, label %loop, label %exit
+loop:
+ %i = phi i64 [ %start, %entry ], [ %i.next, %loop ]
+ %gep = getelementptr inbounds i32, ptr %p, i64 %i
+ store i32 -1, ptr %gep, align 4
+ %i.next = add nuw nsw i64 %i, 1
+ %cmp = icmp ult i64 %i.next, 4
+ br i1 %cmp, label %loop, label %exit
+exit:
+ ret void
+}
+
+; void bounded_by_guard(char *p, unsigned long n) {
+; if (n <= 100 && n != 0)
+; for (unsigned long i = 0; i < n; ++i)
+; p[i] = 0;
+; }
+
+define void @bounded_by_guard(ptr %p, i64 %n) {
+; CHECK-LABEL: define void @bounded_by_guard(
+; CHECK: call void @llvm.memset.p0.i64(ptr align 1 {{%.*}}, i8 0, i64 range(i64 0, 101) {{%.*}}, i1 false)
+entry:
+ %small = icmp ule i64 %n, 100
+ %nonzero = icmp ne i64 %n, 0
+ %guard = and i1 %small, %nonzero
+ br i1 %guard, label %loop, label %exit
+loop:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+ %gep = getelementptr inbounds i8, ptr %p, i64 %i
+ store i8 0, ptr %gep, align 1
+ %i.next = add nuw nsw i64 %i, 1
+ %cmp = icmp ult i64 %i.next, %n
+ br i1 %cmp, label %loop, label %exit
+exit:
+ ret void
+}
+
+; void unbounded(int *p, unsigned long n) {
+; if (n != 0)
+; for (unsigned long i = 0; i < n; ++i)
+; p[i] = -1;
+; }
+
+define void @unbounded(ptr %p, i64 %n) {
+; CHECK-LABEL: define void @unbounded(
+; CHECK-NEXT: call void @llvm.memset.p0.i64(ptr align 4 {{%.*}}, i8 -1, i64 {{%.*}}, i1 false)
+entry:
+ %nonzero = icmp ne i64 %n, 0
+ br i1 %nonzero, label %loop, label %exit
+loop:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+ %gep = getelementptr inbounds i32, ptr %p, i64 %i
+ store i32 -1, ptr %gep, align 4
+ %i.next = add nuw nsw i64 %i, 1
+ %cmp = icmp ult i64 %i.next, %n
+ br i1 %cmp, label %loop, label %exit
+exit:
+ ret void
+}
More information about the llvm-commits
mailing list