[llvm] [LoopIdiom] Don't convert `memset.inline` into `memset` (PR #227650)

Ömer Sinan Ağacan via llvm-commits llvm-commits at lists.llvm.org
Thu Oct 1 01:49:20 PDT 2026


https://github.com/osa1 updated https://github.com/llvm/llvm-project/pull/227650

>From fe74ed6e207bdc3b162edd47a1df3c6d9d2ddf27 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?=C3=96mer=20Sinan=20A=C4=9Facan?= <omer at osa1.net>
Date: Wed, 30 Sep 2026 11:27:07 +0100
Subject: [PATCH 1/5] [LoopIdiom] Add test for `memset.inline` in a loop (NFC)

Currently loop-idiom converts `llvm.memset.inline` into `llvm.memset`,
which is not allowed as the inline intrinsic is guaranteed to not make a
libcall.
---
 .../LoopIdiom/memset-inline-intrinsic.ll      | 33 +++++++++++++++++++
 1 file changed, 33 insertions(+)
 create mode 100644 llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll

diff --git a/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll b/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll
new file mode 100644
index 0000000000000..2dacaedd47dd0
--- /dev/null
+++ b/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll
@@ -0,0 +1,33 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -passes=loop-idiom < %s -S | FileCheck %s
+
+define void @f(ptr %a, i64 %n) {
+; CHECK-LABEL: @f(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N:%.*]], i64 1)
+; CHECK-NEXT:    [[TMP0:%.*]] = shl nuw i64 [[UMAX]], 4
+; CHECK-NEXT:    call void @llvm.memset.p0.i64(ptr [[A:%.*]], i8 0, i64 [[TMP0]], i1 false)
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INC:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds [16 x i8], ptr [[A]], i64 [[I]]
+; CHECK-NEXT:    [[INC]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp ult i64 [[INC]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %inc, %loop ]
+  %p = getelementptr inbounds [16 x i8], ptr %a, i64 %i
+  call void @llvm.memset.inline.p0.i64(ptr %p, i8 0, i64 16, i1 false)
+  %inc = add nuw nsw i64 %i, 1
+  %c = icmp ult i64 %inc, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}

>From a4fe80f77689a74f5ac07afda24ddd286fd6554b Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?=C3=96mer=20Sinan=20A=C4=9Facan?= <omer at osa1.net>
Date: Wed, 30 Sep 2026 11:27:08 +0100
Subject: [PATCH 2/5] [LoopIdiom] Don't convert `memset.inline` into `memset`

---
 .../Transforms/Scalar/LoopIdiomRecognize.cpp  | 27 +++++++++++++------
 .../LoopIdiom/memset-inline-intrinsic.ll      |  8 +++---
 2 files changed, 22 insertions(+), 13 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
index bbb0bd370899e..5605b4a55a315 100644
--- a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
@@ -837,8 +837,10 @@ bool LoopIdiomRecognize::processLoopMemCpy(MemCpyInst *MCI,
   if (MCI->isVolatile() || !isa<ConstantInt>(MCI->getLength()))
     return false;
 
-  // If we're not allowed to hack on memcpy, we fail.
-  if ((!HasMemcpy && !MCI->isForceInlined()) || DisableLIRP::Memcpy)
+  // If we're not allowed to hack on memcpy, we fail. We don't mess with the
+  // inlined version as generating a larger inline mempcy could affect code
+  // size.
+  if (!HasMemcpy || MCI->isForceInlined() || DisableLIRP::Memcpy)
     return false;
 
   Value *Dest = MCI->getDest();
@@ -900,8 +902,9 @@ bool LoopIdiomRecognize::processLoopMemSet(MemSetInst *MSI,
   if (MSI->isVolatile())
     return false;
 
-  // If we're not allowed to hack on memset, we fail.
-  if (!HasMemset || DisableLIRP::Memset)
+  // If we're not allowed to hack on memset, we fail. We don't mess with the
+  // inlined version as generating a larger memset could affect code size.
+  if (!HasMemset || MSI->isForceInlined() || DisableLIRP::Memset)
     return false;
 
   Value *Pointer = MSI->getDest();
@@ -1096,6 +1099,10 @@ bool LoopIdiomRecognize::processLoopStridedStore(
     Value *StoredVal, Instruction *TheStore,
     SmallPtrSetImpl<Instruction *> &Stores, const SCEVAddRecExpr *Ev,
     const SCEV *BECount, bool IsNegStride, bool IsLoopMemset) {
+  // The same check as in `processLoopStoreOfLoopLoad`, see the comments there.
+  if (auto *MSI = dyn_cast<MemSetInst>(TheStore); MSI && MSI->isForceInlined())
+    return false;
+
   Module *M = TheStore->getModule();
 
   // The trip count of the loop and the base pointer of the addrec SCEV is
@@ -1352,10 +1359,14 @@ bool LoopIdiomRecognize::processLoopStoreOfLoopLoad(
     MaybeAlign StoreAlign, MaybeAlign LoadAlign, Instruction *TheStore,
     Instruction *TheLoad, const SCEVAddRecExpr *StoreEv,
     const SCEVAddRecExpr *LoadEv, const SCEV *BECount) {
-
-  // FIXME: until llvm.memcpy.inline supports dynamic sizes, we need to
-  // conservatively bail here, since otherwise we may have to transform
-  // llvm.memcpy.inline into llvm.memcpy which is illegal.
+  // Avoid converting `llvm.memcpy.inline` into `llvm.memcpy`, as the inline
+  // intrinsic is guaranteed to not make a libcall.
+  //
+  // We could generate `llvm.memcpy.inline` when the store instruction is the
+  // inline version, but that can potentially generate more code. For now, do
+  // the conservative thing and bail.
+  //
+  // The same check for `llvm.memset.inline` is in `processLoopStridedStore`.
   if (auto *MCI = dyn_cast<MemCpyInst>(TheStore); MCI && MCI->isForceInlined())
     return false;
 
diff --git a/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll b/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll
index 2dacaedd47dd0..d2fe3f5d3073d 100644
--- a/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll
+++ b/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll
@@ -4,15 +4,13 @@
 define void @f(ptr %a, i64 %n) {
 ; CHECK-LABEL: @f(
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N:%.*]], i64 1)
-; CHECK-NEXT:    [[TMP0:%.*]] = shl nuw i64 [[UMAX]], 4
-; CHECK-NEXT:    call void @llvm.memset.p0.i64(ptr [[A:%.*]], i8 0, i64 [[TMP0]], i1 false)
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INC:%.*]], [[LOOP]] ]
-; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds [16 x i8], ptr [[A]], i64 [[I]]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds [16 x i8], ptr [[A:%.*]], i64 [[I]]
+; CHECK-NEXT:    call void @llvm.memset.inline.p0.i64(ptr [[P]], i8 0, i64 16, i1 false)
 ; CHECK-NEXT:    [[INC]] = add nuw nsw i64 [[I]], 1
-; CHECK-NEXT:    [[C:%.*]] = icmp ult i64 [[INC]], [[N]]
+; CHECK-NEXT:    [[C:%.*]] = icmp ult i64 [[INC]], [[N:%.*]]
 ; CHECK-NEXT:    br i1 [[C]], label [[LOOP]], label [[EXIT:%.*]]
 ; CHECK:       exit:
 ; CHECK-NEXT:    ret void

>From 2e42c70cecc5f00c23aa9d338c968c27ec7635ab Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?=C3=96mer=20Sinan=20A=C4=9Facan?= <omer at osa1.net>
Date: Wed, 30 Sep 2026 12:22:00 +0100
Subject: [PATCH 3/5] Convert redundant checks into assertions

---
 .../Transforms/Scalar/LoopIdiomRecognize.cpp  | 25 +++++++++----------
 1 file changed, 12 insertions(+), 13 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
index 5605b4a55a315..5599352f416dc 100644
--- a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
@@ -1099,9 +1099,12 @@ bool LoopIdiomRecognize::processLoopStridedStore(
     Value *StoredVal, Instruction *TheStore,
     SmallPtrSetImpl<Instruction *> &Stores, const SCEVAddRecExpr *Ev,
     const SCEV *BECount, bool IsNegStride, bool IsLoopMemset) {
-  // The same check as in `processLoopStoreOfLoopLoad`, see the comments there.
-  if (auto *MSI = dyn_cast<MemSetInst>(TheStore); MSI && MSI->isForceInlined())
-    return false;
+  // We currently don't convert inline intrinsics into larger ones, to avoid
+  // code size increase. `processLoopMemSet` checks that the intrinsic is not
+  // inline before calling this function.
+  assert((!isa<MemIntrinsic>(TheStore) ||
+          !cast<MemIntrinsic>(TheStore)->isForceInlined()) &&
+         "inline mem intrinsics should be filtered out by callers");
 
   Module *M = TheStore->getModule();
 
@@ -1359,16 +1362,12 @@ bool LoopIdiomRecognize::processLoopStoreOfLoopLoad(
     MaybeAlign StoreAlign, MaybeAlign LoadAlign, Instruction *TheStore,
     Instruction *TheLoad, const SCEVAddRecExpr *StoreEv,
     const SCEVAddRecExpr *LoadEv, const SCEV *BECount) {
-  // Avoid converting `llvm.memcpy.inline` into `llvm.memcpy`, as the inline
-  // intrinsic is guaranteed to not make a libcall.
-  //
-  // We could generate `llvm.memcpy.inline` when the store instruction is the
-  // inline version, but that can potentially generate more code. For now, do
-  // the conservative thing and bail.
-  //
-  // The same check for `llvm.memset.inline` is in `processLoopStridedStore`.
-  if (auto *MCI = dyn_cast<MemCpyInst>(TheStore); MCI && MCI->isForceInlined())
-    return false;
+  // We currently don't convert inline intrinsics into larger ones, to avoid
+  // code size increase. `processLoopMemCpy` checks that the intrinsic is not
+  // inline before calling this function.
+  assert((!isa<MemIntrinsic>(TheStore) ||
+          !cast<MemIntrinsic>(TheStore)->isForceInlined()) &&
+         "inline mem intrinsics should be filtered out by callers");
 
   // The trip count of the loop and the base pointer of the addrec SCEV is
   // guaranteed to be loop invariant, which means that it should dominate the

>From 0733f20fbb8775e8e9342f314e6a141657a34180 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?=C3=96mer=20Sinan=20A=C4=9Facan?= <omer at osa1.net>
Date: Wed, 30 Sep 2026 12:26:43 +0100
Subject: [PATCH 4/5] Add variable-length `memset.inline` test

---
 .../LoopIdiom/memset-inline-intrinsic.ll      | 35 +++++++++++++++++--
 1 file changed, 33 insertions(+), 2 deletions(-)

diff --git a/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll b/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll
index d2fe3f5d3073d..42d00b11fe696 100644
--- a/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll
+++ b/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll
@@ -1,8 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
 ; RUN: opt -passes=loop-idiom < %s -S | FileCheck %s
 
-define void @f(ptr %a, i64 %n) {
-; CHECK-LABEL: @f(
+define void @fixed_length(ptr %a, i64 %n) {
+; CHECK-LABEL: @fixed_length(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
@@ -29,3 +29,34 @@ loop:
 exit:
   ret void
 }
+
+define void @variable_length(ptr %a, i64 %n, i64 %m) {
+; CHECK-LABEL: @variable_length(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INC:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[OFF:%.*]] = mul i64 [[I]], [[M:%.*]]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A:%.*]], i64 [[OFF]]
+; CHECK-NEXT:    call void @llvm.memset.inline.p0.i64(ptr [[P]], i8 0, i64 [[M]], i1 false)
+; CHECK-NEXT:    [[INC]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp ult i64 [[INC]], [[N:%.*]]
+; CHECK-NEXT:    br i1 [[C]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %inc, %loop ]
+  %off = mul i64 %i, %m
+  %p = getelementptr inbounds i8, ptr %a, i64 %off
+  call void @llvm.memset.inline.p0.i64(ptr %p, i8 0, i64 %m, i1 false)
+  %inc = add nuw nsw i64 %i, 1
+  %c = icmp ult i64 %inc, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}

>From b23e039f6058dd4f85b1a76c3f178a299af233e4 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?=C3=96mer=20Sinan=20A=C4=9Facan?= <omer at osa1.net>
Date: Thu, 1 Oct 2026 09:48:52 +0100
Subject: [PATCH 5/5] Address review comments

---
 llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
index 5599352f416dc..b3f095f5ba80d 100644
--- a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
@@ -1102,7 +1102,7 @@ bool LoopIdiomRecognize::processLoopStridedStore(
   // We currently don't convert inline intrinsics into larger ones, to avoid
   // code size increase. `processLoopMemSet` checks that the intrinsic is not
   // inline before calling this function.
-  assert((!isa<MemIntrinsic>(TheStore) ||
+  assert((isa<StoreInst>(TheStore) ||
           !cast<MemIntrinsic>(TheStore)->isForceInlined()) &&
          "inline mem intrinsics should be filtered out by callers");
 
@@ -1365,7 +1365,7 @@ bool LoopIdiomRecognize::processLoopStoreOfLoopLoad(
   // We currently don't convert inline intrinsics into larger ones, to avoid
   // code size increase. `processLoopMemCpy` checks that the intrinsic is not
   // inline before calling this function.
-  assert((!isa<MemIntrinsic>(TheStore) ||
+  assert((isa<StoreInst>(TheStore) ||
           !cast<MemIntrinsic>(TheStore)->isForceInlined()) &&
          "inline mem intrinsics should be filtered out by callers");
 



More information about the llvm-commits mailing list