[llvm] da95f72 - [LoopIdiom] Don't convert `memset.inline` into `memset` (#227650)

via llvm-commits llvm-commits at lists.llvm.org
Thu Oct 1 02:29:10 PDT 2026


Author: Ömer Sinan Ağacan
Date: 2026-10-01T09:28:58Z
New Revision: da95f72aa9c168846411134cbc93e2e0b267a28e

URL: https://github.com/llvm/llvm-project/commit/da95f72aa9c168846411134cbc93e2e0b267a28e
DIFF: https://github.com/llvm/llvm-project/commit/da95f72aa9c168846411134cbc93e2e0b267a28e.diff

LOG: [LoopIdiom] Don't convert `memset.inline` into `memset` (#227650)

Also remove an old and invalid FIXME and reword comments to clarify why we
don't turn inline intrinsics into non-inline versions.

There were 2 places where we check for `memcpy.inline`:

1. `processLoopMemCpy`
2. `processLoopStoreOfLoopLoad`

(1) calls (2). Because (1) didn't properly check for inline intrinsics (2)
needed to check for them as well. With (1) fixed, (2) doesn't need to check. So
the check is made an assertion.

A copy of (1) exists for `memset.inline` as well, in `processLoopMemSet`, but
(2) didn't exist in `processLoopStridedStore`, causing the bug where we convert
a `memset.inline` into `memset`.

Update `processLoopMemSet` checks the same way as (1) and add the same
assertion in (2) to `processLoopStridedStore`.

- Fixes the bug where we convert `memset.inline` to `memset`.
- Makes the checks in `memset` and `memcpy` handling paths consistent with each
  other.

Only tests for `memset` are added. `memcpy` paths already covered by existing
test file `memcpy-inline-intrinsic.ll` in the same directory.

Added: 
    llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll

Modified: 
    llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
index bbb0bd370899e..b3f095f5ba80d 100644
--- a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
@@ -837,8 +837,10 @@ bool LoopIdiomRecognize::processLoopMemCpy(MemCpyInst *MCI,
   if (MCI->isVolatile() || !isa<ConstantInt>(MCI->getLength()))
     return false;
 
-  // If we're not allowed to hack on memcpy, we fail.
-  if ((!HasMemcpy && !MCI->isForceInlined()) || DisableLIRP::Memcpy)
+  // If we're not allowed to hack on memcpy, we fail. We don't mess with the
+  // inlined version as generating a larger inline mempcy could affect code
+  // size.
+  if (!HasMemcpy || MCI->isForceInlined() || DisableLIRP::Memcpy)
     return false;
 
   Value *Dest = MCI->getDest();
@@ -900,8 +902,9 @@ bool LoopIdiomRecognize::processLoopMemSet(MemSetInst *MSI,
   if (MSI->isVolatile())
     return false;
 
-  // If we're not allowed to hack on memset, we fail.
-  if (!HasMemset || DisableLIRP::Memset)
+  // If we're not allowed to hack on memset, we fail. We don't mess with the
+  // inlined version as generating a larger memset could affect code size.
+  if (!HasMemset || MSI->isForceInlined() || DisableLIRP::Memset)
     return false;
 
   Value *Pointer = MSI->getDest();
@@ -1096,6 +1099,13 @@ bool LoopIdiomRecognize::processLoopStridedStore(
     Value *StoredVal, Instruction *TheStore,
     SmallPtrSetImpl<Instruction *> &Stores, const SCEVAddRecExpr *Ev,
     const SCEV *BECount, bool IsNegStride, bool IsLoopMemset) {
+  // We currently don't convert inline intrinsics into larger ones, to avoid
+  // code size increase. `processLoopMemSet` checks that the intrinsic is not
+  // inline before calling this function.
+  assert((isa<StoreInst>(TheStore) ||
+          !cast<MemIntrinsic>(TheStore)->isForceInlined()) &&
+         "inline mem intrinsics should be filtered out by callers");
+
   Module *M = TheStore->getModule();
 
   // The trip count of the loop and the base pointer of the addrec SCEV is
@@ -1352,12 +1362,12 @@ bool LoopIdiomRecognize::processLoopStoreOfLoopLoad(
     MaybeAlign StoreAlign, MaybeAlign LoadAlign, Instruction *TheStore,
     Instruction *TheLoad, const SCEVAddRecExpr *StoreEv,
     const SCEVAddRecExpr *LoadEv, const SCEV *BECount) {
-
-  // FIXME: until llvm.memcpy.inline supports dynamic sizes, we need to
-  // conservatively bail here, since otherwise we may have to transform
-  // llvm.memcpy.inline into llvm.memcpy which is illegal.
-  if (auto *MCI = dyn_cast<MemCpyInst>(TheStore); MCI && MCI->isForceInlined())
-    return false;
+  // We currently don't convert inline intrinsics into larger ones, to avoid
+  // code size increase. `processLoopMemCpy` checks that the intrinsic is not
+  // inline before calling this function.
+  assert((isa<StoreInst>(TheStore) ||
+          !cast<MemIntrinsic>(TheStore)->isForceInlined()) &&
+         "inline mem intrinsics should be filtered out by callers");
 
   // The trip count of the loop and the base pointer of the addrec SCEV is
   // guaranteed to be loop invariant, which means that it should dominate the

diff  --git a/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll b/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll
new file mode 100644
index 0000000000000..42d00b11fe696
--- /dev/null
+++ b/llvm/test/Transforms/LoopIdiom/memset-inline-intrinsic.ll
@@ -0,0 +1,62 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -passes=loop-idiom < %s -S | FileCheck %s
+
+define void @fixed_length(ptr %a, i64 %n) {
+; CHECK-LABEL: @fixed_length(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INC:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds [16 x i8], ptr [[A:%.*]], i64 [[I]]
+; CHECK-NEXT:    call void @llvm.memset.inline.p0.i64(ptr [[P]], i8 0, i64 16, i1 false)
+; CHECK-NEXT:    [[INC]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp ult i64 [[INC]], [[N:%.*]]
+; CHECK-NEXT:    br i1 [[C]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %inc, %loop ]
+  %p = getelementptr inbounds [16 x i8], ptr %a, i64 %i
+  call void @llvm.memset.inline.p0.i64(ptr %p, i8 0, i64 16, i1 false)
+  %inc = add nuw nsw i64 %i, 1
+  %c = icmp ult i64 %inc, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+define void @variable_length(ptr %a, i64 %n, i64 %m) {
+; CHECK-LABEL: @variable_length(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INC:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[OFF:%.*]] = mul i64 [[I]], [[M:%.*]]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A:%.*]], i64 [[OFF]]
+; CHECK-NEXT:    call void @llvm.memset.inline.p0.i64(ptr [[P]], i8 0, i64 [[M]], i1 false)
+; CHECK-NEXT:    [[INC]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp ult i64 [[INC]], [[N:%.*]]
+; CHECK-NEXT:    br i1 [[C]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %inc, %loop ]
+  %off = mul i64 %i, %m
+  %p = getelementptr inbounds i8, ptr %a, i64 %off
+  call void @llvm.memset.inline.p0.i64(ptr %p, i8 0, i64 %m, i1 false)
+  %inc = add nuw nsw i64 %i, 1
+  %c = icmp ult i64 %inc, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}


        


More information about the llvm-commits mailing list