[llvm] [LowerMemIntrinsics] Expand small variable-length memsets inline (PR #228420)

via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 05:38:29 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Kuba (Brecka) Mracek (kubamracek)

<details>
<summary>Changes</summary>

Part 2 of the motivation mentioned in https://github.com/llvm/llvm-project/pull/226801: Optimizing a recognized memset from a loop. Lowering a (non-constant) llvm.memset with a small upper bound into a memset libcall is more expensive. Instead emit branch(es) for per-size class direct stores.

Commit message:

> A memset whose length is not a constant is always lowered to a call to memset, even when the length is known to be small. Expand such a memset inline instead when the upper bound of its length is at most a target-provided threshold.
> 
> The bounds of the length come from (1) range attribute on the call, (2) known bits, (3) computeConstantRange from ValueTracking.
> 
> The expansion branches on the length to the largest power-of-two size class, and covers a length in [Size, 2 * Size] with one "regular" store at the start + one "tail" store at the end, which may overlap.
> 
> A new `getMaxBoundedMemSetInlineSize` TTI hook is added, which can be overridden with -bounded-memset-inline-size.

Assisted-by: Claude

---

Patch is 22.29 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/228420.diff


8 Files Affected:

- (modified) llvm/include/llvm/Analysis/TargetTransformInfo.h (+6) 
- (modified) llvm/include/llvm/Analysis/TargetTransformInfoImpl.h (+2) 
- (modified) llvm/include/llvm/Transforms/Utils/LowerMemIntrinsics.h (+7) 
- (modified) llvm/lib/Analysis/TargetTransformInfo.cpp (+4) 
- (modified) llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp (+51) 
- (modified) llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h (+6) 
- (modified) llvm/lib/Transforms/Utils/LowerMemIntrinsics.cpp (+117) 
- (added) llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/bounded-memset.ll (+164) 


``````````diff
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index a33e6f62e941ed7..6068abe3b3bfece 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -487,6 +487,12 @@ class TargetTransformInfo {
   /// profitable to inline the call.
   LLVM_ABI uint64_t getMaxMemIntrinsicInlineSizeThreshold() const;
 
+  /// Returns the maximum length in bytes for which a memset with a
+  /// non-constant length, known to be bounded by that value, is preferably
+  /// expanded inline instead of calling the library. Zero disables the
+  /// expansion.
+  LLVM_ABI uint64_t getMaxBoundedMemSetInlineSize() const;
+
   /// \return The estimated number of case clusters when lowering \p 'SI'.
   /// \p JTSize Set a jump table size only when \p SI is suitable for a jump
   /// table.
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index a19c122c16f20c5..f0a2d755a298256 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -116,6 +116,8 @@ class LLVM_ABI TargetTransformInfoImplBase {
 
   virtual uint64_t getMaxMemIntrinsicInlineSizeThreshold() const { return 64; }
 
+  virtual uint64_t getMaxBoundedMemSetInlineSize() const { return 0; }
+
   // Although this default value is arbitrary, it is not random. It is assumed
   // that a condition that evaluates the same way by a higher percentage than
   // this is best represented as control flow. Therefore, the default value N
diff --git a/llvm/include/llvm/Transforms/Utils/LowerMemIntrinsics.h b/llvm/include/llvm/Transforms/Utils/LowerMemIntrinsics.h
index 55c7119cd8fdbec..3509574b8080c69 100644
--- a/llvm/include/llvm/Transforms/Utils/LowerMemIntrinsics.h
+++ b/llvm/include/llvm/Transforms/Utils/LowerMemIntrinsics.h
@@ -23,6 +23,7 @@ namespace llvm {
 class AnyMemCpyInst;
 class ConstantInt;
 class Instruction;
+struct KnownBits;
 class MemCpyInst;
 class MemMoveInst;
 class MemSetInst;
@@ -71,6 +72,12 @@ LLVM_ABI void expandMemSetAsLoop(MemSetInst *MemSet,
 LLVM_ABI void expandMemSetAsLoop(MemSetInst *MemSet,
                                  const TargetTransformInfo &TTI);
 
+/// Expand \p MemSet, whose length is known to be in [\p MinLen, \p MaxLen] and
+/// to have the known bits \p LenKnown, into a tree of branches on the length
+/// leading to at most two overlapping stores each. \p MemSet is not deleted.
+LLVM_ABI void expandBoundedMemSet(MemSetInst *MemSet, uint64_t MinLen,
+                                  uint64_t MaxLen, const KnownBits &LenKnown);
+
 /// Expand \p MemSetPattern as a loop. \p MemSet is not deleted.
 /// If \p TTI is provided, the memset.pattern is expanded according to the
 /// target's preferences. Otherwise, it is expanded as an element-wise loop.
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index af73a615f25fa61..5d4e3c7811cde84 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -1336,6 +1336,10 @@ uint64_t TargetTransformInfo::getMaxMemIntrinsicInlineSizeThreshold() const {
   return TTIImpl->getMaxMemIntrinsicInlineSizeThreshold();
 }
 
+uint64_t TargetTransformInfo::getMaxBoundedMemSetInlineSize() const {
+  return TTIImpl->getMaxBoundedMemSetInlineSize();
+}
+
 InstructionCost TargetTransformInfo::getArithmeticReductionCost(
     unsigned Opcode, VectorType *Ty, std::optional<FastMathFlags> FMF,
     TTI::TargetCostKind CostKind) const {
diff --git a/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp b/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
index 77d8d44ea3586d3..560a81d9f32f707 100644
--- a/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
+++ b/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
@@ -17,12 +17,14 @@
 #include "llvm/Analysis/ObjCARCUtil.h"
 #include "llvm/Analysis/TargetLibraryInfo.h"
 #include "llvm/Analysis/TargetTransformInfo.h"
+#include "llvm/Analysis/ValueTracking.h"
 #include "llvm/CodeGen/ExpandVectorPredication.h"
 #include "llvm/CodeGen/LibcallLoweringInfo.h"
 #include "llvm/CodeGen/Passes.h"
 #include "llvm/CodeGen/RuntimeLibcallUtil.h"
 #include "llvm/CodeGen/TargetLowering.h"
 #include "llvm/CodeGen/TargetPassConfig.h"
+#include "llvm/IR/ConstantRange.h"
 #include "llvm/IR/Function.h"
 #include "llvm/IR/GlobalValue.h"
 #include "llvm/IR/IRBuilder.h"
@@ -56,6 +58,13 @@ static cl::opt<int64_t> MemIntrinsicExpandSizeThresholdOpt(
     cl::desc("Set minimum mem intrinsic size to expand in IR"), cl::init(-1),
     cl::Hidden);
 
+/// Maximum bound of a memset with non-constant length for it to be expanded
+/// inline. Overrides TargetTransformInfo::getMaxBoundedMemSetInlineSize.
+static cl::opt<uint64_t> BoundedMemSetInlineSizeOpt(
+    "bounded-memset-inline-size",
+    cl::desc("Set maximum bound of a variable-length memset to expand inline"),
+    cl::Hidden);
+
 namespace {
 
 struct PreISelIntrinsicLowering {
@@ -81,6 +90,8 @@ struct PreISelIntrinsicLowering {
 
   static bool shouldExpandMemIntrinsicWithSize(Value *Size,
                                                const TargetTransformInfo &TTI);
+  bool tryExpandBoundedMemSet(MemSetInst *Memset,
+                              const TargetTransformInfo &TTI) const;
   bool
   expandMemIntrinsicUses(Function &F,
                          DenseMap<Constant *, GlobalVariable *> &CMap) const;
@@ -288,6 +299,41 @@ static bool canEmitMemcpy(const ModuleLibcallLoweringInfo &ModuleLowering,
   return Lowering.getMemcpyImpl() != RTLIB::Unsupported;
 }
 
+// Expand a memset whose length is not constant but is known to be small, which
+// is faster than calling the library.
+bool PreISelIntrinsicLowering::tryExpandBoundedMemSet(
+    MemSetInst *Memset, const TargetTransformInfo &TTI) const {
+  // This pass runs even at -O0, so skip if e.g. llc -O0 is used.
+  if (!TM || TM->getOptLevel() == CodeGenOptLevel::None)
+    return false;
+
+  Value *Len = Memset->getLength();
+  const Function *F = Memset->getFunction();
+  if (isa<Constant>(Len) || Memset->isVolatile() || F->hasOptNone() ||
+      F->hasMinSize())
+    return false;
+  uint64_t Threshold = BoundedMemSetInlineSizeOpt.getNumOccurrences()
+                           ? BoundedMemSetInlineSizeOpt
+                           : TTI.getMaxBoundedMemSetInlineSize();
+  if (!Threshold)
+    return false;
+
+  const DataLayout &DL = Memset->getDataLayout();
+  KnownBits Known = computeKnownBits(Len, DL);
+  ConstantRange CR = computeConstantRangeIncludingKnownBits(
+      {Len, Known}, /*ForSigned=*/false, SimplifyQuery(DL, Memset));
+  Attribute RangeAttr = Memset->getParamAttr(
+      Memset->getLengthUse().getOperandNo(), Attribute::Range);
+  if (RangeAttr.isValid())
+    CR = CR.intersectWith(RangeAttr.getRange());
+  if (CR.isEmptySet() || CR.getUnsignedMax().ugt(Threshold))
+    return false;
+
+  expandBoundedMemSet(Memset, CR.getUnsignedMin().getZExtValue(),
+                      CR.getUnsignedMax().getZExtValue(), Known);
+  return true;
+}
+
 // Return a value appropriate for use with the memset_pattern16 libcall, if
 // possible and if we know how. (Adapted from equivalent helper in
 // LoopIdiomRecognize).
@@ -406,6 +452,11 @@ bool PreISelIntrinsicLowering::expandMemIntrinsicUses(
       auto *Memset = cast<MemSetInst>(Inst);
       Function *ParentFunc = Memset->getFunction();
       const TargetTransformInfo &TTI = LookupTTI(*ParentFunc);
+      if (tryExpandBoundedMemSet(Memset, TTI)) {
+        Changed = true;
+        Memset->eraseFromParent();
+        break;
+      }
       if (shouldExpandMemIntrinsicWithSize(Memset->getLength(), TTI)) {
         if (UseMemIntrinsicLibFunc &&
             canEmitLibcall(ModuleLibcalls, TM, ParentFunc, RTLIB::MEMSET))
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index d090f69c1476a4b..dc2647c2959079d 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -268,6 +268,12 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override;
   bool useNeonVector(const Type *Ty) const;
 
+  // The expansion uses overlapping unaligned accesses, which are split up under
+  // strict alignment, and MOPS has its own variable-length lowering.
+  uint64_t getMaxBoundedMemSetInlineSize() const override {
+    return ST->requiresStrictAlign() || ST->hasMOPS() ? 0 : 64;
+  }
+
   InstructionCost getMemoryOpCost(
       unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace,
       TTI::TargetCostKind CostKind,
diff --git a/llvm/lib/Transforms/Utils/LowerMemIntrinsics.cpp b/llvm/lib/Transforms/Utils/LowerMemIntrinsics.cpp
index 84131f6592e5005..b5236cb0d471719 100644
--- a/llvm/lib/Transforms/Utils/LowerMemIntrinsics.cpp
+++ b/llvm/lib/Transforms/Utils/LowerMemIntrinsics.cpp
@@ -15,6 +15,7 @@
 #include "llvm/IR/ProfDataUtils.h"
 #include "llvm/ProfileData/InstrProf.h"
 #include "llvm/Support/Debug.h"
+#include "llvm/Support/KnownBits.h"
 #include "llvm/Support/MathExtras.h"
 #include "llvm/Transforms/Utils/BasicBlockUtils.h"
 #include "llvm/Transforms/Utils/LoopUtils.h"
@@ -1141,6 +1142,20 @@ static Value *createMemSetSplat(const DataLayout &DL, IRBuilderBase &B,
   return Result;
 }
 
+static Value *createBoundedMemSetSplat(IRBuilderBase &B, Value *SetValue,
+                                       uint64_t Size) {
+  if (Size == 1)
+    return SetValue;
+  if (Size <= 8) {
+    Type *IntTy = B.getIntNTy(Size * 8);
+    return B.CreateMul(
+        B.CreateZExt(SetValue, IntTy, "setvalue.zext"),
+        ConstantInt::get(IntTy, APInt::getSplat(Size * 8, APInt(8, 1))),
+        "setvalue.splat");
+  }
+  return B.CreateVectorSplat(Size, SetValue, "setvalue.splat");
+}
+
 static void
 createMemSetLoopKnownSize(Instruction *InsertBefore, Value *DstAddr,
                           ConstantInt *Len, Value *SetValue, Align DstAlign,
@@ -1536,6 +1551,108 @@ void llvm::expandMemSetAsLoop(MemSetInst *MemSet,
   expandMemSetAsLoop(MemSet, &TTI);
 }
 
+void llvm::expandBoundedMemSet(MemSetInst *MemSet, uint64_t MinLen,
+                               uint64_t MaxLen, const KnownBits &LenKnown) {
+  assert(MinLen <= MaxLen && "Invalid length bounds");
+  assert(!MemSet->isVolatile() && "Cannot split a volatile memset");
+
+  // Known bits make the length a multiple of MinStoreSize, so no smaller store
+  // is needed, and a length below it can only be zero.
+  uint64_t MinStoreSize = uint64_t(1)
+                          << std::min(LenKnown.countMinTrailingZeros(), 63u);
+  if (MaxLen < MinStoreSize)
+    return;
+
+  Value *Dst = MemSet->getRawDest();
+  Value *Len = MemSet->getLength();
+  Value *SetValue = MemSet->getValue();
+  Type *LenTy = Len->getType();
+  Align DstAlign = MemSet->getDestAlign().valueOrOne();
+  Align TailAlign = commonAlignment(DstAlign, MinStoreSize);
+
+  // Branch to the largest power-of-two Size not exceeding Len. A length in
+  // [Size, 2 * Size] is covered by a store of Size bytes at the start and one
+  // at the end, which may overlap. E.g. for a length of 0, 4, 8 or 12:
+  //
+  //     %len.ge8 = icmp uge i64 %len, 8
+  //     br i1 %len.ge8, label %memset_store8, label %memset_lt8
+  //   memset_store8:                 ; length 8 or 12
+  //     store i64 0, ptr %dst
+  //     %tail.off = sub i64 %len, 8
+  //     %tail = getelementptr inbounds i8, ptr %dst, i64 %tail.off
+  //     store i64 0, ptr %tail       ; <<< may overlap!
+  //     br label %memset_done
+  //   memset_lt8:
+  //     %len.ge4 = icmp uge i64 %len, 4
+  //     br i1 %len.ge4, label %memset_store4, label %memset_done
+  //   memset_store4:                 ; length 4, so no tail store
+  //     store i32 0, ptr %dst
+  //     br label %memset_done
+  //
+  // And for a length in [0, 16] with no known bits:
+  //
+  //     %len.ge16 = icmp uge i64 %len, 16
+  //     br i1 %len.ge16, label %memset_store16, label %memset_lt16
+  //   memset_store16:                ; length 16, so no tail store
+  //     store <16 x i8> zeroinitializer, ptr %dst
+  //     br label %memset_done
+  //   memset_lt16:
+  //     %len.ge8 = icmp uge i64 %len, 8
+  //     br i1 %len.ge8, label %memset_store8, label %memset_lt8
+  //   memset_store8:                 ; length 8 to 15
+  //     store i64 0, ptr %dst
+  //     %tail.off = sub i64 %len, 8
+  //     %tail = getelementptr inbounds i8, ptr %dst, i64 %tail.off
+  //     store i64 0, ptr %tail       ; <<< may overlap!
+  //     br label %memset_done
+  //   ...
+  //   memset_lt2:
+  //     %len.ge1 = icmp uge i64 %len, 1
+  //     br i1 %len.ge1, label %memset_store1, label %memset_done
+  //   memset_store1:                 ; length 1
+  //     store i8 0, ptr %dst
+  //     br label %memset_done
+  uint64_t TopSize = llvm::bit_floor(MaxLen);
+  Instruction *InsertBefore = MemSet;
+  for (uint64_t Size = TopSize; Size >= MinStoreSize; Size /= 2) {
+    Instruction *StoreBefore = InsertBefore;
+    if (MinLen < Size) {
+      IRBuilder<> B(InsertBefore);
+      Value *Cond = B.CreateICmpUGE(Len, ConstantInt::get(LenTy, Size),
+                                    "len.ge" + Twine(Size));
+      if (Size == MinStoreSize) {
+        StoreBefore = SplitBlockAndInsertIfThen(Cond, InsertBefore, false);
+      } else {
+        Instruction *ElseTerm;
+        SplitBlockAndInsertIfThenElse(Cond, InsertBefore, &StoreBefore,
+                                      &ElseTerm);
+        ElseTerm->getParent()->setName("memset_lt" + Twine(Size));
+        InsertBefore = ElseTerm;
+      }
+      StoreBefore->getParent()->setName("memset_store" + Twine(Size));
+    }
+
+    // Do the Size-sized store
+    IRBuilder<> B(StoreBefore);
+    Value *Val = createBoundedMemSetSplat(B, SetValue, Size);
+    B.CreateAlignedStore(Val, Dst, DstAlign);
+
+    // The possibly-overlapping "remainder" store
+    uint64_t MaxLenForSize = Size == TopSize ? MaxLen : 2 * Size - MinStoreSize;
+    if (MaxLenForSize > Size) {
+      Value *TailOff =
+          B.CreateSub(Len, ConstantInt::get(LenTy, Size), "tail.off");
+      Value *TailPtr = B.CreateInBoundsGEP(B.getInt8Ty(), Dst, TailOff, "tail");
+      B.CreateAlignedStore(Val, TailPtr, TailAlign);
+    }
+
+    if (MinLen >= Size)
+      break;
+  }
+
+  MemSet->getParent()->setName("memset_done");
+}
+
 void llvm::expandMemSetPatternAsLoop(MemSetPatternInst *Memset,
                                      const TargetTransformInfo *TTI) {
   createMemSetPatternLoop(
diff --git a/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/bounded-memset.ll b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/bounded-memset.ll
new file mode 100644
index 000000000000000..3dc518876ec6dd6
--- /dev/null
+++ b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/bounded-memset.ll
@@ -0,0 +1,164 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -codegen-opt-level=2 -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefixes=COMMON,EXPAND
+; RUN: opt -codegen-opt-level=0 -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefixes=COMMON,CALL
+
+target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128-Fn32"
+target triple = "aarch64"
+
+define void @range_0_16(ptr %p, i64 %len) {
+; EXPAND-LABEL: define void @range_0_16(
+; EXPAND-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; EXPAND-NEXT:    [[LEN_GE8:%.*]] = icmp uge i64 [[LEN]], 8
+; EXPAND-NEXT:    br i1 [[LEN_GE8]], label %[[MEMSET_STORE8:.*]], label %[[MEMSET_LT8:.*]]
+; EXPAND:       [[MEMSET_STORE8]]:
+; EXPAND-NEXT:    store i64 0, ptr [[P]], align 1
+; EXPAND-NEXT:    [[TAIL_OFF:%.*]] = sub i64 [[LEN]], 8
+; EXPAND-NEXT:    [[TAIL:%.*]] = getelementptr inbounds i8, ptr [[P]], i64 [[TAIL_OFF]]
+; EXPAND-NEXT:    store i64 0, ptr [[TAIL]], align 1
+; EXPAND-NEXT:    br label %[[MEMSET_DONE:.*]]
+; EXPAND:       [[MEMSET_LT8]]:
+; EXPAND-NEXT:    [[LEN_GE4:%.*]] = icmp uge i64 [[LEN]], 4
+; EXPAND-NEXT:    br i1 [[LEN_GE4]], label %[[MEMSET_STORE4:.*]], label %[[MEMSET_LT4:.*]]
+; EXPAND:       [[MEMSET_STORE4]]:
+; EXPAND-NEXT:    store i32 0, ptr [[P]], align 1
+; EXPAND-NEXT:    [[TAIL_OFF1:%.*]] = sub i64 [[LEN]], 4
+; EXPAND-NEXT:    [[TAIL2:%.*]] = getelementptr inbounds i8, ptr [[P]], i64 [[TAIL_OFF1]]
+; EXPAND-NEXT:    store i32 0, ptr [[TAIL2]], align 1
+; EXPAND-NEXT:    br label %[[BB3:.*]]
+; EXPAND:       [[MEMSET_LT4]]:
+; EXPAND-NEXT:    [[LEN_GE2:%.*]] = icmp uge i64 [[LEN]], 2
+; EXPAND-NEXT:    br i1 [[LEN_GE2]], label %[[MEMSET_STORE2:.*]], label %[[MEMSET_LT2:.*]]
+; EXPAND:       [[MEMSET_STORE2]]:
+; EXPAND-NEXT:    store i16 0, ptr [[P]], align 1
+; EXPAND-NEXT:    [[TAIL_OFF3:%.*]] = sub i64 [[LEN]], 2
+; EXPAND-NEXT:    [[TAIL4:%.*]] = getelementptr inbounds i8, ptr [[P]], i64 [[TAIL_OFF3]]
+; EXPAND-NEXT:    store i16 0, ptr [[TAIL4]], align 1
+; EXPAND-NEXT:    br label %[[BB2:.*]]
+; EXPAND:       [[MEMSET_LT2]]:
+; EXPAND-NEXT:    [[LEN_GE1:%.*]] = icmp uge i64 [[LEN]], 1
+; EXPAND-NEXT:    br i1 [[LEN_GE1]], label %[[MEMSET_STORE1:.*]], label %[[BB1:.*]]
+; EXPAND:       [[MEMSET_STORE1]]:
+; EXPAND-NEXT:    store i8 0, ptr [[P]], align 1
+; EXPAND-NEXT:    br label %[[BB1]]
+; EXPAND:       [[BB1]]:
+; EXPAND-NEXT:    br label %[[BB2]]
+; EXPAND:       [[BB2]]:
+; EXPAND-NEXT:    br label %[[BB3]]
+; EXPAND:       [[BB3]]:
+; EXPAND-NEXT:    br label %[[MEMSET_DONE]]
+; EXPAND:       [[MEMSET_DONE]]:
+; EXPAND-NEXT:    ret void
+;
+; CALL-LABEL: define void @range_0_16(
+; CALL-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; CALL-NEXT:    call void @llvm.memset.p0.i64(ptr align 1 [[P]], i8 0, i64 range(i64 0, 16) [[LEN]], i1 false)
+; CALL-NEXT:    ret void
+;
+  call void @llvm.memset.p0.i64(ptr align 1 %p, i8 0, i64 range(i64 0, 16) %len, i1 false)
+  ret void
+}
+
+define void @range_min_8(ptr %p, i64 %len) {
+; EXPAND-LABEL: define void @range_min_8(
+; EXPAND-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; EXPAND-NEXT:    [[LEN_GE16:%.*]] = icmp uge i64 [[LEN]], 16
+; EXPAND-NEXT:    br i1 [[LEN_GE16]], label %[[MEMSET_STORE16:.*]], label %[[MEMSET_LT16:.*]]
+; EXPAND:       [[MEMSET_STORE16]]:
+; EXPAND-NEXT:    store <16 x i8> zeroinitializer, ptr [[P]], align 1
+; EXPAND-NEXT:    br label %[[MEMSET_DONE:.*]]
+; EXPAND:       [[MEMSET_LT16]]:
+; EXPAND-NEXT:    store i64 0, ptr [[P]], align 1
+; EXPAND-NEXT:    [[TAIL_OFF:%.*]] = sub i64 [[LEN]], 8
+; EXPAND-NEXT:    [[TAIL:%.*]] = getelementptr inbounds i8, ptr [[P]], i64 [[TAIL_OFF]]
+; EXPAND-NEXT:    store i64 0, ptr [[TAIL]], align 1
+; EXPAND-NEXT:    br label %[[MEMSET_DONE]]
+; EXPAND:       [[MEMSET_DONE]]:
+; EXPAND-NEXT:    ret void
+;
+; CALL-LABEL: define void @range_min_8(
+; CALL-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; CALL-NEXT:    call void @llvm.memset.p0.i64(ptr align 1 [[P]], i8 0, i64 range(i64 8, 17) [[LEN]], i1 false)
+; CALL-NEXT:    ret void
+;
+  call void @llvm.memset.p0.i64(ptr align 1 %p, i8 0, i64 range(i64 8, 17) %len, i1 false)
+  ret void
+}
+
+define void @over_threshold(ptr %p, i64 %len) {
+; COMMON-LABEL: define void @over_threshold(
+; COMMON-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; COMMON-NEXT:    call void @llvm.memset.p0.i64(ptr align 1 [[P]], i8 0, i64 range(i64 0, 66) [[LEN]], i1 false)
+; COMMON-NEXT:    ret void
+;
+  call void @llvm.memset.p0.i64(ptr align 1 %p, i8 0, i64 range(i64 0, 66) %len, i1 false)
+  ret void
+}
+
+define void @volatile(ptr %p, i64 %len) {
+; COMMON-LABEL: define void @volatile(
+; COMMON-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; COMMON-NEXT:    call void @llvm.memset.p0.i64(ptr align 1 [[P]], i8 0, i64 range(i64 0, 17) [[LEN]], i1 true)
+; COMMON-NEXT:    ret void
+;
+  call void @llvm.memset.p0.i64(ptr align 1 %p, i8 0, i64 range(i64 0, 17) %len, i1 true)
+  ret void
+}
+
+define void @strict_align(ptr %p, i64 %len) "target-features"="+strict-align" {
+; COMMON-LABEL: define void @strict_align(
+; COMMON-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) #[[ATTR0:[0-9]+]] {
+; COMMON-NEXT:    call void @llvm.memset.p0.i64(ptr align 1 [[P]], i8 0, i64 range(i64 0, 17) [[LEN]], i1 false)
+; COMMON-NEXT:    ret void
+;
+  call void @llvm.memset.p0.i64(ptr align 1 %p, i8 0, i64 range(i64 0, 17) %len, i1 false)
+  ret void
+}
+
+defi...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/228420


More information about the llvm-commits mailing list