[llvm] [LowerMemIntrinsics] Expand small variable-length memsets inline (PR #228420)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 05:38:29 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Kuba (Brecka) Mracek (kubamracek)
<details>
<summary>Changes</summary>
Part 2 of the motivation mentioned in https://github.com/llvm/llvm-project/pull/226801: Optimizing a recognized memset from a loop. Lowering a (non-constant) llvm.memset with a small upper bound into a memset libcall is more expensive. Instead emit branch(es) for per-size class direct stores.
Commit message:
> A memset whose length is not a constant is always lowered to a call to memset, even when the length is known to be small. Expand such a memset inline instead when the upper bound of its length is at most a target-provided threshold.
>
> The bounds of the length come from (1) range attribute on the call, (2) known bits, (3) computeConstantRange from ValueTracking.
>
> The expansion branches on the length to the largest power-of-two size class, and covers a length in [Size, 2 * Size] with one "regular" store at the start + one "tail" store at the end, which may overlap.
>
> A new `getMaxBoundedMemSetInlineSize` TTI hook is added, which can be overridden with -bounded-memset-inline-size.
Assisted-by: Claude
---
Patch is 22.29 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/228420.diff
8 Files Affected:
- (modified) llvm/include/llvm/Analysis/TargetTransformInfo.h (+6)
- (modified) llvm/include/llvm/Analysis/TargetTransformInfoImpl.h (+2)
- (modified) llvm/include/llvm/Transforms/Utils/LowerMemIntrinsics.h (+7)
- (modified) llvm/lib/Analysis/TargetTransformInfo.cpp (+4)
- (modified) llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp (+51)
- (modified) llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h (+6)
- (modified) llvm/lib/Transforms/Utils/LowerMemIntrinsics.cpp (+117)
- (added) llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/bounded-memset.ll (+164)
``````````diff
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index a33e6f62e941ed7..6068abe3b3bfece 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -487,6 +487,12 @@ class TargetTransformInfo {
/// profitable to inline the call.
LLVM_ABI uint64_t getMaxMemIntrinsicInlineSizeThreshold() const;
+ /// Returns the maximum length in bytes for which a memset with a
+ /// non-constant length, known to be bounded by that value, is preferably
+ /// expanded inline instead of calling the library. Zero disables the
+ /// expansion.
+ LLVM_ABI uint64_t getMaxBoundedMemSetInlineSize() const;
+
/// \return The estimated number of case clusters when lowering \p 'SI'.
/// \p JTSize Set a jump table size only when \p SI is suitable for a jump
/// table.
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index a19c122c16f20c5..f0a2d755a298256 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -116,6 +116,8 @@ class LLVM_ABI TargetTransformInfoImplBase {
virtual uint64_t getMaxMemIntrinsicInlineSizeThreshold() const { return 64; }
+ virtual uint64_t getMaxBoundedMemSetInlineSize() const { return 0; }
+
// Although this default value is arbitrary, it is not random. It is assumed
// that a condition that evaluates the same way by a higher percentage than
// this is best represented as control flow. Therefore, the default value N
diff --git a/llvm/include/llvm/Transforms/Utils/LowerMemIntrinsics.h b/llvm/include/llvm/Transforms/Utils/LowerMemIntrinsics.h
index 55c7119cd8fdbec..3509574b8080c69 100644
--- a/llvm/include/llvm/Transforms/Utils/LowerMemIntrinsics.h
+++ b/llvm/include/llvm/Transforms/Utils/LowerMemIntrinsics.h
@@ -23,6 +23,7 @@ namespace llvm {
class AnyMemCpyInst;
class ConstantInt;
class Instruction;
+struct KnownBits;
class MemCpyInst;
class MemMoveInst;
class MemSetInst;
@@ -71,6 +72,12 @@ LLVM_ABI void expandMemSetAsLoop(MemSetInst *MemSet,
LLVM_ABI void expandMemSetAsLoop(MemSetInst *MemSet,
const TargetTransformInfo &TTI);
+/// Expand \p MemSet, whose length is known to be in [\p MinLen, \p MaxLen] and
+/// to have the known bits \p LenKnown, into a tree of branches on the length
+/// leading to at most two overlapping stores each. \p MemSet is not deleted.
+LLVM_ABI void expandBoundedMemSet(MemSetInst *MemSet, uint64_t MinLen,
+ uint64_t MaxLen, const KnownBits &LenKnown);
+
/// Expand \p MemSetPattern as a loop. \p MemSet is not deleted.
/// If \p TTI is provided, the memset.pattern is expanded according to the
/// target's preferences. Otherwise, it is expanded as an element-wise loop.
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index af73a615f25fa61..5d4e3c7811cde84 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -1336,6 +1336,10 @@ uint64_t TargetTransformInfo::getMaxMemIntrinsicInlineSizeThreshold() const {
return TTIImpl->getMaxMemIntrinsicInlineSizeThreshold();
}
+uint64_t TargetTransformInfo::getMaxBoundedMemSetInlineSize() const {
+ return TTIImpl->getMaxBoundedMemSetInlineSize();
+}
+
InstructionCost TargetTransformInfo::getArithmeticReductionCost(
unsigned Opcode, VectorType *Ty, std::optional<FastMathFlags> FMF,
TTI::TargetCostKind CostKind) const {
diff --git a/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp b/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
index 77d8d44ea3586d3..560a81d9f32f707 100644
--- a/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
+++ b/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
@@ -17,12 +17,14 @@
#include "llvm/Analysis/ObjCARCUtil.h"
#include "llvm/Analysis/TargetLibraryInfo.h"
#include "llvm/Analysis/TargetTransformInfo.h"
+#include "llvm/Analysis/ValueTracking.h"
#include "llvm/CodeGen/ExpandVectorPredication.h"
#include "llvm/CodeGen/LibcallLoweringInfo.h"
#include "llvm/CodeGen/Passes.h"
#include "llvm/CodeGen/RuntimeLibcallUtil.h"
#include "llvm/CodeGen/TargetLowering.h"
#include "llvm/CodeGen/TargetPassConfig.h"
+#include "llvm/IR/ConstantRange.h"
#include "llvm/IR/Function.h"
#include "llvm/IR/GlobalValue.h"
#include "llvm/IR/IRBuilder.h"
@@ -56,6 +58,13 @@ static cl::opt<int64_t> MemIntrinsicExpandSizeThresholdOpt(
cl::desc("Set minimum mem intrinsic size to expand in IR"), cl::init(-1),
cl::Hidden);
+/// Maximum bound of a memset with non-constant length for it to be expanded
+/// inline. Overrides TargetTransformInfo::getMaxBoundedMemSetInlineSize.
+static cl::opt<uint64_t> BoundedMemSetInlineSizeOpt(
+ "bounded-memset-inline-size",
+ cl::desc("Set maximum bound of a variable-length memset to expand inline"),
+ cl::Hidden);
+
namespace {
struct PreISelIntrinsicLowering {
@@ -81,6 +90,8 @@ struct PreISelIntrinsicLowering {
static bool shouldExpandMemIntrinsicWithSize(Value *Size,
const TargetTransformInfo &TTI);
+ bool tryExpandBoundedMemSet(MemSetInst *Memset,
+ const TargetTransformInfo &TTI) const;
bool
expandMemIntrinsicUses(Function &F,
DenseMap<Constant *, GlobalVariable *> &CMap) const;
@@ -288,6 +299,41 @@ static bool canEmitMemcpy(const ModuleLibcallLoweringInfo &ModuleLowering,
return Lowering.getMemcpyImpl() != RTLIB::Unsupported;
}
+// Expand a memset whose length is not constant but is known to be small, which
+// is faster than calling the library.
+bool PreISelIntrinsicLowering::tryExpandBoundedMemSet(
+ MemSetInst *Memset, const TargetTransformInfo &TTI) const {
+ // This pass runs even at -O0, so skip if e.g. llc -O0 is used.
+ if (!TM || TM->getOptLevel() == CodeGenOptLevel::None)
+ return false;
+
+ Value *Len = Memset->getLength();
+ const Function *F = Memset->getFunction();
+ if (isa<Constant>(Len) || Memset->isVolatile() || F->hasOptNone() ||
+ F->hasMinSize())
+ return false;
+ uint64_t Threshold = BoundedMemSetInlineSizeOpt.getNumOccurrences()
+ ? BoundedMemSetInlineSizeOpt
+ : TTI.getMaxBoundedMemSetInlineSize();
+ if (!Threshold)
+ return false;
+
+ const DataLayout &DL = Memset->getDataLayout();
+ KnownBits Known = computeKnownBits(Len, DL);
+ ConstantRange CR = computeConstantRangeIncludingKnownBits(
+ {Len, Known}, /*ForSigned=*/false, SimplifyQuery(DL, Memset));
+ Attribute RangeAttr = Memset->getParamAttr(
+ Memset->getLengthUse().getOperandNo(), Attribute::Range);
+ if (RangeAttr.isValid())
+ CR = CR.intersectWith(RangeAttr.getRange());
+ if (CR.isEmptySet() || CR.getUnsignedMax().ugt(Threshold))
+ return false;
+
+ expandBoundedMemSet(Memset, CR.getUnsignedMin().getZExtValue(),
+ CR.getUnsignedMax().getZExtValue(), Known);
+ return true;
+}
+
// Return a value appropriate for use with the memset_pattern16 libcall, if
// possible and if we know how. (Adapted from equivalent helper in
// LoopIdiomRecognize).
@@ -406,6 +452,11 @@ bool PreISelIntrinsicLowering::expandMemIntrinsicUses(
auto *Memset = cast<MemSetInst>(Inst);
Function *ParentFunc = Memset->getFunction();
const TargetTransformInfo &TTI = LookupTTI(*ParentFunc);
+ if (tryExpandBoundedMemSet(Memset, TTI)) {
+ Changed = true;
+ Memset->eraseFromParent();
+ break;
+ }
if (shouldExpandMemIntrinsicWithSize(Memset->getLength(), TTI)) {
if (UseMemIntrinsicLibFunc &&
canEmitLibcall(ModuleLibcalls, TM, ParentFunc, RTLIB::MEMSET))
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index d090f69c1476a4b..dc2647c2959079d 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -268,6 +268,12 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override;
bool useNeonVector(const Type *Ty) const;
+ // The expansion uses overlapping unaligned accesses, which are split up under
+ // strict alignment, and MOPS has its own variable-length lowering.
+ uint64_t getMaxBoundedMemSetInlineSize() const override {
+ return ST->requiresStrictAlign() || ST->hasMOPS() ? 0 : 64;
+ }
+
InstructionCost getMemoryOpCost(
unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace,
TTI::TargetCostKind CostKind,
diff --git a/llvm/lib/Transforms/Utils/LowerMemIntrinsics.cpp b/llvm/lib/Transforms/Utils/LowerMemIntrinsics.cpp
index 84131f6592e5005..b5236cb0d471719 100644
--- a/llvm/lib/Transforms/Utils/LowerMemIntrinsics.cpp
+++ b/llvm/lib/Transforms/Utils/LowerMemIntrinsics.cpp
@@ -15,6 +15,7 @@
#include "llvm/IR/ProfDataUtils.h"
#include "llvm/ProfileData/InstrProf.h"
#include "llvm/Support/Debug.h"
+#include "llvm/Support/KnownBits.h"
#include "llvm/Support/MathExtras.h"
#include "llvm/Transforms/Utils/BasicBlockUtils.h"
#include "llvm/Transforms/Utils/LoopUtils.h"
@@ -1141,6 +1142,20 @@ static Value *createMemSetSplat(const DataLayout &DL, IRBuilderBase &B,
return Result;
}
+static Value *createBoundedMemSetSplat(IRBuilderBase &B, Value *SetValue,
+ uint64_t Size) {
+ if (Size == 1)
+ return SetValue;
+ if (Size <= 8) {
+ Type *IntTy = B.getIntNTy(Size * 8);
+ return B.CreateMul(
+ B.CreateZExt(SetValue, IntTy, "setvalue.zext"),
+ ConstantInt::get(IntTy, APInt::getSplat(Size * 8, APInt(8, 1))),
+ "setvalue.splat");
+ }
+ return B.CreateVectorSplat(Size, SetValue, "setvalue.splat");
+}
+
static void
createMemSetLoopKnownSize(Instruction *InsertBefore, Value *DstAddr,
ConstantInt *Len, Value *SetValue, Align DstAlign,
@@ -1536,6 +1551,108 @@ void llvm::expandMemSetAsLoop(MemSetInst *MemSet,
expandMemSetAsLoop(MemSet, &TTI);
}
+void llvm::expandBoundedMemSet(MemSetInst *MemSet, uint64_t MinLen,
+ uint64_t MaxLen, const KnownBits &LenKnown) {
+ assert(MinLen <= MaxLen && "Invalid length bounds");
+ assert(!MemSet->isVolatile() && "Cannot split a volatile memset");
+
+ // Known bits make the length a multiple of MinStoreSize, so no smaller store
+ // is needed, and a length below it can only be zero.
+ uint64_t MinStoreSize = uint64_t(1)
+ << std::min(LenKnown.countMinTrailingZeros(), 63u);
+ if (MaxLen < MinStoreSize)
+ return;
+
+ Value *Dst = MemSet->getRawDest();
+ Value *Len = MemSet->getLength();
+ Value *SetValue = MemSet->getValue();
+ Type *LenTy = Len->getType();
+ Align DstAlign = MemSet->getDestAlign().valueOrOne();
+ Align TailAlign = commonAlignment(DstAlign, MinStoreSize);
+
+ // Branch to the largest power-of-two Size not exceeding Len. A length in
+ // [Size, 2 * Size] is covered by a store of Size bytes at the start and one
+ // at the end, which may overlap. E.g. for a length of 0, 4, 8 or 12:
+ //
+ // %len.ge8 = icmp uge i64 %len, 8
+ // br i1 %len.ge8, label %memset_store8, label %memset_lt8
+ // memset_store8: ; length 8 or 12
+ // store i64 0, ptr %dst
+ // %tail.off = sub i64 %len, 8
+ // %tail = getelementptr inbounds i8, ptr %dst, i64 %tail.off
+ // store i64 0, ptr %tail ; <<< may overlap!
+ // br label %memset_done
+ // memset_lt8:
+ // %len.ge4 = icmp uge i64 %len, 4
+ // br i1 %len.ge4, label %memset_store4, label %memset_done
+ // memset_store4: ; length 4, so no tail store
+ // store i32 0, ptr %dst
+ // br label %memset_done
+ //
+ // And for a length in [0, 16] with no known bits:
+ //
+ // %len.ge16 = icmp uge i64 %len, 16
+ // br i1 %len.ge16, label %memset_store16, label %memset_lt16
+ // memset_store16: ; length 16, so no tail store
+ // store <16 x i8> zeroinitializer, ptr %dst
+ // br label %memset_done
+ // memset_lt16:
+ // %len.ge8 = icmp uge i64 %len, 8
+ // br i1 %len.ge8, label %memset_store8, label %memset_lt8
+ // memset_store8: ; length 8 to 15
+ // store i64 0, ptr %dst
+ // %tail.off = sub i64 %len, 8
+ // %tail = getelementptr inbounds i8, ptr %dst, i64 %tail.off
+ // store i64 0, ptr %tail ; <<< may overlap!
+ // br label %memset_done
+ // ...
+ // memset_lt2:
+ // %len.ge1 = icmp uge i64 %len, 1
+ // br i1 %len.ge1, label %memset_store1, label %memset_done
+ // memset_store1: ; length 1
+ // store i8 0, ptr %dst
+ // br label %memset_done
+ uint64_t TopSize = llvm::bit_floor(MaxLen);
+ Instruction *InsertBefore = MemSet;
+ for (uint64_t Size = TopSize; Size >= MinStoreSize; Size /= 2) {
+ Instruction *StoreBefore = InsertBefore;
+ if (MinLen < Size) {
+ IRBuilder<> B(InsertBefore);
+ Value *Cond = B.CreateICmpUGE(Len, ConstantInt::get(LenTy, Size),
+ "len.ge" + Twine(Size));
+ if (Size == MinStoreSize) {
+ StoreBefore = SplitBlockAndInsertIfThen(Cond, InsertBefore, false);
+ } else {
+ Instruction *ElseTerm;
+ SplitBlockAndInsertIfThenElse(Cond, InsertBefore, &StoreBefore,
+ &ElseTerm);
+ ElseTerm->getParent()->setName("memset_lt" + Twine(Size));
+ InsertBefore = ElseTerm;
+ }
+ StoreBefore->getParent()->setName("memset_store" + Twine(Size));
+ }
+
+ // Do the Size-sized store
+ IRBuilder<> B(StoreBefore);
+ Value *Val = createBoundedMemSetSplat(B, SetValue, Size);
+ B.CreateAlignedStore(Val, Dst, DstAlign);
+
+ // The possibly-overlapping "remainder" store
+ uint64_t MaxLenForSize = Size == TopSize ? MaxLen : 2 * Size - MinStoreSize;
+ if (MaxLenForSize > Size) {
+ Value *TailOff =
+ B.CreateSub(Len, ConstantInt::get(LenTy, Size), "tail.off");
+ Value *TailPtr = B.CreateInBoundsGEP(B.getInt8Ty(), Dst, TailOff, "tail");
+ B.CreateAlignedStore(Val, TailPtr, TailAlign);
+ }
+
+ if (MinLen >= Size)
+ break;
+ }
+
+ MemSet->getParent()->setName("memset_done");
+}
+
void llvm::expandMemSetPatternAsLoop(MemSetPatternInst *Memset,
const TargetTransformInfo *TTI) {
createMemSetPatternLoop(
diff --git a/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/bounded-memset.ll b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/bounded-memset.ll
new file mode 100644
index 000000000000000..3dc518876ec6dd6
--- /dev/null
+++ b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/bounded-memset.ll
@@ -0,0 +1,164 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -codegen-opt-level=2 -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefixes=COMMON,EXPAND
+; RUN: opt -codegen-opt-level=0 -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefixes=COMMON,CALL
+
+target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128-Fn32"
+target triple = "aarch64"
+
+define void @range_0_16(ptr %p, i64 %len) {
+; EXPAND-LABEL: define void @range_0_16(
+; EXPAND-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; EXPAND-NEXT: [[LEN_GE8:%.*]] = icmp uge i64 [[LEN]], 8
+; EXPAND-NEXT: br i1 [[LEN_GE8]], label %[[MEMSET_STORE8:.*]], label %[[MEMSET_LT8:.*]]
+; EXPAND: [[MEMSET_STORE8]]:
+; EXPAND-NEXT: store i64 0, ptr [[P]], align 1
+; EXPAND-NEXT: [[TAIL_OFF:%.*]] = sub i64 [[LEN]], 8
+; EXPAND-NEXT: [[TAIL:%.*]] = getelementptr inbounds i8, ptr [[P]], i64 [[TAIL_OFF]]
+; EXPAND-NEXT: store i64 0, ptr [[TAIL]], align 1
+; EXPAND-NEXT: br label %[[MEMSET_DONE:.*]]
+; EXPAND: [[MEMSET_LT8]]:
+; EXPAND-NEXT: [[LEN_GE4:%.*]] = icmp uge i64 [[LEN]], 4
+; EXPAND-NEXT: br i1 [[LEN_GE4]], label %[[MEMSET_STORE4:.*]], label %[[MEMSET_LT4:.*]]
+; EXPAND: [[MEMSET_STORE4]]:
+; EXPAND-NEXT: store i32 0, ptr [[P]], align 1
+; EXPAND-NEXT: [[TAIL_OFF1:%.*]] = sub i64 [[LEN]], 4
+; EXPAND-NEXT: [[TAIL2:%.*]] = getelementptr inbounds i8, ptr [[P]], i64 [[TAIL_OFF1]]
+; EXPAND-NEXT: store i32 0, ptr [[TAIL2]], align 1
+; EXPAND-NEXT: br label %[[BB3:.*]]
+; EXPAND: [[MEMSET_LT4]]:
+; EXPAND-NEXT: [[LEN_GE2:%.*]] = icmp uge i64 [[LEN]], 2
+; EXPAND-NEXT: br i1 [[LEN_GE2]], label %[[MEMSET_STORE2:.*]], label %[[MEMSET_LT2:.*]]
+; EXPAND: [[MEMSET_STORE2]]:
+; EXPAND-NEXT: store i16 0, ptr [[P]], align 1
+; EXPAND-NEXT: [[TAIL_OFF3:%.*]] = sub i64 [[LEN]], 2
+; EXPAND-NEXT: [[TAIL4:%.*]] = getelementptr inbounds i8, ptr [[P]], i64 [[TAIL_OFF3]]
+; EXPAND-NEXT: store i16 0, ptr [[TAIL4]], align 1
+; EXPAND-NEXT: br label %[[BB2:.*]]
+; EXPAND: [[MEMSET_LT2]]:
+; EXPAND-NEXT: [[LEN_GE1:%.*]] = icmp uge i64 [[LEN]], 1
+; EXPAND-NEXT: br i1 [[LEN_GE1]], label %[[MEMSET_STORE1:.*]], label %[[BB1:.*]]
+; EXPAND: [[MEMSET_STORE1]]:
+; EXPAND-NEXT: store i8 0, ptr [[P]], align 1
+; EXPAND-NEXT: br label %[[BB1]]
+; EXPAND: [[BB1]]:
+; EXPAND-NEXT: br label %[[BB2]]
+; EXPAND: [[BB2]]:
+; EXPAND-NEXT: br label %[[BB3]]
+; EXPAND: [[BB3]]:
+; EXPAND-NEXT: br label %[[MEMSET_DONE]]
+; EXPAND: [[MEMSET_DONE]]:
+; EXPAND-NEXT: ret void
+;
+; CALL-LABEL: define void @range_0_16(
+; CALL-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; CALL-NEXT: call void @llvm.memset.p0.i64(ptr align 1 [[P]], i8 0, i64 range(i64 0, 16) [[LEN]], i1 false)
+; CALL-NEXT: ret void
+;
+ call void @llvm.memset.p0.i64(ptr align 1 %p, i8 0, i64 range(i64 0, 16) %len, i1 false)
+ ret void
+}
+
+define void @range_min_8(ptr %p, i64 %len) {
+; EXPAND-LABEL: define void @range_min_8(
+; EXPAND-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; EXPAND-NEXT: [[LEN_GE16:%.*]] = icmp uge i64 [[LEN]], 16
+; EXPAND-NEXT: br i1 [[LEN_GE16]], label %[[MEMSET_STORE16:.*]], label %[[MEMSET_LT16:.*]]
+; EXPAND: [[MEMSET_STORE16]]:
+; EXPAND-NEXT: store <16 x i8> zeroinitializer, ptr [[P]], align 1
+; EXPAND-NEXT: br label %[[MEMSET_DONE:.*]]
+; EXPAND: [[MEMSET_LT16]]:
+; EXPAND-NEXT: store i64 0, ptr [[P]], align 1
+; EXPAND-NEXT: [[TAIL_OFF:%.*]] = sub i64 [[LEN]], 8
+; EXPAND-NEXT: [[TAIL:%.*]] = getelementptr inbounds i8, ptr [[P]], i64 [[TAIL_OFF]]
+; EXPAND-NEXT: store i64 0, ptr [[TAIL]], align 1
+; EXPAND-NEXT: br label %[[MEMSET_DONE]]
+; EXPAND: [[MEMSET_DONE]]:
+; EXPAND-NEXT: ret void
+;
+; CALL-LABEL: define void @range_min_8(
+; CALL-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; CALL-NEXT: call void @llvm.memset.p0.i64(ptr align 1 [[P]], i8 0, i64 range(i64 8, 17) [[LEN]], i1 false)
+; CALL-NEXT: ret void
+;
+ call void @llvm.memset.p0.i64(ptr align 1 %p, i8 0, i64 range(i64 8, 17) %len, i1 false)
+ ret void
+}
+
+define void @over_threshold(ptr %p, i64 %len) {
+; COMMON-LABEL: define void @over_threshold(
+; COMMON-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; COMMON-NEXT: call void @llvm.memset.p0.i64(ptr align 1 [[P]], i8 0, i64 range(i64 0, 66) [[LEN]], i1 false)
+; COMMON-NEXT: ret void
+;
+ call void @llvm.memset.p0.i64(ptr align 1 %p, i8 0, i64 range(i64 0, 66) %len, i1 false)
+ ret void
+}
+
+define void @volatile(ptr %p, i64 %len) {
+; COMMON-LABEL: define void @volatile(
+; COMMON-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) {
+; COMMON-NEXT: call void @llvm.memset.p0.i64(ptr align 1 [[P]], i8 0, i64 range(i64 0, 17) [[LEN]], i1 true)
+; COMMON-NEXT: ret void
+;
+ call void @llvm.memset.p0.i64(ptr align 1 %p, i8 0, i64 range(i64 0, 17) %len, i1 true)
+ ret void
+}
+
+define void @strict_align(ptr %p, i64 %len) "target-features"="+strict-align" {
+; COMMON-LABEL: define void @strict_align(
+; COMMON-SAME: ptr [[P:%.*]], i64 [[LEN:%.*]]) #[[ATTR0:[0-9]+]] {
+; COMMON-NEXT: call void @llvm.memset.p0.i64(ptr align 1 [[P]], i8 0, i64 range(i64 0, 17) [[LEN]], i1 false)
+; COMMON-NEXT: ret void
+;
+ call void @llvm.memset.p0.i64(ptr align 1 %p, i8 0, i64 range(i64 0, 17) %len, i1 false)
+ ret void
+}
+
+defi...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/228420
More information about the llvm-commits
mailing list