[llvm] [AggressiveInstCombine] Guard memset with length in [0, 1] (PR #213240)

via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 01:24:18 PDT 2026


https://github.com/hsnaveen2u updated https://github.com/llvm/llvm-project/pull/213240

>From 86e30728f1369deaceae7b61020e1af6c47b77a3 Mon Sep 17 00:00:00 2001
From: Naveen <naveen.siddegowda at oss.qualcomm.com>
Date: Mon, 3 Aug 2026 13:12:41 +0900
Subject: [PATCH] [AggressiveInstCombine] Guard memset with length in [0, 1]

Use computeKnownBits to identify nonconstant memset lengths whose
possible values are limited to zero and one. The check is integrated
into the existing instruction loop in foldUnusualPatterns via the
dedicated helper foldMemSetZeroOrOneLength.

Insert a conditional branch around the memset and specialize the
executed path to a constant length of one. A following InstCombine
pass can then replace it with a byte store, including for a
nonconstant fill value.

Do not transform wider ranges such as [0, 2].

Fixes #213027.
Assisted by GPT-5
Signed-off-by: Naveen <naveen.siddegowda at oss.qualcomm.com>
---
 .../AggressiveInstCombine.cpp                 |  36 ++++++
 .../AggressiveInstCombine/memset.ll           | 104 ++++++++++++++++++
 2 files changed, 140 insertions(+)
 create mode 100644 llvm/test/Transforms/AggressiveInstCombine/memset.ll

diff --git a/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp b/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp
index f72ff61f028db..6f7ed018ea103 100644
--- a/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp
+++ b/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp
@@ -29,6 +29,7 @@
 #include "llvm/IR/Function.h"
 #include "llvm/IR/IRBuilder.h"
 #include "llvm/IR/Instruction.h"
+#include "llvm/IR/IntrinsicInst.h"
 #include "llvm/IR/MDBuilder.h"
 #include "llvm/IR/PatternMatch.h"
 #include "llvm/IR/ProfDataUtils.h"
@@ -57,6 +58,7 @@ STATISTIC(NumSelectCTTZFolded,
           "Number of select-based split cttz patterns folded");
 STATISTIC(NumSelectCTLZFolded,
           "Number of select-based split ctlz patterns folded");
+STATISTIC(NumMemSetsGuarded, "Number of memsets guarded for a zero length");
 
 static cl::opt<unsigned> MaxInstrsToScan(
     "aggressive-instcombine-max-scan-instrs", cl::init(64), cl::Hidden,
@@ -2462,6 +2464,37 @@ static bool foldMulHigh(Instruction &I) {
   return false;
 }
 
+/// Guard a memset whose nonconstant length is known to be in [0, 1].
+/// Inserts a conditional branch around the memset and specialises the
+/// executed path to a constant length of one.
+static bool foldMemSetZeroOrOneLength(MemSetInst *MI, const DataLayout &DL,
+                                      TargetLibraryInfo &TLI, DominatorTree &DT,
+                                      AssumptionCache &AC,
+                                      bool &MadeCFGChange) {
+  if (isa<ConstantInt>(MI->getLength()))
+    return false;
+
+  SimplifyQuery SQ(DL, &TLI, &DT, &AC, nullptr,
+                   /*UseInstrInfo=*/true, /*CanUseUndef=*/false);
+  KnownBits KnownLen =
+      computeKnownBits(MI->getLength(), SQ.getWithInstruction(MI));
+  if (!KnownLen.getMaxValue().isOne())
+    return false;
+
+  DomTreeUpdater DTU(DT, DomTreeUpdater::UpdateStrategy::Lazy);
+  IRBuilder<> B(MI);
+  B.SetCurrentDebugLocation(MI->getDebugLoc());
+  Value *IsNonZero = B.CreateIsNotNull(MI->getLength(), "memset.notzero");
+  Instruction *ThenTerm = SplitBlockAndInsertIfThen(
+      IsNonZero, MI->getIterator(), /*Unreachable=*/false,
+      /*BranchWeights=*/nullptr, &DTU);
+  MI->moveBefore(ThenTerm->getIterator());
+  MI->setLength((uint64_t)1);
+  ++NumMemSetsGuarded;
+  MadeCFGChange = true;
+  return true;
+}
+
 /// This is the entry point for folds that could be implemented in regular
 /// InstCombine, but they are separated because they are not expected to
 /// occur frequently and/or have more than a constant-length pattern match.
@@ -2495,6 +2528,9 @@ static bool foldUnusualPatterns(Function &F, DominatorTree &DT,
       MadeChange |= foldPatternedLoads(I, DL);
       MadeChange |= foldICmpOrChain(I, DL, TTI, AA, DT);
       MadeChange |= foldMulHigh(I);
+      if (auto *MI = dyn_cast<MemSetInst>(&I))
+        MadeChange |=
+            foldMemSetZeroOrOneLength(MI, DL, TLI, DT, AC, MadeCFGChange);
       // NOTE: This function introduces erasing of the instruction `I`, so it
       // needs to be called at the end of this sequence, otherwise we may make
       // bugs.
diff --git a/llvm/test/Transforms/AggressiveInstCombine/memset.ll b/llvm/test/Transforms/AggressiveInstCombine/memset.ll
new file mode 100644
index 0000000000000..6cf8a27d14460
--- /dev/null
+++ b/llvm/test/Transforms/AggressiveInstCombine/memset.ll
@@ -0,0 +1,104 @@
+; RUN: opt -passes=aggressive-instcombine -S < %s | FileCheck %s --check-prefix=AIC
+; RUN: opt -passes='aggressive-instcombine,instcombine' -S < %s | FileCheck %s --check-prefix=COMBINED
+
+declare void @llvm.memset.p0.i64(ptr nocapture writeonly, i8, i64, i1 immarg)
+
+define void @range_0_1(ptr %dst, i8 %value, i64 %n) {
+; AIC-LABEL: define void @range_0_1(
+; AIC-SAME: ptr [[DST:%.*]], i8 [[VALUE:%.*]], i64 [[N:%.*]]) {
+; AIC:       entry:
+; AIC-NEXT:    [[LEN:%.*]] = and i64 [[N]], 1
+; AIC-NEXT:    [[MEMSET_NOTZERO:%.*]] = icmp ne i64 [[LEN]], 0
+; AIC-NEXT:    br i1 [[MEMSET_NOTZERO]], label %[[DO_MEMSET:.*]], label %[[END:.*]]
+; AIC:       [[DO_MEMSET]]:
+; AIC-NEXT:    call void @llvm.memset.p0.i64(ptr align 1 [[DST]], i8 [[VALUE]], i64 1, i1 false)
+; AIC-NEXT:    br label %[[END]]
+; AIC:       [[END]]:
+; AIC-NEXT:    ret void
+;
+; COMBINED-LABEL: define void @range_0_1(
+; COMBINED-SAME: ptr [[DST:%.*]], i8 [[VALUE:%.*]], i64 [[N:%.*]]) {
+; COMBINED:       [[LEN:%.*]] = and i64 [[N]], 1
+; COMBINED:       icmp {{eq|ne}} i64 [[LEN]], 0
+; COMBINED:       br i1
+; COMBINED:       store i8 [[VALUE]], ptr [[DST]], align 1
+; COMBINED-NOT:   call void @llvm.memset
+; COMBINED:       ret void
+entry:
+  %len = and i64 %n, 1
+  call void @llvm.memset.p0.i64(ptr align 1 %dst, i8 %value, i64 %len, i1 false)
+  ret void
+}
+
+define void @range_0_1_zext(ptr %dst, i8 %value, i32 %n) {
+; AIC-LABEL: define void @range_0_1_zext(
+; AIC-SAME: ptr [[DST:%.*]], i8 [[VALUE:%.*]], i32 [[N:%.*]]) {
+; AIC:       entry:
+; AIC-NEXT:    [[MASKED:%.*]] = and i32 [[N]], 1
+; AIC-NEXT:    [[LEN:%.*]] = zext i32 [[MASKED]] to i64
+; AIC-NEXT:    [[MEMSET_NOTZERO:%.*]] = icmp ne i64 [[LEN]], 0
+; AIC-NEXT:    br i1 [[MEMSET_NOTZERO]], label %[[DO_MEMSET:.*]], label %[[END:.*]]
+; AIC:       [[DO_MEMSET]]:
+; AIC-NEXT:    call void @llvm.memset.p0.i64(ptr align 1 [[DST]], i8 [[VALUE]], i64 1, i1 false)
+; AIC-NEXT:    br label %[[END]]
+; AIC:       [[END]]:
+; AIC-NEXT:    ret void
+;
+; COMBINED-LABEL: define void @range_0_1_zext(
+; COMBINED-SAME: ptr [[DST:%.*]], i8 [[VALUE:%.*]], i32 [[N:%.*]]) {
+; COMBINED:       [[MASKED:%.*]] = and i32 [[N]], 1
+; COMBINED:       icmp {{eq|ne}} i32 [[MASKED]], 0
+; COMBINED:       br i1
+; COMBINED:       store i8 [[VALUE]], ptr [[DST]], align 1
+; COMBINED-NOT:   call void @llvm.memset
+; COMBINED:       ret void
+entry:
+  %masked = and i32 %n, 1
+  %len = zext i32 %masked to i64
+  call void @llvm.memset.p0.i64(ptr align 1 %dst, i8 %value, i64 %len, i1 false)
+  ret void
+}
+
+define void @range_0_1_volatile(ptr %dst, i8 %value, i64 %n) {
+; AIC-LABEL: define void @range_0_1_volatile(
+; AIC-SAME: ptr [[DST:%.*]], i8 [[VALUE:%.*]], i64 [[N:%.*]]) {
+; AIC:       entry:
+; AIC-NEXT:    [[LEN:%.*]] = and i64 [[N]], 1
+; AIC-NEXT:    [[MEMSET_NOTZERO:%.*]] = icmp ne i64 [[LEN]], 0
+; AIC-NEXT:    br i1 [[MEMSET_NOTZERO]], label %[[DO_MEMSET:.*]], label %[[END:.*]]
+; AIC:       [[DO_MEMSET]]:
+; AIC-NEXT:    call void @llvm.memset.p0.i64(ptr align 1 [[DST]], i8 [[VALUE]], i64 1, i1 true)
+; AIC-NEXT:    br label %[[END]]
+; AIC:       [[END]]:
+; AIC-NEXT:    ret void
+;
+; COMBINED-LABEL: define void @range_0_1_volatile(
+; COMBINED-SAME: ptr [[DST:%.*]], i8 [[VALUE:%.*]], i64 [[N:%.*]]) {
+; COMBINED:       br i1
+; COMBINED:       store volatile i8 [[VALUE]], ptr [[DST]], align 1
+; COMBINED-NOT:   call void @llvm.memset
+; COMBINED:       ret void
+entry:
+  %len = and i64 %n, 1
+  call void @llvm.memset.p0.i64(ptr align 1 %dst, i8 %value, i64 %len, i1 true)
+  ret void
+}
+
+define void @range_0_2(ptr %dst, i8 %value, i64 %n) {
+; AIC-LABEL: define void @range_0_2(
+; AIC-SAME: ptr [[DST:%.*]], i8 [[VALUE:%.*]], i64 [[N:%.*]]) {
+; AIC:       entry:
+; AIC-NEXT:    [[LEN:%.*]] = urem i64 [[N]], 3
+; AIC-NEXT:    call void @llvm.memset.p0.i64(ptr align 1 [[DST]], i8 [[VALUE]], i64 [[LEN]], i1 false)
+; AIC-NEXT:    ret void
+;
+; COMBINED-LABEL: define void @range_0_2(
+; COMBINED-SAME: ptr [[DST:%.*]], i8 [[VALUE:%.*]], i64 [[N:%.*]]) {
+; COMBINED:       [[LEN:%.*]] = urem i64 [[N]], 3
+; COMBINED-NEXT:  call void @llvm.memset.p0.i64(ptr align 1 [[DST]], i8 [[VALUE]], i64 [[LEN]], i1 false)
+; COMBINED-NEXT:  ret void
+entry:
+  %len = urem i64 %n, 3
+  call void @llvm.memset.p0.i64(ptr align 1 %dst, i8 %value, i64 %len, i1 false)
+  ret void
+}



More information about the llvm-commits mailing list