[llvm] [InlineCost] Use store-to-load forwarding to resolve call arguments (PR #190607)
Jiří Filek via llvm-commits
llvm-commits at lists.llvm.org
Mon Apr 6 06:45:44 PDT 2026
https://github.com/fileho created https://github.com/llvm/llvm-project/pull/190607
Uses `FindAvailableLoadedValue` to resolve load instructions in call arguments to constants before inline cost analysis. This gives the inliner more precise cost estimate and option to inline functions which would not be inlined otherwise.
The `-O3` doesn't inline empty `std::set` and `std::map` because node deletion is recursive. The inliner doesn't know that `nullptr` is passed in as it is a `load` from a member.
This addresses both `libstdc++` and `libc++`:
- `libstdc++` - `FindAvailableLoadedValue` requires `MaxInstToScan=0`, because relevant store is 7 instructions away and `DefMaxInstsToScan = 6`. Benchmarking on large LLVM TUs showed no measurable compile-time difference between limit=6 and whole basic block
- `libc++` - uses `memset` to zero all members in ctor, this patch handles only `memset` to zero (the type mismatch case), which could be generalized but seems very rare
The store-to-load pattern is created and consumed within the same CGSCC inliner invocation: the ctor is inlined first (creating stores to the object), and then the dtor's inline cost is evaluated (seeing loads from the same object). No pass has an opportunity to simplify the IR in between.
The `-flto` build eliminates empty `std::set` because the IR is simplified enough in the regular optimization pass. However, when the code is not header-only in a different TU, `-flto` doesn't help.
The change is much more general than just `std::set` and `std::map`. I saw several impacts of it on LLVM codebase with `-O3`. Some function reduce in size due to better dead-code elimination. Some increase due to more aggressive inlining opportunities, and some are greatly simplified.
In my experiments I saw no measurable regression in compile times compiling many large LLVM TUs. I measured ~1% faster compilation due to following opt passes being faster. However, this needs more benchmarks.
Closes #183994
>From 41169e693445fbbd246f3a394a08eed8efc25292 Mon Sep 17 00:00:00 2001
From: Jiri Filek <jiri.filek at gmail.com>
Date: Mon, 6 Apr 2026 14:35:59 +0200
Subject: [PATCH] [InlineCost] Use store-to-load forwarding to resolve call
arguments
Use FindAvailableLoadedValue to resolve load instructions in call arguments to constants before inline cost analysis. This gives inliner more precise cost estimate
---
llvm/lib/Analysis/InlineCost.cpp | 24 +++++-
.../Transforms/Inline/inline_store_to_load.ll | 73 +++++++++++++++++++
2 files changed, 94 insertions(+), 3 deletions(-)
create mode 100644 llvm/test/Transforms/Inline/inline_store_to_load.ll
diff --git a/llvm/lib/Analysis/InlineCost.cpp b/llvm/lib/Analysis/InlineCost.cpp
index 06ff9cc80f638..34d781ae29338 100644
--- a/llvm/lib/Analysis/InlineCost.cpp
+++ b/llvm/lib/Analysis/InlineCost.cpp
@@ -23,6 +23,7 @@
#include "llvm/Analysis/DomConditionCache.h"
#include "llvm/Analysis/EphemeralValuesCache.h"
#include "llvm/Analysis/InstructionSimplify.h"
+#include "llvm/Analysis/Loads.h"
#include "llvm/Analysis/LoopInfo.h"
#include "llvm/Analysis/MemoryBuiltins.h"
#include "llvm/Analysis/OptimizationRemarkEmitter.h"
@@ -2931,11 +2932,28 @@ InlineResult CallAnalyzer::analyze() {
auto CAI = CandidateCall.arg_begin();
for (Argument &FAI : F.args()) {
assert(CAI != CandidateCall.arg_end());
- SimplifiedValues[&FAI] = *CAI;
- if (isa<Constant>(*CAI))
+ Value *CallerArg = *CAI;
+
+ // Simple store-to-load forwarding of arguments in the caller.
+ // This can greatly reduce the function cost via early returns.
+ if (auto *LI = dyn_cast<LoadInst>(CallerArg)) {
+ BasicBlock::iterator ScanFrom = LI->getIterator();
+ if (Value *Loaded =
+ FindAvailableLoadedValue(LI, LI->getParent(), ScanFrom, 0)) {
+ if (auto *C = dyn_cast<Constant>(Loaded)) {
+ if (C->getType() == LI->getType())
+ CallerArg = C;
+ else if (C->isNullValue())
+ CallerArg = Constant::getNullValue(LI->getType());
+ }
+ }
+ }
+
+ SimplifiedValues[&FAI] = CallerArg;
+ if (isa<Constant>(CallerArg))
++NumConstantArgs;
- Value *PtrArg = *CAI;
+ Value *PtrArg = CallerArg;
if (ConstantInt *C = stripAndComputeInBoundsConstantOffsets(PtrArg)) {
ConstantOffsetPtrs[&FAI] = std::make_pair(PtrArg, C->getValue());
diff --git a/llvm/test/Transforms/Inline/inline_store_to_load.ll b/llvm/test/Transforms/Inline/inline_store_to_load.ll
new file mode 100644
index 0000000000000..919dc9aafa403
--- /dev/null
+++ b/llvm/test/Transforms/Inline/inline_store_to_load.ll
@@ -0,0 +1,73 @@
+; RUN: opt < %s -passes=inline -inline-threshold=20 -S | FileCheck %s
+
+; Test that the inliner can use store-to-load forwarding to resolve call
+; arguments to constants. We use -inline-threshold=20 so that @callee is
+; only inlined when the constant argument enables dead-branch elimination.
+
+target datalayout = "p:64:64"
+
+; Two paths: mode==0 is trivial (ret), otherwise too expensive to inline.
+define i32 @callee(i32 %mode, ptr %p) {
+entry:
+ %cmp = icmp eq i32 %mode, 0
+ br i1 %cmp, label %fast, label %slow
+fast:
+ %v = load i32, ptr %p
+ ret i32 %v
+slow:
+ %a1 = load volatile i32, ptr %p
+ %a2 = load volatile i32, ptr %p
+ %x1 = add i32 %a1, %a2
+ %a3 = load volatile i32, ptr %p
+ %x2 = add i32 %x1, %a3
+ %a4 = load volatile i32, ptr %p
+ %x3 = add i32 %x2, %a4
+ %a5 = load volatile i32, ptr %p
+ %x4 = add i32 %x3, %a5
+ ret i32 %x4
+}
+
+; Trivial when called with null, otherwise too expensive to inline.
+define void @recursive_callee(ptr %x) {
+entry:
+ %cmp = icmp eq ptr %x, null
+ br i1 %cmp, label %done, label %recurse
+recurse:
+ %next = load ptr, ptr %x
+ call void @recursive_callee(ptr %next)
+ %v = load volatile i32, ptr %x
+ br label %done
+done:
+ ret void
+}
+
+; Store-to-load forwarding resolves %mode to 0, making only the fast path
+; reachable and the callee cheap enough to inline.
+; CHECK-LABEL: define i32 @caller_store_forward(
+; CHECK: %cmp.i = icmp eq i32 %mode, 0
+; CHECK: callee.exit:
+; CHECK-NEXT: %{{.*}} = phi i32
+; CHECK-NEXT: ret i32
+define i32 @caller_store_forward() {
+entry:
+ %p = alloca i32
+ store i32 0, ptr %p
+ %mode = load i32, ptr %p
+ %r = call i32 @callee(i32 %mode, ptr %p)
+ ret i32 %r
+}
+
+; Memset-to-load forwarding converts the zero-filled integer to a null
+; pointer, making the null check take the early exit.
+; CHECK-LABEL: define void @caller_memset_ptr(
+; CHECK: %cmp.i = icmp eq ptr %x, null
+; CHECK: recursive_callee.exit:
+; CHECK-NEXT: ret void
+define void @caller_memset_ptr() {
+entry:
+ %p = alloca ptr
+ call void @llvm.memset.p0.i64(ptr %p, i8 0, i64 8, i1 false)
+ %x = load ptr, ptr %p
+ call void @recursive_callee(ptr %x)
+ ret void
+}
More information about the llvm-commits
mailing list