[llvm] 7bdca28 - [InlineCost] Never inline functions with incompatible target features (#205113)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Jul 6 00:28:53 PDT 2026
Author: Nikita Popov
Date: 2026-07-06T07:28:48Z
New Revision: 7bdca287b102ee935e982613dba2264cbc7a32ad
URL: https://github.com/llvm/llvm-project/commit/7bdca287b102ee935e982613dba2264cbc7a32ad
DIFF: https://github.com/llvm/llvm-project/commit/7bdca287b102ee935e982613dba2264cbc7a32ad.diff
LOG: [InlineCost] Never inline functions with incompatible target features (#205113)
If inlining is unsound due to incompatible target feature attributes, we
should not inline the call even if alwaysinline is set. This will likely
result in a crash during instruction selection.
We tried this previously in
https://github.com/llvm/llvm-project/commit/d6f994acb3d545b80161e24ab742c9c69d4bbf33,
but the change had to be reverted because the quality of our
areInlineCompatible() hooks was very bad at the time, which resulted in
inlining not happening in many cases where it was safe.
I think we're in a much better position now. Most notably, we now have a
default areInlineCompatible() implementation that actually does
something sensible (https://github.com/llvm/llvm-project/pull/117493),
inlining compatibility for target features is now specified in TableGen
(https://github.com/llvm/llvm-project/pull/205348). Various
target-specific issues have been fixed as well, e.g. ARM's overly strict
feature whitelist (https://github.com/llvm/llvm-project/pull/205763) and
X86's overly conservative ABI compatibility checks
(https://github.com/llvm/llvm-project/pull/205106).
Added:
llvm/test/Transforms/Inline/X86/always-inline-features.ll
Modified:
llvm/docs/ReleaseNotes.md
llvm/lib/Analysis/InlineCost.cpp
Removed:
################################################################################
diff --git a/llvm/docs/ReleaseNotes.md b/llvm/docs/ReleaseNotes.md
index e42a959ab3101..11d3ebc80fca6 100644
--- a/llvm/docs/ReleaseNotes.md
+++ b/llvm/docs/ReleaseNotes.md
@@ -111,6 +111,9 @@ Makes programs 10x faster by doing Special New Thing.
corresponding metadata) now apply only at the point of definition, instead of
for the execution of the function (for arguments) or forever (for returns).
+* `alwaysinline` no longer bypasses inlining compatibility checks based on
+ target features. Inlining will only be performed if it is safe to do so.
+
### Changes to LLVM infrastructure
* Removed ``Constant::isZeroValue``. It was functionally identical to
diff --git a/llvm/lib/Analysis/InlineCost.cpp b/llvm/lib/Analysis/InlineCost.cpp
index a2e321884e708..c3823931a7bcb 100644
--- a/llvm/lib/Analysis/InlineCost.cpp
+++ b/llvm/lib/Analysis/InlineCost.cpp
@@ -3081,16 +3081,14 @@ LLVM_DUMP_METHOD void InlineCostCallAnalyzer::dump() { print(dbgs()); }
/// Test that there are no attribute conflicts between Caller and Callee
/// that prevent inlining.
static bool functionsHaveCompatibleAttributes(
- Function *Caller, Function *Callee, TargetTransformInfo &TTI,
+ Function *Caller, Function *Callee,
function_ref<const TargetLibraryInfo &(Function &)> &GetTLI) {
// Note that CalleeTLI must be a copy not a reference. The legacy pass manager
// caches the most recently created TLI in the TargetLibraryInfoWrapperPass
// object, and always returns the same object (which is overwritten on each
// GetTLI call). Therefore we copy the first result.
auto CalleeTLI = GetTLI(*Callee);
- return (IgnoreTTIInlineCompatible ||
- TTI.areInlineCompatible(Caller, Callee)) &&
- GetTLI(*Caller).areInlineCompatible(CalleeTLI,
+ return GetTLI(*Caller).areInlineCompatible(CalleeTLI,
InlineCallerSupersetNoBuiltin) &&
AttributeFuncs::areInlineCompatible(*Caller, *Callee);
}
@@ -3213,6 +3211,13 @@ std::optional<InlineResult> llvm::getAttributeBasedInliningDecision(
" address space");
}
+ // Inlining into a function with less target features is unsound, so enforce
+ // this even if alwaysinline is used.
+ Function *Caller = Call.getCaller();
+ if (!IgnoreTTIInlineCompatible &&
+ !CalleeTTI.areInlineCompatible(Caller, Callee))
+ return InlineResult::failure("conflicting target features");
+
// Calls to functions with always-inline attributes should be inlined
// whenever possible.
if (Call.hasFnAttr(Attribute::AlwaysInline)) {
@@ -3227,8 +3232,7 @@ std::optional<InlineResult> llvm::getAttributeBasedInliningDecision(
// Never inline functions with conflicting attributes (unless callee has
// always-inline attribute).
- Function *Caller = Call.getCaller();
- if (!functionsHaveCompatibleAttributes(Caller, Callee, CalleeTTI, GetTLI))
+ if (!functionsHaveCompatibleAttributes(Caller, Callee, GetTLI))
return InlineResult::failure("conflicting attributes");
// Flatten: inline all viable calls from flatten functions regardless of cost.
diff --git a/llvm/test/Transforms/Inline/X86/always-inline-features.ll b/llvm/test/Transforms/Inline/X86/always-inline-features.ll
new file mode 100644
index 0000000000000..500356471c9ae
--- /dev/null
+++ b/llvm/test/Transforms/Inline/X86/always-inline-features.ll
@@ -0,0 +1,32 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=inline < %s | FileCheck %s
+
+target triple = "x86_64-unknown-linux-gnu"
+
+; Even with alwaysinline, the callee cannot be inlined, because the target
+; features are incompatible. Inlining it would result in a selection failure
+; crash.
+
+define <16 x float> @caller(<16 x float> %a, <16 x float> %b) {
+; CHECK-LABEL: define <16 x float> @caller(
+; CHECK-SAME: <16 x float> [[A:%.*]], <16 x float> [[B:%.*]]) {
+; CHECK-NEXT: [[RES:%.*]] = call <16 x float> @callee(<16 x float> [[A]], <16 x float> [[B]]) #[[ATTR2:[0-9]+]]
+; CHECK-NEXT: ret <16 x float> [[RES]]
+;
+ %res = call <16 x float> @callee(<16 x float> %a, <16 x float> %b) alwaysinline
+ ret <16 x float> %res
+}
+
+define <16 x float> @callee(<16 x float> %a, <16 x float> %b) #0 {
+; CHECK-LABEL: define <16 x float> @callee(
+; CHECK-SAME: <16 x float> [[A:%.*]], <16 x float> [[B:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[RES:%.*]] = call <16 x float> @llvm.x86.avx512.min.ps.512(<16 x float> [[A]], <16 x float> [[B]], i32 0)
+; CHECK-NEXT: ret <16 x float> [[RES]]
+;
+ %res = call <16 x float> @llvm.x86.avx512.min.ps.512(<16 x float> %a, <16 x float> %b, i32 0)
+ ret <16 x float> %res
+}
+
+declare <16 x float> @llvm.x86.avx512.min.ps.512(<16 x float>, <16 x float>, i32 immarg)
+
+attributes #0 = { "target-features"="+avx512bw,+avx512dq,+avx512f" }
More information about the llvm-commits
mailing list