[llvm] [TTI] Provide conservative costs for @llvm.speculative.load. (PR #180036)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 23 02:11:44 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/180036
>From 60fa4678bfef2240ae6979e7d8bcbcccf25a5903 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sat, 31 Jan 2026 14:46:20 +0000
Subject: [PATCH 1/2] [TTI] Provide conservative costs for
@llvm.speculative.load.
Add TTI support for @llvm.speculative.load, defaulting to Invalid if not
implemented by the target.
Provide implementation for AArch64, which checks if the loaded type is
<= 16 bytes.
---
.../llvm/Analysis/TargetTransformInfoImpl.h | 2 +
llvm/include/llvm/CodeGen/BasicTTIImpl.h | 3 +
.../AArch64/AArch64TargetTransformInfo.cpp | 19 ++++++
.../CostModel/AArch64/speculative-load.ll | 65 +++++++++++--------
.../CostModel/X86/speculative-load.ll | 18 ++---
5 files changed, 71 insertions(+), 36 deletions(-)
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 84cb3a6e664b9e..9805d88ec269f2 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -921,6 +921,8 @@ class LLVM_ABI TargetTransformInfoImplBase {
switch (ICA.getID()) {
default:
break;
+ case Intrinsic::speculative_load:
+ return InstructionCost::getInvalid();
case Intrinsic::allow_runtime_check:
case Intrinsic::allow_ubsan_check:
case Intrinsic::annotation:
diff --git a/llvm/include/llvm/CodeGen/BasicTTIImpl.h b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
index cb01ef40282170..25276c5e95e02f 100644
--- a/llvm/include/llvm/CodeGen/BasicTTIImpl.h
+++ b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
@@ -2049,6 +2049,9 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
// The cost of materialising a constant integer vector.
return TargetTransformInfo::TCC_Basic;
}
+ case Intrinsic::speculative_load:
+ // Delegate to base; targets must opt-in with a valid cost.
+ return BaseT::getIntrinsicInstrCost(ICA, CostKind);
case Intrinsic::vector_extract: {
// FIXME: Handle case where a scalable vector is extracted from a scalable
// vector
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 8e0b88dbad0eb4..ac6776af86403e 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -619,6 +619,25 @@ AArch64TTIImpl::getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
return InstructionCost::getInvalid();
switch (ICA.getID()) {
+ case Intrinsic::speculative_load: {
+ // Speculative loads are only valid for types <= 16 bytes due to MTE
+ // (Memory Tagging Extension) using 16-byte tag granules. Loads larger
+ // than 16 bytes could cross a tag granule boundary.
+ auto LT = getTypeLegalizationCost(RetTy);
+ if (!LT.first.isValid())
+ return InstructionCost::getInvalid();
+ // For scalable vectors, check that we use a single register (which means
+ // <= 16 bytes at minimum vscale). For fixed types, compute the actual size.
+ if (isa<ScalableVectorType>(RetTy)) {
+ if (LT.first.getValue() != 1)
+ return InstructionCost::getInvalid();
+ } else {
+ if (LT.first.getValue() * LT.second.getStoreSize() > 16)
+ return InstructionCost::getInvalid();
+ }
+ // Return cost of a regular load.
+ return getMemoryOpCost(Instruction::Load, RetTy, Align(1), 0, CostKind);
+ }
case Intrinsic::experimental_vector_histogram_add: {
InstructionCost HistCost = getHistogramCost(ST, ICA);
// If the cost isn't valid, we may still be able to scalarize
diff --git a/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll b/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
index 4b7b906ce3feea..397b63d765726d 100644
--- a/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
@@ -1,6 +1,6 @@
; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py
-; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 < %s | FileCheck %s --check-prefixes=COMMON
-; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 -mattr=+sve < %s | FileCheck %s --check-prefixes=COMMON
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 < %s | FileCheck %s --check-prefixes=COMMON,NEON
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 -mattr=+sve < %s | FileCheck %s --check-prefixes=COMMON,SVE
define void @speculative_load_cost_fixed(ptr %p) {
; Scalar types - all valid (<= 16 bytes)
@@ -9,22 +9,22 @@ define void @speculative_load_cost_fixed(ptr %p) {
; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %5 = call b128 (ptr, i1, ...) @llvm.speculative.load.b128.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %6 = call <2 x i32> (ptr, i1, ...) @llvm.speculative.load.v2i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %7 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %8 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %9 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %10 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %11 = call <8 x i8> (ptr, i1, ...) @llvm.speculative.load.v8i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 48 for instruction: %12 = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %13 = call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %14 = call <8 x i16> (ptr, i1, ...) @llvm.speculative.load.v8i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %15 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %16 = call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 96 for instruction: %17 = call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 48 for instruction: %18 = call <16 x i16> (ptr, i1, ...) @llvm.speculative.load.v16i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %19 = call <8 x float> (ptr, i1, ...) @llvm.speculative.load.v8f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %20 = call <4 x double> (ptr, i1, ...) @llvm.speculative.load.v4f64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %5 = call b128 (ptr, i1, ...) @llvm.speculative.load.b128.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %6 = call <2 x i32> (ptr, i1, ...) @llvm.speculative.load.v2i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %7 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %8 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %9 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %10 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %11 = call <8 x i8> (ptr, i1, ...) @llvm.speculative.load.v8i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %12 = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %13 = call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %14 = call <8 x i16> (ptr, i1, ...) @llvm.speculative.load.v8i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Invalid cost for instruction: %15 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Invalid cost for instruction: %16 = call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Invalid cost for instruction: %17 = call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Invalid cost for instruction: %18 = call <16 x i16> (ptr, i1, ...) @llvm.speculative.load.v16i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Invalid cost for instruction: %19 = call <8 x float> (ptr, i1, ...) @llvm.speculative.load.v8f32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Invalid cost for instruction: %20 = call <4 x double> (ptr, i1, ...) @llvm.speculative.load.v4f64.p0(ptr %p, i1 false, i64 0)
; COMMON-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
@@ -55,16 +55,27 @@ define void @speculative_load_cost_fixed(ptr %p) {
}
define void @speculative_load_cost_scalable(ptr %p) {
-; COMMON-LABEL: 'speculative_load_cost_scalable'
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %1 = call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %2 = call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %3 = call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+; NEON-LABEL: 'speculative_load_cost_scalable'
+; NEON-NEXT: Cost Model: Invalid cost for instruction: %1 = call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT: Cost Model: Invalid cost for instruction: %2 = call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT: Cost Model: Invalid cost for instruction: %3 = call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT: Cost Model: Invalid cost for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT: Cost Model: Invalid cost for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT: Cost Model: Invalid cost for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT: Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; SVE-LABEL: 'speculative_load_cost_scalable'
+; SVE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %1 = call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %2 = call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %3 = call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT: Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
+ ; Scalable vector types - invalid without SVE, valid with SVE if <= 16 bytes
call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
diff --git a/llvm/test/Analysis/CostModel/X86/speculative-load.ll b/llvm/test/Analysis/CostModel/X86/speculative-load.ll
index 81b3b56eeb10c3..d5ed066dd83ceb 100644
--- a/llvm/test/Analysis/CostModel/X86/speculative-load.ll
+++ b/llvm/test/Analysis/CostModel/X86/speculative-load.ll
@@ -3,15 +3,15 @@
define void @speculative_load_cost(ptr %p) {
; CHECK-LABEL: 'speculative_load_cost'
-; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %5 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Found an estimated cost of 22 for instruction: %6 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Found an estimated cost of 5 for instruction: %7 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Found an estimated cost of 7 for instruction: %8 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %9 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Invalid cost for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Invalid cost for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Invalid cost for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Invalid cost for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Invalid cost for instruction: %5 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Invalid cost for instruction: %6 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Invalid cost for instruction: %7 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Invalid cost for instruction: %8 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Invalid cost for instruction: %9 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
; CHECK-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
>From c777075e7207adbd6123565c325f291db088d323 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 23 Sep 2026 10:01:46 +0100
Subject: [PATCH 2/2] !fixup add isLegalSpeculativeLoad
---
.../llvm/Analysis/TargetTransformInfo.h | 6 +++
.../llvm/Analysis/TargetTransformInfoImpl.h | 8 +++-
llvm/include/llvm/CodeGen/BasicTTIImpl.h | 14 +++++--
llvm/lib/Analysis/TargetTransformInfo.cpp | 5 +++
.../AArch64/AArch64TargetTransformInfo.cpp | 39 ++++++++++---------
.../AArch64/AArch64TargetTransformInfo.h | 3 ++
.../CostModel/AArch64/speculative-load.ll | 21 +++++-----
.../CostModel/X86/speculative-load.ll | 18 ++++-----
8 files changed, 70 insertions(+), 44 deletions(-)
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index e30cbc61a5420b..a683a40c14e76d 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -933,6 +933,12 @@ class TargetTransformInfo {
isLegalMaskedLoad(Type *DataType, Align Alignment, unsigned AddressSpace,
MaskKind MaskKind = VariableOrConstantMask) const;
+ /// Return true if the target supports speculatively loading \p DataType from
+ /// address space \p AddressSpace, i.e. @llvm.can.load.speculatively can
+ /// return true for the store size of \p DataType.
+ LLVM_ABI bool isLegalSpeculativeLoad(Type *DataType,
+ unsigned AddressSpace) const;
+
/// Return true if the target supports nontemporal store.
LLVM_ABI bool isLegalNTStore(Type *DataType, Align Alignment) const;
/// Return true if the target supports nontemporal load.
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index f487ec556a0b98..71dead1c3c5cd5 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -368,6 +368,11 @@ class LLVM_ABI TargetTransformInfoImplBase {
return false;
}
+ virtual bool isLegalSpeculativeLoad(Type *DataType,
+ unsigned AddressSpace) const {
+ return false;
+ }
+
virtual bool isLegalNTStore(Type *DataType, Align Alignment) const {
// By default, assume nontemporal memory stores are available for stores
// that are aligned and have a size that is a power of 2.
@@ -926,8 +931,6 @@ class LLVM_ABI TargetTransformInfoImplBase {
switch (ICA.getID()) {
default:
break;
- case Intrinsic::speculative_load:
- return InstructionCost::getInvalid();
case Intrinsic::allow_runtime_check:
case Intrinsic::allow_ubsan_check:
case Intrinsic::annotation:
@@ -993,6 +996,7 @@ class LLVM_ABI TargetTransformInfoImplBase {
case Intrinsic::vp_gather:
case Intrinsic::masked_compressstore:
case Intrinsic::masked_expandload:
+ case Intrinsic::speculative_load:
return 1;
}
return InstructionCost::getInvalid();
diff --git a/llvm/include/llvm/CodeGen/BasicTTIImpl.h b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
index 4751a5c3713583..bb7aa02bef3aba 100644
--- a/llvm/include/llvm/CodeGen/BasicTTIImpl.h
+++ b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
@@ -2051,9 +2051,6 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
// The cost of materialising a constant integer vector.
return TargetTransformInfo::TCC_Basic;
}
- case Intrinsic::speculative_load:
- // Delegate to base; targets must opt-in with a valid cost.
- return BaseT::getIntrinsicInstrCost(ICA, CostKind);
case Intrinsic::vector_extract: {
// FIXME: Handle case where a scalable vector is extracted from a scalable
// vector
@@ -2523,6 +2520,13 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
return thisT()->getMemIntrinsicInstrCost(
MemIntrinsicCostAttributes(IID, Ty, TyAlign, 0), CostKind);
}
+ case Intrinsic::speculative_load: {
+ const IntrinsicInst *I = ICA.getInst();
+ Align Alignment = I ? I->getParamAlign(0).valueOrOne() : Align(1);
+ unsigned AS = Tys[0]->getPointerAddressSpace();
+ return thisT()->getMemIntrinsicInstrCost(
+ MemIntrinsicCostAttributes(IID, RetTy, Alignment, AS), CostKind);
+ }
case Intrinsic::experimental_vp_strided_store: {
auto *Ty = cast<VectorType>(ICA.getArgTypes()[0]);
Align Alignment = thisT()->DL.getABITypeAlign(Ty->getElementType());
@@ -3279,6 +3283,10 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
}
case Intrinsic::vp_load_ff:
return InstructionCost::getInvalid();
+ case Intrinsic::speculative_load:
+ // Speculative loads are lowered to regular loads of the full type.
+ return thisT()->getMemoryOpCost(Instruction::Load, DataTy, Alignment,
+ MICA.getAddressSpace(), CostKind);
default:
llvm_unreachable("unexpected intrinsic");
}
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 4c2cac9c440a08..1223fba8d4fea1 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -494,6 +494,11 @@ bool TargetTransformInfo::isLegalMaskedLoad(Type *DataType, Align Alignment,
MaskKind);
}
+bool TargetTransformInfo::isLegalSpeculativeLoad(Type *DataType,
+ unsigned AddressSpace) const {
+ return TTIImpl->isLegalSpeculativeLoad(DataType, AddressSpace);
+}
+
bool TargetTransformInfo::isLegalNTStore(Type *DataType,
Align Alignment) const {
return TTIImpl->isLegalNTStore(DataType, Alignment);
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 3944b100db7ed0..8b77caf6d40cb8 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -619,25 +619,6 @@ AArch64TTIImpl::getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
return InstructionCost::getInvalid();
switch (ICA.getID()) {
- case Intrinsic::speculative_load: {
- // Speculative loads are only valid for types <= 16 bytes due to MTE
- // (Memory Tagging Extension) using 16-byte tag granules. Loads larger
- // than 16 bytes could cross a tag granule boundary.
- auto LT = getTypeLegalizationCost(RetTy);
- if (!LT.first.isValid())
- return InstructionCost::getInvalid();
- // For scalable vectors, check that we use a single register (which means
- // <= 16 bytes at minimum vscale). For fixed types, compute the actual size.
- if (isa<ScalableVectorType>(RetTy)) {
- if (LT.first.getValue() != 1)
- return InstructionCost::getInvalid();
- } else {
- if (LT.first.getValue() * LT.second.getStoreSize() > 16)
- return InstructionCost::getInvalid();
- }
- // Return cost of a regular load.
- return getMemoryOpCost(Instruction::Load, RetTy, Align(1), 0, CostKind);
- }
case Intrinsic::experimental_vector_histogram_add: {
InstructionCost HistCost = getHistogramCost(ST, ICA);
// If the cost isn't valid, we may still be able to scalarize
@@ -5939,6 +5920,26 @@ bool AArch64TTIImpl::isLegalMaskedExpandLoad(Type *DataTy,
(ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2());
}
+bool AArch64TTIImpl::isLegalSpeculativeLoad(Type *DataType,
+ unsigned AddressSpace) const {
+ // Matches AArch64TargetLowering::emitCanLoadSpeculatively: only address
+ // space 0 and sizes up to the 16-byte MTE tag granule are supported.
+ if (AddressSpace != 0)
+ return false;
+ TypeSize Size = DL.getTypeStoreSize(DataType);
+ uint64_t MinSize = Size.getKnownMinValue();
+ // Scalable types are at least the minimum vscale times their known minimum
+ // size.
+ if (Size.isScalable()) {
+ if (!ST->isSVEorStreamingSVEAvailable())
+ return false;
+ MinSize *=
+ std::max(ST->getMinSVEVectorSizeInBits(), AArch64::SVEBitsPerBlock) /
+ AArch64::SVEBitsPerBlock;
+ }
+ return MinSize <= 16;
+}
+
unsigned
AArch64TTIImpl::getMaxInterleaveFactor(ElementCount VF,
bool HasUnorderedReductions) const {
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index c1d8fc787ef623..323dc5424b9221 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -279,6 +279,9 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
bool isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const override;
+ bool isLegalSpeculativeLoad(Type *DataType,
+ unsigned AddressSpace) const override;
+
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
TTI::UnrollingPreferences &UP,
OptimizationRemarkEmitter *ORE) const override;
diff --git a/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll b/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
index 397b63d765726d..db19fcf2c58358 100644
--- a/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
@@ -3,7 +3,7 @@
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 -mattr=+sve < %s | FileCheck %s --check-prefixes=COMMON,SVE
define void @speculative_load_cost_fixed(ptr %p) {
- ; Scalar types - all valid (<= 16 bytes)
+ ; Scalar types (<= 16 bytes)
; COMMON-LABEL: 'speculative_load_cost_fixed'
; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
@@ -19,12 +19,12 @@ define void @speculative_load_cost_fixed(ptr %p) {
; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %12 = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr %p, i1 false, i64 0)
; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %13 = call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr %p, i1 false, i64 0)
; COMMON-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %14 = call <8 x i16> (ptr, i1, ...) @llvm.speculative.load.v8i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %15 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %16 = call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %17 = call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %18 = call <16 x i16> (ptr, i1, ...) @llvm.speculative.load.v16i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %19 = call <8 x float> (ptr, i1, ...) @llvm.speculative.load.v8f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT: Cost Model: Invalid cost for instruction: %20 = call <4 x double> (ptr, i1, ...) @llvm.speculative.load.v4f64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %15 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %16 = call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %17 = call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %18 = call <16 x i16> (ptr, i1, ...) @llvm.speculative.load.v16i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %19 = call <8 x float> (ptr, i1, ...) @llvm.speculative.load.v8f32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %20 = call <4 x double> (ptr, i1, ...) @llvm.speculative.load.v4f64.p0(ptr %p, i1 false, i64 0)
; COMMON-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
@@ -33,7 +33,7 @@ define void @speculative_load_cost_fixed(ptr %p) {
call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
call b128 (ptr, i1, ...) @llvm.speculative.load.b128.p0(ptr %p, i1 false, i64 0)
- ; Vector types <= 16 bytes - valid
+ ; Vector types <= 16 bytes
call <2 x i32> (ptr, i1, ...) @llvm.speculative.load.v2i32.p0(ptr %p, i1 false, i64 0)
call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
@@ -44,7 +44,7 @@ define void @speculative_load_cost_fixed(ptr %p) {
call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr %p, i1 false, i64 0)
call <8 x i16> (ptr, i1, ...) @llvm.speculative.load.v8i16.p0(ptr %p, i1 false, i64 0)
- ; Vector types > 16 bytes - invalid
+ ; Vector types > 16 bytes
call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
@@ -72,10 +72,9 @@ define void @speculative_load_cost_scalable(ptr %p) {
; SVE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
; SVE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
; SVE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
-; SVE-NEXT: Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
; SVE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
- ; Scalable vector types - invalid without SVE, valid with SVE if <= 16 bytes
call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
diff --git a/llvm/test/Analysis/CostModel/X86/speculative-load.ll b/llvm/test/Analysis/CostModel/X86/speculative-load.ll
index d5ed066dd83ceb..2dba86ceabff97 100644
--- a/llvm/test/Analysis/CostModel/X86/speculative-load.ll
+++ b/llvm/test/Analysis/CostModel/X86/speculative-load.ll
@@ -3,15 +3,15 @@
define void @speculative_load_cost(ptr %p) {
; CHECK-LABEL: 'speculative_load_cost'
-; CHECK-NEXT: Cost Model: Invalid cost for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Invalid cost for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Invalid cost for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Invalid cost for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Invalid cost for instruction: %5 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Invalid cost for instruction: %6 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Invalid cost for instruction: %7 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Invalid cost for instruction: %8 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT: Cost Model: Invalid cost for instruction: %9 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %5 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %6 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %7 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %8 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %9 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
; CHECK-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
More information about the llvm-commits
mailing list