[llvm] [TTI] Provide conservative costs for @llvm.speculative.load. (PR #180036)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 23 02:11:44 PDT 2026


https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/180036

>From 60fa4678bfef2240ae6979e7d8bcbcccf25a5903 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sat, 31 Jan 2026 14:46:20 +0000
Subject: [PATCH 1/2] [TTI] Provide conservative costs for
 @llvm.speculative.load.

Add TTI support for @llvm.speculative.load, defaulting to Invalid if not
implemented by the target.

Provide implementation for AArch64, which checks if the loaded type is
<= 16 bytes.
---
 .../llvm/Analysis/TargetTransformInfoImpl.h   |  2 +
 llvm/include/llvm/CodeGen/BasicTTIImpl.h      |  3 +
 .../AArch64/AArch64TargetTransformInfo.cpp    | 19 ++++++
 .../CostModel/AArch64/speculative-load.ll     | 65 +++++++++++--------
 .../CostModel/X86/speculative-load.ll         | 18 ++---
 5 files changed, 71 insertions(+), 36 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 84cb3a6e664b9e..9805d88ec269f2 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -921,6 +921,8 @@ class LLVM_ABI TargetTransformInfoImplBase {
     switch (ICA.getID()) {
     default:
       break;
+    case Intrinsic::speculative_load:
+      return InstructionCost::getInvalid();
     case Intrinsic::allow_runtime_check:
     case Intrinsic::allow_ubsan_check:
     case Intrinsic::annotation:
diff --git a/llvm/include/llvm/CodeGen/BasicTTIImpl.h b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
index cb01ef40282170..25276c5e95e02f 100644
--- a/llvm/include/llvm/CodeGen/BasicTTIImpl.h
+++ b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
@@ -2049,6 +2049,9 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
       // The cost of materialising a constant integer vector.
       return TargetTransformInfo::TCC_Basic;
     }
+    case Intrinsic::speculative_load:
+      // Delegate to base; targets must opt-in with a valid cost.
+      return BaseT::getIntrinsicInstrCost(ICA, CostKind);
     case Intrinsic::vector_extract: {
       // FIXME: Handle case where a scalable vector is extracted from a scalable
       // vector
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 8e0b88dbad0eb4..ac6776af86403e 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -619,6 +619,25 @@ AArch64TTIImpl::getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
       return InstructionCost::getInvalid();
 
   switch (ICA.getID()) {
+  case Intrinsic::speculative_load: {
+    // Speculative loads are only valid for types <= 16 bytes due to MTE
+    // (Memory Tagging Extension) using 16-byte tag granules. Loads larger
+    // than 16 bytes could cross a tag granule boundary.
+    auto LT = getTypeLegalizationCost(RetTy);
+    if (!LT.first.isValid())
+      return InstructionCost::getInvalid();
+    // For scalable vectors, check that we use a single register (which means
+    // <= 16 bytes at minimum vscale). For fixed types, compute the actual size.
+    if (isa<ScalableVectorType>(RetTy)) {
+      if (LT.first.getValue() != 1)
+        return InstructionCost::getInvalid();
+    } else {
+      if (LT.first.getValue() * LT.second.getStoreSize() > 16)
+        return InstructionCost::getInvalid();
+    }
+    // Return cost of a regular load.
+    return getMemoryOpCost(Instruction::Load, RetTy, Align(1), 0, CostKind);
+  }
   case Intrinsic::experimental_vector_histogram_add: {
     InstructionCost HistCost = getHistogramCost(ST, ICA);
     // If the cost isn't valid, we may still be able to scalarize
diff --git a/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll b/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
index 4b7b906ce3feea..397b63d765726d 100644
--- a/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py
-; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 < %s | FileCheck %s --check-prefixes=COMMON
-; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 -mattr=+sve < %s | FileCheck %s --check-prefixes=COMMON
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 < %s | FileCheck %s --check-prefixes=COMMON,NEON
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 -mattr=+sve < %s | FileCheck %s --check-prefixes=COMMON,SVE
 
 define void @speculative_load_cost_fixed(ptr %p) {
   ; Scalar types - all valid (<= 16 bytes)
@@ -9,22 +9,22 @@ define void @speculative_load_cost_fixed(ptr %p) {
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %5 = call b128 (ptr, i1, ...) @llvm.speculative.load.b128.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 6 for instruction: %6 = call <2 x i32> (ptr, i1, ...) @llvm.speculative.load.v2i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 12 for instruction: %7 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 6 for instruction: %8 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 10 for instruction: %9 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %10 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 24 for instruction: %11 = call <8 x i8> (ptr, i1, ...) @llvm.speculative.load.v8i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 48 for instruction: %12 = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 12 for instruction: %13 = call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 24 for instruction: %14 = call <8 x i16> (ptr, i1, ...) @llvm.speculative.load.v8i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 24 for instruction: %15 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 12 for instruction: %16 = call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 96 for instruction: %17 = call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 48 for instruction: %18 = call <16 x i16> (ptr, i1, ...) @llvm.speculative.load.v16i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 20 for instruction: %19 = call <8 x float> (ptr, i1, ...) @llvm.speculative.load.v8f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %20 = call <4 x double> (ptr, i1, ...) @llvm.speculative.load.v4f64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %5 = call b128 (ptr, i1, ...) @llvm.speculative.load.b128.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %6 = call <2 x i32> (ptr, i1, ...) @llvm.speculative.load.v2i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %7 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %8 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %9 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %10 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %11 = call <8 x i8> (ptr, i1, ...) @llvm.speculative.load.v8i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %12 = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %13 = call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %14 = call <8 x i16> (ptr, i1, ...) @llvm.speculative.load.v8i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %15 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %16 = call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %17 = call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %18 = call <16 x i16> (ptr, i1, ...) @llvm.speculative.load.v16i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %19 = call <8 x float> (ptr, i1, ...) @llvm.speculative.load.v8f32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %20 = call <4 x double> (ptr, i1, ...) @llvm.speculative.load.v4f64.p0(ptr %p, i1 false, i64 0)
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
 ;
   call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
@@ -55,16 +55,27 @@ define void @speculative_load_cost_fixed(ptr %p) {
 }
 
 define void @speculative_load_cost_scalable(ptr %p) {
-; COMMON-LABEL: 'speculative_load_cost_scalable'
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %1 = call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %2 = call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %3 = call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
+; NEON-LABEL: 'speculative_load_cost_scalable'
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %1 = call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %2 = call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %3 = call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; SVE-LABEL: 'speculative_load_cost_scalable'
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %1 = call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %2 = call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %3 = call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
 ;
+  ; Scalable vector types - invalid without SVE, valid with SVE if <= 16 bytes
   call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
   call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
   call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
diff --git a/llvm/test/Analysis/CostModel/X86/speculative-load.ll b/llvm/test/Analysis/CostModel/X86/speculative-load.ll
index 81b3b56eeb10c3..d5ed066dd83ceb 100644
--- a/llvm/test/Analysis/CostModel/X86/speculative-load.ll
+++ b/llvm/test/Analysis/CostModel/X86/speculative-load.ll
@@ -3,15 +3,15 @@
 
 define void @speculative_load_cost(ptr %p) {
 ; CHECK-LABEL: 'speculative_load_cost'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 11 for instruction: %5 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %6 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 5 for instruction: %7 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 7 for instruction: %8 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 3 for instruction: %9 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %5 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %6 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %7 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %8 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %9 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
 ; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
 ;
   call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)

>From c777075e7207adbd6123565c325f291db088d323 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 23 Sep 2026 10:01:46 +0100
Subject: [PATCH 2/2] !fixup add isLegalSpeculativeLoad

---
 .../llvm/Analysis/TargetTransformInfo.h       |  6 +++
 .../llvm/Analysis/TargetTransformInfoImpl.h   |  8 +++-
 llvm/include/llvm/CodeGen/BasicTTIImpl.h      | 14 +++++--
 llvm/lib/Analysis/TargetTransformInfo.cpp     |  5 +++
 .../AArch64/AArch64TargetTransformInfo.cpp    | 39 ++++++++++---------
 .../AArch64/AArch64TargetTransformInfo.h      |  3 ++
 .../CostModel/AArch64/speculative-load.ll     | 21 +++++-----
 .../CostModel/X86/speculative-load.ll         | 18 ++++-----
 8 files changed, 70 insertions(+), 44 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index e30cbc61a5420b..a683a40c14e76d 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -933,6 +933,12 @@ class TargetTransformInfo {
   isLegalMaskedLoad(Type *DataType, Align Alignment, unsigned AddressSpace,
                     MaskKind MaskKind = VariableOrConstantMask) const;
 
+  /// Return true if the target supports speculatively loading \p DataType from
+  /// address space \p AddressSpace, i.e. @llvm.can.load.speculatively can
+  /// return true for the store size of \p DataType.
+  LLVM_ABI bool isLegalSpeculativeLoad(Type *DataType,
+                                       unsigned AddressSpace) const;
+
   /// Return true if the target supports nontemporal store.
   LLVM_ABI bool isLegalNTStore(Type *DataType, Align Alignment) const;
   /// Return true if the target supports nontemporal load.
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index f487ec556a0b98..71dead1c3c5cd5 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -368,6 +368,11 @@ class LLVM_ABI TargetTransformInfoImplBase {
     return false;
   }
 
+  virtual bool isLegalSpeculativeLoad(Type *DataType,
+                                      unsigned AddressSpace) const {
+    return false;
+  }
+
   virtual bool isLegalNTStore(Type *DataType, Align Alignment) const {
     // By default, assume nontemporal memory stores are available for stores
     // that are aligned and have a size that is a power of 2.
@@ -926,8 +931,6 @@ class LLVM_ABI TargetTransformInfoImplBase {
     switch (ICA.getID()) {
     default:
       break;
-    case Intrinsic::speculative_load:
-      return InstructionCost::getInvalid();
     case Intrinsic::allow_runtime_check:
     case Intrinsic::allow_ubsan_check:
     case Intrinsic::annotation:
@@ -993,6 +996,7 @@ class LLVM_ABI TargetTransformInfoImplBase {
     case Intrinsic::vp_gather:
     case Intrinsic::masked_compressstore:
     case Intrinsic::masked_expandload:
+    case Intrinsic::speculative_load:
       return 1;
     }
     return InstructionCost::getInvalid();
diff --git a/llvm/include/llvm/CodeGen/BasicTTIImpl.h b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
index 4751a5c3713583..bb7aa02bef3aba 100644
--- a/llvm/include/llvm/CodeGen/BasicTTIImpl.h
+++ b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
@@ -2051,9 +2051,6 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
       // The cost of materialising a constant integer vector.
       return TargetTransformInfo::TCC_Basic;
     }
-    case Intrinsic::speculative_load:
-      // Delegate to base; targets must opt-in with a valid cost.
-      return BaseT::getIntrinsicInstrCost(ICA, CostKind);
     case Intrinsic::vector_extract: {
       // FIXME: Handle case where a scalable vector is extracted from a scalable
       // vector
@@ -2523,6 +2520,13 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
       return thisT()->getMemIntrinsicInstrCost(
           MemIntrinsicCostAttributes(IID, Ty, TyAlign, 0), CostKind);
     }
+    case Intrinsic::speculative_load: {
+      const IntrinsicInst *I = ICA.getInst();
+      Align Alignment = I ? I->getParamAlign(0).valueOrOne() : Align(1);
+      unsigned AS = Tys[0]->getPointerAddressSpace();
+      return thisT()->getMemIntrinsicInstrCost(
+          MemIntrinsicCostAttributes(IID, RetTy, Alignment, AS), CostKind);
+    }
     case Intrinsic::experimental_vp_strided_store: {
       auto *Ty = cast<VectorType>(ICA.getArgTypes()[0]);
       Align Alignment = thisT()->DL.getABITypeAlign(Ty->getElementType());
@@ -3279,6 +3283,10 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
     }
     case Intrinsic::vp_load_ff:
       return InstructionCost::getInvalid();
+    case Intrinsic::speculative_load:
+      // Speculative loads are lowered to regular loads of the full type.
+      return thisT()->getMemoryOpCost(Instruction::Load, DataTy, Alignment,
+                                      MICA.getAddressSpace(), CostKind);
     default:
       llvm_unreachable("unexpected intrinsic");
     }
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 4c2cac9c440a08..1223fba8d4fea1 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -494,6 +494,11 @@ bool TargetTransformInfo::isLegalMaskedLoad(Type *DataType, Align Alignment,
                                     MaskKind);
 }
 
+bool TargetTransformInfo::isLegalSpeculativeLoad(Type *DataType,
+                                                 unsigned AddressSpace) const {
+  return TTIImpl->isLegalSpeculativeLoad(DataType, AddressSpace);
+}
+
 bool TargetTransformInfo::isLegalNTStore(Type *DataType,
                                          Align Alignment) const {
   return TTIImpl->isLegalNTStore(DataType, Alignment);
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 3944b100db7ed0..8b77caf6d40cb8 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -619,25 +619,6 @@ AArch64TTIImpl::getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
       return InstructionCost::getInvalid();
 
   switch (ICA.getID()) {
-  case Intrinsic::speculative_load: {
-    // Speculative loads are only valid for types <= 16 bytes due to MTE
-    // (Memory Tagging Extension) using 16-byte tag granules. Loads larger
-    // than 16 bytes could cross a tag granule boundary.
-    auto LT = getTypeLegalizationCost(RetTy);
-    if (!LT.first.isValid())
-      return InstructionCost::getInvalid();
-    // For scalable vectors, check that we use a single register (which means
-    // <= 16 bytes at minimum vscale). For fixed types, compute the actual size.
-    if (isa<ScalableVectorType>(RetTy)) {
-      if (LT.first.getValue() != 1)
-        return InstructionCost::getInvalid();
-    } else {
-      if (LT.first.getValue() * LT.second.getStoreSize() > 16)
-        return InstructionCost::getInvalid();
-    }
-    // Return cost of a regular load.
-    return getMemoryOpCost(Instruction::Load, RetTy, Align(1), 0, CostKind);
-  }
   case Intrinsic::experimental_vector_histogram_add: {
     InstructionCost HistCost = getHistogramCost(ST, ICA);
     // If the cost isn't valid, we may still be able to scalarize
@@ -5939,6 +5920,26 @@ bool AArch64TTIImpl::isLegalMaskedExpandLoad(Type *DataTy,
          (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2());
 }
 
+bool AArch64TTIImpl::isLegalSpeculativeLoad(Type *DataType,
+                                            unsigned AddressSpace) const {
+  // Matches AArch64TargetLowering::emitCanLoadSpeculatively: only address
+  // space 0 and sizes up to the 16-byte MTE tag granule are supported.
+  if (AddressSpace != 0)
+    return false;
+  TypeSize Size = DL.getTypeStoreSize(DataType);
+  uint64_t MinSize = Size.getKnownMinValue();
+  // Scalable types are at least the minimum vscale times their known minimum
+  // size.
+  if (Size.isScalable()) {
+    if (!ST->isSVEorStreamingSVEAvailable())
+      return false;
+    MinSize *=
+        std::max(ST->getMinSVEVectorSizeInBits(), AArch64::SVEBitsPerBlock) /
+        AArch64::SVEBitsPerBlock;
+  }
+  return MinSize <= 16;
+}
+
 unsigned
 AArch64TTIImpl::getMaxInterleaveFactor(ElementCount VF,
                                        bool HasUnorderedReductions) const {
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index c1d8fc787ef623..323dc5424b9221 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -279,6 +279,9 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
 
   bool isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const override;
 
+  bool isLegalSpeculativeLoad(Type *DataType,
+                              unsigned AddressSpace) const override;
+
   void getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
                                TTI::UnrollingPreferences &UP,
                                OptimizationRemarkEmitter *ORE) const override;
diff --git a/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll b/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
index 397b63d765726d..db19fcf2c58358 100644
--- a/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
@@ -3,7 +3,7 @@
 ; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 -mattr=+sve < %s | FileCheck %s --check-prefixes=COMMON,SVE
 
 define void @speculative_load_cost_fixed(ptr %p) {
-  ; Scalar types - all valid (<= 16 bytes)
+  ; Scalar types (<= 16 bytes)
 ; COMMON-LABEL: 'speculative_load_cost_fixed'
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
@@ -19,12 +19,12 @@ define void @speculative_load_cost_fixed(ptr %p) {
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %12 = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr %p, i1 false, i64 0)
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %13 = call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr %p, i1 false, i64 0)
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %14 = call <8 x i16> (ptr, i1, ...) @llvm.speculative.load.v8i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %15 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %16 = call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %17 = call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %18 = call <16 x i16> (ptr, i1, ...) @llvm.speculative.load.v16i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %19 = call <8 x float> (ptr, i1, ...) @llvm.speculative.load.v8f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %20 = call <4 x double> (ptr, i1, ...) @llvm.speculative.load.v4f64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %15 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %16 = call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %17 = call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %18 = call <16 x i16> (ptr, i1, ...) @llvm.speculative.load.v16i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %19 = call <8 x float> (ptr, i1, ...) @llvm.speculative.load.v8f32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %20 = call <4 x double> (ptr, i1, ...) @llvm.speculative.load.v4f64.p0(ptr %p, i1 false, i64 0)
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
 ;
   call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
@@ -33,7 +33,7 @@ define void @speculative_load_cost_fixed(ptr %p) {
   call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
   call b128 (ptr, i1, ...) @llvm.speculative.load.b128.p0(ptr %p, i1 false, i64 0)
 
-  ; Vector types <= 16 bytes - valid
+  ; Vector types <= 16 bytes
   call <2 x i32> (ptr, i1, ...) @llvm.speculative.load.v2i32.p0(ptr %p, i1 false, i64 0)
   call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
   call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
@@ -44,7 +44,7 @@ define void @speculative_load_cost_fixed(ptr %p) {
   call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr %p, i1 false, i64 0)
   call <8 x i16> (ptr, i1, ...) @llvm.speculative.load.v8i16.p0(ptr %p, i1 false, i64 0)
 
-  ; Vector types > 16 bytes - invalid
+  ; Vector types > 16 bytes
   call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
   call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
   call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
@@ -72,10 +72,9 @@ define void @speculative_load_cost_scalable(ptr %p) {
 ; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
 ; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
 ; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
-; SVE-NEXT:  Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
 ; SVE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
 ;
-  ; Scalable vector types - invalid without SVE, valid with SVE if <= 16 bytes
   call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
   call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
   call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
diff --git a/llvm/test/Analysis/CostModel/X86/speculative-load.ll b/llvm/test/Analysis/CostModel/X86/speculative-load.ll
index d5ed066dd83ceb..2dba86ceabff97 100644
--- a/llvm/test/Analysis/CostModel/X86/speculative-load.ll
+++ b/llvm/test/Analysis/CostModel/X86/speculative-load.ll
@@ -3,15 +3,15 @@
 
 define void @speculative_load_cost(ptr %p) {
 ; CHECK-LABEL: 'speculative_load_cost'
-; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %5 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %6 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %7 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %8 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %9 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %5 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %6 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %7 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %8 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %9 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
 ; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
 ;
   call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)



More information about the llvm-commits mailing list