[llvm] [LV] Use speculative load intrinsics for early-exit vectorization (PR #180039)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 10:45:59 PDT 2026


https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/180039

>From f3b6f3988c798598604a2b14312400a6b3a15ada Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sat, 31 Jan 2026 14:46:20 +0000
Subject: [PATCH 1/3] [TTI] Provide conservative costs for
 @llvm.speculative.load.

Add TTI support for @llvm.speculative.load, defaulting to Invalid if not
implemented by the target.

Provide implementation for AArch64, which checks if the loaded type is
<= 16 bytes.
---
 .../llvm/Analysis/TargetTransformInfoImpl.h   |  2 +
 llvm/include/llvm/CodeGen/BasicTTIImpl.h      |  3 +
 .../AArch64/AArch64TargetTransformInfo.cpp    | 19 ++++++
 .../CostModel/AArch64/speculative-load.ll     | 65 +++++++++++--------
 .../CostModel/X86/speculative-load.ll         | 18 ++---
 5 files changed, 71 insertions(+), 36 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 84cb3a6e664b9..9805d88ec269f 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -921,6 +921,8 @@ class LLVM_ABI TargetTransformInfoImplBase {
     switch (ICA.getID()) {
     default:
       break;
+    case Intrinsic::speculative_load:
+      return InstructionCost::getInvalid();
     case Intrinsic::allow_runtime_check:
     case Intrinsic::allow_ubsan_check:
     case Intrinsic::annotation:
diff --git a/llvm/include/llvm/CodeGen/BasicTTIImpl.h b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
index cb01ef4028217..25276c5e95e02 100644
--- a/llvm/include/llvm/CodeGen/BasicTTIImpl.h
+++ b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
@@ -2049,6 +2049,9 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
       // The cost of materialising a constant integer vector.
       return TargetTransformInfo::TCC_Basic;
     }
+    case Intrinsic::speculative_load:
+      // Delegate to base; targets must opt-in with a valid cost.
+      return BaseT::getIntrinsicInstrCost(ICA, CostKind);
     case Intrinsic::vector_extract: {
       // FIXME: Handle case where a scalable vector is extracted from a scalable
       // vector
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 8e0b88dbad0eb..ac6776af86403 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -619,6 +619,25 @@ AArch64TTIImpl::getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
       return InstructionCost::getInvalid();
 
   switch (ICA.getID()) {
+  case Intrinsic::speculative_load: {
+    // Speculative loads are only valid for types <= 16 bytes due to MTE
+    // (Memory Tagging Extension) using 16-byte tag granules. Loads larger
+    // than 16 bytes could cross a tag granule boundary.
+    auto LT = getTypeLegalizationCost(RetTy);
+    if (!LT.first.isValid())
+      return InstructionCost::getInvalid();
+    // For scalable vectors, check that we use a single register (which means
+    // <= 16 bytes at minimum vscale). For fixed types, compute the actual size.
+    if (isa<ScalableVectorType>(RetTy)) {
+      if (LT.first.getValue() != 1)
+        return InstructionCost::getInvalid();
+    } else {
+      if (LT.first.getValue() * LT.second.getStoreSize() > 16)
+        return InstructionCost::getInvalid();
+    }
+    // Return cost of a regular load.
+    return getMemoryOpCost(Instruction::Load, RetTy, Align(1), 0, CostKind);
+  }
   case Intrinsic::experimental_vector_histogram_add: {
     InstructionCost HistCost = getHistogramCost(ST, ICA);
     // If the cost isn't valid, we may still be able to scalarize
diff --git a/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll b/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
index 4b7b906ce3fee..397b63d765726 100644
--- a/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/speculative-load.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py
-; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 < %s | FileCheck %s --check-prefixes=COMMON
-; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 -mattr=+sve < %s | FileCheck %s --check-prefixes=COMMON
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 < %s | FileCheck %s --check-prefixes=COMMON,NEON
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=aarch64 -mattr=+sve < %s | FileCheck %s --check-prefixes=COMMON,SVE
 
 define void @speculative_load_cost_fixed(ptr %p) {
   ; Scalar types - all valid (<= 16 bytes)
@@ -9,22 +9,22 @@ define void @speculative_load_cost_fixed(ptr %p) {
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %5 = call b128 (ptr, i1, ...) @llvm.speculative.load.b128.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 6 for instruction: %6 = call <2 x i32> (ptr, i1, ...) @llvm.speculative.load.v2i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 12 for instruction: %7 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 6 for instruction: %8 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 10 for instruction: %9 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %10 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 24 for instruction: %11 = call <8 x i8> (ptr, i1, ...) @llvm.speculative.load.v8i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 48 for instruction: %12 = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 12 for instruction: %13 = call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 24 for instruction: %14 = call <8 x i16> (ptr, i1, ...) @llvm.speculative.load.v8i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 24 for instruction: %15 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 12 for instruction: %16 = call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 96 for instruction: %17 = call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 48 for instruction: %18 = call <16 x i16> (ptr, i1, ...) @llvm.speculative.load.v16i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 20 for instruction: %19 = call <8 x float> (ptr, i1, ...) @llvm.speculative.load.v8f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %20 = call <4 x double> (ptr, i1, ...) @llvm.speculative.load.v4f64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %5 = call b128 (ptr, i1, ...) @llvm.speculative.load.b128.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %6 = call <2 x i32> (ptr, i1, ...) @llvm.speculative.load.v2i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %7 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %8 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %9 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %10 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %11 = call <8 x i8> (ptr, i1, ...) @llvm.speculative.load.v8i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %12 = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %13 = call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %14 = call <8 x i16> (ptr, i1, ...) @llvm.speculative.load.v8i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %15 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %16 = call <4 x i64> (ptr, i1, ...) @llvm.speculative.load.v4i64.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %17 = call <32 x i8> (ptr, i1, ...) @llvm.speculative.load.v32i8.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %18 = call <16 x i16> (ptr, i1, ...) @llvm.speculative.load.v16i16.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %19 = call <8 x float> (ptr, i1, ...) @llvm.speculative.load.v8f32.p0(ptr %p, i1 false, i64 0)
+; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %20 = call <4 x double> (ptr, i1, ...) @llvm.speculative.load.v4f64.p0(ptr %p, i1 false, i64 0)
 ; COMMON-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
 ;
   call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
@@ -55,16 +55,27 @@ define void @speculative_load_cost_fixed(ptr %p) {
 }
 
 define void @speculative_load_cost_scalable(ptr %p) {
-; COMMON-LABEL: 'speculative_load_cost_scalable'
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %1 = call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %2 = call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %3 = call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
-; COMMON-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
+; NEON-LABEL: 'speculative_load_cost_scalable'
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %1 = call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %2 = call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %3 = call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
+; NEON-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; SVE-LABEL: 'speculative_load_cost_scalable'
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %1 = call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %2 = call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %3 = call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %4 = call <vscale x 16 x i8> (ptr, i1, ...) @llvm.speculative.load.nxv16i8.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %5 = call <vscale x 2 x double> (ptr, i1, ...) @llvm.speculative.load.nxv2f64.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %6 = call <vscale x 4 x float> (ptr, i1, ...) @llvm.speculative.load.nxv4f32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Invalid cost for instruction: %7 = call <vscale x 8 x float> (ptr, i1, ...) @llvm.speculative.load.nxv8f32.p0(ptr %p, i1 false, i64 0)
+; SVE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
 ;
+  ; Scalable vector types - invalid without SVE, valid with SVE if <= 16 bytes
   call <vscale x 2 x i64> (ptr, i1, ...) @llvm.speculative.load.nxv2i64.p0(ptr %p, i1 false, i64 0)
   call <vscale x 4 x i32> (ptr, i1, ...) @llvm.speculative.load.nxv4i32.p0(ptr %p, i1 false, i64 0)
   call <vscale x 8 x i16> (ptr, i1, ...) @llvm.speculative.load.nxv8i16.p0(ptr %p, i1 false, i64 0)
diff --git a/llvm/test/Analysis/CostModel/X86/speculative-load.ll b/llvm/test/Analysis/CostModel/X86/speculative-load.ll
index 81b3b56eeb10c..d5ed066dd83ce 100644
--- a/llvm/test/Analysis/CostModel/X86/speculative-load.ll
+++ b/llvm/test/Analysis/CostModel/X86/speculative-load.ll
@@ -3,15 +3,15 @@
 
 define void @speculative_load_cost(ptr %p) {
 ; CHECK-LABEL: 'speculative_load_cost'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 11 for instruction: %5 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %6 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 5 for instruction: %7 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 7 for instruction: %8 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 3 for instruction: %9 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %1 = call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %2 = call b16 (ptr, i1, ...) @llvm.speculative.load.b16.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %3 = call b32 (ptr, i1, ...) @llvm.speculative.load.b32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %4 = call b64 (ptr, i1, ...) @llvm.speculative.load.b64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %5 = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %6 = call <8 x i32> (ptr, i1, ...) @llvm.speculative.load.v8i32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %7 = call <2 x i64> (ptr, i1, ...) @llvm.speculative.load.v2i64.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %8 = call <4 x float> (ptr, i1, ...) @llvm.speculative.load.v4f32.p0(ptr %p, i1 false, i64 0)
+; CHECK-NEXT:  Cost Model: Invalid cost for instruction: %9 = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr %p, i1 false, i64 0)
 ; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret void
 ;
   call b8 (ptr, i1, ...) @llvm.speculative.load.b8.p0(ptr %p, i1 false, i64 0)

>From 314e61e8b9b0b21056953d487563b4544a2b6c3e Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 10 Mar 2026 15:04:36 +0000
Subject: [PATCH 2/3] [VPlan] Handle broadcast in getSCEVExprForVPValue (NFC)

A broadcast just replicates a scalar value, so its SCEV expression is
the same as its operand's. This allows getSCEVExprForVPValue to see
through broadcasts when computing SCEV expressions for VPValues.
---
 llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 3 +++
 1 file changed, 3 insertions(+)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 6b2233f606f91..ea07220c279c2 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -165,6 +165,9 @@ const SCEV *vputils::getSCEVExprForVPValue(const VPValue *V,
   };
 
   VPValue *LHSVal, *RHSVal;
+  // Broadcast just replicates a scalar, so the SCEV is the same as its operand.
+  if (match(V, m_Broadcast(m_VPValue(LHSVal))))
+    return getSCEVExprForVPValue(LHSVal, PSE, L);
   if (match(V, m_Add(m_VPValue(LHSVal), m_VPValue(RHSVal))))
     return CreateSCEV({LHSVal, RHSVal}, [&](ArrayRef<SCEVUse> Ops) {
       return SE.getAddExpr(Ops[0], Ops[1], SCEV::FlagAnyWrap, 0);

>From 8565f1064d911ac38b79a1b5ce00e0aa9f598942 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sat, 21 Mar 2026 21:49:41 +0000
Subject: [PATCH 3/3] [LV] Use speculative load intrinsics for early-exit
 vectorization

Add initial support for vectorizing loops using llvm.speculative.load
when loads are not known dereferenceable. The intrinsic was added in
https://github.com/llvm/llvm-project/pull/179642.

LV has to generate an oracle function, which returns the number of
elements the original scalar loop would have read, starting at the
current index of the vector loop.

To do so, introduce a new VPSpeculativeLoadRecipe, which models the
speculative load and also contains a VPlan that is used to generate the
oracle function during execute.

The oracle VPlan is the based on the original scalar VPlan. It replays
the scalar loop, starting at the current index of the vector loop
(passed to the oracle function as argument, same as other live-ins like
base pointers). The exits are updated to return the index of the IV that
exits the loop.

A new LiveIn VPInstruction has been added to model the live-in arguments
to the oracle; at the point where the oracle plan is created, no IR
values exist for them yet. During execute, they will be replaced by the
created function arguments.

The initial implementation limits the supported loops to cases where all
speculative loads share the same element type, access are guaranteed
consecutive and do not have multiple early exits, to keep the
implementation simple.
---
 llvm/lib/Analysis/VectorUtils.cpp             |   3 +
 .../Vectorize/LoopVectorizationPlanner.h      |   7 +
 .../Transforms/Vectorize/LoopVectorize.cpp    |  21 +-
 llvm/lib/Transforms/Vectorize/VPlan.cpp       |  27 +-
 llvm/lib/Transforms/Vectorize/VPlan.h         |  55 ++
 .../Vectorize/VPlanConstruction.cpp           | 285 ++++++-
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp | 112 ++-
 .../Transforms/Vectorize/VPlanTransforms.cpp  |   7 +-
 .../Transforms/Vectorize/VPlanTransforms.h    |  21 +-
 llvm/lib/Transforms/Vectorize/VPlanUtils.cpp  |   7 +
 llvm/lib/Transforms/Vectorize/VPlanUtils.h    |   4 +
 .../Transforms/Vectorize/VPlanVerifier.cpp    |  30 +
 .../early-exit-with-speculative-load-basic.ll | 396 ++++++++-
 ...y-exit-with-speculative-load-load-types.ll | 230 ++++-
 ...rly-exit-with-speculative-load-scalable.ll |  69 +-
 .../LoopVectorize/AArch64/early_exit_cost.ll  | 109 ++-
 ...early-exit-with-speculative-load-oracle.ll | 789 +++++++++++++++++-
 .../VPlan/vplan-print-before-after-all.ll     |   1 +
 .../LoopVectorize/early_exit_legality.ll      |   9 +-
 .../Transforms/Vectorize/VPlanTest.cpp        |  20 +
 .../Vectorize/VPlanVerifierTest.cpp           |  71 ++
 21 files changed, 2134 insertions(+), 139 deletions(-)

diff --git a/llvm/lib/Analysis/VectorUtils.cpp b/llvm/lib/Analysis/VectorUtils.cpp
index 1c105ebb772b3..7c620c2a9a577 100644
--- a/llvm/lib/Analysis/VectorUtils.cpp
+++ b/llvm/lib/Analysis/VectorUtils.cpp
@@ -158,6 +158,8 @@ bool llvm::isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID,
   case Intrinsic::powi:
   case Intrinsic::vector_extract:
     return (ScalarOpdIdx == 1);
+  case Intrinsic::speculative_load:
+    return true;
   case Intrinsic::smul_fix:
   case Intrinsic::smul_fix_sat:
   case Intrinsic::umul_fix:
@@ -196,6 +198,7 @@ bool llvm::isVectorIntrinsicWithOverloadTypeAtArg(
   case Intrinsic::scmp:
   case Intrinsic::vector_extract:
   case Intrinsic::loop_dependence_war_mask:
+  case Intrinsic::speculative_load:
     return OpdIdx == -1 || OpdIdx == 0;
   case Intrinsic::modf:
   case Intrinsic::sincos:
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index e14c335701f62..816db57ea3daf 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -469,6 +469,13 @@ class VPBuilder {
     return createNaryOp(Instruction::Freeze, Op, DL, Name);
   }
 
+  VPInstruction *createLiveIn(unsigned Idx, Type *ResultTy,
+                              const Twine &Name = "") {
+    return tryInsertInstruction(new VPInstruction(
+        VPInstruction::LiveIn, getPlan().getConstantInt(32, Idx), {}, {},
+        DebugLoc::getUnknown(), Name, ResultTy));
+  }
+
   VPWidenCastRecipe *createWidenCast(Instruction::CastOps Opcode, VPValue *Op,
                                      Type *ResultTy) {
     assert(Op->getScalarType() != ResultTy &&
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 716c73a7c63b3..da035e1d998c9 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3273,6 +3273,7 @@ static bool willGenerateVectors(VPlan &Plan, ElementCount VF,
       case VPRecipeBase::VPExpandSCEVSC:
       case VPRecipeBase::VPPredInstPHISC:
       case VPRecipeBase::VPBranchOnMaskSC:
+      case VPRecipeBase::VPSpeculativeLoadOracleSC:
         continue;
       case VPRecipeBase::VPReductionSC:
       case VPRecipeBase::VPActiveLaneMaskPHISC:
@@ -5570,7 +5571,7 @@ LoopVectorizationPlanner::computeBestVF() {
     return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
   }
 
-  if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF() && VPlans.size() == 2) {
+  if (FirstPlan.hasVF(UserVF) && hasForcedEpilogueVF() && VPlans.size() == 2) {
     assert(VPlans[0]->getSingleVF() == UserVF &&
            "expected second plan to be for the forced UserVF");
     assert(VPlans[1]->getSingleVF() == EpilogueVectorizationForceVF &&
@@ -5695,11 +5696,13 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   // TODO: Move to VPlan transform stage once the transition to the VPlan-based
   // cost model is complete for better cost estimates.
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
+  bool HasBranchWeights =
+      hasBranchWeightMD(*OrigLoop->getLoopLatch()->getTerminator());
+  RUN_VPLAN_PASS(VPlanTransforms::attachSpeculativeLoadChecks, BestVPlan,
+                 BestVF, PSE, OrigLoop, HasBranchWeights);
   RUN_VPLAN_PASS(VPlanTransforms::materializePacksAndUnpacks, BestVPlan);
   RUN_VPLAN_PASS(VPlanTransforms::materializeBroadcasts, BestVPlan);
   RUN_VPLAN_PASS(VPlanTransforms::replicateByVF, BestVPlan, BestVF);
-  bool HasBranchWeights =
-      hasBranchWeightMD(*OrigLoop->getLoopLatch()->getTerminator());
   if (HasBranchWeights) {
     std::optional<unsigned> VScale = Config.getVScaleForTuning();
     RUN_VPLAN_PASS(VPlanTransforms::addBranchWeightToMiddleTerminator,
@@ -6620,7 +6623,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
       if (isa<VPWidenCanonicalIVRecipe, VPBlendRecipe, VPReductionRecipe,
               VPReplicateRecipe, VPWidenLoadRecipe, VPWidenStoreRecipe,
               VPWidenCallRecipe, VPWidenIntrinsicRecipe, VPVectorPointerRecipe,
-              VPVectorEndPointerRecipe, VPHistogramRecipe>(&R) ||
+              VPVectorEndPointerRecipe, VPHistogramRecipe,
+              VPSpeculativeLoadOracleRecipe>(&R) ||
           (Instruction::isCast(cast<VPInstruction>(R).getOpcode()) &&
            vputils::onlyFirstLaneUsed(R.getVPSingleValue())))
         continue;
@@ -8032,6 +8036,15 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     IC = 1;
   }
 
+  // FIXME: Enable interleaving for plans with speculative loads.
+  if (InterleaveLoop && vputils::findSpeculativeLoadOracle(*BestPlanPtr)) {
+    LLVM_DEBUG(dbgs() << "LV: Not interleaving loop with speculative loads.\n");
+    IntDiagMsg = {"SpeculativeLoadPreventsInterleaving",
+                  "Unable to interleave loop using speculative loads."};
+    InterleaveLoop = false;
+    IC = 1;
+  }
+
   // Emit diagnostic messages, if any.
   if (!VectorizeLoop && !InterleaveLoop) {
     // Do not vectorize or interleaving the loop.
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.cpp b/llvm/lib/Transforms/Vectorize/VPlan.cpp
index d1f8230b8eb06..8b34846b8688c 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlan.cpp
@@ -569,7 +569,7 @@ const VPRegionBlock *VPBasicBlock::getEnclosingLoopRegion() const {
   return getEnclosingLoopRegionForRegion(getParent());
 }
 
-static bool hasConditionalTerminator(const VPBasicBlock *VPBB) {
+static bool hasTerminator(const VPBasicBlock *VPBB) {
   if (VPBB->empty()) {
     assert(
         VPBB->getNumSuccessors() < 2 &&
@@ -578,6 +578,8 @@ static bool hasConditionalTerminator(const VPBasicBlock *VPBB) {
   }
 
   const VPRecipeBase *R = &VPBB->back();
+  if (match(R, m_VPInstruction<Instruction::Ret>()))
+    return true;
   [[maybe_unused]] bool IsSwitch =
       isa<VPInstruction>(R) &&
       cast<VPInstruction>(R)->getOpcode() == Instruction::Switch;
@@ -608,13 +610,13 @@ static bool hasConditionalTerminator(const VPBasicBlock *VPBB) {
 }
 
 VPRecipeBase *VPBasicBlock::getTerminator() {
-  if (hasConditionalTerminator(this))
+  if (hasTerminator(this))
     return &back();
   return nullptr;
 }
 
 const VPRecipeBase *VPBasicBlock::getTerminator() const {
-  if (hasConditionalTerminator(this))
+  if (hasTerminator(this))
     return &back();
   return nullptr;
 }
@@ -1101,11 +1103,10 @@ void VPlan::printLiveIns(raw_ostream &O) const {
   }
 }
 
-LLVM_DUMP_METHOD
-void VPlan::print(raw_ostream &O) const {
+void VPlan::print(raw_ostream &O, const Twine &Title) const {
   VPSlotTracker SlotTracker(this);
 
-  O << "VPlan '" << getName() << "' {";
+  O << Title << " {";
 
   printLiveIns(O);
 
@@ -1117,6 +1118,20 @@ void VPlan::print(raw_ostream &O) const {
   }
 
   O << "}\n";
+
+  for (const VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<const VPBasicBlock>(
+           vp_depth_first_deep(getEntry())))
+    for (const auto &Oracle :
+         make_isa_range<const VPSpeculativeLoadOracleRecipe>(*VPBB)) {
+      O << '\n';
+      Oracle.getOraclePlan().print(O, "VPlan for speculative-load oracle " +
+                                          SlotTracker.getOrCreateName(&Oracle));
+    }
+}
+
+LLVM_DUMP_METHOD
+void VPlan::print(raw_ostream &O) const {
+  print(O, "VPlan '" + getName() + "'");
 }
 
 std::string VPlan::getName() const {
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 91aaa37f798a0..b6be54f8ab24f 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -436,6 +436,7 @@ class LLVM_ABI_FOR_TEST VPRecipeBase
     VPReductionSC,
     VPReplicateSC,
     VPScalarIVStepsSC,
+    VPSpeculativeLoadOracleSC,
     VPVectorPointerSC,
     VPVectorEndPointerSC,
     VPWidenCallSC,
@@ -640,6 +641,7 @@ class LLVM_ABI_FOR_TEST VPSingleDefRecipe : public VPRecipeBase,
     case VPRecipeBase::VPReductionSC:
     case VPRecipeBase::VPReplicateSC:
     case VPRecipeBase::VPScalarIVStepsSC:
+    case VPRecipeBase::VPSpeculativeLoadOracleSC:
     case VPRecipeBase::VPVectorPointerSC:
     case VPRecipeBase::VPVectorEndPointerSC:
     case VPRecipeBase::VPWidenCallSC:
@@ -1328,6 +1330,9 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     // Represents the incoming loop-invariant alias-mask. All memory accesses
     // in the loop must stay within the active lanes.
     IncomingAliasMask,
+    // Yields the value of the plan's live-in at the index given by operand 0,
+    // supplied externally when the plan is executed.
+    LiveIn,
     // Increment the canonical IV separately for each unrolled part.
     CanonicalIVIncrementForPart,
     // Abstract instruction that compares two values and branches. This is
@@ -3544,6 +3549,53 @@ class LLVM_ABI_FOR_TEST VPBranchOnMaskRecipe : public VPRecipeBase,
   }
 };
 
+/// Defines the oracle function for a @llvm.speculative.load intrinsic call. It
+/// contains a nested VPlan which is used to generate the oracle function,
+/// returning the number of bytes read by the intrinsic.
+class VPSpeculativeLoadOracleRecipe : public VPSingleDefRecipe {
+  std::unique_ptr<VPlan> OraclePlan;
+
+  /// The number of bytes to load per lane.
+  uint64_t AccessSize;
+
+  /// Symbolic value for the base index, which will be passed as argument.
+  std::unique_ptr<VPSymbolicValue> BaseIndex;
+
+public:
+  VPSpeculativeLoadOracleRecipe(
+      std::unique_ptr<VPlan> Oracle, uint64_t AccessSize,
+      ArrayRef<VPValue *> Args,
+      std::unique_ptr<VPSymbolicValue> BaseIndex = nullptr);
+
+  VP_CLASSOF_IMPL(VPRecipeBase::VPSpeculativeLoadOracleSC)
+
+  VPSpeculativeLoadOracleRecipe *clone() override;
+  void execute(VPTransformState &State) override;
+
+  InstructionCost computeCost(ElementCount VF, VPCostContext &) const override {
+    return VF.isScalable() ? InstructionCost::getInvalid() : InstructionCost(0);
+  }
+
+  bool usesFirstLaneOnly(const VPValue *Op) const override {
+    assert(is_contained(operands(), Op) &&
+           "Op must be an operand of the recipe");
+    return true;
+  }
+
+  /// Returns the plan the oracle function is generated from.
+  const VPlan &getOraclePlan() const { return *OraclePlan; }
+
+  /// Returns the number of bytes each lane of a speculative load using this
+  /// oracle reads.
+  uint64_t getAccessSize() const { return AccessSize; }
+
+protected:
+#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
+  void printRecipe(raw_ostream &O, const Twine &Indent,
+                   VPSlotTracker &Tracker) const override;
+#endif
+};
+
 /// A recipe to combine multiple recipes into a single 'expression' recipe,
 /// which should be considered a single entity for cost-modeling and transforms.
 /// The recipe needs to be 'decomposed', i.e. replaced by its individual
@@ -5174,6 +5226,9 @@ class VPlan {
   /// Print the live-ins of this VPlan to \p O.
   void printLiveIns(raw_ostream &O) const;
 
+  /// Print this VPlan to \p O, headed by \p Title.
+  void print(raw_ostream &O, const Twine &Title) const;
+
   /// Print this VPlan to \p O.
   LLVM_ABI_FOR_TEST void print(raw_ostream &O) const;
 
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 18498084a601a..3cdd72287dcaf 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1229,14 +1229,157 @@ void VPlanTransforms::createInLoopReductionRecipes(VPlan &Plan,
     R->eraseFromParent();
 }
 
-bool VPlanTransforms::areAllLoadsDereferenceable(VPBasicBlock *HeaderVPBB,
-                                                 Loop *TheLoop,
-                                                 PredicatedScalarEvolution &PSE,
-                                                 DominatorTree &DT,
-                                                 AssumptionCache *AC) {
+/// Clone plain CFG \p Plan into an oracle plan: a scalar loop with trip count
+/// VF replaying the exit conditions, returning safe lanes * \p AccessSize
+/// bytes.
+static VPSpeculativeLoadOracleRecipe *buildOraclePlan(VPlan &Plan,
+                                                      uint64_t AccessSize) {
+  std::unique_ptr<VPlan> OraclePlan(Plan.duplicate());
+
+  auto [HeaderVPBB, LatchVPBB] =
+      VPBlockUtils::getPlainCFGHeaderAndLatch(*OraclePlan);
+  Type *IVTy = OraclePlan->getVectorTripCount().getType();
+  VPValue *Zero = OraclePlan->getZero(IVTy);
+  VPValue *One = OraclePlan->getConstantInt(IVTy, 1);
+  auto *LatchTerm = cast<VPInstruction>(LatchVPBB->getTerminator());
+  assert(match(LatchTerm, m_BranchOnCond()) && "Unexpected terminator");
+  auto *ScalarIVPHI =
+      VPBuilder(HeaderVPBB, HeaderVPBB->begin())
+          .createScalarPhi({Zero}, DebugLoc::getUnknown(), "index");
+  VPBuilder LatchBuilder(LatchTerm);
+  VPValue *IVInc = LatchBuilder.createAdd(
+      ScalarIVPHI, One, DebugLoc::getUnknown(), "index.next", {true, false});
+  ScalarIVPHI->addIncoming(IVInc);
+  LatchBuilder.createNaryOp(VPInstruction::BranchOnCount,
+                            {IVInc, &OraclePlan->getVF()},
+                            LatchTerm->getDebugLoc());
+  LatchTerm->eraseFromParent();
+
+  // Make the header the entry's single successor; the original entry and the
+  // vector skeleton become unreachable.
+  auto *NewEntry = OraclePlan->createVPBasicBlock("oracle.entry");
+  OraclePlan->setEntry(NewEntry);
+  VPBlockUtils::disconnectBlocks(HeaderVPBB->getPredecessors()[0], HeaderVPBB);
+  VPBlockUtils::connectBlocks(NewEntry, HeaderVPBB);
+  HeaderVPBB->swapPredecessors();
+
+  // Update the IV to start at the current vector iteration, passed as argument
+  // to the oracle function.
+  VPValue *BaseIndex = VPBuilder(NewEntry).createLiveIn(0, IVTy, "base.index");
+  VPBuilder Builder(HeaderVPBB, HeaderVPBB->getFirstNonPhi());
+  auto *AdjustedIV = Builder.createAdd(ScalarIVPHI, BaseIndex,
+                                       DebugLoc::getUnknown(), "adjusted.iv");
+
+  // Replace wide inductions with scalar equivalents based on the adjusted IV.
+  assert(
+      none_of(HeaderVPBB->phis(),
+              IsaPred<VPReductionPHIRecipe, VPFirstOrderRecurrencePHIRecipe>) &&
+      "unsupported header phi in the oracle plan");
+  for (VPWidenIntOrFpInductionRecipe &WideIV : make_early_inc_range(
+           make_isa_range<VPWidenIntOrFpInductionRecipe>(HeaderVPBB->phis()))) {
+    const InductionDescriptor &ID = WideIV.getInductionDescriptor();
+    WideIV.replaceAllUsesWith(Builder.createDerivedIV(
+        ID.getKind(), dyn_cast_or_null<FPMathOperator>(ID.getInductionBinOp()),
+        WideIV.getStartValue(), AdjustedIV, WideIV.getStepValue()));
+    WideIV.eraseFromParent();
+  }
+
+  // Redirect all early exits and the latch exit to a single new exit block.
+  auto *MiddleVPBB = VPBlockUtils::getPlainCFGMiddleBlock(*OraclePlan);
+  auto *ExitVPBB = OraclePlan->createVPBasicBlock("oracle.exit");
+  for (auto [Pred, EarlyExitVPBB] :
+       vputils::getEarlyExits(*OraclePlan, MiddleVPBB)) {
+    for (VPRecipeBase &R : EarlyExitVPBB->phis())
+      cast<VPIRPhi>(&R)->removeIncomingValueFor(Pred);
+    VPBlockUtils::replaceSuccessor(Pred, EarlyExitVPBB, ExitVPBB);
+  }
+  VPBlockUtils::replaceSuccessor(LatchVPBB, MiddleVPBB, ExitVPBB);
+
+  // The scalar header is unreachable.
+  VPBlockUtils::disconnectBlocks(OraclePlan->getScalarPreheader(),
+                                 OraclePlan->getScalarHeader());
+
+  // Reset a recipe-defined trip count; it now lives in an unreachable block.
+  VPValue *TC = OraclePlan->getTripCount();
+  if (TC->getDefiningRecipe()) {
+    TC->replaceAllUsesWith(Zero);
+    OraclePlan->resetTripCount(Zero);
+  }
+  // Return the number of valid leading bytes: (exit IV + 1) * store size.
+  Type *I64Ty = Type::getInt64Ty(OraclePlan->getContext());
+  VPBuilder ExitBuilder(ExitVPBB);
+  VPValue *LaneCount = ExitBuilder.createScalarZExtOrTrunc(
+      ExitBuilder.createAdd(ScalarIVPHI, One, DebugLoc::getUnknown(), "lanes"),
+      I64Ty, {});
+  VPValue *ByteCount = ExitBuilder.createOverflowingOp(
+      Instruction::Mul,
+      {LaneCount, OraclePlan->getConstantInt(I64Ty, AccessSize)}, {},
+      DebugLoc::getUnknown(), "bytes");
+  ExitBuilder.createNaryOp(Instruction::Ret, ByteCount);
+
+  VPlanTransforms::removeDeadRecipes(*OraclePlan);
+  VPlanTransforms::convertToConcreteRecipes(*OraclePlan);
+  VPlanTransforms::combineRecipes(*OraclePlan);
+
+  // Scalarize the remaining Load and GEP recipes in the executable CFG.
+  ReversePostOrderTraversal<VPBlockShallowTraversalWrapper<VPBlockBase *>> RPOT(
+      HeaderVPBB);
+  auto OracleBlocks = VPBlockUtils::blocksAs<VPBasicBlock>(RPOT);
+  for (VPBasicBlock *VPBB : OracleBlocks) {
+    for (VPInstruction &VPI :
+         make_early_inc_range(make_isa_range<VPInstruction>(*VPBB))) {
+      unsigned Opc = VPI.getOpcode();
+      if (Opc != Instruction::Load && Opc != Instruction::GetElementPtr)
+        continue;
+      auto *Replicate = VPBuilder::createSingleScalarOp(
+          Opc, VPI.operands(), /*Mask=*/nullptr, VPI, VPI, VPI.getDebugLoc(),
+          VPI.getUnderlyingInstr());
+      Replicate->insertBefore(&VPI);
+      VPI.replaceAllUsesWith(Replicate);
+      VPI.eraseFromParent();
+    }
+  }
+
+  // Collect the live-ins the oracle needs, those will become function
+  // arguments.
+  SetVector<VPIRValue *> LiveIns;
+  for (VPBasicBlock *VPBB : OracleBlocks)
+    for (VPRecipeBase &R : *VPBB)
+      for (VPValue *Op : R.operands()) {
+        auto *LiveIn = dyn_cast<VPIRValue>(Op);
+        if (LiveIn && !isa<ConstantData>(LiveIn->getValue()))
+          LiveIns.insert(LiveIn);
+      }
+
+  auto SymbolicBaseIndex = std::make_unique<VPSymbolicValue>(IVTy);
+  SmallVector<VPValue *> Operands = {SymbolicBaseIndex.get()};
+  VPBuilder EntryBuilder(NewEntry);
+  for (VPIRValue *LiveIn : LiveIns) {
+    Value *V = LiveIn->getValue();
+    LiveIn->replaceAllUsesWith(
+        EntryBuilder.createLiveIn(Operands.size(), V->getType(), V->getName()));
+    Operands.push_back(Plan.getOrAddLiveIn(V));
+  }
+
+  return new VPSpeculativeLoadOracleRecipe(std::move(OraclePlan), AccessSize,
+                                           Operands,
+                                           std::move(SymbolicBaseIndex));
+}
+
+bool VPlanTransforms::replaceUnsafeLoadsWithSpeculative(
+    VPlan &Plan, Loop *TheLoop, PredicatedScalarEvolution &PSE,
+    DominatorTree &DT, AssumptionCache *AC) {
   ScalarEvolution &SE = *PSE.getSE();
-  const DataLayout &DL = TheLoop->getHeader()->getDataLayout();
-  for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(HeaderVPBB)) {
+  const DataLayout &DL = Plan.getDataLayout();
+  auto [HeaderVPBB, LatchVPBB] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
+  auto Bail = [](const char *Reason) {
+    LLVM_DEBUG(dbgs() << "LV: Not vectorizing: " << Reason << ".\n");
+    return false;
+  };
+
+  auto LoopBody = vp_rpo_plain_cfg_loop_body(HeaderVPBB);
+  SmallVector<VPInstruction *> UnsafeLoads;
+  for (VPBasicBlock *VPBB : LoopBody) {
     for (VPRecipeBase &R : *VPBB) {
       auto *VPI = dyn_cast<VPInstruction>(&R);
       if (!VPI || VPI->getOpcode() != Instruction::Load) {
@@ -1245,30 +1388,89 @@ bool VPlanTransforms::areAllLoadsDereferenceable(VPBasicBlock *HeaderVPBB,
       }
 
       // Get the pointer SCEV for dereferenceability checking.
-      VPValue *Ptr = VPI->getOperand(0);
-      const SCEV *PtrSCEV = vputils::getSCEVExprForVPValue(Ptr, PSE, TheLoop);
-      if (isa<SCEVCouldNotCompute>(PtrSCEV)) {
-        LLVM_DEBUG(dbgs() << "LV: Not vectorizing: Found non-dereferenceable "
-                             "load with SCEVCouldNotCompute pointer\n");
-        return false;
-      }
-
-      // Check dereferenceability using the SCEV-based version.
-      Type *LoadTy = VPI->getScalarType();
-      const SCEV *SizeSCEV =
-          SE.getStoreSizeOfExpr(DL.getIndexType(PtrSCEV->getType()), LoadTy);
+      const SCEV *PtrSCEV =
+          vputils::getSCEVExprForVPValue(VPI->getOperand(0), PSE, TheLoop);
+      if (isa<SCEVCouldNotCompute>(PtrSCEV))
+        return Bail("load pointer SCEV cannot be computed");
+
+      // TODO: Support safe loop-invariant loads here.
+      const SCEV *SizeSCEV = SE.getStoreSizeOfExpr(
+          DL.getIndexType(PtrSCEV->getType()), VPI->getScalarType());
       auto *Load = cast<LoadInst>(VPI->getUnderlyingValue());
       SmallVector<const SCEVPredicate *> Preds;
       if (isDereferenceableAndAlignedInLoop(PtrSCEV, Load->getAlign(), SizeSCEV,
                                             TheLoop, SE, DT, AC, &Preds))
         continue;
 
-      LLVM_DEBUG(
-          dbgs() << "LV: Not vectorizing: Auto-vectorization of loops with "
-                    "potentially faulting load is not supported.\n");
-      return false;
+      UnsafeLoads.push_back(VPI);
+    }
+  }
+
+  if (UnsafeLoads.empty())
+    return true;
+
+  // TODO: Support loads with different element types.
+  Type *EltTy = UnsafeLoads.front()->getScalarType();
+  uint64_t AccessSize = DL.getTypeStoreSize(EltTy).getFixedValue();
+  if (EltTy->getPrimitiveSizeInBits().getKnownMinValue() != AccessSize * 8 ||
+      !isPowerOf2_64(AccessSize) ||
+      any_of(UnsafeLoads,
+             [EltTy](VPInstruction *L) { return L->getScalarType() != EltTy; }))
+    return Bail("unsupported element type for speculative loads");
+
+  // TODO: Support non-unit-strided loads.
+  if (any_of(UnsafeLoads, [&](VPInstruction *L) {
+        return vputils::getConstantStride(L->getOperand(0), EltTy, PSE,
+                                          TheLoop) != 1;
+      }))
+    return Bail("speculative load is not a consecutive access");
+
+  auto EarlyExits =
+      vputils::getEarlyExits(Plan, VPBlockUtils::getPlainCFGMiddleBlock(Plan));
+  if (EarlyExits.size() != 1)
+    return Bail("loop has multiple early exits");
+
+  VPBasicBlock *EarlyExitingVPBB = EarlyExits.front().first;
+  VPDominatorTree VPDT(Plan);
+  for (VPInstruction *VPI : UnsafeLoads)
+    if (!VPDT.dominates(VPI->getParent(), EarlyExitingVPBB) ||
+        !VPDT.dominates(VPI->getParent(), LatchVPBB))
+      return Bail("conditionally executed unsafe load");
+
+  // TODO: Extend oracle logic to support remaining recipes.
+  auto ContainsUnsupportedRecipes = [](const VPRecipeBase &R) {
+    if (auto *VPI = dyn_cast<VPInstruction>(&R)) {
+      unsigned Opc = VPI->getOpcode();
+      // Independent freezes can make the oracle exit before the vector loop,
+      // replacing a later, defined exit's loaded value with poison.
+      return Opc == Instruction::Call || Opc == Instruction::FCmp ||
+             Opc == Instruction::Freeze || Instruction::isUnaryOp(Opc);
     }
+    if (isa<VPWidenPointerInductionRecipe>(&R))
+      return true;
+    auto *IV = dyn_cast<VPWidenIntOrFpInductionRecipe>(&R);
+    return IV && IV->getStepValue()->getDefiningRecipe();
+  };
+  if (any_of(LoopBody, [&](VPBasicBlock *VPBB) {
+        return any_of(*VPBB, ContainsUnsupportedRecipes);
+      }))
+    return Bail("loop body cannot be replayed by a speculative-load oracle");
+
+  auto *Oracle = buildOraclePlan(Plan, AccessSize);
+  Oracle->insertBefore(*HeaderVPBB, HeaderVPBB->getFirstNonPhi());
+
+  for (VPInstruction *VPI : UnsafeLoads) {
+    SmallVector<VPValue *> Ops = {VPI->getOperand(0),
+                                  /*FromEnd=*/Plan.getFalse(), Oracle};
+    append_range(Ops, Oracle->operands());
+    VPBuilder Builder(VPI);
+    VPValue *SpecLoad = Builder.insert(new VPWidenIntrinsicRecipe(
+        Intrinsic::speculative_load, Ops, VPI->getScalarType(), {}, {},
+        VPI->getDebugLoc()));
+    VPI->replaceAllUsesWith(Builder.createFreeze(SpecLoad, VPI->getDebugLoc()));
+    VPI->eraseFromParent();
   }
+
   return true;
 }
 
@@ -1333,6 +1535,11 @@ void VPlanTransforms::createLoopRegions(VPlan &Plan, DebugLoc DL) {
   VPRegionBlock *TopRegion = Plan.getVectorLoopRegion();
   TopRegion->setName("vector loop");
   TopRegion->getEntryBasicBlock()->setName("vector.body");
+
+  // A speculative-load oracles start at the current canonical IV.
+  if (auto *Oracle = vputils::findSpeculativeLoadOracle(Plan))
+    cast<VPSymbolicValue>(Oracle->getOperand(0))
+        ->replaceAllUsesWith(TopRegion->getCanonicalIV());
 }
 
 void VPlanTransforms::foldTailByMasking(VPlan &Plan) {
@@ -1493,6 +1700,38 @@ void VPlanTransforms::attachCheckBlock(VPlan &Plan, Value *Cond,
   attachVPCheckBlock(Plan, CondVPV, CheckBlockVPBB, AddBranchWeights);
 }
 
+void VPlanTransforms::attachSpeculativeLoadChecks(
+    VPlan &Plan, ElementCount VF, PredicatedScalarEvolution &PSE, Loop *TheLoop,
+    bool AddBranchWeights) {
+  auto *Oracle = vputils::findSpeculativeLoadOracle(Plan);
+  if (!Oracle)
+    return;
+
+  VPBasicBlock *CheckBlockVPBB = Plan.createVPBasicBlock("spec.load.check");
+  VPBuilder Builder(CheckBlockVPBB);
+  Type *I1Ty = Type::getInt1Ty(Plan.getContext());
+  Type *I64Ty = Type::getInt64Ty(Plan.getContext());
+  VPValue *SizeVal =
+      Builder.createElementCount(I64Ty, VF * Oracle->getAccessSize());
+  VPValue *AllChecksPassed = Plan.getTrue();
+  for (VPUser *U : Oracle->users()) {
+    auto *R = cast<VPWidenIntrinsicRecipe>(U);
+    const SCEV *PtrSCEV =
+        vputils::getSCEVExprForVPValue(R->getOperand(0), PSE, TheLoop);
+    assert(!isa<SCEVCouldNotCompute>(PtrSCEV) && "non-computable pointer SCEV");
+    auto *AR = cast<SCEVAddRecExpr>(PtrSCEV);
+    PtrSCEV = AR->getStart();
+    VPValue *StartPtr = vputils::getOrCreateVPValueForSCEVExpr(Plan, PtrSCEV);
+    VPValue *IsSafe = Builder.createScalarIntrinsic(
+        Intrinsic::can_load_speculatively, {StartPtr, SizeVal}, I1Ty,
+        DebugLoc::getUnknown());
+    AllChecksPassed = Builder.createAnd(AllChecksPassed, IsSafe);
+  }
+
+  attachVPCheckBlock(Plan, Builder.createNot(AllChecksPassed), CheckBlockVPBB,
+                     AddBranchWeights);
+}
+
 void VPlanTransforms::addMinimumIterationCheck(
     VPlan &Plan, ElementCount VF, unsigned UF,
     ElementCount MinProfitableTripCount, bool RequiresScalarEpilogue,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 602aff8090bf8..80cdb6a065bca 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -13,9 +13,11 @@
 
 #include "LoopVectorizationPlanner.h"
 #include "VPlan.h"
+#include "VPlanCFG.h"
 #include "VPlanHelpers.h"
 #include "VPlanPatternMatch.h"
 #include "VPlanUtils.h"
+#include "llvm/ADT/PostOrderIterator.h"
 #include "llvm/ADT/STLExtras.h"
 #include "llvm/ADT/SmallVector.h"
 #include "llvm/ADT/SmallVectorExtras.h"
@@ -25,6 +27,7 @@
 #include "llvm/Analysis/LoopInfo.h"
 #include "llvm/Analysis/ScalarEvolutionExpressions.h"
 #include "llvm/IR/BasicBlock.h"
+#include "llvm/IR/Dominators.h"
 #include "llvm/IR/IRBuilder.h"
 #include "llvm/IR/Instruction.h"
 #include "llvm/IR/Instructions.h"
@@ -96,6 +99,7 @@ bool VPRecipeBase::mayWriteToMemory() const {
   case VPScalarIVStepsSC:
   case VPPredInstPHISC:
   case VPExpandSCEVSC:
+  case VPSpeculativeLoadOracleSC:
     return false;
   case VPBlendSC:
   case VPReductionEVLSC:
@@ -151,6 +155,7 @@ bool VPRecipeBase::mayReadFromMemory() const {
   case VPWidenStoreEVLSC:
   case VPWidenStoreSC:
   case VPExpandSCEVSC:
+  case VPSpeculativeLoadOracleSC:
     return false;
   case VPBlendSC:
   case VPReductionEVLSC:
@@ -188,14 +193,12 @@ bool VPRecipeBase::mayHaveSideEffects() const {
   case VPPredInstPHISC:
   case VPVectorEndPointerSC:
   case VPExpandSCEVSC:
+  case VPSpeculativeLoadOracleSC:
     return false;
-  case VPInstructionSC: {
-    auto *VPI = cast<VPInstruction>(this);
+  case VPInstructionSC:
     return mayWriteToMemory() ||
-           VPI->getOpcode() == VPInstruction::BranchOnCount ||
-           VPI->getOpcode() == VPInstruction::BranchOnCond ||
-           VPI->getOpcode() == VPInstruction::BranchOnTwoConds;
-  }
+           match(this,
+                 m_CombineOr(m_Branch(), m_VPInstruction<Instruction::Ret>()));
   case VPWidenCallSC: {
     Function *Fn = cast<VPWidenCallRecipe>(this)->getCalledScalarFunction();
     return mayWriteToMemory() || !Fn->doesNotThrow() || !Fn->willReturn();
@@ -499,6 +502,7 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
     for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
       AssertOperandType(Idx, Op0Ty);
     return Type::getVoidTy(Ctx);
+  case Instruction::Ret:
   case Instruction::Store:
     return Type::getVoidTy(Ctx);
   case Instruction::ICmp:
@@ -575,6 +579,7 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
   case VPInstruction::FirstActiveLane:
   case VPInstruction::LastActiveLane:
   case VPInstruction::NumActiveLanes:
+  case VPInstruction::LiveIn:
   case VPInstruction::IncomingAliasMask:
   case Instruction::Load:
   case Instruction::Alloca:
@@ -644,6 +649,8 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case VPInstruction::IncomingAliasMask:
     return 0;
   case Instruction::Alloca:
+  case VPInstruction::LiveIn:
+  case Instruction::Ret:
   case Instruction::ExtractValue:
   case Instruction::Freeze:
   case Instruction::Load:
@@ -882,6 +889,18 @@ Value *VPInstruction::generate(VPTransformState &State) {
         {AVL, VFArg, Builder.getTrue()});
     return EVL;
   }
+  case VPInstruction::LiveIn: {
+    Argument *Arg = Builder.GetInsertBlock()->getParent()->getArg(
+        cast<VPConstantInt>(getOperand(0))->getZExtValue());
+    Arg->setName(getName());
+    return Arg;
+  }
+  case Instruction::Ret: {
+    auto *Ret = Builder.CreateRet(State.get(getOperand(0), /*IsScalar=*/true));
+    // Remove the temporary unreachable terminator.
+    Builder.GetInsertBlock()->getTerminator()->eraseFromParent();
+    return Ret;
+  }
   case VPInstruction::BranchOnCond: {
     Value *Cond = State.get(getOperand(0), VPLane(0));
     // Replace the temporary unreachable terminator with a new conditional
@@ -1582,6 +1601,7 @@ bool VPInstruction::isSingleScalar() const {
   switch (getOpcode()) {
   case Instruction::Load:
   case Instruction::PHI:
+  case VPInstruction::LiveIn:
   case VPInstruction::ExplicitVectorLength:
   case VPInstruction::ResumeForEpilogue:
   case VPInstruction::Intrinsic:
@@ -1669,6 +1689,7 @@ bool VPInstruction::opcodeMayReadOrWriteFromMemory() const {
       Instruction::isUnaryOp(getOpcode()) || Instruction::isCast(getOpcode()))
     return false;
   switch (getOpcode()) {
+  case Instruction::Ret:
   case Instruction::ExtractValue:
   case Instruction::InsertValue:
   case Instruction::GetElementPtr:
@@ -1694,6 +1715,7 @@ bool VPInstruction::opcodeMayReadOrWriteFromMemory() const {
   case VPInstruction::ExtractPenultimateElement:
   case VPInstruction::ActiveLaneMask:
   case VPInstruction::WideActiveLaneMask:
+  case VPInstruction::LiveIn:
   case VPInstruction::IncomingAliasMask:
   case VPInstruction::ExitingIVValue:
   case VPInstruction::ExplicitVectorLength:
@@ -1739,7 +1761,9 @@ bool VPInstruction::usesFirstLaneOnly(const VPValue *Op) const {
     return Op == getOperand(1);
   case Instruction::InsertElement:
     return Op == getOperand(1) || Op == getOperand(2);
+  case Instruction::Ret:
   case Instruction::PHI:
+  case VPInstruction::LiveIn:
     return true;
   case Instruction::FCmp:
   case Instruction::ICmp:
@@ -1826,6 +1850,9 @@ void VPInstruction::printRecipe(raw_ostream &O, const Twine &Indent,
   case VPInstruction::WideActiveLaneMask:
     O << "wide active lane mask";
     break;
+  case VPInstruction::LiveIn:
+    O << "live-in";
+    break;
   case VPInstruction::IncomingAliasMask:
     O << "incoming-alias-mask";
     break;
@@ -3605,6 +3632,79 @@ InstructionCost VPReductionRecipe::computeCost(ElementCount VF,
                                             Ctx.CostKind);
 }
 
+VPSpeculativeLoadOracleRecipe::VPSpeculativeLoadOracleRecipe(
+    std::unique_ptr<VPlan> Oracle, uint64_t AccessSize,
+    ArrayRef<VPValue *> Args, std::unique_ptr<VPSymbolicValue> BaseIndex)
+    : VPSingleDefRecipe(VPRecipeBase::VPSpeculativeLoadOracleSC, Args,
+                        PointerType::getUnqual(Oracle->getContext())),
+      OraclePlan(std::move(Oracle)), AccessSize(AccessSize),
+      BaseIndex(std::move(BaseIndex)) {}
+
+VPSpeculativeLoadOracleRecipe *VPSpeculativeLoadOracleRecipe::clone() {
+  assert((!BaseIndex || BaseIndex->isMaterialized()) &&
+         "must materialize the base index before cloning");
+  return new VPSpeculativeLoadOracleRecipe(
+      std::unique_ptr<VPlan>(OraclePlan->duplicate()), AccessSize, operands());
+}
+
+void VPSpeculativeLoadOracleRecipe::execute(VPTransformState &State) {
+  VPlan &Plan = *OraclePlan;
+  LLVMContext &Ctx = Plan.getContext();
+  auto ParamTypes = map_to_vector(
+      operands(), [](VPValue *Op) { return Op->getScalarType(); });
+
+  Module *M = State.Builder.GetInsertBlock()->getModule();
+  Function *OracleFn = Function::Create(
+      FunctionType::get(Type::getInt64Ty(Ctx), ParamTypes, /*isVarArg=*/false),
+      GlobalValue::InternalLinkage, "speculativeLoadOracle", M);
+  OracleFn->setMemoryEffects(MemoryEffects::argMemOnly(ModRefInfo::Ref));
+  OracleFn->addFnAttr(Attribute::NoInline);
+  OracleFn->addFnAttr(Attribute::OptimizeNone);
+  OracleFn->addFnAttr(Attribute::NoUnwind);
+  OracleFn->addFnAttr(Attribute::NoSync);
+  OracleFn->addFnAttr(Attribute::MustProgress);
+  OracleFn->addFnAttr(Attribute::WillReturn);
+
+  BasicBlock *EntryBB = BasicBlock::Create(Ctx, "oracle.entry", OracleFn);
+  IRBuilder<> OracleBuilder(EntryBB);
+  OracleBuilder.SetInsertPoint(OracleBuilder.CreateUnreachable());
+
+  // Replace the VF placeholder with the runtime lane count.
+  Plan.getVF().replaceAllUsesWith(Plan.getOrAddLiveIn(
+      getRuntimeVF(OracleBuilder, Plan.getVF().getType(), State.VF)));
+
+  // Execute the plan with VF=1.
+  DominatorTree OracleDT(*OracleFn);
+  LoopInfo OracleLI(OracleDT);
+  VPTransformState OracleState(State.TTI, ElementCount::getFixed(1), &OracleLI,
+                               /*DT=*/nullptr, /*AC=*/nullptr, OracleBuilder,
+                               &Plan, /*CurrentParentLoop=*/nullptr);
+  OracleState.CFG.PrevBB = EntryBB;
+  OracleState.CFG.VPBB2IRBB[Plan.getEntry()] = EntryBB;
+
+  for (VPRecipeBase &R : *Plan.getEntry())
+    R.execute(OracleState);
+  ReversePostOrderTraversal<VPBlockShallowTraversalWrapper<VPBlockBase *>> RPOT(
+      Plan.getEntry());
+  for (VPBlockBase *Block : drop_begin(RPOT))
+    Block->execute(&OracleState);
+  OracleState.fixupHeaderPhis();
+
+  State.set(this, OracleFn, /*IsScalar=*/true);
+}
+
+#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
+void VPSpeculativeLoadOracleRecipe::printRecipe(raw_ostream &O,
+                                                const Twine &Indent,
+                                                VPSlotTracker &Tracker) const {
+  O << Indent << "SPECULATIVE-LOAD-ORACLE ";
+  printAsOperand(O, Tracker);
+  O << " = fn(";
+  printOperands(O, Tracker);
+  O << ")";
+}
+#endif
+
 VPExpressionRecipe::VPExpressionRecipe(
     ExpressionTypes ExpressionType,
     ArrayRef<VPSingleDefRecipe *> ExpressionRecipes)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index a2f2a2997086e..96ff129f196c5 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -3318,11 +3318,10 @@ bool VPlanTransforms::handleUncountableEarlyExits(
   auto *MiddleVPBB = VPBlockUtils::getPlainCFGMiddleBlock(Plan);
   auto [HeaderVPBB, LatchVPBB] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
 
-  // Dereferenceability is checked separately for uncountable exit loops with
-  // stores, as only the loads contributing to the exit condition need to
-  // be checked.
+  // Dereferenceability is checked separately for uncountable exit loops without
+  // stores; loads that may fault are replaced by speculative loads.
   if (Style == UncountableExitStyle::ReadOnly &&
-      !areAllLoadsDereferenceable(HeaderVPBB, TheLoop, PSE, DT, AC))
+      !replaceUnsafeLoadsWithSpeculative(Plan, TheLoop, PSE, DT, AC))
     return false;
 
   VPBuilder LatchBuilder(LatchVPBB->getTerminator());
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 6e5a2184270a5..dfb170826e0e9 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -236,6 +236,12 @@ struct VPlanTransforms {
   static void attachCheckBlock(VPlan &Plan, Value *Cond, BasicBlock *CheckBlock,
                                bool AddBranchWeights);
 
+  /// Attach @llvm.can.load.speculatively checks for \p Plan's speculative
+  /// loads, bypassing the vector loop if any fails.
+  static void attachSpeculativeLoadChecks(VPlan &Plan, ElementCount VF,
+                                          PredicatedScalarEvolution &PSE,
+                                          Loop *TheLoop, bool AddBranchWeights);
+
   /// Replaces the VPInstructions in \p Plan with corresponding
   /// widen recipes. Returns false if any VPInstructions could not be converted
   /// to a wide recipe if needed. Uses \p PSE to detect contiguous memory
@@ -373,14 +379,13 @@ struct VPlanTransforms {
   /// Remove dead recipes from \p Plan.
   static void removeDeadRecipes(VPlan &Plan);
 
-  /// Check if all loads in the loop are dereferenceable. Iterates over the
-  /// loop body blocks reachable from \p HeaderVPBB. Returns false if any
-  /// non-dereferenceable load is found.
-  static bool areAllLoadsDereferenceable(VPBasicBlock *HeaderVPBB,
-                                         Loop *TheLoop,
-                                         PredicatedScalarEvolution &PSE,
-                                         DominatorTree &DT,
-                                         AssumptionCache *AC);
+  /// Replace loads that may fault with @llvm.speculative.load, backed by an
+  /// oracle plan replaying \p Plan's exit conditions. Must run before the early
+  /// exits are flattened. Returns false if a load cannot be replaced.
+  static bool replaceUnsafeLoadsWithSpeculative(VPlan &Plan, Loop *TheLoop,
+                                                PredicatedScalarEvolution &PSE,
+                                                DominatorTree &DT,
+                                                AssumptionCache *AC);
 
   /// Update \p Plan to account for uncountable early exits by introducing
   /// appropriate branching logic in the latch that handles early exits and the
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index ea07220c279c2..27b8ac04ff56d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -616,6 +616,13 @@ VPValue *vputils::findIncomingAliasMask(const VPlan &Plan) {
   return nullptr;
 }
 
+VPSpeculativeLoadOracleRecipe *vputils::findSpeculativeLoadOracle(VPlan &Plan) {
+  for (VPRecipeBase &R : *Plan.getVectorLoopRegion()->getEntryBasicBlock())
+    if (auto *Oracle = dyn_cast<VPSpeculativeLoadOracleRecipe>(&R))
+      return Oracle;
+  return nullptr;
+}
+
 SmallVector<std::pair<VPBasicBlock *, VPIRBasicBlock *>>
 vputils::getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB) {
   SmallVector<std::pair<VPBasicBlock *, VPIRBasicBlock *>> Exits;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.h b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
index 913fab53223c9..c77ceb7e36ecc 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
@@ -192,6 +192,10 @@ VPInstruction *findComputeReductionResult(VPReductionPHIRecipe *PhiR);
 /// Finds the incoming alias-mask within the vector preheader.
 VPValue *findIncomingAliasMask(const VPlan &Plan);
 
+/// Finds the speculative-load oracle in \p Plan's vector loop header, if any.
+/// Requires loop regions to be created.
+VPSpeculativeLoadOracleRecipe *findSpeculativeLoadOracle(VPlan &Plan);
+
 /// Returns the (early exiting block, exit block) pairs of \p Plan, i.e. all
 /// edges to an exit block that do not come from \p MiddleVPBB.
 SmallVector<std::pair<VPBasicBlock *, VPIRBasicBlock *>>
diff --git a/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp b/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
index e36ad81cfae2a..4c9449a3ac0f1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
@@ -321,12 +321,42 @@ bool VPlanVerifier::verifyVPBasicBlock(const VPBasicBlock *VPBB) {
         return false;
       }
     }
+    if (const auto *Oracle = dyn_cast<VPSpeculativeLoadOracleRecipe>(&R)) {
+      if (!verifyVPlanIsValid(Oracle->getOraclePlan())) {
+        errs() << "Invalid speculative-load oracle plan!\n";
+        return false;
+      }
+    }
     if (const auto *VPI = dyn_cast<VPInstruction>(&R)) {
       switch (VPI->getOpcode()) {
+      case Instruction::Ret:
+        if (&R != &VPBB->back() || VPBB->getParent() ||
+            VPBB->getNumSuccessors() != 0) {
+          errs() << "Return must terminate a top-level block without "
+                    "successors!\n";
+          return false;
+        }
+        break;
       case VPInstruction::LastActiveLane:
         if (!verifyLastActiveLaneRecipe(*VPI))
           return false;
         break;
+      case VPInstruction::LiveIn: {
+        if (VPBB != VPBB->getPlan()->getEntry()) {
+          errs() << "Live-in must be in the plan's entry block!\n";
+          return false;
+        }
+        uint64_t Idx = cast<VPConstantInt>(VPI->getOperand(0))->getZExtValue();
+        if (any_of(make_range(VPBB->begin(), VPI->getIterator()),
+                   [Idx](const VPRecipeBase &R) {
+                     return match(&R, m_VPInstruction<VPInstruction::LiveIn>(
+                                          m_SpecificInt(Idx)));
+                   })) {
+          errs() << "Multiple live-ins with index " << Idx << "!\n";
+          return false;
+        }
+        break;
+      }
       default:
         break;
       }
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-basic.ll b/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-basic.ll
index 8ca8b9ea7737c..f01b8a4429714 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-basic.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-basic.ll
@@ -154,21 +154,58 @@ exit:
 ; CHECK-LABEL: define i64 @find_first_mismatch(
 ; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[SPEC_LOAD_CHECK:.*]]
+; CHECK:       [[SPEC_LOAD_CHECK]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[A]], i64 4)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[B]], i64 4)
+; CHECK-NEXT:    [[TMP2:%.*]] = and i1 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = xor i1 [[TMP2]], true
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP4]]
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT:    [[LD_A:%.*]] = load i8, ptr [[GEP_A]], align 1
-; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP6:%.*]] = call <4 x i8> (ptr, i1, ...) @llvm.speculative.load.v4i8.p0(ptr [[GEP_A]], i1 false, ptr @speculativeLoadOracle, i64 [[IV]], ptr [[A]], ptr [[B]])
+; CHECK-NEXT:    [[TMP7:%.*]] = freeze <4 x i8> [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP9:%.*]] = call <4 x i8> (ptr, i1, ...) @llvm.speculative.load.v4i8.p0(ptr [[TMP8]], i1 false, ptr @speculativeLoadOracle, i64 [[IV]], ptr [[A]], ptr [[B]])
+; CHECK-NEXT:    [[TMP10:%.*]] = freeze <4 x i8> [[TMP9]]
+; CHECK-NEXT:    [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = freeze <4 x i1> [[TMP11]]
+; CHECK-NEXT:    [[TMP13:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP12]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP14:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP14]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    [[TMP15:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP11]], i1 false)
+; CHECK-NEXT:    [[TMP16:%.*]] = add i64 [[IV]], [[TMP15]]
+; CHECK-NEXT:    br label %[[EARLY_EXIT:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[SPEC_LOAD_CHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
+; CHECK:       [[LOOP_HEADER]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_A1:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV1]]
+; CHECK-NEXT:    [[LD_A:%.*]] = load i8, ptr [[GEP_A1]], align 1
+; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV1]]
 ; CHECK-NEXT:    [[LD_B:%.*]] = load i8, ptr [[GEP_B]], align 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i8 [[LD_A]], [[LD_B]]
-; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT]], label %[[LOOP_LATCH]]
 ; CHECK:       [[LOOP_LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
 ; CHECK:       [[EARLY_EXIT]]:
-; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[LOOP]] ]
+; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV1]], %[[LOOP_HEADER]] ], [ [[TMP16]], %[[VECTOR_EARLY_EXIT]] ]
 ; CHECK-NEXT:    ret i64 [[IV_LCSSA]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret i64 -1
@@ -177,19 +214,54 @@ exit:
 ; CHECK-LABEL: define i64 @find_first_mismatch_no_live_out(
 ; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[SPEC_LOAD_CHECK:.*]]
+; CHECK:       [[SPEC_LOAD_CHECK]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[A]], i64 4)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[B]], i64 4)
+; CHECK-NEXT:    [[TMP2:%.*]] = and i1 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = xor i1 [[TMP2]], true
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP4]]
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT:    [[LD_A:%.*]] = load i8, ptr [[GEP_A]], align 1
-; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP6:%.*]] = call <4 x i8> (ptr, i1, ...) @llvm.speculative.load.v4i8.p0(ptr [[GEP_A]], i1 false, ptr @speculativeLoadOracle.1, i64 [[IV]], ptr [[A]], ptr [[B]])
+; CHECK-NEXT:    [[TMP7:%.*]] = freeze <4 x i8> [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP9:%.*]] = call <4 x i8> (ptr, i1, ...) @llvm.speculative.load.v4i8.p0(ptr [[TMP8]], i1 false, ptr @speculativeLoadOracle.1, i64 [[IV]], ptr [[A]], ptr [[B]])
+; CHECK-NEXT:    [[TMP10:%.*]] = freeze <4 x i8> [[TMP9]]
+; CHECK-NEXT:    [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = freeze <4 x i1> [[TMP11]]
+; CHECK-NEXT:    [[TMP13:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP12]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP14:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP14]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    br label %[[EARLY_EXIT:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[SPEC_LOAD_CHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
+; CHECK:       [[LOOP_HEADER]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_A1:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV1]]
+; CHECK-NEXT:    [[LD_A:%.*]] = load i8, ptr [[GEP_A1]], align 1
+; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV1]]
 ; CHECK-NEXT:    [[LD_B:%.*]] = load i8, ptr [[GEP_B]], align 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i8 [[LD_A]], [[LD_B]]
-; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT]], label %[[LOOP_LATCH]]
 ; CHECK:       [[LOOP_LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
 ; CHECK:       [[EARLY_EXIT]]:
 ; CHECK-NEXT:    ret i64 1
 ; CHECK:       [[EXIT]]:
@@ -199,38 +271,112 @@ exit:
 ; CHECK-LABEL: define i64 @early_and_latch_exit_to_same_block(
 ; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[SPEC_LOAD_CHECK:.*]]
+; CHECK:       [[SPEC_LOAD_CHECK]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[A]], i64 4)
+; CHECK-NEXT:    [[TMP1:%.*]] = xor i1 [[TMP0]], true
+; CHECK-NEXT:    br i1 [[TMP1]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP2]]
 ; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
 ; CHECK:       [[LOOP_HEADER]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT:    [[LD_A:%.*]] = load i8, ptr [[GEP_A]], align 1
+; CHECK-NEXT:    [[TMP4:%.*]] = call <4 x i8> (ptr, i1, ...) @llvm.speculative.load.v4i8.p0(ptr [[GEP_A]], i1 false, ptr @speculativeLoadOracle.2, i64 [[IV]], ptr [[A]])
+; CHECK-NEXT:    [[TMP5:%.*]] = freeze <4 x i8> [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq <4 x i8> [[TMP5]], splat (i8 42)
+; CHECK-NEXT:    [[TMP7:%.*]] = freeze <4 x i1> [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP7]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    [[TMP10:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP6]], i1 false)
+; CHECK-NEXT:    [[TMP11:%.*]] = add i64 [[IV]], [[TMP10]]
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[SPEC_LOAD_CHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP_HEADER1:.*]]
+; CHECK:       [[LOOP_HEADER1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_A1:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV1]]
+; CHECK-NEXT:    [[LD_A:%.*]] = load i8, ptr [[GEP_A1]], align 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i8 [[LD_A]], 42
-; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT]], label %[[LOOP_LATCH]]
 ; CHECK:       [[LOOP_LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP_HEADER1]], label %[[EXIT]], !llvm.loop [[LOOP7:![0-9]+]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[RES:%.*]] = phi i64 [ [[IV]], %[[LOOP_HEADER]] ], [ -1, %[[LOOP_LATCH]] ]
+; CHECK-NEXT:    [[RES:%.*]] = phi i64 [ [[IV1]], %[[LOOP_HEADER1]] ], [ -1, %[[LOOP_LATCH]] ], [ -1, %[[MIDDLE_BLOCK]] ], [ [[TMP11]], %[[VECTOR_EARLY_EXIT]] ]
 ; CHECK-NEXT:    ret i64 [[RES]]
 ;
 ;
 ; CHECK-LABEL: define ptr @address_live_out(
 ; CHECK-SAME: ptr [[P:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[SPEC_LOAD_CHECK:.*]]
+; CHECK:       [[SPEC_LOAD_CHECK]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[P]], i64 4)
+; CHECK-NEXT:    [[TMP1:%.*]] = xor i1 [[TMP0]], true
+; CHECK-NEXT:    br i1 [[TMP1]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP2]]
 ; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
 ; CHECK:       [[LOOP_HEADER]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[TMP4:%.*]] = add i64 [[IV]], 2
+; CHECK-NEXT:    [[TMP5:%.*]] = add i64 [[IV]], 3
 ; CHECK-NEXT:    [[GEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[IV]]
-; CHECK-NEXT:    [[L:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP5]]
+; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <4 x ptr> poison, ptr [[GEP]], i64 0
+; CHECK-NEXT:    [[TMP11:%.*]] = insertelement <4 x ptr> [[TMP10]], ptr [[TMP7]], i64 1
+; CHECK-NEXT:    [[TMP12:%.*]] = insertelement <4 x ptr> [[TMP11]], ptr [[TMP8]], i64 2
+; CHECK-NEXT:    [[TMP13:%.*]] = insertelement <4 x ptr> [[TMP12]], ptr [[TMP9]], i64 3
+; CHECK-NEXT:    [[TMP14:%.*]] = call <4 x i8> (ptr, i1, ...) @llvm.speculative.load.v4i8.p0(ptr [[GEP]], i1 false, ptr @speculativeLoadOracle.3, i64 [[IV]], ptr [[P]])
+; CHECK-NEXT:    [[TMP15:%.*]] = freeze <4 x i8> [[TMP14]]
+; CHECK-NEXT:    [[TMP16:%.*]] = icmp eq <4 x i8> [[TMP15]], zeroinitializer
+; CHECK-NEXT:    [[TMP17:%.*]] = freeze <4 x i1> [[TMP16]]
+; CHECK-NEXT:    [[TMP18:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP17]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP19:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP18]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP19]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    [[FIRST_ACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP16]], i1 false)
+; CHECK-NEXT:    [[TMP20:%.*]] = extractelement <4 x ptr> [[TMP13]], i64 [[FIRST_ACTIVE_LANE]]
+; CHECK-NEXT:    br label %[[EARLY_EXIT:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[SPEC_LOAD_CHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP_HEADER1:.*]]
+; CHECK:       [[LOOP_HEADER1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[IV1]]
+; CHECK-NEXT:    [[L:%.*]] = load i8, ptr [[GEP1]], align 1
 ; CHECK-NEXT:    [[C:%.*]] = icmp eq i8 [[L]], 0
-; CHECK-NEXT:    br i1 [[C]], label %[[EARLY_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK-NEXT:    br i1 [[C]], label %[[EARLY_EXIT]], label %[[LOOP_LATCH]]
 ; CHECK:       [[LOOP_LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV1]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP_HEADER]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP_HEADER1]], !llvm.loop [[LOOP9:![0-9]+]]
 ; CHECK:       [[EARLY_EXIT]]:
-; CHECK-NEXT:    [[GEP_LCSSA:%.*]] = phi ptr [ [[GEP]], %[[LOOP_HEADER]] ]
+; CHECK-NEXT:    [[GEP_LCSSA:%.*]] = phi ptr [ [[GEP1]], %[[LOOP_HEADER1]] ], [ [[TMP20]], %[[VECTOR_EARLY_EXIT]] ]
 ; CHECK-NEXT:    ret ptr [[GEP_LCSSA]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret ptr null
@@ -239,20 +385,63 @@ exit:
 ; CHECK-LABEL: define i64 @index_cast_live_out(
 ; CHECK-SAME: ptr [[P:%.*]], i32 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[SPEC_LOAD_CHECK:.*]]
+; CHECK:       [[SPEC_LOAD_CHECK]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[P]], i64 4)
+; CHECK-NEXT:    [[TMP1:%.*]] = xor i1 [[TMP0]], true
+; CHECK-NEXT:    br i1 [[TMP1]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = and i32 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[N]], [[TMP2]]
 ; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
 ; CHECK:       [[LOOP_HEADER]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = add i32 [[IV]], 1
+; CHECK-NEXT:    [[TMP4:%.*]] = add i32 [[IV]], 2
+; CHECK-NEXT:    [[TMP5:%.*]] = add i32 [[IV]], 3
 ; CHECK-NEXT:    [[IDX:%.*]] = zext i32 [[IV]] to i64
+; CHECK-NEXT:    [[TMP7:%.*]] = zext i32 [[TMP3]] to i64
+; CHECK-NEXT:    [[TMP8:%.*]] = zext i32 [[TMP4]] to i64
+; CHECK-NEXT:    [[TMP9:%.*]] = zext i32 [[TMP5]] to i64
+; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <4 x i64> poison, i64 [[IDX]], i64 0
+; CHECK-NEXT:    [[TMP11:%.*]] = insertelement <4 x i64> [[TMP10]], i64 [[TMP7]], i64 1
+; CHECK-NEXT:    [[TMP12:%.*]] = insertelement <4 x i64> [[TMP11]], i64 [[TMP8]], i64 2
+; CHECK-NEXT:    [[TMP13:%.*]] = insertelement <4 x i64> [[TMP12]], i64 [[TMP9]], i64 3
 ; CHECK-NEXT:    [[GEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[IDX]]
-; CHECK-NEXT:    [[L:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[TMP15:%.*]] = call <4 x i8> (ptr, i1, ...) @llvm.speculative.load.v4i8.p0(ptr [[GEP]], i1 false, ptr @speculativeLoadOracle.4, i32 [[IV]], ptr [[P]])
+; CHECK-NEXT:    [[TMP16:%.*]] = freeze <4 x i8> [[TMP15]]
+; CHECK-NEXT:    [[TMP17:%.*]] = icmp eq <4 x i8> [[TMP16]], zeroinitializer
+; CHECK-NEXT:    [[TMP18:%.*]] = freeze <4 x i1> [[TMP17]]
+; CHECK-NEXT:    [[TMP19:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP18]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[IV]], 4
+; CHECK-NEXT:    [[TMP20:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP19]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP20]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    [[FIRST_ACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP17]], i1 false)
+; CHECK-NEXT:    [[TMP21:%.*]] = extractelement <4 x i64> [[TMP13]], i64 [[FIRST_ACTIVE_LANE]]
+; CHECK-NEXT:    br label %[[EARLY_EXIT:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[SPEC_LOAD_CHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP_HEADER1:.*]]
+; CHECK:       [[LOOP_HEADER1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[IDX1:%.*]] = zext i32 [[IV1]] to i64
+; CHECK-NEXT:    [[GEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[IDX1]]
+; CHECK-NEXT:    [[L:%.*]] = load i8, ptr [[GEP1]], align 1
 ; CHECK-NEXT:    [[C:%.*]] = icmp eq i8 [[L]], 0
-; CHECK-NEXT:    br i1 [[C]], label %[[EARLY_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK-NEXT:    br i1 [[C]], label %[[EARLY_EXIT]], label %[[LOOP_LATCH]]
 ; CHECK:       [[LOOP_LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV1]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp eq i32 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP_HEADER]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP_HEADER1]], !llvm.loop [[LOOP11:![0-9]+]]
 ; CHECK:       [[EARLY_EXIT]]:
-; CHECK-NEXT:    [[IDX_LCSSA:%.*]] = phi i64 [ [[IDX]], %[[LOOP_HEADER]] ]
+; CHECK-NEXT:    [[IDX_LCSSA:%.*]] = phi i64 [ [[IDX1]], %[[LOOP_HEADER1]] ], [ [[TMP21]], %[[VECTOR_EARLY_EXIT]] ]
 ; CHECK-NEXT:    ret i64 [[IDX_LCSSA]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret i64 -1
@@ -260,25 +449,150 @@ exit:
 ;
 ; CHECK-LABEL: define i32 @masked_index_broadcast(
 ; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
-; CHECK-NEXT:  [[ENTRY:.*]]:
-; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
-; CHECK:       [[LOOP_HEADER]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
-; CHECK-NEXT:    [[ADD:%.*]] = add i64 [[IV]], [[N]]
-; CHECK-NEXT:    [[CMP:%.*]] = icmp ugt i64 [[ADD]], 0
+; CHECK-NEXT:  [[SCALAR_PH:.*]]:
+; CHECK-NEXT:    br label %[[LOOP_HEADER1:.*]]
+; CHECK:       [[LOOP_HEADER1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[ADD1:%.*]] = add i64 [[IV1]], [[N]]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ugt i64 [[ADD1]], 0
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP_LATCH]], label %[[BODY:.*]]
 ; CHECK:       [[BODY]]:
-; CHECK-NEXT:    [[IDX:%.*]] = and i64 [[ADD]], 4294967295
+; CHECK-NEXT:    [[IDX:%.*]] = and i64 [[ADD1]], 4294967295
 ; CHECK-NEXT:    [[GEP:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[IDX]]
 ; CHECK-NEXT:    [[L:%.*]] = load i64, ptr [[GEP]], align 8
 ; CHECK-NEXT:    [[C:%.*]] = icmp eq i64 [[L]], 0
 ; CHECK-NEXT:    br i1 [[C]], label %[[LOOP_LATCH]], label %[[EARLY_EXIT:.*]]
 ; CHECK:       [[LOOP_LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
-; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP_HEADER]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV1]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV1]], [[N]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP_HEADER1]]
 ; CHECK:       [[EARLY_EXIT]]:
 ; CHECK-NEXT:    ret i32 1
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret i32 0
 ;
+;
+; CHECK-LABEL: define internal i64 @speculativeLoadOracle(
+; CHECK-SAME: i64 [[BASE_INDEX:%.*]], ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT:  [[ORACLE_ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
+; CHECK:       [[LOOP_HEADER]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ORACLE_ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[ADJUSTED_IV:%.*]] = add i64 [[INDEX]], [[BASE_INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load i8, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP5:%.*]] = load i8, ptr [[TMP4]], align 1
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp ne i8 [[TMP3]], [[TMP5]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[ORACLE_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[ORACLE_EXIT]], label %[[LOOP_HEADER]]
+; CHECK:       [[ORACLE_EXIT]]:
+; CHECK-NEXT:    [[LANES:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    ret i64 [[LANES]]
+;
+;
+; CHECK-LABEL: define internal i64 @speculativeLoadOracle.1(
+; CHECK-SAME: i64 [[BASE_INDEX:%.*]], ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR1]] {
+; CHECK-NEXT:  [[ORACLE_ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
+; CHECK:       [[LOOP_HEADER]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ORACLE_ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[ADJUSTED_IV:%.*]] = add i64 [[INDEX]], [[BASE_INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load i8, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP5:%.*]] = load i8, ptr [[TMP4]], align 1
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp ne i8 [[TMP3]], [[TMP5]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[ORACLE_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[ORACLE_EXIT]], label %[[LOOP_HEADER]]
+; CHECK:       [[ORACLE_EXIT]]:
+; CHECK-NEXT:    [[LANES:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    ret i64 [[LANES]]
+;
+;
+; CHECK-LABEL: define internal i64 @speculativeLoadOracle.2(
+; CHECK-SAME: i64 [[BASE_INDEX:%.*]], ptr [[A:%.*]]) #[[ATTR1]] {
+; CHECK-NEXT:  [[ORACLE_ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
+; CHECK:       [[LOOP_HEADER]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ORACLE_ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[ADJUSTED_IV:%.*]] = add i64 [[INDEX]], [[BASE_INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load i8, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i8 [[TMP3]], 42
+; CHECK-NEXT:    br i1 [[TMP4]], label %[[ORACLE_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[ORACLE_EXIT]], label %[[LOOP_HEADER]]
+; CHECK:       [[ORACLE_EXIT]]:
+; CHECK-NEXT:    [[LANES:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    ret i64 [[LANES]]
+;
+;
+; CHECK-LABEL: define internal i64 @speculativeLoadOracle.3(
+; CHECK-SAME: i64 [[BASE_INDEX:%.*]], ptr [[P:%.*]]) #[[ATTR1]] {
+; CHECK-NEXT:  [[ORACLE_ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
+; CHECK:       [[LOOP_HEADER]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ORACLE_ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[ADJUSTED_IV:%.*]] = add i64 [[INDEX]], [[BASE_INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr i8, ptr [[P]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load i8, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i8 [[TMP3]], 0
+; CHECK-NEXT:    br i1 [[TMP4]], label %[[ORACLE_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[ORACLE_EXIT]], label %[[LOOP_HEADER]]
+; CHECK:       [[ORACLE_EXIT]]:
+; CHECK-NEXT:    [[LANES:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    ret i64 [[LANES]]
+;
+;
+; CHECK-LABEL: define internal i64 @speculativeLoadOracle.4(
+; CHECK-SAME: i32 [[BASE_INDEX:%.*]], ptr [[P:%.*]]) #[[ATTR1]] {
+; CHECK-NEXT:  [[ORACLE_ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
+; CHECK:       [[LOOP_HEADER]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[ORACLE_ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[ADJUSTED_IV:%.*]] = add i32 [[INDEX]], [[BASE_INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = zext i32 [[ADJUSTED_IV]] to i64
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = load i8, ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i8 [[TMP4]], 0
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[ORACLE_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i32 [[INDEX_NEXT]], 4
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[ORACLE_EXIT]], label %[[LOOP_HEADER]]
+; CHECK:       [[ORACLE_EXIT]]:
+; CHECK-NEXT:    [[LANES:%.*]] = add i32 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP7:%.*]] = zext i32 [[LANES]] to i64
+; CHECK-NEXT:    ret i64 [[TMP7]]
+;
+;.
+; CHECK: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
+; CHECK: attributes #[[ATTR1]] = { mustprogress noinline nosync nounwind optnone willreturn memory(argmem: read) }
+; CHECK: attributes #[[ATTR2:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(argmem: read) }
+; CHECK: attributes #[[ATTR3:[0-9]+]] = { nocallback nocreateundeforpoison nofree nosync nounwind speculatable willreturn memory(none) }
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
+; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
+; CHECK: [[LOOP6]] = distinct !{[[LOOP6]], [[META1]], [[META2]]}
+; CHECK: [[LOOP7]] = distinct !{[[LOOP7]], [[META2]], [[META1]]}
+; CHECK: [[LOOP8]] = distinct !{[[LOOP8]], [[META1]], [[META2]]}
+; CHECK: [[LOOP9]] = distinct !{[[LOOP9]], [[META2]], [[META1]]}
+; CHECK: [[LOOP10]] = distinct !{[[LOOP10]], [[META1]], [[META2]]}
+; CHECK: [[LOOP11]] = distinct !{[[LOOP11]], [[META2]], [[META1]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-load-types.ll b/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-load-types.ll
index 757856f2fbeaf..d02c9287a8342 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-load-types.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-load-types.ll
@@ -124,21 +124,58 @@ exit:
 ; CHECK-LABEL: define i64 @load_i16(
 ; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[SPEC_LOAD_CHECK:.*]]
+; CHECK:       [[SPEC_LOAD_CHECK]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[A]], i64 8)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[B]], i64 8)
+; CHECK-NEXT:    [[TMP2:%.*]] = and i1 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = xor i1 [[TMP2]], true
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP4]]
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT:    [[LD_A:%.*]] = load i16, ptr [[GEP_A]], align 2
-; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP6:%.*]] = call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr [[GEP_A]], i1 false, ptr @speculativeLoadOracle, i64 [[IV]], ptr [[A]], ptr [[B]])
+; CHECK-NEXT:    [[TMP7:%.*]] = freeze <4 x i16> [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP9:%.*]] = call <4 x i16> (ptr, i1, ...) @llvm.speculative.load.v4i16.p0(ptr [[TMP8]], i1 false, ptr @speculativeLoadOracle, i64 [[IV]], ptr [[A]], ptr [[B]])
+; CHECK-NEXT:    [[TMP10:%.*]] = freeze <4 x i16> [[TMP9]]
+; CHECK-NEXT:    [[TMP11:%.*]] = icmp ne <4 x i16> [[TMP7]], [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = freeze <4 x i1> [[TMP11]]
+; CHECK-NEXT:    [[TMP13:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP12]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP14:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP14]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    [[TMP15:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP11]], i1 false)
+; CHECK-NEXT:    [[TMP16:%.*]] = add i64 [[IV]], [[TMP15]]
+; CHECK-NEXT:    br label %[[EARLY_EXIT:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[SPEC_LOAD_CHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP1:.*]]
+; CHECK:       [[LOOP1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_A1:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[IV1]]
+; CHECK-NEXT:    [[LD_A:%.*]] = load i16, ptr [[GEP_A1]], align 2
+; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[IV1]]
 ; CHECK-NEXT:    [[LD_B:%.*]] = load i16, ptr [[GEP_B]], align 2
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i16 [[LD_A]], [[LD_B]]
-; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT]], label %[[LOOP_LATCH]]
 ; CHECK:       [[LOOP_LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP1]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
 ; CHECK:       [[EARLY_EXIT]]:
-; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[LOOP]] ]
+; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV1]], %[[LOOP1]] ], [ [[TMP16]], %[[VECTOR_EARLY_EXIT]] ]
 ; CHECK-NEXT:    ret i64 [[IV_LCSSA]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret i64 -1
@@ -147,21 +184,58 @@ exit:
 ; CHECK-LABEL: define i64 @load_i32(
 ; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[SPEC_LOAD_CHECK:.*]]
+; CHECK:       [[SPEC_LOAD_CHECK]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[A]], i64 16)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[B]], i64 16)
+; CHECK-NEXT:    [[TMP2:%.*]] = and i1 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = xor i1 [[TMP2]], true
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP4]]
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT:    [[LD_A:%.*]] = load i32, ptr [[GEP_A]], align 4
-; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP6:%.*]] = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr [[GEP_A]], i1 false, ptr @speculativeLoadOracle.1, i64 [[IV]], ptr [[A]], ptr [[B]])
+; CHECK-NEXT:    [[TMP7:%.*]] = freeze <4 x i32> [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP9:%.*]] = call <4 x i32> (ptr, i1, ...) @llvm.speculative.load.v4i32.p0(ptr [[TMP8]], i1 false, ptr @speculativeLoadOracle.1, i64 [[IV]], ptr [[A]], ptr [[B]])
+; CHECK-NEXT:    [[TMP10:%.*]] = freeze <4 x i32> [[TMP9]]
+; CHECK-NEXT:    [[TMP11:%.*]] = icmp ne <4 x i32> [[TMP7]], [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = freeze <4 x i1> [[TMP11]]
+; CHECK-NEXT:    [[TMP13:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP12]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP14:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP14]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    [[TMP15:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP11]], i1 false)
+; CHECK-NEXT:    [[TMP16:%.*]] = add i64 [[IV]], [[TMP15]]
+; CHECK-NEXT:    br label %[[EARLY_EXIT:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[SPEC_LOAD_CHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP1:.*]]
+; CHECK:       [[LOOP1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_A1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV1]]
+; CHECK-NEXT:    [[LD_A:%.*]] = load i32, ptr [[GEP_A1]], align 4
+; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV1]]
 ; CHECK-NEXT:    [[LD_B:%.*]] = load i32, ptr [[GEP_B]], align 4
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i32 [[LD_A]], [[LD_B]]
-; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT]], label %[[LOOP_LATCH]]
 ; CHECK:       [[LOOP_LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP1]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
 ; CHECK:       [[EARLY_EXIT]]:
-; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[LOOP]] ]
+; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV1]], %[[LOOP1]] ], [ [[TMP16]], %[[VECTOR_EARLY_EXIT]] ]
 ; CHECK-NEXT:    ret i64 [[IV_LCSSA]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret i64 -1
@@ -170,20 +244,53 @@ exit:
 ; CHECK-LABEL: define i64 @load_double(
 ; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[SPEC_LOAD_CHECK:.*]]
+; CHECK:       [[SPEC_LOAD_CHECK]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[A]], i64 16)
+; CHECK-NEXT:    [[TMP1:%.*]] = xor i1 [[TMP0]], true
+; CHECK-NEXT:    br i1 [[TMP1]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = and i64 [[N]], 1
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP2]]
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds double, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT:    [[LD_A:%.*]] = load double, ptr [[GEP_A]], align 8
+; CHECK-NEXT:    [[TMP4:%.*]] = call <2 x double> (ptr, i1, ...) @llvm.speculative.load.v2f64.p0(ptr [[GEP_A]], i1 false, ptr @speculativeLoadOracle.2, i64 [[IV]], ptr [[A]])
+; CHECK-NEXT:    [[TMP5:%.*]] = freeze <2 x double> [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = bitcast <2 x double> [[TMP5]] to <2 x i64>
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq <2 x i64> [[TMP6]], zeroinitializer
+; CHECK-NEXT:    [[TMP8:%.*]] = freeze <2 x i1> [[TMP7]]
+; CHECK-NEXT:    [[TMP9:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP8]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 2
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP9]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    [[TMP11:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v2i1(<2 x i1> [[TMP7]], i1 false)
+; CHECK-NEXT:    [[TMP12:%.*]] = add i64 [[IV]], [[TMP11]]
+; CHECK-NEXT:    br label %[[EARLY_EXIT:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[SPEC_LOAD_CHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP1:.*]]
+; CHECK:       [[LOOP1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_A1:%.*]] = getelementptr inbounds double, ptr [[A]], i64 [[IV1]]
+; CHECK-NEXT:    [[LD_A:%.*]] = load double, ptr [[GEP_A1]], align 8
 ; CHECK-NEXT:    [[BITS:%.*]] = bitcast double [[LD_A]] to i64
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[BITS]], 0
-; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT]], label %[[LOOP_LATCH]]
 ; CHECK:       [[LOOP_LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP1]], label %[[EXIT]], !llvm.loop [[LOOP7:![0-9]+]]
 ; CHECK:       [[EARLY_EXIT]]:
-; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[LOOP]] ]
+; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV1]], %[[LOOP1]] ], [ [[TMP12]], %[[VECTOR_EARLY_EXIT]] ]
 ; CHECK-NEXT:    ret i64 [[IV_LCSSA]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret i64 -1
@@ -229,3 +336,86 @@ exit:
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret i64 -1
 ;
+;
+; CHECK-LABEL: define internal i64 @speculativeLoadOracle(
+; CHECK-SAME: i64 [[BASE_INDEX:%.*]], ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT:  [[ORACLE_ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ORACLE_ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[ADJUSTED_IV:%.*]] = add i64 [[INDEX]], [[BASE_INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load i16, ptr [[TMP2]], align 2
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP5:%.*]] = load i16, ptr [[TMP4]], align 2
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp ne i16 [[TMP3]], [[TMP5]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[ORACLE_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[ORACLE_EXIT]], label %[[LOOP]]
+; CHECK:       [[ORACLE_EXIT]]:
+; CHECK-NEXT:    [[LANES:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    [[BYTES:%.*]] = shl i64 [[LANES]], 1
+; CHECK-NEXT:    ret i64 [[BYTES]]
+;
+;
+; CHECK-LABEL: define internal i64 @speculativeLoadOracle.1(
+; CHECK-SAME: i64 [[BASE_INDEX:%.*]], ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR1]] {
+; CHECK-NEXT:  [[ORACLE_ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ORACLE_ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[ADJUSTED_IV:%.*]] = add i64 [[INDEX]], [[BASE_INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load i32, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP5:%.*]] = load i32, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp ne i32 [[TMP3]], [[TMP5]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[ORACLE_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[ORACLE_EXIT]], label %[[LOOP]]
+; CHECK:       [[ORACLE_EXIT]]:
+; CHECK-NEXT:    [[LANES:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    [[BYTES:%.*]] = shl i64 [[LANES]], 2
+; CHECK-NEXT:    ret i64 [[BYTES]]
+;
+;
+; CHECK-LABEL: define internal i64 @speculativeLoadOracle.2(
+; CHECK-SAME: i64 [[BASE_INDEX:%.*]], ptr [[A:%.*]]) #[[ATTR1]] {
+; CHECK-NEXT:  [[ORACLE_ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ORACLE_ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[ADJUSTED_IV:%.*]] = add i64 [[INDEX]], [[BASE_INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds double, ptr [[A]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load double, ptr [[TMP2]], align 8
+; CHECK-NEXT:    [[TMP4:%.*]] = bitcast double [[TMP3]] to i64
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[TMP4]], 0
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[ORACLE_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 2
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[ORACLE_EXIT]], label %[[LOOP]]
+; CHECK:       [[ORACLE_EXIT]]:
+; CHECK-NEXT:    [[LANES:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    [[BYTES:%.*]] = shl i64 [[LANES]], 3
+; CHECK-NEXT:    ret i64 [[BYTES]]
+;
+;.
+; CHECK: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
+; CHECK: attributes #[[ATTR1]] = { mustprogress noinline nosync nounwind optnone willreturn memory(argmem: read) }
+; CHECK: attributes #[[ATTR2:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(argmem: read) }
+; CHECK: attributes #[[ATTR3:[0-9]+]] = { nocallback nocreateundeforpoison nofree nosync nounwind speculatable willreturn memory(none) }
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
+; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
+; CHECK: [[LOOP6]] = distinct !{[[LOOP6]], [[META1]], [[META2]]}
+; CHECK: [[LOOP7]] = distinct !{[[LOOP7]], [[META2]], [[META1]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-scalable.ll b/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-scalable.ll
index 827179661f997..915f24233091b 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-scalable.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/early-exit-with-speculative-load-scalable.ll
@@ -32,23 +32,84 @@ exit:
 ; CHECK-LABEL: define i64 @find_first_eq_const(
 ; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 16
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[SPEC_LOAD_CHECK:.*]]
+; CHECK:       [[SPEC_LOAD_CHECK]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[A]], i64 16)
+; CHECK-NEXT:    [[TMP1:%.*]] = xor i1 [[TMP0]], true
+; CHECK-NEXT:    br i1 [[TMP1]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = and i64 [[N]], 15
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP2]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP4:%.*]] = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr [[TMP3]], i1 false, ptr @speculativeLoadOracle, i64 [[INDEX]], ptr [[A]])
+; CHECK-NEXT:    [[TMP5:%.*]] = freeze <16 x i8> [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq <16 x i8> [[TMP5]], splat (i8 42)
+; CHECK-NEXT:    [[TMP7:%.*]] = freeze <16 x i1> [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP7]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT:    [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    [[TMP10:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v16i1(<16 x i1> [[TMP6]], i1 false)
+; CHECK-NEXT:    [[TMP11:%.*]] = add i64 [[INDEX]], [[TMP10]]
+; CHECK-NEXT:    br label %[[EARLY_EXIT:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[SPEC_LOAD_CHECK]] ]
 ; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
 ; CHECK:       [[LOOP_HEADER]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
 ; CHECK-NEXT:    [[LD_A:%.*]] = load i8, ptr [[GEP_A]], align 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i8 [[LD_A]], 42
-; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[EARLY_EXIT]], label %[[LOOP_LATCH]]
 ; CHECK:       [[LOOP_LATCH]]:
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT:.*]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
 ; CHECK:       [[EARLY_EXIT]]:
-; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[LOOP_HEADER]] ]
+; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[LOOP_HEADER]] ], [ [[TMP11]], %[[VECTOR_EARLY_EXIT]] ]
 ; CHECK-NEXT:    ret i64 [[IV_LCSSA]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret i64 -1
 ;
+;
+; CHECK-LABEL: define internal i64 @speculativeLoadOracle(
+; CHECK-SAME: i64 [[BASE_INDEX:%.*]], ptr [[A:%.*]]) #[[ATTR2:[0-9]+]] {
+; CHECK-NEXT:  [[ORACLE_ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
+; CHECK:       [[LOOP_HEADER]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ORACLE_ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[ADJUSTED_IV:%.*]] = add i64 [[INDEX]], [[BASE_INDEX]]
+; CHECK-NEXT:    [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[ADJUSTED_IV]]
+; CHECK-NEXT:    [[TMP1:%.*]] = load i8, ptr [[TMP0]], align 1
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i8 [[TMP1]], 42
+; CHECK-NEXT:    br i1 [[TMP2]], label %[[ORACLE_EXIT:.*]], label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 16
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[ORACLE_EXIT]], label %[[LOOP_HEADER]]
+; CHECK:       [[ORACLE_EXIT]]:
+; CHECK-NEXT:    [[LANES:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    ret i64 [[LANES]]
+;
 ;.
 ; CHECK: attributes #[[ATTR0]] = { "target-features"="+sve" }
+; CHECK: attributes #[[ATTR1:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
+; CHECK: attributes #[[ATTR2]] = { mustprogress noinline nosync nounwind optnone willreturn memory(argmem: read) }
+; CHECK: attributes #[[ATTR3:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(argmem: read) }
+; CHECK: attributes #[[ATTR4:[0-9]+]] = { nocallback nocreateundeforpoison nofree nosync nounwind speculatable willreturn memory(none) }
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
 ;.
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/early_exit_cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/early_exit_cost.ll
index 5879557b7c3a1..798eb64c8aaa1 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/early_exit_cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/early_exit_cost.ll
@@ -7,42 +7,108 @@ target triple = "arm64-apple-macosx"
 define i64 @early_exit_with_without_dereferenceable(ptr %p1, ptr %p2) {
 ; CHECK-LABEL: define i64 @early_exit_with_without_dereferenceable(
 ; CHECK-SAME: ptr [[P1:%.*]], ptr [[P2:%.*]]) {
-; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:  [[ENTRY:.*:]]
 ; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
 ; CHECK:       [[LOOP_HEADER]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[P1]], i64 16)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[P2]], i64 16)
+; CHECK-NEXT:    [[TMP2:%.*]] = and i1 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = xor i1 [[TMP2]], true
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
 ; CHECK-NEXT:    [[GEP_P1:%.*]] = getelementptr inbounds i8, ptr [[P1]], i64 [[IV]]
-; CHECK-NEXT:    [[LD1:%.*]] = load i8, ptr [[GEP_P1]], align 1
-; CHECK-NEXT:    [[GEP_P2:%.*]] = getelementptr inbounds i8, ptr [[P2]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP5:%.*]] = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr [[GEP_P1]], i1 false, ptr @speculativeLoadOracle, i64 [[IV]], ptr [[P1]], ptr [[P2]])
+; CHECK-NEXT:    [[TMP6:%.*]] = freeze <16 x i8> [[TMP5]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i8, ptr [[P2]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP8:%.*]] = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr [[TMP7]], i1 false, ptr @speculativeLoadOracle, i64 [[IV]], ptr [[P1]], ptr [[P2]])
+; CHECK-NEXT:    [[TMP9:%.*]] = freeze <16 x i8> [[TMP8]]
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp ne <16 x i8> [[TMP6]], [[TMP9]]
+; CHECK-NEXT:    [[TMP11:%.*]] = freeze <16 x i1> [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP11]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 16
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], 96
+; CHECK-NEXT:    br i1 [[TMP12]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    [[TMP14:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v16i1(<16 x i1> [[TMP10]], i1 false)
+; CHECK-NEXT:    [[TMP15:%.*]] = add i64 [[IV]], [[TMP14]]
+; CHECK-NEXT:    br label %[[EXIT:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ 96, %[[MIDDLE_BLOCK]] ], [ 0, %[[LOOP_HEADER]] ]
+; CHECK-NEXT:    br label %[[LOOP_HEADER1:.*]]
+; CHECK:       [[LOOP_HEADER1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_P3:%.*]] = getelementptr inbounds i8, ptr [[P1]], i64 [[IV1]]
+; CHECK-NEXT:    [[LD1:%.*]] = load i8, ptr [[GEP_P3]], align 1
+; CHECK-NEXT:    [[GEP_P2:%.*]] = getelementptr inbounds i8, ptr [[P2]], i64 [[IV1]]
 ; CHECK-NEXT:    [[LD2:%.*]] = load i8, ptr [[GEP_P2]], align 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i8 [[LD1]], [[LD2]]
-; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP_LATCH]], label %[[EXIT:.*]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP_LATCH]], label %[[EXIT]]
 ; CHECK:       [[LOOP_LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV1]], 1
 ; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp ne i64 [[IV_NEXT]], 100
-; CHECK-NEXT:    br i1 [[EXITCOND]], label %[[LOOP_HEADER]], label %[[EXIT]]
+; CHECK-NEXT:    br i1 [[EXITCOND]], label %[[LOOP_HEADER1]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[LOOP_LATCH]] ], [ [[IV]], %[[LOOP_HEADER]] ]
+; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV1]], %[[LOOP_LATCH]] ], [ [[IV1]], %[[LOOP_HEADER1]] ], [ [[TMP15]], %[[VECTOR_EARLY_EXIT]] ]
 ; CHECK-NEXT:    ret i64 [[IV_LCSSA]]
 ;
 ; MAX-BW-LABEL: define i64 @early_exit_with_without_dereferenceable(
 ; MAX-BW-SAME: ptr [[P1:%.*]], ptr [[P2:%.*]]) {
-; MAX-BW-NEXT:  [[ENTRY:.*]]:
+; MAX-BW-NEXT:  [[ENTRY:.*:]]
 ; MAX-BW-NEXT:    br label %[[LOOP_HEADER:.*]]
 ; MAX-BW:       [[LOOP_HEADER]]:
-; MAX-BW-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; MAX-BW-NEXT:    [[TMP0:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[P1]], i64 16)
+; MAX-BW-NEXT:    [[TMP1:%.*]] = call i1 @llvm.can.load.speculatively.p0(ptr [[P2]], i64 16)
+; MAX-BW-NEXT:    [[TMP2:%.*]] = and i1 [[TMP0]], [[TMP1]]
+; MAX-BW-NEXT:    [[TMP3:%.*]] = xor i1 [[TMP2]], true
+; MAX-BW-NEXT:    br i1 [[TMP3]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; MAX-BW:       [[VECTOR_PH]]:
+; MAX-BW-NEXT:    br label %[[VECTOR_BODY:.*]]
+; MAX-BW:       [[VECTOR_BODY]]:
+; MAX-BW-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
 ; MAX-BW-NEXT:    [[GEP_P1:%.*]] = getelementptr inbounds i8, ptr [[P1]], i64 [[IV]]
-; MAX-BW-NEXT:    [[LD1:%.*]] = load i8, ptr [[GEP_P1]], align 1
-; MAX-BW-NEXT:    [[GEP_P2:%.*]] = getelementptr inbounds i8, ptr [[P2]], i64 [[IV]]
+; MAX-BW-NEXT:    [[TMP5:%.*]] = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr [[GEP_P1]], i1 false, ptr @speculativeLoadOracle, i64 [[IV]], ptr [[P1]], ptr [[P2]])
+; MAX-BW-NEXT:    [[TMP6:%.*]] = freeze <16 x i8> [[TMP5]]
+; MAX-BW-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i8, ptr [[P2]], i64 [[IV]]
+; MAX-BW-NEXT:    [[TMP8:%.*]] = call <16 x i8> (ptr, i1, ...) @llvm.speculative.load.v16i8.p0(ptr [[TMP7]], i1 false, ptr @speculativeLoadOracle, i64 [[IV]], ptr [[P1]], ptr [[P2]])
+; MAX-BW-NEXT:    [[TMP9:%.*]] = freeze <16 x i8> [[TMP8]]
+; MAX-BW-NEXT:    [[TMP10:%.*]] = icmp ne <16 x i8> [[TMP6]], [[TMP9]]
+; MAX-BW-NEXT:    [[TMP11:%.*]] = freeze <16 x i1> [[TMP10]]
+; MAX-BW-NEXT:    [[TMP12:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP11]])
+; MAX-BW-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 16
+; MAX-BW-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], 96
+; MAX-BW-NEXT:    br i1 [[TMP12]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; MAX-BW:       [[VECTOR_BODY_INTERIM]]:
+; MAX-BW-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; MAX-BW:       [[MIDDLE_BLOCK]]:
+; MAX-BW-NEXT:    br label %[[SCALAR_PH]]
+; MAX-BW:       [[VECTOR_EARLY_EXIT]]:
+; MAX-BW-NEXT:    [[TMP14:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v16i1(<16 x i1> [[TMP10]], i1 false)
+; MAX-BW-NEXT:    [[TMP15:%.*]] = add i64 [[IV]], [[TMP14]]
+; MAX-BW-NEXT:    br label %[[EXIT:.*]]
+; MAX-BW:       [[SCALAR_PH]]:
+; MAX-BW-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ 96, %[[MIDDLE_BLOCK]] ], [ 0, %[[LOOP_HEADER]] ]
+; MAX-BW-NEXT:    br label %[[LOOP_HEADER1:.*]]
+; MAX-BW:       [[LOOP_HEADER1]]:
+; MAX-BW-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; MAX-BW-NEXT:    [[GEP_P3:%.*]] = getelementptr inbounds i8, ptr [[P1]], i64 [[IV1]]
+; MAX-BW-NEXT:    [[LD1:%.*]] = load i8, ptr [[GEP_P3]], align 1
+; MAX-BW-NEXT:    [[GEP_P2:%.*]] = getelementptr inbounds i8, ptr [[P2]], i64 [[IV1]]
 ; MAX-BW-NEXT:    [[LD2:%.*]] = load i8, ptr [[GEP_P2]], align 1
 ; MAX-BW-NEXT:    [[CMP:%.*]] = icmp eq i8 [[LD1]], [[LD2]]
-; MAX-BW-NEXT:    br i1 [[CMP]], label %[[LOOP_LATCH]], label %[[EXIT:.*]]
+; MAX-BW-NEXT:    br i1 [[CMP]], label %[[LOOP_LATCH]], label %[[EXIT]]
 ; MAX-BW:       [[LOOP_LATCH]]:
-; MAX-BW-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; MAX-BW-NEXT:    [[IV_NEXT]] = add i64 [[IV1]], 1
 ; MAX-BW-NEXT:    [[EXITCOND:%.*]] = icmp ne i64 [[IV_NEXT]], 100
-; MAX-BW-NEXT:    br i1 [[EXITCOND]], label %[[LOOP_HEADER]], label %[[EXIT]]
+; MAX-BW-NEXT:    br i1 [[EXITCOND]], label %[[LOOP_HEADER1]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
 ; MAX-BW:       [[EXIT]]:
-; MAX-BW-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[LOOP_LATCH]] ], [ [[IV]], %[[LOOP_HEADER]] ]
+; MAX-BW-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV1]], %[[LOOP_LATCH]] ], [ [[IV1]], %[[LOOP_HEADER1]] ], [ [[TMP15]], %[[VECTOR_EARLY_EXIT]] ]
 ; MAX-BW-NEXT:    ret i64 [[IV_LCSSA]]
 ;
 entry:
@@ -130,3 +196,14 @@ loop.latch:
 exit:
   ret i64 %iv
 }
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.
+; MAX-BW: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; MAX-BW: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; MAX-BW: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; MAX-BW: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/early-exit-with-speculative-load-oracle.ll b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/early-exit-with-speculative-load-oracle.ll
index 7ae491ba18af8..5f948b8dee5ae 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/early-exit-with-speculative-load-oracle.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/early-exit-with-speculative-load-oracle.ll
@@ -1,12 +1,101 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt -p loop-vectorize -force-vector-width=4 -disable-output \
-; RUN:     -vplan-print-after=printOptimizedVPlan %s 2>&1 | FileCheck %s --allow-empty
-; CHECK-NOT: VPlan for loop in
+; RUN:     -vplan-print-after=printOptimizedVPlan %s 2>&1 | FileCheck %s
 
 target triple = "arm64-apple-macosx"
 
 @G = external global [1024 x i8]
 
 define i64 @find_first_eq_const(ptr %A, i64 %n) {
+; CHECK-LABEL: VPlan for loop in 'find_first_eq_const'
+; CHECK:  VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT:  Live-in ir<%n> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<entry>:
+; CHECK-NEXT:  Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.ph:
+; CHECK-NEXT:  Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT:  <x1> vector loop: {
+; CHECK-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT:    vector.body:
+; CHECK-NEXT:      vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT:      SPECULATIVE-LOAD-ORACLE vp<[[VP5:%[0-9]+]]> = fn(vp<[[VP3]]>, ir<%A>)
+; CHECK-NEXT:      CLONE ir<%gep.A> = getelementptr inbounds ir<%A>, vp<[[VP4]]>
+; CHECK-NEXT:      WIDEN-INTRINSIC vp<[[VP6:%[0-9]+]]> = call llvm.speculative.load(ir<%gep.A>, ir<false>, vp<[[VP5]]>, vp<[[VP3]]>, ir<%A>)
+; CHECK-NEXT:      EMIT vp<[[VP7:%[0-9]+]]> = freeze vp<[[VP6]]>
+; CHECK-NEXT:      WIDEN ir<%cmp> = icmp eq vp<[[VP7]]>, ir<42>
+; CHECK-NEXT:      EMIT vp<[[VP8:%[0-9]+]]> = freeze ir<%cmp>
+; CHECK-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = any-of vp<[[VP8]]>
+; CHECK-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT:      EMIT branch-on-two-conds vp<[[VP9]]>, vp<[[VP10]]>
+; CHECK-NEXT:    No successors
+; CHECK-NEXT:  }
+; CHECK-NEXT:  Successor(s): vector.early.exit, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT:  middle.block:
+; CHECK-NEXT:    EMIT vp<%cmp.n> = icmp eq ir<%n>, vp<[[VP2]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<%cmp.n>
+; CHECK-NEXT:  Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<exit>:
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.early.exit:
+; CHECK-NEXT:    EMIT vp<[[VP13:%[0-9]+]]> = first-active-lane ir<%cmp>
+; CHECK-NEXT:    EMIT vp<[[VP14:%[0-9]+]]> = add vp<[[VP3]]>, vp<[[VP13]]>
+; CHECK-NEXT:  Successor(s): ir-bb<early.exit>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<early.exit>:
+; CHECK-NEXT:    IR   %iv.lcssa = phi i64 [ %iv, %loop.header ] (extra operand: vp<[[VP14]]> from vector.early.exit)
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  scalar.ph:
+; CHECK-NEXT:    EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT:  Successor(s): ir-bb<loop.header>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<loop.header>:
+; CHECK-NEXT:    IR   %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT:    IR   %gep.A = getelementptr inbounds i8, ptr %A, i64 %iv
+; CHECK-NEXT:    IR   %ld.A = load i8, ptr %gep.A, align 1
+; CHECK-NEXT:    IR   %cmp = icmp eq i8 %ld.A, 42
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
+; CHECK-EMPTY:
+; CHECK:  VPlan for speculative-load oracle vp<[[VP5]]> {
+; CHECK-NEXT:  Live-in vp<[[VP0]]> = VF
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.entry:
+; CHECK-NEXT:    EMIT-SCALAR vp<%base.index> = live-in ir<0>
+; CHECK-NEXT:    EMIT-SCALAR vp<%A> = live-in ir<1>
+; CHECK-NEXT:  Successor(s): loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.header:
+; CHECK-NEXT:    EMIT-SCALAR vp<%index> = phi [ ir<0>, oracle.entry ], [ vp<%index.next>, loop.latch ]
+; CHECK-NEXT:    EMIT vp<%adjusted.iv> = add vp<%index>, vp<%base.index>
+; CHECK-NEXT:    CLONE ir<%gep.A> = getelementptr inbounds vp<%A>, vp<%adjusted.iv>
+; CHECK-NEXT:    CLONE ir<%ld.A> = load ir<%gep.A>
+; CHECK-NEXT:    EMIT ir<%cmp> = icmp eq ir<%ld.A>, ir<42>
+; CHECK-NEXT:    EMIT branch-on-cond ir<%cmp> (!vplan.prof.estimated estimated {67108864, 2080374784})
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.latch
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.latch:
+; CHECK-NEXT:    EMIT vp<%index.next> = add nuw vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT vp<[[VP2]]> = icmp eq vp<%index.next>, vp<[[VP0]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<[[VP2]]>
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.exit:
+; CHECK-NEXT:    EMIT vp<%lanes> = add vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT ret vp<%lanes>
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
 ;
 entry:
   br label %loop.header
@@ -31,6 +120,96 @@ exit:
 }
 
 define i64 @find_first_eq_live_in(ptr %A, i64 %n, i8 %val) {
+; CHECK-LABEL: VPlan for loop in 'find_first_eq_live_in'
+; CHECK:  VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT:  Live-in ir<%n> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<entry>:
+; CHECK-NEXT:  Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.ph:
+; CHECK-NEXT:  Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT:  <x1> vector loop: {
+; CHECK-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT:    vector.body:
+; CHECK-NEXT:      vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT:      SPECULATIVE-LOAD-ORACLE vp<[[VP5:%[0-9]+]]> = fn(vp<[[VP3]]>, ir<%A>, ir<%val>)
+; CHECK-NEXT:      CLONE ir<%gep.A> = getelementptr inbounds ir<%A>, vp<[[VP4]]>
+; CHECK-NEXT:      WIDEN-INTRINSIC vp<[[VP6:%[0-9]+]]> = call llvm.speculative.load(ir<%gep.A>, ir<false>, vp<[[VP5]]>, vp<[[VP3]]>, ir<%A>, ir<%val>)
+; CHECK-NEXT:      EMIT vp<[[VP7:%[0-9]+]]> = freeze vp<[[VP6]]>
+; CHECK-NEXT:      WIDEN ir<%cmp> = icmp eq vp<[[VP7]]>, ir<%val>
+; CHECK-NEXT:      EMIT vp<[[VP8:%[0-9]+]]> = freeze ir<%cmp>
+; CHECK-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = any-of vp<[[VP8]]>
+; CHECK-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT:      EMIT branch-on-two-conds vp<[[VP9]]>, vp<[[VP10]]>
+; CHECK-NEXT:    No successors
+; CHECK-NEXT:  }
+; CHECK-NEXT:  Successor(s): vector.early.exit, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT:  middle.block:
+; CHECK-NEXT:    EMIT vp<%cmp.n> = icmp eq ir<%n>, vp<[[VP2]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<%cmp.n>
+; CHECK-NEXT:  Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<exit>:
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.early.exit:
+; CHECK-NEXT:    EMIT vp<[[VP13:%[0-9]+]]> = first-active-lane ir<%cmp>
+; CHECK-NEXT:    EMIT vp<[[VP14:%[0-9]+]]> = add vp<[[VP3]]>, vp<[[VP13]]>
+; CHECK-NEXT:  Successor(s): ir-bb<early.exit>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<early.exit>:
+; CHECK-NEXT:    IR   %iv.lcssa = phi i64 [ %iv, %loop.header ] (extra operand: vp<[[VP14]]> from vector.early.exit)
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  scalar.ph:
+; CHECK-NEXT:    EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT:  Successor(s): ir-bb<loop.header>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<loop.header>:
+; CHECK-NEXT:    IR   %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT:    IR   %gep.A = getelementptr inbounds i8, ptr %A, i64 %iv
+; CHECK-NEXT:    IR   %ld.A = load i8, ptr %gep.A, align 1
+; CHECK-NEXT:    IR   %cmp = icmp eq i8 %ld.A, %val
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
+; CHECK-EMPTY:
+; CHECK:  VPlan for speculative-load oracle vp<[[VP5]]> {
+; CHECK-NEXT:  Live-in vp<[[VP0]]> = VF
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.entry:
+; CHECK-NEXT:    EMIT-SCALAR vp<%base.index> = live-in ir<0>
+; CHECK-NEXT:    EMIT-SCALAR vp<%A> = live-in ir<1>
+; CHECK-NEXT:    EMIT-SCALAR vp<%val> = live-in ir<2>
+; CHECK-NEXT:  Successor(s): loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.header:
+; CHECK-NEXT:    EMIT-SCALAR vp<%index> = phi [ ir<0>, oracle.entry ], [ vp<%index.next>, loop.latch ]
+; CHECK-NEXT:    EMIT vp<%adjusted.iv> = add vp<%index>, vp<%base.index>
+; CHECK-NEXT:    CLONE ir<%gep.A> = getelementptr inbounds vp<%A>, vp<%adjusted.iv>
+; CHECK-NEXT:    CLONE ir<%ld.A> = load ir<%gep.A>
+; CHECK-NEXT:    EMIT ir<%cmp> = icmp eq ir<%ld.A>, vp<%val>
+; CHECK-NEXT:    EMIT branch-on-cond ir<%cmp> (!vplan.prof.estimated estimated {67108864, 2080374784})
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.latch
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.latch:
+; CHECK-NEXT:    EMIT vp<%index.next> = add nuw vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT vp<[[VP2]]> = icmp eq vp<%index.next>, vp<[[VP0]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<[[VP2]]>
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.exit:
+; CHECK-NEXT:    EMIT vp<%lanes> = add vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT ret vp<%lanes>
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
 ;
 entry:
   br label %loop.header
@@ -55,6 +234,105 @@ exit:
 }
 
 define i32 @i32_induction(ptr %A, ptr %B, i32 %n) {
+; CHECK-LABEL: VPlan for loop in 'i32_induction'
+; CHECK:  VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT:  Live-in ir<%n> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<entry>:
+; CHECK-NEXT:  Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.ph:
+; CHECK-NEXT:  Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT:  <x1> vector loop: {
+; CHECK-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT:    vector.body:
+; CHECK-NEXT:      vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT:      SPECULATIVE-LOAD-ORACLE vp<[[VP5:%[0-9]+]]> = fn(vp<[[VP3]]>, ir<%A>, ir<%B>)
+; CHECK-NEXT:      CLONE ir<%gep.A> = getelementptr inbounds ir<%A>, vp<[[VP4]]>
+; CHECK-NEXT:      WIDEN-INTRINSIC vp<[[VP6:%[0-9]+]]> = call llvm.speculative.load(ir<%gep.A>, ir<false>, vp<[[VP5]]>, vp<[[VP3]]>, ir<%A>, ir<%B>)
+; CHECK-NEXT:      EMIT vp<[[VP7:%[0-9]+]]> = freeze vp<[[VP6]]>
+; CHECK-NEXT:      CLONE ir<%gep.B> = getelementptr inbounds ir<%B>, vp<[[VP4]]>
+; CHECK-NEXT:      WIDEN-INTRINSIC vp<[[VP8:%[0-9]+]]> = call llvm.speculative.load(ir<%gep.B>, ir<false>, vp<[[VP5]]>, vp<[[VP3]]>, ir<%A>, ir<%B>)
+; CHECK-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = freeze vp<[[VP8]]>
+; CHECK-NEXT:      WIDEN ir<%cmp> = icmp ne vp<[[VP7]]>, vp<[[VP9]]>
+; CHECK-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = freeze ir<%cmp>
+; CHECK-NEXT:      EMIT vp<[[VP11:%[0-9]+]]> = any-of vp<[[VP10]]>
+; CHECK-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT:      EMIT branch-on-two-conds vp<[[VP11]]>, vp<[[VP12]]>
+; CHECK-NEXT:    No successors
+; CHECK-NEXT:  }
+; CHECK-NEXT:  Successor(s): vector.early.exit, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT:  middle.block:
+; CHECK-NEXT:    EMIT vp<%cmp.n> = icmp eq ir<%n>, vp<[[VP2]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<%cmp.n>
+; CHECK-NEXT:  Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<exit>:
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.early.exit:
+; CHECK-NEXT:    EMIT vp<[[VP15:%[0-9]+]]> = first-active-lane ir<%cmp>
+; CHECK-NEXT:    EMIT-SCALAR vp<[[VP16:%[0-9]+]]> = trunc vp<[[VP15]]> to i32
+; CHECK-NEXT:    EMIT vp<[[VP17:%[0-9]+]]> = add vp<[[VP3]]>, vp<[[VP16]]>
+; CHECK-NEXT:  Successor(s): ir-bb<early.exit>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<early.exit>:
+; CHECK-NEXT:    IR   %iv.lcssa = phi i32 [ %iv, %loop.header ] (extra operand: vp<[[VP17]]> from vector.early.exit)
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  scalar.ph:
+; CHECK-NEXT:    EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT:  Successor(s): ir-bb<loop.header>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<loop.header>:
+; CHECK-NEXT:    IR   %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT:    IR   %gep.A = getelementptr inbounds i8, ptr %A, i32 %iv
+; CHECK-NEXT:    IR   %ld.A = load i8, ptr %gep.A, align 1
+; CHECK-NEXT:    IR   %gep.B = getelementptr inbounds i8, ptr %B, i32 %iv
+; CHECK-NEXT:    IR   %ld.B = load i8, ptr %gep.B, align 1
+; CHECK-NEXT:    IR   %cmp = icmp ne i8 %ld.A, %ld.B
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
+; CHECK-EMPTY:
+; CHECK:  VPlan for speculative-load oracle vp<[[VP5]]> {
+; CHECK-NEXT:  Live-in vp<[[VP0]]> = VF
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.entry:
+; CHECK-NEXT:    EMIT-SCALAR vp<%base.index> = live-in ir<0>
+; CHECK-NEXT:    EMIT-SCALAR vp<%A> = live-in ir<1>
+; CHECK-NEXT:    EMIT-SCALAR vp<%B> = live-in ir<2>
+; CHECK-NEXT:  Successor(s): loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.header:
+; CHECK-NEXT:    EMIT-SCALAR vp<%index> = phi [ ir<0>, oracle.entry ], [ vp<%index.next>, loop.latch ]
+; CHECK-NEXT:    EMIT vp<%adjusted.iv> = add vp<%index>, vp<%base.index>
+; CHECK-NEXT:    CLONE ir<%gep.A> = getelementptr inbounds vp<%A>, vp<%adjusted.iv>
+; CHECK-NEXT:    CLONE ir<%ld.A> = load ir<%gep.A>
+; CHECK-NEXT:    CLONE ir<%gep.B> = getelementptr inbounds vp<%B>, vp<%adjusted.iv>
+; CHECK-NEXT:    CLONE ir<%ld.B> = load ir<%gep.B>
+; CHECK-NEXT:    EMIT ir<%cmp> = icmp ne ir<%ld.A>, ir<%ld.B>
+; CHECK-NEXT:    EMIT branch-on-cond ir<%cmp> (!vplan.prof.estimated estimated {67108864, 2080374784})
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.latch
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.latch:
+; CHECK-NEXT:    EMIT vp<%index.next> = add nuw vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT vp<[[VP2]]> = icmp eq vp<%index.next>, vp<[[VP0]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<[[VP2]]>
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.exit:
+; CHECK-NEXT:    EMIT vp<%lanes> = add vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT-SCALAR vp<[[VP4]]> = zext vp<%lanes> to i64
+; CHECK-NEXT:    EMIT ret vp<[[VP4]]>
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
 ;
 entry:
   br label %loop.header
@@ -80,9 +358,9 @@ exit:
   ret i32 -1
 }
 
-; Covers replaying scalar binops, div/rem with a constant divisor and freeze.
+; The oracle cannot replay freeze independently of the vector loop.
+; CHECK-NOT: VPlan for loop in 'binop_chain'
 define i64 @binop_chain(ptr %A, ptr %B, ptr %C, i64 %n) {
-;
 entry:
   br label %loop.header
 
@@ -186,6 +464,108 @@ exit:
 }
 
 define i64 @derived_induction_start(ptr %A, ptr %B, i64 %n) {
+; CHECK-LABEL: VPlan for loop in 'derived_induction_start'
+; CHECK:  VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT:  Live-in ir<%n> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<entry>:
+; CHECK-NEXT:  Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.ph:
+; CHECK-NEXT:    vp<[[VP3:%[0-9]+]]> = DERIVED-IV ir<10> + vp<[[VP2]]> * ir<1>
+; CHECK-NEXT:  Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT:  <x1> vector loop: {
+; CHECK-NEXT:  vp<[[VP4:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT:    vector.body:
+; CHECK-NEXT:      vp<[[VP5:%[0-9]+]]> = DERIVED-IV nuw ir<10> + vp<[[VP4]]> * ir<1>
+; CHECK-NEXT:      vp<[[VP6:%[0-9]+]]> = SCALAR-STEPS vp<[[VP5]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT:      SPECULATIVE-LOAD-ORACLE vp<[[VP7:%[0-9]+]]> = fn(vp<[[VP4]]>, ir<%A>, ir<%B>)
+; CHECK-NEXT:      CLONE ir<%gep.A> = getelementptr inbounds ir<%A>, vp<[[VP6]]>
+; CHECK-NEXT:      WIDEN-INTRINSIC vp<[[VP8:%[0-9]+]]> = call llvm.speculative.load(ir<%gep.A>, ir<false>, vp<[[VP7]]>, vp<[[VP4]]>, ir<%A>, ir<%B>)
+; CHECK-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = freeze vp<[[VP8]]>
+; CHECK-NEXT:      CLONE ir<%gep.B> = getelementptr inbounds ir<%B>, vp<[[VP6]]>
+; CHECK-NEXT:      WIDEN-INTRINSIC vp<[[VP10:%[0-9]+]]> = call llvm.speculative.load(ir<%gep.B>, ir<false>, vp<[[VP7]]>, vp<[[VP4]]>, ir<%A>, ir<%B>)
+; CHECK-NEXT:      EMIT vp<[[VP11:%[0-9]+]]> = freeze vp<[[VP10]]>
+; CHECK-NEXT:      WIDEN ir<%cmp> = icmp ne vp<[[VP9]]>, vp<[[VP11]]>
+; CHECK-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = freeze ir<%cmp>
+; CHECK-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = any-of vp<[[VP12]]>
+; CHECK-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP4]]>, vp<[[VP1]]>
+; CHECK-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT:      EMIT branch-on-two-conds vp<[[VP13]]>, vp<[[VP14]]>
+; CHECK-NEXT:    No successors
+; CHECK-NEXT:  }
+; CHECK-NEXT:  Successor(s): vector.early.exit, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT:  middle.block:
+; CHECK-NEXT:    EMIT vp<%cmp.n> = icmp eq ir<%n>, vp<[[VP2]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<%cmp.n>
+; CHECK-NEXT:  Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<exit>:
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.early.exit:
+; CHECK-NEXT:    EMIT vp<[[VP17:%[0-9]+]]> = first-active-lane ir<%cmp>
+; CHECK-NEXT:    EMIT vp<[[VP18:%[0-9]+]]> = add vp<[[VP4]]>, vp<[[VP17]]>
+; CHECK-NEXT:  Successor(s): ir-bb<early.exit>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<early.exit>:
+; CHECK-NEXT:    IR   %iv.lcssa = phi i64 [ %iv, %loop.header ] (extra operand: vp<[[VP18]]> from vector.early.exit)
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  scalar.ph:
+; CHECK-NEXT:    EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT:    EMIT-SCALAR vp<%bc.resume.val>.1 = phi [ vp<[[VP3]]>, middle.block ], [ ir<10>, ir-bb<entry> ]
+; CHECK-NEXT:  Successor(s): ir-bb<loop.header>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<loop.header>:
+; CHECK-NEXT:    IR   %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT:    IR   %iv.off = phi i64 [ 10, %entry ], [ %iv.off.next, %loop.latch ] (extra operand: vp<%bc.resume.val>.1 from scalar.ph)
+; CHECK-NEXT:    IR   %gep.A = getelementptr inbounds i8, ptr %A, i64 %iv.off
+; CHECK-NEXT:    IR   %ld.A = load i8, ptr %gep.A, align 1
+; CHECK-NEXT:    IR   %gep.B = getelementptr inbounds i8, ptr %B, i64 %iv.off
+; CHECK-NEXT:    IR   %ld.B = load i8, ptr %gep.B, align 1
+; CHECK-NEXT:    IR   %cmp = icmp ne i8 %ld.A, %ld.B
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
+; CHECK-EMPTY:
+; CHECK:  VPlan for speculative-load oracle vp<[[VP7]]> {
+; CHECK-NEXT:  Live-in vp<[[VP0]]> = VF
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.entry:
+; CHECK-NEXT:    EMIT-SCALAR vp<%base.index> = live-in ir<0>
+; CHECK-NEXT:    EMIT-SCALAR vp<%A> = live-in ir<1>
+; CHECK-NEXT:    EMIT-SCALAR vp<%B> = live-in ir<2>
+; CHECK-NEXT:  Successor(s): loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.header:
+; CHECK-NEXT:    EMIT-SCALAR vp<%index> = phi [ ir<0>, oracle.entry ], [ vp<%index.next>, loop.latch ]
+; CHECK-NEXT:    EMIT vp<%adjusted.iv> = add vp<%index>, vp<%base.index>
+; CHECK-NEXT:    EMIT vp<[[VP2]]> = add ir<10>, vp<%adjusted.iv>
+; CHECK-NEXT:    CLONE ir<%gep.A> = getelementptr inbounds vp<%A>, vp<[[VP2]]>
+; CHECK-NEXT:    CLONE ir<%ld.A> = load ir<%gep.A>
+; CHECK-NEXT:    CLONE ir<%gep.B> = getelementptr inbounds vp<%B>, vp<[[VP2]]>
+; CHECK-NEXT:    CLONE ir<%ld.B> = load ir<%gep.B>
+; CHECK-NEXT:    EMIT ir<%cmp> = icmp ne ir<%ld.A>, ir<%ld.B>
+; CHECK-NEXT:    EMIT branch-on-cond ir<%cmp> (!vplan.prof.estimated estimated {67108864, 2080374784})
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.latch
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.latch:
+; CHECK-NEXT:    EMIT vp<%index.next> = add nuw vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT vp<[[VP3]]> = icmp eq vp<%index.next>, vp<[[VP0]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<[[VP3]]>
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.exit:
+; CHECK-NEXT:    EMIT vp<%lanes> = add vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT ret vp<%lanes>
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
 ;
 entry:
   br label %loop.header
@@ -215,6 +595,102 @@ exit:
 
 ; Test with induction with step 2.
 define i64 @derived_induction_step(ptr %A, i64 %n) {
+; CHECK-LABEL: VPlan for loop in 'derived_induction_step'
+; CHECK:  VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT:  Live-in ir<%n> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<entry>:
+; CHECK-NEXT:  Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.ph:
+; CHECK-NEXT:    vp<[[VP3:%[0-9]+]]> = DERIVED-IV ir<0> + vp<[[VP2]]> * ir<2>
+; CHECK-NEXT:  Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT:  <x1> vector loop: {
+; CHECK-NEXT:  vp<[[VP4:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT:    vector.body:
+; CHECK-NEXT:      ir<%iv.2> = WIDEN-INDUCTION ir<0>, ir<2>, vp<[[VP0]]> (truncated to i8)
+; CHECK-NEXT:      vp<[[VP5:%[0-9]+]]> = SCALAR-STEPS vp<[[VP4]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT:      SPECULATIVE-LOAD-ORACLE vp<[[VP6:%[0-9]+]]> = fn(vp<[[VP4]]>, ir<%A>)
+; CHECK-NEXT:      CLONE ir<%gep.A> = getelementptr inbounds ir<%A>, vp<[[VP5]]>
+; CHECK-NEXT:      WIDEN-INTRINSIC vp<[[VP7:%[0-9]+]]> = call llvm.speculative.load(ir<%gep.A>, ir<false>, vp<[[VP6]]>, vp<[[VP4]]>, ir<%A>)
+; CHECK-NEXT:      EMIT vp<[[VP8:%[0-9]+]]> = freeze vp<[[VP7]]>
+; CHECK-NEXT:      WIDEN ir<%cmp> = icmp ne vp<[[VP8]]>, ir<%iv.2>
+; CHECK-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = freeze ir<%cmp>
+; CHECK-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = any-of vp<[[VP9]]>
+; CHECK-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP4]]>, vp<[[VP1]]>
+; CHECK-NEXT:      EMIT vp<[[VP11:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT:      EMIT branch-on-two-conds vp<[[VP10]]>, vp<[[VP11]]>
+; CHECK-NEXT:    No successors
+; CHECK-NEXT:  }
+; CHECK-NEXT:  Successor(s): vector.early.exit, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT:  middle.block:
+; CHECK-NEXT:    EMIT vp<%cmp.n> = icmp eq ir<%n>, vp<[[VP2]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<%cmp.n>
+; CHECK-NEXT:  Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<exit>:
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.early.exit:
+; CHECK-NEXT:    EMIT vp<[[VP14:%[0-9]+]]> = first-active-lane ir<%cmp>
+; CHECK-NEXT:    EMIT vp<[[VP15:%[0-9]+]]> = add vp<[[VP4]]>, vp<[[VP14]]>
+; CHECK-NEXT:  Successor(s): ir-bb<early.exit>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<early.exit>:
+; CHECK-NEXT:    IR   %iv.lcssa = phi i64 [ %iv, %loop.header ] (extra operand: vp<[[VP15]]> from vector.early.exit)
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  scalar.ph:
+; CHECK-NEXT:    EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT:    EMIT-SCALAR vp<%bc.resume.val>.1 = phi [ vp<[[VP3]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT:  Successor(s): ir-bb<loop.header>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<loop.header>:
+; CHECK-NEXT:    IR   %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT:    IR   %iv.2 = phi i64 [ 0, %entry ], [ %iv.2.next, %loop.latch ] (extra operand: vp<%bc.resume.val>.1 from scalar.ph)
+; CHECK-NEXT:    IR   %gep.A = getelementptr inbounds i8, ptr %A, i64 %iv
+; CHECK-NEXT:    IR   %ld.A = load i8, ptr %gep.A, align 1
+; CHECK-NEXT:    IR   %trunc = trunc i64 %iv.2 to i8
+; CHECK-NEXT:    IR   %cmp = icmp ne i8 %ld.A, %trunc
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
+; CHECK-EMPTY:
+; CHECK:  VPlan for speculative-load oracle vp<[[VP6]]> {
+; CHECK-NEXT:  Live-in vp<[[VP0]]> = VF
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.entry:
+; CHECK-NEXT:    EMIT-SCALAR vp<%base.index> = live-in ir<0>
+; CHECK-NEXT:    EMIT-SCALAR vp<%A> = live-in ir<1>
+; CHECK-NEXT:  Successor(s): loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.header:
+; CHECK-NEXT:    EMIT-SCALAR vp<%index> = phi [ ir<0>, oracle.entry ], [ vp<%index.next>, loop.latch ]
+; CHECK-NEXT:    EMIT vp<%adjusted.iv> = add vp<%index>, vp<%base.index>
+; CHECK-NEXT:    EMIT vp<[[VP2]]> = shl vp<%adjusted.iv>, ir<1>
+; CHECK-NEXT:    CLONE ir<%gep.A> = getelementptr inbounds vp<%A>, vp<%adjusted.iv>
+; CHECK-NEXT:    CLONE ir<%ld.A> = load ir<%gep.A>
+; CHECK-NEXT:    EMIT-SCALAR ir<%trunc> = trunc vp<[[VP2]]> to i8
+; CHECK-NEXT:    EMIT ir<%cmp> = icmp ne ir<%ld.A>, ir<%trunc>
+; CHECK-NEXT:    EMIT branch-on-cond ir<%cmp> (!vplan.prof.estimated estimated {67108864, 2080374784})
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.latch
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.latch:
+; CHECK-NEXT:    EMIT vp<%index.next> = add nuw vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT vp<[[VP3]]> = icmp eq vp<%index.next>, vp<[[VP0]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<[[VP3]]>
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.exit:
+; CHECK-NEXT:    EMIT vp<%lanes> = add vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT ret vp<%lanes>
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
 ;
 entry:
   br label %loop.header
@@ -243,6 +719,97 @@ exit:
 
 ; Dead recipes are dropped before the oracle plan is built.
 define i64 @dead_recipes_in_loop(ptr %A, i64 %n) {
+; CHECK-LABEL: VPlan for loop in 'dead_recipes_in_loop'
+; CHECK:  VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT:  Live-in ir<%n> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<entry>:
+; CHECK-NEXT:  Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.ph:
+; CHECK-NEXT:  Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT:  <x1> vector loop: {
+; CHECK-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT:    vector.body:
+; CHECK-NEXT:      vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT:      SPECULATIVE-LOAD-ORACLE vp<[[VP5:%[0-9]+]]> = fn(vp<[[VP3]]>, ir<%A>)
+; CHECK-NEXT:      CLONE ir<%gep.A> = getelementptr inbounds ir<%A>, vp<[[VP4]]>
+; CHECK-NEXT:      WIDEN-INTRINSIC vp<[[VP6:%[0-9]+]]> = call llvm.speculative.load(ir<%gep.A>, ir<false>, vp<[[VP5]]>, vp<[[VP3]]>, ir<%A>)
+; CHECK-NEXT:      EMIT vp<[[VP7:%[0-9]+]]> = freeze vp<[[VP6]]>
+; CHECK-NEXT:      WIDEN ir<%cmp> = icmp eq vp<[[VP7]]>, ir<42>
+; CHECK-NEXT:      EMIT vp<[[VP8:%[0-9]+]]> = freeze ir<%cmp>
+; CHECK-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = any-of vp<[[VP8]]>
+; CHECK-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT:      EMIT branch-on-two-conds vp<[[VP9]]>, vp<[[VP10]]>
+; CHECK-NEXT:    No successors
+; CHECK-NEXT:  }
+; CHECK-NEXT:  Successor(s): vector.early.exit, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT:  middle.block:
+; CHECK-NEXT:    EMIT vp<%cmp.n> = icmp eq ir<%n>, vp<[[VP2]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<%cmp.n>
+; CHECK-NEXT:  Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<exit>:
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.early.exit:
+; CHECK-NEXT:    EMIT vp<[[VP13:%[0-9]+]]> = first-active-lane ir<%cmp>
+; CHECK-NEXT:    EMIT vp<[[VP14:%[0-9]+]]> = add vp<[[VP3]]>, vp<[[VP13]]>
+; CHECK-NEXT:  Successor(s): ir-bb<early.exit>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<early.exit>:
+; CHECK-NEXT:    IR   %iv.lcssa = phi i64 [ %iv, %loop.header ] (extra operand: vp<[[VP14]]> from vector.early.exit)
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  scalar.ph:
+; CHECK-NEXT:    EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT:  Successor(s): ir-bb<loop.header>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<loop.header>:
+; CHECK-NEXT:    IR   %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT:    IR   %dead.call = call i8 @llvm.abs.i8(i8 0, i1 false)
+; CHECK-NEXT:    IR   %dead.gep = getelementptr i8, ptr null, i64 8
+; CHECK-NEXT:    IR   %gep.A = getelementptr inbounds i8, ptr %A, i64 %iv
+; CHECK-NEXT:    IR   %ld.A = load i8, ptr %gep.A, align 1
+; CHECK-NEXT:    IR   %cmp = icmp eq i8 %ld.A, 42
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
+; CHECK-EMPTY:
+; CHECK:  VPlan for speculative-load oracle vp<[[VP5]]> {
+; CHECK-NEXT:  Live-in vp<[[VP0]]> = VF
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.entry:
+; CHECK-NEXT:    EMIT-SCALAR vp<%base.index> = live-in ir<0>
+; CHECK-NEXT:    EMIT-SCALAR vp<%A> = live-in ir<1>
+; CHECK-NEXT:  Successor(s): loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.header:
+; CHECK-NEXT:    EMIT-SCALAR vp<%index> = phi [ ir<0>, oracle.entry ], [ vp<%index.next>, loop.latch ]
+; CHECK-NEXT:    EMIT vp<%adjusted.iv> = add vp<%index>, vp<%base.index>
+; CHECK-NEXT:    CLONE ir<%gep.A> = getelementptr inbounds vp<%A>, vp<%adjusted.iv>
+; CHECK-NEXT:    CLONE ir<%ld.A> = load ir<%gep.A>
+; CHECK-NEXT:    EMIT ir<%cmp> = icmp eq ir<%ld.A>, ir<42>
+; CHECK-NEXT:    EMIT branch-on-cond ir<%cmp> (!vplan.prof.estimated estimated {67108864, 2080374784})
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.latch
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.latch:
+; CHECK-NEXT:    EMIT vp<%index.next> = add nuw vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT vp<[[VP2]]> = icmp eq vp<%index.next>, vp<[[VP0]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<[[VP2]]>
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.exit:
+; CHECK-NEXT:    EMIT vp<%lanes> = add vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT ret vp<%lanes>
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
 ;
 entry:
   br label %loop.header
@@ -270,6 +837,95 @@ exit:
 
 ; The oracle only reads memory through its arguments, so @G is passed in.
 define i64 @global_base(i64 %n) {
+; CHECK-LABEL: VPlan for loop in 'global_base'
+; CHECK:  VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT:  Live-in ir<%n> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<entry>:
+; CHECK-NEXT:  Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.ph:
+; CHECK-NEXT:  Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT:  <x1> vector loop: {
+; CHECK-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT:    vector.body:
+; CHECK-NEXT:      vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT:      SPECULATIVE-LOAD-ORACLE vp<[[VP5:%[0-9]+]]> = fn(vp<[[VP3]]>, ir<@G>)
+; CHECK-NEXT:      CLONE ir<%gep> = getelementptr inbounds ir<@G>, vp<[[VP4]]>
+; CHECK-NEXT:      WIDEN-INTRINSIC vp<[[VP6:%[0-9]+]]> = call llvm.speculative.load(ir<%gep>, ir<false>, vp<[[VP5]]>, vp<[[VP3]]>, ir<@G>)
+; CHECK-NEXT:      EMIT vp<[[VP7:%[0-9]+]]> = freeze vp<[[VP6]]>
+; CHECK-NEXT:      WIDEN ir<%c> = icmp eq vp<[[VP7]]>, ir<42>
+; CHECK-NEXT:      EMIT vp<[[VP8:%[0-9]+]]> = freeze ir<%c>
+; CHECK-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = any-of vp<[[VP8]]>
+; CHECK-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT:      EMIT branch-on-two-conds vp<[[VP9]]>, vp<[[VP10]]>
+; CHECK-NEXT:    No successors
+; CHECK-NEXT:  }
+; CHECK-NEXT:  Successor(s): vector.early.exit, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT:  middle.block:
+; CHECK-NEXT:    EMIT vp<%cmp.n> = icmp eq ir<%n>, vp<[[VP2]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<%cmp.n>
+; CHECK-NEXT:  Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<exit>:
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.early.exit:
+; CHECK-NEXT:    EMIT vp<[[VP13:%[0-9]+]]> = first-active-lane ir<%c>
+; CHECK-NEXT:    EMIT vp<[[VP14:%[0-9]+]]> = add vp<[[VP3]]>, vp<[[VP13]]>
+; CHECK-NEXT:  Successor(s): ir-bb<early.exit>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<early.exit>:
+; CHECK-NEXT:    IR   %iv.lcssa = phi i64 [ %iv, %loop.header ] (extra operand: vp<[[VP14]]> from vector.early.exit)
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  scalar.ph:
+; CHECK-NEXT:    EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT:  Successor(s): ir-bb<loop.header>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<loop.header>:
+; CHECK-NEXT:    IR   %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT:    IR   %gep = getelementptr inbounds i8, ptr @G, i64 %iv
+; CHECK-NEXT:    IR   %l = load i8, ptr %gep, align 1
+; CHECK-NEXT:    IR   %c = icmp eq i8 %l, 42
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
+; CHECK-EMPTY:
+; CHECK:  VPlan for speculative-load oracle vp<[[VP5]]> {
+; CHECK-NEXT:  Live-in vp<[[VP0]]> = VF
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.entry:
+; CHECK-NEXT:    EMIT-SCALAR vp<%base.index> = live-in ir<0>
+; CHECK-NEXT:    EMIT-SCALAR vp<%G> = live-in ir<1>
+; CHECK-NEXT:  Successor(s): loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.header:
+; CHECK-NEXT:    EMIT-SCALAR vp<%index> = phi [ ir<0>, oracle.entry ], [ vp<%index.next>, loop.latch ]
+; CHECK-NEXT:    EMIT vp<%adjusted.iv> = add vp<%index>, vp<%base.index>
+; CHECK-NEXT:    CLONE ir<%gep> = getelementptr inbounds vp<%G>, vp<%adjusted.iv>
+; CHECK-NEXT:    CLONE ir<%l> = load ir<%gep>
+; CHECK-NEXT:    EMIT ir<%c> = icmp eq ir<%l>, ir<42>
+; CHECK-NEXT:    EMIT branch-on-cond ir<%c> (!vplan.prof.estimated estimated {67108864, 2080374784})
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.latch
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.latch:
+; CHECK-NEXT:    EMIT vp<%index.next> = add nuw vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT vp<[[VP2]]> = icmp eq vp<%index.next>, vp<[[VP0]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<[[VP2]]>
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.exit:
+; CHECK-NEXT:    EMIT vp<%lanes> = add vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT ret vp<%lanes>
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
 ;
 entry:
   br label %loop.header
@@ -295,6 +951,8 @@ exit:
 
 ; TODO: support calls in the oracle.
 define i64 @call_in_exit_condition(ptr %A, i64 %n) {
+; CHECK-NOT: VPlan for loop in 'call_in_exit_condition'
+;
 entry:
   br label %loop.header
 
@@ -320,6 +978,8 @@ exit:
 
 ; TODO: support fcmp in the oracle.
 define i64 @fcmp_in_exit_condition(ptr %A, i64 %n) {
+; CHECK-NOT: VPlan for loop in 'fcmp_in_exit_condition'
+;
 entry:
   br label %loop.header
 
@@ -369,6 +1029,8 @@ exit:
 
 ; TODO: support pointer inductions in the oracle.
 define i64 @pointer_induction(ptr %begin, ptr %end) {
+; CHECK-NOT: VPlan for loop in 'pointer_induction'
+;
 entry:
   %is.empty = icmp eq ptr %begin, %end
   br i1 %is.empty, label %exit, label %loop.header
@@ -393,6 +1055,8 @@ exit:
 }
 
 define i64 @early_exit_runtime_induction_step(ptr %A, i64 %n, i64 %s) {
+; CHECK-NOT: VPlan for loop in 'early_exit_runtime_induction_step'
+;
 entry:
   %step = mul i64 %s, 4
   br label %loop.header
@@ -417,3 +1081,120 @@ early.exit:
 exit:
   ret i64 -1
 }
+
+; An exit live-out may use %scale, but the oracle only needs the exit test.
+define i8 @liveout_only(ptr %A, i64 %n, i8 %scale) {
+; CHECK-LABEL: VPlan for loop in 'liveout_only'
+; CHECK:  VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT:  Live-in ir<%n> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<entry>:
+; CHECK-NEXT:  Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.ph:
+; CHECK-NEXT:  Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT:  <x1> vector loop: {
+; CHECK-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT:    vector.body:
+; CHECK-NEXT:      vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT:      SPECULATIVE-LOAD-ORACLE vp<[[VP5:%[0-9]+]]> = fn(vp<[[VP3]]>, ir<%A>)
+; CHECK-NEXT:      CLONE ir<%gep> = getelementptr inbounds ir<%A>, vp<[[VP4]]>
+; CHECK-NEXT:      WIDEN-INTRINSIC vp<[[VP6:%[0-9]+]]> = call llvm.speculative.load(ir<%gep>, ir<false>, vp<[[VP5]]>, vp<[[VP3]]>, ir<%A>)
+; CHECK-NEXT:      EMIT vp<[[VP7:%[0-9]+]]> = freeze vp<[[VP6]]>
+; CHECK-NEXT:      WIDEN ir<%cmp> = icmp eq vp<[[VP7]]>, ir<42>
+; CHECK-NEXT:      EMIT vp<[[VP8:%[0-9]+]]> = freeze ir<%cmp>
+; CHECK-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = any-of vp<[[VP8]]>
+; CHECK-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT:      EMIT branch-on-two-conds vp<[[VP9]]>, vp<[[VP10]]>
+; CHECK-NEXT:    No successors
+; CHECK-NEXT:  }
+; CHECK-NEXT:  Successor(s): vector.early.exit, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT:  middle.block:
+; CHECK-NEXT:    EMIT vp<%cmp.n> = icmp eq ir<%n>, vp<[[VP2]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<%cmp.n>
+; CHECK-NEXT:  Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<exit>:
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  vector.early.exit:
+; CHECK-NEXT:    WIDEN ir<%val> = add vp<[[VP7]]>, ir<%scale>
+; CHECK-NEXT:    EMIT vp<%first.active.lane> = first-active-lane ir<%cmp>
+; CHECK-NEXT:    EMIT vp<%early.exit.value> = extract-lane vp<%first.active.lane>, ir<%val>
+; CHECK-NEXT:  Successor(s): ir-bb<early.exit>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<early.exit>:
+; CHECK-NEXT:    IR   %val.lcssa = phi i8 [ %val, %loop.header ] (extra operand: vp<%early.exit.value> from vector.early.exit)
+; CHECK-NEXT:  No successors
+; CHECK-EMPTY:
+; CHECK-NEXT:  scalar.ph:
+; CHECK-NEXT:    EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT:  Successor(s): ir-bb<loop.header>
+; CHECK-EMPTY:
+; CHECK-NEXT:  ir-bb<loop.header>:
+; CHECK-NEXT:    IR   %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT:    IR   %gep = getelementptr inbounds i8, ptr %A, i64 %iv
+; CHECK-NEXT:    IR   %ld = load i8, ptr %gep, align 1
+; CHECK-NEXT:    IR   %val = add i8 %ld, %scale
+; CHECK-NEXT:    IR   %cmp = icmp eq i8 %ld, 42
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
+; CHECK-EMPTY:
+; CHECK:  VPlan for speculative-load oracle vp<[[VP5]]> {
+; CHECK-NEXT:  Live-in vp<[[VP0]]> = VF
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.entry:
+; CHECK-NEXT:    EMIT-SCALAR vp<%base.index> = live-in ir<0>
+; CHECK-NEXT:    EMIT-SCALAR vp<%A> = live-in ir<1>
+; CHECK-NEXT:  Successor(s): loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.header:
+; CHECK-NEXT:    EMIT-SCALAR vp<%index> = phi [ ir<0>, oracle.entry ], [ vp<%index.next>, loop.latch ]
+; CHECK-NEXT:    EMIT vp<%adjusted.iv> = add vp<%index>, vp<%base.index>
+; CHECK-NEXT:    CLONE ir<%gep> = getelementptr inbounds vp<%A>, vp<%adjusted.iv>
+; CHECK-NEXT:    CLONE ir<%ld> = load ir<%gep>
+; CHECK-NEXT:    EMIT ir<%cmp> = icmp eq ir<%ld>, ir<42>
+; CHECK-NEXT:    EMIT branch-on-cond ir<%cmp> (!vplan.prof.estimated estimated {67108864, 2080374784})
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.latch
+; CHECK-EMPTY:
+; CHECK-NEXT:  loop.latch:
+; CHECK-NEXT:    EMIT vp<%index.next> = add nuw vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT vp<[[VP2]]> = icmp eq vp<%index.next>, vp<[[VP0]]>
+; CHECK-NEXT:    EMIT branch-on-cond vp<[[VP2]]>
+; CHECK-NEXT:  Successor(s): oracle.exit, loop.header
+; CHECK-EMPTY:
+; CHECK-NEXT:  oracle.exit:
+; CHECK-NEXT:    EMIT vp<%lanes> = add vp<%index>, ir<1>
+; CHECK-NEXT:    EMIT ret vp<%lanes>
+; CHECK-NEXT:  No successors
+; CHECK-NEXT:  }
+;
+entry:
+  br label %loop.header
+
+loop.header:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+  %gep = getelementptr inbounds i8, ptr %A, i64 %iv
+  %ld = load i8, ptr %gep, align 1
+  %val = add i8 %ld, %scale
+  %cmp = icmp eq i8 %ld, 42
+  br i1 %cmp, label %early.exit, label %loop.latch
+
+loop.latch:
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp ne i64 %iv.next, %n
+  br i1 %ec, label %loop.header, label %exit
+
+early.exit:
+  ret i8 %val
+
+exit:
+  ret i8 0
+}
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
index fe71cee7c3492..962422263e8a8 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
@@ -71,6 +71,7 @@
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::addMinimumIterationCheck
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::replaceWideCanonicalIVWithWideIV
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::unrollByUF
+; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::attachSpeculativeLoadChecks
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::materializePacksAndUnpacks
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::materializeBroadcasts
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::replicateByVF
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_legality.ll b/llvm/test/Transforms/LoopVectorize/early_exit_legality.ll
index 7768aac2ec309..d8c4d275db792 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_legality.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_legality.ll
@@ -208,7 +208,8 @@ loop.end:
 
 define i64 @same_exit_block_pre_inc_use1_too_small_allocas() {
 ; CHECK-LABEL: LV: Checking a loop in 'same_exit_block_pre_inc_use1_too_small_allocas'
-; CHECK:       LV: Not vectorizing: Auto-vectorization of loops with potentially faulting load is not supported.
+; CHECK:       LV: We can vectorize this loop!
+; CHECK-NOT:   LV: Not vectorizing:
 entry:
   %p1 = alloca [42 x i8]
   %p2 = alloca [42 x i8]
@@ -238,7 +239,8 @@ loop.end:
 
 define i64 @same_exit_block_pre_inc_use1_too_small_deref_ptrs(ptr dereferenceable(42) %p1, ptr dereferenceable(42) %p2) {
 ; CHECK-LABEL: LV: Checking a loop in 'same_exit_block_pre_inc_use1_too_small_deref_ptrs'
-; CHECK:       LV: Not vectorizing: Auto-vectorization of loops with potentially faulting load is not supported.
+; CHECK:       LV: We can vectorize this loop!
+; CHECK-NOT:   LV: Not vectorizing:
 entry:
   br label %loop
 
@@ -264,7 +266,8 @@ loop.end:
 
 define i64 @same_exit_block_pre_inc_use1_unknown_ptrs(ptr %p1, ptr %p2) {
 ; CHECK-LABEL: LV: Checking a loop in 'same_exit_block_pre_inc_use1_unknown_ptrs'
-; CHECK:       LV: Not vectorizing: Auto-vectorization of loops with potentially faulting load is not supported.
+; CHECK:       LV: We can vectorize this loop!
+; CHECK-NOT:   LV: Not vectorizing:
 entry:
   br label %loop
 
diff --git a/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp b/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
index ffc3a0d303474..8b37aa5676528 100644
--- a/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
+++ b/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
@@ -37,6 +37,26 @@ namespace {
 using VPInstructionTest = VPlanTestBase;
 using VPlanSCEVTest = VPlanTestIRBase;
 
+TEST_F(VPInstructionTest, ReturnKeepsOperandAlive) {
+  VPlan &Plan = getPlan();
+  auto *Exit = Plan.createVPBasicBlock("exit");
+  Plan.setEntry(Exit);
+  VPBuilder Builder(Exit);
+  VPValue *One = Plan.getConstantInt(64, 1);
+  VPInstruction *Value = Builder.createAdd(One, One);
+  Builder.createAdd(Value, One);
+  VPInstruction *Ret = Builder.createNaryOp(Instruction::Ret, Value);
+
+  VPlanTransforms::removeDeadRecipes(Plan);
+  EXPECT_EQ(Exit->size(), 2u);
+  EXPECT_EQ(&Exit->front(), Value);
+  EXPECT_EQ(Exit->getTerminator(), Ret);
+  EXPECT_EQ(std::as_const(*Exit).getTerminator(), Ret);
+  EXPECT_TRUE(Ret->getScalarType()->isVoidTy());
+  EXPECT_FALSE(Ret->mayReadOrWriteMemory());
+  EXPECT_TRUE(Ret->usesFirstLaneOnly(Value));
+}
+
 TEST_F(VPlanSCEVTest, GetSCEVExprForVPValueAbs) {
   const char *ModuleString = R"(
 define void @f(i32 %x) {
diff --git a/llvm/unittests/Transforms/Vectorize/VPlanVerifierTest.cpp b/llvm/unittests/Transforms/Vectorize/VPlanVerifierTest.cpp
index 7ce87be841c2d..776d883e4cd56 100644
--- a/llvm/unittests/Transforms/Vectorize/VPlanVerifierTest.cpp
+++ b/llvm/unittests/Transforms/Vectorize/VPlanVerifierTest.cpp
@@ -23,6 +23,77 @@ LLVM_ABI extern cl::opt<bool> VerifyEachVPlan;
 using VPVerifierTest = VPlanTestBase;
 
 namespace {
+TEST_F(VPVerifierTest, ReturnTerminator) {
+  VPlan &Plan = getPlan();
+  auto *Exit = Plan.createVPBasicBlock("exit");
+  Plan.setEntry(Exit);
+  VPValue *One = Plan.getConstantInt(64, 1);
+  VPBuilder Builder(Exit);
+  Builder.createNaryOp(Instruction::Ret, One);
+  EXPECT_TRUE(verifyVPlanIsValid(Plan));
+
+  auto CheckInvalidReturn = [&]() {
+#if GTEST_HAS_STREAM_REDIRECTION
+    ::testing::internal::CaptureStderr();
+#endif
+    EXPECT_FALSE(verifyVPlanIsValid(Plan));
+#if GTEST_HAS_STREAM_REDIRECTION
+    EXPECT_STREQ(
+        "Return must terminate a top-level block without successors!\n",
+        ::testing::internal::GetCapturedStderr().c_str());
+#endif
+  };
+
+  // A return must be the last recipe in its block.
+  VPInstruction *AfterRet = Builder.createAdd(One, One);
+  CheckInvalidReturn();
+  AfterRet->eraseFromParent();
+
+  // A return cannot have a successor.
+  VPBlockUtils::connectBlocks(Exit, Plan.getScalarHeader());
+  CheckInvalidReturn();
+  VPBlockUtils::disconnectBlocks(Exit, Plan.getScalarHeader());
+
+  // A return cannot terminate a block inside a loop region.
+  auto *Entry = Plan.createVPBasicBlock("entry");
+  Plan.setEntry(Entry);
+  auto *Region = Plan.createLoopRegion(One->getScalarType(), DebugLoc(), "loop",
+                                       Exit, Exit);
+  VPBlockUtils::connectBlocks(Entry, Region);
+  CheckInvalidReturn();
+}
+
+TEST_F(VPVerifierTest, LiveIn) {
+  VPlan &Plan = getPlan();
+  VPBasicBlock *Entry = Plan.getEntry();
+  Type *I64Ty = Type::getInt64Ty(C);
+  VPBuilder Builder(Entry);
+  Builder.createLiveIn(0, I64Ty);
+  VPInstruction *LiveIn1 = Builder.createLiveIn(1, I64Ty);
+  EXPECT_TRUE(verifyVPlanIsValid(Plan));
+
+  auto CheckInvalidPlan = [&](const char *Msg) {
+#if GTEST_HAS_STREAM_REDIRECTION
+    ::testing::internal::CaptureStderr();
+#endif
+    EXPECT_FALSE(verifyVPlanIsValid(Plan));
+#if GTEST_HAS_STREAM_REDIRECTION
+    EXPECT_STREQ(Msg, ::testing::internal::GetCapturedStderr().c_str());
+#endif
+  };
+
+  // Live-ins must have distinct indices.
+  VPInstruction *Duplicate = Builder.createLiveIn(1, I64Ty);
+  CheckInvalidPlan("Multiple live-ins with index 1!\n");
+  Duplicate->eraseFromParent();
+
+  // Live-ins must be in the plan's entry block.
+  VPBasicBlock *VPBB = Plan.createVPBasicBlock("bb");
+  VPBlockUtils::connectBlocks(Entry, VPBB);
+  LiveIn1->moveBefore(*VPBB, VPBB->end());
+  CheckInvalidPlan("Live-in must be in the plan's entry block!\n");
+}
+
 TEST_F(VPVerifierTest, VPInstructionUseBeforeDefSameBB) {
   VPlan &Plan = getPlan();
   VPIRValue *Zero = Plan.getConstantInt(32, 0);



More information about the llvm-commits mailing list