[llvm] [TTI] Add distinguished shouldVectorizeNTStore/Load (PR #177936)

Tomer Shafir via llvm-commits llvm-commits at lists.llvm.org
Mon Jan 26 04:14:03 PST 2026


https://github.com/tomershafir updated https://github.com/llvm/llvm-project/pull/177936

>From 70e0f0d74a048196926ba36b2c2f6f0b7d52fce1 Mon Sep 17 00:00:00 2001
From: tomershafir <tomer.shafir8 at gmail.com>
Date: Thu, 22 Jan 2026 23:08:54 +0200
Subject: [PATCH 1/5] [AArch64] Align nontemporal store/load little-endian
 checks

This patch aims to align all nontemporal store/load handling to systematically enforce a little-endian target. This has been the effective support LLVM had for NT store/load lowering (there has been no effective support for big-endian, even with the inconsistencies).

The change in `llvm/lib/Target/AArch64/AArch64InstrInfo.td` is effectively a NFC, because the only lowering of LDNP, in `llvm/lib/Target/AArch64/AArch64ISelLowering.cpp`, have already checked for `isLittleEndian`.
The change in `llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h` affects its single caller `llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp`. The previous logic has been wrong, enabling vectorization of effectively illegal nontemporal store/load instructions on big-endian.
---
 .../Target/AArch64/AArch64ISelLowering.cpp    |  10 +
 llvm/lib/Target/AArch64/AArch64InstrInfo.td   |  16 +-
 .../AArch64/AArch64TargetTransformInfo.h      |  30 +-
 llvm/test/CodeGen/AArch64/nontemporal-load.ll | 378 +++++++++---------
 .../AArch64/nontemporal-load-store.ll         | 245 ++++++++----
 5 files changed, 401 insertions(+), 278 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 05d5f6b323706..eef52867107e9 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -7322,6 +7322,10 @@ static SDValue LowerADDRSPACECAST(SDValue Op, SelectionDAG &DAG) {
 }
 
 // Lower non-temporal stores that would otherwise be broken by legalization.
+//
+// Coordinated with LDNP constraints in
+// `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
+// `AArch64TargetLowering::ReplaceNodeResults`
 static SDValue LowerNTStore(StoreSDNode *StoreNode, EVT VT, EVT MemVT,
                             const SDLoc &DL, SelectionDAG &DAG) {
   assert(StoreNode && "Expected a store operation");
@@ -29564,6 +29568,12 @@ void AArch64TargetLowering::ReplaceNodeResults(
     EVT MemVT = LoadNode->getMemoryVT();
     // Handle lowering 256 bit non temporal loads into LDNP for little-endian
     // targets.
+    //
+    // Currently we only support NT loads lowering for little-endian targets.
+    //
+    // Coordinated with LDNP constraints in
+    // `llvm/lib/Target/AArch64/AArch64InstrInfo.td`
+    // and `AArch64TTIImpl::isLegalNTLoad`.
     if (LoadNode->isNonTemporal() && Subtarget->isLittleEndian() &&
         MemVT.getSizeInBits() == 256u &&
         (MemVT.getScalarSizeInBits() == 8u ||
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.td b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
index 59a3b2d36e0f0..ee9e3fa886a5c 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
@@ -3786,8 +3786,15 @@ defm LDNPQ : LoadPairNoAlloc<0b10, 1, FPR128Op, simm7s16, "ldnp">;
 def : Pat<(AArch64ldp (am_indexed7s64 GPR64sp:$Rn, simm7s8:$offset)),
           (LDPXi GPR64sp:$Rn, simm7s8:$offset)>;
 
-def : Pat<(AArch64ldnp (am_indexed7s128 GPR64sp:$Rn, simm7s16:$offset)),
-          (LDNPQi GPR64sp:$Rn, simm7s16:$offset)>;
+// Currently we only support NT loads lowering for little-endian targets.
+//
+// Coordinated with LDNP constraints in `AArch64TargetLowering::ReplaceNodeResults`
+// and `AArch64TTIImpl::isLegalNTLoad`.
+let Predicates = [IsLE] in {
+  def : Pat<(AArch64ldnp(am_indexed7s128 GPR64sp:$Rn, simm7s16:$offset)),
+            (LDNPQi GPR64sp:$Rn, simm7s16:$offset)>;
+}
+
 //---
 // (register offset)
 //---
@@ -10724,6 +10731,11 @@ def : Pat<(i64 (int_aarch64_neon_urshl (i64 FPR64:$Rn), (i64 FPR64:$Rm))),
 // We have to resort to tricks to turn a single-input store into a store pair,
 // because there is no single-input nontemporal store, only STNP.
 //
+// Currently we only support NT stores lowering for little-endian targets.
+//
+// Coordinated with STNP constraints in `AArch64TargetLowering::LowerNTStore`
+// and `AArch64TTIImpl::isLegalNTStore`.
+//
 // Currently, STNP lowering can only either keep or increase code size, thus
 // we predicate it to not apply when optimizing for code size.
 let Predicates = [IsLE, NotForCodeSize] in {
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index c9bf44b15144a..b789e13bf25da 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -391,7 +391,8 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
     return false;
   }
 
-  bool isLegalNTStoreLoad(Type *DataType, Align Alignment) const {
+  std::optional<bool> isLegalNTStoreLoad(Type *DataType,
+                                         Align Alignment) const {
     // NOTE: The logic below is mostly geared towards LV, which calls it with
     //       vectors with 2 elements. We might want to improve that, if other
     //       users show up.
@@ -405,17 +406,34 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
       return NumElements > 1 && isPowerOf2_64(NumElements) && EltSize >= 8 &&
              EltSize <= 128 && isPowerOf2_64(EltSize);
     }
-    return BaseT::isLegalNTStore(DataType, Alignment);
+    return std::nullopt;
   }
 
   bool isLegalNTStore(Type *DataType, Align Alignment) const override {
-    return isLegalNTStoreLoad(DataType, Alignment);
+    // Currently we only support NT stores lowering for little-endian targets.
+    //
+    // Coordinated with STNP constraints in
+    // `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
+    // `AArch64TargetLowering::LowerNTStore`
+    if (!ST->isLittleEndian())
+      return false;
+    if (auto Result = isLegalNTStoreLoad(DataType, Alignment))
+      return *Result;
+    // Fallback to target independent logic
+    return BaseT::isLegalNTStore(DataType, Alignment);
   }
 
   bool isLegalNTLoad(Type *DataType, Align Alignment) const override {
-    // Only supports little-endian targets.
-    if (ST->isLittleEndian())
-      return isLegalNTStoreLoad(DataType, Alignment);
+    // Currently we only support NT loads lowering for little-endian targets.
+    //
+    // Coordinated with LDNP constraints in
+    // `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
+    // `AArch64TargetLowering::ReplaceNodeResults`
+    if (!ST->isLittleEndian())
+      return false;
+    if (auto Result = isLegalNTStoreLoad(DataType, Alignment))
+      return *Result;
+    // Fallback to target independent logic
     return BaseT::isLegalNTLoad(DataType, Alignment);
   }
 
diff --git a/llvm/test/CodeGen/AArch64/nontemporal-load.ll b/llvm/test/CodeGen/AArch64/nontemporal-load.ll
index ad92530eabf08..62b3e5651423f 100644
--- a/llvm/test/CodeGen/AArch64/nontemporal-load.ll
+++ b/llvm/test/CodeGen/AArch64/nontemporal-load.ll
@@ -1,12 +1,12 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc --mattr=+sve -aarch64-enable-sink-fold=true < %s -mtriple aarch64-apple-darwin | FileCheck %s
+; RUN: llc --mattr=+sve -aarch64-enable-sink-fold=true < %s -mtriple aarch64-apple-darwin | FileCheck --check-prefix CHECK-LE %s
 ; RUN: llc --mattr=+sve -aarch64-enable-sink-fold=true < %s -mtriple aarch64_be-unknown-unknown | FileCheck --check-prefix CHECK-BE %s
 
 define <4 x double> @test_ldnp_v4f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4f64:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v4f64:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v4f64:
 ; CHECK-BE:       // %bb.0:
@@ -17,10 +17,10 @@ define <4 x double> @test_ldnp_v4f64(ptr %A) {
 }
 
 define <4 x i64> @test_ldnp_v4i64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4i64:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v4i64:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v4i64:
 ; CHECK-BE:       // %bb.0:
@@ -31,10 +31,10 @@ define <4 x i64> @test_ldnp_v4i64(ptr %A) {
 }
 
 define <8 x i32> @test_ldnp_v8i32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v8i32:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v8i32:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v8i32:
 ; CHECK-BE:       // %bb.0:
@@ -45,10 +45,10 @@ define <8 x i32> @test_ldnp_v8i32(ptr %A) {
 }
 
 define <8 x float> @test_ldnp_v8f32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v8f32:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v8f32:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v8f32:
 ; CHECK-BE:       // %bb.0:
@@ -59,10 +59,10 @@ define <8 x float> @test_ldnp_v8f32(ptr %A) {
 }
 
 define <16 x i16> @test_ldnp_v16i16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16i16:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v16i16:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v16i16:
 ; CHECK-BE:       // %bb.0:
@@ -73,10 +73,10 @@ define <16 x i16> @test_ldnp_v16i16(ptr %A) {
 }
 
 define <16 x half> @test_ldnp_v16f16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16f16:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v16f16:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v16f16:
 ; CHECK-BE:       // %bb.0:
@@ -87,10 +87,10 @@ define <16 x half> @test_ldnp_v16f16(ptr %A) {
 }
 
 define <32 x i8> @test_ldnp_v32i8(ptr %A) {
-; CHECK-LABEL: test_ldnp_v32i8:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v32i8:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v32i8:
 ; CHECK-BE:       // %bb.0:
@@ -101,10 +101,10 @@ define <32 x i8> @test_ldnp_v32i8(ptr %A) {
 }
 
 define <4 x i32> @test_ldnp_v4i32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4i32:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr q0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v4i32:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr q0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v4i32:
 ; CHECK-BE:       // %bb.0:
@@ -115,10 +115,10 @@ define <4 x i32> @test_ldnp_v4i32(ptr %A) {
 }
 
 define <4 x float> @test_ldnp_v4f32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4f32:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr q0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v4f32:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr q0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v4f32:
 ; CHECK-BE:       // %bb.0:
@@ -129,10 +129,10 @@ define <4 x float> @test_ldnp_v4f32(ptr %A) {
 }
 
 define <8 x i16> @test_ldnp_v8i16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v8i16:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr q0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v8i16:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr q0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v8i16:
 ; CHECK-BE:       // %bb.0:
@@ -143,10 +143,10 @@ define <8 x i16> @test_ldnp_v8i16(ptr %A) {
 }
 
 define <16 x i8> @test_ldnp_v16i8(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16i8:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr q0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v16i8:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr q0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v16i8:
 ; CHECK-BE:       // %bb.0:
@@ -156,10 +156,10 @@ define <16 x i8> @test_ldnp_v16i8(ptr %A) {
   ret <16 x i8> %lv
 }
 define <2 x double> @test_ldnp_v2f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v2f64:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr q0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v2f64:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr q0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v2f64:
 ; CHECK-BE:       // %bb.0:
@@ -170,10 +170,10 @@ define <2 x double> @test_ldnp_v2f64(ptr %A) {
 }
 
 define <2 x i32> @test_ldnp_v2i32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v2i32:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr d0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v2i32:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr d0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v2i32:
 ; CHECK-BE:       // %bb.0:
@@ -184,10 +184,10 @@ define <2 x i32> @test_ldnp_v2i32(ptr %A) {
 }
 
 define <2 x float> @test_ldnp_v2f32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v2f32:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr d0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v2f32:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr d0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v2f32:
 ; CHECK-BE:       // %bb.0:
@@ -198,10 +198,10 @@ define <2 x float> @test_ldnp_v2f32(ptr %A) {
 }
 
 define <4 x i16> @test_ldnp_v4i16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4i16:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr d0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v4i16:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr d0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v4i16:
 ; CHECK-BE:       // %bb.0:
@@ -212,10 +212,10 @@ define <4 x i16> @test_ldnp_v4i16(ptr %A) {
 }
 
 define <8 x i8> @test_ldnp_v8i8(ptr %A) {
-; CHECK-LABEL: test_ldnp_v8i8:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr d0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v8i8:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr d0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v8i8:
 ; CHECK-BE:       // %bb.0:
@@ -226,10 +226,10 @@ define <8 x i8> @test_ldnp_v8i8(ptr %A) {
 }
 
 define <1 x double> @test_ldnp_v1f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v1f64:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr d0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v1f64:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr d0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v1f64:
 ; CHECK-BE:       // %bb.0:
@@ -240,10 +240,10 @@ define <1 x double> @test_ldnp_v1f64(ptr %A) {
 }
 
 define <1 x i64> @test_ldnp_v1i64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v1i64:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldr d0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v1i64:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldr d0, [x0]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v1i64:
 ; CHECK-BE:       // %bb.0:
@@ -254,11 +254,11 @@ define <1 x i64> @test_ldnp_v1i64(ptr %A) {
 }
 
 define <32 x i16> @test_ldnp_v32i16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v32i16:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ldnp q2, q3, [x0, #32]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v32i16:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v32i16:
 ; CHECK-BE:       // %bb.0:
@@ -270,11 +270,11 @@ define <32 x i16> @test_ldnp_v32i16(ptr %A) {
 }
 
 define <32 x half> @test_ldnp_v32f16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v32f16:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ldnp q2, q3, [x0, #32]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v32f16:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v32f16:
 ; CHECK-BE:       // %bb.0:
@@ -286,11 +286,11 @@ define <32 x half> @test_ldnp_v32f16(ptr %A) {
 }
 
 define <16 x i32> @test_ldnp_v16i32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16i32:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ldnp q2, q3, [x0, #32]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v16i32:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v16i32:
 ; CHECK-BE:       // %bb.0:
@@ -302,11 +302,11 @@ define <16 x i32> @test_ldnp_v16i32(ptr %A) {
 }
 
 define <16 x float> @test_ldnp_v16f32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16f32:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ldnp q2, q3, [x0, #32]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v16f32:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v16f32:
 ; CHECK-BE:       // %bb.0:
@@ -318,15 +318,15 @@ define <16 x float> @test_ldnp_v16f32(ptr %A) {
 }
 
 define <17 x float> @test_ldnp_v17f32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v17f32:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0, #32]
-; CHECK-NEXT:    ldr s2, [x0, #64]
-; CHECK-NEXT:    ldnp q3, q4, [x0]
-; CHECK-NEXT:    stp q0, q1, [x8, #32]
-; CHECK-NEXT:    stp q3, q4, [x8]
-; CHECK-NEXT:    str s2, [x8, #64]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v17f32:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0, #32]
+; CHECK-LE-NEXT:    ldr s2, [x0, #64]
+; CHECK-LE-NEXT:    ldnp q3, q4, [x0]
+; CHECK-LE-NEXT:    stp q0, q1, [x8, #32]
+; CHECK-LE-NEXT:    stp q3, q4, [x8]
+; CHECK-LE-NEXT:    str s2, [x8, #64]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v17f32:
 ; CHECK-BE:       // %bb.0:
@@ -352,27 +352,27 @@ define <17 x float> @test_ldnp_v17f32(ptr %A) {
 }
 
 define <33 x double> @test_ldnp_v33f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v33f64:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ldr d20, [x0, #256]
-; CHECK-NEXT:    ldnp q2, q3, [x0, #32]
-; CHECK-NEXT:    ldnp q4, q5, [x0, #64]
-; CHECK-NEXT:    ldnp q6, q7, [x0, #96]
-; CHECK-NEXT:    ldnp q16, q17, [x0, #128]
-; CHECK-NEXT:    ldnp q18, q19, [x0, #224]
-; CHECK-NEXT:    ldnp q21, q22, [x0, #160]
-; CHECK-NEXT:    ldnp q23, q24, [x0, #192]
-; CHECK-NEXT:    stp q0, q1, [x8]
-; CHECK-NEXT:    stp q2, q3, [x8, #32]
-; CHECK-NEXT:    stp q4, q5, [x8, #64]
-; CHECK-NEXT:    stp q6, q7, [x8, #96]
-; CHECK-NEXT:    stp q16, q17, [x8, #128]
-; CHECK-NEXT:    stp q21, q22, [x8, #160]
-; CHECK-NEXT:    stp q23, q24, [x8, #192]
-; CHECK-NEXT:    stp q18, q19, [x8, #224]
-; CHECK-NEXT:    str d20, [x8, #256]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v33f64:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ldr d20, [x0, #256]
+; CHECK-LE-NEXT:    ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT:    ldnp q4, q5, [x0, #64]
+; CHECK-LE-NEXT:    ldnp q6, q7, [x0, #96]
+; CHECK-LE-NEXT:    ldnp q16, q17, [x0, #128]
+; CHECK-LE-NEXT:    ldnp q18, q19, [x0, #224]
+; CHECK-LE-NEXT:    ldnp q21, q22, [x0, #160]
+; CHECK-LE-NEXT:    ldnp q23, q24, [x0, #192]
+; CHECK-LE-NEXT:    stp q0, q1, [x8]
+; CHECK-LE-NEXT:    stp q2, q3, [x8, #32]
+; CHECK-LE-NEXT:    stp q4, q5, [x8, #64]
+; CHECK-LE-NEXT:    stp q6, q7, [x8, #96]
+; CHECK-LE-NEXT:    stp q16, q17, [x8, #128]
+; CHECK-LE-NEXT:    stp q21, q22, [x8, #160]
+; CHECK-LE-NEXT:    stp q23, q24, [x8, #192]
+; CHECK-LE-NEXT:    stp q18, q19, [x8, #224]
+; CHECK-LE-NEXT:    str d20, [x8, #256]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v33f64:
 ; CHECK-BE:       // %bb.0:
@@ -446,13 +446,13 @@ define <33 x double> @test_ldnp_v33f64(ptr %A) {
 }
 
 define <33 x i8> @test_ldnp_v33i8(ptr %A) {
-; CHECK-LABEL: test_ldnp_v33i8:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ldr b2, [x0, #32]
-; CHECK-NEXT:    stp q0, q1, [x8]
-; CHECK-NEXT:    stur b2, [x8, #32]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v33i8:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ldr b2, [x0, #32]
+; CHECK-LE-NEXT:    stp q0, q1, [x8]
+; CHECK-LE-NEXT:    stur b2, [x8, #32]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v33i8:
 ; CHECK-BE:       // %bb.0:
@@ -470,20 +470,20 @@ define <33 x i8> @test_ldnp_v33i8(ptr %A) {
 }
 
 define <4 x i65> @test_ldnp_v4i65(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4i65:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldp x8, x9, [x0, #8]
-; CHECK-NEXT:    ldr x10, [x0, #24]
-; CHECK-NEXT:    ldrb w11, [x0, #32]
-; CHECK-NEXT:    ldr x0, [x0]
-; CHECK-NEXT:    ubfx x5, x10, #2, #1
-; CHECK-NEXT:    extr x2, x9, x8, #1
-; CHECK-NEXT:    extr x4, x10, x9, #2
-; CHECK-NEXT:    extr x6, x11, x10, #3
-; CHECK-NEXT:    ubfx x3, x9, #1, #1
-; CHECK-NEXT:    ubfx x7, x11, #3, #1
-; CHECK-NEXT:    and x1, x8, #0x1
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v4i65:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldp x8, x9, [x0, #8]
+; CHECK-LE-NEXT:    ldr x10, [x0, #24]
+; CHECK-LE-NEXT:    ldrb w11, [x0, #32]
+; CHECK-LE-NEXT:    ldr x0, [x0]
+; CHECK-LE-NEXT:    ubfx x5, x10, #2, #1
+; CHECK-LE-NEXT:    extr x2, x9, x8, #1
+; CHECK-LE-NEXT:    extr x4, x10, x9, #2
+; CHECK-LE-NEXT:    extr x6, x11, x10, #3
+; CHECK-LE-NEXT:    ubfx x3, x9, #1, #1
+; CHECK-LE-NEXT:    ubfx x7, x11, #3, #1
+; CHECK-LE-NEXT:    and x1, x8, #0x1
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v4i65:
 ; CHECK-BE:       // %bb.0:
@@ -510,17 +510,17 @@ define <4 x i65> @test_ldnp_v4i65(ptr %A) {
 }
 
 define <4 x i63> @test_ldnp_v4i63(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4i63:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldp x8, x9, [x0, #16]
-; CHECK-NEXT:    ldp x10, x11, [x0]
-; CHECK-NEXT:    extr x3, x9, x8, #61
-; CHECK-NEXT:    extr x9, x11, x10, #63
-; CHECK-NEXT:    extr x8, x8, x11, #62
-; CHECK-NEXT:    and x0, x10, #0x7fffffffffffffff
-; CHECK-NEXT:    and x1, x9, #0x7fffffffffffffff
-; CHECK-NEXT:    and x2, x8, #0x7fffffffffffffff
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v4i63:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldp x8, x9, [x0, #16]
+; CHECK-LE-NEXT:    ldp x10, x11, [x0]
+; CHECK-LE-NEXT:    extr x3, x9, x8, #61
+; CHECK-LE-NEXT:    extr x9, x11, x10, #63
+; CHECK-LE-NEXT:    extr x8, x8, x11, #62
+; CHECK-LE-NEXT:    and x0, x10, #0x7fffffffffffffff
+; CHECK-LE-NEXT:    and x1, x9, #0x7fffffffffffffff
+; CHECK-LE-NEXT:    and x2, x8, #0x7fffffffffffffff
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v4i63:
 ; CHECK-BE:       // %bb.0:
@@ -539,17 +539,17 @@ define <4 x i63> @test_ldnp_v4i63(ptr %A) {
 }
 
 define <5 x double> @test_ldnp_v5f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v5f64:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q2, [x0]
-; CHECK-NEXT:    ldr d4, [x0, #32]
-; CHECK-NEXT:    ext.16b v1, v0, v0, #8
-; CHECK-NEXT:    ext.16b v3, v2, v2, #8
-; CHECK-NEXT:    ; kill: def $d0 killed $d0 killed $q0
-; CHECK-NEXT:    ; kill: def $d2 killed $d2 killed $q2
-; CHECK-NEXT:    ; kill: def $d1 killed $d1 killed $q1
-; CHECK-NEXT:    ; kill: def $d3 killed $d3 killed $q3
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v5f64:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q2, [x0]
+; CHECK-LE-NEXT:    ldr d4, [x0, #32]
+; CHECK-LE-NEXT:    ext.16b v1, v0, v0, #8
+; CHECK-LE-NEXT:    ext.16b v3, v2, v2, #8
+; CHECK-LE-NEXT:    ; kill: def $d0 killed $d0 killed $q0
+; CHECK-LE-NEXT:    ; kill: def $d2 killed $d2 killed $q2
+; CHECK-LE-NEXT:    ; kill: def $d1 killed $d1 killed $q1
+; CHECK-LE-NEXT:    ; kill: def $d3 killed $d3 killed $q3
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v5f64:
 ; CHECK-BE:       // %bb.0:
@@ -570,13 +570,13 @@ define <5 x double> @test_ldnp_v5f64(ptr %A) {
 }
 
 define <16 x i64> @test_ldnp_v16i64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16i64:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ldnp q2, q3, [x0, #32]
-; CHECK-NEXT:    ldnp q4, q5, [x0, #64]
-; CHECK-NEXT:    ldnp q6, q7, [x0, #96]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v16i64:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT:    ldnp q4, q5, [x0, #64]
+; CHECK-LE-NEXT:    ldnp q6, q7, [x0, #96]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v16i64:
 ; CHECK-BE:       // %bb.0:
@@ -590,13 +590,13 @@ define <16 x i64> @test_ldnp_v16i64(ptr %A) {
 }
 
 define <16 x double> @test_ldnp_v16f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16f64:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ldnp q0, q1, [x0]
-; CHECK-NEXT:    ldnp q2, q3, [x0, #32]
-; CHECK-NEXT:    ldnp q4, q5, [x0, #64]
-; CHECK-NEXT:    ldnp q6, q7, [x0, #96]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v16f64:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ldnp q0, q1, [x0]
+; CHECK-LE-NEXT:    ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT:    ldnp q4, q5, [x0, #64]
+; CHECK-LE-NEXT:    ldnp q6, q7, [x0, #96]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v16f64:
 ; CHECK-BE:       // %bb.0:
@@ -610,15 +610,15 @@ define <16 x double> @test_ldnp_v16f64(ptr %A) {
 }
 
 define <vscale x 20 x float> @test_ldnp_v20f32_vscale(ptr %A) {
-; CHECK-LABEL: test_ldnp_v20f32_vscale:
-; CHECK:       ; %bb.0:
-; CHECK-NEXT:    ptrue p0.s
-; CHECK-NEXT:    ldnt1w { z0.s }, p0/z, [x0]
-; CHECK-NEXT:    ldnt1w { z1.s }, p0/z, [x0, #1, mul vl]
-; CHECK-NEXT:    ldnt1w { z2.s }, p0/z, [x0, #2, mul vl]
-; CHECK-NEXT:    ldnt1w { z3.s }, p0/z, [x0, #3, mul vl]
-; CHECK-NEXT:    ldnt1w { z4.s }, p0/z, [x0, #4, mul vl]
-; CHECK-NEXT:    ret
+; CHECK-LE-LABEL: test_ldnp_v20f32_vscale:
+; CHECK-LE:       ; %bb.0:
+; CHECK-LE-NEXT:    ptrue p0.s
+; CHECK-LE-NEXT:    ldnt1w { z0.s }, p0/z, [x0]
+; CHECK-LE-NEXT:    ldnt1w { z1.s }, p0/z, [x0, #1, mul vl]
+; CHECK-LE-NEXT:    ldnt1w { z2.s }, p0/z, [x0, #2, mul vl]
+; CHECK-LE-NEXT:    ldnt1w { z3.s }, p0/z, [x0, #3, mul vl]
+; CHECK-LE-NEXT:    ldnt1w { z4.s }, p0/z, [x0, #4, mul vl]
+; CHECK-LE-NEXT:    ret
 ;
 ; CHECK-BE-LABEL: test_ldnp_v20f32_vscale:
 ; CHECK-BE:       // %bb.0:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll b/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
index c7edf9bdfaf6b..83a5d1b60a400 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
@@ -1,11 +1,17 @@
-; RUN: opt -passes=loop-vectorize -mtriple=arm64-apple-iphones -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64 -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK-LE
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64_be -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK-BE
 
 ; Vectors with i4 elements may not legal with nontemporal stores.
 define void @test_i4_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_i4_store(
-; CHECK-NOT:   vector.body:
-; CHECK:        ret void
+; CHECK-LE-LABEL: define void @test_i4_store(
+; CHECK-LE-NOT:   vector.body:
+; CHECK-LE:        store i4 {{.*}} !nontemporal !0
+; CHECK-LE:        ret void
 ;
+; CHECK-BE-LABEL: define void @test_i4_store(
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE:        store i4 {{.*}} !nontemporal !0
+; CHECK-BE:        ret void
 entry:
   br label %for.body
 
@@ -23,11 +29,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_i8_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_i8_store(
-; CHECK-LABEL: vector.body:
-; CHECK:         store <4 x i8> {{.*}} !nontemporal !0
-; CHECK:         br
+; CHECK-LE-LABEL: define void @test_i8_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE:         store <4 x i8> {{.*}} !nontemporal !0
+; CHECK-LE:         br
 ;
+; CHECK-BE-LABEL: define void @test_i8_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE:        store i8 {{.*}} !nontemporal !0
+; CHECK-BE:         br
 entry:
   br label %for.body
 
@@ -45,11 +55,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_half_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_half_store(
-; CHECK-LABEL: vector.body:
-; CHECK:         store <4 x half> {{.*}} !nontemporal !0
-; CHECK:         br
+; CHECK-LE-LABEL: define void @test_half_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE:         store <4 x half> {{.*}} !nontemporal !0
+; CHECK-LE:         br
 ;
+; CHECK-BE-LABEL: define void @test_half_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE:         store half {{.*}} !nontemporal !0
+; CHECK-BE:         br
 entry:
   br label %for.body
 
@@ -67,11 +81,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_i16_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_i16_store(
-; CHECK-LABEL: vector.body:
-; CHECK:         store <4 x i16> {{.*}} !nontemporal !0
-; CHECK:         br
+; CHECK-LE-LABEL: define void @test_i16_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE:         store <4 x i16> {{.*}} !nontemporal !0
+; CHECK-LE:         br
 ;
+; CHECK-BE-LABEL: define void @test_i16_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE:         store i16 {{.*}} !nontemporal !0
+; CHECK-BE:         br
 entry:
   br label %for.body
 
@@ -89,11 +107,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_i32_store(ptr nocapture %ddst) {
-; CHECK-LABEL: define void @test_i32_store(
-; CHECK-LABEL: vector.body:
-; CHECK:         store <16 x i32> {{.*}} !nontemporal !0
-; CHECK:         br
+; CHECK-LE-LABEL: define void @test_i32_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE:         store <16 x i32> {{.*}} !nontemporal !0
+; CHECK-LE:         br
 ;
+; CHECK-BE-LABEL: define void @test_i32_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE:         store i32 {{.*}} !nontemporal !0
+; CHECK-BE:         br
 entry:
   br label %for.body
 
@@ -117,10 +139,13 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_i33_store(ptr nocapture %ddst) {
-; CHECK-LABEL: define void @test_i33_store(
-; CHECK-NOT:   vector.body:
-; CHECK:         ret
+; CHECK-LE-LABEL: define void @test_i33_store(
+; CHECK-LE-NOT:   vector.body:
+; CHECK-LE:         ret
 ;
+; CHECK-BE-LABEL: define void @test_i33_store(
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE:         ret
 entry:
   br label %for.body
 
@@ -144,10 +169,13 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_i40_store(ptr nocapture %ddst) {
-; CHECK-LABEL: define void @test_i40_store(
-; CHECK-NOT:   vector.body:
-; CHECK:         ret
+; CHECK-LE-LABEL: define void @test_i40_store(
+; CHECK-LE-NOT:   vector.body:
+; CHECK-LE:         ret
 ;
+; CHECK-BE-LABEL: define void @test_i40_store(
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE:         ret
 entry:
   br label %for.body
 
@@ -170,11 +198,15 @@ for.cond.cleanup:                                 ; preds = %for.body
   ret void
 }
 define void @test_i64_store(ptr nocapture %ddst) local_unnamed_addr #0 {
-; CHECK-LABEL: define void @test_i64_store(
-; CHECK-LABEL: vector.body:
-; CHECK:         store <4 x i64> {{.*}} !nontemporal !0
-; CHECK:         br
+; CHECK-LE-LABEL: define void @test_i64_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE:         store <4 x i64> {{.*}} !nontemporal !0
+; CHECK-LE:         br
 ;
+; CHECK-BE-LABEL: define void @test_i64_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE:         store i64 {{.*}} !nontemporal !0
+; CHECK-BE:         br
 entry:
   br label %for.body
 
@@ -192,11 +224,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_double_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_double_store(
-; CHECK-LABEL: vector.body:
-; CHECK:         store <4 x double> {{.*}} !nontemporal !0
-; CHECK:         br
+; CHECK-LE-LABEL: define void @test_double_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE:         store <4 x double> {{.*}} !nontemporal !0
+; CHECK-LE:         br
 ;
+; CHECK-BE-LABEL: define void @test_double_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE:         store double {{.*}} !nontemporal !0
+; CHECK-BE:         br
 entry:
   br label %for.body
 
@@ -214,11 +250,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_i128_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_i128_store(
-; CHECK-LABEL: vector.body:
-; CHECK:         store <4 x i128> {{.*}} !nontemporal !0
-; CHECK:         br
+; CHECK-LE-LABEL: define void @test_i128_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE:         store <4 x i128> {{.*}} !nontemporal !0
+; CHECK-LE:         br
 ;
+; CHECK-BE-LABEL: define void @test_i128_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE:         store i128 {{.*}} !nontemporal !0
+; CHECK-BE:         br
 entry:
   br label %for.body
 
@@ -236,10 +276,13 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_i256_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_i256_store(
-; CHECK-NOT:   vector.body:
-; CHECK:        ret void
+; CHECK-LE-LABEL: define void @test_i256_store(
+; CHECK-LE-NOT:   vector.body:
+; CHECK-LE:        ret void
 ;
+; CHECK-BE-LABEL: define void @test_i256_store(
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE:        ret void
 entry:
   br label %for.body
 
@@ -257,10 +300,13 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i4 @test_i4_load(ptr %ddst) {
-; CHECK-LABEL: define i4 @test_i4_load
-; CHECK-NOT: vector.body:
-; CHECK: ret i4 %{{.*}}
+; CHECK-LE-LABEL: define i4 @test_i4_load
+; CHECK-LE-NOT: vector.body:
+; CHECK-LE: ret i4 %{{.*}}
 ;
+; CHECK-BE-LABEL: define i4 @test_i4_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: ret i4 %{{.*}}
 entry:
   br label %for.body
 
@@ -279,11 +325,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i8 @test_load_i8(ptr %ddst) {
-; CHECK-LABEL: @test_load_i8(
-; CHECK:   vector.body:
-; CHECK: load <4 x i8>, ptr {{.*}}, align 1, !nontemporal !0
-; CHECK: ret i8 %{{.*}}
+; CHECK-LE-LABEL: @test_load_i8(
+; CHECK-LE-LABEL:   vector.body:
+; CHECK-LE: load <4 x i8>, ptr {{.*}}, align 1, !nontemporal !0
+; CHECK-LE: ret i8 %{{.*}}
 ;
+; CHECK-BE-LABEL: @test_load_i8(
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE: load i8, ptr {{.*}}, align 1, !nontemporal !0
+; CHECK-BE: ret i8 %{{.*}}
 entry:
   br label %for.body
 
@@ -302,11 +352,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define half @test_half_load(ptr %ddst) {
-; CHECK-LABEL: @test_half_load
-; CHECK-LABEL:   vector.body:
-; CHECK: load <4 x half>, ptr {{.*}}, align 2, !nontemporal !0
-; CHECK: ret half %{{.*}}
+; CHECK-LE-LABEL: @test_half_load
+; CHECK-LE-LABEL:   vector.body:
+; CHECK-LE: load <4 x half>, ptr {{.*}}, align 2, !nontemporal !0
+; CHECK-LE: ret half %{{.*}}
 ;
+; CHECK-BE-LABEL: @test_half_load
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE: load half, ptr {{.*}}, align 2, !nontemporal !0
+; CHECK-BE: ret half %{{.*}}
 entry:
   br label %for.body
 
@@ -325,11 +379,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i16 @test_i16_load(ptr %ddst) {
-; CHECK-LABEL: @test_i16_load
-; CHECK-LABEL:   vector.body:
-; CHECK: load <4 x i16>, ptr {{.*}}, align 2, !nontemporal !0
-; CHECK: ret i16 %{{.*}}
+; CHECK-LE-LABEL: @test_i16_load
+; CHECK-LE-LABEL:   vector.body:
+; CHECK-LE: load <4 x i16>, ptr {{.*}}, align 2, !nontemporal !0
+; CHECK-LE: ret i16 %{{.*}}
 ;
+; CHECK-BE-LABEL: @test_i16_load
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE: load i16, ptr {{.*}}, align 2, !nontemporal !0
+; CHECK-BE: ret i16 %{{.*}}
 entry:
   br label %for.body
 
@@ -348,11 +406,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i32 @test_i32_load(ptr %ddst) {
-; CHECK-LABEL: @test_i32_load
-; CHECK-LABEL:   vector.body:
-; CHECK: load <4 x i32>, ptr {{.*}}, align 4, !nontemporal !0
-; CHECK: ret i32 %{{.*}}
+; CHECK-LE-LABEL: @test_i32_load
+; CHECK-LE-LABEL:   vector.body:
+; CHECK-LE: load <4 x i32>, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-LE: ret i32 %{{.*}}
 ;
+; CHECK-BE-LABEL: @test_i32_load
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE: load i32, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-BE: ret i32 %{{.*}}
 entry:
   br label %for.body
 
@@ -371,10 +433,13 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i33 @test_i33_load(ptr %ddst) {
-; CHECK-LABEL: @test_i33_load
-; CHECK-NOT:   vector.body:
-; CHECK: ret i33 %{{.*}}
+; CHECK-LE-LABEL: @test_i33_load
+; CHECK-LE-NOT:   vector.body:
+; CHECK-LE: ret i33 %{{.*}}
 ;
+; CHECK-BE-LABEL: @test_i33_load
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE: ret i33 %{{.*}}
 entry:
   br label %for.body
 
@@ -393,10 +458,13 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i40 @test_i40_load(ptr %ddst) {
-; CHECK-LABEL: @test_i40_load
-; CHECK-NOT:   vector.body:
-; CHECK: ret i40 %{{.*}}
+; CHECK-LE-LABEL: @test_i40_load
+; CHECK-LE-NOT:   vector.body:
+; CHECK-LE: ret i40 %{{.*}}
 ;
+; CHECK-BE-LABEL: @test_i40_load
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE: ret i40 %{{.*}}
 entry:
   br label %for.body
 
@@ -415,11 +483,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i64 @test_i64_load(ptr %ddst) {
-; CHECK-LABEL: @test_i64_load
-; CHECK-LABEL:   vector.body:
-; CHECK: load <4 x i64>, ptr {{.*}}, align 4, !nontemporal !0
-; CHECK: ret i64 %{{.*}}
+; CHECK-LE-LABEL: @test_i64_load
+; CHECK-LE-LABEL:   vector.body:
+; CHECK-LE: load <4 x i64>, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-LE: ret i64 %{{.*}}
 ;
+; CHECK-BE-LABEL: @test_i64_load
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE: load i64, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-BE: ret i64 %{{.*}}
 entry:
   br label %for.body
 
@@ -438,11 +510,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define double @test_double_load(ptr %ddst) {
-; CHECK-LABEL: @test_double_load
-; CHECK-LABEL:   vector.body:
-; CHECK: load <4 x double>, ptr {{.*}}, align 4, !nontemporal !0
-; CHECK: ret double %{{.*}}
+; CHECK-LE-LABEL: @test_double_load
+; CHECK-LE-LABEL:   vector.body:
+; CHECK-LE: load <4 x double>, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-LE: ret double %{{.*}}
 ;
+; CHECK-BE-LABEL: @test_double_load
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE: load double, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-BE: ret double %{{.*}}
 entry:
   br label %for.body
 
@@ -461,11 +537,15 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i128 @test_i128_load(ptr %ddst) {
-; CHECK-LABEL: @test_i128_load
-; CHECK-LABEL:   vector.body:
-; CHECK: load <4 x i128>, ptr {{.*}}, align 4, !nontemporal !0
-; CHECK: ret i128 %{{.*}}
+; CHECK-LE-LABEL: @test_i128_load
+; CHECK-LE-LABEL:   vector.body:
+; CHECK-LE: load <4 x i128>, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-LE: ret i128 %{{.*}}
 ;
+; CHECK-BE-LABEL: @test_i128_load
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE: load i128, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-BE: ret i128 %{{.*}}
 entry:
   br label %for.body
 
@@ -484,10 +564,13 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i256 @test_256_load(ptr %ddst) {
-; CHECK-LABEL: @test_256_load
-; CHECK-NOT:   vector.body:
-; CHECK: ret i256 %{{.*}}
+; CHECK-LE-LABEL: @test_256_load
+; CHECK-LE-NOT:   vector.body:
+; CHECK-LE: ret i256 %{{.*}}
 ;
+; CHECK-BE-LABEL: @test_256_load
+; CHECK-BE-NOT:   vector.body:
+; CHECK-BE: ret i256 %{{.*}}
 entry:
   br label %for.body
 

>From 05cdd55ed4bc37dab8c7ac7328a404f62049b9ed Mon Sep 17 00:00:00 2001
From: tomershafir <tomer.shafir8 at gmail.com>
Date: Mon, 26 Jan 2026 13:19:21 +0200
Subject: [PATCH 2/5] fix typo

---
 llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index eef52867107e9..871e42e4a3061 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -7323,7 +7323,7 @@ static SDValue LowerADDRSPACECAST(SDValue Op, SelectionDAG &DAG) {
 
 // Lower non-temporal stores that would otherwise be broken by legalization.
 //
-// Coordinated with LDNP constraints in
+// Coordinated with SDNP constraints in
 // `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
 // `AArch64TargetLowering::ReplaceNodeResults`
 static SDValue LowerNTStore(StoreSDNode *StoreNode, EVT VT, EVT MemVT,

>From d2348536b9cbaafa79ad2946277cbb38b3e9754b Mon Sep 17 00:00:00 2001
From: tomershafir <tomer.shafir8 at gmail.com>
Date: Mon, 26 Jan 2026 13:26:32 +0200
Subject: [PATCH 3/5] match common code with CHECK prefix

---
 .../AArch64/nontemporal-load-store.ll         | 87 ++++++-------------
 1 file changed, 27 insertions(+), 60 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll b/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
index 83a5d1b60a400..1ddc18142d127 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
@@ -1,17 +1,12 @@
-; RUN: opt -passes=loop-vectorize -mtriple=aarch64 -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK-LE
-; RUN: opt -passes=loop-vectorize -mtriple=aarch64_be -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK-BE
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64 -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK,CHECK-LE
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64_be -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK,CHECK-BE
 
 ; Vectors with i4 elements may not legal with nontemporal stores.
 define void @test_i4_store(ptr %ddst) {
-; CHECK-LE-LABEL: define void @test_i4_store(
-; CHECK-LE-NOT:   vector.body:
-; CHECK-LE:        store i4 {{.*}} !nontemporal !0
-; CHECK-LE:        ret void
-;
-; CHECK-BE-LABEL: define void @test_i4_store(
-; CHECK-BE-NOT:   vector.body:
-; CHECK-BE:        store i4 {{.*}} !nontemporal !0
-; CHECK-BE:        ret void
+; CHECK-LABEL: define void @test_i4_store(
+; CHECK-NOT:   vector.body:
+; CHECK:        store i4 {{.*}} !nontemporal !0
+; CHECK:        ret void
 entry:
   br label %for.body
 
@@ -139,13 +134,9 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_i33_store(ptr nocapture %ddst) {
-; CHECK-LE-LABEL: define void @test_i33_store(
-; CHECK-LE-NOT:   vector.body:
-; CHECK-LE:         ret
-;
-; CHECK-BE-LABEL: define void @test_i33_store(
-; CHECK-BE-NOT:   vector.body:
-; CHECK-BE:         ret
+; CHECK-LABEL: define void @test_i33_store(
+; CHECK-NOT:   vector.body:
+; CHECK:         ret
 entry:
   br label %for.body
 
@@ -169,13 +160,9 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_i40_store(ptr nocapture %ddst) {
-; CHECK-LE-LABEL: define void @test_i40_store(
-; CHECK-LE-NOT:   vector.body:
-; CHECK-LE:         ret
-;
-; CHECK-BE-LABEL: define void @test_i40_store(
-; CHECK-BE-NOT:   vector.body:
-; CHECK-BE:         ret
+; CHECK-LABEL: define void @test_i40_store(
+; CHECK-NOT:   vector.body:
+; CHECK:         ret
 entry:
   br label %for.body
 
@@ -276,13 +263,9 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define void @test_i256_store(ptr %ddst) {
-; CHECK-LE-LABEL: define void @test_i256_store(
-; CHECK-LE-NOT:   vector.body:
-; CHECK-LE:        ret void
-;
-; CHECK-BE-LABEL: define void @test_i256_store(
-; CHECK-BE-NOT:   vector.body:
-; CHECK-BE:        ret void
+; CHECK-LABEL: define void @test_i256_store(
+; CHECK-NOT:   vector.body:
+; CHECK:        ret void
 entry:
   br label %for.body
 
@@ -300,13 +283,9 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i4 @test_i4_load(ptr %ddst) {
-; CHECK-LE-LABEL: define i4 @test_i4_load
-; CHECK-LE-NOT: vector.body:
-; CHECK-LE: ret i4 %{{.*}}
-;
-; CHECK-BE-LABEL: define i4 @test_i4_load
-; CHECK-BE-NOT: vector.body:
-; CHECK-BE: ret i4 %{{.*}}
+; CHECK-LABEL: define i4 @test_i4_load
+; CHECK-NOT: vector.body:
+; CHECK: ret i4 %{{.*}}
 entry:
   br label %for.body
 
@@ -433,13 +412,9 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i33 @test_i33_load(ptr %ddst) {
-; CHECK-LE-LABEL: @test_i33_load
-; CHECK-LE-NOT:   vector.body:
-; CHECK-LE: ret i33 %{{.*}}
-;
-; CHECK-BE-LABEL: @test_i33_load
-; CHECK-BE-NOT:   vector.body:
-; CHECK-BE: ret i33 %{{.*}}
+; CHECK-LABEL: @test_i33_load
+; CHECK-NOT:   vector.body:
+; CHECK: ret i33 %{{.*}}
 entry:
   br label %for.body
 
@@ -458,13 +433,9 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i40 @test_i40_load(ptr %ddst) {
-; CHECK-LE-LABEL: @test_i40_load
-; CHECK-LE-NOT:   vector.body:
-; CHECK-LE: ret i40 %{{.*}}
-;
-; CHECK-BE-LABEL: @test_i40_load
-; CHECK-BE-NOT:   vector.body:
-; CHECK-BE: ret i40 %{{.*}}
+; CHECK-LABEL: @test_i40_load
+; CHECK-NOT:   vector.body:
+; CHECK: ret i40 %{{.*}}
 entry:
   br label %for.body
 
@@ -564,13 +535,9 @@ for.cond.cleanup:                                 ; preds = %for.body
 }
 
 define i256 @test_256_load(ptr %ddst) {
-; CHECK-LE-LABEL: @test_256_load
-; CHECK-LE-NOT:   vector.body:
-; CHECK-LE: ret i256 %{{.*}}
-;
-; CHECK-BE-LABEL: @test_256_load
-; CHECK-BE-NOT:   vector.body:
-; CHECK-BE: ret i256 %{{.*}}
+; CHECK-LABEL: @test_256_load
+; CHECK-NOT:   vector.body:
+; CHECK: ret i256 %{{.*}}
 entry:
   br label %for.body
 

>From 7482684e80ff96062a52602ef6fce10f2bbfdd29 Mon Sep 17 00:00:00 2001
From: tomershafir <tomer.shafir8 at gmail.com>
Date: Mon, 26 Jan 2026 13:40:26 +0200
Subject: [PATCH 4/5] fix typo

---
 llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 871e42e4a3061..b1b8100391338 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -7323,7 +7323,7 @@ static SDValue LowerADDRSPACECAST(SDValue Op, SelectionDAG &DAG) {
 
 // Lower non-temporal stores that would otherwise be broken by legalization.
 //
-// Coordinated with SDNP constraints in
+// Coordinated with STNP constraints in
 // `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
 // `AArch64TargetLowering::ReplaceNodeResults`
 static SDValue LowerNTStore(StoreSDNode *StoreNode, EVT VT, EVT MemVT,

>From e3c6b7ff5f53f69ec4a0a47426f4574b2fde1c4d Mon Sep 17 00:00:00 2001
From: tomershafir <tomer.shafir8 at gmail.com>
Date: Mon, 26 Jan 2026 14:13:30 +0200
Subject: [PATCH 5/5] [TTI] Add distinguished shouldVectorizeNTStore/Load

The issue this patch tries to solve is to differentiate between is an NT store/load directly legal on the target, and should an NT store/load be considered for vectorization early before we know the final vectorized store type. The differentiation is necessary for later passes in the pipeline to check legality without being over-permissive.

It adds 2 new TTI hooks `shouldVectorizeNTStore()` and `shouldVectorizeNTLoad()`, to be used by `LoopVectorizationLegality` when checking instruction legality before vectorization. In addition, it sinks `isLegalNTStore` and `isLegalNTLoad` to `TargetLowering.h` so that they can be easily called from the backend through TLI. I intend to submit a followup patch to introduce a new user of the TLI hooks, but I think this semantic differentiation deserves a patch of its own - this is effectively a NFC.

The default implementation of the new hooks has not changed, and they delegate to corresponding `isLegal` checks.

Note: this patch removes the AArch64 override for `isLegalNTStore` as its currently not used anymore, only to re-introduce it in a followup patch with a fixed implementation based on what the target actually supports, and a new user.
---
 .../llvm/Analysis/TargetTransformInfo.h       | 11 ++++--
 .../llvm/Analysis/TargetTransformInfoImpl.h   | 18 +++------
 llvm/include/llvm/CodeGen/BasicTTIImpl.h      | 10 +++++
 llvm/include/llvm/CodeGen/TargetLowering.h    | 18 +++++++++
 llvm/lib/Analysis/TargetTransformInfo.cpp     | 11 +++---
 .../Target/AArch64/AArch64ISelLowering.cpp    |  2 +-
 llvm/lib/Target/AArch64/AArch64InstrInfo.td   |  6 +--
 .../AArch64/AArch64TargetTransformInfo.h      | 18 ++++-----
 llvm/lib/Target/X86/X86ISelLowering.cpp       | 37 +++++++++++++++++++
 llvm/lib/Target/X86/X86ISelLowering.h         |  5 +++
 .../lib/Target/X86/X86TargetTransformInfo.cpp | 35 ------------------
 llvm/lib/Target/X86/X86TargetTransformInfo.h  |  2 -
 .../Vectorize/LoopVectorizationLegality.cpp   | 13 ++++---
 13 files changed, 110 insertions(+), 76 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index 9db2e3977f71c..dc1dd88cad140 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -896,10 +896,13 @@ class TargetTransformInfo {
   isLegalMaskedLoad(Type *DataType, Align Alignment, unsigned AddressSpace,
                     MaskKind MaskKind = VariableOrConstantMask) const;
 
-  /// Return true if the target supports nontemporal store.
-  LLVM_ABI bool isLegalNTStore(Type *DataType, Align Alignment) const;
-  /// Return true if the target supports nontemporal load.
-  LLVM_ABI bool isLegalNTLoad(Type *DataType, Align Alignment) const;
+  /// Returns true if the vectorization should try vectorize nontemporal stores.
+  /// Currently only used by the loop vectorizer.
+  LLVM_ABI bool shouldVectorizeNTStore(Type *DataType, Align Alignment) const;
+
+  /// Returns true if the vectorization should try vectorize nontemporal loads.
+  /// Currently only used by the loop vectorizer.
+  LLVM_ABI bool shouldVectorizeNTLoad(Type *DataType, Align Alignment) const;
 
   /// \Returns true if the target supports broadcasting a load to a vector of
   /// type <NumElements x ElementTy>.
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 07b3755924fd1..06dae244b260e 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -362,18 +362,12 @@ class TargetTransformInfoImplBase {
     return false;
   }
 
-  virtual bool isLegalNTStore(Type *DataType, Align Alignment) const {
-    // By default, assume nontemporal memory stores are available for stores
-    // that are aligned and have a size that is a power of 2.
-    unsigned DataSize = DL.getTypeStoreSize(DataType);
-    return Alignment >= DataSize && isPowerOf2_32(DataSize);
-  }
-
-  virtual bool isLegalNTLoad(Type *DataType, Align Alignment) const {
-    // By default, assume nontemporal memory loads are available for loads that
-    // are aligned and have a size that is a power of 2.
-    unsigned DataSize = DL.getTypeStoreSize(DataType);
-    return Alignment >= DataSize && isPowerOf2_32(DataSize);
+  virtual bool shouldVectorizeNTStore(Type *DataType, Align Alignment) const {
+    return false;
+  }
+
+  virtual bool shouldVectorizeNTLoad(Type *DataType, Align Alignment) const {
+    return false;
   }
 
   virtual bool isLegalBroadcastLoad(Type *ElementTy,
diff --git a/llvm/include/llvm/CodeGen/BasicTTIImpl.h b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
index c430e11168f73..b296fdd6b1fd5 100644
--- a/llvm/include/llvm/CodeGen/BasicTTIImpl.h
+++ b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
@@ -476,6 +476,16 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
     return getTLI()->isLegalAddressingMode(DL, AM, Ty, AddrSpace, I);
   }
 
+  bool shouldVectorizeNTStore(Type *DataType, Align Alignment) const override {
+    const DataLayout &DL = this->getDataLayout();
+    return getTLI()->isLegalNTStore(DataType, Alignment, DL);
+  }
+
+  bool shouldVectorizeNTLoad(Type *DataType, Align Alignment) const override {
+    const DataLayout &DL = this->getDataLayout();
+    return getTLI()->isLegalNTLoad(DataType, Alignment, DL);
+  }
+
   int64_t getPreferredLargeGEPBaseOffset(int64_t MinOffset, int64_t MaxOffset) {
     return getTLI()->getPreferredLargeGEPBaseOffset(MinOffset, MaxOffset);
   }
diff --git a/llvm/include/llvm/CodeGen/TargetLowering.h b/llvm/include/llvm/CodeGen/TargetLowering.h
index d8b5274d37c85..d8c53db1c8ad7 100644
--- a/llvm/include/llvm/CodeGen/TargetLowering.h
+++ b/llvm/include/llvm/CodeGen/TargetLowering.h
@@ -3003,6 +3003,24 @@ class LLVM_ABI TargetLoweringBase {
     return Value == 0;
   }
 
+  /// Return true if the target supports nontemporal store.
+  virtual bool isLegalNTStore(Type *DataType, Align Alignment,
+                              const DataLayout &DL) const {
+    // By default, assume nontemporal memory stores are available for stores
+    // that are aligned and have a size that is a power of 2.
+    unsigned DataSize = DL.getTypeStoreSize(DataType);
+    return Alignment >= DataSize && isPowerOf2_32(DataSize);
+  }
+
+  /// Return true if the target supports nontemporal load.
+  virtual bool isLegalNTLoad(Type *DataType, Align Alignment,
+                             const DataLayout &DL) const {
+    // By default, assume nontemporal memory loads are available for loads that
+    // are aligned and have a size that is a power of 2.
+    unsigned DataSize = DL.getTypeStoreSize(DataType);
+    return Alignment >= DataSize && isPowerOf2_32(DataSize);
+  }
+
   /// Given a shuffle vector SVI representing a vector splat, return a new
   /// scalar type of size equal to SVI's scalar type if the new type is more
   /// profitable. Returns nullptr otherwise. For example under MVE float splats
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index b2b77da4914d6..bdf8610eaeac7 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -490,13 +490,14 @@ bool TargetTransformInfo::isLegalMaskedLoad(Type *DataType, Align Alignment,
                                     MaskKind);
 }
 
-bool TargetTransformInfo::isLegalNTStore(Type *DataType,
-                                         Align Alignment) const {
-  return TTIImpl->isLegalNTStore(DataType, Alignment);
+bool TargetTransformInfo::shouldVectorizeNTStore(Type *DataType,
+                                                 Align Alignment) const {
+  return TTIImpl->shouldVectorizeNTStore(DataType, Alignment);
 }
 
-bool TargetTransformInfo::isLegalNTLoad(Type *DataType, Align Alignment) const {
-  return TTIImpl->isLegalNTLoad(DataType, Alignment);
+bool TargetTransformInfo::shouldVectorizeNTLoad(Type *DataType,
+                                                Align Alignment) const {
+  return TTIImpl->shouldVectorizeNTLoad(DataType, Alignment);
 }
 
 bool TargetTransformInfo::isLegalBroadcastLoad(Type *ElementTy,
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index b1b8100391338..8113375ce52c3 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -29573,7 +29573,7 @@ void AArch64TargetLowering::ReplaceNodeResults(
     //
     // Coordinated with LDNP constraints in
     // `llvm/lib/Target/AArch64/AArch64InstrInfo.td`
-    // and `AArch64TTIImpl::isLegalNTLoad`.
+    // and `AArch64TargetLowering::isLegalNTLoad`.
     if (LoadNode->isNonTemporal() && Subtarget->isLittleEndian() &&
         MemVT.getSizeInBits() == 256u &&
         (MemVT.getScalarSizeInBits() == 8u ||
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.td b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
index ee9e3fa886a5c..ba5a4e0fa4b53 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
@@ -3789,7 +3789,7 @@ def : Pat<(AArch64ldp (am_indexed7s64 GPR64sp:$Rn, simm7s8:$offset)),
 // Currently we only support NT loads lowering for little-endian targets.
 //
 // Coordinated with LDNP constraints in `AArch64TargetLowering::ReplaceNodeResults`
-// and `AArch64TTIImpl::isLegalNTLoad`.
+// and `AArch64TargetLowering::isLegalNTLoad`.
 let Predicates = [IsLE] in {
   def : Pat<(AArch64ldnp(am_indexed7s128 GPR64sp:$Rn, simm7s16:$offset)),
             (LDNPQi GPR64sp:$Rn, simm7s16:$offset)>;
@@ -10733,8 +10733,8 @@ def : Pat<(i64 (int_aarch64_neon_urshl (i64 FPR64:$Rn), (i64 FPR64:$Rm))),
 //
 // Currently we only support NT stores lowering for little-endian targets.
 //
-// Coordinated with STNP constraints in `AArch64TargetLowering::LowerNTStore`
-// and `AArch64TTIImpl::isLegalNTStore`.
+// Coordinated with STNP constraints in `LowerNTStore`
+// and `AArch64TargetLowering::isLegalNTStore`.
 //
 // Currently, STNP lowering can only either keep or increase code size, thus
 // we predicate it to not apply when optimizing for code size.
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index b789e13bf25da..1b6da612e76d4 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -391,8 +391,8 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
     return false;
   }
 
-  std::optional<bool> isLegalNTStoreLoad(Type *DataType,
-                                         Align Alignment) const {
+  std::optional<bool> shouldVectorizeNTStoreLoad(Type *DataType,
+                                                 Align Alignment) const {
     // NOTE: The logic below is mostly geared towards LV, which calls it with
     //       vectors with 2 elements. We might want to improve that, if other
     //       users show up.
@@ -409,21 +409,21 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
     return std::nullopt;
   }
 
-  bool isLegalNTStore(Type *DataType, Align Alignment) const override {
+  bool shouldVectorizeNTStore(Type *DataType, Align Alignment) const override {
     // Currently we only support NT stores lowering for little-endian targets.
     //
     // Coordinated with STNP constraints in
     // `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
-    // `AArch64TargetLowering::LowerNTStore`
+    // `LowerNTStore`
     if (!ST->isLittleEndian())
       return false;
-    if (auto Result = isLegalNTStoreLoad(DataType, Alignment))
+    if (auto Result = shouldVectorizeNTStoreLoad(DataType, Alignment))
       return *Result;
     // Fallback to target independent logic
-    return BaseT::isLegalNTStore(DataType, Alignment);
+    return BaseT::shouldVectorizeNTStore(DataType, Alignment);
   }
 
-  bool isLegalNTLoad(Type *DataType, Align Alignment) const override {
+  bool shouldVectorizeNTLoad(Type *DataType, Align Alignment) const override {
     // Currently we only support NT loads lowering for little-endian targets.
     //
     // Coordinated with LDNP constraints in
@@ -431,10 +431,10 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
     // `AArch64TargetLowering::ReplaceNodeResults`
     if (!ST->isLittleEndian())
       return false;
-    if (auto Result = isLegalNTStoreLoad(DataType, Alignment))
+    if (auto Result = shouldVectorizeNTStoreLoad(DataType, Alignment))
       return *Result;
     // Fallback to target independent logic
-    return BaseT::isLegalNTLoad(DataType, Alignment);
+    return BaseT::shouldVectorizeNTLoad(DataType, Alignment);
   }
 
   InstructionCost getPartialReductionCost(
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index e54f4ed2fb26c..dc5cd618f4e19 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -36037,6 +36037,43 @@ bool X86TargetLowering::isLegalStoreImmediate(int64_t Imm) const {
   return isInt<32>(Imm);
 }
 
+bool X86TTIImpl::isLegalNTLoad(Type *DataType, Align Alignment,
+                               const DataLayout &DL) const {
+  unsigned DataSize = DL.getTypeStoreSize(DataType);
+  // The only supported nontemporal loads are for aligned vectors of 16 or 32
+  // bytes.  Note that 32-byte nontemporal vector loads are supported by AVX2
+  // (the equivalent stores only require AVX).
+  if (Alignment >= DataSize && (DataSize == 16 || DataSize == 32))
+    return DataSize == 16 ? ST->hasSSE1() : ST->hasAVX2();
+
+  return false;
+}
+
+bool X86TTIImpl::isLegalNTStore(Type *DataType, Align Alignment,
+                                const DataLayout &DL) const {
+  unsigned DataSize = DL.getTypeStoreSize(DataType);
+
+  // SSE4A supports nontemporal stores of float and double at arbitrary
+  // alignment.
+  if (ST->hasSSE4A() && (DataType->isFloatTy() || DataType->isDoubleTy()))
+    return true;
+
+  // Besides the SSE4A subtarget exception above, only aligned stores are
+  // available nontemporaly on any other subtarget.  And only stores with a size
+  // of 4..32 bytes (powers of 2, only) are permitted.
+  if (Alignment < DataSize || DataSize < 4 || DataSize > 32 ||
+      !isPowerOf2_32(DataSize))
+    return false;
+
+  // 32-byte vector nontemporal stores are supported by AVX (the equivalent
+  // loads require AVX2).
+  if (DataSize == 32)
+    return ST->hasAVX();
+  if (DataSize == 16)
+    return ST->hasSSE1();
+  return true;
+}
+
 bool X86TargetLowering::isTruncateFree(EVT VT1, EVT VT2) const {
   if (!VT1.isScalarInteger() || !VT2.isScalarInteger())
     return false;
diff --git a/llvm/lib/Target/X86/X86ISelLowering.h b/llvm/lib/Target/X86/X86ISelLowering.h
index 30faca54e13f6..17c424add229a 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.h
+++ b/llvm/lib/Target/X86/X86ISelLowering.h
@@ -1456,6 +1456,11 @@ namespace llvm {
 
     bool isLegalStoreImmediate(int64_t Imm) const override;
 
+    bool isLegalNTLoad(Type *DataType, Align Alignment,
+                       const DataLayout &DL) const override;
+    bool isLegalNTStore(Type *DataType, Align Alignment,
+                        const DataLayout &DL) const override;
+
     /// Add x86-specific opcodes to the default list.
     bool isBinOp(unsigned Opcode) const override;
 
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index 608727b745925..40ca2665c0d94 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -6367,41 +6367,6 @@ bool X86TTIImpl::isLegalMaskedStore(Type *DataTy, Align Alignment,
   return isLegalMaskedLoadStore(ScalarTy, ST);
 }
 
-bool X86TTIImpl::isLegalNTLoad(Type *DataType, Align Alignment) const {
-  unsigned DataSize = DL.getTypeStoreSize(DataType);
-  // The only supported nontemporal loads are for aligned vectors of 16 or 32
-  // bytes.  Note that 32-byte nontemporal vector loads are supported by AVX2
-  // (the equivalent stores only require AVX).
-  if (Alignment >= DataSize && (DataSize == 16 || DataSize == 32))
-    return DataSize == 16 ?  ST->hasSSE1() : ST->hasAVX2();
-
-  return false;
-}
-
-bool X86TTIImpl::isLegalNTStore(Type *DataType, Align Alignment) const {
-  unsigned DataSize = DL.getTypeStoreSize(DataType);
-
-  // SSE4A supports nontemporal stores of float and double at arbitrary
-  // alignment.
-  if (ST->hasSSE4A() && (DataType->isFloatTy() || DataType->isDoubleTy()))
-    return true;
-
-  // Besides the SSE4A subtarget exception above, only aligned stores are
-  // available nontemporaly on any other subtarget.  And only stores with a size
-  // of 4..32 bytes (powers of 2, only) are permitted.
-  if (Alignment < DataSize || DataSize < 4 || DataSize > 32 ||
-      !isPowerOf2_32(DataSize))
-    return false;
-
-  // 32-byte vector nontemporal stores are supported by AVX (the equivalent
-  // loads require AVX2).
-  if (DataSize == 32)
-    return ST->hasAVX();
-  if (DataSize == 16)
-    return ST->hasSSE1();
-  return true;
-}
-
 bool X86TTIImpl::isLegalBroadcastLoad(Type *ElementTy,
                                       ElementCount NumElements) const {
   // movddup
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.h b/llvm/lib/Target/X86/X86TargetTransformInfo.h
index 4f672793a0fcf..451d03fce44a5 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.h
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.h
@@ -274,8 +274,6 @@ class X86TTIImpl final : public BasicTTIImplBase<X86TTIImpl> {
   isLegalMaskedStore(Type *DataType, Align Alignment, unsigned AddressSpace,
                      TTI::MaskKind MaskKind =
                          TTI::MaskKind::VariableOrConstantMask) const override;
-  bool isLegalNTLoad(Type *DataType, Align Alignment) const override;
-  bool isLegalNTStore(Type *DataType, Align Alignment) const override;
   bool isLegalBroadcastLoad(Type *ElementTy,
                             ElementCount NumElements) const override;
   bool forceScalarizeMaskedGather(VectorType *VTy,
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index ea431b4fa5df8..a0dd100f0c903 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -1045,10 +1045,11 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
     // For nontemporal stores, check that a nontemporal vector version is
     // supported on the target.
     if (ST->getMetadata(LLVMContext::MD_nontemporal)) {
-      // Arbitrarily try a vector of 2 elements.
+      // Check a 2-element vector type, which implictly covers any power-of-2
+      // sized vector type by logically splitting to pairs
       auto *VecTy = FixedVectorType::get(T, /*NumElts=*/2);
       assert(VecTy && "did not find vectorized version of stored type");
-      if (!TTI->isLegalNTStore(VecTy, ST->getAlign())) {
+      if (!TTI->shouldVectorizeNTStore(VecTy, ST->getAlign())) {
         reportVectorizationFailure(
             "nontemporal store instruction cannot be vectorized",
             "CantVectorizeNontemporalStore", ORE, TheLoop, ST);
@@ -1057,12 +1058,14 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
     }
 
   } else if (auto *LD = dyn_cast<LoadInst>(&I)) {
+    // For nontemporal loads, check that a nontemporal vector version is
+    // supported on the target.
     if (LD->getMetadata(LLVMContext::MD_nontemporal)) {
-      // For nontemporal loads, check that a nontemporal vector version is
-      // supported on the target (arbitrarily try a vector of 2 elements).
+      // Check a 2-element vector type, which implictly covers any power-of-2
+      // sized vector type by logically splitting to pairs
       auto *VecTy = FixedVectorType::get(I.getType(), /*NumElts=*/2);
       assert(VecTy && "did not find vectorized version of load type");
-      if (!TTI->isLegalNTLoad(VecTy, LD->getAlign())) {
+      if (!TTI->shouldVectorizeNTLoad(VecTy, LD->getAlign())) {
         reportVectorizationFailure(
             "nontemporal load instruction cannot be vectorized",
             "CantVectorizeNontemporalLoad", ORE, TheLoop, LD);



More information about the llvm-commits mailing list