[llvm] [TTI] Add distinguished shouldVectorizeNTStore/Load (PR #177936)
Tomer Shafir via llvm-commits
llvm-commits at lists.llvm.org
Mon Jan 26 04:14:03 PST 2026
https://github.com/tomershafir updated https://github.com/llvm/llvm-project/pull/177936
>From 70e0f0d74a048196926ba36b2c2f6f0b7d52fce1 Mon Sep 17 00:00:00 2001
From: tomershafir <tomer.shafir8 at gmail.com>
Date: Thu, 22 Jan 2026 23:08:54 +0200
Subject: [PATCH 1/5] [AArch64] Align nontemporal store/load little-endian
checks
This patch aims to align all nontemporal store/load handling to systematically enforce a little-endian target. This has been the effective support LLVM had for NT store/load lowering (there has been no effective support for big-endian, even with the inconsistencies).
The change in `llvm/lib/Target/AArch64/AArch64InstrInfo.td` is effectively a NFC, because the only lowering of LDNP, in `llvm/lib/Target/AArch64/AArch64ISelLowering.cpp`, have already checked for `isLittleEndian`.
The change in `llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h` affects its single caller `llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp`. The previous logic has been wrong, enabling vectorization of effectively illegal nontemporal store/load instructions on big-endian.
---
.../Target/AArch64/AArch64ISelLowering.cpp | 10 +
llvm/lib/Target/AArch64/AArch64InstrInfo.td | 16 +-
.../AArch64/AArch64TargetTransformInfo.h | 30 +-
llvm/test/CodeGen/AArch64/nontemporal-load.ll | 378 +++++++++---------
.../AArch64/nontemporal-load-store.ll | 245 ++++++++----
5 files changed, 401 insertions(+), 278 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 05d5f6b323706..eef52867107e9 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -7322,6 +7322,10 @@ static SDValue LowerADDRSPACECAST(SDValue Op, SelectionDAG &DAG) {
}
// Lower non-temporal stores that would otherwise be broken by legalization.
+//
+// Coordinated with LDNP constraints in
+// `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
+// `AArch64TargetLowering::ReplaceNodeResults`
static SDValue LowerNTStore(StoreSDNode *StoreNode, EVT VT, EVT MemVT,
const SDLoc &DL, SelectionDAG &DAG) {
assert(StoreNode && "Expected a store operation");
@@ -29564,6 +29568,12 @@ void AArch64TargetLowering::ReplaceNodeResults(
EVT MemVT = LoadNode->getMemoryVT();
// Handle lowering 256 bit non temporal loads into LDNP for little-endian
// targets.
+ //
+ // Currently we only support NT loads lowering for little-endian targets.
+ //
+ // Coordinated with LDNP constraints in
+ // `llvm/lib/Target/AArch64/AArch64InstrInfo.td`
+ // and `AArch64TTIImpl::isLegalNTLoad`.
if (LoadNode->isNonTemporal() && Subtarget->isLittleEndian() &&
MemVT.getSizeInBits() == 256u &&
(MemVT.getScalarSizeInBits() == 8u ||
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.td b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
index 59a3b2d36e0f0..ee9e3fa886a5c 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
@@ -3786,8 +3786,15 @@ defm LDNPQ : LoadPairNoAlloc<0b10, 1, FPR128Op, simm7s16, "ldnp">;
def : Pat<(AArch64ldp (am_indexed7s64 GPR64sp:$Rn, simm7s8:$offset)),
(LDPXi GPR64sp:$Rn, simm7s8:$offset)>;
-def : Pat<(AArch64ldnp (am_indexed7s128 GPR64sp:$Rn, simm7s16:$offset)),
- (LDNPQi GPR64sp:$Rn, simm7s16:$offset)>;
+// Currently we only support NT loads lowering for little-endian targets.
+//
+// Coordinated with LDNP constraints in `AArch64TargetLowering::ReplaceNodeResults`
+// and `AArch64TTIImpl::isLegalNTLoad`.
+let Predicates = [IsLE] in {
+ def : Pat<(AArch64ldnp(am_indexed7s128 GPR64sp:$Rn, simm7s16:$offset)),
+ (LDNPQi GPR64sp:$Rn, simm7s16:$offset)>;
+}
+
//---
// (register offset)
//---
@@ -10724,6 +10731,11 @@ def : Pat<(i64 (int_aarch64_neon_urshl (i64 FPR64:$Rn), (i64 FPR64:$Rm))),
// We have to resort to tricks to turn a single-input store into a store pair,
// because there is no single-input nontemporal store, only STNP.
//
+// Currently we only support NT stores lowering for little-endian targets.
+//
+// Coordinated with STNP constraints in `AArch64TargetLowering::LowerNTStore`
+// and `AArch64TTIImpl::isLegalNTStore`.
+//
// Currently, STNP lowering can only either keep or increase code size, thus
// we predicate it to not apply when optimizing for code size.
let Predicates = [IsLE, NotForCodeSize] in {
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index c9bf44b15144a..b789e13bf25da 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -391,7 +391,8 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
return false;
}
- bool isLegalNTStoreLoad(Type *DataType, Align Alignment) const {
+ std::optional<bool> isLegalNTStoreLoad(Type *DataType,
+ Align Alignment) const {
// NOTE: The logic below is mostly geared towards LV, which calls it with
// vectors with 2 elements. We might want to improve that, if other
// users show up.
@@ -405,17 +406,34 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
return NumElements > 1 && isPowerOf2_64(NumElements) && EltSize >= 8 &&
EltSize <= 128 && isPowerOf2_64(EltSize);
}
- return BaseT::isLegalNTStore(DataType, Alignment);
+ return std::nullopt;
}
bool isLegalNTStore(Type *DataType, Align Alignment) const override {
- return isLegalNTStoreLoad(DataType, Alignment);
+ // Currently we only support NT stores lowering for little-endian targets.
+ //
+ // Coordinated with STNP constraints in
+ // `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
+ // `AArch64TargetLowering::LowerNTStore`
+ if (!ST->isLittleEndian())
+ return false;
+ if (auto Result = isLegalNTStoreLoad(DataType, Alignment))
+ return *Result;
+ // Fallback to target independent logic
+ return BaseT::isLegalNTStore(DataType, Alignment);
}
bool isLegalNTLoad(Type *DataType, Align Alignment) const override {
- // Only supports little-endian targets.
- if (ST->isLittleEndian())
- return isLegalNTStoreLoad(DataType, Alignment);
+ // Currently we only support NT loads lowering for little-endian targets.
+ //
+ // Coordinated with LDNP constraints in
+ // `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
+ // `AArch64TargetLowering::ReplaceNodeResults`
+ if (!ST->isLittleEndian())
+ return false;
+ if (auto Result = isLegalNTStoreLoad(DataType, Alignment))
+ return *Result;
+ // Fallback to target independent logic
return BaseT::isLegalNTLoad(DataType, Alignment);
}
diff --git a/llvm/test/CodeGen/AArch64/nontemporal-load.ll b/llvm/test/CodeGen/AArch64/nontemporal-load.ll
index ad92530eabf08..62b3e5651423f 100644
--- a/llvm/test/CodeGen/AArch64/nontemporal-load.ll
+++ b/llvm/test/CodeGen/AArch64/nontemporal-load.ll
@@ -1,12 +1,12 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc --mattr=+sve -aarch64-enable-sink-fold=true < %s -mtriple aarch64-apple-darwin | FileCheck %s
+; RUN: llc --mattr=+sve -aarch64-enable-sink-fold=true < %s -mtriple aarch64-apple-darwin | FileCheck --check-prefix CHECK-LE %s
; RUN: llc --mattr=+sve -aarch64-enable-sink-fold=true < %s -mtriple aarch64_be-unknown-unknown | FileCheck --check-prefix CHECK-BE %s
define <4 x double> @test_ldnp_v4f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4f64:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v4f64:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v4f64:
; CHECK-BE: // %bb.0:
@@ -17,10 +17,10 @@ define <4 x double> @test_ldnp_v4f64(ptr %A) {
}
define <4 x i64> @test_ldnp_v4i64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4i64:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v4i64:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v4i64:
; CHECK-BE: // %bb.0:
@@ -31,10 +31,10 @@ define <4 x i64> @test_ldnp_v4i64(ptr %A) {
}
define <8 x i32> @test_ldnp_v8i32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v8i32:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v8i32:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v8i32:
; CHECK-BE: // %bb.0:
@@ -45,10 +45,10 @@ define <8 x i32> @test_ldnp_v8i32(ptr %A) {
}
define <8 x float> @test_ldnp_v8f32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v8f32:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v8f32:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v8f32:
; CHECK-BE: // %bb.0:
@@ -59,10 +59,10 @@ define <8 x float> @test_ldnp_v8f32(ptr %A) {
}
define <16 x i16> @test_ldnp_v16i16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16i16:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v16i16:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v16i16:
; CHECK-BE: // %bb.0:
@@ -73,10 +73,10 @@ define <16 x i16> @test_ldnp_v16i16(ptr %A) {
}
define <16 x half> @test_ldnp_v16f16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16f16:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v16f16:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v16f16:
; CHECK-BE: // %bb.0:
@@ -87,10 +87,10 @@ define <16 x half> @test_ldnp_v16f16(ptr %A) {
}
define <32 x i8> @test_ldnp_v32i8(ptr %A) {
-; CHECK-LABEL: test_ldnp_v32i8:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v32i8:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v32i8:
; CHECK-BE: // %bb.0:
@@ -101,10 +101,10 @@ define <32 x i8> @test_ldnp_v32i8(ptr %A) {
}
define <4 x i32> @test_ldnp_v4i32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4i32:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr q0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v4i32:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr q0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v4i32:
; CHECK-BE: // %bb.0:
@@ -115,10 +115,10 @@ define <4 x i32> @test_ldnp_v4i32(ptr %A) {
}
define <4 x float> @test_ldnp_v4f32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4f32:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr q0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v4f32:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr q0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v4f32:
; CHECK-BE: // %bb.0:
@@ -129,10 +129,10 @@ define <4 x float> @test_ldnp_v4f32(ptr %A) {
}
define <8 x i16> @test_ldnp_v8i16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v8i16:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr q0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v8i16:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr q0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v8i16:
; CHECK-BE: // %bb.0:
@@ -143,10 +143,10 @@ define <8 x i16> @test_ldnp_v8i16(ptr %A) {
}
define <16 x i8> @test_ldnp_v16i8(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16i8:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr q0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v16i8:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr q0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v16i8:
; CHECK-BE: // %bb.0:
@@ -156,10 +156,10 @@ define <16 x i8> @test_ldnp_v16i8(ptr %A) {
ret <16 x i8> %lv
}
define <2 x double> @test_ldnp_v2f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v2f64:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr q0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v2f64:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr q0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v2f64:
; CHECK-BE: // %bb.0:
@@ -170,10 +170,10 @@ define <2 x double> @test_ldnp_v2f64(ptr %A) {
}
define <2 x i32> @test_ldnp_v2i32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v2i32:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr d0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v2i32:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr d0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v2i32:
; CHECK-BE: // %bb.0:
@@ -184,10 +184,10 @@ define <2 x i32> @test_ldnp_v2i32(ptr %A) {
}
define <2 x float> @test_ldnp_v2f32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v2f32:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr d0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v2f32:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr d0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v2f32:
; CHECK-BE: // %bb.0:
@@ -198,10 +198,10 @@ define <2 x float> @test_ldnp_v2f32(ptr %A) {
}
define <4 x i16> @test_ldnp_v4i16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4i16:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr d0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v4i16:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr d0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v4i16:
; CHECK-BE: // %bb.0:
@@ -212,10 +212,10 @@ define <4 x i16> @test_ldnp_v4i16(ptr %A) {
}
define <8 x i8> @test_ldnp_v8i8(ptr %A) {
-; CHECK-LABEL: test_ldnp_v8i8:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr d0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v8i8:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr d0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v8i8:
; CHECK-BE: // %bb.0:
@@ -226,10 +226,10 @@ define <8 x i8> @test_ldnp_v8i8(ptr %A) {
}
define <1 x double> @test_ldnp_v1f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v1f64:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr d0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v1f64:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr d0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v1f64:
; CHECK-BE: // %bb.0:
@@ -240,10 +240,10 @@ define <1 x double> @test_ldnp_v1f64(ptr %A) {
}
define <1 x i64> @test_ldnp_v1i64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v1i64:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldr d0, [x0]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v1i64:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldr d0, [x0]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v1i64:
; CHECK-BE: // %bb.0:
@@ -254,11 +254,11 @@ define <1 x i64> @test_ldnp_v1i64(ptr %A) {
}
define <32 x i16> @test_ldnp_v32i16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v32i16:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ldnp q2, q3, [x0, #32]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v32i16:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v32i16:
; CHECK-BE: // %bb.0:
@@ -270,11 +270,11 @@ define <32 x i16> @test_ldnp_v32i16(ptr %A) {
}
define <32 x half> @test_ldnp_v32f16(ptr %A) {
-; CHECK-LABEL: test_ldnp_v32f16:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ldnp q2, q3, [x0, #32]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v32f16:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v32f16:
; CHECK-BE: // %bb.0:
@@ -286,11 +286,11 @@ define <32 x half> @test_ldnp_v32f16(ptr %A) {
}
define <16 x i32> @test_ldnp_v16i32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16i32:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ldnp q2, q3, [x0, #32]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v16i32:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v16i32:
; CHECK-BE: // %bb.0:
@@ -302,11 +302,11 @@ define <16 x i32> @test_ldnp_v16i32(ptr %A) {
}
define <16 x float> @test_ldnp_v16f32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16f32:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ldnp q2, q3, [x0, #32]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v16f32:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v16f32:
; CHECK-BE: // %bb.0:
@@ -318,15 +318,15 @@ define <16 x float> @test_ldnp_v16f32(ptr %A) {
}
define <17 x float> @test_ldnp_v17f32(ptr %A) {
-; CHECK-LABEL: test_ldnp_v17f32:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0, #32]
-; CHECK-NEXT: ldr s2, [x0, #64]
-; CHECK-NEXT: ldnp q3, q4, [x0]
-; CHECK-NEXT: stp q0, q1, [x8, #32]
-; CHECK-NEXT: stp q3, q4, [x8]
-; CHECK-NEXT: str s2, [x8, #64]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v17f32:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0, #32]
+; CHECK-LE-NEXT: ldr s2, [x0, #64]
+; CHECK-LE-NEXT: ldnp q3, q4, [x0]
+; CHECK-LE-NEXT: stp q0, q1, [x8, #32]
+; CHECK-LE-NEXT: stp q3, q4, [x8]
+; CHECK-LE-NEXT: str s2, [x8, #64]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v17f32:
; CHECK-BE: // %bb.0:
@@ -352,27 +352,27 @@ define <17 x float> @test_ldnp_v17f32(ptr %A) {
}
define <33 x double> @test_ldnp_v33f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v33f64:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ldr d20, [x0, #256]
-; CHECK-NEXT: ldnp q2, q3, [x0, #32]
-; CHECK-NEXT: ldnp q4, q5, [x0, #64]
-; CHECK-NEXT: ldnp q6, q7, [x0, #96]
-; CHECK-NEXT: ldnp q16, q17, [x0, #128]
-; CHECK-NEXT: ldnp q18, q19, [x0, #224]
-; CHECK-NEXT: ldnp q21, q22, [x0, #160]
-; CHECK-NEXT: ldnp q23, q24, [x0, #192]
-; CHECK-NEXT: stp q0, q1, [x8]
-; CHECK-NEXT: stp q2, q3, [x8, #32]
-; CHECK-NEXT: stp q4, q5, [x8, #64]
-; CHECK-NEXT: stp q6, q7, [x8, #96]
-; CHECK-NEXT: stp q16, q17, [x8, #128]
-; CHECK-NEXT: stp q21, q22, [x8, #160]
-; CHECK-NEXT: stp q23, q24, [x8, #192]
-; CHECK-NEXT: stp q18, q19, [x8, #224]
-; CHECK-NEXT: str d20, [x8, #256]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v33f64:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ldr d20, [x0, #256]
+; CHECK-LE-NEXT: ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT: ldnp q4, q5, [x0, #64]
+; CHECK-LE-NEXT: ldnp q6, q7, [x0, #96]
+; CHECK-LE-NEXT: ldnp q16, q17, [x0, #128]
+; CHECK-LE-NEXT: ldnp q18, q19, [x0, #224]
+; CHECK-LE-NEXT: ldnp q21, q22, [x0, #160]
+; CHECK-LE-NEXT: ldnp q23, q24, [x0, #192]
+; CHECK-LE-NEXT: stp q0, q1, [x8]
+; CHECK-LE-NEXT: stp q2, q3, [x8, #32]
+; CHECK-LE-NEXT: stp q4, q5, [x8, #64]
+; CHECK-LE-NEXT: stp q6, q7, [x8, #96]
+; CHECK-LE-NEXT: stp q16, q17, [x8, #128]
+; CHECK-LE-NEXT: stp q21, q22, [x8, #160]
+; CHECK-LE-NEXT: stp q23, q24, [x8, #192]
+; CHECK-LE-NEXT: stp q18, q19, [x8, #224]
+; CHECK-LE-NEXT: str d20, [x8, #256]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v33f64:
; CHECK-BE: // %bb.0:
@@ -446,13 +446,13 @@ define <33 x double> @test_ldnp_v33f64(ptr %A) {
}
define <33 x i8> @test_ldnp_v33i8(ptr %A) {
-; CHECK-LABEL: test_ldnp_v33i8:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ldr b2, [x0, #32]
-; CHECK-NEXT: stp q0, q1, [x8]
-; CHECK-NEXT: stur b2, [x8, #32]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v33i8:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ldr b2, [x0, #32]
+; CHECK-LE-NEXT: stp q0, q1, [x8]
+; CHECK-LE-NEXT: stur b2, [x8, #32]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v33i8:
; CHECK-BE: // %bb.0:
@@ -470,20 +470,20 @@ define <33 x i8> @test_ldnp_v33i8(ptr %A) {
}
define <4 x i65> @test_ldnp_v4i65(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4i65:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldp x8, x9, [x0, #8]
-; CHECK-NEXT: ldr x10, [x0, #24]
-; CHECK-NEXT: ldrb w11, [x0, #32]
-; CHECK-NEXT: ldr x0, [x0]
-; CHECK-NEXT: ubfx x5, x10, #2, #1
-; CHECK-NEXT: extr x2, x9, x8, #1
-; CHECK-NEXT: extr x4, x10, x9, #2
-; CHECK-NEXT: extr x6, x11, x10, #3
-; CHECK-NEXT: ubfx x3, x9, #1, #1
-; CHECK-NEXT: ubfx x7, x11, #3, #1
-; CHECK-NEXT: and x1, x8, #0x1
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v4i65:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldp x8, x9, [x0, #8]
+; CHECK-LE-NEXT: ldr x10, [x0, #24]
+; CHECK-LE-NEXT: ldrb w11, [x0, #32]
+; CHECK-LE-NEXT: ldr x0, [x0]
+; CHECK-LE-NEXT: ubfx x5, x10, #2, #1
+; CHECK-LE-NEXT: extr x2, x9, x8, #1
+; CHECK-LE-NEXT: extr x4, x10, x9, #2
+; CHECK-LE-NEXT: extr x6, x11, x10, #3
+; CHECK-LE-NEXT: ubfx x3, x9, #1, #1
+; CHECK-LE-NEXT: ubfx x7, x11, #3, #1
+; CHECK-LE-NEXT: and x1, x8, #0x1
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v4i65:
; CHECK-BE: // %bb.0:
@@ -510,17 +510,17 @@ define <4 x i65> @test_ldnp_v4i65(ptr %A) {
}
define <4 x i63> @test_ldnp_v4i63(ptr %A) {
-; CHECK-LABEL: test_ldnp_v4i63:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldp x8, x9, [x0, #16]
-; CHECK-NEXT: ldp x10, x11, [x0]
-; CHECK-NEXT: extr x3, x9, x8, #61
-; CHECK-NEXT: extr x9, x11, x10, #63
-; CHECK-NEXT: extr x8, x8, x11, #62
-; CHECK-NEXT: and x0, x10, #0x7fffffffffffffff
-; CHECK-NEXT: and x1, x9, #0x7fffffffffffffff
-; CHECK-NEXT: and x2, x8, #0x7fffffffffffffff
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v4i63:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldp x8, x9, [x0, #16]
+; CHECK-LE-NEXT: ldp x10, x11, [x0]
+; CHECK-LE-NEXT: extr x3, x9, x8, #61
+; CHECK-LE-NEXT: extr x9, x11, x10, #63
+; CHECK-LE-NEXT: extr x8, x8, x11, #62
+; CHECK-LE-NEXT: and x0, x10, #0x7fffffffffffffff
+; CHECK-LE-NEXT: and x1, x9, #0x7fffffffffffffff
+; CHECK-LE-NEXT: and x2, x8, #0x7fffffffffffffff
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v4i63:
; CHECK-BE: // %bb.0:
@@ -539,17 +539,17 @@ define <4 x i63> @test_ldnp_v4i63(ptr %A) {
}
define <5 x double> @test_ldnp_v5f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v5f64:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q2, [x0]
-; CHECK-NEXT: ldr d4, [x0, #32]
-; CHECK-NEXT: ext.16b v1, v0, v0, #8
-; CHECK-NEXT: ext.16b v3, v2, v2, #8
-; CHECK-NEXT: ; kill: def $d0 killed $d0 killed $q0
-; CHECK-NEXT: ; kill: def $d2 killed $d2 killed $q2
-; CHECK-NEXT: ; kill: def $d1 killed $d1 killed $q1
-; CHECK-NEXT: ; kill: def $d3 killed $d3 killed $q3
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v5f64:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q2, [x0]
+; CHECK-LE-NEXT: ldr d4, [x0, #32]
+; CHECK-LE-NEXT: ext.16b v1, v0, v0, #8
+; CHECK-LE-NEXT: ext.16b v3, v2, v2, #8
+; CHECK-LE-NEXT: ; kill: def $d0 killed $d0 killed $q0
+; CHECK-LE-NEXT: ; kill: def $d2 killed $d2 killed $q2
+; CHECK-LE-NEXT: ; kill: def $d1 killed $d1 killed $q1
+; CHECK-LE-NEXT: ; kill: def $d3 killed $d3 killed $q3
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v5f64:
; CHECK-BE: // %bb.0:
@@ -570,13 +570,13 @@ define <5 x double> @test_ldnp_v5f64(ptr %A) {
}
define <16 x i64> @test_ldnp_v16i64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16i64:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ldnp q2, q3, [x0, #32]
-; CHECK-NEXT: ldnp q4, q5, [x0, #64]
-; CHECK-NEXT: ldnp q6, q7, [x0, #96]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v16i64:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT: ldnp q4, q5, [x0, #64]
+; CHECK-LE-NEXT: ldnp q6, q7, [x0, #96]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v16i64:
; CHECK-BE: // %bb.0:
@@ -590,13 +590,13 @@ define <16 x i64> @test_ldnp_v16i64(ptr %A) {
}
define <16 x double> @test_ldnp_v16f64(ptr %A) {
-; CHECK-LABEL: test_ldnp_v16f64:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ldnp q0, q1, [x0]
-; CHECK-NEXT: ldnp q2, q3, [x0, #32]
-; CHECK-NEXT: ldnp q4, q5, [x0, #64]
-; CHECK-NEXT: ldnp q6, q7, [x0, #96]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v16f64:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ldnp q0, q1, [x0]
+; CHECK-LE-NEXT: ldnp q2, q3, [x0, #32]
+; CHECK-LE-NEXT: ldnp q4, q5, [x0, #64]
+; CHECK-LE-NEXT: ldnp q6, q7, [x0, #96]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v16f64:
; CHECK-BE: // %bb.0:
@@ -610,15 +610,15 @@ define <16 x double> @test_ldnp_v16f64(ptr %A) {
}
define <vscale x 20 x float> @test_ldnp_v20f32_vscale(ptr %A) {
-; CHECK-LABEL: test_ldnp_v20f32_vscale:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: ptrue p0.s
-; CHECK-NEXT: ldnt1w { z0.s }, p0/z, [x0]
-; CHECK-NEXT: ldnt1w { z1.s }, p0/z, [x0, #1, mul vl]
-; CHECK-NEXT: ldnt1w { z2.s }, p0/z, [x0, #2, mul vl]
-; CHECK-NEXT: ldnt1w { z3.s }, p0/z, [x0, #3, mul vl]
-; CHECK-NEXT: ldnt1w { z4.s }, p0/z, [x0, #4, mul vl]
-; CHECK-NEXT: ret
+; CHECK-LE-LABEL: test_ldnp_v20f32_vscale:
+; CHECK-LE: ; %bb.0:
+; CHECK-LE-NEXT: ptrue p0.s
+; CHECK-LE-NEXT: ldnt1w { z0.s }, p0/z, [x0]
+; CHECK-LE-NEXT: ldnt1w { z1.s }, p0/z, [x0, #1, mul vl]
+; CHECK-LE-NEXT: ldnt1w { z2.s }, p0/z, [x0, #2, mul vl]
+; CHECK-LE-NEXT: ldnt1w { z3.s }, p0/z, [x0, #3, mul vl]
+; CHECK-LE-NEXT: ldnt1w { z4.s }, p0/z, [x0, #4, mul vl]
+; CHECK-LE-NEXT: ret
;
; CHECK-BE-LABEL: test_ldnp_v20f32_vscale:
; CHECK-BE: // %bb.0:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll b/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
index c7edf9bdfaf6b..83a5d1b60a400 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
@@ -1,11 +1,17 @@
-; RUN: opt -passes=loop-vectorize -mtriple=arm64-apple-iphones -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64 -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK-LE
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64_be -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK-BE
; Vectors with i4 elements may not legal with nontemporal stores.
define void @test_i4_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_i4_store(
-; CHECK-NOT: vector.body:
-; CHECK: ret void
+; CHECK-LE-LABEL: define void @test_i4_store(
+; CHECK-LE-NOT: vector.body:
+; CHECK-LE: store i4 {{.*}} !nontemporal !0
+; CHECK-LE: ret void
;
+; CHECK-BE-LABEL: define void @test_i4_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: store i4 {{.*}} !nontemporal !0
+; CHECK-BE: ret void
entry:
br label %for.body
@@ -23,11 +29,15 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_i8_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_i8_store(
-; CHECK-LABEL: vector.body:
-; CHECK: store <4 x i8> {{.*}} !nontemporal !0
-; CHECK: br
+; CHECK-LE-LABEL: define void @test_i8_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: store <4 x i8> {{.*}} !nontemporal !0
+; CHECK-LE: br
;
+; CHECK-BE-LABEL: define void @test_i8_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: store i8 {{.*}} !nontemporal !0
+; CHECK-BE: br
entry:
br label %for.body
@@ -45,11 +55,15 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_half_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_half_store(
-; CHECK-LABEL: vector.body:
-; CHECK: store <4 x half> {{.*}} !nontemporal !0
-; CHECK: br
+; CHECK-LE-LABEL: define void @test_half_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: store <4 x half> {{.*}} !nontemporal !0
+; CHECK-LE: br
;
+; CHECK-BE-LABEL: define void @test_half_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: store half {{.*}} !nontemporal !0
+; CHECK-BE: br
entry:
br label %for.body
@@ -67,11 +81,15 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_i16_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_i16_store(
-; CHECK-LABEL: vector.body:
-; CHECK: store <4 x i16> {{.*}} !nontemporal !0
-; CHECK: br
+; CHECK-LE-LABEL: define void @test_i16_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: store <4 x i16> {{.*}} !nontemporal !0
+; CHECK-LE: br
;
+; CHECK-BE-LABEL: define void @test_i16_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: store i16 {{.*}} !nontemporal !0
+; CHECK-BE: br
entry:
br label %for.body
@@ -89,11 +107,15 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_i32_store(ptr nocapture %ddst) {
-; CHECK-LABEL: define void @test_i32_store(
-; CHECK-LABEL: vector.body:
-; CHECK: store <16 x i32> {{.*}} !nontemporal !0
-; CHECK: br
+; CHECK-LE-LABEL: define void @test_i32_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: store <16 x i32> {{.*}} !nontemporal !0
+; CHECK-LE: br
;
+; CHECK-BE-LABEL: define void @test_i32_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: store i32 {{.*}} !nontemporal !0
+; CHECK-BE: br
entry:
br label %for.body
@@ -117,10 +139,13 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_i33_store(ptr nocapture %ddst) {
-; CHECK-LABEL: define void @test_i33_store(
-; CHECK-NOT: vector.body:
-; CHECK: ret
+; CHECK-LE-LABEL: define void @test_i33_store(
+; CHECK-LE-NOT: vector.body:
+; CHECK-LE: ret
;
+; CHECK-BE-LABEL: define void @test_i33_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: ret
entry:
br label %for.body
@@ -144,10 +169,13 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_i40_store(ptr nocapture %ddst) {
-; CHECK-LABEL: define void @test_i40_store(
-; CHECK-NOT: vector.body:
-; CHECK: ret
+; CHECK-LE-LABEL: define void @test_i40_store(
+; CHECK-LE-NOT: vector.body:
+; CHECK-LE: ret
;
+; CHECK-BE-LABEL: define void @test_i40_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: ret
entry:
br label %for.body
@@ -170,11 +198,15 @@ for.cond.cleanup: ; preds = %for.body
ret void
}
define void @test_i64_store(ptr nocapture %ddst) local_unnamed_addr #0 {
-; CHECK-LABEL: define void @test_i64_store(
-; CHECK-LABEL: vector.body:
-; CHECK: store <4 x i64> {{.*}} !nontemporal !0
-; CHECK: br
+; CHECK-LE-LABEL: define void @test_i64_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: store <4 x i64> {{.*}} !nontemporal !0
+; CHECK-LE: br
;
+; CHECK-BE-LABEL: define void @test_i64_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: store i64 {{.*}} !nontemporal !0
+; CHECK-BE: br
entry:
br label %for.body
@@ -192,11 +224,15 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_double_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_double_store(
-; CHECK-LABEL: vector.body:
-; CHECK: store <4 x double> {{.*}} !nontemporal !0
-; CHECK: br
+; CHECK-LE-LABEL: define void @test_double_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: store <4 x double> {{.*}} !nontemporal !0
+; CHECK-LE: br
;
+; CHECK-BE-LABEL: define void @test_double_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: store double {{.*}} !nontemporal !0
+; CHECK-BE: br
entry:
br label %for.body
@@ -214,11 +250,15 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_i128_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_i128_store(
-; CHECK-LABEL: vector.body:
-; CHECK: store <4 x i128> {{.*}} !nontemporal !0
-; CHECK: br
+; CHECK-LE-LABEL: define void @test_i128_store(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: store <4 x i128> {{.*}} !nontemporal !0
+; CHECK-LE: br
;
+; CHECK-BE-LABEL: define void @test_i128_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: store i128 {{.*}} !nontemporal !0
+; CHECK-BE: br
entry:
br label %for.body
@@ -236,10 +276,13 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_i256_store(ptr %ddst) {
-; CHECK-LABEL: define void @test_i256_store(
-; CHECK-NOT: vector.body:
-; CHECK: ret void
+; CHECK-LE-LABEL: define void @test_i256_store(
+; CHECK-LE-NOT: vector.body:
+; CHECK-LE: ret void
;
+; CHECK-BE-LABEL: define void @test_i256_store(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: ret void
entry:
br label %for.body
@@ -257,10 +300,13 @@ for.cond.cleanup: ; preds = %for.body
}
define i4 @test_i4_load(ptr %ddst) {
-; CHECK-LABEL: define i4 @test_i4_load
-; CHECK-NOT: vector.body:
-; CHECK: ret i4 %{{.*}}
+; CHECK-LE-LABEL: define i4 @test_i4_load
+; CHECK-LE-NOT: vector.body:
+; CHECK-LE: ret i4 %{{.*}}
;
+; CHECK-BE-LABEL: define i4 @test_i4_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: ret i4 %{{.*}}
entry:
br label %for.body
@@ -279,11 +325,15 @@ for.cond.cleanup: ; preds = %for.body
}
define i8 @test_load_i8(ptr %ddst) {
-; CHECK-LABEL: @test_load_i8(
-; CHECK: vector.body:
-; CHECK: load <4 x i8>, ptr {{.*}}, align 1, !nontemporal !0
-; CHECK: ret i8 %{{.*}}
+; CHECK-LE-LABEL: @test_load_i8(
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: load <4 x i8>, ptr {{.*}}, align 1, !nontemporal !0
+; CHECK-LE: ret i8 %{{.*}}
;
+; CHECK-BE-LABEL: @test_load_i8(
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: load i8, ptr {{.*}}, align 1, !nontemporal !0
+; CHECK-BE: ret i8 %{{.*}}
entry:
br label %for.body
@@ -302,11 +352,15 @@ for.cond.cleanup: ; preds = %for.body
}
define half @test_half_load(ptr %ddst) {
-; CHECK-LABEL: @test_half_load
-; CHECK-LABEL: vector.body:
-; CHECK: load <4 x half>, ptr {{.*}}, align 2, !nontemporal !0
-; CHECK: ret half %{{.*}}
+; CHECK-LE-LABEL: @test_half_load
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: load <4 x half>, ptr {{.*}}, align 2, !nontemporal !0
+; CHECK-LE: ret half %{{.*}}
;
+; CHECK-BE-LABEL: @test_half_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: load half, ptr {{.*}}, align 2, !nontemporal !0
+; CHECK-BE: ret half %{{.*}}
entry:
br label %for.body
@@ -325,11 +379,15 @@ for.cond.cleanup: ; preds = %for.body
}
define i16 @test_i16_load(ptr %ddst) {
-; CHECK-LABEL: @test_i16_load
-; CHECK-LABEL: vector.body:
-; CHECK: load <4 x i16>, ptr {{.*}}, align 2, !nontemporal !0
-; CHECK: ret i16 %{{.*}}
+; CHECK-LE-LABEL: @test_i16_load
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: load <4 x i16>, ptr {{.*}}, align 2, !nontemporal !0
+; CHECK-LE: ret i16 %{{.*}}
;
+; CHECK-BE-LABEL: @test_i16_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: load i16, ptr {{.*}}, align 2, !nontemporal !0
+; CHECK-BE: ret i16 %{{.*}}
entry:
br label %for.body
@@ -348,11 +406,15 @@ for.cond.cleanup: ; preds = %for.body
}
define i32 @test_i32_load(ptr %ddst) {
-; CHECK-LABEL: @test_i32_load
-; CHECK-LABEL: vector.body:
-; CHECK: load <4 x i32>, ptr {{.*}}, align 4, !nontemporal !0
-; CHECK: ret i32 %{{.*}}
+; CHECK-LE-LABEL: @test_i32_load
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: load <4 x i32>, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-LE: ret i32 %{{.*}}
;
+; CHECK-BE-LABEL: @test_i32_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: load i32, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-BE: ret i32 %{{.*}}
entry:
br label %for.body
@@ -371,10 +433,13 @@ for.cond.cleanup: ; preds = %for.body
}
define i33 @test_i33_load(ptr %ddst) {
-; CHECK-LABEL: @test_i33_load
-; CHECK-NOT: vector.body:
-; CHECK: ret i33 %{{.*}}
+; CHECK-LE-LABEL: @test_i33_load
+; CHECK-LE-NOT: vector.body:
+; CHECK-LE: ret i33 %{{.*}}
;
+; CHECK-BE-LABEL: @test_i33_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: ret i33 %{{.*}}
entry:
br label %for.body
@@ -393,10 +458,13 @@ for.cond.cleanup: ; preds = %for.body
}
define i40 @test_i40_load(ptr %ddst) {
-; CHECK-LABEL: @test_i40_load
-; CHECK-NOT: vector.body:
-; CHECK: ret i40 %{{.*}}
+; CHECK-LE-LABEL: @test_i40_load
+; CHECK-LE-NOT: vector.body:
+; CHECK-LE: ret i40 %{{.*}}
;
+; CHECK-BE-LABEL: @test_i40_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: ret i40 %{{.*}}
entry:
br label %for.body
@@ -415,11 +483,15 @@ for.cond.cleanup: ; preds = %for.body
}
define i64 @test_i64_load(ptr %ddst) {
-; CHECK-LABEL: @test_i64_load
-; CHECK-LABEL: vector.body:
-; CHECK: load <4 x i64>, ptr {{.*}}, align 4, !nontemporal !0
-; CHECK: ret i64 %{{.*}}
+; CHECK-LE-LABEL: @test_i64_load
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: load <4 x i64>, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-LE: ret i64 %{{.*}}
;
+; CHECK-BE-LABEL: @test_i64_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: load i64, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-BE: ret i64 %{{.*}}
entry:
br label %for.body
@@ -438,11 +510,15 @@ for.cond.cleanup: ; preds = %for.body
}
define double @test_double_load(ptr %ddst) {
-; CHECK-LABEL: @test_double_load
-; CHECK-LABEL: vector.body:
-; CHECK: load <4 x double>, ptr {{.*}}, align 4, !nontemporal !0
-; CHECK: ret double %{{.*}}
+; CHECK-LE-LABEL: @test_double_load
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: load <4 x double>, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-LE: ret double %{{.*}}
;
+; CHECK-BE-LABEL: @test_double_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: load double, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-BE: ret double %{{.*}}
entry:
br label %for.body
@@ -461,11 +537,15 @@ for.cond.cleanup: ; preds = %for.body
}
define i128 @test_i128_load(ptr %ddst) {
-; CHECK-LABEL: @test_i128_load
-; CHECK-LABEL: vector.body:
-; CHECK: load <4 x i128>, ptr {{.*}}, align 4, !nontemporal !0
-; CHECK: ret i128 %{{.*}}
+; CHECK-LE-LABEL: @test_i128_load
+; CHECK-LE-LABEL: vector.body:
+; CHECK-LE: load <4 x i128>, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-LE: ret i128 %{{.*}}
;
+; CHECK-BE-LABEL: @test_i128_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: load i128, ptr {{.*}}, align 4, !nontemporal !0
+; CHECK-BE: ret i128 %{{.*}}
entry:
br label %for.body
@@ -484,10 +564,13 @@ for.cond.cleanup: ; preds = %for.body
}
define i256 @test_256_load(ptr %ddst) {
-; CHECK-LABEL: @test_256_load
-; CHECK-NOT: vector.body:
-; CHECK: ret i256 %{{.*}}
+; CHECK-LE-LABEL: @test_256_load
+; CHECK-LE-NOT: vector.body:
+; CHECK-LE: ret i256 %{{.*}}
;
+; CHECK-BE-LABEL: @test_256_load
+; CHECK-BE-NOT: vector.body:
+; CHECK-BE: ret i256 %{{.*}}
entry:
br label %for.body
>From 05cdd55ed4bc37dab8c7ac7328a404f62049b9ed Mon Sep 17 00:00:00 2001
From: tomershafir <tomer.shafir8 at gmail.com>
Date: Mon, 26 Jan 2026 13:19:21 +0200
Subject: [PATCH 2/5] fix typo
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index eef52867107e9..871e42e4a3061 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -7323,7 +7323,7 @@ static SDValue LowerADDRSPACECAST(SDValue Op, SelectionDAG &DAG) {
// Lower non-temporal stores that would otherwise be broken by legalization.
//
-// Coordinated with LDNP constraints in
+// Coordinated with SDNP constraints in
// `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
// `AArch64TargetLowering::ReplaceNodeResults`
static SDValue LowerNTStore(StoreSDNode *StoreNode, EVT VT, EVT MemVT,
>From d2348536b9cbaafa79ad2946277cbb38b3e9754b Mon Sep 17 00:00:00 2001
From: tomershafir <tomer.shafir8 at gmail.com>
Date: Mon, 26 Jan 2026 13:26:32 +0200
Subject: [PATCH 3/5] match common code with CHECK prefix
---
.../AArch64/nontemporal-load-store.ll | 87 ++++++-------------
1 file changed, 27 insertions(+), 60 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll b/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
index 83a5d1b60a400..1ddc18142d127 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/nontemporal-load-store.ll
@@ -1,17 +1,12 @@
-; RUN: opt -passes=loop-vectorize -mtriple=aarch64 -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK-LE
-; RUN: opt -passes=loop-vectorize -mtriple=aarch64_be -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK-BE
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64 -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK,CHECK-LE
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64_be -force-vector-width=4 -force-vector-interleave=1 %s -S | FileCheck %s --check-prefixes=CHECK,CHECK-BE
; Vectors with i4 elements may not legal with nontemporal stores.
define void @test_i4_store(ptr %ddst) {
-; CHECK-LE-LABEL: define void @test_i4_store(
-; CHECK-LE-NOT: vector.body:
-; CHECK-LE: store i4 {{.*}} !nontemporal !0
-; CHECK-LE: ret void
-;
-; CHECK-BE-LABEL: define void @test_i4_store(
-; CHECK-BE-NOT: vector.body:
-; CHECK-BE: store i4 {{.*}} !nontemporal !0
-; CHECK-BE: ret void
+; CHECK-LABEL: define void @test_i4_store(
+; CHECK-NOT: vector.body:
+; CHECK: store i4 {{.*}} !nontemporal !0
+; CHECK: ret void
entry:
br label %for.body
@@ -139,13 +134,9 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_i33_store(ptr nocapture %ddst) {
-; CHECK-LE-LABEL: define void @test_i33_store(
-; CHECK-LE-NOT: vector.body:
-; CHECK-LE: ret
-;
-; CHECK-BE-LABEL: define void @test_i33_store(
-; CHECK-BE-NOT: vector.body:
-; CHECK-BE: ret
+; CHECK-LABEL: define void @test_i33_store(
+; CHECK-NOT: vector.body:
+; CHECK: ret
entry:
br label %for.body
@@ -169,13 +160,9 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_i40_store(ptr nocapture %ddst) {
-; CHECK-LE-LABEL: define void @test_i40_store(
-; CHECK-LE-NOT: vector.body:
-; CHECK-LE: ret
-;
-; CHECK-BE-LABEL: define void @test_i40_store(
-; CHECK-BE-NOT: vector.body:
-; CHECK-BE: ret
+; CHECK-LABEL: define void @test_i40_store(
+; CHECK-NOT: vector.body:
+; CHECK: ret
entry:
br label %for.body
@@ -276,13 +263,9 @@ for.cond.cleanup: ; preds = %for.body
}
define void @test_i256_store(ptr %ddst) {
-; CHECK-LE-LABEL: define void @test_i256_store(
-; CHECK-LE-NOT: vector.body:
-; CHECK-LE: ret void
-;
-; CHECK-BE-LABEL: define void @test_i256_store(
-; CHECK-BE-NOT: vector.body:
-; CHECK-BE: ret void
+; CHECK-LABEL: define void @test_i256_store(
+; CHECK-NOT: vector.body:
+; CHECK: ret void
entry:
br label %for.body
@@ -300,13 +283,9 @@ for.cond.cleanup: ; preds = %for.body
}
define i4 @test_i4_load(ptr %ddst) {
-; CHECK-LE-LABEL: define i4 @test_i4_load
-; CHECK-LE-NOT: vector.body:
-; CHECK-LE: ret i4 %{{.*}}
-;
-; CHECK-BE-LABEL: define i4 @test_i4_load
-; CHECK-BE-NOT: vector.body:
-; CHECK-BE: ret i4 %{{.*}}
+; CHECK-LABEL: define i4 @test_i4_load
+; CHECK-NOT: vector.body:
+; CHECK: ret i4 %{{.*}}
entry:
br label %for.body
@@ -433,13 +412,9 @@ for.cond.cleanup: ; preds = %for.body
}
define i33 @test_i33_load(ptr %ddst) {
-; CHECK-LE-LABEL: @test_i33_load
-; CHECK-LE-NOT: vector.body:
-; CHECK-LE: ret i33 %{{.*}}
-;
-; CHECK-BE-LABEL: @test_i33_load
-; CHECK-BE-NOT: vector.body:
-; CHECK-BE: ret i33 %{{.*}}
+; CHECK-LABEL: @test_i33_load
+; CHECK-NOT: vector.body:
+; CHECK: ret i33 %{{.*}}
entry:
br label %for.body
@@ -458,13 +433,9 @@ for.cond.cleanup: ; preds = %for.body
}
define i40 @test_i40_load(ptr %ddst) {
-; CHECK-LE-LABEL: @test_i40_load
-; CHECK-LE-NOT: vector.body:
-; CHECK-LE: ret i40 %{{.*}}
-;
-; CHECK-BE-LABEL: @test_i40_load
-; CHECK-BE-NOT: vector.body:
-; CHECK-BE: ret i40 %{{.*}}
+; CHECK-LABEL: @test_i40_load
+; CHECK-NOT: vector.body:
+; CHECK: ret i40 %{{.*}}
entry:
br label %for.body
@@ -564,13 +535,9 @@ for.cond.cleanup: ; preds = %for.body
}
define i256 @test_256_load(ptr %ddst) {
-; CHECK-LE-LABEL: @test_256_load
-; CHECK-LE-NOT: vector.body:
-; CHECK-LE: ret i256 %{{.*}}
-;
-; CHECK-BE-LABEL: @test_256_load
-; CHECK-BE-NOT: vector.body:
-; CHECK-BE: ret i256 %{{.*}}
+; CHECK-LABEL: @test_256_load
+; CHECK-NOT: vector.body:
+; CHECK: ret i256 %{{.*}}
entry:
br label %for.body
>From 7482684e80ff96062a52602ef6fce10f2bbfdd29 Mon Sep 17 00:00:00 2001
From: tomershafir <tomer.shafir8 at gmail.com>
Date: Mon, 26 Jan 2026 13:40:26 +0200
Subject: [PATCH 4/5] fix typo
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 871e42e4a3061..b1b8100391338 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -7323,7 +7323,7 @@ static SDValue LowerADDRSPACECAST(SDValue Op, SelectionDAG &DAG) {
// Lower non-temporal stores that would otherwise be broken by legalization.
//
-// Coordinated with SDNP constraints in
+// Coordinated with STNP constraints in
// `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
// `AArch64TargetLowering::ReplaceNodeResults`
static SDValue LowerNTStore(StoreSDNode *StoreNode, EVT VT, EVT MemVT,
>From e3c6b7ff5f53f69ec4a0a47426f4574b2fde1c4d Mon Sep 17 00:00:00 2001
From: tomershafir <tomer.shafir8 at gmail.com>
Date: Mon, 26 Jan 2026 14:13:30 +0200
Subject: [PATCH 5/5] [TTI] Add distinguished shouldVectorizeNTStore/Load
The issue this patch tries to solve is to differentiate between is an NT store/load directly legal on the target, and should an NT store/load be considered for vectorization early before we know the final vectorized store type. The differentiation is necessary for later passes in the pipeline to check legality without being over-permissive.
It adds 2 new TTI hooks `shouldVectorizeNTStore()` and `shouldVectorizeNTLoad()`, to be used by `LoopVectorizationLegality` when checking instruction legality before vectorization. In addition, it sinks `isLegalNTStore` and `isLegalNTLoad` to `TargetLowering.h` so that they can be easily called from the backend through TLI. I intend to submit a followup patch to introduce a new user of the TLI hooks, but I think this semantic differentiation deserves a patch of its own - this is effectively a NFC.
The default implementation of the new hooks has not changed, and they delegate to corresponding `isLegal` checks.
Note: this patch removes the AArch64 override for `isLegalNTStore` as its currently not used anymore, only to re-introduce it in a followup patch with a fixed implementation based on what the target actually supports, and a new user.
---
.../llvm/Analysis/TargetTransformInfo.h | 11 ++++--
.../llvm/Analysis/TargetTransformInfoImpl.h | 18 +++------
llvm/include/llvm/CodeGen/BasicTTIImpl.h | 10 +++++
llvm/include/llvm/CodeGen/TargetLowering.h | 18 +++++++++
llvm/lib/Analysis/TargetTransformInfo.cpp | 11 +++---
.../Target/AArch64/AArch64ISelLowering.cpp | 2 +-
llvm/lib/Target/AArch64/AArch64InstrInfo.td | 6 +--
.../AArch64/AArch64TargetTransformInfo.h | 18 ++++-----
llvm/lib/Target/X86/X86ISelLowering.cpp | 37 +++++++++++++++++++
llvm/lib/Target/X86/X86ISelLowering.h | 5 +++
.../lib/Target/X86/X86TargetTransformInfo.cpp | 35 ------------------
llvm/lib/Target/X86/X86TargetTransformInfo.h | 2 -
.../Vectorize/LoopVectorizationLegality.cpp | 13 ++++---
13 files changed, 110 insertions(+), 76 deletions(-)
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index 9db2e3977f71c..dc1dd88cad140 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -896,10 +896,13 @@ class TargetTransformInfo {
isLegalMaskedLoad(Type *DataType, Align Alignment, unsigned AddressSpace,
MaskKind MaskKind = VariableOrConstantMask) const;
- /// Return true if the target supports nontemporal store.
- LLVM_ABI bool isLegalNTStore(Type *DataType, Align Alignment) const;
- /// Return true if the target supports nontemporal load.
- LLVM_ABI bool isLegalNTLoad(Type *DataType, Align Alignment) const;
+ /// Returns true if the vectorization should try vectorize nontemporal stores.
+ /// Currently only used by the loop vectorizer.
+ LLVM_ABI bool shouldVectorizeNTStore(Type *DataType, Align Alignment) const;
+
+ /// Returns true if the vectorization should try vectorize nontemporal loads.
+ /// Currently only used by the loop vectorizer.
+ LLVM_ABI bool shouldVectorizeNTLoad(Type *DataType, Align Alignment) const;
/// \Returns true if the target supports broadcasting a load to a vector of
/// type <NumElements x ElementTy>.
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 07b3755924fd1..06dae244b260e 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -362,18 +362,12 @@ class TargetTransformInfoImplBase {
return false;
}
- virtual bool isLegalNTStore(Type *DataType, Align Alignment) const {
- // By default, assume nontemporal memory stores are available for stores
- // that are aligned and have a size that is a power of 2.
- unsigned DataSize = DL.getTypeStoreSize(DataType);
- return Alignment >= DataSize && isPowerOf2_32(DataSize);
- }
-
- virtual bool isLegalNTLoad(Type *DataType, Align Alignment) const {
- // By default, assume nontemporal memory loads are available for loads that
- // are aligned and have a size that is a power of 2.
- unsigned DataSize = DL.getTypeStoreSize(DataType);
- return Alignment >= DataSize && isPowerOf2_32(DataSize);
+ virtual bool shouldVectorizeNTStore(Type *DataType, Align Alignment) const {
+ return false;
+ }
+
+ virtual bool shouldVectorizeNTLoad(Type *DataType, Align Alignment) const {
+ return false;
}
virtual bool isLegalBroadcastLoad(Type *ElementTy,
diff --git a/llvm/include/llvm/CodeGen/BasicTTIImpl.h b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
index c430e11168f73..b296fdd6b1fd5 100644
--- a/llvm/include/llvm/CodeGen/BasicTTIImpl.h
+++ b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
@@ -476,6 +476,16 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
return getTLI()->isLegalAddressingMode(DL, AM, Ty, AddrSpace, I);
}
+ bool shouldVectorizeNTStore(Type *DataType, Align Alignment) const override {
+ const DataLayout &DL = this->getDataLayout();
+ return getTLI()->isLegalNTStore(DataType, Alignment, DL);
+ }
+
+ bool shouldVectorizeNTLoad(Type *DataType, Align Alignment) const override {
+ const DataLayout &DL = this->getDataLayout();
+ return getTLI()->isLegalNTLoad(DataType, Alignment, DL);
+ }
+
int64_t getPreferredLargeGEPBaseOffset(int64_t MinOffset, int64_t MaxOffset) {
return getTLI()->getPreferredLargeGEPBaseOffset(MinOffset, MaxOffset);
}
diff --git a/llvm/include/llvm/CodeGen/TargetLowering.h b/llvm/include/llvm/CodeGen/TargetLowering.h
index d8b5274d37c85..d8c53db1c8ad7 100644
--- a/llvm/include/llvm/CodeGen/TargetLowering.h
+++ b/llvm/include/llvm/CodeGen/TargetLowering.h
@@ -3003,6 +3003,24 @@ class LLVM_ABI TargetLoweringBase {
return Value == 0;
}
+ /// Return true if the target supports nontemporal store.
+ virtual bool isLegalNTStore(Type *DataType, Align Alignment,
+ const DataLayout &DL) const {
+ // By default, assume nontemporal memory stores are available for stores
+ // that are aligned and have a size that is a power of 2.
+ unsigned DataSize = DL.getTypeStoreSize(DataType);
+ return Alignment >= DataSize && isPowerOf2_32(DataSize);
+ }
+
+ /// Return true if the target supports nontemporal load.
+ virtual bool isLegalNTLoad(Type *DataType, Align Alignment,
+ const DataLayout &DL) const {
+ // By default, assume nontemporal memory loads are available for loads that
+ // are aligned and have a size that is a power of 2.
+ unsigned DataSize = DL.getTypeStoreSize(DataType);
+ return Alignment >= DataSize && isPowerOf2_32(DataSize);
+ }
+
/// Given a shuffle vector SVI representing a vector splat, return a new
/// scalar type of size equal to SVI's scalar type if the new type is more
/// profitable. Returns nullptr otherwise. For example under MVE float splats
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index b2b77da4914d6..bdf8610eaeac7 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -490,13 +490,14 @@ bool TargetTransformInfo::isLegalMaskedLoad(Type *DataType, Align Alignment,
MaskKind);
}
-bool TargetTransformInfo::isLegalNTStore(Type *DataType,
- Align Alignment) const {
- return TTIImpl->isLegalNTStore(DataType, Alignment);
+bool TargetTransformInfo::shouldVectorizeNTStore(Type *DataType,
+ Align Alignment) const {
+ return TTIImpl->shouldVectorizeNTStore(DataType, Alignment);
}
-bool TargetTransformInfo::isLegalNTLoad(Type *DataType, Align Alignment) const {
- return TTIImpl->isLegalNTLoad(DataType, Alignment);
+bool TargetTransformInfo::shouldVectorizeNTLoad(Type *DataType,
+ Align Alignment) const {
+ return TTIImpl->shouldVectorizeNTLoad(DataType, Alignment);
}
bool TargetTransformInfo::isLegalBroadcastLoad(Type *ElementTy,
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index b1b8100391338..8113375ce52c3 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -29573,7 +29573,7 @@ void AArch64TargetLowering::ReplaceNodeResults(
//
// Coordinated with LDNP constraints in
// `llvm/lib/Target/AArch64/AArch64InstrInfo.td`
- // and `AArch64TTIImpl::isLegalNTLoad`.
+ // and `AArch64TargetLowering::isLegalNTLoad`.
if (LoadNode->isNonTemporal() && Subtarget->isLittleEndian() &&
MemVT.getSizeInBits() == 256u &&
(MemVT.getScalarSizeInBits() == 8u ||
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.td b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
index ee9e3fa886a5c..ba5a4e0fa4b53 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
@@ -3789,7 +3789,7 @@ def : Pat<(AArch64ldp (am_indexed7s64 GPR64sp:$Rn, simm7s8:$offset)),
// Currently we only support NT loads lowering for little-endian targets.
//
// Coordinated with LDNP constraints in `AArch64TargetLowering::ReplaceNodeResults`
-// and `AArch64TTIImpl::isLegalNTLoad`.
+// and `AArch64TargetLowering::isLegalNTLoad`.
let Predicates = [IsLE] in {
def : Pat<(AArch64ldnp(am_indexed7s128 GPR64sp:$Rn, simm7s16:$offset)),
(LDNPQi GPR64sp:$Rn, simm7s16:$offset)>;
@@ -10733,8 +10733,8 @@ def : Pat<(i64 (int_aarch64_neon_urshl (i64 FPR64:$Rn), (i64 FPR64:$Rm))),
//
// Currently we only support NT stores lowering for little-endian targets.
//
-// Coordinated with STNP constraints in `AArch64TargetLowering::LowerNTStore`
-// and `AArch64TTIImpl::isLegalNTStore`.
+// Coordinated with STNP constraints in `LowerNTStore`
+// and `AArch64TargetLowering::isLegalNTStore`.
//
// Currently, STNP lowering can only either keep or increase code size, thus
// we predicate it to not apply when optimizing for code size.
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index b789e13bf25da..1b6da612e76d4 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -391,8 +391,8 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
return false;
}
- std::optional<bool> isLegalNTStoreLoad(Type *DataType,
- Align Alignment) const {
+ std::optional<bool> shouldVectorizeNTStoreLoad(Type *DataType,
+ Align Alignment) const {
// NOTE: The logic below is mostly geared towards LV, which calls it with
// vectors with 2 elements. We might want to improve that, if other
// users show up.
@@ -409,21 +409,21 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
return std::nullopt;
}
- bool isLegalNTStore(Type *DataType, Align Alignment) const override {
+ bool shouldVectorizeNTStore(Type *DataType, Align Alignment) const override {
// Currently we only support NT stores lowering for little-endian targets.
//
// Coordinated with STNP constraints in
// `llvm/lib/Target/AArch64/AArch64InstrInfo.td` and
- // `AArch64TargetLowering::LowerNTStore`
+ // `LowerNTStore`
if (!ST->isLittleEndian())
return false;
- if (auto Result = isLegalNTStoreLoad(DataType, Alignment))
+ if (auto Result = shouldVectorizeNTStoreLoad(DataType, Alignment))
return *Result;
// Fallback to target independent logic
- return BaseT::isLegalNTStore(DataType, Alignment);
+ return BaseT::shouldVectorizeNTStore(DataType, Alignment);
}
- bool isLegalNTLoad(Type *DataType, Align Alignment) const override {
+ bool shouldVectorizeNTLoad(Type *DataType, Align Alignment) const override {
// Currently we only support NT loads lowering for little-endian targets.
//
// Coordinated with LDNP constraints in
@@ -431,10 +431,10 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
// `AArch64TargetLowering::ReplaceNodeResults`
if (!ST->isLittleEndian())
return false;
- if (auto Result = isLegalNTStoreLoad(DataType, Alignment))
+ if (auto Result = shouldVectorizeNTStoreLoad(DataType, Alignment))
return *Result;
// Fallback to target independent logic
- return BaseT::isLegalNTLoad(DataType, Alignment);
+ return BaseT::shouldVectorizeNTLoad(DataType, Alignment);
}
InstructionCost getPartialReductionCost(
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index e54f4ed2fb26c..dc5cd618f4e19 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -36037,6 +36037,43 @@ bool X86TargetLowering::isLegalStoreImmediate(int64_t Imm) const {
return isInt<32>(Imm);
}
+bool X86TTIImpl::isLegalNTLoad(Type *DataType, Align Alignment,
+ const DataLayout &DL) const {
+ unsigned DataSize = DL.getTypeStoreSize(DataType);
+ // The only supported nontemporal loads are for aligned vectors of 16 or 32
+ // bytes. Note that 32-byte nontemporal vector loads are supported by AVX2
+ // (the equivalent stores only require AVX).
+ if (Alignment >= DataSize && (DataSize == 16 || DataSize == 32))
+ return DataSize == 16 ? ST->hasSSE1() : ST->hasAVX2();
+
+ return false;
+}
+
+bool X86TTIImpl::isLegalNTStore(Type *DataType, Align Alignment,
+ const DataLayout &DL) const {
+ unsigned DataSize = DL.getTypeStoreSize(DataType);
+
+ // SSE4A supports nontemporal stores of float and double at arbitrary
+ // alignment.
+ if (ST->hasSSE4A() && (DataType->isFloatTy() || DataType->isDoubleTy()))
+ return true;
+
+ // Besides the SSE4A subtarget exception above, only aligned stores are
+ // available nontemporaly on any other subtarget. And only stores with a size
+ // of 4..32 bytes (powers of 2, only) are permitted.
+ if (Alignment < DataSize || DataSize < 4 || DataSize > 32 ||
+ !isPowerOf2_32(DataSize))
+ return false;
+
+ // 32-byte vector nontemporal stores are supported by AVX (the equivalent
+ // loads require AVX2).
+ if (DataSize == 32)
+ return ST->hasAVX();
+ if (DataSize == 16)
+ return ST->hasSSE1();
+ return true;
+}
+
bool X86TargetLowering::isTruncateFree(EVT VT1, EVT VT2) const {
if (!VT1.isScalarInteger() || !VT2.isScalarInteger())
return false;
diff --git a/llvm/lib/Target/X86/X86ISelLowering.h b/llvm/lib/Target/X86/X86ISelLowering.h
index 30faca54e13f6..17c424add229a 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.h
+++ b/llvm/lib/Target/X86/X86ISelLowering.h
@@ -1456,6 +1456,11 @@ namespace llvm {
bool isLegalStoreImmediate(int64_t Imm) const override;
+ bool isLegalNTLoad(Type *DataType, Align Alignment,
+ const DataLayout &DL) const override;
+ bool isLegalNTStore(Type *DataType, Align Alignment,
+ const DataLayout &DL) const override;
+
/// Add x86-specific opcodes to the default list.
bool isBinOp(unsigned Opcode) const override;
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index 608727b745925..40ca2665c0d94 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -6367,41 +6367,6 @@ bool X86TTIImpl::isLegalMaskedStore(Type *DataTy, Align Alignment,
return isLegalMaskedLoadStore(ScalarTy, ST);
}
-bool X86TTIImpl::isLegalNTLoad(Type *DataType, Align Alignment) const {
- unsigned DataSize = DL.getTypeStoreSize(DataType);
- // The only supported nontemporal loads are for aligned vectors of 16 or 32
- // bytes. Note that 32-byte nontemporal vector loads are supported by AVX2
- // (the equivalent stores only require AVX).
- if (Alignment >= DataSize && (DataSize == 16 || DataSize == 32))
- return DataSize == 16 ? ST->hasSSE1() : ST->hasAVX2();
-
- return false;
-}
-
-bool X86TTIImpl::isLegalNTStore(Type *DataType, Align Alignment) const {
- unsigned DataSize = DL.getTypeStoreSize(DataType);
-
- // SSE4A supports nontemporal stores of float and double at arbitrary
- // alignment.
- if (ST->hasSSE4A() && (DataType->isFloatTy() || DataType->isDoubleTy()))
- return true;
-
- // Besides the SSE4A subtarget exception above, only aligned stores are
- // available nontemporaly on any other subtarget. And only stores with a size
- // of 4..32 bytes (powers of 2, only) are permitted.
- if (Alignment < DataSize || DataSize < 4 || DataSize > 32 ||
- !isPowerOf2_32(DataSize))
- return false;
-
- // 32-byte vector nontemporal stores are supported by AVX (the equivalent
- // loads require AVX2).
- if (DataSize == 32)
- return ST->hasAVX();
- if (DataSize == 16)
- return ST->hasSSE1();
- return true;
-}
-
bool X86TTIImpl::isLegalBroadcastLoad(Type *ElementTy,
ElementCount NumElements) const {
// movddup
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.h b/llvm/lib/Target/X86/X86TargetTransformInfo.h
index 4f672793a0fcf..451d03fce44a5 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.h
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.h
@@ -274,8 +274,6 @@ class X86TTIImpl final : public BasicTTIImplBase<X86TTIImpl> {
isLegalMaskedStore(Type *DataType, Align Alignment, unsigned AddressSpace,
TTI::MaskKind MaskKind =
TTI::MaskKind::VariableOrConstantMask) const override;
- bool isLegalNTLoad(Type *DataType, Align Alignment) const override;
- bool isLegalNTStore(Type *DataType, Align Alignment) const override;
bool isLegalBroadcastLoad(Type *ElementTy,
ElementCount NumElements) const override;
bool forceScalarizeMaskedGather(VectorType *VTy,
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index ea431b4fa5df8..a0dd100f0c903 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -1045,10 +1045,11 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
// For nontemporal stores, check that a nontemporal vector version is
// supported on the target.
if (ST->getMetadata(LLVMContext::MD_nontemporal)) {
- // Arbitrarily try a vector of 2 elements.
+ // Check a 2-element vector type, which implictly covers any power-of-2
+ // sized vector type by logically splitting to pairs
auto *VecTy = FixedVectorType::get(T, /*NumElts=*/2);
assert(VecTy && "did not find vectorized version of stored type");
- if (!TTI->isLegalNTStore(VecTy, ST->getAlign())) {
+ if (!TTI->shouldVectorizeNTStore(VecTy, ST->getAlign())) {
reportVectorizationFailure(
"nontemporal store instruction cannot be vectorized",
"CantVectorizeNontemporalStore", ORE, TheLoop, ST);
@@ -1057,12 +1058,14 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
}
} else if (auto *LD = dyn_cast<LoadInst>(&I)) {
+ // For nontemporal loads, check that a nontemporal vector version is
+ // supported on the target.
if (LD->getMetadata(LLVMContext::MD_nontemporal)) {
- // For nontemporal loads, check that a nontemporal vector version is
- // supported on the target (arbitrarily try a vector of 2 elements).
+ // Check a 2-element vector type, which implictly covers any power-of-2
+ // sized vector type by logically splitting to pairs
auto *VecTy = FixedVectorType::get(I.getType(), /*NumElts=*/2);
assert(VecTy && "did not find vectorized version of load type");
- if (!TTI->isLegalNTLoad(VecTy, LD->getAlign())) {
+ if (!TTI->shouldVectorizeNTLoad(VecTy, LD->getAlign())) {
reportVectorizationFailure(
"nontemporal load instruction cannot be vectorized",
"CantVectorizeNontemporalLoad", ORE, TheLoop, LD);
More information about the llvm-commits
mailing list