[llvm] [AArch64] Disable consecutive store merging when Neon is unavailable (PR #111519)
Benjamin Maxwell via llvm-commits
llvm-commits at lists.llvm.org
Tue Oct 8 06:56:31 PDT 2024
================
@@ -0,0 +1,138 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -mtriple=aarch64-none-linux-gnu -mattr=+sve -O3 < %s -o - | FileCheck %s --check-prefixes=CHECK
+
+; Tests consecutive stores of @llvm.aarch64.sve.faddv. Within SDAG faddv is
+; lowered as a FADDV + EXTRACT_VECTOR_ELT (of lane 0). Stores of extracts can
+; be matched by DAGCombiner::mergeConsecutiveStores(), which we want to avoid in
+; some cases as it can lead to worse codegen.
+
+; TODO: A single `stp s0, s1, [x0]` may be preferred here.
+define void @consecutive_stores_pair(ptr noalias %dest0, ptr noalias %src0) {
+; CHECK-LABEL: consecutive_stores_pair:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: ld1w { z0.s }, p0/z, [x1]
+; CHECK-NEXT: ld1w { z1.s }, p0/z, [x1, #1, mul vl]
+; CHECK-NEXT: faddv s0, p0, z0.s
+; CHECK-NEXT: faddv s1, p0, z1.s
+; CHECK-NEXT: mov v0.s[1], v1.s[0]
+; CHECK-NEXT: str d0, [x0]
+; CHECK-NEXT: ret
+ %ptrue = call <vscale x 4 x i1> @llvm.aarch64.sve.ptrue.nxv4i1(i32 31)
+ %vscale = call i64 @llvm.vscale.i64()
+ %c4_vscale = shl i64 %vscale, 2
+ %src1 = getelementptr inbounds float, ptr %src0, i64 %c4_vscale
+ %dest1 = getelementptr inbounds i8, ptr %dest0, i64 4
+ %vec0 = load <vscale x 4 x float>, ptr %src0, align 4
+ %vec1 = load <vscale x 4 x float>, ptr %src1, align 4
----------------
MacDue wrote:
Just leftovers from when I was reducing the issue, removed now :+1:
https://github.com/llvm/llvm-project/pull/111519
More information about the llvm-commits
mailing list