[llvm] [SLP][NFC]Add a test with the non-profitable vectorization, NFC (PR #228063)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Thu Oct 1 05:56:04 PDT 2026
https://github.com/alexey-bataev created https://github.com/llvm/llvm-project/pull/228063
None
>From ce16283c552ae277da70eee2e6e0b69f322a7b8c Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Thu, 1 Oct 2026 05:55:49 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../runtime-alias-checks-vectorized-loop.ll | 373 ++++++++++++++++++
1 file changed, 373 insertions(+)
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/runtime-alias-checks-vectorized-loop.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/runtime-alias-checks-vectorized-loop.ll
index d95203173bf7b..99f2b3145fdb8 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/runtime-alias-checks-vectorized-loop.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/runtime-alias-checks-vectorized-loop.ll
@@ -187,15 +187,388 @@ exit:
ret void
}
+; The fmul and fadd pairs, fused by the backend, cost as the fmuladd in the
+; scalar region cost, so the tighter bound rejects the versioning.
+define void @vectorized_loop_contract(ptr %m, ptr %in, ptr %out, i64 %n) {
+; CHECK-LABEL: define void @vectorized_loop_contract(
+; CHECK-SAME: ptr [[M:%.*]], ptr [[IN:%.*]], ptr [[OUT:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[M16:%.*]] = ptrtoaddr ptr [[M]] to i64
+; CHECK-NEXT: [[M1_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 4
+; CHECK-NEXT: [[M2_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 8
+; CHECK-NEXT: [[M3_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 12
+; CHECK-NEXT: [[M4_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 16
+; CHECK-NEXT: [[M5_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 20
+; CHECK-NEXT: [[M6_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 24
+; CHECK-NEXT: [[M7_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 28
+; CHECK-NEXT: [[M8_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 32
+; CHECK-NEXT: [[M9_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 36
+; CHECK-NEXT: [[M10_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 40
+; CHECK-NEXT: [[M11_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 44
+; CHECK-NEXT: [[M12_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 48
+; CHECK-NEXT: [[M13_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 52
+; CHECK-NEXT: [[M14_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 56
+; CHECK-NEXT: [[M15_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 60
+; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[M16]], 64
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT_RTMERGE:%.*]], %[[EXIT:.*]] ]
+; CHECK-NEXT: [[SRC:%.*]] = phi ptr [ [[IN]], %[[ENTRY]] ], [ [[SRC_NEXT_RTMERGE:%.*]], %[[EXIT]] ]
+; CHECK-NEXT: [[DST:%.*]] = phi ptr [ [[OUT]], %[[ENTRY]] ], [ [[DST_NEXT_RTMERGE:%.*]], %[[EXIT]] ]
+; CHECK-NEXT: [[DST17:%.*]] = ptrtoaddr ptr [[DST]] to i64
+; CHECK-NEXT: [[TMP1:%.*]] = add i64 [[DST17]], 16
+; CHECK-NEXT: [[RT_BOUND0:%.*]] = icmp ult i64 [[DST17]], [[TMP0]]
+; CHECK-NEXT: [[RT_BOUND1:%.*]] = icmp ult i64 [[M16]], [[TMP1]]
+; CHECK-NEXT: [[RT_CONFLICT:%.*]] = and i1 [[RT_BOUND0]], [[RT_BOUND1]]
+; CHECK-NEXT: [[RT_GUARD:%.*]] = freeze i1 [[RT_CONFLICT]]
+; CHECK-NEXT: br i1 [[RT_GUARD]], label %[[LOOP_RTSCALAR:.*]], label %[[LOOP_RTVEC:.*]], !prof [[PROF2:![0-9]+]]
+; CHECK: [[EXIT1:.*]]:
+; CHECK-NEXT: ret void
+; CHECK: [[LOOP_RTVEC]]:
+; CHECK-NEXT: [[R:%.*]] = load float, ptr [[SRC]], align 4
+; CHECK-NEXT: [[G_P:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 4
+; CHECK-NEXT: [[G:%.*]] = load float, ptr [[G_P]], align 4
+; CHECK-NEXT: [[B_P:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 8
+; CHECK-NEXT: [[B:%.*]] = load float, ptr [[B_P]], align 4
+; CHECK-NEXT: [[A_P:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 12
+; CHECK-NEXT: [[A:%.*]] = load float, ptr [[A_P]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[M]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x float> poison, float [[R]], i64 0
+; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <4 x float> [[TMP3]], <4 x float> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP5:%.*]] = fmul contract <4 x float> [[TMP2]], [[TMP4]]
+; CHECK-NEXT: [[TMP6:%.*]] = load <4 x float>, ptr [[M4_P]], align 4
+; CHECK-NEXT: [[TMP7:%.*]] = insertelement <4 x float> poison, float [[G]], i64 0
+; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <4 x float> [[TMP7]], <4 x float> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP9:%.*]] = fmul contract <4 x float> [[TMP6]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = fadd contract <4 x float> [[TMP9]], [[TMP5]]
+; CHECK-NEXT: [[TMP11:%.*]] = load <4 x float>, ptr [[M8_P]], align 4
+; CHECK-NEXT: [[TMP12:%.*]] = insertelement <4 x float> poison, float [[B]], i64 0
+; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <4 x float> [[TMP12]], <4 x float> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP14:%.*]] = fmul contract <4 x float> [[TMP11]], [[TMP13]]
+; CHECK-NEXT: [[TMP15:%.*]] = fadd contract <4 x float> [[TMP10]], [[TMP14]]
+; CHECK-NEXT: [[TMP16:%.*]] = load <4 x float>, ptr [[M12_P]], align 4
+; CHECK-NEXT: [[TMP17:%.*]] = insertelement <4 x float> poison, float [[A]], i64 0
+; CHECK-NEXT: [[TMP18:%.*]] = shufflevector <4 x float> [[TMP17]], <4 x float> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP19:%.*]] = fmul contract <4 x float> [[TMP16]], [[TMP18]]
+; CHECK-NEXT: [[TMP20:%.*]] = fadd contract <4 x float> [[TMP15]], [[TMP19]]
+; CHECK-NEXT: store <4 x float> [[TMP20]], ptr [[DST]], align 4
+; CHECK-NEXT: [[SRC_NEXT1:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 16
+; CHECK-NEXT: [[DST_NEXT1:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 16
+; CHECK-NEXT: [[IV_NEXT1:%.*]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[DONE1:%.*]] = icmp eq i64 [[IV_NEXT1]], [[N]]
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[LOOP_RTSCALAR]]:
+; CHECK-NEXT: [[R_SCALAR:%.*]] = load float, ptr [[SRC]], align 4
+; CHECK-NEXT: [[G_P_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 4
+; CHECK-NEXT: [[G_SCALAR:%.*]] = load float, ptr [[G_P_SCALAR]], align 4
+; CHECK-NEXT: [[B_P_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 8
+; CHECK-NEXT: [[B_SCALAR:%.*]] = load float, ptr [[B_P_SCALAR]], align 4
+; CHECK-NEXT: [[A_P_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 12
+; CHECK-NEXT: [[A_SCALAR:%.*]] = load float, ptr [[A_P_SCALAR]], align 4
+; CHECK-NEXT: [[M0:%.*]] = load float, ptr [[M]], align 4
+; CHECK-NEXT: [[R0:%.*]] = fmul contract float [[M0]], [[R_SCALAR]]
+; CHECK-NEXT: [[M4:%.*]] = load float, ptr [[M4_P]], align 4
+; CHECK-NEXT: [[G0:%.*]] = fmul contract float [[M4]], [[G_SCALAR]]
+; CHECK-NEXT: [[S0:%.*]] = fadd contract float [[G0]], [[R0]]
+; CHECK-NEXT: [[M8:%.*]] = load float, ptr [[M8_P]], align 4
+; CHECK-NEXT: [[B0:%.*]] = fmul contract float [[M8]], [[B_SCALAR]]
+; CHECK-NEXT: [[T0:%.*]] = fadd contract float [[S0]], [[B0]]
+; CHECK-NEXT: [[M12:%.*]] = load float, ptr [[M12_P]], align 4
+; CHECK-NEXT: [[A0:%.*]] = fmul contract float [[M12]], [[A_SCALAR]]
+; CHECK-NEXT: [[X0:%.*]] = fadd contract float [[T0]], [[A0]]
+; CHECK-NEXT: store float [[X0]], ptr [[DST]], align 4
+; CHECK-NEXT: [[M1:%.*]] = load float, ptr [[M1_P]], align 4
+; CHECK-NEXT: [[R1:%.*]] = fmul contract float [[M1]], [[R_SCALAR]]
+; CHECK-NEXT: [[M5:%.*]] = load float, ptr [[M5_P]], align 4
+; CHECK-NEXT: [[G1:%.*]] = fmul contract float [[M5]], [[G_SCALAR]]
+; CHECK-NEXT: [[S1:%.*]] = fadd contract float [[G1]], [[R1]]
+; CHECK-NEXT: [[M9:%.*]] = load float, ptr [[M9_P]], align 4
+; CHECK-NEXT: [[B1:%.*]] = fmul contract float [[M9]], [[B_SCALAR]]
+; CHECK-NEXT: [[T1:%.*]] = fadd contract float [[S1]], [[B1]]
+; CHECK-NEXT: [[M13:%.*]] = load float, ptr [[M13_P]], align 4
+; CHECK-NEXT: [[A1:%.*]] = fmul contract float [[M13]], [[A_SCALAR]]
+; CHECK-NEXT: [[X1:%.*]] = fadd contract float [[T1]], [[A1]]
+; CHECK-NEXT: [[DST1:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 4
+; CHECK-NEXT: store float [[X1]], ptr [[DST1]], align 4
+; CHECK-NEXT: [[M2:%.*]] = load float, ptr [[M2_P]], align 4
+; CHECK-NEXT: [[R2:%.*]] = fmul contract float [[M2]], [[R_SCALAR]]
+; CHECK-NEXT: [[M6:%.*]] = load float, ptr [[M6_P]], align 4
+; CHECK-NEXT: [[G2:%.*]] = fmul contract float [[M6]], [[G_SCALAR]]
+; CHECK-NEXT: [[S2:%.*]] = fadd contract float [[G2]], [[R2]]
+; CHECK-NEXT: [[M10:%.*]] = load float, ptr [[M10_P]], align 4
+; CHECK-NEXT: [[B2:%.*]] = fmul contract float [[M10]], [[B_SCALAR]]
+; CHECK-NEXT: [[T2:%.*]] = fadd contract float [[S2]], [[B2]]
+; CHECK-NEXT: [[M14:%.*]] = load float, ptr [[M14_P]], align 4
+; CHECK-NEXT: [[A2:%.*]] = fmul contract float [[M14]], [[A_SCALAR]]
+; CHECK-NEXT: [[X2:%.*]] = fadd contract float [[T2]], [[A2]]
+; CHECK-NEXT: [[DST2:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 8
+; CHECK-NEXT: store float [[X2]], ptr [[DST2]], align 4
+; CHECK-NEXT: [[M3:%.*]] = load float, ptr [[M3_P]], align 4
+; CHECK-NEXT: [[R3:%.*]] = fmul contract float [[M3]], [[R_SCALAR]]
+; CHECK-NEXT: [[M7:%.*]] = load float, ptr [[M7_P]], align 4
+; CHECK-NEXT: [[G3:%.*]] = fmul contract float [[M7]], [[G_SCALAR]]
+; CHECK-NEXT: [[S3:%.*]] = fadd contract float [[G3]], [[R3]]
+; CHECK-NEXT: [[M11:%.*]] = load float, ptr [[M11_P]], align 4
+; CHECK-NEXT: [[B3:%.*]] = fmul contract float [[M11]], [[B_SCALAR]]
+; CHECK-NEXT: [[T3:%.*]] = fadd contract float [[S3]], [[B3]]
+; CHECK-NEXT: [[M15:%.*]] = load float, ptr [[M15_P]], align 4
+; CHECK-NEXT: [[A3:%.*]] = fmul contract float [[M15]], [[A_SCALAR]]
+; CHECK-NEXT: [[X3:%.*]] = fadd contract float [[T3]], [[A3]]
+; CHECK-NEXT: [[DST3:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 12
+; CHECK-NEXT: store float [[X3]], ptr [[DST3]], align 4
+; CHECK-NEXT: [[SRC_NEXT:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 16
+; CHECK-NEXT: [[DST_NEXT:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 16
+; CHECK-NEXT: [[IV_NEXT:%.*]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[SRC_NEXT_RTMERGE]] = phi ptr [ [[SRC_NEXT1]], %[[LOOP_RTVEC]] ], [ [[SRC_NEXT]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT: [[DST_NEXT_RTMERGE]] = phi ptr [ [[DST_NEXT1]], %[[LOOP_RTVEC]] ], [ [[DST_NEXT]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT: [[IV_NEXT_RTMERGE]] = phi i64 [ [[IV_NEXT1]], %[[LOOP_RTVEC]] ], [ [[IV_NEXT]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT: [[DONE_RTMERGE:%.*]] = phi i1 [ [[DONE1]], %[[LOOP_RTVEC]] ], [ [[DONE]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT: br i1 [[DONE_RTMERGE]], label %[[EXIT1]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+;
+; VER-LABEL: define void @vectorized_loop_contract(
+; VER-SAME: ptr [[M:%.*]], ptr [[IN:%.*]], ptr [[OUT:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; VER-NEXT: [[ENTRY:.*]]:
+; VER-NEXT: [[M16:%.*]] = ptrtoaddr ptr [[M]] to i64
+; VER-NEXT: [[M1_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 4
+; VER-NEXT: [[M2_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 8
+; VER-NEXT: [[M3_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 12
+; VER-NEXT: [[M4_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 16
+; VER-NEXT: [[M5_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 20
+; VER-NEXT: [[M6_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 24
+; VER-NEXT: [[M7_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 28
+; VER-NEXT: [[M8_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 32
+; VER-NEXT: [[M9_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 36
+; VER-NEXT: [[M10_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 40
+; VER-NEXT: [[M11_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 44
+; VER-NEXT: [[M12_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 48
+; VER-NEXT: [[M13_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 52
+; VER-NEXT: [[M14_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 56
+; VER-NEXT: [[M15_P:%.*]] = getelementptr inbounds i8, ptr [[M]], i64 60
+; VER-NEXT: [[TMP0:%.*]] = add i64 [[M16]], 64
+; VER-NEXT: br label %[[LOOP:.*]]
+; VER: [[LOOP]]:
+; VER-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT_RTMERGE:%.*]], %[[LOOP_RTCONT:.*]] ]
+; VER-NEXT: [[SRC:%.*]] = phi ptr [ [[IN]], %[[ENTRY]] ], [ [[SRC_NEXT_RTMERGE:%.*]], %[[LOOP_RTCONT]] ]
+; VER-NEXT: [[DST:%.*]] = phi ptr [ [[OUT]], %[[ENTRY]] ], [ [[DST_NEXT_RTMERGE:%.*]], %[[LOOP_RTCONT]] ]
+; VER-NEXT: [[DST17:%.*]] = ptrtoaddr ptr [[DST]] to i64
+; VER-NEXT: [[TMP1:%.*]] = add i64 [[DST17]], 16
+; VER-NEXT: [[RT_BOUND0:%.*]] = icmp ult i64 [[DST17]], [[TMP0]]
+; VER-NEXT: [[RT_BOUND1:%.*]] = icmp ult i64 [[M16]], [[TMP1]]
+; VER-NEXT: [[RT_CONFLICT:%.*]] = and i1 [[RT_BOUND0]], [[RT_BOUND1]]
+; VER-NEXT: [[RT_GUARD:%.*]] = freeze i1 [[RT_CONFLICT]]
+; VER-NEXT: br i1 [[RT_GUARD]], label %[[LOOP_RTSCALAR:.*]], label %[[LOOP_RTVEC:.*]], !prof [[PROF0]]
+; VER: [[EXIT:.*]]:
+; VER-NEXT: ret void
+; VER: [[LOOP_RTVEC]]:
+; VER-NEXT: [[R:%.*]] = load float, ptr [[SRC]], align 4
+; VER-NEXT: [[G_P:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 4
+; VER-NEXT: [[G:%.*]] = load float, ptr [[G_P]], align 4
+; VER-NEXT: [[B_P:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 8
+; VER-NEXT: [[B:%.*]] = load float, ptr [[B_P]], align 4
+; VER-NEXT: [[A_P:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 12
+; VER-NEXT: [[A:%.*]] = load float, ptr [[A_P]], align 4
+; VER-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[M]], align 4
+; VER-NEXT: [[TMP3:%.*]] = insertelement <4 x float> poison, float [[R]], i64 0
+; VER-NEXT: [[TMP4:%.*]] = shufflevector <4 x float> [[TMP3]], <4 x float> poison, <4 x i32> zeroinitializer
+; VER-NEXT: [[TMP5:%.*]] = fmul contract <4 x float> [[TMP2]], [[TMP4]]
+; VER-NEXT: [[TMP6:%.*]] = load <4 x float>, ptr [[M4_P]], align 4
+; VER-NEXT: [[TMP7:%.*]] = insertelement <4 x float> poison, float [[G]], i64 0
+; VER-NEXT: [[TMP8:%.*]] = shufflevector <4 x float> [[TMP7]], <4 x float> poison, <4 x i32> zeroinitializer
+; VER-NEXT: [[TMP9:%.*]] = fmul contract <4 x float> [[TMP6]], [[TMP8]]
+; VER-NEXT: [[TMP10:%.*]] = fadd contract <4 x float> [[TMP9]], [[TMP5]]
+; VER-NEXT: [[TMP11:%.*]] = load <4 x float>, ptr [[M8_P]], align 4
+; VER-NEXT: [[TMP12:%.*]] = insertelement <4 x float> poison, float [[B]], i64 0
+; VER-NEXT: [[TMP13:%.*]] = shufflevector <4 x float> [[TMP12]], <4 x float> poison, <4 x i32> zeroinitializer
+; VER-NEXT: [[TMP14:%.*]] = fmul contract <4 x float> [[TMP11]], [[TMP13]]
+; VER-NEXT: [[TMP15:%.*]] = fadd contract <4 x float> [[TMP10]], [[TMP14]]
+; VER-NEXT: [[TMP16:%.*]] = load <4 x float>, ptr [[M12_P]], align 4
+; VER-NEXT: [[TMP17:%.*]] = insertelement <4 x float> poison, float [[A]], i64 0
+; VER-NEXT: [[TMP18:%.*]] = shufflevector <4 x float> [[TMP17]], <4 x float> poison, <4 x i32> zeroinitializer
+; VER-NEXT: [[TMP19:%.*]] = fmul contract <4 x float> [[TMP16]], [[TMP18]]
+; VER-NEXT: [[TMP20:%.*]] = fadd contract <4 x float> [[TMP15]], [[TMP19]]
+; VER-NEXT: store <4 x float> [[TMP20]], ptr [[DST]], align 4
+; VER-NEXT: [[SRC_NEXT:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 16
+; VER-NEXT: [[DST_NEXT:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 16
+; VER-NEXT: [[IV_NEXT:%.*]] = add nuw nsw i64 [[IV]], 1
+; VER-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; VER-NEXT: br label %[[LOOP_RTCONT]]
+; VER: [[LOOP_RTSCALAR]]:
+; VER-NEXT: [[R_SCALAR:%.*]] = load float, ptr [[SRC]], align 4
+; VER-NEXT: [[G_P_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 4
+; VER-NEXT: [[G_SCALAR:%.*]] = load float, ptr [[G_P_SCALAR]], align 4
+; VER-NEXT: [[B_P_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 8
+; VER-NEXT: [[B_SCALAR:%.*]] = load float, ptr [[B_P_SCALAR]], align 4
+; VER-NEXT: [[A_P_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 12
+; VER-NEXT: [[A_SCALAR:%.*]] = load float, ptr [[A_P_SCALAR]], align 4
+; VER-NEXT: [[M0_SCALAR:%.*]] = load float, ptr [[M]], align 4
+; VER-NEXT: [[R0_SCALAR:%.*]] = fmul contract float [[M0_SCALAR]], [[R_SCALAR]]
+; VER-NEXT: [[M4_SCALAR:%.*]] = load float, ptr [[M4_P]], align 4
+; VER-NEXT: [[G0_SCALAR:%.*]] = fmul contract float [[M4_SCALAR]], [[G_SCALAR]]
+; VER-NEXT: [[S0_SCALAR:%.*]] = fadd contract float [[G0_SCALAR]], [[R0_SCALAR]]
+; VER-NEXT: [[M8_SCALAR:%.*]] = load float, ptr [[M8_P]], align 4
+; VER-NEXT: [[B0_SCALAR:%.*]] = fmul contract float [[M8_SCALAR]], [[B_SCALAR]]
+; VER-NEXT: [[T0_SCALAR:%.*]] = fadd contract float [[S0_SCALAR]], [[B0_SCALAR]]
+; VER-NEXT: [[M12_SCALAR:%.*]] = load float, ptr [[M12_P]], align 4
+; VER-NEXT: [[A0_SCALAR:%.*]] = fmul contract float [[M12_SCALAR]], [[A_SCALAR]]
+; VER-NEXT: [[X0_SCALAR:%.*]] = fadd contract float [[T0_SCALAR]], [[A0_SCALAR]]
+; VER-NEXT: store float [[X0_SCALAR]], ptr [[DST]], align 4
+; VER-NEXT: [[M1_SCALAR:%.*]] = load float, ptr [[M1_P]], align 4
+; VER-NEXT: [[R1_SCALAR:%.*]] = fmul contract float [[M1_SCALAR]], [[R_SCALAR]]
+; VER-NEXT: [[M5_SCALAR:%.*]] = load float, ptr [[M5_P]], align 4
+; VER-NEXT: [[G1_SCALAR:%.*]] = fmul contract float [[M5_SCALAR]], [[G_SCALAR]]
+; VER-NEXT: [[S1_SCALAR:%.*]] = fadd contract float [[G1_SCALAR]], [[R1_SCALAR]]
+; VER-NEXT: [[M9_SCALAR:%.*]] = load float, ptr [[M9_P]], align 4
+; VER-NEXT: [[B1_SCALAR:%.*]] = fmul contract float [[M9_SCALAR]], [[B_SCALAR]]
+; VER-NEXT: [[T1_SCALAR:%.*]] = fadd contract float [[S1_SCALAR]], [[B1_SCALAR]]
+; VER-NEXT: [[M13_SCALAR:%.*]] = load float, ptr [[M13_P]], align 4
+; VER-NEXT: [[A1_SCALAR:%.*]] = fmul contract float [[M13_SCALAR]], [[A_SCALAR]]
+; VER-NEXT: [[X1_SCALAR:%.*]] = fadd contract float [[T1_SCALAR]], [[A1_SCALAR]]
+; VER-NEXT: [[DST1_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 4
+; VER-NEXT: store float [[X1_SCALAR]], ptr [[DST1_SCALAR]], align 4
+; VER-NEXT: [[M2_SCALAR:%.*]] = load float, ptr [[M2_P]], align 4
+; VER-NEXT: [[R2_SCALAR:%.*]] = fmul contract float [[M2_SCALAR]], [[R_SCALAR]]
+; VER-NEXT: [[M6_SCALAR:%.*]] = load float, ptr [[M6_P]], align 4
+; VER-NEXT: [[G2_SCALAR:%.*]] = fmul contract float [[M6_SCALAR]], [[G_SCALAR]]
+; VER-NEXT: [[S2_SCALAR:%.*]] = fadd contract float [[G2_SCALAR]], [[R2_SCALAR]]
+; VER-NEXT: [[M10_SCALAR:%.*]] = load float, ptr [[M10_P]], align 4
+; VER-NEXT: [[B2_SCALAR:%.*]] = fmul contract float [[M10_SCALAR]], [[B_SCALAR]]
+; VER-NEXT: [[T2_SCALAR:%.*]] = fadd contract float [[S2_SCALAR]], [[B2_SCALAR]]
+; VER-NEXT: [[M14_SCALAR:%.*]] = load float, ptr [[M14_P]], align 4
+; VER-NEXT: [[A2_SCALAR:%.*]] = fmul contract float [[M14_SCALAR]], [[A_SCALAR]]
+; VER-NEXT: [[X2_SCALAR:%.*]] = fadd contract float [[T2_SCALAR]], [[A2_SCALAR]]
+; VER-NEXT: [[DST2_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 8
+; VER-NEXT: store float [[X2_SCALAR]], ptr [[DST2_SCALAR]], align 4
+; VER-NEXT: [[M3_SCALAR:%.*]] = load float, ptr [[M3_P]], align 4
+; VER-NEXT: [[R3_SCALAR:%.*]] = fmul contract float [[M3_SCALAR]], [[R_SCALAR]]
+; VER-NEXT: [[M7_SCALAR:%.*]] = load float, ptr [[M7_P]], align 4
+; VER-NEXT: [[G3_SCALAR:%.*]] = fmul contract float [[M7_SCALAR]], [[G_SCALAR]]
+; VER-NEXT: [[S3_SCALAR:%.*]] = fadd contract float [[G3_SCALAR]], [[R3_SCALAR]]
+; VER-NEXT: [[M11_SCALAR:%.*]] = load float, ptr [[M11_P]], align 4
+; VER-NEXT: [[B3_SCALAR:%.*]] = fmul contract float [[M11_SCALAR]], [[B_SCALAR]]
+; VER-NEXT: [[T3_SCALAR:%.*]] = fadd contract float [[S3_SCALAR]], [[B3_SCALAR]]
+; VER-NEXT: [[M15_SCALAR:%.*]] = load float, ptr [[M15_P]], align 4
+; VER-NEXT: [[A3_SCALAR:%.*]] = fmul contract float [[M15_SCALAR]], [[A_SCALAR]]
+; VER-NEXT: [[X3_SCALAR:%.*]] = fadd contract float [[T3_SCALAR]], [[A3_SCALAR]]
+; VER-NEXT: [[DST3_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 12
+; VER-NEXT: store float [[X3_SCALAR]], ptr [[DST3_SCALAR]], align 4
+; VER-NEXT: [[SRC_NEXT_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 16
+; VER-NEXT: [[DST_NEXT_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 16
+; VER-NEXT: [[IV_NEXT_SCALAR:%.*]] = add nuw nsw i64 [[IV]], 1
+; VER-NEXT: [[DONE_SCALAR:%.*]] = icmp eq i64 [[IV_NEXT_SCALAR]], [[N]]
+; VER-NEXT: br label %[[LOOP_RTCONT]]
+; VER: [[LOOP_RTCONT]]:
+; VER-NEXT: [[SRC_NEXT_RTMERGE]] = phi ptr [ [[SRC_NEXT]], %[[LOOP_RTVEC]] ], [ [[SRC_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; VER-NEXT: [[DST_NEXT_RTMERGE]] = phi ptr [ [[DST_NEXT]], %[[LOOP_RTVEC]] ], [ [[DST_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; VER-NEXT: [[IV_NEXT_RTMERGE]] = phi i64 [ [[IV_NEXT]], %[[LOOP_RTVEC]] ], [ [[IV_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; VER-NEXT: [[DONE_RTMERGE:%.*]] = phi i1 [ [[DONE]], %[[LOOP_RTVEC]] ], [ [[DONE_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; VER-NEXT: br i1 [[DONE_RTMERGE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+;
+entry:
+ %m1.p = getelementptr inbounds i8, ptr %m, i64 4
+ %m2.p = getelementptr inbounds i8, ptr %m, i64 8
+ %m3.p = getelementptr inbounds i8, ptr %m, i64 12
+ %m4.p = getelementptr inbounds i8, ptr %m, i64 16
+ %m5.p = getelementptr inbounds i8, ptr %m, i64 20
+ %m6.p = getelementptr inbounds i8, ptr %m, i64 24
+ %m7.p = getelementptr inbounds i8, ptr %m, i64 28
+ %m8.p = getelementptr inbounds i8, ptr %m, i64 32
+ %m9.p = getelementptr inbounds i8, ptr %m, i64 36
+ %m10.p = getelementptr inbounds i8, ptr %m, i64 40
+ %m11.p = getelementptr inbounds i8, ptr %m, i64 44
+ %m12.p = getelementptr inbounds i8, ptr %m, i64 48
+ %m13.p = getelementptr inbounds i8, ptr %m, i64 52
+ %m14.p = getelementptr inbounds i8, ptr %m, i64 56
+ %m15.p = getelementptr inbounds i8, ptr %m, i64 60
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %src = phi ptr [ %in, %entry ], [ %src.next, %loop ]
+ %dst = phi ptr [ %out, %entry ], [ %dst.next, %loop ]
+ %r = load float, ptr %src, align 4
+ %g.p = getelementptr inbounds i8, ptr %src, i64 4
+ %g = load float, ptr %g.p, align 4
+ %b.p = getelementptr inbounds i8, ptr %src, i64 8
+ %b = load float, ptr %b.p, align 4
+ %a.p = getelementptr inbounds i8, ptr %src, i64 12
+ %a = load float, ptr %a.p, align 4
+ %m0 = load float, ptr %m, align 4
+ %r0 = fmul contract float %m0, %r
+ %m4 = load float, ptr %m4.p, align 4
+ %g0 = fmul contract float %m4, %g
+ %s0 = fadd contract float %g0, %r0
+ %m8 = load float, ptr %m8.p, align 4
+ %b0 = fmul contract float %m8, %b
+ %t0 = fadd contract float %s0, %b0
+ %m12 = load float, ptr %m12.p, align 4
+ %a0 = fmul contract float %m12, %a
+ %x0 = fadd contract float %t0, %a0
+ store float %x0, ptr %dst, align 4
+ %m1 = load float, ptr %m1.p, align 4
+ %r1 = fmul contract float %m1, %r
+ %m5 = load float, ptr %m5.p, align 4
+ %g1 = fmul contract float %m5, %g
+ %s1 = fadd contract float %g1, %r1
+ %m9 = load float, ptr %m9.p, align 4
+ %b1 = fmul contract float %m9, %b
+ %t1 = fadd contract float %s1, %b1
+ %m13 = load float, ptr %m13.p, align 4
+ %a1 = fmul contract float %m13, %a
+ %x1 = fadd contract float %t1, %a1
+ %dst1 = getelementptr inbounds i8, ptr %dst, i64 4
+ store float %x1, ptr %dst1, align 4
+ %m2 = load float, ptr %m2.p, align 4
+ %r2 = fmul contract float %m2, %r
+ %m6 = load float, ptr %m6.p, align 4
+ %g2 = fmul contract float %m6, %g
+ %s2 = fadd contract float %g2, %r2
+ %m10 = load float, ptr %m10.p, align 4
+ %b2 = fmul contract float %m10, %b
+ %t2 = fadd contract float %s2, %b2
+ %m14 = load float, ptr %m14.p, align 4
+ %a2 = fmul contract float %m14, %a
+ %x2 = fadd contract float %t2, %a2
+ %dst2 = getelementptr inbounds i8, ptr %dst, i64 8
+ store float %x2, ptr %dst2, align 4
+ %m3 = load float, ptr %m3.p, align 4
+ %r3 = fmul contract float %m3, %r
+ %m7 = load float, ptr %m7.p, align 4
+ %g3 = fmul contract float %m7, %g
+ %s3 = fadd contract float %g3, %r3
+ %m11 = load float, ptr %m11.p, align 4
+ %b3 = fmul contract float %m11, %b
+ %t3 = fadd contract float %s3, %b3
+ %m15 = load float, ptr %m15.p, align 4
+ %a3 = fmul contract float %m15, %a
+ %x3 = fadd contract float %t3, %a3
+ %dst3 = getelementptr inbounds i8, ptr %dst, i64 12
+ store float %x3, ptr %dst3, align 4
+ %src.next = getelementptr inbounds i8, ptr %src, i64 16
+ %dst.next = getelementptr inbounds i8, ptr %dst, i64 16
+ %iv.next = add nuw nsw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop, !llvm.loop !2
+
+exit:
+ ret void
+}
+
declare float @llvm.fmuladd.f32(float, float, float)
!0 = distinct !{!0, !1}
!1 = !{!"llvm.loop.isvectorized", i32 1}
+!2 = distinct !{!2, !1}
;.
; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]]}
; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[PROF2]] = !{!"branch_weights", i32 1, i32 1048575}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META1]]}
;.
; VER: [[PROF0]] = !{!"branch_weights", i32 1, i32 1048575}
; VER: [[LOOP1]] = distinct !{[[LOOP1]], [[META2:![0-9]+]]}
; VER: [[META2]] = !{!"llvm.loop.isvectorized", i32 1}
+; VER: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]]}
;.
More information about the llvm-commits
mailing list