[llvm] [LoopVectorize] Improve Vectorization of Low Trip Count Loops (PR #195823)
Jack Styles via llvm-commits
llvm-commits at lists.llvm.org
Tue Aug 18 05:56:20 PDT 2026
================
@@ -0,0 +1,162 @@
+; REQUIRES: asserts
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve -S %s | FileCheck %s --check-prefix=IR
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve -debug-only=loop-vectorize -disable-output %s 2>&1 | FileCheck %s --check-prefix=DBG
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @tc3_udiv_i8_reject(ptr noalias %a, ptr noalias %b,
+ ptr noalias %c) #0 {
+; IR-LABEL: define void @tc3_udiv_i8_reject(
+; IR-NOT: vector.body:
+;
+; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_reject'
+; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
+; DBG: LV: Scalar loop costs: 9.
+; DBG: Cost for VF 2: 19
+; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 28 >= scalar cost 27.
+; DBG: LV: Selecting VF: 1.
+; DBG: LV: Vectorization is possible but not beneficial.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds i8, ptr %a, i64 %iv
+ %pb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %pc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %va = load i8, ptr %pa, align 1
+ %vb = load i8, ptr %pb, align 1
+ %div = udiv i8 %va, %vb
+ store i8 %div, ptr %pc, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @tc3_smin_i8_accept(ptr noalias %a, ptr noalias %b) #0 {
+; IR-LABEL: define void @tc3_smin_i8_accept(
+; IR: vector.body
+
+; DBG-LABEL: LV: Checking a loop in 'tc3_smin_i8_accept'
+; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
+; DBG: LV: Scalar loop costs: 10.
+; DBG: Cost for VF 2: 19
+; DBG: LV: Accepting VF 2 for one-scalar-tail low trip count: vector cost 29 < scalar cost 30.
+; DBG: LV: Selecting VF: 2.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %arrayidx = getelementptr inbounds i8, ptr %b, i64 %iv
+ %0 = load i8, ptr %arrayidx, align 1
+ %arrayidx2 = getelementptr inbounds i8, ptr %a, i64 %iv
+ %1 = load i8, ptr %arrayidx2, align 1
+ %min = tail call i8 @llvm.smin.i8(i8 %0, i8 %1)
+ store i8 %min, ptr %arrayidx, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @tc3_udiv_i8_forced(ptr noalias %a, ptr noalias %b,
+ ptr noalias %c) #0 {
+; IR-LABEL: define void @tc3_udiv_i8_forced(
+; IR: vector.body
+
+; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_forced'
+; DBG-NOT: Rejecting VF 2
+; DBG: LV: Selecting VF: 2.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds i8, ptr %a, i64 %iv
+ %pb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %pc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %va = load i8, ptr %pa, align 1
+ %vb = load i8, ptr %pb, align 1
+ %div = udiv i8 %va, %vb
+ store i8 %div, ptr %pc, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !0
+
+exit:
+ ret void
+}
+
+define void @tc5_udiv_i8_reject_ic2(ptr noalias %a, ptr noalias %b, ptr noalias %c) #0{
+; IR-LABEL: define void @tc5_udiv_i8_reject_ic2(
+; IR: vector.body
+
+; DBG-LABEL: LV: Checking a loop in 'tc5_udiv_i8_reject_ic2'
+; DBG-NOT: LV: Selecting VF: 2.
+; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 47 >= scalar cost 45.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds i8, ptr %a, i64 %iv
+ %pb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %pc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %va = load i8, ptr %pa, align 1
+ %vb = load i8, ptr %pb, align 1
+ %div = udiv i8 %va, %vb
+ store i8 %div, ptr %pc, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !2
+
+exit:
+ ret void
+}
+
+define void @tc5_sin_f32_select_smaller_vf(ptr noalias %a,
+ ptr noalias %c) #0 {
+; IR-LABEL: define void @tc5_sin_f32_select_smaller_vf(
+; IR: vector.body
+
+; DBG-LABEL: LV: Checking a loop in 'tc5_sin_f32_select_smaller_vf'
+; DBG: Picking MaxVF=4 with 1 scalar iteration remaining.
+; DBG: LV: Accepting VF 4 for one-scalar-tail low trip count: vector cost 72 < scalar cost 80.
+; DBG-NOT: Selecting VF: 4
+; DBG: LV: Selecting VF: 2.
----------------
Stylie777 wrote:
I disagree. The objective of this test is to prove the Loop Vectorizer can still pick the best VF when the MaxVF is not the best option, as is the case in this example. I have added a comment to the test, and expanded the testing CHECK lines to make it clearer.
https://github.com/llvm/llvm-project/pull/195823
More information about the llvm-commits
mailing list