[llvm] [LoopVectorize] Improve Vectorization of Low Trip Count Loops (PR #195823)
Sander de Smalen via llvm-commits
llvm-commits at lists.llvm.org
Wed Aug 26 05:01:45 PDT 2026
================
@@ -0,0 +1,160 @@
+; REQUIRES: asserts
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve -S %s | FileCheck %s --check-prefix=IR
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve -debug-only=loop-vectorize -disable-output %s 2>&1 | FileCheck %s --check-prefix=DBG
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @tc3_udiv_i8_reject(ptr noalias %a, ptr noalias %b,
+ ptr noalias %c) #0 {
+; IR-LABEL: define void @tc3_udiv_i8_reject(
+; IR-NOT: vector.body:
+;
+; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_reject'
+; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
+; DBG: LV: Scalar loop costs: 9.
+; DBG: Cost for VF 2: 19
+; DBG: LV: Selecting VF: 1.
+; DBG: LV: Vectorization is possible but not beneficial.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds i8, ptr %a, i64 %iv
+ %pb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %pc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %va = load i8, ptr %pa, align 1
+ %vb = load i8, ptr %pb, align 1
+ %div = udiv i8 %va, %vb
+ store i8 %div, ptr %pc, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @tc3_smin_i8_accept(ptr noalias %a, ptr noalias %b) #0 {
+; IR-LABEL: define void @tc3_smin_i8_accept(
+; IR: vector.body
+
+; DBG-LABEL: LV: Checking a loop in 'tc3_smin_i8_accept'
+; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
+; DBG: LV: Scalar loop costs: 10.
+; DBG: Cost for VF 2: 19
+; DBG: LV: Selecting VF: 2.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %arrayidx = getelementptr inbounds i8, ptr %b, i64 %iv
+ %0 = load i8, ptr %arrayidx, align 1
+ %arrayidx2 = getelementptr inbounds i8, ptr %a, i64 %iv
+ %1 = load i8, ptr %arrayidx2, align 1
+ %min = tail call i8 @llvm.smin.i8(i8 %0, i8 %1)
+ store i8 %min, ptr %arrayidx, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @tc3_udiv_i8_forced(ptr noalias %a, ptr noalias %b,
+ ptr noalias %c) #0 {
+; IR-LABEL: define void @tc3_udiv_i8_forced(
+; IR: vector.body
+
+; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_forced'
+; DBG-NOT: Rejecting VF 2
+; DBG: LV: Selecting VF: 2.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds i8, ptr %a, i64 %iv
+ %pb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %pc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %va = load i8, ptr %pa, align 1
+ %vb = load i8, ptr %pb, align 1
+ %div = udiv i8 %va, %vb
+ store i8 %div, ptr %pc, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !0
+
+exit:
+ ret void
+}
+
+define void @tc5_udiv_i8_reject_ic2(ptr noalias %a, ptr noalias %b, ptr noalias %c) #0{
+; IR-LABEL: define void @tc5_udiv_i8_reject_ic2(
+; IR: vector.body
+
+; DBG-LABEL: LV: Checking a loop in 'tc5_udiv_i8_reject_ic2'
+; DBG-NOT: LV: Selecting VF: 2.
----------------
sdesmalen-arm wrote:
I think this check is not particularly insightful. MaxVF=2, so it does consider a VF of 2, but it doesn't select it because it's considered less profitable:
```
LV: Selecting VF: 1
LV: Vectorization is possible but not beneficial
```
https://github.com/llvm/llvm-project/pull/195823
More information about the llvm-commits
mailing list