[llvm] [LV] Add tests for zero branch weights and execution frequencies (NFC). (PR #226553)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 25 10:55:45 PDT 2026
https://github.com/fhahn created https://github.com/llvm/llvm-project/pull/226553
Add missing test coverage for VPlan-based execution frequency computation:
* blocks entered via an edge with zero branch weight,
* blocks whose sibling edge has zero weight,
* nested branches whose skipping edges have zero weight,
* a block executing on all paths through the loop after a switch whose probabilities do not add up to 1,
* frequencies below the smallest representable non-zero one
* all-zero weights making a join's frequency unknown
* estimated frequencies propagating through a branch with weights.
Also updates llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
to check frequencies directly after the frequencies have been recorded.
>From 7b219ef7c6bbd072ccbe096e44a2b2636587d7fd Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 24 Sep 2026 18:37:30 +0100
Subject: [PATCH] [LV] Add tests for zero branch weights and execution
frequencies (NFC).
Add tests for blocks entered via an edge with zero branch weight, blocks
whose sibling edge has zero weight, nested branches whose skipping edges
have zero weight, a block executing on all paths through the loop after a
switch whose probabilities do not add up to 1, frequencies below the
smallest representable non-zero one, all-zero weights making a join's
frequency unknown and estimated frequencies propagating through a branch
with weights.
---
.../VPlan/execution-frequencies-match-bfi.ll | 1013 ++++++++++++-----
.../X86/predicated-instruction-cost.ll | 54 +
.../replicate-region-branch-weights.ll | 92 ++
3 files changed, 905 insertions(+), 254 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll b/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
index 96b539f4d7f5e..a99a536578d6f 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
@@ -1,11 +1,11 @@
; RUN: opt -passes='print<block-freq>' -disable-output %s 2>&1 \
; RUN: | FileCheck --check-prefix=BFI %s
; RUN: opt -passes=loop-vectorize -force-vector-width=2 -force-vector-interleave=1 \
-; RUN: -vplan-print-after=introduceMasksAndLinearize -disable-output %s 2>&1 \
+; RUN: -vplan-print-after=recordExecutionFrequencies -disable-output %s 2>&1 \
; RUN: | FileCheck --check-prefix=VPLAN %s
-; Check that the execution frequencies VPlan records on the masked recipes of
-; a block match the block frequencies BlockFrequencyInfo computes for the
+; Check that the execution frequencies VPlan records on the recipes of a block
+; match the block frequencies BlockFrequencyInfo computes for the
; corresponding block of the original scalar loop.
define void @single_pred(ptr noalias %a, ptr noalias %b, ptr noalias %idx) {
@@ -23,23 +23,26 @@ define void @single_pred(ptr noalias %a, ptr noalias %b, ptr noalias %idx) {
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'single_pred'
-; VPLAN: vector.body:
-; VPLAN-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<%{{.+}}>
-; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>
-; VPLAN-NEXT: EMIT-SCALAR ir<%i> = load ir<%gep.idx>
-; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.b>{{$}}
-; VPLAN-NEXT: EMIT ir<%c.0> = icmp sgt ir<%i>, ir<0>
-; VPLAN-NEXT: Successor(s): if.then
-; VPLAN-EMPTY:
-; VPLAN-NEXT: if.then:
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.a>, ir<%c.0> (!vplan.execution.frequency 2305843009213693952 (25%))
-; VPLAN-NEXT: Successor(s): latch
-; VPLAN-EMPTY:
-; VPLAN-NEXT: latch:
-; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
-; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%i> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.b>{{$}}
+; VPLAN-NEXT: EMIT ir<%c.0> = icmp sgt ir<%i>, ir<0>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.0> (!prof {1, 3}){{$}}
+; VPLAN-NEXT: Successor(s): if.then, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 2305843009213693952 (25%))
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.a> (!vplan.execution.frequency 2305843009213693952 (25%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
;
entry:
br label %loop
@@ -67,6 +70,136 @@ exit:
ret void
}
+define void @single_pred_zero_weight(ptr noalias %a, ptr noalias %b, ptr noalias %idx) {
+; %if.then is only entered via an edge with zero branch weight. BFI treats the
+; edge as cold, not as never taken.
+;
+; %loop 1 = 1
+; %if.then 2^-31 ~ 4.66e-10
+; %latch 1 = 1
+;
+; TODO: VPlan currently records a frequency of 0 for %if.then.
+;
+; BFI-LABEL: block-frequency-info: single_pred_zero_weight
+; BFI-NEXT: - entry: float = 1.0,
+; BFI-NEXT: - loop: float = 1000.0,
+; BFI-NEXT: - if.then: float = 0.00000046566,
+; BFI-NEXT: - latch: float = 1000.0,
+; BFI-NEXT: - exit: float = 1.0,
+;
+; VPLAN-LABEL: VPlan for loop in 'single_pred_zero_weight'
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%i> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.b>{{$}}
+; VPLAN-NEXT: EMIT ir<%c.0> = icmp sgt ir<%i>, ir<0>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.0> (!prof {0, 1000}){{$}}
+; VPLAN-NEXT: Successor(s): if.then, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.a>{{$}}
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %gep.idx = getelementptr inbounds i32, ptr %idx, i64 %iv
+ %i = load i32, ptr %gep.idx, align 4
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ store i32 %i, ptr %gep.b, align 4
+ %c.0 = icmp sgt i32 %i, 0
+ br i1 %c.0, label %if.then, label %latch, !prof !13
+
+if.then:
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ store i32 %i, ptr %gep.a, align 4
+ br label %latch
+
+latch:
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop, !prof !3
+
+exit:
+ ret void
+}
+
+define void @single_pred_zero_weight_sibling(ptr noalias %a, ptr noalias %b, ptr noalias %idx) {
+; The edge skipping %if.then has zero branch weight. BFI treats it as cold, not
+; as never taken.
+;
+; %loop 1 = 1
+; %if.then 1 - 2^-31 ~ 1 - 4.66e-10
+; %latch 1 = 1
+;
+; TODO: VPlan currently records %if.then as always executing.
+;
+; BFI-LABEL: block-frequency-info: single_pred_zero_weight_sibling
+; BFI-NEXT: - entry: float = 1.0,
+; BFI-NEXT: - loop: float = 1000.0, int = 18014398509481984
+; BFI-NEXT: - if.then: float = 1000.0, int = 18014398501093376
+; BFI-NEXT: - latch: float = 1000.0, int = 18014398509481984
+; BFI-NEXT: - exit: float = 1.0,
+;
+; VPLAN-LABEL: VPlan for loop in 'single_pred_zero_weight_sibling'
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%i> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.b>{{$}}
+; VPLAN-NEXT: EMIT ir<%c.0> = icmp sgt ir<%i>, ir<0>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.0> (!prof {1000, 0}){{$}}
+; VPLAN-NEXT: Successor(s): if.then, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.a>{{$}}
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %gep.idx = getelementptr inbounds i32, ptr %idx, i64 %iv
+ %i = load i32, ptr %gep.idx, align 4
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ store i32 %i, ptr %gep.b, align 4
+ %c.0 = icmp sgt i32 %i, 0
+ br i1 %c.0, label %if.then, label %latch, !prof !14
+
+if.then:
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ store i32 %i, ptr %gep.a, align 4
+ br label %latch
+
+latch:
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop, !prof !3
+
+exit:
+ ret void
+}
+
define void @two_preds(ptr noalias %a, ptr noalias %b, ptr noalias %c, ptr noalias %idx) {
; Execution frequency of each block of the loop. %merge is reached from both
; %then (1/4) and %else (3/4 * 1/3 = 1/4).
@@ -87,35 +220,36 @@ define void @two_preds(ptr noalias %a, ptr noalias %b, ptr noalias %c, ptr noali
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'two_preds'
-; VPLAN: vector.body:
-; VPLAN-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<%{{.+}}>
-; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>
-; VPLAN-NEXT: EMIT-SCALAR ir<%i> = load ir<%gep.idx>
-; VPLAN-NEXT: EMIT ir<%c.0> = icmp sgt ir<%i>, ir<0>
-; VPLAN-NEXT: Successor(s): else
-; VPLAN-EMPTY:
-; VPLAN-NEXT: else:
-; VPLAN-NEXT: EMIT vp<[[NOT_C0:%.+]]> = not ir<%c.0>
-; VPLAN-NEXT: EMIT ir<%gep.c> = getelementptr inbounds ir<%c>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.c>, vp<[[NOT_C0]]> (!vplan.execution.frequency 6917529027641081856 (75%))
-; VPLAN-NEXT: EMIT ir<%c.1> = icmp slt ir<%i>, ir<-100>, vp<[[NOT_C0]]> (!vplan.execution.frequency 6917529027641081856 (75%))
-; VPLAN-NEXT: Successor(s): then
-; VPLAN-EMPTY:
-; VPLAN-NEXT: then:
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.a>, ir<%c.0> (!vplan.execution.frequency 2305843009213693952 (25%))
-; VPLAN-NEXT: Successor(s): merge
-; VPLAN-EMPTY:
-; VPLAN-NEXT: merge:
-; VPLAN-NEXT: EMIT vp<[[AND:%.+]]> = logical-and vp<[[NOT_C0]]>, ir<%c.1>
-; VPLAN-NEXT: EMIT vp<[[MASK:%.+]]> = or vp<[[AND]]>, ir<%c.0>
-; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.b>, vp<[[MASK]]> (!vplan.execution.frequency 4611686019501129728 (50%))
-; VPLAN-NEXT: Successor(s): latch
-; VPLAN-EMPTY:
-; VPLAN-NEXT: latch:
-; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
-; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%i> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT ir<%c.0> = icmp sgt ir<%i>, ir<0>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.0> (!prof {1, 3}){{$}}
+; VPLAN-NEXT: Successor(s): then, else
+; VPLAN-EMPTY:
+; VPLAN-NEXT: else:
+; VPLAN-NEXT: EMIT ir<%gep.c> = getelementptr inbounds ir<%c>, ir<%iv> (!vplan.execution.frequency 6917529027641081856 (75%))
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.c> (!vplan.execution.frequency 6917529027641081856 (75%))
+; VPLAN-NEXT: EMIT ir<%c.1> = icmp slt ir<%i>, ir<-100> (!vplan.execution.frequency 6917529027641081856 (75%))
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.1> (!prof {1, 2}, !vplan.execution.frequency 6917529027641081856 (75%))
+; VPLAN-NEXT: Successor(s): merge, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 2305843009213693952 (25%))
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.a> (!vplan.execution.frequency 2305843009213693952 (25%))
+; VPLAN-NEXT: Successor(s): merge
+; VPLAN-EMPTY:
+; VPLAN-NEXT: merge:
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv> (!vplan.execution.frequency 4611686019501129728 (50%))
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.b> (!vplan.execution.frequency 4611686019501129728 (50%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
;
entry:
br label %loop
@@ -169,28 +303,31 @@ define void @nested_ifs(ptr noalias %a, ptr noalias %b, ptr noalias %idx) {
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'nested_ifs'
-; VPLAN: vector.body:
-; VPLAN-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<%{{.+}}>
-; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>
-; VPLAN-NEXT: EMIT-SCALAR ir<%i> = load ir<%gep.idx>
-; VPLAN-NEXT: EMIT ir<%c.0> = icmp sgt ir<%i>, ir<0>
-; VPLAN-NEXT: Successor(s): if.0
-; VPLAN-EMPTY:
-; VPLAN-NEXT: if.0:
-; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.b>, ir<%c.0> (!vplan.execution.frequency 2305843009213693952 (25%))
-; VPLAN-NEXT: EMIT ir<%c.1> = icmp slt ir<%i>, ir<100>, ir<%c.0> (!vplan.execution.frequency 2305843009213693952 (25%))
-; VPLAN-NEXT: Successor(s): if.1
-; VPLAN-EMPTY:
-; VPLAN-NEXT: if.1:
-; VPLAN-NEXT: EMIT vp<[[MASK:%.+]]> = logical-and ir<%c.0>, ir<%c.1>
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.a>, vp<[[MASK]]> (!vplan.execution.frequency 1152921504606846976 (12.5%))
-; VPLAN-NEXT: Successor(s): latch
-; VPLAN-EMPTY:
-; VPLAN-NEXT: latch:
-; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
-; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%i> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT ir<%c.0> = icmp sgt ir<%i>, ir<0>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.0> (!prof {1, 3}){{$}}
+; VPLAN-NEXT: Successor(s): if.0, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.0:
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv> (!vplan.execution.frequency 2305843009213693952 (25%))
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.b> (!vplan.execution.frequency 2305843009213693952 (25%))
+; VPLAN-NEXT: EMIT ir<%c.1> = icmp slt ir<%i>, ir<100> (!vplan.execution.frequency 2305843009213693952 (25%))
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.1> (!prof {1, 1}, !vplan.execution.frequency 2305843009213693952 (25%))
+; VPLAN-NEXT: Successor(s): if.1, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.1:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 1152921504606846976 (12.5%))
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.a> (!vplan.execution.frequency 1152921504606846976 (12.5%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
;
entry:
br label %loop
@@ -243,18 +380,19 @@ define void @second_branch_without_weights(ptr noalias %a, ptr noalias %b, ptr n
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'second_branch_without_weights'
-; VPLAN: if.then.1:
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.a>, ir<%c.0> (!vplan.execution.frequency 2305843009213693952 (25%))
-; VPLAN-NEXT: Successor(s): merge
+; VPLAN: if.then.1:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 2305843009213693952 (25%))
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.a> (!vplan.execution.frequency 2305843009213693952 (25%))
+; VPLAN-NEXT: Successor(s): merge
; VPLAN-EMPTY:
-; VPLAN-NEXT: merge:
-; VPLAN-NEXT: Successor(s): if.then.2
+; VPLAN-NEXT: merge:
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.0> (!vplan.prof.estimated estimated {1342177280, 805306368}){{$}}
+; VPLAN-NEXT: Successor(s): if.then.2, latch
; VPLAN-EMPTY:
-; VPLAN-NEXT: if.then.2:
-; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.b>, ir<%c.0> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; VPLAN-NEXT: Successor(s): latch
+; VPLAN-NEXT: if.then.2:
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; VPLAN-NEXT: EMIT store ir<%i>, ir<%gep.b> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; VPLAN-NEXT: Successor(s): latch
;
entry:
br label %loop
@@ -308,36 +446,33 @@ define void @switch_common_dest(ptr noalias %a, ptr noalias %b, ptr noalias %c,
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'switch_common_dest'
-; VPLAN: vector.body:
-; VPLAN-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<%{{.+}}>
-; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>
-; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>
-; VPLAN-NEXT: Successor(s): other
-; VPLAN-EMPTY:
-; VPLAN-NEXT: other:
-; VPLAN-NEXT: EMIT vp<[[C0:%.+]]> = icmp eq ir<%l>, ir<0>
-; VPLAN-NEXT: EMIT vp<[[C1:%.+]]> = icmp eq ir<%l>, ir<1>
-; VPLAN-NEXT: EMIT vp<[[C2:%.+]]> = icmp eq ir<%l>, ir<2>
-; VPLAN-NEXT: EMIT vp<[[C0_OR_C1:%.+]]> = or vp<[[C0]]>, vp<[[C1]]>
-; VPLAN-NEXT: EMIT vp<[[ANY:%.+]]> = or vp<[[C0_OR_C1]]>, vp<[[C2]]>
-; VPLAN-NEXT: EMIT vp<[[DEFAULT:%.+]]> = not vp<[[ANY]]>
-; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<2>, ir<%gep.b>, vp<[[C2]]> (!vplan.execution.frequency 1152921504606846976 (12.5%))
-; VPLAN-NEXT: Successor(s): if.then
-; VPLAN-EMPTY:
-; VPLAN-NEXT: if.then:
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a>, vp<[[C0_OR_C1]]> (!vplan.execution.frequency 3458764513820540928 (37.5%))
-; VPLAN-NEXT: Successor(s): default
-; VPLAN-EMPTY:
-; VPLAN-NEXT: default:
-; VPLAN-NEXT: EMIT ir<%gep.c> = getelementptr inbounds ir<%c>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.c>, vp<[[DEFAULT]]> (!vplan.execution.frequency 4611686018427387904 (50%))
-; VPLAN-NEXT: Successor(s): latch
-; VPLAN-EMPTY:
-; VPLAN-NEXT: latch:
-; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
-; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT switch ir<%l>, ir<0>, ir<1>, ir<2> (!prof {500, 125, 250, 125}){{$}}
+; VPLAN-NEXT: Successor(s): default, if.then, if.then, other
+; VPLAN-EMPTY:
+; VPLAN-NEXT: other:
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv> (!vplan.execution.frequency 1152921504606846976 (12.5%))
+; VPLAN-NEXT: EMIT store ir<2>, ir<%gep.b> (!vplan.execution.frequency 1152921504606846976 (12.5%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 3458764513820540928 (37.5%))
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 3458764513820540928 (37.5%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: default:
+; VPLAN-NEXT: EMIT ir<%gep.c> = getelementptr inbounds ir<%c>, ir<%iv> (!vplan.execution.frequency 4611686018427387904 (50%))
+; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.c> (!vplan.execution.frequency 4611686018427387904 (50%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
;
entry:
br label %loop
@@ -395,29 +530,28 @@ define void @switch_common_dest_weight_sum_not_a_power_of_two(ptr noalias %a, pt
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'switch_common_dest_weight_sum_not_a_power_of_two'
-; VPLAN: vector.body:
-; VPLAN-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<%{{.+}}>
-; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>
-; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>
-; VPLAN-NEXT: Successor(s): if.then
-; VPLAN-EMPTY:
-; VPLAN-NEXT: if.then:
-; VPLAN-NEXT: EMIT vp<[[C0:%.+]]> = icmp eq ir<%l>, ir<0>
-; VPLAN-NEXT: EMIT vp<[[C1:%.+]]> = icmp eq ir<%l>, ir<1>
-; VPLAN-NEXT: EMIT vp<[[C0_OR_C1:%.+]]> = or vp<[[C0]]>, vp<[[C1]]>
-; VPLAN-NEXT: EMIT vp<[[DEFAULT:%.+]]> = not vp<[[C0_OR_C1]]>
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a>, vp<[[C0_OR_C1]]> (!vplan.execution.frequency 6148914689804861440 (66.67%))
-; VPLAN-NEXT: Successor(s): default
-; VPLAN-EMPTY:
-; VPLAN-NEXT: default:
-; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.b>, vp<[[DEFAULT]]> (!vplan.execution.frequency 3074457347049914368 (33.33%))
-; VPLAN-NEXT: Successor(s): latch
-; VPLAN-EMPTY:
-; VPLAN-NEXT: latch:
-; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
-; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT switch ir<%l>, ir<0>, ir<1> (!prof {1, 1, 1}){{$}}
+; VPLAN-NEXT: Successor(s): default, if.then, if.then
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 6148914689804861440 (66.67%))
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 6148914689804861440 (66.67%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: default:
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv> (!vplan.execution.frequency 3074457347049914368 (33.33%))
+; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.b> (!vplan.execution.frequency 3074457347049914368 (33.33%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
;
entry:
br label %loop
@@ -468,29 +602,28 @@ define void @switch_common_dest_almost_always_taken(ptr noalias %a, ptr noalias
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'switch_common_dest_almost_always_taken'
-; VPLAN: vector.body:
-; VPLAN-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<%{{.+}}>
-; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>
-; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>
-; VPLAN-NEXT: Successor(s): if.then
-; VPLAN-EMPTY:
-; VPLAN-NEXT: if.then:
-; VPLAN-NEXT: EMIT vp<[[C0:%.+]]> = icmp eq ir<%l>, ir<0>
-; VPLAN-NEXT: EMIT vp<[[C1:%.+]]> = icmp eq ir<%l>, ir<1>
-; VPLAN-NEXT: EMIT vp<[[C0_OR_C1:%.+]]> = or vp<[[C0]]>, vp<[[C1]]>
-; VPLAN-NEXT: EMIT vp<[[DEFAULT:%.+]]> = not vp<[[C0_OR_C1]]>
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a>, vp<[[C0_OR_C1]]> (!vplan.execution.frequency 9223372032559808512 (100%))
-; VPLAN-NEXT: Successor(s): default
-; VPLAN-EMPTY:
-; VPLAN-NEXT: default:
-; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.b>, vp<[[DEFAULT]]> (!vplan.execution.frequency 4294967296 (4.657E-8%))
-; VPLAN-NEXT: Successor(s): latch
-; VPLAN-EMPTY:
-; VPLAN-NEXT: latch:
-; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
-; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT switch ir<%l>, ir<0>, ir<1> (!prof {1, 2147483647, 2147483647}){{$}}
+; VPLAN-NEXT: Successor(s): default, if.then, if.then
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: default:
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.b> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
;
entry:
br label %loop
@@ -543,36 +676,30 @@ define void @switch_common_dest_almost_never_taken(ptr noalias %a, ptr noalias %
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'switch_common_dest_almost_never_taken'
-; VPLAN: vector.body:
-; VPLAN-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<%{{.+}}>
-; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>
-; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>
-; VPLAN-NEXT: EMIT ir<%c> = icmp sgt ir<%l>, ir<0>
-; VPLAN-NEXT: Successor(s): mid
-; VPLAN-EMPTY:
-; VPLAN-NEXT: mid:
-; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.b>, ir<%c> (!vplan.execution.frequency 4294967296 (4.657E-8%))
-; VPLAN-NEXT: Successor(s): if.then
-; VPLAN-EMPTY:
-; VPLAN-NEXT: if.then:
-; VPLAN-NEXT: EMIT vp<[[C1:%.+]]> = icmp eq ir<%l>, ir<1>
-; VPLAN-NEXT: EMIT vp<[[C2:%.+]]> = icmp eq ir<%l>, ir<2>
-; VPLAN-NEXT: EMIT vp<[[C3:%.+]]> = icmp eq ir<%l>, ir<3>
-; VPLAN-NEXT: EMIT vp<[[C4:%.+]]> = icmp eq ir<%l>, ir<4>
-; VPLAN-NEXT: EMIT vp<[[OR_0:%.+]]> = or vp<[[C1]]>, vp<[[C2]]>
-; VPLAN-NEXT: EMIT vp<[[OR_1:%.+]]> = or vp<[[OR_0]]>, vp<[[C3]]>
-; VPLAN-NEXT: EMIT vp<[[ANY:%.+]]> = or vp<[[OR_1]]>, vp<[[C4]]>
-; VPLAN-NEXT: EMIT vp<[[MASK:%.+]]> = logical-and ir<%c>, vp<[[ANY]]>
-; VPLAN-NEXT: EMIT vp<[[NOT_MASK:%.+]]> = not vp<[[MASK]]>
-; VPLAN-NEXT: EMIT vp<[[DEFAULT:%.+]]> = logical-and ir<%c>, vp<[[NOT_MASK]]>
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a>, vp<[[MASK]]> (!vplan.execution.frequency 3435973836 (3.725E-8%))
-; VPLAN-NEXT: Successor(s): latch
-; VPLAN-EMPTY:
-; VPLAN-NEXT: latch:
-; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
-; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT ir<%c> = icmp sgt ir<%l>, ir<0>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c> (!prof {1, 2147483647}){{$}}
+; VPLAN-NEXT: Successor(s): mid, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: mid:
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.b> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: EMIT switch ir<%l>, ir<1>, ir<2>, ir<3>, ir<4> (!prof {1, 1, 1, 1, 1}, !vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: Successor(s): latch, if.then, if.then, if.then, if.then
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 3435973836 (3.725E-8%))
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 3435973836 (3.725E-8%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
;
entry:
br label %loop
@@ -627,29 +754,30 @@ define void @switch_common_dest_many_edges_almost_never_taken(ptr noalias %a, pt
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'switch_common_dest_many_edges_almost_never_taken'
-; VPLAN: vector.body:
-; VPLAN-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<%{{.+}}>
-; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>
-; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>
-; VPLAN-NEXT: EMIT ir<%c> = icmp sgt ir<%l>, ir<0>
-; VPLAN-NEXT: Successor(s): mid
-; VPLAN-EMPTY:
-; VPLAN-NEXT: mid:
-; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.b>, ir<%c> (!vplan.execution.frequency 4294967296 (4.657E-8%))
-; VPLAN-NEXT: Successor(s): if.then
-; VPLAN-EMPTY:
-; VPLAN-NEXT: if.then:
-; VPLAN: EMIT vp<[[MASK:%.+]]> = logical-and ir<%c>, vp<%{{.+}}>
-; VPLAN-NEXT: EMIT vp<[[NOT_MASK:%.+]]> = not vp<[[MASK]]>
-; VPLAN-NEXT: EMIT vp<[[DEFAULT:%.+]]> = logical-and ir<%c>, vp<[[NOT_MASK]]>
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a>, vp<[[MASK]]> (!vplan.execution.frequency 3817748708 (4.139E-8%))
-; VPLAN-NEXT: Successor(s): latch
-; VPLAN-EMPTY:
-; VPLAN-NEXT: latch:
-; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
-; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT ir<%c> = icmp sgt ir<%l>, ir<0>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c> (!prof {1, 2147483647}){{$}}
+; VPLAN-NEXT: Successor(s): mid, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: mid:
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.b> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: EMIT switch ir<%l>, ir<1>, ir<2>, ir<3>, ir<4>, ir<5>, ir<6>, ir<7>, ir<8> (!prof {1, 1, 1, 1, 1, 1, 1, 1, 1}, !vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: Successor(s): latch, if.then, if.then, if.then, if.then, if.then, if.then, if.then, if.then
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 3817748708 (4.139E-8%))
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 3817748708 (4.139E-8%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
;
entry:
br label %loop
@@ -707,26 +835,17 @@ define void @switch_common_dest_weight_sum_exceeds_32_bits(ptr noalias %a, ptr n
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'switch_common_dest_weight_sum_exceeds_32_bits'
-; VPLAN: vector.body:
-; VPLAN-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<%{{.+}}>
-; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>
-; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>
-; VPLAN-NEXT: Successor(s): if.then
-; VPLAN-EMPTY:
-; VPLAN-NEXT: if.then:
-; VPLAN-NEXT: EMIT vp<[[C0:%.+]]> = icmp eq ir<%l>, ir<0>
-; VPLAN-NEXT: EMIT vp<[[C1:%.+]]> = icmp eq ir<%l>, ir<1>
-; VPLAN-NEXT: EMIT vp<[[C2:%.+]]> = icmp eq ir<%l>, ir<2>
-; VPLAN-NEXT: EMIT vp<[[C3:%.+]]> = icmp eq ir<%l>, ir<3>
-; VPLAN-NEXT: EMIT vp<[[C4:%.+]]> = icmp eq ir<%l>, ir<4>
-; VPLAN-NEXT: EMIT vp<[[OR0:%.+]]> = or vp<[[C0]]>, vp<[[C1]]>
-; VPLAN-NEXT: EMIT vp<[[OR1:%.+]]> = or vp<[[OR0]]>, vp<[[C2]]>
-; VPLAN-NEXT: EMIT vp<[[OR2:%.+]]> = or vp<[[OR1]]>, vp<[[C3]]>
-; VPLAN-NEXT: EMIT vp<[[MASK:%.+]]> = or vp<[[OR2]]>, vp<[[C4]]>
-; VPLAN-NEXT: EMIT vp<{{.+}}> = not vp<[[MASK]]>
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a>, vp<[[MASK]]> (!vplan.execution.frequency 9223372023969873920 (100%))
-; VPLAN-NEXT: Successor(s): latch
+; VPLAN: loop:
+; VPLAN-NEXT: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, latch ]{{$}}
+; VPLAN-NEXT: EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>{{$}}
+; VPLAN-NEXT: EMIT-SCALAR ir<%l> = load ir<%gep.idx>{{$}}
+; VPLAN-NEXT: EMIT switch ir<%l>, ir<0>, ir<1>, ir<2>, ir<3>, ir<4> (!prof {36, 4294967295, 4294967295, 4294967295, 4294967295, 4294967295}){{$}}
+; VPLAN-NEXT: Successor(s): latch, if.then, if.then, if.then, if.then, if.then
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 9223372023969873920 (100%))
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 9223372023969873920 (100%))
+; VPLAN-NEXT: Successor(s): latch
;
entry:
br label %loop
@@ -775,9 +894,10 @@ define void @switch_many_edges_to_latch(ptr noalias %a, ptr noalias %idx) {
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'switch_many_edges_to_latch'
-; VPLAN: if.then:
-; VPLAN: EMIT store ir<1>, ir<%gep.a>, vp<{{.+}}> (!vplan.execution.frequency 7535434672457646080 (81.7%))
-; VPLAN-NEXT: Successor(s): latch
+; VPLAN: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 7535434672457646080 (81.7%))
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 7535434672457646080 (81.7%))
+; VPLAN-NEXT: Successor(s): latch
;
entry:
br label %loop
@@ -855,16 +975,15 @@ define void @switch_weights_clamped_at_both_ends(ptr noalias %a, ptr noalias %b,
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'switch_weights_clamped_at_both_ends'
-; VPLAN: if.then:
-; VPLAN: EMIT vp<[[DEFAULT:%.+]]> = not vp<[[MASK:%.+]]>
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a>, vp<[[MASK]]> (!vplan.execution.frequency 9223372032559808512 (100%))
-; VPLAN-NEXT: Successor(s): default
+; VPLAN: if.then:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT: Successor(s): latch
; VPLAN-EMPTY:
-; VPLAN-NEXT: default:
-; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.b>, vp<[[DEFAULT]]> (!vplan.execution.frequency 4294967296 (4.657E-8%))
-; VPLAN-NEXT: Successor(s): latch
+; VPLAN-NEXT: default:
+; VPLAN-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.b> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: Successor(s): latch
;
entry:
br label %loop
@@ -917,20 +1036,20 @@ define void @nested_blocks_almost_never_entered(ptr noalias %a, ptr noalias %idx
; BFI-NEXT: - exit: float = 1.0,
;
; VPLAN-LABEL: VPlan for loop in 'nested_blocks_almost_never_entered'
-; VPLAN: if.then.1:
-; VPLAN-NEXT: EMIT ir<%c.1> = icmp sgt ir<%l>, ir<1>, ir<%c.0> (!vplan.execution.frequency 4294967296 (4.657E-8%))
-; VPLAN-NEXT: Successor(s): if.then.2
+; VPLAN: if.then.1:
+; VPLAN-NEXT: EMIT ir<%c.1> = icmp sgt ir<%l>, ir<1> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.1> (!prof {1, 2147483647}, !vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT: Successor(s): if.then.2, latch
; VPLAN-EMPTY:
-; VPLAN-NEXT: if.then.2:
-; VPLAN-NEXT: EMIT vp<[[AND:%.+]]> = logical-and ir<%c.0>, ir<%c.1>
-; VPLAN-NEXT: EMIT ir<%c.2> = icmp sgt ir<%l>, ir<2>, vp<[[AND]]> (!vplan.execution.frequency 2 (2.168E-17%))
-; VPLAN-NEXT: Successor(s): if.then.3
+; VPLAN-NEXT: if.then.2:
+; VPLAN-NEXT: EMIT ir<%c.2> = icmp sgt ir<%l>, ir<2> (!vplan.execution.frequency 2 (2.168E-17%))
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.2> (!prof {1, 2147483647}, !vplan.execution.frequency 2 (2.168E-17%))
+; VPLAN-NEXT: Successor(s): if.then.3, latch
; VPLAN-EMPTY:
-; VPLAN-NEXT: if.then.3:
-; VPLAN-NEXT: EMIT vp<[[MASK:%.+]]> = logical-and vp<[[AND]]>, ir<%c.2>
-; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a>, vp<[[MASK]]> (!vplan.execution.frequency 1 (1.084E-17%))
-; VPLAN-NEXT: Successor(s): latch
+; VPLAN-NEXT: if.then.3:
+; VPLAN-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 1 (1.084E-17%))
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 1 (1.084E-17%))
+; VPLAN-NEXT: Successor(s): latch
;
entry:
br label %loop
@@ -964,6 +1083,388 @@ exit:
ret void
}
+define void @switch_join_always(ptr noalias %a, ptr noalias %idx) {
+; %join executes on every path through the loop, so it always executes, even
+; though none of the switch's probabilities (5/7, 1/7, 1/7) is exact.
+;
+; %loop 1 = 1
+; %case.1 1/7 ~ 0.143
+; %case.2 1/7 ~ 0.143
+; %join 1 = 1
+; %latch 1 = 1
+;
+; TODO: %join and %latch are currently recorded as executing slightly less
+; often, as the rounded probabilities do not add up to 1.
+;
+; BFI-LABEL: block-frequency-info: switch_join_always
+; BFI-NEXT: - entry: float = 1.0,
+; BFI-NEXT: - loop: float = 1000.0,
+; BFI-NEXT: - case.1: float = 142.86,
+; BFI-NEXT: - case.2: float = 142.86,
+; BFI-NEXT: - join: float = 1000.0,
+; BFI-NEXT: - latch: float = 1000.0,
+; BFI-NEXT: - exit: float = 1.0,
+;
+; VPLAN-LABEL: VPlan for loop in 'switch_join_always'
+; VPLAN: case.2:
+; VPLAN-NEXT: EMIT store ir<2>, ir<%gep.a> (!vplan.execution.frequency 1317624575466405888 (14.29%))
+; VPLAN-NEXT: Successor(s): join
+; VPLAN-EMPTY:
+; VPLAN-NEXT: case.1:
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 1317624575466405888 (14.29%))
+; VPLAN-NEXT: Successor(s): join
+; VPLAN-EMPTY:
+; VPLAN-NEXT: join:
+; VPLAN-NEXT: EMIT ir<%add> = add ir<%i>, ir<10> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT: EMIT store ir<%add>, ir<%gep.a> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}, !vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT: Successor(s): middle.block, loop
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %gep.idx = getelementptr inbounds i32, ptr %idx, i64 %iv
+ %i = load i32, ptr %gep.idx, align 4
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ switch i32 %i, label %join [
+ i32 1, label %case.1
+ i32 2, label %case.2
+ ], !prof !15
+
+case.1:
+ store i32 1, ptr %gep.a, align 4
+ br label %join
+
+case.2:
+ store i32 2, ptr %gep.a, align 4
+ br label %join
+
+join:
+ %add = add i32 %i, 10
+ store i32 %add, ptr %gep.a, align 4
+ br label %latch
+
+latch:
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop, !prof !3
+
+exit:
+ ret void
+}
+
+define void @all_zero_weights(ptr noalias %a, ptr noalias %idx) {
+; The branch in %if.then has all-zero weights. BranchProbabilityInfo takes both
+; of its edges as equally likely.
+;
+; %loop 1 = 1
+; %if.then 1/2 = 1/2
+; %if.then.2 1/4 = 1/4
+; %join 3/4 = 3/4
+; %if.then.3 3/8 = 3/8
+; %latch 1 = 1
+;
+; TODO: VPlan currently treats all-zero weights as unknown, so it records no
+; frequency for %if.then.2 and the blocks it reaches.
+;
+; BFI-LABEL: block-frequency-info: all_zero_weights
+; BFI-NEXT: - entry: float = 1.0,
+; BFI-NEXT: - loop: float = 1000.0,
+; BFI-NEXT: - if.then: float = 500.0,
+; BFI-NEXT: - if.then.2: float = 250.0,
+; BFI-NEXT: - join: float = 750.0,
+; BFI-NEXT: - if.then.3: float = 375.0,
+; BFI-NEXT: - latch: float = 1000.0,
+; BFI-NEXT: - exit: float = 1.0,
+;
+; VPLAN-LABEL: VPlan for loop in 'all_zero_weights'
+; VPLAN: if.then:
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 4611686018427387904 (50%))
+; VPLAN-NEXT: EMIT ir<%c.1> = icmp sgt ir<%i>, ir<10> (!vplan.execution.frequency 4611686018427387904 (50%))
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.1> (!prof {0, 0}, !vplan.execution.frequency 4611686018427387904 (50%))
+; VPLAN-NEXT: Successor(s): if.then.2, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then.2:
+; VPLAN-NEXT: EMIT store ir<2>, ir<%gep.a>{{$}}
+; VPLAN-NEXT: Successor(s): join
+; VPLAN-EMPTY:
+; VPLAN-NEXT: join:
+; VPLAN-NEXT: EMIT ir<%c.2> = icmp sgt ir<%i>, ir<20>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.2> (!prof {1, 1}){{$}}
+; VPLAN-NEXT: Successor(s): if.then.3, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.then.3:
+; VPLAN-NEXT: EMIT store ir<3>, ir<%gep.a>{{$}}
+; VPLAN-NEXT: Successor(s): latch
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %gep.idx = getelementptr inbounds i32, ptr %idx, i64 %iv
+ %i = load i32, ptr %gep.idx, align 4
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %c.0 = icmp sgt i32 %i, 0
+ br i1 %c.0, label %if.then, label %join, !prof !2
+
+if.then:
+ store i32 1, ptr %gep.a, align 4
+ %c.1 = icmp sgt i32 %i, 10
+ br i1 %c.1, label %if.then.2, label %latch, !prof !16
+
+if.then.2:
+ store i32 2, ptr %gep.a, align 4
+ br label %join
+
+join:
+ %c.2 = icmp sgt i32 %i, 20
+ br i1 %c.2, label %if.then.3, label %latch, !prof !2
+
+if.then.3:
+ store i32 3, ptr %gep.a, align 4
+ br label %latch
+
+latch:
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop, !prof !3
+
+exit:
+ ret void
+}
+
+define void @estimated_through_weighted_branch(ptr noalias %a, ptr noalias %idx) {
+;
+; %loop 1 = 1
+; %outer.then 5/8 = 5/8 (estimated)
+; %inner.then 5/32 = 5/32 (estimated)
+; %latch 1 = 1
+;
+; BFI-LABEL: block-frequency-info: estimated_through_weighted_branch
+; BFI-NEXT: - entry: float = 1.0,
+; BFI-NEXT: - loop: float = 1000.0,
+; BFI-NEXT: - outer.then: float = 625.0,
+; BFI-NEXT: - inner.then: float = 156.25,
+; BFI-NEXT: - latch: float = 1000.0,
+; BFI-NEXT: - exit: float = 1.0,
+;
+; VPLAN-LABEL: VPlan for loop in 'estimated_through_weighted_branch'
+; VPLAN: outer.then:
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; VPLAN-NEXT: EMIT ir<%c.1> = icmp sgt ir<%i>, ir<10> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.1> (!prof {1, 3}, !vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; VPLAN-NEXT: Successor(s): inner.then, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: inner.then:
+; VPLAN-NEXT: EMIT store ir<2>, ir<%gep.a> (!vplan.execution.frequency 1441151880758558720 (15.63%, estimated))
+; VPLAN-NEXT: Successor(s): latch
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %gep.idx = getelementptr inbounds i32, ptr %idx, i64 %iv
+ %i = load i32, ptr %gep.idx, align 4
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %c.0 = icmp sgt i32 %i, 0
+ br i1 %c.0, label %outer.then, label %latch
+
+outer.then:
+ store i32 1, ptr %gep.a, align 4
+ %c.1 = icmp sgt i32 %i, 10
+ br i1 %c.1, label %inner.then, label %latch, !prof !0
+
+inner.then:
+ store i32 2, ptr %gep.a, align 4
+ br label %latch
+
+latch:
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop, !prof !3
+
+exit:
+ ret void
+}
+
+define void @rarely_executed_chain(ptr noalias %a, ptr noalias %idx) {
+;aaaa
+; %loop 1 = 1
+; %if.a 2^-31 ~ 4.66e-10
+; %if.b 2^-62 ~ 2.17e-19
+; %if.c 2^-63 ~ 1.08e-19
+; %if.d 2^-64 ~ 5.42e-20
+; %if.e 2^-64 ~ 5.42e-20 (clamped up from 2^-65)
+; %latch 1 = 1
+;
+; TODO: %if.a and the blocks it reaches are currently recorded as never
+; executing.
+;
+; BFI-LABEL: block-frequency-info: rarely_executed_chain
+; BFI-NEXT: - entry: float = 1.0,
+; BFI-NEXT: - loop: float = 1000.0,
+; BFI-NEXT: - if.a: float = 0.00000046566,
+; BFI-NEXT: - if.b: float = 0.00000000000000021684,
+; BFI-NEXT: - if.c: float = 0.00000000000000010842,
+; BFI-NEXT: - if.d: float = 0.00000000000000005421,
+; BFI-NEXT: - if.e: float = 0.00000000000000005421,
+; BFI-NEXT: - latch: float = 1000.0,
+; BFI-NEXT: - exit: float = 1.0,
+;
+; VPLAN-LABEL: VPlan for loop in 'rarely_executed_chain'
+; VPLAN: if.a:
+; VPLAN-NEXT: EMIT ir<%c.1> = icmp sgt ir<%i>, ir<10>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.1> (!prof {1000, 0}){{$}}
+; VPLAN-NEXT: Successor(s): latch, if.b
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.b:
+; VPLAN-NEXT: EMIT ir<%c.2> = icmp sgt ir<%i>, ir<20>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.2> (!prof {1, 1}){{$}}
+; VPLAN-NEXT: Successor(s): if.c, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.c:
+; VPLAN-NEXT: EMIT ir<%c.3> = icmp sgt ir<%i>, ir<30>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.3> (!prof {1, 1}){{$}}
+; VPLAN-NEXT: Successor(s): latch, if.d
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.d:
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a>{{$}}
+; VPLAN-NEXT: EMIT ir<%c.4> = icmp sgt ir<%i>, ir<40>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.4> (!prof {1, 1}){{$}}
+; VPLAN-NEXT: Successor(s): if.e, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: if.e:
+; VPLAN-NEXT: EMIT store ir<2>, ir<%gep.a>{{$}}
+; VPLAN-NEXT: Successor(s): latch
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %gep.idx = getelementptr inbounds i32, ptr %idx, i64 %iv
+ %i = load i32, ptr %gep.idx, align 4
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %c.0 = icmp sgt i32 %i, 0
+ br i1 %c.0, label %latch, label %if.a, !prof !14
+
+if.a:
+ %c.1 = icmp sgt i32 %i, 10
+ br i1 %c.1, label %latch, label %if.b, !prof !14
+
+if.b:
+ %c.2 = icmp sgt i32 %i, 20
+ br i1 %c.2, label %if.c, label %latch, !prof !2
+
+if.c:
+ %c.3 = icmp sgt i32 %i, 30
+ br i1 %c.3, label %latch, label %if.d, !prof !2
+
+if.d:
+ store i32 1, ptr %gep.a, align 4
+ %c.4 = icmp sgt i32 %i, 40
+ br i1 %c.4, label %if.e, label %latch, !prof !2
+
+if.e:
+ store i32 2, ptr %gep.a, align 4
+ br label %latch
+
+latch:
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop, !prof !3
+
+exit:
+ ret void
+}
+
+define void @nested_zero_weight_siblings(ptr noalias %a, ptr noalias %idx) {
+; Nested branches whose skipping edges have zero weight. Each is treated as
+; cold.
+;
+; %loop 1 = 1
+; %then.0 1 - 2^-31 ~ 1 - 4.66e-10
+; %then.1 (1 - 2^-31)^2 ~ 1 - 9.31e-10
+; %then.2 (1 - 2^-31)^3 ~ 1 - 1.40e-9
+; %latch 1 = 1
+;
+; TODO: %then.0, %then.1 and %then.2 are currently recorded as always executing.
+;
+; BFI-LABEL: block-frequency-info: nested_zero_weight_siblings
+; BFI-NEXT: - entry: float = 1.0,
+; BFI-NEXT: - loop: float = 1000.0, int = 18014398509481984
+; BFI-NEXT: - then.0: float = 1000.0, int = 18014398501093376
+; BFI-NEXT: - then.1: float = 1000.0, int = 18014398492704768
+; BFI-NEXT: - then.2: float = 1000.0, int = 18014398484316160
+; BFI-NEXT: - latch: float = 1000.0, int = 18014398509481984
+; BFI-NEXT: - exit: float = 1.0,
+;
+; VPLAN-LABEL: VPlan for loop in 'nested_zero_weight_siblings'
+; VPLAN: then.0:
+; VPLAN-NEXT: EMIT store ir<0>, ir<%gep.a>{{$}}
+; VPLAN-NEXT: EMIT ir<%c.1> = icmp sgt ir<%i>, ir<10>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.1> (!prof {1000, 0}){{$}}
+; VPLAN-NEXT: Successor(s): then.1, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: then.1:
+; VPLAN-NEXT: EMIT store ir<1>, ir<%gep.a>{{$}}
+; VPLAN-NEXT: EMIT ir<%c.2> = icmp sgt ir<%i>, ir<20>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%c.2> (!prof {1000, 0}){{$}}
+; VPLAN-NEXT: Successor(s): then.2, latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: then.2:
+; VPLAN-NEXT: EMIT store ir<2>, ir<%gep.a>{{$}}
+; VPLAN-NEXT: Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT: latch:
+; VPLAN-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT: EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
+; VPLAN-NEXT: Successor(s): middle.block, loop
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %gep.idx = getelementptr inbounds i32, ptr %idx, i64 %iv
+ %i = load i32, ptr %gep.idx, align 4
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %c.0 = icmp sgt i32 %i, 0
+ br i1 %c.0, label %then.0, label %latch, !prof !14
+
+then.0:
+ store i32 0, ptr %gep.a, align 4
+ %c.1 = icmp sgt i32 %i, 10
+ br i1 %c.1, label %then.1, label %latch, !prof !14
+
+then.1:
+ store i32 1, ptr %gep.a, align 4
+ %c.2 = icmp sgt i32 %i, 20
+ br i1 %c.2, label %then.2, label %latch, !prof !14
+
+then.2:
+ store i32 2, ptr %gep.a, align 4
+ br label %latch
+
+latch:
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop, !prof !3
+
+exit:
+ ret void
+}
+
!0 = !{!"branch_weights", i32 1, i32 3}
!1 = !{!"branch_weights", i32 1, i32 2}
!2 = !{!"branch_weights", i32 1, i32 1}
@@ -975,3 +1476,7 @@ exit:
!8 = !{!"branch_weights", i32 1, i32 1, i32 1, i32 1, i32 1}
!9 = !{!"branch_weights", i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1}
!12 = !{!"branch_weights", i32 2, i32 2863311531, i32 2863311531, i32 2863311531}
+!13 = !{!"branch_weights", i32 0, i32 1000}
+!14 = !{!"branch_weights", i32 1000, i32 0}
+!15 = !{!"branch_weights", i32 5, i32 1, i32 1}
+!16 = !{!"branch_weights", i32 0, i32 0}
diff --git a/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
index bf996014137e9..730b7fa2404bb 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
@@ -299,4 +299,58 @@ exit:
ret i32 0
}
+; %then is only entered via an edge with zero branch weight. The cost of its
+; replicate region must be scaled by the minimal possible execution probability.
+define void @predicated_sdiv_zero_weight(ptr noalias %a, ptr noalias %b, i64 %n) {
+; CHECK-LABEL: define void @predicated_sdiv_zero_weight(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[C:%.*]] = icmp sgt i32 [[L]], 0
+; CHECK-NEXT: br i1 [[C]], label %[[THEN:.*]], label %[[LATCH]], !prof [[PROF4:![0-9]+]]
+; CHECK: [[THEN]]:
+; CHECK-NEXT: [[D:%.*]] = sdiv i32 1000, [[L]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[D]], ptr [[GEP_B]], align 4
+; CHECK-NEXT: br label %[[LATCH]]
+; CHECK: [[LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l = load i32, ptr %gep.a, align 4
+ %c = icmp sgt i32 %l, 0
+ br i1 %c, label %then, label %latch, !prof !0
+
+then:
+ %d = sdiv i32 1000, %l
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ store i32 %d, ptr %gep.b, align 4
+ br label %latch
+
+latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, %n
+ br i1 %ec, label %exit, label %loop, !llvm.loop !1
+
+exit:
+ ret void
+}
+
attributes #0 = { "target-cpu"="skylake-avx512" }
+
+!0 = !{!"branch_weights", i32 0, i32 1000}
+!1 = distinct !{!1, !2}
+!2 = !{!"llvm.loop.interleave.count", i32 1}
diff --git a/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
index b8861ee4590bb..2f7fdeda4c08f 100644
--- a/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
@@ -1304,6 +1304,91 @@ exit:
ret void
}
+; Predicated store where the condition has a zero taken weight. As in
+; BlockFrequencyInfo, a zero weight marks the edge as cold, not as never taken,
+; so the replicate region branches must still get (very unlikely) weights.
+define void @predicated_store_zero_taken_weight(ptr %a, i32 %n) {
+; VF4IC1-LABEL: define void @predicated_store_zero_taken_weight(
+; VF4IC1-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) {
+; VF4IC1: [[ENTRY:.*:]]
+; VF4IC1: br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
+; VF4IC1: [[VECTOR_PH]]:
+; VF4IC1: [[VECTOR_BODY:.*]]:
+; VF4IC1: br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF4IC1: [[PRED_STORE_IF]]:
+; VF4IC1: [[PRED_STORE_CONTINUE]]:
+; VF4IC1: br i1 [[TMP4:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
+; VF4IC1: [[PRED_STORE_IF1]]:
+; VF4IC1: [[PRED_STORE_CONTINUE2]]:
+; VF4IC1: br i1 [[TMP7:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
+; VF4IC1: [[PRED_STORE_IF3]]:
+; VF4IC1: [[PRED_STORE_CONTINUE4]]:
+; VF4IC1: br i1 [[TMP10:%.*]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6:.*]]
+; VF4IC1: [[PRED_STORE_IF5]]:
+; VF4IC1: [[PRED_STORE_CONTINUE6]]:
+; VF4IC1: br i1 [[TMP13:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP50:![0-9]+]]
+; VF4IC1: [[MIDDLE_BLOCK]]:
+; VF4IC1: br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]], !prof [[PROF7]]
+; VF4IC1: [[SCALAR_PH]]:
+; VF4IC1: [[LOOP:.*]]:
+; VF4IC1: br i1 [[C:%.*]], label %[[IF_THEN:.*]], label %[[LATCH:.*]], !prof [[PROF51:![0-9]+]]
+; VF4IC1: [[IF_THEN]]:
+; VF4IC1: [[LATCH]]:
+; VF4IC1: br i1 [[EXITCOND:%.*]], label %[[EXIT]], label %[[LOOP]], !prof [[PROF8]], !llvm.loop [[LOOP52:![0-9]+]]
+; VF4IC1: [[EXIT]]:
+;
+; VF2IC2-LABEL: define void @predicated_store_zero_taken_weight(
+; VF2IC2-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) {
+; VF2IC2: [[ENTRY:.*:]]
+; VF2IC2: br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
+; VF2IC2: [[VECTOR_PH]]:
+; VF2IC2: [[VECTOR_BODY:.*]]:
+; VF2IC2: br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF2IC2: [[PRED_STORE_IF]]:
+; VF2IC2: [[PRED_STORE_CONTINUE]]:
+; VF2IC2: br i1 [[TMP6:%.*]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]]
+; VF2IC2: [[PRED_STORE_IF2]]:
+; VF2IC2: [[PRED_STORE_CONTINUE3]]:
+; VF2IC2: br i1 [[TMP9:%.*]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]]
+; VF2IC2: [[PRED_STORE_IF4]]:
+; VF2IC2: [[PRED_STORE_CONTINUE5]]:
+; VF2IC2: br i1 [[TMP12:%.*]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7:.*]]
+; VF2IC2: [[PRED_STORE_IF6]]:
+; VF2IC2: [[PRED_STORE_CONTINUE7]]:
+; VF2IC2: br i1 [[TMP15:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP50:![0-9]+]]
+; VF2IC2: [[MIDDLE_BLOCK]]:
+; VF2IC2: br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]], !prof [[PROF7]]
+; VF2IC2: [[SCALAR_PH]]:
+; VF2IC2: [[LOOP:.*]]:
+; VF2IC2: br i1 [[C:%.*]], label %[[IF_THEN:.*]], label %[[LATCH:.*]], !prof [[PROF51:![0-9]+]]
+; VF2IC2: [[IF_THEN]]:
+; VF2IC2: [[LATCH]]:
+; VF2IC2: br i1 [[EXITCOND:%.*]], label %[[EXIT]], label %[[LOOP]], !prof [[PROF8]], !llvm.loop [[LOOP52:![0-9]+]]
+; VF2IC2: [[EXIT]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %latch ]
+ %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+ %val = load i32, ptr %gep, align 4
+ %c = icmp sgt i32 %val, 0
+ br i1 %c, label %if.then, label %latch, !prof !10
+
+if.then:
+ store i32 0, ptr %gep, align 4
+ br label %latch
+
+latch:
+ %iv.next = add nuw nsw i32 %iv, 1
+ %exitcond = icmp eq i32 %iv.next, %n
+ br i1 %exitcond, label %exit, label %loop, !prof !0
+
+exit:
+ ret void
+}
+
!0 = !{!"branch_weights", i32 1, i32 1000}
!1 = !{!"branch_weights", i32 1, i32 7}
!2 = !{!"branch_weights", i32 1, i32 1}
@@ -1314,6 +1399,7 @@ exit:
!7 = !{!"branch_weights", i32 4, i32 1, i32 2, i32 1}
!8 = !{!"branch_weights", i32 4294967295, i32 1}
!9 = !{!"branch_weights", i32 1, i32 0}
+!10 = !{!"branch_weights", i32 0, i32 1000}
;.
; VF4IC1: [[PROF0]] = !{!"branch_weights", i32 1, i32 127}
; VF4IC1: [[PROF1]] = !{!"branch_weights", i32 1, i32 7}
@@ -1365,6 +1451,9 @@ exit:
; VF4IC1: [[LOOP47]] = distinct !{[[LOOP47]], [[META5]], [[META4]], [[META10]]}
; VF4IC1: [[LOOP48]] = distinct !{[[LOOP48]], [[META4]], [[META5]], [[META6]]}
; VF4IC1: [[LOOP49]] = distinct !{[[LOOP49]], [[META5]], [[META4]], [[META10]]}
+; VF4IC1: [[LOOP50]] = distinct !{[[LOOP50]], [[META4]], [[META5]], [[META6]]}
+; VF4IC1: [[PROF51]] = !{!"branch_weights", i32 0, i32 1000}
+; VF4IC1: [[LOOP52]] = distinct !{[[LOOP52]], [[META5]], [[META4]], [[META10]]}
;.
; VF2IC2: [[PROF0]] = !{!"branch_weights", i32 1, i32 127}
; VF2IC2: [[PROF1]] = !{!"branch_weights", i32 1, i32 7}
@@ -1416,4 +1505,7 @@ exit:
; VF2IC2: [[LOOP47]] = distinct !{[[LOOP47]], [[META5]], [[META4]], [[META10]]}
; VF2IC2: [[LOOP48]] = distinct !{[[LOOP48]], [[META4]], [[META5]], [[META6]]}
; VF2IC2: [[LOOP49]] = distinct !{[[LOOP49]], [[META5]], [[META4]], [[META10]]}
+; VF2IC2: [[LOOP50]] = distinct !{[[LOOP50]], [[META4]], [[META5]], [[META6]]}
+; VF2IC2: [[PROF51]] = !{!"branch_weights", i32 0, i32 1000}
+; VF2IC2: [[LOOP52]] = distinct !{[[LOOP52]], [[META5]], [[META4]], [[META10]]}
;.
More information about the llvm-commits
mailing list