[llvm] 5950bbf - [VPlan] Fix merged replicate region frequency when only one is known. (#221607)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 7 00:51:13 PDT 2026
Author: Florian Hahn
Date: 2026-09-07T08:51:07+01:00
New Revision: 5950bbf1e12fa2445c9de948c3f602f03ba12d62
URL: https://github.com/llvm/llvm-project/commit/5950bbf1e12fa2445c9de948c3f602f03ba12d62
DIFF: https://github.com/llvm/llvm-project/commit/5950bbf1e12fa2445c9de948c3f602f03ba12d62.diff
LOG: [VPlan] Fix merged replicate region frequency when only one is known. (#221607)
When merging two replicate regions guarded by the same mask, the merged
region's entry frequency should be the higher of the two original
frequencies, since it now fires whenever either original region did.
The existing check only updated the frequency when both were known,
leaving the second region's own (possibly much lower) frequency in place
whenever the first was unknown. This means we might set a misleadingly
low frequency. Conservatively set to unknown instead.
Added:
Modified:
llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index a93bad3b471c5..ab60aaa8b833e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -494,12 +494,18 @@ static bool mergeReplicateRegionsIntoSuccessors(VPlan &Plan) {
// The merged region is entered whenever either of the original regions was,
// so use the higher, i.e. more conservative, of their entry frequencies.
+ // If only one of the two is known, the higher one is unknown, so the
+ // result must be unknown too.
VPBranchOnMaskRecipe *Guard2 = Region2->getEntryBranchOnMask();
std::optional<BlockFrequency> Freq1 =
Region1->getEntryBranchOnMask()->getExecutionFrequency();
std::optional<BlockFrequency> Freq2 = Guard2->getExecutionFrequency();
- if (Freq1 && Freq2 && *Freq2 < *Freq1)
- Guard2->setExecutionFrequency(Freq1, Plan.getContext());
+ if (Freq1 && Freq2) {
+ if (*Freq2 < *Freq1)
+ Guard2->setExecutionFrequency(Freq1, Plan.getContext());
+ } else if (Freq2) {
+ Guard2->clearExecutionFrequency();
+ }
// Note: No fusion-preventing memory dependencies are expected in either
// region. Such dependencies should be rejected during earlier dependence
diff --git a/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
index 0227d7b99c9b6..b8861ee4590bb 100644
--- a/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
@@ -1204,6 +1204,106 @@ exit:
ret void
}
+; Two predicated stores guarded by the same condition. The first branch is
+; always taken (so it needs no weights of its own), while the second is
+; rarely taken. The merged region fires whenever either original region did,
+; i.e. (near) always, so it must not keep the second region's low frequency.
+define void @merged_replicate_regions_first_always_taken(ptr noalias %a, ptr noalias %b, i32 %n) {
+; VF4IC1-LABEL: define void @merged_replicate_regions_first_always_taken(
+; VF4IC1-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i32 [[N:%.*]]) {
+; VF4IC1: [[ENTRY:.*:]]
+; VF4IC1: br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
+; VF4IC1: [[VECTOR_PH]]:
+; VF4IC1: [[VECTOR_BODY:.*]]:
+; VF4IC1: br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF4IC1: [[PRED_STORE_IF]]:
+; VF4IC1: [[PRED_STORE_CONTINUE]]:
+; VF4IC1: br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
+; VF4IC1: [[PRED_STORE_IF1]]:
+; VF4IC1: [[PRED_STORE_CONTINUE2]]:
+; VF4IC1: br i1 [[TMP9:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
+; VF4IC1: [[PRED_STORE_IF3]]:
+; VF4IC1: [[PRED_STORE_CONTINUE4]]:
+; VF4IC1: br i1 [[TMP13:%.*]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6:.*]]
+; VF4IC1: [[PRED_STORE_IF5]]:
+; VF4IC1: [[PRED_STORE_CONTINUE6]]:
+; VF4IC1: br i1 [[TMP17:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP48:![0-9]+]]
+; VF4IC1: [[MIDDLE_BLOCK]]:
+; VF4IC1: br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]], !prof [[PROF7]]
+; VF4IC1: [[SCALAR_PH]]:
+; VF4IC1: [[LOOP:.*]]:
+; VF4IC1: br i1 [[CMP:%.*]], label %[[IF_THEN_1:.*]], label %[[MERGE:.*]], !prof [[PROF8]]
+; VF4IC1: [[IF_THEN_1]]:
+; VF4IC1: [[MERGE]]:
+; VF4IC1: br i1 [[CMP]], label %[[IF_THEN_2:.*]], label %[[LATCH:.*]], !prof [[PROF42]]
+; VF4IC1: [[IF_THEN_2]]:
+; VF4IC1: [[LATCH]]:
+; VF4IC1: br i1 [[EXITCOND:%.*]], label %[[EXIT]], label %[[LOOP]], !prof [[PROF8]], !llvm.loop [[LOOP49:![0-9]+]]
+; VF4IC1: [[EXIT]]:
+;
+; VF2IC2-LABEL: define void @merged_replicate_regions_first_always_taken(
+; VF2IC2-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i32 [[N:%.*]]) {
+; VF2IC2: [[ENTRY:.*:]]
+; VF2IC2: br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
+; VF2IC2: [[VECTOR_PH]]:
+; VF2IC2: [[VECTOR_BODY:.*]]:
+; VF2IC2: br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF2IC2: [[PRED_STORE_IF]]:
+; VF2IC2: [[PRED_STORE_CONTINUE]]:
+; VF2IC2: br i1 [[TMP7:%.*]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]]
+; VF2IC2: [[PRED_STORE_IF2]]:
+; VF2IC2: [[PRED_STORE_CONTINUE3]]:
+; VF2IC2: br i1 [[TMP11:%.*]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]]
+; VF2IC2: [[PRED_STORE_IF4]]:
+; VF2IC2: [[PRED_STORE_CONTINUE5]]:
+; VF2IC2: br i1 [[TMP15:%.*]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7:.*]]
+; VF2IC2: [[PRED_STORE_IF6]]:
+; VF2IC2: [[PRED_STORE_CONTINUE7]]:
+; VF2IC2: br i1 [[TMP19:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP48:![0-9]+]]
+; VF2IC2: [[MIDDLE_BLOCK]]:
+; VF2IC2: br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]], !prof [[PROF7]]
+; VF2IC2: [[SCALAR_PH]]:
+; VF2IC2: [[LOOP:.*]]:
+; VF2IC2: br i1 [[CMP:%.*]], label %[[IF_THEN_1:.*]], label %[[MERGE:.*]], !prof [[PROF8]]
+; VF2IC2: [[IF_THEN_1]]:
+; VF2IC2: [[MERGE]]:
+; VF2IC2: br i1 [[CMP]], label %[[IF_THEN_2:.*]], label %[[LATCH:.*]], !prof [[PROF42]]
+; VF2IC2: [[IF_THEN_2]]:
+; VF2IC2: [[LATCH]]:
+; VF2IC2: br i1 [[EXITCOND:%.*]], label %[[EXIT]], label %[[LOOP]], !prof [[PROF8]], !llvm.loop [[LOOP49:![0-9]+]]
+; VF2IC2: [[EXIT]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %latch ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i32 %iv
+ %val = load i32, ptr %gep.a, align 4
+ %cmp = icmp sgt i32 %val, 0
+ br i1 %cmp, label %if.then.1, label %merge, !prof !9
+
+if.then.1:
+ store i32 0, ptr %gep.a, align 4
+ br label %merge
+
+merge:
+ br i1 %cmp, label %if.then.2, label %latch, !prof !0
+
+if.then.2:
+ %gep.b = getelementptr inbounds i32, ptr %b, i32 %iv
+ store i32 0, ptr %gep.b, align 4
+ br label %latch
+
+latch:
+ %iv.next = add nuw nsw i32 %iv, 1
+ %exitcond = icmp eq i32 %iv.next, %n
+ br i1 %exitcond, label %exit, label %loop, !prof !0
+
+exit:
+ ret void
+}
+
!0 = !{!"branch_weights", i32 1, i32 1000}
!1 = !{!"branch_weights", i32 1, i32 7}
!2 = !{!"branch_weights", i32 1, i32 1}
@@ -1213,6 +1313,7 @@ exit:
!6 = !{!"branch_weights", i32 1, i32 100000}
!7 = !{!"branch_weights", i32 4, i32 1, i32 2, i32 1}
!8 = !{!"branch_weights", i32 4294967295, i32 1}
+!9 = !{!"branch_weights", i32 1, i32 0}
;.
; VF4IC1: [[PROF0]] = !{!"branch_weights", i32 1, i32 127}
; VF4IC1: [[PROF1]] = !{!"branch_weights", i32 1, i32 7}
@@ -1262,6 +1363,8 @@ exit:
; VF4IC1: [[LOOP45]] = distinct !{[[LOOP45]], [[META4]], [[META5]], [[META6]]}
; VF4IC1: [[PROF46]] = !{!"branch_weights", i32 -1, i32 1}
; VF4IC1: [[LOOP47]] = distinct !{[[LOOP47]], [[META5]], [[META4]], [[META10]]}
+; VF4IC1: [[LOOP48]] = distinct !{[[LOOP48]], [[META4]], [[META5]], [[META6]]}
+; VF4IC1: [[LOOP49]] = distinct !{[[LOOP49]], [[META5]], [[META4]], [[META10]]}
;.
; VF2IC2: [[PROF0]] = !{!"branch_weights", i32 1, i32 127}
; VF2IC2: [[PROF1]] = !{!"branch_weights", i32 1, i32 7}
@@ -1311,4 +1414,6 @@ exit:
; VF2IC2: [[LOOP45]] = distinct !{[[LOOP45]], [[META4]], [[META5]], [[META6]]}
; VF2IC2: [[PROF46]] = !{!"branch_weights", i32 -1, i32 1}
; VF2IC2: [[LOOP47]] = distinct !{[[LOOP47]], [[META5]], [[META4]], [[META10]]}
+; VF2IC2: [[LOOP48]] = distinct !{[[LOOP48]], [[META4]], [[META5]], [[META6]]}
+; VF2IC2: [[LOOP49]] = distinct !{[[LOOP49]], [[META5]], [[META4]], [[META10]]}
;.
More information about the llvm-commits
mailing list