[llvm] [LoopVectorize] Fix nondeterminism in loop-vectorize (PR #200833)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Jun 1 07:20:54 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Orlando Cazalet-Hyams (OCHyams)
<details>
<summary>Changes</summary>
The nondeterministic iteration over `AddrDefs` (SmallPtrSet) causes nondeterministic output for the test case in this patch (reduced from a C codebase). One of two different outputs is generated arbitrarily, chosen roughly equally.
Between the two different outputs sometimes the instruction
`%3 = load i64, ptr %2, align 8`
has an associated cost of 4 and othertimes 9. The instruction is visited twice in `setCostBasedWideningDecision` in the `AddrDefs` loop: once directly as an elemement of `AddrDefs`, and the other time indirectly in the lambda `UpdateMemOpUserCost` as a User of another `AddrDefs` element. Each of those times `setWideningDecision` is called with a different cost value; the final of the two calls sets the final value (previous is overwritten). Because `AddrDefs` iteration is nondeterministic, the order of those two calls to `setWideningDecision` is also nondeterministic, hence we see two different costs arbitrarily between runs.
This patch fixes the nondeterministic iteration order. The fact that two costs for the same instruction are arbitrarily chosen between may (or may not) be a deeper issue, but that falls outside my expertise.
Some additional info:
We observe the issue reproducing (frequently) in llvm-22 but not llvm-21.
With the test case in this patch we observe the nondeterminism frequently reproduces (< 3 tries) from 1f331e453fa9c328c165c739f61e17a2815ece82 building on and targeting x86_64 linux. Before that we haven't observed the nondeterminism (> 500 tries), but can't rule out that it may be observed with different inputs based on the fact the SmallPtrSet has been in use since 2017. N.B. Building on Windows bisects to a different commit.
---
Full diff: https://github.com/llvm/llvm-project/pull/200833.diff
2 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+2-2)
- (added) llvm/lib/Transforms/Vectorize/nondetermisitic-widening-cost.ll (+128)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 30456819602b2..d27973eb003b8 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -4831,7 +4831,7 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
return;
// Start with all scalar pointer uses.
- SmallPtrSet<Instruction *, 8> AddrDefs;
+ SmallSetVector<Instruction *, 8> AddrDefs;
for (BasicBlock *BB : TheLoop->blocks())
for (Instruction &I : *BB) {
Instruction *PtrDef =
@@ -4849,7 +4849,7 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
for (auto &Op : I->operands())
if (auto *InstOp = dyn_cast<Instruction>(Op))
if (TheLoop->contains(InstOp) && !isa<PHINode>(InstOp) &&
- AddrDefs.insert(InstOp).second)
+ AddrDefs.insert(InstOp))
Worklist.push_back(InstOp);
}
diff --git a/llvm/lib/Transforms/Vectorize/nondetermisitic-widening-cost.ll b/llvm/lib/Transforms/Vectorize/nondetermisitic-widening-cost.ll
new file mode 100644
index 0000000000000..938f177c37854
--- /dev/null
+++ b/llvm/lib/Transforms/Vectorize/nondetermisitic-widening-cost.ll
@@ -0,0 +1,128 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt --passes=loop-vectorize %s -S | FileCheck %s
+
+; Check that we see expected deterministic (over multiple test runs) output.
+; NOTE: Beware, if this test fails it may be due to non-determinism.
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux"
+
+define i32 @fun(i64 %0, float %1, ptr %a, ptr %b, i64 %len) #0 {
+; CHECK-LABEL: define i32 @fun(
+; CHECK-SAME: i64 [[TMP0:%.*]], float [[TMP1:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[LEN:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[VLA:%.*]] = alloca float, i64 [[LEN]], align 16
+; CHECK-NEXT: [[TMP2:%.*]] = add i64 [[TMP0]], 1
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP2]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
+; CHECK: [[VECTOR_MEMCHECK]]:
+; CHECK-NEXT: [[TMP3:%.*]] = shl i64 [[TMP0]], 2
+; CHECK-NEXT: [[DIFF_CHECK:%.*]] = icmp ult i64 [[TMP3]], 16
+; CHECK-NEXT: br i1 [[DIFF_CHECK]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[TMP2]], 4
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP2]], [[N_MOD_VF]]
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr [4 x i8], ptr [[VLA]], i64 [[TMP0]]
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x float> poison, float [[TMP1]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x float> [[BROADCAST_SPLATINSERT]], <4 x float> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[INDEX]], 0
+; CHECK-NEXT: [[TMP6:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT: [[TMP7:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT: [[TMP8:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP5]]
+; CHECK-NEXT: [[TMP10:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT: [[TMP11:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP7]]
+; CHECK-NEXT: [[TMP12:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP8]]
+; CHECK-NEXT: [[TMP13:%.*]] = load ptr, ptr [[TMP9]], align 8
+; CHECK-NEXT: [[TMP14:%.*]] = load ptr, ptr [[TMP10]], align 8
+; CHECK-NEXT: [[TMP15:%.*]] = load ptr, ptr [[TMP11]], align 8
+; CHECK-NEXT: [[TMP16:%.*]] = load ptr, ptr [[TMP12]], align 8
+; CHECK-NEXT: [[TMP17:%.*]] = load i64, ptr [[TMP13]], align 8
+; CHECK-NEXT: [[TMP18:%.*]] = load i64, ptr [[TMP14]], align 8
+; CHECK-NEXT: [[TMP19:%.*]] = load i64, ptr [[TMP15]], align 8
+; CHECK-NEXT: [[TMP20:%.*]] = load i64, ptr [[TMP16]], align 8
+; CHECK-NEXT: [[TMP21:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP17]]
+; CHECK-NEXT: [[TMP22:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP18]]
+; CHECK-NEXT: [[TMP23:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP19]]
+; CHECK-NEXT: [[TMP24:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP20]]
+; CHECK-NEXT: [[TMP25:%.*]] = load float, ptr [[TMP21]], align 4
+; CHECK-NEXT: [[TMP26:%.*]] = load float, ptr [[TMP22]], align 4
+; CHECK-NEXT: [[TMP27:%.*]] = load float, ptr [[TMP23]], align 4
+; CHECK-NEXT: [[TMP28:%.*]] = load float, ptr [[TMP24]], align 4
+; CHECK-NEXT: [[TMP29:%.*]] = insertelement <4 x float> poison, float [[TMP25]], i32 0
+; CHECK-NEXT: [[TMP30:%.*]] = insertelement <4 x float> [[TMP29]], float [[TMP26]], i32 1
+; CHECK-NEXT: [[TMP31:%.*]] = insertelement <4 x float> [[TMP30]], float [[TMP27]], i32 2
+; CHECK-NEXT: [[TMP32:%.*]] = insertelement <4 x float> [[TMP31]], float [[TMP28]], i32 3
+; CHECK-NEXT: [[TMP33:%.*]] = getelementptr [4 x i8], ptr [[VLA]], i64 [[TMP5]]
+; CHECK-NEXT: [[TMP34:%.*]] = getelementptr float, ptr [[TMP33]], i32 0
+; CHECK-NEXT: store <4 x float> [[TMP32]], ptr [[TMP34]], align 4
+; CHECK-NEXT: [[TMP35:%.*]] = getelementptr [4 x i8], ptr [[TMP4]], i64 [[TMP5]]
+; CHECK-NEXT: [[TMP36:%.*]] = getelementptr float, ptr [[TMP35]], i32 0
+; CHECK-NEXT: store <4 x float> [[BROADCAST_SPLAT]], ptr [[TMP36]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP37:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP37]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP2]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[FOR_END_LOOPEXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_MEMCHECK]] ]
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[ARRAYIDX16:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[INDVARS_IV]]
+; CHECK-NEXT: [[TMP38:%.*]] = load ptr, ptr [[ARRAYIDX16]], align 8
+; CHECK-NEXT: [[TMP39:%.*]] = load i64, ptr [[TMP38]], align 8
+; CHECK-NEXT: [[ARRAYIDX18:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP39]]
+; CHECK-NEXT: [[TMP40:%.*]] = load float, ptr [[ARRAYIDX18]], align 4
+; CHECK-NEXT: [[ARRAYIDX21:%.*]] = getelementptr [4 x i8], ptr [[VLA]], i64 [[INDVARS_IV]]
+; CHECK-NEXT: store float [[TMP40]], ptr [[ARRAYIDX21]], align 4
+; CHECK-NEXT: [[TMP41:%.*]] = load i64, ptr [[B]], align 8
+; CHECK-NEXT: [[ARRAYIDX27:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP41]]
+; CHECK-NEXT: [[TMP42:%.*]] = load float, ptr [[ARRAYIDX27]], align 4
+; CHECK-NEXT: [[ARRAYIDX28:%.*]] = getelementptr [4 x i8], ptr [[VLA]], i64 [[TMP0]]
+; CHECK-NEXT: [[ARRAYIDX30:%.*]] = getelementptr [4 x i8], ptr [[ARRAYIDX28]], i64 [[INDVARS_IV]]
+; CHECK-NEXT: store float [[TMP1]], ptr [[ARRAYIDX30]], align 4
+; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add i64 [[INDVARS_IV]], 1
+; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV]], [[TMP0]]
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label %[[FOR_END_LOOPEXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[FOR_END_LOOPEXIT]]:
+; CHECK-NEXT: ret i32 0
+;
+entry:
+ %vla = alloca float, i64 %len, align 16
+ br label %for.body
+
+for.body: ; preds = %for.body, %entry
+ %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.body ]
+ %arrayidx16 = getelementptr [8 x i8], ptr %a, i64 %indvars.iv
+ %2 = load ptr, ptr %arrayidx16, align 8
+ %3 = load i64, ptr %2, align 8
+ %arrayidx18 = getelementptr [4 x i8], ptr %a, i64 %3
+ %4 = load float, ptr %arrayidx18, align 4
+ %arrayidx21 = getelementptr [4 x i8], ptr %vla, i64 %indvars.iv
+ store float %4, ptr %arrayidx21, align 4
+ %5 = load i64, ptr %b, align 8
+ %arrayidx27 = getelementptr [4 x i8], ptr %a, i64 %5
+ %6 = load float, ptr %arrayidx27, align 4
+ %arrayidx28 = getelementptr [4 x i8], ptr %vla, i64 %0
+ %arrayidx30 = getelementptr [4 x i8], ptr %arrayidx28, i64 %indvars.iv
+ store float %1, ptr %arrayidx30, align 4
+ %indvars.iv.next = add i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv, %0
+ br i1 %exitcond.not, label %for.end.loopexit, label %for.body
+
+for.end.loopexit: ; preds = %for.body
+ ret i32 0
+}
+
+attributes #0 = { "target-features"="+avx2" }
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META1]]}
+;.
``````````
</details>
https://github.com/llvm/llvm-project/pull/200833
More information about the llvm-commits
mailing list