[llvm] [LoopVectorize] Fix crash in VPWidenRecipe (PR #177357)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Fri Aug 28 04:43:36 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/177357
>From f9de623de2dcb31b644d0853cd594455eeab84fd Mon Sep 17 00:00:00 2001
From: Manasij Mukherjee <manasijm at nvidia.com>
Date: Thu, 22 Jan 2026 05:24:23 -0800
Subject: [PATCH 1/3] [LoopVectorize] Fix crash in VPWidenRecipe
VPWidenRecipe crashed when the ExtractValue operand was a struct constant.
Use the extracted value instead.
Fixes #176663
---
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 4 ++-
.../multiple-result-intrinsics.ll | 27 +++++++++++++++++++
2 files changed, 30 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 584b430c8e78e..59209c80d1be1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -2273,9 +2273,11 @@ void VPWidenRecipe::execute(VPTransformState &State) {
}
case Instruction::ExtractValue: {
assert(getNumOperands() == 2 && "expected single level extractvalue");
- Value *Op = State.get(getOperand(0));
+ Value *Op = State.get(getOperand(0), /*NeedsScalar=*/true);
Value *Extract = Builder.CreateExtractValue(
Op, cast<VPConstantInt>(getOperand(1))->getZExtValue());
+ if (!Extract->getType()->isVectorTy())
+ Extract = Builder.CreateVectorSplat(State.VF, Extract);
State.set(this, Extract);
break;
}
diff --git a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
index f64d43adecfb8..a1650416760cc 100644
--- a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
+++ b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
@@ -816,3 +816,30 @@ for.body:
exit:
ret void
}
+
+define i32 @smul_with_overflow_constant_args() {
+;
+; CHECK-LABEL: define i32 @smul_with_overflow_constant_args() {
+; CHECK: [[ENTRY:.*:]]
+; CHECK: [[VECTOR_PH:.*:]]
+; CHECK: [[VECTOR_BODY:.*:]]
+; CHECK: [[MIDDLE_BLOCK:.*:]]
+; CHECK: [[EXIT:.*:]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc = phi i32 [ 0, %entry ], [ %or, %loop ]
+ %call = tail call { i32, i1 } @llvm.smul.with.overflow.i32(i32 0, i32 0)
+ %overflow = extractvalue { i32, i1 } %call, 1
+ %ext = zext i1 %overflow to i32
+ %or = or i32 %acc, %ext
+ %iv.next = add i64 %iv, 1
+ %done = icmp eq i64 %iv, 1
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret i32 %or
+}
>From 08f565a8440cb07affae818c98103f21b49f0257 Mon Sep 17 00:00:00 2001
From: Manasij Mukherjee <manasijm at nvidia.com>
Date: Thu, 22 Jan 2026 16:11:01 -0800
Subject: [PATCH 2/3] Better test, checks
---
.../LoopVectorize/multiple-result-intrinsics.ll | 12 +++++++-----
1 file changed, 7 insertions(+), 5 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
index a1650416760cc..cc0e553bac511 100644
--- a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
+++ b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
@@ -817,14 +817,16 @@ exit:
ret void
}
-define i32 @smul_with_overflow_constant_args() {
-;
-; CHECK-LABEL: define i32 @smul_with_overflow_constant_args() {
-; CHECK: [[ENTRY:.*:]]
+define i32 @smul_with_overflow_vars(i32 %a, i32 %b) {
+; CHECK-LABEL: define i32 @smul_with_overflow_vars(
+; CHECK-SAME: i32 [[A:%.*]], i32 [[B:%.*]]) {
; CHECK: [[VECTOR_PH:.*:]]
; CHECK: [[VECTOR_BODY:.*:]]
+; CHECK: [[TMP0:%.*]] = call { <2 x i32>, <2 x i1> } @llvm.smul.with.overflow.v2i32(<2 x i32> [[BROADCAST_SPLAT2:%.*]], <2 x i32> [[BROADCAST_SPLAT:%.*]])
+; CHECK: [[TMP1:%.*]] = extractvalue { <2 x i32>, <2 x i1> } [[TMP0]], 1
; CHECK: [[MIDDLE_BLOCK:.*:]]
; CHECK: [[EXIT:.*:]]
+; CHECK: [[EXIT1:.*:]]
;
entry:
br label %loop
@@ -832,7 +834,7 @@ entry:
loop:
%iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
%acc = phi i32 [ 0, %entry ], [ %or, %loop ]
- %call = tail call { i32, i1 } @llvm.smul.with.overflow.i32(i32 0, i32 0)
+ %call = tail call { i32, i1 } @llvm.smul.with.overflow.i32(i32 %a, i32 %b)
%overflow = extractvalue { i32, i1 } %call, 1
%ext = zext i1 %overflow to i32
%or = or i32 %acc, %ext
>From 99c47298084f890e0eb9bcb07fa8bf51091e6d74 Mon Sep 17 00:00:00 2001
From: Manasij Mukherjee <manasijm at nvidia.com>
Date: Mon, 9 Feb 2026 13:41:00 -0800
Subject: [PATCH 3/3] Add comment, split out test, update script filters out
less context
---
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 3 ++
.../multiple-result-intrinsics.ll | 29 -----------------
.../widen-extractvalue-struct.ll | 32 +++++++++++++++++++
3 files changed, 35 insertions(+), 29 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/widen-extractvalue-struct.ll
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 59209c80d1be1..5bce025d1d39e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -2273,6 +2273,9 @@ void VPWidenRecipe::execute(VPTransformState &State) {
}
case Instruction::ExtractValue: {
assert(getNumOperands() == 2 && "expected single level extractvalue");
+ // Fetch the struct operand as a scalar so that extractvalue is valid
+ // on the struct type, regardless of whether it is a widened struct of
+ // vectors or a scalar struct constant folded by VPlan simplification.
Value *Op = State.get(getOperand(0), /*NeedsScalar=*/true);
Value *Extract = Builder.CreateExtractValue(
Op, cast<VPConstantInt>(getOperand(1))->getZExtValue());
diff --git a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
index cc0e553bac511..f64d43adecfb8 100644
--- a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
+++ b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
@@ -816,32 +816,3 @@ for.body:
exit:
ret void
}
-
-define i32 @smul_with_overflow_vars(i32 %a, i32 %b) {
-; CHECK-LABEL: define i32 @smul_with_overflow_vars(
-; CHECK-SAME: i32 [[A:%.*]], i32 [[B:%.*]]) {
-; CHECK: [[VECTOR_PH:.*:]]
-; CHECK: [[VECTOR_BODY:.*:]]
-; CHECK: [[TMP0:%.*]] = call { <2 x i32>, <2 x i1> } @llvm.smul.with.overflow.v2i32(<2 x i32> [[BROADCAST_SPLAT2:%.*]], <2 x i32> [[BROADCAST_SPLAT:%.*]])
-; CHECK: [[TMP1:%.*]] = extractvalue { <2 x i32>, <2 x i1> } [[TMP0]], 1
-; CHECK: [[MIDDLE_BLOCK:.*:]]
-; CHECK: [[EXIT:.*:]]
-; CHECK: [[EXIT1:.*:]]
-;
-entry:
- br label %loop
-
-loop:
- %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
- %acc = phi i32 [ 0, %entry ], [ %or, %loop ]
- %call = tail call { i32, i1 } @llvm.smul.with.overflow.i32(i32 %a, i32 %b)
- %overflow = extractvalue { i32, i1 } %call, 1
- %ext = zext i1 %overflow to i32
- %or = or i32 %acc, %ext
- %iv.next = add i64 %iv, 1
- %done = icmp eq i64 %iv, 1
- br i1 %done, label %exit, label %loop
-
-exit:
- ret i32 %or
-}
diff --git a/llvm/test/Transforms/LoopVectorize/widen-extractvalue-struct.ll b/llvm/test/Transforms/LoopVectorize/widen-extractvalue-struct.ll
new file mode 100644
index 0000000000000..685dcd2bb707f
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/widen-extractvalue-struct.ll
@@ -0,0 +1,32 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter "(:|sincos|extract|broadcast)" --version 5
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=2 < %s -S -o - | FileCheck %s
+
+define float @sincos_invariant_arg(float %x) {
+; CHECK-LABEL: define float @sincos_invariant_arg(
+; CHECK-SAME: float [[X:%.*]]) {
+; CHECK: [[ENTRY:.*:]]
+; CHECK: [[VECTOR_PH:.*:]]
+; CHECK: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x float> poison, float [[X]], i64 0
+; CHECK: [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x float> [[BROADCAST_SPLATINSERT]], <2 x float> poison, <2 x i32> zeroinitializer
+; CHECK: [[TMP0:%.*]] = call { <2 x float>, <2 x float> } @llvm.sincos.v2f32(<2 x float> [[BROADCAST_SPLAT]])
+; CHECK: [[TMP1:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP0]], 0
+; CHECK: [[VECTOR_BODY:.*:]]
+; CHECK: [[MIDDLE_BLOCK:.*:]]
+; CHECK: [[EXIT:.*:]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc = phi float [ 0.0, %entry ], [ %sum, %loop ]
+ %call = tail call { float, float } @llvm.sincos.f32(float %x)
+ %sin_val = extractvalue { float, float } %call, 0
+ %sum = fadd float %acc, %sin_val
+ %iv.next = add i64 %iv, 1
+ %done = icmp eq i64 %iv, 1
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret float %sum
+}
More information about the llvm-commits
mailing list