[llvm] [LoopVectorize] Fix crash in VPWidenRecipe (PR #177357)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Fri Aug 28 04:43:36 PDT 2026


https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/177357

>From f9de623de2dcb31b644d0853cd594455eeab84fd Mon Sep 17 00:00:00 2001
From: Manasij Mukherjee <manasijm at nvidia.com>
Date: Thu, 22 Jan 2026 05:24:23 -0800
Subject: [PATCH 1/3] [LoopVectorize] Fix crash in VPWidenRecipe

VPWidenRecipe crashed when the ExtractValue operand was a struct constant.
Use the extracted value instead.

Fixes #176663
---
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp |  4 ++-
 .../multiple-result-intrinsics.ll             | 27 +++++++++++++++++++
 2 files changed, 30 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 584b430c8e78e..59209c80d1be1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -2273,9 +2273,11 @@ void VPWidenRecipe::execute(VPTransformState &State) {
   }
   case Instruction::ExtractValue: {
     assert(getNumOperands() == 2 && "expected single level extractvalue");
-    Value *Op = State.get(getOperand(0));
+    Value *Op = State.get(getOperand(0), /*NeedsScalar=*/true);
     Value *Extract = Builder.CreateExtractValue(
         Op, cast<VPConstantInt>(getOperand(1))->getZExtValue());
+    if (!Extract->getType()->isVectorTy())
+      Extract = Builder.CreateVectorSplat(State.VF, Extract);
     State.set(this, Extract);
     break;
   }
diff --git a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
index f64d43adecfb8..a1650416760cc 100644
--- a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
+++ b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
@@ -816,3 +816,30 @@ for.body:
 exit:
   ret void
 }
+
+define i32 @smul_with_overflow_constant_args() {
+;
+; CHECK-LABEL: define i32 @smul_with_overflow_constant_args() {
+; CHECK:  [[ENTRY:.*:]]
+; CHECK:  [[VECTOR_PH:.*:]]
+; CHECK:  [[VECTOR_BODY:.*:]]
+; CHECK:  [[MIDDLE_BLOCK:.*:]]
+; CHECK:  [[EXIT:.*:]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i32 [ 0, %entry ], [ %or, %loop ]
+  %call = tail call { i32, i1 } @llvm.smul.with.overflow.i32(i32 0, i32 0)
+  %overflow = extractvalue { i32, i1 } %call, 1
+  %ext = zext i1 %overflow to i32
+  %or = or i32 %acc, %ext
+  %iv.next = add i64 %iv, 1
+  %done = icmp eq i64 %iv, 1
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret i32 %or
+}

>From 08f565a8440cb07affae818c98103f21b49f0257 Mon Sep 17 00:00:00 2001
From: Manasij Mukherjee <manasijm at nvidia.com>
Date: Thu, 22 Jan 2026 16:11:01 -0800
Subject: [PATCH 2/3] Better test, checks

---
 .../LoopVectorize/multiple-result-intrinsics.ll      | 12 +++++++-----
 1 file changed, 7 insertions(+), 5 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
index a1650416760cc..cc0e553bac511 100644
--- a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
+++ b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
@@ -817,14 +817,16 @@ exit:
   ret void
 }
 
-define i32 @smul_with_overflow_constant_args() {
-;
-; CHECK-LABEL: define i32 @smul_with_overflow_constant_args() {
-; CHECK:  [[ENTRY:.*:]]
+define i32 @smul_with_overflow_vars(i32 %a, i32 %b) {
+; CHECK-LABEL: define i32 @smul_with_overflow_vars(
+; CHECK-SAME: i32 [[A:%.*]], i32 [[B:%.*]]) {
 ; CHECK:  [[VECTOR_PH:.*:]]
 ; CHECK:  [[VECTOR_BODY:.*:]]
+; CHECK:    [[TMP0:%.*]] = call { <2 x i32>, <2 x i1> } @llvm.smul.with.overflow.v2i32(<2 x i32> [[BROADCAST_SPLAT2:%.*]], <2 x i32> [[BROADCAST_SPLAT:%.*]])
+; CHECK:    [[TMP1:%.*]] = extractvalue { <2 x i32>, <2 x i1> } [[TMP0]], 1
 ; CHECK:  [[MIDDLE_BLOCK:.*:]]
 ; CHECK:  [[EXIT:.*:]]
+; CHECK:  [[EXIT1:.*:]]
 ;
 entry:
   br label %loop
@@ -832,7 +834,7 @@ entry:
 loop:
   %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
   %acc = phi i32 [ 0, %entry ], [ %or, %loop ]
-  %call = tail call { i32, i1 } @llvm.smul.with.overflow.i32(i32 0, i32 0)
+  %call = tail call { i32, i1 } @llvm.smul.with.overflow.i32(i32 %a, i32 %b)
   %overflow = extractvalue { i32, i1 } %call, 1
   %ext = zext i1 %overflow to i32
   %or = or i32 %acc, %ext

>From 99c47298084f890e0eb9bcb07fa8bf51091e6d74 Mon Sep 17 00:00:00 2001
From: Manasij Mukherjee <manasijm at nvidia.com>
Date: Mon, 9 Feb 2026 13:41:00 -0800
Subject: [PATCH 3/3] Add comment, split out test, update script filters out
 less context

---
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp |  3 ++
 .../multiple-result-intrinsics.ll             | 29 -----------------
 .../widen-extractvalue-struct.ll              | 32 +++++++++++++++++++
 3 files changed, 35 insertions(+), 29 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/widen-extractvalue-struct.ll

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 59209c80d1be1..5bce025d1d39e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -2273,6 +2273,9 @@ void VPWidenRecipe::execute(VPTransformState &State) {
   }
   case Instruction::ExtractValue: {
     assert(getNumOperands() == 2 && "expected single level extractvalue");
+    // Fetch the struct operand as a scalar so that extractvalue is valid
+    // on the struct type, regardless of whether it is a widened struct of
+    // vectors or a scalar struct constant folded by VPlan simplification.
     Value *Op = State.get(getOperand(0), /*NeedsScalar=*/true);
     Value *Extract = Builder.CreateExtractValue(
         Op, cast<VPConstantInt>(getOperand(1))->getZExtValue());
diff --git a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
index cc0e553bac511..f64d43adecfb8 100644
--- a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
+++ b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
@@ -816,32 +816,3 @@ for.body:
 exit:
   ret void
 }
-
-define i32 @smul_with_overflow_vars(i32 %a, i32 %b) {
-; CHECK-LABEL: define i32 @smul_with_overflow_vars(
-; CHECK-SAME: i32 [[A:%.*]], i32 [[B:%.*]]) {
-; CHECK:  [[VECTOR_PH:.*:]]
-; CHECK:  [[VECTOR_BODY:.*:]]
-; CHECK:    [[TMP0:%.*]] = call { <2 x i32>, <2 x i1> } @llvm.smul.with.overflow.v2i32(<2 x i32> [[BROADCAST_SPLAT2:%.*]], <2 x i32> [[BROADCAST_SPLAT:%.*]])
-; CHECK:    [[TMP1:%.*]] = extractvalue { <2 x i32>, <2 x i1> } [[TMP0]], 1
-; CHECK:  [[MIDDLE_BLOCK:.*:]]
-; CHECK:  [[EXIT:.*:]]
-; CHECK:  [[EXIT1:.*:]]
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %acc = phi i32 [ 0, %entry ], [ %or, %loop ]
-  %call = tail call { i32, i1 } @llvm.smul.with.overflow.i32(i32 %a, i32 %b)
-  %overflow = extractvalue { i32, i1 } %call, 1
-  %ext = zext i1 %overflow to i32
-  %or = or i32 %acc, %ext
-  %iv.next = add i64 %iv, 1
-  %done = icmp eq i64 %iv, 1
-  br i1 %done, label %exit, label %loop
-
-exit:
-  ret i32 %or
-}
diff --git a/llvm/test/Transforms/LoopVectorize/widen-extractvalue-struct.ll b/llvm/test/Transforms/LoopVectorize/widen-extractvalue-struct.ll
new file mode 100644
index 0000000000000..685dcd2bb707f
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/widen-extractvalue-struct.ll
@@ -0,0 +1,32 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter "(:|sincos|extract|broadcast)" --version 5
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=2 < %s -S -o - | FileCheck %s
+
+define float @sincos_invariant_arg(float %x) {
+; CHECK-LABEL: define float @sincos_invariant_arg(
+; CHECK-SAME: float [[X:%.*]]) {
+; CHECK:  [[ENTRY:.*:]]
+; CHECK:  [[VECTOR_PH:.*:]]
+; CHECK:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x float> poison, float [[X]], i64 0
+; CHECK:    [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x float> [[BROADCAST_SPLATINSERT]], <2 x float> poison, <2 x i32> zeroinitializer
+; CHECK:    [[TMP0:%.*]] = call { <2 x float>, <2 x float> } @llvm.sincos.v2f32(<2 x float> [[BROADCAST_SPLAT]])
+; CHECK:    [[TMP1:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP0]], 0
+; CHECK:  [[VECTOR_BODY:.*:]]
+; CHECK:  [[MIDDLE_BLOCK:.*:]]
+; CHECK:  [[EXIT:.*:]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi float [ 0.0, %entry ], [ %sum, %loop ]
+  %call = tail call { float, float } @llvm.sincos.f32(float %x)
+  %sin_val = extractvalue { float, float } %call, 0
+  %sum = fadd float %acc, %sin_val
+  %iv.next = add i64 %iv, 1
+  %done = icmp eq i64 %iv, 1
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret float %sum
+}



More information about the llvm-commits mailing list