[llvm] [VPlan] Don't replicate extractvalues that would require extracting lanes from a struct (PR #219941)

Luke Lau via llvm-commits llvm-commits at lists.llvm.org
Mon Aug 31 03:47:11 PDT 2026


https://github.com/lukel97 created https://github.com/llvm/llvm-project/pull/219941

- **[RISCV] Add further constraints on vsetvli intrinsics range**
- **Pass full simplifyquery, add a test for a non-constant avl <= vlmax**
- **Add arg comment**
- **[VPlan] Don't replicate extractvalues that would require extracting lanes from a struct**


>From f29494ceedb8047d5d10804122336878960eee20 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Fri, 28 Aug 2026 16:38:21 +0800
Subject: [PATCH 1/4] [RISCV] Add further constraints on vsetvli intrinsics
 range
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

The spec also mandates:

- `ceil(AVL / 2) ≤ vl ≤ VLMAX if AVL < (2 * VLMAX)`
- `vl = VLMAX if AVL ≥ (2 * VLMAX)`

This also reworks the constraints to be based on ConstantRange so `vl = AVL if AVL ≤ VLMAX` is no longer restricted to constants.
---
 .../Target/RISCV/RISCVTargetTransformInfo.cpp | 40 +++++++++++++------
 .../InstCombine/RISCV/riscv-vsetvli-range.ll  | 19 +++++----
 2 files changed, 39 insertions(+), 20 deletions(-)

diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index f51e47895c4bb..55129d1e87dfc 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -3786,21 +3786,35 @@ RISCVTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
     // only the VLMAX upper bound is sound.
     ConstantRange VLRange = VLMAXRange;
     if (HasAVL) {
-      APInt MaxVL = VLMAXRange.getUnsignedMax();
+      // vl ≤ VLMAX
+      VLRange =
+          ConstantRange::makeAllowedICmpRegion(CmpInst::ICMP_ULE, VLMAXRange);
+
       Value *AVL = II.getArgOperand(0);
+      ConstantRange AVLRange =
+          computeConstantRangeIncludingKnownBits(AVL, false, DL);
+
+      // vl = AVL if AVL ≤ VLMAX
+      if (AVLRange.icmp(CmpInst::ICMP_ULE, VLMAXRange))
+        return IC.replaceInstUsesWith(II, AVL);
+
+      // vl ≤ AVL
+      VLRange = VLRange.umin(AVLRange.getUnsignedMax());
+
       // vl > 0 if AVL > 0
-      APInt MinVL = APInt(BitWidth, isKnownNonZero(AVL, DL) ? 1 : 0);
-      if (auto *AVLC = dyn_cast<ConstantInt>(AVL)) {
-        const APInt &C = AVLC->getValue();
-        // A constant AVL not exceeding the smallest possible VLMAX means vl is
-        // exactly AVL, so replace the intrinsic with that constant.
-        if (C.ule(VLMAXRange.getUnsignedMin()))
-          return IC.replaceInstUsesWith(II, ConstantInt::get(II.getType(), C));
-        VLRange =
-            ConstantRange::getNonEmpty(MinVL, APIntOps::umin(C, MaxVL) + 1);
-      } else {
-        VLRange = ConstantRange::getNonEmpty(MinVL, MaxVL + 1);
-      }
+      if (AVLRange.icmp(CmpInst::ICMP_UGT, APInt::getZero(BitWidth)))
+        VLRange = VLRange.umax(APInt(BitWidth, 1));
+
+      // vl = VLMAX if AVL ≥ (2 * VLMAX)
+      ConstantRange TwoVLMAX = VLMAXRange.multiply(APInt(BitWidth, 2));
+      if (AVLRange.icmp(CmpInst::ICMP_UGE, TwoVLMAX))
+        VLRange = VLRange.intersectWith(VLMAXRange);
+
+      // ceil(AVL / 2) ≤ vl ≤ VLMAX if AVL < (2 * VLMAX)
+      if (AVLRange.icmp(CmpInst::ICMP_ULT, TwoVLMAX))
+        VLRange = VLRange.umax(APIntOps::RoundingUDiv(AVLRange.getUnsignedMin(),
+                                                      APInt(BitWidth, 2),
+                                                      APInt::Rounding::UP));
     }
 
     ConstantRange OldRange =
diff --git a/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
index 837ae6a9eb702..63e6c40cbf4b4 100644
--- a/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
+++ b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
@@ -34,11 +34,11 @@ define i64 @vsetvli_const_avl_at_min128() {
 }
 
 ; AVL == 20 sits in (MinVLMAX, ...) at VLEN128, so vl is only known to be
-; [0, 20]; at VLEN512 it is still below VLMAX (64), so vl == 20 folds.
+; [10, 20]; at VLEN512 it is still below VLMAX (64), so vl == 20 folds.
 define i64 @vsetvli_const_avl_mid() {
 ; VLEN128-LABEL: define i64 @vsetvli_const_avl_mid(
 ; VLEN128-SAME: ) #[[ATTR0]] {
-; VLEN128-NEXT:    [[VL:%.*]] = call range(i64 1, 21) i64 @llvm.riscv.vsetvli.i64(i64 20, i64 0, i64 0)
+; VLEN128-NEXT:    [[VL:%.*]] = call range(i64 10, 21) i64 @llvm.riscv.vsetvli.i64(i64 20, i64 0, i64 0)
 ; VLEN128-NEXT:    ret i64 [[VL]]
 ;
 ; VLEN512-LABEL: define i64 @vsetvli_const_avl_mid(
@@ -50,12 +50,17 @@ define i64 @vsetvli_const_avl_mid() {
 }
 
 ; AVL far above the largest VLMAX (8192): vl is capped by VLMAX, giving the
-; VLEN-independent range [0, 8192].
+; VLEN-independent range [MinVLMAX, 8192].
 define i64 @vsetvli_const_avl_above_max() {
-; CHECK-LABEL: define i64 @vsetvli_const_avl_above_max(
-; CHECK-SAME: ) #[[ATTR0]] {
-; CHECK-NEXT:    [[VL:%.*]] = call range(i64 1, 8193) i64 @llvm.riscv.vsetvli.i64(i64 100000, i64 0, i64 0)
-; CHECK-NEXT:    ret i64 [[VL]]
+; VLEN128-LABEL: define i64 @vsetvli_const_avl_above_max(
+; VLEN128-SAME: ) #[[ATTR0]] {
+; VLEN128-NEXT:    [[VL:%.*]] = call range(i64 16, 8193) i64 @llvm.riscv.vsetvli.i64(i64 100000, i64 0, i64 0)
+; VLEN128-NEXT:    ret i64 [[VL]]
+;
+; VLEN512-LABEL: define i64 @vsetvli_const_avl_above_max(
+; VLEN512-SAME: ) #[[ATTR0]] {
+; VLEN512-NEXT:    [[VL:%.*]] = call range(i64 64, 8193) i64 @llvm.riscv.vsetvli.i64(i64 100000, i64 0, i64 0)
+; VLEN512-NEXT:    ret i64 [[VL]]
 ;
   %vl = call i64 @llvm.riscv.vsetvli.i64(i64 100000, i64 0, i64 0)
   ret i64 %vl

>From b2aa3232cd18df332dda92a200a2f68f0cf6aaf6 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Fri, 28 Aug 2026 17:35:09 +0800
Subject: [PATCH 2/4] Pass full simplifyquery, add a test for a non-constant
 avl <= vlmax

---
 llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp |  4 ++--
 .../InstCombine/RISCV/riscv-vsetvli-range.ll       | 14 +++++++++++++-
 2 files changed, 15 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 55129d1e87dfc..902cf875de1df 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -3791,8 +3791,8 @@ RISCVTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
           ConstantRange::makeAllowedICmpRegion(CmpInst::ICMP_ULE, VLMAXRange);
 
       Value *AVL = II.getArgOperand(0);
-      ConstantRange AVLRange =
-          computeConstantRangeIncludingKnownBits(AVL, false, DL);
+      ConstantRange AVLRange = computeConstantRangeIncludingKnownBits(
+          AVL, false, IC.getSimplifyQuery().getWithInstruction(&II));
 
       // vl = AVL if AVL ≤ VLMAX
       if (AVLRange.icmp(CmpInst::ICMP_ULE, VLMAXRange))
diff --git a/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
index 63e6c40cbf4b4..f74ec414077ab 100644
--- a/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
+++ b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
@@ -129,7 +129,7 @@ define i1 @vsetvli_runtime_avl_gt_max_folds(i64 %avl) {
   ret i1 %c
 }
 
-; vl known > 0, so range must be > 0
+; avl known > 0, so range must be > 0
 define i64 @vsetvli_vl_known_nonzero(i64 %x) {
 ; CHECK-LABEL: define i64 @vsetvli_vl_known_nonzero(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
@@ -141,3 +141,15 @@ define i64 @vsetvli_vl_known_nonzero(i64 %x) {
   %vl = call i64 @llvm.riscv.vsetvli.i64(i64 %avl, i64 0, i64 0)
   ret i64 %vl
 }
+
+; avl known <= vlmax, so vl = avl
+define i64 @vsetvli_avl_known_lt_vlmax(i64 %x) {
+; CHECK-LABEL: define i64 @vsetvli_avl_known_lt_vlmax(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[VL:%.*]] = and i64 [[X]], 7
+; CHECK-NEXT:    ret i64 [[VL]]
+;
+  %avl = and i64 %x, 7
+  %vl = call i64 @llvm.riscv.vsetvli.i64(i64 %avl, i64 0, i64 0)
+  ret i64 %vl
+}

>From 118e79c6ddbfecfbc9580de8b942b4c012b88874 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Mon, 31 Aug 2026 11:22:31 +0800
Subject: [PATCH 3/4] Add arg comment

---
 llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp | 3 ++-
 1 file changed, 2 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 902cf875de1df..cd14ff3d2dded 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -3792,7 +3792,8 @@ RISCVTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
 
       Value *AVL = II.getArgOperand(0);
       ConstantRange AVLRange = computeConstantRangeIncludingKnownBits(
-          AVL, false, IC.getSimplifyQuery().getWithInstruction(&II));
+          AVL, /*ForSigned=*/false,
+          IC.getSimplifyQuery().getWithInstruction(&II));
 
       // vl = AVL if AVL ≤ VLMAX
       if (AVLRange.icmp(CmpInst::ICMP_ULE, VLMAXRange))

>From fe2d33b051eab6a390729ef6da97d6f2cdfdec49 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Mon, 31 Aug 2026 17:27:51 +0800
Subject: [PATCH 4/4] [VPlan] Don't replicate extractvalues that would require
 extracting lanes from a struct

In the added test cases, we have an extractvalue in a replicate region:

    vector.ph:
      WIDEN-INTRINSIC ir<%sincos> = call llvm.sincos(ir<0.000000e+00>)

    ...

    pred.store.if:
      EMIT vp<%3> = extractelement ir<%sincos>, ir<0>
      CLONE ir<%cos> = extractvalue vp<%3>
      CLONE store ir<%cos>, ir<%cos_dst>

However the operand

a) doesn't generate per lane
b) isn't defined in the same replicate region

so replicateRegionByVF tries to extract a lane from a struct, which doesn't work. This fixes it by choosing to widen the extractvalue earlier on when we know we'll end up trying to extract lanes otherwise.
---
 .../Transforms/Vectorize/LoopVectorize.cpp    |   8 +
 .../multiple-result-intrinsics.ll             |  49 +++++
 .../LoopVectorize/struct-return-replicate.ll  | 175 ++++++++++++++++++
 3 files changed, 232 insertions(+)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 1f153afd6bedd..0678f74c07c3e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -6431,6 +6431,14 @@ VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
                        VPI->getOpcode()) &&
          "Should have been handled prior to this!");
 
+  // We can only replicate an extractvalue if its operand generates per lane in
+  // the same replicate region.
+  if (VPI->getOpcode() == Instruction::ExtractValue)
+    if (VPRecipeBase *OpR = VPI->getOperand(0)->getDefiningRecipe())
+      if (!vputils::doesGeneratePerAllLanes(OpR) ||
+          OpR->getParent() != VPI->getParent())
+        return tryToWiden(VPI);
+
   if (!shouldWiden(Instr, Range))
     return nullptr;
 
diff --git a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
index 8c204f7bb4e32..8a102523bf914 100644
--- a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
+++ b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
@@ -816,3 +816,52 @@ for.body:
 exit:
   ret void
 }
+
+define void @sincos_predicated_store(i1 %c, double %x, ptr noalias %dst, ptr noalias %cos_dst) {
+; CHECK-LABEL: define void @sincos_predicated_store(
+; CHECK-SAME: i1 [[C:%.*]], double [[X:%.*]], ptr noalias [[DST:%.*]], ptr noalias [[COS_DST:%.*]]) {
+; CHECK:  [[ENTRY:.*:]]
+; CHECK:  [[VECTOR_PH:.*]]:
+; CHECK:    [[TMP0:%.*]] = call { <2 x double>, <2 x double> } @llvm.sincos.v2f64(<2 x double> [[BROADCAST_SPLAT:%.*]])
+; CHECK:    [[TMP2:%.*]] = extractvalue { <2 x double>, <2 x double> } [[TMP0]], 1
+; CHECK:  [[VECTOR_BODY:.*:]]
+; CHECK:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE2:.*]] ]
+; CHECK:    store <2 x double> zeroinitializer, ptr [[TMP3:%.*]], align 8
+; CHECK:    br i1 [[TMP1:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; CHECK:  [[PRED_STORE_IF]]:
+; CHECK:    [[TMP4:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
+; CHECK:    store double [[TMP4]], ptr [[COS_DST]], align 8
+; CHECK:    br label %[[PRED_STORE_CONTINUE]]
+; CHECK:  [[PRED_STORE_CONTINUE]]:
+; CHECK:    br i1 [[TMP1]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2]]
+; CHECK:  [[PRED_STORE_IF1]]:
+; CHECK:    [[TMP5:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK:    store double [[TMP5]], ptr [[COS_DST]], align 8
+; CHECK:    br label %[[PRED_STORE_CONTINUE2]]
+; CHECK:  [[PRED_STORE_CONTINUE2]]:
+; CHECK:  [[MIDDLE_BLOCK:.*:]]
+; CHECK:  [[EXIT:.*:]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %iv.next, %latch ], [ 0, %entry ]
+  %sincos = tail call { double, double } @llvm.sincos.f64(double %x)
+  %gep = getelementptr double, ptr %dst, i64 %iv
+  store double 0.0, ptr %gep
+  br i1 %c, label %latch, label %if
+
+if:
+  %cos = extractvalue { double, double } %sincos, 1
+  store double %cos, ptr %cos_dst
+  br label %latch
+
+latch:
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, 1024
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/struct-return-replicate.ll b/llvm/test/Transforms/LoopVectorize/struct-return-replicate.ll
index 92da4ca663969..c02b82db45514 100644
--- a/llvm/test/Transforms/LoopVectorize/struct-return-replicate.ll
+++ b/llvm/test/Transforms/LoopVectorize/struct-return-replicate.ll
@@ -656,6 +656,181 @@ exit:
   ret void
 }
 
+define void @struct_return_predicated_extractvalue(i1 %c, ptr noalias %p, ptr noalias %q) {
+; VF4-LABEL: define void @struct_return_predicated_extractvalue(
+; VF4-SAME: i1 [[C:%.*]], ptr noalias [[P:%.*]], ptr noalias [[Q:%.*]]) {
+; VF4-NEXT:  [[ENTRY:.*:]]
+; VF4-NEXT:    br label %[[VECTOR_PH:.*]]
+; VF4:       [[VECTOR_PH]]:
+; VF4-NEXT:    [[TMP0:%.*]] = xor i1 [[C]], true
+; VF4-NEXT:    br label %[[VECTOR_BODY:.*]]
+; VF4:       [[VECTOR_BODY]]:
+; VF4-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE6:.*]] ]
+; VF4-NEXT:    [[TMP1:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF4-NEXT:    [[TMP2:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF4-NEXT:    [[TMP3:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF4-NEXT:    [[TMP4:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF4-NEXT:    [[TMP5:%.*]] = extractvalue { float, float } [[TMP1]], 0
+; VF4-NEXT:    [[TMP6:%.*]] = insertelement <4 x float> poison, float [[TMP5]], i64 0
+; VF4-NEXT:    [[TMP7:%.*]] = insertvalue { <4 x float>, <4 x float> } poison, <4 x float> [[TMP6]], 0
+; VF4-NEXT:    [[TMP8:%.*]] = extractvalue { float, float } [[TMP1]], 1
+; VF4-NEXT:    [[TMP9:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP7]], 1
+; VF4-NEXT:    [[TMP10:%.*]] = insertelement <4 x float> [[TMP9]], float [[TMP8]], i64 0
+; VF4-NEXT:    [[TMP11:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP7]], <4 x float> [[TMP10]], 1
+; VF4-NEXT:    [[TMP12:%.*]] = extractvalue { float, float } [[TMP2]], 0
+; VF4-NEXT:    [[TMP13:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP11]], 0
+; VF4-NEXT:    [[TMP14:%.*]] = insertelement <4 x float> [[TMP13]], float [[TMP12]], i64 1
+; VF4-NEXT:    [[TMP15:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP11]], <4 x float> [[TMP14]], 0
+; VF4-NEXT:    [[TMP16:%.*]] = extractvalue { float, float } [[TMP2]], 1
+; VF4-NEXT:    [[TMP17:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP15]], 1
+; VF4-NEXT:    [[TMP18:%.*]] = insertelement <4 x float> [[TMP17]], float [[TMP16]], i64 1
+; VF4-NEXT:    [[TMP19:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP15]], <4 x float> [[TMP18]], 1
+; VF4-NEXT:    [[TMP20:%.*]] = extractvalue { float, float } [[TMP3]], 0
+; VF4-NEXT:    [[TMP21:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP19]], 0
+; VF4-NEXT:    [[TMP22:%.*]] = insertelement <4 x float> [[TMP21]], float [[TMP20]], i64 2
+; VF4-NEXT:    [[TMP23:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP19]], <4 x float> [[TMP22]], 0
+; VF4-NEXT:    [[TMP24:%.*]] = extractvalue { float, float } [[TMP3]], 1
+; VF4-NEXT:    [[TMP25:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP23]], 1
+; VF4-NEXT:    [[TMP26:%.*]] = insertelement <4 x float> [[TMP25]], float [[TMP24]], i64 2
+; VF4-NEXT:    [[TMP27:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP23]], <4 x float> [[TMP26]], 1
+; VF4-NEXT:    [[TMP28:%.*]] = extractvalue { float, float } [[TMP4]], 0
+; VF4-NEXT:    [[TMP29:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP27]], 0
+; VF4-NEXT:    [[TMP30:%.*]] = insertelement <4 x float> [[TMP29]], float [[TMP28]], i64 3
+; VF4-NEXT:    [[TMP31:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP27]], <4 x float> [[TMP30]], 0
+; VF4-NEXT:    [[TMP32:%.*]] = extractvalue { float, float } [[TMP4]], 1
+; VF4-NEXT:    [[TMP33:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP31]], 1
+; VF4-NEXT:    [[TMP34:%.*]] = insertelement <4 x float> [[TMP33]], float [[TMP32]], i64 3
+; VF4-NEXT:    [[TMP35:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP31]], <4 x float> [[TMP34]], 1
+; VF4-NEXT:    store float 0.000000e+00, ptr [[P]], align 4
+; VF4-NEXT:    [[TMP36:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP35]], 1
+; VF4-NEXT:    br i1 [[TMP0]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF4:       [[PRED_STORE_IF]]:
+; VF4-NEXT:    [[TMP37:%.*]] = extractelement <4 x float> [[TMP36]], i64 0
+; VF4-NEXT:    store float [[TMP37]], ptr [[Q]], align 4
+; VF4-NEXT:    br label %[[PRED_STORE_CONTINUE]]
+; VF4:       [[PRED_STORE_CONTINUE]]:
+; VF4-NEXT:    br i1 [[TMP0]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
+; VF4:       [[PRED_STORE_IF1]]:
+; VF4-NEXT:    [[TMP38:%.*]] = extractelement <4 x float> [[TMP36]], i64 1
+; VF4-NEXT:    store float [[TMP38]], ptr [[Q]], align 4
+; VF4-NEXT:    br label %[[PRED_STORE_CONTINUE2]]
+; VF4:       [[PRED_STORE_CONTINUE2]]:
+; VF4-NEXT:    br i1 [[TMP0]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
+; VF4:       [[PRED_STORE_IF3]]:
+; VF4-NEXT:    [[TMP39:%.*]] = extractelement <4 x float> [[TMP36]], i64 2
+; VF4-NEXT:    store float [[TMP39]], ptr [[Q]], align 4
+; VF4-NEXT:    br label %[[PRED_STORE_CONTINUE4]]
+; VF4:       [[PRED_STORE_CONTINUE4]]:
+; VF4-NEXT:    br i1 [[TMP0]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6]]
+; VF4:       [[PRED_STORE_IF5]]:
+; VF4-NEXT:    [[TMP40:%.*]] = extractelement <4 x float> [[TMP36]], i64 3
+; VF4-NEXT:    store float [[TMP40]], ptr [[Q]], align 4
+; VF4-NEXT:    br label %[[PRED_STORE_CONTINUE6]]
+; VF4:       [[PRED_STORE_CONTINUE6]]:
+; VF4-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; VF4-NEXT:    [[TMP41:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; VF4-NEXT:    br i1 [[TMP41]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; VF4:       [[MIDDLE_BLOCK]]:
+;
+; VF2IC2-LABEL: define void @struct_return_predicated_extractvalue(
+; VF2IC2-SAME: i1 [[C:%.*]], ptr noalias [[P:%.*]], ptr noalias [[Q:%.*]]) {
+; VF2IC2-NEXT:  [[ENTRY:.*:]]
+; VF2IC2-NEXT:    br label %[[VECTOR_PH:.*]]
+; VF2IC2:       [[VECTOR_PH]]:
+; VF2IC2-NEXT:    [[TMP0:%.*]] = xor i1 [[C]], true
+; VF2IC2-NEXT:    br label %[[VECTOR_BODY:.*]]
+; VF2IC2:       [[VECTOR_BODY]]:
+; VF2IC2-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE6:.*]] ]
+; VF2IC2-NEXT:    [[TMP1:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF2IC2-NEXT:    [[TMP2:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF2IC2-NEXT:    [[TMP3:%.*]] = extractvalue { float, float } [[TMP1]], 0
+; VF2IC2-NEXT:    [[TMP4:%.*]] = insertelement <2 x float> poison, float [[TMP3]], i64 0
+; VF2IC2-NEXT:    [[TMP5:%.*]] = insertvalue { <2 x float>, <2 x float> } poison, <2 x float> [[TMP4]], 0
+; VF2IC2-NEXT:    [[TMP6:%.*]] = extractvalue { float, float } [[TMP1]], 1
+; VF2IC2-NEXT:    [[TMP7:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP5]], 1
+; VF2IC2-NEXT:    [[TMP8:%.*]] = insertelement <2 x float> [[TMP7]], float [[TMP6]], i64 0
+; VF2IC2-NEXT:    [[TMP9:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP5]], <2 x float> [[TMP8]], 1
+; VF2IC2-NEXT:    [[TMP10:%.*]] = extractvalue { float, float } [[TMP2]], 0
+; VF2IC2-NEXT:    [[TMP11:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP9]], 0
+; VF2IC2-NEXT:    [[TMP12:%.*]] = insertelement <2 x float> [[TMP11]], float [[TMP10]], i64 1
+; VF2IC2-NEXT:    [[TMP13:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP9]], <2 x float> [[TMP12]], 0
+; VF2IC2-NEXT:    [[TMP14:%.*]] = extractvalue { float, float } [[TMP2]], 1
+; VF2IC2-NEXT:    [[TMP15:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP13]], 1
+; VF2IC2-NEXT:    [[TMP16:%.*]] = insertelement <2 x float> [[TMP15]], float [[TMP14]], i64 1
+; VF2IC2-NEXT:    [[TMP17:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP13]], <2 x float> [[TMP16]], 1
+; VF2IC2-NEXT:    [[TMP18:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF2IC2-NEXT:    [[TMP19:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF2IC2-NEXT:    [[TMP20:%.*]] = extractvalue { float, float } [[TMP18]], 0
+; VF2IC2-NEXT:    [[TMP21:%.*]] = insertelement <2 x float> poison, float [[TMP20]], i64 0
+; VF2IC2-NEXT:    [[TMP22:%.*]] = insertvalue { <2 x float>, <2 x float> } poison, <2 x float> [[TMP21]], 0
+; VF2IC2-NEXT:    [[TMP23:%.*]] = extractvalue { float, float } [[TMP18]], 1
+; VF2IC2-NEXT:    [[TMP24:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP22]], 1
+; VF2IC2-NEXT:    [[TMP25:%.*]] = insertelement <2 x float> [[TMP24]], float [[TMP23]], i64 0
+; VF2IC2-NEXT:    [[TMP26:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP22]], <2 x float> [[TMP25]], 1
+; VF2IC2-NEXT:    [[TMP27:%.*]] = extractvalue { float, float } [[TMP19]], 0
+; VF2IC2-NEXT:    [[TMP28:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP26]], 0
+; VF2IC2-NEXT:    [[TMP29:%.*]] = insertelement <2 x float> [[TMP28]], float [[TMP27]], i64 1
+; VF2IC2-NEXT:    [[TMP30:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP26]], <2 x float> [[TMP29]], 0
+; VF2IC2-NEXT:    [[TMP31:%.*]] = extractvalue { float, float } [[TMP19]], 1
+; VF2IC2-NEXT:    [[TMP32:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP30]], 1
+; VF2IC2-NEXT:    [[TMP33:%.*]] = insertelement <2 x float> [[TMP32]], float [[TMP31]], i64 1
+; VF2IC2-NEXT:    [[TMP34:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP30]], <2 x float> [[TMP33]], 1
+; VF2IC2-NEXT:    store float 0.000000e+00, ptr [[P]], align 4
+; VF2IC2-NEXT:    [[TMP35:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP17]], 1
+; VF2IC2-NEXT:    [[TMP36:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP34]], 1
+; VF2IC2-NEXT:    br i1 [[TMP0]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF2IC2:       [[PRED_STORE_IF]]:
+; VF2IC2-NEXT:    [[TMP37:%.*]] = extractelement <2 x float> [[TMP35]], i64 0
+; VF2IC2-NEXT:    store float [[TMP37]], ptr [[Q]], align 4
+; VF2IC2-NEXT:    br label %[[PRED_STORE_CONTINUE]]
+; VF2IC2:       [[PRED_STORE_CONTINUE]]:
+; VF2IC2-NEXT:    br i1 [[TMP0]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
+; VF2IC2:       [[PRED_STORE_IF1]]:
+; VF2IC2-NEXT:    [[TMP38:%.*]] = extractelement <2 x float> [[TMP35]], i64 1
+; VF2IC2-NEXT:    store float [[TMP38]], ptr [[Q]], align 4
+; VF2IC2-NEXT:    br label %[[PRED_STORE_CONTINUE2]]
+; VF2IC2:       [[PRED_STORE_CONTINUE2]]:
+; VF2IC2-NEXT:    br i1 [[TMP0]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
+; VF2IC2:       [[PRED_STORE_IF3]]:
+; VF2IC2-NEXT:    [[TMP39:%.*]] = extractelement <2 x float> [[TMP36]], i64 0
+; VF2IC2-NEXT:    store float [[TMP39]], ptr [[Q]], align 4
+; VF2IC2-NEXT:    br label %[[PRED_STORE_CONTINUE4]]
+; VF2IC2:       [[PRED_STORE_CONTINUE4]]:
+; VF2IC2-NEXT:    br i1 [[TMP0]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6]]
+; VF2IC2:       [[PRED_STORE_IF5]]:
+; VF2IC2-NEXT:    [[TMP40:%.*]] = extractelement <2 x float> [[TMP36]], i64 1
+; VF2IC2-NEXT:    store float [[TMP40]], ptr [[Q]], align 4
+; VF2IC2-NEXT:    br label %[[PRED_STORE_CONTINUE6]]
+; VF2IC2:       [[PRED_STORE_CONTINUE6]]:
+; VF2IC2-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; VF2IC2-NEXT:    [[TMP41:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; VF2IC2-NEXT:    br i1 [[TMP41]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; VF2IC2:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %iv.next, %latch ], [ 0, %entry ]
+  %call = tail call { float, float } @fn2(float 0.0) #3
+  %gep = getelementptr float, ptr %p, i64 %iv
+  store float 0.0, ptr %p
+  br i1 %c, label %latch, label %if
+
+if:
+  %val = extractvalue { float, float } %call, 1
+  store float %val, ptr %q
+  br label %latch
+
+latch:
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, 1024
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  ret void
+}
+
 declare { i64 } @fn1(float)
 declare { float, float } @fn2(float)
 declare { i32, i32, i32 } @fn3(i32)



More information about the llvm-commits mailing list