[llvm] [VPlan] Don't replicate extractvalues that would require extracting lanes from a struct (PR #219941)
Luke Lau via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 31 03:47:11 PDT 2026
https://github.com/lukel97 created https://github.com/llvm/llvm-project/pull/219941
- **[RISCV] Add further constraints on vsetvli intrinsics range**
- **Pass full simplifyquery, add a test for a non-constant avl <= vlmax**
- **Add arg comment**
- **[VPlan] Don't replicate extractvalues that would require extracting lanes from a struct**
>From f29494ceedb8047d5d10804122336878960eee20 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Fri, 28 Aug 2026 16:38:21 +0800
Subject: [PATCH 1/4] [RISCV] Add further constraints on vsetvli intrinsics
range
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The spec also mandates:
- `ceil(AVL / 2) ≤ vl ≤ VLMAX if AVL < (2 * VLMAX)`
- `vl = VLMAX if AVL ≥ (2 * VLMAX)`
This also reworks the constraints to be based on ConstantRange so `vl = AVL if AVL ≤ VLMAX` is no longer restricted to constants.
---
.../Target/RISCV/RISCVTargetTransformInfo.cpp | 40 +++++++++++++------
.../InstCombine/RISCV/riscv-vsetvli-range.ll | 19 +++++----
2 files changed, 39 insertions(+), 20 deletions(-)
diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index f51e47895c4bb..55129d1e87dfc 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -3786,21 +3786,35 @@ RISCVTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
// only the VLMAX upper bound is sound.
ConstantRange VLRange = VLMAXRange;
if (HasAVL) {
- APInt MaxVL = VLMAXRange.getUnsignedMax();
+ // vl ≤ VLMAX
+ VLRange =
+ ConstantRange::makeAllowedICmpRegion(CmpInst::ICMP_ULE, VLMAXRange);
+
Value *AVL = II.getArgOperand(0);
+ ConstantRange AVLRange =
+ computeConstantRangeIncludingKnownBits(AVL, false, DL);
+
+ // vl = AVL if AVL ≤ VLMAX
+ if (AVLRange.icmp(CmpInst::ICMP_ULE, VLMAXRange))
+ return IC.replaceInstUsesWith(II, AVL);
+
+ // vl ≤ AVL
+ VLRange = VLRange.umin(AVLRange.getUnsignedMax());
+
// vl > 0 if AVL > 0
- APInt MinVL = APInt(BitWidth, isKnownNonZero(AVL, DL) ? 1 : 0);
- if (auto *AVLC = dyn_cast<ConstantInt>(AVL)) {
- const APInt &C = AVLC->getValue();
- // A constant AVL not exceeding the smallest possible VLMAX means vl is
- // exactly AVL, so replace the intrinsic with that constant.
- if (C.ule(VLMAXRange.getUnsignedMin()))
- return IC.replaceInstUsesWith(II, ConstantInt::get(II.getType(), C));
- VLRange =
- ConstantRange::getNonEmpty(MinVL, APIntOps::umin(C, MaxVL) + 1);
- } else {
- VLRange = ConstantRange::getNonEmpty(MinVL, MaxVL + 1);
- }
+ if (AVLRange.icmp(CmpInst::ICMP_UGT, APInt::getZero(BitWidth)))
+ VLRange = VLRange.umax(APInt(BitWidth, 1));
+
+ // vl = VLMAX if AVL ≥ (2 * VLMAX)
+ ConstantRange TwoVLMAX = VLMAXRange.multiply(APInt(BitWidth, 2));
+ if (AVLRange.icmp(CmpInst::ICMP_UGE, TwoVLMAX))
+ VLRange = VLRange.intersectWith(VLMAXRange);
+
+ // ceil(AVL / 2) ≤ vl ≤ VLMAX if AVL < (2 * VLMAX)
+ if (AVLRange.icmp(CmpInst::ICMP_ULT, TwoVLMAX))
+ VLRange = VLRange.umax(APIntOps::RoundingUDiv(AVLRange.getUnsignedMin(),
+ APInt(BitWidth, 2),
+ APInt::Rounding::UP));
}
ConstantRange OldRange =
diff --git a/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
index 837ae6a9eb702..63e6c40cbf4b4 100644
--- a/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
+++ b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
@@ -34,11 +34,11 @@ define i64 @vsetvli_const_avl_at_min128() {
}
; AVL == 20 sits in (MinVLMAX, ...) at VLEN128, so vl is only known to be
-; [0, 20]; at VLEN512 it is still below VLMAX (64), so vl == 20 folds.
+; [10, 20]; at VLEN512 it is still below VLMAX (64), so vl == 20 folds.
define i64 @vsetvli_const_avl_mid() {
; VLEN128-LABEL: define i64 @vsetvli_const_avl_mid(
; VLEN128-SAME: ) #[[ATTR0]] {
-; VLEN128-NEXT: [[VL:%.*]] = call range(i64 1, 21) i64 @llvm.riscv.vsetvli.i64(i64 20, i64 0, i64 0)
+; VLEN128-NEXT: [[VL:%.*]] = call range(i64 10, 21) i64 @llvm.riscv.vsetvli.i64(i64 20, i64 0, i64 0)
; VLEN128-NEXT: ret i64 [[VL]]
;
; VLEN512-LABEL: define i64 @vsetvli_const_avl_mid(
@@ -50,12 +50,17 @@ define i64 @vsetvli_const_avl_mid() {
}
; AVL far above the largest VLMAX (8192): vl is capped by VLMAX, giving the
-; VLEN-independent range [0, 8192].
+; VLEN-independent range [MinVLMAX, 8192].
define i64 @vsetvli_const_avl_above_max() {
-; CHECK-LABEL: define i64 @vsetvli_const_avl_above_max(
-; CHECK-SAME: ) #[[ATTR0]] {
-; CHECK-NEXT: [[VL:%.*]] = call range(i64 1, 8193) i64 @llvm.riscv.vsetvli.i64(i64 100000, i64 0, i64 0)
-; CHECK-NEXT: ret i64 [[VL]]
+; VLEN128-LABEL: define i64 @vsetvli_const_avl_above_max(
+; VLEN128-SAME: ) #[[ATTR0]] {
+; VLEN128-NEXT: [[VL:%.*]] = call range(i64 16, 8193) i64 @llvm.riscv.vsetvli.i64(i64 100000, i64 0, i64 0)
+; VLEN128-NEXT: ret i64 [[VL]]
+;
+; VLEN512-LABEL: define i64 @vsetvli_const_avl_above_max(
+; VLEN512-SAME: ) #[[ATTR0]] {
+; VLEN512-NEXT: [[VL:%.*]] = call range(i64 64, 8193) i64 @llvm.riscv.vsetvli.i64(i64 100000, i64 0, i64 0)
+; VLEN512-NEXT: ret i64 [[VL]]
;
%vl = call i64 @llvm.riscv.vsetvli.i64(i64 100000, i64 0, i64 0)
ret i64 %vl
>From b2aa3232cd18df332dda92a200a2f68f0cf6aaf6 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Fri, 28 Aug 2026 17:35:09 +0800
Subject: [PATCH 2/4] Pass full simplifyquery, add a test for a non-constant
avl <= vlmax
---
llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp | 4 ++--
.../InstCombine/RISCV/riscv-vsetvli-range.ll | 14 +++++++++++++-
2 files changed, 15 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 55129d1e87dfc..902cf875de1df 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -3791,8 +3791,8 @@ RISCVTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
ConstantRange::makeAllowedICmpRegion(CmpInst::ICMP_ULE, VLMAXRange);
Value *AVL = II.getArgOperand(0);
- ConstantRange AVLRange =
- computeConstantRangeIncludingKnownBits(AVL, false, DL);
+ ConstantRange AVLRange = computeConstantRangeIncludingKnownBits(
+ AVL, false, IC.getSimplifyQuery().getWithInstruction(&II));
// vl = AVL if AVL ≤ VLMAX
if (AVLRange.icmp(CmpInst::ICMP_ULE, VLMAXRange))
diff --git a/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
index 63e6c40cbf4b4..f74ec414077ab 100644
--- a/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
+++ b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
@@ -129,7 +129,7 @@ define i1 @vsetvli_runtime_avl_gt_max_folds(i64 %avl) {
ret i1 %c
}
-; vl known > 0, so range must be > 0
+; avl known > 0, so range must be > 0
define i64 @vsetvli_vl_known_nonzero(i64 %x) {
; CHECK-LABEL: define i64 @vsetvli_vl_known_nonzero(
; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
@@ -141,3 +141,15 @@ define i64 @vsetvli_vl_known_nonzero(i64 %x) {
%vl = call i64 @llvm.riscv.vsetvli.i64(i64 %avl, i64 0, i64 0)
ret i64 %vl
}
+
+; avl known <= vlmax, so vl = avl
+define i64 @vsetvli_avl_known_lt_vlmax(i64 %x) {
+; CHECK-LABEL: define i64 @vsetvli_avl_known_lt_vlmax(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[VL:%.*]] = and i64 [[X]], 7
+; CHECK-NEXT: ret i64 [[VL]]
+;
+ %avl = and i64 %x, 7
+ %vl = call i64 @llvm.riscv.vsetvli.i64(i64 %avl, i64 0, i64 0)
+ ret i64 %vl
+}
>From 118e79c6ddbfecfbc9580de8b942b4c012b88874 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Mon, 31 Aug 2026 11:22:31 +0800
Subject: [PATCH 3/4] Add arg comment
---
llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp | 3 ++-
1 file changed, 2 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 902cf875de1df..cd14ff3d2dded 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -3792,7 +3792,8 @@ RISCVTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
Value *AVL = II.getArgOperand(0);
ConstantRange AVLRange = computeConstantRangeIncludingKnownBits(
- AVL, false, IC.getSimplifyQuery().getWithInstruction(&II));
+ AVL, /*ForSigned=*/false,
+ IC.getSimplifyQuery().getWithInstruction(&II));
// vl = AVL if AVL ≤ VLMAX
if (AVLRange.icmp(CmpInst::ICMP_ULE, VLMAXRange))
>From fe2d33b051eab6a390729ef6da97d6f2cdfdec49 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Mon, 31 Aug 2026 17:27:51 +0800
Subject: [PATCH 4/4] [VPlan] Don't replicate extractvalues that would require
extracting lanes from a struct
In the added test cases, we have an extractvalue in a replicate region:
vector.ph:
WIDEN-INTRINSIC ir<%sincos> = call llvm.sincos(ir<0.000000e+00>)
...
pred.store.if:
EMIT vp<%3> = extractelement ir<%sincos>, ir<0>
CLONE ir<%cos> = extractvalue vp<%3>
CLONE store ir<%cos>, ir<%cos_dst>
However the operand
a) doesn't generate per lane
b) isn't defined in the same replicate region
so replicateRegionByVF tries to extract a lane from a struct, which doesn't work. This fixes it by choosing to widen the extractvalue earlier on when we know we'll end up trying to extract lanes otherwise.
---
.../Transforms/Vectorize/LoopVectorize.cpp | 8 +
.../multiple-result-intrinsics.ll | 49 +++++
.../LoopVectorize/struct-return-replicate.ll | 175 ++++++++++++++++++
3 files changed, 232 insertions(+)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 1f153afd6bedd..0678f74c07c3e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -6431,6 +6431,14 @@ VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
VPI->getOpcode()) &&
"Should have been handled prior to this!");
+ // We can only replicate an extractvalue if its operand generates per lane in
+ // the same replicate region.
+ if (VPI->getOpcode() == Instruction::ExtractValue)
+ if (VPRecipeBase *OpR = VPI->getOperand(0)->getDefiningRecipe())
+ if (!vputils::doesGeneratePerAllLanes(OpR) ||
+ OpR->getParent() != VPI->getParent())
+ return tryToWiden(VPI);
+
if (!shouldWiden(Instr, Range))
return nullptr;
diff --git a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
index 8c204f7bb4e32..8a102523bf914 100644
--- a/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
+++ b/llvm/test/Transforms/LoopVectorize/multiple-result-intrinsics.ll
@@ -816,3 +816,52 @@ for.body:
exit:
ret void
}
+
+define void @sincos_predicated_store(i1 %c, double %x, ptr noalias %dst, ptr noalias %cos_dst) {
+; CHECK-LABEL: define void @sincos_predicated_store(
+; CHECK-SAME: i1 [[C:%.*]], double [[X:%.*]], ptr noalias [[DST:%.*]], ptr noalias [[COS_DST:%.*]]) {
+; CHECK: [[ENTRY:.*:]]
+; CHECK: [[VECTOR_PH:.*]]:
+; CHECK: [[TMP0:%.*]] = call { <2 x double>, <2 x double> } @llvm.sincos.v2f64(<2 x double> [[BROADCAST_SPLAT:%.*]])
+; CHECK: [[TMP2:%.*]] = extractvalue { <2 x double>, <2 x double> } [[TMP0]], 1
+; CHECK: [[VECTOR_BODY:.*:]]
+; CHECK: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE2:.*]] ]
+; CHECK: store <2 x double> zeroinitializer, ptr [[TMP3:%.*]], align 8
+; CHECK: br i1 [[TMP1:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; CHECK: [[PRED_STORE_IF]]:
+; CHECK: [[TMP4:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
+; CHECK: store double [[TMP4]], ptr [[COS_DST]], align 8
+; CHECK: br label %[[PRED_STORE_CONTINUE]]
+; CHECK: [[PRED_STORE_CONTINUE]]:
+; CHECK: br i1 [[TMP1]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2]]
+; CHECK: [[PRED_STORE_IF1]]:
+; CHECK: [[TMP5:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK: store double [[TMP5]], ptr [[COS_DST]], align 8
+; CHECK: br label %[[PRED_STORE_CONTINUE2]]
+; CHECK: [[PRED_STORE_CONTINUE2]]:
+; CHECK: [[MIDDLE_BLOCK:.*:]]
+; CHECK: [[EXIT:.*:]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ %iv.next, %latch ], [ 0, %entry ]
+ %sincos = tail call { double, double } @llvm.sincos.f64(double %x)
+ %gep = getelementptr double, ptr %dst, i64 %iv
+ store double 0.0, ptr %gep
+ br i1 %c, label %latch, label %if
+
+if:
+ %cos = extractvalue { double, double } %sincos, 1
+ store double %cos, ptr %cos_dst
+ br label %latch
+
+latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/struct-return-replicate.ll b/llvm/test/Transforms/LoopVectorize/struct-return-replicate.ll
index 92da4ca663969..c02b82db45514 100644
--- a/llvm/test/Transforms/LoopVectorize/struct-return-replicate.ll
+++ b/llvm/test/Transforms/LoopVectorize/struct-return-replicate.ll
@@ -656,6 +656,181 @@ exit:
ret void
}
+define void @struct_return_predicated_extractvalue(i1 %c, ptr noalias %p, ptr noalias %q) {
+; VF4-LABEL: define void @struct_return_predicated_extractvalue(
+; VF4-SAME: i1 [[C:%.*]], ptr noalias [[P:%.*]], ptr noalias [[Q:%.*]]) {
+; VF4-NEXT: [[ENTRY:.*:]]
+; VF4-NEXT: br label %[[VECTOR_PH:.*]]
+; VF4: [[VECTOR_PH]]:
+; VF4-NEXT: [[TMP0:%.*]] = xor i1 [[C]], true
+; VF4-NEXT: br label %[[VECTOR_BODY:.*]]
+; VF4: [[VECTOR_BODY]]:
+; VF4-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE6:.*]] ]
+; VF4-NEXT: [[TMP1:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF4-NEXT: [[TMP2:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF4-NEXT: [[TMP3:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF4-NEXT: [[TMP4:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF4-NEXT: [[TMP5:%.*]] = extractvalue { float, float } [[TMP1]], 0
+; VF4-NEXT: [[TMP6:%.*]] = insertelement <4 x float> poison, float [[TMP5]], i64 0
+; VF4-NEXT: [[TMP7:%.*]] = insertvalue { <4 x float>, <4 x float> } poison, <4 x float> [[TMP6]], 0
+; VF4-NEXT: [[TMP8:%.*]] = extractvalue { float, float } [[TMP1]], 1
+; VF4-NEXT: [[TMP9:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP7]], 1
+; VF4-NEXT: [[TMP10:%.*]] = insertelement <4 x float> [[TMP9]], float [[TMP8]], i64 0
+; VF4-NEXT: [[TMP11:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP7]], <4 x float> [[TMP10]], 1
+; VF4-NEXT: [[TMP12:%.*]] = extractvalue { float, float } [[TMP2]], 0
+; VF4-NEXT: [[TMP13:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP11]], 0
+; VF4-NEXT: [[TMP14:%.*]] = insertelement <4 x float> [[TMP13]], float [[TMP12]], i64 1
+; VF4-NEXT: [[TMP15:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP11]], <4 x float> [[TMP14]], 0
+; VF4-NEXT: [[TMP16:%.*]] = extractvalue { float, float } [[TMP2]], 1
+; VF4-NEXT: [[TMP17:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP15]], 1
+; VF4-NEXT: [[TMP18:%.*]] = insertelement <4 x float> [[TMP17]], float [[TMP16]], i64 1
+; VF4-NEXT: [[TMP19:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP15]], <4 x float> [[TMP18]], 1
+; VF4-NEXT: [[TMP20:%.*]] = extractvalue { float, float } [[TMP3]], 0
+; VF4-NEXT: [[TMP21:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP19]], 0
+; VF4-NEXT: [[TMP22:%.*]] = insertelement <4 x float> [[TMP21]], float [[TMP20]], i64 2
+; VF4-NEXT: [[TMP23:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP19]], <4 x float> [[TMP22]], 0
+; VF4-NEXT: [[TMP24:%.*]] = extractvalue { float, float } [[TMP3]], 1
+; VF4-NEXT: [[TMP25:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP23]], 1
+; VF4-NEXT: [[TMP26:%.*]] = insertelement <4 x float> [[TMP25]], float [[TMP24]], i64 2
+; VF4-NEXT: [[TMP27:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP23]], <4 x float> [[TMP26]], 1
+; VF4-NEXT: [[TMP28:%.*]] = extractvalue { float, float } [[TMP4]], 0
+; VF4-NEXT: [[TMP29:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP27]], 0
+; VF4-NEXT: [[TMP30:%.*]] = insertelement <4 x float> [[TMP29]], float [[TMP28]], i64 3
+; VF4-NEXT: [[TMP31:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP27]], <4 x float> [[TMP30]], 0
+; VF4-NEXT: [[TMP32:%.*]] = extractvalue { float, float } [[TMP4]], 1
+; VF4-NEXT: [[TMP33:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP31]], 1
+; VF4-NEXT: [[TMP34:%.*]] = insertelement <4 x float> [[TMP33]], float [[TMP32]], i64 3
+; VF4-NEXT: [[TMP35:%.*]] = insertvalue { <4 x float>, <4 x float> } [[TMP31]], <4 x float> [[TMP34]], 1
+; VF4-NEXT: store float 0.000000e+00, ptr [[P]], align 4
+; VF4-NEXT: [[TMP36:%.*]] = extractvalue { <4 x float>, <4 x float> } [[TMP35]], 1
+; VF4-NEXT: br i1 [[TMP0]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF4: [[PRED_STORE_IF]]:
+; VF4-NEXT: [[TMP37:%.*]] = extractelement <4 x float> [[TMP36]], i64 0
+; VF4-NEXT: store float [[TMP37]], ptr [[Q]], align 4
+; VF4-NEXT: br label %[[PRED_STORE_CONTINUE]]
+; VF4: [[PRED_STORE_CONTINUE]]:
+; VF4-NEXT: br i1 [[TMP0]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
+; VF4: [[PRED_STORE_IF1]]:
+; VF4-NEXT: [[TMP38:%.*]] = extractelement <4 x float> [[TMP36]], i64 1
+; VF4-NEXT: store float [[TMP38]], ptr [[Q]], align 4
+; VF4-NEXT: br label %[[PRED_STORE_CONTINUE2]]
+; VF4: [[PRED_STORE_CONTINUE2]]:
+; VF4-NEXT: br i1 [[TMP0]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
+; VF4: [[PRED_STORE_IF3]]:
+; VF4-NEXT: [[TMP39:%.*]] = extractelement <4 x float> [[TMP36]], i64 2
+; VF4-NEXT: store float [[TMP39]], ptr [[Q]], align 4
+; VF4-NEXT: br label %[[PRED_STORE_CONTINUE4]]
+; VF4: [[PRED_STORE_CONTINUE4]]:
+; VF4-NEXT: br i1 [[TMP0]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6]]
+; VF4: [[PRED_STORE_IF5]]:
+; VF4-NEXT: [[TMP40:%.*]] = extractelement <4 x float> [[TMP36]], i64 3
+; VF4-NEXT: store float [[TMP40]], ptr [[Q]], align 4
+; VF4-NEXT: br label %[[PRED_STORE_CONTINUE6]]
+; VF4: [[PRED_STORE_CONTINUE6]]:
+; VF4-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; VF4-NEXT: [[TMP41:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; VF4-NEXT: br i1 [[TMP41]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; VF4: [[MIDDLE_BLOCK]]:
+;
+; VF2IC2-LABEL: define void @struct_return_predicated_extractvalue(
+; VF2IC2-SAME: i1 [[C:%.*]], ptr noalias [[P:%.*]], ptr noalias [[Q:%.*]]) {
+; VF2IC2-NEXT: [[ENTRY:.*:]]
+; VF2IC2-NEXT: br label %[[VECTOR_PH:.*]]
+; VF2IC2: [[VECTOR_PH]]:
+; VF2IC2-NEXT: [[TMP0:%.*]] = xor i1 [[C]], true
+; VF2IC2-NEXT: br label %[[VECTOR_BODY:.*]]
+; VF2IC2: [[VECTOR_BODY]]:
+; VF2IC2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE6:.*]] ]
+; VF2IC2-NEXT: [[TMP1:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF2IC2-NEXT: [[TMP2:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF2IC2-NEXT: [[TMP3:%.*]] = extractvalue { float, float } [[TMP1]], 0
+; VF2IC2-NEXT: [[TMP4:%.*]] = insertelement <2 x float> poison, float [[TMP3]], i64 0
+; VF2IC2-NEXT: [[TMP5:%.*]] = insertvalue { <2 x float>, <2 x float> } poison, <2 x float> [[TMP4]], 0
+; VF2IC2-NEXT: [[TMP6:%.*]] = extractvalue { float, float } [[TMP1]], 1
+; VF2IC2-NEXT: [[TMP7:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP5]], 1
+; VF2IC2-NEXT: [[TMP8:%.*]] = insertelement <2 x float> [[TMP7]], float [[TMP6]], i64 0
+; VF2IC2-NEXT: [[TMP9:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP5]], <2 x float> [[TMP8]], 1
+; VF2IC2-NEXT: [[TMP10:%.*]] = extractvalue { float, float } [[TMP2]], 0
+; VF2IC2-NEXT: [[TMP11:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP9]], 0
+; VF2IC2-NEXT: [[TMP12:%.*]] = insertelement <2 x float> [[TMP11]], float [[TMP10]], i64 1
+; VF2IC2-NEXT: [[TMP13:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP9]], <2 x float> [[TMP12]], 0
+; VF2IC2-NEXT: [[TMP14:%.*]] = extractvalue { float, float } [[TMP2]], 1
+; VF2IC2-NEXT: [[TMP15:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP13]], 1
+; VF2IC2-NEXT: [[TMP16:%.*]] = insertelement <2 x float> [[TMP15]], float [[TMP14]], i64 1
+; VF2IC2-NEXT: [[TMP17:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP13]], <2 x float> [[TMP16]], 1
+; VF2IC2-NEXT: [[TMP18:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF2IC2-NEXT: [[TMP19:%.*]] = tail call { float, float } @fn2(float 0.000000e+00) #[[ATTR3]]
+; VF2IC2-NEXT: [[TMP20:%.*]] = extractvalue { float, float } [[TMP18]], 0
+; VF2IC2-NEXT: [[TMP21:%.*]] = insertelement <2 x float> poison, float [[TMP20]], i64 0
+; VF2IC2-NEXT: [[TMP22:%.*]] = insertvalue { <2 x float>, <2 x float> } poison, <2 x float> [[TMP21]], 0
+; VF2IC2-NEXT: [[TMP23:%.*]] = extractvalue { float, float } [[TMP18]], 1
+; VF2IC2-NEXT: [[TMP24:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP22]], 1
+; VF2IC2-NEXT: [[TMP25:%.*]] = insertelement <2 x float> [[TMP24]], float [[TMP23]], i64 0
+; VF2IC2-NEXT: [[TMP26:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP22]], <2 x float> [[TMP25]], 1
+; VF2IC2-NEXT: [[TMP27:%.*]] = extractvalue { float, float } [[TMP19]], 0
+; VF2IC2-NEXT: [[TMP28:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP26]], 0
+; VF2IC2-NEXT: [[TMP29:%.*]] = insertelement <2 x float> [[TMP28]], float [[TMP27]], i64 1
+; VF2IC2-NEXT: [[TMP30:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP26]], <2 x float> [[TMP29]], 0
+; VF2IC2-NEXT: [[TMP31:%.*]] = extractvalue { float, float } [[TMP19]], 1
+; VF2IC2-NEXT: [[TMP32:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP30]], 1
+; VF2IC2-NEXT: [[TMP33:%.*]] = insertelement <2 x float> [[TMP32]], float [[TMP31]], i64 1
+; VF2IC2-NEXT: [[TMP34:%.*]] = insertvalue { <2 x float>, <2 x float> } [[TMP30]], <2 x float> [[TMP33]], 1
+; VF2IC2-NEXT: store float 0.000000e+00, ptr [[P]], align 4
+; VF2IC2-NEXT: [[TMP35:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP17]], 1
+; VF2IC2-NEXT: [[TMP36:%.*]] = extractvalue { <2 x float>, <2 x float> } [[TMP34]], 1
+; VF2IC2-NEXT: br i1 [[TMP0]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF2IC2: [[PRED_STORE_IF]]:
+; VF2IC2-NEXT: [[TMP37:%.*]] = extractelement <2 x float> [[TMP35]], i64 0
+; VF2IC2-NEXT: store float [[TMP37]], ptr [[Q]], align 4
+; VF2IC2-NEXT: br label %[[PRED_STORE_CONTINUE]]
+; VF2IC2: [[PRED_STORE_CONTINUE]]:
+; VF2IC2-NEXT: br i1 [[TMP0]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
+; VF2IC2: [[PRED_STORE_IF1]]:
+; VF2IC2-NEXT: [[TMP38:%.*]] = extractelement <2 x float> [[TMP35]], i64 1
+; VF2IC2-NEXT: store float [[TMP38]], ptr [[Q]], align 4
+; VF2IC2-NEXT: br label %[[PRED_STORE_CONTINUE2]]
+; VF2IC2: [[PRED_STORE_CONTINUE2]]:
+; VF2IC2-NEXT: br i1 [[TMP0]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
+; VF2IC2: [[PRED_STORE_IF3]]:
+; VF2IC2-NEXT: [[TMP39:%.*]] = extractelement <2 x float> [[TMP36]], i64 0
+; VF2IC2-NEXT: store float [[TMP39]], ptr [[Q]], align 4
+; VF2IC2-NEXT: br label %[[PRED_STORE_CONTINUE4]]
+; VF2IC2: [[PRED_STORE_CONTINUE4]]:
+; VF2IC2-NEXT: br i1 [[TMP0]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6]]
+; VF2IC2: [[PRED_STORE_IF5]]:
+; VF2IC2-NEXT: [[TMP40:%.*]] = extractelement <2 x float> [[TMP36]], i64 1
+; VF2IC2-NEXT: store float [[TMP40]], ptr [[Q]], align 4
+; VF2IC2-NEXT: br label %[[PRED_STORE_CONTINUE6]]
+; VF2IC2: [[PRED_STORE_CONTINUE6]]:
+; VF2IC2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; VF2IC2-NEXT: [[TMP41:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; VF2IC2-NEXT: br i1 [[TMP41]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; VF2IC2: [[MIDDLE_BLOCK]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ %iv.next, %latch ], [ 0, %entry ]
+ %call = tail call { float, float } @fn2(float 0.0) #3
+ %gep = getelementptr float, ptr %p, i64 %iv
+ store float 0.0, ptr %p
+ br i1 %c, label %latch, label %if
+
+if:
+ %val = extractvalue { float, float } %call, 1
+ store float %val, ptr %q
+ br label %latch
+
+latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
+
declare { i64 } @fn1(float)
declare { float, float } @fn2(float)
declare { i32, i32, i32 } @fn3(i32)
More information about the llvm-commits
mailing list