[llvm] [VectorCombine] Fold interleave and widen chained operations (PR #224005)

Kamlesh Kumar via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 01:25:16 PDT 2026


https://github.com/kamleshbhalui updated https://github.com/llvm/llvm-project/pull/224005

>From b633894f5df14e31749ef49de620ded8d2a71db3 Mon Sep 17 00:00:00 2001
From: Kamlesh Kumar <kamlesh.kumar at arm.com>
Date: Wed, 16 Sep 2026 13:44:25 +0100
Subject: [PATCH 1/3] added test

---
 .../deinterleave-interleave-tree.ll           | 541 ++++++++++++++++++
 1 file changed, 541 insertions(+)
 create mode 100644 llvm/test/Transforms/VectorCombine/deinterleave-interleave-tree.ll

diff --git a/llvm/test/Transforms/VectorCombine/deinterleave-interleave-tree.ll b/llvm/test/Transforms/VectorCombine/deinterleave-interleave-tree.ll
new file mode 100644
index 0000000000000..35b8d5bc8ca74
--- /dev/null
+++ b/llvm/test/Transforms/VectorCombine/deinterleave-interleave-tree.ll
@@ -0,0 +1,541 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=vector-combine %s -S -o - | FileCheck %s
+
+define <8 x i16> @two_deinterleave2_mul_xor_trunc(<8 x i32> %a, <8 x i32> %b) {
+; CHECK-LABEL: define <8 x i16> @two_deinterleave2_mul_xor_trunc(
+; CHECK-SAME: <8 x i32> [[A:%.*]], <8 x i32> [[B:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 1
+; CHECK-NEXT:    [[DB:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[B]])
+; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 0
+; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 1
+; CHECK-NEXT:    [[M0:%.*]] = mul <4 x i32> [[A0]], [[B0]]
+; CHECK-NEXT:    [[M1:%.*]] = mul <4 x i32> [[A1]], [[B1]]
+; CHECK-NEXT:    [[X0:%.*]] = xor <4 x i32> [[M0]], splat (i32 7)
+; CHECK-NEXT:    [[X1:%.*]] = xor <4 x i32> [[M1]], splat (i32 7)
+; CHECK-NEXT:    [[T0:%.*]] = trunc <4 x i32> [[X0]] to <4 x i16>
+; CHECK-NEXT:    [[T1:%.*]] = trunc <4 x i32> [[X1]] to <4 x i16>
+; CHECK-NEXT:    [[R:%.*]] = call <8 x i16> @llvm.vector.interleave2.v8i16(<4 x i16> [[T0]], <4 x i16> [[T1]])
+; CHECK-NEXT:    ret <8 x i16> [[R]]
+;
+  %da = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %a)
+  %a0 = extractvalue { <4 x i32>, <4 x i32> } %da, 0
+  %a1 = extractvalue { <4 x i32>, <4 x i32> } %da, 1
+  %db = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %b)
+  %b0 = extractvalue { <4 x i32>, <4 x i32> } %db, 0
+  %b1 = extractvalue { <4 x i32>, <4 x i32> } %db, 1
+  %m0 = mul <4 x i32> %a0, %b0
+  %m1 = mul <4 x i32> %a1, %b1
+  %x0 = xor <4 x i32> %m0, splat (i32 7)
+  %x1 = xor <4 x i32> %m1, splat (i32 7)
+  %t0 = trunc <4 x i32> %x0 to <4 x i16>
+  %t1 = trunc <4 x i32> %x1 to <4 x i16>
+  %r = call <8 x i16> @llvm.vector.interleave2.v8i16(<4 x i16> %t0, <4 x i16> %t1)
+  ret <8 x i16> %r
+}
+
+define <vscale x 8 x i16> @two_deinterleave2_independent_chains(<vscale x 8 x i16> %a, <vscale x 8 x i16> %b) {
+; CHECK-LABEL: define <vscale x 8 x i16> @two_deinterleave2_independent_chains(
+; CHECK-SAME: <vscale x 8 x i16> [[A:%.*]], <vscale x 8 x i16> [[B:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave2.nxv8i16(<vscale x 8 x i16> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } [[DA]], 1
+; CHECK-NEXT:    [[DB:%.*]] = call { <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave2.nxv8i16(<vscale x 8 x i16> [[B]])
+; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } [[DB]], 0
+; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } [[DB]], 1
+; CHECK-NEXT:    [[AA0:%.*]] = add <vscale x 4 x i16> [[A0]], splat (i16 1)
+; CHECK-NEXT:    [[AA1:%.*]] = add <vscale x 4 x i16> [[A1]], splat (i16 1)
+; CHECK-NEXT:    [[BB0:%.*]] = xor <vscale x 4 x i16> [[B0]], splat (i16 3)
+; CHECK-NEXT:    [[BB1:%.*]] = xor <vscale x 4 x i16> [[B1]], splat (i16 3)
+; CHECK-NEXT:    [[M0:%.*]] = mul <vscale x 4 x i16> [[AA0]], [[BB0]]
+; CHECK-NEXT:    [[M1:%.*]] = mul <vscale x 4 x i16> [[AA1]], [[BB1]]
+; CHECK-NEXT:    [[S0:%.*]] = sub <vscale x 4 x i16> [[M0]], splat (i16 5)
+; CHECK-NEXT:    [[S1:%.*]] = sub <vscale x 4 x i16> [[M1]], splat (i16 5)
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 8 x i16> @llvm.vector.interleave2.nxv8i16(<vscale x 4 x i16> [[S0]], <vscale x 4 x i16> [[S1]])
+; CHECK-NEXT:    ret <vscale x 8 x i16> [[R]]
+;
+  %da = call { <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave2.nxv8i16(<vscale x 8 x i16> %a)
+  %a0 = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } %da, 0
+  %a1 = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } %da, 1
+  %db = call { <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave2.nxv8i16(<vscale x 8 x i16> %b)
+  %b0 = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } %db, 0
+  %b1 = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } %db, 1
+  %aa0 = add <vscale x 4 x i16> %a0, splat (i16 1)
+  %aa1 = add <vscale x 4 x i16> %a1, splat (i16 1)
+  %bb0 = xor <vscale x 4 x i16> %b0, splat (i16 3)
+  %bb1 = xor <vscale x 4 x i16> %b1, splat (i16 3)
+  %m0 = mul <vscale x 4 x i16> %aa0, %bb0
+  %m1 = mul <vscale x 4 x i16> %aa1, %bb1
+  %s0 = sub <vscale x 4 x i16> %m0, splat (i16 5)
+  %s1 = sub <vscale x 4 x i16> %m1, splat (i16 5)
+  %r = call <vscale x 8 x i16> @llvm.vector.interleave2.nxv8i16(<vscale x 4 x i16> %s0, <vscale x 4 x i16> %s1)
+  ret <vscale x 8 x i16> %r
+}
+
+define <vscale x 6 x i32> @two_deinterleave3_add(<vscale x 6 x i32> %a, <vscale x 6 x i32> %b) {
+; CHECK-LABEL: define <vscale x 6 x i32> @two_deinterleave3_add(
+; CHECK-SAME: <vscale x 6 x i32> [[A:%.*]], <vscale x 6 x i32> [[B:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave3.nxv6i32(<vscale x 6 x i32> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DA]], 1
+; CHECK-NEXT:    [[A2:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DA]], 2
+; CHECK-NEXT:    [[DB:%.*]] = call { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave3.nxv6i32(<vscale x 6 x i32> [[B]])
+; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DB]], 0
+; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DB]], 1
+; CHECK-NEXT:    [[B2:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DB]], 2
+; CHECK-NEXT:    [[S0:%.*]] = add <vscale x 2 x i32> [[A0]], [[B0]]
+; CHECK-NEXT:    [[S1:%.*]] = add <vscale x 2 x i32> [[A1]], [[B1]]
+; CHECK-NEXT:    [[S2:%.*]] = add <vscale x 2 x i32> [[A2]], [[B2]]
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 6 x i32> @llvm.vector.interleave3.nxv6i32(<vscale x 2 x i32> [[S0]], <vscale x 2 x i32> [[S1]], <vscale x 2 x i32> [[S2]])
+; CHECK-NEXT:    ret <vscale x 6 x i32> [[R]]
+;
+  %da = call { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave3.nxv6i32(<vscale x 6 x i32> %a)
+  %a0 = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } %da, 0
+  %a1 = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } %da, 1
+  %a2 = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } %da, 2
+  %db = call { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave3.nxv6i32(<vscale x 6 x i32> %b)
+  %b0 = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } %db, 0
+  %b1 = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } %db, 1
+  %b2 = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } %db, 2
+  %s0 = add <vscale x 2 x i32> %a0, %b0
+  %s1 = add <vscale x 2 x i32> %a1, %b1
+  %s2 = add <vscale x 2 x i32> %a2, %b2
+  %r = call <vscale x 6 x i32> @llvm.vector.interleave3.nxv6i32(<vscale x 2 x i32> %s0, <vscale x 2 x i32> %s1, <vscale x 2 x i32> %s2)
+  ret <vscale x 6 x i32> %r
+}
+
+define <4 x i32> @negative_deinterleave2_mixed_sources(
+; CHECK-LABEL: define <4 x i32> @negative_deinterleave2_mixed_sources(
+; CHECK-SAME: <4 x i32> [[A:%.*]], <4 x i32> [[B:%.*]], <4 x i32> [[C:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 1
+; CHECK-NEXT:    [[DB:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[B]])
+; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DB]], 0
+; CHECK-NEXT:    [[DC:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[C]])
+; CHECK-NEXT:    [[C1:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DC]], 1
+; CHECK-NEXT:    [[S0:%.*]] = add <2 x i32> [[A0]], [[B0]]
+; CHECK-NEXT:    [[S1:%.*]] = add <2 x i32> [[A1]], [[C1]]
+; CHECK-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[S0]], <2 x i32> [[S1]])
+; CHECK-NEXT:    ret <4 x i32> [[R]]
+;
+  <4 x i32> %a, <4 x i32> %b, <4 x i32> %c) {
+  %da = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %a)
+  %a0 = extractvalue { <2 x i32>, <2 x i32> } %da, 0
+  %a1 = extractvalue { <2 x i32>, <2 x i32> } %da, 1
+  %db = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %b)
+  %b0 = extractvalue { <2 x i32>, <2 x i32> } %db, 0
+  %dc = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %c)
+  %c1 = extractvalue { <2 x i32>, <2 x i32> } %dc, 1
+  %s0 = add <2 x i32> %a0, %b0
+  %s1 = add <2 x i32> %a1, %c1
+  %r = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %s0, <2 x i32> %s1)
+  ret <4 x i32> %r
+}
+
+define <8 x i32> @three_deinterleave2_tree_with_splats(<8 x i32> %a, <8 x i32> %b, <8 x i32> %c) {
+; CHECK-LABEL: define <8 x i32> @three_deinterleave2_tree_with_splats(
+; CHECK-SAME: <8 x i32> [[A:%.*]], <8 x i32> [[B:%.*]], <8 x i32> [[C:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 1
+; CHECK-NEXT:    [[DB:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[B]])
+; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 0
+; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 1
+; CHECK-NEXT:    [[DC:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[C]])
+; CHECK-NEXT:    [[C0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DC]], 0
+; CHECK-NEXT:    [[C1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DC]], 1
+; CHECK-NEXT:    [[M0:%.*]] = mul <4 x i32> [[A0]], [[B0]]
+; CHECK-NEXT:    [[M1:%.*]] = mul <4 x i32> [[A1]], [[B1]]
+; CHECK-NEXT:    [[X0:%.*]] = xor <4 x i32> [[M0]], splat (i32 7)
+; CHECK-NEXT:    [[X1:%.*]] = xor <4 x i32> [[M1]], splat (i32 7)
+; CHECK-NEXT:    [[S0:%.*]] = add <4 x i32> [[X0]], [[C0]]
+; CHECK-NEXT:    [[S1:%.*]] = add <4 x i32> [[X1]], [[C1]]
+; CHECK-NEXT:    [[T0:%.*]] = shl <4 x i32> [[S0]], splat (i32 1)
+; CHECK-NEXT:    [[T1:%.*]] = shl <4 x i32> [[S1]], splat (i32 1)
+; CHECK-NEXT:    [[R:%.*]] = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> [[T0]], <4 x i32> [[T1]])
+; CHECK-NEXT:    ret <8 x i32> [[R]]
+;
+  %da = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %a)
+  %a0 = extractvalue { <4 x i32>, <4 x i32> } %da, 0
+  %a1 = extractvalue { <4 x i32>, <4 x i32> } %da, 1
+  %db = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %b)
+  %b0 = extractvalue { <4 x i32>, <4 x i32> } %db, 0
+  %b1 = extractvalue { <4 x i32>, <4 x i32> } %db, 1
+  %dc = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %c)
+  %c0 = extractvalue { <4 x i32>, <4 x i32> } %dc, 0
+  %c1 = extractvalue { <4 x i32>, <4 x i32> } %dc, 1
+  %m0 = mul <4 x i32> %a0, %b0
+  %m1 = mul <4 x i32> %a1, %b1
+  %x0 = xor <4 x i32> %m0, splat (i32 7)
+  %x1 = xor <4 x i32> %m1, splat (i32 7)
+  %s0 = add <4 x i32> %x0, %c0
+  %s1 = add <4 x i32> %x1, %c1
+  %t0 = shl <4 x i32> %s0, splat (i32 1)
+  %t1 = shl <4 x i32> %s1, splat (i32 1)
+  %r = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> %t0, <4 x i32> %t1)
+  ret <8 x i32> %r
+}
+
+define <4 x i32> @deinterleave2_extract_used_twice_in_member(<4 x i32> %a) {
+; CHECK-LABEL: define <4 x i32> @deinterleave2_extract_used_twice_in_member(
+; CHECK-SAME: <4 x i32> [[A:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 1
+; CHECK-NEXT:    [[M0:%.*]] = mul <2 x i32> [[A0]], [[A0]]
+; CHECK-NEXT:    [[M1:%.*]] = mul <2 x i32> [[A1]], [[A1]]
+; CHECK-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[M0]], <2 x i32> [[M1]])
+; CHECK-NEXT:    ret <4 x i32> [[R]]
+;
+  %da = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %a)
+  %a0 = extractvalue { <2 x i32>, <2 x i32> } %da, 0
+  %a1 = extractvalue { <2 x i32>, <2 x i32> } %da, 1
+  %m0 = mul <2 x i32> %a0, %a0
+  %m1 = mul <2 x i32> %a1, %a1
+  %r = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %m0, <2 x i32> %m1)
+  ret <4 x i32> %r
+}
+
+define <4 x i32> @negative_deinterleave2_shared_non_splat_operand(<4 x i32> %a, <2 x i32> %x) {
+; CHECK-LABEL: define <4 x i32> @negative_deinterleave2_shared_non_splat_operand(
+; CHECK-SAME: <4 x i32> [[A:%.*]], <2 x i32> [[X:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 1
+; CHECK-NEXT:    [[S0:%.*]] = add <2 x i32> [[A0]], [[X]]
+; CHECK-NEXT:    [[S1:%.*]] = add <2 x i32> [[A1]], [[X]]
+; CHECK-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[S0]], <2 x i32> [[S1]])
+; CHECK-NEXT:    ret <4 x i32> [[R]]
+;
+  %da = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %a)
+  %a0 = extractvalue { <2 x i32>, <2 x i32> } %da, 0
+  %a1 = extractvalue { <2 x i32>, <2 x i32> } %da, 1
+  %s0 = add <2 x i32> %a0, %x
+  %s1 = add <2 x i32> %a1, %x
+  %r = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %s0, <2 x i32> %s1)
+  ret <4 x i32> %r
+}
+
+define <4 x i32> @deinterleave2_extract_with_unfolded_user(<4 x i32> %a, ptr %p) {
+; CHECK-LABEL: define <4 x i32> @deinterleave2_extract_with_unfolded_user(
+; CHECK-SAME: <4 x i32> [[A:%.*]], ptr [[P:%.*]]) {
+; CHECK-NEXT:    [[D:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[A]])
+; CHECK-NEXT:    [[F0:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[D]], 0
+; CHECK-NEXT:    [[F1:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[D]], 1
+; CHECK-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[F0]], <2 x i32> [[F1]])
+; CHECK-NEXT:    store <2 x i32> [[F0]], ptr [[P]], align 8
+; CHECK-NEXT:    ret <4 x i32> [[R]]
+;
+  %d = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %a)
+  %f0 = extractvalue { <2 x i32>, <2 x i32> } %d, 0
+  %f1 = extractvalue { <2 x i32>, <2 x i32> } %d, 1
+  %r = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %f0, <2 x i32> %f1)
+  store <2 x i32> %f0, ptr %p
+  ret <4 x i32> %r
+}
+
+define <4 x i32> @deinterleave2_extracts_feed_two_interleaves(<4 x i32> %a) {
+; CHECK-LABEL: define <4 x i32> @deinterleave2_extracts_feed_two_interleaves(
+; CHECK-SAME: <4 x i32> [[A:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 1
+; CHECK-NEXT:    [[S0:%.*]] = add <2 x i32> [[A0]], splat (i32 1)
+; CHECK-NEXT:    [[S1:%.*]] = add <2 x i32> [[A1]], splat (i32 1)
+; CHECK-NEXT:    [[R1:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[S0]], <2 x i32> [[S1]])
+; CHECK-NEXT:    [[T0:%.*]] = sub <2 x i32> [[A0]], splat (i32 1)
+; CHECK-NEXT:    [[T1:%.*]] = sub <2 x i32> [[A1]], splat (i32 1)
+; CHECK-NEXT:    [[R2:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[T0]], <2 x i32> [[T1]])
+; CHECK-NEXT:    [[R:%.*]] = xor <4 x i32> [[R1]], [[R2]]
+; CHECK-NEXT:    ret <4 x i32> [[R]]
+;
+  %da = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %a)
+  %a0 = extractvalue { <2 x i32>, <2 x i32> } %da, 0
+  %a1 = extractvalue { <2 x i32>, <2 x i32> } %da, 1
+  %s0 = add <2 x i32> %a0, splat (i32 1)
+  %s1 = add <2 x i32> %a1, splat (i32 1)
+  %r1 = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %s0, <2 x i32> %s1)
+  %t0 = sub <2 x i32> %a0, splat (i32 1)
+  %t1 = sub <2 x i32> %a1, splat (i32 1)
+  %r2 = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %t0, <2 x i32> %t1)
+  %r = xor <4 x i32> %r1, %r2
+  ret <4 x i32> %r
+}
+
+define <4 x float> @deinterleave2_fabs_nnan(<4 x float> %v) {
+; CHECK-LABEL: define <4 x float> @deinterleave2_fabs_nnan(
+; CHECK-SAME: <4 x float> [[V:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = call nnan <4 x float> @llvm.fabs.v4f32(<4 x float> [[V]])
+; CHECK-NEXT:    ret <4 x float> [[R]]
+;
+  %d = call { <2 x float>, <2 x float> } @llvm.vector.deinterleave2.v4f32(<4 x float> %v)
+  %f0 = extractvalue { <2 x float>, <2 x float> } %d, 0
+  %f1 = extractvalue { <2 x float>, <2 x float> } %d, 1
+  %u0 = call nnan <2 x float> @llvm.fabs.v2f32(<2 x float> %f0)
+  %u1 = call nnan <2 x float> @llvm.fabs.v2f32(<2 x float> %f1)
+  %r = call <4 x float> @llvm.vector.interleave2.v4f32(<2 x float> %u0, <2 x float> %u1)
+  ret <4 x float> %r
+}
+
+define <4 x float> @deinterleave2_fabs_mismatched_fmf(<4 x float> %v) {
+; CHECK-LABEL: define <4 x float> @deinterleave2_fabs_mismatched_fmf(
+; CHECK-SAME: <4 x float> [[V:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = call ninf <4 x float> @llvm.fabs.v4f32(<4 x float> [[V]])
+; CHECK-NEXT:    ret <4 x float> [[R]]
+;
+  %d = call { <2 x float>, <2 x float> } @llvm.vector.deinterleave2.v4f32(<4 x float> %v)
+  %f0 = extractvalue { <2 x float>, <2 x float> } %d, 0
+  %f1 = extractvalue { <2 x float>, <2 x float> } %d, 1
+  %u0 = call nnan ninf <2 x float> @llvm.fabs.v2f32(<2 x float> %f0)
+  %u1 = call ninf nsz <2 x float> @llvm.fabs.v2f32(<2 x float> %f1)
+  %r = call <4 x float> @llvm.vector.interleave2.v4f32(<2 x float> %u0, <2 x float> %u1)
+  ret <4 x float> %r
+}
+
+define <8 x i32> @two_deinterleave2_smax(<8 x i32> %a, <8 x i32> %b) {
+; CHECK-LABEL: define <8 x i32> @two_deinterleave2_smax(
+; CHECK-SAME: <8 x i32> [[A:%.*]], <8 x i32> [[B:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 1
+; CHECK-NEXT:    [[DB:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[B]])
+; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 0
+; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 1
+; CHECK-NEXT:    [[M0:%.*]] = call <4 x i32> @llvm.smax.v4i32(<4 x i32> [[A0]], <4 x i32> [[B0]])
+; CHECK-NEXT:    [[M1:%.*]] = call <4 x i32> @llvm.smax.v4i32(<4 x i32> [[A1]], <4 x i32> [[B1]])
+; CHECK-NEXT:    [[R:%.*]] = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> [[M0]], <4 x i32> [[M1]])
+; CHECK-NEXT:    ret <8 x i32> [[R]]
+;
+  %da = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %a)
+  %a0 = extractvalue { <4 x i32>, <4 x i32> } %da, 0
+  %a1 = extractvalue { <4 x i32>, <4 x i32> } %da, 1
+  %db = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %b)
+  %b0 = extractvalue { <4 x i32>, <4 x i32> } %db, 0
+  %b1 = extractvalue { <4 x i32>, <4 x i32> } %db, 1
+  %m0 = call <4 x i32> @llvm.smax.v4i32(<4 x i32> %a0, <4 x i32> %b0)
+  %m1 = call <4 x i32> @llvm.smax.v4i32(<4 x i32> %a1, <4 x i32> %b1)
+  %r = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> %m0, <4 x i32> %m1)
+  ret <8 x i32> %r
+}
+
+define <vscale x 4 x double> @two_deinterleave2_fma_splat(<vscale x 4 x double> %a, <vscale x 4 x double> %b) {
+; CHECK-LABEL: define <vscale x 4 x double> @two_deinterleave2_fma_splat(
+; CHECK-SAME: <vscale x 4 x double> [[A:%.*]], <vscale x 4 x double> [[B:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.vector.deinterleave2.nxv4f64(<vscale x 4 x double> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } [[DA]], 1
+; CHECK-NEXT:    [[DB:%.*]] = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.vector.deinterleave2.nxv4f64(<vscale x 4 x double> [[B]])
+; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } [[DB]], 0
+; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } [[DB]], 1
+; CHECK-NEXT:    [[M0:%.*]] = call contract <vscale x 2 x double> @llvm.fma.nxv2f64(<vscale x 2 x double> [[A0]], <vscale x 2 x double> [[B0]], <vscale x 2 x double> splat (double 1.000000e+00))
+; CHECK-NEXT:    [[M1:%.*]] = call contract <vscale x 2 x double> @llvm.fma.nxv2f64(<vscale x 2 x double> [[A1]], <vscale x 2 x double> [[B1]], <vscale x 2 x double> splat (double 1.000000e+00))
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x double> @llvm.vector.interleave2.nxv4f64(<vscale x 2 x double> [[M0]], <vscale x 2 x double> [[M1]])
+; CHECK-NEXT:    ret <vscale x 4 x double> [[R]]
+;
+  %da = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.vector.deinterleave2.nxv4f64(<vscale x 4 x double> %a)
+  %a0 = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } %da, 0
+  %a1 = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } %da, 1
+  %db = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.vector.deinterleave2.nxv4f64(<vscale x 4 x double> %b)
+  %b0 = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } %db, 0
+  %b1 = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } %db, 1
+  %m0 = call contract <vscale x 2 x double> @llvm.fma.nxv2f64(<vscale x 2 x double> %a0, <vscale x 2 x double> %b0, <vscale x 2 x double> splat (double 1.0))
+  %m1 = call contract <vscale x 2 x double> @llvm.fma.nxv2f64(<vscale x 2 x double> %a1, <vscale x 2 x double> %b1, <vscale x 2 x double> splat (double 1.0))
+  %r = call <vscale x 4 x double> @llvm.vector.interleave2.nxv4f64(<vscale x 2 x double> %m0, <vscale x 2 x double> %m1)
+  ret <vscale x 4 x double> %r
+}
+
+define <8 x i16> @deinterleave2_umin_scalar_splat(<8 x i16> %v, i16 %s) {
+; CHECK-LABEL: define <8 x i16> @deinterleave2_umin_scalar_splat(
+; CHECK-SAME: <8 x i16> [[V:%.*]], i16 [[S:%.*]]) {
+; CHECK-NEXT:    [[DOTSPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[S]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT:%.*]] = shufflevector <8 x i16> [[DOTSPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[R:%.*]] = call <8 x i16> @llvm.umin.v8i16(<8 x i16> [[V]], <8 x i16> [[DOTSPLAT]])
+; CHECK-NEXT:    ret <8 x i16> [[R]]
+;
+  %s.ins = insertelement <4 x i16> poison, i16 %s, i64 0
+  %s.splat = shufflevector <4 x i16> %s.ins, <4 x i16> poison, <4 x i32> zeroinitializer
+  %d = call { <4 x i16>, <4 x i16> } @llvm.vector.deinterleave2.v8i16(<8 x i16> %v)
+  %f0 = extractvalue { <4 x i16>, <4 x i16> } %d, 0
+  %f1 = extractvalue { <4 x i16>, <4 x i16> } %d, 1
+  %u0 = call <4 x i16> @llvm.umin.v4i16(<4 x i16> %f0, <4 x i16> %s.splat)
+  %u1 = call <4 x i16> @llvm.umin.v4i16(<4 x i16> %f1, <4 x i16> %s.splat)
+  %r = call <8 x i16> @llvm.vector.interleave2.v8i16(<4 x i16> %u0, <4 x i16> %u1)
+  ret <8 x i16> %r
+}
+
+define <4 x float> @deinterleave2_powi_same_exponent(<4 x float> %v, i32 %n) {
+; CHECK-LABEL: define <4 x float> @deinterleave2_powi_same_exponent(
+; CHECK-SAME: <4 x float> [[V:%.*]], i32 [[N:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = call <4 x float> @llvm.powi.v4f32.i32(<4 x float> [[V]], i32 [[N]])
+; CHECK-NEXT:    ret <4 x float> [[R]]
+;
+  %d = call { <2 x float>, <2 x float> } @llvm.vector.deinterleave2.v4f32(<4 x float> %v)
+  %f0 = extractvalue { <2 x float>, <2 x float> } %d, 0
+  %f1 = extractvalue { <2 x float>, <2 x float> } %d, 1
+  %u0 = call <2 x float> @llvm.powi.v2f32.i32(<2 x float> %f0, i32 %n)
+  %u1 = call <2 x float> @llvm.powi.v2f32.i32(<2 x float> %f1, i32 %n)
+  %r = call <4 x float> @llvm.vector.interleave2.v4f32(<2 x float> %u0, <2 x float> %u1)
+  ret <4 x float> %r
+}
+
+define <4 x float> @negative_deinterleave2_powi_different_exponent(<4 x float> %v, i32 %n, i32 %m) {
+; CHECK-LABEL: define <4 x float> @negative_deinterleave2_powi_different_exponent(
+; CHECK-SAME: <4 x float> [[V:%.*]], i32 [[N:%.*]], i32 [[M:%.*]]) {
+; CHECK-NEXT:    [[D:%.*]] = call { <2 x float>, <2 x float> } @llvm.vector.deinterleave2.v4f32(<4 x float> [[V]])
+; CHECK-NEXT:    [[F0:%.*]] = extractvalue { <2 x float>, <2 x float> } [[D]], 0
+; CHECK-NEXT:    [[F1:%.*]] = extractvalue { <2 x float>, <2 x float> } [[D]], 1
+; CHECK-NEXT:    [[U0:%.*]] = call <2 x float> @llvm.powi.v2f32.i32(<2 x float> [[F0]], i32 [[N]])
+; CHECK-NEXT:    [[U1:%.*]] = call <2 x float> @llvm.powi.v2f32.i32(<2 x float> [[F1]], i32 [[M]])
+; CHECK-NEXT:    [[R:%.*]] = call <4 x float> @llvm.vector.interleave2.v4f32(<2 x float> [[U0]], <2 x float> [[U1]])
+; CHECK-NEXT:    ret <4 x float> [[R]]
+;
+  %d = call { <2 x float>, <2 x float> } @llvm.vector.deinterleave2.v4f32(<4 x float> %v)
+  %f0 = extractvalue { <2 x float>, <2 x float> } %d, 0
+  %f1 = extractvalue { <2 x float>, <2 x float> } %d, 1
+  %u0 = call <2 x float> @llvm.powi.v2f32.i32(<2 x float> %f0, i32 %n)
+  %u1 = call <2 x float> @llvm.powi.v2f32.i32(<2 x float> %f1, i32 %m)
+  %r = call <4 x float> @llvm.vector.interleave2.v4f32(<2 x float> %u0, <2 x float> %u1)
+  ret <4 x float> %r
+}
+
+define <8 x i8> @deinterleave2_ctlz_same_immarg(<8 x i8> %v) {
+; CHECK-LABEL: define <8 x i8> @deinterleave2_ctlz_same_immarg(
+; CHECK-SAME: <8 x i8> [[V:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = call <8 x i8> @llvm.ctlz.v8i8(<8 x i8> [[V]], i1 true)
+; CHECK-NEXT:    ret <8 x i8> [[R]]
+;
+  %d = call { <4 x i8>, <4 x i8> } @llvm.vector.deinterleave2.v8i8(<8 x i8> %v)
+  %f0 = extractvalue { <4 x i8>, <4 x i8> } %d, 0
+  %f1 = extractvalue { <4 x i8>, <4 x i8> } %d, 1
+  %u0 = call <4 x i8> @llvm.ctlz.v4i8(<4 x i8> %f0, i1 true)
+  %u1 = call <4 x i8> @llvm.ctlz.v4i8(<4 x i8> %f1, i1 true)
+  %r = call <8 x i8> @llvm.vector.interleave2.v8i8(<4 x i8> %u0, <4 x i8> %u1)
+  ret <8 x i8> %r
+}
+
+define <8 x i8> @negative_deinterleave2_ctlz_different_immarg(<8 x i8> %v) {
+; CHECK-LABEL: define <8 x i8> @negative_deinterleave2_ctlz_different_immarg(
+; CHECK-SAME: <8 x i8> [[V:%.*]]) {
+; CHECK-NEXT:    [[D:%.*]] = call { <4 x i8>, <4 x i8> } @llvm.vector.deinterleave2.v8i8(<8 x i8> [[V]])
+; CHECK-NEXT:    [[F0:%.*]] = extractvalue { <4 x i8>, <4 x i8> } [[D]], 0
+; CHECK-NEXT:    [[F1:%.*]] = extractvalue { <4 x i8>, <4 x i8> } [[D]], 1
+; CHECK-NEXT:    [[U0:%.*]] = call <4 x i8> @llvm.ctlz.v4i8(<4 x i8> [[F0]], i1 true)
+; CHECK-NEXT:    [[U1:%.*]] = call <4 x i8> @llvm.ctlz.v4i8(<4 x i8> [[F1]], i1 false)
+; CHECK-NEXT:    [[R:%.*]] = call <8 x i8> @llvm.vector.interleave2.v8i8(<4 x i8> [[U0]], <4 x i8> [[U1]])
+; CHECK-NEXT:    ret <8 x i8> [[R]]
+;
+  %d = call { <4 x i8>, <4 x i8> } @llvm.vector.deinterleave2.v8i8(<8 x i8> %v)
+  %f0 = extractvalue { <4 x i8>, <4 x i8> } %d, 0
+  %f1 = extractvalue { <4 x i8>, <4 x i8> } %d, 1
+  %u0 = call <4 x i8> @llvm.ctlz.v4i8(<4 x i8> %f0, i1 true)
+  %u1 = call <4 x i8> @llvm.ctlz.v4i8(<4 x i8> %f1, i1 false)
+  %r = call <8 x i8> @llvm.vector.interleave2.v8i8(<4 x i8> %u0, <4 x i8> %u1)
+  ret <8 x i8> %r
+}
+
+define <8 x i32> @negative_two_deinterleave2_smax_smin(<8 x i32> %a, <8 x i32> %b) {
+; CHECK-LABEL: define <8 x i32> @negative_two_deinterleave2_smax_smin(
+; CHECK-SAME: <8 x i32> [[A:%.*]], <8 x i32> [[B:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 1
+; CHECK-NEXT:    [[DB:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[B]])
+; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 0
+; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 1
+; CHECK-NEXT:    [[M0:%.*]] = call <4 x i32> @llvm.smax.v4i32(<4 x i32> [[A0]], <4 x i32> [[B0]])
+; CHECK-NEXT:    [[M1:%.*]] = call <4 x i32> @llvm.smin.v4i32(<4 x i32> [[A1]], <4 x i32> [[B1]])
+; CHECK-NEXT:    [[R:%.*]] = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> [[M0]], <4 x i32> [[M1]])
+; CHECK-NEXT:    ret <8 x i32> [[R]]
+;
+  %da = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %a)
+  %a0 = extractvalue { <4 x i32>, <4 x i32> } %da, 0
+  %a1 = extractvalue { <4 x i32>, <4 x i32> } %da, 1
+  %db = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %b)
+  %b0 = extractvalue { <4 x i32>, <4 x i32> } %db, 0
+  %b1 = extractvalue { <4 x i32>, <4 x i32> } %db, 1
+  %m0 = call <4 x i32> @llvm.smax.v4i32(<4 x i32> %a0, <4 x i32> %b0)
+  %m1 = call <4 x i32> @llvm.smin.v4i32(<4 x i32> %a1, <4 x i32> %b1)
+  %r = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> %m0, <4 x i32> %m1)
+  ret <8 x i32> %r
+}
+
+define <4 x i32> @negative_deinterleave2_vector_reverse(<4 x i32> %v) {
+; CHECK-LABEL: define <4 x i32> @negative_deinterleave2_vector_reverse(
+; CHECK-SAME: <4 x i32> [[V:%.*]]) {
+; CHECK-NEXT:    [[D:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[V]])
+; CHECK-NEXT:    [[F0:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[D]], 0
+; CHECK-NEXT:    [[F1:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[D]], 1
+; CHECK-NEXT:    [[U0:%.*]] = call <2 x i32> @llvm.vector.reverse.v2i32(<2 x i32> [[F0]])
+; CHECK-NEXT:    [[U1:%.*]] = call <2 x i32> @llvm.vector.reverse.v2i32(<2 x i32> [[F1]])
+; CHECK-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[U0]], <2 x i32> [[U1]])
+; CHECK-NEXT:    ret <4 x i32> [[R]]
+;
+  %d = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %v)
+  %f0 = extractvalue { <2 x i32>, <2 x i32> } %d, 0
+  %f1 = extractvalue { <2 x i32>, <2 x i32> } %d, 1
+  %u0 = call <2 x i32> @llvm.vector.reverse.v2i32(<2 x i32> %f0)
+  %u1 = call <2 x i32> @llvm.vector.reverse.v2i32(<2 x i32> %f1)
+  %r = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %u0, <2 x i32> %u1)
+  ret <4 x i32> %r
+}
+
+define <vscale x 12 x i16> @two_deinterleave3_abs_mul_sat(<vscale x 12 x i16> %a, <vscale x 12 x i16> %b) {
+; CHECK-LABEL: define <vscale x 12 x i16> @two_deinterleave3_abs_mul_sat(
+; CHECK-SAME: <vscale x 12 x i16> [[A:%.*]], <vscale x 12 x i16> [[B:%.*]]) {
+; CHECK-NEXT:    [[DA:%.*]] = call { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave3.nxv12i16(<vscale x 12 x i16> [[A]])
+; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DA]], 0
+; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DA]], 1
+; CHECK-NEXT:    [[A2:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DA]], 2
+; CHECK-NEXT:    [[DB:%.*]] = call { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave3.nxv12i16(<vscale x 12 x i16> [[B]])
+; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DB]], 0
+; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DB]], 1
+; CHECK-NEXT:    [[B2:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DB]], 2
+; CHECK-NEXT:    [[ABS0:%.*]] = call <vscale x 4 x i16> @llvm.abs.nxv4i16(<vscale x 4 x i16> [[A0]], i1 false)
+; CHECK-NEXT:    [[ABS1:%.*]] = call <vscale x 4 x i16> @llvm.abs.nxv4i16(<vscale x 4 x i16> [[A1]], i1 false)
+; CHECK-NEXT:    [[ABS2:%.*]] = call <vscale x 4 x i16> @llvm.abs.nxv4i16(<vscale x 4 x i16> [[A2]], i1 false)
+; CHECK-NEXT:    [[M0:%.*]] = mul nsw <vscale x 4 x i16> [[ABS0]], [[B0]]
+; CHECK-NEXT:    [[M1:%.*]] = mul nsw <vscale x 4 x i16> [[ABS1]], [[B1]]
+; CHECK-NEXT:    [[M2:%.*]] = mul nsw <vscale x 4 x i16> [[ABS2]], [[B2]]
+; CHECK-NEXT:    [[S0:%.*]] = call <vscale x 4 x i16> @llvm.sadd.sat.nxv4i16(<vscale x 4 x i16> [[M0]], <vscale x 4 x i16> splat (i16 7))
+; CHECK-NEXT:    [[S1:%.*]] = call <vscale x 4 x i16> @llvm.sadd.sat.nxv4i16(<vscale x 4 x i16> [[M1]], <vscale x 4 x i16> splat (i16 7))
+; CHECK-NEXT:    [[S2:%.*]] = call <vscale x 4 x i16> @llvm.sadd.sat.nxv4i16(<vscale x 4 x i16> [[M2]], <vscale x 4 x i16> splat (i16 7))
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 12 x i16> @llvm.vector.interleave3.nxv12i16(<vscale x 4 x i16> [[S0]], <vscale x 4 x i16> [[S1]], <vscale x 4 x i16> [[S2]])
+; CHECK-NEXT:    ret <vscale x 12 x i16> [[R]]
+;
+  %da = call { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave3.nxv12i16(<vscale x 12 x i16> %a)
+  %a0 = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } %da, 0
+  %a1 = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } %da, 1
+  %a2 = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } %da, 2
+  %db = call { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave3.nxv12i16(<vscale x 12 x i16> %b)
+  %b0 = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } %db, 0
+  %b1 = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } %db, 1
+  %b2 = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } %db, 2
+  %abs0 = call <vscale x 4 x i16> @llvm.abs.nxv4i16(<vscale x 4 x i16> %a0, i1 false)
+  %abs1 = call <vscale x 4 x i16> @llvm.abs.nxv4i16(<vscale x 4 x i16> %a1, i1 false)
+  %abs2 = call <vscale x 4 x i16> @llvm.abs.nxv4i16(<vscale x 4 x i16> %a2, i1 false)
+  %m0 = mul nsw <vscale x 4 x i16> %abs0, %b0
+  %m1 = mul nsw <vscale x 4 x i16> %abs1, %b1
+  %m2 = mul nsw <vscale x 4 x i16> %abs2, %b2
+  %s0 = call <vscale x 4 x i16> @llvm.sadd.sat.nxv4i16(<vscale x 4 x i16> %m0, <vscale x 4 x i16> splat (i16 7))
+  %s1 = call <vscale x 4 x i16> @llvm.sadd.sat.nxv4i16(<vscale x 4 x i16> %m1, <vscale x 4 x i16> splat (i16 7))
+  %s2 = call <vscale x 4 x i16> @llvm.sadd.sat.nxv4i16(<vscale x 4 x i16> %m2, <vscale x 4 x i16> splat (i16 7))
+  %r = call <vscale x 12 x i16> @llvm.vector.interleave3.nxv12i16(<vscale x 4 x i16> %s0, <vscale x 4 x i16> %s1, <vscale x 4 x i16> %s2)
+  ret <vscale x 12 x i16> %r
+}
+
+define <8 x i1> @deinterleave2_is_fpclass(<8 x float> %v) {
+; CHECK-LABEL: define <8 x i1> @deinterleave2_is_fpclass(
+; CHECK-SAME: <8 x float> [[V:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = call <8 x i1> @llvm.is.fpclass.v8f32(<8 x float> [[V]], /* (nan) */ i32 3)
+; CHECK-NEXT:    ret <8 x i1> [[R]]
+;
+  %d = call { <4 x float>, <4 x float> } @llvm.vector.deinterleave2.v8f32(<8 x float> %v)
+  %f0 = extractvalue { <4 x float>, <4 x float> } %d, 0
+  %f1 = extractvalue { <4 x float>, <4 x float> } %d, 1
+  %c0 = call <4 x i1> @llvm.is.fpclass.v4f32(<4 x float> %f0, i32 3)
+  %c1 = call <4 x i1> @llvm.is.fpclass.v4f32(<4 x float> %f1, i32 3)
+  %r = call <8 x i1> @llvm.vector.interleave2.v8i1(<4 x i1> %c0, <4 x i1> %c1)
+  ret <8 x i1> %r
+}

>From e197f38ede2402d5159216bbccd6dc4b6fcb1e85 Mon Sep 17 00:00:00 2001
From: Kamlesh Kumar <kamlesh.kumar at arm.com>
Date: Wed, 16 Sep 2026 13:45:49 +0100
Subject: [PATCH 2/3] [VectorCombine] fold interleave and widen the linked
 operation

When operands of element wise operation comes from one or more
deinterleave and terminate at interleave, then all linked operation
can be widened.
It is always profitable because it folds the interleave operation.
---
 .../Transforms/Vectorize/VectorCombine.cpp    | 405 +++++++++---------
 .../deinterleave-interleave-tree.ll           | 133 +-----
 2 files changed, 228 insertions(+), 310 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index e8c5885f9b1e3..ce62a45d5f488 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -163,7 +163,7 @@ class VectorCombine {
   bool shrinkType(Instruction &I);
   bool shrinkLoadForShuffles(Instruction &I);
   bool shrinkPhiOfShuffles(Instruction &I);
-  bool foldDeinterleaveInterleavePair(Instruction &I);
+  bool foldInterleaveOfDeinterleaveChains(Instruction &I);
 
   void replaceValue(Instruction &Old, Value &New, bool Erase = true) {
     LLVM_DEBUG(dbgs() << "VC: Replacing: " << Old << '\n');
@@ -6032,233 +6032,242 @@ bool VectorCombine::foldInsExtVectorToShuffle(Instruction &I) {
   return true;
 }
 
-/// Fold away a matched pair of vector.deinterleave/interleave intrinsics
-/// with a chain of elementwise operations on each between the
-/// deinterleave and interleave.
-///
-/// For example:
-///  ```
-///  %d = call { <2 x i16>, <2 x i16> } @deinterleave2.v4i16(<4 x i16> %v)
-///  %f0 = extractvalue { <2 x i16>, <2 x i16> } %d, 0
-///  %f1 = extractvalue { <2 x i16>, <2 x i16> } %d, 1
-///
-///  %u0 = add <2 x i16> %f0, splat (i16 3)
-///  %u1 = add <2 x i16> %f1, splat (i16 3)
-///
-///  %r = call <4 x i16> @interleave2.v4i16(<2 x i16> %u0, <2 x i16> %u1)
-///  ```
-/// Folds to:
-///  ```
-///  %r = add <4 x i16> %v, splat (i16 3)
-///  ```
-bool VectorCombine::foldDeinterleaveInterleavePair(Instruction &I) {
-  auto *Deinterleave = dyn_cast<IntrinsicInst>(&I);
-  if (!Deinterleave)
-    return false;
+static unsigned getNumDataOperands(const Instruction *Inst) {
+  if (auto *CB = dyn_cast<CallBase>(Inst))
+    return CB->arg_size(); // Exclude callee operand and bundles.
+  return Inst->getNumOperands();
+}
 
-  unsigned Factor =
-      getDeinterleaveIntrinsicFactor(Deinterleave->getIntrinsicID());
-  if (!Factor || Deinterleave->hasOperandBundles() ||
-      !Deinterleave->hasNUndroppableUses(Factor))
+/// Return true if \p Inst is an elementwise operation that can be rebuilt at a
+/// wider element count.
+static bool isSupportedElementwise(Instruction *Inst) {
+  auto *ResultTy = dyn_cast<VectorType>(Inst->getType());
+  if (!ResultTy || !isSafeToSpeculativelyExecute(Inst))
     return false;
 
-  const Intrinsic::ID ExpectedInterleaveIID =
-      Intrinsic::getInterleaveIntrinsicID(Factor);
-
-  // Collect one extract for each deinterleaved field.
-  SmallVector<Use *, 8> CurrentUses(Factor, nullptr);
-  for (Use &U : Deinterleave->uses()) {
-    if (U.getUser()->isDroppable())
-      continue;
-
-    auto *Extract = dyn_cast<ExtractValueInst>(U.getUser());
-    if (!Extract || Extract->getNumIndices() != 1)
+  if (auto *II = dyn_cast<IntrinsicInst>(Inst)) {
+    if (II->hasOperandBundles() ||
+        !isTriviallyVectorizable(II->getIntrinsicID()))
       return false;
+  } else if (!isa<BinaryOperator, UnaryOperator, CastInst, CmpInst, SelectInst,
+                  FreezeInst>(Inst)) {
+    return false;
+  }
 
-    unsigned Index = *Extract->idx_begin();
-    if (Index >= Factor || CurrentUses[Index])
+  // Reject operations that change the element-count.
+  // E.g., bitcast <vscale x 4 x i16> %v to <vscale x 8 x i8>
+  for (unsigned Op = 0, E = getNumDataOperands(Inst); Op != E; ++Op) {
+    auto *OperandTy = dyn_cast<VectorType>(Inst->getOperand(Op)->getType());
+    if (OperandTy &&
+        OperandTy->getElementCount() != ResultTy->getElementCount())
       return false;
-
-    CurrentUses[Index] = &U;
   }
 
-  using ElementwiseStep = SmallVector<Use *, 8>;
-  SmallVector<ElementwiseStep, 4> Steps;
-  IntrinsicInst *Interleave = nullptr;
-  unsigned NumVisited = 0;
+  return true;
+}
 
-  auto GetNumDataOperands = [](Instruction *Inst) {
-    if (auto *CB = dyn_cast<CallBase>(Inst))
-      return CB->arg_size(); // Exclude callee operand and bundles.
-    return Inst->getNumOperands();
+static Value *getCommonSplatValue(ArrayRef<Value *> Values) {
+  auto GetSplatOrScalar = [](Value *V) {
+    return isa<VectorType>(V->getType()) ? getSplatValue(V) : V;
   };
 
-  auto IsSupportedElementwise = [&](Instruction *Inst) {
-    auto *ResultTy = dyn_cast<VectorType>(Inst->getType());
-    if (!ResultTy || !isSafeToSpeculativelyExecute(Inst))
-      return false;
+  Value *CommonValue = GetSplatOrScalar(Values.front());
+  if (!CommonValue)
+    return nullptr;
+  for (Value *V : Values.drop_front())
+    if (GetSplatOrScalar(V) != CommonValue)
+      return nullptr;
+  return CommonValue;
+}
 
-    if (auto *II = dyn_cast<IntrinsicInst>(Inst)) {
-      if (II->hasOperandBundles() ||
-          !isTriviallyVectorizable(II->getIntrinsicID()))
-        return false;
-    } else if (!isa<BinaryOperator, UnaryOperator, CastInst, CmpInst,
-                    SelectInst, FreezeInst>(Inst)) {
-      return false;
-    }
+/// Return the common deinterleave intrinsic if \p Members are its extracts in
+/// field order.
+static IntrinsicInst *getDeinterleaveForMembers(ArrayRef<Value *> Members,
+                                                unsigned Factor) {
+  IntrinsicInst *Deinterleave = nullptr;
+  for (const auto &[Index, Member] : enumerate(Members)) {
+    auto *Extract = dyn_cast<ExtractValueInst>(Member);
+    if (!Extract || Extract->getNumIndices() != 1 ||
+        *Extract->idx_begin() != Index)
+      return nullptr;
 
-    // Reject operations that change the element-count.
-    // E.g., bitcast <vscale x 4 x i16> %v to <vscale x 8 x i8>
-    for (unsigned Op = 0, E = GetNumDataOperands(Inst); Op != E; ++Op) {
-      auto *OperandTy = dyn_cast<VectorType>(Inst->getOperand(Op)->getType());
-      if (OperandTy &&
-          OperandTy->getElementCount() != ResultTy->getElementCount())
-        return false;
-    }
+    auto *Current = dyn_cast<IntrinsicInst>(Extract->getAggregateOperand());
+    if (!Current || Current->hasOperandBundles() ||
+        getDeinterleaveIntrinsicFactor(Current->getIntrinsicID()) != Factor ||
+        (Deinterleave && Current != Deinterleave))
+      return nullptr;
+    Deinterleave = Current;
+  }
 
-    return true;
-  };
+  if (!Deinterleave || !Deinterleave->hasNUndroppableUses(Factor))
+    return nullptr;
+  return Deinterleave;
+}
 
-  // Traverse the Factor use chains with a breadth-first search.
-  // At each level, expect every chain to perform the same operation with the
-  // preceding chain value at the same operand position, until they all reach
-  // the matching interleave.
-  while (NumVisited + Factor <= MaxInstrsToScan) {
-    NumVisited += Factor;
-
-    for (Use *&CurrentUse : CurrentUses) {
-      Use *NextUse = CurrentUse->getUser()->getSingleUndroppableUse();
-      auto *Next =
-          NextUse ? dyn_cast<Instruction>(NextUse->getUser()) : nullptr;
-      if (!Next)
-        return false;
+static SmallVector<Value *, 8> getMemberOperands(ArrayRef<Value *> Members,
+                                                 unsigned OperandIndex) {
+  SmallVector<Value *, 8> Operands;
+  for (Value *Member : Members)
+    Operands.push_back(cast<Instruction>(Member)->getOperand(OperandIndex));
+  return Operands;
+}
 
-      CurrentUse = NextUse;
-    }
+/// Check whether the tree of elementwise operations each feeding \p Members
+/// can be rebuilt at the interleaved width.
+static bool canWidenOperations(ArrayRef<Value *> Members, unsigned Factor,
+                               unsigned &NumScanned) {
+  assert(Members.size() == Factor && "expected one member per field");
+  if (getDeinterleaveForMembers(Members, Factor))
+    return true;
+  if (NumScanned + Factor > MaxInstrsToScan)
+    return false;
+  NumScanned += Factor;
 
-    // Check whether every chain has reached the same interleave.
-    if (auto *II = dyn_cast<IntrinsicInst>(CurrentUses.front()->getUser());
-        II && II->getIntrinsicID() == ExpectedInterleaveIID) {
-      if (II->hasOperandBundles())
-        return false;
+  auto *FirstInst = dyn_cast<Instruction>(Members.front());
+  if (!FirstInst || !isSupportedElementwise(FirstInst) ||
+      !FirstInst->getSingleUndroppableUse())
+    return false;
 
-      for (unsigned Index = 0; Index != Factor; ++Index)
-        if (CurrentUses[Index]->getUser() != II ||
-            CurrentUses[Index]->getOperandNo() != Index)
-          return false;
+  for (Value *Member : Members.drop_front()) {
+    auto *Inst = dyn_cast<Instruction>(Member);
+    if (!Inst || !isSupportedElementwise(Inst) ||
+        !Inst->getSingleUndroppableUse() ||
+        !FirstInst->isSameOperationAs(Inst, Instruction::CompareCallTargets))
+      return false;
+  }
 
-      Interleave = II;
-      break;
+  for (unsigned Op = 0, E = getNumDataOperands(FirstInst); Op != E; ++Op) {
+    SmallVector<Value *, 8> Operands = getMemberOperands(Members, Op);
+    if (!isa<VectorType>(Operands.front()->getType())) {
+      if (!all_equal(Operands))
+        return false;
+      continue;
     }
 
-    auto *FirstInst = cast<Instruction>(CurrentUses.front()->getUser());
-    if (!IsSupportedElementwise(FirstInst))
+    if (!getCommonSplatValue(Operands) &&
+        !canWidenOperations(Operands, Factor, NumScanned))
       return false;
+  }
+  return true;
+}
 
-    unsigned ChainOperand = CurrentUses.front()->getOperandNo();
-    bool MismatchedUse = any_of(CurrentUses, [&](Use *U) {
-      auto *Inst = cast<Instruction>(U->getUser());
-      return Inst != FirstInst && (U->getOperandNo() != ChainOperand ||
-                                   !FirstInst->isSameOperationAs(
-                                       Inst, Instruction::CompareCallTargets));
-    });
-    if (MismatchedUse)
-      return false;
-
-    auto GetSplatOrScalar = [](Value *V) {
-      return isa<VectorType>(V->getType()) ? getSplatValue(V) : V;
-    };
-
-    // Non-chain operands must be either the same scalar or splats of that
-    // scalar. This intentionally rejects differing poison/undef or non-splat
-    // vector operands between chains.
-    for (unsigned Op = 0, E = GetNumDataOperands(FirstInst); Op != E; ++Op) {
-      if (Op == ChainOperand)
-        continue;
+static Value *createWideInstruction(Instruction *NarrowInst,
+                                    ArrayRef<Value *> NewOperands,
+                                    VectorType *WideResultTy,
+                                    IRBuilder<InstSimplifyFolder> &Builder) {
+  if (isa<BinaryOperator, UnaryOperator>(NarrowInst))
+    return Builder.CreateNAryOp(NarrowInst->getOpcode(), NewOperands);
+  if (auto *Cast = dyn_cast<CastInst>(NarrowInst))
+    return Builder.CreateCast(Cast->getOpcode(), NewOperands[0], WideResultTy);
+  if (auto *Cmp = dyn_cast<CmpInst>(NarrowInst))
+    return Builder.CreateCmp(Cmp->getPredicate(), NewOperands[0],
+                             NewOperands[1]);
+  if (isa<SelectInst>(NarrowInst))
+    return Builder.CreateSelect(
+        NewOperands[0], NewOperands[1], NewOperands[2], /*Name=*/"",
+        ProfcheckDisableMetadataFixes ? nullptr : NarrowInst);
+  if (isa<FreezeInst>(NarrowInst))
+    return Builder.CreateFreeze(NewOperands[0]);
+  if (auto *II = dyn_cast<IntrinsicInst>(NarrowInst))
+    return Builder.CreateIntrinsic(WideResultTy, II->getIntrinsicID(),
+                                   NewOperands);
+  llvm_unreachable("Unsupported instruction");
+}
 
-      Value *CommonValue = GetSplatOrScalar(FirstInst->getOperand(Op));
-      if (!CommonValue || any_of(CurrentUses, [&](Use *U) {
-            Instruction *Inst = cast<Instruction>(U->getUser());
-            return Inst != FirstInst &&
-                   GetSplatOrScalar(Inst->getOperand(Op)) != CommonValue;
-          }))
-        return false;
+static Value *widenOperations(ArrayRef<Value *> Members, unsigned Factor,
+                              ElementCount WideEC,
+                              IRBuilder<InstSimplifyFolder> &Builder) {
+  if (auto *Deinterleave = getDeinterleaveForMembers(Members, Factor)) {
+    Value *Source = Deinterleave->getArgOperand(0);
+    assert(cast<VectorType>(Source->getType())->getElementCount() == WideEC &&
+           "deinterleave source must have the interleaved element count");
+    return Source;
+  }
+
+  auto *NarrowInst = cast<Instruction>(Members.front());
+  unsigned NumOperands = getNumDataOperands(NarrowInst);
+  SmallVector<Value *, 4> NewOperands;
+  NewOperands.reserve(NumOperands);
+  for (unsigned Op = 0; Op != NumOperands; ++Op) {
+    SmallVector<Value *, 8> Operands = getMemberOperands(Members, Op);
+    Value *NewOperand = Operands.front();
+    if (isa<VectorType>(NewOperand->getType())) {
+      if (Value *CommonValue = getCommonSplatValue(Operands)) {
+        Builder.SetCurrentDebugLocation(NarrowInst->getDebugLoc());
+        NewOperand = Builder.CreateVectorSplat(WideEC, CommonValue);
+      } else {
+        NewOperand = widenOperations(Operands, Factor, WideEC, Builder);
+      }
     }
+    NewOperands.push_back(NewOperand);
+  }
 
-    Steps.push_back(CurrentUses);
+  Builder.SetCurrentDebugLocation(NarrowInst->getDebugLoc());
+  auto *WideResultTy =
+      VectorType::get(NarrowInst->getType()->getScalarType(), WideEC);
+  Value *NewValue =
+      createWideInstruction(NarrowInst, NewOperands, WideResultTy, Builder);
+  auto *NewInst = dyn_cast<Instruction>(NewValue);
+  if (NewInst) {
+    propagateIRFlags(NewInst, Members);
+    propagateMetadata(NewInst, Members);
   }
+  return NewValue;
+}
 
+/// Fold away vector.deinterleave/interleave intrinsics with matching trees of
+/// elementwise operations between them.
+///
+/// For example:
+///  ```
+///  %d = call { <2 x i16>, <2 x i16> } @deinterleave2.v4i16(<4 x i16> %v)
+///  %f0 = extractvalue { <2 x i16>, <2 x i16> } %d, 0
+///  %f1 = extractvalue { <2 x i16>, <2 x i16> } %d, 1
+///
+///  %u0 = add <2 x i16> %f0, splat (i16 3)
+///  %u1 = add <2 x i16> %f1, splat (i16 3)
+///
+///  %r = call <4 x i16> @interleave2.v4i16(<2 x i16> %u0, <2 x i16> %u1)
+///  ```
+/// Folds to:
+///  ```
+///  %r = add <4 x i16> %v, splat (i16 3)
+///  ```
+/// And with two sources:
+///  ```
+///  %da = call { <2 x i16>, <2 x i16> } @deinterleave2.v4i16(<4 x i16> %a)
+///  %a0 = extractvalue { <2 x i16>, <2 x i16> } %da, 0
+///  %a1 = extractvalue { <2 x i16>, <2 x i16> } %da, 1
+///  %db = call { <2 x i16>, <2 x i16> } @deinterleave2.v4i16(<4 x i16> %b)
+///  %b0 = extractvalue { <2 x i16>, <2 x i16> } %db, 0
+///  %b1 = extractvalue { <2 x i16>, <2 x i16> } %db, 1
+///
+///  %m0 = mul <2 x i16> %a0, %b0
+///  %m1 = mul <2 x i16> %a1, %b1
+///
+///  %r = call <4 x i16> @interleave2.v4i16(<2 x i16> %m0, <2 x i16> %m1)
+///  ```
+/// Folds to:
+///  ```
+///  %r = mul <4 x i16> %a, %b
+///  ```
+bool VectorCombine::foldInterleaveOfDeinterleaveChains(Instruction &I) {
+  auto *Interleave = dyn_cast<IntrinsicInst>(&I);
   if (!Interleave)
     return false;
 
-  // Rebuild the matched elementwise chain at the original vector width.
-  Value *WideValue = Deinterleave->getArgOperand(0);
-  ElementCount WideEC =
-      cast<VectorType>(WideValue->getType())->getElementCount();
-
-  auto CreateWideInstruction = [&](Instruction *NarrowInst,
-                                   ArrayRef<Value *> NewOperands,
-                                   VectorType *WideResultTy) -> Value * {
-    assert(IsSupportedElementwise(NarrowInst) &&
-           "Expected supported elementwise");
-    if (isa<BinaryOperator, UnaryOperator>(NarrowInst))
-      return Builder.CreateNAryOp(NarrowInst->getOpcode(), NewOperands);
-    if (auto *Cast = dyn_cast<CastInst>(NarrowInst))
-      return Builder.CreateCast(Cast->getOpcode(), NewOperands[0],
-                                WideResultTy);
-    if (auto *Cmp = dyn_cast<CmpInst>(NarrowInst))
-      return Builder.CreateCmp(Cmp->getPredicate(), NewOperands[0],
-                               NewOperands[1]);
-    if (isa<SelectInst>(NarrowInst))
-      return Builder.CreateSelect(
-          NewOperands[0], NewOperands[1], NewOperands[2], /*Name=*/"",
-          ProfcheckDisableMetadataFixes ? nullptr : NarrowInst);
-    if (isa<FreezeInst>(NarrowInst))
-      return Builder.CreateFreeze(NewOperands[0]);
-    if (auto *II = dyn_cast<IntrinsicInst>(NarrowInst))
-      return Builder.CreateIntrinsic(WideResultTy, II->getIntrinsicID(),
-                                     NewOperands);
-    llvm_unreachable("Unsupported instruction");
-  };
-
-  // The BFS has succeeded and collected multiple levels of instructions that
-  // can be SLP-widened into a chain of wider instructions.
-  for (const ElementwiseStep &Step : Steps) {
-    Instruction *NarrowInst = cast<Instruction>(Step.front()->getUser());
-    unsigned ChainOperand = Step.front()->getOperandNo();
-
-    Builder.SetInsertPoint(NarrowInst);
-    Builder.SetCurrentDebugLocation(NarrowInst->getDebugLoc());
-
-    unsigned NumOperands = GetNumDataOperands(NarrowInst);
-    SmallVector<Value *, 4> NewOperands;
-    NewOperands.reserve(NumOperands);
-
-    for (unsigned Op = 0; Op != NumOperands; ++Op) {
-      Value *Operand = NarrowInst->getOperand(Op);
-
-      if (Op == ChainOperand)
-        Operand = WideValue;
-      else if (isa<VectorType>(Operand->getType()))
-        Operand = Builder.CreateVectorSplat(WideEC, getSplatValue(Operand));
-      NewOperands.push_back(Operand);
-    }
-
-    auto *WideResultTy =
-        VectorType::get(NarrowInst->getType()->getScalarType(), WideEC);
-    Value *NewValue =
-        CreateWideInstruction(NarrowInst, NewOperands, WideResultTy);
-
-    SmallVector<Value *> NarrowInsts =
-        map_to_vector(Step, [](Use *U) { return cast<Value>(U->getUser()); });
-    propagateIRFlags(NewValue, NarrowInsts);
-
-    if (auto *NewInst = dyn_cast<Instruction>(NewValue))
-      propagateMetadata(NewInst, NarrowInsts);
+  unsigned Factor = getInterleaveIntrinsicFactor(Interleave->getIntrinsicID());
+  if (!Factor || Interleave->hasOperandBundles())
+    return false;
 
-    WideValue = NewValue;
-  }
+  SmallVector<Value *, 8> RootMembers(Interleave->args());
+  unsigned NumScanned = 0;
+  if (!canWidenOperations(RootMembers, Factor, NumScanned))
+    return false;
 
+  ElementCount WideEC = cast<VectorType>(I.getType())->getElementCount();
+  Builder.SetInsertPoint(Interleave);
+  Value *WideValue = widenOperations(RootMembers, Factor, WideEC, Builder);
   assert(WideValue->getType() == Interleave->getType());
   replaceValue(*Interleave, *WideValue);
   return true;
@@ -6269,6 +6278,9 @@ bool VectorCombine::foldDeinterleaveInterleavePair(Instruction &I) {
 /// larger splat `<vscale x 8 x i64> <splat of ((777 << 32) | 666)>` first
 /// before casting it back into `<vscale x 16 x i32>`.
 bool VectorCombine::foldInterleaveIntrinsics(Instruction &I) {
+  if (foldInterleaveOfDeinterleaveChains(I))
+    return true;
+
   const APInt *SplatVal0, *SplatVal1;
   if (!match(&I, m_Intrinsic<Intrinsic::vector_interleave2>(
                      m_APInt(SplatVal0), m_APInt(SplatVal1))))
@@ -6334,9 +6346,6 @@ bool VectorCombine::foldInterleaveIntrinsics(Instruction &I) {
 /// %merge1 = bitcast <vscale x 16 x i16> %f1 to <vscale x 8 x i32>
 /// ```
 bool VectorCombine::foldDeinterleaveIntrinsics(Instruction &I) {
-  if (foldDeinterleaveInterleavePair(I))
-    return true;
-
   // This pattern involves bitcast that is not compatible with big endian.
   if (DL->isBigEndian())
     return false;
diff --git a/llvm/test/Transforms/VectorCombine/deinterleave-interleave-tree.ll b/llvm/test/Transforms/VectorCombine/deinterleave-interleave-tree.ll
index 35b8d5bc8ca74..0e00d36ba4d42 100644
--- a/llvm/test/Transforms/VectorCombine/deinterleave-interleave-tree.ll
+++ b/llvm/test/Transforms/VectorCombine/deinterleave-interleave-tree.ll
@@ -4,19 +4,9 @@
 define <8 x i16> @two_deinterleave2_mul_xor_trunc(<8 x i32> %a, <8 x i32> %b) {
 ; CHECK-LABEL: define <8 x i16> @two_deinterleave2_mul_xor_trunc(
 ; CHECK-SAME: <8 x i32> [[A:%.*]], <8 x i32> [[B:%.*]]) {
-; CHECK-NEXT:    [[DA:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[A]])
-; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 0
-; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 1
-; CHECK-NEXT:    [[DB:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[B]])
-; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 0
-; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 1
-; CHECK-NEXT:    [[M0:%.*]] = mul <4 x i32> [[A0]], [[B0]]
-; CHECK-NEXT:    [[M1:%.*]] = mul <4 x i32> [[A1]], [[B1]]
-; CHECK-NEXT:    [[X0:%.*]] = xor <4 x i32> [[M0]], splat (i32 7)
-; CHECK-NEXT:    [[X1:%.*]] = xor <4 x i32> [[M1]], splat (i32 7)
-; CHECK-NEXT:    [[T0:%.*]] = trunc <4 x i32> [[X0]] to <4 x i16>
-; CHECK-NEXT:    [[T1:%.*]] = trunc <4 x i32> [[X1]] to <4 x i16>
-; CHECK-NEXT:    [[R:%.*]] = call <8 x i16> @llvm.vector.interleave2.v8i16(<4 x i16> [[T0]], <4 x i16> [[T1]])
+; CHECK-NEXT:    [[TMP1:%.*]] = mul <8 x i32> [[A]], [[B]]
+; CHECK-NEXT:    [[TMP2:%.*]] = xor <8 x i32> [[TMP1]], splat (i32 7)
+; CHECK-NEXT:    [[R:%.*]] = trunc <8 x i32> [[TMP2]] to <8 x i16>
 ; CHECK-NEXT:    ret <8 x i16> [[R]]
 ;
   %da = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %a)
@@ -38,21 +28,10 @@ define <8 x i16> @two_deinterleave2_mul_xor_trunc(<8 x i32> %a, <8 x i32> %b) {
 define <vscale x 8 x i16> @two_deinterleave2_independent_chains(<vscale x 8 x i16> %a, <vscale x 8 x i16> %b) {
 ; CHECK-LABEL: define <vscale x 8 x i16> @two_deinterleave2_independent_chains(
 ; CHECK-SAME: <vscale x 8 x i16> [[A:%.*]], <vscale x 8 x i16> [[B:%.*]]) {
-; CHECK-NEXT:    [[DA:%.*]] = call { <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave2.nxv8i16(<vscale x 8 x i16> [[A]])
-; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } [[DA]], 0
-; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } [[DA]], 1
-; CHECK-NEXT:    [[DB:%.*]] = call { <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave2.nxv8i16(<vscale x 8 x i16> [[B]])
-; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } [[DB]], 0
-; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16> } [[DB]], 1
-; CHECK-NEXT:    [[AA0:%.*]] = add <vscale x 4 x i16> [[A0]], splat (i16 1)
-; CHECK-NEXT:    [[AA1:%.*]] = add <vscale x 4 x i16> [[A1]], splat (i16 1)
-; CHECK-NEXT:    [[BB0:%.*]] = xor <vscale x 4 x i16> [[B0]], splat (i16 3)
-; CHECK-NEXT:    [[BB1:%.*]] = xor <vscale x 4 x i16> [[B1]], splat (i16 3)
-; CHECK-NEXT:    [[M0:%.*]] = mul <vscale x 4 x i16> [[AA0]], [[BB0]]
-; CHECK-NEXT:    [[M1:%.*]] = mul <vscale x 4 x i16> [[AA1]], [[BB1]]
-; CHECK-NEXT:    [[S0:%.*]] = sub <vscale x 4 x i16> [[M0]], splat (i16 5)
-; CHECK-NEXT:    [[S1:%.*]] = sub <vscale x 4 x i16> [[M1]], splat (i16 5)
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 8 x i16> @llvm.vector.interleave2.nxv8i16(<vscale x 4 x i16> [[S0]], <vscale x 4 x i16> [[S1]])
+; CHECK-NEXT:    [[TMP1:%.*]] = add <vscale x 8 x i16> [[A]], splat (i16 1)
+; CHECK-NEXT:    [[TMP2:%.*]] = xor <vscale x 8 x i16> [[B]], splat (i16 3)
+; CHECK-NEXT:    [[TMP3:%.*]] = mul <vscale x 8 x i16> [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[R:%.*]] = sub <vscale x 8 x i16> [[TMP3]], splat (i16 5)
 ; CHECK-NEXT:    ret <vscale x 8 x i16> [[R]]
 ;
   %da = call { <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave2.nxv8i16(<vscale x 8 x i16> %a)
@@ -76,18 +55,7 @@ define <vscale x 8 x i16> @two_deinterleave2_independent_chains(<vscale x 8 x i1
 define <vscale x 6 x i32> @two_deinterleave3_add(<vscale x 6 x i32> %a, <vscale x 6 x i32> %b) {
 ; CHECK-LABEL: define <vscale x 6 x i32> @two_deinterleave3_add(
 ; CHECK-SAME: <vscale x 6 x i32> [[A:%.*]], <vscale x 6 x i32> [[B:%.*]]) {
-; CHECK-NEXT:    [[DA:%.*]] = call { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave3.nxv6i32(<vscale x 6 x i32> [[A]])
-; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DA]], 0
-; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DA]], 1
-; CHECK-NEXT:    [[A2:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DA]], 2
-; CHECK-NEXT:    [[DB:%.*]] = call { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave3.nxv6i32(<vscale x 6 x i32> [[B]])
-; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DB]], 0
-; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DB]], 1
-; CHECK-NEXT:    [[B2:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } [[DB]], 2
-; CHECK-NEXT:    [[S0:%.*]] = add <vscale x 2 x i32> [[A0]], [[B0]]
-; CHECK-NEXT:    [[S1:%.*]] = add <vscale x 2 x i32> [[A1]], [[B1]]
-; CHECK-NEXT:    [[S2:%.*]] = add <vscale x 2 x i32> [[A2]], [[B2]]
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 6 x i32> @llvm.vector.interleave3.nxv6i32(<vscale x 2 x i32> [[S0]], <vscale x 2 x i32> [[S1]], <vscale x 2 x i32> [[S2]])
+; CHECK-NEXT:    [[R:%.*]] = add <vscale x 6 x i32> [[A]], [[B]]
 ; CHECK-NEXT:    ret <vscale x 6 x i32> [[R]]
 ;
   %da = call { <vscale x 2 x i32>, <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave3.nxv6i32(<vscale x 6 x i32> %a)
@@ -137,24 +105,10 @@ define <4 x i32> @negative_deinterleave2_mixed_sources(
 define <8 x i32> @three_deinterleave2_tree_with_splats(<8 x i32> %a, <8 x i32> %b, <8 x i32> %c) {
 ; CHECK-LABEL: define <8 x i32> @three_deinterleave2_tree_with_splats(
 ; CHECK-SAME: <8 x i32> [[A:%.*]], <8 x i32> [[B:%.*]], <8 x i32> [[C:%.*]]) {
-; CHECK-NEXT:    [[DA:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[A]])
-; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 0
-; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 1
-; CHECK-NEXT:    [[DB:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[B]])
-; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 0
-; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 1
-; CHECK-NEXT:    [[DC:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[C]])
-; CHECK-NEXT:    [[C0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DC]], 0
-; CHECK-NEXT:    [[C1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DC]], 1
-; CHECK-NEXT:    [[M0:%.*]] = mul <4 x i32> [[A0]], [[B0]]
-; CHECK-NEXT:    [[M1:%.*]] = mul <4 x i32> [[A1]], [[B1]]
-; CHECK-NEXT:    [[X0:%.*]] = xor <4 x i32> [[M0]], splat (i32 7)
-; CHECK-NEXT:    [[X1:%.*]] = xor <4 x i32> [[M1]], splat (i32 7)
-; CHECK-NEXT:    [[S0:%.*]] = add <4 x i32> [[X0]], [[C0]]
-; CHECK-NEXT:    [[S1:%.*]] = add <4 x i32> [[X1]], [[C1]]
-; CHECK-NEXT:    [[T0:%.*]] = shl <4 x i32> [[S0]], splat (i32 1)
-; CHECK-NEXT:    [[T1:%.*]] = shl <4 x i32> [[S1]], splat (i32 1)
-; CHECK-NEXT:    [[R:%.*]] = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> [[T0]], <4 x i32> [[T1]])
+; CHECK-NEXT:    [[TMP1:%.*]] = mul <8 x i32> [[A]], [[B]]
+; CHECK-NEXT:    [[TMP2:%.*]] = xor <8 x i32> [[TMP1]], splat (i32 7)
+; CHECK-NEXT:    [[TMP3:%.*]] = add <8 x i32> [[TMP2]], [[C]]
+; CHECK-NEXT:    [[R:%.*]] = shl <8 x i32> [[TMP3]], splat (i32 1)
 ; CHECK-NEXT:    ret <8 x i32> [[R]]
 ;
   %da = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %a)
@@ -181,12 +135,7 @@ define <8 x i32> @three_deinterleave2_tree_with_splats(<8 x i32> %a, <8 x i32> %
 define <4 x i32> @deinterleave2_extract_used_twice_in_member(<4 x i32> %a) {
 ; CHECK-LABEL: define <4 x i32> @deinterleave2_extract_used_twice_in_member(
 ; CHECK-SAME: <4 x i32> [[A:%.*]]) {
-; CHECK-NEXT:    [[DA:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[A]])
-; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 0
-; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 1
-; CHECK-NEXT:    [[M0:%.*]] = mul <2 x i32> [[A0]], [[A0]]
-; CHECK-NEXT:    [[M1:%.*]] = mul <2 x i32> [[A1]], [[A1]]
-; CHECK-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[M0]], <2 x i32> [[M1]])
+; CHECK-NEXT:    [[R:%.*]] = mul <4 x i32> [[A]], [[A]]
 ; CHECK-NEXT:    ret <4 x i32> [[R]]
 ;
   %da = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %a)
@@ -223,10 +172,8 @@ define <4 x i32> @deinterleave2_extract_with_unfolded_user(<4 x i32> %a, ptr %p)
 ; CHECK-SAME: <4 x i32> [[A:%.*]], ptr [[P:%.*]]) {
 ; CHECK-NEXT:    [[D:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[A]])
 ; CHECK-NEXT:    [[F0:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[D]], 0
-; CHECK-NEXT:    [[F1:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[D]], 1
-; CHECK-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[F0]], <2 x i32> [[F1]])
 ; CHECK-NEXT:    store <2 x i32> [[F0]], ptr [[P]], align 8
-; CHECK-NEXT:    ret <4 x i32> [[R]]
+; CHECK-NEXT:    ret <4 x i32> [[A]]
 ;
   %d = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %a)
   %f0 = extractvalue { <2 x i32>, <2 x i32> } %d, 0
@@ -239,15 +186,8 @@ define <4 x i32> @deinterleave2_extract_with_unfolded_user(<4 x i32> %a, ptr %p)
 define <4 x i32> @deinterleave2_extracts_feed_two_interleaves(<4 x i32> %a) {
 ; CHECK-LABEL: define <4 x i32> @deinterleave2_extracts_feed_two_interleaves(
 ; CHECK-SAME: <4 x i32> [[A:%.*]]) {
-; CHECK-NEXT:    [[DA:%.*]] = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> [[A]])
-; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 0
-; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <2 x i32>, <2 x i32> } [[DA]], 1
-; CHECK-NEXT:    [[S0:%.*]] = add <2 x i32> [[A0]], splat (i32 1)
-; CHECK-NEXT:    [[S1:%.*]] = add <2 x i32> [[A1]], splat (i32 1)
-; CHECK-NEXT:    [[R1:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[S0]], <2 x i32> [[S1]])
-; CHECK-NEXT:    [[T0:%.*]] = sub <2 x i32> [[A0]], splat (i32 1)
-; CHECK-NEXT:    [[T1:%.*]] = sub <2 x i32> [[A1]], splat (i32 1)
-; CHECK-NEXT:    [[R2:%.*]] = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> [[T0]], <2 x i32> [[T1]])
+; CHECK-NEXT:    [[R1:%.*]] = add <4 x i32> [[A]], splat (i32 1)
+; CHECK-NEXT:    [[R2:%.*]] = sub <4 x i32> [[A]], splat (i32 1)
 ; CHECK-NEXT:    [[R:%.*]] = xor <4 x i32> [[R1]], [[R2]]
 ; CHECK-NEXT:    ret <4 x i32> [[R]]
 ;
@@ -297,15 +237,7 @@ define <4 x float> @deinterleave2_fabs_mismatched_fmf(<4 x float> %v) {
 define <8 x i32> @two_deinterleave2_smax(<8 x i32> %a, <8 x i32> %b) {
 ; CHECK-LABEL: define <8 x i32> @two_deinterleave2_smax(
 ; CHECK-SAME: <8 x i32> [[A:%.*]], <8 x i32> [[B:%.*]]) {
-; CHECK-NEXT:    [[DA:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[A]])
-; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 0
-; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DA]], 1
-; CHECK-NEXT:    [[DB:%.*]] = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> [[B]])
-; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 0
-; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <4 x i32>, <4 x i32> } [[DB]], 1
-; CHECK-NEXT:    [[M0:%.*]] = call <4 x i32> @llvm.smax.v4i32(<4 x i32> [[A0]], <4 x i32> [[B0]])
-; CHECK-NEXT:    [[M1:%.*]] = call <4 x i32> @llvm.smax.v4i32(<4 x i32> [[A1]], <4 x i32> [[B1]])
-; CHECK-NEXT:    [[R:%.*]] = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> [[M0]], <4 x i32> [[M1]])
+; CHECK-NEXT:    [[R:%.*]] = call <8 x i32> @llvm.smax.v8i32(<8 x i32> [[A]], <8 x i32> [[B]])
 ; CHECK-NEXT:    ret <8 x i32> [[R]]
 ;
   %da = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %a)
@@ -323,15 +255,7 @@ define <8 x i32> @two_deinterleave2_smax(<8 x i32> %a, <8 x i32> %b) {
 define <vscale x 4 x double> @two_deinterleave2_fma_splat(<vscale x 4 x double> %a, <vscale x 4 x double> %b) {
 ; CHECK-LABEL: define <vscale x 4 x double> @two_deinterleave2_fma_splat(
 ; CHECK-SAME: <vscale x 4 x double> [[A:%.*]], <vscale x 4 x double> [[B:%.*]]) {
-; CHECK-NEXT:    [[DA:%.*]] = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.vector.deinterleave2.nxv4f64(<vscale x 4 x double> [[A]])
-; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } [[DA]], 0
-; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } [[DA]], 1
-; CHECK-NEXT:    [[DB:%.*]] = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.vector.deinterleave2.nxv4f64(<vscale x 4 x double> [[B]])
-; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } [[DB]], 0
-; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <vscale x 2 x double>, <vscale x 2 x double> } [[DB]], 1
-; CHECK-NEXT:    [[M0:%.*]] = call contract <vscale x 2 x double> @llvm.fma.nxv2f64(<vscale x 2 x double> [[A0]], <vscale x 2 x double> [[B0]], <vscale x 2 x double> splat (double 1.000000e+00))
-; CHECK-NEXT:    [[M1:%.*]] = call contract <vscale x 2 x double> @llvm.fma.nxv2f64(<vscale x 2 x double> [[A1]], <vscale x 2 x double> [[B1]], <vscale x 2 x double> splat (double 1.000000e+00))
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x double> @llvm.vector.interleave2.nxv4f64(<vscale x 2 x double> [[M0]], <vscale x 2 x double> [[M1]])
+; CHECK-NEXT:    [[R:%.*]] = call contract <vscale x 4 x double> @llvm.fma.nxv4f64(<vscale x 4 x double> [[A]], <vscale x 4 x double> [[B]], <vscale x 4 x double> splat (double 1.000000e+00))
 ; CHECK-NEXT:    ret <vscale x 4 x double> [[R]]
 ;
   %da = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.vector.deinterleave2.nxv4f64(<vscale x 4 x double> %a)
@@ -484,24 +408,9 @@ define <4 x i32> @negative_deinterleave2_vector_reverse(<4 x i32> %v) {
 define <vscale x 12 x i16> @two_deinterleave3_abs_mul_sat(<vscale x 12 x i16> %a, <vscale x 12 x i16> %b) {
 ; CHECK-LABEL: define <vscale x 12 x i16> @two_deinterleave3_abs_mul_sat(
 ; CHECK-SAME: <vscale x 12 x i16> [[A:%.*]], <vscale x 12 x i16> [[B:%.*]]) {
-; CHECK-NEXT:    [[DA:%.*]] = call { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave3.nxv12i16(<vscale x 12 x i16> [[A]])
-; CHECK-NEXT:    [[A0:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DA]], 0
-; CHECK-NEXT:    [[A1:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DA]], 1
-; CHECK-NEXT:    [[A2:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DA]], 2
-; CHECK-NEXT:    [[DB:%.*]] = call { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave3.nxv12i16(<vscale x 12 x i16> [[B]])
-; CHECK-NEXT:    [[B0:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DB]], 0
-; CHECK-NEXT:    [[B1:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DB]], 1
-; CHECK-NEXT:    [[B2:%.*]] = extractvalue { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } [[DB]], 2
-; CHECK-NEXT:    [[ABS0:%.*]] = call <vscale x 4 x i16> @llvm.abs.nxv4i16(<vscale x 4 x i16> [[A0]], i1 false)
-; CHECK-NEXT:    [[ABS1:%.*]] = call <vscale x 4 x i16> @llvm.abs.nxv4i16(<vscale x 4 x i16> [[A1]], i1 false)
-; CHECK-NEXT:    [[ABS2:%.*]] = call <vscale x 4 x i16> @llvm.abs.nxv4i16(<vscale x 4 x i16> [[A2]], i1 false)
-; CHECK-NEXT:    [[M0:%.*]] = mul nsw <vscale x 4 x i16> [[ABS0]], [[B0]]
-; CHECK-NEXT:    [[M1:%.*]] = mul nsw <vscale x 4 x i16> [[ABS1]], [[B1]]
-; CHECK-NEXT:    [[M2:%.*]] = mul nsw <vscale x 4 x i16> [[ABS2]], [[B2]]
-; CHECK-NEXT:    [[S0:%.*]] = call <vscale x 4 x i16> @llvm.sadd.sat.nxv4i16(<vscale x 4 x i16> [[M0]], <vscale x 4 x i16> splat (i16 7))
-; CHECK-NEXT:    [[S1:%.*]] = call <vscale x 4 x i16> @llvm.sadd.sat.nxv4i16(<vscale x 4 x i16> [[M1]], <vscale x 4 x i16> splat (i16 7))
-; CHECK-NEXT:    [[S2:%.*]] = call <vscale x 4 x i16> @llvm.sadd.sat.nxv4i16(<vscale x 4 x i16> [[M2]], <vscale x 4 x i16> splat (i16 7))
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 12 x i16> @llvm.vector.interleave3.nxv12i16(<vscale x 4 x i16> [[S0]], <vscale x 4 x i16> [[S1]], <vscale x 4 x i16> [[S2]])
+; CHECK-NEXT:    [[TMP1:%.*]] = call <vscale x 12 x i16> @llvm.abs.nxv12i16(<vscale x 12 x i16> [[A]], i1 false)
+; CHECK-NEXT:    [[TMP2:%.*]] = mul nsw <vscale x 12 x i16> [[TMP1]], [[B]]
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 12 x i16> @llvm.sadd.sat.nxv12i16(<vscale x 12 x i16> [[TMP2]], <vscale x 12 x i16> splat (i16 7))
 ; CHECK-NEXT:    ret <vscale x 12 x i16> [[R]]
 ;
   %da = call { <vscale x 4 x i16>, <vscale x 4 x i16>, <vscale x 4 x i16> } @llvm.vector.deinterleave3.nxv12i16(<vscale x 12 x i16> %a)

>From 3a76c229229de79ed6dc7b5ba90dd9875d6a9969 Mon Sep 17 00:00:00 2001
From: Kamlesh Kumar <kamlesh.kumar at arm.com>
Date: Tue, 22 Sep 2026 09:23:26 +0100
Subject: [PATCH 3/3] fixip address review comments

---
 .../Transforms/Vectorize/VectorCombine.cpp    | 51 +++++++++++--------
 1 file changed, 29 insertions(+), 22 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index ce62a45d5f488..c3ae1fb856a9f 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -6032,6 +6032,7 @@ bool VectorCombine::foldInsExtVectorToShuffle(Instruction &I) {
   return true;
 }
 
+/// Return the number of data operands of \p Inst.
 static unsigned getNumDataOperands(const Instruction *Inst) {
   if (auto *CB = dyn_cast<CallBase>(Inst))
     return CB->arg_size(); // Exclude callee operand and bundles.
@@ -6066,6 +6067,7 @@ static bool isSupportedElementwise(Instruction *Inst) {
   return true;
 }
 
+/// Return the common splat value of \p Values.
 static Value *getCommonSplatValue(ArrayRef<Value *> Values) {
   auto GetSplatOrScalar = [](Value *V) {
     return isa<VectorType>(V->getType()) ? getSplatValue(V) : V;
@@ -6092,20 +6094,21 @@ static IntrinsicInst *getDeinterleaveForMembers(ArrayRef<Value *> Members,
       return nullptr;
 
     auto *Current = dyn_cast<IntrinsicInst>(Extract->getAggregateOperand());
-    if (!Current || Current->hasOperandBundles() ||
-        getDeinterleaveIntrinsicFactor(Current->getIntrinsicID()) != Factor ||
-        (Deinterleave && Current != Deinterleave))
+    if (!Current || (Deinterleave && Current != Deinterleave))
       return nullptr;
     Deinterleave = Current;
   }
-
-  if (!Deinterleave || !Deinterleave->hasNUndroppableUses(Factor))
+  if (Deinterleave->hasOperandBundles() ||
+      getDeinterleaveIntrinsicFactor(Deinterleave->getIntrinsicID()) !=
+          Factor ||
+      !Deinterleave->hasNUndroppableUses(Factor))
     return nullptr;
   return Deinterleave;
 }
 
-static SmallVector<Value *, 8> getMemberOperands(ArrayRef<Value *> Members,
-                                                 unsigned OperandIndex) {
+/// Return the operand at \p OperandIndex of each \p Members.
+static SmallVector<Value *, 8>
+getDeinterleavedOperands(ArrayRef<Value *> Members, unsigned OperandIndex) {
   SmallVector<Value *, 8> Operands;
   for (Value *Member : Members)
     Operands.push_back(cast<Instruction>(Member)->getOperand(OperandIndex));
@@ -6114,8 +6117,9 @@ static SmallVector<Value *, 8> getMemberOperands(ArrayRef<Value *> Members,
 
 /// Check whether the tree of elementwise operations each feeding \p Members
 /// can be rebuilt at the interleaved width.
-static bool canWidenOperations(ArrayRef<Value *> Members, unsigned Factor,
-                               unsigned &NumScanned) {
+static bool canWidenDeinterleavedOperations(ArrayRef<Value *> Members,
+                                            unsigned Factor,
+                                            unsigned &NumScanned) {
   assert(Members.size() == Factor && "expected one member per field");
   if (getDeinterleaveForMembers(Members, Factor))
     return true;
@@ -6130,14 +6134,15 @@ static bool canWidenOperations(ArrayRef<Value *> Members, unsigned Factor,
 
   for (Value *Member : Members.drop_front()) {
     auto *Inst = dyn_cast<Instruction>(Member);
-    if (!Inst || !isSupportedElementwise(Inst) ||
-        !Inst->getSingleUndroppableUse() ||
+    if (!Inst || !Inst->getSingleUndroppableUse() ||
         !FirstInst->isSameOperationAs(Inst, Instruction::CompareCallTargets))
       return false;
   }
 
+  // Scalars operands should be equal among all members.
+  // Vector operands should be a common splat value or can be widened.
   for (unsigned Op = 0, E = getNumDataOperands(FirstInst); Op != E; ++Op) {
-    SmallVector<Value *, 8> Operands = getMemberOperands(Members, Op);
+    SmallVector<Value *, 8> Operands = getDeinterleavedOperands(Members, Op);
     if (!isa<VectorType>(Operands.front()->getType())) {
       if (!all_equal(Operands))
         return false;
@@ -6145,7 +6150,7 @@ static bool canWidenOperations(ArrayRef<Value *> Members, unsigned Factor,
     }
 
     if (!getCommonSplatValue(Operands) &&
-        !canWidenOperations(Operands, Factor, NumScanned))
+        !canWidenDeinterleavedOperations(Operands, Factor, NumScanned))
       return false;
   }
   return true;
@@ -6174,9 +6179,10 @@ static Value *createWideInstruction(Instruction *NarrowInst,
   llvm_unreachable("Unsupported instruction");
 }
 
-static Value *widenOperations(ArrayRef<Value *> Members, unsigned Factor,
-                              ElementCount WideEC,
-                              IRBuilder<InstSimplifyFolder> &Builder) {
+static Value *
+widenDeinterleavedOperations(ArrayRef<Value *> Members, unsigned Factor,
+                             ElementCount WideEC,
+                             IRBuilder<InstSimplifyFolder> &Builder) {
   if (auto *Deinterleave = getDeinterleaveForMembers(Members, Factor)) {
     Value *Source = Deinterleave->getArgOperand(0);
     assert(cast<VectorType>(Source->getType())->getElementCount() == WideEC &&
@@ -6189,14 +6195,15 @@ static Value *widenOperations(ArrayRef<Value *> Members, unsigned Factor,
   SmallVector<Value *, 4> NewOperands;
   NewOperands.reserve(NumOperands);
   for (unsigned Op = 0; Op != NumOperands; ++Op) {
-    SmallVector<Value *, 8> Operands = getMemberOperands(Members, Op);
+    SmallVector<Value *, 8> Operands = getDeinterleavedOperands(Members, Op);
     Value *NewOperand = Operands.front();
     if (isa<VectorType>(NewOperand->getType())) {
       if (Value *CommonValue = getCommonSplatValue(Operands)) {
         Builder.SetCurrentDebugLocation(NarrowInst->getDebugLoc());
         NewOperand = Builder.CreateVectorSplat(WideEC, CommonValue);
       } else {
-        NewOperand = widenOperations(Operands, Factor, WideEC, Builder);
+        NewOperand =
+            widenDeinterleavedOperations(Operands, Factor, WideEC, Builder);
       }
     }
     NewOperands.push_back(NewOperand);
@@ -6207,8 +6214,7 @@ static Value *widenOperations(ArrayRef<Value *> Members, unsigned Factor,
       VectorType::get(NarrowInst->getType()->getScalarType(), WideEC);
   Value *NewValue =
       createWideInstruction(NarrowInst, NewOperands, WideResultTy, Builder);
-  auto *NewInst = dyn_cast<Instruction>(NewValue);
-  if (NewInst) {
+  if (auto *NewInst = dyn_cast<Instruction>(NewValue)) {
     propagateIRFlags(NewInst, Members);
     propagateMetadata(NewInst, Members);
   }
@@ -6262,12 +6268,13 @@ bool VectorCombine::foldInterleaveOfDeinterleaveChains(Instruction &I) {
 
   SmallVector<Value *, 8> RootMembers(Interleave->args());
   unsigned NumScanned = 0;
-  if (!canWidenOperations(RootMembers, Factor, NumScanned))
+  if (!canWidenDeinterleavedOperations(RootMembers, Factor, NumScanned))
     return false;
 
   ElementCount WideEC = cast<VectorType>(I.getType())->getElementCount();
   Builder.SetInsertPoint(Interleave);
-  Value *WideValue = widenOperations(RootMembers, Factor, WideEC, Builder);
+  Value *WideValue =
+      widenDeinterleavedOperations(RootMembers, Factor, WideEC, Builder);
   assert(WideValue->getType() == Interleave->getType());
   replaceValue(*Interleave, *WideValue);
   return true;



More information about the llvm-commits mailing list