[llvm] [DAGCombiner] Fold reordered scalar parts to a bitcast and shuffle (PR #226224)

Tim Besard via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 23:59:36 PDT 2026


https://github.com/maleadt updated https://github.com/llvm/llvm-project/pull/226224

>From 2ea41b9f153b47206722be7a74964252fcbbc4d2 Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Thu, 24 Sep 2026 22:14:57 +0200
Subject: [PATCH 1/2] [VectorCombine] Add tests for insertelement chains of
 scalar parts (NFC)

Precommit tests for chains of insertelements whose elements are all
truncated parts of the same scalar, and a PhaseOrdering test for the
loop in which SLP produces such a chain.

Assisted-by: Claude Code, Codex
---
 .../X86/vector-reduction-of-scalar-parts.ll   |  74 +++
 .../AArch64/insert-scalar-parts.ll            | 138 +++++
 .../VectorCombine/X86/insert-scalar-parts.ll  | 554 ++++++++++++++++++
 3 files changed, 766 insertions(+)
 create mode 100644 llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
 create mode 100644 llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
 create mode 100644 llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll

diff --git a/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
new file mode 100644
index 00000000000000..30031078a5548c
--- /dev/null
+++ b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
@@ -0,0 +1,74 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -O2 -S -mtriple=x86_64-- -mcpu=x86-64 | FileCheck %s
+; RUN: opt < %s -O2 -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s
+
+; Sum the two float fields of records that SROA loaded as one i64. SLP
+; vectorizes the two reductions and gathers the fields in reverse order; the
+; gather should become a shuffle of the loaded vector.
+
+define [2 x float] @sum_pairs(ptr %p, i64 %n) {
+; CHECK-LABEL: define [2 x float] @sum_pairs(
+; CHECK-SAME: ptr nofree readonly captures(none) [[P:%.*]], i64 [[N:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[SKIP:%.*]] = icmp slt i64 [[N]], 1
+; CHECK-NEXT:    br i1 [[SKIP]], label %[[EXIT:.*]], label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[PTR:%.*]] = phi ptr [ [[PTR_NEXT:%.*]], %[[LOOP]] ], [ [[P]], %[[ENTRY]] ]
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ [[I_NEXT:%.*]], %[[LOOP]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = phi <2 x float> [ [[TMP6:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
+; CHECK-NEXT:    [[X:%.*]] = load i64, ptr [[PTR]], align 1
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[TMP1:%.*]] = trunc nuw i64 [[HI]] to i32
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x i32> poison, i32 [[TMP1]], i64 0
+; CHECK-NEXT:    [[TMP3:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <2 x i32> [[TMP2]], i32 [[TMP3]], i64 1
+; CHECK-NEXT:    [[TMP5:%.*]] = bitcast <2 x i32> [[TMP4]] to <2 x float>
+; CHECK-NEXT:    [[TMP6]] = fadd <2 x float> [[TMP0]], [[TMP5]]
+; CHECK-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 8
+; CHECK-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = phi <2 x float> [ zeroinitializer, %[[ENTRY]] ], [ [[TMP6]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x float> [[TMP7]], i64 1
+; CHECK-NEXT:    [[R0:%.*]] = insertvalue [2 x float] poison, float [[TMP8]], 0
+; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <2 x float> [[TMP7]], i64 0
+; CHECK-NEXT:    [[R1:%.*]] = insertvalue [2 x float] [[R0]], float [[TMP9]], 1
+; CHECK-NEXT:    ret [2 x float] [[R1]]
+;
+entry:
+  %skip = icmp slt i64 %n, 1
+  br i1 %skip, label %exit, label %loop
+
+loop:
+  %ptr = phi ptr [ %p, %entry ], [ %ptr.next, %loop ]
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  %acc.a = phi float [ 0.0, %entry ], [ %sum.a, %loop ]
+  %acc.b = phi float [ 0.0, %entry ], [ %sum.b, %loop ]
+  %x = load i64, ptr %ptr, align 1
+  %hi = lshr i64 %x, 32
+  %b.i = trunc nuw i64 %hi to i32
+  %a.i = trunc i64 %x to i32
+  %a = bitcast i32 %a.i to float
+  %sum.a = fadd float %acc.a, %a
+  %b = bitcast i32 %b.i to float
+  %sum.b = fadd float %acc.b, %b
+  %ptr.next = getelementptr i8, ptr %ptr, i64 8
+  %i.next = add i64 %i, 1
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop, !llvm.loop !0
+
+exit:
+  %r.b = phi float [ 0.0, %entry ], [ %sum.b, %loop ]
+  %r.a = phi float [ 0.0, %entry ], [ %sum.a, %loop ]
+  %r0 = insertvalue [2 x float] poison, float %r.a, 0
+  %r1 = insertvalue [2 x float] %r0, float %r.b, 1
+  ret [2 x float] %r1
+}
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.unroll.disable"}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.unroll.disable"}
+;.
diff --git a/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll b/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
new file mode 100644
index 00000000000000..2c7d80440ff271
--- /dev/null
+++ b/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
@@ -0,0 +1,138 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=vector-combine -S -mtriple=aarch64 | FileCheck %s --check-prefix=LE
+; RUN: opt < %s -passes=vector-combine -S -mtriple=aarch64_be | FileCheck %s --check-prefix=BE
+
+; The first element of the bitcast scalar is its low part on little-endian
+; targets and its high part on big-endian targets.
+
+define <2 x i32> @high_half_first(i64 %x) {
+; LE-LABEL: define <2 x i32> @high_half_first(
+; LE-SAME: i64 [[X:%.*]]) {
+; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; LE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; LE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; LE-NEXT:    ret <2 x i32> [[V1]]
+;
+; BE-LABEL: define <2 x i32> @high_half_first(
+; BE-SAME: i64 [[X:%.*]]) {
+; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; BE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; BE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; BE-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @low_half_first(i64 %x) {
+; LE-LABEL: define <2 x i32> @low_half_first(
+; LE-SAME: i64 [[X:%.*]]) {
+; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; LE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T0]], i64 0
+; LE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; LE-NEXT:    ret <2 x i32> [[V1]]
+;
+; BE-LABEL: define <2 x i32> @low_half_first(
+; BE-SAME: i64 [[X:%.*]]) {
+; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; BE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T0]], i64 0
+; BE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; BE-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t0, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t1, i64 1
+  ret <2 x i32> %v1
+}
+
+define <4 x i16> @reversed_quarters(i64 %x) {
+; LE-LABEL: define <4 x i16> @reversed_quarters(
+; LE-SAME: i64 [[X:%.*]]) {
+; LE-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
+; LE-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; LE-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
+; LE-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; LE-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; LE-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; LE-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
+; LE-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
+; LE-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
+; LE-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; LE-NEXT:    ret <4 x i16> [[V3]]
+;
+; BE-LABEL: define <4 x i16> @reversed_quarters(
+; BE-SAME: i64 [[X:%.*]]) {
+; BE-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
+; BE-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; BE-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
+; BE-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; BE-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; BE-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; BE-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
+; BE-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
+; BE-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
+; BE-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; BE-NEXT:    ret <4 x i16> [[V3]]
+;
+  %s1 = lshr i64 %x, 16
+  %s2 = lshr i64 %x, 32
+  %s3 = lshr i64 %x, 48
+  %t0 = trunc i64 %x to i16
+  %t1 = trunc i64 %s1 to i16
+  %t2 = trunc i64 %s2 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <4 x i16> poison, i16 %t3, i64 0
+  %v1 = insertelement <4 x i16> %v0, i16 %t2, i64 1
+  %v2 = insertelement <4 x i16> %v1, i16 %t1, i64 2
+  %v3 = insertelement <4 x i16> %v2, i16 %t0, i64 3
+  ret <4 x i16> %v3
+}
+
+define <4 x i32> @repeated_halves(i64 %x) {
+; LE-LABEL: define <4 x i32> @repeated_halves(
+; LE-SAME: i64 [[X:%.*]]) {
+; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; LE-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
+; LE-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
+; LE-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
+; LE-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; LE-NEXT:    ret <4 x i32> [[V3]]
+;
+; BE-LABEL: define <4 x i32> @repeated_halves(
+; BE-SAME: i64 [[X:%.*]]) {
+; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; BE-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
+; BE-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
+; BE-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
+; BE-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; BE-NEXT:    ret <4 x i32> [[V3]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <4 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %t0, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %t1, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %t0, i64 3
+  ret <4 x i32> %v3
+}
diff --git a/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll b/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll
new file mode 100644
index 00000000000000..82d2766983b54c
--- /dev/null
+++ b/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll
@@ -0,0 +1,554 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=+sse2 | FileCheck %s --check-prefixes=CHECK,SSE
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s --check-prefixes=CHECK,AVX
+
+; Insertelement chains whose elements are all parts of the same scalar can be
+; a bitcast of the scalar and a shuffle.
+
+define <2 x i32> @swapped_halves(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @swapped_halves(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x float> @swapped_halves_float(i64 %x) {
+; CHECK-LABEL: define <2 x float> @swapped_halves_float(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[F1:%.*]] = bitcast i32 [[T1]] to float
+; CHECK-NEXT:    [[F0:%.*]] = bitcast i32 [[T0]] to float
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x float> poison, float [[F1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x float> [[V0]], float [[F0]], i64 1
+; CHECK-NEXT:    ret <2 x float> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %f1 = bitcast i32 %t1 to float
+  %f0 = bitcast i32 %t0 to float
+  %v0 = insertelement <2 x float> poison, float %f1, i64 0
+  %v1 = insertelement <2 x float> %v0, float %f0, i64 1
+  ret <2 x float> %v1
+}
+
+define <4 x i16> @reversed_quarters(i64 %x) {
+; CHECK-LABEL: define <4 x i16> @reversed_quarters(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
+; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; CHECK-NEXT:    ret <4 x i16> [[V3]]
+;
+  %s1 = lshr i64 %x, 16
+  %s2 = lshr i64 %x, 32
+  %s3 = lshr i64 %x, 48
+  %t0 = trunc i64 %x to i16
+  %t1 = trunc i64 %s1 to i16
+  %t2 = trunc i64 %s2 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <4 x i16> poison, i16 %t3, i64 0
+  %v1 = insertelement <4 x i16> %v0, i16 %t2, i64 1
+  %v2 = insertelement <4 x i16> %v1, i16 %t1, i64 2
+  %v3 = insertelement <4 x i16> %v2, i16 %t0, i64 3
+  ret <4 x i16> %v3
+}
+
+define <4 x i32> @reversed_i128(i128 %x) {
+; CHECK-LABEL: define <4 x i32> @reversed_i128(
+; CHECK-SAME: i128 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S1:%.*]] = lshr i128 [[X]], 32
+; CHECK-NEXT:    [[S2:%.*]] = lshr i128 [[X]], 64
+; CHECK-NEXT:    [[S3:%.*]] = lshr i128 [[X]], 96
+; CHECK-NEXT:    [[T0:%.*]] = trunc i128 [[X]] to i32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i128 [[S1]] to i32
+; CHECK-NEXT:    [[T2:%.*]] = trunc i128 [[S2]] to i32
+; CHECK-NEXT:    [[T3:%.*]] = trunc i128 [[S3]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T3]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T2]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
+; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; CHECK-NEXT:    ret <4 x i32> [[V3]]
+;
+  %s1 = lshr i128 %x, 32
+  %s2 = lshr i128 %x, 64
+  %s3 = lshr i128 %x, 96
+  %t0 = trunc i128 %x to i32
+  %t1 = trunc i128 %s1 to i32
+  %t2 = trunc i128 %s2 to i32
+  %t3 = trunc i128 %s3 to i32
+  %v0 = insertelement <4 x i32> poison, i32 %t3, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %t2, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %t1, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %t0, i64 3
+  ret <4 x i32> %v3
+}
+
+; The scalar has fewer parts than the vector has elements.
+define <4 x i32> @repeated_halves(i64 %x) {
+; CHECK-LABEL: define <4 x i32> @repeated_halves(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
+; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; CHECK-NEXT:    ret <4 x i32> [[V3]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <4 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %t0, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %t1, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %t0, i64 3
+  ret <4 x i32> %v3
+}
+
+; The scalar has more parts than the vector has elements.
+define <2 x i16> @odd_quarters(i64 %x) {
+; CHECK-LABEL: define <2 x i16> @odd_quarters(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i16> poison, i16 [[T3]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i16> [[V0]], i16 [[T1]], i64 1
+; CHECK-NEXT:    ret <2 x i16> [[V1]]
+;
+  %s1 = lshr i64 %x, 16
+  %s3 = lshr i64 %x, 48
+  %t1 = trunc i64 %s1 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <2 x i16> poison, i16 %t3, i64 0
+  %v1 = insertelement <2 x i16> %v0, i16 %t1, i64 1
+  ret <2 x i16> %v1
+}
+
+define <4 x i16> @in_order_quarters(i64 %x) {
+; CHECK-LABEL: define <4 x i16> @in_order_quarters(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T0]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T1]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T2]], i64 2
+; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T3]], i64 3
+; CHECK-NEXT:    ret <4 x i16> [[V3]]
+;
+  %s1 = lshr i64 %x, 16
+  %s2 = lshr i64 %x, 32
+  %s3 = lshr i64 %x, 48
+  %t0 = trunc i64 %x to i16
+  %t1 = trunc i64 %s1 to i16
+  %t2 = trunc i64 %s2 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <4 x i16> poison, i16 %t0, i64 0
+  %v1 = insertelement <4 x i16> %v0, i16 %t1, i64 1
+  %v2 = insertelement <4 x i16> %v1, i16 %t2, i64 2
+  %v3 = insertelement <4 x i16> %v2, i16 %t3, i64 3
+  ret <4 x i16> %v3
+}
+
+; Elements that are not inserted stay poison.
+define <4 x i16> @missing_elts(i64 %x) {
+; CHECK-LABEL: define <4 x i16> @missing_elts(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 2
+; CHECK-NEXT:    ret <4 x i16> [[V1]]
+;
+  %s2 = lshr i64 %x, 32
+  %s3 = lshr i64 %x, 48
+  %t2 = trunc i64 %s2 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <4 x i16> poison, i16 %t3, i64 0
+  %v1 = insertelement <4 x i16> %v0, i16 %t2, i64 2
+  ret <4 x i16> %v1
+}
+
+define <2 x i32> @undef_base(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @undef_base(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> undef, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> undef, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @splat_high_half(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @splat_high_half(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t1, i64 1
+  ret <2 x i32> %v1
+}
+
+; Only the last insert to an element counts.
+define <2 x i32> @overwritten_elt(i64 %x, i32 %y) {
+; CHECK-LABEL: define <2 x i32> @overwritten_elt(
+; CHECK-SAME: i64 [[X:%.*]], i32 [[Y:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[Y]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <2 x i32> [[V1]], i32 [[T1]], i64 0
+; CHECK-NEXT:    ret <2 x i32> [[V2]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %y, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  %v2 = insertelement <2 x i32> %v1, i32 %t1, i64 0
+  ret <2 x i32> %v2
+}
+
+define <2 x i32> @double_source(double %d) {
+; CHECK-LABEL: define <2 x i32> @double_source(
+; CHECK-SAME: double [[D:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[X:%.*]] = bitcast double [[D]] to i64
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %x = bitcast double %d to i64
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+; Every element is inserted, so the base does not matter.
+define <2 x i32> @overwritten_base(i64 %x, <2 x i32> %base) {
+; CHECK-LABEL: define <2 x i32> @overwritten_base(
+; CHECK-SAME: i64 [[X:%.*]], <2 x i32> [[BASE:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> [[BASE]], i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> %base, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @swapped_halves_extra_use(i64 %x, ptr %p) {
+; CHECK-LABEL: define <2 x i32> @swapped_halves_extra_use(
+; CHECK-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    store i32 [[T1]], ptr [[P]], align 4
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  store i32 %t1, ptr %p
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+; Negative tests
+
+define <2 x i32> @different_sources(i64 %x, i64 %y) {
+; CHECK-LABEL: define <2 x i32> @different_sources(
+; CHECK-SAME: i64 [[X:%.*]], i64 [[Y:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[Y]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %y to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @unaligned_shift(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @unaligned_shift(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 16
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 16
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @ashr(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @ashr(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = ashr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = ashr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @variable_shift(i64 %x, i64 %s) {
+; CHECK-LABEL: define <2 x i32> @variable_shift(
+; CHECK-SAME: i64 [[X:%.*]], i64 [[S:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], [[S]]
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, %s
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @variable_index(i64 %x, i64 %i) {
+; CHECK-LABEL: define <2 x i32> @variable_index(
+; CHECK-SAME: i64 [[X:%.*]], i64 [[I:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 [[I]]
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 %i
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @non_poison_base(i64 %x, <2 x i32> %base) {
+; CHECK-LABEL: define <2 x i32> @non_poison_base(
+; CHECK-SAME: i64 [[X:%.*]], <2 x i32> [[BASE:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> [[BASE]], i32 [[T1]], i64 0
+; CHECK-NEXT:    ret <2 x i32> [[V0]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %v0 = insertelement <2 x i32> %base, i32 %t1, i64 0
+  ret <2 x i32> %v0
+}
+
+; The elements that are not inserted would become poison instead of undef.
+define <4 x i16> @undef_base_missing_elts(i64 %x) {
+; CHECK-LABEL: define <4 x i16> @undef_base_missing_elts(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> undef, i16 [[T3]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 2
+; CHECK-NEXT:    ret <4 x i16> [[V1]]
+;
+  %s2 = lshr i64 %x, 32
+  %s3 = lshr i64 %x, 48
+  %t2 = trunc i64 %s2 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <4 x i16> undef, i16 %t3, i64 0
+  %v1 = insertelement <4 x i16> %v0, i16 %t2, i64 2
+  ret <4 x i16> %v1
+}
+
+define <2 x i32> @out_of_range_index(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @out_of_range_index(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 2
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 2
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @out_of_range_shift(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @out_of_range_shift(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 64
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 64
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+; The intermediate vector is used elsewhere, so the chain ends there.
+define <2 x i32> @extra_use_of_insert(i64 %x, ptr %p) {
+; CHECK-LABEL: define <2 x i32> @extra_use_of_insert(
+; CHECK-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    store <2 x i32> [[V0]], ptr [[P]], align 8
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  store <2 x i32> %v0, ptr %p
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+; The scalar is not a whole number of elements.
+define <2 x i32> @partial_source(i48 %x) {
+; CHECK-LABEL: define <2 x i32> @partial_source(
+; CHECK-SAME: i48 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i48 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i48 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i48 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i48 %x, 32
+  %t1 = trunc i48 %hi to i32
+  %t0 = trunc i48 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+; Elements are truncated from each element of a vector, not from its bits.
+define <2 x i32> @vector_trunc(<2 x i32> %x) {
+; CHECK-LABEL: define <2 x i32> @vector_trunc(
+; CHECK-SAME: <2 x i32> [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[T:%.*]] = trunc <2 x i32> [[X]] to <2 x i16>
+; CHECK-NEXT:    [[B:%.*]] = bitcast <2 x i16> [[T]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[B]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V0]]
+;
+  %t = trunc <2 x i32> %x to <2 x i16>
+  %b = bitcast <2 x i16> %t to i32
+  %v0 = insertelement <2 x i32> poison, i32 %b, i64 1
+  ret <2 x i32> %v0
+}
+
+define <8 x i1> @bool_elts(i8 %x) {
+; CHECK-LABEL: define <8 x i1> @bool_elts(
+; CHECK-SAME: i8 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S1:%.*]] = lshr i8 [[X]], 1
+; CHECK-NEXT:    [[T1:%.*]] = trunc i8 [[S1]] to i1
+; CHECK-NEXT:    [[T0:%.*]] = trunc i8 [[X]] to i1
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <8 x i1> poison, i1 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <8 x i1> [[V0]], i1 [[T0]], i64 1
+; CHECK-NEXT:    ret <8 x i1> [[V1]]
+;
+  %s1 = lshr i8 %x, 1
+  %t1 = trunc i8 %s1 to i1
+  %t0 = trunc i8 %x to i1
+  %v0 = insertelement <8 x i1> poison, i1 %t1, i64 0
+  %v1 = insertelement <8 x i1> %v0, i1 %t0, i64 1
+  ret <8 x i1> %v1
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; AVX: {{.*}}
+; SSE: {{.*}}

>From 459001730230d73e3652c5c11b3037dd03ae359c Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Thu, 24 Sep 2026 22:34:22 +0200
Subject: [PATCH 2/2] [VectorCombine] Fold insertelement chains of scalar parts
 to a bitcast and shuffle

An insertelement chain whose elements are all truncated parts of the
same scalar is lowered element by element, unless InstCombine can turn
an in-order pair of halves into a bitcast. The SLP vectorizer produces
such chains for the fields of a struct that SROA loaded as one integer,
e.g. when summing two float fields in a loop:

  %hi = lshr i64 %x, 32
  %h = trunc i64 %hi to i32
  %l = trunc i64 %x to i32
  %v0 = insertelement <2 x i32> poison, i32 %h, i64 0
  %v1 = insertelement <2 x i32> %v0, i32 %l, i64 1

which X86 lowers to shrq + vmovd + vpinsrd. If TTI says it is cheaper,
replace the chain by a shuffle of the bitcast scalar:

  %b = bitcast i64 %x to <2 x i32>
  %v1 = shufflevector <2 x i32> %b, <2 x i32> poison, <2 x i32> <i32 1, i32 0>

The elements may be bitcast to FP, the scalar may have more or fewer
parts than the vector has elements, and big-endian targets are handled.
Elements that are not inserted become poison, so the chain must start
from poison unless it inserts every element.

Assisted-by: Claude Code, Codex
---
 .../Transforms/Vectorize/VectorCombine.cpp    | 111 ++++++++
 .../AArch64/block_scaling_decompr_8bit.ll     |   5 +-
 llvm/test/Transforms/PhaseOrdering/X86/avg.ll | 259 ++++++++----------
 .../X86/vector-reduction-of-scalar-parts.ll   |   9 +-
 .../AArch64/insert-scalar-parts.ll            |  69 +----
 .../VectorCombine/X86/insert-scalar-parts.ll  | 185 +++++--------
 6 files changed, 313 insertions(+), 325 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index ddb06610d6572f..95d1adac12350d 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -125,6 +125,7 @@ class VectorCombine {
   bool foldInsExtFNeg(Instruction &I);
   bool foldInsExtBinop(Instruction &I);
   bool foldInsExtVectorToShuffle(Instruction &I);
+  bool foldInsertScalarPartsToShuffle(Instruction &I);
   bool foldBitOpOfCastops(Instruction &I);
   bool foldBitOpOfCastConstant(Instruction &I);
   bool foldBitcastShuffle(Instruction &I);
@@ -6036,6 +6037,114 @@ bool VectorCombine::foldInsExtVectorToShuffle(Instruction &I) {
   return true;
 }
 
+/// Try to replace a chain of insertelements of parts of the same scalar with a
+/// bitcast and a shuffle (little endian):
+///   insert (insert poison, (trunc (lshr X, 32)), 0), (trunc X), 1 -->
+///   shuffle (bitcast X to <2 x i32>), poison, <1, 0>
+bool VectorCombine::foldInsertScalarPartsToShuffle(Instruction &I) {
+  auto *VecTy = dyn_cast<FixedVectorType>(I.getType());
+  if (!VecTy)
+    return false;
+
+  // Start from the last insertelement of the chain.
+  if (I.hasOneUse() && isa<InsertElementInst>(I.user_back()) &&
+      I.user_back()->getOperand(0) == &I)
+    return false;
+
+  Type *EltTy = VecTy->getElementType();
+  if ((!EltTy->isIntegerTy() && !EltTy->isIEEELikeFPTy()) ||
+      !DL->typeSizeEqualsStoreSize(EltTy))
+    return false;
+  unsigned EltBits = EltTy->getPrimitiveSizeInBits();
+  unsigned NumElts = VecTy->getNumElements();
+
+  Value *Src = nullptr;
+  unsigned NumSrcElts = 0;
+  SmallVector<int> Mask(NumElts, PoisonMaskElem);
+  APInt DemandedElts = APInt::getZero(NumElts);
+  InstructionCost OldCost = 0;
+  Value *Vec = &I;
+  while (auto *Ins = dyn_cast<InsertElementInst>(Vec)) {
+    if (Ins != &I && !Ins->hasOneUse())
+      return false;
+    uint64_t Idx;
+    if (!match(Ins->getOperand(2), m_ConstantInt(Idx)) || Idx >= NumElts)
+      return false;
+    Vec = Ins->getOperand(0);
+    // A later insert to the same element overrides this one.
+    if (DemandedElts[Idx])
+      continue;
+    DemandedElts.setBit(Idx);
+
+    // Match (bitcast (trunc (lshr X, ShAmt))), the bitcast and shift being
+    // optional.
+    Value *Elt = Ins->getOperand(1);
+    Value *Trunc = Elt;
+    match(Trunc, m_BitCast(m_Value(Trunc)));
+    Value *X;
+    if (!match(Trunc, m_Trunc(m_Value(X))) || !X->getType()->isIntegerTy() ||
+        Trunc->getType()->getPrimitiveSizeInBits() != EltBits)
+      return false;
+    Value *Shift = nullptr;
+    uint64_t ShAmt = 0;
+    if (match(X, m_LShr(m_Value(), m_ConstantInt(ShAmt)))) {
+      Shift = X;
+      X = cast<Instruction>(Shift)->getOperand(0);
+    }
+
+    if (!Src) {
+      unsigned SrcBits = X->getType()->getIntegerBitWidth();
+      if (SrcBits % EltBits)
+        return false;
+      Src = X;
+      NumSrcElts = SrcBits / EltBits;
+    } else if (X != Src) {
+      return false;
+    }
+    if (ShAmt % EltBits || ShAmt / EltBits >= NumSrcElts)
+      return false;
+    unsigned Part = ShAmt / EltBits;
+    Mask[Idx] = DL->isBigEndian() ? NumSrcElts - 1 - Part : Part;
+
+    // The scalar ops die with the chain if it is their only user.
+    for (Value *V : {Elt == Trunc ? nullptr : Elt, Trunc, Shift}) {
+      if (!V)
+        continue;
+      if (!V->hasOneUse())
+        break;
+      OldCost += TTI.getInstructionCost(cast<Instruction>(V), CostKind);
+    }
+  }
+  // Elements that are not inserted become poison, so the base must be poison
+  // unless every element is inserted.
+  if (!Src || (!isa<PoisonValue>(Vec) && !DemandedElts.isAllOnes()))
+    return false;
+
+  OldCost += TTI.getScalarizationOverhead(VecTy, DemandedElts, /*Insert=*/true,
+                                          /*Extract=*/false, CostKind);
+
+  auto *SrcVecTy = FixedVectorType::get(EltTy, NumSrcElts);
+  InstructionCost NewCost =
+      TTI.getCastInstrCost(Instruction::BitCast, SrcVecTy, Src->getType(),
+                           TTI::CastContextHint::None, CostKind);
+  bool IsIdentity = NumSrcElts == NumElts &&
+                    ShuffleVectorInst::isIdentityMask(Mask, NumSrcElts);
+  if (!IsIdentity)
+    NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, VecTy, SrcVecTy,
+                                  CostKind, Mask);
+
+  LLVM_DEBUG(dbgs() << "Found an insertelement chain of scalar parts: " << I
+                    << "\n  OldCost: " << OldCost << " vs NewCost: " << NewCost
+                    << "\n");
+  if (!OldCost.isValid() || !NewCost.isValid() || NewCost >= OldCost)
+    return false;
+
+  Value *Cast = Builder.CreateBitCast(Src, SrcVecTy);
+  Value *Shuf = IsIdentity ? Cast : Builder.CreateShuffleVector(Cast, Mask);
+  replaceValue(I, *Shuf);
+  return true;
+}
+
 /// Fold away a matched pair of vector.deinterleave/interleave intrinsics
 /// with a chain of elementwise operations on each between the
 /// deinterleave and interleave.
@@ -6953,6 +7062,8 @@ bool VectorCombine::run() {
           return true;
         if (foldInsExtVectorToShuffle(I))
           return true;
+        if (foldInsertScalarPartsToShuffle(I))
+          return true;
         break;
       case Instruction::ShuffleVector:
         if (foldPermuteOfBinops(I))
diff --git a/llvm/test/Transforms/PhaseOrdering/AArch64/block_scaling_decompr_8bit.ll b/llvm/test/Transforms/PhaseOrdering/AArch64/block_scaling_decompr_8bit.ll
index 381fba8257394c..a78937cec54f37 100644
--- a/llvm/test/Transforms/PhaseOrdering/AArch64/block_scaling_decompr_8bit.ll
+++ b/llvm/test/Transforms/PhaseOrdering/AArch64/block_scaling_decompr_8bit.ll
@@ -413,17 +413,16 @@ define internal noundef <8 x i16> @_ZL24cmplx_mul_combined_re_im11__Int16x8_t20c
 ; CHECK-LABEL: define internal fastcc noundef <8 x i16> @_ZL24cmplx_mul_combined_re_im11__Int16x8_t20cmplx_int16_t(
 ; CHECK-SAME: <8 x i16> noundef [[A:%.*]], i64 [[SCALE_COERCE:%.*]]) unnamed_addr #[[ATTR1:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[SCALE_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[SCALE_COERCE]] to i16
 ; CHECK-NEXT:    [[SCALE_SROA_2_0_EXTRACT_SHIFT36:%.*]] = lshr i64 [[SCALE_COERCE]], 16
 ; CHECK-NEXT:    [[SCALE_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[SCALE_SROA_2_0_EXTRACT_SHIFT36]] to i16
 ; CHECK-NEXT:    [[SHUFFLE_I:%.*]] = shufflevector <8 x i16> [[A]], <8 x i16> poison, <8 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6>
-; CHECK-NEXT:    [[VECINIT_I19:%.*]] = insertelement <8 x i16> poison, i16 [[SCALE_SROA_0_0_EXTRACT_TRUNC]], i64 0
+; CHECK-NEXT:    [[TMP2:%.*]] = bitcast i64 [[SCALE_COERCE]] to <4 x i16>
 ; CHECK-NEXT:    [[VECINIT_I:%.*]] = insertelement <8 x i16> poison, i16 [[SCALE_SROA_2_0_EXTRACT_TRUNC]], i64 0
 ; CHECK-NEXT:    [[VECINIT7_I:%.*]] = shufflevector <8 x i16> [[VECINIT_I]], <8 x i16> poison, <8 x i32> zeroinitializer
 ; CHECK-NEXT:    [[VQNEGQ_V1_I:%.*]] = tail call <8 x i16> @llvm.aarch64.neon.sqneg.v8i16(<8 x i16> [[VECINIT7_I]])
 ; CHECK-NEXT:    [[VBSL5_I:%.*]] = shufflevector <8 x i16> [[VQNEGQ_V1_I]], <8 x i16> [[VECINIT_I]], <8 x i32> <i32 0, i32 8, i32 2, i32 8, i32 4, i32 8, i32 6, i32 8>
 ; CHECK-NEXT:    [[SHUFFLE_I85:%.*]] = shufflevector <8 x i16> [[A]], <8 x i16> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[SHUFFLE_I82:%.*]] = shufflevector <8 x i16> [[VECINIT_I19]], <8 x i16> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[SHUFFLE_I82:%.*]] = shufflevector <4 x i16> [[TMP2]], <4 x i16> poison, <4 x i32> zeroinitializer
 ; CHECK-NEXT:    [[VQDMULL_V2_I72:%.*]] = tail call <4 x i32> @llvm.aarch64.neon.sqdmull.v4i32(<4 x i16> [[SHUFFLE_I85]], <4 x i16> [[SHUFFLE_I82]])
 ; CHECK-NEXT:    [[SHUFFLE_I97:%.*]] = shufflevector <8 x i16> [[A]], <8 x i16> poison, <4 x i32> <i32 4, i32 5, i32 6, i32 7>
 ; CHECK-NEXT:    [[VQDMULL_V2_I:%.*]] = tail call <4 x i32> @llvm.aarch64.neon.sqdmull.v4i32(<4 x i16> [[SHUFFLE_I97]], <4 x i16> [[SHUFFLE_I82]])
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
index 8f8899d285ad37..403952c9b591e8 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64    | FileCheck %s --check-prefixes=CHECK,SSE2
 ; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s --check-prefixes=CHECK,SSE4
 ; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
@@ -14,10 +14,11 @@
 %"struct.std::array16" = type { [16 x i8] }
 
 define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0, i64 %b.coerce1) {
-; SSE2-LABEL: @avgr_16_u8(
-; SSE2-NEXT:  entry:
-; SSE2-NEXT:    [[TMP1:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; SSE2-NEXT:    [[TMP2:%.*]] = insertelement <2 x i64> [[TMP1]], i64 [[A_COERCE1:%.*]], i64 1
+; SSE2-LABEL: define { i64, i64 } @avgr_16_u8(
+; SSE2-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; SSE2-NEXT:  [[ENTRY:.*:]]
+; SSE2-NEXT:    [[TMP1:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0]], i64 0
+; SSE2-NEXT:    [[TMP2:%.*]] = insertelement <2 x i64> [[TMP1]], i64 [[A_COERCE1]], i64 1
 ; SSE2-NEXT:    [[TMP7:%.*]] = trunc <2 x i64> [[TMP2]] to <2 x i16>
 ; SSE2-NEXT:    [[TMP3:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 16)
 ; SSE2-NEXT:    [[TMP4:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 24)
@@ -25,8 +26,8 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE2-NEXT:    [[TMP6:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 40)
 ; SSE2-NEXT:    [[TMP45:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 48)
 ; SSE2-NEXT:    [[TMP46:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 56)
-; SSE2-NEXT:    [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; SSE2-NEXT:    [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[B_COERCE1:%.*]], i64 1
+; SSE2-NEXT:    [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0]], i64 0
+; SSE2-NEXT:    [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[B_COERCE1]], i64 1
 ; SSE2-NEXT:    [[TMP15:%.*]] = trunc <2 x i64> [[TMP10]] to <2 x i16>
 ; SSE2-NEXT:    [[TMP11:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 16)
 ; SSE2-NEXT:    [[TMP12:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 24)
@@ -93,9 +94,10 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE2-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP69]], 1
 ; SSE2-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
-; SSE4-LABEL: @avgr_16_u8(
-; SSE4-NEXT:  entry:
-; SSE4-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i16
+; SSE4-LABEL: define { i64, i64 } @avgr_16_u8(
+; SSE4-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; SSE4-NEXT:  [[ENTRY:.*:]]
+; SSE4-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0]] to i16
 ; SSE4-NEXT:    [[TMP1:%.*]] = lshr i16 [[TMP0]], 8
 ; SSE4-NEXT:    [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
 ; SSE4-NEXT:    [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 24
@@ -103,7 +105,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[A_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 40
 ; SSE4-NEXT:    [[A_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 48
 ; SSE4-NEXT:    [[A_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 56
-; SSE4-NEXT:    [[TMP2:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i16
+; SSE4-NEXT:    [[TMP2:%.*]] = trunc i64 [[A_COERCE1]] to i16
 ; SSE4-NEXT:    [[TMP3:%.*]] = lshr i16 [[TMP2]], 8
 ; SSE4-NEXT:    [[A_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 16
 ; SSE4-NEXT:    [[A_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 24
@@ -111,7 +113,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[A_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 40
 ; SSE4-NEXT:    [[A_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 48
 ; SSE4-NEXT:    [[A_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 56
-; SSE4-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i16
+; SSE4-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0]] to i16
 ; SSE4-NEXT:    [[TMP5:%.*]] = lshr i16 [[TMP4]], 8
 ; SSE4-NEXT:    [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
 ; SSE4-NEXT:    [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 24
@@ -119,7 +121,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[B_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 40
 ; SSE4-NEXT:    [[B_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 48
 ; SSE4-NEXT:    [[B_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 56
-; SSE4-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
+; SSE4-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1]] to i16
 ; SSE4-NEXT:    [[TMP7:%.*]] = lshr i16 [[TMP6]], 8
 ; SSE4-NEXT:    [[B_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
 ; SSE4-NEXT:    [[B_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 24
@@ -213,15 +215,16 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP69]], 1
 ; SSE4-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
-; AVX2-LABEL: @avgr_16_u8(
-; AVX2-NEXT:  entry:
-; AVX2-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i16
+; AVX2-LABEL: define { i64, i64 } @avgr_16_u8(
+; AVX2-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; AVX2-NEXT:  [[ENTRY:.*:]]
+; AVX2-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0]] to i16
 ; AVX2-NEXT:    [[TMP1:%.*]] = insertelement <2 x i16> poison, i16 [[TMP0]], i64 0
-; AVX2-NEXT:    [[TMP2:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i16
+; AVX2-NEXT:    [[TMP2:%.*]] = trunc i64 [[A_COERCE1]] to i16
 ; AVX2-NEXT:    [[TMP3:%.*]] = insertelement <2 x i16> [[TMP1]], i16 [[TMP2]], i64 1
-; AVX2-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i16
+; AVX2-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0]] to i16
 ; AVX2-NEXT:    [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
-; AVX2-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
+; AVX2-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1]] to i16
 ; AVX2-NEXT:    [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
 ; AVX2-NEXT:    [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
 ; AVX2-NEXT:    [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
@@ -275,15 +278,16 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; AVX2-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP43]], 1
 ; AVX2-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
-; AVX512-LABEL: @avgr_16_u8(
-; AVX512-NEXT:  entry:
-; AVX512-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i16
+; AVX512-LABEL: define { i64, i64 } @avgr_16_u8(
+; AVX512-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; AVX512-NEXT:  [[ENTRY:.*:]]
+; AVX512-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0]] to i16
 ; AVX512-NEXT:    [[TMP1:%.*]] = insertelement <2 x i16> poison, i16 [[TMP0]], i64 0
-; AVX512-NEXT:    [[TMP2:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i16
+; AVX512-NEXT:    [[TMP2:%.*]] = trunc i64 [[A_COERCE1]] to i16
 ; AVX512-NEXT:    [[TMP3:%.*]] = insertelement <2 x i16> [[TMP1]], i16 [[TMP2]], i64 1
-; AVX512-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i16
+; AVX512-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0]] to i16
 ; AVX512-NEXT:    [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
-; AVX512-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
+; AVX512-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1]] to i16
 ; AVX512-NEXT:    [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
 ; AVX512-NEXT:    [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
 ; AVX512-NEXT:    [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
@@ -380,9 +384,10 @@ for.body:                                         ; preds = %for.cond
 }
 
 define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0, i64 %b.coerce1) {
-; SSE2-LABEL: @avgr_16_u8_alt(
-; SSE2-NEXT:  entry:
-; SSE2-NEXT:    [[A_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i8
+; SSE2-LABEL: define { i64, i64 } @avgr_16_u8_alt(
+; SSE2-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; SSE2-NEXT:  [[ENTRY:.*:]]
+; SSE2-NEXT:    [[A_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_COERCE0]] to i8
 ; SSE2-NEXT:    [[A_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 8
 ; SSE2-NEXT:    [[A_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_2_0_EXTRACT_SHIFT]] to i8
 ; SSE2-NEXT:    [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
@@ -397,7 +402,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; SSE2-NEXT:    [[A_SROA_7_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_7_0_EXTRACT_SHIFT]] to i8
 ; SSE2-NEXT:    [[A_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 56
 ; SSE2-NEXT:    [[A_SROA_8_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_8_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i8
+; SSE2-NEXT:    [[A_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_COERCE1]] to i8
 ; SSE2-NEXT:    [[A_SROA_11_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 8
 ; SSE2-NEXT:    [[A_SROA_11_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_11_8_EXTRACT_SHIFT]] to i8
 ; SSE2-NEXT:    [[A_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 16
@@ -412,7 +417,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; SSE2-NEXT:    [[A_SROA_16_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_16_8_EXTRACT_SHIFT]] to i8
 ; SSE2-NEXT:    [[A_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 56
 ; SSE2-NEXT:    [[A_SROA_17_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_17_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i8
+; SSE2-NEXT:    [[B_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_COERCE0]] to i8
 ; SSE2-NEXT:    [[B_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 8
 ; SSE2-NEXT:    [[B_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_2_0_EXTRACT_SHIFT]] to i8
 ; SSE2-NEXT:    [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
@@ -427,7 +432,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; SSE2-NEXT:    [[B_SROA_7_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_7_0_EXTRACT_SHIFT]] to i8
 ; SSE2-NEXT:    [[B_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 56
 ; SSE2-NEXT:    [[B_SROA_8_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_8_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i8
+; SSE2-NEXT:    [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_COERCE1]] to i8
 ; SSE2-NEXT:    [[B_SROA_11_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 8
 ; SSE2-NEXT:    [[B_SROA_11_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_11_8_EXTRACT_SHIFT]] to i8
 ; SSE2-NEXT:    [[B_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
@@ -586,13 +591,14 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; SSE2-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[RETVAL_SROA_9_8_INSERT_INSERT]], 1
 ; SSE2-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
-; SSE4-LABEL: @avgr_16_u8_alt(
-; SSE4-NEXT:  entry:
-; SSE4-NEXT:    [[TMP0:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; SSE4-LABEL: define { i64, i64 } @avgr_16_u8_alt(
+; SSE4-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; SSE4-NEXT:  [[ENTRY:.*:]]
+; SSE4-NEXT:    [[TMP0:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0]], i64 0
 ; SSE4-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i64> [[TMP0]], <8 x i64> poison, <8 x i32> zeroinitializer
 ; SSE4-NEXT:    [[TMP2:%.*]] = lshr <8 x i64> [[TMP1]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
 ; SSE4-NEXT:    [[TMP3:%.*]] = trunc <8 x i64> [[TMP2]] to <8 x i8>
-; SSE4-NEXT:    [[TMP4:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; SSE4-NEXT:    [[TMP4:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE0]], i64 0
 ; SSE4-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i64> [[TMP4]], <8 x i64> poison, <8 x i32> zeroinitializer
 ; SSE4-NEXT:    [[TMP6:%.*]] = lshr <8 x i64> [[TMP5]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
 ; SSE4-NEXT:    [[TMP7:%.*]] = trunc <8 x i64> [[TMP6]] to <8 x i8>
@@ -604,11 +610,11 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; SSE4-NEXT:    [[TMP13:%.*]] = add nuw <8 x i8> [[TMP10]], [[TMP12]]
 ; SSE4-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = bitcast <8 x i8> [[TMP13]] to i64
 ; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
-; SSE4-NEXT:    [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
+; SSE4-NEXT:    [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1]], i64 0
 ; SSE4-NEXT:    [[TMP18:%.*]] = shufflevector <8 x i64> [[TMP17]], <8 x i64> poison, <8 x i32> zeroinitializer
 ; SSE4-NEXT:    [[TMP19:%.*]] = lshr <8 x i64> [[TMP18]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
 ; SSE4-NEXT:    [[TMP20:%.*]] = trunc <8 x i64> [[TMP19]] to <8 x i8>
-; SSE4-NEXT:    [[TMP21:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE1:%.*]], i64 0
+; SSE4-NEXT:    [[TMP21:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE1]], i64 0
 ; SSE4-NEXT:    [[TMP22:%.*]] = shufflevector <8 x i64> [[TMP21]], <8 x i64> poison, <8 x i32> zeroinitializer
 ; SSE4-NEXT:    [[TMP23:%.*]] = lshr <8 x i64> [[TMP22]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
 ; SSE4-NEXT:    [[TMP24:%.*]] = trunc <8 x i64> [[TMP23]] to <8 x i8>
@@ -622,13 +628,14 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP49]], 1
 ; SSE4-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
-; AVX-LABEL: @avgr_16_u8_alt(
-; AVX-NEXT:  entry:
-; AVX-NEXT:    [[TMP0:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; AVX-LABEL: define { i64, i64 } @avgr_16_u8_alt(
+; AVX-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; AVX-NEXT:  [[ENTRY:.*:]]
+; AVX-NEXT:    [[TMP0:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0]], i64 0
 ; AVX-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i64> [[TMP0]], <8 x i64> poison, <8 x i32> zeroinitializer
 ; AVX-NEXT:    [[TMP2:%.*]] = lshr <8 x i64> [[TMP1]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
 ; AVX-NEXT:    [[TMP3:%.*]] = trunc <8 x i64> [[TMP2]] to <8 x i8>
-; AVX-NEXT:    [[TMP4:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; AVX-NEXT:    [[TMP4:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE0]], i64 0
 ; AVX-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i64> [[TMP4]], <8 x i64> poison, <8 x i32> zeroinitializer
 ; AVX-NEXT:    [[TMP6:%.*]] = lshr <8 x i64> [[TMP5]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
 ; AVX-NEXT:    [[TMP7:%.*]] = trunc <8 x i64> [[TMP6]] to <8 x i8>
@@ -640,11 +647,11 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; AVX-NEXT:    [[TMP13:%.*]] = add nuw <8 x i8> [[TMP10]], [[TMP12]]
 ; AVX-NEXT:    [[TMP16:%.*]] = bitcast <8 x i8> [[TMP13]] to i64
 ; AVX-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
-; AVX-NEXT:    [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
+; AVX-NEXT:    [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1]], i64 0
 ; AVX-NEXT:    [[TMP18:%.*]] = shufflevector <8 x i64> [[TMP17]], <8 x i64> poison, <8 x i32> zeroinitializer
 ; AVX-NEXT:    [[TMP19:%.*]] = lshr <8 x i64> [[TMP18]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
 ; AVX-NEXT:    [[TMP20:%.*]] = trunc <8 x i64> [[TMP19]] to <8 x i8>
-; AVX-NEXT:    [[TMP21:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE1:%.*]], i64 0
+; AVX-NEXT:    [[TMP21:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE1]], i64 0
 ; AVX-NEXT:    [[TMP22:%.*]] = shufflevector <8 x i64> [[TMP21]], <8 x i64> poison, <8 x i32> zeroinitializer
 ; AVX-NEXT:    [[TMP23:%.*]] = lshr <8 x i64> [[TMP22]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
 ; AVX-NEXT:    [[TMP24:%.*]] = trunc <8 x i64> [[TMP23]] to <8 x i8>
@@ -701,15 +708,16 @@ for.body:                                         ; preds = %for.cond
 }
 
 define { i64, i64 } @avgr_8_u16(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0, i64 %b.coerce1) {
-; CHECK-LABEL: @avgr_8_u16(
-; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x i64> [[TMP1]], i64 [[A_COERCE1:%.*]], i64 1
+; CHECK-LABEL: define { i64, i64 } @avgr_8_u16(
+; CHECK-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0]], i64 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x i64> [[TMP1]], i64 [[A_COERCE1]], i64 1
 ; CHECK-NEXT:    [[TMP15:%.*]] = trunc <2 x i64> [[TMP2]] to <2 x i32>
 ; CHECK-NEXT:    [[TMP3:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 32)
 ; CHECK-NEXT:    [[TMP4:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 48)
-; CHECK-NEXT:    [[TMP7:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <2 x i64> [[TMP7]], i64 [[B_COERCE1:%.*]], i64 1
+; CHECK-NEXT:    [[TMP7:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0]], i64 0
+; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <2 x i64> [[TMP7]], i64 [[B_COERCE1]], i64 1
 ; CHECK-NEXT:    [[TMP18:%.*]] = trunc <2 x i64> [[TMP8]] to <2 x i32>
 ; CHECK-NEXT:    [[TMP9:%.*]] = lshr <2 x i64> [[TMP8]], splat (i64 32)
 ; CHECK-NEXT:    [[TMP10:%.*]] = lshr <2 x i64> [[TMP8]], splat (i64 48)
@@ -787,42 +795,30 @@ for.body:                                         ; preds = %for.cond
 }
 
 define { i64, i64 } @avgr_8_u16_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0, i64 %b.coerce1)  {
-; SSE2-LABEL: @avgr_8_u16_alt(
-; SSE2-NEXT:  entry:
-; SSE2-NEXT:    [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0:%.*]], 48
+; SSE2-LABEL: define { i64, i64 } @avgr_8_u16_alt(
+; SSE2-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; SSE2-NEXT:  [[ENTRY:.*:]]
+; SSE2-NEXT:    [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 48
 ; SSE2-NEXT:    [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
-; SSE2-NEXT:    [[A_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
-; SSE2-NEXT:    [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0:%.*]], 48
+; SSE2-NEXT:    [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 48
 ; SSE2-NEXT:    [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
-; SSE2-NEXT:    [[B_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
-; SSE2-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0]] to i16
-; SSE2-NEXT:    [[TMP14:%.*]] = insertelement <2 x i16> poison, i16 [[TMP0]], i64 0
-; SSE2-NEXT:    [[TMP17:%.*]] = trunc i64 [[A_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[TMP18:%.*]] = insertelement <2 x i16> [[TMP14]], i16 [[TMP17]], i64 1
-; SSE2-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0]] to i16
-; SSE2-NEXT:    [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
-; SSE2-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; SSE2-NEXT:    [[TMP19:%.*]] = lshr <2 x i16> [[TMP18]], splat (i16 1)
-; SSE2-NEXT:    [[TMP21:%.*]] = lshr <2 x i16> [[TMP7]], splat (i16 1)
-; SSE2-NEXT:    [[TMP10:%.*]] = add nuw <2 x i16> [[TMP21]], [[TMP19]]
-; SSE2-NEXT:    [[TMP37:%.*]] = shufflevector <2 x i16> [[TMP7]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT:    [[TMP1:%.*]] = shufflevector <2 x i16> [[TMP18]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT:    [[TMP44:%.*]] = shufflevector <2 x i16> [[TMP10]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT:    [[A_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1:%.*]], 48
+; SSE2-NEXT:    [[TMP1:%.*]] = bitcast i64 [[A_COERCE0]] to <4 x i16>
+; SSE2-NEXT:    [[TMP37:%.*]] = bitcast i64 [[B_COERCE0]] to <4 x i16>
+; SSE2-NEXT:    [[TMP4:%.*]] = lshr <4 x i16> [[TMP37]], splat (i16 1)
+; SSE2-NEXT:    [[TMP5:%.*]] = lshr <4 x i16> [[TMP1]], splat (i16 1)
+; SSE2-NEXT:    [[TMP44:%.*]] = add nuw <4 x i16> [[TMP4]], [[TMP5]]
+; SSE2-NEXT:    [[A_SROA_9_8_EXTRACT_SHIFT1:%.*]] = lshr i64 [[A_COERCE1]], 48
+; SSE2-NEXT:    [[A_SROA_8_8_EXTRACT_SHIFT1:%.*]] = lshr i64 [[A_COERCE1]], 32
+; SSE2-NEXT:    [[A_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 48
 ; SSE2-NEXT:    [[A_SROA_8_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
-; SSE2-NEXT:    [[A_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
-; SSE2-NEXT:    [[B_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2:%.*]], 48
-; SSE2-NEXT:    [[B_SROA_8_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2]], 32
-; SSE2-NEXT:    [[B_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2]], 16
-; SSE2-NEXT:    [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_8_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[B_SROA_8_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_9_8_EXTRACT_SHIFT]] to i16
+; SSE2-NEXT:    [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_8_8_EXTRACT_SHIFT]] to i16
+; SSE2-NEXT:    [[B_SROA_8_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_9_8_EXTRACT_SHIFT]] to i16
 ; SSE2-NEXT:    [[B_SROA_3_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_3_0_EXTRACT_SHIFT]] to i16
 ; SSE2-NEXT:    [[B_SROA_4_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_4_0_EXTRACT_SHIFT]] to i16
 ; SSE2-NEXT:    [[TMP38:%.*]] = insertelement <4 x i16> [[TMP37]], i16 [[B_SROA_3_0_EXTRACT_TRUNC]], i64 2
 ; SSE2-NEXT:    [[TMP39:%.*]] = insertelement <4 x i16> [[TMP38]], i16 [[B_SROA_4_0_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT:    [[NARROW:%.*]] = trunc i64 [[A_SROA_8_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[A_SROA_5_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_9_8_EXTRACT_SHIFT]] to i16
+; SSE2-NEXT:    [[NARROW:%.*]] = trunc i64 [[A_SROA_8_8_EXTRACT_SHIFT1]] to i16
+; SSE2-NEXT:    [[A_SROA_5_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_9_8_EXTRACT_SHIFT1]] to i16
 ; SSE2-NEXT:    [[B_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_3_0_EXTRACT_SHIFT]] to i16
 ; SSE2-NEXT:    [[B_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_4_0_EXTRACT_SHIFT]] to i16
 ; SSE2-NEXT:    [[TMP2:%.*]] = insertelement <4 x i16> [[TMP1]], i16 [[B_SROA_2_0_EXTRACT_TRUNC]], i64 2
@@ -844,125 +840,96 @@ define { i64, i64 } @avgr_8_u16_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; SSE2-NEXT:    [[TMP15:%.*]] = add nuw <4 x i16> [[TMP57]], [[TMP9]]
 ; SSE2-NEXT:    [[TMP16:%.*]] = bitcast <4 x i16> [[TMP15]] to i64
 ; SSE2-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
-; SSE2-NEXT:    [[TMP40:%.*]] = trunc i64 [[B_COERCE1]] to i16
-; SSE2-NEXT:    [[TMP41:%.*]] = insertelement <2 x i16> poison, i16 [[TMP40]], i64 0
-; SSE2-NEXT:    [[TMP42:%.*]] = trunc i64 [[A_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[TMP27:%.*]] = insertelement <2 x i16> [[TMP41]], i16 [[TMP42]], i64 1
-; SSE2-NEXT:    [[TMP28:%.*]] = trunc i64 [[B_COERCE2]] to i16
-; SSE2-NEXT:    [[TMP29:%.*]] = insertelement <2 x i16> poison, i16 [[TMP28]], i64 0
-; SSE2-NEXT:    [[TMP45:%.*]] = trunc i64 [[B_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[TMP46:%.*]] = insertelement <2 x i16> [[TMP29]], i16 [[TMP45]], i64 1
-; SSE2-NEXT:    [[TMP32:%.*]] = lshr <2 x i16> [[TMP27]], splat (i16 1)
-; SSE2-NEXT:    [[TMP47:%.*]] = lshr <2 x i16> [[TMP46]], splat (i16 1)
-; SSE2-NEXT:    [[TMP34:%.*]] = add nuw <2 x i16> [[TMP47]], [[TMP32]]
-; SSE2-NEXT:    [[TMP35:%.*]] = shufflevector <2 x i16> [[TMP46]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE2-NEXT:    [[TMP22:%.*]] = bitcast i64 [[A_COERCE1]] to <4 x i16>
+; SSE2-NEXT:    [[TMP35:%.*]] = bitcast i64 [[B_COERCE1]] to <4 x i16>
 ; SSE2-NEXT:    [[TMP36:%.*]] = insertelement <4 x i16> [[TMP35]], i16 [[B_SROA_9_8_EXTRACT_TRUNC]], i64 2
 ; SSE2-NEXT:    [[TMP20:%.*]] = insertelement <4 x i16> [[TMP36]], i16 [[B_SROA_8_8_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT:    [[TMP22:%.*]] = shufflevector <2 x i16> [[TMP27]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
 ; SSE2-NEXT:    [[TMP23:%.*]] = insertelement <4 x i16> [[TMP22]], i16 [[NARROW]], i64 2
 ; SSE2-NEXT:    [[TMP24:%.*]] = insertelement <4 x i16> [[TMP23]], i16 [[A_SROA_5_8_EXTRACT_TRUNC]], i64 3
 ; SSE2-NEXT:    [[TMP25:%.*]] = or <4 x i16> [[TMP20]], [[TMP24]]
 ; SSE2-NEXT:    [[TMP26:%.*]] = and <4 x i16> [[TMP25]], splat (i16 1)
-; SSE2-NEXT:    [[TMP43:%.*]] = shufflevector <2 x i16> [[TMP34]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE2-NEXT:    [[TMP40:%.*]] = lshr <4 x i16> [[TMP35]], splat (i16 1)
+; SSE2-NEXT:    [[TMP34:%.*]] = lshr <4 x i16> [[TMP22]], splat (i16 1)
+; SSE2-NEXT:    [[TMP43:%.*]] = add nuw <4 x i16> [[TMP40]], [[TMP34]]
 ; SSE2-NEXT:    [[TMP30:%.*]] = shufflevector <4 x i16> [[TMP43]], <4 x i16> [[TMP56]], <4 x i32> <i32 0, i32 1, i32 7, i32 5>
 ; SSE2-NEXT:    [[TMP31:%.*]] = add nuw <4 x i16> [[TMP30]], [[TMP26]]
 ; SSE2-NEXT:    [[TMP33:%.*]] = bitcast <4 x i16> [[TMP31]] to i64
 ; SSE2-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP33]], 1
 ; SSE2-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
-; SSE4-LABEL: @avgr_8_u16_alt(
-; SSE4-NEXT:  entry:
-; SSE4-NEXT:    [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0:%.*]], 48
+; SSE4-LABEL: define { i64, i64 } @avgr_8_u16_alt(
+; SSE4-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; SSE4-NEXT:  [[ENTRY:.*:]]
+; SSE4-NEXT:    [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 48
 ; SSE4-NEXT:    [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
-; SSE4-NEXT:    [[A_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
 ; SSE4-NEXT:    [[B_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_4_0_EXTRACT_SHIFT]] to i16
 ; SSE4-NEXT:    [[B_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_3_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0:%.*]], 48
+; SSE4-NEXT:    [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 48
 ; SSE4-NEXT:    [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
-; SSE4-NEXT:    [[B_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
 ; SSE4-NEXT:    [[B_SROA_4_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_4_0_EXTRACT_SHIFT]] to i16
 ; SSE4-NEXT:    [[B_SROA_3_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_3_0_EXTRACT_SHIFT]] to i16
 ; SSE4-NEXT:    [[SHR5:%.*]] = lshr i16 [[B_SROA_0_0_EXTRACT_TRUNC]], 1
 ; SSE4-NEXT:    [[SHR5_1:%.*]] = lshr i16 [[B_SROA_2_0_EXTRACT_TRUNC]], 1
 ; SSE4-NEXT:    [[SHR5_3:%.*]] = lshr i16 [[B_SROA_4_0_EXTRACT_TRUNC]], 1
 ; SSE4-NEXT:    [[SHR5_2:%.*]] = lshr i16 [[B_SROA_3_0_EXTRACT_TRUNC]], 1
-; SSE4-NEXT:    [[NARROW:%.*]] = add nuw i16 [[SHR5_3]], [[SHR5]]
-; SSE4-NEXT:    [[NARROW_1:%.*]] = add nuw i16 [[SHR5_2]], [[SHR5_1]]
-; SSE4-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0]] to i16
-; SSE4-NEXT:    [[TMP14:%.*]] = insertelement <2 x i16> poison, i16 [[TMP0]], i64 0
-; SSE4-NEXT:    [[TMP17:%.*]] = trunc i64 [[A_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[TMP18:%.*]] = insertelement <2 x i16> [[TMP14]], i16 [[TMP17]], i64 1
-; SSE4-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0]] to i16
-; SSE4-NEXT:    [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
-; SSE4-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; SSE4-NEXT:    [[TMP19:%.*]] = lshr <2 x i16> [[TMP18]], splat (i16 1)
-; SSE4-NEXT:    [[TMP21:%.*]] = lshr <2 x i16> [[TMP7]], splat (i16 1)
-; SSE4-NEXT:    [[TMP10:%.*]] = add nuw <2 x i16> [[TMP21]], [[TMP19]]
-; SSE4-NEXT:    [[TMP37:%.*]] = shufflevector <2 x i16> [[TMP7]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT:    [[TMP1:%.*]] = bitcast i64 [[A_COERCE0]] to <4 x i16>
+; SSE4-NEXT:    [[TMP37:%.*]] = bitcast i64 [[B_COERCE0]] to <4 x i16>
 ; SSE4-NEXT:    [[TMP38:%.*]] = insertelement <4 x i16> [[TMP37]], i16 [[B_SROA_3_0_EXTRACT_TRUNC]], i64 2
 ; SSE4-NEXT:    [[TMP39:%.*]] = insertelement <4 x i16> [[TMP38]], i16 [[B_SROA_4_0_EXTRACT_TRUNC]], i64 3
-; SSE4-NEXT:    [[TMP1:%.*]] = shufflevector <2 x i16> [[TMP18]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
 ; SSE4-NEXT:    [[TMP2:%.*]] = insertelement <4 x i16> [[TMP1]], i16 [[B_SROA_2_0_EXTRACT_TRUNC]], i64 2
 ; SSE4-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> [[TMP2]], i16 [[B_SROA_0_0_EXTRACT_TRUNC]], i64 3
 ; SSE4-NEXT:    [[TMP8:%.*]] = or <4 x i16> [[TMP39]], [[TMP3]]
 ; SSE4-NEXT:    [[TMP9:%.*]] = and <4 x i16> [[TMP8]], splat (i16 1)
-; SSE4-NEXT:    [[TMP11:%.*]] = shufflevector <2 x i16> [[TMP10]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT:    [[TMP12:%.*]] = insertelement <4 x i16> [[TMP11]], i16 [[NARROW_1]], i64 2
-; SSE4-NEXT:    [[TMP13:%.*]] = insertelement <4 x i16> [[TMP12]], i16 [[NARROW]], i64 3
+; SSE4-NEXT:    [[TMP14:%.*]] = lshr <4 x i16> [[TMP37]], splat (i16 1)
+; SSE4-NEXT:    [[TMP17:%.*]] = lshr <4 x i16> [[TMP1]], splat (i16 1)
+; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <4 x i16> [[TMP14]], i16 [[SHR5_2]], i64 2
+; SSE4-NEXT:    [[TMP11:%.*]] = insertelement <4 x i16> [[TMP17]], i16 [[SHR5_1]], i64 2
+; SSE4-NEXT:    [[TMP12:%.*]] = insertelement <4 x i16> [[TMP10]], i16 [[SHR5_3]], i64 3
+; SSE4-NEXT:    [[TMP18:%.*]] = insertelement <4 x i16> [[TMP11]], i16 [[SHR5]], i64 3
+; SSE4-NEXT:    [[TMP13:%.*]] = add nuw <4 x i16> [[TMP12]], [[TMP18]]
 ; SSE4-NEXT:    [[TMP15:%.*]] = add nuw <4 x i16> [[TMP13]], [[TMP9]]
 ; SSE4-NEXT:    [[TMP16:%.*]] = bitcast <4 x i16> [[TMP15]] to i64
 ; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
-; SSE4-NEXT:    [[A_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1:%.*]], 48
-; SSE4-NEXT:    [[B_SROA_8_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
-; SSE4-NEXT:    [[A_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
+; SSE4-NEXT:    [[A_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 48
+; SSE4-NEXT:    [[B_SROA_8_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 32
 ; SSE4-NEXT:    [[A_SROA_5_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_9_8_EXTRACT_SHIFT]] to i16
 ; SSE4-NEXT:    [[A_SROA_7_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_8_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[B_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2:%.*]], 48
-; SSE4-NEXT:    [[B_SROA_8_8_EXTRACT_SHIFT1:%.*]] = lshr i64 [[B_COERCE2]], 32
-; SSE4-NEXT:    [[B_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2]], 16
+; SSE4-NEXT:    [[B_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 48
+; SSE4-NEXT:    [[B_SROA_8_8_EXTRACT_SHIFT1:%.*]] = lshr i64 [[B_COERCE1]], 32
 ; SSE4-NEXT:    [[B_SROA_8_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_9_8_EXTRACT_SHIFT]] to i16
 ; SSE4-NEXT:    [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_8_8_EXTRACT_SHIFT1]] to i16
 ; SSE4-NEXT:    [[SHR_6:%.*]] = lshr i16 [[A_SROA_5_8_EXTRACT_TRUNC]], 1
 ; SSE4-NEXT:    [[SHR_7:%.*]] = lshr i16 [[A_SROA_7_8_EXTRACT_TRUNC]], 1
 ; SSE4-NEXT:    [[SHR5_6:%.*]] = lshr i16 [[B_SROA_8_8_EXTRACT_TRUNC]], 1
 ; SSE4-NEXT:    [[SHR5_7:%.*]] = lshr i16 [[B_SROA_9_8_EXTRACT_TRUNC]], 1
-; SSE4-NEXT:    [[NARROW_6:%.*]] = add nuw i16 [[SHR5_6]], [[SHR_6]]
-; SSE4-NEXT:    [[NARROW_7:%.*]] = add nuw i16 [[SHR5_7]], [[SHR_7]]
-; SSE4-NEXT:    [[TMP40:%.*]] = trunc i64 [[B_COERCE1]] to i16
-; SSE4-NEXT:    [[TMP41:%.*]] = insertelement <2 x i16> poison, i16 [[TMP40]], i64 0
-; SSE4-NEXT:    [[TMP42:%.*]] = trunc i64 [[A_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[TMP27:%.*]] = insertelement <2 x i16> [[TMP41]], i16 [[TMP42]], i64 1
-; SSE4-NEXT:    [[TMP28:%.*]] = trunc i64 [[B_COERCE2]] to i16
-; SSE4-NEXT:    [[TMP29:%.*]] = insertelement <2 x i16> poison, i16 [[TMP28]], i64 0
-; SSE4-NEXT:    [[TMP45:%.*]] = trunc i64 [[B_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[TMP46:%.*]] = insertelement <2 x i16> [[TMP29]], i16 [[TMP45]], i64 1
-; SSE4-NEXT:    [[TMP32:%.*]] = lshr <2 x i16> [[TMP27]], splat (i16 1)
-; SSE4-NEXT:    [[TMP47:%.*]] = lshr <2 x i16> [[TMP46]], splat (i16 1)
-; SSE4-NEXT:    [[TMP34:%.*]] = add nuw <2 x i16> [[TMP47]], [[TMP32]]
-; SSE4-NEXT:    [[TMP35:%.*]] = shufflevector <2 x i16> [[TMP46]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT:    [[TMP22:%.*]] = bitcast i64 [[A_COERCE1]] to <4 x i16>
+; SSE4-NEXT:    [[TMP35:%.*]] = bitcast i64 [[B_COERCE1]] to <4 x i16>
 ; SSE4-NEXT:    [[TMP36:%.*]] = insertelement <4 x i16> [[TMP35]], i16 [[B_SROA_9_8_EXTRACT_TRUNC]], i64 2
 ; SSE4-NEXT:    [[TMP20:%.*]] = insertelement <4 x i16> [[TMP36]], i16 [[B_SROA_8_8_EXTRACT_TRUNC]], i64 3
-; SSE4-NEXT:    [[TMP22:%.*]] = shufflevector <2 x i16> [[TMP27]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
 ; SSE4-NEXT:    [[TMP23:%.*]] = insertelement <4 x i16> [[TMP22]], i16 [[A_SROA_7_8_EXTRACT_TRUNC]], i64 2
 ; SSE4-NEXT:    [[TMP24:%.*]] = insertelement <4 x i16> [[TMP23]], i16 [[A_SROA_5_8_EXTRACT_TRUNC]], i64 3
 ; SSE4-NEXT:    [[TMP25:%.*]] = or <4 x i16> [[TMP20]], [[TMP24]]
 ; SSE4-NEXT:    [[TMP26:%.*]] = and <4 x i16> [[TMP25]], splat (i16 1)
-; SSE4-NEXT:    [[TMP43:%.*]] = shufflevector <2 x i16> [[TMP34]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT:    [[TMP44:%.*]] = insertelement <4 x i16> [[TMP43]], i16 [[NARROW_7]], i64 2
-; SSE4-NEXT:    [[TMP30:%.*]] = insertelement <4 x i16> [[TMP44]], i16 [[NARROW_6]], i64 3
+; SSE4-NEXT:    [[TMP32:%.*]] = lshr <4 x i16> [[TMP35]], splat (i16 1)
+; SSE4-NEXT:    [[TMP34:%.*]] = lshr <4 x i16> [[TMP22]], splat (i16 1)
+; SSE4-NEXT:    [[TMP27:%.*]] = insertelement <4 x i16> [[TMP32]], i16 [[SHR5_7]], i64 2
+; SSE4-NEXT:    [[TMP28:%.*]] = insertelement <4 x i16> [[TMP34]], i16 [[SHR_7]], i64 2
+; SSE4-NEXT:    [[TMP29:%.*]] = insertelement <4 x i16> [[TMP27]], i16 [[SHR5_6]], i64 3
+; SSE4-NEXT:    [[TMP40:%.*]] = insertelement <4 x i16> [[TMP28]], i16 [[SHR_6]], i64 3
+; SSE4-NEXT:    [[TMP30:%.*]] = add nuw <4 x i16> [[TMP29]], [[TMP40]]
 ; SSE4-NEXT:    [[TMP31:%.*]] = add nuw <4 x i16> [[TMP30]], [[TMP26]]
 ; SSE4-NEXT:    [[TMP33:%.*]] = bitcast <4 x i16> [[TMP31]] to i64
 ; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP33]], 1
 ; SSE4-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
-; AVX-LABEL: @avgr_8_u16_alt(
-; AVX-NEXT:  entry:
-; AVX-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; AVX-LABEL: define { i64, i64 } @avgr_8_u16_alt(
+; AVX-SAME: i64 [[A_COERCE0:%.*]], i64 [[A_COERCE1:%.*]], i64 [[B_COERCE0:%.*]], i64 [[B_COERCE1:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; AVX-NEXT:  [[ENTRY:.*:]]
+; AVX-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE0]], i64 0
 ; AVX-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
 ; AVX-NEXT:    [[TMP2:%.*]] = lshr <4 x i64> [[TMP1]], <i64 0, i64 16, i64 32, i64 48>
 ; AVX-NEXT:    [[TMP3:%.*]] = trunc <4 x i64> [[TMP2]] to <4 x i16>
-; AVX-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; AVX-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[B_COERCE0]], i64 0
 ; AVX-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
 ; AVX-NEXT:    [[TMP6:%.*]] = lshr <4 x i64> [[TMP5]], <i64 0, i64 16, i64 32, i64 48>
 ; AVX-NEXT:    [[TMP7:%.*]] = trunc <4 x i64> [[TMP6]] to <4 x i16>
@@ -974,11 +941,11 @@ define { i64, i64 } @avgr_8_u16_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; AVX-NEXT:    [[TMP13:%.*]] = add nuw <4 x i16> [[TMP10]], [[TMP12]]
 ; AVX-NEXT:    [[TMP14:%.*]] = bitcast <4 x i16> [[TMP13]] to i64
 ; AVX-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP14]], 0
-; AVX-NEXT:    [[TMP15:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
+; AVX-NEXT:    [[TMP15:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE1]], i64 0
 ; AVX-NEXT:    [[TMP16:%.*]] = shufflevector <4 x i64> [[TMP15]], <4 x i64> poison, <4 x i32> zeroinitializer
 ; AVX-NEXT:    [[TMP17:%.*]] = lshr <4 x i64> [[TMP16]], <i64 0, i64 16, i64 32, i64 48>
 ; AVX-NEXT:    [[TMP18:%.*]] = trunc <4 x i64> [[TMP17]] to <4 x i16>
-; AVX-NEXT:    [[TMP19:%.*]] = insertelement <4 x i64> poison, i64 [[B_COERCE1:%.*]], i64 0
+; AVX-NEXT:    [[TMP19:%.*]] = insertelement <4 x i64> poison, i64 [[B_COERCE1]], i64 0
 ; AVX-NEXT:    [[TMP20:%.*]] = shufflevector <4 x i64> [[TMP19]], <4 x i64> poison, <4 x i32> zeroinitializer
 ; AVX-NEXT:    [[TMP21:%.*]] = lshr <4 x i64> [[TMP20]], <i64 0, i64 16, i64 32, i64 48>
 ; AVX-NEXT:    [[TMP22:%.*]] = trunc <4 x i64> [[TMP21]] to <4 x i16>
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
index 30031078a5548c..a345eb42da6965 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
@@ -16,13 +16,8 @@ define [2 x float] @sum_pairs(ptr %p, i64 %n) {
 ; CHECK-NEXT:    [[PTR:%.*]] = phi ptr [ [[PTR_NEXT:%.*]], %[[LOOP]] ], [ [[P]], %[[ENTRY]] ]
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ [[I_NEXT:%.*]], %[[LOOP]] ], [ 0, %[[ENTRY]] ]
 ; CHECK-NEXT:    [[TMP0:%.*]] = phi <2 x float> [ [[TMP6:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
-; CHECK-NEXT:    [[X:%.*]] = load i64, ptr [[PTR]], align 1
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[TMP1:%.*]] = trunc nuw i64 [[HI]] to i32
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x i32> poison, i32 [[TMP1]], i64 0
-; CHECK-NEXT:    [[TMP3:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <2 x i32> [[TMP2]], i32 [[TMP3]], i64 1
-; CHECK-NEXT:    [[TMP5:%.*]] = bitcast <2 x i32> [[TMP4]] to <2 x float>
+; CHECK-NEXT:    [[X1:%.*]] = load <2 x float>, ptr [[PTR]], align 1
+; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <2 x float> [[X1]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    [[TMP6]] = fadd <2 x float> [[TMP0]], [[TMP5]]
 ; CHECK-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 8
 ; CHECK-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 1
diff --git a/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll b/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
index 2c7d80440ff271..28ea2686da6993 100644
--- a/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
+++ b/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
@@ -8,20 +8,13 @@
 define <2 x i32> @high_half_first(i64 %x) {
 ; LE-LABEL: define <2 x i32> @high_half_first(
 ; LE-SAME: i64 [[X:%.*]]) {
-; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; LE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; LE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; LE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; LE-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; LE-NEXT:    ret <2 x i32> [[V1]]
 ;
 ; BE-LABEL: define <2 x i32> @high_half_first(
 ; BE-SAME: i64 [[X:%.*]]) {
-; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; BE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; BE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; BE-NEXT:    [[V1:%.*]] = bitcast i64 [[X]] to <2 x i32>
 ; BE-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -35,20 +28,13 @@ define <2 x i32> @high_half_first(i64 %x) {
 define <2 x i32> @low_half_first(i64 %x) {
 ; LE-LABEL: define <2 x i32> @low_half_first(
 ; LE-SAME: i64 [[X:%.*]]) {
-; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; LE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T0]], i64 0
-; LE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; LE-NEXT:    [[V1:%.*]] = bitcast i64 [[X]] to <2 x i32>
 ; LE-NEXT:    ret <2 x i32> [[V1]]
 ;
 ; BE-LABEL: define <2 x i32> @low_half_first(
 ; BE-SAME: i64 [[X:%.*]]) {
-; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; BE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T0]], i64 0
-; BE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; BE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; BE-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; BE-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -62,32 +48,13 @@ define <2 x i32> @low_half_first(i64 %x) {
 define <4 x i16> @reversed_quarters(i64 %x) {
 ; LE-LABEL: define <4 x i16> @reversed_quarters(
 ; LE-SAME: i64 [[X:%.*]]) {
-; LE-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
-; LE-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
-; LE-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
-; LE-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; LE-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; LE-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; LE-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
-; LE-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
-; LE-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
-; LE-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; LE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <4 x i16>
+; LE-NEXT:    [[V3:%.*]] = shufflevector <4 x i16> [[TMP1]], <4 x i16> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
 ; LE-NEXT:    ret <4 x i16> [[V3]]
 ;
 ; BE-LABEL: define <4 x i16> @reversed_quarters(
 ; BE-SAME: i64 [[X:%.*]]) {
-; BE-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
-; BE-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
-; BE-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
-; BE-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; BE-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; BE-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; BE-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
-; BE-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
-; BE-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
-; BE-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; BE-NEXT:    [[V3:%.*]] = bitcast i64 [[X]] to <4 x i16>
 ; BE-NEXT:    ret <4 x i16> [[V3]]
 ;
   %s1 = lshr i64 %x, 16
@@ -107,24 +74,14 @@ define <4 x i16> @reversed_quarters(i64 %x) {
 define <4 x i32> @repeated_halves(i64 %x) {
 ; LE-LABEL: define <4 x i32> @repeated_halves(
 ; LE-SAME: i64 [[X:%.*]]) {
-; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; LE-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
-; LE-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
-; LE-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
-; LE-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; LE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; LE-NEXT:    [[V3:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> <i32 1, i32 0, i32 1, i32 0>
 ; LE-NEXT:    ret <4 x i32> [[V3]]
 ;
 ; BE-LABEL: define <4 x i32> @repeated_halves(
 ; BE-SAME: i64 [[X:%.*]]) {
-; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; BE-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
-; BE-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
-; BE-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
-; BE-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; BE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; BE-NEXT:    [[V3:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
 ; BE-NEXT:    ret <4 x i32> [[V3]]
 ;
   %hi = lshr i64 %x, 32
diff --git a/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll b/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll
index 82d2766983b54c..27ac96fc18b7bd 100644
--- a/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll
@@ -8,11 +8,8 @@
 define <2 x i32> @swapped_halves(i64 %x) {
 ; CHECK-LABEL: define <2 x i32> @swapped_halves(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -26,13 +23,8 @@ define <2 x i32> @swapped_halves(i64 %x) {
 define <2 x float> @swapped_halves_float(i64 %x) {
 ; CHECK-LABEL: define <2 x float> @swapped_halves_float(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[F1:%.*]] = bitcast i32 [[T1]] to float
-; CHECK-NEXT:    [[F0:%.*]] = bitcast i32 [[T0]] to float
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x float> poison, float [[F1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x float> [[V0]], float [[F0]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x float>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <2 x float> [[TMP1]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x float> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -48,17 +40,8 @@ define <2 x float> @swapped_halves_float(i64 %x) {
 define <4 x i16> @reversed_quarters(i64 %x) {
 ; CHECK-LABEL: define <4 x i16> @reversed_quarters(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
-; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
-; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <4 x i16>
+; CHECK-NEXT:    [[V3:%.*]] = shufflevector <4 x i16> [[TMP1]], <4 x i16> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
 ; CHECK-NEXT:    ret <4 x i16> [[V3]]
 ;
   %s1 = lshr i64 %x, 16
@@ -78,17 +61,8 @@ define <4 x i16> @reversed_quarters(i64 %x) {
 define <4 x i32> @reversed_i128(i128 %x) {
 ; CHECK-LABEL: define <4 x i32> @reversed_i128(
 ; CHECK-SAME: i128 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[S1:%.*]] = lshr i128 [[X]], 32
-; CHECK-NEXT:    [[S2:%.*]] = lshr i128 [[X]], 64
-; CHECK-NEXT:    [[S3:%.*]] = lshr i128 [[X]], 96
-; CHECK-NEXT:    [[T0:%.*]] = trunc i128 [[X]] to i32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i128 [[S1]] to i32
-; CHECK-NEXT:    [[T2:%.*]] = trunc i128 [[S2]] to i32
-; CHECK-NEXT:    [[T3:%.*]] = trunc i128 [[S3]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T3]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T2]], i64 1
-; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
-; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i128 [[X]] to <4 x i32>
+; CHECK-NEXT:    [[V3:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
 ; CHECK-NEXT:    ret <4 x i32> [[V3]]
 ;
   %s1 = lshr i128 %x, 32
@@ -109,13 +83,8 @@ define <4 x i32> @reversed_i128(i128 %x) {
 define <4 x i32> @repeated_halves(i64 %x) {
 ; CHECK-LABEL: define <4 x i32> @repeated_halves(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
-; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
-; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V3:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> <i32 1, i32 0, i32 1, i32 0>
 ; CHECK-NEXT:    ret <4 x i32> [[V3]]
 ;
   %hi = lshr i64 %x, 32
@@ -132,12 +101,8 @@ define <4 x i32> @repeated_halves(i64 %x) {
 define <2 x i16> @odd_quarters(i64 %x) {
 ; CHECK-LABEL: define <2 x i16> @odd_quarters(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i16> poison, i16 [[T3]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i16> [[V0]], i16 [[T1]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <4 x i16>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <4 x i16> [[TMP1]], <4 x i16> poison, <2 x i32> <i32 3, i32 1>
 ; CHECK-NEXT:    ret <2 x i16> [[V1]]
 ;
   %s1 = lshr i64 %x, 16
@@ -152,17 +117,7 @@ define <2 x i16> @odd_quarters(i64 %x) {
 define <4 x i16> @in_order_quarters(i64 %x) {
 ; CHECK-LABEL: define <4 x i16> @in_order_quarters(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T0]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T1]], i64 1
-; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T2]], i64 2
-; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T3]], i64 3
+; CHECK-NEXT:    [[V3:%.*]] = bitcast i64 [[X]] to <4 x i16>
 ; CHECK-NEXT:    ret <4 x i16> [[V3]]
 ;
   %s1 = lshr i64 %x, 16
@@ -183,12 +138,8 @@ define <4 x i16> @in_order_quarters(i64 %x) {
 define <4 x i16> @missing_elts(i64 %x) {
 ; CHECK-LABEL: define <4 x i16> @missing_elts(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 2
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <4 x i16>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <4 x i16> [[TMP1]], <4 x i16> poison, <4 x i32> <i32 3, i32 poison, i32 2, i32 poison>
 ; CHECK-NEXT:    ret <4 x i16> [[V1]]
 ;
   %s2 = lshr i64 %x, 32
@@ -203,11 +154,8 @@ define <4 x i16> @missing_elts(i64 %x) {
 define <2 x i32> @undef_base(i64 %x) {
 ; CHECK-LABEL: define <2 x i32> @undef_base(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> undef, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -219,13 +167,19 @@ define <2 x i32> @undef_base(i64 %x) {
 }
 
 define <2 x i32> @splat_high_half(i64 %x) {
-; CHECK-LABEL: define <2 x i32> @splat_high_half(
-; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
-; CHECK-NEXT:    ret <2 x i32> [[V1]]
+; SSE-LABEL: define <2 x i32> @splat_high_half(
+; SSE-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; SSE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; SSE-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 1>
+; SSE-NEXT:    ret <2 x i32> [[V1]]
+;
+; AVX-LABEL: define <2 x i32> @splat_high_half(
+; AVX-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; AVX-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; AVX-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; AVX-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; AVX-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; AVX-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
   %t1 = trunc i64 %hi to i32
@@ -238,12 +192,8 @@ define <2 x i32> @splat_high_half(i64 %x) {
 define <2 x i32> @overwritten_elt(i64 %x, i32 %y) {
 ; CHECK-LABEL: define <2 x i32> @overwritten_elt(
 ; CHECK-SAME: i64 [[X:%.*]], i32 [[Y:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[Y]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
-; CHECK-NEXT:    [[V2:%.*]] = insertelement <2 x i32> [[V1]], i32 [[T1]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V2:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x i32> [[V2]]
 ;
   %hi = lshr i64 %x, 32
@@ -259,11 +209,8 @@ define <2 x i32> @double_source(double %d) {
 ; CHECK-LABEL: define <2 x i32> @double_source(
 ; CHECK-SAME: double [[D:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:    [[X:%.*]] = bitcast double [[D]] to i64
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x i32> [[V1]]
 ;
   %x = bitcast double %d to i64
@@ -279,11 +226,8 @@ define <2 x i32> @double_source(double %d) {
 define <2 x i32> @overwritten_base(i64 %x, <2 x i32> %base) {
 ; CHECK-LABEL: define <2 x i32> @overwritten_base(
 ; CHECK-SAME: i64 [[X:%.*]], <2 x i32> [[BASE:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> [[BASE]], i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -295,15 +239,24 @@ define <2 x i32> @overwritten_base(i64 %x, <2 x i32> %base) {
 }
 
 define <2 x i32> @swapped_halves_extra_use(i64 %x, ptr %p) {
-; CHECK-LABEL: define <2 x i32> @swapped_halves_extra_use(
-; CHECK-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    store i32 [[T1]], ptr [[P]], align 4
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
-; CHECK-NEXT:    ret <2 x i32> [[V1]]
+; SSE-LABEL: define <2 x i32> @swapped_halves_extra_use(
+; SSE-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; SSE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; SSE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; SSE-NEXT:    store i32 [[T1]], ptr [[P]], align 4
+; SSE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; SSE-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
+; SSE-NEXT:    ret <2 x i32> [[V1]]
+;
+; AVX-LABEL: define <2 x i32> @swapped_halves_extra_use(
+; AVX-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; AVX-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; AVX-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; AVX-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; AVX-NEXT:    store i32 [[T1]], ptr [[P]], align 4
+; AVX-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; AVX-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; AVX-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
   %t1 = trunc i64 %hi to i32
@@ -479,15 +432,24 @@ define <2 x i32> @out_of_range_shift(i64 %x) {
 
 ; The intermediate vector is used elsewhere, so the chain ends there.
 define <2 x i32> @extra_use_of_insert(i64 %x, ptr %p) {
-; CHECK-LABEL: define <2 x i32> @extra_use_of_insert(
-; CHECK-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    store <2 x i32> [[V0]], ptr [[P]], align 8
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
-; CHECK-NEXT:    ret <2 x i32> [[V1]]
+; SSE-LABEL: define <2 x i32> @extra_use_of_insert(
+; SSE-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; SSE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; SSE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; SSE-NEXT:    [[V0:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 poison>
+; SSE-NEXT:    store <2 x i32> [[V0]], ptr [[P]], align 8
+; SSE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; SSE-NEXT:    ret <2 x i32> [[V1]]
+;
+; AVX-LABEL: define <2 x i32> @extra_use_of_insert(
+; AVX-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; AVX-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; AVX-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; AVX-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; AVX-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; AVX-NEXT:    store <2 x i32> [[V0]], ptr [[P]], align 8
+; AVX-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; AVX-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
   %t1 = trunc i64 %hi to i32
@@ -549,6 +511,3 @@ define <8 x i1> @bool_elts(i8 %x) {
   %v1 = insertelement <8 x i1> %v0, i1 %t0, i64 1
   ret <8 x i1> %v1
 }
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; AVX: {{.*}}
-; SSE: {{.*}}



More information about the llvm-commits mailing list