[llvm] [VectorCombine] Fold insertelement chains of scalar parts to a bitcast and shuffle (PR #226224)

Tim Besard via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 00:19:05 PDT 2026


https://github.com/maleadt updated https://github.com/llvm/llvm-project/pull/226224

>From 34f3aab2ab77cefcbfeabc636c82464131589365 Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Thu, 24 Sep 2026 22:14:57 +0200
Subject: [PATCH 1/2] [VectorCombine] Add tests for insertelement chains of
 scalar parts (NFC)

Precommit tests for chains of insertelements whose elements are all
truncated parts of the same scalar, and a PhaseOrdering test for the
loop in which SLP produces such a chain.

Assisted-by: Claude Code, Codex
---
 .../X86/vector-reduction-of-scalar-parts.ll   | 103 ++++
 .../AArch64/insert-scalar-parts.ll            | 138 +++++
 .../VectorCombine/X86/insert-scalar-parts.ll  | 554 ++++++++++++++++++
 3 files changed, 795 insertions(+)
 create mode 100644 llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
 create mode 100644 llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
 create mode 100644 llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll

diff --git a/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
new file mode 100644
index 00000000000000..a0b4c118b9542a
--- /dev/null
+++ b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
@@ -0,0 +1,103 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -O2 -S -mtriple=x86_64-- -mcpu=x86-64 | FileCheck %s --check-prefixes=CHECK,SSE2
+; RUN: opt < %s -O2 -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX2
+
+; Sum the two float fields of records that SROA loaded as one i64. SLP
+; vectorizes the two reductions and gathers the fields in reverse order; the
+; gather should become a shuffle of the loaded vector.
+
+define [2 x float] @sum_pairs(ptr %p, i64 %n) {
+; SSE2-LABEL: define [2 x float] @sum_pairs(
+; SSE2-SAME: ptr nofree readonly captures(none) [[P:%.*]], i64 [[N:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; SSE2-NEXT:  [[ENTRY:.*]]:
+; SSE2-NEXT:    [[SKIP:%.*]] = icmp slt i64 [[N]], 1
+; SSE2-NEXT:    br i1 [[SKIP]], label %[[EXIT:.*]], label %[[LOOP:.*]]
+; SSE2:       [[LOOP]]:
+; SSE2-NEXT:    [[PTR:%.*]] = phi ptr [ [[PTR_NEXT:%.*]], %[[LOOP]] ], [ [[P]], %[[ENTRY]] ]
+; SSE2-NEXT:    [[I:%.*]] = phi i64 [ [[I_NEXT:%.*]], %[[LOOP]] ], [ 0, %[[ENTRY]] ]
+; SSE2-NEXT:    [[TMP0:%.*]] = phi <2 x float> [ [[TMP2:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
+; SSE2-NEXT:    [[X1:%.*]] = load <2 x float>, ptr [[PTR]], align 1
+; SSE2-NEXT:    [[TMP1:%.*]] = shufflevector <2 x float> [[X1]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
+; SSE2-NEXT:    [[TMP2]] = fadd <2 x float> [[TMP0]], [[TMP1]]
+; SSE2-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 8
+; SSE2-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 1
+; SSE2-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
+; SSE2-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
+; SSE2:       [[EXIT]]:
+; SSE2-NEXT:    [[TMP3:%.*]] = phi <2 x float> [ zeroinitializer, %[[ENTRY]] ], [ [[TMP2]], %[[LOOP]] ]
+; SSE2-NEXT:    [[TMP4:%.*]] = extractelement <2 x float> [[TMP3]], i64 1
+; SSE2-NEXT:    [[R0:%.*]] = insertvalue [2 x float] poison, float [[TMP4]], 0
+; SSE2-NEXT:    [[TMP5:%.*]] = extractelement <2 x float> [[TMP3]], i64 0
+; SSE2-NEXT:    [[R1:%.*]] = insertvalue [2 x float] [[R0]], float [[TMP5]], 1
+; SSE2-NEXT:    ret [2 x float] [[R1]]
+;
+; AVX2-LABEL: define [2 x float] @sum_pairs(
+; AVX2-SAME: ptr nofree readonly captures(none) [[P:%.*]], i64 [[N:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; AVX2-NEXT:  [[ENTRY:.*]]:
+; AVX2-NEXT:    [[SKIP:%.*]] = icmp slt i64 [[N]], 1
+; AVX2-NEXT:    br i1 [[SKIP]], label %[[EXIT:.*]], label %[[LOOP:.*]]
+; AVX2:       [[LOOP]]:
+; AVX2-NEXT:    [[PTR:%.*]] = phi ptr [ [[PTR_NEXT:%.*]], %[[LOOP]] ], [ [[P]], %[[ENTRY]] ]
+; AVX2-NEXT:    [[I:%.*]] = phi i64 [ [[I_NEXT:%.*]], %[[LOOP]] ], [ 0, %[[ENTRY]] ]
+; AVX2-NEXT:    [[TMP0:%.*]] = phi <2 x float> [ [[TMP6:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
+; AVX2-NEXT:    [[X:%.*]] = load i64, ptr [[PTR]], align 1
+; AVX2-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; AVX2-NEXT:    [[TMP1:%.*]] = trunc nuw i64 [[HI]] to i32
+; AVX2-NEXT:    [[TMP2:%.*]] = insertelement <2 x i32> poison, i32 [[TMP1]], i64 0
+; AVX2-NEXT:    [[TMP3:%.*]] = trunc i64 [[X]] to i32
+; AVX2-NEXT:    [[TMP4:%.*]] = insertelement <2 x i32> [[TMP2]], i32 [[TMP3]], i64 1
+; AVX2-NEXT:    [[TMP5:%.*]] = bitcast <2 x i32> [[TMP4]] to <2 x float>
+; AVX2-NEXT:    [[TMP6]] = fadd <2 x float> [[TMP0]], [[TMP5]]
+; AVX2-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 8
+; AVX2-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 1
+; AVX2-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
+; AVX2-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
+; AVX2:       [[EXIT]]:
+; AVX2-NEXT:    [[TMP7:%.*]] = phi <2 x float> [ zeroinitializer, %[[ENTRY]] ], [ [[TMP6]], %[[LOOP]] ]
+; AVX2-NEXT:    [[TMP8:%.*]] = extractelement <2 x float> [[TMP7]], i64 1
+; AVX2-NEXT:    [[R0:%.*]] = insertvalue [2 x float] poison, float [[TMP8]], 0
+; AVX2-NEXT:    [[TMP9:%.*]] = extractelement <2 x float> [[TMP7]], i64 0
+; AVX2-NEXT:    [[R1:%.*]] = insertvalue [2 x float] [[R0]], float [[TMP9]], 1
+; AVX2-NEXT:    ret [2 x float] [[R1]]
+;
+entry:
+  %skip = icmp slt i64 %n, 1
+  br i1 %skip, label %exit, label %loop
+
+loop:
+  %ptr = phi ptr [ %p, %entry ], [ %ptr.next, %loop ]
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  %acc.a = phi float [ 0.0, %entry ], [ %sum.a, %loop ]
+  %acc.b = phi float [ 0.0, %entry ], [ %sum.b, %loop ]
+  %x = load i64, ptr %ptr, align 1
+  %hi = lshr i64 %x, 32
+  %b.i = trunc nuw i64 %hi to i32
+  %a.i = trunc i64 %x to i32
+  %a = bitcast i32 %a.i to float
+  %sum.a = fadd float %acc.a, %a
+  %b = bitcast i32 %b.i to float
+  %sum.b = fadd float %acc.b, %b
+  %ptr.next = getelementptr i8, ptr %ptr, i64 8
+  %i.next = add i64 %i, 1
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop, !llvm.loop !0
+
+exit:
+  %r.b = phi float [ 0.0, %entry ], [ %sum.b, %loop ]
+  %r.a = phi float [ 0.0, %entry ], [ %sum.a, %loop ]
+  %r0 = insertvalue [2 x float] poison, float %r.a, 0
+  %r1 = insertvalue [2 x float] %r0, float %r.b, 1
+  ret [2 x float] %r1
+}
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.unroll.disable"}
+;.
+; SSE2: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]]}
+; SSE2: [[META1]] = !{!"llvm.loop.unroll.disable"}
+;.
+; AVX2: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]]}
+; AVX2: [[META1]] = !{!"llvm.loop.unroll.disable"}
+;.
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}
diff --git a/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll b/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
new file mode 100644
index 00000000000000..2c7d80440ff271
--- /dev/null
+++ b/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
@@ -0,0 +1,138 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=vector-combine -S -mtriple=aarch64 | FileCheck %s --check-prefix=LE
+; RUN: opt < %s -passes=vector-combine -S -mtriple=aarch64_be | FileCheck %s --check-prefix=BE
+
+; The first element of the bitcast scalar is its low part on little-endian
+; targets and its high part on big-endian targets.
+
+define <2 x i32> @high_half_first(i64 %x) {
+; LE-LABEL: define <2 x i32> @high_half_first(
+; LE-SAME: i64 [[X:%.*]]) {
+; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; LE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; LE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; LE-NEXT:    ret <2 x i32> [[V1]]
+;
+; BE-LABEL: define <2 x i32> @high_half_first(
+; BE-SAME: i64 [[X:%.*]]) {
+; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; BE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; BE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; BE-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @low_half_first(i64 %x) {
+; LE-LABEL: define <2 x i32> @low_half_first(
+; LE-SAME: i64 [[X:%.*]]) {
+; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; LE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T0]], i64 0
+; LE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; LE-NEXT:    ret <2 x i32> [[V1]]
+;
+; BE-LABEL: define <2 x i32> @low_half_first(
+; BE-SAME: i64 [[X:%.*]]) {
+; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; BE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T0]], i64 0
+; BE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; BE-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t0, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t1, i64 1
+  ret <2 x i32> %v1
+}
+
+define <4 x i16> @reversed_quarters(i64 %x) {
+; LE-LABEL: define <4 x i16> @reversed_quarters(
+; LE-SAME: i64 [[X:%.*]]) {
+; LE-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
+; LE-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; LE-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
+; LE-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; LE-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; LE-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; LE-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
+; LE-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
+; LE-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
+; LE-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; LE-NEXT:    ret <4 x i16> [[V3]]
+;
+; BE-LABEL: define <4 x i16> @reversed_quarters(
+; BE-SAME: i64 [[X:%.*]]) {
+; BE-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
+; BE-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; BE-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
+; BE-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; BE-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; BE-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; BE-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
+; BE-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
+; BE-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
+; BE-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; BE-NEXT:    ret <4 x i16> [[V3]]
+;
+  %s1 = lshr i64 %x, 16
+  %s2 = lshr i64 %x, 32
+  %s3 = lshr i64 %x, 48
+  %t0 = trunc i64 %x to i16
+  %t1 = trunc i64 %s1 to i16
+  %t2 = trunc i64 %s2 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <4 x i16> poison, i16 %t3, i64 0
+  %v1 = insertelement <4 x i16> %v0, i16 %t2, i64 1
+  %v2 = insertelement <4 x i16> %v1, i16 %t1, i64 2
+  %v3 = insertelement <4 x i16> %v2, i16 %t0, i64 3
+  ret <4 x i16> %v3
+}
+
+define <4 x i32> @repeated_halves(i64 %x) {
+; LE-LABEL: define <4 x i32> @repeated_halves(
+; LE-SAME: i64 [[X:%.*]]) {
+; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; LE-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
+; LE-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
+; LE-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
+; LE-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; LE-NEXT:    ret <4 x i32> [[V3]]
+;
+; BE-LABEL: define <4 x i32> @repeated_halves(
+; BE-SAME: i64 [[X:%.*]]) {
+; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; BE-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
+; BE-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
+; BE-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
+; BE-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; BE-NEXT:    ret <4 x i32> [[V3]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <4 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %t0, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %t1, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %t0, i64 3
+  ret <4 x i32> %v3
+}
diff --git a/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll b/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll
new file mode 100644
index 00000000000000..82d2766983b54c
--- /dev/null
+++ b/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll
@@ -0,0 +1,554 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=+sse2 | FileCheck %s --check-prefixes=CHECK,SSE
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s --check-prefixes=CHECK,AVX
+
+; Insertelement chains whose elements are all parts of the same scalar can be
+; a bitcast of the scalar and a shuffle.
+
+define <2 x i32> @swapped_halves(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @swapped_halves(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x float> @swapped_halves_float(i64 %x) {
+; CHECK-LABEL: define <2 x float> @swapped_halves_float(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[F1:%.*]] = bitcast i32 [[T1]] to float
+; CHECK-NEXT:    [[F0:%.*]] = bitcast i32 [[T0]] to float
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x float> poison, float [[F1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x float> [[V0]], float [[F0]], i64 1
+; CHECK-NEXT:    ret <2 x float> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %f1 = bitcast i32 %t1 to float
+  %f0 = bitcast i32 %t0 to float
+  %v0 = insertelement <2 x float> poison, float %f1, i64 0
+  %v1 = insertelement <2 x float> %v0, float %f0, i64 1
+  ret <2 x float> %v1
+}
+
+define <4 x i16> @reversed_quarters(i64 %x) {
+; CHECK-LABEL: define <4 x i16> @reversed_quarters(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
+; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; CHECK-NEXT:    ret <4 x i16> [[V3]]
+;
+  %s1 = lshr i64 %x, 16
+  %s2 = lshr i64 %x, 32
+  %s3 = lshr i64 %x, 48
+  %t0 = trunc i64 %x to i16
+  %t1 = trunc i64 %s1 to i16
+  %t2 = trunc i64 %s2 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <4 x i16> poison, i16 %t3, i64 0
+  %v1 = insertelement <4 x i16> %v0, i16 %t2, i64 1
+  %v2 = insertelement <4 x i16> %v1, i16 %t1, i64 2
+  %v3 = insertelement <4 x i16> %v2, i16 %t0, i64 3
+  ret <4 x i16> %v3
+}
+
+define <4 x i32> @reversed_i128(i128 %x) {
+; CHECK-LABEL: define <4 x i32> @reversed_i128(
+; CHECK-SAME: i128 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S1:%.*]] = lshr i128 [[X]], 32
+; CHECK-NEXT:    [[S2:%.*]] = lshr i128 [[X]], 64
+; CHECK-NEXT:    [[S3:%.*]] = lshr i128 [[X]], 96
+; CHECK-NEXT:    [[T0:%.*]] = trunc i128 [[X]] to i32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i128 [[S1]] to i32
+; CHECK-NEXT:    [[T2:%.*]] = trunc i128 [[S2]] to i32
+; CHECK-NEXT:    [[T3:%.*]] = trunc i128 [[S3]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T3]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T2]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
+; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; CHECK-NEXT:    ret <4 x i32> [[V3]]
+;
+  %s1 = lshr i128 %x, 32
+  %s2 = lshr i128 %x, 64
+  %s3 = lshr i128 %x, 96
+  %t0 = trunc i128 %x to i32
+  %t1 = trunc i128 %s1 to i32
+  %t2 = trunc i128 %s2 to i32
+  %t3 = trunc i128 %s3 to i32
+  %v0 = insertelement <4 x i32> poison, i32 %t3, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %t2, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %t1, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %t0, i64 3
+  ret <4 x i32> %v3
+}
+
+; The scalar has fewer parts than the vector has elements.
+define <4 x i32> @repeated_halves(i64 %x) {
+; CHECK-LABEL: define <4 x i32> @repeated_halves(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
+; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; CHECK-NEXT:    ret <4 x i32> [[V3]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <4 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %t0, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %t1, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %t0, i64 3
+  ret <4 x i32> %v3
+}
+
+; The scalar has more parts than the vector has elements.
+define <2 x i16> @odd_quarters(i64 %x) {
+; CHECK-LABEL: define <2 x i16> @odd_quarters(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i16> poison, i16 [[T3]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i16> [[V0]], i16 [[T1]], i64 1
+; CHECK-NEXT:    ret <2 x i16> [[V1]]
+;
+  %s1 = lshr i64 %x, 16
+  %s3 = lshr i64 %x, 48
+  %t1 = trunc i64 %s1 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <2 x i16> poison, i16 %t3, i64 0
+  %v1 = insertelement <2 x i16> %v0, i16 %t1, i64 1
+  ret <2 x i16> %v1
+}
+
+define <4 x i16> @in_order_quarters(i64 %x) {
+; CHECK-LABEL: define <4 x i16> @in_order_quarters(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T0]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T1]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T2]], i64 2
+; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T3]], i64 3
+; CHECK-NEXT:    ret <4 x i16> [[V3]]
+;
+  %s1 = lshr i64 %x, 16
+  %s2 = lshr i64 %x, 32
+  %s3 = lshr i64 %x, 48
+  %t0 = trunc i64 %x to i16
+  %t1 = trunc i64 %s1 to i16
+  %t2 = trunc i64 %s2 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <4 x i16> poison, i16 %t0, i64 0
+  %v1 = insertelement <4 x i16> %v0, i16 %t1, i64 1
+  %v2 = insertelement <4 x i16> %v1, i16 %t2, i64 2
+  %v3 = insertelement <4 x i16> %v2, i16 %t3, i64 3
+  ret <4 x i16> %v3
+}
+
+; Elements that are not inserted stay poison.
+define <4 x i16> @missing_elts(i64 %x) {
+; CHECK-LABEL: define <4 x i16> @missing_elts(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 2
+; CHECK-NEXT:    ret <4 x i16> [[V1]]
+;
+  %s2 = lshr i64 %x, 32
+  %s3 = lshr i64 %x, 48
+  %t2 = trunc i64 %s2 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <4 x i16> poison, i16 %t3, i64 0
+  %v1 = insertelement <4 x i16> %v0, i16 %t2, i64 2
+  ret <4 x i16> %v1
+}
+
+define <2 x i32> @undef_base(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @undef_base(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> undef, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> undef, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @splat_high_half(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @splat_high_half(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t1, i64 1
+  ret <2 x i32> %v1
+}
+
+; Only the last insert to an element counts.
+define <2 x i32> @overwritten_elt(i64 %x, i32 %y) {
+; CHECK-LABEL: define <2 x i32> @overwritten_elt(
+; CHECK-SAME: i64 [[X:%.*]], i32 [[Y:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[Y]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <2 x i32> [[V1]], i32 [[T1]], i64 0
+; CHECK-NEXT:    ret <2 x i32> [[V2]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %y, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  %v2 = insertelement <2 x i32> %v1, i32 %t1, i64 0
+  ret <2 x i32> %v2
+}
+
+define <2 x i32> @double_source(double %d) {
+; CHECK-LABEL: define <2 x i32> @double_source(
+; CHECK-SAME: double [[D:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[X:%.*]] = bitcast double [[D]] to i64
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %x = bitcast double %d to i64
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+; Every element is inserted, so the base does not matter.
+define <2 x i32> @overwritten_base(i64 %x, <2 x i32> %base) {
+; CHECK-LABEL: define <2 x i32> @overwritten_base(
+; CHECK-SAME: i64 [[X:%.*]], <2 x i32> [[BASE:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> [[BASE]], i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> %base, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @swapped_halves_extra_use(i64 %x, ptr %p) {
+; CHECK-LABEL: define <2 x i32> @swapped_halves_extra_use(
+; CHECK-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    store i32 [[T1]], ptr [[P]], align 4
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  store i32 %t1, ptr %p
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+; Negative tests
+
+define <2 x i32> @different_sources(i64 %x, i64 %y) {
+; CHECK-LABEL: define <2 x i32> @different_sources(
+; CHECK-SAME: i64 [[X:%.*]], i64 [[Y:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[Y]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %y to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @unaligned_shift(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @unaligned_shift(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 16
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 16
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @ashr(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @ashr(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = ashr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = ashr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @variable_shift(i64 %x, i64 %s) {
+; CHECK-LABEL: define <2 x i32> @variable_shift(
+; CHECK-SAME: i64 [[X:%.*]], i64 [[S:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], [[S]]
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, %s
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @variable_index(i64 %x, i64 %i) {
+; CHECK-LABEL: define <2 x i32> @variable_index(
+; CHECK-SAME: i64 [[X:%.*]], i64 [[I:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 [[I]]
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 %i
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @non_poison_base(i64 %x, <2 x i32> %base) {
+; CHECK-LABEL: define <2 x i32> @non_poison_base(
+; CHECK-SAME: i64 [[X:%.*]], <2 x i32> [[BASE:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> [[BASE]], i32 [[T1]], i64 0
+; CHECK-NEXT:    ret <2 x i32> [[V0]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %v0 = insertelement <2 x i32> %base, i32 %t1, i64 0
+  ret <2 x i32> %v0
+}
+
+; The elements that are not inserted would become poison instead of undef.
+define <4 x i16> @undef_base_missing_elts(i64 %x) {
+; CHECK-LABEL: define <4 x i16> @undef_base_missing_elts(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> undef, i16 [[T3]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 2
+; CHECK-NEXT:    ret <4 x i16> [[V1]]
+;
+  %s2 = lshr i64 %x, 32
+  %s3 = lshr i64 %x, 48
+  %t2 = trunc i64 %s2 to i16
+  %t3 = trunc i64 %s3 to i16
+  %v0 = insertelement <4 x i16> undef, i16 %t3, i64 0
+  %v1 = insertelement <4 x i16> %v0, i16 %t2, i64 2
+  ret <4 x i16> %v1
+}
+
+define <2 x i32> @out_of_range_index(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @out_of_range_index(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 2
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 2
+  ret <2 x i32> %v1
+}
+
+define <2 x i32> @out_of_range_shift(i64 %x) {
+; CHECK-LABEL: define <2 x i32> @out_of_range_shift(
+; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 64
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 64
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+; The intermediate vector is used elsewhere, so the chain ends there.
+define <2 x i32> @extra_use_of_insert(i64 %x, ptr %p) {
+; CHECK-LABEL: define <2 x i32> @extra_use_of_insert(
+; CHECK-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    store <2 x i32> [[V0]], ptr [[P]], align 8
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i64 %x, 32
+  %t1 = trunc i64 %hi to i32
+  %t0 = trunc i64 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  store <2 x i32> %v0, ptr %p
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+; The scalar is not a whole number of elements.
+define <2 x i32> @partial_source(i48 %x) {
+; CHECK-LABEL: define <2 x i32> @partial_source(
+; CHECK-SAME: i48 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[HI:%.*]] = lshr i48 [[X]], 32
+; CHECK-NEXT:    [[T1:%.*]] = trunc i48 [[HI]] to i32
+; CHECK-NEXT:    [[T0:%.*]] = trunc i48 [[X]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V1]]
+;
+  %hi = lshr i48 %x, 32
+  %t1 = trunc i48 %hi to i32
+  %t0 = trunc i48 %x to i32
+  %v0 = insertelement <2 x i32> poison, i32 %t1, i64 0
+  %v1 = insertelement <2 x i32> %v0, i32 %t0, i64 1
+  ret <2 x i32> %v1
+}
+
+; Elements are truncated from each element of a vector, not from its bits.
+define <2 x i32> @vector_trunc(<2 x i32> %x) {
+; CHECK-LABEL: define <2 x i32> @vector_trunc(
+; CHECK-SAME: <2 x i32> [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[T:%.*]] = trunc <2 x i32> [[X]] to <2 x i16>
+; CHECK-NEXT:    [[B:%.*]] = bitcast <2 x i16> [[T]] to i32
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[B]], i64 1
+; CHECK-NEXT:    ret <2 x i32> [[V0]]
+;
+  %t = trunc <2 x i32> %x to <2 x i16>
+  %b = bitcast <2 x i16> %t to i32
+  %v0 = insertelement <2 x i32> poison, i32 %b, i64 1
+  ret <2 x i32> %v0
+}
+
+define <8 x i1> @bool_elts(i8 %x) {
+; CHECK-LABEL: define <8 x i1> @bool_elts(
+; CHECK-SAME: i8 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[S1:%.*]] = lshr i8 [[X]], 1
+; CHECK-NEXT:    [[T1:%.*]] = trunc i8 [[S1]] to i1
+; CHECK-NEXT:    [[T0:%.*]] = trunc i8 [[X]] to i1
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <8 x i1> poison, i1 [[T1]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <8 x i1> [[V0]], i1 [[T0]], i64 1
+; CHECK-NEXT:    ret <8 x i1> [[V1]]
+;
+  %s1 = lshr i8 %x, 1
+  %t1 = trunc i8 %s1 to i1
+  %t0 = trunc i8 %x to i1
+  %v0 = insertelement <8 x i1> poison, i1 %t1, i64 0
+  %v1 = insertelement <8 x i1> %v0, i1 %t0, i64 1
+  ret <8 x i1> %v1
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; AVX: {{.*}}
+; SSE: {{.*}}

>From 0432c53054c5f7baba79cb0eb647482d6feadc81 Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Thu, 24 Sep 2026 22:34:22 +0200
Subject: [PATCH 2/2] [VectorCombine] Fold insertelement chains of scalar parts
 to a bitcast and shuffle

An insertelement chain whose elements are all truncated parts of the
same scalar is lowered element by element, unless InstCombine can turn
an in-order pair of halves into a bitcast. The SLP vectorizer produces
such chains for the fields of a struct that SROA loaded as one integer,
e.g. when summing two float fields in a loop:

  %hi = lshr i64 %x, 32
  %h = trunc i64 %hi to i32
  %l = trunc i64 %x to i32
  %v0 = insertelement <2 x i32> poison, i32 %h, i64 0
  %v1 = insertelement <2 x i32> %v0, i32 %l, i64 1

which X86 lowers to shrq + vmovd + vpinsrd. If TTI says it is cheaper,
replace the chain by a shuffle of the bitcast scalar:

  %b = bitcast i64 %x to <2 x i32>
  %v1 = shufflevector <2 x i32> %b, <2 x i32> poison, <2 x i32> <i32 1, i32 0>

The elements may be bitcast to FP, the scalar may have more or fewer
parts than the vector has elements, and big-endian targets are handled.
Elements that are not inserted become poison, so the chain must start
from poison unless it inserts every element.

Assisted-by: Claude Code, Codex
---
 .../Transforms/Vectorize/VectorCombine.cpp    | 111 +++++++++++
 .../AArch64/block_scaling_decompr_8bit.ll     |   5 +-
 .../X86/vector-reduction-of-scalar-parts.ll   |  78 +++-----
 .../AArch64/insert-scalar-parts.ll            |  69 ++-----
 .../VectorCombine/X86/insert-scalar-parts.ll  | 185 +++++++-----------
 5 files changed, 223 insertions(+), 225 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index ddb06610d6572f..95d1adac12350d 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -125,6 +125,7 @@ class VectorCombine {
   bool foldInsExtFNeg(Instruction &I);
   bool foldInsExtBinop(Instruction &I);
   bool foldInsExtVectorToShuffle(Instruction &I);
+  bool foldInsertScalarPartsToShuffle(Instruction &I);
   bool foldBitOpOfCastops(Instruction &I);
   bool foldBitOpOfCastConstant(Instruction &I);
   bool foldBitcastShuffle(Instruction &I);
@@ -6036,6 +6037,114 @@ bool VectorCombine::foldInsExtVectorToShuffle(Instruction &I) {
   return true;
 }
 
+/// Try to replace a chain of insertelements of parts of the same scalar with a
+/// bitcast and a shuffle (little endian):
+///   insert (insert poison, (trunc (lshr X, 32)), 0), (trunc X), 1 -->
+///   shuffle (bitcast X to <2 x i32>), poison, <1, 0>
+bool VectorCombine::foldInsertScalarPartsToShuffle(Instruction &I) {
+  auto *VecTy = dyn_cast<FixedVectorType>(I.getType());
+  if (!VecTy)
+    return false;
+
+  // Start from the last insertelement of the chain.
+  if (I.hasOneUse() && isa<InsertElementInst>(I.user_back()) &&
+      I.user_back()->getOperand(0) == &I)
+    return false;
+
+  Type *EltTy = VecTy->getElementType();
+  if ((!EltTy->isIntegerTy() && !EltTy->isIEEELikeFPTy()) ||
+      !DL->typeSizeEqualsStoreSize(EltTy))
+    return false;
+  unsigned EltBits = EltTy->getPrimitiveSizeInBits();
+  unsigned NumElts = VecTy->getNumElements();
+
+  Value *Src = nullptr;
+  unsigned NumSrcElts = 0;
+  SmallVector<int> Mask(NumElts, PoisonMaskElem);
+  APInt DemandedElts = APInt::getZero(NumElts);
+  InstructionCost OldCost = 0;
+  Value *Vec = &I;
+  while (auto *Ins = dyn_cast<InsertElementInst>(Vec)) {
+    if (Ins != &I && !Ins->hasOneUse())
+      return false;
+    uint64_t Idx;
+    if (!match(Ins->getOperand(2), m_ConstantInt(Idx)) || Idx >= NumElts)
+      return false;
+    Vec = Ins->getOperand(0);
+    // A later insert to the same element overrides this one.
+    if (DemandedElts[Idx])
+      continue;
+    DemandedElts.setBit(Idx);
+
+    // Match (bitcast (trunc (lshr X, ShAmt))), the bitcast and shift being
+    // optional.
+    Value *Elt = Ins->getOperand(1);
+    Value *Trunc = Elt;
+    match(Trunc, m_BitCast(m_Value(Trunc)));
+    Value *X;
+    if (!match(Trunc, m_Trunc(m_Value(X))) || !X->getType()->isIntegerTy() ||
+        Trunc->getType()->getPrimitiveSizeInBits() != EltBits)
+      return false;
+    Value *Shift = nullptr;
+    uint64_t ShAmt = 0;
+    if (match(X, m_LShr(m_Value(), m_ConstantInt(ShAmt)))) {
+      Shift = X;
+      X = cast<Instruction>(Shift)->getOperand(0);
+    }
+
+    if (!Src) {
+      unsigned SrcBits = X->getType()->getIntegerBitWidth();
+      if (SrcBits % EltBits)
+        return false;
+      Src = X;
+      NumSrcElts = SrcBits / EltBits;
+    } else if (X != Src) {
+      return false;
+    }
+    if (ShAmt % EltBits || ShAmt / EltBits >= NumSrcElts)
+      return false;
+    unsigned Part = ShAmt / EltBits;
+    Mask[Idx] = DL->isBigEndian() ? NumSrcElts - 1 - Part : Part;
+
+    // The scalar ops die with the chain if it is their only user.
+    for (Value *V : {Elt == Trunc ? nullptr : Elt, Trunc, Shift}) {
+      if (!V)
+        continue;
+      if (!V->hasOneUse())
+        break;
+      OldCost += TTI.getInstructionCost(cast<Instruction>(V), CostKind);
+    }
+  }
+  // Elements that are not inserted become poison, so the base must be poison
+  // unless every element is inserted.
+  if (!Src || (!isa<PoisonValue>(Vec) && !DemandedElts.isAllOnes()))
+    return false;
+
+  OldCost += TTI.getScalarizationOverhead(VecTy, DemandedElts, /*Insert=*/true,
+                                          /*Extract=*/false, CostKind);
+
+  auto *SrcVecTy = FixedVectorType::get(EltTy, NumSrcElts);
+  InstructionCost NewCost =
+      TTI.getCastInstrCost(Instruction::BitCast, SrcVecTy, Src->getType(),
+                           TTI::CastContextHint::None, CostKind);
+  bool IsIdentity = NumSrcElts == NumElts &&
+                    ShuffleVectorInst::isIdentityMask(Mask, NumSrcElts);
+  if (!IsIdentity)
+    NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, VecTy, SrcVecTy,
+                                  CostKind, Mask);
+
+  LLVM_DEBUG(dbgs() << "Found an insertelement chain of scalar parts: " << I
+                    << "\n  OldCost: " << OldCost << " vs NewCost: " << NewCost
+                    << "\n");
+  if (!OldCost.isValid() || !NewCost.isValid() || NewCost >= OldCost)
+    return false;
+
+  Value *Cast = Builder.CreateBitCast(Src, SrcVecTy);
+  Value *Shuf = IsIdentity ? Cast : Builder.CreateShuffleVector(Cast, Mask);
+  replaceValue(I, *Shuf);
+  return true;
+}
+
 /// Fold away a matched pair of vector.deinterleave/interleave intrinsics
 /// with a chain of elementwise operations on each between the
 /// deinterleave and interleave.
@@ -6953,6 +7062,8 @@ bool VectorCombine::run() {
           return true;
         if (foldInsExtVectorToShuffle(I))
           return true;
+        if (foldInsertScalarPartsToShuffle(I))
+          return true;
         break;
       case Instruction::ShuffleVector:
         if (foldPermuteOfBinops(I))
diff --git a/llvm/test/Transforms/PhaseOrdering/AArch64/block_scaling_decompr_8bit.ll b/llvm/test/Transforms/PhaseOrdering/AArch64/block_scaling_decompr_8bit.ll
index 381fba8257394c..a78937cec54f37 100644
--- a/llvm/test/Transforms/PhaseOrdering/AArch64/block_scaling_decompr_8bit.ll
+++ b/llvm/test/Transforms/PhaseOrdering/AArch64/block_scaling_decompr_8bit.ll
@@ -413,17 +413,16 @@ define internal noundef <8 x i16> @_ZL24cmplx_mul_combined_re_im11__Int16x8_t20c
 ; CHECK-LABEL: define internal fastcc noundef <8 x i16> @_ZL24cmplx_mul_combined_re_im11__Int16x8_t20cmplx_int16_t(
 ; CHECK-SAME: <8 x i16> noundef [[A:%.*]], i64 [[SCALE_COERCE:%.*]]) unnamed_addr #[[ATTR1:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[SCALE_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[SCALE_COERCE]] to i16
 ; CHECK-NEXT:    [[SCALE_SROA_2_0_EXTRACT_SHIFT36:%.*]] = lshr i64 [[SCALE_COERCE]], 16
 ; CHECK-NEXT:    [[SCALE_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[SCALE_SROA_2_0_EXTRACT_SHIFT36]] to i16
 ; CHECK-NEXT:    [[SHUFFLE_I:%.*]] = shufflevector <8 x i16> [[A]], <8 x i16> poison, <8 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6>
-; CHECK-NEXT:    [[VECINIT_I19:%.*]] = insertelement <8 x i16> poison, i16 [[SCALE_SROA_0_0_EXTRACT_TRUNC]], i64 0
+; CHECK-NEXT:    [[TMP2:%.*]] = bitcast i64 [[SCALE_COERCE]] to <4 x i16>
 ; CHECK-NEXT:    [[VECINIT_I:%.*]] = insertelement <8 x i16> poison, i16 [[SCALE_SROA_2_0_EXTRACT_TRUNC]], i64 0
 ; CHECK-NEXT:    [[VECINIT7_I:%.*]] = shufflevector <8 x i16> [[VECINIT_I]], <8 x i16> poison, <8 x i32> zeroinitializer
 ; CHECK-NEXT:    [[VQNEGQ_V1_I:%.*]] = tail call <8 x i16> @llvm.aarch64.neon.sqneg.v8i16(<8 x i16> [[VECINIT7_I]])
 ; CHECK-NEXT:    [[VBSL5_I:%.*]] = shufflevector <8 x i16> [[VQNEGQ_V1_I]], <8 x i16> [[VECINIT_I]], <8 x i32> <i32 0, i32 8, i32 2, i32 8, i32 4, i32 8, i32 6, i32 8>
 ; CHECK-NEXT:    [[SHUFFLE_I85:%.*]] = shufflevector <8 x i16> [[A]], <8 x i16> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[SHUFFLE_I82:%.*]] = shufflevector <8 x i16> [[VECINIT_I19]], <8 x i16> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[SHUFFLE_I82:%.*]] = shufflevector <4 x i16> [[TMP2]], <4 x i16> poison, <4 x i32> zeroinitializer
 ; CHECK-NEXT:    [[VQDMULL_V2_I72:%.*]] = tail call <4 x i32> @llvm.aarch64.neon.sqdmull.v4i32(<4 x i16> [[SHUFFLE_I85]], <4 x i16> [[SHUFFLE_I82]])
 ; CHECK-NEXT:    [[SHUFFLE_I97:%.*]] = shufflevector <8 x i16> [[A]], <8 x i16> poison, <4 x i32> <i32 4, i32 5, i32 6, i32 7>
 ; CHECK-NEXT:    [[VQDMULL_V2_I:%.*]] = tail call <4 x i32> @llvm.aarch64.neon.sqdmull.v4i32(<4 x i16> [[SHUFFLE_I97]], <4 x i16> [[SHUFFLE_I82]])
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
index a0b4c118b9542a..3cd21931f0c3e2 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-of-scalar-parts.ll
@@ -7,58 +7,29 @@
 ; gather should become a shuffle of the loaded vector.
 
 define [2 x float] @sum_pairs(ptr %p, i64 %n) {
-; SSE2-LABEL: define [2 x float] @sum_pairs(
-; SSE2-SAME: ptr nofree readonly captures(none) [[P:%.*]], i64 [[N:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
-; SSE2-NEXT:  [[ENTRY:.*]]:
-; SSE2-NEXT:    [[SKIP:%.*]] = icmp slt i64 [[N]], 1
-; SSE2-NEXT:    br i1 [[SKIP]], label %[[EXIT:.*]], label %[[LOOP:.*]]
-; SSE2:       [[LOOP]]:
-; SSE2-NEXT:    [[PTR:%.*]] = phi ptr [ [[PTR_NEXT:%.*]], %[[LOOP]] ], [ [[P]], %[[ENTRY]] ]
-; SSE2-NEXT:    [[I:%.*]] = phi i64 [ [[I_NEXT:%.*]], %[[LOOP]] ], [ 0, %[[ENTRY]] ]
-; SSE2-NEXT:    [[TMP0:%.*]] = phi <2 x float> [ [[TMP2:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
-; SSE2-NEXT:    [[X1:%.*]] = load <2 x float>, ptr [[PTR]], align 1
-; SSE2-NEXT:    [[TMP1:%.*]] = shufflevector <2 x float> [[X1]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
-; SSE2-NEXT:    [[TMP2]] = fadd <2 x float> [[TMP0]], [[TMP1]]
-; SSE2-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 8
-; SSE2-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 1
-; SSE2-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
-; SSE2-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
-; SSE2:       [[EXIT]]:
-; SSE2-NEXT:    [[TMP3:%.*]] = phi <2 x float> [ zeroinitializer, %[[ENTRY]] ], [ [[TMP2]], %[[LOOP]] ]
-; SSE2-NEXT:    [[TMP4:%.*]] = extractelement <2 x float> [[TMP3]], i64 1
-; SSE2-NEXT:    [[R0:%.*]] = insertvalue [2 x float] poison, float [[TMP4]], 0
-; SSE2-NEXT:    [[TMP5:%.*]] = extractelement <2 x float> [[TMP3]], i64 0
-; SSE2-NEXT:    [[R1:%.*]] = insertvalue [2 x float] [[R0]], float [[TMP5]], 1
-; SSE2-NEXT:    ret [2 x float] [[R1]]
-;
-; AVX2-LABEL: define [2 x float] @sum_pairs(
-; AVX2-SAME: ptr nofree readonly captures(none) [[P:%.*]], i64 [[N:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
-; AVX2-NEXT:  [[ENTRY:.*]]:
-; AVX2-NEXT:    [[SKIP:%.*]] = icmp slt i64 [[N]], 1
-; AVX2-NEXT:    br i1 [[SKIP]], label %[[EXIT:.*]], label %[[LOOP:.*]]
-; AVX2:       [[LOOP]]:
-; AVX2-NEXT:    [[PTR:%.*]] = phi ptr [ [[PTR_NEXT:%.*]], %[[LOOP]] ], [ [[P]], %[[ENTRY]] ]
-; AVX2-NEXT:    [[I:%.*]] = phi i64 [ [[I_NEXT:%.*]], %[[LOOP]] ], [ 0, %[[ENTRY]] ]
-; AVX2-NEXT:    [[TMP0:%.*]] = phi <2 x float> [ [[TMP6:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
-; AVX2-NEXT:    [[X:%.*]] = load i64, ptr [[PTR]], align 1
-; AVX2-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; AVX2-NEXT:    [[TMP1:%.*]] = trunc nuw i64 [[HI]] to i32
-; AVX2-NEXT:    [[TMP2:%.*]] = insertelement <2 x i32> poison, i32 [[TMP1]], i64 0
-; AVX2-NEXT:    [[TMP3:%.*]] = trunc i64 [[X]] to i32
-; AVX2-NEXT:    [[TMP4:%.*]] = insertelement <2 x i32> [[TMP2]], i32 [[TMP3]], i64 1
-; AVX2-NEXT:    [[TMP5:%.*]] = bitcast <2 x i32> [[TMP4]] to <2 x float>
-; AVX2-NEXT:    [[TMP6]] = fadd <2 x float> [[TMP0]], [[TMP5]]
-; AVX2-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 8
-; AVX2-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 1
-; AVX2-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
-; AVX2-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
-; AVX2:       [[EXIT]]:
-; AVX2-NEXT:    [[TMP7:%.*]] = phi <2 x float> [ zeroinitializer, %[[ENTRY]] ], [ [[TMP6]], %[[LOOP]] ]
-; AVX2-NEXT:    [[TMP8:%.*]] = extractelement <2 x float> [[TMP7]], i64 1
-; AVX2-NEXT:    [[R0:%.*]] = insertvalue [2 x float] poison, float [[TMP8]], 0
-; AVX2-NEXT:    [[TMP9:%.*]] = extractelement <2 x float> [[TMP7]], i64 0
-; AVX2-NEXT:    [[R1:%.*]] = insertvalue [2 x float] [[R0]], float [[TMP9]], 1
-; AVX2-NEXT:    ret [2 x float] [[R1]]
+; CHECK-LABEL: define [2 x float] @sum_pairs(
+; CHECK-SAME: ptr nofree readonly captures(none) [[P:%.*]], i64 [[N:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[SKIP:%.*]] = icmp slt i64 [[N]], 1
+; CHECK-NEXT:    br i1 [[SKIP]], label %[[EXIT:.*]], label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[PTR:%.*]] = phi ptr [ [[PTR_NEXT:%.*]], %[[LOOP]] ], [ [[P]], %[[ENTRY]] ]
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ [[I_NEXT:%.*]], %[[LOOP]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = phi <2 x float> [ [[TMP2:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
+; CHECK-NEXT:    [[X1:%.*]] = load <2 x float>, ptr [[PTR]], align 1
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <2 x float> [[X1]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
+; CHECK-NEXT:    [[TMP2]] = fadd <2 x float> [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 8
+; CHECK-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[TMP3:%.*]] = phi <2 x float> [ zeroinitializer, %[[ENTRY]] ], [ [[TMP2]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x float> [[TMP3]], i64 1
+; CHECK-NEXT:    [[R0:%.*]] = insertvalue [2 x float] poison, float [[TMP4]], 0
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <2 x float> [[TMP3]], i64 0
+; CHECK-NEXT:    [[R1:%.*]] = insertvalue [2 x float] [[R0]], float [[TMP5]], 1
+; CHECK-NEXT:    ret [2 x float] [[R1]]
 ;
 entry:
   %skip = icmp slt i64 %n, 1
@@ -100,4 +71,5 @@ exit:
 ; AVX2: [[META1]] = !{!"llvm.loop.unroll.disable"}
 ;.
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; CHECK: {{.*}}
+; AVX2: {{.*}}
+; SSE2: {{.*}}
diff --git a/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll b/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
index 2c7d80440ff271..28ea2686da6993 100644
--- a/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
+++ b/llvm/test/Transforms/VectorCombine/AArch64/insert-scalar-parts.ll
@@ -8,20 +8,13 @@
 define <2 x i32> @high_half_first(i64 %x) {
 ; LE-LABEL: define <2 x i32> @high_half_first(
 ; LE-SAME: i64 [[X:%.*]]) {
-; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; LE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; LE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; LE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; LE-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; LE-NEXT:    ret <2 x i32> [[V1]]
 ;
 ; BE-LABEL: define <2 x i32> @high_half_first(
 ; BE-SAME: i64 [[X:%.*]]) {
-; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; BE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; BE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; BE-NEXT:    [[V1:%.*]] = bitcast i64 [[X]] to <2 x i32>
 ; BE-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -35,20 +28,13 @@ define <2 x i32> @high_half_first(i64 %x) {
 define <2 x i32> @low_half_first(i64 %x) {
 ; LE-LABEL: define <2 x i32> @low_half_first(
 ; LE-SAME: i64 [[X:%.*]]) {
-; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; LE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T0]], i64 0
-; LE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; LE-NEXT:    [[V1:%.*]] = bitcast i64 [[X]] to <2 x i32>
 ; LE-NEXT:    ret <2 x i32> [[V1]]
 ;
 ; BE-LABEL: define <2 x i32> @low_half_first(
 ; BE-SAME: i64 [[X:%.*]]) {
-; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; BE-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T0]], i64 0
-; BE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; BE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; BE-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; BE-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -62,32 +48,13 @@ define <2 x i32> @low_half_first(i64 %x) {
 define <4 x i16> @reversed_quarters(i64 %x) {
 ; LE-LABEL: define <4 x i16> @reversed_quarters(
 ; LE-SAME: i64 [[X:%.*]]) {
-; LE-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
-; LE-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
-; LE-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
-; LE-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; LE-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; LE-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; LE-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
-; LE-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
-; LE-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
-; LE-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; LE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <4 x i16>
+; LE-NEXT:    [[V3:%.*]] = shufflevector <4 x i16> [[TMP1]], <4 x i16> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
 ; LE-NEXT:    ret <4 x i16> [[V3]]
 ;
 ; BE-LABEL: define <4 x i16> @reversed_quarters(
 ; BE-SAME: i64 [[X:%.*]]) {
-; BE-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
-; BE-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
-; BE-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
-; BE-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; BE-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; BE-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; BE-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
-; BE-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
-; BE-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
-; BE-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; BE-NEXT:    [[V3:%.*]] = bitcast i64 [[X]] to <4 x i16>
 ; BE-NEXT:    ret <4 x i16> [[V3]]
 ;
   %s1 = lshr i64 %x, 16
@@ -107,24 +74,14 @@ define <4 x i16> @reversed_quarters(i64 %x) {
 define <4 x i32> @repeated_halves(i64 %x) {
 ; LE-LABEL: define <4 x i32> @repeated_halves(
 ; LE-SAME: i64 [[X:%.*]]) {
-; LE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; LE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; LE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; LE-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
-; LE-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
-; LE-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
-; LE-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; LE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; LE-NEXT:    [[V3:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> <i32 1, i32 0, i32 1, i32 0>
 ; LE-NEXT:    ret <4 x i32> [[V3]]
 ;
 ; BE-LABEL: define <4 x i32> @repeated_halves(
 ; BE-SAME: i64 [[X:%.*]]) {
-; BE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; BE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; BE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; BE-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
-; BE-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
-; BE-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
-; BE-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; BE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; BE-NEXT:    [[V3:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
 ; BE-NEXT:    ret <4 x i32> [[V3]]
 ;
   %hi = lshr i64 %x, 32
diff --git a/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll b/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll
index 82d2766983b54c..27ac96fc18b7bd 100644
--- a/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/insert-scalar-parts.ll
@@ -8,11 +8,8 @@
 define <2 x i32> @swapped_halves(i64 %x) {
 ; CHECK-LABEL: define <2 x i32> @swapped_halves(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -26,13 +23,8 @@ define <2 x i32> @swapped_halves(i64 %x) {
 define <2 x float> @swapped_halves_float(i64 %x) {
 ; CHECK-LABEL: define <2 x float> @swapped_halves_float(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[F1:%.*]] = bitcast i32 [[T1]] to float
-; CHECK-NEXT:    [[F0:%.*]] = bitcast i32 [[T0]] to float
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x float> poison, float [[F1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x float> [[V0]], float [[F0]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x float>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <2 x float> [[TMP1]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x float> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -48,17 +40,8 @@ define <2 x float> @swapped_halves_float(i64 %x) {
 define <4 x i16> @reversed_quarters(i64 %x) {
 ; CHECK-LABEL: define <4 x i16> @reversed_quarters(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 1
-; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T1]], i64 2
-; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T0]], i64 3
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <4 x i16>
+; CHECK-NEXT:    [[V3:%.*]] = shufflevector <4 x i16> [[TMP1]], <4 x i16> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
 ; CHECK-NEXT:    ret <4 x i16> [[V3]]
 ;
   %s1 = lshr i64 %x, 16
@@ -78,17 +61,8 @@ define <4 x i16> @reversed_quarters(i64 %x) {
 define <4 x i32> @reversed_i128(i128 %x) {
 ; CHECK-LABEL: define <4 x i32> @reversed_i128(
 ; CHECK-SAME: i128 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[S1:%.*]] = lshr i128 [[X]], 32
-; CHECK-NEXT:    [[S2:%.*]] = lshr i128 [[X]], 64
-; CHECK-NEXT:    [[S3:%.*]] = lshr i128 [[X]], 96
-; CHECK-NEXT:    [[T0:%.*]] = trunc i128 [[X]] to i32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i128 [[S1]] to i32
-; CHECK-NEXT:    [[T2:%.*]] = trunc i128 [[S2]] to i32
-; CHECK-NEXT:    [[T3:%.*]] = trunc i128 [[S3]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T3]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T2]], i64 1
-; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
-; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i128 [[X]] to <4 x i32>
+; CHECK-NEXT:    [[V3:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
 ; CHECK-NEXT:    ret <4 x i32> [[V3]]
 ;
   %s1 = lshr i128 %x, 32
@@ -109,13 +83,8 @@ define <4 x i32> @reversed_i128(i128 %x) {
 define <4 x i32> @repeated_halves(i64 %x) {
 ; CHECK-LABEL: define <4 x i32> @repeated_halves(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[T0]], i64 1
-; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[T1]], i64 2
-; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[T0]], i64 3
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V3:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> <i32 1, i32 0, i32 1, i32 0>
 ; CHECK-NEXT:    ret <4 x i32> [[V3]]
 ;
   %hi = lshr i64 %x, 32
@@ -132,12 +101,8 @@ define <4 x i32> @repeated_halves(i64 %x) {
 define <2 x i16> @odd_quarters(i64 %x) {
 ; CHECK-LABEL: define <2 x i16> @odd_quarters(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i16> poison, i16 [[T3]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i16> [[V0]], i16 [[T1]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <4 x i16>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <4 x i16> [[TMP1]], <4 x i16> poison, <2 x i32> <i32 3, i32 1>
 ; CHECK-NEXT:    ret <2 x i16> [[V1]]
 ;
   %s1 = lshr i64 %x, 16
@@ -152,17 +117,7 @@ define <2 x i16> @odd_quarters(i64 %x) {
 define <4 x i16> @in_order_quarters(i64 %x) {
 ; CHECK-LABEL: define <4 x i16> @in_order_quarters(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[X]], 16
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i16
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T0]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T1]], i64 1
-; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i16> [[V1]], i16 [[T2]], i64 2
-; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i16> [[V2]], i16 [[T3]], i64 3
+; CHECK-NEXT:    [[V3:%.*]] = bitcast i64 [[X]] to <4 x i16>
 ; CHECK-NEXT:    ret <4 x i16> [[V3]]
 ;
   %s1 = lshr i64 %x, 16
@@ -183,12 +138,8 @@ define <4 x i16> @in_order_quarters(i64 %x) {
 define <4 x i16> @missing_elts(i64 %x) {
 ; CHECK-LABEL: define <4 x i16> @missing_elts(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[X]], 48
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i16> poison, i16 [[T3]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i16> [[V0]], i16 [[T2]], i64 2
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <4 x i16>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <4 x i16> [[TMP1]], <4 x i16> poison, <4 x i32> <i32 3, i32 poison, i32 2, i32 poison>
 ; CHECK-NEXT:    ret <4 x i16> [[V1]]
 ;
   %s2 = lshr i64 %x, 32
@@ -203,11 +154,8 @@ define <4 x i16> @missing_elts(i64 %x) {
 define <2 x i32> @undef_base(i64 %x) {
 ; CHECK-LABEL: define <2 x i32> @undef_base(
 ; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> undef, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -219,13 +167,19 @@ define <2 x i32> @undef_base(i64 %x) {
 }
 
 define <2 x i32> @splat_high_half(i64 %x) {
-; CHECK-LABEL: define <2 x i32> @splat_high_half(
-; CHECK-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
-; CHECK-NEXT:    ret <2 x i32> [[V1]]
+; SSE-LABEL: define <2 x i32> @splat_high_half(
+; SSE-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; SSE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; SSE-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 1>
+; SSE-NEXT:    ret <2 x i32> [[V1]]
+;
+; AVX-LABEL: define <2 x i32> @splat_high_half(
+; AVX-SAME: i64 [[X:%.*]]) #[[ATTR0]] {
+; AVX-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; AVX-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; AVX-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; AVX-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T1]], i64 1
+; AVX-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
   %t1 = trunc i64 %hi to i32
@@ -238,12 +192,8 @@ define <2 x i32> @splat_high_half(i64 %x) {
 define <2 x i32> @overwritten_elt(i64 %x, i32 %y) {
 ; CHECK-LABEL: define <2 x i32> @overwritten_elt(
 ; CHECK-SAME: i64 [[X:%.*]], i32 [[Y:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[Y]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
-; CHECK-NEXT:    [[V2:%.*]] = insertelement <2 x i32> [[V1]], i32 [[T1]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V2:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x i32> [[V2]]
 ;
   %hi = lshr i64 %x, 32
@@ -259,11 +209,8 @@ define <2 x i32> @double_source(double %d) {
 ; CHECK-LABEL: define <2 x i32> @double_source(
 ; CHECK-SAME: double [[D:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:    [[X:%.*]] = bitcast double [[D]] to i64
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x i32> [[V1]]
 ;
   %x = bitcast double %d to i64
@@ -279,11 +226,8 @@ define <2 x i32> @double_source(double %d) {
 define <2 x i32> @overwritten_base(i64 %x, <2 x i32> %base) {
 ; CHECK-LABEL: define <2 x i32> @overwritten_base(
 ; CHECK-SAME: i64 [[X:%.*]], <2 x i32> [[BASE:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> [[BASE]], i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; CHECK-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
 ; CHECK-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
@@ -295,15 +239,24 @@ define <2 x i32> @overwritten_base(i64 %x, <2 x i32> %base) {
 }
 
 define <2 x i32> @swapped_halves_extra_use(i64 %x, ptr %p) {
-; CHECK-LABEL: define <2 x i32> @swapped_halves_extra_use(
-; CHECK-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    store i32 [[T1]], ptr [[P]], align 4
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
-; CHECK-NEXT:    ret <2 x i32> [[V1]]
+; SSE-LABEL: define <2 x i32> @swapped_halves_extra_use(
+; SSE-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; SSE-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; SSE-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; SSE-NEXT:    store i32 [[T1]], ptr [[P]], align 4
+; SSE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; SSE-NEXT:    [[V1:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 0>
+; SSE-NEXT:    ret <2 x i32> [[V1]]
+;
+; AVX-LABEL: define <2 x i32> @swapped_halves_extra_use(
+; AVX-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; AVX-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; AVX-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; AVX-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; AVX-NEXT:    store i32 [[T1]], ptr [[P]], align 4
+; AVX-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; AVX-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; AVX-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
   %t1 = trunc i64 %hi to i32
@@ -479,15 +432,24 @@ define <2 x i32> @out_of_range_shift(i64 %x) {
 
 ; The intermediate vector is used elsewhere, so the chain ends there.
 define <2 x i32> @extra_use_of_insert(i64 %x, ptr %p) {
-; CHECK-LABEL: define <2 x i32> @extra_use_of_insert(
-; CHECK-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
-; CHECK-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
-; CHECK-NEXT:    store <2 x i32> [[V0]], ptr [[P]], align 8
-; CHECK-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
-; CHECK-NEXT:    ret <2 x i32> [[V1]]
+; SSE-LABEL: define <2 x i32> @extra_use_of_insert(
+; SSE-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; SSE-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; SSE-NEXT:    [[TMP1:%.*]] = bitcast i64 [[X]] to <2 x i32>
+; SSE-NEXT:    [[V0:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 poison>
+; SSE-NEXT:    store <2 x i32> [[V0]], ptr [[P]], align 8
+; SSE-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; SSE-NEXT:    ret <2 x i32> [[V1]]
+;
+; AVX-LABEL: define <2 x i32> @extra_use_of_insert(
+; AVX-SAME: i64 [[X:%.*]], ptr [[P:%.*]]) #[[ATTR0]] {
+; AVX-NEXT:    [[HI:%.*]] = lshr i64 [[X]], 32
+; AVX-NEXT:    [[T1:%.*]] = trunc i64 [[HI]] to i32
+; AVX-NEXT:    [[T0:%.*]] = trunc i64 [[X]] to i32
+; AVX-NEXT:    [[V0:%.*]] = insertelement <2 x i32> poison, i32 [[T1]], i64 0
+; AVX-NEXT:    store <2 x i32> [[V0]], ptr [[P]], align 8
+; AVX-NEXT:    [[V1:%.*]] = insertelement <2 x i32> [[V0]], i32 [[T0]], i64 1
+; AVX-NEXT:    ret <2 x i32> [[V1]]
 ;
   %hi = lshr i64 %x, 32
   %t1 = trunc i64 %hi to i32
@@ -549,6 +511,3 @@ define <8 x i1> @bool_elts(i8 %x) {
   %v1 = insertelement <8 x i1> %v0, i1 %t0, i64 1
   ret <8 x i1> %v1
 }
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; AVX: {{.*}}
-; SSE: {{.*}}



More information about the llvm-commits mailing list