[llvm] [X86] Fold vpextrq $1 from a spilled vector into a direct memory load (PR #203339)
Ye Tian via llvm-commits
llvm-commits at lists.llvm.org
Wed Jul 1 10:49:01 PDT 2026
https://github.com/TianYe717 updated https://github.com/llvm/llvm-project/pull/203339
>From b7c0d08358a780440c30fd042ff0540e1c586026 Mon Sep 17 00:00:00 2001
From: TianYe <939808194 at qq.com>
Date: Fri, 12 Jun 2026 00:41:15 +0800
Subject: [PATCH 1/5] [X86] Fold vpextrq $1 from a spilled vector into a direct
memory load
Instead of reloading the full vector and extracting lane 1 with vpextrq,
load the upper 8 bytes directly from the spill slot at offset 8.
---
llvm/lib/Target/X86/X86InstrInfo.cpp | 15 ++++++
llvm/test/CodeGen/X86/vpextrq-fold.ll | 68 +++++++++++++++++++++++++++
2 files changed, 83 insertions(+)
create mode 100644 llvm/test/CodeGen/X86/vpextrq-fold.ll
diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index 6c86a3cbece2e..4f076afb615d8 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -7444,6 +7444,21 @@ MachineInstr *X86InstrInfo::foldMemoryOperandCustom(
InsertPt, MI))
return NewMI;
break;
+ case X86::VPEXTRQrri:
+ case X86::VPEXTRQZrri:
+ case X86::PEXTRQrri:
+ // Fold: extractelt(v2i64 vector, 1) where vector is a spilled stack slot.
+ // Instead of reloading the full vector and extracting with vpextrq, load
+ // the upper 8 bytes directly from the spill slot at offset 8.
+ if (OpNum == 1 && MI.getOperand(2).getImm() == 1 &&
+ Alignment >= Align(8)) {
+ MachineInstrBuilder MIB = BuildMI(*InsertPt->getParent(), InsertPt,
+ MI.getDebugLoc(), get(X86::MOV64rm));
+ MIB.add(MI.getOperand(0)); // dst register
+ addOperands(MIB, MOs, 8); // offset 8 for upper lane
+ return MIB;
+ }
+ break;
}
return nullptr;
diff --git a/llvm/test/CodeGen/X86/vpextrq-fold.ll b/llvm/test/CodeGen/X86/vpextrq-fold.ll
new file mode 100644
index 0000000000000..197aebb18cf5a
--- /dev/null
+++ b/llvm/test/CodeGen/X86/vpextrq-fold.ll
@@ -0,0 +1,68 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mcpu=emeraldrapids -O3 %s -o - | FileCheck %s
+
+; Test that extractelement from a spilled v2i64 vector lane 1 is folded into a
+; direct memory load from the spill slot at offset 8, instead of reloading the
+; full vector and extracting with vpextrq.
+
+declare <2 x i64> @llvm.masked.load.v2i64.p0(ptr, i32 immarg, <2 x i1>, <2 x i64>)
+declare void @clobber()
+declare i64 @llvm.fshl.i64(i64, i64, i64)
+
+define void @repro(ptr %src, ptr %dst, i64 %sh) {
+; CHECK-LABEL: repro:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: pushq %r15
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: pushq %r14
+; CHECK-NEXT: .cfi_def_cfa_offset 24
+; CHECK-NEXT: pushq %rbx
+; CHECK-NEXT: .cfi_def_cfa_offset 32
+; CHECK-NEXT: subq $16, %rsp
+; CHECK-NEXT: .cfi_def_cfa_offset 48
+; CHECK-NEXT: .cfi_offset %rbx, -32
+; CHECK-NEXT: .cfi_offset %r14, -24
+; CHECK-NEXT: .cfi_offset %r15, -16
+; CHECK-NEXT: movq %rdx, %rbx
+; CHECK-NEXT: movq %rsi, %r14
+; CHECK-NEXT: movl $3, %eax
+; CHECK-NEXT: kmovd %eax, %k1
+; CHECK-NEXT: vmovdqu64 (%rdi), %xmm0 {%k1} {z}
+; CHECK-NEXT: vmovdqa %xmm0, (%rsp) # 16-byte Spill
+; CHECK-NEXT: vmovq %xmm0, %r15
+; CHECK-NEXT: vmovq %xmm0, (%rsi)
+; CHECK-NEXT: callq clobber at PLT
+; CHECK-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; CHECK-NEXT: movl %ebx, %ecx
+; CHECK-NEXT: shldq %cl, %r15, %rax
+; CHECK-NEXT: movq %rax, 8(%r14)
+; CHECK-NEXT: addq $16, %rsp
+; CHECK-NEXT: .cfi_def_cfa_offset 32
+; CHECK-NEXT: popq %rbx
+; CHECK-NEXT: .cfi_def_cfa_offset 24
+; CHECK-NEXT: popq %r14
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: popq %r15
+; CHECK-NEXT: .cfi_def_cfa_offset 8
+; CHECK-NEXT: retq
+entry:
+ %mask32 = bitcast i32 3 to <32 x i1>
+ %mask = shufflevector <32 x i1> %mask32, <32 x i1> poison, <2 x i32> <i32 0, i32 1>
+
+ %vec = call <2 x i64> @llvm.masked.load.v2i64.p0(
+ ptr align 8 %src,
+ i32 8,
+ <2 x i1> %mask,
+ <2 x i64> zeroinitializer)
+
+ %lo = extractelement <2 x i64> %vec, i64 0
+ store i64 %lo, ptr %dst, align 8
+
+ call void @clobber()
+
+ %hi = extractelement <2 x i64> %vec, i64 1
+ %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
+ %out = getelementptr i8, ptr %dst, i64 8
+ store i64 %r, ptr %out, align 8
+ ret void
+}
>From 7b61b02c9eadd57a7a2232c0a838de01138c63e7 Mon Sep 17 00:00:00 2001
From: TianYe <939808194 at qq.com>
Date: Sun, 14 Jun 2026 13:00:49 +0800
Subject: [PATCH 2/5] [X86] reduce test case
---
llvm/lib/Target/X86/X86InstrInfo.cpp | 2 -
.../test/CodeGen/X86/stack-folding-vpextrq.ll | 25 +++++++
llvm/test/CodeGen/X86/vpextrq-fold.ll | 68 -------------------
3 files changed, 25 insertions(+), 70 deletions(-)
create mode 100644 llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
delete mode 100644 llvm/test/CodeGen/X86/vpextrq-fold.ll
diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index 4f076afb615d8..fc46bb3041276 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -7444,9 +7444,7 @@ MachineInstr *X86InstrInfo::foldMemoryOperandCustom(
InsertPt, MI))
return NewMI;
break;
- case X86::VPEXTRQrri:
case X86::VPEXTRQZrri:
- case X86::PEXTRQrri:
// Fold: extractelt(v2i64 vector, 1) where vector is a spilled stack slot.
// Instead of reloading the full vector and extracting with vpextrq, load
// the upper 8 bytes directly from the spill slot at offset 8.
diff --git a/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll b/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
new file mode 100644
index 0000000000000..095e3793e56a9
--- /dev/null
+++ b/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
@@ -0,0 +1,25 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -O3 -mtriple=x86_64-unknown-unknown -mcpu=emeraldrapids < %s | FileCheck %s
+
+; Stack reload folding test for vpextrq.
+; By including a function call with sideeffects we can force a spill of the
+; vector register and check that the reload is correctly folded into a direct
+; memory load instead of vpextrq.
+
+declare void @clobber()
+
+define i64 @stack_fold_vpextrq(<2 x i64> %a0) {
+; CHECK-LABEL: stack_fold_vpextrq:
+; CHECK: # %bb.0:
+; CHECK-NEXT: subq $24, %rsp
+; CHECK-NEXT: .cfi_def_cfa_offset 32
+; CHECK-NEXT: vmovaps %xmm0, (%rsp) # 16-byte Spill
+; CHECK-NEXT: callq clobber at PLT
+; CHECK-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; CHECK-NEXT: addq $24, %rsp
+; CHECK-NEXT: .cfi_def_cfa_offset 8
+; CHECK-NEXT: retq
+ call void @clobber()
+ %1 = extractelement <2 x i64> %a0, i64 1
+ ret i64 %1
+}
diff --git a/llvm/test/CodeGen/X86/vpextrq-fold.ll b/llvm/test/CodeGen/X86/vpextrq-fold.ll
deleted file mode 100644
index 197aebb18cf5a..0000000000000
--- a/llvm/test/CodeGen/X86/vpextrq-fold.ll
+++ /dev/null
@@ -1,68 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mcpu=emeraldrapids -O3 %s -o - | FileCheck %s
-
-; Test that extractelement from a spilled v2i64 vector lane 1 is folded into a
-; direct memory load from the spill slot at offset 8, instead of reloading the
-; full vector and extracting with vpextrq.
-
-declare <2 x i64> @llvm.masked.load.v2i64.p0(ptr, i32 immarg, <2 x i1>, <2 x i64>)
-declare void @clobber()
-declare i64 @llvm.fshl.i64(i64, i64, i64)
-
-define void @repro(ptr %src, ptr %dst, i64 %sh) {
-; CHECK-LABEL: repro:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: pushq %r15
-; CHECK-NEXT: .cfi_def_cfa_offset 16
-; CHECK-NEXT: pushq %r14
-; CHECK-NEXT: .cfi_def_cfa_offset 24
-; CHECK-NEXT: pushq %rbx
-; CHECK-NEXT: .cfi_def_cfa_offset 32
-; CHECK-NEXT: subq $16, %rsp
-; CHECK-NEXT: .cfi_def_cfa_offset 48
-; CHECK-NEXT: .cfi_offset %rbx, -32
-; CHECK-NEXT: .cfi_offset %r14, -24
-; CHECK-NEXT: .cfi_offset %r15, -16
-; CHECK-NEXT: movq %rdx, %rbx
-; CHECK-NEXT: movq %rsi, %r14
-; CHECK-NEXT: movl $3, %eax
-; CHECK-NEXT: kmovd %eax, %k1
-; CHECK-NEXT: vmovdqu64 (%rdi), %xmm0 {%k1} {z}
-; CHECK-NEXT: vmovdqa %xmm0, (%rsp) # 16-byte Spill
-; CHECK-NEXT: vmovq %xmm0, %r15
-; CHECK-NEXT: vmovq %xmm0, (%rsi)
-; CHECK-NEXT: callq clobber at PLT
-; CHECK-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; CHECK-NEXT: movl %ebx, %ecx
-; CHECK-NEXT: shldq %cl, %r15, %rax
-; CHECK-NEXT: movq %rax, 8(%r14)
-; CHECK-NEXT: addq $16, %rsp
-; CHECK-NEXT: .cfi_def_cfa_offset 32
-; CHECK-NEXT: popq %rbx
-; CHECK-NEXT: .cfi_def_cfa_offset 24
-; CHECK-NEXT: popq %r14
-; CHECK-NEXT: .cfi_def_cfa_offset 16
-; CHECK-NEXT: popq %r15
-; CHECK-NEXT: .cfi_def_cfa_offset 8
-; CHECK-NEXT: retq
-entry:
- %mask32 = bitcast i32 3 to <32 x i1>
- %mask = shufflevector <32 x i1> %mask32, <32 x i1> poison, <2 x i32> <i32 0, i32 1>
-
- %vec = call <2 x i64> @llvm.masked.load.v2i64.p0(
- ptr align 8 %src,
- i32 8,
- <2 x i1> %mask,
- <2 x i64> zeroinitializer)
-
- %lo = extractelement <2 x i64> %vec, i64 0
- store i64 %lo, ptr %dst, align 8
-
- call void @clobber()
-
- %hi = extractelement <2 x i64> %vec, i64 1
- %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
- %out = getelementptr i8, ptr %dst, i64 8
- store i64 %r, ptr %out, align 8
- ret void
-}
>From b43209e4c3dcab5f709b5e22153245e14e6782cc Mon Sep 17 00:00:00 2001
From: TianYe <939808194 at qq.com>
Date: Sun, 14 Jun 2026 13:46:17 +0800
Subject: [PATCH 3/5] format code
---
llvm/lib/Target/X86/X86InstrInfo.cpp | 3 +--
1 file changed, 1 insertion(+), 2 deletions(-)
diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index fc46bb3041276..a9604108f3b27 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -7448,8 +7448,7 @@ MachineInstr *X86InstrInfo::foldMemoryOperandCustom(
// Fold: extractelt(v2i64 vector, 1) where vector is a spilled stack slot.
// Instead of reloading the full vector and extracting with vpextrq, load
// the upper 8 bytes directly from the spill slot at offset 8.
- if (OpNum == 1 && MI.getOperand(2).getImm() == 1 &&
- Alignment >= Align(8)) {
+ if (OpNum == 1 && MI.getOperand(2).getImm() == 1 && Alignment >= Align(8)) {
MachineInstrBuilder MIB = BuildMI(*InsertPt->getParent(), InsertPt,
MI.getDebugLoc(), get(X86::MOV64rm));
MIB.add(MI.getOperand(0)); // dst register
>From dc3d0d42e7c0d4beb0489acb9bbcdb8c5c32910f Mon Sep 17 00:00:00 2001
From: TianYe <939808194 at qq.com>
Date: Thu, 2 Jul 2026 01:38:37 +0800
Subject: [PATCH 4/5] Add instructions and cases
---
llvm/lib/Target/X86/X86InstrInfo.cpp | 2 +
llvm/test/CodeGen/X86/vpextrq-fold.ll | 282 ++++++++++++++++++++++++++
2 files changed, 284 insertions(+)
create mode 100644 llvm/test/CodeGen/X86/vpextrq-fold.ll
diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index a9604108f3b27..21526af60d657 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -7444,6 +7444,8 @@ MachineInstr *X86InstrInfo::foldMemoryOperandCustom(
InsertPt, MI))
return NewMI;
break;
+ case X86::PEXTRQrri:
+ case X86::VPEXTRQrri:
case X86::VPEXTRQZrri:
// Fold: extractelt(v2i64 vector, 1) where vector is a spilled stack slot.
// Instead of reloading the full vector and extracting with vpextrq, load
diff --git a/llvm/test/CodeGen/X86/vpextrq-fold.ll b/llvm/test/CodeGen/X86/vpextrq-fold.ll
new file mode 100644
index 0000000000000..a08ccddc8af84
--- /dev/null
+++ b/llvm/test/CodeGen/X86/vpextrq-fold.ll
@@ -0,0 +1,282 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+sse4.1 -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,SSE41
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+avx -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,AVX
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+avx512f -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,AVX512
+
+declare <2 x i64> @llvm.masked.load.v2i64.p0(ptr, i32 immarg, <2 x i1>, <2 x i64>)
+declare void @clobber()
+declare i64 @llvm.fshl.i64(i64, i64, i64)
+
+; For SSE4.1 and AVX-512F, use masked load which creates a vector that must be
+; spilled across the clobber call, then extract lane 1 which should fold to a
+; direct load from the spill slot at offset 8.
+define void @repro_masked(ptr %src, ptr %dst, i64 %sh) {
+; SSE41-LABEL: repro_masked:
+; SSE41: # %bb.0: # %entry
+; SSE41-NEXT: pushq %r15
+; SSE41-NEXT: .cfi_def_cfa_offset 16
+; SSE41-NEXT: pushq %r14
+; SSE41-NEXT: .cfi_def_cfa_offset 24
+; SSE41-NEXT: pushq %rbx
+; SSE41-NEXT: .cfi_def_cfa_offset 32
+; SSE41-NEXT: subq $16, %rsp
+; SSE41-NEXT: .cfi_def_cfa_offset 48
+; SSE41-NEXT: .cfi_offset %rbx, -32
+; SSE41-NEXT: .cfi_offset %r14, -24
+; SSE41-NEXT: .cfi_offset %r15, -16
+; SSE41-NEXT: movq %rdx, %r14
+; SSE41-NEXT: movq %rsi, %rbx
+; SSE41-NEXT: movb $3, %al
+; SSE41-NEXT: pxor %xmm0, %xmm0
+; SSE41-NEXT: xorl %ecx, %ecx
+; SSE41-NEXT: testb %cl, %cl
+; SSE41-NEXT: jne .LBB0_2
+; SSE41-NEXT: # %bb.1: # %cond.load
+; SSE41-NEXT: movq {{.*#+}} xmm0 = mem[0],zero
+; SSE41-NEXT: .LBB0_2: # %else
+; SSE41-NEXT: testb $2, %al
+; SSE41-NEXT: je .LBB0_4
+; SSE41-NEXT: # %bb.3: # %cond.load1
+; SSE41-NEXT: pinsrq $1, 8(%rdi), %xmm0
+; SSE41-NEXT: .LBB0_4: # %else2
+; SSE41-NEXT: movdqa %xmm0, (%rsp) # 16-byte Spill
+; SSE41-NEXT: movq %xmm0, %r15
+; SSE41-NEXT: movq %xmm0, (%rbx)
+; SSE41-NEXT: callq clobber at PLT
+; SSE41-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; SSE41-NEXT: movl %r14d, %ecx
+; SSE41-NEXT: shldq %cl, %r15, %rax
+; SSE41-NEXT: movq %rax, 8(%rbx)
+; SSE41-NEXT: addq $16, %rsp
+; SSE41-NEXT: .cfi_def_cfa_offset 32
+; SSE41-NEXT: popq %rbx
+; SSE41-NEXT: .cfi_def_cfa_offset 24
+; SSE41-NEXT: popq %r14
+; SSE41-NEXT: .cfi_def_cfa_offset 16
+; SSE41-NEXT: popq %r15
+; SSE41-NEXT: .cfi_def_cfa_offset 8
+; SSE41-NEXT: retq
+;
+; AVX-LABEL: repro_masked:
+; AVX: # %bb.0: # %entry
+; AVX-NEXT: pushq %r15
+; AVX-NEXT: .cfi_def_cfa_offset 16
+; AVX-NEXT: pushq %r14
+; AVX-NEXT: .cfi_def_cfa_offset 24
+; AVX-NEXT: pushq %r12
+; AVX-NEXT: .cfi_def_cfa_offset 32
+; AVX-NEXT: pushq %rbx
+; AVX-NEXT: .cfi_def_cfa_offset 40
+; AVX-NEXT: pushq %rax
+; AVX-NEXT: .cfi_def_cfa_offset 48
+; AVX-NEXT: .cfi_offset %rbx, -40
+; AVX-NEXT: .cfi_offset %r12, -32
+; AVX-NEXT: .cfi_offset %r14, -24
+; AVX-NEXT: .cfi_offset %r15, -16
+; AVX-NEXT: movq %rdx, %rbx
+; AVX-NEXT: movq %rsi, %r14
+; AVX-NEXT: movq 8(%rdi), %r15
+; AVX-NEXT: vmovdqu (%rdi), %xmm0
+; AVX-NEXT: vmovq %xmm0, %r12
+; AVX-NEXT: vmovq %xmm0, (%rsi)
+; AVX-NEXT: callq clobber at PLT
+; AVX-NEXT: movl %ebx, %ecx
+; AVX-NEXT: shldq %cl, %r12, %r15
+; AVX-NEXT: movq %r15, 8(%r14)
+; AVX-NEXT: addq $8, %rsp
+; AVX-NEXT: .cfi_def_cfa_offset 40
+; AVX-NEXT: popq %rbx
+; AVX-NEXT: .cfi_def_cfa_offset 32
+; AVX-NEXT: popq %r12
+; AVX-NEXT: .cfi_def_cfa_offset 24
+; AVX-NEXT: popq %r14
+; AVX-NEXT: .cfi_def_cfa_offset 16
+; AVX-NEXT: popq %r15
+; AVX-NEXT: .cfi_def_cfa_offset 8
+; AVX-NEXT: retq
+;
+; AVX512-LABEL: repro_masked:
+; AVX512: # %bb.0: # %entry
+; AVX512-NEXT: pushq %r15
+; AVX512-NEXT: .cfi_def_cfa_offset 16
+; AVX512-NEXT: pushq %r14
+; AVX512-NEXT: .cfi_def_cfa_offset 24
+; AVX512-NEXT: pushq %rbx
+; AVX512-NEXT: .cfi_def_cfa_offset 32
+; AVX512-NEXT: subq $64, %rsp
+; AVX512-NEXT: .cfi_def_cfa_offset 96
+; AVX512-NEXT: .cfi_offset %rbx, -32
+; AVX512-NEXT: .cfi_offset %r14, -24
+; AVX512-NEXT: .cfi_offset %r15, -16
+; AVX512-NEXT: movq %rdx, %rbx
+; AVX512-NEXT: movq %rsi, %r14
+; AVX512-NEXT: movw $3, %ax
+; AVX512-NEXT: kmovw %eax, %k0
+; AVX512-NEXT: kshiftlw $14, %k0, %k0
+; AVX512-NEXT: kshiftrw $14, %k0, %k1
+; AVX512-NEXT: vmovdqu64 (%rdi), %zmm0 {%k1} {z}
+; AVX512-NEXT: vmovdqu64 %zmm0, (%rsp) # 64-byte Spill
+; AVX512-NEXT: vmovq %xmm0, %r15
+; AVX512-NEXT: vmovq %xmm0, (%rsi)
+; AVX512-NEXT: vzeroupper
+; AVX512-NEXT: callq clobber at PLT
+; AVX512-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; AVX512-NEXT: movl %ebx, %ecx
+; AVX512-NEXT: shldq %cl, %r15, %rax
+; AVX512-NEXT: movq %rax, 8(%r14)
+; AVX512-NEXT: addq $64, %rsp
+; AVX512-NEXT: .cfi_def_cfa_offset 32
+; AVX512-NEXT: popq %rbx
+; AVX512-NEXT: .cfi_def_cfa_offset 24
+; AVX512-NEXT: popq %r14
+; AVX512-NEXT: .cfi_def_cfa_offset 16
+; AVX512-NEXT: popq %r15
+; AVX512-NEXT: .cfi_def_cfa_offset 8
+; AVX512-NEXT: retq
+entry:
+ %mask32 = bitcast i32 3 to <32 x i1>
+ %mask = shufflevector <32 x i1> %mask32, <32 x i1> poison, <2 x i32> <i32 0, i32 1>
+
+ %vec = call <2 x i64> @llvm.masked.load.v2i64.p0(
+ ptr align 8 %src,
+ i32 8,
+ <2 x i1> %mask,
+ <2 x i64> zeroinitializer)
+
+ %lo = extractelement <2 x i64> %vec, i64 0
+ store i64 %lo, ptr %dst, align 8
+
+ call void @clobber()
+
+ %hi = extractelement <2 x i64> %vec, i64 1
+ %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
+ %out = getelementptr i8, ptr %dst, i64 8
+ store i64 %r, ptr %out, align 8
+ ret void
+}
+
+; For AVX, use ALU on loaded vectors to create a vector that can't be
+; scalarized, forcing a spill across the clobber call.
+define void @repro_avx(ptr %src, ptr %dst, ptr %src2, i64 %sh) {
+; SSE41-LABEL: repro_avx:
+; SSE41: # %bb.0:
+; SSE41-NEXT: pushq %r15
+; SSE41-NEXT: .cfi_def_cfa_offset 16
+; SSE41-NEXT: pushq %r14
+; SSE41-NEXT: .cfi_def_cfa_offset 24
+; SSE41-NEXT: pushq %rbx
+; SSE41-NEXT: .cfi_def_cfa_offset 32
+; SSE41-NEXT: subq $16, %rsp
+; SSE41-NEXT: .cfi_def_cfa_offset 48
+; SSE41-NEXT: .cfi_offset %rbx, -32
+; SSE41-NEXT: .cfi_offset %r14, -24
+; SSE41-NEXT: .cfi_offset %r15, -16
+; SSE41-NEXT: movq %rcx, %rbx
+; SSE41-NEXT: movq %rsi, %r14
+; SSE41-NEXT: movdqa (%rdi), %xmm0
+; SSE41-NEXT: paddq (%rdx), %xmm0
+; SSE41-NEXT: movdqa %xmm0, (%rsp) # 16-byte Spill
+; SSE41-NEXT: movq %xmm0, %r15
+; SSE41-NEXT: movq %xmm0, (%rsi)
+; SSE41-NEXT: callq clobber at PLT
+; SSE41-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; SSE41-NEXT: movl %ebx, %ecx
+; SSE41-NEXT: shldq %cl, %r15, %rax
+; SSE41-NEXT: movq %rax, 8(%r14)
+; SSE41-NEXT: addq $16, %rsp
+; SSE41-NEXT: .cfi_def_cfa_offset 32
+; SSE41-NEXT: popq %rbx
+; SSE41-NEXT: .cfi_def_cfa_offset 24
+; SSE41-NEXT: popq %r14
+; SSE41-NEXT: .cfi_def_cfa_offset 16
+; SSE41-NEXT: popq %r15
+; SSE41-NEXT: .cfi_def_cfa_offset 8
+; SSE41-NEXT: retq
+;
+; AVX-LABEL: repro_avx:
+; AVX: # %bb.0:
+; AVX-NEXT: pushq %r15
+; AVX-NEXT: .cfi_def_cfa_offset 16
+; AVX-NEXT: pushq %r14
+; AVX-NEXT: .cfi_def_cfa_offset 24
+; AVX-NEXT: pushq %rbx
+; AVX-NEXT: .cfi_def_cfa_offset 32
+; AVX-NEXT: subq $16, %rsp
+; AVX-NEXT: .cfi_def_cfa_offset 48
+; AVX-NEXT: .cfi_offset %rbx, -32
+; AVX-NEXT: .cfi_offset %r14, -24
+; AVX-NEXT: .cfi_offset %r15, -16
+; AVX-NEXT: movq %rcx, %rbx
+; AVX-NEXT: movq %rsi, %r14
+; AVX-NEXT: vmovdqa (%rdi), %xmm0
+; AVX-NEXT: vpaddq (%rdx), %xmm0, %xmm0
+; AVX-NEXT: vmovdqa %xmm0, (%rsp) # 16-byte Spill
+; AVX-NEXT: vmovq %xmm0, %r15
+; AVX-NEXT: vmovq %xmm0, (%rsi)
+; AVX-NEXT: callq clobber at PLT
+; AVX-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; AVX-NEXT: movl %ebx, %ecx
+; AVX-NEXT: shldq %cl, %r15, %rax
+; AVX-NEXT: movq %rax, 8(%r14)
+; AVX-NEXT: addq $16, %rsp
+; AVX-NEXT: .cfi_def_cfa_offset 32
+; AVX-NEXT: popq %rbx
+; AVX-NEXT: .cfi_def_cfa_offset 24
+; AVX-NEXT: popq %r14
+; AVX-NEXT: .cfi_def_cfa_offset 16
+; AVX-NEXT: popq %r15
+; AVX-NEXT: .cfi_def_cfa_offset 8
+; AVX-NEXT: retq
+;
+; AVX512-LABEL: repro_avx:
+; AVX512: # %bb.0:
+; AVX512-NEXT: pushq %r15
+; AVX512-NEXT: .cfi_def_cfa_offset 16
+; AVX512-NEXT: pushq %r14
+; AVX512-NEXT: .cfi_def_cfa_offset 24
+; AVX512-NEXT: pushq %rbx
+; AVX512-NEXT: .cfi_def_cfa_offset 32
+; AVX512-NEXT: subq $16, %rsp
+; AVX512-NEXT: .cfi_def_cfa_offset 48
+; AVX512-NEXT: .cfi_offset %rbx, -32
+; AVX512-NEXT: .cfi_offset %r14, -24
+; AVX512-NEXT: .cfi_offset %r15, -16
+; AVX512-NEXT: movq %rcx, %rbx
+; AVX512-NEXT: movq %rsi, %r14
+; AVX512-NEXT: vmovdqa (%rdi), %xmm0
+; AVX512-NEXT: vpaddq (%rdx), %xmm0, %xmm0
+; AVX512-NEXT: vmovdqa %xmm0, (%rsp) # 16-byte Spill
+; AVX512-NEXT: vmovq %xmm0, %r15
+; AVX512-NEXT: vmovq %xmm0, (%rsi)
+; AVX512-NEXT: callq clobber at PLT
+; AVX512-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; AVX512-NEXT: movl %ebx, %ecx
+; AVX512-NEXT: shldq %cl, %r15, %rax
+; AVX512-NEXT: movq %rax, 8(%r14)
+; AVX512-NEXT: addq $16, %rsp
+; AVX512-NEXT: .cfi_def_cfa_offset 32
+; AVX512-NEXT: popq %rbx
+; AVX512-NEXT: .cfi_def_cfa_offset 24
+; AVX512-NEXT: popq %r14
+; AVX512-NEXT: .cfi_def_cfa_offset 16
+; AVX512-NEXT: popq %r15
+; AVX512-NEXT: .cfi_def_cfa_offset 8
+; AVX512-NEXT: retq
+
+ %v1 = load <2 x i64>, ptr %src, align 16
+ %v2 = load <2 x i64>, ptr %src2, align 16
+ %vec = add <2 x i64> %v1, %v2
+
+ %lo = extractelement <2 x i64> %vec, i64 0
+ store i64 %lo, ptr %dst, align 8
+
+ call void @clobber()
+
+ %hi = extractelement <2 x i64> %vec, i64 1
+ %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
+ %out = getelementptr i8, ptr %dst, i64 8
+ store i64 %r, ptr %out, align 8
+ ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}
>From a26e17c0b78fd5788029400f0336617dd6ee739b Mon Sep 17 00:00:00 2001
From: TianYe <939808194 at qq.com>
Date: Thu, 2 Jul 2026 01:48:38 +0800
Subject: [PATCH 5/5] Clean up vpextrq fold tests
---
.../test/CodeGen/X86/stack-folding-vpextrq.ll | 48 ++-
llvm/test/CodeGen/X86/vpextrq-fold.ll | 282 ------------------
2 files changed, 37 insertions(+), 293 deletions(-)
delete mode 100644 llvm/test/CodeGen/X86/vpextrq-fold.ll
diff --git a/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll b/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
index 095e3793e56a9..8a81355e25a02 100644
--- a/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
+++ b/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
@@ -1,5 +1,7 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -O3 -mtriple=x86_64-unknown-unknown -mcpu=emeraldrapids < %s | FileCheck %s
+; RUN: llc -O3 -mtriple=x86_64-unknown-unknown -mattr=+sse4.1 < %s | FileCheck %s --check-prefixes=CHECK,SSE41
+; RUN: llc -O3 -mtriple=x86_64-unknown-unknown -mattr=+avx < %s | FileCheck %s --check-prefixes=CHECK,AVX
+; RUN: llc -O3 -mtriple=x86_64-unknown-unknown -mattr=+avx512f < %s | FileCheck %s --check-prefixes=CHECK,AVX512
; Stack reload folding test for vpextrq.
; By including a function call with sideeffects we can force a spill of the
@@ -9,17 +11,41 @@
declare void @clobber()
define i64 @stack_fold_vpextrq(<2 x i64> %a0) {
-; CHECK-LABEL: stack_fold_vpextrq:
-; CHECK: # %bb.0:
-; CHECK-NEXT: subq $24, %rsp
-; CHECK-NEXT: .cfi_def_cfa_offset 32
-; CHECK-NEXT: vmovaps %xmm0, (%rsp) # 16-byte Spill
-; CHECK-NEXT: callq clobber at PLT
-; CHECK-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; CHECK-NEXT: addq $24, %rsp
-; CHECK-NEXT: .cfi_def_cfa_offset 8
-; CHECK-NEXT: retq
+; SSE41-LABEL: stack_fold_vpextrq:
+; SSE41: # %bb.0:
+; SSE41-NEXT: subq $24, %rsp
+; SSE41-NEXT: .cfi_def_cfa_offset 32
+; SSE41-NEXT: movaps %xmm0, (%rsp) # 16-byte Spill
+; SSE41-NEXT: callq clobber at PLT
+; SSE41-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; SSE41-NEXT: addq $24, %rsp
+; SSE41-NEXT: .cfi_def_cfa_offset 8
+; SSE41-NEXT: retq
+;
+; AVX-LABEL: stack_fold_vpextrq:
+; AVX: # %bb.0:
+; AVX-NEXT: subq $24, %rsp
+; AVX-NEXT: .cfi_def_cfa_offset 32
+; AVX-NEXT: vmovaps %xmm0, (%rsp) # 16-byte Spill
+; AVX-NEXT: callq clobber at PLT
+; AVX-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; AVX-NEXT: addq $24, %rsp
+; AVX-NEXT: .cfi_def_cfa_offset 8
+; AVX-NEXT: retq
+;
+; AVX512-LABEL: stack_fold_vpextrq:
+; AVX512: # %bb.0:
+; AVX512-NEXT: subq $24, %rsp
+; AVX512-NEXT: .cfi_def_cfa_offset 32
+; AVX512-NEXT: vmovaps %xmm0, (%rsp) # 16-byte Spill
+; AVX512-NEXT: callq clobber at PLT
+; AVX512-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; AVX512-NEXT: addq $24, %rsp
+; AVX512-NEXT: .cfi_def_cfa_offset 8
+; AVX512-NEXT: retq
call void @clobber()
%1 = extractelement <2 x i64> %a0, i64 1
ret i64 %1
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}
diff --git a/llvm/test/CodeGen/X86/vpextrq-fold.ll b/llvm/test/CodeGen/X86/vpextrq-fold.ll
deleted file mode 100644
index a08ccddc8af84..0000000000000
--- a/llvm/test/CodeGen/X86/vpextrq-fold.ll
+++ /dev/null
@@ -1,282 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+sse4.1 -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,SSE41
-; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+avx -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,AVX
-; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+avx512f -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,AVX512
-
-declare <2 x i64> @llvm.masked.load.v2i64.p0(ptr, i32 immarg, <2 x i1>, <2 x i64>)
-declare void @clobber()
-declare i64 @llvm.fshl.i64(i64, i64, i64)
-
-; For SSE4.1 and AVX-512F, use masked load which creates a vector that must be
-; spilled across the clobber call, then extract lane 1 which should fold to a
-; direct load from the spill slot at offset 8.
-define void @repro_masked(ptr %src, ptr %dst, i64 %sh) {
-; SSE41-LABEL: repro_masked:
-; SSE41: # %bb.0: # %entry
-; SSE41-NEXT: pushq %r15
-; SSE41-NEXT: .cfi_def_cfa_offset 16
-; SSE41-NEXT: pushq %r14
-; SSE41-NEXT: .cfi_def_cfa_offset 24
-; SSE41-NEXT: pushq %rbx
-; SSE41-NEXT: .cfi_def_cfa_offset 32
-; SSE41-NEXT: subq $16, %rsp
-; SSE41-NEXT: .cfi_def_cfa_offset 48
-; SSE41-NEXT: .cfi_offset %rbx, -32
-; SSE41-NEXT: .cfi_offset %r14, -24
-; SSE41-NEXT: .cfi_offset %r15, -16
-; SSE41-NEXT: movq %rdx, %r14
-; SSE41-NEXT: movq %rsi, %rbx
-; SSE41-NEXT: movb $3, %al
-; SSE41-NEXT: pxor %xmm0, %xmm0
-; SSE41-NEXT: xorl %ecx, %ecx
-; SSE41-NEXT: testb %cl, %cl
-; SSE41-NEXT: jne .LBB0_2
-; SSE41-NEXT: # %bb.1: # %cond.load
-; SSE41-NEXT: movq {{.*#+}} xmm0 = mem[0],zero
-; SSE41-NEXT: .LBB0_2: # %else
-; SSE41-NEXT: testb $2, %al
-; SSE41-NEXT: je .LBB0_4
-; SSE41-NEXT: # %bb.3: # %cond.load1
-; SSE41-NEXT: pinsrq $1, 8(%rdi), %xmm0
-; SSE41-NEXT: .LBB0_4: # %else2
-; SSE41-NEXT: movdqa %xmm0, (%rsp) # 16-byte Spill
-; SSE41-NEXT: movq %xmm0, %r15
-; SSE41-NEXT: movq %xmm0, (%rbx)
-; SSE41-NEXT: callq clobber at PLT
-; SSE41-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; SSE41-NEXT: movl %r14d, %ecx
-; SSE41-NEXT: shldq %cl, %r15, %rax
-; SSE41-NEXT: movq %rax, 8(%rbx)
-; SSE41-NEXT: addq $16, %rsp
-; SSE41-NEXT: .cfi_def_cfa_offset 32
-; SSE41-NEXT: popq %rbx
-; SSE41-NEXT: .cfi_def_cfa_offset 24
-; SSE41-NEXT: popq %r14
-; SSE41-NEXT: .cfi_def_cfa_offset 16
-; SSE41-NEXT: popq %r15
-; SSE41-NEXT: .cfi_def_cfa_offset 8
-; SSE41-NEXT: retq
-;
-; AVX-LABEL: repro_masked:
-; AVX: # %bb.0: # %entry
-; AVX-NEXT: pushq %r15
-; AVX-NEXT: .cfi_def_cfa_offset 16
-; AVX-NEXT: pushq %r14
-; AVX-NEXT: .cfi_def_cfa_offset 24
-; AVX-NEXT: pushq %r12
-; AVX-NEXT: .cfi_def_cfa_offset 32
-; AVX-NEXT: pushq %rbx
-; AVX-NEXT: .cfi_def_cfa_offset 40
-; AVX-NEXT: pushq %rax
-; AVX-NEXT: .cfi_def_cfa_offset 48
-; AVX-NEXT: .cfi_offset %rbx, -40
-; AVX-NEXT: .cfi_offset %r12, -32
-; AVX-NEXT: .cfi_offset %r14, -24
-; AVX-NEXT: .cfi_offset %r15, -16
-; AVX-NEXT: movq %rdx, %rbx
-; AVX-NEXT: movq %rsi, %r14
-; AVX-NEXT: movq 8(%rdi), %r15
-; AVX-NEXT: vmovdqu (%rdi), %xmm0
-; AVX-NEXT: vmovq %xmm0, %r12
-; AVX-NEXT: vmovq %xmm0, (%rsi)
-; AVX-NEXT: callq clobber at PLT
-; AVX-NEXT: movl %ebx, %ecx
-; AVX-NEXT: shldq %cl, %r12, %r15
-; AVX-NEXT: movq %r15, 8(%r14)
-; AVX-NEXT: addq $8, %rsp
-; AVX-NEXT: .cfi_def_cfa_offset 40
-; AVX-NEXT: popq %rbx
-; AVX-NEXT: .cfi_def_cfa_offset 32
-; AVX-NEXT: popq %r12
-; AVX-NEXT: .cfi_def_cfa_offset 24
-; AVX-NEXT: popq %r14
-; AVX-NEXT: .cfi_def_cfa_offset 16
-; AVX-NEXT: popq %r15
-; AVX-NEXT: .cfi_def_cfa_offset 8
-; AVX-NEXT: retq
-;
-; AVX512-LABEL: repro_masked:
-; AVX512: # %bb.0: # %entry
-; AVX512-NEXT: pushq %r15
-; AVX512-NEXT: .cfi_def_cfa_offset 16
-; AVX512-NEXT: pushq %r14
-; AVX512-NEXT: .cfi_def_cfa_offset 24
-; AVX512-NEXT: pushq %rbx
-; AVX512-NEXT: .cfi_def_cfa_offset 32
-; AVX512-NEXT: subq $64, %rsp
-; AVX512-NEXT: .cfi_def_cfa_offset 96
-; AVX512-NEXT: .cfi_offset %rbx, -32
-; AVX512-NEXT: .cfi_offset %r14, -24
-; AVX512-NEXT: .cfi_offset %r15, -16
-; AVX512-NEXT: movq %rdx, %rbx
-; AVX512-NEXT: movq %rsi, %r14
-; AVX512-NEXT: movw $3, %ax
-; AVX512-NEXT: kmovw %eax, %k0
-; AVX512-NEXT: kshiftlw $14, %k0, %k0
-; AVX512-NEXT: kshiftrw $14, %k0, %k1
-; AVX512-NEXT: vmovdqu64 (%rdi), %zmm0 {%k1} {z}
-; AVX512-NEXT: vmovdqu64 %zmm0, (%rsp) # 64-byte Spill
-; AVX512-NEXT: vmovq %xmm0, %r15
-; AVX512-NEXT: vmovq %xmm0, (%rsi)
-; AVX512-NEXT: vzeroupper
-; AVX512-NEXT: callq clobber at PLT
-; AVX512-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; AVX512-NEXT: movl %ebx, %ecx
-; AVX512-NEXT: shldq %cl, %r15, %rax
-; AVX512-NEXT: movq %rax, 8(%r14)
-; AVX512-NEXT: addq $64, %rsp
-; AVX512-NEXT: .cfi_def_cfa_offset 32
-; AVX512-NEXT: popq %rbx
-; AVX512-NEXT: .cfi_def_cfa_offset 24
-; AVX512-NEXT: popq %r14
-; AVX512-NEXT: .cfi_def_cfa_offset 16
-; AVX512-NEXT: popq %r15
-; AVX512-NEXT: .cfi_def_cfa_offset 8
-; AVX512-NEXT: retq
-entry:
- %mask32 = bitcast i32 3 to <32 x i1>
- %mask = shufflevector <32 x i1> %mask32, <32 x i1> poison, <2 x i32> <i32 0, i32 1>
-
- %vec = call <2 x i64> @llvm.masked.load.v2i64.p0(
- ptr align 8 %src,
- i32 8,
- <2 x i1> %mask,
- <2 x i64> zeroinitializer)
-
- %lo = extractelement <2 x i64> %vec, i64 0
- store i64 %lo, ptr %dst, align 8
-
- call void @clobber()
-
- %hi = extractelement <2 x i64> %vec, i64 1
- %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
- %out = getelementptr i8, ptr %dst, i64 8
- store i64 %r, ptr %out, align 8
- ret void
-}
-
-; For AVX, use ALU on loaded vectors to create a vector that can't be
-; scalarized, forcing a spill across the clobber call.
-define void @repro_avx(ptr %src, ptr %dst, ptr %src2, i64 %sh) {
-; SSE41-LABEL: repro_avx:
-; SSE41: # %bb.0:
-; SSE41-NEXT: pushq %r15
-; SSE41-NEXT: .cfi_def_cfa_offset 16
-; SSE41-NEXT: pushq %r14
-; SSE41-NEXT: .cfi_def_cfa_offset 24
-; SSE41-NEXT: pushq %rbx
-; SSE41-NEXT: .cfi_def_cfa_offset 32
-; SSE41-NEXT: subq $16, %rsp
-; SSE41-NEXT: .cfi_def_cfa_offset 48
-; SSE41-NEXT: .cfi_offset %rbx, -32
-; SSE41-NEXT: .cfi_offset %r14, -24
-; SSE41-NEXT: .cfi_offset %r15, -16
-; SSE41-NEXT: movq %rcx, %rbx
-; SSE41-NEXT: movq %rsi, %r14
-; SSE41-NEXT: movdqa (%rdi), %xmm0
-; SSE41-NEXT: paddq (%rdx), %xmm0
-; SSE41-NEXT: movdqa %xmm0, (%rsp) # 16-byte Spill
-; SSE41-NEXT: movq %xmm0, %r15
-; SSE41-NEXT: movq %xmm0, (%rsi)
-; SSE41-NEXT: callq clobber at PLT
-; SSE41-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; SSE41-NEXT: movl %ebx, %ecx
-; SSE41-NEXT: shldq %cl, %r15, %rax
-; SSE41-NEXT: movq %rax, 8(%r14)
-; SSE41-NEXT: addq $16, %rsp
-; SSE41-NEXT: .cfi_def_cfa_offset 32
-; SSE41-NEXT: popq %rbx
-; SSE41-NEXT: .cfi_def_cfa_offset 24
-; SSE41-NEXT: popq %r14
-; SSE41-NEXT: .cfi_def_cfa_offset 16
-; SSE41-NEXT: popq %r15
-; SSE41-NEXT: .cfi_def_cfa_offset 8
-; SSE41-NEXT: retq
-;
-; AVX-LABEL: repro_avx:
-; AVX: # %bb.0:
-; AVX-NEXT: pushq %r15
-; AVX-NEXT: .cfi_def_cfa_offset 16
-; AVX-NEXT: pushq %r14
-; AVX-NEXT: .cfi_def_cfa_offset 24
-; AVX-NEXT: pushq %rbx
-; AVX-NEXT: .cfi_def_cfa_offset 32
-; AVX-NEXT: subq $16, %rsp
-; AVX-NEXT: .cfi_def_cfa_offset 48
-; AVX-NEXT: .cfi_offset %rbx, -32
-; AVX-NEXT: .cfi_offset %r14, -24
-; AVX-NEXT: .cfi_offset %r15, -16
-; AVX-NEXT: movq %rcx, %rbx
-; AVX-NEXT: movq %rsi, %r14
-; AVX-NEXT: vmovdqa (%rdi), %xmm0
-; AVX-NEXT: vpaddq (%rdx), %xmm0, %xmm0
-; AVX-NEXT: vmovdqa %xmm0, (%rsp) # 16-byte Spill
-; AVX-NEXT: vmovq %xmm0, %r15
-; AVX-NEXT: vmovq %xmm0, (%rsi)
-; AVX-NEXT: callq clobber at PLT
-; AVX-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; AVX-NEXT: movl %ebx, %ecx
-; AVX-NEXT: shldq %cl, %r15, %rax
-; AVX-NEXT: movq %rax, 8(%r14)
-; AVX-NEXT: addq $16, %rsp
-; AVX-NEXT: .cfi_def_cfa_offset 32
-; AVX-NEXT: popq %rbx
-; AVX-NEXT: .cfi_def_cfa_offset 24
-; AVX-NEXT: popq %r14
-; AVX-NEXT: .cfi_def_cfa_offset 16
-; AVX-NEXT: popq %r15
-; AVX-NEXT: .cfi_def_cfa_offset 8
-; AVX-NEXT: retq
-;
-; AVX512-LABEL: repro_avx:
-; AVX512: # %bb.0:
-; AVX512-NEXT: pushq %r15
-; AVX512-NEXT: .cfi_def_cfa_offset 16
-; AVX512-NEXT: pushq %r14
-; AVX512-NEXT: .cfi_def_cfa_offset 24
-; AVX512-NEXT: pushq %rbx
-; AVX512-NEXT: .cfi_def_cfa_offset 32
-; AVX512-NEXT: subq $16, %rsp
-; AVX512-NEXT: .cfi_def_cfa_offset 48
-; AVX512-NEXT: .cfi_offset %rbx, -32
-; AVX512-NEXT: .cfi_offset %r14, -24
-; AVX512-NEXT: .cfi_offset %r15, -16
-; AVX512-NEXT: movq %rcx, %rbx
-; AVX512-NEXT: movq %rsi, %r14
-; AVX512-NEXT: vmovdqa (%rdi), %xmm0
-; AVX512-NEXT: vpaddq (%rdx), %xmm0, %xmm0
-; AVX512-NEXT: vmovdqa %xmm0, (%rsp) # 16-byte Spill
-; AVX512-NEXT: vmovq %xmm0, %r15
-; AVX512-NEXT: vmovq %xmm0, (%rsi)
-; AVX512-NEXT: callq clobber at PLT
-; AVX512-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; AVX512-NEXT: movl %ebx, %ecx
-; AVX512-NEXT: shldq %cl, %r15, %rax
-; AVX512-NEXT: movq %rax, 8(%r14)
-; AVX512-NEXT: addq $16, %rsp
-; AVX512-NEXT: .cfi_def_cfa_offset 32
-; AVX512-NEXT: popq %rbx
-; AVX512-NEXT: .cfi_def_cfa_offset 24
-; AVX512-NEXT: popq %r14
-; AVX512-NEXT: .cfi_def_cfa_offset 16
-; AVX512-NEXT: popq %r15
-; AVX512-NEXT: .cfi_def_cfa_offset 8
-; AVX512-NEXT: retq
-
- %v1 = load <2 x i64>, ptr %src, align 16
- %v2 = load <2 x i64>, ptr %src2, align 16
- %vec = add <2 x i64> %v1, %v2
-
- %lo = extractelement <2 x i64> %vec, i64 0
- store i64 %lo, ptr %dst, align 8
-
- call void @clobber()
-
- %hi = extractelement <2 x i64> %vec, i64 1
- %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
- %out = getelementptr i8, ptr %dst, i64 8
- store i64 %r, ptr %out, align 8
- ret void
-}
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; CHECK: {{.*}}
More information about the llvm-commits
mailing list