[llvm] [X86] Fold vpextrq $1 from a spilled vector into a direct memory load (PR #203339)

Ye Tian via llvm-commits llvm-commits at lists.llvm.org
Wed Jul 1 10:49:01 PDT 2026


https://github.com/TianYe717 updated https://github.com/llvm/llvm-project/pull/203339

>From b7c0d08358a780440c30fd042ff0540e1c586026 Mon Sep 17 00:00:00 2001
From: TianYe <939808194 at qq.com>
Date: Fri, 12 Jun 2026 00:41:15 +0800
Subject: [PATCH 1/5] [X86] Fold vpextrq $1 from a spilled vector into a direct
 memory load

Instead of reloading the full vector and extracting lane 1 with vpextrq,
load the upper 8 bytes directly from the spill slot at offset 8.
---
 llvm/lib/Target/X86/X86InstrInfo.cpp  | 15 ++++++
 llvm/test/CodeGen/X86/vpextrq-fold.ll | 68 +++++++++++++++++++++++++++
 2 files changed, 83 insertions(+)
 create mode 100644 llvm/test/CodeGen/X86/vpextrq-fold.ll

diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index 6c86a3cbece2e..4f076afb615d8 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -7444,6 +7444,21 @@ MachineInstr *X86InstrInfo::foldMemoryOperandCustom(
                        InsertPt, MI))
       return NewMI;
     break;
+  case X86::VPEXTRQrri:
+  case X86::VPEXTRQZrri:
+  case X86::PEXTRQrri:
+    // Fold: extractelt(v2i64 vector, 1) where vector is a spilled stack slot.
+    // Instead of reloading the full vector and extracting with vpextrq, load
+    // the upper 8 bytes directly from the spill slot at offset 8.
+    if (OpNum == 1 && MI.getOperand(2).getImm() == 1 &&
+        Alignment >= Align(8)) {
+      MachineInstrBuilder MIB = BuildMI(*InsertPt->getParent(), InsertPt,
+                                        MI.getDebugLoc(), get(X86::MOV64rm));
+      MIB.add(MI.getOperand(0)); // dst register
+      addOperands(MIB, MOs, 8);  // offset 8 for upper lane
+      return MIB;
+    }
+    break;
   }
 
   return nullptr;
diff --git a/llvm/test/CodeGen/X86/vpextrq-fold.ll b/llvm/test/CodeGen/X86/vpextrq-fold.ll
new file mode 100644
index 0000000000000..197aebb18cf5a
--- /dev/null
+++ b/llvm/test/CodeGen/X86/vpextrq-fold.ll
@@ -0,0 +1,68 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mcpu=emeraldrapids -O3 %s -o - | FileCheck %s
+
+; Test that extractelement from a spilled v2i64 vector lane 1 is folded into a
+; direct memory load from the spill slot at offset 8, instead of reloading the
+; full vector and extracting with vpextrq.
+
+declare <2 x i64> @llvm.masked.load.v2i64.p0(ptr, i32 immarg, <2 x i1>, <2 x i64>)
+declare void @clobber()
+declare i64 @llvm.fshl.i64(i64, i64, i64)
+
+define void @repro(ptr %src, ptr %dst, i64 %sh) {
+; CHECK-LABEL: repro:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    pushq %r15
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    pushq %r14
+; CHECK-NEXT:    .cfi_def_cfa_offset 24
+; CHECK-NEXT:    pushq %rbx
+; CHECK-NEXT:    .cfi_def_cfa_offset 32
+; CHECK-NEXT:    subq $16, %rsp
+; CHECK-NEXT:    .cfi_def_cfa_offset 48
+; CHECK-NEXT:    .cfi_offset %rbx, -32
+; CHECK-NEXT:    .cfi_offset %r14, -24
+; CHECK-NEXT:    .cfi_offset %r15, -16
+; CHECK-NEXT:    movq %rdx, %rbx
+; CHECK-NEXT:    movq %rsi, %r14
+; CHECK-NEXT:    movl $3, %eax
+; CHECK-NEXT:    kmovd %eax, %k1
+; CHECK-NEXT:    vmovdqu64 (%rdi), %xmm0 {%k1} {z}
+; CHECK-NEXT:    vmovdqa %xmm0, (%rsp) # 16-byte Spill
+; CHECK-NEXT:    vmovq %xmm0, %r15
+; CHECK-NEXT:    vmovq %xmm0, (%rsi)
+; CHECK-NEXT:    callq clobber at PLT
+; CHECK-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; CHECK-NEXT:    movl %ebx, %ecx
+; CHECK-NEXT:    shldq %cl, %r15, %rax
+; CHECK-NEXT:    movq %rax, 8(%r14)
+; CHECK-NEXT:    addq $16, %rsp
+; CHECK-NEXT:    .cfi_def_cfa_offset 32
+; CHECK-NEXT:    popq %rbx
+; CHECK-NEXT:    .cfi_def_cfa_offset 24
+; CHECK-NEXT:    popq %r14
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    popq %r15
+; CHECK-NEXT:    .cfi_def_cfa_offset 8
+; CHECK-NEXT:    retq
+entry:
+  %mask32 = bitcast i32 3 to <32 x i1>
+  %mask = shufflevector <32 x i1> %mask32, <32 x i1> poison, <2 x i32> <i32 0, i32 1>
+
+  %vec = call <2 x i64> @llvm.masked.load.v2i64.p0(
+      ptr align 8 %src,
+      i32 8,
+      <2 x i1> %mask,
+      <2 x i64> zeroinitializer)
+
+  %lo = extractelement <2 x i64> %vec, i64 0
+  store i64 %lo, ptr %dst, align 8
+
+  call void @clobber()
+
+  %hi = extractelement <2 x i64> %vec, i64 1
+  %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
+  %out = getelementptr i8, ptr %dst, i64 8
+  store i64 %r, ptr %out, align 8
+  ret void
+}

>From 7b61b02c9eadd57a7a2232c0a838de01138c63e7 Mon Sep 17 00:00:00 2001
From: TianYe <939808194 at qq.com>
Date: Sun, 14 Jun 2026 13:00:49 +0800
Subject: [PATCH 2/5] [X86] reduce test case

---
 llvm/lib/Target/X86/X86InstrInfo.cpp          |  2 -
 .../test/CodeGen/X86/stack-folding-vpextrq.ll | 25 +++++++
 llvm/test/CodeGen/X86/vpextrq-fold.ll         | 68 -------------------
 3 files changed, 25 insertions(+), 70 deletions(-)
 create mode 100644 llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
 delete mode 100644 llvm/test/CodeGen/X86/vpextrq-fold.ll

diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index 4f076afb615d8..fc46bb3041276 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -7444,9 +7444,7 @@ MachineInstr *X86InstrInfo::foldMemoryOperandCustom(
                        InsertPt, MI))
       return NewMI;
     break;
-  case X86::VPEXTRQrri:
   case X86::VPEXTRQZrri:
-  case X86::PEXTRQrri:
     // Fold: extractelt(v2i64 vector, 1) where vector is a spilled stack slot.
     // Instead of reloading the full vector and extracting with vpextrq, load
     // the upper 8 bytes directly from the spill slot at offset 8.
diff --git a/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll b/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
new file mode 100644
index 0000000000000..095e3793e56a9
--- /dev/null
+++ b/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
@@ -0,0 +1,25 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -O3 -mtriple=x86_64-unknown-unknown -mcpu=emeraldrapids < %s | FileCheck %s
+
+; Stack reload folding test for vpextrq.
+; By including a function call with sideeffects we can force a spill of the
+; vector register and check that the reload is correctly folded into a direct
+; memory load instead of vpextrq.
+
+declare void @clobber()
+
+define i64 @stack_fold_vpextrq(<2 x i64> %a0) {
+; CHECK-LABEL: stack_fold_vpextrq:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    subq $24, %rsp
+; CHECK-NEXT:    .cfi_def_cfa_offset 32
+; CHECK-NEXT:    vmovaps %xmm0, (%rsp) # 16-byte Spill
+; CHECK-NEXT:    callq clobber at PLT
+; CHECK-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; CHECK-NEXT:    addq $24, %rsp
+; CHECK-NEXT:    .cfi_def_cfa_offset 8
+; CHECK-NEXT:    retq
+  call void @clobber()
+  %1 = extractelement <2 x i64> %a0, i64 1
+  ret i64 %1
+}
diff --git a/llvm/test/CodeGen/X86/vpextrq-fold.ll b/llvm/test/CodeGen/X86/vpextrq-fold.ll
deleted file mode 100644
index 197aebb18cf5a..0000000000000
--- a/llvm/test/CodeGen/X86/vpextrq-fold.ll
+++ /dev/null
@@ -1,68 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mcpu=emeraldrapids -O3 %s -o - | FileCheck %s
-
-; Test that extractelement from a spilled v2i64 vector lane 1 is folded into a
-; direct memory load from the spill slot at offset 8, instead of reloading the
-; full vector and extracting with vpextrq.
-
-declare <2 x i64> @llvm.masked.load.v2i64.p0(ptr, i32 immarg, <2 x i1>, <2 x i64>)
-declare void @clobber()
-declare i64 @llvm.fshl.i64(i64, i64, i64)
-
-define void @repro(ptr %src, ptr %dst, i64 %sh) {
-; CHECK-LABEL: repro:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    pushq %r15
-; CHECK-NEXT:    .cfi_def_cfa_offset 16
-; CHECK-NEXT:    pushq %r14
-; CHECK-NEXT:    .cfi_def_cfa_offset 24
-; CHECK-NEXT:    pushq %rbx
-; CHECK-NEXT:    .cfi_def_cfa_offset 32
-; CHECK-NEXT:    subq $16, %rsp
-; CHECK-NEXT:    .cfi_def_cfa_offset 48
-; CHECK-NEXT:    .cfi_offset %rbx, -32
-; CHECK-NEXT:    .cfi_offset %r14, -24
-; CHECK-NEXT:    .cfi_offset %r15, -16
-; CHECK-NEXT:    movq %rdx, %rbx
-; CHECK-NEXT:    movq %rsi, %r14
-; CHECK-NEXT:    movl $3, %eax
-; CHECK-NEXT:    kmovd %eax, %k1
-; CHECK-NEXT:    vmovdqu64 (%rdi), %xmm0 {%k1} {z}
-; CHECK-NEXT:    vmovdqa %xmm0, (%rsp) # 16-byte Spill
-; CHECK-NEXT:    vmovq %xmm0, %r15
-; CHECK-NEXT:    vmovq %xmm0, (%rsi)
-; CHECK-NEXT:    callq clobber at PLT
-; CHECK-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; CHECK-NEXT:    movl %ebx, %ecx
-; CHECK-NEXT:    shldq %cl, %r15, %rax
-; CHECK-NEXT:    movq %rax, 8(%r14)
-; CHECK-NEXT:    addq $16, %rsp
-; CHECK-NEXT:    .cfi_def_cfa_offset 32
-; CHECK-NEXT:    popq %rbx
-; CHECK-NEXT:    .cfi_def_cfa_offset 24
-; CHECK-NEXT:    popq %r14
-; CHECK-NEXT:    .cfi_def_cfa_offset 16
-; CHECK-NEXT:    popq %r15
-; CHECK-NEXT:    .cfi_def_cfa_offset 8
-; CHECK-NEXT:    retq
-entry:
-  %mask32 = bitcast i32 3 to <32 x i1>
-  %mask = shufflevector <32 x i1> %mask32, <32 x i1> poison, <2 x i32> <i32 0, i32 1>
-
-  %vec = call <2 x i64> @llvm.masked.load.v2i64.p0(
-      ptr align 8 %src,
-      i32 8,
-      <2 x i1> %mask,
-      <2 x i64> zeroinitializer)
-
-  %lo = extractelement <2 x i64> %vec, i64 0
-  store i64 %lo, ptr %dst, align 8
-
-  call void @clobber()
-
-  %hi = extractelement <2 x i64> %vec, i64 1
-  %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
-  %out = getelementptr i8, ptr %dst, i64 8
-  store i64 %r, ptr %out, align 8
-  ret void
-}

>From b43209e4c3dcab5f709b5e22153245e14e6782cc Mon Sep 17 00:00:00 2001
From: TianYe <939808194 at qq.com>
Date: Sun, 14 Jun 2026 13:46:17 +0800
Subject: [PATCH 3/5] format code

---
 llvm/lib/Target/X86/X86InstrInfo.cpp | 3 +--
 1 file changed, 1 insertion(+), 2 deletions(-)

diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index fc46bb3041276..a9604108f3b27 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -7448,8 +7448,7 @@ MachineInstr *X86InstrInfo::foldMemoryOperandCustom(
     // Fold: extractelt(v2i64 vector, 1) where vector is a spilled stack slot.
     // Instead of reloading the full vector and extracting with vpextrq, load
     // the upper 8 bytes directly from the spill slot at offset 8.
-    if (OpNum == 1 && MI.getOperand(2).getImm() == 1 &&
-        Alignment >= Align(8)) {
+    if (OpNum == 1 && MI.getOperand(2).getImm() == 1 && Alignment >= Align(8)) {
       MachineInstrBuilder MIB = BuildMI(*InsertPt->getParent(), InsertPt,
                                         MI.getDebugLoc(), get(X86::MOV64rm));
       MIB.add(MI.getOperand(0)); // dst register

>From dc3d0d42e7c0d4beb0489acb9bbcdb8c5c32910f Mon Sep 17 00:00:00 2001
From: TianYe <939808194 at qq.com>
Date: Thu, 2 Jul 2026 01:38:37 +0800
Subject: [PATCH 4/5] Add instructions and cases

---
 llvm/lib/Target/X86/X86InstrInfo.cpp  |   2 +
 llvm/test/CodeGen/X86/vpextrq-fold.ll | 282 ++++++++++++++++++++++++++
 2 files changed, 284 insertions(+)
 create mode 100644 llvm/test/CodeGen/X86/vpextrq-fold.ll

diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index a9604108f3b27..21526af60d657 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -7444,6 +7444,8 @@ MachineInstr *X86InstrInfo::foldMemoryOperandCustom(
                        InsertPt, MI))
       return NewMI;
     break;
+  case X86::PEXTRQrri:
+  case X86::VPEXTRQrri:
   case X86::VPEXTRQZrri:
     // Fold: extractelt(v2i64 vector, 1) where vector is a spilled stack slot.
     // Instead of reloading the full vector and extracting with vpextrq, load
diff --git a/llvm/test/CodeGen/X86/vpextrq-fold.ll b/llvm/test/CodeGen/X86/vpextrq-fold.ll
new file mode 100644
index 0000000000000..a08ccddc8af84
--- /dev/null
+++ b/llvm/test/CodeGen/X86/vpextrq-fold.ll
@@ -0,0 +1,282 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+sse4.1 -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,SSE41
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+avx -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,AVX
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+avx512f -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,AVX512
+
+declare <2 x i64> @llvm.masked.load.v2i64.p0(ptr, i32 immarg, <2 x i1>, <2 x i64>)
+declare void @clobber()
+declare i64 @llvm.fshl.i64(i64, i64, i64)
+
+; For SSE4.1 and AVX-512F, use masked load which creates a vector that must be
+; spilled across the clobber call, then extract lane 1 which should fold to a
+; direct load from the spill slot at offset 8.
+define void @repro_masked(ptr %src, ptr %dst, i64 %sh) {
+; SSE41-LABEL: repro_masked:
+; SSE41:       # %bb.0: # %entry
+; SSE41-NEXT:    pushq %r15
+; SSE41-NEXT:    .cfi_def_cfa_offset 16
+; SSE41-NEXT:    pushq %r14
+; SSE41-NEXT:    .cfi_def_cfa_offset 24
+; SSE41-NEXT:    pushq %rbx
+; SSE41-NEXT:    .cfi_def_cfa_offset 32
+; SSE41-NEXT:    subq $16, %rsp
+; SSE41-NEXT:    .cfi_def_cfa_offset 48
+; SSE41-NEXT:    .cfi_offset %rbx, -32
+; SSE41-NEXT:    .cfi_offset %r14, -24
+; SSE41-NEXT:    .cfi_offset %r15, -16
+; SSE41-NEXT:    movq %rdx, %r14
+; SSE41-NEXT:    movq %rsi, %rbx
+; SSE41-NEXT:    movb $3, %al
+; SSE41-NEXT:    pxor %xmm0, %xmm0
+; SSE41-NEXT:    xorl %ecx, %ecx
+; SSE41-NEXT:    testb %cl, %cl
+; SSE41-NEXT:    jne .LBB0_2
+; SSE41-NEXT:  # %bb.1: # %cond.load
+; SSE41-NEXT:    movq {{.*#+}} xmm0 = mem[0],zero
+; SSE41-NEXT:  .LBB0_2: # %else
+; SSE41-NEXT:    testb $2, %al
+; SSE41-NEXT:    je .LBB0_4
+; SSE41-NEXT:  # %bb.3: # %cond.load1
+; SSE41-NEXT:    pinsrq $1, 8(%rdi), %xmm0
+; SSE41-NEXT:  .LBB0_4: # %else2
+; SSE41-NEXT:    movdqa %xmm0, (%rsp) # 16-byte Spill
+; SSE41-NEXT:    movq %xmm0, %r15
+; SSE41-NEXT:    movq %xmm0, (%rbx)
+; SSE41-NEXT:    callq clobber at PLT
+; SSE41-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; SSE41-NEXT:    movl %r14d, %ecx
+; SSE41-NEXT:    shldq %cl, %r15, %rax
+; SSE41-NEXT:    movq %rax, 8(%rbx)
+; SSE41-NEXT:    addq $16, %rsp
+; SSE41-NEXT:    .cfi_def_cfa_offset 32
+; SSE41-NEXT:    popq %rbx
+; SSE41-NEXT:    .cfi_def_cfa_offset 24
+; SSE41-NEXT:    popq %r14
+; SSE41-NEXT:    .cfi_def_cfa_offset 16
+; SSE41-NEXT:    popq %r15
+; SSE41-NEXT:    .cfi_def_cfa_offset 8
+; SSE41-NEXT:    retq
+;
+; AVX-LABEL: repro_masked:
+; AVX:       # %bb.0: # %entry
+; AVX-NEXT:    pushq %r15
+; AVX-NEXT:    .cfi_def_cfa_offset 16
+; AVX-NEXT:    pushq %r14
+; AVX-NEXT:    .cfi_def_cfa_offset 24
+; AVX-NEXT:    pushq %r12
+; AVX-NEXT:    .cfi_def_cfa_offset 32
+; AVX-NEXT:    pushq %rbx
+; AVX-NEXT:    .cfi_def_cfa_offset 40
+; AVX-NEXT:    pushq %rax
+; AVX-NEXT:    .cfi_def_cfa_offset 48
+; AVX-NEXT:    .cfi_offset %rbx, -40
+; AVX-NEXT:    .cfi_offset %r12, -32
+; AVX-NEXT:    .cfi_offset %r14, -24
+; AVX-NEXT:    .cfi_offset %r15, -16
+; AVX-NEXT:    movq %rdx, %rbx
+; AVX-NEXT:    movq %rsi, %r14
+; AVX-NEXT:    movq 8(%rdi), %r15
+; AVX-NEXT:    vmovdqu (%rdi), %xmm0
+; AVX-NEXT:    vmovq %xmm0, %r12
+; AVX-NEXT:    vmovq %xmm0, (%rsi)
+; AVX-NEXT:    callq clobber at PLT
+; AVX-NEXT:    movl %ebx, %ecx
+; AVX-NEXT:    shldq %cl, %r12, %r15
+; AVX-NEXT:    movq %r15, 8(%r14)
+; AVX-NEXT:    addq $8, %rsp
+; AVX-NEXT:    .cfi_def_cfa_offset 40
+; AVX-NEXT:    popq %rbx
+; AVX-NEXT:    .cfi_def_cfa_offset 32
+; AVX-NEXT:    popq %r12
+; AVX-NEXT:    .cfi_def_cfa_offset 24
+; AVX-NEXT:    popq %r14
+; AVX-NEXT:    .cfi_def_cfa_offset 16
+; AVX-NEXT:    popq %r15
+; AVX-NEXT:    .cfi_def_cfa_offset 8
+; AVX-NEXT:    retq
+;
+; AVX512-LABEL: repro_masked:
+; AVX512:       # %bb.0: # %entry
+; AVX512-NEXT:    pushq %r15
+; AVX512-NEXT:    .cfi_def_cfa_offset 16
+; AVX512-NEXT:    pushq %r14
+; AVX512-NEXT:    .cfi_def_cfa_offset 24
+; AVX512-NEXT:    pushq %rbx
+; AVX512-NEXT:    .cfi_def_cfa_offset 32
+; AVX512-NEXT:    subq $64, %rsp
+; AVX512-NEXT:    .cfi_def_cfa_offset 96
+; AVX512-NEXT:    .cfi_offset %rbx, -32
+; AVX512-NEXT:    .cfi_offset %r14, -24
+; AVX512-NEXT:    .cfi_offset %r15, -16
+; AVX512-NEXT:    movq %rdx, %rbx
+; AVX512-NEXT:    movq %rsi, %r14
+; AVX512-NEXT:    movw $3, %ax
+; AVX512-NEXT:    kmovw %eax, %k0
+; AVX512-NEXT:    kshiftlw $14, %k0, %k0
+; AVX512-NEXT:    kshiftrw $14, %k0, %k1
+; AVX512-NEXT:    vmovdqu64 (%rdi), %zmm0 {%k1} {z}
+; AVX512-NEXT:    vmovdqu64 %zmm0, (%rsp) # 64-byte Spill
+; AVX512-NEXT:    vmovq %xmm0, %r15
+; AVX512-NEXT:    vmovq %xmm0, (%rsi)
+; AVX512-NEXT:    vzeroupper
+; AVX512-NEXT:    callq clobber at PLT
+; AVX512-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; AVX512-NEXT:    movl %ebx, %ecx
+; AVX512-NEXT:    shldq %cl, %r15, %rax
+; AVX512-NEXT:    movq %rax, 8(%r14)
+; AVX512-NEXT:    addq $64, %rsp
+; AVX512-NEXT:    .cfi_def_cfa_offset 32
+; AVX512-NEXT:    popq %rbx
+; AVX512-NEXT:    .cfi_def_cfa_offset 24
+; AVX512-NEXT:    popq %r14
+; AVX512-NEXT:    .cfi_def_cfa_offset 16
+; AVX512-NEXT:    popq %r15
+; AVX512-NEXT:    .cfi_def_cfa_offset 8
+; AVX512-NEXT:    retq
+entry:
+  %mask32 = bitcast i32 3 to <32 x i1>
+  %mask = shufflevector <32 x i1> %mask32, <32 x i1> poison, <2 x i32> <i32 0, i32 1>
+
+  %vec = call <2 x i64> @llvm.masked.load.v2i64.p0(
+      ptr align 8 %src,
+      i32 8,
+      <2 x i1> %mask,
+      <2 x i64> zeroinitializer)
+
+  %lo = extractelement <2 x i64> %vec, i64 0
+  store i64 %lo, ptr %dst, align 8
+
+  call void @clobber()
+
+  %hi = extractelement <2 x i64> %vec, i64 1
+  %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
+  %out = getelementptr i8, ptr %dst, i64 8
+  store i64 %r, ptr %out, align 8
+  ret void
+}
+
+; For AVX, use ALU on loaded vectors to create a vector that can't be
+; scalarized, forcing a spill across the clobber call.
+define void @repro_avx(ptr %src, ptr %dst, ptr %src2, i64 %sh) {
+; SSE41-LABEL: repro_avx:
+; SSE41:       # %bb.0:
+; SSE41-NEXT:    pushq %r15
+; SSE41-NEXT:    .cfi_def_cfa_offset 16
+; SSE41-NEXT:    pushq %r14
+; SSE41-NEXT:    .cfi_def_cfa_offset 24
+; SSE41-NEXT:    pushq %rbx
+; SSE41-NEXT:    .cfi_def_cfa_offset 32
+; SSE41-NEXT:    subq $16, %rsp
+; SSE41-NEXT:    .cfi_def_cfa_offset 48
+; SSE41-NEXT:    .cfi_offset %rbx, -32
+; SSE41-NEXT:    .cfi_offset %r14, -24
+; SSE41-NEXT:    .cfi_offset %r15, -16
+; SSE41-NEXT:    movq %rcx, %rbx
+; SSE41-NEXT:    movq %rsi, %r14
+; SSE41-NEXT:    movdqa (%rdi), %xmm0
+; SSE41-NEXT:    paddq (%rdx), %xmm0
+; SSE41-NEXT:    movdqa %xmm0, (%rsp) # 16-byte Spill
+; SSE41-NEXT:    movq %xmm0, %r15
+; SSE41-NEXT:    movq %xmm0, (%rsi)
+; SSE41-NEXT:    callq clobber at PLT
+; SSE41-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; SSE41-NEXT:    movl %ebx, %ecx
+; SSE41-NEXT:    shldq %cl, %r15, %rax
+; SSE41-NEXT:    movq %rax, 8(%r14)
+; SSE41-NEXT:    addq $16, %rsp
+; SSE41-NEXT:    .cfi_def_cfa_offset 32
+; SSE41-NEXT:    popq %rbx
+; SSE41-NEXT:    .cfi_def_cfa_offset 24
+; SSE41-NEXT:    popq %r14
+; SSE41-NEXT:    .cfi_def_cfa_offset 16
+; SSE41-NEXT:    popq %r15
+; SSE41-NEXT:    .cfi_def_cfa_offset 8
+; SSE41-NEXT:    retq
+;
+; AVX-LABEL: repro_avx:
+; AVX:       # %bb.0:
+; AVX-NEXT:    pushq %r15
+; AVX-NEXT:    .cfi_def_cfa_offset 16
+; AVX-NEXT:    pushq %r14
+; AVX-NEXT:    .cfi_def_cfa_offset 24
+; AVX-NEXT:    pushq %rbx
+; AVX-NEXT:    .cfi_def_cfa_offset 32
+; AVX-NEXT:    subq $16, %rsp
+; AVX-NEXT:    .cfi_def_cfa_offset 48
+; AVX-NEXT:    .cfi_offset %rbx, -32
+; AVX-NEXT:    .cfi_offset %r14, -24
+; AVX-NEXT:    .cfi_offset %r15, -16
+; AVX-NEXT:    movq %rcx, %rbx
+; AVX-NEXT:    movq %rsi, %r14
+; AVX-NEXT:    vmovdqa (%rdi), %xmm0
+; AVX-NEXT:    vpaddq (%rdx), %xmm0, %xmm0
+; AVX-NEXT:    vmovdqa %xmm0, (%rsp) # 16-byte Spill
+; AVX-NEXT:    vmovq %xmm0, %r15
+; AVX-NEXT:    vmovq %xmm0, (%rsi)
+; AVX-NEXT:    callq clobber at PLT
+; AVX-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; AVX-NEXT:    movl %ebx, %ecx
+; AVX-NEXT:    shldq %cl, %r15, %rax
+; AVX-NEXT:    movq %rax, 8(%r14)
+; AVX-NEXT:    addq $16, %rsp
+; AVX-NEXT:    .cfi_def_cfa_offset 32
+; AVX-NEXT:    popq %rbx
+; AVX-NEXT:    .cfi_def_cfa_offset 24
+; AVX-NEXT:    popq %r14
+; AVX-NEXT:    .cfi_def_cfa_offset 16
+; AVX-NEXT:    popq %r15
+; AVX-NEXT:    .cfi_def_cfa_offset 8
+; AVX-NEXT:    retq
+;
+; AVX512-LABEL: repro_avx:
+; AVX512:       # %bb.0:
+; AVX512-NEXT:    pushq %r15
+; AVX512-NEXT:    .cfi_def_cfa_offset 16
+; AVX512-NEXT:    pushq %r14
+; AVX512-NEXT:    .cfi_def_cfa_offset 24
+; AVX512-NEXT:    pushq %rbx
+; AVX512-NEXT:    .cfi_def_cfa_offset 32
+; AVX512-NEXT:    subq $16, %rsp
+; AVX512-NEXT:    .cfi_def_cfa_offset 48
+; AVX512-NEXT:    .cfi_offset %rbx, -32
+; AVX512-NEXT:    .cfi_offset %r14, -24
+; AVX512-NEXT:    .cfi_offset %r15, -16
+; AVX512-NEXT:    movq %rcx, %rbx
+; AVX512-NEXT:    movq %rsi, %r14
+; AVX512-NEXT:    vmovdqa (%rdi), %xmm0
+; AVX512-NEXT:    vpaddq (%rdx), %xmm0, %xmm0
+; AVX512-NEXT:    vmovdqa %xmm0, (%rsp) # 16-byte Spill
+; AVX512-NEXT:    vmovq %xmm0, %r15
+; AVX512-NEXT:    vmovq %xmm0, (%rsi)
+; AVX512-NEXT:    callq clobber at PLT
+; AVX512-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; AVX512-NEXT:    movl %ebx, %ecx
+; AVX512-NEXT:    shldq %cl, %r15, %rax
+; AVX512-NEXT:    movq %rax, 8(%r14)
+; AVX512-NEXT:    addq $16, %rsp
+; AVX512-NEXT:    .cfi_def_cfa_offset 32
+; AVX512-NEXT:    popq %rbx
+; AVX512-NEXT:    .cfi_def_cfa_offset 24
+; AVX512-NEXT:    popq %r14
+; AVX512-NEXT:    .cfi_def_cfa_offset 16
+; AVX512-NEXT:    popq %r15
+; AVX512-NEXT:    .cfi_def_cfa_offset 8
+; AVX512-NEXT:    retq
+
+  %v1 = load <2 x i64>, ptr %src, align 16
+  %v2 = load <2 x i64>, ptr %src2, align 16
+  %vec = add <2 x i64> %v1, %v2
+
+  %lo = extractelement <2 x i64> %vec, i64 0
+  store i64 %lo, ptr %dst, align 8
+
+  call void @clobber()
+
+  %hi = extractelement <2 x i64> %vec, i64 1
+  %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
+  %out = getelementptr i8, ptr %dst, i64 8
+  store i64 %r, ptr %out, align 8
+  ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}

>From a26e17c0b78fd5788029400f0336617dd6ee739b Mon Sep 17 00:00:00 2001
From: TianYe <939808194 at qq.com>
Date: Thu, 2 Jul 2026 01:48:38 +0800
Subject: [PATCH 5/5] Clean up vpextrq fold tests

---
 .../test/CodeGen/X86/stack-folding-vpextrq.ll |  48 ++-
 llvm/test/CodeGen/X86/vpextrq-fold.ll         | 282 ------------------
 2 files changed, 37 insertions(+), 293 deletions(-)
 delete mode 100644 llvm/test/CodeGen/X86/vpextrq-fold.ll

diff --git a/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll b/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
index 095e3793e56a9..8a81355e25a02 100644
--- a/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
+++ b/llvm/test/CodeGen/X86/stack-folding-vpextrq.ll
@@ -1,5 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -O3 -mtriple=x86_64-unknown-unknown -mcpu=emeraldrapids < %s | FileCheck %s
+; RUN: llc -O3 -mtriple=x86_64-unknown-unknown -mattr=+sse4.1 < %s | FileCheck %s --check-prefixes=CHECK,SSE41
+; RUN: llc -O3 -mtriple=x86_64-unknown-unknown -mattr=+avx < %s | FileCheck %s --check-prefixes=CHECK,AVX
+; RUN: llc -O3 -mtriple=x86_64-unknown-unknown -mattr=+avx512f < %s | FileCheck %s --check-prefixes=CHECK,AVX512
 
 ; Stack reload folding test for vpextrq.
 ; By including a function call with sideeffects we can force a spill of the
@@ -9,17 +11,41 @@
 declare void @clobber()
 
 define i64 @stack_fold_vpextrq(<2 x i64> %a0) {
-; CHECK-LABEL: stack_fold_vpextrq:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    subq $24, %rsp
-; CHECK-NEXT:    .cfi_def_cfa_offset 32
-; CHECK-NEXT:    vmovaps %xmm0, (%rsp) # 16-byte Spill
-; CHECK-NEXT:    callq clobber at PLT
-; CHECK-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; CHECK-NEXT:    addq $24, %rsp
-; CHECK-NEXT:    .cfi_def_cfa_offset 8
-; CHECK-NEXT:    retq
+; SSE41-LABEL: stack_fold_vpextrq:
+; SSE41:       # %bb.0:
+; SSE41-NEXT:    subq $24, %rsp
+; SSE41-NEXT:    .cfi_def_cfa_offset 32
+; SSE41-NEXT:    movaps %xmm0, (%rsp) # 16-byte Spill
+; SSE41-NEXT:    callq clobber at PLT
+; SSE41-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; SSE41-NEXT:    addq $24, %rsp
+; SSE41-NEXT:    .cfi_def_cfa_offset 8
+; SSE41-NEXT:    retq
+;
+; AVX-LABEL: stack_fold_vpextrq:
+; AVX:       # %bb.0:
+; AVX-NEXT:    subq $24, %rsp
+; AVX-NEXT:    .cfi_def_cfa_offset 32
+; AVX-NEXT:    vmovaps %xmm0, (%rsp) # 16-byte Spill
+; AVX-NEXT:    callq clobber at PLT
+; AVX-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; AVX-NEXT:    addq $24, %rsp
+; AVX-NEXT:    .cfi_def_cfa_offset 8
+; AVX-NEXT:    retq
+;
+; AVX512-LABEL: stack_fold_vpextrq:
+; AVX512:       # %bb.0:
+; AVX512-NEXT:    subq $24, %rsp
+; AVX512-NEXT:    .cfi_def_cfa_offset 32
+; AVX512-NEXT:    vmovaps %xmm0, (%rsp) # 16-byte Spill
+; AVX512-NEXT:    callq clobber at PLT
+; AVX512-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
+; AVX512-NEXT:    addq $24, %rsp
+; AVX512-NEXT:    .cfi_def_cfa_offset 8
+; AVX512-NEXT:    retq
   call void @clobber()
   %1 = extractelement <2 x i64> %a0, i64 1
   ret i64 %1
 }
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}
diff --git a/llvm/test/CodeGen/X86/vpextrq-fold.ll b/llvm/test/CodeGen/X86/vpextrq-fold.ll
deleted file mode 100644
index a08ccddc8af84..0000000000000
--- a/llvm/test/CodeGen/X86/vpextrq-fold.ll
+++ /dev/null
@@ -1,282 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+sse4.1 -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,SSE41
-; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+avx -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,AVX
-; RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=+avx512f -O3 %s -o - | FileCheck %s --check-prefixes=CHECK,AVX512
-
-declare <2 x i64> @llvm.masked.load.v2i64.p0(ptr, i32 immarg, <2 x i1>, <2 x i64>)
-declare void @clobber()
-declare i64 @llvm.fshl.i64(i64, i64, i64)
-
-; For SSE4.1 and AVX-512F, use masked load which creates a vector that must be
-; spilled across the clobber call, then extract lane 1 which should fold to a
-; direct load from the spill slot at offset 8.
-define void @repro_masked(ptr %src, ptr %dst, i64 %sh) {
-; SSE41-LABEL: repro_masked:
-; SSE41:       # %bb.0: # %entry
-; SSE41-NEXT:    pushq %r15
-; SSE41-NEXT:    .cfi_def_cfa_offset 16
-; SSE41-NEXT:    pushq %r14
-; SSE41-NEXT:    .cfi_def_cfa_offset 24
-; SSE41-NEXT:    pushq %rbx
-; SSE41-NEXT:    .cfi_def_cfa_offset 32
-; SSE41-NEXT:    subq $16, %rsp
-; SSE41-NEXT:    .cfi_def_cfa_offset 48
-; SSE41-NEXT:    .cfi_offset %rbx, -32
-; SSE41-NEXT:    .cfi_offset %r14, -24
-; SSE41-NEXT:    .cfi_offset %r15, -16
-; SSE41-NEXT:    movq %rdx, %r14
-; SSE41-NEXT:    movq %rsi, %rbx
-; SSE41-NEXT:    movb $3, %al
-; SSE41-NEXT:    pxor %xmm0, %xmm0
-; SSE41-NEXT:    xorl %ecx, %ecx
-; SSE41-NEXT:    testb %cl, %cl
-; SSE41-NEXT:    jne .LBB0_2
-; SSE41-NEXT:  # %bb.1: # %cond.load
-; SSE41-NEXT:    movq {{.*#+}} xmm0 = mem[0],zero
-; SSE41-NEXT:  .LBB0_2: # %else
-; SSE41-NEXT:    testb $2, %al
-; SSE41-NEXT:    je .LBB0_4
-; SSE41-NEXT:  # %bb.3: # %cond.load1
-; SSE41-NEXT:    pinsrq $1, 8(%rdi), %xmm0
-; SSE41-NEXT:  .LBB0_4: # %else2
-; SSE41-NEXT:    movdqa %xmm0, (%rsp) # 16-byte Spill
-; SSE41-NEXT:    movq %xmm0, %r15
-; SSE41-NEXT:    movq %xmm0, (%rbx)
-; SSE41-NEXT:    callq clobber at PLT
-; SSE41-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; SSE41-NEXT:    movl %r14d, %ecx
-; SSE41-NEXT:    shldq %cl, %r15, %rax
-; SSE41-NEXT:    movq %rax, 8(%rbx)
-; SSE41-NEXT:    addq $16, %rsp
-; SSE41-NEXT:    .cfi_def_cfa_offset 32
-; SSE41-NEXT:    popq %rbx
-; SSE41-NEXT:    .cfi_def_cfa_offset 24
-; SSE41-NEXT:    popq %r14
-; SSE41-NEXT:    .cfi_def_cfa_offset 16
-; SSE41-NEXT:    popq %r15
-; SSE41-NEXT:    .cfi_def_cfa_offset 8
-; SSE41-NEXT:    retq
-;
-; AVX-LABEL: repro_masked:
-; AVX:       # %bb.0: # %entry
-; AVX-NEXT:    pushq %r15
-; AVX-NEXT:    .cfi_def_cfa_offset 16
-; AVX-NEXT:    pushq %r14
-; AVX-NEXT:    .cfi_def_cfa_offset 24
-; AVX-NEXT:    pushq %r12
-; AVX-NEXT:    .cfi_def_cfa_offset 32
-; AVX-NEXT:    pushq %rbx
-; AVX-NEXT:    .cfi_def_cfa_offset 40
-; AVX-NEXT:    pushq %rax
-; AVX-NEXT:    .cfi_def_cfa_offset 48
-; AVX-NEXT:    .cfi_offset %rbx, -40
-; AVX-NEXT:    .cfi_offset %r12, -32
-; AVX-NEXT:    .cfi_offset %r14, -24
-; AVX-NEXT:    .cfi_offset %r15, -16
-; AVX-NEXT:    movq %rdx, %rbx
-; AVX-NEXT:    movq %rsi, %r14
-; AVX-NEXT:    movq 8(%rdi), %r15
-; AVX-NEXT:    vmovdqu (%rdi), %xmm0
-; AVX-NEXT:    vmovq %xmm0, %r12
-; AVX-NEXT:    vmovq %xmm0, (%rsi)
-; AVX-NEXT:    callq clobber at PLT
-; AVX-NEXT:    movl %ebx, %ecx
-; AVX-NEXT:    shldq %cl, %r12, %r15
-; AVX-NEXT:    movq %r15, 8(%r14)
-; AVX-NEXT:    addq $8, %rsp
-; AVX-NEXT:    .cfi_def_cfa_offset 40
-; AVX-NEXT:    popq %rbx
-; AVX-NEXT:    .cfi_def_cfa_offset 32
-; AVX-NEXT:    popq %r12
-; AVX-NEXT:    .cfi_def_cfa_offset 24
-; AVX-NEXT:    popq %r14
-; AVX-NEXT:    .cfi_def_cfa_offset 16
-; AVX-NEXT:    popq %r15
-; AVX-NEXT:    .cfi_def_cfa_offset 8
-; AVX-NEXT:    retq
-;
-; AVX512-LABEL: repro_masked:
-; AVX512:       # %bb.0: # %entry
-; AVX512-NEXT:    pushq %r15
-; AVX512-NEXT:    .cfi_def_cfa_offset 16
-; AVX512-NEXT:    pushq %r14
-; AVX512-NEXT:    .cfi_def_cfa_offset 24
-; AVX512-NEXT:    pushq %rbx
-; AVX512-NEXT:    .cfi_def_cfa_offset 32
-; AVX512-NEXT:    subq $64, %rsp
-; AVX512-NEXT:    .cfi_def_cfa_offset 96
-; AVX512-NEXT:    .cfi_offset %rbx, -32
-; AVX512-NEXT:    .cfi_offset %r14, -24
-; AVX512-NEXT:    .cfi_offset %r15, -16
-; AVX512-NEXT:    movq %rdx, %rbx
-; AVX512-NEXT:    movq %rsi, %r14
-; AVX512-NEXT:    movw $3, %ax
-; AVX512-NEXT:    kmovw %eax, %k0
-; AVX512-NEXT:    kshiftlw $14, %k0, %k0
-; AVX512-NEXT:    kshiftrw $14, %k0, %k1
-; AVX512-NEXT:    vmovdqu64 (%rdi), %zmm0 {%k1} {z}
-; AVX512-NEXT:    vmovdqu64 %zmm0, (%rsp) # 64-byte Spill
-; AVX512-NEXT:    vmovq %xmm0, %r15
-; AVX512-NEXT:    vmovq %xmm0, (%rsi)
-; AVX512-NEXT:    vzeroupper
-; AVX512-NEXT:    callq clobber at PLT
-; AVX512-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; AVX512-NEXT:    movl %ebx, %ecx
-; AVX512-NEXT:    shldq %cl, %r15, %rax
-; AVX512-NEXT:    movq %rax, 8(%r14)
-; AVX512-NEXT:    addq $64, %rsp
-; AVX512-NEXT:    .cfi_def_cfa_offset 32
-; AVX512-NEXT:    popq %rbx
-; AVX512-NEXT:    .cfi_def_cfa_offset 24
-; AVX512-NEXT:    popq %r14
-; AVX512-NEXT:    .cfi_def_cfa_offset 16
-; AVX512-NEXT:    popq %r15
-; AVX512-NEXT:    .cfi_def_cfa_offset 8
-; AVX512-NEXT:    retq
-entry:
-  %mask32 = bitcast i32 3 to <32 x i1>
-  %mask = shufflevector <32 x i1> %mask32, <32 x i1> poison, <2 x i32> <i32 0, i32 1>
-
-  %vec = call <2 x i64> @llvm.masked.load.v2i64.p0(
-      ptr align 8 %src,
-      i32 8,
-      <2 x i1> %mask,
-      <2 x i64> zeroinitializer)
-
-  %lo = extractelement <2 x i64> %vec, i64 0
-  store i64 %lo, ptr %dst, align 8
-
-  call void @clobber()
-
-  %hi = extractelement <2 x i64> %vec, i64 1
-  %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
-  %out = getelementptr i8, ptr %dst, i64 8
-  store i64 %r, ptr %out, align 8
-  ret void
-}
-
-; For AVX, use ALU on loaded vectors to create a vector that can't be
-; scalarized, forcing a spill across the clobber call.
-define void @repro_avx(ptr %src, ptr %dst, ptr %src2, i64 %sh) {
-; SSE41-LABEL: repro_avx:
-; SSE41:       # %bb.0:
-; SSE41-NEXT:    pushq %r15
-; SSE41-NEXT:    .cfi_def_cfa_offset 16
-; SSE41-NEXT:    pushq %r14
-; SSE41-NEXT:    .cfi_def_cfa_offset 24
-; SSE41-NEXT:    pushq %rbx
-; SSE41-NEXT:    .cfi_def_cfa_offset 32
-; SSE41-NEXT:    subq $16, %rsp
-; SSE41-NEXT:    .cfi_def_cfa_offset 48
-; SSE41-NEXT:    .cfi_offset %rbx, -32
-; SSE41-NEXT:    .cfi_offset %r14, -24
-; SSE41-NEXT:    .cfi_offset %r15, -16
-; SSE41-NEXT:    movq %rcx, %rbx
-; SSE41-NEXT:    movq %rsi, %r14
-; SSE41-NEXT:    movdqa (%rdi), %xmm0
-; SSE41-NEXT:    paddq (%rdx), %xmm0
-; SSE41-NEXT:    movdqa %xmm0, (%rsp) # 16-byte Spill
-; SSE41-NEXT:    movq %xmm0, %r15
-; SSE41-NEXT:    movq %xmm0, (%rsi)
-; SSE41-NEXT:    callq clobber at PLT
-; SSE41-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; SSE41-NEXT:    movl %ebx, %ecx
-; SSE41-NEXT:    shldq %cl, %r15, %rax
-; SSE41-NEXT:    movq %rax, 8(%r14)
-; SSE41-NEXT:    addq $16, %rsp
-; SSE41-NEXT:    .cfi_def_cfa_offset 32
-; SSE41-NEXT:    popq %rbx
-; SSE41-NEXT:    .cfi_def_cfa_offset 24
-; SSE41-NEXT:    popq %r14
-; SSE41-NEXT:    .cfi_def_cfa_offset 16
-; SSE41-NEXT:    popq %r15
-; SSE41-NEXT:    .cfi_def_cfa_offset 8
-; SSE41-NEXT:    retq
-;
-; AVX-LABEL: repro_avx:
-; AVX:       # %bb.0:
-; AVX-NEXT:    pushq %r15
-; AVX-NEXT:    .cfi_def_cfa_offset 16
-; AVX-NEXT:    pushq %r14
-; AVX-NEXT:    .cfi_def_cfa_offset 24
-; AVX-NEXT:    pushq %rbx
-; AVX-NEXT:    .cfi_def_cfa_offset 32
-; AVX-NEXT:    subq $16, %rsp
-; AVX-NEXT:    .cfi_def_cfa_offset 48
-; AVX-NEXT:    .cfi_offset %rbx, -32
-; AVX-NEXT:    .cfi_offset %r14, -24
-; AVX-NEXT:    .cfi_offset %r15, -16
-; AVX-NEXT:    movq %rcx, %rbx
-; AVX-NEXT:    movq %rsi, %r14
-; AVX-NEXT:    vmovdqa (%rdi), %xmm0
-; AVX-NEXT:    vpaddq (%rdx), %xmm0, %xmm0
-; AVX-NEXT:    vmovdqa %xmm0, (%rsp) # 16-byte Spill
-; AVX-NEXT:    vmovq %xmm0, %r15
-; AVX-NEXT:    vmovq %xmm0, (%rsi)
-; AVX-NEXT:    callq clobber at PLT
-; AVX-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; AVX-NEXT:    movl %ebx, %ecx
-; AVX-NEXT:    shldq %cl, %r15, %rax
-; AVX-NEXT:    movq %rax, 8(%r14)
-; AVX-NEXT:    addq $16, %rsp
-; AVX-NEXT:    .cfi_def_cfa_offset 32
-; AVX-NEXT:    popq %rbx
-; AVX-NEXT:    .cfi_def_cfa_offset 24
-; AVX-NEXT:    popq %r14
-; AVX-NEXT:    .cfi_def_cfa_offset 16
-; AVX-NEXT:    popq %r15
-; AVX-NEXT:    .cfi_def_cfa_offset 8
-; AVX-NEXT:    retq
-;
-; AVX512-LABEL: repro_avx:
-; AVX512:       # %bb.0:
-; AVX512-NEXT:    pushq %r15
-; AVX512-NEXT:    .cfi_def_cfa_offset 16
-; AVX512-NEXT:    pushq %r14
-; AVX512-NEXT:    .cfi_def_cfa_offset 24
-; AVX512-NEXT:    pushq %rbx
-; AVX512-NEXT:    .cfi_def_cfa_offset 32
-; AVX512-NEXT:    subq $16, %rsp
-; AVX512-NEXT:    .cfi_def_cfa_offset 48
-; AVX512-NEXT:    .cfi_offset %rbx, -32
-; AVX512-NEXT:    .cfi_offset %r14, -24
-; AVX512-NEXT:    .cfi_offset %r15, -16
-; AVX512-NEXT:    movq %rcx, %rbx
-; AVX512-NEXT:    movq %rsi, %r14
-; AVX512-NEXT:    vmovdqa (%rdi), %xmm0
-; AVX512-NEXT:    vpaddq (%rdx), %xmm0, %xmm0
-; AVX512-NEXT:    vmovdqa %xmm0, (%rsp) # 16-byte Spill
-; AVX512-NEXT:    vmovq %xmm0, %r15
-; AVX512-NEXT:    vmovq %xmm0, (%rsi)
-; AVX512-NEXT:    callq clobber at PLT
-; AVX512-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 16-byte Reload
-; AVX512-NEXT:    movl %ebx, %ecx
-; AVX512-NEXT:    shldq %cl, %r15, %rax
-; AVX512-NEXT:    movq %rax, 8(%r14)
-; AVX512-NEXT:    addq $16, %rsp
-; AVX512-NEXT:    .cfi_def_cfa_offset 32
-; AVX512-NEXT:    popq %rbx
-; AVX512-NEXT:    .cfi_def_cfa_offset 24
-; AVX512-NEXT:    popq %r14
-; AVX512-NEXT:    .cfi_def_cfa_offset 16
-; AVX512-NEXT:    popq %r15
-; AVX512-NEXT:    .cfi_def_cfa_offset 8
-; AVX512-NEXT:    retq
-
-  %v1 = load <2 x i64>, ptr %src, align 16
-  %v2 = load <2 x i64>, ptr %src2, align 16
-  %vec = add <2 x i64> %v1, %v2
-
-  %lo = extractelement <2 x i64> %vec, i64 0
-  store i64 %lo, ptr %dst, align 8
-
-  call void @clobber()
-
-  %hi = extractelement <2 x i64> %vec, i64 1
-  %r = call i64 @llvm.fshl.i64(i64 %hi, i64 %lo, i64 %sh)
-  %out = getelementptr i8, ptr %dst, i64 8
-  store i64 %r, ptr %out, align 8
-  ret void
-}
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; CHECK: {{.*}}



More information about the llvm-commits mailing list