[llvm] [X86] Don't allocate unused padding in front of realigned locals (PR #227495)
Timur Golubovich via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 07:00:38 PDT 2026
https://github.com/timurgol007 updated https://github.com/llvm/llvm-project/pull/227495
>From 650afe8e04f409adb3bdd592d3f6787317afcc21 Mon Sep 17 00:00:00 2001
From: Timur Golubovich <timur.golubovich at intel.com>
Date: Fri, 25 Sep 2026 18:22:09 +0200
Subject: [PATCH] [X86] Don't allocate unused padding in front of realigned
locals
PEI aligns local objects relative to the incoming stack pointer, leaving
a gap between the CSRs and the first local. With the stack realigned after
the pushes, locals are addressed from the final stack pointer, so that gap
is never used: exclude it from the prologue's allocation.
Fixes #224278
---
llvm/lib/Target/X86/X86FrameLowering.cpp | 37 ++++++-
.../X86/2009-06-05-VariableIndexInsert.ll | 2 +-
llvm/test/CodeGen/X86/AMX/amx-across-func.ll | 6 +-
llvm/test/CodeGen/X86/AMX/amx-configO2toO0.ll | 2 +-
llvm/test/CodeGen/X86/AMX/amx-zero-config.ll | 6 +-
.../CodeGen/X86/amx-across-func-tilemovrow.ll | 2 +-
llvm/test/CodeGen/X86/amx_movrs_intrinsics.ll | 12 +-
llvm/test/CodeGen/X86/andnot-sink-not.ll | 4 +-
llvm/test/CodeGen/X86/arg-copy-elide.ll | 2 +-
llvm/test/CodeGen/X86/atomic-fp.ll | 8 +-
.../X86/atomic-idempotent-syncscope.ll | 4 +-
llvm/test/CodeGen/X86/atomic-idempotent.ll | 4 +-
llvm/test/CodeGen/X86/atomic-xor.ll | 2 +-
llvm/test/CodeGen/X86/atomic64.ll | 8 +-
llvm/test/CodeGen/X86/avx2-vbroadcast.ll | 16 +--
.../test/CodeGen/X86/avx512-insert-extract.ll | 48 ++++----
.../CodeGen/X86/avx512-insert-extract_i1.ll | 2 +-
llvm/test/CodeGen/X86/avx512-intel-ocl.ll | 12 +-
llvm/test/CodeGen/X86/avx512fp16-cvt.ll | 6 +-
llvm/test/CodeGen/X86/avx512fp16-mov.ll | 8 +-
.../X86/bfloat-calling-conv-no-sse2.ll | 4 +-
llvm/test/CodeGen/X86/bittest-big-integer.ll | 6 +-
.../X86/dagcombine-tokenfactor-limit-crash.ll | 2 +-
.../X86/div-rem-pair-recomposition-signed.ll | 2 +-
.../div-rem-pair-recomposition-unsigned.ll | 2 +-
llvm/test/CodeGen/X86/extractelement-index.ll | 8 +-
llvm/test/CodeGen/X86/extractelement-load.ll | 6 +-
llvm/test/CodeGen/X86/fma.ll | 4 +-
.../test/CodeGen/X86/fp128-libcalls-strict.ll | 8 +-
llvm/test/CodeGen/X86/fp128-libcalls.ll | 16 +--
llvm/test/CodeGen/X86/gep-expanded-vector.ll | 2 +-
llvm/test/CodeGen/X86/i128-fp128-abi.ll | 8 +-
llvm/test/CodeGen/X86/i128-udiv.ll | 18 +--
llvm/test/CodeGen/X86/i64-mem-copy.ll | 4 +-
llvm/test/CodeGen/X86/inline-sse.ll | 2 +-
.../CodeGen/X86/insertelement-var-index.ll | 64 +++++------
llvm/test/CodeGen/X86/isel-x87.ll | 4 +-
.../test/CodeGen/X86/long-double-abi-align.ll | 4 +-
llvm/test/CodeGen/X86/matrix-multiply.ll | 2 +-
.../X86/memset-sse-stack-realignment.ll | 8 +-
llvm/test/CodeGen/X86/mingw-alloca.ll | 4 +-
llvm/test/CodeGen/X86/mmx-arith.ll | 2 +-
llvm/test/CodeGen/X86/mmx-intrinsics.ll | 2 +-
llvm/test/CodeGen/X86/musttail-varargs.ll | 4 +-
llvm/test/CodeGen/X86/nontemporal-loads-2.ll | 60 +++++-----
llvm/test/CodeGen/X86/nosse-vector.ll | 2 +-
llvm/test/CodeGen/X86/packus.ll | 2 +-
llvm/test/CodeGen/X86/pr216504.ll | 4 +-
llvm/test/CodeGen/X86/pr32284.ll | 2 +-
llvm/test/CodeGen/X86/pr34080-2.ll | 2 +-
llvm/test/CodeGen/X86/pr34592.ll | 2 +-
llvm/test/CodeGen/X86/pr34653.ll | 2 +-
llvm/test/CodeGen/X86/pr38539.ll | 2 +-
llvm/test/CodeGen/X86/pr43866.ll | 2 +-
llvm/test/CodeGen/X86/pr50782.ll | 2 +-
llvm/test/CodeGen/X86/sdiv_fix.ll | 2 +-
llvm/test/CodeGen/X86/sdiv_fix_sat.ll | 4 +-
llvm/test/CodeGen/X86/shift-i128.ll | 16 +--
llvm/test/CodeGen/X86/shift-i256.ll | 28 ++---
.../CodeGen/X86/shuffle-combine-crash-4.ll | 2 +-
llvm/test/CodeGen/X86/sse-intel-ocl.ll | 4 +-
.../CodeGen/X86/sse-intrinsics-fast-isel.ll | 4 +-
llvm/test/CodeGen/X86/sse-regcall.ll | 4 +-
llvm/test/CodeGen/X86/sse-regcall4.ll | 4 +-
.../stack-clash-small-alloc-medium-align.ll | 2 +-
.../X86/stack-realign-local-padding.ll | 104 ++++++++++++++++++
.../X86/statepoint-no-realign-stack.ll | 4 +-
llvm/test/CodeGen/X86/sttni.ll | 16 +--
llvm/test/CodeGen/X86/udiv_fix.ll | 2 +-
llvm/test/CodeGen/X86/udiv_fix_sat.ll | 2 +-
llvm/test/CodeGen/X86/var-permute-256.ll | 6 +-
llvm/test/CodeGen/X86/var-permute-512.ll | 12 +-
llvm/test/CodeGen/X86/vec-strict-128.ll | 2 +-
.../CodeGen/X86/vec-strict-fptoint-256.ll | 8 +-
llvm/test/CodeGen/X86/vec_ins_extract-1.ll | 8 +-
llvm/test/CodeGen/X86/vec_insert-8.ll | 4 +-
llvm/test/CodeGen/X86/vector-compress.ll | 38 +++----
llvm/test/CodeGen/X86/vector-extend-inreg.ll | 6 +-
.../CodeGen/X86/vector-extract-last-active.ll | 12 +-
llvm/test/CodeGen/X86/vector-llrint.ll | 10 +-
llvm/test/CodeGen/X86/vector-lrint.ll | 8 +-
llvm/test/CodeGen/X86/vector-reduce-ctpop.ll | 10 +-
llvm/test/CodeGen/X86/vector-reduce-smax.ll | 4 +-
llvm/test/CodeGen/X86/vector-reduce-smin.ll | 2 +-
llvm/test/CodeGen/X86/vector-reduce-umax.ll | 6 +-
llvm/test/CodeGen/X86/vector-reduce-umin.ll | 2 +-
.../CodeGen/X86/vector-shuffle-512-v16.ll | 2 +-
.../X86/vector-shuffle-variable-256.ll | 16 +--
llvm/test/CodeGen/X86/widen_arith-6.ll | 2 +-
llvm/test/CodeGen/X86/x86-64-baseptr.ll | 2 +-
.../test/DebugInfo/COFF/fpo-realign-alloca.ll | 4 +-
llvm/test/DebugInfo/COFF/vframe-csr.ll | 12 +-
92 files changed, 484 insertions(+), 349 deletions(-)
create mode 100644 llvm/test/CodeGen/X86/stack-realign-local-padding.ll
diff --git a/llvm/lib/Target/X86/X86FrameLowering.cpp b/llvm/lib/Target/X86/X86FrameLowering.cpp
index a25aba6d0afe0a..146e191c2997ba 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.cpp
+++ b/llvm/lib/Target/X86/X86FrameLowering.cpp
@@ -1589,6 +1589,32 @@ static bool isOpcodeRep(unsigned Opcode) {
return false;
}
+/// Returns the number of bytes between the end of the fixed and callee-save
+/// area and the first local object, which PEI leaves as padding when it aligns
+/// the local objects relative to the incoming stack pointer.
+static uint64_t getUnusedLocalAreaPadding(const MachineFrameInfo &MFI) {
+ int64_t FixedEnd = 0;
+ int64_t LocalsTop = std::numeric_limits<int64_t>::max();
+ for (int I : seq(MFI.getObjectIndexBegin(), MFI.getObjectIndexEnd())) {
+ if (MFI.isDeadObjectIndex(I) || MFI.isVariableSizedObjectIndex(I) ||
+ MFI.getStackID(I) != TargetStackID::Default)
+ continue;
+
+ int64_t ObjOffset = MFI.getObjectOffset(I);
+ int64_t ObjSize = MFI.getObjectSize(I);
+ // Offsets are negative, measured from the incoming stack pointer.
+ if (MFI.isFixedObjectIndex(I))
+ FixedEnd = std::max(FixedEnd, -ObjOffset);
+ else
+ LocalsTop = std::min<int64_t>(LocalsTop, -ObjOffset - ObjSize);
+ }
+ if (LocalsTop == std::numeric_limits<int64_t>::max())
+ return 0;
+
+ assert(LocalsTop >= FixedEnd && "Local object overlaps the fixed area");
+ return LocalsTop - FixedEnd;
+}
+
/// emitPrologue - Push callee-saved registers onto the stack, which
/// automatically adjust the stack pointer. Adjust the stack pointer to allocate
/// space for local variables. Also emit labels used by the exception handler to
@@ -1901,9 +1927,14 @@ void X86FrameLowering::emitPrologue(MachineFunction &MF,
NumBytes =
FrameSize - (X86FI->getCalleeSavedFrameSize() + TailCallArgReserveSize);
- // Callee-saved registers are pushed on stack before the stack is realigned.
- if (TRI->hasStackRealignment(MF) && !IsWin64Prologue)
- NumBytes = alignTo(NumBytes, MaxAlign);
+ // Callee-saved registers are pushed on stack before the stack is realigned,
+ // and the realignment itself already provides the local objects' alignment,
+ // so leave out the padding PEI put in front of them for it.
+ if (TRI->hasStackRealignment(MF) && !IsWin64Prologue) {
+ uint64_t Padding = getUnusedLocalAreaPadding(MFI);
+ assert(Padding <= NumBytes && "Padding exceeds the local area");
+ NumBytes = alignTo(NumBytes - Padding, MaxAlign);
+ }
// Save EBP/RBP into the appropriate stack slot.
auto EmitSEHPushFramePtr = [&]() {
diff --git a/llvm/test/CodeGen/X86/2009-06-05-VariableIndexInsert.ll b/llvm/test/CodeGen/X86/2009-06-05-VariableIndexInsert.ll
index 695a2d0cd806e0..99913ae148811f 100644
--- a/llvm/test/CodeGen/X86/2009-06-05-VariableIndexInsert.ll
+++ b/llvm/test/CodeGen/X86/2009-06-05-VariableIndexInsert.ll
@@ -8,7 +8,7 @@ define <2 x i64> @_mm_insert_epi16(<2 x i64> %a, i32 %b, i32 %imm) nounwind read
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $32, %esp
+; X86-NEXT: subl $16, %esp
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: movzwl 8(%ebp), %ecx
; X86-NEXT: andl $7, %eax
diff --git a/llvm/test/CodeGen/X86/AMX/amx-across-func.ll b/llvm/test/CodeGen/X86/AMX/amx-across-func.ll
index cbd4ee03706426..ba447c494b728d 100644
--- a/llvm/test/CodeGen/X86/AMX/amx-across-func.ll
+++ b/llvm/test/CodeGen/X86/AMX/amx-across-func.ll
@@ -103,7 +103,7 @@ define dso_local void @test_api(i16 signext %0, i16 signext %1) nounwind {
; O0-NEXT: pushq %rbp
; O0-NEXT: movq %rsp, %rbp
; O0-NEXT: andq $-1024, %rsp # imm = 0xFC00
-; O0-NEXT: subq $8192, %rsp # imm = 0x2000
+; O0-NEXT: subq $7168, %rsp # imm = 0x1C00
; O0-NEXT: vxorps %xmm0, %xmm0, %xmm0
; O0-NEXT: # kill: def $zmm0 killed $xmm0
; O0-NEXT: vmovups %zmm0, {{[0-9]+}}(%rsp)
@@ -337,7 +337,7 @@ define dso_local i32 @test_loop(i32 %0) nounwind {
; O0-NEXT: pushq %rbp
; O0-NEXT: movq %rsp, %rbp
; O0-NEXT: andq $-1024, %rsp # imm = 0xFC00
-; O0-NEXT: subq $4096, %rsp # imm = 0x1000
+; O0-NEXT: subq $3072, %rsp # imm = 0xC00
; O0-NEXT: vxorps %xmm0, %xmm0, %xmm0
; O0-NEXT: # kill: def $zmm0 killed $xmm0
; O0-NEXT: vmovups %zmm0, {{[0-9]+}}(%rsp)
@@ -557,7 +557,7 @@ define dso_local void @test_loop2(i32 %0) nounwind {
; O0-NEXT: pushq %rbp
; O0-NEXT: movq %rsp, %rbp
; O0-NEXT: andq $-1024, %rsp # imm = 0xFC00
-; O0-NEXT: subq $3072, %rsp # imm = 0xC00
+; O0-NEXT: subq $2048, %rsp # imm = 0x800
; O0-NEXT: vxorps %xmm0, %xmm0, %xmm0
; O0-NEXT: # kill: def $zmm0 killed $xmm0
; O0-NEXT: vmovups %zmm0, {{[0-9]+}}(%rsp)
diff --git a/llvm/test/CodeGen/X86/AMX/amx-configO2toO0.ll b/llvm/test/CodeGen/X86/AMX/amx-configO2toO0.ll
index 5eb036ca47563e..0690bdcf9cc2ff 100644
--- a/llvm/test/CodeGen/X86/AMX/amx-configO2toO0.ll
+++ b/llvm/test/CodeGen/X86/AMX/amx-configO2toO0.ll
@@ -10,7 +10,7 @@ define dso_local void @test_api(i32 %cond, i16 signext %row, i16 signext %col) n
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-1024, %rsp # imm = 0xFC00
-; AVX512-NEXT: subq $8192, %rsp # imm = 0x2000
+; AVX512-NEXT: subq $7168, %rsp # imm = 0x1C00
; AVX512-NEXT: vxorps %xmm0, %xmm0, %xmm0
; AVX512-NEXT: # kill: def $zmm0 killed $xmm0
; AVX512-NEXT: vmovups %zmm0, {{[0-9]+}}(%rsp)
diff --git a/llvm/test/CodeGen/X86/AMX/amx-zero-config.ll b/llvm/test/CodeGen/X86/AMX/amx-zero-config.ll
index c4901ab136c71e..96d6fb4e7aa5f7 100644
--- a/llvm/test/CodeGen/X86/AMX/amx-zero-config.ll
+++ b/llvm/test/CodeGen/X86/AMX/amx-zero-config.ll
@@ -66,7 +66,7 @@ define void @foo(ptr %buf) nounwind {
; AVX512-O0-NEXT: pushq %rbp
; AVX512-O0-NEXT: movq %rsp, %rbp
; AVX512-O0-NEXT: andq $-1024, %rsp # imm = 0xFC00
-; AVX512-O0-NEXT: subq $3072, %rsp # imm = 0xC00
+; AVX512-O0-NEXT: subq $2048, %rsp # imm = 0x800
; AVX512-O0-NEXT: vxorps %xmm0, %xmm0, %xmm0
; AVX512-O0-NEXT: # kill: def $zmm0 killed $xmm0
; AVX512-O0-NEXT: vmovups %zmm0, {{[0-9]+}}(%rsp)
@@ -107,7 +107,7 @@ define void @foo(ptr %buf) nounwind {
; AVX2-O0-NEXT: pushq %rbp
; AVX2-O0-NEXT: movq %rsp, %rbp
; AVX2-O0-NEXT: andq $-1024, %rsp # imm = 0xFC00
-; AVX2-O0-NEXT: subq $3072, %rsp # imm = 0xC00
+; AVX2-O0-NEXT: subq $2048, %rsp # imm = 0x800
; AVX2-O0-NEXT: vxorps %xmm0, %xmm0, %xmm0
; AVX2-O0-NEXT: # kill: def $ymm0 killed $xmm0
; AVX2-O0-NEXT: vmovups %ymm0, {{[0-9]+}}(%rsp)
@@ -149,7 +149,7 @@ define void @foo(ptr %buf) nounwind {
; SSE2-O0-NEXT: pushq %rbp
; SSE2-O0-NEXT: movq %rsp, %rbp
; SSE2-O0-NEXT: andq $-1024, %rsp # imm = 0xFC00
-; SSE2-O0-NEXT: subq $3072, %rsp # imm = 0xC00
+; SSE2-O0-NEXT: subq $2048, %rsp # imm = 0x800
; SSE2-O0-NEXT: xorps %xmm0, %xmm0
; SSE2-O0-NEXT: movups %xmm0, {{[0-9]+}}(%rsp)
; SSE2-O0-NEXT: movups %xmm0, {{[0-9]+}}(%rsp)
diff --git a/llvm/test/CodeGen/X86/amx-across-func-tilemovrow.ll b/llvm/test/CodeGen/X86/amx-across-func-tilemovrow.ll
index d317deadb0354e..bdaafc4e847441 100644
--- a/llvm/test/CodeGen/X86/amx-across-func-tilemovrow.ll
+++ b/llvm/test/CodeGen/X86/amx-across-func-tilemovrow.ll
@@ -94,7 +94,7 @@ define dso_local <16 x i32> @test_api(i16 signext %0, i16 signext %1) nounwind {
; O0-NEXT: pushq %rbp
; O0-NEXT: movq %rsp, %rbp
; O0-NEXT: andq $-1024, %rsp # imm = 0xFC00
-; O0-NEXT: subq $4096, %rsp # imm = 0x1000
+; O0-NEXT: subq $3072, %rsp # imm = 0xC00
; O0-NEXT: vpxor %xmm0, %xmm0, %xmm0
; O0-NEXT: # kill: def $zmm0 killed $xmm0
; O0-NEXT: vmovups %zmm0, {{[0-9]+}}(%rsp)
diff --git a/llvm/test/CodeGen/X86/amx_movrs_intrinsics.ll b/llvm/test/CodeGen/X86/amx_movrs_intrinsics.ll
index 1b93ae029f27b9..7e31d502ead2bd 100755
--- a/llvm/test/CodeGen/X86/amx_movrs_intrinsics.ll
+++ b/llvm/test/CodeGen/X86/amx_movrs_intrinsics.ll
@@ -11,7 +11,7 @@ define void @test_amx_internal(i16 %m, i16 %n, ptr %buf, i64 %s) {
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: .cfi_def_cfa_register %rbp
; CHECK-NEXT: andq $-1024, %rsp # imm = 0xFC00
-; CHECK-NEXT: subq $3072, %rsp # imm = 0xC00
+; CHECK-NEXT: subq $2048, %rsp # imm = 0x800
; CHECK-NEXT: xorps %xmm0, %xmm0
; CHECK-NEXT: movups %xmm0, {{[0-9]+}}(%rsp)
; CHECK-NEXT: movups %xmm0, {{[0-9]+}}(%rsp)
@@ -46,8 +46,8 @@ define void @test_amx_internal(i16 %m, i16 %n, ptr %buf, i64 %s) {
; EGPR-NEXT: .cfi_def_cfa_register %rbp
; EGPR-NEXT: andq $-1024, %rsp # encoding: [0x48,0x81,0xe4,0x00,0xfc,0xff,0xff]
; EGPR-NEXT: # imm = 0xFC00
-; EGPR-NEXT: subq $3072, %rsp # encoding: [0x48,0x81,0xec,0x00,0x0c,0x00,0x00]
-; EGPR-NEXT: # imm = 0xC00
+; EGPR-NEXT: subq $2048, %rsp # encoding: [0x48,0x81,0xec,0x00,0x08,0x00,0x00]
+; EGPR-NEXT: # imm = 0x800
; EGPR-NEXT: xorps %xmm0, %xmm0 # encoding: [0x0f,0x57,0xc0]
; EGPR-NEXT: movups %xmm0, {{[0-9]+}}(%rsp) # encoding: [0x0f,0x11,0x84,0x24,0xc0,0x03,0x00,0x00]
; EGPR-NEXT: movups %xmm0, {{[0-9]+}}(%rsp) # encoding: [0x0f,0x11,0x84,0x24,0xd0,0x03,0x00,0x00]
@@ -108,7 +108,7 @@ define void @test_amx_t1_internal(i16 %m, i16 %n, ptr %buf, i64 %s) {
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: .cfi_def_cfa_register %rbp
; CHECK-NEXT: andq $-1024, %rsp # imm = 0xFC00
-; CHECK-NEXT: subq $3072, %rsp # imm = 0xC00
+; CHECK-NEXT: subq $2048, %rsp # imm = 0x800
; CHECK-NEXT: xorps %xmm0, %xmm0
; CHECK-NEXT: movups %xmm0, {{[0-9]+}}(%rsp)
; CHECK-NEXT: movups %xmm0, {{[0-9]+}}(%rsp)
@@ -143,8 +143,8 @@ define void @test_amx_t1_internal(i16 %m, i16 %n, ptr %buf, i64 %s) {
; EGPR-NEXT: .cfi_def_cfa_register %rbp
; EGPR-NEXT: andq $-1024, %rsp # encoding: [0x48,0x81,0xe4,0x00,0xfc,0xff,0xff]
; EGPR-NEXT: # imm = 0xFC00
-; EGPR-NEXT: subq $3072, %rsp # encoding: [0x48,0x81,0xec,0x00,0x0c,0x00,0x00]
-; EGPR-NEXT: # imm = 0xC00
+; EGPR-NEXT: subq $2048, %rsp # encoding: [0x48,0x81,0xec,0x00,0x08,0x00,0x00]
+; EGPR-NEXT: # imm = 0x800
; EGPR-NEXT: xorps %xmm0, %xmm0 # encoding: [0x0f,0x57,0xc0]
; EGPR-NEXT: movups %xmm0, {{[0-9]+}}(%rsp) # encoding: [0x0f,0x11,0x84,0x24,0xc0,0x03,0x00,0x00]
; EGPR-NEXT: movups %xmm0, {{[0-9]+}}(%rsp) # encoding: [0x0f,0x11,0x84,0x24,0xd0,0x03,0x00,0x00]
diff --git a/llvm/test/CodeGen/X86/andnot-sink-not.ll b/llvm/test/CodeGen/X86/andnot-sink-not.ll
index fefbdc84699f44..bd6442d5edb0a8 100644
--- a/llvm/test/CodeGen/X86/andnot-sink-not.ll
+++ b/llvm/test/CodeGen/X86/andnot-sink-not.ll
@@ -1018,7 +1018,7 @@ define <4 x i32> @and_sink_not_v4i32(<4 x i32> %x, <4 x i32> %m, i1 zeroext %con
; X86-SSE-NEXT: pushl %edi
; X86-SSE-NEXT: pushl %esi
; X86-SSE-NEXT: andl $-16, %esp
-; X86-SSE-NEXT: subl $64, %esp
+; X86-SSE-NEXT: subl $48, %esp
; X86-SSE-NEXT: movl 8(%ebp), %eax
; X86-SSE-NEXT: movl 24(%ebp), %ecx
; X86-SSE-NEXT: movl 20(%ebp), %edx
@@ -1190,7 +1190,7 @@ define <4 x i32> @and_sink_not_v4i32_swapped(<4 x i32> %x, <4 x i32> %m, i1 zero
; X86-SSE-NEXT: pushl %edi
; X86-SSE-NEXT: pushl %esi
; X86-SSE-NEXT: andl $-16, %esp
-; X86-SSE-NEXT: subl $64, %esp
+; X86-SSE-NEXT: subl $48, %esp
; X86-SSE-NEXT: movl 8(%ebp), %eax
; X86-SSE-NEXT: movl 24(%ebp), %ecx
; X86-SSE-NEXT: movl 20(%ebp), %edx
diff --git a/llvm/test/CodeGen/X86/arg-copy-elide.ll b/llvm/test/CodeGen/X86/arg-copy-elide.ll
index 15edb612d7649d..b51f32f2b9675d 100644
--- a/llvm/test/CodeGen/X86/arg-copy-elide.ll
+++ b/llvm/test/CodeGen/X86/arg-copy-elide.ll
@@ -187,7 +187,7 @@ define void @split_i128(ptr %sret, i128 %x) {
; CHECK-NEXT: pushl %edi
; CHECK-NEXT: pushl %esi
; CHECK-NEXT: andl $-16, %esp
-; CHECK-NEXT: subl $48, %esp
+; CHECK-NEXT: subl $32, %esp
; CHECK-NEXT: movl 24(%ebp), %eax
; CHECK-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; CHECK-NEXT: movl 28(%ebp), %ebx
diff --git a/llvm/test/CodeGen/X86/atomic-fp.ll b/llvm/test/CodeGen/X86/atomic-fp.ll
index 2dee1d12e72558..36540422a139c0 100644
--- a/llvm/test/CodeGen/X86/atomic-fp.ll
+++ b/llvm/test/CodeGen/X86/atomic-fp.ll
@@ -625,7 +625,7 @@ define dso_local void @fadd_array(ptr %arg, double %arg1, i64 %arg2) nounwind {
; X86-NOSSE-NEXT: movl %esp, %ebp
; X86-NOSSE-NEXT: pushl %esi
; X86-NOSSE-NEXT: andl $-8, %esp
-; X86-NOSSE-NEXT: subl $32, %esp
+; X86-NOSSE-NEXT: subl $24, %esp
; X86-NOSSE-NEXT: movl 20(%ebp), %eax
; X86-NOSSE-NEXT: movl 8(%ebp), %ecx
; X86-NOSSE-NEXT: fildll (%ecx,%eax,8)
@@ -1339,7 +1339,7 @@ define dso_local void @fsub_array(ptr %arg, double %arg1, i64 %arg2) nounwind {
; X86-NOSSE-NEXT: movl %esp, %ebp
; X86-NOSSE-NEXT: pushl %esi
; X86-NOSSE-NEXT: andl $-8, %esp
-; X86-NOSSE-NEXT: subl $32, %esp
+; X86-NOSSE-NEXT: subl $24, %esp
; X86-NOSSE-NEXT: movl 20(%ebp), %eax
; X86-NOSSE-NEXT: movl 8(%ebp), %ecx
; X86-NOSSE-NEXT: fildll (%ecx,%eax,8)
@@ -2043,7 +2043,7 @@ define dso_local void @fmul_array(ptr %arg, double %arg1, i64 %arg2) nounwind {
; X86-NOSSE-NEXT: movl %esp, %ebp
; X86-NOSSE-NEXT: pushl %esi
; X86-NOSSE-NEXT: andl $-8, %esp
-; X86-NOSSE-NEXT: subl $32, %esp
+; X86-NOSSE-NEXT: subl $24, %esp
; X86-NOSSE-NEXT: movl 20(%ebp), %eax
; X86-NOSSE-NEXT: movl 8(%ebp), %ecx
; X86-NOSSE-NEXT: fildll (%ecx,%eax,8)
@@ -2751,7 +2751,7 @@ define dso_local void @fdiv_array(ptr %arg, double %arg1, i64 %arg2) nounwind {
; X86-NOSSE-NEXT: movl %esp, %ebp
; X86-NOSSE-NEXT: pushl %esi
; X86-NOSSE-NEXT: andl $-8, %esp
-; X86-NOSSE-NEXT: subl $32, %esp
+; X86-NOSSE-NEXT: subl $24, %esp
; X86-NOSSE-NEXT: movl 20(%ebp), %eax
; X86-NOSSE-NEXT: movl 8(%ebp), %ecx
; X86-NOSSE-NEXT: fildll (%ecx,%eax,8)
diff --git a/llvm/test/CodeGen/X86/atomic-idempotent-syncscope.ll b/llvm/test/CodeGen/X86/atomic-idempotent-syncscope.ll
index 9e20fdb59f552c..2502e212f9ffd3 100644
--- a/llvm/test/CodeGen/X86/atomic-idempotent-syncscope.ll
+++ b/llvm/test/CodeGen/X86/atomic-idempotent-syncscope.ll
@@ -143,7 +143,7 @@ define i128 @or128(ptr %p) #0 {
; X86-GENERIC-NEXT: pushl %edi
; X86-GENERIC-NEXT: pushl %esi
; X86-GENERIC-NEXT: andl $-16, %esp
-; X86-GENERIC-NEXT: subl $48, %esp
+; X86-GENERIC-NEXT: subl $32, %esp
; X86-GENERIC-NEXT: movl 12(%ebp), %edi
; X86-GENERIC-NEXT: movl 12(%edi), %ecx
; X86-GENERIC-NEXT: movl 8(%edi), %edx
@@ -460,7 +460,7 @@ define void @or128_nouse_seq_cst(ptr %p) #0 {
; X86-GENERIC-NEXT: pushl %edi
; X86-GENERIC-NEXT: pushl %esi
; X86-GENERIC-NEXT: andl $-16, %esp
-; X86-GENERIC-NEXT: subl $48, %esp
+; X86-GENERIC-NEXT: subl $32, %esp
; X86-GENERIC-NEXT: movl 8(%ebp), %esi
; X86-GENERIC-NEXT: movl 12(%esi), %ecx
; X86-GENERIC-NEXT: movl 8(%esi), %edi
diff --git a/llvm/test/CodeGen/X86/atomic-idempotent.ll b/llvm/test/CodeGen/X86/atomic-idempotent.ll
index 01c3e7999a92ce..2516604b4d2af2 100644
--- a/llvm/test/CodeGen/X86/atomic-idempotent.ll
+++ b/llvm/test/CodeGen/X86/atomic-idempotent.ll
@@ -158,7 +158,7 @@ define i128 @or128(ptr %p) #0 {
; X86-GENERIC-NEXT: pushl %edi
; X86-GENERIC-NEXT: pushl %esi
; X86-GENERIC-NEXT: andl $-16, %esp
-; X86-GENERIC-NEXT: subl $48, %esp
+; X86-GENERIC-NEXT: subl $32, %esp
; X86-GENERIC-NEXT: movl 12(%ebp), %edi
; X86-GENERIC-NEXT: movl 12(%edi), %ecx
; X86-GENERIC-NEXT: movl 8(%edi), %edx
@@ -478,7 +478,7 @@ define void @or128_nouse_seq_cst(ptr %p) #0 {
; X86-GENERIC-NEXT: pushl %edi
; X86-GENERIC-NEXT: pushl %esi
; X86-GENERIC-NEXT: andl $-16, %esp
-; X86-GENERIC-NEXT: subl $48, %esp
+; X86-GENERIC-NEXT: subl $32, %esp
; X86-GENERIC-NEXT: movl 8(%ebp), %esi
; X86-GENERIC-NEXT: movl 12(%esi), %ecx
; X86-GENERIC-NEXT: movl 8(%esi), %edi
diff --git a/llvm/test/CodeGen/X86/atomic-xor.ll b/llvm/test/CodeGen/X86/atomic-xor.ll
index c648ecdfbe674b..cad3f2e84c3b67 100644
--- a/llvm/test/CodeGen/X86/atomic-xor.ll
+++ b/llvm/test/CodeGen/X86/atomic-xor.ll
@@ -26,7 +26,7 @@ define i128 @xor128_signbit_used(ptr %p) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: movl 12(%ebp), %edi
; X86-NEXT: movl 12(%edi), %ecx
; X86-NEXT: movl 8(%edi), %edx
diff --git a/llvm/test/CodeGen/X86/atomic64.ll b/llvm/test/CodeGen/X86/atomic64.ll
index 8f4da356e06cbb..d3bafbb06cc109 100644
--- a/llvm/test/CodeGen/X86/atomic64.ll
+++ b/llvm/test/CodeGen/X86/atomic64.ll
@@ -328,7 +328,7 @@ define void @atomic_fetch_max64(i64 %x) nounwind {
; I486-NEXT: movl %esp, %ebp
; I486-NEXT: pushl %esi
; I486-NEXT: andl $-8, %esp
-; I486-NEXT: subl $72, %esp
+; I486-NEXT: subl $64, %esp
; I486-NEXT: movl 12(%ebp), %eax
; I486-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; I486-NEXT: movl 8(%ebp), %eax
@@ -420,7 +420,7 @@ define void @atomic_fetch_min64(i64 %x) nounwind {
; I486-NEXT: movl %esp, %ebp
; I486-NEXT: pushl %esi
; I486-NEXT: andl $-8, %esp
-; I486-NEXT: subl $72, %esp
+; I486-NEXT: subl $64, %esp
; I486-NEXT: movl 12(%ebp), %eax
; I486-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; I486-NEXT: movl 8(%ebp), %eax
@@ -512,7 +512,7 @@ define void @atomic_fetch_umax64(i64 %x) nounwind {
; I486-NEXT: movl %esp, %ebp
; I486-NEXT: pushl %esi
; I486-NEXT: andl $-8, %esp
-; I486-NEXT: subl $72, %esp
+; I486-NEXT: subl $64, %esp
; I486-NEXT: movl 12(%ebp), %eax
; I486-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; I486-NEXT: movl 8(%ebp), %eax
@@ -604,7 +604,7 @@ define void @atomic_fetch_umin64(i64 %x) nounwind {
; I486-NEXT: movl %esp, %ebp
; I486-NEXT: pushl %esi
; I486-NEXT: andl $-8, %esp
-; I486-NEXT: subl $72, %esp
+; I486-NEXT: subl $64, %esp
; I486-NEXT: movl 12(%ebp), %eax
; I486-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; I486-NEXT: movl 8(%ebp), %eax
diff --git a/llvm/test/CodeGen/X86/avx2-vbroadcast.ll b/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
index a5601f17256828..311f74a2bce9ec 100644
--- a/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
+++ b/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
@@ -1133,7 +1133,7 @@ define void @isel_crash_32b(ptr %cV_R.addr) {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: .cfi_def_cfa_register %ebp
; X86-NEXT: andl $-32, %esp
-; X86-NEXT: addl $-128, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X86-NEXT: vmovaps %ymm0, (%esp)
@@ -1153,7 +1153,7 @@ define void @isel_crash_32b(ptr %cV_R.addr) {
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: .cfi_def_cfa_register %rbp
; X64-NEXT: andq $-32, %rsp
-; X64-NEXT: addq $-128, %rsp
+; X64-NEXT: subq $96, %rsp
; X64-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X64-NEXT: vmovaps %ymm0, (%rsp)
; X64-NEXT: vpbroadcastb (%rdi), %ymm1
@@ -1224,7 +1224,7 @@ define void @isel_crash_16w(ptr %cV_R.addr) {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: .cfi_def_cfa_register %ebp
; X86-NEXT: andl $-32, %esp
-; X86-NEXT: addl $-128, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X86-NEXT: vmovaps %ymm0, (%esp)
@@ -1244,7 +1244,7 @@ define void @isel_crash_16w(ptr %cV_R.addr) {
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: .cfi_def_cfa_register %rbp
; X64-NEXT: andq $-32, %rsp
-; X64-NEXT: addq $-128, %rsp
+; X64-NEXT: subq $96, %rsp
; X64-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X64-NEXT: vmovaps %ymm0, (%rsp)
; X64-NEXT: vpbroadcastw (%rdi), %ymm1
@@ -1315,7 +1315,7 @@ define void @isel_crash_8d(ptr %cV_R.addr) {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: .cfi_def_cfa_register %ebp
; X86-NEXT: andl $-32, %esp
-; X86-NEXT: addl $-128, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X86-NEXT: vmovaps %ymm0, (%esp)
@@ -1335,7 +1335,7 @@ define void @isel_crash_8d(ptr %cV_R.addr) {
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: .cfi_def_cfa_register %rbp
; X64-NEXT: andq $-32, %rsp
-; X64-NEXT: addq $-128, %rsp
+; X64-NEXT: subq $96, %rsp
; X64-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X64-NEXT: vmovaps %ymm0, (%rsp)
; X64-NEXT: vbroadcastss (%rdi), %ymm1
@@ -1405,7 +1405,7 @@ define void @isel_crash_4q(ptr %cV_R.addr) {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: .cfi_def_cfa_register %ebp
; X86-NEXT: andl $-32, %esp
-; X86-NEXT: addl $-128, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X86-NEXT: vmovaps %ymm0, (%esp)
@@ -1425,7 +1425,7 @@ define void @isel_crash_4q(ptr %cV_R.addr) {
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: .cfi_def_cfa_register %rbp
; X64-NEXT: andq $-32, %rsp
-; X64-NEXT: addq $-128, %rsp
+; X64-NEXT: subq $96, %rsp
; X64-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X64-NEXT: vmovaps %ymm0, (%rsp)
; X64-NEXT: vbroadcastsd (%rdi), %ymm1
diff --git a/llvm/test/CodeGen/X86/avx512-insert-extract.ll b/llvm/test/CodeGen/X86/avx512-insert-extract.ll
index 4efc4678f1d002..9458f25ac85155 100644
--- a/llvm/test/CodeGen/X86/avx512-insert-extract.ll
+++ b/llvm/test/CodeGen/X86/avx512-insert-extract.ll
@@ -101,7 +101,7 @@ define float @test7(<16 x float> %x, i32 %ind) nounwind {
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: subq $64, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $15, %edi
@@ -120,7 +120,7 @@ define double @test8(<8 x double> %x, i32 %ind) nounwind {
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: subq $64, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $7, %edi
@@ -139,7 +139,7 @@ define float @test9(<8 x float> %x, i32 %ind) nounwind {
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %ymm0, (%rsp)
; CHECK-NEXT: andl $7, %edi
@@ -158,7 +158,7 @@ define i32 @test10(<16 x i32> %x, i32 %ind) nounwind {
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: subq $64, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $15, %edi
@@ -1129,7 +1129,7 @@ define i64 @test_extractelement_variable_v4i64(<4 x i64> %t1, i32 %index) nounwi
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %ymm0, (%rsp)
; CHECK-NEXT: andl $3, %edi
@@ -1148,7 +1148,7 @@ define i64 @test_extractelement_variable_v8i64(<8 x i64> %t1, i32 %index) nounwi
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: subq $64, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $7, %edi
@@ -1179,7 +1179,7 @@ define double @test_extractelement_variable_v4f64(<4 x double> %t1, i32 %index)
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %ymm0, (%rsp)
; CHECK-NEXT: andl $3, %edi
@@ -1198,7 +1198,7 @@ define double @test_extractelement_variable_v8f64(<8 x double> %t1, i32 %index)
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: subq $64, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $7, %edi
@@ -1229,7 +1229,7 @@ define i32 @test_extractelement_variable_v8i32(<8 x i32> %t1, i32 %index) nounwi
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %ymm0, (%rsp)
; CHECK-NEXT: andl $7, %edi
@@ -1248,7 +1248,7 @@ define i32 @test_extractelement_variable_v16i32(<16 x i32> %t1, i32 %index) noun
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: subq $64, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $15, %edi
@@ -1279,7 +1279,7 @@ define float @test_extractelement_variable_v8f32(<8 x float> %t1, i32 %index) no
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %ymm0, (%rsp)
; CHECK-NEXT: andl $7, %edi
@@ -1298,7 +1298,7 @@ define float @test_extractelement_variable_v16f32(<16 x float> %t1, i32 %index)
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: subq $64, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $15, %edi
@@ -1329,7 +1329,7 @@ define i16 @test_extractelement_variable_v16i16(<16 x i16> %t1, i32 %index) noun
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %ymm0, (%rsp)
; CHECK-NEXT: andl $15, %edi
@@ -1348,7 +1348,7 @@ define i16 @test_extractelement_variable_v32i16(<32 x i16> %t1, i32 %index) noun
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: subq $64, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $31, %edi
@@ -1379,7 +1379,7 @@ define i8 @test_extractelement_variable_v32i8(<32 x i8> %t1, i32 %index) nounwin
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %ymm0, (%rsp)
; CHECK-NEXT: andl $31, %edi
@@ -1399,7 +1399,7 @@ define i8 @test_extractelement_variable_v64i8(<64 x i8> %t1, i32 %index) nounwin
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: subq $64, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $63, %edi
@@ -1419,7 +1419,7 @@ define i8 @test_extractelement_variable_v64i8_indexi8(<64 x i8> %t1, i8 %index)
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: subq $64, %rsp
; CHECK-NEXT: addb %dil, %dil
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: movzbl %dil, %eax
@@ -1577,7 +1577,7 @@ define zeroext i8 @test_extractelement_varible_v32i1(<32 x i8> %a, <32 x i8> %b,
; SKX-NEXT: pushq %rbp
; SKX-NEXT: movq %rsp, %rbp
; SKX-NEXT: andq $-32, %rsp
-; SKX-NEXT: subq $64, %rsp
+; SKX-NEXT: subq $32, %rsp
; SKX-NEXT: ## kill: def $edi killed $edi def $rdi
; SKX-NEXT: vpcmpnleub %ymm1, %ymm0, %k0
; SKX-NEXT: vpmovm2b %k0, %ymm0
@@ -1613,7 +1613,7 @@ define i32 @test_insertelement_variable_v32i1(<32 x i8> %a, i8 %b, i32 %index) n
; KNL-NEXT: pushq %rbp
; KNL-NEXT: movq %rsp, %rbp
; KNL-NEXT: andq $-32, %rsp
-; KNL-NEXT: subq $64, %rsp
+; KNL-NEXT: subq $32, %rsp
; KNL-NEXT: ## kill: def $esi killed $esi def $rsi
; KNL-NEXT: vpxor %xmm1, %xmm1, %xmm1
; KNL-NEXT: vpcmpeqb %ymm1, %ymm0, %ymm0
@@ -1663,7 +1663,7 @@ define i64 @test_insertelement_variable_v64i1(<64 x i8> %a, i8 %b, i32 %index) n
; KNL-NEXT: pushq %rbp
; KNL-NEXT: movq %rsp, %rbp
; KNL-NEXT: andq $-64, %rsp
-; KNL-NEXT: addq $-128, %rsp
+; KNL-NEXT: subq $64, %rsp
; KNL-NEXT: ## kill: def $esi killed $esi def $rsi
; KNL-NEXT: vpxor %xmm1, %xmm1, %xmm1
; KNL-NEXT: vextracti64x4 $1, %zmm0, %ymm2
@@ -1729,7 +1729,7 @@ define i96 @test_insertelement_variable_v96i1(<96 x i8> %a, i8 %b, i32 %index) n
; KNL-NEXT: pushq %rbp
; KNL-NEXT: movq %rsp, %rbp
; KNL-NEXT: andq $-64, %rsp
-; KNL-NEXT: subq $192, %rsp
+; KNL-NEXT: addq $-128, %rsp
; KNL-NEXT: movl 744(%rbp), %eax
; KNL-NEXT: andl $127, %eax
; KNL-NEXT: vmovdqu64 224(%rbp), %zmm0
@@ -1852,7 +1852,7 @@ define i96 @test_insertelement_variable_v96i1(<96 x i8> %a, i8 %b, i32 %index) n
; SKX-NEXT: pushq %rbp
; SKX-NEXT: movq %rsp, %rbp
; SKX-NEXT: andq $-64, %rsp
-; SKX-NEXT: subq $192, %rsp
+; SKX-NEXT: addq $-128, %rsp
; SKX-NEXT: vmovd %edi, %xmm0
; SKX-NEXT: vpinsrb $1, %esi, %xmm0, %xmm0
; SKX-NEXT: vpinsrb $2, %edx, %xmm0, %xmm0
@@ -1934,7 +1934,7 @@ define i128 @test_insertelement_variable_v128i1(<128 x i8> %a, i8 %b, i32 %index
; KNL-NEXT: pushq %rbp
; KNL-NEXT: movq %rsp, %rbp
; KNL-NEXT: andq $-64, %rsp
-; KNL-NEXT: subq $192, %rsp
+; KNL-NEXT: addq $-128, %rsp
; KNL-NEXT: ## kill: def $esi killed $esi def $rsi
; KNL-NEXT: vpxor %xmm2, %xmm2, %xmm2
; KNL-NEXT: vextracti64x4 $1, %zmm0, %ymm3
@@ -2006,7 +2006,7 @@ define i128 @test_insertelement_variable_v128i1(<128 x i8> %a, i8 %b, i32 %index
; SKX-NEXT: pushq %rbp
; SKX-NEXT: movq %rsp, %rbp
; SKX-NEXT: andq $-64, %rsp
-; SKX-NEXT: subq $192, %rsp
+; SKX-NEXT: addq $-128, %rsp
; SKX-NEXT: ## kill: def $esi killed $esi def $rsi
; SKX-NEXT: vptestmb %zmm0, %zmm0, %k0
; SKX-NEXT: vptestmb %zmm1, %zmm1, %k1
diff --git a/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll b/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
index adb8bec6bb5bf8..b77e7095e8aa74 100644
--- a/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
+++ b/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
@@ -12,7 +12,7 @@ define zeroext i8 @test_extractelement_varible_v64i1(<64 x i8> %a, <64 x i8> %b,
; SKX-NEXT: movq %rsp, %rbp
; SKX-NEXT: .cfi_def_cfa_register %rbp
; SKX-NEXT: andq $-64, %rsp
-; SKX-NEXT: addq $-128, %rsp
+; SKX-NEXT: subq $64, %rsp
; SKX-NEXT: ## kill: def $edi killed $edi def $rdi
; SKX-NEXT: vpcmpnleub %zmm1, %zmm0, %k0
; SKX-NEXT: vpmovm2b %k0, %zmm0
diff --git a/llvm/test/CodeGen/X86/avx512-intel-ocl.ll b/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
index 0fa67a264fcf01..310762d1ba9ee7 100644
--- a/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
+++ b/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
@@ -19,7 +19,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
; X32-NEXT: pushl %ebp
; X32-NEXT: movl %esp, %ebp
; X32-NEXT: andl $-64, %esp
-; X32-NEXT: subl $192, %esp
+; X32-NEXT: addl $-128, %esp
; X32-NEXT: vaddps %zmm1, %zmm0, %zmm0
; X32-NEXT: leal {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl %eax, (%esp)
@@ -34,7 +34,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
; WIN32-NEXT: pushl %ebp
; WIN32-NEXT: movl %esp, %ebp
; WIN32-NEXT: andl $-64, %esp
-; WIN32-NEXT: addl $-128, %esp
+; WIN32-NEXT: subl $64, %esp
; WIN32-NEXT: vaddps %zmm1, %zmm0, %zmm0
; WIN32-NEXT: movl %esp, %eax
; WIN32-NEXT: pushl %eax
@@ -67,7 +67,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
; X64-NEXT: pushq %r13
; X64-NEXT: pushq %r12
; X64-NEXT: andq $-64, %rsp
-; X64-NEXT: addq $-128, %rsp
+; X64-NEXT: subq $64, %rsp
; X64-NEXT: vaddps %zmm1, %zmm0, %zmm0
; X64-NEXT: movq %rsp, %rdi
; X64-NEXT: pushq %rbp
@@ -97,7 +97,7 @@ define <16 x float> @testf16_regs(<16 x float> %a, <16 x float> %b) nounwind {
; X32-NEXT: pushl %ebp
; X32-NEXT: movl %esp, %ebp
; X32-NEXT: andl $-64, %esp
-; X32-NEXT: subl $256, %esp ## imm = 0x100
+; X32-NEXT: subl $192, %esp
; X32-NEXT: vmovaps %zmm1, {{[-0-9]+}}(%e{{[sb]}}p) ## 64-byte Spill
; X32-NEXT: vaddps %zmm1, %zmm0, %zmm0
; X32-NEXT: leal {{[0-9]+}}(%esp), %eax
@@ -114,7 +114,7 @@ define <16 x float> @testf16_regs(<16 x float> %a, <16 x float> %b) nounwind {
; WIN32-NEXT: pushl %ebp
; WIN32-NEXT: movl %esp, %ebp
; WIN32-NEXT: andl $-64, %esp
-; WIN32-NEXT: subl $192, %esp
+; WIN32-NEXT: addl $-128, %esp
; WIN32-NEXT: vmovaps %zmm1, (%esp) # 64-byte Spill
; WIN32-NEXT: vaddps %zmm1, %zmm0, %zmm0
; WIN32-NEXT: leal {{[0-9]+}}(%esp), %eax
@@ -150,7 +150,7 @@ define <16 x float> @testf16_regs(<16 x float> %a, <16 x float> %b) nounwind {
; X64-NEXT: pushq %r13
; X64-NEXT: pushq %r12
; X64-NEXT: andq $-64, %rsp
-; X64-NEXT: addq $-128, %rsp
+; X64-NEXT: subq $64, %rsp
; X64-NEXT: vmovaps %zmm1, %zmm16
; X64-NEXT: vaddps %zmm1, %zmm0, %zmm0
; X64-NEXT: movq %rsp, %rdi
diff --git a/llvm/test/CodeGen/X86/avx512fp16-cvt.ll b/llvm/test/CodeGen/X86/avx512fp16-cvt.ll
index cc58bc1e44f37c..9dda386ca88db6 100644
--- a/llvm/test/CodeGen/X86/avx512fp16-cvt.ll
+++ b/llvm/test/CodeGen/X86/avx512fp16-cvt.ll
@@ -819,7 +819,7 @@ define i128 @half_to_s128(half %x) {
; X86-NEXT: .cfi_def_cfa_register %ebp
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: .cfi_offset %esi, -12
; X86-NEXT: movl 8(%ebp), %esi
; X86-NEXT: vmovsh {{.*#+}} xmm0 = mem[0],zero,zero,zero,zero,zero,zero,zero
@@ -922,7 +922,7 @@ define i128 @half_to_u128(half %x) {
; X86-NEXT: .cfi_def_cfa_register %ebp
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: .cfi_offset %esi, -12
; X86-NEXT: movl 8(%ebp), %esi
; X86-NEXT: vmovsh {{.*#+}} xmm0 = mem[0],zero,zero,zero,zero,zero,zero,zero
@@ -1002,7 +1002,7 @@ define fp128 @half_to_f128(half %x) nounwind {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: movl 8(%ebp), %esi
; X86-NEXT: vmovsh {{.*#+}} xmm0 = mem[0],zero,zero,zero,zero,zero,zero,zero
; X86-NEXT: vcvtsh2ss %xmm0, %xmm0, %xmm0
diff --git a/llvm/test/CodeGen/X86/avx512fp16-mov.ll b/llvm/test/CodeGen/X86/avx512fp16-mov.ll
index e2f2688b1d9f3c..24bcfdfb2639bd 100644
--- a/llvm/test/CodeGen/X86/avx512fp16-mov.ll
+++ b/llvm/test/CodeGen/X86/avx512fp16-mov.ll
@@ -1640,7 +1640,7 @@ define half @extract_f16_8(<32 x half> %x, i64 %idx) nounwind {
; X64-NEXT: pushq %rbp
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: andq $-64, %rsp
-; X64-NEXT: addq $-128, %rsp
+; X64-NEXT: subq $64, %rsp
; X64-NEXT: andl $31, %edi
; X64-NEXT: vmovaps %zmm0, (%rsp)
; X64-NEXT: vmovsh {{.*#+}} xmm0 = mem[0],zero,zero,zero,zero,zero,zero,zero
@@ -1654,7 +1654,7 @@ define half @extract_f16_8(<32 x half> %x, i64 %idx) nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-64, %esp
-; X86-NEXT: addl $-128, %esp
+; X86-NEXT: subl $64, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: andl $31, %eax
; X86-NEXT: vmovaps %zmm0, (%esp)
@@ -1673,7 +1673,7 @@ define half @extract_f16_9(<64 x half> %x, i64 %idx) nounwind {
; X64-NEXT: pushq %rbp
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: andq $-64, %rsp
-; X64-NEXT: subq $192, %rsp
+; X64-NEXT: addq $-128, %rsp
; X64-NEXT: andl $63, %edi
; X64-NEXT: vmovaps %zmm1, {{[0-9]+}}(%rsp)
; X64-NEXT: vmovaps %zmm0, (%rsp)
@@ -1688,7 +1688,7 @@ define half @extract_f16_9(<64 x half> %x, i64 %idx) nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-64, %esp
-; X86-NEXT: subl $192, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: andl $63, %eax
; X86-NEXT: vmovaps %zmm1, {{[0-9]+}}(%esp)
diff --git a/llvm/test/CodeGen/X86/bfloat-calling-conv-no-sse2.ll b/llvm/test/CodeGen/X86/bfloat-calling-conv-no-sse2.ll
index f363cad816dfb2..1f59f4bf9d10df 100644
--- a/llvm/test/CodeGen/X86/bfloat-calling-conv-no-sse2.ll
+++ b/llvm/test/CodeGen/X86/bfloat-calling-conv-no-sse2.ll
@@ -949,7 +949,7 @@ define void @call_ret_v16bf16(ptr %ptr) #0 {
; NOSSE-NEXT: pushl %edi
; NOSSE-NEXT: pushl %esi
; NOSSE-NEXT: andl $-32, %esp
-; NOSSE-NEXT: subl $256, %esp # imm = 0x100
+; NOSSE-NEXT: subl $224, %esp
; NOSSE-NEXT: movl 8(%ebp), %esi
; NOSSE-NEXT: movzwl 2(%esi), %eax
; NOSSE-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -1096,7 +1096,7 @@ define void @call_ret_v16bf16(ptr %ptr) #0 {
; SSE-NEXT: pushl %edi
; SSE-NEXT: pushl %esi
; SSE-NEXT: andl $-32, %esp
-; SSE-NEXT: subl $256, %esp # imm = 0x100
+; SSE-NEXT: subl $224, %esp
; SSE-NEXT: movl 8(%ebp), %esi
; SSE-NEXT: movzwl 2(%esi), %eax
; SSE-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
diff --git a/llvm/test/CodeGen/X86/bittest-big-integer.ll b/llvm/test/CodeGen/X86/bittest-big-integer.ll
index 6767cc45a2c5e7..a470f8f1ac7636 100644
--- a/llvm/test/CodeGen/X86/bittest-big-integer.ll
+++ b/llvm/test/CodeGen/X86/bittest-big-integer.ll
@@ -894,7 +894,7 @@ define <8 x i16> @complement_ne_i128_bitcast(ptr %word, i32 %position) nounwind
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $80, %esp
+; X86-NEXT: subl $64, %esp
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: movzwl (%eax), %ecx
; X86-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -1208,7 +1208,7 @@ define i1 @sequence_i128(ptr %word, i32 %pos0, i32 %pos1, i32 %pos2) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $144, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movb 20(%ebp), %ch
; X86-NEXT: movb 12(%ebp), %cl
; X86-NEXT: movl $0, {{[0-9]+}}(%esp)
@@ -1439,7 +1439,7 @@ define i32 @blsr_u512(ptr %word) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $240, %esp
+; X86-NEXT: subl $224, %esp
; X86-NEXT: movl 8(%ebp), %ebx
; X86-NEXT: movl 12(%ebx), %esi
; X86-NEXT: movl 28(%ebx), %eax
diff --git a/llvm/test/CodeGen/X86/dagcombine-tokenfactor-limit-crash.ll b/llvm/test/CodeGen/X86/dagcombine-tokenfactor-limit-crash.ll
index fce59ff485efee..51021cdeb1f751 100644
--- a/llvm/test/CodeGen/X86/dagcombine-tokenfactor-limit-crash.ll
+++ b/llvm/test/CodeGen/X86/dagcombine-tokenfactor-limit-crash.ll
@@ -8,7 +8,7 @@ target triple = "x86_64-unknown-linux-gnu"
; CHECK: pushq %rbx
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $66144, %rsp # imm = 0x10260
+; CHECK-NEXT: subq $66112, %rsp # imm = 0x10240
; CHECK-NEXT: .cfi_offset %rbx, -24
; CHECK-NEXT: movabsq $-868076584853899022, %rax # imm = 0xF3F3F8F201F2F8F2
; CHECK-NEXT: movq %rax, (%rsp)
diff --git a/llvm/test/CodeGen/X86/div-rem-pair-recomposition-signed.ll b/llvm/test/CodeGen/X86/div-rem-pair-recomposition-signed.ll
index 5cfdbb7af38ad3..080dc54677e6e1 100644
--- a/llvm/test/CodeGen/X86/div-rem-pair-recomposition-signed.ll
+++ b/llvm/test/CodeGen/X86/div-rem-pair-recomposition-signed.ll
@@ -151,7 +151,7 @@ define i128 @scalar_i128(i128 %x, i128 %y, ptr %divdst) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $176, %esp
+; X86-NEXT: subl $160, %esp
; X86-NEXT: movl 32(%ebp), %eax
; X86-NEXT: movl 36(%ebp), %ecx
; X86-NEXT: movl %ecx, %edx
diff --git a/llvm/test/CodeGen/X86/div-rem-pair-recomposition-unsigned.ll b/llvm/test/CodeGen/X86/div-rem-pair-recomposition-unsigned.ll
index 1006d0fcae9ed0..b418d952385974 100644
--- a/llvm/test/CodeGen/X86/div-rem-pair-recomposition-unsigned.ll
+++ b/llvm/test/CodeGen/X86/div-rem-pair-recomposition-unsigned.ll
@@ -151,7 +151,7 @@ define i128 @scalar_i128(i128 %x, i128 %y, ptr %divdst) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $160, %esp
+; X86-NEXT: subl $144, %esp
; X86-NEXT: movl 48(%ebp), %ebx
; X86-NEXT: movl 40(%ebp), %ecx
; X86-NEXT: movl 52(%ebp), %esi
diff --git a/llvm/test/CodeGen/X86/extractelement-index.ll b/llvm/test/CodeGen/X86/extractelement-index.ll
index 077351b9718d5f..bbafb0b8c73b8d 100644
--- a/llvm/test/CodeGen/X86/extractelement-index.ll
+++ b/llvm/test/CodeGen/X86/extractelement-index.ll
@@ -454,7 +454,7 @@ define i8 @extractelement_v32i8_var(<32 x i8> %a, i256 %i) nounwind {
; AVX-NEXT: pushq %rbp
; AVX-NEXT: movq %rsp, %rbp
; AVX-NEXT: andq $-32, %rsp
-; AVX-NEXT: subq $64, %rsp
+; AVX-NEXT: subq $32, %rsp
; AVX-NEXT: andl $31, %edi
; AVX-NEXT: vmovaps %ymm0, (%rsp)
; AVX-NEXT: movzbl (%rsp,%rdi), %eax
@@ -498,7 +498,7 @@ define i16 @extractelement_v16i16_var(<16 x i16> %a, i256 %i) nounwind {
; AVX-NEXT: pushq %rbp
; AVX-NEXT: movq %rsp, %rbp
; AVX-NEXT: andq $-32, %rsp
-; AVX-NEXT: subq $64, %rsp
+; AVX-NEXT: subq $32, %rsp
; AVX-NEXT: andl $15, %edi
; AVX-NEXT: vmovaps %ymm0, (%rsp)
; AVX-NEXT: movzwl (%rsp,%rdi,2), %eax
@@ -542,7 +542,7 @@ define i32 @extractelement_v8i32_var(<8 x i32> %a, i256 %i) nounwind {
; AVX-NEXT: pushq %rbp
; AVX-NEXT: movq %rsp, %rbp
; AVX-NEXT: andq $-32, %rsp
-; AVX-NEXT: subq $64, %rsp
+; AVX-NEXT: subq $32, %rsp
; AVX-NEXT: andl $7, %edi
; AVX-NEXT: vmovaps %ymm0, (%rsp)
; AVX-NEXT: movl (%rsp,%rdi,4), %eax
@@ -586,7 +586,7 @@ define i64 @extractelement_v4i64_var(<4 x i64> %a, i256 %i) nounwind {
; AVX-NEXT: pushq %rbp
; AVX-NEXT: movq %rsp, %rbp
; AVX-NEXT: andq $-32, %rsp
-; AVX-NEXT: subq $64, %rsp
+; AVX-NEXT: subq $32, %rsp
; AVX-NEXT: andl $3, %edi
; AVX-NEXT: vmovaps %ymm0, (%rsp)
; AVX-NEXT: movq (%rsp,%rdi,8), %rax
diff --git a/llvm/test/CodeGen/X86/extractelement-load.ll b/llvm/test/CodeGen/X86/extractelement-load.ll
index ce68eebd5b752b..90d478603ab99d 100644
--- a/llvm/test/CodeGen/X86/extractelement-load.ll
+++ b/llvm/test/CodeGen/X86/extractelement-load.ll
@@ -427,7 +427,7 @@ define i32 @main() nounwind {
; X86-SSE2-NEXT: pushl %edi
; X86-SSE2-NEXT: pushl %esi
; X86-SSE2-NEXT: andl $-32, %esp
-; X86-SSE2-NEXT: subl $64, %esp
+; X86-SSE2-NEXT: subl $32, %esp
; X86-SSE2-NEXT: movaps n1+16, %xmm0
; X86-SSE2-NEXT: movaps n1, %xmm1
; X86-SSE2-NEXT: movl zero+4, %ecx
@@ -461,7 +461,7 @@ define i32 @main() nounwind {
; X64-SSSE3-NEXT: pushq %rbp
; X64-SSSE3-NEXT: movq %rsp, %rbp
; X64-SSSE3-NEXT: andq $-32, %rsp
-; X64-SSSE3-NEXT: subq $64, %rsp
+; X64-SSSE3-NEXT: subq $32, %rsp
; X64-SSSE3-NEXT: movq n1 at GOTPCREL(%rip), %rax
; X64-SSSE3-NEXT: movaps (%rax), %xmm0
; X64-SSSE3-NEXT: movaps 16(%rax), %xmm1
@@ -494,7 +494,7 @@ define i32 @main() nounwind {
; X64-AVX-NEXT: pushq %rbp
; X64-AVX-NEXT: movq %rsp, %rbp
; X64-AVX-NEXT: andq $-32, %rsp
-; X64-AVX-NEXT: subq $64, %rsp
+; X64-AVX-NEXT: subq $32, %rsp
; X64-AVX-NEXT: movq n1 at GOTPCREL(%rip), %rax
; X64-AVX-NEXT: vmovaps (%rax), %ymm0
; X64-AVX-NEXT: movl zero+4(%rip), %ecx
diff --git a/llvm/test/CodeGen/X86/fma.ll b/llvm/test/CodeGen/X86/fma.ll
index 6d865628ba7157..a4cb79d53acb75 100644
--- a/llvm/test/CodeGen/X86/fma.ll
+++ b/llvm/test/CodeGen/X86/fma.ll
@@ -1122,8 +1122,8 @@ define <16 x float> @test_v16f32(<16 x float> %a, <16 x float> %b, <16 x float>
; FMACALL32_BDVER2-NEXT: pushl %ebp ## encoding: [0x55]
; FMACALL32_BDVER2-NEXT: movl %esp, %ebp ## encoding: [0x89,0xe5]
; FMACALL32_BDVER2-NEXT: andl $-32, %esp ## encoding: [0x83,0xe4,0xe0]
-; FMACALL32_BDVER2-NEXT: subl $448, %esp ## encoding: [0x81,0xec,0xc0,0x01,0x00,0x00]
-; FMACALL32_BDVER2-NEXT: ## imm = 0x1C0
+; FMACALL32_BDVER2-NEXT: subl $416, %esp ## encoding: [0x81,0xec,0xa0,0x01,0x00,0x00]
+; FMACALL32_BDVER2-NEXT: ## imm = 0x1A0
; FMACALL32_BDVER2-NEXT: vmovaps 56(%ebp), %xmm4 ## encoding: [0xc5,0xf8,0x28,0x65,0x38]
; FMACALL32_BDVER2-NEXT: vmovaps %ymm2, {{[-0-9]+}}(%e{{[sb]}}p) ## 32-byte Spill
; FMACALL32_BDVER2-NEXT: ## encoding: [0xc5,0xfc,0x29,0x94,0x24,0x60,0x01,0x00,0x00]
diff --git a/llvm/test/CodeGen/X86/fp128-libcalls-strict.ll b/llvm/test/CodeGen/X86/fp128-libcalls-strict.ll
index 31af5bc924d61b..150073244767a6 100644
--- a/llvm/test/CodeGen/X86/fp128-libcalls-strict.ll
+++ b/llvm/test/CodeGen/X86/fp128-libcalls-strict.ll
@@ -111,7 +111,7 @@ define fp128 @add(fp128 %x, fp128 %y) nounwind strictfp {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 8(%ebp), %esi
; WIN-X86-NEXT: movl 36(%ebp), %edi
; WIN-X86-NEXT: movl 40(%ebp), %ebx
@@ -240,7 +240,7 @@ define fp128 @sub(fp128 %x, fp128 %y) nounwind strictfp {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 8(%ebp), %esi
; WIN-X86-NEXT: movl 36(%ebp), %edi
; WIN-X86-NEXT: movl 40(%ebp), %ebx
@@ -369,7 +369,7 @@ define fp128 @mul(fp128 %x, fp128 %y) nounwind strictfp {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 8(%ebp), %esi
; WIN-X86-NEXT: movl 36(%ebp), %edi
; WIN-X86-NEXT: movl 40(%ebp), %ebx
@@ -498,7 +498,7 @@ define fp128 @div(fp128 %x, fp128 %y) nounwind strictfp {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 8(%ebp), %esi
; WIN-X86-NEXT: movl 36(%ebp), %edi
; WIN-X86-NEXT: movl 40(%ebp), %ebx
diff --git a/llvm/test/CodeGen/X86/fp128-libcalls.ll b/llvm/test/CodeGen/X86/fp128-libcalls.ll
index 3cb21b5fd59c61..e3ba851952081b 100644
--- a/llvm/test/CodeGen/X86/fp128-libcalls.ll
+++ b/llvm/test/CodeGen/X86/fp128-libcalls.ll
@@ -105,7 +105,7 @@ define dso_local void @Test128Add(fp128 %d1, fp128 %d2) nounwind {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 16(%ebp), %edx
; WIN-X86-NEXT: movl 20(%ebp), %esi
; WIN-X86-NEXT: movl 24(%ebp), %edi
@@ -236,7 +236,7 @@ define dso_local void @Test128_1Add(fp128 %d1) nounwind {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 16(%ebp), %esi
; WIN-X86-NEXT: movl 20(%ebp), %edi
; WIN-X86-NEXT: movl _vf128, %edx
@@ -362,7 +362,7 @@ define dso_local void @Test128Sub(fp128 %d1, fp128 %d2) nounwind {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 16(%ebp), %edx
; WIN-X86-NEXT: movl 20(%ebp), %esi
; WIN-X86-NEXT: movl 24(%ebp), %edi
@@ -493,7 +493,7 @@ define dso_local void @Test128_1Sub(fp128 %d1) nounwind {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 16(%ebp), %esi
; WIN-X86-NEXT: movl 20(%ebp), %edi
; WIN-X86-NEXT: movl _vf128, %edx
@@ -619,7 +619,7 @@ define dso_local void @Test128Mul(fp128 %d1, fp128 %d2) nounwind {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 16(%ebp), %edx
; WIN-X86-NEXT: movl 20(%ebp), %esi
; WIN-X86-NEXT: movl 24(%ebp), %edi
@@ -750,7 +750,7 @@ define dso_local void @Test128_1Mul(fp128 %d1) nounwind {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 16(%ebp), %esi
; WIN-X86-NEXT: movl 20(%ebp), %edi
; WIN-X86-NEXT: movl _vf128, %edx
@@ -876,7 +876,7 @@ define dso_local void @Test128Div(fp128 %d1, fp128 %d2) nounwind {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 16(%ebp), %edx
; WIN-X86-NEXT: movl 20(%ebp), %esi
; WIN-X86-NEXT: movl 24(%ebp), %edi
@@ -1007,7 +1007,7 @@ define dso_local void @Test128_1Div(fp128 %d1) nounwind {
; WIN-X86-NEXT: pushl %edi
; WIN-X86-NEXT: pushl %esi
; WIN-X86-NEXT: andl $-16, %esp
-; WIN-X86-NEXT: subl $80, %esp
+; WIN-X86-NEXT: subl $64, %esp
; WIN-X86-NEXT: movl 16(%ebp), %esi
; WIN-X86-NEXT: movl 20(%ebp), %edi
; WIN-X86-NEXT: movl _vf128, %edx
diff --git a/llvm/test/CodeGen/X86/gep-expanded-vector.ll b/llvm/test/CodeGen/X86/gep-expanded-vector.ll
index 943cd3610c9d32..f0f722ca5d7fed 100644
--- a/llvm/test/CodeGen/X86/gep-expanded-vector.ll
+++ b/llvm/test/CodeGen/X86/gep-expanded-vector.ll
@@ -9,7 +9,7 @@ define ptr @malloc_init_state(<64 x ptr> %tmp, i32 %ind) nounwind {
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $576, %rsp # imm = 0x240
+; CHECK-NEXT: subq $512, %rsp # imm = 0x200
; CHECK-NEXT: # kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vpbroadcastq {{.*#+}} zmm8 = [16,16,16,16,16,16,16,16]
; CHECK-NEXT: vpaddq %zmm8, %zmm0, %zmm0
diff --git a/llvm/test/CodeGen/X86/i128-fp128-abi.ll b/llvm/test/CodeGen/X86/i128-fp128-abi.ll
index 9f385ee2faf4e3..77881225408dd1 100644
--- a/llvm/test/CodeGen/X86/i128-fp128-abi.ll
+++ b/llvm/test/CodeGen/X86/i128-fp128-abi.ll
@@ -657,7 +657,7 @@ define void @call_first_arg(PrimTy %x) nounwind {
; CHECK-MSVC32-NEXT: movl %esp, %ebp
; CHECK-MSVC32-NEXT: pushl %esi
; CHECK-MSVC32-NEXT: andl $-16, %esp
-; CHECK-MSVC32-NEXT: subl $64, %esp
+; CHECK-MSVC32-NEXT: subl $48, %esp
; CHECK-MSVC32-NEXT: movl 8(%ebp), %eax
; CHECK-MSVC32-NEXT: movl 12(%ebp), %ecx
; CHECK-MSVC32-NEXT: movl 16(%ebp), %edx
@@ -793,7 +793,7 @@ define void @call_leading_args(PrimTy %x) nounwind {
; CHECK-MSVC32-NEXT: movl %esp, %ebp
; CHECK-MSVC32-NEXT: pushl %esi
; CHECK-MSVC32-NEXT: andl $-16, %esp
-; CHECK-MSVC32-NEXT: subl $96, %esp
+; CHECK-MSVC32-NEXT: subl $80, %esp
; CHECK-MSVC32-NEXT: movl 8(%ebp), %eax
; CHECK-MSVC32-NEXT: movl 12(%ebp), %ecx
; CHECK-MSVC32-NEXT: movl 16(%ebp), %edx
@@ -960,7 +960,7 @@ define void @call_many_leading_args(PrimTy %x) nounwind {
; CHECK-MSVC32-NEXT: movl %esp, %ebp
; CHECK-MSVC32-NEXT: pushl %esi
; CHECK-MSVC32-NEXT: andl $-16, %esp
-; CHECK-MSVC32-NEXT: subl $112, %esp
+; CHECK-MSVC32-NEXT: subl $96, %esp
; CHECK-MSVC32-NEXT: movl 8(%ebp), %eax
; CHECK-MSVC32-NEXT: movl 12(%ebp), %ecx
; CHECK-MSVC32-NEXT: movl 16(%ebp), %edx
@@ -1108,7 +1108,7 @@ define void @call_trailing_arg(PrimTy %x) nounwind {
; CHECK-MSVC32-NEXT: movl %esp, %ebp
; CHECK-MSVC32-NEXT: pushl %esi
; CHECK-MSVC32-NEXT: andl $-16, %esp
-; CHECK-MSVC32-NEXT: subl $96, %esp
+; CHECK-MSVC32-NEXT: subl $80, %esp
; CHECK-MSVC32-NEXT: movl 8(%ebp), %eax
; CHECK-MSVC32-NEXT: movl 12(%ebp), %ecx
; CHECK-MSVC32-NEXT: movl 16(%ebp), %edx
diff --git a/llvm/test/CodeGen/X86/i128-udiv.ll b/llvm/test/CodeGen/X86/i128-udiv.ll
index b5355c460c999d..543e9d13c50bff 100644
--- a/llvm/test/CodeGen/X86/i128-udiv.ll
+++ b/llvm/test/CodeGen/X86/i128-udiv.ll
@@ -43,7 +43,7 @@ define i128 @test2(i128 %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $144, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movl 32(%ebp), %esi
; X86-NEXT: movl 36(%ebp), %edi
; X86-NEXT: movl 28(%ebp), %ecx
@@ -358,7 +358,7 @@ define i128 @test3(i128 %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $160, %esp
+; X86-NEXT: subl $144, %esp
; X86-NEXT: movl 32(%ebp), %edi
; X86-NEXT: movl 36(%ebp), %edx
; X86-NEXT: movl 28(%ebp), %esi
@@ -688,7 +688,7 @@ define i128 @div_by_7(i128 %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $160, %esp
+; X86-NEXT: subl $144, %esp
; X86-NEXT: movl 32(%ebp), %edi
; X86-NEXT: movl 36(%ebp), %ebx
; X86-NEXT: movl 28(%ebp), %edx
@@ -1014,7 +1014,7 @@ define i128 @div_by_11(i128 %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $160, %esp
+; X86-NEXT: subl $144, %esp
; X86-NEXT: movl 32(%ebp), %edi
; X86-NEXT: movl 36(%ebp), %ebx
; X86-NEXT: movl 28(%ebp), %edx
@@ -1338,7 +1338,7 @@ define i128 @div_by_22(i128 %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $160, %esp
+; X86-NEXT: subl $144, %esp
; X86-NEXT: movl 32(%ebp), %edi
; X86-NEXT: movl 36(%ebp), %ebx
; X86-NEXT: movl 28(%ebp), %edx
@@ -1664,7 +1664,7 @@ define i128 @div_by_56(i128 %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $160, %esp
+; X86-NEXT: subl $144, %esp
; X86-NEXT: movl 32(%ebp), %edi
; X86-NEXT: movl 36(%ebp), %ebx
; X86-NEXT: movl 28(%ebp), %edx
@@ -1990,7 +1990,7 @@ define i128 @rem_by_7(i128 %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $144, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movl 36(%ebp), %ebx
; X86-NEXT: movl 28(%ebp), %eax
; X86-NEXT: testl %ebx, %ebx
@@ -2320,7 +2320,7 @@ define i128 @rem_by_14(i128 %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $144, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movl 36(%ebp), %ebx
; X86-NEXT: movl 28(%ebp), %eax
; X86-NEXT: testl %ebx, %ebx
@@ -2653,7 +2653,7 @@ define i128 @div_by_67(i128 %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $160, %esp
+; X86-NEXT: subl $144, %esp
; X86-NEXT: movl 32(%ebp), %edi
; X86-NEXT: movl 36(%ebp), %ebx
; X86-NEXT: movl 28(%ebp), %edx
diff --git a/llvm/test/CodeGen/X86/i64-mem-copy.ll b/llvm/test/CodeGen/X86/i64-mem-copy.ll
index 9b866183d88ed0..8a89ab3714a6ac 100644
--- a/llvm/test/CodeGen/X86/i64-mem-copy.ll
+++ b/llvm/test/CodeGen/X86/i64-mem-copy.ll
@@ -126,7 +126,7 @@ define void @PR23476(<5 x i64> %in, ptr %out, i32 %index) nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $80, %esp
+; X86-NEXT: subl $64, %esp
; X86-NEXT: movl 52(%ebp), %eax
; X86-NEXT: andl $7, %eax
; X86-NEXT: movl 48(%ebp), %ecx
@@ -147,7 +147,7 @@ define void @PR23476(<5 x i64> %in, ptr %out, i32 %index) nounwind {
; X86AVX-NEXT: pushl %ebp
; X86AVX-NEXT: movl %esp, %ebp
; X86AVX-NEXT: andl $-32, %esp
-; X86AVX-NEXT: subl $96, %esp
+; X86AVX-NEXT: subl $64, %esp
; X86AVX-NEXT: movl 52(%ebp), %eax
; X86AVX-NEXT: andl $7, %eax
; X86AVX-NEXT: movl 48(%ebp), %ecx
diff --git a/llvm/test/CodeGen/X86/inline-sse.ll b/llvm/test/CodeGen/X86/inline-sse.ll
index 87aa882a1f498c..3beba4b407abe2 100644
--- a/llvm/test/CodeGen/X86/inline-sse.ll
+++ b/llvm/test/CodeGen/X86/inline-sse.ll
@@ -11,7 +11,7 @@ define void @nop() nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $32, %esp
+; X86-NEXT: subl $16, %esp
; X86-NEXT: #APP
; X86-NEXT: #NO_APP
; X86-NEXT: movaps %xmm0, (%esp)
diff --git a/llvm/test/CodeGen/X86/insertelement-var-index.ll b/llvm/test/CodeGen/X86/insertelement-var-index.ll
index a4af339b5ebb23..2bc67ce8c2bb93 100644
--- a/llvm/test/CodeGen/X86/insertelement-var-index.ll
+++ b/llvm/test/CodeGen/X86/insertelement-var-index.ll
@@ -862,7 +862,7 @@ define <16 x i8> @arg_i8_v16i8(<16 x i8> %v, i8 %x, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-16, %esp
-; X86AVX2-NEXT: subl $32, %esp
+; X86AVX2-NEXT: subl $16, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $15, %eax
; X86AVX2-NEXT: movzbl 8(%ebp), %ecx
@@ -916,7 +916,7 @@ define <8 x i16> @arg_i16_v8i16(<8 x i16> %v, i16 %x, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-16, %esp
-; X86AVX2-NEXT: subl $32, %esp
+; X86AVX2-NEXT: subl $16, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $7, %eax
; X86AVX2-NEXT: movzwl 8(%ebp), %ecx
@@ -961,7 +961,7 @@ define <4 x i32> @arg_i32_v4i32(<4 x i32> %v, i32 %x, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-16, %esp
-; X86AVX2-NEXT: subl $32, %esp
+; X86AVX2-NEXT: subl $16, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $3, %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
@@ -1008,7 +1008,7 @@ define <2 x i64> @arg_i64_v2i64(<2 x i64> %v, i64 %x, i32 %y) nounwind {
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: pushl %esi
; X86AVX2-NEXT: andl $-16, %esp
-; X86AVX2-NEXT: subl $48, %esp
+; X86AVX2-NEXT: subl $32, %esp
; X86AVX2-NEXT: movl 16(%ebp), %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
; X86AVX2-NEXT: movl 12(%ebp), %edx
@@ -1139,7 +1139,7 @@ define <2 x double> @arg_f64_v2f64(<2 x double> %v, double %x, i32 %y) nounwind
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-16, %esp
-; X86AVX2-NEXT: subl $32, %esp
+; X86AVX2-NEXT: subl $16, %esp
; X86AVX2-NEXT: movl 16(%ebp), %eax
; X86AVX2-NEXT: andl $1, %eax
; X86AVX2-NEXT: vmovsd {{.*#+}} xmm1 = mem[0],zero
@@ -1196,7 +1196,7 @@ define <16 x i8> @load_i8_v16i8(<16 x i8> %v, ptr %p, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-16, %esp
-; X86AVX2-NEXT: subl $32, %esp
+; X86AVX2-NEXT: subl $16, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $15, %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
@@ -1255,7 +1255,7 @@ define <8 x i16> @load_i16_v8i16(<8 x i16> %v, ptr %p, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-16, %esp
-; X86AVX2-NEXT: subl $32, %esp
+; X86AVX2-NEXT: subl $16, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $7, %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
@@ -1304,7 +1304,7 @@ define <4 x i32> @load_i32_v4i32(<4 x i32> %v, ptr %p, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-16, %esp
-; X86AVX2-NEXT: subl $32, %esp
+; X86AVX2-NEXT: subl $16, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $3, %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
@@ -1355,7 +1355,7 @@ define <2 x i64> @load_i64_v2i64(<2 x i64> %v, ptr %p, i32 %y) nounwind {
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: pushl %esi
; X86AVX2-NEXT: andl $-16, %esp
-; X86AVX2-NEXT: subl $48, %esp
+; X86AVX2-NEXT: subl $32, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
; X86AVX2-NEXT: movl (%ecx), %edx
@@ -1493,7 +1493,7 @@ define <2 x double> @load_f64_v2f64(<2 x double> %v, ptr %p, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-16, %esp
-; X86AVX2-NEXT: subl $32, %esp
+; X86AVX2-NEXT: subl $16, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $1, %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
@@ -1526,7 +1526,7 @@ define <32 x i8> @arg_i8_v32i8(<32 x i8> %v, i8 %x, i32 %y) nounwind {
; AVX1OR2-NEXT: pushq %rbp
; AVX1OR2-NEXT: movq %rsp, %rbp
; AVX1OR2-NEXT: andq $-32, %rsp
-; AVX1OR2-NEXT: subq $64, %rsp
+; AVX1OR2-NEXT: subq $32, %rsp
; AVX1OR2-NEXT: # kill: def $esi killed $esi def $rsi
; AVX1OR2-NEXT: vmovaps %ymm0, (%rsp)
; AVX1OR2-NEXT: andl $31, %esi
@@ -1541,7 +1541,7 @@ define <32 x i8> @arg_i8_v32i8(<32 x i8> %v, i8 %x, i32 %y) nounwind {
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-32, %rsp
-; AVX512F-NEXT: subq $64, %rsp
+; AVX512F-NEXT: subq $32, %rsp
; AVX512F-NEXT: # kill: def $esi killed $esi def $rsi
; AVX512F-NEXT: vmovaps %ymm0, (%rsp)
; AVX512F-NEXT: andl $31, %esi
@@ -1563,7 +1563,7 @@ define <32 x i8> @arg_i8_v32i8(<32 x i8> %v, i8 %x, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-32, %esp
-; X86AVX2-NEXT: subl $64, %esp
+; X86AVX2-NEXT: subl $32, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $31, %eax
; X86AVX2-NEXT: movzbl 8(%ebp), %ecx
@@ -1594,7 +1594,7 @@ define <16 x i16> @arg_i16_v16i16(<16 x i16> %v, i16 %x, i32 %y) nounwind {
; AVX1OR2-NEXT: pushq %rbp
; AVX1OR2-NEXT: movq %rsp, %rbp
; AVX1OR2-NEXT: andq $-32, %rsp
-; AVX1OR2-NEXT: subq $64, %rsp
+; AVX1OR2-NEXT: subq $32, %rsp
; AVX1OR2-NEXT: # kill: def $esi killed $esi def $rsi
; AVX1OR2-NEXT: vmovaps %ymm0, (%rsp)
; AVX1OR2-NEXT: andl $15, %esi
@@ -1609,7 +1609,7 @@ define <16 x i16> @arg_i16_v16i16(<16 x i16> %v, i16 %x, i32 %y) nounwind {
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-32, %rsp
-; AVX512F-NEXT: subq $64, %rsp
+; AVX512F-NEXT: subq $32, %rsp
; AVX512F-NEXT: # kill: def $esi killed $esi def $rsi
; AVX512F-NEXT: vmovaps %ymm0, (%rsp)
; AVX512F-NEXT: andl $15, %esi
@@ -1631,7 +1631,7 @@ define <16 x i16> @arg_i16_v16i16(<16 x i16> %v, i16 %x, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-32, %esp
-; X86AVX2-NEXT: subl $64, %esp
+; X86AVX2-NEXT: subl $32, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $15, %eax
; X86AVX2-NEXT: movzwl 8(%ebp), %ecx
@@ -1662,7 +1662,7 @@ define <8 x i32> @arg_i32_v8i32(<8 x i32> %v, i32 %x, i32 %y) nounwind {
; AVX1OR2-NEXT: pushq %rbp
; AVX1OR2-NEXT: movq %rsp, %rbp
; AVX1OR2-NEXT: andq $-32, %rsp
-; AVX1OR2-NEXT: subq $64, %rsp
+; AVX1OR2-NEXT: subq $32, %rsp
; AVX1OR2-NEXT: # kill: def $esi killed $esi def $rsi
; AVX1OR2-NEXT: vmovaps %ymm0, (%rsp)
; AVX1OR2-NEXT: andl $7, %esi
@@ -1684,7 +1684,7 @@ define <8 x i32> @arg_i32_v8i32(<8 x i32> %v, i32 %x, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-32, %esp
-; X86AVX2-NEXT: subl $64, %esp
+; X86AVX2-NEXT: subl $32, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $7, %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
@@ -1715,7 +1715,7 @@ define <4 x i64> @arg_i64_v4i64(<4 x i64> %v, i64 %x, i32 %y) nounwind {
; AVX1OR2-NEXT: pushq %rbp
; AVX1OR2-NEXT: movq %rsp, %rbp
; AVX1OR2-NEXT: andq $-32, %rsp
-; AVX1OR2-NEXT: subq $64, %rsp
+; AVX1OR2-NEXT: subq $32, %rsp
; AVX1OR2-NEXT: # kill: def $esi killed $esi def $rsi
; AVX1OR2-NEXT: vmovaps %ymm0, (%rsp)
; AVX1OR2-NEXT: andl $3, %esi
@@ -1739,7 +1739,7 @@ define <4 x i64> @arg_i64_v4i64(<4 x i64> %v, i64 %x, i32 %y) nounwind {
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: pushl %esi
; X86AVX2-NEXT: andl $-32, %esp
-; X86AVX2-NEXT: subl $96, %esp
+; X86AVX2-NEXT: subl $64, %esp
; X86AVX2-NEXT: movl 16(%ebp), %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
; X86AVX2-NEXT: movl 12(%ebp), %edx
@@ -1859,7 +1859,7 @@ define <4 x double> @arg_f64_v4f64(<4 x double> %v, double %x, i32 %y) nounwind
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-32, %esp
-; X86AVX2-NEXT: subl $64, %esp
+; X86AVX2-NEXT: subl $32, %esp
; X86AVX2-NEXT: movl 16(%ebp), %eax
; X86AVX2-NEXT: andl $3, %eax
; X86AVX2-NEXT: vmovsd {{.*#+}} xmm1 = mem[0],zero
@@ -1891,7 +1891,7 @@ define <32 x i8> @load_i8_v32i8(<32 x i8> %v, ptr %p, i32 %y) nounwind {
; AVX1OR2-NEXT: pushq %rbp
; AVX1OR2-NEXT: movq %rsp, %rbp
; AVX1OR2-NEXT: andq $-32, %rsp
-; AVX1OR2-NEXT: subq $64, %rsp
+; AVX1OR2-NEXT: subq $32, %rsp
; AVX1OR2-NEXT: # kill: def $esi killed $esi def $rsi
; AVX1OR2-NEXT: movzbl (%rdi), %eax
; AVX1OR2-NEXT: vmovaps %ymm0, (%rsp)
@@ -1907,7 +1907,7 @@ define <32 x i8> @load_i8_v32i8(<32 x i8> %v, ptr %p, i32 %y) nounwind {
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-32, %rsp
-; AVX512F-NEXT: subq $64, %rsp
+; AVX512F-NEXT: subq $32, %rsp
; AVX512F-NEXT: # kill: def $esi killed $esi def $rsi
; AVX512F-NEXT: movzbl (%rdi), %eax
; AVX512F-NEXT: vmovaps %ymm0, (%rsp)
@@ -1930,7 +1930,7 @@ define <32 x i8> @load_i8_v32i8(<32 x i8> %v, ptr %p, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-32, %esp
-; X86AVX2-NEXT: subl $64, %esp
+; X86AVX2-NEXT: subl $32, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $31, %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
@@ -1964,7 +1964,7 @@ define <16 x i16> @load_i16_v16i16(<16 x i16> %v, ptr %p, i32 %y) nounwind {
; AVX1OR2-NEXT: pushq %rbp
; AVX1OR2-NEXT: movq %rsp, %rbp
; AVX1OR2-NEXT: andq $-32, %rsp
-; AVX1OR2-NEXT: subq $64, %rsp
+; AVX1OR2-NEXT: subq $32, %rsp
; AVX1OR2-NEXT: # kill: def $esi killed $esi def $rsi
; AVX1OR2-NEXT: movzwl (%rdi), %eax
; AVX1OR2-NEXT: vmovaps %ymm0, (%rsp)
@@ -1980,7 +1980,7 @@ define <16 x i16> @load_i16_v16i16(<16 x i16> %v, ptr %p, i32 %y) nounwind {
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-32, %rsp
-; AVX512F-NEXT: subq $64, %rsp
+; AVX512F-NEXT: subq $32, %rsp
; AVX512F-NEXT: # kill: def $esi killed $esi def $rsi
; AVX512F-NEXT: movzwl (%rdi), %eax
; AVX512F-NEXT: vmovaps %ymm0, (%rsp)
@@ -2003,7 +2003,7 @@ define <16 x i16> @load_i16_v16i16(<16 x i16> %v, ptr %p, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-32, %esp
-; X86AVX2-NEXT: subl $64, %esp
+; X86AVX2-NEXT: subl $32, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $15, %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
@@ -2037,7 +2037,7 @@ define <8 x i32> @load_i32_v8i32(<8 x i32> %v, ptr %p, i32 %y) nounwind {
; AVX1OR2-NEXT: pushq %rbp
; AVX1OR2-NEXT: movq %rsp, %rbp
; AVX1OR2-NEXT: andq $-32, %rsp
-; AVX1OR2-NEXT: subq $64, %rsp
+; AVX1OR2-NEXT: subq $32, %rsp
; AVX1OR2-NEXT: # kill: def $esi killed $esi def $rsi
; AVX1OR2-NEXT: movl (%rdi), %eax
; AVX1OR2-NEXT: vmovaps %ymm0, (%rsp)
@@ -2060,7 +2060,7 @@ define <8 x i32> @load_i32_v8i32(<8 x i32> %v, ptr %p, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-32, %esp
-; X86AVX2-NEXT: subl $64, %esp
+; X86AVX2-NEXT: subl $32, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $7, %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
@@ -2094,7 +2094,7 @@ define <4 x i64> @load_i64_v4i64(<4 x i64> %v, ptr %p, i32 %y) nounwind {
; AVX1OR2-NEXT: pushq %rbp
; AVX1OR2-NEXT: movq %rsp, %rbp
; AVX1OR2-NEXT: andq $-32, %rsp
-; AVX1OR2-NEXT: subq $64, %rsp
+; AVX1OR2-NEXT: subq $32, %rsp
; AVX1OR2-NEXT: # kill: def $esi killed $esi def $rsi
; AVX1OR2-NEXT: movq (%rdi), %rax
; AVX1OR2-NEXT: vmovaps %ymm0, (%rsp)
@@ -2119,7 +2119,7 @@ define <4 x i64> @load_i64_v4i64(<4 x i64> %v, ptr %p, i32 %y) nounwind {
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: pushl %esi
; X86AVX2-NEXT: andl $-32, %esp
-; X86AVX2-NEXT: subl $96, %esp
+; X86AVX2-NEXT: subl $64, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
; X86AVX2-NEXT: movl (%ecx), %edx
@@ -2243,7 +2243,7 @@ define <4 x double> @load_f64_v4f64(<4 x double> %v, ptr %p, i32 %y) nounwind {
; X86AVX2-NEXT: pushl %ebp
; X86AVX2-NEXT: movl %esp, %ebp
; X86AVX2-NEXT: andl $-32, %esp
-; X86AVX2-NEXT: subl $64, %esp
+; X86AVX2-NEXT: subl $32, %esp
; X86AVX2-NEXT: movl 12(%ebp), %eax
; X86AVX2-NEXT: andl $3, %eax
; X86AVX2-NEXT: movl 8(%ebp), %ecx
diff --git a/llvm/test/CodeGen/X86/isel-x87.ll b/llvm/test/CodeGen/X86/isel-x87.ll
index 492faaa19cd660..fdd5e69db91486 100644
--- a/llvm/test/CodeGen/X86/isel-x87.ll
+++ b/llvm/test/CodeGen/X86/isel-x87.ll
@@ -12,7 +12,7 @@ define x86_fp80 @f0(x86_fp80 noundef %a) nounwind {
; GISEL_X86-NEXT: pushl %ebp
; GISEL_X86-NEXT: movl %esp, %ebp
; GISEL_X86-NEXT: andl $-16, %esp
-; GISEL_X86-NEXT: subl $48, %esp
+; GISEL_X86-NEXT: subl $32, %esp
; GISEL_X86-NEXT: fldt 8(%ebp)
; GISEL_X86-NEXT: fldt {{\.?LCPI[0-9]+_[0-9]+}}
; GISEL_X86-NEXT: fxch %st(1)
@@ -30,7 +30,7 @@ define x86_fp80 @f0(x86_fp80 noundef %a) nounwind {
; SDAG_X86-NEXT: pushl %ebp
; SDAG_X86-NEXT: movl %esp, %ebp
; SDAG_X86-NEXT: andl $-16, %esp
-; SDAG_X86-NEXT: subl $48, %esp
+; SDAG_X86-NEXT: subl $32, %esp
; SDAG_X86-NEXT: fldt 8(%ebp)
; SDAG_X86-NEXT: fld %st(0)
; SDAG_X86-NEXT: fstpt {{[0-9]+}}(%esp)
diff --git a/llvm/test/CodeGen/X86/long-double-abi-align.ll b/llvm/test/CodeGen/X86/long-double-abi-align.ll
index 02d68ada9a8d44..3dd92b60c7c8a0 100644
--- a/llvm/test/CodeGen/X86/long-double-abi-align.ll
+++ b/llvm/test/CodeGen/X86/long-double-abi-align.ll
@@ -10,7 +10,7 @@ define void @foo(i32 %0, x86_fp80 %1, i32 %2) nounwind {
; MSVC-NEXT: pushl %ebp
; MSVC-NEXT: movl %esp, %ebp
; MSVC-NEXT: andl $-16, %esp
-; MSVC-NEXT: subl $32, %esp
+; MSVC-NEXT: subl $16, %esp
; MSVC-NEXT: fldt 24(%ebp)
; MSVC-NEXT: fstpt (%esp)
; MSVC-NEXT: leal 8(%ebp), %eax
@@ -34,7 +34,7 @@ define void @foo(i32 %0, x86_fp80 %1, i32 %2) nounwind {
; MINGW-NEXT: pushl %ebp
; MINGW-NEXT: movl %esp, %ebp
; MINGW-NEXT: andl $-16, %esp
-; MINGW-NEXT: subl $32, %esp
+; MINGW-NEXT: subl $16, %esp
; MINGW-NEXT: fldt 12(%ebp)
; MINGW-NEXT: fstpt (%esp)
; MINGW-NEXT: leal 8(%ebp), %eax
diff --git a/llvm/test/CodeGen/X86/matrix-multiply.ll b/llvm/test/CodeGen/X86/matrix-multiply.ll
index f38b769fe49872..7ed7d0b166e0cc 100644
--- a/llvm/test/CodeGen/X86/matrix-multiply.ll
+++ b/llvm/test/CodeGen/X86/matrix-multiply.ll
@@ -3415,7 +3415,7 @@ define <64 x double> @test_mul8x8_f64(<64 x double> %a0, <64 x double> %a1) noun
; AVX1OR2-NEXT: pushq %rbp
; AVX1OR2-NEXT: movq %rsp, %rbp
; AVX1OR2-NEXT: andq $-32, %rsp
-; AVX1OR2-NEXT: subq $448, %rsp # imm = 0x1C0
+; AVX1OR2-NEXT: subq $416, %rsp # imm = 0x1A0
; AVX1OR2-NEXT: vmovapd %ymm2, %ymm12
; AVX1OR2-NEXT: vmovapd %ymm0, (%rsp) # 32-byte Spill
; AVX1OR2-NEXT: movq %rdi, %rax
diff --git a/llvm/test/CodeGen/X86/memset-sse-stack-realignment.ll b/llvm/test/CodeGen/X86/memset-sse-stack-realignment.ll
index a5ecdab880a6a6..00dd58be4f3444 100644
--- a/llvm/test/CodeGen/X86/memset-sse-stack-realignment.ll
+++ b/llvm/test/CodeGen/X86/memset-sse-stack-realignment.ll
@@ -39,7 +39,7 @@ define void @test1(i32 %t) nounwind {
; SSE-NEXT: movl %esp, %ebp
; SSE-NEXT: pushl %esi
; SSE-NEXT: andl $-16, %esp
-; SSE-NEXT: subl $48, %esp
+; SSE-NEXT: subl $32, %esp
; SSE-NEXT: movl %esp, %esi
; SSE-NEXT: movl 8(%ebp), %eax
; SSE-NEXT: xorps %xmm0, %xmm0
@@ -62,7 +62,7 @@ define void @test1(i32 %t) nounwind {
; AVX-NEXT: movl %esp, %ebp
; AVX-NEXT: pushl %esi
; AVX-NEXT: andl $-32, %esp
-; AVX-NEXT: subl $64, %esp
+; AVX-NEXT: subl $32, %esp
; AVX-NEXT: movl %esp, %esi
; AVX-NEXT: movl 8(%ebp), %eax
; AVX-NEXT: vxorps %xmm0, %xmm0, %xmm0
@@ -112,7 +112,7 @@ define void @test2(i32 %t) nounwind {
; SSE-NEXT: movl %esp, %ebp
; SSE-NEXT: pushl %esi
; SSE-NEXT: andl $-16, %esp
-; SSE-NEXT: subl $32, %esp
+; SSE-NEXT: subl $16, %esp
; SSE-NEXT: movl %esp, %esi
; SSE-NEXT: movl 8(%ebp), %eax
; SSE-NEXT: xorps %xmm0, %xmm0
@@ -134,7 +134,7 @@ define void @test2(i32 %t) nounwind {
; AVX-NEXT: movl %esp, %ebp
; AVX-NEXT: pushl %esi
; AVX-NEXT: andl $-16, %esp
-; AVX-NEXT: subl $32, %esp
+; AVX-NEXT: subl $16, %esp
; AVX-NEXT: movl %esp, %esi
; AVX-NEXT: movl 8(%ebp), %eax
; AVX-NEXT: vxorps %xmm0, %xmm0, %xmm0
diff --git a/llvm/test/CodeGen/X86/mingw-alloca.ll b/llvm/test/CodeGen/X86/mingw-alloca.ll
index fe60f103a9587a..493f8323e07d09 100644
--- a/llvm/test/CodeGen/X86/mingw-alloca.ll
+++ b/llvm/test/CodeGen/X86/mingw-alloca.ll
@@ -22,12 +22,12 @@ entry:
; COFF: andl $-16, %esp
; COFF: pushl %eax
; COFF: calll __alloca
-; COFF: movl 8012(%esp), %eax
+; COFF: movl 7996(%esp), %eax
; ELF: foo2:
; ELF: andl $-16, %esp
; ELF: pushl %eax
; ELF: calll _alloca
-; ELF: movl 8012(%esp), %eax
+; ELF: movl 7996(%esp), %eax
%A2 = alloca [2000 x i32], align 16 ; <ptr> [#uses=1]
%A2.sub = getelementptr [2000 x i32], ptr %A2, i32 0, i32 0 ; <ptr> [#uses=1]
call void @bar2( ptr %A2.sub, i32 %N )
diff --git a/llvm/test/CodeGen/X86/mmx-arith.ll b/llvm/test/CodeGen/X86/mmx-arith.ll
index 8f97d2652bc531..832e46b1367a6d 100644
--- a/llvm/test/CodeGen/X86/mmx-arith.ll
+++ b/llvm/test/CodeGen/X86/mmx-arith.ll
@@ -243,7 +243,7 @@ define void @test2(ptr %A, ptr %B) nounwind {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-8, %esp
-; X86-NEXT: subl $16, %esp
+; X86-NEXT: subl $8, %esp
; X86-NEXT: movl 12(%ebp), %ecx
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: movq {{.*#+}} xmm0 = mem[0],zero
diff --git a/llvm/test/CodeGen/X86/mmx-intrinsics.ll b/llvm/test/CodeGen/X86/mmx-intrinsics.ll
index a7b6ed416622ef..c7e311dbbb3fa8 100644
--- a/llvm/test/CodeGen/X86/mmx-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/mmx-intrinsics.ll
@@ -2864,7 +2864,7 @@ define void @test23(<1 x i64> %d, <1 x i64> %n, ptr %p) nounwind optsize ssp {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: pushl %edi
; X86-NEXT: andl $-8, %esp
-; X86-NEXT: subl $24, %esp
+; X86-NEXT: subl $16, %esp
; X86-NEXT: movl 16(%ebp), %eax
; X86-NEXT: movl 20(%ebp), %ecx
; X86-NEXT: movl %ecx, {{[0-9]+}}(%esp)
diff --git a/llvm/test/CodeGen/X86/musttail-varargs.ll b/llvm/test/CodeGen/X86/musttail-varargs.ll
index 65cd1edd92e318..4db78c6fed599f 100644
--- a/llvm/test/CodeGen/X86/musttail-varargs.ll
+++ b/llvm/test/CodeGen/X86/musttail-varargs.ll
@@ -245,7 +245,7 @@ define void @f_thunk(ptr %this, ...) {
; X86-NOSSE-NEXT: movl %esp, %ebp
; X86-NOSSE-NEXT: pushl %esi
; X86-NOSSE-NEXT: andl $-16, %esp
-; X86-NOSSE-NEXT: subl $32, %esp
+; X86-NOSSE-NEXT: subl $16, %esp
; X86-NOSSE-NEXT: movl 8(%ebp), %esi
; X86-NOSSE-NEXT: leal 12(%ebp), %eax
; X86-NOSSE-NEXT: movl %eax, (%esp)
@@ -264,7 +264,7 @@ define void @f_thunk(ptr %this, ...) {
; X86-SSE-NEXT: movl %esp, %ebp
; X86-SSE-NEXT: pushl %esi
; X86-SSE-NEXT: andl $-16, %esp
-; X86-SSE-NEXT: subl $80, %esp
+; X86-SSE-NEXT: subl $64, %esp
; X86-SSE-NEXT: movaps %xmm2, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
; X86-SSE-NEXT: movaps %xmm1, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
; X86-SSE-NEXT: movaps %xmm0, (%esp) # 16-byte Spill
diff --git a/llvm/test/CodeGen/X86/nontemporal-loads-2.ll b/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
index 71c23d88085ec8..a1e351683bd7db 100644
--- a/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
+++ b/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
@@ -204,7 +204,7 @@ define <4 x double> @test_v4f64_align16(ptr %src) nounwind {
; AVX-NEXT: pushq %rbp
; AVX-NEXT: movq %rsp, %rbp
; AVX-NEXT: andq $-32, %rsp
-; AVX-NEXT: subq $64, %rsp
+; AVX-NEXT: subq $32, %rsp
; AVX-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX-NEXT: vmovntdqa (%rdi), %xmm0
@@ -235,7 +235,7 @@ define <8 x float> @test_v8f32_align16(ptr %src) nounwind {
; AVX-NEXT: pushq %rbp
; AVX-NEXT: movq %rsp, %rbp
; AVX-NEXT: andq $-32, %rsp
-; AVX-NEXT: subq $64, %rsp
+; AVX-NEXT: subq $32, %rsp
; AVX-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX-NEXT: vmovntdqa (%rdi), %xmm0
@@ -266,7 +266,7 @@ define <4 x i64> @test_v4i64_align16(ptr %src) nounwind {
; AVX-NEXT: pushq %rbp
; AVX-NEXT: movq %rsp, %rbp
; AVX-NEXT: andq $-32, %rsp
-; AVX-NEXT: subq $64, %rsp
+; AVX-NEXT: subq $32, %rsp
; AVX-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX-NEXT: vmovntdqa (%rdi), %xmm0
@@ -297,7 +297,7 @@ define <8 x i32> @test_v8i32_align16(ptr %src) nounwind {
; AVX-NEXT: pushq %rbp
; AVX-NEXT: movq %rsp, %rbp
; AVX-NEXT: andq $-32, %rsp
-; AVX-NEXT: subq $64, %rsp
+; AVX-NEXT: subq $32, %rsp
; AVX-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX-NEXT: vmovntdqa (%rdi), %xmm0
@@ -328,7 +328,7 @@ define <16 x i16> @test_v16i16_align16(ptr %src) nounwind {
; AVX-NEXT: pushq %rbp
; AVX-NEXT: movq %rsp, %rbp
; AVX-NEXT: andq $-32, %rsp
-; AVX-NEXT: subq $64, %rsp
+; AVX-NEXT: subq $32, %rsp
; AVX-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX-NEXT: vmovntdqa (%rdi), %xmm0
@@ -359,7 +359,7 @@ define <32 x i8> @test_v32i8_align16(ptr %src) nounwind {
; AVX-NEXT: pushq %rbp
; AVX-NEXT: movq %rsp, %rbp
; AVX-NEXT: andq $-32, %rsp
-; AVX-NEXT: subq $64, %rsp
+; AVX-NEXT: subq $32, %rsp
; AVX-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX-NEXT: vmovntdqa (%rdi), %xmm0
@@ -570,7 +570,7 @@ define <8 x double> @test_v8f64_align16(ptr %src) nounwind {
; AVX1-NEXT: pushq %rbp
; AVX1-NEXT: movq %rsp, %rbp
; AVX1-NEXT: andq $-32, %rsp
-; AVX1-NEXT: subq $96, %rsp
+; AVX1-NEXT: subq $64, %rsp
; AVX1-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX1-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX1-NEXT: vmovntdqa (%rdi), %xmm0
@@ -590,7 +590,7 @@ define <8 x double> @test_v8f64_align16(ptr %src) nounwind {
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX2-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX2-NEXT: vmovntdqa (%rdi), %xmm0
@@ -610,7 +610,7 @@ define <8 x double> @test_v8f64_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -649,7 +649,7 @@ define <16 x float> @test_v16f32_align16(ptr %src) nounwind {
; AVX1-NEXT: pushq %rbp
; AVX1-NEXT: movq %rsp, %rbp
; AVX1-NEXT: andq $-32, %rsp
-; AVX1-NEXT: subq $96, %rsp
+; AVX1-NEXT: subq $64, %rsp
; AVX1-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX1-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX1-NEXT: vmovntdqa (%rdi), %xmm0
@@ -669,7 +669,7 @@ define <16 x float> @test_v16f32_align16(ptr %src) nounwind {
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX2-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX2-NEXT: vmovntdqa (%rdi), %xmm0
@@ -689,7 +689,7 @@ define <16 x float> @test_v16f32_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -728,7 +728,7 @@ define <8 x i64> @test_v8i64_align16(ptr %src) nounwind {
; AVX1-NEXT: pushq %rbp
; AVX1-NEXT: movq %rsp, %rbp
; AVX1-NEXT: andq $-32, %rsp
-; AVX1-NEXT: subq $96, %rsp
+; AVX1-NEXT: subq $64, %rsp
; AVX1-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX1-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX1-NEXT: vmovntdqa (%rdi), %xmm0
@@ -748,7 +748,7 @@ define <8 x i64> @test_v8i64_align16(ptr %src) nounwind {
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX2-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX2-NEXT: vmovntdqa (%rdi), %xmm0
@@ -768,7 +768,7 @@ define <8 x i64> @test_v8i64_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -807,7 +807,7 @@ define <16 x i32> @test_v16i32_align16(ptr %src) nounwind {
; AVX1-NEXT: pushq %rbp
; AVX1-NEXT: movq %rsp, %rbp
; AVX1-NEXT: andq $-32, %rsp
-; AVX1-NEXT: subq $96, %rsp
+; AVX1-NEXT: subq $64, %rsp
; AVX1-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX1-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX1-NEXT: vmovntdqa (%rdi), %xmm0
@@ -827,7 +827,7 @@ define <16 x i32> @test_v16i32_align16(ptr %src) nounwind {
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX2-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX2-NEXT: vmovntdqa (%rdi), %xmm0
@@ -847,7 +847,7 @@ define <16 x i32> @test_v16i32_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -886,7 +886,7 @@ define <32 x i16> @test_v32i16_align16(ptr %src) nounwind {
; AVX1-NEXT: pushq %rbp
; AVX1-NEXT: movq %rsp, %rbp
; AVX1-NEXT: andq $-32, %rsp
-; AVX1-NEXT: subq $96, %rsp
+; AVX1-NEXT: subq $64, %rsp
; AVX1-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX1-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX1-NEXT: vmovntdqa (%rdi), %xmm0
@@ -906,7 +906,7 @@ define <32 x i16> @test_v32i16_align16(ptr %src) nounwind {
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX2-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX2-NEXT: vmovntdqa (%rdi), %xmm0
@@ -926,7 +926,7 @@ define <32 x i16> @test_v32i16_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -965,7 +965,7 @@ define <64 x i8> @test_v64i8_align16(ptr %src) nounwind {
; AVX1-NEXT: pushq %rbp
; AVX1-NEXT: movq %rsp, %rbp
; AVX1-NEXT: andq $-32, %rsp
-; AVX1-NEXT: subq $96, %rsp
+; AVX1-NEXT: subq $64, %rsp
; AVX1-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX1-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX1-NEXT: vmovntdqa (%rdi), %xmm0
@@ -985,7 +985,7 @@ define <64 x i8> @test_v64i8_align16(ptr %src) nounwind {
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vmovntdqa 16(%rdi), %xmm0
; AVX2-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX2-NEXT: vmovntdqa (%rdi), %xmm0
@@ -1005,7 +1005,7 @@ define <64 x i8> @test_v64i8_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -1060,7 +1060,7 @@ define <8 x double> @test_v8f64_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
@@ -1111,7 +1111,7 @@ define <16 x float> @test_v16f32_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
@@ -1162,7 +1162,7 @@ define <8 x i64> @test_v8i64_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
@@ -1213,7 +1213,7 @@ define <16 x i32> @test_v16i32_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
@@ -1264,7 +1264,7 @@ define <32 x i16> @test_v32i16_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
@@ -1315,7 +1315,7 @@ define <64 x i8> @test_v64i8_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
diff --git a/llvm/test/CodeGen/X86/nosse-vector.ll b/llvm/test/CodeGen/X86/nosse-vector.ll
index 9807d1b09d8ef6..1af1aad58a376d 100644
--- a/llvm/test/CodeGen/X86/nosse-vector.ll
+++ b/llvm/test/CodeGen/X86/nosse-vector.ll
@@ -143,7 +143,7 @@ define void @sitofp_4i64_4f32_mem(ptr %p0, ptr %p1) nounwind {
; X32-NEXT: pushl %edi
; X32-NEXT: pushl %esi
; X32-NEXT: andl $-8, %esp
-; X32-NEXT: subl $48, %esp
+; X32-NEXT: subl $40, %esp
; X32-NEXT: movl 8(%ebp), %edx
; X32-NEXT: movl 24(%edx), %eax
; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
diff --git a/llvm/test/CodeGen/X86/packus.ll b/llvm/test/CodeGen/X86/packus.ll
index 7de7caff18d6b0..8e5c383de8db2a 100644
--- a/llvm/test/CodeGen/X86/packus.ll
+++ b/llvm/test/CodeGen/X86/packus.ll
@@ -901,7 +901,7 @@ define <32 x i16> @_mm512_packus_epi32_manual(<16 x i32> %a, <16 x i32> %b) noun
; X86-SSE2-NEXT: pushl %ebp
; X86-SSE2-NEXT: movl %esp, %ebp
; X86-SSE2-NEXT: andl $-16, %esp
-; X86-SSE2-NEXT: subl $64, %esp
+; X86-SSE2-NEXT: subl $48, %esp
; X86-SSE2-NEXT: movdqa %xmm2, %xmm4
; X86-SSE2-NEXT: movdqa %xmm1, %xmm6
; X86-SSE2-NEXT: movdqa %xmm0, %xmm2
diff --git a/llvm/test/CodeGen/X86/pr216504.ll b/llvm/test/CodeGen/X86/pr216504.ll
index 6cf96095260886..098d974a8ac27e 100644
--- a/llvm/test/CodeGen/X86/pr216504.ll
+++ b/llvm/test/CodeGen/X86/pr216504.ll
@@ -14,7 +14,7 @@ define void @f25() nounwind {
; X64-NEXT: pushq %rbp
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: andq $-32, %rsp
-; X64-NEXT: subq $96, %rsp
+; X64-NEXT: subq $64, %rsp
; X64-NEXT: movq g2 at GOTPCREL(%rip), %rax
; X64-NEXT: movzwl (%rax), %ecx
; X64-NEXT: andl $7, %ecx
@@ -33,7 +33,7 @@ define void @f25() nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-32, %esp
-; X86-NEXT: subl $96, %esp
+; X86-NEXT: subl $64, %esp
; X86-NEXT: movzwl g2, %eax
; X86-NEXT: andl $7, %eax
; X86-NEXT: vmovss {{.*#+}} xmm0 = [60,0,0,0]
diff --git a/llvm/test/CodeGen/X86/pr32284.ll b/llvm/test/CodeGen/X86/pr32284.ll
index 745ec17a63f435..cf2adb8e2231e9 100644
--- a/llvm/test/CodeGen/X86/pr32284.ll
+++ b/llvm/test/CodeGen/X86/pr32284.ll
@@ -746,7 +746,7 @@ define void @f3() #0 {
; X86-O0-NEXT: .cfi_def_cfa_register %ebp
; X86-O0-NEXT: pushl %esi
; X86-O0-NEXT: andl $-8, %esp
-; X86-O0-NEXT: subl $16, %esp
+; X86-O0-NEXT: subl $8, %esp
; X86-O0-NEXT: .cfi_offset %esi, -12
; X86-O0-NEXT: movl var_13, %ecx
; X86-O0-NEXT: movl %ecx, %eax
diff --git a/llvm/test/CodeGen/X86/pr34080-2.ll b/llvm/test/CodeGen/X86/pr34080-2.ll
index 279373a7aab3fa..3f228dc2241300 100644
--- a/llvm/test/CodeGen/X86/pr34080-2.ll
+++ b/llvm/test/CodeGen/X86/pr34080-2.ll
@@ -12,7 +12,7 @@ define void @computeJD(ptr) nounwind {
; CHECK-NEXT: pushl %edi
; CHECK-NEXT: pushl %esi
; CHECK-NEXT: andl $-8, %esp
-; CHECK-NEXT: subl $40, %esp
+; CHECK-NEXT: subl $32, %esp
; CHECK-NEXT: movl 8(%ebp), %ebx
; CHECK-NEXT: movl 8(%ebx), %esi
; CHECK-NEXT: xorl %eax, %eax
diff --git a/llvm/test/CodeGen/X86/pr34592.ll b/llvm/test/CodeGen/X86/pr34592.ll
index e0423d595303ec..7ddb62c1851d39 100644
--- a/llvm/test/CodeGen/X86/pr34592.ll
+++ b/llvm/test/CodeGen/X86/pr34592.ll
@@ -8,7 +8,7 @@ define <16 x i64> @pluto(<16 x i64> %arg, <16 x i64> %arg1, <16 x i64> %arg2, <1
; CHECK-O0-NEXT: pushq %rbp
; CHECK-O0-NEXT: movq %rsp, %rbp
; CHECK-O0-NEXT: andq $-32, %rsp
-; CHECK-O0-NEXT: subq $64, %rsp
+; CHECK-O0-NEXT: subq $32, %rsp
; CHECK-O0-NEXT: vmovaps %ymm4, %ymm10
; CHECK-O0-NEXT: vmovaps %ymm3, %ymm9
; CHECK-O0-NEXT: vmovaps %ymm2, (%rsp) # 32-byte Spill
diff --git a/llvm/test/CodeGen/X86/pr34653.ll b/llvm/test/CodeGen/X86/pr34653.ll
index d46cd2091856eb..ca6e47dccfc5fc 100644
--- a/llvm/test/CodeGen/X86/pr34653.ll
+++ b/llvm/test/CodeGen/X86/pr34653.ll
@@ -12,7 +12,7 @@ define void @pr34653() {
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: .cfi_def_cfa_register %rbp
; CHECK-NEXT: andq $-512, %rsp # imm = 0xFE00
-; CHECK-NEXT: subq $1024, %rsp # imm = 0x400
+; CHECK-NEXT: subq $512, %rsp # imm = 0x200
; CHECK-NEXT: movq %rsp, %rdi
; CHECK-NEXT: callq test at PLT
; CHECK-NEXT: vmovsd {{.*#+}} xmm0 = mem[0],zero
diff --git a/llvm/test/CodeGen/X86/pr38539.ll b/llvm/test/CodeGen/X86/pr38539.ll
index e0e6f6bec48764..722154e3bdd114 100644
--- a/llvm/test/CodeGen/X86/pr38539.ll
+++ b/llvm/test/CodeGen/X86/pr38539.ll
@@ -22,7 +22,7 @@ define void @f() nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $96, %esp
+; X86-NEXT: subl $80, %esp
; X86-NEXT: movl {{[0-9]+}}(%esp), %edx
; X86-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X86-NEXT: movl {{[0-9]+}}(%esp), %esi
diff --git a/llvm/test/CodeGen/X86/pr43866.ll b/llvm/test/CodeGen/X86/pr43866.ll
index 20eedbc942277a..28505e8daa4959 100644
--- a/llvm/test/CodeGen/X86/pr43866.ll
+++ b/llvm/test/CodeGen/X86/pr43866.ll
@@ -12,7 +12,7 @@ define dso_local void @test() {
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: .cfi_def_cfa_register %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: vmovsd {{.*#+}} xmm0 = mem[0],zero
; CHECK-NEXT: vpcmpeqd %xmm1, %xmm1, %xmm1
; CHECK-NEXT: vshufps {{.*#+}} xmm2 = xmm1[1,0],xmm0[1,0]
diff --git a/llvm/test/CodeGen/X86/pr50782.ll b/llvm/test/CodeGen/X86/pr50782.ll
index 591a33446d4e30..b9e0e1ab17e165 100644
--- a/llvm/test/CodeGen/X86/pr50782.ll
+++ b/llvm/test/CodeGen/X86/pr50782.ll
@@ -20,7 +20,7 @@ define void @h(float %i) {
; CHECK-NEXT: .cfi_def_cfa_register %ebp
; CHECK-NEXT: pushl %esi
; CHECK-NEXT: andl $-16, %esp
-; CHECK-NEXT: subl $32, %esp
+; CHECK-NEXT: subl $16, %esp
; CHECK-NEXT: movl %esp, %esi
; CHECK-NEXT: .cfi_offset %esi, -12
; CHECK-NEXT: flds 8(%ebp)
diff --git a/llvm/test/CodeGen/X86/sdiv_fix.ll b/llvm/test/CodeGen/X86/sdiv_fix.ll
index 392bc83d9d5d8d..70ae75906f071a 100644
--- a/llvm/test/CodeGen/X86/sdiv_fix.ll
+++ b/llvm/test/CodeGen/X86/sdiv_fix.ll
@@ -307,7 +307,7 @@ define i64 @func5(i64 %x, i64 %y) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movl 8(%ebp), %ecx
; X86-NEXT: movl 12(%ebp), %edi
; X86-NEXT: movl 16(%ebp), %eax
diff --git a/llvm/test/CodeGen/X86/sdiv_fix_sat.ll b/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
index eac2bb7ebe9620..ecdb82078c55b7 100644
--- a/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
+++ b/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
@@ -370,7 +370,7 @@ define i64 @func5(i64 %x, i64 %y) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: addl $-128, %esp
+; X86-NEXT: subl $112, %esp
; X86-NEXT: movl 8(%ebp), %esi
; X86-NEXT: movl 12(%ebp), %edi
; X86-NEXT: movl 16(%ebp), %ecx
@@ -799,7 +799,7 @@ define <4 x i32> @vec(<4 x i32> %x, <4 x i32> %y) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $240, %esp
+; X86-NEXT: subl $224, %esp
; X86-NEXT: movl 20(%ebp), %esi
; X86-NEXT: movl 36(%ebp), %ebx
; X86-NEXT: movl 16(%ebp), %ecx
diff --git a/llvm/test/CodeGen/X86/shift-i128.ll b/llvm/test/CodeGen/X86/shift-i128.ll
index 6c9cdb1b62ca4a..d100c6f69536e0 100644
--- a/llvm/test/CodeGen/X86/shift-i128.ll
+++ b/llvm/test/CodeGen/X86/shift-i128.ll
@@ -15,7 +15,7 @@ define void @test_lshr_i128(i128 %x, i128 %a, ptr nocapture %r) nounwind {
; i686-NEXT: pushl %edi
; i686-NEXT: pushl %esi
; i686-NEXT: andl $-16, %esp
-; i686-NEXT: subl $48, %esp
+; i686-NEXT: subl $32, %esp
; i686-NEXT: movzbl 24(%ebp), %ecx
; i686-NEXT: movl 8(%ebp), %eax
; i686-NEXT: movl 12(%ebp), %edx
@@ -81,7 +81,7 @@ define void @test_ashr_i128(i128 %x, i128 %a, ptr nocapture %r) nounwind {
; i686-NEXT: pushl %edi
; i686-NEXT: pushl %esi
; i686-NEXT: andl $-16, %esp
-; i686-NEXT: subl $48, %esp
+; i686-NEXT: subl $32, %esp
; i686-NEXT: movzbl 24(%ebp), %ecx
; i686-NEXT: movl 8(%ebp), %eax
; i686-NEXT: movl 12(%ebp), %edx
@@ -149,7 +149,7 @@ define void @test_shl_i128(i128 %x, i128 %a, ptr nocapture %r) nounwind {
; i686-NEXT: pushl %edi
; i686-NEXT: pushl %esi
; i686-NEXT: andl $-16, %esp
-; i686-NEXT: subl $48, %esp
+; i686-NEXT: subl $32, %esp
; i686-NEXT: movzbl 24(%ebp), %ecx
; i686-NEXT: movl 8(%ebp), %eax
; i686-NEXT: movl 12(%ebp), %edx
@@ -278,7 +278,7 @@ define void @test_lshr_v2i128(<2 x i128> %x, <2 x i128> %a, ptr nocapture %r) no
; i686-NEXT: pushl %edi
; i686-NEXT: pushl %esi
; i686-NEXT: andl $-16, %esp
-; i686-NEXT: subl $112, %esp
+; i686-NEXT: subl $96, %esp
; i686-NEXT: movl 40(%ebp), %edx
; i686-NEXT: movl 24(%ebp), %eax
; i686-NEXT: movl 28(%ebp), %ecx
@@ -405,7 +405,7 @@ define void @test_ashr_v2i128(<2 x i128> %x, <2 x i128> %a, ptr nocapture %r) no
; i686-NEXT: pushl %edi
; i686-NEXT: pushl %esi
; i686-NEXT: andl $-16, %esp
-; i686-NEXT: subl $112, %esp
+; i686-NEXT: subl $96, %esp
; i686-NEXT: movl 40(%ebp), %edx
; i686-NEXT: movl 24(%ebp), %eax
; i686-NEXT: movl 28(%ebp), %ecx
@@ -538,7 +538,7 @@ define void @test_shl_v2i128(<2 x i128> %x, <2 x i128> %a, ptr nocapture %r) nou
; i686-NEXT: pushl %edi
; i686-NEXT: pushl %esi
; i686-NEXT: andl $-16, %esp
-; i686-NEXT: addl $-128, %esp
+; i686-NEXT: subl $112, %esp
; i686-NEXT: movl 40(%ebp), %edi
; i686-NEXT: movl 24(%ebp), %eax
; i686-NEXT: movl 28(%ebp), %ecx
@@ -1004,7 +1004,7 @@ define i128 @shift_i128_limited_shamt_no_nuw(i128 noundef %a, i32 noundef %b) no
; i686-NEXT: pushl %edi
; i686-NEXT: pushl %esi
; i686-NEXT: andl $-16, %esp
-; i686-NEXT: subl $48, %esp
+; i686-NEXT: subl $32, %esp
; i686-NEXT: movzbl 40(%ebp), %eax
; i686-NEXT: movl 24(%ebp), %ecx
; i686-NEXT: movl 28(%ebp), %edx
@@ -1076,7 +1076,7 @@ define i128 @shift_i128_limited_shamt_unknown_lhs(i128 noundef %a, i32 noundef %
; i686-NEXT: pushl %edi
; i686-NEXT: pushl %esi
; i686-NEXT: andl $-16, %esp
-; i686-NEXT: subl $48, %esp
+; i686-NEXT: subl $32, %esp
; i686-NEXT: movl 24(%ebp), %eax
; i686-NEXT: movl 28(%ebp), %edx
; i686-NEXT: movl 32(%ebp), %esi
diff --git a/llvm/test/CodeGen/X86/shift-i256.ll b/llvm/test/CodeGen/X86/shift-i256.ll
index 4933ef441de6b1..68ee0715aaff7a 100644
--- a/llvm/test/CodeGen/X86/shift-i256.ll
+++ b/llvm/test/CodeGen/X86/shift-i256.ll
@@ -171,7 +171,7 @@ define i256 @shl_i256(i256 %a0, i256 %a1) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movzbl 44(%ebp), %ecx
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: movl 16(%ebp), %edx
@@ -406,7 +406,7 @@ define i256 @lshr_i256(i256 %a0, i256 %a1) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movzbl 44(%ebp), %ecx
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: movl 16(%ebp), %edx
@@ -644,7 +644,7 @@ define i256 @ashr_i256(i256 %a0, i256 %a1) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movzbl 44(%ebp), %ecx
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: movl 16(%ebp), %edx
@@ -878,7 +878,7 @@ define i256 @shl_i256_load(ptr %p0, i256 %a1) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movl 12(%ebp), %ecx
; X86-NEXT: movl (%ecx), %eax
; X86-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -1112,7 +1112,7 @@ define i256 @lshr_i256_load(ptr %p0, i256 %a1) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movl 12(%ebp), %ecx
; X86-NEXT: movl (%ecx), %eax
; X86-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -1367,7 +1367,7 @@ define i256 @ashr_i256_load(ptr %p0, i256 %a1) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: movl (%eax), %ecx
; X86-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -2041,7 +2041,7 @@ define i256 @shl_1_i256(i256 %a0) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movzbl 12(%ebp), %ecx
; X86-NEXT: movl $0, {{[0-9]+}}(%esp)
; X86-NEXT: movl $0, {{[0-9]+}}(%esp)
@@ -2259,7 +2259,7 @@ define i256 @lshr_signbit_i256(i256 %a0) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movzbl 12(%ebp), %ecx
; X86-NEXT: movl $0, {{[0-9]+}}(%esp)
; X86-NEXT: movl $0, {{[0-9]+}}(%esp)
@@ -2472,7 +2472,7 @@ define i256 @ashr_signbit_i256(i256 %a0) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movzbl 12(%ebp), %ecx
; X86-NEXT: movl $-1, {{[0-9]+}}(%esp)
; X86-NEXT: movl $-1, {{[0-9]+}}(%esp)
@@ -2696,7 +2696,7 @@ define i256 @shl_allbits_i256(i256 %a0) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movzbl 12(%ebp), %ecx
; X86-NEXT: movl $-1, {{[0-9]+}}(%esp)
; X86-NEXT: movl $-1, {{[0-9]+}}(%esp)
@@ -2915,7 +2915,7 @@ define i256 @lshr_allbits_i256(i256 %a0) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $112, %esp
+; X86-NEXT: subl $96, %esp
; X86-NEXT: movzbl 12(%ebp), %ecx
; X86-NEXT: movl $0, {{[0-9]+}}(%esp)
; X86-NEXT: movl $0, {{[0-9]+}}(%esp)
@@ -3292,7 +3292,7 @@ define i64 @lshr_extract_load_i256_i64(ptr %p0, i256 %a1) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $96, %esp
+; X86-NEXT: subl $80, %esp
; X86-NEXT: movl 8(%ebp), %ecx
; X86-NEXT: movl (%ecx), %eax
; X86-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -3422,7 +3422,7 @@ define i64 @ashr_extract_load_i256_i64(ptr %p0, i256 %a1) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $96, %esp
+; X86-NEXT: subl $80, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: movl (%eax), %ecx
; X86-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -3560,7 +3560,7 @@ define i64 @ashr_extract_idx_load_i256_i64(ptr %p0, i256 %a1) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $96, %esp
+; X86-NEXT: subl $80, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: movl (%eax), %ecx
; X86-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
diff --git a/llvm/test/CodeGen/X86/shuffle-combine-crash-4.ll b/llvm/test/CodeGen/X86/shuffle-combine-crash-4.ll
index 8ea99b313838a8..fa9861d69bf0b0 100644
--- a/llvm/test/CodeGen/X86/shuffle-combine-crash-4.ll
+++ b/llvm/test/CodeGen/X86/shuffle-combine-crash-4.ll
@@ -13,7 +13,7 @@ define void @infiloop() {
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: .cfi_def_cfa_register %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: movabsq $506097522914230528, %rax # imm = 0x706050403020100
; CHECK-NEXT: movq %rax, test24_id5239(%rip)
; CHECK-NEXT: vmovaps {{.*#+}} ymm0 = [4,1,6,7,6,7,2,3,6,7,4,5,6,7,2,3,6,7,2,3,2,3,2,3,4,5,4,5,2,3,0,1]
diff --git a/llvm/test/CodeGen/X86/sse-intel-ocl.ll b/llvm/test/CodeGen/X86/sse-intel-ocl.ll
index b2de7545ff5f51..f9f1c593dce78e 100644
--- a/llvm/test/CodeGen/X86/sse-intel-ocl.ll
+++ b/llvm/test/CodeGen/X86/sse-intel-ocl.ll
@@ -13,7 +13,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
; WIN32-NEXT: pushl %ebp
; WIN32-NEXT: movl %esp, %ebp
; WIN32-NEXT: andl $-16, %esp
-; WIN32-NEXT: subl $80, %esp
+; WIN32-NEXT: subl $64, %esp
; WIN32-NEXT: movups 72(%ebp), %xmm4
; WIN32-NEXT: movups 8(%ebp), %xmm3
; WIN32-NEXT: addps %xmm4, %xmm3
@@ -91,7 +91,7 @@ define <16 x float> @testf16_regs(<16 x float> %a, <16 x float> %b) nounwind {
; WIN32-NEXT: pushl %ebp
; WIN32-NEXT: movl %esp, %ebp
; WIN32-NEXT: andl $-16, %esp
-; WIN32-NEXT: subl $80, %esp
+; WIN32-NEXT: subl $64, %esp
; WIN32-NEXT: movups 72(%ebp), %xmm6
; WIN32-NEXT: movups 8(%ebp), %xmm3
; WIN32-NEXT: movups 56(%ebp), %xmm7
diff --git a/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll b/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
index 2e2e78a6da51e2..913e54873fc621 100644
--- a/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
+++ b/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
@@ -2754,7 +2754,7 @@ define void @test_mm_storeh_pi(ptr %a0, <4 x float> %a1) nounwind {
; X86-SSE1-NEXT: pushl %ebp # encoding: [0x55]
; X86-SSE1-NEXT: movl %esp, %ebp # encoding: [0x89,0xe5]
; X86-SSE1-NEXT: andl $-16, %esp # encoding: [0x83,0xe4,0xf0]
-; X86-SSE1-NEXT: subl $32, %esp # encoding: [0x83,0xec,0x20]
+; X86-SSE1-NEXT: subl $16, %esp # encoding: [0x83,0xec,0x10]
; X86-SSE1-NEXT: movl 8(%ebp), %eax # encoding: [0x8b,0x45,0x08]
; X86-SSE1-NEXT: movaps %xmm0, (%esp) # encoding: [0x0f,0x29,0x04,0x24]
; X86-SSE1-NEXT: movl {{[0-9]+}}(%esp), %ecx # encoding: [0x8b,0x4c,0x24,0x08]
@@ -2861,7 +2861,7 @@ define void @test_mm_storel_pi(ptr %a0, <4 x float> %a1) nounwind {
; X86-SSE1-NEXT: pushl %ebp # encoding: [0x55]
; X86-SSE1-NEXT: movl %esp, %ebp # encoding: [0x89,0xe5]
; X86-SSE1-NEXT: andl $-16, %esp # encoding: [0x83,0xe4,0xf0]
-; X86-SSE1-NEXT: subl $32, %esp # encoding: [0x83,0xec,0x20]
+; X86-SSE1-NEXT: subl $16, %esp # encoding: [0x83,0xec,0x10]
; X86-SSE1-NEXT: movl 8(%ebp), %eax # encoding: [0x8b,0x45,0x08]
; X86-SSE1-NEXT: movaps %xmm0, (%esp) # encoding: [0x0f,0x29,0x04,0x24]
; X86-SSE1-NEXT: movl (%esp), %ecx # encoding: [0x8b,0x0c,0x24]
diff --git a/llvm/test/CodeGen/X86/sse-regcall.ll b/llvm/test/CodeGen/X86/sse-regcall.ll
index 03b9e123eea486..e7c02b58e4d967 100644
--- a/llvm/test/CodeGen/X86/sse-regcall.ll
+++ b/llvm/test/CodeGen/X86/sse-regcall.ll
@@ -75,7 +75,7 @@ define x86_regcallcc <16 x float> @testf32_inp(<16 x float> %a, <16 x float> %b,
; WIN32-NEXT: pushl %ebp
; WIN32-NEXT: movl %esp, %ebp
; WIN32-NEXT: andl $-16, %esp
-; WIN32-NEXT: subl $32, %esp
+; WIN32-NEXT: subl $16, %esp
; WIN32-NEXT: movaps %xmm7, (%esp) # 16-byte Spill
; WIN32-NEXT: movaps %xmm6, %xmm7
; WIN32-NEXT: movaps %xmm5, %xmm6
@@ -364,7 +364,7 @@ define x86_regcallcc <32 x float> @testf32_stack(<32 x float> %a, <32 x float> %
; WIN32-NEXT: pushl %ebp
; WIN32-NEXT: movl %esp, %ebp
; WIN32-NEXT: andl $-16, %esp
-; WIN32-NEXT: subl $48, %esp
+; WIN32-NEXT: subl $32, %esp
; WIN32-NEXT: movaps %xmm7, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
; WIN32-NEXT: movaps %xmm6, (%esp) # 16-byte Spill
; WIN32-NEXT: movaps %xmm5, %xmm6
diff --git a/llvm/test/CodeGen/X86/sse-regcall4.ll b/llvm/test/CodeGen/X86/sse-regcall4.ll
index 6f964f0a88ea3d..2aae7af393c10d 100644
--- a/llvm/test/CodeGen/X86/sse-regcall4.ll
+++ b/llvm/test/CodeGen/X86/sse-regcall4.ll
@@ -75,7 +75,7 @@ define x86_regcallcc <16 x float> @testf32_inp(<16 x float> %a, <16 x float> %b,
; WIN32-NEXT: pushl %ebp
; WIN32-NEXT: movl %esp, %ebp
; WIN32-NEXT: andl $-16, %esp
-; WIN32-NEXT: subl $32, %esp
+; WIN32-NEXT: subl $16, %esp
; WIN32-NEXT: movaps %xmm7, (%esp) # 16-byte Spill
; WIN32-NEXT: movaps %xmm6, %xmm7
; WIN32-NEXT: movaps %xmm5, %xmm6
@@ -363,7 +363,7 @@ define x86_regcallcc <32 x float> @testf32_stack(<32 x float> %a, <32 x float> %
; WIN32-NEXT: pushl %ebp
; WIN32-NEXT: movl %esp, %ebp
; WIN32-NEXT: andl $-16, %esp
-; WIN32-NEXT: subl $48, %esp
+; WIN32-NEXT: subl $32, %esp
; WIN32-NEXT: movaps %xmm7, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
; WIN32-NEXT: movaps %xmm6, (%esp) # 16-byte Spill
; WIN32-NEXT: movaps %xmm5, %xmm6
diff --git a/llvm/test/CodeGen/X86/stack-clash-small-alloc-medium-align.ll b/llvm/test/CodeGen/X86/stack-clash-small-alloc-medium-align.ll
index 01a1cb136a49ac..af82f350245867 100644
--- a/llvm/test/CodeGen/X86/stack-clash-small-alloc-medium-align.ll
+++ b/llvm/test/CodeGen/X86/stack-clash-small-alloc-medium-align.ll
@@ -93,7 +93,7 @@ define i32 @foo4(i64 %i) local_unnamed_addr #0 {
; CHECK-NEXT: .cfi_def_cfa_register %rbp
; CHECK-NEXT: pushq %rbx
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $896, %rsp # imm = 0x380
+; CHECK-NEXT: subq $832, %rsp # imm = 0x340
; CHECK-NEXT: movq %rsp, %rbx
; CHECK-NEXT: .cfi_offset %rbx, -24
; CHECK-NEXT: movl $1, (%rbx,%rdi,4)
diff --git a/llvm/test/CodeGen/X86/stack-realign-local-padding.ll b/llvm/test/CodeGen/X86/stack-realign-local-padding.ll
new file mode 100644
index 00000000000000..2d1d084dfba0d6
--- /dev/null
+++ b/llvm/test/CodeGen/X86/stack-realign-local-padding.ll
@@ -0,0 +1,104 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+
+; PEI realigns the stack but leaves locals relative to the incoming SP,
+; so padding between saved registers and locals must not be allocated.
+; GPR/XMM saves must not affect it, two-block objects still need both.
+
+declare void @use(ptr)
+
+define void @one_block() nounwind {
+; CHECK-LABEL: one_block:
+; CHECK: # %bb.0:
+; CHECK-NEXT: pushq %rbp
+; CHECK-NEXT: movq %rsp, %rbp
+; CHECK-NEXT: andq $-64, %rsp
+; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: movq %rsp, %rdi
+; CHECK-NEXT: callq use at PLT
+; CHECK-NEXT: movq %rbp, %rsp
+; CHECK-NEXT: popq %rbp
+; CHECK-NEXT: retq
+ %a = alloca [64 x i8], align 64
+ call void @use(ptr %a)
+ ret void
+}
+
+define void @two_blocks() nounwind {
+; CHECK-LABEL: two_blocks:
+; CHECK: # %bb.0:
+; CHECK-NEXT: pushq %rbp
+; CHECK-NEXT: movq %rsp, %rbp
+; CHECK-NEXT: andq $-64, %rsp
+; CHECK-NEXT: addq $-128, %rsp
+; CHECK-NEXT: movq %rsp, %rdi
+; CHECK-NEXT: callq use at PLT
+; CHECK-NEXT: movq %rbp, %rsp
+; CHECK-NEXT: popq %rbp
+; CHECK-NEXT: retq
+ %a = alloca [65 x i8], align 64
+ call void @use(ptr %a)
+ ret void
+}
+
+define void @gpr_csrs() nounwind {
+; CHECK-LABEL: gpr_csrs:
+; CHECK: # %bb.0:
+; CHECK-NEXT: pushq %rbp
+; CHECK-NEXT: movq %rsp, %rbp
+; CHECK-NEXT: pushq %r13
+; CHECK-NEXT: pushq %r12
+; CHECK-NEXT: pushq %rbx
+; CHECK-NEXT: andq $-64, %rsp
+; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: #APP
+; CHECK-NEXT: #NO_APP
+; CHECK-NEXT: movq %rsp, %rdi
+; CHECK-NEXT: callq use at PLT
+; CHECK-NEXT: leaq -24(%rbp), %rsp
+; CHECK-NEXT: popq %rbx
+; CHECK-NEXT: popq %r12
+; CHECK-NEXT: popq %r13
+; CHECK-NEXT: popq %rbp
+; CHECK-NEXT: retq
+ %a = alloca [64 x i8], align 64
+ call void asm sideeffect "", "~{rbx},~{r12},~{r13}"()
+ call void @use(ptr %a)
+ ret void
+}
+
+define x86_regcallcc void @xmm_csrs() nounwind {
+; CHECK-LABEL: xmm_csrs:
+; CHECK: # %bb.0:
+; CHECK-NEXT: pushq %rbp
+; CHECK-NEXT: movq %rsp, %rbp
+; CHECK-NEXT: andq $-64, %rsp
+; CHECK-NEXT: subq $192, %rsp
+; CHECK-NEXT: movaps %xmm15, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT: movaps %xmm14, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT: movaps %xmm13, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT: movaps %xmm12, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT: movaps %xmm11, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT: movaps %xmm10, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT: movaps %xmm9, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT: movaps %xmm8, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT: #APP
+; CHECK-NEXT: #NO_APP
+; CHECK-NEXT: movq %rsp, %rdi
+; CHECK-NEXT: callq use at PLT
+; CHECK-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm8 # 16-byte Reload
+; CHECK-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm9 # 16-byte Reload
+; CHECK-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm10 # 16-byte Reload
+; CHECK-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm11 # 16-byte Reload
+; CHECK-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm12 # 16-byte Reload
+; CHECK-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm13 # 16-byte Reload
+; CHECK-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm14 # 16-byte Reload
+; CHECK-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm15 # 16-byte Reload
+; CHECK-NEXT: movq %rbp, %rsp
+; CHECK-NEXT: popq %rbp
+; CHECK-NEXT: retq
+ %a = alloca [64 x i8], align 64
+ call void asm sideeffect "", "~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15}"()
+ call void @use(ptr %a)
+ ret void
+}
diff --git a/llvm/test/CodeGen/X86/statepoint-no-realign-stack.ll b/llvm/test/CodeGen/X86/statepoint-no-realign-stack.ll
index b531f00b4be5b9..87e8a2094da4c7 100644
--- a/llvm/test/CodeGen/X86/statepoint-no-realign-stack.ll
+++ b/llvm/test/CodeGen/X86/statepoint-no-realign-stack.ll
@@ -20,7 +20,7 @@ define void @can_realign(ptr %p) {
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: .cfi_def_cfa_register %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: vmovaps (%rdi), %ymm0
; CHECK-NEXT: vmovaps %ymm0, (%rsp)
; CHECK-NEXT: vzeroupper
@@ -65,7 +65,7 @@ define <4 x ptr addrspace(1)> @spillfill_can_realign(<4 x ptr addrspace(1)> %obj
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: .cfi_def_cfa_register %rbp
; CHECK-NEXT: andq $-32, %rsp
-; CHECK-NEXT: subq $64, %rsp
+; CHECK-NEXT: subq $32, %rsp
; CHECK-NEXT: vmovaps %ymm0, (%rsp)
; CHECK-NEXT: vzeroupper
; CHECK-NEXT: callq do_safepoint at PLT
diff --git a/llvm/test/CodeGen/X86/sttni.ll b/llvm/test/CodeGen/X86/sttni.ll
index 39cbee54737c38..7cbfd46c390793 100644
--- a/llvm/test/CodeGen/X86/sttni.ll
+++ b/llvm/test/CodeGen/X86/sttni.ll
@@ -58,7 +58,7 @@ define i32 @pcmpestri_reg_diff_i8(<16 x i8> %lhs, i32 %lhs_len, <16 x i8> %rhs,
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: movl 12(%ebp), %edx
; X86-NEXT: pcmpestri $24, %xmm1, %xmm0
@@ -184,7 +184,7 @@ define i32 @pcmpestri_mem_diff_i8(ptr %lhs_ptr, i32 %lhs_len, ptr %rhs_ptr, i32
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: movl 20(%ebp), %edx
; X86-NEXT: movl 16(%ebp), %ecx
@@ -304,7 +304,7 @@ define i32 @pcmpestri_reg_diff_i16(<8 x i16> %lhs, i32 %lhs_len, <8 x i16> %rhs,
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: movl 12(%ebp), %edx
; X86-NEXT: pcmpestri $24, %xmm1, %xmm0
@@ -438,7 +438,7 @@ define i32 @pcmpestri_mem_diff_i16(ptr %lhs_ptr, i32 %lhs_len, ptr %rhs_ptr, i32
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: movl 20(%ebp), %edx
; X86-NEXT: movl 16(%ebp), %ecx
@@ -558,7 +558,7 @@ define i32 @pcmpistri_reg_diff_i8(<16 x i8> %lhs, <16 x i8> %rhs) nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: movdqa %xmm0, (%esp)
; X86-NEXT: andl $15, %ecx
; X86-NEXT: movzbl (%esp,%ecx), %eax
@@ -657,7 +657,7 @@ define i32 @pcmpistri_mem_diff_i8(ptr %lhs_ptr, ptr %rhs_ptr) nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: movl 8(%ebp), %ecx
; X86-NEXT: movdqu (%ecx), %xmm1
@@ -772,7 +772,7 @@ define i32 @pcmpistri_reg_diff_i16(<8 x i16> %lhs, <8 x i16> %rhs) nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: movdqa %xmm0, (%esp)
; X86-NEXT: addl %ecx, %ecx
; X86-NEXT: andl $14, %ecx
@@ -879,7 +879,7 @@ define i32 @pcmpistri_mem_diff_i16(ptr %lhs_ptr, ptr %rhs_ptr) nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $48, %esp
+; X86-NEXT: subl $32, %esp
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: movl 8(%ebp), %ecx
; X86-NEXT: movdqu (%ecx), %xmm1
diff --git a/llvm/test/CodeGen/X86/udiv_fix.ll b/llvm/test/CodeGen/X86/udiv_fix.ll
index 8c3698c535b005..584121d34d6f38 100644
--- a/llvm/test/CodeGen/X86/udiv_fix.ll
+++ b/llvm/test/CodeGen/X86/udiv_fix.ll
@@ -143,7 +143,7 @@ define i64 @func5(i64 %x, i64 %y) nounwind {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $80, %esp
+; X86-NEXT: subl $64, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: movl 12(%ebp), %ecx
; X86-NEXT: movl 16(%ebp), %edx
diff --git a/llvm/test/CodeGen/X86/udiv_fix_sat.ll b/llvm/test/CodeGen/X86/udiv_fix_sat.ll
index 659ab7e69e9aeb..81bf490e24bec9 100644
--- a/llvm/test/CodeGen/X86/udiv_fix_sat.ll
+++ b/llvm/test/CodeGen/X86/udiv_fix_sat.ll
@@ -176,7 +176,7 @@ define i64 @func5(i64 %x, i64 %y) nounwind {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $80, %esp
+; X86-NEXT: subl $64, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: movl 12(%ebp), %ecx
; X86-NEXT: movl 16(%ebp), %edx
diff --git a/llvm/test/CodeGen/X86/var-permute-256.ll b/llvm/test/CodeGen/X86/var-permute-256.ll
index 0ca24b4c85f90e..5155077f41df09 100644
--- a/llvm/test/CodeGen/X86/var-permute-256.ll
+++ b/llvm/test/CodeGen/X86/var-permute-256.ll
@@ -2010,7 +2010,7 @@ define <4 x i64> @PR50356(<4 x i64> %0, <4 x i32> %1, <4 x i64> %2) unnamed_addr
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $64, %rsp
+; AVX2-NEXT: subq $32, %rsp
; AVX2-NEXT: vmovd %xmm1, %eax
; AVX2-NEXT: vmovaps %ymm0, (%rsp)
; AVX2-NEXT: andl $3, %eax
@@ -2031,7 +2031,7 @@ define <4 x i64> @PR50356(<4 x i64> %0, <4 x i32> %1, <4 x i64> %2) unnamed_addr
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-32, %rsp
-; AVX512-NEXT: subq $64, %rsp
+; AVX512-NEXT: subq $32, %rsp
; AVX512-NEXT: # kill: def $ymm2 killed $ymm2 def $zmm2
; AVX512-NEXT: vmovd %xmm1, %eax
; AVX512-NEXT: vmovaps %ymm0, (%rsp)
@@ -2055,7 +2055,7 @@ define <4 x i64> @PR50356(<4 x i64> %0, <4 x i32> %1, <4 x i64> %2) unnamed_addr
; AVX512VL-NEXT: pushq %rbp
; AVX512VL-NEXT: movq %rsp, %rbp
; AVX512VL-NEXT: andq $-32, %rsp
-; AVX512VL-NEXT: subq $64, %rsp
+; AVX512VL-NEXT: subq $32, %rsp
; AVX512VL-NEXT: vmovd %xmm1, %eax
; AVX512VL-NEXT: vmovaps %ymm0, (%rsp)
; AVX512VL-NEXT: andl $3, %eax
diff --git a/llvm/test/CodeGen/X86/var-permute-512.ll b/llvm/test/CodeGen/X86/var-permute-512.ll
index 6b893f834f50de..9511b6c35f6160 100644
--- a/llvm/test/CodeGen/X86/var-permute-512.ll
+++ b/llvm/test/CodeGen/X86/var-permute-512.ll
@@ -97,7 +97,7 @@ define <32 x i16> @var_shuffle_v32i16(<32 x i16> %v, <32 x i16> %indices) nounwi
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-64, %rsp
-; AVX512F-NEXT: addq $-128, %rsp
+; AVX512F-NEXT: subq $64, %rsp
; AVX512F-NEXT: vextracti128 $1, %ymm1, %xmm2
; AVX512F-NEXT: vextracti32x4 $2, %zmm1, %xmm3
; AVX512F-NEXT: vextracti32x4 $3, %zmm1, %xmm4
@@ -326,7 +326,7 @@ define <64 x i8> @var_shuffle_v64i8(<64 x i8> %v, <64 x i8> %indices) nounwind {
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-64, %rsp
-; AVX512F-NEXT: addq $-128, %rsp
+; AVX512F-NEXT: subq $64, %rsp
; AVX512F-NEXT: vextracti128 $1, %ymm1, %xmm2
; AVX512F-NEXT: vextracti32x4 $2, %zmm1, %xmm3
; AVX512F-NEXT: vextracti32x4 $3, %zmm1, %xmm4
@@ -551,7 +551,7 @@ define <64 x i8> @var_shuffle_v64i8(<64 x i8> %v, <64 x i8> %indices) nounwind {
; AVX512BW-NEXT: pushq %rbp
; AVX512BW-NEXT: movq %rsp, %rbp
; AVX512BW-NEXT: andq $-64, %rsp
-; AVX512BW-NEXT: addq $-128, %rsp
+; AVX512BW-NEXT: subq $64, %rsp
; AVX512BW-NEXT: vextracti128 $1, %ymm1, %xmm2
; AVX512BW-NEXT: vextracti32x4 $2, %zmm1, %xmm3
; AVX512BW-NEXT: vextracti32x4 $3, %zmm1, %xmm4
@@ -1064,7 +1064,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-64, %rsp
-; AVX512F-NEXT: addq $-128, %rsp
+; AVX512F-NEXT: subq $64, %rsp
; AVX512F-NEXT: # kill: def $esi killed $esi def $rsi
; AVX512F-NEXT: vpbroadcastd %esi, %zmm2
; AVX512F-NEXT: vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm1 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
@@ -1315,7 +1315,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
; AVX512BW-NEXT: pushq %rbp
; AVX512BW-NEXT: movq %rsp, %rbp
; AVX512BW-NEXT: andq $-64, %rsp
-; AVX512BW-NEXT: addq $-128, %rsp
+; AVX512BW-NEXT: subq $64, %rsp
; AVX512BW-NEXT: # kill: def $esi killed $esi def $rsi
; AVX512BW-NEXT: vpbroadcastd %esi, %zmm2
; AVX512BW-NEXT: vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm1 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
@@ -1566,7 +1566,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
; AVX512VBMI-NEXT: pushq %rbp
; AVX512VBMI-NEXT: movq %rsp, %rbp
; AVX512VBMI-NEXT: andq $-64, %rsp
-; AVX512VBMI-NEXT: addq $-128, %rsp
+; AVX512VBMI-NEXT: subq $64, %rsp
; AVX512VBMI-NEXT: # kill: def $esi killed $esi def $rsi
; AVX512VBMI-NEXT: vpbroadcastd %esi, %zmm1
; AVX512VBMI-NEXT: vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm1, %zmm2 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
diff --git a/llvm/test/CodeGen/X86/vec-strict-128.ll b/llvm/test/CodeGen/X86/vec-strict-128.ll
index 84c0ff61cf9b7d..7e898213c1e6fb 100644
--- a/llvm/test/CodeGen/X86/vec-strict-128.ll
+++ b/llvm/test/CodeGen/X86/vec-strict-128.ll
@@ -348,7 +348,7 @@ define <2 x double> @f14(<2 x double> %a, <2 x double> %b, <2 x double> %c) #0 {
; SSE-X86-NEXT: movl %esp, %ebp
; SSE-X86-NEXT: .cfi_def_cfa_register %ebp
; SSE-X86-NEXT: andl $-16, %esp
-; SSE-X86-NEXT: subl $112, %esp
+; SSE-X86-NEXT: subl $96, %esp
; SSE-X86-NEXT: movaps %xmm2, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
; SSE-X86-NEXT: movaps %xmm1, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
; SSE-X86-NEXT: movaps %xmm0, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
diff --git a/llvm/test/CodeGen/X86/vec-strict-fptoint-256.ll b/llvm/test/CodeGen/X86/vec-strict-fptoint-256.ll
index 179e8ad69672b1..8cbf576677d1af 100644
--- a/llvm/test/CodeGen/X86/vec-strict-fptoint-256.ll
+++ b/llvm/test/CodeGen/X86/vec-strict-fptoint-256.ll
@@ -375,7 +375,7 @@ define <4 x i64> @strict_vector_fptoui_v4f64_to_v4i64(<4 x double> %a) #0 {
; AVX512F-32-NEXT: .cfi_def_cfa_register %ebp
; AVX512F-32-NEXT: pushl %ebx
; AVX512F-32-NEXT: andl $-8, %esp
-; AVX512F-32-NEXT: subl $40, %esp
+; AVX512F-32-NEXT: subl $32, %esp
; AVX512F-32-NEXT: .cfi_offset %ebx, -12
; AVX512F-32-NEXT: vextractf128 $1, %ymm0, %xmm2
; AVX512F-32-NEXT: vshufpd {{.*#+}} xmm3 = xmm2[1,0]
@@ -468,7 +468,7 @@ define <4 x i64> @strict_vector_fptoui_v4f64_to_v4i64(<4 x double> %a) #0 {
; AVX512VL-32-NEXT: .cfi_def_cfa_register %ebp
; AVX512VL-32-NEXT: pushl %ebx
; AVX512VL-32-NEXT: andl $-8, %esp
-; AVX512VL-32-NEXT: subl $40, %esp
+; AVX512VL-32-NEXT: subl $32, %esp
; AVX512VL-32-NEXT: .cfi_offset %ebx, -12
; AVX512VL-32-NEXT: vextractf128 $1, %ymm0, %xmm2
; AVX512VL-32-NEXT: vshufpd {{.*#+}} xmm3 = xmm2[1,0]
@@ -906,7 +906,7 @@ define <4 x i64> @strict_vector_fptoui_v4f32_to_v4i64(<4 x float> %a) #0 {
; AVX512F-32-NEXT: .cfi_def_cfa_register %ebp
; AVX512F-32-NEXT: pushl %ebx
; AVX512F-32-NEXT: andl $-8, %esp
-; AVX512F-32-NEXT: subl $40, %esp
+; AVX512F-32-NEXT: subl $32, %esp
; AVX512F-32-NEXT: .cfi_offset %ebx, -12
; AVX512F-32-NEXT: vshufps {{.*#+}} xmm2 = xmm0[3,3,3,3]
; AVX512F-32-NEXT: vmovss {{.*#+}} xmm1 = [9.22337203E+18,0.0E+0,0.0E+0,0.0E+0]
@@ -999,7 +999,7 @@ define <4 x i64> @strict_vector_fptoui_v4f32_to_v4i64(<4 x float> %a) #0 {
; AVX512VL-32-NEXT: .cfi_def_cfa_register %ebp
; AVX512VL-32-NEXT: pushl %ebx
; AVX512VL-32-NEXT: andl $-8, %esp
-; AVX512VL-32-NEXT: subl $40, %esp
+; AVX512VL-32-NEXT: subl $32, %esp
; AVX512VL-32-NEXT: .cfi_offset %ebx, -12
; AVX512VL-32-NEXT: vshufps {{.*#+}} xmm2 = xmm0[3,3,3,3]
; AVX512VL-32-NEXT: vmovss {{.*#+}} xmm1 = [9.22337203E+18,0.0E+0,0.0E+0,0.0E+0]
diff --git a/llvm/test/CodeGen/X86/vec_ins_extract-1.ll b/llvm/test/CodeGen/X86/vec_ins_extract-1.ll
index cf70d5d7f1edfd..bdb24fac73047f 100644
--- a/llvm/test/CodeGen/X86/vec_ins_extract-1.ll
+++ b/llvm/test/CodeGen/X86/vec_ins_extract-1.ll
@@ -11,7 +11,7 @@ define i32 @t0(i32 inreg %t7, <4 x i32> inreg %t8) nounwind {
; X32-NEXT: pushl %ebp
; X32-NEXT: movl %esp, %ebp
; X32-NEXT: andl $-16, %esp
-; X32-NEXT: subl $32, %esp
+; X32-NEXT: subl $16, %esp
; X32-NEXT: andl $3, %eax
; X32-NEXT: movaps %xmm0, (%esp)
; X32-NEXT: movl $76, (%esp,%eax,4)
@@ -39,7 +39,7 @@ define i32 @t1(i32 inreg %t7, <4 x i32> inreg %t8) nounwind {
; X32-NEXT: pushl %ebp
; X32-NEXT: movl %esp, %ebp
; X32-NEXT: andl $-16, %esp
-; X32-NEXT: subl $32, %esp
+; X32-NEXT: subl $16, %esp
; X32-NEXT: andl $3, %eax
; X32-NEXT: movl $76, %ecx
; X32-NEXT: pinsrd $0, %ecx, %xmm0
@@ -69,7 +69,7 @@ define <4 x i32> @t2(i32 inreg %t7, <4 x i32> inreg %t8) nounwind {
; X32-NEXT: pushl %ebp
; X32-NEXT: movl %esp, %ebp
; X32-NEXT: andl $-16, %esp
-; X32-NEXT: subl $32, %esp
+; X32-NEXT: subl $16, %esp
; X32-NEXT: andl $3, %eax
; X32-NEXT: movdqa %xmm0, (%esp)
; X32-NEXT: pinsrd $0, (%esp,%eax,4), %xmm0
@@ -95,7 +95,7 @@ define <4 x i32> @t3(i32 inreg %t7, <4 x i32> inreg %t8) nounwind {
; X32-NEXT: pushl %ebp
; X32-NEXT: movl %esp, %ebp
; X32-NEXT: andl $-16, %esp
-; X32-NEXT: subl $32, %esp
+; X32-NEXT: subl $16, %esp
; X32-NEXT: andl $3, %eax
; X32-NEXT: movaps %xmm0, (%esp)
; X32-NEXT: movss %xmm0, (%esp,%eax,4)
diff --git a/llvm/test/CodeGen/X86/vec_insert-8.ll b/llvm/test/CodeGen/X86/vec_insert-8.ll
index aa3364b31d66f8..ae78705644b992 100644
--- a/llvm/test/CodeGen/X86/vec_insert-8.ll
+++ b/llvm/test/CodeGen/X86/vec_insert-8.ll
@@ -10,7 +10,7 @@ define <4 x i32> @var_insert(<4 x i32> %x, i32 %val, i32 %idx) nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $32, %esp
+; X86-NEXT: subl $16, %esp
; X86-NEXT: movl 12(%ebp), %eax
; X86-NEXT: andl $3, %eax
; X86-NEXT: movl 8(%ebp), %ecx
@@ -40,7 +40,7 @@ define i32 @var_extract(<4 x i32> %x, i32 %idx) nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $32, %esp
+; X86-NEXT: subl $16, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: andl $3, %eax
; X86-NEXT: movaps %xmm0, (%esp)
diff --git a/llvm/test/CodeGen/X86/vector-compress.ll b/llvm/test/CodeGen/X86/vector-compress.ll
index d5872b1a2a351c..e0216248717a6d 100644
--- a/llvm/test/CodeGen/X86/vector-compress.ll
+++ b/llvm/test/CodeGen/X86/vector-compress.ll
@@ -240,7 +240,7 @@ define <8 x i32> @test_compress_v8i32(<8 x i32> %vec, <8 x i1> %mask, <8 x i32>
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: pushq %rbx
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $64, %rsp
+; AVX2-NEXT: subq $32, %rsp
; AVX2-NEXT: vpmovzxwd {{.*#+}} ymm1 = xmm1[0],zero,xmm1[1],zero,xmm1[2],zero,xmm1[3],zero,xmm1[4],zero,xmm1[5],zero,xmm1[6],zero,xmm1[7],zero
; AVX2-NEXT: vpslld $31, %ymm1, %ymm1
; AVX2-NEXT: vmovaps %ymm2, (%rsp)
@@ -336,7 +336,7 @@ define <8 x float> @test_compress_v8f32(<8 x float> %vec, <8 x i1> %mask, <8 x f
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $64, %rsp
+; AVX2-NEXT: subq $32, %rsp
; AVX2-NEXT: vpmovzxwd {{.*#+}} ymm1 = xmm1[0],zero,xmm1[1],zero,xmm1[2],zero,xmm1[3],zero,xmm1[4],zero,xmm1[5],zero,xmm1[6],zero,xmm1[7],zero
; AVX2-NEXT: vpslld $31, %ymm1, %ymm1
; AVX2-NEXT: vmovaps %ymm2, (%rsp)
@@ -439,7 +439,7 @@ define <4 x i64> @test_compress_v4i64(<4 x i64> %vec, <4 x i1> %mask, <4 x i64>
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $64, %rsp
+; AVX2-NEXT: subq $32, %rsp
; AVX2-NEXT: vpslld $31, %xmm1, %xmm1
; AVX2-NEXT: vpsrad $31, %xmm1, %xmm1
; AVX2-NEXT: vpmovsxdq %xmm1, %ymm1
@@ -513,7 +513,7 @@ define <4 x double> @test_compress_v4f64(<4 x double> %vec, <4 x i1> %mask, <4 x
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $64, %rsp
+; AVX2-NEXT: subq $32, %rsp
; AVX2-NEXT: vpslld $31, %xmm1, %xmm1
; AVX2-NEXT: vpsrad $31, %xmm1, %xmm1
; AVX2-NEXT: vmovaps %ymm2, (%rsp)
@@ -742,7 +742,7 @@ define <16 x float> @test_compress_v16f32(<16 x float> %vec, <16 x i1> %mask, <1
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vmovaps %ymm4, {{[0-9]+}}(%rsp)
; AVX2-NEXT: vmovaps %ymm3, (%rsp)
; AVX2-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2, %xmm3
@@ -890,7 +890,7 @@ define <8 x i64> @test_compress_v8i64(<8 x i64> %vec, <8 x i1> %mask, <8 x i64>
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: pushq %rbx
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vmovaps %ymm4, {{[0-9]+}}(%rsp)
; AVX2-NEXT: vmovaps %ymm3, (%rsp)
; AVX2-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2, %xmm3
@@ -982,7 +982,7 @@ define <8 x double> @test_compress_v8f64(<8 x double> %vec, <8 x i1> %mask, <8 x
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vmovaps %ymm4, {{[0-9]+}}(%rsp)
; AVX2-NEXT: vmovaps %ymm3, (%rsp)
; AVX2-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2, %xmm3
@@ -1349,7 +1349,7 @@ define <32 x i8> @test_compress_v32i8(<32 x i8> %vec, <32 x i1> %mask, <32 x i8>
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $64, %rsp
+; AVX2-NEXT: subq $32, %rsp
; AVX2-NEXT: vpsllw $7, %ymm1, %ymm1
; AVX2-NEXT: vpxor %xmm3, %xmm3, %xmm3
; AVX2-NEXT: vmovaps %ymm2, (%rsp)
@@ -1568,7 +1568,7 @@ define <32 x i8> @test_compress_v32i8(<32 x i8> %vec, <32 x i1> %mask, <32 x i8>
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-32, %rsp
-; AVX512F-NEXT: subq $64, %rsp
+; AVX512F-NEXT: subq $32, %rsp
; AVX512F-NEXT: vpmovzxbd {{.*#+}} zmm4 = xmm1[0],zero,zero,zero,xmm1[1],zero,zero,zero,xmm1[2],zero,zero,zero,xmm1[3],zero,zero,zero,xmm1[4],zero,zero,zero,xmm1[5],zero,zero,zero,xmm1[6],zero,zero,zero,xmm1[7],zero,zero,zero,xmm1[8],zero,zero,zero,xmm1[9],zero,zero,zero,xmm1[10],zero,zero,zero,xmm1[11],zero,zero,zero,xmm1[12],zero,zero,zero,xmm1[13],zero,zero,zero,xmm1[14],zero,zero,zero,xmm1[15],zero,zero,zero
; AVX512F-NEXT: vextracti128 $1, %ymm1, %xmm3
; AVX512F-NEXT: vpmovsxbd %xmm3, %zmm5
@@ -1629,7 +1629,7 @@ define <32 x i8> @test_compress_v32i8(<32 x i8> %vec, <32 x i1> %mask, <32 x i8>
; AVX512VL-ONLY-NEXT: pushq %rbp
; AVX512VL-ONLY-NEXT: movq %rsp, %rbp
; AVX512VL-ONLY-NEXT: andq $-32, %rsp
-; AVX512VL-ONLY-NEXT: subq $64, %rsp
+; AVX512VL-ONLY-NEXT: subq $32, %rsp
; AVX512VL-ONLY-NEXT: vpsllw $7, %ymm1, %ymm1
; AVX512VL-ONLY-NEXT: vpmovb2m %ymm1, %k6
; AVX512VL-ONLY-NEXT: kshiftrd $31, %k6, %k0
@@ -2090,7 +2090,7 @@ define <64 x i8> @test_compress_v64i8(<64 x i8> %vec, <64 x i1> %mask, <64 x i8>
; AVX2-NEXT: pushq %r12
; AVX2-NEXT: pushq %rbx
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: movzbl 360(%rbp), %eax
; AVX2-NEXT: movzbl 352(%rbp), %r10d
; AVX2-NEXT: vmovd %r10d, %xmm4
@@ -2684,7 +2684,7 @@ define <64 x i8> @test_compress_v64i8(<64 x i8> %vec, <64 x i1> %mask, <64 x i8>
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-64, %rsp
-; AVX512F-NEXT: subq $192, %rsp
+; AVX512F-NEXT: addq $-128, %rsp
; AVX512F-NEXT: vmovdqu64 352(%rbp), %zmm2
; AVX512F-NEXT: vmovdqu64 416(%rbp), %zmm3
; AVX512F-NEXT: vpmovqw %zmm2, %xmm2
@@ -2823,7 +2823,7 @@ define <64 x i8> @test_compress_v64i8(<64 x i8> %vec, <64 x i1> %mask, <64 x i8>
; AVX512VL-ONLY-NEXT: pushq %rbp
; AVX512VL-ONLY-NEXT: movq %rsp, %rbp
; AVX512VL-ONLY-NEXT: andq $-64, %rsp
-; AVX512VL-ONLY-NEXT: addq $-128, %rsp
+; AVX512VL-ONLY-NEXT: subq $64, %rsp
; AVX512VL-ONLY-NEXT: vpsllw $7, %zmm1, %zmm1
; AVX512VL-ONLY-NEXT: vpmovb2m %zmm1, %k6
; AVX512VL-ONLY-NEXT: kshiftrq $63, %k6, %k0
@@ -3620,7 +3620,7 @@ define <32 x i16> @test_compress_v32i16(<32 x i16> %vec, <32 x i1> %mask, <32 x
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-64, %rsp
-; AVX512F-NEXT: addq $-128, %rsp
+; AVX512F-NEXT: subq $64, %rsp
; AVX512F-NEXT: vpmovzxbd {{.*#+}} zmm4 = xmm1[0],zero,zero,zero,xmm1[1],zero,zero,zero,xmm1[2],zero,zero,zero,xmm1[3],zero,zero,zero,xmm1[4],zero,zero,zero,xmm1[5],zero,zero,zero,xmm1[6],zero,zero,zero,xmm1[7],zero,zero,zero,xmm1[8],zero,zero,zero,xmm1[9],zero,zero,zero,xmm1[10],zero,zero,zero,xmm1[11],zero,zero,zero,xmm1[12],zero,zero,zero,xmm1[13],zero,zero,zero,xmm1[14],zero,zero,zero,xmm1[15],zero,zero,zero
; AVX512F-NEXT: vextracti128 $1, %ymm1, %xmm3
; AVX512F-NEXT: vpmovsxbd %xmm3, %zmm5
@@ -4015,7 +4015,7 @@ define <64 x i32> @test_compress_large(<64 x i1> %mask, <64 x i32> %vec, <64 x i
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $288, %rsp # imm = 0x120
+; AVX2-NEXT: subq $256, %rsp # imm = 0x100
; AVX2-NEXT: # kill: def $esi killed $esi def $rsi
; AVX2-NEXT: vmovss %xmm0, (%rsp)
; AVX2-NEXT: andl $1, %esi
@@ -4474,7 +4474,7 @@ define <64 x i32> @test_compress_large(<64 x i1> %mask, <64 x i32> %vec, <64 x i
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-64, %rsp
-; AVX512F-NEXT: subq $576, %rsp # imm = 0x240
+; AVX512F-NEXT: subq $512, %rsp # imm = 0x200
; AVX512F-NEXT: vmovdqu64 352(%rbp), %zmm4
; AVX512F-NEXT: vmovdqu64 416(%rbp), %zmm5
; AVX512F-NEXT: vpmovqw %zmm4, %xmm4
@@ -4581,7 +4581,7 @@ define <64 x i32> @test_compress_large(<64 x i1> %mask, <64 x i32> %vec, <64 x i
; AVX512VL-NEXT: pushq %rbp
; AVX512VL-NEXT: movq %rsp, %rbp
; AVX512VL-NEXT: andq $-64, %rsp
-; AVX512VL-NEXT: subq $576, %rsp # imm = 0x240
+; AVX512VL-NEXT: subq $512, %rsp # imm = 0x200
; AVX512VL-NEXT: vpsllw $7, %zmm0, %zmm0
; AVX512VL-NEXT: vpmovb2m %zmm0, %k3
; AVX512VL-NEXT: kshiftrq $48, %k3, %k1
@@ -4658,7 +4658,7 @@ define <8 x i64> @test_compress_knownbits_zext_v8i16_8i64(<8 x i16> %vec, <8 x
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: pushq %rbx
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vpshufd {{.*#+}} xmm4 = xmm0[2,3,2,3]
; AVX2-NEXT: vpmovzxwq {{.*#+}} ymm4 = xmm4[0],zero,zero,zero,xmm4[1],zero,zero,zero,xmm4[2],zero,zero,zero,xmm4[3],zero,zero,zero
; AVX2-NEXT: vbroadcastsd {{.*#+}} ymm5 = [3,3,3,3]
@@ -4762,7 +4762,7 @@ define <8 x i64> @test_compress_knownbits_sext_v8i16_8i64(<8 x i16> %vec, <8 x i
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: pushq %rbx
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vpshufd {{.*#+}} xmm4 = xmm0[2,3,2,3]
; AVX2-NEXT: vpmovsxwq %xmm4, %ymm4
; AVX2-NEXT: vbroadcastsd {{.*#+}} ymm5 = [3,3,3,3]
diff --git a/llvm/test/CodeGen/X86/vector-extend-inreg.ll b/llvm/test/CodeGen/X86/vector-extend-inreg.ll
index 9648eb5fc584a1..5cf7751d4ac89d 100644
--- a/llvm/test/CodeGen/X86/vector-extend-inreg.ll
+++ b/llvm/test/CodeGen/X86/vector-extend-inreg.ll
@@ -10,7 +10,7 @@ define i64 @extract_any_extend_vector_inreg_v16i64(<16 x i64> %a0, i32 %a1) noun
; X86-SSE-NEXT: pushl %ebp
; X86-SSE-NEXT: movl %esp, %ebp
; X86-SSE-NEXT: andl $-16, %esp
-; X86-SSE-NEXT: subl $272, %esp # imm = 0x110
+; X86-SSE-NEXT: subl $256, %esp # imm = 0x100
; X86-SSE-NEXT: movl 88(%ebp), %eax
; X86-SSE-NEXT: movdqa 72(%ebp), %xmm0
; X86-SSE-NEXT: psrldq {{.*#+}} xmm0 = xmm0[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero
@@ -64,7 +64,7 @@ define i64 @extract_any_extend_vector_inreg_v16i64(<16 x i64> %a0, i32 %a1) noun
; X86-AVX-NEXT: pushl %ebp
; X86-AVX-NEXT: movl %esp, %ebp
; X86-AVX-NEXT: andl $-32, %esp
-; X86-AVX-NEXT: subl $288, %esp # imm = 0x120
+; X86-AVX-NEXT: subl $256, %esp # imm = 0x100
; X86-AVX-NEXT: movl 40(%ebp), %eax
; X86-AVX-NEXT: vmovsd {{.*#+}} xmm0 = mem[0],zero
; X86-AVX-NEXT: vxorps %xmm1, %xmm1, %xmm1
@@ -91,7 +91,7 @@ define i64 @extract_any_extend_vector_inreg_v16i64(<16 x i64> %a0, i32 %a1) noun
; X64-AVX-NEXT: pushq %rbp
; X64-AVX-NEXT: movq %rsp, %rbp
; X64-AVX-NEXT: andq $-32, %rsp
-; X64-AVX-NEXT: subq $160, %rsp
+; X64-AVX-NEXT: addq $-128, %rsp
; X64-AVX-NEXT: # kill: def $edi killed $edi def $rdi
; X64-AVX-NEXT: vpermq {{.*#+}} ymm0 = ymm3[3,3,3,3]
; X64-AVX-NEXT: vmovq {{.*#+}} xmm0 = xmm0[0],zero
diff --git a/llvm/test/CodeGen/X86/vector-extract-last-active.ll b/llvm/test/CodeGen/X86/vector-extract-last-active.ll
index 16a2de9783893c..4a2857d57f3772 100644
--- a/llvm/test/CodeGen/X86/vector-extract-last-active.ll
+++ b/llvm/test/CodeGen/X86/vector-extract-last-active.ll
@@ -435,7 +435,7 @@ define i32 @extract_last_active_v8i32(<8 x i32> %a, <8 x i1> %c) nounwind {
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $64, %rsp
+; AVX2-NEXT: subq $32, %rsp
; AVX2-NEXT: vmovaps %ymm0, (%rsp)
; AVX2-NEXT: vpsllw $15, %xmm1, %xmm0
; AVX2-NEXT: vpacksswb %xmm0, %xmm0, %xmm1
@@ -462,7 +462,7 @@ define i32 @extract_last_active_v8i32(<8 x i32> %a, <8 x i1> %c) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-32, %rsp
-; AVX512-NEXT: subq $64, %rsp
+; AVX512-NEXT: subq $32, %rsp
; AVX512-NEXT: vpsllw $15, %xmm1, %xmm1
; AVX512-NEXT: vpmovw2m %xmm1, %k1
; AVX512-NEXT: vmovdqa %ymm0, (%rsp)
@@ -552,7 +552,7 @@ define i32 @extract_last_active_v16i32(<16 x i32> %a, <16 x i1> %c) nounwind {
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $96, %rsp
+; AVX2-NEXT: subq $64, %rsp
; AVX2-NEXT: vpxor %xmm3, %xmm3, %xmm3
; AVX2-NEXT: vpsllw $7, %xmm2, %xmm2
; AVX2-NEXT: vpcmpgtb %xmm2, %xmm3, %xmm3
@@ -583,7 +583,7 @@ define i32 @extract_last_active_v16i32(<16 x i32> %a, <16 x i1> %c) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: addq $-128, %rsp
+; AVX512-NEXT: subq $64, %rsp
; AVX512-NEXT: vpsllw $7, %xmm1, %xmm1
; AVX512-NEXT: vpmovb2m %xmm1, %k1
; AVX512-NEXT: vmovdqa64 %zmm0, (%rsp)
@@ -723,7 +723,7 @@ define i8 @extract_last_active_split(<32 x i8> %data, <32 x i8> %mask, i8 %passt
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $64, %rsp
+; AVX2-NEXT: subq $32, %rsp
; AVX2-NEXT: vpxor %xmm2, %xmm2, %xmm2
; AVX2-NEXT: vpcmpeqb %ymm2, %ymm1, %ymm2
; AVX2-NEXT: vmovaps %ymm0, (%rsp)
@@ -753,7 +753,7 @@ define i8 @extract_last_active_split(<32 x i8> %data, <32 x i8> %mask, i8 %passt
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-32, %rsp
-; AVX512-NEXT: subq $64, %rsp
+; AVX512-NEXT: subq $32, %rsp
; AVX512-NEXT: vptestmb %ymm1, %ymm1, %k1
; AVX512-NEXT: vmovaps %ymm0, (%rsp)
; AVX512-NEXT: vmovdqu8 {{.*#+}} ymm0 {%k1} {z} = [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31]
diff --git a/llvm/test/CodeGen/X86/vector-llrint.ll b/llvm/test/CodeGen/X86/vector-llrint.ll
index 50b631c54d8751..352009e8d8d606 100644
--- a/llvm/test/CodeGen/X86/vector-llrint.ll
+++ b/llvm/test/CodeGen/X86/vector-llrint.ll
@@ -141,7 +141,7 @@ define <4 x i64> @llrint_v4i64_v4f32(<4 x float> %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-8, %esp
-; X86-NEXT: subl $56, %esp
+; X86-NEXT: subl $48, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: flds 24(%ebp)
; X86-NEXT: flds 20(%ebp)
@@ -301,7 +301,7 @@ define <8 x i64> @llrint_v8i64_v8f32(<8 x float> %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-8, %esp
-; X86-NEXT: subl $120, %esp
+; X86-NEXT: subl $112, %esp
; X86-NEXT: flds 12(%ebp)
; X86-NEXT: fistpll {{[0-9]+}}(%esp)
; X86-NEXT: flds 16(%ebp)
@@ -586,7 +586,7 @@ define <16 x i64> @llrint_v16i64_v16f32(<16 x float> %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-8, %esp
-; X86-NEXT: subl $248, %esp
+; X86-NEXT: subl $240, %esp
; X86-NEXT: flds 12(%ebp)
; X86-NEXT: fistpll {{[0-9]+}}(%esp)
; X86-NEXT: flds 16(%ebp)
@@ -1254,7 +1254,7 @@ define <4 x i64> @llrint_v4i64_v4f64(<4 x double> %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-8, %esp
-; X86-NEXT: subl $56, %esp
+; X86-NEXT: subl $48, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: fldl 36(%ebp)
; X86-NEXT: fldl 28(%ebp)
@@ -1417,7 +1417,7 @@ define <8 x i64> @llrint_v8i64_v8f64(<8 x double> %x) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-8, %esp
-; X86-NEXT: subl $120, %esp
+; X86-NEXT: subl $112, %esp
; X86-NEXT: fldl 12(%ebp)
; X86-NEXT: fistpll {{[0-9]+}}(%esp)
; X86-NEXT: fldl 20(%ebp)
diff --git a/llvm/test/CodeGen/X86/vector-lrint.ll b/llvm/test/CodeGen/X86/vector-lrint.ll
index d7261cf772c522..55a2e083414de3 100644
--- a/llvm/test/CodeGen/X86/vector-lrint.ll
+++ b/llvm/test/CodeGen/X86/vector-lrint.ll
@@ -276,7 +276,7 @@ define <4 x iXLen> @lrint_v4f32(<4 x float> %x) nounwind {
; X86-I64-NEXT: pushl %edi
; X86-I64-NEXT: pushl %esi
; X86-I64-NEXT: andl $-8, %esp
-; X86-I64-NEXT: subl $56, %esp
+; X86-I64-NEXT: subl $48, %esp
; X86-I64-NEXT: movl 8(%ebp), %eax
; X86-I64-NEXT: flds 24(%ebp)
; X86-I64-NEXT: flds 20(%ebp)
@@ -544,7 +544,7 @@ define <8 x iXLen> @lrint_v8f32(<8 x float> %x) nounwind {
; X86-I64-NEXT: pushl %edi
; X86-I64-NEXT: pushl %esi
; X86-I64-NEXT: andl $-8, %esp
-; X86-I64-NEXT: subl $120, %esp
+; X86-I64-NEXT: subl $112, %esp
; X86-I64-NEXT: flds 12(%ebp)
; X86-I64-NEXT: fistpll {{[0-9]+}}(%esp)
; X86-I64-NEXT: flds 16(%ebp)
@@ -1081,7 +1081,7 @@ define <4 x iXLen> @lrint_v4f64(<4 x double> %x) nounwind {
; X86-I64-NEXT: pushl %edi
; X86-I64-NEXT: pushl %esi
; X86-I64-NEXT: andl $-8, %esp
-; X86-I64-NEXT: subl $56, %esp
+; X86-I64-NEXT: subl $48, %esp
; X86-I64-NEXT: movl 8(%ebp), %eax
; X86-I64-NEXT: fldl 36(%ebp)
; X86-I64-NEXT: fldl 28(%ebp)
@@ -1366,7 +1366,7 @@ define <8 x iXLen> @lrint_v8f64(<8 x double> %x) nounwind {
; X86-I64-NEXT: pushl %edi
; X86-I64-NEXT: pushl %esi
; X86-I64-NEXT: andl $-8, %esp
-; X86-I64-NEXT: subl $120, %esp
+; X86-I64-NEXT: subl $112, %esp
; X86-I64-NEXT: fldl 12(%ebp)
; X86-I64-NEXT: fistpll {{[0-9]+}}(%esp)
; X86-I64-NEXT: fldl 20(%ebp)
diff --git a/llvm/test/CodeGen/X86/vector-reduce-ctpop.ll b/llvm/test/CodeGen/X86/vector-reduce-ctpop.ll
index ffa915336d92fa..7876fcccd43692 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-ctpop.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-ctpop.ll
@@ -2443,7 +2443,7 @@ define i64 @reduce_ctpop_v16i64(<16 x i64> %a0) nounwind {
; X86-SSE4-NEXT: pushl %ebp
; X86-SSE4-NEXT: movl %esp, %ebp
; X86-SSE4-NEXT: andl $-16, %esp
-; X86-SSE4-NEXT: subl $32, %esp
+; X86-SSE4-NEXT: subl $16, %esp
; X86-SSE4-NEXT: movaps %xmm2, (%esp) # 16-byte Spill
; X86-SSE4-NEXT: movdqa %xmm0, %xmm2
; X86-SSE4-NEXT: movdqa 40(%ebp), %xmm5
@@ -2636,7 +2636,7 @@ define i64 @reduce_ctpop_v16i64(<16 x i64> %a0) nounwind {
; X86-AVX1-NEXT: pushl %ebp
; X86-AVX1-NEXT: movl %esp, %ebp
; X86-AVX1-NEXT: andl $-32, %esp
-; X86-AVX1-NEXT: subl $96, %esp
+; X86-AVX1-NEXT: subl $64, %esp
; X86-AVX1-NEXT: vmovaps %ymm1, {{[-0-9]+}}(%e{{[sb]}}p) # 32-byte Spill
; X86-AVX1-NEXT: vextractf128 $1, %ymm2, %xmm5
; X86-AVX1-NEXT: vbroadcastss {{.*#+}} xmm3 = [15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15]
@@ -3743,7 +3743,7 @@ define <8 x i32> @reduce_ctpop_v4i64_buildvector_v8i32(<4 x i64> %a0, <4 x i64>
; X86-SSE2-NEXT: pushl %ebp
; X86-SSE2-NEXT: movl %esp, %ebp
; X86-SSE2-NEXT: andl $-16, %esp
-; X86-SSE2-NEXT: subl $80, %esp
+; X86-SSE2-NEXT: subl $64, %esp
; X86-SSE2-NEXT: movdqa 24(%ebp), %xmm6
; X86-SSE2-NEXT: movdqa 8(%ebp), %xmm5
; X86-SSE2-NEXT: movdqa %xmm1, %xmm3
@@ -4295,7 +4295,7 @@ define <8 x i32> @reduce_ctpop_v4i64_buildvector_v8i32(<4 x i64> %a0, <4 x i64>
; X86-SSE4-NEXT: pushl %ebp
; X86-SSE4-NEXT: movl %esp, %ebp
; X86-SSE4-NEXT: andl $-16, %esp
-; X86-SSE4-NEXT: subl $80, %esp
+; X86-SSE4-NEXT: subl $64, %esp
; X86-SSE4-NEXT: movdqa 24(%ebp), %xmm6
; X86-SSE4-NEXT: movdqa {{.*#+}} xmm4 = [15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15]
; X86-SSE4-NEXT: movdqa %xmm1, %xmm5
@@ -5037,7 +5037,7 @@ define <8 x i32> @reduce_ctpop_v4i64_buildvector_v8i32(<4 x i64> %a0, <4 x i64>
; X86-AVX2-NEXT: pushl %ebp
; X86-AVX2-NEXT: movl %esp, %ebp
; X86-AVX2-NEXT: andl $-32, %esp
-; X86-AVX2-NEXT: subl $192, %esp
+; X86-AVX2-NEXT: subl $160, %esp
; X86-AVX2-NEXT: vmovdqa 40(%ebp), %ymm6
; X86-AVX2-NEXT: vmovdqa 8(%ebp), %ymm5
; X86-AVX2-NEXT: vpbroadcastb {{.*#+}} ymm3 = [15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15]
diff --git a/llvm/test/CodeGen/X86/vector-reduce-smax.ll b/llvm/test/CodeGen/X86/vector-reduce-smax.ll
index b114dcc46696d8..8861735443338d 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-smax.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-smax.ll
@@ -728,7 +728,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
; X86-SSE2-NEXT: pushl %ebp
; X86-SSE2-NEXT: movl %esp, %ebp
; X86-SSE2-NEXT: andl $-16, %esp
-; X86-SSE2-NEXT: subl $32, %esp
+; X86-SSE2-NEXT: subl $16, %esp
; X86-SSE2-NEXT: movdqa %xmm2, %xmm7
; X86-SSE2-NEXT: movdqa %xmm1, %xmm2
; X86-SSE2-NEXT: movaps %xmm0, (%esp) # 16-byte Spill
@@ -988,7 +988,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
; X86-SSE41-NEXT: pushl %ebp
; X86-SSE41-NEXT: movl %esp, %ebp
; X86-SSE41-NEXT: andl $-16, %esp
-; X86-SSE41-NEXT: subl $32, %esp
+; X86-SSE41-NEXT: subl $16, %esp
; X86-SSE41-NEXT: movdqa %xmm2, %xmm3
; X86-SSE41-NEXT: movdqa %xmm1, %xmm2
; X86-SSE41-NEXT: movaps %xmm0, (%esp) # 16-byte Spill
diff --git a/llvm/test/CodeGen/X86/vector-reduce-smin.ll b/llvm/test/CodeGen/X86/vector-reduce-smin.ll
index fa440378975e33..00b7d8d6c562ec 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-smin.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-smin.ll
@@ -983,7 +983,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
; X86-SSE41-NEXT: pushl %ebp
; X86-SSE41-NEXT: movl %esp, %ebp
; X86-SSE41-NEXT: andl $-16, %esp
-; X86-SSE41-NEXT: subl $48, %esp
+; X86-SSE41-NEXT: subl $32, %esp
; X86-SSE41-NEXT: movaps %xmm1, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
; X86-SSE41-NEXT: movdqa %xmm0, %xmm3
; X86-SSE41-NEXT: movdqa 24(%ebp), %xmm6
diff --git a/llvm/test/CodeGen/X86/vector-reduce-umax.ll b/llvm/test/CodeGen/X86/vector-reduce-umax.ll
index f9a6a685e449ff..dade241b6e81e7 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-umax.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-umax.ll
@@ -835,7 +835,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
; X86-SSE2-NEXT: pushl %ebp
; X86-SSE2-NEXT: movl %esp, %ebp
; X86-SSE2-NEXT: andl $-16, %esp
-; X86-SSE2-NEXT: subl $32, %esp
+; X86-SSE2-NEXT: subl $16, %esp
; X86-SSE2-NEXT: movdqa %xmm2, %xmm7
; X86-SSE2-NEXT: movdqa %xmm1, %xmm2
; X86-SSE2-NEXT: movaps %xmm0, (%esp) # 16-byte Spill
@@ -1095,7 +1095,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
; X86-SSE41-NEXT: pushl %ebp
; X86-SSE41-NEXT: movl %esp, %ebp
; X86-SSE41-NEXT: andl $-16, %esp
-; X86-SSE41-NEXT: subl $32, %esp
+; X86-SSE41-NEXT: subl $16, %esp
; X86-SSE41-NEXT: movdqa %xmm2, %xmm3
; X86-SSE41-NEXT: movdqa %xmm1, %xmm2
; X86-SSE41-NEXT: movaps %xmm0, (%esp) # 16-byte Spill
@@ -1440,7 +1440,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
; X86-AVX1-NEXT: pushl %ebp
; X86-AVX1-NEXT: movl %esp, %ebp
; X86-AVX1-NEXT: andl $-32, %esp
-; X86-AVX1-NEXT: subl $96, %esp
+; X86-AVX1-NEXT: subl $64, %esp
; X86-AVX1-NEXT: vmovaps %ymm2, {{[-0-9]+}}(%e{{[sb]}}p) # 32-byte Spill
; X86-AVX1-NEXT: vmovaps %ymm0, (%esp) # 32-byte Spill
; X86-AVX1-NEXT: vmovddup {{.*#+}} xmm3 = [0,2147483648,0,2147483648]
diff --git a/llvm/test/CodeGen/X86/vector-reduce-umin.ll b/llvm/test/CodeGen/X86/vector-reduce-umin.ll
index 12bfb53cb715f1..6c5c31ded85003 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-umin.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-umin.ll
@@ -1091,7 +1091,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
; X86-SSE41-NEXT: pushl %ebp
; X86-SSE41-NEXT: movl %esp, %ebp
; X86-SSE41-NEXT: andl $-16, %esp
-; X86-SSE41-NEXT: subl $48, %esp
+; X86-SSE41-NEXT: subl $32, %esp
; X86-SSE41-NEXT: movaps %xmm1, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
; X86-SSE41-NEXT: movdqa %xmm0, %xmm3
; X86-SSE41-NEXT: movdqa 24(%ebp), %xmm6
diff --git a/llvm/test/CodeGen/X86/vector-shuffle-512-v16.ll b/llvm/test/CodeGen/X86/vector-shuffle-512-v16.ll
index 79643cdeebabcf..2fb038760711a9 100644
--- a/llvm/test/CodeGen/X86/vector-shuffle-512-v16.ll
+++ b/llvm/test/CodeGen/X86/vector-shuffle-512-v16.ll
@@ -968,7 +968,7 @@ define void @ispc_1864(ptr %arg) {
; ALL-NEXT: movq %rsp, %rbp
; ALL-NEXT: .cfi_def_cfa_register %rbp
; ALL-NEXT: andq $-64, %rsp
-; ALL-NEXT: subq $4864, %rsp # imm = 0x1300
+; ALL-NEXT: subq $4800, %rsp # imm = 0x12C0
; ALL-NEXT: vbroadcastss {{.*#+}} ymm0 = [-5.0E+0,-5.0E+0,-5.0E+0,-5.0E+0,-5.0E+0,-5.0E+0,-5.0E+0,-5.0E+0]
; ALL-NEXT: vmulps 32(%rdi), %ymm0, %ymm0
; ALL-NEXT: vcvtps2pd %ymm0, %zmm0
diff --git a/llvm/test/CodeGen/X86/vector-shuffle-variable-256.ll b/llvm/test/CodeGen/X86/vector-shuffle-variable-256.ll
index 8f78438dedf92d..d647973d309fdf 100644
--- a/llvm/test/CodeGen/X86/vector-shuffle-variable-256.ll
+++ b/llvm/test/CodeGen/X86/vector-shuffle-variable-256.ll
@@ -12,7 +12,7 @@ define <4 x double> @var_shuffle_v4f64_v4f64_xxxx_i64(<4 x double> %x, i64 %i0,
; ALL-NEXT: pushq %rbp
; ALL-NEXT: movq %rsp, %rbp
; ALL-NEXT: andq $-32, %rsp
-; ALL-NEXT: subq $64, %rsp
+; ALL-NEXT: subq $32, %rsp
; ALL-NEXT: andl $3, %esi
; ALL-NEXT: andl $3, %edi
; ALL-NEXT: andl $3, %ecx
@@ -43,7 +43,7 @@ define <4 x double> @var_shuffle_v4f64_v4f64_uxx0_i64(<4 x double> %x, i64 %i0,
; ALL-NEXT: pushq %rbp
; ALL-NEXT: movq %rsp, %rbp
; ALL-NEXT: andq $-32, %rsp
-; ALL-NEXT: subq $64, %rsp
+; ALL-NEXT: subq $32, %rsp
; ALL-NEXT: andl $3, %edx
; ALL-NEXT: andl $3, %esi
; ALL-NEXT: vmovaps %ymm0, (%rsp)
@@ -95,7 +95,7 @@ define <4 x i64> @var_shuffle_v4i64_v4i64_xxxx_i64(<4 x i64> %x, i64 %i0, i64 %i
; ALL-NEXT: pushq %rbp
; ALL-NEXT: movq %rsp, %rbp
; ALL-NEXT: andq $-32, %rsp
-; ALL-NEXT: subq $64, %rsp
+; ALL-NEXT: subq $32, %rsp
; ALL-NEXT: andl $3, %edi
; ALL-NEXT: andl $3, %esi
; ALL-NEXT: andl $3, %edx
@@ -128,7 +128,7 @@ define <4 x i64> @var_shuffle_v4i64_v4i64_xx00_i64(<4 x i64> %x, i64 %i0, i64 %i
; ALL-NEXT: pushq %rbp
; ALL-NEXT: movq %rsp, %rbp
; ALL-NEXT: andq $-32, %rsp
-; ALL-NEXT: subq $64, %rsp
+; ALL-NEXT: subq $32, %rsp
; ALL-NEXT: andl $3, %edi
; ALL-NEXT: andl $3, %esi
; ALL-NEXT: vmovaps %ymm0, (%rsp)
@@ -182,7 +182,7 @@ define <8 x float> @var_shuffle_v8f32_v8f32_xxxxxxxx_i32(<8 x float> %x, i32 %i0
; ALL-NEXT: pushq %rbp
; ALL-NEXT: movq %rsp, %rbp
; ALL-NEXT: andq $-32, %rsp
-; ALL-NEXT: subq $64, %rsp
+; ALL-NEXT: subq $32, %rsp
; ALL-NEXT: # kill: def $r9d killed $r9d def $r9
; ALL-NEXT: # kill: def $r8d killed $r8d def $r8
; ALL-NEXT: # kill: def $ecx killed $ecx def $rcx
@@ -286,7 +286,7 @@ define <16 x i16> @var_shuffle_v16i16_v16i16_xxxxxxxxxxxxxxxx_i16(<16 x i16> %x,
; AVX1-NEXT: pushq %rbp
; AVX1-NEXT: movq %rsp, %rbp
; AVX1-NEXT: andq $-32, %rsp
-; AVX1-NEXT: subq $64, %rsp
+; AVX1-NEXT: subq $32, %rsp
; AVX1-NEXT: # kill: def $r9d killed $r9d def $r9
; AVX1-NEXT: # kill: def $r8d killed $r8d def $r8
; AVX1-NEXT: # kill: def $ecx killed $ecx def $rcx
@@ -348,7 +348,7 @@ define <16 x i16> @var_shuffle_v16i16_v16i16_xxxxxxxxxxxxxxxx_i16(<16 x i16> %x,
; AVX2-NEXT: pushq %rbp
; AVX2-NEXT: movq %rsp, %rbp
; AVX2-NEXT: andq $-32, %rsp
-; AVX2-NEXT: subq $64, %rsp
+; AVX2-NEXT: subq $32, %rsp
; AVX2-NEXT: # kill: def $r9d killed $r9d def $r9
; AVX2-NEXT: # kill: def $r8d killed $r8d def $r8
; AVX2-NEXT: # kill: def $ecx killed $ecx def $rcx
@@ -596,7 +596,7 @@ define <4 x i64> @mem_shuffle_v4i64_v4i64_xxxx_i64(<4 x i64> %x, ptr %i) nounwin
; ALL-NEXT: pushq %rbp
; ALL-NEXT: movq %rsp, %rbp
; ALL-NEXT: andq $-32, %rsp
-; ALL-NEXT: subq $64, %rsp
+; ALL-NEXT: subq $32, %rsp
; ALL-NEXT: movl (%rdi), %eax
; ALL-NEXT: movl 8(%rdi), %ecx
; ALL-NEXT: andl $3, %eax
diff --git a/llvm/test/CodeGen/X86/widen_arith-6.ll b/llvm/test/CodeGen/X86/widen_arith-6.ll
index 6fa232f4d3227d..7f84fd2f546d50 100644
--- a/llvm/test/CodeGen/X86/widen_arith-6.ll
+++ b/llvm/test/CodeGen/X86/widen_arith-6.ll
@@ -9,7 +9,7 @@ define void @update(ptr %dst, ptr %src, i32 %n) nounwind {
; CHECK-NEXT: pushl %ebp
; CHECK-NEXT: movl %esp, %ebp
; CHECK-NEXT: andl $-16, %esp
-; CHECK-NEXT: subl $48, %esp
+; CHECK-NEXT: subl $32, %esp
; CHECK-NEXT: movl $1077936128, {{[0-9]+}}(%esp) # imm = 0x40400000
; CHECK-NEXT: movl $1073741824, {{[0-9]+}}(%esp) # imm = 0x40000000
; CHECK-NEXT: movl $1065353216, {{[0-9]+}}(%esp) # imm = 0x3F800000
diff --git a/llvm/test/CodeGen/X86/x86-64-baseptr.ll b/llvm/test/CodeGen/X86/x86-64-baseptr.ll
index 4f366ba3144a89..1dabef12521e3d 100644
--- a/llvm/test/CodeGen/X86/x86-64-baseptr.ll
+++ b/llvm/test/CodeGen/X86/x86-64-baseptr.ll
@@ -359,7 +359,7 @@ define void @vmw_host_printf(ptr %fmt, ...) nounwind {
; X32ABI-NEXT: movl %esp, %ebp
; X32ABI-NEXT: pushq %rbx
; X32ABI-NEXT: andl $-16, %esp
-; X32ABI-NEXT: subl $208, %esp
+; X32ABI-NEXT: subl $192, %esp
; X32ABI-NEXT: movl %esp, %ebx
; X32ABI-NEXT: movq %rsi, 24(%ebx)
; X32ABI-NEXT: movq %rdx, 32(%ebx)
diff --git a/llvm/test/DebugInfo/COFF/fpo-realign-alloca.ll b/llvm/test/DebugInfo/COFF/fpo-realign-alloca.ll
index d6f45b166d22fd..c9972b0b10306f 100644
--- a/llvm/test/DebugInfo/COFF/fpo-realign-alloca.ll
+++ b/llvm/test/DebugInfo/COFF/fpo-realign-alloca.ll
@@ -20,8 +20,8 @@
; CHECK: .cv_fpo_pushreg %esi
; CHECK: andl $-16, %esp
; CHECK: .cv_fpo_stackalign 16
-; CHECK: subl $32, %esp
-; CHECK: .cv_fpo_stackalloc 32
+; CHECK: subl $16, %esp
+; CHECK: .cv_fpo_stackalloc 16
; CHECK: .cv_fpo_endprologue
; CHECK: movl %esp, %esi
; CHECK: leal 8(%esi),
diff --git a/llvm/test/DebugInfo/COFF/vframe-csr.ll b/llvm/test/DebugInfo/COFF/vframe-csr.ll
index f46965afd7da55..3e8e941d1e2d94 100644
--- a/llvm/test/DebugInfo/COFF/vframe-csr.ll
+++ b/llvm/test/DebugInfo/COFF/vframe-csr.ll
@@ -16,9 +16,9 @@
; ASM: .cv_fpo_setframe %ebp
; ASM: andl $-8, %esp
; ASM: .cv_fpo_stackalign 8
-; FIXME: Why 24 bytes? We only need 12 bytes of data.
-; ASM: subl $24, %esp
-; ASM: .cv_fpo_stackalloc 24
+; 12 bytes of data, rounded up to the 8-byte realignment.
+; ASM: subl $16, %esp
+; ASM: .cv_fpo_stackalloc 16
; ASM: .cv_fpo_endprologue
; 'x' should be EBP-relative, 'a' and 'force_alignment' ESP relative.
@@ -70,7 +70,7 @@
; OBJ: LocalFramePtrReg: VFRAME (0x7536)
; OBJ: ParamFramePtrReg: EBP (0x16)
; OBJ: }
-; ESP is VFRAME - 24, ESP offset of 'a' is 4, so -20.
+; ESP is VFRAME - 16, ESP offset of 'a' is 4, so -12.
; OBJ: LocalSym {
; OBJ: Kind: S_LOCAL (0x113E)
; OBJ: Type: int (0x74)
@@ -80,7 +80,7 @@
; OBJ: }
; OBJ: DefRangeFramePointerRelSym {
; OBJ: Kind: S_DEFRANGE_FRAMEPOINTER_REL (0x1142)
-; OBJ: Offset: -20
+; OBJ: Offset: -12
; OBJ: }
; ESP is VFRAME - 16, ESP offset of 'force_alignment' is 8, so -8.
; OBJ: LocalSym {
@@ -92,7 +92,7 @@
; OBJ: }
; OBJ: DefRangeFramePointerRelSym {
; OBJ: Kind: S_DEFRANGE_FRAMEPOINTER_REL (0x1142)
-; OBJ: Offset: -16
+; OBJ: Offset: -8
; OBJ: }
; OBJ: ProcEnd {
; OBJ: Kind: S_PROC_ID_END (0x114F)
More information about the llvm-commits
mailing list