[llvm] [X86] Don't allocate unused padding in front of realigned locals (PR #227495)

Timur Golubovich via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 07:00:38 PDT 2026


https://github.com/timurgol007 updated https://github.com/llvm/llvm-project/pull/227495

>From 650afe8e04f409adb3bdd592d3f6787317afcc21 Mon Sep 17 00:00:00 2001
From: Timur Golubovich <timur.golubovich at intel.com>
Date: Fri, 25 Sep 2026 18:22:09 +0200
Subject: [PATCH] [X86] Don't allocate unused padding in front of realigned
 locals

PEI aligns local objects relative to the incoming stack pointer, leaving
a gap between the CSRs and the first local. With the stack realigned after
the pushes, locals are addressed from the final stack pointer, so that gap
is never used: exclude it from the prologue's allocation.

Fixes #224278
---
 llvm/lib/Target/X86/X86FrameLowering.cpp      |  37 ++++++-
 .../X86/2009-06-05-VariableIndexInsert.ll     |   2 +-
 llvm/test/CodeGen/X86/AMX/amx-across-func.ll  |   6 +-
 llvm/test/CodeGen/X86/AMX/amx-configO2toO0.ll |   2 +-
 llvm/test/CodeGen/X86/AMX/amx-zero-config.ll  |   6 +-
 .../CodeGen/X86/amx-across-func-tilemovrow.ll |   2 +-
 llvm/test/CodeGen/X86/amx_movrs_intrinsics.ll |  12 +-
 llvm/test/CodeGen/X86/andnot-sink-not.ll      |   4 +-
 llvm/test/CodeGen/X86/arg-copy-elide.ll       |   2 +-
 llvm/test/CodeGen/X86/atomic-fp.ll            |   8 +-
 .../X86/atomic-idempotent-syncscope.ll        |   4 +-
 llvm/test/CodeGen/X86/atomic-idempotent.ll    |   4 +-
 llvm/test/CodeGen/X86/atomic-xor.ll           |   2 +-
 llvm/test/CodeGen/X86/atomic64.ll             |   8 +-
 llvm/test/CodeGen/X86/avx2-vbroadcast.ll      |  16 +--
 .../test/CodeGen/X86/avx512-insert-extract.ll |  48 ++++----
 .../CodeGen/X86/avx512-insert-extract_i1.ll   |   2 +-
 llvm/test/CodeGen/X86/avx512-intel-ocl.ll     |  12 +-
 llvm/test/CodeGen/X86/avx512fp16-cvt.ll       |   6 +-
 llvm/test/CodeGen/X86/avx512fp16-mov.ll       |   8 +-
 .../X86/bfloat-calling-conv-no-sse2.ll        |   4 +-
 llvm/test/CodeGen/X86/bittest-big-integer.ll  |   6 +-
 .../X86/dagcombine-tokenfactor-limit-crash.ll |   2 +-
 .../X86/div-rem-pair-recomposition-signed.ll  |   2 +-
 .../div-rem-pair-recomposition-unsigned.ll    |   2 +-
 llvm/test/CodeGen/X86/extractelement-index.ll |   8 +-
 llvm/test/CodeGen/X86/extractelement-load.ll  |   6 +-
 llvm/test/CodeGen/X86/fma.ll                  |   4 +-
 .../test/CodeGen/X86/fp128-libcalls-strict.ll |   8 +-
 llvm/test/CodeGen/X86/fp128-libcalls.ll       |  16 +--
 llvm/test/CodeGen/X86/gep-expanded-vector.ll  |   2 +-
 llvm/test/CodeGen/X86/i128-fp128-abi.ll       |   8 +-
 llvm/test/CodeGen/X86/i128-udiv.ll            |  18 +--
 llvm/test/CodeGen/X86/i64-mem-copy.ll         |   4 +-
 llvm/test/CodeGen/X86/inline-sse.ll           |   2 +-
 .../CodeGen/X86/insertelement-var-index.ll    |  64 +++++------
 llvm/test/CodeGen/X86/isel-x87.ll             |   4 +-
 .../test/CodeGen/X86/long-double-abi-align.ll |   4 +-
 llvm/test/CodeGen/X86/matrix-multiply.ll      |   2 +-
 .../X86/memset-sse-stack-realignment.ll       |   8 +-
 llvm/test/CodeGen/X86/mingw-alloca.ll         |   4 +-
 llvm/test/CodeGen/X86/mmx-arith.ll            |   2 +-
 llvm/test/CodeGen/X86/mmx-intrinsics.ll       |   2 +-
 llvm/test/CodeGen/X86/musttail-varargs.ll     |   4 +-
 llvm/test/CodeGen/X86/nontemporal-loads-2.ll  |  60 +++++-----
 llvm/test/CodeGen/X86/nosse-vector.ll         |   2 +-
 llvm/test/CodeGen/X86/packus.ll               |   2 +-
 llvm/test/CodeGen/X86/pr216504.ll             |   4 +-
 llvm/test/CodeGen/X86/pr32284.ll              |   2 +-
 llvm/test/CodeGen/X86/pr34080-2.ll            |   2 +-
 llvm/test/CodeGen/X86/pr34592.ll              |   2 +-
 llvm/test/CodeGen/X86/pr34653.ll              |   2 +-
 llvm/test/CodeGen/X86/pr38539.ll              |   2 +-
 llvm/test/CodeGen/X86/pr43866.ll              |   2 +-
 llvm/test/CodeGen/X86/pr50782.ll              |   2 +-
 llvm/test/CodeGen/X86/sdiv_fix.ll             |   2 +-
 llvm/test/CodeGen/X86/sdiv_fix_sat.ll         |   4 +-
 llvm/test/CodeGen/X86/shift-i128.ll           |  16 +--
 llvm/test/CodeGen/X86/shift-i256.ll           |  28 ++---
 .../CodeGen/X86/shuffle-combine-crash-4.ll    |   2 +-
 llvm/test/CodeGen/X86/sse-intel-ocl.ll        |   4 +-
 .../CodeGen/X86/sse-intrinsics-fast-isel.ll   |   4 +-
 llvm/test/CodeGen/X86/sse-regcall.ll          |   4 +-
 llvm/test/CodeGen/X86/sse-regcall4.ll         |   4 +-
 .../stack-clash-small-alloc-medium-align.ll   |   2 +-
 .../X86/stack-realign-local-padding.ll        | 104 ++++++++++++++++++
 .../X86/statepoint-no-realign-stack.ll        |   4 +-
 llvm/test/CodeGen/X86/sttni.ll                |  16 +--
 llvm/test/CodeGen/X86/udiv_fix.ll             |   2 +-
 llvm/test/CodeGen/X86/udiv_fix_sat.ll         |   2 +-
 llvm/test/CodeGen/X86/var-permute-256.ll      |   6 +-
 llvm/test/CodeGen/X86/var-permute-512.ll      |  12 +-
 llvm/test/CodeGen/X86/vec-strict-128.ll       |   2 +-
 .../CodeGen/X86/vec-strict-fptoint-256.ll     |   8 +-
 llvm/test/CodeGen/X86/vec_ins_extract-1.ll    |   8 +-
 llvm/test/CodeGen/X86/vec_insert-8.ll         |   4 +-
 llvm/test/CodeGen/X86/vector-compress.ll      |  38 +++----
 llvm/test/CodeGen/X86/vector-extend-inreg.ll  |   6 +-
 .../CodeGen/X86/vector-extract-last-active.ll |  12 +-
 llvm/test/CodeGen/X86/vector-llrint.ll        |  10 +-
 llvm/test/CodeGen/X86/vector-lrint.ll         |   8 +-
 llvm/test/CodeGen/X86/vector-reduce-ctpop.ll  |  10 +-
 llvm/test/CodeGen/X86/vector-reduce-smax.ll   |   4 +-
 llvm/test/CodeGen/X86/vector-reduce-smin.ll   |   2 +-
 llvm/test/CodeGen/X86/vector-reduce-umax.ll   |   6 +-
 llvm/test/CodeGen/X86/vector-reduce-umin.ll   |   2 +-
 .../CodeGen/X86/vector-shuffle-512-v16.ll     |   2 +-
 .../X86/vector-shuffle-variable-256.ll        |  16 +--
 llvm/test/CodeGen/X86/widen_arith-6.ll        |   2 +-
 llvm/test/CodeGen/X86/x86-64-baseptr.ll       |   2 +-
 .../test/DebugInfo/COFF/fpo-realign-alloca.ll |   4 +-
 llvm/test/DebugInfo/COFF/vframe-csr.ll        |  12 +-
 92 files changed, 484 insertions(+), 349 deletions(-)
 create mode 100644 llvm/test/CodeGen/X86/stack-realign-local-padding.ll

diff --git a/llvm/lib/Target/X86/X86FrameLowering.cpp b/llvm/lib/Target/X86/X86FrameLowering.cpp
index a25aba6d0afe0a..146e191c2997ba 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.cpp
+++ b/llvm/lib/Target/X86/X86FrameLowering.cpp
@@ -1589,6 +1589,32 @@ static bool isOpcodeRep(unsigned Opcode) {
   return false;
 }
 
+/// Returns the number of bytes between the end of the fixed and callee-save
+/// area and the first local object, which PEI leaves as padding when it aligns
+/// the local objects relative to the incoming stack pointer.
+static uint64_t getUnusedLocalAreaPadding(const MachineFrameInfo &MFI) {
+  int64_t FixedEnd = 0;
+  int64_t LocalsTop = std::numeric_limits<int64_t>::max();
+  for (int I : seq(MFI.getObjectIndexBegin(), MFI.getObjectIndexEnd())) {
+    if (MFI.isDeadObjectIndex(I) || MFI.isVariableSizedObjectIndex(I) ||
+        MFI.getStackID(I) != TargetStackID::Default)
+      continue;
+
+    int64_t ObjOffset = MFI.getObjectOffset(I);
+    int64_t ObjSize = MFI.getObjectSize(I);
+    // Offsets are negative, measured from the incoming stack pointer.
+    if (MFI.isFixedObjectIndex(I))
+      FixedEnd = std::max(FixedEnd, -ObjOffset);
+    else
+      LocalsTop = std::min<int64_t>(LocalsTop, -ObjOffset - ObjSize);
+  }
+  if (LocalsTop == std::numeric_limits<int64_t>::max())
+    return 0;
+
+  assert(LocalsTop >= FixedEnd && "Local object overlaps the fixed area");
+  return LocalsTop - FixedEnd;
+}
+
 /// emitPrologue - Push callee-saved registers onto the stack, which
 /// automatically adjust the stack pointer. Adjust the stack pointer to allocate
 /// space for local variables. Also emit labels used by the exception handler to
@@ -1901,9 +1927,14 @@ void X86FrameLowering::emitPrologue(MachineFunction &MF,
     NumBytes =
         FrameSize - (X86FI->getCalleeSavedFrameSize() + TailCallArgReserveSize);
 
-    // Callee-saved registers are pushed on stack before the stack is realigned.
-    if (TRI->hasStackRealignment(MF) && !IsWin64Prologue)
-      NumBytes = alignTo(NumBytes, MaxAlign);
+    // Callee-saved registers are pushed on stack before the stack is realigned,
+    // and the realignment itself already provides the local objects' alignment,
+    // so leave out the padding PEI put in front of them for it.
+    if (TRI->hasStackRealignment(MF) && !IsWin64Prologue) {
+      uint64_t Padding = getUnusedLocalAreaPadding(MFI);
+      assert(Padding <= NumBytes && "Padding exceeds the local area");
+      NumBytes = alignTo(NumBytes - Padding, MaxAlign);
+    }
 
     // Save EBP/RBP into the appropriate stack slot.
     auto EmitSEHPushFramePtr = [&]() {
diff --git a/llvm/test/CodeGen/X86/2009-06-05-VariableIndexInsert.ll b/llvm/test/CodeGen/X86/2009-06-05-VariableIndexInsert.ll
index 695a2d0cd806e0..99913ae148811f 100644
--- a/llvm/test/CodeGen/X86/2009-06-05-VariableIndexInsert.ll
+++ b/llvm/test/CodeGen/X86/2009-06-05-VariableIndexInsert.ll
@@ -8,7 +8,7 @@ define <2 x i64> @_mm_insert_epi16(<2 x i64> %a, i32 %b, i32 %imm) nounwind read
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $32, %esp
+; X86-NEXT:    subl $16, %esp
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    movzwl 8(%ebp), %ecx
 ; X86-NEXT:    andl $7, %eax
diff --git a/llvm/test/CodeGen/X86/AMX/amx-across-func.ll b/llvm/test/CodeGen/X86/AMX/amx-across-func.ll
index cbd4ee03706426..ba447c494b728d 100644
--- a/llvm/test/CodeGen/X86/AMX/amx-across-func.ll
+++ b/llvm/test/CodeGen/X86/AMX/amx-across-func.ll
@@ -103,7 +103,7 @@ define dso_local void @test_api(i16 signext %0, i16 signext %1) nounwind {
 ; O0-NEXT:    pushq %rbp
 ; O0-NEXT:    movq %rsp, %rbp
 ; O0-NEXT:    andq $-1024, %rsp # imm = 0xFC00
-; O0-NEXT:    subq $8192, %rsp # imm = 0x2000
+; O0-NEXT:    subq $7168, %rsp # imm = 0x1C00
 ; O0-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; O0-NEXT:    # kill: def $zmm0 killed $xmm0
 ; O0-NEXT:    vmovups %zmm0, {{[0-9]+}}(%rsp)
@@ -337,7 +337,7 @@ define dso_local i32 @test_loop(i32 %0) nounwind {
 ; O0-NEXT:    pushq %rbp
 ; O0-NEXT:    movq %rsp, %rbp
 ; O0-NEXT:    andq $-1024, %rsp # imm = 0xFC00
-; O0-NEXT:    subq $4096, %rsp # imm = 0x1000
+; O0-NEXT:    subq $3072, %rsp # imm = 0xC00
 ; O0-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; O0-NEXT:    # kill: def $zmm0 killed $xmm0
 ; O0-NEXT:    vmovups %zmm0, {{[0-9]+}}(%rsp)
@@ -557,7 +557,7 @@ define dso_local void @test_loop2(i32 %0) nounwind {
 ; O0-NEXT:    pushq %rbp
 ; O0-NEXT:    movq %rsp, %rbp
 ; O0-NEXT:    andq $-1024, %rsp # imm = 0xFC00
-; O0-NEXT:    subq $3072, %rsp # imm = 0xC00
+; O0-NEXT:    subq $2048, %rsp # imm = 0x800
 ; O0-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; O0-NEXT:    # kill: def $zmm0 killed $xmm0
 ; O0-NEXT:    vmovups %zmm0, {{[0-9]+}}(%rsp)
diff --git a/llvm/test/CodeGen/X86/AMX/amx-configO2toO0.ll b/llvm/test/CodeGen/X86/AMX/amx-configO2toO0.ll
index 5eb036ca47563e..0690bdcf9cc2ff 100644
--- a/llvm/test/CodeGen/X86/AMX/amx-configO2toO0.ll
+++ b/llvm/test/CodeGen/X86/AMX/amx-configO2toO0.ll
@@ -10,7 +10,7 @@ define dso_local void @test_api(i32 %cond, i16 signext %row, i16 signext %col) n
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-1024, %rsp # imm = 0xFC00
-; AVX512-NEXT:    subq $8192, %rsp # imm = 0x2000
+; AVX512-NEXT:    subq $7168, %rsp # imm = 0x1C00
 ; AVX512-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; AVX512-NEXT:    # kill: def $zmm0 killed $xmm0
 ; AVX512-NEXT:    vmovups %zmm0, {{[0-9]+}}(%rsp)
diff --git a/llvm/test/CodeGen/X86/AMX/amx-zero-config.ll b/llvm/test/CodeGen/X86/AMX/amx-zero-config.ll
index c4901ab136c71e..96d6fb4e7aa5f7 100644
--- a/llvm/test/CodeGen/X86/AMX/amx-zero-config.ll
+++ b/llvm/test/CodeGen/X86/AMX/amx-zero-config.ll
@@ -66,7 +66,7 @@ define void @foo(ptr %buf) nounwind {
 ; AVX512-O0-NEXT:    pushq %rbp
 ; AVX512-O0-NEXT:    movq %rsp, %rbp
 ; AVX512-O0-NEXT:    andq $-1024, %rsp # imm = 0xFC00
-; AVX512-O0-NEXT:    subq $3072, %rsp # imm = 0xC00
+; AVX512-O0-NEXT:    subq $2048, %rsp # imm = 0x800
 ; AVX512-O0-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; AVX512-O0-NEXT:    # kill: def $zmm0 killed $xmm0
 ; AVX512-O0-NEXT:    vmovups %zmm0, {{[0-9]+}}(%rsp)
@@ -107,7 +107,7 @@ define void @foo(ptr %buf) nounwind {
 ; AVX2-O0-NEXT:    pushq %rbp
 ; AVX2-O0-NEXT:    movq %rsp, %rbp
 ; AVX2-O0-NEXT:    andq $-1024, %rsp # imm = 0xFC00
-; AVX2-O0-NEXT:    subq $3072, %rsp # imm = 0xC00
+; AVX2-O0-NEXT:    subq $2048, %rsp # imm = 0x800
 ; AVX2-O0-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; AVX2-O0-NEXT:    # kill: def $ymm0 killed $xmm0
 ; AVX2-O0-NEXT:    vmovups %ymm0, {{[0-9]+}}(%rsp)
@@ -149,7 +149,7 @@ define void @foo(ptr %buf) nounwind {
 ; SSE2-O0-NEXT:    pushq %rbp
 ; SSE2-O0-NEXT:    movq %rsp, %rbp
 ; SSE2-O0-NEXT:    andq $-1024, %rsp # imm = 0xFC00
-; SSE2-O0-NEXT:    subq $3072, %rsp # imm = 0xC00
+; SSE2-O0-NEXT:    subq $2048, %rsp # imm = 0x800
 ; SSE2-O0-NEXT:    xorps %xmm0, %xmm0
 ; SSE2-O0-NEXT:    movups %xmm0, {{[0-9]+}}(%rsp)
 ; SSE2-O0-NEXT:    movups %xmm0, {{[0-9]+}}(%rsp)
diff --git a/llvm/test/CodeGen/X86/amx-across-func-tilemovrow.ll b/llvm/test/CodeGen/X86/amx-across-func-tilemovrow.ll
index d317deadb0354e..bdaafc4e847441 100644
--- a/llvm/test/CodeGen/X86/amx-across-func-tilemovrow.ll
+++ b/llvm/test/CodeGen/X86/amx-across-func-tilemovrow.ll
@@ -94,7 +94,7 @@ define dso_local <16 x i32> @test_api(i16 signext %0, i16 signext %1) nounwind {
 ; O0-NEXT:    pushq %rbp
 ; O0-NEXT:    movq %rsp, %rbp
 ; O0-NEXT:    andq $-1024, %rsp # imm = 0xFC00
-; O0-NEXT:    subq $4096, %rsp # imm = 0x1000
+; O0-NEXT:    subq $3072, %rsp # imm = 0xC00
 ; O0-NEXT:    vpxor %xmm0, %xmm0, %xmm0
 ; O0-NEXT:    # kill: def $zmm0 killed $xmm0
 ; O0-NEXT:    vmovups %zmm0, {{[0-9]+}}(%rsp)
diff --git a/llvm/test/CodeGen/X86/amx_movrs_intrinsics.ll b/llvm/test/CodeGen/X86/amx_movrs_intrinsics.ll
index 1b93ae029f27b9..7e31d502ead2bd 100755
--- a/llvm/test/CodeGen/X86/amx_movrs_intrinsics.ll
+++ b/llvm/test/CodeGen/X86/amx_movrs_intrinsics.ll
@@ -11,7 +11,7 @@ define void @test_amx_internal(i16 %m, i16 %n, ptr %buf, i64 %s) {
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    .cfi_def_cfa_register %rbp
 ; CHECK-NEXT:    andq $-1024, %rsp # imm = 0xFC00
-; CHECK-NEXT:    subq $3072, %rsp # imm = 0xC00
+; CHECK-NEXT:    subq $2048, %rsp # imm = 0x800
 ; CHECK-NEXT:    xorps %xmm0, %xmm0
 ; CHECK-NEXT:    movups %xmm0, {{[0-9]+}}(%rsp)
 ; CHECK-NEXT:    movups %xmm0, {{[0-9]+}}(%rsp)
@@ -46,8 +46,8 @@ define void @test_amx_internal(i16 %m, i16 %n, ptr %buf, i64 %s) {
 ; EGPR-NEXT:    .cfi_def_cfa_register %rbp
 ; EGPR-NEXT:    andq $-1024, %rsp # encoding: [0x48,0x81,0xe4,0x00,0xfc,0xff,0xff]
 ; EGPR-NEXT:    # imm = 0xFC00
-; EGPR-NEXT:    subq $3072, %rsp # encoding: [0x48,0x81,0xec,0x00,0x0c,0x00,0x00]
-; EGPR-NEXT:    # imm = 0xC00
+; EGPR-NEXT:    subq $2048, %rsp # encoding: [0x48,0x81,0xec,0x00,0x08,0x00,0x00]
+; EGPR-NEXT:    # imm = 0x800
 ; EGPR-NEXT:    xorps %xmm0, %xmm0 # encoding: [0x0f,0x57,0xc0]
 ; EGPR-NEXT:    movups %xmm0, {{[0-9]+}}(%rsp) # encoding: [0x0f,0x11,0x84,0x24,0xc0,0x03,0x00,0x00]
 ; EGPR-NEXT:    movups %xmm0, {{[0-9]+}}(%rsp) # encoding: [0x0f,0x11,0x84,0x24,0xd0,0x03,0x00,0x00]
@@ -108,7 +108,7 @@ define void @test_amx_t1_internal(i16 %m, i16 %n, ptr %buf, i64 %s) {
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    .cfi_def_cfa_register %rbp
 ; CHECK-NEXT:    andq $-1024, %rsp # imm = 0xFC00
-; CHECK-NEXT:    subq $3072, %rsp # imm = 0xC00
+; CHECK-NEXT:    subq $2048, %rsp # imm = 0x800
 ; CHECK-NEXT:    xorps %xmm0, %xmm0
 ; CHECK-NEXT:    movups %xmm0, {{[0-9]+}}(%rsp)
 ; CHECK-NEXT:    movups %xmm0, {{[0-9]+}}(%rsp)
@@ -143,8 +143,8 @@ define void @test_amx_t1_internal(i16 %m, i16 %n, ptr %buf, i64 %s) {
 ; EGPR-NEXT:    .cfi_def_cfa_register %rbp
 ; EGPR-NEXT:    andq $-1024, %rsp # encoding: [0x48,0x81,0xe4,0x00,0xfc,0xff,0xff]
 ; EGPR-NEXT:    # imm = 0xFC00
-; EGPR-NEXT:    subq $3072, %rsp # encoding: [0x48,0x81,0xec,0x00,0x0c,0x00,0x00]
-; EGPR-NEXT:    # imm = 0xC00
+; EGPR-NEXT:    subq $2048, %rsp # encoding: [0x48,0x81,0xec,0x00,0x08,0x00,0x00]
+; EGPR-NEXT:    # imm = 0x800
 ; EGPR-NEXT:    xorps %xmm0, %xmm0 # encoding: [0x0f,0x57,0xc0]
 ; EGPR-NEXT:    movups %xmm0, {{[0-9]+}}(%rsp) # encoding: [0x0f,0x11,0x84,0x24,0xc0,0x03,0x00,0x00]
 ; EGPR-NEXT:    movups %xmm0, {{[0-9]+}}(%rsp) # encoding: [0x0f,0x11,0x84,0x24,0xd0,0x03,0x00,0x00]
diff --git a/llvm/test/CodeGen/X86/andnot-sink-not.ll b/llvm/test/CodeGen/X86/andnot-sink-not.ll
index fefbdc84699f44..bd6442d5edb0a8 100644
--- a/llvm/test/CodeGen/X86/andnot-sink-not.ll
+++ b/llvm/test/CodeGen/X86/andnot-sink-not.ll
@@ -1018,7 +1018,7 @@ define <4 x i32> @and_sink_not_v4i32(<4 x i32> %x, <4 x i32> %m, i1 zeroext %con
 ; X86-SSE-NEXT:    pushl %edi
 ; X86-SSE-NEXT:    pushl %esi
 ; X86-SSE-NEXT:    andl $-16, %esp
-; X86-SSE-NEXT:    subl $64, %esp
+; X86-SSE-NEXT:    subl $48, %esp
 ; X86-SSE-NEXT:    movl 8(%ebp), %eax
 ; X86-SSE-NEXT:    movl 24(%ebp), %ecx
 ; X86-SSE-NEXT:    movl 20(%ebp), %edx
@@ -1190,7 +1190,7 @@ define <4 x i32> @and_sink_not_v4i32_swapped(<4 x i32> %x, <4 x i32> %m, i1 zero
 ; X86-SSE-NEXT:    pushl %edi
 ; X86-SSE-NEXT:    pushl %esi
 ; X86-SSE-NEXT:    andl $-16, %esp
-; X86-SSE-NEXT:    subl $64, %esp
+; X86-SSE-NEXT:    subl $48, %esp
 ; X86-SSE-NEXT:    movl 8(%ebp), %eax
 ; X86-SSE-NEXT:    movl 24(%ebp), %ecx
 ; X86-SSE-NEXT:    movl 20(%ebp), %edx
diff --git a/llvm/test/CodeGen/X86/arg-copy-elide.ll b/llvm/test/CodeGen/X86/arg-copy-elide.ll
index 15edb612d7649d..b51f32f2b9675d 100644
--- a/llvm/test/CodeGen/X86/arg-copy-elide.ll
+++ b/llvm/test/CodeGen/X86/arg-copy-elide.ll
@@ -187,7 +187,7 @@ define void @split_i128(ptr %sret, i128 %x) {
 ; CHECK-NEXT:    pushl %edi
 ; CHECK-NEXT:    pushl %esi
 ; CHECK-NEXT:    andl $-16, %esp
-; CHECK-NEXT:    subl $48, %esp
+; CHECK-NEXT:    subl $32, %esp
 ; CHECK-NEXT:    movl 24(%ebp), %eax
 ; CHECK-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
 ; CHECK-NEXT:    movl 28(%ebp), %ebx
diff --git a/llvm/test/CodeGen/X86/atomic-fp.ll b/llvm/test/CodeGen/X86/atomic-fp.ll
index 2dee1d12e72558..36540422a139c0 100644
--- a/llvm/test/CodeGen/X86/atomic-fp.ll
+++ b/llvm/test/CodeGen/X86/atomic-fp.ll
@@ -625,7 +625,7 @@ define dso_local void @fadd_array(ptr %arg, double %arg1, i64 %arg2) nounwind {
 ; X86-NOSSE-NEXT:    movl %esp, %ebp
 ; X86-NOSSE-NEXT:    pushl %esi
 ; X86-NOSSE-NEXT:    andl $-8, %esp
-; X86-NOSSE-NEXT:    subl $32, %esp
+; X86-NOSSE-NEXT:    subl $24, %esp
 ; X86-NOSSE-NEXT:    movl 20(%ebp), %eax
 ; X86-NOSSE-NEXT:    movl 8(%ebp), %ecx
 ; X86-NOSSE-NEXT:    fildll (%ecx,%eax,8)
@@ -1339,7 +1339,7 @@ define dso_local void @fsub_array(ptr %arg, double %arg1, i64 %arg2) nounwind {
 ; X86-NOSSE-NEXT:    movl %esp, %ebp
 ; X86-NOSSE-NEXT:    pushl %esi
 ; X86-NOSSE-NEXT:    andl $-8, %esp
-; X86-NOSSE-NEXT:    subl $32, %esp
+; X86-NOSSE-NEXT:    subl $24, %esp
 ; X86-NOSSE-NEXT:    movl 20(%ebp), %eax
 ; X86-NOSSE-NEXT:    movl 8(%ebp), %ecx
 ; X86-NOSSE-NEXT:    fildll (%ecx,%eax,8)
@@ -2043,7 +2043,7 @@ define dso_local void @fmul_array(ptr %arg, double %arg1, i64 %arg2) nounwind {
 ; X86-NOSSE-NEXT:    movl %esp, %ebp
 ; X86-NOSSE-NEXT:    pushl %esi
 ; X86-NOSSE-NEXT:    andl $-8, %esp
-; X86-NOSSE-NEXT:    subl $32, %esp
+; X86-NOSSE-NEXT:    subl $24, %esp
 ; X86-NOSSE-NEXT:    movl 20(%ebp), %eax
 ; X86-NOSSE-NEXT:    movl 8(%ebp), %ecx
 ; X86-NOSSE-NEXT:    fildll (%ecx,%eax,8)
@@ -2751,7 +2751,7 @@ define dso_local void @fdiv_array(ptr %arg, double %arg1, i64 %arg2) nounwind {
 ; X86-NOSSE-NEXT:    movl %esp, %ebp
 ; X86-NOSSE-NEXT:    pushl %esi
 ; X86-NOSSE-NEXT:    andl $-8, %esp
-; X86-NOSSE-NEXT:    subl $32, %esp
+; X86-NOSSE-NEXT:    subl $24, %esp
 ; X86-NOSSE-NEXT:    movl 20(%ebp), %eax
 ; X86-NOSSE-NEXT:    movl 8(%ebp), %ecx
 ; X86-NOSSE-NEXT:    fildll (%ecx,%eax,8)
diff --git a/llvm/test/CodeGen/X86/atomic-idempotent-syncscope.ll b/llvm/test/CodeGen/X86/atomic-idempotent-syncscope.ll
index 9e20fdb59f552c..2502e212f9ffd3 100644
--- a/llvm/test/CodeGen/X86/atomic-idempotent-syncscope.ll
+++ b/llvm/test/CodeGen/X86/atomic-idempotent-syncscope.ll
@@ -143,7 +143,7 @@ define i128 @or128(ptr %p) #0 {
 ; X86-GENERIC-NEXT:    pushl %edi
 ; X86-GENERIC-NEXT:    pushl %esi
 ; X86-GENERIC-NEXT:    andl $-16, %esp
-; X86-GENERIC-NEXT:    subl $48, %esp
+; X86-GENERIC-NEXT:    subl $32, %esp
 ; X86-GENERIC-NEXT:    movl 12(%ebp), %edi
 ; X86-GENERIC-NEXT:    movl 12(%edi), %ecx
 ; X86-GENERIC-NEXT:    movl 8(%edi), %edx
@@ -460,7 +460,7 @@ define void @or128_nouse_seq_cst(ptr %p) #0 {
 ; X86-GENERIC-NEXT:    pushl %edi
 ; X86-GENERIC-NEXT:    pushl %esi
 ; X86-GENERIC-NEXT:    andl $-16, %esp
-; X86-GENERIC-NEXT:    subl $48, %esp
+; X86-GENERIC-NEXT:    subl $32, %esp
 ; X86-GENERIC-NEXT:    movl 8(%ebp), %esi
 ; X86-GENERIC-NEXT:    movl 12(%esi), %ecx
 ; X86-GENERIC-NEXT:    movl 8(%esi), %edi
diff --git a/llvm/test/CodeGen/X86/atomic-idempotent.ll b/llvm/test/CodeGen/X86/atomic-idempotent.ll
index 01c3e7999a92ce..2516604b4d2af2 100644
--- a/llvm/test/CodeGen/X86/atomic-idempotent.ll
+++ b/llvm/test/CodeGen/X86/atomic-idempotent.ll
@@ -158,7 +158,7 @@ define i128 @or128(ptr %p) #0 {
 ; X86-GENERIC-NEXT:    pushl %edi
 ; X86-GENERIC-NEXT:    pushl %esi
 ; X86-GENERIC-NEXT:    andl $-16, %esp
-; X86-GENERIC-NEXT:    subl $48, %esp
+; X86-GENERIC-NEXT:    subl $32, %esp
 ; X86-GENERIC-NEXT:    movl 12(%ebp), %edi
 ; X86-GENERIC-NEXT:    movl 12(%edi), %ecx
 ; X86-GENERIC-NEXT:    movl 8(%edi), %edx
@@ -478,7 +478,7 @@ define void @or128_nouse_seq_cst(ptr %p) #0 {
 ; X86-GENERIC-NEXT:    pushl %edi
 ; X86-GENERIC-NEXT:    pushl %esi
 ; X86-GENERIC-NEXT:    andl $-16, %esp
-; X86-GENERIC-NEXT:    subl $48, %esp
+; X86-GENERIC-NEXT:    subl $32, %esp
 ; X86-GENERIC-NEXT:    movl 8(%ebp), %esi
 ; X86-GENERIC-NEXT:    movl 12(%esi), %ecx
 ; X86-GENERIC-NEXT:    movl 8(%esi), %edi
diff --git a/llvm/test/CodeGen/X86/atomic-xor.ll b/llvm/test/CodeGen/X86/atomic-xor.ll
index c648ecdfbe674b..cad3f2e84c3b67 100644
--- a/llvm/test/CodeGen/X86/atomic-xor.ll
+++ b/llvm/test/CodeGen/X86/atomic-xor.ll
@@ -26,7 +26,7 @@ define i128 @xor128_signbit_used(ptr %p) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    movl 12(%ebp), %edi
 ; X86-NEXT:    movl 12(%edi), %ecx
 ; X86-NEXT:    movl 8(%edi), %edx
diff --git a/llvm/test/CodeGen/X86/atomic64.ll b/llvm/test/CodeGen/X86/atomic64.ll
index 8f4da356e06cbb..d3bafbb06cc109 100644
--- a/llvm/test/CodeGen/X86/atomic64.ll
+++ b/llvm/test/CodeGen/X86/atomic64.ll
@@ -328,7 +328,7 @@ define void @atomic_fetch_max64(i64 %x) nounwind {
 ; I486-NEXT:    movl %esp, %ebp
 ; I486-NEXT:    pushl %esi
 ; I486-NEXT:    andl $-8, %esp
-; I486-NEXT:    subl $72, %esp
+; I486-NEXT:    subl $64, %esp
 ; I486-NEXT:    movl 12(%ebp), %eax
 ; I486-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
 ; I486-NEXT:    movl 8(%ebp), %eax
@@ -420,7 +420,7 @@ define void @atomic_fetch_min64(i64 %x) nounwind {
 ; I486-NEXT:    movl %esp, %ebp
 ; I486-NEXT:    pushl %esi
 ; I486-NEXT:    andl $-8, %esp
-; I486-NEXT:    subl $72, %esp
+; I486-NEXT:    subl $64, %esp
 ; I486-NEXT:    movl 12(%ebp), %eax
 ; I486-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
 ; I486-NEXT:    movl 8(%ebp), %eax
@@ -512,7 +512,7 @@ define void @atomic_fetch_umax64(i64 %x) nounwind {
 ; I486-NEXT:    movl %esp, %ebp
 ; I486-NEXT:    pushl %esi
 ; I486-NEXT:    andl $-8, %esp
-; I486-NEXT:    subl $72, %esp
+; I486-NEXT:    subl $64, %esp
 ; I486-NEXT:    movl 12(%ebp), %eax
 ; I486-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
 ; I486-NEXT:    movl 8(%ebp), %eax
@@ -604,7 +604,7 @@ define void @atomic_fetch_umin64(i64 %x) nounwind {
 ; I486-NEXT:    movl %esp, %ebp
 ; I486-NEXT:    pushl %esi
 ; I486-NEXT:    andl $-8, %esp
-; I486-NEXT:    subl $72, %esp
+; I486-NEXT:    subl $64, %esp
 ; I486-NEXT:    movl 12(%ebp), %eax
 ; I486-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
 ; I486-NEXT:    movl 8(%ebp), %eax
diff --git a/llvm/test/CodeGen/X86/avx2-vbroadcast.ll b/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
index a5601f17256828..311f74a2bce9ec 100644
--- a/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
+++ b/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
@@ -1133,7 +1133,7 @@ define void @isel_crash_32b(ptr %cV_R.addr) {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-NEXT:    andl $-32, %esp
-; X86-NEXT:    addl $-128, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X86-NEXT:    vmovaps %ymm0, (%esp)
@@ -1153,7 +1153,7 @@ define void @isel_crash_32b(ptr %cV_R.addr) {
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    .cfi_def_cfa_register %rbp
 ; X64-NEXT:    andq $-32, %rsp
-; X64-NEXT:    addq $-128, %rsp
+; X64-NEXT:    subq $96, %rsp
 ; X64-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X64-NEXT:    vmovaps %ymm0, (%rsp)
 ; X64-NEXT:    vpbroadcastb (%rdi), %ymm1
@@ -1224,7 +1224,7 @@ define void @isel_crash_16w(ptr %cV_R.addr) {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-NEXT:    andl $-32, %esp
-; X86-NEXT:    addl $-128, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X86-NEXT:    vmovaps %ymm0, (%esp)
@@ -1244,7 +1244,7 @@ define void @isel_crash_16w(ptr %cV_R.addr) {
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    .cfi_def_cfa_register %rbp
 ; X64-NEXT:    andq $-32, %rsp
-; X64-NEXT:    addq $-128, %rsp
+; X64-NEXT:    subq $96, %rsp
 ; X64-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X64-NEXT:    vmovaps %ymm0, (%rsp)
 ; X64-NEXT:    vpbroadcastw (%rdi), %ymm1
@@ -1315,7 +1315,7 @@ define void @isel_crash_8d(ptr %cV_R.addr) {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-NEXT:    andl $-32, %esp
-; X86-NEXT:    addl $-128, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X86-NEXT:    vmovaps %ymm0, (%esp)
@@ -1335,7 +1335,7 @@ define void @isel_crash_8d(ptr %cV_R.addr) {
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    .cfi_def_cfa_register %rbp
 ; X64-NEXT:    andq $-32, %rsp
-; X64-NEXT:    addq $-128, %rsp
+; X64-NEXT:    subq $96, %rsp
 ; X64-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X64-NEXT:    vmovaps %ymm0, (%rsp)
 ; X64-NEXT:    vbroadcastss (%rdi), %ymm1
@@ -1405,7 +1405,7 @@ define void @isel_crash_4q(ptr %cV_R.addr) {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-NEXT:    andl $-32, %esp
-; X86-NEXT:    addl $-128, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X86-NEXT:    vmovaps %ymm0, (%esp)
@@ -1425,7 +1425,7 @@ define void @isel_crash_4q(ptr %cV_R.addr) {
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    .cfi_def_cfa_register %rbp
 ; X64-NEXT:    andq $-32, %rsp
-; X64-NEXT:    addq $-128, %rsp
+; X64-NEXT:    subq $96, %rsp
 ; X64-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X64-NEXT:    vmovaps %ymm0, (%rsp)
 ; X64-NEXT:    vbroadcastsd (%rdi), %ymm1
diff --git a/llvm/test/CodeGen/X86/avx512-insert-extract.ll b/llvm/test/CodeGen/X86/avx512-insert-extract.ll
index 4efc4678f1d002..9458f25ac85155 100644
--- a/llvm/test/CodeGen/X86/avx512-insert-extract.ll
+++ b/llvm/test/CodeGen/X86/avx512-insert-extract.ll
@@ -101,7 +101,7 @@ define float @test7(<16 x float> %x, i32 %ind) nounwind {
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    subq $64, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $15, %edi
@@ -120,7 +120,7 @@ define double @test8(<8 x double> %x, i32 %ind) nounwind {
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    subq $64, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $7, %edi
@@ -139,7 +139,7 @@ define float @test9(<8 x float> %x, i32 %ind) nounwind {
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %ymm0, (%rsp)
 ; CHECK-NEXT:    andl $7, %edi
@@ -158,7 +158,7 @@ define i32 @test10(<16 x i32> %x, i32 %ind) nounwind {
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    subq $64, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $15, %edi
@@ -1129,7 +1129,7 @@ define i64 @test_extractelement_variable_v4i64(<4 x i64> %t1, i32 %index) nounwi
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %ymm0, (%rsp)
 ; CHECK-NEXT:    andl $3, %edi
@@ -1148,7 +1148,7 @@ define i64 @test_extractelement_variable_v8i64(<8 x i64> %t1, i32 %index) nounwi
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    subq $64, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $7, %edi
@@ -1179,7 +1179,7 @@ define double @test_extractelement_variable_v4f64(<4 x double> %t1, i32 %index)
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %ymm0, (%rsp)
 ; CHECK-NEXT:    andl $3, %edi
@@ -1198,7 +1198,7 @@ define double @test_extractelement_variable_v8f64(<8 x double> %t1, i32 %index)
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    subq $64, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $7, %edi
@@ -1229,7 +1229,7 @@ define i32 @test_extractelement_variable_v8i32(<8 x i32> %t1, i32 %index) nounwi
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %ymm0, (%rsp)
 ; CHECK-NEXT:    andl $7, %edi
@@ -1248,7 +1248,7 @@ define i32 @test_extractelement_variable_v16i32(<16 x i32> %t1, i32 %index) noun
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    subq $64, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $15, %edi
@@ -1279,7 +1279,7 @@ define float @test_extractelement_variable_v8f32(<8 x float> %t1, i32 %index) no
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %ymm0, (%rsp)
 ; CHECK-NEXT:    andl $7, %edi
@@ -1298,7 +1298,7 @@ define float @test_extractelement_variable_v16f32(<16 x float> %t1, i32 %index)
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    subq $64, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $15, %edi
@@ -1329,7 +1329,7 @@ define i16 @test_extractelement_variable_v16i16(<16 x i16> %t1, i32 %index) noun
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %ymm0, (%rsp)
 ; CHECK-NEXT:    andl $15, %edi
@@ -1348,7 +1348,7 @@ define i16 @test_extractelement_variable_v32i16(<32 x i16> %t1, i32 %index) noun
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    subq $64, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $31, %edi
@@ -1379,7 +1379,7 @@ define i8 @test_extractelement_variable_v32i8(<32 x i8> %t1, i32 %index) nounwin
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %ymm0, (%rsp)
 ; CHECK-NEXT:    andl $31, %edi
@@ -1399,7 +1399,7 @@ define i8 @test_extractelement_variable_v64i8(<64 x i8> %t1, i32 %index) nounwin
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    subq $64, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $63, %edi
@@ -1419,7 +1419,7 @@ define i8 @test_extractelement_variable_v64i8_indexi8(<64 x i8> %t1, i8 %index)
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    subq $64, %rsp
 ; CHECK-NEXT:    addb %dil, %dil
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    movzbl %dil, %eax
@@ -1577,7 +1577,7 @@ define zeroext i8 @test_extractelement_varible_v32i1(<32 x i8> %a, <32 x i8> %b,
 ; SKX-NEXT:    pushq %rbp
 ; SKX-NEXT:    movq %rsp, %rbp
 ; SKX-NEXT:    andq $-32, %rsp
-; SKX-NEXT:    subq $64, %rsp
+; SKX-NEXT:    subq $32, %rsp
 ; SKX-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; SKX-NEXT:    vpcmpnleub %ymm1, %ymm0, %k0
 ; SKX-NEXT:    vpmovm2b %k0, %ymm0
@@ -1613,7 +1613,7 @@ define i32 @test_insertelement_variable_v32i1(<32 x i8> %a, i8 %b, i32 %index) n
 ; KNL-NEXT:    pushq %rbp
 ; KNL-NEXT:    movq %rsp, %rbp
 ; KNL-NEXT:    andq $-32, %rsp
-; KNL-NEXT:    subq $64, %rsp
+; KNL-NEXT:    subq $32, %rsp
 ; KNL-NEXT:    ## kill: def $esi killed $esi def $rsi
 ; KNL-NEXT:    vpxor %xmm1, %xmm1, %xmm1
 ; KNL-NEXT:    vpcmpeqb %ymm1, %ymm0, %ymm0
@@ -1663,7 +1663,7 @@ define i64 @test_insertelement_variable_v64i1(<64 x i8> %a, i8 %b, i32 %index) n
 ; KNL-NEXT:    pushq %rbp
 ; KNL-NEXT:    movq %rsp, %rbp
 ; KNL-NEXT:    andq $-64, %rsp
-; KNL-NEXT:    addq $-128, %rsp
+; KNL-NEXT:    subq $64, %rsp
 ; KNL-NEXT:    ## kill: def $esi killed $esi def $rsi
 ; KNL-NEXT:    vpxor %xmm1, %xmm1, %xmm1
 ; KNL-NEXT:    vextracti64x4 $1, %zmm0, %ymm2
@@ -1729,7 +1729,7 @@ define i96 @test_insertelement_variable_v96i1(<96 x i8> %a, i8 %b, i32 %index) n
 ; KNL-NEXT:    pushq %rbp
 ; KNL-NEXT:    movq %rsp, %rbp
 ; KNL-NEXT:    andq $-64, %rsp
-; KNL-NEXT:    subq $192, %rsp
+; KNL-NEXT:    addq $-128, %rsp
 ; KNL-NEXT:    movl 744(%rbp), %eax
 ; KNL-NEXT:    andl $127, %eax
 ; KNL-NEXT:    vmovdqu64 224(%rbp), %zmm0
@@ -1852,7 +1852,7 @@ define i96 @test_insertelement_variable_v96i1(<96 x i8> %a, i8 %b, i32 %index) n
 ; SKX-NEXT:    pushq %rbp
 ; SKX-NEXT:    movq %rsp, %rbp
 ; SKX-NEXT:    andq $-64, %rsp
-; SKX-NEXT:    subq $192, %rsp
+; SKX-NEXT:    addq $-128, %rsp
 ; SKX-NEXT:    vmovd %edi, %xmm0
 ; SKX-NEXT:    vpinsrb $1, %esi, %xmm0, %xmm0
 ; SKX-NEXT:    vpinsrb $2, %edx, %xmm0, %xmm0
@@ -1934,7 +1934,7 @@ define i128 @test_insertelement_variable_v128i1(<128 x i8> %a, i8 %b, i32 %index
 ; KNL-NEXT:    pushq %rbp
 ; KNL-NEXT:    movq %rsp, %rbp
 ; KNL-NEXT:    andq $-64, %rsp
-; KNL-NEXT:    subq $192, %rsp
+; KNL-NEXT:    addq $-128, %rsp
 ; KNL-NEXT:    ## kill: def $esi killed $esi def $rsi
 ; KNL-NEXT:    vpxor %xmm2, %xmm2, %xmm2
 ; KNL-NEXT:    vextracti64x4 $1, %zmm0, %ymm3
@@ -2006,7 +2006,7 @@ define i128 @test_insertelement_variable_v128i1(<128 x i8> %a, i8 %b, i32 %index
 ; SKX-NEXT:    pushq %rbp
 ; SKX-NEXT:    movq %rsp, %rbp
 ; SKX-NEXT:    andq $-64, %rsp
-; SKX-NEXT:    subq $192, %rsp
+; SKX-NEXT:    addq $-128, %rsp
 ; SKX-NEXT:    ## kill: def $esi killed $esi def $rsi
 ; SKX-NEXT:    vptestmb %zmm0, %zmm0, %k0
 ; SKX-NEXT:    vptestmb %zmm1, %zmm1, %k1
diff --git a/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll b/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
index adb8bec6bb5bf8..b77e7095e8aa74 100644
--- a/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
+++ b/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
@@ -12,7 +12,7 @@ define zeroext i8 @test_extractelement_varible_v64i1(<64 x i8> %a, <64 x i8> %b,
 ; SKX-NEXT:    movq %rsp, %rbp
 ; SKX-NEXT:    .cfi_def_cfa_register %rbp
 ; SKX-NEXT:    andq $-64, %rsp
-; SKX-NEXT:    addq $-128, %rsp
+; SKX-NEXT:    subq $64, %rsp
 ; SKX-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; SKX-NEXT:    vpcmpnleub %zmm1, %zmm0, %k0
 ; SKX-NEXT:    vpmovm2b %k0, %zmm0
diff --git a/llvm/test/CodeGen/X86/avx512-intel-ocl.ll b/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
index 0fa67a264fcf01..310762d1ba9ee7 100644
--- a/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
+++ b/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
@@ -19,7 +19,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
 ; X32-NEXT:    pushl %ebp
 ; X32-NEXT:    movl %esp, %ebp
 ; X32-NEXT:    andl $-64, %esp
-; X32-NEXT:    subl $192, %esp
+; X32-NEXT:    addl $-128, %esp
 ; X32-NEXT:    vaddps %zmm1, %zmm0, %zmm0
 ; X32-NEXT:    leal {{[0-9]+}}(%esp), %eax
 ; X32-NEXT:    movl %eax, (%esp)
@@ -34,7 +34,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
 ; WIN32-NEXT:    pushl %ebp
 ; WIN32-NEXT:    movl %esp, %ebp
 ; WIN32-NEXT:    andl $-64, %esp
-; WIN32-NEXT:    addl $-128, %esp
+; WIN32-NEXT:    subl $64, %esp
 ; WIN32-NEXT:    vaddps %zmm1, %zmm0, %zmm0
 ; WIN32-NEXT:    movl %esp, %eax
 ; WIN32-NEXT:    pushl %eax
@@ -67,7 +67,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
 ; X64-NEXT:    pushq %r13
 ; X64-NEXT:    pushq %r12
 ; X64-NEXT:    andq $-64, %rsp
-; X64-NEXT:    addq $-128, %rsp
+; X64-NEXT:    subq $64, %rsp
 ; X64-NEXT:    vaddps %zmm1, %zmm0, %zmm0
 ; X64-NEXT:    movq %rsp, %rdi
 ; X64-NEXT:    pushq %rbp
@@ -97,7 +97,7 @@ define <16 x float> @testf16_regs(<16 x float> %a, <16 x float> %b) nounwind {
 ; X32-NEXT:    pushl %ebp
 ; X32-NEXT:    movl %esp, %ebp
 ; X32-NEXT:    andl $-64, %esp
-; X32-NEXT:    subl $256, %esp ## imm = 0x100
+; X32-NEXT:    subl $192, %esp
 ; X32-NEXT:    vmovaps %zmm1, {{[-0-9]+}}(%e{{[sb]}}p) ## 64-byte Spill
 ; X32-NEXT:    vaddps %zmm1, %zmm0, %zmm0
 ; X32-NEXT:    leal {{[0-9]+}}(%esp), %eax
@@ -114,7 +114,7 @@ define <16 x float> @testf16_regs(<16 x float> %a, <16 x float> %b) nounwind {
 ; WIN32-NEXT:    pushl %ebp
 ; WIN32-NEXT:    movl %esp, %ebp
 ; WIN32-NEXT:    andl $-64, %esp
-; WIN32-NEXT:    subl $192, %esp
+; WIN32-NEXT:    addl $-128, %esp
 ; WIN32-NEXT:    vmovaps %zmm1, (%esp) # 64-byte Spill
 ; WIN32-NEXT:    vaddps %zmm1, %zmm0, %zmm0
 ; WIN32-NEXT:    leal {{[0-9]+}}(%esp), %eax
@@ -150,7 +150,7 @@ define <16 x float> @testf16_regs(<16 x float> %a, <16 x float> %b) nounwind {
 ; X64-NEXT:    pushq %r13
 ; X64-NEXT:    pushq %r12
 ; X64-NEXT:    andq $-64, %rsp
-; X64-NEXT:    addq $-128, %rsp
+; X64-NEXT:    subq $64, %rsp
 ; X64-NEXT:    vmovaps %zmm1, %zmm16
 ; X64-NEXT:    vaddps %zmm1, %zmm0, %zmm0
 ; X64-NEXT:    movq %rsp, %rdi
diff --git a/llvm/test/CodeGen/X86/avx512fp16-cvt.ll b/llvm/test/CodeGen/X86/avx512fp16-cvt.ll
index cc58bc1e44f37c..9dda386ca88db6 100644
--- a/llvm/test/CodeGen/X86/avx512fp16-cvt.ll
+++ b/llvm/test/CodeGen/X86/avx512fp16-cvt.ll
@@ -819,7 +819,7 @@ define i128 @half_to_s128(half %x) {
 ; X86-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    .cfi_offset %esi, -12
 ; X86-NEXT:    movl 8(%ebp), %esi
 ; X86-NEXT:    vmovsh {{.*#+}} xmm0 = mem[0],zero,zero,zero,zero,zero,zero,zero
@@ -922,7 +922,7 @@ define i128 @half_to_u128(half %x) {
 ; X86-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    .cfi_offset %esi, -12
 ; X86-NEXT:    movl 8(%ebp), %esi
 ; X86-NEXT:    vmovsh {{.*#+}} xmm0 = mem[0],zero,zero,zero,zero,zero,zero,zero
@@ -1002,7 +1002,7 @@ define fp128 @half_to_f128(half %x) nounwind {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    movl 8(%ebp), %esi
 ; X86-NEXT:    vmovsh {{.*#+}} xmm0 = mem[0],zero,zero,zero,zero,zero,zero,zero
 ; X86-NEXT:    vcvtsh2ss %xmm0, %xmm0, %xmm0
diff --git a/llvm/test/CodeGen/X86/avx512fp16-mov.ll b/llvm/test/CodeGen/X86/avx512fp16-mov.ll
index e2f2688b1d9f3c..24bcfdfb2639bd 100644
--- a/llvm/test/CodeGen/X86/avx512fp16-mov.ll
+++ b/llvm/test/CodeGen/X86/avx512fp16-mov.ll
@@ -1640,7 +1640,7 @@ define half @extract_f16_8(<32 x half> %x, i64 %idx) nounwind {
 ; X64-NEXT:    pushq %rbp
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    andq $-64, %rsp
-; X64-NEXT:    addq $-128, %rsp
+; X64-NEXT:    subq $64, %rsp
 ; X64-NEXT:    andl $31, %edi
 ; X64-NEXT:    vmovaps %zmm0, (%rsp)
 ; X64-NEXT:    vmovsh {{.*#+}} xmm0 = mem[0],zero,zero,zero,zero,zero,zero,zero
@@ -1654,7 +1654,7 @@ define half @extract_f16_8(<32 x half> %x, i64 %idx) nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-64, %esp
-; X86-NEXT:    addl $-128, %esp
+; X86-NEXT:    subl $64, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    andl $31, %eax
 ; X86-NEXT:    vmovaps %zmm0, (%esp)
@@ -1673,7 +1673,7 @@ define half @extract_f16_9(<64 x half> %x, i64 %idx) nounwind {
 ; X64-NEXT:    pushq %rbp
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    andq $-64, %rsp
-; X64-NEXT:    subq $192, %rsp
+; X64-NEXT:    addq $-128, %rsp
 ; X64-NEXT:    andl $63, %edi
 ; X64-NEXT:    vmovaps %zmm1, {{[0-9]+}}(%rsp)
 ; X64-NEXT:    vmovaps %zmm0, (%rsp)
@@ -1688,7 +1688,7 @@ define half @extract_f16_9(<64 x half> %x, i64 %idx) nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-64, %esp
-; X86-NEXT:    subl $192, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    andl $63, %eax
 ; X86-NEXT:    vmovaps %zmm1, {{[0-9]+}}(%esp)
diff --git a/llvm/test/CodeGen/X86/bfloat-calling-conv-no-sse2.ll b/llvm/test/CodeGen/X86/bfloat-calling-conv-no-sse2.ll
index f363cad816dfb2..1f59f4bf9d10df 100644
--- a/llvm/test/CodeGen/X86/bfloat-calling-conv-no-sse2.ll
+++ b/llvm/test/CodeGen/X86/bfloat-calling-conv-no-sse2.ll
@@ -949,7 +949,7 @@ define void @call_ret_v16bf16(ptr %ptr) #0 {
 ; NOSSE-NEXT:    pushl %edi
 ; NOSSE-NEXT:    pushl %esi
 ; NOSSE-NEXT:    andl $-32, %esp
-; NOSSE-NEXT:    subl $256, %esp # imm = 0x100
+; NOSSE-NEXT:    subl $224, %esp
 ; NOSSE-NEXT:    movl 8(%ebp), %esi
 ; NOSSE-NEXT:    movzwl 2(%esi), %eax
 ; NOSSE-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -1096,7 +1096,7 @@ define void @call_ret_v16bf16(ptr %ptr) #0 {
 ; SSE-NEXT:    pushl %edi
 ; SSE-NEXT:    pushl %esi
 ; SSE-NEXT:    andl $-32, %esp
-; SSE-NEXT:    subl $256, %esp # imm = 0x100
+; SSE-NEXT:    subl $224, %esp
 ; SSE-NEXT:    movl 8(%ebp), %esi
 ; SSE-NEXT:    movzwl 2(%esi), %eax
 ; SSE-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
diff --git a/llvm/test/CodeGen/X86/bittest-big-integer.ll b/llvm/test/CodeGen/X86/bittest-big-integer.ll
index 6767cc45a2c5e7..a470f8f1ac7636 100644
--- a/llvm/test/CodeGen/X86/bittest-big-integer.ll
+++ b/llvm/test/CodeGen/X86/bittest-big-integer.ll
@@ -894,7 +894,7 @@ define <8 x i16> @complement_ne_i128_bitcast(ptr %word, i32 %position) nounwind
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $80, %esp
+; X86-NEXT:    subl $64, %esp
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    movzwl (%eax), %ecx
 ; X86-NEXT:    movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -1208,7 +1208,7 @@ define i1 @sequence_i128(ptr %word, i32 %pos0, i32 %pos1, i32 %pos2) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $144, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movb 20(%ebp), %ch
 ; X86-NEXT:    movb 12(%ebp), %cl
 ; X86-NEXT:    movl $0, {{[0-9]+}}(%esp)
@@ -1439,7 +1439,7 @@ define i32 @blsr_u512(ptr %word) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $240, %esp
+; X86-NEXT:    subl $224, %esp
 ; X86-NEXT:    movl 8(%ebp), %ebx
 ; X86-NEXT:    movl 12(%ebx), %esi
 ; X86-NEXT:    movl 28(%ebx), %eax
diff --git a/llvm/test/CodeGen/X86/dagcombine-tokenfactor-limit-crash.ll b/llvm/test/CodeGen/X86/dagcombine-tokenfactor-limit-crash.ll
index fce59ff485efee..51021cdeb1f751 100644
--- a/llvm/test/CodeGen/X86/dagcombine-tokenfactor-limit-crash.ll
+++ b/llvm/test/CodeGen/X86/dagcombine-tokenfactor-limit-crash.ll
@@ -8,7 +8,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 ; CHECK:          pushq   %rbx
 ; CHECK-NEXT:     andq    $-32, %rsp
-; CHECK-NEXT:     subq    $66144, %rsp            # imm = 0x10260
+; CHECK-NEXT:     subq    $66112, %rsp            # imm = 0x10240
 ; CHECK-NEXT:     .cfi_offset %rbx, -24
 ; CHECK-NEXT:     movabsq $-868076584853899022, %rax # imm = 0xF3F3F8F201F2F8F2
 ; CHECK-NEXT:     movq    %rax, (%rsp)
diff --git a/llvm/test/CodeGen/X86/div-rem-pair-recomposition-signed.ll b/llvm/test/CodeGen/X86/div-rem-pair-recomposition-signed.ll
index 5cfdbb7af38ad3..080dc54677e6e1 100644
--- a/llvm/test/CodeGen/X86/div-rem-pair-recomposition-signed.ll
+++ b/llvm/test/CodeGen/X86/div-rem-pair-recomposition-signed.ll
@@ -151,7 +151,7 @@ define i128 @scalar_i128(i128 %x, i128 %y, ptr %divdst) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $176, %esp
+; X86-NEXT:    subl $160, %esp
 ; X86-NEXT:    movl 32(%ebp), %eax
 ; X86-NEXT:    movl 36(%ebp), %ecx
 ; X86-NEXT:    movl %ecx, %edx
diff --git a/llvm/test/CodeGen/X86/div-rem-pair-recomposition-unsigned.ll b/llvm/test/CodeGen/X86/div-rem-pair-recomposition-unsigned.ll
index 1006d0fcae9ed0..b418d952385974 100644
--- a/llvm/test/CodeGen/X86/div-rem-pair-recomposition-unsigned.ll
+++ b/llvm/test/CodeGen/X86/div-rem-pair-recomposition-unsigned.ll
@@ -151,7 +151,7 @@ define i128 @scalar_i128(i128 %x, i128 %y, ptr %divdst) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $160, %esp
+; X86-NEXT:    subl $144, %esp
 ; X86-NEXT:    movl 48(%ebp), %ebx
 ; X86-NEXT:    movl 40(%ebp), %ecx
 ; X86-NEXT:    movl 52(%ebp), %esi
diff --git a/llvm/test/CodeGen/X86/extractelement-index.ll b/llvm/test/CodeGen/X86/extractelement-index.ll
index 077351b9718d5f..bbafb0b8c73b8d 100644
--- a/llvm/test/CodeGen/X86/extractelement-index.ll
+++ b/llvm/test/CodeGen/X86/extractelement-index.ll
@@ -454,7 +454,7 @@ define i8 @extractelement_v32i8_var(<32 x i8> %a, i256 %i) nounwind {
 ; AVX-NEXT:    pushq %rbp
 ; AVX-NEXT:    movq %rsp, %rbp
 ; AVX-NEXT:    andq $-32, %rsp
-; AVX-NEXT:    subq $64, %rsp
+; AVX-NEXT:    subq $32, %rsp
 ; AVX-NEXT:    andl $31, %edi
 ; AVX-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX-NEXT:    movzbl (%rsp,%rdi), %eax
@@ -498,7 +498,7 @@ define i16 @extractelement_v16i16_var(<16 x i16> %a, i256 %i) nounwind {
 ; AVX-NEXT:    pushq %rbp
 ; AVX-NEXT:    movq %rsp, %rbp
 ; AVX-NEXT:    andq $-32, %rsp
-; AVX-NEXT:    subq $64, %rsp
+; AVX-NEXT:    subq $32, %rsp
 ; AVX-NEXT:    andl $15, %edi
 ; AVX-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX-NEXT:    movzwl (%rsp,%rdi,2), %eax
@@ -542,7 +542,7 @@ define i32 @extractelement_v8i32_var(<8 x i32> %a, i256 %i) nounwind {
 ; AVX-NEXT:    pushq %rbp
 ; AVX-NEXT:    movq %rsp, %rbp
 ; AVX-NEXT:    andq $-32, %rsp
-; AVX-NEXT:    subq $64, %rsp
+; AVX-NEXT:    subq $32, %rsp
 ; AVX-NEXT:    andl $7, %edi
 ; AVX-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX-NEXT:    movl (%rsp,%rdi,4), %eax
@@ -586,7 +586,7 @@ define i64 @extractelement_v4i64_var(<4 x i64> %a, i256 %i) nounwind {
 ; AVX-NEXT:    pushq %rbp
 ; AVX-NEXT:    movq %rsp, %rbp
 ; AVX-NEXT:    andq $-32, %rsp
-; AVX-NEXT:    subq $64, %rsp
+; AVX-NEXT:    subq $32, %rsp
 ; AVX-NEXT:    andl $3, %edi
 ; AVX-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX-NEXT:    movq (%rsp,%rdi,8), %rax
diff --git a/llvm/test/CodeGen/X86/extractelement-load.ll b/llvm/test/CodeGen/X86/extractelement-load.ll
index ce68eebd5b752b..90d478603ab99d 100644
--- a/llvm/test/CodeGen/X86/extractelement-load.ll
+++ b/llvm/test/CodeGen/X86/extractelement-load.ll
@@ -427,7 +427,7 @@ define i32 @main() nounwind {
 ; X86-SSE2-NEXT:    pushl %edi
 ; X86-SSE2-NEXT:    pushl %esi
 ; X86-SSE2-NEXT:    andl $-32, %esp
-; X86-SSE2-NEXT:    subl $64, %esp
+; X86-SSE2-NEXT:    subl $32, %esp
 ; X86-SSE2-NEXT:    movaps n1+16, %xmm0
 ; X86-SSE2-NEXT:    movaps n1, %xmm1
 ; X86-SSE2-NEXT:    movl zero+4, %ecx
@@ -461,7 +461,7 @@ define i32 @main() nounwind {
 ; X64-SSSE3-NEXT:    pushq %rbp
 ; X64-SSSE3-NEXT:    movq %rsp, %rbp
 ; X64-SSSE3-NEXT:    andq $-32, %rsp
-; X64-SSSE3-NEXT:    subq $64, %rsp
+; X64-SSSE3-NEXT:    subq $32, %rsp
 ; X64-SSSE3-NEXT:    movq n1 at GOTPCREL(%rip), %rax
 ; X64-SSSE3-NEXT:    movaps (%rax), %xmm0
 ; X64-SSSE3-NEXT:    movaps 16(%rax), %xmm1
@@ -494,7 +494,7 @@ define i32 @main() nounwind {
 ; X64-AVX-NEXT:    pushq %rbp
 ; X64-AVX-NEXT:    movq %rsp, %rbp
 ; X64-AVX-NEXT:    andq $-32, %rsp
-; X64-AVX-NEXT:    subq $64, %rsp
+; X64-AVX-NEXT:    subq $32, %rsp
 ; X64-AVX-NEXT:    movq n1 at GOTPCREL(%rip), %rax
 ; X64-AVX-NEXT:    vmovaps (%rax), %ymm0
 ; X64-AVX-NEXT:    movl zero+4(%rip), %ecx
diff --git a/llvm/test/CodeGen/X86/fma.ll b/llvm/test/CodeGen/X86/fma.ll
index 6d865628ba7157..a4cb79d53acb75 100644
--- a/llvm/test/CodeGen/X86/fma.ll
+++ b/llvm/test/CodeGen/X86/fma.ll
@@ -1122,8 +1122,8 @@ define <16 x float> @test_v16f32(<16 x float> %a, <16 x float> %b, <16 x float>
 ; FMACALL32_BDVER2-NEXT:    pushl %ebp ## encoding: [0x55]
 ; FMACALL32_BDVER2-NEXT:    movl %esp, %ebp ## encoding: [0x89,0xe5]
 ; FMACALL32_BDVER2-NEXT:    andl $-32, %esp ## encoding: [0x83,0xe4,0xe0]
-; FMACALL32_BDVER2-NEXT:    subl $448, %esp ## encoding: [0x81,0xec,0xc0,0x01,0x00,0x00]
-; FMACALL32_BDVER2-NEXT:    ## imm = 0x1C0
+; FMACALL32_BDVER2-NEXT:    subl $416, %esp ## encoding: [0x81,0xec,0xa0,0x01,0x00,0x00]
+; FMACALL32_BDVER2-NEXT:    ## imm = 0x1A0
 ; FMACALL32_BDVER2-NEXT:    vmovaps 56(%ebp), %xmm4 ## encoding: [0xc5,0xf8,0x28,0x65,0x38]
 ; FMACALL32_BDVER2-NEXT:    vmovaps %ymm2, {{[-0-9]+}}(%e{{[sb]}}p) ## 32-byte Spill
 ; FMACALL32_BDVER2-NEXT:    ## encoding: [0xc5,0xfc,0x29,0x94,0x24,0x60,0x01,0x00,0x00]
diff --git a/llvm/test/CodeGen/X86/fp128-libcalls-strict.ll b/llvm/test/CodeGen/X86/fp128-libcalls-strict.ll
index 31af5bc924d61b..150073244767a6 100644
--- a/llvm/test/CodeGen/X86/fp128-libcalls-strict.ll
+++ b/llvm/test/CodeGen/X86/fp128-libcalls-strict.ll
@@ -111,7 +111,7 @@ define fp128 @add(fp128 %x, fp128 %y) nounwind strictfp {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 8(%ebp), %esi
 ; WIN-X86-NEXT:    movl 36(%ebp), %edi
 ; WIN-X86-NEXT:    movl 40(%ebp), %ebx
@@ -240,7 +240,7 @@ define fp128 @sub(fp128 %x, fp128 %y) nounwind strictfp {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 8(%ebp), %esi
 ; WIN-X86-NEXT:    movl 36(%ebp), %edi
 ; WIN-X86-NEXT:    movl 40(%ebp), %ebx
@@ -369,7 +369,7 @@ define fp128 @mul(fp128 %x, fp128 %y) nounwind strictfp {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 8(%ebp), %esi
 ; WIN-X86-NEXT:    movl 36(%ebp), %edi
 ; WIN-X86-NEXT:    movl 40(%ebp), %ebx
@@ -498,7 +498,7 @@ define fp128 @div(fp128 %x, fp128 %y) nounwind strictfp {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 8(%ebp), %esi
 ; WIN-X86-NEXT:    movl 36(%ebp), %edi
 ; WIN-X86-NEXT:    movl 40(%ebp), %ebx
diff --git a/llvm/test/CodeGen/X86/fp128-libcalls.ll b/llvm/test/CodeGen/X86/fp128-libcalls.ll
index 3cb21b5fd59c61..e3ba851952081b 100644
--- a/llvm/test/CodeGen/X86/fp128-libcalls.ll
+++ b/llvm/test/CodeGen/X86/fp128-libcalls.ll
@@ -105,7 +105,7 @@ define dso_local void @Test128Add(fp128 %d1, fp128 %d2) nounwind {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 16(%ebp), %edx
 ; WIN-X86-NEXT:    movl 20(%ebp), %esi
 ; WIN-X86-NEXT:    movl 24(%ebp), %edi
@@ -236,7 +236,7 @@ define dso_local void @Test128_1Add(fp128 %d1) nounwind {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 16(%ebp), %esi
 ; WIN-X86-NEXT:    movl 20(%ebp), %edi
 ; WIN-X86-NEXT:    movl _vf128, %edx
@@ -362,7 +362,7 @@ define dso_local void @Test128Sub(fp128 %d1, fp128 %d2) nounwind {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 16(%ebp), %edx
 ; WIN-X86-NEXT:    movl 20(%ebp), %esi
 ; WIN-X86-NEXT:    movl 24(%ebp), %edi
@@ -493,7 +493,7 @@ define dso_local void @Test128_1Sub(fp128 %d1) nounwind {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 16(%ebp), %esi
 ; WIN-X86-NEXT:    movl 20(%ebp), %edi
 ; WIN-X86-NEXT:    movl _vf128, %edx
@@ -619,7 +619,7 @@ define dso_local void @Test128Mul(fp128 %d1, fp128 %d2) nounwind {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 16(%ebp), %edx
 ; WIN-X86-NEXT:    movl 20(%ebp), %esi
 ; WIN-X86-NEXT:    movl 24(%ebp), %edi
@@ -750,7 +750,7 @@ define dso_local void @Test128_1Mul(fp128 %d1) nounwind {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 16(%ebp), %esi
 ; WIN-X86-NEXT:    movl 20(%ebp), %edi
 ; WIN-X86-NEXT:    movl _vf128, %edx
@@ -876,7 +876,7 @@ define dso_local void @Test128Div(fp128 %d1, fp128 %d2) nounwind {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 16(%ebp), %edx
 ; WIN-X86-NEXT:    movl 20(%ebp), %esi
 ; WIN-X86-NEXT:    movl 24(%ebp), %edi
@@ -1007,7 +1007,7 @@ define dso_local void @Test128_1Div(fp128 %d1) nounwind {
 ; WIN-X86-NEXT:    pushl %edi
 ; WIN-X86-NEXT:    pushl %esi
 ; WIN-X86-NEXT:    andl $-16, %esp
-; WIN-X86-NEXT:    subl $80, %esp
+; WIN-X86-NEXT:    subl $64, %esp
 ; WIN-X86-NEXT:    movl 16(%ebp), %esi
 ; WIN-X86-NEXT:    movl 20(%ebp), %edi
 ; WIN-X86-NEXT:    movl _vf128, %edx
diff --git a/llvm/test/CodeGen/X86/gep-expanded-vector.ll b/llvm/test/CodeGen/X86/gep-expanded-vector.ll
index 943cd3610c9d32..f0f722ca5d7fed 100644
--- a/llvm/test/CodeGen/X86/gep-expanded-vector.ll
+++ b/llvm/test/CodeGen/X86/gep-expanded-vector.ll
@@ -9,7 +9,7 @@ define ptr @malloc_init_state(<64 x ptr> %tmp, i32 %ind) nounwind {
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $576, %rsp # imm = 0x240
+; CHECK-NEXT:    subq $512, %rsp # imm = 0x200
 ; CHECK-NEXT:    # kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vpbroadcastq {{.*#+}} zmm8 = [16,16,16,16,16,16,16,16]
 ; CHECK-NEXT:    vpaddq %zmm8, %zmm0, %zmm0
diff --git a/llvm/test/CodeGen/X86/i128-fp128-abi.ll b/llvm/test/CodeGen/X86/i128-fp128-abi.ll
index 9f385ee2faf4e3..77881225408dd1 100644
--- a/llvm/test/CodeGen/X86/i128-fp128-abi.ll
+++ b/llvm/test/CodeGen/X86/i128-fp128-abi.ll
@@ -657,7 +657,7 @@ define void @call_first_arg(PrimTy %x) nounwind {
 ; CHECK-MSVC32-NEXT:    movl %esp, %ebp
 ; CHECK-MSVC32-NEXT:    pushl %esi
 ; CHECK-MSVC32-NEXT:    andl $-16, %esp
-; CHECK-MSVC32-NEXT:    subl $64, %esp
+; CHECK-MSVC32-NEXT:    subl $48, %esp
 ; CHECK-MSVC32-NEXT:    movl 8(%ebp), %eax
 ; CHECK-MSVC32-NEXT:    movl 12(%ebp), %ecx
 ; CHECK-MSVC32-NEXT:    movl 16(%ebp), %edx
@@ -793,7 +793,7 @@ define void @call_leading_args(PrimTy %x) nounwind {
 ; CHECK-MSVC32-NEXT:    movl %esp, %ebp
 ; CHECK-MSVC32-NEXT:    pushl %esi
 ; CHECK-MSVC32-NEXT:    andl $-16, %esp
-; CHECK-MSVC32-NEXT:    subl $96, %esp
+; CHECK-MSVC32-NEXT:    subl $80, %esp
 ; CHECK-MSVC32-NEXT:    movl 8(%ebp), %eax
 ; CHECK-MSVC32-NEXT:    movl 12(%ebp), %ecx
 ; CHECK-MSVC32-NEXT:    movl 16(%ebp), %edx
@@ -960,7 +960,7 @@ define void @call_many_leading_args(PrimTy %x) nounwind {
 ; CHECK-MSVC32-NEXT:    movl %esp, %ebp
 ; CHECK-MSVC32-NEXT:    pushl %esi
 ; CHECK-MSVC32-NEXT:    andl $-16, %esp
-; CHECK-MSVC32-NEXT:    subl $112, %esp
+; CHECK-MSVC32-NEXT:    subl $96, %esp
 ; CHECK-MSVC32-NEXT:    movl 8(%ebp), %eax
 ; CHECK-MSVC32-NEXT:    movl 12(%ebp), %ecx
 ; CHECK-MSVC32-NEXT:    movl 16(%ebp), %edx
@@ -1108,7 +1108,7 @@ define void @call_trailing_arg(PrimTy %x) nounwind {
 ; CHECK-MSVC32-NEXT:    movl %esp, %ebp
 ; CHECK-MSVC32-NEXT:    pushl %esi
 ; CHECK-MSVC32-NEXT:    andl $-16, %esp
-; CHECK-MSVC32-NEXT:    subl $96, %esp
+; CHECK-MSVC32-NEXT:    subl $80, %esp
 ; CHECK-MSVC32-NEXT:    movl 8(%ebp), %eax
 ; CHECK-MSVC32-NEXT:    movl 12(%ebp), %ecx
 ; CHECK-MSVC32-NEXT:    movl 16(%ebp), %edx
diff --git a/llvm/test/CodeGen/X86/i128-udiv.ll b/llvm/test/CodeGen/X86/i128-udiv.ll
index b5355c460c999d..543e9d13c50bff 100644
--- a/llvm/test/CodeGen/X86/i128-udiv.ll
+++ b/llvm/test/CodeGen/X86/i128-udiv.ll
@@ -43,7 +43,7 @@ define i128 @test2(i128 %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $144, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movl 32(%ebp), %esi
 ; X86-NEXT:    movl 36(%ebp), %edi
 ; X86-NEXT:    movl 28(%ebp), %ecx
@@ -358,7 +358,7 @@ define i128 @test3(i128 %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $160, %esp
+; X86-NEXT:    subl $144, %esp
 ; X86-NEXT:    movl 32(%ebp), %edi
 ; X86-NEXT:    movl 36(%ebp), %edx
 ; X86-NEXT:    movl 28(%ebp), %esi
@@ -688,7 +688,7 @@ define i128 @div_by_7(i128 %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $160, %esp
+; X86-NEXT:    subl $144, %esp
 ; X86-NEXT:    movl 32(%ebp), %edi
 ; X86-NEXT:    movl 36(%ebp), %ebx
 ; X86-NEXT:    movl 28(%ebp), %edx
@@ -1014,7 +1014,7 @@ define i128 @div_by_11(i128 %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $160, %esp
+; X86-NEXT:    subl $144, %esp
 ; X86-NEXT:    movl 32(%ebp), %edi
 ; X86-NEXT:    movl 36(%ebp), %ebx
 ; X86-NEXT:    movl 28(%ebp), %edx
@@ -1338,7 +1338,7 @@ define i128 @div_by_22(i128 %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $160, %esp
+; X86-NEXT:    subl $144, %esp
 ; X86-NEXT:    movl 32(%ebp), %edi
 ; X86-NEXT:    movl 36(%ebp), %ebx
 ; X86-NEXT:    movl 28(%ebp), %edx
@@ -1664,7 +1664,7 @@ define i128 @div_by_56(i128 %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $160, %esp
+; X86-NEXT:    subl $144, %esp
 ; X86-NEXT:    movl 32(%ebp), %edi
 ; X86-NEXT:    movl 36(%ebp), %ebx
 ; X86-NEXT:    movl 28(%ebp), %edx
@@ -1990,7 +1990,7 @@ define i128 @rem_by_7(i128 %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $144, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movl 36(%ebp), %ebx
 ; X86-NEXT:    movl 28(%ebp), %eax
 ; X86-NEXT:    testl %ebx, %ebx
@@ -2320,7 +2320,7 @@ define i128 @rem_by_14(i128 %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $144, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movl 36(%ebp), %ebx
 ; X86-NEXT:    movl 28(%ebp), %eax
 ; X86-NEXT:    testl %ebx, %ebx
@@ -2653,7 +2653,7 @@ define i128 @div_by_67(i128 %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $160, %esp
+; X86-NEXT:    subl $144, %esp
 ; X86-NEXT:    movl 32(%ebp), %edi
 ; X86-NEXT:    movl 36(%ebp), %ebx
 ; X86-NEXT:    movl 28(%ebp), %edx
diff --git a/llvm/test/CodeGen/X86/i64-mem-copy.ll b/llvm/test/CodeGen/X86/i64-mem-copy.ll
index 9b866183d88ed0..8a89ab3714a6ac 100644
--- a/llvm/test/CodeGen/X86/i64-mem-copy.ll
+++ b/llvm/test/CodeGen/X86/i64-mem-copy.ll
@@ -126,7 +126,7 @@ define void @PR23476(<5 x i64> %in, ptr %out, i32 %index) nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $80, %esp
+; X86-NEXT:    subl $64, %esp
 ; X86-NEXT:    movl 52(%ebp), %eax
 ; X86-NEXT:    andl $7, %eax
 ; X86-NEXT:    movl 48(%ebp), %ecx
@@ -147,7 +147,7 @@ define void @PR23476(<5 x i64> %in, ptr %out, i32 %index) nounwind {
 ; X86AVX-NEXT:    pushl %ebp
 ; X86AVX-NEXT:    movl %esp, %ebp
 ; X86AVX-NEXT:    andl $-32, %esp
-; X86AVX-NEXT:    subl $96, %esp
+; X86AVX-NEXT:    subl $64, %esp
 ; X86AVX-NEXT:    movl 52(%ebp), %eax
 ; X86AVX-NEXT:    andl $7, %eax
 ; X86AVX-NEXT:    movl 48(%ebp), %ecx
diff --git a/llvm/test/CodeGen/X86/inline-sse.ll b/llvm/test/CodeGen/X86/inline-sse.ll
index 87aa882a1f498c..3beba4b407abe2 100644
--- a/llvm/test/CodeGen/X86/inline-sse.ll
+++ b/llvm/test/CodeGen/X86/inline-sse.ll
@@ -11,7 +11,7 @@ define void @nop() nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $32, %esp
+; X86-NEXT:    subl $16, %esp
 ; X86-NEXT:    #APP
 ; X86-NEXT:    #NO_APP
 ; X86-NEXT:    movaps %xmm0, (%esp)
diff --git a/llvm/test/CodeGen/X86/insertelement-var-index.ll b/llvm/test/CodeGen/X86/insertelement-var-index.ll
index a4af339b5ebb23..2bc67ce8c2bb93 100644
--- a/llvm/test/CodeGen/X86/insertelement-var-index.ll
+++ b/llvm/test/CodeGen/X86/insertelement-var-index.ll
@@ -862,7 +862,7 @@ define <16 x i8> @arg_i8_v16i8(<16 x i8> %v, i8 %x, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-16, %esp
-; X86AVX2-NEXT:    subl $32, %esp
+; X86AVX2-NEXT:    subl $16, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $15, %eax
 ; X86AVX2-NEXT:    movzbl 8(%ebp), %ecx
@@ -916,7 +916,7 @@ define <8 x i16> @arg_i16_v8i16(<8 x i16> %v, i16 %x, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-16, %esp
-; X86AVX2-NEXT:    subl $32, %esp
+; X86AVX2-NEXT:    subl $16, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $7, %eax
 ; X86AVX2-NEXT:    movzwl 8(%ebp), %ecx
@@ -961,7 +961,7 @@ define <4 x i32> @arg_i32_v4i32(<4 x i32> %v, i32 %x, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-16, %esp
-; X86AVX2-NEXT:    subl $32, %esp
+; X86AVX2-NEXT:    subl $16, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $3, %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
@@ -1008,7 +1008,7 @@ define <2 x i64> @arg_i64_v2i64(<2 x i64> %v, i64 %x, i32 %y) nounwind {
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    pushl %esi
 ; X86AVX2-NEXT:    andl $-16, %esp
-; X86AVX2-NEXT:    subl $48, %esp
+; X86AVX2-NEXT:    subl $32, %esp
 ; X86AVX2-NEXT:    movl 16(%ebp), %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
 ; X86AVX2-NEXT:    movl 12(%ebp), %edx
@@ -1139,7 +1139,7 @@ define <2 x double> @arg_f64_v2f64(<2 x double> %v, double %x, i32 %y) nounwind
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-16, %esp
-; X86AVX2-NEXT:    subl $32, %esp
+; X86AVX2-NEXT:    subl $16, %esp
 ; X86AVX2-NEXT:    movl 16(%ebp), %eax
 ; X86AVX2-NEXT:    andl $1, %eax
 ; X86AVX2-NEXT:    vmovsd {{.*#+}} xmm1 = mem[0],zero
@@ -1196,7 +1196,7 @@ define <16 x i8> @load_i8_v16i8(<16 x i8> %v, ptr %p, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-16, %esp
-; X86AVX2-NEXT:    subl $32, %esp
+; X86AVX2-NEXT:    subl $16, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $15, %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
@@ -1255,7 +1255,7 @@ define <8 x i16> @load_i16_v8i16(<8 x i16> %v, ptr %p, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-16, %esp
-; X86AVX2-NEXT:    subl $32, %esp
+; X86AVX2-NEXT:    subl $16, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $7, %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
@@ -1304,7 +1304,7 @@ define <4 x i32> @load_i32_v4i32(<4 x i32> %v, ptr %p, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-16, %esp
-; X86AVX2-NEXT:    subl $32, %esp
+; X86AVX2-NEXT:    subl $16, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $3, %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
@@ -1355,7 +1355,7 @@ define <2 x i64> @load_i64_v2i64(<2 x i64> %v, ptr %p, i32 %y) nounwind {
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    pushl %esi
 ; X86AVX2-NEXT:    andl $-16, %esp
-; X86AVX2-NEXT:    subl $48, %esp
+; X86AVX2-NEXT:    subl $32, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
 ; X86AVX2-NEXT:    movl (%ecx), %edx
@@ -1493,7 +1493,7 @@ define <2 x double> @load_f64_v2f64(<2 x double> %v, ptr %p, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-16, %esp
-; X86AVX2-NEXT:    subl $32, %esp
+; X86AVX2-NEXT:    subl $16, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $1, %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
@@ -1526,7 +1526,7 @@ define <32 x i8> @arg_i8_v32i8(<32 x i8> %v, i8 %x, i32 %y) nounwind {
 ; AVX1OR2-NEXT:    pushq %rbp
 ; AVX1OR2-NEXT:    movq %rsp, %rbp
 ; AVX1OR2-NEXT:    andq $-32, %rsp
-; AVX1OR2-NEXT:    subq $64, %rsp
+; AVX1OR2-NEXT:    subq $32, %rsp
 ; AVX1OR2-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX1OR2-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX1OR2-NEXT:    andl $31, %esi
@@ -1541,7 +1541,7 @@ define <32 x i8> @arg_i8_v32i8(<32 x i8> %v, i8 %x, i32 %y) nounwind {
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-32, %rsp
-; AVX512F-NEXT:    subq $64, %rsp
+; AVX512F-NEXT:    subq $32, %rsp
 ; AVX512F-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX512F-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX512F-NEXT:    andl $31, %esi
@@ -1563,7 +1563,7 @@ define <32 x i8> @arg_i8_v32i8(<32 x i8> %v, i8 %x, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-32, %esp
-; X86AVX2-NEXT:    subl $64, %esp
+; X86AVX2-NEXT:    subl $32, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $31, %eax
 ; X86AVX2-NEXT:    movzbl 8(%ebp), %ecx
@@ -1594,7 +1594,7 @@ define <16 x i16> @arg_i16_v16i16(<16 x i16> %v, i16 %x, i32 %y) nounwind {
 ; AVX1OR2-NEXT:    pushq %rbp
 ; AVX1OR2-NEXT:    movq %rsp, %rbp
 ; AVX1OR2-NEXT:    andq $-32, %rsp
-; AVX1OR2-NEXT:    subq $64, %rsp
+; AVX1OR2-NEXT:    subq $32, %rsp
 ; AVX1OR2-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX1OR2-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX1OR2-NEXT:    andl $15, %esi
@@ -1609,7 +1609,7 @@ define <16 x i16> @arg_i16_v16i16(<16 x i16> %v, i16 %x, i32 %y) nounwind {
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-32, %rsp
-; AVX512F-NEXT:    subq $64, %rsp
+; AVX512F-NEXT:    subq $32, %rsp
 ; AVX512F-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX512F-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX512F-NEXT:    andl $15, %esi
@@ -1631,7 +1631,7 @@ define <16 x i16> @arg_i16_v16i16(<16 x i16> %v, i16 %x, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-32, %esp
-; X86AVX2-NEXT:    subl $64, %esp
+; X86AVX2-NEXT:    subl $32, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $15, %eax
 ; X86AVX2-NEXT:    movzwl 8(%ebp), %ecx
@@ -1662,7 +1662,7 @@ define <8 x i32> @arg_i32_v8i32(<8 x i32> %v, i32 %x, i32 %y) nounwind {
 ; AVX1OR2-NEXT:    pushq %rbp
 ; AVX1OR2-NEXT:    movq %rsp, %rbp
 ; AVX1OR2-NEXT:    andq $-32, %rsp
-; AVX1OR2-NEXT:    subq $64, %rsp
+; AVX1OR2-NEXT:    subq $32, %rsp
 ; AVX1OR2-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX1OR2-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX1OR2-NEXT:    andl $7, %esi
@@ -1684,7 +1684,7 @@ define <8 x i32> @arg_i32_v8i32(<8 x i32> %v, i32 %x, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-32, %esp
-; X86AVX2-NEXT:    subl $64, %esp
+; X86AVX2-NEXT:    subl $32, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $7, %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
@@ -1715,7 +1715,7 @@ define <4 x i64> @arg_i64_v4i64(<4 x i64> %v, i64 %x, i32 %y) nounwind {
 ; AVX1OR2-NEXT:    pushq %rbp
 ; AVX1OR2-NEXT:    movq %rsp, %rbp
 ; AVX1OR2-NEXT:    andq $-32, %rsp
-; AVX1OR2-NEXT:    subq $64, %rsp
+; AVX1OR2-NEXT:    subq $32, %rsp
 ; AVX1OR2-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX1OR2-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX1OR2-NEXT:    andl $3, %esi
@@ -1739,7 +1739,7 @@ define <4 x i64> @arg_i64_v4i64(<4 x i64> %v, i64 %x, i32 %y) nounwind {
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    pushl %esi
 ; X86AVX2-NEXT:    andl $-32, %esp
-; X86AVX2-NEXT:    subl $96, %esp
+; X86AVX2-NEXT:    subl $64, %esp
 ; X86AVX2-NEXT:    movl 16(%ebp), %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
 ; X86AVX2-NEXT:    movl 12(%ebp), %edx
@@ -1859,7 +1859,7 @@ define <4 x double> @arg_f64_v4f64(<4 x double> %v, double %x, i32 %y) nounwind
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-32, %esp
-; X86AVX2-NEXT:    subl $64, %esp
+; X86AVX2-NEXT:    subl $32, %esp
 ; X86AVX2-NEXT:    movl 16(%ebp), %eax
 ; X86AVX2-NEXT:    andl $3, %eax
 ; X86AVX2-NEXT:    vmovsd {{.*#+}} xmm1 = mem[0],zero
@@ -1891,7 +1891,7 @@ define <32 x i8> @load_i8_v32i8(<32 x i8> %v, ptr %p, i32 %y) nounwind {
 ; AVX1OR2-NEXT:    pushq %rbp
 ; AVX1OR2-NEXT:    movq %rsp, %rbp
 ; AVX1OR2-NEXT:    andq $-32, %rsp
-; AVX1OR2-NEXT:    subq $64, %rsp
+; AVX1OR2-NEXT:    subq $32, %rsp
 ; AVX1OR2-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX1OR2-NEXT:    movzbl (%rdi), %eax
 ; AVX1OR2-NEXT:    vmovaps %ymm0, (%rsp)
@@ -1907,7 +1907,7 @@ define <32 x i8> @load_i8_v32i8(<32 x i8> %v, ptr %p, i32 %y) nounwind {
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-32, %rsp
-; AVX512F-NEXT:    subq $64, %rsp
+; AVX512F-NEXT:    subq $32, %rsp
 ; AVX512F-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX512F-NEXT:    movzbl (%rdi), %eax
 ; AVX512F-NEXT:    vmovaps %ymm0, (%rsp)
@@ -1930,7 +1930,7 @@ define <32 x i8> @load_i8_v32i8(<32 x i8> %v, ptr %p, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-32, %esp
-; X86AVX2-NEXT:    subl $64, %esp
+; X86AVX2-NEXT:    subl $32, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $31, %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
@@ -1964,7 +1964,7 @@ define <16 x i16> @load_i16_v16i16(<16 x i16> %v, ptr %p, i32 %y) nounwind {
 ; AVX1OR2-NEXT:    pushq %rbp
 ; AVX1OR2-NEXT:    movq %rsp, %rbp
 ; AVX1OR2-NEXT:    andq $-32, %rsp
-; AVX1OR2-NEXT:    subq $64, %rsp
+; AVX1OR2-NEXT:    subq $32, %rsp
 ; AVX1OR2-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX1OR2-NEXT:    movzwl (%rdi), %eax
 ; AVX1OR2-NEXT:    vmovaps %ymm0, (%rsp)
@@ -1980,7 +1980,7 @@ define <16 x i16> @load_i16_v16i16(<16 x i16> %v, ptr %p, i32 %y) nounwind {
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-32, %rsp
-; AVX512F-NEXT:    subq $64, %rsp
+; AVX512F-NEXT:    subq $32, %rsp
 ; AVX512F-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX512F-NEXT:    movzwl (%rdi), %eax
 ; AVX512F-NEXT:    vmovaps %ymm0, (%rsp)
@@ -2003,7 +2003,7 @@ define <16 x i16> @load_i16_v16i16(<16 x i16> %v, ptr %p, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-32, %esp
-; X86AVX2-NEXT:    subl $64, %esp
+; X86AVX2-NEXT:    subl $32, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $15, %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
@@ -2037,7 +2037,7 @@ define <8 x i32> @load_i32_v8i32(<8 x i32> %v, ptr %p, i32 %y) nounwind {
 ; AVX1OR2-NEXT:    pushq %rbp
 ; AVX1OR2-NEXT:    movq %rsp, %rbp
 ; AVX1OR2-NEXT:    andq $-32, %rsp
-; AVX1OR2-NEXT:    subq $64, %rsp
+; AVX1OR2-NEXT:    subq $32, %rsp
 ; AVX1OR2-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX1OR2-NEXT:    movl (%rdi), %eax
 ; AVX1OR2-NEXT:    vmovaps %ymm0, (%rsp)
@@ -2060,7 +2060,7 @@ define <8 x i32> @load_i32_v8i32(<8 x i32> %v, ptr %p, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-32, %esp
-; X86AVX2-NEXT:    subl $64, %esp
+; X86AVX2-NEXT:    subl $32, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $7, %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
@@ -2094,7 +2094,7 @@ define <4 x i64> @load_i64_v4i64(<4 x i64> %v, ptr %p, i32 %y) nounwind {
 ; AVX1OR2-NEXT:    pushq %rbp
 ; AVX1OR2-NEXT:    movq %rsp, %rbp
 ; AVX1OR2-NEXT:    andq $-32, %rsp
-; AVX1OR2-NEXT:    subq $64, %rsp
+; AVX1OR2-NEXT:    subq $32, %rsp
 ; AVX1OR2-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX1OR2-NEXT:    movq (%rdi), %rax
 ; AVX1OR2-NEXT:    vmovaps %ymm0, (%rsp)
@@ -2119,7 +2119,7 @@ define <4 x i64> @load_i64_v4i64(<4 x i64> %v, ptr %p, i32 %y) nounwind {
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    pushl %esi
 ; X86AVX2-NEXT:    andl $-32, %esp
-; X86AVX2-NEXT:    subl $96, %esp
+; X86AVX2-NEXT:    subl $64, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
 ; X86AVX2-NEXT:    movl (%ecx), %edx
@@ -2243,7 +2243,7 @@ define <4 x double> @load_f64_v4f64(<4 x double> %v, ptr %p, i32 %y) nounwind {
 ; X86AVX2-NEXT:    pushl %ebp
 ; X86AVX2-NEXT:    movl %esp, %ebp
 ; X86AVX2-NEXT:    andl $-32, %esp
-; X86AVX2-NEXT:    subl $64, %esp
+; X86AVX2-NEXT:    subl $32, %esp
 ; X86AVX2-NEXT:    movl 12(%ebp), %eax
 ; X86AVX2-NEXT:    andl $3, %eax
 ; X86AVX2-NEXT:    movl 8(%ebp), %ecx
diff --git a/llvm/test/CodeGen/X86/isel-x87.ll b/llvm/test/CodeGen/X86/isel-x87.ll
index 492faaa19cd660..fdd5e69db91486 100644
--- a/llvm/test/CodeGen/X86/isel-x87.ll
+++ b/llvm/test/CodeGen/X86/isel-x87.ll
@@ -12,7 +12,7 @@ define x86_fp80 @f0(x86_fp80 noundef %a) nounwind {
 ; GISEL_X86-NEXT:    pushl %ebp
 ; GISEL_X86-NEXT:    movl %esp, %ebp
 ; GISEL_X86-NEXT:    andl $-16, %esp
-; GISEL_X86-NEXT:    subl $48, %esp
+; GISEL_X86-NEXT:    subl $32, %esp
 ; GISEL_X86-NEXT:    fldt 8(%ebp)
 ; GISEL_X86-NEXT:    fldt {{\.?LCPI[0-9]+_[0-9]+}}
 ; GISEL_X86-NEXT:    fxch %st(1)
@@ -30,7 +30,7 @@ define x86_fp80 @f0(x86_fp80 noundef %a) nounwind {
 ; SDAG_X86-NEXT:    pushl %ebp
 ; SDAG_X86-NEXT:    movl %esp, %ebp
 ; SDAG_X86-NEXT:    andl $-16, %esp
-; SDAG_X86-NEXT:    subl $48, %esp
+; SDAG_X86-NEXT:    subl $32, %esp
 ; SDAG_X86-NEXT:    fldt 8(%ebp)
 ; SDAG_X86-NEXT:    fld %st(0)
 ; SDAG_X86-NEXT:    fstpt {{[0-9]+}}(%esp)
diff --git a/llvm/test/CodeGen/X86/long-double-abi-align.ll b/llvm/test/CodeGen/X86/long-double-abi-align.ll
index 02d68ada9a8d44..3dd92b60c7c8a0 100644
--- a/llvm/test/CodeGen/X86/long-double-abi-align.ll
+++ b/llvm/test/CodeGen/X86/long-double-abi-align.ll
@@ -10,7 +10,7 @@ define void @foo(i32 %0, x86_fp80 %1, i32 %2) nounwind {
 ; MSVC-NEXT:    pushl %ebp
 ; MSVC-NEXT:    movl %esp, %ebp
 ; MSVC-NEXT:    andl $-16, %esp
-; MSVC-NEXT:    subl $32, %esp
+; MSVC-NEXT:    subl $16, %esp
 ; MSVC-NEXT:    fldt 24(%ebp)
 ; MSVC-NEXT:    fstpt (%esp)
 ; MSVC-NEXT:    leal 8(%ebp), %eax
@@ -34,7 +34,7 @@ define void @foo(i32 %0, x86_fp80 %1, i32 %2) nounwind {
 ; MINGW-NEXT:    pushl %ebp
 ; MINGW-NEXT:    movl %esp, %ebp
 ; MINGW-NEXT:    andl $-16, %esp
-; MINGW-NEXT:    subl $32, %esp
+; MINGW-NEXT:    subl $16, %esp
 ; MINGW-NEXT:    fldt 12(%ebp)
 ; MINGW-NEXT:    fstpt (%esp)
 ; MINGW-NEXT:    leal 8(%ebp), %eax
diff --git a/llvm/test/CodeGen/X86/matrix-multiply.ll b/llvm/test/CodeGen/X86/matrix-multiply.ll
index f38b769fe49872..7ed7d0b166e0cc 100644
--- a/llvm/test/CodeGen/X86/matrix-multiply.ll
+++ b/llvm/test/CodeGen/X86/matrix-multiply.ll
@@ -3415,7 +3415,7 @@ define <64 x double> @test_mul8x8_f64(<64 x double> %a0, <64 x double> %a1) noun
 ; AVX1OR2-NEXT:    pushq %rbp
 ; AVX1OR2-NEXT:    movq %rsp, %rbp
 ; AVX1OR2-NEXT:    andq $-32, %rsp
-; AVX1OR2-NEXT:    subq $448, %rsp # imm = 0x1C0
+; AVX1OR2-NEXT:    subq $416, %rsp # imm = 0x1A0
 ; AVX1OR2-NEXT:    vmovapd %ymm2, %ymm12
 ; AVX1OR2-NEXT:    vmovapd %ymm0, (%rsp) # 32-byte Spill
 ; AVX1OR2-NEXT:    movq %rdi, %rax
diff --git a/llvm/test/CodeGen/X86/memset-sse-stack-realignment.ll b/llvm/test/CodeGen/X86/memset-sse-stack-realignment.ll
index a5ecdab880a6a6..00dd58be4f3444 100644
--- a/llvm/test/CodeGen/X86/memset-sse-stack-realignment.ll
+++ b/llvm/test/CodeGen/X86/memset-sse-stack-realignment.ll
@@ -39,7 +39,7 @@ define void @test1(i32 %t) nounwind {
 ; SSE-NEXT:    movl %esp, %ebp
 ; SSE-NEXT:    pushl %esi
 ; SSE-NEXT:    andl $-16, %esp
-; SSE-NEXT:    subl $48, %esp
+; SSE-NEXT:    subl $32, %esp
 ; SSE-NEXT:    movl %esp, %esi
 ; SSE-NEXT:    movl 8(%ebp), %eax
 ; SSE-NEXT:    xorps %xmm0, %xmm0
@@ -62,7 +62,7 @@ define void @test1(i32 %t) nounwind {
 ; AVX-NEXT:    movl %esp, %ebp
 ; AVX-NEXT:    pushl %esi
 ; AVX-NEXT:    andl $-32, %esp
-; AVX-NEXT:    subl $64, %esp
+; AVX-NEXT:    subl $32, %esp
 ; AVX-NEXT:    movl %esp, %esi
 ; AVX-NEXT:    movl 8(%ebp), %eax
 ; AVX-NEXT:    vxorps %xmm0, %xmm0, %xmm0
@@ -112,7 +112,7 @@ define void @test2(i32 %t) nounwind {
 ; SSE-NEXT:    movl %esp, %ebp
 ; SSE-NEXT:    pushl %esi
 ; SSE-NEXT:    andl $-16, %esp
-; SSE-NEXT:    subl $32, %esp
+; SSE-NEXT:    subl $16, %esp
 ; SSE-NEXT:    movl %esp, %esi
 ; SSE-NEXT:    movl 8(%ebp), %eax
 ; SSE-NEXT:    xorps %xmm0, %xmm0
@@ -134,7 +134,7 @@ define void @test2(i32 %t) nounwind {
 ; AVX-NEXT:    movl %esp, %ebp
 ; AVX-NEXT:    pushl %esi
 ; AVX-NEXT:    andl $-16, %esp
-; AVX-NEXT:    subl $32, %esp
+; AVX-NEXT:    subl $16, %esp
 ; AVX-NEXT:    movl %esp, %esi
 ; AVX-NEXT:    movl 8(%ebp), %eax
 ; AVX-NEXT:    vxorps %xmm0, %xmm0, %xmm0
diff --git a/llvm/test/CodeGen/X86/mingw-alloca.ll b/llvm/test/CodeGen/X86/mingw-alloca.ll
index fe60f103a9587a..493f8323e07d09 100644
--- a/llvm/test/CodeGen/X86/mingw-alloca.ll
+++ b/llvm/test/CodeGen/X86/mingw-alloca.ll
@@ -22,12 +22,12 @@ entry:
 ; COFF: andl $-16, %esp
 ; COFF: pushl %eax
 ; COFF: calll __alloca
-; COFF: movl	8012(%esp), %eax
+; COFF: movl	7996(%esp), %eax
 ; ELF: foo2:
 ; ELF: andl $-16, %esp
 ; ELF: pushl %eax
 ; ELF: calll _alloca
-; ELF: movl	8012(%esp), %eax
+; ELF: movl	7996(%esp), %eax
 	%A2 = alloca [2000 x i32], align 16		; <ptr> [#uses=1]
 	%A2.sub = getelementptr [2000 x i32], ptr %A2, i32 0, i32 0		; <ptr> [#uses=1]
 	call void @bar2( ptr %A2.sub, i32 %N )
diff --git a/llvm/test/CodeGen/X86/mmx-arith.ll b/llvm/test/CodeGen/X86/mmx-arith.ll
index 8f97d2652bc531..832e46b1367a6d 100644
--- a/llvm/test/CodeGen/X86/mmx-arith.ll
+++ b/llvm/test/CodeGen/X86/mmx-arith.ll
@@ -243,7 +243,7 @@ define void @test2(ptr %A, ptr %B) nounwind {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-8, %esp
-; X86-NEXT:    subl $16, %esp
+; X86-NEXT:    subl $8, %esp
 ; X86-NEXT:    movl 12(%ebp), %ecx
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    movq {{.*#+}} xmm0 = mem[0],zero
diff --git a/llvm/test/CodeGen/X86/mmx-intrinsics.ll b/llvm/test/CodeGen/X86/mmx-intrinsics.ll
index a7b6ed416622ef..c7e311dbbb3fa8 100644
--- a/llvm/test/CodeGen/X86/mmx-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/mmx-intrinsics.ll
@@ -2864,7 +2864,7 @@ define void @test23(<1 x i64> %d, <1 x i64> %n, ptr %p) nounwind optsize ssp {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    andl $-8, %esp
-; X86-NEXT:    subl $24, %esp
+; X86-NEXT:    subl $16, %esp
 ; X86-NEXT:    movl 16(%ebp), %eax
 ; X86-NEXT:    movl 20(%ebp), %ecx
 ; X86-NEXT:    movl %ecx, {{[0-9]+}}(%esp)
diff --git a/llvm/test/CodeGen/X86/musttail-varargs.ll b/llvm/test/CodeGen/X86/musttail-varargs.ll
index 65cd1edd92e318..4db78c6fed599f 100644
--- a/llvm/test/CodeGen/X86/musttail-varargs.ll
+++ b/llvm/test/CodeGen/X86/musttail-varargs.ll
@@ -245,7 +245,7 @@ define void @f_thunk(ptr %this, ...) {
 ; X86-NOSSE-NEXT:    movl %esp, %ebp
 ; X86-NOSSE-NEXT:    pushl %esi
 ; X86-NOSSE-NEXT:    andl $-16, %esp
-; X86-NOSSE-NEXT:    subl $32, %esp
+; X86-NOSSE-NEXT:    subl $16, %esp
 ; X86-NOSSE-NEXT:    movl 8(%ebp), %esi
 ; X86-NOSSE-NEXT:    leal 12(%ebp), %eax
 ; X86-NOSSE-NEXT:    movl %eax, (%esp)
@@ -264,7 +264,7 @@ define void @f_thunk(ptr %this, ...) {
 ; X86-SSE-NEXT:    movl %esp, %ebp
 ; X86-SSE-NEXT:    pushl %esi
 ; X86-SSE-NEXT:    andl $-16, %esp
-; X86-SSE-NEXT:    subl $80, %esp
+; X86-SSE-NEXT:    subl $64, %esp
 ; X86-SSE-NEXT:    movaps %xmm2, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
 ; X86-SSE-NEXT:    movaps %xmm1, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
 ; X86-SSE-NEXT:    movaps %xmm0, (%esp) # 16-byte Spill
diff --git a/llvm/test/CodeGen/X86/nontemporal-loads-2.ll b/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
index 71c23d88085ec8..a1e351683bd7db 100644
--- a/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
+++ b/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
@@ -204,7 +204,7 @@ define <4 x double> @test_v4f64_align16(ptr %src) nounwind {
 ; AVX-NEXT:    pushq %rbp
 ; AVX-NEXT:    movq %rsp, %rbp
 ; AVX-NEXT:    andq $-32, %rsp
-; AVX-NEXT:    subq $64, %rsp
+; AVX-NEXT:    subq $32, %rsp
 ; AVX-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -235,7 +235,7 @@ define <8 x float> @test_v8f32_align16(ptr %src) nounwind {
 ; AVX-NEXT:    pushq %rbp
 ; AVX-NEXT:    movq %rsp, %rbp
 ; AVX-NEXT:    andq $-32, %rsp
-; AVX-NEXT:    subq $64, %rsp
+; AVX-NEXT:    subq $32, %rsp
 ; AVX-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -266,7 +266,7 @@ define <4 x i64> @test_v4i64_align16(ptr %src) nounwind {
 ; AVX-NEXT:    pushq %rbp
 ; AVX-NEXT:    movq %rsp, %rbp
 ; AVX-NEXT:    andq $-32, %rsp
-; AVX-NEXT:    subq $64, %rsp
+; AVX-NEXT:    subq $32, %rsp
 ; AVX-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -297,7 +297,7 @@ define <8 x i32> @test_v8i32_align16(ptr %src) nounwind {
 ; AVX-NEXT:    pushq %rbp
 ; AVX-NEXT:    movq %rsp, %rbp
 ; AVX-NEXT:    andq $-32, %rsp
-; AVX-NEXT:    subq $64, %rsp
+; AVX-NEXT:    subq $32, %rsp
 ; AVX-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -328,7 +328,7 @@ define <16 x i16> @test_v16i16_align16(ptr %src) nounwind {
 ; AVX-NEXT:    pushq %rbp
 ; AVX-NEXT:    movq %rsp, %rbp
 ; AVX-NEXT:    andq $-32, %rsp
-; AVX-NEXT:    subq $64, %rsp
+; AVX-NEXT:    subq $32, %rsp
 ; AVX-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -359,7 +359,7 @@ define <32 x i8> @test_v32i8_align16(ptr %src) nounwind {
 ; AVX-NEXT:    pushq %rbp
 ; AVX-NEXT:    movq %rsp, %rbp
 ; AVX-NEXT:    andq $-32, %rsp
-; AVX-NEXT:    subq $64, %rsp
+; AVX-NEXT:    subq $32, %rsp
 ; AVX-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -570,7 +570,7 @@ define <8 x double> @test_v8f64_align16(ptr %src) nounwind {
 ; AVX1-NEXT:    pushq %rbp
 ; AVX1-NEXT:    movq %rsp, %rbp
 ; AVX1-NEXT:    andq $-32, %rsp
-; AVX1-NEXT:    subq $96, %rsp
+; AVX1-NEXT:    subq $64, %rsp
 ; AVX1-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX1-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX1-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -590,7 +590,7 @@ define <8 x double> @test_v8f64_align16(ptr %src) nounwind {
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX2-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX2-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -610,7 +610,7 @@ define <8 x double> @test_v8f64_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -649,7 +649,7 @@ define <16 x float> @test_v16f32_align16(ptr %src) nounwind {
 ; AVX1-NEXT:    pushq %rbp
 ; AVX1-NEXT:    movq %rsp, %rbp
 ; AVX1-NEXT:    andq $-32, %rsp
-; AVX1-NEXT:    subq $96, %rsp
+; AVX1-NEXT:    subq $64, %rsp
 ; AVX1-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX1-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX1-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -669,7 +669,7 @@ define <16 x float> @test_v16f32_align16(ptr %src) nounwind {
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX2-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX2-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -689,7 +689,7 @@ define <16 x float> @test_v16f32_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -728,7 +728,7 @@ define <8 x i64> @test_v8i64_align16(ptr %src) nounwind {
 ; AVX1-NEXT:    pushq %rbp
 ; AVX1-NEXT:    movq %rsp, %rbp
 ; AVX1-NEXT:    andq $-32, %rsp
-; AVX1-NEXT:    subq $96, %rsp
+; AVX1-NEXT:    subq $64, %rsp
 ; AVX1-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX1-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX1-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -748,7 +748,7 @@ define <8 x i64> @test_v8i64_align16(ptr %src) nounwind {
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX2-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX2-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -768,7 +768,7 @@ define <8 x i64> @test_v8i64_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -807,7 +807,7 @@ define <16 x i32> @test_v16i32_align16(ptr %src) nounwind {
 ; AVX1-NEXT:    pushq %rbp
 ; AVX1-NEXT:    movq %rsp, %rbp
 ; AVX1-NEXT:    andq $-32, %rsp
-; AVX1-NEXT:    subq $96, %rsp
+; AVX1-NEXT:    subq $64, %rsp
 ; AVX1-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX1-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX1-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -827,7 +827,7 @@ define <16 x i32> @test_v16i32_align16(ptr %src) nounwind {
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX2-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX2-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -847,7 +847,7 @@ define <16 x i32> @test_v16i32_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -886,7 +886,7 @@ define <32 x i16> @test_v32i16_align16(ptr %src) nounwind {
 ; AVX1-NEXT:    pushq %rbp
 ; AVX1-NEXT:    movq %rsp, %rbp
 ; AVX1-NEXT:    andq $-32, %rsp
-; AVX1-NEXT:    subq $96, %rsp
+; AVX1-NEXT:    subq $64, %rsp
 ; AVX1-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX1-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX1-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -906,7 +906,7 @@ define <32 x i16> @test_v32i16_align16(ptr %src) nounwind {
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX2-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX2-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -926,7 +926,7 @@ define <32 x i16> @test_v32i16_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -965,7 +965,7 @@ define <64 x i8> @test_v64i8_align16(ptr %src) nounwind {
 ; AVX1-NEXT:    pushq %rbp
 ; AVX1-NEXT:    movq %rsp, %rbp
 ; AVX1-NEXT:    andq $-32, %rsp
-; AVX1-NEXT:    subq $96, %rsp
+; AVX1-NEXT:    subq $64, %rsp
 ; AVX1-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX1-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX1-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -985,7 +985,7 @@ define <64 x i8> @test_v64i8_align16(ptr %src) nounwind {
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vmovntdqa 16(%rdi), %xmm0
 ; AVX2-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX2-NEXT:    vmovntdqa (%rdi), %xmm0
@@ -1005,7 +1005,7 @@ define <64 x i8> @test_v64i8_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -1060,7 +1060,7 @@ define <8 x double> @test_v8f64_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
@@ -1111,7 +1111,7 @@ define <16 x float> @test_v16f32_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
@@ -1162,7 +1162,7 @@ define <8 x i64> @test_v8i64_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
@@ -1213,7 +1213,7 @@ define <16 x i32> @test_v16i32_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
@@ -1264,7 +1264,7 @@ define <32 x i16> @test_v32i16_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
@@ -1315,7 +1315,7 @@ define <64 x i8> @test_v64i8_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
diff --git a/llvm/test/CodeGen/X86/nosse-vector.ll b/llvm/test/CodeGen/X86/nosse-vector.ll
index 9807d1b09d8ef6..1af1aad58a376d 100644
--- a/llvm/test/CodeGen/X86/nosse-vector.ll
+++ b/llvm/test/CodeGen/X86/nosse-vector.ll
@@ -143,7 +143,7 @@ define void @sitofp_4i64_4f32_mem(ptr %p0, ptr %p1) nounwind {
 ; X32-NEXT:    pushl %edi
 ; X32-NEXT:    pushl %esi
 ; X32-NEXT:    andl $-8, %esp
-; X32-NEXT:    subl $48, %esp
+; X32-NEXT:    subl $40, %esp
 ; X32-NEXT:    movl 8(%ebp), %edx
 ; X32-NEXT:    movl 24(%edx), %eax
 ; X32-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
diff --git a/llvm/test/CodeGen/X86/packus.ll b/llvm/test/CodeGen/X86/packus.ll
index 7de7caff18d6b0..8e5c383de8db2a 100644
--- a/llvm/test/CodeGen/X86/packus.ll
+++ b/llvm/test/CodeGen/X86/packus.ll
@@ -901,7 +901,7 @@ define <32 x i16> @_mm512_packus_epi32_manual(<16 x i32> %a, <16 x i32> %b) noun
 ; X86-SSE2-NEXT:    pushl %ebp
 ; X86-SSE2-NEXT:    movl %esp, %ebp
 ; X86-SSE2-NEXT:    andl $-16, %esp
-; X86-SSE2-NEXT:    subl $64, %esp
+; X86-SSE2-NEXT:    subl $48, %esp
 ; X86-SSE2-NEXT:    movdqa %xmm2, %xmm4
 ; X86-SSE2-NEXT:    movdqa %xmm1, %xmm6
 ; X86-SSE2-NEXT:    movdqa %xmm0, %xmm2
diff --git a/llvm/test/CodeGen/X86/pr216504.ll b/llvm/test/CodeGen/X86/pr216504.ll
index 6cf96095260886..098d974a8ac27e 100644
--- a/llvm/test/CodeGen/X86/pr216504.ll
+++ b/llvm/test/CodeGen/X86/pr216504.ll
@@ -14,7 +14,7 @@ define void @f25() nounwind {
 ; X64-NEXT:    pushq %rbp
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    andq $-32, %rsp
-; X64-NEXT:    subq $96, %rsp
+; X64-NEXT:    subq $64, %rsp
 ; X64-NEXT:    movq g2 at GOTPCREL(%rip), %rax
 ; X64-NEXT:    movzwl (%rax), %ecx
 ; X64-NEXT:    andl $7, %ecx
@@ -33,7 +33,7 @@ define void @f25() nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-32, %esp
-; X86-NEXT:    subl $96, %esp
+; X86-NEXT:    subl $64, %esp
 ; X86-NEXT:    movzwl g2, %eax
 ; X86-NEXT:    andl $7, %eax
 ; X86-NEXT:    vmovss {{.*#+}} xmm0 = [60,0,0,0]
diff --git a/llvm/test/CodeGen/X86/pr32284.ll b/llvm/test/CodeGen/X86/pr32284.ll
index 745ec17a63f435..cf2adb8e2231e9 100644
--- a/llvm/test/CodeGen/X86/pr32284.ll
+++ b/llvm/test/CodeGen/X86/pr32284.ll
@@ -746,7 +746,7 @@ define void @f3() #0 {
 ; X86-O0-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-O0-NEXT:    pushl %esi
 ; X86-O0-NEXT:    andl $-8, %esp
-; X86-O0-NEXT:    subl $16, %esp
+; X86-O0-NEXT:    subl $8, %esp
 ; X86-O0-NEXT:    .cfi_offset %esi, -12
 ; X86-O0-NEXT:    movl var_13, %ecx
 ; X86-O0-NEXT:    movl %ecx, %eax
diff --git a/llvm/test/CodeGen/X86/pr34080-2.ll b/llvm/test/CodeGen/X86/pr34080-2.ll
index 279373a7aab3fa..3f228dc2241300 100644
--- a/llvm/test/CodeGen/X86/pr34080-2.ll
+++ b/llvm/test/CodeGen/X86/pr34080-2.ll
@@ -12,7 +12,7 @@ define void @computeJD(ptr) nounwind {
 ; CHECK-NEXT:    pushl %edi
 ; CHECK-NEXT:    pushl %esi
 ; CHECK-NEXT:    andl $-8, %esp
-; CHECK-NEXT:    subl $40, %esp
+; CHECK-NEXT:    subl $32, %esp
 ; CHECK-NEXT:    movl 8(%ebp), %ebx
 ; CHECK-NEXT:    movl 8(%ebx), %esi
 ; CHECK-NEXT:    xorl %eax, %eax
diff --git a/llvm/test/CodeGen/X86/pr34592.ll b/llvm/test/CodeGen/X86/pr34592.ll
index e0423d595303ec..7ddb62c1851d39 100644
--- a/llvm/test/CodeGen/X86/pr34592.ll
+++ b/llvm/test/CodeGen/X86/pr34592.ll
@@ -8,7 +8,7 @@ define <16 x i64> @pluto(<16 x i64> %arg, <16 x i64> %arg1, <16 x i64> %arg2, <1
 ; CHECK-O0-NEXT:    pushq %rbp
 ; CHECK-O0-NEXT:    movq %rsp, %rbp
 ; CHECK-O0-NEXT:    andq $-32, %rsp
-; CHECK-O0-NEXT:    subq $64, %rsp
+; CHECK-O0-NEXT:    subq $32, %rsp
 ; CHECK-O0-NEXT:    vmovaps %ymm4, %ymm10
 ; CHECK-O0-NEXT:    vmovaps %ymm3, %ymm9
 ; CHECK-O0-NEXT:    vmovaps %ymm2, (%rsp) # 32-byte Spill
diff --git a/llvm/test/CodeGen/X86/pr34653.ll b/llvm/test/CodeGen/X86/pr34653.ll
index d46cd2091856eb..ca6e47dccfc5fc 100644
--- a/llvm/test/CodeGen/X86/pr34653.ll
+++ b/llvm/test/CodeGen/X86/pr34653.ll
@@ -12,7 +12,7 @@ define void @pr34653() {
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    .cfi_def_cfa_register %rbp
 ; CHECK-NEXT:    andq $-512, %rsp # imm = 0xFE00
-; CHECK-NEXT:    subq $1024, %rsp # imm = 0x400
+; CHECK-NEXT:    subq $512, %rsp # imm = 0x200
 ; CHECK-NEXT:    movq %rsp, %rdi
 ; CHECK-NEXT:    callq test at PLT
 ; CHECK-NEXT:    vmovsd {{.*#+}} xmm0 = mem[0],zero
diff --git a/llvm/test/CodeGen/X86/pr38539.ll b/llvm/test/CodeGen/X86/pr38539.ll
index e0e6f6bec48764..722154e3bdd114 100644
--- a/llvm/test/CodeGen/X86/pr38539.ll
+++ b/llvm/test/CodeGen/X86/pr38539.ll
@@ -22,7 +22,7 @@ define void @f() nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $96, %esp
+; X86-NEXT:    subl $80, %esp
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %edx
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %ebx
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %esi
diff --git a/llvm/test/CodeGen/X86/pr43866.ll b/llvm/test/CodeGen/X86/pr43866.ll
index 20eedbc942277a..28505e8daa4959 100644
--- a/llvm/test/CodeGen/X86/pr43866.ll
+++ b/llvm/test/CodeGen/X86/pr43866.ll
@@ -12,7 +12,7 @@ define dso_local void @test()  {
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    .cfi_def_cfa_register %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    vmovsd {{.*#+}} xmm0 = mem[0],zero
 ; CHECK-NEXT:    vpcmpeqd %xmm1, %xmm1, %xmm1
 ; CHECK-NEXT:    vshufps {{.*#+}} xmm2 = xmm1[1,0],xmm0[1,0]
diff --git a/llvm/test/CodeGen/X86/pr50782.ll b/llvm/test/CodeGen/X86/pr50782.ll
index 591a33446d4e30..b9e0e1ab17e165 100644
--- a/llvm/test/CodeGen/X86/pr50782.ll
+++ b/llvm/test/CodeGen/X86/pr50782.ll
@@ -20,7 +20,7 @@ define void @h(float %i) {
 ; CHECK-NEXT:    .cfi_def_cfa_register %ebp
 ; CHECK-NEXT:    pushl %esi
 ; CHECK-NEXT:    andl $-16, %esp
-; CHECK-NEXT:    subl $32, %esp
+; CHECK-NEXT:    subl $16, %esp
 ; CHECK-NEXT:    movl %esp, %esi
 ; CHECK-NEXT:    .cfi_offset %esi, -12
 ; CHECK-NEXT:    flds 8(%ebp)
diff --git a/llvm/test/CodeGen/X86/sdiv_fix.ll b/llvm/test/CodeGen/X86/sdiv_fix.ll
index 392bc83d9d5d8d..70ae75906f071a 100644
--- a/llvm/test/CodeGen/X86/sdiv_fix.ll
+++ b/llvm/test/CodeGen/X86/sdiv_fix.ll
@@ -307,7 +307,7 @@ define i64 @func5(i64 %x, i64 %y) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movl 8(%ebp), %ecx
 ; X86-NEXT:    movl 12(%ebp), %edi
 ; X86-NEXT:    movl 16(%ebp), %eax
diff --git a/llvm/test/CodeGen/X86/sdiv_fix_sat.ll b/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
index eac2bb7ebe9620..ecdb82078c55b7 100644
--- a/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
+++ b/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
@@ -370,7 +370,7 @@ define i64 @func5(i64 %x, i64 %y) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    addl $-128, %esp
+; X86-NEXT:    subl $112, %esp
 ; X86-NEXT:    movl 8(%ebp), %esi
 ; X86-NEXT:    movl 12(%ebp), %edi
 ; X86-NEXT:    movl 16(%ebp), %ecx
@@ -799,7 +799,7 @@ define <4 x i32> @vec(<4 x i32> %x, <4 x i32> %y) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $240, %esp
+; X86-NEXT:    subl $224, %esp
 ; X86-NEXT:    movl 20(%ebp), %esi
 ; X86-NEXT:    movl 36(%ebp), %ebx
 ; X86-NEXT:    movl 16(%ebp), %ecx
diff --git a/llvm/test/CodeGen/X86/shift-i128.ll b/llvm/test/CodeGen/X86/shift-i128.ll
index 6c9cdb1b62ca4a..d100c6f69536e0 100644
--- a/llvm/test/CodeGen/X86/shift-i128.ll
+++ b/llvm/test/CodeGen/X86/shift-i128.ll
@@ -15,7 +15,7 @@ define void @test_lshr_i128(i128 %x, i128 %a, ptr nocapture %r) nounwind {
 ; i686-NEXT:    pushl %edi
 ; i686-NEXT:    pushl %esi
 ; i686-NEXT:    andl $-16, %esp
-; i686-NEXT:    subl $48, %esp
+; i686-NEXT:    subl $32, %esp
 ; i686-NEXT:    movzbl 24(%ebp), %ecx
 ; i686-NEXT:    movl 8(%ebp), %eax
 ; i686-NEXT:    movl 12(%ebp), %edx
@@ -81,7 +81,7 @@ define void @test_ashr_i128(i128 %x, i128 %a, ptr nocapture %r) nounwind {
 ; i686-NEXT:    pushl %edi
 ; i686-NEXT:    pushl %esi
 ; i686-NEXT:    andl $-16, %esp
-; i686-NEXT:    subl $48, %esp
+; i686-NEXT:    subl $32, %esp
 ; i686-NEXT:    movzbl 24(%ebp), %ecx
 ; i686-NEXT:    movl 8(%ebp), %eax
 ; i686-NEXT:    movl 12(%ebp), %edx
@@ -149,7 +149,7 @@ define void @test_shl_i128(i128 %x, i128 %a, ptr nocapture %r) nounwind {
 ; i686-NEXT:    pushl %edi
 ; i686-NEXT:    pushl %esi
 ; i686-NEXT:    andl $-16, %esp
-; i686-NEXT:    subl $48, %esp
+; i686-NEXT:    subl $32, %esp
 ; i686-NEXT:    movzbl 24(%ebp), %ecx
 ; i686-NEXT:    movl 8(%ebp), %eax
 ; i686-NEXT:    movl 12(%ebp), %edx
@@ -278,7 +278,7 @@ define void @test_lshr_v2i128(<2 x i128> %x, <2 x i128> %a, ptr nocapture %r) no
 ; i686-NEXT:    pushl %edi
 ; i686-NEXT:    pushl %esi
 ; i686-NEXT:    andl $-16, %esp
-; i686-NEXT:    subl $112, %esp
+; i686-NEXT:    subl $96, %esp
 ; i686-NEXT:    movl 40(%ebp), %edx
 ; i686-NEXT:    movl 24(%ebp), %eax
 ; i686-NEXT:    movl 28(%ebp), %ecx
@@ -405,7 +405,7 @@ define void @test_ashr_v2i128(<2 x i128> %x, <2 x i128> %a, ptr nocapture %r) no
 ; i686-NEXT:    pushl %edi
 ; i686-NEXT:    pushl %esi
 ; i686-NEXT:    andl $-16, %esp
-; i686-NEXT:    subl $112, %esp
+; i686-NEXT:    subl $96, %esp
 ; i686-NEXT:    movl 40(%ebp), %edx
 ; i686-NEXT:    movl 24(%ebp), %eax
 ; i686-NEXT:    movl 28(%ebp), %ecx
@@ -538,7 +538,7 @@ define void @test_shl_v2i128(<2 x i128> %x, <2 x i128> %a, ptr nocapture %r) nou
 ; i686-NEXT:    pushl %edi
 ; i686-NEXT:    pushl %esi
 ; i686-NEXT:    andl $-16, %esp
-; i686-NEXT:    addl $-128, %esp
+; i686-NEXT:    subl $112, %esp
 ; i686-NEXT:    movl 40(%ebp), %edi
 ; i686-NEXT:    movl 24(%ebp), %eax
 ; i686-NEXT:    movl 28(%ebp), %ecx
@@ -1004,7 +1004,7 @@ define i128 @shift_i128_limited_shamt_no_nuw(i128 noundef %a, i32 noundef %b) no
 ; i686-NEXT:    pushl %edi
 ; i686-NEXT:    pushl %esi
 ; i686-NEXT:    andl $-16, %esp
-; i686-NEXT:    subl $48, %esp
+; i686-NEXT:    subl $32, %esp
 ; i686-NEXT:    movzbl 40(%ebp), %eax
 ; i686-NEXT:    movl 24(%ebp), %ecx
 ; i686-NEXT:    movl 28(%ebp), %edx
@@ -1076,7 +1076,7 @@ define i128 @shift_i128_limited_shamt_unknown_lhs(i128 noundef %a, i32 noundef %
 ; i686-NEXT:    pushl %edi
 ; i686-NEXT:    pushl %esi
 ; i686-NEXT:    andl $-16, %esp
-; i686-NEXT:    subl $48, %esp
+; i686-NEXT:    subl $32, %esp
 ; i686-NEXT:    movl 24(%ebp), %eax
 ; i686-NEXT:    movl 28(%ebp), %edx
 ; i686-NEXT:    movl 32(%ebp), %esi
diff --git a/llvm/test/CodeGen/X86/shift-i256.ll b/llvm/test/CodeGen/X86/shift-i256.ll
index 4933ef441de6b1..68ee0715aaff7a 100644
--- a/llvm/test/CodeGen/X86/shift-i256.ll
+++ b/llvm/test/CodeGen/X86/shift-i256.ll
@@ -171,7 +171,7 @@ define i256 @shl_i256(i256 %a0, i256 %a1) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movzbl 44(%ebp), %ecx
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    movl 16(%ebp), %edx
@@ -406,7 +406,7 @@ define i256 @lshr_i256(i256 %a0, i256 %a1) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movzbl 44(%ebp), %ecx
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    movl 16(%ebp), %edx
@@ -644,7 +644,7 @@ define i256 @ashr_i256(i256 %a0, i256 %a1) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movzbl 44(%ebp), %ecx
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    movl 16(%ebp), %edx
@@ -878,7 +878,7 @@ define i256 @shl_i256_load(ptr %p0, i256 %a1) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movl 12(%ebp), %ecx
 ; X86-NEXT:    movl (%ecx), %eax
 ; X86-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -1112,7 +1112,7 @@ define i256 @lshr_i256_load(ptr %p0, i256 %a1) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movl 12(%ebp), %ecx
 ; X86-NEXT:    movl (%ecx), %eax
 ; X86-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -1367,7 +1367,7 @@ define i256 @ashr_i256_load(ptr %p0, i256 %a1) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    movl (%eax), %ecx
 ; X86-NEXT:    movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -2041,7 +2041,7 @@ define i256 @shl_1_i256(i256 %a0) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movzbl 12(%ebp), %ecx
 ; X86-NEXT:    movl $0, {{[0-9]+}}(%esp)
 ; X86-NEXT:    movl $0, {{[0-9]+}}(%esp)
@@ -2259,7 +2259,7 @@ define i256 @lshr_signbit_i256(i256 %a0) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movzbl 12(%ebp), %ecx
 ; X86-NEXT:    movl $0, {{[0-9]+}}(%esp)
 ; X86-NEXT:    movl $0, {{[0-9]+}}(%esp)
@@ -2472,7 +2472,7 @@ define i256 @ashr_signbit_i256(i256 %a0) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movzbl 12(%ebp), %ecx
 ; X86-NEXT:    movl $-1, {{[0-9]+}}(%esp)
 ; X86-NEXT:    movl $-1, {{[0-9]+}}(%esp)
@@ -2696,7 +2696,7 @@ define i256 @shl_allbits_i256(i256 %a0) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movzbl 12(%ebp), %ecx
 ; X86-NEXT:    movl $-1, {{[0-9]+}}(%esp)
 ; X86-NEXT:    movl $-1, {{[0-9]+}}(%esp)
@@ -2915,7 +2915,7 @@ define i256 @lshr_allbits_i256(i256 %a0) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $112, %esp
+; X86-NEXT:    subl $96, %esp
 ; X86-NEXT:    movzbl 12(%ebp), %ecx
 ; X86-NEXT:    movl $0, {{[0-9]+}}(%esp)
 ; X86-NEXT:    movl $0, {{[0-9]+}}(%esp)
@@ -3292,7 +3292,7 @@ define i64 @lshr_extract_load_i256_i64(ptr %p0, i256 %a1) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $96, %esp
+; X86-NEXT:    subl $80, %esp
 ; X86-NEXT:    movl 8(%ebp), %ecx
 ; X86-NEXT:    movl (%ecx), %eax
 ; X86-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -3422,7 +3422,7 @@ define i64 @ashr_extract_load_i256_i64(ptr %p0, i256 %a1) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $96, %esp
+; X86-NEXT:    subl $80, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    movl (%eax), %ecx
 ; X86-NEXT:    movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
@@ -3560,7 +3560,7 @@ define i64 @ashr_extract_idx_load_i256_i64(ptr %p0, i256 %a1) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $96, %esp
+; X86-NEXT:    subl $80, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    movl (%eax), %ecx
 ; X86-NEXT:    movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
diff --git a/llvm/test/CodeGen/X86/shuffle-combine-crash-4.ll b/llvm/test/CodeGen/X86/shuffle-combine-crash-4.ll
index 8ea99b313838a8..fa9861d69bf0b0 100644
--- a/llvm/test/CodeGen/X86/shuffle-combine-crash-4.ll
+++ b/llvm/test/CodeGen/X86/shuffle-combine-crash-4.ll
@@ -13,7 +13,7 @@ define void @infiloop() {
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    .cfi_def_cfa_register %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    movabsq $506097522914230528, %rax # imm = 0x706050403020100
 ; CHECK-NEXT:    movq %rax, test24_id5239(%rip)
 ; CHECK-NEXT:    vmovaps {{.*#+}} ymm0 = [4,1,6,7,6,7,2,3,6,7,4,5,6,7,2,3,6,7,2,3,2,3,2,3,4,5,4,5,2,3,0,1]
diff --git a/llvm/test/CodeGen/X86/sse-intel-ocl.ll b/llvm/test/CodeGen/X86/sse-intel-ocl.ll
index b2de7545ff5f51..f9f1c593dce78e 100644
--- a/llvm/test/CodeGen/X86/sse-intel-ocl.ll
+++ b/llvm/test/CodeGen/X86/sse-intel-ocl.ll
@@ -13,7 +13,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
 ; WIN32-NEXT:    pushl %ebp
 ; WIN32-NEXT:    movl %esp, %ebp
 ; WIN32-NEXT:    andl $-16, %esp
-; WIN32-NEXT:    subl $80, %esp
+; WIN32-NEXT:    subl $64, %esp
 ; WIN32-NEXT:    movups 72(%ebp), %xmm4
 ; WIN32-NEXT:    movups 8(%ebp), %xmm3
 ; WIN32-NEXT:    addps %xmm4, %xmm3
@@ -91,7 +91,7 @@ define <16 x float> @testf16_regs(<16 x float> %a, <16 x float> %b) nounwind {
 ; WIN32-NEXT:    pushl %ebp
 ; WIN32-NEXT:    movl %esp, %ebp
 ; WIN32-NEXT:    andl $-16, %esp
-; WIN32-NEXT:    subl $80, %esp
+; WIN32-NEXT:    subl $64, %esp
 ; WIN32-NEXT:    movups 72(%ebp), %xmm6
 ; WIN32-NEXT:    movups 8(%ebp), %xmm3
 ; WIN32-NEXT:    movups 56(%ebp), %xmm7
diff --git a/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll b/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
index 2e2e78a6da51e2..913e54873fc621 100644
--- a/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
+++ b/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
@@ -2754,7 +2754,7 @@ define void @test_mm_storeh_pi(ptr %a0, <4 x float> %a1) nounwind {
 ; X86-SSE1-NEXT:    pushl %ebp # encoding: [0x55]
 ; X86-SSE1-NEXT:    movl %esp, %ebp # encoding: [0x89,0xe5]
 ; X86-SSE1-NEXT:    andl $-16, %esp # encoding: [0x83,0xe4,0xf0]
-; X86-SSE1-NEXT:    subl $32, %esp # encoding: [0x83,0xec,0x20]
+; X86-SSE1-NEXT:    subl $16, %esp # encoding: [0x83,0xec,0x10]
 ; X86-SSE1-NEXT:    movl 8(%ebp), %eax # encoding: [0x8b,0x45,0x08]
 ; X86-SSE1-NEXT:    movaps %xmm0, (%esp) # encoding: [0x0f,0x29,0x04,0x24]
 ; X86-SSE1-NEXT:    movl {{[0-9]+}}(%esp), %ecx # encoding: [0x8b,0x4c,0x24,0x08]
@@ -2861,7 +2861,7 @@ define void @test_mm_storel_pi(ptr %a0, <4 x float> %a1) nounwind {
 ; X86-SSE1-NEXT:    pushl %ebp # encoding: [0x55]
 ; X86-SSE1-NEXT:    movl %esp, %ebp # encoding: [0x89,0xe5]
 ; X86-SSE1-NEXT:    andl $-16, %esp # encoding: [0x83,0xe4,0xf0]
-; X86-SSE1-NEXT:    subl $32, %esp # encoding: [0x83,0xec,0x20]
+; X86-SSE1-NEXT:    subl $16, %esp # encoding: [0x83,0xec,0x10]
 ; X86-SSE1-NEXT:    movl 8(%ebp), %eax # encoding: [0x8b,0x45,0x08]
 ; X86-SSE1-NEXT:    movaps %xmm0, (%esp) # encoding: [0x0f,0x29,0x04,0x24]
 ; X86-SSE1-NEXT:    movl (%esp), %ecx # encoding: [0x8b,0x0c,0x24]
diff --git a/llvm/test/CodeGen/X86/sse-regcall.ll b/llvm/test/CodeGen/X86/sse-regcall.ll
index 03b9e123eea486..e7c02b58e4d967 100644
--- a/llvm/test/CodeGen/X86/sse-regcall.ll
+++ b/llvm/test/CodeGen/X86/sse-regcall.ll
@@ -75,7 +75,7 @@ define x86_regcallcc <16 x float> @testf32_inp(<16 x float> %a, <16 x float> %b,
 ; WIN32-NEXT:    pushl %ebp
 ; WIN32-NEXT:    movl %esp, %ebp
 ; WIN32-NEXT:    andl $-16, %esp
-; WIN32-NEXT:    subl $32, %esp
+; WIN32-NEXT:    subl $16, %esp
 ; WIN32-NEXT:    movaps %xmm7, (%esp) # 16-byte Spill
 ; WIN32-NEXT:    movaps %xmm6, %xmm7
 ; WIN32-NEXT:    movaps %xmm5, %xmm6
@@ -364,7 +364,7 @@ define x86_regcallcc <32 x float> @testf32_stack(<32 x float> %a, <32 x float> %
 ; WIN32-NEXT:    pushl %ebp
 ; WIN32-NEXT:    movl %esp, %ebp
 ; WIN32-NEXT:    andl $-16, %esp
-; WIN32-NEXT:    subl $48, %esp
+; WIN32-NEXT:    subl $32, %esp
 ; WIN32-NEXT:    movaps %xmm7, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
 ; WIN32-NEXT:    movaps %xmm6, (%esp) # 16-byte Spill
 ; WIN32-NEXT:    movaps %xmm5, %xmm6
diff --git a/llvm/test/CodeGen/X86/sse-regcall4.ll b/llvm/test/CodeGen/X86/sse-regcall4.ll
index 6f964f0a88ea3d..2aae7af393c10d 100644
--- a/llvm/test/CodeGen/X86/sse-regcall4.ll
+++ b/llvm/test/CodeGen/X86/sse-regcall4.ll
@@ -75,7 +75,7 @@ define x86_regcallcc <16 x float> @testf32_inp(<16 x float> %a, <16 x float> %b,
 ; WIN32-NEXT:    pushl %ebp
 ; WIN32-NEXT:    movl %esp, %ebp
 ; WIN32-NEXT:    andl $-16, %esp
-; WIN32-NEXT:    subl $32, %esp
+; WIN32-NEXT:    subl $16, %esp
 ; WIN32-NEXT:    movaps %xmm7, (%esp) # 16-byte Spill
 ; WIN32-NEXT:    movaps %xmm6, %xmm7
 ; WIN32-NEXT:    movaps %xmm5, %xmm6
@@ -363,7 +363,7 @@ define x86_regcallcc <32 x float> @testf32_stack(<32 x float> %a, <32 x float> %
 ; WIN32-NEXT:    pushl %ebp
 ; WIN32-NEXT:    movl %esp, %ebp
 ; WIN32-NEXT:    andl $-16, %esp
-; WIN32-NEXT:    subl $48, %esp
+; WIN32-NEXT:    subl $32, %esp
 ; WIN32-NEXT:    movaps %xmm7, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
 ; WIN32-NEXT:    movaps %xmm6, (%esp) # 16-byte Spill
 ; WIN32-NEXT:    movaps %xmm5, %xmm6
diff --git a/llvm/test/CodeGen/X86/stack-clash-small-alloc-medium-align.ll b/llvm/test/CodeGen/X86/stack-clash-small-alloc-medium-align.ll
index 01a1cb136a49ac..af82f350245867 100644
--- a/llvm/test/CodeGen/X86/stack-clash-small-alloc-medium-align.ll
+++ b/llvm/test/CodeGen/X86/stack-clash-small-alloc-medium-align.ll
@@ -93,7 +93,7 @@ define i32 @foo4(i64 %i) local_unnamed_addr #0 {
 ; CHECK-NEXT:    .cfi_def_cfa_register %rbp
 ; CHECK-NEXT:    pushq %rbx
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $896, %rsp # imm = 0x380
+; CHECK-NEXT:    subq $832, %rsp # imm = 0x340
 ; CHECK-NEXT:    movq %rsp, %rbx
 ; CHECK-NEXT:    .cfi_offset %rbx, -24
 ; CHECK-NEXT:    movl $1, (%rbx,%rdi,4)
diff --git a/llvm/test/CodeGen/X86/stack-realign-local-padding.ll b/llvm/test/CodeGen/X86/stack-realign-local-padding.ll
new file mode 100644
index 00000000000000..2d1d084dfba0d6
--- /dev/null
+++ b/llvm/test/CodeGen/X86/stack-realign-local-padding.ll
@@ -0,0 +1,104 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+
+; PEI realigns the stack but leaves locals relative to the incoming SP,
+; so padding between saved registers and locals must not be allocated.
+; GPR/XMM saves must not affect it, two-block objects still need both.
+
+declare void @use(ptr)
+
+define void @one_block() nounwind {
+; CHECK-LABEL: one_block:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    pushq %rbp
+; CHECK-NEXT:    movq %rsp, %rbp
+; CHECK-NEXT:    andq $-64, %rsp
+; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    movq %rsp, %rdi
+; CHECK-NEXT:    callq use at PLT
+; CHECK-NEXT:    movq %rbp, %rsp
+; CHECK-NEXT:    popq %rbp
+; CHECK-NEXT:    retq
+  %a = alloca [64 x i8], align 64
+  call void @use(ptr %a)
+  ret void
+}
+
+define void @two_blocks() nounwind {
+; CHECK-LABEL: two_blocks:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    pushq %rbp
+; CHECK-NEXT:    movq %rsp, %rbp
+; CHECK-NEXT:    andq $-64, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
+; CHECK-NEXT:    movq %rsp, %rdi
+; CHECK-NEXT:    callq use at PLT
+; CHECK-NEXT:    movq %rbp, %rsp
+; CHECK-NEXT:    popq %rbp
+; CHECK-NEXT:    retq
+  %a = alloca [65 x i8], align 64
+  call void @use(ptr %a)
+  ret void
+}
+
+define void @gpr_csrs() nounwind {
+; CHECK-LABEL: gpr_csrs:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    pushq %rbp
+; CHECK-NEXT:    movq %rsp, %rbp
+; CHECK-NEXT:    pushq %r13
+; CHECK-NEXT:    pushq %r12
+; CHECK-NEXT:    pushq %rbx
+; CHECK-NEXT:    andq $-64, %rsp
+; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    #APP
+; CHECK-NEXT:    #NO_APP
+; CHECK-NEXT:    movq %rsp, %rdi
+; CHECK-NEXT:    callq use at PLT
+; CHECK-NEXT:    leaq -24(%rbp), %rsp
+; CHECK-NEXT:    popq %rbx
+; CHECK-NEXT:    popq %r12
+; CHECK-NEXT:    popq %r13
+; CHECK-NEXT:    popq %rbp
+; CHECK-NEXT:    retq
+  %a = alloca [64 x i8], align 64
+  call void asm sideeffect "", "~{rbx},~{r12},~{r13}"()
+  call void @use(ptr %a)
+  ret void
+}
+
+define x86_regcallcc void @xmm_csrs() nounwind {
+; CHECK-LABEL: xmm_csrs:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    pushq %rbp
+; CHECK-NEXT:    movq %rsp, %rbp
+; CHECK-NEXT:    andq $-64, %rsp
+; CHECK-NEXT:    subq $192, %rsp
+; CHECK-NEXT:    movaps %xmm15, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT:    movaps %xmm14, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT:    movaps %xmm13, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT:    movaps %xmm12, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT:    movaps %xmm11, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT:    movaps %xmm10, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT:    movaps %xmm9, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT:    movaps %xmm8, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; CHECK-NEXT:    #APP
+; CHECK-NEXT:    #NO_APP
+; CHECK-NEXT:    movq %rsp, %rdi
+; CHECK-NEXT:    callq use at PLT
+; CHECK-NEXT:    movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm8 # 16-byte Reload
+; CHECK-NEXT:    movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm9 # 16-byte Reload
+; CHECK-NEXT:    movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm10 # 16-byte Reload
+; CHECK-NEXT:    movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm11 # 16-byte Reload
+; CHECK-NEXT:    movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm12 # 16-byte Reload
+; CHECK-NEXT:    movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm13 # 16-byte Reload
+; CHECK-NEXT:    movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm14 # 16-byte Reload
+; CHECK-NEXT:    movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm15 # 16-byte Reload
+; CHECK-NEXT:    movq %rbp, %rsp
+; CHECK-NEXT:    popq %rbp
+; CHECK-NEXT:    retq
+  %a = alloca [64 x i8], align 64
+  call void asm sideeffect "", "~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15}"()
+  call void @use(ptr %a)
+  ret void
+}
diff --git a/llvm/test/CodeGen/X86/statepoint-no-realign-stack.ll b/llvm/test/CodeGen/X86/statepoint-no-realign-stack.ll
index b531f00b4be5b9..87e8a2094da4c7 100644
--- a/llvm/test/CodeGen/X86/statepoint-no-realign-stack.ll
+++ b/llvm/test/CodeGen/X86/statepoint-no-realign-stack.ll
@@ -20,7 +20,7 @@ define void @can_realign(ptr %p) {
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    .cfi_def_cfa_register %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    vmovaps (%rdi), %ymm0
 ; CHECK-NEXT:    vmovaps %ymm0, (%rsp)
 ; CHECK-NEXT:    vzeroupper
@@ -65,7 +65,7 @@ define <4 x ptr addrspace(1)> @spillfill_can_realign(<4 x ptr addrspace(1)> %obj
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    .cfi_def_cfa_register %rbp
 ; CHECK-NEXT:    andq $-32, %rsp
-; CHECK-NEXT:    subq $64, %rsp
+; CHECK-NEXT:    subq $32, %rsp
 ; CHECK-NEXT:    vmovaps %ymm0, (%rsp)
 ; CHECK-NEXT:    vzeroupper
 ; CHECK-NEXT:    callq do_safepoint at PLT
diff --git a/llvm/test/CodeGen/X86/sttni.ll b/llvm/test/CodeGen/X86/sttni.ll
index 39cbee54737c38..7cbfd46c390793 100644
--- a/llvm/test/CodeGen/X86/sttni.ll
+++ b/llvm/test/CodeGen/X86/sttni.ll
@@ -58,7 +58,7 @@ define i32 @pcmpestri_reg_diff_i8(<16 x i8> %lhs, i32 %lhs_len, <16 x i8> %rhs,
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    movl 12(%ebp), %edx
 ; X86-NEXT:    pcmpestri $24, %xmm1, %xmm0
@@ -184,7 +184,7 @@ define i32 @pcmpestri_mem_diff_i8(ptr %lhs_ptr, i32 %lhs_len, ptr %rhs_ptr, i32
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    movl 20(%ebp), %edx
 ; X86-NEXT:    movl 16(%ebp), %ecx
@@ -304,7 +304,7 @@ define i32 @pcmpestri_reg_diff_i16(<8 x i16> %lhs, i32 %lhs_len, <8 x i16> %rhs,
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    movl 12(%ebp), %edx
 ; X86-NEXT:    pcmpestri $24, %xmm1, %xmm0
@@ -438,7 +438,7 @@ define i32 @pcmpestri_mem_diff_i16(ptr %lhs_ptr, i32 %lhs_len, ptr %rhs_ptr, i32
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    movl 20(%ebp), %edx
 ; X86-NEXT:    movl 16(%ebp), %ecx
@@ -558,7 +558,7 @@ define i32 @pcmpistri_reg_diff_i8(<16 x i8> %lhs, <16 x i8> %rhs) nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    movdqa %xmm0, (%esp)
 ; X86-NEXT:    andl $15, %ecx
 ; X86-NEXT:    movzbl (%esp,%ecx), %eax
@@ -657,7 +657,7 @@ define i32 @pcmpistri_mem_diff_i8(ptr %lhs_ptr, ptr %rhs_ptr) nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    movl 8(%ebp), %ecx
 ; X86-NEXT:    movdqu (%ecx), %xmm1
@@ -772,7 +772,7 @@ define i32 @pcmpistri_reg_diff_i16(<8 x i16> %lhs, <8 x i16> %rhs) nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    movdqa %xmm0, (%esp)
 ; X86-NEXT:    addl %ecx, %ecx
 ; X86-NEXT:    andl $14, %ecx
@@ -879,7 +879,7 @@ define i32 @pcmpistri_mem_diff_i16(ptr %lhs_ptr, ptr %rhs_ptr) nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $48, %esp
+; X86-NEXT:    subl $32, %esp
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    movl 8(%ebp), %ecx
 ; X86-NEXT:    movdqu (%ecx), %xmm1
diff --git a/llvm/test/CodeGen/X86/udiv_fix.ll b/llvm/test/CodeGen/X86/udiv_fix.ll
index 8c3698c535b005..584121d34d6f38 100644
--- a/llvm/test/CodeGen/X86/udiv_fix.ll
+++ b/llvm/test/CodeGen/X86/udiv_fix.ll
@@ -143,7 +143,7 @@ define i64 @func5(i64 %x, i64 %y) nounwind {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $80, %esp
+; X86-NEXT:    subl $64, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    movl 12(%ebp), %ecx
 ; X86-NEXT:    movl 16(%ebp), %edx
diff --git a/llvm/test/CodeGen/X86/udiv_fix_sat.ll b/llvm/test/CodeGen/X86/udiv_fix_sat.ll
index 659ab7e69e9aeb..81bf490e24bec9 100644
--- a/llvm/test/CodeGen/X86/udiv_fix_sat.ll
+++ b/llvm/test/CodeGen/X86/udiv_fix_sat.ll
@@ -176,7 +176,7 @@ define i64 @func5(i64 %x, i64 %y) nounwind {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $80, %esp
+; X86-NEXT:    subl $64, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    movl 12(%ebp), %ecx
 ; X86-NEXT:    movl 16(%ebp), %edx
diff --git a/llvm/test/CodeGen/X86/var-permute-256.ll b/llvm/test/CodeGen/X86/var-permute-256.ll
index 0ca24b4c85f90e..5155077f41df09 100644
--- a/llvm/test/CodeGen/X86/var-permute-256.ll
+++ b/llvm/test/CodeGen/X86/var-permute-256.ll
@@ -2010,7 +2010,7 @@ define <4 x i64> @PR50356(<4 x i64> %0, <4 x i32> %1, <4 x i64> %2) unnamed_addr
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $64, %rsp
+; AVX2-NEXT:    subq $32, %rsp
 ; AVX2-NEXT:    vmovd %xmm1, %eax
 ; AVX2-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX2-NEXT:    andl $3, %eax
@@ -2031,7 +2031,7 @@ define <4 x i64> @PR50356(<4 x i64> %0, <4 x i32> %1, <4 x i64> %2) unnamed_addr
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-32, %rsp
-; AVX512-NEXT:    subq $64, %rsp
+; AVX512-NEXT:    subq $32, %rsp
 ; AVX512-NEXT:    # kill: def $ymm2 killed $ymm2 def $zmm2
 ; AVX512-NEXT:    vmovd %xmm1, %eax
 ; AVX512-NEXT:    vmovaps %ymm0, (%rsp)
@@ -2055,7 +2055,7 @@ define <4 x i64> @PR50356(<4 x i64> %0, <4 x i32> %1, <4 x i64> %2) unnamed_addr
 ; AVX512VL-NEXT:    pushq %rbp
 ; AVX512VL-NEXT:    movq %rsp, %rbp
 ; AVX512VL-NEXT:    andq $-32, %rsp
-; AVX512VL-NEXT:    subq $64, %rsp
+; AVX512VL-NEXT:    subq $32, %rsp
 ; AVX512VL-NEXT:    vmovd %xmm1, %eax
 ; AVX512VL-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX512VL-NEXT:    andl $3, %eax
diff --git a/llvm/test/CodeGen/X86/var-permute-512.ll b/llvm/test/CodeGen/X86/var-permute-512.ll
index 6b893f834f50de..9511b6c35f6160 100644
--- a/llvm/test/CodeGen/X86/var-permute-512.ll
+++ b/llvm/test/CodeGen/X86/var-permute-512.ll
@@ -97,7 +97,7 @@ define <32 x i16> @var_shuffle_v32i16(<32 x i16> %v, <32 x i16> %indices) nounwi
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-64, %rsp
-; AVX512F-NEXT:    addq $-128, %rsp
+; AVX512F-NEXT:    subq $64, %rsp
 ; AVX512F-NEXT:    vextracti128 $1, %ymm1, %xmm2
 ; AVX512F-NEXT:    vextracti32x4 $2, %zmm1, %xmm3
 ; AVX512F-NEXT:    vextracti32x4 $3, %zmm1, %xmm4
@@ -326,7 +326,7 @@ define <64 x i8> @var_shuffle_v64i8(<64 x i8> %v, <64 x i8> %indices) nounwind {
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-64, %rsp
-; AVX512F-NEXT:    addq $-128, %rsp
+; AVX512F-NEXT:    subq $64, %rsp
 ; AVX512F-NEXT:    vextracti128 $1, %ymm1, %xmm2
 ; AVX512F-NEXT:    vextracti32x4 $2, %zmm1, %xmm3
 ; AVX512F-NEXT:    vextracti32x4 $3, %zmm1, %xmm4
@@ -551,7 +551,7 @@ define <64 x i8> @var_shuffle_v64i8(<64 x i8> %v, <64 x i8> %indices) nounwind {
 ; AVX512BW-NEXT:    pushq %rbp
 ; AVX512BW-NEXT:    movq %rsp, %rbp
 ; AVX512BW-NEXT:    andq $-64, %rsp
-; AVX512BW-NEXT:    addq $-128, %rsp
+; AVX512BW-NEXT:    subq $64, %rsp
 ; AVX512BW-NEXT:    vextracti128 $1, %ymm1, %xmm2
 ; AVX512BW-NEXT:    vextracti32x4 $2, %zmm1, %xmm3
 ; AVX512BW-NEXT:    vextracti32x4 $3, %zmm1, %xmm4
@@ -1064,7 +1064,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-64, %rsp
-; AVX512F-NEXT:    addq $-128, %rsp
+; AVX512F-NEXT:    subq $64, %rsp
 ; AVX512F-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX512F-NEXT:    vpbroadcastd %esi, %zmm2
 ; AVX512F-NEXT:    vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm1 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
@@ -1315,7 +1315,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
 ; AVX512BW-NEXT:    pushq %rbp
 ; AVX512BW-NEXT:    movq %rsp, %rbp
 ; AVX512BW-NEXT:    andq $-64, %rsp
-; AVX512BW-NEXT:    addq $-128, %rsp
+; AVX512BW-NEXT:    subq $64, %rsp
 ; AVX512BW-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX512BW-NEXT:    vpbroadcastd %esi, %zmm2
 ; AVX512BW-NEXT:    vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm1 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
@@ -1566,7 +1566,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
 ; AVX512VBMI-NEXT:    pushq %rbp
 ; AVX512VBMI-NEXT:    movq %rsp, %rbp
 ; AVX512VBMI-NEXT:    andq $-64, %rsp
-; AVX512VBMI-NEXT:    addq $-128, %rsp
+; AVX512VBMI-NEXT:    subq $64, %rsp
 ; AVX512VBMI-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX512VBMI-NEXT:    vpbroadcastd %esi, %zmm1
 ; AVX512VBMI-NEXT:    vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm1, %zmm2 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
diff --git a/llvm/test/CodeGen/X86/vec-strict-128.ll b/llvm/test/CodeGen/X86/vec-strict-128.ll
index 84c0ff61cf9b7d..7e898213c1e6fb 100644
--- a/llvm/test/CodeGen/X86/vec-strict-128.ll
+++ b/llvm/test/CodeGen/X86/vec-strict-128.ll
@@ -348,7 +348,7 @@ define <2 x double> @f14(<2 x double> %a, <2 x double> %b, <2 x double> %c) #0 {
 ; SSE-X86-NEXT:    movl %esp, %ebp
 ; SSE-X86-NEXT:    .cfi_def_cfa_register %ebp
 ; SSE-X86-NEXT:    andl $-16, %esp
-; SSE-X86-NEXT:    subl $112, %esp
+; SSE-X86-NEXT:    subl $96, %esp
 ; SSE-X86-NEXT:    movaps %xmm2, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
 ; SSE-X86-NEXT:    movaps %xmm1, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
 ; SSE-X86-NEXT:    movaps %xmm0, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
diff --git a/llvm/test/CodeGen/X86/vec-strict-fptoint-256.ll b/llvm/test/CodeGen/X86/vec-strict-fptoint-256.ll
index 179e8ad69672b1..8cbf576677d1af 100644
--- a/llvm/test/CodeGen/X86/vec-strict-fptoint-256.ll
+++ b/llvm/test/CodeGen/X86/vec-strict-fptoint-256.ll
@@ -375,7 +375,7 @@ define <4 x i64> @strict_vector_fptoui_v4f64_to_v4i64(<4 x double> %a) #0 {
 ; AVX512F-32-NEXT:    .cfi_def_cfa_register %ebp
 ; AVX512F-32-NEXT:    pushl %ebx
 ; AVX512F-32-NEXT:    andl $-8, %esp
-; AVX512F-32-NEXT:    subl $40, %esp
+; AVX512F-32-NEXT:    subl $32, %esp
 ; AVX512F-32-NEXT:    .cfi_offset %ebx, -12
 ; AVX512F-32-NEXT:    vextractf128 $1, %ymm0, %xmm2
 ; AVX512F-32-NEXT:    vshufpd {{.*#+}} xmm3 = xmm2[1,0]
@@ -468,7 +468,7 @@ define <4 x i64> @strict_vector_fptoui_v4f64_to_v4i64(<4 x double> %a) #0 {
 ; AVX512VL-32-NEXT:    .cfi_def_cfa_register %ebp
 ; AVX512VL-32-NEXT:    pushl %ebx
 ; AVX512VL-32-NEXT:    andl $-8, %esp
-; AVX512VL-32-NEXT:    subl $40, %esp
+; AVX512VL-32-NEXT:    subl $32, %esp
 ; AVX512VL-32-NEXT:    .cfi_offset %ebx, -12
 ; AVX512VL-32-NEXT:    vextractf128 $1, %ymm0, %xmm2
 ; AVX512VL-32-NEXT:    vshufpd {{.*#+}} xmm3 = xmm2[1,0]
@@ -906,7 +906,7 @@ define <4 x i64> @strict_vector_fptoui_v4f32_to_v4i64(<4 x float> %a) #0 {
 ; AVX512F-32-NEXT:    .cfi_def_cfa_register %ebp
 ; AVX512F-32-NEXT:    pushl %ebx
 ; AVX512F-32-NEXT:    andl $-8, %esp
-; AVX512F-32-NEXT:    subl $40, %esp
+; AVX512F-32-NEXT:    subl $32, %esp
 ; AVX512F-32-NEXT:    .cfi_offset %ebx, -12
 ; AVX512F-32-NEXT:    vshufps {{.*#+}} xmm2 = xmm0[3,3,3,3]
 ; AVX512F-32-NEXT:    vmovss {{.*#+}} xmm1 = [9.22337203E+18,0.0E+0,0.0E+0,0.0E+0]
@@ -999,7 +999,7 @@ define <4 x i64> @strict_vector_fptoui_v4f32_to_v4i64(<4 x float> %a) #0 {
 ; AVX512VL-32-NEXT:    .cfi_def_cfa_register %ebp
 ; AVX512VL-32-NEXT:    pushl %ebx
 ; AVX512VL-32-NEXT:    andl $-8, %esp
-; AVX512VL-32-NEXT:    subl $40, %esp
+; AVX512VL-32-NEXT:    subl $32, %esp
 ; AVX512VL-32-NEXT:    .cfi_offset %ebx, -12
 ; AVX512VL-32-NEXT:    vshufps {{.*#+}} xmm2 = xmm0[3,3,3,3]
 ; AVX512VL-32-NEXT:    vmovss {{.*#+}} xmm1 = [9.22337203E+18,0.0E+0,0.0E+0,0.0E+0]
diff --git a/llvm/test/CodeGen/X86/vec_ins_extract-1.ll b/llvm/test/CodeGen/X86/vec_ins_extract-1.ll
index cf70d5d7f1edfd..bdb24fac73047f 100644
--- a/llvm/test/CodeGen/X86/vec_ins_extract-1.ll
+++ b/llvm/test/CodeGen/X86/vec_ins_extract-1.ll
@@ -11,7 +11,7 @@ define i32 @t0(i32 inreg %t7, <4 x i32> inreg %t8) nounwind {
 ; X32-NEXT:    pushl %ebp
 ; X32-NEXT:    movl %esp, %ebp
 ; X32-NEXT:    andl $-16, %esp
-; X32-NEXT:    subl $32, %esp
+; X32-NEXT:    subl $16, %esp
 ; X32-NEXT:    andl $3, %eax
 ; X32-NEXT:    movaps %xmm0, (%esp)
 ; X32-NEXT:    movl $76, (%esp,%eax,4)
@@ -39,7 +39,7 @@ define i32 @t1(i32 inreg %t7, <4 x i32> inreg %t8) nounwind {
 ; X32-NEXT:    pushl %ebp
 ; X32-NEXT:    movl %esp, %ebp
 ; X32-NEXT:    andl $-16, %esp
-; X32-NEXT:    subl $32, %esp
+; X32-NEXT:    subl $16, %esp
 ; X32-NEXT:    andl $3, %eax
 ; X32-NEXT:    movl $76, %ecx
 ; X32-NEXT:    pinsrd $0, %ecx, %xmm0
@@ -69,7 +69,7 @@ define <4 x i32> @t2(i32 inreg %t7, <4 x i32> inreg %t8) nounwind {
 ; X32-NEXT:    pushl %ebp
 ; X32-NEXT:    movl %esp, %ebp
 ; X32-NEXT:    andl $-16, %esp
-; X32-NEXT:    subl $32, %esp
+; X32-NEXT:    subl $16, %esp
 ; X32-NEXT:    andl $3, %eax
 ; X32-NEXT:    movdqa %xmm0, (%esp)
 ; X32-NEXT:    pinsrd $0, (%esp,%eax,4), %xmm0
@@ -95,7 +95,7 @@ define <4 x i32> @t3(i32 inreg %t7, <4 x i32> inreg %t8) nounwind {
 ; X32-NEXT:    pushl %ebp
 ; X32-NEXT:    movl %esp, %ebp
 ; X32-NEXT:    andl $-16, %esp
-; X32-NEXT:    subl $32, %esp
+; X32-NEXT:    subl $16, %esp
 ; X32-NEXT:    andl $3, %eax
 ; X32-NEXT:    movaps %xmm0, (%esp)
 ; X32-NEXT:    movss %xmm0, (%esp,%eax,4)
diff --git a/llvm/test/CodeGen/X86/vec_insert-8.ll b/llvm/test/CodeGen/X86/vec_insert-8.ll
index aa3364b31d66f8..ae78705644b992 100644
--- a/llvm/test/CodeGen/X86/vec_insert-8.ll
+++ b/llvm/test/CodeGen/X86/vec_insert-8.ll
@@ -10,7 +10,7 @@ define <4 x i32> @var_insert(<4 x i32> %x, i32 %val, i32 %idx) nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $32, %esp
+; X86-NEXT:    subl $16, %esp
 ; X86-NEXT:    movl 12(%ebp), %eax
 ; X86-NEXT:    andl $3, %eax
 ; X86-NEXT:    movl 8(%ebp), %ecx
@@ -40,7 +40,7 @@ define i32 @var_extract(<4 x i32> %x, i32 %idx) nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $32, %esp
+; X86-NEXT:    subl $16, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    andl $3, %eax
 ; X86-NEXT:    movaps %xmm0, (%esp)
diff --git a/llvm/test/CodeGen/X86/vector-compress.ll b/llvm/test/CodeGen/X86/vector-compress.ll
index d5872b1a2a351c..e0216248717a6d 100644
--- a/llvm/test/CodeGen/X86/vector-compress.ll
+++ b/llvm/test/CodeGen/X86/vector-compress.ll
@@ -240,7 +240,7 @@ define <8 x i32> @test_compress_v8i32(<8 x i32> %vec, <8 x i1> %mask, <8 x i32>
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    pushq %rbx
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $64, %rsp
+; AVX2-NEXT:    subq $32, %rsp
 ; AVX2-NEXT:    vpmovzxwd {{.*#+}} ymm1 = xmm1[0],zero,xmm1[1],zero,xmm1[2],zero,xmm1[3],zero,xmm1[4],zero,xmm1[5],zero,xmm1[6],zero,xmm1[7],zero
 ; AVX2-NEXT:    vpslld $31, %ymm1, %ymm1
 ; AVX2-NEXT:    vmovaps %ymm2, (%rsp)
@@ -336,7 +336,7 @@ define <8 x float> @test_compress_v8f32(<8 x float> %vec, <8 x i1> %mask, <8 x f
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $64, %rsp
+; AVX2-NEXT:    subq $32, %rsp
 ; AVX2-NEXT:    vpmovzxwd {{.*#+}} ymm1 = xmm1[0],zero,xmm1[1],zero,xmm1[2],zero,xmm1[3],zero,xmm1[4],zero,xmm1[5],zero,xmm1[6],zero,xmm1[7],zero
 ; AVX2-NEXT:    vpslld $31, %ymm1, %ymm1
 ; AVX2-NEXT:    vmovaps %ymm2, (%rsp)
@@ -439,7 +439,7 @@ define <4 x i64> @test_compress_v4i64(<4 x i64> %vec, <4 x i1> %mask, <4 x i64>
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $64, %rsp
+; AVX2-NEXT:    subq $32, %rsp
 ; AVX2-NEXT:    vpslld $31, %xmm1, %xmm1
 ; AVX2-NEXT:    vpsrad $31, %xmm1, %xmm1
 ; AVX2-NEXT:    vpmovsxdq %xmm1, %ymm1
@@ -513,7 +513,7 @@ define <4 x double> @test_compress_v4f64(<4 x double> %vec, <4 x i1> %mask, <4 x
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $64, %rsp
+; AVX2-NEXT:    subq $32, %rsp
 ; AVX2-NEXT:    vpslld $31, %xmm1, %xmm1
 ; AVX2-NEXT:    vpsrad $31, %xmm1, %xmm1
 ; AVX2-NEXT:    vmovaps %ymm2, (%rsp)
@@ -742,7 +742,7 @@ define <16 x float> @test_compress_v16f32(<16 x float> %vec, <16 x i1> %mask, <1
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vmovaps %ymm4, {{[0-9]+}}(%rsp)
 ; AVX2-NEXT:    vmovaps %ymm3, (%rsp)
 ; AVX2-NEXT:    vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2, %xmm3
@@ -890,7 +890,7 @@ define <8 x i64> @test_compress_v8i64(<8 x i64> %vec, <8 x i1> %mask, <8 x i64>
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    pushq %rbx
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vmovaps %ymm4, {{[0-9]+}}(%rsp)
 ; AVX2-NEXT:    vmovaps %ymm3, (%rsp)
 ; AVX2-NEXT:    vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2, %xmm3
@@ -982,7 +982,7 @@ define <8 x double> @test_compress_v8f64(<8 x double> %vec, <8 x i1> %mask, <8 x
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vmovaps %ymm4, {{[0-9]+}}(%rsp)
 ; AVX2-NEXT:    vmovaps %ymm3, (%rsp)
 ; AVX2-NEXT:    vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2, %xmm3
@@ -1349,7 +1349,7 @@ define <32 x i8> @test_compress_v32i8(<32 x i8> %vec, <32 x i1> %mask, <32 x i8>
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $64, %rsp
+; AVX2-NEXT:    subq $32, %rsp
 ; AVX2-NEXT:    vpsllw $7, %ymm1, %ymm1
 ; AVX2-NEXT:    vpxor %xmm3, %xmm3, %xmm3
 ; AVX2-NEXT:    vmovaps %ymm2, (%rsp)
@@ -1568,7 +1568,7 @@ define <32 x i8> @test_compress_v32i8(<32 x i8> %vec, <32 x i1> %mask, <32 x i8>
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-32, %rsp
-; AVX512F-NEXT:    subq $64, %rsp
+; AVX512F-NEXT:    subq $32, %rsp
 ; AVX512F-NEXT:    vpmovzxbd {{.*#+}} zmm4 = xmm1[0],zero,zero,zero,xmm1[1],zero,zero,zero,xmm1[2],zero,zero,zero,xmm1[3],zero,zero,zero,xmm1[4],zero,zero,zero,xmm1[5],zero,zero,zero,xmm1[6],zero,zero,zero,xmm1[7],zero,zero,zero,xmm1[8],zero,zero,zero,xmm1[9],zero,zero,zero,xmm1[10],zero,zero,zero,xmm1[11],zero,zero,zero,xmm1[12],zero,zero,zero,xmm1[13],zero,zero,zero,xmm1[14],zero,zero,zero,xmm1[15],zero,zero,zero
 ; AVX512F-NEXT:    vextracti128 $1, %ymm1, %xmm3
 ; AVX512F-NEXT:    vpmovsxbd %xmm3, %zmm5
@@ -1629,7 +1629,7 @@ define <32 x i8> @test_compress_v32i8(<32 x i8> %vec, <32 x i1> %mask, <32 x i8>
 ; AVX512VL-ONLY-NEXT:    pushq %rbp
 ; AVX512VL-ONLY-NEXT:    movq %rsp, %rbp
 ; AVX512VL-ONLY-NEXT:    andq $-32, %rsp
-; AVX512VL-ONLY-NEXT:    subq $64, %rsp
+; AVX512VL-ONLY-NEXT:    subq $32, %rsp
 ; AVX512VL-ONLY-NEXT:    vpsllw $7, %ymm1, %ymm1
 ; AVX512VL-ONLY-NEXT:    vpmovb2m %ymm1, %k6
 ; AVX512VL-ONLY-NEXT:    kshiftrd $31, %k6, %k0
@@ -2090,7 +2090,7 @@ define <64 x i8> @test_compress_v64i8(<64 x i8> %vec, <64 x i1> %mask, <64 x i8>
 ; AVX2-NEXT:    pushq %r12
 ; AVX2-NEXT:    pushq %rbx
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    movzbl 360(%rbp), %eax
 ; AVX2-NEXT:    movzbl 352(%rbp), %r10d
 ; AVX2-NEXT:    vmovd %r10d, %xmm4
@@ -2684,7 +2684,7 @@ define <64 x i8> @test_compress_v64i8(<64 x i8> %vec, <64 x i1> %mask, <64 x i8>
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-64, %rsp
-; AVX512F-NEXT:    subq $192, %rsp
+; AVX512F-NEXT:    addq $-128, %rsp
 ; AVX512F-NEXT:    vmovdqu64 352(%rbp), %zmm2
 ; AVX512F-NEXT:    vmovdqu64 416(%rbp), %zmm3
 ; AVX512F-NEXT:    vpmovqw %zmm2, %xmm2
@@ -2823,7 +2823,7 @@ define <64 x i8> @test_compress_v64i8(<64 x i8> %vec, <64 x i1> %mask, <64 x i8>
 ; AVX512VL-ONLY-NEXT:    pushq %rbp
 ; AVX512VL-ONLY-NEXT:    movq %rsp, %rbp
 ; AVX512VL-ONLY-NEXT:    andq $-64, %rsp
-; AVX512VL-ONLY-NEXT:    addq $-128, %rsp
+; AVX512VL-ONLY-NEXT:    subq $64, %rsp
 ; AVX512VL-ONLY-NEXT:    vpsllw $7, %zmm1, %zmm1
 ; AVX512VL-ONLY-NEXT:    vpmovb2m %zmm1, %k6
 ; AVX512VL-ONLY-NEXT:    kshiftrq $63, %k6, %k0
@@ -3620,7 +3620,7 @@ define <32 x i16> @test_compress_v32i16(<32 x i16> %vec, <32 x i1> %mask, <32 x
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-64, %rsp
-; AVX512F-NEXT:    addq $-128, %rsp
+; AVX512F-NEXT:    subq $64, %rsp
 ; AVX512F-NEXT:    vpmovzxbd {{.*#+}} zmm4 = xmm1[0],zero,zero,zero,xmm1[1],zero,zero,zero,xmm1[2],zero,zero,zero,xmm1[3],zero,zero,zero,xmm1[4],zero,zero,zero,xmm1[5],zero,zero,zero,xmm1[6],zero,zero,zero,xmm1[7],zero,zero,zero,xmm1[8],zero,zero,zero,xmm1[9],zero,zero,zero,xmm1[10],zero,zero,zero,xmm1[11],zero,zero,zero,xmm1[12],zero,zero,zero,xmm1[13],zero,zero,zero,xmm1[14],zero,zero,zero,xmm1[15],zero,zero,zero
 ; AVX512F-NEXT:    vextracti128 $1, %ymm1, %xmm3
 ; AVX512F-NEXT:    vpmovsxbd %xmm3, %zmm5
@@ -4015,7 +4015,7 @@ define <64 x i32> @test_compress_large(<64 x i1> %mask, <64 x i32> %vec, <64 x i
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $288, %rsp # imm = 0x120
+; AVX2-NEXT:    subq $256, %rsp # imm = 0x100
 ; AVX2-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX2-NEXT:    vmovss %xmm0, (%rsp)
 ; AVX2-NEXT:    andl $1, %esi
@@ -4474,7 +4474,7 @@ define <64 x i32> @test_compress_large(<64 x i1> %mask, <64 x i32> %vec, <64 x i
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-64, %rsp
-; AVX512F-NEXT:    subq $576, %rsp # imm = 0x240
+; AVX512F-NEXT:    subq $512, %rsp # imm = 0x200
 ; AVX512F-NEXT:    vmovdqu64 352(%rbp), %zmm4
 ; AVX512F-NEXT:    vmovdqu64 416(%rbp), %zmm5
 ; AVX512F-NEXT:    vpmovqw %zmm4, %xmm4
@@ -4581,7 +4581,7 @@ define <64 x i32> @test_compress_large(<64 x i1> %mask, <64 x i32> %vec, <64 x i
 ; AVX512VL-NEXT:    pushq %rbp
 ; AVX512VL-NEXT:    movq %rsp, %rbp
 ; AVX512VL-NEXT:    andq $-64, %rsp
-; AVX512VL-NEXT:    subq $576, %rsp # imm = 0x240
+; AVX512VL-NEXT:    subq $512, %rsp # imm = 0x200
 ; AVX512VL-NEXT:    vpsllw $7, %zmm0, %zmm0
 ; AVX512VL-NEXT:    vpmovb2m %zmm0, %k3
 ; AVX512VL-NEXT:    kshiftrq $48, %k3, %k1
@@ -4658,7 +4658,7 @@ define <8 x i64> @test_compress_knownbits_zext_v8i16_8i64(<8 x i16> %vec, <8 x
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    pushq %rbx
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vpshufd {{.*#+}} xmm4 = xmm0[2,3,2,3]
 ; AVX2-NEXT:    vpmovzxwq {{.*#+}} ymm4 = xmm4[0],zero,zero,zero,xmm4[1],zero,zero,zero,xmm4[2],zero,zero,zero,xmm4[3],zero,zero,zero
 ; AVX2-NEXT:    vbroadcastsd {{.*#+}} ymm5 = [3,3,3,3]
@@ -4762,7 +4762,7 @@ define <8 x i64> @test_compress_knownbits_sext_v8i16_8i64(<8 x i16> %vec, <8 x i
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    pushq %rbx
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vpshufd {{.*#+}} xmm4 = xmm0[2,3,2,3]
 ; AVX2-NEXT:    vpmovsxwq %xmm4, %ymm4
 ; AVX2-NEXT:    vbroadcastsd {{.*#+}} ymm5 = [3,3,3,3]
diff --git a/llvm/test/CodeGen/X86/vector-extend-inreg.ll b/llvm/test/CodeGen/X86/vector-extend-inreg.ll
index 9648eb5fc584a1..5cf7751d4ac89d 100644
--- a/llvm/test/CodeGen/X86/vector-extend-inreg.ll
+++ b/llvm/test/CodeGen/X86/vector-extend-inreg.ll
@@ -10,7 +10,7 @@ define i64 @extract_any_extend_vector_inreg_v16i64(<16 x i64> %a0, i32 %a1) noun
 ; X86-SSE-NEXT:    pushl %ebp
 ; X86-SSE-NEXT:    movl %esp, %ebp
 ; X86-SSE-NEXT:    andl $-16, %esp
-; X86-SSE-NEXT:    subl $272, %esp # imm = 0x110
+; X86-SSE-NEXT:    subl $256, %esp # imm = 0x100
 ; X86-SSE-NEXT:    movl 88(%ebp), %eax
 ; X86-SSE-NEXT:    movdqa 72(%ebp), %xmm0
 ; X86-SSE-NEXT:    psrldq {{.*#+}} xmm0 = xmm0[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero
@@ -64,7 +64,7 @@ define i64 @extract_any_extend_vector_inreg_v16i64(<16 x i64> %a0, i32 %a1) noun
 ; X86-AVX-NEXT:    pushl %ebp
 ; X86-AVX-NEXT:    movl %esp, %ebp
 ; X86-AVX-NEXT:    andl $-32, %esp
-; X86-AVX-NEXT:    subl $288, %esp # imm = 0x120
+; X86-AVX-NEXT:    subl $256, %esp # imm = 0x100
 ; X86-AVX-NEXT:    movl 40(%ebp), %eax
 ; X86-AVX-NEXT:    vmovsd {{.*#+}} xmm0 = mem[0],zero
 ; X86-AVX-NEXT:    vxorps %xmm1, %xmm1, %xmm1
@@ -91,7 +91,7 @@ define i64 @extract_any_extend_vector_inreg_v16i64(<16 x i64> %a0, i32 %a1) noun
 ; X64-AVX-NEXT:    pushq %rbp
 ; X64-AVX-NEXT:    movq %rsp, %rbp
 ; X64-AVX-NEXT:    andq $-32, %rsp
-; X64-AVX-NEXT:    subq $160, %rsp
+; X64-AVX-NEXT:    addq $-128, %rsp
 ; X64-AVX-NEXT:    # kill: def $edi killed $edi def $rdi
 ; X64-AVX-NEXT:    vpermq {{.*#+}} ymm0 = ymm3[3,3,3,3]
 ; X64-AVX-NEXT:    vmovq {{.*#+}} xmm0 = xmm0[0],zero
diff --git a/llvm/test/CodeGen/X86/vector-extract-last-active.ll b/llvm/test/CodeGen/X86/vector-extract-last-active.ll
index 16a2de9783893c..4a2857d57f3772 100644
--- a/llvm/test/CodeGen/X86/vector-extract-last-active.ll
+++ b/llvm/test/CodeGen/X86/vector-extract-last-active.ll
@@ -435,7 +435,7 @@ define i32 @extract_last_active_v8i32(<8 x i32> %a, <8 x i1> %c) nounwind {
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $64, %rsp
+; AVX2-NEXT:    subq $32, %rsp
 ; AVX2-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX2-NEXT:    vpsllw $15, %xmm1, %xmm0
 ; AVX2-NEXT:    vpacksswb %xmm0, %xmm0, %xmm1
@@ -462,7 +462,7 @@ define i32 @extract_last_active_v8i32(<8 x i32> %a, <8 x i1> %c) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-32, %rsp
-; AVX512-NEXT:    subq $64, %rsp
+; AVX512-NEXT:    subq $32, %rsp
 ; AVX512-NEXT:    vpsllw $15, %xmm1, %xmm1
 ; AVX512-NEXT:    vpmovw2m %xmm1, %k1
 ; AVX512-NEXT:    vmovdqa %ymm0, (%rsp)
@@ -552,7 +552,7 @@ define i32 @extract_last_active_v16i32(<16 x i32> %a, <16 x i1> %c) nounwind {
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $96, %rsp
+; AVX2-NEXT:    subq $64, %rsp
 ; AVX2-NEXT:    vpxor %xmm3, %xmm3, %xmm3
 ; AVX2-NEXT:    vpsllw $7, %xmm2, %xmm2
 ; AVX2-NEXT:    vpcmpgtb %xmm2, %xmm3, %xmm3
@@ -583,7 +583,7 @@ define i32 @extract_last_active_v16i32(<16 x i32> %a, <16 x i1> %c) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    addq $-128, %rsp
+; AVX512-NEXT:    subq $64, %rsp
 ; AVX512-NEXT:    vpsllw $7, %xmm1, %xmm1
 ; AVX512-NEXT:    vpmovb2m %xmm1, %k1
 ; AVX512-NEXT:    vmovdqa64 %zmm0, (%rsp)
@@ -723,7 +723,7 @@ define i8 @extract_last_active_split(<32 x i8> %data, <32 x i8> %mask, i8 %passt
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $64, %rsp
+; AVX2-NEXT:    subq $32, %rsp
 ; AVX2-NEXT:    vpxor %xmm2, %xmm2, %xmm2
 ; AVX2-NEXT:    vpcmpeqb %ymm2, %ymm1, %ymm2
 ; AVX2-NEXT:    vmovaps %ymm0, (%rsp)
@@ -753,7 +753,7 @@ define i8 @extract_last_active_split(<32 x i8> %data, <32 x i8> %mask, i8 %passt
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-32, %rsp
-; AVX512-NEXT:    subq $64, %rsp
+; AVX512-NEXT:    subq $32, %rsp
 ; AVX512-NEXT:    vptestmb %ymm1, %ymm1, %k1
 ; AVX512-NEXT:    vmovaps %ymm0, (%rsp)
 ; AVX512-NEXT:    vmovdqu8 {{.*#+}} ymm0 {%k1} {z} = [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31]
diff --git a/llvm/test/CodeGen/X86/vector-llrint.ll b/llvm/test/CodeGen/X86/vector-llrint.ll
index 50b631c54d8751..352009e8d8d606 100644
--- a/llvm/test/CodeGen/X86/vector-llrint.ll
+++ b/llvm/test/CodeGen/X86/vector-llrint.ll
@@ -141,7 +141,7 @@ define <4 x i64> @llrint_v4i64_v4f32(<4 x float> %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-8, %esp
-; X86-NEXT:    subl $56, %esp
+; X86-NEXT:    subl $48, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    flds 24(%ebp)
 ; X86-NEXT:    flds 20(%ebp)
@@ -301,7 +301,7 @@ define <8 x i64> @llrint_v8i64_v8f32(<8 x float> %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-8, %esp
-; X86-NEXT:    subl $120, %esp
+; X86-NEXT:    subl $112, %esp
 ; X86-NEXT:    flds 12(%ebp)
 ; X86-NEXT:    fistpll {{[0-9]+}}(%esp)
 ; X86-NEXT:    flds 16(%ebp)
@@ -586,7 +586,7 @@ define <16 x i64> @llrint_v16i64_v16f32(<16 x float> %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-8, %esp
-; X86-NEXT:    subl $248, %esp
+; X86-NEXT:    subl $240, %esp
 ; X86-NEXT:    flds 12(%ebp)
 ; X86-NEXT:    fistpll {{[0-9]+}}(%esp)
 ; X86-NEXT:    flds 16(%ebp)
@@ -1254,7 +1254,7 @@ define <4 x i64> @llrint_v4i64_v4f64(<4 x double> %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-8, %esp
-; X86-NEXT:    subl $56, %esp
+; X86-NEXT:    subl $48, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    fldl 36(%ebp)
 ; X86-NEXT:    fldl 28(%ebp)
@@ -1417,7 +1417,7 @@ define <8 x i64> @llrint_v8i64_v8f64(<8 x double> %x) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-8, %esp
-; X86-NEXT:    subl $120, %esp
+; X86-NEXT:    subl $112, %esp
 ; X86-NEXT:    fldl 12(%ebp)
 ; X86-NEXT:    fistpll {{[0-9]+}}(%esp)
 ; X86-NEXT:    fldl 20(%ebp)
diff --git a/llvm/test/CodeGen/X86/vector-lrint.ll b/llvm/test/CodeGen/X86/vector-lrint.ll
index d7261cf772c522..55a2e083414de3 100644
--- a/llvm/test/CodeGen/X86/vector-lrint.ll
+++ b/llvm/test/CodeGen/X86/vector-lrint.ll
@@ -276,7 +276,7 @@ define <4 x iXLen> @lrint_v4f32(<4 x float> %x) nounwind {
 ; X86-I64-NEXT:    pushl %edi
 ; X86-I64-NEXT:    pushl %esi
 ; X86-I64-NEXT:    andl $-8, %esp
-; X86-I64-NEXT:    subl $56, %esp
+; X86-I64-NEXT:    subl $48, %esp
 ; X86-I64-NEXT:    movl 8(%ebp), %eax
 ; X86-I64-NEXT:    flds 24(%ebp)
 ; X86-I64-NEXT:    flds 20(%ebp)
@@ -544,7 +544,7 @@ define <8 x iXLen> @lrint_v8f32(<8 x float> %x) nounwind {
 ; X86-I64-NEXT:    pushl %edi
 ; X86-I64-NEXT:    pushl %esi
 ; X86-I64-NEXT:    andl $-8, %esp
-; X86-I64-NEXT:    subl $120, %esp
+; X86-I64-NEXT:    subl $112, %esp
 ; X86-I64-NEXT:    flds 12(%ebp)
 ; X86-I64-NEXT:    fistpll {{[0-9]+}}(%esp)
 ; X86-I64-NEXT:    flds 16(%ebp)
@@ -1081,7 +1081,7 @@ define <4 x iXLen> @lrint_v4f64(<4 x double> %x) nounwind {
 ; X86-I64-NEXT:    pushl %edi
 ; X86-I64-NEXT:    pushl %esi
 ; X86-I64-NEXT:    andl $-8, %esp
-; X86-I64-NEXT:    subl $56, %esp
+; X86-I64-NEXT:    subl $48, %esp
 ; X86-I64-NEXT:    movl 8(%ebp), %eax
 ; X86-I64-NEXT:    fldl 36(%ebp)
 ; X86-I64-NEXT:    fldl 28(%ebp)
@@ -1366,7 +1366,7 @@ define <8 x iXLen> @lrint_v8f64(<8 x double> %x) nounwind {
 ; X86-I64-NEXT:    pushl %edi
 ; X86-I64-NEXT:    pushl %esi
 ; X86-I64-NEXT:    andl $-8, %esp
-; X86-I64-NEXT:    subl $120, %esp
+; X86-I64-NEXT:    subl $112, %esp
 ; X86-I64-NEXT:    fldl 12(%ebp)
 ; X86-I64-NEXT:    fistpll {{[0-9]+}}(%esp)
 ; X86-I64-NEXT:    fldl 20(%ebp)
diff --git a/llvm/test/CodeGen/X86/vector-reduce-ctpop.ll b/llvm/test/CodeGen/X86/vector-reduce-ctpop.ll
index ffa915336d92fa..7876fcccd43692 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-ctpop.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-ctpop.ll
@@ -2443,7 +2443,7 @@ define i64 @reduce_ctpop_v16i64(<16 x i64> %a0) nounwind {
 ; X86-SSE4-NEXT:    pushl %ebp
 ; X86-SSE4-NEXT:    movl %esp, %ebp
 ; X86-SSE4-NEXT:    andl $-16, %esp
-; X86-SSE4-NEXT:    subl $32, %esp
+; X86-SSE4-NEXT:    subl $16, %esp
 ; X86-SSE4-NEXT:    movaps %xmm2, (%esp) # 16-byte Spill
 ; X86-SSE4-NEXT:    movdqa %xmm0, %xmm2
 ; X86-SSE4-NEXT:    movdqa 40(%ebp), %xmm5
@@ -2636,7 +2636,7 @@ define i64 @reduce_ctpop_v16i64(<16 x i64> %a0) nounwind {
 ; X86-AVX1-NEXT:    pushl %ebp
 ; X86-AVX1-NEXT:    movl %esp, %ebp
 ; X86-AVX1-NEXT:    andl $-32, %esp
-; X86-AVX1-NEXT:    subl $96, %esp
+; X86-AVX1-NEXT:    subl $64, %esp
 ; X86-AVX1-NEXT:    vmovaps %ymm1, {{[-0-9]+}}(%e{{[sb]}}p) # 32-byte Spill
 ; X86-AVX1-NEXT:    vextractf128 $1, %ymm2, %xmm5
 ; X86-AVX1-NEXT:    vbroadcastss {{.*#+}} xmm3 = [15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15]
@@ -3743,7 +3743,7 @@ define <8 x i32> @reduce_ctpop_v4i64_buildvector_v8i32(<4 x i64> %a0, <4 x i64>
 ; X86-SSE2-NEXT:    pushl %ebp
 ; X86-SSE2-NEXT:    movl %esp, %ebp
 ; X86-SSE2-NEXT:    andl $-16, %esp
-; X86-SSE2-NEXT:    subl $80, %esp
+; X86-SSE2-NEXT:    subl $64, %esp
 ; X86-SSE2-NEXT:    movdqa 24(%ebp), %xmm6
 ; X86-SSE2-NEXT:    movdqa 8(%ebp), %xmm5
 ; X86-SSE2-NEXT:    movdqa %xmm1, %xmm3
@@ -4295,7 +4295,7 @@ define <8 x i32> @reduce_ctpop_v4i64_buildvector_v8i32(<4 x i64> %a0, <4 x i64>
 ; X86-SSE4-NEXT:    pushl %ebp
 ; X86-SSE4-NEXT:    movl %esp, %ebp
 ; X86-SSE4-NEXT:    andl $-16, %esp
-; X86-SSE4-NEXT:    subl $80, %esp
+; X86-SSE4-NEXT:    subl $64, %esp
 ; X86-SSE4-NEXT:    movdqa 24(%ebp), %xmm6
 ; X86-SSE4-NEXT:    movdqa {{.*#+}} xmm4 = [15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15]
 ; X86-SSE4-NEXT:    movdqa %xmm1, %xmm5
@@ -5037,7 +5037,7 @@ define <8 x i32> @reduce_ctpop_v4i64_buildvector_v8i32(<4 x i64> %a0, <4 x i64>
 ; X86-AVX2-NEXT:    pushl %ebp
 ; X86-AVX2-NEXT:    movl %esp, %ebp
 ; X86-AVX2-NEXT:    andl $-32, %esp
-; X86-AVX2-NEXT:    subl $192, %esp
+; X86-AVX2-NEXT:    subl $160, %esp
 ; X86-AVX2-NEXT:    vmovdqa 40(%ebp), %ymm6
 ; X86-AVX2-NEXT:    vmovdqa 8(%ebp), %ymm5
 ; X86-AVX2-NEXT:    vpbroadcastb {{.*#+}} ymm3 = [15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15,15]
diff --git a/llvm/test/CodeGen/X86/vector-reduce-smax.ll b/llvm/test/CodeGen/X86/vector-reduce-smax.ll
index b114dcc46696d8..8861735443338d 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-smax.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-smax.ll
@@ -728,7 +728,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
 ; X86-SSE2-NEXT:    pushl %ebp
 ; X86-SSE2-NEXT:    movl %esp, %ebp
 ; X86-SSE2-NEXT:    andl $-16, %esp
-; X86-SSE2-NEXT:    subl $32, %esp
+; X86-SSE2-NEXT:    subl $16, %esp
 ; X86-SSE2-NEXT:    movdqa %xmm2, %xmm7
 ; X86-SSE2-NEXT:    movdqa %xmm1, %xmm2
 ; X86-SSE2-NEXT:    movaps %xmm0, (%esp) # 16-byte Spill
@@ -988,7 +988,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
 ; X86-SSE41-NEXT:    pushl %ebp
 ; X86-SSE41-NEXT:    movl %esp, %ebp
 ; X86-SSE41-NEXT:    andl $-16, %esp
-; X86-SSE41-NEXT:    subl $32, %esp
+; X86-SSE41-NEXT:    subl $16, %esp
 ; X86-SSE41-NEXT:    movdqa %xmm2, %xmm3
 ; X86-SSE41-NEXT:    movdqa %xmm1, %xmm2
 ; X86-SSE41-NEXT:    movaps %xmm0, (%esp) # 16-byte Spill
diff --git a/llvm/test/CodeGen/X86/vector-reduce-smin.ll b/llvm/test/CodeGen/X86/vector-reduce-smin.ll
index fa440378975e33..00b7d8d6c562ec 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-smin.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-smin.ll
@@ -983,7 +983,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
 ; X86-SSE41-NEXT:    pushl %ebp
 ; X86-SSE41-NEXT:    movl %esp, %ebp
 ; X86-SSE41-NEXT:    andl $-16, %esp
-; X86-SSE41-NEXT:    subl $48, %esp
+; X86-SSE41-NEXT:    subl $32, %esp
 ; X86-SSE41-NEXT:    movaps %xmm1, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
 ; X86-SSE41-NEXT:    movdqa %xmm0, %xmm3
 ; X86-SSE41-NEXT:    movdqa 24(%ebp), %xmm6
diff --git a/llvm/test/CodeGen/X86/vector-reduce-umax.ll b/llvm/test/CodeGen/X86/vector-reduce-umax.ll
index f9a6a685e449ff..dade241b6e81e7 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-umax.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-umax.ll
@@ -835,7 +835,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
 ; X86-SSE2-NEXT:    pushl %ebp
 ; X86-SSE2-NEXT:    movl %esp, %ebp
 ; X86-SSE2-NEXT:    andl $-16, %esp
-; X86-SSE2-NEXT:    subl $32, %esp
+; X86-SSE2-NEXT:    subl $16, %esp
 ; X86-SSE2-NEXT:    movdqa %xmm2, %xmm7
 ; X86-SSE2-NEXT:    movdqa %xmm1, %xmm2
 ; X86-SSE2-NEXT:    movaps %xmm0, (%esp) # 16-byte Spill
@@ -1095,7 +1095,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
 ; X86-SSE41-NEXT:    pushl %ebp
 ; X86-SSE41-NEXT:    movl %esp, %ebp
 ; X86-SSE41-NEXT:    andl $-16, %esp
-; X86-SSE41-NEXT:    subl $32, %esp
+; X86-SSE41-NEXT:    subl $16, %esp
 ; X86-SSE41-NEXT:    movdqa %xmm2, %xmm3
 ; X86-SSE41-NEXT:    movdqa %xmm1, %xmm2
 ; X86-SSE41-NEXT:    movaps %xmm0, (%esp) # 16-byte Spill
@@ -1440,7 +1440,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
 ; X86-AVX1-NEXT:    pushl %ebp
 ; X86-AVX1-NEXT:    movl %esp, %ebp
 ; X86-AVX1-NEXT:    andl $-32, %esp
-; X86-AVX1-NEXT:    subl $96, %esp
+; X86-AVX1-NEXT:    subl $64, %esp
 ; X86-AVX1-NEXT:    vmovaps %ymm2, {{[-0-9]+}}(%e{{[sb]}}p) # 32-byte Spill
 ; X86-AVX1-NEXT:    vmovaps %ymm0, (%esp) # 32-byte Spill
 ; X86-AVX1-NEXT:    vmovddup {{.*#+}} xmm3 = [0,2147483648,0,2147483648]
diff --git a/llvm/test/CodeGen/X86/vector-reduce-umin.ll b/llvm/test/CodeGen/X86/vector-reduce-umin.ll
index 12bfb53cb715f1..6c5c31ded85003 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-umin.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-umin.ll
@@ -1091,7 +1091,7 @@ define i64 @test_v16i64(<16 x i64> %a0) nounwind {
 ; X86-SSE41-NEXT:    pushl %ebp
 ; X86-SSE41-NEXT:    movl %esp, %ebp
 ; X86-SSE41-NEXT:    andl $-16, %esp
-; X86-SSE41-NEXT:    subl $48, %esp
+; X86-SSE41-NEXT:    subl $32, %esp
 ; X86-SSE41-NEXT:    movaps %xmm1, {{[-0-9]+}}(%e{{[sb]}}p) # 16-byte Spill
 ; X86-SSE41-NEXT:    movdqa %xmm0, %xmm3
 ; X86-SSE41-NEXT:    movdqa 24(%ebp), %xmm6
diff --git a/llvm/test/CodeGen/X86/vector-shuffle-512-v16.ll b/llvm/test/CodeGen/X86/vector-shuffle-512-v16.ll
index 79643cdeebabcf..2fb038760711a9 100644
--- a/llvm/test/CodeGen/X86/vector-shuffle-512-v16.ll
+++ b/llvm/test/CodeGen/X86/vector-shuffle-512-v16.ll
@@ -968,7 +968,7 @@ define void @ispc_1864(ptr %arg) {
 ; ALL-NEXT:    movq %rsp, %rbp
 ; ALL-NEXT:    .cfi_def_cfa_register %rbp
 ; ALL-NEXT:    andq $-64, %rsp
-; ALL-NEXT:    subq $4864, %rsp # imm = 0x1300
+; ALL-NEXT:    subq $4800, %rsp # imm = 0x12C0
 ; ALL-NEXT:    vbroadcastss {{.*#+}} ymm0 = [-5.0E+0,-5.0E+0,-5.0E+0,-5.0E+0,-5.0E+0,-5.0E+0,-5.0E+0,-5.0E+0]
 ; ALL-NEXT:    vmulps 32(%rdi), %ymm0, %ymm0
 ; ALL-NEXT:    vcvtps2pd %ymm0, %zmm0
diff --git a/llvm/test/CodeGen/X86/vector-shuffle-variable-256.ll b/llvm/test/CodeGen/X86/vector-shuffle-variable-256.ll
index 8f78438dedf92d..d647973d309fdf 100644
--- a/llvm/test/CodeGen/X86/vector-shuffle-variable-256.ll
+++ b/llvm/test/CodeGen/X86/vector-shuffle-variable-256.ll
@@ -12,7 +12,7 @@ define <4 x double> @var_shuffle_v4f64_v4f64_xxxx_i64(<4 x double> %x, i64 %i0,
 ; ALL-NEXT:    pushq %rbp
 ; ALL-NEXT:    movq %rsp, %rbp
 ; ALL-NEXT:    andq $-32, %rsp
-; ALL-NEXT:    subq $64, %rsp
+; ALL-NEXT:    subq $32, %rsp
 ; ALL-NEXT:    andl $3, %esi
 ; ALL-NEXT:    andl $3, %edi
 ; ALL-NEXT:    andl $3, %ecx
@@ -43,7 +43,7 @@ define <4 x double> @var_shuffle_v4f64_v4f64_uxx0_i64(<4 x double> %x, i64 %i0,
 ; ALL-NEXT:    pushq %rbp
 ; ALL-NEXT:    movq %rsp, %rbp
 ; ALL-NEXT:    andq $-32, %rsp
-; ALL-NEXT:    subq $64, %rsp
+; ALL-NEXT:    subq $32, %rsp
 ; ALL-NEXT:    andl $3, %edx
 ; ALL-NEXT:    andl $3, %esi
 ; ALL-NEXT:    vmovaps %ymm0, (%rsp)
@@ -95,7 +95,7 @@ define <4 x i64> @var_shuffle_v4i64_v4i64_xxxx_i64(<4 x i64> %x, i64 %i0, i64 %i
 ; ALL-NEXT:    pushq %rbp
 ; ALL-NEXT:    movq %rsp, %rbp
 ; ALL-NEXT:    andq $-32, %rsp
-; ALL-NEXT:    subq $64, %rsp
+; ALL-NEXT:    subq $32, %rsp
 ; ALL-NEXT:    andl $3, %edi
 ; ALL-NEXT:    andl $3, %esi
 ; ALL-NEXT:    andl $3, %edx
@@ -128,7 +128,7 @@ define <4 x i64> @var_shuffle_v4i64_v4i64_xx00_i64(<4 x i64> %x, i64 %i0, i64 %i
 ; ALL-NEXT:    pushq %rbp
 ; ALL-NEXT:    movq %rsp, %rbp
 ; ALL-NEXT:    andq $-32, %rsp
-; ALL-NEXT:    subq $64, %rsp
+; ALL-NEXT:    subq $32, %rsp
 ; ALL-NEXT:    andl $3, %edi
 ; ALL-NEXT:    andl $3, %esi
 ; ALL-NEXT:    vmovaps %ymm0, (%rsp)
@@ -182,7 +182,7 @@ define <8 x float> @var_shuffle_v8f32_v8f32_xxxxxxxx_i32(<8 x float> %x, i32 %i0
 ; ALL-NEXT:    pushq %rbp
 ; ALL-NEXT:    movq %rsp, %rbp
 ; ALL-NEXT:    andq $-32, %rsp
-; ALL-NEXT:    subq $64, %rsp
+; ALL-NEXT:    subq $32, %rsp
 ; ALL-NEXT:    # kill: def $r9d killed $r9d def $r9
 ; ALL-NEXT:    # kill: def $r8d killed $r8d def $r8
 ; ALL-NEXT:    # kill: def $ecx killed $ecx def $rcx
@@ -286,7 +286,7 @@ define <16 x i16> @var_shuffle_v16i16_v16i16_xxxxxxxxxxxxxxxx_i16(<16 x i16> %x,
 ; AVX1-NEXT:    pushq %rbp
 ; AVX1-NEXT:    movq %rsp, %rbp
 ; AVX1-NEXT:    andq $-32, %rsp
-; AVX1-NEXT:    subq $64, %rsp
+; AVX1-NEXT:    subq $32, %rsp
 ; AVX1-NEXT:    # kill: def $r9d killed $r9d def $r9
 ; AVX1-NEXT:    # kill: def $r8d killed $r8d def $r8
 ; AVX1-NEXT:    # kill: def $ecx killed $ecx def $rcx
@@ -348,7 +348,7 @@ define <16 x i16> @var_shuffle_v16i16_v16i16_xxxxxxxxxxxxxxxx_i16(<16 x i16> %x,
 ; AVX2-NEXT:    pushq %rbp
 ; AVX2-NEXT:    movq %rsp, %rbp
 ; AVX2-NEXT:    andq $-32, %rsp
-; AVX2-NEXT:    subq $64, %rsp
+; AVX2-NEXT:    subq $32, %rsp
 ; AVX2-NEXT:    # kill: def $r9d killed $r9d def $r9
 ; AVX2-NEXT:    # kill: def $r8d killed $r8d def $r8
 ; AVX2-NEXT:    # kill: def $ecx killed $ecx def $rcx
@@ -596,7 +596,7 @@ define <4 x i64> @mem_shuffle_v4i64_v4i64_xxxx_i64(<4 x i64> %x, ptr %i) nounwin
 ; ALL-NEXT:    pushq %rbp
 ; ALL-NEXT:    movq %rsp, %rbp
 ; ALL-NEXT:    andq $-32, %rsp
-; ALL-NEXT:    subq $64, %rsp
+; ALL-NEXT:    subq $32, %rsp
 ; ALL-NEXT:    movl (%rdi), %eax
 ; ALL-NEXT:    movl 8(%rdi), %ecx
 ; ALL-NEXT:    andl $3, %eax
diff --git a/llvm/test/CodeGen/X86/widen_arith-6.ll b/llvm/test/CodeGen/X86/widen_arith-6.ll
index 6fa232f4d3227d..7f84fd2f546d50 100644
--- a/llvm/test/CodeGen/X86/widen_arith-6.ll
+++ b/llvm/test/CodeGen/X86/widen_arith-6.ll
@@ -9,7 +9,7 @@ define void @update(ptr %dst, ptr %src, i32 %n) nounwind {
 ; CHECK-NEXT:    pushl %ebp
 ; CHECK-NEXT:    movl %esp, %ebp
 ; CHECK-NEXT:    andl $-16, %esp
-; CHECK-NEXT:    subl $48, %esp
+; CHECK-NEXT:    subl $32, %esp
 ; CHECK-NEXT:    movl $1077936128, {{[0-9]+}}(%esp) # imm = 0x40400000
 ; CHECK-NEXT:    movl $1073741824, {{[0-9]+}}(%esp) # imm = 0x40000000
 ; CHECK-NEXT:    movl $1065353216, {{[0-9]+}}(%esp) # imm = 0x3F800000
diff --git a/llvm/test/CodeGen/X86/x86-64-baseptr.ll b/llvm/test/CodeGen/X86/x86-64-baseptr.ll
index 4f366ba3144a89..1dabef12521e3d 100644
--- a/llvm/test/CodeGen/X86/x86-64-baseptr.ll
+++ b/llvm/test/CodeGen/X86/x86-64-baseptr.ll
@@ -359,7 +359,7 @@ define void @vmw_host_printf(ptr %fmt, ...) nounwind {
 ; X32ABI-NEXT:    movl %esp, %ebp
 ; X32ABI-NEXT:    pushq %rbx
 ; X32ABI-NEXT:    andl $-16, %esp
-; X32ABI-NEXT:    subl $208, %esp
+; X32ABI-NEXT:    subl $192, %esp
 ; X32ABI-NEXT:    movl %esp, %ebx
 ; X32ABI-NEXT:    movq %rsi, 24(%ebx)
 ; X32ABI-NEXT:    movq %rdx, 32(%ebx)
diff --git a/llvm/test/DebugInfo/COFF/fpo-realign-alloca.ll b/llvm/test/DebugInfo/COFF/fpo-realign-alloca.ll
index d6f45b166d22fd..c9972b0b10306f 100644
--- a/llvm/test/DebugInfo/COFF/fpo-realign-alloca.ll
+++ b/llvm/test/DebugInfo/COFF/fpo-realign-alloca.ll
@@ -20,8 +20,8 @@
 ; CHECK:         .cv_fpo_pushreg %esi
 ; CHECK:         andl    $-16, %esp
 ; CHECK:         .cv_fpo_stackalign      16
-; CHECK:         subl    $32, %esp
-; CHECK:         .cv_fpo_stackalloc      32
+; CHECK:         subl    $16, %esp
+; CHECK:         .cv_fpo_stackalloc      16
 ; CHECK:         .cv_fpo_endprologue
 ; CHECK:         movl    %esp, %esi
 ; CHECK:         leal    8(%esi),
diff --git a/llvm/test/DebugInfo/COFF/vframe-csr.ll b/llvm/test/DebugInfo/COFF/vframe-csr.ll
index f46965afd7da55..3e8e941d1e2d94 100644
--- a/llvm/test/DebugInfo/COFF/vframe-csr.ll
+++ b/llvm/test/DebugInfo/COFF/vframe-csr.ll
@@ -16,9 +16,9 @@
 ; ASM:         .cv_fpo_setframe        %ebp
 ; ASM:         andl    $-8, %esp
 ; ASM:         .cv_fpo_stackalign      8
-; FIXME: Why 24 bytes? We only need 12 bytes of data.
-; ASM:         subl    $24, %esp
-; ASM:         .cv_fpo_stackalloc      24
+; 12 bytes of data, rounded up to the 8-byte realignment.
+; ASM:         subl    $16, %esp
+; ASM:         .cv_fpo_stackalloc      16
 ; ASM:         .cv_fpo_endprologue
 
 ; 'x' should be EBP-relative, 'a' and 'force_alignment' ESP relative.
@@ -70,7 +70,7 @@
 ; OBJ:     LocalFramePtrReg: VFRAME (0x7536)
 ; OBJ:     ParamFramePtrReg: EBP (0x16)
 ; OBJ:   }
-; 	ESP is VFRAME - 24, ESP offset of 'a' is 4, so -20.
+; 	ESP is VFRAME - 16, ESP offset of 'a' is 4, so -12.
 ; OBJ:   LocalSym {
 ; OBJ:     Kind: S_LOCAL (0x113E)
 ; OBJ:     Type: int (0x74)
@@ -80,7 +80,7 @@
 ; OBJ:   }
 ; OBJ:   DefRangeFramePointerRelSym {
 ; OBJ:     Kind: S_DEFRANGE_FRAMEPOINTER_REL (0x1142)
-; OBJ:     Offset: -20
+; OBJ:     Offset: -12
 ; OBJ:   }
 ; 	ESP is VFRAME - 16, ESP offset of 'force_alignment' is 8, so -8.
 ; OBJ:   LocalSym {
@@ -92,7 +92,7 @@
 ; OBJ:   }
 ; OBJ:   DefRangeFramePointerRelSym {
 ; OBJ:     Kind: S_DEFRANGE_FRAMEPOINTER_REL (0x1142)
-; OBJ:     Offset: -16
+; OBJ:     Offset: -8
 ; OBJ:   }
 ; OBJ:   ProcEnd {
 ; OBJ:     Kind: S_PROC_ID_END (0x114F)



More information about the llvm-commits mailing list