[llvm] [X86] Use 8-bit immediates for 128-byte stack adjustments and LEA splits (PR #219866)

Andrew Gaul via llvm-commits llvm-commits at lists.llvm.org
Sun Aug 30 21:21:13 PDT 2026


https://github.com/gaul updated https://github.com/llvm/llvm-project/pull/219866

>From 7c1581e6aef72d281e63e0943226e3c2d072e5d9 Mon Sep 17 00:00:00 2001
From: Andrew Gaul <andrew at gaul.org>
Date: Sun, 30 Aug 2026 21:19:23 -0700
Subject: [PATCH 1/2] [X86] Add test coverage for 128-byte stack adjustments
 and slow-LEA splits

Both currently use the 32-bit immediate encoding of ADD/SUB with 128
where the sign-extended 8-bit encoding of the inverse operation with
-128 would be three bytes shorter.

Co-Authored-By: Claude Fable 5 <noreply at anthropic.com>
---
 llvm/test/CodeGen/X86/lea-fixup-disp128.mir |  65 ++++++++++
 llvm/test/CodeGen/X86/stack-adjust-128.ll   | 136 ++++++++++++++++++++
 2 files changed, 201 insertions(+)
 create mode 100644 llvm/test/CodeGen/X86/lea-fixup-disp128.mir
 create mode 100644 llvm/test/CodeGen/X86/stack-adjust-128.ll

diff --git a/llvm/test/CodeGen/X86/lea-fixup-disp128.mir b/llvm/test/CodeGen/X86/lea-fixup-disp128.mir
new file mode 100644
index 0000000000000..8c9088882af74
--- /dev/null
+++ b/llvm/test/CodeGen/X86/lea-fixup-disp128.mir
@@ -0,0 +1,65 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 3
+# RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=slow-3ops-lea -run-pass x86-fixup-leas -o -  %s | FileCheck %s
+
+# When splitting a slow 3-operand LEA, a displacement of exactly 128 should
+# become SUB of -128 rather than ADD of 128: only -128 fits the sign-extended
+# 8-bit immediate, saving three bytes. Adjacent displacements keep the ADD.
+
+--- |
+  define void @disp128() nounwind {
+    ret void
+  }
+  define void @disp128_32() nounwind {
+    ret void
+  }
+  define void @disp129() nounwind {
+    ret void
+  }
+  define void @disp_minus128() nounwind {
+    ret void
+  }
+
+---
+name:            disp128
+body:             |
+  bb.0:
+    ; CHECK-LABEL: name: disp128
+    ; CHECK: renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 0, $noreg
+    ; CHECK-NEXT: $rax = ADD64ri32 $rax, 128, implicit-def $eflags
+    ; CHECK-NEXT: JMP64r renamable $rax
+    renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 128, $noreg
+    JMP64r renamable $rax
+...
+---
+name:            disp128_32
+body:             |
+  bb.0:
+    ; CHECK-LABEL: name: disp128_32
+    ; CHECK: renamable $eax = LEA64_32r renamable $rdi, 1, renamable $rsi, 0, $noreg
+    ; CHECK-NEXT: $eax = ADD32ri $eax, 128, implicit-def $eflags
+    ; CHECK-NEXT: RET64 implicit $eax
+    renamable $eax = LEA64_32r renamable $rdi, 1, renamable $rsi, 128, $noreg
+    RET64 implicit $eax
+...
+---
+name:            disp129
+body:             |
+  bb.0:
+    ; CHECK-LABEL: name: disp129
+    ; CHECK: renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 0, $noreg
+    ; CHECK-NEXT: $rax = ADD64ri32 $rax, 129, implicit-def $eflags
+    ; CHECK-NEXT: JMP64r renamable $rax
+    renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 129, $noreg
+    JMP64r renamable $rax
+...
+---
+name:            disp_minus128
+body:             |
+  bb.0:
+    ; CHECK-LABEL: name: disp_minus128
+    ; CHECK: renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 0, $noreg
+    ; CHECK-NEXT: $rax = ADD64ri32 $rax, -128, implicit-def $eflags
+    ; CHECK-NEXT: JMP64r renamable $rax
+    renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, -128, $noreg
+    JMP64r renamable $rax
+...
diff --git a/llvm/test/CodeGen/X86/stack-adjust-128.ll b/llvm/test/CodeGen/X86/stack-adjust-128.ll
new file mode 100644
index 0000000000000..a19631248e7d6
--- /dev/null
+++ b/llvm/test/CodeGen/X86/stack-adjust-128.ll
@@ -0,0 +1,136 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -verify-machineinstrs | FileCheck %s --check-prefix=LINUX
+; RUN: llc < %s -mtriple=x86_64-pc-windows-msvc -verify-machineinstrs | FileCheck %s --check-prefix=WIN64
+
+; A 128-byte stack adjustment should be emitted as `add rsp, -128` in the
+; prologue and `sub rsp, -128` in the epilogue: 128 misses the sign-extended
+; 8-bit immediate form, but -128 fits it, saving three bytes each. Windows
+; keeps the canonical SUB/ADD spelling because the Win64 unwinder recognizes
+; epilogues by disassembling for `add rsp, imm`.
+
+declare void @escape(ptr)
+
+; One value lives across the call, forcing a single CSR push; the 120 bytes
+; of locals then round up to a 128-byte adjustment on Linux.
+define ptr @frame128(ptr %p) {
+; LINUX-LABEL: frame128:
+; LINUX:       # %bb.0:
+; LINUX-NEXT:    pushq %rbx
+; LINUX-NEXT:    .cfi_def_cfa_offset 16
+; LINUX-NEXT:    subq $128, %rsp
+; LINUX-NEXT:    .cfi_def_cfa_offset 144
+; LINUX-NEXT:    .cfi_offset %rbx, -16
+; LINUX-NEXT:    movq %rdi, %rbx
+; LINUX-NEXT:    leaq {{[0-9]+}}(%rsp), %rdi
+; LINUX-NEXT:    callq escape at PLT
+; LINUX-NEXT:    movq %rbx, %rax
+; LINUX-NEXT:    addq $128, %rsp
+; LINUX-NEXT:    .cfi_def_cfa_offset 16
+; LINUX-NEXT:    popq %rbx
+; LINUX-NEXT:    .cfi_def_cfa_offset 8
+; LINUX-NEXT:    retq
+;
+; WIN64-LABEL: frame128:
+; WIN64:       # %bb.0:
+; WIN64-NEXT:    pushq %rsi
+; WIN64-NEXT:    .seh_pushreg %rsi
+; WIN64-NEXT:    subq $160, %rsp
+; WIN64-NEXT:    .seh_stackalloc 160
+; WIN64-NEXT:    .seh_endprologue
+; WIN64-NEXT:    movq %rcx, %rsi
+; WIN64-NEXT:    leaq {{[0-9]+}}(%rsp), %rcx
+; WIN64-NEXT:    callq escape
+; WIN64-NEXT:    movq %rsi, %rax
+; WIN64-NEXT:    .seh_startepilogue
+; WIN64-NEXT:    addq $160, %rsp
+; WIN64-NEXT:    popq %rsi
+; WIN64-NEXT:    .seh_endepilogue
+; WIN64-NEXT:    retq
+; WIN64-NEXT:    .seh_endproc
+  %buf = alloca [120 x i8]
+  call void @escape(ptr %buf)
+  ret ptr %p
+}
+
+; The same shape with 96 bytes of locals reaches a 128-byte adjustment on
+; Win64 (32-byte shadow area included), which must keep `sub rsp, 128`.
+define ptr @frame128_win(ptr %p) {
+; LINUX-LABEL: frame128_win:
+; LINUX:       # %bb.0:
+; LINUX-NEXT:    pushq %rbx
+; LINUX-NEXT:    .cfi_def_cfa_offset 16
+; LINUX-NEXT:    subq $96, %rsp
+; LINUX-NEXT:    .cfi_def_cfa_offset 112
+; LINUX-NEXT:    .cfi_offset %rbx, -16
+; LINUX-NEXT:    movq %rdi, %rbx
+; LINUX-NEXT:    movq %rsp, %rdi
+; LINUX-NEXT:    callq escape at PLT
+; LINUX-NEXT:    movq %rbx, %rax
+; LINUX-NEXT:    addq $96, %rsp
+; LINUX-NEXT:    .cfi_def_cfa_offset 16
+; LINUX-NEXT:    popq %rbx
+; LINUX-NEXT:    .cfi_def_cfa_offset 8
+; LINUX-NEXT:    retq
+;
+; WIN64-LABEL: frame128_win:
+; WIN64:       # %bb.0:
+; WIN64-NEXT:    pushq %rsi
+; WIN64-NEXT:    .seh_pushreg %rsi
+; WIN64-NEXT:    subq $128, %rsp
+; WIN64-NEXT:    .seh_stackalloc 128
+; WIN64-NEXT:    .seh_endprologue
+; WIN64-NEXT:    movq %rcx, %rsi
+; WIN64-NEXT:    leaq {{[0-9]+}}(%rsp), %rcx
+; WIN64-NEXT:    callq escape
+; WIN64-NEXT:    movq %rsi, %rax
+; WIN64-NEXT:    .seh_startepilogue
+; WIN64-NEXT:    addq $128, %rsp
+; WIN64-NEXT:    popq %rsi
+; WIN64-NEXT:    .seh_endepilogue
+; WIN64-NEXT:    retq
+; WIN64-NEXT:    .seh_endproc
+  %buf = alloca [96 x i8]
+  call void @escape(ptr %buf)
+  ret ptr %p
+}
+
+; An adjacent size keeps the plain 32-bit immediate.
+define ptr @frame144(ptr %p) {
+; LINUX-LABEL: frame144:
+; LINUX:       # %bb.0:
+; LINUX-NEXT:    pushq %rbx
+; LINUX-NEXT:    .cfi_def_cfa_offset 16
+; LINUX-NEXT:    subq $144, %rsp
+; LINUX-NEXT:    .cfi_def_cfa_offset 160
+; LINUX-NEXT:    .cfi_offset %rbx, -16
+; LINUX-NEXT:    movq %rdi, %rbx
+; LINUX-NEXT:    leaq {{[0-9]+}}(%rsp), %rdi
+; LINUX-NEXT:    callq escape at PLT
+; LINUX-NEXT:    movq %rbx, %rax
+; LINUX-NEXT:    addq $144, %rsp
+; LINUX-NEXT:    .cfi_def_cfa_offset 16
+; LINUX-NEXT:    popq %rbx
+; LINUX-NEXT:    .cfi_def_cfa_offset 8
+; LINUX-NEXT:    retq
+;
+; WIN64-LABEL: frame144:
+; WIN64:       # %bb.0:
+; WIN64-NEXT:    pushq %rsi
+; WIN64-NEXT:    .seh_pushreg %rsi
+; WIN64-NEXT:    subq $176, %rsp
+; WIN64-NEXT:    .seh_stackalloc 176
+; WIN64-NEXT:    .seh_endprologue
+; WIN64-NEXT:    movq %rcx, %rsi
+; WIN64-NEXT:    leaq {{[0-9]+}}(%rsp), %rcx
+; WIN64-NEXT:    callq escape
+; WIN64-NEXT:    movq %rsi, %rax
+; WIN64-NEXT:    .seh_startepilogue
+; WIN64-NEXT:    addq $176, %rsp
+; WIN64-NEXT:    popq %rsi
+; WIN64-NEXT:    .seh_endepilogue
+; WIN64-NEXT:    retq
+; WIN64-NEXT:    .seh_endproc
+  %buf = alloca [136 x i8]
+  call void @escape(ptr %buf)
+  ret ptr %p
+}

>From 02172e59485de9dc5a241a6efb83529568636c9b Mon Sep 17 00:00:00 2001
From: Andrew Gaul <andrew at gaul.org>
Date: Sun, 30 Aug 2026 21:19:23 -0700
Subject: [PATCH 2/2] [X86] Use 8-bit immediates for 128-byte stack adjustments
 and LEA splits

ADD and SUB of +128 need a 32-bit immediate, but the equivalent
operation with -128 fits the sign-extended 8-bit form, three bytes
shorter. ISel has long done this through X86InstrCompiler.td patterns
(extended to flag-producing adds in c0e01d29a467), but two post-ISel
paths still emit the long form:

* X86FrameLowering::BuildStackAdjustment emits sub rsp, 128 and
  add rsp, 128 for 128-byte adjustments. Flip them to add rsp, -128
  and sub rsp, -128. EFLAGS is dead whenever this branch is taken,
  since the code already prefers a flag-clobbering ADD/SUB over LEA on
  that basis. Windows CFI targets keep the canonical spelling because
  the Win64 unwinder recognizes an epilogue by disassembling forward
  for add rsp, imm.

* X86FixupLEAs::processInstrForSlow3OpLEA splits a slow three-operand
  LEA into a two-operand LEA plus an ADD of the displacement, guarded
  on EFLAGS being provably dead; emit SUB of -128 when that
  displacement is exactly 128.

Together these cover all 55 remaining oversized 128-immediate
encodings in uutils coreutils (48 stack adjustments, 7 LEA splits).

Found via x86lint.

Co-Authored-By: Claude Fable 5 <noreply at anthropic.com>
---
 llvm/lib/Target/X86/X86FixupLEAs.cpp          | 21 ++++++++++++++++
 llvm/lib/Target/X86/X86FrameLowering.cpp      | 20 +++++++++++++---
 llvm/test/CodeGen/X86/arg-copy-elide.ll       |  2 +-
 llvm/test/CodeGen/X86/avx2-vbroadcast.ll      | 16 ++++++-------
 llvm/test/CodeGen/X86/avx512-calling-conv.ll  |  4 ++--
 .../test/CodeGen/X86/avx512-insert-extract.ll | 22 ++++++++---------
 .../CodeGen/X86/avx512-insert-extract_i1.ll   |  2 +-
 llvm/test/CodeGen/X86/avx512-intel-ocl.ll     |  6 ++---
 .../avx512-shuffles/shuffle-chained-bf16.ll   |  2 +-
 llvm/test/CodeGen/X86/avx512fp16-mov.ll       |  4 ++--
 llvm/test/CodeGen/X86/frem.ll                 |  4 ++--
 llvm/test/CodeGen/X86/i386-baseptr.ll         |  2 +-
 llvm/test/CodeGen/X86/lea-fixup-disp128.mir   |  4 ++--
 llvm/test/CodeGen/X86/nontemporal-loads-2.ll  | 24 +++++++++----------
 llvm/test/CodeGen/X86/sdiv_fix_sat.ll         |  2 +-
 llvm/test/CodeGen/X86/shift-i128.ll           |  2 +-
 .../X86/smulo-128-legalisation-lowering.ll    |  4 ++--
 llvm/test/CodeGen/X86/stack-adjust-128.ll     |  4 ++--
 llvm/test/CodeGen/X86/var-permute-512.ll      | 12 +++++-----
 .../CodeGen/X86/vec-strict-inttofp-512.ll     |  4 ++--
 llvm/test/CodeGen/X86/vector-compress.ll      |  4 ++--
 .../CodeGen/X86/vector-extract-last-active.ll |  2 +-
 ...lar-shift-by-byte-multiple-legalization.ll |  8 +++----
 ...ad-of-small-alloca-with-zero-upper-half.ll |  4 ++--
 llvm/test/CodeGen/X86/x86-64-baseptr.ll       |  2 +-
 25 files changed, 108 insertions(+), 73 deletions(-)

diff --git a/llvm/lib/Target/X86/X86FixupLEAs.cpp b/llvm/lib/Target/X86/X86FixupLEAs.cpp
index aa5ecd94ed828..5c9c74e64aa73 100644
--- a/llvm/lib/Target/X86/X86FixupLEAs.cpp
+++ b/llvm/lib/Target/X86/X86FixupLEAs.cpp
@@ -385,6 +385,18 @@ static inline unsigned getADDriFromLEA(unsigned LEAOpcode,
   }
 }
 
+static inline unsigned getSUBriFromLEA(unsigned LEAOpcode) {
+  switch (LEAOpcode) {
+  default:
+    llvm_unreachable("Unexpected LEA instruction");
+  case X86::LEA32r:
+  case X86::LEA64_32r:
+    return X86::SUB32ri;
+  case X86::LEA64r:
+    return X86::SUB64ri32;
+  }
+}
+
 static inline unsigned getINCDECFromLEA(unsigned LEAOpcode, bool IsINC) {
   switch (LEAOpcode) {
   default:
@@ -851,6 +863,15 @@ void FixupLEAsImpl::processInstrForSlow3OpLEA(MachineBasicBlock::iterator &I,
         NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
                     .addReg(DestReg);
         LLVM_DEBUG(NewMI->dump(););
+      } else if (Offset.isImm() && Offset.getImm() == 128) {
+        // ADD of +128 needs a 32-bit immediate, while SUB of -128 fits the
+        // sign-extended 8-bit form, three bytes shorter. EFLAGS was proved
+        // dead above, so the different flag results don't matter.
+        unsigned NewOpc = getSUBriFromLEA(MI.getOpcode());
+        NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
+                    .addReg(DestReg)
+                    .addImm(-128);
+        LLVM_DEBUG(NewMI->dump(););
       } else {
         unsigned NewOpc = getADDriFromLEA(MI.getOpcode(), Offset);
         NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
diff --git a/llvm/lib/Target/X86/X86FrameLowering.cpp b/llvm/lib/Target/X86/X86FrameLowering.cpp
index 7251bdda1dd05..4cbed4c4510e2 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.cpp
+++ b/llvm/lib/Target/X86/X86FrameLowering.cpp
@@ -414,11 +414,25 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
                               StackPtr),
                       StackPtr, false, Offset);
   } else {
-    const unsigned Opc = IsSub ? getSUBriOpcode(Uses64BitFramePtr)
-                               : getADDriOpcode(Uses64BitFramePtr);
+    unsigned Opc = IsSub ? getSUBriOpcode(Uses64BitFramePtr)
+                         : getADDriOpcode(Uses64BitFramePtr);
+    int64_t Imm = AbsOffset;
+    // Prefer `add rsp, -128` over `sub rsp, 128` (and vice versa in the
+    // epilogue): 128 is the one magnitude whose negation fits the
+    // sign-extended 8-bit immediate while the value itself does not, so the
+    // flipped operation is three bytes shorter. EFLAGS is dead here (this
+    // branch clobbers it anyway). Skip under Windows CFI: the Win64 unwinder
+    // recognizes an epilogue by disassembling forward for `add rsp, imm` and
+    // must not see the SUB spelling.
+    if (AbsOffset == 128 &&
+        !MBB.getParent()->getTarget().getMCAsmInfo().usesWindowsCFI()) {
+      Opc = IsSub ? getADDriOpcode(Uses64BitFramePtr)
+                  : getSUBriOpcode(Uses64BitFramePtr);
+      Imm = -128;
+    }
     MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
              .addReg(StackPtr)
-             .addImm(AbsOffset);
+             .addImm(Imm);
     MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
   }
   return MI;
diff --git a/llvm/test/CodeGen/X86/arg-copy-elide.ll b/llvm/test/CodeGen/X86/arg-copy-elide.ll
index f13627b55856f..15edb612d7649 100644
--- a/llvm/test/CodeGen/X86/arg-copy-elide.ll
+++ b/llvm/test/CodeGen/X86/arg-copy-elide.ll
@@ -130,7 +130,7 @@ define void @high_alignment(i32 %x) {
 ; CHECK-NEXT:    pushl %ebp
 ; CHECK-NEXT:    movl %esp, %ebp
 ; CHECK-NEXT:    andl $-128, %esp
-; CHECK-NEXT:    subl $128, %esp
+; CHECK-NEXT:    addl $-128, %esp
 ; CHECK-NEXT:    movl 8(%ebp), %eax
 ; CHECK-NEXT:    movl %eax, (%esp)
 ; CHECK-NEXT:    movl %esp, %eax
diff --git a/llvm/test/CodeGen/X86/avx2-vbroadcast.ll b/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
index c50af6968f5bb..a5601f1725682 100644
--- a/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
+++ b/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
@@ -1133,7 +1133,7 @@ define void @isel_crash_32b(ptr %cV_R.addr) {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-NEXT:    andl $-32, %esp
-; X86-NEXT:    subl $128, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X86-NEXT:    vmovaps %ymm0, (%esp)
@@ -1153,7 +1153,7 @@ define void @isel_crash_32b(ptr %cV_R.addr) {
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    .cfi_def_cfa_register %rbp
 ; X64-NEXT:    andq $-32, %rsp
-; X64-NEXT:    subq $128, %rsp
+; X64-NEXT:    addq $-128, %rsp
 ; X64-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X64-NEXT:    vmovaps %ymm0, (%rsp)
 ; X64-NEXT:    vpbroadcastb (%rdi), %ymm1
@@ -1224,7 +1224,7 @@ define void @isel_crash_16w(ptr %cV_R.addr) {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-NEXT:    andl $-32, %esp
-; X86-NEXT:    subl $128, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X86-NEXT:    vmovaps %ymm0, (%esp)
@@ -1244,7 +1244,7 @@ define void @isel_crash_16w(ptr %cV_R.addr) {
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    .cfi_def_cfa_register %rbp
 ; X64-NEXT:    andq $-32, %rsp
-; X64-NEXT:    subq $128, %rsp
+; X64-NEXT:    addq $-128, %rsp
 ; X64-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X64-NEXT:    vmovaps %ymm0, (%rsp)
 ; X64-NEXT:    vpbroadcastw (%rdi), %ymm1
@@ -1315,7 +1315,7 @@ define void @isel_crash_8d(ptr %cV_R.addr) {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-NEXT:    andl $-32, %esp
-; X86-NEXT:    subl $128, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X86-NEXT:    vmovaps %ymm0, (%esp)
@@ -1335,7 +1335,7 @@ define void @isel_crash_8d(ptr %cV_R.addr) {
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    .cfi_def_cfa_register %rbp
 ; X64-NEXT:    andq $-32, %rsp
-; X64-NEXT:    subq $128, %rsp
+; X64-NEXT:    addq $-128, %rsp
 ; X64-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X64-NEXT:    vmovaps %ymm0, (%rsp)
 ; X64-NEXT:    vbroadcastss (%rdi), %ymm1
@@ -1405,7 +1405,7 @@ define void @isel_crash_4q(ptr %cV_R.addr) {
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    .cfi_def_cfa_register %ebp
 ; X86-NEXT:    andl $-32, %esp
-; X86-NEXT:    subl $128, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X86-NEXT:    vmovaps %ymm0, (%esp)
@@ -1425,7 +1425,7 @@ define void @isel_crash_4q(ptr %cV_R.addr) {
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    .cfi_def_cfa_register %rbp
 ; X64-NEXT:    andq $-32, %rsp
-; X64-NEXT:    subq $128, %rsp
+; X64-NEXT:    addq $-128, %rsp
 ; X64-NEXT:    vxorps %xmm0, %xmm0, %xmm0
 ; X64-NEXT:    vmovaps %ymm0, (%rsp)
 ; X64-NEXT:    vbroadcastsd (%rdi), %ymm1
diff --git a/llvm/test/CodeGen/X86/avx512-calling-conv.ll b/llvm/test/CodeGen/X86/avx512-calling-conv.ll
index 0a19a36f5cb03..c6aa14a8d30b6 100644
--- a/llvm/test/CodeGen/X86/avx512-calling-conv.ll
+++ b/llvm/test/CodeGen/X86/avx512-calling-conv.ll
@@ -2116,7 +2116,7 @@ define void @v64i1_mem(<128 x i32> %x, <64 x i1> %y) {
 ; SKX-NEXT:    movq %rsp, %rbp
 ; SKX-NEXT:    .cfi_def_cfa_register %rbp
 ; SKX-NEXT:    andq $-64, %rsp
-; SKX-NEXT:    subq $128, %rsp
+; SKX-NEXT:    addq $-128, %rsp
 ; SKX-NEXT:    vmovaps 16(%rbp), %zmm8
 ; SKX-NEXT:    vmovaps %zmm8, (%rsp)
 ; SKX-NEXT:    callq _v64i1_mem_callee
@@ -2283,7 +2283,7 @@ define void @v64i1_mem(<128 x i32> %x, <64 x i1> %y) {
 ; FASTISEL-NEXT:    movq %rsp, %rbp
 ; FASTISEL-NEXT:    .cfi_def_cfa_register %rbp
 ; FASTISEL-NEXT:    andq $-64, %rsp
-; FASTISEL-NEXT:    subq $128, %rsp
+; FASTISEL-NEXT:    addq $-128, %rsp
 ; FASTISEL-NEXT:    vpsllw $7, 16(%rbp), %zmm8
 ; FASTISEL-NEXT:    vpmovb2m %zmm8, %k0
 ; FASTISEL-NEXT:    vpmovm2b %k0, %zmm8
diff --git a/llvm/test/CodeGen/X86/avx512-insert-extract.ll b/llvm/test/CodeGen/X86/avx512-insert-extract.ll
index e3e7cf6085907..4efc4678f1d00 100644
--- a/llvm/test/CodeGen/X86/avx512-insert-extract.ll
+++ b/llvm/test/CodeGen/X86/avx512-insert-extract.ll
@@ -101,7 +101,7 @@ define float @test7(<16 x float> %x, i32 %ind) nounwind {
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $15, %edi
@@ -120,7 +120,7 @@ define double @test8(<8 x double> %x, i32 %ind) nounwind {
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $7, %edi
@@ -158,7 +158,7 @@ define i32 @test10(<16 x i32> %x, i32 %ind) nounwind {
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $15, %edi
@@ -1148,7 +1148,7 @@ define i64 @test_extractelement_variable_v8i64(<8 x i64> %t1, i32 %index) nounwi
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $7, %edi
@@ -1198,7 +1198,7 @@ define double @test_extractelement_variable_v8f64(<8 x double> %t1, i32 %index)
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $7, %edi
@@ -1248,7 +1248,7 @@ define i32 @test_extractelement_variable_v16i32(<16 x i32> %t1, i32 %index) noun
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $15, %edi
@@ -1298,7 +1298,7 @@ define float @test_extractelement_variable_v16f32(<16 x float> %t1, i32 %index)
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $15, %edi
@@ -1348,7 +1348,7 @@ define i16 @test_extractelement_variable_v32i16(<32 x i16> %t1, i32 %index) noun
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $31, %edi
@@ -1399,7 +1399,7 @@ define i8 @test_extractelement_variable_v64i8(<64 x i8> %t1, i32 %index) nounwin
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    andl $63, %edi
@@ -1419,7 +1419,7 @@ define i8 @test_extractelement_variable_v64i8_indexi8(<64 x i8> %t1, i8 %index)
 ; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    addb %dil, %dil
 ; CHECK-NEXT:    vmovaps %zmm0, (%rsp)
 ; CHECK-NEXT:    movzbl %dil, %eax
@@ -1663,7 +1663,7 @@ define i64 @test_insertelement_variable_v64i1(<64 x i8> %a, i8 %b, i32 %index) n
 ; KNL-NEXT:    pushq %rbp
 ; KNL-NEXT:    movq %rsp, %rbp
 ; KNL-NEXT:    andq $-64, %rsp
-; KNL-NEXT:    subq $128, %rsp
+; KNL-NEXT:    addq $-128, %rsp
 ; KNL-NEXT:    ## kill: def $esi killed $esi def $rsi
 ; KNL-NEXT:    vpxor %xmm1, %xmm1, %xmm1
 ; KNL-NEXT:    vextracti64x4 $1, %zmm0, %ymm2
diff --git a/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll b/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
index ee2bd96be099c..adb8bec6bb5bf 100644
--- a/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
+++ b/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
@@ -12,7 +12,7 @@ define zeroext i8 @test_extractelement_varible_v64i1(<64 x i8> %a, <64 x i8> %b,
 ; SKX-NEXT:    movq %rsp, %rbp
 ; SKX-NEXT:    .cfi_def_cfa_register %rbp
 ; SKX-NEXT:    andq $-64, %rsp
-; SKX-NEXT:    subq $128, %rsp
+; SKX-NEXT:    addq $-128, %rsp
 ; SKX-NEXT:    ## kill: def $edi killed $edi def $rdi
 ; SKX-NEXT:    vpcmpnleub %zmm1, %zmm0, %k0
 ; SKX-NEXT:    vpmovm2b %k0, %zmm0
diff --git a/llvm/test/CodeGen/X86/avx512-intel-ocl.ll b/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
index eb8bff26f3b77..0fa67a264fcf0 100644
--- a/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
+++ b/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
@@ -34,7 +34,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
 ; WIN32-NEXT:    pushl %ebp
 ; WIN32-NEXT:    movl %esp, %ebp
 ; WIN32-NEXT:    andl $-64, %esp
-; WIN32-NEXT:    subl $128, %esp
+; WIN32-NEXT:    addl $-128, %esp
 ; WIN32-NEXT:    vaddps %zmm1, %zmm0, %zmm0
 ; WIN32-NEXT:    movl %esp, %eax
 ; WIN32-NEXT:    pushl %eax
@@ -67,7 +67,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
 ; X64-NEXT:    pushq %r13
 ; X64-NEXT:    pushq %r12
 ; X64-NEXT:    andq $-64, %rsp
-; X64-NEXT:    subq $128, %rsp
+; X64-NEXT:    addq $-128, %rsp
 ; X64-NEXT:    vaddps %zmm1, %zmm0, %zmm0
 ; X64-NEXT:    movq %rsp, %rdi
 ; X64-NEXT:    pushq %rbp
@@ -150,7 +150,7 @@ define <16 x float> @testf16_regs(<16 x float> %a, <16 x float> %b) nounwind {
 ; X64-NEXT:    pushq %r13
 ; X64-NEXT:    pushq %r12
 ; X64-NEXT:    andq $-64, %rsp
-; X64-NEXT:    subq $128, %rsp
+; X64-NEXT:    addq $-128, %rsp
 ; X64-NEXT:    vmovaps %zmm1, %zmm16
 ; X64-NEXT:    vaddps %zmm1, %zmm0, %zmm0
 ; X64-NEXT:    movq %rsp, %rdi
diff --git a/llvm/test/CodeGen/X86/avx512-shuffles/shuffle-chained-bf16.ll b/llvm/test/CodeGen/X86/avx512-shuffles/shuffle-chained-bf16.ll
index f646f609d7e70..a7c79016334fa 100644
--- a/llvm/test/CodeGen/X86/avx512-shuffles/shuffle-chained-bf16.ll
+++ b/llvm/test/CodeGen/X86/avx512-shuffles/shuffle-chained-bf16.ll
@@ -12,7 +12,7 @@ define <2 x bfloat> @shuffle_chained_v32bf16_v2bf16(<32 x bfloat> %a) {
 ; CHECK-NEXT:    movq %rsp, %rbp
 ; CHECK-NEXT:    .cfi_def_cfa_register %rbp
 ; CHECK-NEXT:    andq $-64, %rsp
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    vmovd {{.*#+}} xmm1 = [0,16,0,0,0,0,0,0]
 ; CHECK-NEXT:    vpermw %zmm0, %zmm1, %zmm0
 ; CHECK-NEXT:    # kill: def $xmm0 killed $xmm0 killed $zmm0
diff --git a/llvm/test/CodeGen/X86/avx512fp16-mov.ll b/llvm/test/CodeGen/X86/avx512fp16-mov.ll
index 79fc699f09932..e2f2688b1d9f3 100644
--- a/llvm/test/CodeGen/X86/avx512fp16-mov.ll
+++ b/llvm/test/CodeGen/X86/avx512fp16-mov.ll
@@ -1640,7 +1640,7 @@ define half @extract_f16_8(<32 x half> %x, i64 %idx) nounwind {
 ; X64-NEXT:    pushq %rbp
 ; X64-NEXT:    movq %rsp, %rbp
 ; X64-NEXT:    andq $-64, %rsp
-; X64-NEXT:    subq $128, %rsp
+; X64-NEXT:    addq $-128, %rsp
 ; X64-NEXT:    andl $31, %edi
 ; X64-NEXT:    vmovaps %zmm0, (%rsp)
 ; X64-NEXT:    vmovsh {{.*#+}} xmm0 = mem[0],zero,zero,zero,zero,zero,zero,zero
@@ -1654,7 +1654,7 @@ define half @extract_f16_8(<32 x half> %x, i64 %idx) nounwind {
 ; X86-NEXT:    pushl %ebp
 ; X86-NEXT:    movl %esp, %ebp
 ; X86-NEXT:    andl $-64, %esp
-; X86-NEXT:    subl $128, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movl 8(%ebp), %eax
 ; X86-NEXT:    andl $31, %eax
 ; X86-NEXT:    vmovaps %zmm0, (%esp)
diff --git a/llvm/test/CodeGen/X86/frem.ll b/llvm/test/CodeGen/X86/frem.ll
index 959265d08299a..c259a1c65d9e5 100644
--- a/llvm/test/CodeGen/X86/frem.ll
+++ b/llvm/test/CodeGen/X86/frem.ll
@@ -1397,7 +1397,7 @@ define void @frem_v4f80(<4 x x86_fp80> %a0, <4 x x86_fp80> %a1, ptr%p3) nounwind
 ; CHECK-LABEL: frem_v4f80:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    pushq %rbx
-; CHECK-NEXT:    subq $128, %rsp
+; CHECK-NEXT:    addq $-128, %rsp
 ; CHECK-NEXT:    movq %rdi, %rbx
 ; CHECK-NEXT:    fldt {{[0-9]+}}(%rsp)
 ; CHECK-NEXT:    fstpt {{[-0-9]+}}(%r{{[sb]}}p) # 10-byte Folded Spill
@@ -1441,7 +1441,7 @@ define void @frem_v4f80(<4 x x86_fp80> %a0, <4 x x86_fp80> %a1, ptr%p3) nounwind
 ; CHECK-NEXT:    fstpt 10(%rbx)
 ; CHECK-NEXT:    fldt {{[-0-9]+}}(%r{{[sb]}}p) # 10-byte Folded Reload
 ; CHECK-NEXT:    fstpt (%rbx)
-; CHECK-NEXT:    addq $128, %rsp
+; CHECK-NEXT:    subq $-128, %rsp
 ; CHECK-NEXT:    popq %rbx
 ; CHECK-NEXT:    retq
   %frem = frem <4 x x86_fp80> %a0, %a1
diff --git a/llvm/test/CodeGen/X86/i386-baseptr.ll b/llvm/test/CodeGen/X86/i386-baseptr.ll
index 777eb838b84cc..01702cbeade50 100644
--- a/llvm/test/CodeGen/X86/i386-baseptr.ll
+++ b/llvm/test/CodeGen/X86/i386-baseptr.ll
@@ -97,7 +97,7 @@ define x86_regcallcc void @clobber_baseptr_argptr(i32 %param1, i32 %param2, i32
 ; CHECK-NEXT:    .cfi_def_cfa_register %ebp
 ; CHECK-NEXT:    pushl %ebx
 ; CHECK-NEXT:    andl $-128, %esp
-; CHECK-NEXT:    subl $128, %esp
+; CHECK-NEXT:    addl $-128, %esp
 ; CHECK-NEXT:    movl %esp, %esi
 ; CHECK-NEXT:    .cfi_offset %ebx, -12
 ; CHECK-NEXT:    movl 8(%ebp), %edi
diff --git a/llvm/test/CodeGen/X86/lea-fixup-disp128.mir b/llvm/test/CodeGen/X86/lea-fixup-disp128.mir
index 8c9088882af74..653bd2b58da17 100644
--- a/llvm/test/CodeGen/X86/lea-fixup-disp128.mir
+++ b/llvm/test/CodeGen/X86/lea-fixup-disp128.mir
@@ -25,7 +25,7 @@ body:             |
   bb.0:
     ; CHECK-LABEL: name: disp128
     ; CHECK: renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 0, $noreg
-    ; CHECK-NEXT: $rax = ADD64ri32 $rax, 128, implicit-def $eflags
+    ; CHECK-NEXT: $rax = SUB64ri32 $rax, -128, implicit-def $eflags
     ; CHECK-NEXT: JMP64r renamable $rax
     renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 128, $noreg
     JMP64r renamable $rax
@@ -36,7 +36,7 @@ body:             |
   bb.0:
     ; CHECK-LABEL: name: disp128_32
     ; CHECK: renamable $eax = LEA64_32r renamable $rdi, 1, renamable $rsi, 0, $noreg
-    ; CHECK-NEXT: $eax = ADD32ri $eax, 128, implicit-def $eflags
+    ; CHECK-NEXT: $eax = SUB32ri $eax, -128, implicit-def $eflags
     ; CHECK-NEXT: RET64 implicit $eax
     renamable $eax = LEA64_32r renamable $rdi, 1, renamable $rsi, 128, $noreg
     RET64 implicit $eax
diff --git a/llvm/test/CodeGen/X86/nontemporal-loads-2.ll b/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
index 28ddfe5ab62dc..71c23d88085ec 100644
--- a/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
+++ b/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
@@ -610,7 +610,7 @@ define <8 x double> @test_v8f64_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -689,7 +689,7 @@ define <16 x float> @test_v16f32_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -768,7 +768,7 @@ define <8 x i64> @test_v8i64_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -847,7 +847,7 @@ define <16 x i32> @test_v16i32_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -926,7 +926,7 @@ define <32 x i16> @test_v32i16_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -1005,7 +1005,7 @@ define <64 x i8> @test_v64i8_align16(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 48(%rdi), %xmm0
 ; AVX512-NEXT:    vmovdqa %xmm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %xmm0
@@ -1060,7 +1060,7 @@ define <8 x double> @test_v8f64_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
@@ -1111,7 +1111,7 @@ define <16 x float> @test_v16f32_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
@@ -1162,7 +1162,7 @@ define <8 x i64> @test_v8i64_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
@@ -1213,7 +1213,7 @@ define <16 x i32> @test_v16i32_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
@@ -1264,7 +1264,7 @@ define <32 x i16> @test_v32i16_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
@@ -1315,7 +1315,7 @@ define <64 x i8> @test_v64i8_align32(ptr %src) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vmovntdqa 32(%rdi), %ymm0
 ; AVX512-NEXT:    vmovdqa %ymm0, {{[0-9]+}}(%rsp)
 ; AVX512-NEXT:    vmovntdqa (%rdi), %ymm0
diff --git a/llvm/test/CodeGen/X86/sdiv_fix_sat.ll b/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
index e7d41d5bebc8a..eac2bb7ebe962 100644
--- a/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
+++ b/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
@@ -370,7 +370,7 @@ define i64 @func5(i64 %x, i64 %y) nounwind {
 ; X86-NEXT:    pushl %edi
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    andl $-16, %esp
-; X86-NEXT:    subl $128, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    movl 8(%ebp), %esi
 ; X86-NEXT:    movl 12(%ebp), %edi
 ; X86-NEXT:    movl 16(%ebp), %ecx
diff --git a/llvm/test/CodeGen/X86/shift-i128.ll b/llvm/test/CodeGen/X86/shift-i128.ll
index 73755f5b2a5ee..6c9cdb1b62ca4 100644
--- a/llvm/test/CodeGen/X86/shift-i128.ll
+++ b/llvm/test/CodeGen/X86/shift-i128.ll
@@ -538,7 +538,7 @@ define void @test_shl_v2i128(<2 x i128> %x, <2 x i128> %a, ptr nocapture %r) nou
 ; i686-NEXT:    pushl %edi
 ; i686-NEXT:    pushl %esi
 ; i686-NEXT:    andl $-16, %esp
-; i686-NEXT:    subl $128, %esp
+; i686-NEXT:    addl $-128, %esp
 ; i686-NEXT:    movl 40(%ebp), %edi
 ; i686-NEXT:    movl 24(%ebp), %eax
 ; i686-NEXT:    movl 28(%ebp), %ecx
diff --git a/llvm/test/CodeGen/X86/smulo-128-legalisation-lowering.ll b/llvm/test/CodeGen/X86/smulo-128-legalisation-lowering.ll
index 13596e1b18768..6e09f66980aa2 100644
--- a/llvm/test/CodeGen/X86/smulo-128-legalisation-lowering.ll
+++ b/llvm/test/CodeGen/X86/smulo-128-legalisation-lowering.ll
@@ -523,7 +523,7 @@ define zeroext i1 @smuloi256(i256 %v1, i256 %v2, ptr %res) {
 ; X86-NEXT:    .cfi_def_cfa_offset 16
 ; X86-NEXT:    pushl %esi
 ; X86-NEXT:    .cfi_def_cfa_offset 20
-; X86-NEXT:    subl $128, %esp
+; X86-NEXT:    addl $-128, %esp
 ; X86-NEXT:    .cfi_def_cfa_offset 148
 ; X86-NEXT:    .cfi_offset %esi, -20
 ; X86-NEXT:    .cfi_offset %edi, -16
@@ -1280,7 +1280,7 @@ define zeroext i1 @smuloi256(i256 %v1, i256 %v2, ptr %res) {
 ; X86-NEXT:    movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx ## 4-byte Reload
 ; X86-NEXT:    movl %ecx, 24(%eax)
 ; X86-NEXT:    setne %al
-; X86-NEXT:    addl $128, %esp
+; X86-NEXT:    subl $-128, %esp
 ; X86-NEXT:    popl %esi
 ; X86-NEXT:    popl %edi
 ; X86-NEXT:    popl %ebx
diff --git a/llvm/test/CodeGen/X86/stack-adjust-128.ll b/llvm/test/CodeGen/X86/stack-adjust-128.ll
index a19631248e7d6..15cee691a4e69 100644
--- a/llvm/test/CodeGen/X86/stack-adjust-128.ll
+++ b/llvm/test/CodeGen/X86/stack-adjust-128.ll
@@ -17,14 +17,14 @@ define ptr @frame128(ptr %p) {
 ; LINUX:       # %bb.0:
 ; LINUX-NEXT:    pushq %rbx
 ; LINUX-NEXT:    .cfi_def_cfa_offset 16
-; LINUX-NEXT:    subq $128, %rsp
+; LINUX-NEXT:    addq $-128, %rsp
 ; LINUX-NEXT:    .cfi_def_cfa_offset 144
 ; LINUX-NEXT:    .cfi_offset %rbx, -16
 ; LINUX-NEXT:    movq %rdi, %rbx
 ; LINUX-NEXT:    leaq {{[0-9]+}}(%rsp), %rdi
 ; LINUX-NEXT:    callq escape at PLT
 ; LINUX-NEXT:    movq %rbx, %rax
-; LINUX-NEXT:    addq $128, %rsp
+; LINUX-NEXT:    subq $-128, %rsp
 ; LINUX-NEXT:    .cfi_def_cfa_offset 16
 ; LINUX-NEXT:    popq %rbx
 ; LINUX-NEXT:    .cfi_def_cfa_offset 8
diff --git a/llvm/test/CodeGen/X86/var-permute-512.ll b/llvm/test/CodeGen/X86/var-permute-512.ll
index 0a3b9fa5947db..6b893f834f50d 100644
--- a/llvm/test/CodeGen/X86/var-permute-512.ll
+++ b/llvm/test/CodeGen/X86/var-permute-512.ll
@@ -97,7 +97,7 @@ define <32 x i16> @var_shuffle_v32i16(<32 x i16> %v, <32 x i16> %indices) nounwi
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-64, %rsp
-; AVX512F-NEXT:    subq $128, %rsp
+; AVX512F-NEXT:    addq $-128, %rsp
 ; AVX512F-NEXT:    vextracti128 $1, %ymm1, %xmm2
 ; AVX512F-NEXT:    vextracti32x4 $2, %zmm1, %xmm3
 ; AVX512F-NEXT:    vextracti32x4 $3, %zmm1, %xmm4
@@ -326,7 +326,7 @@ define <64 x i8> @var_shuffle_v64i8(<64 x i8> %v, <64 x i8> %indices) nounwind {
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-64, %rsp
-; AVX512F-NEXT:    subq $128, %rsp
+; AVX512F-NEXT:    addq $-128, %rsp
 ; AVX512F-NEXT:    vextracti128 $1, %ymm1, %xmm2
 ; AVX512F-NEXT:    vextracti32x4 $2, %zmm1, %xmm3
 ; AVX512F-NEXT:    vextracti32x4 $3, %zmm1, %xmm4
@@ -551,7 +551,7 @@ define <64 x i8> @var_shuffle_v64i8(<64 x i8> %v, <64 x i8> %indices) nounwind {
 ; AVX512BW-NEXT:    pushq %rbp
 ; AVX512BW-NEXT:    movq %rsp, %rbp
 ; AVX512BW-NEXT:    andq $-64, %rsp
-; AVX512BW-NEXT:    subq $128, %rsp
+; AVX512BW-NEXT:    addq $-128, %rsp
 ; AVX512BW-NEXT:    vextracti128 $1, %ymm1, %xmm2
 ; AVX512BW-NEXT:    vextracti32x4 $2, %zmm1, %xmm3
 ; AVX512BW-NEXT:    vextracti32x4 $3, %zmm1, %xmm4
@@ -1064,7 +1064,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-64, %rsp
-; AVX512F-NEXT:    subq $128, %rsp
+; AVX512F-NEXT:    addq $-128, %rsp
 ; AVX512F-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX512F-NEXT:    vpbroadcastd %esi, %zmm2
 ; AVX512F-NEXT:    vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm1 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
@@ -1315,7 +1315,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
 ; AVX512BW-NEXT:    pushq %rbp
 ; AVX512BW-NEXT:    movq %rsp, %rbp
 ; AVX512BW-NEXT:    andq $-64, %rsp
-; AVX512BW-NEXT:    subq $128, %rsp
+; AVX512BW-NEXT:    addq $-128, %rsp
 ; AVX512BW-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX512BW-NEXT:    vpbroadcastd %esi, %zmm2
 ; AVX512BW-NEXT:    vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm1 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
@@ -1566,7 +1566,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
 ; AVX512VBMI-NEXT:    pushq %rbp
 ; AVX512VBMI-NEXT:    movq %rsp, %rbp
 ; AVX512VBMI-NEXT:    andq $-64, %rsp
-; AVX512VBMI-NEXT:    subq $128, %rsp
+; AVX512VBMI-NEXT:    addq $-128, %rsp
 ; AVX512VBMI-NEXT:    # kill: def $esi killed $esi def $rsi
 ; AVX512VBMI-NEXT:    vpbroadcastd %esi, %zmm1
 ; AVX512VBMI-NEXT:    vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm1, %zmm2 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
diff --git a/llvm/test/CodeGen/X86/vec-strict-inttofp-512.ll b/llvm/test/CodeGen/X86/vec-strict-inttofp-512.ll
index bf60f0631749d..0b83512b8e688 100644
--- a/llvm/test/CodeGen/X86/vec-strict-inttofp-512.ll
+++ b/llvm/test/CodeGen/X86/vec-strict-inttofp-512.ll
@@ -270,7 +270,7 @@ define <8 x double> @sitofp_v8i64_v8f64(<8 x i64> %x) #0 {
 ; NODQ-32-NEXT:    movl %esp, %ebp
 ; NODQ-32-NEXT:    .cfi_def_cfa_register %ebp
 ; NODQ-32-NEXT:    andl $-8, %esp
-; NODQ-32-NEXT:    subl $128, %esp
+; NODQ-32-NEXT:    addl $-128, %esp
 ; NODQ-32-NEXT:    vextractf32x4 $2, %zmm0, %xmm1
 ; NODQ-32-NEXT:    vmovlps %xmm1, {{[0-9]+}}(%esp)
 ; NODQ-32-NEXT:    vshufps {{.*#+}} xmm1 = xmm1[2,3,2,3]
@@ -368,7 +368,7 @@ define <8 x double> @uitofp_v8i64_v8f64(<8 x i64> %x) #0 {
 ; NODQ-32-NEXT:    movl %esp, %ebp
 ; NODQ-32-NEXT:    .cfi_def_cfa_register %ebp
 ; NODQ-32-NEXT:    andl $-8, %esp
-; NODQ-32-NEXT:    subl $128, %esp
+; NODQ-32-NEXT:    addl $-128, %esp
 ; NODQ-32-NEXT:    vextractf32x4 $2, %zmm0, %xmm3
 ; NODQ-32-NEXT:    vmovlps %xmm3, {{[0-9]+}}(%esp)
 ; NODQ-32-NEXT:    vshufps {{.*#+}} xmm1 = xmm3[2,3,2,3]
diff --git a/llvm/test/CodeGen/X86/vector-compress.ll b/llvm/test/CodeGen/X86/vector-compress.ll
index 39ef06af99340..7f19806114894 100644
--- a/llvm/test/CodeGen/X86/vector-compress.ll
+++ b/llvm/test/CodeGen/X86/vector-compress.ll
@@ -2778,7 +2778,7 @@ define <64 x i8> @test_compress_v64i8(<64 x i8> %vec, <64 x i1> %mask, <64 x i8>
 ; AVX512VL-ONLY-NEXT:    pushq %rbp
 ; AVX512VL-ONLY-NEXT:    movq %rsp, %rbp
 ; AVX512VL-ONLY-NEXT:    andq $-64, %rsp
-; AVX512VL-ONLY-NEXT:    subq $128, %rsp
+; AVX512VL-ONLY-NEXT:    addq $-128, %rsp
 ; AVX512VL-ONLY-NEXT:    vpsllw $7, %zmm1, %zmm1
 ; AVX512VL-ONLY-NEXT:    vpmovb2m %zmm1, %k6
 ; AVX512VL-ONLY-NEXT:    kshiftrq $63, %k6, %k0
@@ -3575,7 +3575,7 @@ define <32 x i16> @test_compress_v32i16(<32 x i16> %vec, <32 x i1> %mask, <32 x
 ; AVX512F-NEXT:    pushq %rbp
 ; AVX512F-NEXT:    movq %rsp, %rbp
 ; AVX512F-NEXT:    andq $-64, %rsp
-; AVX512F-NEXT:    subq $128, %rsp
+; AVX512F-NEXT:    addq $-128, %rsp
 ; AVX512F-NEXT:    vextracti128 $1, %ymm1, %xmm4
 ; AVX512F-NEXT:    vpmovzxbw {{.*#+}} ymm3 = xmm4[0],zero,xmm4[1],zero,xmm4[2],zero,xmm4[3],zero,xmm4[4],zero,xmm4[5],zero,xmm4[6],zero,xmm4[7],zero,xmm4[8],zero,xmm4[9],zero,xmm4[10],zero,xmm4[11],zero,xmm4[12],zero,xmm4[13],zero,xmm4[14],zero,xmm4[15],zero
 ; AVX512F-NEXT:    vpmovsxbd %xmm4, %zmm4
diff --git a/llvm/test/CodeGen/X86/vector-extract-last-active.ll b/llvm/test/CodeGen/X86/vector-extract-last-active.ll
index bd6f35df29490..16a2de9783893 100644
--- a/llvm/test/CodeGen/X86/vector-extract-last-active.ll
+++ b/llvm/test/CodeGen/X86/vector-extract-last-active.ll
@@ -583,7 +583,7 @@ define i32 @extract_last_active_v16i32(<16 x i32> %a, <16 x i1> %c) nounwind {
 ; AVX512-NEXT:    pushq %rbp
 ; AVX512-NEXT:    movq %rsp, %rbp
 ; AVX512-NEXT:    andq $-64, %rsp
-; AVX512-NEXT:    subq $128, %rsp
+; AVX512-NEXT:    addq $-128, %rsp
 ; AVX512-NEXT:    vpsllw $7, %xmm1, %xmm1
 ; AVX512-NEXT:    vpmovb2m %xmm1, %k1
 ; AVX512-NEXT:    vmovdqa64 %zmm0, (%rsp)
diff --git a/llvm/test/CodeGen/X86/wide-scalar-shift-by-byte-multiple-legalization.ll b/llvm/test/CodeGen/X86/wide-scalar-shift-by-byte-multiple-legalization.ll
index e6e20d03c4138..4abf5ccbd81ed 100644
--- a/llvm/test/CodeGen/X86/wide-scalar-shift-by-byte-multiple-legalization.ll
+++ b/llvm/test/CodeGen/X86/wide-scalar-shift-by-byte-multiple-legalization.ll
@@ -19940,7 +19940,7 @@ define void @ashr_64bytes_qwordOff(ptr %src.ptr, ptr %qwordOff.ptr, ptr %dst) no
 ; X86-SSE42-NEXT:    pushl %ebx
 ; X86-SSE42-NEXT:    pushl %edi
 ; X86-SSE42-NEXT:    pushl %esi
-; X86-SSE42-NEXT:    subl $128, %esp
+; X86-SSE42-NEXT:    addl $-128, %esp
 ; X86-SSE42-NEXT:    movl {{[0-9]+}}(%esp), %eax
 ; X86-SSE42-NEXT:    movl {{[0-9]+}}(%esp), %ecx
 ; X86-SSE42-NEXT:    movl {{[0-9]+}}(%esp), %edx
@@ -19985,7 +19985,7 @@ define void @ashr_64bytes_qwordOff(ptr %src.ptr, ptr %qwordOff.ptr, ptr %dst) no
 ; X86-SSE42-NEXT:    movups %xmm2, 32(%eax)
 ; X86-SSE42-NEXT:    movups %xmm1, 16(%eax)
 ; X86-SSE42-NEXT:    movups %xmm0, (%eax)
-; X86-SSE42-NEXT:    addl $128, %esp
+; X86-SSE42-NEXT:    subl $-128, %esp
 ; X86-SSE42-NEXT:    popl %esi
 ; X86-SSE42-NEXT:    popl %edi
 ; X86-SSE42-NEXT:    popl %ebx
@@ -19996,7 +19996,7 @@ define void @ashr_64bytes_qwordOff(ptr %src.ptr, ptr %qwordOff.ptr, ptr %dst) no
 ; X86-AVX1-NEXT:    pushl %ebx
 ; X86-AVX1-NEXT:    pushl %edi
 ; X86-AVX1-NEXT:    pushl %esi
-; X86-AVX1-NEXT:    subl $128, %esp
+; X86-AVX1-NEXT:    addl $-128, %esp
 ; X86-AVX1-NEXT:    movl {{[0-9]+}}(%esp), %eax
 ; X86-AVX1-NEXT:    movl {{[0-9]+}}(%esp), %ecx
 ; X86-AVX1-NEXT:    movl {{[0-9]+}}(%esp), %edx
@@ -20039,7 +20039,7 @@ define void @ashr_64bytes_qwordOff(ptr %src.ptr, ptr %qwordOff.ptr, ptr %dst) no
 ; X86-AVX1-NEXT:    vmovups %xmm2, 32(%eax)
 ; X86-AVX1-NEXT:    vmovups %xmm1, 16(%eax)
 ; X86-AVX1-NEXT:    vmovups %xmm0, (%eax)
-; X86-AVX1-NEXT:    addl $128, %esp
+; X86-AVX1-NEXT:    subl $-128, %esp
 ; X86-AVX1-NEXT:    popl %esi
 ; X86-AVX1-NEXT:    popl %edi
 ; X86-AVX1-NEXT:    popl %ebx
diff --git a/llvm/test/CodeGen/X86/widen-load-of-small-alloca-with-zero-upper-half.ll b/llvm/test/CodeGen/X86/widen-load-of-small-alloca-with-zero-upper-half.ll
index fde915247760a..dedf5b4150908 100644
--- a/llvm/test/CodeGen/X86/widen-load-of-small-alloca-with-zero-upper-half.ll
+++ b/llvm/test/CodeGen/X86/widen-load-of-small-alloca-with-zero-upper-half.ll
@@ -2488,7 +2488,7 @@ define void @load_8byte_chunk_of_64byte_alloca_with_zero_upper_half(ptr %src, i6
 ; X86-SHLD-NEXT:    pushl %ebx
 ; X86-SHLD-NEXT:    pushl %edi
 ; X86-SHLD-NEXT:    pushl %esi
-; X86-SHLD-NEXT:    subl $128, %esp
+; X86-SHLD-NEXT:    addl $-128, %esp
 ; X86-SHLD-NEXT:    movl {{[0-9]+}}(%esp), %ecx
 ; X86-SHLD-NEXT:    movl {{[0-9]+}}(%esp), %eax
 ; X86-SHLD-NEXT:    movl {{[0-9]+}}(%esp), %edx
@@ -2516,7 +2516,7 @@ define void @load_8byte_chunk_of_64byte_alloca_with_zero_upper_half(ptr %src, i6
 ; X86-SHLD-NEXT:    shrdl %cl, %esi, %edx
 ; X86-SHLD-NEXT:    movl %ebx, 4(%eax)
 ; X86-SHLD-NEXT:    movl %edx, (%eax)
-; X86-SHLD-NEXT:    addl $128, %esp
+; X86-SHLD-NEXT:    subl $-128, %esp
 ; X86-SHLD-NEXT:    popl %esi
 ; X86-SHLD-NEXT:    popl %edi
 ; X86-SHLD-NEXT:    popl %ebx
diff --git a/llvm/test/CodeGen/X86/x86-64-baseptr.ll b/llvm/test/CodeGen/X86/x86-64-baseptr.ll
index 62c63d5defe60..4f366ba3144a8 100644
--- a/llvm/test/CodeGen/X86/x86-64-baseptr.ll
+++ b/llvm/test/CodeGen/X86/x86-64-baseptr.ll
@@ -116,7 +116,7 @@ define void @clobber_base() #0 {
 ; X32ABI-NEXT:    .cfi_def_cfa_register %rbp
 ; X32ABI-NEXT:    pushq %rbx
 ; X32ABI-NEXT:    andl $-128, %esp
-; X32ABI-NEXT:    subl $128, %esp
+; X32ABI-NEXT:    addl $-128, %esp
 ; X32ABI-NEXT:    movl %esp, %ebx
 ; X32ABI-NEXT:    .cfi_offset %rbx, -24
 ; X32ABI-NEXT:    callq helper at PLT



More information about the llvm-commits mailing list