[llvm] [X86] Use 8-bit immediates for 128-byte stack adjustments and LEA splits (PR #219866)
Andrew Gaul via llvm-commits
llvm-commits at lists.llvm.org
Sun Aug 30 21:21:13 PDT 2026
https://github.com/gaul updated https://github.com/llvm/llvm-project/pull/219866
>From 7c1581e6aef72d281e63e0943226e3c2d072e5d9 Mon Sep 17 00:00:00 2001
From: Andrew Gaul <andrew at gaul.org>
Date: Sun, 30 Aug 2026 21:19:23 -0700
Subject: [PATCH 1/2] [X86] Add test coverage for 128-byte stack adjustments
and slow-LEA splits
Both currently use the 32-bit immediate encoding of ADD/SUB with 128
where the sign-extended 8-bit encoding of the inverse operation with
-128 would be three bytes shorter.
Co-Authored-By: Claude Fable 5 <noreply at anthropic.com>
---
llvm/test/CodeGen/X86/lea-fixup-disp128.mir | 65 ++++++++++
llvm/test/CodeGen/X86/stack-adjust-128.ll | 136 ++++++++++++++++++++
2 files changed, 201 insertions(+)
create mode 100644 llvm/test/CodeGen/X86/lea-fixup-disp128.mir
create mode 100644 llvm/test/CodeGen/X86/stack-adjust-128.ll
diff --git a/llvm/test/CodeGen/X86/lea-fixup-disp128.mir b/llvm/test/CodeGen/X86/lea-fixup-disp128.mir
new file mode 100644
index 0000000000000..8c9088882af74
--- /dev/null
+++ b/llvm/test/CodeGen/X86/lea-fixup-disp128.mir
@@ -0,0 +1,65 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 3
+# RUN: llc -mtriple=x86_64-unknown-linux-gnu -mattr=slow-3ops-lea -run-pass x86-fixup-leas -o - %s | FileCheck %s
+
+# When splitting a slow 3-operand LEA, a displacement of exactly 128 should
+# become SUB of -128 rather than ADD of 128: only -128 fits the sign-extended
+# 8-bit immediate, saving three bytes. Adjacent displacements keep the ADD.
+
+--- |
+ define void @disp128() nounwind {
+ ret void
+ }
+ define void @disp128_32() nounwind {
+ ret void
+ }
+ define void @disp129() nounwind {
+ ret void
+ }
+ define void @disp_minus128() nounwind {
+ ret void
+ }
+
+---
+name: disp128
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: disp128
+ ; CHECK: renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 0, $noreg
+ ; CHECK-NEXT: $rax = ADD64ri32 $rax, 128, implicit-def $eflags
+ ; CHECK-NEXT: JMP64r renamable $rax
+ renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 128, $noreg
+ JMP64r renamable $rax
+...
+---
+name: disp128_32
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: disp128_32
+ ; CHECK: renamable $eax = LEA64_32r renamable $rdi, 1, renamable $rsi, 0, $noreg
+ ; CHECK-NEXT: $eax = ADD32ri $eax, 128, implicit-def $eflags
+ ; CHECK-NEXT: RET64 implicit $eax
+ renamable $eax = LEA64_32r renamable $rdi, 1, renamable $rsi, 128, $noreg
+ RET64 implicit $eax
+...
+---
+name: disp129
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: disp129
+ ; CHECK: renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 0, $noreg
+ ; CHECK-NEXT: $rax = ADD64ri32 $rax, 129, implicit-def $eflags
+ ; CHECK-NEXT: JMP64r renamable $rax
+ renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 129, $noreg
+ JMP64r renamable $rax
+...
+---
+name: disp_minus128
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: disp_minus128
+ ; CHECK: renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 0, $noreg
+ ; CHECK-NEXT: $rax = ADD64ri32 $rax, -128, implicit-def $eflags
+ ; CHECK-NEXT: JMP64r renamable $rax
+ renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, -128, $noreg
+ JMP64r renamable $rax
+...
diff --git a/llvm/test/CodeGen/X86/stack-adjust-128.ll b/llvm/test/CodeGen/X86/stack-adjust-128.ll
new file mode 100644
index 0000000000000..a19631248e7d6
--- /dev/null
+++ b/llvm/test/CodeGen/X86/stack-adjust-128.ll
@@ -0,0 +1,136 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -verify-machineinstrs | FileCheck %s --check-prefix=LINUX
+; RUN: llc < %s -mtriple=x86_64-pc-windows-msvc -verify-machineinstrs | FileCheck %s --check-prefix=WIN64
+
+; A 128-byte stack adjustment should be emitted as `add rsp, -128` in the
+; prologue and `sub rsp, -128` in the epilogue: 128 misses the sign-extended
+; 8-bit immediate form, but -128 fits it, saving three bytes each. Windows
+; keeps the canonical SUB/ADD spelling because the Win64 unwinder recognizes
+; epilogues by disassembling for `add rsp, imm`.
+
+declare void @escape(ptr)
+
+; One value lives across the call, forcing a single CSR push; the 120 bytes
+; of locals then round up to a 128-byte adjustment on Linux.
+define ptr @frame128(ptr %p) {
+; LINUX-LABEL: frame128:
+; LINUX: # %bb.0:
+; LINUX-NEXT: pushq %rbx
+; LINUX-NEXT: .cfi_def_cfa_offset 16
+; LINUX-NEXT: subq $128, %rsp
+; LINUX-NEXT: .cfi_def_cfa_offset 144
+; LINUX-NEXT: .cfi_offset %rbx, -16
+; LINUX-NEXT: movq %rdi, %rbx
+; LINUX-NEXT: leaq {{[0-9]+}}(%rsp), %rdi
+; LINUX-NEXT: callq escape at PLT
+; LINUX-NEXT: movq %rbx, %rax
+; LINUX-NEXT: addq $128, %rsp
+; LINUX-NEXT: .cfi_def_cfa_offset 16
+; LINUX-NEXT: popq %rbx
+; LINUX-NEXT: .cfi_def_cfa_offset 8
+; LINUX-NEXT: retq
+;
+; WIN64-LABEL: frame128:
+; WIN64: # %bb.0:
+; WIN64-NEXT: pushq %rsi
+; WIN64-NEXT: .seh_pushreg %rsi
+; WIN64-NEXT: subq $160, %rsp
+; WIN64-NEXT: .seh_stackalloc 160
+; WIN64-NEXT: .seh_endprologue
+; WIN64-NEXT: movq %rcx, %rsi
+; WIN64-NEXT: leaq {{[0-9]+}}(%rsp), %rcx
+; WIN64-NEXT: callq escape
+; WIN64-NEXT: movq %rsi, %rax
+; WIN64-NEXT: .seh_startepilogue
+; WIN64-NEXT: addq $160, %rsp
+; WIN64-NEXT: popq %rsi
+; WIN64-NEXT: .seh_endepilogue
+; WIN64-NEXT: retq
+; WIN64-NEXT: .seh_endproc
+ %buf = alloca [120 x i8]
+ call void @escape(ptr %buf)
+ ret ptr %p
+}
+
+; The same shape with 96 bytes of locals reaches a 128-byte adjustment on
+; Win64 (32-byte shadow area included), which must keep `sub rsp, 128`.
+define ptr @frame128_win(ptr %p) {
+; LINUX-LABEL: frame128_win:
+; LINUX: # %bb.0:
+; LINUX-NEXT: pushq %rbx
+; LINUX-NEXT: .cfi_def_cfa_offset 16
+; LINUX-NEXT: subq $96, %rsp
+; LINUX-NEXT: .cfi_def_cfa_offset 112
+; LINUX-NEXT: .cfi_offset %rbx, -16
+; LINUX-NEXT: movq %rdi, %rbx
+; LINUX-NEXT: movq %rsp, %rdi
+; LINUX-NEXT: callq escape at PLT
+; LINUX-NEXT: movq %rbx, %rax
+; LINUX-NEXT: addq $96, %rsp
+; LINUX-NEXT: .cfi_def_cfa_offset 16
+; LINUX-NEXT: popq %rbx
+; LINUX-NEXT: .cfi_def_cfa_offset 8
+; LINUX-NEXT: retq
+;
+; WIN64-LABEL: frame128_win:
+; WIN64: # %bb.0:
+; WIN64-NEXT: pushq %rsi
+; WIN64-NEXT: .seh_pushreg %rsi
+; WIN64-NEXT: subq $128, %rsp
+; WIN64-NEXT: .seh_stackalloc 128
+; WIN64-NEXT: .seh_endprologue
+; WIN64-NEXT: movq %rcx, %rsi
+; WIN64-NEXT: leaq {{[0-9]+}}(%rsp), %rcx
+; WIN64-NEXT: callq escape
+; WIN64-NEXT: movq %rsi, %rax
+; WIN64-NEXT: .seh_startepilogue
+; WIN64-NEXT: addq $128, %rsp
+; WIN64-NEXT: popq %rsi
+; WIN64-NEXT: .seh_endepilogue
+; WIN64-NEXT: retq
+; WIN64-NEXT: .seh_endproc
+ %buf = alloca [96 x i8]
+ call void @escape(ptr %buf)
+ ret ptr %p
+}
+
+; An adjacent size keeps the plain 32-bit immediate.
+define ptr @frame144(ptr %p) {
+; LINUX-LABEL: frame144:
+; LINUX: # %bb.0:
+; LINUX-NEXT: pushq %rbx
+; LINUX-NEXT: .cfi_def_cfa_offset 16
+; LINUX-NEXT: subq $144, %rsp
+; LINUX-NEXT: .cfi_def_cfa_offset 160
+; LINUX-NEXT: .cfi_offset %rbx, -16
+; LINUX-NEXT: movq %rdi, %rbx
+; LINUX-NEXT: leaq {{[0-9]+}}(%rsp), %rdi
+; LINUX-NEXT: callq escape at PLT
+; LINUX-NEXT: movq %rbx, %rax
+; LINUX-NEXT: addq $144, %rsp
+; LINUX-NEXT: .cfi_def_cfa_offset 16
+; LINUX-NEXT: popq %rbx
+; LINUX-NEXT: .cfi_def_cfa_offset 8
+; LINUX-NEXT: retq
+;
+; WIN64-LABEL: frame144:
+; WIN64: # %bb.0:
+; WIN64-NEXT: pushq %rsi
+; WIN64-NEXT: .seh_pushreg %rsi
+; WIN64-NEXT: subq $176, %rsp
+; WIN64-NEXT: .seh_stackalloc 176
+; WIN64-NEXT: .seh_endprologue
+; WIN64-NEXT: movq %rcx, %rsi
+; WIN64-NEXT: leaq {{[0-9]+}}(%rsp), %rcx
+; WIN64-NEXT: callq escape
+; WIN64-NEXT: movq %rsi, %rax
+; WIN64-NEXT: .seh_startepilogue
+; WIN64-NEXT: addq $176, %rsp
+; WIN64-NEXT: popq %rsi
+; WIN64-NEXT: .seh_endepilogue
+; WIN64-NEXT: retq
+; WIN64-NEXT: .seh_endproc
+ %buf = alloca [136 x i8]
+ call void @escape(ptr %buf)
+ ret ptr %p
+}
>From 02172e59485de9dc5a241a6efb83529568636c9b Mon Sep 17 00:00:00 2001
From: Andrew Gaul <andrew at gaul.org>
Date: Sun, 30 Aug 2026 21:19:23 -0700
Subject: [PATCH 2/2] [X86] Use 8-bit immediates for 128-byte stack adjustments
and LEA splits
ADD and SUB of +128 need a 32-bit immediate, but the equivalent
operation with -128 fits the sign-extended 8-bit form, three bytes
shorter. ISel has long done this through X86InstrCompiler.td patterns
(extended to flag-producing adds in c0e01d29a467), but two post-ISel
paths still emit the long form:
* X86FrameLowering::BuildStackAdjustment emits sub rsp, 128 and
add rsp, 128 for 128-byte adjustments. Flip them to add rsp, -128
and sub rsp, -128. EFLAGS is dead whenever this branch is taken,
since the code already prefers a flag-clobbering ADD/SUB over LEA on
that basis. Windows CFI targets keep the canonical spelling because
the Win64 unwinder recognizes an epilogue by disassembling forward
for add rsp, imm.
* X86FixupLEAs::processInstrForSlow3OpLEA splits a slow three-operand
LEA into a two-operand LEA plus an ADD of the displacement, guarded
on EFLAGS being provably dead; emit SUB of -128 when that
displacement is exactly 128.
Together these cover all 55 remaining oversized 128-immediate
encodings in uutils coreutils (48 stack adjustments, 7 LEA splits).
Found via x86lint.
Co-Authored-By: Claude Fable 5 <noreply at anthropic.com>
---
llvm/lib/Target/X86/X86FixupLEAs.cpp | 21 ++++++++++++++++
llvm/lib/Target/X86/X86FrameLowering.cpp | 20 +++++++++++++---
llvm/test/CodeGen/X86/arg-copy-elide.ll | 2 +-
llvm/test/CodeGen/X86/avx2-vbroadcast.ll | 16 ++++++-------
llvm/test/CodeGen/X86/avx512-calling-conv.ll | 4 ++--
.../test/CodeGen/X86/avx512-insert-extract.ll | 22 ++++++++---------
.../CodeGen/X86/avx512-insert-extract_i1.ll | 2 +-
llvm/test/CodeGen/X86/avx512-intel-ocl.ll | 6 ++---
.../avx512-shuffles/shuffle-chained-bf16.ll | 2 +-
llvm/test/CodeGen/X86/avx512fp16-mov.ll | 4 ++--
llvm/test/CodeGen/X86/frem.ll | 4 ++--
llvm/test/CodeGen/X86/i386-baseptr.ll | 2 +-
llvm/test/CodeGen/X86/lea-fixup-disp128.mir | 4 ++--
llvm/test/CodeGen/X86/nontemporal-loads-2.ll | 24 +++++++++----------
llvm/test/CodeGen/X86/sdiv_fix_sat.ll | 2 +-
llvm/test/CodeGen/X86/shift-i128.ll | 2 +-
.../X86/smulo-128-legalisation-lowering.ll | 4 ++--
llvm/test/CodeGen/X86/stack-adjust-128.ll | 4 ++--
llvm/test/CodeGen/X86/var-permute-512.ll | 12 +++++-----
.../CodeGen/X86/vec-strict-inttofp-512.ll | 4 ++--
llvm/test/CodeGen/X86/vector-compress.ll | 4 ++--
.../CodeGen/X86/vector-extract-last-active.ll | 2 +-
...lar-shift-by-byte-multiple-legalization.ll | 8 +++----
...ad-of-small-alloca-with-zero-upper-half.ll | 4 ++--
llvm/test/CodeGen/X86/x86-64-baseptr.ll | 2 +-
25 files changed, 108 insertions(+), 73 deletions(-)
diff --git a/llvm/lib/Target/X86/X86FixupLEAs.cpp b/llvm/lib/Target/X86/X86FixupLEAs.cpp
index aa5ecd94ed828..5c9c74e64aa73 100644
--- a/llvm/lib/Target/X86/X86FixupLEAs.cpp
+++ b/llvm/lib/Target/X86/X86FixupLEAs.cpp
@@ -385,6 +385,18 @@ static inline unsigned getADDriFromLEA(unsigned LEAOpcode,
}
}
+static inline unsigned getSUBriFromLEA(unsigned LEAOpcode) {
+ switch (LEAOpcode) {
+ default:
+ llvm_unreachable("Unexpected LEA instruction");
+ case X86::LEA32r:
+ case X86::LEA64_32r:
+ return X86::SUB32ri;
+ case X86::LEA64r:
+ return X86::SUB64ri32;
+ }
+}
+
static inline unsigned getINCDECFromLEA(unsigned LEAOpcode, bool IsINC) {
switch (LEAOpcode) {
default:
@@ -851,6 +863,15 @@ void FixupLEAsImpl::processInstrForSlow3OpLEA(MachineBasicBlock::iterator &I,
NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
.addReg(DestReg);
LLVM_DEBUG(NewMI->dump(););
+ } else if (Offset.isImm() && Offset.getImm() == 128) {
+ // ADD of +128 needs a 32-bit immediate, while SUB of -128 fits the
+ // sign-extended 8-bit form, three bytes shorter. EFLAGS was proved
+ // dead above, so the different flag results don't matter.
+ unsigned NewOpc = getSUBriFromLEA(MI.getOpcode());
+ NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
+ .addReg(DestReg)
+ .addImm(-128);
+ LLVM_DEBUG(NewMI->dump(););
} else {
unsigned NewOpc = getADDriFromLEA(MI.getOpcode(), Offset);
NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
diff --git a/llvm/lib/Target/X86/X86FrameLowering.cpp b/llvm/lib/Target/X86/X86FrameLowering.cpp
index 7251bdda1dd05..4cbed4c4510e2 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.cpp
+++ b/llvm/lib/Target/X86/X86FrameLowering.cpp
@@ -414,11 +414,25 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
StackPtr),
StackPtr, false, Offset);
} else {
- const unsigned Opc = IsSub ? getSUBriOpcode(Uses64BitFramePtr)
- : getADDriOpcode(Uses64BitFramePtr);
+ unsigned Opc = IsSub ? getSUBriOpcode(Uses64BitFramePtr)
+ : getADDriOpcode(Uses64BitFramePtr);
+ int64_t Imm = AbsOffset;
+ // Prefer `add rsp, -128` over `sub rsp, 128` (and vice versa in the
+ // epilogue): 128 is the one magnitude whose negation fits the
+ // sign-extended 8-bit immediate while the value itself does not, so the
+ // flipped operation is three bytes shorter. EFLAGS is dead here (this
+ // branch clobbers it anyway). Skip under Windows CFI: the Win64 unwinder
+ // recognizes an epilogue by disassembling forward for `add rsp, imm` and
+ // must not see the SUB spelling.
+ if (AbsOffset == 128 &&
+ !MBB.getParent()->getTarget().getMCAsmInfo().usesWindowsCFI()) {
+ Opc = IsSub ? getADDriOpcode(Uses64BitFramePtr)
+ : getSUBriOpcode(Uses64BitFramePtr);
+ Imm = -128;
+ }
MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
.addReg(StackPtr)
- .addImm(AbsOffset);
+ .addImm(Imm);
MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
}
return MI;
diff --git a/llvm/test/CodeGen/X86/arg-copy-elide.ll b/llvm/test/CodeGen/X86/arg-copy-elide.ll
index f13627b55856f..15edb612d7649 100644
--- a/llvm/test/CodeGen/X86/arg-copy-elide.ll
+++ b/llvm/test/CodeGen/X86/arg-copy-elide.ll
@@ -130,7 +130,7 @@ define void @high_alignment(i32 %x) {
; CHECK-NEXT: pushl %ebp
; CHECK-NEXT: movl %esp, %ebp
; CHECK-NEXT: andl $-128, %esp
-; CHECK-NEXT: subl $128, %esp
+; CHECK-NEXT: addl $-128, %esp
; CHECK-NEXT: movl 8(%ebp), %eax
; CHECK-NEXT: movl %eax, (%esp)
; CHECK-NEXT: movl %esp, %eax
diff --git a/llvm/test/CodeGen/X86/avx2-vbroadcast.ll b/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
index c50af6968f5bb..a5601f1725682 100644
--- a/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
+++ b/llvm/test/CodeGen/X86/avx2-vbroadcast.ll
@@ -1133,7 +1133,7 @@ define void @isel_crash_32b(ptr %cV_R.addr) {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: .cfi_def_cfa_register %ebp
; X86-NEXT: andl $-32, %esp
-; X86-NEXT: subl $128, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X86-NEXT: vmovaps %ymm0, (%esp)
@@ -1153,7 +1153,7 @@ define void @isel_crash_32b(ptr %cV_R.addr) {
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: .cfi_def_cfa_register %rbp
; X64-NEXT: andq $-32, %rsp
-; X64-NEXT: subq $128, %rsp
+; X64-NEXT: addq $-128, %rsp
; X64-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X64-NEXT: vmovaps %ymm0, (%rsp)
; X64-NEXT: vpbroadcastb (%rdi), %ymm1
@@ -1224,7 +1224,7 @@ define void @isel_crash_16w(ptr %cV_R.addr) {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: .cfi_def_cfa_register %ebp
; X86-NEXT: andl $-32, %esp
-; X86-NEXT: subl $128, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X86-NEXT: vmovaps %ymm0, (%esp)
@@ -1244,7 +1244,7 @@ define void @isel_crash_16w(ptr %cV_R.addr) {
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: .cfi_def_cfa_register %rbp
; X64-NEXT: andq $-32, %rsp
-; X64-NEXT: subq $128, %rsp
+; X64-NEXT: addq $-128, %rsp
; X64-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X64-NEXT: vmovaps %ymm0, (%rsp)
; X64-NEXT: vpbroadcastw (%rdi), %ymm1
@@ -1315,7 +1315,7 @@ define void @isel_crash_8d(ptr %cV_R.addr) {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: .cfi_def_cfa_register %ebp
; X86-NEXT: andl $-32, %esp
-; X86-NEXT: subl $128, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X86-NEXT: vmovaps %ymm0, (%esp)
@@ -1335,7 +1335,7 @@ define void @isel_crash_8d(ptr %cV_R.addr) {
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: .cfi_def_cfa_register %rbp
; X64-NEXT: andq $-32, %rsp
-; X64-NEXT: subq $128, %rsp
+; X64-NEXT: addq $-128, %rsp
; X64-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X64-NEXT: vmovaps %ymm0, (%rsp)
; X64-NEXT: vbroadcastss (%rdi), %ymm1
@@ -1405,7 +1405,7 @@ define void @isel_crash_4q(ptr %cV_R.addr) {
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: .cfi_def_cfa_register %ebp
; X86-NEXT: andl $-32, %esp
-; X86-NEXT: subl $128, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X86-NEXT: vmovaps %ymm0, (%esp)
@@ -1425,7 +1425,7 @@ define void @isel_crash_4q(ptr %cV_R.addr) {
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: .cfi_def_cfa_register %rbp
; X64-NEXT: andq $-32, %rsp
-; X64-NEXT: subq $128, %rsp
+; X64-NEXT: addq $-128, %rsp
; X64-NEXT: vxorps %xmm0, %xmm0, %xmm0
; X64-NEXT: vmovaps %ymm0, (%rsp)
; X64-NEXT: vbroadcastsd (%rdi), %ymm1
diff --git a/llvm/test/CodeGen/X86/avx512-calling-conv.ll b/llvm/test/CodeGen/X86/avx512-calling-conv.ll
index 0a19a36f5cb03..c6aa14a8d30b6 100644
--- a/llvm/test/CodeGen/X86/avx512-calling-conv.ll
+++ b/llvm/test/CodeGen/X86/avx512-calling-conv.ll
@@ -2116,7 +2116,7 @@ define void @v64i1_mem(<128 x i32> %x, <64 x i1> %y) {
; SKX-NEXT: movq %rsp, %rbp
; SKX-NEXT: .cfi_def_cfa_register %rbp
; SKX-NEXT: andq $-64, %rsp
-; SKX-NEXT: subq $128, %rsp
+; SKX-NEXT: addq $-128, %rsp
; SKX-NEXT: vmovaps 16(%rbp), %zmm8
; SKX-NEXT: vmovaps %zmm8, (%rsp)
; SKX-NEXT: callq _v64i1_mem_callee
@@ -2283,7 +2283,7 @@ define void @v64i1_mem(<128 x i32> %x, <64 x i1> %y) {
; FASTISEL-NEXT: movq %rsp, %rbp
; FASTISEL-NEXT: .cfi_def_cfa_register %rbp
; FASTISEL-NEXT: andq $-64, %rsp
-; FASTISEL-NEXT: subq $128, %rsp
+; FASTISEL-NEXT: addq $-128, %rsp
; FASTISEL-NEXT: vpsllw $7, 16(%rbp), %zmm8
; FASTISEL-NEXT: vpmovb2m %zmm8, %k0
; FASTISEL-NEXT: vpmovm2b %k0, %zmm8
diff --git a/llvm/test/CodeGen/X86/avx512-insert-extract.ll b/llvm/test/CodeGen/X86/avx512-insert-extract.ll
index e3e7cf6085907..4efc4678f1d00 100644
--- a/llvm/test/CodeGen/X86/avx512-insert-extract.ll
+++ b/llvm/test/CodeGen/X86/avx512-insert-extract.ll
@@ -101,7 +101,7 @@ define float @test7(<16 x float> %x, i32 %ind) nounwind {
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $15, %edi
@@ -120,7 +120,7 @@ define double @test8(<8 x double> %x, i32 %ind) nounwind {
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $7, %edi
@@ -158,7 +158,7 @@ define i32 @test10(<16 x i32> %x, i32 %ind) nounwind {
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $15, %edi
@@ -1148,7 +1148,7 @@ define i64 @test_extractelement_variable_v8i64(<8 x i64> %t1, i32 %index) nounwi
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $7, %edi
@@ -1198,7 +1198,7 @@ define double @test_extractelement_variable_v8f64(<8 x double> %t1, i32 %index)
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $7, %edi
@@ -1248,7 +1248,7 @@ define i32 @test_extractelement_variable_v16i32(<16 x i32> %t1, i32 %index) noun
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $15, %edi
@@ -1298,7 +1298,7 @@ define float @test_extractelement_variable_v16f32(<16 x float> %t1, i32 %index)
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $15, %edi
@@ -1348,7 +1348,7 @@ define i16 @test_extractelement_variable_v32i16(<32 x i16> %t1, i32 %index) noun
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $31, %edi
@@ -1399,7 +1399,7 @@ define i8 @test_extractelement_variable_v64i8(<64 x i8> %t1, i32 %index) nounwin
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: ## kill: def $edi killed $edi def $rdi
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: andl $63, %edi
@@ -1419,7 +1419,7 @@ define i8 @test_extractelement_variable_v64i8_indexi8(<64 x i8> %t1, i8 %index)
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: addb %dil, %dil
; CHECK-NEXT: vmovaps %zmm0, (%rsp)
; CHECK-NEXT: movzbl %dil, %eax
@@ -1663,7 +1663,7 @@ define i64 @test_insertelement_variable_v64i1(<64 x i8> %a, i8 %b, i32 %index) n
; KNL-NEXT: pushq %rbp
; KNL-NEXT: movq %rsp, %rbp
; KNL-NEXT: andq $-64, %rsp
-; KNL-NEXT: subq $128, %rsp
+; KNL-NEXT: addq $-128, %rsp
; KNL-NEXT: ## kill: def $esi killed $esi def $rsi
; KNL-NEXT: vpxor %xmm1, %xmm1, %xmm1
; KNL-NEXT: vextracti64x4 $1, %zmm0, %ymm2
diff --git a/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll b/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
index ee2bd96be099c..adb8bec6bb5bf 100644
--- a/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
+++ b/llvm/test/CodeGen/X86/avx512-insert-extract_i1.ll
@@ -12,7 +12,7 @@ define zeroext i8 @test_extractelement_varible_v64i1(<64 x i8> %a, <64 x i8> %b,
; SKX-NEXT: movq %rsp, %rbp
; SKX-NEXT: .cfi_def_cfa_register %rbp
; SKX-NEXT: andq $-64, %rsp
-; SKX-NEXT: subq $128, %rsp
+; SKX-NEXT: addq $-128, %rsp
; SKX-NEXT: ## kill: def $edi killed $edi def $rdi
; SKX-NEXT: vpcmpnleub %zmm1, %zmm0, %k0
; SKX-NEXT: vpmovm2b %k0, %zmm0
diff --git a/llvm/test/CodeGen/X86/avx512-intel-ocl.ll b/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
index eb8bff26f3b77..0fa67a264fcf0 100644
--- a/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
+++ b/llvm/test/CodeGen/X86/avx512-intel-ocl.ll
@@ -34,7 +34,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
; WIN32-NEXT: pushl %ebp
; WIN32-NEXT: movl %esp, %ebp
; WIN32-NEXT: andl $-64, %esp
-; WIN32-NEXT: subl $128, %esp
+; WIN32-NEXT: addl $-128, %esp
; WIN32-NEXT: vaddps %zmm1, %zmm0, %zmm0
; WIN32-NEXT: movl %esp, %eax
; WIN32-NEXT: pushl %eax
@@ -67,7 +67,7 @@ define <16 x float> @testf16_inp(<16 x float> %a, <16 x float> %b) nounwind {
; X64-NEXT: pushq %r13
; X64-NEXT: pushq %r12
; X64-NEXT: andq $-64, %rsp
-; X64-NEXT: subq $128, %rsp
+; X64-NEXT: addq $-128, %rsp
; X64-NEXT: vaddps %zmm1, %zmm0, %zmm0
; X64-NEXT: movq %rsp, %rdi
; X64-NEXT: pushq %rbp
@@ -150,7 +150,7 @@ define <16 x float> @testf16_regs(<16 x float> %a, <16 x float> %b) nounwind {
; X64-NEXT: pushq %r13
; X64-NEXT: pushq %r12
; X64-NEXT: andq $-64, %rsp
-; X64-NEXT: subq $128, %rsp
+; X64-NEXT: addq $-128, %rsp
; X64-NEXT: vmovaps %zmm1, %zmm16
; X64-NEXT: vaddps %zmm1, %zmm0, %zmm0
; X64-NEXT: movq %rsp, %rdi
diff --git a/llvm/test/CodeGen/X86/avx512-shuffles/shuffle-chained-bf16.ll b/llvm/test/CodeGen/X86/avx512-shuffles/shuffle-chained-bf16.ll
index f646f609d7e70..a7c79016334fa 100644
--- a/llvm/test/CodeGen/X86/avx512-shuffles/shuffle-chained-bf16.ll
+++ b/llvm/test/CodeGen/X86/avx512-shuffles/shuffle-chained-bf16.ll
@@ -12,7 +12,7 @@ define <2 x bfloat> @shuffle_chained_v32bf16_v2bf16(<32 x bfloat> %a) {
; CHECK-NEXT: movq %rsp, %rbp
; CHECK-NEXT: .cfi_def_cfa_register %rbp
; CHECK-NEXT: andq $-64, %rsp
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: vmovd {{.*#+}} xmm1 = [0,16,0,0,0,0,0,0]
; CHECK-NEXT: vpermw %zmm0, %zmm1, %zmm0
; CHECK-NEXT: # kill: def $xmm0 killed $xmm0 killed $zmm0
diff --git a/llvm/test/CodeGen/X86/avx512fp16-mov.ll b/llvm/test/CodeGen/X86/avx512fp16-mov.ll
index 79fc699f09932..e2f2688b1d9f3 100644
--- a/llvm/test/CodeGen/X86/avx512fp16-mov.ll
+++ b/llvm/test/CodeGen/X86/avx512fp16-mov.ll
@@ -1640,7 +1640,7 @@ define half @extract_f16_8(<32 x half> %x, i64 %idx) nounwind {
; X64-NEXT: pushq %rbp
; X64-NEXT: movq %rsp, %rbp
; X64-NEXT: andq $-64, %rsp
-; X64-NEXT: subq $128, %rsp
+; X64-NEXT: addq $-128, %rsp
; X64-NEXT: andl $31, %edi
; X64-NEXT: vmovaps %zmm0, (%rsp)
; X64-NEXT: vmovsh {{.*#+}} xmm0 = mem[0],zero,zero,zero,zero,zero,zero,zero
@@ -1654,7 +1654,7 @@ define half @extract_f16_8(<32 x half> %x, i64 %idx) nounwind {
; X86-NEXT: pushl %ebp
; X86-NEXT: movl %esp, %ebp
; X86-NEXT: andl $-64, %esp
-; X86-NEXT: subl $128, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movl 8(%ebp), %eax
; X86-NEXT: andl $31, %eax
; X86-NEXT: vmovaps %zmm0, (%esp)
diff --git a/llvm/test/CodeGen/X86/frem.ll b/llvm/test/CodeGen/X86/frem.ll
index 959265d08299a..c259a1c65d9e5 100644
--- a/llvm/test/CodeGen/X86/frem.ll
+++ b/llvm/test/CodeGen/X86/frem.ll
@@ -1397,7 +1397,7 @@ define void @frem_v4f80(<4 x x86_fp80> %a0, <4 x x86_fp80> %a1, ptr%p3) nounwind
; CHECK-LABEL: frem_v4f80:
; CHECK: # %bb.0:
; CHECK-NEXT: pushq %rbx
-; CHECK-NEXT: subq $128, %rsp
+; CHECK-NEXT: addq $-128, %rsp
; CHECK-NEXT: movq %rdi, %rbx
; CHECK-NEXT: fldt {{[0-9]+}}(%rsp)
; CHECK-NEXT: fstpt {{[-0-9]+}}(%r{{[sb]}}p) # 10-byte Folded Spill
@@ -1441,7 +1441,7 @@ define void @frem_v4f80(<4 x x86_fp80> %a0, <4 x x86_fp80> %a1, ptr%p3) nounwind
; CHECK-NEXT: fstpt 10(%rbx)
; CHECK-NEXT: fldt {{[-0-9]+}}(%r{{[sb]}}p) # 10-byte Folded Reload
; CHECK-NEXT: fstpt (%rbx)
-; CHECK-NEXT: addq $128, %rsp
+; CHECK-NEXT: subq $-128, %rsp
; CHECK-NEXT: popq %rbx
; CHECK-NEXT: retq
%frem = frem <4 x x86_fp80> %a0, %a1
diff --git a/llvm/test/CodeGen/X86/i386-baseptr.ll b/llvm/test/CodeGen/X86/i386-baseptr.ll
index 777eb838b84cc..01702cbeade50 100644
--- a/llvm/test/CodeGen/X86/i386-baseptr.ll
+++ b/llvm/test/CodeGen/X86/i386-baseptr.ll
@@ -97,7 +97,7 @@ define x86_regcallcc void @clobber_baseptr_argptr(i32 %param1, i32 %param2, i32
; CHECK-NEXT: .cfi_def_cfa_register %ebp
; CHECK-NEXT: pushl %ebx
; CHECK-NEXT: andl $-128, %esp
-; CHECK-NEXT: subl $128, %esp
+; CHECK-NEXT: addl $-128, %esp
; CHECK-NEXT: movl %esp, %esi
; CHECK-NEXT: .cfi_offset %ebx, -12
; CHECK-NEXT: movl 8(%ebp), %edi
diff --git a/llvm/test/CodeGen/X86/lea-fixup-disp128.mir b/llvm/test/CodeGen/X86/lea-fixup-disp128.mir
index 8c9088882af74..653bd2b58da17 100644
--- a/llvm/test/CodeGen/X86/lea-fixup-disp128.mir
+++ b/llvm/test/CodeGen/X86/lea-fixup-disp128.mir
@@ -25,7 +25,7 @@ body: |
bb.0:
; CHECK-LABEL: name: disp128
; CHECK: renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 0, $noreg
- ; CHECK-NEXT: $rax = ADD64ri32 $rax, 128, implicit-def $eflags
+ ; CHECK-NEXT: $rax = SUB64ri32 $rax, -128, implicit-def $eflags
; CHECK-NEXT: JMP64r renamable $rax
renamable $rax = LEA64r renamable $rdi, 1, renamable $rsi, 128, $noreg
JMP64r renamable $rax
@@ -36,7 +36,7 @@ body: |
bb.0:
; CHECK-LABEL: name: disp128_32
; CHECK: renamable $eax = LEA64_32r renamable $rdi, 1, renamable $rsi, 0, $noreg
- ; CHECK-NEXT: $eax = ADD32ri $eax, 128, implicit-def $eflags
+ ; CHECK-NEXT: $eax = SUB32ri $eax, -128, implicit-def $eflags
; CHECK-NEXT: RET64 implicit $eax
renamable $eax = LEA64_32r renamable $rdi, 1, renamable $rsi, 128, $noreg
RET64 implicit $eax
diff --git a/llvm/test/CodeGen/X86/nontemporal-loads-2.ll b/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
index 28ddfe5ab62dc..71c23d88085ec 100644
--- a/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
+++ b/llvm/test/CodeGen/X86/nontemporal-loads-2.ll
@@ -610,7 +610,7 @@ define <8 x double> @test_v8f64_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -689,7 +689,7 @@ define <16 x float> @test_v16f32_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -768,7 +768,7 @@ define <8 x i64> @test_v8i64_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -847,7 +847,7 @@ define <16 x i32> @test_v16i32_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -926,7 +926,7 @@ define <32 x i16> @test_v32i16_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -1005,7 +1005,7 @@ define <64 x i8> @test_v64i8_align16(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 48(%rdi), %xmm0
; AVX512-NEXT: vmovdqa %xmm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa 32(%rdi), %xmm0
@@ -1060,7 +1060,7 @@ define <8 x double> @test_v8f64_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
@@ -1111,7 +1111,7 @@ define <16 x float> @test_v16f32_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
@@ -1162,7 +1162,7 @@ define <8 x i64> @test_v8i64_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
@@ -1213,7 +1213,7 @@ define <16 x i32> @test_v16i32_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
@@ -1264,7 +1264,7 @@ define <32 x i16> @test_v32i16_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
@@ -1315,7 +1315,7 @@ define <64 x i8> @test_v64i8_align32(ptr %src) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vmovntdqa 32(%rdi), %ymm0
; AVX512-NEXT: vmovdqa %ymm0, {{[0-9]+}}(%rsp)
; AVX512-NEXT: vmovntdqa (%rdi), %ymm0
diff --git a/llvm/test/CodeGen/X86/sdiv_fix_sat.ll b/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
index e7d41d5bebc8a..eac2bb7ebe962 100644
--- a/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
+++ b/llvm/test/CodeGen/X86/sdiv_fix_sat.ll
@@ -370,7 +370,7 @@ define i64 @func5(i64 %x, i64 %y) nounwind {
; X86-NEXT: pushl %edi
; X86-NEXT: pushl %esi
; X86-NEXT: andl $-16, %esp
-; X86-NEXT: subl $128, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: movl 8(%ebp), %esi
; X86-NEXT: movl 12(%ebp), %edi
; X86-NEXT: movl 16(%ebp), %ecx
diff --git a/llvm/test/CodeGen/X86/shift-i128.ll b/llvm/test/CodeGen/X86/shift-i128.ll
index 73755f5b2a5ee..6c9cdb1b62ca4 100644
--- a/llvm/test/CodeGen/X86/shift-i128.ll
+++ b/llvm/test/CodeGen/X86/shift-i128.ll
@@ -538,7 +538,7 @@ define void @test_shl_v2i128(<2 x i128> %x, <2 x i128> %a, ptr nocapture %r) nou
; i686-NEXT: pushl %edi
; i686-NEXT: pushl %esi
; i686-NEXT: andl $-16, %esp
-; i686-NEXT: subl $128, %esp
+; i686-NEXT: addl $-128, %esp
; i686-NEXT: movl 40(%ebp), %edi
; i686-NEXT: movl 24(%ebp), %eax
; i686-NEXT: movl 28(%ebp), %ecx
diff --git a/llvm/test/CodeGen/X86/smulo-128-legalisation-lowering.ll b/llvm/test/CodeGen/X86/smulo-128-legalisation-lowering.ll
index 13596e1b18768..6e09f66980aa2 100644
--- a/llvm/test/CodeGen/X86/smulo-128-legalisation-lowering.ll
+++ b/llvm/test/CodeGen/X86/smulo-128-legalisation-lowering.ll
@@ -523,7 +523,7 @@ define zeroext i1 @smuloi256(i256 %v1, i256 %v2, ptr %res) {
; X86-NEXT: .cfi_def_cfa_offset 16
; X86-NEXT: pushl %esi
; X86-NEXT: .cfi_def_cfa_offset 20
-; X86-NEXT: subl $128, %esp
+; X86-NEXT: addl $-128, %esp
; X86-NEXT: .cfi_def_cfa_offset 148
; X86-NEXT: .cfi_offset %esi, -20
; X86-NEXT: .cfi_offset %edi, -16
@@ -1280,7 +1280,7 @@ define zeroext i1 @smuloi256(i256 %v1, i256 %v2, ptr %res) {
; X86-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx ## 4-byte Reload
; X86-NEXT: movl %ecx, 24(%eax)
; X86-NEXT: setne %al
-; X86-NEXT: addl $128, %esp
+; X86-NEXT: subl $-128, %esp
; X86-NEXT: popl %esi
; X86-NEXT: popl %edi
; X86-NEXT: popl %ebx
diff --git a/llvm/test/CodeGen/X86/stack-adjust-128.ll b/llvm/test/CodeGen/X86/stack-adjust-128.ll
index a19631248e7d6..15cee691a4e69 100644
--- a/llvm/test/CodeGen/X86/stack-adjust-128.ll
+++ b/llvm/test/CodeGen/X86/stack-adjust-128.ll
@@ -17,14 +17,14 @@ define ptr @frame128(ptr %p) {
; LINUX: # %bb.0:
; LINUX-NEXT: pushq %rbx
; LINUX-NEXT: .cfi_def_cfa_offset 16
-; LINUX-NEXT: subq $128, %rsp
+; LINUX-NEXT: addq $-128, %rsp
; LINUX-NEXT: .cfi_def_cfa_offset 144
; LINUX-NEXT: .cfi_offset %rbx, -16
; LINUX-NEXT: movq %rdi, %rbx
; LINUX-NEXT: leaq {{[0-9]+}}(%rsp), %rdi
; LINUX-NEXT: callq escape at PLT
; LINUX-NEXT: movq %rbx, %rax
-; LINUX-NEXT: addq $128, %rsp
+; LINUX-NEXT: subq $-128, %rsp
; LINUX-NEXT: .cfi_def_cfa_offset 16
; LINUX-NEXT: popq %rbx
; LINUX-NEXT: .cfi_def_cfa_offset 8
diff --git a/llvm/test/CodeGen/X86/var-permute-512.ll b/llvm/test/CodeGen/X86/var-permute-512.ll
index 0a3b9fa5947db..6b893f834f50d 100644
--- a/llvm/test/CodeGen/X86/var-permute-512.ll
+++ b/llvm/test/CodeGen/X86/var-permute-512.ll
@@ -97,7 +97,7 @@ define <32 x i16> @var_shuffle_v32i16(<32 x i16> %v, <32 x i16> %indices) nounwi
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-64, %rsp
-; AVX512F-NEXT: subq $128, %rsp
+; AVX512F-NEXT: addq $-128, %rsp
; AVX512F-NEXT: vextracti128 $1, %ymm1, %xmm2
; AVX512F-NEXT: vextracti32x4 $2, %zmm1, %xmm3
; AVX512F-NEXT: vextracti32x4 $3, %zmm1, %xmm4
@@ -326,7 +326,7 @@ define <64 x i8> @var_shuffle_v64i8(<64 x i8> %v, <64 x i8> %indices) nounwind {
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-64, %rsp
-; AVX512F-NEXT: subq $128, %rsp
+; AVX512F-NEXT: addq $-128, %rsp
; AVX512F-NEXT: vextracti128 $1, %ymm1, %xmm2
; AVX512F-NEXT: vextracti32x4 $2, %zmm1, %xmm3
; AVX512F-NEXT: vextracti32x4 $3, %zmm1, %xmm4
@@ -551,7 +551,7 @@ define <64 x i8> @var_shuffle_v64i8(<64 x i8> %v, <64 x i8> %indices) nounwind {
; AVX512BW-NEXT: pushq %rbp
; AVX512BW-NEXT: movq %rsp, %rbp
; AVX512BW-NEXT: andq $-64, %rsp
-; AVX512BW-NEXT: subq $128, %rsp
+; AVX512BW-NEXT: addq $-128, %rsp
; AVX512BW-NEXT: vextracti128 $1, %ymm1, %xmm2
; AVX512BW-NEXT: vextracti32x4 $2, %zmm1, %xmm3
; AVX512BW-NEXT: vextracti32x4 $3, %zmm1, %xmm4
@@ -1064,7 +1064,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-64, %rsp
-; AVX512F-NEXT: subq $128, %rsp
+; AVX512F-NEXT: addq $-128, %rsp
; AVX512F-NEXT: # kill: def $esi killed $esi def $rsi
; AVX512F-NEXT: vpbroadcastd %esi, %zmm2
; AVX512F-NEXT: vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm1 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
@@ -1315,7 +1315,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
; AVX512BW-NEXT: pushq %rbp
; AVX512BW-NEXT: movq %rsp, %rbp
; AVX512BW-NEXT: andq $-64, %rsp
-; AVX512BW-NEXT: subq $128, %rsp
+; AVX512BW-NEXT: addq $-128, %rsp
; AVX512BW-NEXT: # kill: def $esi killed $esi def $rsi
; AVX512BW-NEXT: vpbroadcastd %esi, %zmm2
; AVX512BW-NEXT: vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm1 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
@@ -1566,7 +1566,7 @@ define void @var_cvt_shuffle_v64f32_v64i8_idx(ptr %dst, <64 x i8> %src, i32 %b)
; AVX512VBMI-NEXT: pushq %rbp
; AVX512VBMI-NEXT: movq %rsp, %rbp
; AVX512VBMI-NEXT: andq $-64, %rsp
-; AVX512VBMI-NEXT: subq $128, %rsp
+; AVX512VBMI-NEXT: addq $-128, %rsp
; AVX512VBMI-NEXT: # kill: def $esi killed $esi def $rsi
; AVX512VBMI-NEXT: vpbroadcastd %esi, %zmm1
; AVX512VBMI-NEXT: vpaddd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm1, %zmm2 # [10,9,8,7,6,5,4,3,2,1,u,4294967295,4294967294,4294967293,4294967292,4294967291]
diff --git a/llvm/test/CodeGen/X86/vec-strict-inttofp-512.ll b/llvm/test/CodeGen/X86/vec-strict-inttofp-512.ll
index bf60f0631749d..0b83512b8e688 100644
--- a/llvm/test/CodeGen/X86/vec-strict-inttofp-512.ll
+++ b/llvm/test/CodeGen/X86/vec-strict-inttofp-512.ll
@@ -270,7 +270,7 @@ define <8 x double> @sitofp_v8i64_v8f64(<8 x i64> %x) #0 {
; NODQ-32-NEXT: movl %esp, %ebp
; NODQ-32-NEXT: .cfi_def_cfa_register %ebp
; NODQ-32-NEXT: andl $-8, %esp
-; NODQ-32-NEXT: subl $128, %esp
+; NODQ-32-NEXT: addl $-128, %esp
; NODQ-32-NEXT: vextractf32x4 $2, %zmm0, %xmm1
; NODQ-32-NEXT: vmovlps %xmm1, {{[0-9]+}}(%esp)
; NODQ-32-NEXT: vshufps {{.*#+}} xmm1 = xmm1[2,3,2,3]
@@ -368,7 +368,7 @@ define <8 x double> @uitofp_v8i64_v8f64(<8 x i64> %x) #0 {
; NODQ-32-NEXT: movl %esp, %ebp
; NODQ-32-NEXT: .cfi_def_cfa_register %ebp
; NODQ-32-NEXT: andl $-8, %esp
-; NODQ-32-NEXT: subl $128, %esp
+; NODQ-32-NEXT: addl $-128, %esp
; NODQ-32-NEXT: vextractf32x4 $2, %zmm0, %xmm3
; NODQ-32-NEXT: vmovlps %xmm3, {{[0-9]+}}(%esp)
; NODQ-32-NEXT: vshufps {{.*#+}} xmm1 = xmm3[2,3,2,3]
diff --git a/llvm/test/CodeGen/X86/vector-compress.ll b/llvm/test/CodeGen/X86/vector-compress.ll
index 39ef06af99340..7f19806114894 100644
--- a/llvm/test/CodeGen/X86/vector-compress.ll
+++ b/llvm/test/CodeGen/X86/vector-compress.ll
@@ -2778,7 +2778,7 @@ define <64 x i8> @test_compress_v64i8(<64 x i8> %vec, <64 x i1> %mask, <64 x i8>
; AVX512VL-ONLY-NEXT: pushq %rbp
; AVX512VL-ONLY-NEXT: movq %rsp, %rbp
; AVX512VL-ONLY-NEXT: andq $-64, %rsp
-; AVX512VL-ONLY-NEXT: subq $128, %rsp
+; AVX512VL-ONLY-NEXT: addq $-128, %rsp
; AVX512VL-ONLY-NEXT: vpsllw $7, %zmm1, %zmm1
; AVX512VL-ONLY-NEXT: vpmovb2m %zmm1, %k6
; AVX512VL-ONLY-NEXT: kshiftrq $63, %k6, %k0
@@ -3575,7 +3575,7 @@ define <32 x i16> @test_compress_v32i16(<32 x i16> %vec, <32 x i1> %mask, <32 x
; AVX512F-NEXT: pushq %rbp
; AVX512F-NEXT: movq %rsp, %rbp
; AVX512F-NEXT: andq $-64, %rsp
-; AVX512F-NEXT: subq $128, %rsp
+; AVX512F-NEXT: addq $-128, %rsp
; AVX512F-NEXT: vextracti128 $1, %ymm1, %xmm4
; AVX512F-NEXT: vpmovzxbw {{.*#+}} ymm3 = xmm4[0],zero,xmm4[1],zero,xmm4[2],zero,xmm4[3],zero,xmm4[4],zero,xmm4[5],zero,xmm4[6],zero,xmm4[7],zero,xmm4[8],zero,xmm4[9],zero,xmm4[10],zero,xmm4[11],zero,xmm4[12],zero,xmm4[13],zero,xmm4[14],zero,xmm4[15],zero
; AVX512F-NEXT: vpmovsxbd %xmm4, %zmm4
diff --git a/llvm/test/CodeGen/X86/vector-extract-last-active.ll b/llvm/test/CodeGen/X86/vector-extract-last-active.ll
index bd6f35df29490..16a2de9783893 100644
--- a/llvm/test/CodeGen/X86/vector-extract-last-active.ll
+++ b/llvm/test/CodeGen/X86/vector-extract-last-active.ll
@@ -583,7 +583,7 @@ define i32 @extract_last_active_v16i32(<16 x i32> %a, <16 x i1> %c) nounwind {
; AVX512-NEXT: pushq %rbp
; AVX512-NEXT: movq %rsp, %rbp
; AVX512-NEXT: andq $-64, %rsp
-; AVX512-NEXT: subq $128, %rsp
+; AVX512-NEXT: addq $-128, %rsp
; AVX512-NEXT: vpsllw $7, %xmm1, %xmm1
; AVX512-NEXT: vpmovb2m %xmm1, %k1
; AVX512-NEXT: vmovdqa64 %zmm0, (%rsp)
diff --git a/llvm/test/CodeGen/X86/wide-scalar-shift-by-byte-multiple-legalization.ll b/llvm/test/CodeGen/X86/wide-scalar-shift-by-byte-multiple-legalization.ll
index e6e20d03c4138..4abf5ccbd81ed 100644
--- a/llvm/test/CodeGen/X86/wide-scalar-shift-by-byte-multiple-legalization.ll
+++ b/llvm/test/CodeGen/X86/wide-scalar-shift-by-byte-multiple-legalization.ll
@@ -19940,7 +19940,7 @@ define void @ashr_64bytes_qwordOff(ptr %src.ptr, ptr %qwordOff.ptr, ptr %dst) no
; X86-SSE42-NEXT: pushl %ebx
; X86-SSE42-NEXT: pushl %edi
; X86-SSE42-NEXT: pushl %esi
-; X86-SSE42-NEXT: subl $128, %esp
+; X86-SSE42-NEXT: addl $-128, %esp
; X86-SSE42-NEXT: movl {{[0-9]+}}(%esp), %eax
; X86-SSE42-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X86-SSE42-NEXT: movl {{[0-9]+}}(%esp), %edx
@@ -19985,7 +19985,7 @@ define void @ashr_64bytes_qwordOff(ptr %src.ptr, ptr %qwordOff.ptr, ptr %dst) no
; X86-SSE42-NEXT: movups %xmm2, 32(%eax)
; X86-SSE42-NEXT: movups %xmm1, 16(%eax)
; X86-SSE42-NEXT: movups %xmm0, (%eax)
-; X86-SSE42-NEXT: addl $128, %esp
+; X86-SSE42-NEXT: subl $-128, %esp
; X86-SSE42-NEXT: popl %esi
; X86-SSE42-NEXT: popl %edi
; X86-SSE42-NEXT: popl %ebx
@@ -19996,7 +19996,7 @@ define void @ashr_64bytes_qwordOff(ptr %src.ptr, ptr %qwordOff.ptr, ptr %dst) no
; X86-AVX1-NEXT: pushl %ebx
; X86-AVX1-NEXT: pushl %edi
; X86-AVX1-NEXT: pushl %esi
-; X86-AVX1-NEXT: subl $128, %esp
+; X86-AVX1-NEXT: addl $-128, %esp
; X86-AVX1-NEXT: movl {{[0-9]+}}(%esp), %eax
; X86-AVX1-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X86-AVX1-NEXT: movl {{[0-9]+}}(%esp), %edx
@@ -20039,7 +20039,7 @@ define void @ashr_64bytes_qwordOff(ptr %src.ptr, ptr %qwordOff.ptr, ptr %dst) no
; X86-AVX1-NEXT: vmovups %xmm2, 32(%eax)
; X86-AVX1-NEXT: vmovups %xmm1, 16(%eax)
; X86-AVX1-NEXT: vmovups %xmm0, (%eax)
-; X86-AVX1-NEXT: addl $128, %esp
+; X86-AVX1-NEXT: subl $-128, %esp
; X86-AVX1-NEXT: popl %esi
; X86-AVX1-NEXT: popl %edi
; X86-AVX1-NEXT: popl %ebx
diff --git a/llvm/test/CodeGen/X86/widen-load-of-small-alloca-with-zero-upper-half.ll b/llvm/test/CodeGen/X86/widen-load-of-small-alloca-with-zero-upper-half.ll
index fde915247760a..dedf5b4150908 100644
--- a/llvm/test/CodeGen/X86/widen-load-of-small-alloca-with-zero-upper-half.ll
+++ b/llvm/test/CodeGen/X86/widen-load-of-small-alloca-with-zero-upper-half.ll
@@ -2488,7 +2488,7 @@ define void @load_8byte_chunk_of_64byte_alloca_with_zero_upper_half(ptr %src, i6
; X86-SHLD-NEXT: pushl %ebx
; X86-SHLD-NEXT: pushl %edi
; X86-SHLD-NEXT: pushl %esi
-; X86-SHLD-NEXT: subl $128, %esp
+; X86-SHLD-NEXT: addl $-128, %esp
; X86-SHLD-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X86-SHLD-NEXT: movl {{[0-9]+}}(%esp), %eax
; X86-SHLD-NEXT: movl {{[0-9]+}}(%esp), %edx
@@ -2516,7 +2516,7 @@ define void @load_8byte_chunk_of_64byte_alloca_with_zero_upper_half(ptr %src, i6
; X86-SHLD-NEXT: shrdl %cl, %esi, %edx
; X86-SHLD-NEXT: movl %ebx, 4(%eax)
; X86-SHLD-NEXT: movl %edx, (%eax)
-; X86-SHLD-NEXT: addl $128, %esp
+; X86-SHLD-NEXT: subl $-128, %esp
; X86-SHLD-NEXT: popl %esi
; X86-SHLD-NEXT: popl %edi
; X86-SHLD-NEXT: popl %ebx
diff --git a/llvm/test/CodeGen/X86/x86-64-baseptr.ll b/llvm/test/CodeGen/X86/x86-64-baseptr.ll
index 62c63d5defe60..4f366ba3144a8 100644
--- a/llvm/test/CodeGen/X86/x86-64-baseptr.ll
+++ b/llvm/test/CodeGen/X86/x86-64-baseptr.ll
@@ -116,7 +116,7 @@ define void @clobber_base() #0 {
; X32ABI-NEXT: .cfi_def_cfa_register %rbp
; X32ABI-NEXT: pushq %rbx
; X32ABI-NEXT: andl $-128, %esp
-; X32ABI-NEXT: subl $128, %esp
+; X32ABI-NEXT: addl $-128, %esp
; X32ABI-NEXT: movl %esp, %ebx
; X32ABI-NEXT: .cfi_offset %rbx, -24
; X32ABI-NEXT: callq helper at PLT
More information about the llvm-commits
mailing list