[llvm] [X86][APX] Implement push+push2+push pre-alignment strategy for PP2 (PR #205031)
Feng Zou via llvm-commits
llvm-commits at lists.llvm.org
Tue Jun 23 21:25:30 PDT 2026
https://github.com/fzou1 updated https://github.com/llvm/llvm-project/pull/205031
>From ce30358cf3a9fb7ad954050ffb284f0bfe72ee7b Mon Sep 17 00:00:00 2001
From: "Zou, Feng" <feng.zou at intel.com>
Date: Mon, 22 Jun 2026 05:53:35 +0200
Subject: [PATCH 1/3] [X86][APX] Implement push+push2+push pre-alignment
strategy for PP2
Replace the dummy "push %rax" stack-alignment padding for APX push2/pop2
(PP2) with a push+push2+push strategy: when an even number of callee-saved
GPRs is involved, a single CSR push provides the 16-byte alignment instead
of a throwaway push %rax, and the remaining registers use push2/pop2. The
padForPush2Pop2 flag and its associated dummy push, SUB/LEA padding, and
SEH_StackAlloc emission in spill/restoreCalleeSavedRegisters are removed.
BuildStackAdjustment now uses NF (no-flags) variants of ADD/SUB when the
target supports them, avoiding gratuitous EFLAGS clobber. NF is disabled on
Win64 prologues because the OS epilogue unwinder does not recognize
EVEX-encoded instructions; those fall back to LEA, which it does recognize.
ADD64ri32_NF/SUB64ri32_NF are recognized in mergeSPUpdates and the epilogue
backward scan.
Fix the epilogue DWARF CFI loop to emit .cfi_def_cfa_offset for FrameDestroy
ADD/LEA/NF stack adjustments (previously only POP opcodes were handled),
reading the actual immediate from the instruction rather than assuming a
fixed amount.
Update LIT tests accordingly.
Assisted-by: Claude Opus 4.8 (1M context) <noreply at anthropic.com>
---
llvm/lib/Target/X86/X86FrameLowering.cpp | 64 ++---
llvm/test/CodeGen/X86/apx/add.ll | 2 +-
.../apx/pp2-with-stack-clash-protection.ll | 36 +--
.../CodeGen/X86/apx/push2-pop2-cfi-seh-v3.ll | 80 +++---
.../CodeGen/X86/apx/push2-pop2-cfi-seh.ll | 159 +++++++-----
llvm/test/CodeGen/X86/apx/push2-pop2.ll | 231 +++++++++++-------
llvm/test/CodeGen/X86/apx/sub.ll | 2 +-
llvm/test/CodeGen/X86/apx/win64-abi.ll | 4 +-
.../X86/win64-eh-unwindv3-push2pop2.ll | 32 +--
9 files changed, 352 insertions(+), 258 deletions(-)
diff --git a/llvm/lib/Target/X86/X86FrameLowering.cpp b/llvm/lib/Target/X86/X86FrameLowering.cpp
index 8fe09bea456c8..b703fc24b6a6f 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.cpp
+++ b/llvm/lib/Target/X86/X86FrameLowering.cpp
@@ -383,7 +383,11 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
}
MachineInstrBuilder MI;
- if (UseLEA) {
+ // Use NF (no-flags) variants when the target supports it, avoiding
+ // gratuitous EFLAGS clobber. Not on Windows where the OS epilogue unwinder
+ // doesn't recognize EVEX-encoded instructions.
+ bool UseNF = STI.hasNF() && !isWin64Prologue(*MBB.getParent());
+ if (UseLEA && !UseNF) {
MI = addRegOffset(BuildMI(MBB, MBBI, DL,
TII.get(getLEArOpcode(Uses64BitFramePtr)),
StackPtr),
@@ -391,12 +395,16 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
} else {
bool IsSub = Offset < 0;
uint64_t AbsOffset = IsSub ? -Offset : Offset;
- const unsigned Opc = IsSub ? getSUBriOpcode(Uses64BitFramePtr)
- : getADDriOpcode(Uses64BitFramePtr);
+ const unsigned Opc =
+ IsSub ? (UseNF ? (unsigned)X86::SUB64ri32_NF
+ : getSUBriOpcode(Uses64BitFramePtr))
+ : (UseNF ? (unsigned)X86::ADD64ri32_NF
+ : getADDriOpcode(Uses64BitFramePtr));
MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
.addReg(StackPtr)
.addImm(AbsOffset);
- MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
+ if (!UseNF)
+ MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
}
return MI;
}
@@ -432,7 +440,8 @@ int64_t X86FrameLowering::mergeSPUpdates(MachineBasicBlock &MBB,
for (;;) {
unsigned Opc = PI->getOpcode();
- if ((Opc == X86::ADD64ri32 || Opc == X86::ADD32ri) &&
+ if ((Opc == X86::ADD64ri32 || Opc == X86::ADD32ri ||
+ Opc == X86::ADD64ri32_NF) &&
PI->getOperand(0).getReg() == StackPtr) {
assert(PI->getOperand(1).getReg() == StackPtr);
Offset = PI->getOperand(2).getImm();
@@ -444,7 +453,8 @@ int64_t X86FrameLowering::mergeSPUpdates(MachineBasicBlock &MBB,
PI->getOperand(5).getReg() == X86::NoRegister) {
// For LEAs we have: def = lea SP, FI, noreg, Offset, noreg.
Offset = PI->getOperand(4).getImm();
- } else if ((Opc == X86::SUB64ri32 || Opc == X86::SUB32ri) &&
+ } else if ((Opc == X86::SUB64ri32 || Opc == X86::SUB32ri ||
+ Opc == X86::SUB64ri32_NF) &&
PI->getOperand(0).getReg() == StackPtr) {
assert(PI->getOperand(1).getReg() == StackPtr);
Offset = -PI->getOperand(2).getImm();
@@ -2625,7 +2635,8 @@ void X86FrameLowering::emitEpilogue(MachineFunction &MF,
(Opc != X86::POP32r && Opc != X86::POP64r && Opc != X86::BTR64ri8 &&
Opc != X86::ADD64ri32 && Opc != X86::POPP64r && Opc != X86::POP2 &&
Opc != X86::POP2P && Opc != X86::LEA64r && Opc != X86::SEH_PushReg &&
- Opc != X86::SEH_Push2Regs && Opc != X86::SEH_StackAlloc))
+ Opc != X86::SEH_Push2Regs && Opc != X86::SEH_StackAlloc &&
+ Opc != X86::ADD64ri32_NF))
break;
FirstCSPop = PI;
}
@@ -2772,6 +2783,17 @@ void X86FrameLowering::emitEpilogue(MachineFunction &MF,
BuildCFI(MBB, MBBI, DL,
MCCFIInstruction::cfiDefCfaOffset(nullptr, -Offset),
MachineInstr::FrameDestroy);
+ } else if (PI->getFlag(MachineInstr::FrameDestroy) &&
+ PI->getOperand(0).getReg() == StackPtr &&
+ (Opc == X86::ADD64ri32 || Opc == X86::ADD64ri32_NF ||
+ Opc == X86::ADD32ri || Opc == X86::LEA64r)) {
+ int64_t SPAdj = Opc == X86::LEA64r
+ ? PI->getOperand(4).getImm()
+ : PI->getOperand(2).getImm();
+ Offset += SPAdj;
+ BuildCFI(MBB, MBBI, DL,
+ MCCFIInstruction::cfiDefCfaOffset(nullptr, -Offset),
+ MachineInstr::FrameDestroy);
}
}
}
@@ -3029,6 +3051,7 @@ bool X86FrameLowering::assignCalleeSavedSpillSlots(
}
}
+ bool IsFPRemovedFromCSI = false;
if (hasFP(MF)) {
// emitPrologue always spills frame register the first thing.
SpillSlotOffset -= SlotSize;
@@ -3049,6 +3072,7 @@ bool X86FrameLowering::assignCalleeSavedSpillSlots(
for (unsigned i = 0; i < CSI.size(); ++i) {
if (TRI->regsOverlap(CSI[i].getReg(), FPReg)) {
CSI.erase(CSI.begin() + i);
+ IsFPRemovedFromCSI = true;
break;
}
}
@@ -3069,14 +3093,11 @@ bool X86FrameLowering::assignCalleeSavedSpillSlots(
unsigned NumCSGPR = llvm::count_if(CSI, [](const CalleeSavedInfo &I) {
return X86::GR64RegClass.contains(I.getReg());
});
- bool NeedPadding = (SpillSlotOffset % 16 != 0) && (NumCSGPR % 2 == 0);
- bool UsePush2Pop2 = NeedPadding ? NumCSGPR > 2 : NumCSGPR > 1;
- X86FI->setPadForPush2Pop2(NeedPadding && UsePush2Pop2);
- NumRegsForPush2 = UsePush2Pop2 ? alignDown(NumCSGPR, 2) : 0;
- if (X86FI->padForPush2Pop2()) {
- SpillSlotOffset -= SlotSize;
- MFI.CreateFixedSpillStackObject(SlotSize, SpillSlotOffset);
- }
+ bool UsePush2Pop2 = !IsFPRemovedFromCSI ? NumCSGPR > 2 : NumCSGPR > 1;
+ NumRegsForPush2 =
+ UsePush2Pop2
+ ? alignDown(IsFPRemovedFromCSI ? NumCSGPR : NumCSGPR - 1, 2)
+ : 0;
}
// Assign slots for GPRs. It increases frame size.
@@ -3159,12 +3180,6 @@ bool X86FrameLowering::spillCalleeSavedRegisters(
// Push GPRs. It increases frame size.
const MachineFunction &MF = *MBB.getParent();
const X86MachineFunctionInfo *X86FI = MF.getInfo<X86MachineFunctionInfo>();
- if (X86FI->padForPush2Pop2()) {
- assert(SlotSize == 8 && "Unexpected slot size for padding!");
- BuildMI(MBB, MI, DL, TII.get(X86::PUSH64r))
- .addReg(X86::RAX, RegState::Undef)
- .setMIFlag(MachineInstr::FrameSetup);
- }
// Update LiveIn of the basic block and decide whether we can add a kill flag
// to the use.
@@ -3343,13 +3358,6 @@ bool X86FrameLowering::restoreCalleeSavedRegisters(
.setMIFlag(MachineInstr::FrameDestroy);
}
}
- if (X86FI->padForPush2Pop2()) {
- if (IsWin64UnwindV3)
- BuildMI(MBB, MI, DL, TII.get(X86::SEH_StackAlloc))
- .addImm(SlotSize)
- .setMIFlag(MachineInstr::FrameDestroy);
- emitSPUpdate(MBB, MI, DL, SlotSize, /*InEpilogue=*/true);
- }
return true;
}
diff --git a/llvm/test/CodeGen/X86/apx/add.ll b/llvm/test/CodeGen/X86/apx/add.ll
index d7c5635b617c1..45c56645237d1 100644
--- a/llvm/test/CodeGen/X86/apx/add.ll
+++ b/llvm/test/CodeGen/X86/apx/add.ll
@@ -1248,7 +1248,7 @@ define i32 @two_address_no_subreg(i32 %arg0, ptr %arg1, i1 %arg2) nounwind {
; NF-NEXT: xorl %ecx, %ecx # encoding: [0x31,0xc9]
; NF-NEXT: callq *%rax # encoding: [0xff,0xd0]
; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: addq $8, %rsp # encoding: [0x48,0x83,0xc4,0x08]
+; NF-NEXT: {nf} addq $8, %rsp # encoding: [0x62,0xf4,0xfc,0x0c,0x83,0xc4,0x08]
; NF-NEXT: popq %rbx # encoding: [0x5b]
; NF-NEXT: popq %r12 # encoding: [0x41,0x5c]
; NF-NEXT: popq %r13 # encoding: [0x41,0x5d]
diff --git a/llvm/test/CodeGen/X86/apx/pp2-with-stack-clash-protection.ll b/llvm/test/CodeGen/X86/apx/pp2-with-stack-clash-protection.ll
index a7795f0b6bc5d..7dc97b14da812 100644
--- a/llvm/test/CodeGen/X86/apx/pp2-with-stack-clash-protection.ll
+++ b/llvm/test/CodeGen/X86/apx/pp2-with-stack-clash-protection.ll
@@ -8,22 +8,22 @@
define i32 @foo(ptr %src1, i32 %len, ptr %src2, ptr %dst, i1 %cmp, i32 %sub) #0 {
; CHECK-LABEL: foo:
; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: pushq %rax
+; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: .cfi_def_cfa_offset 16
-; CHECK-NEXT: push2 %r15, %rbp
+; CHECK-NEXT: push2 %r14, %r15
; CHECK-NEXT: .cfi_def_cfa_offset 32
-; CHECK-NEXT: push2 %r13, %r14
+; CHECK-NEXT: push2 %r12, %r13
; CHECK-NEXT: .cfi_def_cfa_offset 48
-; CHECK-NEXT: push2 %rbx, %r12
+; CHECK-NEXT: pushq %rbx
+; CHECK-NEXT: .cfi_def_cfa_offset 56
+; CHECK-NEXT: pushq %rax
; CHECK-NEXT: .cfi_def_cfa_offset 64
-; CHECK-NEXT: subq $16, %rsp
-; CHECK-NEXT: .cfi_def_cfa_offset 80
-; CHECK-NEXT: .cfi_offset %rbx, -64
-; CHECK-NEXT: .cfi_offset %r12, -56
-; CHECK-NEXT: .cfi_offset %r13, -48
-; CHECK-NEXT: .cfi_offset %r14, -40
-; CHECK-NEXT: .cfi_offset %r15, -32
-; CHECK-NEXT: .cfi_offset %rbp, -24
+; CHECK-NEXT: .cfi_offset %rbx, -56
+; CHECK-NEXT: .cfi_offset %r12, -48
+; CHECK-NEXT: .cfi_offset %r13, -40
+; CHECK-NEXT: .cfi_offset %r14, -32
+; CHECK-NEXT: .cfi_offset %r15, -24
+; CHECK-NEXT: .cfi_offset %rbp, -16
; CHECK-NEXT: movl %r9d, {{[-0-9]+}}(%r{{[sb]}}p) # 4-byte Spill
; CHECK-NEXT: movl %r8d, %ebp
; CHECK-NEXT: movq %rcx, %r14
@@ -44,15 +44,15 @@ define i32 @foo(ptr %src1, i32 %len, ptr %src2, ptr %dst, i1 %cmp, i32 %sub) #0
; CHECK-NEXT: jne .LBB0_1
; CHECK-NEXT: # %bb.2: # %while.end
; CHECK-NEXT: movl {{[-0-9]+}}(%r{{[sb]}}p), %eax # 4-byte Reload
-; CHECK-NEXT: addq $16, %rsp
-; CHECK-NEXT: .cfi_def_cfa_offset 64
-; CHECK-NEXT: pop2 %r12, %rbx
+; CHECK-NEXT: addq $8, %rsp
+; CHECK-NEXT: .cfi_def_cfa_offset 56
+; CHECK-NEXT: popq %rbx
; CHECK-NEXT: .cfi_def_cfa_offset 48
-; CHECK-NEXT: pop2 %r14, %r13
+; CHECK-NEXT: pop2 %r13, %r12
; CHECK-NEXT: .cfi_def_cfa_offset 32
-; CHECK-NEXT: pop2 %rbp, %r15
+; CHECK-NEXT: pop2 %r15, %r14
; CHECK-NEXT: .cfi_def_cfa_offset 16
-; CHECK-NEXT: popq %rcx
+; CHECK-NEXT: popq %rbp
; CHECK-NEXT: .cfi_def_cfa_offset 8
; CHECK-NEXT: retq
entry:
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh-v3.ll b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh-v3.ll
index 671146a8daea4..a10f554821445 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh-v3.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh-v3.ll
@@ -52,16 +52,16 @@ define i32 @csr6_alloc16(ptr %argv) {
;
; WIN-V3-LABEL: csr6_alloc16:
; WIN-V3: # %bb.0: # %entry
-; WIN-V3-NEXT: .seh_pushreg %rax
-; WIN-V3-NEXT: pushq %rax
-; WIN-V3-NEXT: .seh_push2regs %r15, %r14
-; WIN-V3-NEXT: push2 %r14, %r15
-; WIN-V3-NEXT: .seh_push2regs %r13, %r12
-; WIN-V3-NEXT: push2 %r12, %r13
-; WIN-V3-NEXT: .seh_push2regs %rbp, %rbx
-; WIN-V3-NEXT: push2 %rbx, %rbp
-; WIN-V3-NEXT: .seh_stackalloc 64
-; WIN-V3-NEXT: subq $64, %rsp
+; WIN-V3-NEXT: .seh_pushreg %r15
+; WIN-V3-NEXT: pushq %r15
+; WIN-V3-NEXT: .seh_push2regs %r14, %r13
+; WIN-V3-NEXT: push2 %r13, %r14
+; WIN-V3-NEXT: .seh_push2regs %r12, %rbp
+; WIN-V3-NEXT: push2 %rbp, %r12
+; WIN-V3-NEXT: .seh_pushreg %rbx
+; WIN-V3-NEXT: pushq %rbx
+; WIN-V3-NEXT: .seh_stackalloc 56
+; WIN-V3-NEXT: subq $56, %rsp
; WIN-V3-NEXT: .seh_endprologue
; WIN-V3-NEXT: #APP
; WIN-V3-NEXT: #NO_APP
@@ -69,32 +69,32 @@ define i32 @csr6_alloc16(ptr %argv) {
; WIN-V3-NEXT: callq *%rax
; WIN-V3-NEXT: nop
; WIN-V3-NEXT: .seh_startepilogue
-; WIN-V3-NEXT: .seh_stackalloc 64
-; WIN-V3-NEXT: addq $64, %rsp
-; WIN-V3-NEXT: .seh_push2regs %rbx, %rbp
-; WIN-V3-NEXT: pop2 %rbp, %rbx
-; WIN-V3-NEXT: .seh_push2regs %r12, %r13
-; WIN-V3-NEXT: pop2 %r13, %r12
-; WIN-V3-NEXT: .seh_push2regs %r14, %r15
-; WIN-V3-NEXT: pop2 %r15, %r14
-; WIN-V3-NEXT: .seh_stackalloc 8
-; WIN-V3-NEXT: popq %rax
+; WIN-V3-NEXT: .seh_stackalloc 56
+; WIN-V3-NEXT: addq $56, %rsp
+; WIN-V3-NEXT: .seh_pushreg %rbx
+; WIN-V3-NEXT: popq %rbx
+; WIN-V3-NEXT: .seh_push2regs %rbp, %r12
+; WIN-V3-NEXT: pop2 %r12, %rbp
+; WIN-V3-NEXT: .seh_push2regs %r13, %r14
+; WIN-V3-NEXT: pop2 %r14, %r13
+; WIN-V3-NEXT: .seh_pushreg %r15
+; WIN-V3-NEXT: popq %r15
; WIN-V3-NEXT: .seh_endepilogue
; WIN-V3-NEXT: retq
; WIN-V3-NEXT: .seh_endproc
;
; WIN-V3-PPX-LABEL: csr6_alloc16:
; WIN-V3-PPX: # %bb.0: # %entry
-; WIN-V3-PPX-NEXT: .seh_pushreg %rax
-; WIN-V3-PPX-NEXT: pushq %rax
-; WIN-V3-PPX-NEXT: .seh_push2regs %r15, %r14
-; WIN-V3-PPX-NEXT: push2p %r14, %r15
-; WIN-V3-PPX-NEXT: .seh_push2regs %r13, %r12
-; WIN-V3-PPX-NEXT: push2p %r12, %r13
-; WIN-V3-PPX-NEXT: .seh_push2regs %rbp, %rbx
-; WIN-V3-PPX-NEXT: push2p %rbx, %rbp
-; WIN-V3-PPX-NEXT: .seh_stackalloc 64
-; WIN-V3-PPX-NEXT: subq $64, %rsp
+; WIN-V3-PPX-NEXT: .seh_pushreg %r15
+; WIN-V3-PPX-NEXT: pushp %r15
+; WIN-V3-PPX-NEXT: .seh_push2regs %r14, %r13
+; WIN-V3-PPX-NEXT: push2p %r13, %r14
+; WIN-V3-PPX-NEXT: .seh_push2regs %r12, %rbp
+; WIN-V3-PPX-NEXT: push2p %rbp, %r12
+; WIN-V3-PPX-NEXT: .seh_pushreg %rbx
+; WIN-V3-PPX-NEXT: pushp %rbx
+; WIN-V3-PPX-NEXT: .seh_stackalloc 56
+; WIN-V3-PPX-NEXT: subq $56, %rsp
; WIN-V3-PPX-NEXT: .seh_endprologue
; WIN-V3-PPX-NEXT: #APP
; WIN-V3-PPX-NEXT: #NO_APP
@@ -102,16 +102,16 @@ define i32 @csr6_alloc16(ptr %argv) {
; WIN-V3-PPX-NEXT: callq *%rax
; WIN-V3-PPX-NEXT: nop
; WIN-V3-PPX-NEXT: .seh_startepilogue
-; WIN-V3-PPX-NEXT: .seh_stackalloc 64
-; WIN-V3-PPX-NEXT: addq $64, %rsp
-; WIN-V3-PPX-NEXT: .seh_push2regs %rbx, %rbp
-; WIN-V3-PPX-NEXT: pop2p %rbp, %rbx
-; WIN-V3-PPX-NEXT: .seh_push2regs %r12, %r13
-; WIN-V3-PPX-NEXT: pop2p %r13, %r12
-; WIN-V3-PPX-NEXT: .seh_push2regs %r14, %r15
-; WIN-V3-PPX-NEXT: pop2p %r15, %r14
-; WIN-V3-PPX-NEXT: .seh_stackalloc 8
-; WIN-V3-PPX-NEXT: popq %rax
+; WIN-V3-PPX-NEXT: .seh_stackalloc 56
+; WIN-V3-PPX-NEXT: addq $56, %rsp
+; WIN-V3-PPX-NEXT: .seh_pushreg %rbx
+; WIN-V3-PPX-NEXT: popp %rbx
+; WIN-V3-PPX-NEXT: .seh_push2regs %rbp, %r12
+; WIN-V3-PPX-NEXT: pop2p %r12, %rbp
+; WIN-V3-PPX-NEXT: .seh_push2regs %r13, %r14
+; WIN-V3-PPX-NEXT: pop2p %r14, %r13
+; WIN-V3-PPX-NEXT: .seh_pushreg %r15
+; WIN-V3-PPX-NEXT: popp %r15
; WIN-V3-PPX-NEXT: .seh_endepilogue
; WIN-V3-PPX-NEXT: retq
; WIN-V3-PPX-NEXT: .seh_endproc
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
index aabaa968a11b9..5d25f8826cddd 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
@@ -2,7 +2,7 @@
; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu | FileCheck %s --check-prefix=LIN-REF
; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+push2pop2 | FileCheck %s --check-prefix=LIN
; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+push2pop2,+ppx | FileCheck %s --check-prefix=LIN-PPX
-; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mcpu=diamondrapids | FileCheck %s --check-prefix=LIN-PPX
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mcpu=diamondrapids | FileCheck %s --check-prefix=LIN-DR
; RUN: llc < %s -mtriple=x86_64-windows-msvc | FileCheck %s --check-prefix=WIN-REF
; RUN: llc < %s -mtriple=x86_64-windows-msvc -mattr=+push2pop2 | FileCheck %s --check-prefix=WIN
; RUN: llc < %s -mtriple=x86_64-windows-msvc -mattr=+push2pop2,+ppx | FileCheck %s --check-prefix=WIN-PPX
@@ -58,74 +58,109 @@ define i32 @csr6_alloc16(ptr %argv) {
;
; LIN-LABEL: csr6_alloc16:
; LIN: # %bb.0: # %entry
-; LIN-NEXT: pushq %rax
+; LIN-NEXT: pushq %rbp
; LIN-NEXT: .cfi_def_cfa_offset 16
-; LIN-NEXT: push2 %r15, %rbp
+; LIN-NEXT: push2 %r14, %r15
; LIN-NEXT: .cfi_def_cfa_offset 32
-; LIN-NEXT: push2 %r13, %r14
+; LIN-NEXT: push2 %r12, %r13
; LIN-NEXT: .cfi_def_cfa_offset 48
-; LIN-NEXT: push2 %rbx, %r12
-; LIN-NEXT: .cfi_def_cfa_offset 64
-; LIN-NEXT: subq $32, %rsp
-; LIN-NEXT: .cfi_def_cfa_offset 96
-; LIN-NEXT: .cfi_offset %rbx, -64
-; LIN-NEXT: .cfi_offset %r12, -56
-; LIN-NEXT: .cfi_offset %r13, -48
-; LIN-NEXT: .cfi_offset %r14, -40
-; LIN-NEXT: .cfi_offset %r15, -32
-; LIN-NEXT: .cfi_offset %rbp, -24
+; LIN-NEXT: pushq %rbx
+; LIN-NEXT: .cfi_def_cfa_offset 56
+; LIN-NEXT: subq $24, %rsp
+; LIN-NEXT: .cfi_def_cfa_offset 80
+; LIN-NEXT: .cfi_offset %rbx, -56
+; LIN-NEXT: .cfi_offset %r12, -48
+; LIN-NEXT: .cfi_offset %r13, -40
+; LIN-NEXT: .cfi_offset %r14, -32
+; LIN-NEXT: .cfi_offset %r15, -24
+; LIN-NEXT: .cfi_offset %rbp, -16
; LIN-NEXT: #APP
; LIN-NEXT: #NO_APP
; LIN-NEXT: xorl %ecx, %ecx
; LIN-NEXT: xorl %eax, %eax
; LIN-NEXT: callq *%rcx
-; LIN-NEXT: addq $32, %rsp
-; LIN-NEXT: .cfi_def_cfa_offset 64
-; LIN-NEXT: pop2 %r12, %rbx
+; LIN-NEXT: addq $24, %rsp
+; LIN-NEXT: .cfi_def_cfa_offset 56
+; LIN-NEXT: popq %rbx
; LIN-NEXT: .cfi_def_cfa_offset 48
-; LIN-NEXT: pop2 %r14, %r13
+; LIN-NEXT: pop2 %r13, %r12
; LIN-NEXT: .cfi_def_cfa_offset 32
-; LIN-NEXT: pop2 %rbp, %r15
+; LIN-NEXT: pop2 %r15, %r14
; LIN-NEXT: .cfi_def_cfa_offset 16
-; LIN-NEXT: popq %rax
+; LIN-NEXT: popq %rbp
; LIN-NEXT: .cfi_def_cfa_offset 8
; LIN-NEXT: retq
;
; LIN-PPX-LABEL: csr6_alloc16:
; LIN-PPX: # %bb.0: # %entry
-; LIN-PPX-NEXT: pushq %rax
+; LIN-PPX-NEXT: pushp %rbp
; LIN-PPX-NEXT: .cfi_def_cfa_offset 16
-; LIN-PPX-NEXT: push2p %r15, %rbp
+; LIN-PPX-NEXT: push2p %r14, %r15
; LIN-PPX-NEXT: .cfi_def_cfa_offset 32
-; LIN-PPX-NEXT: push2p %r13, %r14
+; LIN-PPX-NEXT: push2p %r12, %r13
; LIN-PPX-NEXT: .cfi_def_cfa_offset 48
-; LIN-PPX-NEXT: push2p %rbx, %r12
-; LIN-PPX-NEXT: .cfi_def_cfa_offset 64
-; LIN-PPX-NEXT: subq $32, %rsp
-; LIN-PPX-NEXT: .cfi_def_cfa_offset 96
-; LIN-PPX-NEXT: .cfi_offset %rbx, -64
-; LIN-PPX-NEXT: .cfi_offset %r12, -56
-; LIN-PPX-NEXT: .cfi_offset %r13, -48
-; LIN-PPX-NEXT: .cfi_offset %r14, -40
-; LIN-PPX-NEXT: .cfi_offset %r15, -32
-; LIN-PPX-NEXT: .cfi_offset %rbp, -24
+; LIN-PPX-NEXT: pushp %rbx
+; LIN-PPX-NEXT: .cfi_def_cfa_offset 56
+; LIN-PPX-NEXT: subq $24, %rsp
+; LIN-PPX-NEXT: .cfi_def_cfa_offset 80
+; LIN-PPX-NEXT: .cfi_offset %rbx, -56
+; LIN-PPX-NEXT: .cfi_offset %r12, -48
+; LIN-PPX-NEXT: .cfi_offset %r13, -40
+; LIN-PPX-NEXT: .cfi_offset %r14, -32
+; LIN-PPX-NEXT: .cfi_offset %r15, -24
+; LIN-PPX-NEXT: .cfi_offset %rbp, -16
; LIN-PPX-NEXT: #APP
; LIN-PPX-NEXT: #NO_APP
; LIN-PPX-NEXT: xorl %ecx, %ecx
; LIN-PPX-NEXT: xorl %eax, %eax
; LIN-PPX-NEXT: callq *%rcx
-; LIN-PPX-NEXT: addq $32, %rsp
-; LIN-PPX-NEXT: .cfi_def_cfa_offset 64
-; LIN-PPX-NEXT: pop2p %r12, %rbx
+; LIN-PPX-NEXT: addq $24, %rsp
+; LIN-PPX-NEXT: .cfi_def_cfa_offset 56
+; LIN-PPX-NEXT: popp %rbx
; LIN-PPX-NEXT: .cfi_def_cfa_offset 48
-; LIN-PPX-NEXT: pop2p %r14, %r13
+; LIN-PPX-NEXT: pop2p %r13, %r12
; LIN-PPX-NEXT: .cfi_def_cfa_offset 32
-; LIN-PPX-NEXT: pop2p %rbp, %r15
+; LIN-PPX-NEXT: pop2p %r15, %r14
; LIN-PPX-NEXT: .cfi_def_cfa_offset 16
-; LIN-PPX-NEXT: popq %rax
+; LIN-PPX-NEXT: popp %rbp
; LIN-PPX-NEXT: .cfi_def_cfa_offset 8
; LIN-PPX-NEXT: retq
;
+; LIN-DR-LABEL: csr6_alloc16:
+; LIN-DR: # %bb.0: # %entry
+; LIN-DR-NEXT: pushp %rbp
+; LIN-DR-NEXT: .cfi_def_cfa_offset 16
+; LIN-DR-NEXT: push2p %r14, %r15
+; LIN-DR-NEXT: .cfi_def_cfa_offset 32
+; LIN-DR-NEXT: push2p %r12, %r13
+; LIN-DR-NEXT: .cfi_def_cfa_offset 48
+; LIN-DR-NEXT: pushp %rbx
+; LIN-DR-NEXT: .cfi_def_cfa_offset 56
+; LIN-DR-NEXT: {nf} subq $24, %rsp
+; LIN-DR-NEXT: .cfi_def_cfa_offset 80
+; LIN-DR-NEXT: .cfi_offset %rbx, -56
+; LIN-DR-NEXT: .cfi_offset %r12, -48
+; LIN-DR-NEXT: .cfi_offset %r13, -40
+; LIN-DR-NEXT: .cfi_offset %r14, -32
+; LIN-DR-NEXT: .cfi_offset %r15, -24
+; LIN-DR-NEXT: .cfi_offset %rbp, -16
+; LIN-DR-NEXT: #APP
+; LIN-DR-NEXT: #NO_APP
+; LIN-DR-NEXT: xorl %ecx, %ecx
+; LIN-DR-NEXT: xorl %eax, %eax
+; LIN-DR-NEXT: callq *%rcx
+; LIN-DR-NEXT: {nf} addq $24, %rsp
+; LIN-DR-NEXT: .cfi_def_cfa_offset 56
+; LIN-DR-NEXT: popp %rbx
+; LIN-DR-NEXT: .cfi_def_cfa_offset 48
+; LIN-DR-NEXT: pop2p %r13, %r12
+; LIN-DR-NEXT: .cfi_def_cfa_offset 32
+; LIN-DR-NEXT: pop2p %r15, %r14
+; LIN-DR-NEXT: .cfi_def_cfa_offset 16
+; LIN-DR-NEXT: popp %rbp
+; LIN-DR-NEXT: .cfi_def_cfa_offset 8
+; LIN-DR-NEXT: retq
+;
; WIN-REF-LABEL: csr6_alloc16:
; WIN-REF: # %bb.0: # %entry
; WIN-REF-NEXT: pushq %r15
@@ -162,19 +197,18 @@ define i32 @csr6_alloc16(ptr %argv) {
;
; WIN-LABEL: csr6_alloc16:
; WIN: # %bb.0: # %entry
-; WIN-NEXT: pushq %rax
-; WIN-NEXT: .seh_pushreg %rax
-; WIN-NEXT: push2 %r14, %r15
+; WIN-NEXT: pushq %r15
; WIN-NEXT: .seh_pushreg %r15
+; WIN-NEXT: push2 %r13, %r14
; WIN-NEXT: .seh_pushreg %r14
-; WIN-NEXT: push2 %r12, %r13
; WIN-NEXT: .seh_pushreg %r13
+; WIN-NEXT: push2 %rbp, %r12
; WIN-NEXT: .seh_pushreg %r12
-; WIN-NEXT: push2 %rbx, %rbp
; WIN-NEXT: .seh_pushreg %rbp
+; WIN-NEXT: pushq %rbx
; WIN-NEXT: .seh_pushreg %rbx
-; WIN-NEXT: subq $64, %rsp
-; WIN-NEXT: .seh_stackalloc 64
+; WIN-NEXT: subq $56, %rsp
+; WIN-NEXT: .seh_stackalloc 56
; WIN-NEXT: .seh_endprologue
; WIN-NEXT: #APP
; WIN-NEXT: #NO_APP
@@ -182,30 +216,29 @@ define i32 @csr6_alloc16(ptr %argv) {
; WIN-NEXT: callq *%rax
; WIN-NEXT: nop
; WIN-NEXT: .seh_startepilogue
-; WIN-NEXT: addq $64, %rsp
-; WIN-NEXT: pop2 %rbp, %rbx
-; WIN-NEXT: pop2 %r13, %r12
-; WIN-NEXT: pop2 %r15, %r14
-; WIN-NEXT: popq %rax
+; WIN-NEXT: addq $56, %rsp
+; WIN-NEXT: popq %rbx
+; WIN-NEXT: pop2 %r12, %rbp
+; WIN-NEXT: pop2 %r14, %r13
+; WIN-NEXT: popq %r15
; WIN-NEXT: .seh_endepilogue
; WIN-NEXT: retq
; WIN-NEXT: .seh_endproc
;
; WIN-PPX-LABEL: csr6_alloc16:
; WIN-PPX: # %bb.0: # %entry
-; WIN-PPX-NEXT: pushq %rax
-; WIN-PPX-NEXT: .seh_pushreg %rax
-; WIN-PPX-NEXT: push2p %r14, %r15
+; WIN-PPX-NEXT: pushp %r15
; WIN-PPX-NEXT: .seh_pushreg %r15
+; WIN-PPX-NEXT: push2p %r13, %r14
; WIN-PPX-NEXT: .seh_pushreg %r14
-; WIN-PPX-NEXT: push2p %r12, %r13
; WIN-PPX-NEXT: .seh_pushreg %r13
+; WIN-PPX-NEXT: push2p %rbp, %r12
; WIN-PPX-NEXT: .seh_pushreg %r12
-; WIN-PPX-NEXT: push2p %rbx, %rbp
; WIN-PPX-NEXT: .seh_pushreg %rbp
+; WIN-PPX-NEXT: pushp %rbx
; WIN-PPX-NEXT: .seh_pushreg %rbx
-; WIN-PPX-NEXT: subq $64, %rsp
-; WIN-PPX-NEXT: .seh_stackalloc 64
+; WIN-PPX-NEXT: subq $56, %rsp
+; WIN-PPX-NEXT: .seh_stackalloc 56
; WIN-PPX-NEXT: .seh_endprologue
; WIN-PPX-NEXT: #APP
; WIN-PPX-NEXT: #NO_APP
@@ -213,11 +246,11 @@ define i32 @csr6_alloc16(ptr %argv) {
; WIN-PPX-NEXT: callq *%rax
; WIN-PPX-NEXT: nop
; WIN-PPX-NEXT: .seh_startepilogue
-; WIN-PPX-NEXT: addq $64, %rsp
-; WIN-PPX-NEXT: pop2p %rbp, %rbx
-; WIN-PPX-NEXT: pop2p %r13, %r12
-; WIN-PPX-NEXT: pop2p %r15, %r14
-; WIN-PPX-NEXT: popq %rax
+; WIN-PPX-NEXT: addq $56, %rsp
+; WIN-PPX-NEXT: popp %rbx
+; WIN-PPX-NEXT: pop2p %r12, %rbp
+; WIN-PPX-NEXT: pop2p %r14, %r13
+; WIN-PPX-NEXT: popp %r15
; WIN-PPX-NEXT: .seh_endepilogue
; WIN-PPX-NEXT: retq
; WIN-PPX-NEXT: .seh_endproc
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2.ll b/llvm/test/CodeGen/X86/apx/push2-pop2.ll
index f5be484be2b1a..ec13253c08e2f 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2.ll
@@ -1,7 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+push2pop2 | FileCheck %s --check-prefix=CHECK
-; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+push2pop2,+ppx | FileCheck %s --check-prefix=PPX
+; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+push2pop2,+ppx | FileCheck %s --check-prefixes=PPX,PPX-NO-NF
; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+push2pop2 -frame-pointer=all | FileCheck %s --check-prefix=FRAME
+; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+push2pop2,+ppx,+nf | FileCheck %s --check-prefixes=PPX,PPX-NF
define void @csr1() nounwind {
; CHECK-LABEL: csr1:
@@ -120,26 +121,26 @@ entry:
define void @csr4() nounwind {
; CHECK-LABEL: csr4:
; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: pushq %rax
-; CHECK-NEXT: push2 %r15, %rbp
-; CHECK-NEXT: push2 %r13, %r14
+; CHECK-NEXT: pushq %rbp
+; CHECK-NEXT: push2 %r14, %r15
+; CHECK-NEXT: pushq %r13
; CHECK-NEXT: #APP
; CHECK-NEXT: #NO_APP
-; CHECK-NEXT: pop2 %r14, %r13
-; CHECK-NEXT: pop2 %rbp, %r15
-; CHECK-NEXT: popq %rax
+; CHECK-NEXT: popq %r13
+; CHECK-NEXT: pop2 %r15, %r14
+; CHECK-NEXT: popq %rbp
; CHECK-NEXT: retq
;
; PPX-LABEL: csr4:
; PPX: # %bb.0: # %entry
-; PPX-NEXT: pushq %rax
-; PPX-NEXT: push2p %r15, %rbp
-; PPX-NEXT: push2p %r13, %r14
+; PPX-NEXT: pushp %rbp
+; PPX-NEXT: push2p %r14, %r15
+; PPX-NEXT: pushp %r13
; PPX-NEXT: #APP
; PPX-NEXT: #NO_APP
-; PPX-NEXT: pop2p %r14, %r13
-; PPX-NEXT: pop2p %rbp, %r15
-; PPX-NEXT: popq %rax
+; PPX-NEXT: popp %r13
+; PPX-NEXT: pop2p %r15, %r14
+; PPX-NEXT: popp %rbp
; PPX-NEXT: retq
;
; FRAME-LABEL: csr4:
@@ -212,30 +213,30 @@ entry:
define void @csr6() nounwind {
; CHECK-LABEL: csr6:
; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: pushq %rax
-; CHECK-NEXT: push2 %r15, %rbp
-; CHECK-NEXT: push2 %r13, %r14
-; CHECK-NEXT: push2 %rbx, %r12
+; CHECK-NEXT: pushq %rbp
+; CHECK-NEXT: push2 %r14, %r15
+; CHECK-NEXT: push2 %r12, %r13
+; CHECK-NEXT: pushq %rbx
; CHECK-NEXT: #APP
; CHECK-NEXT: #NO_APP
-; CHECK-NEXT: pop2 %r12, %rbx
-; CHECK-NEXT: pop2 %r14, %r13
-; CHECK-NEXT: pop2 %rbp, %r15
-; CHECK-NEXT: popq %rax
+; CHECK-NEXT: popq %rbx
+; CHECK-NEXT: pop2 %r13, %r12
+; CHECK-NEXT: pop2 %r15, %r14
+; CHECK-NEXT: popq %rbp
; CHECK-NEXT: retq
;
; PPX-LABEL: csr6:
; PPX: # %bb.0: # %entry
-; PPX-NEXT: pushq %rax
-; PPX-NEXT: push2p %r15, %rbp
-; PPX-NEXT: push2p %r13, %r14
-; PPX-NEXT: push2p %rbx, %r12
+; PPX-NEXT: pushp %rbp
+; PPX-NEXT: push2p %r14, %r15
+; PPX-NEXT: push2p %r12, %r13
+; PPX-NEXT: pushp %rbx
; PPX-NEXT: #APP
; PPX-NEXT: #NO_APP
-; PPX-NEXT: pop2p %r12, %rbx
-; PPX-NEXT: pop2p %r14, %r13
-; PPX-NEXT: pop2p %rbp, %r15
-; PPX-NEXT: popq %rax
+; PPX-NEXT: popp %rbx
+; PPX-NEXT: pop2p %r13, %r12
+; PPX-NEXT: pop2p %r15, %r14
+; PPX-NEXT: popp %rbp
; PPX-NEXT: retq
;
; FRAME-LABEL: csr6:
@@ -269,11 +270,11 @@ define void @lea_in_epilog(i1 %arg, ptr %arg1, ptr %arg2, i64 %arg3, i64 %arg4,
; CHECK-NEXT: testb $1, %dil
; CHECK-NEXT: je .LBB6_5
; CHECK-NEXT: # %bb.1: # %bb13
-; CHECK-NEXT: pushq %rax
-; CHECK-NEXT: push2 %r15, %rbp
-; CHECK-NEXT: push2 %r13, %r14
-; CHECK-NEXT: push2 %rbx, %r12
-; CHECK-NEXT: subq $16, %rsp
+; CHECK-NEXT: pushq %rbp
+; CHECK-NEXT: push2 %r14, %r15
+; CHECK-NEXT: push2 %r12, %r13
+; CHECK-NEXT: pushq %rbx
+; CHECK-NEXT: subq $24, %rsp
; CHECK-NEXT: movq %r9, %r14
; CHECK-NEXT: movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
; CHECK-NEXT: addq {{[0-9]+}}(%rsp), %r14
@@ -306,67 +307,67 @@ define void @lea_in_epilog(i1 %arg, ptr %arg1, ptr %arg2, i64 %arg3, i64 %arg4,
; CHECK-NEXT: # %bb.3: # %bb11
; CHECK-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
; CHECK-NEXT: leaq {{[0-9]+}}(%rsp), %rsp
-; CHECK-NEXT: pop2 %r12, %rbx
-; CHECK-NEXT: pop2 %r14, %r13
-; CHECK-NEXT: pop2 %rbp, %r15
-; CHECK-NEXT: leaq {{[0-9]+}}(%rsp), %rsp
+; CHECK-NEXT: popq %rbx
+; CHECK-NEXT: pop2 %r13, %r12
+; CHECK-NEXT: pop2 %r15, %r14
+; CHECK-NEXT: popq %rbp
; CHECK-NEXT: jne .LBB6_5
; CHECK-NEXT: # %bb.4: # %bb12
; CHECK-NEXT: movq $0, (%rax)
; CHECK-NEXT: .LBB6_5: # %bb14
; CHECK-NEXT: retq
;
-; PPX-LABEL: lea_in_epilog:
-; PPX: # %bb.0: # %bb
-; PPX-NEXT: testb $1, %dil
-; PPX-NEXT: je .LBB6_5
-; PPX-NEXT: # %bb.1: # %bb13
-; PPX-NEXT: pushq %rax
-; PPX-NEXT: push2p %r15, %rbp
-; PPX-NEXT: push2p %r13, %r14
-; PPX-NEXT: push2p %rbx, %r12
-; PPX-NEXT: subq $16, %rsp
-; PPX-NEXT: movq %r9, %r14
-; PPX-NEXT: movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
-; PPX-NEXT: addq {{[0-9]+}}(%rsp), %r14
-; PPX-NEXT: movq {{[0-9]+}}(%rsp), %r13
-; PPX-NEXT: addq %r14, %r13
-; PPX-NEXT: movq {{[0-9]+}}(%rsp), %r15
-; PPX-NEXT: addq %r14, %r15
-; PPX-NEXT: movq {{[0-9]+}}(%rsp), %rbx
-; PPX-NEXT: addq %r14, %rbx
-; PPX-NEXT: xorl %ebp, %ebp
-; PPX-NEXT: xorl %r12d, %r12d
-; PPX-NEXT: movl %edi, {{[-0-9]+}}(%r{{[sb]}}p) # 4-byte Spill
-; PPX-NEXT: .p2align 4
-; PPX-NEXT: .LBB6_2: # %bb15
-; PPX-NEXT: # =>This Inner Loop Header: Depth=1
-; PPX-NEXT: incq %r12
-; PPX-NEXT: movl $432, %edx # imm = 0x1B0
-; PPX-NEXT: xorl %edi, %edi
-; PPX-NEXT: movq %r15, %rsi
-; PPX-NEXT: callq memcpy at PLT
-; PPX-NEXT: movl {{[-0-9]+}}(%r{{[sb]}}p), %edi # 4-byte Reload
-; PPX-NEXT: movq {{[0-9]+}}(%rsp), %rax
-; PPX-NEXT: addq %rax, %r13
-; PPX-NEXT: addq %rax, %r15
-; PPX-NEXT: addq %rax, %rbx
-; PPX-NEXT: addq %rax, %r14
-; PPX-NEXT: addq $8, %rbp
-; PPX-NEXT: testb $1, %dil
-; PPX-NEXT: je .LBB6_2
-; PPX-NEXT: # %bb.3: # %bb11
-; PPX-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
-; PPX-NEXT: leaq {{[0-9]+}}(%rsp), %rsp
-; PPX-NEXT: pop2p %r12, %rbx
-; PPX-NEXT: pop2p %r14, %r13
-; PPX-NEXT: pop2p %rbp, %r15
-; PPX-NEXT: leaq {{[0-9]+}}(%rsp), %rsp
-; PPX-NEXT: jne .LBB6_5
-; PPX-NEXT: # %bb.4: # %bb12
-; PPX-NEXT: movq $0, (%rax)
-; PPX-NEXT: .LBB6_5: # %bb14
-; PPX-NEXT: retq
+; PPX-NO-NF-LABEL: lea_in_epilog:
+; PPX-NO-NF: # %bb.0: # %bb
+; PPX-NO-NF-NEXT: testb $1, %dil
+; PPX-NO-NF-NEXT: je .LBB6_5
+; PPX-NO-NF-NEXT: # %bb.1: # %bb13
+; PPX-NO-NF-NEXT: pushp %rbp
+; PPX-NO-NF-NEXT: push2p %r14, %r15
+; PPX-NO-NF-NEXT: push2p %r12, %r13
+; PPX-NO-NF-NEXT: pushp %rbx
+; PPX-NO-NF-NEXT: subq $24, %rsp
+; PPX-NO-NF-NEXT: movq %r9, %r14
+; PPX-NO-NF-NEXT: movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; PPX-NO-NF-NEXT: addq {{[0-9]+}}(%rsp), %r14
+; PPX-NO-NF-NEXT: movq {{[0-9]+}}(%rsp), %r13
+; PPX-NO-NF-NEXT: addq %r14, %r13
+; PPX-NO-NF-NEXT: movq {{[0-9]+}}(%rsp), %r15
+; PPX-NO-NF-NEXT: addq %r14, %r15
+; PPX-NO-NF-NEXT: movq {{[0-9]+}}(%rsp), %rbx
+; PPX-NO-NF-NEXT: addq %r14, %rbx
+; PPX-NO-NF-NEXT: xorl %ebp, %ebp
+; PPX-NO-NF-NEXT: xorl %r12d, %r12d
+; PPX-NO-NF-NEXT: movl %edi, {{[-0-9]+}}(%r{{[sb]}}p) # 4-byte Spill
+; PPX-NO-NF-NEXT: .p2align 4
+; PPX-NO-NF-NEXT: .LBB6_2: # %bb15
+; PPX-NO-NF-NEXT: # =>This Inner Loop Header: Depth=1
+; PPX-NO-NF-NEXT: incq %r12
+; PPX-NO-NF-NEXT: movl $432, %edx # imm = 0x1B0
+; PPX-NO-NF-NEXT: xorl %edi, %edi
+; PPX-NO-NF-NEXT: movq %r15, %rsi
+; PPX-NO-NF-NEXT: callq memcpy at PLT
+; PPX-NO-NF-NEXT: movl {{[-0-9]+}}(%r{{[sb]}}p), %edi # 4-byte Reload
+; PPX-NO-NF-NEXT: movq {{[0-9]+}}(%rsp), %rax
+; PPX-NO-NF-NEXT: addq %rax, %r13
+; PPX-NO-NF-NEXT: addq %rax, %r15
+; PPX-NO-NF-NEXT: addq %rax, %rbx
+; PPX-NO-NF-NEXT: addq %rax, %r14
+; PPX-NO-NF-NEXT: addq $8, %rbp
+; PPX-NO-NF-NEXT: testb $1, %dil
+; PPX-NO-NF-NEXT: je .LBB6_2
+; PPX-NO-NF-NEXT: # %bb.3: # %bb11
+; PPX-NO-NF-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; PPX-NO-NF-NEXT: leaq {{[0-9]+}}(%rsp), %rsp
+; PPX-NO-NF-NEXT: popp %rbx
+; PPX-NO-NF-NEXT: pop2p %r13, %r12
+; PPX-NO-NF-NEXT: pop2p %r15, %r14
+; PPX-NO-NF-NEXT: popp %rbp
+; PPX-NO-NF-NEXT: jne .LBB6_5
+; PPX-NO-NF-NEXT: # %bb.4: # %bb12
+; PPX-NO-NF-NEXT: movq $0, (%rax)
+; PPX-NO-NF-NEXT: .LBB6_5: # %bb14
+; PPX-NO-NF-NEXT: retq
;
; FRAME-LABEL: lea_in_epilog:
; FRAME: # %bb.0: # %bb
@@ -421,6 +422,58 @@ define void @lea_in_epilog(i1 %arg, ptr %arg1, ptr %arg2, i64 %arg3, i64 %arg4,
; FRAME-NEXT: movq $0, (%rax)
; FRAME-NEXT: .LBB6_5: # %bb14
; FRAME-NEXT: retq
+;
+; PPX-NF-LABEL: lea_in_epilog:
+; PPX-NF: # %bb.0: # %bb
+; PPX-NF-NEXT: testb $1, %dil
+; PPX-NF-NEXT: je .LBB6_5
+; PPX-NF-NEXT: # %bb.1: # %bb13
+; PPX-NF-NEXT: pushp %rbp
+; PPX-NF-NEXT: push2p %r14, %r15
+; PPX-NF-NEXT: push2p %r12, %r13
+; PPX-NF-NEXT: pushp %rbx
+; PPX-NF-NEXT: {nf} subq $24, %rsp
+; PPX-NF-NEXT: movq %r9, %r14
+; PPX-NF-NEXT: movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; PPX-NF-NEXT: addq {{[0-9]+}}(%rsp), %r14
+; PPX-NF-NEXT: movq {{[0-9]+}}(%rsp), %r13
+; PPX-NF-NEXT: addq %r14, %r13
+; PPX-NF-NEXT: movq {{[0-9]+}}(%rsp), %r15
+; PPX-NF-NEXT: addq %r14, %r15
+; PPX-NF-NEXT: movq {{[0-9]+}}(%rsp), %rbx
+; PPX-NF-NEXT: addq %r14, %rbx
+; PPX-NF-NEXT: xorl %ebp, %ebp
+; PPX-NF-NEXT: xorl %r12d, %r12d
+; PPX-NF-NEXT: movl %edi, {{[-0-9]+}}(%r{{[sb]}}p) # 4-byte Spill
+; PPX-NF-NEXT: .p2align 4
+; PPX-NF-NEXT: .LBB6_2: # %bb15
+; PPX-NF-NEXT: # =>This Inner Loop Header: Depth=1
+; PPX-NF-NEXT: incq %r12
+; PPX-NF-NEXT: movl $432, %edx # imm = 0x1B0
+; PPX-NF-NEXT: xorl %edi, %edi
+; PPX-NF-NEXT: movq %r15, %rsi
+; PPX-NF-NEXT: callq memcpy at PLT
+; PPX-NF-NEXT: movl {{[-0-9]+}}(%r{{[sb]}}p), %edi # 4-byte Reload
+; PPX-NF-NEXT: movq {{[0-9]+}}(%rsp), %rax
+; PPX-NF-NEXT: addq %rax, %r13
+; PPX-NF-NEXT: addq %rax, %r15
+; PPX-NF-NEXT: addq %rax, %rbx
+; PPX-NF-NEXT: addq %rax, %r14
+; PPX-NF-NEXT: addq $8, %rbp
+; PPX-NF-NEXT: testb $1, %dil
+; PPX-NF-NEXT: je .LBB6_2
+; PPX-NF-NEXT: # %bb.3: # %bb11
+; PPX-NF-NEXT: movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; PPX-NF-NEXT: {nf} addq $24, %rsp
+; PPX-NF-NEXT: popp %rbx
+; PPX-NF-NEXT: pop2p %r13, %r12
+; PPX-NF-NEXT: pop2p %r15, %r14
+; PPX-NF-NEXT: popp %rbp
+; PPX-NF-NEXT: jne .LBB6_5
+; PPX-NF-NEXT: # %bb.4: # %bb12
+; PPX-NF-NEXT: movq $0, (%rax)
+; PPX-NF-NEXT: .LBB6_5: # %bb14
+; PPX-NF-NEXT: retq
bb:
br i1 %arg, label %bb13, label %bb14
diff --git a/llvm/test/CodeGen/X86/apx/sub.ll b/llvm/test/CodeGen/X86/apx/sub.ll
index 34af966465d93..d999795cb848f 100644
--- a/llvm/test/CodeGen/X86/apx/sub.ll
+++ b/llvm/test/CodeGen/X86/apx/sub.ll
@@ -1182,7 +1182,7 @@ define fastcc void @fold_with_physical_reg(ptr %arg0, i64 %arg1, ptr %arg2, i1 %
; NF-NEXT: pushq %r13 # encoding: [0x41,0x55]
; NF-NEXT: pushq %r12 # encoding: [0x41,0x54]
; NF-NEXT: pushq %rbx # encoding: [0x53]
-; NF-NEXT: subq $24, %rsp # encoding: [0x48,0x83,0xec,0x18]
+; NF-NEXT: {nf} subq $24, %rsp # encoding: [0x62,0xf4,0xfc,0x0c,0x83,0xec,0x18]
; NF-NEXT: movl %ecx, %r14d # encoding: [0x41,0x89,0xce]
; NF-NEXT: movq %rdx, %rbx # encoding: [0x48,0x89,0xd3]
; NF-NEXT: movq %rsi, %r12 # encoding: [0x49,0x89,0xf4]
diff --git a/llvm/test/CodeGen/X86/apx/win64-abi.ll b/llvm/test/CodeGen/X86/apx/win64-abi.ll
index 8aba657f868e7..3a6ed2cea8b56 100644
--- a/llvm/test/CodeGen/X86/apx/win64-abi.ll
+++ b/llvm/test/CodeGen/X86/apx/win64-abi.ll
@@ -151,14 +151,14 @@ entry:
}
; PUSH2/POP2 must be used for callee-saved registers including R30+R31 pair.
-define void @test_push2pop2_r30_r31() nounwind "target-features"="+egpr,+push2pop2" {
+define void @test_push2pop2_r30_r31() nounwind "target-features"="+egpr,+push2pop2" "frame-pointer"="all" {
; CHECK-LABEL: test_push2pop2_r30_r31:
; CHECK: push2 %r30, %r31
; CHECK: callq external
; CHECK: pop2 %r31, %r30
; CHECK: retq
call void @external()
- call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15},~{r16},~{r17},~{r18},~{r19},~{r20},~{r21},~{r22},~{r23},~{r24},~{r25},~{r26},~{r27},~{r28},~{r29},~{r30},~{r31},~{rbp},~{xmm0},~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15}"()
+ call void asm sideeffect "", "~{rax},~{rcx},~{rdx},~{rsi},~{rdi},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15},~{r16},~{r17},~{r18},~{r19},~{r20},~{r21},~{r22},~{r23},~{r24},~{r25},~{r26},~{r27},~{r28},~{r29},~{r30},~{r31},~{xmm0},~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15}"()
ret void
}
diff --git a/llvm/test/CodeGen/X86/win64-eh-unwindv3-push2pop2.ll b/llvm/test/CodeGen/X86/win64-eh-unwindv3-push2pop2.ll
index f24050f2db77c..848db2ec9822e 100644
--- a/llvm/test/CodeGen/X86/win64-eh-unwindv3-push2pop2.ll
+++ b/llvm/test/CodeGen/X86/win64-eh-unwindv3-push2pop2.ll
@@ -16,26 +16,26 @@ declare i32 @c(i32) local_unnamed_addr
define dso_local i32 @push2pop2_padding(i32 %x) local_unnamed_addr {
; CHECK-LABEL: push2pop2_padding:
; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: .seh_pushreg %rax
-; CHECK-NEXT: pushq %rax
-; CHECK-NEXT: .seh_push2regs %r15, %r14
-; CHECK-NEXT: push2 %r14, %r15
-; CHECK-NEXT: .seh_push2regs %r13, %r12
-; CHECK-NEXT: push2 %r12, %r13
-; CHECK-NEXT: .seh_push2regs %rbp, %rbx
-; CHECK-NEXT: push2 %rbx, %rbp
+; CHECK-NEXT: .seh_pushreg %r15
+; CHECK-NEXT: pushq %r15
+; CHECK-NEXT: .seh_push2regs %r14, %r13
+; CHECK-NEXT: push2 %r13, %r14
+; CHECK-NEXT: .seh_push2regs %r12, %rbp
+; CHECK-NEXT: push2 %rbp, %r12
+; CHECK-NEXT: .seh_pushreg %rbx
+; CHECK-NEXT: pushq %rbx
; CHECK-NEXT: .seh_endprologue
; CHECK-NEXT: #APP
; CHECK-NEXT: #NO_APP
; CHECK-NEXT: .seh_startepilogue
-; CHECK-NEXT: .seh_push2regs %rbx, %rbp
-; CHECK-NEXT: pop2 %rbp, %rbx
-; CHECK-NEXT: .seh_push2regs %r12, %r13
-; CHECK-NEXT: pop2 %r13, %r12
-; CHECK-NEXT: .seh_push2regs %r14, %r15
-; CHECK-NEXT: pop2 %r15, %r14
-; CHECK-NEXT: .seh_stackalloc 8
-; CHECK-NEXT: popq %rax
+; CHECK-NEXT: .seh_pushreg %rbx
+; CHECK-NEXT: popq %rbx
+; CHECK-NEXT: .seh_push2regs %rbp, %r12
+; CHECK-NEXT: pop2 %r12, %rbp
+; CHECK-NEXT: .seh_push2regs %r13, %r14
+; CHECK-NEXT: pop2 %r14, %r13
+; CHECK-NEXT: .seh_pushreg %r15
+; CHECK-NEXT: popq %r15
; CHECK-NEXT: .seh_endepilogue
; CHECK-NEXT: jmp c # TAILCALL
; CHECK-NEXT: .seh_endproc
>From 10c1412d0080a4b3719eba80b41ac6916a433e88 Mon Sep 17 00:00:00 2001
From: "Zou, Feng" <feng.zou at intel.com>
Date: Mon, 22 Jun 2026 15:35:05 +0200
Subject: [PATCH 2/3] [X86][APX] Allow NF stack adjustment in Windows x64
epilogue under unwind v3
The Windows prologue unwinder is data-driven and never disassembles the
prologue, so an EVEX NF {nf} sub of RSP is always safe there. The v1/v2
epilogue unwinder, however, disassembles the epilogue to recognize the
canonical "add/lea rsp; pop...; ret" sequence and does not understand EVEX NF
add/sub, so NF must not be used in a v1/v2 epilogue. Unwind v3 encodes epilog
operations declaratively (no disassembly), so {nf} add of RSP is allowed in a
v3 epilogue. The restriction only applies when the function actually emits
unwind info, so nounwind functions may use NF in the epilogue too.
- Replace the over-broad "no NF on Windows" gate in BuildStackAdjustment with
prologue-always / epilogue-only-when-(no-unwind-info-or-v3) gating, factored
into a needsWin64NonV3EpilogueUnwind() helper.
- The NF stack-adjust opcodes are 64-bit (SUB64ri32_NF/ADD64ri32_NF), so don't
use them for the x32 ABI where the stack pointer is the 32-bit ESP.
- push2/pop2 are EVEX-encoded and equally undisassemblable by the v1/v2
epilogue unwinder, so gate push2/pop2 candidacy on the same condition.
- Add a split-file test covering v1, v2, v3, nounwind and x32; regenerate
push2-pop2-cfi-seh.ll (push2/pop2 now suppressed on Windows v1/v2, and split
the diamondrapids RUN line which enables +nf into its own prefix).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply at anthropic.com>
---
llvm/lib/Target/X86/X86FrameLowering.cpp | 43 ++++--
llvm/lib/Target/X86/X86FrameLowering.h | 6 +
.../test/CodeGen/X86/apx/nf-stackalloc-seh.ll | 136 ++++++++++++++++++
.../CodeGen/X86/apx/push2-pop2-cfi-seh.ll | 63 ++++++--
4 files changed, 226 insertions(+), 22 deletions(-)
create mode 100644 llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll
diff --git a/llvm/lib/Target/X86/X86FrameLowering.cpp b/llvm/lib/Target/X86/X86FrameLowering.cpp
index b703fc24b6a6f..1cc990426aed8 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.cpp
+++ b/llvm/lib/Target/X86/X86FrameLowering.cpp
@@ -384,9 +384,17 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
MachineInstrBuilder MI;
// Use NF (no-flags) variants when the target supports it, avoiding
- // gratuitous EFLAGS clobber. Not on Windows where the OS epilogue unwinder
- // doesn't recognize EVEX-encoded instructions.
- bool UseNF = STI.hasNF() && !isWin64Prologue(*MBB.getParent());
+ // gratuitous EFLAGS clobber. The NF stack-adjust opcodes below are 64-bit
+ // (SUB64ri32_NF/ADD64ri32_NF), so don't use them for the x32 ABI where the
+ // stack pointer is 32-bit. On Windows whether NF is safe in the epilogue
+ // depends on how the OS unwinder reads the code: the prologue unwinder is
+ // data-driven (it walks the unwind codes and never disassembles the
+ // prologue), so NF is always safe there; but the v1/v2 epilogue unwinder
+ // disassembles the epilogue and doesn't recognize EVEX NF add/sub
+ // instructions, so only use NF in a Windows epilogue under unwind v3 (which
+ // encodes epilog operations declaratively and needs no disassembly).
+ bool UseNF = STI.hasNF() && Uses64BitFramePtr &&
+ !(InEpilogue && needsWin64NonV3EpilogueUnwind(*MBB.getParent()));
if (UseLEA && !UseNF) {
MI = addRegOffset(BuildMI(MBB, MBBI, DL,
TII.get(getLEArOpcode(Uses64BitFramePtr)),
@@ -395,11 +403,10 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
} else {
bool IsSub = Offset < 0;
uint64_t AbsOffset = IsSub ? -Offset : Offset;
- const unsigned Opc =
- IsSub ? (UseNF ? (unsigned)X86::SUB64ri32_NF
- : getSUBriOpcode(Uses64BitFramePtr))
- : (UseNF ? (unsigned)X86::ADD64ri32_NF
- : getADDriOpcode(Uses64BitFramePtr));
+ const unsigned Opc = IsSub ? (UseNF ? (unsigned)X86::SUB64ri32_NF
+ : getSUBriOpcode(Uses64BitFramePtr))
+ : (UseNF ? (unsigned)X86::ADD64ri32_NF
+ : getADDriOpcode(Uses64BitFramePtr));
MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
.addReg(StackPtr)
.addImm(AbsOffset);
@@ -1485,6 +1492,14 @@ bool X86FrameLowering::isWin64Prologue(const MachineFunction &MF) const {
return MF.getTarget().getMCAsmInfo().usesWindowsCFI();
}
+bool X86FrameLowering::needsWin64NonV3EpilogueUnwind(
+ const MachineFunction &MF) const {
+ const Function &Fn = MF.getFunction();
+ if (!isWin64Prologue(MF) || !Fn.needsUnwindTableEntry())
+ return false;
+ return Fn.getParent()->getWinX64EHUnwindMode() != WinX64EHUnwindMode::V3;
+}
+
bool X86FrameLowering::needsDwarfCFI(const MachineFunction &MF) const {
return !isWin64Prologue(MF) && MF.needsFrameMoves();
}
@@ -2787,9 +2802,8 @@ void X86FrameLowering::emitEpilogue(MachineFunction &MF,
PI->getOperand(0).getReg() == StackPtr &&
(Opc == X86::ADD64ri32 || Opc == X86::ADD64ri32_NF ||
Opc == X86::ADD32ri || Opc == X86::LEA64r)) {
- int64_t SPAdj = Opc == X86::LEA64r
- ? PI->getOperand(4).getImm()
- : PI->getOperand(2).getImm();
+ int64_t SPAdj = Opc == X86::LEA64r ? PI->getOperand(4).getImm()
+ : PI->getOperand(2).getImm();
Offset += SPAdj;
BuildCFI(MBB, MBBI, DL,
MCCFIInstruction::cfiDefCfaOffset(nullptr, -Offset),
@@ -3089,7 +3103,12 @@ bool X86FrameLowering::assignCalleeSavedSpillSlots(
// 3. When the number of CSR push is even, start to use push2 from the 1st
// push and make the stack 16B aligned before the push
unsigned NumRegsForPush2 = 0;
- if (STI.hasPush2Pop2() && getStackAlignment() >= 16) {
+ // push2/pop2 are EVEX-encoded. The Windows v1/v2 epilogue unwinder
+ // disassembles the epilogue and can't decode EVEX, and only unwind v3 emits
+ // the declarative .seh_push2regs, so don't use push2/pop2 when the function
+ // needs v1/v2 Win64 unwind info.
+ if (STI.hasPush2Pop2() && getStackAlignment() >= 16 &&
+ !needsWin64NonV3EpilogueUnwind(MF)) {
unsigned NumCSGPR = llvm::count_if(CSI, [](const CalleeSavedInfo &I) {
return X86::GR64RegClass.contains(I.getReg());
});
diff --git a/llvm/lib/Target/X86/X86FrameLowering.h b/llvm/lib/Target/X86/X86FrameLowering.h
index f1e3796f5fddd..242445a95673f 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.h
+++ b/llvm/lib/Target/X86/X86FrameLowering.h
@@ -244,6 +244,12 @@ class X86FrameLowering : public TargetFrameLowering {
private:
bool isWin64Prologue(const MachineFunction &MF) const;
+ /// Returns true if MF emits Win64 unwind info using a version (v1/v2) whose
+ /// OS epilogue unwinder disassembles the epilogue, and therefore cannot
+ /// decode EVEX-encoded (APX) instructions placed in the epilogue. Unwind v3
+ /// encodes epilog operations declaratively, so it returns false there.
+ bool needsWin64NonV3EpilogueUnwind(const MachineFunction &MF) const;
+
bool needsDwarfCFI(const MachineFunction &MF) const;
uint64_t calculateMaxStackAlign(const MachineFunction &MF) const;
diff --git a/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll b/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll
new file mode 100644
index 0000000000000..8d41143b787a8
--- /dev/null
+++ b/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll
@@ -0,0 +1,136 @@
+; Verify how APX NF (no-flags) ADD/SUB are used for the stack-pointer
+; adjustment in the Windows x64 prologue/epilogue, and how that interacts with
+; the unwind information version.
+;
+; The Windows prologue unwinder is data-driven: it walks the unwind codes and
+; never disassembles the prologue, so an EVEX NF {nf} sub of RSP in the
+; prologue is always safe (v1, v2 and v3 all emit it). The v1/v2 epilogue
+; unwinder, however, disassembles the epilogue to recognize the canonical
+; "add/lea rsp; pop...; ret" sequence, and does not understand EVEX NF add/sub
+; -- so NF must NOT be used in a v1/v2 epilogue. Unwind v3 encodes epilog
+; operations declaratively (no disassembly), so {nf} add of RSP is allowed in a
+; v3 epilogue. The v1/v2 epilogue restriction only applies when the function
+; actually emits unwind info; a nounwind function has no unwind table entry, so
+; its epilogue is never disassembled and {nf} add is allowed there too.
+;
+; The unwind version is selected by a module-wide flag, so each case lives in
+; its own split-file section.
+;
+; RUN: split-file %s %t
+; RUN: llc < %t/v1.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V1
+; RUN: llc < %t/v2.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V2
+; RUN: llc < %t/v3.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V3
+; RUN: llc < %t/nounwind.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=NOUNWIND
+;
+; The NF stack-adjust opcodes are 64-bit (SUB64ri32_NF/ADD64ri32_NF), so for
+; the x32 ABI -- where the stack pointer is the 32-bit ESP -- NF must not be
+; used; the sized non-NF SUB32ri/ADD32ri are emitted instead.
+; RUN: llc < %t/nounwind.ll -mtriple=x86_64-linux-gnux32 -mattr=+nf -verify-machineinstrs | FileCheck %s --check-prefix=X32
+
+;--- v1.ll
+declare void @callee(ptr)
+
+define void @f() {
+; Prologue: NF is always safe on Windows regardless of unwind version, because
+; the prologue unwinder does not disassemble. The stack is allocated with a
+; {nf} subq, with the matching .seh_stackalloc.
+;
+; V1-LABEL: f:
+; V1: {nf} subq ${{[0-9]+}}, %rsp
+; V1: .seh_stackalloc
+; V1: .seh_endprologue
+;
+; Epilogue under v1/v2: the OS unwinder disassembles the epilogue, so the stack
+; deallocation must be a legacy-encoded addq, NOT {nf} addq.
+; V1: .seh_startepilogue
+; V1-NOT: {nf} addq {{.*}}, %rsp
+; V1: addq ${{[0-9]+}}, %rsp
+; V1: .seh_endepilogue
+; V1: retq
+entry:
+ %p = alloca [64 x i8], align 16
+ call void @callee(ptr %p)
+ ret void
+}
+
+;--- v2.ll
+declare void @callee(ptr)
+
+define void @f() {
+; Like v1, the v2 epilogue unwinder disassembles the epilogue (and looks for
+; the .seh_unwindv2start marker), so NF is used in the prologue but NOT in the
+; epilogue.
+; V2-LABEL: f:
+; V2: .seh_unwindversion 2
+; V2: {nf} subq ${{[0-9]+}}, %rsp
+; V2: .seh_stackalloc
+; V2: .seh_endprologue
+;
+; V2: .seh_startepilogue
+; V2-NOT: {nf} addq {{.*}}, %rsp
+; V2: addq ${{[0-9]+}}, %rsp
+; V2: .seh_endepilogue
+; V2: retq
+entry:
+ %p = alloca [64 x i8], align 16
+ call void @callee(ptr %p)
+ ret void
+}
+
+!llvm.module.flags = !{!0}
+!0 = !{i32 1, !"winx64-eh-unwind", i32 2}
+
+;--- v3.ll
+declare void @callee(ptr)
+
+define void @f() {
+; In v3 mode the SEH directive is emitted *before* its instruction.
+; V3: .seh_unwindversion 3
+; V3-LABEL: f:
+; V3: .seh_stackalloc
+; V3: {nf} subq ${{[0-9]+}}, %rsp
+; V3: .seh_endprologue
+;
+; Epilogue under v3: epilog ops are encoded declaratively, so {nf} addq is
+; allowed for the stack deallocation.
+; V3: .seh_startepilogue
+; V3: .seh_stackalloc
+; V3: {nf} addq ${{[0-9]+}}, %rsp
+; V3: .seh_endepilogue
+; V3: retq
+entry:
+ %p = alloca [64 x i8], align 16
+ call void @callee(ptr %p)
+ ret void
+}
+
+!llvm.module.flags = !{!0}
+!0 = !{i32 1, !"winx64-eh-unwind", i32 3}
+
+;--- nounwind.ll
+declare void @callee(ptr)
+
+; A nounwind function emits no SEH unwind info, so the OS never disassembles
+; its epilogue. NF is therefore allowed in both the prologue and the epilogue
+; even under the default (v1) unwind mode, and no .seh_ directives are emitted.
+define void @f() nounwind {
+; NOUNWIND-LABEL: f:
+; NOUNWIND-NOT: .seh_
+; NOUNWIND: {nf} subq ${{[0-9]+}}, %rsp
+; NOUNWIND: {nf} addq ${{[0-9]+}}, %rsp
+; NOUNWIND: retq
+; NOUNWIND-NOT: .seh_
+;
+; For the x32 ABI the stack pointer is ESP, so the 64-bit NF opcodes can't be
+; used: plain SUB32ri/ADD32ri on %esp are emitted, with no {nf} prefix.
+; X32-LABEL: f:
+; X32-NOT: {nf}
+; X32: subl ${{[0-9]+}}, %esp
+; X32: addl ${{[0-9]+}}, %esp
+; X32: retq
+; X32-NOT: {nf}
+entry:
+ %p = alloca [64 x i8], align 16
+ call void @callee(ptr %p)
+ ret void
+}
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
index 5d25f8826cddd..f8476cc7f1c1a 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
@@ -9,8 +9,9 @@
; EPGR normally required unwind v3 info, but that changes the SEH directives
; that get emitted, so disable epgr so that we can validate diamondrapids
-; enables push2pop2
-; RUN: llc < %s -mtriple=x86_64-windows-msvc -mcpu=diamondrapids -mattr=-egpr | FileCheck %s --check-prefix=WIN-PPX
+; enables push2pop2. diamondrapids also enables +nf, so the prologue stack
+; adjustment uses an NF sub; use a separate prefix from the +ppx line above.
+; RUN: llc < %s -mtriple=x86_64-windows-msvc -mcpu=diamondrapids -mattr=-egpr | FileCheck %s --check-prefix=WIN-DR
define i32 @csr6_alloc16(ptr %argv) {
; LIN-REF-LABEL: csr6_alloc16:
@@ -199,11 +200,13 @@ define i32 @csr6_alloc16(ptr %argv) {
; WIN: # %bb.0: # %entry
; WIN-NEXT: pushq %r15
; WIN-NEXT: .seh_pushreg %r15
-; WIN-NEXT: push2 %r13, %r14
+; WIN-NEXT: pushq %r14
; WIN-NEXT: .seh_pushreg %r14
+; WIN-NEXT: pushq %r13
; WIN-NEXT: .seh_pushreg %r13
-; WIN-NEXT: push2 %rbp, %r12
+; WIN-NEXT: pushq %r12
; WIN-NEXT: .seh_pushreg %r12
+; WIN-NEXT: pushq %rbp
; WIN-NEXT: .seh_pushreg %rbp
; WIN-NEXT: pushq %rbx
; WIN-NEXT: .seh_pushreg %rbx
@@ -218,8 +221,10 @@ define i32 @csr6_alloc16(ptr %argv) {
; WIN-NEXT: .seh_startepilogue
; WIN-NEXT: addq $56, %rsp
; WIN-NEXT: popq %rbx
-; WIN-NEXT: pop2 %r12, %rbp
-; WIN-NEXT: pop2 %r14, %r13
+; WIN-NEXT: popq %rbp
+; WIN-NEXT: popq %r12
+; WIN-NEXT: popq %r13
+; WIN-NEXT: popq %r14
; WIN-NEXT: popq %r15
; WIN-NEXT: .seh_endepilogue
; WIN-NEXT: retq
@@ -229,11 +234,13 @@ define i32 @csr6_alloc16(ptr %argv) {
; WIN-PPX: # %bb.0: # %entry
; WIN-PPX-NEXT: pushp %r15
; WIN-PPX-NEXT: .seh_pushreg %r15
-; WIN-PPX-NEXT: push2p %r13, %r14
+; WIN-PPX-NEXT: pushp %r14
; WIN-PPX-NEXT: .seh_pushreg %r14
+; WIN-PPX-NEXT: pushp %r13
; WIN-PPX-NEXT: .seh_pushreg %r13
-; WIN-PPX-NEXT: push2p %rbp, %r12
+; WIN-PPX-NEXT: pushp %r12
; WIN-PPX-NEXT: .seh_pushreg %r12
+; WIN-PPX-NEXT: pushp %rbp
; WIN-PPX-NEXT: .seh_pushreg %rbp
; WIN-PPX-NEXT: pushp %rbx
; WIN-PPX-NEXT: .seh_pushreg %rbx
@@ -248,12 +255,48 @@ define i32 @csr6_alloc16(ptr %argv) {
; WIN-PPX-NEXT: .seh_startepilogue
; WIN-PPX-NEXT: addq $56, %rsp
; WIN-PPX-NEXT: popp %rbx
-; WIN-PPX-NEXT: pop2p %r12, %rbp
-; WIN-PPX-NEXT: pop2p %r14, %r13
+; WIN-PPX-NEXT: popp %rbp
+; WIN-PPX-NEXT: popp %r12
+; WIN-PPX-NEXT: popp %r13
+; WIN-PPX-NEXT: popp %r14
; WIN-PPX-NEXT: popp %r15
; WIN-PPX-NEXT: .seh_endepilogue
; WIN-PPX-NEXT: retq
; WIN-PPX-NEXT: .seh_endproc
+;
+; WIN-DR-LABEL: csr6_alloc16:
+; WIN-DR: # %bb.0: # %entry
+; WIN-DR-NEXT: pushp %r15
+; WIN-DR-NEXT: .seh_pushreg %r15
+; WIN-DR-NEXT: pushp %r14
+; WIN-DR-NEXT: .seh_pushreg %r14
+; WIN-DR-NEXT: pushp %r13
+; WIN-DR-NEXT: .seh_pushreg %r13
+; WIN-DR-NEXT: pushp %r12
+; WIN-DR-NEXT: .seh_pushreg %r12
+; WIN-DR-NEXT: pushp %rbp
+; WIN-DR-NEXT: .seh_pushreg %rbp
+; WIN-DR-NEXT: pushp %rbx
+; WIN-DR-NEXT: .seh_pushreg %rbx
+; WIN-DR-NEXT: {nf} subq $56, %rsp
+; WIN-DR-NEXT: .seh_stackalloc 56
+; WIN-DR-NEXT: .seh_endprologue
+; WIN-DR-NEXT: #APP
+; WIN-DR-NEXT: #NO_APP
+; WIN-DR-NEXT: xorl %eax, %eax
+; WIN-DR-NEXT: callq *%rax
+; WIN-DR-NEXT: nop
+; WIN-DR-NEXT: .seh_startepilogue
+; WIN-DR-NEXT: addq $56, %rsp
+; WIN-DR-NEXT: popp %rbx
+; WIN-DR-NEXT: popp %rbp
+; WIN-DR-NEXT: popp %r12
+; WIN-DR-NEXT: popp %r13
+; WIN-DR-NEXT: popp %r14
+; WIN-DR-NEXT: popp %r15
+; WIN-DR-NEXT: .seh_endepilogue
+; WIN-DR-NEXT: retq
+; WIN-DR-NEXT: .seh_endproc
entry:
tail call void asm sideeffect "", "~{rbp},~{r15},~{r14},~{r13},~{r12},~{rbx},~{dirflag},~{fpsr},~{flags}"()
%a = alloca [3 x ptr], align 8
>From c643e1569fbaa612143d93839ae0cfabdf3dde98 Mon Sep 17 00:00:00 2001
From: "Zou, Feng" <feng.zou at intel.com>
Date: Wed, 24 Jun 2026 03:18:38 +0200
Subject: [PATCH 3/3] [X86][APX] Address review: prefer NF only as LEA
replacement, drop dead code
Apply review feedback on the push+push2+push / NF stack-adjustment work:
- BuildStackAdjustment: use an NF (no-flags) SUB/ADD only as a smaller
replacement for LEA, i.e. only when EFLAGS must be preserved. When EFLAGS is
dead, prefer the plain SUB/ADD, which is shorter than the EVEX-encoded NF
form. Reformat into a clear three-way branch (NF / LEA / plain SUB+ADD).
- Because NF is now gated on UseLEA, it can never reach a Win64 epilogue (which
only uses LEA for the SP adjustment when it has a frame pointer, and that path
does not go through BuildStackAdjustment). The Windows epilogue unwinder thus
never sees an undisassemblable NF add/sub, so drop the unwind-v3 special case
from the NF gate. needsWin64NonV3EpilogueUnwind is retained: it still gates
push2/pop2 candidacy, which is an independent EVEX-encoding concern.
- Remove the epilogue DWARF CFI branch that handled FrameDestroy ADD/LEA/NF
stack adjustments. It is unreachable: the only post-FirstCSPop SP adjustment
was the push2/pop2 padding deallocation, which was removed when padding moved
to a real CSR push/pop; the local-frame restore is emitted before FirstCSPop
with its own CFI.
- Remove the now-unused PadForPush2Pop2 flag and its accessors; its only writer
was deleted with the old dummy-push padding, so getCalleeSavedFrameSize() no
longer needs the padding term.
- Replace nf-stackalloc-seh.ll (which tested the now-unreachable Windows NF
stack-adjustment path) with nf-stackadjust.ll, which exercises NF in both the
prologue ({nf} subq) and epilogue ({nf} addq) via the lea-sp tuning, and
checks the LEA, plain-SUB/ADD (EFLAGS dead) and x32 (no NF) fallbacks. Update
the push2-pop2-cfi-seh.ll WIN-DR comment to reflect that it validates
push2/pop2 suppression under V1 unwind, and regenerate add.ll, sub.ll and
push2-pop2.ll (gratuitous NF reverts to plain SUB/ADD).
Assisted-by: Claude Opus 4.8 (1M context) <noreply at anthropic.com>
---
llvm/lib/Target/X86/X86FrameLowering.cpp | 53 +++----
llvm/lib/Target/X86/X86MachineFunctionInfo.h | 10 +-
llvm/test/CodeGen/X86/apx/add.ll | 2 +-
llvm/test/CodeGen/X86/apx/nf-stackadjust.ll | 65 +++++++++
.../test/CodeGen/X86/apx/nf-stackalloc-seh.ll | 136 ------------------
.../CodeGen/X86/apx/push2-pop2-cfi-seh.ll | 17 ++-
llvm/test/CodeGen/X86/apx/push2-pop2.ll | 2 +-
llvm/test/CodeGen/X86/apx/sub.ll | 2 +-
8 files changed, 101 insertions(+), 186 deletions(-)
create mode 100644 llvm/test/CodeGen/X86/apx/nf-stackadjust.ll
delete mode 100644 llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll
diff --git a/llvm/lib/Target/X86/X86FrameLowering.cpp b/llvm/lib/Target/X86/X86FrameLowering.cpp
index 1cc990426aed8..dce161fa6316f 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.cpp
+++ b/llvm/lib/Target/X86/X86FrameLowering.cpp
@@ -383,35 +383,36 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
}
MachineInstrBuilder MI;
- // Use NF (no-flags) variants when the target supports it, avoiding
- // gratuitous EFLAGS clobber. The NF stack-adjust opcodes below are 64-bit
- // (SUB64ri32_NF/ADD64ri32_NF), so don't use them for the x32 ABI where the
- // stack pointer is 32-bit. On Windows whether NF is safe in the epilogue
- // depends on how the OS unwinder reads the code: the prologue unwinder is
- // data-driven (it walks the unwind codes and never disassembles the
- // prologue), so NF is always safe there; but the v1/v2 epilogue unwinder
- // disassembles the epilogue and doesn't recognize EVEX NF add/sub
- // instructions, so only use NF in a Windows epilogue under unwind v3 (which
- // encodes epilog operations declaratively and needs no disassembly).
- bool UseNF = STI.hasNF() && Uses64BitFramePtr &&
- !(InEpilogue && needsWin64NonV3EpilogueUnwind(*MBB.getParent()));
- if (UseLEA && !UseNF) {
+ // Use an NF (no-flags) variant as a smaller replacement for LEA when EFLAGS
+ // must be preserved (i.e. only when we would otherwise emit LEA). If EFLAGS
+ // is dead we prefer the plain SUB/ADD, which is shorter than the EVEX-encoded
+ // NF form. The NF stack-adjust opcodes below are 64-bit (SUB64ri32_NF/
+ // ADD64ri32_NF), so don't use them for the x32 ABI where the stack pointer is
+ // 32-bit. NF cannot reach a Win64 epilogue (which never uses LEA for the SP
+ // adjustment unless it has a frame pointer, and that path doesn't go through
+ // here), so the Windows epilogue unwinder never sees an undisassemblable NF
+ // add/sub.
+ bool UseNF = UseLEA && STI.hasNF() && Uses64BitFramePtr;
+ bool IsSub = Offset < 0;
+ uint64_t AbsOffset = IsSub ? -Offset : Offset;
+ if (UseNF) {
+ const unsigned Opc = IsSub ? X86::SUB64ri32_NF : X86::ADD64ri32_NF;
+ MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
+ .addReg(StackPtr)
+ .addImm(AbsOffset);
+ // NF instructions define no EFLAGS, so there is nothing to mark dead.
+ } else if (UseLEA) {
MI = addRegOffset(BuildMI(MBB, MBBI, DL,
TII.get(getLEArOpcode(Uses64BitFramePtr)),
StackPtr),
StackPtr, false, Offset);
} else {
- bool IsSub = Offset < 0;
- uint64_t AbsOffset = IsSub ? -Offset : Offset;
- const unsigned Opc = IsSub ? (UseNF ? (unsigned)X86::SUB64ri32_NF
- : getSUBriOpcode(Uses64BitFramePtr))
- : (UseNF ? (unsigned)X86::ADD64ri32_NF
- : getADDriOpcode(Uses64BitFramePtr));
+ const unsigned Opc = IsSub ? getSUBriOpcode(Uses64BitFramePtr)
+ : getADDriOpcode(Uses64BitFramePtr);
MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
.addReg(StackPtr)
.addImm(AbsOffset);
- if (!UseNF)
- MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
+ MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
}
return MI;
}
@@ -2798,16 +2799,6 @@ void X86FrameLowering::emitEpilogue(MachineFunction &MF,
BuildCFI(MBB, MBBI, DL,
MCCFIInstruction::cfiDefCfaOffset(nullptr, -Offset),
MachineInstr::FrameDestroy);
- } else if (PI->getFlag(MachineInstr::FrameDestroy) &&
- PI->getOperand(0).getReg() == StackPtr &&
- (Opc == X86::ADD64ri32 || Opc == X86::ADD64ri32_NF ||
- Opc == X86::ADD32ri || Opc == X86::LEA64r)) {
- int64_t SPAdj = Opc == X86::LEA64r ? PI->getOperand(4).getImm()
- : PI->getOperand(2).getImm();
- Offset += SPAdj;
- BuildCFI(MBB, MBBI, DL,
- MCCFIInstruction::cfiDefCfaOffset(nullptr, -Offset),
- MachineInstr::FrameDestroy);
}
}
}
diff --git a/llvm/lib/Target/X86/X86MachineFunctionInfo.h b/llvm/lib/Target/X86/X86MachineFunctionInfo.h
index 1bda505ed39f1..1cfe86d57071f 100644
--- a/llvm/lib/Target/X86/X86MachineFunctionInfo.h
+++ b/llvm/lib/Target/X86/X86MachineFunctionInfo.h
@@ -149,9 +149,6 @@ class X86MachineFunctionInfo : public MachineFunctionInfo {
/// other tools to detect the extended record.
bool HasSwiftAsyncContext = false;
- /// Adjust stack for push2/pop2
- bool PadForPush2Pop2 = false;
-
/// Candidate registers for push2/pop2
std::set<Register> CandidatesForPush2Pop2;
@@ -211,9 +208,7 @@ class X86MachineFunctionInfo : public MachineFunctionInfo {
const DenseMap<int, unsigned>& getWinEHXMMSlotInfo() const {
return WinEHXMMSlotInfo; }
- unsigned getCalleeSavedFrameSize() const {
- return CalleeSavedFrameSize + 8 * padForPush2Pop2();
- }
+ unsigned getCalleeSavedFrameSize() const { return CalleeSavedFrameSize; }
void setCalleeSavedFrameSize(unsigned bytes) { CalleeSavedFrameSize = bytes; }
unsigned getBytesToPopOnReturn() const { return BytesToPopOnReturn; }
@@ -284,9 +279,6 @@ class X86MachineFunctionInfo : public MachineFunctionInfo {
bool hasSwiftAsyncContext() const { return HasSwiftAsyncContext; }
void setHasSwiftAsyncContext(bool v) { HasSwiftAsyncContext = v; }
- bool padForPush2Pop2() const { return PadForPush2Pop2; }
- void setPadForPush2Pop2(bool V) { PadForPush2Pop2 = V; }
-
bool isCandidateForPush2Pop2(Register Reg) const {
return CandidatesForPush2Pop2.find(Reg) != CandidatesForPush2Pop2.end();
}
diff --git a/llvm/test/CodeGen/X86/apx/add.ll b/llvm/test/CodeGen/X86/apx/add.ll
index 45c56645237d1..d7c5635b617c1 100644
--- a/llvm/test/CodeGen/X86/apx/add.ll
+++ b/llvm/test/CodeGen/X86/apx/add.ll
@@ -1248,7 +1248,7 @@ define i32 @two_address_no_subreg(i32 %arg0, ptr %arg1, i1 %arg2) nounwind {
; NF-NEXT: xorl %ecx, %ecx # encoding: [0x31,0xc9]
; NF-NEXT: callq *%rax # encoding: [0xff,0xd0]
; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: {nf} addq $8, %rsp # encoding: [0x62,0xf4,0xfc,0x0c,0x83,0xc4,0x08]
+; NF-NEXT: addq $8, %rsp # encoding: [0x48,0x83,0xc4,0x08]
; NF-NEXT: popq %rbx # encoding: [0x5b]
; NF-NEXT: popq %r12 # encoding: [0x41,0x5c]
; NF-NEXT: popq %r13 # encoding: [0x41,0x5d]
diff --git a/llvm/test/CodeGen/X86/apx/nf-stackadjust.ll b/llvm/test/CodeGen/X86/apx/nf-stackadjust.ll
new file mode 100644
index 0000000000000..f54956115adb8
--- /dev/null
+++ b/llvm/test/CodeGen/X86/apx/nf-stackadjust.ll
@@ -0,0 +1,65 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; Verify APX NF (no-flags) ADD/SUB used for the stack-pointer adjustment.
+;
+; NF is preferred only as a smaller replacement for LEA, i.e. only when EFLAGS
+; must be preserved across the adjustment. The lea-sp tuning forces LEA for the
+; SP, so it is the simplest way to exercise NF in both the prologue ({nf} subq)
+; and the epilogue ({nf} addq). Without lea-sp the adjustment has dead EFLAGS
+; and uses a plain (shorter) SUB/ADD instead.
+;
+; RUN: llc < %s -mtriple=x86_64-linux-gnu -mattr=+nf,+lea-sp | FileCheck %s --check-prefix=NF
+; RUN: llc < %s -mtriple=x86_64-linux-gnu -mattr=+lea-sp | FileCheck %s --check-prefix=LEA
+; RUN: llc < %s -mtriple=x86_64-linux-gnu -mattr=+nf | FileCheck %s --check-prefix=NONF
+;
+; The NF stack-adjust opcodes are 64-bit (SUB64ri32_NF/ADD64ri32_NF), so for the
+; x32 ABI -- where the stack pointer is the 32-bit ESP -- NF must not be used;
+; LEA on %esp is emitted instead.
+; RUN: llc < %s -mtriple=x86_64-linux-gnux32 -mattr=+nf,+lea-sp -verify-machineinstrs | FileCheck %s --check-prefix=X32
+
+declare void @callee(ptr)
+
+define void @f() {
+; NF-LABEL: f:
+; NF: # %bb.0: # %entry
+; NF-NEXT: {nf} subq $72, %rsp
+; NF-NEXT: .cfi_def_cfa_offset 80
+; NF-NEXT: movq %rsp, %rdi
+; NF-NEXT: callq callee at PLT
+; NF-NEXT: {nf} addq $72, %rsp
+; NF-NEXT: .cfi_def_cfa_offset 8
+; NF-NEXT: retq
+;
+; LEA-LABEL: f:
+; LEA: # %bb.0: # %entry
+; LEA-NEXT: leaq -{{[0-9]+}}(%rsp), %rsp
+; LEA-NEXT: .cfi_def_cfa_offset 80
+; LEA-NEXT: movq %rsp, %rdi
+; LEA-NEXT: callq callee at PLT
+; LEA-NEXT: leaq {{[0-9]+}}(%rsp), %rsp
+; LEA-NEXT: .cfi_def_cfa_offset 8
+; LEA-NEXT: retq
+;
+; NONF-LABEL: f:
+; NONF: # %bb.0: # %entry
+; NONF-NEXT: subq $72, %rsp
+; NONF-NEXT: .cfi_def_cfa_offset 80
+; NONF-NEXT: movq %rsp, %rdi
+; NONF-NEXT: callq callee at PLT
+; NONF-NEXT: addq $72, %rsp
+; NONF-NEXT: .cfi_def_cfa_offset 8
+; NONF-NEXT: retq
+;
+; X32-LABEL: f:
+; X32: # %bb.0: # %entry
+; X32-NEXT: leal -{{[0-9]+}}(%esp), %esp
+; X32-NEXT: .cfi_def_cfa_offset 80
+; X32-NEXT: movl %esp, %edi
+; X32-NEXT: callq callee at PLT
+; X32-NEXT: leal {{[0-9]+}}(%esp), %esp
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: retq
+entry:
+ %p = alloca [64 x i8], align 16
+ call void @callee(ptr %p)
+ ret void
+}
diff --git a/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll b/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll
deleted file mode 100644
index 8d41143b787a8..0000000000000
--- a/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll
+++ /dev/null
@@ -1,136 +0,0 @@
-; Verify how APX NF (no-flags) ADD/SUB are used for the stack-pointer
-; adjustment in the Windows x64 prologue/epilogue, and how that interacts with
-; the unwind information version.
-;
-; The Windows prologue unwinder is data-driven: it walks the unwind codes and
-; never disassembles the prologue, so an EVEX NF {nf} sub of RSP in the
-; prologue is always safe (v1, v2 and v3 all emit it). The v1/v2 epilogue
-; unwinder, however, disassembles the epilogue to recognize the canonical
-; "add/lea rsp; pop...; ret" sequence, and does not understand EVEX NF add/sub
-; -- so NF must NOT be used in a v1/v2 epilogue. Unwind v3 encodes epilog
-; operations declaratively (no disassembly), so {nf} add of RSP is allowed in a
-; v3 epilogue. The v1/v2 epilogue restriction only applies when the function
-; actually emits unwind info; a nounwind function has no unwind table entry, so
-; its epilogue is never disassembled and {nf} add is allowed there too.
-;
-; The unwind version is selected by a module-wide flag, so each case lives in
-; its own split-file section.
-;
-; RUN: split-file %s %t
-; RUN: llc < %t/v1.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V1
-; RUN: llc < %t/v2.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V2
-; RUN: llc < %t/v3.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V3
-; RUN: llc < %t/nounwind.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=NOUNWIND
-;
-; The NF stack-adjust opcodes are 64-bit (SUB64ri32_NF/ADD64ri32_NF), so for
-; the x32 ABI -- where the stack pointer is the 32-bit ESP -- NF must not be
-; used; the sized non-NF SUB32ri/ADD32ri are emitted instead.
-; RUN: llc < %t/nounwind.ll -mtriple=x86_64-linux-gnux32 -mattr=+nf -verify-machineinstrs | FileCheck %s --check-prefix=X32
-
-;--- v1.ll
-declare void @callee(ptr)
-
-define void @f() {
-; Prologue: NF is always safe on Windows regardless of unwind version, because
-; the prologue unwinder does not disassemble. The stack is allocated with a
-; {nf} subq, with the matching .seh_stackalloc.
-;
-; V1-LABEL: f:
-; V1: {nf} subq ${{[0-9]+}}, %rsp
-; V1: .seh_stackalloc
-; V1: .seh_endprologue
-;
-; Epilogue under v1/v2: the OS unwinder disassembles the epilogue, so the stack
-; deallocation must be a legacy-encoded addq, NOT {nf} addq.
-; V1: .seh_startepilogue
-; V1-NOT: {nf} addq {{.*}}, %rsp
-; V1: addq ${{[0-9]+}}, %rsp
-; V1: .seh_endepilogue
-; V1: retq
-entry:
- %p = alloca [64 x i8], align 16
- call void @callee(ptr %p)
- ret void
-}
-
-;--- v2.ll
-declare void @callee(ptr)
-
-define void @f() {
-; Like v1, the v2 epilogue unwinder disassembles the epilogue (and looks for
-; the .seh_unwindv2start marker), so NF is used in the prologue but NOT in the
-; epilogue.
-; V2-LABEL: f:
-; V2: .seh_unwindversion 2
-; V2: {nf} subq ${{[0-9]+}}, %rsp
-; V2: .seh_stackalloc
-; V2: .seh_endprologue
-;
-; V2: .seh_startepilogue
-; V2-NOT: {nf} addq {{.*}}, %rsp
-; V2: addq ${{[0-9]+}}, %rsp
-; V2: .seh_endepilogue
-; V2: retq
-entry:
- %p = alloca [64 x i8], align 16
- call void @callee(ptr %p)
- ret void
-}
-
-!llvm.module.flags = !{!0}
-!0 = !{i32 1, !"winx64-eh-unwind", i32 2}
-
-;--- v3.ll
-declare void @callee(ptr)
-
-define void @f() {
-; In v3 mode the SEH directive is emitted *before* its instruction.
-; V3: .seh_unwindversion 3
-; V3-LABEL: f:
-; V3: .seh_stackalloc
-; V3: {nf} subq ${{[0-9]+}}, %rsp
-; V3: .seh_endprologue
-;
-; Epilogue under v3: epilog ops are encoded declaratively, so {nf} addq is
-; allowed for the stack deallocation.
-; V3: .seh_startepilogue
-; V3: .seh_stackalloc
-; V3: {nf} addq ${{[0-9]+}}, %rsp
-; V3: .seh_endepilogue
-; V3: retq
-entry:
- %p = alloca [64 x i8], align 16
- call void @callee(ptr %p)
- ret void
-}
-
-!llvm.module.flags = !{!0}
-!0 = !{i32 1, !"winx64-eh-unwind", i32 3}
-
-;--- nounwind.ll
-declare void @callee(ptr)
-
-; A nounwind function emits no SEH unwind info, so the OS never disassembles
-; its epilogue. NF is therefore allowed in both the prologue and the epilogue
-; even under the default (v1) unwind mode, and no .seh_ directives are emitted.
-define void @f() nounwind {
-; NOUNWIND-LABEL: f:
-; NOUNWIND-NOT: .seh_
-; NOUNWIND: {nf} subq ${{[0-9]+}}, %rsp
-; NOUNWIND: {nf} addq ${{[0-9]+}}, %rsp
-; NOUNWIND: retq
-; NOUNWIND-NOT: .seh_
-;
-; For the x32 ABI the stack pointer is ESP, so the 64-bit NF opcodes can't be
-; used: plain SUB32ri/ADD32ri on %esp are emitted, with no {nf} prefix.
-; X32-LABEL: f:
-; X32-NOT: {nf}
-; X32: subl ${{[0-9]+}}, %esp
-; X32: addl ${{[0-9]+}}, %esp
-; X32: retq
-; X32-NOT: {nf}
-entry:
- %p = alloca [64 x i8], align 16
- call void @callee(ptr %p)
- ret void
-}
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
index f8476cc7f1c1a..d9814e44632e6 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
@@ -7,10 +7,13 @@
; RUN: llc < %s -mtriple=x86_64-windows-msvc -mattr=+push2pop2 | FileCheck %s --check-prefix=WIN
; RUN: llc < %s -mtriple=x86_64-windows-msvc -mattr=+push2pop2,+ppx | FileCheck %s --check-prefix=WIN-PPX
-; EPGR normally required unwind v3 info, but that changes the SEH directives
-; that get emitted, so disable epgr so that we can validate diamondrapids
-; enables push2pop2. diamondrapids also enables +nf, so the prologue stack
-; adjustment uses an NF sub; use a separate prefix from the +ppx line above.
+; diamondrapids enables EGPR, which would require V3 unwind info and emit
+; V3-style SEH directives; disable EGPR so this runs with default (V1) unwind.
+; This validates that the push2/pop2 candidacy gate suppresses push2/pop2 under
+; V1 (individual pushp/popp emitted): the function emits unwind info, and the
+; V1/V2 epilogue unwinder can't decode EVEX push2/pop2. diamondrapids also
+; enables +nf, but the stack adjustment here has dead EFLAGS so it stays a plain
+; (non-NF) subq/addq.
; RUN: llc < %s -mtriple=x86_64-windows-msvc -mcpu=diamondrapids -mattr=-egpr | FileCheck %s --check-prefix=WIN-DR
define i32 @csr6_alloc16(ptr %argv) {
@@ -137,7 +140,7 @@ define i32 @csr6_alloc16(ptr %argv) {
; LIN-DR-NEXT: .cfi_def_cfa_offset 48
; LIN-DR-NEXT: pushp %rbx
; LIN-DR-NEXT: .cfi_def_cfa_offset 56
-; LIN-DR-NEXT: {nf} subq $24, %rsp
+; LIN-DR-NEXT: subq $24, %rsp
; LIN-DR-NEXT: .cfi_def_cfa_offset 80
; LIN-DR-NEXT: .cfi_offset %rbx, -56
; LIN-DR-NEXT: .cfi_offset %r12, -48
@@ -150,7 +153,7 @@ define i32 @csr6_alloc16(ptr %argv) {
; LIN-DR-NEXT: xorl %ecx, %ecx
; LIN-DR-NEXT: xorl %eax, %eax
; LIN-DR-NEXT: callq *%rcx
-; LIN-DR-NEXT: {nf} addq $24, %rsp
+; LIN-DR-NEXT: addq $24, %rsp
; LIN-DR-NEXT: .cfi_def_cfa_offset 56
; LIN-DR-NEXT: popp %rbx
; LIN-DR-NEXT: .cfi_def_cfa_offset 48
@@ -278,7 +281,7 @@ define i32 @csr6_alloc16(ptr %argv) {
; WIN-DR-NEXT: .seh_pushreg %rbp
; WIN-DR-NEXT: pushp %rbx
; WIN-DR-NEXT: .seh_pushreg %rbx
-; WIN-DR-NEXT: {nf} subq $56, %rsp
+; WIN-DR-NEXT: subq $56, %rsp
; WIN-DR-NEXT: .seh_stackalloc 56
; WIN-DR-NEXT: .seh_endprologue
; WIN-DR-NEXT: #APP
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2.ll b/llvm/test/CodeGen/X86/apx/push2-pop2.ll
index ec13253c08e2f..abac95b35adce 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2.ll
@@ -432,7 +432,7 @@ define void @lea_in_epilog(i1 %arg, ptr %arg1, ptr %arg2, i64 %arg3, i64 %arg4,
; PPX-NF-NEXT: push2p %r14, %r15
; PPX-NF-NEXT: push2p %r12, %r13
; PPX-NF-NEXT: pushp %rbx
-; PPX-NF-NEXT: {nf} subq $24, %rsp
+; PPX-NF-NEXT: subq $24, %rsp
; PPX-NF-NEXT: movq %r9, %r14
; PPX-NF-NEXT: movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
; PPX-NF-NEXT: addq {{[0-9]+}}(%rsp), %r14
diff --git a/llvm/test/CodeGen/X86/apx/sub.ll b/llvm/test/CodeGen/X86/apx/sub.ll
index d999795cb848f..34af966465d93 100644
--- a/llvm/test/CodeGen/X86/apx/sub.ll
+++ b/llvm/test/CodeGen/X86/apx/sub.ll
@@ -1182,7 +1182,7 @@ define fastcc void @fold_with_physical_reg(ptr %arg0, i64 %arg1, ptr %arg2, i1 %
; NF-NEXT: pushq %r13 # encoding: [0x41,0x55]
; NF-NEXT: pushq %r12 # encoding: [0x41,0x54]
; NF-NEXT: pushq %rbx # encoding: [0x53]
-; NF-NEXT: {nf} subq $24, %rsp # encoding: [0x62,0xf4,0xfc,0x0c,0x83,0xec,0x18]
+; NF-NEXT: subq $24, %rsp # encoding: [0x48,0x83,0xec,0x18]
; NF-NEXT: movl %ecx, %r14d # encoding: [0x41,0x89,0xce]
; NF-NEXT: movq %rdx, %rbx # encoding: [0x48,0x89,0xd3]
; NF-NEXT: movq %rsi, %r12 # encoding: [0x49,0x89,0xf4]
More information about the llvm-commits
mailing list