[llvm] [X86][APX] Implement push+push2+push pre-alignment strategy for PP2 (PR #205031)

Feng Zou via llvm-commits llvm-commits at lists.llvm.org
Tue Jun 23 21:25:30 PDT 2026


https://github.com/fzou1 updated https://github.com/llvm/llvm-project/pull/205031

>From ce30358cf3a9fb7ad954050ffb284f0bfe72ee7b Mon Sep 17 00:00:00 2001
From: "Zou, Feng" <feng.zou at intel.com>
Date: Mon, 22 Jun 2026 05:53:35 +0200
Subject: [PATCH 1/3] [X86][APX] Implement push+push2+push pre-alignment
 strategy for PP2

Replace the dummy "push %rax" stack-alignment padding for APX push2/pop2
(PP2) with a push+push2+push strategy: when an even number of callee-saved
GPRs is involved, a single CSR push provides the 16-byte alignment instead
of a throwaway push %rax, and the remaining registers use push2/pop2. The
padForPush2Pop2 flag and its associated dummy push, SUB/LEA padding, and
SEH_StackAlloc emission in spill/restoreCalleeSavedRegisters are removed.

BuildStackAdjustment now uses NF (no-flags) variants of ADD/SUB when the
target supports them, avoiding gratuitous EFLAGS clobber. NF is disabled on
Win64 prologues because the OS epilogue unwinder does not recognize
EVEX-encoded instructions; those fall back to LEA, which it does recognize.
ADD64ri32_NF/SUB64ri32_NF are recognized in mergeSPUpdates and the epilogue
backward scan.

Fix the epilogue DWARF CFI loop to emit .cfi_def_cfa_offset for FrameDestroy
ADD/LEA/NF stack adjustments (previously only POP opcodes were handled),
reading the actual immediate from the instruction rather than assuming a
fixed amount.

Update LIT tests accordingly.

Assisted-by: Claude Opus 4.8 (1M context) <noreply at anthropic.com>
---
 llvm/lib/Target/X86/X86FrameLowering.cpp      |  64 ++---
 llvm/test/CodeGen/X86/apx/add.ll              |   2 +-
 .../apx/pp2-with-stack-clash-protection.ll    |  36 +--
 .../CodeGen/X86/apx/push2-pop2-cfi-seh-v3.ll  |  80 +++---
 .../CodeGen/X86/apx/push2-pop2-cfi-seh.ll     | 159 +++++++-----
 llvm/test/CodeGen/X86/apx/push2-pop2.ll       | 231 +++++++++++-------
 llvm/test/CodeGen/X86/apx/sub.ll              |   2 +-
 llvm/test/CodeGen/X86/apx/win64-abi.ll        |   4 +-
 .../X86/win64-eh-unwindv3-push2pop2.ll        |  32 +--
 9 files changed, 352 insertions(+), 258 deletions(-)

diff --git a/llvm/lib/Target/X86/X86FrameLowering.cpp b/llvm/lib/Target/X86/X86FrameLowering.cpp
index 8fe09bea456c8..b703fc24b6a6f 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.cpp
+++ b/llvm/lib/Target/X86/X86FrameLowering.cpp
@@ -383,7 +383,11 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
   }
 
   MachineInstrBuilder MI;
-  if (UseLEA) {
+  // Use NF (no-flags) variants when the target supports it, avoiding
+  // gratuitous EFLAGS clobber. Not on Windows where the OS epilogue unwinder
+  // doesn't recognize EVEX-encoded instructions.
+  bool UseNF = STI.hasNF() && !isWin64Prologue(*MBB.getParent());
+  if (UseLEA && !UseNF) {
     MI = addRegOffset(BuildMI(MBB, MBBI, DL,
                               TII.get(getLEArOpcode(Uses64BitFramePtr)),
                               StackPtr),
@@ -391,12 +395,16 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
   } else {
     bool IsSub = Offset < 0;
     uint64_t AbsOffset = IsSub ? -Offset : Offset;
-    const unsigned Opc = IsSub ? getSUBriOpcode(Uses64BitFramePtr)
-                               : getADDriOpcode(Uses64BitFramePtr);
+    const unsigned Opc =
+        IsSub ? (UseNF ? (unsigned)X86::SUB64ri32_NF
+                       : getSUBriOpcode(Uses64BitFramePtr))
+              : (UseNF ? (unsigned)X86::ADD64ri32_NF
+                       : getADDriOpcode(Uses64BitFramePtr));
     MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
              .addReg(StackPtr)
              .addImm(AbsOffset);
-    MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
+    if (!UseNF)
+      MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
   }
   return MI;
 }
@@ -432,7 +440,8 @@ int64_t X86FrameLowering::mergeSPUpdates(MachineBasicBlock &MBB,
   for (;;) {
     unsigned Opc = PI->getOpcode();
 
-    if ((Opc == X86::ADD64ri32 || Opc == X86::ADD32ri) &&
+    if ((Opc == X86::ADD64ri32 || Opc == X86::ADD32ri ||
+         Opc == X86::ADD64ri32_NF) &&
         PI->getOperand(0).getReg() == StackPtr) {
       assert(PI->getOperand(1).getReg() == StackPtr);
       Offset = PI->getOperand(2).getImm();
@@ -444,7 +453,8 @@ int64_t X86FrameLowering::mergeSPUpdates(MachineBasicBlock &MBB,
                PI->getOperand(5).getReg() == X86::NoRegister) {
       // For LEAs we have: def = lea SP, FI, noreg, Offset, noreg.
       Offset = PI->getOperand(4).getImm();
-    } else if ((Opc == X86::SUB64ri32 || Opc == X86::SUB32ri) &&
+    } else if ((Opc == X86::SUB64ri32 || Opc == X86::SUB32ri ||
+                Opc == X86::SUB64ri32_NF) &&
                PI->getOperand(0).getReg() == StackPtr) {
       assert(PI->getOperand(1).getReg() == StackPtr);
       Offset = -PI->getOperand(2).getImm();
@@ -2625,7 +2635,8 @@ void X86FrameLowering::emitEpilogue(MachineFunction &MF,
           (Opc != X86::POP32r && Opc != X86::POP64r && Opc != X86::BTR64ri8 &&
            Opc != X86::ADD64ri32 && Opc != X86::POPP64r && Opc != X86::POP2 &&
            Opc != X86::POP2P && Opc != X86::LEA64r && Opc != X86::SEH_PushReg &&
-           Opc != X86::SEH_Push2Regs && Opc != X86::SEH_StackAlloc))
+           Opc != X86::SEH_Push2Regs && Opc != X86::SEH_StackAlloc &&
+           Opc != X86::ADD64ri32_NF))
         break;
       FirstCSPop = PI;
     }
@@ -2772,6 +2783,17 @@ void X86FrameLowering::emitEpilogue(MachineFunction &MF,
         BuildCFI(MBB, MBBI, DL,
                  MCCFIInstruction::cfiDefCfaOffset(nullptr, -Offset),
                  MachineInstr::FrameDestroy);
+      } else if (PI->getFlag(MachineInstr::FrameDestroy) &&
+                 PI->getOperand(0).getReg() == StackPtr &&
+                 (Opc == X86::ADD64ri32 || Opc == X86::ADD64ri32_NF ||
+                  Opc == X86::ADD32ri || Opc == X86::LEA64r)) {
+        int64_t SPAdj = Opc == X86::LEA64r
+                            ? PI->getOperand(4).getImm()
+                            : PI->getOperand(2).getImm();
+        Offset += SPAdj;
+        BuildCFI(MBB, MBBI, DL,
+                 MCCFIInstruction::cfiDefCfaOffset(nullptr, -Offset),
+                 MachineInstr::FrameDestroy);
       }
     }
   }
@@ -3029,6 +3051,7 @@ bool X86FrameLowering::assignCalleeSavedSpillSlots(
     }
   }
 
+  bool IsFPRemovedFromCSI = false;
   if (hasFP(MF)) {
     // emitPrologue always spills frame register the first thing.
     SpillSlotOffset -= SlotSize;
@@ -3049,6 +3072,7 @@ bool X86FrameLowering::assignCalleeSavedSpillSlots(
     for (unsigned i = 0; i < CSI.size(); ++i) {
       if (TRI->regsOverlap(CSI[i].getReg(), FPReg)) {
         CSI.erase(CSI.begin() + i);
+        IsFPRemovedFromCSI = true;
         break;
       }
     }
@@ -3069,14 +3093,11 @@ bool X86FrameLowering::assignCalleeSavedSpillSlots(
     unsigned NumCSGPR = llvm::count_if(CSI, [](const CalleeSavedInfo &I) {
       return X86::GR64RegClass.contains(I.getReg());
     });
-    bool NeedPadding = (SpillSlotOffset % 16 != 0) && (NumCSGPR % 2 == 0);
-    bool UsePush2Pop2 = NeedPadding ? NumCSGPR > 2 : NumCSGPR > 1;
-    X86FI->setPadForPush2Pop2(NeedPadding && UsePush2Pop2);
-    NumRegsForPush2 = UsePush2Pop2 ? alignDown(NumCSGPR, 2) : 0;
-    if (X86FI->padForPush2Pop2()) {
-      SpillSlotOffset -= SlotSize;
-      MFI.CreateFixedSpillStackObject(SlotSize, SpillSlotOffset);
-    }
+    bool UsePush2Pop2 = !IsFPRemovedFromCSI ? NumCSGPR > 2 : NumCSGPR > 1;
+    NumRegsForPush2 =
+        UsePush2Pop2
+            ? alignDown(IsFPRemovedFromCSI ? NumCSGPR : NumCSGPR - 1, 2)
+            : 0;
   }
 
   // Assign slots for GPRs. It increases frame size.
@@ -3159,12 +3180,6 @@ bool X86FrameLowering::spillCalleeSavedRegisters(
   // Push GPRs. It increases frame size.
   const MachineFunction &MF = *MBB.getParent();
   const X86MachineFunctionInfo *X86FI = MF.getInfo<X86MachineFunctionInfo>();
-  if (X86FI->padForPush2Pop2()) {
-    assert(SlotSize == 8 && "Unexpected slot size for padding!");
-    BuildMI(MBB, MI, DL, TII.get(X86::PUSH64r))
-        .addReg(X86::RAX, RegState::Undef)
-        .setMIFlag(MachineInstr::FrameSetup);
-  }
 
   // Update LiveIn of the basic block and decide whether we can add a kill flag
   // to the use.
@@ -3343,13 +3358,6 @@ bool X86FrameLowering::restoreCalleeSavedRegisters(
           .setMIFlag(MachineInstr::FrameDestroy);
     }
   }
-  if (X86FI->padForPush2Pop2()) {
-    if (IsWin64UnwindV3)
-      BuildMI(MBB, MI, DL, TII.get(X86::SEH_StackAlloc))
-          .addImm(SlotSize)
-          .setMIFlag(MachineInstr::FrameDestroy);
-    emitSPUpdate(MBB, MI, DL, SlotSize, /*InEpilogue=*/true);
-  }
 
   return true;
 }
diff --git a/llvm/test/CodeGen/X86/apx/add.ll b/llvm/test/CodeGen/X86/apx/add.ll
index d7c5635b617c1..45c56645237d1 100644
--- a/llvm/test/CodeGen/X86/apx/add.ll
+++ b/llvm/test/CodeGen/X86/apx/add.ll
@@ -1248,7 +1248,7 @@ define i32 @two_address_no_subreg(i32 %arg0, ptr %arg1, i1 %arg2) nounwind {
 ; NF-NEXT:    xorl %ecx, %ecx # encoding: [0x31,0xc9]
 ; NF-NEXT:    callq *%rax # encoding: [0xff,0xd0]
 ; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    addq $8, %rsp # encoding: [0x48,0x83,0xc4,0x08]
+; NF-NEXT:    {nf} addq $8, %rsp # encoding: [0x62,0xf4,0xfc,0x0c,0x83,0xc4,0x08]
 ; NF-NEXT:    popq %rbx # encoding: [0x5b]
 ; NF-NEXT:    popq %r12 # encoding: [0x41,0x5c]
 ; NF-NEXT:    popq %r13 # encoding: [0x41,0x5d]
diff --git a/llvm/test/CodeGen/X86/apx/pp2-with-stack-clash-protection.ll b/llvm/test/CodeGen/X86/apx/pp2-with-stack-clash-protection.ll
index a7795f0b6bc5d..7dc97b14da812 100644
--- a/llvm/test/CodeGen/X86/apx/pp2-with-stack-clash-protection.ll
+++ b/llvm/test/CodeGen/X86/apx/pp2-with-stack-clash-protection.ll
@@ -8,22 +8,22 @@
 define i32 @foo(ptr %src1, i32 %len, ptr %src2, ptr %dst, i1 %cmp, i32 %sub) #0 {
 ; CHECK-LABEL: foo:
 ; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    pushq %rax
+; CHECK-NEXT:    pushq %rbp
 ; CHECK-NEXT:    .cfi_def_cfa_offset 16
-; CHECK-NEXT:    push2 %r15, %rbp
+; CHECK-NEXT:    push2 %r14, %r15
 ; CHECK-NEXT:    .cfi_def_cfa_offset 32
-; CHECK-NEXT:    push2 %r13, %r14
+; CHECK-NEXT:    push2 %r12, %r13
 ; CHECK-NEXT:    .cfi_def_cfa_offset 48
-; CHECK-NEXT:    push2 %rbx, %r12
+; CHECK-NEXT:    pushq %rbx
+; CHECK-NEXT:    .cfi_def_cfa_offset 56
+; CHECK-NEXT:    pushq %rax
 ; CHECK-NEXT:    .cfi_def_cfa_offset 64
-; CHECK-NEXT:    subq $16, %rsp
-; CHECK-NEXT:    .cfi_def_cfa_offset 80
-; CHECK-NEXT:    .cfi_offset %rbx, -64
-; CHECK-NEXT:    .cfi_offset %r12, -56
-; CHECK-NEXT:    .cfi_offset %r13, -48
-; CHECK-NEXT:    .cfi_offset %r14, -40
-; CHECK-NEXT:    .cfi_offset %r15, -32
-; CHECK-NEXT:    .cfi_offset %rbp, -24
+; CHECK-NEXT:    .cfi_offset %rbx, -56
+; CHECK-NEXT:    .cfi_offset %r12, -48
+; CHECK-NEXT:    .cfi_offset %r13, -40
+; CHECK-NEXT:    .cfi_offset %r14, -32
+; CHECK-NEXT:    .cfi_offset %r15, -24
+; CHECK-NEXT:    .cfi_offset %rbp, -16
 ; CHECK-NEXT:    movl %r9d, {{[-0-9]+}}(%r{{[sb]}}p) # 4-byte Spill
 ; CHECK-NEXT:    movl %r8d, %ebp
 ; CHECK-NEXT:    movq %rcx, %r14
@@ -44,15 +44,15 @@ define i32 @foo(ptr %src1, i32 %len, ptr %src2, ptr %dst, i1 %cmp, i32 %sub) #0
 ; CHECK-NEXT:    jne .LBB0_1
 ; CHECK-NEXT:  # %bb.2: # %while.end
 ; CHECK-NEXT:    movl {{[-0-9]+}}(%r{{[sb]}}p), %eax # 4-byte Reload
-; CHECK-NEXT:    addq $16, %rsp
-; CHECK-NEXT:    .cfi_def_cfa_offset 64
-; CHECK-NEXT:    pop2 %r12, %rbx
+; CHECK-NEXT:    addq $8, %rsp
+; CHECK-NEXT:    .cfi_def_cfa_offset 56
+; CHECK-NEXT:    popq %rbx
 ; CHECK-NEXT:    .cfi_def_cfa_offset 48
-; CHECK-NEXT:    pop2 %r14, %r13
+; CHECK-NEXT:    pop2 %r13, %r12
 ; CHECK-NEXT:    .cfi_def_cfa_offset 32
-; CHECK-NEXT:    pop2 %rbp, %r15
+; CHECK-NEXT:    pop2 %r15, %r14
 ; CHECK-NEXT:    .cfi_def_cfa_offset 16
-; CHECK-NEXT:    popq %rcx
+; CHECK-NEXT:    popq %rbp
 ; CHECK-NEXT:    .cfi_def_cfa_offset 8
 ; CHECK-NEXT:    retq
 entry:
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh-v3.ll b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh-v3.ll
index 671146a8daea4..a10f554821445 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh-v3.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh-v3.ll
@@ -52,16 +52,16 @@ define i32 @csr6_alloc16(ptr %argv) {
 ;
 ; WIN-V3-LABEL: csr6_alloc16:
 ; WIN-V3:       # %bb.0: # %entry
-; WIN-V3-NEXT:    .seh_pushreg %rax
-; WIN-V3-NEXT:    pushq %rax
-; WIN-V3-NEXT:    .seh_push2regs %r15, %r14
-; WIN-V3-NEXT:    push2 %r14, %r15
-; WIN-V3-NEXT:    .seh_push2regs %r13, %r12
-; WIN-V3-NEXT:    push2 %r12, %r13
-; WIN-V3-NEXT:    .seh_push2regs %rbp, %rbx
-; WIN-V3-NEXT:    push2 %rbx, %rbp
-; WIN-V3-NEXT:    .seh_stackalloc 64
-; WIN-V3-NEXT:    subq $64, %rsp
+; WIN-V3-NEXT:    .seh_pushreg %r15
+; WIN-V3-NEXT:    pushq %r15
+; WIN-V3-NEXT:    .seh_push2regs %r14, %r13
+; WIN-V3-NEXT:    push2 %r13, %r14
+; WIN-V3-NEXT:    .seh_push2regs %r12, %rbp
+; WIN-V3-NEXT:    push2 %rbp, %r12
+; WIN-V3-NEXT:    .seh_pushreg %rbx
+; WIN-V3-NEXT:    pushq %rbx
+; WIN-V3-NEXT:    .seh_stackalloc 56
+; WIN-V3-NEXT:    subq $56, %rsp
 ; WIN-V3-NEXT:    .seh_endprologue
 ; WIN-V3-NEXT:    #APP
 ; WIN-V3-NEXT:    #NO_APP
@@ -69,32 +69,32 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; WIN-V3-NEXT:    callq *%rax
 ; WIN-V3-NEXT:    nop
 ; WIN-V3-NEXT:    .seh_startepilogue
-; WIN-V3-NEXT:    .seh_stackalloc 64
-; WIN-V3-NEXT:    addq $64, %rsp
-; WIN-V3-NEXT:    .seh_push2regs %rbx, %rbp
-; WIN-V3-NEXT:    pop2 %rbp, %rbx
-; WIN-V3-NEXT:    .seh_push2regs %r12, %r13
-; WIN-V3-NEXT:    pop2 %r13, %r12
-; WIN-V3-NEXT:    .seh_push2regs %r14, %r15
-; WIN-V3-NEXT:    pop2 %r15, %r14
-; WIN-V3-NEXT:    .seh_stackalloc 8
-; WIN-V3-NEXT:    popq %rax
+; WIN-V3-NEXT:    .seh_stackalloc 56
+; WIN-V3-NEXT:    addq $56, %rsp
+; WIN-V3-NEXT:    .seh_pushreg %rbx
+; WIN-V3-NEXT:    popq %rbx
+; WIN-V3-NEXT:    .seh_push2regs %rbp, %r12
+; WIN-V3-NEXT:    pop2 %r12, %rbp
+; WIN-V3-NEXT:    .seh_push2regs %r13, %r14
+; WIN-V3-NEXT:    pop2 %r14, %r13
+; WIN-V3-NEXT:    .seh_pushreg %r15
+; WIN-V3-NEXT:    popq %r15
 ; WIN-V3-NEXT:    .seh_endepilogue
 ; WIN-V3-NEXT:    retq
 ; WIN-V3-NEXT:    .seh_endproc
 ;
 ; WIN-V3-PPX-LABEL: csr6_alloc16:
 ; WIN-V3-PPX:       # %bb.0: # %entry
-; WIN-V3-PPX-NEXT:    .seh_pushreg %rax
-; WIN-V3-PPX-NEXT:    pushq %rax
-; WIN-V3-PPX-NEXT:    .seh_push2regs %r15, %r14
-; WIN-V3-PPX-NEXT:    push2p %r14, %r15
-; WIN-V3-PPX-NEXT:    .seh_push2regs %r13, %r12
-; WIN-V3-PPX-NEXT:    push2p %r12, %r13
-; WIN-V3-PPX-NEXT:    .seh_push2regs %rbp, %rbx
-; WIN-V3-PPX-NEXT:    push2p %rbx, %rbp
-; WIN-V3-PPX-NEXT:    .seh_stackalloc 64
-; WIN-V3-PPX-NEXT:    subq $64, %rsp
+; WIN-V3-PPX-NEXT:    .seh_pushreg %r15
+; WIN-V3-PPX-NEXT:    pushp %r15
+; WIN-V3-PPX-NEXT:    .seh_push2regs %r14, %r13
+; WIN-V3-PPX-NEXT:    push2p %r13, %r14
+; WIN-V3-PPX-NEXT:    .seh_push2regs %r12, %rbp
+; WIN-V3-PPX-NEXT:    push2p %rbp, %r12
+; WIN-V3-PPX-NEXT:    .seh_pushreg %rbx
+; WIN-V3-PPX-NEXT:    pushp %rbx
+; WIN-V3-PPX-NEXT:    .seh_stackalloc 56
+; WIN-V3-PPX-NEXT:    subq $56, %rsp
 ; WIN-V3-PPX-NEXT:    .seh_endprologue
 ; WIN-V3-PPX-NEXT:    #APP
 ; WIN-V3-PPX-NEXT:    #NO_APP
@@ -102,16 +102,16 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; WIN-V3-PPX-NEXT:    callq *%rax
 ; WIN-V3-PPX-NEXT:    nop
 ; WIN-V3-PPX-NEXT:    .seh_startepilogue
-; WIN-V3-PPX-NEXT:    .seh_stackalloc 64
-; WIN-V3-PPX-NEXT:    addq $64, %rsp
-; WIN-V3-PPX-NEXT:    .seh_push2regs %rbx, %rbp
-; WIN-V3-PPX-NEXT:    pop2p %rbp, %rbx
-; WIN-V3-PPX-NEXT:    .seh_push2regs %r12, %r13
-; WIN-V3-PPX-NEXT:    pop2p %r13, %r12
-; WIN-V3-PPX-NEXT:    .seh_push2regs %r14, %r15
-; WIN-V3-PPX-NEXT:    pop2p %r15, %r14
-; WIN-V3-PPX-NEXT:    .seh_stackalloc 8
-; WIN-V3-PPX-NEXT:    popq %rax
+; WIN-V3-PPX-NEXT:    .seh_stackalloc 56
+; WIN-V3-PPX-NEXT:    addq $56, %rsp
+; WIN-V3-PPX-NEXT:    .seh_pushreg %rbx
+; WIN-V3-PPX-NEXT:    popp %rbx
+; WIN-V3-PPX-NEXT:    .seh_push2regs %rbp, %r12
+; WIN-V3-PPX-NEXT:    pop2p %r12, %rbp
+; WIN-V3-PPX-NEXT:    .seh_push2regs %r13, %r14
+; WIN-V3-PPX-NEXT:    pop2p %r14, %r13
+; WIN-V3-PPX-NEXT:    .seh_pushreg %r15
+; WIN-V3-PPX-NEXT:    popp %r15
 ; WIN-V3-PPX-NEXT:    .seh_endepilogue
 ; WIN-V3-PPX-NEXT:    retq
 ; WIN-V3-PPX-NEXT:    .seh_endproc
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
index aabaa968a11b9..5d25f8826cddd 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
@@ -2,7 +2,7 @@
 ; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu | FileCheck %s --check-prefix=LIN-REF
 ; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+push2pop2 | FileCheck %s --check-prefix=LIN
 ; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+push2pop2,+ppx | FileCheck %s --check-prefix=LIN-PPX
-; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mcpu=diamondrapids | FileCheck %s --check-prefix=LIN-PPX
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mcpu=diamondrapids | FileCheck %s --check-prefix=LIN-DR
 ; RUN: llc < %s -mtriple=x86_64-windows-msvc | FileCheck %s --check-prefix=WIN-REF
 ; RUN: llc < %s -mtriple=x86_64-windows-msvc -mattr=+push2pop2 | FileCheck %s --check-prefix=WIN
 ; RUN: llc < %s -mtriple=x86_64-windows-msvc -mattr=+push2pop2,+ppx | FileCheck %s --check-prefix=WIN-PPX
@@ -58,74 +58,109 @@ define i32 @csr6_alloc16(ptr %argv) {
 ;
 ; LIN-LABEL: csr6_alloc16:
 ; LIN:       # %bb.0: # %entry
-; LIN-NEXT:    pushq %rax
+; LIN-NEXT:    pushq %rbp
 ; LIN-NEXT:    .cfi_def_cfa_offset 16
-; LIN-NEXT:    push2 %r15, %rbp
+; LIN-NEXT:    push2 %r14, %r15
 ; LIN-NEXT:    .cfi_def_cfa_offset 32
-; LIN-NEXT:    push2 %r13, %r14
+; LIN-NEXT:    push2 %r12, %r13
 ; LIN-NEXT:    .cfi_def_cfa_offset 48
-; LIN-NEXT:    push2 %rbx, %r12
-; LIN-NEXT:    .cfi_def_cfa_offset 64
-; LIN-NEXT:    subq $32, %rsp
-; LIN-NEXT:    .cfi_def_cfa_offset 96
-; LIN-NEXT:    .cfi_offset %rbx, -64
-; LIN-NEXT:    .cfi_offset %r12, -56
-; LIN-NEXT:    .cfi_offset %r13, -48
-; LIN-NEXT:    .cfi_offset %r14, -40
-; LIN-NEXT:    .cfi_offset %r15, -32
-; LIN-NEXT:    .cfi_offset %rbp, -24
+; LIN-NEXT:    pushq %rbx
+; LIN-NEXT:    .cfi_def_cfa_offset 56
+; LIN-NEXT:    subq $24, %rsp
+; LIN-NEXT:    .cfi_def_cfa_offset 80
+; LIN-NEXT:    .cfi_offset %rbx, -56
+; LIN-NEXT:    .cfi_offset %r12, -48
+; LIN-NEXT:    .cfi_offset %r13, -40
+; LIN-NEXT:    .cfi_offset %r14, -32
+; LIN-NEXT:    .cfi_offset %r15, -24
+; LIN-NEXT:    .cfi_offset %rbp, -16
 ; LIN-NEXT:    #APP
 ; LIN-NEXT:    #NO_APP
 ; LIN-NEXT:    xorl %ecx, %ecx
 ; LIN-NEXT:    xorl %eax, %eax
 ; LIN-NEXT:    callq *%rcx
-; LIN-NEXT:    addq $32, %rsp
-; LIN-NEXT:    .cfi_def_cfa_offset 64
-; LIN-NEXT:    pop2 %r12, %rbx
+; LIN-NEXT:    addq $24, %rsp
+; LIN-NEXT:    .cfi_def_cfa_offset 56
+; LIN-NEXT:    popq %rbx
 ; LIN-NEXT:    .cfi_def_cfa_offset 48
-; LIN-NEXT:    pop2 %r14, %r13
+; LIN-NEXT:    pop2 %r13, %r12
 ; LIN-NEXT:    .cfi_def_cfa_offset 32
-; LIN-NEXT:    pop2 %rbp, %r15
+; LIN-NEXT:    pop2 %r15, %r14
 ; LIN-NEXT:    .cfi_def_cfa_offset 16
-; LIN-NEXT:    popq %rax
+; LIN-NEXT:    popq %rbp
 ; LIN-NEXT:    .cfi_def_cfa_offset 8
 ; LIN-NEXT:    retq
 ;
 ; LIN-PPX-LABEL: csr6_alloc16:
 ; LIN-PPX:       # %bb.0: # %entry
-; LIN-PPX-NEXT:    pushq %rax
+; LIN-PPX-NEXT:    pushp %rbp
 ; LIN-PPX-NEXT:    .cfi_def_cfa_offset 16
-; LIN-PPX-NEXT:    push2p %r15, %rbp
+; LIN-PPX-NEXT:    push2p %r14, %r15
 ; LIN-PPX-NEXT:    .cfi_def_cfa_offset 32
-; LIN-PPX-NEXT:    push2p %r13, %r14
+; LIN-PPX-NEXT:    push2p %r12, %r13
 ; LIN-PPX-NEXT:    .cfi_def_cfa_offset 48
-; LIN-PPX-NEXT:    push2p %rbx, %r12
-; LIN-PPX-NEXT:    .cfi_def_cfa_offset 64
-; LIN-PPX-NEXT:    subq $32, %rsp
-; LIN-PPX-NEXT:    .cfi_def_cfa_offset 96
-; LIN-PPX-NEXT:    .cfi_offset %rbx, -64
-; LIN-PPX-NEXT:    .cfi_offset %r12, -56
-; LIN-PPX-NEXT:    .cfi_offset %r13, -48
-; LIN-PPX-NEXT:    .cfi_offset %r14, -40
-; LIN-PPX-NEXT:    .cfi_offset %r15, -32
-; LIN-PPX-NEXT:    .cfi_offset %rbp, -24
+; LIN-PPX-NEXT:    pushp %rbx
+; LIN-PPX-NEXT:    .cfi_def_cfa_offset 56
+; LIN-PPX-NEXT:    subq $24, %rsp
+; LIN-PPX-NEXT:    .cfi_def_cfa_offset 80
+; LIN-PPX-NEXT:    .cfi_offset %rbx, -56
+; LIN-PPX-NEXT:    .cfi_offset %r12, -48
+; LIN-PPX-NEXT:    .cfi_offset %r13, -40
+; LIN-PPX-NEXT:    .cfi_offset %r14, -32
+; LIN-PPX-NEXT:    .cfi_offset %r15, -24
+; LIN-PPX-NEXT:    .cfi_offset %rbp, -16
 ; LIN-PPX-NEXT:    #APP
 ; LIN-PPX-NEXT:    #NO_APP
 ; LIN-PPX-NEXT:    xorl %ecx, %ecx
 ; LIN-PPX-NEXT:    xorl %eax, %eax
 ; LIN-PPX-NEXT:    callq *%rcx
-; LIN-PPX-NEXT:    addq $32, %rsp
-; LIN-PPX-NEXT:    .cfi_def_cfa_offset 64
-; LIN-PPX-NEXT:    pop2p %r12, %rbx
+; LIN-PPX-NEXT:    addq $24, %rsp
+; LIN-PPX-NEXT:    .cfi_def_cfa_offset 56
+; LIN-PPX-NEXT:    popp %rbx
 ; LIN-PPX-NEXT:    .cfi_def_cfa_offset 48
-; LIN-PPX-NEXT:    pop2p %r14, %r13
+; LIN-PPX-NEXT:    pop2p %r13, %r12
 ; LIN-PPX-NEXT:    .cfi_def_cfa_offset 32
-; LIN-PPX-NEXT:    pop2p %rbp, %r15
+; LIN-PPX-NEXT:    pop2p %r15, %r14
 ; LIN-PPX-NEXT:    .cfi_def_cfa_offset 16
-; LIN-PPX-NEXT:    popq %rax
+; LIN-PPX-NEXT:    popp %rbp
 ; LIN-PPX-NEXT:    .cfi_def_cfa_offset 8
 ; LIN-PPX-NEXT:    retq
 ;
+; LIN-DR-LABEL: csr6_alloc16:
+; LIN-DR:       # %bb.0: # %entry
+; LIN-DR-NEXT:    pushp %rbp
+; LIN-DR-NEXT:    .cfi_def_cfa_offset 16
+; LIN-DR-NEXT:    push2p %r14, %r15
+; LIN-DR-NEXT:    .cfi_def_cfa_offset 32
+; LIN-DR-NEXT:    push2p %r12, %r13
+; LIN-DR-NEXT:    .cfi_def_cfa_offset 48
+; LIN-DR-NEXT:    pushp %rbx
+; LIN-DR-NEXT:    .cfi_def_cfa_offset 56
+; LIN-DR-NEXT:    {nf} subq $24, %rsp
+; LIN-DR-NEXT:    .cfi_def_cfa_offset 80
+; LIN-DR-NEXT:    .cfi_offset %rbx, -56
+; LIN-DR-NEXT:    .cfi_offset %r12, -48
+; LIN-DR-NEXT:    .cfi_offset %r13, -40
+; LIN-DR-NEXT:    .cfi_offset %r14, -32
+; LIN-DR-NEXT:    .cfi_offset %r15, -24
+; LIN-DR-NEXT:    .cfi_offset %rbp, -16
+; LIN-DR-NEXT:    #APP
+; LIN-DR-NEXT:    #NO_APP
+; LIN-DR-NEXT:    xorl %ecx, %ecx
+; LIN-DR-NEXT:    xorl %eax, %eax
+; LIN-DR-NEXT:    callq *%rcx
+; LIN-DR-NEXT:    {nf} addq $24, %rsp
+; LIN-DR-NEXT:    .cfi_def_cfa_offset 56
+; LIN-DR-NEXT:    popp %rbx
+; LIN-DR-NEXT:    .cfi_def_cfa_offset 48
+; LIN-DR-NEXT:    pop2p %r13, %r12
+; LIN-DR-NEXT:    .cfi_def_cfa_offset 32
+; LIN-DR-NEXT:    pop2p %r15, %r14
+; LIN-DR-NEXT:    .cfi_def_cfa_offset 16
+; LIN-DR-NEXT:    popp %rbp
+; LIN-DR-NEXT:    .cfi_def_cfa_offset 8
+; LIN-DR-NEXT:    retq
+;
 ; WIN-REF-LABEL: csr6_alloc16:
 ; WIN-REF:       # %bb.0: # %entry
 ; WIN-REF-NEXT:    pushq %r15
@@ -162,19 +197,18 @@ define i32 @csr6_alloc16(ptr %argv) {
 ;
 ; WIN-LABEL: csr6_alloc16:
 ; WIN:       # %bb.0: # %entry
-; WIN-NEXT:    pushq %rax
-; WIN-NEXT:    .seh_pushreg %rax
-; WIN-NEXT:    push2 %r14, %r15
+; WIN-NEXT:    pushq %r15
 ; WIN-NEXT:    .seh_pushreg %r15
+; WIN-NEXT:    push2 %r13, %r14
 ; WIN-NEXT:    .seh_pushreg %r14
-; WIN-NEXT:    push2 %r12, %r13
 ; WIN-NEXT:    .seh_pushreg %r13
+; WIN-NEXT:    push2 %rbp, %r12
 ; WIN-NEXT:    .seh_pushreg %r12
-; WIN-NEXT:    push2 %rbx, %rbp
 ; WIN-NEXT:    .seh_pushreg %rbp
+; WIN-NEXT:    pushq %rbx
 ; WIN-NEXT:    .seh_pushreg %rbx
-; WIN-NEXT:    subq $64, %rsp
-; WIN-NEXT:    .seh_stackalloc 64
+; WIN-NEXT:    subq $56, %rsp
+; WIN-NEXT:    .seh_stackalloc 56
 ; WIN-NEXT:    .seh_endprologue
 ; WIN-NEXT:    #APP
 ; WIN-NEXT:    #NO_APP
@@ -182,30 +216,29 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; WIN-NEXT:    callq *%rax
 ; WIN-NEXT:    nop
 ; WIN-NEXT:    .seh_startepilogue
-; WIN-NEXT:    addq $64, %rsp
-; WIN-NEXT:    pop2 %rbp, %rbx
-; WIN-NEXT:    pop2 %r13, %r12
-; WIN-NEXT:    pop2 %r15, %r14
-; WIN-NEXT:    popq %rax
+; WIN-NEXT:    addq $56, %rsp
+; WIN-NEXT:    popq %rbx
+; WIN-NEXT:    pop2 %r12, %rbp
+; WIN-NEXT:    pop2 %r14, %r13
+; WIN-NEXT:    popq %r15
 ; WIN-NEXT:    .seh_endepilogue
 ; WIN-NEXT:    retq
 ; WIN-NEXT:    .seh_endproc
 ;
 ; WIN-PPX-LABEL: csr6_alloc16:
 ; WIN-PPX:       # %bb.0: # %entry
-; WIN-PPX-NEXT:    pushq %rax
-; WIN-PPX-NEXT:    .seh_pushreg %rax
-; WIN-PPX-NEXT:    push2p %r14, %r15
+; WIN-PPX-NEXT:    pushp %r15
 ; WIN-PPX-NEXT:    .seh_pushreg %r15
+; WIN-PPX-NEXT:    push2p %r13, %r14
 ; WIN-PPX-NEXT:    .seh_pushreg %r14
-; WIN-PPX-NEXT:    push2p %r12, %r13
 ; WIN-PPX-NEXT:    .seh_pushreg %r13
+; WIN-PPX-NEXT:    push2p %rbp, %r12
 ; WIN-PPX-NEXT:    .seh_pushreg %r12
-; WIN-PPX-NEXT:    push2p %rbx, %rbp
 ; WIN-PPX-NEXT:    .seh_pushreg %rbp
+; WIN-PPX-NEXT:    pushp %rbx
 ; WIN-PPX-NEXT:    .seh_pushreg %rbx
-; WIN-PPX-NEXT:    subq $64, %rsp
-; WIN-PPX-NEXT:    .seh_stackalloc 64
+; WIN-PPX-NEXT:    subq $56, %rsp
+; WIN-PPX-NEXT:    .seh_stackalloc 56
 ; WIN-PPX-NEXT:    .seh_endprologue
 ; WIN-PPX-NEXT:    #APP
 ; WIN-PPX-NEXT:    #NO_APP
@@ -213,11 +246,11 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; WIN-PPX-NEXT:    callq *%rax
 ; WIN-PPX-NEXT:    nop
 ; WIN-PPX-NEXT:    .seh_startepilogue
-; WIN-PPX-NEXT:    addq $64, %rsp
-; WIN-PPX-NEXT:    pop2p %rbp, %rbx
-; WIN-PPX-NEXT:    pop2p %r13, %r12
-; WIN-PPX-NEXT:    pop2p %r15, %r14
-; WIN-PPX-NEXT:    popq %rax
+; WIN-PPX-NEXT:    addq $56, %rsp
+; WIN-PPX-NEXT:    popp %rbx
+; WIN-PPX-NEXT:    pop2p %r12, %rbp
+; WIN-PPX-NEXT:    pop2p %r14, %r13
+; WIN-PPX-NEXT:    popp %r15
 ; WIN-PPX-NEXT:    .seh_endepilogue
 ; WIN-PPX-NEXT:    retq
 ; WIN-PPX-NEXT:    .seh_endproc
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2.ll b/llvm/test/CodeGen/X86/apx/push2-pop2.ll
index f5be484be2b1a..ec13253c08e2f 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2.ll
@@ -1,7 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
 ; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+push2pop2 | FileCheck %s --check-prefix=CHECK
-; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+push2pop2,+ppx | FileCheck %s --check-prefix=PPX
+; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+push2pop2,+ppx | FileCheck %s --check-prefixes=PPX,PPX-NO-NF
 ; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+push2pop2 -frame-pointer=all | FileCheck %s --check-prefix=FRAME
+; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+push2pop2,+ppx,+nf | FileCheck %s --check-prefixes=PPX,PPX-NF
 
 define void @csr1() nounwind {
 ; CHECK-LABEL: csr1:
@@ -120,26 +121,26 @@ entry:
 define void @csr4() nounwind {
 ; CHECK-LABEL: csr4:
 ; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    pushq %rax
-; CHECK-NEXT:    push2 %r15, %rbp
-; CHECK-NEXT:    push2 %r13, %r14
+; CHECK-NEXT:    pushq %rbp
+; CHECK-NEXT:    push2 %r14, %r15
+; CHECK-NEXT:    pushq %r13
 ; CHECK-NEXT:    #APP
 ; CHECK-NEXT:    #NO_APP
-; CHECK-NEXT:    pop2 %r14, %r13
-; CHECK-NEXT:    pop2 %rbp, %r15
-; CHECK-NEXT:    popq %rax
+; CHECK-NEXT:    popq %r13
+; CHECK-NEXT:    pop2 %r15, %r14
+; CHECK-NEXT:    popq %rbp
 ; CHECK-NEXT:    retq
 ;
 ; PPX-LABEL: csr4:
 ; PPX:       # %bb.0: # %entry
-; PPX-NEXT:    pushq %rax
-; PPX-NEXT:    push2p %r15, %rbp
-; PPX-NEXT:    push2p %r13, %r14
+; PPX-NEXT:    pushp %rbp
+; PPX-NEXT:    push2p %r14, %r15
+; PPX-NEXT:    pushp %r13
 ; PPX-NEXT:    #APP
 ; PPX-NEXT:    #NO_APP
-; PPX-NEXT:    pop2p %r14, %r13
-; PPX-NEXT:    pop2p %rbp, %r15
-; PPX-NEXT:    popq %rax
+; PPX-NEXT:    popp %r13
+; PPX-NEXT:    pop2p %r15, %r14
+; PPX-NEXT:    popp %rbp
 ; PPX-NEXT:    retq
 ;
 ; FRAME-LABEL: csr4:
@@ -212,30 +213,30 @@ entry:
 define void @csr6() nounwind {
 ; CHECK-LABEL: csr6:
 ; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    pushq %rax
-; CHECK-NEXT:    push2 %r15, %rbp
-; CHECK-NEXT:    push2 %r13, %r14
-; CHECK-NEXT:    push2 %rbx, %r12
+; CHECK-NEXT:    pushq %rbp
+; CHECK-NEXT:    push2 %r14, %r15
+; CHECK-NEXT:    push2 %r12, %r13
+; CHECK-NEXT:    pushq %rbx
 ; CHECK-NEXT:    #APP
 ; CHECK-NEXT:    #NO_APP
-; CHECK-NEXT:    pop2 %r12, %rbx
-; CHECK-NEXT:    pop2 %r14, %r13
-; CHECK-NEXT:    pop2 %rbp, %r15
-; CHECK-NEXT:    popq %rax
+; CHECK-NEXT:    popq %rbx
+; CHECK-NEXT:    pop2 %r13, %r12
+; CHECK-NEXT:    pop2 %r15, %r14
+; CHECK-NEXT:    popq %rbp
 ; CHECK-NEXT:    retq
 ;
 ; PPX-LABEL: csr6:
 ; PPX:       # %bb.0: # %entry
-; PPX-NEXT:    pushq %rax
-; PPX-NEXT:    push2p %r15, %rbp
-; PPX-NEXT:    push2p %r13, %r14
-; PPX-NEXT:    push2p %rbx, %r12
+; PPX-NEXT:    pushp %rbp
+; PPX-NEXT:    push2p %r14, %r15
+; PPX-NEXT:    push2p %r12, %r13
+; PPX-NEXT:    pushp %rbx
 ; PPX-NEXT:    #APP
 ; PPX-NEXT:    #NO_APP
-; PPX-NEXT:    pop2p %r12, %rbx
-; PPX-NEXT:    pop2p %r14, %r13
-; PPX-NEXT:    pop2p %rbp, %r15
-; PPX-NEXT:    popq %rax
+; PPX-NEXT:    popp %rbx
+; PPX-NEXT:    pop2p %r13, %r12
+; PPX-NEXT:    pop2p %r15, %r14
+; PPX-NEXT:    popp %rbp
 ; PPX-NEXT:    retq
 ;
 ; FRAME-LABEL: csr6:
@@ -269,11 +270,11 @@ define void @lea_in_epilog(i1 %arg, ptr %arg1, ptr %arg2, i64 %arg3, i64 %arg4,
 ; CHECK-NEXT:    testb $1, %dil
 ; CHECK-NEXT:    je .LBB6_5
 ; CHECK-NEXT:  # %bb.1: # %bb13
-; CHECK-NEXT:    pushq %rax
-; CHECK-NEXT:    push2 %r15, %rbp
-; CHECK-NEXT:    push2 %r13, %r14
-; CHECK-NEXT:    push2 %rbx, %r12
-; CHECK-NEXT:    subq $16, %rsp
+; CHECK-NEXT:    pushq %rbp
+; CHECK-NEXT:    push2 %r14, %r15
+; CHECK-NEXT:    push2 %r12, %r13
+; CHECK-NEXT:    pushq %rbx
+; CHECK-NEXT:    subq $24, %rsp
 ; CHECK-NEXT:    movq %r9, %r14
 ; CHECK-NEXT:    movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
 ; CHECK-NEXT:    addq {{[0-9]+}}(%rsp), %r14
@@ -306,67 +307,67 @@ define void @lea_in_epilog(i1 %arg, ptr %arg1, ptr %arg2, i64 %arg3, i64 %arg4,
 ; CHECK-NEXT:  # %bb.3: # %bb11
 ; CHECK-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
 ; CHECK-NEXT:    leaq {{[0-9]+}}(%rsp), %rsp
-; CHECK-NEXT:    pop2 %r12, %rbx
-; CHECK-NEXT:    pop2 %r14, %r13
-; CHECK-NEXT:    pop2 %rbp, %r15
-; CHECK-NEXT:    leaq {{[0-9]+}}(%rsp), %rsp
+; CHECK-NEXT:    popq %rbx
+; CHECK-NEXT:    pop2 %r13, %r12
+; CHECK-NEXT:    pop2 %r15, %r14
+; CHECK-NEXT:    popq %rbp
 ; CHECK-NEXT:    jne .LBB6_5
 ; CHECK-NEXT:  # %bb.4: # %bb12
 ; CHECK-NEXT:    movq $0, (%rax)
 ; CHECK-NEXT:  .LBB6_5: # %bb14
 ; CHECK-NEXT:    retq
 ;
-; PPX-LABEL: lea_in_epilog:
-; PPX:       # %bb.0: # %bb
-; PPX-NEXT:    testb $1, %dil
-; PPX-NEXT:    je .LBB6_5
-; PPX-NEXT:  # %bb.1: # %bb13
-; PPX-NEXT:    pushq %rax
-; PPX-NEXT:    push2p %r15, %rbp
-; PPX-NEXT:    push2p %r13, %r14
-; PPX-NEXT:    push2p %rbx, %r12
-; PPX-NEXT:    subq $16, %rsp
-; PPX-NEXT:    movq %r9, %r14
-; PPX-NEXT:    movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
-; PPX-NEXT:    addq {{[0-9]+}}(%rsp), %r14
-; PPX-NEXT:    movq {{[0-9]+}}(%rsp), %r13
-; PPX-NEXT:    addq %r14, %r13
-; PPX-NEXT:    movq {{[0-9]+}}(%rsp), %r15
-; PPX-NEXT:    addq %r14, %r15
-; PPX-NEXT:    movq {{[0-9]+}}(%rsp), %rbx
-; PPX-NEXT:    addq %r14, %rbx
-; PPX-NEXT:    xorl %ebp, %ebp
-; PPX-NEXT:    xorl %r12d, %r12d
-; PPX-NEXT:    movl %edi, {{[-0-9]+}}(%r{{[sb]}}p) # 4-byte Spill
-; PPX-NEXT:    .p2align 4
-; PPX-NEXT:  .LBB6_2: # %bb15
-; PPX-NEXT:    # =>This Inner Loop Header: Depth=1
-; PPX-NEXT:    incq %r12
-; PPX-NEXT:    movl $432, %edx # imm = 0x1B0
-; PPX-NEXT:    xorl %edi, %edi
-; PPX-NEXT:    movq %r15, %rsi
-; PPX-NEXT:    callq memcpy at PLT
-; PPX-NEXT:    movl {{[-0-9]+}}(%r{{[sb]}}p), %edi # 4-byte Reload
-; PPX-NEXT:    movq {{[0-9]+}}(%rsp), %rax
-; PPX-NEXT:    addq %rax, %r13
-; PPX-NEXT:    addq %rax, %r15
-; PPX-NEXT:    addq %rax, %rbx
-; PPX-NEXT:    addq %rax, %r14
-; PPX-NEXT:    addq $8, %rbp
-; PPX-NEXT:    testb $1, %dil
-; PPX-NEXT:    je .LBB6_2
-; PPX-NEXT:  # %bb.3: # %bb11
-; PPX-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
-; PPX-NEXT:    leaq {{[0-9]+}}(%rsp), %rsp
-; PPX-NEXT:    pop2p %r12, %rbx
-; PPX-NEXT:    pop2p %r14, %r13
-; PPX-NEXT:    pop2p %rbp, %r15
-; PPX-NEXT:    leaq {{[0-9]+}}(%rsp), %rsp
-; PPX-NEXT:    jne .LBB6_5
-; PPX-NEXT:  # %bb.4: # %bb12
-; PPX-NEXT:    movq $0, (%rax)
-; PPX-NEXT:  .LBB6_5: # %bb14
-; PPX-NEXT:    retq
+; PPX-NO-NF-LABEL: lea_in_epilog:
+; PPX-NO-NF:       # %bb.0: # %bb
+; PPX-NO-NF-NEXT:    testb $1, %dil
+; PPX-NO-NF-NEXT:    je .LBB6_5
+; PPX-NO-NF-NEXT:  # %bb.1: # %bb13
+; PPX-NO-NF-NEXT:    pushp %rbp
+; PPX-NO-NF-NEXT:    push2p %r14, %r15
+; PPX-NO-NF-NEXT:    push2p %r12, %r13
+; PPX-NO-NF-NEXT:    pushp %rbx
+; PPX-NO-NF-NEXT:    subq $24, %rsp
+; PPX-NO-NF-NEXT:    movq %r9, %r14
+; PPX-NO-NF-NEXT:    movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; PPX-NO-NF-NEXT:    addq {{[0-9]+}}(%rsp), %r14
+; PPX-NO-NF-NEXT:    movq {{[0-9]+}}(%rsp), %r13
+; PPX-NO-NF-NEXT:    addq %r14, %r13
+; PPX-NO-NF-NEXT:    movq {{[0-9]+}}(%rsp), %r15
+; PPX-NO-NF-NEXT:    addq %r14, %r15
+; PPX-NO-NF-NEXT:    movq {{[0-9]+}}(%rsp), %rbx
+; PPX-NO-NF-NEXT:    addq %r14, %rbx
+; PPX-NO-NF-NEXT:    xorl %ebp, %ebp
+; PPX-NO-NF-NEXT:    xorl %r12d, %r12d
+; PPX-NO-NF-NEXT:    movl %edi, {{[-0-9]+}}(%r{{[sb]}}p) # 4-byte Spill
+; PPX-NO-NF-NEXT:    .p2align 4
+; PPX-NO-NF-NEXT:  .LBB6_2: # %bb15
+; PPX-NO-NF-NEXT:    # =>This Inner Loop Header: Depth=1
+; PPX-NO-NF-NEXT:    incq %r12
+; PPX-NO-NF-NEXT:    movl $432, %edx # imm = 0x1B0
+; PPX-NO-NF-NEXT:    xorl %edi, %edi
+; PPX-NO-NF-NEXT:    movq %r15, %rsi
+; PPX-NO-NF-NEXT:    callq memcpy at PLT
+; PPX-NO-NF-NEXT:    movl {{[-0-9]+}}(%r{{[sb]}}p), %edi # 4-byte Reload
+; PPX-NO-NF-NEXT:    movq {{[0-9]+}}(%rsp), %rax
+; PPX-NO-NF-NEXT:    addq %rax, %r13
+; PPX-NO-NF-NEXT:    addq %rax, %r15
+; PPX-NO-NF-NEXT:    addq %rax, %rbx
+; PPX-NO-NF-NEXT:    addq %rax, %r14
+; PPX-NO-NF-NEXT:    addq $8, %rbp
+; PPX-NO-NF-NEXT:    testb $1, %dil
+; PPX-NO-NF-NEXT:    je .LBB6_2
+; PPX-NO-NF-NEXT:  # %bb.3: # %bb11
+; PPX-NO-NF-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; PPX-NO-NF-NEXT:    leaq {{[0-9]+}}(%rsp), %rsp
+; PPX-NO-NF-NEXT:    popp %rbx
+; PPX-NO-NF-NEXT:    pop2p %r13, %r12
+; PPX-NO-NF-NEXT:    pop2p %r15, %r14
+; PPX-NO-NF-NEXT:    popp %rbp
+; PPX-NO-NF-NEXT:    jne .LBB6_5
+; PPX-NO-NF-NEXT:  # %bb.4: # %bb12
+; PPX-NO-NF-NEXT:    movq $0, (%rax)
+; PPX-NO-NF-NEXT:  .LBB6_5: # %bb14
+; PPX-NO-NF-NEXT:    retq
 ;
 ; FRAME-LABEL: lea_in_epilog:
 ; FRAME:       # %bb.0: # %bb
@@ -421,6 +422,58 @@ define void @lea_in_epilog(i1 %arg, ptr %arg1, ptr %arg2, i64 %arg3, i64 %arg4,
 ; FRAME-NEXT:    movq $0, (%rax)
 ; FRAME-NEXT:  .LBB6_5: # %bb14
 ; FRAME-NEXT:    retq
+;
+; PPX-NF-LABEL: lea_in_epilog:
+; PPX-NF:       # %bb.0: # %bb
+; PPX-NF-NEXT:    testb $1, %dil
+; PPX-NF-NEXT:    je .LBB6_5
+; PPX-NF-NEXT:  # %bb.1: # %bb13
+; PPX-NF-NEXT:    pushp %rbp
+; PPX-NF-NEXT:    push2p %r14, %r15
+; PPX-NF-NEXT:    push2p %r12, %r13
+; PPX-NF-NEXT:    pushp %rbx
+; PPX-NF-NEXT:    {nf} subq $24, %rsp
+; PPX-NF-NEXT:    movq %r9, %r14
+; PPX-NF-NEXT:    movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
+; PPX-NF-NEXT:    addq {{[0-9]+}}(%rsp), %r14
+; PPX-NF-NEXT:    movq {{[0-9]+}}(%rsp), %r13
+; PPX-NF-NEXT:    addq %r14, %r13
+; PPX-NF-NEXT:    movq {{[0-9]+}}(%rsp), %r15
+; PPX-NF-NEXT:    addq %r14, %r15
+; PPX-NF-NEXT:    movq {{[0-9]+}}(%rsp), %rbx
+; PPX-NF-NEXT:    addq %r14, %rbx
+; PPX-NF-NEXT:    xorl %ebp, %ebp
+; PPX-NF-NEXT:    xorl %r12d, %r12d
+; PPX-NF-NEXT:    movl %edi, {{[-0-9]+}}(%r{{[sb]}}p) # 4-byte Spill
+; PPX-NF-NEXT:    .p2align 4
+; PPX-NF-NEXT:  .LBB6_2: # %bb15
+; PPX-NF-NEXT:    # =>This Inner Loop Header: Depth=1
+; PPX-NF-NEXT:    incq %r12
+; PPX-NF-NEXT:    movl $432, %edx # imm = 0x1B0
+; PPX-NF-NEXT:    xorl %edi, %edi
+; PPX-NF-NEXT:    movq %r15, %rsi
+; PPX-NF-NEXT:    callq memcpy at PLT
+; PPX-NF-NEXT:    movl {{[-0-9]+}}(%r{{[sb]}}p), %edi # 4-byte Reload
+; PPX-NF-NEXT:    movq {{[0-9]+}}(%rsp), %rax
+; PPX-NF-NEXT:    addq %rax, %r13
+; PPX-NF-NEXT:    addq %rax, %r15
+; PPX-NF-NEXT:    addq %rax, %rbx
+; PPX-NF-NEXT:    addq %rax, %r14
+; PPX-NF-NEXT:    addq $8, %rbp
+; PPX-NF-NEXT:    testb $1, %dil
+; PPX-NF-NEXT:    je .LBB6_2
+; PPX-NF-NEXT:  # %bb.3: # %bb11
+; PPX-NF-NEXT:    movq {{[-0-9]+}}(%r{{[sb]}}p), %rax # 8-byte Reload
+; PPX-NF-NEXT:    {nf} addq $24, %rsp
+; PPX-NF-NEXT:    popp %rbx
+; PPX-NF-NEXT:    pop2p %r13, %r12
+; PPX-NF-NEXT:    pop2p %r15, %r14
+; PPX-NF-NEXT:    popp %rbp
+; PPX-NF-NEXT:    jne .LBB6_5
+; PPX-NF-NEXT:  # %bb.4: # %bb12
+; PPX-NF-NEXT:    movq $0, (%rax)
+; PPX-NF-NEXT:  .LBB6_5: # %bb14
+; PPX-NF-NEXT:    retq
 bb:
   br i1 %arg, label %bb13, label %bb14
 
diff --git a/llvm/test/CodeGen/X86/apx/sub.ll b/llvm/test/CodeGen/X86/apx/sub.ll
index 34af966465d93..d999795cb848f 100644
--- a/llvm/test/CodeGen/X86/apx/sub.ll
+++ b/llvm/test/CodeGen/X86/apx/sub.ll
@@ -1182,7 +1182,7 @@ define fastcc void @fold_with_physical_reg(ptr %arg0, i64 %arg1, ptr %arg2, i1 %
 ; NF-NEXT:    pushq %r13 # encoding: [0x41,0x55]
 ; NF-NEXT:    pushq %r12 # encoding: [0x41,0x54]
 ; NF-NEXT:    pushq %rbx # encoding: [0x53]
-; NF-NEXT:    subq $24, %rsp # encoding: [0x48,0x83,0xec,0x18]
+; NF-NEXT:    {nf} subq $24, %rsp # encoding: [0x62,0xf4,0xfc,0x0c,0x83,0xec,0x18]
 ; NF-NEXT:    movl %ecx, %r14d # encoding: [0x41,0x89,0xce]
 ; NF-NEXT:    movq %rdx, %rbx # encoding: [0x48,0x89,0xd3]
 ; NF-NEXT:    movq %rsi, %r12 # encoding: [0x49,0x89,0xf4]
diff --git a/llvm/test/CodeGen/X86/apx/win64-abi.ll b/llvm/test/CodeGen/X86/apx/win64-abi.ll
index 8aba657f868e7..3a6ed2cea8b56 100644
--- a/llvm/test/CodeGen/X86/apx/win64-abi.ll
+++ b/llvm/test/CodeGen/X86/apx/win64-abi.ll
@@ -151,14 +151,14 @@ entry:
 }
 
 ; PUSH2/POP2 must be used for callee-saved registers including R30+R31 pair.
-define void @test_push2pop2_r30_r31() nounwind "target-features"="+egpr,+push2pop2" {
+define void @test_push2pop2_r30_r31() nounwind "target-features"="+egpr,+push2pop2" "frame-pointer"="all" {
 ; CHECK-LABEL: test_push2pop2_r30_r31:
 ; CHECK: push2 %r30, %r31
 ; CHECK: callq external
 ; CHECK: pop2 %r31, %r30
 ; CHECK: retq
   call void @external()
-  call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15},~{r16},~{r17},~{r18},~{r19},~{r20},~{r21},~{r22},~{r23},~{r24},~{r25},~{r26},~{r27},~{r28},~{r29},~{r30},~{r31},~{rbp},~{xmm0},~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15}"()
+  call void asm sideeffect "", "~{rax},~{rcx},~{rdx},~{rsi},~{rdi},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15},~{r16},~{r17},~{r18},~{r19},~{r20},~{r21},~{r22},~{r23},~{r24},~{r25},~{r26},~{r27},~{r28},~{r29},~{r30},~{r31},~{xmm0},~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15}"()
   ret void
 }
 
diff --git a/llvm/test/CodeGen/X86/win64-eh-unwindv3-push2pop2.ll b/llvm/test/CodeGen/X86/win64-eh-unwindv3-push2pop2.ll
index f24050f2db77c..848db2ec9822e 100644
--- a/llvm/test/CodeGen/X86/win64-eh-unwindv3-push2pop2.ll
+++ b/llvm/test/CodeGen/X86/win64-eh-unwindv3-push2pop2.ll
@@ -16,26 +16,26 @@ declare i32 @c(i32) local_unnamed_addr
 define dso_local i32 @push2pop2_padding(i32 %x) local_unnamed_addr {
 ; CHECK-LABEL: push2pop2_padding:
 ; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    .seh_pushreg %rax
-; CHECK-NEXT:    pushq %rax
-; CHECK-NEXT:    .seh_push2regs %r15, %r14
-; CHECK-NEXT:    push2 %r14, %r15
-; CHECK-NEXT:    .seh_push2regs %r13, %r12
-; CHECK-NEXT:    push2 %r12, %r13
-; CHECK-NEXT:    .seh_push2regs %rbp, %rbx
-; CHECK-NEXT:    push2 %rbx, %rbp
+; CHECK-NEXT:    .seh_pushreg %r15
+; CHECK-NEXT:    pushq %r15
+; CHECK-NEXT:    .seh_push2regs %r14, %r13
+; CHECK-NEXT:    push2 %r13, %r14
+; CHECK-NEXT:    .seh_push2regs %r12, %rbp
+; CHECK-NEXT:    push2 %rbp, %r12
+; CHECK-NEXT:    .seh_pushreg %rbx
+; CHECK-NEXT:    pushq %rbx
 ; CHECK-NEXT:    .seh_endprologue
 ; CHECK-NEXT:    #APP
 ; CHECK-NEXT:    #NO_APP
 ; CHECK-NEXT:    .seh_startepilogue
-; CHECK-NEXT:    .seh_push2regs %rbx, %rbp
-; CHECK-NEXT:    pop2 %rbp, %rbx
-; CHECK-NEXT:    .seh_push2regs %r12, %r13
-; CHECK-NEXT:    pop2 %r13, %r12
-; CHECK-NEXT:    .seh_push2regs %r14, %r15
-; CHECK-NEXT:    pop2 %r15, %r14
-; CHECK-NEXT:    .seh_stackalloc 8
-; CHECK-NEXT:    popq %rax
+; CHECK-NEXT:    .seh_pushreg %rbx
+; CHECK-NEXT:    popq %rbx
+; CHECK-NEXT:    .seh_push2regs %rbp, %r12
+; CHECK-NEXT:    pop2 %r12, %rbp
+; CHECK-NEXT:    .seh_push2regs %r13, %r14
+; CHECK-NEXT:    pop2 %r14, %r13
+; CHECK-NEXT:    .seh_pushreg %r15
+; CHECK-NEXT:    popq %r15
 ; CHECK-NEXT:    .seh_endepilogue
 ; CHECK-NEXT:    jmp c # TAILCALL
 ; CHECK-NEXT:    .seh_endproc

>From 10c1412d0080a4b3719eba80b41ac6916a433e88 Mon Sep 17 00:00:00 2001
From: "Zou, Feng" <feng.zou at intel.com>
Date: Mon, 22 Jun 2026 15:35:05 +0200
Subject: [PATCH 2/3] [X86][APX] Allow NF stack adjustment in Windows x64
 epilogue under unwind v3

The Windows prologue unwinder is data-driven and never disassembles the
prologue, so an EVEX NF {nf} sub of RSP is always safe there. The v1/v2
epilogue unwinder, however, disassembles the epilogue to recognize the
canonical "add/lea rsp; pop...; ret" sequence and does not understand EVEX NF
add/sub, so NF must not be used in a v1/v2 epilogue. Unwind v3 encodes epilog
operations declaratively (no disassembly), so {nf} add of RSP is allowed in a
v3 epilogue. The restriction only applies when the function actually emits
unwind info, so nounwind functions may use NF in the epilogue too.

- Replace the over-broad "no NF on Windows" gate in BuildStackAdjustment with
  prologue-always / epilogue-only-when-(no-unwind-info-or-v3) gating, factored
  into a needsWin64NonV3EpilogueUnwind() helper.
- The NF stack-adjust opcodes are 64-bit (SUB64ri32_NF/ADD64ri32_NF), so don't
  use them for the x32 ABI where the stack pointer is the 32-bit ESP.
- push2/pop2 are EVEX-encoded and equally undisassemblable by the v1/v2
  epilogue unwinder, so gate push2/pop2 candidacy on the same condition.
- Add a split-file test covering v1, v2, v3, nounwind and x32; regenerate
  push2-pop2-cfi-seh.ll (push2/pop2 now suppressed on Windows v1/v2, and split
  the diamondrapids RUN line which enables +nf into its own prefix).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply at anthropic.com>
---
 llvm/lib/Target/X86/X86FrameLowering.cpp      |  43 ++++--
 llvm/lib/Target/X86/X86FrameLowering.h        |   6 +
 .../test/CodeGen/X86/apx/nf-stackalloc-seh.ll | 136 ++++++++++++++++++
 .../CodeGen/X86/apx/push2-pop2-cfi-seh.ll     |  63 ++++++--
 4 files changed, 226 insertions(+), 22 deletions(-)
 create mode 100644 llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll

diff --git a/llvm/lib/Target/X86/X86FrameLowering.cpp b/llvm/lib/Target/X86/X86FrameLowering.cpp
index b703fc24b6a6f..1cc990426aed8 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.cpp
+++ b/llvm/lib/Target/X86/X86FrameLowering.cpp
@@ -384,9 +384,17 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
 
   MachineInstrBuilder MI;
   // Use NF (no-flags) variants when the target supports it, avoiding
-  // gratuitous EFLAGS clobber. Not on Windows where the OS epilogue unwinder
-  // doesn't recognize EVEX-encoded instructions.
-  bool UseNF = STI.hasNF() && !isWin64Prologue(*MBB.getParent());
+  // gratuitous EFLAGS clobber. The NF stack-adjust opcodes below are 64-bit
+  // (SUB64ri32_NF/ADD64ri32_NF), so don't use them for the x32 ABI where the
+  // stack pointer is 32-bit. On Windows whether NF is safe in the epilogue
+  // depends on how the OS unwinder reads the code: the prologue unwinder is
+  // data-driven (it walks the unwind codes and never disassembles the
+  // prologue), so NF is always safe there; but the v1/v2 epilogue unwinder
+  // disassembles the epilogue and doesn't recognize EVEX NF add/sub
+  // instructions, so only use NF in a Windows epilogue under unwind v3 (which
+  // encodes epilog operations declaratively and needs no disassembly).
+  bool UseNF = STI.hasNF() && Uses64BitFramePtr &&
+               !(InEpilogue && needsWin64NonV3EpilogueUnwind(*MBB.getParent()));
   if (UseLEA && !UseNF) {
     MI = addRegOffset(BuildMI(MBB, MBBI, DL,
                               TII.get(getLEArOpcode(Uses64BitFramePtr)),
@@ -395,11 +403,10 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
   } else {
     bool IsSub = Offset < 0;
     uint64_t AbsOffset = IsSub ? -Offset : Offset;
-    const unsigned Opc =
-        IsSub ? (UseNF ? (unsigned)X86::SUB64ri32_NF
-                       : getSUBriOpcode(Uses64BitFramePtr))
-              : (UseNF ? (unsigned)X86::ADD64ri32_NF
-                       : getADDriOpcode(Uses64BitFramePtr));
+    const unsigned Opc = IsSub ? (UseNF ? (unsigned)X86::SUB64ri32_NF
+                                        : getSUBriOpcode(Uses64BitFramePtr))
+                               : (UseNF ? (unsigned)X86::ADD64ri32_NF
+                                        : getADDriOpcode(Uses64BitFramePtr));
     MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
              .addReg(StackPtr)
              .addImm(AbsOffset);
@@ -1485,6 +1492,14 @@ bool X86FrameLowering::isWin64Prologue(const MachineFunction &MF) const {
   return MF.getTarget().getMCAsmInfo().usesWindowsCFI();
 }
 
+bool X86FrameLowering::needsWin64NonV3EpilogueUnwind(
+    const MachineFunction &MF) const {
+  const Function &Fn = MF.getFunction();
+  if (!isWin64Prologue(MF) || !Fn.needsUnwindTableEntry())
+    return false;
+  return Fn.getParent()->getWinX64EHUnwindMode() != WinX64EHUnwindMode::V3;
+}
+
 bool X86FrameLowering::needsDwarfCFI(const MachineFunction &MF) const {
   return !isWin64Prologue(MF) && MF.needsFrameMoves();
 }
@@ -2787,9 +2802,8 @@ void X86FrameLowering::emitEpilogue(MachineFunction &MF,
                  PI->getOperand(0).getReg() == StackPtr &&
                  (Opc == X86::ADD64ri32 || Opc == X86::ADD64ri32_NF ||
                   Opc == X86::ADD32ri || Opc == X86::LEA64r)) {
-        int64_t SPAdj = Opc == X86::LEA64r
-                            ? PI->getOperand(4).getImm()
-                            : PI->getOperand(2).getImm();
+        int64_t SPAdj = Opc == X86::LEA64r ? PI->getOperand(4).getImm()
+                                           : PI->getOperand(2).getImm();
         Offset += SPAdj;
         BuildCFI(MBB, MBBI, DL,
                  MCCFIInstruction::cfiDefCfaOffset(nullptr, -Offset),
@@ -3089,7 +3103,12 @@ bool X86FrameLowering::assignCalleeSavedSpillSlots(
   // 3. When the number of CSR push is even, start to use push2 from the 1st
   //    push and make the stack 16B aligned before the push
   unsigned NumRegsForPush2 = 0;
-  if (STI.hasPush2Pop2() && getStackAlignment() >= 16) {
+  // push2/pop2 are EVEX-encoded. The Windows v1/v2 epilogue unwinder
+  // disassembles the epilogue and can't decode EVEX, and only unwind v3 emits
+  // the declarative .seh_push2regs, so don't use push2/pop2 when the function
+  // needs v1/v2 Win64 unwind info.
+  if (STI.hasPush2Pop2() && getStackAlignment() >= 16 &&
+      !needsWin64NonV3EpilogueUnwind(MF)) {
     unsigned NumCSGPR = llvm::count_if(CSI, [](const CalleeSavedInfo &I) {
       return X86::GR64RegClass.contains(I.getReg());
     });
diff --git a/llvm/lib/Target/X86/X86FrameLowering.h b/llvm/lib/Target/X86/X86FrameLowering.h
index f1e3796f5fddd..242445a95673f 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.h
+++ b/llvm/lib/Target/X86/X86FrameLowering.h
@@ -244,6 +244,12 @@ class X86FrameLowering : public TargetFrameLowering {
 private:
   bool isWin64Prologue(const MachineFunction &MF) const;
 
+  /// Returns true if MF emits Win64 unwind info using a version (v1/v2) whose
+  /// OS epilogue unwinder disassembles the epilogue, and therefore cannot
+  /// decode EVEX-encoded (APX) instructions placed in the epilogue. Unwind v3
+  /// encodes epilog operations declaratively, so it returns false there.
+  bool needsWin64NonV3EpilogueUnwind(const MachineFunction &MF) const;
+
   bool needsDwarfCFI(const MachineFunction &MF) const;
 
   uint64_t calculateMaxStackAlign(const MachineFunction &MF) const;
diff --git a/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll b/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll
new file mode 100644
index 0000000000000..8d41143b787a8
--- /dev/null
+++ b/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll
@@ -0,0 +1,136 @@
+; Verify how APX NF (no-flags) ADD/SUB are used for the stack-pointer
+; adjustment in the Windows x64 prologue/epilogue, and how that interacts with
+; the unwind information version.
+;
+; The Windows prologue unwinder is data-driven: it walks the unwind codes and
+; never disassembles the prologue, so an EVEX NF {nf} sub of RSP in the
+; prologue is always safe (v1, v2 and v3 all emit it). The v1/v2 epilogue
+; unwinder, however, disassembles the epilogue to recognize the canonical
+; "add/lea rsp; pop...; ret" sequence, and does not understand EVEX NF add/sub
+; -- so NF must NOT be used in a v1/v2 epilogue. Unwind v3 encodes epilog
+; operations declaratively (no disassembly), so {nf} add of RSP is allowed in a
+; v3 epilogue. The v1/v2 epilogue restriction only applies when the function
+; actually emits unwind info; a nounwind function has no unwind table entry, so
+; its epilogue is never disassembled and {nf} add is allowed there too.
+;
+; The unwind version is selected by a module-wide flag, so each case lives in
+; its own split-file section.
+;
+; RUN: split-file %s %t
+; RUN: llc < %t/v1.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V1
+; RUN: llc < %t/v2.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V2
+; RUN: llc < %t/v3.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V3
+; RUN: llc < %t/nounwind.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=NOUNWIND
+;
+; The NF stack-adjust opcodes are 64-bit (SUB64ri32_NF/ADD64ri32_NF), so for
+; the x32 ABI -- where the stack pointer is the 32-bit ESP -- NF must not be
+; used; the sized non-NF SUB32ri/ADD32ri are emitted instead.
+; RUN: llc < %t/nounwind.ll -mtriple=x86_64-linux-gnux32 -mattr=+nf -verify-machineinstrs | FileCheck %s --check-prefix=X32
+
+;--- v1.ll
+declare void @callee(ptr)
+
+define void @f() {
+; Prologue: NF is always safe on Windows regardless of unwind version, because
+; the prologue unwinder does not disassemble. The stack is allocated with a
+; {nf} subq, with the matching .seh_stackalloc.
+;
+; V1-LABEL: f:
+; V1:         {nf} subq ${{[0-9]+}}, %rsp
+; V1:         .seh_stackalloc
+; V1:         .seh_endprologue
+;
+; Epilogue under v1/v2: the OS unwinder disassembles the epilogue, so the stack
+; deallocation must be a legacy-encoded addq, NOT {nf} addq.
+; V1:         .seh_startepilogue
+; V1-NOT:     {nf} addq {{.*}}, %rsp
+; V1:         addq ${{[0-9]+}}, %rsp
+; V1:         .seh_endepilogue
+; V1:         retq
+entry:
+  %p = alloca [64 x i8], align 16
+  call void @callee(ptr %p)
+  ret void
+}
+
+;--- v2.ll
+declare void @callee(ptr)
+
+define void @f() {
+; Like v1, the v2 epilogue unwinder disassembles the epilogue (and looks for
+; the .seh_unwindv2start marker), so NF is used in the prologue but NOT in the
+; epilogue.
+; V2-LABEL: f:
+; V2:         .seh_unwindversion 2
+; V2:         {nf} subq ${{[0-9]+}}, %rsp
+; V2:         .seh_stackalloc
+; V2:         .seh_endprologue
+;
+; V2:         .seh_startepilogue
+; V2-NOT:     {nf} addq {{.*}}, %rsp
+; V2:         addq ${{[0-9]+}}, %rsp
+; V2:         .seh_endepilogue
+; V2:         retq
+entry:
+  %p = alloca [64 x i8], align 16
+  call void @callee(ptr %p)
+  ret void
+}
+
+!llvm.module.flags = !{!0}
+!0 = !{i32 1, !"winx64-eh-unwind", i32 2}
+
+;--- v3.ll
+declare void @callee(ptr)
+
+define void @f() {
+; In v3 mode the SEH directive is emitted *before* its instruction.
+; V3:         .seh_unwindversion 3
+; V3-LABEL: f:
+; V3:         .seh_stackalloc
+; V3:         {nf} subq ${{[0-9]+}}, %rsp
+; V3:         .seh_endprologue
+;
+; Epilogue under v3: epilog ops are encoded declaratively, so {nf} addq is
+; allowed for the stack deallocation.
+; V3:         .seh_startepilogue
+; V3:         .seh_stackalloc
+; V3:         {nf} addq ${{[0-9]+}}, %rsp
+; V3:         .seh_endepilogue
+; V3:         retq
+entry:
+  %p = alloca [64 x i8], align 16
+  call void @callee(ptr %p)
+  ret void
+}
+
+!llvm.module.flags = !{!0}
+!0 = !{i32 1, !"winx64-eh-unwind", i32 3}
+
+;--- nounwind.ll
+declare void @callee(ptr)
+
+; A nounwind function emits no SEH unwind info, so the OS never disassembles
+; its epilogue. NF is therefore allowed in both the prologue and the epilogue
+; even under the default (v1) unwind mode, and no .seh_ directives are emitted.
+define void @f() nounwind {
+; NOUNWIND-LABEL: f:
+; NOUNWIND-NOT:    .seh_
+; NOUNWIND:        {nf} subq ${{[0-9]+}}, %rsp
+; NOUNWIND:        {nf} addq ${{[0-9]+}}, %rsp
+; NOUNWIND:        retq
+; NOUNWIND-NOT:    .seh_
+;
+; For the x32 ABI the stack pointer is ESP, so the 64-bit NF opcodes can't be
+; used: plain SUB32ri/ADD32ri on %esp are emitted, with no {nf} prefix.
+; X32-LABEL: f:
+; X32-NOT:        {nf}
+; X32:            subl ${{[0-9]+}}, %esp
+; X32:            addl ${{[0-9]+}}, %esp
+; X32:            retq
+; X32-NOT:        {nf}
+entry:
+  %p = alloca [64 x i8], align 16
+  call void @callee(ptr %p)
+  ret void
+}
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
index 5d25f8826cddd..f8476cc7f1c1a 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
@@ -9,8 +9,9 @@
 
 ; EPGR normally required unwind v3 info, but that changes the SEH directives
 ; that get emitted, so disable epgr so that we can validate diamondrapids
-; enables push2pop2
-; RUN: llc < %s -mtriple=x86_64-windows-msvc -mcpu=diamondrapids -mattr=-egpr | FileCheck %s --check-prefix=WIN-PPX
+; enables push2pop2. diamondrapids also enables +nf, so the prologue stack
+; adjustment uses an NF sub; use a separate prefix from the +ppx line above.
+; RUN: llc < %s -mtriple=x86_64-windows-msvc -mcpu=diamondrapids -mattr=-egpr | FileCheck %s --check-prefix=WIN-DR
 
 define i32 @csr6_alloc16(ptr %argv) {
 ; LIN-REF-LABEL: csr6_alloc16:
@@ -199,11 +200,13 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; WIN:       # %bb.0: # %entry
 ; WIN-NEXT:    pushq %r15
 ; WIN-NEXT:    .seh_pushreg %r15
-; WIN-NEXT:    push2 %r13, %r14
+; WIN-NEXT:    pushq %r14
 ; WIN-NEXT:    .seh_pushreg %r14
+; WIN-NEXT:    pushq %r13
 ; WIN-NEXT:    .seh_pushreg %r13
-; WIN-NEXT:    push2 %rbp, %r12
+; WIN-NEXT:    pushq %r12
 ; WIN-NEXT:    .seh_pushreg %r12
+; WIN-NEXT:    pushq %rbp
 ; WIN-NEXT:    .seh_pushreg %rbp
 ; WIN-NEXT:    pushq %rbx
 ; WIN-NEXT:    .seh_pushreg %rbx
@@ -218,8 +221,10 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; WIN-NEXT:    .seh_startepilogue
 ; WIN-NEXT:    addq $56, %rsp
 ; WIN-NEXT:    popq %rbx
-; WIN-NEXT:    pop2 %r12, %rbp
-; WIN-NEXT:    pop2 %r14, %r13
+; WIN-NEXT:    popq %rbp
+; WIN-NEXT:    popq %r12
+; WIN-NEXT:    popq %r13
+; WIN-NEXT:    popq %r14
 ; WIN-NEXT:    popq %r15
 ; WIN-NEXT:    .seh_endepilogue
 ; WIN-NEXT:    retq
@@ -229,11 +234,13 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; WIN-PPX:       # %bb.0: # %entry
 ; WIN-PPX-NEXT:    pushp %r15
 ; WIN-PPX-NEXT:    .seh_pushreg %r15
-; WIN-PPX-NEXT:    push2p %r13, %r14
+; WIN-PPX-NEXT:    pushp %r14
 ; WIN-PPX-NEXT:    .seh_pushreg %r14
+; WIN-PPX-NEXT:    pushp %r13
 ; WIN-PPX-NEXT:    .seh_pushreg %r13
-; WIN-PPX-NEXT:    push2p %rbp, %r12
+; WIN-PPX-NEXT:    pushp %r12
 ; WIN-PPX-NEXT:    .seh_pushreg %r12
+; WIN-PPX-NEXT:    pushp %rbp
 ; WIN-PPX-NEXT:    .seh_pushreg %rbp
 ; WIN-PPX-NEXT:    pushp %rbx
 ; WIN-PPX-NEXT:    .seh_pushreg %rbx
@@ -248,12 +255,48 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; WIN-PPX-NEXT:    .seh_startepilogue
 ; WIN-PPX-NEXT:    addq $56, %rsp
 ; WIN-PPX-NEXT:    popp %rbx
-; WIN-PPX-NEXT:    pop2p %r12, %rbp
-; WIN-PPX-NEXT:    pop2p %r14, %r13
+; WIN-PPX-NEXT:    popp %rbp
+; WIN-PPX-NEXT:    popp %r12
+; WIN-PPX-NEXT:    popp %r13
+; WIN-PPX-NEXT:    popp %r14
 ; WIN-PPX-NEXT:    popp %r15
 ; WIN-PPX-NEXT:    .seh_endepilogue
 ; WIN-PPX-NEXT:    retq
 ; WIN-PPX-NEXT:    .seh_endproc
+;
+; WIN-DR-LABEL: csr6_alloc16:
+; WIN-DR:       # %bb.0: # %entry
+; WIN-DR-NEXT:    pushp %r15
+; WIN-DR-NEXT:    .seh_pushreg %r15
+; WIN-DR-NEXT:    pushp %r14
+; WIN-DR-NEXT:    .seh_pushreg %r14
+; WIN-DR-NEXT:    pushp %r13
+; WIN-DR-NEXT:    .seh_pushreg %r13
+; WIN-DR-NEXT:    pushp %r12
+; WIN-DR-NEXT:    .seh_pushreg %r12
+; WIN-DR-NEXT:    pushp %rbp
+; WIN-DR-NEXT:    .seh_pushreg %rbp
+; WIN-DR-NEXT:    pushp %rbx
+; WIN-DR-NEXT:    .seh_pushreg %rbx
+; WIN-DR-NEXT:    {nf} subq $56, %rsp
+; WIN-DR-NEXT:    .seh_stackalloc 56
+; WIN-DR-NEXT:    .seh_endprologue
+; WIN-DR-NEXT:    #APP
+; WIN-DR-NEXT:    #NO_APP
+; WIN-DR-NEXT:    xorl %eax, %eax
+; WIN-DR-NEXT:    callq *%rax
+; WIN-DR-NEXT:    nop
+; WIN-DR-NEXT:    .seh_startepilogue
+; WIN-DR-NEXT:    addq $56, %rsp
+; WIN-DR-NEXT:    popp %rbx
+; WIN-DR-NEXT:    popp %rbp
+; WIN-DR-NEXT:    popp %r12
+; WIN-DR-NEXT:    popp %r13
+; WIN-DR-NEXT:    popp %r14
+; WIN-DR-NEXT:    popp %r15
+; WIN-DR-NEXT:    .seh_endepilogue
+; WIN-DR-NEXT:    retq
+; WIN-DR-NEXT:    .seh_endproc
 entry:
   tail call void asm sideeffect "", "~{rbp},~{r15},~{r14},~{r13},~{r12},~{rbx},~{dirflag},~{fpsr},~{flags}"()
   %a = alloca [3 x ptr], align 8

>From c643e1569fbaa612143d93839ae0cfabdf3dde98 Mon Sep 17 00:00:00 2001
From: "Zou, Feng" <feng.zou at intel.com>
Date: Wed, 24 Jun 2026 03:18:38 +0200
Subject: [PATCH 3/3] [X86][APX] Address review: prefer NF only as LEA
 replacement, drop dead code

Apply review feedback on the push+push2+push / NF stack-adjustment work:

- BuildStackAdjustment: use an NF (no-flags) SUB/ADD only as a smaller
  replacement for LEA, i.e. only when EFLAGS must be preserved. When EFLAGS is
  dead, prefer the plain SUB/ADD, which is shorter than the EVEX-encoded NF
  form. Reformat into a clear three-way branch (NF / LEA / plain SUB+ADD).

- Because NF is now gated on UseLEA, it can never reach a Win64 epilogue (which
  only uses LEA for the SP adjustment when it has a frame pointer, and that path
  does not go through BuildStackAdjustment). The Windows epilogue unwinder thus
  never sees an undisassemblable NF add/sub, so drop the unwind-v3 special case
  from the NF gate. needsWin64NonV3EpilogueUnwind is retained: it still gates
  push2/pop2 candidacy, which is an independent EVEX-encoding concern.

- Remove the epilogue DWARF CFI branch that handled FrameDestroy ADD/LEA/NF
  stack adjustments. It is unreachable: the only post-FirstCSPop SP adjustment
  was the push2/pop2 padding deallocation, which was removed when padding moved
  to a real CSR push/pop; the local-frame restore is emitted before FirstCSPop
  with its own CFI.

- Remove the now-unused PadForPush2Pop2 flag and its accessors; its only writer
  was deleted with the old dummy-push padding, so getCalleeSavedFrameSize() no
  longer needs the padding term.

- Replace nf-stackalloc-seh.ll (which tested the now-unreachable Windows NF
  stack-adjustment path) with nf-stackadjust.ll, which exercises NF in both the
  prologue ({nf} subq) and epilogue ({nf} addq) via the lea-sp tuning, and
  checks the LEA, plain-SUB/ADD (EFLAGS dead) and x32 (no NF) fallbacks. Update
  the push2-pop2-cfi-seh.ll WIN-DR comment to reflect that it validates
  push2/pop2 suppression under V1 unwind, and regenerate add.ll, sub.ll and
  push2-pop2.ll (gratuitous NF reverts to plain SUB/ADD).

Assisted-by: Claude Opus 4.8 (1M context) <noreply at anthropic.com>
---
 llvm/lib/Target/X86/X86FrameLowering.cpp      |  53 +++----
 llvm/lib/Target/X86/X86MachineFunctionInfo.h  |  10 +-
 llvm/test/CodeGen/X86/apx/add.ll              |   2 +-
 llvm/test/CodeGen/X86/apx/nf-stackadjust.ll   |  65 +++++++++
 .../test/CodeGen/X86/apx/nf-stackalloc-seh.ll | 136 ------------------
 .../CodeGen/X86/apx/push2-pop2-cfi-seh.ll     |  17 ++-
 llvm/test/CodeGen/X86/apx/push2-pop2.ll       |   2 +-
 llvm/test/CodeGen/X86/apx/sub.ll              |   2 +-
 8 files changed, 101 insertions(+), 186 deletions(-)
 create mode 100644 llvm/test/CodeGen/X86/apx/nf-stackadjust.ll
 delete mode 100644 llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll

diff --git a/llvm/lib/Target/X86/X86FrameLowering.cpp b/llvm/lib/Target/X86/X86FrameLowering.cpp
index 1cc990426aed8..dce161fa6316f 100644
--- a/llvm/lib/Target/X86/X86FrameLowering.cpp
+++ b/llvm/lib/Target/X86/X86FrameLowering.cpp
@@ -383,35 +383,36 @@ MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
   }
 
   MachineInstrBuilder MI;
-  // Use NF (no-flags) variants when the target supports it, avoiding
-  // gratuitous EFLAGS clobber. The NF stack-adjust opcodes below are 64-bit
-  // (SUB64ri32_NF/ADD64ri32_NF), so don't use them for the x32 ABI where the
-  // stack pointer is 32-bit. On Windows whether NF is safe in the epilogue
-  // depends on how the OS unwinder reads the code: the prologue unwinder is
-  // data-driven (it walks the unwind codes and never disassembles the
-  // prologue), so NF is always safe there; but the v1/v2 epilogue unwinder
-  // disassembles the epilogue and doesn't recognize EVEX NF add/sub
-  // instructions, so only use NF in a Windows epilogue under unwind v3 (which
-  // encodes epilog operations declaratively and needs no disassembly).
-  bool UseNF = STI.hasNF() && Uses64BitFramePtr &&
-               !(InEpilogue && needsWin64NonV3EpilogueUnwind(*MBB.getParent()));
-  if (UseLEA && !UseNF) {
+  // Use an NF (no-flags) variant as a smaller replacement for LEA when EFLAGS
+  // must be preserved (i.e. only when we would otherwise emit LEA). If EFLAGS
+  // is dead we prefer the plain SUB/ADD, which is shorter than the EVEX-encoded
+  // NF form. The NF stack-adjust opcodes below are 64-bit (SUB64ri32_NF/
+  // ADD64ri32_NF), so don't use them for the x32 ABI where the stack pointer is
+  // 32-bit. NF cannot reach a Win64 epilogue (which never uses LEA for the SP
+  // adjustment unless it has a frame pointer, and that path doesn't go through
+  // here), so the Windows epilogue unwinder never sees an undisassemblable NF
+  // add/sub.
+  bool UseNF = UseLEA && STI.hasNF() && Uses64BitFramePtr;
+  bool IsSub = Offset < 0;
+  uint64_t AbsOffset = IsSub ? -Offset : Offset;
+  if (UseNF) {
+    const unsigned Opc = IsSub ? X86::SUB64ri32_NF : X86::ADD64ri32_NF;
+    MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
+             .addReg(StackPtr)
+             .addImm(AbsOffset);
+    // NF instructions define no EFLAGS, so there is nothing to mark dead.
+  } else if (UseLEA) {
     MI = addRegOffset(BuildMI(MBB, MBBI, DL,
                               TII.get(getLEArOpcode(Uses64BitFramePtr)),
                               StackPtr),
                       StackPtr, false, Offset);
   } else {
-    bool IsSub = Offset < 0;
-    uint64_t AbsOffset = IsSub ? -Offset : Offset;
-    const unsigned Opc = IsSub ? (UseNF ? (unsigned)X86::SUB64ri32_NF
-                                        : getSUBriOpcode(Uses64BitFramePtr))
-                               : (UseNF ? (unsigned)X86::ADD64ri32_NF
-                                        : getADDriOpcode(Uses64BitFramePtr));
+    const unsigned Opc = IsSub ? getSUBriOpcode(Uses64BitFramePtr)
+                               : getADDriOpcode(Uses64BitFramePtr);
     MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
              .addReg(StackPtr)
              .addImm(AbsOffset);
-    if (!UseNF)
-      MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
+    MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
   }
   return MI;
 }
@@ -2798,16 +2799,6 @@ void X86FrameLowering::emitEpilogue(MachineFunction &MF,
         BuildCFI(MBB, MBBI, DL,
                  MCCFIInstruction::cfiDefCfaOffset(nullptr, -Offset),
                  MachineInstr::FrameDestroy);
-      } else if (PI->getFlag(MachineInstr::FrameDestroy) &&
-                 PI->getOperand(0).getReg() == StackPtr &&
-                 (Opc == X86::ADD64ri32 || Opc == X86::ADD64ri32_NF ||
-                  Opc == X86::ADD32ri || Opc == X86::LEA64r)) {
-        int64_t SPAdj = Opc == X86::LEA64r ? PI->getOperand(4).getImm()
-                                           : PI->getOperand(2).getImm();
-        Offset += SPAdj;
-        BuildCFI(MBB, MBBI, DL,
-                 MCCFIInstruction::cfiDefCfaOffset(nullptr, -Offset),
-                 MachineInstr::FrameDestroy);
       }
     }
   }
diff --git a/llvm/lib/Target/X86/X86MachineFunctionInfo.h b/llvm/lib/Target/X86/X86MachineFunctionInfo.h
index 1bda505ed39f1..1cfe86d57071f 100644
--- a/llvm/lib/Target/X86/X86MachineFunctionInfo.h
+++ b/llvm/lib/Target/X86/X86MachineFunctionInfo.h
@@ -149,9 +149,6 @@ class X86MachineFunctionInfo : public MachineFunctionInfo {
   /// other tools to detect the extended record.
   bool HasSwiftAsyncContext = false;
 
-  /// Adjust stack for push2/pop2
-  bool PadForPush2Pop2 = false;
-
   /// Candidate registers for push2/pop2
   std::set<Register> CandidatesForPush2Pop2;
 
@@ -211,9 +208,7 @@ class X86MachineFunctionInfo : public MachineFunctionInfo {
   const DenseMap<int, unsigned>& getWinEHXMMSlotInfo() const {
     return WinEHXMMSlotInfo; }
 
-  unsigned getCalleeSavedFrameSize() const {
-    return CalleeSavedFrameSize + 8 * padForPush2Pop2();
-  }
+  unsigned getCalleeSavedFrameSize() const { return CalleeSavedFrameSize; }
   void setCalleeSavedFrameSize(unsigned bytes) { CalleeSavedFrameSize = bytes; }
 
   unsigned getBytesToPopOnReturn() const { return BytesToPopOnReturn; }
@@ -284,9 +279,6 @@ class X86MachineFunctionInfo : public MachineFunctionInfo {
   bool hasSwiftAsyncContext() const { return HasSwiftAsyncContext; }
   void setHasSwiftAsyncContext(bool v) { HasSwiftAsyncContext = v; }
 
-  bool padForPush2Pop2() const { return PadForPush2Pop2; }
-  void setPadForPush2Pop2(bool V) { PadForPush2Pop2 = V; }
-
   bool isCandidateForPush2Pop2(Register Reg) const {
     return CandidatesForPush2Pop2.find(Reg) != CandidatesForPush2Pop2.end();
   }
diff --git a/llvm/test/CodeGen/X86/apx/add.ll b/llvm/test/CodeGen/X86/apx/add.ll
index 45c56645237d1..d7c5635b617c1 100644
--- a/llvm/test/CodeGen/X86/apx/add.ll
+++ b/llvm/test/CodeGen/X86/apx/add.ll
@@ -1248,7 +1248,7 @@ define i32 @two_address_no_subreg(i32 %arg0, ptr %arg1, i1 %arg2) nounwind {
 ; NF-NEXT:    xorl %ecx, %ecx # encoding: [0x31,0xc9]
 ; NF-NEXT:    callq *%rax # encoding: [0xff,0xd0]
 ; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    {nf} addq $8, %rsp # encoding: [0x62,0xf4,0xfc,0x0c,0x83,0xc4,0x08]
+; NF-NEXT:    addq $8, %rsp # encoding: [0x48,0x83,0xc4,0x08]
 ; NF-NEXT:    popq %rbx # encoding: [0x5b]
 ; NF-NEXT:    popq %r12 # encoding: [0x41,0x5c]
 ; NF-NEXT:    popq %r13 # encoding: [0x41,0x5d]
diff --git a/llvm/test/CodeGen/X86/apx/nf-stackadjust.ll b/llvm/test/CodeGen/X86/apx/nf-stackadjust.ll
new file mode 100644
index 0000000000000..f54956115adb8
--- /dev/null
+++ b/llvm/test/CodeGen/X86/apx/nf-stackadjust.ll
@@ -0,0 +1,65 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; Verify APX NF (no-flags) ADD/SUB used for the stack-pointer adjustment.
+;
+; NF is preferred only as a smaller replacement for LEA, i.e. only when EFLAGS
+; must be preserved across the adjustment. The lea-sp tuning forces LEA for the
+; SP, so it is the simplest way to exercise NF in both the prologue ({nf} subq)
+; and the epilogue ({nf} addq). Without lea-sp the adjustment has dead EFLAGS
+; and uses a plain (shorter) SUB/ADD instead.
+;
+; RUN: llc < %s -mtriple=x86_64-linux-gnu -mattr=+nf,+lea-sp | FileCheck %s --check-prefix=NF
+; RUN: llc < %s -mtriple=x86_64-linux-gnu -mattr=+lea-sp | FileCheck %s --check-prefix=LEA
+; RUN: llc < %s -mtriple=x86_64-linux-gnu -mattr=+nf | FileCheck %s --check-prefix=NONF
+;
+; The NF stack-adjust opcodes are 64-bit (SUB64ri32_NF/ADD64ri32_NF), so for the
+; x32 ABI -- where the stack pointer is the 32-bit ESP -- NF must not be used;
+; LEA on %esp is emitted instead.
+; RUN: llc < %s -mtriple=x86_64-linux-gnux32 -mattr=+nf,+lea-sp -verify-machineinstrs | FileCheck %s --check-prefix=X32
+
+declare void @callee(ptr)
+
+define void @f() {
+; NF-LABEL: f:
+; NF:       # %bb.0: # %entry
+; NF-NEXT:    {nf} subq $72, %rsp
+; NF-NEXT:    .cfi_def_cfa_offset 80
+; NF-NEXT:    movq %rsp, %rdi
+; NF-NEXT:    callq callee at PLT
+; NF-NEXT:    {nf} addq $72, %rsp
+; NF-NEXT:    .cfi_def_cfa_offset 8
+; NF-NEXT:    retq
+;
+; LEA-LABEL: f:
+; LEA:       # %bb.0: # %entry
+; LEA-NEXT:    leaq -{{[0-9]+}}(%rsp), %rsp
+; LEA-NEXT:    .cfi_def_cfa_offset 80
+; LEA-NEXT:    movq %rsp, %rdi
+; LEA-NEXT:    callq callee at PLT
+; LEA-NEXT:    leaq {{[0-9]+}}(%rsp), %rsp
+; LEA-NEXT:    .cfi_def_cfa_offset 8
+; LEA-NEXT:    retq
+;
+; NONF-LABEL: f:
+; NONF:       # %bb.0: # %entry
+; NONF-NEXT:    subq $72, %rsp
+; NONF-NEXT:    .cfi_def_cfa_offset 80
+; NONF-NEXT:    movq %rsp, %rdi
+; NONF-NEXT:    callq callee at PLT
+; NONF-NEXT:    addq $72, %rsp
+; NONF-NEXT:    .cfi_def_cfa_offset 8
+; NONF-NEXT:    retq
+;
+; X32-LABEL: f:
+; X32:       # %bb.0: # %entry
+; X32-NEXT:    leal -{{[0-9]+}}(%esp), %esp
+; X32-NEXT:    .cfi_def_cfa_offset 80
+; X32-NEXT:    movl %esp, %edi
+; X32-NEXT:    callq callee at PLT
+; X32-NEXT:    leal {{[0-9]+}}(%esp), %esp
+; X32-NEXT:    .cfi_def_cfa_offset 8
+; X32-NEXT:    retq
+entry:
+  %p = alloca [64 x i8], align 16
+  call void @callee(ptr %p)
+  ret void
+}
diff --git a/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll b/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll
deleted file mode 100644
index 8d41143b787a8..0000000000000
--- a/llvm/test/CodeGen/X86/apx/nf-stackalloc-seh.ll
+++ /dev/null
@@ -1,136 +0,0 @@
-; Verify how APX NF (no-flags) ADD/SUB are used for the stack-pointer
-; adjustment in the Windows x64 prologue/epilogue, and how that interacts with
-; the unwind information version.
-;
-; The Windows prologue unwinder is data-driven: it walks the unwind codes and
-; never disassembles the prologue, so an EVEX NF {nf} sub of RSP in the
-; prologue is always safe (v1, v2 and v3 all emit it). The v1/v2 epilogue
-; unwinder, however, disassembles the epilogue to recognize the canonical
-; "add/lea rsp; pop...; ret" sequence, and does not understand EVEX NF add/sub
-; -- so NF must NOT be used in a v1/v2 epilogue. Unwind v3 encodes epilog
-; operations declaratively (no disassembly), so {nf} add of RSP is allowed in a
-; v3 epilogue. The v1/v2 epilogue restriction only applies when the function
-; actually emits unwind info; a nounwind function has no unwind table entry, so
-; its epilogue is never disassembled and {nf} add is allowed there too.
-;
-; The unwind version is selected by a module-wide flag, so each case lives in
-; its own split-file section.
-;
-; RUN: split-file %s %t
-; RUN: llc < %t/v1.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V1
-; RUN: llc < %t/v2.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V2
-; RUN: llc < %t/v3.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=V3
-; RUN: llc < %t/nounwind.ll -mtriple=x86_64-windows-msvc -mattr=+nf | FileCheck %s --check-prefix=NOUNWIND
-;
-; The NF stack-adjust opcodes are 64-bit (SUB64ri32_NF/ADD64ri32_NF), so for
-; the x32 ABI -- where the stack pointer is the 32-bit ESP -- NF must not be
-; used; the sized non-NF SUB32ri/ADD32ri are emitted instead.
-; RUN: llc < %t/nounwind.ll -mtriple=x86_64-linux-gnux32 -mattr=+nf -verify-machineinstrs | FileCheck %s --check-prefix=X32
-
-;--- v1.ll
-declare void @callee(ptr)
-
-define void @f() {
-; Prologue: NF is always safe on Windows regardless of unwind version, because
-; the prologue unwinder does not disassemble. The stack is allocated with a
-; {nf} subq, with the matching .seh_stackalloc.
-;
-; V1-LABEL: f:
-; V1:         {nf} subq ${{[0-9]+}}, %rsp
-; V1:         .seh_stackalloc
-; V1:         .seh_endprologue
-;
-; Epilogue under v1/v2: the OS unwinder disassembles the epilogue, so the stack
-; deallocation must be a legacy-encoded addq, NOT {nf} addq.
-; V1:         .seh_startepilogue
-; V1-NOT:     {nf} addq {{.*}}, %rsp
-; V1:         addq ${{[0-9]+}}, %rsp
-; V1:         .seh_endepilogue
-; V1:         retq
-entry:
-  %p = alloca [64 x i8], align 16
-  call void @callee(ptr %p)
-  ret void
-}
-
-;--- v2.ll
-declare void @callee(ptr)
-
-define void @f() {
-; Like v1, the v2 epilogue unwinder disassembles the epilogue (and looks for
-; the .seh_unwindv2start marker), so NF is used in the prologue but NOT in the
-; epilogue.
-; V2-LABEL: f:
-; V2:         .seh_unwindversion 2
-; V2:         {nf} subq ${{[0-9]+}}, %rsp
-; V2:         .seh_stackalloc
-; V2:         .seh_endprologue
-;
-; V2:         .seh_startepilogue
-; V2-NOT:     {nf} addq {{.*}}, %rsp
-; V2:         addq ${{[0-9]+}}, %rsp
-; V2:         .seh_endepilogue
-; V2:         retq
-entry:
-  %p = alloca [64 x i8], align 16
-  call void @callee(ptr %p)
-  ret void
-}
-
-!llvm.module.flags = !{!0}
-!0 = !{i32 1, !"winx64-eh-unwind", i32 2}
-
-;--- v3.ll
-declare void @callee(ptr)
-
-define void @f() {
-; In v3 mode the SEH directive is emitted *before* its instruction.
-; V3:         .seh_unwindversion 3
-; V3-LABEL: f:
-; V3:         .seh_stackalloc
-; V3:         {nf} subq ${{[0-9]+}}, %rsp
-; V3:         .seh_endprologue
-;
-; Epilogue under v3: epilog ops are encoded declaratively, so {nf} addq is
-; allowed for the stack deallocation.
-; V3:         .seh_startepilogue
-; V3:         .seh_stackalloc
-; V3:         {nf} addq ${{[0-9]+}}, %rsp
-; V3:         .seh_endepilogue
-; V3:         retq
-entry:
-  %p = alloca [64 x i8], align 16
-  call void @callee(ptr %p)
-  ret void
-}
-
-!llvm.module.flags = !{!0}
-!0 = !{i32 1, !"winx64-eh-unwind", i32 3}
-
-;--- nounwind.ll
-declare void @callee(ptr)
-
-; A nounwind function emits no SEH unwind info, so the OS never disassembles
-; its epilogue. NF is therefore allowed in both the prologue and the epilogue
-; even under the default (v1) unwind mode, and no .seh_ directives are emitted.
-define void @f() nounwind {
-; NOUNWIND-LABEL: f:
-; NOUNWIND-NOT:    .seh_
-; NOUNWIND:        {nf} subq ${{[0-9]+}}, %rsp
-; NOUNWIND:        {nf} addq ${{[0-9]+}}, %rsp
-; NOUNWIND:        retq
-; NOUNWIND-NOT:    .seh_
-;
-; For the x32 ABI the stack pointer is ESP, so the 64-bit NF opcodes can't be
-; used: plain SUB32ri/ADD32ri on %esp are emitted, with no {nf} prefix.
-; X32-LABEL: f:
-; X32-NOT:        {nf}
-; X32:            subl ${{[0-9]+}}, %esp
-; X32:            addl ${{[0-9]+}}, %esp
-; X32:            retq
-; X32-NOT:        {nf}
-entry:
-  %p = alloca [64 x i8], align 16
-  call void @callee(ptr %p)
-  ret void
-}
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
index f8476cc7f1c1a..d9814e44632e6 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2-cfi-seh.ll
@@ -7,10 +7,13 @@
 ; RUN: llc < %s -mtriple=x86_64-windows-msvc -mattr=+push2pop2 | FileCheck %s --check-prefix=WIN
 ; RUN: llc < %s -mtriple=x86_64-windows-msvc -mattr=+push2pop2,+ppx | FileCheck %s --check-prefix=WIN-PPX
 
-; EPGR normally required unwind v3 info, but that changes the SEH directives
-; that get emitted, so disable epgr so that we can validate diamondrapids
-; enables push2pop2. diamondrapids also enables +nf, so the prologue stack
-; adjustment uses an NF sub; use a separate prefix from the +ppx line above.
+; diamondrapids enables EGPR, which would require V3 unwind info and emit
+; V3-style SEH directives; disable EGPR so this runs with default (V1) unwind.
+; This validates that the push2/pop2 candidacy gate suppresses push2/pop2 under
+; V1 (individual pushp/popp emitted): the function emits unwind info, and the
+; V1/V2 epilogue unwinder can't decode EVEX push2/pop2. diamondrapids also
+; enables +nf, but the stack adjustment here has dead EFLAGS so it stays a plain
+; (non-NF) subq/addq.
 ; RUN: llc < %s -mtriple=x86_64-windows-msvc -mcpu=diamondrapids -mattr=-egpr | FileCheck %s --check-prefix=WIN-DR
 
 define i32 @csr6_alloc16(ptr %argv) {
@@ -137,7 +140,7 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; LIN-DR-NEXT:    .cfi_def_cfa_offset 48
 ; LIN-DR-NEXT:    pushp %rbx
 ; LIN-DR-NEXT:    .cfi_def_cfa_offset 56
-; LIN-DR-NEXT:    {nf} subq $24, %rsp
+; LIN-DR-NEXT:    subq $24, %rsp
 ; LIN-DR-NEXT:    .cfi_def_cfa_offset 80
 ; LIN-DR-NEXT:    .cfi_offset %rbx, -56
 ; LIN-DR-NEXT:    .cfi_offset %r12, -48
@@ -150,7 +153,7 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; LIN-DR-NEXT:    xorl %ecx, %ecx
 ; LIN-DR-NEXT:    xorl %eax, %eax
 ; LIN-DR-NEXT:    callq *%rcx
-; LIN-DR-NEXT:    {nf} addq $24, %rsp
+; LIN-DR-NEXT:    addq $24, %rsp
 ; LIN-DR-NEXT:    .cfi_def_cfa_offset 56
 ; LIN-DR-NEXT:    popp %rbx
 ; LIN-DR-NEXT:    .cfi_def_cfa_offset 48
@@ -278,7 +281,7 @@ define i32 @csr6_alloc16(ptr %argv) {
 ; WIN-DR-NEXT:    .seh_pushreg %rbp
 ; WIN-DR-NEXT:    pushp %rbx
 ; WIN-DR-NEXT:    .seh_pushreg %rbx
-; WIN-DR-NEXT:    {nf} subq $56, %rsp
+; WIN-DR-NEXT:    subq $56, %rsp
 ; WIN-DR-NEXT:    .seh_stackalloc 56
 ; WIN-DR-NEXT:    .seh_endprologue
 ; WIN-DR-NEXT:    #APP
diff --git a/llvm/test/CodeGen/X86/apx/push2-pop2.ll b/llvm/test/CodeGen/X86/apx/push2-pop2.ll
index ec13253c08e2f..abac95b35adce 100644
--- a/llvm/test/CodeGen/X86/apx/push2-pop2.ll
+++ b/llvm/test/CodeGen/X86/apx/push2-pop2.ll
@@ -432,7 +432,7 @@ define void @lea_in_epilog(i1 %arg, ptr %arg1, ptr %arg2, i64 %arg3, i64 %arg4,
 ; PPX-NF-NEXT:    push2p %r14, %r15
 ; PPX-NF-NEXT:    push2p %r12, %r13
 ; PPX-NF-NEXT:    pushp %rbx
-; PPX-NF-NEXT:    {nf} subq $24, %rsp
+; PPX-NF-NEXT:    subq $24, %rsp
 ; PPX-NF-NEXT:    movq %r9, %r14
 ; PPX-NF-NEXT:    movq %rsi, {{[-0-9]+}}(%r{{[sb]}}p) # 8-byte Spill
 ; PPX-NF-NEXT:    addq {{[0-9]+}}(%rsp), %r14
diff --git a/llvm/test/CodeGen/X86/apx/sub.ll b/llvm/test/CodeGen/X86/apx/sub.ll
index d999795cb848f..34af966465d93 100644
--- a/llvm/test/CodeGen/X86/apx/sub.ll
+++ b/llvm/test/CodeGen/X86/apx/sub.ll
@@ -1182,7 +1182,7 @@ define fastcc void @fold_with_physical_reg(ptr %arg0, i64 %arg1, ptr %arg2, i1 %
 ; NF-NEXT:    pushq %r13 # encoding: [0x41,0x55]
 ; NF-NEXT:    pushq %r12 # encoding: [0x41,0x54]
 ; NF-NEXT:    pushq %rbx # encoding: [0x53]
-; NF-NEXT:    {nf} subq $24, %rsp # encoding: [0x62,0xf4,0xfc,0x0c,0x83,0xec,0x18]
+; NF-NEXT:    subq $24, %rsp # encoding: [0x48,0x83,0xec,0x18]
 ; NF-NEXT:    movl %ecx, %r14d # encoding: [0x41,0x89,0xce]
 ; NF-NEXT:    movq %rdx, %rbx # encoding: [0x48,0x89,0xd3]
 ; NF-NEXT:    movq %rsi, %r12 # encoding: [0x49,0x89,0xf4]



More information about the llvm-commits mailing list