[llvm] 2fd8d41 - [X86] Don't fold loads from non-fixed stack objects into tail calls (#221243)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 7 02:44:07 PDT 2026


Author: Akash Manna
Date: 2026-09-07T09:44:02Z
New Revision: 2fd8d41d7b8824637be32d569c1e1482128b057b

URL: https://github.com/llvm/llvm-project/commit/2fd8d41d7b8824637be32d569c1e1482128b057b
DIFF: https://github.com/llvm/llvm-project/commit/2fd8d41d7b8824637be32d569c1e1482128b057b.diff

LOG: [X86] Don't fold loads from non-fixed stack objects into tail calls (#221243)

Fixes #216504

The callee of a tail call can be folded into the jump as a memory
operand (`TCRETURNmi64`). That jump runs after the epilogue, so the
operand is resolved relative to the incoming stack pointer, which only
works for fixed objects such as incoming arguments. Here the callee was
a `volatile` local, so the load stayed on the stack and got folded,
while a variable-index extract from a 256-bit vector went through a
32-byte stack temporary and forced dynamic realignment. Once the stack
is realigned a local has no static offset from the incoming stack
pointer, and PEI tripped the assertion. The sibcall eligibility check
does look at realignment, but it runs before legalization, so it never
saw the temporary. Spill slots created during register allocation can
cause the same thing, so no check at that point can be complete.

The fold is now refused whenever the load's address may use a non-fixed
frame index, in `checkTCRetEnoughRegs`, which gates both the
`TCRETURNmi` patterns and the callee-load hoisting in
`PreprocessISelDAG`. The callee is loaded into a register before the
epilogue and the call is still emitted as a tail call. Loads from fixed
objects fold as before. As a side effect this also stops the jump from
reading a slot below the restored stack pointer, which was only safe
inside the red zone.

Added: 
    llvm/test/CodeGen/X86/pr216504.ll

Modified: 
    llvm/lib/Target/X86/X86ISelDAGToDAG.cpp

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index 779e4dfce513a..4478930016f63 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -16,6 +16,7 @@
 #include "X86Subtarget.h"
 #include "X86TargetMachine.h"
 #include "llvm/ADT/Statistic.h"
+#include "llvm/CodeGen/MachineFrameInfo.h"
 #include "llvm/CodeGen/MachineModuleInfo.h"
 #include "llvm/CodeGen/SelectionDAGISel.h"
 #include "llvm/Config/llvm-config.h"
@@ -3682,7 +3683,40 @@ static bool mayUseCarryFlag(X86::CondCode CC) {
   return true;
 }
 
+/// Return true if \p Addr may be matched with a non-fixed frame index as base.
+static bool addrMayUseNonFixedFrameIndex(SDValue Addr,
+                                         const MachineFrameInfo &MFI,
+                                         unsigned Depth = 0) {
+  if (auto *FI = dyn_cast<FrameIndexSDNode>(Addr))
+    return !MFI.isFixedObjectIndex(FI->getIndex());
+  // Assume the worst if we can't see the whole address expression.
+  if (Depth >= SelectionDAG::MaxRecursionDepth)
+    return true;
+  switch (Addr.getOpcode()) {
+  case ISD::ADD:
+  case ISD::OR:
+  case ISD::XOR:
+    return addrMayUseNonFixedFrameIndex(Addr.getOperand(0), MFI, Depth + 1) ||
+           addrMayUseNonFixedFrameIndex(Addr.getOperand(1), MFI, Depth + 1);
+  case ISD::SUB:
+    return addrMayUseNonFixedFrameIndex(Addr.getOperand(0), MFI, Depth + 1);
+  default:
+    // Only add-like nodes and the LHS of a SUB can fold a frame index into the
+    // base; anything else is matched as a register or symbol base.
+    return false;
+  }
+}
+
 bool X86DAGToDAGISel::checkTCRetEnoughRegs(SDNode *N) const {
+  assert(N->getOpcode() == X86ISD::TC_RETURN);
+  // X86tcret args: (*chain, ptr, imm, regs..., glue)
+  const SDValue &BasePtr = cast<LoadSDNode>(N->getOperand(1))->getBasePtr();
+
+  // The tail call executes after the epilogue, where only fixed stack objects
+  // can still be addressed (the stack may end up realigned).
+  if (addrMayUseNonFixedFrameIndex(BasePtr, MF->getFrameInfo()))
+    return false;
+
   // Check that there is enough volatile registers to load the callee address.
 
   const X86RegisterInfo *RI = Subtarget->getRegisterInfo();
@@ -3711,13 +3745,9 @@ bool X86DAGToDAGISel::checkTCRetEnoughRegs(SDNode *N) const {
   // The load's base and index need up to two registers.
   unsigned LoadGPRs = 2;
 
-  assert(N->getOpcode() == X86ISD::TC_RETURN);
-  // X86tcret args: (*chain, ptr, imm, regs..., glue)
-
   if (Subtarget->is32Bit()) {
     // FIXME: This was carried from X86tcret_1reg which was used for 32-bit,
     // but it could apply to 64-bit too.
-    const SDValue &BasePtr = cast<LoadSDNode>(N->getOperand(1))->getBasePtr();
     if (isa<FrameIndexSDNode>(BasePtr)) {
       LoadGPRs -= 2; // Base is fixed index off ESP; no regs needed.
     } else if (BasePtr.getOpcode() == X86ISD::Wrapper &&

diff  --git a/llvm/test/CodeGen/X86/pr216504.ll b/llvm/test/CodeGen/X86/pr216504.ll
new file mode 100644
index 0000000000000..6cf9609526088
--- /dev/null
+++ b/llvm/test/CodeGen/X86/pr216504.ll
@@ -0,0 +1,57 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+avx2 -verify-machineinstrs | FileCheck %s --check-prefix=X64
+; RUN: llc < %s -mtriple=i686-unknown-linux-gnu -mattr=+avx2 -verify-machineinstrs | FileCheck %s --check-prefix=X86
+
+; The variable-index extract needs a 32-byte aligned stack temporary, so the
+; stack is realigned. The callee load from a local must not be folded into the
+; tail call, which executes after the epilogue.
+
+ at g2 = global i16 0, align 2
+
+define void @f25() nounwind {
+; X64-LABEL: f25:
+; X64:       # %bb.0: # %entry
+; X64-NEXT:    pushq %rbp
+; X64-NEXT:    movq %rsp, %rbp
+; X64-NEXT:    andq $-32, %rsp
+; X64-NEXT:    subq $96, %rsp
+; X64-NEXT:    movq g2 at GOTPCREL(%rip), %rax
+; X64-NEXT:    movzwl (%rax), %ecx
+; X64-NEXT:    andl $7, %ecx
+; X64-NEXT:    vmovss {{.*#+}} xmm0 = [60,0,0,0]
+; X64-NEXT:    vmovaps %ymm0, {{[0-9]+}}(%rsp)
+; X64-NEXT:    movl 32(%rsp,%rcx,4), %ecx
+; X64-NEXT:    movw %cx, (%rax)
+; X64-NEXT:    movq {{[0-9]+}}(%rsp), %rax
+; X64-NEXT:    movq %rbp, %rsp
+; X64-NEXT:    popq %rbp
+; X64-NEXT:    vzeroupper
+; X64-NEXT:    jmpq *%rax # TAILCALL
+;
+; X86-LABEL: f25:
+; X86:       # %bb.0: # %entry
+; X86-NEXT:    pushl %ebp
+; X86-NEXT:    movl %esp, %ebp
+; X86-NEXT:    andl $-32, %esp
+; X86-NEXT:    subl $96, %esp
+; X86-NEXT:    movzwl g2, %eax
+; X86-NEXT:    andl $7, %eax
+; X86-NEXT:    vmovss {{.*#+}} xmm0 = [60,0,0,0]
+; X86-NEXT:    vmovaps %ymm0, {{[0-9]+}}(%esp)
+; X86-NEXT:    movl 32(%esp,%eax,4), %eax
+; X86-NEXT:    movw %ax, g2
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movl %ebp, %esp
+; X86-NEXT:    popl %ebp
+; X86-NEXT:    vzeroupper
+; X86-NEXT:    jmpl *%eax # TAILCALL
+entry:
+  %fp5 = alloca ptr, align 8
+  %0 = load i16, ptr @g2, align 2
+  %vecext = extractelement <8 x i32> <i32 60, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0>, i16 %0
+  %conv = trunc i32 %vecext to i16
+  store i16 %conv, ptr @g2, align 2
+  %fp = load volatile ptr, ptr %fp5, align 8
+  tail call void %fp()
+  ret void
+}


        


More information about the llvm-commits mailing list