[llvm-branch-commits] [llvm] release/23.x: [llvm][AArch64] Fix FPDiff founding direction in non-sibcall tail calls (#223545) (PR #224137)
Jon Roelofs via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Wed Sep 16 14:00:20 PDT 2026
https://github.com/jroelofs created https://github.com/llvm/llvm-project/pull/224137
This fixes another subtle bug in frame accounting (see: #217156 /
#220406), for tail calls that have a non-multiple of 16 bytes worth of
stack arg area, and need that stack arg re-use to be increased to cover
the alignment requirement. This is best illustrated with callers
containing 8 formal arguments covering the first 8 GPRs (x0-x7),
followed by 9 bytes of argument passed on the stack.
In a callee-pops tail call (e.g. tailcc/swifttailcc), the set of
reusable stack arg area bytes has already been sufficiently aligned by
LowerFormalArguments, so growing NumBytes up to StackAlign is enough to
consume that excess. Otherwise (e.g. a plain C-convention call, forced
off the sibcall path, as in the aarch64_inout_za tests), we can't rely
on either having been pre-aligned, so we round NumBytes up to the same
residue mod StackAlign as NumReusableBytes, which cancels the residue
out of their difference (FPDiff), thus keeping the stack aligned going
into the callee.
Concretely, #220406 computed the Realign distance needed to round FPDiff
up to the next multiple of StackAlign, then subtracted that from FPDiff.
Subtracting that round-up distance overshoots rounding down to the
nearest multiple, and in general arrives at a value that may not even be
a multiple of StackAlign. For example, in @caller_last_stack_arg_i64 we
had NumReusableBytes=8 and NumBytes=1 for an FPDiff of 7, which computed
Realign=9, FPDiff=-2, NumBytes=10 instead of the intended NumBytes=8,
FPDiff=0.
rdar://187331378
(cherry picked from commit ec4d25ab1d0f47f078fd749f6073f04cc820ddf3)
>From 1811a378851a47a15f79ff8ca1e9d4c73643d18d Mon Sep 17 00:00:00 2001
From: Jon Roelofs <jonathan_roelofs at apple.com>
Date: Fri, 11 Sep 2026 14:07:28 -0700
Subject: [PATCH 1/2] Reland "[llvm][AArch64] Ensure stack alignment in
non-sibcall tail calls with FPDiff" (#220406)
When a caller has a non-multiple of the minimum stack alignment more
argument stack space than the callee it is tail calling, we need to ensure that
the resulting FPDiff's magnitude has been rounded up to a multiple of the
platform's minimum stack alignment.
In this re-land there is a fixup to 234ce03692ede13ffd2fcb35570d801c1e332814,
which, because of the wrong order, wasn't actually moving NumBytes after
FPDiff had been re-aligned.
rdar://184474075
(cherry picked from commit 393710bbf74c4e7b92226603a18ead0bba70a8a0)
---
.../Target/AArch64/AArch64ISelLowering.cpp | 14 +++-
llvm/test/CodeGen/AArch64/arm64ec-varargs.ll | 31 ++++++++
.../AArch64/sme-za-tailcall-fpdiff-align.ll | 79 +++++++++++++++++++
3 files changed, 120 insertions(+), 4 deletions(-)
create mode 100644 llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index d134e7f911f83..c6d27746a10ea 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -10286,19 +10286,25 @@ AArch64TargetLowering::LowerCall(CallLoweringInfo &CLI,
// caller will deallocate the entire stack and the callee still expects its
// arguments to begin at SP+0. Completely unused for non-tail calls.
int FPDiff = 0;
+ const Align StackAlign = Subtarget->getFrameLowering()->getStackAlign();
if (IsTailCall && !IsSibCall) {
unsigned NumReusableBytes = FuncInfo->getBytesInStackArgArea();
- // Since callee will pop argument stack as a tail call, we must keep the
- // popped size 16-byte aligned.
- NumBytes = alignTo(NumBytes, 16);
-
// FPDiff will be negative if this tail call requires more space than we
// would automatically have in our incoming argument space. Positive if we
// can actually shrink the stack.
FPDiff = NumReusableBytes - NumBytes;
+ // Since callee will pop the argument stack as a tail call, we must keep the
+ // popped size aligned to the stack alignment. Either or both of NumBytes
+ // and NumReusableBytes may not have been aligned, so we further increase by
+ // the amount needed to keep FPDiff aligned, and therefore preserve the
+ // required alignment going into the callee.
+ uint64_t Realign = offsetToAlignment(FPDiff, StackAlign);
+ FPDiff -= Realign;
+ NumBytes += Realign;
+
// Update the required reserved area if this is the tail call requiring the
// most argument stack space.
if (FPDiff < 0 && FuncInfo->getTailCallReservedStack() < (unsigned)-FPDiff)
diff --git a/llvm/test/CodeGen/AArch64/arm64ec-varargs.ll b/llvm/test/CodeGen/AArch64/arm64ec-varargs.ll
index 979e09cfb5fac..6f13b9606afeb 100644
--- a/llvm/test/CodeGen/AArch64/arm64ec-varargs.ll
+++ b/llvm/test/CodeGen/AArch64/arm64ec-varargs.ll
@@ -157,3 +157,34 @@ define void @varargs_thunk(ptr noundef %0, ...) "thunk" {
musttail call void (ptr, ...) %vtablefn(ptr noundef %0, ...)
ret void
}
+
+declare tailcc void @callee_9_variadic(i64, i64, i64, i64, i64, i64, i64, i64, i64, ...)
+
+define tailcc void @caller_8_args(i64, i64, i64, i64, i64, i64, i64, i64) {
+; CHECK-LABEL: caller_8_args:
+; CHECK: .seh_proc caller_8_args
+; CHECK-NEXT: // %bb.0: // %entry
+; CHECK-NEXT: sub sp, sp, #16
+; CHECK-NEXT: .seh_stackalloc 16
+; CHECK-NEXT: .seh_endprologue
+; CHECK-NEXT: mov w8, #9 // =0x9
+; CHECK-NEXT: mov x4, sp
+; CHECK-NEXT: mov w0, #1 // =0x1
+; CHECK-NEXT: mov w1, #2 // =0x2
+; CHECK-NEXT: mov w2, #3 // =0x3
+; CHECK-NEXT: mov w3, #4 // =0x4
+; CHECK-NEXT: mov w6, #7 // =0x7
+; CHECK-NEXT: mov w7, #8 // =0x8
+; CHECK-NEXT: mov w5, #16 // =0x10
+; CHECK-NEXT: str x8, [sp]
+; CHECK-NEXT: .weak_anti_dep callee_9_variadic
+; CHECK-NEXT: callee_9_variadic = "#callee_9_variadic"
+; CHECK-NEXT: .weak_anti_dep "#callee_9_variadic"
+; CHECK-NEXT: "#callee_9_variadic" = callee_9_variadic
+; CHECK-NEXT: b "#callee_9_variadic"
+; CHECK-NEXT: .seh_endfunclet
+; CHECK-NEXT: .seh_endproc
+entry:
+ tail call tailcc void (i64, i64, i64, i64, i64, i64, i64, i64, i64, ...) @callee_9_variadic(i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9)
+ ret void
+}
diff --git a/llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll b/llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll
new file mode 100644
index 0000000000000..9c449b12f2127
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll
@@ -0,0 +1,79 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=arm64-apple-ios -mattr=+sme -o - | FileCheck %s
+; RUN: llc < %s -mtriple=arm64-apple-ios -mattr=+sme -global-isel-abort=2 -o - 2>&1 | FileCheck %s --check-prefix=CHECK,GISEL
+
+; GlobalISel doesn't implement SME yet, but once it does, we should be mindful of this case.
+; GISEL: warning: Instruction selection used fallback path for caller_more_args
+; GISEL: warning: Instruction selection used fallback path for caller_same_args
+
+declare void @external_func() "aarch64_inout_za"
+
+; ZA state being live for both caller and callee forces the non-sibcall tail
+; call fast path. FPDiff = (caller's incoming argument stack) - (callee's
+; outgoing argument stack) can come out either positive or negative, and in
+; either case its magnitude must be rounded away from zero to a multiple of
+; 16 to keep the stack pointer 16-byte aligned at all times.
+declare void @callee_fewer_args(i64, i64, i64, i64, i64, i64, i64, i64) "aarch64_inout_za"
+define void @caller_more_args(i64, i64, i64, i64, i64, i64, i64, i64, i64, i64) "aarch64_inout_za" uwtable {
+; CHECK-LABEL: caller_more_args:
+; CHECK: ; %bb.0: ; %entry
+; CHECK-NEXT: stp x29, x30, [sp, #-16]! ; 16-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -8
+; CHECK-NEXT: .cfi_offset w29, -16
+; CHECK-NEXT: bl _external_func
+; CHECK-NEXT: mov w0, #1 ; =0x1
+; CHECK-NEXT: mov w1, #2 ; =0x2
+; CHECK-NEXT: mov w2, #3 ; =0x3
+; CHECK-NEXT: mov w3, #4 ; =0x4
+; CHECK-NEXT: mov w4, #5 ; =0x5
+; CHECK-NEXT: mov w5, #6 ; =0x6
+; CHECK-NEXT: mov w6, #7 ; =0x7
+; CHECK-NEXT: mov w7, #8 ; =0x8
+; CHECK-NEXT: ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
+; CHECK-NEXT: .cfi_def_cfa_offset 0
+; CHECK-NEXT: add sp, sp, #16
+; CHECK-NEXT: .cfi_def_cfa_offset -16
+; CHECK-NEXT: .cfi_restore w30
+; CHECK-NEXT: .cfi_restore w29
+; CHECK-NEXT: b _callee_fewer_args
+entry:
+ tail call void @external_func()
+ tail call void @callee_fewer_args(i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8) "aarch64_inout_za"
+ ret void
+}
+
+; Likewise for a callee that has the same amount of argument stack as the
+; caller, we must round up the amount of used stack space in order to keep
+; FPDiff a multiple of 16, thus keeping the stack aligned for the callee.
+declare void @callee_same_args(i64, i64, i64, i64, i64, i64, i64, i64, i64) "aarch64_inout_za"
+define void @caller_same_args(i64, i64, i64, i64, i64, i64, i64, i64, i64) "aarch64_inout_za" uwtable {
+; CHECK-LABEL: caller_same_args:
+; CHECK: ; %bb.0: ; %entry
+; CHECK-NEXT: stp x29, x30, [sp, #-16]! ; 16-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -8
+; CHECK-NEXT: .cfi_offset w29, -16
+; CHECK-NEXT: bl _external_func
+; CHECK-NEXT: mov w8, #9 ; =0x9
+; CHECK-NEXT: mov w0, #1 ; =0x1
+; CHECK-NEXT: mov w1, #2 ; =0x2
+; CHECK-NEXT: str x8, [sp, #16]
+; CHECK-NEXT: mov w2, #3 ; =0x3
+; CHECK-NEXT: mov w3, #4 ; =0x4
+; CHECK-NEXT: mov w4, #5 ; =0x5
+; CHECK-NEXT: mov w5, #6 ; =0x6
+; CHECK-NEXT: mov w6, #7 ; =0x7
+; CHECK-NEXT: mov w7, #8 ; =0x8
+; CHECK-NEXT: ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
+; CHECK-NEXT: .cfi_def_cfa_offset 0
+; CHECK-NEXT: .cfi_restore w30
+; CHECK-NEXT: .cfi_restore w29
+; CHECK-NEXT: b _callee_same_args
+entry:
+ tail call void @external_func()
+ tail call void @callee_same_args(i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9) "aarch64_inout_za"
+ ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GISEL: {{.*}}
>From bd257a893be9402dcd39dec4191ca0b549bb48c6 Mon Sep 17 00:00:00 2001
From: Jon Roelofs <jonathan_roelofs at apple.com>
Date: Wed, 16 Sep 2026 12:11:16 -0700
Subject: [PATCH 2/2] [llvm][AArch64] Fix FPDiff founding direction in
non-sibcall tail calls (#223545)
This fixes another subtle bug in frame accounting (see: #217156 /
#220406), for tail calls that have a non-multiple of 16 bytes worth of
stack arg area, and need that stack arg re-use to be increased to cover
the alignment requirement. This is best illustrated with callers
containing 8 formal arguments covering the first 8 GPRs (x0-x7),
followed by 9 bytes of argument passed on the stack.
In a callee-pops tail call (e.g. tailcc/swifttailcc), the set of
reusable stack arg area bytes has already been sufficiently aligned by
LowerFormalArguments, so growing NumBytes up to StackAlign is enough to
consume that excess. Otherwise (e.g. a plain C-convention call, forced
off the sibcall path, as in the aarch64_inout_za tests), we can't rely
on either having been pre-aligned, so we round NumBytes up to the same
residue mod StackAlign as NumReusableBytes, which cancels the residue
out of their difference (FPDiff), thus keeping the stack aligned going
into the callee.
Concretely, #220406 computed the Realign distance needed to round FPDiff
up to the next multiple of StackAlign, then subtracted that from FPDiff.
Subtracting that round-up distance overshoots rounding down to the
nearest multiple, and in general arrives at a value that may not even be
a multiple of StackAlign. For example, in @caller_last_stack_arg_i64 we
had NumReusableBytes=8 and NumBytes=1 for an FPDiff of 7, which computed
Realign=9, FPDiff=-2, NumBytes=10 instead of the intended NumBytes=8,
FPDiff=0.
rdar://187331378
(cherry picked from commit ec4d25ab1d0f47f078fd749f6073f04cc820ddf3)
---
.../Target/AArch64/AArch64ISelLowering.cpp | 37 +++---
.../AArch64/sme-za-tailcall-fpdiff-align.ll | 35 ++++++
.../CodeGen/AArch64/swifttail-fpdiff-align.ll | 108 ++++++++++++++++++
.../CodeGen/AArch64/tail-call-stack-args.ll | 21 ++++
4 files changed, 185 insertions(+), 16 deletions(-)
create mode 100644 llvm/test/CodeGen/AArch64/swifttail-fpdiff-align.ll
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index c6d27746a10ea..f71c13c1085a0 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -10291,31 +10291,35 @@ AArch64TargetLowering::LowerCall(CallLoweringInfo &CLI,
if (IsTailCall && !IsSibCall) {
unsigned NumReusableBytes = FuncInfo->getBytesInStackArgArea();
+ // In general, neither NumBytes nor NumReusableBytes is guaranteed to be
+ // aligned, so we round NumBytes up to the same residue mod StackAlign as
+ // NumReusableBytes, which keeps their difference (FPDiff) a multiple of
+ // StackAlign, and therefore preserve the required stack alignment going
+ // into the callee. When the callee's convention can guarantee TCO,
+ // LowerFormalArguments will have force-aligned the stack arg area for us
+ // already, so we can count on our own alignment of NumBytes below to result
+ // in an aligned FPDiff.
+ assert((!DoesCalleeRestoreStack(CallConv, TailCallOpt) ||
+ isAligned(StackAlign, NumReusableBytes)) &&
+ "expected LowerFormalArguments to force-align stack arg area");
+ NumBytes += offsetToAlignment(NumBytes - NumReusableBytes, StackAlign);
+
// FPDiff will be negative if this tail call requires more space than we
// would automatically have in our incoming argument space. Positive if we
// can actually shrink the stack.
FPDiff = NumReusableBytes - NumBytes;
- // Since callee will pop the argument stack as a tail call, we must keep the
- // popped size aligned to the stack alignment. Either or both of NumBytes
- // and NumReusableBytes may not have been aligned, so we further increase by
- // the amount needed to keep FPDiff aligned, and therefore preserve the
- // required alignment going into the callee.
- uint64_t Realign = offsetToAlignment(FPDiff, StackAlign);
- FPDiff -= Realign;
- NumBytes += Realign;
-
// Update the required reserved area if this is the tail call requiring the
// most argument stack space.
if (FPDiff < 0 && FuncInfo->getTailCallReservedStack() < (unsigned)-FPDiff)
FuncInfo->setTailCallReservedStack(-FPDiff);
- // The stack pointer must be 16-byte aligned at all times it's used for a
- // memory operation, which in practice means at *all* times and in
+ // The stack pointer must be at least 16-byte aligned at all times it's used
+ // for a memory operation, which in practice means at *all* times and in
// particular across call boundaries. Therefore our own arguments started at
- // a 16-byte aligned SP and the delta applied for the tail call should
- // satisfy the same constraint.
- assert(FPDiff % 16 == 0 && "unaligned stack on tail call");
+ // an aligned SP and the delta applied for the tail call should satisfy the
+ // same constraint.
+ assert(isAligned(StackAlign, FPDiff) && "unaligned stack on tail call");
}
auto DescribeCallsite =
@@ -10862,8 +10866,9 @@ AArch64TargetLowering::LowerCall(CallLoweringInfo &CLI,
MF.getFunction().getParent()->getModuleFlag("import-call-optimization"))
DAG.addCalledGlobal(Chain.getNode(), CalledGlobal, OpFlags);
- uint64_t CalleePopBytes =
- DoesCalleeRestoreStack(CallConv, TailCallOpt) ? alignTo(NumBytes, 16) : 0;
+ uint64_t CalleePopBytes = DoesCalleeRestoreStack(CallConv, TailCallOpt)
+ ? alignTo(NumBytes, StackAlign)
+ : 0;
Chain = DAG.getCALLSEQ_END(Chain, NumBytes, CalleePopBytes, InGlue, DL);
InGlue = Chain.getValue(1);
diff --git a/llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll b/llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll
index 9c449b12f2127..5fbb4ecfad5a2 100644
--- a/llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll
+++ b/llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll
@@ -75,5 +75,40 @@ entry:
tail call void @callee_same_args(i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9) "aarch64_inout_za"
ret void
}
+
+; The caller's first 8 formal arguments land in registers, followed by i64 %9 on
+; the stack, which leaves those 8 bytes available for reuse. However the callee
+; only needs 1 byte of outgoing stack arg area for the i1. To keep the FPDiff
+; aligned, we must grow NumBytes by an additional 8 bytes, covering the entire
+; reusable area.
+declare void @callee_last_stack_arg_i1(i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8, i1) "aarch64_inout_za"
+define void @caller_last_stack_arg_i64(i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8, i64 %9) "aarch64_inout_za" uwtable {
+; CHECK-LABEL: caller_last_stack_arg_i64:
+; CHECK: ; %bb.0: ; %entry
+; CHECK-NEXT: stp x29, x30, [sp, #-16]! ; 16-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -8
+; CHECK-NEXT: .cfi_offset w29, -16
+; CHECK-NEXT: bl _external_func
+; CHECK-NEXT: mov w8, #1 ; =0x1
+; CHECK-NEXT: mov w0, #1 ; =0x1
+; CHECK-NEXT: mov w1, #2 ; =0x2
+; CHECK-NEXT: strb w8, [sp, #16]
+; CHECK-NEXT: mov w2, #3 ; =0x3
+; CHECK-NEXT: mov w3, #4 ; =0x4
+; CHECK-NEXT: mov w4, #5 ; =0x5
+; CHECK-NEXT: mov w5, #6 ; =0x6
+; CHECK-NEXT: mov w6, #7 ; =0x7
+; CHECK-NEXT: mov w7, #8 ; =0x8
+; CHECK-NEXT: ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
+; CHECK-NEXT: .cfi_def_cfa_offset 0
+; CHECK-NEXT: .cfi_restore w30
+; CHECK-NEXT: .cfi_restore w29
+; CHECK-NEXT: b _callee_last_stack_arg_i1
+entry:
+ tail call void @external_func()
+ tail call void @callee_last_stack_arg_i1(i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i1 1) "aarch64_inout_za"
+ ret void
+}
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
; GISEL: {{.*}}
diff --git a/llvm/test/CodeGen/AArch64/swifttail-fpdiff-align.ll b/llvm/test/CodeGen/AArch64/swifttail-fpdiff-align.ll
new file mode 100644
index 0000000000000..56124709d7478
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/swifttail-fpdiff-align.ll
@@ -0,0 +1,108 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=arm64-apple-ios -o - | FileCheck %s
+
+declare swiftcc ptr @swift_task_alloc(i64)
+
+; A musttail call modelling a swiftasync context reallocation thunk, whose 10
+; ordinary arguments overflow the 8 GPR argument registers, leaving 9 bytes of
+; stack arg area:
+; * Formal arguments %1-%8 are forwarded in registers x0-x7
+; * swiftasync and swiftself don't count agaginst the usual 8 GPR argument
+; list, and are are passed in dedicated registers (x22/x20)
+; * ptr %9 (8 bytes) and i1 %10 (1 byte) are passed on the stack, packed in 9
+; contiguous bytes
+; We round up that stack arg area to keep the FPDiff (and therefore the stack)
+; 16-byte aligned going into the callee.
+define swifttailcc void @thunk(ptr %1, ptr swiftasync %sa, i64 %2, i64 %3, i64 %4, ptr %5, i64 %6, ptr %7, ptr %8, ptr %9, i1 %10, ptr swiftself %self) #0 {
+; CHECK-LABEL: thunk:
+; CHECK: ; %bb.0: ; %entry
+; CHECK-NEXT: orr x29, x29, #0x1000000000000000
+; CHECK-NEXT: sub sp, sp, #96
+; CHECK-NEXT: stp x28, x27, [sp, #8] ; 16-byte Folded Spill
+; CHECK-NEXT: mov x27, x1
+; CHECK-NEXT: mov x28, x0
+; CHECK-NEXT: stp x26, x25, [sp, #24] ; 16-byte Folded Spill
+; CHECK-NEXT: mov w0, #64 ; =0x40
+; CHECK-NEXT: mov x25, x3
+; CHECK-NEXT: stp x24, x23, [sp, #40] ; 16-byte Folded Spill
+; CHECK-NEXT: mov x23, x5
+; CHECK-NEXT: mov x24, x4
+; CHECK-NEXT: stp x21, x19, [sp, #56] ; 16-byte Folded Spill
+; CHECK-NEXT: mov x19, x7
+; CHECK-NEXT: mov x21, x6
+; CHECK-NEXT: stp x29, x30, [sp, #80] ; 16-byte Folded Spill
+; CHECK-NEXT: add x29, sp, #80
+; CHECK-NEXT: mov x26, x2
+; CHECK-NEXT: str x22, [sp, #72]
+; CHECK-NEXT: ldr x1, [x20]
+; CHECK-NEXT: str x1, [sp] ; 8-byte Spill
+; CHECK-NEXT: bl _swift_task_alloc
+; CHECK-NEXT: ldp x29, x30, [sp, #80] ; 16-byte Folded Reload
+; CHECK-NEXT: mov x8, x0
+; CHECK-NEXT: str x0, [x22, #16]
+; CHECK-NEXT: mov x0, x28
+; CHECK-NEXT: mov x1, x27
+; CHECK-NEXT: mov x2, x26
+; CHECK-NEXT: mov x3, x25
+; CHECK-NEXT: mov x4, x24
+; CHECK-NEXT: mov x5, x23
+; CHECK-NEXT: mov x6, x21
+; CHECK-NEXT: mov x7, x19
+; CHECK-NEXT: ldp x21, x19, [sp, #56] ; 16-byte Folded Reload
+; CHECK-NEXT: mov x22, x8
+; CHECK-NEXT: ldp x24, x23, [sp, #40] ; 16-byte Folded Reload
+; CHECK-NEXT: ldr x8, [sp] ; 8-byte Reload
+; CHECK-NEXT: ldp x26, x25, [sp, #24] ; 16-byte Folded Reload
+; CHECK-NEXT: and x29, x29, #0xefffffffffffffff
+; CHECK-NEXT: ldp x28, x27, [sp, #8] ; 16-byte Folded Reload
+; CHECK-NEXT: add sp, sp, #96
+; CHECK-NEXT: br x8
+entry:
+ %oldCtx = getelementptr inbounds i8, ptr %sa, i64 16
+ %fn = load ptr, ptr %self, align 8
+ %newCtx = call swiftcc ptr @swift_task_alloc(i64 64)
+ store ptr %newCtx, ptr %oldCtx, align 8
+ musttail call swifttailcc void %fn(ptr %1, ptr swiftasync %newCtx, i64 %2, i64 %3, i64 %4, ptr %5, i64 %6, ptr %7, ptr %8, ptr %9, i1 %10, ptr swiftself %self)
+ ret void
+}
+
+; A musttail call from a function with no arguments (and therefore no reusable
+; stack arg area) to a function with 11 i32 arguments that overflow the 8 GPR
+; argument registers, leaving 4 bytes of stack arg area:
+; * The first 8 i32 arguments land in registers w0-w7
+; * The arg is passed on the stack, consuming 4 bytes
+; We round up that stack arg area to keep the FPDiff (and therefore the stack)
+; 16-byte aligned going into the callee.
+define swifttailcc void @caller_no_args() {
+; CHECK-LABEL: caller_no_args:
+; CHECK: ; %bb.0: ; %entry
+; CHECK-NEXT: mov w8, #8 ; =0x8
+; CHECK-NEXT: str w8, [sp, #-16]!
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: mov w0, wzr
+; CHECK-NEXT: mov w1, #1 ; =0x1
+; CHECK-NEXT: mov w2, #2 ; =0x2
+; CHECK-NEXT: mov w3, #3 ; =0x3
+; CHECK-NEXT: mov w4, #4 ; =0x4
+; CHECK-NEXT: mov w5, #5 ; =0x5
+; CHECK-NEXT: mov w6, #6 ; =0x6
+; CHECK-NEXT: mov w7, #7 ; =0x7
+; CHECK-NEXT: b _callee_11_args
+entry:
+ musttail call swifttailcc void @callee_11_args(i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8)
+ ret void
+}
+
+; The corresponding callee pops swifttail callee, with 11 i32 arguments that
+; overflow the 8 GPR argument registers, leaving 4 bytes of stack arg area,
+; which is rounded up to keep the stack 16-byte aligned. The amount popped here
+; must match what was pushed in @caller_no_args.
+define swifttailcc void @callee_11_args(i32, i32, i32, i32, i32, i32, i32, i32, i32) "noinline" {
+; CHECK-LABEL: callee_11_args:
+; CHECK: ; %bb.0:
+; CHECK-NEXT: add sp, sp, #16
+; CHECK-NEXT: ret
+ ret void
+}
+
+attributes #0 = { nounwind "frame-pointer"="non-leaf-no-reserve" }
diff --git a/llvm/test/CodeGen/AArch64/tail-call-stack-args.ll b/llvm/test/CodeGen/AArch64/tail-call-stack-args.ll
index bbeea28f18bda..5bf3bffdac1a8 100644
--- a/llvm/test/CodeGen/AArch64/tail-call-stack-args.ll
+++ b/llvm/test/CodeGen/AArch64/tail-call-stack-args.ll
@@ -78,3 +78,24 @@ define void @wrapper_func_i64(i32 %a, i32 %b, i32 %c, i32 %d, i32 %e, i32 %f, i3
tail call void @func_i64(i32 %a, i32 %b, i32 %c, i32 %d, i32 %e, i32 %f, i32 %g, i32 %h, i32 %i, i64 %conv)
ret void
}
+
+; A musttail call with 9 bytes of stack arg area: the first 8 formal arguments
+; land in registers, followed by ptr %8 (8 bytes) and i1 %9 (1 byte) packed
+; tightly on the stack without padding. 9 is neither 8 nor 16-byte aligned, so
+; we need to round up NumBytes of frame size in order to keep the FPDiff aligned
+; for a stack-restoring (tailcc) callee.
+declare tailcc void @func_9_bytes_argspace(i64, i64, i64, i64, i64, i64, i64, i64, ptr, i1)
+
+define tailcc void @wrapper_func_9_bytes_argspace(i64 %0, i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i64 %6, i64 %7, ptr %8, i1 %9) {
+; SDAG-LABEL: wrapper_func_9_bytes_argspace:
+; SDAG: // %bb.0:
+; SDAG-NEXT: b func_9_bytes_argspace
+;
+; GI-LABEL: wrapper_func_9_bytes_argspace:
+; GI: // %bb.0:
+; GI-NEXT: ldrb w8, [sp, #8]
+; GI-NEXT: strb w8, [sp, #8]
+; GI-NEXT: b func_9_bytes_argspace
+ musttail call tailcc void @func_9_bytes_argspace(i64 %0, i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i64 %6, i64 %7, ptr %8, i1 %9)
+ ret void
+}
More information about the llvm-branch-commits
mailing list