[llvm-branch-commits] [llvm] release/23.x: [llvm][AArch64] Fix FPDiff founding direction in non-sibcall tail calls (#223545) (PR #224137)

via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Wed Sep 16 14:00:59 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-aarch64

Author: Jon Roelofs (jroelofs)

<details>
<summary>Changes</summary>

This fixes another subtle bug in frame accounting (see: #<!-- -->217156 /
#<!-- -->220406), for tail calls that have a non-multiple of 16 bytes worth of
stack arg area, and need that stack arg re-use to be increased to cover
the alignment requirement. This is best illustrated with callers
containing 8 formal arguments covering the first 8 GPRs (x0-x7),
followed by 9 bytes of argument passed on the stack.
    
In a callee-pops tail call (e.g. tailcc/swifttailcc), the set of
reusable stack arg area bytes has already been sufficiently aligned by
LowerFormalArguments, so growing NumBytes up to StackAlign is enough to
consume that excess. Otherwise (e.g. a plain C-convention call, forced
off the sibcall path, as in the aarch64_inout_za tests), we can't rely
on either having been pre-aligned, so we round NumBytes up to the same
residue mod StackAlign as NumReusableBytes, which cancels the residue
out of their difference (FPDiff), thus keeping the stack aligned going
into the callee.
    
Concretely, #<!-- -->220406 computed the Realign distance needed to round FPDiff
up to the next multiple of StackAlign, then subtracted that from FPDiff.
Subtracting that round-up distance overshoots rounding down to the
nearest multiple, and in general arrives at a value that may not even be
a multiple of StackAlign. For example, in @<!-- -->caller_last_stack_arg_i64 we
had NumReusableBytes=8 and NumBytes=1 for an FPDiff of 7, which computed
Realign=9, FPDiff=-2, NumBytes=10 instead of the intended NumBytes=8,
FPDiff=0.
    
rdar://187331378
(cherry picked from commit ec4d25ab1d0f47f078fd749f6073f04cc820ddf3)

---
Full diff: https://github.com/llvm/llvm-project/pull/224137.diff


5 Files Affected:

- (modified) llvm/lib/Target/AArch64/AArch64ISelLowering.cpp (+21-10) 
- (modified) llvm/test/CodeGen/AArch64/arm64ec-varargs.ll (+31) 
- (added) llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll (+114) 
- (added) llvm/test/CodeGen/AArch64/swifttail-fpdiff-align.ll (+108) 
- (modified) llvm/test/CodeGen/AArch64/tail-call-stack-args.ll (+21) 


``````````diff
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index d134e7f911f83..f71c13c1085a0 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -10286,13 +10286,23 @@ AArch64TargetLowering::LowerCall(CallLoweringInfo &CLI,
   // caller will deallocate the entire stack and the callee still expects its
   // arguments to begin at SP+0. Completely unused for non-tail calls.
   int FPDiff = 0;
+  const Align StackAlign = Subtarget->getFrameLowering()->getStackAlign();
 
   if (IsTailCall && !IsSibCall) {
     unsigned NumReusableBytes = FuncInfo->getBytesInStackArgArea();
 
-    // Since callee will pop argument stack as a tail call, we must keep the
-    // popped size 16-byte aligned.
-    NumBytes = alignTo(NumBytes, 16);
+    // In general, neither NumBytes nor NumReusableBytes is guaranteed to be
+    // aligned, so we round NumBytes up to the same residue mod StackAlign as
+    // NumReusableBytes, which keeps their difference (FPDiff) a multiple of
+    // StackAlign, and therefore preserve the required stack alignment going
+    // into the callee.  When the callee's convention can guarantee TCO,
+    // LowerFormalArguments will have force-aligned the stack arg area for us
+    // already, so we can count on our own alignment of NumBytes below to result
+    // in an aligned FPDiff.
+    assert((!DoesCalleeRestoreStack(CallConv, TailCallOpt) ||
+            isAligned(StackAlign, NumReusableBytes)) &&
+           "expected LowerFormalArguments to force-align stack arg area");
+    NumBytes += offsetToAlignment(NumBytes - NumReusableBytes, StackAlign);
 
     // FPDiff will be negative if this tail call requires more space than we
     // would automatically have in our incoming argument space. Positive if we
@@ -10304,12 +10314,12 @@ AArch64TargetLowering::LowerCall(CallLoweringInfo &CLI,
     if (FPDiff < 0 && FuncInfo->getTailCallReservedStack() < (unsigned)-FPDiff)
       FuncInfo->setTailCallReservedStack(-FPDiff);
 
-    // The stack pointer must be 16-byte aligned at all times it's used for a
-    // memory operation, which in practice means at *all* times and in
+    // The stack pointer must be at least 16-byte aligned at all times it's used
+    // for a memory operation, which in practice means at *all* times and in
     // particular across call boundaries. Therefore our own arguments started at
-    // a 16-byte aligned SP and the delta applied for the tail call should
-    // satisfy the same constraint.
-    assert(FPDiff % 16 == 0 && "unaligned stack on tail call");
+    // an aligned SP and the delta applied for the tail call should satisfy the
+    // same constraint.
+    assert(isAligned(StackAlign, FPDiff) && "unaligned stack on tail call");
   }
 
   auto DescribeCallsite =
@@ -10856,8 +10866,9 @@ AArch64TargetLowering::LowerCall(CallLoweringInfo &CLI,
       MF.getFunction().getParent()->getModuleFlag("import-call-optimization"))
     DAG.addCalledGlobal(Chain.getNode(), CalledGlobal, OpFlags);
 
-  uint64_t CalleePopBytes =
-      DoesCalleeRestoreStack(CallConv, TailCallOpt) ? alignTo(NumBytes, 16) : 0;
+  uint64_t CalleePopBytes = DoesCalleeRestoreStack(CallConv, TailCallOpt)
+                                ? alignTo(NumBytes, StackAlign)
+                                : 0;
 
   Chain = DAG.getCALLSEQ_END(Chain, NumBytes, CalleePopBytes, InGlue, DL);
   InGlue = Chain.getValue(1);
diff --git a/llvm/test/CodeGen/AArch64/arm64ec-varargs.ll b/llvm/test/CodeGen/AArch64/arm64ec-varargs.ll
index 979e09cfb5fac..6f13b9606afeb 100644
--- a/llvm/test/CodeGen/AArch64/arm64ec-varargs.ll
+++ b/llvm/test/CodeGen/AArch64/arm64ec-varargs.ll
@@ -157,3 +157,34 @@ define void @varargs_thunk(ptr noundef %0, ...) "thunk" {
   musttail call void (ptr, ...) %vtablefn(ptr noundef %0, ...)
   ret void
 }
+
+declare tailcc void @callee_9_variadic(i64, i64, i64, i64, i64, i64, i64, i64, i64, ...)
+
+define tailcc void @caller_8_args(i64, i64, i64, i64, i64, i64, i64, i64) {
+; CHECK-LABEL: caller_8_args:
+; CHECK:       .seh_proc caller_8_args
+; CHECK-NEXT:  // %bb.0: // %entry
+; CHECK-NEXT:    sub sp, sp, #16
+; CHECK-NEXT:    .seh_stackalloc 16
+; CHECK-NEXT:    .seh_endprologue
+; CHECK-NEXT:    mov w8, #9 // =0x9
+; CHECK-NEXT:    mov x4, sp
+; CHECK-NEXT:    mov w0, #1 // =0x1
+; CHECK-NEXT:    mov w1, #2 // =0x2
+; CHECK-NEXT:    mov w2, #3 // =0x3
+; CHECK-NEXT:    mov w3, #4 // =0x4
+; CHECK-NEXT:    mov w6, #7 // =0x7
+; CHECK-NEXT:    mov w7, #8 // =0x8
+; CHECK-NEXT:    mov w5, #16 // =0x10
+; CHECK-NEXT:    str x8, [sp]
+; CHECK-NEXT:    .weak_anti_dep callee_9_variadic
+; CHECK-NEXT:  callee_9_variadic = "#callee_9_variadic"
+; CHECK-NEXT:    .weak_anti_dep "#callee_9_variadic"
+; CHECK-NEXT:  "#callee_9_variadic" = callee_9_variadic
+; CHECK-NEXT:    b "#callee_9_variadic"
+; CHECK-NEXT:    .seh_endfunclet
+; CHECK-NEXT:    .seh_endproc
+entry:
+  tail call tailcc void (i64, i64, i64, i64, i64, i64, i64, i64, i64, ...) @callee_9_variadic(i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9)
+  ret void
+}
diff --git a/llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll b/llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll
new file mode 100644
index 0000000000000..5fbb4ecfad5a2
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll
@@ -0,0 +1,114 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=arm64-apple-ios -mattr=+sme -o - | FileCheck %s
+; RUN: llc < %s -mtriple=arm64-apple-ios -mattr=+sme -global-isel-abort=2 -o - 2>&1 | FileCheck %s --check-prefix=CHECK,GISEL
+
+; GlobalISel doesn't implement SME yet, but once it does, we should be mindful of this case.
+; GISEL: warning: Instruction selection used fallback path for caller_more_args
+; GISEL: warning: Instruction selection used fallback path for caller_same_args
+
+declare void @external_func() "aarch64_inout_za"
+
+; ZA state being live for both caller and callee forces the non-sibcall tail
+; call fast path. FPDiff = (caller's incoming argument stack) - (callee's
+; outgoing argument stack) can come out either positive or negative, and in
+; either case its magnitude must be rounded away from zero to a multiple of
+; 16 to keep the stack pointer 16-byte aligned at all times.
+declare void @callee_fewer_args(i64, i64, i64, i64, i64, i64, i64, i64) "aarch64_inout_za"
+define void @caller_more_args(i64, i64, i64, i64, i64, i64, i64, i64, i64, i64) "aarch64_inout_za" uwtable {
+; CHECK-LABEL: caller_more_args:
+; CHECK:       ; %bb.0: ; %entry
+; CHECK-NEXT:    stp x29, x30, [sp, #-16]! ; 16-byte Folded Spill
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    .cfi_offset w30, -8
+; CHECK-NEXT:    .cfi_offset w29, -16
+; CHECK-NEXT:    bl _external_func
+; CHECK-NEXT:    mov w0, #1 ; =0x1
+; CHECK-NEXT:    mov w1, #2 ; =0x2
+; CHECK-NEXT:    mov w2, #3 ; =0x3
+; CHECK-NEXT:    mov w3, #4 ; =0x4
+; CHECK-NEXT:    mov w4, #5 ; =0x5
+; CHECK-NEXT:    mov w5, #6 ; =0x6
+; CHECK-NEXT:    mov w6, #7 ; =0x7
+; CHECK-NEXT:    mov w7, #8 ; =0x8
+; CHECK-NEXT:    ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
+; CHECK-NEXT:    .cfi_def_cfa_offset 0
+; CHECK-NEXT:    add sp, sp, #16
+; CHECK-NEXT:    .cfi_def_cfa_offset -16
+; CHECK-NEXT:    .cfi_restore w30
+; CHECK-NEXT:    .cfi_restore w29
+; CHECK-NEXT:    b _callee_fewer_args
+entry:
+  tail call void @external_func()
+  tail call void @callee_fewer_args(i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8) "aarch64_inout_za"
+  ret void
+}
+
+; Likewise for a callee that has the same amount of argument stack as the
+; caller, we must round up the amount of used stack space in order to keep
+; FPDiff a multiple of 16, thus keeping the stack aligned for the callee.
+declare void @callee_same_args(i64, i64, i64, i64, i64, i64, i64, i64, i64) "aarch64_inout_za"
+define void @caller_same_args(i64, i64, i64, i64, i64, i64, i64, i64, i64) "aarch64_inout_za" uwtable {
+; CHECK-LABEL: caller_same_args:
+; CHECK:       ; %bb.0: ; %entry
+; CHECK-NEXT:    stp x29, x30, [sp, #-16]! ; 16-byte Folded Spill
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    .cfi_offset w30, -8
+; CHECK-NEXT:    .cfi_offset w29, -16
+; CHECK-NEXT:    bl _external_func
+; CHECK-NEXT:    mov w8, #9 ; =0x9
+; CHECK-NEXT:    mov w0, #1 ; =0x1
+; CHECK-NEXT:    mov w1, #2 ; =0x2
+; CHECK-NEXT:    str x8, [sp, #16]
+; CHECK-NEXT:    mov w2, #3 ; =0x3
+; CHECK-NEXT:    mov w3, #4 ; =0x4
+; CHECK-NEXT:    mov w4, #5 ; =0x5
+; CHECK-NEXT:    mov w5, #6 ; =0x6
+; CHECK-NEXT:    mov w6, #7 ; =0x7
+; CHECK-NEXT:    mov w7, #8 ; =0x8
+; CHECK-NEXT:    ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
+; CHECK-NEXT:    .cfi_def_cfa_offset 0
+; CHECK-NEXT:    .cfi_restore w30
+; CHECK-NEXT:    .cfi_restore w29
+; CHECK-NEXT:    b _callee_same_args
+entry:
+  tail call void @external_func()
+  tail call void @callee_same_args(i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9) "aarch64_inout_za"
+  ret void
+}
+
+; The caller's first 8 formal arguments land in registers, followed by i64 %9 on
+; the stack, which leaves those 8 bytes available for reuse. However the callee
+; only needs 1 byte of outgoing stack arg area for the i1. To keep the FPDiff
+; aligned, we must grow NumBytes by an additional 8 bytes, covering the entire
+; reusable area.
+declare void @callee_last_stack_arg_i1(i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8, i1) "aarch64_inout_za"
+define void @caller_last_stack_arg_i64(i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8, i64 %9) "aarch64_inout_za" uwtable {
+; CHECK-LABEL: caller_last_stack_arg_i64:
+; CHECK:       ; %bb.0: ; %entry
+; CHECK-NEXT:    stp x29, x30, [sp, #-16]! ; 16-byte Folded Spill
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    .cfi_offset w30, -8
+; CHECK-NEXT:    .cfi_offset w29, -16
+; CHECK-NEXT:    bl _external_func
+; CHECK-NEXT:    mov w8, #1 ; =0x1
+; CHECK-NEXT:    mov w0, #1 ; =0x1
+; CHECK-NEXT:    mov w1, #2 ; =0x2
+; CHECK-NEXT:    strb w8, [sp, #16]
+; CHECK-NEXT:    mov w2, #3 ; =0x3
+; CHECK-NEXT:    mov w3, #4 ; =0x4
+; CHECK-NEXT:    mov w4, #5 ; =0x5
+; CHECK-NEXT:    mov w5, #6 ; =0x6
+; CHECK-NEXT:    mov w6, #7 ; =0x7
+; CHECK-NEXT:    mov w7, #8 ; =0x8
+; CHECK-NEXT:    ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
+; CHECK-NEXT:    .cfi_def_cfa_offset 0
+; CHECK-NEXT:    .cfi_restore w30
+; CHECK-NEXT:    .cfi_restore w29
+; CHECK-NEXT:    b _callee_last_stack_arg_i1
+entry:
+  tail call void @external_func()
+  tail call void @callee_last_stack_arg_i1(i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i1 1) "aarch64_inout_za"
+  ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GISEL: {{.*}}
diff --git a/llvm/test/CodeGen/AArch64/swifttail-fpdiff-align.ll b/llvm/test/CodeGen/AArch64/swifttail-fpdiff-align.ll
new file mode 100644
index 0000000000000..56124709d7478
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/swifttail-fpdiff-align.ll
@@ -0,0 +1,108 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=arm64-apple-ios -o - | FileCheck %s
+
+declare swiftcc ptr @swift_task_alloc(i64)
+
+; A musttail call modelling a swiftasync context reallocation thunk, whose 10
+; ordinary arguments overflow the 8 GPR argument registers, leaving 9 bytes of
+; stack arg area:
+;   * Formal arguments %1-%8 are forwarded in registers x0-x7
+;   * swiftasync and swiftself don't count agaginst the usual 8 GPR argument
+;     list, and are are passed in dedicated registers (x22/x20)
+;   * ptr %9 (8 bytes) and i1 %10 (1 byte) are passed on the stack, packed in 9
+;     contiguous bytes
+; We round up that stack arg area to keep the FPDiff (and therefore the stack)
+; 16-byte aligned going into the callee.
+define swifttailcc void @thunk(ptr %1, ptr swiftasync %sa, i64 %2, i64 %3, i64 %4, ptr %5, i64 %6, ptr %7, ptr %8, ptr %9, i1 %10, ptr swiftself %self) #0 {
+; CHECK-LABEL: thunk:
+; CHECK:       ; %bb.0: ; %entry
+; CHECK-NEXT:    orr x29, x29, #0x1000000000000000
+; CHECK-NEXT:    sub sp, sp, #96
+; CHECK-NEXT:    stp x28, x27, [sp, #8] ; 16-byte Folded Spill
+; CHECK-NEXT:    mov x27, x1
+; CHECK-NEXT:    mov x28, x0
+; CHECK-NEXT:    stp x26, x25, [sp, #24] ; 16-byte Folded Spill
+; CHECK-NEXT:    mov w0, #64 ; =0x40
+; CHECK-NEXT:    mov x25, x3
+; CHECK-NEXT:    stp x24, x23, [sp, #40] ; 16-byte Folded Spill
+; CHECK-NEXT:    mov x23, x5
+; CHECK-NEXT:    mov x24, x4
+; CHECK-NEXT:    stp x21, x19, [sp, #56] ; 16-byte Folded Spill
+; CHECK-NEXT:    mov x19, x7
+; CHECK-NEXT:    mov x21, x6
+; CHECK-NEXT:    stp x29, x30, [sp, #80] ; 16-byte Folded Spill
+; CHECK-NEXT:    add x29, sp, #80
+; CHECK-NEXT:    mov x26, x2
+; CHECK-NEXT:    str x22, [sp, #72]
+; CHECK-NEXT:    ldr x1, [x20]
+; CHECK-NEXT:    str x1, [sp] ; 8-byte Spill
+; CHECK-NEXT:    bl _swift_task_alloc
+; CHECK-NEXT:    ldp x29, x30, [sp, #80] ; 16-byte Folded Reload
+; CHECK-NEXT:    mov x8, x0
+; CHECK-NEXT:    str x0, [x22, #16]
+; CHECK-NEXT:    mov x0, x28
+; CHECK-NEXT:    mov x1, x27
+; CHECK-NEXT:    mov x2, x26
+; CHECK-NEXT:    mov x3, x25
+; CHECK-NEXT:    mov x4, x24
+; CHECK-NEXT:    mov x5, x23
+; CHECK-NEXT:    mov x6, x21
+; CHECK-NEXT:    mov x7, x19
+; CHECK-NEXT:    ldp x21, x19, [sp, #56] ; 16-byte Folded Reload
+; CHECK-NEXT:    mov x22, x8
+; CHECK-NEXT:    ldp x24, x23, [sp, #40] ; 16-byte Folded Reload
+; CHECK-NEXT:    ldr x8, [sp] ; 8-byte Reload
+; CHECK-NEXT:    ldp x26, x25, [sp, #24] ; 16-byte Folded Reload
+; CHECK-NEXT:    and x29, x29, #0xefffffffffffffff
+; CHECK-NEXT:    ldp x28, x27, [sp, #8] ; 16-byte Folded Reload
+; CHECK-NEXT:    add sp, sp, #96
+; CHECK-NEXT:    br x8
+entry:
+  %oldCtx = getelementptr inbounds i8, ptr %sa, i64 16
+  %fn = load ptr, ptr %self, align 8
+  %newCtx = call swiftcc ptr @swift_task_alloc(i64 64)
+  store ptr %newCtx, ptr %oldCtx, align 8
+  musttail call swifttailcc void %fn(ptr %1, ptr swiftasync %newCtx, i64 %2, i64 %3, i64 %4, ptr %5, i64 %6, ptr %7, ptr %8, ptr %9, i1 %10, ptr swiftself %self)
+  ret void
+}
+
+; A musttail call from a function with no arguments (and therefore no reusable
+; stack arg area) to a function with 11 i32 arguments that overflow the 8 GPR
+; argument registers, leaving 4 bytes of stack arg area:
+;   * The first 8 i32 arguments land in registers w0-w7
+;   * The arg is passed on the stack, consuming 4 bytes
+; We round up that stack arg area to keep the FPDiff (and therefore the stack)
+; 16-byte aligned going into the callee.
+define swifttailcc void @caller_no_args() {
+; CHECK-LABEL: caller_no_args:
+; CHECK:       ; %bb.0: ; %entry
+; CHECK-NEXT:    mov w8, #8 ; =0x8
+; CHECK-NEXT:    str w8, [sp, #-16]!
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    mov w0, wzr
+; CHECK-NEXT:    mov w1, #1 ; =0x1
+; CHECK-NEXT:    mov w2, #2 ; =0x2
+; CHECK-NEXT:    mov w3, #3 ; =0x3
+; CHECK-NEXT:    mov w4, #4 ; =0x4
+; CHECK-NEXT:    mov w5, #5 ; =0x5
+; CHECK-NEXT:    mov w6, #6 ; =0x6
+; CHECK-NEXT:    mov w7, #7 ; =0x7
+; CHECK-NEXT:    b _callee_11_args
+entry:
+  musttail call swifttailcc void @callee_11_args(i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8)
+  ret void
+}
+
+; The corresponding callee pops swifttail callee, with 11 i32 arguments that
+; overflow the 8 GPR argument registers, leaving 4 bytes of stack arg area,
+; which is rounded up to keep the stack 16-byte aligned. The amount popped here
+; must match what was pushed in @caller_no_args.
+define swifttailcc void @callee_11_args(i32, i32, i32, i32, i32, i32, i32, i32, i32) "noinline" {
+; CHECK-LABEL: callee_11_args:
+; CHECK:       ; %bb.0:
+; CHECK-NEXT:    add sp, sp, #16
+; CHECK-NEXT:    ret
+  ret void
+}
+
+attributes #0 = { nounwind "frame-pointer"="non-leaf-no-reserve" }
diff --git a/llvm/test/CodeGen/AArch64/tail-call-stack-args.ll b/llvm/test/CodeGen/AArch64/tail-call-stack-args.ll
index bbeea28f18bda..5bf3bffdac1a8 100644
--- a/llvm/test/CodeGen/AArch64/tail-call-stack-args.ll
+++ b/llvm/test/CodeGen/AArch64/tail-call-stack-args.ll
@@ -78,3 +78,24 @@ define void @wrapper_func_i64(i32 %a, i32 %b, i32 %c, i32 %d, i32 %e, i32 %f, i3
   tail call void @func_i64(i32 %a, i32 %b, i32 %c, i32 %d, i32 %e, i32 %f, i32 %g, i32 %h, i32 %i, i64 %conv)
   ret void
 }
+
+; A musttail call with 9 bytes of stack arg area: the first 8 formal arguments
+; land in registers, followed by ptr %8 (8 bytes) and i1 %9 (1 byte) packed
+; tightly on the stack without padding. 9 is neither 8 nor 16-byte aligned, so
+; we need to round up NumBytes of frame size in order to keep the FPDiff aligned
+; for a stack-restoring (tailcc) callee.
+declare tailcc void @func_9_bytes_argspace(i64, i64, i64, i64, i64, i64, i64, i64, ptr, i1)
+
+define tailcc void @wrapper_func_9_bytes_argspace(i64 %0, i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i64 %6, i64 %7, ptr %8, i1 %9) {
+; SDAG-LABEL: wrapper_func_9_bytes_argspace:
+; SDAG:       // %bb.0:
+; SDAG-NEXT:    b func_9_bytes_argspace
+;
+; GI-LABEL: wrapper_func_9_bytes_argspace:
+; GI:       // %bb.0:
+; GI-NEXT:    ldrb w8, [sp, #8]
+; GI-NEXT:    strb w8, [sp, #8]
+; GI-NEXT:    b func_9_bytes_argspace
+  musttail call tailcc void @func_9_bytes_argspace(i64 %0, i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i64 %6, i64 %7, ptr %8, i1 %9)
+  ret void
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/224137


More information about the llvm-branch-commits mailing list