[llvm] ad695d4 - [AArch64] Don't emit stack restore for SME ZA non-sibling tail calls (#224721)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 23 13:54:23 PDT 2026


Author: David Tellenbach
Date: 2026-09-23T20:54:15Z
New Revision: ad695d46e4535f953558705541b71c2d938fb72b

URL: https://github.com/llvm/llvm-project/commit/ad695d46e4535f953558705541b71c2d938fb72b
DIFF: https://github.com/llvm/llvm-project/commit/ad695d46e4535f953558705541b71c2d938fb72b.diff

LOG: [AArch64] Don't emit stack restore for SME ZA non-sibling tail calls (#224721)

A tail call from a function with live ZA state was previously prevented
from being lowered as a sibling call to have a CALLSEQ_START to glue
INOUT_ZA_USE to. This led to bogus stack restores.

We now drop the INOUT_ZA_USE marker on tail calls and can thus lower as
sibling calls if needed to take advantage of the existing correct stack
restore behavior.

This is safe to do since tail calls use TCRETURN and the MachineSMEABI
pass requires live ZA state to stay live across returns anyway.

Added: 
    llvm/test/CodeGen/AArch64/sme-za-non-sibling-tailcall.ll

Modified: 
    llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
    llvm/test/CodeGen/AArch64/sme-za-tailcall-fpdiff-align.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 7332eb95cb845..6467976b42add 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -10600,6 +10600,19 @@ AArch64TargetLowering::LowerCall(CallLoweringInfo &CLI,
     // Check if it's really possible to do a tail call.
     IsTailCall = isEligibleForTailCallOptimization(CLI);
 
+    // If we have a tail-call, it's safe to drop ZAMarkerNode since
+    // 1. INOUT_ZA_USE is the only marker node that can reach this point
+    // (otherwise, the call is not elegible for tail call optimization at all),
+    // and
+    // 2. INOUT_ZA_USE is redunant on a tail call, since a tail call is a return
+    // for which MachineSMEABIPass requires an acitve ZA state anyway, the
+    // marker node doesn't add anything.
+    if (IsTailCall && ZAMarkerNode) {
+      assert(ZAMarkerNode == AArch64ISD::INOUT_ZA_USE &&
+             "Unexpected SME ZA marker node");
+      ZAMarkerNode = std::nullopt;
+    }
+
     // A sibling call is one where we're under the usual C ABI and not planning
     // to change that but can still do a tail call:
     if (!ZAMarkerNode && !TailCallOpt && IsTailCall &&

diff  --git a/llvm/test/CodeGen/AArch64/sme-za-non-sibling-tailcall.ll b/llvm/test/CodeGen/AArch64/sme-za-non-sibling-tailcall.ll
new file mode 100644
index 0000000000000..d66333e9cfec0
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sme-za-non-sibling-tailcall.ll
@@ -0,0 +1,110 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple arm64-apple-ios -mattr +sme -o - < %s | FileCheck %s
+
+declare void @callee() "aarch64_inout_za"
+declare void @clobber()
+declare void @za1() "aarch64_inout_za"
+declare void @za2() "aarch64_inout_za"
+
+define void @caller0(i64 %0, i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8) "aarch64_inout_za" {
+; CHECK-LABEL: caller0:
+; CHECK:       ; %bb.0:
+; CHECK-NEXT:    b _callee
+  tail call void @callee() "aarch64_inout_za"
+  ret void
+}
+
+define void @caller1(i64 %0, i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8, i64 %9) "aarch64_inout_za" {
+; CHECK-LABEL: caller1:
+; CHECK:       ; %bb.0:
+; CHECK-NEXT:    b _callee
+  tail call void @callee() "aarch64_inout_za"
+  ret void
+}
+
+define void @caller_with_clobber() "aarch64_inout_za" {
+; CHECK-LABEL: caller_with_clobber:
+; CHECK:       ; %bb.0:
+; CHECK-NEXT:    stp x29, x30, [sp, #-16]! ; 16-byte Folded Spill
+; CHECK-NEXT:    mov x29, sp
+; CHECK-NEXT:    sub sp, sp, #16
+; CHECK-NEXT:    .cfi_def_cfa w29, 16
+; CHECK-NEXT:    .cfi_offset w30, -8
+; CHECK-NEXT:    .cfi_offset w29, -16
+; CHECK-NEXT:    rdsvl x8, #1
+; CHECK-NEXT:    mov x9, sp
+; CHECK-NEXT:    msub x9, x8, x8, x9
+; CHECK-NEXT:    mov sp, x9
+; CHECK-NEXT:    sub x10, x29, #16
+; CHECK-NEXT:    stp x9, x8, [x29, #-16]
+; CHECK-NEXT:    msr TPIDR2_EL0, x10
+; CHECK-NEXT:    bl _clobber
+; CHECK-NEXT:    smstart za
+; CHECK-NEXT:    mrs x8, TPIDR2_EL0
+; CHECK-NEXT:    sub x0, x29, #16
+; CHECK-NEXT:    cbnz x8, LBB2_2
+; CHECK-NEXT:  ; %bb.1:
+; CHECK-NEXT:    bl ___arm_tpidr2_restore
+; CHECK-NEXT:  LBB2_2:
+; CHECK-NEXT:    msr TPIDR2_EL0, xzr
+; CHECK-NEXT:    mov sp, x29
+; CHECK-NEXT:    ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
+; CHECK-NEXT:    b _callee
+  call void @clobber()
+  tail call void @callee() "aarch64_inout_za"
+  ret void
+}
+
+define void @live_chain() "aarch64_inout_za" {
+; CHECK-LABEL: live_chain:
+; CHECK:       ; %bb.0:
+; CHECK-NEXT:    stp x29, x30, [sp, #-16]! ; 16-byte Folded Spill
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    .cfi_offset w30, -8
+; CHECK-NEXT:    .cfi_offset w29, -16
+; CHECK-NEXT:    bl _za1
+; CHECK-NEXT:    bl _za2
+; CHECK-NEXT:    ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
+; CHECK-NEXT:    b _callee
+  call void @za1()
+  call void @za2()
+  tail call void @callee() "aarch64_inout_za"
+  ret void
+}
+
+define void @mixed_chain() "aarch64_inout_za" {
+; CHECK-LABEL: mixed_chain:
+; CHECK:       ; %bb.0:
+; CHECK-NEXT:    stp x29, x30, [sp, #-16]! ; 16-byte Folded Spill
+; CHECK-NEXT:    mov x29, sp
+; CHECK-NEXT:    sub sp, sp, #16
+; CHECK-NEXT:    .cfi_def_cfa w29, 16
+; CHECK-NEXT:    .cfi_offset w30, -8
+; CHECK-NEXT:    .cfi_offset w29, -16
+; CHECK-NEXT:    rdsvl x8, #1
+; CHECK-NEXT:    mov x9, sp
+; CHECK-NEXT:    msub x9, x8, x8, x9
+; CHECK-NEXT:    mov sp, x9
+; CHECK-NEXT:    stp x9, x8, [x29, #-16]
+; CHECK-NEXT:    bl _za1
+; CHECK-NEXT:    sub x8, x29, #16
+; CHECK-NEXT:    msr TPIDR2_EL0, x8
+; CHECK-NEXT:    bl _clobber
+; CHECK-NEXT:    smstart za
+; CHECK-NEXT:    mrs x8, TPIDR2_EL0
+; CHECK-NEXT:    sub x0, x29, #16
+; CHECK-NEXT:    cbnz x8, LBB4_2
+; CHECK-NEXT:  ; %bb.1:
+; CHECK-NEXT:    bl ___arm_tpidr2_restore
+; CHECK-NEXT:  LBB4_2:
+; CHECK-NEXT:    msr TPIDR2_EL0, xzr
+; CHECK-NEXT:    bl _za2
+; CHECK-NEXT:    mov sp, x29
+; CHECK-NEXT:    ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
+; CHECK-NEXT:    b _callee
+  call void @za1()
+  call void @clobber()
+  call void @za2()
+  tail call void @callee() "aarch64_inout_za"
+  ret void
+}

diff  --git a/llvm/test/CodeGen/AArch64/sme-za-tailcall-fp
diff -align.ll b/llvm/test/CodeGen/AArch64/sme-za-tailcall-fp
diff -align.ll
index 5fbb4ecfad5a2..2a3352a168c7e 100644
--- a/llvm/test/CodeGen/AArch64/sme-za-tailcall-fp
diff -align.ll
+++ b/llvm/test/CodeGen/AArch64/sme-za-tailcall-fp
diff -align.ll
@@ -32,8 +32,6 @@ define void @caller_more_args(i64, i64, i64, i64, i64, i64, i64, i64, i64, i64)
 ; CHECK-NEXT:    mov w7, #8 ; =0x8
 ; CHECK-NEXT:    ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
 ; CHECK-NEXT:    .cfi_def_cfa_offset 0
-; CHECK-NEXT:    add sp, sp, #16
-; CHECK-NEXT:    .cfi_def_cfa_offset -16
 ; CHECK-NEXT:    .cfi_restore w30
 ; CHECK-NEXT:    .cfi_restore w29
 ; CHECK-NEXT:    b _callee_fewer_args
@@ -58,13 +56,13 @@ define void @caller_same_args(i64, i64, i64, i64, i64, i64, i64, i64, i64) "aarc
 ; CHECK-NEXT:    mov w8, #9 ; =0x9
 ; CHECK-NEXT:    mov w0, #1 ; =0x1
 ; CHECK-NEXT:    mov w1, #2 ; =0x2
-; CHECK-NEXT:    str x8, [sp, #16]
 ; CHECK-NEXT:    mov w2, #3 ; =0x3
 ; CHECK-NEXT:    mov w3, #4 ; =0x4
 ; CHECK-NEXT:    mov w4, #5 ; =0x5
 ; CHECK-NEXT:    mov w5, #6 ; =0x6
 ; CHECK-NEXT:    mov w6, #7 ; =0x7
 ; CHECK-NEXT:    mov w7, #8 ; =0x8
+; CHECK-NEXT:    str x8, [sp, #16]
 ; CHECK-NEXT:    ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
 ; CHECK-NEXT:    .cfi_def_cfa_offset 0
 ; CHECK-NEXT:    .cfi_restore w30
@@ -93,13 +91,13 @@ define void @caller_last_stack_arg_i64(i64 %1, i64 %2, i64 %3, i64 %4, i64 %5, i
 ; CHECK-NEXT:    mov w8, #1 ; =0x1
 ; CHECK-NEXT:    mov w0, #1 ; =0x1
 ; CHECK-NEXT:    mov w1, #2 ; =0x2
-; CHECK-NEXT:    strb w8, [sp, #16]
 ; CHECK-NEXT:    mov w2, #3 ; =0x3
 ; CHECK-NEXT:    mov w3, #4 ; =0x4
 ; CHECK-NEXT:    mov w4, #5 ; =0x5
 ; CHECK-NEXT:    mov w5, #6 ; =0x6
 ; CHECK-NEXT:    mov w6, #7 ; =0x7
 ; CHECK-NEXT:    mov w7, #8 ; =0x8
+; CHECK-NEXT:    strb w8, [sp, #16]
 ; CHECK-NEXT:    ldp x29, x30, [sp], #16 ; 16-byte Folded Reload
 ; CHECK-NEXT:    .cfi_def_cfa_offset 0
 ; CHECK-NEXT:    .cfi_restore w30


        


More information about the llvm-commits mailing list