[llvm] ca24243 - [AMDGPU][GlobalISel] Reject guaranteed tail calls in call lowering (#216260)

via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 13 06:57:52 PDT 2026


Author: Arseniy Obolenskiy
Date: 2026-09-13T15:57:48+02:00
New Revision: ca2424325cd9247ac0c11cd2d1c8ea3fb07ab16e

URL: https://github.com/llvm/llvm-project/commit/ca2424325cd9247ac0c11cd2d1c8ea3fb07ab16e
DIFF: https://github.com/llvm/llvm-project/commit/ca2424325cd9247ac0c11cd2d1c8ea3fb07ab16e.diff

LOG: [AMDGPU][GlobalISel] Reject guaranteed tail calls in call lowering (#216260)

The FPDiff immediate was being written to the wrong operand, corrupting
the callee GlobalAddress

Reject required tail calls like SelectionDAG does, and fix the stale
index for the one path that still needs it (cs.chain lowering)

Added: 
    

Modified: 
    llvm/lib/Target/AMDGPU/AMDGPUCallLowering.cpp
    llvm/test/CodeGen/AMDGPU/unsupported-calls.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/AMDGPU/AMDGPUCallLowering.cpp b/llvm/lib/Target/AMDGPU/AMDGPUCallLowering.cpp
index 3c78bb1dd7be3..2a78237cf7e20 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUCallLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUCallLowering.cpp
@@ -1484,7 +1484,7 @@ bool AMDGPUCallLowering::lowerTailCall(
   // If we have -tailcallopt, we need to adjust the stack. We'll do the call
   // sequence start and end here.
   if (!IsSibCall) {
-    MIB->getOperand(CalleeIdx + 1).setImm(FPDiff);
+    MIB->getOperand(CalleeIdx + 2).setImm(FPDiff);
     CallSeqStart.addImm(NumBytes).addImm(0);
     // End the call sequence *before* emitting the call. Normally, we would
     // tidy the frame up after the call. However, here, we've laid out the
@@ -1613,6 +1613,17 @@ bool AMDGPUCallLowering::lowerCall(MachineIRBuilder &MIRBuilder,
   if (Info.CanLowerReturn && !Info.OrigRet.Ty->isVoidTy())
     splitToValueTypes(Info.OrigRet, InArgs, DL, Info.CallConv);
 
+  if (Info.IsTailCall && MF.getTarget().Options.GuaranteedTailCallOpt) {
+    StringRef CalleeName = Info.Callee.isGlobal()
+                               ? Info.Callee.getGlobal()->getName()
+                               : "<unknown>";
+    F.getContext().diagnose(DiagnosticInfoUnsupported(
+        F, "unsupported required tail call to function " + CalleeName));
+    for (Register ResReg : Info.OrigRet.Regs)
+      MIRBuilder.buildUndef(ResReg);
+    return true;
+  }
+
   // If we can lower as a tail call, do that instead.
   bool CanTailCallOpt =
       isEligibleForTailCallOptimization(MIRBuilder, Info, InArgs, OutArgs);

diff  --git a/llvm/test/CodeGen/AMDGPU/unsupported-calls.ll b/llvm/test/CodeGen/AMDGPU/unsupported-calls.ll
index 61140d383a151..c11aef236c485 100644
--- a/llvm/test/CodeGen/AMDGPU/unsupported-calls.ll
+++ b/llvm/test/CodeGen/AMDGPU/unsupported-calls.ll
@@ -1,4 +1,5 @@
 ; RUN: not llc -mtriple=amdgpu6.00-mesa-mesa3d -tailcallopt < %s 2>&1 | FileCheck --check-prefix=GCN %s
+; RUN: not llc -global-isel -mtriple=amdgpu6.00-mesa-mesa3d -tailcallopt < %s 2>&1 | FileCheck --check-prefix=GCN %s
 ; RUN: not llc -mtriple=amdgpu6.00--amdpal -tailcallopt < %s 2>&1 | FileCheck --check-prefix=GCN %s
 ; RUN: not llc -mtriple=r600-- -mcpu=cypress -tailcallopt < %s 2>&1 | FileCheck -check-prefix=R600 %s
 
@@ -43,6 +44,21 @@ define i32 @test_tail_call(ptr addrspace(1) %out, ptr addrspace(1) %in) {
   ret i32 %c
 }
 
+define fastcc i32 @defined_fastcc_function(i32 %x) nounwind noinline {
+  %y = add i32 %x, 8
+  ret i32 %y
+}
+
+; GCN: error: <unknown>:0:0: in function test_tail_call_fastcc i32 (ptr addrspace(1), ptr addrspace(1)): unsupported required tail call to function defined_fastcc_function
+; R600: in function test_tail_call_fastcc{{.*}}: unsupported call to function defined_fastcc_function
+define fastcc i32 @test_tail_call_fastcc(ptr addrspace(1) %out, ptr addrspace(1) %in) {
+  %b_ptr = getelementptr i32, ptr addrspace(1) %in, i32 1
+  %a = load i32, ptr addrspace(1) %in
+  %b = load i32, ptr addrspace(1) %b_ptr
+  %c = tail call fastcc i32 @defined_fastcc_function(i32 %b)
+  ret i32 %c
+}
+
 ; R600: in function test_c_call{{.*}}: unsupported call to function defined_function
 define amdgpu_ps i32 @test_c_call_from_shader() {
   %call = call i32 @defined_function(i32 0)


        


More information about the llvm-commits mailing list