[llvm] [PrologEpilogInserter] Scan all blocks for inline stack probe insertion (PR #195456)
Rong Mantle Bao via llvm-commits
llvm-commits at lists.llvm.org
Sat May 2 09:07:06 PDT 2026
https://github.com/CSharperMantle updated https://github.com/llvm/llvm-project/pull/195456
>From 1aa9bdcb98140e79dd759e5b25ecb5b2954541d3 Mon Sep 17 00:00:00 2001
From: Rong Bao <rong.bao at csmantle.top>
Date: Sat, 2 May 2026 22:45:53 +0800
Subject: [PATCH 1/2] [PrologEpilogInserter] Scan all blocks for inline stack
probe insertion
This prevents incomplete stack probe generation in non-entry blocks, for
example in this snippet targeting RISC-V:
target triple = "riscv64-unknown-linux-gnu"
define void @f(i64 %n) #0 {
entry:
%v = alloca i32, i64 %n, align 4
call void @g(ptr %v, [3000 x i64] poison)
ret void
}
declare void @g(ptr, [3000 x i64])
attributes #0 = { uwtable "frame-pointer"="none" "probe-stack"="inline-asm" }
---
llvm/lib/CodeGen/PrologEpilogInserter.cpp | 10 ++++++++--
1 file changed, 8 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/CodeGen/PrologEpilogInserter.cpp b/llvm/lib/CodeGen/PrologEpilogInserter.cpp
index 68fd54cf00146..4fc9807e5d214 100644
--- a/llvm/lib/CodeGen/PrologEpilogInserter.cpp
+++ b/llvm/lib/CodeGen/PrologEpilogInserter.cpp
@@ -1181,8 +1181,14 @@ void PEIImpl::insertPrologEpilogCode(MachineFunction &MF) {
// Zero call used registers before restoring callee-saved registers.
insertZeroCallUsedRegs(MF);
- for (MachineBasicBlock *SaveBlock : SaveBlocks)
- TFI.inlineStackProbe(MF, *SaveBlock);
+ {
+ SmallVector<MachineBasicBlock *> Blocks;
+ for (MachineBasicBlock &MBB : MF)
+ Blocks.push_back(&MBB);
+
+ for (MachineBasicBlock *MBB : Blocks)
+ TFI.inlineStackProbe(MF, *MBB);
+ }
// Emit additional code that is required to support segmented stacks, if
// we've been asked for it. This, when linked with a runtime with support
>From d8fa841963cdad64801ad2b0d07d71a3bb8a55ee Mon Sep 17 00:00:00 2001
From: Rong Bao <rong.bao at csmantle.top>
Date: Sat, 2 May 2026 23:29:04 +0800
Subject: [PATCH 2/2] [test][RISCV] Add test case for non-entry
PROBED_STACKALLOC pseudoinstruction
---
.../RISCV/stack-probing-dynamic-nonentry.ll | 115 ++++++++++++++++++
1 file changed, 115 insertions(+)
create mode 100644 llvm/test/CodeGen/RISCV/stack-probing-dynamic-nonentry.ll
diff --git a/llvm/test/CodeGen/RISCV/stack-probing-dynamic-nonentry.ll b/llvm/test/CodeGen/RISCV/stack-probing-dynamic-nonentry.ll
new file mode 100644
index 0000000000000..bad23c3ae67fa
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/stack-probing-dynamic-nonentry.ll
@@ -0,0 +1,115 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=riscv64 -mattr=+m -O2 < %s | FileCheck %s -check-prefix=RV64
+; RUN: llc -mtriple=riscv32 -mattr=+m -O2 < %s | FileCheck %s -check-prefix=RV32
+
+; Test that very large outgoing call frames in functions with variable-sized
+; objects get proper stack probing. The outgoing args are large enough to force
+; the PROBED_STACKALLOC path, which must be expanded in a non-entry block.
+
+define void @f(i64 %n) #0 {
+; RV64-LABEL: f:
+; RV64: # %bb.0: # %entry
+; RV64-NEXT: addi sp, sp, -16
+; RV64-NEXT: .cfi_def_cfa_offset 16
+; RV64-NEXT: sd zero, 0(sp)
+; RV64-NEXT: sd ra, 8(sp) # 8-byte Folded Spill
+; RV64-NEXT: sd s0, 0(sp) # 8-byte Folded Spill
+; RV64-NEXT: .cfi_offset ra, -8
+; RV64-NEXT: .cfi_offset s0, -16
+; RV64-NEXT: addi s0, sp, 16
+; RV64-NEXT: .cfi_def_cfa s0, 0
+; RV64-NEXT: slli a0, a0, 2
+; RV64-NEXT: addi a0, a0, 15
+; RV64-NEXT: andi a0, a0, -16
+; RV64-NEXT: sub a0, sp, a0
+; RV64-NEXT: lui a1, 1
+; RV64-NEXT: .LBB0_1: # %entry
+; RV64-NEXT: # =>This Inner Loop Header: Depth=1
+; RV64-NEXT: sub sp, sp, a1
+; RV64-NEXT: sd zero, 0(sp)
+; RV64-NEXT: bltu a0, sp, .LBB0_1
+; RV64-NEXT: # %bb.2: # %entry
+; RV64-NEXT: mv sp, a0
+; RV64-NEXT: lui a1, 5
+; RV64-NEXT: sub t1, sp, a1
+; RV64-NEXT: lui t2, 1
+; RV64-NEXT: .LBB0_3: # %entry
+; RV64-NEXT: # =>This Inner Loop Header: Depth=1
+; RV64-NEXT: sub sp, sp, t2
+; RV64-NEXT: sd zero, 0(sp)
+; RV64-NEXT: bne sp, t1, .LBB0_3
+; RV64-NEXT: # %bb.4: # %entry
+; RV64-NEXT: addi sp, sp, -2048
+; RV64-NEXT: addi sp, sp, -1424
+; RV64-NEXT: sd zero, 0(sp)
+; RV64-NEXT: call g
+; RV64-NEXT: lui a0, 6
+; RV64-NEXT: addi a0, a0, -624
+; RV64-NEXT: add sp, sp, a0
+; RV64-NEXT: addi sp, s0, -16
+; RV64-NEXT: .cfi_def_cfa sp, 16
+; RV64-NEXT: ld ra, 8(sp) # 8-byte Folded Reload
+; RV64-NEXT: ld s0, 0(sp) # 8-byte Folded Reload
+; RV64-NEXT: .cfi_restore ra
+; RV64-NEXT: .cfi_restore s0
+; RV64-NEXT: addi sp, sp, 16
+; RV64-NEXT: .cfi_def_cfa_offset 0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: f:
+; RV32: # %bb.0: # %entry
+; RV32-NEXT: addi sp, sp, -16
+; RV32-NEXT: .cfi_def_cfa_offset 16
+; RV32-NEXT: sw zero, 0(sp)
+; RV32-NEXT: sw ra, 12(sp) # 4-byte Folded Spill
+; RV32-NEXT: sw s0, 8(sp) # 4-byte Folded Spill
+; RV32-NEXT: .cfi_offset ra, -4
+; RV32-NEXT: .cfi_offset s0, -8
+; RV32-NEXT: addi s0, sp, 16
+; RV32-NEXT: .cfi_def_cfa s0, 0
+; RV32-NEXT: slli a0, a0, 2
+; RV32-NEXT: addi a0, a0, 15
+; RV32-NEXT: andi a0, a0, -16
+; RV32-NEXT: sub a0, sp, a0
+; RV32-NEXT: lui a1, 1
+; RV32-NEXT: .LBB0_1: # %entry
+; RV32-NEXT: # =>This Inner Loop Header: Depth=1
+; RV32-NEXT: sub sp, sp, a1
+; RV32-NEXT: sw zero, 0(sp)
+; RV32-NEXT: bltu a0, sp, .LBB0_1
+; RV32-NEXT: # %bb.2: # %entry
+; RV32-NEXT: mv sp, a0
+; RV32-NEXT: lui a1, 5
+; RV32-NEXT: sub t1, sp, a1
+; RV32-NEXT: lui t2, 1
+; RV32-NEXT: .LBB0_3: # %entry
+; RV32-NEXT: # =>This Inner Loop Header: Depth=1
+; RV32-NEXT: sub sp, sp, t2
+; RV32-NEXT: sw zero, 0(sp)
+; RV32-NEXT: bne sp, t1, .LBB0_3
+; RV32-NEXT: # %bb.4: # %entry
+; RV32-NEXT: addi sp, sp, -2048
+; RV32-NEXT: addi sp, sp, -1456
+; RV32-NEXT: sw zero, 0(sp)
+; RV32-NEXT: call g
+; RV32-NEXT: lui a0, 6
+; RV32-NEXT: addi a0, a0, -592
+; RV32-NEXT: add sp, sp, a0
+; RV32-NEXT: addi sp, s0, -16
+; RV32-NEXT: .cfi_def_cfa sp, 16
+; RV32-NEXT: lw ra, 12(sp) # 4-byte Folded Reload
+; RV32-NEXT: lw s0, 8(sp) # 4-byte Folded Reload
+; RV32-NEXT: .cfi_restore ra
+; RV32-NEXT: .cfi_restore s0
+; RV32-NEXT: addi sp, sp, 16
+; RV32-NEXT: .cfi_def_cfa_offset 0
+; RV32-NEXT: ret
+entry:
+ %v = alloca i32, i64 %n
+ call void @g(ptr %v, [3000 x i64] poison)
+ ret void
+}
+
+declare void @g(ptr, [3000 x i64])
+
+attributes #0 = { "probe-stack"="inline-asm" }
More information about the llvm-commits
mailing list