[llvm] [SystemZ] Add shrink-wrapping support for ELF prologue/epilogue (PR #225240)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 28 06:03:40 PDT 2026
https://github.com/MarkVeerasingam updated https://github.com/llvm/llvm-project/pull/225240
>From 161489ed117ac71c4b84570239a0e12da89c1d9d Mon Sep 17 00:00:00 2001
From: MarkVeerasingam <markveer70 at gmail.com>
Date: Sun, 20 Sep 2026 14:51:32 -0500
Subject: [PATCH 1/5] Added a shrinkwrap test for SystemZ codegen
---
llvm/test/CodeGen/SystemZ/shrinkwrap.ll | 22 ++++++++++++++++++++++
1 file changed, 22 insertions(+)
create mode 100644 llvm/test/CodeGen/SystemZ/shrinkwrap.ll
diff --git a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
new file mode 100644
index 0000000000000..10bba42ddd862
--- /dev/null
+++ b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
@@ -0,0 +1,22 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+;
+; RUN: llc -mtriple=s390x-linux-gnu -O2 < %s | FileCheck %s -check-prefix=Z
+; RUN: llc -mtriple=s390x-linux-gnu -O2 -enable-shrink-wrap < %s | FileCheck %s -check-prefix=SW
+
+declare void @notdead(ptr)
+
+define void @conditional_alloca(i64 %n) nounwind {
+; Z-LABEL: conditional_alloca:
+;
+; SW-LABEL: conditional_alloca:
+ %cmp = icmp eq i64 %n, 0
+ br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+ %addr = alloca i8, i64 %n, align 16
+ call void @notdead(ptr %addr)
+ br label %if.end
+
+if.end:
+ ret void
+}
>From e36cce6e6ae08ddcc70f1b5292f8123461f5e2cc Mon Sep 17 00:00:00 2001
From: MarkVeerasingam <markveer70 at gmail.com>
Date: Mon, 21 Sep 2026 14:30:40 -0500
Subject: [PATCH 2/5] [SystemZ] Make ELF function prologue/epilogue shrink-wrap
capable
---
.../Target/SystemZ/SystemZFrameLowering.cpp | 24 ++++++++----
llvm/test/CodeGen/SystemZ/shrinkwrap.ll | 37 ++++++++++++++++++-
2 files changed, 52 insertions(+), 9 deletions(-)
diff --git a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
index 14688a53c1c63..de4c0d2331442 100644
--- a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
@@ -530,7 +530,6 @@ static void buildDefCFAReg(MachineBasicBlock &MBB,
void SystemZELFFrameLowering::emitPrologue(MachineFunction &MF,
MachineBasicBlock &MBB) const {
- assert(&MF.front() == &MBB && "Shrink-wrapping not yet supported");
const SystemZSubtarget &STI = MF.getSubtarget<SystemZSubtarget>();
const SystemZTargetLowering &TLI = *STI.getTargetLowering();
MachineFrameInfo &MFFrame = MF.getFrameInfo();
@@ -720,19 +719,32 @@ void SystemZELFFrameLowering::emitPrologue(MachineFunction &MF,
void SystemZELFFrameLowering::emitEpilogue(MachineFunction &MF,
MachineBasicBlock &MBB) const {
- MachineBasicBlock::iterator MBBI = MBB.getLastNonDebugInstr();
auto *ZII =
static_cast<const SystemZInstrInfo *>(MF.getSubtarget().getInstrInfo());
SystemZMachineFunctionInfo *ZFI = MF.getInfo<SystemZMachineFunctionInfo>();
MachineFrameInfo &MFFrame = MF.getFrameInfo();
+ // A shrink-wrapped epilogue can be inserted into a block without a return
+ // instruction. Insert before the first terminator, or after the
+ // non-debug instruction when the block has no terminator.
+ MachineBasicBlock::iterator MBBI = MBB.end();
+ DebugLoc DL;
+ if (!MBB.empty()) {
+ MBBI = MBB.getFirstTerminator();
+ if (MBBI == MBB.end())
+ MBBI = MBB.getLastNonDebugInstr();
+
+ if (MBBI != MBB.end()) {
+ DL = MBBI->getDebugLoc();
+ if (!MBBI->isTerminator())
+ MBBI = std::next(MBBI);
+ }
+ }
+
// See SystemZELFFrameLowering::emitPrologue
if (MF.getFunction().getCallingConv() == CallingConv::GHC)
return;
- // Skip the return instruction.
- assert(MBBI->isReturn() && "Can only insert epilogue into returning blocks");
-
uint64_t StackSize = MFFrame.getStackSize();
if (ZFI->getRestoreGPRRegs().LowGPR) {
--MBBI;
@@ -741,7 +753,6 @@ void SystemZELFFrameLowering::emitEpilogue(MachineFunction &MF,
llvm_unreachable("Expected to see callee-save register restore code");
unsigned AddrOpNo = 2;
- DebugLoc DL = MBBI->getDebugLoc();
uint64_t Offset = StackSize + MBBI->getOperand(AddrOpNo + 1).getImm();
unsigned NewOpcode = ZII->getOpcodeForOffset(Opcode, Offset);
@@ -759,7 +770,6 @@ void SystemZELFFrameLowering::emitEpilogue(MachineFunction &MF,
MBBI->setDesc(ZII->get(NewOpcode));
MBBI->getOperand(AddrOpNo + 1).ChangeToImmediate(Offset);
} else if (StackSize) {
- DebugLoc DL = MBBI->getDebugLoc();
emitIncrement(MBB, MBBI, DL, SystemZ::R15D, StackSize, ZII);
}
}
diff --git a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
index 10bba42ddd862..d1660535d77b0 100644
--- a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
+++ b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
@@ -1,14 +1,47 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
;
-; RUN: llc -mtriple=s390x-linux-gnu -O2 < %s | FileCheck %s -check-prefix=Z
-; RUN: llc -mtriple=s390x-linux-gnu -O2 -enable-shrink-wrap < %s | FileCheck %s -check-prefix=SW
+; RUN: llc -mtriple=s390x-linux-gnu -O2 -verify-machineinstrs < %s | FileCheck %s -check-prefix=Z
+; RUN: llc -mtriple=s390x-linux-gnu -O2 -enable-shrink-wrap -verify-machineinstrs < %s | FileCheck %s -check-prefix=SW
declare void @notdead(ptr)
define void @conditional_alloca(i64 %n) nounwind {
; Z-LABEL: conditional_alloca:
+; Z: # %bb.0:
+; Z-NEXT: stmg %r11, %r15, 88(%r15)
+; Z-NEXT: aghi %r15, -160
+; Z-NEXT: lgr %r11, %r15
+; Z-NEXT: cgije %r2, 0, .LBB0_2
+; Z-NEXT: # %bb.1: # %if.then
+; Z-NEXT: lgr %r1, %r15
+; Z-NEXT: la %r0, 7(%r2)
+; Z-NEXT: nill %r0, 65528
+; Z-NEXT: sgr %r1, %r0
+; Z-NEXT: la %r2, 160(%r1)
+; Z-NEXT: nill %r2, 65520
+; Z-NEXT: lay %r15, -8(%r1)
+; Z-NEXT: brasl %r14, notdead at PLT
+; Z-NEXT: .LBB0_2: # %if.end
+; Z-NEXT: lmg %r11, %r15, 248(%r11)
+; Z-NEXT: br %r14
;
; SW-LABEL: conditional_alloca:
+; SW: # %bb.0:
+; SW-NEXT: cgibe %r2, 0, 0(%r14)
+; SW-NEXT: .LBB0_1: # %if.then
+; SW-NEXT: stmg %r11, %r15, 88(%r15)
+; SW-NEXT: aghi %r15, -160
+; SW-NEXT: lgr %r11, %r15
+; SW-NEXT: lgr %r1, %r15
+; SW-NEXT: la %r0, 7(%r2)
+; SW-NEXT: nill %r0, 65528
+; SW-NEXT: sgr %r1, %r0
+; SW-NEXT: la %r2, 160(%r1)
+; SW-NEXT: nill %r2, 65520
+; SW-NEXT: lay %r15, -8(%r1)
+; SW-NEXT: brasl %r14, notdead at PLT
+; SW-NEXT: lmg %r11, %r15, 248(%r11)
+; SW-NEXT: br %r14
%cmp = icmp eq i64 %n, 0
br i1 %cmp, label %if.end, label %if.then
>From d7f2a146c9a0013e5aaa2d261f5b5475a93c8671 Mon Sep 17 00:00:00 2001
From: MarkVeerasingam <markveer70 at gmail.com>
Date: Mon, 21 Sep 2026 17:41:35 -0500
Subject: [PATCH 3/5] [SystemZ] Updated shrinkwrap test to cover vararg
clobbering instructions
---
llvm/test/CodeGen/SystemZ/shrinkwrap.ll | 64 +++++++++++++++++++++++++
1 file changed, 64 insertions(+)
diff --git a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
index d1660535d77b0..0209d94f777d1 100644
--- a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
+++ b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
@@ -6,6 +6,7 @@
declare void @notdead(ptr)
define void @conditional_alloca(i64 %n) nounwind {
+; existing generated checks...
; Z-LABEL: conditional_alloca:
; Z: # %bb.0:
; Z-NEXT: stmg %r11, %r15, 88(%r15)
@@ -53,3 +54,66 @@ if.then:
if.end:
ret void
}
+
+; Varargs prologue must remain in the entry block because it saves incoming
+; GPR argument registers before they can be clobbered.
+declare void @llvm.va_start.p0(ptr)
+declare void @llvm.va_end.p0(ptr)
+declare void @take(ptr)
+
+define void @va_fpr(double %a, double %b, double %c, double %d,
+; Z-LABEL: va_fpr:
+; Z: # %bb.0:
+; Z-NEXT: stmg %r3, %r15, 24(%r15)
+; Z-NEXT: aghi %r15, -168
+; Z-NEXT: #APP
+; Z-NEXT: lghi %r3, 0
+; Z-NEXT: #NO_APP
+; Z-NEXT: cije %r2, 0, .LBB1_2
+; Z-NEXT: # %bb.1: # %call
+; Z-NEXT: la %r0, 168(%r15)
+; Z-NEXT: stg %r0, 184(%r15)
+; Z-NEXT: la %r0, 328(%r15)
+; Z-NEXT: stg %r0, 176(%r15)
+; Z-NEXT: mvghi 168(%r15), 4
+; Z-NEXT: la %r2, 160(%r15)
+; Z-NEXT: mvghi 160(%r15), 1
+; Z-NEXT: brasl %r14, take at PLT
+; Z-NEXT: .LBB1_2: # %ret
+; Z-NEXT: lmg %r6, %r15, 216(%r15)
+; Z-NEXT: br %r14
+;
+; SW-LABEL: va_fpr:
+; SW: # %bb.0:
+; SW-NEXT: #APP
+; SW-NEXT: lghi %r3, 0
+; SW-NEXT: #NO_APP
+; SW-NEXT: cibe %r2, 0, 0(%r14)
+; SW-NEXT: .LBB1_1: # %call
+; SW-NEXT: stmg %r3, %r15, 24(%r15)
+; SW-NEXT: aghi %r15, -168
+; SW-NEXT: la %r0, 168(%r15)
+; SW-NEXT: stg %r0, 184(%r15)
+; SW-NEXT: la %r0, 328(%r15)
+; SW-NEXT: stg %r0, 176(%r15)
+; SW-NEXT: mvghi 168(%r15), 4
+; SW-NEXT: la %r2, 160(%r15)
+; SW-NEXT: mvghi 160(%r15), 1
+; SW-NEXT: brasl %r14, take at PLT
+; SW-NEXT: lmg %r6, %r15, 216(%r15)
+; SW-NEXT: br %r14
+ i32 %n, ...) nounwind {
+ call void asm sideeffect "lghi %r3, 0", "~{r3}"()
+ %ap = alloca ptr
+ %z = icmp eq i32 %n, 0
+ br i1 %z, label %ret, label %call
+
+call:
+ call void @llvm.va_start.p0(ptr %ap)
+ call void @take(ptr %ap)
+ call void @llvm.va_end.p0(ptr %ap)
+ br label %ret
+
+ret:
+ ret void
+}
>From 2288306b46c9a271ae435a7cbd6ac1b720b13d93 Mon Sep 17 00:00:00 2001
From: MarkVeerasingam <markveer70 at gmail.com>
Date: Mon, 21 Sep 2026 18:07:34 -0500
Subject: [PATCH 4/5] [SystemZ] Keep varargs prologue in entry block
The varargs prologue saves incoming GPR argument registers, so it must
remain in the entry block when shrink-wrapping is enabled. Otherwise an
instruction before the shrink-wrapped prologue can clobber an incoming
argument register before it is saved.
Add a SystemZ-specific canUseAsPrologue restriction for varargs functions
and update the shrink-wrap test to cover the case.
---
llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp | 11 +++++++++++
llvm/lib/Target/SystemZ/SystemZFrameLowering.h | 2 ++
llvm/test/CodeGen/SystemZ/shrinkwrap.ll | 9 +++++----
3 files changed, 18 insertions(+), 4 deletions(-)
diff --git a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
index de4c0d2331442..ee7a66c6ee6a0 100644
--- a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
@@ -1555,6 +1555,17 @@ void SystemZXPLINKFrameLowering::processFunctionBeforeFrameFinalized(
}
}
+bool SystemZELFFrameLowering::canUseAsPrologue(
+ const MachineBasicBlock &MBB) const {
+ const MachineFunction &MF = *MBB.getParent();
+
+ // Keep the varargs prologue in the entry block so incoming GPRs are saved
+ // before they can be clobbered.
+ if (MF.getFunction().isVarArg())
+ return &MBB == &MF.front();
+ return true;
+}
+
// Determines the size of the frame, and creates the deferred spill objects.
void SystemZXPLINKFrameLowering::determineFrameLayout(
MachineFunction &MF) const {
diff --git a/llvm/lib/Target/SystemZ/SystemZFrameLowering.h b/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
index c67139aac203f..c3941103d5b9f 100644
--- a/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
+++ b/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
@@ -118,6 +118,8 @@ class SystemZELFFrameLowering : public SystemZFrameLowering {
// Get or create the frame index of where the old frame pointer is stored.
int getOrCreateFramePointerSaveIndex(MachineFunction &MF) const override;
+ bool canUseAsPrologue(const MachineBasicBlock &MBB) const override;
+
protected:
bool hasFPImpl(const MachineFunction &MF) const override;
};
diff --git a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
index 0209d94f777d1..687837921122c 100644
--- a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
+++ b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
@@ -85,13 +85,13 @@ define void @va_fpr(double %a, double %b, double %c, double %d,
;
; SW-LABEL: va_fpr:
; SW: # %bb.0:
+; SW-NEXT: stmg %r3, %r15, 24(%r15)
+; SW-NEXT: aghi %r15, -168
; SW-NEXT: #APP
; SW-NEXT: lghi %r3, 0
; SW-NEXT: #NO_APP
-; SW-NEXT: cibe %r2, 0, 0(%r14)
-; SW-NEXT: .LBB1_1: # %call
-; SW-NEXT: stmg %r3, %r15, 24(%r15)
-; SW-NEXT: aghi %r15, -168
+; SW-NEXT: cije %r2, 0, .LBB1_2
+; SW-NEXT: # %bb.1: # %call
; SW-NEXT: la %r0, 168(%r15)
; SW-NEXT: stg %r0, 184(%r15)
; SW-NEXT: la %r0, 328(%r15)
@@ -100,6 +100,7 @@ define void @va_fpr(double %a, double %b, double %c, double %d,
; SW-NEXT: la %r2, 160(%r15)
; SW-NEXT: mvghi 160(%r15), 1
; SW-NEXT: brasl %r14, take at PLT
+; SW-NEXT: .LBB1_2: # %ret
; SW-NEXT: lmg %r6, %r15, 216(%r15)
; SW-NEXT: br %r14
i32 %n, ...) nounwind {
>From ee97ee47432d265743374008107770dffa5d2d98 Mon Sep 17 00:00:00 2001
From: MarkVeerasingam <markveer70 at gmail.com>
Date: Mon, 28 Sep 2026 08:02:08 -0500
Subject: [PATCH 5/5] [SystemZ] Enable shrink-wrapping for SystemZ ELF
Enable shrink-wrapping for SystemZ ELF, with a few guards for cases where
the prologue must remain in the entry block.
Disable shrink-wrapping for GHC functions, mcount instrumentation, stack
backchains, and inline stack probing.
Also avoid updating Frame Pointer liveness when the prologue is emitted in
a non-entry block, and reject candidate blocks where SystemZ::CC is live-in
since the prologue may clobber it.
Add regression tests covering these cases.
---
.../Target/SystemZ/SystemZFrameLowering.cpp | 38 ++-
.../lib/Target/SystemZ/SystemZFrameLowering.h | 2 +
llvm/test/CodeGen/SystemZ/int-cmp-02.ll | 143 +++++++----
llvm/test/CodeGen/SystemZ/shrinkwrap.ll | 237 +++++++++++++++++-
4 files changed, 363 insertions(+), 57 deletions(-)
diff --git a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
index ee7a66c6ee6a0..1f14d4b4295c0 100644
--- a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
@@ -674,8 +674,10 @@ void SystemZELFFrameLowering::emitPrologue(MachineFunction &MF,
// Mark the FramePtr as live at the beginning of every block except
// the entry block. (We'll have marked R11 as live on entry when
// saving the GPRs.)
- for (MachineBasicBlock &MBBJ : llvm::drop_begin(MF))
- MBBJ.addLiveIn(SystemZ::R11D);
+ if (&MBB == &MF.front()) {
+ for (MachineBasicBlock &MBBJ : llvm::drop_begin(MF))
+ MBBJ.addLiveIn(SystemZ::R11D);
+ }
}
// Skip over the FPR/VR saves.
@@ -1555,6 +1557,32 @@ void SystemZXPLINKFrameLowering::processFunctionBeforeFrameFinalized(
}
}
+bool SystemZELFFrameLowering::enableShrinkWrapping(
+ const MachineFunction &MF) const {
+ const Function &F = MF.getFunction();
+ const SystemZSubtarget &Subtarget = MF.getSubtarget<SystemZSubtarget>();
+
+ // GHC calling convention does not use standard prologue/epilogue.
+ if (F.getCallingConv() == CallingConv::GHC)
+ return false;
+
+ // mcount instrumentation must be called at the function entry.
+ if (F.hasFnAttribute("systemz-instrument-function-entry"))
+ return false;
+
+ // Backchain setup currently assumes %r1 is free at the entry block.
+ // TODO: Investigate if we need to generalise the temp reg handling.
+ // Right now we just disable shrink-wrapping if Backchain is enabled
+ if (Subtarget.hasBackChain())
+ return false;
+
+ // stack probing uses %r0/%r1 scratch registers.
+ if (Subtarget.getTargetLowering()->hasInlineStackProbe(MF))
+ return false;
+
+ return true;
+}
+
bool SystemZELFFrameLowering::canUseAsPrologue(
const MachineBasicBlock &MBB) const {
const MachineFunction &MF = *MBB.getParent();
@@ -1563,6 +1591,12 @@ bool SystemZELFFrameLowering::canUseAsPrologue(
// before they can be clobbered.
if (MF.getFunction().isVarArg())
return &MBB == &MF.front();
+
+ // If CC is live into MBB, prologue instructions (e.g. AGHI/AGFI for stack
+ // adjustments) will clobber CC.
+ if (MBB.isLiveIn(SystemZ::CC))
+ return false;
+
return true;
}
diff --git a/llvm/lib/Target/SystemZ/SystemZFrameLowering.h b/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
index c3941103d5b9f..b33bf577b98a5 100644
--- a/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
+++ b/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
@@ -118,6 +118,8 @@ class SystemZELFFrameLowering : public SystemZFrameLowering {
// Get or create the frame index of where the old frame pointer is stored.
int getOrCreateFramePointerSaveIndex(MachineFunction &MF) const override;
+ bool enableShrinkWrapping(const MachineFunction &MF) const override;
+
bool canUseAsPrologue(const MachineBasicBlock &MBB) const override;
protected:
diff --git a/llvm/test/CodeGen/SystemZ/int-cmp-02.ll b/llvm/test/CodeGen/SystemZ/int-cmp-02.ll
index 3fd2f24d353a7..b5ce7aa4cf211 100644
--- a/llvm/test/CodeGen/SystemZ/int-cmp-02.ll
+++ b/llvm/test/CodeGen/SystemZ/int-cmp-02.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
; Test 32-bit signed comparison in which the second operand is a variable.
;
; RUN: llc < %s -mtriple=s390x-linux-gnu | FileCheck %s
@@ -7,9 +8,11 @@ declare i32 @foo()
; Check register comparison.
define double @f1(double %a, double %b, i32 %i1, i32 %i2) {
; CHECK-LABEL: f1:
-; CHECK: crbl %r2, %r3, 0(%r14)
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: crbl %r2, %r3, 0(%r14)
+; CHECK-NEXT: .LBB0_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%cond = icmp slt i32 %i1, %i2
%res = select i1 %cond, double %a, double %b
ret double %res
@@ -18,10 +21,12 @@ define double @f1(double %a, double %b, i32 %i1, i32 %i2) {
; Check the low end of the C range.
define double @f2(double %a, double %b, i32 %i1, ptr %ptr) {
; CHECK-LABEL: f2:
-; CHECK: c %r2, 0(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: c %r2, 0(%r3)
+; CHECK-NEXT: blr %r14
+; CHECK-NEXT: .LBB1_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%i2 = load i32, ptr %ptr
%cond = icmp slt i32 %i1, %i2
%res = select i1 %cond, double %a, double %b
@@ -31,10 +36,12 @@ define double @f2(double %a, double %b, i32 %i1, ptr %ptr) {
; Check the high end of the aligned C range.
define double @f3(double %a, double %b, i32 %i1, ptr %base) {
; CHECK-LABEL: f3:
-; CHECK: c %r2, 4092(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: c %r2, 4092(%r3)
+; CHECK-NEXT: blr %r14
+; CHECK-NEXT: .LBB2_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%ptr = getelementptr i32, ptr %base, i64 1023
%i2 = load i32, ptr %ptr
%cond = icmp slt i32 %i1, %i2
@@ -45,10 +52,12 @@ define double @f3(double %a, double %b, i32 %i1, ptr %base) {
; Check the next word up, which should use CY instead of C.
define double @f4(double %a, double %b, i32 %i1, ptr %base) {
; CHECK-LABEL: f4:
-; CHECK: cy %r2, 4096(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: cy %r2, 4096(%r3)
+; CHECK-NEXT: blr %r14
+; CHECK-NEXT: .LBB3_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%ptr = getelementptr i32, ptr %base, i64 1024
%i2 = load i32, ptr %ptr
%cond = icmp slt i32 %i1, %i2
@@ -59,10 +68,12 @@ define double @f4(double %a, double %b, i32 %i1, ptr %base) {
; Check the high end of the aligned CY range.
define double @f5(double %a, double %b, i32 %i1, ptr %base) {
; CHECK-LABEL: f5:
-; CHECK: cy %r2, 524284(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: cy %r2, 524284(%r3)
+; CHECK-NEXT: blr %r14
+; CHECK-NEXT: .LBB4_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%ptr = getelementptr i32, ptr %base, i64 131071
%i2 = load i32, ptr %ptr
%cond = icmp slt i32 %i1, %i2
@@ -74,11 +85,13 @@ define double @f5(double %a, double %b, i32 %i1, ptr %base) {
; Other sequences besides this one would be OK.
define double @f6(double %a, double %b, i32 %i1, ptr %base) {
; CHECK-LABEL: f6:
-; CHECK: agfi %r3, 524288
-; CHECK: c %r2, 0(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: agfi %r3, 524288
+; CHECK-NEXT: c %r2, 0(%r3)
+; CHECK-NEXT: blr %r14
+; CHECK-NEXT: .LBB5_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%ptr = getelementptr i32, ptr %base, i64 131072
%i2 = load i32, ptr %ptr
%cond = icmp slt i32 %i1, %i2
@@ -89,10 +102,12 @@ define double @f6(double %a, double %b, i32 %i1, ptr %base) {
; Check the high end of the negative aligned CY range.
define double @f7(double %a, double %b, i32 %i1, ptr %base) {
; CHECK-LABEL: f7:
-; CHECK: cy %r2, -4(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: cy %r2, -4(%r3)
+; CHECK-NEXT: blr %r14
+; CHECK-NEXT: .LBB6_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%ptr = getelementptr i32, ptr %base, i64 -1
%i2 = load i32, ptr %ptr
%cond = icmp slt i32 %i1, %i2
@@ -103,10 +118,12 @@ define double @f7(double %a, double %b, i32 %i1, ptr %base) {
; Check the low end of the CY range.
define double @f8(double %a, double %b, i32 %i1, ptr %base) {
; CHECK-LABEL: f8:
-; CHECK: cy %r2, -524288(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: cy %r2, -524288(%r3)
+; CHECK-NEXT: blr %r14
+; CHECK-NEXT: .LBB7_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%ptr = getelementptr i32, ptr %base, i64 -131072
%i2 = load i32, ptr %ptr
%cond = icmp slt i32 %i1, %i2
@@ -118,11 +135,13 @@ define double @f8(double %a, double %b, i32 %i1, ptr %base) {
; Other sequences besides this one would be OK.
define double @f9(double %a, double %b, i32 %i1, ptr %base) {
; CHECK-LABEL: f9:
-; CHECK: agfi %r3, -524292
-; CHECK: c %r2, 0(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: agfi %r3, -524292
+; CHECK-NEXT: c %r2, 0(%r3)
+; CHECK-NEXT: blr %r14
+; CHECK-NEXT: .LBB8_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%ptr = getelementptr i32, ptr %base, i64 -131073
%i2 = load i32, ptr %ptr
%cond = icmp slt i32 %i1, %i2
@@ -133,10 +152,12 @@ define double @f9(double %a, double %b, i32 %i1, ptr %base) {
; Check that C allows an index.
define double @f10(double %a, double %b, i32 %i1, i64 %base, i64 %index) {
; CHECK-LABEL: f10:
-; CHECK: c %r2, 4092({{%r4,%r3|%r3,%r4}})
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: c %r2, 4092(%r4,%r3)
+; CHECK-NEXT: blr %r14
+; CHECK-NEXT: .LBB9_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%add1 = add i64 %base, %index
%add2 = add i64 %add1, 4092
%ptr = inttoptr i64 %add2 to ptr
@@ -149,10 +170,12 @@ define double @f10(double %a, double %b, i32 %i1, i64 %base, i64 %index) {
; Check that CY allows an index.
define double @f11(double %a, double %b, i32 %i1, i64 %base, i64 %index) {
; CHECK-LABEL: f11:
-; CHECK: cy %r2, 4096({{%r4,%r3|%r3,%r4}})
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: cy %r2, 4096(%r4,%r3)
+; CHECK-NEXT: blr %r14
+; CHECK-NEXT: .LBB10_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%add1 = add i64 %base, %index
%add2 = add i64 %add1, 4096
%ptr = inttoptr i64 %add2 to ptr
@@ -166,9 +189,23 @@ define double @f11(double %a, double %b, i32 %i1, i64 %base, i64 %index) {
; critical edge %entry->%while.body, which lost the kills information for CC.
define void @f12(i32 %a, i32 %b) {
; CHECK-LABEL: f12:
-; CHECK: cije %r2, 0
-; CHECK: crjlh %r2,
-; CHECK: br %r14
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: cibe %r2, 0, 0(%r14)
+; CHECK-NEXT: .LBB11_1: # %while.body.preheader
+; CHECK-NEXT: stmg %r13, %r15, 104(%r15)
+; CHECK-NEXT: .cfi_offset %r13, -56
+; CHECK-NEXT: .cfi_offset %r14, -48
+; CHECK-NEXT: .cfi_offset %r15, -40
+; CHECK-NEXT: aghi %r15, -160
+; CHECK-NEXT: .cfi_def_cfa_offset 320
+; CHECK-NEXT: lr %r13, %r3
+; CHECK-NEXT: .LBB11_2: # %while.body
+; CHECK-NEXT: # =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: brasl %r14, foo at PLT
+; CHECK-NEXT: crjlh %r2, %r13, .LBB11_2
+; CHECK-NEXT: # %bb.3:
+; CHECK-NEXT: lmg %r13, %r15, 264(%r15)
+; CHECK-NEXT: br %r14
entry:
%cmp11 = icmp eq i32 %a, 0
br i1 %cmp11, label %while.end, label %while.body
@@ -185,10 +222,12 @@ while.end:
; Check the comparison can be reversed if that allows C to be used.
define double @f13(double %a, double %b, i32 %i2, ptr %ptr) {
; CHECK-LABEL: f13:
-; CHECK: c %r2, 0(%r3)
-; CHECK-NEXT: bhr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK: # %bb.0:
+; CHECK-NEXT: c %r2, 0(%r3)
+; CHECK-NEXT: bhr %r14
+; CHECK-NEXT: .LBB12_1:
+; CHECK-NEXT: ldr %f0, %f2
+; CHECK-NEXT: br %r14
%i1 = load i32, ptr %ptr
%cond = icmp slt i32 %i1, %i2
%res = select i1 %cond, double %a, double %b
diff --git a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
index 687837921122c..efe8586173def 100644
--- a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
+++ b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
@@ -9,11 +9,11 @@ define void @conditional_alloca(i64 %n) nounwind {
; existing generated checks...
; Z-LABEL: conditional_alloca:
; Z: # %bb.0:
+; Z-NEXT: cgibe %r2, 0, 0(%r14)
+; Z-NEXT: .LBB0_1: # %if.then
; Z-NEXT: stmg %r11, %r15, 88(%r15)
; Z-NEXT: aghi %r15, -160
; Z-NEXT: lgr %r11, %r15
-; Z-NEXT: cgije %r2, 0, .LBB0_2
-; Z-NEXT: # %bb.1: # %if.then
; Z-NEXT: lgr %r1, %r15
; Z-NEXT: la %r0, 7(%r2)
; Z-NEXT: nill %r0, 65528
@@ -22,7 +22,6 @@ define void @conditional_alloca(i64 %n) nounwind {
; Z-NEXT: nill %r2, 65520
; Z-NEXT: lay %r15, -8(%r1)
; Z-NEXT: brasl %r14, notdead at PLT
-; Z-NEXT: .LBB0_2: # %if.end
; Z-NEXT: lmg %r11, %r15, 248(%r11)
; Z-NEXT: br %r14
;
@@ -118,3 +117,235 @@ call:
ret:
ret void
}
+
+define void @test_backchain(i64 %n) nounwind "backchain" {
+; CHECK-LABEL: test_backchain:
+; CHECK: # %bb.0:
+; CHECK-NEXT: stmg %r11, %r15, 88(%r15)
+; CHECK-NEXT: lgr %r1, %r15
+; CHECK-NEXT: aghi %r15, -160
+; Z-LABEL: test_backchain:
+; Z: # %bb.0:
+; Z-NEXT: stmg %r11, %r15, 88(%r15)
+; Z-NEXT: lgr %r1, %r15
+; Z-NEXT: aghi %r15, -160
+; Z-NEXT: stg %r1, 0(%r15)
+; Z-NEXT: lgr %r11, %r15
+; Z-NEXT: cgije %r2, 0, .LBB2_2
+; Z-NEXT: # %bb.1: # %if.then
+; Z-NEXT: lgr %r1, %r15
+; Z-NEXT: la %r0, 7(%r2)
+; Z-NEXT: lg %r3, 0(%r15)
+; Z-NEXT: nill %r0, 65528
+; Z-NEXT: sgr %r1, %r0
+; Z-NEXT: lay %r15, -8(%r1)
+; Z-NEXT: la %r2, 160(%r1)
+; Z-NEXT: nill %r2, 65520
+; Z-NEXT: stg %r3, -8(%r1)
+; Z-NEXT: brasl %r14, notdead at PLT
+; Z-NEXT: .LBB2_2: # %if.end
+; Z-NEXT: lmg %r11, %r15, 248(%r11)
+; Z-NEXT: br %r14
+;
+; SW-LABEL: test_backchain:
+; SW: # %bb.0:
+; SW-NEXT: cgibe %r2, 0, 0(%r14)
+; SW-NEXT: .LBB2_1: # %if.then
+; SW-NEXT: stmg %r11, %r15, 88(%r15)
+; SW-NEXT: lgr %r1, %r15
+; SW-NEXT: aghi %r15, -160
+; SW-NEXT: stg %r1, 0(%r15)
+; SW-NEXT: lgr %r11, %r15
+; SW-NEXT: lgr %r1, %r15
+; SW-NEXT: la %r0, 7(%r2)
+; SW-NEXT: lg %r3, 0(%r15)
+; SW-NEXT: nill %r0, 65528
+; SW-NEXT: sgr %r1, %r0
+; SW-NEXT: lay %r15, -8(%r1)
+; SW-NEXT: la %r2, 160(%r1)
+; SW-NEXT: nill %r2, 65520
+; SW-NEXT: stg %r3, -8(%r1)
+; SW-NEXT: brasl %r14, notdead at PLT
+; SW-NEXT: lmg %r11, %r15, 248(%r11)
+; SW-NEXT: br %r14
+ %cmp = icmp eq i64 %n, 0
+ br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+ %addr = alloca i8, i64 %n, align 16
+ call void @notdead(ptr %addr)
+ br label %if.end
+
+if.end:
+ ret void
+}
+
+define void @test_mcount(i64 %n) nounwind "systemz-instrument-function-entry"="mcount" {
+; CHECK-LABEL: test_mcount:
+; CHECK: # %bb.0:
+; CHECK-NEXT: brasl %r0, mcount at PLT
+; Z-LABEL: test_mcount:
+; Z: # %bb.0:
+; Z-NEXT: stg %r14, 8(%r15)
+; Z-NEXT: brasl %r14, mcount at PLT
+; Z-NEXT: lg %r14, 8(%r15)
+; Z-NEXT: stmg %r11, %r15, 88(%r15)
+; Z-NEXT: aghi %r15, -160
+; Z-NEXT: lgr %r11, %r15
+; Z-NEXT: cgije %r2, 0, .LBB3_2
+; Z-NEXT: # %bb.1: # %if.then
+; Z-NEXT: lgr %r1, %r15
+; Z-NEXT: la %r0, 7(%r2)
+; Z-NEXT: nill %r0, 65528
+; Z-NEXT: sgr %r1, %r0
+; Z-NEXT: la %r2, 160(%r1)
+; Z-NEXT: nill %r2, 65520
+; Z-NEXT: lay %r15, -8(%r1)
+; Z-NEXT: brasl %r14, notdead at PLT
+; Z-NEXT: .LBB3_2: # %if.end
+; Z-NEXT: lmg %r11, %r15, 248(%r11)
+; Z-NEXT: br %r14
+;
+; SW-LABEL: test_mcount:
+; SW: # %bb.0:
+; SW-NEXT: cgibe %r2, 0, 0(%r14)
+; SW-NEXT: .LBB3_1: # %if.then
+; SW-NEXT: stg %r14, 8(%r15)
+; SW-NEXT: brasl %r14, mcount at PLT
+; SW-NEXT: lg %r14, 8(%r15)
+; SW-NEXT: stmg %r11, %r15, 88(%r15)
+; SW-NEXT: aghi %r15, -160
+; SW-NEXT: lgr %r11, %r15
+; SW-NEXT: lgr %r1, %r15
+; SW-NEXT: la %r0, 7(%r2)
+; SW-NEXT: nill %r0, 65528
+; SW-NEXT: sgr %r1, %r0
+; SW-NEXT: la %r2, 160(%r1)
+; SW-NEXT: nill %r2, 65520
+; SW-NEXT: lay %r15, -8(%r1)
+; SW-NEXT: brasl %r14, notdead at PLT
+; SW-NEXT: lmg %r11, %r15, 248(%r11)
+; SW-NEXT: br %r14
+ %cmp = icmp eq i64 %n, 0
+ br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+ %addr = alloca i8, i64 %n, align 16
+ call void @notdead(ptr %addr)
+ br label %if.end
+
+if.end:
+ ret void
+}
+
+declare ghccc void @ghc_callee(i64)
+
+; to test that GHC functions disable shrink-wrapping without triggering the fatal
+; error in SystemZELFFrameLowering::emitPrologue. This test writes a standard
+; GHC function that calls another GHC function conditionally
+define ghccc void @test_ghc(i64 %n) nounwind {
+; CHECK-LABEL: test_ghc:
+; CHECK: # %bb.0:
+; CHECK-NEXT: cgije %r2, 0, .LBB3_2
+; CHECK-NEXT: # %bb.1: # %if.then
+; CHECK-NEXT: brasl %r14, ghc_callee
+; CHECK-NEXT: .LBB3_2: # %if.end
+; CHECK-NEXT: br %r14
+; Z-LABEL: test_ghc:
+; Z: # %bb.0:
+; Z-NEXT: cghi %r7, 0
+; Z-NEXT: jglh ghc_callee at PLT
+; Z-NEXT: .LBB4_1: # %if.end
+; Z-NEXT: br %r14
+;
+; SW-LABEL: test_ghc:
+; SW: # %bb.0:
+; SW-NEXT: cghi %r7, 0
+; SW-NEXT: jglh ghc_callee at PLT
+; SW-NEXT: .LBB4_1: # %if.end
+; SW-NEXT: br %r14
+ %cmp = icmp eq i64 %n, 0
+ br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+ tail call ghccc void @ghc_callee(i64 %n)
+ br label %if.end
+
+if.end:
+ ret void
+}
+
+
+
+define void @test_stack_probe(i64 %n) nounwind "probe-stack"="inline-asm" {
+; CHECK-LABEL: test_stack_probe:
+; CHECK: # %bb.0:
+; Z-LABEL: test_stack_probe:
+; Z: # %bb.0:
+; Z-NEXT: stmg %r11, %r15, 88(%r15)
+; Z-NEXT: aghi %r15, -160
+; Z-NEXT: lgr %r11, %r15
+; Z-NEXT: cgije %r2, 0, .LBB5_6
+; Z-NEXT: # %bb.1: # %if.then
+; Z-NEXT: lghi %r1, 8200
+; Z-NEXT: clgfi %r1, 4096
+; Z-NEXT: jl .LBB5_3
+; Z-NEXT: .LBB5_2: # %if.then
+; Z-NEXT: # =>This Inner Loop Header: Depth=1
+; Z-NEXT: slgfi %r1, 4096
+; Z-NEXT: slgfi %r15, 4096
+; Z-NEXT: cg %r15, 4088(%r15)
+; Z-NEXT: clgfi %r1, 4096
+; Z-NEXT: jhe .LBB5_2
+; Z-NEXT: .LBB5_3: # %if.then
+; Z-NEXT: cgije %r1, 0, .LBB5_5
+; Z-NEXT: # %bb.4: # %if.then
+; Z-NEXT: slgr %r15, %r1
+; Z-NEXT: cg %r15, -8(%r1,%r15)
+; Z-NEXT: .LBB5_5: # %if.then
+; Z-NEXT: la %r2, 168(%r15)
+; Z-NEXT: nill %r2, 65520
+; Z-NEXT: brasl %r14, notdead at PLT
+; Z-NEXT: .LBB5_6: # %if.end
+; Z-NEXT: lmg %r11, %r15, 248(%r11)
+; Z-NEXT: br %r14
+;
+; SW-LABEL: test_stack_probe:
+; SW: # %bb.0:
+; SW-NEXT: cgibe %r2, 0, 0(%r14)
+; SW-NEXT: .LBB5_1: # %if.then
+; SW-NEXT: stmg %r11, %r15, 88(%r15)
+; SW-NEXT: aghi %r15, -160
+; SW-NEXT: lgr %r11, %r15
+; SW-NEXT: lghi %r1, 8200
+; SW-NEXT: clgfi %r1, 4096
+; SW-NEXT: jl .LBB5_3
+; SW-NEXT: .LBB5_2: # %if.then
+; SW-NEXT: # =>This Inner Loop Header: Depth=1
+; SW-NEXT: slgfi %r1, 4096
+; SW-NEXT: slgfi %r15, 4096
+; SW-NEXT: cg %r15, 4088(%r15)
+; SW-NEXT: clgfi %r1, 4096
+; SW-NEXT: jhe .LBB5_2
+; SW-NEXT: .LBB5_3: # %if.then
+; SW-NEXT: cgije %r1, 0, .LBB5_5
+; SW-NEXT: # %bb.4: # %if.then
+; SW-NEXT: slgr %r15, %r1
+; SW-NEXT: cg %r15, -8(%r1,%r15)
+; SW-NEXT: .LBB5_5: # %if.then
+; SW-NEXT: la %r2, 168(%r15)
+; SW-NEXT: nill %r2, 65520
+; SW-NEXT: brasl %r14, notdead at PLT
+; SW-NEXT: lmg %r11, %r15, 248(%r11)
+; SW-NEXT: br %r14
+ %cmp = icmp eq i64 %n, 0
+ br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+ %addr = alloca i8, i64 8192, align 16
+ call void @notdead(ptr %addr)
+ br label %if.end
+
+if.end:
+ ret void
+}
More information about the llvm-commits
mailing list