[llvm] [SystemZ] Add shrink-wrapping support for ELF prologue/epilogue (PR #225240)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 28 06:03:40 PDT 2026


https://github.com/MarkVeerasingam updated https://github.com/llvm/llvm-project/pull/225240

>From 161489ed117ac71c4b84570239a0e12da89c1d9d Mon Sep 17 00:00:00 2001
From: MarkVeerasingam <markveer70 at gmail.com>
Date: Sun, 20 Sep 2026 14:51:32 -0500
Subject: [PATCH 1/5] Added a shrinkwrap test for SystemZ codegen

---
 llvm/test/CodeGen/SystemZ/shrinkwrap.ll | 22 ++++++++++++++++++++++
 1 file changed, 22 insertions(+)
 create mode 100644 llvm/test/CodeGen/SystemZ/shrinkwrap.ll

diff --git a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
new file mode 100644
index 0000000000000..10bba42ddd862
--- /dev/null
+++ b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
@@ -0,0 +1,22 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+;
+; RUN: llc -mtriple=s390x-linux-gnu -O2 < %s | FileCheck %s -check-prefix=Z
+; RUN: llc -mtriple=s390x-linux-gnu -O2 -enable-shrink-wrap < %s | FileCheck %s -check-prefix=SW
+
+declare void @notdead(ptr)
+
+define void @conditional_alloca(i64 %n) nounwind {
+; Z-LABEL: conditional_alloca:
+;
+; SW-LABEL: conditional_alloca:
+  %cmp = icmp eq i64 %n, 0
+  br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+  %addr = alloca i8, i64 %n, align 16
+  call void @notdead(ptr %addr)
+  br label %if.end
+
+if.end:
+  ret void
+}

>From e36cce6e6ae08ddcc70f1b5292f8123461f5e2cc Mon Sep 17 00:00:00 2001
From: MarkVeerasingam <markveer70 at gmail.com>
Date: Mon, 21 Sep 2026 14:30:40 -0500
Subject: [PATCH 2/5] [SystemZ] Make ELF function prologue/epilogue shrink-wrap
 capable

---
 .../Target/SystemZ/SystemZFrameLowering.cpp   | 24 ++++++++----
 llvm/test/CodeGen/SystemZ/shrinkwrap.ll       | 37 ++++++++++++++++++-
 2 files changed, 52 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
index 14688a53c1c63..de4c0d2331442 100644
--- a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
@@ -530,7 +530,6 @@ static void buildDefCFAReg(MachineBasicBlock &MBB,
 
 void SystemZELFFrameLowering::emitPrologue(MachineFunction &MF,
                                            MachineBasicBlock &MBB) const {
-  assert(&MF.front() == &MBB && "Shrink-wrapping not yet supported");
   const SystemZSubtarget &STI = MF.getSubtarget<SystemZSubtarget>();
   const SystemZTargetLowering &TLI = *STI.getTargetLowering();
   MachineFrameInfo &MFFrame = MF.getFrameInfo();
@@ -720,19 +719,32 @@ void SystemZELFFrameLowering::emitPrologue(MachineFunction &MF,
 
 void SystemZELFFrameLowering::emitEpilogue(MachineFunction &MF,
                                            MachineBasicBlock &MBB) const {
-  MachineBasicBlock::iterator MBBI = MBB.getLastNonDebugInstr();
   auto *ZII =
       static_cast<const SystemZInstrInfo *>(MF.getSubtarget().getInstrInfo());
   SystemZMachineFunctionInfo *ZFI = MF.getInfo<SystemZMachineFunctionInfo>();
   MachineFrameInfo &MFFrame = MF.getFrameInfo();
 
+  // A shrink-wrapped epilogue can be inserted into a block without a return
+  // instruction. Insert before the first terminator, or after the
+  // non-debug instruction when the block has no terminator.
+  MachineBasicBlock::iterator MBBI = MBB.end();
+  DebugLoc DL;
+  if (!MBB.empty()) {
+    MBBI = MBB.getFirstTerminator();
+    if (MBBI == MBB.end())
+      MBBI = MBB.getLastNonDebugInstr();
+
+    if (MBBI != MBB.end()) {
+      DL = MBBI->getDebugLoc();
+      if (!MBBI->isTerminator())
+        MBBI = std::next(MBBI);
+    }
+  }
+
   // See SystemZELFFrameLowering::emitPrologue
   if (MF.getFunction().getCallingConv() == CallingConv::GHC)
     return;
 
-  // Skip the return instruction.
-  assert(MBBI->isReturn() && "Can only insert epilogue into returning blocks");
-
   uint64_t StackSize = MFFrame.getStackSize();
   if (ZFI->getRestoreGPRRegs().LowGPR) {
     --MBBI;
@@ -741,7 +753,6 @@ void SystemZELFFrameLowering::emitEpilogue(MachineFunction &MF,
       llvm_unreachable("Expected to see callee-save register restore code");
 
     unsigned AddrOpNo = 2;
-    DebugLoc DL = MBBI->getDebugLoc();
     uint64_t Offset = StackSize + MBBI->getOperand(AddrOpNo + 1).getImm();
     unsigned NewOpcode = ZII->getOpcodeForOffset(Opcode, Offset);
 
@@ -759,7 +770,6 @@ void SystemZELFFrameLowering::emitEpilogue(MachineFunction &MF,
     MBBI->setDesc(ZII->get(NewOpcode));
     MBBI->getOperand(AddrOpNo + 1).ChangeToImmediate(Offset);
   } else if (StackSize) {
-    DebugLoc DL = MBBI->getDebugLoc();
     emitIncrement(MBB, MBBI, DL, SystemZ::R15D, StackSize, ZII);
   }
 }
diff --git a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
index 10bba42ddd862..d1660535d77b0 100644
--- a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
+++ b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
@@ -1,14 +1,47 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
 ;
-; RUN: llc -mtriple=s390x-linux-gnu -O2 < %s | FileCheck %s -check-prefix=Z
-; RUN: llc -mtriple=s390x-linux-gnu -O2 -enable-shrink-wrap < %s | FileCheck %s -check-prefix=SW
+; RUN: llc -mtriple=s390x-linux-gnu -O2 -verify-machineinstrs < %s | FileCheck %s -check-prefix=Z
+; RUN: llc -mtriple=s390x-linux-gnu -O2 -enable-shrink-wrap -verify-machineinstrs < %s | FileCheck %s -check-prefix=SW
 
 declare void @notdead(ptr)
 
 define void @conditional_alloca(i64 %n) nounwind {
 ; Z-LABEL: conditional_alloca:
+; Z:       # %bb.0:
+; Z-NEXT:    stmg %r11, %r15, 88(%r15)
+; Z-NEXT:    aghi %r15, -160
+; Z-NEXT:    lgr %r11, %r15
+; Z-NEXT:    cgije %r2, 0, .LBB0_2
+; Z-NEXT:  # %bb.1: # %if.then
+; Z-NEXT:    lgr %r1, %r15
+; Z-NEXT:    la %r0, 7(%r2)
+; Z-NEXT:    nill %r0, 65528
+; Z-NEXT:    sgr %r1, %r0
+; Z-NEXT:    la %r2, 160(%r1)
+; Z-NEXT:    nill %r2, 65520
+; Z-NEXT:    lay %r15, -8(%r1)
+; Z-NEXT:    brasl %r14, notdead at PLT
+; Z-NEXT:  .LBB0_2: # %if.end
+; Z-NEXT:    lmg %r11, %r15, 248(%r11)
+; Z-NEXT:    br %r14
 ;
 ; SW-LABEL: conditional_alloca:
+; SW:       # %bb.0:
+; SW-NEXT:    cgibe %r2, 0, 0(%r14)
+; SW-NEXT:  .LBB0_1: # %if.then
+; SW-NEXT:    stmg %r11, %r15, 88(%r15)
+; SW-NEXT:    aghi %r15, -160
+; SW-NEXT:    lgr %r11, %r15
+; SW-NEXT:    lgr %r1, %r15
+; SW-NEXT:    la %r0, 7(%r2)
+; SW-NEXT:    nill %r0, 65528
+; SW-NEXT:    sgr %r1, %r0
+; SW-NEXT:    la %r2, 160(%r1)
+; SW-NEXT:    nill %r2, 65520
+; SW-NEXT:    lay %r15, -8(%r1)
+; SW-NEXT:    brasl %r14, notdead at PLT
+; SW-NEXT:    lmg %r11, %r15, 248(%r11)
+; SW-NEXT:    br %r14
   %cmp = icmp eq i64 %n, 0
   br i1 %cmp, label %if.end, label %if.then
 

>From d7f2a146c9a0013e5aaa2d261f5b5475a93c8671 Mon Sep 17 00:00:00 2001
From: MarkVeerasingam <markveer70 at gmail.com>
Date: Mon, 21 Sep 2026 17:41:35 -0500
Subject: [PATCH 3/5] [SystemZ] Updated shrinkwrap test to cover vararg
 clobbering instructions

---
 llvm/test/CodeGen/SystemZ/shrinkwrap.ll | 64 +++++++++++++++++++++++++
 1 file changed, 64 insertions(+)

diff --git a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
index d1660535d77b0..0209d94f777d1 100644
--- a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
+++ b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
@@ -6,6 +6,7 @@
 declare void @notdead(ptr)
 
 define void @conditional_alloca(i64 %n) nounwind {
+; existing generated checks...
 ; Z-LABEL: conditional_alloca:
 ; Z:       # %bb.0:
 ; Z-NEXT:    stmg %r11, %r15, 88(%r15)
@@ -53,3 +54,66 @@ if.then:
 if.end:
   ret void
 }
+
+; Varargs prologue must remain in the entry block because it saves incoming
+; GPR argument registers before they can be clobbered.
+declare void @llvm.va_start.p0(ptr)
+declare void @llvm.va_end.p0(ptr)
+declare void @take(ptr)
+
+define void @va_fpr(double %a, double %b, double %c, double %d,
+; Z-LABEL: va_fpr:
+; Z:       # %bb.0:
+; Z-NEXT:    stmg %r3, %r15, 24(%r15)
+; Z-NEXT:    aghi %r15, -168
+; Z-NEXT:    #APP
+; Z-NEXT:    lghi %r3, 0
+; Z-NEXT:    #NO_APP
+; Z-NEXT:    cije %r2, 0, .LBB1_2
+; Z-NEXT:  # %bb.1: # %call
+; Z-NEXT:    la %r0, 168(%r15)
+; Z-NEXT:    stg %r0, 184(%r15)
+; Z-NEXT:    la %r0, 328(%r15)
+; Z-NEXT:    stg %r0, 176(%r15)
+; Z-NEXT:    mvghi 168(%r15), 4
+; Z-NEXT:    la %r2, 160(%r15)
+; Z-NEXT:    mvghi 160(%r15), 1
+; Z-NEXT:    brasl %r14, take at PLT
+; Z-NEXT:  .LBB1_2: # %ret
+; Z-NEXT:    lmg %r6, %r15, 216(%r15)
+; Z-NEXT:    br %r14
+;
+; SW-LABEL: va_fpr:
+; SW:       # %bb.0:
+; SW-NEXT:    #APP
+; SW-NEXT:    lghi %r3, 0
+; SW-NEXT:    #NO_APP
+; SW-NEXT:    cibe %r2, 0, 0(%r14)
+; SW-NEXT:  .LBB1_1: # %call
+; SW-NEXT:    stmg %r3, %r15, 24(%r15)
+; SW-NEXT:    aghi %r15, -168
+; SW-NEXT:    la %r0, 168(%r15)
+; SW-NEXT:    stg %r0, 184(%r15)
+; SW-NEXT:    la %r0, 328(%r15)
+; SW-NEXT:    stg %r0, 176(%r15)
+; SW-NEXT:    mvghi 168(%r15), 4
+; SW-NEXT:    la %r2, 160(%r15)
+; SW-NEXT:    mvghi 160(%r15), 1
+; SW-NEXT:    brasl %r14, take at PLT
+; SW-NEXT:    lmg %r6, %r15, 216(%r15)
+; SW-NEXT:    br %r14
+                    i32 %n, ...) nounwind {
+  call void asm sideeffect "lghi %r3, 0", "~{r3}"()
+  %ap = alloca ptr
+  %z = icmp eq i32 %n, 0
+  br i1 %z, label %ret, label %call
+
+call:
+  call void @llvm.va_start.p0(ptr %ap)
+  call void @take(ptr %ap)
+  call void @llvm.va_end.p0(ptr %ap)
+  br label %ret
+
+ret:
+  ret void
+}

>From 2288306b46c9a271ae435a7cbd6ac1b720b13d93 Mon Sep 17 00:00:00 2001
From: MarkVeerasingam <markveer70 at gmail.com>
Date: Mon, 21 Sep 2026 18:07:34 -0500
Subject: [PATCH 4/5] [SystemZ] Keep varargs prologue in entry block

The varargs prologue saves incoming GPR argument registers, so it must
remain in the entry block when shrink-wrapping is enabled. Otherwise an
instruction before the shrink-wrapped prologue can clobber an incoming
argument register before it is saved.

Add a SystemZ-specific canUseAsPrologue restriction for varargs functions
and update the shrink-wrap test to cover the case.
---
 llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp | 11 +++++++++++
 llvm/lib/Target/SystemZ/SystemZFrameLowering.h   |  2 ++
 llvm/test/CodeGen/SystemZ/shrinkwrap.ll          |  9 +++++----
 3 files changed, 18 insertions(+), 4 deletions(-)

diff --git a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
index de4c0d2331442..ee7a66c6ee6a0 100644
--- a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
@@ -1555,6 +1555,17 @@ void SystemZXPLINKFrameLowering::processFunctionBeforeFrameFinalized(
   }
 }
 
+bool SystemZELFFrameLowering::canUseAsPrologue(
+    const MachineBasicBlock &MBB) const {
+  const MachineFunction &MF = *MBB.getParent();
+
+  // Keep the varargs prologue in the entry block so incoming GPRs are saved
+  // before they can be clobbered.
+  if (MF.getFunction().isVarArg())
+    return &MBB == &MF.front();
+  return true;
+}
+
 // Determines the size of the frame, and creates the deferred spill objects.
 void SystemZXPLINKFrameLowering::determineFrameLayout(
     MachineFunction &MF) const {
diff --git a/llvm/lib/Target/SystemZ/SystemZFrameLowering.h b/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
index c67139aac203f..c3941103d5b9f 100644
--- a/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
+++ b/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
@@ -118,6 +118,8 @@ class SystemZELFFrameLowering : public SystemZFrameLowering {
   // Get or create the frame index of where the old frame pointer is stored.
   int getOrCreateFramePointerSaveIndex(MachineFunction &MF) const override;
 
+  bool canUseAsPrologue(const MachineBasicBlock &MBB) const override;
+
 protected:
   bool hasFPImpl(const MachineFunction &MF) const override;
 };
diff --git a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
index 0209d94f777d1..687837921122c 100644
--- a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
+++ b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
@@ -85,13 +85,13 @@ define void @va_fpr(double %a, double %b, double %c, double %d,
 ;
 ; SW-LABEL: va_fpr:
 ; SW:       # %bb.0:
+; SW-NEXT:    stmg %r3, %r15, 24(%r15)
+; SW-NEXT:    aghi %r15, -168
 ; SW-NEXT:    #APP
 ; SW-NEXT:    lghi %r3, 0
 ; SW-NEXT:    #NO_APP
-; SW-NEXT:    cibe %r2, 0, 0(%r14)
-; SW-NEXT:  .LBB1_1: # %call
-; SW-NEXT:    stmg %r3, %r15, 24(%r15)
-; SW-NEXT:    aghi %r15, -168
+; SW-NEXT:    cije %r2, 0, .LBB1_2
+; SW-NEXT:  # %bb.1: # %call
 ; SW-NEXT:    la %r0, 168(%r15)
 ; SW-NEXT:    stg %r0, 184(%r15)
 ; SW-NEXT:    la %r0, 328(%r15)
@@ -100,6 +100,7 @@ define void @va_fpr(double %a, double %b, double %c, double %d,
 ; SW-NEXT:    la %r2, 160(%r15)
 ; SW-NEXT:    mvghi 160(%r15), 1
 ; SW-NEXT:    brasl %r14, take at PLT
+; SW-NEXT:  .LBB1_2: # %ret
 ; SW-NEXT:    lmg %r6, %r15, 216(%r15)
 ; SW-NEXT:    br %r14
                     i32 %n, ...) nounwind {

>From ee97ee47432d265743374008107770dffa5d2d98 Mon Sep 17 00:00:00 2001
From: MarkVeerasingam <markveer70 at gmail.com>
Date: Mon, 28 Sep 2026 08:02:08 -0500
Subject: [PATCH 5/5] [SystemZ] Enable shrink-wrapping for SystemZ ELF

Enable shrink-wrapping for SystemZ ELF, with a few guards for cases where
the prologue must remain in the entry block.

Disable shrink-wrapping for GHC functions, mcount instrumentation, stack
backchains, and inline stack probing.

Also avoid updating Frame Pointer liveness when the prologue is emitted in
a non-entry block, and reject candidate blocks where SystemZ::CC is live-in
since the prologue may clobber it.

Add regression tests covering these cases.
---
 .../Target/SystemZ/SystemZFrameLowering.cpp   |  38 ++-
 .../lib/Target/SystemZ/SystemZFrameLowering.h |   2 +
 llvm/test/CodeGen/SystemZ/int-cmp-02.ll       | 143 +++++++----
 llvm/test/CodeGen/SystemZ/shrinkwrap.ll       | 237 +++++++++++++++++-
 4 files changed, 363 insertions(+), 57 deletions(-)

diff --git a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
index ee7a66c6ee6a0..1f14d4b4295c0 100644
--- a/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZFrameLowering.cpp
@@ -674,8 +674,10 @@ void SystemZELFFrameLowering::emitPrologue(MachineFunction &MF,
     // Mark the FramePtr as live at the beginning of every block except
     // the entry block.  (We'll have marked R11 as live on entry when
     // saving the GPRs.)
-    for (MachineBasicBlock &MBBJ : llvm::drop_begin(MF))
-      MBBJ.addLiveIn(SystemZ::R11D);
+    if (&MBB == &MF.front()) {
+      for (MachineBasicBlock &MBBJ : llvm::drop_begin(MF))
+        MBBJ.addLiveIn(SystemZ::R11D);
+    }
   }
 
   // Skip over the FPR/VR saves.
@@ -1555,6 +1557,32 @@ void SystemZXPLINKFrameLowering::processFunctionBeforeFrameFinalized(
   }
 }
 
+bool SystemZELFFrameLowering::enableShrinkWrapping(
+    const MachineFunction &MF) const {
+  const Function &F = MF.getFunction();
+  const SystemZSubtarget &Subtarget = MF.getSubtarget<SystemZSubtarget>();
+
+  // GHC calling convention does not use standard prologue/epilogue.
+  if (F.getCallingConv() == CallingConv::GHC)
+    return false;
+
+  // mcount instrumentation must be called at the function entry.
+  if (F.hasFnAttribute("systemz-instrument-function-entry"))
+    return false;
+
+  // Backchain setup currently assumes %r1 is free at the entry block.
+  // TODO: Investigate if we need to generalise the temp reg handling.
+  // Right now we just disable shrink-wrapping if Backchain is enabled
+  if (Subtarget.hasBackChain())
+    return false;
+
+  // stack probing uses %r0/%r1 scratch registers.
+  if (Subtarget.getTargetLowering()->hasInlineStackProbe(MF))
+    return false;
+
+  return true;
+}
+
 bool SystemZELFFrameLowering::canUseAsPrologue(
     const MachineBasicBlock &MBB) const {
   const MachineFunction &MF = *MBB.getParent();
@@ -1563,6 +1591,12 @@ bool SystemZELFFrameLowering::canUseAsPrologue(
   // before they can be clobbered.
   if (MF.getFunction().isVarArg())
     return &MBB == &MF.front();
+
+  // If CC is live into MBB, prologue instructions (e.g. AGHI/AGFI for stack
+  // adjustments) will clobber CC.
+  if (MBB.isLiveIn(SystemZ::CC))
+    return false;
+
   return true;
 }
 
diff --git a/llvm/lib/Target/SystemZ/SystemZFrameLowering.h b/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
index c3941103d5b9f..b33bf577b98a5 100644
--- a/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
+++ b/llvm/lib/Target/SystemZ/SystemZFrameLowering.h
@@ -118,6 +118,8 @@ class SystemZELFFrameLowering : public SystemZFrameLowering {
   // Get or create the frame index of where the old frame pointer is stored.
   int getOrCreateFramePointerSaveIndex(MachineFunction &MF) const override;
 
+  bool enableShrinkWrapping(const MachineFunction &MF) const override;
+
   bool canUseAsPrologue(const MachineBasicBlock &MBB) const override;
 
 protected:
diff --git a/llvm/test/CodeGen/SystemZ/int-cmp-02.ll b/llvm/test/CodeGen/SystemZ/int-cmp-02.ll
index 3fd2f24d353a7..b5ce7aa4cf211 100644
--- a/llvm/test/CodeGen/SystemZ/int-cmp-02.ll
+++ b/llvm/test/CodeGen/SystemZ/int-cmp-02.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
 ; Test 32-bit signed comparison in which the second operand is a variable.
 ;
 ; RUN: llc < %s -mtriple=s390x-linux-gnu | FileCheck %s
@@ -7,9 +8,11 @@ declare i32 @foo()
 ; Check register comparison.
 define double @f1(double %a, double %b, i32 %i1, i32 %i2) {
 ; CHECK-LABEL: f1:
-; CHECK: crbl %r2, %r3, 0(%r14)
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    crbl %r2, %r3, 0(%r14)
+; CHECK-NEXT:  .LBB0_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %cond = icmp slt i32 %i1, %i2
   %res = select i1 %cond, double %a, double %b
   ret double %res
@@ -18,10 +21,12 @@ define double @f1(double %a, double %b, i32 %i1, i32 %i2) {
 ; Check the low end of the C range.
 define double @f2(double %a, double %b, i32 %i1, ptr %ptr) {
 ; CHECK-LABEL: f2:
-; CHECK: c %r2, 0(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    c %r2, 0(%r3)
+; CHECK-NEXT:    blr %r14
+; CHECK-NEXT:  .LBB1_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %i2 = load i32, ptr %ptr
   %cond = icmp slt i32 %i1, %i2
   %res = select i1 %cond, double %a, double %b
@@ -31,10 +36,12 @@ define double @f2(double %a, double %b, i32 %i1, ptr %ptr) {
 ; Check the high end of the aligned C range.
 define double @f3(double %a, double %b, i32 %i1, ptr %base) {
 ; CHECK-LABEL: f3:
-; CHECK: c %r2, 4092(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    c %r2, 4092(%r3)
+; CHECK-NEXT:    blr %r14
+; CHECK-NEXT:  .LBB2_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %ptr = getelementptr i32, ptr %base, i64 1023
   %i2 = load i32, ptr %ptr
   %cond = icmp slt i32 %i1, %i2
@@ -45,10 +52,12 @@ define double @f3(double %a, double %b, i32 %i1, ptr %base) {
 ; Check the next word up, which should use CY instead of C.
 define double @f4(double %a, double %b, i32 %i1, ptr %base) {
 ; CHECK-LABEL: f4:
-; CHECK: cy %r2, 4096(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    cy %r2, 4096(%r3)
+; CHECK-NEXT:    blr %r14
+; CHECK-NEXT:  .LBB3_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %ptr = getelementptr i32, ptr %base, i64 1024
   %i2 = load i32, ptr %ptr
   %cond = icmp slt i32 %i1, %i2
@@ -59,10 +68,12 @@ define double @f4(double %a, double %b, i32 %i1, ptr %base) {
 ; Check the high end of the aligned CY range.
 define double @f5(double %a, double %b, i32 %i1, ptr %base) {
 ; CHECK-LABEL: f5:
-; CHECK: cy %r2, 524284(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    cy %r2, 524284(%r3)
+; CHECK-NEXT:    blr %r14
+; CHECK-NEXT:  .LBB4_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %ptr = getelementptr i32, ptr %base, i64 131071
   %i2 = load i32, ptr %ptr
   %cond = icmp slt i32 %i1, %i2
@@ -74,11 +85,13 @@ define double @f5(double %a, double %b, i32 %i1, ptr %base) {
 ; Other sequences besides this one would be OK.
 define double @f6(double %a, double %b, i32 %i1, ptr %base) {
 ; CHECK-LABEL: f6:
-; CHECK: agfi %r3, 524288
-; CHECK: c %r2, 0(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    agfi %r3, 524288
+; CHECK-NEXT:    c %r2, 0(%r3)
+; CHECK-NEXT:    blr %r14
+; CHECK-NEXT:  .LBB5_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %ptr = getelementptr i32, ptr %base, i64 131072
   %i2 = load i32, ptr %ptr
   %cond = icmp slt i32 %i1, %i2
@@ -89,10 +102,12 @@ define double @f6(double %a, double %b, i32 %i1, ptr %base) {
 ; Check the high end of the negative aligned CY range.
 define double @f7(double %a, double %b, i32 %i1, ptr %base) {
 ; CHECK-LABEL: f7:
-; CHECK: cy %r2, -4(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    cy %r2, -4(%r3)
+; CHECK-NEXT:    blr %r14
+; CHECK-NEXT:  .LBB6_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %ptr = getelementptr i32, ptr %base, i64 -1
   %i2 = load i32, ptr %ptr
   %cond = icmp slt i32 %i1, %i2
@@ -103,10 +118,12 @@ define double @f7(double %a, double %b, i32 %i1, ptr %base) {
 ; Check the low end of the CY range.
 define double @f8(double %a, double %b, i32 %i1, ptr %base) {
 ; CHECK-LABEL: f8:
-; CHECK: cy %r2, -524288(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    cy %r2, -524288(%r3)
+; CHECK-NEXT:    blr %r14
+; CHECK-NEXT:  .LBB7_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %ptr = getelementptr i32, ptr %base, i64 -131072
   %i2 = load i32, ptr %ptr
   %cond = icmp slt i32 %i1, %i2
@@ -118,11 +135,13 @@ define double @f8(double %a, double %b, i32 %i1, ptr %base) {
 ; Other sequences besides this one would be OK.
 define double @f9(double %a, double %b, i32 %i1, ptr %base) {
 ; CHECK-LABEL: f9:
-; CHECK: agfi %r3, -524292
-; CHECK: c %r2, 0(%r3)
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    agfi %r3, -524292
+; CHECK-NEXT:    c %r2, 0(%r3)
+; CHECK-NEXT:    blr %r14
+; CHECK-NEXT:  .LBB8_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %ptr = getelementptr i32, ptr %base, i64 -131073
   %i2 = load i32, ptr %ptr
   %cond = icmp slt i32 %i1, %i2
@@ -133,10 +152,12 @@ define double @f9(double %a, double %b, i32 %i1, ptr %base) {
 ; Check that C allows an index.
 define double @f10(double %a, double %b, i32 %i1, i64 %base, i64 %index) {
 ; CHECK-LABEL: f10:
-; CHECK: c %r2, 4092({{%r4,%r3|%r3,%r4}})
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    c %r2, 4092(%r4,%r3)
+; CHECK-NEXT:    blr %r14
+; CHECK-NEXT:  .LBB9_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %add1 = add i64 %base, %index
   %add2 = add i64 %add1, 4092
   %ptr = inttoptr i64 %add2 to ptr
@@ -149,10 +170,12 @@ define double @f10(double %a, double %b, i32 %i1, i64 %base, i64 %index) {
 ; Check that CY allows an index.
 define double @f11(double %a, double %b, i32 %i1, i64 %base, i64 %index) {
 ; CHECK-LABEL: f11:
-; CHECK: cy %r2, 4096({{%r4,%r3|%r3,%r4}})
-; CHECK-NEXT: blr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    cy %r2, 4096(%r4,%r3)
+; CHECK-NEXT:    blr %r14
+; CHECK-NEXT:  .LBB10_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %add1 = add i64 %base, %index
   %add2 = add i64 %add1, 4096
   %ptr = inttoptr i64 %add2 to ptr
@@ -166,9 +189,23 @@ define double @f11(double %a, double %b, i32 %i1, i64 %base, i64 %index) {
 ; critical edge %entry->%while.body, which lost the kills information for CC.
 define void @f12(i32 %a, i32 %b) {
 ; CHECK-LABEL: f12:
-; CHECK: cije %r2, 0
-; CHECK: crjlh %r2,
-; CHECK: br %r14
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    cibe %r2, 0, 0(%r14)
+; CHECK-NEXT:  .LBB11_1: # %while.body.preheader
+; CHECK-NEXT:    stmg %r13, %r15, 104(%r15)
+; CHECK-NEXT:    .cfi_offset %r13, -56
+; CHECK-NEXT:    .cfi_offset %r14, -48
+; CHECK-NEXT:    .cfi_offset %r15, -40
+; CHECK-NEXT:    aghi %r15, -160
+; CHECK-NEXT:    .cfi_def_cfa_offset 320
+; CHECK-NEXT:    lr %r13, %r3
+; CHECK-NEXT:  .LBB11_2: # %while.body
+; CHECK-NEXT:    # =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    brasl %r14, foo at PLT
+; CHECK-NEXT:    crjlh %r2, %r13, .LBB11_2
+; CHECK-NEXT:  # %bb.3:
+; CHECK-NEXT:    lmg %r13, %r15, 264(%r15)
+; CHECK-NEXT:    br %r14
 entry:
   %cmp11 = icmp eq i32 %a, 0
   br i1 %cmp11, label %while.end, label %while.body
@@ -185,10 +222,12 @@ while.end:
 ; Check the comparison can be reversed if that allows C to be used.
 define double @f13(double %a, double %b, i32 %i2, ptr %ptr) {
 ; CHECK-LABEL: f13:
-; CHECK: c %r2, 0(%r3)
-; CHECK-NEXT: bhr %r14
-; CHECK: ldr %f0, %f2
-; CHECK: br %r14
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    c %r2, 0(%r3)
+; CHECK-NEXT:    bhr %r14
+; CHECK-NEXT:  .LBB12_1:
+; CHECK-NEXT:    ldr %f0, %f2
+; CHECK-NEXT:    br %r14
   %i1 = load i32, ptr %ptr
   %cond = icmp slt i32 %i1, %i2
   %res = select i1 %cond, double %a, double %b
diff --git a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
index 687837921122c..efe8586173def 100644
--- a/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
+++ b/llvm/test/CodeGen/SystemZ/shrinkwrap.ll
@@ -9,11 +9,11 @@ define void @conditional_alloca(i64 %n) nounwind {
 ; existing generated checks...
 ; Z-LABEL: conditional_alloca:
 ; Z:       # %bb.0:
+; Z-NEXT:    cgibe %r2, 0, 0(%r14)
+; Z-NEXT:  .LBB0_1: # %if.then
 ; Z-NEXT:    stmg %r11, %r15, 88(%r15)
 ; Z-NEXT:    aghi %r15, -160
 ; Z-NEXT:    lgr %r11, %r15
-; Z-NEXT:    cgije %r2, 0, .LBB0_2
-; Z-NEXT:  # %bb.1: # %if.then
 ; Z-NEXT:    lgr %r1, %r15
 ; Z-NEXT:    la %r0, 7(%r2)
 ; Z-NEXT:    nill %r0, 65528
@@ -22,7 +22,6 @@ define void @conditional_alloca(i64 %n) nounwind {
 ; Z-NEXT:    nill %r2, 65520
 ; Z-NEXT:    lay %r15, -8(%r1)
 ; Z-NEXT:    brasl %r14, notdead at PLT
-; Z-NEXT:  .LBB0_2: # %if.end
 ; Z-NEXT:    lmg %r11, %r15, 248(%r11)
 ; Z-NEXT:    br %r14
 ;
@@ -118,3 +117,235 @@ call:
 ret:
   ret void
 }
+
+define void @test_backchain(i64 %n) nounwind "backchain" {
+; CHECK-LABEL: test_backchain:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    stmg %r11, %r15, 88(%r15)
+; CHECK-NEXT:    lgr  %r1, %r15
+; CHECK-NEXT:    aghi %r15, -160
+; Z-LABEL: test_backchain:
+; Z:       # %bb.0:
+; Z-NEXT:    stmg %r11, %r15, 88(%r15)
+; Z-NEXT:    lgr %r1, %r15
+; Z-NEXT:    aghi %r15, -160
+; Z-NEXT:    stg %r1, 0(%r15)
+; Z-NEXT:    lgr %r11, %r15
+; Z-NEXT:    cgije %r2, 0, .LBB2_2
+; Z-NEXT:  # %bb.1: # %if.then
+; Z-NEXT:    lgr %r1, %r15
+; Z-NEXT:    la %r0, 7(%r2)
+; Z-NEXT:    lg %r3, 0(%r15)
+; Z-NEXT:    nill %r0, 65528
+; Z-NEXT:    sgr %r1, %r0
+; Z-NEXT:    lay %r15, -8(%r1)
+; Z-NEXT:    la %r2, 160(%r1)
+; Z-NEXT:    nill %r2, 65520
+; Z-NEXT:    stg %r3, -8(%r1)
+; Z-NEXT:    brasl %r14, notdead at PLT
+; Z-NEXT:  .LBB2_2: # %if.end
+; Z-NEXT:    lmg %r11, %r15, 248(%r11)
+; Z-NEXT:    br %r14
+;
+; SW-LABEL: test_backchain:
+; SW:       # %bb.0:
+; SW-NEXT:    cgibe %r2, 0, 0(%r14)
+; SW-NEXT:  .LBB2_1: # %if.then
+; SW-NEXT:    stmg %r11, %r15, 88(%r15)
+; SW-NEXT:    lgr %r1, %r15
+; SW-NEXT:    aghi %r15, -160
+; SW-NEXT:    stg %r1, 0(%r15)
+; SW-NEXT:    lgr %r11, %r15
+; SW-NEXT:    lgr %r1, %r15
+; SW-NEXT:    la %r0, 7(%r2)
+; SW-NEXT:    lg %r3, 0(%r15)
+; SW-NEXT:    nill %r0, 65528
+; SW-NEXT:    sgr %r1, %r0
+; SW-NEXT:    lay %r15, -8(%r1)
+; SW-NEXT:    la %r2, 160(%r1)
+; SW-NEXT:    nill %r2, 65520
+; SW-NEXT:    stg %r3, -8(%r1)
+; SW-NEXT:    brasl %r14, notdead at PLT
+; SW-NEXT:    lmg %r11, %r15, 248(%r11)
+; SW-NEXT:    br %r14
+  %cmp = icmp eq i64 %n, 0
+  br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+  %addr = alloca i8, i64 %n, align 16
+  call void @notdead(ptr %addr)
+  br label %if.end
+
+if.end:
+  ret void
+}
+
+define void @test_mcount(i64 %n) nounwind "systemz-instrument-function-entry"="mcount" {
+; CHECK-LABEL: test_mcount:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    brasl %r0, mcount at PLT
+; Z-LABEL: test_mcount:
+; Z:       # %bb.0:
+; Z-NEXT:    stg %r14, 8(%r15)
+; Z-NEXT:    brasl %r14, mcount at PLT
+; Z-NEXT:    lg %r14, 8(%r15)
+; Z-NEXT:    stmg %r11, %r15, 88(%r15)
+; Z-NEXT:    aghi %r15, -160
+; Z-NEXT:    lgr %r11, %r15
+; Z-NEXT:    cgije %r2, 0, .LBB3_2
+; Z-NEXT:  # %bb.1: # %if.then
+; Z-NEXT:    lgr %r1, %r15
+; Z-NEXT:    la %r0, 7(%r2)
+; Z-NEXT:    nill %r0, 65528
+; Z-NEXT:    sgr %r1, %r0
+; Z-NEXT:    la %r2, 160(%r1)
+; Z-NEXT:    nill %r2, 65520
+; Z-NEXT:    lay %r15, -8(%r1)
+; Z-NEXT:    brasl %r14, notdead at PLT
+; Z-NEXT:  .LBB3_2: # %if.end
+; Z-NEXT:    lmg %r11, %r15, 248(%r11)
+; Z-NEXT:    br %r14
+;
+; SW-LABEL: test_mcount:
+; SW:       # %bb.0:
+; SW-NEXT:    cgibe %r2, 0, 0(%r14)
+; SW-NEXT:  .LBB3_1: # %if.then
+; SW-NEXT:    stg %r14, 8(%r15)
+; SW-NEXT:    brasl %r14, mcount at PLT
+; SW-NEXT:    lg %r14, 8(%r15)
+; SW-NEXT:    stmg %r11, %r15, 88(%r15)
+; SW-NEXT:    aghi %r15, -160
+; SW-NEXT:    lgr %r11, %r15
+; SW-NEXT:    lgr %r1, %r15
+; SW-NEXT:    la %r0, 7(%r2)
+; SW-NEXT:    nill %r0, 65528
+; SW-NEXT:    sgr %r1, %r0
+; SW-NEXT:    la %r2, 160(%r1)
+; SW-NEXT:    nill %r2, 65520
+; SW-NEXT:    lay %r15, -8(%r1)
+; SW-NEXT:    brasl %r14, notdead at PLT
+; SW-NEXT:    lmg %r11, %r15, 248(%r11)
+; SW-NEXT:    br %r14
+  %cmp = icmp eq i64 %n, 0
+  br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+  %addr = alloca i8, i64 %n, align 16
+  call void @notdead(ptr %addr)
+  br label %if.end
+
+if.end:
+  ret void
+}
+
+declare ghccc void @ghc_callee(i64)
+
+; to test that GHC functions disable shrink-wrapping without triggering the fatal
+; error in SystemZELFFrameLowering::emitPrologue. This test writes a standard
+; GHC function that calls another GHC function conditionally
+define ghccc void @test_ghc(i64 %n) nounwind {
+; CHECK-LABEL: test_ghc:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    cgije %r2, 0, .LBB3_2
+; CHECK-NEXT:  # %bb.1: # %if.then
+; CHECK-NEXT:    brasl %r14, ghc_callee
+; CHECK-NEXT:  .LBB3_2: # %if.end
+; CHECK-NEXT:    br %r14
+; Z-LABEL: test_ghc:
+; Z:       # %bb.0:
+; Z-NEXT:    cghi %r7, 0
+; Z-NEXT:    jglh ghc_callee at PLT
+; Z-NEXT:  .LBB4_1: # %if.end
+; Z-NEXT:    br %r14
+;
+; SW-LABEL: test_ghc:
+; SW:       # %bb.0:
+; SW-NEXT:    cghi %r7, 0
+; SW-NEXT:    jglh ghc_callee at PLT
+; SW-NEXT:  .LBB4_1: # %if.end
+; SW-NEXT:    br %r14
+  %cmp = icmp eq i64 %n, 0
+  br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+  tail call ghccc void @ghc_callee(i64 %n)
+  br label %if.end
+
+if.end:
+  ret void
+}
+
+
+
+define void @test_stack_probe(i64 %n) nounwind "probe-stack"="inline-asm" {
+; CHECK-LABEL: test_stack_probe:
+; CHECK:       # %bb.0:
+; Z-LABEL: test_stack_probe:
+; Z:       # %bb.0:
+; Z-NEXT:    stmg %r11, %r15, 88(%r15)
+; Z-NEXT:    aghi %r15, -160
+; Z-NEXT:    lgr %r11, %r15
+; Z-NEXT:    cgije %r2, 0, .LBB5_6
+; Z-NEXT:  # %bb.1: # %if.then
+; Z-NEXT:    lghi %r1, 8200
+; Z-NEXT:    clgfi %r1, 4096
+; Z-NEXT:    jl .LBB5_3
+; Z-NEXT:  .LBB5_2: # %if.then
+; Z-NEXT:    # =>This Inner Loop Header: Depth=1
+; Z-NEXT:    slgfi %r1, 4096
+; Z-NEXT:    slgfi %r15, 4096
+; Z-NEXT:    cg %r15, 4088(%r15)
+; Z-NEXT:    clgfi %r1, 4096
+; Z-NEXT:    jhe .LBB5_2
+; Z-NEXT:  .LBB5_3: # %if.then
+; Z-NEXT:    cgije %r1, 0, .LBB5_5
+; Z-NEXT:  # %bb.4: # %if.then
+; Z-NEXT:    slgr %r15, %r1
+; Z-NEXT:    cg %r15, -8(%r1,%r15)
+; Z-NEXT:  .LBB5_5: # %if.then
+; Z-NEXT:    la %r2, 168(%r15)
+; Z-NEXT:    nill %r2, 65520
+; Z-NEXT:    brasl %r14, notdead at PLT
+; Z-NEXT:  .LBB5_6: # %if.end
+; Z-NEXT:    lmg %r11, %r15, 248(%r11)
+; Z-NEXT:    br %r14
+;
+; SW-LABEL: test_stack_probe:
+; SW:       # %bb.0:
+; SW-NEXT:    cgibe %r2, 0, 0(%r14)
+; SW-NEXT:  .LBB5_1: # %if.then
+; SW-NEXT:    stmg %r11, %r15, 88(%r15)
+; SW-NEXT:    aghi %r15, -160
+; SW-NEXT:    lgr %r11, %r15
+; SW-NEXT:    lghi %r1, 8200
+; SW-NEXT:    clgfi %r1, 4096
+; SW-NEXT:    jl .LBB5_3
+; SW-NEXT:  .LBB5_2: # %if.then
+; SW-NEXT:    # =>This Inner Loop Header: Depth=1
+; SW-NEXT:    slgfi %r1, 4096
+; SW-NEXT:    slgfi %r15, 4096
+; SW-NEXT:    cg %r15, 4088(%r15)
+; SW-NEXT:    clgfi %r1, 4096
+; SW-NEXT:    jhe .LBB5_2
+; SW-NEXT:  .LBB5_3: # %if.then
+; SW-NEXT:    cgije %r1, 0, .LBB5_5
+; SW-NEXT:  # %bb.4: # %if.then
+; SW-NEXT:    slgr %r15, %r1
+; SW-NEXT:    cg %r15, -8(%r1,%r15)
+; SW-NEXT:  .LBB5_5: # %if.then
+; SW-NEXT:    la %r2, 168(%r15)
+; SW-NEXT:    nill %r2, 65520
+; SW-NEXT:    brasl %r14, notdead at PLT
+; SW-NEXT:    lmg %r11, %r15, 248(%r11)
+; SW-NEXT:    br %r14
+  %cmp = icmp eq i64 %n, 0
+  br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+  %addr = alloca i8, i64 8192, align 16
+  call void @notdead(ptr %addr)
+  br label %if.end
+
+if.end:
+  ret void
+}



More information about the llvm-commits mailing list