[llvm] [SystemZ] Improve handling of memmoves. (PR #196285)

via llvm-commits llvm-commits at lists.llvm.org
Thu May 7 04:07:28 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-systemz

Author: Jonas Paulsson (JonPsson1)

<details>
<summary>Changes</summary>

Handle memmoves with constant-length up to 256 bytes with target instructions instead of libcall:

- Extend TargetLowering:MemOp so that it is possible to distinguish between a Memcpy and Memmove.
- Use loads/stores up to 16 or 32 bytes (under evaluation - see below).
- Use VLL/VSTL up to 15 bytes.
- Use MVC/MVCRL for anything else up to 256 bytes.

Patch still has experimental options as next step is to do benchmarking and see if there is any setting that gives
any improvement compared to that which seems to be GCCs default (max 1 load/store).

A slight annoyance is that FrameObjects also need to be handled, although it is not currently possible to detect theses cases and treat them as Memcpys. As it is, this ignores this optimization and simply loads the FI address into a register to do the compare before MVC/MVCRL.

@<!-- -->gchatelet Are the changes to MemOp acceptable?

---

Patch is 67.42 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/196285.diff


9 Files Affected:

- (modified) llvm/include/llvm/CodeGen/TargetLowering.h (+8-4) 
- (modified) llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp (+2-1) 
- (modified) llvm/lib/Target/SystemZ/SystemZISelLowering.cpp (+98-1) 
- (modified) llvm/lib/Target/SystemZ/SystemZISelLowering.h (+2) 
- (modified) llvm/lib/Target/SystemZ/SystemZInstrInfo.td (+7) 
- (modified) llvm/lib/Target/SystemZ/SystemZOperators.td (+3) 
- (modified) llvm/lib/Target/SystemZ/SystemZSelectionDAGInfo.cpp (+17) 
- (modified) llvm/lib/Target/SystemZ/SystemZSelectionDAGInfo.h (+7) 
- (modified) llvm/test/CodeGen/SystemZ/memmove-01.ll (+434-511) 


``````````diff
diff --git a/llvm/include/llvm/CodeGen/TargetLowering.h b/llvm/include/llvm/CodeGen/TargetLowering.h
index 9b0bfa111f2d5..8257af3da9397 100644
--- a/llvm/include/llvm/CodeGen/TargetLowering.h
+++ b/llvm/include/llvm/CodeGen/TargetLowering.h
@@ -131,12 +131,14 @@ struct MemOp {
   // memcpy only
   bool MemcpyStrSrc; // Indicates whether the memcpy source is an in-register
                      // constant so it does not need to be loaded.
-  Align SrcAlign;    // Inferred alignment of the source or default value if the
-                     // memory operation does not need to load the value.
+  bool SrcDstMayOverlap; // True if the source and destination memory regions
+                         // may overlap (memmove).
+  Align SrcAlign; // Inferred alignment of the source or default value if the
+                  // memory operation does not need to load the value.
 public:
   static MemOp Copy(uint64_t Size, bool DstAlignCanChange, Align DstAlign,
-                    Align SrcAlign, bool IsVolatile,
-                    bool MemcpyStrSrc = false) {
+                    Align SrcAlign, bool IsVolatile, bool MemcpyStrSrc = false,
+                    bool SrcDstMayOverlap = false) {
     MemOp Op;
     Op.Size = Size;
     Op.DstAlignCanChange = DstAlignCanChange;
@@ -144,6 +146,7 @@ struct MemOp {
     Op.AllowOverlap = !IsVolatile;
     Op.IsMemset = false;
     Op.ZeroMemset = false;
+    Op.SrcDstMayOverlap = SrcDstMayOverlap;
     Op.MemcpyStrSrc = MemcpyStrSrc;
     Op.SrcAlign = SrcAlign;
     return Op;
@@ -171,6 +174,7 @@ struct MemOp {
   bool allowOverlap() const { return AllowOverlap; }
   bool isMemset() const { return IsMemset; }
   bool isMemcpy() const { return !IsMemset; }
+  bool isMemmove() const { return isMemcpy() && SrcDstMayOverlap; }
   bool isMemcpyWithFixedDstAlign() const {
     return isMemcpy() && !DstAlignCanChange;
   }
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
index 857ff98f84b32..e469a5cefe46f 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
@@ -9450,7 +9450,8 @@ static SDValue getMemmoveLoadsAndStores(SelectionDAG &DAG, const SDLoc &dl,
   unsigned Limit = AlwaysInline ? ~0U : TLI.getMaxStoresPerMemmove(OptSize);
   if (!TLI.findOptimalMemOpLowering(
           C, MemOps, Limit,
-          MemOp::Copy(Size, DstAlignCanChange, Alignment, *SrcAlign, isVol),
+          MemOp::Copy(Size, DstAlignCanChange, Alignment, *SrcAlign, isVol,
+                      /*MemcpyStrSrc*/ false, /*SrcDstMayOverlap*/ true),
           DstPtrInfo.getAddrSpace(), SrcPtrInfo.getAddrSpace(),
           MF.getFunction().getAttributes(), nullptr))
     return SDValue();
diff --git a/llvm/lib/Target/SystemZ/SystemZISelLowering.cpp b/llvm/lib/Target/SystemZ/SystemZISelLowering.cpp
index f1f6872b68fbe..7a6dc3fe4acdf 100644
--- a/llvm/lib/Target/SystemZ/SystemZISelLowering.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZISelLowering.cpp
@@ -43,6 +43,10 @@ static cl::opt<bool> EnableIntArgExtCheck(
     cl::desc("Verify that narrow int args are properly extended per the "
              "SystemZ ABI."));
 
+// EXPERIMENTAL
+static cl::opt<unsigned> MEMMOVESTORES("memmove-stores", cl::init(1));
+static cl::opt<bool> MEMMOVEVLL("memmove-vll", cl::init(true));
+
 namespace {
 // Represents information about a comparison.
 struct Comparison {
@@ -815,7 +819,7 @@ SystemZTargetLowering::SystemZTargetLowering(const TargetMachine &TM,
   MaxStoresPerMemcpyOptSize = 0;
 
   // Same with memmove.
-  MaxStoresPerMemmove = Subtarget.hasVector() ? 2 : 0;
+  MaxStoresPerMemmove = Subtarget.hasVector() ? MEMMOVESTORES : 0;
   MaxStoresPerMemmoveOptSize = 0;
 
   // The main memset sequence is a byte store followed by an MVC.
@@ -1466,6 +1470,14 @@ bool SystemZTargetLowering::findOptimalMemOpLowering(
   assert(Limit != ~0U &&
          "Expected EmitTargetCodeForMemXXX() to handle AlwaysInline cases.");
 
+  if (Op.isMemmove()) {
+    if (Op.size() >= 16 &&
+        (!Op.isAligned(Align(8)) || (Op.size() >= 25 && Op.size() <= 31)))
+      return false;
+    return TargetLowering::findOptimalMemOpLowering(
+        Context, MemOps, Limit, Op, DstAS, SrcAS, FuncAttributes, LargestVT);
+  }
+
   if (Op.isZeroMemset())
     return false; // Memset zero: Use XC.
 
@@ -10820,6 +10832,89 @@ SystemZTargetLowering::emitMemMemWrapper(MachineInstr &MI,
   return MBB;
 }
 
+MachineBasicBlock *
+SystemZTargetLowering::emitMemmoveImm(MachineInstr &MI,
+                                      MachineBasicBlock *MBB) const {
+  MachineFunction &MF = *MBB->getParent();
+  const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
+  MachineRegisterInfo &MRI = MF.getRegInfo();
+
+  DebugLoc DL = MI.getDebugLoc();
+  MachineOperand DstBase = earlyUseOperand(MI.getOperand(0));
+  uint64_t DstDisp = MI.getOperand(1).getImm();
+  MachineOperand SrcBase = earlyUseOperand(MI.getOperand(2));
+  uint64_t SrcDisp = MI.getOperand(3).getImm();
+  uint64_t Len = MI.getOperand(4).getImm();
+  assert(Len >= 1 && Len <= 256 &&
+         "Memmove of of unsupported constant length.");
+  assert(isUInt<12>(DstDisp) && isUInt<12>(SrcDisp) &&
+         "Unexpected large displacement.");
+
+  // Fold any displacement (or frame index reference) into a new register.
+  auto foldAddressIfNeeded = [&](MachineOperand &Base, uint64_t &Disp) -> void {
+    if (Disp || Base.isFI()) {
+      Register Reg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
+      unsigned Opcode = TII->getOpcodeForOffset(SystemZ::LA, Disp);
+      BuildMI(*MBB, MI, DL, TII->get(Opcode), Reg)
+          .add(Base).addImm(Disp).addReg(0);
+      Base = MachineOperand::CreateReg(Reg, false);
+      Disp = 0;
+    }
+  };
+
+  if (Len <= 15 && MEMMOVEVLL) {
+    Register HighByteReg = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
+    BuildMI(*MBB, MI, DL, TII->get(SystemZ::LHI), HighByteReg).addImm(Len - 1);
+
+    Register VecReg = MRI.createVirtualRegister(&SystemZ::VR128BitRegClass);
+    BuildMI(*MBB, MI, DL, TII->get(SystemZ::VLL), VecReg)
+        .addReg(HighByteReg)
+        .add(SrcBase).addImm(SrcDisp);
+
+    BuildMI(*MBB, MI, DL, TII->get(SystemZ::VSTL))
+        .addReg(VecReg)
+        .addReg(HighByteReg)
+        .add(DstBase).addImm(DstDisp);
+
+    MI.eraseFromParent();
+    return MBB;
+  }
+
+  // Use MVC or MVCRL after comparing the addresses.
+  MachineBasicBlock *DoneMBB = SystemZ::splitBlockAfter(MI, MBB);
+  MachineBasicBlock *MvcMBB = SystemZ::emitBlockAfter(MBB);
+  MachineBasicBlock *MvcrlMBB = SystemZ::emitBlockAfter(MvcMBB);
+  MBB->addSuccessor(MvcMBB);
+  MBB->addSuccessor(MvcrlMBB);
+  MvcMBB->addSuccessor(DoneMBB);
+  MvcrlMBB->addSuccessor(DoneMBB);
+
+  // Fold any displacements in order to do the compare.
+  foldAddressIfNeeded(SrcBase, SrcDisp);
+  foldAddressIfNeeded(DstBase, DstDisp);
+
+  BuildMI(MBB, DL, TII->get(SystemZ::CLGR)).add(SrcBase).add(DstBase);
+  BuildMI(MBB, DL, TII->get(SystemZ::BRC))
+      .addImm(SystemZ::CCMASK_ICMP).addImm(SystemZ::CCMASK_CMP_LT)
+      .addMBB(MvcrlMBB);
+
+  BuildMI(MvcMBB, DL, TII->get(SystemZ::MVC))
+      .add(DstBase).addImm(DstDisp)
+      .addImm(Len)
+      .add(SrcBase).addImm(SrcDisp)
+      .setMemRefs(MI.memoperands());
+  BuildMI(MvcMBB, DL, TII->get(SystemZ::J)).addMBB(DoneMBB);
+
+  BuildMI(MvcrlMBB, DL, TII->get(SystemZ::LHI), SystemZ::R0L).addImm(Len - 1);
+  BuildMI(MvcrlMBB, DL, TII->get(SystemZ::MVCRL))
+      .add(DstBase).addImm(DstDisp)
+      .add(SrcBase).addImm(SrcDisp)
+      .setMemRefs(MI.memoperands());
+
+  MI.eraseFromParent();
+  return DoneMBB;
+}
+
 // Decompose string pseudo-instruction MI into a loop that continually performs
 // Opcode until CC != 3.
 MachineBasicBlock *SystemZTargetLowering::emitStringWrapper(
@@ -11175,6 +11270,8 @@ MachineBasicBlock *SystemZTargetLowering::EmitInstrWithCustomInserter(
   case SystemZ::MemsetRegImm:
   case SystemZ::MemsetRegReg:
     return emitMemMemWrapper(MI, MBB, SystemZ::MVC, true/*IsMemset*/);
+  case SystemZ::MemmoveImm:
+    return emitMemmoveImm(MI, MBB);
   case SystemZ::CLSTLoop:
     return emitStringWrapper(MI, MBB, SystemZ::CLST);
   case SystemZ::MVSTLoop:
diff --git a/llvm/lib/Target/SystemZ/SystemZISelLowering.h b/llvm/lib/Target/SystemZ/SystemZISelLowering.h
index bb3eeba6446d2..374fb58416fa0 100644
--- a/llvm/lib/Target/SystemZ/SystemZISelLowering.h
+++ b/llvm/lib/Target/SystemZ/SystemZISelLowering.h
@@ -463,6 +463,8 @@ class SystemZTargetLowering : public TargetLowering {
   MachineBasicBlock *emitMemMemWrapper(MachineInstr &MI, MachineBasicBlock *BB,
                                        unsigned Opcode,
                                        bool IsMemset = false) const;
+  MachineBasicBlock *emitMemmoveImm(MachineInstr &MI,
+                                    MachineBasicBlock *BB) const;
   MachineBasicBlock *emitStringWrapper(MachineInstr &MI, MachineBasicBlock *BB,
                                        unsigned Opcode) const;
   MachineBasicBlock *emitTransactionBegin(MachineInstr &MI,
diff --git a/llvm/lib/Target/SystemZ/SystemZInstrInfo.td b/llvm/lib/Target/SystemZ/SystemZInstrInfo.td
index 35a923d070e3e..d7b9a1496c8eb 100644
--- a/llvm/lib/Target/SystemZ/SystemZInstrInfo.td
+++ b/llvm/lib/Target/SystemZ/SystemZInstrInfo.td
@@ -553,6 +553,13 @@ let Predicates = [FeatureMiscellaneousExtensions3],
     mayLoad = 1, mayStore = 1, Uses = [R0L] in
   def MVCRL : SideEffectBinarySSE<"mvcrl", 0xE50A>;
 
+let usesCustomInserter = 1, hasNoSchedulingInfo = 1, mayLoad = 1, mayStore = 1 in
+  def MemmoveImm : Pseudo<(outs),
+                          (ins bdaddr12only:$dest, bdaddr12only:$src,
+                               imm64:$length),
+                          [(z_memmove bdaddr12only:$dest, bdaddr12only:$src,
+                                      imm64:$length)]>;
+
 // String moves.
 let mayLoad = 1, mayStore = 1, Defs = [CC] in
   defm MVST : StringRRE<"mvst", 0xB255, z_stpcpy>;
diff --git a/llvm/lib/Target/SystemZ/SystemZOperators.td b/llvm/lib/Target/SystemZ/SystemZOperators.td
index 758445e2a566d..628c3ebb09a7e 100644
--- a/llvm/lib/Target/SystemZ/SystemZOperators.td
+++ b/llvm/lib/Target/SystemZ/SystemZOperators.td
@@ -665,6 +665,9 @@ def z_atomic_cmp_swap_128 : SDNode<"SystemZISD::ATOMIC_CMP_SWAP_128",
 def z_mvc               : SDNode<"SystemZISD::MVC", SDT_ZMemMemLength,
                                  [SDNPHasChain, SDNPMayStore, SDNPMayLoad]>;
 
+def z_memmove           : SDNode<"SystemZISD::MEMMOVE", SDT_ZMemMemLength,
+                                 [SDNPHasChain, SDNPMayStore, SDNPMayLoad]>;
+
 // Similar to MVC, but for logic operations (AND, OR, XOR).
 def z_nc                : SDNode<"SystemZISD::NC", SDT_ZMemMemLength,
                                   [SDNPHasChain, SDNPMayStore, SDNPMayLoad]>;
diff --git a/llvm/lib/Target/SystemZ/SystemZSelectionDAGInfo.cpp b/llvm/lib/Target/SystemZ/SystemZSelectionDAGInfo.cpp
index eec37a9df386f..01bbd5024e409 100644
--- a/llvm/lib/Target/SystemZ/SystemZSelectionDAGInfo.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZSelectionDAGInfo.cpp
@@ -87,6 +87,23 @@ SDValue SystemZSelectionDAGInfo::EmitTargetCodeForMemcpy(
   return emitMemMemReg(DAG, DL, SystemZISD::MVC, Chain, Dst, Src, Size);
 }
 
+SDValue SystemZSelectionDAGInfo::EmitTargetCodeForMemmove(
+    SelectionDAG &DAG, const SDLoc &DL, SDValue Chain, SDValue Dst, SDValue Src,
+    SDValue Size, Align Alignment, bool IsVolatile,
+    MachinePointerInfo DstPtrInfo, MachinePointerInfo SrcPtrInfo) const {
+  if (IsVolatile)
+    return SDValue();
+
+  // XXX VLL FeatureVector
+  // XXX MVCRL FeatureMiscellaneousExtensions3
+  if (auto *CSize = dyn_cast<ConstantSDNode>(Size))
+    if (CSize->getZExtValue() <= 256)
+      return DAG.getNode(SystemZISD::MEMMOVE, DL, MVT::Other,
+                         {Chain, Dst, Src, Size});
+
+  return SDValue();
+}
+
 // Handle a memset of 1, 2, 4 or 8 bytes with the operands given by
 // Chain, Dst, ByteVal and Size.  These cases are expected to use
 // MVI, MVHHI, MVHI and MVGHI respectively.
diff --git a/llvm/lib/Target/SystemZ/SystemZSelectionDAGInfo.h b/llvm/lib/Target/SystemZ/SystemZSelectionDAGInfo.h
index 8e6da4fe8b0ae..96286d0c192f5 100644
--- a/llvm/lib/Target/SystemZ/SystemZSelectionDAGInfo.h
+++ b/llvm/lib/Target/SystemZ/SystemZSelectionDAGInfo.h
@@ -51,6 +51,13 @@ class SystemZSelectionDAGInfo : public SelectionDAGGenTargetInfo {
                                   MachinePointerInfo DstPtrInfo,
                                   MachinePointerInfo SrcPtrInfo) const override;
 
+  SDValue EmitTargetCodeForMemmove(SelectionDAG &DAG, const SDLoc &DL,
+                                   SDValue Chain, SDValue Dst, SDValue Src,
+                                   SDValue Size, Align Alignment,
+                                   bool IsVolatile,
+                                   MachinePointerInfo DstPtrInfo,
+                                   MachinePointerInfo SrcPtrInfo) const override;
+
   SDValue EmitTargetCodeForMemset(SelectionDAG &DAG, const SDLoc &DL,
                                   SDValue Chain, SDValue Dst, SDValue Byte,
                                   SDValue Size, Align Alignment,
diff --git a/llvm/test/CodeGen/SystemZ/memmove-01.ll b/llvm/test/CodeGen/SystemZ/memmove-01.ll
index 62459ccf487c8..653033c31a6bf 100644
--- a/llvm/test/CodeGen/SystemZ/memmove-01.ll
+++ b/llvm/test/CodeGen/SystemZ/memmove-01.ll
@@ -1,5 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mcpu=z17 < %s -mtriple=s390x-linux-gnu | FileCheck %s
+; RUN: llc -mcpu=z17 < %s -mtriple=s390x-linux-gnu -verify-machineinstrs \
+; RUN:   | FileCheck %s
 ;
 ; Test non-volatile memmoves of small constant lengths in both aligned and
 ; unaligned cases.
@@ -9,14 +10,8 @@ declare void @llvm.memmove.p0.p0.i64(ptr nocapture, ptr nocapture, i64, i1) noun
 define void @fun1(ptr %Dst, ptr %Src) {
 ; CHECK-LABEL: fun1:
 ; CHECK:       # %bb.0:
-; CHECK-NEXT:    stmg %r14, %r15, 112(%r15)
-; CHECK-NEXT:    .cfi_offset %r14, -48
-; CHECK-NEXT:    .cfi_offset %r15, -40
-; CHECK-NEXT:    aghi %r15, -160
-; CHECK-NEXT:    .cfi_def_cfa_offset 320
-; CHECK-NEXT:    lghi %r4, 1
-; CHECK-NEXT:    brasl %r14, memmove at PLT
-; CHECK-NEXT:    lmg %r14, %r15, 272(%r15)
+; CHECK-NEXT:    lb %r0, 0(%r3)
+; CHECK-NEXT:    stc %r0, 0(%r2)
 ; CHECK-NEXT:    br %r14
   call void @llvm.memmove.p0.p0.i64(ptr align 8 %Dst, ptr align 8 %Src, i64 1, i1 false)
   ret void
@@ -25,14 +20,8 @@ define void @fun1(ptr %Dst, ptr %Src) {
 define void @fun1_unaligned(ptr %Dst, ptr %Src) {
 ; CHECK-LABEL: fun1_unaligned:
 ; CHECK:       # %bb.0:
-; CHECK-NEXT:    stmg %r14, %r15, 112(%r15)
-; CHECK-NEXT:    .cfi_offset %r14, -48
-; CHECK-NEXT:    .cfi_offset %r15, -40
-; CHECK-NEXT:    aghi %r15, -160
-; CHECK-NEXT:    .cfi_def_cfa_offset 320
-; CHECK-NEXT:    lghi %r4, 1
-; CHECK-NEXT:    brasl %r14, memmove at PLT
-; CHECK-NEXT:    lmg %r14, %r15, 272(%r15)
+; CHECK-NEXT:    lb %r0, 0(%r3)
+; CHECK-NEXT:    stc %r0, 0(%r2)
 ; CHECK-NEXT:    br %r14
   call void @llvm.memmove.p0.p0.i64(ptr align 1 %Dst, ptr align 1 %Src, i64 1, i1 false)
   ret void
@@ -41,14 +30,8 @@ define void @fun1_unaligned(ptr %Dst, ptr %Src) {
 define void @fun2(ptr %Dst, ptr %Src) {
 ; CHECK-LABEL: fun2:
 ; CHECK:       # %bb.0:
-; CHECK-NEXT:    stmg %r14, %r15, 112(%r15)
-; CHECK-NEXT:    .cfi_offset %r14, -48
-; CHECK-NEXT:    .cfi_offset %r15, -40
-; CHECK-NEXT:    aghi %r15, -160
-; CHECK-NEXT:    .cfi_def_cfa_offset 320
-; CHECK-NEXT:    lghi %r4, 2
-; CHECK-NEXT:    brasl %r14, memmove at PLT
-; CHECK-NEXT:    lmg %r14, %r15, 272(%r15)
+; CHECK-NEXT:    lh %r0, 0(%r3)
+; CHECK-NEXT:    sth %r0, 0(%r2)
 ; CHECK-NEXT:    br %r14
   call void @llvm.memmove.p0.p0.i64(ptr align 8 %Dst, ptr align 8 %Src, i64 2, i1 false)
   ret void
@@ -57,14 +40,8 @@ define void @fun2(ptr %Dst, ptr %Src) {
 define void @fun2_unaligned(ptr %Dst, ptr %Src) {
 ; CHECK-LABEL: fun2_unaligned:
 ; CHECK:       # %bb.0:
-; CHECK-NEXT:    stmg %r14, %r15, 112(%r15)
-; CHECK-NEXT:    .cfi_offset %r14, -48
-; CHECK-NEXT:    .cfi_offset %r15, -40
-; CHECK-NEXT:    aghi %r15, -160
-; CHECK-NEXT:    .cfi_def_cfa_offset 320
-; CHECK-NEXT:    lghi %r4, 2
-; CHECK-NEXT:    brasl %r14, memmove at PLT
-; CHECK-NEXT:    lmg %r14, %r15, 272(%r15)
+; CHECK-NEXT:    lh %r0, 0(%r3)
+; CHECK-NEXT:    sth %r0, 0(%r2)
 ; CHECK-NEXT:    br %r14
   call void @llvm.memmove.p0.p0.i64(ptr align 1 %Dst, ptr align 1 %Src, i64 2, i1 false)
   ret void
@@ -73,14 +50,9 @@ define void @fun2_unaligned(ptr %Dst, ptr %Src) {
 define void @fun3(ptr %Dst, ptr %Src) {
 ; CHECK-LABEL: fun3:
 ; CHECK:       # %bb.0:
-; CHECK-NEXT:    stmg %r14, %r15, 112(%r15)
-; CHECK-NEXT:    .cfi_offset %r14, -48
-; CHECK-NEXT:    .cfi_offset %r15, -40
-; CHECK-NEXT:    aghi %r15, -160
-; CHECK-NEXT:    .cfi_def_cfa_offset 320
-; CHECK-NEXT:    lghi %r4, 3
-; CHECK-NEXT:    brasl %r14, memmove at PLT
-; CHECK-NEXT:    lmg %r14, %r15, 272(%r15)
+; CHECK-NEXT:    lhi %r0, 2
+; CHECK-NEXT:    vll %v0, %r0, 0(%r3)
+; CHECK-NEXT:    vstl %v0, %r0, 0(%r2)
 ; CHECK-NEXT:    br %r14
   call void @llvm.memmove.p0.p0.i64(ptr align 8 %Dst, ptr align 8 %Src, i64 3, i1 false)
   ret void
@@ -89,14 +61,9 @@ define void @fun3(ptr %Dst, ptr %Src) {
 define void @fun3_unaligned(ptr %Dst, ptr %Src) {
 ; CHECK-LABEL: fun3_unaligned:
 ; CHECK:       # %bb.0:
-; CHECK-NEXT:    stmg %r14, %r15, 112(%r15)
-; CHECK-NEXT:    .cfi_offset %r14, -48
-; CHECK-NEXT:    .cfi_offset %r15, -40
-; CHECK-NEXT:    aghi %r15, -160
-; CHECK-NEXT:    .cfi_def_cfa_offset 320
-; CHECK-NEXT:    lghi %r4, 3
-; CHECK-NEXT:    brasl %r14, memmove at PLT
-; CHECK-NEXT:    lmg %r14, %r15, 272(%r15)
+; CHECK-NEXT:    lhi %r0, 2
+; CHECK-NEXT:    vll %v0, %r0, 0(%r3)
+; CHECK-NEXT:    vstl %v0, %r0, 0(%r2)
 ; CHECK-NEXT:    br %r14
   call void @llvm.memmove.p0.p0.i64(ptr align 1 %Dst, ptr align 1 %Src, i64 3, i1 false)
   ret void
@@ -105,14 +72,8 @@ define void @fun3_unaligned(ptr %Dst, ptr %Src) {
 define void @fun4(ptr %Dst, ptr %Src) {
 ; CHECK-LABEL: fun4:
 ; CHECK:       # %bb.0:
-; CHECK-NEXT:    stmg %r14, %r15, 112(%r15)
-; CHECK-NEXT:    .cfi_offset %r14, -48
-; CHECK-NEXT:    .cfi_offset %r15, -40
-; CHECK-NEXT:    aghi %r15, -160
-; CHECK-NEXT:    .cfi_def_cfa_offset 320
-; CHECK-NEXT:    lghi %r4, 4
-; CHECK-NEXT:    brasl %r14, memmove at PLT
-; CHECK-NEXT:    lmg %r14, %r15, 272(%r15)
+; CHECK-NEXT:    l %r0, 0(%r3)
+; CHECK-NEXT:    st %r0, 0(%r2)
 ; CHECK-NEXT:    br %r14
   call void @llvm.memmove.p0.p0.i64(ptr align 8 %Dst, ptr align 8 %Src, i64 4, i1 false)
   ret void
@@ -121,14 +82,8 @@ define void @fun4(ptr %Dst, ptr %Src) {
 define void @fun4_unaligned(ptr %Dst, ptr %Src) {
 ; CHECK-LABEL: fun4_unaligned:
 ; CHECK:       # %bb.0:
-; CHECK-NEXT:    stmg %r14, %r15, 112(%r15)
-; CHECK-NEXT:    .cfi_offset %r14, -48
-; CHECK-NEXT:    .cfi_offset %r15, -40
-; CHECK-NEXT:    aghi %r15, -160
-; CHECK-NEXT:    .cfi_def_cfa_offset 320
-; CHECK-NEXT:    lghi %r4, 4
-; CHECK-NEXT:    brasl %r14, memmove at PLT
-; CHECK-NEXT:    lmg %r14, %r15, 272(%r15)
+; CHECK-NEXT:    l %r0, 0(%r3)
+; CHECK-NEXT:    st %r0, 0(%r2)
 ; CHECK-NEXT:    br %r14
   call void @llvm.memmove.p0.p0.i64(ptr align 1 %Dst, ptr align 1 %Src, i64 4, i1 false)
   ret void
@@ -137,14 +92,9 @@ define void @fun4_unaligned(ptr %Dst, ptr %Src) {
 define void @fun5(ptr %Dst, ptr %Src) {
 ; CHECK-LABEL: fun5:
 ; CHECK:       # %bb.0:
-; CHECK-NEXT:    stmg %r14, %r15, 112(%r15)
-; CHECK-NEXT:    .cfi_offset %r14, -48
-; CHECK-NEXT:    .cfi_offset %r15, -40
-; CHECK-NEXT:    aghi %r15, -160
-; CHECK-NEXT:    .cfi_def_cfa_offset 320
-; CHECK-NEXT:    lghi %r4, 5
-; CHECK-NEXT:    brasl %r14, memmove at PLT
-; CHECK-NEXT:    lmg %r14, %r15, 272(%r15)
+; CHECK-NEXT:    lhi %r0, 4
+; CHECK-NEXT:    vll %v0, %r0, 0(%r3)
+; CHECK-NEXT:    vstl %v0, %r0, 0(%r2)
 ; CHECK-NEXT:    br %r14
   call void @llvm.memmove.p0.p0.i64(ptr align 8 %Dst, ptr align 8 %Src, i64 5, i1 false)
   ret void
@@ -153,14 +103,9 @@ define void @fun5(ptr %Dst, ptr %Src) {
 define void @fun5_unaligned(ptr %Dst, ptr %Src) {
 ; CHECK-LABEL: fun5_unaligned:
 ; CHECK:       # %bb.0:
-; CHECK-NEXT:    stmg %r14, %r15, 112(%r15)
-; CHECK-NEXT:    .cfi_offset %r14, -48
-; CHECK-NEXT:    .cfi_offset %r15, -40
-; CHECK-NEXT:    aghi %r15, -160
-; CHECK-NEXT:    .cfi_def_cfa_offset 320
-; CHECK-NEXT:    lghi %r4, 5
-; CHECK-NEXT:    brasl %r14, memmo...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/196285


More information about the llvm-commits mailing list