[llvm] [AArch64][SVE] Support copy of PPR2 register class in copyPhysReg (PR #216303)

Kerry McLaughlin via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 25 06:43:09 PDT 2026


https://github.com/kmclaughlin-arm updated https://github.com/llvm/llvm-project/pull/216303

>From 3bd0abb61eb5e74d22ee62d1e36f5ec466a1b9f3 Mon Sep 17 00:00:00 2001
From: Kerry McLaughlin <kerry.mclaughlin at arm.com>
Date: Fri, 14 Aug 2026 10:50:01 +0000
Subject: [PATCH 1/4] [AArch64][SVE] Support copy of PPR2 register class in
 copyPhysReg.

This fixes the following crash, which was observed after landing #209484:
https://github.com/llvm/llvm-project/pull/209484
---
 llvm/lib/Target/AArch64/AArch64InstrInfo.cpp  | 12 ++++++++++
 .../test/CodeGen/AArch64/sve-copy-pprpair.mir | 24 +++++++++++++++++++
 2 files changed, 36 insertions(+)
 create mode 100644 llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir

diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 522b263df5de1..42062c8021135 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -5773,6 +5773,8 @@ void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
     const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode));
     AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI);
     AddSubReg(MIB, SrcReg, Indices[SubReg], {}, TRI);
+    if (Opcode == AArch64::ORR_PPzPP)
+      AddSubReg(MIB, SrcReg, Indices[SubReg], {}, TRI);
     AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI);
   }
 }
@@ -5993,6 +5995,16 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
     return;
   }
 
+  if (AArch64::PPR2RegClass.contains(DestReg) &&
+      AArch64::PPR2RegClass.contains(SrcReg)) {
+    assert(Subtarget.isSVEorStreamingSVEAvailable() &&
+           "Unexpected SVE predicate register.");
+    static const unsigned Indices[] = {AArch64::psub0, AArch64::psub1};
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_PPzPP,
+                     Indices);
+    return;
+  }
+
   // Copy a Z register by ORRing with itself.
   if (AArch64::ZPRRegClass.contains(DestReg) &&
       AArch64::ZPRRegClass.contains(SrcReg)) {
diff --git a/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir b/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
new file mode 100644
index 0000000000000..336886f699b49
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
@@ -0,0 +1,24 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
+# RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve -run-pass=postrapseudos -simplify-mir -verify-machineinstrs %s -o - | FileCheck %s
+
+---
+name:            copy_ppr2
+alignment:       4
+tracksRegLiveness: true
+liveins:
+  - { reg: '$p0_p1' }
+frameInfo:
+  maxCallFrameSize: 0
+body:             |
+  bb.0:
+    liveins: $p0_p1
+    ; CHECK-LABEL: name: copy_ppr2
+    ; CHECK: liveins: $p0_p1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $p2 = ORR_PPzPP $p0, $p0, killed $p0
+    ; CHECK-NEXT: $p3 = ORR_PPzPP $p1, $p1, killed $p1
+    ; CHECK-NEXT: RET_ReallyLR
+    $p2_p3 = COPY killed renamable $p0_p1
+    RET_ReallyLR
+
+...

>From 3f14e68668c45ab8a6bf77ffe9e91bd9d181aaad Mon Sep 17 00:00:00 2001
From: Kerry McLaughlin <kerry.mclaughlin at arm.com>
Date: Fri, 14 Aug 2026 13:56:38 +0000
Subject: [PATCH 2/4] - Call copyPhysReg directly from copyPhysRegTuple -
 Changed forwardCopyWillClobberTuple to consider number of predicate registers
 - Added tests with overlap & test for forwardCopyWillClobberTuple

---
 llvm/lib/Target/AArch64/AArch64InstrInfo.cpp  | 51 ++++++++-----------
 llvm/lib/Target/AArch64/AArch64InstrInfo.h    |  2 +-
 llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll | 22 ++++----
 .../test/CodeGen/AArch64/sve-copy-pprpair.mir | 42 +++++++++++++++
 4 files changed, 74 insertions(+), 43 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 42062c8021135..3ffa080021e39 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -5744,38 +5744,36 @@ static const MachineInstrBuilder &AddSubReg(const MachineInstrBuilder &MIB,
 }
 
 static bool forwardCopyWillClobberTuple(unsigned DestReg, unsigned SrcReg,
-                                        unsigned NumRegs) {
-  // We really want the positive remainder mod 32 here, that happens to be
+                                        unsigned NumRegs, bool IsPred) {
+  // We really want the positive remainder mod 16/32 here, that happens to be
   // easily obtainable with a mask.
-  return ((DestReg - SrcReg) & 0x1f) < NumRegs;
+  unsigned MaxRegs = IsPred ? 0xf : 0x1f;
+  return ((DestReg - SrcReg) & MaxRegs) < NumRegs;
 }
 
 void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
                                         MachineBasicBlock::iterator I,
                                         const DebugLoc &DL, MCRegister DestReg,
                                         MCRegister SrcReg, bool KillSrc,
-                                        unsigned Opcode,
                                         ArrayRef<unsigned> Indices) const {
   assert(Subtarget.hasNEON() && "Unexpected register copy without NEON");
   const TargetRegisterInfo *TRI = &getRegisterInfo();
   uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
   uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
   unsigned NumRegs = Indices.size();
+  bool IsPred = AArch64::PPR2RegClass.contains(DestReg);
 
   int SubReg = 0, End = NumRegs, Incr = 1;
-  if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs)) {
+  if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs, IsPred)) {
     SubReg = NumRegs - 1;
     End = -1;
     Incr = -1;
   }
 
   for (; SubReg != End; SubReg += Incr) {
-    const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode));
-    AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI);
-    AddSubReg(MIB, SrcReg, Indices[SubReg], {}, TRI);
-    if (Opcode == AArch64::ORR_PPzPP)
-      AddSubReg(MIB, SrcReg, Indices[SubReg], {}, TRI);
-    AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI);
+    MCRegister DestSubReg = TRI->getSubReg(DestReg, Indices[SubReg]);
+    MCRegister SrcSubReg = TRI->getSubReg(SrcReg, Indices[SubReg]);
+    copyPhysReg(MBB, I, DL, DestSubReg, SrcSubReg, KillSrc);
   }
 }
 
@@ -5995,13 +5993,13 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
     return;
   }
 
+  // Copy a predicate register pair by copying the individual sub-registers.
   if (AArch64::PPR2RegClass.contains(DestReg) &&
       AArch64::PPR2RegClass.contains(SrcReg)) {
     assert(Subtarget.isSVEorStreamingSVEAvailable() &&
            "Unexpected SVE predicate register.");
     static const unsigned Indices[] = {AArch64::psub0, AArch64::psub1};
-    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_PPzPP,
-                     Indices);
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
     return;
   }
 
@@ -6024,8 +6022,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
     assert(Subtarget.isSVEorStreamingSVEAvailable() &&
            "Unexpected SVE register.");
     static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1};
-    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
-                     Indices);
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
     return;
   }
 
@@ -6036,8 +6033,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
            "Unexpected SVE register.");
     static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1,
                                        AArch64::zsub2};
-    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
-                     Indices);
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
     return;
   }
 
@@ -6050,8 +6046,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
            "Unexpected SVE register.");
     static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1,
                                        AArch64::zsub2, AArch64::zsub3};
-    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
-                     Indices);
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
     return;
   }
 
@@ -6060,8 +6055,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
       AArch64::DDDDRegClass.contains(SrcReg)) {
     static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
                                        AArch64::dsub2, AArch64::dsub3};
-    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
-                     Indices);
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
     return;
   }
 
@@ -6070,8 +6064,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
       AArch64::DDDRegClass.contains(SrcReg)) {
     static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
                                        AArch64::dsub2};
-    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
-                     Indices);
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
     return;
   }
 
@@ -6079,8 +6072,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
   if (AArch64::DDRegClass.contains(DestReg) &&
       AArch64::DDRegClass.contains(SrcReg)) {
     static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1};
-    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
-                     Indices);
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
     return;
   }
 
@@ -6089,8 +6081,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
       AArch64::QQQQRegClass.contains(SrcReg)) {
     static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
                                        AArch64::qsub2, AArch64::qsub3};
-    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
-                     Indices);
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
     return;
   }
 
@@ -6099,8 +6090,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
       AArch64::QQQRegClass.contains(SrcReg)) {
     static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
                                        AArch64::qsub2};
-    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
-                     Indices);
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
     return;
   }
 
@@ -6108,8 +6098,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
   if (AArch64::QQRegClass.contains(DestReg) &&
       AArch64::QQRegClass.contains(SrcReg)) {
     static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1};
-    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
-                     Indices);
+    copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
     return;
   }
 
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.h b/llvm/lib/Target/AArch64/AArch64InstrInfo.h
index 15bd832de8d25..821c2f3a7af42 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.h
@@ -355,7 +355,7 @@ class AArch64InstrInfo final : public AArch64GenInstrInfo {
 
   void copyPhysRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
                         const DebugLoc &DL, MCRegister DestReg,
-                        MCRegister SrcReg, bool KillSrc, unsigned Opcode,
+                        MCRegister SrcReg, bool KillSrc,
                         llvm::ArrayRef<unsigned> Indices) const;
   void copyGPRRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
                        const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg,
diff --git a/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll b/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll
index 9d71338c9c97a..8d517d47c75a1 100644
--- a/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll
@@ -9,8 +9,8 @@
 
 define void @test_D1D2_from_D0D1(ptr %addr) #0 {
 ; CHECK-LABEL: test_D1D2_from_D0D1:
-; CHECK: mov.8b v2, v1
-; CHECK: mov.8b v1, v0
+; CHECK: fmov d2, d1
+; CHECK: fmov d1, d0
 entry:
   %vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
   %vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -25,8 +25,8 @@ entry:
 
 define void @test_D0D1_from_D1D2(ptr %addr) #0 {
 ; CHECK-LABEL: test_D0D1_from_D1D2:
-; CHECK: mov.8b v0, v1
-; CHECK: mov.8b v1, v2
+; CHECK: fmov d0, d1
+; CHECK: fmov d1, d2
 entry:
   %vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
   %vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -41,8 +41,8 @@ entry:
 
 define void @test_D0D1_from_D31D0(ptr %addr) #0 {
 ; CHECK-LABEL: test_D0D1_from_D31D0:
-; CHECK: mov.8b v1, v0
-; CHECK: mov.8b v0, v31
+; CHECK: fmov d1, d0
+; CHECK: fmov d0, d31
 entry:
   %vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
   %vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -57,8 +57,8 @@ entry:
 
 define void @test_D31D0_from_D0D1(ptr %addr) #0 {
 ; CHECK-LABEL: test_D31D0_from_D0D1:
-; CHECK: mov.8b v31, v0
-; CHECK: mov.8b v0, v1
+; CHECK: fmov d31, d0
+; CHECK: fmov d0, d1
 entry:
   %vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
   %vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -73,9 +73,9 @@ entry:
 
 define void @test_D2D3D4_from_D0D1D2(ptr %addr) #0 {
 ; CHECK-LABEL: test_D2D3D4_from_D0D1D2:
-; CHECK: mov.8b v4, v2
-; CHECK: mov.8b v3, v1
-; CHECK: mov.8b v2, v0
+; CHECK: fmov d4, d2
+; CHECK: fmov d3, d1
+; CHECK: fmov d2, d0
 entry:
   %vec = tail call { <8 x i8>, <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld3.v8i8.p0(ptr %addr)
   %vec0 = extractvalue { <8 x i8>, <8 x i8>, <8 x i8> } %vec, 0
diff --git a/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir b/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
index 336886f699b49..799528d6e3774 100644
--- a/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
+++ b/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
@@ -22,3 +22,45 @@ body:             |
     RET_ReallyLR
 
 ...
+---
+name:            copy_ppr2_overlap
+alignment:       4
+tracksRegLiveness: true
+liveins:
+  - { reg: '$p0_p1' }
+frameInfo:
+  maxCallFrameSize: 0
+body:             |
+  bb.0:
+    liveins: $p0_p1
+    ; CHECK-LABEL: name: copy_ppr2_overlap
+    ; CHECK: liveins: $p0_p1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $p2 = ORR_PPzPP $p1, $p1, killed $p1
+    ; CHECK-NEXT: $p1 = ORR_PPzPP $p0, $p0, killed $p0
+    ; CHECK-NEXT: RET_ReallyLR
+    $p1_p2 = COPY killed renamable $p0_p1
+    RET_ReallyLR
+
+...
+---
+name:            copy_ppr2_max
+alignment:       4
+tracksRegLiveness: true
+liveins:
+  - { reg: '$p15_p0' }
+frameInfo:
+  maxCallFrameSize: 0
+body:             |
+  bb.0:
+    liveins: $p15_p0
+    ; CHECK-LABEL: name: copy_ppr2_max
+    ; CHECK: liveins: $p15_p0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $p1 = ORR_PPzPP $p0, $p0, killed $p0
+    ; CHECK-NEXT: $p0 = ORR_PPzPP $p15, $p15, killed $p15
+    ; CHECK-NEXT: RET_ReallyLR
+    $p0_p1 = COPY killed renamable $p15_p0
+    RET_ReallyLR
+
+...

>From 137b9959152ffae9f7839633ab3e5ba9d6531992 Mon Sep 17 00:00:00 2001
From: Kerry McLaughlin <kerry.mclaughlin at arm.com>
Date: Mon, 24 Aug 2026 12:39:04 +0000
Subject: [PATCH 3/4] - Check whether sub-register is a PPR in copyPhysRegTuple
 - Move most of copyPhysReg to copyPhysRegImpl to avoid multiple
 ++NumCopyInstrs

---
 llvm/lib/Target/AArch64/AArch64InstrInfo.cpp | 30 ++++++++++++++------
 llvm/lib/Target/AArch64/AArch64InstrInfo.h   |  4 +++
 2 files changed, 25 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 3ffa080021e39..66c66cfdd1e25 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -5761,7 +5761,8 @@ void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
   uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
   uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
   unsigned NumRegs = Indices.size();
-  bool IsPred = AArch64::PPR2RegClass.contains(DestReg);
+  bool IsPred =
+      AArch64::PPRRegClass.contains(TRI->getSubReg(DestReg, Indices[0]));
 
   int SubReg = 0, End = NumRegs, Incr = 1;
   if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs, IsPred)) {
@@ -5773,7 +5774,7 @@ void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
   for (; SubReg != End; SubReg += Incr) {
     MCRegister DestSubReg = TRI->getSubReg(DestReg, Indices[SubReg]);
     MCRegister SrcSubReg = TRI->getSubReg(SrcReg, Indices[SubReg]);
-    copyPhysReg(MBB, I, DL, DestSubReg, SrcSubReg, KillSrc);
+    copyPhysRegImpl(MBB, I, DL, DestSubReg, SrcSubReg, KillSrc);
   }
 }
 
@@ -5834,13 +5835,12 @@ static bool mustAvoidNeonAtMBBI(const AArch64Subtarget &Subtarget,
   return !Subtarget.hasSMEFA64() && isInStreamingCallSiteRegion(MBB, I);
 }
 
-void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
-                                   MachineBasicBlock::iterator I,
-                                   const DebugLoc &DL, Register DestReg,
-                                   Register SrcReg, bool KillSrc,
-                                   bool RenamableDest,
-                                   bool RenamableSrc) const {
-  ++NumCopyInstrs;
+void AArch64InstrInfo::copyPhysRegImpl(MachineBasicBlock &MBB,
+                                       MachineBasicBlock::iterator I,
+                                       const DebugLoc &DL, Register DestReg,
+                                       Register SrcReg, bool KillSrc,
+                                       bool RenamableDest,
+                                       bool RenamableSrc) const {
   if (AArch64::GPR32spRegClass.contains(DestReg) &&
       AArch64::GPR32spRegClass.contains(SrcReg)) {
     if (DestReg == AArch64::WSP || SrcReg == AArch64::WSP) {
@@ -6364,6 +6364,18 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
   llvm_unreachable("unimplemented reg-to-reg copy");
 }
 
+void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
+                                   MachineBasicBlock::iterator I,
+                                   const DebugLoc &DL, Register DestReg,
+                                   Register SrcReg, bool KillSrc,
+                                   bool RenamableDest,
+                                   bool RenamableSrc) const {
+  ++NumCopyInstrs;
+  copyPhysRegImpl(MBB, I, DL, DestReg, SrcReg, KillSrc, RenamableDest,
+                  RenamableSrc);
+  return;
+}
+
 static void storeRegPairToStackSlot(const TargetRegisterInfo &TRI,
                                     MachineBasicBlock &MBB,
                                     MachineBasicBlock::iterator InsertBefore,
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.h b/llvm/lib/Target/AArch64/AArch64InstrInfo.h
index 821c2f3a7af42..d9e34365479f1 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.h
@@ -361,6 +361,10 @@ class AArch64InstrInfo final : public AArch64GenInstrInfo {
                        const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg,
                        bool KillSrc, unsigned Opcode, unsigned ZeroReg,
                        llvm::ArrayRef<unsigned> Indices) const;
+  void copyPhysRegImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
+                       const DebugLoc &DL, Register DestReg, Register SrcReg,
+                       bool KillSrc, bool RenamableDest = false,
+                       bool RenamableSrc = false) const;
   void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
                    const DebugLoc &DL, Register DestReg, Register SrcReg,
                    bool KillSrc, bool RenamableDest = false,

>From 63fa4d3a9cb69f7f7ffd9f3e21042d5966abcf58 Mon Sep 17 00:00:00 2001
From: Kerry McLaughlin <kerry.mclaughlin at arm.com>
Date: Tue, 25 Aug 2026 12:19:28 +0000
Subject: [PATCH 4/4] - Add an assert for PNRRegClass - Inline
 forwardCopyWillClobberTuple

---
 llvm/lib/Target/AArch64/AArch64InstrInfo.cpp | 19 +++++++------------
 1 file changed, 7 insertions(+), 12 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 66c66cfdd1e25..d6472b93a3af9 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -5743,14 +5743,6 @@ static const MachineInstrBuilder &AddSubReg(const MachineInstrBuilder &MIB,
   return MIB.addReg(Reg, State, SubIdx);
 }
 
-static bool forwardCopyWillClobberTuple(unsigned DestReg, unsigned SrcReg,
-                                        unsigned NumRegs, bool IsPred) {
-  // We really want the positive remainder mod 16/32 here, that happens to be
-  // easily obtainable with a mask.
-  unsigned MaxRegs = IsPred ? 0xf : 0x1f;
-  return ((DestReg - SrcReg) & MaxRegs) < NumRegs;
-}
-
 void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
                                         MachineBasicBlock::iterator I,
                                         const DebugLoc &DL, MCRegister DestReg,
@@ -5761,18 +5753,21 @@ void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
   uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
   uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
   unsigned NumRegs = Indices.size();
-  bool IsPred =
-      AArch64::PPRRegClass.contains(TRI->getSubReg(DestReg, Indices[0]));
+  MCRegister DestSubReg = TRI->getSubReg(DestReg, Indices[0]);
+  assert(!AArch64::PNRRegClass.contains(DestSubReg) &&
+         "Unexpected predicate tuple copy");
+  unsigned MaxRegs = AArch64::PPRRegClass.contains(DestSubReg) ? 15 : 31;
 
   int SubReg = 0, End = NumRegs, Incr = 1;
-  if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs, IsPred)) {
+  // Copy in reverse if a forward copy will clobber the tuple
+  if (((DestEncoding - SrcEncoding) & MaxRegs) < NumRegs) {
     SubReg = NumRegs - 1;
     End = -1;
     Incr = -1;
   }
 
   for (; SubReg != End; SubReg += Incr) {
-    MCRegister DestSubReg = TRI->getSubReg(DestReg, Indices[SubReg]);
+    DestSubReg = TRI->getSubReg(DestReg, Indices[SubReg]);
     MCRegister SrcSubReg = TRI->getSubReg(SrcReg, Indices[SubReg]);
     copyPhysRegImpl(MBB, I, DL, DestSubReg, SrcSubReg, KillSrc);
   }



More information about the llvm-commits mailing list