[llvm] [AArch64][SVE] Support copy of PPR2 register class in copyPhysReg (PR #216303)
Kerry McLaughlin via llvm-commits
llvm-commits at lists.llvm.org
Tue Aug 25 06:43:09 PDT 2026
https://github.com/kmclaughlin-arm updated https://github.com/llvm/llvm-project/pull/216303
>From 3bd0abb61eb5e74d22ee62d1e36f5ec466a1b9f3 Mon Sep 17 00:00:00 2001
From: Kerry McLaughlin <kerry.mclaughlin at arm.com>
Date: Fri, 14 Aug 2026 10:50:01 +0000
Subject: [PATCH 1/4] [AArch64][SVE] Support copy of PPR2 register class in
copyPhysReg.
This fixes the following crash, which was observed after landing #209484:
https://github.com/llvm/llvm-project/pull/209484
---
llvm/lib/Target/AArch64/AArch64InstrInfo.cpp | 12 ++++++++++
.../test/CodeGen/AArch64/sve-copy-pprpair.mir | 24 +++++++++++++++++++
2 files changed, 36 insertions(+)
create mode 100644 llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 522b263df5de1..42062c8021135 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -5773,6 +5773,8 @@ void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode));
AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI);
AddSubReg(MIB, SrcReg, Indices[SubReg], {}, TRI);
+ if (Opcode == AArch64::ORR_PPzPP)
+ AddSubReg(MIB, SrcReg, Indices[SubReg], {}, TRI);
AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI);
}
}
@@ -5993,6 +5995,16 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
return;
}
+ if (AArch64::PPR2RegClass.contains(DestReg) &&
+ AArch64::PPR2RegClass.contains(SrcReg)) {
+ assert(Subtarget.isSVEorStreamingSVEAvailable() &&
+ "Unexpected SVE predicate register.");
+ static const unsigned Indices[] = {AArch64::psub0, AArch64::psub1};
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_PPzPP,
+ Indices);
+ return;
+ }
+
// Copy a Z register by ORRing with itself.
if (AArch64::ZPRRegClass.contains(DestReg) &&
AArch64::ZPRRegClass.contains(SrcReg)) {
diff --git a/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir b/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
new file mode 100644
index 0000000000000..336886f699b49
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
@@ -0,0 +1,24 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
+# RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve -run-pass=postrapseudos -simplify-mir -verify-machineinstrs %s -o - | FileCheck %s
+
+---
+name: copy_ppr2
+alignment: 4
+tracksRegLiveness: true
+liveins:
+ - { reg: '$p0_p1' }
+frameInfo:
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ liveins: $p0_p1
+ ; CHECK-LABEL: name: copy_ppr2
+ ; CHECK: liveins: $p0_p1
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: $p2 = ORR_PPzPP $p0, $p0, killed $p0
+ ; CHECK-NEXT: $p3 = ORR_PPzPP $p1, $p1, killed $p1
+ ; CHECK-NEXT: RET_ReallyLR
+ $p2_p3 = COPY killed renamable $p0_p1
+ RET_ReallyLR
+
+...
>From 3f14e68668c45ab8a6bf77ffe9e91bd9d181aaad Mon Sep 17 00:00:00 2001
From: Kerry McLaughlin <kerry.mclaughlin at arm.com>
Date: Fri, 14 Aug 2026 13:56:38 +0000
Subject: [PATCH 2/4] - Call copyPhysReg directly from copyPhysRegTuple -
Changed forwardCopyWillClobberTuple to consider number of predicate registers
- Added tests with overlap & test for forwardCopyWillClobberTuple
---
llvm/lib/Target/AArch64/AArch64InstrInfo.cpp | 51 ++++++++-----------
llvm/lib/Target/AArch64/AArch64InstrInfo.h | 2 +-
llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll | 22 ++++----
.../test/CodeGen/AArch64/sve-copy-pprpair.mir | 42 +++++++++++++++
4 files changed, 74 insertions(+), 43 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 42062c8021135..3ffa080021e39 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -5744,38 +5744,36 @@ static const MachineInstrBuilder &AddSubReg(const MachineInstrBuilder &MIB,
}
static bool forwardCopyWillClobberTuple(unsigned DestReg, unsigned SrcReg,
- unsigned NumRegs) {
- // We really want the positive remainder mod 32 here, that happens to be
+ unsigned NumRegs, bool IsPred) {
+ // We really want the positive remainder mod 16/32 here, that happens to be
// easily obtainable with a mask.
- return ((DestReg - SrcReg) & 0x1f) < NumRegs;
+ unsigned MaxRegs = IsPred ? 0xf : 0x1f;
+ return ((DestReg - SrcReg) & MaxRegs) < NumRegs;
}
void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
MachineBasicBlock::iterator I,
const DebugLoc &DL, MCRegister DestReg,
MCRegister SrcReg, bool KillSrc,
- unsigned Opcode,
ArrayRef<unsigned> Indices) const {
assert(Subtarget.hasNEON() && "Unexpected register copy without NEON");
const TargetRegisterInfo *TRI = &getRegisterInfo();
uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
unsigned NumRegs = Indices.size();
+ bool IsPred = AArch64::PPR2RegClass.contains(DestReg);
int SubReg = 0, End = NumRegs, Incr = 1;
- if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs)) {
+ if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs, IsPred)) {
SubReg = NumRegs - 1;
End = -1;
Incr = -1;
}
for (; SubReg != End; SubReg += Incr) {
- const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode));
- AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI);
- AddSubReg(MIB, SrcReg, Indices[SubReg], {}, TRI);
- if (Opcode == AArch64::ORR_PPzPP)
- AddSubReg(MIB, SrcReg, Indices[SubReg], {}, TRI);
- AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI);
+ MCRegister DestSubReg = TRI->getSubReg(DestReg, Indices[SubReg]);
+ MCRegister SrcSubReg = TRI->getSubReg(SrcReg, Indices[SubReg]);
+ copyPhysReg(MBB, I, DL, DestSubReg, SrcSubReg, KillSrc);
}
}
@@ -5995,13 +5993,13 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
return;
}
+ // Copy a predicate register pair by copying the individual sub-registers.
if (AArch64::PPR2RegClass.contains(DestReg) &&
AArch64::PPR2RegClass.contains(SrcReg)) {
assert(Subtarget.isSVEorStreamingSVEAvailable() &&
"Unexpected SVE predicate register.");
static const unsigned Indices[] = {AArch64::psub0, AArch64::psub1};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_PPzPP,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6024,8 +6022,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
assert(Subtarget.isSVEorStreamingSVEAvailable() &&
"Unexpected SVE register.");
static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6036,8 +6033,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
"Unexpected SVE register.");
static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1,
AArch64::zsub2};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6050,8 +6046,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
"Unexpected SVE register.");
static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1,
AArch64::zsub2, AArch64::zsub3};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6060,8 +6055,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
AArch64::DDDDRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
AArch64::dsub2, AArch64::dsub3};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6070,8 +6064,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
AArch64::DDDRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
AArch64::dsub2};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6079,8 +6072,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
if (AArch64::DDRegClass.contains(DestReg) &&
AArch64::DDRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6089,8 +6081,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
AArch64::QQQQRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
AArch64::qsub2, AArch64::qsub3};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6099,8 +6090,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
AArch64::QQQRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
AArch64::qsub2};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6108,8 +6098,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
if (AArch64::QQRegClass.contains(DestReg) &&
AArch64::QQRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.h b/llvm/lib/Target/AArch64/AArch64InstrInfo.h
index 15bd832de8d25..821c2f3a7af42 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.h
@@ -355,7 +355,7 @@ class AArch64InstrInfo final : public AArch64GenInstrInfo {
void copyPhysRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
const DebugLoc &DL, MCRegister DestReg,
- MCRegister SrcReg, bool KillSrc, unsigned Opcode,
+ MCRegister SrcReg, bool KillSrc,
llvm::ArrayRef<unsigned> Indices) const;
void copyGPRRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg,
diff --git a/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll b/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll
index 9d71338c9c97a..8d517d47c75a1 100644
--- a/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll
@@ -9,8 +9,8 @@
define void @test_D1D2_from_D0D1(ptr %addr) #0 {
; CHECK-LABEL: test_D1D2_from_D0D1:
-; CHECK: mov.8b v2, v1
-; CHECK: mov.8b v1, v0
+; CHECK: fmov d2, d1
+; CHECK: fmov d1, d0
entry:
%vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
%vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -25,8 +25,8 @@ entry:
define void @test_D0D1_from_D1D2(ptr %addr) #0 {
; CHECK-LABEL: test_D0D1_from_D1D2:
-; CHECK: mov.8b v0, v1
-; CHECK: mov.8b v1, v2
+; CHECK: fmov d0, d1
+; CHECK: fmov d1, d2
entry:
%vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
%vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -41,8 +41,8 @@ entry:
define void @test_D0D1_from_D31D0(ptr %addr) #0 {
; CHECK-LABEL: test_D0D1_from_D31D0:
-; CHECK: mov.8b v1, v0
-; CHECK: mov.8b v0, v31
+; CHECK: fmov d1, d0
+; CHECK: fmov d0, d31
entry:
%vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
%vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -57,8 +57,8 @@ entry:
define void @test_D31D0_from_D0D1(ptr %addr) #0 {
; CHECK-LABEL: test_D31D0_from_D0D1:
-; CHECK: mov.8b v31, v0
-; CHECK: mov.8b v0, v1
+; CHECK: fmov d31, d0
+; CHECK: fmov d0, d1
entry:
%vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
%vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -73,9 +73,9 @@ entry:
define void @test_D2D3D4_from_D0D1D2(ptr %addr) #0 {
; CHECK-LABEL: test_D2D3D4_from_D0D1D2:
-; CHECK: mov.8b v4, v2
-; CHECK: mov.8b v3, v1
-; CHECK: mov.8b v2, v0
+; CHECK: fmov d4, d2
+; CHECK: fmov d3, d1
+; CHECK: fmov d2, d0
entry:
%vec = tail call { <8 x i8>, <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld3.v8i8.p0(ptr %addr)
%vec0 = extractvalue { <8 x i8>, <8 x i8>, <8 x i8> } %vec, 0
diff --git a/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir b/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
index 336886f699b49..799528d6e3774 100644
--- a/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
+++ b/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
@@ -22,3 +22,45 @@ body: |
RET_ReallyLR
...
+---
+name: copy_ppr2_overlap
+alignment: 4
+tracksRegLiveness: true
+liveins:
+ - { reg: '$p0_p1' }
+frameInfo:
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ liveins: $p0_p1
+ ; CHECK-LABEL: name: copy_ppr2_overlap
+ ; CHECK: liveins: $p0_p1
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: $p2 = ORR_PPzPP $p1, $p1, killed $p1
+ ; CHECK-NEXT: $p1 = ORR_PPzPP $p0, $p0, killed $p0
+ ; CHECK-NEXT: RET_ReallyLR
+ $p1_p2 = COPY killed renamable $p0_p1
+ RET_ReallyLR
+
+...
+---
+name: copy_ppr2_max
+alignment: 4
+tracksRegLiveness: true
+liveins:
+ - { reg: '$p15_p0' }
+frameInfo:
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ liveins: $p15_p0
+ ; CHECK-LABEL: name: copy_ppr2_max
+ ; CHECK: liveins: $p15_p0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: $p1 = ORR_PPzPP $p0, $p0, killed $p0
+ ; CHECK-NEXT: $p0 = ORR_PPzPP $p15, $p15, killed $p15
+ ; CHECK-NEXT: RET_ReallyLR
+ $p0_p1 = COPY killed renamable $p15_p0
+ RET_ReallyLR
+
+...
>From 137b9959152ffae9f7839633ab3e5ba9d6531992 Mon Sep 17 00:00:00 2001
From: Kerry McLaughlin <kerry.mclaughlin at arm.com>
Date: Mon, 24 Aug 2026 12:39:04 +0000
Subject: [PATCH 3/4] - Check whether sub-register is a PPR in copyPhysRegTuple
- Move most of copyPhysReg to copyPhysRegImpl to avoid multiple
++NumCopyInstrs
---
llvm/lib/Target/AArch64/AArch64InstrInfo.cpp | 30 ++++++++++++++------
llvm/lib/Target/AArch64/AArch64InstrInfo.h | 4 +++
2 files changed, 25 insertions(+), 9 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 3ffa080021e39..66c66cfdd1e25 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -5761,7 +5761,8 @@ void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
unsigned NumRegs = Indices.size();
- bool IsPred = AArch64::PPR2RegClass.contains(DestReg);
+ bool IsPred =
+ AArch64::PPRRegClass.contains(TRI->getSubReg(DestReg, Indices[0]));
int SubReg = 0, End = NumRegs, Incr = 1;
if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs, IsPred)) {
@@ -5773,7 +5774,7 @@ void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
for (; SubReg != End; SubReg += Incr) {
MCRegister DestSubReg = TRI->getSubReg(DestReg, Indices[SubReg]);
MCRegister SrcSubReg = TRI->getSubReg(SrcReg, Indices[SubReg]);
- copyPhysReg(MBB, I, DL, DestSubReg, SrcSubReg, KillSrc);
+ copyPhysRegImpl(MBB, I, DL, DestSubReg, SrcSubReg, KillSrc);
}
}
@@ -5834,13 +5835,12 @@ static bool mustAvoidNeonAtMBBI(const AArch64Subtarget &Subtarget,
return !Subtarget.hasSMEFA64() && isInStreamingCallSiteRegion(MBB, I);
}
-void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
- MachineBasicBlock::iterator I,
- const DebugLoc &DL, Register DestReg,
- Register SrcReg, bool KillSrc,
- bool RenamableDest,
- bool RenamableSrc) const {
- ++NumCopyInstrs;
+void AArch64InstrInfo::copyPhysRegImpl(MachineBasicBlock &MBB,
+ MachineBasicBlock::iterator I,
+ const DebugLoc &DL, Register DestReg,
+ Register SrcReg, bool KillSrc,
+ bool RenamableDest,
+ bool RenamableSrc) const {
if (AArch64::GPR32spRegClass.contains(DestReg) &&
AArch64::GPR32spRegClass.contains(SrcReg)) {
if (DestReg == AArch64::WSP || SrcReg == AArch64::WSP) {
@@ -6364,6 +6364,18 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
llvm_unreachable("unimplemented reg-to-reg copy");
}
+void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
+ MachineBasicBlock::iterator I,
+ const DebugLoc &DL, Register DestReg,
+ Register SrcReg, bool KillSrc,
+ bool RenamableDest,
+ bool RenamableSrc) const {
+ ++NumCopyInstrs;
+ copyPhysRegImpl(MBB, I, DL, DestReg, SrcReg, KillSrc, RenamableDest,
+ RenamableSrc);
+ return;
+}
+
static void storeRegPairToStackSlot(const TargetRegisterInfo &TRI,
MachineBasicBlock &MBB,
MachineBasicBlock::iterator InsertBefore,
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.h b/llvm/lib/Target/AArch64/AArch64InstrInfo.h
index 821c2f3a7af42..d9e34365479f1 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.h
@@ -361,6 +361,10 @@ class AArch64InstrInfo final : public AArch64GenInstrInfo {
const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg,
bool KillSrc, unsigned Opcode, unsigned ZeroReg,
llvm::ArrayRef<unsigned> Indices) const;
+ void copyPhysRegImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
+ const DebugLoc &DL, Register DestReg, Register SrcReg,
+ bool KillSrc, bool RenamableDest = false,
+ bool RenamableSrc = false) const;
void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
const DebugLoc &DL, Register DestReg, Register SrcReg,
bool KillSrc, bool RenamableDest = false,
>From 63fa4d3a9cb69f7f7ffd9f3e21042d5966abcf58 Mon Sep 17 00:00:00 2001
From: Kerry McLaughlin <kerry.mclaughlin at arm.com>
Date: Tue, 25 Aug 2026 12:19:28 +0000
Subject: [PATCH 4/4] - Add an assert for PNRRegClass - Inline
forwardCopyWillClobberTuple
---
llvm/lib/Target/AArch64/AArch64InstrInfo.cpp | 19 +++++++------------
1 file changed, 7 insertions(+), 12 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 66c66cfdd1e25..d6472b93a3af9 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -5743,14 +5743,6 @@ static const MachineInstrBuilder &AddSubReg(const MachineInstrBuilder &MIB,
return MIB.addReg(Reg, State, SubIdx);
}
-static bool forwardCopyWillClobberTuple(unsigned DestReg, unsigned SrcReg,
- unsigned NumRegs, bool IsPred) {
- // We really want the positive remainder mod 16/32 here, that happens to be
- // easily obtainable with a mask.
- unsigned MaxRegs = IsPred ? 0xf : 0x1f;
- return ((DestReg - SrcReg) & MaxRegs) < NumRegs;
-}
-
void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
MachineBasicBlock::iterator I,
const DebugLoc &DL, MCRegister DestReg,
@@ -5761,18 +5753,21 @@ void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
unsigned NumRegs = Indices.size();
- bool IsPred =
- AArch64::PPRRegClass.contains(TRI->getSubReg(DestReg, Indices[0]));
+ MCRegister DestSubReg = TRI->getSubReg(DestReg, Indices[0]);
+ assert(!AArch64::PNRRegClass.contains(DestSubReg) &&
+ "Unexpected predicate tuple copy");
+ unsigned MaxRegs = AArch64::PPRRegClass.contains(DestSubReg) ? 15 : 31;
int SubReg = 0, End = NumRegs, Incr = 1;
- if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs, IsPred)) {
+ // Copy in reverse if a forward copy will clobber the tuple
+ if (((DestEncoding - SrcEncoding) & MaxRegs) < NumRegs) {
SubReg = NumRegs - 1;
End = -1;
Incr = -1;
}
for (; SubReg != End; SubReg += Incr) {
- MCRegister DestSubReg = TRI->getSubReg(DestReg, Indices[SubReg]);
+ DestSubReg = TRI->getSubReg(DestReg, Indices[SubReg]);
MCRegister SrcSubReg = TRI->getSubReg(SrcReg, Indices[SubReg]);
copyPhysRegImpl(MBB, I, DL, DestSubReg, SrcSubReg, KillSrc);
}
More information about the llvm-commits
mailing list