[llvm] 622789f - [AArch64][SVE] Support copy of PPR2 register class in copyPhysReg (#216303)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 01:45:40 PDT 2026
Author: Kerry McLaughlin
Date: 2026-08-27T09:45:35+01:00
New Revision: 622789fbc0fb8fd77ed71fe13858d0a99f6b70bb
URL: https://github.com/llvm/llvm-project/commit/622789fbc0fb8fd77ed71fe13858d0a99f6b70bb
DIFF: https://github.com/llvm/llvm-project/commit/622789fbc0fb8fd77ed71fe13858d0a99f6b70bb.diff
LOG: [AArch64][SVE] Support copy of PPR2 register class in copyPhysReg (#216303)
This fixes the following crash, which was observed after landing
#209484: https://clang.godbolt.org/z/YG1r63oT3
Added:
llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
Modified:
llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
llvm/lib/Target/AArch64/AArch64InstrInfo.h
llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll
Removed:
################################################################################
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 9867164af2e41..43aab1400e81f 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -5743,37 +5743,33 @@ static const MachineInstrBuilder &AddSubReg(const MachineInstrBuilder &MIB,
return MIB.addReg(Reg, State, SubIdx);
}
-static bool forwardCopyWillClobberTuple(unsigned DestReg, unsigned SrcReg,
- unsigned NumRegs) {
- // We really want the positive remainder mod 32 here, that happens to be
- // easily obtainable with a mask.
- return ((DestReg - SrcReg) & 0x1f) < NumRegs;
-}
-
void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB,
MachineBasicBlock::iterator I,
const DebugLoc &DL, MCRegister DestReg,
MCRegister SrcReg, bool KillSrc,
- unsigned Opcode,
ArrayRef<unsigned> Indices) const {
assert(Subtarget.hasNEON() && "Unexpected register copy without NEON");
const TargetRegisterInfo *TRI = &getRegisterInfo();
uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
unsigned NumRegs = Indices.size();
+ MCRegister DestSubReg = TRI->getSubReg(DestReg, Indices[0]);
+ assert(!AArch64::PNRRegClass.contains(DestSubReg) &&
+ "Unexpected predicate tuple copy");
+ unsigned MaxRegs = AArch64::PPRRegClass.contains(DestSubReg) ? 15 : 31;
int SubReg = 0, End = NumRegs, Incr = 1;
- if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs)) {
+ // Copy in reverse if a forward copy will clobber the tuple
+ if (((DestEncoding - SrcEncoding) & MaxRegs) < NumRegs) {
SubReg = NumRegs - 1;
End = -1;
Incr = -1;
}
for (; SubReg != End; SubReg += Incr) {
- const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode));
- AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI);
- AddSubReg(MIB, SrcReg, Indices[SubReg], {}, TRI);
- AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI);
+ DestSubReg = TRI->getSubReg(DestReg, Indices[SubReg]);
+ MCRegister SrcSubReg = TRI->getSubReg(SrcReg, Indices[SubReg]);
+ copyPhysRegImpl(MBB, I, DL, DestSubReg, SrcSubReg, KillSrc);
}
}
@@ -5834,13 +5830,12 @@ static bool mustAvoidNeonAtMBBI(const AArch64Subtarget &Subtarget,
return !Subtarget.hasSMEFA64() && isInStreamingCallSiteRegion(MBB, I);
}
-void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
- MachineBasicBlock::iterator I,
- const DebugLoc &DL, Register DestReg,
- Register SrcReg, bool KillSrc,
- bool RenamableDest,
- bool RenamableSrc) const {
- ++NumCopyInstrs;
+void AArch64InstrInfo::copyPhysRegImpl(MachineBasicBlock &MBB,
+ MachineBasicBlock::iterator I,
+ const DebugLoc &DL, Register DestReg,
+ Register SrcReg, bool KillSrc,
+ bool RenamableDest,
+ bool RenamableSrc) const {
if (AArch64::GPR32spRegClass.contains(DestReg) &&
AArch64::GPR32spRegClass.contains(SrcReg)) {
if (DestReg == AArch64::WSP || SrcReg == AArch64::WSP) {
@@ -5993,6 +5988,16 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
return;
}
+ // Copy a predicate register pair by copying the individual sub-registers.
+ if (AArch64::PPR2RegClass.contains(DestReg) &&
+ AArch64::PPR2RegClass.contains(SrcReg)) {
+ assert(Subtarget.isSVEorStreamingSVEAvailable() &&
+ "Unexpected SVE predicate register.");
+ static const unsigned Indices[] = {AArch64::psub0, AArch64::psub1};
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
+ return;
+ }
+
// Copy a Z register by ORRing with itself.
if (AArch64::ZPRRegClass.contains(DestReg) &&
AArch64::ZPRRegClass.contains(SrcReg)) {
@@ -6012,8 +6017,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
assert(Subtarget.isSVEorStreamingSVEAvailable() &&
"Unexpected SVE register.");
static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6024,8 +6028,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
"Unexpected SVE register.");
static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1,
AArch64::zsub2};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6038,8 +6041,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
"Unexpected SVE register.");
static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1,
AArch64::zsub2, AArch64::zsub3};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORR_ZZZ,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6048,8 +6050,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
AArch64::DDDDRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
AArch64::dsub2, AArch64::dsub3};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6058,8 +6059,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
AArch64::DDDRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
AArch64::dsub2};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6067,8 +6067,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
if (AArch64::DDRegClass.contains(DestReg) &&
AArch64::DDRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6077,8 +6076,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
AArch64::QQQQRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
AArch64::qsub2, AArch64::qsub3};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6087,8 +6085,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
AArch64::QQQRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
AArch64::qsub2};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6096,8 +6093,7 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
if (AArch64::QQRegClass.contains(DestReg) &&
AArch64::QQRegClass.contains(SrcReg)) {
static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1};
- copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8,
- Indices);
+ copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
return;
}
@@ -6363,6 +6359,18 @@ void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
llvm_unreachable("unimplemented reg-to-reg copy");
}
+void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB,
+ MachineBasicBlock::iterator I,
+ const DebugLoc &DL, Register DestReg,
+ Register SrcReg, bool KillSrc,
+ bool RenamableDest,
+ bool RenamableSrc) const {
+ ++NumCopyInstrs;
+ copyPhysRegImpl(MBB, I, DL, DestReg, SrcReg, KillSrc, RenamableDest,
+ RenamableSrc);
+ return;
+}
+
static void storeRegPairToStackSlot(const TargetRegisterInfo &TRI,
MachineBasicBlock &MBB,
MachineBasicBlock::iterator InsertBefore,
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.h b/llvm/lib/Target/AArch64/AArch64InstrInfo.h
index 15bd832de8d25..d9e34365479f1 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.h
@@ -355,12 +355,16 @@ class AArch64InstrInfo final : public AArch64GenInstrInfo {
void copyPhysRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
const DebugLoc &DL, MCRegister DestReg,
- MCRegister SrcReg, bool KillSrc, unsigned Opcode,
+ MCRegister SrcReg, bool KillSrc,
llvm::ArrayRef<unsigned> Indices) const;
void copyGPRRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg,
bool KillSrc, unsigned Opcode, unsigned ZeroReg,
llvm::ArrayRef<unsigned> Indices) const;
+ void copyPhysRegImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
+ const DebugLoc &DL, Register DestReg, Register SrcReg,
+ bool KillSrc, bool RenamableDest = false,
+ bool RenamableSrc = false) const;
void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
const DebugLoc &DL, Register DestReg, Register SrcReg,
bool KillSrc, bool RenamableDest = false,
diff --git a/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll b/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll
index 9d71338c9c97a..8d517d47c75a1 100644
--- a/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-copy-tuple.ll
@@ -9,8 +9,8 @@
define void @test_D1D2_from_D0D1(ptr %addr) #0 {
; CHECK-LABEL: test_D1D2_from_D0D1:
-; CHECK: mov.8b v2, v1
-; CHECK: mov.8b v1, v0
+; CHECK: fmov d2, d1
+; CHECK: fmov d1, d0
entry:
%vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
%vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -25,8 +25,8 @@ entry:
define void @test_D0D1_from_D1D2(ptr %addr) #0 {
; CHECK-LABEL: test_D0D1_from_D1D2:
-; CHECK: mov.8b v0, v1
-; CHECK: mov.8b v1, v2
+; CHECK: fmov d0, d1
+; CHECK: fmov d1, d2
entry:
%vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
%vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -41,8 +41,8 @@ entry:
define void @test_D0D1_from_D31D0(ptr %addr) #0 {
; CHECK-LABEL: test_D0D1_from_D31D0:
-; CHECK: mov.8b v1, v0
-; CHECK: mov.8b v0, v31
+; CHECK: fmov d1, d0
+; CHECK: fmov d0, d31
entry:
%vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
%vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -57,8 +57,8 @@ entry:
define void @test_D31D0_from_D0D1(ptr %addr) #0 {
; CHECK-LABEL: test_D31D0_from_D0D1:
-; CHECK: mov.8b v31, v0
-; CHECK: mov.8b v0, v1
+; CHECK: fmov d31, d0
+; CHECK: fmov d0, d1
entry:
%vec = tail call { <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld2.v8i8.p0(ptr %addr)
%vec0 = extractvalue { <8 x i8>, <8 x i8> } %vec, 0
@@ -73,9 +73,9 @@ entry:
define void @test_D2D3D4_from_D0D1D2(ptr %addr) #0 {
; CHECK-LABEL: test_D2D3D4_from_D0D1D2:
-; CHECK: mov.8b v4, v2
-; CHECK: mov.8b v3, v1
-; CHECK: mov.8b v2, v0
+; CHECK: fmov d4, d2
+; CHECK: fmov d3, d1
+; CHECK: fmov d2, d0
entry:
%vec = tail call { <8 x i8>, <8 x i8>, <8 x i8> } @llvm.aarch64.neon.ld3.v8i8.p0(ptr %addr)
%vec0 = extractvalue { <8 x i8>, <8 x i8>, <8 x i8> } %vec, 0
diff --git a/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir b/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
new file mode 100644
index 0000000000000..799528d6e3774
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-copy-pprpair.mir
@@ -0,0 +1,66 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
+# RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve -run-pass=postrapseudos -simplify-mir -verify-machineinstrs %s -o - | FileCheck %s
+
+---
+name: copy_ppr2
+alignment: 4
+tracksRegLiveness: true
+liveins:
+ - { reg: '$p0_p1' }
+frameInfo:
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ liveins: $p0_p1
+ ; CHECK-LABEL: name: copy_ppr2
+ ; CHECK: liveins: $p0_p1
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: $p2 = ORR_PPzPP $p0, $p0, killed $p0
+ ; CHECK-NEXT: $p3 = ORR_PPzPP $p1, $p1, killed $p1
+ ; CHECK-NEXT: RET_ReallyLR
+ $p2_p3 = COPY killed renamable $p0_p1
+ RET_ReallyLR
+
+...
+---
+name: copy_ppr2_overlap
+alignment: 4
+tracksRegLiveness: true
+liveins:
+ - { reg: '$p0_p1' }
+frameInfo:
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ liveins: $p0_p1
+ ; CHECK-LABEL: name: copy_ppr2_overlap
+ ; CHECK: liveins: $p0_p1
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: $p2 = ORR_PPzPP $p1, $p1, killed $p1
+ ; CHECK-NEXT: $p1 = ORR_PPzPP $p0, $p0, killed $p0
+ ; CHECK-NEXT: RET_ReallyLR
+ $p1_p2 = COPY killed renamable $p0_p1
+ RET_ReallyLR
+
+...
+---
+name: copy_ppr2_max
+alignment: 4
+tracksRegLiveness: true
+liveins:
+ - { reg: '$p15_p0' }
+frameInfo:
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ liveins: $p15_p0
+ ; CHECK-LABEL: name: copy_ppr2_max
+ ; CHECK: liveins: $p15_p0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: $p1 = ORR_PPzPP $p0, $p0, killed $p0
+ ; CHECK-NEXT: $p0 = ORR_PPzPP $p15, $p15, killed $p15
+ ; CHECK-NEXT: RET_ReallyLR
+ $p0_p1 = COPY killed renamable $p15_p0
+ RET_ReallyLR
+
+...
More information about the llvm-commits
mailing list