[llvm-branch-commits] [llvm] PPC: Fold 64-bit zero-extending word load feeding extsw subregister (PR #211107)
Matt Arsenault via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Thu Oct 8 04:26:49 PDT 2026
https://github.com/arsenm updated https://github.com/llvm/llvm-project/pull/211107
>From 24e9116cab6af2774cd921cf8b15a8d5742de444 Mon Sep 17 00:00:00 2001
From: Matt Arsenault <Matthew.Arsenault at amd.com>
Date: Tue, 21 Jul 2026 19:35:27 +0200
Subject: [PATCH] PPC: Fold 64-bit zero-extending word load feeding extsw
subregister
A gprc LWZ/LWZX feeding EXTSW_32_64 is rewritten into a sign-extending
LWA/LWAX load. Extend the same fold to the 64-bit zero-extending word
loads LWZ8/LWZX8 when the EXTSW_32_64 reads their sub_32 subregister,
producing a single LWA/LWAX instead of a redundant lwz+extsw pair.
Co-authored-by: Claude (Claude Opus 4.8, claude-opus-4-8) <noreply at anthropic.com>
---
llvm/lib/Target/PowerPC/PPCMIPeephole.cpp | 16 ++++++++++---
.../peephole-elim-extsw-subreg-input.mir | 24 +++++++++----------
2 files changed, 25 insertions(+), 15 deletions(-)
diff --git a/llvm/lib/Target/PowerPC/PPCMIPeephole.cpp b/llvm/lib/Target/PowerPC/PPCMIPeephole.cpp
index 4ca258a2d2555..73bb0d25f8a7b 100644
--- a/llvm/lib/Target/PowerPC/PPCMIPeephole.cpp
+++ b/llvm/lib/Target/PowerPC/PPCMIPeephole.cpp
@@ -1013,9 +1013,18 @@ bool PPCMIPeephole::simplifyCode() {
MachineInstr *SrcMI = MRI->getVRegDef(NarrowReg);
unsigned SrcOpcode = SrcMI->getOpcode();
+ Register NarrowSubReg = MI.getOperand(1).getSubReg();
+
// If we've used a zero-extending load that we will sign-extend,
- // just do a sign-extending load.
- if (SrcOpcode == PPC::LWZ || SrcOpcode == PPC::LWZX) {
+ // just do a sign-extending load. The source may be a 32-bit gprc load
+ // consumed directly, or a 64-bit g8rc load consumed through its sub_32
+ // subregister.
+ bool SrcIsGPRCWordZextLoad =
+ SrcOpcode == PPC::LWZ || SrcOpcode == PPC::LWZX;
+ bool SrcIsG8RCWordZextLoad =
+ SrcOpcode == PPC::LWZ8 || SrcOpcode == PPC::LWZX8;
+ if ((!NarrowSubReg && SrcIsGPRCWordZextLoad) ||
+ (NarrowSubReg == PPC::sub_32 && SrcIsG8RCWordZextLoad)) {
if (!MRI->hasOneNonDBGUse(SrcMI->getOperand(0).getReg()))
break;
@@ -1041,7 +1050,8 @@ bool PPCMIPeephole::simplifyCode() {
// Likewise if the source is X-Form the new opcode should also be
// X-Form.
unsigned Opc = PPC::LWA_32;
- bool SourceIsXForm = SrcOpcode == PPC::LWZX;
+ bool SourceIsXForm =
+ SrcOpcode == PPC::LWZX || SrcOpcode == PPC::LWZX8;
bool MIIs64Bit = MI.getOpcode() == PPC::EXTSW ||
MI.getOpcode() == PPC::EXTSW_32_64;
diff --git a/llvm/test/CodeGen/PowerPC/peephole-elim-extsw-subreg-input.mir b/llvm/test/CodeGen/PowerPC/peephole-elim-extsw-subreg-input.mir
index 90f67b81b1b15..daaed25113537 100644
--- a/llvm/test/CodeGen/PowerPC/peephole-elim-extsw-subreg-input.mir
+++ b/llvm/test/CodeGen/PowerPC/peephole-elim-extsw-subreg-input.mir
@@ -40,9 +40,9 @@ body: |
; CHECK: liveins: $x3
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[COPY:%[0-9]+]]:g8rc_and_g8rc_nox0 = COPY $x3
- ; CHECK-NEXT: [[LWZ8_:%[0-9]+]]:g8rc = LWZ8 0, [[COPY]] :: (load (s32))
- ; CHECK-NEXT: [[EXTSW_32_64_:%[0-9]+]]:g8rc = EXTSW_32_64 [[LWZ8_]].sub_32
- ; CHECK-NEXT: $x3 = COPY [[EXTSW_32_64_]]
+ ; CHECK-NEXT: [[LWA:%[0-9]+]]:g8rc = LWA 0, [[COPY]] :: (load (s32))
+ ; CHECK-NEXT: [[DEF:%[0-9]+]]:g8rc = IMPLICIT_DEF
+ ; CHECK-NEXT: $x3 = COPY [[LWA]]
; CHECK-NEXT: BLR8 implicit $lr8, implicit $rm, implicit $x3
%0:g8rc_and_g8rc_nox0 = COPY $x3
%1:g8rc = LWZ8 0, %0 :: (load (s32))
@@ -63,9 +63,9 @@ body: |
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[COPY:%[0-9]+]]:g8rc_and_g8rc_nox0 = COPY $x3
; CHECK-NEXT: [[COPY1:%[0-9]+]]:g8rc = COPY $x4
- ; CHECK-NEXT: [[LWZX8_:%[0-9]+]]:g8rc = LWZX8 [[COPY]], [[COPY1]] :: (load (s32))
- ; CHECK-NEXT: [[EXTSW_32_64_:%[0-9]+]]:g8rc = EXTSW_32_64 [[LWZX8_]].sub_32
- ; CHECK-NEXT: $x3 = COPY [[EXTSW_32_64_]]
+ ; CHECK-NEXT: [[LWAX:%[0-9]+]]:g8rc = LWAX [[COPY]], [[COPY1]] :: (load (s32))
+ ; CHECK-NEXT: [[DEF:%[0-9]+]]:g8rc = IMPLICIT_DEF
+ ; CHECK-NEXT: $x3 = COPY [[LWAX]]
; CHECK-NEXT: BLR8 implicit $lr8, implicit $rm, implicit $x3
%0:g8rc_and_g8rc_nox0 = COPY $x3
%3:g8rc = COPY $x4
@@ -86,10 +86,10 @@ body: |
; CHECK: liveins: $x3
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[COPY:%[0-9]+]]:g8rc_and_g8rc_nox0 = COPY $x3
- ; CHECK-NEXT: [[LWZ8_:%[0-9]+]]:g8rc = LWZ8 0, [[COPY]] :: (load (s32))
- ; CHECK-NEXT: [[EXTSW_32_:%[0-9]+]]:gprc = EXTSW_32 [[LWZ8_]].sub_32
+ ; CHECK-NEXT: [[LWA_32_:%[0-9]+]]:gprc = LWA_32 0, [[COPY]] :: (load (s32))
; CHECK-NEXT: [[DEF:%[0-9]+]]:g8rc = IMPLICIT_DEF
- ; CHECK-NEXT: [[INSERT_SUBREG:%[0-9]+]]:g8rc = INSERT_SUBREG [[DEF]], [[EXTSW_32_]], %subreg.sub_32
+ ; CHECK-NEXT: [[DEF1:%[0-9]+]]:g8rc = IMPLICIT_DEF
+ ; CHECK-NEXT: [[INSERT_SUBREG:%[0-9]+]]:g8rc = INSERT_SUBREG [[DEF1]], [[LWA_32_]], %subreg.sub_32
; CHECK-NEXT: $x3 = COPY [[INSERT_SUBREG]]
; CHECK-NEXT: BLR8 implicit $lr8, implicit $rm, implicit $x3
%0:g8rc_and_g8rc_nox0 = COPY $x3
@@ -113,10 +113,10 @@ body: |
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[COPY:%[0-9]+]]:g8rc_and_g8rc_nox0 = COPY $x3
; CHECK-NEXT: [[COPY1:%[0-9]+]]:g8rc = COPY $x4
- ; CHECK-NEXT: [[LWZX8_:%[0-9]+]]:g8rc = LWZX8 [[COPY]], [[COPY1]] :: (load (s32))
- ; CHECK-NEXT: [[EXTSW_32_:%[0-9]+]]:gprc = EXTSW_32 [[LWZX8_]].sub_32
+ ; CHECK-NEXT: [[LWAX_32_:%[0-9]+]]:gprc = LWAX_32 [[COPY]], [[COPY1]] :: (load (s32))
; CHECK-NEXT: [[DEF:%[0-9]+]]:g8rc = IMPLICIT_DEF
- ; CHECK-NEXT: [[INSERT_SUBREG:%[0-9]+]]:g8rc = INSERT_SUBREG [[DEF]], [[EXTSW_32_]], %subreg.sub_32
+ ; CHECK-NEXT: [[DEF1:%[0-9]+]]:g8rc = IMPLICIT_DEF
+ ; CHECK-NEXT: [[INSERT_SUBREG:%[0-9]+]]:g8rc = INSERT_SUBREG [[DEF1]], [[LWAX_32_]], %subreg.sub_32
; CHECK-NEXT: $x3 = COPY [[INSERT_SUBREG]]
; CHECK-NEXT: BLR8 implicit $lr8, implicit $rm, implicit $x3
%0:g8rc_and_g8rc_nox0 = COPY $x3
More information about the llvm-branch-commits
mailing list