[llvm] [AArch64] Share materialized constant between add x, k and sub x,k (PR #225430)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 27 02:58:16 PDT 2026
https://github.com/prajapati-git updated https://github.com/llvm/llvm-project/pull/225430
>From d8c27917457d205498ba9dee277feb1218c5e16c Mon Sep 17 00:00:00 2001
From: Pratham <prathamkumar882 at gmail.com>
Date: Sun, 27 Sep 2026 15:27:07 +0530
Subject: [PATCH] [AArch64] Move add/sub constant sharing to a post-isel
MachineIR pass
---
.../Target/AArch64/AArch64MIPeepholeOpt.cpp | 90 +++++++++++++++
llvm/test/CodeGen/AArch64/pr222095.ll | 105 ++++++++++++++++++
2 files changed, 195 insertions(+)
create mode 100644 llvm/test/CodeGen/AArch64/pr222095.ll
diff --git a/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp b/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp
index 554d5938cf2cd..a26a6296b024b 100644
--- a/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp
+++ b/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp
@@ -72,6 +72,7 @@
#include "AArch64ExpandImm.h"
#include "AArch64InstrInfo.h"
#include "MCTargetDesc/AArch64AddressingModes.h"
+#include "llvm/ADT/DenseMap.h"
#include "llvm/CodeGen/MachineDominators.h"
#include "llvm/CodeGen/MachineLoopInfo.h"
@@ -122,6 +123,10 @@ class AArch64MIPeepholeOptImpl {
bool checkMovImmInstr(MachineInstr &MI, MachineInstr *&MovMI,
MachineInstr *&SubregToRegMI);
+ template <typename T>
+ bool foldSharedAddSubConstant(MachineBasicBlock &MBB, unsigned AddOpc,
+ unsigned SubOpc);
+
template <typename T>
bool visitADDSUB(unsigned PosOpc, unsigned NegOpc, MachineInstr &MI);
template <typename T>
@@ -576,6 +581,86 @@ bool AArch64MIPeepholeOptImpl::checkMovImmInstr(MachineInstr &MI,
return true;
}
+template <typename T>
+bool AArch64MIPeepholeOptImpl::foldSharedAddSubConstant(MachineBasicBlock &MBB,
+ unsigned AddOpc,
+ unsigned SubOpc) {
+ using SignedT = std::make_signed_t<T>;
+
+ struct CanonicalConst {
+ Register Reg;
+ bool IsNegative;
+ };
+ DenseMap<T, CanonicalConst> Canonical;
+ bool Changed = false;
+
+ for (MachineInstr &MI : make_early_inc_range(MBB)) {
+ unsigned Opc = MI.getOpcode();
+ if (Opc != AddOpc && Opc != SubOpc)
+ continue;
+
+ Register SrcReg = MI.getOperand(2).getReg();
+ if (!SrcReg.isVirtual())
+ continue;
+ MachineInstr *MovMI = MRI->getUniqueVRegDef(SrcReg);
+ if (!MovMI)
+ continue;
+ MachineInstr *SubregToRegMI = nullptr;
+ if (MovMI->getOpcode() == TargetOpcode::SUBREG_TO_REG) {
+ SubregToRegMI = MovMI;
+ MovMI = MRI->getUniqueVRegDef(MovMI->getOperand(1).getReg());
+ if (!MovMI)
+ continue;
+ }
+ if (MovMI->getOpcode() != AArch64::MOVi32imm &&
+ MovMI->getOpcode() != AArch64::MOVi64imm)
+ continue;
+
+ MachineLoop *L = MLI->getLoopFor(&MBB);
+ if (L && !L->isLoopInvariant(MI))
+ continue;
+
+ T Imm = static_cast<T>(MovMI->getOperand(1).getImm());
+ if (SubregToRegMI)
+ Imm &= 0xFFFFFFFF;
+ if (Imm == 0)
+ continue;
+
+ SignedT SImm = static_cast<SignedT>(Imm);
+ bool IsNegative = SImm < 0;
+ if (IsNegative && static_cast<T>(-SImm) == Imm)
+ continue;
+ T Key = IsNegative ? static_cast<T>(-SImm) : Imm;
+
+ auto It = Canonical.find(Key);
+ if (It == Canonical.end()) {
+ Canonical[Key] = {SrcReg, IsNegative};
+ continue;
+ }
+
+ CanonicalConst &Canon = It->second;
+ if (Canon.Reg == SrcReg)
+ continue;
+
+ bool EffectIsAdd = (Opc == AddOpc) == !IsNegative;
+ bool NewIsAdd = EffectIsAdd == !Canon.IsNegative;
+ unsigned NewOpc = NewIsAdd ? AddOpc : SubOpc;
+
+ MI.getOperand(2).setReg(Canon.Reg);
+ MI.getOperand(2).setIsKill(false);
+ MI.setDesc(TII->get(NewOpc));
+
+ if (MRI->use_nodbg_empty(SrcReg)) {
+ if (SubregToRegMI)
+ SubregToRegMI->eraseFromParent();
+ MovMI->eraseFromParent();
+ }
+ Changed = true;
+ }
+
+ return Changed;
+}
+
template <typename T>
bool AArch64MIPeepholeOptImpl::splitTwoPartImm(MachineInstr &MI,
SplitAndOpcFunc<T> SplitAndOpc,
@@ -968,6 +1053,11 @@ bool AArch64MIPeepholeOptImpl::run(MachineFunction &MF) {
bool Changed = false;
for (MachineBasicBlock &MBB : MF) {
+ Changed |= foldSharedAddSubConstant<uint32_t>(MBB, AArch64::ADDWrr,
+ AArch64::SUBWrr);
+ Changed |= foldSharedAddSubConstant<uint64_t>(MBB, AArch64::ADDXrr,
+ AArch64::SUBXrr);
+
for (MachineInstr &MI : make_early_inc_range(MBB)) {
switch (MI.getOpcode()) {
default:
diff --git a/llvm/test/CodeGen/AArch64/pr222095.ll b/llvm/test/CodeGen/AArch64/pr222095.ll
new file mode 100644
index 0000000000000..35388a4c1f865
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/pr222095.ll
@@ -0,0 +1,105 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=aarch64-linux-gnu < %s | FileCheck %s
+
+; https://github.com/llvm/llvm-project/issues/222095
+; When the same constant magnitude is needed once added and once subtracted,
+; and that magnitude doesn't fit as a direct immediate on this target, the
+; positive form should be materialized once and shared (via a plain `sub`)
+; rather than also materializing an unrelated negated constant.
+
+define i32 @add_sub_shared_const_i32(i32 %a, i32 %b) {
+; CHECK-LABEL: add_sub_shared_const_i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: mov w8, #4097 // =0x1001
+; CHECK-NEXT: add w0, w0, w8
+; CHECK-NEXT: sub w1, w1, w8
+; CHECK-NEXT: bl use32
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: ret
+ %add = add i32 %a, 4097
+ %sub = add i32 %b, -4097
+ %r = call i32 @use32(i32 %add, i32 %sub)
+ ret i32 %r
+}
+declare i32 @use32(i32, i32)
+
+define i64 @add_sub_shared_const_i64(i64 %a, i64 %b) {
+; CHECK-LABEL: add_sub_shared_const_i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: mov w8, #4097 // =0x1001
+; CHECK-NEXT: add x0, x0, x8
+; CHECK-NEXT: sub x1, x1, x8
+; CHECK-NEXT: bl use64
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: ret
+ %add = add i64 %a, 4097
+ %sub = add i64 %b, -4097
+ %r = call i64 @use64(i64 %add, i64 %sub)
+ ret i64 %r
+}
+declare i64 @use64(i64, i64)
+
+; Three or more shared uses in the same block, mixing add and sub, should all
+; converge on one materialized register.
+define i32 @add_sub_shared_const_multi_use(i32 %a, i32 %b, i32 %c) {
+; CHECK-LABEL: add_sub_shared_const_multi_use:
+; CHECK: // %bb.0:
+; CHECK-NEXT: str x30, [sp, #-32]! // 8-byte Folded Spill
+; CHECK-NEXT: stp x20, x19, [sp, #16] // 16-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 32
+; CHECK-NEXT: .cfi_offset w19, -8
+; CHECK-NEXT: .cfi_offset w20, -16
+; CHECK-NEXT: .cfi_offset w30, -32
+; CHECK-NEXT: mov w20, #4097 // =0x1001
+; CHECK-NEXT: mov w19, w2
+; CHECK-NEXT: add w0, w0, w20
+; CHECK-NEXT: sub w1, w1, w20
+; CHECK-NEXT: bl use32
+; CHECK-NEXT: sub w1, w19, w20
+; CHECK-NEXT: bl use32
+; CHECK-NEXT: ldp x20, x19, [sp, #16] // 16-byte Folded Reload
+; CHECK-NEXT: ldr x30, [sp], #32 // 8-byte Folded Reload
+; CHECK-NEXT: ret
+ %add = add i32 %a, 4097
+ %sub1 = add i32 %b, -4097
+ %sub2 = add i32 %c, -4097
+ %r1 = call i32 @use32(i32 %add, i32 %sub1)
+ %r2 = call i32 @use32(i32 %r1, i32 %sub2)
+ ret i32 %r2
+}
+
+; The same tie-break must not fire when the other operand is a multiply:
+; madd can fold the constant directly into its accumulator, which is
+; strictly better than materializing the positive magnitude and using a
+; separate sub. This is a regression guard for that specific interaction.
+define i64 @add_neg_tied_const_does_not_break_madd(i64 %a) {
+; CHECK-LABEL: add_neg_tied_const_does_not_break_madd:
+; CHECK: // %bb.0:
+; CHECK-NEXT: mov w8, #37 // =0x25
+; CHECK-NEXT: mov x9, #-32888 // =0xffffffffffff7f88
+; CHECK-NEXT: movk x9, #65518, lsl #16
+; CHECK-NEXT: madd x0, x0, x8, x9
+; CHECK-NEXT: ret
+ %tmp0 = add i64 %a, -31000
+ %tmp1 = mul i64 %tmp0, 37
+ ret i64 %tmp1
+}
+
+; Self-negating edge case: for i32, -2147483648 (INT_MIN) negated wraps back
+; to itself, so there is no positive magnitude to converge on. Must not
+; needlessly flip the opcode for no benefit.
+define i32 @add_self_negating_const(i32 %a) {
+; CHECK-LABEL: add_self_negating_const:
+; CHECK: // %bb.0:
+; CHECK-NEXT: mov w8, #-2147483648 // =0x80000000
+; CHECK-NEXT: add w0, w0, w8
+; CHECK-NEXT: ret
+ %add = add i32 %a, -2147483648
+ ret i32 %add
+}
More information about the llvm-commits
mailing list