[llvm] [AArch64] Share materialized constant between add x, k and sub x,k (PR #225430)

via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 27 02:58:16 PDT 2026


https://github.com/prajapati-git updated https://github.com/llvm/llvm-project/pull/225430

>From d8c27917457d205498ba9dee277feb1218c5e16c Mon Sep 17 00:00:00 2001
From: Pratham <prathamkumar882 at gmail.com>
Date: Sun, 27 Sep 2026 15:27:07 +0530
Subject: [PATCH] [AArch64] Move add/sub constant sharing to a post-isel
 MachineIR pass

---
 .../Target/AArch64/AArch64MIPeepholeOpt.cpp   |  90 +++++++++++++++
 llvm/test/CodeGen/AArch64/pr222095.ll         | 105 ++++++++++++++++++
 2 files changed, 195 insertions(+)
 create mode 100644 llvm/test/CodeGen/AArch64/pr222095.ll

diff --git a/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp b/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp
index 554d5938cf2cd..a26a6296b024b 100644
--- a/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp
+++ b/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp
@@ -72,6 +72,7 @@
 #include "AArch64ExpandImm.h"
 #include "AArch64InstrInfo.h"
 #include "MCTargetDesc/AArch64AddressingModes.h"
+#include "llvm/ADT/DenseMap.h"
 #include "llvm/CodeGen/MachineDominators.h"
 #include "llvm/CodeGen/MachineLoopInfo.h"
 
@@ -122,6 +123,10 @@ class AArch64MIPeepholeOptImpl {
   bool checkMovImmInstr(MachineInstr &MI, MachineInstr *&MovMI,
                         MachineInstr *&SubregToRegMI);
 
+  template <typename T>
+  bool foldSharedAddSubConstant(MachineBasicBlock &MBB, unsigned AddOpc,
+                                unsigned SubOpc);
+
   template <typename T>
   bool visitADDSUB(unsigned PosOpc, unsigned NegOpc, MachineInstr &MI);
   template <typename T>
@@ -576,6 +581,86 @@ bool AArch64MIPeepholeOptImpl::checkMovImmInstr(MachineInstr &MI,
   return true;
 }
 
+template <typename T>
+bool AArch64MIPeepholeOptImpl::foldSharedAddSubConstant(MachineBasicBlock &MBB,
+                                                        unsigned AddOpc,
+                                                        unsigned SubOpc) {
+  using SignedT = std::make_signed_t<T>;
+
+  struct CanonicalConst {
+    Register Reg;
+    bool IsNegative;
+  };
+  DenseMap<T, CanonicalConst> Canonical;
+  bool Changed = false;
+
+  for (MachineInstr &MI : make_early_inc_range(MBB)) {
+    unsigned Opc = MI.getOpcode();
+    if (Opc != AddOpc && Opc != SubOpc)
+      continue;
+
+    Register SrcReg = MI.getOperand(2).getReg();
+    if (!SrcReg.isVirtual())
+      continue;
+    MachineInstr *MovMI = MRI->getUniqueVRegDef(SrcReg);
+    if (!MovMI)
+      continue;
+    MachineInstr *SubregToRegMI = nullptr;
+    if (MovMI->getOpcode() == TargetOpcode::SUBREG_TO_REG) {
+      SubregToRegMI = MovMI;
+      MovMI = MRI->getUniqueVRegDef(MovMI->getOperand(1).getReg());
+      if (!MovMI)
+        continue;
+    }
+    if (MovMI->getOpcode() != AArch64::MOVi32imm &&
+        MovMI->getOpcode() != AArch64::MOVi64imm)
+      continue;
+
+    MachineLoop *L = MLI->getLoopFor(&MBB);
+    if (L && !L->isLoopInvariant(MI))
+      continue;
+
+    T Imm = static_cast<T>(MovMI->getOperand(1).getImm());
+    if (SubregToRegMI)
+      Imm &= 0xFFFFFFFF;
+    if (Imm == 0)
+      continue;
+
+    SignedT SImm = static_cast<SignedT>(Imm);
+    bool IsNegative = SImm < 0;
+    if (IsNegative && static_cast<T>(-SImm) == Imm)
+      continue;
+    T Key = IsNegative ? static_cast<T>(-SImm) : Imm;
+
+    auto It = Canonical.find(Key);
+    if (It == Canonical.end()) {
+      Canonical[Key] = {SrcReg, IsNegative};
+      continue;
+    }
+
+    CanonicalConst &Canon = It->second;
+    if (Canon.Reg == SrcReg)
+      continue;
+
+    bool EffectIsAdd = (Opc == AddOpc) == !IsNegative;
+    bool NewIsAdd = EffectIsAdd == !Canon.IsNegative;
+    unsigned NewOpc = NewIsAdd ? AddOpc : SubOpc;
+
+    MI.getOperand(2).setReg(Canon.Reg);
+    MI.getOperand(2).setIsKill(false);
+    MI.setDesc(TII->get(NewOpc));
+
+    if (MRI->use_nodbg_empty(SrcReg)) {
+      if (SubregToRegMI)
+        SubregToRegMI->eraseFromParent();
+      MovMI->eraseFromParent();
+    }
+    Changed = true;
+  }
+
+  return Changed;
+}
+
 template <typename T>
 bool AArch64MIPeepholeOptImpl::splitTwoPartImm(MachineInstr &MI,
                                                SplitAndOpcFunc<T> SplitAndOpc,
@@ -968,6 +1053,11 @@ bool AArch64MIPeepholeOptImpl::run(MachineFunction &MF) {
   bool Changed = false;
 
   for (MachineBasicBlock &MBB : MF) {
+    Changed |= foldSharedAddSubConstant<uint32_t>(MBB, AArch64::ADDWrr,
+                                                  AArch64::SUBWrr);
+    Changed |= foldSharedAddSubConstant<uint64_t>(MBB, AArch64::ADDXrr,
+                                                  AArch64::SUBXrr);
+
     for (MachineInstr &MI : make_early_inc_range(MBB)) {
       switch (MI.getOpcode()) {
       default:
diff --git a/llvm/test/CodeGen/AArch64/pr222095.ll b/llvm/test/CodeGen/AArch64/pr222095.ll
new file mode 100644
index 0000000000000..35388a4c1f865
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/pr222095.ll
@@ -0,0 +1,105 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=aarch64-linux-gnu < %s | FileCheck %s
+
+; https://github.com/llvm/llvm-project/issues/222095
+; When the same constant magnitude is needed once added and once subtracted,
+; and that magnitude doesn't fit as a direct immediate on this target, the
+; positive form should be materialized once and shared (via a plain `sub`)
+; rather than also materializing an unrelated negated constant.
+
+define i32 @add_sub_shared_const_i32(i32 %a, i32 %b) {
+; CHECK-LABEL: add_sub_shared_const_i32:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    .cfi_offset w30, -16
+; CHECK-NEXT:    mov w8, #4097 // =0x1001
+; CHECK-NEXT:    add w0, w0, w8
+; CHECK-NEXT:    sub w1, w1, w8
+; CHECK-NEXT:    bl use32
+; CHECK-NEXT:    ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT:    ret
+  %add = add i32 %a, 4097
+  %sub = add i32 %b, -4097
+  %r = call i32 @use32(i32 %add, i32 %sub)
+  ret i32 %r
+}
+declare i32 @use32(i32, i32)
+
+define i64 @add_sub_shared_const_i64(i64 %a, i64 %b) {
+; CHECK-LABEL: add_sub_shared_const_i64:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    .cfi_offset w30, -16
+; CHECK-NEXT:    mov w8, #4097 // =0x1001
+; CHECK-NEXT:    add x0, x0, x8
+; CHECK-NEXT:    sub x1, x1, x8
+; CHECK-NEXT:    bl use64
+; CHECK-NEXT:    ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT:    ret
+  %add = add i64 %a, 4097
+  %sub = add i64 %b, -4097
+  %r = call i64 @use64(i64 %add, i64 %sub)
+  ret i64 %r
+}
+declare i64 @use64(i64, i64)
+
+; Three or more shared uses in the same block, mixing add and sub, should all
+; converge on one materialized register.
+define i32 @add_sub_shared_const_multi_use(i32 %a, i32 %b, i32 %c) {
+; CHECK-LABEL: add_sub_shared_const_multi_use:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    str x30, [sp, #-32]! // 8-byte Folded Spill
+; CHECK-NEXT:    stp x20, x19, [sp, #16] // 16-byte Folded Spill
+; CHECK-NEXT:    .cfi_def_cfa_offset 32
+; CHECK-NEXT:    .cfi_offset w19, -8
+; CHECK-NEXT:    .cfi_offset w20, -16
+; CHECK-NEXT:    .cfi_offset w30, -32
+; CHECK-NEXT:    mov w20, #4097 // =0x1001
+; CHECK-NEXT:    mov w19, w2
+; CHECK-NEXT:    add w0, w0, w20
+; CHECK-NEXT:    sub w1, w1, w20
+; CHECK-NEXT:    bl use32
+; CHECK-NEXT:    sub w1, w19, w20
+; CHECK-NEXT:    bl use32
+; CHECK-NEXT:    ldp x20, x19, [sp, #16] // 16-byte Folded Reload
+; CHECK-NEXT:    ldr x30, [sp], #32 // 8-byte Folded Reload
+; CHECK-NEXT:    ret
+  %add = add i32 %a, 4097
+  %sub1 = add i32 %b, -4097
+  %sub2 = add i32 %c, -4097
+  %r1 = call i32 @use32(i32 %add, i32 %sub1)
+  %r2 = call i32 @use32(i32 %r1, i32 %sub2)
+  ret i32 %r2
+}
+
+; The same tie-break must not fire when the other operand is a multiply:
+; madd can fold the constant directly into its accumulator, which is
+; strictly better than materializing the positive magnitude and using a
+; separate sub. This is a regression guard for that specific interaction.
+define i64 @add_neg_tied_const_does_not_break_madd(i64 %a) {
+; CHECK-LABEL: add_neg_tied_const_does_not_break_madd:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    mov w8, #37 // =0x25
+; CHECK-NEXT:    mov x9, #-32888 // =0xffffffffffff7f88
+; CHECK-NEXT:    movk x9, #65518, lsl #16
+; CHECK-NEXT:    madd x0, x0, x8, x9
+; CHECK-NEXT:    ret
+  %tmp0 = add i64 %a, -31000
+  %tmp1 = mul i64 %tmp0, 37
+  ret i64 %tmp1
+}
+
+; Self-negating edge case: for i32, -2147483648 (INT_MIN) negated wraps back
+; to itself, so there is no positive magnitude to converge on. Must not
+; needlessly flip the opcode for no benefit.
+define i32 @add_self_negating_const(i32 %a) {
+; CHECK-LABEL: add_self_negating_const:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    mov w8, #-2147483648 // =0x80000000
+; CHECK-NEXT:    add w0, w0, w8
+; CHECK-NEXT:    ret
+  %add = add i32 %a, -2147483648
+  ret i32 %add
+}



More information about the llvm-commits mailing list