[llvm] b885cfb - [LFI][AArch64] Add guard elimination optimization (#204693)

via llvm-commits llvm-commits at lists.llvm.org
Tue Jun 30 02:19:38 PDT 2026


Author: Zachary Yedidia
Date: 2026-06-30T02:19:34-07:00
New Revision: b885cfbb055839ea0aa9b2c0fdd95357cae8dfa6

URL: https://github.com/llvm/llvm-project/commit/b885cfbb055839ea0aa9b2c0fdd95357cae8dfa6
DIFF: https://github.com/llvm/llvm-project/commit/b885cfbb055839ea0aa9b2c0fdd95357cae8dfa6.diff

LOG: [LFI][AArch64] Add guard elimination optimization (#204693)

This adds support for the guard elimination optimization to the AArch64
LFI rewriter. Redundant guards (`add x28, x27, wN, uxtw` instructions)
will be skipped when possible. See the LFI.rst documentation for an
example of the optimization.

Added: 
    llvm/test/MC/AArch64/LFI/guard-elim.s

Modified: 
    llvm/docs/LFI.rst
    llvm/lib/Target/AArch64/MCTargetDesc/AArch64MCLFIRewriter.cpp
    llvm/lib/Target/AArch64/MCTargetDesc/AArch64MCLFIRewriter.h
    llvm/test/MC/AArch64/LFI/exclusive.s
    llvm/test/MC/AArch64/LFI/fp.s
    llvm/test/MC/AArch64/LFI/lse.s
    llvm/test/MC/AArch64/LFI/mem-lr.s
    llvm/test/MC/AArch64/LFI/mem.s
    llvm/test/MC/AArch64/LFI/prefetch.s
    llvm/test/MC/AArch64/LFI/rcpc.s
    llvm/test/MC/AArch64/LFI/simd.s

Removed: 
    


################################################################################
diff  --git a/llvm/docs/LFI.rst b/llvm/docs/LFI.rst
index b0f5e31d87ee9..74cfe02c76f21 100644
--- a/llvm/docs/LFI.rst
+++ b/llvm/docs/LFI.rst
@@ -365,8 +365,6 @@ Optimizations
 Basic guard elimination
 ~~~~~~~~~~~~~~~~~~~~~~~
 
-**Note**: not yet implemented.
-
 If a register is guarded multiple times in the same basic block without any
 modifications to it during the intervening instructions, then subsequent guards
 can be removed.

diff  --git a/llvm/lib/Target/AArch64/MCTargetDesc/AArch64MCLFIRewriter.cpp b/llvm/lib/Target/AArch64/MCTargetDesc/AArch64MCLFIRewriter.cpp
index 03a5e83db008f..29f2ace01ade3 100644
--- a/llvm/lib/Target/AArch64/MCTargetDesc/AArch64MCLFIRewriter.cpp
+++ b/llvm/lib/Target/AArch64/MCTargetDesc/AArch64MCLFIRewriter.cpp
@@ -22,9 +22,15 @@
 #include "llvm/MC/MCInstrInfo.h"
 #include "llvm/MC/MCStreamer.h"
 #include "llvm/MC/MCSubtargetInfo.h"
+#include "llvm/Support/CommandLine.h"
 
 using namespace llvm;
 
+static cl::opt<bool>
+    LFIGuardElim("aarch64-lfi-guard-elim", cl::Hidden,
+                 cl::desc("Enable the LFI guard elimination optimization"),
+                 cl::init(true));
+
 namespace llvm::AArch64 {
 struct LFIVariantEntry {
   unsigned Inst;
@@ -245,14 +251,35 @@ MCRegister AArch64MCLFIRewriter::mayModifyReserved(const MCInst &Inst) const {
   return {};
 }
 
+void AArch64MCLFIRewriter::onLabel(const MCSymbol *) {
+  // Invalidate guard state since the label is a potential branch target.
+  ActiveGuardReg = std::nullopt;
+}
+
 void AArch64MCLFIRewriter::emitInst(const MCInst &Inst, MCStreamer &Out,
                                     const MCSubtargetInfo &STI) {
+  // Invalidate the active guard if this instruction modifies the guarded
+  // register, modifies x28 itself, or may affect control flow.
+  if (ActiveGuardReg) {
+    const MCInstrDesc &Desc = InstInfo->get(Inst.getOpcode());
+    if (Desc.mayAffectControlFlow(Inst, *RegInfo) ||
+        mayModifyRegister(Inst, *ActiveGuardReg) ||
+        mayModifyRegister(Inst, getWRegFromXReg(*ActiveGuardReg)) ||
+        mayModifyRegister(Inst, LFIAddrReg))
+      ActiveGuardReg = std::nullopt;
+  }
+
   Out.emitInstruction(Inst, STI);
 }
 
 void AArch64MCLFIRewriter::emitAddMask(MCRegister Dest, MCRegister Src,
                                        MCStreamer &Out,
                                        const MCSubtargetInfo &STI) {
+  // If x28 already holds the guarded value of Src, this guard is redundant and
+  // can be skipped.
+  if (LFIGuardElim && Dest == LFIAddrReg && ActiveGuardReg == Src)
+    return;
+
   // add Dest, LFIBaseReg, W(Src), uxtw
   MCInst Inst;
   Inst.setOpcode(AArch64::ADDXrx);
@@ -262,6 +289,10 @@ void AArch64MCLFIRewriter::emitAddMask(MCRegister Dest, MCRegister Src,
   Inst.addOperand(
       MCOperand::createImm(AArch64_AM::getArithExtendImm(AArch64_AM::UXTW, 0)));
   emitInst(Inst, Out, STI);
+
+  // Record Src as the new active guard.
+  if (Dest == LFIAddrReg)
+    ActiveGuardReg = Src;
 }
 
 void AArch64MCLFIRewriter::emitBranch(unsigned Opcode, MCRegister Target,
@@ -818,8 +849,12 @@ bool llvm::isLFIPrePostMemAccess(unsigned Opcode) {
 
 bool AArch64MCLFIRewriter::rewriteInst(const MCInst &Inst, MCStreamer &Out,
                                        const MCSubtargetInfo &STI) {
-  // The guard prevents rewrite-recursion when we emit instructions from inside
-  // the rewriter (such instructions should not be rewritten).
+  // Invalidate guard state if the rewriter was manually disabled.
+  if (!Enabled)
+    ActiveGuardReg = std::nullopt;
+
+  // This recursion guard prevents rewrite-recursion when we emit instructions
+  // from inside the rewriter (such instructions should not be rewritten).
   if (!Enabled || Guard)
     return false;
   Guard = true;

diff  --git a/llvm/lib/Target/AArch64/MCTargetDesc/AArch64MCLFIRewriter.h b/llvm/lib/Target/AArch64/MCTargetDesc/AArch64MCLFIRewriter.h
index b04da5c2c2d75..b21c381393cfd 100644
--- a/llvm/lib/Target/AArch64/MCTargetDesc/AArch64MCLFIRewriter.h
+++ b/llvm/lib/Target/AArch64/MCTargetDesc/AArch64MCLFIRewriter.h
@@ -19,6 +19,8 @@
 #include "llvm/MC/MCRegister.h"
 #include "llvm/MC/MCRegisterInfo.h"
 
+#include <optional>
+
 namespace llvm {
 class MCContext;
 class MCExpr;
@@ -26,6 +28,7 @@ class MCInst;
 class MCOperand;
 class MCStreamer;
 class MCSubtargetInfo;
+class MCSymbol;
 
 /// Rewrites AArch64 instructions for LFI sandboxing.
 ///
@@ -49,6 +52,8 @@ class AArch64MCLFIRewriter : public MCLFIRewriter {
   bool rewriteInst(const MCInst &Inst, MCStreamer &Out,
                    const MCSubtargetInfo &STI) override;
 
+  void onLabel(const MCSymbol *Symbol) override;
+
 private:
   /// Recursion guard to prevent infinite loops when emitting instructions.
   bool Guard = false;
@@ -59,6 +64,11 @@ class AArch64MCLFIRewriter : public MCLFIRewriter {
   /// the guard and the branch so the relocation stays on the BLR.
   const MCExpr *PendingTLSDescCall = nullptr;
 
+  /// Rewriter state for implementing the guard-elimination optimization, which
+  /// allows redundant add masks to be skipped. When it holds a value, x28 is
+  /// known to already hold the guarded value of that register.
+  std::optional<MCRegister> ActiveGuardReg;
+
   // Instruction classification. Returns the reserved register that may be
   // modified, or an invalid register if no reserved register is touched.
   MCRegister mayModifyReserved(const MCInst &Inst) const;

diff  --git a/llvm/test/MC/AArch64/LFI/exclusive.s b/llvm/test/MC/AArch64/LFI/exclusive.s
index eb42bd234848c..d9b961c4fe8c6 100644
--- a/llvm/test/MC/AArch64/LFI/exclusive.s
+++ b/llvm/test/MC/AArch64/LFI/exclusive.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple aarch64_lfi %s | FileCheck %s
+// RUN: llvm-mc -triple aarch64_lfi --aarch64-lfi-guard-elim=false %s | FileCheck %s
 
 // Load exclusive
 ldxr x0, [x1]

diff  --git a/llvm/test/MC/AArch64/LFI/fp.s b/llvm/test/MC/AArch64/LFI/fp.s
index d1fc4aac25a23..ca00e60c03cc2 100644
--- a/llvm/test/MC/AArch64/LFI/fp.s
+++ b/llvm/test/MC/AArch64/LFI/fp.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple aarch64_lfi %s | FileCheck %s
+// RUN: llvm-mc -triple aarch64_lfi --aarch64-lfi-guard-elim=false %s | FileCheck %s
 
 // FP/SIMD scalar loads (zero offset -> RoW)
 ldr b0, [x1]

diff  --git a/llvm/test/MC/AArch64/LFI/guard-elim.s b/llvm/test/MC/AArch64/LFI/guard-elim.s
new file mode 100644
index 0000000000000..8a684027f4180
--- /dev/null
+++ b/llvm/test/MC/AArch64/LFI/guard-elim.s
@@ -0,0 +1,176 @@
+// RUN: llvm-mc -triple aarch64_lfi %s | FileCheck %s
+
+// Consecutive loads from the same register share a guard.
+ldr x0, [x1, #8]
+ldr x2, [x1, #16]
+// CHECK:      add x28, x27, w1, uxtw
+// CHECK-NEXT: ldr x0, [x28, #8]
+// CHECK-NEXT: ldr x2, [x28, #16]
+
+// Modifying the base register invalidates the guard.
+ldr x4, [x3, #8]
+add x3, x3, #24
+ldr x5, [x3, #8]
+// CHECK:      add x28, x27, w3, uxtw
+// CHECK-NEXT: ldr x4, [x28, #8]
+// CHECK-NEXT: add x3, x3, #24
+// CHECK-NEXT: add x28, x27, w3, uxtw
+// CHECK-NEXT: ldr x5, [x28, #8]
+
+// A 
diff erent base register requires a new guard.
+ldr x6, [x4, #8]
+ldr x7, [x5, #8]
+// CHECK:      add x28, x27, w4, uxtw
+// CHECK-NEXT: ldr x6, [x28, #8]
+// CHECK-NEXT: add x28, x27, w5, uxtw
+// CHECK-NEXT: ldr x7, [x28, #8]
+
+// Labels invalidate the guard.
+label_boundary_test:
+ldr x8, [x6, #8]
+label1:
+ldr x9, [x6, #16]
+// CHECK-LABEL: label_boundary_test:
+// CHECK-NEXT: add x28, x27, w6, uxtw
+// CHECK-NEXT: ldr x8, [x28, #8]
+// CHECK-NEXT: label1:
+// CHECK-NEXT: add x28, x27, w6, uxtw
+// CHECK-NEXT: ldr x9, [x28, #16]
+
+// Branches invalidate the guard.
+control_flow_test:
+ldr x10, [x7, #8]
+b label2
+ldr x11, [x7, #16]
+label2:
+// CHECK-LABEL: control_flow_test:
+// CHECK-NEXT: add x28, x27, w7, uxtw
+// CHECK-NEXT: ldr x10, [x28, #8]
+// CHECK-NEXT: b label2
+// CHECK-NEXT: add x28, x27, w7, uxtw
+// CHECK-NEXT: ldr x11, [x28, #16]
+// CHECK-NEXT: label2:
+
+// Modifying the W subregister invalidates the X guard.
+w_reg_modification:
+ldr x12, [x8, #8]
+mov w8, #0
+ldr x13, [x8, #16]
+// CHECK-LABEL: w_reg_modification:
+// CHECK-NEXT: add x28, x27, w8, uxtw
+// CHECK-NEXT: ldr x12, [x28, #8]
+// CHECK-NEXT: mov w8, #0
+// CHECK-NEXT: add x28, x27, w8, uxtw
+// CHECK-NEXT: ldr x13, [x28, #16]
+
+// Multiple consecutive accesses share a single guard.
+multiple_accesses:
+ldr x14, [x9, #8]
+ldr x15, [x9, #16]
+ldr x16, [x9, #24]
+str x17, [x9, #32]
+// CHECK-LABEL: multiple_accesses:
+// CHECK-NEXT: add x28, x27, w9, uxtw
+// CHECK-NEXT: ldr x14, [x28, #8]
+// CHECK-NEXT: ldr x15, [x28, #16]
+// CHECK-NEXT: ldr x16, [x28, #24]
+// CHECK-NEXT: str x17, [x28, #32]
+
+// Mixed loads and stores share a guard.
+mixed_load_store:
+str x18, [x10, #8]
+ldr x19, [x10, #16]
+str x20, [x10, #24]
+// CHECK-LABEL: mixed_load_store:
+// CHECK-NEXT: add x28, x27, w10, uxtw
+// CHECK-NEXT: str x18, [x28, #8]
+// CHECK-NEXT: ldr x19, [x28, #16]
+// CHECK-NEXT: str x20, [x28, #24]
+
+// Instructions that don't touch x28 or the guarded register keep the guard.
+non_modifying_between:
+ldr x21, [x11, #8]
+mov x0, x1
+add x2, x3, x4
+ldr x22, [x11, #16]
+// CHECK-LABEL: non_modifying_between:
+// CHECK-NEXT: add x28, x27, w11, uxtw
+// CHECK-NEXT: ldr x21, [x28, #8]
+// CHECK-NEXT: mov x0, x1
+// CHECK-NEXT: add x2, x3, x4
+// CHECK-NEXT: ldr x22, [x28, #16]
+
+// Post-index pair writeback invalidates the guard.
+prepost_ldp:
+ldp x0, x1, [x2]
+ldp x3, x4, [x2], #16
+ldp x5, x6, [x2]
+// CHECK-LABEL: prepost_ldp:
+// CHECK-NEXT: add x28, x27, w2, uxtw
+// CHECK-NEXT: ldp x0, x1, [x28]
+// CHECK-NEXT: ldp x3, x4, [x28]
+// CHECK-NEXT: add x2, x2, #16
+// CHECK-NEXT: add x28, x27, w2, uxtw
+// CHECK-NEXT: ldp x5, x6, [x28]
+
+prepost_stp:
+stp x0, x1, [x2]
+stp x3, x4, [x2], #16
+stp x5, x6, [x2]
+// CHECK-LABEL: prepost_stp:
+// CHECK-NEXT: add x28, x27, w2, uxtw
+// CHECK-NEXT: stp x0, x1, [x28]
+// CHECK-NEXT: stp x3, x4, [x28]
+// CHECK-NEXT: add x2, x2, #16
+// CHECK-NEXT: add x28, x27, w2, uxtw
+// CHECK-NEXT: stp x5, x6, [x28]
+
+// Pre-index pair writeback invalidates the guard.
+prepost_ldp_pre:
+ldp x0, x1, [x2]
+ldp x3, x4, [x2, #16]!
+ldp x5, x6, [x2]
+// CHECK-LABEL: prepost_ldp_pre:
+// CHECK-NEXT: add x28, x27, w2, uxtw
+// CHECK-NEXT: ldp x0, x1, [x28]
+// CHECK-NEXT: ldp x3, x4, [x28, #16]
+// CHECK-NEXT: add x2, x2, #16
+// CHECK-NEXT: add x28, x27, w2, uxtw
+// CHECK-NEXT: ldp x5, x6, [x28]
+
+// A load into the base register invalidates the guard.
+load_into_base:
+ldr x1, [x1, #8]
+ldr x2, [x1, #16]
+// CHECK-LABEL: load_into_base:
+// CHECK-NEXT: add x28, x27, w1, uxtw
+// CHECK-NEXT: ldr x1, [x28, #8]
+// CHECK-NEXT: add x28, x27, w1, uxtw
+// CHECK-NEXT: ldr x2, [x28, #16]
+
+// The guard for x28 carries across an LR mask, since the mask touches only x30.
+lr_mask_between:
+ldr x0, [x12, #8]
+ldr x30, [x12, #16]
+ldr x1, [x12, #24]
+// CHECK-LABEL: lr_mask_between:
+// CHECK-NEXT: add x28, x27, w12, uxtw
+// CHECK-NEXT: ldr x0, [x28, #8]
+// CHECK-NEXT: ldr x30, [x28, #16]
+// CHECK-NEXT: add x30, x27, w30, uxtw
+// CHECK-NEXT: ldr x1, [x28, #24]
+
+// A .lfi_rewrite_disable region invalidates the guard, since the instructions
+// inside it bypass the rewriter and may modify the base register or x28.
+rewrite_disable_boundary:
+ldr x0, [x14, #8]
+.lfi_rewrite_disable
+mov x14, x15
+.lfi_rewrite_enable
+ldr x1, [x14, #16]
+// CHECK-LABEL: rewrite_disable_boundary:
+// CHECK-NEXT: add x28, x27, w14, uxtw
+// CHECK-NEXT: ldr x0, [x28, #8]
+// CHECK-NEXT: mov x14, x15
+// CHECK-NEXT: add x28, x27, w14, uxtw
+// CHECK-NEXT: ldr x1, [x28, #16]

diff  --git a/llvm/test/MC/AArch64/LFI/lse.s b/llvm/test/MC/AArch64/LFI/lse.s
index dc9faace32827..488d6bcd48d50 100644
--- a/llvm/test/MC/AArch64/LFI/lse.s
+++ b/llvm/test/MC/AArch64/LFI/lse.s
@@ -1,5 +1,5 @@
-// RUN: llvm-mc -triple aarch64_lfi %s | FileCheck %s
-// RUN: llvm-mc -triple aarch64_lfi -mattr=+no-lfi-loads %s | FileCheck %s
+// RUN: llvm-mc -triple aarch64_lfi --aarch64-lfi-guard-elim=false %s | FileCheck %s
+// RUN: llvm-mc -triple aarch64_lfi -mattr=+no-lfi-loads --aarch64-lfi-guard-elim=false %s | FileCheck %s
 
 .arch_extension lse
 

diff  --git a/llvm/test/MC/AArch64/LFI/mem-lr.s b/llvm/test/MC/AArch64/LFI/mem-lr.s
index c04ede69f5bab..e9b53131042f6 100644
--- a/llvm/test/MC/AArch64/LFI/mem-lr.s
+++ b/llvm/test/MC/AArch64/LFI/mem-lr.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple aarch64_lfi %s | FileCheck %s
+// RUN: llvm-mc -triple aarch64_lfi --aarch64-lfi-guard-elim=false %s | FileCheck %s
 
 // Memory accesses that define LR (x30) must sandbox the base register
 // in addition to masking LR after the access.

diff  --git a/llvm/test/MC/AArch64/LFI/mem.s b/llvm/test/MC/AArch64/LFI/mem.s
index 7e763efdeb803..ded1b26ffbcb3 100644
--- a/llvm/test/MC/AArch64/LFI/mem.s
+++ b/llvm/test/MC/AArch64/LFI/mem.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple aarch64_lfi %s | FileCheck %s
+// RUN: llvm-mc -triple aarch64_lfi --aarch64-lfi-guard-elim=false %s | FileCheck %s
 
 // SP-relative loads/stores (no sandboxing needed)
 ldr x0, [sp]

diff  --git a/llvm/test/MC/AArch64/LFI/prefetch.s b/llvm/test/MC/AArch64/LFI/prefetch.s
index 55b6bb78fa85c..1eeec73c708ba 100644
--- a/llvm/test/MC/AArch64/LFI/prefetch.s
+++ b/llvm/test/MC/AArch64/LFI/prefetch.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple aarch64_lfi %s | FileCheck %s
+// RUN: llvm-mc -triple aarch64_lfi --aarch64-lfi-guard-elim=false %s | FileCheck %s
 
 prfm pldl1keep, [x0]
 // CHECK: prfm pldl1keep, [x27, w0, uxtw]

diff  --git a/llvm/test/MC/AArch64/LFI/rcpc.s b/llvm/test/MC/AArch64/LFI/rcpc.s
index eab3d8ed263d3..b9350c947ef92 100644
--- a/llvm/test/MC/AArch64/LFI/rcpc.s
+++ b/llvm/test/MC/AArch64/LFI/rcpc.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple aarch64_lfi %s | FileCheck %s
+// RUN: llvm-mc -triple aarch64_lfi --aarch64-lfi-guard-elim=false %s | FileCheck %s
 
 .arch_extension rcpc
 

diff  --git a/llvm/test/MC/AArch64/LFI/simd.s b/llvm/test/MC/AArch64/LFI/simd.s
index 8adb203a8df44..b01026f38b78a 100644
--- a/llvm/test/MC/AArch64/LFI/simd.s
+++ b/llvm/test/MC/AArch64/LFI/simd.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple aarch64_lfi %s | FileCheck %s
+// RUN: llvm-mc -triple aarch64_lfi --aarch64-lfi-guard-elim=false %s | FileCheck %s
 
 // LD1/ST1 single structure (no post-index)
 


        


More information about the llvm-commits mailing list