[llvm] [LFI][X86] Add control-flow rewrites (PR #210913)

Zachary Yedidia via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 25 23:35:25 PDT 2026


https://github.com/zyedidia updated https://github.com/llvm/llvm-project/pull/210913

>From 8f7e587c8ab73dced15e51d4ad47c49da4e305fa Mon Sep 17 00:00:00 2001
From: Zachary Yedidia <zyedidia at gmail.com>
Date: Tue, 21 Jul 2026 01:58:03 -0700
Subject: [PATCH 01/10] [LFI][X86] Add control-flow rewrites

---
 llvm/docs/LFI.md                              |  93 +++++++++++-
 llvm/lib/Target/X86/CMakeLists.txt            |   1 +
 .../X86/MCTargetDesc/X86MCLFIRewriter.cpp     | 143 +++++++++++++++++-
 .../X86/MCTargetDesc/X86MCLFIRewriter.h       |  12 ++
 llvm/lib/Target/X86/X86.h                     |   9 ++
 llvm/lib/Target/X86/X86CodeGenPassBuilder.cpp |   3 +
 llvm/lib/Target/X86/X86LFIRewritePass.cpp     | 101 +++++++++++++
 llvm/lib/Target/X86/X86PassRegistry.def       |   1 +
 llvm/lib/Target/X86/X86TargetMachine.cpp      |   3 +
 llvm/test/CodeGen/X86/lfi-align.ll            |  65 ++++++++
 llvm/test/MC/X86/LFI/control-flow-errors.s    |  34 +++++
 llvm/test/MC/X86/LFI/control-flow.s           |  78 ++++++++++
 12 files changed, 540 insertions(+), 3 deletions(-)
 create mode 100644 llvm/lib/Target/X86/X86LFIRewritePass.cpp
 create mode 100644 llvm/test/CodeGen/X86/lfi-align.ll
 create mode 100644 llvm/test/MC/X86/LFI/control-flow-errors.s
 create mode 100644 llvm/test/MC/X86/LFI/control-flow.s

diff --git a/llvm/docs/LFI.md b/llvm/docs/LFI.md
index c9b9815847f52..142381dd12bbb 100644
--- a/llvm/docs/LFI.md
+++ b/llvm/docs/LFI.md
@@ -639,10 +639,101 @@ In the following assembly rewrites, some shorthand is used.
 
 - `%rN` or `%eN`: refers to any general-purpose non-reserved register.
 - `{a,b,c}`: matches any of `a`, `b`, or `c`.
+- `N(...)`: refers to any memory addressing mode.
+
+#### Bundles
+
+The X86-64 target divides the code region into 32-byte aligned *bundles*.
+Indirect branch targets are masked so that they are always bundle-aligned,
+which restricts the set of reachable instructions to bundle boundaries. For
+this to be sound, two additional properties are required:
+
+- A rewrite sequence must never be split across a bundle boundary, otherwise
+  control could be transferred into the middle of the sequence, skipping the
+  mask.
+- The return address pushed by a call must be bundle-aligned, otherwise a
+  masked `ret` would not return to the instruction following the call.
+
+Both properties are enforced by instruction bundling in the assembler.
+
+**Note**: instruction bundling has not been implemented yet, so the rewrites
+below are currently emitted without it. Until bundling is added, the emitted
+code is not yet a complete sandbox.
+
+To make sure that valid indirect branch targets remain reachable after masking,
+the compiler aligns function entry points, address-taken basic blocks, jump
+table targets, and exception handling landing pads to a bundle boundary.
+
+The targets of direct branches do not need to be aligned, since they are
+resolved at build time. The hidden `-x86-lfi-align-direct-branches` option
+aligns every basic block anyway, so that every branch target in the program is
+bundle-aligned, which can simplify verification.
 
 #### Control flow
 
-**Note**: these rewrites have not been implemented.
+Indirect jumps are rewritten to first apply a mask that zeroes the top 32 bits
+and bottom 5 bits of the target. An `addq` instruction is then used to fill
+in the top 32 bits with the sandbox base, producing an address that is both
+inside the sandbox and bundle-aligned.
+
+Indirect branches through memory first load the branch target into the scratch
+register (`%r11`), and then dispatch through it.
+
+Returns are rewritten to pop the return address into the scratch register,
+followed by a sandboxed indirect jump.
+
+Direct jumps and direct calls do not need to be rewritten, since their targets
+are resolved at link time. Direct calls are placed at the end of a bundle.
+
+:::{list-table}
+:header-rows: 1
+
+* - Original
+  - Rewritten
+* - ```gas
+    jmpq *%rX
+    ```
+  - ```gas
+    andl $-32, %eX
+    addq %r14, %rX
+    jmpq *%rX
+    ```
+* - ```gas
+    jmpq *N(...)
+    ```
+  - ```gas
+    movq N(...), %r11
+    andl $-32, %r11d
+    addq %r14, %r11
+    jmpq *%r11
+    ```
+* - ```gas
+    callq *%rX
+    ```
+  - ```gas
+    andl $-32, %eX
+    addq %r14, %rX
+    callq *%rX
+    ```
+* - ```gas
+    callq *N(...)
+    ```
+  - ```gas
+    movq N(...), %r11
+    andl $-32, %r11d
+    addq %r14, %r11
+    callq *%r11
+    ```
+* - ```gas
+    ret
+    ```
+  - ```gas
+    popq %r11
+    andl $-32, %r11d
+    addq %r14, %r11
+    jmpq *%r11
+    ```
+:::
 
 #### Memory accesses
 
diff --git a/llvm/lib/Target/X86/CMakeLists.txt b/llvm/lib/Target/X86/CMakeLists.txt
index 62987bdbd1c2b..3c81a825a5e33 100644
--- a/llvm/lib/Target/X86/CMakeLists.txt
+++ b/llvm/lib/Target/X86/CMakeLists.txt
@@ -69,6 +69,7 @@ set(sources
   X86InstrFoldTables.cpp
   X86InstrInfo.cpp
   X86CompressEVEX.cpp
+  X86LFIRewritePass.cpp
   X86LoadValueInjectionLoadHardening.cpp
   X86LoadValueInjectionRetHardening.cpp
   X86MCInstLower.cpp
diff --git a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
index 00a767e86c009..94a55cdff9b7e 100644
--- a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
+++ b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
@@ -35,6 +35,32 @@ static bool isSyscall(const MCInst &Inst) {
   return Inst.getOpcode() == X86::SYSCALL;
 }
 
+static bool isDirectCall(const MCInst &Inst) {
+  switch (Inst.getOpcode()) {
+  case X86::CALLpcrel32:
+  case X86::CALL64pcrel32:
+    return true;
+  default:
+    return false;
+  }
+}
+
+static bool isSupportedIndirectBranch(const MCInst &Inst) {
+  switch (Inst.getOpcode()) {
+  case X86::JMP64r:
+  case X86::JMP64r_NT:
+  case X86::JMP64m:
+  case X86::JMP64m_NT:
+  case X86::CALL64r:
+  case X86::CALL64r_NT:
+  case X86::CALL64m:
+  case X86::CALL64m_NT:
+    return true;
+  default:
+    return false;
+  }
+}
+
 // Find the index of the memory operand if it has an %fs segment override.
 // Returns -1 if there is no memory operand or no %fs override.
 static int findFSMemOperand(const MCInst &Inst, const MCInstrInfo &InstInfo) {
@@ -92,7 +118,7 @@ void X86::X86MCLFIRewriter::rewriteSyscall(const MCInst &Inst, MCStreamer &Out,
   Lea.addOperand(MCOperand::createReg(X86::NoRegister));
   Out.emitInstruction(Lea, STI);
 
-  // jmpq *(%r14)
+  // jmpq *-8(%r14)
   MCInst Jmp;
   Jmp.setOpcode(X86::JMP64m);
   Jmp.addOperand(MCOperand::createReg(LFIBaseReg));
@@ -106,6 +132,111 @@ void X86::X86MCLFIRewriter::rewriteSyscall(const MCInst &Inst, MCStreamer &Out,
   Out.emitBundleUnlock(STI);
 }
 
+// andl $-LFIBundleSize, %eX
+// addq %r14, %rX
+void X86::X86MCLFIRewriter::emitSandboxBranchReg(MCRegister Reg,
+                                                 MCStreamer &Out,
+                                                 const MCSubtargetInfo &STI) {
+  MCRegister Reg32 = getX86SubSuperRegister(Reg, 32);
+
+  MCInst And;
+  And.setOpcode(X86::AND32ri8);
+  And.addOperand(MCOperand::createReg(Reg32));
+  And.addOperand(MCOperand::createReg(Reg32));
+  And.addOperand(MCOperand::createImm(-static_cast<int64_t>(LFIBundleSize)));
+  Out.emitInstruction(And, STI);
+
+  MCInst Add;
+  Add.setOpcode(X86::ADD64rr);
+  Add.addOperand(MCOperand::createReg(Reg));
+  Add.addOperand(MCOperand::createReg(Reg));
+  Add.addOperand(MCOperand::createReg(LFIBaseReg));
+  Out.emitInstruction(Add, STI);
+}
+
+// Rewrite an indirect jump or call so that it can only target a bundle
+// boundary inside the sandbox.
+//
+// jmpq *%rX
+// ->
+// andl $-32, %eX
+// addq %r14, %rX
+// jmpq *%rX
+//
+// A branch through memory loads its target into the scratch register first,
+// and then dispatches through it.
+//
+// jmpq *(%rdi)
+// ->
+// movq (%rdi), %r11
+// andl $-32, %r11d
+// addq %r14, %r11
+// jmpq *%r11
+void X86::X86MCLFIRewriter::rewriteIndirectBranch(const MCInst &Inst,
+                                                  MCStreamer &Out,
+                                                  const MCSubtargetInfo &STI) {
+  MCRegister Target;
+  int MemIdx = X86II::getMemoryOperandIdx(InstInfo->get(Inst.getOpcode()));
+  if (MemIdx >= 0) {
+    Target = LFIScratchReg;
+
+    // Construct the load and then apply the rewriter to it.
+    MCInst Mov;
+    Mov.setOpcode(X86::MOV64rm);
+    Mov.addOperand(MCOperand::createReg(Target));
+    for (unsigned I = 0; I < X86::AddrNumOperands; ++I)
+      Mov.addOperand(Inst.getOperand(MemIdx + I));
+    doRewriteInst(Mov, Out, STI);
+  } else {
+    Target = Inst.getOperand(0).getReg();
+
+    if (Target == LFIBaseReg || Target == LFITPReg || Target == X86::RSP)
+      return error(Inst, "indirect branch through reserved register");
+  }
+
+  emitSandboxBranchReg(Target, Out, STI);
+
+  MCInst Branch;
+  Branch.setOpcode(isCall(Inst) ? X86::CALL64r : X86::JMP64r);
+  Branch.addOperand(MCOperand::createReg(Target));
+  Out.emitInstruction(Branch, STI);
+}
+
+// ret
+// ->
+// popq %r11
+// andl $-32, %r11d
+// addq %r14, %r11
+// jmpq *%r11
+void X86::X86MCLFIRewriter::rewriteReturn(const MCInst &Inst, MCStreamer &Out,
+                                          const MCSubtargetInfo &STI) {
+  if (Inst.getOpcode() != X86::RET64 && Inst.getOpcode() != X86::RETI64)
+    return error(Inst, "unsupported return instruction");
+
+  MCInst Pop;
+  Pop.setOpcode(X86::POP64r);
+  Pop.addOperand(MCOperand::createReg(LFIScratchReg));
+  Out.emitInstruction(Pop, STI);
+
+  if (Inst.getOpcode() == X86::RETI64) {
+    // Return with an immediate is rewritten recursively so that the stack
+    // pointer modification goes through the rewriter.
+    MCInst Add;
+    Add.setOpcode(X86::ADD64ri32);
+    Add.addOperand(MCOperand::createReg(X86::RSP));
+    Add.addOperand(MCOperand::createReg(X86::RSP));
+    Add.addOperand(Inst.getOperand(0));
+    doRewriteInst(Add, Out, STI);
+  }
+
+  emitSandboxBranchReg(LFIScratchReg, Out, STI);
+
+  MCInst Jmp;
+  Jmp.setOpcode(X86::JMP64r);
+  Jmp.addOperand(MCOperand::createReg(LFIScratchReg));
+  Out.emitInstruction(Jmp, STI);
+}
+
 // Emit: movq TPOffset(%r15), %Reg
 static void emitTPLoad(MCRegister Reg, MCStreamer &Out,
                        const MCSubtargetInfo &STI) {
@@ -229,10 +360,18 @@ void X86::X86MCLFIRewriter::doRewriteInst(const MCInst &Inst, MCStreamer &Out,
   if (isSyscall(Inst))
     return rewriteSyscall(Inst, Out, STI);
 
+  if (isReturn(Inst))
+    return rewriteReturn(Inst, Out, STI);
+
+  if ((isIndirectBranch(Inst) || isCall(Inst)) && !isDirectCall(Inst)) {
+    if (!isSupportedIndirectBranch(Inst))
+      return error(Inst, "unsupported indirect branch");
+    return rewriteIndirectBranch(Inst, Out, STI);
+  }
+
   if (isFSAccess(Inst))
     return rewriteFSAccess(Inst, Out, STI);
 
-  // Pass through all other instructions unchanged.
   Out.emitInstruction(Inst, STI);
 }
 
diff --git a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.h b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.h
index d74f875311d94..c29547b9b3f3f 100644
--- a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.h
+++ b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.h
@@ -25,6 +25,8 @@ class MCSubtargetInfo;
 
 namespace X86 {
 
+constexpr unsigned LFIBundleSize = 32;
+
 class X86MCLFIRewriter : public MCLFIRewriter {
 public:
   X86MCLFIRewriter(MCContext &Ctx, std::unique_ptr<MCRegisterInfo> &&RI,
@@ -44,6 +46,16 @@ class X86MCLFIRewriter : public MCLFIRewriter {
   void rewriteSyscall(const MCInst &Inst, MCStreamer &Out,
                       const MCSubtargetInfo &STI);
 
+  /// Emit the mask sequence that turns an arbitrary value in Reg into a
+  /// bundle-aligned address inside the sandbox.
+  void emitSandboxBranchReg(MCRegister Reg, MCStreamer &Out,
+                            const MCSubtargetInfo &STI);
+
+  void rewriteIndirectBranch(const MCInst &Inst, MCStreamer &Out,
+                             const MCSubtargetInfo &STI);
+  void rewriteReturn(const MCInst &Inst, MCStreamer &Out,
+                     const MCSubtargetInfo &STI);
+
   bool isFSAccess(const MCInst &Inst);
   void rewriteFSAccess(const MCInst &Inst, MCStreamer &Out,
                        const MCSubtargetInfo &STI);
diff --git a/llvm/lib/Target/X86/X86.h b/llvm/lib/Target/X86/X86.h
index eef4de389a7de..57187fe93fcb6 100644
--- a/llvm/lib/Target/X86/X86.h
+++ b/llvm/lib/Target/X86/X86.h
@@ -75,6 +75,15 @@ class X86FPStackifierPass : public OptionalPassInfoMixin<X86FPStackifierPass> {
 
 FunctionPass *createX86FPStackifierLegacyPass();
 
+/// This pass aligns the code so that it conforms to the LFI sandboxing rules.
+class X86LFIRewritePass : public RequiredPassInfoMixin<X86LFIRewritePass> {
+public:
+  PreservedAnalyses run(MachineFunction &MF,
+                        MachineFunctionAnalysisManager &MFAM);
+};
+
+FunctionPass *createX86LFIRewritePass();
+
 /// This pass inserts AVX vzeroupper instructions before each call to avoid
 /// transition penalty between functions encoded with AVX and SSE.
 class X86InsertVZeroUpperPass
diff --git a/llvm/lib/Target/X86/X86CodeGenPassBuilder.cpp b/llvm/lib/Target/X86/X86CodeGenPassBuilder.cpp
index fc2eb6134d1fd..8bbd26710150a 100644
--- a/llvm/lib/Target/X86/X86CodeGenPassBuilder.cpp
+++ b/llvm/lib/Target/X86/X86CodeGenPassBuilder.cpp
@@ -198,6 +198,9 @@ void X86CodeGenPassBuilder::addPreEmitPass(PassManagerWrapper &PMW) const {
   }
   addMachineFunctionPass(X86CompressEVEXPass(), PMW);
   addMachineFunctionPass(X86InsertX87WaitPass(), PMW);
+
+  if (TM.getTargetTriple().isLFI())
+    addMachineFunctionPass(X86LFIRewritePass(), PMW);
 }
 
 void X86CodeGenPassBuilder::addPreEmitPass2(PassManagerWrapper &PMW) const {
diff --git a/llvm/lib/Target/X86/X86LFIRewritePass.cpp b/llvm/lib/Target/X86/X86LFIRewritePass.cpp
new file mode 100644
index 0000000000000..9bee84d918714
--- /dev/null
+++ b/llvm/lib/Target/X86/X86LFIRewritePass.cpp
@@ -0,0 +1,101 @@
+//===- X86LFIRewritePass.cpp - Modify code generation for LFI ---*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// This file implements the X86LFIRewritePass, which prepares machine code for
+// LFI sandboxing by making sure that every address which may legitimately be
+// the destination of an indirect branch is aligned to a bundle boundary.
+//
+//===----------------------------------------------------------------------===//
+
+#include "MCTargetDesc/X86MCLFIRewriter.h"
+#include "X86.h"
+#include "llvm/CodeGen/MachineFunctionPass.h"
+#include "llvm/CodeGen/MachineJumpTableInfo.h"
+#include "llvm/Support/Alignment.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/Target/TargetMachine.h"
+
+using namespace llvm;
+
+static cl::opt<bool> AlignDirectBranches(
+    "x86-lfi-align-direct-branches",
+    cl::desc("Align the targets of direct branches to a bundle boundary"),
+    cl::init(false), cl::Hidden);
+
+static constexpr Align BundleAlign = Align::Constant<X86::LFIBundleSize>();
+
+namespace {
+class X86LFIRewriteLegacy : public MachineFunctionPass {
+public:
+  static char ID;
+  X86LFIRewriteLegacy() : MachineFunctionPass(ID) {}
+
+  bool runOnMachineFunction(MachineFunction &MF) override;
+
+  StringRef getPassName() const override { return "X86 LFI rewrites"; }
+};
+} // namespace
+
+char X86LFIRewriteLegacy::ID = 0;
+
+static void alignToBundle(MachineBasicBlock &MBB) {
+  MBB.setAlignment(std::max(MBB.getAlignment(), BundleAlign), /*MaxBytes=*/0);
+}
+
+// Returns true if MBB may be reached by an indirect branch.
+static bool isIndirectlyReachable(
+    MachineFunction &MF, const MachineBasicBlock &MBB,
+    const SmallPtrSetImpl<MachineBasicBlock *> &JumpTableTargets) {
+  if (MBB.hasAddressTaken() || JumpTableTargets.contains(&MBB))
+    return true;
+
+  if (MBB.isEHPad())
+    return true;
+
+  // With SJLJ exception handling, the dispatch block jumps indirectly to the
+  // block holding the call site's landing pad label, which is no longer marked
+  // as an EH pad by that point.
+  if (MF.getTarget().Options.ExceptionModel == ExceptionHandling::SjLj)
+    for (const MachineInstr &MI : MBB)
+      if (MI.isEHLabel() &&
+          MF.hasCallSiteLandingPad(MI.getOperand(0).getMCSymbol()))
+        return true;
+
+  return false;
+}
+
+static void alignIndirectBranchTargets(MachineFunction &MF) {
+  // Function entry points are reachable through function pointers.
+  MF.ensureAlignment(BundleAlign);
+
+  // Blocks that are the target of a jump table are not considered
+  // address-taken by LLVM, but they are still reached by an indirect branch.
+  SmallPtrSet<MachineBasicBlock *, 8> JumpTableTargets;
+  if (const MachineJumpTableInfo *JTI = MF.getJumpTableInfo())
+    for (const MachineJumpTableEntry &JTE : JTI->getJumpTables())
+      JumpTableTargets.insert_range(JTE.MBBs);
+
+  for (MachineBasicBlock &MBB : MF)
+    if (AlignDirectBranches || isIndirectlyReachable(MF, MBB, JumpTableTargets))
+      alignToBundle(MBB);
+}
+
+bool X86LFIRewriteLegacy::runOnMachineFunction(MachineFunction &MF) {
+  alignIndirectBranchTargets(MF);
+  return true;
+}
+
+PreservedAnalyses X86LFIRewritePass::run(MachineFunction &MF,
+                                         MachineFunctionAnalysisManager &) {
+  alignIndirectBranchTargets(MF);
+  return PreservedAnalyses::all();
+}
+
+FunctionPass *llvm::createX86LFIRewritePass() {
+  return new X86LFIRewriteLegacy();
+}
diff --git a/llvm/lib/Target/X86/X86PassRegistry.def b/llvm/lib/Target/X86/X86PassRegistry.def
index 8744dbfaed33a..15d71779e4dc7 100644
--- a/llvm/lib/Target/X86/X86PassRegistry.def
+++ b/llvm/lib/Target/X86/X86PassRegistry.def
@@ -56,6 +56,7 @@ MACHINE_FUNCTION_PASS("x86-indirect-branch-tracking", X86IndirectBranchTrackingP
 MACHINE_FUNCTION_PASS("x86-insert-vzeroupper", X86InsertVZeroUpperPass())
 MACHINE_FUNCTION_PASS("x86-insert-x87-wait", X86InsertX87WaitPass())
 MACHINE_FUNCTION_PASS("x86-isel", X86ISelDAGToDAGPass(*this))
+MACHINE_FUNCTION_PASS("x86-lfi-rewrite", X86LFIRewritePass())
 MACHINE_FUNCTION_PASS("x86-lower-tile-copy", X86LowerTileCopyPass())
 MACHINE_FUNCTION_PASS("x86-lvi-load", X86LoadValueInjectionLoadHardeningPass())
 MACHINE_FUNCTION_PASS("x86-lvi-ret", X86LoadValueInjectionRetHardeningPass())
diff --git a/llvm/lib/Target/X86/X86TargetMachine.cpp b/llvm/lib/Target/X86/X86TargetMachine.cpp
index 5a578ae2a33f5..0086d1291b6ff 100644
--- a/llvm/lib/Target/X86/X86TargetMachine.cpp
+++ b/llvm/lib/Target/X86/X86TargetMachine.cpp
@@ -573,6 +573,9 @@ void X86PassConfig::addPreEmitPass() {
   }
   addPass(createX86CompressEVEXLegacyPass());
   addPass(createX86InsertX87WaitLegacyPass());
+
+  if (TM->getTargetTriple().isLFI())
+    addPass(createX86LFIRewritePass());
 }
 
 void X86PassConfig::addPreEmitPass2() {
diff --git a/llvm/test/CodeGen/X86/lfi-align.ll b/llvm/test/CodeGen/X86/lfi-align.ll
new file mode 100644
index 0000000000000..d36061b480118
--- /dev/null
+++ b/llvm/test/CodeGen/X86/lfi-align.ll
@@ -0,0 +1,65 @@
+; RUN: llc < %s -mtriple=x86_64_lfi | FileCheck %s
+
+; LFI masks indirect branch targets down to a 32-byte bundle boundary, so every
+; address that may be reached indirectly has to be aligned to one.
+
+declare void @f(i32)
+
+; Function entry points are reachable through function pointers.
+define void @entry_aligned() {
+; CHECK:      .p2align 5
+; CHECK-NEXT: .type entry_aligned, at function
+; CHECK-NEXT: entry_aligned:
+  ret void
+}
+
+; Jump table targets are not marked address-taken by LLVM, but are still
+; reached by the indirect branch that dispatches through the table.
+define void @jump_table(i32 %x) {
+; CHECK-LABEL: jump_table:
+; CHECK:      jmpq *.LJTI
+; CHECK:      .p2align 5
+; CHECK-NEXT: .LBB{{[0-9_]+}}:
+; CHECK:      .p2align 5
+; CHECK-NEXT: .LBB{{[0-9_]+}}:
+; CHECK:      .p2align 5
+; CHECK-NEXT: .LBB{{[0-9_]+}}:
+; CHECK:      .p2align 5
+; CHECK-NEXT: .LBB{{[0-9_]+}}:
+; CHECK:      .p2align 5
+; CHECK-NEXT: .LBB{{[0-9_]+}}:
+entry:
+  switch i32 %x, label %exit [ i32 0, label %a
+                               i32 1, label %b
+                               i32 2, label %c
+                               i32 3, label %d
+                               i32 4, label %e ]
+a:
+  call void @f(i32 0)
+  br label %exit
+b:
+  call void @f(i32 1)
+  br label %exit
+c:
+  call void @f(i32 2)
+  br label %exit
+d:
+  call void @f(i32 3)
+  br label %exit
+e:
+  call void @f(i32 4)
+  br label %exit
+exit:
+  ret void
+}
+
+; Blocks whose address is taken may be the target of an indirect branch.
+define ptr @block_address() {
+; CHECK-LABEL: block_address:
+; CHECK:      .p2align 5
+; CHECK-NEXT: .Ltmp{{[0-9]+}}:
+entry:
+  br label %target
+target:
+  ret ptr blockaddress(@block_address, %target)
+}
diff --git a/llvm/test/MC/X86/LFI/control-flow-errors.s b/llvm/test/MC/X86/LFI/control-flow-errors.s
new file mode 100644
index 0000000000000..226dfda2826b8
--- /dev/null
+++ b/llvm/test/MC/X86/LFI/control-flow-errors.s
@@ -0,0 +1,34 @@
+// RUN: not llvm-mc -triple x86_64_lfi %s 2>&1 | FileCheck %s
+
+// Masking the target in place would clobber a register the sandbox relies on.
+
+jmpq *%r14
+// CHECK: error: indirect branch through reserved register
+
+callq *%r15
+// CHECK: error: indirect branch through reserved register
+
+jmpq *%rsp
+// CHECK: error: indirect branch through reserved register
+
+// Far branches cannot be sandboxed.
+
+ljmpq *(%rax)
+// CHECK: error: unsupported indirect branch
+
+lcallq *(%rax)
+// CHECK: error: unsupported indirect branch
+
+// Only 64-bit near returns are supported.
+
+retw
+// CHECK: error: unsupported return instruction
+
+retw $8
+// CHECK: error: unsupported return instruction
+
+lret
+// CHECK: error: unsupported return instruction
+
+iretq
+// CHECK: error: unsupported return instruction
diff --git a/llvm/test/MC/X86/LFI/control-flow.s b/llvm/test/MC/X86/LFI/control-flow.s
new file mode 100644
index 0000000000000..3fce7fb8990ec
--- /dev/null
+++ b/llvm/test/MC/X86/LFI/control-flow.s
@@ -0,0 +1,78 @@
+// RUN: llvm-mc -triple x86_64_lfi %s | FileCheck %s
+
+jmpq *%rax
+// CHECK:      andl $-32, %eax
+// CHECK-NEXT: addq %r14, %rax
+// CHECK-NEXT: jmpq *%rax
+
+// The scratch register may be used as a branch target.
+jmpq *%r11
+// CHECK:      andl $-32, %r11d
+// CHECK-NEXT: addq %r14, %r11
+// CHECK-NEXT: jmpq *%r11
+
+jmpq *(%rdi)
+// CHECK:      movq (%rdi), %r11
+// CHECK-NEXT: andl $-32, %r11d
+// CHECK-NEXT: addq %r14, %r11
+// CHECK-NEXT: jmpq *%r11
+
+jmpq *8(%rdi,%rsi,4)
+// CHECK:      movq 8(%rdi,%rsi,4), %r11
+// CHECK-NEXT: andl $-32, %r11d
+// CHECK-NEXT: addq %r14, %r11
+// CHECK-NEXT: jmpq *%r11
+
+jmpq *foo(%rip)
+// CHECK:      movq foo(%rip), %r11
+// CHECK-NEXT: andl $-32, %r11d
+// CHECK-NEXT: addq %r14, %r11
+// CHECK-NEXT: jmpq *%r11
+
+// The target load is itself rewritten, so an %fs-relative branch target is
+// resolved against the virtual thread pointer.
+jmpq *%fs:(%rdi)
+// CHECK:      movq 16(%r15), %r11
+// CHECK-NEXT: movq (%r11,%rdi), %r11
+// CHECK-NEXT: andl $-32, %r11d
+// CHECK-NEXT: addq %r14, %r11
+// CHECK-NEXT: jmpq *%r11
+
+// The notrack prefix is dropped; LFI masks the target instead.
+notrack jmpq *%rax
+// CHECK:      andl $-32, %eax
+// CHECK-NEXT: addq %r14, %rax
+// CHECK-NEXT: jmpq *%rax
+
+callq *%rcx
+// CHECK:      andl $-32, %ecx
+// CHECK-NEXT: addq %r14, %rcx
+// CHECK-NEXT: callq *%rcx
+
+callq *(%rdx)
+// CHECK:      movq (%rdx), %r11
+// CHECK-NEXT: andl $-32, %r11d
+// CHECK-NEXT: addq %r14, %r11
+// CHECK-NEXT: callq *%r11
+
+ret
+// CHECK:      popq %r11
+// CHECK-NEXT: andl $-32, %r11d
+// CHECK-NEXT: addq %r14, %r11
+// CHECK-NEXT: jmpq *%r11
+
+retq $16
+// CHECK:      popq %r11
+// CHECK-NEXT: addq $16, %rsp
+// CHECK-NEXT: andl $-32, %r11d
+// CHECK-NEXT: addq %r14, %r11
+// CHECK-NEXT: jmpq *%r11
+
+callq foo
+// CHECK: callq foo
+
+jmp foo
+// CHECK: jmp foo
+
+je foo
+// CHECK: je foo

>From 4fbcb9cf1383b0efa14e3b545298777b64ded636 Mon Sep 17 00:00:00 2001
From: Zachary Yedidia <zyedidia at gmail.com>
Date: Wed, 22 Jul 2026 20:35:39 -0700
Subject: [PATCH 02/10] Preserve notrack prefix

---
 llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp | 7 +++++++
 llvm/test/MC/X86/LFI/control-flow.s                   | 9 +++++++--
 2 files changed, 14 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
index 94a55cdff9b7e..b8742859d3c7d 100644
--- a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
+++ b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
@@ -61,6 +61,11 @@ static bool isSupportedIndirectBranch(const MCInst &Inst) {
   }
 }
 
+static bool hasNoTrackPrefix(const MCInst &Inst, const MCInstrInfo &InstInfo) {
+  return (InstInfo.get(Inst.getOpcode()).TSFlags & X86II::NOTRACK) ||
+         (Inst.getFlags() & X86::IP_HAS_NOTRACK);
+}
+
 // Find the index of the memory operand if it has an %fs segment override.
 // Returns -1 if there is no memory operand or no %fs override.
 static int findFSMemOperand(const MCInst &Inst, const MCInstrInfo &InstInfo) {
@@ -199,6 +204,8 @@ void X86::X86MCLFIRewriter::rewriteIndirectBranch(const MCInst &Inst,
   MCInst Branch;
   Branch.setOpcode(isCall(Inst) ? X86::CALL64r : X86::JMP64r);
   Branch.addOperand(MCOperand::createReg(Target));
+  if (hasNoTrackPrefix(Inst, *InstInfo))
+    Branch.setFlags(Branch.getFlags() | X86::IP_HAS_NOTRACK);
   Out.emitInstruction(Branch, STI);
 }
 
diff --git a/llvm/test/MC/X86/LFI/control-flow.s b/llvm/test/MC/X86/LFI/control-flow.s
index 3fce7fb8990ec..97ac304c1f870 100644
--- a/llvm/test/MC/X86/LFI/control-flow.s
+++ b/llvm/test/MC/X86/LFI/control-flow.s
@@ -38,11 +38,16 @@ jmpq *%fs:(%rdi)
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: jmpq *%r11
 
-// The notrack prefix is dropped; LFI masks the target instead.
 notrack jmpq *%rax
 // CHECK:      andl $-32, %eax
 // CHECK-NEXT: addq %r14, %rax
-// CHECK-NEXT: jmpq *%rax
+// CHECK-NEXT: notrack jmpq *%rax
+
+notrack callq *(%rdx)
+// CHECK:      movq (%rdx), %r11
+// CHECK-NEXT: andl $-32, %r11d
+// CHECK-NEXT: addq %r14, %r11
+// CHECK-NEXT: notrack callq *%r11
 
 callq *%rcx
 // CHECK:      andl $-32, %ecx

>From 6758641c8392a239a54af2ecbcc75bdb862d6e72 Mon Sep 17 00:00:00 2001
From: Zachary Yedidia <zyedidia at gmail.com>
Date: Wed, 22 Jul 2026 20:37:54 -0700
Subject: [PATCH 03/10] Add test case for rep ret

---
 llvm/test/MC/X86/LFI/control-flow.s | 6 ++++++
 1 file changed, 6 insertions(+)

diff --git a/llvm/test/MC/X86/LFI/control-flow.s b/llvm/test/MC/X86/LFI/control-flow.s
index 97ac304c1f870..295f0166b13fa 100644
--- a/llvm/test/MC/X86/LFI/control-flow.s
+++ b/llvm/test/MC/X86/LFI/control-flow.s
@@ -66,6 +66,12 @@ ret
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: jmpq *%r11
 
+rep ret
+// CHECK:      popq %r11
+// CHECK-NEXT: andl $-32, %r11d
+// CHECK-NEXT: addq %r14, %r11
+// CHECK-NEXT: jmpq *%r11
+
 retq $16
 // CHECK:      popq %r11
 // CHECK-NEXT: addq $16, %rsp

>From f89e7bdde9693d4c3a46cff5323f1c6c714a783d Mon Sep 17 00:00:00 2001
From: Zachary Yedidia <zyedidia at gmail.com>
Date: Wed, 22 Jul 2026 20:52:20 -0700
Subject: [PATCH 04/10] Use getSubReg instead of getX86SubSuperRegister

---
 llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
index b8742859d3c7d..98b9f96931e90 100644
--- a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
+++ b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
@@ -142,7 +142,7 @@ void X86::X86MCLFIRewriter::rewriteSyscall(const MCInst &Inst, MCStreamer &Out,
 void X86::X86MCLFIRewriter::emitSandboxBranchReg(MCRegister Reg,
                                                  MCStreamer &Out,
                                                  const MCSubtargetInfo &STI) {
-  MCRegister Reg32 = getX86SubSuperRegister(Reg, 32);
+  MCRegister Reg32 = RegInfo->getSubReg(Reg, X86::sub_32bit);
 
   MCInst And;
   And.setOpcode(X86::AND32ri8);

>From 0b21c8378caee4717744117ac3130f8e3cc27a8b Mon Sep 17 00:00:00 2001
From: Zachary Yedidia <zyedidia at gmail.com>
Date: Thu, 23 Jul 2026 14:40:30 -0700
Subject: [PATCH 05/10] Simplify branch alignment and remove direct branch
 alignment option

---
 llvm/docs/LFI.md                          |  4 +---
 llvm/lib/Target/X86/X86LFIRewritePass.cpp | 25 ++++++++---------------
 2 files changed, 9 insertions(+), 20 deletions(-)

diff --git a/llvm/docs/LFI.md b/llvm/docs/LFI.md
index 142381dd12bbb..de06e99d21775 100644
--- a/llvm/docs/LFI.md
+++ b/llvm/docs/LFI.md
@@ -665,9 +665,7 @@ the compiler aligns function entry points, address-taken basic blocks, jump
 table targets, and exception handling landing pads to a bundle boundary.
 
 The targets of direct branches do not need to be aligned, since they are
-resolved at build time. The hidden `-x86-lfi-align-direct-branches` option
-aligns every basic block anyway, so that every branch target in the program is
-bundle-aligned, which can simplify verification.
+resolved at build time.
 
 #### Control flow
 
diff --git a/llvm/lib/Target/X86/X86LFIRewritePass.cpp b/llvm/lib/Target/X86/X86LFIRewritePass.cpp
index 9bee84d918714..261efa209b253 100644
--- a/llvm/lib/Target/X86/X86LFIRewritePass.cpp
+++ b/llvm/lib/Target/X86/X86LFIRewritePass.cpp
@@ -17,16 +17,10 @@
 #include "llvm/CodeGen/MachineFunctionPass.h"
 #include "llvm/CodeGen/MachineJumpTableInfo.h"
 #include "llvm/Support/Alignment.h"
-#include "llvm/Support/CommandLine.h"
 #include "llvm/Target/TargetMachine.h"
 
 using namespace llvm;
 
-static cl::opt<bool> AlignDirectBranches(
-    "x86-lfi-align-direct-branches",
-    cl::desc("Align the targets of direct branches to a bundle boundary"),
-    cl::init(false), cl::Hidden);
-
 static constexpr Align BundleAlign = Align::Constant<X86::LFIBundleSize>();
 
 namespace {
@@ -47,14 +41,11 @@ static void alignToBundle(MachineBasicBlock &MBB) {
   MBB.setAlignment(std::max(MBB.getAlignment(), BundleAlign), /*MaxBytes=*/0);
 }
 
-// Returns true if MBB may be reached by an indirect branch.
-static bool isIndirectlyReachable(
-    MachineFunction &MF, const MachineBasicBlock &MBB,
-    const SmallPtrSetImpl<MachineBasicBlock *> &JumpTableTargets) {
-  if (MBB.hasAddressTaken() || JumpTableTargets.contains(&MBB))
-    return true;
-
-  if (MBB.isEHPad())
+// Returns true if MBB may be reached by an indirect branch (does not include
+// jump table targets).
+static bool isIndirectlyReachable(MachineFunction &MF,
+                                  const MachineBasicBlock &MBB) {
+  if (MBB.hasAddressTaken() || MBB.isEHPad())
     return true;
 
   // With SJLJ exception handling, the dispatch block jumps indirectly to the
@@ -75,13 +66,13 @@ static void alignIndirectBranchTargets(MachineFunction &MF) {
 
   // Blocks that are the target of a jump table are not considered
   // address-taken by LLVM, but they are still reached by an indirect branch.
-  SmallPtrSet<MachineBasicBlock *, 8> JumpTableTargets;
   if (const MachineJumpTableInfo *JTI = MF.getJumpTableInfo())
     for (const MachineJumpTableEntry &JTE : JTI->getJumpTables())
-      JumpTableTargets.insert_range(JTE.MBBs);
+      for (MachineBasicBlock *MBB : JTE.MBBs)
+        alignToBundle(*MBB);
 
   for (MachineBasicBlock &MBB : MF)
-    if (AlignDirectBranches || isIndirectlyReachable(MF, MBB, JumpTableTargets))
+    if (isIndirectlyReachable(MF, MBB))
       alignToBundle(MBB);
 }
 

>From d7e92ba56bedf325b5d1c73a1455ce6a203ba496 Mon Sep 17 00:00:00 2001
From: Zachary Yedidia <zyedidia at gmail.com>
Date: Mon, 17 Aug 2026 15:24:51 -0700
Subject: [PATCH 06/10] Improve test for rep ret and 32-bit instructions

---
 llvm/test/MC/X86/LFI/control-flow-errors.s | 9 +++++++++
 llvm/test/MC/X86/LFI/control-flow.s        | 3 +++
 2 files changed, 12 insertions(+)

diff --git a/llvm/test/MC/X86/LFI/control-flow-errors.s b/llvm/test/MC/X86/LFI/control-flow-errors.s
index 226dfda2826b8..d525909c4ccdc 100644
--- a/llvm/test/MC/X86/LFI/control-flow-errors.s
+++ b/llvm/test/MC/X86/LFI/control-flow-errors.s
@@ -32,3 +32,12 @@ lret
 
 iretq
 // CHECK: error: unsupported return instruction
+
+// LFI only supports x86-64, so instructions in .code32 are rejected.
+.code32
+
+ret
+// CHECK: error: unsupported return instruction
+
+jmp *%eax
+// CHECK: error: unsupported indirect branch
diff --git a/llvm/test/MC/X86/LFI/control-flow.s b/llvm/test/MC/X86/LFI/control-flow.s
index 295f0166b13fa..6a3238742914b 100644
--- a/llvm/test/MC/X86/LFI/control-flow.s
+++ b/llvm/test/MC/X86/LFI/control-flow.s
@@ -66,7 +66,10 @@ ret
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: jmpq *%r11
 
+// The rep prefix has no effect on ret and the return is fully replaced, so the
+// prefix is dropped.
 rep ret
+// CHECK-NOT:  rep
 // CHECK:      popq %r11
 // CHECK-NEXT: andl $-32, %r11d
 // CHECK-NEXT: addq %r14, %r11

>From 0a7bf28291ac3bc4b354e5d487479c05c40d957c Mon Sep 17 00:00:00 2001
From: Zachary Yedidia <zyedidia at gmail.com>
Date: Mon, 17 Aug 2026 15:57:15 -0700
Subject: [PATCH 07/10] Remove test that expects behavior for rewriting
 reserved register r11

---
 llvm/test/MC/X86/LFI/control-flow.s | 6 ------
 1 file changed, 6 deletions(-)

diff --git a/llvm/test/MC/X86/LFI/control-flow.s b/llvm/test/MC/X86/LFI/control-flow.s
index 6a3238742914b..3b63652970b4e 100644
--- a/llvm/test/MC/X86/LFI/control-flow.s
+++ b/llvm/test/MC/X86/LFI/control-flow.s
@@ -5,12 +5,6 @@ jmpq *%rax
 // CHECK-NEXT: addq %r14, %rax
 // CHECK-NEXT: jmpq *%rax
 
-// The scratch register may be used as a branch target.
-jmpq *%r11
-// CHECK:      andl $-32, %r11d
-// CHECK-NEXT: addq %r14, %r11
-// CHECK-NEXT: jmpq *%r11
-
 jmpq *(%rdi)
 // CHECK:      movq (%rdi), %r11
 // CHECK-NEXT: andl $-32, %r11d

>From e0abd61d91f1e6fda95c6a55ed136fed754de55b Mon Sep 17 00:00:00 2001
From: Zachary Yedidia <zyedidia at gmail.com>
Date: Tue, 18 Aug 2026 14:42:11 -0700
Subject: [PATCH 08/10] Use bundle grouping directives where necessary

---
 .../X86/MCTargetDesc/X86MCLFIRewriter.cpp     | 29 +++++++++++++++-
 .../X86/MCTargetDesc/X86MCLFIRewriter.h       |  2 ++
 llvm/test/MC/X86/LFI/control-flow.s           | 34 ++++++++++++++++---
 3 files changed, 60 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
index 98b9f96931e90..46cd4009de9a6 100644
--- a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
+++ b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
@@ -164,9 +164,11 @@ void X86::X86MCLFIRewriter::emitSandboxBranchReg(MCRegister Reg,
 //
 // jmpq *%rX
 // ->
+// .bundle_lock
 // andl $-32, %eX
 // addq %r14, %rX
 // jmpq *%rX
+// .bundle_unlock
 //
 // A branch through memory loads its target into the scratch register first,
 // and then dispatches through it.
@@ -174,9 +176,11 @@ void X86::X86MCLFIRewriter::emitSandboxBranchReg(MCRegister Reg,
 // jmpq *(%rdi)
 // ->
 // movq (%rdi), %r11
+// .bundle_lock
 // andl $-32, %r11d
 // addq %r14, %r11
 // jmpq *%r11
+// .bundle_unlock
 void X86::X86MCLFIRewriter::rewriteIndirectBranch(const MCInst &Inst,
                                                   MCStreamer &Out,
                                                   const MCSubtargetInfo &STI) {
@@ -199,6 +203,8 @@ void X86::X86MCLFIRewriter::rewriteIndirectBranch(const MCInst &Inst,
       return error(Inst, "indirect branch through reserved register");
   }
 
+  Out.emitBundleLock(/*AlignToEnd=*/isCall(Inst), STI);
+
   emitSandboxBranchReg(Target, Out, STI);
 
   MCInst Branch;
@@ -207,14 +213,28 @@ void X86::X86MCLFIRewriter::rewriteIndirectBranch(const MCInst &Inst,
   if (hasNoTrackPrefix(Inst, *InstInfo))
     Branch.setFlags(Branch.getFlags() | X86::IP_HAS_NOTRACK);
   Out.emitInstruction(Branch, STI);
+
+  Out.emitBundleUnlock(STI);
+}
+
+// Direct calls are not rewritten, but must be placed at the end of a bundle
+// so that the return address they push is bundle-aligned.
+void X86::X86MCLFIRewriter::rewriteDirectCall(const MCInst &Inst,
+                                              MCStreamer &Out,
+                                              const MCSubtargetInfo &STI) {
+  Out.emitBundleLock(/*AlignToEnd=*/true, STI);
+  Out.emitInstruction(Inst, STI);
+  Out.emitBundleUnlock(STI);
 }
 
 // ret
 // ->
 // popq %r11
+// .bundle_lock
 // andl $-32, %r11d
 // addq %r14, %r11
 // jmpq *%r11
+// .bundle_unlock
 void X86::X86MCLFIRewriter::rewriteReturn(const MCInst &Inst, MCStreamer &Out,
                                           const MCSubtargetInfo &STI) {
   if (Inst.getOpcode() != X86::RET64 && Inst.getOpcode() != X86::RETI64)
@@ -236,12 +256,16 @@ void X86::X86MCLFIRewriter::rewriteReturn(const MCInst &Inst, MCStreamer &Out,
     doRewriteInst(Add, Out, STI);
   }
 
+  Out.emitBundleLock(/*AlignToEnd=*/false, STI);
+
   emitSandboxBranchReg(LFIScratchReg, Out, STI);
 
   MCInst Jmp;
   Jmp.setOpcode(X86::JMP64r);
   Jmp.addOperand(MCOperand::createReg(LFIScratchReg));
   Out.emitInstruction(Jmp, STI);
+
+  Out.emitBundleUnlock(STI);
 }
 
 // Emit: movq TPOffset(%r15), %Reg
@@ -370,7 +394,10 @@ void X86::X86MCLFIRewriter::doRewriteInst(const MCInst &Inst, MCStreamer &Out,
   if (isReturn(Inst))
     return rewriteReturn(Inst, Out, STI);
 
-  if ((isIndirectBranch(Inst) || isCall(Inst)) && !isDirectCall(Inst)) {
+  if (isDirectCall(Inst))
+    return rewriteDirectCall(Inst, Out, STI);
+
+  if (isIndirectBranch(Inst) || isCall(Inst)) {
     if (!isSupportedIndirectBranch(Inst))
       return error(Inst, "unsupported indirect branch");
     return rewriteIndirectBranch(Inst, Out, STI);
diff --git a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.h b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.h
index c29547b9b3f3f..e182d3e6df38e 100644
--- a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.h
+++ b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.h
@@ -53,6 +53,8 @@ class X86MCLFIRewriter : public MCLFIRewriter {
 
   void rewriteIndirectBranch(const MCInst &Inst, MCStreamer &Out,
                              const MCSubtargetInfo &STI);
+  void rewriteDirectCall(const MCInst &Inst, MCStreamer &Out,
+                         const MCSubtargetInfo &STI);
   void rewriteReturn(const MCInst &Inst, MCStreamer &Out,
                      const MCSubtargetInfo &STI);
 
diff --git a/llvm/test/MC/X86/LFI/control-flow.s b/llvm/test/MC/X86/LFI/control-flow.s
index 3b63652970b4e..2ef8419b16b35 100644
--- a/llvm/test/MC/X86/LFI/control-flow.s
+++ b/llvm/test/MC/X86/LFI/control-flow.s
@@ -1,83 +1,109 @@
 // RUN: llvm-mc -triple x86_64_lfi %s | FileCheck %s
 
 jmpq *%rax
-// CHECK:      andl $-32, %eax
+// CHECK:      .bundle_lock
+// CHECK-NEXT: andl $-32, %eax
 // CHECK-NEXT: addq %r14, %rax
 // CHECK-NEXT: jmpq *%rax
+// CHECK-NEXT: .bundle_unlock
 
 jmpq *(%rdi)
 // CHECK:      movq (%rdi), %r11
+// CHECK-NEXT: .bundle_lock
 // CHECK-NEXT: andl $-32, %r11d
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: jmpq *%r11
+// CHECK-NEXT: .bundle_unlock
 
 jmpq *8(%rdi,%rsi,4)
 // CHECK:      movq 8(%rdi,%rsi,4), %r11
+// CHECK-NEXT: .bundle_lock
 // CHECK-NEXT: andl $-32, %r11d
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: jmpq *%r11
+// CHECK-NEXT: .bundle_unlock
 
 jmpq *foo(%rip)
 // CHECK:      movq foo(%rip), %r11
+// CHECK-NEXT: .bundle_lock
 // CHECK-NEXT: andl $-32, %r11d
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: jmpq *%r11
+// CHECK-NEXT: .bundle_unlock
 
 // The target load is itself rewritten, so an %fs-relative branch target is
 // resolved against the virtual thread pointer.
 jmpq *%fs:(%rdi)
 // CHECK:      movq 16(%r15), %r11
 // CHECK-NEXT: movq (%r11,%rdi), %r11
+// CHECK-NEXT: .bundle_lock
 // CHECK-NEXT: andl $-32, %r11d
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: jmpq *%r11
+// CHECK-NEXT: .bundle_unlock
 
 notrack jmpq *%rax
-// CHECK:      andl $-32, %eax
+// CHECK:      .bundle_lock
+// CHECK-NEXT: andl $-32, %eax
 // CHECK-NEXT: addq %r14, %rax
 // CHECK-NEXT: notrack jmpq *%rax
+// CHECK-NEXT: .bundle_unlock
 
 notrack callq *(%rdx)
 // CHECK:      movq (%rdx), %r11
+// CHECK-NEXT: .bundle_lock align_to_end
 // CHECK-NEXT: andl $-32, %r11d
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: notrack callq *%r11
+// CHECK-NEXT: .bundle_unlock
 
 callq *%rcx
-// CHECK:      andl $-32, %ecx
+// CHECK:      .bundle_lock align_to_end
+// CHECK-NEXT: andl $-32, %ecx
 // CHECK-NEXT: addq %r14, %rcx
 // CHECK-NEXT: callq *%rcx
+// CHECK-NEXT: .bundle_unlock
 
 callq *(%rdx)
 // CHECK:      movq (%rdx), %r11
+// CHECK-NEXT: .bundle_lock align_to_end
 // CHECK-NEXT: andl $-32, %r11d
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: callq *%r11
+// CHECK-NEXT: .bundle_unlock
 
 ret
 // CHECK:      popq %r11
+// CHECK-NEXT: .bundle_lock
 // CHECK-NEXT: andl $-32, %r11d
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: jmpq *%r11
+// CHECK-NEXT: .bundle_unlock
 
 // The rep prefix has no effect on ret and the return is fully replaced, so the
 // prefix is dropped.
 rep ret
 // CHECK-NOT:  rep
 // CHECK:      popq %r11
+// CHECK-NEXT: .bundle_lock
 // CHECK-NEXT: andl $-32, %r11d
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: jmpq *%r11
+// CHECK-NEXT: .bundle_unlock
 
 retq $16
 // CHECK:      popq %r11
 // CHECK-NEXT: addq $16, %rsp
+// CHECK-NEXT: .bundle_lock
 // CHECK-NEXT: andl $-32, %r11d
 // CHECK-NEXT: addq %r14, %r11
 // CHECK-NEXT: jmpq *%r11
+// CHECK-NEXT: .bundle_unlock
 
 callq foo
-// CHECK: callq foo
+// CHECK:      .bundle_lock align_to_end
+// CHECK-NEXT: callq foo
+// CHECK-NEXT: .bundle_unlock
 
 jmp foo
 // CHECK: jmp foo

>From b99f50c2a3fa0240f73fde88c2654a89b1b2a6ff Mon Sep 17 00:00:00 2001
From: Zachary Yedidia <zyedidia at gmail.com>
Date: Tue, 18 Aug 2026 16:44:02 -0700
Subject: [PATCH 09/10] Update documentation to clarify bundling is now
 implemented

---
 llvm/docs/LFI.md | 36 ++++++++----------------------------
 1 file changed, 8 insertions(+), 28 deletions(-)

diff --git a/llvm/docs/LFI.md b/llvm/docs/LFI.md
index de06e99d21775..7b5eb453357db 100644
--- a/llvm/docs/LFI.md
+++ b/llvm/docs/LFI.md
@@ -624,8 +624,14 @@ entered in the middle wraps that sequence in `.bundle_lock` / `.bundle_unlock`.
 This keeps the whole group within a single bundle, so no masked indirect branch
 can land between its instructions.
 
-**Note**: the masking of indirect branch targets is part of the control flow
-rewrites, which have not been implemented yet.
+Calls are emitted in a `.bundle_lock align_to_end` group, which places the call
+at the end of its bundle. This makes the return address pushed by the call
+bundle-aligned, so that a masked `ret` returns to the instruction following the
+call.
+
+To make sure that indirect branch targets remain reachable after masking, the
+compiler aligns them to a bundle boundary. The targets of direct branches do
+not need to be aligned, since they are resolved statically.
 
 Bundling is specific to X86-64. AArch64 instructions are fixed-width and
 naturally aligned, and the AArch64 LFI target confines indirect branches by
@@ -641,32 +647,6 @@ In the following assembly rewrites, some shorthand is used.
 - `{a,b,c}`: matches any of `a`, `b`, or `c`.
 - `N(...)`: refers to any memory addressing mode.
 
-#### Bundles
-
-The X86-64 target divides the code region into 32-byte aligned *bundles*.
-Indirect branch targets are masked so that they are always bundle-aligned,
-which restricts the set of reachable instructions to bundle boundaries. For
-this to be sound, two additional properties are required:
-
-- A rewrite sequence must never be split across a bundle boundary, otherwise
-  control could be transferred into the middle of the sequence, skipping the
-  mask.
-- The return address pushed by a call must be bundle-aligned, otherwise a
-  masked `ret` would not return to the instruction following the call.
-
-Both properties are enforced by instruction bundling in the assembler.
-
-**Note**: instruction bundling has not been implemented yet, so the rewrites
-below are currently emitted without it. Until bundling is added, the emitted
-code is not yet a complete sandbox.
-
-To make sure that valid indirect branch targets remain reachable after masking,
-the compiler aligns function entry points, address-taken basic blocks, jump
-table targets, and exception handling landing pads to a bundle boundary.
-
-The targets of direct branches do not need to be aligned, since they are
-resolved at build time.
-
 #### Control flow
 
 Indirect jumps are rewritten to first apply a mask that zeroes the top 32 bits

>From 3bf1146f402ba3dcae364f849f7624f907cd0452 Mon Sep 17 00:00:00 2001
From: Zachary Yedidia <zyedidia at gmail.com>
Date: Tue, 25 Aug 2026 23:35:07 -0700
Subject: [PATCH 10/10] Refactor to use MCInstBuilder pattern

---
 .../X86/MCTargetDesc/X86MCLFIRewriter.cpp     | 122 ++++++++----------
 1 file changed, 53 insertions(+), 69 deletions(-)

diff --git a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
index 46cd4009de9a6..34fdad32ea327 100644
--- a/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
+++ b/llvm/lib/Target/X86/MCTargetDesc/X86MCLFIRewriter.cpp
@@ -17,6 +17,7 @@
 #include "llvm/MC/MCContext.h"
 #include "llvm/MC/MCExpr.h"
 #include "llvm/MC/MCInst.h"
+#include "llvm/MC/MCInstBuilder.h"
 #include "llvm/MC/MCStreamer.h"
 #include "llvm/MC/MCSubtargetInfo.h"
 
@@ -112,26 +113,24 @@ void X86::X86MCLFIRewriter::rewriteSyscall(const MCInst &Inst, MCStreamer &Out,
   MCSymbol *Symbol = Out.getContext().createTempSymbol();
 
   // leaq .Ltmp(%rip), %r11
-  MCInst Lea;
-  Lea.setOpcode(X86::LEA64r);
-  Lea.addOperand(MCOperand::createReg(LFIScratchReg));
-  Lea.addOperand(MCOperand::createReg(X86::RIP));
-  Lea.addOperand(MCOperand::createImm(1));
-  Lea.addOperand(MCOperand::createReg(X86::NoRegister));
-  Lea.addOperand(
-      MCOperand::createExpr(MCSymbolRefExpr::create(Symbol, Out.getContext())));
-  Lea.addOperand(MCOperand::createReg(X86::NoRegister));
-  Out.emitInstruction(Lea, STI);
+  Out.emitInstruction(
+      MCInstBuilder(X86::LEA64r)
+          .addReg(LFIScratchReg)
+          .addReg(X86::RIP)
+          .addImm(1)
+          .addReg(X86::NoRegister)
+          .addExpr(MCSymbolRefExpr::create(Symbol, Out.getContext()))
+          .addReg(X86::NoRegister),
+      STI);
 
   // jmpq *-8(%r14)
-  MCInst Jmp;
-  Jmp.setOpcode(X86::JMP64m);
-  Jmp.addOperand(MCOperand::createReg(LFIBaseReg));
-  Jmp.addOperand(MCOperand::createImm(1));
-  Jmp.addOperand(MCOperand::createReg(X86::NoRegister));
-  Jmp.addOperand(MCOperand::createImm(-8));
-  Jmp.addOperand(MCOperand::createReg(X86::NoRegister));
-  Out.emitInstruction(Jmp, STI);
+  Out.emitInstruction(MCInstBuilder(X86::JMP64m)
+                          .addReg(LFIBaseReg)
+                          .addImm(1)
+                          .addReg(X86::NoRegister)
+                          .addImm(-8)
+                          .addReg(X86::NoRegister),
+                      STI);
 
   Out.emitLabel(Symbol);
   Out.emitBundleUnlock(STI);
@@ -144,19 +143,15 @@ void X86::X86MCLFIRewriter::emitSandboxBranchReg(MCRegister Reg,
                                                  const MCSubtargetInfo &STI) {
   MCRegister Reg32 = RegInfo->getSubReg(Reg, X86::sub_32bit);
 
-  MCInst And;
-  And.setOpcode(X86::AND32ri8);
-  And.addOperand(MCOperand::createReg(Reg32));
-  And.addOperand(MCOperand::createReg(Reg32));
-  And.addOperand(MCOperand::createImm(-static_cast<int64_t>(LFIBundleSize)));
-  Out.emitInstruction(And, STI);
-
-  MCInst Add;
-  Add.setOpcode(X86::ADD64rr);
-  Add.addOperand(MCOperand::createReg(Reg));
-  Add.addOperand(MCOperand::createReg(Reg));
-  Add.addOperand(MCOperand::createReg(LFIBaseReg));
-  Out.emitInstruction(Add, STI);
+  Out.emitInstruction(MCInstBuilder(X86::AND32ri8)
+                          .addReg(Reg32)
+                          .addReg(Reg32)
+                          .addImm(-static_cast<int64_t>(LFIBundleSize)),
+                      STI);
+
+  Out.emitInstruction(
+      MCInstBuilder(X86::ADD64rr).addReg(Reg).addReg(Reg).addReg(LFIBaseReg),
+      STI);
 }
 
 // Rewrite an indirect jump or call so that it can only target a bundle
@@ -190,9 +185,8 @@ void X86::X86MCLFIRewriter::rewriteIndirectBranch(const MCInst &Inst,
     Target = LFIScratchReg;
 
     // Construct the load and then apply the rewriter to it.
-    MCInst Mov;
-    Mov.setOpcode(X86::MOV64rm);
-    Mov.addOperand(MCOperand::createReg(Target));
+    MCInstBuilder Mov(X86::MOV64rm);
+    Mov.addReg(Target);
     for (unsigned I = 0; I < X86::AddrNumOperands; ++I)
       Mov.addOperand(Inst.getOperand(MemIdx + I));
     doRewriteInst(Mov, Out, STI);
@@ -207,9 +201,8 @@ void X86::X86MCLFIRewriter::rewriteIndirectBranch(const MCInst &Inst,
 
   emitSandboxBranchReg(Target, Out, STI);
 
-  MCInst Branch;
-  Branch.setOpcode(isCall(Inst) ? X86::CALL64r : X86::JMP64r);
-  Branch.addOperand(MCOperand::createReg(Target));
+  MCInst Branch =
+      MCInstBuilder(isCall(Inst) ? X86::CALL64r : X86::JMP64r).addReg(Target);
   if (hasNoTrackPrefix(Inst, *InstInfo))
     Branch.setFlags(Branch.getFlags() | X86::IP_HAS_NOTRACK);
   Out.emitInstruction(Branch, STI);
@@ -240,30 +233,23 @@ void X86::X86MCLFIRewriter::rewriteReturn(const MCInst &Inst, MCStreamer &Out,
   if (Inst.getOpcode() != X86::RET64 && Inst.getOpcode() != X86::RETI64)
     return error(Inst, "unsupported return instruction");
 
-  MCInst Pop;
-  Pop.setOpcode(X86::POP64r);
-  Pop.addOperand(MCOperand::createReg(LFIScratchReg));
-  Out.emitInstruction(Pop, STI);
+  Out.emitInstruction(MCInstBuilder(X86::POP64r).addReg(LFIScratchReg), STI);
 
   if (Inst.getOpcode() == X86::RETI64) {
     // Return with an immediate is rewritten recursively so that the stack
     // pointer modification goes through the rewriter.
-    MCInst Add;
-    Add.setOpcode(X86::ADD64ri32);
-    Add.addOperand(MCOperand::createReg(X86::RSP));
-    Add.addOperand(MCOperand::createReg(X86::RSP));
-    Add.addOperand(Inst.getOperand(0));
-    doRewriteInst(Add, Out, STI);
+    doRewriteInst(MCInstBuilder(X86::ADD64ri32)
+                      .addReg(X86::RSP)
+                      .addReg(X86::RSP)
+                      .addOperand(Inst.getOperand(0)),
+                  Out, STI);
   }
 
   Out.emitBundleLock(/*AlignToEnd=*/false, STI);
 
   emitSandboxBranchReg(LFIScratchReg, Out, STI);
 
-  MCInst Jmp;
-  Jmp.setOpcode(X86::JMP64r);
-  Jmp.addOperand(MCOperand::createReg(LFIScratchReg));
-  Out.emitInstruction(Jmp, STI);
+  Out.emitInstruction(MCInstBuilder(X86::JMP64r).addReg(LFIScratchReg), STI);
 
   Out.emitBundleUnlock(STI);
 }
@@ -271,15 +257,14 @@ void X86::X86MCLFIRewriter::rewriteReturn(const MCInst &Inst, MCStreamer &Out,
 // Emit: movq TPOffset(%r15), %Reg
 static void emitTPLoad(MCRegister Reg, MCStreamer &Out,
                        const MCSubtargetInfo &STI) {
-  MCInst Mov;
-  Mov.setOpcode(X86::MOV64rm);
-  Mov.addOperand(MCOperand::createReg(Reg));
-  Mov.addOperand(MCOperand::createReg(LFITPReg));
-  Mov.addOperand(MCOperand::createImm(1));
-  Mov.addOperand(MCOperand::createReg(X86::NoRegister));
-  Mov.addOperand(MCOperand::createImm(TPOffset));
-  Mov.addOperand(MCOperand::createReg(X86::NoRegister));
-  Out.emitInstruction(Mov, STI);
+  Out.emitInstruction(MCInstBuilder(X86::MOV64rm)
+                          .addReg(Reg)
+                          .addReg(LFITPReg)
+                          .addImm(1)
+                          .addReg(X86::NoRegister)
+                          .addImm(TPOffset)
+                          .addReg(X86::NoRegister),
+                      STI);
 }
 
 bool X86::X86MCLFIRewriter::isFSAccess(const MCInst &Inst) {
@@ -357,15 +342,14 @@ void X86::X86MCLFIRewriter::rewriteFSAccess(const MCInst &Inst, MCStreamer &Out,
   // leaq (%rax,%rdi), %rax
   // movq 8(%rax,%rsi,2), %rax
   if (HasBase && HasIndex) {
-    MCInst Lea;
-    Lea.setOpcode(X86::LEA64r);
-    Lea.addOperand(MCOperand::createReg(TPDest));
-    Lea.addOperand(MCOperand::createReg(TPDest));
-    Lea.addOperand(MCOperand::createImm(1));
-    Lea.addOperand(MCOperand::createReg(BaseReg));
-    Lea.addOperand(MCOperand::createImm(0));
-    Lea.addOperand(MCOperand::createReg(X86::NoRegister));
-    Out.emitInstruction(Lea, STI);
+    Out.emitInstruction(MCInstBuilder(X86::LEA64r)
+                            .addReg(TPDest)
+                            .addReg(TPDest)
+                            .addImm(1)
+                            .addReg(BaseReg)
+                            .addImm(0)
+                            .addReg(X86::NoRegister),
+                        STI);
   }
 
   // Emit the access with TPDest as the new base, and the original base



More information about the llvm-commits mailing list