[llvm] [X86] Fold an LEA into a following memory operand's addressing mode (PR #221863)
Andrew Gaul via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 7 22:09:30 PDT 2026
https://github.com/gaul updated https://github.com/llvm/llvm-project/pull/221863
>From 912879b75dd6e78af2fd1de2c1d6a5c36dfcf421 Mon Sep 17 00:00:00 2001
From: Andrew Gaul <andrew at gaul.org>
Date: Fri, 4 Sep 2026 11:18:17 -0700
Subject: [PATCH 1/2] [X86] Add tests for folding an address computation into a
memory operand
An LEA whose sole use is the base register of a following load or store
can fold into that instruction's addressing mode, deleting the LEA.
SelectionDAG does this within a basic block, but when the address is
computed in one block and dereferenced in another the two never share a
DAG; tail duplication later brings them adjacent, yet no pass rejoins
them, so the pair ships as "lea; mov (%lea)".
Add tests capturing the current unfolded codegen. A following change adds
a late peephole that performs the fold and updates these checks.
Co-Authored-By: Claude Opus 4.8 <noreply at anthropic.com>
---
llvm/test/CodeGen/X86/fold-addr-into-memop.ll | 151 ++++++++++++++++++
1 file changed, 151 insertions(+)
create mode 100644 llvm/test/CodeGen/X86/fold-addr-into-memop.ll
diff --git a/llvm/test/CodeGen/X86/fold-addr-into-memop.ll b/llvm/test/CodeGen/X86/fold-addr-into-memop.ll
new file mode 100644
index 0000000000000..c6c00b8ec5169
--- /dev/null
+++ b/llvm/test/CodeGen/X86/fold-addr-into-memop.ll
@@ -0,0 +1,151 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -verify-machineinstrs | FileCheck %s
+
+; An address computed in one block and dereferenced in another is brought
+; adjacent (lea then memory op) only after tail duplication, past the point
+; where SelectionDAG folds an address into a memory operand. These check that
+; the late peephole rejoins the pair: the lea disappears into the load/store's
+; addressing mode.
+
+; base + index*4 load
+define i32 @fold_idx4(ptr %a, ptr %b, i64 %i, i1 %c) {
+; CHECK-LABEL: fold_idx4:
+; CHECK: # %bb.0:
+; CHECK-NEXT: testb $1, %cl
+; CHECK-NEXT: je .LBB0_2
+; CHECK-NEXT: # %bb.1: # %la
+; CHECK-NEXT: leaq (%rdi,%rdx,4), %rax
+; CHECK-NEXT: movl (%rax), %eax
+; CHECK-NEXT: retq
+; CHECK-NEXT: .LBB0_2: # %lb
+; CHECK-NEXT: leaq (%rsi,%rdx,4), %rax
+; CHECK-NEXT: movl (%rax), %eax
+; CHECK-NEXT: retq
+ br i1 %c, label %la, label %lb
+la:
+ %qa = getelementptr i32, ptr %a, i64 %i
+ br label %join
+lb:
+ %qb = getelementptr i32, ptr %b, i64 %i
+ br label %join
+join:
+ %q = phi ptr [ %qa, %la ], [ %qb, %lb ]
+ %v = load i32, ptr %q
+ ret i32 %v
+}
+
+; base + index*8 load; the load also fully rewrites the base register
+define i64 @fold_idx8(ptr %a, ptr %b, i64 %i, i1 %c) {
+; CHECK-LABEL: fold_idx8:
+; CHECK: # %bb.0:
+; CHECK-NEXT: testb $1, %cl
+; CHECK-NEXT: je .LBB1_2
+; CHECK-NEXT: # %bb.1: # %la
+; CHECK-NEXT: leaq (%rdi,%rdx,8), %rax
+; CHECK-NEXT: movq (%rax), %rax
+; CHECK-NEXT: retq
+; CHECK-NEXT: .LBB1_2: # %lb
+; CHECK-NEXT: leaq (%rsi,%rdx,8), %rax
+; CHECK-NEXT: movq (%rax), %rax
+; CHECK-NEXT: retq
+ br i1 %c, label %la, label %lb
+la:
+ %qa = getelementptr i64, ptr %a, i64 %i
+ br label %join
+lb:
+ %qb = getelementptr i64, ptr %b, i64 %i
+ br label %join
+join:
+ %q = phi ptr [ %qa, %la ], [ %qb, %lb ]
+ %v = load i64, ptr %q
+ ret i64 %v
+}
+
+; store consumer
+define void @fold_store_idx4(ptr %a, ptr %b, i64 %i, i32 %x, i1 %c) {
+; CHECK-LABEL: fold_store_idx4:
+; CHECK: # %bb.0:
+; CHECK-NEXT: testb $1, %r8b
+; CHECK-NEXT: je .LBB2_2
+; CHECK-NEXT: # %bb.1: # %la
+; CHECK-NEXT: leaq (%rdi,%rdx,4), %rax
+; CHECK-NEXT: movl %ecx, (%rax)
+; CHECK-NEXT: retq
+; CHECK-NEXT: .LBB2_2: # %lb
+; CHECK-NEXT: leaq (%rsi,%rdx,4), %rax
+; CHECK-NEXT: movl %ecx, (%rax)
+; CHECK-NEXT: retq
+ br i1 %c, label %la, label %lb
+la:
+ %qa = getelementptr i32, ptr %a, i64 %i
+ br label %join
+lb:
+ %qb = getelementptr i32, ptr %b, i64 %i
+ br label %join
+join:
+ %q = phi ptr [ %qa, %la ], [ %qb, %lb ]
+ store i32 %x, ptr %q
+ ret void
+}
+
+; load-op consumer (add reg, mem): the folded address feeds the add's memory
+; operand
+define i32 @fold_addrm(ptr %a, ptr %b, i64 %i, i32 %x, i1 %c) {
+; CHECK-LABEL: fold_addrm:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movl %ecx, %eax
+; CHECK-NEXT: testb $1, %r8b
+; CHECK-NEXT: je .LBB3_2
+; CHECK-NEXT: # %bb.1: # %la
+; CHECK-NEXT: leaq (%rdi,%rdx,4), %rcx
+; CHECK-NEXT: addl (%rcx), %eax
+; CHECK-NEXT: retq
+; CHECK-NEXT: .LBB3_2: # %lb
+; CHECK-NEXT: leaq (%rsi,%rdx,4), %rcx
+; CHECK-NEXT: addl (%rcx), %eax
+; CHECK-NEXT: retq
+ br i1 %c, label %la, label %lb
+la:
+ %qa = getelementptr i32, ptr %a, i64 %i
+ br label %join
+lb:
+ %qb = getelementptr i32, ptr %b, i64 %i
+ br label %join
+join:
+ %q = phi ptr [ %qa, %la ], [ %qb, %lb ]
+ %v = load i32, ptr %q
+ %s = add i32 %v, %x
+ ret i32 %s
+}
+
+; negative: the address has a second use after the load, so the base stays
+; live and the lea must remain.
+define i64 @no_fold_multiuse(ptr %a, ptr %b, i64 %i, i1 %c) {
+; CHECK-LABEL: no_fold_multiuse:
+; CHECK: # %bb.0:
+; CHECK-NEXT: testb $1, %cl
+; CHECK-NEXT: je .LBB4_2
+; CHECK-NEXT: # %bb.1: # %la
+; CHECK-NEXT: leaq (%rdi,%rdx,8), %rcx
+; CHECK-NEXT: jmp .LBB4_3
+; CHECK-NEXT: .LBB4_2: # %lb
+; CHECK-NEXT: leaq (%rsi,%rdx,8), %rcx
+; CHECK-NEXT: .LBB4_3: # %join
+; CHECK-NEXT: movq (%rcx), %rax
+; CHECK-NEXT: addq 8(%rcx), %rax
+; CHECK-NEXT: retq
+ br i1 %c, label %la, label %lb
+la:
+ %qa = getelementptr i64, ptr %a, i64 %i
+ br label %join
+lb:
+ %qb = getelementptr i64, ptr %b, i64 %i
+ br label %join
+join:
+ %q = phi ptr [ %qa, %la ], [ %qb, %lb ]
+ %v = load i64, ptr %q
+ %q2 = getelementptr i64, ptr %q, i64 1
+ %w = load i64, ptr %q2
+ %s = add i64 %v, %w
+ ret i64 %s
+}
>From 98df9d705d0cd77ecab372e457228a0bb51e947f Mon Sep 17 00:00:00 2001
From: Andrew Gaul <andrew at gaul.org>
Date: Fri, 4 Sep 2026 11:25:58 -0700
Subject: [PATCH 2/2] [X86] Fold an LEA into a following memory operand's
addressing mode
Add X86FoldAddrIntoMemOp, a late peephole that folds an LEA whose only
use is the base register of the immediately following load or store into
that instruction's addressing mode, deleting the LEA:
lea rax, [rdi + rdx*4] movl (%rdi,%rdx,4), %eax
mov eax, [rax] -->
SelectionDAG already performs this fold within a basic block. The residue
it leaves is the cross-block shape: an address computed in one block and
dereferenced in another is never part of a single DAG, so isel does not
fold it; tail duplication and block placement later make the LEA and its
consumer adjacent, but no pass rejoins them. Running after those passes
picks up the adjacent pair. It also catches addresses fast-isel emits as a
separate LEA (the stmxcsr/ldmxcsr change in the fallout test).
The fold requires the LEA's result to appear in the consumer only as the
memory base and to be dead afterward, at most one index between the two,
and the displacements to sum within signed 32 bits.
Motivated by an x86lint audit: a release build of the Rust compiler's
librustc_driver has 5,709 "LEA foldable into memory" sites -- all
register-base, dominated by array/slice element addresses and struct
field offsets -- roughly 4x the count in Firefox's libxul.so, the extra
coming from the cross-block control flow of Rust's enum and slice-bounds
codegen.
Prototype: LEA64r producers only. The arithmetic siblings (ADD/SUB/INC/DEC
advancing a pointer, gated on dead EFLAGS) and looking past strictly
adjacent uses are planned extensions.
Co-Authored-By: Claude Opus 4.8 <noreply at anthropic.com>
---
llvm/lib/Target/X86/CMakeLists.txt | 1 +
llvm/lib/Target/X86/X86.h | 5 +
llvm/lib/Target/X86/X86FoldAddrIntoMemOp.cpp | 208 ++++++++++++++++++
llvm/lib/Target/X86/X86TargetMachine.cpp | 2 +
llvm/test/CodeGen/X86/fold-addr-into-memop.ll | 24 +-
llvm/test/CodeGen/X86/opt-pipeline.ll | 1 +
.../CodeGen/X86/sse-intrinsics-fast-isel.ll | 36 +--
7 files changed, 237 insertions(+), 40 deletions(-)
create mode 100644 llvm/lib/Target/X86/X86FoldAddrIntoMemOp.cpp
diff --git a/llvm/lib/Target/X86/CMakeLists.txt b/llvm/lib/Target/X86/CMakeLists.txt
index a053eb8501701..58ae2d919bb7d 100644
--- a/llvm/lib/Target/X86/CMakeLists.txt
+++ b/llvm/lib/Target/X86/CMakeLists.txt
@@ -53,6 +53,7 @@ set(sources
X86AvoidStoreForwardingBlocks.cpp
X86DynAllocaExpander.cpp
X86FixupSetCC.cpp
+ X86FoldAddrIntoMemOp.cpp
X86FlagsCopyLowering.cpp
X86FloatingPoint.cpp
X86FrameLowering.cpp
diff --git a/llvm/lib/Target/X86/X86.h b/llvm/lib/Target/X86/X86.h
index eef4de389a7de..48ea02e2736cd 100644
--- a/llvm/lib/Target/X86/X86.h
+++ b/llvm/lib/Target/X86/X86.h
@@ -142,6 +142,10 @@ class X86OptimizeLEAsPass : public OptionalPassInfoMixin<X86OptimizeLEAsPass> {
FunctionPass *createX86OptimizeLEAsLegacyPass();
+/// Return a pass that folds an LEA whose only use is the base of the following
+/// memory instruction into that instruction's addressing mode.
+FunctionPass *createX86FoldAddrIntoMemOpPass();
+
/// Return a pass that transforms setcc + movzx pairs into xor + setcc.
class X86FixupSetCCPass : public OptionalPassInfoMixin<X86FixupSetCCPass> {
public:
@@ -511,6 +515,7 @@ void initializeX86LoadValueInjectionRetHardeningLegacyPass(PassRegistry &);
void initializeX86LowerAMXIntrinsicsLegacyPassPass(PassRegistry &);
void initializeX86LowerAMXTypeLegacyPassPass(PassRegistry &);
void initializeX86LowerTileCopyLegacyPass(PassRegistry &);
+void initializeX86FoldAddrIntoMemOpPass(PassRegistry &);
void initializeX86OptimizeLEAsLegacyPass(PassRegistry &);
void initializeX86PartialReductionLegacyPass(PassRegistry &);
void initializeX86PreTileConfigLegacyPass(PassRegistry &);
diff --git a/llvm/lib/Target/X86/X86FoldAddrIntoMemOp.cpp b/llvm/lib/Target/X86/X86FoldAddrIntoMemOp.cpp
new file mode 100644
index 0000000000000..ea1aa51014483
--- /dev/null
+++ b/llvm/lib/Target/X86/X86FoldAddrIntoMemOp.cpp
@@ -0,0 +1,208 @@
+//===-- X86FoldAddrIntoMemOp.cpp - Fold address into a memory operand -----===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// This file defines a late peephole that folds an LEA whose only use is the
+// base register of the immediately following memory instruction into that
+// instruction's addressing mode, deleting the LEA:
+//
+// lea rax, [rdi + rdx*4] movl (%rdi,%rdx,4), %eax
+// mov eax, [rax] -->
+//
+// SelectionDAG already performs this fold within a basic block. The residue
+// this pass targets is the cross-block shape: an address computed in one block
+// and dereferenced in another is never part of a single DAG, so the fold does
+// not happen at isel; tail duplication and block placement later bring the LEA
+// and its consumer adjacent, but by then no pass rejoins them. Running after
+// those passes lets the fold fire on the adjacent pair.
+//
+// This is a prototype. It handles LEA64r producers only; the arithmetic
+// siblings (ADD/SUB/INC/DEC advancing a pointer, which need an EFLAGS-dead
+// gate) are a planned extension, as is looking past strictly-adjacent uses.
+//
+//===----------------------------------------------------------------------===//
+
+#include "X86.h"
+#include "X86InstrInfo.h"
+#include "X86RegisterInfo.h"
+#include "X86Subtarget.h"
+#include "llvm/ADT/Statistic.h"
+#include "llvm/CodeGen/LivePhysRegs.h"
+#include "llvm/CodeGen/MachineFunctionPass.h"
+#include "llvm/CodeGen/MachineInstr.h"
+#include "llvm/CodeGen/MachineRegisterInfo.h"
+#include "llvm/Support/MathExtras.h"
+
+using namespace llvm;
+
+#define DEBUG_TYPE "x86-fold-addr-into-memop"
+
+STATISTIC(NumFolded, "Number of LEAs folded into a memory operand");
+
+namespace {
+class X86FoldAddrIntoMemOp : public MachineFunctionPass {
+public:
+ static char ID;
+
+ X86FoldAddrIntoMemOp() : MachineFunctionPass(ID) {}
+
+ StringRef getPassName() const override {
+ return "X86 Fold Address Into Memory Operand";
+ }
+
+ bool runOnMachineFunction(MachineFunction &MF) override;
+
+ // Runs after regalloc; operates on physical registers.
+ MachineFunctionProperties getRequiredProperties() const override {
+ return MachineFunctionProperties().setNoVRegs();
+ }
+
+private:
+ // If \p LEA's result is the base register of \p Use's memory operand and can
+ // be folded into it, rewrite \p Use in place and return true (leaving \p LEA
+ // dead for the caller to erase). \p LiveAfterUse is the set of registers live
+ // after \p Use.
+ bool tryFold(MachineInstr &LEA, MachineInstr &Use,
+ const LivePhysRegs &LiveAfterUse);
+
+ const X86RegisterInfo *TRI = nullptr;
+ const MachineRegisterInfo *MRI = nullptr;
+};
+} // namespace
+
+char X86FoldAddrIntoMemOp::ID = 0;
+
+INITIALIZE_PASS(X86FoldAddrIntoMemOp, DEBUG_TYPE,
+ "X86 fold address into memory operand", false, false)
+
+FunctionPass *llvm::createX86FoldAddrIntoMemOpPass() {
+ return new X86FoldAddrIntoMemOp();
+}
+
+// A register operand that names no register (base or index absent).
+static bool isNoReg(const MachineOperand &MO) {
+ return MO.isReg() && MO.getReg() == X86::NoRegister;
+}
+
+bool X86FoldAddrIntoMemOp::tryFold(MachineInstr &LEA, MachineInstr &Use,
+ const LivePhysRegs &LiveAfterUse) {
+ // The consumer must be a plain load/store with a single, addressable memory
+ // operand. Control-transfer consumers (call/jmp/ret through memory) are left
+ // alone for the prototype.
+ if (!Use.mayLoadOrStore() || Use.isCall() || Use.isBranch() || Use.isReturn())
+ return false;
+ const MCInstrDesc &Desc = Use.getDesc();
+ int MemIdx = X86II::getMemoryOperandNo(Desc.TSFlags);
+ if (MemIdx < 0)
+ return false;
+ MemIdx += X86II::getOperandBias(Desc);
+
+ Register DefReg = LEA.getOperand(0).getReg();
+
+ // LEA64r operands: dst, then base, scale, index, disp, segment.
+ const MachineOperand &LBase = LEA.getOperand(1 + X86::AddrBaseReg);
+ const MachineOperand &LScale = LEA.getOperand(1 + X86::AddrScaleAmt);
+ const MachineOperand &LIndex = LEA.getOperand(1 + X86::AddrIndexReg);
+ const MachineOperand &LDisp = LEA.getOperand(1 + X86::AddrDisp);
+ const MachineOperand &LSeg = LEA.getOperand(1 + X86::AddrSegmentReg);
+
+ // Prototype restrictions: a real GPR base, no segment, an immediate
+ // displacement (excludes RIP-relative and global-address LEAs).
+ if (!LBase.isReg() || !LBase.getReg().isValid() || LBase.getReg() == X86::RIP)
+ return false;
+ if (!isNoReg(LSeg) || !LDisp.isImm())
+ return false;
+
+ MachineOperand &UBase = Use.getOperand(MemIdx + X86::AddrBaseReg);
+ MachineOperand &UScale = Use.getOperand(MemIdx + X86::AddrScaleAmt);
+ MachineOperand &UIndex = Use.getOperand(MemIdx + X86::AddrIndexReg);
+ MachineOperand &UDisp = Use.getOperand(MemIdx + X86::AddrDisp);
+ MachineOperand &USeg = Use.getOperand(MemIdx + X86::AddrSegmentReg);
+
+ // The LEA's result must be exactly the consumer's base, with an immediate
+ // displacement and no segment override.
+ if (!UBase.isReg() || UBase.getReg() != DefReg)
+ return false;
+ if (!isNoReg(USeg) || !UDisp.isImm())
+ return false;
+
+ // An addressing mode holds one index; a VSIB (vector) index is not a GPR and
+ // is left alone.
+ bool LHasIndex = LIndex.getReg().isValid();
+ bool UHasIndex = UIndex.getReg().isValid();
+ if (LHasIndex && UHasIndex)
+ return false;
+ if (UHasIndex && !X86::GR64RegClass.contains(UIndex.getReg()) &&
+ !X86::GR32RegClass.contains(UIndex.getReg()))
+ return false;
+
+ // The combined displacement must fit the 32-bit field.
+ int64_t Disp = LDisp.getImm() + UDisp.getImm();
+ if (!isInt<32>(Disp))
+ return false;
+
+ // DefReg may appear in the consumer only as the memory base. If it is read
+ // anywhere else (as the index or a data operand) the base cannot fold away.
+ // If the consumer fully rewrites DefReg, the folded base is dead regardless
+ // of downstream liveness.
+ bool WritesDefReg = false;
+ for (unsigned I = 0, E = Use.getNumOperands(); I != E; ++I) {
+ if (I == unsigned(MemIdx + X86::AddrBaseReg))
+ continue;
+ const MachineOperand &MO = Use.getOperand(I);
+ if (!MO.isReg() || !MO.getReg().isValid())
+ continue;
+ if (!TRI->regsOverlap(MO.getReg(), DefReg))
+ continue;
+ if (MO.isUse())
+ return false;
+ WritesDefReg = true;
+ }
+
+ // The base value must be dead after the consumer.
+ if (!WritesDefReg && !LiveAfterUse.available(*MRI, DefReg))
+ return false;
+
+ // Rewrite the consumer's addressing mode to the LEA's address plus its own
+ // displacement, then report the LEA as removable.
+ UBase.setReg(LBase.getReg());
+ if (LHasIndex) {
+ UIndex.setReg(LIndex.getReg());
+ UScale.setImm(LScale.getImm());
+ }
+ UDisp.setImm(Disp);
+ return true;
+}
+
+bool X86FoldAddrIntoMemOp::runOnMachineFunction(MachineFunction &MF) {
+ const X86Subtarget &ST = MF.getSubtarget<X86Subtarget>();
+ TRI = ST.getRegisterInfo();
+ MRI = &MF.getRegInfo();
+
+ bool Changed = false;
+ SmallVector<MachineInstr *, 8> DeadLEAs;
+ for (MachineBasicBlock &MBB : MF) {
+ // Walk the block backward tracking liveness. At the top of each iteration
+ // LiveRegs holds the registers live after the current instruction.
+ LivePhysRegs LiveRegs(*TRI);
+ LiveRegs.addLiveOuts(MBB);
+ for (MachineInstr &MI : reverse(MBB)) {
+ if (MI.getIterator() != MBB.begin()) {
+ MachineInstr &Prev = *std::prev(MI.getIterator());
+ if (Prev.getOpcode() == X86::LEA64r && tryFold(Prev, MI, LiveRegs)) {
+ DeadLEAs.push_back(&Prev);
+ ++NumFolded;
+ Changed = true;
+ }
+ }
+ LiveRegs.stepBackward(MI);
+ }
+ }
+ for (MachineInstr *LEA : DeadLEAs)
+ LEA->eraseFromParent();
+ return Changed;
+}
diff --git a/llvm/lib/Target/X86/X86TargetMachine.cpp b/llvm/lib/Target/X86/X86TargetMachine.cpp
index 886405a0c7bae..8310567337c08 100644
--- a/llvm/lib/Target/X86/X86TargetMachine.cpp
+++ b/llvm/lib/Target/X86/X86TargetMachine.cpp
@@ -97,6 +97,7 @@ extern "C" LLVM_C_ABI void LLVMInitializeX86Target() {
initializeX86LoadValueInjectionLoadHardeningLegacyPass(PR);
initializeX86LoadValueInjectionRetHardeningLegacyPass(PR);
initializeX86OptimizeLEAsLegacyPass(PR);
+ initializeX86FoldAddrIntoMemOpPass(PR);
initializeX86PartialReductionLegacyPass(PR);
initializeX86ReturnThunksLegacyPass(PR);
initializeX86DAGToDAGISelLegacyPass(PR);
@@ -568,6 +569,7 @@ void X86PassConfig::addPreEmitPass() {
addPass(createX86FixupBWInstsLegacyPass());
addPass(createX86PadShortFunctions());
addPass(createX86FixupLEAsLegacyPass());
+ addPass(createX86FoldAddrIntoMemOpPass());
addPass(createX86FixupInstTuningLegacyPass());
addPass(createX86FixupVectorConstantsLegacyPass());
}
diff --git a/llvm/test/CodeGen/X86/fold-addr-into-memop.ll b/llvm/test/CodeGen/X86/fold-addr-into-memop.ll
index c6c00b8ec5169..9e19a4dc369c1 100644
--- a/llvm/test/CodeGen/X86/fold-addr-into-memop.ll
+++ b/llvm/test/CodeGen/X86/fold-addr-into-memop.ll
@@ -14,12 +14,10 @@ define i32 @fold_idx4(ptr %a, ptr %b, i64 %i, i1 %c) {
; CHECK-NEXT: testb $1, %cl
; CHECK-NEXT: je .LBB0_2
; CHECK-NEXT: # %bb.1: # %la
-; CHECK-NEXT: leaq (%rdi,%rdx,4), %rax
-; CHECK-NEXT: movl (%rax), %eax
+; CHECK-NEXT: movl (%rdi,%rdx,4), %eax
; CHECK-NEXT: retq
; CHECK-NEXT: .LBB0_2: # %lb
-; CHECK-NEXT: leaq (%rsi,%rdx,4), %rax
-; CHECK-NEXT: movl (%rax), %eax
+; CHECK-NEXT: movl (%rsi,%rdx,4), %eax
; CHECK-NEXT: retq
br i1 %c, label %la, label %lb
la:
@@ -41,12 +39,10 @@ define i64 @fold_idx8(ptr %a, ptr %b, i64 %i, i1 %c) {
; CHECK-NEXT: testb $1, %cl
; CHECK-NEXT: je .LBB1_2
; CHECK-NEXT: # %bb.1: # %la
-; CHECK-NEXT: leaq (%rdi,%rdx,8), %rax
-; CHECK-NEXT: movq (%rax), %rax
+; CHECK-NEXT: movq (%rdi,%rdx,8), %rax
; CHECK-NEXT: retq
; CHECK-NEXT: .LBB1_2: # %lb
-; CHECK-NEXT: leaq (%rsi,%rdx,8), %rax
-; CHECK-NEXT: movq (%rax), %rax
+; CHECK-NEXT: movq (%rsi,%rdx,8), %rax
; CHECK-NEXT: retq
br i1 %c, label %la, label %lb
la:
@@ -68,12 +64,10 @@ define void @fold_store_idx4(ptr %a, ptr %b, i64 %i, i32 %x, i1 %c) {
; CHECK-NEXT: testb $1, %r8b
; CHECK-NEXT: je .LBB2_2
; CHECK-NEXT: # %bb.1: # %la
-; CHECK-NEXT: leaq (%rdi,%rdx,4), %rax
-; CHECK-NEXT: movl %ecx, (%rax)
+; CHECK-NEXT: movl %ecx, (%rdi,%rdx,4)
; CHECK-NEXT: retq
; CHECK-NEXT: .LBB2_2: # %lb
-; CHECK-NEXT: leaq (%rsi,%rdx,4), %rax
-; CHECK-NEXT: movl %ecx, (%rax)
+; CHECK-NEXT: movl %ecx, (%rsi,%rdx,4)
; CHECK-NEXT: retq
br i1 %c, label %la, label %lb
la:
@@ -97,12 +91,10 @@ define i32 @fold_addrm(ptr %a, ptr %b, i64 %i, i32 %x, i1 %c) {
; CHECK-NEXT: testb $1, %r8b
; CHECK-NEXT: je .LBB3_2
; CHECK-NEXT: # %bb.1: # %la
-; CHECK-NEXT: leaq (%rdi,%rdx,4), %rcx
-; CHECK-NEXT: addl (%rcx), %eax
+; CHECK-NEXT: addl (%rdi,%rdx,4), %eax
; CHECK-NEXT: retq
; CHECK-NEXT: .LBB3_2: # %lb
-; CHECK-NEXT: leaq (%rsi,%rdx,4), %rcx
-; CHECK-NEXT: addl (%rcx), %eax
+; CHECK-NEXT: addl (%rsi,%rdx,4), %eax
; CHECK-NEXT: retq
br i1 %c, label %la, label %lb
la:
diff --git a/llvm/test/CodeGen/X86/opt-pipeline.ll b/llvm/test/CodeGen/X86/opt-pipeline.ll
index e0256b66fff89..b942daefaa7d3 100644
--- a/llvm/test/CodeGen/X86/opt-pipeline.ll
+++ b/llvm/test/CodeGen/X86/opt-pipeline.ll
@@ -210,6 +210,7 @@
; CHECK-NEXT: Lazy Machine Block Frequency Analysis
; CHECK-NEXT: X86 Atom pad short functions
; CHECK-NEXT: X86 LEA Fixup
+; CHECK-NEXT: X86 Fold Address Into Memory Operand
; CHECK-NEXT: X86 Fixup Inst Tuning
; CHECK-NEXT: X86 Fixup Vector Constants
; CHECK-NEXT: Compressing EVEX instrs when possible
diff --git a/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll b/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
index 2e2e78a6da51e..38bdc17e71e09 100644
--- a/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
+++ b/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
@@ -955,8 +955,7 @@ define i32 @test_MM_GET_EXCEPTION_MASK() nounwind {
;
; X64-SSE-LABEL: test_MM_GET_EXCEPTION_MASK:
; X64-SSE: # %bb.0:
-; X64-SSE-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-SSE-NEXT: stmxcsr (%rax) # encoding: [0x0f,0xae,0x18]
+; X64-SSE-NEXT: stmxcsr -{{[0-9]+}}(%rsp) # encoding: [0x0f,0xae,0x5c,0x24,0xfc]
; X64-SSE-NEXT: movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
; X64-SSE-NEXT: andl $8064, %eax # encoding: [0x25,0x80,0x1f,0x00,0x00]
; X64-SSE-NEXT: # imm = 0x1F80
@@ -964,8 +963,7 @@ define i32 @test_MM_GET_EXCEPTION_MASK() nounwind {
;
; X64-AVX-LABEL: test_MM_GET_EXCEPTION_MASK:
; X64-AVX: # %bb.0:
-; X64-AVX-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-AVX-NEXT: vstmxcsr (%rax) # encoding: [0xc5,0xf8,0xae,0x18]
+; X64-AVX-NEXT: vstmxcsr -{{[0-9]+}}(%rsp) # encoding: [0xc5,0xf8,0xae,0x5c,0x24,0xfc]
; X64-AVX-NEXT: movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
; X64-AVX-NEXT: andl $8064, %eax # encoding: [0x25,0x80,0x1f,0x00,0x00]
; X64-AVX-NEXT: # imm = 0x1F80
@@ -1002,16 +1000,14 @@ define i32 @test_MM_GET_EXCEPTION_STATE() nounwind {
;
; X64-SSE-LABEL: test_MM_GET_EXCEPTION_STATE:
; X64-SSE: # %bb.0:
-; X64-SSE-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-SSE-NEXT: stmxcsr (%rax) # encoding: [0x0f,0xae,0x18]
+; X64-SSE-NEXT: stmxcsr -{{[0-9]+}}(%rsp) # encoding: [0x0f,0xae,0x5c,0x24,0xfc]
; X64-SSE-NEXT: movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
; X64-SSE-NEXT: andl $63, %eax # encoding: [0x83,0xe0,0x3f]
; X64-SSE-NEXT: retq # encoding: [0xc3]
;
; X64-AVX-LABEL: test_MM_GET_EXCEPTION_STATE:
; X64-AVX: # %bb.0:
-; X64-AVX-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-AVX-NEXT: vstmxcsr (%rax) # encoding: [0xc5,0xf8,0xae,0x18]
+; X64-AVX-NEXT: vstmxcsr -{{[0-9]+}}(%rsp) # encoding: [0xc5,0xf8,0xae,0x5c,0x24,0xfc]
; X64-AVX-NEXT: movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
; X64-AVX-NEXT: andl $63, %eax # encoding: [0x83,0xe0,0x3f]
; X64-AVX-NEXT: retq # encoding: [0xc3]
@@ -1048,8 +1044,7 @@ define i32 @test_MM_GET_FLUSH_ZERO_MODE() nounwind {
;
; X64-SSE-LABEL: test_MM_GET_FLUSH_ZERO_MODE:
; X64-SSE: # %bb.0:
-; X64-SSE-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-SSE-NEXT: stmxcsr (%rax) # encoding: [0x0f,0xae,0x18]
+; X64-SSE-NEXT: stmxcsr -{{[0-9]+}}(%rsp) # encoding: [0x0f,0xae,0x5c,0x24,0xfc]
; X64-SSE-NEXT: movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
; X64-SSE-NEXT: andl $32768, %eax # encoding: [0x25,0x00,0x80,0x00,0x00]
; X64-SSE-NEXT: # imm = 0x8000
@@ -1057,8 +1052,7 @@ define i32 @test_MM_GET_FLUSH_ZERO_MODE() nounwind {
;
; X64-AVX-LABEL: test_MM_GET_FLUSH_ZERO_MODE:
; X64-AVX: # %bb.0:
-; X64-AVX-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-AVX-NEXT: vstmxcsr (%rax) # encoding: [0xc5,0xf8,0xae,0x18]
+; X64-AVX-NEXT: vstmxcsr -{{[0-9]+}}(%rsp) # encoding: [0xc5,0xf8,0xae,0x5c,0x24,0xfc]
; X64-AVX-NEXT: movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
; X64-AVX-NEXT: andl $32768, %eax # encoding: [0x25,0x00,0x80,0x00,0x00]
; X64-AVX-NEXT: # imm = 0x8000
@@ -1096,8 +1090,7 @@ define i32 @test_MM_GET_ROUNDING_MODE() nounwind {
;
; X64-SSE-LABEL: test_MM_GET_ROUNDING_MODE:
; X64-SSE: # %bb.0:
-; X64-SSE-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-SSE-NEXT: stmxcsr (%rax) # encoding: [0x0f,0xae,0x18]
+; X64-SSE-NEXT: stmxcsr -{{[0-9]+}}(%rsp) # encoding: [0x0f,0xae,0x5c,0x24,0xfc]
; X64-SSE-NEXT: movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
; X64-SSE-NEXT: andl $24576, %eax # encoding: [0x25,0x00,0x60,0x00,0x00]
; X64-SSE-NEXT: # imm = 0x6000
@@ -1105,8 +1098,7 @@ define i32 @test_MM_GET_ROUNDING_MODE() nounwind {
;
; X64-AVX-LABEL: test_MM_GET_ROUNDING_MODE:
; X64-AVX: # %bb.0:
-; X64-AVX-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-AVX-NEXT: vstmxcsr (%rax) # encoding: [0xc5,0xf8,0xae,0x18]
+; X64-AVX-NEXT: vstmxcsr -{{[0-9]+}}(%rsp) # encoding: [0xc5,0xf8,0xae,0x5c,0x24,0xfc]
; X64-AVX-NEXT: movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
; X64-AVX-NEXT: andl $24576, %eax # encoding: [0x25,0x00,0x60,0x00,0x00]
; X64-AVX-NEXT: # imm = 0x6000
@@ -1140,15 +1132,13 @@ define i32 @test_mm_getcsr() nounwind {
;
; X64-SSE-LABEL: test_mm_getcsr:
; X64-SSE: # %bb.0:
-; X64-SSE-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-SSE-NEXT: stmxcsr (%rax) # encoding: [0x0f,0xae,0x18]
+; X64-SSE-NEXT: stmxcsr -{{[0-9]+}}(%rsp) # encoding: [0x0f,0xae,0x5c,0x24,0xfc]
; X64-SSE-NEXT: movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
; X64-SSE-NEXT: retq # encoding: [0xc3]
;
; X64-AVX-LABEL: test_mm_getcsr:
; X64-AVX: # %bb.0:
-; X64-AVX-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-AVX-NEXT: vstmxcsr (%rax) # encoding: [0xc5,0xf8,0xae,0x18]
+; X64-AVX-NEXT: vstmxcsr -{{[0-9]+}}(%rsp) # encoding: [0xc5,0xf8,0xae,0x5c,0x24,0xfc]
; X64-AVX-NEXT: movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
; X64-AVX-NEXT: retq # encoding: [0xc3]
%1 = alloca i32, align 4
@@ -2316,15 +2306,13 @@ define void @test_mm_setcsr(i32 %a0) nounwind {
; X64-SSE-LABEL: test_mm_setcsr:
; X64-SSE: # %bb.0:
; X64-SSE-NEXT: movl %edi, -{{[0-9]+}}(%rsp) # encoding: [0x89,0x7c,0x24,0xfc]
-; X64-SSE-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-SSE-NEXT: ldmxcsr (%rax) # encoding: [0x0f,0xae,0x10]
+; X64-SSE-NEXT: ldmxcsr -{{[0-9]+}}(%rsp) # encoding: [0x0f,0xae,0x54,0x24,0xfc]
; X64-SSE-NEXT: retq # encoding: [0xc3]
;
; X64-AVX-LABEL: test_mm_setcsr:
; X64-AVX: # %bb.0:
; X64-AVX-NEXT: movl %edi, -{{[0-9]+}}(%rsp) # encoding: [0x89,0x7c,0x24,0xfc]
-; X64-AVX-NEXT: leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-AVX-NEXT: vldmxcsr (%rax) # encoding: [0xc5,0xf8,0xae,0x10]
+; X64-AVX-NEXT: vldmxcsr -{{[0-9]+}}(%rsp) # encoding: [0xc5,0xf8,0xae,0x54,0x24,0xfc]
; X64-AVX-NEXT: retq # encoding: [0xc3]
%st = alloca i32, align 4
store i32 %a0, ptr %st, align 4
More information about the llvm-commits
mailing list