[llvm] [X86] Fold an LEA into a following memory operand's addressing mode (PR #221863)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 7 18:58:16 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-x86

Author: Andrew Gaul (gaul)

<details>
<summary>Changes</summary>

Add X86FoldAddrIntoMemOp, a late peephole that folds an LEA whose only use is the base register of the immediately following load or store into that instruction's addressing mode, deleting the LEA:
```
    lea  rax, [rdi + rdx*4]        movl (%rdi,%rdx,4), %eax
    mov  eax, [rax]          -->
```
SelectionDAG already performs this fold within a basic block. The residue it leaves is the cross-block shape: an address computed in one block and dereferenced in another is never part of a single DAG, so isel does not fold it; tail duplication and block placement later make the LEA and its consumer adjacent, but no pass rejoins them. Running after those passes picks up the adjacent pair. It also catches addresses fast-isel emits as a separate LEA (the stmxcsr/ldmxcsr change in the fallout test).

The fold requires the LEA's result to appear in the consumer only as the memory base and to be dead afterward, at most one index between the two, and the displacements to sum within signed 32 bits.

Motivated by an x86lint audit: a release build of the Rust compiler's librustc_driver has 5,709 "LEA foldable into memory" sites -- all register-base, dominated by array/slice element addresses and struct field offsets -- roughly 4x the count in Firefox's libxul.so, the extra coming from the cross-block control flow of Rust's enum and slice-bounds codegen.

Prototype: LEA64r producers only. The arithmetic siblings (ADD/SUB/INC/DEC advancing a pointer, gated on dead EFLAGS) and looking past strictly adjacent uses are planned extensions.

---

Patch is 21.83 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/221863.diff


6 Files Affected:

- (modified) llvm/lib/Target/X86/CMakeLists.txt (+1) 
- (modified) llvm/lib/Target/X86/X86.h (+5) 
- (added) llvm/lib/Target/X86/X86FoldAddrIntoMemOp.cpp (+208) 
- (modified) llvm/lib/Target/X86/X86TargetMachine.cpp (+2) 
- (added) llvm/test/CodeGen/X86/fold-addr-into-memop.ll (+143) 
- (modified) llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll (+12-24) 


``````````diff
diff --git a/llvm/lib/Target/X86/CMakeLists.txt b/llvm/lib/Target/X86/CMakeLists.txt
index a053eb8501701..58ae2d919bb7d 100644
--- a/llvm/lib/Target/X86/CMakeLists.txt
+++ b/llvm/lib/Target/X86/CMakeLists.txt
@@ -53,6 +53,7 @@ set(sources
   X86AvoidStoreForwardingBlocks.cpp
   X86DynAllocaExpander.cpp
   X86FixupSetCC.cpp
+  X86FoldAddrIntoMemOp.cpp
   X86FlagsCopyLowering.cpp
   X86FloatingPoint.cpp
   X86FrameLowering.cpp
diff --git a/llvm/lib/Target/X86/X86.h b/llvm/lib/Target/X86/X86.h
index eef4de389a7de..48ea02e2736cd 100644
--- a/llvm/lib/Target/X86/X86.h
+++ b/llvm/lib/Target/X86/X86.h
@@ -142,6 +142,10 @@ class X86OptimizeLEAsPass : public OptionalPassInfoMixin<X86OptimizeLEAsPass> {
 
 FunctionPass *createX86OptimizeLEAsLegacyPass();
 
+/// Return a pass that folds an LEA whose only use is the base of the following
+/// memory instruction into that instruction's addressing mode.
+FunctionPass *createX86FoldAddrIntoMemOpPass();
+
 /// Return a pass that transforms setcc + movzx pairs into xor + setcc.
 class X86FixupSetCCPass : public OptionalPassInfoMixin<X86FixupSetCCPass> {
 public:
@@ -511,6 +515,7 @@ void initializeX86LoadValueInjectionRetHardeningLegacyPass(PassRegistry &);
 void initializeX86LowerAMXIntrinsicsLegacyPassPass(PassRegistry &);
 void initializeX86LowerAMXTypeLegacyPassPass(PassRegistry &);
 void initializeX86LowerTileCopyLegacyPass(PassRegistry &);
+void initializeX86FoldAddrIntoMemOpPass(PassRegistry &);
 void initializeX86OptimizeLEAsLegacyPass(PassRegistry &);
 void initializeX86PartialReductionLegacyPass(PassRegistry &);
 void initializeX86PreTileConfigLegacyPass(PassRegistry &);
diff --git a/llvm/lib/Target/X86/X86FoldAddrIntoMemOp.cpp b/llvm/lib/Target/X86/X86FoldAddrIntoMemOp.cpp
new file mode 100644
index 0000000000000..98ecd8dff60fa
--- /dev/null
+++ b/llvm/lib/Target/X86/X86FoldAddrIntoMemOp.cpp
@@ -0,0 +1,208 @@
+//===-- X86FoldAddrIntoMemOp.cpp - fold an address into a memory operand ---===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// This file defines a late peephole that folds an LEA whose only use is the
+// base register of the immediately following memory instruction into that
+// instruction's addressing mode, deleting the LEA:
+//
+//     lea  rax, [rdi + rdx*4]        movl (%rdi,%rdx,4), %eax
+//     mov  eax, [rax]          -->
+//
+// SelectionDAG already performs this fold within a basic block. The residue
+// this pass targets is the cross-block shape: an address computed in one block
+// and dereferenced in another is never part of a single DAG, so the fold does
+// not happen at isel; tail duplication and block placement later bring the LEA
+// and its consumer adjacent, but by then no pass rejoins them. Running after
+// those passes lets the fold fire on the adjacent pair.
+//
+// This is a prototype. It handles LEA64r producers only; the arithmetic
+// siblings (ADD/SUB/INC/DEC advancing a pointer, which need an EFLAGS-dead
+// gate) are a planned extension, as is looking past strictly-adjacent uses.
+//
+//===----------------------------------------------------------------------===//
+
+#include "X86.h"
+#include "X86InstrInfo.h"
+#include "X86RegisterInfo.h"
+#include "X86Subtarget.h"
+#include "llvm/ADT/Statistic.h"
+#include "llvm/CodeGen/LivePhysRegs.h"
+#include "llvm/CodeGen/MachineFunctionPass.h"
+#include "llvm/CodeGen/MachineInstr.h"
+#include "llvm/CodeGen/MachineRegisterInfo.h"
+#include "llvm/Support/MathExtras.h"
+
+using namespace llvm;
+
+#define DEBUG_TYPE "x86-fold-addr-into-memop"
+
+STATISTIC(NumFolded, "Number of LEAs folded into a memory operand");
+
+namespace {
+class X86FoldAddrIntoMemOp : public MachineFunctionPass {
+public:
+  static char ID;
+
+  X86FoldAddrIntoMemOp() : MachineFunctionPass(ID) {}
+
+  StringRef getPassName() const override {
+    return "X86 Fold Address Into Memory Operand";
+  }
+
+  bool runOnMachineFunction(MachineFunction &MF) override;
+
+  // Runs after regalloc; operates on physical registers.
+  MachineFunctionProperties getRequiredProperties() const override {
+    return MachineFunctionProperties().setNoVRegs();
+  }
+
+private:
+  // If \p LEA's result is the base register of \p Use's memory operand and can
+  // be folded into it, rewrite \p Use in place and return true (leaving \p LEA
+  // dead for the caller to erase). \p LiveAfterUse is the set of registers live
+  // after \p Use.
+  bool tryFold(MachineInstr &LEA, MachineInstr &Use,
+               const LivePhysRegs &LiveAfterUse);
+
+  const X86RegisterInfo *TRI = nullptr;
+  const MachineRegisterInfo *MRI = nullptr;
+};
+} // namespace
+
+char X86FoldAddrIntoMemOp::ID = 0;
+
+INITIALIZE_PASS(X86FoldAddrIntoMemOp, DEBUG_TYPE,
+                "X86 fold address into memory operand", false, false)
+
+FunctionPass *llvm::createX86FoldAddrIntoMemOpPass() {
+  return new X86FoldAddrIntoMemOp();
+}
+
+// A register operand that names no register (base or index absent).
+static bool isNoReg(const MachineOperand &MO) {
+  return MO.isReg() && MO.getReg() == X86::NoRegister;
+}
+
+bool X86FoldAddrIntoMemOp::tryFold(MachineInstr &LEA, MachineInstr &Use,
+                                   const LivePhysRegs &LiveAfterUse) {
+  // The consumer must be a plain load/store with a single, addressable memory
+  // operand. Control-transfer consumers (call/jmp/ret through memory) are left
+  // alone for the prototype.
+  if (!Use.mayLoadOrStore() || Use.isCall() || Use.isBranch() || Use.isReturn())
+    return false;
+  const MCInstrDesc &Desc = Use.getDesc();
+  int MemIdx = X86II::getMemoryOperandNo(Desc.TSFlags);
+  if (MemIdx < 0)
+    return false;
+  MemIdx += X86II::getOperandBias(Desc);
+
+  Register DefReg = LEA.getOperand(0).getReg();
+
+  // LEA64r operands: dst, then base, scale, index, disp, segment.
+  const MachineOperand &LBase = LEA.getOperand(1 + X86::AddrBaseReg);
+  const MachineOperand &LScale = LEA.getOperand(1 + X86::AddrScaleAmt);
+  const MachineOperand &LIndex = LEA.getOperand(1 + X86::AddrIndexReg);
+  const MachineOperand &LDisp = LEA.getOperand(1 + X86::AddrDisp);
+  const MachineOperand &LSeg = LEA.getOperand(1 + X86::AddrSegmentReg);
+
+  // Prototype restrictions: a real GPR base, no segment, an immediate
+  // displacement (excludes RIP-relative and global-address LEAs).
+  if (!LBase.isReg() || !LBase.getReg().isValid() || LBase.getReg() == X86::RIP)
+    return false;
+  if (!isNoReg(LSeg) || !LDisp.isImm())
+    return false;
+
+  MachineOperand &UBase = Use.getOperand(MemIdx + X86::AddrBaseReg);
+  MachineOperand &UScale = Use.getOperand(MemIdx + X86::AddrScaleAmt);
+  MachineOperand &UIndex = Use.getOperand(MemIdx + X86::AddrIndexReg);
+  MachineOperand &UDisp = Use.getOperand(MemIdx + X86::AddrDisp);
+  MachineOperand &USeg = Use.getOperand(MemIdx + X86::AddrSegmentReg);
+
+  // The LEA's result must be exactly the consumer's base, with an immediate
+  // displacement and no segment override.
+  if (!UBase.isReg() || UBase.getReg() != DefReg)
+    return false;
+  if (!isNoReg(USeg) || !UDisp.isImm())
+    return false;
+
+  // An addressing mode holds one index; a VSIB (vector) index is not a GPR and
+  // is left alone.
+  bool LHasIndex = LIndex.getReg().isValid();
+  bool UHasIndex = UIndex.getReg().isValid();
+  if (LHasIndex && UHasIndex)
+    return false;
+  if (UHasIndex && !X86::GR64RegClass.contains(UIndex.getReg()) &&
+      !X86::GR32RegClass.contains(UIndex.getReg()))
+    return false;
+
+  // The combined displacement must fit the 32-bit field.
+  int64_t Disp = LDisp.getImm() + UDisp.getImm();
+  if (!isInt<32>(Disp))
+    return false;
+
+  // DefReg may appear in the consumer only as the memory base. If it is read
+  // anywhere else (as the index or a data operand) the base cannot fold away.
+  // If the consumer fully rewrites DefReg, the folded base is dead regardless
+  // of downstream liveness.
+  bool WritesDefReg = false;
+  for (unsigned I = 0, E = Use.getNumOperands(); I != E; ++I) {
+    if (I == unsigned(MemIdx + X86::AddrBaseReg))
+      continue;
+    const MachineOperand &MO = Use.getOperand(I);
+    if (!MO.isReg() || !MO.getReg().isValid())
+      continue;
+    if (!TRI->regsOverlap(MO.getReg(), DefReg))
+      continue;
+    if (MO.isUse())
+      return false;
+    WritesDefReg = true;
+  }
+
+  // The base value must be dead after the consumer.
+  if (!WritesDefReg && !LiveAfterUse.available(*MRI, DefReg))
+    return false;
+
+  // Rewrite the consumer's addressing mode to the LEA's address plus its own
+  // displacement, then report the LEA as removable.
+  UBase.setReg(LBase.getReg());
+  if (LHasIndex) {
+    UIndex.setReg(LIndex.getReg());
+    UScale.setImm(LScale.getImm());
+  }
+  UDisp.setImm(Disp);
+  return true;
+}
+
+bool X86FoldAddrIntoMemOp::runOnMachineFunction(MachineFunction &MF) {
+  const X86Subtarget &ST = MF.getSubtarget<X86Subtarget>();
+  TRI = ST.getRegisterInfo();
+  MRI = &MF.getRegInfo();
+
+  bool Changed = false;
+  SmallVector<MachineInstr *, 8> DeadLEAs;
+  for (MachineBasicBlock &MBB : MF) {
+    // Walk the block backward tracking liveness. At the top of each iteration
+    // LiveRegs holds the registers live after the current instruction.
+    LivePhysRegs LiveRegs(*TRI);
+    LiveRegs.addLiveOuts(MBB);
+    for (MachineInstr &MI : reverse(MBB)) {
+      if (MI.getIterator() != MBB.begin()) {
+        MachineInstr &Prev = *std::prev(MI.getIterator());
+        if (Prev.getOpcode() == X86::LEA64r && tryFold(Prev, MI, LiveRegs)) {
+          DeadLEAs.push_back(&Prev);
+          ++NumFolded;
+          Changed = true;
+        }
+      }
+      LiveRegs.stepBackward(MI);
+    }
+  }
+  for (MachineInstr *LEA : DeadLEAs)
+    LEA->eraseFromParent();
+  return Changed;
+}
diff --git a/llvm/lib/Target/X86/X86TargetMachine.cpp b/llvm/lib/Target/X86/X86TargetMachine.cpp
index 886405a0c7bae..8310567337c08 100644
--- a/llvm/lib/Target/X86/X86TargetMachine.cpp
+++ b/llvm/lib/Target/X86/X86TargetMachine.cpp
@@ -97,6 +97,7 @@ extern "C" LLVM_C_ABI void LLVMInitializeX86Target() {
   initializeX86LoadValueInjectionLoadHardeningLegacyPass(PR);
   initializeX86LoadValueInjectionRetHardeningLegacyPass(PR);
   initializeX86OptimizeLEAsLegacyPass(PR);
+  initializeX86FoldAddrIntoMemOpPass(PR);
   initializeX86PartialReductionLegacyPass(PR);
   initializeX86ReturnThunksLegacyPass(PR);
   initializeX86DAGToDAGISelLegacyPass(PR);
@@ -568,6 +569,7 @@ void X86PassConfig::addPreEmitPass() {
     addPass(createX86FixupBWInstsLegacyPass());
     addPass(createX86PadShortFunctions());
     addPass(createX86FixupLEAsLegacyPass());
+    addPass(createX86FoldAddrIntoMemOpPass());
     addPass(createX86FixupInstTuningLegacyPass());
     addPass(createX86FixupVectorConstantsLegacyPass());
   }
diff --git a/llvm/test/CodeGen/X86/fold-addr-into-memop.ll b/llvm/test/CodeGen/X86/fold-addr-into-memop.ll
new file mode 100644
index 0000000000000..9e19a4dc369c1
--- /dev/null
+++ b/llvm/test/CodeGen/X86/fold-addr-into-memop.ll
@@ -0,0 +1,143 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -verify-machineinstrs | FileCheck %s
+
+; An address computed in one block and dereferenced in another is brought
+; adjacent (lea then memory op) only after tail duplication, past the point
+; where SelectionDAG folds an address into a memory operand. These check that
+; the late peephole rejoins the pair: the lea disappears into the load/store's
+; addressing mode.
+
+; base + index*4 load
+define i32 @fold_idx4(ptr %a, ptr %b, i64 %i, i1 %c) {
+; CHECK-LABEL: fold_idx4:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    testb $1, %cl
+; CHECK-NEXT:    je .LBB0_2
+; CHECK-NEXT:  # %bb.1: # %la
+; CHECK-NEXT:    movl (%rdi,%rdx,4), %eax
+; CHECK-NEXT:    retq
+; CHECK-NEXT:  .LBB0_2: # %lb
+; CHECK-NEXT:    movl (%rsi,%rdx,4), %eax
+; CHECK-NEXT:    retq
+  br i1 %c, label %la, label %lb
+la:
+  %qa = getelementptr i32, ptr %a, i64 %i
+  br label %join
+lb:
+  %qb = getelementptr i32, ptr %b, i64 %i
+  br label %join
+join:
+  %q = phi ptr [ %qa, %la ], [ %qb, %lb ]
+  %v = load i32, ptr %q
+  ret i32 %v
+}
+
+; base + index*8 load; the load also fully rewrites the base register
+define i64 @fold_idx8(ptr %a, ptr %b, i64 %i, i1 %c) {
+; CHECK-LABEL: fold_idx8:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    testb $1, %cl
+; CHECK-NEXT:    je .LBB1_2
+; CHECK-NEXT:  # %bb.1: # %la
+; CHECK-NEXT:    movq (%rdi,%rdx,8), %rax
+; CHECK-NEXT:    retq
+; CHECK-NEXT:  .LBB1_2: # %lb
+; CHECK-NEXT:    movq (%rsi,%rdx,8), %rax
+; CHECK-NEXT:    retq
+  br i1 %c, label %la, label %lb
+la:
+  %qa = getelementptr i64, ptr %a, i64 %i
+  br label %join
+lb:
+  %qb = getelementptr i64, ptr %b, i64 %i
+  br label %join
+join:
+  %q = phi ptr [ %qa, %la ], [ %qb, %lb ]
+  %v = load i64, ptr %q
+  ret i64 %v
+}
+
+; store consumer
+define void @fold_store_idx4(ptr %a, ptr %b, i64 %i, i32 %x, i1 %c) {
+; CHECK-LABEL: fold_store_idx4:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    testb $1, %r8b
+; CHECK-NEXT:    je .LBB2_2
+; CHECK-NEXT:  # %bb.1: # %la
+; CHECK-NEXT:    movl %ecx, (%rdi,%rdx,4)
+; CHECK-NEXT:    retq
+; CHECK-NEXT:  .LBB2_2: # %lb
+; CHECK-NEXT:    movl %ecx, (%rsi,%rdx,4)
+; CHECK-NEXT:    retq
+  br i1 %c, label %la, label %lb
+la:
+  %qa = getelementptr i32, ptr %a, i64 %i
+  br label %join
+lb:
+  %qb = getelementptr i32, ptr %b, i64 %i
+  br label %join
+join:
+  %q = phi ptr [ %qa, %la ], [ %qb, %lb ]
+  store i32 %x, ptr %q
+  ret void
+}
+
+; load-op consumer (add reg, mem): the folded address feeds the add's memory
+; operand
+define i32 @fold_addrm(ptr %a, ptr %b, i64 %i, i32 %x, i1 %c) {
+; CHECK-LABEL: fold_addrm:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    movl %ecx, %eax
+; CHECK-NEXT:    testb $1, %r8b
+; CHECK-NEXT:    je .LBB3_2
+; CHECK-NEXT:  # %bb.1: # %la
+; CHECK-NEXT:    addl (%rdi,%rdx,4), %eax
+; CHECK-NEXT:    retq
+; CHECK-NEXT:  .LBB3_2: # %lb
+; CHECK-NEXT:    addl (%rsi,%rdx,4), %eax
+; CHECK-NEXT:    retq
+  br i1 %c, label %la, label %lb
+la:
+  %qa = getelementptr i32, ptr %a, i64 %i
+  br label %join
+lb:
+  %qb = getelementptr i32, ptr %b, i64 %i
+  br label %join
+join:
+  %q = phi ptr [ %qa, %la ], [ %qb, %lb ]
+  %v = load i32, ptr %q
+  %s = add i32 %v, %x
+  ret i32 %s
+}
+
+; negative: the address has a second use after the load, so the base stays
+; live and the lea must remain.
+define i64 @no_fold_multiuse(ptr %a, ptr %b, i64 %i, i1 %c) {
+; CHECK-LABEL: no_fold_multiuse:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    testb $1, %cl
+; CHECK-NEXT:    je .LBB4_2
+; CHECK-NEXT:  # %bb.1: # %la
+; CHECK-NEXT:    leaq (%rdi,%rdx,8), %rcx
+; CHECK-NEXT:    jmp .LBB4_3
+; CHECK-NEXT:  .LBB4_2: # %lb
+; CHECK-NEXT:    leaq (%rsi,%rdx,8), %rcx
+; CHECK-NEXT:  .LBB4_3: # %join
+; CHECK-NEXT:    movq (%rcx), %rax
+; CHECK-NEXT:    addq 8(%rcx), %rax
+; CHECK-NEXT:    retq
+  br i1 %c, label %la, label %lb
+la:
+  %qa = getelementptr i64, ptr %a, i64 %i
+  br label %join
+lb:
+  %qb = getelementptr i64, ptr %b, i64 %i
+  br label %join
+join:
+  %q = phi ptr [ %qa, %la ], [ %qb, %lb ]
+  %v = load i64, ptr %q
+  %q2 = getelementptr i64, ptr %q, i64 1
+  %w = load i64, ptr %q2
+  %s = add i64 %v, %w
+  ret i64 %s
+}
diff --git a/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll b/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
index 2e2e78a6da51e..38bdc17e71e09 100644
--- a/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
+++ b/llvm/test/CodeGen/X86/sse-intrinsics-fast-isel.ll
@@ -955,8 +955,7 @@ define i32 @test_MM_GET_EXCEPTION_MASK() nounwind {
 ;
 ; X64-SSE-LABEL: test_MM_GET_EXCEPTION_MASK:
 ; X64-SSE:       # %bb.0:
-; X64-SSE-NEXT:    leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-SSE-NEXT:    stmxcsr (%rax) # encoding: [0x0f,0xae,0x18]
+; X64-SSE-NEXT:    stmxcsr -{{[0-9]+}}(%rsp) # encoding: [0x0f,0xae,0x5c,0x24,0xfc]
 ; X64-SSE-NEXT:    movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
 ; X64-SSE-NEXT:    andl $8064, %eax # encoding: [0x25,0x80,0x1f,0x00,0x00]
 ; X64-SSE-NEXT:    # imm = 0x1F80
@@ -964,8 +963,7 @@ define i32 @test_MM_GET_EXCEPTION_MASK() nounwind {
 ;
 ; X64-AVX-LABEL: test_MM_GET_EXCEPTION_MASK:
 ; X64-AVX:       # %bb.0:
-; X64-AVX-NEXT:    leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-AVX-NEXT:    vstmxcsr (%rax) # encoding: [0xc5,0xf8,0xae,0x18]
+; X64-AVX-NEXT:    vstmxcsr -{{[0-9]+}}(%rsp) # encoding: [0xc5,0xf8,0xae,0x5c,0x24,0xfc]
 ; X64-AVX-NEXT:    movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
 ; X64-AVX-NEXT:    andl $8064, %eax # encoding: [0x25,0x80,0x1f,0x00,0x00]
 ; X64-AVX-NEXT:    # imm = 0x1F80
@@ -1002,16 +1000,14 @@ define i32 @test_MM_GET_EXCEPTION_STATE() nounwind {
 ;
 ; X64-SSE-LABEL: test_MM_GET_EXCEPTION_STATE:
 ; X64-SSE:       # %bb.0:
-; X64-SSE-NEXT:    leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-SSE-NEXT:    stmxcsr (%rax) # encoding: [0x0f,0xae,0x18]
+; X64-SSE-NEXT:    stmxcsr -{{[0-9]+}}(%rsp) # encoding: [0x0f,0xae,0x5c,0x24,0xfc]
 ; X64-SSE-NEXT:    movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
 ; X64-SSE-NEXT:    andl $63, %eax # encoding: [0x83,0xe0,0x3f]
 ; X64-SSE-NEXT:    retq # encoding: [0xc3]
 ;
 ; X64-AVX-LABEL: test_MM_GET_EXCEPTION_STATE:
 ; X64-AVX:       # %bb.0:
-; X64-AVX-NEXT:    leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-AVX-NEXT:    vstmxcsr (%rax) # encoding: [0xc5,0xf8,0xae,0x18]
+; X64-AVX-NEXT:    vstmxcsr -{{[0-9]+}}(%rsp) # encoding: [0xc5,0xf8,0xae,0x5c,0x24,0xfc]
 ; X64-AVX-NEXT:    movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
 ; X64-AVX-NEXT:    andl $63, %eax # encoding: [0x83,0xe0,0x3f]
 ; X64-AVX-NEXT:    retq # encoding: [0xc3]
@@ -1048,8 +1044,7 @@ define i32 @test_MM_GET_FLUSH_ZERO_MODE() nounwind {
 ;
 ; X64-SSE-LABEL: test_MM_GET_FLUSH_ZERO_MODE:
 ; X64-SSE:       # %bb.0:
-; X64-SSE-NEXT:    leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-SSE-NEXT:    stmxcsr (%rax) # encoding: [0x0f,0xae,0x18]
+; X64-SSE-NEXT:    stmxcsr -{{[0-9]+}}(%rsp) # encoding: [0x0f,0xae,0x5c,0x24,0xfc]
 ; X64-SSE-NEXT:    movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
 ; X64-SSE-NEXT:    andl $32768, %eax # encoding: [0x25,0x00,0x80,0x00,0x00]
 ; X64-SSE-NEXT:    # imm = 0x8000
@@ -1057,8 +1052,7 @@ define i32 @test_MM_GET_FLUSH_ZERO_MODE() nounwind {
 ;
 ; X64-AVX-LABEL: test_MM_GET_FLUSH_ZERO_MODE:
 ; X64-AVX:       # %bb.0:
-; X64-AVX-NEXT:    leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-AVX-NEXT:    vstmxcsr (%rax) # encoding: [0xc5,0xf8,0xae,0x18]
+; X64-AVX-NEXT:    vstmxcsr -{{[0-9]+}}(%rsp) # encoding: [0xc5,0xf8,0xae,0x5c,0x24,0xfc]
 ; X64-AVX-NEXT:    movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
 ; X64-AVX-NEXT:    andl $32768, %eax # encoding: [0x25,0x00,0x80,0x00,0x00]
 ; X64-AVX-NEXT:    # imm = 0x8000
@@ -1096,8 +1090,7 @@ define i32 @test_MM_GET_ROUNDING_MODE() nounwind {
 ;
 ; X64-SSE-LABEL: test_MM_GET_ROUNDING_MODE:
 ; X64-SSE:       # %bb.0:
-; X64-SSE-NEXT:    leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-SSE-NEXT:    stmxcsr (%rax) # encoding: [0x0f,0xae,0x18]
+; X64-SSE-NEXT:    stmxcsr -{{[0-9]+}}(%rsp) # encoding: [0x0f,0xae,0x5c,0x24,0xfc]
 ; X64-SSE-NEXT:    movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
 ; X64-SSE-NEXT:    andl $24576, %eax # encoding: [0x25,0x00,0x60,0x00,0x00]
 ; X64-SSE-NEXT:    # imm = 0x6000
@@ -1105,8 +1098,7 @@ define i32 @test_MM_GET_ROUNDING_MODE() nounwind {
 ;
 ; X64-AVX-LABEL: test_MM_GET_ROUNDING_MODE:
 ; X64-AVX:       # %bb.0:
-; X64-AVX-NEXT:    leaq -{{[0-9]+}}(%rsp), %rax # encoding: [0x48,0x8d,0x44,0x24,0xfc]
-; X64-AVX-NEXT:    vstmxcsr (%rax) # encoding: [0xc5,0xf8,0xae,0x18]
+; X64-AVX-NEXT:    vstmxcsr -{{[0-9]+}}(%rsp) # encoding: [0xc5,0xf8,0xae,0x5c,0x24,0xfc]
 ; X64-AVX-NEXT:    movl -{{[0-9]+}}(%rsp), %eax # encoding: [0x8b,0x44,0x24,0xfc]
 ; X64-AVX-NEXT:    andl $24576, %eax # encoding: [0x25,0x00,0x60,0x00,0x00]
 ; X64-AVX-NEXT:    # imm = 0x6000
@@ -1140,15 +1132,13 @@ define i32 @test_mm_getcsr() nounwind {
 ;
 ; X64-SSE-LABEL: test_m...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/221863


More information about the llvm-commits mailing list