[llvm-branch-commits] [llvm] [TargetInstrInfo] Fix folding inline asm operands next to tied operands (PR #229627)

Bill Wendling via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Thu Oct 8 00:19:50 PDT 2026


https://github.com/isanbard updated https://github.com/llvm/llvm-project/pull/229627

>From a03834104d226c02aa1e295e052f927d5190abf3 Mon Sep 17 00:00:00 2001
From: Bill Wendling <isanbard at gmail.com>
Date: Tue, 6 Oct 2026 06:11:01 -0700
Subject: [PATCH 1/4] [TargetInstrInfo] Fix folding inline asm operands next to
 tied operands

foldInlineAsmMemOperand() swapped a register operand for the target's
memory operands with MachineInstr::removeOperand(), which asserts when a
later operand is tied, because moving it would break the tie. Inline asm
lists every input after every output, so folding any operand that comes
before another tied pair asserted, e.g. an "rm" input followed by the
input of a "+r" operand, or one of two "+rm" operands. Without
assertions, the moved operands kept stale tie indices. Untie the
operands, rebuild the operand list, and re-tie the remaining pairs at
their new positions.

It also gave up when the register appears in more than one operand, e.g.
one value passed to two "rm" operands, which left the greedy allocator
unable to spill that value at all. Fold every such operand into the
stack slot.

Finally, take MayLoad from the folded operands rather than from every
read of the register: a folded def whose tied use is another virtual
register (as when the fast register allocator lowers tied operands
itself) still reads the slot, and an unrelated register read of the
same value doesn't.

Nothing in tree marks an inline asm register operand foldable yet, so
the test sets the "foldable" flag in MIR and runs the greedy allocator.

Assisted-by: Claude Opus 5.5
---
 llvm/lib/CodeGen/TargetInstrInfo.cpp          | 119 +++++++++++++-----
 .../CodeGen/X86/greedy-inline-asm-fold.mir    |  80 ++++++++++++
 2 files changed, 166 insertions(+), 33 deletions(-)
 create mode 100644 llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir

diff --git a/llvm/lib/CodeGen/TargetInstrInfo.cpp b/llvm/lib/CodeGen/TargetInstrInfo.cpp
index e5bb73dc4b5811..eeb6885497e6ce 100644
--- a/llvm/lib/CodeGen/TargetInstrInfo.cpp
+++ b/llvm/lib/CodeGen/TargetInstrInfo.cpp
@@ -650,29 +650,70 @@ static MachineInstr *foldPatchpoint(MachineFunction &MF, MachineInstr &MI,
   return NewMI;
 }
 
-static void foldInlineAsmMemOperand(MachineInstr *MI, unsigned OpNo, int FI,
-                                    const TargetInstrInfo &TII) {
-  // If the machine operand is tied, untie it first.
-  if (MI->getOperand(OpNo).isTied()) {
-    unsigned TiedTo = MI->findTiedOperandIdx(OpNo);
-    MI->untieRegOperand(OpNo);
-    // Intentional recursion!
-    foldInlineAsmMemOperand(MI, TiedTo, FI, TII);
+/// Rewrite the register operands \p Ops of the inline asm \p MI to refer to
+/// stack slot \p FI. The use tied to a folded def is folded along with it,
+/// since the two name a single location.
+static void foldInlineAsmMemOperands(MachineInstr &MI, ArrayRef<unsigned> Ops,
+                                     int FI, const TargetInstrInfo &TII) {
+  SmallVector<MachineOperand, 5> MemOps;
+  TII.getFrameIndexOperands(MemOps, FI);
+  assert(!MemOps.empty() && "getFrameIndexOperands didn't create any operands");
+  InlineAsm::Flag MemFlag(InlineAsm::Kind::Mem, MemOps.size());
+  MemFlag.setMemConstraint(InlineAsm::ConstraintCode::m);
+
+  // Find the uses to fold along with their defs, and the ties to keep.
+  SmallVector<unsigned, 4> FoldOps(Ops);
+  SmallVector<std::pair<unsigned, unsigned>, 4> Ties;
+  for (unsigned I = InlineAsm::MIOp_FirstOperand, E = MI.getNumOperands();
+       I != E; ++I) {
+    const MachineOperand &MO = MI.getOperand(I);
+    if (!MO.isReg() || !MO.isTied() || !MO.isUse())
+      continue;
+
+    unsigned DefIdx = MI.findTiedOperandIdx(I);
+    if (is_contained(Ops, DefIdx))
+      FoldOps.push_back(I);
+    else
+      Ties.emplace_back(DefIdx, I);
   }
 
-  SmallVector<MachineOperand, 5> NewOps;
-  TII.getFrameIndexOperands(NewOps, FI);
-  assert(!NewOps.empty() && "getFrameIndexOperands didn't create any operands");
-  MI->removeOperand(OpNo);
-  MI->insert(MI->operands_begin() + OpNo, NewOps);
+  // A folded operand becomes MemOps.size() operands, moving every later one,
+  // and MachineInstr can't move a tied operand. So untie everything, re-add
+  // the operands from the first folded one on, and re-tie the pairs that are
+  // still registers at their new positions.
+  for (unsigned I = InlineAsm::MIOp_FirstOperand, E = MI.getNumOperands();
+       I != E; ++I)
+    MI.untieRegOperand(I);
+
+  unsigned First = *llvm::min_element(FoldOps);
+  SmallVector<MachineOperand, 16> Tail(MI.operands_begin() + First,
+                                       MI.operands_end());
+  while (MI.getNumOperands() > First)
+    MI.removeOperand(MI.getNumOperands() - 1);
+
+  SmallVector<unsigned, 16> NewIdx;
+  for (auto [I, MO] : enumerate(Tail)) {
+    NewIdx.push_back(MI.getNumOperands());
+    if (!is_contained(FoldOps, First + I)) {
+      MI.addOperand(MO);
+      continue;
+    }
+
+    // Only the first register after a flag can be folded, so the operand just
+    // before this one is its flag. It now describes the memory operand.
+    MachineOperand &FlagMO = MI.getOperand(MI.getNumOperands() - 1);
+    assert(InlineAsm::Flag(FlagMO.getImm()).getNumOperandRegisters() == 1 &&
+           "cannot fold one register of a multi-register operand");
+    FlagMO.setImm(MemFlag);
+    for (const MachineOperand &MemOp : MemOps)
+      MI.addOperand(MemOp);
+  }
 
-  // Change the previous operand to a MemKind InlineAsm::Flag. The second param
-  // is the per-target number of operands that represent the memory operand
-  // excluding this one (MD). This includes MO.
-  InlineAsm::Flag F(InlineAsm::Kind::Mem, NewOps.size());
-  F.setMemConstraint(InlineAsm::ConstraintCode::m);
-  MachineOperand &MD = MI->getOperand(OpNo - 1);
-  MD.setImm(F);
+  auto Remap = [&](unsigned Idx) {
+    return Idx < First ? Idx : NewIdx[Idx - First];
+  };
+  for (auto [DefIdx, UseIdx] : Ties)
+    MI.tieOperands(Remap(DefIdx), Remap(UseIdx));
 }
 
 // Returns nullptr if not possible to fold.
@@ -680,29 +721,41 @@ static MachineInstr *foldInlineAsmMemOperand(MachineInstr &MI,
                                              ArrayRef<unsigned> Ops, int FI,
                                              const TargetInstrInfo &TII) {
   assert(MI.isInlineAsm() && "wrong opcode");
-  if (Ops.size() > 1)
-    return nullptr;
-  unsigned Op = Ops[0];
-  assert(Op && "should never be first operand");
-  assert(MI.getOperand(Op).isReg() && "shouldn't be folding non-reg operands");
 
-  if (!MI.mayFoldInlineAsmRegOp(Op))
-    return nullptr;
+  // Every operand in Ops holds the same register (e.g. one value passed to two
+  // "rm" operands), so they all fold to the same stack slot. The asm reads
+  // the slot through a use, and through a def tied to a use, which need not
+  // be the same virtual register when the allocator lowers ties itself.
+  bool Reads = false;
+  bool Writes = false;
+  for (unsigned Op : Ops) {
+    assert(Op && "should never be first operand");
+    const MachineOperand &MO = MI.getOperand(Op);
+    assert(MO.isReg() && "shouldn't be folding non-reg operands");
 
-  MachineInstr &NewMI = TII.duplicate(*MI.getParent(), MI.getIterator(), MI);
+    // A tied use is only ever folded along with its def.
+    if (MI.isRegTiedToDefOperand(Op) || !MI.mayFoldInlineAsmRegOp(Op))
+      return nullptr;
 
-  foldInlineAsmMemOperand(&NewMI, Op, FI, TII);
+    Reads |= MO.readsReg();
+    if (MO.isDef()) {
+      Writes = true;
+      Reads |=
+          MO.isTied() && MI.getOperand(MI.findTiedOperandIdx(Op)).readsReg();
+    }
+  }
+
+  MachineInstr &NewMI = TII.duplicate(*MI.getParent(), MI.getIterator(), MI);
+  foldInlineAsmMemOperands(NewMI, Ops, FI, TII);
 
   // Update mayload/maystore metadata, and memoperands.
-  const VirtRegInfo &RI =
-      AnalyzeVirtRegInBundle(MI, MI.getOperand(Op).getReg());
   MachineOperand &ExtraMO = NewMI.getOperand(InlineAsm::MIOp_ExtraInfo);
   MachineMemOperand::Flags Flags = MachineMemOperand::MONone;
-  if (RI.Reads) {
+  if (Reads) {
     ExtraMO.setImm(ExtraMO.getImm() | InlineAsm::Extra_MayLoad);
     Flags |= MachineMemOperand::MOLoad;
   }
-  if (RI.Writes) {
+  if (Writes) {
     ExtraMO.setImm(ExtraMO.getImm() | InlineAsm::Extra_MayStore);
     Flags |= MachineMemOperand::MOStore;
   }
diff --git a/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir b/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir
new file mode 100644
index 00000000000000..3d7842964304b3
--- /dev/null
+++ b/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir
@@ -0,0 +1,80 @@
+# RUN: llc -mtriple=x86_64-unknown-linux-gnu -run-pass=greedy \
+# RUN:   -verify-machineinstrs %s -o - | FileCheck %s
+
+# The greedy register allocator spills a value used by a foldable ("rm")
+# inline asm operand by folding the operand to the value's stack slot. The
+# clobbers below leave six GPRs, and the asm needs seven registers, so one
+# value is spilled. Folding replaces an operand with the five operands of a
+# memory reference, moving every operand after it, which must keep any tied
+# operands tied.
+
+---
+# One value passed to two "rm" operands: both operands fold to its stack slot.
+name: rm_same_value_twice
+tracksRegLiveness: true
+fixedStack:
+  - { id: 0, size: 8, alignment: 16, isImmutable: true }
+body: |
+  bb.0:
+    liveins: $rdi, $rsi, $rdx, $rcx, $r8, $r9
+
+    ; CHECK-LABEL: name: rm_same_value_twice
+    ; CHECK: INLINEASM &"", sideeffect mayload attdialect, mem:m, %stack.[[SLOT:[0-9]+]], 1, $noreg, 0, $noreg, mem:m, %stack.[[SLOT]], 1, $noreg, 0, $noreg, reguse:GR64,
+    %0:gr64 = COPY $rdi
+    %1:gr64 = COPY $rsi
+    %2:gr64 = COPY $rdx
+    %3:gr64 = COPY $rcx
+    %4:gr64 = COPY $r8
+    %5:gr64 = COPY $r9
+    %6:gr64 = MOV64rm %fixed-stack.0, 1, $noreg, 0, $noreg :: (load (s64) from %fixed-stack.0, align 16)
+    INLINEASM &"", sideeffect attdialect, reguse:GR64 foldable, %0, reguse:GR64 foldable, %0, reguse:GR64, %1, reguse:GR64, %2, reguse:GR64, %3, reguse:GR64, %4, reguse:GR64, %5, reguse:GR64, %6, clobber, implicit-def dead early-clobber $rax, clobber, implicit-def dead early-clobber $rbx, clobber, implicit-def dead early-clobber $rbp, clobber, implicit-def dead early-clobber $r10, clobber, implicit-def dead early-clobber $r11, clobber, implicit-def dead early-clobber $r12, clobber, implicit-def dead early-clobber $r13, clobber, implicit-def dead early-clobber $r14, clobber, implicit-def dead early-clobber $r15
+    RET 0
+...
+---
+# Folding the "rm" input moves the input tied to the "=r" output after it.
+name: rm_before_tied
+tracksRegLiveness: true
+fixedStack:
+  - { id: 0, size: 8, alignment: 16, isImmutable: true }
+body: |
+  bb.0:
+    liveins: $rdi, $rsi, $rdx, $rcx, $r8, $r9
+
+    ; CHECK-LABEL: name: rm_before_tied
+    ; CHECK: INLINEASM &"", sideeffect mayload attdialect, regdef:GR64, def %{{[0-9]+}}, mem:m, %stack.{{[0-9]+}}, 1, $noreg, 0, $noreg, {{.*}}, reguse tiedto:$0, %{{[0-9]+}}(tied-def 3), clobber,
+    %0:gr64 = COPY $rdi
+    %1:gr64 = COPY $rsi
+    %2:gr64 = COPY $rdx
+    %3:gr64 = COPY $rcx
+    %4:gr64 = COPY $r8
+    %5:gr64 = COPY $r9
+    %6:gr64 = MOV64rm %fixed-stack.0, 1, $noreg, 0, $noreg :: (load (s64) from %fixed-stack.0, align 16)
+    INLINEASM &"", sideeffect attdialect, regdef:GR64, def %6, reguse:GR64 foldable, %0, reguse:GR64, %1, reguse:GR64, %2, reguse:GR64, %3, reguse:GR64, %4, reguse:GR64, %5, reguse tiedto:$0, %6(tied-def 3), clobber, implicit-def dead early-clobber $rax, clobber, implicit-def dead early-clobber $rbx, clobber, implicit-def dead early-clobber $rbp, clobber, implicit-def dead early-clobber $r10, clobber, implicit-def dead early-clobber $r11, clobber, implicit-def dead early-clobber $r12, clobber, implicit-def dead early-clobber $r13, clobber, implicit-def dead early-clobber $r14, clobber, implicit-def dead early-clobber $r15
+    $rax = COPY %6
+    RET 0, $rax
+...
+---
+# Two read-write "+rm" operands: folding one tied pair, which reads and writes
+# the stack slot, keeps the other pair tied.
+name: two_tied_rm
+tracksRegLiveness: true
+fixedStack:
+  - { id: 0, size: 8, alignment: 16, isImmutable: true }
+body: |
+  bb.0:
+    liveins: $rdi, $rsi, $rdx, $rcx, $r8, $r9
+
+    ; CHECK-LABEL: name: two_tied_rm
+    ; CHECK: INLINEASM &"", sideeffect mayload maystore attdialect, regdef:GR64 foldable, def %{{[0-9]+}}, mem:m, %stack.[[SLOT:[0-9]+]], 1, $noreg, 0, $noreg, {{.*}}, reguse tiedto:$0, %{{[0-9]+}}(tied-def 3), mem:m, %stack.[[SLOT]], 1, $noreg, 0, $noreg, clobber,
+    %0:gr64 = COPY $rdi
+    %1:gr64 = COPY $rsi
+    %2:gr64 = COPY $rdx
+    %3:gr64 = COPY $rcx
+    %4:gr64 = COPY $r8
+    %5:gr64 = COPY $r9
+    %6:gr64 = MOV64rm %fixed-stack.0, 1, $noreg, 0, $noreg :: (load (s64) from %fixed-stack.0, align 16)
+    INLINEASM &"", sideeffect attdialect, regdef:GR64 foldable, def %0, regdef:GR64 foldable, def %1, reguse:GR64, %2, reguse:GR64, %3, reguse:GR64, %4, reguse:GR64, %5, reguse:GR64, %6, reguse tiedto:$0, %0(tied-def 3), reguse tiedto:$1, %1(tied-def 5), clobber, implicit-def dead early-clobber $rax, clobber, implicit-def dead early-clobber $rbx, clobber, implicit-def dead early-clobber $rbp, clobber, implicit-def dead early-clobber $r10, clobber, implicit-def dead early-clobber $r11, clobber, implicit-def dead early-clobber $r12, clobber, implicit-def dead early-clobber $r13, clobber, implicit-def dead early-clobber $r14, clobber, implicit-def dead early-clobber $r15
+    $rax = COPY %0
+    $rdx = COPY %1
+    RET 0, $rax, $rdx
+...

>From e2a0915d8155ea472950ec59ff8b447967254a93 Mon Sep 17 00:00:00 2001
From: Bill Wendling <isanbard at gmail.com>
Date: Wed, 7 Oct 2026 23:34:38 -0700
Subject: [PATCH 2/4] Apply suggestion from @nickdesaulniers

Co-authored-by: Nick Desaulniers <ndesaulniers at google.com>
---
 llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir b/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir
index 3d7842964304b3..6d352d9bbe5d33 100644
--- a/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir
+++ b/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir
@@ -9,7 +9,7 @@
 # operands tied.
 
 ---
-# One value passed to two "rm" operands: both operands fold to its stack slot.
+# One value passed to two "rm" operands: both operands fold to the same stack slot.
 name: rm_same_value_twice
 tracksRegLiveness: true
 fixedStack:

>From 46847372ac8ae155e544f111c7c26d43ecc25572 Mon Sep 17 00:00:00 2001
From: Bill Wendling <isanbard at gmail.com>
Date: Wed, 7 Oct 2026 23:41:24 -0700
Subject: [PATCH 3/4] [TargetInstrInfo] Add C inline asm statements to
 greedy-inline-asm-fold.mir

Show the C source behind each test so it's clear which MIR operand
corresponds to which asm constraint. Also reword the first test's
comment per review.

Co-Authored-By: Claude Opus 5.5 <noreply at anthropic.com>
---
 .../CodeGen/X86/greedy-inline-asm-fold.mir    | 30 ++++++++++++++++++-
 1 file changed, 29 insertions(+), 1 deletion(-)

diff --git a/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir b/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir
index 6d352d9bbe5d33..f7e1c695ebfc48 100644
--- a/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir
+++ b/llvm/test/CodeGen/X86/greedy-inline-asm-fold.mir
@@ -7,9 +7,21 @@
 # value is spilled. Folding replaces an operand with the five operands of a
 # memory reference, moving every operand after it, which must keep any tied
 # operands tied.
+#
+# Each test is derived from the C function and inline asm statement in its
+# comment, where CLOBBERS is:
+#
+#   "rax", "rbx", "rbp", "r10", "r11", "r12", "r13", "r14", "r15"
 
 ---
-# One value passed to two "rm" operands: both operands fold to the same stack slot.
+# One value passed to two "rm" operands: both operands fold to the same stack
+# slot.
+#
+#   void rm_same_value_twice(long a, long b, long c, long d, long e, long f,
+#                            long g) {
+#     asm volatile("" : : "rm"(a), "rm"(a), "r"(b), "r"(c), "r"(d), "r"(e),
+#                         "r"(f), "r"(g) : CLOBBERS);
+#   }
 name: rm_same_value_twice
 tracksRegLiveness: true
 fixedStack:
@@ -32,6 +44,13 @@ body: |
 ...
 ---
 # Folding the "rm" input moves the input tied to the "=r" output after it.
+#
+#   long rm_before_tied(long a, long b, long c, long d, long e, long f,
+#                       long g) {
+#     asm volatile("" : "+r"(g) : "rm"(a), "r"(b), "r"(c), "r"(d), "r"(e),
+#                                 "r"(f) : CLOBBERS);
+#     return g;
+#   }
 name: rm_before_tied
 tracksRegLiveness: true
 fixedStack:
@@ -56,6 +75,15 @@ body: |
 ---
 # Two read-write "+rm" operands: folding one tied pair, which reads and writes
 # the stack slot, keeps the other pair tied.
+#
+#   struct pair { long a, b; };
+#
+#   struct pair two_tied_rm(long a, long b, long c, long d, long e, long f,
+#                           long g) {
+#     asm volatile("" : "+rm"(a), "+rm"(b) : "r"(c), "r"(d), "r"(e), "r"(f),
+#                                            "r"(g) : CLOBBERS);
+#     return (struct pair){a, b};
+#   }
 name: two_tied_rm
 tracksRegLiveness: true
 fixedStack:

>From d1ef2e8b72832453cc95b8c9223cff5477766143 Mon Sep 17 00:00:00 2001
From: Bill Wendling <isanbard at gmail.com>
Date: Thu, 8 Oct 2026 00:18:59 -0700
Subject: [PATCH 4/4] [CodeGen] Add a sanity check that there are operands to
 fold

---
 llvm/lib/CodeGen/TargetInstrInfo.cpp | 4 ++++
 1 file changed, 4 insertions(+)

diff --git a/llvm/lib/CodeGen/TargetInstrInfo.cpp b/llvm/lib/CodeGen/TargetInstrInfo.cpp
index eeb6885497e6ce..f50754ccae5f86 100644
--- a/llvm/lib/CodeGen/TargetInstrInfo.cpp
+++ b/llvm/lib/CodeGen/TargetInstrInfo.cpp
@@ -722,6 +722,10 @@ static MachineInstr *foldInlineAsmMemOperand(MachineInstr &MI,
                                              const TargetInstrInfo &TII) {
   assert(MI.isInlineAsm() && "wrong opcode");
 
+  // Sanity check that there are operands to fold.
+  if (Ops.empty())
+    return nullptr;
+
   // Every operand in Ops holds the same register (e.g. one value passed to two
   // "rm" operands), so they all fold to the same stack slot. The asm reads
   // the slot through a use, and through a def tied to a use, which need not



More information about the llvm-branch-commits mailing list