[llvm] [SelectionDAG] Don't unfold a load in the pre-RA schedulers if the unfolded load already exists (PR #226652)
Akash Manna via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 27 04:38:22 PDT 2026
https://github.com/akash-manna-sky updated https://github.com/llvm/llvm-project/pull/226652
>From c6d0681026e1b8929519df5825f4642f50cf1688 Mon Sep 17 00:00:00 2001
From: Akash Manna <akash.manna.mymail at gmail.com>
Date: Sat, 26 Sep 2026 12:41:04 +0530
Subject: [PATCH 1/2] [X86] Add test for unfolding a load onto an existing load
in the pre-RA scheduler (NFC)
Pre-commit for #204079.
---
.../CodeGen/X86/sched-unfold-existing-load.ll | 53 +++++++++++++++++++
1 file changed, 53 insertions(+)
create mode 100644 llvm/test/CodeGen/X86/sched-unfold-existing-load.ll
diff --git a/llvm/test/CodeGen/X86/sched-unfold-existing-load.ll b/llvm/test/CodeGen/X86/sched-unfold-existing-load.ll
new file mode 100644
index 0000000000000..d9b34aeed2ff7
--- /dev/null
+++ b/llvm/test/CodeGen/X86/sched-unfold-existing-load.ll
@@ -0,0 +1,53 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -O1 < %s | FileCheck %s --check-prefix=LIST
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -O1 -pre-RA-sched=fast < %s | FileCheck %s --check-prefix=FAST
+
+ at val = dso_local global i32 0, align 4
+ at other = dso_local global i32 0, align 4
+
+; The compare of %y selects to a CMP32mi with the load folded in. The pre-RA
+; scheduler tries to unfold it to break the EFLAGS interference between the two
+; setcc stores, but the unfolded load already exists: it is the load of %x
+; (!nontemporal only keeps the two loads from being the same node).
+define void @existing_load(i64 %a, ptr %p, ptr %q) {
+; LIST-LABEL: existing_load:
+; LIST: # %bb.0:
+; LIST-NEXT: movl $1, other(%rip)
+; LIST-NEXT: movl val(%rip), %eax
+; LIST-NEXT: testq %rdi, %rdi
+; LIST-NEXT: setne (%rdx)
+; LIST-NEXT: testl %eax, %eax
+; LIST-NEXT: movl (%rsi), %ecx
+; LIST-NEXT: movl %ecx, val(%rip)
+; LIST-NEXT: setne (%rdx)
+; LIST-NEXT: movl %eax, (%rsi)
+; LIST-NEXT: retq
+;
+; FAST-LABEL: existing_load:
+; FAST: # %bb.0:
+; FAST-NEXT: movl $1, other(%rip)
+; FAST-NEXT: movl val(%rip), %eax
+; FAST-NEXT: testq %rdi, %rdi
+; FAST-NEXT: setne (%rdx)
+; FAST-NEXT: testl %eax, %eax
+; FAST-NEXT: movl (%rsi), %ecx
+; FAST-NEXT: movl %ecx, val(%rip)
+; FAST-NEXT: setne (%rdx)
+; FAST-NEXT: movl %eax, (%rsi)
+; FAST-NEXT: retq
+ store i32 1, ptr @other
+ %x = load i32, ptr @val
+ %y = load i32, ptr @val, !nontemporal !0
+ %c1 = icmp ne i64 %a, 0
+ %z1 = zext i1 %c1 to i8
+ store i8 %z1, ptr %q
+ %v = load i32, ptr %p
+ store i32 %v, ptr @val
+ %c2 = icmp ne i32 %y, 0
+ %z2 = zext i1 %c2 to i8
+ store i8 %z2, ptr %q
+ store i32 %x, ptr %p
+ ret void
+}
+
+!0 = !{i32 1}
>From 2e5812ec579e922231d94af34d674456cb043b77 Mon Sep 17 00:00:00 2001
From: Akash Manna <akash.manna.mymail at gmail.com>
Date: Sat, 26 Sep 2026 12:45:51 +0530
Subject: [PATCH 2/2] [SelectionDAG] Don't unfold a load in the pre-RA
schedulers if the unfolded load already exists
TryUnfoldSU and its ScheduleDAGFast twin redirected the chain uses of the
folded node to a load that already had users. That can CSE one of those
users into an existing node and delete it while an SUnit still refers to
it, which then crashed in InstrEmitter with a <<Deleted Node!>>. Give up
on unfolding in that case and let the caller insert flag copies instead.
Fixes #204079
---
llvm/docs/ReleaseNotes.md | 5 +
.../CodeGen/SelectionDAG/ScheduleDAGFast.cpp | 47 +++----
.../SelectionDAG/ScheduleDAGRRList.cpp | 86 +++++--------
llvm/test/CodeGen/X86/pr204079.ll | 119 ++++++++++++++++++
llvm/test/CodeGen/X86/pr37916.ll | 7 +-
.../CodeGen/X86/sched-unfold-existing-load.ll | 18 +--
6 files changed, 188 insertions(+), 94 deletions(-)
create mode 100644 llvm/test/CodeGen/X86/pr204079.ll
diff --git a/llvm/docs/ReleaseNotes.md b/llvm/docs/ReleaseNotes.md
index 985117034e2be..2a2d36ae889f3 100644
--- a/llvm/docs/ReleaseNotes.md
+++ b/llvm/docs/ReleaseNotes.md
@@ -288,6 +288,11 @@ Makes programs 10x faster by doing Special New Thing.
compiling a function containing a static alloca of `(size_t)-1` bytes, whose
size collided with the sentinel value MachineFrameInfo used to mark dead
stack objects.
+* Fixed a crash
+ ([#204079](https://github.com/llvm/llvm-project/issues/204079)) in the
+ SelectionDAG pre-RA schedulers when they tried to unfold a load-folded
+ instruction to break a physical register dependency and the unfolded load
+ already existed in the DAG.
### Changes to the Metadata Info
diff --git a/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGFast.cpp b/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGFast.cpp
index 0f67cee757407..3b457c4d3d881 100644
--- a/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGFast.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGFast.cpp
@@ -225,20 +225,26 @@ SUnit *ScheduleDAGFast::CopyAndMoveSuccessors(SUnit *SU) {
if (!TII->unfoldMemoryOperand(*DAG, N, NewNodes))
return nullptr;
- LLVM_DEBUG(dbgs() << "Unfolding SU # " << SU->NodeNum << "\n");
assert(NewNodes.size() == 2 && "Expected a load folding node!");
N = NewNodes[1];
SDNode *LoadNode = NewNodes[0];
unsigned NumVals = N->getNumValues();
unsigned OldNumVals = SU->getNode()->getNumValues();
+
+ // Give up if LoadNode already exists, see ScheduleDAGRRList::TryUnfoldSU.
+ if (LoadNode->getNodeId() != -1)
+ return nullptr;
+ assert(N->getNodeId() == -1 && "Node using a new load can't exist yet!");
+
+ LLVM_DEBUG(dbgs() << "Unfolding SU # " << SU->NodeNum << "\n");
+
for (unsigned i = 0; i != NumVals; ++i)
DAG->ReplaceAllUsesOfValueWith(SDValue(SU->getNode(), i), SDValue(N, i));
- DAG->ReplaceAllUsesOfValueWith(SDValue(SU->getNode(), OldNumVals-1),
+ DAG->ReplaceAllUsesOfValueWith(SDValue(SU->getNode(), OldNumVals - 1),
SDValue(LoadNode, 1));
SUnit *NewSU = newSUnit(N);
- assert(N->getNodeId() == -1 && "Node already inserted!");
N->setNodeId(NewSU->NodeNum);
const MCInstrDesc &MCID = TII->get(N->getMachineOpcode());
@@ -251,18 +257,8 @@ SUnit *ScheduleDAGFast::CopyAndMoveSuccessors(SUnit *SU) {
if (MCID.isCommutable())
NewSU->isCommutable = true;
- // LoadNode may already exist. This can happen when there is another
- // load from the same location and producing the same type of value
- // but it has different alignment or volatileness.
- bool isNewLoad = true;
- SUnit *LoadSU;
- if (LoadNode->getNodeId() != -1) {
- LoadSU = &SUnits[LoadNode->getNodeId()];
- isNewLoad = false;
- } else {
- LoadSU = newSUnit(LoadNode);
- LoadNode->setNodeId(LoadSU->NodeNum);
- }
+ SUnit *LoadSU = newSUnit(LoadNode);
+ LoadNode->setNodeId(LoadSU->NodeNum);
SDep ChainPred;
SmallVector<SDep, 4> ChainSuccs;
@@ -287,14 +283,11 @@ SUnit *ScheduleDAGFast::CopyAndMoveSuccessors(SUnit *SU) {
if (ChainPred.getSUnit()) {
RemovePred(SU, ChainPred);
- if (isNewLoad)
- AddPred(LoadSU, ChainPred);
+ AddPred(LoadSU, ChainPred);
}
for (const SDep &Pred : LoadPreds) {
RemovePred(SU, Pred);
- if (isNewLoad) {
- AddPred(LoadSU, Pred);
- }
+ AddPred(LoadSU, Pred);
}
for (const SDep &Pred : NodePreds) {
RemovePred(SU, Pred);
@@ -311,16 +304,12 @@ SUnit *ScheduleDAGFast::CopyAndMoveSuccessors(SUnit *SU) {
SUnit *SuccDep = D.getSUnit();
D.setSUnit(SU);
RemovePred(SuccDep, D);
- if (isNewLoad) {
- D.setSUnit(LoadSU);
- AddPred(SuccDep, D);
- }
- }
- if (isNewLoad) {
- SDep D(LoadSU, SDep::Barrier);
- D.setLatency(LoadSU->Latency);
- AddPred(NewSU, D);
+ D.setSUnit(LoadSU);
+ AddPred(SuccDep, D);
}
+ SDep D(LoadSU, SDep::Barrier);
+ D.setLatency(LoadSU->Latency);
+ AddPred(NewSU, D);
++NumUnfolds;
diff --git a/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp b/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp
index 520fe43a8cfc2..e3cad4e7ab2e8 100644
--- a/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp
@@ -986,54 +986,36 @@ SUnit *ScheduleDAGRRList::TryUnfoldSU(SUnit *SU) {
unsigned NumVals = N->getNumValues();
unsigned OldNumVals = SU->getNode()->getNumValues();
- // LoadNode may already exist. This can happen when there is another
- // load from the same location and producing the same type of value
- // but it has different alignment or volatileness.
- bool isNewLoad = true;
- SUnit *LoadSU;
- if (LoadNode->getNodeId() != -1) {
- LoadSU = &SUnits[LoadNode->getNodeId()];
- // If LoadSU has already been scheduled, we should clone it but
- // this would negate the benefit to unfolding so just return SU.
- if (LoadSU->isScheduled)
- return SU;
- isNewLoad = false;
- } else {
- LoadSU = CreateNewSUnit(LoadNode);
- LoadNode->setNodeId(LoadSU->NodeNum);
+ // LoadNode may already exist (e.g. a load from the same location that was
+ // not folded). Its chain result then already has users, and redirecting the
+ // chain uses of SU's node to it below can CSE one of them into an existing
+ // node and delete it while an SUnit still refers to it. Cloning SU is not
+ // safe for a memory access either, so give up and let the caller insert
+ // physical register copies instead.
+ if (LoadNode->getNodeId() != -1)
+ return nullptr;
+ assert(N->getNodeId() == -1 && "Node using a new load can't exist yet!");
- InitNumRegDefsLeft(LoadSU);
- computeLatency(LoadSU);
- }
+ SUnit *LoadSU = CreateNewSUnit(LoadNode);
+ LoadNode->setNodeId(LoadSU->NodeNum);
+ InitNumRegDefsLeft(LoadSU);
+ computeLatency(LoadSU);
- bool isNewN = true;
- SUnit *NewSU;
- // This can only happen when isNewLoad is false.
- if (N->getNodeId() != -1) {
- NewSU = &SUnits[N->getNodeId()];
- // If NewSU has already been scheduled, we need to clone it, but this
- // negates the benefit to unfolding so just return SU.
- if (NewSU->isScheduled) {
- return SU;
- }
- isNewN = false;
- } else {
- NewSU = CreateNewSUnit(N);
- N->setNodeId(NewSU->NodeNum);
+ SUnit *NewSU = CreateNewSUnit(N);
+ N->setNodeId(NewSU->NodeNum);
- const MCInstrDesc &MCID = TII->get(N->getMachineOpcode());
- for (unsigned i = 0; i != MCID.getNumOperands(); ++i) {
- if (MCID.getOperandConstraint(i, MCOI::TIED_TO) != -1) {
- NewSU->isTwoAddress = true;
- break;
- }
+ const MCInstrDesc &MCID = TII->get(N->getMachineOpcode());
+ for (unsigned i = 0; i != MCID.getNumOperands(); ++i) {
+ if (MCID.getOperandConstraint(i, MCOI::TIED_TO) != -1) {
+ NewSU->isTwoAddress = true;
+ break;
}
- if (MCID.isCommutable())
- NewSU->isCommutable = true;
-
- InitNumRegDefsLeft(NewSU);
- computeLatency(NewSU);
}
+ if (MCID.isCommutable())
+ NewSU->isCommutable = true;
+
+ InitNumRegDefsLeft(NewSU);
+ computeLatency(NewSU);
LLVM_DEBUG(dbgs() << "Unfolding SU #" << SU->NodeNum << "\n");
@@ -1067,13 +1049,11 @@ SUnit *ScheduleDAGRRList::TryUnfoldSU(SUnit *SU) {
// Now assign edges to the newly-created nodes.
for (const SDep &Pred : ChainPreds) {
RemovePred(SU, Pred);
- if (isNewLoad)
- AddPredQueued(LoadSU, Pred);
+ AddPredQueued(LoadSU, Pred);
}
for (const SDep &Pred : LoadPreds) {
RemovePred(SU, Pred);
- if (isNewLoad)
- AddPredQueued(LoadSU, Pred);
+ AddPredQueued(LoadSU, Pred);
}
for (const SDep &Pred : NodePreds) {
RemovePred(SU, Pred);
@@ -1094,10 +1074,8 @@ SUnit *ScheduleDAGRRList::TryUnfoldSU(SUnit *SU) {
SUnit *SuccDep = D.getSUnit();
D.setSUnit(SU);
RemovePred(SuccDep, D);
- if (isNewLoad) {
- D.setSUnit(LoadSU);
- AddPredQueued(SuccDep, D);
- }
+ D.setSUnit(LoadSU);
+ AddPredQueued(SuccDep, D);
}
// Add a data dependency to reflect that NewSU reads the value defined
@@ -1106,10 +1084,8 @@ SUnit *ScheduleDAGRRList::TryUnfoldSU(SUnit *SU) {
D.setLatency(LoadSU->Latency);
AddPredQueued(NewSU, D);
- if (isNewLoad)
- AvailableQueue->addNode(LoadSU);
- if (isNewN)
- AvailableQueue->addNode(NewSU);
+ AvailableQueue->addNode(LoadSU);
+ AvailableQueue->addNode(NewSU);
++NumUnfolds;
diff --git a/llvm/test/CodeGen/X86/pr204079.ll b/llvm/test/CodeGen/X86/pr204079.ll
new file mode 100644
index 0000000000000..fbb9806024a66
--- /dev/null
+++ b/llvm/test/CodeGen/X86/pr204079.ll
@@ -0,0 +1,119 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -O1 < %s | FileCheck %s --check-prefix=LIST
+; RUN: llc -mtriple=x86_64-unknown-linux-gnu -O1 -pre-RA-sched=fast < %s | FileCheck %s --check-prefix=FAST
+
+; The two loads of @val select to a standalone MOV32rm and a CMP32mi with the
+; load folded in (the byte loads are combined into one i32 load that only
+; feeds the compare). To break the EFLAGS interference between the two setcc
+; stores to @cond, the pre-RA scheduler tried to unfold the CMP32mi, but the
+; unfolded load was CSE'd with the existing MOV32rm and redirecting the chain
+; uses of the CMP32mi to it deleted a TokenFactor that still had an SUnit.
+
+ at ptr = dso_local global ptr null, align 8
+ at s0 = dso_local global i16 0, align 2
+ at s1 = dso_local global i16 0, align 2
+ at s2 = dso_local global i16 0, align 2
+ at s3 = dso_local global i16 0, align 2
+ at s4 = dso_local global i16 0, align 2
+ at c0 = dso_local global i8 0, align 1
+ at c1 = dso_local global i8 0, align 1
+ at val = dso_local global i32 0, align 4
+ at cond = dso_local global i8 0, align 1
+
+define void @f12(i64 %a1, ptr %p) {
+; LIST-LABEL: f12:
+; LIST: # %bb.0: # %entry
+; LIST-NEXT: movzbl (%rsi), %eax
+; LIST-NEXT: movw %ax, s0(%rip)
+; LIST-NEXT: movzwl s2(%rip), %eax
+; LIST-NEXT: movzwl s3(%rip), %ecx
+; LIST-NEXT: addl %eax, %ecx
+; LIST-NEXT: movw %cx, s1(%rip)
+; LIST-NEXT: movb $0, c1(%rip)
+; LIST-NEXT: movzwl s4(%rip), %eax
+; LIST-NEXT: movw %ax, s3(%rip)
+; LIST-NEXT: movw $0, s2(%rip)
+; LIST-NEXT: movb $0, c0(%rip)
+; LIST-NEXT: movw $0, s4(%rip)
+; LIST-NEXT: movb $0, -{{[0-9]+}}(%rsp)
+; LIST-NEXT: movl val(%rip), %eax
+; LIST-NEXT: cmpl $0, val(%rip)
+; LIST-NEXT: setne %cl
+; LIST-NEXT: testq %rdi, %rdi
+; LIST-NEXT: setne cond(%rip)
+; LIST-NEXT: movl -{{[0-9]+}}(%rsp), %edx
+; LIST-NEXT: movl %edx, val(%rip)
+; LIST-NEXT: movb %cl, cond(%rip)
+; LIST-NEXT: movl %eax, -{{[0-9]+}}(%rsp)
+; LIST-NEXT: retq
+;
+; FAST-LABEL: f12:
+; FAST: # %bb.0: # %entry
+; FAST-NEXT: movzbl (%rsi), %eax
+; FAST-NEXT: movb $0, -{{[0-9]+}}(%rsp)
+; FAST-NEXT: movzwl s4(%rip), %ecx
+; FAST-NEXT: movw $0, s4(%rip)
+; FAST-NEXT: movzwl s3(%rip), %edx
+; FAST-NEXT: movzwl s2(%rip), %esi
+; FAST-NEXT: addl %edx, %esi
+; FAST-NEXT: movw %si, s1(%rip)
+; FAST-NEXT: movw %cx, s3(%rip)
+; FAST-NEXT: movw $0, s2(%rip)
+; FAST-NEXT: movb $0, c1(%rip)
+; FAST-NEXT: movw %ax, s0(%rip)
+; FAST-NEXT: movb $0, c0(%rip)
+; FAST-NEXT: movl val(%rip), %eax
+; FAST-NEXT: cmpl $0, val(%rip)
+; FAST-NEXT: setne %cl
+; FAST-NEXT: testq %rdi, %rdi
+; FAST-NEXT: setne cond(%rip)
+; FAST-NEXT: movl -{{[0-9]+}}(%rsp), %edx
+; FAST-NEXT: movl %eax, -{{[0-9]+}}(%rsp)
+; FAST-NEXT: movb %cl, cond(%rip)
+; FAST-NEXT: movl %edx, val(%rip)
+; FAST-NEXT: retq
+entry:
+ %local = alloca i32, align 4
+ %local2 = alloca i8, align 1
+ %0 = load i8, ptr %p, align 1
+ %1 = zext nneg i8 %0 to i16
+ store i16 %1, ptr @s0, align 2
+ %2 = load i8, ptr @c0, align 1
+ store i8 %2, ptr @c1, align 1
+ %3 = load i16, ptr @s2, align 2
+ %4 = load i16, ptr @s3, align 2
+ %5 = sext i16 %3 to i17
+ %6 = sext i16 %4 to i17
+ %7 = add nsw i17 %6, %5
+ %8 = trunc i17 %7 to i16
+ %9 = lshr i17 %7, 16
+ %10 = trunc nuw nsw i17 %9 to i8
+ store i8 %10, ptr @c1, align 1
+ store i16 %8, ptr @s1, align 2
+ store i8 0, ptr @c1, align 1
+ %11 = load ptr, ptr @ptr, align 8
+ %12 = load i16, ptr @s4, align 2
+ store i16 %12, ptr @s3, align 2
+ store i16 0, ptr @s2, align 2
+ store i8 0, ptr @c0, align 1
+ store i16 0, ptr @s4, align 2
+ store i16 0, ptr @s2, align 2
+ store i8 0, ptr %local2, align 1
+ %13 = load i8, ptr @val, align 4
+ %14 = load i24, ptr getelementptr inbounds nuw (i8, ptr @val, i64 1), align 1
+ %15 = load i32, ptr @val, align 4
+ %tobool = icmp ne i64 %a1, 0
+ %16 = zext i1 %tobool to i8
+ store i8 %16, ptr @cond, align 1
+ %17 = load i32, ptr %local, align 4
+ store i32 %17, ptr @val, align 4
+ %18 = zext i24 %14 to i32
+ %19 = shl nuw i32 %18, 8
+ %20 = zext i8 %13 to i32
+ %21 = or disjoint i32 %19, %20
+ %tobool12 = icmp ne i32 %21, 0
+ %22 = zext i1 %tobool12 to i8
+ store i8 %22, ptr @cond, align 1
+ store i32 %15, ptr %local, align 4
+ ret void
+}
diff --git a/llvm/test/CodeGen/X86/pr37916.ll b/llvm/test/CodeGen/X86/pr37916.ll
index e6639a11ca5ea..34b616991658b 100644
--- a/llvm/test/CodeGen/X86/pr37916.ll
+++ b/llvm/test/CodeGen/X86/pr37916.ll
@@ -10,12 +10,15 @@ define void @fn1() local_unnamed_addr {
; CHECK-NEXT: .p2align 4
; CHECK-NEXT: .LBB0_1: # %if.end
; CHECK-NEXT: # =>This Inner Loop Header: Depth=1
-; CHECK-NEXT: movl a+4, %eax
-; CHECK-NEXT: orl a, %eax
+; CHECK-NEXT: movl a, %eax
+; CHECK-NEXT: movl a+4, %ecx
+; CHECK-NEXT: orl %eax, %ecx
+; CHECK-NEXT: movl a+4, %ecx
; CHECK-NEXT: movl $a, f
; CHECK-NEXT: je .LBB0_3
; CHECK-NEXT: # %bb.2: # %if.end
; CHECK-NEXT: # in Loop: Header=BB0_1 Depth=1
+; CHECK-NEXT: orl %ecx, %eax
; CHECK-NEXT: jne .LBB0_1
; CHECK-NEXT: .LBB0_3: # %cond.false
entry:
diff --git a/llvm/test/CodeGen/X86/sched-unfold-existing-load.ll b/llvm/test/CodeGen/X86/sched-unfold-existing-load.ll
index d9b34aeed2ff7..b1dda2f7359cf 100644
--- a/llvm/test/CodeGen/X86/sched-unfold-existing-load.ll
+++ b/llvm/test/CodeGen/X86/sched-unfold-existing-load.ll
@@ -14,25 +14,27 @@ define void @existing_load(i64 %a, ptr %p, ptr %q) {
; LIST: # %bb.0:
; LIST-NEXT: movl $1, other(%rip)
; LIST-NEXT: movl val(%rip), %eax
+; LIST-NEXT: cmpl $0, val(%rip)
+; LIST-NEXT: setne %cl
; LIST-NEXT: testq %rdi, %rdi
; LIST-NEXT: setne (%rdx)
-; LIST-NEXT: testl %eax, %eax
-; LIST-NEXT: movl (%rsi), %ecx
-; LIST-NEXT: movl %ecx, val(%rip)
-; LIST-NEXT: setne (%rdx)
+; LIST-NEXT: movl (%rsi), %edi
+; LIST-NEXT: movl %edi, val(%rip)
+; LIST-NEXT: movb %cl, (%rdx)
; LIST-NEXT: movl %eax, (%rsi)
; LIST-NEXT: retq
;
; FAST-LABEL: existing_load:
; FAST: # %bb.0:
+; FAST-NEXT: cmpl $0, val(%rip)
; FAST-NEXT: movl $1, other(%rip)
; FAST-NEXT: movl val(%rip), %eax
+; FAST-NEXT: setne %cl
; FAST-NEXT: testq %rdi, %rdi
; FAST-NEXT: setne (%rdx)
-; FAST-NEXT: testl %eax, %eax
-; FAST-NEXT: movl (%rsi), %ecx
-; FAST-NEXT: movl %ecx, val(%rip)
-; FAST-NEXT: setne (%rdx)
+; FAST-NEXT: movl (%rsi), %edi
+; FAST-NEXT: movl %edi, val(%rip)
+; FAST-NEXT: movb %cl, (%rdx)
; FAST-NEXT: movl %eax, (%rsi)
; FAST-NEXT: retq
store i32 1, ptr @other
More information about the llvm-commits
mailing list