[llvm] [AMDGPU] Avoid quadratic erase in loop patterns in SGPR/VALU legalization (PR #211740)
Arseniy Obolenskiy via llvm-commits
llvm-commits at lists.llvm.org
Thu Jul 23 23:57:46 PDT 2026
https://github.com/aobolensk created https://github.com/llvm/llvm-project/pull/211740
None
>From 1e80be0c453b0ccd8fab53938dd46b14cfaceb62 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Fri, 24 Jul 2026 08:50:28 +0200
Subject: [PATCH] [AMDGPU] Avoid quadratic erase in loop patterns in SGPR/VALU
legalization
---
llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp | 13 +++++++------
llvm/lib/Target/AMDGPU/SIInstrInfo.cpp | 3 ++-
llvm/lib/Target/AMDGPU/SIInstrInfo.h | 18 ++++++++++--------
.../Target/AMDGPU/SIOptimizeVGPRLiveRange.cpp | 15 ++++++---------
4 files changed, 25 insertions(+), 24 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
index 8230c84260067..ac6f7b50de75a 100644
--- a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
@@ -100,6 +100,8 @@ class V2SCopyInfo {
unsigned NumReadfirstlanes = 0;
// Current score state. To speedup selection V2SCopyInfos for processing
bool NeedToBeConvertedToVALU = false;
+ // Marks entries lowered to VALU for bulk removal from V2SCopies.
+ bool Erased = false;
// Unique ID. Used as a key for mapping to keep permanent order.
unsigned ID;
@@ -1085,12 +1087,12 @@ void SIFixSGPRCopies::lowerVGPR2SGPRCopies(MachineFunction &MF) {
while (!LoweringWorklist.empty()) {
unsigned CurID = LoweringWorklist.pop_back_val();
auto *CurInfoIt = V2SCopies.find(CurID);
- if (CurInfoIt != V2SCopies.end()) {
- const V2SCopyInfo &C = CurInfoIt->second;
+ if (CurInfoIt != V2SCopies.end() && !CurInfoIt->second.Erased) {
+ V2SCopyInfo &C = CurInfoIt->second;
LLVM_DEBUG(dbgs() << "Processing ...\n"; C.dump());
for (auto S : C.Siblings) {
auto *SibInfoIt = V2SCopies.find(S);
- if (SibInfoIt != V2SCopies.end()) {
+ if (SibInfoIt != V2SCopies.end() && !SibInfoIt->second.Erased) {
V2SCopyInfo &SI = SibInfoIt->second;
LLVM_DEBUG(dbgs() << "Sibling:\n"; SI.dump());
if (!SI.NeedToBeConvertedToVALU) {
@@ -1104,11 +1106,10 @@ void SIFixSGPRCopies::lowerVGPR2SGPRCopies(MachineFunction &MF) {
LLVM_DEBUG(dbgs() << "V2S copy " << *C.Copy
<< " is being turned to VALU\n");
Copies.insert(C.Copy);
- // TODO: MapVector::erase is inefficient. Do bulk removal with remove_if
- // instead.
- V2SCopies.erase(C.ID);
+ C.Erased = true;
}
}
+ V2SCopies.remove_if([](const auto &P) { return P.second.Erased; });
TII->moveToVALU(Copies, MDT);
Copies.clear();
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index 28914d598a72d..633a815447d9a 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -7852,7 +7852,8 @@ SIInstrInfo::legalizeOperands(MachineInstr &MI,
}
void SIInstrWorklist::insert(MachineInstr *MI) {
- InstrList.insert(MI);
+ if (InSet.insert(MI).second)
+ InstrList.push_back(MI);
// Add MBUF instructiosn to deferred list.
int RsrcIdx =
AMDGPU::getNamedOperandIdx(MI->getOpcode(), AMDGPU::OpName::srsrc);
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.h b/llvm/lib/Target/AMDGPU/SIInstrInfo.h
index 8e15b7b45b609..e57e80b88c2c6 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.h
@@ -18,6 +18,7 @@
#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
#include "SIRegisterInfo.h"
#include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/ADT/DenseSet.h"
#include "llvm/ADT/SetVector.h"
#include "llvm/CodeGen/TargetInstrInfo.h"
#include "llvm/CodeGen/TargetSchedule.h"
@@ -70,20 +71,19 @@ struct SIInstrWorklist {
void insert(MachineInstr *MI);
- MachineInstr *top() const {
- const auto *iter = InstrList.begin();
- return *iter;
- }
+ MachineInstr *top() const { return InstrList[Front]; }
void erase_top() {
- const auto *iter = InstrList.begin();
- InstrList.erase(iter);
+ InSet.erase(InstrList[Front]);
+ ++Front;
}
- bool empty() const { return InstrList.empty(); }
+ bool empty() const { return Front == InstrList.size(); }
void clear() {
InstrList.clear();
+ Front = 0;
+ InSet.clear();
DeferredList.clear();
}
@@ -93,7 +93,9 @@ struct SIInstrWorklist {
private:
/// InstrList contains the MachineInstrs.
- SetVector<MachineInstr *> InstrList;
+ SmallVector<MachineInstr *> InstrList;
+ DenseSet<MachineInstr *> InSet;
+ unsigned Front = 0;
/// Deferred instructions are specific MachineInstr
/// that will be added by insert method.
SetVector<MachineInstr *> DeferredList;
diff --git a/llvm/lib/Target/AMDGPU/SIOptimizeVGPRLiveRange.cpp b/llvm/lib/Target/AMDGPU/SIOptimizeVGPRLiveRange.cpp
index b6a3bc7380057..de3fba95254be 100644
--- a/llvm/lib/Target/AMDGPU/SIOptimizeVGPRLiveRange.cpp
+++ b/llvm/lib/Target/AMDGPU/SIOptimizeVGPRLiveRange.cpp
@@ -494,15 +494,12 @@ void SIOptimizeVGPRLiveRange::updateLiveRangeInElseRegion(
}
// Transfer the possible Kills in ElseBlocks from Reg to NewReg
- auto I = OldVarInfo.Kills.begin();
- while (I != OldVarInfo.Kills.end()) {
- if (ElseBlocks.contains((*I)->getParent())) {
- NewVarInfo.Kills.push_back(*I);
- I = OldVarInfo.Kills.erase(I);
- } else {
- ++I;
- }
- }
+ llvm::erase_if(OldVarInfo.Kills, [&](MachineInstr *MI) {
+ if (!ElseBlocks.contains(MI->getParent()))
+ return false;
+ NewVarInfo.Kills.push_back(MI);
+ return true;
+ });
}
void SIOptimizeVGPRLiveRange::optimizeLiveRange(
More information about the llvm-commits
mailing list