[llvm] AMDGPU: Use LiveIntervals instead of LiveVariables in SIOptimizeVGPRLiveRange (PR #227252)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 29 03:09:52 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Matt Arsenault (arsenm)
<details>
<summary>Changes</summary>
Drop the LiveVariables dependency and the hand-written VarInfo
maintenance; the pass already knew how to recompute the affected
intervals with LiveIntervals, so make that the only path.
The legacy pass manager cannot schedule a pass requiring both
LiveIntervals and LiveVariables here, since LiveVariables (and its
UnreachableMachineBlockElim dependency) invalidates the LiveIntervals
just computed for it. In the AMDGPU pipeline, anchor the pass after
MachineLoopInfo instead of PHIElimination so LiveIntervals is computed
before PHIElimination, which is required anyway: the pass needs SSA and
introduces new PHIs. Preserving SlotIndexes keeps the transitive
last-user chain intact.
Since LiveVariables is no longer maintained past this point, PHIElimination
now splits critical edges using LiveIntervals, which accounts for most of
the test churn: LiveIntervalCalc drops kill flags and adds dead flags.
Co-Authored-By: Claude Opus 5 <noreply@<!-- -->anthropic.com>
---
Patch is 8.13 MiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/227252.diff
60 Files Affected:
- (modified) llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp (+2-10)
- (modified) llvm/lib/Target/AMDGPU/SIOptimizeVGPRLiveRange.cpp (+27-206)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll (+9-9)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll (+14-14)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll (+11-11)
- (modified) llvm/test/CodeGen/AMDGPU/a-v-flat-atomic-cmpxchg.ll (+64-64)
- (modified) llvm/test/CodeGen/AMDGPU/a-v-flat-atomicrmw.ll (+8-8)
- (modified) llvm/test/CodeGen/AMDGPU/absdiff.ll (+5-5)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll (+34901-34751)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.128bit.ll (+108-107)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.256bit.ll (+404-403)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.320bit.ll (+253-263)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.32bit.ll (+99-84)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.48bit.ll (+20-20)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll (+3648-3607)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.640bit.ll (+30-32)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.64bit.ll (+31-31)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.704bit.ll (+565-545)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.768bit.ll (+1758-1716)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.832bit.ll (+2600-2556)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.896bit.ll (+2958-2949)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.960bit.ll (+3144-3155)
- (modified) llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.96bit.ll (+88-88)
- (modified) llvm/test/CodeGen/AMDGPU/asyncmark-gfx12plus.ll (+12-12)
- (modified) llvm/test/CodeGen/AMDGPU/asyncmark-pregfx12.ll (+26-26)
- (modified) llvm/test/CodeGen/AMDGPU/atomicrmw-nand.ll (+6-6)
- (modified) llvm/test/CodeGen/AMDGPU/block-should-not-be-in-alive-blocks.mir (+15-15)
- (modified) llvm/test/CodeGen/AMDGPU/branch-folding-implicit-def-subreg.ll (+254-275)
- (modified) llvm/test/CodeGen/AMDGPU/branch-relaxation-gfx1250.ll (+14-14)
- (modified) llvm/test/CodeGen/AMDGPU/bypass-div.ll (+21-21)
- (modified) llvm/test/CodeGen/AMDGPU/div_i128.ll (+43-43)
- (modified) llvm/test/CodeGen/AMDGPU/div_v2i128.ll (+147-147)
- (modified) llvm/test/CodeGen/AMDGPU/flat_atomics_i64_noprivate.ll (+72-60)
- (modified) llvm/test/CodeGen/AMDGPU/flat_atomics_i64_system.ll (+168-148)
- (modified) llvm/test/CodeGen/AMDGPU/fptoi.i128.ll (+14-14)
- (modified) llvm/test/CodeGen/AMDGPU/fract-match.ll (+42-30)
- (modified) llvm/test/CodeGen/AMDGPU/frem.ll (+77-78)
- (modified) llvm/test/CodeGen/AMDGPU/global_atomics_scan_fmax.ll (+54-54)
- (modified) llvm/test/CodeGen/AMDGPU/global_atomics_scan_fmin.ll (+54-54)
- (modified) llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll (+2-2)
- (modified) llvm/test/CodeGen/AMDGPU/llc-pipeline.ll (+8-8)
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.init.whole.wave-w32.ll (+20-20)
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fadd.ll (+27-27)
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fsub.ll (+27-27)
- (modified) llvm/test/CodeGen/AMDGPU/local-atomicrmw-fadd.ll (+91-91)
- (modified) llvm/test/CodeGen/AMDGPU/local-atomicrmw-fmax.ll (+100-100)
- (modified) llvm/test/CodeGen/AMDGPU/local-atomicrmw-fmin.ll (+100-100)
- (modified) llvm/test/CodeGen/AMDGPU/local-atomicrmw-fsub.ll (+175-175)
- (modified) llvm/test/CodeGen/AMDGPU/lower-control-flow-live-variables-update.mir (+87)
- (removed) llvm/test/CodeGen/AMDGPU/lower-control-flow-live-variables-update.xfail.mir (-42)
- (modified) llvm/test/CodeGen/AMDGPU/mad_uint24.ll (+36-36)
- (modified) llvm/test/CodeGen/AMDGPU/rewrite-vgpr-mfma-to-agpr-spill-multi-store-codegen.ll (+337-438)
- (modified) llvm/test/CodeGen/AMDGPU/si-opt-vgpr-liverange-bug-deadlanes.mir (+8-10)
- (renamed) llvm/test/CodeGen/AMDGPU/si-opt-vgpr-liverange-undef-use.mir (+11-11)
- (modified) llvm/test/CodeGen/AMDGPU/si-optimize-vgpr-live-range-dbg-instr.mir (+23-22)
- (modified) llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll (+410-397)
- (modified) llvm/test/CodeGen/AMDGPU/srem64.ll (+24-24)
- (modified) llvm/test/CodeGen/AMDGPU/urem64.ll (+24-24)
- (modified) llvm/test/CodeGen/AMDGPU/vgpr-liverange-ir.ll (+126-126)
- (modified) llvm/test/CodeGen/AMDGPU/vni8-across-blocks.ll (+16-16)
``````````diff
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
index 72c028edbaef5..c52970188eea8 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
@@ -1856,12 +1856,8 @@ void GCNPassConfig::addOptimizedRegAlloc() {
if (EnableDCEInRA)
insertPass(&DetectDeadLanesID, &DeadMachineInstructionElimID);
- // FIXME: when an instruction has a Killed operand, and the instruction is
- // inside a bundle, seems only the BUNDLE instruction appears as the Kills of
- // the register in LiveVariables, this would trigger a failure in verifier,
- // we should fix it and enable the verifier.
if (OptVGPRLiveRange)
- insertPass(&LiveVariablesID, &SIOptimizeVGPRLiveRangeLegacyID);
+ insertPass(&MachineLoopInfoID, &SIOptimizeVGPRLiveRangeLegacyID);
// This must be run immediately after phi elimination and before
// TwoAddressInstructions, otherwise the processing of the tied operand of
@@ -2662,12 +2658,8 @@ Error AMDGPUCodeGenPassBuilder::addOptimizedRegAlloc(PassManagerWrapper &PMW) {
if (EnableDCEInRA)
insertPass<DetectDeadLanesPass>(DeadMachineInstructionElimPass());
- // FIXME: when an instruction has a Killed operand, and the instruction is
- // inside a bundle, seems only the BUNDLE instruction appears as the Kills of
- // the register in LiveVariables, this would trigger a failure in verifier,
- // we should fix it and enable the verifier.
if (OptVGPRLiveRange)
- insertPass<RequireAnalysisPass<LiveVariablesAnalysis, MachineFunction>>(
+ insertPass<RequireAnalysisPass<MachineLoopAnalysis, MachineFunction>>(
SIOptimizeVGPRLiveRangePass());
// This must be run immediately after phi elimination and before
diff --git a/llvm/lib/Target/AMDGPU/SIOptimizeVGPRLiveRange.cpp b/llvm/lib/Target/AMDGPU/SIOptimizeVGPRLiveRange.cpp
index c72b40e14ee0e..1120c5209e2fc 100644
--- a/llvm/lib/Target/AMDGPU/SIOptimizeVGPRLiveRange.cpp
+++ b/llvm/lib/Target/AMDGPU/SIOptimizeVGPRLiveRange.cpp
@@ -76,7 +76,6 @@
#include "GCNSubtarget.h"
#include "SIMachineFunctionInfo.h"
#include "llvm/CodeGen/LiveIntervals.h"
-#include "llvm/CodeGen/LiveVariables.h"
#include "llvm/CodeGen/MachineDominators.h"
#include "llvm/CodeGen/MachineLoopInfo.h"
#include "llvm/CodeGen/TargetRegisterInfo.h"
@@ -94,7 +93,6 @@ class SIOptimizeVGPRLiveRange {
const SIRegisterInfo *TRI = nullptr;
const SIInstrInfo *TII = nullptr;
LiveIntervals *LIS = nullptr;
- LiveVariables *LV = nullptr;
MachineDominatorTree *MDT = nullptr;
const MachineLoopInfo *Loops = nullptr;
MachineRegisterInfo *MRI = nullptr;
@@ -109,9 +107,9 @@ class SIOptimizeVGPRLiveRange {
bool isLiveIntoMBB(Register Reg, const MachineBasicBlock *MBB) const;
public:
- SIOptimizeVGPRLiveRange(LiveIntervals *LIS, LiveVariables *LV,
- MachineDominatorTree *MDT, MachineLoopInfo *Loops)
- : LIS(LIS), LV(LV), MDT(MDT), Loops(Loops) {}
+ SIOptimizeVGPRLiveRange(LiveIntervals *LIS, MachineDominatorTree *MDT,
+ MachineLoopInfo *Loops)
+ : LIS(LIS), MDT(MDT), Loops(Loops) {}
bool run(MachineFunction &MF);
MachineBasicBlock *getElseTarget(MachineBasicBlock *MBB) const;
@@ -132,17 +130,6 @@ class SIOptimizeVGPRLiveRange {
SmallSetVector<MachineBasicBlock *, 2> &Blocks,
SmallVectorImpl<MachineInstr *> &Instructions) const;
- void findNonPHIUsesInBlock(Register Reg, MachineBasicBlock *MBB,
- SmallVectorImpl<MachineInstr *> &Uses) const;
-
- void updateLiveRangeInThenRegion(Register Reg, MachineBasicBlock *If,
- MachineBasicBlock *Flow) const;
-
- void updateLiveRangeInElseRegion(
- Register Reg, Register NewReg, MachineBasicBlock *Flow,
- MachineBasicBlock *Endif,
- SmallSetVector<MachineBasicBlock *, 16> &ElseBlocks) const;
-
void
optimizeLiveRange(Register Reg, MachineBasicBlock *If,
MachineBasicBlock *Flow, MachineBasicBlock *Endif,
@@ -168,12 +155,11 @@ class SIOptimizeVGPRLiveRangeLegacy : public MachineFunctionPass {
void getAnalysisUsage(AnalysisUsage &AU) const override {
AU.setPreservesCFG();
- AU.addUsedIfAvailable<LiveIntervalsWrapperPass>();
+ AU.addRequired<LiveIntervalsWrapperPass>();
+ AU.addPreserved<SlotIndexesWrapperPass>();
AU.addPreserved<LiveIntervalsWrapperPass>();
- AU.addRequired<LiveVariablesWrapperPass>();
AU.addRequired<MachineDominatorTreeWrapperPass>();
AU.addRequired<MachineLoopInfoWrapperPass>();
- AU.addPreserved<LiveVariablesWrapperPass>();
MachineFunctionPass::getAnalysisUsage(AU);
}
@@ -201,18 +187,12 @@ SIOptimizeVGPRLiveRange::getElseTarget(MachineBasicBlock *MBB) const {
bool SIOptimizeVGPRLiveRange::isLiveThrough(
Register Reg, const MachineBasicBlock *MBB) const {
- if (!LIS)
- return LV->getVarInfo(Reg).AliveBlocks.test(MBB->getNumber());
-
const LiveInterval &LI = LIS->getInterval(Reg);
return LIS->isLiveInToMBB(LI, MBB) && LIS->isLiveOutOfMBB(LI, MBB);
}
bool SIOptimizeVGPRLiveRange::isLiveIntoMBB(
Register Reg, const MachineBasicBlock *MBB) const {
- if (!LIS)
- return LV->getVarInfo(Reg).isLiveIn(*MBB, Reg, *MRI);
-
const LiveInterval &LI = LIS->getInterval(Reg);
return LIS->isLiveInToMBB(LI, MBB);
}
@@ -244,17 +224,6 @@ void SIOptimizeVGPRLiveRange::collectElseRegionBlocks(
});
}
-/// Find the instructions(excluding phi) in \p MBB that uses the \p Reg.
-void SIOptimizeVGPRLiveRange::findNonPHIUsesInBlock(
- Register Reg, MachineBasicBlock *MBB,
- SmallVectorImpl<MachineInstr *> &Uses) const {
- for (auto &UseMI : MRI->use_nodbg_instructions(Reg)) {
- if (UseMI.getParent() == MBB && !UseMI.isPHI() &&
- UseMI.readsVirtualRegister(Reg))
- Uses.push_back(&UseMI);
- }
-}
-
/// Collect the killed registers in the ELSE region which are not alive through
/// the whole THEN region.
void SIOptimizeVGPRLiveRange::collectCandidateRegisters(
@@ -428,100 +397,6 @@ void SIOptimizeVGPRLiveRange::collectWaterfallCandidateRegisters(
}
}
-// Re-calculate the liveness of \p Reg in the THEN-region
-void SIOptimizeVGPRLiveRange::updateLiveRangeInThenRegion(
- Register Reg, MachineBasicBlock *If, MachineBasicBlock *Flow) const {
- SetVector<MachineBasicBlock *> Blocks;
- SmallVector<MachineBasicBlock *> WorkList({If});
-
- // Collect all successors until we see the flow block, where we should
- // reconverge.
- while (!WorkList.empty()) {
- auto *MBB = WorkList.pop_back_val();
- for (auto *Succ : MBB->successors()) {
- if (Succ != Flow && Blocks.insert(Succ))
- WorkList.push_back(Succ);
- }
- }
-
- LiveVariables::VarInfo &OldVarInfo = LV->getVarInfo(Reg);
- for (MachineBasicBlock *MBB : Blocks) {
- // Clear Live bit, as we will recalculate afterwards
- LLVM_DEBUG(dbgs() << "Clear AliveBlock " << printMBBReference(*MBB)
- << '\n');
- OldVarInfo.AliveBlocks.reset(MBB->getNumber());
- }
-
- SmallPtrSet<MachineBasicBlock *, 4> PHIIncoming;
-
- // Get the blocks the Reg should be alive through
- for (auto I = MRI->use_nodbg_begin(Reg), E = MRI->use_nodbg_end(); I != E;
- ++I) {
- auto *UseMI = I->getParent();
- if (UseMI->isPHI() && I->readsReg()) {
- if (Blocks.contains(UseMI->getParent()))
- PHIIncoming.insert(UseMI->getOperand(I.getOperandNo() + 1).getMBB());
- }
- }
-
- for (MachineBasicBlock *MBB : Blocks) {
- SmallVector<MachineInstr *> Uses;
- // PHI instructions has been processed before.
- findNonPHIUsesInBlock(Reg, MBB, Uses);
-
- if (Uses.size() == 1) {
- LLVM_DEBUG(dbgs() << "Found one Non-PHI use in "
- << printMBBReference(*MBB) << '\n');
- LV->HandleVirtRegUse(Reg, MBB, *(*Uses.begin()));
- } else if (Uses.size() > 1) {
- // Process the instructions in-order
- LLVM_DEBUG(dbgs() << "Found " << Uses.size() << " Non-PHI uses in "
- << printMBBReference(*MBB) << '\n');
- for (MachineInstr &MI : *MBB) {
- if (llvm::is_contained(Uses, &MI))
- LV->HandleVirtRegUse(Reg, MBB, MI);
- }
- }
-
- // Mark Reg alive through the block if this is a PHI incoming block
- if (PHIIncoming.contains(MBB))
- LV->MarkVirtRegAliveInBlock(OldVarInfo, MRI->getDefBlock(Reg), MBB);
- }
-
- // Set the isKilled flag if we get new Kills in the THEN region.
- for (auto *MI : OldVarInfo.Kills) {
- if (Blocks.contains(MI->getParent()))
- MI->addRegisterKilled(Reg, TRI);
- }
-}
-
-void SIOptimizeVGPRLiveRange::updateLiveRangeInElseRegion(
- Register Reg, Register NewReg, MachineBasicBlock *Flow,
- MachineBasicBlock *Endif,
- SmallSetVector<MachineBasicBlock *, 16> &ElseBlocks) const {
- LiveVariables::VarInfo &NewVarInfo = LV->getVarInfo(NewReg);
- LiveVariables::VarInfo &OldVarInfo = LV->getVarInfo(Reg);
-
- // Transfer aliveBlocks from Reg to NewReg
- for (auto *MBB : ElseBlocks) {
- unsigned BBNum = MBB->getNumber();
- if (OldVarInfo.AliveBlocks.test(BBNum)) {
- NewVarInfo.AliveBlocks.set(BBNum);
- LLVM_DEBUG(dbgs() << "Removing AliveBlock " << printMBBReference(*MBB)
- << '\n');
- OldVarInfo.AliveBlocks.reset(BBNum);
- }
- }
-
- // Transfer the possible Kills in ElseBlocks from Reg to NewReg
- llvm::erase_if(OldVarInfo.Kills, [&](MachineInstr *MI) {
- if (!ElseBlocks.contains(MI->getParent()))
- return false;
- NewVarInfo.Kills.push_back(MI);
- return true;
- });
-}
-
void SIOptimizeVGPRLiveRange::optimizeLiveRange(
Register Reg, MachineBasicBlock *If, MachineBasicBlock *Flow,
MachineBasicBlock *Endif,
@@ -567,28 +442,17 @@ void SIOptimizeVGPRLiveRange::optimizeLiveRange(
O.setReg(NewReg);
}
- if (LIS) {
- // The new PHI is a def of NewReg and a use of Reg and UndefReg; the uses of
- // Reg in the Else/Endif region were rewritten to NewReg. Kill flags moved
- // with the rewritten operands may no longer mark the last use, so drop them
- // and let the recomputed intervals be the source of truth.
- MRI->clearKillFlags(Reg);
- MRI->clearKillFlags(NewReg);
- LIS->InsertMachineInstrInMaps(*PHI);
- LIS->removeInterval(Reg);
- LIS->createAndComputeVirtRegInterval(Reg);
- LIS->createAndComputeVirtRegInterval(NewReg);
- LIS->createAndComputeVirtRegInterval(UndefReg);
- }
-
- if (LV) {
- // The optimized Reg is not alive through Flow blocks anymore.
- LiveVariables::VarInfo &OldVarInfo = LV->getVarInfo(Reg);
- OldVarInfo.AliveBlocks.reset(Flow->getNumber());
-
- updateLiveRangeInElseRegion(Reg, NewReg, Flow, Endif, ElseBlocks);
- updateLiveRangeInThenRegion(Reg, If, Flow);
- }
+ // The new PHI is a def of NewReg and a use of Reg and UndefReg; the uses of
+ // Reg in the Else/Endif region were rewritten to NewReg. Kill flags moved
+ // with the rewritten operands may no longer mark the last use, so drop them
+ // and let the recomputed intervals be the source of truth.
+ MRI->clearKillFlags(Reg);
+ MRI->clearKillFlags(NewReg);
+ LIS->InsertMachineInstrInMaps(*PHI);
+ LIS->removeInterval(Reg);
+ LIS->createAndComputeVirtRegInterval(Reg);
+ LIS->createAndComputeVirtRegInterval(NewReg);
+ LIS->createAndComputeVirtRegInterval(UndefReg);
}
void SIOptimizeVGPRLiveRange::optimizeWaterfallLiveRange(
@@ -621,49 +485,11 @@ void SIOptimizeVGPRLiveRange::optimizeWaterfallLiveRange(
PHI.addReg(Reg).addMBB(Pred);
}
- if (LIS) {
- LIS->InsertMachineInstrInMaps(*PHI);
- LIS->removeInterval(Reg);
- LIS->createAndComputeVirtRegInterval(Reg);
- LIS->createAndComputeVirtRegInterval(NewReg);
- LIS->createAndComputeVirtRegInterval(UndefReg);
- }
-
- if (LV) {
- LiveVariables::VarInfo &NewVarInfo = LV->getVarInfo(NewReg);
- LiveVariables::VarInfo &OldVarInfo = LV->getVarInfo(Reg);
-
- // Find last use and mark as kill
- MachineInstr *Kill = nullptr;
- for (auto *MI : reverse(Instructions)) {
- if (MI->readsRegister(NewReg, TRI)) {
- MI->addRegisterKilled(NewReg, TRI);
- NewVarInfo.Kills.push_back(MI);
- Kill = MI;
- break;
- }
- }
- assert(Kill && "Failed to find last usage of register in loop");
-
- MachineBasicBlock *KillBlock = Kill->getParent();
- bool PostKillBlock = false;
- for (auto *Block : Blocks) {
- auto BBNum = Block->getNumber();
-
- // collectWaterfallCandidateRegisters only collects registers that are
- // dead after the loop. So we know that the old reg is no longer live
- // throughout the waterfall loop.
- OldVarInfo.AliveBlocks.reset(BBNum);
-
- // The new register is live up to (and including) the block that kills it.
- PostKillBlock |= (Block == KillBlock);
- if (PostKillBlock) {
- NewVarInfo.AliveBlocks.reset(BBNum);
- } else if (Block != LoopHeader) {
- NewVarInfo.AliveBlocks.set(BBNum);
- }
- }
- }
+ LIS->InsertMachineInstrInMaps(*PHI);
+ LIS->removeInterval(Reg);
+ LIS->createAndComputeVirtRegInterval(Reg);
+ LIS->createAndComputeVirtRegInterval(NewReg);
+ LIS->createAndComputeVirtRegInterval(UndefReg);
}
char SIOptimizeVGPRLiveRangeLegacy::ID = 0;
@@ -672,7 +498,7 @@ INITIALIZE_PASS_BEGIN(SIOptimizeVGPRLiveRangeLegacy, DEBUG_TYPE,
"SI Optimize VGPR LiveRange", false, false)
INITIALIZE_PASS_DEPENDENCY(MachineDominatorTreeWrapperPass)
INITIALIZE_PASS_DEPENDENCY(MachineLoopInfoWrapperPass)
-INITIALIZE_PASS_DEPENDENCY(LiveVariablesWrapperPass)
+INITIALIZE_PASS_DEPENDENCY(LiveIntervalsWrapperPass)
INITIALIZE_PASS_END(SIOptimizeVGPRLiveRangeLegacy, DEBUG_TYPE,
"SI Optimize VGPR LiveRange", false, false)
@@ -686,33 +512,28 @@ bool SIOptimizeVGPRLiveRangeLegacy::runOnMachineFunction(MachineFunction &MF) {
if (skipFunction(MF.getFunction()))
return false;
- auto *LISWrapper = getAnalysisIfAvailable<LiveIntervalsWrapperPass>();
- LiveIntervals *LIS = LISWrapper ? &LISWrapper->getLIS() : nullptr;
- LiveVariables *LV = &getAnalysis<LiveVariablesWrapperPass>().getLV();
+ LiveIntervals *LIS = &getAnalysis<LiveIntervalsWrapperPass>().getLIS();
MachineDominatorTree *MDT =
&getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
MachineLoopInfo *Loops = &getAnalysis<MachineLoopInfoWrapperPass>().getLI();
- return SIOptimizeVGPRLiveRange(LIS, LV, MDT, Loops).run(MF);
+ return SIOptimizeVGPRLiveRange(LIS, MDT, Loops).run(MF);
}
PreservedAnalyses
SIOptimizeVGPRLiveRangePass::run(MachineFunction &MF,
MachineFunctionAnalysisManager &MFAM) {
MFPropsModifier _(*this, MF);
- LiveIntervals *LIS = MFAM.getCachedResult<LiveIntervalsAnalysis>(MF);
- LiveVariables *LV = MFAM.getCachedResult<LiveVariablesAnalysis>(MF);
- if (!LIS && !LV)
- LV = &MFAM.getResult<LiveVariablesAnalysis>(MF);
+ LiveIntervals *LIS = &MFAM.getResult<LiveIntervalsAnalysis>(MF);
MachineDominatorTree *MDT = &MFAM.getResult<MachineDominatorTreeAnalysis>(MF);
MachineLoopInfo *Loops = &MFAM.getResult<MachineLoopAnalysis>(MF);
- bool Changed = SIOptimizeVGPRLiveRange(LIS, LV, MDT, Loops).run(MF);
+ bool Changed = SIOptimizeVGPRLiveRange(LIS, MDT, Loops).run(MF);
if (!Changed)
return PreservedAnalyses::all();
auto PA = getMachineFunctionPassPreservedAnalyses();
+ PA.preserve<SlotIndexesAnalysis>();
PA.preserve<LiveIntervalsAnalysis>();
- PA.preserve<LiveVariablesAnalysis>();
PA.preserveSet<CFGAnalyses>();
return PA;
}
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll
index 03f235a1ee26f..01e0ad2d6ea42 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll
@@ -343,23 +343,23 @@ define void @divergent_i1_icmp_used_outside_loop(i32 %v0, i32 %v1, ptr addrspace
; GFX10: ; %bb.0: ; %entry
; GFX10-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX10-NEXT: s_mov_b32 s4, 0
-; GFX10-NEXT: s_mov_b32 s6, 0
-; GFX10-NEXT: ; implicit-def: $sgpr7
+; GFX10-NEXT: s_mov_b32 s7, 0
+; GFX10-NEXT: ; implicit-def: $sgpr6
; GFX10-NEXT: s_branch .LBB5_2
; GFX10-NEXT: .LBB5_1: ; %Flow
; GFX10-NEXT: ; in Loop: Header=BB5_2 Depth=1
; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s8
; GFX10-NEXT: s_and_b32 s5, exec_lo, s5
-; GFX10-NEXT: s_or_b32 s6, s5, s6
-; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s6
+; GFX10-NEXT: s_or_b32 s7, s5, s7
+; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s7
; GFX10-NEXT: s_cbranch_execz .LBB5_6
; GFX10-NEXT: .LBB5_2: ; %cond.block.0
; GFX10-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX10-NEXT: v_cmp_eq_u32_e32 vcc_lo, s4, v0
; GFX10-NEXT: v_mov_b32_e32 v4, s4
-; GFX10-NEXT: s_andn2_b32 s5, s7, exec_lo
-; GFX10-NEXT: s_and_b32 s7, exec_lo, vcc_lo
-; GFX10-NEXT: s_or_b32 s7, s5, s7
+; GFX10-NEXT: s_andn2_b32 s5, s6, exec_lo
+; GFX10-NEXT: s_and_b32 s6, exec_lo, vcc_lo
+; GFX10-NEXT: s_or_b32 s6, s5, s6
; GFX10-NEXT: s_and_saveexec_b32 s8, vcc_lo
; GFX10-NEXT: s_cbranch_execz .LBB5_4
; GFX10-NEXT: ; %bb.3: ; %if.block.0
@@ -386,8 +386,8 @@ define void @divergent_i1_icmp_used_outside_loop(i32 %v0, i32 %v1, ptr addrspace
; GFX10-NEXT: s_andn2_b32 s5, s5, exec_lo
; GFX10-NEXT: s_branch .LBB5_1
; GFX10-NEXT: .LBB5_6: ; %cond.block.1
-; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s6
-; GFX10-NEXT: s_and_saveexec_b32 s4, s7
+; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s7
+; GFX10-NEXT: s_and_saveexec_b32 s4, s6
; GFX10-NEXT: s_cbranch_execz .LBB5_8
; GFX10-NEXT: ; %bb.7: ; %if.block.1
; GFX10-NEXT: global_store_dword v[6:7], v4, off
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll
index 6f3245a39a9a7..f56b06ca2376d 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll
@@ -493,7 +493,7 @@ define amdgpu_ps i32 @irreducible_cfg(i32 %x, i32 %y, i32 %a0, i32 %a1, i32 %a2,
; GFX10-NEXT: .LBB6_1: ; %Flow2
; GFX10-NEXT: ; in Loop: Header=BB6_2 Depth=1
; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s6
-; GFX10-NEXT: s_and_b32 s4, exec_lo, s5
+; GFX10-NEXT: s_and_b32 s4, exec_lo, s4
; GFX10-NEXT: s_mov_b32 s0, exec_lo
; GFX10-NEXT: s_or_b32 s1, s4, s1
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s1
@@ -501,9 +501,9 @@ define amdgpu_ps i32 @irreducible_cfg(i32 %x, i32 %y, i32 %a0, i32 %a1, i32 %a2,
; GFX10-NEXT: .LBB6_2: ; %irr.guard
; GFX10-NEXT: ; =>This Loop Header: Depth=1
; GFX10-NEXT: ; Child Loop BB6_6 Depth 2
-; GFX10-NEXT: s_mov_b32 s4, exec_lo
-; GFX10-NEXT: s_and_saveexec_b32 s5, s0
-; GFX10-NEXT: s_xor_b32 s5, exec_lo, s5
+; GFX10-NEXT: s_mov_b32 s5, exec_lo
+; GFX10-NEXT: s_and_saveexec_b32 s4, s0
+; GFX10-NEXT: s_xor_b32 s4, exec_lo, s4
; GFX10-NEXT: ; %bb.3: ; %.loopexit
; GFX10-NEXT: ; in Loop: Header=BB6_2 Depth=1
; GFX10-NEXT: v_cmp_gt_i32_e64 s0, v5, v0
@@ -514,34 +514,34 @@ define amdgpu_ps i32 @irreducible_cfg(i32 %x, i32 %y, i32 %a0, i32 %a1, i32 %a2,
; GFX10-NEXT: s_or_b32 s6, s0, s6
; GFX10-NEXT: s_and_b32 s0, exec_lo, s0
; GFX10-NEXT: s_xor_b32 s6, s6, s7
-; GFX10-NEXT: s_andn2_b32 s4, s4, exec_lo
+; GFX10-NEXT: s_andn2_b32 s5, s5, exec_lo
; GFX10-NEXT: s_and_b32 s6, exec_lo, s6
; GFX10-NEXT: s_or_b32 s3, s3, s0
-; GFX10-NEXT: s_or_b32 s4, s4, s6
+; GFX10-NEXT: s_or_b32 s5, s5, s6
; GFX10-NEXT: ; %bb.4: ; %Flow1
; GFX10-NEXT: ; in Loop: Header=BB6_2 Depth=1
-; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s4
; GFX10-NEXT: s_andn2_b32 s0, s2, exec_lo
; GFX10-NEXT: s_and_b32 s2, exec_lo, s3
-; GFX10-NEXT: s_mov_b32 s5, exec_lo
+; GFX10-NEXT: s_mov_b32 s4, exec_lo
; GFX10-NEXT: s_or_b32 s2, s0, s2
-; GFX10-NEXT: s_and_saveexec_b32 s6, s4
+; GFX10-NEXT: s_and_saveexec_b32 s6, s5
; GFX10-NEXT: s_cbranch_execz .LBB6_1
; GFX10-NEXT: ; %bb.5: ; %.preheader
; GFX10-NEXT: ; in Loop: Header=BB6_2 Depth=1
; GFX10-NEXT: ...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/227252
More information about the llvm-commits
mailing list