[llvm] [NVPTX][AtomicExpandPass] Complete support for AtomicRMW in NVPTX (PR #176015)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Jan 22 06:19:21 PST 2026
================
@@ -7065,65 +7065,97 @@ NVPTXTargetLowering::shouldExpandAtomicRMWInIR(const AtomicRMWInst *AI) const {
bool NVPTXTargetLowering::shouldInsertFencesForAtomic(
const Instruction *I) const {
- auto *CI = dyn_cast<AtomicCmpXchgInst>(I);
// When CAS bitwidth is not supported on the hardware, the CAS is emulated
- // using a retry loop that uses a higher-bitwidth monotonic CAS. We enforce
- // the memory order using explicit fences around the retry loop.
- // The memory order of natively supported CAS operations can be enforced
- // by lowering to an atom.cas with the right memory synchronizing effect.
- // However, atom.cas only supports relaxed, acquire, release and acq_rel.
- // So we also use explicit fences for enforcing memory order for
- // seq_cast CAS with natively-supported bitwidths.
- return CI &&
- (cast<IntegerType>(CI->getCompareOperand()->getType())->getBitWidth() <
- STI.getMinCmpXchgSizeInBits() ||
- CI->getMergedOrdering() == AtomicOrdering::SequentiallyConsistent);
+ // using a retry loop that uses a higher-bitwidth monotonic CAS. Similarly, if
+ // the atomicrmw operation is not supported on hardware, we emulate it with a
+ // cmpxchg loop. In such cases, we enforce the memory order using explicit
+ // fences around the retry loop.
+ // The memory order of natively supported CAS or RMW operations can be
+ // enforced by lowering to an `atom.<op>` instr with the right memory
+ // synchronization effect. However, atom only supports relaxed, acquire,
+ // release and acq_rel. So we also use explicit fences to enforce memory
+ // order in seq_cast CAS or RMW instructions that can be lowered as acq_rel.
+ if (auto *CI = dyn_cast<AtomicCmpXchgInst>(I))
+ return (cast<IntegerType>(CI->getCompareOperand()->getType())
+ ->getBitWidth() < STI.getMinCmpXchgSizeInBits()) ||
+ CI->getMergedOrdering() == AtomicOrdering::SequentiallyConsistent;
+ if (auto *RI = dyn_cast<AtomicRMWInst>(I))
+ return shouldExpandAtomicRMWInIR(RI) == AtomicExpansionKind::CmpXChg ||
+ RI->getOrdering() == AtomicOrdering::SequentiallyConsistent;
+ return false;
}
AtomicOrdering NVPTXTargetLowering::atomicOperationOrderAfterFenceSplit(
const Instruction *I) const {
- auto *CI = dyn_cast<AtomicCmpXchgInst>(I);
- bool BitwidthSupportedAndIsSeqCst =
+ // Only lower to atom.<op>.acquire if the operation is not emulated, and its
+ // ordering is seq_cst. This produces a sequence of the form:
+ // fence.sc
+ // atom.<op>.acquire
+ // Instead of
+ // fence.sc
+ // atom.<op>
+ // fence.acquire
+ // The two-instruction sequence is weaker than the alternative, but guarantees
+ // seq_cst ordering.
+ //
+ // In all other cases, lower to atom.<op>.relaxed
+ if (auto *CI = dyn_cast<AtomicCmpXchgInst>(I);
CI && CI->getMergedOrdering() == AtomicOrdering::SequentiallyConsistent &&
cast<IntegerType>(CI->getCompareOperand()->getType())->getBitWidth() >=
- STI.getMinCmpXchgSizeInBits();
- return BitwidthSupportedAndIsSeqCst ? AtomicOrdering::Acquire
- : AtomicOrdering::Monotonic;
+ STI.getMinCmpXchgSizeInBits())
+ return AtomicOrdering::Acquire;
+ else if (auto *RI = dyn_cast<AtomicRMWInst>(I);
+ RI && RI->getOrdering() == AtomicOrdering::SequentiallyConsistent &&
+ shouldExpandAtomicRMWInIR(RI) == AtomicExpansionKind::None)
+ return AtomicOrdering::Acquire;
+
+ return AtomicOrdering::Monotonic;
}
+// prerequisites: shouldInsertFencesForAtomic() returns true for Inst
Instruction *NVPTXTargetLowering::emitLeadingFence(IRBuilderBase &Builder,
Instruction *Inst,
AtomicOrdering Ord) const {
- if (!isa<AtomicCmpXchgInst>(Inst))
+ auto *CI = dyn_cast<AtomicCmpXchgInst>(Inst);
+ auto *RI = dyn_cast<AtomicRMWInst>(Inst);
+ if (!CI && !RI)
return TargetLoweringBase::emitLeadingFence(Builder, Inst, Ord);
- // Specialize for cmpxchg
+ // Specialize for cmpxchg and rmw
// Emit a fence.sc leading fence for cmpxchg seq_cst which are not emulated
- SyncScope::ID SSID = cast<AtomicCmpXchgInst>(Inst)->getSyncScopeID();
+ auto SSID = getAtomicSyncScopeID(Inst);
+ assert(SSID.has_value() && "Expected an atomic operation");
----------------
gonzalobg wrote:
No, it was only to hide the assert.
Doesn't `.value()` fail if the optional is `nullopt`?
If so, then we probably dont need the `assert` at all cause we are always checking that later anyways.
https://github.com/llvm/llvm-project/pull/176015
More information about the llvm-commits
mailing list