[llvm] [InstCombine] Preserve elementwise atomic access sizes (PR #223897)
Yonah Goldberg via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 10:52:19 PDT 2026
https://github.com/YonahGoldberg updated https://github.com/llvm/llvm-project/pull/223897
>From 7769f3424f6d4274b81feef8a7b18660b7e182f6 Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 16 Sep 2026 01:49:36 +0000
Subject: [PATCH 1/5] [InstCombine] Add tests for elementwise atomic access
sizes
---
llvm/test/Transforms/InstCombine/atomic.ll | 119 +++++++++++++++++++++
1 file changed, 119 insertions(+)
diff --git a/llvm/test/Transforms/InstCombine/atomic.ll b/llvm/test/Transforms/InstCombine/atomic.ll
index 062aa3db34759..f5e22f4c7c8c0 100644
--- a/llvm/test/Transforms/InstCombine/atomic.ll
+++ b/llvm/test/Transforms/InstCombine/atomic.ll
@@ -474,4 +474,123 @@ define void @store_elementwise_bitcast(ptr %p, i64 %v) {
ret void
}
+define void @store_merge_elementwise_different_element_size(
+; CHECK-LABEL: @store_merge_elementwise_different_element_size(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br i1 [[C:%.*]], label [[THEN:%.*]], label [[ELSE:%.*]]
+; CHECK: then:
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast <2 x i64> [[B:%.*]] to <4 x i32>
+; CHECK-NEXT: br label [[END:%.*]]
+; CHECK: else:
+; CHECK-NEXT: br label [[END]]
+; CHECK: end:
+; CHECK-NEXT: [[STOREMERGE:%.*]] = phi <4 x i32> [ [[A:%.*]], [[ELSE]] ], [ [[TMP0]], [[THEN]] ]
+; CHECK-NEXT: store atomic elementwise <4 x i32> [[STOREMERGE]], ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT: ret void
+;
+ ptr %p, i1 %c, <4 x i32> %a, <2 x i64> %b) {
+entry:
+ br i1 %c, label %then, label %else
+then:
+ store atomic elementwise <2 x i64> %b, ptr %p unordered, align 16
+ br label %end
+else:
+ store atomic elementwise <4 x i32> %a, ptr %p unordered, align 16
+ br label %end
+end:
+ ret void
+}
+
+define void @store_merge_elementwise_same_element_size(
+; CHECK-LABEL: @store_merge_elementwise_same_element_size(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br i1 [[C:%.*]], label [[THEN:%.*]], label [[ELSE:%.*]]
+; CHECK: then:
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast <4 x float> [[B:%.*]] to <4 x i32>
+; CHECK-NEXT: br label [[END:%.*]]
+; CHECK: else:
+; CHECK-NEXT: br label [[END]]
+; CHECK: end:
+; CHECK-NEXT: [[STOREMERGE:%.*]] = phi <4 x i32> [ [[A:%.*]], [[ELSE]] ], [ [[TMP0]], [[THEN]] ]
+; CHECK-NEXT: store atomic elementwise <4 x i32> [[STOREMERGE]], ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT: ret void
+;
+ ptr %p, i1 %c, <4 x i32> %a, <4 x float> %b) {
+entry:
+ br i1 %c, label %then, label %else
+then:
+ store atomic elementwise <4 x float> %b, ptr %p unordered, align 16
+ br label %end
+else:
+ store atomic elementwise <4 x i32> %a, ptr %p unordered, align 16
+ br label %end
+end:
+ ret void
+}
+
+declare void @use_v2i64(<2 x i64>) memory(none)
+declare void @use_v4i32(<4 x i32>) memory(none)
+
+define <4 x i32> @load_cse_elementwise_different_element_size(ptr %p) {
+; CHECK-LABEL: @load_cse_elementwise_different_element_size(
+; CHECK-NEXT: [[A:%.*]] = load atomic elementwise <2 x i64>, ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT: call void @use_v2i64(<2 x i64> [[A]])
+; CHECK-NEXT: [[B_CAST:%.*]] = bitcast <2 x i64> [[A]] to <4 x i32>
+; CHECK-NEXT: ret <4 x i32> [[B_CAST]]
+;
+ %a = load atomic elementwise <2 x i64>, ptr %p unordered, align 16
+ call void @use_v2i64(<2 x i64> %a)
+ %b = load atomic elementwise <4 x i32>, ptr %p unordered, align 16
+ ret <4 x i32> %b
+}
+
+define <4 x float> @load_cse_elementwise_same_element_size(ptr %p) {
+; CHECK-LABEL: @load_cse_elementwise_same_element_size(
+; CHECK-NEXT: [[A:%.*]] = load atomic elementwise <4 x i32>, ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT: call void @use_v4i32(<4 x i32> [[A]])
+; CHECK-NEXT: [[B_CAST:%.*]] = bitcast <4 x i32> [[A]] to <4 x float>
+; CHECK-NEXT: ret <4 x float> [[B_CAST]]
+;
+ %a = load atomic elementwise <4 x i32>, ptr %p unordered, align 16
+ call void @use_v4i32(<4 x i32> %a)
+ %b = load atomic elementwise <4 x float>, ptr %p unordered, align 16
+ ret <4 x float> %b
+}
+
+define <4 x i32> @store_to_load_elementwise_different_element_size(
+; CHECK-LABEL: @store_to_load_elementwise_different_element_size(
+; CHECK-NEXT: store atomic elementwise <2 x i64> [[A:%.*]], ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT: [[B_CAST:%.*]] = bitcast <2 x i64> [[A]] to <4 x i32>
+; CHECK-NEXT: ret <4 x i32> [[B_CAST]]
+;
+ ptr %p, <2 x i64> %a) {
+ store atomic elementwise <2 x i64> %a, ptr %p unordered, align 16
+ %b = load atomic elementwise <4 x i32>, ptr %p unordered, align 16
+ ret <4 x i32> %b
+}
+
+define <4 x float> @store_to_load_elementwise_same_element_size(
+; CHECK-LABEL: @store_to_load_elementwise_same_element_size(
+; CHECK-NEXT: store atomic elementwise <4 x i32> [[A:%.*]], ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT: [[B_CAST:%.*]] = bitcast <4 x i32> [[A]] to <4 x float>
+; CHECK-NEXT: ret <4 x float> [[B_CAST]]
+;
+ ptr %p, <4 x i32> %a) {
+ store atomic elementwise <4 x i32> %a, ptr %p unordered, align 16
+ %b = load atomic elementwise <4 x float>, ptr %p unordered, align 16
+ ret <4 x float> %b
+}
+
+define <4 x i32> @load_cse_whole_to_elementwise(ptr %p) {
+; CHECK-LABEL: @load_cse_whole_to_elementwise(
+; CHECK-NEXT: [[A:%.*]] = load atomic <4 x i32>, ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT: call void @use_v4i32(<4 x i32> [[A]])
+; CHECK-NEXT: ret <4 x i32> [[A]]
+;
+ %a = load atomic <4 x i32>, ptr %p unordered, align 16
+ call void @use_v4i32(<4 x i32> %a)
+ %b = load atomic elementwise <4 x i32>, ptr %p unordered, align 16
+ ret <4 x i32> %b
+}
+
attributes #0 = { null_pointer_is_valid }
>From df4404f69cda1ac1a057e2760a161f14f98a2928 Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 16 Sep 2026 01:51:16 +0000
Subject: [PATCH 2/5] [InstCombine] Preserve elementwise atomic access sizes
---
llvm/include/llvm/Analysis/Loads.h | 6 ++-
llvm/lib/Analysis/Loads.cpp | 44 ++++++++++++++-----
.../InstCombineLoadStoreAlloca.cpp | 12 ++++-
llvm/lib/Transforms/Scalar/JumpThreading.cpp | 10 ++---
llvm/test/Transforms/InstCombine/atomic.ll | 12 ++---
5 files changed, 59 insertions(+), 25 deletions(-)
diff --git a/llvm/include/llvm/Analysis/Loads.h b/llvm/include/llvm/Analysis/Loads.h
index 22ce31d61ddb8..4213f6c073405 100644
--- a/llvm/include/llvm/Analysis/Loads.h
+++ b/llvm/include/llvm/Analysis/Loads.h
@@ -177,6 +177,7 @@ FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA, bool *IsLoadCSE,
/// \param AtLeastAtomic Are we looking for at-least an atomic load/store ? In
/// case it is false, we can return an atomic or non-atomic load or store. In
/// case it is true, we need to return an atomic load or store.
+/// \param IsElementwise Whether the requested atomic access is elementwise.
/// \param ScanBB The basic block to scan.
/// \param [in,out] ScanFrom The location to start scanning from. When this
/// function returns, it points at the last instruction scanned.
@@ -190,8 +191,9 @@ FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA, bool *IsLoadCSE,
/// \returns The found value, or nullptr if no value is found.
LLVM_ABI Value *findAvailablePtrLoadStore(
const MemoryLocation &Loc, Type *AccessTy, bool AtLeastAtomic,
- BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom, unsigned MaxInstsToScan,
- BatchAAResults *AA, bool *IsLoadCSE, unsigned *NumScanedInst);
+ bool IsElementwise, BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
+ unsigned MaxInstsToScan, BatchAAResults *AA, bool *IsLoadCSE,
+ unsigned *NumScanedInst);
/// Returns true if a pointer value \p From can be replaced with another pointer
/// value \To if they are deemed equal through some means (e.g. information from
diff --git a/llvm/lib/Analysis/Loads.cpp b/llvm/lib/Analysis/Loads.cpp
index de9022c540d42..65e81c1624773 100644
--- a/llvm/lib/Analysis/Loads.cpp
+++ b/llvm/lib/Analysis/Loads.cpp
@@ -559,9 +559,9 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BasicBlock *ScanBB,
return nullptr;
MemoryLocation Loc = MemoryLocation::get(Load);
- return findAvailablePtrLoadStore(Loc, Load->getType(), Load->isAtomic(),
- ScanBB, ScanFrom, MaxInstsToScan, AA, IsLoad,
- NumScanedInst);
+ return findAvailablePtrLoadStore(
+ Loc, Load->getType(), Load->isAtomic(), Load->isElementwise(), ScanBB,
+ ScanFrom, MaxInstsToScan, AA, IsLoad, NumScanedInst);
}
// Check if the load and the store have the same base, constant offsets and
@@ -592,7 +592,24 @@ static bool areNonOverlapSameBaseLoadAndStore(const Value *LoadPtr,
static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
Type *AccessTy, bool AtLeastAtomic,
- const DataLayout &DL, bool *IsLoadCSE) {
+ bool IsElementwise, const DataLayout &DL,
+ bool *IsLoadCSE) {
+ // For elementwise atomics, each vector element is a separate atomic access.
+ // Reusing an operation with a different access size would change atomicity.
+ auto hasCompatibleAtomicAccessSize = [&](Type *OtherAccessTy,
+ bool OtherIsElementwise) {
+ if (!AtLeastAtomic)
+ return true;
+
+ Type *AtomicAccessTy =
+ IsElementwise ? AccessTy->getScalarType() : AccessTy;
+ Type *OtherAtomicAccessTy = OtherIsElementwise
+ ? OtherAccessTy->getScalarType()
+ : OtherAccessTy;
+ return DL.getTypeStoreSize(AtomicAccessTy) ==
+ DL.getTypeStoreSize(OtherAtomicAccessTy);
+ };
+
// If this is a load of Ptr, the loaded value is available.
// (This is true even if the load is volatile or atomic, although
// those cases are unlikely.)
@@ -606,7 +623,8 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
if (!AreEquivalentAddressValues(LoadPtr, Ptr))
return nullptr;
- if (CastInst::isBitOrNoopPointerCastable(LI->getType(), AccessTy, DL)) {
+ if (hasCompatibleAtomicAccessSize(LI->getType(), LI->isElementwise()) &&
+ CastInst::isBitOrNoopPointerCastable(LI->getType(), AccessTy, DL)) {
if (IsLoadCSE)
*IsLoadCSE = true;
return LI;
@@ -630,6 +648,9 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
*IsLoadCSE = false;
Value *Val = SI->getValueOperand();
+ if (!hasCompatibleAtomicAccessSize(Val->getType(), SI->isElementwise()))
+ return nullptr;
+
if (CastInst::isBitOrNoopPointerCastable(Val->getType(), AccessTy, DL))
return Val;
@@ -686,8 +707,9 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
Value *llvm::findAvailablePtrLoadStore(
const MemoryLocation &Loc, Type *AccessTy, bool AtLeastAtomic,
- BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom, unsigned MaxInstsToScan,
- BatchAAResults *AA, bool *IsLoadCSE, unsigned *NumScanedInst) {
+ bool IsElementwise, BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
+ unsigned MaxInstsToScan, BatchAAResults *AA, bool *IsLoadCSE,
+ unsigned *NumScanedInst) {
if (MaxInstsToScan == 0)
MaxInstsToScan = ~0U;
@@ -713,8 +735,9 @@ Value *llvm::findAvailablePtrLoadStore(
--ScanFrom;
- if (Value *Available = getAvailableLoadStore(Inst, StrippedPtr, AccessTy,
- AtLeastAtomic, DL, IsLoadCSE))
+ if (Value *Available = getAvailableLoadStore(
+ Inst, StrippedPtr, AccessTy, AtLeastAtomic, IsElementwise, DL,
+ IsLoadCSE))
return Available;
// Try to get the store size for the type.
@@ -793,7 +816,8 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA,
return nullptr;
Available = getAvailableLoadStore(&Inst, StrippedPtr, AccessTy,
- AtLeastAtomic, DL, IsLoadCSE);
+ AtLeastAtomic, Load->isElementwise(), DL,
+ IsLoadCSE);
if (Available)
break;
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp b/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp
index 8e2a0c5376d8e..74322d599bd28 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp
@@ -1669,8 +1669,16 @@ bool InstCombinerImpl::mergeStoreIntoSuccessor(StoreInst &SI) {
auto *SIVTy = SI.getValueOperand()->getType();
auto *OSVTy = OtherStore->getValueOperand()->getType();
- return CastInst::isBitOrNoopPointerCastable(OSVTy, SIVTy, DL) &&
- SI.hasSameSpecialState(OtherStore);
+ if (!CastInst::isBitOrNoopPointerCastable(OSVTy, SIVTy, DL) ||
+ !SI.hasSameSpecialState(OtherStore))
+ return false;
+
+ // Elementwise atomic stores behave as one atomic store per vector
+ // element. Do not split or merge those atomic accesses by changing the
+ // element size.
+ return !SI.isElementwise() ||
+ DL.getTypeStoreSize(SIVTy->getScalarType()) ==
+ DL.getTypeStoreSize(OSVTy->getScalarType());
};
// If the other block ends in an unconditional branch, check for the 'if then
diff --git a/llvm/lib/Transforms/Scalar/JumpThreading.cpp b/llvm/lib/Transforms/Scalar/JumpThreading.cpp
index 66233407a5548..4f1ae940d3b7e 100644
--- a/llvm/lib/Transforms/Scalar/JumpThreading.cpp
+++ b/llvm/lib/Transforms/Scalar/JumpThreading.cpp
@@ -1314,8 +1314,8 @@ bool JumpThreadingPass::simplifyPartiallyRedundantLoad(LoadInst *LoadI) {
LocationSize::precise(DL.getTypeStoreSize(AccessTy)),
AATags);
PredAvailable = findAvailablePtrLoadStore(
- Loc, AccessTy, LoadI->isAtomic(), PredBB, BBIt, DefMaxInstsToScan,
- &BatchAA, &IsLoadCSE, &NumScanedInst);
+ Loc, AccessTy, LoadI->isAtomic(), LoadI->isElementwise(), PredBB, BBIt,
+ DefMaxInstsToScan, &BatchAA, &IsLoadCSE, &NumScanedInst);
// If PredBB has a single predecessor, continue scanning through the
// single predecessor.
@@ -1326,9 +1326,9 @@ bool JumpThreadingPass::simplifyPartiallyRedundantLoad(LoadInst *LoadI) {
if (SinglePredBB) {
BBIt = SinglePredBB->end();
PredAvailable = findAvailablePtrLoadStore(
- Loc, AccessTy, LoadI->isAtomic(), SinglePredBB, BBIt,
- (DefMaxInstsToScan - NumScanedInst), &BatchAA, &IsLoadCSE,
- &NumScanedInst);
+ Loc, AccessTy, LoadI->isAtomic(), LoadI->isElementwise(),
+ SinglePredBB, BBIt, (DefMaxInstsToScan - NumScanedInst), &BatchAA,
+ &IsLoadCSE, &NumScanedInst);
}
}
diff --git a/llvm/test/Transforms/InstCombine/atomic.ll b/llvm/test/Transforms/InstCombine/atomic.ll
index f5e22f4c7c8c0..660194f6e7e19 100644
--- a/llvm/test/Transforms/InstCombine/atomic.ll
+++ b/llvm/test/Transforms/InstCombine/atomic.ll
@@ -479,13 +479,12 @@ define void @store_merge_elementwise_different_element_size(
; CHECK-NEXT: entry:
; CHECK-NEXT: br i1 [[C:%.*]], label [[THEN:%.*]], label [[ELSE:%.*]]
; CHECK: then:
-; CHECK-NEXT: [[TMP0:%.*]] = bitcast <2 x i64> [[B:%.*]] to <4 x i32>
+; CHECK-NEXT: store atomic elementwise <2 x i64> [[B:%.*]], ptr [[P:%.*]] unordered, align 16
; CHECK-NEXT: br label [[END:%.*]]
; CHECK: else:
+; CHECK-NEXT: store atomic elementwise <4 x i32> [[A:%.*]], ptr [[P]] unordered, align 16
; CHECK-NEXT: br label [[END]]
; CHECK: end:
-; CHECK-NEXT: [[STOREMERGE:%.*]] = phi <4 x i32> [ [[A:%.*]], [[ELSE]] ], [ [[TMP0]], [[THEN]] ]
-; CHECK-NEXT: store atomic elementwise <4 x i32> [[STOREMERGE]], ptr [[P:%.*]] unordered, align 16
; CHECK-NEXT: ret void
;
ptr %p, i1 %c, <4 x i32> %a, <2 x i64> %b) {
@@ -535,7 +534,7 @@ define <4 x i32> @load_cse_elementwise_different_element_size(ptr %p) {
; CHECK-LABEL: @load_cse_elementwise_different_element_size(
; CHECK-NEXT: [[A:%.*]] = load atomic elementwise <2 x i64>, ptr [[P:%.*]] unordered, align 16
; CHECK-NEXT: call void @use_v2i64(<2 x i64> [[A]])
-; CHECK-NEXT: [[B_CAST:%.*]] = bitcast <2 x i64> [[A]] to <4 x i32>
+; CHECK-NEXT: [[B_CAST:%.*]] = load atomic elementwise <4 x i32>, ptr [[P]] unordered, align 16
; CHECK-NEXT: ret <4 x i32> [[B_CAST]]
;
%a = load atomic elementwise <2 x i64>, ptr %p unordered, align 16
@@ -560,7 +559,7 @@ define <4 x float> @load_cse_elementwise_same_element_size(ptr %p) {
define <4 x i32> @store_to_load_elementwise_different_element_size(
; CHECK-LABEL: @store_to_load_elementwise_different_element_size(
; CHECK-NEXT: store atomic elementwise <2 x i64> [[A:%.*]], ptr [[P:%.*]] unordered, align 16
-; CHECK-NEXT: [[B_CAST:%.*]] = bitcast <2 x i64> [[A]] to <4 x i32>
+; CHECK-NEXT: [[B_CAST:%.*]] = load atomic elementwise <4 x i32>, ptr [[P]] unordered, align 16
; CHECK-NEXT: ret <4 x i32> [[B_CAST]]
;
ptr %p, <2 x i64> %a) {
@@ -585,7 +584,8 @@ define <4 x i32> @load_cse_whole_to_elementwise(ptr %p) {
; CHECK-LABEL: @load_cse_whole_to_elementwise(
; CHECK-NEXT: [[A:%.*]] = load atomic <4 x i32>, ptr [[P:%.*]] unordered, align 16
; CHECK-NEXT: call void @use_v4i32(<4 x i32> [[A]])
-; CHECK-NEXT: ret <4 x i32> [[A]]
+; CHECK-NEXT: [[B:%.*]] = load atomic elementwise <4 x i32>, ptr [[P]] unordered, align 16
+; CHECK-NEXT: ret <4 x i32> [[B]]
;
%a = load atomic <4 x i32>, ptr %p unordered, align 16
call void @use_v4i32(<4 x i32> %a)
>From cadcce7c4661b19934c49fdd834f67154aae60f3 Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 16 Sep 2026 02:52:45 +0000
Subject: [PATCH 3/5] format
---
llvm/include/llvm/Analysis/Loads.h | 11 +++++----
llvm/lib/Analysis/Loads.cpp | 38 +++++++++++++++---------------
2 files changed, 25 insertions(+), 24 deletions(-)
diff --git a/llvm/include/llvm/Analysis/Loads.h b/llvm/include/llvm/Analysis/Loads.h
index 4213f6c073405..30319b97838bc 100644
--- a/llvm/include/llvm/Analysis/Loads.h
+++ b/llvm/include/llvm/Analysis/Loads.h
@@ -189,11 +189,12 @@ FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA, bool *IsLoadCSE,
/// location in memory, as opposed to the value operand of a store.
///
/// \returns The found value, or nullptr if no value is found.
-LLVM_ABI Value *findAvailablePtrLoadStore(
- const MemoryLocation &Loc, Type *AccessTy, bool AtLeastAtomic,
- bool IsElementwise, BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
- unsigned MaxInstsToScan, BatchAAResults *AA, bool *IsLoadCSE,
- unsigned *NumScanedInst);
+LLVM_ABI Value *
+findAvailablePtrLoadStore(const MemoryLocation &Loc, Type *AccessTy,
+ bool AtLeastAtomic, bool IsElementwise,
+ BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
+ unsigned MaxInstsToScan, BatchAAResults *AA,
+ bool *IsLoadCSE, unsigned *NumScanedInst);
/// Returns true if a pointer value \p From can be replaced with another pointer
/// value \To if they are deemed equal through some means (e.g. information from
diff --git a/llvm/lib/Analysis/Loads.cpp b/llvm/lib/Analysis/Loads.cpp
index 65e81c1624773..422883f26a986 100644
--- a/llvm/lib/Analysis/Loads.cpp
+++ b/llvm/lib/Analysis/Loads.cpp
@@ -559,9 +559,9 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BasicBlock *ScanBB,
return nullptr;
MemoryLocation Loc = MemoryLocation::get(Load);
- return findAvailablePtrLoadStore(
- Loc, Load->getType(), Load->isAtomic(), Load->isElementwise(), ScanBB,
- ScanFrom, MaxInstsToScan, AA, IsLoad, NumScanedInst);
+ return findAvailablePtrLoadStore(Loc, Load->getType(), Load->isAtomic(),
+ Load->isElementwise(), ScanBB, ScanFrom,
+ MaxInstsToScan, AA, IsLoad, NumScanedInst);
}
// Check if the load and the store have the same base, constant offsets and
@@ -601,11 +601,9 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
if (!AtLeastAtomic)
return true;
- Type *AtomicAccessTy =
- IsElementwise ? AccessTy->getScalarType() : AccessTy;
- Type *OtherAtomicAccessTy = OtherIsElementwise
- ? OtherAccessTy->getScalarType()
- : OtherAccessTy;
+ Type *AtomicAccessTy = IsElementwise ? AccessTy->getScalarType() : AccessTy;
+ Type *OtherAtomicAccessTy =
+ OtherIsElementwise ? OtherAccessTy->getScalarType() : OtherAccessTy;
return DL.getTypeStoreSize(AtomicAccessTy) ==
DL.getTypeStoreSize(OtherAtomicAccessTy);
};
@@ -705,11 +703,13 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
return nullptr;
}
-Value *llvm::findAvailablePtrLoadStore(
- const MemoryLocation &Loc, Type *AccessTy, bool AtLeastAtomic,
- bool IsElementwise, BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
- unsigned MaxInstsToScan, BatchAAResults *AA, bool *IsLoadCSE,
- unsigned *NumScanedInst) {
+Value *llvm::findAvailablePtrLoadStore(const MemoryLocation &Loc,
+ Type *AccessTy, bool AtLeastAtomic,
+ bool IsElementwise, BasicBlock *ScanBB,
+ BasicBlock::iterator &ScanFrom,
+ unsigned MaxInstsToScan,
+ BatchAAResults *AA, bool *IsLoadCSE,
+ unsigned *NumScanedInst) {
if (MaxInstsToScan == 0)
MaxInstsToScan = ~0U;
@@ -735,9 +735,9 @@ Value *llvm::findAvailablePtrLoadStore(
--ScanFrom;
- if (Value *Available = getAvailableLoadStore(
- Inst, StrippedPtr, AccessTy, AtLeastAtomic, IsElementwise, DL,
- IsLoadCSE))
+ if (Value *Available =
+ getAvailableLoadStore(Inst, StrippedPtr, AccessTy, AtLeastAtomic,
+ IsElementwise, DL, IsLoadCSE))
return Available;
// Try to get the store size for the type.
@@ -815,9 +815,9 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA,
if (MaxInstsToScan-- == 0)
return nullptr;
- Available = getAvailableLoadStore(&Inst, StrippedPtr, AccessTy,
- AtLeastAtomic, Load->isElementwise(), DL,
- IsLoadCSE);
+ Available =
+ getAvailableLoadStore(&Inst, StrippedPtr, AccessTy, AtLeastAtomic,
+ Load->isElementwise(), DL, IsLoadCSE);
if (Available)
break;
>From bb6fa09c8a758ab0fda0a085c3d4c4ddccbb6f7f Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 16 Sep 2026 17:28:33 +0000
Subject: [PATCH 4/5] Use LoadStoreInstProperties
---
llvm/include/llvm/Analysis/Loads.h | 8 +--
llvm/lib/Analysis/Loads.cpp | 74 +++++++++++---------
llvm/lib/Transforms/Scalar/JumpThreading.cpp | 10 +--
3 files changed, 47 insertions(+), 45 deletions(-)
diff --git a/llvm/include/llvm/Analysis/Loads.h b/llvm/include/llvm/Analysis/Loads.h
index 30319b97838bc..19f1d7ba7e30a 100644
--- a/llvm/include/llvm/Analysis/Loads.h
+++ b/llvm/include/llvm/Analysis/Loads.h
@@ -28,6 +28,7 @@ class DataLayout;
class DominatorTree;
class Instruction;
class LoadInst;
+struct LoadStoreInstProperties;
class Loop;
class MemoryLocation;
class SCEV;
@@ -174,10 +175,7 @@ FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA, bool *IsLoadCSE,
///
/// \param Loc The location we want the load and store to originate from.
/// \param AccessTy The access type of the pointer.
-/// \param AtLeastAtomic Are we looking for at-least an atomic load/store ? In
-/// case it is false, we can return an atomic or non-atomic load or store. In
-/// case it is true, we need to return an atomic load or store.
-/// \param IsElementwise Whether the requested atomic access is elementwise.
+/// \param AccessProps The properties of the load we want to replace.
/// \param ScanBB The basic block to scan.
/// \param [in,out] ScanFrom The location to start scanning from. When this
/// function returns, it points at the last instruction scanned.
@@ -191,7 +189,7 @@ FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA, bool *IsLoadCSE,
/// \returns The found value, or nullptr if no value is found.
LLVM_ABI Value *
findAvailablePtrLoadStore(const MemoryLocation &Loc, Type *AccessTy,
- bool AtLeastAtomic, bool IsElementwise,
+ const LoadStoreInstProperties &AccessProps,
BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
unsigned MaxInstsToScan, BatchAAResults *AA,
bool *IsLoadCSE, unsigned *NumScanedInst);
diff --git a/llvm/lib/Analysis/Loads.cpp b/llvm/lib/Analysis/Loads.cpp
index 422883f26a986..7e8759a7138d2 100644
--- a/llvm/lib/Analysis/Loads.cpp
+++ b/llvm/lib/Analysis/Loads.cpp
@@ -559,9 +559,9 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BasicBlock *ScanBB,
return nullptr;
MemoryLocation Loc = MemoryLocation::get(Load);
- return findAvailablePtrLoadStore(Loc, Load->getType(), Load->isAtomic(),
- Load->isElementwise(), ScanBB, ScanFrom,
- MaxInstsToScan, AA, IsLoad, NumScanedInst);
+ return findAvailablePtrLoadStore(Loc, Load->getType(), Load->getProperties(),
+ ScanBB, ScanFrom, MaxInstsToScan, AA, IsLoad,
+ NumScanedInst);
}
// Check if the load and the store have the same base, constant offsets and
@@ -590,23 +590,29 @@ static bool areNonOverlapSameBaseLoadAndStore(const Value *LoadPtr,
return LoadRange.intersectWith(StoreRange).isEmptySet();
}
-static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
- Type *AccessTy, bool AtLeastAtomic,
- bool IsElementwise, const DataLayout &DL,
- bool *IsLoadCSE) {
- // For elementwise atomics, each vector element is a separate atomic access.
- // Reusing an operation with a different access size would change atomicity.
- auto hasCompatibleAtomicAccessSize = [&](Type *OtherAccessTy,
- bool OtherIsElementwise) {
- if (!AtLeastAtomic)
- return true;
+// For elementwise atomics, each vector element is a separate atomic access.
+// Reusing an operation with a different access size would change atomicity.
+static bool hasCompatibleAtomicAccessSize(
+ Type *AccessTy, const LoadStoreInstProperties &AccessProps,
+ Type *OtherAccessTy, const LoadStoreInstProperties &OtherProps,
+ const DataLayout &DL) {
+ if (AccessProps.Ordering == AtomicOrdering::NotAtomic)
+ return true;
- Type *AtomicAccessTy = IsElementwise ? AccessTy->getScalarType() : AccessTy;
- Type *OtherAtomicAccessTy =
- OtherIsElementwise ? OtherAccessTy->getScalarType() : OtherAccessTy;
- return DL.getTypeStoreSize(AtomicAccessTy) ==
- DL.getTypeStoreSize(OtherAtomicAccessTy);
- };
+ Type *AtomicAccessTy =
+ AccessProps.IsElementwise ? AccessTy->getScalarType() : AccessTy;
+ Type *OtherAtomicAccessTy =
+ OtherProps.IsElementwise ? OtherAccessTy->getScalarType() : OtherAccessTy;
+ return DL.getTypeStoreSize(AtomicAccessTy) ==
+ DL.getTypeStoreSize(OtherAtomicAccessTy);
+}
+
+static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
+ Type *AccessTy,
+ const LoadStoreInstProperties &AccessProps,
+ const DataLayout &DL, bool *IsLoadCSE) {
+ const bool AtLeastAtomic =
+ AccessProps.Ordering != AtomicOrdering::NotAtomic;
// If this is a load of Ptr, the loaded value is available.
// (This is true even if the load is volatile or atomic, although
@@ -621,7 +627,8 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
if (!AreEquivalentAddressValues(LoadPtr, Ptr))
return nullptr;
- if (hasCompatibleAtomicAccessSize(LI->getType(), LI->isElementwise()) &&
+ if (hasCompatibleAtomicAccessSize(AccessTy, AccessProps, LI->getType(),
+ LI->getProperties(), DL) &&
CastInst::isBitOrNoopPointerCastable(LI->getType(), AccessTy, DL)) {
if (IsLoadCSE)
*IsLoadCSE = true;
@@ -646,7 +653,8 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
*IsLoadCSE = false;
Value *Val = SI->getValueOperand();
- if (!hasCompatibleAtomicAccessSize(Val->getType(), SI->isElementwise()))
+ if (!hasCompatibleAtomicAccessSize(AccessTy, AccessProps, Val->getType(),
+ SI->getProperties(), DL))
return nullptr;
if (CastInst::isBitOrNoopPointerCastable(Val->getType(), AccessTy, DL))
@@ -703,13 +711,11 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
return nullptr;
}
-Value *llvm::findAvailablePtrLoadStore(const MemoryLocation &Loc,
- Type *AccessTy, bool AtLeastAtomic,
- bool IsElementwise, BasicBlock *ScanBB,
- BasicBlock::iterator &ScanFrom,
- unsigned MaxInstsToScan,
- BatchAAResults *AA, bool *IsLoadCSE,
- unsigned *NumScanedInst) {
+Value *llvm::findAvailablePtrLoadStore(
+ const MemoryLocation &Loc, Type *AccessTy,
+ const LoadStoreInstProperties &AccessProps, BasicBlock *ScanBB,
+ BasicBlock::iterator &ScanFrom, unsigned MaxInstsToScan, BatchAAResults *AA,
+ bool *IsLoadCSE, unsigned *NumScanedInst) {
if (MaxInstsToScan == 0)
MaxInstsToScan = ~0U;
@@ -735,9 +741,8 @@ Value *llvm::findAvailablePtrLoadStore(const MemoryLocation &Loc,
--ScanFrom;
- if (Value *Available =
- getAvailableLoadStore(Inst, StrippedPtr, AccessTy, AtLeastAtomic,
- IsElementwise, DL, IsLoadCSE))
+ if (Value *Available = getAvailableLoadStore(Inst, StrippedPtr, AccessTy,
+ AccessProps, DL, IsLoadCSE))
return Available;
// Try to get the store size for the type.
@@ -798,7 +803,7 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA,
Value *StrippedPtr = Load->getPointerOperand()->stripPointerCasts();
BasicBlock *ScanBB = Load->getParent();
Type *AccessTy = Load->getType();
- bool AtLeastAtomic = Load->isAtomic();
+ LoadStoreInstProperties AccessProps = Load->getProperties();
if (!Load->isUnordered())
return nullptr;
@@ -815,9 +820,8 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA,
if (MaxInstsToScan-- == 0)
return nullptr;
- Available =
- getAvailableLoadStore(&Inst, StrippedPtr, AccessTy, AtLeastAtomic,
- Load->isElementwise(), DL, IsLoadCSE);
+ Available = getAvailableLoadStore(&Inst, StrippedPtr, AccessTy, AccessProps,
+ DL, IsLoadCSE);
if (Available)
break;
diff --git a/llvm/lib/Transforms/Scalar/JumpThreading.cpp b/llvm/lib/Transforms/Scalar/JumpThreading.cpp
index 4f1ae940d3b7e..37315fbd49d3a 100644
--- a/llvm/lib/Transforms/Scalar/JumpThreading.cpp
+++ b/llvm/lib/Transforms/Scalar/JumpThreading.cpp
@@ -1314,8 +1314,8 @@ bool JumpThreadingPass::simplifyPartiallyRedundantLoad(LoadInst *LoadI) {
LocationSize::precise(DL.getTypeStoreSize(AccessTy)),
AATags);
PredAvailable = findAvailablePtrLoadStore(
- Loc, AccessTy, LoadI->isAtomic(), LoadI->isElementwise(), PredBB, BBIt,
- DefMaxInstsToScan, &BatchAA, &IsLoadCSE, &NumScanedInst);
+ Loc, AccessTy, LoadI->getProperties(), PredBB, BBIt, DefMaxInstsToScan,
+ &BatchAA, &IsLoadCSE, &NumScanedInst);
// If PredBB has a single predecessor, continue scanning through the
// single predecessor.
@@ -1326,9 +1326,9 @@ bool JumpThreadingPass::simplifyPartiallyRedundantLoad(LoadInst *LoadI) {
if (SinglePredBB) {
BBIt = SinglePredBB->end();
PredAvailable = findAvailablePtrLoadStore(
- Loc, AccessTy, LoadI->isAtomic(), LoadI->isElementwise(),
- SinglePredBB, BBIt, (DefMaxInstsToScan - NumScanedInst), &BatchAA,
- &IsLoadCSE, &NumScanedInst);
+ Loc, AccessTy, LoadI->getProperties(), SinglePredBB, BBIt,
+ (DefMaxInstsToScan - NumScanedInst), &BatchAA, &IsLoadCSE,
+ &NumScanedInst);
}
}
>From 35ddc438f2d0686f90fc55532a27ac1ee0055013 Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 16 Sep 2026 17:28:43 +0000
Subject: [PATCH 5/5] format
---
llvm/lib/Analysis/Loads.cpp | 3 +--
1 file changed, 1 insertion(+), 2 deletions(-)
diff --git a/llvm/lib/Analysis/Loads.cpp b/llvm/lib/Analysis/Loads.cpp
index 7e8759a7138d2..4d96fbe85d7cd 100644
--- a/llvm/lib/Analysis/Loads.cpp
+++ b/llvm/lib/Analysis/Loads.cpp
@@ -611,8 +611,7 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
Type *AccessTy,
const LoadStoreInstProperties &AccessProps,
const DataLayout &DL, bool *IsLoadCSE) {
- const bool AtLeastAtomic =
- AccessProps.Ordering != AtomicOrdering::NotAtomic;
+ const bool AtLeastAtomic = AccessProps.Ordering != AtomicOrdering::NotAtomic;
// If this is a load of Ptr, the loaded value is available.
// (This is true even if the load is volatile or atomic, although
More information about the llvm-commits
mailing list