[llvm] [InstCombine] Preserve elementwise atomic access sizes (PR #223897)

Yonah Goldberg via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 10:52:19 PDT 2026


https://github.com/YonahGoldberg updated https://github.com/llvm/llvm-project/pull/223897

>From 7769f3424f6d4274b81feef8a7b18660b7e182f6 Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 16 Sep 2026 01:49:36 +0000
Subject: [PATCH 1/5] [InstCombine] Add tests for elementwise atomic access
 sizes

---
 llvm/test/Transforms/InstCombine/atomic.ll | 119 +++++++++++++++++++++
 1 file changed, 119 insertions(+)

diff --git a/llvm/test/Transforms/InstCombine/atomic.ll b/llvm/test/Transforms/InstCombine/atomic.ll
index 062aa3db34759..f5e22f4c7c8c0 100644
--- a/llvm/test/Transforms/InstCombine/atomic.ll
+++ b/llvm/test/Transforms/InstCombine/atomic.ll
@@ -474,4 +474,123 @@ define void @store_elementwise_bitcast(ptr %p, i64 %v) {
   ret void
 }
 
+define void @store_merge_elementwise_different_element_size(
+; CHECK-LABEL: @store_merge_elementwise_different_element_size(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br i1 [[C:%.*]], label [[THEN:%.*]], label [[ELSE:%.*]]
+; CHECK:       then:
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast <2 x i64> [[B:%.*]] to <4 x i32>
+; CHECK-NEXT:    br label [[END:%.*]]
+; CHECK:       else:
+; CHECK-NEXT:    br label [[END]]
+; CHECK:       end:
+; CHECK-NEXT:    [[STOREMERGE:%.*]] = phi <4 x i32> [ [[A:%.*]], [[ELSE]] ], [ [[TMP0]], [[THEN]] ]
+; CHECK-NEXT:    store atomic elementwise <4 x i32> [[STOREMERGE]], ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT:    ret void
+;
+  ptr %p, i1 %c, <4 x i32> %a, <2 x i64> %b) {
+entry:
+  br i1 %c, label %then, label %else
+then:
+  store atomic elementwise <2 x i64> %b, ptr %p unordered, align 16
+  br label %end
+else:
+  store atomic elementwise <4 x i32> %a, ptr %p unordered, align 16
+  br label %end
+end:
+  ret void
+}
+
+define void @store_merge_elementwise_same_element_size(
+; CHECK-LABEL: @store_merge_elementwise_same_element_size(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br i1 [[C:%.*]], label [[THEN:%.*]], label [[ELSE:%.*]]
+; CHECK:       then:
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast <4 x float> [[B:%.*]] to <4 x i32>
+; CHECK-NEXT:    br label [[END:%.*]]
+; CHECK:       else:
+; CHECK-NEXT:    br label [[END]]
+; CHECK:       end:
+; CHECK-NEXT:    [[STOREMERGE:%.*]] = phi <4 x i32> [ [[A:%.*]], [[ELSE]] ], [ [[TMP0]], [[THEN]] ]
+; CHECK-NEXT:    store atomic elementwise <4 x i32> [[STOREMERGE]], ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT:    ret void
+;
+  ptr %p, i1 %c, <4 x i32> %a, <4 x float> %b) {
+entry:
+  br i1 %c, label %then, label %else
+then:
+  store atomic elementwise <4 x float> %b, ptr %p unordered, align 16
+  br label %end
+else:
+  store atomic elementwise <4 x i32> %a, ptr %p unordered, align 16
+  br label %end
+end:
+  ret void
+}
+
+declare void @use_v2i64(<2 x i64>) memory(none)
+declare void @use_v4i32(<4 x i32>) memory(none)
+
+define <4 x i32> @load_cse_elementwise_different_element_size(ptr %p) {
+; CHECK-LABEL: @load_cse_elementwise_different_element_size(
+; CHECK-NEXT:    [[A:%.*]] = load atomic elementwise <2 x i64>, ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT:    call void @use_v2i64(<2 x i64> [[A]])
+; CHECK-NEXT:    [[B_CAST:%.*]] = bitcast <2 x i64> [[A]] to <4 x i32>
+; CHECK-NEXT:    ret <4 x i32> [[B_CAST]]
+;
+  %a = load atomic elementwise <2 x i64>, ptr %p unordered, align 16
+  call void @use_v2i64(<2 x i64> %a)
+  %b = load atomic elementwise <4 x i32>, ptr %p unordered, align 16
+  ret <4 x i32> %b
+}
+
+define <4 x float> @load_cse_elementwise_same_element_size(ptr %p) {
+; CHECK-LABEL: @load_cse_elementwise_same_element_size(
+; CHECK-NEXT:    [[A:%.*]] = load atomic elementwise <4 x i32>, ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT:    call void @use_v4i32(<4 x i32> [[A]])
+; CHECK-NEXT:    [[B_CAST:%.*]] = bitcast <4 x i32> [[A]] to <4 x float>
+; CHECK-NEXT:    ret <4 x float> [[B_CAST]]
+;
+  %a = load atomic elementwise <4 x i32>, ptr %p unordered, align 16
+  call void @use_v4i32(<4 x i32> %a)
+  %b = load atomic elementwise <4 x float>, ptr %p unordered, align 16
+  ret <4 x float> %b
+}
+
+define <4 x i32> @store_to_load_elementwise_different_element_size(
+; CHECK-LABEL: @store_to_load_elementwise_different_element_size(
+; CHECK-NEXT:    store atomic elementwise <2 x i64> [[A:%.*]], ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT:    [[B_CAST:%.*]] = bitcast <2 x i64> [[A]] to <4 x i32>
+; CHECK-NEXT:    ret <4 x i32> [[B_CAST]]
+;
+  ptr %p, <2 x i64> %a) {
+  store atomic elementwise <2 x i64> %a, ptr %p unordered, align 16
+  %b = load atomic elementwise <4 x i32>, ptr %p unordered, align 16
+  ret <4 x i32> %b
+}
+
+define <4 x float> @store_to_load_elementwise_same_element_size(
+; CHECK-LABEL: @store_to_load_elementwise_same_element_size(
+; CHECK-NEXT:    store atomic elementwise <4 x i32> [[A:%.*]], ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT:    [[B_CAST:%.*]] = bitcast <4 x i32> [[A]] to <4 x float>
+; CHECK-NEXT:    ret <4 x float> [[B_CAST]]
+;
+  ptr %p, <4 x i32> %a) {
+  store atomic elementwise <4 x i32> %a, ptr %p unordered, align 16
+  %b = load atomic elementwise <4 x float>, ptr %p unordered, align 16
+  ret <4 x float> %b
+}
+
+define <4 x i32> @load_cse_whole_to_elementwise(ptr %p) {
+; CHECK-LABEL: @load_cse_whole_to_elementwise(
+; CHECK-NEXT:    [[A:%.*]] = load atomic <4 x i32>, ptr [[P:%.*]] unordered, align 16
+; CHECK-NEXT:    call void @use_v4i32(<4 x i32> [[A]])
+; CHECK-NEXT:    ret <4 x i32> [[A]]
+;
+  %a = load atomic <4 x i32>, ptr %p unordered, align 16
+  call void @use_v4i32(<4 x i32> %a)
+  %b = load atomic elementwise <4 x i32>, ptr %p unordered, align 16
+  ret <4 x i32> %b
+}
+
 attributes #0 = { null_pointer_is_valid }

>From df4404f69cda1ac1a057e2760a161f14f98a2928 Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 16 Sep 2026 01:51:16 +0000
Subject: [PATCH 2/5] [InstCombine] Preserve elementwise atomic access sizes

---
 llvm/include/llvm/Analysis/Loads.h            |  6 ++-
 llvm/lib/Analysis/Loads.cpp                   | 44 ++++++++++++++-----
 .../InstCombineLoadStoreAlloca.cpp            | 12 ++++-
 llvm/lib/Transforms/Scalar/JumpThreading.cpp  | 10 ++---
 llvm/test/Transforms/InstCombine/atomic.ll    | 12 ++---
 5 files changed, 59 insertions(+), 25 deletions(-)

diff --git a/llvm/include/llvm/Analysis/Loads.h b/llvm/include/llvm/Analysis/Loads.h
index 22ce31d61ddb8..4213f6c073405 100644
--- a/llvm/include/llvm/Analysis/Loads.h
+++ b/llvm/include/llvm/Analysis/Loads.h
@@ -177,6 +177,7 @@ FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA, bool *IsLoadCSE,
 /// \param AtLeastAtomic Are we looking for at-least an atomic load/store ? In
 /// case it is false, we can return an atomic or non-atomic load or store. In
 /// case it is true, we need to return an atomic load or store.
+/// \param IsElementwise Whether the requested atomic access is elementwise.
 /// \param ScanBB The basic block to scan.
 /// \param [in,out] ScanFrom The location to start scanning from. When this
 /// function returns, it points at the last instruction scanned.
@@ -190,8 +191,9 @@ FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA, bool *IsLoadCSE,
 /// \returns The found value, or nullptr if no value is found.
 LLVM_ABI Value *findAvailablePtrLoadStore(
     const MemoryLocation &Loc, Type *AccessTy, bool AtLeastAtomic,
-    BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom, unsigned MaxInstsToScan,
-    BatchAAResults *AA, bool *IsLoadCSE, unsigned *NumScanedInst);
+    bool IsElementwise, BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
+    unsigned MaxInstsToScan, BatchAAResults *AA, bool *IsLoadCSE,
+    unsigned *NumScanedInst);
 
 /// Returns true if a pointer value \p From can be replaced with another pointer
 /// value \To if they are deemed equal through some means (e.g. information from
diff --git a/llvm/lib/Analysis/Loads.cpp b/llvm/lib/Analysis/Loads.cpp
index de9022c540d42..65e81c1624773 100644
--- a/llvm/lib/Analysis/Loads.cpp
+++ b/llvm/lib/Analysis/Loads.cpp
@@ -559,9 +559,9 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BasicBlock *ScanBB,
     return nullptr;
 
   MemoryLocation Loc = MemoryLocation::get(Load);
-  return findAvailablePtrLoadStore(Loc, Load->getType(), Load->isAtomic(),
-                                   ScanBB, ScanFrom, MaxInstsToScan, AA, IsLoad,
-                                   NumScanedInst);
+  return findAvailablePtrLoadStore(
+      Loc, Load->getType(), Load->isAtomic(), Load->isElementwise(), ScanBB,
+      ScanFrom, MaxInstsToScan, AA, IsLoad, NumScanedInst);
 }
 
 // Check if the load and the store have the same base, constant offsets and
@@ -592,7 +592,24 @@ static bool areNonOverlapSameBaseLoadAndStore(const Value *LoadPtr,
 
 static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
                                     Type *AccessTy, bool AtLeastAtomic,
-                                    const DataLayout &DL, bool *IsLoadCSE) {
+                                    bool IsElementwise, const DataLayout &DL,
+                                    bool *IsLoadCSE) {
+  // For elementwise atomics, each vector element is a separate atomic access.
+  // Reusing an operation with a different access size would change atomicity.
+  auto hasCompatibleAtomicAccessSize = [&](Type *OtherAccessTy,
+                                           bool OtherIsElementwise) {
+    if (!AtLeastAtomic)
+      return true;
+
+    Type *AtomicAccessTy =
+        IsElementwise ? AccessTy->getScalarType() : AccessTy;
+    Type *OtherAtomicAccessTy = OtherIsElementwise
+                                    ? OtherAccessTy->getScalarType()
+                                    : OtherAccessTy;
+    return DL.getTypeStoreSize(AtomicAccessTy) ==
+           DL.getTypeStoreSize(OtherAtomicAccessTy);
+  };
+
   // If this is a load of Ptr, the loaded value is available.
   // (This is true even if the load is volatile or atomic, although
   // those cases are unlikely.)
@@ -606,7 +623,8 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
     if (!AreEquivalentAddressValues(LoadPtr, Ptr))
       return nullptr;
 
-    if (CastInst::isBitOrNoopPointerCastable(LI->getType(), AccessTy, DL)) {
+    if (hasCompatibleAtomicAccessSize(LI->getType(), LI->isElementwise()) &&
+        CastInst::isBitOrNoopPointerCastable(LI->getType(), AccessTy, DL)) {
       if (IsLoadCSE)
         *IsLoadCSE = true;
       return LI;
@@ -630,6 +648,9 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
       *IsLoadCSE = false;
 
     Value *Val = SI->getValueOperand();
+    if (!hasCompatibleAtomicAccessSize(Val->getType(), SI->isElementwise()))
+      return nullptr;
+
     if (CastInst::isBitOrNoopPointerCastable(Val->getType(), AccessTy, DL))
       return Val;
 
@@ -686,8 +707,9 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
 
 Value *llvm::findAvailablePtrLoadStore(
     const MemoryLocation &Loc, Type *AccessTy, bool AtLeastAtomic,
-    BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom, unsigned MaxInstsToScan,
-    BatchAAResults *AA, bool *IsLoadCSE, unsigned *NumScanedInst) {
+    bool IsElementwise, BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
+    unsigned MaxInstsToScan, BatchAAResults *AA, bool *IsLoadCSE,
+    unsigned *NumScanedInst) {
   if (MaxInstsToScan == 0)
     MaxInstsToScan = ~0U;
 
@@ -713,8 +735,9 @@ Value *llvm::findAvailablePtrLoadStore(
 
     --ScanFrom;
 
-    if (Value *Available = getAvailableLoadStore(Inst, StrippedPtr, AccessTy,
-                                                 AtLeastAtomic, DL, IsLoadCSE))
+    if (Value *Available = getAvailableLoadStore(
+            Inst, StrippedPtr, AccessTy, AtLeastAtomic, IsElementwise, DL,
+            IsLoadCSE))
       return Available;
 
     // Try to get the store size for the type.
@@ -793,7 +816,8 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA,
       return nullptr;
 
     Available = getAvailableLoadStore(&Inst, StrippedPtr, AccessTy,
-                                      AtLeastAtomic, DL, IsLoadCSE);
+                                      AtLeastAtomic, Load->isElementwise(), DL,
+                                      IsLoadCSE);
     if (Available)
       break;
 
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp b/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp
index 8e2a0c5376d8e..74322d599bd28 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp
@@ -1669,8 +1669,16 @@ bool InstCombinerImpl::mergeStoreIntoSuccessor(StoreInst &SI) {
 
     auto *SIVTy = SI.getValueOperand()->getType();
     auto *OSVTy = OtherStore->getValueOperand()->getType();
-    return CastInst::isBitOrNoopPointerCastable(OSVTy, SIVTy, DL) &&
-           SI.hasSameSpecialState(OtherStore);
+    if (!CastInst::isBitOrNoopPointerCastable(OSVTy, SIVTy, DL) ||
+        !SI.hasSameSpecialState(OtherStore))
+      return false;
+
+    // Elementwise atomic stores behave as one atomic store per vector
+    // element. Do not split or merge those atomic accesses by changing the
+    // element size.
+    return !SI.isElementwise() ||
+           DL.getTypeStoreSize(SIVTy->getScalarType()) ==
+               DL.getTypeStoreSize(OSVTy->getScalarType());
   };
 
   // If the other block ends in an unconditional branch, check for the 'if then
diff --git a/llvm/lib/Transforms/Scalar/JumpThreading.cpp b/llvm/lib/Transforms/Scalar/JumpThreading.cpp
index 66233407a5548..4f1ae940d3b7e 100644
--- a/llvm/lib/Transforms/Scalar/JumpThreading.cpp
+++ b/llvm/lib/Transforms/Scalar/JumpThreading.cpp
@@ -1314,8 +1314,8 @@ bool JumpThreadingPass::simplifyPartiallyRedundantLoad(LoadInst *LoadI) {
                        LocationSize::precise(DL.getTypeStoreSize(AccessTy)),
                        AATags);
     PredAvailable = findAvailablePtrLoadStore(
-        Loc, AccessTy, LoadI->isAtomic(), PredBB, BBIt, DefMaxInstsToScan,
-        &BatchAA, &IsLoadCSE, &NumScanedInst);
+        Loc, AccessTy, LoadI->isAtomic(), LoadI->isElementwise(), PredBB, BBIt,
+        DefMaxInstsToScan, &BatchAA, &IsLoadCSE, &NumScanedInst);
 
     // If PredBB has a single predecessor, continue scanning through the
     // single predecessor.
@@ -1326,9 +1326,9 @@ bool JumpThreadingPass::simplifyPartiallyRedundantLoad(LoadInst *LoadI) {
       if (SinglePredBB) {
         BBIt = SinglePredBB->end();
         PredAvailable = findAvailablePtrLoadStore(
-            Loc, AccessTy, LoadI->isAtomic(), SinglePredBB, BBIt,
-            (DefMaxInstsToScan - NumScanedInst), &BatchAA, &IsLoadCSE,
-            &NumScanedInst);
+            Loc, AccessTy, LoadI->isAtomic(), LoadI->isElementwise(),
+            SinglePredBB, BBIt, (DefMaxInstsToScan - NumScanedInst), &BatchAA,
+            &IsLoadCSE, &NumScanedInst);
       }
     }
 
diff --git a/llvm/test/Transforms/InstCombine/atomic.ll b/llvm/test/Transforms/InstCombine/atomic.ll
index f5e22f4c7c8c0..660194f6e7e19 100644
--- a/llvm/test/Transforms/InstCombine/atomic.ll
+++ b/llvm/test/Transforms/InstCombine/atomic.ll
@@ -479,13 +479,12 @@ define void @store_merge_elementwise_different_element_size(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br i1 [[C:%.*]], label [[THEN:%.*]], label [[ELSE:%.*]]
 ; CHECK:       then:
-; CHECK-NEXT:    [[TMP0:%.*]] = bitcast <2 x i64> [[B:%.*]] to <4 x i32>
+; CHECK-NEXT:    store atomic elementwise <2 x i64> [[B:%.*]], ptr [[P:%.*]] unordered, align 16
 ; CHECK-NEXT:    br label [[END:%.*]]
 ; CHECK:       else:
+; CHECK-NEXT:    store atomic elementwise <4 x i32> [[A:%.*]], ptr [[P]] unordered, align 16
 ; CHECK-NEXT:    br label [[END]]
 ; CHECK:       end:
-; CHECK-NEXT:    [[STOREMERGE:%.*]] = phi <4 x i32> [ [[A:%.*]], [[ELSE]] ], [ [[TMP0]], [[THEN]] ]
-; CHECK-NEXT:    store atomic elementwise <4 x i32> [[STOREMERGE]], ptr [[P:%.*]] unordered, align 16
 ; CHECK-NEXT:    ret void
 ;
   ptr %p, i1 %c, <4 x i32> %a, <2 x i64> %b) {
@@ -535,7 +534,7 @@ define <4 x i32> @load_cse_elementwise_different_element_size(ptr %p) {
 ; CHECK-LABEL: @load_cse_elementwise_different_element_size(
 ; CHECK-NEXT:    [[A:%.*]] = load atomic elementwise <2 x i64>, ptr [[P:%.*]] unordered, align 16
 ; CHECK-NEXT:    call void @use_v2i64(<2 x i64> [[A]])
-; CHECK-NEXT:    [[B_CAST:%.*]] = bitcast <2 x i64> [[A]] to <4 x i32>
+; CHECK-NEXT:    [[B_CAST:%.*]] = load atomic elementwise <4 x i32>, ptr [[P]] unordered, align 16
 ; CHECK-NEXT:    ret <4 x i32> [[B_CAST]]
 ;
   %a = load atomic elementwise <2 x i64>, ptr %p unordered, align 16
@@ -560,7 +559,7 @@ define <4 x float> @load_cse_elementwise_same_element_size(ptr %p) {
 define <4 x i32> @store_to_load_elementwise_different_element_size(
 ; CHECK-LABEL: @store_to_load_elementwise_different_element_size(
 ; CHECK-NEXT:    store atomic elementwise <2 x i64> [[A:%.*]], ptr [[P:%.*]] unordered, align 16
-; CHECK-NEXT:    [[B_CAST:%.*]] = bitcast <2 x i64> [[A]] to <4 x i32>
+; CHECK-NEXT:    [[B_CAST:%.*]] = load atomic elementwise <4 x i32>, ptr [[P]] unordered, align 16
 ; CHECK-NEXT:    ret <4 x i32> [[B_CAST]]
 ;
   ptr %p, <2 x i64> %a) {
@@ -585,7 +584,8 @@ define <4 x i32> @load_cse_whole_to_elementwise(ptr %p) {
 ; CHECK-LABEL: @load_cse_whole_to_elementwise(
 ; CHECK-NEXT:    [[A:%.*]] = load atomic <4 x i32>, ptr [[P:%.*]] unordered, align 16
 ; CHECK-NEXT:    call void @use_v4i32(<4 x i32> [[A]])
-; CHECK-NEXT:    ret <4 x i32> [[A]]
+; CHECK-NEXT:    [[B:%.*]] = load atomic elementwise <4 x i32>, ptr [[P]] unordered, align 16
+; CHECK-NEXT:    ret <4 x i32> [[B]]
 ;
   %a = load atomic <4 x i32>, ptr %p unordered, align 16
   call void @use_v4i32(<4 x i32> %a)

>From cadcce7c4661b19934c49fdd834f67154aae60f3 Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 16 Sep 2026 02:52:45 +0000
Subject: [PATCH 3/5] format

---
 llvm/include/llvm/Analysis/Loads.h | 11 +++++----
 llvm/lib/Analysis/Loads.cpp        | 38 +++++++++++++++---------------
 2 files changed, 25 insertions(+), 24 deletions(-)

diff --git a/llvm/include/llvm/Analysis/Loads.h b/llvm/include/llvm/Analysis/Loads.h
index 4213f6c073405..30319b97838bc 100644
--- a/llvm/include/llvm/Analysis/Loads.h
+++ b/llvm/include/llvm/Analysis/Loads.h
@@ -189,11 +189,12 @@ FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA, bool *IsLoadCSE,
 /// location in memory, as opposed to the value operand of a store.
 ///
 /// \returns The found value, or nullptr if no value is found.
-LLVM_ABI Value *findAvailablePtrLoadStore(
-    const MemoryLocation &Loc, Type *AccessTy, bool AtLeastAtomic,
-    bool IsElementwise, BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
-    unsigned MaxInstsToScan, BatchAAResults *AA, bool *IsLoadCSE,
-    unsigned *NumScanedInst);
+LLVM_ABI Value *
+findAvailablePtrLoadStore(const MemoryLocation &Loc, Type *AccessTy,
+                          bool AtLeastAtomic, bool IsElementwise,
+                          BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
+                          unsigned MaxInstsToScan, BatchAAResults *AA,
+                          bool *IsLoadCSE, unsigned *NumScanedInst);
 
 /// Returns true if a pointer value \p From can be replaced with another pointer
 /// value \To if they are deemed equal through some means (e.g. information from
diff --git a/llvm/lib/Analysis/Loads.cpp b/llvm/lib/Analysis/Loads.cpp
index 65e81c1624773..422883f26a986 100644
--- a/llvm/lib/Analysis/Loads.cpp
+++ b/llvm/lib/Analysis/Loads.cpp
@@ -559,9 +559,9 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BasicBlock *ScanBB,
     return nullptr;
 
   MemoryLocation Loc = MemoryLocation::get(Load);
-  return findAvailablePtrLoadStore(
-      Loc, Load->getType(), Load->isAtomic(), Load->isElementwise(), ScanBB,
-      ScanFrom, MaxInstsToScan, AA, IsLoad, NumScanedInst);
+  return findAvailablePtrLoadStore(Loc, Load->getType(), Load->isAtomic(),
+                                   Load->isElementwise(), ScanBB, ScanFrom,
+                                   MaxInstsToScan, AA, IsLoad, NumScanedInst);
 }
 
 // Check if the load and the store have the same base, constant offsets and
@@ -601,11 +601,9 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
     if (!AtLeastAtomic)
       return true;
 
-    Type *AtomicAccessTy =
-        IsElementwise ? AccessTy->getScalarType() : AccessTy;
-    Type *OtherAtomicAccessTy = OtherIsElementwise
-                                    ? OtherAccessTy->getScalarType()
-                                    : OtherAccessTy;
+    Type *AtomicAccessTy = IsElementwise ? AccessTy->getScalarType() : AccessTy;
+    Type *OtherAtomicAccessTy =
+        OtherIsElementwise ? OtherAccessTy->getScalarType() : OtherAccessTy;
     return DL.getTypeStoreSize(AtomicAccessTy) ==
            DL.getTypeStoreSize(OtherAtomicAccessTy);
   };
@@ -705,11 +703,13 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
   return nullptr;
 }
 
-Value *llvm::findAvailablePtrLoadStore(
-    const MemoryLocation &Loc, Type *AccessTy, bool AtLeastAtomic,
-    bool IsElementwise, BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
-    unsigned MaxInstsToScan, BatchAAResults *AA, bool *IsLoadCSE,
-    unsigned *NumScanedInst) {
+Value *llvm::findAvailablePtrLoadStore(const MemoryLocation &Loc,
+                                       Type *AccessTy, bool AtLeastAtomic,
+                                       bool IsElementwise, BasicBlock *ScanBB,
+                                       BasicBlock::iterator &ScanFrom,
+                                       unsigned MaxInstsToScan,
+                                       BatchAAResults *AA, bool *IsLoadCSE,
+                                       unsigned *NumScanedInst) {
   if (MaxInstsToScan == 0)
     MaxInstsToScan = ~0U;
 
@@ -735,9 +735,9 @@ Value *llvm::findAvailablePtrLoadStore(
 
     --ScanFrom;
 
-    if (Value *Available = getAvailableLoadStore(
-            Inst, StrippedPtr, AccessTy, AtLeastAtomic, IsElementwise, DL,
-            IsLoadCSE))
+    if (Value *Available =
+            getAvailableLoadStore(Inst, StrippedPtr, AccessTy, AtLeastAtomic,
+                                  IsElementwise, DL, IsLoadCSE))
       return Available;
 
     // Try to get the store size for the type.
@@ -815,9 +815,9 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA,
     if (MaxInstsToScan-- == 0)
       return nullptr;
 
-    Available = getAvailableLoadStore(&Inst, StrippedPtr, AccessTy,
-                                      AtLeastAtomic, Load->isElementwise(), DL,
-                                      IsLoadCSE);
+    Available =
+        getAvailableLoadStore(&Inst, StrippedPtr, AccessTy, AtLeastAtomic,
+                              Load->isElementwise(), DL, IsLoadCSE);
     if (Available)
       break;
 

>From bb6fa09c8a758ab0fda0a085c3d4c4ddccbb6f7f Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 16 Sep 2026 17:28:33 +0000
Subject: [PATCH 4/5] Use LoadStoreInstProperties

---
 llvm/include/llvm/Analysis/Loads.h           |  8 +--
 llvm/lib/Analysis/Loads.cpp                  | 74 +++++++++++---------
 llvm/lib/Transforms/Scalar/JumpThreading.cpp | 10 +--
 3 files changed, 47 insertions(+), 45 deletions(-)

diff --git a/llvm/include/llvm/Analysis/Loads.h b/llvm/include/llvm/Analysis/Loads.h
index 30319b97838bc..19f1d7ba7e30a 100644
--- a/llvm/include/llvm/Analysis/Loads.h
+++ b/llvm/include/llvm/Analysis/Loads.h
@@ -28,6 +28,7 @@ class DataLayout;
 class DominatorTree;
 class Instruction;
 class LoadInst;
+struct LoadStoreInstProperties;
 class Loop;
 class MemoryLocation;
 class SCEV;
@@ -174,10 +175,7 @@ FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA, bool *IsLoadCSE,
 ///
 /// \param Loc The location we want the load and store to originate from.
 /// \param AccessTy The access type of the pointer.
-/// \param AtLeastAtomic Are we looking for at-least an atomic load/store ? In
-/// case it is false, we can return an atomic or non-atomic load or store. In
-/// case it is true, we need to return an atomic load or store.
-/// \param IsElementwise Whether the requested atomic access is elementwise.
+/// \param AccessProps The properties of the load we want to replace.
 /// \param ScanBB The basic block to scan.
 /// \param [in,out] ScanFrom The location to start scanning from. When this
 /// function returns, it points at the last instruction scanned.
@@ -191,7 +189,7 @@ FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA, bool *IsLoadCSE,
 /// \returns The found value, or nullptr if no value is found.
 LLVM_ABI Value *
 findAvailablePtrLoadStore(const MemoryLocation &Loc, Type *AccessTy,
-                          bool AtLeastAtomic, bool IsElementwise,
+                          const LoadStoreInstProperties &AccessProps,
                           BasicBlock *ScanBB, BasicBlock::iterator &ScanFrom,
                           unsigned MaxInstsToScan, BatchAAResults *AA,
                           bool *IsLoadCSE, unsigned *NumScanedInst);
diff --git a/llvm/lib/Analysis/Loads.cpp b/llvm/lib/Analysis/Loads.cpp
index 422883f26a986..7e8759a7138d2 100644
--- a/llvm/lib/Analysis/Loads.cpp
+++ b/llvm/lib/Analysis/Loads.cpp
@@ -559,9 +559,9 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BasicBlock *ScanBB,
     return nullptr;
 
   MemoryLocation Loc = MemoryLocation::get(Load);
-  return findAvailablePtrLoadStore(Loc, Load->getType(), Load->isAtomic(),
-                                   Load->isElementwise(), ScanBB, ScanFrom,
-                                   MaxInstsToScan, AA, IsLoad, NumScanedInst);
+  return findAvailablePtrLoadStore(Loc, Load->getType(), Load->getProperties(),
+                                   ScanBB, ScanFrom, MaxInstsToScan, AA, IsLoad,
+                                   NumScanedInst);
 }
 
 // Check if the load and the store have the same base, constant offsets and
@@ -590,23 +590,29 @@ static bool areNonOverlapSameBaseLoadAndStore(const Value *LoadPtr,
   return LoadRange.intersectWith(StoreRange).isEmptySet();
 }
 
-static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
-                                    Type *AccessTy, bool AtLeastAtomic,
-                                    bool IsElementwise, const DataLayout &DL,
-                                    bool *IsLoadCSE) {
-  // For elementwise atomics, each vector element is a separate atomic access.
-  // Reusing an operation with a different access size would change atomicity.
-  auto hasCompatibleAtomicAccessSize = [&](Type *OtherAccessTy,
-                                           bool OtherIsElementwise) {
-    if (!AtLeastAtomic)
-      return true;
+// For elementwise atomics, each vector element is a separate atomic access.
+// Reusing an operation with a different access size would change atomicity.
+static bool hasCompatibleAtomicAccessSize(
+    Type *AccessTy, const LoadStoreInstProperties &AccessProps,
+    Type *OtherAccessTy, const LoadStoreInstProperties &OtherProps,
+    const DataLayout &DL) {
+  if (AccessProps.Ordering == AtomicOrdering::NotAtomic)
+    return true;
 
-    Type *AtomicAccessTy = IsElementwise ? AccessTy->getScalarType() : AccessTy;
-    Type *OtherAtomicAccessTy =
-        OtherIsElementwise ? OtherAccessTy->getScalarType() : OtherAccessTy;
-    return DL.getTypeStoreSize(AtomicAccessTy) ==
-           DL.getTypeStoreSize(OtherAtomicAccessTy);
-  };
+  Type *AtomicAccessTy =
+      AccessProps.IsElementwise ? AccessTy->getScalarType() : AccessTy;
+  Type *OtherAtomicAccessTy =
+      OtherProps.IsElementwise ? OtherAccessTy->getScalarType() : OtherAccessTy;
+  return DL.getTypeStoreSize(AtomicAccessTy) ==
+         DL.getTypeStoreSize(OtherAtomicAccessTy);
+}
+
+static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
+                                    Type *AccessTy,
+                                    const LoadStoreInstProperties &AccessProps,
+                                    const DataLayout &DL, bool *IsLoadCSE) {
+  const bool AtLeastAtomic =
+      AccessProps.Ordering != AtomicOrdering::NotAtomic;
 
   // If this is a load of Ptr, the loaded value is available.
   // (This is true even if the load is volatile or atomic, although
@@ -621,7 +627,8 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
     if (!AreEquivalentAddressValues(LoadPtr, Ptr))
       return nullptr;
 
-    if (hasCompatibleAtomicAccessSize(LI->getType(), LI->isElementwise()) &&
+    if (hasCompatibleAtomicAccessSize(AccessTy, AccessProps, LI->getType(),
+                                      LI->getProperties(), DL) &&
         CastInst::isBitOrNoopPointerCastable(LI->getType(), AccessTy, DL)) {
       if (IsLoadCSE)
         *IsLoadCSE = true;
@@ -646,7 +653,8 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
       *IsLoadCSE = false;
 
     Value *Val = SI->getValueOperand();
-    if (!hasCompatibleAtomicAccessSize(Val->getType(), SI->isElementwise()))
+    if (!hasCompatibleAtomicAccessSize(AccessTy, AccessProps, Val->getType(),
+                                       SI->getProperties(), DL))
       return nullptr;
 
     if (CastInst::isBitOrNoopPointerCastable(Val->getType(), AccessTy, DL))
@@ -703,13 +711,11 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
   return nullptr;
 }
 
-Value *llvm::findAvailablePtrLoadStore(const MemoryLocation &Loc,
-                                       Type *AccessTy, bool AtLeastAtomic,
-                                       bool IsElementwise, BasicBlock *ScanBB,
-                                       BasicBlock::iterator &ScanFrom,
-                                       unsigned MaxInstsToScan,
-                                       BatchAAResults *AA, bool *IsLoadCSE,
-                                       unsigned *NumScanedInst) {
+Value *llvm::findAvailablePtrLoadStore(
+    const MemoryLocation &Loc, Type *AccessTy,
+    const LoadStoreInstProperties &AccessProps, BasicBlock *ScanBB,
+    BasicBlock::iterator &ScanFrom, unsigned MaxInstsToScan, BatchAAResults *AA,
+    bool *IsLoadCSE, unsigned *NumScanedInst) {
   if (MaxInstsToScan == 0)
     MaxInstsToScan = ~0U;
 
@@ -735,9 +741,8 @@ Value *llvm::findAvailablePtrLoadStore(const MemoryLocation &Loc,
 
     --ScanFrom;
 
-    if (Value *Available =
-            getAvailableLoadStore(Inst, StrippedPtr, AccessTy, AtLeastAtomic,
-                                  IsElementwise, DL, IsLoadCSE))
+    if (Value *Available = getAvailableLoadStore(Inst, StrippedPtr, AccessTy,
+                                                 AccessProps, DL, IsLoadCSE))
       return Available;
 
     // Try to get the store size for the type.
@@ -798,7 +803,7 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA,
   Value *StrippedPtr = Load->getPointerOperand()->stripPointerCasts();
   BasicBlock *ScanBB = Load->getParent();
   Type *AccessTy = Load->getType();
-  bool AtLeastAtomic = Load->isAtomic();
+  LoadStoreInstProperties AccessProps = Load->getProperties();
 
   if (!Load->isUnordered())
     return nullptr;
@@ -815,9 +820,8 @@ Value *llvm::FindAvailableLoadedValue(LoadInst *Load, BatchAAResults &AA,
     if (MaxInstsToScan-- == 0)
       return nullptr;
 
-    Available =
-        getAvailableLoadStore(&Inst, StrippedPtr, AccessTy, AtLeastAtomic,
-                              Load->isElementwise(), DL, IsLoadCSE);
+    Available = getAvailableLoadStore(&Inst, StrippedPtr, AccessTy, AccessProps,
+                                      DL, IsLoadCSE);
     if (Available)
       break;
 
diff --git a/llvm/lib/Transforms/Scalar/JumpThreading.cpp b/llvm/lib/Transforms/Scalar/JumpThreading.cpp
index 4f1ae940d3b7e..37315fbd49d3a 100644
--- a/llvm/lib/Transforms/Scalar/JumpThreading.cpp
+++ b/llvm/lib/Transforms/Scalar/JumpThreading.cpp
@@ -1314,8 +1314,8 @@ bool JumpThreadingPass::simplifyPartiallyRedundantLoad(LoadInst *LoadI) {
                        LocationSize::precise(DL.getTypeStoreSize(AccessTy)),
                        AATags);
     PredAvailable = findAvailablePtrLoadStore(
-        Loc, AccessTy, LoadI->isAtomic(), LoadI->isElementwise(), PredBB, BBIt,
-        DefMaxInstsToScan, &BatchAA, &IsLoadCSE, &NumScanedInst);
+        Loc, AccessTy, LoadI->getProperties(), PredBB, BBIt, DefMaxInstsToScan,
+        &BatchAA, &IsLoadCSE, &NumScanedInst);
 
     // If PredBB has a single predecessor, continue scanning through the
     // single predecessor.
@@ -1326,9 +1326,9 @@ bool JumpThreadingPass::simplifyPartiallyRedundantLoad(LoadInst *LoadI) {
       if (SinglePredBB) {
         BBIt = SinglePredBB->end();
         PredAvailable = findAvailablePtrLoadStore(
-            Loc, AccessTy, LoadI->isAtomic(), LoadI->isElementwise(),
-            SinglePredBB, BBIt, (DefMaxInstsToScan - NumScanedInst), &BatchAA,
-            &IsLoadCSE, &NumScanedInst);
+            Loc, AccessTy, LoadI->getProperties(), SinglePredBB, BBIt,
+            (DefMaxInstsToScan - NumScanedInst), &BatchAA, &IsLoadCSE,
+            &NumScanedInst);
       }
     }
 

>From 35ddc438f2d0686f90fc55532a27ac1ee0055013 Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 16 Sep 2026 17:28:43 +0000
Subject: [PATCH 5/5] format

---
 llvm/lib/Analysis/Loads.cpp | 3 +--
 1 file changed, 1 insertion(+), 2 deletions(-)

diff --git a/llvm/lib/Analysis/Loads.cpp b/llvm/lib/Analysis/Loads.cpp
index 7e8759a7138d2..4d96fbe85d7cd 100644
--- a/llvm/lib/Analysis/Loads.cpp
+++ b/llvm/lib/Analysis/Loads.cpp
@@ -611,8 +611,7 @@ static Value *getAvailableLoadStore(Instruction *Inst, const Value *Ptr,
                                     Type *AccessTy,
                                     const LoadStoreInstProperties &AccessProps,
                                     const DataLayout &DL, bool *IsLoadCSE) {
-  const bool AtLeastAtomic =
-      AccessProps.Ordering != AtomicOrdering::NotAtomic;
+  const bool AtLeastAtomic = AccessProps.Ordering != AtomicOrdering::NotAtomic;
 
   // If this is a load of Ptr, the loaded value is available.
   // (This is true even if the load is volatile or atomic, although



More information about the llvm-commits mailing list