[llvm] [IR] Add elementwise modifier to atomic stores (PR #210672)

Harrison Hao via llvm-commits llvm-commits at lists.llvm.org
Mon Aug 3 07:53:37 PDT 2026


https://github.com/harrisonGPU updated https://github.com/llvm/llvm-project/pull/210672

>From e328cfd8c20249f85838d56b0776380ea57cc30c Mon Sep 17 00:00:00 2001
From: Harrison Hao <tsworld1314 at gmail.com>
Date: Mon, 20 Jul 2026 17:35:22 +0800
Subject: [PATCH 1/4] [IR] Add elementwise modifier to atomic stores

---
 llvm/docs/LangRef.md                          |  29 +++--
 llvm/include/llvm/IR/Instructions.h           |  17 ++-
 llvm/lib/AsmParser/LLParser.cpp               |  30 ++++-
 llvm/lib/Bitcode/Reader/BitcodeReader.cpp     |  14 ++-
 llvm/lib/Bitcode/Writer/BitcodeWriter.cpp     |  20 ++--
 llvm/lib/IR/AsmWriter.cpp                     |   4 +-
 llvm/lib/IR/Instruction.cpp                   |   1 +
 llvm/lib/IR/Instructions.cpp                  |   8 +-
 llvm/lib/IR/Verifier.cpp                      |  25 +++-
 .../Transforms/Utils/FunctionComparator.cpp   |   3 +
 llvm/test/Assembler/atomic.ll                 |   4 +
 .../invalid-load-store-atomic-elementwise.ll  |  56 +++++++++
 .../Bitcode/atomic-load-store-elementwise.ll  |  14 +++
 llvm/test/Bitcode/compatibility.ll            |   6 +
 llvm/unittests/IR/VerifierTest.cpp            | 111 ++++++++++++++++++
 15 files changed, 305 insertions(+), 37 deletions(-)

diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index 1fc6ea1a7ed9d..84f7f11e27b2c 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -11986,7 +11986,7 @@ store i32 3, ptr %ptr                           ; yields void
 
 ```
 store [volatile] <ty> <value>, ptr <pointer>[, align <alignment>][, !nontemporal !<nontemp_node>][, !invariant.group !<empty_node>]        ; yields void
-store atomic [volatile] <ty> <value>, ptr <pointer> [syncscope("<target-scope>")] <ordering>, align <alignment> [, !invariant.group !<empty_node>] ; yields void
+store atomic [volatile] [elementwise] <ty> <value>, ptr <pointer> [syncscope("<target-scope>")] <ordering>, align <alignment> [, !invariant.group !<empty_node>] ; yields void
 !<nontemp_node> = !{ i32 1 }
 !<empty_node> = !{}
 ```
@@ -12004,16 +12004,25 @@ operand. If the `store` is marked as `volatile`, then the optimizer is not
 allowed to modify the number or order of execution of this `store` with other
 {ref}`volatile operations <volatile>`.  Only values of {ref}`first class <t_firstclass>` types of known size (i.e., not containing an {ref}`opaque structural type <t_opaque>`) can be stored.
 
-If the `store` is marked as `atomic`, it takes an extra {ref}`ordering <ordering>` and optional `syncscope("<target-scope>")` argument. The
-`acquire` and `acq_rel` orderings aren't valid on `store` instructions.
-Atomic loads produce {ref}`defined <memmodel>` results when they may see
-multiple atomic stores. The type of the pointee must be an integer, pointer,
+If the `store` is marked as `atomic`, it takes an extra
+{ref}`ordering <ordering>` and optional `syncscope("<target-scope>")`
+argument. The `acquire` and `acq_rel` orderings are not valid on `store`
+instructions. The type of the stored value must be an integer, pointer,
 floating-point, or vector type whose bit width is a power of two greater than
-or equal to eight. `align` must be
-explicitly specified on atomic stores. Note: if the alignment is not greater or
-equal to the size of the `<value>` type, the atomic operation is likely to
-require a lock and have poor performance. `!nontemporal` does not have any
-defined semantics for atomic stores.
+or equal to eight.
+
+If the `elementwise` modifier is present, the instruction has
+{ref}`elementwise atomic semantics <elementwise-atomics>`. The stored value
+must have a fixed vector type whose total bit width is a power of two greater
+than or equal to eight, and whose element type is supported by scalar atomic
+stores.
+
+`align` must be explicitly specified on atomic stores, and is otherwise
+optional on non-atomic stores. Note: if the alignment is not greater than or
+equal to the size of the `<value>` type, or the element type for an
+`elementwise` store, the atomic operation is likely to require a lock and have
+poor performance. `!nontemporal` does not have any defined semantics for
+atomic stores.
 
 The optional constant `align` argument specifies the alignment of the
 operation (that is, the alignment of the memory address). It is the
diff --git a/llvm/include/llvm/IR/Instructions.h b/llvm/include/llvm/IR/Instructions.h
index e0b26c62d7854..0475798b9463d 100644
--- a/llvm/include/llvm/IR/Instructions.h
+++ b/llvm/include/llvm/IR/Instructions.h
@@ -332,9 +332,10 @@ class StoreInst : public Instruction {
   using VolatileField = BoolBitfieldElementT<0>;
   using AlignmentField = AlignmentBitfieldElementT<VolatileField::NextBit>;
   using OrderingField = AtomicOrderingBitfieldElementT<AlignmentField::NextBit>;
-  static_assert(
-      Bitfield::areContiguous<VolatileField, AlignmentField, OrderingField>(),
-      "Bitfields must be contiguous");
+  using ElementWiseField = BoolBitfieldElementT<OrderingField::NextBit>;
+  static_assert(Bitfield::areContiguous<VolatileField, AlignmentField,
+                                        OrderingField, ElementWiseField>(),
+                "Bitfields must be contiguous");
 
   void AssertOK();
 
@@ -370,6 +371,12 @@ class StoreInst : public Instruction {
   /// Specify whether this is a volatile store or not.
   void setVolatile(bool V) { setSubclassData<VolatileField>(V); }
 
+  /// Return true if this is an elementwise atomic store.
+  bool isElementwise() const { return getSubclassData<ElementWiseField>(); }
+
+  /// Specify whether this is an elementwise atomic store or not.
+  void setElementwise(bool V) { setSubclassData<ElementWiseField>(V); }
+
   /// Transparently provide more efficient getOperand methods.
   DECLARE_TRANSPARENT_OPERAND_ACCESSORS(Value);
 
@@ -412,7 +419,8 @@ class StoreInst : public Instruction {
 
   /// Returns the properties of this store instruction.
   LoadStoreInstProperties getProperties() const {
-    return {isVolatile(), getAlign(), getOrdering(), getSyncScopeID()};
+    return {isVolatile(), getAlign(), getOrdering(), getSyncScopeID(),
+            isElementwise()};
   }
 
   /// Sets the properties of this store instruction.
@@ -421,6 +429,7 @@ class StoreInst : public Instruction {
     setAlignment(Props.Alignment);
     setOrdering(Props.Ordering);
     setSyncScopeID(Props.SSID);
+    setElementwise(Props.IsElementwise);
   }
 
   bool isSimple() const { return !isAtomic() && !isVolatile(); }
diff --git a/llvm/lib/AsmParser/LLParser.cpp b/llvm/lib/AsmParser/LLParser.cpp
index edff818b3b152..93a79a7035e6f 100644
--- a/llvm/lib/AsmParser/LLParser.cpp
+++ b/llvm/lib/AsmParser/LLParser.cpp
@@ -9033,10 +9033,11 @@ int LLParser::parseLoad(Instruction *&Inst, PerFunctionState &PFS) {
 /// parseStore
 
 ///   ::= 'store' 'volatile'? TypeAndValue ',' TypeAndValue (',' 'align' i32)?
-///   ::= 'store' 'atomic' 'volatile'? TypeAndValue ',' TypeAndValue
-///       'singlethread'? AtomicOrdering (',' 'align' i32)?
+///   ::= 'store' 'atomic' 'volatile'? 'elementwise'? TypeAndValue ','
+///       TypeAndValue 'singlethread'? AtomicOrdering (',' 'align' i32)?
 int LLParser::parseStore(Instruction *&Inst, PerFunctionState &PFS) {
-  Value *Val, *Ptr; LocTy Loc, PtrLoc;
+  Value *Val, *Ptr;
+  LocTy Loc, PtrLoc;
   MaybeAlign Alignment;
   bool AteExtraComma = false;
   bool isAtomic = false;
@@ -9054,6 +9055,12 @@ int LLParser::parseStore(Instruction *&Inst, PerFunctionState &PFS) {
     Lex.Lex();
   }
 
+  bool IsElementwise = false;
+  if (Lex.getKind() == lltok::kw_elementwise) {
+    IsElementwise = true;
+    Lex.Lex();
+  }
+
   if (parseTypeAndValue(Val, Loc, PFS) ||
       parseToken(lltok::comma, "expected ',' after store operand") ||
       parseTypeAndValue(Ptr, PtrLoc, PFS) ||
@@ -9070,13 +9077,28 @@ int LLParser::parseStore(Instruction *&Inst, PerFunctionState &PFS) {
   if (Ordering == AtomicOrdering::Acquire ||
       Ordering == AtomicOrdering::AcquireRelease)
     return error(Loc, "atomic store cannot use Acquire ordering");
+
+  if (IsElementwise && !isAtomic)
+    return error(Loc, "elementwise store must be atomic");
+
+  if (IsElementwise && !isa<FixedVectorType>(Val->getType()))
+    return error(
+        Loc, "atomic elementwise store operand must have fixed vector type");
+
+  if (IsElementwise && Ordering == AtomicOrdering::SequentiallyConsistent)
+    return error(Loc,
+                 "atomic elementwise store cannot be sequentially consistent");
+
   SmallPtrSet<Type *, 4> Visited;
   if (!Alignment && !Val->getType()->isSized(&Visited))
     return error(Loc, "storing unsized types is not allowed");
   if (!Alignment)
     Alignment = M->getDataLayout().getABITypeAlign(Val->getType());
 
-  Inst = new StoreInst(Val, Ptr, isVolatile, *Alignment, Ordering, SSID);
+  Inst = new StoreInst(Val, Ptr,
+                       LoadStoreInstProperties{isVolatile, *Alignment, Ordering,
+                                               SSID, IsElementwise},
+                       /*InsertBefore=*/nullptr);
   return AteExtraComma ? InstExtraComma : InstNormal;
 }
 
diff --git a/llvm/lib/Bitcode/Reader/BitcodeReader.cpp b/llvm/lib/Bitcode/Reader/BitcodeReader.cpp
index ac61ede6395af..d7d5c90a980a3 100644
--- a/llvm/lib/Bitcode/Reader/BitcodeReader.cpp
+++ b/llvm/lib/Bitcode/Reader/BitcodeReader.cpp
@@ -6541,7 +6541,8 @@ Error BitcodeReader::parseFunctionBody(Function *F) {
     }
     case bitc::FUNC_CODE_INST_STOREATOMIC:
     case bitc::FUNC_CODE_INST_STOREATOMIC_OLD: {
-      // STOREATOMIC: [ptrty, ptr, val, align, vol, ordering, ssid]
+      // STOREATOMIC: [ptrty, ptr, val, align, vol, ordering, ssid,
+      // elementwise?]
       unsigned OpNum = 0;
       Value *Val, *Ptr;
       unsigned PtrTypeID, ValTypeID;
@@ -6558,7 +6559,7 @@ Error BitcodeReader::parseFunctionBody(Function *F) {
           return error("Invalid store atomic record");
       }
 
-      if (OpNum + 4 != Record.size())
+      if (OpNum + 4 != Record.size() && OpNum + 5 != Record.size())
         return error("Invalid store atomic record");
 
       if (Error Err = typeCheckLoadStoreInst(Val->getType(), Ptr->getType()))
@@ -6577,7 +6578,14 @@ Error BitcodeReader::parseFunctionBody(Function *F) {
         return Err;
       if (!Align)
         return error("Alignment missing from atomic store");
-      I = new StoreInst(Val, Ptr, Record[OpNum + 1], *Align, Ordering, SSID);
+
+      bool IsElementwise = Record.size() > OpNum + 4 && Record[OpNum + 4];
+
+      I = new StoreInst(
+          Val, Ptr,
+          LoadStoreInstProperties{/*IsVolatile=*/Record[OpNum + 1] != 0, *Align,
+                                  Ordering, SSID, IsElementwise},
+          /*InsertBefore=*/nullptr);
       InstructionList.push_back(I);
       break;
     }
diff --git a/llvm/lib/Bitcode/Writer/BitcodeWriter.cpp b/llvm/lib/Bitcode/Writer/BitcodeWriter.cpp
index 571336c217797..9fd39e2f901ec 100644
--- a/llvm/lib/Bitcode/Writer/BitcodeWriter.cpp
+++ b/llvm/lib/Bitcode/Writer/BitcodeWriter.cpp
@@ -3590,8 +3590,9 @@ void ModuleBitcodeWriter::writeInstruction(const Instruction &I,
     break;
   }
 
-  case Instruction::Store:
-    if (cast<StoreInst>(I).isAtomic()) {
+  case Instruction::Store: {
+    const auto &SI = cast<StoreInst>(I);
+    if (SI.isAtomic()) {
       Code = bitc::FUNC_CODE_INST_STOREATOMIC;
     } else {
       Code = bitc::FUNC_CODE_INST_STORE;
@@ -3601,14 +3602,17 @@ void ModuleBitcodeWriter::writeInstruction(const Instruction &I,
       AbbrevToUse = 0;
     if (pushValueAndType(I.getOperand(0), InstID, Vals)) // valty + val
       AbbrevToUse = 0;
-    Vals.push_back(getEncodedAlign(cast<StoreInst>(I).getAlign()));
-    Vals.push_back(cast<StoreInst>(I).isVolatile());
-    if (cast<StoreInst>(I).isAtomic()) {
-      Vals.push_back(getEncodedOrdering(cast<StoreInst>(I).getOrdering()));
-      Vals.push_back(
-          getEncodedSyncScopeID(cast<StoreInst>(I).getSyncScopeID()));
+    Vals.push_back(getEncodedAlign(SI.getAlign()));
+    Vals.push_back(SI.isVolatile());
+    if (SI.isAtomic()) {
+      Vals.push_back(getEncodedOrdering(SI.getOrdering()));
+      Vals.push_back(getEncodedSyncScopeID(SI.getSyncScopeID()));
+      if (SI.isElementwise())
+        Vals.push_back(1);
     }
     break;
+  }
+
   case Instruction::AtomicCmpXchg:
     Code = bitc::FUNC_CODE_INST_CMPXCHG;
     pushValueAndType(I.getOperand(0), InstID, Vals); // ptrty + ptr
diff --git a/llvm/lib/IR/AsmWriter.cpp b/llvm/lib/IR/AsmWriter.cpp
index cad4f17b0db91..2555da6cfd87f 100644
--- a/llvm/lib/IR/AsmWriter.cpp
+++ b/llvm/lib/IR/AsmWriter.cpp
@@ -4480,7 +4480,9 @@ void AssemblyWriter::printInstruction(const Instruction &I) {
       (isa<AtomicRMWInst>(I) && cast<AtomicRMWInst>(I).isVolatile()))
     Out << " volatile";
 
-  if (isa<LoadInst>(I) && cast<LoadInst>(I).isElementwise())
+  // Print the elementwise marker for atomic loads and stores.
+  if ((isa<LoadInst>(I) && cast<LoadInst>(I).isElementwise()) ||
+      (isa<StoreInst>(I) && cast<StoreInst>(I).isElementwise()))
     Out << " elementwise";
 
   // Print out optimization information.
diff --git a/llvm/lib/IR/Instruction.cpp b/llvm/lib/IR/Instruction.cpp
index 099a54e684c96..1ac5bd6b636d1 100644
--- a/llvm/lib/IR/Instruction.cpp
+++ b/llvm/lib/IR/Instruction.cpp
@@ -931,6 +931,7 @@ bool Instruction::hasSameSpecialState(const Instruction *I2,
            LI->getSyncScopeID() == cast<LoadInst>(I2)->getSyncScopeID();
   if (const StoreInst *SI = dyn_cast<StoreInst>(I1))
     return SI->isVolatile() == cast<StoreInst>(I2)->isVolatile() &&
+           SI->isElementwise() == cast<StoreInst>(I2)->isElementwise() &&
            (SI->getAlign() == cast<StoreInst>(I2)->getAlign() ||
             IgnoreAlignment) &&
            SI->getOrdering() == cast<StoreInst>(I2)->getOrdering() &&
diff --git a/llvm/lib/IR/Instructions.cpp b/llvm/lib/IR/Instructions.cpp
index 1f37ff69991d8..a43a28a5268c0 100644
--- a/llvm/lib/IR/Instructions.cpp
+++ b/llvm/lib/IR/Instructions.cpp
@@ -1406,7 +1406,9 @@ StoreInst::StoreInst(Value *Val, Value *Ptr,
                      const LoadStoreInstProperties &Props,
                      InsertPosition InsertBefore)
     : StoreInst(Val, Ptr, Props.IsVolatile, Props.Alignment, Props.Ordering,
-                Props.SSID, InsertBefore) {}
+                Props.SSID, InsertBefore) {
+  setElementwise(Props.IsElementwise);
+}
 
 StoreInst::StoreInst(Value *val, Value *addr, bool isVolatile, Align Align,
                      AtomicOrdering Order, SyncScope::ID SSID,
@@ -4454,8 +4456,8 @@ LoadInst *LoadInst::cloneImpl() const {
 }
 
 StoreInst *StoreInst::cloneImpl() const {
-  return new StoreInst(getOperand(0), getOperand(1), isVolatile(), getAlign(),
-                       getOrdering(), getSyncScopeID());
+  return new StoreInst(getOperand(0), getOperand(1), getProperties(),
+                       /*InsertBefore=*/nullptr);
 }
 
 AtomicCmpXchgInst *AtomicCmpXchgInst::cloneImpl() const {
diff --git a/llvm/lib/IR/Verifier.cpp b/llvm/lib/IR/Verifier.cpp
index af5cd296bc909..b5ecd335db856 100644
--- a/llvm/lib/IR/Verifier.cpp
+++ b/llvm/lib/IR/Verifier.cpp
@@ -4668,14 +4668,31 @@ void Verifier::visitStoreInst(StoreInst &SI) {
     Check(SI.getOrdering() != AtomicOrdering::Acquire &&
               SI.getOrdering() != AtomicOrdering::AcquireRelease,
           "Store cannot have Acquire ordering", &SI);
-    Check(ElTy->getScalarType()->isIntOrPtrTy() ||
-              ElTy->getScalarType()->isByteTy() ||
-              ElTy->getScalarType()->isFloatingPointTy(),
+
+    Type *ScalarTy = ElTy;
+    if (SI.isElementwise()) {
+      Check(SI.getOrdering() != AtomicOrdering::SequentiallyConsistent,
+            "atomic elementwise store cannot be sequentially consistent.", &SI);
+
+      auto *VecTy = dyn_cast<FixedVectorType>(ElTy);
+      Check(VecTy,
+            "atomic elementwise store operand must have fixed vector type!",
+            &SI, ElTy);
+      if (VecTy) {
+        checkAtomicMemAccessSize(ScalarTy, &SI);
+        ScalarTy = VecTy->getElementType();
+      }
+    }
+
+    Check(ScalarTy->getScalarType()->isIntOrPtrTy() ||
+              ScalarTy->getScalarType()->isByteTy() ||
+              ScalarTy->getScalarType()->isFloatingPointTy(),
           "atomic store operand must have integer, byte, pointer, floating "
           "point, or vector type!",
           ElTy, &SI);
-    checkAtomicMemAccessSize(ElTy, &SI);
+    checkAtomicMemAccessSize(ScalarTy, &SI);
   } else {
+    Check(!SI.isElementwise(), "non-atomic store cannot be elementwise", &SI);
     Check(SI.getSyncScopeID() == SyncScope::System,
           "Non-atomic store cannot have SynchronizationScope specified", &SI);
   }
diff --git a/llvm/lib/Transforms/Utils/FunctionComparator.cpp b/llvm/lib/Transforms/Utils/FunctionComparator.cpp
index c9cc62adc2604..05c24af6efb39 100644
--- a/llvm/lib/Transforms/Utils/FunctionComparator.cpp
+++ b/llvm/lib/Transforms/Utils/FunctionComparator.cpp
@@ -716,6 +716,9 @@ int FunctionComparator::cmpOperations(const Instruction *L,
     if (int Res =
             cmpNumbers(SI->isVolatile(), cast<StoreInst>(R)->isVolatile()))
       return Res;
+    if (int Res = cmpNumbers(SI->isElementwise(),
+                             cast<StoreInst>(R)->isElementwise()))
+      return Res;
     if (int Res = cmpAligns(SI->getAlign(), cast<StoreInst>(R)->getAlign()))
       return Res;
     if (int Res =
diff --git a/llvm/test/Assembler/atomic.ll b/llvm/test/Assembler/atomic.ll
index a1ac41f645b37..1c873b2310c2e 100644
--- a/llvm/test/Assembler/atomic.ll
+++ b/llvm/test/Assembler/atomic.ll
@@ -75,6 +75,10 @@ define void @f(ptr %x) {
   load atomic elementwise <2 x float>, ptr %x syncscope("agent") monotonic, align 4
   ; CHECK: load atomic volatile elementwise <2 x i32>, ptr %x monotonic, align 4
   load atomic volatile elementwise <2 x i32>, ptr %x monotonic, align 4
+  ; CHECK: store atomic elementwise <2 x float> <float 3.000000e+00, float 4.000000e+00>, ptr %x syncscope("agent") monotonic, align 4
+  store atomic elementwise <2 x float> <float 3.0, float 4.0>, ptr %x syncscope("agent") monotonic, align 4
+  ; CHECK: store atomic volatile elementwise <2 x i32> <i32 3, i32 4>, ptr %x monotonic, align 4
+  store atomic volatile elementwise <2 x i32> <i32 3, i32 4>, ptr %x monotonic, align 4
 
   ; CHECK: fence syncscope("singlethread") release
   fence syncscope("singlethread") release
diff --git a/llvm/test/Assembler/invalid-load-store-atomic-elementwise.ll b/llvm/test/Assembler/invalid-load-store-atomic-elementwise.ll
index f51ae512a2207..f4339501e9bb3 100644
--- a/llvm/test/Assembler/invalid-load-store-atomic-elementwise.ll
+++ b/llvm/test/Assembler/invalid-load-store-atomic-elementwise.ll
@@ -6,6 +6,13 @@
 ; RUN: not llvm-as -disable-output %t/load-non-byte.ll 2>&1 | FileCheck %t/load-non-byte.ll
 ; RUN: not llvm-as -disable-output %t/load-non-byte-element.ll 2>&1 | FileCheck %t/load-non-byte-element.ll
 ; RUN: not llvm-as -disable-output %t/load-seq-cst.ll 2>&1 | FileCheck %t/load-seq-cst.ll
+; RUN: not llvm-as -disable-output %t/store-non-atomic.ll 2>&1 | FileCheck %t/store-non-atomic.ll
+; RUN: not llvm-as -disable-output %t/store-scalar.ll 2>&1 | FileCheck %t/store-scalar.ll
+; RUN: not llvm-as -disable-output %t/store-scalable.ll 2>&1 | FileCheck %t/store-scalable.ll
+; RUN: not llvm-as -disable-output %t/store-odd-sized.ll 2>&1 | FileCheck %t/store-odd-sized.ll
+; RUN: not llvm-as -disable-output %t/store-non-byte.ll 2>&1 | FileCheck %t/store-non-byte.ll
+; RUN: not llvm-as -disable-output %t/store-non-byte-element.ll 2>&1 | FileCheck %t/store-non-byte-element.ll
+; RUN: not llvm-as -disable-output %t/store-seq-cst.ll 2>&1 | FileCheck %t/store-seq-cst.ll
 
 ;--- load-non-atomic.ll
 ; CHECK: elementwise load must be atomic
@@ -55,3 +62,52 @@ define <4 x i32> @bad_seq_cst(ptr %p) {
   %v = load atomic elementwise <4 x i32>, ptr %p seq_cst, align 4
   ret <4 x i32> %v
 }
+
+;--- store-non-atomic.ll
+; CHECK: elementwise store must be atomic
+define void @bad_non_atomic_store(ptr %p, <2 x float> %v) {
+  store elementwise <2 x float> %v, ptr %p, align 4
+  ret void
+}
+
+;--- store-scalar.ll
+; CHECK: atomic elementwise store operand must have fixed vector type
+define void @bad_scalar_store(ptr %p, float %v) {
+  store atomic elementwise float %v, ptr %p monotonic, align 4
+  ret void
+}
+
+;--- store-scalable.ll
+; CHECK: atomic elementwise store operand must have fixed vector type
+define void @bad_scalable_store(ptr %p, <vscale x 2 x i32> %v) {
+  store atomic elementwise <vscale x 2 x i32> %v, ptr %p monotonic, align 4
+  ret void
+}
+
+;--- store-odd-sized.ll
+; CHECK: atomic memory access' operand must have a power-of-two size
+define void @bad_odd_sized_vector_store(ptr %p, <5 x i32> %v) {
+  store atomic elementwise <5 x i32> %v, ptr %p monotonic, align 4
+  ret void
+}
+
+;--- store-non-byte.ll
+; CHECK: atomic memory access' size must be byte-sized
+define void @bad_non_byte_store(ptr %p, <4 x i1> %v) {
+  store atomic elementwise <4 x i1> %v, ptr %p monotonic, align 4
+  ret void
+}
+
+;--- store-non-byte-element.ll
+; CHECK: atomic memory access' size must be byte-sized
+define void @bad_non_byte_element_store(ptr %p, <8 x i1> %v) {
+  store atomic elementwise <8 x i1> %v, ptr %p monotonic, align 1
+  ret void
+}
+
+;--- store-seq-cst.ll
+; CHECK: atomic elementwise store cannot be sequentially consistent
+define void @bad_store_seq_cst(ptr %p, <4 x i32> %v) {
+  store atomic elementwise <4 x i32> %v, ptr %p seq_cst, align 4
+  ret void
+}
diff --git a/llvm/test/Bitcode/atomic-load-store-elementwise.ll b/llvm/test/Bitcode/atomic-load-store-elementwise.ll
index 8426ad16f371e..2322e835bf6ce 100644
--- a/llvm/test/Bitcode/atomic-load-store-elementwise.ll
+++ b/llvm/test/Bitcode/atomic-load-store-elementwise.ll
@@ -14,3 +14,17 @@ define <4 x i32> @load_elem_i32_volatile(ptr %p) {
   %v = load atomic volatile elementwise <4 x i32>, ptr %p acquire, align 4
   ret <4 x i32> %v
 }
+
+define void @store_elem_f32(ptr %p, <2 x float> %v) {
+; CHECK-LABEL: @store_elem_f32(
+; CHECK: store atomic elementwise <2 x float> %v, ptr %p syncscope("agent") monotonic, align 4
+  store atomic elementwise <2 x float> %v, ptr %p syncscope("agent") monotonic, align 4
+  ret void
+}
+
+define void @store_elem_i32_volatile(ptr %p, <4 x i32> %v) {
+; CHECK-LABEL: @store_elem_i32_volatile(
+; CHECK: store atomic volatile elementwise <4 x i32> %v, ptr %p monotonic, align 4
+  store atomic volatile elementwise <4 x i32> %v, ptr %p monotonic, align 4
+  ret void
+}
diff --git a/llvm/test/Bitcode/compatibility.ll b/llvm/test/Bitcode/compatibility.ll
index 2500b6fba06a5..768fa6d6ec735 100644
--- a/llvm/test/Bitcode/compatibility.ll
+++ b/llvm/test/Bitcode/compatibility.ll
@@ -1044,6 +1044,12 @@ define void @elementwise_atomics(ptr %word, <4 x i32> %ival, <4 x float> %fval)
 ; CHECK: %load.elementwise.volatile = load atomic volatile elementwise <4 x float>, ptr %word acquire, align 4
   %load.elementwise.volatile = load atomic volatile elementwise <4 x float>, ptr %word acquire, align 4
 
+; CHECK: store atomic elementwise <4 x i32> <i32 1, i32 2, i32 3, i32 4>, ptr %word monotonic, align 4
+  store atomic elementwise <4 x i32> <i32 1, i32 2, i32 3, i32 4>, ptr %word monotonic, align 4
+
+; CHECK: store atomic volatile elementwise <4 x float> <float 1.000000e+00, float 2.000000e+00, float 3.000000e+00, float 4.000000e+00>, ptr %word monotonic, align 4
+  store atomic volatile elementwise <4 x float> <float 1.0, float 2.0, float 3.0, float 4.0>, ptr %word monotonic, align 4
+
   ret void
 }
 
diff --git a/llvm/unittests/IR/VerifierTest.cpp b/llvm/unittests/IR/VerifierTest.cpp
index 755d9dabff9cb..98d204c851951 100644
--- a/llvm/unittests/IR/VerifierTest.cpp
+++ b/llvm/unittests/IR/VerifierTest.cpp
@@ -711,6 +711,117 @@ TEST(VerifierTest, ElementwiseLoadOddSizedVector) {
       << Error;
 }
 
+TEST(VerifierTest, ElementwiseStoreNonAtomic) {
+  LLVMContext C;
+  Module M("M", C);
+  FunctionType *FTy = FunctionType::get(Type::getVoidTy(C), /*isVarArg=*/false);
+  Function *F = Function::Create(FTy, Function::ExternalLinkage, "foo", M);
+  BasicBlock *Entry = BasicBlock::Create(C, "entry", F);
+  Value *Ptr = PoisonValue::get(PointerType::get(C, 0));
+
+  Type *I32Ty = Type::getInt32Ty(C);
+  Constant *Value = ConstantVector::getSplat(ElementCount::getFixed(4),
+                                             ConstantInt::get(I32Ty, 0));
+
+  new StoreInst(Value, Ptr,
+                LoadStoreInstProperties{/*IsVolatile=*/false, Align(4),
+                                        AtomicOrdering::NotAtomic,
+                                        SyncScope::System,
+                                        /*IsElementwise=*/true},
+                Entry);
+  ReturnInst::Create(C, Entry);
+
+  std::string Error;
+  raw_string_ostream ErrorOS(Error);
+  EXPECT_TRUE(verifyFunction(*F, &ErrorOS));
+  EXPECT_TRUE(
+      StringRef(Error).starts_with("non-atomic store cannot be elementwise"))
+      << Error;
+}
+
+TEST(VerifierTest, ElementwiseStoreScalar) {
+  LLVMContext C;
+  Module M("M", C);
+  FunctionType *FTy = FunctionType::get(Type::getVoidTy(C), /*isVarArg=*/false);
+  Function *F = Function::Create(FTy, Function::ExternalLinkage, "foo", M);
+  BasicBlock *Entry = BasicBlock::Create(C, "entry", F);
+  Value *Ptr = PoisonValue::get(PointerType::get(C, 0));
+
+  Type *I32Ty = Type::getInt32Ty(C);
+  Constant *Value = ConstantInt::get(I32Ty, 0);
+
+  new StoreInst(Value, Ptr,
+                LoadStoreInstProperties{/*IsVolatile=*/false, Align(4),
+                                        AtomicOrdering::Monotonic,
+                                        SyncScope::System,
+                                        /*IsElementwise=*/true},
+                Entry);
+  ReturnInst::Create(C, Entry);
+
+  std::string Error;
+  raw_string_ostream ErrorOS(Error);
+  EXPECT_TRUE(verifyFunction(*F, &ErrorOS));
+  EXPECT_TRUE(StringRef(Error).starts_with(
+      "atomic elementwise store operand must have fixed vector type!"))
+      << Error;
+}
+
+TEST(VerifierTest, ElementwiseStoreOddSizedVector) {
+  LLVMContext C;
+  Module M("M", C);
+  FunctionType *FTy = FunctionType::get(Type::getVoidTy(C), /*isVarArg=*/false);
+  Function *F = Function::Create(FTy, Function::ExternalLinkage, "foo", M);
+  BasicBlock *Entry = BasicBlock::Create(C, "entry", F);
+  Value *Ptr = PoisonValue::get(PointerType::get(C, 0));
+
+  Type *I32Ty = Type::getInt32Ty(C);
+  Constant *Value = ConstantVector::getSplat(ElementCount::getFixed(5),
+                                             ConstantInt::get(I32Ty, 0));
+
+  new StoreInst(Value, Ptr,
+                LoadStoreInstProperties{/*IsVolatile=*/false, Align(4),
+                                        AtomicOrdering::Monotonic,
+                                        SyncScope::System,
+                                        /*IsElementwise=*/true},
+                Entry);
+  ReturnInst::Create(C, Entry);
+
+  std::string Error;
+  raw_string_ostream ErrorOS(Error);
+  EXPECT_TRUE(verifyFunction(*F, &ErrorOS));
+  EXPECT_TRUE(StringRef(Error).starts_with(
+      "atomic memory access' operand must have a power-of-two size"))
+      << Error;
+}
+
+TEST(VerifierTest, ElementwiseStoreSequentiallyConsistent) {
+  LLVMContext C;
+  Module M("M", C);
+  FunctionType *FTy = FunctionType::get(Type::getVoidTy(C), false);
+  Function *F = Function::Create(FTy, Function::ExternalLinkage, "foo", M);
+  BasicBlock *Entry = BasicBlock::Create(C, "entry", F);
+  Value *Ptr = PoisonValue::get(PointerType::get(C, 0));
+
+  Type *I32Ty = Type::getInt32Ty(C);
+  Constant *Value = ConstantVector::getSplat(ElementCount::getFixed(4),
+                                             ConstantInt::get(I32Ty, 0));
+
+  new StoreInst(Value, Ptr,
+                LoadStoreInstProperties{/*IsVolatile=*/false, Align(4),
+                                        AtomicOrdering::SequentiallyConsistent,
+                                        SyncScope::System,
+                                        /*IsElementwise=*/true},
+                Entry);
+  ReturnInst::Create(C, Entry);
+
+  std::string Error;
+  raw_string_ostream ErrorOS(Error);
+  EXPECT_TRUE(verifyFunction(*F, &ErrorOS));
+  EXPECT_TRUE(StringRef(Error).starts_with(
+      "atomic elementwise store cannot be sequentially consistent."))
+      << Error;
+}
+
 TEST(VerifierTest, GetElementPtrInst) {
   LLVMContext C;
   Module M("M", C);

>From 7482a61b9ccc2a69c6210b2cba15f70a7642c87e Mon Sep 17 00:00:00 2001
From: Harrison Hao <tsworld1314 at gmail.com>
Date: Thu, 23 Jul 2026 15:26:23 +0800
Subject: [PATCH 2/4] Add mergeFunc elementwise test.

---
 .../MergeFunc/atomic-elementwise.ll           | 20 +++++++++++++++++++
 1 file changed, 20 insertions(+)

diff --git a/llvm/test/Transforms/MergeFunc/atomic-elementwise.ll b/llvm/test/Transforms/MergeFunc/atomic-elementwise.ll
index 9f756c3aaa25b..ad736bb6ebf01 100644
--- a/llvm/test/Transforms/MergeFunc/atomic-elementwise.ll
+++ b/llvm/test/Transforms/MergeFunc/atomic-elementwise.ll
@@ -21,6 +21,26 @@ define internal <2 x i32> @elementwise_load(ptr %p) {
   ret <2 x i32> %value
 }
 
+define internal void @whole_vector_store(ptr %p, <2 x i32> %value) {
+; CHECK-LABEL: define internal void @whole_vector_store(
+; CHECK-SAME: ptr [[P:%.*]], <2 x i32> [[VALUE:%.*]]) {
+; CHECK-NEXT:    store atomic <2 x i32> [[VALUE]], ptr [[P]] monotonic, align 4
+; CHECK-NEXT:    ret void
+;
+  store atomic <2 x i32> %value, ptr %p monotonic, align 4
+  ret void
+}
+
+define internal void @elementwise_store(ptr %p, <2 x i32> %value) {
+; CHECK-LABEL: define internal void @elementwise_store(
+; CHECK-SAME: ptr [[P:%.*]], <2 x i32> [[VALUE:%.*]]) {
+; CHECK-NEXT:    store atomic elementwise <2 x i32> [[VALUE]], ptr [[P]] monotonic, align 4
+; CHECK-NEXT:    ret void
+;
+  store atomic elementwise <2 x i32> %value, ptr %p monotonic, align 4
+  ret void
+}
+
 define internal <2 x i32> @whole_vector_atomicrmw(ptr %p, <2 x i32> %value) {
 ; CHECK-LABEL: define internal <2 x i32> @whole_vector_atomicrmw(
 ; CHECK-SAME: ptr [[P:%.*]], <2 x i32> [[VALUE:%.*]]) {

>From 678bb0c5132a67fa175f6575ddd08f26f443b5fe Mon Sep 17 00:00:00 2001
From: Harrison Hao <tsworld1314 at gmail.com>
Date: Sat, 25 Jul 2026 11:55:03 +0800
Subject: [PATCH 3/4] Update doc

---
 llvm/docs/LangRef.md | 10 +++++-----
 1 file changed, 5 insertions(+), 5 deletions(-)

diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index 84f7f11e27b2c..1a1e83fb2cd94 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -4208,11 +4208,11 @@ monotonic modification order with other operations that are not marked
 ### Elementwise Atomic Operations
 
 Certain atomic instructions, such as {ref}`atomicrmw <i_atomicrmw>`,
-and {ref}`atomic load <i_load>`, may be marked `elementwise`. The access type
-must then be a fixed vector type whose total bit width is a power of two and
-whose element type is supported by the corresponding scalar atomic instruction.
-The {ref}`ordering <ordering>` of an `elementwise` instruction may not be
-`seq_cst`.
+{ref}`atomic load <i_load>`, and {ref}`atomic store <i_store>`, may be marked
+`elementwise`. The access type must then be a fixed vector type whose total bit
+width is a power of two and whose element type is supported by the corresponding
+scalar atomic instruction. The {ref}`ordering <ordering>` of an `elementwise`
+instruction may not be `seq_cst`.
 
 An `elementwise` atomic instruction behaves as if it were expanded into one
 scalar version of that instruction for each vector element. Each resulting

>From e7b0fbbf79f2393bc299b65f78bb8471eed7d6ab Mon Sep 17 00:00:00 2001
From: Harrison Hao <tsworld1314 at gmail.com>
Date: Mon, 3 Aug 2026 22:52:44 +0800
Subject: [PATCH 4/4] Update for comments.

---
 llvm/docs/LangRef.md                          | 22 +++++++++----------
 .../InstCombineLoadStoreAlloca.cpp            |  3 +++
 .../lib/Transforms/Scalar/LowerAtomicPass.cpp |  1 +
 llvm/test/Transforms/InstCombine/atomic.ll    | 11 ++++++++++
 .../atomic-load-store-elementwise.ll          | 10 +++++++++
 5 files changed, 36 insertions(+), 11 deletions(-)

diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index 1a1e83fb2cd94..aa102935ae170 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -12005,17 +12005,17 @@ allowed to modify the number or order of execution of this `store` with other
 {ref}`volatile operations <volatile>`.  Only values of {ref}`first class <t_firstclass>` types of known size (i.e., not containing an {ref}`opaque structural type <t_opaque>`) can be stored.
 
 If the `store` is marked as `atomic`, it takes an extra
-{ref}`ordering <ordering>` and optional `syncscope("<target-scope>")`
-argument. The `acquire` and `acq_rel` orderings are not valid on `store`
-instructions. The type of the stored value must be an integer, pointer,
-floating-point, or vector type whose bit width is a power of two greater than
-or equal to eight.
-
-If the `elementwise` modifier is present, the instruction has
-{ref}`elementwise atomic semantics <elementwise-atomics>`. The stored value
-must have a fixed vector type whose total bit width is a power of two greater
-than or equal to eight, and whose element type is supported by scalar atomic
-stores.
+{ref}`ordering <ordering>`, an optional `syncscope("<target-scope>")`, and an
+optional {ref}`elementwise <elementwise-atomics>` argument. The `acquire` and
+`acq_rel` orderings are not valid on `store` instructions. Atomic loads produce
+{ref}`defined <memmodel>` results when they may see multiple atomic stores. The
+type of the pointee must be an integer, pointer, floating-point, or vector type
+whose bit width is a power of two greater than or equal to eight.
+
+If the `store` is marked `elementwise`, the instruction has
+{ref}`elementwise atomic semantics <elementwise-atomics>`. The stored type must
+be a fixed vector type whose total bit width is a power of two and whose
+element type is supported by scalar atomic stores.
 
 `align` must be explicitly specified on atomic stores, and is otherwise
 optional on non-atomic stores. Note: if the alignment is not greater than or
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp b/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp
index fe6c3d6c893ac..6f433d533db39 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineLoadStoreAlloca.cpp
@@ -1294,6 +1294,9 @@ static bool combineStoreToValueType(InstCombinerImpl &IC, StoreInst &SI) {
   if (!SI.isUnordered())
     return false;
 
+  if (SI.isElementwise())
+    return false;
+
   // swifterror values can't be bitcasted.
   if (SI.getPointerOperand()->isSwiftError())
     return false;
diff --git a/llvm/lib/Transforms/Scalar/LowerAtomicPass.cpp b/llvm/lib/Transforms/Scalar/LowerAtomicPass.cpp
index 297902181a00d..712655a1fcd95 100644
--- a/llvm/lib/Transforms/Scalar/LowerAtomicPass.cpp
+++ b/llvm/lib/Transforms/Scalar/LowerAtomicPass.cpp
@@ -34,6 +34,7 @@ static bool LowerLoadInst(LoadInst *LI) {
 
 static bool LowerStoreInst(StoreInst *SI) {
   SI->setAtomic(AtomicOrdering::NotAtomic);
+  SI->setElementwise(false);
   return true;
 }
 
diff --git a/llvm/test/Transforms/InstCombine/atomic.ll b/llvm/test/Transforms/InstCombine/atomic.ll
index 02f7929951035..062aa3db34759 100644
--- a/llvm/test/Transforms/InstCombine/atomic.ll
+++ b/llvm/test/Transforms/InstCombine/atomic.ll
@@ -463,4 +463,15 @@ define i64 @load_elementwise_bitcast(ptr %p) {
   ret i64 %r
 }
 
+define void @store_elementwise_bitcast(ptr %p, i64 %v) {
+; CHECK-LABEL: @store_elementwise_bitcast(
+; CHECK-NEXT:    [[R:%.*]] = bitcast i64 [[V:%.*]] to <2 x i32>
+; CHECK-NEXT:    store atomic elementwise <2 x i32> [[R]], ptr [[P:%.*]] unordered, align 8
+; CHECK-NEXT:    ret void
+;
+  %r = bitcast i64 %v to <2 x i32>
+  store atomic elementwise <2 x i32> %r, ptr %p unordered, align 8
+  ret void
+}
+
 attributes #0 = { null_pointer_is_valid }
diff --git a/llvm/test/Transforms/LowerAtomic/atomic-load-store-elementwise.ll b/llvm/test/Transforms/LowerAtomic/atomic-load-store-elementwise.ll
index c50a679a634f0..a39891e499382 100644
--- a/llvm/test/Transforms/LowerAtomic/atomic-load-store-elementwise.ll
+++ b/llvm/test/Transforms/LowerAtomic/atomic-load-store-elementwise.ll
@@ -10,3 +10,13 @@ define <2 x i32> @load_elementwise(ptr %p) {
   %v = load atomic elementwise <2 x i32>, ptr %p monotonic, align 4
   ret <2 x i32> %v
 }
+
+define void @store_elementwise(ptr %p, <2 x i32> %v) {
+; CHECK-LABEL: define void @store_elementwise(
+; CHECK-SAME: ptr [[P:%.*]], <2 x i32> [[V:%.*]]) {
+; CHECK-NEXT:    store <2 x i32> [[V]], ptr [[P]], align 4
+; CHECK-NEXT:    ret void
+;
+  store atomic elementwise <2 x i32> %v, ptr %p monotonic, align 4
+  ret void
+}



More information about the llvm-commits mailing list