[llvm-branch-commits] [clang] [llvm] [HLSL] Add float overload for `InterlockedExchange` (PR #222163)

Joshua Batista via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Wed Sep 23 12:54:16 PDT 2026


https://github.com/bob80905 updated https://github.com/llvm/llvm-project/pull/222163

>From 945e156c2ed79bc8a96cb2f095447482f4dcffb8 Mon Sep 17 00:00:00 2001
From: Joshua Batista <jbatista at microsoft.com>
Date: Fri, 4 Sep 2026 14:54:47 -0700
Subject: [PATCH 1/6] First attempt implementing float InterlockedExchange

---
 clang/include/clang/Basic/Builtins.td         |  3 +-
 .../clang/Basic/DiagnosticSemaKinds.td        |  2 +-
 clang/lib/CodeGen/CGHLSLBuiltins.cpp          |  9 ++++-
 clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp |  6 +++
 clang/lib/Sema/HLSLExternalSemaSource.cpp     | 14 +++++--
 clang/lib/Sema/SemaHLSL.cpp                   | 12 ++++--
 .../builtins/InterlockedExchange.hlsl         | 11 +++++
 ...ByteAddressBuffer-InterlockedExchange.hlsl | 16 ++++++++
 ...sBuffer-InterlockedExchangeFloat-sm60.hlsl | 40 +++++++++++++++++++
 .../BuiltIns/InterlockedExchange-errors.hlsl  | 35 +++++++++-------
 llvm/lib/Target/DirectX/DXILLegalizePass.cpp  | 29 ++++++++++++++
 .../lib/Target/DirectX/DXILResourceAccess.cpp | 20 ++++++++--
 .../DirectX/LegalizeAtomicExchangeFloat.ll    | 39 ++++++++++++++++++
 .../DirectX/ResourceAtomicExchangeFloat.ll    | 38 ++++++++++++++++++
 14 files changed, 246 insertions(+), 28 deletions(-)
 create mode 100644 clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchangeFloat-sm60.hlsl
 create mode 100644 llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll
 create mode 100644 llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll

diff --git a/clang/include/clang/Basic/Builtins.td b/clang/include/clang/Basic/Builtins.td
index b91c56444e17e3..19a678ae67a3b4 100644
--- a/clang/include/clang/Basic/Builtins.td
+++ b/clang/include/clang/Basic/Builtins.td
@@ -5561,7 +5561,8 @@ def HLSLInterlockedAnd : LangBuiltin<"HLSL_LANG"> {
 
 def HLSLInterlockedExchange : LangBuiltin<"HLSL_LANG"> {
   let Spellings = ["__builtin_hlsl_interlocked_exchange"];
-  let Attributes = [NoThrow];
+  // Prevent inadvertent float -> double arg promotion.
+  let Attributes = [NoThrow, CustomTypeChecking];
   let Prototype = "void (...)";
 }
 
diff --git a/clang/include/clang/Basic/DiagnosticSemaKinds.td b/clang/include/clang/Basic/DiagnosticSemaKinds.td
index 36a18473f4d4cc..43b3a48edf04ae 100644
--- a/clang/include/clang/Basic/DiagnosticSemaKinds.td
+++ b/clang/include/clang/Basic/DiagnosticSemaKinds.td
@@ -13409,7 +13409,7 @@ def err_builtin_invalid_arg_type: Error<
   // An 'or' if non-empty second and third components are combined
   "%plural{0:|:%plural{0:|:or }2}3"
   // Third component: floating-point types
-  "%select{|floating-point|16 or 32 bit floating-point}3"
+  "%select{|floating-point|16 or 32 bit floating-point|32 bit floating-point}3"
   // A space after a non-empty third component
   "%plural{0:|: }3"
   "%plural{[0,3]:type|:types}1 (was %4)">;
diff --git a/clang/lib/CodeGen/CGHLSLBuiltins.cpp b/clang/lib/CodeGen/CGHLSLBuiltins.cpp
index 84808e7f88c6ac..0ce891a27550b6 100644
--- a/clang/lib/CodeGen/CGHLSLBuiltins.cpp
+++ b/clang/lib/CodeGen/CGHLSLBuiltins.cpp
@@ -317,8 +317,13 @@ static Value *handleInterlockedOp(CodeGenFunction &CGF, const CallExpr *E,
   LValue DestLV = CGF.EmitLValue(E->getArg(0));
   Address DestAddr = DestLV.getAddress();
   Value *Val = CGF.EmitScalarExpr(E->getArg(1));
-  assert(E->getArg(1)->getType()->isIntegerType() &&
-         "Intrinsic InterlockedOp value operand must be an integer");
+  [[maybe_unused]] QualType ValTy = E->getArg(1)->getType();
+  if (Op == llvm::AtomicRMWInst::Xchg)
+    assert((ValTy->isIntegerType() || ValTy->isFloatingType()) &&
+           "InterlockedExchange value operand must be an integer or a float");
+  else
+    assert(ValTy->isIntegerType() &&
+           "Intrinsic InterlockedOp value operand must be an integer");
 
   // Scopeless atomics will default to CrossDevice, which is illegal in Vulkan.
   // Set the memory scope: Workgroup for groupshared, otherwise Device.
diff --git a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp
index 4922f67d0aec1b..91874dbca65b80 100644
--- a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp
+++ b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp
@@ -1765,6 +1765,12 @@ BuiltinTypeDeclBuilder::addByteAddressBufferInterlockedMethods() {
   addByteAddressBufferInterlockedMethod(
       "InterlockedExchange", AST.UnsignedIntTy,
       "__builtin_hlsl_interlocked_exchange", /*RequiresOriginalValue=*/true);
+  // The float exchange reuses the 32-bit integer DXIL operation, so it needs
+  // no capability bits and works from SM 6.0. ByteAddressBuffer carries no
+  // element type, so the method name states the type.
+  addByteAddressBufferInterlockedMethod("InterlockedExchangeFloat", AST.FloatTy,
+                                        "__builtin_hlsl_interlocked_exchange",
+                                        /*RequiresOriginalValue=*/true);
   addByteAddressBufferInterlockedMethod("InterlockedMax", AST.IntTy,
                                         "__builtin_hlsl_interlocked_max");
   addByteAddressBufferInterlockedMethod("InterlockedMax", AST.UnsignedIntTy,
diff --git a/clang/lib/Sema/HLSLExternalSemaSource.cpp b/clang/lib/Sema/HLSLExternalSemaSource.cpp
index bf3becac043b3a..a25d38e865335c 100644
--- a/clang/lib/Sema/HLSLExternalSemaSource.cpp
+++ b/clang/lib/Sema/HLSLExternalSemaSource.cpp
@@ -888,13 +888,18 @@ static void buildAtomicOverload(Sema &S, NamespaceDecl *NS, StringRef FuncName,
 // Synthesize the InterlockedFunc overload set: {int, uint, int64_t, uint64_t}
 // x {groupshared, device} x {2-arg, 3-arg}. Operations that always report the
 // previous value, such as InterlockedExchange, only get the 3-arg form.
+// InterlockedExchange also accepts float, which lowers to a bitwise exchange
+// of the 32-bit pattern.
 static void defineHLSLInterlockedFunc(Sema &S, NamespaceDecl *NS,
                                       StringRef FuncName, StringRef BuiltinName,
-                                      bool RequiresOriginalValue = false) {
+                                      bool RequiresOriginalValue = false,
+                                      bool SupportsFloat = false) {
   ASTContext &AST = S.getASTContext();
   // HLSL: int64_t == long, uint64_t == unsigned long (see hlsl_basic_types.h).
-  QualType Elems[] = {AST.IntTy, AST.UnsignedIntTy, AST.LongTy,
-                      AST.UnsignedLongTy};
+  SmallVector<QualType, 5> Elems = {AST.IntTy, AST.UnsignedIntTy, AST.LongTy,
+                                    AST.UnsignedLongTy};
+  if (SupportsFloat)
+    Elems.push_back(AST.FloatTy);
   LangAS AddrSpaces[] = {LangAS::hlsl_groupshared, LangAS::hlsl_device};
 
   for (QualType ElemTy : Elems)
@@ -914,7 +919,8 @@ void HLSLExternalSemaSource::defineHLSLAtomicIntrinsics() {
                             "__builtin_hlsl_interlocked_and");
   defineHLSLInterlockedFunc(*SemaPtr, HLSLNamespace, "InterlockedExchange",
                             "__builtin_hlsl_interlocked_exchange",
-                            /*RequiresOriginalValue=*/true);
+                            /*RequiresOriginalValue=*/true,
+                            /*SupportsFloat=*/true);
   defineHLSLInterlockedFunc(*SemaPtr, HLSLNamespace, "InterlockedMax",
                             "__builtin_hlsl_interlocked_max");
   defineHLSLInterlockedFunc(*SemaPtr, HLSLNamespace, "InterlockedMin",
diff --git a/clang/lib/Sema/SemaHLSL.cpp b/clang/lib/Sema/SemaHLSL.cpp
index 84bf12fc680734..dfaaacf41ebee0 100644
--- a/clang/lib/Sema/SemaHLSL.cpp
+++ b/clang/lib/Sema/SemaHLSL.cpp
@@ -4632,11 +4632,17 @@ bool SemaHLSL::CheckBuiltinFunctionCall(unsigned BuiltinID, CallExpr *TheCall) {
     }
 
     QualType DestTy = TheCall->getArg(0)->getType().getUnqualifiedType();
-    if (!DestTy->isIntegerType()) {
+    // InterlockedExchange also operates on float. DXIL lowers that as a
+    // bitwise exchange of the value's bit pattern, and DXC accepts 32-bit
+    // float only, so half and double are rejected.
+    const bool AllowsFloat =
+        BuiltinID == Builtin::BI__builtin_hlsl_interlocked_exchange;
+    if (!DestTy->isIntegerType() &&
+        !(AllowsFloat && DestTy->isSpecificBuiltinType(BuiltinType::Float))) {
       SemaRef.Diag(TheCall->getArg(0)->getBeginLoc(),
                    diag::err_builtin_invalid_arg_type)
-          << /*ordinal=*/1 << /*scalar*/ 1 << /*integer*/ 1 << /*no float*/ 0
-          << DestTy;
+          << /*ordinal=*/1 << /*scalar*/ 1 << /*integer*/ 1
+          << /*32 bit floating-point*/ (AllowsFloat ? 3 : 0) << DestTy;
       return true;
     }
 
diff --git a/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl b/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl
index 36240bf9f19ba1..5099a83ce42752 100644
--- a/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl
+++ b/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl
@@ -14,6 +14,7 @@ groupshared int  gs_i32;
 groupshared uint gs_u32;
 groupshared int64_t  gs_i64;
 groupshared uint64_t gs_u64;
+groupshared float gs_f32;
 
 // CHECK-LABEL: define {{.*}}void @{{.*}}test_int_3arg
 // CHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_i32{{.*}}, i32 %{{.*}} syncscope("workgroup") monotonic
@@ -42,3 +43,13 @@ export void test_int64_3arg(int64_t v, out int64_t orig) {
 export void test_uint64_3arg(uint64_t v, out uint64_t orig) {
   InterlockedExchange(gs_u64, v, orig);
 }
+
+// The float overload keeps the float type in the IR. DXIL converts it to an
+// i32 exchange later, and SPIR-V selects OpAtomicExchange directly.
+// CHECK-LABEL: define {{.*}}void @{{.*}}test_float_3arg
+// DXCHECK:  %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_f32{{.*}}, float %{{.*}} syncscope("workgroup") monotonic
+// SPVCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_f32{{.*}}, float %{{.*}} syncscope("workgroup") monotonic
+// CHECK:    store float %[[R]], ptr {{.*}}
+export void test_float_3arg(float v, out float orig) {
+  InterlockedExchange(gs_f32, v, orig);
+}
diff --git a/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl b/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl
index ce03bd0504053e..5185983e6b6359 100644
--- a/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl
+++ b/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl
@@ -38,3 +38,19 @@ export void test_bab_uint_3arg(uint off, uint v, out uint orig) {
 export void test_bab_uint64_3arg(uint off, uint64_t v, out uint64_t orig) {
   BAB.InterlockedExchange64(off, v, orig);
 }
+
+// ByteAddressBuffer holds no element type, so the float method carries the
+// type in its name. The value keeps the float type here; DXIL converts it to
+// an i32 exchange later.
+// CHECK-LABEL: define {{.*}}void @{{.*}}test_bab_float_3arg
+// DXCHECK:  %[[HANDLE:.*]] = load target("dx.RawBuffer", i8, 1, 0), ptr {{.*}}
+// DXCHECK:  %[[PTR:.*]] = call ptr @llvm.dx.resource.getpointer.p0.tdx.RawBuffer_i8_1_0t.i32(target("dx.RawBuffer", i8, 1, 0) %[[HANDLE]], i32 %{{.*}})
+// DXCHECK:  %[[R:.*]] = atomicrmw xchg ptr %[[PTR]], float %{{.*}} syncscope("device") monotonic
+// DXCHECK:  store float %[[R]], ptr {{.*}}
+// SPVCHECK: %[[HANDLE:.*]] = load target("spirv.VulkanBuffer", [0 x i8], 12, 1), ptr {{.*}}
+// SPVCHECK: %[[PTR:.*]] = call ptr addrspace(11) @llvm.spv.resource.getpointer.p11.tspirv.VulkanBuffer_a0i8_12_1t.i32(target("spirv.VulkanBuffer", [0 x i8], 12, 1) %[[HANDLE]], i32 %{{.*}})
+// SPVCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(11) %[[PTR]], float %{{.*}} syncscope("device") monotonic
+// SPVCHECK: store float %[[R]], ptr {{.*}}
+export void test_bab_float_3arg(uint off, float v, out float orig) {
+  BAB.InterlockedExchangeFloat(off, v, orig);
+}
diff --git a/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchangeFloat-sm60.hlsl b/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchangeFloat-sm60.hlsl
new file mode 100644
index 00000000000000..c4b1f34455f14f
--- /dev/null
+++ b/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchangeFloat-sm60.hlsl
@@ -0,0 +1,40 @@
+// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header \
+// RUN:   -triple dxil-pc-shadermodel6.0-library %s -fsyntax-only -verify \
+// RUN:   -verify-ignore-unexpected=warning
+
+// The float exchange reuses the 32-bit integer DXIL operation, so it needs no
+// capability bits and works from SM 6.0. The 64-bit exchange needs SM 6.6.
+// This file checks both halves, so it proves the two are gated differently.
+
+RWByteAddressBuffer BAB : register(u0);
+RasterizerOrderedByteAddressBuffer ROVB : register(u1);
+groupshared float gs_f32;
+groupshared int64_t gs_i64;
+
+void sm60_bab_float_ok(uint off, float v, out float orig) {
+  BAB.InterlockedExchangeFloat(off, v, orig);
+}
+
+void sm60_rovb_float_ok(uint off, float v, out float orig) {
+  ROVB.InterlockedExchangeFloat(off, v, orig);
+}
+
+void sm60_free_function_ok(float v) {
+  float orig;
+  InterlockedExchange(gs_f32, v, orig);
+}
+
+void sm60_direct_builtin_ok(float v) {
+  float orig;
+  __builtin_hlsl_interlocked_exchange(gs_f32, v, orig);
+}
+
+void sm60_no_bab_exchange64(uint off, uint64_t v, out uint64_t orig) {
+  BAB.InterlockedExchange64(off, v, orig);
+  // expected-error at -1 {{no member named 'InterlockedExchange64' in 'hlsl::RWByteAddressBuffer'}}
+}
+
+void sm60_no_direct_builtin_i64(int64_t v, out int64_t orig) {
+  __builtin_hlsl_interlocked_exchange(gs_i64, v, orig);
+  // expected-error at -1 {{'__builtin_hlsl_interlocked_exchange' requires shader model 6.6 or newer}}
+}
diff --git a/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl b/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl
index bf6a1757b42f2c..3d70256bcddef9 100644
--- a/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl
+++ b/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl
@@ -3,53 +3,54 @@
 // RUN:   -disable-llvm-passes -verify
 
 // InterlockedExchange is provided as a set of address-space-qualified
-// overloads (groupshared/device, {int,uint,int64_t,uint64_t}). It always
-// reports the previous value, so there is no 2-argument form.
+// overloads (groupshared/device, {int,uint,int64_t,uint64_t,float}). It always
+// reports the previous value, so there is no 2-argument form. Only 32-bit
+// float is accepted, so double has no overload.
 
 groupshared int gs_i32;
-groupshared float gs_f32;
+groupshared double gs_f64;
 struct S { int x; };
 groupshared S gs_s;
 
 void too_few() {
   InterlockedExchange(gs_i32); // expected-error{{no matching function for call to 'InterlockedExchange'}}
-  // expected-note@*:* 8 {{candidate function}}
+  // expected-note@*:* 10 {{candidate function}}
 }
 
 void missing_original_value(int v) {
   InterlockedExchange(gs_i32, v); // expected-error{{no matching function for call to 'InterlockedExchange'}}
-  // expected-note@*:* 8 {{candidate function}}
+  // expected-note@*:* 10 {{candidate function}}
 }
 
 void too_many(int v, int extra) {
   int orig;
   InterlockedExchange(gs_i32, v, orig, extra); // expected-error{{no matching function for call to 'InterlockedExchange'}}
-  // expected-note@*:* 8 {{candidate function}}
+  // expected-note@*:* 10 {{candidate function}}
 }
 
 void local_dest(int v) {
   int dest;
   int orig;
   InterlockedExchange(dest, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}}
-  // expected-note@*:* 8 {{candidate function}}
+  // expected-note@*:* 10 {{candidate function}}
 }
 
-void float_dest(float v) {
-  float orig;
-  InterlockedExchange(gs_f32, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}}
-  // expected-note@*:* 8 {{candidate function}}
+void double_dest(double v) {
+  double orig;
+  InterlockedExchange(gs_f64, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}}
+  // expected-note@*:* 10 {{candidate function}}
 }
 
 void struct_dest(int v) {
   int orig;
   InterlockedExchange(gs_s, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}}
-  // expected-note@*:* 8 {{candidate function}}
+  // expected-note@*:* 10 {{candidate function}}
 }
 
 void mismatched_orig_type(int v) {
   uint orig;
   InterlockedExchange(gs_i32, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}}
-  // expected-note@*:* 8 {{candidate function}}
+  // expected-note@*:* 10 {{candidate function}}
 }
 
 void direct_too_few() {
@@ -72,7 +73,13 @@ void direct_non_integer_dest() {
   S local_s;
   S orig;
   __builtin_hlsl_interlocked_exchange(local_s, 1, orig);
-  // expected-error at -1 {{1st argument must be a scalar integer type (was 'S')}}
+  // expected-error at -1 {{1st argument must be a scalar integer or 32 bit floating-point type (was 'S')}}
+}
+
+void direct_double_dest(double v) {
+  double orig;
+  __builtin_hlsl_interlocked_exchange(gs_f64, v, orig);
+  // expected-error at -1 {{1st argument must be a scalar integer or 32 bit floating-point type (was 'double')}}
 }
 
 void direct_nonlvalue_dest(int v) {
diff --git a/llvm/lib/Target/DirectX/DXILLegalizePass.cpp b/llvm/lib/Target/DirectX/DXILLegalizePass.cpp
index 9d4d7661ac9c92..c8037fbd84d0cc 100644
--- a/llvm/lib/Target/DirectX/DXILLegalizePass.cpp
+++ b/llvm/lib/Target/DirectX/DXILLegalizePass.cpp
@@ -356,6 +356,34 @@ static bool updateFnegToFsub(Instruction &I,
   return true;
 }
 
+// DXIL has no floating-point atomic operation. A float exchange only moves the
+// bit pattern, so exchange an integer of the same width instead. Opaque
+// pointers keep the pointer operand type-agnostic, so only the value and the
+// result need a cast. This matches what DXC emits for groupshared memory.
+static bool
+legalizeFloatAtomicExchange(Instruction &I,
+                            SmallVectorImpl<Instruction *> &ToRemove,
+                            DenseMap<Value *, Value *> &) {
+  auto *AI = dyn_cast<AtomicRMWInst>(&I);
+  if (!AI || AI->getOperation() != AtomicRMWInst::Xchg)
+    return false;
+
+  Type *ValTy = AI->getValOperand()->getType();
+  if (!ValTy->isFloatingPointTy())
+    return false;
+
+  IRBuilder<> Builder(AI);
+  Type *IntTy = Builder.getIntNTy(ValTy->getPrimitiveSizeInBits());
+  Value *Val = Builder.CreateBitCast(AI->getValOperand(), IntTy);
+  AtomicRMWInst *NewAI = Builder.CreateAtomicRMW(
+      AtomicRMWInst::Xchg, AI->getPointerOperand(), Val, AI->getAlign(),
+      AI->getOrdering(), AI->getSyncScopeID());
+  NewAI->copyMetadata(*AI);
+  AI->replaceAllUsesWith(Builder.CreateBitCast(NewAI, ValTy));
+  ToRemove.push_back(AI);
+  return true;
+}
+
 static bool
 resolveUnreachableSwitchDefault(Instruction &I,
                                 SmallVectorImpl<Instruction *> &ToRemove,
@@ -496,6 +524,7 @@ class DXILLegalizationPipeline {
     LegalizationPipeline[Stage1].push_back(fixI8UseChain);
     LegalizationPipeline[Stage1].push_back(legalizeFreeze);
     LegalizationPipeline[Stage1].push_back(updateFnegToFsub);
+    LegalizationPipeline[Stage1].push_back(legalizeFloatAtomicExchange);
     LegalizationPipeline[Stage1].push_back(
         downcastI64toI32InsertExtractElements);
     LegalizationPipeline[Stage2].push_back(legalizeScalarLoadStoreOnArrays);
diff --git a/llvm/lib/Target/DirectX/DXILResourceAccess.cpp b/llvm/lib/Target/DirectX/DXILResourceAccess.cpp
index a5cb99e17b2431..0dd5f69eff7cb9 100644
--- a/llvm/lib/Target/DirectX/DXILResourceAccess.cpp
+++ b/llvm/lib/Target/DirectX/DXILResourceAccess.cpp
@@ -383,17 +383,31 @@ static void emitAtomicBinOp(IRBuilder<> &Builder, AtomicRMWInst *AI,
     return;
   }
 
+  // DXIL has no floating-point atomic op. A float exchange only moves the bit
+  // pattern, so cast the value to an integer of the same width, exchange, and
+  // cast the result back. This matches what DXC emits.
+  Value *Val = AI->getValOperand();
+  Type *ValTy = Val->getType();
+  Type *OpTy = ValTy;
+  if (ValTy->isFloatingPointTy()) {
+    OpTy = Builder.getIntNTy(ValTy->getPrimitiveSizeInBits());
+    Val = Builder.CreateBitCast(Val, OpTy);
+  }
+
   SmallVector<Value *, 6> Args{
       Handle, Builder.getInt32(static_cast<uint32_t>(*BinOpCode))};
   append_range(Args, Coords);
   Args.append(3 - Coords.size(), PoisonValue::get(Builder.getInt32Ty()));
-  Args.push_back(AI->getValOperand());
+  Args.push_back(Val);
 
   // Emit the target-independent intrinsic; DXILOpLowering lowers it to the
   // DXIL `AtomicBinOp` op and handles the target-ext-typed handle cast via
   // its `createTmpHandleCast` bookkeeping.
-  Value *Result = Builder.CreateIntrinsic(
-      AI->getType(), Intrinsic::dx_resource_atomic_binop, Args);
+  Value *Result =
+      Builder.CreateIntrinsic(OpTy, Intrinsic::dx_resource_atomic_binop, Args);
+
+  if (OpTy != ValTy)
+    Result = Builder.CreateBitCast(Result, ValTy);
 
   AI->replaceAllUsesWith(Result);
 }
diff --git a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll
new file mode 100644
index 00000000000000..446c76a2b801db
--- /dev/null
+++ b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll
@@ -0,0 +1,39 @@
+; RUN: opt -S -dxil-legalize -mtriple=dxil-pc-shadermodel6.0-compute %s | FileCheck %s
+
+; DXIL has no floating-point atomic op. A float exchange on groupshared memory
+; only moves the bit pattern, so it becomes an i32 exchange with a bitcast on
+; the value and on the result. Opaque pointers leave the pointer operand
+; unchanged.
+
+target triple = "dxil-pc-shadermodel6.0-compute"
+
+ at gs = external addrspace(3) global float
+
+; CHECK-LABEL: define float @gs_xchg_float
+define float @gs_xchg_float(float %val) {
+  ; CHECK: [[CAST:%.*]] = bitcast float %val to i32
+  ; CHECK: [[OLD:%.*]] = atomicrmw xchg ptr addrspace(3) @gs, i32 [[CAST]] syncscope("workgroup") monotonic
+  ; CHECK: [[RES:%.*]] = bitcast i32 [[OLD]] to float
+  %old = atomicrmw xchg ptr addrspace(3) @gs, float %val syncscope("workgroup") monotonic
+  ; CHECK: ret float [[RES]]
+  ret float %old
+}
+
+; An integer exchange must pass through with no bitcast.
+; CHECK-LABEL: define i32 @gs_xchg_i32
+define i32 @gs_xchg_i32(ptr addrspace(3) %p, i32 %val) {
+  ; CHECK-NOT: bitcast
+  ; CHECK: atomicrmw xchg ptr addrspace(3) %p, i32 %val syncscope("workgroup") monotonic
+  %old = atomicrmw xchg ptr addrspace(3) %p, i32 %val syncscope("workgroup") monotonic
+  ret i32 %old
+}
+
+; Only exchange is rewritten. DXIL does not support float atomic add anywhere,
+; but fadd must not be silently turned into an integer add.
+; CHECK-LABEL: define float @gs_fadd_float
+define float @gs_fadd_float(ptr addrspace(3) %p, float %val) {
+  ; CHECK-NOT: bitcast
+  ; CHECK: atomicrmw fadd ptr addrspace(3) %p, float %val syncscope("workgroup") monotonic
+  %old = atomicrmw fadd ptr addrspace(3) %p, float %val syncscope("workgroup") monotonic
+  ret float %old
+}
diff --git a/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll b/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll
new file mode 100644
index 00000000000000..a35fce262f9a60
--- /dev/null
+++ b/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll
@@ -0,0 +1,38 @@
+; RUN: opt -S -dxil-resource-access -dxil-op-lower -mtriple=dxil-pc-shadermodel6.0-compute %s | FileCheck %s
+
+; DXIL has no floating-point atomic op. A float exchange only moves the bit
+; pattern, so it lowers to an i32 AtomicBinOp with a bitcast on the value and
+; on the returned original value. This needs no capability bits, so it works
+; from SM 6.0.
+
+target triple = "dxil-pc-shadermodel6.0-compute"
+
+; CHECK-LABEL: define float @bab_xchg_float
+define float @bab_xchg_float(i32 %offset, float %val) {
+  %buffer = call target("dx.RawBuffer", i8, 1, 0, 0)
+      @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null)
+  %ptr = call ptr @llvm.dx.resource.getpointer(
+      target("dx.RawBuffer", i8, 1, 0, 0) %buffer, i32 %offset)
+  ; CHECK: [[CAST:%.*]] = bitcast float %val to i32
+  ; CHECK: [[OLD:%.*]] = call i32 @dx.op.atomicBinOp.i32(i32 78, %dx.types.Handle %{{.*}}, i32 8, i32 %offset, i32 poison, i32 poison, i32 [[CAST]])
+  ; CHECK: [[RES:%.*]] = bitcast i32 [[OLD]] to float
+  %old = atomicrmw xchg ptr %ptr, float %val monotonic
+  ; CHECK: ret float [[RES]]
+  ret float %old
+}
+
+; A StructuredBuffer of float keeps the struct index in coord0 and the byte
+; offset in coord1.
+; CHECK-LABEL: define float @sbuf_xchg_float
+define float @sbuf_xchg_float(i32 %index, float %val) {
+  %buffer = call target("dx.RawBuffer", float, 1, 0, 0)
+      @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null)
+  %ptr = call ptr @llvm.dx.resource.getpointer(
+      target("dx.RawBuffer", float, 1, 0, 0) %buffer, i32 %index)
+  ; CHECK: [[CAST:%.*]] = bitcast float %val to i32
+  ; CHECK: [[OLD:%.*]] = call i32 @dx.op.atomicBinOp.i32(i32 78, %dx.types.Handle %{{.*}}, i32 8, i32 %index, i32 0, i32 poison, i32 [[CAST]])
+  ; CHECK: [[RES:%.*]] = bitcast i32 [[OLD]] to float
+  %old = atomicrmw xchg ptr %ptr, float %val monotonic
+  ; CHECK: ret float [[RES]]
+  ret float %old
+}

>From fe2c83e6adf0536ae6fe018d5538987db7f312a6 Mon Sep 17 00:00:00 2001
From: Joshua Batista <jbatista at microsoft.com>
Date: Thu, 17 Sep 2026 11:19:31 -0700
Subject: [PATCH 2/6] Add float InterlockedExchange coverage to the texture
 test

---
 .../test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl  | 8 +++++++-
 1 file changed, 7 insertions(+), 1 deletion(-)

diff --git a/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl b/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl
index 5013b11c62636e..4a5d221a8ad4d8 100644
--- a/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl
+++ b/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl
@@ -19,6 +19,7 @@
 
 RWTexture2D<int> Out : register(u0);
 RWTexture2DArray<uint> UOut : register(u1);
+RWTexture2D<float> FOut : register(u2);
 
 // CHECK-LABEL: define void @main
 // DXCHECK:  %[[PTR1:.*]] = call {{.*}} @llvm.dx.resource.getpointer.{{.*}}(target("dx.Texture", i32, 1, 0, 1, 2) %{{.*}}, <2 x i32> %{{.*}})
@@ -37,6 +38,8 @@ RWTexture2DArray<uint> UOut : register(u1);
 // DXCHECK:  atomicrmw umax ptr %[[PTR7]], i32 1 syncscope("device") monotonic
 // DXCHECK:  %[[PTR8:.*]] = call {{.*}} @llvm.dx.resource.getpointer.{{.*}}(target("dx.Texture", i32, 1, 0, 1, 2) %{{.*}}, <2 x i32> %{{.*}})
 // DXCHECK:  atomicrmw xchg ptr %[[PTR8]], i32 1 syncscope("device") monotonic
+// DXCHECK:  %[[PTR9:.*]] = call {{.*}} @llvm.dx.resource.getpointer.{{.*}}(target("dx.Texture", float, 1, 0, 0, 2) %{{.*}}, <2 x i32> %{{.*}})
+// DXCHECK:  atomicrmw xchg ptr %[[PTR9]], float 1.000000e+00 syncscope("device") monotonic
 // SPVCHECK: %[[PTR1:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, <2 x i32> %{{.*}})
 // SPVCHECK: atomicrmw add ptr addrspace(11) %[[PTR1]], i32 1 syncscope("device") monotonic
 // SPVCHECK: %[[PTR2:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, <2 x i32> %{{.*}})
@@ -53,6 +56,8 @@ RWTexture2DArray<uint> UOut : register(u1);
 // SPVCHECK: atomicrmw umax ptr addrspace(11) %[[PTR7]], i32 1 syncscope("device") monotonic
 // SPVCHECK: %[[PTR8:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, <2 x i32> %{{.*}})
 // SPVCHECK: atomicrmw xchg ptr addrspace(11) %[[PTR8]], i32 1 syncscope("device") monotonic
+// SPVCHECK: %[[PTR9:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.Image", float, {{.*}}) %{{.*}}, <2 x i32> %{{.*}})
+// SPVCHECK: atomicrmw xchg ptr addrspace(11) %[[PTR9]], float 1.000000e+00 syncscope("device") monotonic
 [shader("compute")]
 [numthreads(1,1,1)]
 void main(uint3 id : SV_DispatchThreadID) {
@@ -63,7 +68,8 @@ void main(uint3 id : SV_DispatchThreadID) {
   InterlockedMin(UOut[id], 1u);
   InterlockedMax(Out[id.xy], 1);
   InterlockedMax(UOut[id], 1u);
-  // DXIL has no float texture atomic, so only the integer form is covered.
   int Orig;
   InterlockedExchange(Out[id.xy], 1, Orig);
+  float FOrig;
+  InterlockedExchange(FOut[id.xy], 1.0f, FOrig);
 }

>From 5bfa2d291404f21433a4500192cc86d79fd693a0 Mon Sep 17 00:00:00 2001
From: Joshua Batista <jbatista at microsoft.com>
Date: Thu, 17 Sep 2026 12:43:06 -0700
Subject: [PATCH 3/6] Lower atomicrmw on scalar float texture resources

---
 llvm/lib/Target/DirectX/DXILResourceAccess.cpp |  7 +++++--
 .../ResourceAtomicBinOp-texture-nonscalar.ll   |  8 ++++----
 .../DirectX/ResourceAtomicExchangeFloat.ll     | 18 ++++++++++++++++++
 3 files changed, 27 insertions(+), 6 deletions(-)

diff --git a/llvm/lib/Target/DirectX/DXILResourceAccess.cpp b/llvm/lib/Target/DirectX/DXILResourceAccess.cpp
index 0dd5f69eff7cb9..70f5407aeadcfd 100644
--- a/llvm/lib/Target/DirectX/DXILResourceAccess.cpp
+++ b/llvm/lib/Target/DirectX/DXILResourceAccess.cpp
@@ -424,10 +424,13 @@ static void createBufferAtomicBinOp(IntrinsicInst *II, AtomicRMWInst *AI,
 
 static void createTextureAtomicBinOp(IntrinsicInst *II, AtomicRMWInst *AI,
                                      dxil::ResourceTypeInfo &RTI) {
+  // A texture atomic operates on a whole texel, so a multi-component texel has
+  // no single addressable component. A scalar float texel is allowed, because
+  // emitAtomicBinOp exchanges its bit pattern as an integer.
   Type *ContainedType = RTI.getHandleTy()->getTypeParameter(0);
-  if (!ContainedType->isIntegerTy()) {
+  if (!ContainedType->isIntegerTy() && !ContainedType->isFloatingPointTy()) {
     reportFatalUsageError("DXIL atomicrmw requires a texture resource with a "
-                          "scalar integer element type");
+                          "scalar element type");
     return;
   }
 
diff --git a/llvm/test/CodeGen/DirectX/ResourceAtomicBinOp-texture-nonscalar.ll b/llvm/test/CodeGen/DirectX/ResourceAtomicBinOp-texture-nonscalar.ll
index 2abc27da6ffbd5..6c1b2c2ee2fd38 100644
--- a/llvm/test/CodeGen/DirectX/ResourceAtomicBinOp-texture-nonscalar.ll
+++ b/llvm/test/CodeGen/DirectX/ResourceAtomicBinOp-texture-nonscalar.ll
@@ -11,7 +11,7 @@
 
 target triple = "dxil-pc-shadermodel6.6-compute"
 
-; CHECK: DXIL atomicrmw requires a texture resource with a scalar integer element type
+; CHECK: DXIL atomicrmw requires a texture resource with a scalar element type
 define i32 @atomic_texture1d_int2(i32 %coord, i32 %value) {
   %texture = call target("dx.Texture", <2 x i32>, 1, 0, 0, 1)
       @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null)
@@ -25,7 +25,7 @@ define i32 @atomic_texture1d_int2(i32 %coord, i32 %value) {
 
 target triple = "dxil-pc-shadermodel6.6-compute"
 
-; CHECK: DXIL atomicrmw requires a texture resource with a scalar integer element type
+; CHECK: DXIL atomicrmw requires a texture resource with a scalar element type
 define i32 @atomic_texture2d_int4(<2 x i32> %coords, i32 %value) {
   %texture = call target("dx.Texture", <4 x i32>, 1, 0, 0, 2)
       @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null)
@@ -39,7 +39,7 @@ define i32 @atomic_texture2d_int4(<2 x i32> %coords, i32 %value) {
 
 target triple = "dxil-pc-shadermodel6.6-compute"
 
-; CHECK: DXIL atomicrmw requires a texture resource with a scalar integer element type
+; CHECK: DXIL atomicrmw requires a texture resource with a scalar element type
 define i64 @atomic_texture2darray_i64x2(<3 x i32> %coords, i64 %value) {
   %texture = call target("dx.Texture", <2 x i64>, 1, 0, 0, 7)
       @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null)
@@ -53,7 +53,7 @@ define i64 @atomic_texture2darray_i64x2(<3 x i32> %coords, i64 %value) {
 
 target triple = "dxil-pc-shadermodel6.6-compute"
 
-; CHECK: DXIL atomicrmw requires a texture resource with a scalar integer element type
+; CHECK: DXIL atomicrmw requires a texture resource with a scalar element type
 define i32 @atomic_texture3d_float4(<3 x i32> %coords, i32 %value) {
   %texture = call target("dx.Texture", <4 x float>, 1, 0, 0, 4)
       @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null)
diff --git a/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll b/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll
index a35fce262f9a60..ace0c929afb69b 100644
--- a/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll
+++ b/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll
@@ -36,3 +36,21 @@ define float @sbuf_xchg_float(i32 %index, float %val) {
   ; CHECK: ret float [[RES]]
   ret float %old
 }
+
+; A texture of scalar float lowers the same way, with one coordinate operand
+; per texture dimension.
+; CHECK-LABEL: define float @texture2d_xchg_float
+define float @texture2d_xchg_float(<2 x i32> %coords, float %val) {
+  %texture = call target("dx.Texture", float, 1, 0, 0, 2)
+      @llvm.dx.resource.handlefrombinding(i32 0, i32 1, i32 1, i32 0, ptr null)
+  %ptr = call ptr @llvm.dx.resource.getpointer(
+      target("dx.Texture", float, 1, 0, 0, 2) %texture, <2 x i32> %coords)
+  ; CHECK: %[[X:.*]] = extractelement <2 x i32> %coords, i64 0
+  ; CHECK: %[[Y:.*]] = extractelement <2 x i32> %coords, i64 1
+  ; CHECK: [[CAST:%.*]] = bitcast float %val to i32
+  ; CHECK: [[OLD:%.*]] = call i32 @dx.op.atomicBinOp.i32(i32 78, %dx.types.Handle %{{.*}}, i32 8, i32 %[[X]], i32 %[[Y]], i32 poison, i32 [[CAST]])
+  ; CHECK: [[RES:%.*]] = bitcast i32 [[OLD]] to float
+  %old = atomicrmw xchg ptr %ptr, float %val monotonic
+  ; CHECK: ret float [[RES]]
+  ret float %old
+}

>From 7e3da5cebbc7c6ce63536cee8614de281dc9e7a1 Mon Sep 17 00:00:00 2001
From: Joshua Batista <jbatista at microsoft.com>
Date: Thu, 17 Sep 2026 14:48:40 -0700
Subject: [PATCH 4/6] Remove comments on the float InterlockedExchange
 declarations

---
 clang/include/clang/Basic/Builtins.td         | 1 -
 clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp | 3 ---
 2 files changed, 4 deletions(-)

diff --git a/clang/include/clang/Basic/Builtins.td b/clang/include/clang/Basic/Builtins.td
index 19a678ae67a3b4..fcdbf4b885eb4b 100644
--- a/clang/include/clang/Basic/Builtins.td
+++ b/clang/include/clang/Basic/Builtins.td
@@ -5561,7 +5561,6 @@ def HLSLInterlockedAnd : LangBuiltin<"HLSL_LANG"> {
 
 def HLSLInterlockedExchange : LangBuiltin<"HLSL_LANG"> {
   let Spellings = ["__builtin_hlsl_interlocked_exchange"];
-  // Prevent inadvertent float -> double arg promotion.
   let Attributes = [NoThrow, CustomTypeChecking];
   let Prototype = "void (...)";
 }
diff --git a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp
index 91874dbca65b80..4fc974763a6fe5 100644
--- a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp
+++ b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp
@@ -1765,9 +1765,6 @@ BuiltinTypeDeclBuilder::addByteAddressBufferInterlockedMethods() {
   addByteAddressBufferInterlockedMethod(
       "InterlockedExchange", AST.UnsignedIntTy,
       "__builtin_hlsl_interlocked_exchange", /*RequiresOriginalValue=*/true);
-  // The float exchange reuses the 32-bit integer DXIL operation, so it needs
-  // no capability bits and works from SM 6.0. ByteAddressBuffer carries no
-  // element type, so the method name states the type.
   addByteAddressBufferInterlockedMethod("InterlockedExchangeFloat", AST.FloatTy,
                                         "__builtin_hlsl_interlocked_exchange",
                                         /*RequiresOriginalValue=*/true);

>From ac4fcb79efd4f9ce07f87315ee92f5ce2fa781b8 Mon Sep 17 00:00:00 2001
From: Joshua Batista <jbatista at microsoft.com>
Date: Wed, 23 Sep 2026 12:34:33 -0700
Subject: [PATCH 5/6] address Kaitlin, reject float atomic exchange of
 unsupported width

---
 llvm/lib/Target/DirectX/DXILLegalizePass.cpp  | 10 ++++++-
 ...zeAtomicExchangeFloat-unsupported-width.ll | 30 +++++++++++++++++++
 .../DirectX/LegalizeAtomicExchangeFloat.ll    | 12 ++++++++
 3 files changed, 51 insertions(+), 1 deletion(-)
 create mode 100644 llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll

diff --git a/llvm/lib/Target/DirectX/DXILLegalizePass.cpp b/llvm/lib/Target/DirectX/DXILLegalizePass.cpp
index c8037fbd84d0cc..9cc864c5e47762 100644
--- a/llvm/lib/Target/DirectX/DXILLegalizePass.cpp
+++ b/llvm/lib/Target/DirectX/DXILLegalizePass.cpp
@@ -17,6 +17,7 @@
 #include "llvm/IR/Instructions.h"
 #include "llvm/IR/Module.h"
 #include "llvm/Pass.h"
+#include "llvm/Support/ErrorHandling.h"
 #include "llvm/Transforms/Utils/BasicBlockUtils.h"
 #include "llvm/Transforms/Utils/Local.h"
 #include <functional>
@@ -372,8 +373,15 @@ legalizeFloatAtomicExchange(Instruction &I,
   if (!ValTy->isFloatingPointTy())
     return false;
 
+  // DXIL has 32-bit and 64-bit atomics only. A float of any other width has no
+  // integer exchange to lower to.
+  unsigned Width = ValTy->getPrimitiveSizeInBits();
+  if (Width != 32 && Width != 64)
+    reportFatalUsageError("DXIL atomic exchange requires a 32-bit or 64-bit "
+                          "floating-point value");
+
   IRBuilder<> Builder(AI);
-  Type *IntTy = Builder.getIntNTy(ValTy->getPrimitiveSizeInBits());
+  Type *IntTy = Builder.getIntNTy(Width);
   Value *Val = Builder.CreateBitCast(AI->getValOperand(), IntTy);
   AtomicRMWInst *NewAI = Builder.CreateAtomicRMW(
       AtomicRMWInst::Xchg, AI->getPointerOperand(), Val, AI->getAlign(),
diff --git a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll
new file mode 100644
index 00000000000000..e75d74965da8d6
--- /dev/null
+++ b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll
@@ -0,0 +1,30 @@
+; RUN: split-file %s %t
+; RUN: not opt -S -dxil-legalize -mtriple=dxil-pc-shadermodel6.0-compute %t/half.ll 2>&1 | FileCheck %t/half.ll
+; RUN: not opt -S -dxil-legalize -mtriple=dxil-pc-shadermodel6.0-compute %t/fp128.ll 2>&1 | FileCheck %t/fp128.ll
+
+; DXIL has 32-bit and 64-bit atomics only, so a float exchange of any other
+; width has no integer exchange to lower to.
+
+;--- half.ll
+
+target triple = "dxil-pc-shadermodel6.0-compute"
+
+ at gs = external addrspace(3) global half
+
+; CHECK: DXIL atomic exchange requires a 32-bit or 64-bit floating-point value
+define half @gs_xchg_half(half %val) {
+  %old = atomicrmw xchg ptr addrspace(3) @gs, half %val syncscope("workgroup") monotonic
+  ret half %old
+}
+
+;--- fp128.ll
+
+target triple = "dxil-pc-shadermodel6.0-compute"
+
+ at gs = external addrspace(3) global fp128
+
+; CHECK: DXIL atomic exchange requires a 32-bit or 64-bit floating-point value
+define fp128 @gs_xchg_fp128(fp128 %val) {
+  %old = atomicrmw xchg ptr addrspace(3) @gs, fp128 %val syncscope("workgroup") monotonic
+  ret fp128 %old
+}
\ No newline at end of file
diff --git a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll
index 446c76a2b801db..8af5c533519bf2 100644
--- a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll
+++ b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll
@@ -8,6 +8,7 @@
 target triple = "dxil-pc-shadermodel6.0-compute"
 
 @gs = external addrspace(3) global float
+ at gsd = external addrspace(3) global double
 
 ; CHECK-LABEL: define float @gs_xchg_float
 define float @gs_xchg_float(float %val) {
@@ -19,6 +20,17 @@ define float @gs_xchg_float(float %val) {
   ret float %old
 }
 
+; DXIL has a 64-bit atomic, so a double exchange becomes an i64 exchange.
+; CHECK-LABEL: define double @gs_xchg_double
+define double @gs_xchg_double(double %val) {
+  ; CHECK: [[CAST:%.*]] = bitcast double %val to i64
+  ; CHECK: [[OLD:%.*]] = atomicrmw xchg ptr addrspace(3) @gsd, i64 [[CAST]] syncscope("workgroup") monotonic
+  ; CHECK: [[RES:%.*]] = bitcast i64 [[OLD]] to double
+  %old = atomicrmw xchg ptr addrspace(3) @gsd, double %val syncscope("workgroup") monotonic
+  ; CHECK: ret double [[RES]]
+  ret double %old
+}
+
 ; An integer exchange must pass through with no bitcast.
 ; CHECK-LABEL: define i32 @gs_xchg_i32
 define i32 @gs_xchg_i32(ptr addrspace(3) %p, i32 %val) {

>From 9b76fdf6eee789a7ee9e121d2052392d6721152d Mon Sep 17 00:00:00 2001
From: Joshua Batista <jbatista at microsoft.com>
Date: Wed, 23 Sep 2026 12:53:24 -0700
Subject: [PATCH 6/6] add new line

---
 .../DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll    | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll
index e75d74965da8d6..ed344f8b27c000 100644
--- a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll
+++ b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll
@@ -27,4 +27,4 @@ target triple = "dxil-pc-shadermodel6.0-compute"
 define fp128 @gs_xchg_fp128(fp128 %val) {
   %old = atomicrmw xchg ptr addrspace(3) @gs, fp128 %val syncscope("workgroup") monotonic
   ret fp128 %old
-}
\ No newline at end of file
+}



More information about the llvm-branch-commits mailing list