[compiler-rt] [llvm] [compiler-rt] Add KCSAN inspired GPU ConcurrencySanitizer runtime (PR #207714)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Jul 6 05:24:07 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Joseph Huber (jhuber6)
<details>
<summary>Changes</summary>
Summary:
Add a standalone GPU concurrency sanitizer runtime as a new top-level
compiler-rt component 'csan' (libclang_rt.csan.a), enabled for offloading
device builds. It is a watchpoint-based data race detector modelled on the
Linux kernel's KCSAN and reports races to the host over RPC. Tests live in
the top-level 'test/csan' suite.
---
Patch is 68.78 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/207714.diff
31 Files Affected:
- (modified) compiler-rt/cmake/caches/AMDGPU.cmake (+1-1)
- (modified) compiler-rt/cmake/config-ix.cmake (+7-1)
- (modified) compiler-rt/include/CMakeLists.txt (+1)
- (added) compiler-rt/include/sanitizer/gpu_sanitizer.h (+59)
- (added) compiler-rt/lib/csan/.clang-format (+3)
- (added) compiler-rt/lib/csan/CMakeLists.txt (+31)
- (added) compiler-rt/lib/csan/csan_gpu.cpp (+732)
- (added) compiler-rt/test/csan/CMakeLists.txt (+29)
- (added) compiler-rt/test/csan/array_race.c (+18)
- (added) compiler-rt/test/csan/atomic.c (+10)
- (added) compiler-rt/test/csan/disjoint.c (+13)
- (added) compiler-rt/test/csan/global_race.c (+14)
- (added) compiler-rt/test/csan/lds_disjoint.c (+13)
- (added) compiler-rt/test/csan/lds_race.c (+15)
- (added) compiler-rt/test/csan/lit.cfg.py (+30)
- (added) compiler-rt/test/csan/lit.site.cfg.py.in (+11)
- (added) compiler-rt/test/csan/memcpy_large_race.c (+25)
- (added) compiler-rt/test/csan/memcpy_race.c (+16)
- (added) compiler-rt/test/csan/memmove_race.c (+25)
- (added) compiler-rt/test/csan/race.h (+28)
- (added) compiler-rt/test/csan/single_thread.c (+11)
- (added) compiler-rt/test/csan/write_read_race.c (+20)
- (modified) offload/plugins-nextgen/CMakeLists.txt (+1-1)
- (modified) offload/plugins-nextgen/common/CMakeLists.txt (+3)
- (modified) offload/plugins-nextgen/common/include/PluginInterface.h (+2)
- (modified) offload/plugins-nextgen/common/include/RPC.h (+15)
- (added) offload/plugins-nextgen/common/include/Sanitizer.h (+42)
- (modified) offload/plugins-nextgen/common/include/Utils/ELF.h (+33)
- (modified) offload/plugins-nextgen/common/src/RPC.cpp (+15-1)
- (added) offload/plugins-nextgen/common/src/Sanitizer.cpp (+191)
- (modified) offload/plugins-nextgen/common/src/Utils/ELF.cpp (+55)
``````````diff
diff --git a/compiler-rt/cmake/caches/AMDGPU.cmake b/compiler-rt/cmake/caches/AMDGPU.cmake
index f3a9510c4f311..f9b4be733a5bc 100644
--- a/compiler-rt/cmake/caches/AMDGPU.cmake
+++ b/compiler-rt/cmake/caches/AMDGPU.cmake
@@ -7,7 +7,7 @@ set(COMPILER_RT_BUILD_BUILTINS ON CACHE BOOL "")
set(COMPILER_RT_BAREMETAL_BUILD ON CACHE BOOL "")
set(COMPILER_RT_BUILD_CRT OFF CACHE BOOL "")
set(COMPILER_RT_BUILD_SANITIZERS ON CACHE BOOL "")
-set(COMPILER_RT_SANITIZERS_TO_BUILD "ubsan_minimal" CACHE STRING "")
+set(COMPILER_RT_SANITIZERS_TO_BUILD "ubsan_minimal;csan" CACHE STRING "")
set(COMPILER_RT_BUILD_XRAY OFF CACHE BOOL "")
set(COMPILER_RT_BUILD_LIBFUZZER OFF CACHE BOOL "")
set(COMPILER_RT_BUILD_PROFILE ON CACHE BOOL "")
diff --git a/compiler-rt/cmake/config-ix.cmake b/compiler-rt/cmake/config-ix.cmake
index 083f1c98d0f16..43fa8aa434321 100644
--- a/compiler-rt/cmake/config-ix.cmake
+++ b/compiler-rt/cmake/config-ix.cmake
@@ -765,7 +765,7 @@ if(COMPILER_RT_SUPPORTED_ARCH)
endif()
message(STATUS "Compiler-RT supported architectures: ${COMPILER_RT_SUPPORTED_ARCH}")
-set(ALL_SANITIZERS asan;rtsan;dfsan;msan;hwasan;tsan;tysan;safestack;cfi;scudo_standalone;ubsan_minimal;gwp_asan;nsan;asan_abi)
+set(ALL_SANITIZERS asan;rtsan;dfsan;msan;hwasan;tsan;csan;tysan;safestack;cfi;scudo_standalone;ubsan_minimal;gwp_asan;nsan;asan_abi)
set(COMPILER_RT_SANITIZERS_TO_BUILD all CACHE STRING
"sanitizers to build if supported on the target (all;${ALL_SANITIZERS})")
list_replace(COMPILER_RT_SANITIZERS_TO_BUILD all "${ALL_SANITIZERS}")
@@ -880,6 +880,12 @@ else()
set(COMPILER_RT_HAS_TSAN FALSE)
endif()
+if(COMPILER_RT_GPU_BUILD)
+ set(COMPILER_RT_HAS_CSAN TRUE)
+else()
+ set(COMPILER_RT_HAS_CSAN FALSE)
+endif()
+
if (OS_NAME MATCHES "Linux|FreeBSD|Windows|NetBSD|SunOS")
set(COMPILER_RT_TSAN_HAS_STATIC_RUNTIME TRUE)
else()
diff --git a/compiler-rt/include/CMakeLists.txt b/compiler-rt/include/CMakeLists.txt
index eb998478b081b..05c78eeace345 100644
--- a/compiler-rt/include/CMakeLists.txt
+++ b/compiler-rt/include/CMakeLists.txt
@@ -5,6 +5,7 @@ if (COMPILER_RT_BUILD_SANITIZERS)
sanitizer/common_interface_defs.h
sanitizer/coverage_interface.h
sanitizer/dfsan_interface.h
+ sanitizer/gpu_sanitizer.h
sanitizer/hwasan_interface.h
sanitizer/linux_syscall_hooks.h
sanitizer/lsan_interface.h
diff --git a/compiler-rt/include/sanitizer/gpu_sanitizer.h b/compiler-rt/include/sanitizer/gpu_sanitizer.h
new file mode 100644
index 0000000000000..e38c677953c62
--- /dev/null
+++ b/compiler-rt/include/sanitizer/gpu_sanitizer.h
@@ -0,0 +1,59 @@
+//===-- sanitizer/gpu_sanitizer.h -------------------------------*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// Shared ABI contract for GPU sanitizer reports shipped to the host over RPC.
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef SANITIZER_GPU_SANITIZER_H
+#define SANITIZER_GPU_SANITIZER_H
+
+#include <stdint.h>
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+#define SANITIZER_GPU_OPCODE(n) (('s' << 24) | (n))
+
+#define TSAN_GPU_REPORT_OPCODE SANITIZER_GPU_OPCODE(1)
+
+/// Access classification flags, mirroring the Linux kernel's KCSAN model.
+enum {
+ TSAN_GPU_ACCESS_WRITE = 1 << 0, ///< Non-atomic write.
+ TSAN_GPU_ACCESS_COMPOUND = 1 << 1, ///< Read-modify-write.
+ TSAN_GPU_ACCESS_ATOMIC = 1 << 2, ///< Atomic access.
+};
+
+/// Race report kinds.
+enum {
+ TSAN_GPU_DATA_RACE = 0, ///< Conflicting access.
+ TSAN_GPU_UNKNOWN_ORIGIN = 1, ///< Watched value changed with no finder.
+ TSAN_GPU_INTRA_WAVE = 2, ///< Conflict between lanes of the same wave.
+};
+
+/// The data associated with a single detected race.
+typedef struct __tsan_gpu_race {
+ uint64_t pc; ///< Device PC of the reporting access.
+ uint64_t peer_pc; ///< Device PC of the conflicting access.
+ uint64_t addr; ///< Accessed device address.
+ uint32_t size; ///< Access size in bytes.
+ uint32_t access_type; ///< Bitwise-or of TSAN_GPU_ACCESS_* flags.
+ uint32_t kind; ///< One of the TSAN_GPU_* race kinds.
+ uint32_t block[3]; ///< Block / workgroup id.
+ uint16_t thread[3]; ///< Thread / work-item id within the block.
+ uint8_t lane; ///< Lane id within the wave.
+ uint8_t peer_lane; ///< Conflicting lane for intra-wave races.
+ uint16_t peer_thread[3]; ///< Conflicting thread id for intra-wave races.
+} __tsan_gpu_race;
+
+#ifdef __cplusplus
+} // extern "C"
+#endif
+
+#endif // SANITIZER_GPU_SANITIZER_H
diff --git a/compiler-rt/lib/csan/.clang-format b/compiler-rt/lib/csan/.clang-format
new file mode 100644
index 0000000000000..1f2a97030379d
--- /dev/null
+++ b/compiler-rt/lib/csan/.clang-format
@@ -0,0 +1,3 @@
+BasedOnStyle: Google
+AllowShortIfStatementsOnASingleLine: false
+IndentPPDirectives: AfterHash
diff --git a/compiler-rt/lib/csan/CMakeLists.txt b/compiler-rt/lib/csan/CMakeLists.txt
new file mode 100644
index 0000000000000..646789ec317f6
--- /dev/null
+++ b/compiler-rt/lib/csan/CMakeLists.txt
@@ -0,0 +1,31 @@
+set(CSAN_SOURCES
+ csan_gpu.cpp
+ )
+
+include_directories(..)
+
+include_directories(${COMPILER_RT_SOURCE_DIR}/include)
+include(FindLibcCommonUtils)
+get_target_property(LIBC_COMMON_UTILS_INCLUDE
+ llvm-libc-common-utilities INTERFACE_INCLUDE_DIRECTORIES)
+include_directories(${LIBC_COMMON_UTILS_INCLUDE})
+
+set(CSAN_CFLAGS
+ ${SANITIZER_COMMON_CFLAGS}
+ -DSANITIZER_COMMON_NO_REDEFINE_BUILTINS)
+append_rtti_flag(OFF CSAN_CFLAGS)
+
+add_compiler_rt_component(csan)
+
+add_compiler_rt_object_libraries(RTCsan
+ OS ${SANITIZER_COMMON_SUPPORTED_OS}
+ ARCHS ${UBSAN_COMMON_SUPPORTED_ARCH}
+ SOURCES ${CSAN_SOURCES} CFLAGS ${CSAN_CFLAGS})
+
+add_compiler_rt_runtime(clang_rt.csan
+ STATIC
+ OS ${UBSAN_SUPPORTED_OS}
+ ARCHS ${UBSAN_SUPPORTED_ARCH}
+ OBJECT_LIBS RTCsan
+ CFLAGS ${CSAN_CFLAGS}
+ PARENT_TARGET csan)
diff --git a/compiler-rt/lib/csan/csan_gpu.cpp b/compiler-rt/lib/csan/csan_gpu.cpp
new file mode 100644
index 0000000000000..27056be71fa3a
--- /dev/null
+++ b/compiler-rt/lib/csan/csan_gpu.cpp
@@ -0,0 +1,732 @@
+//===-- csan_gpu.cpp - GPU ConcurrencySanitizer runtime -------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// Watchpoint-based data race detector for GPU targets, modelled on the Linux
+// kernel's KCSAN.
+//
+//===----------------------------------------------------------------------===//
+
+#include <gpuintrin.h>
+#include <stdint.h>
+
+#include "sanitizer/gpu_sanitizer.h"
+#include "sanitizer_common/sanitizer_internal_defs.h"
+#include "shared/rpc.h"
+
+[[gnu::visibility("protected"),
+ gnu::weak]] rpc::Client client asm("__llvm_rpc_client");
+
+// Running total of races this device has reported.
+extern "C" SANITIZER_INTERFACE_ATTRIBUTE uint64_t __tsan_num_data_races = 0;
+
+// Shallow deduplication check to save the host thread work. Keyed on both the
+// PC and the race kind so each distinct kind of race at a PC is reported once.
+static bool should_report(void* pc, unsigned kind) {
+ static uint64_t seen[64] = {};
+ const uint64_t token = (reinterpret_cast<uintptr_t>(pc) >> 4) ^
+ (static_cast<uint64_t>(kind) * 0x9E3779B97F4A7C15ull);
+ uint64_t idx = (token * 0x9E3779B97F4A7C15ull) >> 58;
+ uint64_t last = __scoped_atomic_exchange_n(
+ &seen[idx], token, __ATOMIC_RELAXED, __MEMORY_SCOPE_DEVICE);
+ return last != token;
+}
+
+// Report a data race to the RPC server so it can be symbolized and presented.
+[[gnu::cold]] static void report(unsigned kind, uintptr_t addr, uint32_t size,
+ int access_type, void* pc,
+ void* peer = nullptr, uint8_t peer_lane = 0,
+ const uint16_t (&peer_thread)[3] = {}) {
+ if (!should_report(pc, kind))
+ return;
+
+ __tsan_gpu_race rep;
+ rep.pc = reinterpret_cast<uintptr_t>(pc);
+ rep.peer_pc = reinterpret_cast<uintptr_t>(peer);
+ rep.addr = addr;
+ rep.size = size;
+ rep.access_type = static_cast<unsigned>(access_type);
+ rep.block[0] = __gpu_block_id(__GPU_X_DIM);
+ rep.block[1] = __gpu_block_id(__GPU_Y_DIM);
+ rep.block[2] = __gpu_block_id(__GPU_Z_DIM);
+ rep.thread[0] = __gpu_thread_id(__GPU_X_DIM);
+ rep.thread[1] = __gpu_thread_id(__GPU_Y_DIM);
+ rep.thread[2] = __gpu_thread_id(__GPU_Z_DIM);
+ rep.lane = __gpu_lane_id();
+ rep.peer_lane = peer_lane;
+ rep.peer_thread[0] = peer_thread ? peer_thread[0] : 0;
+ rep.peer_thread[1] = peer_thread ? peer_thread[1] : 0;
+ rep.peer_thread[2] = peer_thread ? peer_thread[2] : 0;
+ rep.kind = kind;
+
+ rpc::Client::Port Port = client.open<TSAN_GPU_REPORT_OPCODE>();
+ Port.send([&](rpc::Buffer* buf, uint32_t) {
+ __builtin_memcpy(buf->data, &rep, sizeof(rep));
+ });
+ static_assert(sizeof(__tsan_gpu_race) <= sizeof(rpc::Buffer),
+ "Report must fit in a single packet");
+
+ __scoped_atomic_fetch_add(&__tsan_num_data_races, 1, __ATOMIC_RELAXED,
+ __MEMORY_SCOPE_DEVICE);
+}
+
+#if defined(__AMDGPU__)
+// AMDGPU does not have a single set frequency. Different architectures and
+// cards can have different values. A frequency of 100MHz is most common so we
+// use it, if it is wrong it just means we sleep longer than expected.
+static constexpr uint64_t CLOCK_FREQ_HZ = 100000000UL;
+#else
+static constexpr uint64_t CLOCK_FREQ_HZ = 1000000000UL;
+#endif
+static constexpr uint64_t TICKS_PER_SEC = 1000000000UL;
+
+// Randomized bounds, in nanoseconds, for the watchpoint stall window. A wider
+// window catches more concurrent races at the cost of extra runtime.
+static constexpr uint64_t SAMPLE_DELAY_MIN_NS = 1000;
+static constexpr uint64_t SAMPLE_DELAY_MAX_NS = 5000;
+
+// Watchpoint table capacity, in bytes.
+static constexpr uint64_t WP_TABLE_SIZE = /*2 MiB=*/2 * 1024ul * 1024ul;
+
+// Number of slots in the watchpoint table, must be a power of two;
+static constexpr uint64_t WP_TABLE_SLOTS = WP_TABLE_SIZE / sizeof(uint64_t);
+
+// The probability that we set up a watchpoint at any given access.
+static constexpr uint32_t WP_CHANCE = 8;
+
+// Whether global accesses use the watchpoint table.
+static constexpr bool WP_ENABLE_TABLE = true;
+
+// The largest size we encode. We check a very narrow range to widen
+// watchpoints. Wide accesses are relatively uncommon on global memory.
+static constexpr uint32_t WP_MAX_SIZE = 8;
+
+static constexpr uint32_t WP_SIZE_BITS =
+ __builtin_popcountg(WP_MAX_SIZE - 1u) + 1;
+static constexpr uint32_t WP_ADDR_BITS = 64 - 2 - WP_SIZE_BITS;
+
+//===----------------------------------------------------------------------===//
+// Watchpoint encoding:
+// [63] is_write
+// [62] consumed
+// [61 : WP_ADDR_BITS] size
+// [WP_ADDR_BITS-1 : 0] address
+//===----------------------------------------------------------------------===//
+
+static constexpr uint64_t WP_INVALID = 0;
+static constexpr uint64_t WP_CONSUMED_MASK = 1ull << 62;
+static constexpr uint64_t WP_WRITE_MASK = 1ull << 63;
+static constexpr uint64_t WP_ADDR_MASK = (1ull << WP_ADDR_BITS) - 1;
+static constexpr uint64_t WP_SIZE_MASK = ((1ull << WP_SIZE_BITS) - 1)
+ << WP_ADDR_BITS;
+static_assert((WP_MAX_SIZE & (WP_MAX_SIZE - 1)) == 0,
+ "WP_MAX_SIZE must be a power of two");
+static_assert((WP_WRITE_MASK ^ WP_CONSUMED_MASK ^ WP_SIZE_MASK ^
+ WP_ADDR_MASK) == ~0ull,
+ "watchpoint fields must partition the 64-bit word");
+
+static constexpr uint64_t encode_watchpoint(uint64_t addr, uint32_t size,
+ bool is_write) {
+ return (is_write ? WP_WRITE_MASK : 0) | ((uint64_t)size << WP_ADDR_BITS) |
+ (addr & WP_ADDR_MASK);
+}
+
+static constexpr bool decode_watchpoint(uint64_t wp, uint64_t& addr,
+ uint32_t& size, bool& is_write) {
+ if (wp == WP_INVALID || (wp & WP_CONSUMED_MASK))
+ return false;
+ is_write = (wp & WP_WRITE_MASK) != 0;
+ size = (wp & WP_SIZE_MASK) >> WP_ADDR_BITS;
+ addr = wp & WP_ADDR_MASK;
+ return true;
+}
+
+static constexpr bool ranges_overlap(uint64_t a1, uint32_t s1, uint64_t a2,
+ uint32_t s2) {
+ return a1 < a2 + s2 && a2 < a1 + s1;
+}
+
+static_assert((WP_TABLE_SLOTS & (WP_TABLE_SLOTS - 1)) == 0,
+ "WP_TABLE_SLOTS must be a power of two");
+
+static constexpr uint32_t watchpoint_slot(uint64_t addr) {
+ const uint64_t word = addr >> (WP_SIZE_BITS - 1);
+ return word & (WP_TABLE_SLOTS - 1);
+}
+
+static inline uint32_t xorshift32(uint32_t& state) {
+ state ^= state << 13;
+ state ^= state >> 17;
+ state ^= state << 5;
+ return state * 0x9e3779bb;
+}
+
+// Wave-uniform Bernoulli trial that is true with probability 1 / N.
+template <uint32_t N>
+static bool bernoulli(uint32_t& rand) {
+ static_assert((N & (N - 1)) == 0,
+ "sample denominator must be a power of two");
+ [[maybe_unused]] const uint32_t r = xorshift32(rand);
+ if constexpr (N == 0)
+ return false;
+ else if constexpr (N == 1)
+ return true;
+ else
+ return (r >> (32 - __builtin_ctzg(N))) == 0;
+}
+
+static uint64_t watchpoints[WP_TABLE_SLOTS] = {};
+
+// Flatten the block-local thread index into a linear lane index.
+static uint32_t flat_thread_id() {
+ return __gpu_thread_id(__GPU_X_DIM) +
+ __gpu_num_threads(__GPU_X_DIM) *
+ (__gpu_thread_id(__GPU_Y_DIM) +
+ __gpu_num_threads(__GPU_Y_DIM) * __gpu_thread_id(__GPU_Z_DIM));
+}
+
+// PRNG seed based off of the current thread's IDs and clock cycle.
+static uint32_t entropy() {
+ return (static_cast<uint32_t>(__builtin_readcyclecounter() | 1u) ^
+ flat_thread_id() ^ (__gpu_block_id(__GPU_X_DIM) * 0x9e3779b9u));
+}
+
+namespace {
+// Per-wavefront analysis state, seeded once at kernel entry and carried for the
+// lifetime of the wave.
+struct Ctx {
+ uint32_t rand;
+};
+} // namespace
+
+// Return the calling wavefront's context from the global LDS.
+static __gpu_local Ctx& get_ctx() {
+ static constexpr uint32_t MAX_WAVEFRONTS = 1024 / 32;
+ static __gpu_local Ctx ctx[MAX_WAVEFRONTS]
+ __attribute__((loader_uninitialized));
+ return ctx[flat_thread_id() / __gpu_num_lanes()];
+}
+
+// Type trait helpers for address spaces.
+template <typename>
+struct is_ptr_local {
+ static constexpr bool value = false;
+};
+template <typename T>
+struct is_ptr_local<T __gpu_local*> {
+ static constexpr bool value = true;
+};
+
+template <typename PtrTy>
+static constexpr bool uses_table() {
+ return WP_ENABLE_TABLE && !is_ptr_local<PtrTy>::value;
+}
+
+// Returns if a wavefront should sample this access for detection. Each eligible
+// access is sampled with a wave-uniform probability of 1/SAMPLE_CHANCE.
+static bool should_watch(uint64_t lane_mask, uint32_t access_type) {
+ // If every access is atomic we cannot have a race.
+ if (!__gpu_ballot(lane_mask, !(access_type & TSAN_GPU_ACCESS_ATOMIC)))
+ return false;
+
+ bool sample = false;
+ if (__gpu_is_first_in_lane(lane_mask))
+ sample = bernoulli<WP_CHANCE>(get_ctx().rand);
+ return __gpu_read_first_lane_u32(lane_mask, sample);
+}
+
+// Scan the watchpoint table to see if there are any points set for this
+// address. Cheap lookup done for each lane in the wavefront.
+template <typename PtrTy>
+static uint64_t* find_watchpoint(uintptr_t addr, uint32_t size,
+ bool expect_write, uint64_t& encoded) {
+ // Local addresses never use the global watchpoint table.
+ if constexpr (!uses_table<PtrTy>())
+ return nullptr;
+
+ uint64_t* wp = &watchpoints[watchpoint_slot(addr)];
+ encoded = __scoped_atomic_load_n(wp, __ATOMIC_ACQUIRE, __MEMORY_SCOPE_DEVICE);
+
+ uint64_t wp_addr;
+ uint32_t wp_size;
+ bool is_write;
+ if (!decode_watchpoint(encoded, wp_addr, wp_size, is_write))
+ return nullptr;
+ if (expect_write && !is_write)
+ return nullptr;
+ if (ranges_overlap(wp_addr, wp_size, addr & WP_ADDR_MASK, size))
+ return wp;
+ return nullptr;
+}
+
+// Claim a free slot for this access by atomically flipping it from INVALID to
+// the encoded watchpoint. Returns nullptr if every candidate slot is taken.
+static uint64_t* insert_watchpoint(uint64_t addr, uint32_t size,
+ bool is_write) {
+ uint64_t* wp = &watchpoints[watchpoint_slot(addr)];
+ const uint64_t encoded = encode_watchpoint(addr, size, is_write);
+ uint64_t expected = WP_INVALID;
+ if (__scoped_atomic_compare_exchange_n(wp, &expected, encoded, false,
+ __ATOMIC_RELEASE, __ATOMIC_RELAXED,
+ __MEMORY_SCOPE_DEVICE))
+ return wp;
+ return nullptr;
+}
+
+// A finder consumes a watchpoint to signal the setter that a race was observed,
+// storing its own PC into the address bits so the setter can report it as the
+// peer. Succeeds only if the slot still holds the value the finder matched.
+static bool try_consume_watchpoint(uint64_t* wp, uint64_t encoded, void* pc) {
+ uint64_t consumed =
+ WP_CONSUMED_MASK | (reinterpret_cast<uintptr_t>(pc) & WP_ADDR_MASK);
+ return __scoped_atomic_compare_exchange_n(wp, &encoded, consumed, false,
+ __ATOMIC_RELEASE, __ATOMIC_RELAXED,
+ __MEMORY_SCOPE_DEVICE);
+}
+
+// A setter tears down its own watchpoint and checks if another thread consumed
+// it and triggered a race, recovering the finder's stashed PC as 'peer'.
+static bool consume_watchpoint(uint64_t* wp, void*& peer) {
+ uint64_t old = __scoped_atomic_exchange_n(
+ wp, WP_CONSUMED_MASK, __ATOMIC_ACQUIRE, __MEMORY_SCOPE_DEVICE);
+ peer = reinterpret_cast<void*>(old & WP_ADDR_MASK);
+ return !(old & WP_CONSUMED_MASK);
+}
+
+// Release the slot back to the pool so it may be reused immediately after.
+static void remove_watchpoint(uint64_t* wp) {
+ __scoped_atomic_store_n(wp, WP_INVALID, __ATOMIC_RELAXED,
+ __MEMORY_SCOPE_DEVICE);
+}
+
+// FNV-1a digest of a byte range so wide accesses can reuse the 64-bit value
+// comparison semantics.
+template <typename BytePtr, typename WordPtr>
+static uint64_t read_range(BytePtr bytes, [[maybe_unused]] WordPtr words,
+ uint32_t size, int scope) {
+ uint64_t sum = 0xcbf29ce484222325ull;
+ uint32_t i = 0;
+
+ // Digest the unaligned prefix byte-by-byte up to a word.
+ for (; i < size && ((reinterpret_cast<uintptr_t>(bytes) + i) & 7u); ++i)
+ sum = (sum ^ __scoped_atomic_load_n(bytes + i, __ATOMIC_RELAXED, scope)) *
+ 0x100000001b3ull;
+
+ // Digest the aligned interior with wide loads.
+ for (; i + 8 <= size; i += 8)
+ sum = (sum ^ __scoped_atomic_load_n(reinterpret_cast<WordPtr>(bytes + i),
+ __ATOMIC_RELAXED, scope)) *
+ 0x100000001b3ull;
+
+ // Digest the trailing bytes that do not fill a word.
+ for (; i < size; ++i)
+ sum = (sum ^ __scoped_atomic_load_n(bytes + i, __ATOMIC_RELAXED, scope)) *
+ 0x100000001b3ull;
+ return sum;
+}
+
+// Snapshot the watched location for value-change detection. Larger sizes get
+// converted into a single checksum.
+static uint64_t read_instrumented_memory(const volatile __gpu_global void* ptr,
+ uint32_t size) {
+ const uintptr_t addr =
+ reinterpret_cast<uintptr_t>(const_cast<const __gpu_global void*>(ptr));
+ if ((addr & (size - 1)) == 0) {
+ switch (size) {
+ case 1:
+ return __scoped_atomic_load_n((const volatile __gpu_global uint8_t*)ptr,
+ __ATOMIC_RELAXED, __MEMORY_SCOPE_SYSTEM);
+ case 2:
+ return __scoped_atomic_load_n(
+ (const volatile __gpu_global uint16_t*)ptr, __ATOMIC_RELAXED,
+...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/207714
More information about the llvm-commits
mailing list