[llvm-branch-commits] [compiler-rt] [compiler-rt] Add 'csan' library for the concurrency sanitizer (PR #225782)

Yaxun Liu via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Thu Sep 24 07:33:03 PDT 2026


================
@@ -0,0 +1,456 @@
+//===----------------------------------------------------------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+///
+/// \file
+/// Watchpoint-based data race detector for GPU targets.
+///
+//===----------------------------------------------------------------------===//
+
+#include <gpuintrin.h>
+
+#include "csan_offload_packet.h"
+#include "csan_watch.h"
+#include "sanitizer_common/sanitizer_internal_defs.h"
+#include "shared/rpc.h"
+
+using namespace __sanitizer;
+
+#define INTERFACE extern "C" SANITIZER_INTERFACE_ATTRIBUTE
+
+extern "C" {
+// Externally initialized by the sanitizer, keeps one table per active device.
+[[gnu::visibility("protected")]] u64 *__csan_watchpoint_table = nullptr;
+}
+
+static constexpr u64 SAMPLE_DELAY_MIN_NS = 1000;
+static constexpr u64 SAMPLE_DELAY_MAX_NS = 10000;
+static constexpr u32 WP_CHANCE = 8;
+static_assert((CSAN_WATCHPOINT_TABLE_ENTRIES &
+               (CSAN_WATCHPOINT_TABLE_ENTRIES - 1)) == 0,
+              "watchpoint table size must be a power of two");
+static_assert(WP_CHANCE >= 2 && (WP_CHANCE & (WP_CHANCE - 1)) == 0,
+              "WP_CHANCE must be a power of two");
+
+// The GPU case does
+static constexpr u32 GPU_MAX_ACCESS_SIZE = 16;
+static constexpr u32 GPU_WATCHPOINT_ENTRIES = CSAN_WATCHPOINT_TABLE_ENTRIES;
+static constexpr u32 GPU_CHECK_ADJACENT_SLOTS = 0;
+static_assert(GPU_WATCHPOINT_ENTRIES * sizeof(u64) == 2 * 1024 * 1024,
+              "GPU watchpoint table must be 2 MiB");
+using GpuWatchpointTable =
+    __csan::WatchpointTable<GPU_MAX_ACCESS_SIZE, GPU_CHECK_ADJACENT_SLOTS>;
+
+static GpuWatchpointTable get_watchpoints() {
+  return GpuWatchpointTable(__csan_watchpoint_table);
+}
+
+// LDS addresses share the global watchpoint table. The original address is
+// combined with the block's linear ID to create a unique global address.
+static constexpr u32 LDS_OFFSET_BITS = 20;
+static constexpr u64 LDS_OFFSET_MASK = (1ull << LDS_OFFSET_BITS) - 1;
+static constexpr u64 LDS_FLAG = 1ull << (GpuWatchpointTable::AddressBits - 1);
+static constexpr u64 GLOBAL_ADDRESS_MASK = LDS_FLAG - 1;
+static constexpr u64 LDS_MAX_BLOCKS =
+    1ull << (GpuWatchpointTable::AddressBits - 1 - LDS_OFFSET_BITS);
+static_assert((LDS_FLAG | ((LDS_MAX_BLOCKS - 1) << LDS_OFFSET_BITS) |
+               LDS_OFFSET_MASK) == GpuWatchpointTable::AddressMask,
+              "LDS key fields must fill the address bits");
+
+[[gnu::visibility("protected"),
+  gnu::weak]] rpc::Client client asm("__llvm_rpc_client");
+
+static u64 __csan_num_data_races = 0;
+
+INTERFACE u64 __csan_get_num_data_races() {
+  return __atomic_load_n(&__csan_num_data_races, __ATOMIC_RELAXED);
+}
+
+// Shallow deduplication check to save the host thread work. Keyed on both the
+// PC and the race kind so each distinct kind of race at a PC is reported once.
+static bool should_report(void *pc, unsigned kind) {
+  static u64 seen[64] = {};
+  const u64 token = (reinterpret_cast<uptr>(pc) >> 4) ^
+                    (static_cast<u64>(kind) * 0x9E3779B97F4A7C15ull);
+  u64 idx = (token * 0x9E3779B97F4A7C15ull) >> 58;
+  u64 last = __scoped_atomic_exchange_n(&seen[idx], token, __ATOMIC_RELAXED,
+                                        __MEMORY_SCOPE_DEVICE);
+  return last != token;
+}
+
+[[gnu::cold, gnu::noinline]] static void
+report(unsigned kind, uptr addr, u32 size, int access_type, uptr pc,
+       void *peer = nullptr, int peer_access = 0, u32 peer_size = 0,
+       u8 peer_lane = 0) {
+  pc = pc ? pc : GET_CALLER_PC();
+  if (!should_report(reinterpret_cast<void *>(pc), kind))
+    return;
+
+  __csan_gpu_race rep = {};
+  rep.pc = pc;
+  rep.peer_pc = reinterpret_cast<uptr>(peer);
+  rep.addr = addr;
+  rep.size = size;
+  rep.access_type = static_cast<unsigned>(access_type);
+  rep.kind = kind;
+  rep.block[0] = __gpu_block_id(__GPU_X_DIM);
+  rep.block[1] = __gpu_block_id(__GPU_Y_DIM);
+  rep.block[2] = __gpu_block_id(__GPU_Z_DIM);
+  rep.thread[0] = __gpu_thread_id(__GPU_X_DIM);
+  rep.thread[1] = __gpu_thread_id(__GPU_Y_DIM);
+  rep.thread[2] = __gpu_thread_id(__GPU_Z_DIM);
+  rep.lane = __gpu_lane_id();
+  rep.peer_lane = peer_lane;
+  rep.peer_access_type = static_cast<u8>(peer_access);
+  rep.peer_size = static_cast<u8>(peer_size);
+
+  rpc::Client::Port Port = client.open<SANITIZER_OFFLOAD_CSAN>();
+  Port.send([&](rpc::Buffer *buf, u32) {
+    __builtin_memcpy(buf->data, &rep, sizeof(rep));
+  });
+  static_assert(sizeof(__csan_gpu_race) <= sizeof(rpc::Buffer),
+                "Report must fit in a single packet");
+
+  __scoped_atomic_fetch_add(&__csan_num_data_races, 1, __ATOMIC_RELAXED,
+                            __MEMORY_SCOPE_DEVICE);
+}
+
+#if defined(__AMDGPU__)
+// AMDGPU does not have a single set frequency. Different architectures and
+// cards can have different values. A frequency of 100MHz is most common so we
+// use it, if it is wrong it just means we sleep longer than expected.
+static constexpr u64 CLOCK_FREQ_HZ = 100000000UL;
+#else
+static constexpr u64 CLOCK_FREQ_HZ = 1000000000UL;
+#endif
+static constexpr u64 TICKS_PER_SEC = 1000000000UL;
+
+// FIXME: Avoids emitting an unresolved reference to the OCLC ABI version.
+static u32 num_blocks(int dim) {
+#ifdef __AMDGPU__
+  return ((const u32 __gpu_constant *)__builtin_amdgcn_implicitarg_ptr())[dim];
----------------
yxsamliu wrote:

`block_count` counts only full workgroups. A non-zero remainder adds another partial group, so this stride can be too small. For global size `(3, 2)` and workgroup size `(2, 1)`, groups `(1, 0)` and `(0, 1)` get the same ID. Could we follow libclc’s `get_num_groups` logic and add a multidimensional partial-group test?

https://github.com/llvm/llvm-project/pull/225782


More information about the llvm-branch-commits mailing list