[llvm-branch-commits] [compiler-rt] [llvm] [PGO] Add GPU wave counters (PR #225589)

Yaxun Liu via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Mon Sep 28 18:03:01 PDT 2026


https://github.com/yxsamliu updated https://github.com/llvm/llvm-project/pull/225589

>From 29cd2c522b10b140cbae4d39a8a674accb7fc0f9 Mon Sep 17 00:00:00 2001
From: "Yaxun (Sam) Liu" <yaxun.liu at amd.com>
Date: Tue, 22 Sep 2026 22:25:58 -0400
Subject: [PATCH 1/3] [PGO] Add GPU wave counters

Lane counts do not show how often a GPU wave reaches an instrumentation
point. A divergent wave can execute both paths even when few lanes take
one path.

Extend the GPU profiling helper to collect a wave count alongside each
existing lane counter. The first active lane increments the wave counter
once, using the same workgroup sampling decision as the lane counters.
Collect both channels together so shared functions use a consistent
counter layout across translation units. Workgroup sampling controls
profiling overhead.

Carry wave counts through raw and indexed profiles and weighted merging
as a separate channel. Preserve ordinary lane-flow counts and leave
blocks without instrumentation unmeasured.
---
 compiler-rt/include/profile/InstrProfData.inc |   7 +-
 compiler-rt/lib/profile/InstrProfiling.h      |   6 +-
 compiler-rt/lib/profile/InstrProfilingMerge.c |   1 +
 .../lib/profile/InstrProfilingPlatformGPU.c   |   6 +-
 llvm/docs/InstrProfileFormat.md               |  32 +++++++
 llvm/include/llvm/ProfileData/InstrProf.h     |  13 ++-
 .../llvm/ProfileData/InstrProfData.inc        |   7 +-
 .../llvm/ProfileData/InstrProfWriter.h        |   8 +-
 llvm/lib/ProfileData/InstrProf.cpp            |  82 +++++++++---------
 llvm/lib/ProfileData/InstrProfReader.cpp      |  25 +++++-
 llvm/lib/ProfileData/InstrProfWriter.cpp      |  36 ++++++--
 .../Instrumentation/InstrProfiling.cpp        |  68 +++++++++++----
 .../InstrProfiling/amdgpu-3d-grid.ll          |   4 +-
 .../amdgpu-contiguous-counters.ll             |   8 +-
 .../InstrProfiling/amdgpu-uniform-counters.ll |   4 +-
 .../InstrProfiling/amdgpu-wave32.ll           |   6 +-
 .../InstrProfiling/amdgpu-wave64.ll           |   6 +-
 .../InstrProfiling/coverage.ll                |   8 +-
 .../InstrProfiling/gpu-wave-convergence.ll    |  14 +++
 .../InstrProfiling/gpu-wave-counts.ll         |  38 ++++++++
 .../InstrProfiling/gpu-wave-inline.ll         |  34 ++++++++
 .../InstrProfiling/gpu-wave-link.ll           |  51 +++++++++++
 .../Transforms/PGOProfile/comdat_internal.ll  |   4 +-
 .../Transforms/PGOProfile/gpu-wave-counts.ll  |  38 ++++++++
 .../instrprof_burst_sampling_fast.ll          |   2 +-
 .../Transforms/PGOProfile/vtable_profile.ll   |   2 +-
 .../llvm-profdata/Inputs/c-general.profraw    | Bin 2152 -> 2248 bytes
 .../llvm-profdata/binary-ids-padding.test     |   8 +-
 .../tools/llvm-profdata/gpu-wave-counts.test  |  50 +++++++++++
 .../insufficient-binary-ids-size.test         |   2 +-
 .../llvm-profdata/large-binary-id-size.test   |   2 +-
 ...alformed-not-space-for-another-header.test |   4 +-
 .../malformed-num-counters-zero.test          |   8 +-
 .../malformed-ptr-to-counter-array.test       |   4 +-
 .../malformed-uniform-counter-array.test      |  12 +--
 .../misaligned-binary-ids-size.test           |   2 +-
 .../tools/llvm-profdata/profile-version.test  |   2 +-
 .../tools/llvm-profdata/raw-32-bits-be.test   |   2 +-
 .../tools/llvm-profdata/raw-32-bits-le.test   |   2 +-
 .../tools/llvm-profdata/raw-64-bits-be.test   |  10 +--
 .../tools/llvm-profdata/raw-64-bits-le.test   |  10 +--
 .../tools/llvm-profdata/raw-two-profiles.test |   8 +-
 .../llvm-profdata/truncated-profile.test      |   6 +-
 llvm/tools/llvm-profdata/llvm-profdata.cpp    |  32 ++++---
 llvm/unittests/ProfileData/InstrProfTest.cpp  |  60 +++++++++++++
 45 files changed, 577 insertions(+), 157 deletions(-)
 create mode 100644 llvm/test/Instrumentation/InstrProfiling/gpu-wave-convergence.ll
 create mode 100644 llvm/test/Instrumentation/InstrProfiling/gpu-wave-counts.ll
 create mode 100644 llvm/test/Instrumentation/InstrProfiling/gpu-wave-inline.ll
 create mode 100644 llvm/test/Instrumentation/InstrProfiling/gpu-wave-link.ll
 create mode 100644 llvm/test/Transforms/PGOProfile/gpu-wave-counts.ll
 create mode 100644 llvm/test/tools/llvm-profdata/gpu-wave-counts.test

diff --git a/compiler-rt/include/profile/InstrProfData.inc b/compiler-rt/include/profile/InstrProfData.inc
index b64026b6798f4..c1957f665cb98 100644
--- a/compiler-rt/include/profile/InstrProfData.inc
+++ b/compiler-rt/include/profile/InstrProfData.inc
@@ -99,6 +99,9 @@ INSTR_PROF_DATA(const uint16_t, llvm::Type::getInt16Ty(Ctx), \
                                  OffloadDeviceWaveSizeVal))
 INSTR_PROF_DATA(const uint32_t, llvm::Type::getInt32Ty(Ctx), NumBitmapBytes, \
                 ConstantInt::get(llvm::Type::getInt32Ty(Ctx), NumBitmapBytes))
+/* The last NumWaveCounters entries in CounterPtr count GPU wave visits. */
+INSTR_PROF_DATA(const uint32_t, llvm::Type::getInt32Ty(Ctx), NumWaveCounters, \
+                ConstantInt::get(llvm::Type::getInt32Ty(Ctx), NumWaveCounters))
 #undef INSTR_PROF_DATA
 /* INSTR_PROF_DATA end. */
 
@@ -773,9 +776,9 @@ serializeValueProfDataFrom(ValueProfRecordClosure *Closure,
         (uint64_t)'f' << 16 | (uint64_t)'R' << 8 | (uint64_t)129
 
 /* Raw profile format version (start from 1). */
-#define INSTR_PROF_RAW_VERSION 11
+#define INSTR_PROF_RAW_VERSION 12
 /* Indexed profile format version (start from 1). */
-#define INSTR_PROF_INDEX_VERSION 14
+#define INSTR_PROF_INDEX_VERSION 15
 /* Coverage mapping format version (start from 0). */
 #define INSTR_PROF_COVMAP_VERSION 6
 
diff --git a/compiler-rt/lib/profile/InstrProfiling.h b/compiler-rt/lib/profile/InstrProfiling.h
index 2c40476e02702..e3d312ba2e966 100644
--- a/compiler-rt/lib/profile/InstrProfiling.h
+++ b/compiler-rt/lib/profile/InstrProfiling.h
@@ -178,11 +178,11 @@ void __llvm_profile_instrument_target_value(uint64_t TargetValue, void *Data,
  * \brief Wave-cooperative counter increment for GPU targets.
  *
  * Reduces per-lane atomic contention by electing a single lane per wave to
- * perform the counter update. \c Uniform is an optional counter tracking the
- * number of uniform.
+ * perform the counter updates. \c Uniform optionally counts lane executions
+ * with a full active mask, and \c Wave counts wave executions.
  */
 void INSTR_PROF_INSTRUMENT_GPU_FUNC(uint64_t *Counter, uint64_t *Uniform,
-                                    uint64_t Step);
+                                    uint64_t Step, uint64_t *Wave);
 
 /*!
  * \brief Write instrumentation data to the current file.
diff --git a/compiler-rt/lib/profile/InstrProfilingMerge.c b/compiler-rt/lib/profile/InstrProfilingMerge.c
index ff84cf8633cf0..58f40e7df509d 100644
--- a/compiler-rt/lib/profile/InstrProfilingMerge.c
+++ b/compiler-rt/lib/profile/InstrProfilingMerge.c
@@ -90,6 +90,7 @@ int __llvm_profile_check_compatibility(const char *ProfileData,
     if (SrcData->NameRef != DstData->NameRef ||
         SrcData->FuncHash != DstData->FuncHash ||
         SrcData->NumCounters != DstData->NumCounters ||
+        SrcData->NumWaveCounters != DstData->NumWaveCounters ||
         SrcData->NumBitmapBytes != DstData->NumBitmapBytes)
       return 1;
   }
diff --git a/compiler-rt/lib/profile/InstrProfilingPlatformGPU.c b/compiler-rt/lib/profile/InstrProfilingPlatformGPU.c
index 22b4c4cdb11e0..5997dee5fa7b2 100644
--- a/compiler-rt/lib/profile/InstrProfilingPlatformGPU.c
+++ b/compiler-rt/lib/profile/InstrProfilingPlatformGPU.c
@@ -29,10 +29,11 @@ static int is_uniform(uint64_t mask) {
 
 // Wave-cooperative counter increment. The instrumentation pass emits calls to
 // this in place of the default non-atomic load/add/store or atomicrmw sequence.
-// The optional uniform counter allows calculating wave uniformity if present.
+// The uniform counter is optional; the wave counter records every wave visit.
 COMPILER_RT_VISIBILITY void INSTR_PROF_INSTRUMENT_GPU_FUNC(uint64_t *counter,
                                                            uint64_t *uniform,
-                                                           uint64_t step) {
+                                                           uint64_t step,
+                                                           uint64_t *wave) {
   uint64_t mask = __gpu_lane_mask();
   if (__gpu_is_first_in_lane(mask)) {
     __scoped_atomic_fetch_add(counter, step * __builtin_popcountg(mask),
@@ -40,6 +41,7 @@ COMPILER_RT_VISIBILITY void INSTR_PROF_INSTRUMENT_GPU_FUNC(uint64_t *counter,
     if (uniform && is_uniform(mask))
       __scoped_atomic_fetch_add(uniform, step * __builtin_popcountg(mask),
                                 __ATOMIC_RELAXED, __MEMORY_SCOPE_DEVICE);
+    __scoped_atomic_fetch_add(wave, 1, __ATOMIC_RELAXED, __MEMORY_SCOPE_DEVICE);
   }
 }
 
diff --git a/llvm/docs/InstrProfileFormat.md b/llvm/docs/InstrProfileFormat.md
index 336139fefd73f..3a53a6b77066e 100644
--- a/llvm/docs/InstrProfileFormat.md
+++ b/llvm/docs/InstrProfileFormat.md
@@ -1,5 +1,37 @@
 # Instrumentation Profile Format
 
+## Experimental GPU wave counters
+
+Raw format version 12 appends a `uint32_t NumWaveCounters` field to each
+profile data record. `NumCounters` is the total number of 64-bit lane and
+wave counters addressed by `CounterPtr`; its final `NumWaveCounters` entries
+are wave visits. The uniform-counter array still has only
+`NumCounters - NumWaveCounters` entries. A nonempty wave tail must leave at
+least one lane counter. This changes the raw record layout and requires a
+matching compiler and profiling runtime.
+
+Indexed format version 15 stores a little-endian 64-bit wave-counter count
+followed by that many 64-bit values after the padded uniformity vector and
+before value-profile data. Zero indicates no wave profile. Readers expose
+the tail separately from the lane-flow counter vector. Merging requires
+matching wave-vector lengths; weighted merging and scaling use saturating
+arithmetic, as for ordinary counters. Text and previous-version export of
+wave profiles is currently unsupported and diagnosed.
+
+GPU counter instrumentation always collects wave counts. Each existing
+instrumentation counter has a wave counter at the same index. The GPU runtime
+updates lane, uniformity, and wave counters in one call: the first active lane
+adds the active-lane count to the lane counter and one to the wave counter.
+Workgroup sampling controls the overhead of all three channels. No extra instrumentation
+points are inserted; blocks without a counter have no direct wave measurement.
+
+Wave counts observe the active groups that execute each instrumentation point.
+They do not require full waves or preserve lane grouping across compiler
+transformations. Both sides of a divergent branch may execute once per wave,
+so wave counts must not participate in scalar flow reconstruction. Existing
+lane counts, uniformity classification, and optimization consumers retain
+their previous meaning.
+
 
 ## Overview
 
diff --git a/llvm/include/llvm/ProfileData/InstrProf.h b/llvm/include/llvm/ProfileData/InstrProf.h
index bff17fcf64781..9381d940f3eab 100644
--- a/llvm/include/llvm/ProfileData/InstrProf.h
+++ b/llvm/include/llvm/ProfileData/InstrProf.h
@@ -907,6 +907,8 @@ struct InstrProfValueSiteRecord {
 /// Profiling information for a single function.
 struct InstrProfRecord {
   std::vector<uint64_t> Counts;
+  /// GPU wave visits at each counter index, separate from lane-flow counts.
+  std::vector<uint64_t> WaveCounts;
   std::vector<uint8_t> BitmapBytes;
   /// For AMDGPU offload profiling: raw or merged uniform counters. One uint64_t
   /// per instrumented block, tracking entries where all lanes were active.
@@ -924,8 +926,9 @@ struct InstrProfRecord {
       : Counts(std::move(Counts)), BitmapBytes(std::move(BitmapBytes)) {}
   InstrProfRecord(InstrProfRecord &&) = default;
   InstrProfRecord(const InstrProfRecord &RHS)
-      : Counts(RHS.Counts), BitmapBytes(RHS.BitmapBytes),
-        UniformCounts(RHS.UniformCounts), UniformityBits(RHS.UniformityBits),
+      : Counts(RHS.Counts), WaveCounts(RHS.WaveCounts),
+        BitmapBytes(RHS.BitmapBytes), UniformCounts(RHS.UniformCounts),
+        UniformityBits(RHS.UniformityBits),
         OffloadDeviceWaveSize(RHS.OffloadDeviceWaveSize),
         ValueData(RHS.ValueData
                       ? std::make_unique<ValueProfData>(*RHS.ValueData)
@@ -933,6 +936,7 @@ struct InstrProfRecord {
   InstrProfRecord &operator=(InstrProfRecord &&) = default;
   InstrProfRecord &operator=(const InstrProfRecord &RHS) {
     Counts = RHS.Counts;
+    WaveCounts = RHS.WaveCounts;
     BitmapBytes = RHS.BitmapBytes;
     UniformCounts = RHS.UniformCounts;
     UniformityBits = RHS.UniformityBits;
@@ -1004,6 +1008,7 @@ struct InstrProfRecord {
   /// Clear value data entries, edge counters, and uniformity data.
   void Clear() {
     Counts.clear();
+    WaveCounts.clear();
     UniformCounts.clear();
     UniformityBits.clear();
     OffloadDeviceWaveSize = 0;
@@ -1231,7 +1236,9 @@ enum ProfVersion {
   Version13 = 13,
   // UniformityBits added for AMDGPU offload profiling divergence detection.
   Version14 = 14,
-  // The current version is 14.
+  // GPU wave counts added to record data.
+  Version15 = 15,
+  // The current version is 15.
   CurrentVersion = INSTR_PROF_INDEX_VERSION
 };
 const uint64_t Version = ProfVersion::CurrentVersion;
diff --git a/llvm/include/llvm/ProfileData/InstrProfData.inc b/llvm/include/llvm/ProfileData/InstrProfData.inc
index b64026b6798f4..c1957f665cb98 100644
--- a/llvm/include/llvm/ProfileData/InstrProfData.inc
+++ b/llvm/include/llvm/ProfileData/InstrProfData.inc
@@ -99,6 +99,9 @@ INSTR_PROF_DATA(const uint16_t, llvm::Type::getInt16Ty(Ctx), \
                                  OffloadDeviceWaveSizeVal))
 INSTR_PROF_DATA(const uint32_t, llvm::Type::getInt32Ty(Ctx), NumBitmapBytes, \
                 ConstantInt::get(llvm::Type::getInt32Ty(Ctx), NumBitmapBytes))
+/* The last NumWaveCounters entries in CounterPtr count GPU wave visits. */
+INSTR_PROF_DATA(const uint32_t, llvm::Type::getInt32Ty(Ctx), NumWaveCounters, \
+                ConstantInt::get(llvm::Type::getInt32Ty(Ctx), NumWaveCounters))
 #undef INSTR_PROF_DATA
 /* INSTR_PROF_DATA end. */
 
@@ -773,9 +776,9 @@ serializeValueProfDataFrom(ValueProfRecordClosure *Closure,
         (uint64_t)'f' << 16 | (uint64_t)'R' << 8 | (uint64_t)129
 
 /* Raw profile format version (start from 1). */
-#define INSTR_PROF_RAW_VERSION 11
+#define INSTR_PROF_RAW_VERSION 12
 /* Indexed profile format version (start from 1). */
-#define INSTR_PROF_INDEX_VERSION 14
+#define INSTR_PROF_INDEX_VERSION 15
 /* Coverage mapping format version (start from 0). */
 #define INSTR_PROF_COVMAP_VERSION 6
 
diff --git a/llvm/include/llvm/ProfileData/InstrProfWriter.h b/llvm/include/llvm/ProfileData/InstrProfWriter.h
index 73fb9432597cf..89e8d3e322c1e 100644
--- a/llvm/include/llvm/ProfileData/InstrProfWriter.h
+++ b/llvm/include/llvm/ProfileData/InstrProfWriter.h
@@ -153,10 +153,10 @@ class InstrProfWriter {
   LLVM_ABI Error validateRecord(const InstrProfRecord &Func);
 
   /// Write \c Record in text format to \c OS
-  LLVM_ABI static void writeRecordInText(StringRef Name, uint64_t Hash,
-                                         const InstrProfRecord &Counters,
-                                         InstrProfSymtab &Symtab,
-                                         raw_fd_ostream &OS);
+  LLVM_ABI static Error writeRecordInText(StringRef Name, uint64_t Hash,
+                                          const InstrProfRecord &Counters,
+                                          InstrProfSymtab &Symtab,
+                                          raw_fd_ostream &OS);
 
   /// Write the profile, returning the raw data. For testing.
   LLVM_ABI std::unique_ptr<MemoryBuffer> writeBuffer();
diff --git a/llvm/lib/ProfileData/InstrProf.cpp b/llvm/lib/ProfileData/InstrProf.cpp
index ca5b17aa3c248..2ffe6ff59cd5c 100644
--- a/llvm/lib/ProfileData/InstrProf.cpp
+++ b/llvm/lib/ProfileData/InstrProf.cpp
@@ -994,6 +994,23 @@ static void mergeUniformityBits(std::vector<uint8_t> &Dst,
     Dst[I] &= Src[I];
 }
 
+static void mergeCounts(MutableArrayRef<uint64_t> Dst, ArrayRef<uint64_t> Src,
+                        uint64_t Weight,
+                        function_ref<void(instrprof_error)> Warn) {
+  assert(Dst.size() == Src.size() && "counter sizes must match");
+  for (size_t I = 0, E = Src.size(); I < E; ++I) {
+    bool Overflowed;
+    uint64_t Value = SaturatingMultiplyAdd(Src[I], Weight, Dst[I], &Overflowed);
+    if (Value > getInstrMaxCountValue()) {
+      Value = getInstrMaxCountValue();
+      Overflowed = true;
+    }
+    Dst[I] = Value;
+    if (Overflowed)
+      Warn(instrprof_error::counter_overflow);
+  }
+}
+
 void InstrProfRecord::merge(InstrProfRecord &Other, uint64_t Weight,
                             function_ref<void(instrprof_error)> Warn) {
   // If the number of counters doesn't match we either have bad data
@@ -1002,6 +1019,11 @@ void InstrProfRecord::merge(InstrProfRecord &Other, uint64_t Weight,
     Warn(instrprof_error::count_mismatch);
     return;
   }
+  // Reject partial wave profiles rather than mixing different populations.
+  if (WaveCounts.size() != Other.WaveCounts.size()) {
+    Warn(instrprof_error::count_mismatch);
+    return;
+  }
 
   computeBlockUniformity();
   Other.computeBlockUniformity();
@@ -1026,35 +1048,14 @@ void InstrProfRecord::merge(InstrProfRecord &Other, uint64_t Weight,
   OffloadDeviceWaveSize = Other.OffloadDeviceWaveSize;
   bool HasUniformCounts = !UniformCounts.empty();
   bool OtherHasUniformCounts = !Other.UniformCounts.empty();
-  for (size_t I = 0, E = Other.Counts.size(); I < E; ++I) {
-    bool Overflowed;
-    uint64_t Value =
-        SaturatingMultiplyAdd(Other.Counts[I], Weight, Counts[I], &Overflowed);
-    if (Value > getInstrMaxCountValue()) {
-      Value = getInstrMaxCountValue();
-      Overflowed = true;
-    }
-    Counts[I] = Value;
-    if (Overflowed)
-      Warn(instrprof_error::counter_overflow);
-  }
+  mergeCounts(Counts, Other.Counts, Weight, Warn);
 
   if (HasUniformCounts && OtherHasUniformCounts) {
     if (UniformCounts.size() != Other.UniformCounts.size()) {
       UniformCounts.clear();
       UniformityBits.clear();
     } else {
-      for (size_t I = 0, E = Other.UniformCounts.size(); I < E; ++I) {
-        bool Overflowed;
-        UniformCounts[I] = SaturatingMultiplyAdd(Other.UniformCounts[I], Weight,
-                                                 UniformCounts[I], &Overflowed);
-        if (UniformCounts[I] > getInstrMaxCountValue()) {
-          UniformCounts[I] = getInstrMaxCountValue();
-          Overflowed = true;
-        }
-        if (Overflowed)
-          Warn(instrprof_error::counter_overflow);
-      }
+      mergeCounts(UniformCounts, Other.UniformCounts, Weight, Warn);
       computeBlockUniformity();
     }
   } else {
@@ -1062,6 +1063,8 @@ void InstrProfRecord::merge(InstrProfRecord &Other, uint64_t Weight,
     mergeUniformityBits(UniformityBits, Other.UniformityBits);
   }
 
+  mergeCounts(WaveCounts, Other.WaveCounts, Weight, Warn);
+
   // If the number of bitmap bytes doesn't match we either have bad data
   // or a hash collision.
   if (BitmapBytes.size() != Other.BitmapBytes.size()) {
@@ -1088,25 +1091,17 @@ void InstrProfRecord::scaleValueProfData(
 void InstrProfRecord::scale(uint64_t N, uint64_t D,
                             function_ref<void(instrprof_error)> Warn) {
   assert(D != 0 && "D cannot be 0");
-  for (auto &Count : this->Counts) {
-    bool Overflowed;
-    Count = SaturatingMultiply(Count, N, &Overflowed) / D;
-    if (Count > getInstrMaxCountValue()) {
-      Count = getInstrMaxCountValue();
-      Overflowed = true;
-    }
-    if (Overflowed)
-      Warn(instrprof_error::counter_overflow);
-  }
-  for (auto &Count : this->UniformCounts) {
-    bool Overflowed;
-    Count = SaturatingMultiply(Count, N, &Overflowed) / D;
-    if (Count > getInstrMaxCountValue()) {
-      Count = getInstrMaxCountValue();
-      Overflowed = true;
+  for (auto *Counters : {&Counts, &UniformCounts, &WaveCounts}) {
+    for (auto &Count : *Counters) {
+      bool Overflowed;
+      Count = SaturatingMultiply(Count, N, &Overflowed) / D;
+      if (Count > getInstrMaxCountValue()) {
+        Count = getInstrMaxCountValue();
+        Overflowed = true;
+      }
+      if (Overflowed)
+        Warn(instrprof_error::counter_overflow);
     }
-    if (Overflowed)
-      Warn(instrprof_error::counter_overflow);
   }
   computeBlockUniformity();
   for (uint32_t Kind = IPVK_First; Kind <= IPVK_Last; ++Kind)
@@ -1799,7 +1794,7 @@ Expected<Header> Header::readFromBuffer(const unsigned char *Buffer) {
       IndexedInstrProf::ProfVersion::CurrentVersion)
     return make_error<InstrProfError>(instrprof_error::unsupported_version);
 
-  static_assert(IndexedInstrProf::ProfVersion::CurrentVersion == Version14,
+  static_assert(IndexedInstrProf::ProfVersion::CurrentVersion == Version15,
                 "Please update the reader as needed when a new field is added "
                 "or when indexed profile version gets bumped.");
 
@@ -1832,10 +1827,11 @@ size_t Header::size() const {
     // of the header, and byte offset of existing fields shouldn't change when
     // indexed profile version gets incremented.
     static_assert(
-        IndexedInstrProf::ProfVersion::CurrentVersion == Version14,
+        IndexedInstrProf::ProfVersion::CurrentVersion == Version15,
         "Please update the size computation below if a new field has "
         "been added to the header; for a version bump without new "
         "fields, add a case statement to fall through to the latest version.");
+  case 15ull: // Wave counts added in record data, no header change
   case 14ull: // UniformityBits added in record data, no header change
   case 13ull:
   case 12ull:
diff --git a/llvm/lib/ProfileData/InstrProfReader.cpp b/llvm/lib/ProfileData/InstrProfReader.cpp
index 4b97546abc068..536acc2eab4de 100644
--- a/llvm/lib/ProfileData/InstrProfReader.cpp
+++ b/llvm/lib/ProfileData/InstrProfReader.cpp
@@ -874,7 +874,7 @@ Error RawInstrProfReader<IntPtrT>::readRawUniformCounters(
   if (UniformCountersStart == UniformCountersEnd)
     return success();
 
-  uint32_t NumCounters = swap(Data->NumCounters);
+  uint32_t NumCounters = swap(Data->NumCounters) - swap(Data->NumWaveCounters);
 
   ptrdiff_t UniformCounterOffset =
       swap(Data->UniformCounterPtr) - UniformCountersDelta;
@@ -964,6 +964,14 @@ Error RawInstrProfReader<IntPtrT>::readNextRecord(NamedInstrProfRecord &Record)
   if (Error E = readRawCounts(Record))
     return error(std::move(E));
 
+  uint32_t NumWaveCounters = swap(Data->NumWaveCounters);
+  if (NumWaveCounters && NumWaveCounters >= Record.Counts.size())
+    return error(instrprof_error::malformed, "invalid number of wave counters");
+  size_t NumLaneCounters = Record.Counts.size() - NumWaveCounters;
+  Record.WaveCounts.assign(Record.Counts.begin() + NumLaneCounters,
+                           Record.Counts.end());
+  Record.Counts.resize(NumLaneCounters);
+
   // Read raw bitmap bytes and set Record.
   if (Error E = readRawBitmapBytes(Record))
     return error(std::move(E));
@@ -1115,6 +1123,21 @@ data_type InstrProfLookupTrait::ReadData(StringRef K, const unsigned char *D,
                             std::move(BitmapByteBuffer),
                             std::move(UniformityBitsBuffer));
 
+    if (GET_VERSION(FormatVersion) >=
+        IndexedInstrProf::ProfVersion::Version15) {
+      if (End - D < ptrdiff_t(sizeof(uint64_t)))
+        return data_type();
+      uint64_t NumWaveCounters =
+          endian::readNext<uint64_t, llvm::endianness::little>(D);
+      if (NumWaveCounters > uint64_t(End - D) / sizeof(uint64_t))
+        return data_type();
+      auto &WaveCounts = DataBuffer.back().WaveCounts;
+      WaveCounts.reserve(NumWaveCounters);
+      for (uint64_t I = 0; I < NumWaveCounters; ++I)
+        WaveCounts.push_back(
+            endian::readNext<uint64_t, llvm::endianness::little>(D));
+    }
+
     // Read value profiling data.
     if (GET_VERSION(FormatVersion) > IndexedInstrProf::ProfVersion::Version2 &&
         !readValueProfilingData(D, End)) {
diff --git a/llvm/lib/ProfileData/InstrProfWriter.cpp b/llvm/lib/ProfileData/InstrProfWriter.cpp
index 52b575ad948a9..0586071faf783 100644
--- a/llvm/lib/ProfileData/InstrProfWriter.cpp
+++ b/llvm/lib/ProfileData/InstrProfWriter.cpp
@@ -83,6 +83,7 @@ class InstrProfRecordWriterTrait {
         M += alignTo(ProfRecord.BitmapBytes.size(), sizeof(uint64_t));
         M += sizeof(uint64_t); // The size of the UniformityBits vector
         M += alignTo(ProfRecord.UniformityBits.size(), sizeof(uint64_t));
+        M += sizeof(uint64_t) * (1 + ProfRecord.WaveCounts.size());
       }
 
       // Value data
@@ -135,6 +136,10 @@ class InstrProfRecordWriterTrait {
              I < alignTo(ProfRecord.UniformityBits.size(), sizeof(uint64_t));
              ++I)
           LE.write<uint8_t>(0);
+
+        LE.write<uint64_t>(ProfRecord.WaveCounts.size());
+        for (uint64_t Count : ProfRecord.WaveCounts)
+          LE.write<uint64_t>(Count);
       }
 
       // Write value data
@@ -539,6 +544,16 @@ Error InstrProfWriter::writeVTableNames(ProfOStream &OS) {
 }
 
 Error InstrProfWriter::writeImpl(ProfOStream &OS) {
+  if (WritePrevVersion) {
+    for (const auto &Function : FunctionData) {
+      for (const auto &Record : Function.getValue()) {
+        if (!Record.second.WaveCounts.empty())
+          return make_error<InstrProfError>(
+              instrprof_error::unsupported_version,
+              "wave counts require indexed profile version 15");
+      }
+    }
+  }
   using namespace IndexedInstrProf;
   using namespace support;
 
@@ -567,7 +582,7 @@ Error InstrProfWriter::writeImpl(ProfOStream &OS) {
   // The WritePrevVersion handling will either need to be removed or updated
   // if the version is advanced beyond 12.
   static_assert(IndexedInstrProf::ProfVersion::CurrentVersion ==
-                IndexedInstrProf::ProfVersion::Version14);
+                IndexedInstrProf::ProfVersion::Version15);
   if (static_cast<bool>(ProfileKind & InstrProfKind::IRInstrumentation))
     Header.Version |= VARIANT_MASK_IR_PROF;
   if (static_cast<bool>(ProfileKind & InstrProfKind::ContextSensitive))
@@ -730,10 +745,15 @@ Error InstrProfWriter::validateRecord(const InstrProfRecord &Func) {
   return Error::success();
 }
 
-void InstrProfWriter::writeRecordInText(StringRef Name, uint64_t Hash,
-                                        const InstrProfRecord &Func,
-                                        InstrProfSymtab &Symtab,
-                                        raw_fd_ostream &OS) {
+Error InstrProfWriter::writeRecordInText(StringRef Name, uint64_t Hash,
+                                         const InstrProfRecord &Func,
+                                         InstrProfSymtab &Symtab,
+                                         raw_fd_ostream &OS) {
+  if (!Func.WaveCounts.empty())
+    return make_error<InstrProfError>(
+        instrprof_error::unsupported_version,
+        "text profiles do not support wave counts");
+
   OS << Name << "\n";
   OS << "# Func Hash:\n" << Hash << "\n";
   OS << "# Num Counters:\n" << Func.Counts.size() << "\n";
@@ -755,7 +775,7 @@ void InstrProfWriter::writeRecordInText(StringRef Name, uint64_t Hash,
   uint32_t NumValueKinds = Func.getNumValueKinds();
   if (!NumValueKinds) {
     OS << "\n";
-    return;
+    return Error::success();
   }
 
   OS << "# Num Value Kinds:\n" << Func.getNumValueKinds() << "\n";
@@ -779,6 +799,7 @@ void InstrProfWriter::writeRecordInText(StringRef Name, uint64_t Hash,
   }
 
   OS << "\n";
+  return Error::success();
 }
 
 Error InstrProfWriter::writeText(raw_fd_ostream &OS) {
@@ -827,7 +848,8 @@ Error InstrProfWriter::writeText(raw_fd_ostream &OS) {
   for (const auto &record : OrderedFuncData) {
     const StringRef &Name = record.first;
     const FuncPair &Func = record.second;
-    writeRecordInText(Name, Func.first, Func.second, Symtab, OS);
+    if (Error E = writeRecordInText(Name, Func.first, Func.second, Symtab, OS))
+      return E;
   }
 
   for (const auto &record : OrderedFuncData) {
diff --git a/llvm/lib/Transforms/Instrumentation/InstrProfiling.cpp b/llvm/lib/Transforms/Instrumentation/InstrProfiling.cpp
index 4a40f826f9ca1..54b4bb02472f1 100644
--- a/llvm/lib/Transforms/Instrumentation/InstrProfiling.cpp
+++ b/llvm/lib/Transforms/Instrumentation/InstrProfiling.cpp
@@ -301,8 +301,15 @@ class InstrLowerer final {
     GlobalVariable *DataVar = nullptr;
     GlobalVariable *RegionBitmaps = nullptr;
     uint32_t NumBitmapBytes = 0;
+    // GPU region counters contain lane counts followed by one wave count for
+    // each lane-counter index. Other targets have no wave counters.
+    uint32_t NumWaveCounters = 0;
 
     PerFunctionProfileData() = default;
+
+    uint64_t getNumRegionCounters() const {
+      return cast<ArrayType>(RegionCounters->getValueType())->getNumElements();
+    }
   };
   DenseMap<GlobalVariable *, PerFunctionProfileData> ProfileDataMap;
   // Key is virtual table variable, value is 'VTableProfData' in the form of
@@ -1056,6 +1063,10 @@ bool InstrLowerer::lower() {
     InstrProfCntrInstBase *FirstProfInst = nullptr;
     for (BasicBlock &BB : F) {
       for (auto I = BB.begin(), E = BB.end(); I != E; I++) {
+        if (isGPUProfTarget(M) &&
+            (isa<InstrProfCoverInst>(I) || isa<InstrProfTimestampInst>(I)))
+          report_fatal_error("wave counts require ordinary counter increments",
+                             false);
         if (auto *Ind = dyn_cast<InstrProfValueProfileInst>(I))
           computeNumValueSiteCounts(Ind);
         else {
@@ -1071,9 +1082,8 @@ bool InstrLowerer::lower() {
 
     // Use a profile intrinsic to create the region counters and data variable.
     // Also create the data variable based on the MCDCParams.
-    if (FirstProfInst != nullptr) {
+    if (FirstProfInst != nullptr)
       static_cast<void>(getOrCreateRegionCounters(FirstProfInst));
-    }
   }
 
   if (EnableVTableValueProfiling)
@@ -1340,14 +1350,13 @@ void InstrLowerer::lowerIncrement(InstrProfIncrementInst *Inc) {
     auto *PtrTy = PointerType::getUnqual(Context);
 
     auto *Addr = getCounterAddress(Inc);
+    auto &PD = ProfileDataMap[Inc->getName()];
 
     // Store the device wave/warp size into the profile data struct once per
     // function. AMDGPU folds llvm.amdgcn.wavefrontsize to the subtarget's
     // constant; other GPUs use their fixed warp size.
     if (!Inv.WaveSizeStored) {
       Inv.WaveSizeStored = true;
-      GlobalVariable *NamePtr = Inc->getName();
-      auto &PD = ProfileDataMap[NamePtr];
       if (PD.DataVar) {
         IRBuilder<> EntryBuilder(&*F->getEntryBlock().getFirstInsertionPt());
         Value *WaveSize16 = nullptr;
@@ -1373,22 +1382,26 @@ void InstrLowerer::lowerIncrement(InstrProfIncrementInst *Inc) {
       }
     }
 
-    GlobalVariable *UniformCounters = getOrCreateUniformCounters(Inc);
-    Value *UniformAddrArg = ConstantPointerNull::get(PtrTy);
-    if (UniformCounters) {
-      Value *UniformIndices[] = {Builder.getInt32(0), Inc->getIndex()};
-      Value *UniformAddr = Builder.CreateInBoundsGEP(
-          UniformCounters->getValueType(), UniformCounters, UniformIndices,
-          "unifctr.addr");
-      UniformAddrArg =
-          Builder.CreatePointerBitCastOrAddrSpaceCast(UniformAddr, PtrTy);
-    }
+    // getCounterAddress creates the lane, uniform, and wave counters together.
+    GlobalVariable *UniformCounters = PD.UniformCounters;
+    Value *UniformIndices[] = {Builder.getInt32(0), Inc->getIndex()};
+    Value *UniformAddr = Builder.CreateInBoundsGEP(
+        UniformCounters->getValueType(), UniformCounters, UniformIndices,
+        "unifctr.addr");
+    Value *UniformAddrArg =
+        Builder.CreatePointerBitCastOrAddrSpaceCast(UniformAddr, PtrTy);
     Value *CastAddr = Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, PtrTy);
     Value *StepI64 =
         Builder.CreateZExtOrTrunc(Inc->getStep(), Int64Ty, "step.i64");
 
+    uint32_t WaveOffset = PD.getNumRegionCounters() - PD.NumWaveCounters;
+    uint32_t WaveIndex = WaveOffset + Inc->getIndex()->getZExtValue();
+    Value *WaveAddr = Builder.CreateConstInBoundsGEP2_32(
+        PD.RegionCounters->getValueType(), PD.RegionCounters, 0, WaveIndex);
+    WaveAddr = Builder.CreatePointerBitCastOrAddrSpaceCast(WaveAddr, PtrTy);
+
     auto *CalleeTy = FunctionType::get(Type::getVoidTy(Context),
-                                       {PtrTy, PtrTy, Int64Ty}, false);
+                                       {PtrTy, PtrTy, Int64Ty, PtrTy}, false);
     FunctionCallee Callee =
         M.getOrInsertFunction(RTLIB::RuntimeLibcallsInfo::getLibcallImplName(
                                   RTLIB::impl___llvm_profile_instrument_gpu),
@@ -1405,10 +1418,11 @@ void InstrLowerer::lowerIncrement(InstrProfIncrementInst *Inc) {
       HeadBuilder.CreateCondBr(Inv.Matched, ThenBB, ContBB);
 
       IRBuilder<> ThenBuilder(ThenBB);
-      ThenBuilder.CreateCall(Callee, {CastAddr, UniformAddrArg, StepI64});
+      ThenBuilder.CreateCall(Callee,
+                             {CastAddr, UniformAddrArg, StepI64, WaveAddr});
       ThenBuilder.CreateBr(ContBB);
     } else {
-      Builder.CreateCall(Callee, {CastAddr, UniformAddrArg, StepI64});
+      Builder.CreateCall(Callee, {CastAddr, UniformAddrArg, StepI64, WaveAddr});
     }
     Inc->eraseFromParent();
     return;
@@ -1919,7 +1933,8 @@ InstrLowerer::getOrCreateRegionBitmaps(InstrProfMCDCBitmapInstBase *Inc) {
 GlobalVariable *
 InstrLowerer::createRegionCounters(InstrProfCntrInstBase *Inc, StringRef Name,
                                    GlobalValue::LinkageTypes Linkage) {
-  uint64_t NumCounters = Inc->getNumCounters()->getZExtValue();
+  uint64_t NumCounters = Inc->getNumCounters()->getZExtValue() +
+                         ProfileDataMap[Inc->getName()].NumWaveCounters;
   auto &Ctx = M.getContext();
   GlobalVariable *GV;
   if (isa<InstrProfCoverInst>(Inc)) {
@@ -1948,6 +1963,20 @@ InstrLowerer::getOrCreateRegionCounters(InstrProfCntrInstBase *Inc) {
   if (PD.RegionCounters)
     return PD.RegionCounters;
 
+  if (isGPUProfTarget(M)) {
+    if (!isa<InstrProfIncrementInst>(Inc) || IsCS)
+      report_fatal_error("wave counts require ordinary counter increments",
+                         false);
+    if (ProfileCorrelate != InstrProfCorrelator::NONE || isSamplingEnabled())
+      report_fatal_error("wave counts do not support profile correlation "
+                         "or lane-level instrumentation sampling",
+                         false);
+    uint64_t NumCounters = Inc->getNumCounters()->getZExtValue();
+    if (NumCounters > UINT32_MAX / 2)
+      report_fatal_error("too many wave profiling counters", false);
+    PD.NumWaveCounters = NumCounters;
+  }
+
   // If RegionCounters doesn't already exist, create it by first setting up
   // the corresponding profile section.
   auto *CounterPtr = setupProfileSection(Inc, IPSK_cnts);
@@ -2098,7 +2127,8 @@ void InstrLowerer::createDataVariable(InstrProfCntrInstBase *Inc) {
         ValuesVar, PointerType::get(Fn->getContext(), 0));
   }
 
-  uint64_t NumCounters = Inc->getNumCounters()->getZExtValue();
+  uint32_t NumWaveCounters = PD.NumWaveCounters;
+  uint64_t NumCounters = PD.getNumRegionCounters();
 
   Constant *CounterPtr = PD.RegionCounters;
   Constant *UniformCounterPtr = PD.UniformCounters;
diff --git a/llvm/test/Instrumentation/InstrProfiling/amdgpu-3d-grid.ll b/llvm/test/Instrumentation/InstrProfiling/amdgpu-3d-grid.ll
index 11f6c39ba3a3d..880c8cb9f3c35 100644
--- a/llvm/test/Instrumentation/InstrProfiling/amdgpu-3d-grid.ll
+++ b/llvm/test/Instrumentation/InstrProfiling/amdgpu-3d-grid.ll
@@ -15,9 +15,9 @@ define amdgpu_kernel void @kernel_3d() {
 declare void @llvm.instrprof.increment(ptr, i64, i32, i32)
 
 ;; Per-function comdat counters (3D grid linearization is handled in the runtime library)
-; CHECK: @__profc_kernel_3d = linkonce_odr protected addrspace(1) global [1 x i64]
+; CHECK: @__profc_kernel_3d = linkonce_odr protected addrspace(1) global [2 x i64]
 ; CHECK: @__llvm_prf_unifcnt_kernel_3d = linkonce_odr protected addrspace(1) global [1 x i64]
 
 ;; Check sampling guard calls library function
 ; CHECK: call i32 @__llvm_profile_sampling_gpu(i32 3)
-; CHECK: call void @__llvm_profile_instrument_gpu(ptr addrspacecast (ptr addrspace(1) @__profc_kernel_3d to ptr), ptr addrspacecast (ptr addrspace(1) @__llvm_prf_unifcnt_kernel_3d to ptr), i64 1)
+; CHECK: call void @__llvm_profile_instrument_gpu(ptr addrspacecast (ptr addrspace(1) @__profc_kernel_3d to ptr), ptr addrspacecast (ptr addrspace(1) @__llvm_prf_unifcnt_kernel_3d to ptr), i64 1, ptr addrspacecast (ptr addrspace(1) getelementptr inbounds ([2 x i64], ptr addrspace(1) @__profc_kernel_3d, i32 0, i32 1) to ptr))
diff --git a/llvm/test/Instrumentation/InstrProfiling/amdgpu-contiguous-counters.ll b/llvm/test/Instrumentation/InstrProfiling/amdgpu-contiguous-counters.ll
index 67865f666832e..f8b2a1c80d11d 100644
--- a/llvm/test/Instrumentation/InstrProfiling/amdgpu-contiguous-counters.ll
+++ b/llvm/test/Instrumentation/InstrProfiling/amdgpu-contiguous-counters.ll
@@ -1,4 +1,4 @@
-;; Per-kernel __profc_* arrays land in section __llvm_prf_cnts with one slot
+;; Per-kernel __profc_* arrays land in section __llvm_prf_cnts with lane and wave slots
 ;; per counter, and a parallel __llvm_prf_unifcnt_* array lands in
 ;; __llvm_prf_ucnts. Counter increments lower to __llvm_profile_instrument_gpu
 ;; calls whose first pointer argument is a GEP into the per-kernel counter
@@ -10,9 +10,9 @@
 @__profn_kernel1 = private constant [7 x i8] c"kernel1"
 @__profn_kernel2 = private constant [7 x i8] c"kernel2"
 
-; CHECK: @__profc_kernel1 = linkonce_odr protected addrspace(1) global [2 x i64] zeroinitializer, section "__llvm_prf_cnts"
+; CHECK: @__profc_kernel1 = linkonce_odr protected addrspace(1) global [4 x i64] zeroinitializer, section "__llvm_prf_cnts"
 ; CHECK: @__llvm_prf_unifcnt_kernel1 = linkonce_odr protected addrspace(1) global [2 x i64] zeroinitializer, section "__llvm_prf_ucnts"
-; CHECK: @__profc_kernel2 = linkonce_odr protected addrspace(1) global [1 x i64] zeroinitializer, section "__llvm_prf_cnts"
+; CHECK: @__profc_kernel2 = linkonce_odr protected addrspace(1) global [2 x i64] zeroinitializer, section "__llvm_prf_cnts"
 
 define amdgpu_kernel void @kernel1() {
   call void @llvm.instrprof.increment(ptr @__profn_kernel1, i64 12345, i32 2, i32 0)
@@ -29,4 +29,4 @@ declare void @llvm.instrprof.increment(ptr, i64, i32, i32)
 
 ;; Second counter slot uses a GEP into the per-kernel counter array and the
 ;; matching uniform-counter array.
-; CHECK: call void @__llvm_profile_instrument_gpu(ptr addrspacecast (ptr addrspace(1) getelementptr inbounds ([2 x i64], ptr addrspace(1) @__profc_kernel1, i32 0, i32 1) to ptr), ptr addrspacecast (ptr addrspace(1) getelementptr inbounds ([2 x i64], ptr addrspace(1) @__llvm_prf_unifcnt_kernel1, i32 0, i32 1) to ptr), i64 1)
+; CHECK: call void @__llvm_profile_instrument_gpu(ptr addrspacecast (ptr addrspace(1) getelementptr inbounds ([4 x i64], ptr addrspace(1) @__profc_kernel1, i32 0, i32 1) to ptr), ptr addrspacecast (ptr addrspace(1) getelementptr inbounds ([2 x i64], ptr addrspace(1) @__llvm_prf_unifcnt_kernel1, i32 0, i32 1) to ptr), i64 1, ptr addrspacecast (ptr addrspace(1) getelementptr inbounds ([4 x i64], ptr addrspace(1) @__profc_kernel1, i32 0, i32 3) to ptr))
diff --git a/llvm/test/Instrumentation/InstrProfiling/amdgpu-uniform-counters.ll b/llvm/test/Instrumentation/InstrProfiling/amdgpu-uniform-counters.ll
index ae1164157c271..c81a9d7c16e78 100644
--- a/llvm/test/Instrumentation/InstrProfiling/amdgpu-uniform-counters.ll
+++ b/llvm/test/Instrumentation/InstrProfiling/amdgpu-uniform-counters.ll
@@ -15,8 +15,8 @@ define amdgpu_kernel void @test_kernel() {
 declare void @llvm.instrprof.increment(ptr, i64, i32, i32)
 
 ;; Per-function counter + uniform-counter globals (comdat)
-; CHECK: @__profc_test_kernel = linkonce_odr protected addrspace(1) global [1 x i64]
+; CHECK: @__profc_test_kernel = linkonce_odr protected addrspace(1) global [2 x i64]
 ; CHECK: @__llvm_prf_unifcnt_test_kernel = linkonce_odr protected addrspace(1) global [1 x i64]
 
 ;; __llvm_profile_instrument_gpu receives counter and uniform-counter bases
-; CHECK: call void @__llvm_profile_instrument_gpu(ptr addrspacecast (ptr addrspace(1) @__profc_test_kernel to ptr), ptr addrspacecast (ptr addrspace(1) @__llvm_prf_unifcnt_test_kernel to ptr), i64 1)
+; CHECK: call void @__llvm_profile_instrument_gpu(ptr addrspacecast (ptr addrspace(1) @__profc_test_kernel to ptr), ptr addrspacecast (ptr addrspace(1) @__llvm_prf_unifcnt_test_kernel to ptr), i64 1, ptr addrspacecast (ptr addrspace(1) getelementptr inbounds ([2 x i64], ptr addrspace(1) @__profc_test_kernel, i32 0, i32 1) to ptr))
diff --git a/llvm/test/Instrumentation/InstrProfiling/amdgpu-wave32.ll b/llvm/test/Instrumentation/InstrProfiling/amdgpu-wave32.ll
index 3a1653ff46c48..fbcfb53d5becb 100644
--- a/llvm/test/Instrumentation/InstrProfiling/amdgpu-wave32.ll
+++ b/llvm/test/Instrumentation/InstrProfiling/amdgpu-wave32.ll
@@ -17,9 +17,9 @@ declare void @llvm.instrprof.increment(ptr, i64, i32, i32)
 attributes #0 = { "target-cpu"="gfx1100" }
 
 ;; Per-function comdat counters + uniform counters
-; CHECK: @__profc_kernel_w32 = linkonce_odr protected addrspace(1) global [1 x i64]
+; CHECK: @__profc_kernel_w32 = linkonce_odr protected addrspace(1) global [2 x i64]
 ; CHECK: @__llvm_prf_unifcnt_kernel_w32 = linkonce_odr protected addrspace(1) global [1 x i64]
-; CHECK: @__profd_kernel_w32 = linkonce_odr protected addrspace(1) global { {{.*}} i16 0, i32 0 }
+; CHECK: @__profd_kernel_w32 = linkonce_odr protected addrspace(1) global { {{.*}} i16 0, i32 0, i32 1 }
 
 ;; Check wave size stored via intrinsic
 ; CHECK: %wavesize.i16 = trunc i32 %{{.*}} to i16
@@ -32,4 +32,4 @@ attributes #0 = { "target-cpu"="gfx1100" }
 
 ;; Check library call
 ; CHECK: po_then:
-; CHECK: call void @__llvm_profile_instrument_gpu(ptr addrspacecast (ptr addrspace(1) @__profc_kernel_w32 to ptr), ptr addrspacecast (ptr addrspace(1) @__llvm_prf_unifcnt_kernel_w32 to ptr), i64 1)
+; CHECK: call void @__llvm_profile_instrument_gpu(ptr addrspacecast (ptr addrspace(1) @__profc_kernel_w32 to ptr), ptr addrspacecast (ptr addrspace(1) @__llvm_prf_unifcnt_kernel_w32 to ptr), i64 1, ptr addrspacecast (ptr addrspace(1) getelementptr inbounds ([2 x i64], ptr addrspace(1) @__profc_kernel_w32, i32 0, i32 1) to ptr))
diff --git a/llvm/test/Instrumentation/InstrProfiling/amdgpu-wave64.ll b/llvm/test/Instrumentation/InstrProfiling/amdgpu-wave64.ll
index 3979653df1f40..aa3f05a168160 100644
--- a/llvm/test/Instrumentation/InstrProfiling/amdgpu-wave64.ll
+++ b/llvm/test/Instrumentation/InstrProfiling/amdgpu-wave64.ll
@@ -16,9 +16,9 @@ declare void @llvm.instrprof.increment(ptr, i64, i32, i32)
 attributes #0 = { "target-cpu"="gfx908" }
 
 ;; Per-function comdat counters
-; CHECK: @__profc_kernel_w64 = linkonce_odr protected addrspace(1) global [1 x i64]
+; CHECK: @__profc_kernel_w64 = linkonce_odr protected addrspace(1) global [2 x i64]
 ; CHECK: @__llvm_prf_unifcnt_kernel_w64 = linkonce_odr protected addrspace(1) global [1 x i64]
-; CHECK: @__profd_kernel_w64 = linkonce_odr protected addrspace(1) global { {{.*}} i16 0, i32 0 }
+; CHECK: @__profd_kernel_w64 = linkonce_odr protected addrspace(1) global { {{.*}} i16 0, i32 0, i32 1 }
 
 ;; Check wave size stored via intrinsic
 ; CHECK: %wavesize.i16 = trunc i32 %{{.*}} to i16
@@ -31,4 +31,4 @@ attributes #0 = { "target-cpu"="gfx908" }
 
 ;; Check library call
 ; CHECK: po_then:
-; CHECK: call void @__llvm_profile_instrument_gpu(ptr addrspacecast (ptr addrspace(1) @__profc_kernel_w64 to ptr), ptr addrspacecast (ptr addrspace(1) @__llvm_prf_unifcnt_kernel_w64 to ptr), i64 1)
+; CHECK: call void @__llvm_profile_instrument_gpu(ptr addrspacecast (ptr addrspace(1) @__profc_kernel_w64 to ptr), ptr addrspacecast (ptr addrspace(1) @__llvm_prf_unifcnt_kernel_w64 to ptr), i64 1, ptr addrspacecast (ptr addrspace(1) getelementptr inbounds ([2 x i64], ptr addrspace(1) @__profc_kernel_w64, i32 0, i32 1) to ptr))
diff --git a/llvm/test/Instrumentation/InstrProfiling/coverage.ll b/llvm/test/Instrumentation/InstrProfiling/coverage.ll
index 695a8829fdf75..6bfb17b656595 100644
--- a/llvm/test/Instrumentation/InstrProfiling/coverage.ll
+++ b/llvm/test/Instrumentation/InstrProfiling/coverage.ll
@@ -5,12 +5,12 @@ target triple = "aarch64-unknown-linux-gnu"
 
 @__profn_foo = private constant [3 x i8] c"foo"
 ; CHECK: @__profc_foo = private global [1 x i8] c"\FF", section "__llvm_prf_cnts", comdat, align 1
-; CHECK: @__profd_foo = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32 } { i64 {{.*}}, i64 {{.*}}, i64 sub (i64 ptrtoint (ptr @__profc_foo to i64)
-; BINARY: @__profd_foo = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32 } { i64 {{.*}}, i64 {{.*}}, i64 ptrtoint (ptr @__profc_foo to i64),
+; CHECK: @__profd_foo = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32, i32 } { i64 {{.*}}, i64 {{.*}}, i64 sub (i64 ptrtoint (ptr @__profc_foo to i64)
+; BINARY: @__profd_foo = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32, i32 } { i64 {{.*}}, i64 {{.*}}, i64 ptrtoint (ptr @__profc_foo to i64),
 @__profn_bar = private constant [3 x i8] c"bar"
 ; CHECK: @__profc_bar = private global [1 x i8] c"\FF", section "__llvm_prf_cnts", comdat, align 1
-; CHECK: @__profd_bar = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32 } { i64 {{.*}}, i64 {{.*}}, i64 sub (i64 ptrtoint (ptr @__profc_bar to i64)
-; BINARY: @__profd_bar = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32 } { i64 {{.*}}, i64 {{.*}}, i64 ptrtoint (ptr @__profc_bar to i64),
+; CHECK: @__profd_bar = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32, i32 } { i64 {{.*}}, i64 {{.*}}, i64 sub (i64 ptrtoint (ptr @__profc_bar to i64)
+; BINARY: @__profd_bar = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32, i32 } { i64 {{.*}}, i64 {{.*}}, i64 ptrtoint (ptr @__profc_bar to i64),
 
 ; CHECK: @__llvm_prf_nm = {{.*}} section "__llvm_prf_names"
 ; BINARY: @__llvm_prf_nm ={{.*}} section "__llvm_covnames"
diff --git a/llvm/test/Instrumentation/InstrProfiling/gpu-wave-convergence.ll b/llvm/test/Instrumentation/InstrProfiling/gpu-wave-convergence.ll
new file mode 100644
index 0000000000000..d3919b5949ccb
--- /dev/null
+++ b/llvm/test/Instrumentation/InstrProfiling/gpu-wave-convergence.ll
@@ -0,0 +1,14 @@
+; RUN: opt -passes='pgo-instr-gen,instrprof,verify' -offload-pgo-sampling=0 -S %s | FileCheck %s
+
+; Wave collection introduces no additional convergent operation.
+target triple = "amdgcn-amd-amdhsa"
+
+; CHECK-LABEL: define amdgpu_kernel void @controlled_kernel
+; CHECK: call void @__llvm_profile_instrument_gpu(
+; CHECK: call token @llvm.experimental.convergence.entry()
+; CHECK: ret void
+define amdgpu_kernel void @controlled_kernel() convergent {
+entry:
+  %token = call token @llvm.experimental.convergence.entry()
+  ret void
+}
diff --git a/llvm/test/Instrumentation/InstrProfiling/gpu-wave-counts.ll b/llvm/test/Instrumentation/InstrProfiling/gpu-wave-counts.ll
new file mode 100644
index 0000000000000..7bdf05a975357
--- /dev/null
+++ b/llvm/test/Instrumentation/InstrProfiling/gpu-wave-counts.ll
@@ -0,0 +1,38 @@
+; RUN: opt -passes=instrprof -offload-pgo-sampling=0 -S %s | FileCheck %s
+; RUN: opt -passes=instrprof -offload-pgo-sampling=3 -S %s | FileCheck %s --check-prefix=SAMPLE
+; RUN: not opt -passes=instrprof -profile-correlate=debug-info -disable-output %s 2>&1 | FileCheck %s --check-prefix=ERROR
+; RUN: not opt -passes=instrprof -sampled-instrumentation -disable-output %s 2>&1 | FileCheck %s --check-prefix=ERROR
+; ERROR: wave counts do not support profile correlation or lane-level instrumentation sampling
+
+target triple = "amdgcn-amd-amdhsa"
+ at __profn_test = private constant [4 x i8] c"test"
+
+; CHECK: @__profc_test = {{.*}}[4 x i64] zeroinitializer
+; CHECK: @__llvm_prf_unifcnt_test = {{.*}}[2 x i64] zeroinitializer
+; CHECK: @__profd_test = {{.*}}i32 4, [3 x i16] zeroinitializer, i16 0, i32 0, i32 2 }
+; CHECK-LABEL: define amdgpu_kernel void @test
+; CHECK: call void @__llvm_profile_instrument_gpu(ptr {{.*}}, ptr {{.*}}, i64 1, ptr {{.*}}i32 2
+; CHECK-NEXT: br i1 %cond, label %a, label %b
+; CHECK: a:
+; CHECK-NEXT: call void @__llvm_profile_instrument_gpu(ptr {{.*}}, ptr {{.*}}, i64 1, ptr {{.*}}i32 3
+; CHECK-NEXT: br label %exit
+; CHECK: b:
+; CHECK-NEXT: br label %exit
+; CHECK: exit:
+; CHECK-NEXT: ret void
+; SAMPLE: call i32 @__llvm_profile_sampling_gpu(i32 3)
+; SAMPLE: br i1
+; SAMPLE: call void @__llvm_profile_instrument_gpu(ptr {{.*}}, ptr {{.*}}, i64 1, ptr {{.*}}i32 2
+; SAMPLE: call void @__llvm_profile_instrument_gpu(ptr {{.*}}, ptr {{.*}}, i64 1, ptr {{.*}}i32 3
+define amdgpu_kernel void @test(i1 %cond) {
+entry:
+  call void @llvm.instrprof.increment(ptr @__profn_test, i64 123, i32 2, i32 0)
+  br i1 %cond, label %a, label %b
+a:
+  call void @llvm.instrprof.increment(ptr @__profn_test, i64 123, i32 2, i32 1)
+  br label %exit
+b:
+  br label %exit
+exit:
+  ret void
+}
diff --git a/llvm/test/Instrumentation/InstrProfiling/gpu-wave-inline.ll b/llvm/test/Instrumentation/InstrProfiling/gpu-wave-inline.ll
new file mode 100644
index 0000000000000..e31c7dc63414f
--- /dev/null
+++ b/llvm/test/Instrumentation/InstrProfiling/gpu-wave-inline.ll
@@ -0,0 +1,34 @@
+; RUN: opt -passes='pgo-instr-gen,always-inline,instrprof,verify' -offload-pgo-sampling=0 -S %s | FileCheck %s
+
+; Inlining leaves several profile names in each caller. Each name needs its own
+; lane, uniform, and wave counters, even after the callee definition is removed.
+source_filename = "wave-inline.c"
+target triple = "amdgcn-amd-amdhsa"
+
+; CHECK-DAG: @__profc_caller = {{.*}}[2 x i64] zeroinitializer
+; CHECK-DAG: @__profc_other = {{.*}}[2 x i64] zeroinitializer
+; CHECK-DAG: @[[CALLEE:__profc_.*callee]] = {{.*}}[2 x i64] zeroinitializer
+; CHECK-DAG: @__llvm_prf_unifcnt_{{.*}}callee = {{.*}}[1 x i64] zeroinitializer
+; CHECK-DAG: @__profd_{{.*}}callee = {{.*}}i32 2, [3 x i16] zeroinitializer, i16 0, i32 0, i32 1 }
+
+; CHECK-LABEL: define i32 @caller(
+; CHECK: call void @__llvm_profile_instrument_gpu({{.*}}i64 1, ptr {{.*}}@__profc_caller{{.*}}i32 1
+; CHECK: call void @__llvm_profile_instrument_gpu({{.*}}i64 1, ptr {{.*}}@[[CALLEE]]{{.*}}i32 1
+define i32 @caller(i32 %x) {
+  %v = call i32 @callee(i32 %x)
+  ret i32 %v
+}
+
+; CHECK-LABEL: define i32 @other(
+; CHECK: call void @__llvm_profile_instrument_gpu({{.*}}i64 1, ptr {{.*}}@__profc_other{{.*}}i32 1
+; CHECK: call void @__llvm_profile_instrument_gpu({{.*}}i64 1, ptr {{.*}}@[[CALLEE]]{{.*}}i32 1
+; CHECK-NOT: define {{.*}}@callee(
+define i32 @other(i32 %x) {
+  %v = call i32 @callee(i32 %x)
+  ret i32 %v
+}
+
+define internal i32 @callee(i32 %x) alwaysinline {
+  %v = add i32 %x, 1
+  ret i32 %v
+}
diff --git a/llvm/test/Instrumentation/InstrProfiling/gpu-wave-link.ll b/llvm/test/Instrumentation/InstrProfiling/gpu-wave-link.ll
new file mode 100644
index 0000000000000..7ff4adfa50e45
--- /dev/null
+++ b/llvm/test/Instrumentation/InstrProfiling/gpu-wave-link.ll
@@ -0,0 +1,51 @@
+; RUN: split-file %s %t
+; RUN: opt -passes='pgo-instr-gen,instrprof,always-inline,globaldce,verify' -offload-pgo-sampling=0 %t/a.ll -o %t/a.bc
+; RUN: opt -passes='pgo-instr-gen,instrprof,always-inline,globaldce,verify' -offload-pgo-sampling=0 %t/b.ll -o %t/b.bc
+; RUN: llvm-link -S %t/a.bc %t/b.bc | FileCheck %s --check-prefixes=CHECK,AB
+; RUN: llvm-link -S %t/b.bc %t/a.bc | FileCheck %s --check-prefixes=CHECK,BA
+
+; Model an inline device function shared by two translation units. Both callers
+; must update the same lane/wave allocation after COMDAT selection, in either
+; link order. No collection option is needed to obtain the wave slots.
+; CHECK: @[[SHARED:__profc_shared[^ ]*]] = {{.*}}[2 x i64] zeroinitializer
+
+; AB-LABEL: define amdgpu_kernel void @caller_a(
+; AB: call void @__llvm_profile_instrument_gpu({{.*}}i64 1, ptr {{.*}}@[[SHARED]]{{.*}}i32 1
+; AB-LABEL: define amdgpu_kernel void @caller_b(
+; AB: call void @__llvm_profile_instrument_gpu({{.*}}i64 1, ptr {{.*}}@[[SHARED]]{{.*}}i32 1
+; BA-LABEL: define amdgpu_kernel void @caller_b(
+; BA: call void @__llvm_profile_instrument_gpu({{.*}}i64 1, ptr {{.*}}@[[SHARED]]{{.*}}i32 1
+; BA-LABEL: define amdgpu_kernel void @caller_a(
+; BA: call void @__llvm_profile_instrument_gpu({{.*}}i64 1, ptr {{.*}}@[[SHARED]]{{.*}}i32 1
+
+;--- a.ll
+source_filename = "a.cpp"
+target triple = "amdgcn-amd-amdhsa"
+$shared = comdat any
+
+define amdgpu_kernel void @caller_a(ptr addrspace(1) %out, i32 %x) {
+  %r = call i32 @shared(i32 %x)
+  store i32 %r, ptr addrspace(1) %out
+  ret void
+}
+
+define linkonce_odr i32 @shared(i32 %x) alwaysinline comdat {
+  %r = add i32 %x, 1
+  ret i32 %r
+}
+
+;--- b.ll
+source_filename = "b.cpp"
+target triple = "amdgcn-amd-amdhsa"
+$shared = comdat any
+
+define amdgpu_kernel void @caller_b(ptr addrspace(1) %out, i32 %x) {
+  %r = call i32 @shared(i32 %x)
+  store i32 %r, ptr addrspace(1) %out
+  ret void
+}
+
+define linkonce_odr i32 @shared(i32 %x) alwaysinline comdat {
+  %r = add i32 %x, 1
+  ret i32 %r
+}
diff --git a/llvm/test/Transforms/PGOProfile/comdat_internal.ll b/llvm/test/Transforms/PGOProfile/comdat_internal.ll
index c948ad84392a0..cc8ae70c1e236 100644
--- a/llvm/test/Transforms/PGOProfile/comdat_internal.ll
+++ b/llvm/test/Transforms/PGOProfile/comdat_internal.ll
@@ -13,9 +13,9 @@ $foo = comdat any
 ; CHECK: @__llvm_profile_raw_version = hidden constant i64 {{[0-9]+}}, comdat
 ; CHECK-NOT: __profn__stdin__foo
 ; CHECK: @__profc__stdin__foo.[[#FOO_HASH]] = private global [1 x i64] zeroinitializer, section "__llvm_prf_cnts", comdat, align 8
-; CHECK: @__profd__stdin__foo.[[#FOO_HASH]] = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32 } { i64 {{.*}}, i64 [[#FOO_HASH]], i64 sub (i64 ptrtoint (ptr @__profc__stdin__foo.742261418966908927 to i64), i64 ptrtoint (ptr @__profd__stdin__foo.742261418966908927 to i64)), i64 0, i64 0, ptr null
+; CHECK: @__profd__stdin__foo.[[#FOO_HASH]] = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32, i32 } { i64 {{.*}}, i64 [[#FOO_HASH]], i64 sub (i64 ptrtoint (ptr @__profc__stdin__foo.742261418966908927 to i64), i64 ptrtoint (ptr @__profd__stdin__foo.742261418966908927 to i64)), i64 0, i64 0, ptr null
 ; CHECK-NOT: @foo
-; CHECK-SAME: , ptr null, i32 1, [3 x i16] zeroinitializer, i16 0, i32 0 }, section "__llvm_prf_data", comdat($__profc__stdin__foo.[[#FOO_HASH]]), align 8
+; CHECK-SAME: , ptr null, i32 1, [3 x i16] zeroinitializer, i16 0, i32 0, i32 0 }, section "__llvm_prf_data", comdat($__profc__stdin__foo.[[#FOO_HASH]]), align 8
 ; CHECK: @__llvm_prf_nm
 ; CHECK: @llvm.compiler.used
 
diff --git a/llvm/test/Transforms/PGOProfile/gpu-wave-counts.ll b/llvm/test/Transforms/PGOProfile/gpu-wave-counts.ll
new file mode 100644
index 0000000000000..9d0ab9f7c4631
--- /dev/null
+++ b/llvm/test/Transforms/PGOProfile/gpu-wave-counts.ll
@@ -0,0 +1,38 @@
+; RUN: opt -passes='pgo-instr-gen,instrprof,mem2reg,verify' -S %s | FileCheck %s
+; RUN: not opt -passes='pgo-instr-gen,instrprof' -pgo-block-coverage -disable-output %s 2>&1 | FileCheck %s --check-prefix=ERROR
+; RUN: not opt -passes='pgo-instr-gen,instrprof' -pgo-temporal-instrumentation -disable-output %s 2>&1 | FileCheck %s --check-prefix=ERROR
+; ERROR: wave counts require ordinary counter increments
+
+target triple = "amdgcn-amd-amdhsa"
+
+; Wave counts use the existing instrumentation sites and intrinsics.
+; CHECK-LABEL: define void @diamond
+; CHECK: call void @__llvm_profile_instrument_gpu(
+; CHECK: call void @__llvm_profile_instrument_gpu(
+define void @diamond(i1 %cond, ptr %p) {
+entry:
+  br i1 %cond, label %a, label %b
+a:
+  store volatile i32 1, ptr %p
+  br label %exit
+b:
+  store volatile i32 2, ptr %p
+  br label %exit
+exit:
+  ret void
+}
+
+; Sampling must leave static allocas in the entry so mem2reg can promote them.
+; CHECK-LABEL: define amdgpu_kernel void @alloca_kernel
+; CHECK-NOT: = alloca
+; CHECK: store i32 %value, ptr addrspace(1) %out
+; CHECK-NOT: = alloca
+; CHECK: ret void
+define amdgpu_kernel void @alloca_kernel(ptr addrspace(1) %out, i32 %value) {
+entry:
+  %slot = alloca i32, align 4, addrspace(5)
+  store i32 %value, ptr addrspace(5) %slot
+  %loaded = load i32, ptr addrspace(5) %slot
+  store i32 %loaded, ptr addrspace(1) %out
+  ret void
+}
diff --git a/llvm/test/Transforms/PGOProfile/instrprof_burst_sampling_fast.ll b/llvm/test/Transforms/PGOProfile/instrprof_burst_sampling_fast.ll
index 7261cc24b0dcf..087afc002efe8 100644
--- a/llvm/test/Transforms/PGOProfile/instrprof_burst_sampling_fast.ll
+++ b/llvm/test/Transforms/PGOProfile/instrprof_burst_sampling_fast.ll
@@ -14,7 +14,7 @@ $__llvm_profile_raw_version = comdat any
 
 ; SAMPLE-VAR: @__llvm_profile_sampling = thread_local global i16 0, comdat
 ; SAMPLE-VAR: @__profc_f = private global [1 x i64] zeroinitializer, section "__llvm_prf_cnts", comdat, align 8
-; SAMPLE-VAR: @__profd_f = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32 } { i64 -3706093650706652785, i64 12884901887, i64 sub (i64 ptrtoint (ptr @__profc_f to i64), i64 ptrtoint (ptr @__profd_f to i64)), i64 0, i64 0, ptr @f.local, ptr null, i32 1, [3 x i16] zeroinitializer, i16 0, i32 0 }, section "__llvm_prf_data", comdat($__profc_f), align 8
+; SAMPLE-VAR: @__profd_f = private global { i64, i64, i64, i64, i64, ptr, ptr, i32, [3 x i16], i16, i32, i32 } { i64 -3706093650706652785, i64 12884901887, i64 sub (i64 ptrtoint (ptr @__profc_f to i64), i64 ptrtoint (ptr @__profd_f to i64)), i64 0, i64 0, ptr @f.local, ptr null, i32 1, [3 x i16] zeroinitializer, i16 0, i32 0, i32 0 }, section "__llvm_prf_data", comdat($__profc_f), align 8
 ; SAMPLE-VAR: @__llvm_prf_nm = private constant {{.*}}, section "__llvm_prf_names", align 1
 ; SAMPLE-VAR: @llvm.compiler.used = appending global [2 x ptr] [ptr @__llvm_profile_sampling, ptr @__profd_f], section "llvm.metadata"
 ; SAMPLE-VAR: @llvm.used = appending global [1 x ptr] [ptr @__llvm_prf_nm], section "llvm.metadata"
diff --git a/llvm/test/Transforms/PGOProfile/vtable_profile.ll b/llvm/test/Transforms/PGOProfile/vtable_profile.ll
index 0c554db05cfb4..ea1c0e51f80f4 100644
--- a/llvm/test/Transforms/PGOProfile/vtable_profile.ll
+++ b/llvm/test/Transforms/PGOProfile/vtable_profile.ll
@@ -49,7 +49,7 @@ target triple = "x86_64-unknown-linux-gnu"
 @llvm.compiler.used = appending global [1 x ptr] [ptr @_ZTV5Base1], section "llvm.metadata"
 
 ; GEN: __llvm_profile_raw_version = comdat any
-; GEN: __llvm_profile_raw_version = hidden constant i64 72057594037927947, comdat
+; GEN: __llvm_profile_raw_version = hidden constant i64 72057594037927948, comdat
 ; GEN: __profn__Z4funci = private constant [8 x i8] c"_Z4funci"
 
 ; LOWER: $__profvt__ZTV7Derived = comdat nodeduplicate
diff --git a/llvm/test/tools/llvm-profdata/Inputs/c-general.profraw b/llvm/test/tools/llvm-profdata/Inputs/c-general.profraw
index dec90151a79352bf898c8c9d1d2d23eaa4bf5eed..93191c38878057d64e71cb4a193cda9324740036 100644
GIT binary patch
delta 199
zcmaDMa6*u?u_!ISs37M*&qU62z5p%L2Rh6D|IeR$O8#=h#JZ%34jdC72rzO?{AfNo
zfSGf$CL=$i!o-j2lM@&PfO3-y7zG$5CO0yggE%{YoC=WS2e9NMCIQ9+lLeW;avng=
z50G2~6IkN}AYWqgLWtxIAjbg6QJ*Zp43<@3hNzx+0nE(+a!-IXKVXrVyaC8*n0%1g
NeDVhtflVwKEC9qDL~H;6

delta 262
zcmX>h_(FiQu_!ISs37M*_e9QgK8LRVl{;7d|1WFC(B>94u`X%i3jq#+KmWmCqM^cM
z1xA6%7K{QM21sHBK(P)qu>(M{8$hucNa_Tb1SV at R32?kX5=#JzRRF~_{vzzy02Dg`
z6bnES1KBIVEWlBKBo at FdFgXJ#=712JctK$D4xo?(LTK^_pfU~?0gf9;);h2V0L3{r
JC$MC&001zQg at 6D6

diff --git a/llvm/test/tools/llvm-profdata/binary-ids-padding.test b/llvm/test/tools/llvm-profdata/binary-ids-padding.test
index 35eb92483603d..fb2e46a0c9f1d 100644
--- a/llvm/test/tools/llvm-profdata/binary-ids-padding.test
+++ b/llvm/test/tools/llvm-profdata/binary-ids-padding.test
@@ -21,7 +21,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-windows, system-zos
 RUN: printf '\201rforpl\377' > %t.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.profraw
 // There will be 2 20-byte binary IDs, so the total Binary IDs size will be 64 bytes.
 //   2 * 8  binary ID sizes
 // + 2 * 20 binary IDs (of size 20)
@@ -72,17 +72,17 @@ RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t.profraw
-RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0' >> %t.profraw
 
 RUN: printf '\067\265\035\031\112\165\023\344' >> %t.profraw
 RUN: printf '\02\0\0\0\0\0\0\0' >> %t.profraw
-RUN: printf '\270\377\3\0\1\0\0\0' >> %t.profraw
+RUN: printf '\260\377\3\0\1\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\02\0\0\0\0\0\0\0' >> %t.profraw
-RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0' >> %t.profraw
 
 RUN: printf '\023\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\067\0\0\0\0\0\0\0' >> %t.profraw
diff --git a/llvm/test/tools/llvm-profdata/gpu-wave-counts.test b/llvm/test/tools/llvm-profdata/gpu-wave-counts.test
new file mode 100644
index 0000000000000..ce87b5bd942a4
--- /dev/null
+++ b/llvm/test/tools/llvm-profdata/gpu-wave-counts.test
@@ -0,0 +1,50 @@
+# RUN: split-file %s %t
+# RUN: %python %t/raw.py little 2 > %t/little.raw
+# RUN: %python %t/raw.py big 2 > %t/big.raw
+# RUN: llvm-profdata merge %t/little.raw -o %t/one.profdata
+# RUN: llvm-profdata show --all-functions --counts %t/one.profdata | FileCheck %s --check-prefix=ONE
+# RUN: llvm-profdata merge %t/one.profdata --weighted-input=2,%t/big.raw -o %t/three.profdata
+# RUN: llvm-profdata show --all-functions --counts %t/three.profdata | FileCheck %s --check-prefix=THREE
+# RUN: not llvm-profdata merge --text %t/one.profdata -o %t/out.txt 2>&1 | FileCheck %s --check-prefix=TEXT
+# RUN: not llvm-profdata show --all-functions --text %t/one.profdata -o %t/show.txt 2>&1 | FileCheck %s --check-prefix=TEXT
+# RUN: not llvm-profdata show --all-functions --text %t/little.raw -o %t/show-raw.txt 2>&1 | FileCheck %s --check-prefix=TEXT
+# RUN: not llvm-profdata merge --write-prev-version %t/one.profdata -o %t/old.profdata 2>&1 | FileCheck %s --check-prefix=OLD
+# RUN: %python %t/raw.py little 5 > %t/bad.raw
+# RUN: not llvm-profdata merge %t/bad.raw -o %t/bad.profdata 2>&1 | FileCheck %s --check-prefix=BAD
+# RUN: %python %t/raw.py little 4 > %t/all-wave.raw
+# RUN: not llvm-profdata merge %t/all-wave.raw -o %t/bad.profdata 2>&1 | FileCheck %s --check-prefix=BAD
+
+# ONE: Block counts: [6400, 3200]
+# ONE: Block wave counts: [100, 100]
+# THREE: Block counts: [19200, 9600]
+# THREE: Block wave counts: [300, 300]
+# TEXT: text profiles do not support wave counts
+# OLD: wave counts require indexed profile version 15
+# BAD: invalid number of wave counters
+
+#--- raw.py
+import hashlib
+import struct
+import sys
+
+endian = '<' if sys.argv[1] == 'little' else '>'
+num_wave = int(sys.argv[2])
+name = b'wave'
+name_ref = int.from_bytes(hashlib.md5(name).digest()[:8], 'little')
+# Raw v12: 80-byte record, 4 counters (2 lane + 2 wave), 2 uniforms.
+record_size = 80
+counter_delta = record_size
+uniform_delta = counter_delta + 4 * 8
+names_delta = uniform_delta + 2 * 8
+names = bytes([len(name), 0]) + name
+header = [0xff6c70726f667281, 12 | (1 << 56), 0, 1, 0, 4, 0, 0, 0,
+          2, 0, uniform_delta, len(names), counter_delta, uniform_delta,
+          names_delta, 0, 0, 2]
+record = struct.pack(endian + '7QI4HII4x', name_ref, 123,
+                     counter_delta, uniform_delta, 0, 0, 0,
+                     4, 0, 0, 0, 64, 0, num_wave)
+assert len(record) == record_size
+sys.stdout.buffer.write(struct.pack(endian + '19Q', *header) + record +
+                        struct.pack(endian + '6Q', 6400, 3200,
+                                    100, 100, 6400, 0) +
+                        names + bytes((-len(names)) % 8))
diff --git a/llvm/test/tools/llvm-profdata/insufficient-binary-ids-size.test b/llvm/test/tools/llvm-profdata/insufficient-binary-ids-size.test
index b6cdf282dd97a..01b77d09ac90e 100644
--- a/llvm/test/tools/llvm-profdata/insufficient-binary-ids-size.test
+++ b/llvm/test/tools/llvm-profdata/insufficient-binary-ids-size.test
@@ -3,7 +3,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-zos
 RUN: printf '\201rforpl\377' > %t.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.profraw
 // We should fail on this because the data buffer (profraw file) is not long
 // enough to hold this binary IDs size. NOTE that this (combined with the 8-byte
 // alignment requirement for binary IDs size) will ensure we can at least read one
diff --git a/llvm/test/tools/llvm-profdata/large-binary-id-size.test b/llvm/test/tools/llvm-profdata/large-binary-id-size.test
index f27bdf4287035..7f77c49f67d70 100644
--- a/llvm/test/tools/llvm-profdata/large-binary-id-size.test
+++ b/llvm/test/tools/llvm-profdata/large-binary-id-size.test
@@ -5,7 +5,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-windows, system-zos
 RUN: printf '\201rforpl\377' > %t.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\40\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
diff --git a/llvm/test/tools/llvm-profdata/malformed-not-space-for-another-header.test b/llvm/test/tools/llvm-profdata/malformed-not-space-for-another-header.test
index aec3323bd0fe9..c454d43221a75 100644
--- a/llvm/test/tools/llvm-profdata/malformed-not-space-for-another-header.test
+++ b/llvm/test/tools/llvm-profdata/malformed-not-space-for-another-header.test
@@ -21,7 +21,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-windows, system-zos
 RUN: printf '\201rforpl\377' > %t.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
@@ -56,7 +56,7 @@ RUN: printf '\0\0\4\0\3\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t.profraw
-RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0' >> %t.profraw
 
 RUN: printf '\023\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\3\0foo\0\0\0' >> %t.profraw
diff --git a/llvm/test/tools/llvm-profdata/malformed-num-counters-zero.test b/llvm/test/tools/llvm-profdata/malformed-num-counters-zero.test
index 139d78341ba85..473a5f5a4231a 100644
--- a/llvm/test/tools/llvm-profdata/malformed-num-counters-zero.test
+++ b/llvm/test/tools/llvm-profdata/malformed-num-counters-zero.test
@@ -21,7 +21,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-windows, system-zos
 RUN: printf '\201rforpl\377' > %t.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
@@ -62,7 +62,7 @@ RUN: cp %t.profraw %t-good.profraw
 // Make NumCounters = 0 so that we get "number of counters is zero" error message
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
-RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0' >> %t.profraw
 
 RUN: printf '\023\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\3\0foo\0\0\0' >> %t.profraw
@@ -74,6 +74,8 @@ ZERO: malformed instrumentation profile data: number of counters is zero
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-bad.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t-bad.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-bad.profraw
+// NumWaveCounters and tail padding.
+RUN: printf '\0\0\0\0\0\0\0\0' >> %t-bad.profraw
 // Counter value is 72057594037927937
 RUN: printf '\1\0\0\0\0\0\0\1' >> %t-bad.profraw
 RUN: printf '\3\0foo\0\0\0' >> %t-bad.profraw
@@ -81,6 +83,8 @@ RUN: printf '\3\0foo\0\0\0' >> %t-bad.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-good.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t-good.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-good.profraw
+// NumWaveCounters and tail padding.
+RUN: printf '\0\0\0\0\0\0\0\0' >> %t-good.profraw
 // Counter value is 72057594037927937
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t-good.profraw
 RUN: printf '\3\0foo\0\0\0' >> %t-good.profraw
diff --git a/llvm/test/tools/llvm-profdata/malformed-ptr-to-counter-array.test b/llvm/test/tools/llvm-profdata/malformed-ptr-to-counter-array.test
index 9fafbeebaa735..98e3c82f36110 100644
--- a/llvm/test/tools/llvm-profdata/malformed-ptr-to-counter-array.test
+++ b/llvm/test/tools/llvm-profdata/malformed-ptr-to-counter-array.test
@@ -21,7 +21,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-windows, system-zos
 RUN: printf '\201rforpl\377' > %t.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
@@ -61,7 +61,7 @@ RUN: printf '\0\0\6\0\2\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\02\0\0\0\0\0\0\0' >> %t.profraw
-RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0' >> %t.profraw
 
 // Counter Section
 
diff --git a/llvm/test/tools/llvm-profdata/malformed-uniform-counter-array.test b/llvm/test/tools/llvm-profdata/malformed-uniform-counter-array.test
index 814c5298fc561..d3963c2c7865c 100644
--- a/llvm/test/tools/llvm-profdata/malformed-uniform-counter-array.test
+++ b/llvm/test/tools/llvm-profdata/malformed-uniform-counter-array.test
@@ -6,7 +6,7 @@ UNSUPPORTED: system-windows
 
 // UniformCounterPtr is one byte before UniformCountersDelta.
 RUN: printf '\201rforpl\377' > %t.neg.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.neg.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.neg.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.neg.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t.neg.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.neg.profraw
@@ -32,7 +32,7 @@ RUN: printf '\130\0\0\0\0\0\0\0' >> %t.neg.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.neg.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.neg.profraw
 RUN: printf '\2\0\0\0\0\0\0\0' >> %t.neg.profraw
-RUN: printf '\0\0\0\0\0\0\0\0' >> %t.neg.profraw
+RUN: printf '\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0' >> %t.neg.profraw
 RUN: printf '\067\0\0\0\0\0\0\0' >> %t.neg.profraw
 RUN: printf '\101\0\0\0\0\0\0\0' >> %t.neg.profraw
 RUN: printf '\5\0\0\0\0\0\0\0' >> %t.neg.profraw
@@ -44,7 +44,7 @@ NEG: error: no profile can be merged
 
 // UniformCounterPtr points one byte past the end of the uniform-counter section.
 RUN: printf '\201rforpl\377' > %t.past.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.past.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.past.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.past.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t.past.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.past.profraw
@@ -70,7 +70,7 @@ RUN: printf '\130\0\0\0\0\0\0\0' >> %t.past.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.past.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.past.profraw
 RUN: printf '\2\0\0\0\0\0\0\0' >> %t.past.profraw
-RUN: printf '\0\0\0\0\0\0\0\0' >> %t.past.profraw
+RUN: printf '\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0' >> %t.past.profraw
 RUN: printf '\067\0\0\0\0\0\0\0' >> %t.past.profraw
 RUN: printf '\101\0\0\0\0\0\0\0' >> %t.past.profraw
 RUN: printf '\5\0\0\0\0\0\0\0' >> %t.past.profraw
@@ -82,7 +82,7 @@ PAST: error: no profile can be merged
 
 // UniformCounterPtr leaves only one uniform counter, but NumCounters is two.
 RUN: printf '\201rforpl\377' > %t.too-many.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.too-many.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.too-many.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.too-many.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t.too-many.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.too-many.profraw
@@ -108,7 +108,7 @@ RUN: printf '\130\0\0\0\0\0\0\0' >> %t.too-many.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.too-many.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.too-many.profraw
 RUN: printf '\2\0\0\0\0\0\0\0' >> %t.too-many.profraw
-RUN: printf '\0\0\0\0\0\0\0\0' >> %t.too-many.profraw
+RUN: printf '\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0' >> %t.too-many.profraw
 RUN: printf '\067\0\0\0\0\0\0\0' >> %t.too-many.profraw
 RUN: printf '\101\0\0\0\0\0\0\0' >> %t.too-many.profraw
 RUN: printf '\5\0\0\0\0\0\0\0' >> %t.too-many.profraw
diff --git a/llvm/test/tools/llvm-profdata/misaligned-binary-ids-size.test b/llvm/test/tools/llvm-profdata/misaligned-binary-ids-size.test
index fb864f9810d48..e454de15c4bc7 100644
--- a/llvm/test/tools/llvm-profdata/misaligned-binary-ids-size.test
+++ b/llvm/test/tools/llvm-profdata/misaligned-binary-ids-size.test
@@ -5,7 +5,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-windows, system-zos
 RUN: printf '\201rforpl\377' > %t.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.profraw
 // We should fail on this because the binary IDs is not a multiple of 8 bytes.
 RUN: printf '\77\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
diff --git a/llvm/test/tools/llvm-profdata/profile-version.test b/llvm/test/tools/llvm-profdata/profile-version.test
index 7f627cb89c7d0..3898a04c428a8 100644
--- a/llvm/test/tools/llvm-profdata/profile-version.test
+++ b/llvm/test/tools/llvm-profdata/profile-version.test
@@ -2,7 +2,7 @@ Test the profile version.
 
 RUN: llvm-profdata merge -o %t.profdata %p/Inputs/basic.proftext
 RUN: llvm-profdata show --profile-version %t.profdata | FileCheck %s
-CHECK: Profile version: 14
+CHECK: Profile version: 15
 
 RUN: llvm-profdata merge -o %t.prev.profdata %p/Inputs/basic.proftext --write-prev-version
 RUN: llvm-profdata show --profile-version %t.prev.profdata | FileCheck %s --check-prefix=PREV
diff --git a/llvm/test/tools/llvm-profdata/raw-32-bits-be.test b/llvm/test/tools/llvm-profdata/raw-32-bits-be.test
index 353989dcaa315..938447f0ae4fd 100644
--- a/llvm/test/tools/llvm-profdata/raw-32-bits-be.test
+++ b/llvm/test/tools/llvm-profdata/raw-32-bits-be.test
@@ -6,7 +6,7 @@
 UNSUPPORTED: system-windows, system-zos
 // Header
 RUN: printf '\377lprofR\201' > %t
-RUN: printf '\0\0\0\0\0\0\0\13' >> %t
+RUN: printf '\0\0\0\0\0\0\0\14' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\2' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
diff --git a/llvm/test/tools/llvm-profdata/raw-32-bits-le.test b/llvm/test/tools/llvm-profdata/raw-32-bits-le.test
index c919850e2293b..97b4f4ad1ada2 100644
--- a/llvm/test/tools/llvm-profdata/raw-32-bits-le.test
+++ b/llvm/test/tools/llvm-profdata/raw-32-bits-le.test
@@ -5,7 +5,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-windows, system-zos
 RUN: printf '\201Rforpl\377' > %t
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\2\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
diff --git a/llvm/test/tools/llvm-profdata/raw-64-bits-be.test b/llvm/test/tools/llvm-profdata/raw-64-bits-be.test
index c4108e95ea36c..af0d9402b1e5d 100644
--- a/llvm/test/tools/llvm-profdata/raw-64-bits-be.test
+++ b/llvm/test/tools/llvm-profdata/raw-64-bits-be.test
@@ -5,7 +5,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-windows, system-zos
 RUN: printf '\377lprofr\201' > %t
-RUN: printf '\0\0\0\0\0\0\0\13' >> %t
+RUN: printf '\0\0\0\0\0\0\0\14' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\2' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
@@ -32,17 +32,17 @@ RUN: printf '\0\0\0\3\0\4\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\1\0\0\0\0' >> %t
-RUN: printf '\0\0\0\0\0\0\0\3' >> %t
+RUN: printf '\0\0\0\0\0\0\0\3\0\0\0\0\0\0\0\0' >> %t
 
 RUN: printf '\344\023\165\112\031\035\265\067' >> %t
 RUN: printf '\0\0\0\0\0\0\0\2' >> %t
-RUN: printf '\0\0\0\1\0\3\377\300' >> %t
+RUN: printf '\0\0\0\1\0\3\377\270' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
-RUN: printf '\0\0\0\3\0\3\377\273' >> %t
+RUN: printf '\0\0\0\3\0\3\377\263' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\2\0\0\0\0' >> %t
-RUN: printf '\0\0\0\0\0\0\0\1' >> %t
+RUN: printf '\0\0\0\0\0\0\0\1\0\0\0\0\0\0\0\0' >> %t
 
 RUN: printf '\0\0\0\0\0\0\0\023' >> %t
 RUN: printf '\0\0\0\0\0\0\0\067' >> %t
diff --git a/llvm/test/tools/llvm-profdata/raw-64-bits-le.test b/llvm/test/tools/llvm-profdata/raw-64-bits-le.test
index 301c34074a8fa..328ac552f9a35 100644
--- a/llvm/test/tools/llvm-profdata/raw-64-bits-le.test
+++ b/llvm/test/tools/llvm-profdata/raw-64-bits-le.test
@@ -5,7 +5,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-windows, system-zos
 RUN: printf '\201rforpl\377' > %t
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\2\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
@@ -32,17 +32,17 @@ RUN: printf '\0\0\4\0\3\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t
-RUN: printf '\0\0\0\0\3\0\0\0' >> %t
+RUN: printf '\0\0\0\0\3\0\0\0\0\0\0\0\0\0\0\0' >> %t
 
 RUN: printf '\067\265\035\031\112\165\023\344' >> %t
 RUN: printf '\02\0\0\0\0\0\0\0' >> %t
-RUN: printf '\300\377\3\0\1\0\0\0' >> %t
+RUN: printf '\270\377\3\0\1\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
-RUN: printf '\273\377\3\0\3\0\0\0' >> %t
+RUN: printf '\263\377\3\0\3\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t
 RUN: printf '\02\0\0\0\0\0\0\0' >> %t
-RUN: printf '\0\0\0\0\1\0\0\0' >> %t
+RUN: printf '\0\0\0\0\1\0\0\0\0\0\0\0\0\0\0\0' >> %t
 
 RUN: printf '\023\0\0\0\0\0\0\0' >> %t
 RUN: printf '\067\0\0\0\0\0\0\0' >> %t
diff --git a/llvm/test/tools/llvm-profdata/raw-two-profiles.test b/llvm/test/tools/llvm-profdata/raw-two-profiles.test
index 80fbba33e8e01..08ab6eedafa40 100644
--- a/llvm/test/tools/llvm-profdata/raw-two-profiles.test
+++ b/llvm/test/tools/llvm-profdata/raw-two-profiles.test
@@ -5,7 +5,7 @@
 // TODO: use a builtin version of printf
 UNSUPPORTED: system-windows, system-zos
 RUN: printf '\201rforpl\377' > %t-foo.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t-foo.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t-foo.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-foo.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t-foo.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-foo.profraw
@@ -32,13 +32,13 @@ RUN: printf '\0\0\0\0\0\0\0\0' >> %t-foo.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-foo.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-foo.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t-foo.profraw
-RUN: printf '\0\0\0\0\0\0\0\0' >> %t-foo.profraw
+RUN: printf '\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0' >> %t-foo.profraw
 
 RUN: printf '\023\0\0\0\0\0\0\0' >> %t-foo.profraw
 RUN: printf '\3\0foo\0\0\0' >> %t-foo.profraw
 
 RUN: printf '\201rforpl\377' > %t-bar.profraw
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t-bar.profraw
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t-bar.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-bar.profraw
 RUN: printf '\1\0\0\0\0\0\0\0' >> %t-bar.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-bar.profraw
@@ -65,7 +65,7 @@ RUN: printf '\0\0\0\0\0\0\0\0' >> %t-bar.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-bar.profraw
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t-bar.profraw
 RUN: printf '\02\0\0\0\0\0\0\0' >> %t-bar.profraw
-RUN: printf '\0\0\0\0\0\0\0\0' >> %t-bar.profraw
+RUN: printf '\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0' >> %t-bar.profraw
 
 RUN: printf '\067\0\0\0\0\0\0\0' >> %t-bar.profraw
 RUN: printf '\101\0\0\0\0\0\0\0' >> %t-bar.profraw
diff --git a/llvm/test/tools/llvm-profdata/truncated-profile.test b/llvm/test/tools/llvm-profdata/truncated-profile.test
index b49549f3a2828..e6418c15b3106 100644
--- a/llvm/test/tools/llvm-profdata/truncated-profile.test
+++ b/llvm/test/tools/llvm-profdata/truncated-profile.test
@@ -5,8 +5,8 @@ UNSUPPORTED: system-zos
 
 // Magic
 RUN: printf '\201rforpl\377' > %t.profraw
-// Version (11 = 0o13)
-RUN: printf '\13\0\0\0\0\0\0\0' >> %t.profraw
+// Version (12 = 0o14)
+RUN: printf '\14\0\0\0\0\0\0\0' >> %t.profraw
 // BinaryIdsSize
 RUN: printf '\0\0\0\0\0\0\0\0' >> %t.profraw
 // NumData
@@ -46,4 +46,4 @@ RUN: printf '\3\0\0\0\0\0\0\0' >> %t.profraw
 RUN: printf '\0\0\0\0' >> %t.profraw
 
 RUN: not llvm-profdata show  %t.profraw 2>&1 | FileCheck %s
-CHECK: invalid instrumentation profile data (file is incomplete or header is corrupt): profile file size (156 bytes) smaller than expected (at least 248 bytes: 152(Header) + 0(BinaryIdSize) + 72(DataSize) + 8(CountersSize) + 0(NumBitmapBytes) + 0(UniformCountersSectionSize) + 16(NamesSize) + 0(VTableSectionSize) + 0(VTableNameSize) + 0(Padding))
+CHECK: invalid instrumentation profile data (file is incomplete or header is corrupt): profile file size (156 bytes) smaller than expected (at least 256 bytes: 152(Header) + 0(BinaryIdSize) + 80(DataSize) + 8(CountersSize) + 0(NumBitmapBytes) + 0(UniformCountersSectionSize) + 16(NamesSize) + 0(VTableSectionSize) + 0(VTableNameSize) + 0(Padding))
diff --git a/llvm/tools/llvm-profdata/llvm-profdata.cpp b/llvm/tools/llvm-profdata/llvm-profdata.cpp
index a8bf8b8760c44..b0c5517473716 100644
--- a/llvm/tools/llvm-profdata/llvm-profdata.cpp
+++ b/llvm/tools/llvm-profdata/llvm-profdata.cpp
@@ -986,14 +986,18 @@ static void writeInstrProfile(StringRef OutputFilename,
   if (EC)
     exitWithErrorCode(EC, OutputFilename);
 
-  if (OutputFormat == PF_Text) {
-    if (Error E = Writer.writeText(Output))
-      warn(std::move(E));
-  } else {
-    if (Output.is_displayed())
-      exitWithError("cannot write a non-text format profile to the terminal");
-    if (Error E = Writer.write(Output))
-      warn(std::move(E));
+  if (OutputFormat != PF_Text && Output.is_displayed())
+    exitWithError("cannot write a non-text format profile to the terminal");
+
+  if (Error E = OutputFormat == PF_Text ? Writer.writeText(Output)
+                                        : Writer.write(Output)) {
+    // Unsupported formats cannot represent the profile at all. Preserve the
+    // existing warning behavior for recoverable profile-data errors.
+    warn(handleErrors(std::move(E), [&](const InstrProfError &IPE) -> Error {
+      if (IPE.get() == instrprof_error::unsupported_version)
+        exitWithError(IPE.message(), OutputFilename);
+      return make_error<InstrProfError>(IPE.get(), IPE.getMessage());
+    }));
   }
 }
 
@@ -2938,8 +2942,9 @@ static int showInstrProfile(ShowFormat SFormat, raw_fd_ostream &OS) {
 
     if (doTextFormatDump) {
       InstrProfSymtab &Symtab = Reader->getSymtab();
-      InstrProfWriter::writeRecordInText(Func.Name, Func.Hash, Func, Symtab,
-                                         OS);
+      if (Error E = InstrProfWriter::writeRecordInText(Func.Name, Func.Hash,
+                                                       Func, Symtab, OS))
+        exitWithError(std::move(E), Filename);
       continue;
     }
 
@@ -3025,6 +3030,13 @@ static int showInstrProfile(ShowFormat SFormat, raw_fd_ostream &OS) {
         }
         OS << "]\n";
 
+        if (!Func.WaveCounts.empty()) {
+          OS << "    Block wave counts: [";
+          for (size_t I = 0, E = Func.WaveCounts.size(); I < E; ++I)
+            OS << (I ? ", " : "") << Func.WaveCounts[I];
+          OS << "]\n";
+        }
+
         // Show uniformity bits if present
         if (!Func.UniformityBits.empty()) {
           OS << "    Block uniformity: [";
diff --git a/llvm/unittests/ProfileData/InstrProfTest.cpp b/llvm/unittests/ProfileData/InstrProfTest.cpp
index d6d39c2e4ea02..94517d7d02192 100644
--- a/llvm/unittests/ProfileData/InstrProfTest.cpp
+++ b/llvm/unittests/ProfileData/InstrProfTest.cpp
@@ -112,6 +112,66 @@ static const auto Err = [](Error E) {
   FAIL();
 };
 
+TEST_F(InstrProfTest, WaveCountsRoundTripAndMerge) {
+  NamedInstrProfRecord A("wave", 123, {6400, 3200});
+  A.WaveCounts = {100, 100};
+  Writer.addRecord(std::move(A), Err);
+  NamedInstrProfRecord B("wave", 123, {640, 320});
+  B.WaveCounts = {10, 10};
+  Writer.addRecord(std::move(B), 2, Err);
+  Writer.addRecord({"plain", 124, {42}}, Err);
+  readProfile(Writer.writeBuffer());
+  auto R = Reader->getInstrProfRecord("wave", 123);
+  ASSERT_THAT_ERROR(R.takeError(), Succeeded());
+  EXPECT_THAT(R->Counts, ElementsAre(7680, 3840));
+  EXPECT_THAT(R->WaveCounts, ElementsAre(120, 120));
+  auto Plain = Reader->getInstrProfRecord("plain", 124);
+  ASSERT_THAT_ERROR(Plain.takeError(), Succeeded());
+  EXPECT_TRUE(Plain->WaveCounts.empty());
+  InstrProfWriter Again;
+  for (auto &Record : *Reader)
+    Again.addRecord(NamedInstrProfRecord(Record), Err);
+  readProfile(Again.writeBuffer());
+  R = Reader->getInstrProfRecord("wave", 123);
+  ASSERT_THAT_ERROR(R.takeError(), Succeeded());
+  EXPECT_THAT(R->WaveCounts, ElementsAre(120, 120));
+}
+
+TEST_F(InstrProfTest, WaveCountsMismatchDoesNotMutate) {
+  InstrProfRecord A({64});
+  A.WaveCounts = {1};
+  InstrProfRecord B({128});
+  bool Warned = false;
+  A.merge(B, 1, [&](instrprof_error E) {
+    EXPECT_EQ(E, instrprof_error::count_mismatch);
+    Warned = true;
+  });
+  EXPECT_TRUE(Warned);
+  EXPECT_THAT(A.Counts, ElementsAre(64));
+  EXPECT_THAT(A.WaveCounts, ElementsAre(1));
+}
+
+TEST_F(InstrProfTest, WaveCountsScaleCopyClearAndOverflow) {
+  InstrProfRecord A({64, 128});
+  A.WaveCounts = {2, 4};
+  A.scale(3, 2, [](instrprof_error) { FAIL(); });
+  EXPECT_THAT(A.WaveCounts, ElementsAre(3, 6));
+  InstrProfRecord B(A), C;
+  C = B;
+  EXPECT_EQ(C.WaveCounts, A.WaveCounts);
+  C.Clear();
+  EXPECT_TRUE(C.WaveCounts.empty());
+  A.WaveCounts = {getInstrMaxCountValue(), 1};
+  B.WaveCounts = {1, 1};
+  bool Overflowed = false;
+  A.merge(B, 1, [&](instrprof_error E) {
+    EXPECT_EQ(E, instrprof_error::counter_overflow);
+    Overflowed = true;
+  });
+  EXPECT_TRUE(Overflowed);
+  EXPECT_THAT(A.WaveCounts, ElementsAre(getInstrMaxCountValue(), 2));
+}
+
 TEST_P(MaybeSparseInstrProfTest, write_and_read_one_function) {
   Writer.addRecord({"foo", 0x1234, {1, 2, 3, 4}}, Err);
   auto Profile = Writer.writeBuffer();

>From 08c0bfe6129e1e3eafdc68324a20feee1e609cd3 Mon Sep 17 00:00:00 2001
From: "Yaxun (Sam) Liu" <yaxun.liu at amd.com>
Date: Wed, 23 Sep 2026 06:45:47 -0400
Subject: [PATCH 2/3] [PGO] Fix wave-counter builds and compressed profile
 fixture

Initialize the wave-counter length to zero in correlated profile records.
These records do not carry wave counts, and Clang builds with warnings as
errors require the new aggregate field to be initialized explicitly.

Regenerate the compressed raw-profile fixture with the current format so
readers built without zlib reach the intended compression diagnostic.
The function identities, counters, and compressed names are unchanged.
---
 llvm/lib/ProfileData/InstrProfCorrelator.cpp  |   1 +
 .../llvm-profdata/Inputs/compressed.profraw   | Bin 2104 -> 2200 bytes
 llvm/test/tools/llvm-profdata/nocompress.test |  17 ++++++++---------
 3 files changed, 9 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/ProfileData/InstrProfCorrelator.cpp b/llvm/lib/ProfileData/InstrProfCorrelator.cpp
index c330badeb4aa5..74ab749070d66 100644
--- a/llvm/lib/ProfileData/InstrProfCorrelator.cpp
+++ b/llvm/lib/ProfileData/InstrProfCorrelator.cpp
@@ -346,6 +346,7 @@ void InstrProfCorrelatorImpl<IntPtrT>::addDataProbe(
       /*NumValueSites=*/{maybeSwap<uint16_t>(0), maybeSwap<uint16_t>(0)},
       /*OffloadDeviceWaveSize=*/maybeSwap<uint16_t>(0),
       maybeSwap<uint32_t>(NumBitmapBytes),
+      /*NumWaveCounters=*/maybeSwap<uint32_t>(0),
   });
 }
 
diff --git a/llvm/test/tools/llvm-profdata/Inputs/compressed.profraw b/llvm/test/tools/llvm-profdata/Inputs/compressed.profraw
index 778e80fce2691a35d9654748763b0aa8c233ec13..d358dfc21e8da60e6bf93824214a58040e83272f 100644
GIT binary patch
delta 198
zcmdlXFhh{Du_!ISs37M*&qU62UX5GfIg3{P|KIR-`Q+G%waF75I3_+2VC0zi(R^|M
zGv{PYMt(+xi67M`Col>C<t7&}3NT7cZe%nEadrSX6(GqEV97~L0*nVH3o?P_Jb;`Z
zAh`x6u*L~MzQp8(5Xl=rjscLPK3RYnEUUl_Q9bbjn41CQo&afnz#=hu1CY}&`5?3T
M<PR(Yn^-JZ0Mmj*ZU6uP

delta 261
zcmbOsxI=)mu_!ISs37M*_e9QgUWZ?ERC`wa|1Wzi=~nT?+T at 8Z1ULl#{0D=Hh6<Av
z7zHL<FbZ%OAc++K#X8W$4gke&0L5k?sS{uln5 at Ah!0`e}ECDE10Tk2ti?Cw at Q0xRy
zEC5LiWUmCX07n6mSOBxY<P4yg143-#1%b&sfI<=op~)YB$~agAIBp<W>%bxa6zABS
Iz+%Ay09e0;nE(I)

diff --git a/llvm/test/tools/llvm-profdata/nocompress.test b/llvm/test/tools/llvm-profdata/nocompress.test
index 23657060e05e8..fe53f4611a181 100644
--- a/llvm/test/tools/llvm-profdata/nocompress.test
+++ b/llvm/test/tools/llvm-profdata/nocompress.test
@@ -1,12 +1,11 @@
-You need a checkout of clang with compiler-rt to generate the
-binary file here.  These shell commands can be used to regenerate
-it.
-$ SRC=path/to/llvm
-$ CFE=$SRC/tools/clang
-$ TESTDIR=$SRC/test/tools/llvm-profdata
-$ CFE_TESTDIR=$CFE/test/Profile
-$ clang -o a.out -fprofile-instr-generate $CFE_TESTDIR/c-general.c
-$ LLVM_PROFILE_FILE=$TESTDIR/Inputs/compressed.profraw ./a.out
+Generate this fixture with a matching Clang and compiler-rt that support zlib.
+Use a local source filename to keep machine paths out of the profile names.
+
+$ SRC=/path/to/llvm-project
+$ cp "$SRC/clang/test/Profile/c-general.c" c-general.c
+$ clang -fprofile-instr-generate c-general.c -o c-general
+$ LLVM_PROFILE_FILE=compressed.profraw ./c-general
+$ cp compressed.profraw "$SRC/llvm/test/tools/llvm-profdata/Inputs/"
 
 RUN: not llvm-profdata show %p/Inputs/compressed.profraw -o %t 2>&1 | FileCheck %s
 

>From 76c7824f3da2727c233bf4736784b501215140bc Mon Sep 17 00:00:00 2001
From: "Yaxun (Sam) Liu" <yaxun.liu at amd.com>
Date: Mon, 28 Sep 2026 20:40:04 -0400
Subject: [PATCH 3/3] [PGO] Collect dense AMDGPU block wave counts

Wave visits are not additive across divergent control flow. Sparse scalar
counter sites can leave repeated loop blocks unmeasured, so their wave
frequencies cannot be reconstructed from entry and edge counts.

Append zero-step instrumentation for eligible unmeasured AMDGPU blocks,
keeping the existing lane and select counter indices unchanged. Exclude the
appended slots from lane-flow reconstruction and uniformity annotation.

Identify the layout with a profile variant bit, preserve it through raw and
indexed readers/writers, and select it automatically during profile use.
Keep sparse profiles readable and reject incompatible merges, including
concatenated raw profiles. Ignore empty merge-worker contexts.

Enable dense collection for ordinary AMDGPU IR-PGO by default, with the
hidden -pgo-instrument-dense-wave-counts option for debugging. Leave
context-sensitive, coverage, and temporal instrumentation unchanged.

Test loop counter placement, critical-edge splitting, EH insertion points,
scalar/select reconstruction, uniformity coverage and merge layout.
---
 llvm/docs/InstrProfileFormat.md               |  20 +++-
 llvm/include/llvm/ProfileData/InstrProf.h     |   4 +-
 .../llvm/ProfileData/InstrProfData.inc        |   4 +-
 .../llvm/ProfileData/InstrProfReader.h        |   6 +
 .../llvm/ProfileData/InstrProfWriter.h        |   7 ++
 llvm/lib/ProfileData/InstrProfReader.cpp      |   8 ++
 llvm/lib/ProfileData/InstrProfWriter.cpp      |   4 +
 .../Instrumentation/PGOInstrumentation.cpp    |  68 +++++++++--
 .../Transforms/PGOProfile/dense-wave-cfg.ll   | 108 ++++++++++++++++++
 .../PGOProfile/dense-wave-instrumentation.ll  |  59 ++++++++++
 .../PGOProfile/dense-wave-profile-use.ll      |  82 +++++++++++++
 .../llvm-profdata/dense-wave-layout.test      |  43 +++++++
 llvm/tools/llvm-profdata/llvm-profdata.cpp    |  11 +-
 llvm/unittests/ProfileData/InstrProfTest.cpp  |  31 +++++
 14 files changed, 441 insertions(+), 14 deletions(-)
 create mode 100644 llvm/test/Transforms/PGOProfile/dense-wave-cfg.ll
 create mode 100644 llvm/test/Transforms/PGOProfile/dense-wave-instrumentation.ll
 create mode 100644 llvm/test/Transforms/PGOProfile/dense-wave-profile-use.ll
 create mode 100644 llvm/test/tools/llvm-profdata/dense-wave-layout.test

diff --git a/llvm/docs/InstrProfileFormat.md b/llvm/docs/InstrProfileFormat.md
index 3a53a6b77066e..abde06ebc0ee8 100644
--- a/llvm/docs/InstrProfileFormat.md
+++ b/llvm/docs/InstrProfileFormat.md
@@ -22,8 +22,24 @@ GPU counter instrumentation always collects wave counts. Each existing
 instrumentation counter has a wave counter at the same index. The GPU runtime
 updates lane, uniformity, and wave counters in one call: the first active lane
 adds the active-lane count to the lane counter and one to the wave counter.
-Workgroup sampling controls the overhead of all three channels. No extra instrumentation
-points are inserted; blocks without a counter have no direct wave measurement.
+Workgroup sampling controls the overhead of all three channels.
+
+Ordinary AMDGPU IR-PGO instrumentation also measures waves in blocks without
+an existing counter. The sparse block counters and select counters retain
+their original indices. Additional block counters follow them in function
+block order, using a zero lane step so that lane-flow reconstruction and
+uniformity annotations retain their original inputs. Blocks without a legal
+insertion point are left unmeasured. Context-sensitive, coverage and temporal
+instrumentation do not use this extension.
+
+`VARIANT_MASK_DENSE_WAVE` (bit 54 in the raw and indexed version fields)
+identifies this layout. The device module supplies the flag to the collector;
+profile readers and writers preserve it through merging. Profile use reads
+the flag to reconstruct the extra indices automatically, including when wave
+metadata production is disabled. Older sparse profiles remain supported.
+Dense and sparse layouts cannot be merged. The hidden generation option
+`-pgo-instrument-dense-wave-counts=false` selects the legacy sparse layout for
+debugging; its value during profile use does not override the recorded layout.
 
 Wave counts observe the active groups that execute each instrumentation point.
 They do not require full waves or preserve lane grouping across compiler
diff --git a/llvm/include/llvm/ProfileData/InstrProf.h b/llvm/include/llvm/ProfileData/InstrProf.h
index 9381d940f3eab..6c3ddd1c8c5f5 100644
--- a/llvm/include/llvm/ProfileData/InstrProf.h
+++ b/llvm/include/llvm/ProfileData/InstrProf.h
@@ -402,7 +402,9 @@ enum class InstrProfKind {
   TemporalProfile = 0x80,
   // A profile with loop entry basic blocks instrumentation.
   LoopEntriesInstrumentation = 0x100,
-  LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/LoopEntriesInstrumentation)
+  // Append zero-step counters to measure waves in otherwise unmeasured blocks.
+  DenseWaveInstrumentation = 0x200,
+  LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/DenseWaveInstrumentation)
 };
 
 LLVM_ABI const std::error_category &instrprof_category();
diff --git a/llvm/include/llvm/ProfileData/InstrProfData.inc b/llvm/include/llvm/ProfileData/InstrProfData.inc
index c1957f665cb98..fe8fe9ba1db30 100644
--- a/llvm/include/llvm/ProfileData/InstrProfData.inc
+++ b/llvm/include/llvm/ProfileData/InstrProfData.inc
@@ -787,7 +787,8 @@ serializeValueProfDataFrom(ValueProfRecordClosure *Closure,
  * (i.e. bit 56) to 1 to indicate if this is an IR-level instrumentation
  * generated profile, and 0 if this is a Clang FE generated profile.
  * 1 in bit 57 indicates there are context-sensitive records in the profile.
- * The 54th bit indicates whether to always instrument loop entry blocks.
+ * The 54th bit indicates dense GPU wave instrumentation.
+ * The 55th bit indicates whether to always instrument loop entry blocks.
  * The 58th bit indicates whether to always instrument function entry blocks.
  * The 59th bit indicates whether to use debug info to correlate profiles.
  * The 60th bit indicates single byte coverage instrumentation.
@@ -797,6 +798,7 @@ serializeValueProfDataFrom(ValueProfRecordClosure *Closure,
  */
 #define VARIANT_MASKS_ALL 0xffffffff00000000ULL
 #define GET_VERSION(V) ((V) & ~VARIANT_MASKS_ALL)
+#define VARIANT_MASK_DENSE_WAVE (0x1ULL << 54)
 #define VARIANT_MASK_INSTR_LOOP_ENTRIES (0x1ULL << 55)
 #define VARIANT_MASK_IR_PROF (0x1ULL << 56)
 #define VARIANT_MASK_CSIR_PROF (0x1ULL << 57)
diff --git a/llvm/include/llvm/ProfileData/InstrProfReader.h b/llvm/include/llvm/ProfileData/InstrProfReader.h
index dc8920040ab50..d1be93188f755 100644
--- a/llvm/include/llvm/ProfileData/InstrProfReader.h
+++ b/llvm/include/llvm/ProfileData/InstrProfReader.h
@@ -143,6 +143,12 @@ class InstrProfReader {
   /// individual attributes prefer using the helpers above.
   virtual InstrProfKind getProfileKind() const = 0;
 
+  /// Whether block wave measurements extend the sparse block/select layout.
+  bool hasDenseWaveProfile() const {
+    return static_cast<bool>(getProfileKind() &
+                             InstrProfKind::DenseWaveInstrumentation);
+  }
+
   /// Return the PGO symtab. There are three different readers:
   /// Raw, Text, and Indexed profile readers. The first two types
   /// of readers are used only by llvm-profdata tool, while the indexed
diff --git a/llvm/include/llvm/ProfileData/InstrProfWriter.h b/llvm/include/llvm/ProfileData/InstrProfWriter.h
index 89e8d3e322c1e..faa6b8922f1c5 100644
--- a/llvm/include/llvm/ProfileData/InstrProfWriter.h
+++ b/llvm/include/llvm/ProfileData/InstrProfWriter.h
@@ -201,6 +201,13 @@ class InstrProfWriter {
       return make_error<InstrProfError>(
           instrprof_error::coverage_count_mismatch);
     }
+    if (static_cast<bool>(
+            (ProfileKind & InstrProfKind::DenseWaveInstrumentation) ^
+            (Other & InstrProfKind::DenseWaveInstrumentation))) {
+      return make_error<InstrProfError>(
+          instrprof_error::unsupported_version,
+          "cannot merge dense and sparse wave instrumentation layouts");
+    }
 
     // Now we update the profile type with the bits that are set.
     ProfileKind |= Other;
diff --git a/llvm/lib/ProfileData/InstrProfReader.cpp b/llvm/lib/ProfileData/InstrProfReader.cpp
index 536acc2eab4de..65d77d27dd732 100644
--- a/llvm/lib/ProfileData/InstrProfReader.cpp
+++ b/llvm/lib/ProfileData/InstrProfReader.cpp
@@ -56,6 +56,9 @@ static InstrProfKind getProfileKindFromVersion(uint64_t Version) {
   if (Version & VARIANT_MASK_INSTR_LOOP_ENTRIES) {
     ProfileKind |= InstrProfKind::LoopEntriesInstrumentation;
   }
+  if (Version & VARIANT_MASK_DENSE_WAVE) {
+    ProfileKind |= InstrProfKind::DenseWaveInstrumentation;
+  }
   if (Version & VARIANT_MASK_BYTE_COVERAGE) {
     ProfileKind |= InstrProfKind::SingleByteCoverage;
   }
@@ -268,6 +271,8 @@ Error TextInstrProfReader::readHeader() {
       ProfileKind &= ~InstrProfKind::FunctionEntryInstrumentation;
     else if (Str.equals_insensitive("instrument_loop_entries"))
       ProfileKind |= InstrProfKind::LoopEntriesInstrumentation;
+    else if (Str.equals_insensitive("dense_wave"))
+      ProfileKind |= InstrProfKind::DenseWaveInstrumentation;
     else if (Str.equals_insensitive("single_byte_coverage"))
       ProfileKind |= InstrProfKind::SingleByteCoverage;
     else if (Str.equals_insensitive("temporal_prof_traces")) {
@@ -546,6 +551,9 @@ Error RawInstrProfReader<IntPtrT>::readNextHeader(const char *CurrentPos) {
 
   // There's another profile to read, so we need to process the header.
   auto *Header = reinterpret_cast<const RawInstrProf::Header *>(CurrentPos);
+  if ((Version ^ swap(Header->Version)) & VARIANT_MASK_DENSE_WAVE)
+    return error(instrprof_error::unsupported_version,
+                 "cannot merge dense and sparse wave instrumentation layouts");
   return readHeader(*Header);
 }
 
diff --git a/llvm/lib/ProfileData/InstrProfWriter.cpp b/llvm/lib/ProfileData/InstrProfWriter.cpp
index 0586071faf783..78ce719f09495 100644
--- a/llvm/lib/ProfileData/InstrProfWriter.cpp
+++ b/llvm/lib/ProfileData/InstrProfWriter.cpp
@@ -593,6 +593,8 @@ Error InstrProfWriter::writeImpl(ProfOStream &OS) {
   if (static_cast<bool>(ProfileKind &
                         InstrProfKind::LoopEntriesInstrumentation))
     Header.Version |= VARIANT_MASK_INSTR_LOOP_ENTRIES;
+  if (static_cast<bool>(ProfileKind & InstrProfKind::DenseWaveInstrumentation))
+    Header.Version |= VARIANT_MASK_DENSE_WAVE;
   if (static_cast<bool>(ProfileKind & InstrProfKind::SingleByteCoverage))
     Header.Version |= VARIANT_MASK_BYTE_COVERAGE;
   if (static_cast<bool>(ProfileKind & InstrProfKind::FunctionEntryOnly))
@@ -818,6 +820,8 @@ Error InstrProfWriter::writeText(raw_fd_ostream &OS) {
           "blocks\n:instrument_loop_entries\n";
   if (static_cast<bool>(ProfileKind & InstrProfKind::SingleByteCoverage))
     OS << "# Instrument block coverage\n:single_byte_coverage\n";
+  if (static_cast<bool>(ProfileKind & InstrProfKind::DenseWaveInstrumentation))
+    OS << "# Dense GPU block wave instrumentation\n:dense_wave\n";
   InstrProfSymtab Symtab;
 
   using FuncPair = detail::DenseMapPair<uint64_t, InstrProfRecord>;
diff --git a/llvm/lib/Transforms/Instrumentation/PGOInstrumentation.cpp b/llvm/lib/Transforms/Instrumentation/PGOInstrumentation.cpp
index 9a0203a03d5c7..516d6b7047856 100644
--- a/llvm/lib/Transforms/Instrumentation/PGOInstrumentation.cpp
+++ b/llvm/lib/Transforms/Instrumentation/PGOInstrumentation.cpp
@@ -264,6 +264,11 @@ static cl::opt<bool>
                              cl::Hidden,
                              cl::desc("Force to instrument loop entries."));
 
+static cl::opt<bool> PGOInstrumentDenseWaveCounts(
+    "pgo-instrument-dense-wave-counts", cl::Hidden, cl::init(true),
+    cl::desc("Measure wave counts in all eligible AMDGPU blocks during IR PGO "
+             "instrumentation"));
+
 static cl::opt<bool> PGOFunctionEntryCoverage(
     "pgo-function-entry-coverage", cl::Hidden,
     cl::desc(
@@ -447,6 +452,15 @@ static const char *ValueProfKindDescr[] = {
 #include "llvm/ProfileData/InstrProfData.inc"
 };
 
+static bool
+shouldInstrumentDenseWaves(const Module &M,
+                           PGOInstrumentationType InstrumentationType) {
+  return PGOInstrumentDenseWaveCounts && M.getTargetTriple().isAMDGPU() &&
+         InstrumentationType == PGOInstrumentationType::FDO &&
+         !PGOFunctionEntryCoverage && !PGOBlockCoverage &&
+         !PGOTemporalInstrumentation;
+}
+
 // Create a COMDAT variable INSTR_PROF_RAW_VERSION_VAR to make the runtime
 // aware this is an ir_level profile so it can set the version flag.
 static GlobalVariable *
@@ -462,6 +476,8 @@ createIRLevelProfileFlagVar(Module &M,
     ProfileVersion |= VARIANT_MASK_INSTR_ENTRY;
   if (PGOInstrumentLoopEntries)
     ProfileVersion |= VARIANT_MASK_INSTR_LOOP_ENTRIES;
+  if (shouldInstrumentDenseWaves(M, InstrumentationType))
+    ProfileVersion |= VARIANT_MASK_DENSE_WAVE;
   if (ProfileCorrelate == InstrProfCorrelator::DEBUG_INFO)
     ProfileVersion |= VARIANT_MASK_DBG_CORRELATE;
   if (PGOFunctionEntryCoverage)
@@ -930,6 +946,17 @@ populateEHOperandBundle(VPCandidateInfo &Cand,
   }
 }
 
+static SmallVector<BasicBlock *>
+getExtraWaveBlocks(Function &F, ArrayRef<BasicBlock *> InstrumentBBs) {
+  SmallVector<BasicBlock *> Extra;
+  SmallPtrSet<BasicBlock *, 32> Measured(InstrumentBBs.begin(),
+                                         InstrumentBBs.end());
+  for (BasicBlock &BB : F)
+    if (!Measured.contains(&BB) && BB.getFirstNonPHIOrDbgOrAlloca() != BB.end())
+      Extra.push_back(&BB);
+  return Extra;
+}
+
 // Visit all edge and instrument the edges not in MST, and do value profiling.
 // Critical edges will be split.
 void FunctionInstrumenter::instrument() {
@@ -966,8 +993,12 @@ void FunctionInstrumenter::instrument() {
 
   std::vector<BasicBlock *> InstrumentBBs;
   FuncInfo.getInstrumentBBs(InstrumentBBs);
-  unsigned NumCounters =
-      InstrumentBBs.size() + FuncInfo.SIVisitor.getNumOfSelectInsts();
+  SmallVector<BasicBlock *> ExtraWaveBBs;
+  if (shouldInstrumentDenseWaves(M, InstrumentationType))
+    ExtraWaveBBs = getExtraWaveBlocks(F, InstrumentBBs);
+  unsigned NumCounters = InstrumentBBs.size() +
+                         FuncInfo.SIVisitor.getNumOfSelectInsts() +
+                         ExtraWaveBBs.size();
 
   if (IsCtxProf) {
     StringSet<> SkipCSInstr(llvm::from_range, CtxPGOSkipCallsiteInstrument);
@@ -1039,6 +1070,15 @@ void FunctionInstrumenter::instrument() {
   // Now instrument select instructions:
   FuncInfo.SIVisitor.instrumentSelects(&I, NumCounters, Name,
                                        FuncInfo.FunctionHash);
+  // Preserve the sparse block/select prefix. A zero lane step still records
+  // one wave visit in the GPU runtime, without affecting lane-flow counts.
+  for (BasicBlock *BB : ExtraWaveBBs) {
+    IRBuilder<> Builder(BB, BB->getFirstNonPHIOrDbgOrAlloca());
+    Builder.CreateIntrinsic(Intrinsic::instrprof_increment_step,
+                            {NormalizedNamePtr, CFGHash,
+                             Builder.getInt32(NumCounters),
+                             Builder.getInt32(I++), Builder.getInt64(0)});
+  }
   assert(I == NumCounters);
 
   if (isValueProfilingDisabled())
@@ -1168,12 +1208,15 @@ class PGOUseFunc {
              BranchProbabilityInfo *BPI, BlockFrequencyInfo *BFIin,
              LoopInfo *LI, ProfileSummaryInfo *PSI, bool IsCS,
              bool InstrumentFuncEntry, bool InstrumentLoopEntries,
-             bool HasSingleByteCoverage)
+             bool HasSingleByteCoverage, bool HasDenseWaveProfile)
       : F(Func), M(Modu), BFI(BFIin), PSI(PSI),
         FuncInfo(Func, TLI, ComdatMembers, false, BPI, BFIin, LI, IsCS,
                  InstrumentFuncEntry, InstrumentLoopEntries,
                  HasSingleByteCoverage),
-        FreqAttr(FFA_Normal), IsCS(IsCS), VPC(Func, TLI) {}
+        FreqAttr(FFA_Normal), IsCS(IsCS),
+        HasDenseWaveProfile(HasDenseWaveProfile && !IsCS &&
+                            Modu->getTargetTriple().isAMDGPU()),
+        VPC(Func, TLI) {}
 
   void handleInstrProfError(Error Err, uint64_t MismatchedFuncSum);
 
@@ -1263,6 +1306,9 @@ class PGOUseFunc {
   // Is to use the context sensitive profile.
   bool IsCS;
 
+  // The input profile owns its counter layout, independently of gen options.
+  bool HasDenseWaveProfile;
+
   ValueProfileCollector VPC;
 
   // Find the Instrumented BB and set the value. Return false on error.
@@ -1316,12 +1362,16 @@ bool PGOUseFunc::setInstrumentedCounts(
   unsigned NumInstrumentedBBs = InstrumentBBs.size();
   unsigned NumSelects = FuncInfo.SIVisitor.getNumOfSelectInsts();
   unsigned NumCounters = NumInstrumentedBBs + NumSelects;
+  SmallVector<BasicBlock *> ExtraWaveBBs;
+  if (HasDenseWaveProfile)
+    ExtraWaveBBs = getExtraWaveBlocks(F, InstrumentBBs);
   // The number of counters here should match the number of counters
   // in profile. Return if they mismatch.
-  if (NumCounters != CountFromProfile.size()) {
+  if (NumCounters + ExtraWaveBBs.size() != CountFromProfile.size()) {
     LLVM_DEBUG({
       dbgs() << "PGO COUNTER MISMATCH for function " << F.getName() << ":\n";
-      dbgs() << "  Expected counters: " << NumCounters << "\n";
+      dbgs() << "  Expected counters: " << NumCounters + ExtraWaveBBs.size()
+             << "\n";
       dbgs() << "    - From instrumented edges: " << NumInstrumentedBBs << "\n";
       for (size_t i = 0; i < InstrumentBBs.size(); ++i) {
         dbgs() << "      " << i << ": ";
@@ -1348,7 +1398,9 @@ bool PGOUseFunc::setInstrumentedCounts(
       CountValue = 1;
     Info.setBBInfoCount(CountValue);
   }
-  ProfileCountSize = CountFromProfile.size();
+  // The appended zero-step slots do not participate in lane-flow or select
+  // reconstruction. Uniformity also retains its original sparse indices.
+  ProfileCountSize = NumCounters;
   CountPosition = I;
 
   // Set the edge count and update the count of unknown edges for BBs.
@@ -2307,7 +2359,7 @@ static bool annotateAllFunctions(
     }
     PGOUseFunc Func(F, &M, TLI, ComdatMembers, BPI, BFI, LI, PSI, IsCS,
                     InstrumentFuncEntry, InstrumentLoopEntries,
-                    HasSingleByteCoverage);
+                    HasSingleByteCoverage, PGOReader->hasDenseWaveProfile());
     if (!Func.getRecord(PGOReader.get()))
       continue;
     if (HasSingleByteCoverage) {
diff --git a/llvm/test/Transforms/PGOProfile/dense-wave-cfg.ll b/llvm/test/Transforms/PGOProfile/dense-wave-cfg.ll
new file mode 100644
index 0000000000000..b6f4924fc6632
--- /dev/null
+++ b/llvm/test/Transforms/PGOProfile/dense-wave-cfg.ll
@@ -0,0 +1,108 @@
+; RUN: split-file %s %t
+; RUN: opt -passes=pgo-instr-gen,verify -pgo-instrument-entry -S %t/loop.ll | FileCheck %s --check-prefix=GEN --implicit-check-not='call void @llvm.instrprof.increment'
+; RUN: %python %t/raw.py > %t/loop.raw
+; RUN: llvm-profdata merge %t/loop.raw -o %t/loop.profdata
+; RUN: opt -passes=pgo-instr-use,verify -pgo-test-profile-file=%t/loop.profdata -S %t/loop.ll -o %t/use.ll
+; RUN: FileCheck %s --check-prefix=COUNTS < %t/use.ll
+; RUN: opt -passes=pgo-instr-gen,verify -pgo-instrument-entry -S %t/eh.ll | FileCheck %s --check-prefix=EH --implicit-check-not='call void @llvm.instrprof.increment'
+
+; Dense instrumentation preserves the entry and split backedge counters,
+; then appends the header and exit measurements.
+; GEN: @__llvm_profile_raw_version = {{.*}}constant i64 378302368699121676
+; GEN-LABEL: define void @loop(
+; GEN: entry:
+; GEN-NEXT: call void @llvm.instrprof.increment({{.*}}i64 [[HASH:784007056844089447]], i32 4, i32 0)
+; GEN-NEXT: br label %header
+; GEN: header:
+; GEN-NEXT: %i = phi i32 [ 0, %entry ], [ %next, %header.header_crit_edge ]
+; GEN-NEXT: call void @llvm.instrprof.increment.step({{.*}}i64 [[HASH]], i32 4, i32 2, i64 0)
+; GEN: header.header_crit_edge:
+; GEN-NEXT: call void @llvm.instrprof.increment({{.*}}i64 [[HASH]], i32 4, i32 1)
+; GEN-NEXT: br label %header
+; GEN: exit:
+; GEN-NEXT: call void @llvm.instrprof.increment.step({{.*}}i64 [[HASH]], i32 4, i32 3, i64 0)
+; GEN-NEXT: ret void
+
+; Profile use reconstructs counter placement from the recorded flags. The
+; header has four times the entry wave count; sparse data cannot supply it.
+; COUNTS-LABEL: define void @loop(
+; COUNTS-SAME: !prof ![[ENTRY_COUNT:[0-9]+]]
+; COUNTS: br i1 %done, label %exit, label %header.header_crit_edge, !prof ![[WEIGHTS:[0-9]+]]
+; COUNTS-DAG: ![[ENTRY_COUNT]] = !{!"function_entry_count", i64 192}
+; COUNTS-DAG: ![[WEIGHTS]] = !{!"branch_weights", i32 192, i32 576}
+
+; A catchswitch has no legal insertion point. Skip dispatch and append only
+; exit; the sparse catch counter must remain after its catchpad.
+; EH-LABEL: define void @eh(
+; EH: entry:
+; EH-NEXT: call void @llvm.instrprof.increment({{.*}}i64 [[EH_HASH:146835646621254984]], i32 3, i32 0)
+; EH: dispatch:
+; EH-NEXT: %cs = catchswitch within none [label %catch] unwind to caller
+; EH: catch:
+; EH-NEXT: %cp = catchpad within %cs [ptr null, i32 64, ptr null]
+; EH-NEXT: call void @llvm.instrprof.increment({{.*}}i64 [[EH_HASH]], i32 3, i32 1)
+; EH-NEXT: catchret from %cp to label %exit
+; EH: exit:
+; EH-NEXT: call void @llvm.instrprof.increment.step({{.*}}i64 [[EH_HASH]], i32 3, i32 2, i64 0)
+; EH-NEXT: ret void
+
+;--- loop.ll
+target triple = "amdgcn-amd-amdhsa"
+
+define void @loop(ptr %p, i32 %n) {
+entry:
+  br label %header
+header:
+  %i = phi i32 [0, %entry], [%next, %header]
+  store volatile i32 %i, ptr %p
+  %next = add i32 %i, 1
+  %done = icmp eq i32 %next, %n
+  br i1 %done, label %exit, label %header
+exit:
+  ret void
+}
+
+;--- eh.ll
+target triple = "amdgcn-amd-amdhsa"
+
+define void @eh() personality ptr @__CxxFrameHandler3 {
+entry:
+  invoke void @may_throw() to label %exit unwind label %dispatch
+dispatch:
+  %cs = catchswitch within none [label %catch] unwind to caller
+catch:
+  %cp = catchpad within %cs [ptr null, i32 64, ptr null]
+  catchret from %cp to label %exit
+exit:
+  ret void
+}
+
+declare void @may_throw()
+declare i32 @__CxxFrameHandler3(...)
+
+;--- raw.py
+import hashlib
+import struct
+import sys
+
+# Three full waves, each taking four loop iterations. Counter order is
+# entry, backedge, header, exit, as checked in the generation above.
+name = b"loop"
+name_ref = int.from_bytes(hashlib.md5(name).digest()[:8], "little")
+lanes = [192, 576, 0, 0]
+waves = [3, 9, 12, 3]
+version = 12 | (1 << 54) | (1 << 56) | (1 << 58)
+names = bytes([len(name), 0]) + name
+# Raw v12: 80-byte data record, eight i64 lane/wave counters, four uniform counters.
+counter_delta = 80
+uniform_delta = counter_delta + 8 * (len(lanes) + len(waves))
+names_delta = uniform_delta + 8 * len(lanes)
+header = [0xff6c70726f667281, version, 0, 1, 0, 8, 0,
+          0, 0, 4, 0, uniform_delta, len(names), counter_delta,
+          uniform_delta, names_delta, 0, 0, 2]
+record = struct.pack("<7QI4HII4x", name_ref, 784007056844089447, counter_delta,
+                     uniform_delta, 0, 0, 0, 8, 0, 0, 0, 64, 0, 4)
+counts = lanes + waves + lanes
+sys.stdout.buffer.write(struct.pack("<19Q", *header) + record +
+                        struct.pack("<12Q", *counts) +
+                        names + bytes((-len(names)) % 8))
diff --git a/llvm/test/Transforms/PGOProfile/dense-wave-instrumentation.ll b/llvm/test/Transforms/PGOProfile/dense-wave-instrumentation.ll
new file mode 100644
index 0000000000000..86c1cab9ce600
--- /dev/null
+++ b/llvm/test/Transforms/PGOProfile/dense-wave-instrumentation.ll
@@ -0,0 +1,59 @@
+; RUN: opt -passes=pgo-instr-gen -pgo-instrument-entry -S %s | FileCheck %s --check-prefix=DENSE
+; RUN: opt -passes=pgo-instr-gen -pgo-instrument-entry -pgo-instrument-dense-wave-counts=false -S %s | FileCheck %s --check-prefix=SPARSE
+; RUN: opt -passes=pgo-instr-gen -pgo-instrument-entry -mtriple=x86_64-unknown-linux-gnu -S %s | FileCheck %s --check-prefix=SPARSE
+; RUN: opt -passes=pgo-instr-gen -pgo-instrument-entry=false -S %s | FileCheck %s --check-prefix=NOENTRY
+; RUN: opt -passes=pgo-instr-gen -pgo-function-entry-coverage -S %s | FileCheck %s --check-prefix=ENTRY-COV --implicit-check-not='i64 0)'
+; RUN: opt -passes=pgo-instr-gen -pgo-block-coverage -S %s | FileCheck %s --check-prefix=BLOCK-COV --implicit-check-not='i64 0)'
+; RUN: opt -passes=pgo-instr-gen -pgo-temporal-instrumentation -S %s | FileCheck %s --check-prefix=TEMPORAL --implicit-check-not='i64 0)'
+; RUN: not opt -passes='default<O2>' -cs-profilegen-file=dense-wave -cspgo-kind=cspgo-instr-gen-pipeline -pgo-instrument-entry -print-before=instrprof -disable-output %s 2>&1 | FileCheck %s --check-prefix=CS --implicit-check-not='i64 0)'
+
+target triple = "amdgcn-amd-amdhsa"
+
+; Dense generation keeps the original block/select prefix, then appends the
+; unmeasured blocks in function order. The zero steps measure waves only.
+; DENSE: @__llvm_profile_raw_version = {{.*}}constant i64 378302368699121676
+; SPARSE: @__llvm_profile_raw_version = {{.*}}constant i64 360287970189639692
+; NOENTRY: @__llvm_profile_raw_version = {{.*}}constant i64 90071992547409932
+; ENTRY-COV: @__llvm_profile_raw_version = {{.*}}constant i64 3530822107858468876
+; BLOCK-COV: @__llvm_profile_raw_version = {{.*}}constant i64 1224979098644774924
+; TEMPORAL: @__llvm_profile_raw_version = {{.*}}constant i64 -9151314442816847860
+; CS: @__llvm_profile_raw_version = {{.*}}constant i64 504403158265495564
+; CS: LLVM ERROR: wave counts require ordinary counter increments
+;
+; DENSE-LABEL: define void @diamond
+; DENSE: entry:
+; DENSE-NEXT: call void @llvm.instrprof.increment({{.*}}i64 [[HASH:942389667449461396]], i32 5, i32 0)
+; DENSE: a:
+; DENSE-NEXT: call void @llvm.instrprof.increment.step({{.*}}i64 [[HASH]], i32 5, i32 3, i64 0)
+; DENSE: b:
+; DENSE-NEXT: call void @llvm.instrprof.increment({{.*}}i64 [[HASH]], i32 5, i32 1)
+; DENSE: exit:
+; DENSE-NEXT: call void @llvm.instrprof.increment.step({{.*}}i64 [[HASH]], i32 5, i32 4, i64 0)
+; DENSE: call void @llvm.instrprof.increment.step({{.*}}i64 [[HASH]], i32 5, i32 2, i64 {{%.*}})
+;
+; SPARSE-LABEL: define void @diamond
+; SPARSE: call void @llvm.instrprof.increment({{.*}}i32 3, i32 0)
+; SPARSE: a:
+; SPARSE-NEXT: store volatile
+; SPARSE: call void @llvm.instrprof.increment({{.*}}i32 3, i32 1)
+; SPARSE: exit:
+; SPARSE: call void @llvm.instrprof.increment.step({{.*}}i32 3, i32 2, i64 {{%.*}})
+;
+; Even without forced entry instrumentation, the entry gets a direct wave site.
+; NOENTRY-LABEL: define void @diamond
+; NOENTRY: entry:
+; NOENTRY-NEXT: call void @llvm.instrprof.increment.step({{.*}}i64 942389667449461396, i32 5, i32 3, i64 0)
+define void @diamond(i1 %cond, i1 %select_cond, ptr %p) {
+entry:
+  br i1 %cond, label %a, label %b
+a:
+  store volatile i32 1, ptr %p
+  br label %exit
+b:
+  store volatile i32 2, ptr %p
+  br label %exit
+exit:
+  %value = select i1 %select_cond, i32 1, i32 2
+  store volatile i32 %value, ptr %p
+  ret void
+}
diff --git a/llvm/test/Transforms/PGOProfile/dense-wave-profile-use.ll b/llvm/test/Transforms/PGOProfile/dense-wave-profile-use.ll
new file mode 100644
index 0000000000000..6229d375f6fb1
--- /dev/null
+++ b/llvm/test/Transforms/PGOProfile/dense-wave-profile-use.ll
@@ -0,0 +1,82 @@
+; RUN: split-file %s %t
+; RUN: %python %t/raw.py dense > %t/dense.raw
+; RUN: llvm-profdata merge %t/dense.raw -o %t/dense.profdata
+; RUN: opt -passes=pgo-instr-use,verify -pgo-test-profile-file=%t/dense.profdata -S %t/input.ll -o %t/dense.ll
+; RUN: FileCheck %s --check-prefix=COUNTS < %t/dense.ll
+; RUN: FileCheck %s --check-prefix=UNIFORM < %t/dense.ll
+; RUN: opt -passes=pgo-instr-use,verify -pgo-test-profile-file=%t/dense.profdata -pgo-instrument-dense-wave-counts=false -S %t/input.ll -o %t/disabled-gen.ll
+; RUN: diff %t/dense.ll %t/disabled-gen.ll
+; RUN: %python %t/raw.py no-entry > %t/no-entry.raw
+; RUN: llvm-profdata merge %t/no-entry.raw -o %t/no-entry.profdata
+; RUN: opt -passes=pgo-instr-use,verify -pgo-test-profile-file=%t/no-entry.profdata -S %t/input.ll | FileCheck %s --check-prefix=COUNTS
+; RUN: %python %t/raw.py sparse > %t/sparse.raw
+; RUN: not llvm-profdata merge %t/dense.raw %t/sparse.raw -o %t/mixed 2>&1 | FileCheck %s --check-prefix=MIXED
+; RUN: not llvm-profdata merge %t/sparse.raw %t/dense.raw -o %t/mixed 2>&1 | FileCheck %s --check-prefix=MIXED
+; RUN: cat %t/dense.raw %t/sparse.raw > %t/mixed.raw
+; RUN: not llvm-profdata merge %t/mixed.raw -o %t/mixed 2>&1 | FileCheck %s --check-prefix=MIXED
+; RUN: cat %t/sparse.raw %t/dense.raw > %t/mixed.raw
+; RUN: not llvm-profdata merge %t/mixed.raw -o %t/mixed 2>&1 | FileCheck %s --check-prefix=MIXED
+
+; The profile flags select the layout independently of generation options.
+; Appended slots do not change scalar/select counts or uniformity coverage.
+; MIXED: cannot merge dense and sparse wave instrumentation layouts
+; COUNTS-DAG: !{!"function_entry_count", i64 6400}
+; COUNTS-DAG: !{!"branch_weights", i32 3200, i32 3200}
+; COUNTS-DAG: !{!"branch_weights", i32 1600, i32 4800}
+; UNIFORM: define void @diamond({{.*}}!uniformity.profile
+; UNIFORM: br i1 %cond, {{.*}}!block.uniformity.profile{{.*}}!branch.uniformity.profile
+; UNIFORM: b:
+; UNIFORM: br label %exit, {{.*}}!block.uniformity.profile
+
+;--- input.ll
+source_filename = "wave-profile-use.ll"
+target triple = "amdgcn-amd-amdhsa"
+define void @diamond(i1 %cond, i1 %select_cond, ptr %p) {
+entry:
+  br i1 %cond, label %a, label %b
+a:
+  store volatile i32 1, ptr %p
+  br label %exit
+b:
+  store volatile i32 2, ptr %p
+  br label %exit
+exit:
+  %value = select i1 %select_cond, i32 1, i32 2
+  store volatile i32 %value, ptr %p
+  ret void
+}
+
+;--- raw.py
+import hashlib
+import struct
+import sys
+
+mode = sys.argv[1]
+name = b"diamond"
+name_ref = int.from_bytes(hashlib.md5(name).digest()[:8], "little")
+# Preserve the sparse block/select prefix and append two zero lane steps.
+lanes = [6400, 3200, 1600]
+waves = [100, 50, 100]
+version = 12 | (1 << 56) | (1 << 58)
+if mode != "sparse":
+    version |= 1 << 54
+    lanes += [0, 0]
+    waves += [100, 100]
+if mode == "no-entry":
+    version &= ~(1 << 58)
+    lanes[0] = 3200
+num_counters = len(lanes) + len(waves)
+counter_delta = 80
+uniform_delta = counter_delta + num_counters * 8
+names_delta = uniform_delta + len(lanes) * 8
+names = bytes([len(name), 0]) + name
+header = [0xff6c70726f667281, version, 0, 1, 0, num_counters, 0,
+          0, 0, len(lanes), 0, uniform_delta, len(names), counter_delta,
+          uniform_delta, names_delta, 0, 0, 2]
+record = struct.pack("<7QI4HII4x", name_ref, 942389667449461396, counter_delta,
+                     uniform_delta, 0, 0, 0, num_counters, 0, 0, 0, 64, 0,
+                     len(waves))
+counts = lanes + waves + lanes
+sys.stdout.buffer.write(struct.pack("<19Q", *header) + record +
+                        struct.pack("<" + "Q" * len(counts), *counts) +
+                        names + bytes((-len(names)) % 8))
diff --git a/llvm/test/tools/llvm-profdata/dense-wave-layout.test b/llvm/test/tools/llvm-profdata/dense-wave-layout.test
new file mode 100644
index 0000000000000..5788a8bf64e79
--- /dev/null
+++ b/llvm/test/tools/llvm-profdata/dense-wave-layout.test
@@ -0,0 +1,43 @@
+# RUN: split-file %s %t
+# RUN: llvm-profdata merge %t/dense.txt -o %t/dense.profdata
+# RUN: llvm-profdata show --profile-version %t/dense.profdata | FileCheck %s --check-prefix=DENSE
+# RUN: llvm-profdata merge --text %t/dense.profdata -o %t/roundtrip.txt
+# RUN: FileCheck %s --check-prefix=TEXT < %t/roundtrip.txt
+# RUN: llvm-profdata merge %t/roundtrip.txt -o %t/roundtrip.profdata
+# RUN: llvm-profdata show --profile-version %t/roundtrip.profdata | FileCheck %s --check-prefix=DENSE
+# RUN: llvm-profdata merge %t/sparse.txt -o %t/sparse.profdata
+# RUN: llvm-profdata show --profile-version %t/sparse.profdata | FileCheck %s --check-prefix=SPARSE
+# RUN: not llvm-profdata merge -j 1 %t/dense.profdata %t/sparse.profdata -o %t/mixed 2>&1 | FileCheck %s --check-prefix=ERROR
+# RUN: not llvm-profdata merge -j 1 %t/sparse.profdata %t/dense.profdata -o %t/mixed 2>&1 | FileCheck %s --check-prefix=ERROR
+# RUN: not llvm-profdata merge -j 2 %t/dense.profdata %t/sparse.profdata -o %t/mixed 2>&1 | FileCheck %s --check-prefix=ERROR
+# RUN: not llvm-profdata merge -j 2 %t/sparse.profdata %t/dense.profdata -o %t/mixed 2>&1 | FileCheck %s --check-prefix=ERROR
+# RUN: llvm-profdata merge -j 2 %t/dense.profdata %t/dense.profdata -o %t/double.profdata
+# RUN: llvm-profdata show --profile-version %t/double.profdata | FileCheck %s --check-prefix=DENSE
+
+# RUN: touch %t/empty.raw
+# RUN: llvm-profdata merge -j 2 %t/dense.profdata %t/empty.raw -o %t/empty-merge.profdata
+# RUN: llvm-profdata show --profile-version %t/empty-merge.profdata | FileCheck %s --check-prefix=DENSE
+# RUN: llvm-profdata merge -j 2 %t/empty.raw %t/dense.profdata -o %t/empty-merge.profdata
+# RUN: llvm-profdata show --profile-version %t/empty-merge.profdata | FileCheck %s --check-prefix=DENSE
+
+# Header round trips use a lane-only record; wave payload text export remains
+# unsupported. Equal-length records must still reject incompatible layouts.
+# DENSE: dense_wave = 1
+# SPARSE: dense_wave = 0
+# TEXT: :dense_wave
+# ERROR: cannot merge dense and sparse wave instrumentation layouts
+
+#--- dense.txt
+:ir
+:dense_wave
+same
+123
+1
+0
+
+#--- sparse.txt
+:ir
+same
+123
+1
+0
diff --git a/llvm/tools/llvm-profdata/llvm-profdata.cpp b/llvm/tools/llvm-profdata/llvm-profdata.cpp
index b0c5517473716..50c9d1486ea22 100644
--- a/llvm/tools/llvm-profdata/llvm-profdata.cpp
+++ b/llvm/tools/llvm-profdata/llvm-profdata.cpp
@@ -910,8 +910,14 @@ static void mergeWriterContexts(WriterContext *Dst, WriterContext *Src) {
     Dst->Errors.push_back(std::move(ErrorPair));
   Src->Errors.clear();
 
-  if (Error E = Dst->Writer.mergeProfileKind(Src->Writer.getProfileKind()))
-    exitWithError(std::move(E));
+  // Empty input files have no layout to reconcile. In particular, a worker
+  // that only saw empty files must not turn a dense wave profile into a mixed
+  // dense/sparse merge.
+  if (Src->Writer.getProfileKind() != InstrProfKind::Unknown ||
+      !Src->Writer.getProfileData().empty()) {
+    if (Error E = Dst->Writer.mergeProfileKind(Src->Writer.getProfileKind()))
+      exitWithError(std::move(E));
+  }
 
   Dst->Writer.mergeRecordsFromWriter(std::move(Src->Writer), [&](Error E) {
     auto [ErrorCode, Msg] = InstrProfError::take(std::move(E));
@@ -3080,6 +3086,7 @@ static int showInstrProfile(ShowFormat SFormat, raw_fd_ostream &OS) {
   if (IsIR) {
     OS << "  entry_first = " << Reader->instrEntryBBEnabled();
     OS << "  instrument_loop_entries = " << Reader->instrLoopEntriesEnabled();
+    OS << "  dense_wave = " << Reader->hasDenseWaveProfile();
   }
   OS << "\n";
   if (ShowAllFunctions || !FuncNameFilter.empty())
diff --git a/llvm/unittests/ProfileData/InstrProfTest.cpp b/llvm/unittests/ProfileData/InstrProfTest.cpp
index 94517d7d02192..bb0492f6afbd5 100644
--- a/llvm/unittests/ProfileData/InstrProfTest.cpp
+++ b/llvm/unittests/ProfileData/InstrProfTest.cpp
@@ -151,6 +151,37 @@ TEST_F(InstrProfTest, WaveCountsMismatchDoesNotMutate) {
   EXPECT_THAT(A.WaveCounts, ElementsAre(1));
 }
 
+TEST_F(InstrProfTest, DenseWaveLayoutRoundTrip) {
+  ASSERT_THAT_ERROR(
+      Writer.mergeProfileKind(InstrProfKind::IRInstrumentation |
+                              InstrProfKind::DenseWaveInstrumentation),
+      Succeeded());
+  NamedInstrProfRecord R("wave", 123, {64, 0});
+  R.WaveCounts = {1, 2};
+  Writer.addRecord(std::move(R), Err);
+  readProfile(Writer.writeBuffer());
+  EXPECT_TRUE(Reader->hasDenseWaveProfile());
+  auto Record = Reader->getInstrProfRecord("wave", 123);
+  ASSERT_THAT_ERROR(Record.takeError(), Succeeded());
+  EXPECT_THAT(Record->Counts, ElementsAre(64, 0));
+  EXPECT_THAT(Record->WaveCounts, ElementsAre(1, 2));
+}
+
+TEST_F(InstrProfTest, DenseWaveLayoutMismatchDoesNotMutate) {
+  auto Sparse = InstrProfKind::IRInstrumentation;
+  auto Dense = Sparse | InstrProfKind::DenseWaveInstrumentation;
+  for (bool DenseFirst : {false, true}) {
+    InstrProfWriter W;
+    auto First = DenseFirst ? Dense : Sparse;
+    auto Second = DenseFirst ? Sparse : Dense;
+    ASSERT_THAT_ERROR(W.mergeProfileKind(First), Succeeded());
+    EXPECT_TRUE(ErrorEquals(instrprof_error::unsupported_version,
+                            W.mergeProfileKind(Second)));
+    EXPECT_EQ(W.getProfileKind(), First);
+    EXPECT_THAT_ERROR(W.mergeProfileKind(First), Succeeded());
+  }
+}
+
 TEST_F(InstrProfTest, WaveCountsScaleCopyClearAndOverflow) {
   InstrProfRecord A({64, 128});
   A.WaveCounts = {2, 4};



More information about the llvm-branch-commits mailing list