[clang] [llvm] [AMDGPU] Add intrinsics and builtins for gfx13 exclusive scan (PR #226078)
Mariusz Sikora via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 28 04:39:52 PDT 2026
https://github.com/mariusz-sikora-at-amd updated https://github.com/llvm/llvm-project/pull/226078
>From 8e706d8b7ca65d064255d19f7cbb723c4969c905 Mon Sep 17 00:00:00 2001
From: Mariusz Sikora <mariusz.sikora at amd.com>
Date: Tue, 22 Sep 2026 08:11:21 -0400
Subject: [PATCH 1/3] [AMDGPU] Add intrinsics and builtins for gfx13 exclusive
scan
---
clang/include/clang/Basic/BuiltinsAMDGPU.td | 21 +
.../include/clang/Basic/BuiltinsAMDGPUDocs.td | 20 +
.../builtins-amdgcn-gfx12-err.cl | 16 +
.../CodeGenOpenCL/builtins-amdgcn-gfx13.cl | 260 +++
...ltins-amdgcn-gfx13-exclusive-scan-types.cl | 42 +
llvm/include/llvm/IR/IntrinsicsAMDGPU.td | 67 +
llvm/lib/Target/AMDGPU/AMDGPU.td | 2 +-
.../AMDGPU/AMDGPURegBankLegalizeRules.cpp | 21 +
.../Target/AMDGPU/AMDGPURegisterBankInfo.cpp | 13 +
llvm/lib/Target/AMDGPU/VOP3Instructions.td | 38 +-
.../AMDGPU/llvm.amdgcn.exclusive.scan.ll | 1594 +++++++++++++++++
11 files changed, 2075 insertions(+), 19 deletions(-)
create mode 100644 clang/test/SemaOpenCL/builtins-amdgcn-gfx13-exclusive-scan-types.cl
create mode 100644 llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exclusive.scan.ll
diff --git a/clang/include/clang/Basic/BuiltinsAMDGPU.td b/clang/include/clang/Basic/BuiltinsAMDGPU.td
index 4fd604390d3ce3..ef789e1aa00283 100644
--- a/clang/include/clang/Basic/BuiltinsAMDGPU.td
+++ b/clang/include/clang/Basic/BuiltinsAMDGPU.td
@@ -928,6 +928,27 @@ def __builtin_amdgcn_swmmac_f32_16x16x32_bf8_fp8_w64 : AMDGPUBuiltin<"_ExtVector
def __builtin_amdgcn_swmmac_f32_16x16x32_bf8_bf8_w64 : AMDGPUBuiltin<"_ExtVector<4, float>(int, _ExtVector<2, int>, _ExtVector<4, float>, int)", [Const], "swmmac-gfx1200-insts,wavefrontsize64">;
def __builtin_amdgcn_prng_b32 : AMDGPUBuiltin<"unsigned int(unsigned int)", [Const], "prng-inst">;
+
+class AMDGPUExclusiveScanBuiltin<string prototype, list<string> argNames>
+ : AMDGPUBuiltin<prototype, [Const], "exclusive-scan-insts"> {
+ let Documentation = [DocExclusiveScan];
+ let ArgNames = argNames;
+}
+
+def __builtin_amdgcn_exclusive_scan_sum_i32 : AMDGPUExclusiveScanBuiltin<"int(int, int, bool)", ["src0", "src1", "clamp"]>;
+def __builtin_amdgcn_exclusive_scan_sum_u32 : AMDGPUExclusiveScanBuiltin<"unsigned int(unsigned int, unsigned int, bool)", ["src0", "src1", "clamp"]>;
+def __builtin_amdgcn_exclusive_scan_xor_b32 : AMDGPUExclusiveScanBuiltin<"int(int, int)", ["src0", "src1"]>;
+def __builtin_amdgcn_exclusive_scan_or_b32 : AMDGPUExclusiveScanBuiltin<"int(int, int)", ["src0", "src1"]>;
+def __builtin_amdgcn_exclusive_scan_and_b32 : AMDGPUExclusiveScanBuiltin<"int(int, int)", ["src0", "src1"]>;
+def __builtin_amdgcn_exclusive_scan_min_i16 : AMDGPUExclusiveScanBuiltin<"short(short, int)", ["src0", "src1"]>;
+def __builtin_amdgcn_exclusive_scan_min_u16 : AMDGPUExclusiveScanBuiltin<"unsigned short(unsigned short, int)", ["src0", "src1"]>;
+def __builtin_amdgcn_exclusive_scan_min_i32 : AMDGPUExclusiveScanBuiltin<"int(int, int)", ["src0", "src1"]>;
+def __builtin_amdgcn_exclusive_scan_min_u32 : AMDGPUExclusiveScanBuiltin<"unsigned int(unsigned int, unsigned int)", ["src0", "src1"]>;
+def __builtin_amdgcn_exclusive_scan_max_i16 : AMDGPUExclusiveScanBuiltin<"short(short, int)", ["src0", "src1"]>;
+def __builtin_amdgcn_exclusive_scan_max_u16 : AMDGPUExclusiveScanBuiltin<"unsigned short(unsigned short, int)", ["src0", "src1"]>;
+def __builtin_amdgcn_exclusive_scan_max_i32 : AMDGPUExclusiveScanBuiltin<"int(int, int)", ["src0", "src1"]>;
+def __builtin_amdgcn_exclusive_scan_max_u32 : AMDGPUExclusiveScanBuiltin<"unsigned int(unsigned int, unsigned int)", ["src0", "src1"]>;
+
def __builtin_amdgcn_cvt_scalef32_pk32_fp6_f16 : AMDGPUBuiltin<"_ExtVector<6, unsigned int>(_ExtVector<32, _Float16>, float)", [Const], "f16bf16-to-fp6bf6-cvt-scale-insts">;
def __builtin_amdgcn_cvt_scalef32_pk32_bf6_f16 : AMDGPUBuiltin<"_ExtVector<6, unsigned int>(_ExtVector<32, _Float16>, float)", [Const], "f16bf16-to-fp6bf6-cvt-scale-insts">;
def __builtin_amdgcn_cvt_scalef32_pk32_fp6_bf16 : AMDGPUBuiltin<"_ExtVector<6, unsigned int>(_ExtVector<32, __bf16>, float)", [Const], "f16bf16-to-fp6bf6-cvt-scale-insts">;
diff --git a/clang/include/clang/Basic/BuiltinsAMDGPUDocs.td b/clang/include/clang/Basic/BuiltinsAMDGPUDocs.td
index bfccd7de6590a2..b9b54cce4f28f2 100644
--- a/clang/include/clang/Basic/BuiltinsAMDGPUDocs.td
+++ b/clang/include/clang/Basic/BuiltinsAMDGPUDocs.td
@@ -653,6 +653,26 @@ nondeterministic value.
}];
}
+def DocExclusiveScan : Documentation {
+ let Category = DocCatWaveDataExchange;
+ let Content = [{
+Performs an exclusive prefix scan of ``src0`` within a subgroup of lanes. Each
+lane receives the running combination of the ``src0`` values of all earlier
+lanes in its subgroup, walking from the lowest to the highest lane index.
+"Exclusive" means a lane's own ``src0`` is not included in its result, so the
+first (lowest) lane of each subgroup receives the operation's identity value.
+
+``src1`` is a 32-bit subgroup mask: bit N is set when lane N belongs to the same
+subgroup as the current lane.
+
+The combining operation is named by the builtin (``sum``, ``xor``, ``or``,
+``and``, ``min``, ``max``).
+
+In wave64 mode the two 32-lane halves of the wave are scanned independently. The
+builtins are convergent: all lanes of a subgroup must execute them together.
+}];
+}
+
def DocWMMA_scale16_GFX1250 : Documentation {
let Category = DocCatWMMA;
let Content = [{
diff --git a/clang/test/CodeGenOpenCL/builtins-amdgcn-gfx12-err.cl b/clang/test/CodeGenOpenCL/builtins-amdgcn-gfx12-err.cl
index d8fb243865f76d..8bc07bf0341289 100644
--- a/clang/test/CodeGenOpenCL/builtins-amdgcn-gfx12-err.cl
+++ b/clang/test/CodeGenOpenCL/builtins-amdgcn-gfx12-err.cl
@@ -3,6 +3,7 @@
// RUN: %clang_cc1 -triple amdgpu12.00-unknown-unknown -verify -emit-llvm -o - %s
typedef unsigned int uint;
+typedef unsigned short ushort;
#pragma OPENCL EXTENSION cl_khr_fp64:enable
@@ -32,6 +33,7 @@ void builtin_test_unsupported(double a_double, float a_float,
v2i a_v2i, v4i a_v4i, v16i a_v16i, v32i a_v32i,
v2f a_v2f, v4f a_v4f, v16f a_v16f, v32f a_v32f,
v4h a_v4h, v8h a_v8h,
+ short a_short, ushort a_ushort,
uint a, uint b) {
@@ -94,4 +96,18 @@ void builtin_test_unsupported(double a_double, float a_float,
a_v16f = __builtin_amdgcn_smfmac_f32_32x32x32_bf8_fp8(a_v2i, a_v4i, a_v16f, a_int, 0, 0); // expected-error {{'__builtin_amdgcn_smfmac_f32_32x32x32_bf8_fp8' needs target feature fp8-insts}}
a_v16f = __builtin_amdgcn_smfmac_f32_32x32x32_fp8_bf8(a_v2i, a_v4i, a_v16f, a_int, 0, 0); // expected-error {{'__builtin_amdgcn_smfmac_f32_32x32x32_fp8_bf8' needs target feature fp8-insts}}
a_v16f = __builtin_amdgcn_smfmac_f32_32x32x32_fp8_fp8(a_v2i, a_v4i, a_v16f, a_int, 0, 0); // expected-error {{'__builtin_amdgcn_smfmac_f32_32x32x32_fp8_fp8' needs target feature fp8-insts}}
+
+ a_int = __builtin_amdgcn_exclusive_scan_sum_i32(a_int, a_int, true); // expected-error {{'__builtin_amdgcn_exclusive_scan_sum_i32' needs target feature exclusive-scan-insts}}
+ a = __builtin_amdgcn_exclusive_scan_sum_u32(a, a, true); // expected-error {{'__builtin_amdgcn_exclusive_scan_sum_u32' needs target feature exclusive-scan-insts}}
+ a_int = __builtin_amdgcn_exclusive_scan_xor_b32(a_int, a_int); // expected-error {{'__builtin_amdgcn_exclusive_scan_xor_b32' needs target feature exclusive-scan-insts}}
+ a_int = __builtin_amdgcn_exclusive_scan_or_b32(a_int, a_int); // expected-error {{'__builtin_amdgcn_exclusive_scan_or_b32' needs target feature exclusive-scan-insts}}
+ a_int = __builtin_amdgcn_exclusive_scan_and_b32(a_int, a_int); // expected-error {{'__builtin_amdgcn_exclusive_scan_and_b32' needs target feature exclusive-scan-insts}}
+ a_short = __builtin_amdgcn_exclusive_scan_min_i16(a_short, a_int); // expected-error {{'__builtin_amdgcn_exclusive_scan_min_i16' needs target feature exclusive-scan-insts}}
+ a_ushort = __builtin_amdgcn_exclusive_scan_min_u16(a_ushort, a_int); // expected-error {{'__builtin_amdgcn_exclusive_scan_min_u16' needs target feature exclusive-scan-insts}}
+ a_int = __builtin_amdgcn_exclusive_scan_min_i32(a_int, a_int); // expected-error {{'__builtin_amdgcn_exclusive_scan_min_i32' needs target feature exclusive-scan-insts}}
+ a = __builtin_amdgcn_exclusive_scan_min_u32(a, a); // expected-error {{'__builtin_amdgcn_exclusive_scan_min_u32' needs target feature exclusive-scan-insts}}
+ a_short = __builtin_amdgcn_exclusive_scan_max_i16(a_short, a_int); // expected-error {{'__builtin_amdgcn_exclusive_scan_max_i16' needs target feature exclusive-scan-insts}}
+ a_ushort = __builtin_amdgcn_exclusive_scan_max_u16(a_ushort, a_int); // expected-error {{'__builtin_amdgcn_exclusive_scan_max_u16' needs target feature exclusive-scan-insts}}
+ a_int = __builtin_amdgcn_exclusive_scan_max_i32(a_int, a_int); // expected-error {{'__builtin_amdgcn_exclusive_scan_max_i32' needs target feature exclusive-scan-insts}}
+ a = __builtin_amdgcn_exclusive_scan_max_u32(a, a); // expected-error {{'__builtin_amdgcn_exclusive_scan_max_u32' needs target feature exclusive-scan-insts}}
}
diff --git a/clang/test/CodeGenOpenCL/builtins-amdgcn-gfx13.cl b/clang/test/CodeGenOpenCL/builtins-amdgcn-gfx13.cl
index 00e65563a18684..21306cb3a083ae 100644
--- a/clang/test/CodeGenOpenCL/builtins-amdgcn-gfx13.cl
+++ b/clang/test/CodeGenOpenCL/builtins-amdgcn-gfx13.cl
@@ -8,6 +8,7 @@ typedef unsigned int __attribute__((ext_vector_type(6))) uint6;
typedef float __attribute__((ext_vector_type(32))) float32;
typedef __bf16 __attribute__((ext_vector_type(2))) bfloat2;
typedef unsigned int uint;
+typedef unsigned short ushort;
// CHECK-LABEL: @test_cvt_scalef32_pk32_bf6_f32(
// CHECK-NEXT: entry:
@@ -211,3 +212,262 @@ void test_s_prefetch_data(global float *gp, unsigned int len)
__builtin_amdgcn_s_prefetch_data(gp, len);
}
+
+// CHECK-LABEL: @test_exclusive_scan_sum_i32(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 [[TMP0]], i32 [[TMP1]], i1 true)
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[TMP3]], align 4
+// CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP5:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP6:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 [[TMP4]], i32 [[TMP5]], i1 false)
+// CHECK-NEXT: [[TMP7:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP6]], ptr addrspace(1) [[TMP7]], align 4
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_sum_i32(global int* out, int src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_sum_i32(src0, src1, true);
+ *out = __builtin_amdgcn_exclusive_scan_sum_i32(src0, src1, false);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_sum_u32(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 [[TMP0]], i32 [[TMP1]], i1 true)
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[TMP3]], align 4
+// CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP5:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP6:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 [[TMP4]], i32 [[TMP5]], i1 false)
+// CHECK-NEXT: [[TMP7:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP6]], ptr addrspace(1) [[TMP7]], align 4
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_sum_u32(global uint* out, uint src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_sum_u32(src0, src1, true);
+ *out = __builtin_amdgcn_exclusive_scan_sum_u32(src0, src1, false);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_xor_b32(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.xor.b32(i32 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[TMP3]], align 4
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_xor_b32(global uint* out, uint src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_xor_b32(src0, src1);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_or_b32(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.or.b32(i32 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[TMP3]], align 4
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_or_b32(global uint* out, uint src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_or_b32(src0, src1);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_and_b32(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.and.b32(i32 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[TMP3]], align 4
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_and_b32(global uint* out, uint src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_and_b32(src0, src1);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_min_i16(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i16, align 2, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i16 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 2
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i16, ptr addrspace(5) [[SRC0_ADDR]], align 2
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i16 @llvm.amdgcn.exclusive.scan.min.i16(i16 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i16 [[TMP2]], ptr addrspace(1) [[TMP3]], align 2
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_min_i16(global short* out, short src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_min_i16(src0, src1);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_min_u16(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i16, align 2, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i16 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 2
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i16, ptr addrspace(5) [[SRC0_ADDR]], align 2
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i16 @llvm.amdgcn.exclusive.scan.min.u16(i16 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i16 [[TMP2]], ptr addrspace(1) [[TMP3]], align 2
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_min_u16(global ushort* out, ushort src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_min_u16(src0, src1);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_min_i32(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.min.i32(i32 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[TMP3]], align 4
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_min_i32(global int* out, int src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_min_i32(src0, src1);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_min_u32(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.min.u32(i32 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[TMP3]], align 4
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_min_u32(global uint* out, uint src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_min_u32(src0, src1);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_max_i16(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i16, align 2, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i16 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 2
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i16, ptr addrspace(5) [[SRC0_ADDR]], align 2
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i16 @llvm.amdgcn.exclusive.scan.max.i16(i16 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i16 [[TMP2]], ptr addrspace(1) [[TMP3]], align 2
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_max_i16(global short* out, short src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_max_i16(src0, src1);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_max_u16(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i16, align 2, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i16 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 2
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i16, ptr addrspace(5) [[SRC0_ADDR]], align 2
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i16 @llvm.amdgcn.exclusive.scan.max.u16(i16 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i16 [[TMP2]], ptr addrspace(1) [[TMP3]], align 2
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_max_u16(global ushort* out, ushort src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_max_u16(src0, src1);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_max_i32(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.max.i32(i32 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[TMP3]], align 4
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_max_i32(global int* out, int src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_max_i32(src0, src1);
+}
+
+// CHECK-LABEL: @test_exclusive_scan_max_u32(
+// CHECK-NEXT: entry:
+// CHECK-NEXT: [[OUT_ADDR:%.*]] = alloca ptr addrspace(1), align 8, addrspace(5)
+// CHECK-NEXT: [[SRC0_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: [[SRC1_ADDR:%.*]] = alloca i32, align 4, addrspace(5)
+// CHECK-NEXT: store ptr addrspace(1) [[OUT:%.*]], ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[SRC0:%.*]], ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: store i32 [[SRC1:%.*]], ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr addrspace(5) [[SRC0_ADDR]], align 4
+// CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr addrspace(5) [[SRC1_ADDR]], align 4
+// CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.exclusive.scan.max.u32(i32 [[TMP0]], i32 [[TMP1]])
+// CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(1), ptr addrspace(5) [[OUT_ADDR]], align 8
+// CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[TMP3]], align 4
+// CHECK-NEXT: ret void
+//
+void test_exclusive_scan_max_u32(global uint* out, uint src0, uint src1) {
+ *out = __builtin_amdgcn_exclusive_scan_max_u32(src0, src1);
+}
diff --git a/clang/test/SemaOpenCL/builtins-amdgcn-gfx13-exclusive-scan-types.cl b/clang/test/SemaOpenCL/builtins-amdgcn-gfx13-exclusive-scan-types.cl
new file mode 100644
index 00000000000000..403689e963392e
--- /dev/null
+++ b/clang/test/SemaOpenCL/builtins-amdgcn-gfx13-exclusive-scan-types.cl
@@ -0,0 +1,42 @@
+// REQUIRES: amdgpu-registered-target
+
+// RUN: %clang_cc1 -cl-std=CL2.0 -triple amdgpu13.10-amd-amdhsa -Wsign-conversion -verify -fsyntax-only %s
+
+typedef unsigned int uint;
+typedef unsigned short ushort;
+
+void test_exclusive_scan_sum_u32_types(int *out_i32, uint *out_u32, int src_i32, uint src_u32) {
+ // unsigned return assigned to signed should warn
+ *out_i32 = __builtin_amdgcn_exclusive_scan_sum_u32(src_u32, src_u32, true); // expected-warning {{implicit conversion changes signedness: 'unsigned int' to 'int'}}
+ // signed args passed to unsigned params should warn
+ *out_u32 = __builtin_amdgcn_exclusive_scan_sum_u32(src_i32, src_u32, true); // expected-warning {{implicit conversion changes signedness: 'int' to 'unsigned int'}}
+ *out_u32 = __builtin_amdgcn_exclusive_scan_sum_u32(src_u32, src_i32, true); // expected-warning {{implicit conversion changes signedness: 'int' to 'unsigned int'}}
+ // correct usage: no warnings
+ *out_u32 = __builtin_amdgcn_exclusive_scan_sum_u32(src_u32, src_u32, true);
+}
+
+void test_exclusive_scan_min_u32_types(int *out_i32, uint *out_u32, int src_i32, uint src_u32) {
+ *out_i32 = __builtin_amdgcn_exclusive_scan_min_u32(src_u32, src_u32); // expected-warning {{implicit conversion changes signedness: 'unsigned int' to 'int'}}
+ *out_u32 = __builtin_amdgcn_exclusive_scan_min_u32(src_i32, src_u32); // expected-warning {{implicit conversion changes signedness: 'int' to 'unsigned int'}}
+ *out_u32 = __builtin_amdgcn_exclusive_scan_min_u32(src_u32, src_i32); // expected-warning {{implicit conversion changes signedness: 'int' to 'unsigned int'}}
+ *out_u32 = __builtin_amdgcn_exclusive_scan_min_u32(src_u32, src_u32);
+}
+
+void test_exclusive_scan_max_u32_types(int *out_i32, uint *out_u32, int src_i32, uint src_u32) {
+ *out_i32 = __builtin_amdgcn_exclusive_scan_max_u32(src_u32, src_u32); // expected-warning {{implicit conversion changes signedness: 'unsigned int' to 'int'}}
+ *out_u32 = __builtin_amdgcn_exclusive_scan_max_u32(src_i32, src_u32); // expected-warning {{implicit conversion changes signedness: 'int' to 'unsigned int'}}
+ *out_u32 = __builtin_amdgcn_exclusive_scan_max_u32(src_u32, src_i32); // expected-warning {{implicit conversion changes signedness: 'int' to 'unsigned int'}}
+ *out_u32 = __builtin_amdgcn_exclusive_scan_max_u32(src_u32, src_u32);
+}
+
+void test_exclusive_scan_min_u16_types(short *out_i16, ushort *out_u16, short src_i16, ushort src_u16) {
+ *out_i16 = __builtin_amdgcn_exclusive_scan_min_u16(src_u16, 0); // expected-warning {{implicit conversion changes signedness: 'unsigned short' to 'short'}}
+ *out_u16 = __builtin_amdgcn_exclusive_scan_min_u16(src_i16, 0); // expected-warning {{implicit conversion changes signedness: 'short' to 'unsigned short'}}
+ *out_u16 = __builtin_amdgcn_exclusive_scan_min_u16(src_u16, 0);
+}
+
+void test_exclusive_scan_max_u16_types(short *out_i16, ushort *out_u16, short src_i16, ushort src_u16) {
+ *out_i16 = __builtin_amdgcn_exclusive_scan_max_u16(src_u16, 0); // expected-warning {{implicit conversion changes signedness: 'unsigned short' to 'short'}}
+ *out_u16 = __builtin_amdgcn_exclusive_scan_max_u16(src_i16, 0); // expected-warning {{implicit conversion changes signedness: 'short' to 'unsigned short'}}
+ *out_u16 = __builtin_amdgcn_exclusive_scan_max_u16(src_u16, 0);
+}
diff --git a/llvm/include/llvm/IR/IntrinsicsAMDGPU.td b/llvm/include/llvm/IR/IntrinsicsAMDGPU.td
index e7829eb29ba34f..309c7ce7dd91e9 100644
--- a/llvm/include/llvm/IR/IntrinsicsAMDGPU.td
+++ b/llvm/include/llvm/IR/IntrinsicsAMDGPU.td
@@ -908,6 +908,73 @@ def int_amdgcn_prng_b32 : DefaultAttrsIntrinsic<
[llvm_i32_ty], [llvm_i32_ty], [IntrNoMem, IntrNoCreateUndefOrPoison]
>, ClangBuiltin<"__builtin_amdgcn_prng_b32">;
+let TargetFeatures = "exclusive-scan-insts" in {
+// i32 @llvm.amdgcn.exclusive.scan.sum.i32 <src0> <src1> <clamp>
+def int_amdgcn_exclusive_scan_sum_i32 :
+ DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty, llvm_i1_ty], [IntrNoMem, IntrConvergent, ImmArg<ArgIndex<2>>]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_sum_i32">;
+
+// i32 @llvm.amdgcn.exclusive.scan.sum.u32 <src0> <src1> <clamp>
+def int_amdgcn_exclusive_scan_sum_u32 :
+ DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty, llvm_i1_ty], [IntrNoMem, IntrConvergent, ImmArg<ArgIndex<2>>]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_sum_u32">;
+
+// i32 @llvm.amdgcn.exclusive.scan.xor.b32 <src0> <src1>
+def int_amdgcn_exclusive_scan_xor_b32 :
+ DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_xor_b32">;
+
+// i32 @llvm.amdgcn.exclusive.scan.or.b32 <src0> <src1>
+def int_amdgcn_exclusive_scan_or_b32 :
+ DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_or_b32">;
+
+// i32 @llvm.amdgcn.exclusive.scan.and.b32 <src0> <src1>
+def int_amdgcn_exclusive_scan_and_b32 :
+ DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_and_b32">;
+
+// i16 @llvm.amdgcn.exclusive.scan.min.i16 <src0> <src1>
+def int_amdgcn_exclusive_scan_min_i16 :
+ DefaultAttrsIntrinsic<[llvm_i16_ty], [llvm_i16_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_min_i16">;
+
+// i16 @llvm.amdgcn.exclusive.scan.min.u16 <src0> <src1>
+def int_amdgcn_exclusive_scan_min_u16 :
+ DefaultAttrsIntrinsic<[llvm_i16_ty], [llvm_i16_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_min_u16">;
+
+// i32 @llvm.amdgcn.exclusive.scan.min.i32 <src0> <src1>
+def int_amdgcn_exclusive_scan_min_i32 :
+ DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_min_i32">;
+
+// i32 @llvm.amdgcn.exclusive.scan.min.u32 <src0> <src1>
+def int_amdgcn_exclusive_scan_min_u32 :
+ DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_min_u32">;
+
+// i16 @llvm.amdgcn.exclusive.scan.max.i16 <src0> <src1>
+def int_amdgcn_exclusive_scan_max_i16 :
+ DefaultAttrsIntrinsic<[llvm_i16_ty], [llvm_i16_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_max_i16">;
+
+// i16 @llvm.amdgcn.exclusive.scan.max.u16 <src0> <src1>
+def int_amdgcn_exclusive_scan_max_u16 :
+ DefaultAttrsIntrinsic<[llvm_i16_ty], [llvm_i16_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_max_u16">;
+
+// i32 @llvm.amdgcn.exclusive.scan.max.i32 <src0> <src1>
+def int_amdgcn_exclusive_scan_max_i32 :
+ DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_max_i32">;
+
+// i32 @llvm.amdgcn.exclusive.scan.max.u32 <src0> <src1>
+def int_amdgcn_exclusive_scan_max_u32 :
+ DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty], [IntrNoMem, IntrConvergent]>,
+ ClangBuiltin<"__builtin_amdgcn_exclusive_scan_max_u32">;
+} // TargetFeatures = "exclusive-scan-insts"
+
def int_amdgcn_bitop3 :
PureIntrinsic<[llvm_anyint_ty],
[LLVMMatchType<0>, LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty],
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index 5fbf30d55947cf..25c99cbb043896 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -3336,7 +3336,7 @@ def AMDGPUFrontendVisibleFeatures {
FeatureDot13Insts, FeatureDot1Insts, FeatureDot2Insts,
FeatureDot3Insts, FeatureDot4Insts, FeatureDot5Insts,
FeatureDot6Insts, FeatureDot7Insts, FeatureDot8Insts,
- FeatureDot9Insts, FeatureExtendedImageInsts, FeatureF16BF16ToFP6BF6ConversionScaleInsts,
+ FeatureDot9Insts, FeatureExclusiveScanInsts, FeatureExtendedImageInsts, FeatureF16BF16ToFP6BF6ConversionScaleInsts,
FeatureF32ToF16BF16ConversionSRInsts, FeatureF32ToFP6BF6ConversionScaleInsts, FeatureFP4ConversionScaleInsts,
FeatureFP6BF6ConversionScaleInsts, FeatureFP8ConversionInsts, FeatureFP8ConversionScaleInsts,
FeatureBlock16ConversionScaleInsts, FeatureFP8E5M3Insts, FeatureFP8Insts,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
index 1d92bca3e12abd..d4c2280230107b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
@@ -2094,6 +2094,27 @@ RegBankLegalizeRules::RegBankLegalizeRules(const GCNSubtarget &_ST,
.Any({{UniS32}, {{UniInVgprS32}, {IntrId, Vgpr32}}})
.Any({{DivS32}, {{Vgpr32}, {IntrId, Vgpr32}}});
+ addRulesForIOpcs(
+ {amdgcn_exclusive_scan_sum_i32, amdgcn_exclusive_scan_sum_u32}, Standard)
+ .Uni(S32, {{UniInVgprS32}, {IntrId, Vgpr32, Vgpr32, Imm}})
+ .Div(S32, {{Vgpr32}, {IntrId, Vgpr32, Vgpr32, Imm}});
+
+ addRulesForIOpcs(
+ {amdgcn_exclusive_scan_xor_b32, amdgcn_exclusive_scan_or_b32,
+ amdgcn_exclusive_scan_and_b32, amdgcn_exclusive_scan_min_i32,
+ amdgcn_exclusive_scan_min_u32, amdgcn_exclusive_scan_max_i32,
+ amdgcn_exclusive_scan_max_u32},
+ Standard)
+ .Uni(S32, {{UniInVgprS32}, {IntrId, Vgpr32, Vgpr32}})
+ .Div(S32, {{Vgpr32}, {IntrId, Vgpr32, Vgpr32}});
+
+ addRulesForIOpcs(
+ {amdgcn_exclusive_scan_min_i16, amdgcn_exclusive_scan_min_u16,
+ amdgcn_exclusive_scan_max_i16, amdgcn_exclusive_scan_max_u16},
+ Standard)
+ .Uni(S16, {{UniInVgprS16}, {IntrId, Vgpr16, Vgpr32}})
+ .Div(S16, {{Vgpr16}, {IntrId, Vgpr16, Vgpr32}});
+
addRulesForIOpcs({amdgcn_sffbh}, Standard)
.Uni(S32, {{Sgpr32}, {IntrId, Sgpr32}})
.Div(S32, {{Vgpr32}, {IntrId, Vgpr32}});
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
index 91bd0d006cc649..f58e3032d45faf 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
@@ -4727,6 +4727,19 @@ AMDGPURegisterBankInfo::getInstrMapping(const MachineInstr &MI) const {
case Intrinsic::amdgcn_alignbyte:
case Intrinsic::amdgcn_perm:
case Intrinsic::amdgcn_prng_b32:
+ case Intrinsic::amdgcn_exclusive_scan_sum_i32:
+ case Intrinsic::amdgcn_exclusive_scan_sum_u32:
+ case Intrinsic::amdgcn_exclusive_scan_xor_b32:
+ case Intrinsic::amdgcn_exclusive_scan_or_b32:
+ case Intrinsic::amdgcn_exclusive_scan_and_b32:
+ case Intrinsic::amdgcn_exclusive_scan_min_i16:
+ case Intrinsic::amdgcn_exclusive_scan_min_u16:
+ case Intrinsic::amdgcn_exclusive_scan_min_i32:
+ case Intrinsic::amdgcn_exclusive_scan_min_u32:
+ case Intrinsic::amdgcn_exclusive_scan_max_i16:
+ case Intrinsic::amdgcn_exclusive_scan_max_u16:
+ case Intrinsic::amdgcn_exclusive_scan_max_i32:
+ case Intrinsic::amdgcn_exclusive_scan_max_u32:
case Intrinsic::amdgcn_fdot2:
case Intrinsic::amdgcn_sdot2:
case Intrinsic::amdgcn_udot2:
diff --git a/llvm/lib/Target/AMDGPU/VOP3Instructions.td b/llvm/lib/Target/AMDGPU/VOP3Instructions.td
index 5e78ba730c20e3..8ad1b767a24463 100644
--- a/llvm/lib/Target/AMDGPU/VOP3Instructions.td
+++ b/llvm/lib/Target/AMDGPU/VOP3Instructions.td
@@ -2136,32 +2136,34 @@ class VOP_EXCLUSIVE_SCAN_fake16<VOPProfile p, VOP3Features f = VOP3_OPSEL_ONLY>
}
// Derives the asm name from the record name.
-multiclass V_EXCLUSIVE_SCAN<VOPProfile p, VOP3Features f = VOP3_REGULAR> {
- defm NAME : VOP3Inst<!tolower(NAME), VOP_EXCLUSIVE_SCAN<p, f>>;
+multiclass V_EXCLUSIVE_SCAN<VOPProfile p, SDPatternOperator node,
+ VOP3Features f = VOP3_REGULAR> {
+ defm NAME : VOP3Inst<!tolower(NAME), VOP_EXCLUSIVE_SCAN<p, f>, node>;
}
// Generates _t16 and _fake16 variants with correctly suffixed asm/PseudoInstr names.
-multiclass V_EXCLUSIVE_SCAN_t16<VOPProfile p, VOP3Features f = VOP3_OPSEL_ONLY> {
+multiclass V_EXCLUSIVE_SCAN_t16<VOPProfile p, SDPatternOperator node,
+ VOP3Features f = VOP3_OPSEL_ONLY> {
let True16Predicate = UseRealTrue16Insts in
- defm _t16 : VOP3Inst<!tolower(NAME#"_t16"), VOP_EXCLUSIVE_SCAN_t16<p, f>>;
+ defm _t16 : VOP3Inst<!tolower(NAME#"_t16"), VOP_EXCLUSIVE_SCAN_t16<p, f>, node>;
let True16Predicate = UseFakeTrue16Insts in
- defm _fake16 : VOP3Inst<!tolower(NAME#"_fake16"), VOP_EXCLUSIVE_SCAN_fake16<p, f>>;
+ defm _fake16 : VOP3Inst<!tolower(NAME#"_fake16"), VOP_EXCLUSIVE_SCAN_fake16<p, f>, node>;
}
let isConvergent = 1, SubtargetPredicate = HasExclusiveScanInsts in {
- defm V_EXCLUSIVE_SCAN_SUM_I32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, VOP3_CLAMP>;
- defm V_EXCLUSIVE_SCAN_SUM_U32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, VOP3_CLAMP>;
- defm V_EXCLUSIVE_SCAN_XOR_B32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32>;
- defm V_EXCLUSIVE_SCAN_OR_B32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32>;
- defm V_EXCLUSIVE_SCAN_AND_B32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32>;
- defm V_EXCLUSIVE_SCAN_MIN_I32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32>;
- defm V_EXCLUSIVE_SCAN_MIN_I16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32>;
- defm V_EXCLUSIVE_SCAN_MAX_I32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32>;
- defm V_EXCLUSIVE_SCAN_MAX_I16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32>;
- defm V_EXCLUSIVE_SCAN_MIN_U32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32>;
- defm V_EXCLUSIVE_SCAN_MIN_U16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32>;
- defm V_EXCLUSIVE_SCAN_MAX_U32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32>;
- defm V_EXCLUSIVE_SCAN_MAX_U16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32>;
+ defm V_EXCLUSIVE_SCAN_SUM_I32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_sum_i32, VOP3_CLAMP>;
+ defm V_EXCLUSIVE_SCAN_SUM_U32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_sum_u32, VOP3_CLAMP>;
+ defm V_EXCLUSIVE_SCAN_XOR_B32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_xor_b32>;
+ defm V_EXCLUSIVE_SCAN_OR_B32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_or_b32>;
+ defm V_EXCLUSIVE_SCAN_AND_B32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_and_b32>;
+ defm V_EXCLUSIVE_SCAN_MIN_I32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_min_i32>;
+ defm V_EXCLUSIVE_SCAN_MIN_I16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32, int_amdgcn_exclusive_scan_min_i16>;
+ defm V_EXCLUSIVE_SCAN_MAX_I32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_max_i32>;
+ defm V_EXCLUSIVE_SCAN_MAX_I16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32, int_amdgcn_exclusive_scan_max_i16>;
+ defm V_EXCLUSIVE_SCAN_MIN_U32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_min_u32>;
+ defm V_EXCLUSIVE_SCAN_MIN_U16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32, int_amdgcn_exclusive_scan_min_u16>;
+ defm V_EXCLUSIVE_SCAN_MAX_U32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_max_u32>;
+ defm V_EXCLUSIVE_SCAN_MAX_U16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32, int_amdgcn_exclusive_scan_max_u16>;
}
//===----------------------------------------------------------------------===//
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exclusive.scan.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exclusive.scan.ll
new file mode 100644
index 00000000000000..8fecf01ff9e09d
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exclusive.scan.ll
@@ -0,0 +1,1594 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -global-isel=0 -mtriple=amdgpu13.10 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-SDAG,GFX13-TRUE16,GFX13-SDAG-TRUE16 %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu13.10 -mattr=-real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-SDAG,GFX13-SDAG-FAKE16 %s
+; RUN: llc -global-isel=1 -mtriple=amdgpu13.10 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-GISEL,GFX13-TRUE16,GFX13-GISEL-TRUE16 %s
+; RUN: llc -global-isel=1 -mtriple=amdgpu13.10 -mattr=-real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-GISEL,GFX13-GISEL-FAKE16 %s
+
+define amdgpu_kernel void @v_exclusive_scan_sum_i32_sgpr(ptr addrspace(1) %out, i32 %src0, i32 %src1) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_sum_i32_sgpr:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_i32 v0, s2, s3 clamp
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_i32 v1, s2, s3
+; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_add_nc_u32 v0, v0, v1
+; GFX13-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_sum_i32_sgpr:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_i32 v0, s2, s3 clamp
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_i32 v1, s2, s3
+; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s2, v0
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_add_co_i32 s2, s2, s3
+; GFX13-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 %src0, i32 %src1, i1 true)
+ %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 %src0, i32 %src1, i1 false)
+ %v = add i32 %vc, %vnc
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_sum_i32_vgpr(i32 addrspace(1)* %out) {
+; GFX13-LABEL: v_exclusive_scan_sum_i32_vgpr:
+; GFX13: ; %bb.0:
+; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT: v_exclusive_scan_sum_i32 v2, v1, v0 clamp
+; GFX13-NEXT: v_exclusive_scan_sum_i32 v0, v1, v0
+; GFX13-NEXT: v_add_nc_u32_e32 v0, v2, v0
+; GFX13-NEXT: s_wait_kmcnt 0x0
+; GFX13-NEXT: global_store_b32 v1, v0, s[0:1] scale_offset
+; GFX13-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 %tidx, i32 %tidy, i1 true)
+ %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 %tidx, i32 %tidy, i1 false)
+ %out_ptr = getelementptr i32, i32 addrspace(1)* %out, i32 %tidx
+ %v = add i32 %vc, %vnc
+ store i32 %v, i32 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_sum_i32_constant(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_sum_i32_constant:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_i32 v0, 7, 5 clamp
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_i32 v1, 7, 5
+; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_add_nc_u32 v0, v0, v1
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_sum_i32_constant:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_i32 v0, 7, 5 clamp
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_i32 v1, 7, 5
+; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s2, v0
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_add_co_i32 s2, s2, s3
+; GFX13-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 7, i32 5, i1 true)
+ %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 7, i32 5, i1 false)
+ %v = add i32 %vc, %vnc
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_sum_i32_undef(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_sum_i32_undef:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_i32 v0, s0, s0 clamp
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_i32 v1, s0, s0
+; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_add_nc_u32 v0, v0, v1
+; GFX13-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_sum_i32_undef:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_i32 v0, s0, s0 clamp
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_i32 v1, s0, s0
+; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s2, v0
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_add_co_i32 s2, s2, s3
+; GFX13-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 undef, i32 undef, i1 true)
+ %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 undef, i32 undef, i1 false)
+ %v = add i32 %vc, %vnc
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_sum_u32_undef(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_sum_u32_undef:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_u32 v0, s0, s0 clamp
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_u32 v1, s0, s0
+; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_add_nc_u32 v0, v0, v1
+; GFX13-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_sum_u32_undef:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_u32 v0, s0, s0 clamp
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_u32 v1, s0, s0
+; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s2, v0
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_add_co_i32 s2, s2, s3
+; GFX13-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 undef, i32 undef, i1 true)
+ %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 undef, i32 undef, i1 false)
+ %v = add i32 %vc, %vnc
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_sum_u32_sgpr(ptr addrspace(1) %out, i32 %src0, i32 %src1) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_sum_u32_sgpr:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_u32 v0, s2, s3 clamp
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_u32 v1, s2, s3
+; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_add_nc_u32 v0, v0, v1
+; GFX13-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_sum_u32_sgpr:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_u32 v0, s2, s3 clamp
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_u32 v1, s2, s3
+; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s2, v0
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_add_co_i32 s2, s2, s3
+; GFX13-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 %src0, i32 %src1, i1 true)
+ %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 %src0, i32 %src1, i1 false)
+ %v = add i32 %vc, %vnc
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_sum_u32_vgpr(i32 addrspace(1)* %out) {
+; GFX13-LABEL: v_exclusive_scan_sum_u32_vgpr:
+; GFX13: ; %bb.0:
+; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT: v_exclusive_scan_sum_u32 v2, v1, v0 clamp
+; GFX13-NEXT: v_exclusive_scan_sum_u32 v0, v1, v0
+; GFX13-NEXT: v_add_nc_u32_e32 v0, v2, v0
+; GFX13-NEXT: s_wait_kmcnt 0x0
+; GFX13-NEXT: global_store_b32 v1, v0, s[0:1] scale_offset
+; GFX13-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 %tidx, i32 %tidy, i1 true)
+ %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 %tidx, i32 %tidy, i1 false)
+ %out_ptr = getelementptr i32, i32 addrspace(1)* %out, i32 %tidx
+ %v = add i32 %vc, %vnc
+ store i32 %v, i32 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_sum_u32_constant(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_sum_u32_constant:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_u32 v0, 7, 5 clamp
+; GFX13-SDAG-NEXT: v_exclusive_scan_sum_u32 v1, 7, 5
+; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_add_nc_u32 v0, v0, v1
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_sum_u32_constant:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_u32 v0, 7, 5 clamp
+; GFX13-GISEL-NEXT: v_exclusive_scan_sum_u32 v1, 7, 5
+; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s2, v0
+; GFX13-GISEL-NEXT: v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_add_co_i32 s2, s2, s3
+; GFX13-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 7, i32 5, i1 true)
+ %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 7, i32 5, i1 false)
+ %v = add i32 %vc, %vnc
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_xor_b32_sgpr(ptr addrspace(1) %out, i32 %src0, i32 %src1) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_xor_b32_sgpr:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_xor_b32 v1, s2, s3
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_xor_b32_sgpr:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_xor_b32 v0, s2, s3
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.xor.b32(i32 %src0, i32 %src1)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_xor_b32_vgpr(i32 addrspace(1)* %out) {
+; GFX13-LABEL: v_exclusive_scan_xor_b32_vgpr:
+; GFX13: ; %bb.0:
+; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-NEXT: v_exclusive_scan_xor_b32 v0, v1, v0
+; GFX13-NEXT: s_wait_kmcnt 0x0
+; GFX13-NEXT: global_store_b32 v1, v0, s[0:1] scale_offset
+; GFX13-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %v = call i32 @llvm.amdgcn.exclusive.scan.xor.b32(i32 %tidx, i32 %tidy)
+ %out_ptr = getelementptr i32, i32 addrspace(1)* %out, i32 %tidx
+ store i32 %v, i32 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_xor_b32_constant(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_xor_b32_constant:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: v_exclusive_scan_xor_b32 v1, 7, 5
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_xor_b32_constant:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_exclusive_scan_xor_b32 v0, 7, 5
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.xor.b32(i32 7, i32 5)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_xor_b32_undef(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_xor_b32_undef:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_xor_b32 v1, s0, s0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_xor_b32_undef:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_xor_b32 v0, s0, s0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.xor.b32(i32 undef, i32 undef)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_or_b32_sgpr(ptr addrspace(1) %out, i32 %src0, i32 %src1) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_or_b32_sgpr:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_or_b32 v1, s2, s3
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_or_b32_sgpr:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_or_b32 v0, s2, s3
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.or.b32(i32 %src0, i32 %src1)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_or_b32_vgpr(i32 addrspace(1)* %out) {
+; GFX13-LABEL: v_exclusive_scan_or_b32_vgpr:
+; GFX13: ; %bb.0:
+; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-NEXT: v_exclusive_scan_or_b32 v0, v1, v0
+; GFX13-NEXT: s_wait_kmcnt 0x0
+; GFX13-NEXT: global_store_b32 v1, v0, s[0:1] scale_offset
+; GFX13-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %v = call i32 @llvm.amdgcn.exclusive.scan.or.b32(i32 %tidx, i32 %tidy)
+ %out_ptr = getelementptr i32, i32 addrspace(1)* %out, i32 %tidx
+ store i32 %v, i32 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_or_b32_constant(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_or_b32_constant:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: v_exclusive_scan_or_b32 v1, 7, 5
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_or_b32_constant:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_exclusive_scan_or_b32 v0, 7, 5
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.or.b32(i32 7, i32 5)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_or_b32_undef(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_or_b32_undef:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_or_b32 v1, s0, s0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_or_b32_undef:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_or_b32 v0, s0, s0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.or.b32(i32 undef, i32 undef)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_and_b32_sgpr(ptr addrspace(1) %out, i32 %src0, i32 %src1) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_and_b32_sgpr:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_and_b32 v1, s2, s3
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_and_b32_sgpr:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_and_b32 v0, s2, s3
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.and.b32(i32 %src0, i32 %src1)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_and_b32_vgpr(i32 addrspace(1)* %out) {
+; GFX13-LABEL: v_exclusive_scan_and_b32_vgpr:
+; GFX13: ; %bb.0:
+; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-NEXT: v_exclusive_scan_and_b32 v0, v1, v0
+; GFX13-NEXT: s_wait_kmcnt 0x0
+; GFX13-NEXT: global_store_b32 v1, v0, s[0:1] scale_offset
+; GFX13-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %v = call i32 @llvm.amdgcn.exclusive.scan.and.b32(i32 %tidx, i32 %tidy)
+ %out_ptr = getelementptr i32, i32 addrspace(1)* %out, i32 %tidx
+ store i32 %v, i32 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_and_b32_constant(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_and_b32_constant:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: v_exclusive_scan_and_b32 v1, 7, 5
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_and_b32_constant:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_exclusive_scan_and_b32 v0, 7, 5
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.and.b32(i32 7, i32 5)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_and_b32_undef(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_and_b32_undef:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_and_b32 v1, s0, s0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_and_b32_undef:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_and_b32 v0, s0, s0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.and.b32(i32 undef, i32 undef)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_i16_sgpr(ptr addrspace(1) %out, i16 %src0, i32 %src1) {
+;
+;
+; GFX13-TRUE16-LABEL: v_exclusive_scan_min_i16_sgpr:
+; GFX13-TRUE16: ; %bb.0:
+; GFX13-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-TRUE16-NEXT: v_exclusive_scan_min_i16 v0.l, s2, s3
+; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_i16_sgpr:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_min_i16 v1, s2, s3
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_i16_sgpr:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_min_i16 v0, s2, s3
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.min.i16(i16 %src0, i32 %src1)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_i16_vgpr(i16 addrspace(1)* %out) {
+;
+;
+;
+; GFX13-SDAG-TRUE16-LABEL: v_exclusive_scan_min_i16_vgpr:
+; GFX13-SDAG-TRUE16: ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-TRUE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-SDAG-TRUE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-TRUE16-NEXT: v_exclusive_scan_min_i16 v0.l, v1.l, v0
+; GFX13-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1] scale_offset
+; GFX13-SDAG-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_i16_vgpr:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-SDAG-FAKE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_min_i16 v0, v1, v0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1] scale_offset
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-TRUE16-LABEL: v_exclusive_scan_min_i16_vgpr:
+; GFX13-GISEL-TRUE16: ; %bb.0:
+; GFX13-GISEL-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-TRUE16-NEXT: v_and_b16 v1.l, 0x3ff, v0.l
+; GFX13-GISEL-TRUE16-NEXT: v_bfe_u32 v2, v0, 10, 10
+; GFX13-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-TRUE16-NEXT: v_lshlrev_b16 v0.l, 1, v1.l
+; GFX13-GISEL-TRUE16-NEXT: v_exclusive_scan_min_i16 v0.h, v1.l, v2
+; GFX13-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX13-GISEL-TRUE16-NEXT: v_cvt_u32_u16_e32 v1, v0.l
+; GFX13-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-TRUE16-NEXT: global_store_d16_hi_b16 v1, v0, s[0:1]
+; GFX13-GISEL-TRUE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_i16_vgpr:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-FAKE16-NEXT: v_lshlrev_b16 v2, 1, v1
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_min_i16 v0, v1, v0
+; GFX13-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX13-GISEL-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v2
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %tidx16 = trunc i32 %tidx to i16
+ %v = call i16 @llvm.amdgcn.exclusive.scan.min.i16(i16 %tidx16, i32 %tidy)
+ %out_ptr = getelementptr i16, i16 addrspace(1)* %out, i16 %tidx16
+ store i16 %v, i16 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_i16_constant(ptr addrspace(1) %out) {
+;
+;
+;
+; GFX13-SDAG-TRUE16-LABEL: v_exclusive_scan_min_i16_constant:
+; GFX13-SDAG-TRUE16: ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-SDAG-TRUE16-NEXT: v_exclusive_scan_min_i16 v0.l, 7, 5
+; GFX13-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-SDAG-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_i16_constant:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_min_i16 v1, 7, 5
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-TRUE16-LABEL: v_exclusive_scan_min_i16_constant:
+; GFX13-GISEL-TRUE16: ; %bb.0:
+; GFX13-GISEL-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-TRUE16-NEXT: v_exclusive_scan_min_i16 v0.l, 7, 5
+; GFX13-GISEL-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-TRUE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_i16_constant:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_min_i16 v0, 7, 5
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.min.i16(i16 7, i32 5)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_i16_undef(ptr addrspace(1) %out) {
+;
+;
+; GFX13-TRUE16-LABEL: v_exclusive_scan_min_i16_undef:
+; GFX13-TRUE16: ; %bb.0:
+; GFX13-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-TRUE16-NEXT: v_exclusive_scan_min_i16 v0.l, s0, s0
+; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_i16_undef:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_min_i16 v1, s0, s0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_i16_undef:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_min_i16 v0, s0, s0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.min.i16(i16 undef, i32 undef)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_u16_sgpr(ptr addrspace(1) %out, i16 %src0, i32 %src1) {
+;
+;
+; GFX13-TRUE16-LABEL: v_exclusive_scan_min_u16_sgpr:
+; GFX13-TRUE16: ; %bb.0:
+; GFX13-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-TRUE16-NEXT: v_exclusive_scan_min_u16 v0.l, s2, s3
+; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_u16_sgpr:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_min_u16 v1, s2, s3
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_u16_sgpr:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_min_u16 v0, s2, s3
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.min.u16(i16 %src0, i32 %src1)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_u16_vgpr(i16 addrspace(1)* %out) {
+;
+;
+;
+; GFX13-SDAG-TRUE16-LABEL: v_exclusive_scan_min_u16_vgpr:
+; GFX13-SDAG-TRUE16: ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-TRUE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-SDAG-TRUE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-TRUE16-NEXT: v_exclusive_scan_min_u16 v0.l, v1.l, v0
+; GFX13-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1] scale_offset
+; GFX13-SDAG-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_u16_vgpr:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-SDAG-FAKE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_min_u16 v0, v1, v0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1] scale_offset
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-TRUE16-LABEL: v_exclusive_scan_min_u16_vgpr:
+; GFX13-GISEL-TRUE16: ; %bb.0:
+; GFX13-GISEL-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-TRUE16-NEXT: v_and_b16 v1.l, 0x3ff, v0.l
+; GFX13-GISEL-TRUE16-NEXT: v_bfe_u32 v2, v0, 10, 10
+; GFX13-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-TRUE16-NEXT: v_lshlrev_b16 v0.l, 1, v1.l
+; GFX13-GISEL-TRUE16-NEXT: v_exclusive_scan_min_u16 v0.h, v1.l, v2
+; GFX13-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX13-GISEL-TRUE16-NEXT: v_cvt_u32_u16_e32 v1, v0.l
+; GFX13-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-TRUE16-NEXT: global_store_d16_hi_b16 v1, v0, s[0:1]
+; GFX13-GISEL-TRUE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_u16_vgpr:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-FAKE16-NEXT: v_lshlrev_b16 v2, 1, v1
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_min_u16 v0, v1, v0
+; GFX13-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX13-GISEL-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v2
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %tidx16 = trunc i32 %tidx to i16
+ %v = call i16 @llvm.amdgcn.exclusive.scan.min.u16(i16 %tidx16, i32 %tidy)
+ %out_ptr = getelementptr i16, i16 addrspace(1)* %out, i16 %tidx16
+ store i16 %v, i16 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_u16_constant(ptr addrspace(1) %out) {
+;
+;
+;
+; GFX13-SDAG-TRUE16-LABEL: v_exclusive_scan_min_u16_constant:
+; GFX13-SDAG-TRUE16: ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-SDAG-TRUE16-NEXT: v_exclusive_scan_min_u16 v0.l, 7, 5
+; GFX13-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-SDAG-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_u16_constant:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_min_u16 v1, 7, 5
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-TRUE16-LABEL: v_exclusive_scan_min_u16_constant:
+; GFX13-GISEL-TRUE16: ; %bb.0:
+; GFX13-GISEL-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-TRUE16-NEXT: v_exclusive_scan_min_u16 v0.l, 7, 5
+; GFX13-GISEL-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-TRUE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_u16_constant:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_min_u16 v0, 7, 5
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.min.u16(i16 7, i32 5)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_u16_undef(ptr addrspace(1) %out) {
+;
+;
+; GFX13-TRUE16-LABEL: v_exclusive_scan_min_u16_undef:
+; GFX13-TRUE16: ; %bb.0:
+; GFX13-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-TRUE16-NEXT: v_exclusive_scan_min_u16 v0.l, s0, s0
+; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_u16_undef:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_min_u16 v1, s0, s0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_u16_undef:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_min_u16 v0, s0, s0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.min.u16(i16 undef, i32 undef)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_i32_sgpr(ptr addrspace(1) %out, i32 %src0, i32 %src1) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_min_i32_sgpr:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_min_i32 v1, s2, s3
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_min_i32_sgpr:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_min_i32 v0, s2, s3
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.min.i32(i32 %src0, i32 %src1)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_i32_vgpr(i32 addrspace(1)* %out) {
+; GFX13-LABEL: v_exclusive_scan_min_i32_vgpr:
+; GFX13: ; %bb.0:
+; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-NEXT: v_exclusive_scan_min_i32 v0, v1, v0
+; GFX13-NEXT: s_wait_kmcnt 0x0
+; GFX13-NEXT: global_store_b32 v1, v0, s[0:1] scale_offset
+; GFX13-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %v = call i32 @llvm.amdgcn.exclusive.scan.min.i32(i32 %tidx, i32 %tidy)
+ %out_ptr = getelementptr i32, i32 addrspace(1)* %out, i32 %tidx
+ store i32 %v, i32 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_i32_constant(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_min_i32_constant:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: v_exclusive_scan_min_i32 v1, 7, 5
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_min_i32_constant:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_exclusive_scan_min_i32 v0, 7, 5
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.min.i32(i32 7, i32 5)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_i32_undef(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_min_i32_undef:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_min_i32 v1, s0, s0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_min_i32_undef:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_min_i32 v0, s0, s0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.min.i32(i32 undef, i32 undef)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_u32_sgpr(ptr addrspace(1) %out, i32 %src0, i32 %src1) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_min_u32_sgpr:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_min_u32 v1, s2, s3
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_min_u32_sgpr:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_min_u32 v0, s2, s3
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.min.u32(i32 %src0, i32 %src1)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_u32_vgpr(i32 addrspace(1)* %out) {
+; GFX13-LABEL: v_exclusive_scan_min_u32_vgpr:
+; GFX13: ; %bb.0:
+; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-NEXT: v_exclusive_scan_min_u32 v0, v1, v0
+; GFX13-NEXT: s_wait_kmcnt 0x0
+; GFX13-NEXT: global_store_b32 v1, v0, s[0:1] scale_offset
+; GFX13-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %v = call i32 @llvm.amdgcn.exclusive.scan.min.u32(i32 %tidx, i32 %tidy)
+ %out_ptr = getelementptr i32, i32 addrspace(1)* %out, i32 %tidx
+ store i32 %v, i32 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_u32_constant(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_min_u32_constant:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: v_exclusive_scan_min_u32 v1, 7, 5
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_min_u32_constant:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_exclusive_scan_min_u32 v0, 7, 5
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.min.u32(i32 7, i32 5)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_min_u32_undef(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_min_u32_undef:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_min_u32 v1, s0, s0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_min_u32_undef:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_min_u32 v0, s0, s0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.min.u32(i32 undef, i32 undef)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_i16_sgpr(ptr addrspace(1) %out, i16 %src0, i32 %src1) {
+;
+;
+; GFX13-TRUE16-LABEL: v_exclusive_scan_max_i16_sgpr:
+; GFX13-TRUE16: ; %bb.0:
+; GFX13-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-TRUE16-NEXT: v_exclusive_scan_max_i16 v0.l, s2, s3
+; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_i16_sgpr:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_max_i16 v1, s2, s3
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_i16_sgpr:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_max_i16 v0, s2, s3
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.max.i16(i16 %src0, i32 %src1)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_i16_vgpr(i16 addrspace(1)* %out) {
+;
+;
+;
+; GFX13-SDAG-TRUE16-LABEL: v_exclusive_scan_max_i16_vgpr:
+; GFX13-SDAG-TRUE16: ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-TRUE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-SDAG-TRUE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-TRUE16-NEXT: v_exclusive_scan_max_i16 v0.l, v1.l, v0
+; GFX13-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1] scale_offset
+; GFX13-SDAG-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_i16_vgpr:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-SDAG-FAKE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_max_i16 v0, v1, v0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1] scale_offset
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-TRUE16-LABEL: v_exclusive_scan_max_i16_vgpr:
+; GFX13-GISEL-TRUE16: ; %bb.0:
+; GFX13-GISEL-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-TRUE16-NEXT: v_and_b16 v1.l, 0x3ff, v0.l
+; GFX13-GISEL-TRUE16-NEXT: v_bfe_u32 v2, v0, 10, 10
+; GFX13-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-TRUE16-NEXT: v_lshlrev_b16 v0.l, 1, v1.l
+; GFX13-GISEL-TRUE16-NEXT: v_exclusive_scan_max_i16 v0.h, v1.l, v2
+; GFX13-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX13-GISEL-TRUE16-NEXT: v_cvt_u32_u16_e32 v1, v0.l
+; GFX13-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-TRUE16-NEXT: global_store_d16_hi_b16 v1, v0, s[0:1]
+; GFX13-GISEL-TRUE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_i16_vgpr:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-FAKE16-NEXT: v_lshlrev_b16 v2, 1, v1
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_max_i16 v0, v1, v0
+; GFX13-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX13-GISEL-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v2
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %tidx16 = trunc i32 %tidx to i16
+ %v = call i16 @llvm.amdgcn.exclusive.scan.max.i16(i16 %tidx16, i32 %tidy)
+ %out_ptr = getelementptr i16, i16 addrspace(1)* %out, i16 %tidx16
+ store i16 %v, i16 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_i16_constant(ptr addrspace(1) %out) {
+;
+;
+;
+; GFX13-SDAG-TRUE16-LABEL: v_exclusive_scan_max_i16_constant:
+; GFX13-SDAG-TRUE16: ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-SDAG-TRUE16-NEXT: v_exclusive_scan_max_i16 v0.l, 7, 5
+; GFX13-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-SDAG-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_i16_constant:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_max_i16 v1, 7, 5
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-TRUE16-LABEL: v_exclusive_scan_max_i16_constant:
+; GFX13-GISEL-TRUE16: ; %bb.0:
+; GFX13-GISEL-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-TRUE16-NEXT: v_exclusive_scan_max_i16 v0.l, 7, 5
+; GFX13-GISEL-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-TRUE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_i16_constant:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_max_i16 v0, 7, 5
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.max.i16(i16 7, i32 5)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_i16_undef(ptr addrspace(1) %out) {
+;
+;
+; GFX13-TRUE16-LABEL: v_exclusive_scan_max_i16_undef:
+; GFX13-TRUE16: ; %bb.0:
+; GFX13-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-TRUE16-NEXT: v_exclusive_scan_max_i16 v0.l, s0, s0
+; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_i16_undef:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_max_i16 v1, s0, s0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_i16_undef:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_max_i16 v0, s0, s0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.max.i16(i16 undef, i32 undef)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_u16_sgpr(ptr addrspace(1) %out, i16 %src0, i32 %src1) {
+;
+;
+; GFX13-TRUE16-LABEL: v_exclusive_scan_max_u16_sgpr:
+; GFX13-TRUE16: ; %bb.0:
+; GFX13-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-TRUE16-NEXT: v_exclusive_scan_max_u16 v0.l, s2, s3
+; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_u16_sgpr:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_max_u16 v1, s2, s3
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_u16_sgpr:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_max_u16 v0, s2, s3
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.max.u16(i16 %src0, i32 %src1)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_u16_vgpr(i16 addrspace(1)* %out) {
+;
+;
+;
+; GFX13-SDAG-TRUE16-LABEL: v_exclusive_scan_max_u16_vgpr:
+; GFX13-SDAG-TRUE16: ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-TRUE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-SDAG-TRUE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-TRUE16-NEXT: v_exclusive_scan_max_u16 v0.l, v1.l, v0
+; GFX13-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1] scale_offset
+; GFX13-SDAG-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_u16_vgpr:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-SDAG-FAKE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_max_u16 v0, v1, v0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1] scale_offset
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-TRUE16-LABEL: v_exclusive_scan_max_u16_vgpr:
+; GFX13-GISEL-TRUE16: ; %bb.0:
+; GFX13-GISEL-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-TRUE16-NEXT: v_and_b16 v1.l, 0x3ff, v0.l
+; GFX13-GISEL-TRUE16-NEXT: v_bfe_u32 v2, v0, 10, 10
+; GFX13-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-TRUE16-NEXT: v_lshlrev_b16 v0.l, 1, v1.l
+; GFX13-GISEL-TRUE16-NEXT: v_exclusive_scan_max_u16 v0.h, v1.l, v2
+; GFX13-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX13-GISEL-TRUE16-NEXT: v_cvt_u32_u16_e32 v1, v0.l
+; GFX13-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-TRUE16-NEXT: global_store_d16_hi_b16 v1, v0, s[0:1]
+; GFX13-GISEL-TRUE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_u16_vgpr:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-GISEL-FAKE16-NEXT: v_lshlrev_b16 v2, 1, v1
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_max_u16 v0, v1, v0
+; GFX13-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX13-GISEL-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v2
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %tidx16 = trunc i32 %tidx to i16
+ %v = call i16 @llvm.amdgcn.exclusive.scan.max.u16(i16 %tidx16, i32 %tidy)
+ %out_ptr = getelementptr i16, i16 addrspace(1)* %out, i16 %tidx16
+ store i16 %v, i16 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_u16_constant(ptr addrspace(1) %out) {
+;
+;
+;
+; GFX13-SDAG-TRUE16-LABEL: v_exclusive_scan_max_u16_constant:
+; GFX13-SDAG-TRUE16: ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-SDAG-TRUE16-NEXT: v_exclusive_scan_max_u16 v0.l, 7, 5
+; GFX13-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-SDAG-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_u16_constant:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_max_u16 v1, 7, 5
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-TRUE16-LABEL: v_exclusive_scan_max_u16_constant:
+; GFX13-GISEL-TRUE16: ; %bb.0:
+; GFX13-GISEL-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-TRUE16-NEXT: v_exclusive_scan_max_u16 v0.l, 7, 5
+; GFX13-GISEL-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-TRUE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_u16_constant:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_max_u16 v0, 7, 5
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.max.u16(i16 7, i32 5)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_u16_undef(ptr addrspace(1) %out) {
+;
+;
+; GFX13-TRUE16-LABEL: v_exclusive_scan_max_u16_undef:
+; GFX13-TRUE16: ; %bb.0:
+; GFX13-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-TRUE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-TRUE16-NEXT: v_exclusive_scan_max_u16 v0.l, s0, s0
+; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-TRUE16-NEXT: s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_u16_undef:
+; GFX13-SDAG-FAKE16: ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT: v_exclusive_scan_max_u16 v1, s0, s0
+; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT: s_endpgm
+;
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_u16_undef:
+; GFX13-GISEL-FAKE16: ; %bb.0:
+; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_max_u16 v0, s0, s0
+; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
+; GFX13-GISEL-FAKE16-NEXT: s_endpgm
+ %v = call i16 @llvm.amdgcn.exclusive.scan.max.u16(i16 undef, i32 undef)
+ store i16 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_i32_sgpr(ptr addrspace(1) %out, i32 %src0, i32 %src1) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_max_i32_sgpr:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_max_i32 v1, s2, s3
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_max_i32_sgpr:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_max_i32 v0, s2, s3
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.max.i32(i32 %src0, i32 %src1)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_i32_vgpr(i32 addrspace(1)* %out) {
+; GFX13-LABEL: v_exclusive_scan_max_i32_vgpr:
+; GFX13: ; %bb.0:
+; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-NEXT: v_exclusive_scan_max_i32 v0, v1, v0
+; GFX13-NEXT: s_wait_kmcnt 0x0
+; GFX13-NEXT: global_store_b32 v1, v0, s[0:1] scale_offset
+; GFX13-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %v = call i32 @llvm.amdgcn.exclusive.scan.max.i32(i32 %tidx, i32 %tidy)
+ %out_ptr = getelementptr i32, i32 addrspace(1)* %out, i32 %tidx
+ store i32 %v, i32 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_i32_constant(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_max_i32_constant:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: v_exclusive_scan_max_i32 v1, 7, 5
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_max_i32_constant:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_exclusive_scan_max_i32 v0, 7, 5
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.max.i32(i32 7, i32 5)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_i32_undef(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_max_i32_undef:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_max_i32 v1, s0, s0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_max_i32_undef:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_max_i32 v0, s0, s0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.max.i32(i32 undef, i32 undef)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_u32_sgpr(ptr addrspace(1) %out, i32 %src0, i32 %src1) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_max_u32_sgpr:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_max_u32 v1, s2, s3
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_max_u32_sgpr:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_max_u32 v0, s2, s3
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.max.u32(i32 %src0, i32 %src1)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_u32_vgpr(i32 addrspace(1)* %out) {
+; GFX13-LABEL: v_exclusive_scan_max_u32_vgpr:
+; GFX13: ; %bb.0:
+; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-NEXT: v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT: v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-NEXT: v_exclusive_scan_max_u32 v0, v1, v0
+; GFX13-NEXT: s_wait_kmcnt 0x0
+; GFX13-NEXT: global_store_b32 v1, v0, s[0:1] scale_offset
+; GFX13-NEXT: s_endpgm
+ %tidx = call i32 @llvm.amdgcn.workitem.id.x()
+ %tidy = call i32 @llvm.amdgcn.workitem.id.y()
+ %v = call i32 @llvm.amdgcn.exclusive.scan.max.u32(i32 %tidx, i32 %tidy)
+ %out_ptr = getelementptr i32, i32 addrspace(1)* %out, i32 %tidx
+ store i32 %v, i32 addrspace(1)* %out_ptr
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_u32_constant(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_max_u32_constant:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: v_exclusive_scan_max_u32 v1, 7, 5
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_max_u32_constant:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_exclusive_scan_max_u32 v0, 7, 5
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.max.u32(i32 7, i32 5)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @v_exclusive_scan_max_u32_undef(ptr addrspace(1) %out) {
+;
+; GFX13-SDAG-LABEL: v_exclusive_scan_max_u32_undef:
+; GFX13-SDAG: ; %bb.0:
+; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
+; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX13-SDAG-NEXT: v_exclusive_scan_max_u32 v1, s0, s0
+; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX13-SDAG-NEXT: s_endpgm
+;
+; GFX13-GISEL-LABEL: v_exclusive_scan_max_u32_undef:
+; GFX13-GISEL: ; %bb.0:
+; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX13-GISEL-NEXT: v_exclusive_scan_max_u32 v0, s0, s0
+; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX13-GISEL-NEXT: s_endpgm
+ %v = call i32 @llvm.amdgcn.exclusive.scan.max.u32(i32 undef, i32 undef)
+ store i32 %v, ptr addrspace(1) %out
+ ret void
+}
>From 59925a86c69f0ce290aa8a9b056d0da2ee24244f Mon Sep 17 00:00:00 2001
From: Mariusz Sikora <mariusz.sikora at amd.com>
Date: Mon, 28 Sep 2026 07:04:27 -0400
Subject: [PATCH 2/3] Fix undef -> poison and multiclass
---
llvm/lib/Target/AMDGPU/VOP3Instructions.td | 41 +++----
.../AMDGPU/llvm.amdgcn.exclusive.scan.ll | 116 +++++++++---------
2 files changed, 75 insertions(+), 82 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/VOP3Instructions.td b/llvm/lib/Target/AMDGPU/VOP3Instructions.td
index 8ad1b767a24463..c8350fe1badd45 100644
--- a/llvm/lib/Target/AMDGPU/VOP3Instructions.td
+++ b/llvm/lib/Target/AMDGPU/VOP3Instructions.td
@@ -2135,35 +2135,28 @@ class VOP_EXCLUSIVE_SCAN_fake16<VOPProfile p, VOP3Features f = VOP3_OPSEL_ONLY>
let HasExtVOP3DPP = 0;
}
-// Derives the asm name from the record name.
-multiclass V_EXCLUSIVE_SCAN<VOPProfile p, SDPatternOperator node,
- VOP3Features f = VOP3_REGULAR> {
- defm NAME : VOP3Inst<!tolower(NAME), VOP_EXCLUSIVE_SCAN<p, f>, node>;
-}
-
-// Generates _t16 and _fake16 variants with correctly suffixed asm/PseudoInstr names.
-multiclass V_EXCLUSIVE_SCAN_t16<VOPProfile p, SDPatternOperator node,
- VOP3Features f = VOP3_OPSEL_ONLY> {
+// Emits the _t16 and _fake16 variants (these i16 ops have no non-true16 form).
+multiclass V_EXCLUSIVE_SCAN_t16<string OpName, VOPProfile p, SDPatternOperator node> {
let True16Predicate = UseRealTrue16Insts in
- defm _t16 : VOP3Inst<!tolower(NAME#"_t16"), VOP_EXCLUSIVE_SCAN_t16<p, f>, node>;
+ defm _t16 : VOP3Inst<OpName#"_t16", VOP_EXCLUSIVE_SCAN_t16<p>, node>;
let True16Predicate = UseFakeTrue16Insts in
- defm _fake16 : VOP3Inst<!tolower(NAME#"_fake16"), VOP_EXCLUSIVE_SCAN_fake16<p, f>, node>;
+ defm _fake16 : VOP3Inst<OpName#"_fake16", VOP_EXCLUSIVE_SCAN_fake16<p>, node>;
}
let isConvergent = 1, SubtargetPredicate = HasExclusiveScanInsts in {
- defm V_EXCLUSIVE_SCAN_SUM_I32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_sum_i32, VOP3_CLAMP>;
- defm V_EXCLUSIVE_SCAN_SUM_U32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_sum_u32, VOP3_CLAMP>;
- defm V_EXCLUSIVE_SCAN_XOR_B32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_xor_b32>;
- defm V_EXCLUSIVE_SCAN_OR_B32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_or_b32>;
- defm V_EXCLUSIVE_SCAN_AND_B32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_and_b32>;
- defm V_EXCLUSIVE_SCAN_MIN_I32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_min_i32>;
- defm V_EXCLUSIVE_SCAN_MIN_I16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32, int_amdgcn_exclusive_scan_min_i16>;
- defm V_EXCLUSIVE_SCAN_MAX_I32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_max_i32>;
- defm V_EXCLUSIVE_SCAN_MAX_I16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32, int_amdgcn_exclusive_scan_max_i16>;
- defm V_EXCLUSIVE_SCAN_MIN_U32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_min_u32>;
- defm V_EXCLUSIVE_SCAN_MIN_U16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32, int_amdgcn_exclusive_scan_min_u16>;
- defm V_EXCLUSIVE_SCAN_MAX_U32 : V_EXCLUSIVE_SCAN<VOP_I32_I32_I32, int_amdgcn_exclusive_scan_max_u32>;
- defm V_EXCLUSIVE_SCAN_MAX_U16 : V_EXCLUSIVE_SCAN_t16<VOP_I16_I16_I32, int_amdgcn_exclusive_scan_max_u16>;
+ defm V_EXCLUSIVE_SCAN_SUM_I32 : VOP3Inst<"v_exclusive_scan_sum_i32", VOP_EXCLUSIVE_SCAN<VOP_I32_I32_I32, VOP3_CLAMP>, int_amdgcn_exclusive_scan_sum_i32>;
+ defm V_EXCLUSIVE_SCAN_SUM_U32 : VOP3Inst<"v_exclusive_scan_sum_u32", VOP_EXCLUSIVE_SCAN<VOP_I32_I32_I32, VOP3_CLAMP>, int_amdgcn_exclusive_scan_sum_u32>;
+ defm V_EXCLUSIVE_SCAN_XOR_B32 : VOP3Inst<"v_exclusive_scan_xor_b32", VOP_EXCLUSIVE_SCAN<VOP_I32_I32_I32>, int_amdgcn_exclusive_scan_xor_b32>;
+ defm V_EXCLUSIVE_SCAN_OR_B32 : VOP3Inst<"v_exclusive_scan_or_b32", VOP_EXCLUSIVE_SCAN<VOP_I32_I32_I32>, int_amdgcn_exclusive_scan_or_b32>;
+ defm V_EXCLUSIVE_SCAN_AND_B32 : VOP3Inst<"v_exclusive_scan_and_b32", VOP_EXCLUSIVE_SCAN<VOP_I32_I32_I32>, int_amdgcn_exclusive_scan_and_b32>;
+ defm V_EXCLUSIVE_SCAN_MIN_I32 : VOP3Inst<"v_exclusive_scan_min_i32", VOP_EXCLUSIVE_SCAN<VOP_I32_I32_I32>, int_amdgcn_exclusive_scan_min_i32>;
+ defm V_EXCLUSIVE_SCAN_MAX_I32 : VOP3Inst<"v_exclusive_scan_max_i32", VOP_EXCLUSIVE_SCAN<VOP_I32_I32_I32>, int_amdgcn_exclusive_scan_max_i32>;
+ defm V_EXCLUSIVE_SCAN_MIN_U32 : VOP3Inst<"v_exclusive_scan_min_u32", VOP_EXCLUSIVE_SCAN<VOP_I32_I32_I32>, int_amdgcn_exclusive_scan_min_u32>;
+ defm V_EXCLUSIVE_SCAN_MAX_U32 : VOP3Inst<"v_exclusive_scan_max_u32", VOP_EXCLUSIVE_SCAN<VOP_I32_I32_I32>, int_amdgcn_exclusive_scan_max_u32>;
+ defm V_EXCLUSIVE_SCAN_MIN_I16 : V_EXCLUSIVE_SCAN_t16<"v_exclusive_scan_min_i16", VOP_I16_I16_I32, int_amdgcn_exclusive_scan_min_i16>;
+ defm V_EXCLUSIVE_SCAN_MAX_I16 : V_EXCLUSIVE_SCAN_t16<"v_exclusive_scan_max_i16", VOP_I16_I16_I32, int_amdgcn_exclusive_scan_max_i16>;
+ defm V_EXCLUSIVE_SCAN_MIN_U16 : V_EXCLUSIVE_SCAN_t16<"v_exclusive_scan_min_u16", VOP_I16_I16_I32, int_amdgcn_exclusive_scan_min_u16>;
+ defm V_EXCLUSIVE_SCAN_MAX_U16 : V_EXCLUSIVE_SCAN_t16<"v_exclusive_scan_max_u16", VOP_I16_I16_I32, int_amdgcn_exclusive_scan_max_u16>;
}
//===----------------------------------------------------------------------===//
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exclusive.scan.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exclusive.scan.ll
index 8fecf01ff9e09d..ec5b5f7e518c23 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exclusive.scan.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exclusive.scan.ll
@@ -97,9 +97,9 @@ define amdgpu_kernel void @v_exclusive_scan_sum_i32_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_sum_i32_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_sum_i32_poison(ptr addrspace(1) %out) {
;
-; GFX13-SDAG-LABEL: v_exclusive_scan_sum_i32_undef:
+; GFX13-SDAG-LABEL: v_exclusive_scan_sum_i32_poison:
; GFX13-SDAG: ; %bb.0:
; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
@@ -110,7 +110,7 @@ define amdgpu_kernel void @v_exclusive_scan_sum_i32_undef(ptr addrspace(1) %out)
; GFX13-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
; GFX13-SDAG-NEXT: s_endpgm
;
-; GFX13-GISEL-LABEL: v_exclusive_scan_sum_i32_undef:
+; GFX13-GISEL-LABEL: v_exclusive_scan_sum_i32_poison:
; GFX13-GISEL: ; %bb.0:
; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
@@ -125,16 +125,16 @@ define amdgpu_kernel void @v_exclusive_scan_sum_i32_undef(ptr addrspace(1) %out)
; GFX13-GISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13-GISEL-NEXT: s_endpgm
- %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 undef, i32 undef, i1 true)
- %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 undef, i32 undef, i1 false)
+ %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 poison, i32 poison, i1 true)
+ %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.i32(i32 poison, i32 poison, i1 false)
%v = add i32 %vc, %vnc
store i32 %v, ptr addrspace(1) %out
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_sum_u32_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_sum_u32_poison(ptr addrspace(1) %out) {
;
-; GFX13-SDAG-LABEL: v_exclusive_scan_sum_u32_undef:
+; GFX13-SDAG-LABEL: v_exclusive_scan_sum_u32_poison:
; GFX13-SDAG: ; %bb.0:
; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-NEXT: s_wait_kmcnt 0x0
@@ -145,7 +145,7 @@ define amdgpu_kernel void @v_exclusive_scan_sum_u32_undef(ptr addrspace(1) %out)
; GFX13-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
; GFX13-SDAG-NEXT: s_endpgm
;
-; GFX13-GISEL-LABEL: v_exclusive_scan_sum_u32_undef:
+; GFX13-GISEL-LABEL: v_exclusive_scan_sum_u32_poison:
; GFX13-GISEL: ; %bb.0:
; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-NEXT: s_wait_kmcnt 0x0
@@ -160,8 +160,8 @@ define amdgpu_kernel void @v_exclusive_scan_sum_u32_undef(ptr addrspace(1) %out)
; GFX13-GISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13-GISEL-NEXT: s_endpgm
- %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 undef, i32 undef, i1 true)
- %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 undef, i32 undef, i1 false)
+ %vc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 poison, i32 poison, i1 true)
+ %vnc = call i32 @llvm.amdgcn.exclusive.scan.sum.u32(i32 poison, i32 poison, i1 false)
%v = add i32 %vc, %vnc
store i32 %v, ptr addrspace(1) %out
ret void
@@ -327,9 +327,9 @@ define amdgpu_kernel void @v_exclusive_scan_xor_b32_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_xor_b32_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_xor_b32_poison(ptr addrspace(1) %out) {
;
-; GFX13-SDAG-LABEL: v_exclusive_scan_xor_b32_undef:
+; GFX13-SDAG-LABEL: v_exclusive_scan_xor_b32_poison:
; GFX13-SDAG: ; %bb.0:
; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
@@ -338,7 +338,7 @@ define amdgpu_kernel void @v_exclusive_scan_xor_b32_undef(ptr addrspace(1) %out)
; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX13-SDAG-NEXT: s_endpgm
;
-; GFX13-GISEL-LABEL: v_exclusive_scan_xor_b32_undef:
+; GFX13-GISEL-LABEL: v_exclusive_scan_xor_b32_poison:
; GFX13-GISEL: ; %bb.0:
; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
@@ -346,7 +346,7 @@ define amdgpu_kernel void @v_exclusive_scan_xor_b32_undef(ptr addrspace(1) %out)
; GFX13-GISEL-NEXT: v_exclusive_scan_xor_b32 v0, s0, s0
; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13-GISEL-NEXT: s_endpgm
- %v = call i32 @llvm.amdgcn.exclusive.scan.xor.b32(i32 undef, i32 undef)
+ %v = call i32 @llvm.amdgcn.exclusive.scan.xor.b32(i32 poison, i32 poison)
store i32 %v, ptr addrspace(1) %out
ret void
}
@@ -418,9 +418,9 @@ define amdgpu_kernel void @v_exclusive_scan_or_b32_constant(ptr addrspace(1) %ou
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_or_b32_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_or_b32_poison(ptr addrspace(1) %out) {
;
-; GFX13-SDAG-LABEL: v_exclusive_scan_or_b32_undef:
+; GFX13-SDAG-LABEL: v_exclusive_scan_or_b32_poison:
; GFX13-SDAG: ; %bb.0:
; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
@@ -429,7 +429,7 @@ define amdgpu_kernel void @v_exclusive_scan_or_b32_undef(ptr addrspace(1) %out)
; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX13-SDAG-NEXT: s_endpgm
;
-; GFX13-GISEL-LABEL: v_exclusive_scan_or_b32_undef:
+; GFX13-GISEL-LABEL: v_exclusive_scan_or_b32_poison:
; GFX13-GISEL: ; %bb.0:
; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
@@ -437,7 +437,7 @@ define amdgpu_kernel void @v_exclusive_scan_or_b32_undef(ptr addrspace(1) %out)
; GFX13-GISEL-NEXT: v_exclusive_scan_or_b32 v0, s0, s0
; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13-GISEL-NEXT: s_endpgm
- %v = call i32 @llvm.amdgcn.exclusive.scan.or.b32(i32 undef, i32 undef)
+ %v = call i32 @llvm.amdgcn.exclusive.scan.or.b32(i32 poison, i32 poison)
store i32 %v, ptr addrspace(1) %out
ret void
}
@@ -509,9 +509,9 @@ define amdgpu_kernel void @v_exclusive_scan_and_b32_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_and_b32_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_and_b32_poison(ptr addrspace(1) %out) {
;
-; GFX13-SDAG-LABEL: v_exclusive_scan_and_b32_undef:
+; GFX13-SDAG-LABEL: v_exclusive_scan_and_b32_poison:
; GFX13-SDAG: ; %bb.0:
; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
@@ -520,7 +520,7 @@ define amdgpu_kernel void @v_exclusive_scan_and_b32_undef(ptr addrspace(1) %out)
; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX13-SDAG-NEXT: s_endpgm
;
-; GFX13-GISEL-LABEL: v_exclusive_scan_and_b32_undef:
+; GFX13-GISEL-LABEL: v_exclusive_scan_and_b32_poison:
; GFX13-GISEL: ; %bb.0:
; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
@@ -528,7 +528,7 @@ define amdgpu_kernel void @v_exclusive_scan_and_b32_undef(ptr addrspace(1) %out)
; GFX13-GISEL-NEXT: v_exclusive_scan_and_b32 v0, s0, s0
; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13-GISEL-NEXT: s_endpgm
- %v = call i32 @llvm.amdgcn.exclusive.scan.and.b32(i32 undef, i32 undef)
+ %v = call i32 @llvm.amdgcn.exclusive.scan.and.b32(i32 poison, i32 poison)
store i32 %v, ptr addrspace(1) %out
ret void
}
@@ -673,10 +673,10 @@ define amdgpu_kernel void @v_exclusive_scan_min_i16_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_min_i16_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_min_i16_poison(ptr addrspace(1) %out) {
;
;
-; GFX13-TRUE16-LABEL: v_exclusive_scan_min_i16_undef:
+; GFX13-TRUE16-LABEL: v_exclusive_scan_min_i16_poison:
; GFX13-TRUE16: ; %bb.0:
; GFX13-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
@@ -685,7 +685,7 @@ define amdgpu_kernel void @v_exclusive_scan_min_i16_undef(ptr addrspace(1) %out)
; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX13-TRUE16-NEXT: s_endpgm
;
-; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_i16_undef:
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_i16_poison:
; GFX13-SDAG-FAKE16: ; %bb.0:
; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
@@ -694,7 +694,7 @@ define amdgpu_kernel void @v_exclusive_scan_min_i16_undef(ptr addrspace(1) %out)
; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
; GFX13-SDAG-FAKE16-NEXT: s_endpgm
;
-; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_i16_undef:
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_i16_poison:
; GFX13-GISEL-FAKE16: ; %bb.0:
; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
@@ -702,7 +702,7 @@ define amdgpu_kernel void @v_exclusive_scan_min_i16_undef(ptr addrspace(1) %out)
; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_min_i16 v0, s0, s0
; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX13-GISEL-FAKE16-NEXT: s_endpgm
- %v = call i16 @llvm.amdgcn.exclusive.scan.min.i16(i16 undef, i32 undef)
+ %v = call i16 @llvm.amdgcn.exclusive.scan.min.i16(i16 poison, i32 poison)
store i16 %v, ptr addrspace(1) %out
ret void
}
@@ -847,10 +847,10 @@ define amdgpu_kernel void @v_exclusive_scan_min_u16_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_min_u16_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_min_u16_poison(ptr addrspace(1) %out) {
;
;
-; GFX13-TRUE16-LABEL: v_exclusive_scan_min_u16_undef:
+; GFX13-TRUE16-LABEL: v_exclusive_scan_min_u16_poison:
; GFX13-TRUE16: ; %bb.0:
; GFX13-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
@@ -859,7 +859,7 @@ define amdgpu_kernel void @v_exclusive_scan_min_u16_undef(ptr addrspace(1) %out)
; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX13-TRUE16-NEXT: s_endpgm
;
-; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_u16_undef:
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_min_u16_poison:
; GFX13-SDAG-FAKE16: ; %bb.0:
; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
@@ -868,7 +868,7 @@ define amdgpu_kernel void @v_exclusive_scan_min_u16_undef(ptr addrspace(1) %out)
; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
; GFX13-SDAG-FAKE16-NEXT: s_endpgm
;
-; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_u16_undef:
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_min_u16_poison:
; GFX13-GISEL-FAKE16: ; %bb.0:
; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
@@ -876,7 +876,7 @@ define amdgpu_kernel void @v_exclusive_scan_min_u16_undef(ptr addrspace(1) %out)
; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_min_u16 v0, s0, s0
; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX13-GISEL-FAKE16-NEXT: s_endpgm
- %v = call i16 @llvm.amdgcn.exclusive.scan.min.u16(i16 undef, i32 undef)
+ %v = call i16 @llvm.amdgcn.exclusive.scan.min.u16(i16 poison, i32 poison)
store i16 %v, ptr addrspace(1) %out
ret void
}
@@ -948,9 +948,9 @@ define amdgpu_kernel void @v_exclusive_scan_min_i32_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_min_i32_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_min_i32_poison(ptr addrspace(1) %out) {
;
-; GFX13-SDAG-LABEL: v_exclusive_scan_min_i32_undef:
+; GFX13-SDAG-LABEL: v_exclusive_scan_min_i32_poison:
; GFX13-SDAG: ; %bb.0:
; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
@@ -959,7 +959,7 @@ define amdgpu_kernel void @v_exclusive_scan_min_i32_undef(ptr addrspace(1) %out)
; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX13-SDAG-NEXT: s_endpgm
;
-; GFX13-GISEL-LABEL: v_exclusive_scan_min_i32_undef:
+; GFX13-GISEL-LABEL: v_exclusive_scan_min_i32_poison:
; GFX13-GISEL: ; %bb.0:
; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
@@ -967,7 +967,7 @@ define amdgpu_kernel void @v_exclusive_scan_min_i32_undef(ptr addrspace(1) %out)
; GFX13-GISEL-NEXT: v_exclusive_scan_min_i32 v0, s0, s0
; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13-GISEL-NEXT: s_endpgm
- %v = call i32 @llvm.amdgcn.exclusive.scan.min.i32(i32 undef, i32 undef)
+ %v = call i32 @llvm.amdgcn.exclusive.scan.min.i32(i32 poison, i32 poison)
store i32 %v, ptr addrspace(1) %out
ret void
}
@@ -1039,9 +1039,9 @@ define amdgpu_kernel void @v_exclusive_scan_min_u32_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_min_u32_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_min_u32_poison(ptr addrspace(1) %out) {
;
-; GFX13-SDAG-LABEL: v_exclusive_scan_min_u32_undef:
+; GFX13-SDAG-LABEL: v_exclusive_scan_min_u32_poison:
; GFX13-SDAG: ; %bb.0:
; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
@@ -1050,7 +1050,7 @@ define amdgpu_kernel void @v_exclusive_scan_min_u32_undef(ptr addrspace(1) %out)
; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX13-SDAG-NEXT: s_endpgm
;
-; GFX13-GISEL-LABEL: v_exclusive_scan_min_u32_undef:
+; GFX13-GISEL-LABEL: v_exclusive_scan_min_u32_poison:
; GFX13-GISEL: ; %bb.0:
; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
@@ -1058,7 +1058,7 @@ define amdgpu_kernel void @v_exclusive_scan_min_u32_undef(ptr addrspace(1) %out)
; GFX13-GISEL-NEXT: v_exclusive_scan_min_u32 v0, s0, s0
; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13-GISEL-NEXT: s_endpgm
- %v = call i32 @llvm.amdgcn.exclusive.scan.min.u32(i32 undef, i32 undef)
+ %v = call i32 @llvm.amdgcn.exclusive.scan.min.u32(i32 poison, i32 poison)
store i32 %v, ptr addrspace(1) %out
ret void
}
@@ -1203,10 +1203,10 @@ define amdgpu_kernel void @v_exclusive_scan_max_i16_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_max_i16_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_max_i16_poison(ptr addrspace(1) %out) {
;
;
-; GFX13-TRUE16-LABEL: v_exclusive_scan_max_i16_undef:
+; GFX13-TRUE16-LABEL: v_exclusive_scan_max_i16_poison:
; GFX13-TRUE16: ; %bb.0:
; GFX13-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
@@ -1215,7 +1215,7 @@ define amdgpu_kernel void @v_exclusive_scan_max_i16_undef(ptr addrspace(1) %out)
; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX13-TRUE16-NEXT: s_endpgm
;
-; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_i16_undef:
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_i16_poison:
; GFX13-SDAG-FAKE16: ; %bb.0:
; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
@@ -1224,7 +1224,7 @@ define amdgpu_kernel void @v_exclusive_scan_max_i16_undef(ptr addrspace(1) %out)
; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
; GFX13-SDAG-FAKE16-NEXT: s_endpgm
;
-; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_i16_undef:
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_i16_poison:
; GFX13-GISEL-FAKE16: ; %bb.0:
; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
@@ -1232,7 +1232,7 @@ define amdgpu_kernel void @v_exclusive_scan_max_i16_undef(ptr addrspace(1) %out)
; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_max_i16 v0, s0, s0
; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX13-GISEL-FAKE16-NEXT: s_endpgm
- %v = call i16 @llvm.amdgcn.exclusive.scan.max.i16(i16 undef, i32 undef)
+ %v = call i16 @llvm.amdgcn.exclusive.scan.max.i16(i16 poison, i32 poison)
store i16 %v, ptr addrspace(1) %out
ret void
}
@@ -1377,10 +1377,10 @@ define amdgpu_kernel void @v_exclusive_scan_max_u16_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_max_u16_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_max_u16_poison(ptr addrspace(1) %out) {
;
;
-; GFX13-TRUE16-LABEL: v_exclusive_scan_max_u16_undef:
+; GFX13-TRUE16-LABEL: v_exclusive_scan_max_u16_poison:
; GFX13-TRUE16: ; %bb.0:
; GFX13-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-TRUE16-NEXT: v_mov_b32_e32 v1, 0
@@ -1389,7 +1389,7 @@ define amdgpu_kernel void @v_exclusive_scan_max_u16_undef(ptr addrspace(1) %out)
; GFX13-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX13-TRUE16-NEXT: s_endpgm
;
-; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_u16_undef:
+; GFX13-SDAG-FAKE16-LABEL: v_exclusive_scan_max_u16_poison:
; GFX13-SDAG-FAKE16: ; %bb.0:
; GFX13-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, 0
@@ -1398,7 +1398,7 @@ define amdgpu_kernel void @v_exclusive_scan_max_u16_undef(ptr addrspace(1) %out)
; GFX13-SDAG-FAKE16-NEXT: global_store_b16 v0, v1, s[0:1]
; GFX13-SDAG-FAKE16-NEXT: s_endpgm
;
-; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_u16_undef:
+; GFX13-GISEL-FAKE16-LABEL: v_exclusive_scan_max_u16_poison:
; GFX13-GISEL-FAKE16: ; %bb.0:
; GFX13-GISEL-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-FAKE16-NEXT: v_mov_b32_e32 v1, 0
@@ -1406,7 +1406,7 @@ define amdgpu_kernel void @v_exclusive_scan_max_u16_undef(ptr addrspace(1) %out)
; GFX13-GISEL-FAKE16-NEXT: v_exclusive_scan_max_u16 v0, s0, s0
; GFX13-GISEL-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX13-GISEL-FAKE16-NEXT: s_endpgm
- %v = call i16 @llvm.amdgcn.exclusive.scan.max.u16(i16 undef, i32 undef)
+ %v = call i16 @llvm.amdgcn.exclusive.scan.max.u16(i16 poison, i32 poison)
store i16 %v, ptr addrspace(1) %out
ret void
}
@@ -1478,9 +1478,9 @@ define amdgpu_kernel void @v_exclusive_scan_max_i32_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_max_i32_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_max_i32_poison(ptr addrspace(1) %out) {
;
-; GFX13-SDAG-LABEL: v_exclusive_scan_max_i32_undef:
+; GFX13-SDAG-LABEL: v_exclusive_scan_max_i32_poison:
; GFX13-SDAG: ; %bb.0:
; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
@@ -1489,7 +1489,7 @@ define amdgpu_kernel void @v_exclusive_scan_max_i32_undef(ptr addrspace(1) %out)
; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX13-SDAG-NEXT: s_endpgm
;
-; GFX13-GISEL-LABEL: v_exclusive_scan_max_i32_undef:
+; GFX13-GISEL-LABEL: v_exclusive_scan_max_i32_poison:
; GFX13-GISEL: ; %bb.0:
; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
@@ -1497,7 +1497,7 @@ define amdgpu_kernel void @v_exclusive_scan_max_i32_undef(ptr addrspace(1) %out)
; GFX13-GISEL-NEXT: v_exclusive_scan_max_i32 v0, s0, s0
; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13-GISEL-NEXT: s_endpgm
- %v = call i32 @llvm.amdgcn.exclusive.scan.max.i32(i32 undef, i32 undef)
+ %v = call i32 @llvm.amdgcn.exclusive.scan.max.i32(i32 poison, i32 poison)
store i32 %v, ptr addrspace(1) %out
ret void
}
@@ -1569,9 +1569,9 @@ define amdgpu_kernel void @v_exclusive_scan_max_u32_constant(ptr addrspace(1) %o
ret void
}
-define amdgpu_kernel void @v_exclusive_scan_max_u32_undef(ptr addrspace(1) %out) {
+define amdgpu_kernel void @v_exclusive_scan_max_u32_poison(ptr addrspace(1) %out) {
;
-; GFX13-SDAG-LABEL: v_exclusive_scan_max_u32_undef:
+; GFX13-SDAG-LABEL: v_exclusive_scan_max_u32_poison:
; GFX13-SDAG: ; %bb.0:
; GFX13-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-SDAG-NEXT: v_mov_b32_e32 v0, 0
@@ -1580,7 +1580,7 @@ define amdgpu_kernel void @v_exclusive_scan_max_u32_undef(ptr addrspace(1) %out)
; GFX13-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX13-SDAG-NEXT: s_endpgm
;
-; GFX13-GISEL-LABEL: v_exclusive_scan_max_u32_undef:
+; GFX13-GISEL-LABEL: v_exclusive_scan_max_u32_poison:
; GFX13-GISEL: ; %bb.0:
; GFX13-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-GISEL-NEXT: v_mov_b32_e32 v1, 0
@@ -1588,7 +1588,7 @@ define amdgpu_kernel void @v_exclusive_scan_max_u32_undef(ptr addrspace(1) %out)
; GFX13-GISEL-NEXT: v_exclusive_scan_max_u32 v0, s0, s0
; GFX13-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13-GISEL-NEXT: s_endpgm
- %v = call i32 @llvm.amdgcn.exclusive.scan.max.u32(i32 undef, i32 undef)
+ %v = call i32 @llvm.amdgcn.exclusive.scan.max.u32(i32 poison, i32 poison)
store i32 %v, ptr addrspace(1) %out
ret void
}
>From dd8c1f0cdc2be2b76473ee07e98f54afb5a54424 Mon Sep 17 00:00:00 2001
From: Mariusz Sikora <mariusz.sikora at amd.com>
Date: Mon, 28 Sep 2026 07:28:38 -0400
Subject: [PATCH 3/3] Add Release Notes for new builtins/intrinsics
---
llvm/docs/AMDGPUUsage.rst | 8 ++++++++
llvm/docs/ReleaseNotes.md | 4 ++++
2 files changed, 12 insertions(+)
diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index 9254114e5044fd..10b4fddde7764d 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -1926,6 +1926,14 @@ The AMDGPU backend implements the following LLVM IR intrinsics.
bfloat, <2 x i16>, <2 x half>, <2 x bfloat>, i64, double, pointers, multiples of the
32-bit vectors.
+ llvm.amdgcn.exclusive.scan.* Provides direct access to the v_exclusive_scan_* instructions. Performs an
+ exclusive prefix scan of the first input operand across a subgroup of lanes,
+ selected by the mask in the second operand. Each lane receives the reduction of
+ the earlier lanes in its subgroup, so the lowest lane gets the identity value.
+ In wave64 mode the two halves of the wave are scanned independently. The operation
+ is part of the name (sum, xor, or, and, min, max).
+ Sum takes an extra i1 clamp operand.
+
llvm.amdgcn.udot2 Provides direct access to v_dot2_u32_u16 across targets which
support such instructions. This performs an unsigned dot product
with two v2i16 operands, summed with the third i32 operand. The
diff --git a/llvm/docs/ReleaseNotes.md b/llvm/docs/ReleaseNotes.md
index 41b2fa83d380f9..f393b5eeb739b6 100644
--- a/llvm/docs/ReleaseNotes.md
+++ b/llvm/docs/ReleaseNotes.md
@@ -247,6 +247,10 @@ Makes programs 10x faster by doing Special New Thing.
* `llvm.amdgcn.icmp`
* `llvm.amdgcn.fcmp`
+* Added `llvm.amdgcn.exclusive.scan.*` intrinsics (and matching
+ `__builtin_amdgcn_exclusive_scan_*` clang builtins) providing direct access to
+ the `v_exclusive_scan_*` subgroup prefix-scan instructions.
+
### Changes to the ARM Backend
* Using the hard-float procedure call standard without floating-point registers
More information about the llvm-commits
mailing list