[llvm] [AMDGPU] Add partial unroll threshold function attribute (PR #223291)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 14 18:55:44 PDT 2026
https://github.com/nina-zhang-0 updated https://github.com/llvm/llvm-project/pull/223291
>From 492e7f25fe9e0564e031c25e02279221f9a70fa9 Mon Sep 17 00:00:00 2001
From: nizhang1 <nizhang1 at amd.com>
Date: Thu, 10 Sep 2026 16:52:40 +0800
Subject: [PATCH 1/3] [AMDGPU] Add partial unroll threshold function attribute
---
llvm/docs/AMDGPUUsage.rst | 4 +++
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 2 ++
.../LoopUnroll/AMDGPU/unroll-threshold.ll | 34 +++++++++++++++++++
3 files changed, 40 insertions(+)
diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index 0d1fcf0700122..70bb7614e71da 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -2756,6 +2756,10 @@ The AMDGPU backend supports the following LLVM IR attributes.
default is 300. Actual threshold may be varied by per-loop metadata or
reduced by heuristics.
+ "amdgpu-partial-unroll-threshold" Set base cost threshold preference for partial and runtime loop unrolling
+ within this function, default is 150. This does not change the threshold
+ used for full unrolling.
+
"amdgpu-max-num-workgroups"="x,y,z" Specify the maximum number of work groups for the kernel dispatch in the
X, Y, and Z dimensions. Each number must be >= 1. Generated by the
``amdgpu_max_num_work_groups`` CLANG attribute [CLANG-ATTR]_. Clang only
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index a7556278b7e0d..fbf4d6ed52888 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -117,6 +117,8 @@ void AMDGPUTTIImpl::getUnrollingPreferences(
const Function &F = *L->getHeader()->getParent();
UP.Threshold =
F.getFnAttributeAsParsedInteger("amdgpu-unroll-threshold", 300);
+ UP.PartialThreshold = F.getFnAttributeAsParsedInteger(
+ "amdgpu-partial-unroll-threshold", UP.PartialThreshold);
UP.MaxCount = std::numeric_limits<unsigned>::max();
UP.Partial = true;
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
index af437ba1d8a6a..a41a13e2c1d5a 100644
--- a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
@@ -104,8 +104,42 @@ do.end: ; preds = %do.body
ret void
}
+; Check that the threshold used for partial unrolling is independent of the
+; threshold used for full unrolling. A partial threshold of 30 produces two
+; copies of the loop body while the threshold for full unrolling remains zero.
+; CHECK-LABEL: @partial_unroll_threshold(
+; CHECK: partial.body:
+; CHECK: store i32
+; CHECK: br i1
+; CHECK: partial.body.1:
+; CHECK: store i32
+; CHECK-NOT: partial.body.2:
+; CHECK: ret void
+
+define void @partial_unroll_threshold(ptr addrspace(1) %a,
+ ptr addrspace(1) %b) #2 {
+entry:
+ br label %partial.body
+
+partial.body: ; preds = %entry, %partial.body
+ %iv = phi i64 [ 1, %entry ], [ %iv.next, %partial.body ]
+ %src = getelementptr inbounds i32, ptr addrspace(1) %b, i64 %iv
+ %value = load i32, ptr addrspace(1) %src, align 4
+ %index = sext i32 %value to i64
+ %dst = getelementptr inbounds i32, ptr addrspace(1) %a, i64 %index
+ %stored = trunc i64 %iv to i32
+ store i32 %stored, ptr addrspace(1) %dst, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 20
+ br i1 %exitcond, label %exit, label %partial.body
+
+exit: ; preds = %partial.body
+ ret void
+}
+
attributes #0 = { "amdgpu-unroll-threshold"="1000" }
attributes #1 = { "amdgpu-unroll-threshold"="100" }
+attributes #2 = { "amdgpu-unroll-threshold"="0" "amdgpu-partial-unroll-threshold"="30" }
!1 = !{!1, !2}
!2 = !{!"amdgpu.loop.unroll.threshold", i32 1000}
>From a7de7c9b0de01a07c3ff5d519ee6123bd65ecc31 Mon Sep 17 00:00:00 2001
From: nizhang1 <nizhang1 at amd.com>
Date: Mon, 14 Sep 2026 14:46:44 +0800
Subject: [PATCH 2/3] AMDGPU: Address review feedback
---
llvm/docs/AMDGPUUsage.rst | 5 +--
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 2 +-
.../AMDGPU/partial-unroll-threshold.ll | 36 +++++++++++++++++++
.../LoopUnroll/AMDGPU/unroll-threshold.ll | 34 ------------------
4 files changed, 40 insertions(+), 37 deletions(-)
create mode 100644 llvm/test/Transforms/LoopUnroll/AMDGPU/partial-unroll-threshold.ll
diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index 70bb7614e71da..14f7bb48eaafb 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -2757,8 +2757,9 @@ The AMDGPU backend supports the following LLVM IR attributes.
reduced by heuristics.
"amdgpu-partial-unroll-threshold" Set base cost threshold preference for partial and runtime loop unrolling
- within this function, default is 150. This does not change the threshold
- used for full unrolling.
+ within this function, default is 150. This is independent of
+ ``amdgpu-unroll-threshold``, which controls the threshold used for full
+ unrolling.
"amdgpu-max-num-workgroups"="x,y,z" Specify the maximum number of work groups for the kernel dispatch in the
X, Y, and Z dimensions. Each number must be >= 1. Generated by the
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index fbf4d6ed52888..be91868651cd1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -118,7 +118,7 @@ void AMDGPUTTIImpl::getUnrollingPreferences(
UP.Threshold =
F.getFnAttributeAsParsedInteger("amdgpu-unroll-threshold", 300);
UP.PartialThreshold = F.getFnAttributeAsParsedInteger(
- "amdgpu-partial-unroll-threshold", UP.PartialThreshold);
+ "amdgpu-partial-unroll-threshold", 150);
UP.MaxCount = std::numeric_limits<unsigned>::max();
UP.Partial = true;
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/partial-unroll-threshold.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/partial-unroll-threshold.ll
new file mode 100644
index 0000000000000..37e33c4e9a94f
--- /dev/null
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/partial-unroll-threshold.ll
@@ -0,0 +1,36 @@
+; RUN: opt < %s -S -mtriple=amdgpu-- -passes=loop-unroll | FileCheck %s
+
+; Check that the threshold used for partial unrolling is independent of the
+; threshold used for full unrolling. A partial threshold of 30 produces two
+; copies of the loop body while the threshold for full unrolling remains zero.
+; CHECK-LABEL: @partial_unroll_threshold(
+; CHECK: partial.body:
+; CHECK: store i32
+; CHECK: br i1
+; CHECK: partial.body.1:
+; CHECK: store i32
+; CHECK-NOT: partial.body.2:
+; CHECK: ret void
+
+define void @partial_unroll_threshold(ptr addrspace(1) %a,
+ ptr addrspace(1) %b) #0 {
+entry:
+ br label %partial.body
+
+partial.body: ; preds = %entry, %partial.body
+ %iv = phi i64 [ 1, %entry ], [ %iv.next, %partial.body ]
+ %src = getelementptr inbounds i32, ptr addrspace(1) %b, i64 %iv
+ %value = load i32, ptr addrspace(1) %src, align 4
+ %index = sext i32 %value to i64
+ %dst = getelementptr inbounds i32, ptr addrspace(1) %a, i64 %index
+ %stored = trunc i64 %iv to i32
+ store i32 %stored, ptr addrspace(1) %dst, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 20
+ br i1 %exitcond, label %exit, label %partial.body
+
+exit: ; preds = %partial.body
+ ret void
+}
+
+attributes #0 = { "amdgpu-unroll-threshold"="0" "amdgpu-partial-unroll-threshold"="30" }
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
index a41a13e2c1d5a..af437ba1d8a6a 100644
--- a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
@@ -104,42 +104,8 @@ do.end: ; preds = %do.body
ret void
}
-; Check that the threshold used for partial unrolling is independent of the
-; threshold used for full unrolling. A partial threshold of 30 produces two
-; copies of the loop body while the threshold for full unrolling remains zero.
-; CHECK-LABEL: @partial_unroll_threshold(
-; CHECK: partial.body:
-; CHECK: store i32
-; CHECK: br i1
-; CHECK: partial.body.1:
-; CHECK: store i32
-; CHECK-NOT: partial.body.2:
-; CHECK: ret void
-
-define void @partial_unroll_threshold(ptr addrspace(1) %a,
- ptr addrspace(1) %b) #2 {
-entry:
- br label %partial.body
-
-partial.body: ; preds = %entry, %partial.body
- %iv = phi i64 [ 1, %entry ], [ %iv.next, %partial.body ]
- %src = getelementptr inbounds i32, ptr addrspace(1) %b, i64 %iv
- %value = load i32, ptr addrspace(1) %src, align 4
- %index = sext i32 %value to i64
- %dst = getelementptr inbounds i32, ptr addrspace(1) %a, i64 %index
- %stored = trunc i64 %iv to i32
- store i32 %stored, ptr addrspace(1) %dst, align 4
- %iv.next = add nuw nsw i64 %iv, 1
- %exitcond = icmp eq i64 %iv.next, 20
- br i1 %exitcond, label %exit, label %partial.body
-
-exit: ; preds = %partial.body
- ret void
-}
-
attributes #0 = { "amdgpu-unroll-threshold"="1000" }
attributes #1 = { "amdgpu-unroll-threshold"="100" }
-attributes #2 = { "amdgpu-unroll-threshold"="0" "amdgpu-partial-unroll-threshold"="30" }
!1 = !{!1, !2}
!2 = !{!"amdgpu.loop.unroll.threshold", i32 1000}
>From 544535b1f56613f560021d8f8f5a64a14ea2a665 Mon Sep 17 00:00:00 2001
From: nizhang1 <Nina.Zhang1 at amd.com>
Date: Tue, 15 Sep 2026 09:46:01 +0800
Subject: [PATCH 3/3] [AMDGPU] Fix format
---
llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index be91868651cd1..fdabf397fc614 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -117,8 +117,8 @@ void AMDGPUTTIImpl::getUnrollingPreferences(
const Function &F = *L->getHeader()->getParent();
UP.Threshold =
F.getFnAttributeAsParsedInteger("amdgpu-unroll-threshold", 300);
- UP.PartialThreshold = F.getFnAttributeAsParsedInteger(
- "amdgpu-partial-unroll-threshold", 150);
+ UP.PartialThreshold =
+ F.getFnAttributeAsParsedInteger("amdgpu-partial-unroll-threshold", 150);
UP.MaxCount = std::numeric_limits<unsigned>::max();
UP.Partial = true;
More information about the llvm-commits
mailing list