[llvm] [AMDGPU] Add partial unroll threshold function attribute (PR #223291)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 14 18:55:44 PDT 2026


https://github.com/nina-zhang-0 updated https://github.com/llvm/llvm-project/pull/223291

>From 492e7f25fe9e0564e031c25e02279221f9a70fa9 Mon Sep 17 00:00:00 2001
From: nizhang1 <nizhang1 at amd.com>
Date: Thu, 10 Sep 2026 16:52:40 +0800
Subject: [PATCH 1/3] [AMDGPU] Add partial unroll threshold function attribute

---
 llvm/docs/AMDGPUUsage.rst                     |  4 +++
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      |  2 ++
 .../LoopUnroll/AMDGPU/unroll-threshold.ll     | 34 +++++++++++++++++++
 3 files changed, 40 insertions(+)

diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index 0d1fcf0700122..70bb7614e71da 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -2756,6 +2756,10 @@ The AMDGPU backend supports the following LLVM IR attributes.
                                                       default is 300. Actual threshold may be varied by per-loop metadata or
                                                       reduced by heuristics.
 
+     "amdgpu-partial-unroll-threshold"                Set base cost threshold preference for partial and runtime loop unrolling
+                                                      within this function, default is 150. This does not change the threshold
+                                                      used for full unrolling.
+
      "amdgpu-max-num-workgroups"="x,y,z"              Specify the maximum number of work groups for the kernel dispatch in the
                                                       X, Y, and Z dimensions. Each number must be >= 1. Generated by the
                                                       ``amdgpu_max_num_work_groups`` CLANG attribute [CLANG-ATTR]_. Clang only
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index a7556278b7e0d..fbf4d6ed52888 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -117,6 +117,8 @@ void AMDGPUTTIImpl::getUnrollingPreferences(
   const Function &F = *L->getHeader()->getParent();
   UP.Threshold =
       F.getFnAttributeAsParsedInteger("amdgpu-unroll-threshold", 300);
+  UP.PartialThreshold = F.getFnAttributeAsParsedInteger(
+      "amdgpu-partial-unroll-threshold", UP.PartialThreshold);
   UP.MaxCount = std::numeric_limits<unsigned>::max();
   UP.Partial = true;
 
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
index af437ba1d8a6a..a41a13e2c1d5a 100644
--- a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
@@ -104,8 +104,42 @@ do.end:                                           ; preds = %do.body
   ret void
 }
 
+; Check that the threshold used for partial unrolling is independent of the
+; threshold used for full unrolling. A partial threshold of 30 produces two
+; copies of the loop body while the threshold for full unrolling remains zero.
+; CHECK-LABEL: @partial_unroll_threshold(
+; CHECK: partial.body:
+; CHECK: store i32
+; CHECK: br i1
+; CHECK: partial.body.1:
+; CHECK: store i32
+; CHECK-NOT: partial.body.2:
+; CHECK: ret void
+
+define void @partial_unroll_threshold(ptr addrspace(1) %a,
+                                      ptr addrspace(1) %b) #2 {
+entry:
+  br label %partial.body
+
+partial.body:                                     ; preds = %entry, %partial.body
+  %iv = phi i64 [ 1, %entry ], [ %iv.next, %partial.body ]
+  %src = getelementptr inbounds i32, ptr addrspace(1) %b, i64 %iv
+  %value = load i32, ptr addrspace(1) %src, align 4
+  %index = sext i32 %value to i64
+  %dst = getelementptr inbounds i32, ptr addrspace(1) %a, i64 %index
+  %stored = trunc i64 %iv to i32
+  store i32 %stored, ptr addrspace(1) %dst, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond = icmp eq i64 %iv.next, 20
+  br i1 %exitcond, label %exit, label %partial.body
+
+exit:                                             ; preds = %partial.body
+  ret void
+}
+
 attributes #0 = { "amdgpu-unroll-threshold"="1000" }
 attributes #1 = { "amdgpu-unroll-threshold"="100" }
+attributes #2 = { "amdgpu-unroll-threshold"="0" "amdgpu-partial-unroll-threshold"="30" }
 
 !1 = !{!1, !2}
 !2 = !{!"amdgpu.loop.unroll.threshold", i32 1000}

>From a7de7c9b0de01a07c3ff5d519ee6123bd65ecc31 Mon Sep 17 00:00:00 2001
From: nizhang1 <nizhang1 at amd.com>
Date: Mon, 14 Sep 2026 14:46:44 +0800
Subject: [PATCH 2/3] AMDGPU: Address review feedback

---
 llvm/docs/AMDGPUUsage.rst                     |  5 +--
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      |  2 +-
 .../AMDGPU/partial-unroll-threshold.ll        | 36 +++++++++++++++++++
 .../LoopUnroll/AMDGPU/unroll-threshold.ll     | 34 ------------------
 4 files changed, 40 insertions(+), 37 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopUnroll/AMDGPU/partial-unroll-threshold.ll

diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index 70bb7614e71da..14f7bb48eaafb 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -2757,8 +2757,9 @@ The AMDGPU backend supports the following LLVM IR attributes.
                                                       reduced by heuristics.
 
      "amdgpu-partial-unroll-threshold"                Set base cost threshold preference for partial and runtime loop unrolling
-                                                      within this function, default is 150. This does not change the threshold
-                                                      used for full unrolling.
+                                                      within this function, default is 150. This is independent of
+                                                      ``amdgpu-unroll-threshold``, which controls the threshold used for full
+                                                      unrolling.
 
      "amdgpu-max-num-workgroups"="x,y,z"              Specify the maximum number of work groups for the kernel dispatch in the
                                                       X, Y, and Z dimensions. Each number must be >= 1. Generated by the
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index fbf4d6ed52888..be91868651cd1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -118,7 +118,7 @@ void AMDGPUTTIImpl::getUnrollingPreferences(
   UP.Threshold =
       F.getFnAttributeAsParsedInteger("amdgpu-unroll-threshold", 300);
   UP.PartialThreshold = F.getFnAttributeAsParsedInteger(
-      "amdgpu-partial-unroll-threshold", UP.PartialThreshold);
+      "amdgpu-partial-unroll-threshold", 150);
   UP.MaxCount = std::numeric_limits<unsigned>::max();
   UP.Partial = true;
 
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/partial-unroll-threshold.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/partial-unroll-threshold.ll
new file mode 100644
index 0000000000000..37e33c4e9a94f
--- /dev/null
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/partial-unroll-threshold.ll
@@ -0,0 +1,36 @@
+; RUN: opt < %s -S -mtriple=amdgpu-- -passes=loop-unroll | FileCheck %s
+
+; Check that the threshold used for partial unrolling is independent of the
+; threshold used for full unrolling. A partial threshold of 30 produces two
+; copies of the loop body while the threshold for full unrolling remains zero.
+; CHECK-LABEL: @partial_unroll_threshold(
+; CHECK: partial.body:
+; CHECK: store i32
+; CHECK: br i1
+; CHECK: partial.body.1:
+; CHECK: store i32
+; CHECK-NOT: partial.body.2:
+; CHECK: ret void
+
+define void @partial_unroll_threshold(ptr addrspace(1) %a,
+                                      ptr addrspace(1) %b) #0 {
+entry:
+  br label %partial.body
+
+partial.body:                                     ; preds = %entry, %partial.body
+  %iv = phi i64 [ 1, %entry ], [ %iv.next, %partial.body ]
+  %src = getelementptr inbounds i32, ptr addrspace(1) %b, i64 %iv
+  %value = load i32, ptr addrspace(1) %src, align 4
+  %index = sext i32 %value to i64
+  %dst = getelementptr inbounds i32, ptr addrspace(1) %a, i64 %index
+  %stored = trunc i64 %iv to i32
+  store i32 %stored, ptr addrspace(1) %dst, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond = icmp eq i64 %iv.next, 20
+  br i1 %exitcond, label %exit, label %partial.body
+
+exit:                                             ; preds = %partial.body
+  ret void
+}
+
+attributes #0 = { "amdgpu-unroll-threshold"="0" "amdgpu-partial-unroll-threshold"="30" }
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
index a41a13e2c1d5a..af437ba1d8a6a 100644
--- a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-threshold.ll
@@ -104,42 +104,8 @@ do.end:                                           ; preds = %do.body
   ret void
 }
 
-; Check that the threshold used for partial unrolling is independent of the
-; threshold used for full unrolling. A partial threshold of 30 produces two
-; copies of the loop body while the threshold for full unrolling remains zero.
-; CHECK-LABEL: @partial_unroll_threshold(
-; CHECK: partial.body:
-; CHECK: store i32
-; CHECK: br i1
-; CHECK: partial.body.1:
-; CHECK: store i32
-; CHECK-NOT: partial.body.2:
-; CHECK: ret void
-
-define void @partial_unroll_threshold(ptr addrspace(1) %a,
-                                      ptr addrspace(1) %b) #2 {
-entry:
-  br label %partial.body
-
-partial.body:                                     ; preds = %entry, %partial.body
-  %iv = phi i64 [ 1, %entry ], [ %iv.next, %partial.body ]
-  %src = getelementptr inbounds i32, ptr addrspace(1) %b, i64 %iv
-  %value = load i32, ptr addrspace(1) %src, align 4
-  %index = sext i32 %value to i64
-  %dst = getelementptr inbounds i32, ptr addrspace(1) %a, i64 %index
-  %stored = trunc i64 %iv to i32
-  store i32 %stored, ptr addrspace(1) %dst, align 4
-  %iv.next = add nuw nsw i64 %iv, 1
-  %exitcond = icmp eq i64 %iv.next, 20
-  br i1 %exitcond, label %exit, label %partial.body
-
-exit:                                             ; preds = %partial.body
-  ret void
-}
-
 attributes #0 = { "amdgpu-unroll-threshold"="1000" }
 attributes #1 = { "amdgpu-unroll-threshold"="100" }
-attributes #2 = { "amdgpu-unroll-threshold"="0" "amdgpu-partial-unroll-threshold"="30" }
 
 !1 = !{!1, !2}
 !2 = !{!"amdgpu.loop.unroll.threshold", i32 1000}

>From 544535b1f56613f560021d8f8f5a64a14ea2a665 Mon Sep 17 00:00:00 2001
From: nizhang1 <Nina.Zhang1 at amd.com>
Date: Tue, 15 Sep 2026 09:46:01 +0800
Subject: [PATCH 3/3] [AMDGPU] Fix format

---
 llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index be91868651cd1..fdabf397fc614 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -117,8 +117,8 @@ void AMDGPUTTIImpl::getUnrollingPreferences(
   const Function &F = *L->getHeader()->getParent();
   UP.Threshold =
       F.getFnAttributeAsParsedInteger("amdgpu-unroll-threshold", 300);
-  UP.PartialThreshold = F.getFnAttributeAsParsedInteger(
-      "amdgpu-partial-unroll-threshold", 150);
+  UP.PartialThreshold =
+      F.getFnAttributeAsParsedInteger("amdgpu-partial-unroll-threshold", 150);
   UP.MaxCount = std::numeric_limits<unsigned>::max();
   UP.Partial = true;
 



More information about the llvm-commits mailing list