[llvm] b09174b - [AMDGPU] Enable runtime loop unrolling (#194924)
via llvm-commits
llvm-commits at lists.llvm.org
Fri May 1 07:05:54 PDT 2026
Author: Adel Ejjeh
Date: 2026-05-01T09:05:49-05:00
New Revision: b09174b41e7ed406fb1ef7aa6881696b00a38170
URL: https://github.com/llvm/llvm-project/commit/b09174b41e7ed406fb1ef7aa6881696b00a38170
DIFF: https://github.com/llvm/llvm-project/commit/b09174b41e7ed406fb1ef7aa6881696b00a38170.diff
LOG: [AMDGPU] Enable runtime loop unrolling (#194924)
Enable auto runtime unrolling for AMDGPU by setting `UP.Runtime = true`
in `getUnrollingPreferences`, with `PartialThreshold = Threshold / 4` to
limit code-size growth.
Benchmarked on **MI350X (gfx950)** and **MI300X (gfx942)** using
Composable Kernel, xpu-perf, and llama.cpp. Results showed some some
improvements and no real regressions.
AI Disclaimer: Cursor was used to evaluate the change and run
benchmarking experiments.
Added:
llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll.ll
Modified:
llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
llvm/test/Transforms/SimpleLoopUnswitch/AMDGPU/uniform-unswitch.ll
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 030a9008c34dc..a087fd2590a6c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -127,7 +127,10 @@ void AMDGPUTTIImpl::getUnrollingPreferences(
// We want to run unroll even for the loops which have been vectorized.
UP.UnrollVectorizedLoop = true;
- // TODO: Do we want runtime unrolling?
+ // Enable runtime unrolling for loops whose trip count is not known at
+ // compile time. Use a reduced PartialThreshold to limit code-size growth.
+ UP.Runtime = true;
+ UP.PartialThreshold = UP.Threshold / 4;
// Maximum alloca size than can fit registers. Reserve 16 registers.
const unsigned MaxAlloca = (256 - 16) * 4;
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll.ll
new file mode 100644
index 0000000000000..769f78f422573
--- /dev/null
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll.ll
@@ -0,0 +1,63 @@
+; RUN: opt -mtriple=amdgcn-- -passes=loop-unroll -S %s | FileCheck %s
+
+; Verify that AMDGPU enables runtime loop unrolling for loops whose trip
+; count is not known at compile time.
+
+; Simple loop with unknown trip count — should be runtime-unrolled.
+define void @runtime_unroll_simple(ptr addrspace(1) %out, i32 %n) {
+; CHECK-LABEL: @runtime_unroll_simple(
+; CHECK: %xtraiter = and i32 %n,
+; CHECK: for.body.epil.preheader:
+; CHECK: for.body.epil:
+; CHECK: %epil.iter
+entry:
+ %cmp = icmp sgt i32 %n, 0
+ br i1 %cmp, label %for.body, label %exit
+
+for.body:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %for.body ]
+ %idx = zext i32 %iv to i64
+ %ptr = getelementptr inbounds i32, ptr addrspace(1) %out, i64 %idx
+ store i32 %iv, ptr addrspace(1) %ptr, align 4
+ %iv.next = add nuw nsw i32 %iv, 1
+ %exitcond = icmp eq i32 %iv.next, %n
+ br i1 %exitcond, label %exit, label %for.body
+
+exit:
+ ret void
+}
+
+; Loop with convergent call — runtime unrolling must NOT fire because the
+; prologue/epilogue would introduce divergent control flow around the
+; convergent operation.
+define void @no_runtime_unroll_convergent(ptr addrspace(1) %out, i32 %n) {
+; CHECK-LABEL: @no_runtime_unroll_convergent(
+; CHECK-NOT: xtraiter
+; CHECK-NOT: epil
+; CHECK: for.body:
+; CHECK: %iv = phi i32
+; CHECK: call i32 @llvm.amdgcn.readfirstlane.i32
+; CHECK: %iv.next = add nuw nsw i32 %iv, 1
+; CHECK: %exitcond = icmp eq i32 %iv.next, %n
+; CHECK: br i1 %exitcond
+entry:
+ %cmp = icmp sgt i32 %n, 0
+ br i1 %cmp, label %for.body, label %exit
+
+for.body:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %for.body ]
+ %lane0 = call i32 @llvm.amdgcn.readfirstlane.i32(i32 %iv)
+ %idx = zext i32 %lane0 to i64
+ %ptr = getelementptr inbounds i32, ptr addrspace(1) %out, i64 %idx
+ store i32 %iv, ptr addrspace(1) %ptr, align 4
+ %iv.next = add nuw nsw i32 %iv, 1
+ %exitcond = icmp eq i32 %iv.next, %n
+ br i1 %exitcond, label %exit, label %for.body
+
+exit:
+ ret void
+}
+
+declare i32 @llvm.amdgcn.readfirstlane.i32(i32) #0
+
+attributes #0 = { nounwind convergent willreturn readnone }
diff --git a/llvm/test/Transforms/SimpleLoopUnswitch/AMDGPU/uniform-unswitch.ll b/llvm/test/Transforms/SimpleLoopUnswitch/AMDGPU/uniform-unswitch.ll
index 209eda54069cd..455eb062a7272 100644
--- a/llvm/test/Transforms/SimpleLoopUnswitch/AMDGPU/uniform-unswitch.ll
+++ b/llvm/test/Transforms/SimpleLoopUnswitch/AMDGPU/uniform-unswitch.ll
@@ -41,7 +41,7 @@ define amdgpu_kernel void @uniform_unswitch(ptr nocapture %out, i32 %n, i32 %x)
; CHECK: for.inc:
; CHECK-NEXT: [[INC]] = add nuw nsw i32 [[I_07]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i32 [[INC]], [[N]]
-; CHECK-NEXT: br i1 [[EXITCOND]], label [[FOR_COND_CLEANUP]], label [[FOR_BODY]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label [[FOR_COND_CLEANUP]], label [[FOR_BODY]], !llvm.loop !0
;
entry:
%cmp6 = icmp sgt i32 %n, 0
@@ -69,9 +69,12 @@ if.then: ; preds = %for.body
for.inc: ; preds = %for.body, %if.then
%inc = add nuw nsw i32 %i.07, 1
%exitcond = icmp eq i32 %inc, %n
- br i1 %exitcond, label %for.cond.cleanup.loopexit, label %for.body
+ br i1 %exitcond, label %for.cond.cleanup.loopexit, label %for.body, !llvm.loop !0
}
declare i32 @llvm.amdgcn.workitem.id.x() #0
attributes #0 = { nounwind readnone }
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.unroll.disable"}
More information about the llvm-commits
mailing list