[llvm] b09174b - [AMDGPU] Enable runtime loop unrolling (#194924)

via llvm-commits llvm-commits at lists.llvm.org
Fri May 1 07:05:54 PDT 2026


Author: Adel Ejjeh
Date: 2026-05-01T09:05:49-05:00
New Revision: b09174b41e7ed406fb1ef7aa6881696b00a38170

URL: https://github.com/llvm/llvm-project/commit/b09174b41e7ed406fb1ef7aa6881696b00a38170
DIFF: https://github.com/llvm/llvm-project/commit/b09174b41e7ed406fb1ef7aa6881696b00a38170.diff

LOG: [AMDGPU] Enable runtime loop unrolling (#194924)

Enable auto runtime unrolling for AMDGPU by setting `UP.Runtime = true`
in `getUnrollingPreferences`, with `PartialThreshold = Threshold / 4` to
limit code-size growth.

Benchmarked on **MI350X (gfx950)** and **MI300X (gfx942)** using
Composable Kernel, xpu-perf, and llama.cpp. Results showed some some
improvements and no real regressions.

AI Disclaimer: Cursor was used to evaluate the change and run
benchmarking experiments.

Added: 
    llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll.ll

Modified: 
    llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
    llvm/test/Transforms/SimpleLoopUnswitch/AMDGPU/uniform-unswitch.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 030a9008c34dc..a087fd2590a6c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -127,7 +127,10 @@ void AMDGPUTTIImpl::getUnrollingPreferences(
   // We want to run unroll even for the loops which have been vectorized.
   UP.UnrollVectorizedLoop = true;
 
-  // TODO: Do we want runtime unrolling?
+  // Enable runtime unrolling for loops whose trip count is not known at
+  // compile time.  Use a reduced PartialThreshold to limit code-size growth.
+  UP.Runtime = true;
+  UP.PartialThreshold = UP.Threshold / 4;
 
   // Maximum alloca size than can fit registers. Reserve 16 registers.
   const unsigned MaxAlloca = (256 - 16) * 4;

diff  --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll.ll
new file mode 100644
index 0000000000000..769f78f422573
--- /dev/null
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll.ll
@@ -0,0 +1,63 @@
+; RUN: opt -mtriple=amdgcn-- -passes=loop-unroll -S %s | FileCheck %s
+
+; Verify that AMDGPU enables runtime loop unrolling for loops whose trip
+; count is not known at compile time.
+
+; Simple loop with unknown trip count — should be runtime-unrolled.
+define void @runtime_unroll_simple(ptr addrspace(1) %out, i32 %n) {
+; CHECK-LABEL: @runtime_unroll_simple(
+; CHECK: %xtraiter = and i32 %n,
+; CHECK: for.body.epil.preheader:
+; CHECK: for.body.epil:
+; CHECK: %epil.iter
+entry:
+  %cmp = icmp sgt i32 %n, 0
+  br i1 %cmp, label %for.body, label %exit
+
+for.body:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %for.body ]
+  %idx = zext i32 %iv to i64
+  %ptr = getelementptr inbounds i32, ptr addrspace(1) %out, i64 %idx
+  store i32 %iv, ptr addrspace(1) %ptr, align 4
+  %iv.next = add nuw nsw i32 %iv, 1
+  %exitcond = icmp eq i32 %iv.next, %n
+  br i1 %exitcond, label %exit, label %for.body
+
+exit:
+  ret void
+}
+
+; Loop with convergent call — runtime unrolling must NOT fire because the
+; prologue/epilogue would introduce divergent control flow around the
+; convergent operation.
+define void @no_runtime_unroll_convergent(ptr addrspace(1) %out, i32 %n) {
+; CHECK-LABEL: @no_runtime_unroll_convergent(
+; CHECK-NOT: xtraiter
+; CHECK-NOT: epil
+; CHECK: for.body:
+; CHECK: %iv = phi i32
+; CHECK: call i32 @llvm.amdgcn.readfirstlane.i32
+; CHECK: %iv.next = add nuw nsw i32 %iv, 1
+; CHECK: %exitcond = icmp eq i32 %iv.next, %n
+; CHECK: br i1 %exitcond
+entry:
+  %cmp = icmp sgt i32 %n, 0
+  br i1 %cmp, label %for.body, label %exit
+
+for.body:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %for.body ]
+  %lane0 = call i32 @llvm.amdgcn.readfirstlane.i32(i32 %iv)
+  %idx = zext i32 %lane0 to i64
+  %ptr = getelementptr inbounds i32, ptr addrspace(1) %out, i64 %idx
+  store i32 %iv, ptr addrspace(1) %ptr, align 4
+  %iv.next = add nuw nsw i32 %iv, 1
+  %exitcond = icmp eq i32 %iv.next, %n
+  br i1 %exitcond, label %exit, label %for.body
+
+exit:
+  ret void
+}
+
+declare i32 @llvm.amdgcn.readfirstlane.i32(i32) #0
+
+attributes #0 = { nounwind convergent willreturn readnone }

diff  --git a/llvm/test/Transforms/SimpleLoopUnswitch/AMDGPU/uniform-unswitch.ll b/llvm/test/Transforms/SimpleLoopUnswitch/AMDGPU/uniform-unswitch.ll
index 209eda54069cd..455eb062a7272 100644
--- a/llvm/test/Transforms/SimpleLoopUnswitch/AMDGPU/uniform-unswitch.ll
+++ b/llvm/test/Transforms/SimpleLoopUnswitch/AMDGPU/uniform-unswitch.ll
@@ -41,7 +41,7 @@ define amdgpu_kernel void @uniform_unswitch(ptr nocapture %out, i32 %n, i32 %x)
 ; CHECK:       for.inc:
 ; CHECK-NEXT:    [[INC]] = add nuw nsw i32 [[I_07]], 1
 ; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp eq i32 [[INC]], [[N]]
-; CHECK-NEXT:    br i1 [[EXITCOND]], label [[FOR_COND_CLEANUP]], label [[FOR_BODY]]
+; CHECK-NEXT:    br i1 [[EXITCOND]], label [[FOR_COND_CLEANUP]], label [[FOR_BODY]], !llvm.loop !0
 ;
 entry:
   %cmp6 = icmp sgt i32 %n, 0
@@ -69,9 +69,12 @@ if.then:                                          ; preds = %for.body
 for.inc:                                          ; preds = %for.body, %if.then
   %inc = add nuw nsw i32 %i.07, 1
   %exitcond = icmp eq i32 %inc, %n
-  br i1 %exitcond, label %for.cond.cleanup.loopexit, label %for.body
+  br i1 %exitcond, label %for.cond.cleanup.loopexit, label %for.body, !llvm.loop !0
 }
 
 declare i32 @llvm.amdgcn.workitem.id.x() #0
 
 attributes #0 = { nounwind readnone }
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.unroll.disable"}


        


More information about the llvm-commits mailing list