[llvm] [AMDGPU] Gate runtime unroll of LDS loops instead of overriding it (PR #222853)

via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 11 01:44:56 PDT 2026


https://github.com/xgxanq updated https://github.com/llvm/llvm-project/pull/222853

>From 0a23edc931d1b1606cfa63ab6fc32e53b8e2efa8 Mon Sep 17 00:00:00 2001
From: anqfu <anqfu at amd.com>
Date: Fri, 11 Sep 2026 06:34:29 +0000
Subject: [PATCH] [AMDGPU] Gate runtime unroll of LDS loops instead of
 overriding it

In AMDGPUTTIImpl::getUnrollingPreferences, change UP.Runtime = UnrollRuntimeLocal
to UP.Runtime &= UnrollRuntimeLocal for loops touching local (LDS, addrspace(3))
memory. The old assignment could force runtime unrolling *on* even when other
preferences had disabled it; with the and, -amdgpu-unroll-runtime-local can only
narrow runtime unrolling of LDS loops, never re-enable it. This gives the knob a
well-defined, per-loop effect (see llvm/llvm-project#147700).

Add lit tests:
- runtime-unroll-local.ll: per-loop gating -- with the knob off, only the LDS
  loop's runtime-unroll epilogue is suppressed; a neighboring global-memory
  loop stays unrolled.
- unroll-runtime-local-mfma.ll: a convergent MFMA K-loop reading LDS with a
  runtime trip count is not runtime-unrolled with the knob off.

Assisted-by: Claude Code
---
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      |  4 +-
 .../LoopUnroll/AMDGPU/runtime-unroll-local.ll | 57 +++++++++++++++++++
 .../AMDGPU/unroll-runtime-local-mfma.ll       | 34 +++++++++++
 3 files changed, 93 insertions(+), 2 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll-local.ll
 create mode 100644 llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-runtime-local-mfma.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index a7556278b7e0d..28f70515d7b97 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -225,9 +225,9 @@ void AMDGPUTTIImpl::getUnrollingPreferences(
             (!isa<GlobalVariable>(GEP->getPointerOperand()) &&
              !isa<Argument>(GEP->getPointerOperand())))
           continue;
-        LLVM_DEBUG(dbgs() << "Allow unroll runtime for loop:\n"
+        LLVM_DEBUG(dbgs() << "Gating unroll runtime by local knob for loop:\n"
                           << *L << " due to LDS use.\n");
-        UP.Runtime = UnrollRuntimeLocal;
+        UP.Runtime &= UnrollRuntimeLocal;
       }
 
       // Check if GEP depends on a value defined by this loop itself.
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll-local.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll-local.ll
new file mode 100644
index 0000000000000..40c8fcc517506
--- /dev/null
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll-local.ll
@@ -0,0 +1,57 @@
+; RUN: opt -mtriple=amdgpu-- -passes=loop-unroll -S %s | FileCheck %s --check-prefixes=CHECK,DEFAULT
+; RUN: opt -mtriple=amdgpu-- -passes=loop-unroll -amdgpu-unroll-runtime-local=false -S %s | FileCheck %s --check-prefixes=CHECK,NOLOCAL
+
+; -amdgpu-unroll-runtime-local gates runtime unrolling per loop, based on
+; whether that loop touches local (LDS, addrspace(3)) memory. With the knob
+; off, only the LDS loop is suppressed; the global-memory loop is unaffected.
+;
+;                     knob=true (default)   knob=false
+;   %lds.loop            unrolled              NOT unrolled
+;   %global.loop         unrolled              unrolled (knob has no effect)
+
+ at lds = internal unnamed_addr addrspace(3) global [256 x i32] poison, align 4
+
+; CHECK-LABEL: @two_loops(
+define void @two_loops(ptr addrspace(1) %out, i32 %n, i32 %m) {
+entry:
+  %cmp = icmp sgt i32 %n, 0
+  br i1 %cmp, label %lds.loop, label %global.preheader
+
+; The LDS loop: gated by the knob. Its runtime-unroll epilogue block
+; (lds.loop.epil) appears only when the knob is on.
+;
+; DEFAULT: lds.loop.epil:
+;
+; NOLOCAL-NOT: lds.loop.epil
+lds.loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %lds.loop ]
+  %idx = zext i32 %iv to i64
+  %ptr = getelementptr inbounds [256 x i32], ptr addrspace(3) @lds, i64 0, i64 %idx
+  store i32 %iv, ptr addrspace(3) %ptr, align 4
+  %iv.next = add nuw nsw i32 %iv, 1
+  %exitcond = icmp eq i32 %iv.next, %n
+  br i1 %exitcond, label %global.preheader, label %lds.loop
+
+global.preheader:
+  %cmp2 = icmp sgt i32 %m, 0
+  br i1 %cmp2, label %global.loop, label %exit
+
+; The global-memory loop: never touches LDS, so the knob must not affect it.
+; Its runtime-unroll epilogue block (global.loop.epil) appears under both
+; knob settings.
+;
+; DEFAULT: global.loop.epil:
+;
+; NOLOCAL: global.loop.epil:
+global.loop:
+  %jv = phi i32 [ 0, %global.preheader ], [ %jv.next, %global.loop ]
+  %jdx = zext i32 %jv to i64
+  %gptr = getelementptr inbounds i32, ptr addrspace(1) %out, i64 %jdx
+  store i32 %jv, ptr addrspace(1) %gptr, align 4
+  %jv.next = add nuw nsw i32 %jv, 1
+  %exitcond2 = icmp eq i32 %jv.next, %m
+  br i1 %exitcond2, label %exit, label %global.loop
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-runtime-local-mfma.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-runtime-local-mfma.ll
new file mode 100644
index 0000000000000..5a2d4b8c558b4
--- /dev/null
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-runtime-local-mfma.ll
@@ -0,0 +1,34 @@
+; RUN: opt -mtriple=amdgpu9.50-amd-amdhsa -passes=loop-unroll -S %s \
+; RUN:   | FileCheck %s --check-prefixes=CHECK,DEFAULT
+; RUN: opt -mtriple=amdgpu9.50-amd-amdhsa -passes=loop-unroll \
+; RUN:   -amdgpu-unroll-runtime-local=false -S %s | FileCheck %s --check-prefixes=CHECK,NOLOCAL
+
+; A convergent MFMA K-loop reading LDS (addrspace(3)) with a runtime trip
+; count, modeled on a triton bf16 GEMM inner loop. By default it is runtime
+; unrolled; with -amdgpu-unroll-runtime-local off the LDS loop is not.
+
+ at global_smem = external addrspace(3) global [0 x i8], align 16
+
+; CHECK-LABEL: @gemm_k_loop(
+; DEFAULT: loop.epil:
+; NOLOCAL-NOT: loop.epil
+define amdgpu_kernel void @gemm_k_loop(<8 x bfloat> %a, i32 %n) {
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi <4 x float> [ zeroinitializer, %entry ], [ %mfma, %loop ]
+  %off = shl i32 %iv, 4
+  %ptr = getelementptr inbounds i8, ptr addrspace(3) @global_smem, i32 %off
+  %b = load <8 x bfloat>, ptr addrspace(3) %ptr, align 16
+  %mfma = tail call <4 x float> @llvm.amdgcn.mfma.f32.16x16x32.bf16(<8 x bfloat> %a, <8 x bfloat> %b, <4 x float> %acc, i32 0, i32 0, i32 0)
+  %iv.next = add nuw nsw i32 %iv, 1
+  %exit = icmp eq i32 %iv.next, %n
+  br i1 %exit, label %end, label %loop
+
+end:
+  ret void
+}
+
+declare <4 x float> @llvm.amdgcn.mfma.f32.16x16x32.bf16(<8 x bfloat>, <8 x bfloat>, <4 x float>, i32 immarg, i32 immarg, i32 immarg)



More information about the llvm-commits mailing list