[llvm] [AMDGPU] Gate runtime unroll of LDS loops instead of overriding it (PR #222853)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 11 01:44:56 PDT 2026
https://github.com/xgxanq updated https://github.com/llvm/llvm-project/pull/222853
>From 0a23edc931d1b1606cfa63ab6fc32e53b8e2efa8 Mon Sep 17 00:00:00 2001
From: anqfu <anqfu at amd.com>
Date: Fri, 11 Sep 2026 06:34:29 +0000
Subject: [PATCH] [AMDGPU] Gate runtime unroll of LDS loops instead of
overriding it
In AMDGPUTTIImpl::getUnrollingPreferences, change UP.Runtime = UnrollRuntimeLocal
to UP.Runtime &= UnrollRuntimeLocal for loops touching local (LDS, addrspace(3))
memory. The old assignment could force runtime unrolling *on* even when other
preferences had disabled it; with the and, -amdgpu-unroll-runtime-local can only
narrow runtime unrolling of LDS loops, never re-enable it. This gives the knob a
well-defined, per-loop effect (see llvm/llvm-project#147700).
Add lit tests:
- runtime-unroll-local.ll: per-loop gating -- with the knob off, only the LDS
loop's runtime-unroll epilogue is suppressed; a neighboring global-memory
loop stays unrolled.
- unroll-runtime-local-mfma.ll: a convergent MFMA K-loop reading LDS with a
runtime trip count is not runtime-unrolled with the knob off.
Assisted-by: Claude Code
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 4 +-
.../LoopUnroll/AMDGPU/runtime-unroll-local.ll | 57 +++++++++++++++++++
.../AMDGPU/unroll-runtime-local-mfma.ll | 34 +++++++++++
3 files changed, 93 insertions(+), 2 deletions(-)
create mode 100644 llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll-local.ll
create mode 100644 llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-runtime-local-mfma.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index a7556278b7e0d..28f70515d7b97 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -225,9 +225,9 @@ void AMDGPUTTIImpl::getUnrollingPreferences(
(!isa<GlobalVariable>(GEP->getPointerOperand()) &&
!isa<Argument>(GEP->getPointerOperand())))
continue;
- LLVM_DEBUG(dbgs() << "Allow unroll runtime for loop:\n"
+ LLVM_DEBUG(dbgs() << "Gating unroll runtime by local knob for loop:\n"
<< *L << " due to LDS use.\n");
- UP.Runtime = UnrollRuntimeLocal;
+ UP.Runtime &= UnrollRuntimeLocal;
}
// Check if GEP depends on a value defined by this loop itself.
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll-local.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll-local.ll
new file mode 100644
index 0000000000000..40c8fcc517506
--- /dev/null
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/runtime-unroll-local.ll
@@ -0,0 +1,57 @@
+; RUN: opt -mtriple=amdgpu-- -passes=loop-unroll -S %s | FileCheck %s --check-prefixes=CHECK,DEFAULT
+; RUN: opt -mtriple=amdgpu-- -passes=loop-unroll -amdgpu-unroll-runtime-local=false -S %s | FileCheck %s --check-prefixes=CHECK,NOLOCAL
+
+; -amdgpu-unroll-runtime-local gates runtime unrolling per loop, based on
+; whether that loop touches local (LDS, addrspace(3)) memory. With the knob
+; off, only the LDS loop is suppressed; the global-memory loop is unaffected.
+;
+; knob=true (default) knob=false
+; %lds.loop unrolled NOT unrolled
+; %global.loop unrolled unrolled (knob has no effect)
+
+ at lds = internal unnamed_addr addrspace(3) global [256 x i32] poison, align 4
+
+; CHECK-LABEL: @two_loops(
+define void @two_loops(ptr addrspace(1) %out, i32 %n, i32 %m) {
+entry:
+ %cmp = icmp sgt i32 %n, 0
+ br i1 %cmp, label %lds.loop, label %global.preheader
+
+; The LDS loop: gated by the knob. Its runtime-unroll epilogue block
+; (lds.loop.epil) appears only when the knob is on.
+;
+; DEFAULT: lds.loop.epil:
+;
+; NOLOCAL-NOT: lds.loop.epil
+lds.loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %lds.loop ]
+ %idx = zext i32 %iv to i64
+ %ptr = getelementptr inbounds [256 x i32], ptr addrspace(3) @lds, i64 0, i64 %idx
+ store i32 %iv, ptr addrspace(3) %ptr, align 4
+ %iv.next = add nuw nsw i32 %iv, 1
+ %exitcond = icmp eq i32 %iv.next, %n
+ br i1 %exitcond, label %global.preheader, label %lds.loop
+
+global.preheader:
+ %cmp2 = icmp sgt i32 %m, 0
+ br i1 %cmp2, label %global.loop, label %exit
+
+; The global-memory loop: never touches LDS, so the knob must not affect it.
+; Its runtime-unroll epilogue block (global.loop.epil) appears under both
+; knob settings.
+;
+; DEFAULT: global.loop.epil:
+;
+; NOLOCAL: global.loop.epil:
+global.loop:
+ %jv = phi i32 [ 0, %global.preheader ], [ %jv.next, %global.loop ]
+ %jdx = zext i32 %jv to i64
+ %gptr = getelementptr inbounds i32, ptr addrspace(1) %out, i64 %jdx
+ store i32 %jv, ptr addrspace(1) %gptr, align 4
+ %jv.next = add nuw nsw i32 %jv, 1
+ %exitcond2 = icmp eq i32 %jv.next, %m
+ br i1 %exitcond2, label %exit, label %global.loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-runtime-local-mfma.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-runtime-local-mfma.ll
new file mode 100644
index 0000000000000..5a2d4b8c558b4
--- /dev/null
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-runtime-local-mfma.ll
@@ -0,0 +1,34 @@
+; RUN: opt -mtriple=amdgpu9.50-amd-amdhsa -passes=loop-unroll -S %s \
+; RUN: | FileCheck %s --check-prefixes=CHECK,DEFAULT
+; RUN: opt -mtriple=amdgpu9.50-amd-amdhsa -passes=loop-unroll \
+; RUN: -amdgpu-unroll-runtime-local=false -S %s | FileCheck %s --check-prefixes=CHECK,NOLOCAL
+
+; A convergent MFMA K-loop reading LDS (addrspace(3)) with a runtime trip
+; count, modeled on a triton bf16 GEMM inner loop. By default it is runtime
+; unrolled; with -amdgpu-unroll-runtime-local off the LDS loop is not.
+
+ at global_smem = external addrspace(3) global [0 x i8], align 16
+
+; CHECK-LABEL: @gemm_k_loop(
+; DEFAULT: loop.epil:
+; NOLOCAL-NOT: loop.epil
+define amdgpu_kernel void @gemm_k_loop(<8 x bfloat> %a, i32 %n) {
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc = phi <4 x float> [ zeroinitializer, %entry ], [ %mfma, %loop ]
+ %off = shl i32 %iv, 4
+ %ptr = getelementptr inbounds i8, ptr addrspace(3) @global_smem, i32 %off
+ %b = load <8 x bfloat>, ptr addrspace(3) %ptr, align 16
+ %mfma = tail call <4 x float> @llvm.amdgcn.mfma.f32.16x16x32.bf16(<8 x bfloat> %a, <8 x bfloat> %b, <4 x float> %acc, i32 0, i32 0, i32 0)
+ %iv.next = add nuw nsw i32 %iv, 1
+ %exit = icmp eq i32 %iv.next, %n
+ br i1 %exit, label %end, label %loop
+
+end:
+ ret void
+}
+
+declare <4 x float> @llvm.amdgcn.mfma.f32.16x16x32.bf16(<8 x bfloat>, <8 x bfloat>, <4 x float>, i32 immarg, i32 immarg, i32 immarg)
More information about the llvm-commits
mailing list