[Mlir-commits] [mlir] [openacc] Attach Parallelism Levels to Auto Loops (PR #200884)

Delaram Talaashrafi llvmlistbot at llvm.org
Mon Jun 1 10:55:28 PDT 2026


https://github.com/delaram-talaashrafi updated https://github.com/llvm/llvm-project/pull/200884

>From 7df55e9452c631002a16afee8edb427af252ef46 Mon Sep 17 00:00:00 2001
From: Delaram Talaashrafi <dtalaashrafi at rome5.pgi.net>
Date: Mon, 1 Jun 2026 10:38:20 -0700
Subject: [PATCH] [openacc] Attach Parallelism Levels to Auto Loops

Auto loops are analyzed by the compiler in later compilation stages to determine
whether they can be parallelized. These loops may carry a parallelism level, but
this does not guarantee that they are parallelizable (the compiler should still
analyze them). However, if a loop is parallelized, the parallelism level specified
in the source should be respected. This change attaches the parallelism level to
auto loops and enables its propagation through subsequent compilation steps.
---
 .../OpenACC/Transforms/ACCComputeLowering.cpp |  7 +++++
 .../OpenACC/acc-compute-lowering-loop.mlir    | 29 +++++++++++++++++++
 2 files changed, 36 insertions(+)

diff --git a/mlir/lib/Dialect/OpenACC/Transforms/ACCComputeLowering.cpp b/mlir/lib/Dialect/OpenACC/Transforms/ACCComputeLowering.cpp
index 9868584a4b699..80b4570587a17 100644
--- a/mlir/lib/Dialect/OpenACC/Transforms/ACCComputeLowering.cpp
+++ b/mlir/lib/Dialect/OpenACC/Transforms/ACCComputeLowering.cpp
@@ -314,6 +314,13 @@ class ACCLoopConversion : public OpRewritePattern<LoopOp> {
           convertACCLoopToSCFFor(loopOp, rewriter, /*enableCollapse=*/true);
       if (!forOp)
         return failure();
+      SmallVector<GPUParallelDimAttr> parDims =
+          getParallelDimensions(loopOp, policy, deviceType);
+      if (!parDims.empty()) {
+        auto parDimsAttr =
+            GPUParallelDimsAttr::get(loopOp->getContext(), parDims);
+        setParDimsAttr(forOp, parDimsAttr);
+      }
       rewriter.replaceOp(loopOp, forOp);
     } else if (!isOpInComputeRegion(loopOp) &&
                !isSpecializedAccRoutine(
diff --git a/mlir/test/Dialect/OpenACC/acc-compute-lowering-loop.mlir b/mlir/test/Dialect/OpenACC/acc-compute-lowering-loop.mlir
index 4032fed217b62..358d1328fa028 100644
--- a/mlir/test/Dialect/OpenACC/acc-compute-lowering-loop.mlir
+++ b/mlir/test/Dialect/OpenACC/acc-compute-lowering-loop.mlir
@@ -174,3 +174,32 @@ func.func @device_routine_vector_with_loop(%buf: memref<8xi32>) attributes {acc.
   } attributes {independent = [#acc.device_type<none>], vector = [#acc.device_type<none>]}
   return
 }
+
+// -----
+
+// Auto loop with gang: lowered to scf.for with predetermined par_dims.
+// CHECK-LABEL: func.func @parallel_loop_auto_gang
+func.func @parallel_loop_auto_gang(%buf: memref<1xi32>) {
+  %c0 = arith.constant 0 : index
+  %c1_i32 = arith.constant 1 : i32
+  %c10_i32 = arith.constant 10 : i32
+  %c100_i32 = arith.constant 100 : i32
+
+  %dev = acc.copyin varPtr(%buf : memref<1xi32>) -> memref<1xi32>
+  // CHECK-NOT: acc.parallel
+  // CHECK: acc.kernel_environment
+  // CHECK: acc.par_width {{.*}} {par_dim = #acc.par_dim<block_x>}
+  // CHECK: acc.compute_region launch(
+  // CHECK: scf.for
+  // CHECK-NOT: scf.parallel
+  // CHECK: acc.par_dims = #acc<par_dims[block_x]>
+  acc.parallel num_gangs({%c10_i32 : i32}) dataOperands(%dev : memref<1xi32>) {
+    acc.loop gang control(%arg0 : i32) = (%c1_i32 : i32) to (%c100_i32 : i32) step (%c1_i32 : i32) {
+      memref.store %arg0, %dev[%c0] : memref<1xi32>
+      acc.yield
+    } attributes {auto_ = [#acc.device_type<none>]}
+    acc.yield
+  }
+  acc.copyout accPtr(%dev : memref<1xi32>) to varPtr(%buf : memref<1xi32>)
+  return
+}



More information about the Mlir-commits mailing list