[flang-commits] [flang] cb9ee31 - Reland [flang] Support scoped LICM and OpenACC capture provenance (#225401) (#226909)
via flang-commits
flang-commits at lists.llvm.org
Mon Sep 28 01:09:19 PDT 2026
Author: jeanPerier
Date: 2026-09-28T08:09:10Z
New Revision: cb9ee31ad93d35af0ee9172cd368dd537546a8d6
URL: https://github.com/llvm/llvm-project/commit/cb9ee31ad93d35af0ee9172cd368dd537546a8d6
DIFF: https://github.com/llvm/llvm-project/commit/cb9ee31ad93d35af0ee9172cd368dd537546a8d6.diff
LOG: Reland [flang] Support scoped LICM and OpenACC capture provenance (#225401) (#226909)
Follow compute-region capture operands when checking whether scalar and
scalar-descriptor loads are safe to speculate. Preserve the existing
optional, array-element, and loop-modification safety checks.
Add an optional only-inside operation-name selector while retaining
function-scoped alias analysis. Resolve the name once per function and
compare interned operation names during ancestor traversal. The default
continues to select all loops. An explicit name selects loops with a
matching ancestor within the function, including the function itself.
Cover capture safety, host exclusion, non-OpenACC and nested scopes,
loop boundaries, unmatched names, and the function boundary.
The motivation for this is to allow running LICM only on device relevant
loop at O0.
Reland #225401 with CMakeFiles.txt change to fix shared library builds.
Added:
flang/test/Transforms/licm-acc-captured-ref.fir
flang/test/Transforms/licm-acc-compute-only.fir
flang/test/Transforms/licm-scope.fir
Modified:
flang/include/flang/Optimizer/Transforms/Passes.td
flang/lib/Optimizer/Transforms/CMakeLists.txt
flang/lib/Optimizer/Transforms/LoopInvariantCodeMotion.cpp
Removed:
################################################################################
diff --git a/flang/include/flang/Optimizer/Transforms/Passes.td b/flang/include/flang/Optimizer/Transforms/Passes.td
index 3034fbf9f386b..51f97efe02885 100644
--- a/flang/include/flang/Optimizer/Transforms/Passes.td
+++ b/flang/include/flang/Optimizer/Transforms/Passes.td
@@ -833,6 +833,11 @@ def LoopInvariantCodeMotion : Pass<"flang-licm", "::mlir::func::FuncOp"> {
fir.declare) and interfaces such as FortranObjectViewOpInterface.
The pass only moves existing operations, so there are no dependent
dialects.
+
+ With only-inside set, a loop must have an ancestor with the given operation
+ name, within the current function (including the function itself). A loop
+ matching the name does not qualify on that basis alone. No matching ancestor
+ means no hoisting. The pass and its analyses remain rooted at the function.
}];
let options = [Option<"hoistFromNestedRegions", "hoist-from-nested-regions",
"::fir::LICMNestedHoistingMode",
@@ -848,7 +853,12 @@ def LoopInvariantCodeMotion : Pass<"flang-licm", "::mlir::func::FuncOp"> {
"loads"),
clEnumValN(::fir::LICMNestedHoistingMode::Aggressive,
"aggressive", "Hoist all safe invariant ops")
- )}]>];
+ )}]>,
+ Option<"onlyInside", "only-inside", "std::string",
+ /*default=*/"std::string{}",
+ "Only hoist from loops nested in an operation with this name "
+ "(empty means all loops)">
+ ];
}
#endif // FLANG_OPTIMIZER_TRANSFORMS_PASSES
diff --git a/flang/lib/Optimizer/Transforms/CMakeLists.txt b/flang/lib/Optimizer/Transforms/CMakeLists.txt
index 7fb77ff1ed224..cb70854eec128 100644
--- a/flang/lib/Optimizer/Transforms/CMakeLists.txt
+++ b/flang/lib/Optimizer/Transforms/CMakeLists.txt
@@ -93,6 +93,7 @@ add_flang_library(FIRTransforms
MLIRMathTransforms
MLIRNVVMDialect
MLIROpenACCDialect
+ MLIROpenACCUtils
MLIROpenMPDialect
MLIRTransformUtils
)
diff --git a/flang/lib/Optimizer/Transforms/LoopInvariantCodeMotion.cpp b/flang/lib/Optimizer/Transforms/LoopInvariantCodeMotion.cpp
index 4003f1cb30b18..f76c8deba5204 100644
--- a/flang/lib/Optimizer/Transforms/LoopInvariantCodeMotion.cpp
+++ b/flang/lib/Optimizer/Transforms/LoopInvariantCodeMotion.cpp
@@ -19,11 +19,13 @@
#include "flang/Optimizer/HLFIR/HLFIROps.h"
#include "flang/Optimizer/Support/Utils.h"
#include "flang/Optimizer/Transforms/Passes.h"
+#include "mlir/Dialect/OpenACC/OpenACCUtils.h"
#include "mlir/Interfaces/LoopLikeInterface.h"
#include "mlir/Pass/Pass.h"
#include "mlir/Transforms/LoopInvariantCodeMotionUtils.h"
#include "llvm/ADT/TypeSwitch.h"
#include "llvm/Support/DebugLog.h"
+#include <optional>
#include <utility>
namespace fir {
@@ -90,6 +92,12 @@ static bool isNonOptionalScalar(Value location) {
}
Operation *defOp = location.getDefiningOp();
if (!defOp) {
+ // A compute-region argument forwards a mapped input. Recover its storage
+ // provenance before checking whether a speculative scalar read is safe.
+ if (Value operand = acc::getACCOperandForBlockArg(location)) {
+ location = operand;
+ continue;
+ }
// If this is a function argument
auto blockArg = cast<BlockArgument>(location);
Block *block = blockArg.getOwner();
@@ -374,7 +382,22 @@ void LoopInvariantCodeMotion::runOnOperation() {
maybeConditionallyExecuted);
};
- getOperation()->walk([&](LoopLikeOpInterface loopLike) {
+ // Resolve the name once: ancestor checks compare interned operation names,
+ // not strings. Keep analysis scope independent of the selected loop scope.
+ std::optional<OperationName> scopeOpName;
+ if (!onlyInside.empty())
+ scopeOpName.emplace(onlyInside, &getContext());
+ Operation *function = getOperation();
+ function->walk([&](LoopLikeOpInterface loopLike) {
+ if (scopeOpName) {
+ Operation *scope = loopLike->getParentOp();
+ while (scope != function && scope->getName() != *scopeOpName)
+ scope = scope->getParentOp();
+ if (scope->getName() != *scopeOpName) {
+ LDBG() << "Skipping loop-like without " << *scopeOpName << " parent";
+ return;
+ }
+ }
if (!fir::canMoveOutOf(loopLike, nullptr)) {
LDBG() << "Cannot hoist anything out of loop operation: ";
LDBG_OS([&](llvm::raw_ostream &os) {
diff --git a/flang/test/Transforms/licm-acc-captured-ref.fir b/flang/test/Transforms/licm-acc-captured-ref.fir
new file mode 100644
index 0000000000000..0ec3be41ab2f7
--- /dev/null
+++ b/flang/test/Transforms/licm-acc-captured-ref.fir
@@ -0,0 +1,137 @@
+// RUN: fir-opt %s --flang-licm | FileCheck %s
+
+// CHECK-LABEL: func.func @captured_descriptor
+// CHECK: acc.compute_region
+// CHECK: fir.load
+// CHECK: scf.for
+// CHECK-NOT: fir.load
+// CHECK: acc.yield
+func.func @captured_descriptor(%x: !fir.ref<!fir.box<!fir.heap<!fir.array<?xf32>>>>, %n: index) {
+ acc.kernel_environment {
+ acc.compute_region ins(%ref = %x, %bound = %n) : (!fir.ref<!fir.box<!fir.heap<!fir.array<?xf32>>>>, index) {
+ %c0 = arith.constant 0 : index
+ %c1 = arith.constant 1 : index
+ scf.for %i = %c0 to %bound step %c1 {
+ %descriptor = fir.load %ref : !fir.ref<!fir.box<!fir.heap<!fir.array<?xf32>>>>
+ }
+ acc.yield
+ } <{origin = "acc.kernels"}>
+ }
+ return
+}
+
+// CHECK-LABEL: func.func @captured_optional_descriptor
+// CHECK: acc.compute_region
+// CHECK-NOT: fir.load
+// CHECK: scf.for
+// CHECK: fir.load
+// CHECK: acc.yield
+func.func @captured_optional_descriptor(%x: !fir.ref<!fir.box<!fir.heap<!fir.array<?xf32>>>> {fir.optional}, %n: index) {
+ acc.kernel_environment {
+ acc.compute_region ins(%ref = %x, %bound = %n) : (!fir.ref<!fir.box<!fir.heap<!fir.array<?xf32>>>>, index) {
+ %c0 = arith.constant 0 : index
+ %c1 = arith.constant 1 : index
+ scf.for %i = %c0 to %bound step %c1 {
+ %descriptor = fir.load %ref : !fir.ref<!fir.box<!fir.heap<!fir.array<?xf32>>>>
+ }
+ acc.yield
+ } <{origin = "acc.kernels"}>
+ }
+ return
+}
+
+// Capturing a reference must preserve the ability to prove that loading a
+// non-optional scalar is safe even when the loop might execute zero times.
+// CHECK-LABEL: func.func @captured_scalar
+// CHECK: acc.compute_region
+// CHECK: fir.load
+// CHECK: scf.for
+// CHECK-NOT: fir.load
+// CHECK: acc.yield
+func.func @captured_scalar(%x: !fir.ref<i32>, %n: index) {
+ acc.kernel_environment {
+ acc.compute_region ins(%ref = %x, %bound = %n) : (!fir.ref<i32>, index) {
+ %c0 = arith.constant 0 : index
+ %c1 = arith.constant 1 : index
+ %zero = arith.constant 0 : i32
+ %result = scf.for %i = %c0 to %bound step %c1 iter_args(%previous = %zero) -> i32 {
+ %value = fir.load %ref : !fir.ref<i32>
+ scf.yield %value : i32
+ }
+ acc.yield
+ } <{origin = "acc.kernels"}>
+ }
+ return
+}
+
+// CHECK-LABEL: func.func @captured_optional
+// CHECK: acc.compute_region
+// CHECK-NOT: fir.load
+// CHECK: scf.for
+// CHECK: fir.load
+// CHECK: acc.yield
+// Capture forwarding must retain OPTIONAL, not treat every input as present.
+func.func @captured_optional(%x: !fir.ref<i32> {fir.optional}, %n: index) {
+ acc.kernel_environment {
+ acc.compute_region ins(%ref = %x, %bound = %n) : (!fir.ref<i32>, index) {
+ %c0 = arith.constant 0 : index
+ %c1 = arith.constant 1 : index
+ %zero = arith.constant 0 : i32
+ %result = scf.for %i = %c0 to %bound step %c1 iter_args(%previous = %zero) -> i32 {
+ %value = fir.load %ref : !fir.ref<i32>
+ scf.yield %value : i32
+ }
+ acc.yield
+ } <{origin = "acc.kernels"}>
+ }
+ return
+}
+
+// CHECK-LABEL: func.func @captured_element
+// CHECK: acc.compute_region
+// CHECK-NOT: fir.load
+// CHECK: scf.for
+// CHECK: fir.load
+// CHECK: acc.yield
+// A captured scalar reference can designate an array element whose subscript
+// is not known to be valid. Its type alone does not justify speculation.
+func.func @captured_element(%x: !fir.ref<!fir.array<10xi32>>, %index: index, %n: index) {
+ %element = fir.coordinate_of %x, %index : (!fir.ref<!fir.array<10xi32>>, index) -> !fir.ref<i32>
+ acc.kernel_environment {
+ acc.compute_region ins(%ref = %element, %bound = %n) : (!fir.ref<i32>, index) {
+ %c0 = arith.constant 0 : index
+ %c1 = arith.constant 1 : index
+ %zero = arith.constant 0 : i32
+ %result = scf.for %i = %c0 to %bound step %c1 iter_args(%previous = %zero) -> i32 {
+ %value = fir.load %ref : !fir.ref<i32>
+ scf.yield %value : i32
+ }
+ acc.yield
+ } <{origin = "acc.kernels"}>
+ }
+ return
+}
+
+// CHECK-LABEL: func.func @captured_modified
+// CHECK: acc.compute_region
+// CHECK-NOT: fir.load
+// CHECK: scf.for
+// CHECK: fir.load
+// CHECK: acc.yield
+// Speculative-read safety does not replace the loop's mod/ref check.
+func.func @captured_modified(%x: !fir.ref<i32>, %n: index) {
+ acc.kernel_environment {
+ acc.compute_region ins(%ref = %x, %bound = %n) : (!fir.ref<i32>, index) {
+ %c0 = arith.constant 0 : index
+ %c1 = arith.constant 1 : index
+ %zero = arith.constant 0 : i32
+ %result = scf.for %i = %c0 to %bound step %c1 iter_args(%previous = %zero) -> i32 {
+ %value = fir.load %ref : !fir.ref<i32>
+ fir.store %zero to %ref : !fir.ref<i32>
+ scf.yield %value : i32
+ }
+ acc.yield
+ } <{origin = "acc.kernels"}>
+ }
+ return
+}
diff --git a/flang/test/Transforms/licm-acc-compute-only.fir b/flang/test/Transforms/licm-acc-compute-only.fir
new file mode 100644
index 0000000000000..acdb3f811af82
--- /dev/null
+++ b/flang/test/Transforms/licm-acc-compute-only.fir
@@ -0,0 +1,44 @@
+// RUN: fir-opt %s --flang-licm="only-inside=acc.compute_region" | FileCheck %s
+// RUN: fir-opt %s --flang-licm | FileCheck %s --check-prefix=GLOBAL
+
+// Filtering must retain host loads even in a loop enclosing a compute region,
+// and find device loops beneath host conditionals. The default is unchanged.
+// CHECK-LABEL: func.func @mixed
+// CHECK: scf.for
+// CHECK: fir.load
+// CHECK: scf.if
+// CHECK: acc.compute_region
+// CHECK: fir.load
+// CHECK: scf.for
+// CHECK-NOT: fir.load
+// CHECK: acc.yield
+// CHECK: } else {
+// CHECK: scf.for
+// CHECK: fir.load
+// GLOBAL-LABEL: func.func @mixed
+// GLOBAL: fir.load
+// GLOBAL: scf.for
+func.func @mixed(%x: !fir.ref<i32>, %n: index, %condition: i1) {
+ %c0 = arith.constant 0 : index
+ %c1 = arith.constant 1 : index
+ scf.for %h = %c0 to %n step %c1 {
+ %host = fir.load %x : !fir.ref<i32>
+ scf.if %condition {
+ acc.kernel_environment {
+ acc.compute_region ins(%ref = %x, %bound = %n) : (!fir.ref<i32>, index) {
+ %zero = arith.constant 0 : index
+ %one = arith.constant 1 : index
+ scf.for %i = %zero to %bound step %one {
+ %device = fir.load %ref : !fir.ref<i32>
+ }
+ acc.yield
+ } <{origin = "acc.kernels"}>
+ }
+ } else {
+ scf.for %i = %c0 to %n step %c1 {
+ %fallback = fir.load %x : !fir.ref<i32>
+ }
+ }
+ }
+ return
+}
diff --git a/flang/test/Transforms/licm-scope.fir b/flang/test/Transforms/licm-scope.fir
new file mode 100644
index 0000000000000..84eeda62dc687
--- /dev/null
+++ b/flang/test/Transforms/licm-scope.fir
@@ -0,0 +1,39 @@
+// RUN: fir-opt %s --flang-licm="only-inside=scf.execute_region" | FileCheck %s --check-prefix=SCOPED
+// RUN: fir-opt %s --flang-licm="only-inside=scf.for" | FileCheck %s --check-prefix=SCOPED
+// RUN: fir-opt %s --flang-licm > %t.default
+// RUN: fir-opt %s --flang-licm="only-inside=func.func" > %t.function
+// RUN:
diff %t.default %t.function
+// RUN: fir-opt %s > %t.input
+// RUN: fir-opt %s --flang-licm="only-inside=scf.if" > %t.unmatched
+// RUN:
diff %t.input %t.unmatched
+// RUN: fir-opt %s --flang-licm="only-inside=builtin.module" > %t.outside-function
+// RUN:
diff %t.input %t.outside-function
+
+// A matching region can be nested under host control flow. Nested matching
+// regions are supported; selecting a loop name does not select that loop itself.
+// SCOPED-LABEL: func.func @nested
+// SCOPED: scf.for
+// SCOPED: arith.addi
+// SCOPED: scf.execute_region
+// SCOPED: scf.execute_region
+// SCOPED: arith.muli
+// SCOPED: scf.for
+// SCOPED-NOT: arith.muli
+// SCOPED: scf.yield
+func.func @nested(%n: index) {
+ %c0 = arith.constant 0 : index
+ %c1 = arith.constant 1 : index
+ scf.for %outer = %c0 to %n step %c1 {
+ %host = arith.addi %n, %n : index
+ scf.execute_region {
+ scf.execute_region {
+ scf.for %inner = %c0 to %n step %c1 {
+ %selected = arith.muli %n, %n : index
+ }
+ scf.yield
+ }
+ scf.yield
+ }
+ }
+ return
+}
More information about the flang-commits
mailing list