[flang-commits] [flang] [flang] Do not honor -fstack-arrays inside offload regions (PR #227537)
Zhen Wang via flang-commits
flang-commits at lists.llvm.org
Tue Sep 29 19:47:48 PDT 2026
https://github.com/wangzpgi updated https://github.com/llvm/llvm-project/pull/227537
>From c8fbc72535fb14f02c77dc596edc2126c6e011ab Mon Sep 17 00:00:00 2001
From: Zhen Wang <zhenw at nvidia.com>
Date: Tue, 29 Sep 2026 14:40:39 -0700
Subject: [PATCH 1/3] [flang] Do not honor -fstack-arrays inside offload
regions
---
.../Transforms/AllocationPlacement.cpp | 12 +-
.../lib/Optimizer/Transforms/StackArrays.cpp | 14 ++-
.../allocation-placement-offload-region.fir | 106 ++++++++++++++++++
.../Transforms/stack-arrays-alloca-scope.fir | 27 +++--
.../stack-arrays-offload-region.fir | 38 +++++++
5 files changed, 181 insertions(+), 16 deletions(-)
create mode 100644 flang/test/Transforms/allocation-placement-offload-region.fir
create mode 100644 flang/test/Transforms/stack-arrays-offload-region.fir
diff --git a/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp b/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp
index 1e6c7f900159d..de6028b92dae5 100644
--- a/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp
+++ b/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp
@@ -17,6 +17,7 @@
//===----------------------------------------------------------------------===//
#include "StackArrays.h"
+#include "flang/Optimizer/Builder/CUFCommon.h"
#include "flang/Optimizer/Dialect/FIRAttr.h"
#include "flang/Optimizer/Dialect/FIRDialect.h"
#include "flang/Optimizer/Dialect/FIROps.h"
@@ -207,12 +208,19 @@ void AllocationPlacementPass::runOnOperation() {
: (allocmem.hasLenParams() || allocmem.hasShapeOperands());
info.byteSize = getConstantByteSize(op, dl, kindMap);
+ // -fstack-arrays cannot be honored in an offload region either: like a
+ // device procedure, it runs on the device stack, which is far smaller than
+ // the host one. The size based part of the policy still applies.
+ fir::AllocationPolicy policy = basePolicy;
+ if (policy.stackArrays && cuf::isExecutingOnDevice(op))
+ policy.stackArrays = false;
+
// A hook, if provided, fully overrides the default policy; it may delegate
// back to decideAllocationPlacement after adjusting the policy.
fir::AllocationPlacement placement =
placementHook
- ? placementHook(info, basePolicy, stackBytesUsed)
- : fir::decideAllocationPlacement(info, basePolicy, stackBytesUsed);
+ ? placementHook(info, policy, stackBytesUsed)
+ : fir::decideAllocationPlacement(info, policy, stackBytesUsed);
// Account for the decision in the running stack budget.
if (endsUpOnStack(placement, info.isCurrentlyOnStack) && info.byteSize)
diff --git a/flang/lib/Optimizer/Transforms/StackArrays.cpp b/flang/lib/Optimizer/Transforms/StackArrays.cpp
index 7d4e6d49642d3..575c853b0bc21 100644
--- a/flang/lib/Optimizer/Transforms/StackArrays.cpp
+++ b/flang/lib/Optimizer/Transforms/StackArrays.cpp
@@ -7,6 +7,7 @@
//===----------------------------------------------------------------------===//
#include "StackArrays.h"
+#include "flang/Optimizer/Builder/CUFCommon.h"
#include "flang/Optimizer/Builder/FIRBuilder.h"
#include "flang/Optimizer/Builder/LowLevelIntrinsics.h"
#include "flang/Optimizer/Dialect/FIRAttr.h"
@@ -791,14 +792,17 @@ void StackArraysPass::runOnOperation() {
return;
}
- if (candidateOps->empty())
- return;
- runCount += candidateOps->size();
-
+ // An offload region runs on the device stack, which is far smaller than the
+ // host one, so its allocations stay on the heap like in a device procedure.
llvm::SmallVector<mlir::Operation *> opsToConvert;
opsToConvert.reserve(candidateOps->size());
for (auto [op, _] : *candidateOps)
- opsToConvert.push_back(op);
+ if (!cuf::isExecutingOnDevice(op))
+ opsToConvert.push_back(op);
+
+ if (opsToConvert.empty())
+ return;
+ runCount += opsToConvert.size();
mlir::MLIRContext &context = getContext();
mlir::RewritePatternSet patterns(&context);
diff --git a/flang/test/Transforms/allocation-placement-offload-region.fir b/flang/test/Transforms/allocation-placement-offload-region.fir
new file mode 100644
index 0000000000000..c0fd11c15b12e
--- /dev/null
+++ b/flang/test/Transforms/allocation-placement-offload-region.fir
@@ -0,0 +1,106 @@
+// Test that -fstack-arrays is not honored inside an offload region: the region
+// runs on the device stack, which is far smaller than the host one, so its
+// runtime-sized temporaries stay on the heap while the same allocation in host
+// code goes on the stack. The size based part of the policy still applies, so
+// a small constant-size temporary goes on the device stack.
+
+// RUN: fir-opt --allocation-placement %s | FileCheck %s
+
+module attributes {fir.allocation_policy =
+ #fir.allocation_policy<stack_arrays = true,
+ small_array_threshold = 1024,
+ total_stack_limit = 4194304>,
+ fir.defaultkind = "a1c4d8i4l4r4", fir.kindmap = "",
+ llvm.data_layout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"} {
+
+// CHECK-LABEL: func.func @dynamic_temp_in_host_code
+// CHECK: fir.alloca !fir.array<?xi32>, %{{.*}}
+// CHECK-NOT: fir.allocmem
+func.func @dynamic_temp_in_host_code(%n: index) {
+ %c0 = arith.constant 0 : index
+ %v = arith.constant 0 : i32
+ %0 = fir.allocmem !fir.array<?xi32>, %n
+ %r = fir.convert %0 : (!fir.heap<!fir.array<?xi32>>) -> !fir.ref<!fir.array<?xi32>>
+ %e = fir.coordinate_of %r, %c0 : (!fir.ref<!fir.array<?xi32>>, index) -> !fir.ref<i32>
+ fir.store %v to %e : !fir.ref<i32>
+ fir.freemem %0 : !fir.heap<!fir.array<?xi32>>
+ return
+}
+
+// CHECK-LABEL: func.func @dynamic_temp_in_acc_kernels
+// CHECK: acc.kernels {
+// CHECK-NEXT: fir.allocmem !fir.array<?xi32>, %{{.*}}
+// CHECK: fir.freemem
+// CHECK-NOT: fir.alloca
+func.func @dynamic_temp_in_acc_kernels(%n: index) {
+ %c0 = arith.constant 0 : index
+ %v = arith.constant 0 : i32
+ acc.kernels {
+ %0 = fir.allocmem !fir.array<?xi32>, %n
+ %r = fir.convert %0 : (!fir.heap<!fir.array<?xi32>>) -> !fir.ref<!fir.array<?xi32>>
+ %e = fir.coordinate_of %r, %c0 : (!fir.ref<!fir.array<?xi32>>, index) -> !fir.ref<i32>
+ fir.store %v to %e : !fir.ref<i32>
+ fir.freemem %0 : !fir.heap<!fir.array<?xi32>>
+ acc.terminator
+ }
+ return
+}
+
+// CHECK-LABEL: func.func @dynamic_temp_in_acc_parallel
+// CHECK: acc.parallel {
+// CHECK-NEXT: fir.allocmem !fir.array<?xi32>, %{{.*}}
+// CHECK: fir.freemem
+// CHECK-NOT: fir.alloca
+func.func @dynamic_temp_in_acc_parallel(%n: index) {
+ %c0 = arith.constant 0 : index
+ %v = arith.constant 0 : i32
+ acc.parallel {
+ %0 = fir.allocmem !fir.array<?xi32>, %n
+ %r = fir.convert %0 : (!fir.heap<!fir.array<?xi32>>) -> !fir.ref<!fir.array<?xi32>>
+ %e = fir.coordinate_of %r, %c0 : (!fir.ref<!fir.array<?xi32>>, index) -> !fir.ref<i32>
+ fir.store %v to %e : !fir.ref<i32>
+ fir.freemem %0 : !fir.heap<!fir.array<?xi32>>
+ acc.yield
+ }
+ return
+}
+
+// CHECK-LABEL: func.func @dynamic_temp_in_acc_serial
+// CHECK: acc.serial {
+// CHECK-NEXT: fir.allocmem !fir.array<?xi32>, %{{.*}}
+// CHECK: fir.freemem
+// CHECK-NOT: fir.alloca
+func.func @dynamic_temp_in_acc_serial(%n: index) {
+ %c0 = arith.constant 0 : index
+ %v = arith.constant 0 : i32
+ acc.serial {
+ %0 = fir.allocmem !fir.array<?xi32>, %n
+ %r = fir.convert %0 : (!fir.heap<!fir.array<?xi32>>) -> !fir.ref<!fir.array<?xi32>>
+ %e = fir.coordinate_of %r, %c0 : (!fir.ref<!fir.array<?xi32>>, index) -> !fir.ref<i32>
+ fir.store %v to %e : !fir.ref<i32>
+ fir.freemem %0 : !fir.heap<!fir.array<?xi32>>
+ acc.yield
+ }
+ return
+}
+
+// <10xi32> is 40 bytes, under the 1024 byte threshold: the size based policy
+// still puts it on the (device) stack.
+// CHECK-LABEL: func.func @small_temp_in_acc_parallel
+// CHECK: acc.parallel {
+// CHECK-NEXT: fir.alloca !fir.array<10xi32>
+// CHECK-NOT: fir.allocmem
+func.func @small_temp_in_acc_parallel() {
+ %c0 = arith.constant 0 : index
+ %v = arith.constant 0 : i32
+ acc.parallel {
+ %0 = fir.allocmem !fir.array<10xi32>
+ %r = fir.convert %0 : (!fir.heap<!fir.array<10xi32>>) -> !fir.ref<!fir.array<10xi32>>
+ %e = fir.coordinate_of %r, %c0 : (!fir.ref<!fir.array<10xi32>>, index) -> !fir.ref<i32>
+ fir.store %v to %e : !fir.ref<i32>
+ fir.freemem %0 : !fir.heap<!fir.array<10xi32>>
+ acc.yield
+ }
+ return
+}
+}
diff --git a/flang/test/Transforms/stack-arrays-alloca-scope.fir b/flang/test/Transforms/stack-arrays-alloca-scope.fir
index 342f09ccfaf6d..9debacd40dd03 100644
--- a/flang/test/Transforms/stack-arrays-alloca-scope.fir
+++ b/flang/test/Transforms/stack-arrays-alloca-scope.fir
@@ -1,9 +1,15 @@
-// RUN: fir-opt --stack-arrays %s | FileCheck %s
+// RUN: fir-opt --allocation-placement %s | FileCheck %s
// Test that stack allocations are created in the block where the enclosing
// construct expects them (fir::getAllocaBlock) and are not hoisted out of it:
// each concurrent execution of a construct modelling parallelism needs its own
// storage.
+//
+// -fstack-arrays is not honored inside offload regions, so the default size
+// based policy is used: the 168 byte <42xi32> temporaries go on the stack,
+// runtime-sized ones stay on the heap.
+
+module attributes {fir.defaultkind = "a1c4d8i4l4r4", fir.kindmap = "", llvm.data_layout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"} {
// The allocation has no operand but must still stay inside the compute
// construct instead of being hoisted to the function entry block.
@@ -63,8 +69,8 @@ func.func @acc_kernels_no_operand() {
// CHECK: acc.kernels {
// CHECK-NEXT: fir.alloca !fir.array<42xi32>
-// The extent operand is available before the compute construct: the allocation
-// can only be hoisted up to the beginning of the construct.
+// A runtime-sized allocation inside a compute construct stays on the heap, even
+// when its extent is available before the construct.
func.func @acc_parallel_operand_outside(%n: index) {
%c0 = arith.constant 0 : index
%c0_i32 = arith.constant 0 : i32
@@ -80,13 +86,13 @@ func.func @acc_parallel_operand_outside(%n: index) {
return
}
// CHECK-LABEL: func.func @acc_parallel_operand_outside(
-// CHECK: %[[SIZE:.*]] = arith.addi
// CHECK-NOT: fir.alloca
// CHECK: acc.parallel {
-// CHECK-NEXT: fir.alloca !fir.array<?xi32>, %[[SIZE]]
+// CHECK-NEXT: fir.allocmem !fir.array<?xi32>
+// CHECK: fir.freemem
+// CHECK-NOT: fir.alloca
-// The extent operand is defined inside the compute construct: the allocation is
-// placed right after it, as it would be in a plain function.
+// Same when the extent is defined inside the compute construct.
func.func @acc_parallel_operand_inside(%n: index) {
%c0 = arith.constant 0 : index
%c0_i32 = arith.constant 0 : i32
@@ -104,8 +110,10 @@ func.func @acc_parallel_operand_inside(%n: index) {
// CHECK-LABEL: func.func @acc_parallel_operand_inside(
// CHECK-NOT: fir.alloca
// CHECK: acc.parallel {
-// CHECK-NEXT: %[[SIZE:.*]] = arith.addi
-// CHECK-NEXT: fir.alloca !fir.array<?xi32>, %[[SIZE]]
+// CHECK-NEXT: arith.addi
+// CHECK-NEXT: fir.allocmem !fir.array<?xi32>
+// CHECK: fir.freemem
+// CHECK-NOT: fir.alloca
// Hoisting out of a sequential loop is still done, but only up to the beginning
// of the compute construct.
@@ -212,3 +220,4 @@ func.func @acc_data_no_operand() {
// CHECK-LABEL: func.func @acc_data_no_operand()
// CHECK: fir.alloca !fir.array<42xi32>
// CHECK: acc.data
+}
diff --git a/flang/test/Transforms/stack-arrays-offload-region.fir b/flang/test/Transforms/stack-arrays-offload-region.fir
new file mode 100644
index 0000000000000..d14b82b7ab921
--- /dev/null
+++ b/flang/test/Transforms/stack-arrays-offload-region.fir
@@ -0,0 +1,38 @@
+// Test that stack-arrays leaves the heap allocations of an offload region
+// alone: the region runs on the device stack, which is far smaller than the
+// host one. The same allocation in host code is moved to the stack.
+
+// RUN: fir-opt --stack-arrays %s | FileCheck %s
+
+// CHECK-LABEL: func.func @dynamic_temp_in_host_code
+// CHECK: fir.alloca !fir.array<?xi32>, %{{.*}}
+// CHECK-NOT: fir.allocmem
+func.func @dynamic_temp_in_host_code(%n: index) {
+ %c0 = arith.constant 0 : index
+ %v = arith.constant 0 : i32
+ %0 = fir.allocmem !fir.array<?xi32>, %n
+ %r = fir.convert %0 : (!fir.heap<!fir.array<?xi32>>) -> !fir.ref<!fir.array<?xi32>>
+ %e = fir.coordinate_of %r, %c0 : (!fir.ref<!fir.array<?xi32>>, index) -> !fir.ref<i32>
+ fir.store %v to %e : !fir.ref<i32>
+ fir.freemem %0 : !fir.heap<!fir.array<?xi32>>
+ return
+}
+
+// CHECK-LABEL: func.func @dynamic_temp_in_acc_parallel
+// CHECK: acc.parallel {
+// CHECK-NEXT: fir.allocmem !fir.array<?xi32>, %{{.*}}
+// CHECK: fir.freemem
+// CHECK-NOT: fir.alloca
+func.func @dynamic_temp_in_acc_parallel(%n: index) {
+ %c0 = arith.constant 0 : index
+ %v = arith.constant 0 : i32
+ acc.parallel {
+ %0 = fir.allocmem !fir.array<?xi32>, %n
+ %r = fir.convert %0 : (!fir.heap<!fir.array<?xi32>>) -> !fir.ref<!fir.array<?xi32>>
+ %e = fir.coordinate_of %r, %c0 : (!fir.ref<!fir.array<?xi32>>, index) -> !fir.ref<i32>
+ fir.store %v to %e : !fir.ref<i32>
+ fir.freemem %0 : !fir.heap<!fir.array<?xi32>>
+ acc.yield
+ }
+ return
+}
>From 9ab1e40bec0c8f6178c396cc486708952af67b2f Mon Sep 17 00:00:00 2001
From: Zhen Wang <zhenw at nvidia.com>
Date: Tue, 29 Sep 2026 18:39:04 -0700
Subject: [PATCH 2/3] add acc routine
---
.../Transforms/AllocationPlacement.cpp | 13 +++++++---
.../lib/Optimizer/Transforms/StackArrays.cpp | 6 ++++-
.../allocation-placement-offload-region.fir | 26 +++++++++++++++----
.../stack-arrays-offload-region.fir | 22 +++++++++++++---
4 files changed, 54 insertions(+), 13 deletions(-)
diff --git a/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp b/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp
index de6028b92dae5..9c61a074e9db0 100644
--- a/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp
+++ b/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp
@@ -32,6 +32,7 @@
#include "mlir/Dialect/DLTI/DLTI.h"
#include "mlir/Dialect/Func/IR/FuncOps.h"
#include "mlir/Dialect/LLVMIR/LLVMDialect.h"
+#include "mlir/Dialect/OpenACC/OpenACC.h"
#include "mlir/IR/Diagnostics.h"
#include "mlir/Pass/Pass.h"
#include "mlir/Transforms/GreedyPatternRewriteDriver.h"
@@ -172,6 +173,10 @@ void AllocationPlacementPass::runOnOperation() {
return;
}
+ // An acc routine is compiled for the device as well, so it is device code
+ // for the purpose of -fstack-arrays even before it is specialized.
+ bool isAccRoutine = mlir::acc::isAccRoutine(func);
+
// Walk allocations in deterministic program order, maintaining the running
// per-function stack budget while collecting the conversions to perform.
std::size_t stackBytesUsed = 0;
@@ -208,11 +213,11 @@ void AllocationPlacementPass::runOnOperation() {
: (allocmem.hasLenParams() || allocmem.hasShapeOperands());
info.byteSize = getConstantByteSize(op, dl, kindMap);
- // -fstack-arrays cannot be honored in an offload region either: like a
- // device procedure, it runs on the device stack, which is far smaller than
- // the host one. The size based part of the policy still applies.
+ // -fstack-arrays cannot be honored in an offload region or an acc routine
+ // either: like a device procedure, they run on the device stack, which is
+ // far smaller than the host one. The size based policy still applies.
fir::AllocationPolicy policy = basePolicy;
- if (policy.stackArrays && cuf::isExecutingOnDevice(op))
+ if (policy.stackArrays && (isAccRoutine || cuf::isExecutingOnDevice(op)))
policy.stackArrays = false;
// A hook, if provided, fully overrides the default policy; it may delegate
diff --git a/flang/lib/Optimizer/Transforms/StackArrays.cpp b/flang/lib/Optimizer/Transforms/StackArrays.cpp
index 575c853b0bc21..06cf3fff7bdcf 100644
--- a/flang/lib/Optimizer/Transforms/StackArrays.cpp
+++ b/flang/lib/Optimizer/Transforms/StackArrays.cpp
@@ -25,6 +25,7 @@
#include "mlir/Dialect/DLTI/DLTI.h"
#include "mlir/Dialect/Func/IR/FuncOps.h"
#include "mlir/Dialect/LLVMIR/LLVMDialect.h"
+#include "mlir/Dialect/OpenACC/OpenACC.h"
#include "mlir/Dialect/OpenMP/OpenMPDialect.h"
#include "mlir/IR/Builders.h"
#include "mlir/IR/Diagnostics.h"
@@ -778,11 +779,14 @@ void StackArraysPass::runOnOperation() {
// This pass only runs under -fstack-arrays, so honor a function that opted
// out in its own policy (device code, where the stack is tiny). Functions
- // without a policy of their own are left to the module setting.
+ // without a policy of their own are left to the module setting. An acc
+ // routine is compiled for the device as well and is treated the same way.
if (std::optional<fir::AllocationPolicy> policy =
fir::getLocalAllocationPolicy(func))
if (!policy->stackArrays)
return;
+ if (mlir::acc::isAccRoutine(func))
+ return;
auto &analysis = getAnalysis<fir::StackArraysAnalysisWrapper>();
const fir::StackArraysAnalysisWrapper::AllocMemMap *candidateOps =
diff --git a/flang/test/Transforms/allocation-placement-offload-region.fir b/flang/test/Transforms/allocation-placement-offload-region.fir
index c0fd11c15b12e..f53c90a242ed9 100644
--- a/flang/test/Transforms/allocation-placement-offload-region.fir
+++ b/flang/test/Transforms/allocation-placement-offload-region.fir
@@ -1,8 +1,8 @@
-// Test that -fstack-arrays is not honored inside an offload region: the region
-// runs on the device stack, which is far smaller than the host one, so its
-// runtime-sized temporaries stay on the heap while the same allocation in host
-// code goes on the stack. The size based part of the policy still applies, so
-// a small constant-size temporary goes on the device stack.
+// Test that -fstack-arrays is not honored inside an offload region or an acc
+// routine: they run on the device stack, which is far smaller than the host
+// one, so their runtime-sized temporaries stay on the heap while the same
+// allocation in host code goes on the stack. The size based part of the policy
+// still applies, so a small constant-size temporary goes on the device stack.
// RUN: fir-opt --allocation-placement %s | FileCheck %s
@@ -103,4 +103,20 @@ func.func @small_temp_in_acc_parallel() {
}
return
}
+
+// An acc routine is compiled for the device as well: its allocations stay on
+// the heap even before it is specialized.
+// CHECK-LABEL: func.func @dynamic_temp_in_acc_routine
+// CHECK: fir.allocmem !fir.array<?xi32>, %{{.*}}
+// CHECK-NOT: fir.alloca
+func.func @dynamic_temp_in_acc_routine(%n: index) attributes {acc.routine_info = #acc.routine_info<[@acc_routine_0]>} {
+ %c0 = arith.constant 0 : index
+ %v = arith.constant 0 : i32
+ %0 = fir.allocmem !fir.array<?xi32>, %n
+ %r = fir.convert %0 : (!fir.heap<!fir.array<?xi32>>) -> !fir.ref<!fir.array<?xi32>>
+ %e = fir.coordinate_of %r, %c0 : (!fir.ref<!fir.array<?xi32>>, index) -> !fir.ref<i32>
+ fir.store %v to %e : !fir.ref<i32>
+ fir.freemem %0 : !fir.heap<!fir.array<?xi32>>
+ return
+}
}
diff --git a/flang/test/Transforms/stack-arrays-offload-region.fir b/flang/test/Transforms/stack-arrays-offload-region.fir
index d14b82b7ab921..45ca5d76bf93a 100644
--- a/flang/test/Transforms/stack-arrays-offload-region.fir
+++ b/flang/test/Transforms/stack-arrays-offload-region.fir
@@ -1,6 +1,6 @@
-// Test that stack-arrays leaves the heap allocations of an offload region
-// alone: the region runs on the device stack, which is far smaller than the
-// host one. The same allocation in host code is moved to the stack.
+// Test that stack-arrays leaves the heap allocations of an offload region and
+// of an acc routine alone: they run on the device stack, which is far smaller
+// than the host one. The same allocation in host code is moved to the stack.
// RUN: fir-opt --stack-arrays %s | FileCheck %s
@@ -36,3 +36,19 @@ func.func @dynamic_temp_in_acc_parallel(%n: index) {
}
return
}
+
+// An acc routine is compiled for the device as well: its allocations stay on
+// the heap even before it is specialized.
+// CHECK-LABEL: func.func @dynamic_temp_in_acc_routine
+// CHECK: fir.allocmem !fir.array<?xi32>, %{{.*}}
+// CHECK-NOT: fir.alloca
+func.func @dynamic_temp_in_acc_routine(%n: index) attributes {acc.routine_info = #acc.routine_info<[@acc_routine_0]>} {
+ %c0 = arith.constant 0 : index
+ %v = arith.constant 0 : i32
+ %0 = fir.allocmem !fir.array<?xi32>, %n
+ %r = fir.convert %0 : (!fir.heap<!fir.array<?xi32>>) -> !fir.ref<!fir.array<?xi32>>
+ %e = fir.coordinate_of %r, %c0 : (!fir.ref<!fir.array<?xi32>>, index) -> !fir.ref<i32>
+ fir.store %v to %e : !fir.ref<i32>
+ fir.freemem %0 : !fir.heap<!fir.array<?xi32>>
+ return
+}
>From 1b7108ff46324794de234bb2cb36d3dadbab5900 Mon Sep 17 00:00:00 2001
From: Zhen Wang <zhenw at nvidia.com>
Date: Tue, 29 Sep 2026 19:47:28 -0700
Subject: [PATCH 3/3] Fold the acc routine check into the base allocation
policy
---
.../Optimizer/Transforms/AllocationPlacement.cpp | 16 ++++++++--------
1 file changed, 8 insertions(+), 8 deletions(-)
diff --git a/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp b/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp
index 9c61a074e9db0..10f7f615e8abb 100644
--- a/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp
+++ b/flang/lib/Optimizer/Transforms/AllocationPlacement.cpp
@@ -154,6 +154,10 @@ void AllocationPlacementPass::runOnOperation() {
smallArrayThresholdBytes);
fir::overrideIfExplicitlySet(basePolicy.totalStackLimitBytes,
totalStackLimitBytes);
+ // An acc routine is compiled for the device as well, so -fstack-arrays
+ // cannot be honored in it, as in a device procedure.
+ if (mlir::acc::isAccRoutine(func))
+ basePolicy.stackArrays = false;
auto module = func->getParentOfType<mlir::ModuleOp>();
std::optional<mlir::DataLayout> dl =
@@ -173,10 +177,6 @@ void AllocationPlacementPass::runOnOperation() {
return;
}
- // An acc routine is compiled for the device as well, so it is device code
- // for the purpose of -fstack-arrays even before it is specialized.
- bool isAccRoutine = mlir::acc::isAccRoutine(func);
-
// Walk allocations in deterministic program order, maintaining the running
// per-function stack budget while collecting the conversions to perform.
std::size_t stackBytesUsed = 0;
@@ -213,11 +213,11 @@ void AllocationPlacementPass::runOnOperation() {
: (allocmem.hasLenParams() || allocmem.hasShapeOperands());
info.byteSize = getConstantByteSize(op, dl, kindMap);
- // -fstack-arrays cannot be honored in an offload region or an acc routine
- // either: like a device procedure, they run on the device stack, which is
- // far smaller than the host one. The size based policy still applies.
+ // -fstack-arrays cannot be honored in an offload region either: like a
+ // device procedure, it runs on the device stack, which is far smaller than
+ // the host one. The size based part of the policy still applies.
fir::AllocationPolicy policy = basePolicy;
- if (policy.stackArrays && (isAccRoutine || cuf::isExecutingOnDevice(op)))
+ if (policy.stackArrays && cuf::isExecutingOnDevice(op))
policy.stackArrays = false;
// A hook, if provided, fully overrides the default policy; it may delegate
More information about the flang-commits
mailing list