[flang-commits] [flang] [flang][cuda] Keep data transfers for allocatable assignments with device data (PR #212855)
Zhen Wang via flang-commits
flang-commits at lists.llvm.org
Wed Jul 29 12:56:36 PDT 2026
https://github.com/wangzpgi updated https://github.com/llvm/llvm-project/pull/212855
>From 1b8582b24323ec05a5fc4e121b19e67324ed08b8 Mon Sep 17 00:00:00 2001
From: Zhen Wang <zhenw at nvidia.com>
Date: Wed, 29 Jul 2026 11:31:57 -0700
Subject: [PATCH 1/2] Keep data transfers for allocatable assignments with
device data
---
flang/include/flang/Evaluate/tools.h | 65 ++++++++++++-------
flang/test/Lower/CUDA/cuda-managed-assign.cuf | 17 +++++
2 files changed, 58 insertions(+), 24 deletions(-)
diff --git a/flang/include/flang/Evaluate/tools.h b/flang/include/flang/Evaluate/tools.h
index 4c4f7912a506a..bf711050a5457 100644
--- a/flang/include/flang/Evaluate/tools.h
+++ b/flang/include/flang/Evaluate/tools.h
@@ -1401,8 +1401,20 @@ template <typename A> inline bool IsWholeManagedArray(const A &expr) {
return expr.Rank() > 0 && sym && IsCUDAManagedOrUnifiedSymbol(*sym);
}
-// Check if any of the symbols part of the lhs or rhs expression has a CUDA
-// device attribute.
+// CUDA Fortran Programming Guide 3.4.1 defines which assignments in host code
+// are copies. A copy that reads or writes device, managed or constant data runs
+// on stream zero, so it waits for previously launched kernels.
+// - Device or constant data on one side and host data on the other is a copy,
+// and so is device data on both sides.
+// - A whole managed variable or array is copied when the other side is a
+// constant, a host variable, a host array or a host array section.
+// - A managed array section is assigned by host code when the other side is
+// host or managed data.
+// - A managed variable, array or array section is copied when the other side is
+// device data, in both directions.
+// One difference from the guide is that a managed array section is copied when
+// the other side is a whole managed array, as the reference compiler does.
+// Return true if the assignment is one of the copies above.
template <typename A, typename B>
inline bool IsCUDADataTransfer(const A &lhs, const B &rhs) {
int lhsNbManagedSymbols{GetNbOfCUDAManagedOrUnifiedSymbols(lhs)};
@@ -1412,35 +1424,40 @@ inline bool IsCUDADataTransfer(const A &lhs, const B &rhs) {
if (HasNonAllocatableModuleCUDAManagedSymbols(lhs))
return false;
- // CUDA Fortran Programming Guide 3.4.1: an assignment involving managed or
- // unified data is a copy when the managed side is a whole variable or array,
- // and is done on the host when it is a section. The host can read and write
- // managed data directly, and copying section by section in a loop is slow.
+ // The host can read and write managed data in place, and copying one section
+ // at a time in a loop is slow, so only whole arrays are copied.
bool wholeLhs{IsWholeManagedArray(lhs)};
bool wholeRhs{IsWholeManagedArray(rhs)};
if (wholeLhs && rhsNbSymbols == 0 && rhsNbManagedSymbols == 0 &&
(IsVariable(rhs) || IsConstantExpr(rhs))) {
- return true; // Initializing a whole managed array is done on the device.
- }
-
- // Assignments done on the host, with no copy:
- // - A whole allocatable left-hand side: the assignment may reallocate it,
- // which is done on the host.
- // - No managed operand is a whole array: they are all sections or elements,
- // which the host can read and write in place.
- // - An expression involving managed data assigned to a managed or host
- // left-hand side: evaluating it on the host avoids a temporary.
- // - A managed left-hand side assigned from host-only data.
- if ((IsAllocatableDesignator(lhs) &&
+ return true; // Whole managed array copied from constant or host data.
+ }
+
+ // The host cannot reach device or constant data, unlike managed and unified
+ // data, so an assignment with such a side is a copy, sections included.
+ bool lhsIsDeviceOnly{lhsNbManagedSymbols == 0 && HasCUDADeviceAttrs(lhs)};
+ // The right-hand side can be an expression, so one device operand is enough.
+ bool rhsHasDeviceOnly{rhsNbSymbols > rhsNbManagedSymbols};
+
+ // Assignments done on the host, with no copy.
+ // - A whole allocatable left-hand side with no device data. The assignment may
+ // reallocate it, which is done on the host.
+ // - A managed left-hand side with no whole managed array on either side. Only
+ // sections and elements are involved, and the host reads and writes them in
+ // place.
+ // - A host left-hand side assigned from a managed section or element.
+ // - An expression involving managed data. Evaluating it on the host avoids a
+ // temporary.
+ // - A managed left-hand side assigned from host data. Whole arrays are copied
+ // by the early return above.
+ if ((IsAllocatableDesignator(lhs) && !lhsIsDeviceOnly && !rhsHasDeviceOnly &&
(lhsNbManagedSymbols >= 1 || rhsNbManagedSymbols >= 1)) ||
- (lhsNbManagedSymbols >= 1 && rhsNbManagedSymbols == rhsNbSymbols &&
+ (lhsNbManagedSymbols >= 1 && !rhsHasDeviceOnly &&
!(wholeLhs || wholeRhs)) ||
- (lhsNbManagedSymbols == 0 && !HasCUDADeviceAttrs(lhs) &&
- rhsNbManagedSymbols >= 1 && rhsNbManagedSymbols == rhsNbSymbols &&
- !wholeRhs) ||
- (rhsNbManagedSymbols >= 1 && !IsVariable(rhs) &&
- (lhsNbManagedSymbols >= 1 || !HasCUDADeviceAttrs(lhs))) ||
+ (!HasCUDADeviceAttrs(lhs) && rhsNbManagedSymbols >= 1 &&
+ !rhsHasDeviceOnly && !wholeRhs) ||
+ (rhsNbManagedSymbols >= 1 && !IsVariable(rhs) && !lhsIsDeviceOnly) ||
(lhsNbManagedSymbols >= 1 && rhsNbSymbols == 0)) {
return false;
}
diff --git a/flang/test/Lower/CUDA/cuda-managed-assign.cuf b/flang/test/Lower/CUDA/cuda-managed-assign.cuf
index 00d087d4eb676..ee9779bb99075 100644
--- a/flang/test/Lower/CUDA/cuda-managed-assign.cuf
+++ b/flang/test/Lower/CUDA/cuda-managed-assign.cuf
@@ -108,6 +108,23 @@ end subroutine
! CHECK: cuf.data_transfer {{.*}}#cuf.cuda_transfer<device_device>
! CHECK: cuf.data_transfer {{.*}}#cuf.cuda_transfer<device_host>
+! An allocatable left-hand side is assigned on the host so that it can be
+! reallocated, but not when either side is device data: an assignment between
+! managed and device data is always a copy.
+subroutine managed_device_alloc_transfer()
+ integer(4), allocatable, device :: da(:)
+ integer(4), allocatable, managed :: ma(:)
+ integer(4), managed :: mb(16)
+ integer(4), device :: db(16)
+ allocate(da(16), ma(16))
+ da = mb ! device allocatable = whole managed
+ ma = db ! managed allocatable = whole device
+end subroutine
+
+! CHECK-LABEL: func.func @_QPmanaged_device_alloc_transfer
+! CHECK: cuf.data_transfer {{.*}}#cuf.cuda_transfer<device_device>
+! CHECK: cuf.data_transfer {{.*}}#cuf.cuda_transfer<device_device>
+
! Only the managed side has to be whole: a host section can be the target of a
! whole managed array copy.
subroutine host_section_from_managed(n)
>From 371b8506ebc3d60365aa5a757324273570a5346b Mon Sep 17 00:00:00 2001
From: Zhen Wang <zhenw at nvidia.com>
Date: Wed, 29 Jul 2026 12:56:19 -0700
Subject: [PATCH 2/2] format
---
flang/include/flang/Evaluate/tools.h | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/flang/include/flang/Evaluate/tools.h b/flang/include/flang/Evaluate/tools.h
index bf711050a5457..7d4eadbc46cb5 100644
--- a/flang/include/flang/Evaluate/tools.h
+++ b/flang/include/flang/Evaluate/tools.h
@@ -1441,8 +1441,8 @@ inline bool IsCUDADataTransfer(const A &lhs, const B &rhs) {
bool rhsHasDeviceOnly{rhsNbSymbols > rhsNbManagedSymbols};
// Assignments done on the host, with no copy.
- // - A whole allocatable left-hand side with no device data. The assignment may
- // reallocate it, which is done on the host.
+ // - A whole allocatable left-hand side with no device data. The assignment
+ // may reallocate it, which is done on the host.
// - A managed left-hand side with no whole managed array on either side. Only
// sections and elements are involved, and the host reads and writes them in
// place.
More information about the flang-commits
mailing list