[flang-commits] [flang] [flang][cuda] Transfer device data used in host operations (PR #228292)

via flang-commits flang-commits at lists.llvm.org
Thu Oct 1 18:00:52 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-flang-fir-hlfir

Author: Valentin Clement (バレンタイン クレメン) (clementval)

<details>
<summary>Changes</summary>

An assignment to host data from an expression with device data is lowered
as an implicit data transfer only when the expression also contains a
constant or a host symbol. Constants in subscripts are ignored, so
expressions such as a(3)*a(4), -a(3), -d1, or a conversion a(3) from
real(8) to real(4) were lowered as plain transfers. The scalar result was
then computed on the host directly from device memory and stored to the
left-hand side, and array expressions failed the cuf.data_transfer
verifier.

Treat any right-hand side with device data that is not a variable as an
implicit transfer. The device data is copied to a host temporary and the
expression is evaluated on the host. An assignment of such an expression
to device data is now diagnosed as an unsupported transfer, as a = a + 10
already is.

---
Full diff: https://github.com/llvm/llvm-project/pull/228292.diff


4 Files Affected:

- (modified) flang/include/flang/Evaluate/tools.h (+2-2) 
- (modified) flang/lib/Evaluate/tools.cpp (+6-2) 
- (modified) flang/test/Lower/CUDA/cuda-data-transfer.cuf (+94-12) 
- (modified) flang/test/Semantics/CUDA/cuf18.cuf (+2) 


``````````diff
diff --git a/flang/include/flang/Evaluate/tools.h b/flang/include/flang/Evaluate/tools.h
index 3a7d234f903ad..515fe6ac75544 100644
--- a/flang/include/flang/Evaluate/tools.h
+++ b/flang/include/flang/Evaluate/tools.h
@@ -1555,8 +1555,8 @@ inline bool IsCUDADataTransfer(const A &lhs, const B &rhs) {
   return !lhsIsHost || rhsNbSymbols > 0;
 }
 
-/// Check if the expression is a mix of host and device variables that require
-/// implicit data transfer.
+/// Check if the expression uses device data in an operation evaluated on the
+/// host, which requires an implicit data transfer.
 bool HasCUDAImplicitTransfer(const Expr<SomeType> &expr);
 
 /// Check if the expression is a mix of host and constant variables.
diff --git a/flang/lib/Evaluate/tools.cpp b/flang/lib/Evaluate/tools.cpp
index 347993d892466..8a08eb2c6e155 100644
--- a/flang/lib/Evaluate/tools.cpp
+++ b/flang/lib/Evaluate/tools.cpp
@@ -1303,8 +1303,12 @@ GetHostAndDeviceSymbols(const Expr<SomeType> &expr) {
 
 bool HasCUDAImplicitTransfer(const Expr<SomeType> &expr) {
   auto [hostSymbols, deviceSymbols] = GetHostAndDeviceSymbols(expr);
-  bool hasConstant{HasConstant(expr)};
-  return (hasConstant || (hostSymbols.size() > 0)) && deviceSymbols.size() > 0;
+  if (deviceSymbols.empty()) {
+    return false;
+  }
+  // Device data used in an operation, even one with no other operand such as
+  // a negation or a conversion, is copied to the host to evaluate it there.
+  return HasConstant(expr) || !hostSymbols.empty() || !IsVariable(expr);
 }
 
 bool HasOnlyCUDAConstntImplicitTransfer(const Expr<SomeType> &expr) {
diff --git a/flang/test/Lower/CUDA/cuda-data-transfer.cuf b/flang/test/Lower/CUDA/cuda-data-transfer.cuf
index 7cc40773d2789..2dac427a1e8dc 100644
--- a/flang/test/Lower/CUDA/cuda-data-transfer.cuf
+++ b/flang/test/Lower/CUDA/cuda-data-transfer.cuf
@@ -520,12 +520,13 @@ end subroutine
 ! CHECK: %[[ALLOC_D:.*]] = cuf.alloc !fir.array<?x?x?xf16>, %{{.*}}, %{{.*}}, %{{.*}} : index, index, index {bindc_name = "d", data_attr = #cuf.cuda<device>, uniq_name = "_QFsub26Ed"} -> !fir.ref<!fir.array<?x?x?xf16>>
 ! CHECK: %[[D:.*]]:2 = hlfir.declare %[[ALLOC_D]](%{{.*}}) {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub26Ed"} : (!fir.ref<!fir.array<?x?x?xf16>>, !fir.shape<3>) -> (!fir.box<!fir.array<?x?x?xf16>>, !fir.ref<!fir.array<?x?x?xf16>>)
 ! CHECK: %[[HD:.*]]:2 = hlfir.declare %{{.*}}(%{{.*}}) {uniq_name = "_QFsub26Ehd"} : (!fir.ref<!fir.array<?x?x?xf32>>, !fir.shape<3>) -> (!fir.box<!fir.array<?x?x?xf32>>, !fir.ref<!fir.array<?x?x?xf32>>)
-! CHECK: %[[ALLOC:.*]] = fir.allocmem !fir.array<?x?x?xf16>, %8, %13, %18 <{bindc_name = ".tmp", uniq_name = ""}>
+! CHECK: %[[ALLOC:.*]] = fir.allocmem !fir.array<?x?x?xf16>, %{{.*}}, %{{.*}}, %{{.*}} <{bindc_name = ".tmp", uniq_name = ""}>
 ! CHECK: %[[TEMP:.*]]:2 = hlfir.declare %[[ALLOC]](%{{.*}}) {uniq_name = ".tmp"} : (!fir.heap<!fir.array<?x?x?xf16>>, !fir.shape<3>) -> (!fir.box<!fir.array<?x?x?xf16>>, !fir.heap<!fir.array<?x?x?xf16>>)
-! CHECK: cuf.data_transfer %[[D]]#0 to %[[TEMP]]#0 {transfer_kind = #cuf.cuda_transfer<device_host>} : !fir.box<!fir.array<?x?x?xf16>>, !fir.box<!fir.array<?x?x?xf16>>
+! CHECK: %[[TEMP_D:.*]]:2 = hlfir.declare %[[TEMP]]#1(%{{.*}}) {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub26Ed"} : (!fir.heap<!fir.array<?x?x?xf16>>, !fir.shape<3>) -> (!fir.box<!fir.array<?x?x?xf16>>, !fir.heap<!fir.array<?x?x?xf16>>)
+! CHECK: cuf.data_transfer %[[D]]#1 to %[[TEMP]]#0 {transfer_kind = #cuf.cuda_transfer<device_host>} : !fir.ref<!fir.array<?x?x?xf16>>, !fir.box<!fir.array<?x?x?xf16>>
 ! CHECK: %[[ELE:.*]] = hlfir.elemental %{{.*}} unordered : (!fir.shape<3>) -> !hlfir.expr<?x?x?xf32> {
 ! CHECK: ^bb0(%{{.*}}: index, %{{.*}}: index, %{{.*}}: index):
-! CHECK: %[[DESIGNATE:.*]] = hlfir.designate %[[TEMP]]#0 (%{{.*}}, %{{.*}}, %{{.*}})  : (!fir.box<!fir.array<?x?x?xf16>>, index, index, index) -> !fir.ref<f16>
+! CHECK: %[[DESIGNATE:.*]] = hlfir.designate %[[TEMP_D]]#0 (%{{.*}}, %{{.*}}, %{{.*}})  : (!fir.box<!fir.array<?x?x?xf16>>, index, index, index) -> !fir.ref<f16>
 ! CHECK: %[[LOAD:.*]] = fir.load %[[DESIGNATE]] : !fir.ref<f16>
 ! CHECK: %[[CONV:.*]] = fir.convert %[[LOAD]] : (f16) -> f32
 ! CHECK: hlfir.yield_element %[[CONV]] : f32
@@ -546,10 +547,11 @@ end subroutine
 ! CHECK: %[[HD:.*]]:2 = hlfir.declare %[[ALLOC_HD]](%{{.*}}) {uniq_name = "_QFsub27Ehd"} : (!fir.ref<!fir.array<10x20x30xf32>>, !fir.shape<3>) -> (!fir.ref<!fir.array<10x20x30xf32>>, !fir.ref<!fir.array<10x20x30xf32>>)
 ! CHECK: %[[ALLOC_TEMP:.*]] = fir.allocmem !fir.array<10x20x30xf16> <{bindc_name = ".tmp", uniq_name = ""}>
 ! CHECK: %[[TEMP:.*]]:2 = hlfir.declare %[[ALLOC_TEMP]](%{{.*}}) {uniq_name = ".tmp"} : (!fir.heap<!fir.array<10x20x30xf16>>, !fir.shape<3>) -> (!fir.heap<!fir.array<10x20x30xf16>>, !fir.heap<!fir.array<10x20x30xf16>>)
+! CHECK: %[[TEMP_D:.*]]:2 = hlfir.declare %[[TEMP]]#0(%{{.*}}) {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub27Ed"} : (!fir.heap<!fir.array<10x20x30xf16>>, !fir.shape<3>) -> (!fir.heap<!fir.array<10x20x30xf16>>, !fir.heap<!fir.array<10x20x30xf16>>)
 ! CHECK: cuf.data_transfer %[[D]]#0 to %[[TEMP]]#0 {transfer_kind = #cuf.cuda_transfer<device_host>} : !fir.ref<!fir.array<10x20x30xf16>>, !fir.heap<!fir.array<10x20x30xf16>>
 ! CHECK: %[[ELE:.*]] = hlfir.elemental %{{.*}} unordered : (!fir.shape<3>) -> !hlfir.expr<10x20x30xf32> {
 ! CHECK: ^bb0(%{{.*}}: index, %{{.*}}: index, %{{.*}}: index):
-! CHECK: %[[DESIGNATE:.*]] = hlfir.designate %[[TEMP]]#0 (%{{.*}}, %{{.*}}, %{{.*}})  : (!fir.heap<!fir.array<10x20x30xf16>>, index, index, index) -> !fir.ref<f16>
+! CHECK: %[[DESIGNATE:.*]] = hlfir.designate %[[TEMP_D]]#0 (%{{.*}}, %{{.*}}, %{{.*}})  : (!fir.heap<!fir.array<10x20x30xf16>>, index, index, index) -> !fir.ref<f16>
 ! CHECK: %[[LOAD:.*]] = fir.load %[[DESIGNATE]] : !fir.ref<f16>
 ! CHECK: %[[CONV:.*]] = fir.convert %[[LOAD]] : (f16) -> f32
 ! CHECK: hlfir.yield_element %[[CONV]] : f32
@@ -741,9 +743,14 @@ subroutine sub42()
 end subroutine
 
 ! CHECK-LABEL: func.func @_QPsub42()
-! CHECK: %[[RES:.*]] = arith.mulf
-! CHECK: fir.store %[[RES]] to %{{.*}} : !fir.ref<f32>
-! CHECK-NOT: cuf.data_transfer
+! CHECK: %[[C1:.*]]:2 = hlfir.declare %{{.*}} {data_attr = #cuf.cuda<constant>, uniq_name = "_QMmod1Ec1"}
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}} {uniq_name = ".tmp"}
+! CHECK: %[[TMP_C1:.*]]:2 = hlfir.declare %[[TMP]]#0 {data_attr = #cuf.cuda<constant>, uniq_name = "_QMmod1Ec1"}
+! CHECK: cuf.data_transfer %[[C1]]#0 to %[[TMP]]#0 {transfer_kind = #cuf.cuda_transfer<device_host>} : !fir.ref<f32>, !fir.ref<f32>
+! CHECK: %[[LHS:.*]] = fir.load %[[TMP_C1]]#0 : !fir.ref<f32>
+! CHECK: %[[RHS:.*]] = fir.load %[[TMP_C1]]#0 : !fir.ref<f32>
+! CHECK: %[[RES:.*]] = arith.mulf %[[LHS]], %[[RHS]]
+! CHECK: hlfir.assign %[[RES]] to %{{.*}} : f32, !fir.ref<f32>
 
 subroutine sub43()
   use mod1
@@ -752,9 +759,14 @@ subroutine sub43()
 end subroutine
 
 ! CHECK-LABEL: func.func @_QPsub43()
-! CHECK: %[[RES:.*]] = arith.mulf
-! CHECK: fir.store %[[RES]] to %{{.*}} : !fir.ref<f32>
-! CHECK-NOT: cuf.data_transfer
+! CHECK: %[[D1:.*]]:2 = hlfir.declare %{{.*}} {data_attr = #cuf.cuda<device>, uniq_name = "_QMmod1Ed1"}
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}} {uniq_name = ".tmp"}
+! CHECK: %[[TMP_D1:.*]]:2 = hlfir.declare %[[TMP]]#0 {data_attr = #cuf.cuda<device>, uniq_name = "_QMmod1Ed1"}
+! CHECK: cuf.data_transfer %[[D1]]#0 to %[[TMP]]#0 {transfer_kind = #cuf.cuda_transfer<device_host>} : !fir.ref<f32>, !fir.ref<f32>
+! CHECK: %[[LHS:.*]] = fir.load %[[TMP_D1]]#0 : !fir.ref<f32>
+! CHECK: %[[RHS:.*]] = fir.load %[[TMP_D1]]#0 : !fir.ref<f32>
+! CHECK: %[[RES:.*]] = arith.mulf %[[LHS]], %[[RHS]]
+! CHECK: hlfir.assign %[[RES]] to %{{.*}} : f32, !fir.ref<f32>
 
 subroutine sub44()
   use mod1
@@ -763,8 +775,13 @@ subroutine sub44()
 end subroutine
 
 ! CHECK-LABEL: func.func @_QPsub44()
-! CHECK: arith.negf
-! CHECK-NOT: cuf.data_transfer
+! CHECK: %[[D1:.*]]:2 = hlfir.declare %{{.*}} {data_attr = #cuf.cuda<device>, uniq_name = "_QMmod1Ed1"}
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}} {uniq_name = ".tmp"}
+! CHECK: %[[TMP_D1:.*]]:2 = hlfir.declare %[[TMP]]#0 {data_attr = #cuf.cuda<device>, uniq_name = "_QMmod1Ed1"}
+! CHECK: cuf.data_transfer %[[D1]]#0 to %[[TMP]]#0 {transfer_kind = #cuf.cuda_transfer<device_host>} : !fir.ref<f32>, !fir.ref<f32>
+! CHECK: %[[VAL:.*]] = fir.load %[[TMP_D1]]#0 : !fir.ref<f32>
+! CHECK: %[[RES:.*]] = arith.negf %[[VAL]]
+! CHECK: hlfir.assign %[[RES]] to %{{.*}} : f32, !fir.ref<f32>
 
 subroutine sub45(c0_d, n)
   implicit none
@@ -795,3 +812,68 @@ end subroutine
 ! CHECK: arith.divf
 ! CHECK: hlfir.assign
 ! CHECK: cuf.data_transfer
+
+subroutine sub47(h, a)
+  double precision :: h
+  double precision, device :: a(10)
+  ! Implicit data transfer of a and then multiplication on the host
+  h = a(3) * a(4)
+end subroutine
+
+! CHECK-LABEL: func.func @_QPsub47
+! CHECK: %[[A:.*]]:2 = hlfir.declare %{{.*}}(%{{.*}}) dummy_scope %{{.*}} arg 2 {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub47Ea"}
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}}(%{{.*}}) {uniq_name = ".tmp"}
+! CHECK: %[[TMP_A:.*]]:2 = hlfir.declare %[[TMP]]#0(%{{.*}}) {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub47Ea"}
+! CHECK: cuf.data_transfer %[[A]]#0 to %[[TMP]]#0 {transfer_kind = #cuf.cuda_transfer<device_host>}
+! CHECK: hlfir.designate %[[TMP_A]]#0 (%c3{{.*}})
+! CHECK: hlfir.designate %[[TMP_A]]#0 (%c4{{.*}})
+! CHECK: arith.mulf
+! CHECK-NOT: hlfir.designate %[[A]]#0
+
+subroutine sub48(h, a)
+  double precision :: h
+  double precision, device :: a(10)
+  ! Implicit data transfer of a and then negation on the host
+  h = -a(3)
+end subroutine
+
+! CHECK-LABEL: func.func @_QPsub48
+! CHECK: %[[A:.*]]:2 = hlfir.declare %{{.*}}(%{{.*}}) dummy_scope %{{.*}} arg 2 {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub48Ea"}
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}}(%{{.*}}) {uniq_name = ".tmp"}
+! CHECK: %[[TMP_A:.*]]:2 = hlfir.declare %[[TMP]]#0(%{{.*}}) {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub48Ea"}
+! CHECK: cuf.data_transfer %[[A]]#0 to %[[TMP]]#0 {transfer_kind = #cuf.cuda_transfer<device_host>}
+! CHECK: hlfir.designate %[[TMP_A]]#0 (%c3{{.*}})
+! CHECK: arith.negf
+! CHECK-NOT: hlfir.designate %[[A]]#0
+
+subroutine sub49(h, a)
+  real :: h
+  double precision, device :: a(10)
+  ! Implicit data transfer of a and then conversion on the host
+  h = a(3)
+end subroutine
+
+! CHECK-LABEL: func.func @_QPsub49
+! CHECK: %[[A:.*]]:2 = hlfir.declare %{{.*}}(%{{.*}}) dummy_scope %{{.*}} arg 2 {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub49Ea"}
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}}(%{{.*}}) {uniq_name = ".tmp"}
+! CHECK: %[[TMP_A:.*]]:2 = hlfir.declare %[[TMP]]#0(%{{.*}}) {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub49Ea"}
+! CHECK: cuf.data_transfer %[[A]]#0 to %[[TMP]]#0 {transfer_kind = #cuf.cuda_transfer<device_host>}
+! CHECK: hlfir.designate %[[TMP_A]]#0 (%c3{{.*}})
+! CHECK: fir.convert %{{.*}} : (f64) -> f32
+! CHECK-NOT: hlfir.designate %[[A]]#0
+
+subroutine sub50(h, a)
+  double precision :: h(10)
+  double precision, device :: a(10)
+  ! Implicit data transfer of a and then negation on the host
+  h = -a
+end subroutine
+
+! CHECK-LABEL: func.func @_QPsub50
+! CHECK: %[[A:.*]]:2 = hlfir.declare %{{.*}}(%{{.*}}) dummy_scope %{{.*}} arg 2 {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub50Ea"}
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}}(%{{.*}}) {uniq_name = ".tmp"}
+! CHECK: %[[TMP_A:.*]]:2 = hlfir.declare %[[TMP]]#0(%{{.*}}) {data_attr = #cuf.cuda<device>, uniq_name = "_QFsub50Ea"}
+! CHECK: cuf.data_transfer %[[A]]#0 to %[[TMP]]#0 {transfer_kind = #cuf.cuda_transfer<device_host>}
+! CHECK: hlfir.elemental
+! CHECK: arith.negf
+! CHECK: hlfir.assign
diff --git a/flang/test/Semantics/CUDA/cuf18.cuf b/flang/test/Semantics/CUDA/cuf18.cuf
index 8148e0de340b6..3d1b06d151c2a 100644
--- a/flang/test/Semantics/CUDA/cuf18.cuf
+++ b/flang/test/Semantics/CUDA/cuf18.cuf
@@ -6,6 +6,8 @@ subroutine sub1()
 
 !ERROR: Unsupported CUDA data transfer
   a = a + 10 ! Illegal expression according to 3.4.2
+!ERROR: Unsupported CUDA data transfer
+  a = -a ! Illegal expression according to 3.4.2
 
   !$cuf kernel do
   do i = 1, 10

``````````

</details>


https://github.com/llvm/llvm-project/pull/228292


More information about the flang-commits mailing list