[flang-commits] [flang] 65a8290 - [flang][cuda] Transfer device data used in host operations (#228292)

via flang-commits flang-commits at lists.llvm.org
Sat Oct 3 15:36:40 PDT 2026


Author: Valentin Clement (バレンタイン クレメン)
Date: 2026-10-03T22:36:32Z
New Revision: 65a82906cff9c7a7a33844812d35ce159d88a354

URL: https://github.com/llvm/llvm-project/commit/65a82906cff9c7a7a33844812d35ce159d88a354
DIFF: https://github.com/llvm/llvm-project/commit/65a82906cff9c7a7a33844812d35ce159d88a354.diff

LOG: [flang][cuda] Transfer device data used in host operations (#228292)

An assignment to host data from an expression with device data is
lowered
as an implicit data transfer only when the expression also contains a
constant or a host symbol. Constants in subscripts are ignored, so
expressions such as a(3)*a(4), -a(3), -d1, or a conversion a(3) from
real(8) to real(4) were lowered as plain transfers. The scalar result
was
then computed on the host directly from device memory and stored to the
left-hand side, and array expressions failed the cuf.data_transfer
verifier.

Treat any right-hand side with device data that is not a variable as an
implicit transfer. The device data is copied to a host temporary and the
expression is evaluated on the host. An assignment of such an expression
to device data is now diagnosed as an unsupported transfer, as a = a +
10
already is.

Added: 
    

Modified: 
    flang/include/flang/Evaluate/tools.h
    flang/lib/Evaluate/tools.cpp
    flang/test/Lower/CUDA/cuda-data-transfer.cuf
    flang/test/Semantics/CUDA/cuf18.cuf

Removed: 
    


################################################################################
diff  --git a/flang/include/flang/Evaluate/tools.h b/flang/include/flang/Evaluate/tools.h
index a57c91472e053..6f57e2d064ace 100644
--- a/flang/include/flang/Evaluate/tools.h
+++ b/flang/include/flang/Evaluate/tools.h
@@ -1556,8 +1556,8 @@ inline bool IsCUDADataTransfer(const A &lhs, const B &rhs) {
   return !lhsIsHost || rhsNbSymbols > 0;
 }
 
-/// Check if the expression is a mix of host and device variables that require
-/// implicit data transfer.
+/// Check if the expression uses device data in an operation evaluated on the
+/// host, which requires an implicit data transfer.
 bool HasCUDAImplicitTransfer(const Expr<SomeType> &expr);
 
 /// Check if the expression is a mix of host and constant variables.

diff  --git a/flang/lib/Evaluate/tools.cpp b/flang/lib/Evaluate/tools.cpp
index 4af26628e8e44..b02dcc5318ded 100644
--- a/flang/lib/Evaluate/tools.cpp
+++ b/flang/lib/Evaluate/tools.cpp
@@ -1327,8 +1327,12 @@ GetHostAndDeviceSymbols(const Expr<SomeType> &expr) {
 
 bool HasCUDAImplicitTransfer(const Expr<SomeType> &expr) {
   auto [hostSymbols, deviceSymbols] = GetHostAndDeviceSymbols(expr);
-  bool hasConstant{HasConstant(expr)};
-  return (hasConstant || (hostSymbols.size() > 0)) && deviceSymbols.size() > 0;
+  if (deviceSymbols.empty()) {
+    return false;
+  }
+  // Device data used in an operation, even one with no other operand such as
+  // a negation or a conversion, is copied to the host to evaluate it there.
+  return HasConstant(expr) || !hostSymbols.empty() || !IsVariable(expr);
 }
 
 bool HasOnlyCUDAConstntImplicitTransfer(const Expr<SomeType> &expr) {

diff  --git a/flang/test/Lower/CUDA/cuda-data-transfer.cuf b/flang/test/Lower/CUDA/cuda-data-transfer.cuf
index a5fc2c1eede58..573e3561bc070 100644
--- a/flang/test/Lower/CUDA/cuda-data-transfer.cuf
+++ b/flang/test/Lower/CUDA/cuda-data-transfer.cuf
@@ -520,12 +520,13 @@ end subroutine
 ! CHECK: %[[ALLOC_D:.*]] = cuf.alloc !fir.array<?x?x?xf16>, %{{.*}}, %{{.*}}, %{{.*}} : index, index, index uniq_name("_QFsub26Ed") bindc_name("d") data_attr(device) -> !fir.ref<!fir.array<?x?x?xf16>>
 ! CHECK: %[[D:.*]]:2 = hlfir.declare %[[ALLOC_D]](%{{.*}}) uniq_name("_QFsub26Ed") data_attr(#cuf.cuda<device>) : (!fir.ref<!fir.array<?x?x?xf16>>, !fir.shape<3>) -> (!fir.box<!fir.array<?x?x?xf16>>, !fir.ref<!fir.array<?x?x?xf16>>)
 ! CHECK: %[[HD:.*]]:2 = hlfir.declare %{{.*}}(%{{.*}}) uniq_name("_QFsub26Ehd") : (!fir.ref<!fir.array<?x?x?xf32>>, !fir.shape<3>) -> (!fir.box<!fir.array<?x?x?xf32>>, !fir.ref<!fir.array<?x?x?xf32>>)
-! CHECK: %[[ALLOC:.*]] = fir.allocmem !fir.array<?x?x?xf16>, %8, %13, %18 <{bindc_name = ".tmp", uniq_name = ""}>
+! CHECK: %[[ALLOC:.*]] = fir.allocmem !fir.array<?x?x?xf16>, %{{.*}}, %{{.*}}, %{{.*}} <{bindc_name = ".tmp", uniq_name = ""}>
 ! CHECK: %[[TEMP:.*]]:2 = hlfir.declare %[[ALLOC]](%{{.*}}) uniq_name(".tmp") : (!fir.heap<!fir.array<?x?x?xf16>>, !fir.shape<3>) -> (!fir.box<!fir.array<?x?x?xf16>>, !fir.heap<!fir.array<?x?x?xf16>>)
-! CHECK: cuf.data_transfer %[[D]]#0 to %[[TEMP]]#0 transfer_kind(device_host) : !fir.box<!fir.array<?x?x?xf16>>, !fir.box<!fir.array<?x?x?xf16>>
+! CHECK: %[[TEMP_D:.*]]:2 = hlfir.declare %[[TEMP]]#1(%{{.*}}) uniq_name("_QFsub26Ed") data_attr(#cuf.cuda<device>) : (!fir.heap<!fir.array<?x?x?xf16>>, !fir.shape<3>) -> (!fir.box<!fir.array<?x?x?xf16>>, !fir.heap<!fir.array<?x?x?xf16>>)
+! CHECK: cuf.data_transfer %[[D]]#1 to %[[TEMP]]#0 transfer_kind(device_host) : !fir.ref<!fir.array<?x?x?xf16>>, !fir.box<!fir.array<?x?x?xf16>>
 ! CHECK: %[[ELE:.*]] = hlfir.elemental %{{.*}} unordered : (!fir.shape<3>) -> !hlfir.expr<?x?x?xf32> {
 ! CHECK: ^bb0(%{{.*}}: index, %{{.*}}: index, %{{.*}}: index):
-! CHECK: %[[DESIGNATE:.*]] = hlfir.designate %[[TEMP]]#0 (%{{.*}}, %{{.*}}, %{{.*}})  : (!fir.box<!fir.array<?x?x?xf16>>, index, index, index) -> !fir.ref<f16>
+! CHECK: %[[DESIGNATE:.*]] = hlfir.designate %[[TEMP_D]]#0 (%{{.*}}, %{{.*}}, %{{.*}})  : (!fir.box<!fir.array<?x?x?xf16>>, index, index, index) -> !fir.ref<f16>
 ! CHECK: %[[LOAD:.*]] = fir.load %[[DESIGNATE]] : !fir.ref<f16>
 ! CHECK: %[[CONV:.*]] = fir.convert %[[LOAD]] : (f16) -> f32
 ! CHECK: hlfir.yield_element %[[CONV]] : f32
@@ -546,10 +547,11 @@ end subroutine
 ! CHECK: %[[HD:.*]]:2 = hlfir.declare %[[ALLOC_HD]](%{{.*}}) uniq_name("_QFsub27Ehd") : (!fir.ref<!fir.array<10x20x30xf32>>, !fir.shape<3>) -> (!fir.ref<!fir.array<10x20x30xf32>>, !fir.ref<!fir.array<10x20x30xf32>>)
 ! CHECK: %[[ALLOC_TEMP:.*]] = fir.allocmem !fir.array<10x20x30xf16> <{bindc_name = ".tmp", uniq_name = ""}>
 ! CHECK: %[[TEMP:.*]]:2 = hlfir.declare %[[ALLOC_TEMP]](%{{.*}}) uniq_name(".tmp") : (!fir.heap<!fir.array<10x20x30xf16>>, !fir.shape<3>) -> (!fir.heap<!fir.array<10x20x30xf16>>, !fir.heap<!fir.array<10x20x30xf16>>)
+! CHECK: %[[TEMP_D:.*]]:2 = hlfir.declare %[[TEMP]]#0(%{{.*}}) uniq_name("_QFsub27Ed") data_attr(#cuf.cuda<device>) : (!fir.heap<!fir.array<10x20x30xf16>>, !fir.shape<3>) -> (!fir.heap<!fir.array<10x20x30xf16>>, !fir.heap<!fir.array<10x20x30xf16>>)
 ! CHECK: cuf.data_transfer %[[D]]#0 to %[[TEMP]]#0 transfer_kind(device_host) : !fir.ref<!fir.array<10x20x30xf16>>, !fir.heap<!fir.array<10x20x30xf16>>
 ! CHECK: %[[ELE:.*]] = hlfir.elemental %{{.*}} unordered : (!fir.shape<3>) -> !hlfir.expr<10x20x30xf32> {
 ! CHECK: ^bb0(%{{.*}}: index, %{{.*}}: index, %{{.*}}: index):
-! CHECK: %[[DESIGNATE:.*]] = hlfir.designate %[[TEMP]]#0 (%{{.*}}, %{{.*}}, %{{.*}})  : (!fir.heap<!fir.array<10x20x30xf16>>, index, index, index) -> !fir.ref<f16>
+! CHECK: %[[DESIGNATE:.*]] = hlfir.designate %[[TEMP_D]]#0 (%{{.*}}, %{{.*}}, %{{.*}})  : (!fir.heap<!fir.array<10x20x30xf16>>, index, index, index) -> !fir.ref<f16>
 ! CHECK: %[[LOAD:.*]] = fir.load %[[DESIGNATE]] : !fir.ref<f16>
 ! CHECK: %[[CONV:.*]] = fir.convert %[[LOAD]] : (f16) -> f32
 ! CHECK: hlfir.yield_element %[[CONV]] : f32
@@ -741,9 +743,14 @@ subroutine sub42()
 end subroutine
 
 ! CHECK-LABEL: func.func @_QPsub42()
-! CHECK: %[[RES:.*]] = arith.mulf
-! CHECK: fir.store %[[RES]] to %{{.*}} : !fir.ref<f32>
-! CHECK-NOT: cuf.data_transfer
+! CHECK: %[[C1:.*]]:2 = hlfir.declare %4 uniq_name("_QMmod1Ec1") data_attr(#cuf.cuda<constant>) : (!fir.ref<f32>) -> (!fir.ref<f32>, !fir.ref<f32>)
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %0 uniq_name(".tmp") : (!fir.ref<f32>) -> (!fir.ref<f32>, !fir.ref<f32>)
+! CHECK: %[[TMP_C1:.*]]:2 = hlfir.declare %19#0 uniq_name("_QMmod1Ec1") data_attr(#cuf.cuda<constant>) : (!fir.ref<f32>) -> (!fir.ref<f32>, !fir.ref<f32>)
+! CHECK: cuf.data_transfer %[[C1]]#0 to %[[TMP]]#0 transfer_kind(device_host) : !fir.ref<f32>, !fir.ref<f32>
+! CHECK: %[[LHS:.*]] = fir.load %[[TMP_C1]]#0 : !fir.ref<f32>
+! CHECK: %[[RHS:.*]] = fir.load %[[TMP_C1]]#0 : !fir.ref<f32>
+! CHECK: %[[RES:.*]] = arith.mulf %[[LHS]], %[[RHS]]
+! CHECK: hlfir.assign %[[RES]] to %{{.*}} : f32, !fir.ref<f32>
 
 subroutine sub43()
   use mod1
@@ -752,9 +759,14 @@ subroutine sub43()
 end subroutine
 
 ! CHECK-LABEL: func.func @_QPsub43()
-! CHECK: %[[RES:.*]] = arith.mulf
-! CHECK: fir.store %[[RES]] to %{{.*}} : !fir.ref<f32>
-! CHECK-NOT: cuf.data_transfer
+! CHECK: %[[D1:.*]]:2 = hlfir.declare %{{.*}} uniq_name("_QMmod1Ed1") data_attr(#cuf.cuda<device>) : (!fir.ref<f32>) -> (!fir.ref<f32>, !fir.ref<f32>)
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}} uniq_name(".tmp") : (!fir.ref<f32>) -> (!fir.ref<f32>, !fir.ref<f32>)
+! CHECK: %[[TMP_D1:.*]]:2 = hlfir.declare %[[TMP]]#0 uniq_name("_QMmod1Ed1") data_attr(#cuf.cuda<device>) : (!fir.ref<f32>) -> (!fir.ref<f32>, !fir.ref<f32>)
+! CHECK: cuf.data_transfer %[[D1]]#0 to %[[TMP]]#0 transfer_kind(device_host) : !fir.ref<f32>, !fir.ref<f32>
+! CHECK: %[[LHS:.*]] = fir.load %[[TMP_D1]]#0 : !fir.ref<f32>
+! CHECK: %[[RHS:.*]] = fir.load %[[TMP_D1]]#0 : !fir.ref<f32>
+! CHECK: %[[RES:.*]] = arith.mulf %[[LHS]], %[[RHS]]
+! CHECK: hlfir.assign %[[RES]] to %{{.*}} : f32, !fir.ref<f32>
 
 subroutine sub44()
   use mod1
@@ -763,8 +775,13 @@ subroutine sub44()
 end subroutine
 
 ! CHECK-LABEL: func.func @_QPsub44()
-! CHECK: arith.negf
-! CHECK-NOT: cuf.data_transfer
+! CHECK: %[[D1:.*]]:2 = hlfir.declare %{{.*}} uniq_name("_QMmod1Ed1") data_attr(#cuf.cuda<device>) : (!fir.ref<f32>) -> (!fir.ref<f32>, !fir.ref<f32>)
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}} uniq_name(".tmp") : (!fir.ref<f32>) -> (!fir.ref<f32>, !fir.ref<f32>)
+! CHECK: %[[TMP_D1:.*]]:2 = hlfir.declare %[[TMP]]#0 uniq_name("_QMmod1Ed1") data_attr(#cuf.cuda<device>) : (!fir.ref<f32>) -> (!fir.ref<f32>, !fir.ref<f32>)
+! CHECK: cuf.data_transfer %[[D1]]#0 to %[[TMP]]#0 transfer_kind(device_host) : !fir.ref<f32>, !fir.ref<f32>
+! CHECK: %[[VAL:.*]] = fir.load %[[TMP_D1]]#0 : !fir.ref<f32>
+! CHECK: %[[RES:.*]] = arith.negf %[[VAL]]
+! CHECK: hlfir.assign %[[RES]] to %{{.*}} : f32, !fir.ref<f32>
 
 subroutine sub45(c0_d, n)
   implicit none
@@ -795,3 +812,68 @@ end subroutine
 ! CHECK: arith.divf
 ! CHECK: hlfir.assign
 ! CHECK: cuf.data_transfer
+
+subroutine sub47(h, a)
+  double precision :: h
+  double precision, device :: a(10)
+  ! Implicit data transfer of a and then multiplication on the host
+  h = a(3) * a(4)
+end subroutine
+
+! CHECK-LABEL: func.func @_QPsub47
+! CHECK: %[[A:.*]]:2 = hlfir.declare %{{.*}} uniq_name("_QFsub47Ea") data_attr(#cuf.cuda<device>) : (!fir.ref<!fir.array<10xf64>>, !fir.shape<1>, !fir.dscope) -> (!fir.ref<!fir.array<10xf64>>, !fir.ref<!fir.array<10xf64>>)
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}} uniq_name(".tmp") : (!fir.heap<!fir.array<10xf64>>, !fir.shape<1>) -> (!fir.heap<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>)
+! CHECK: %[[TMP_A:.*]]:2 = hlfir.declare %[[TMP]]#0(%{{.*}}) uniq_name("_QFsub47Ea") data_attr(#cuf.cuda<device>) : (!fir.heap<!fir.array<10xf64>>, !fir.shape<1>) -> (!fir.heap<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>)
+! CHECK: cuf.data_transfer %[[A]]#0 to %[[TMP]]#0 transfer_kind(device_host) : !fir.ref<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>
+! CHECK: hlfir.designate %[[TMP_A]]#0 (%c3{{.*}})
+! CHECK: hlfir.designate %[[TMP_A]]#0 (%c4{{.*}})
+! CHECK: arith.mulf
+! CHECK-NOT: hlfir.designate %[[A]]#0
+
+subroutine sub48(h, a)
+  double precision :: h
+  double precision, device :: a(10)
+  ! Implicit data transfer of a and then negation on the host
+  h = -a(3)
+end subroutine
+
+! CHECK-LABEL: func.func @_QPsub48
+! CHECK: %[[A:.*]]:2 = hlfir.declare %{{.*}} uniq_name("_QFsub48Ea") data_attr(#cuf.cuda<device>) : (!fir.ref<!fir.array<10xf64>>, !fir.shape<1>, !fir.dscope) -> (!fir.ref<!fir.array<10xf64>>, !fir.ref<!fir.array<10xf64>>)
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}} uniq_name(".tmp") : (!fir.heap<!fir.array<10xf64>>, !fir.shape<1>) -> (!fir.heap<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>)
+! CHECK: %[[TMP_A:.*]]:2 = hlfir.declare %[[TMP]]#0(%{{.*}}) uniq_name("_QFsub48Ea") data_attr(#cuf.cuda<device>) : (!fir.heap<!fir.array<10xf64>>, !fir.shape<1>) -> (!fir.heap<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>)
+! CHECK: cuf.data_transfer %[[A]]#0 to %[[TMP]]#0 transfer_kind(device_host) : !fir.ref<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>
+! CHECK: hlfir.designate %[[TMP_A]]#0 (%c3{{.*}})
+! CHECK: arith.negf
+! CHECK-NOT: hlfir.designate %[[A]]#0
+
+subroutine sub49(h, a)
+  real :: h
+  double precision, device :: a(10)
+  ! Implicit data transfer of a and then conversion on the host
+  h = a(3)
+end subroutine
+
+! CHECK-LABEL: func.func @_QPsub49
+! CHECK: %[[A:.*]]:2 = hlfir.declare %{{.*}} uniq_name("_QFsub49Ea") data_attr(#cuf.cuda<device>) : (!fir.ref<!fir.array<10xf64>>, !fir.shape<1>, !fir.dscope) -> (!fir.ref<!fir.array<10xf64>>, !fir.ref<!fir.array<10xf64>>)
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}} uniq_name(".tmp") : (!fir.heap<!fir.array<10xf64>>, !fir.shape<1>) -> (!fir.heap<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>)
+! CHECK: %[[TMP_A:.*]]:2 = hlfir.declare %[[TMP]]#0(%{{.*}}) uniq_name("_QFsub49Ea") data_attr(#cuf.cuda<device>) : (!fir.heap<!fir.array<10xf64>>, !fir.shape<1>) -> (!fir.heap<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>)
+! CHECK: cuf.data_transfer %[[A]]#0 to %[[TMP]]#0 transfer_kind(device_host) : !fir.ref<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>
+! CHECK: hlfir.designate %[[TMP_A]]#0 (%c3{{.*}})
+! CHECK: fir.convert %{{.*}} : (f64) -> f32
+! CHECK-NOT: hlfir.designate %[[A]]#0
+
+subroutine sub50(h, a)
+  double precision :: h(10)
+  double precision, device :: a(10)
+  ! Implicit data transfer of a and then negation on the host
+  h = -a
+end subroutine
+
+! CHECK-LABEL: func.func @_QPsub50
+! CHECK: %[[A:.*]]:2 = hlfir.declare %{{.*}} uniq_name("_QFsub50Ea") data_attr(#cuf.cuda<device>) : (!fir.ref<!fir.array<10xf64>>, !fir.shape<1>, !fir.dscope) -> (!fir.ref<!fir.array<10xf64>>, !fir.ref<!fir.array<10xf64>>)
+! CHECK: %[[TMP:.*]]:2 = hlfir.declare %{{.*}} uniq_name(".tmp") : (!fir.heap<!fir.array<10xf64>>, !fir.shape<1>) -> (!fir.heap<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>)
+! CHECK: %[[TMP_A:.*]]:2 = hlfir.declare %[[TMP]]#0(%{{.*}}) uniq_name("_QFsub50Ea") data_attr(#cuf.cuda<device>) : (!fir.heap<!fir.array<10xf64>>, !fir.shape<1>) -> (!fir.heap<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>)
+! CHECK: cuf.data_transfer %[[A]]#0 to %[[TMP]]#0 transfer_kind(device_host) : !fir.ref<!fir.array<10xf64>>, !fir.heap<!fir.array<10xf64>>
+! CHECK: hlfir.elemental
+! CHECK: arith.negf
+! CHECK: hlfir.assign

diff  --git a/flang/test/Semantics/CUDA/cuf18.cuf b/flang/test/Semantics/CUDA/cuf18.cuf
index 8148e0de340b6..3d1b06d151c2a 100644
--- a/flang/test/Semantics/CUDA/cuf18.cuf
+++ b/flang/test/Semantics/CUDA/cuf18.cuf
@@ -6,6 +6,8 @@ subroutine sub1()
 
 !ERROR: Unsupported CUDA data transfer
   a = a + 10 ! Illegal expression according to 3.4.2
+!ERROR: Unsupported CUDA data transfer
+  a = -a ! Illegal expression according to 3.4.2
 
   !$cuf kernel do
   do i = 1, 10


        


More information about the flang-commits mailing list