[flang-commits] [flang] [flang][OpenMP] Lower array-section workdistribute assign to element loop (PR #225595)

via flang-commits flang-commits at lists.llvm.org
Mon Oct 5 22:51:01 PDT 2026


================
@@ -595,10 +617,70 @@ static void replaceWithUnorderedDoLoop(OpBuilder &builder, Location loc,
   fir::StoreOp::create(builder, loc, scalar, elemPtr);
 }
 
+/// Return the array descriptor (fir.box) value behind a FortranAAssign arg.
+/// The arg is the address of a descriptor temp: prefer the box value stored
+/// into it, otherwise load the reference.
+static Value getAssignArrayBox(OpBuilder &builder, Location loc, Value box) {
+  if (auto alloca = box.getDefiningOp<fir::AllocaOp>()) {
+    for (auto *user : alloca->getUsers())
+      if (auto storeOp = dyn_cast<fir::StoreOp>(user)) {
+        box = storeOp.getValue();
+        break;
+      }
+  }
+  if (isa<fir::ReferenceType>(box.getType()))
+    box = fir::LoadOp::create(builder, loc, box);
+  return box;
+}
+
+/// Replace an array-to-array FortranAAssign runtime call with an unordered do
+/// loop that copies element by element. Addressing goes through fir.array_coor
+/// on both descriptors, so each section's own strides and bounds are honored -
+/// unlike a flat memcpy, this is correct for strided sections.
+static void replaceArrayToArrayAssignWithUnorderedDoLoop(
+    OpBuilder &builder, Location loc, omp::TeamsOp teamsOp,
+    omp::WorkdistributeOp workdistribute, fir::CallOp callOp) {
+  auto destConvert = callOp.getOperand(0).getDefiningOp<fir::ConvertOp>();
+  auto srcConvert = callOp.getOperand(1).getDefiningOp<fir::ConvertOp>();
+
+  builder.setInsertionPoint(teamsOp);
+  Value destBox = getAssignArrayBox(builder, loc, destConvert.getValue());
+  Value srcBox = getAssignArrayBox(builder, loc, srcConvert.getValue());
+
+  // Element type comes from the destination sequence.
+  auto destBoxType = cast<fir::BoxType>(destBox.getType());
+  auto destSeqType = cast<fir::SequenceType>(destBoxType.getEleTy());
+  Type eleTy = destSeqType.getEleTy();
+  auto eleRefTy = fir::ReferenceType::get(eleTy);
+
+  auto c0 = arith::ConstantIndexOp::create(builder, loc, 0);
+  auto c1 = arith::ConstantIndexOp::create(builder, loc, 1);
+  Value totalElems = CalculateTotalElements(builder, loc, destBox);
+
+  auto *workdistributeBlock = &workdistribute.getRegion().front();
+  builder.setInsertionPointToStart(workdistributeBlock);
+  // Single flattened loop: dest and src conform, so one index set fits both.
+  auto doLoop = fir::DoLoopOp::create(builder, loc, c0, totalElems, c1, true);
+  builder.setInsertionPointToStart(doLoop.getBody());
+
+  auto flatIdx = doLoop.getRegion().front().getArgument(0);
+  SmallVector<Value> indices =
+      convertFlatToMultiDim(builder, loc, flatIdx, destBox);
+
+  auto srcPtr =
+      fir::ArrayCoorOp::create(builder, loc, eleRefTy, srcBox, nullptr, nullptr,
+                               ValueRange{indices}, ValueRange{});
+  Value value = fir::LoadOp::create(builder, loc, srcPtr);
----------------
skc7 wrote:

Fixed in latest patch. 
`array-to-array` assigns now copy through a heap temporary in two loops (src -> tmp -> dest). Added tests for these.

https://github.com/llvm/llvm-project/pull/225595


More information about the flang-commits mailing list