[llvm-branch-commits] [flang] [flang] Lower loops whose branching is confined to their body structurally (PR #225758)

Kareem Ergawy via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Thu Sep 24 01:59:53 PDT 2026


https://github.com/ergawy updated https://github.com/llvm/llvm-project/pull/225758

>From b8f452e9c645f35f169028eb60a5bee7752e6f5d Mon Sep 17 00:00:00 2001
From: ergawy <kareem.ergawy at gmail.com>
Date: Mon, 21 Sep 2026 06:55:52 -0700
Subject: [PATCH] [flang] Lower loops whose branching is confined to their body
 structurally

Such a loop was classified separately by a previous change but still
lowered as a raw CFG, so its structured form was lost.

Lower it structurally instead, with its body folded into a region that can
hold the branching. The loop keeps its bounds on the op, so it remains
available to whatever transforms or parallelizes it. Only the body is
folded: the loop control statements are emitted as they are for any
structured loop, since a branch from outside may target either of them.

Loops an OpenACC or OpenMP directive owns are lowered the same way, so
they keep their form too.
---
 flang/include/flang/Lower/AbstractConverter.h |   8 +
 flang/include/flang/Lower/PFTBuilder.h        |  16 +-
 flang/lib/Lower/Bridge.cpp                    | 130 +++++++++++++--
 flang/lib/Lower/OpenMP/OpenMP.cpp             |   8 +
 flang/lib/Lower/PFTBuilder.cpp                |   4 +
 flang/test/Lower/OpenACC/acc-cache.f90        |  19 +--
 .../OpenACC/acc-unstructured-internals.f90    |  73 +++++++++
 flang/test/Lower/OpenACC/acc-unstructured.f90 |  17 +-
 .../Todo/metadirective-loop-unstructured.f90  |  39 -----
 .../metadirective-loop-unstructured.f90       |  50 ++++++
 .../OpenMP/wsloop-unstructured-cycle.f90      |  62 +++----
 .../OpenMP/wsloop-unstructured-internals.f90  |  44 +++++
 flang/test/Lower/always-execute-loop-body.f90 |   8 +-
 .../Lower/do-loop-branch-to-loop-header.f90   |  55 +++++++
 flang/test/Lower/do_loop_unstructured.f90     | 154 +++---------------
 flang/test/Lower/select-case-statement.f90    |   7 +-
 16 files changed, 453 insertions(+), 241 deletions(-)
 create mode 100644 flang/test/Lower/OpenACC/acc-unstructured-internals.f90
 delete mode 100644 flang/test/Lower/OpenMP/Todo/metadirective-loop-unstructured.f90
 create mode 100644 flang/test/Lower/OpenMP/metadirective-loop-unstructured.f90
 create mode 100644 flang/test/Lower/OpenMP/wsloop-unstructured-internals.f90
 create mode 100644 flang/test/Lower/do-loop-branch-to-loop-header.f90

diff --git a/flang/include/flang/Lower/AbstractConverter.h b/flang/include/flang/Lower/AbstractConverter.h
index ae246d3188bd8..40d94dc9f5bfd 100644
--- a/flang/include/flang/Lower/AbstractConverter.h
+++ b/flang/include/flang/Lower/AbstractConverter.h
@@ -384,6 +384,14 @@ class AbstractConverter {
   virtual void genEval(pft::Evaluation &eval,
                        bool unstructuredContext = true) = 0;
 
+  /// Emit \p loopEval's evaluations, folding the body into an
+  /// scf.execute_region when the loop's branching is confined to that body.
+  /// The loop control statements are emitted outside any wrap, exactly as they
+  /// are for a structured loop. Used by directive lowering, which consumes the
+  /// DO itself and so never reaches genFIR(DoConstruct), where a plain loop's
+  /// body is wrapped.
+  virtual void genLoopBodyEvaluations(pft::Evaluation &loopEval) = 0;
+
   /// Return options controlling lowering behavior.
   const Fortran::lower::LoweringOptions &getLoweringOptions() const {
     return loweringOptions;
diff --git a/flang/include/flang/Lower/PFTBuilder.h b/flang/include/flang/Lower/PFTBuilder.h
index 930214244c74c..f206c08fc19f3 100644
--- a/flang/include/flang/Lower/PFTBuilder.h
+++ b/flang/include/flang/Lower/PFTBuilder.h
@@ -356,17 +356,21 @@ struct Evaluation : EvaluationVariant {
     controlFlow = std::min(controlFlow, kind);
   }
 
-  /// True when control flow is not fully structured, category (c) included.
-  /// Existing consumers ask this to decide whether raw blocks are needed, and
-  /// a category (c) construct still needs them until its body is wrapped, so
-  /// it must answer true here. Use hasUnstructuredInternals() to single out
-  /// category (c) itself.
-  bool isUnstructured() const { return controlFlow != ControlFlow::Structured; }
+  /// True only for fully unstructured control flow, which is lowered as raw
+  /// CFG blocks.
+  bool isUnstructured() const {
+    return controlFlow == ControlFlow::Unstructured;
+  }
 
   bool hasUnstructuredInternals() const {
     return controlFlow == ControlFlow::StructuredWithUnstructuredInternals;
   }
 
+  /// True when this construct is lowered structurally, yet its body holds
+  /// unstructured control flow that has to be folded into an
+  /// scf.execute_region.
+  bool lowerBodyAsWrappedRegion() const;
+
   bool lowerAsStructured() const;
   bool lowerAsUnstructured() const;
   bool forceAsUnstructured() const;
diff --git a/flang/lib/Lower/Bridge.cpp b/flang/lib/Lower/Bridge.cpp
index 6d3331b164dca..ed362f63e7778 100644
--- a/flang/lib/Lower/Bridge.cpp
+++ b/flang/lib/Lower/Bridge.cpp
@@ -1125,6 +1125,11 @@ class FirConverter : public Fortran::lower::AbstractConverter {
     genFIR(eval, unstructuredContext);
   }
 
+  void genLoopBodyEvaluations(
+      Fortran::lower::pft::Evaluation &loopEval) override final {
+    genLoopBodyEvaluations(loopEval, /*unstructuredContext=*/true);
+  }
+
   //===--------------------------------------------------------------------===//
   // Utility methods
   //===--------------------------------------------------------------------===//
@@ -2642,6 +2647,97 @@ class FirConverter : public Fortran::lower::AbstractConverter {
     return wrapOp;
   }
 
+  /// Wrap a loop's *body* -- not the construct, and not the loop control -- in
+  /// an scf.execute_region. A category (c) loop keeps its structured loop op
+  /// while its self-contained raw branching lives inside the region, which may
+  /// hold as many blocks as it needs.
+  ///
+  /// The builder must already be positioned inside the loop body. On return it
+  /// is inside the region. Returns null when the loop has no such internals.
+  mlir::scf::ExecuteRegionOp
+  wrapUnstructuredBody(Fortran::lower::pft::Evaluation &eval,
+                       mlir::Block *&yieldBlock,
+                       mlir::Block *&savedEndDoBlock) {
+    if (!eval.lowerBodyAsWrappedRegion())
+      return nullptr;
+
+    Fortran::lower::pft::EvaluationList &list = eval.getNestedEvaluations();
+    mlir::Location loc = toLocation();
+    auto wrapOp =
+        mlir::scf::ExecuteRegionOp::create(*builder, loc, mlir::TypeRange{},
+                                           /*noInline=*/builder->getUnitAttr());
+    ++wrapUnstructuredCount;
+    mlir::Block *entry = builder->createBlock(&wrapOp.getRegion());
+    builder->setInsertionPointToEnd(entry);
+    // Only the body: both loop control statements live in the enclosing region
+    // and may be branched to from outside the loop, so neither may take its
+    // block from this region. Dropping only the EndDoStmt leaves the DO
+    // statement to be given a block here, which any GOTO targeting the loop
+    // head would then reference across a region boundary.
+    createEmptyBlocksIn(
+        llvm::make_range(std::next(list.begin()), std::prev(list.end())));
+    yieldBlock = builder->createBlock(&wrapOp.getRegion());
+    builder->setInsertionPointToEnd(yieldBlock);
+    mlir::scf::YieldOp::create(*builder, loc);
+
+    // A CYCLE targets the EndDoStmt, which is the boundary between the loop
+    // body and the loop control. Inside the wrap that boundary is the region's
+    // yield block, so a CYCLE leaves the region rather than branching to a
+    // block the region cannot name.
+    savedEndDoBlock = list.back().block;
+    list.back().block = yieldBlock;
+
+    builder->setInsertionPointToEnd(entry);
+    return wrapOp;
+  }
+
+  /// Emit a loop's evaluations, optionally folding the body into an
+  /// scf.execute_region. The loop control statements -- the first and last
+  /// evaluations -- are emitted exactly as they are for a structured loop:
+  /// same context flag, same region, same blocks. They may be branch targets
+  /// from outside the loop, so their blocks have to stay in the enclosing
+  /// region. Only what lies strictly between them goes inside the wrap.
+  void genLoopBodyEvaluations(Fortran::lower::pft::Evaluation &eval,
+                              bool unstructuredContext) {
+    Fortran::lower::pft::EvaluationList &list = eval.getNestedEvaluations();
+    auto iter = list.begin();
+    auto end = std::prev(list.end());
+
+    // The loop control statement, outside any wrap.
+    if (iter != end) {
+      genFIR(*iter, unstructuredContext);
+      ++iter;
+    }
+
+    mlir::Block *yieldBlock = nullptr;
+    mlir::Block *savedEndDoBlock = nullptr;
+    mlir::scf::ExecuteRegionOp wrapOp =
+        wrapUnstructuredBody(eval, yieldBlock, savedEndDoBlock);
+    for (; iter != end; ++iter)
+      genFIR(*iter, unstructuredContext || wrapOp);
+    closeUnstructuredBodyWrap(wrapOp, eval, yieldBlock, savedEndDoBlock);
+  }
+
+  /// Finalize a wrap created by wrapUnstructuredBody: restore the EndDoStmt
+  /// block, fall through to the region's yield, and resume after the wrap op
+  /// so the caller emits the loop control outside it.
+  void closeUnstructuredBodyWrap(mlir::scf::ExecuteRegionOp wrapOp,
+                                 Fortran::lower::pft::Evaluation &eval,
+                                 mlir::Block *yieldBlock,
+                                 mlir::Block *savedEndDoBlock) {
+    if (!wrapOp)
+      return;
+
+    eval.getNestedEvaluations().back().block = savedEndDoBlock;
+
+    if (mlir::Block *current = builder->getBlock())
+      if (current->empty() ||
+          !current->back().hasTrait<mlir::OpTrait::IsTerminator>())
+        genBranch(yieldBlock);
+
+    builder->setInsertionPointAfter(wrapOp);
+  }
+
   /// Finalize a wrap created by wrapUnstructuredConstruct: restore the
   /// original exit block and set the insertion point after the wrap op.
   void closeUnstructuredWrap(mlir::scf::ExecuteRegionOp wrapOp,
@@ -2669,9 +2765,10 @@ class FirConverter : public Fortran::lower::AbstractConverter {
     // skip generating any loop — just lower the body.  The IV value is
     // already available from the parent acc.loop's block argument.
     if (Fortran::lower::isCollapsedDoConstruct(doConstruct)) {
-      auto iter = eval.getNestedEvaluations().begin();
-      for (auto end = --eval.getNestedEvaluations().end(); iter != end; ++iter)
-        genFIR(*iter, unstructuredContext);
+      // The parent acc.loop supplies the iteration, so no loop op is built
+      // here for a wrap to sit inside. The body may still hold self-contained
+      // raw branching, so wrap it directly.
+      genLoopBodyEvaluations(eval, unstructuredContext);
       return;
     }
 
@@ -2694,11 +2791,10 @@ class FirConverter : public Fortran::lower::AbstractConverter {
                 builder->getInsertionPoint()->getBlock()->getParent()) &&
             "builder insertion point is not inside the newly generated loop");
 
-        // Loop body code.
-        auto iter = eval.getNestedEvaluations().begin();
-        for (auto end = --eval.getNestedEvaluations().end(); iter != end;
-             ++iter)
-          genFIR(*iter, unstructuredContext);
+        // The acc.loop supplies the iteration, so the body is emitted here
+        // rather than through the increment-loop path below. Wrap it when it
+        // holds self-contained raw branching.
+        genLoopBodyEvaluations(eval, unstructuredContext);
 
         builder->setInsertionPointAfter(loopOp);
         return;
@@ -2847,10 +2943,12 @@ class FirConverter : public Fortran::lower::AbstractConverter {
     if (!infiniteLoop && !whileCondition)
       genFIRIncrementLoopBegin(incrementLoopNestInfo, doStmtEval.dirs);
 
+    // The loop control is structured, but the body may hold raw branching
+    // confined to it. Wrap the body, leaving the loop control outside, so the
+    // structured loop op's single-block region stays well formed.
     // Loop body code.
-    auto iter = eval.getNestedEvaluations().begin();
-    for (auto end = --eval.getNestedEvaluations().end(); iter != end; ++iter)
-      genFIR(*iter, unstructuredContext);
+    genLoopBodyEvaluations(eval, unstructuredContext);
+    auto iter = std::prev(eval.getNestedEvaluations().end());
 
     // An EndDoStmt in unstructured code may start a new block.
     Fortran::lower::pft::Evaluation &endDoEval = *iter;
@@ -6474,6 +6572,16 @@ class FirConverter : public Fortran::lower::AbstractConverter {
   /// boundaries.
   void createEmptyBlocks(
       std::list<Fortran::lower::pft::Evaluation> &evaluationList) {
+    createEmptyBlocksIn(
+        llvm::make_range(evaluationList.begin(), evaluationList.end()));
+  }
+
+  /// createEmptyBlocks over a sub-range of an evaluation list. A body-only wrap
+  /// must not pre-create blocks for the loop control statements, which live in
+  /// the enclosing region.
+  void createEmptyBlocksIn(
+      llvm::iterator_range<Fortran::lower::pft::EvaluationList::iterator>
+          evaluationList) {
     mlir::Region *region = &builder->getRegion();
     for (Fortran::lower::pft::Evaluation &eval : evaluationList) {
       if (eval.isNewBlock)
diff --git a/flang/lib/Lower/OpenMP/OpenMP.cpp b/flang/lib/Lower/OpenMP/OpenMP.cpp
index a815cb2298719..b23473cf69171 100644
--- a/flang/lib/Lower/OpenMP/OpenMP.cpp
+++ b/flang/lib/Lower/OpenMP/OpenMP.cpp
@@ -1018,6 +1018,14 @@ static void genNestedEvaluations(lower::AbstractConverter &converter,
                                  int collapseValue = 0) {
   lower::pft::Evaluation *curEval = getCollapsedLoopEval(eval, collapseValue);
 
+  // The directive consumes the DO itself, so genFIR(DoConstruct) -- where a
+  // plain loop's body is wrapped -- is never reached. Go through the same
+  // helper here, so the loop control statements are emitted outside the wrap
+  // exactly as they are for a structured loop.
+  if (curEval->isA<parser::DoConstruct>() && curEval->hasNestedEvaluations()) {
+    converter.genLoopBodyEvaluations(*curEval);
+    return;
+  }
   for (lower::pft::Evaluation &e : curEval->getNestedEvaluations())
     converter.genEval(e);
 }
diff --git a/flang/lib/Lower/PFTBuilder.cpp b/flang/lib/Lower/PFTBuilder.cpp
index 948589a11b7d7..779ab41dab58f 100644
--- a/flang/lib/Lower/PFTBuilder.cpp
+++ b/flang/lib/Lower/PFTBuilder.cpp
@@ -1735,6 +1735,10 @@ bool Fortran::lower::pft::Evaluation::lowerAsUnstructured() const {
   return isUnstructured() || clDisableStructuredFir;
 }
 
+bool Fortran::lower::pft::Evaluation::lowerBodyAsWrappedRegion() const {
+  return hasUnstructuredInternals() && !lowerAsUnstructured();
+}
+
 bool Fortran::lower::pft::Evaluation::forceAsUnstructured() const {
   return clDisableStructuredFir;
 }
diff --git a/flang/test/Lower/OpenACC/acc-cache.f90 b/flang/test/Lower/OpenACC/acc-cache.f90
index 12bdee3833870..151ef27a96ddc 100644
--- a/flang/test/Lower/OpenACC/acc-cache.f90
+++ b/flang/test/Lower/OpenACC/acc-cache.f90
@@ -387,11 +387,11 @@ subroutine test_cache_nonunit_lb()
 
 ! For arr(10:20), startIdx = 10, element 15 has lowerbound = 15 - 10 = 5
 ! CHECK: %[[C10:.*]] = arith.constant 10 : index
-! Unstructured loop with SELECT CASE: acc.loop becomes unstructured
-! CHECK: acc.loop private({{.*}}) {
-! CHECK: cf.br ^[[HEADER:.*]]
-! CHECK: ^[[HEADER]]:
-! CHECK: cf.cond_br %{{.*}}, ^[[BODY:.*]], ^[[EXIT:.*]]
+! The SELECT CASE branches only within the loop body, so acc.loop keeps its
+! structured control and the raw blocks live in a wrap inside it.
+! CHECK: acc.loop private({{.*}})
+! CHECK: scf.execute_region no_inline {
+! CHECK: cf.br ^[[BODY:.*]]
 ! CHECK: ^[[BODY]]:
 ! Compute lowerbound = 15 - startIdx = 15 - 10 = 5
 ! CHECK: %[[C1:.*]] = arith.constant 1 : index
@@ -420,13 +420,12 @@ subroutine test_cache_nonunit_lb()
 ! CHECK: hlfir.designate %[[DECL]]#0
 ! CHECK: hlfir.assign
 ! CHECK: cf.br ^[[MERGE]]
-! All SELECT CASE branches converge, then loop back or exit
+! All SELECT CASE branches converge, then leave the wrap. The loop control is
+! acc.loop's own, so the region ends at its yield rather than a back edge.
 ! CHECK: ^[[MERGE]]:
-! CHECK: cf.br ^[[HEADER]]
+! CHECK: cf.br ^[[EXIT:.*]]
 ! CHECK: ^[[EXIT]]:
-! Scope termination: acc.yield marks end of cache scope
-! CHECK: acc.yield
-! CHECK-NEXT: } {{.*}}unstructured{{.*}}
+! CHECK: scf.yield
 end subroutine
 
 ! CHECK-LABEL: func.func @_QPtest_cache_use_after_region()
diff --git a/flang/test/Lower/OpenACC/acc-unstructured-internals.f90 b/flang/test/Lower/OpenACC/acc-unstructured-internals.f90
new file mode 100644
index 0000000000000..f0be621384a6a
--- /dev/null
+++ b/flang/test/Lower/OpenACC/acc-unstructured-internals.f90
@@ -0,0 +1,73 @@
+! RUN: bbc -fopenacc -emit-hlfir -o - %s | FileCheck %s
+
+! Loops under an OpenACC directive whose control flow is structured on the
+! outside but whose branching is confined to the body. The directive's own
+! code-gen consumes the DO, so the body is wrapped at the directive's body
+! lowering site rather than in genFIR(DoConstruct). The loop keeps its bounds
+! on the op -- control(...) rather than a cf trip-count test -- so it is still
+! available to be parallelized.
+
+! A forward GOTO raised inside a nested IF, jumping over a whole inner DO and
+! landing on the last statement of the outer loop body. Both endpoints are
+! inside the body, so the outer loop stays an acc.loop with control(...) and
+! the two inner loops stay acc.loops of their own.
+subroutine kernels_goto_over_inner(qfx, a, its, ite, jts, jte, force, flux)
+  real :: qfx(ite,jte), a(ite,jte)
+  logical :: force
+  integer :: flux
+  !$acc kernels
+  do j = jts, jte
+    do 330 i = its, ite
+      a(i,j) = a(i,j) + 1.0
+330 continue
+    if (force) then
+      if (flux .eq. 1) goto 350
+    endif
+    do i = its, ite
+      qfx(i,j) = 0.
+    enddo
+350 continue
+  enddo
+  !$acc end kernels
+end subroutine
+
+! CHECK-LABEL: func.func @_QPkernels_goto_over_inner
+! CHECK:         acc.kernels {
+! CHECK:           acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK:             scf.execute_region no_inline {
+! The inner loops are untouched by the wrap around the outer body.
+! CHECK:               acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK:                 acc.yield
+! The GOTO and the branches it feeds stay inside the region.
+! CHECK:               cf.cond_br %{{[0-9]+}}, ^bb[[THEN:[0-9]+]], ^bb[[SKIP:[0-9]+]]
+! CHECK:               acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK:                 acc.yield
+! CHECK:               scf.yield
+! CHECK:             acc.yield
+
+! A CYCLE targets the EndDoStmt, the boundary between the loop body and the
+! loop control, so inside the wrap it leaves the region at its yield.
+subroutine parallel_loop_cycle(a, n)
+  real :: a(n)
+  !$acc parallel loop
+  do i = 1, n
+    if (a(i) > 0.0) then
+      a(i) = 1.0
+      cycle
+    end if
+    a(i) = 2.0
+  end do
+end subroutine
+
+! CHECK-LABEL: func.func @_QPparallel_loop_cycle
+! CHECK:         acc.parallel combined(loop) {
+! CHECK:           acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK:             scf.execute_region no_inline {
+! CHECK:               cf.cond_br %{{[0-9]+}}, ^bb[[CYCLE:[0-9]+]], ^bb[[BODY:[0-9]+]]
+! CHECK:             ^bb[[CYCLE]]:
+! CHECK:               cf.br ^bb[[EXIT:[0-9]+]]
+! CHECK:             ^bb[[BODY]]:
+! CHECK:               cf.br ^bb[[EXIT]]
+! CHECK:             ^bb[[EXIT]]:
+! CHECK:               scf.yield
+! CHECK:             acc.yield
diff --git a/flang/test/Lower/OpenACC/acc-unstructured.f90 b/flang/test/Lower/OpenACC/acc-unstructured.f90
index cbb27d74cc96c..6a95f664f78df 100644
--- a/flang/test/Lower/OpenACC/acc-unstructured.f90
+++ b/flang/test/Lower/OpenACC/acc-unstructured.f90
@@ -275,19 +275,18 @@ subroutine test_unstructured_collapse_cycle(a)
 ! Both induction variables (j and i) are privatized:
 ! CHECK: %[[PRIVJ:.*]] = acc.private varPtr(%{{.*}} : !fir.ref<i32>) recipe(@privatization_ref_i32) implicit(true) name("j") -> !fir.ref<i32>
 ! CHECK: %[[PRIVI:.*]] = acc.private varPtr(%{{.*}} : !fir.ref<i32>) recipe(@privatization_ref_i32) implicit(true) name("i") -> !fir.ref<i32>
-! No control(...) on acc.loop — bounds are not on the op:
 ! CHECK: acc.loop combined(serial) private(%[[PRIVJ]], %[[PRIVI]] : !fir.ref<i32>, !fir.ref<i32>) {
-! Outer loop trip-count test (j) emitted as cf:
-! CHECK: arith.cmpi sgt
-! CHECK: cf.cond_br
-! Inner loop trip-count test (i) emitted as cf:
-! CHECK: arith.cmpi sgt
-! CHECK: cf.cond_br
-! The if/cycle is a structured cf branch in the body:
+! The IF-guarded CYCLE branches only within the body, so both loops keep their
+! bounds on the op -- control(...) rather than a cf trip-count test -- and the
+! raw blocks are confined to a wrap inside each body.
+! CHECK: acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK: scf.execute_region no_inline {
+! CHECK: acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK: scf.execute_region no_inline {
 ! CHECK: arith.cmpi eq
 ! CHECK: cf.cond_br
+! CHECK: scf.yield
 ! CHECK: acc.yield
-! CHECK: }
 
 ! `acc serial loop collapse(N)` with STOP in body: wrap-in-execute-region hides
 ! the unstructured if/stop and the three collapsed iterators lower as a single
diff --git a/flang/test/Lower/OpenMP/Todo/metadirective-loop-unstructured.f90 b/flang/test/Lower/OpenMP/Todo/metadirective-loop-unstructured.f90
deleted file mode 100644
index 08ebcc7b4c8ef..0000000000000
--- a/flang/test/Lower/OpenMP/Todo/metadirective-loop-unstructured.f90
+++ /dev/null
@@ -1,39 +0,0 @@
-! Defer unstructured associated loops until every selection path can give its
-! PFT blocks an independent mapping.
-
-! RUN: split-file %s %t
-! RUN: %not_todo_cmd %flang_fc1 -emit-hlfir -fopenmp -fopenmp-version=52 -o - %t/static.f90 2>&1 | FileCheck %s
-! RUN: %not_todo_cmd %flang_fc1 -emit-hlfir -fopenmp -fopenmp-version=52 -o - %t/runtime.f90 2>&1 | FileCheck %s
-
-! CHECK: not yet implemented: unstructured associated DO in loop-associated METADIRECTIVE variant
-
-!--- static.f90
-subroutine test_static(n, a, selector)
-  integer :: n, a(n), selector, i
-  !$omp metadirective &
-  !$omp & when(implementation={vendor(llvm)}: do) &
-  !$omp & otherwise(nothing)
-  do i = 1, n
-    go to (10, 20), selector
-10  a(i) = 1
-    go to 30
-20  a(i) = 2
-30  continue
-  end do
-end subroutine
-
-!--- runtime.f90
-subroutine test_runtime(flag, n, a, selector)
-  logical :: flag
-  integer :: n, a(n), selector, i
-  !$omp metadirective &
-  !$omp & when(user={condition(flag)}: do) &
-  !$omp & otherwise(nothing)
-  do i = 1, n
-    go to (10, 20), selector
-10  a(i) = 1
-    go to 30
-20  a(i) = 2
-30  continue
-  end do
-end subroutine
diff --git a/flang/test/Lower/OpenMP/metadirective-loop-unstructured.f90 b/flang/test/Lower/OpenMP/metadirective-loop-unstructured.f90
new file mode 100644
index 0000000000000..7c4197e7e92d7
--- /dev/null
+++ b/flang/test/Lower/OpenMP/metadirective-loop-unstructured.f90
@@ -0,0 +1,50 @@
+! A DO associated with a METADIRECTIVE variant whose body branches only within
+! itself. The computed GO TO and its targets are all inside the loop body, so
+! the loop keeps its structured form and the raw blocks are confined to an
+! scf.execute_region inside omp.loop_nest. This used to be rejected as not yet
+! implemented, because the body's blocks had nowhere to live.
+
+! RUN: split-file %s %t
+! RUN: %flang_fc1 -emit-hlfir -fopenmp -fopenmp-version=52 -o - %t/static.f90 | FileCheck %s
+! RUN: %flang_fc1 -emit-hlfir -fopenmp -fopenmp-version=52 -o - %t/runtime.f90 | FileCheck %s
+
+! CHECK:   omp.wsloop
+! CHECK:     omp.loop_nest
+! CHECK:       scf.execute_region no_inline {
+! The computed GO TO and the branches it feeds stay inside the region.
+! CHECK:         fir.select %{{[0-9]+}} : i32 [1, ^bb[[L10:[0-9]+]], 2, ^bb[[L20:[0-9]+]], unit, ^bb[[L10]]]
+! CHECK:       ^bb[[L10]]:
+! CHECK:       ^bb[[L20]]:
+! CHECK:         scf.yield
+! CHECK:       omp.yield
+
+!--- static.f90
+subroutine test_static(n, a, selector)
+  integer :: n, a(n), selector, i
+  !$omp metadirective &
+  !$omp & when(implementation={vendor(llvm)}: do) &
+  !$omp & otherwise(nothing)
+  do i = 1, n
+    go to (10, 20), selector
+10  a(i) = 1
+    go to 30
+20  a(i) = 2
+30  continue
+  end do
+end subroutine
+
+!--- runtime.f90
+subroutine test_runtime(flag, n, a, selector)
+  logical :: flag
+  integer :: n, a(n), selector, i
+  !$omp metadirective &
+  !$omp & when(user={condition(flag)}: do) &
+  !$omp & otherwise(nothing)
+  do i = 1, n
+    go to (10, 20), selector
+10  a(i) = 1
+    go to 30
+20  a(i) = 2
+30  continue
+  end do
+end subroutine
diff --git a/flang/test/Lower/OpenMP/wsloop-unstructured-cycle.f90 b/flang/test/Lower/OpenMP/wsloop-unstructured-cycle.f90
index 60fe63b8d29af..5a76ee4b91a6c 100644
--- a/flang/test/Lower/OpenMP/wsloop-unstructured-cycle.f90
+++ b/flang/test/Lower/OpenMP/wsloop-unstructured-cycle.f90
@@ -1,13 +1,10 @@
-! RUN: bbc --wrap-unstructured-constructs-in-execute-region -emit-hlfir -fopenmp -o - %s | FileCheck %s --implicit-check-not=scf.execute_region
+! RUN: bbc --wrap-unstructured-constructs-in-execute-region -emit-hlfir -fopenmp -o - %s | FileCheck %s
 
 ! A DO associated with an OpenMP loop directive is lowered by the directive's
-! own code-gen. Such a DO must never be folded into an
-! scf.execute_region, even when wrapping is enabled and the loop is
-! unstructured -- here the IF-guarded CYCLE makes it so. The body's blocks
-! stay flat inside omp.loop_nest.
-!
-! --implicit-check-not on the RUN line asserts that no wrapping takes place
-! anywhere in the output.
+! own code-gen, which never reaches genFIR(DoConstruct) where a plain loop's
+! body is wrapped. The body is wrapped at the directive's own body-lowering
+! site instead, so a loop whose branching is confined to its body -- here an
+! IF-guarded CYCLE -- keeps its structured form inside omp.loop_nest.
 
 subroutine repro_final(x, y, n)
   implicit none
@@ -27,26 +24,29 @@ subroutine repro_final(x, y, n)
 
 end subroutine repro_final
 
+! The CYCLE targets the EndDoStmt, which inside the wrap is the region's yield
+! block -- so it leaves the region instead of branching to a block outside it.
 ! CHECK-LABEL: func.func @_QPrepro_final(
 ! CHECK:         omp.wsloop
 ! CHECK:           omp.loop_nest
 ! CHECK:             hlfir.assign
-! CHECK:             cf.br ^bb[[TEST:[0-9]+]]
-! CHECK:           ^bb[[TEST]]:
-! CHECK:             arith.cmpf ogt
-! CHECK:             cf.cond_br %{{[0-9]+}}, ^bb[[CYCLE:[0-9]+]], ^bb[[BODY:[0-9]+]]
-! CHECK:           ^bb[[CYCLE]]:
-! CHECK:             hlfir.assign
-! CHECK:             cf.br ^bb[[EXIT:[0-9]+]]
-! CHECK:           ^bb[[BODY]]:
-! CHECK:             hlfir.assign
-! CHECK:             cf.br ^bb[[EXIT]]
-! CHECK:           ^bb[[EXIT]]:
+! CHECK:             scf.execute_region no_inline {
+! CHECK:             ^bb[[TEST:[0-9]+]]:
+! CHECK:               arith.cmpf ogt
+! CHECK:               cf.cond_br %{{[0-9]+}}, ^bb[[CYCLE:[0-9]+]], ^bb[[BODY:[0-9]+]]
+! CHECK:             ^bb[[CYCLE]]:
+! CHECK:               hlfir.assign
+! CHECK:               cf.br ^bb[[EXIT:[0-9]+]]
+! CHECK:             ^bb[[BODY]]:
+! CHECK:               hlfir.assign
+! CHECK:               cf.br ^bb[[EXIT]]
+! CHECK:             ^bb[[EXIT]]:
+! CHECK:               scf.yield
 ! CHECK:             omp.yield
 
-! COLLAPSE(n) and ORDERED(n) both associate n loops with the directive, and
-! the loop transforming directives (TILE, INTERCHANGE, ...) associate as many
-! as their arguments describe. None of the associated loops may be wrapped.
+! COLLAPSE(n) and ORDERED(n) both associate n loops with the directive. The
+! associated loops are collapsed into a single omp.loop_nest, and the wrap goes
+! around the innermost body it encloses.
 
 subroutine collapse_case(x, y, n)
   implicit none
@@ -68,12 +68,14 @@ subroutine collapse_case(x, y, n)
 
 end subroutine collapse_case
 
-! Both loops are associated with the directive, so the body stays flat inside
-! omp.loop_nest.
+! Both loops are associated with the directive, so one wrap covers the body of
+! the collapsed nest.
 ! CHECK-LABEL: func.func @_QPcollapse_case(
 ! CHECK:         omp.wsloop
 ! CHECK:           omp.loop_nest ({{.*}}) {{.*}} collapse(2) {
-! CHECK:             cf.cond_br
+! CHECK:             scf.execute_region no_inline {
+! CHECK:               cf.cond_br
+! CHECK:               scf.yield
 ! CHECK:             omp.yield
 
 subroutine ordered_case(x, y, n)
@@ -96,12 +98,16 @@ subroutine ordered_case(x, y, n)
 
 end subroutine ordered_case
 
-! ORDERED(2) associates the inner loop with the directive as well, so it is
-! not wrapped either.
+! ORDERED(2) keeps the inner loop as a loop of its own inside the nest, so the
+! outer body and the inner body each get a wrap.
 ! CHECK-LABEL: func.func @_QPordered_case(
 ! CHECK:         omp.wsloop ordered(2)
 ! CHECK:           omp.loop_nest
-! CHECK:             cf.cond_br
+! CHECK:             scf.execute_region no_inline {
+! CHECK:               scf.execute_region no_inline {
+! CHECK:                 cf.cond_br
+! CHECK:                 scf.yield
+! CHECK:               scf.yield
 ! CHECK:             omp.yield
 
 ! A TILE case belongs here too, since the SIZES arguments decide how many
diff --git a/flang/test/Lower/OpenMP/wsloop-unstructured-internals.f90 b/flang/test/Lower/OpenMP/wsloop-unstructured-internals.f90
new file mode 100644
index 0000000000000..5ceba4f3d3875
--- /dev/null
+++ b/flang/test/Lower/OpenMP/wsloop-unstructured-internals.f90
@@ -0,0 +1,44 @@
+! RUN: bbc -fopenmp -emit-hlfir -o - %s | FileCheck %s
+
+! A loop associated with an OpenMP loop directive whose branching is confined
+! to its body. The directive's code-gen consumes the DO, so the body is wrapped
+! at the directive's body lowering site rather than in genFIR(DoConstruct). The
+! loop stays a single omp.loop_nest -- it is not dissolved into a cf trip-count
+! loop -- so it is still available to be worksharded.
+
+! A forward GOTO raised inside a nested IF, jumping over a whole inner DO and
+! landing on the last statement of the outer loop body. Both endpoints are in
+! the body, so the associated loop keeps its structured form and the two inner
+! loops stay plain fir.do_loops inside the wrap.
+subroutine omp_goto_over_inner(qfx, a, n, force, flux)
+  real :: qfx(n,n), a(n,n)
+  logical :: force
+  integer :: flux
+  !$omp parallel do
+  do j = 1, n
+    do 330 i = 1, n
+      a(i,j) = a(i,j) + 1.0
+330 continue
+    if (force) then
+      if (flux .eq. 1) goto 350
+    endif
+    do i = 1, n
+      qfx(i,j) = 0.
+    enddo
+350 continue
+  enddo
+  !$omp end parallel do
+end subroutine
+
+! CHECK-LABEL: func.func @_QPomp_goto_over_inner
+! CHECK:         omp.wsloop
+! CHECK:           omp.loop_nest (%{{.*}}) : i32 = (%{{.*}}) to (%{{.*}}) inclusive step (%{{.*}}) {
+! CHECK:             scf.execute_region no_inline {
+! The inner loops are untouched by the wrap around the outer body.
+! CHECK:               fir.do_loop %{{.*}} = %{{.*}} to %{{.*}} step %{{.*}} : i32 {
+! The GOTO and the branches it feeds stay inside the region.
+! CHECK:               cf.cond_br %{{[0-9]+}}, ^bb[[THEN:[0-9]+]], ^bb[[SKIP:[0-9]+]]
+! CHECK:               cf.cond_br %{{[0-9]+}}, ^bb[[GOTO:[0-9]+]], ^bb[[SKIP]]
+! CHECK:               fir.do_loop %{{.*}} = %{{.*}} to %{{.*}} step %{{.*}} : i32 {
+! CHECK:               scf.yield
+! CHECK:             omp.yield
diff --git a/flang/test/Lower/always-execute-loop-body.f90 b/flang/test/Lower/always-execute-loop-body.f90
index 33937390a0804..2df5d5a181dbd 100644
--- a/flang/test/Lower/always-execute-loop-body.f90
+++ b/flang/test/Lower/always-execute-loop-body.f90
@@ -2,6 +2,10 @@
 
 ! Given the flag `--always-execute-loop-body` the compiler emits an extra
 ! code to change the trip count, test tries to verify the extra emitted HLFIR.
+!
+! The trip count only exists on the unstructured lowering path, so the loop has
+! to stay unstructured for the flag to have anything to act on: the GOTO leaves
+! the loop, which keeps it out of the structured form.
 
 ! CHECK-LABEL: func.func @_QPsome
 subroutine some()
@@ -17,7 +21,7 @@ subroutine some()
   ! CHECK: %[[CMP:.*]] = arith.cmpi sgt, %[[LOADED_TRIP]], %c0{{.*}} : i32
   ! CHECK: cf.cond_br %[[CMP]]
   do i=4,1,1
-    stop 2
+    goto 9
   end do
-  return
+9 return
 end
diff --git a/flang/test/Lower/do-loop-branch-to-loop-header.f90 b/flang/test/Lower/do-loop-branch-to-loop-header.f90
new file mode 100644
index 0000000000000..d7a0e8fea430d
--- /dev/null
+++ b/flang/test/Lower/do-loop-branch-to-loop-header.f90
@@ -0,0 +1,55 @@
+! RUN: bbc -emit-fir -o - %s | FileCheck %s
+
+! A DO statement is a valid branch target (F2023 11.2.1), and branching to it
+! restarts the loop: the DO construct becomes active again, its bounds are
+! re-evaluated and the iteration count re-established (F2023 11.1.7.3 p1,
+! 11.1.7.4.1 p1).
+!
+! A loop whose body branching is self-contained keeps its structured form, with
+! the body folded into an scf.execute_region. The loop control statements are
+! not part of that body: they may be branched to from outside the loop, so
+! their blocks have to stay in the enclosing region. Emitting the DO statement
+! inside the wrap gave it a block the enclosing code could not reference.
+
+! The ASSIGN makes a body statement a new block, so the body is wrapped; the
+! GOTO targets the loop header from outside the loop.
+subroutine branch_to_header(a, b)
+  real :: a(10), b(10)
+  integer :: m
+  go to 80
+40 do 42 i = 1, 10
+     assign 41 to m
+41   a(i) = b(i)
+42   continue
+80 continue
+86 go to 40
+end subroutine
+
+! CHECK-LABEL: func.func @_QPbranch_to_header
+! The loop header is a branch target, so it starts a block in the function's
+! own region -- the same region the branch is emitted from.
+! CHECK:         cf.br ^bb[[HEADER:[0-9]+]]
+! CHECK:       ^bb[[HEADER]]:
+! CHECK:         fir.do_loop
+! The body, and only the body, lives in the region.
+! CHECK:           scf.execute_region no_inline {
+! CHECK:             scf.yield
+! CHECK:           }
+! CHECK:         }
+
+! Same shape with the ASSIGN target outside the loop: the body needs no blocks
+! of its own, so no wrap is emitted at all and the loop stays plain.
+subroutine no_wrap_needed(a, b)
+  real :: a(10), b(10)
+  integer :: m
+  go to 80
+40 do 42 i = 1, 10
+     assign 80 to m
+42   a(i) = b(i)
+80 continue
+86 go to 40
+end subroutine
+
+! CHECK-LABEL: func.func @_QPno_wrap_needed
+! CHECK-NOT:     scf.execute_region
+! CHECK:         fir.do_loop
diff --git a/flang/test/Lower/do_loop_unstructured.f90 b/flang/test/Lower/do_loop_unstructured.f90
index a092590077303..502c07a2c2ad1 100644
--- a/flang/test/Lower/do_loop_unstructured.f90
+++ b/flang/test/Lower/do_loop_unstructured.f90
@@ -16,35 +16,13 @@ subroutine simple_unstructured()
     404 continue
   end do
 end subroutine
+! The GOTO targets the statement that follows it, so nothing branches out of
+! the body and the loop keeps its structured form.
 ! CHECK-LABEL: simple_unstructured
-! CHECK:   %[[TRIP_VAR_REF:.*]] = fir.alloca i32
 ! CHECK:   %[[LOOP_VAR_REF:.*]] = fir.alloca i32 {bindc_name = "i", uniq_name = "_QFsimple_unstructuredEi"}
-! CHECK:   %[[LOOP_VAR_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_REF]]
-! CHECK:   %[[ONE:.*]] = arith.constant 1 : i32
-! CHECK:   %[[HUNDRED:.*]] = arith.constant 100 : i32
-! CHECK:   %[[STEP_ONE:.*]] = arith.constant 1 : i32
-! CHECK:   %[[TMP1:.*]] = arith.subi %[[HUNDRED]], %[[ONE]] : i32
-! CHECK:   %[[TMP2:.*]] = arith.addi %[[TMP1]], %[[STEP_ONE]] : i32
-! CHECK:   %[[TRIP_COUNT:.*]] = arith.divsi %[[TMP2]], %[[STEP_ONE]] : i32
-! CHECK:   fir.store %[[TRIP_COUNT]] to %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK:   fir.store %[[ONE]] to %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK:   cf.br ^[[HEADER:.*]]
-! CHECK: ^[[HEADER]]:
-! CHECK:   %[[TRIP_VAR:.*]] = fir.load %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK:   %[[ZERO:.*]] = arith.constant 0 : i32
-! CHECK:   %[[COND:.*]] = arith.cmpi sgt, %[[TRIP_VAR]], %[[ZERO]] : i32
-! CHECK:   cf.cond_br %[[COND]], ^[[BODY:.*]], ^[[EXIT:.*]]
-! CHECK: ^[[BODY]]:
-! CHECK:   %[[TRIP_VAR:.*]] = fir.load %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK:   %[[ONE_1:.*]] = arith.constant 1 : i32
-! CHECK:   %[[TRIP_VAR_NEXT:.*]] = arith.subi %[[TRIP_VAR]], %[[ONE_1]] : i32
-! CHECK:   fir.store %[[TRIP_VAR_NEXT]] to %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK:   %[[LOOP_VAR:.*]] = fir.load %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK:   %[[STEP_ONE_2:.*]] = arith.constant 1 : i32
-! CHECK:   %[[LOOP_VAR_NEXT:.*]] = arith.addi %[[LOOP_VAR]], %[[STEP_ONE_2]] overflow<nsw> : i32
-! CHECK:   fir.store %[[LOOP_VAR_NEXT]] to %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK:   cf.br ^[[HEADER]]
-! CHECK: ^[[EXIT]]:
+! CHECK:   fir.do_loop %[[IV:.*]] = %c1{{.*}} to %c100{{.*}} step %c1{{.*}} : i32 {
+! CHECK:     fir.store %[[IV]] to
+! CHECK:   }
 ! CHECK:   return
 
 ! Test an unstructured loop with a step. Mostly similar to the previous one.
@@ -56,35 +34,11 @@ subroutine simple_unstructured_with_step()
     404 continue
   end do
 end subroutine
+! Same, with an explicit step.
 ! CHECK-LABEL: simple_unstructured_with_step
-! CHECK:   %[[TRIP_VAR_REF:.*]] = fir.alloca i32
-! CHECK:   %[[LOOP_VAR_REF:.*]] = fir.alloca i32 {bindc_name = "i", uniq_name = "_QFsimple_unstructured_with_stepEi"}
-! CHECK:   %[[LOOP_VAR_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_REF]]
-! CHECK:   %[[ONE:.*]] = arith.constant 1 : i32
-! CHECK:   %[[HUNDRED:.*]] = arith.constant 100 : i32
-! CHECK:   %[[STEP:.*]] = arith.constant 2 : i32
-! CHECK:   %[[TMP1:.*]] = arith.subi %[[HUNDRED]], %[[ONE]] : i32
-! CHECK:   %[[TMP2:.*]] = arith.addi %[[TMP1]], %[[STEP]] : i32
-! CHECK:   %[[TRIP_COUNT:.*]] = arith.divsi %[[TMP2]], %[[STEP]] : i32
-! CHECK:   fir.store %[[TRIP_COUNT]] to %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK:   fir.store %[[ONE]] to %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK:   cf.br ^[[HEADER:.*]]
-! CHECK: ^[[HEADER]]:
-! CHECK:   %[[TRIP_VAR:.*]] = fir.load %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK:   %[[ZERO:.*]] = arith.constant 0 : i32
-! CHECK:   %[[COND:.*]] = arith.cmpi sgt, %[[TRIP_VAR]], %[[ZERO]] : i32
-! CHECK:   cf.cond_br %[[COND]], ^[[BODY:.*]], ^[[EXIT:.*]]
-! CHECK: ^[[BODY]]:
-! CHECK:   %[[TRIP_VAR:.*]] = fir.load %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK:   %[[ONE_1:.*]] = arith.constant 1 : i32
-! CHECK:   %[[TRIP_VAR_NEXT:.*]] = arith.subi %[[TRIP_VAR]], %[[ONE_1]] : i32
-! CHECK:   fir.store %[[TRIP_VAR_NEXT]] to %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK:   %[[LOOP_VAR:.*]] = fir.load %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK:   %[[STEP_2:.*]] = arith.constant 2 : i32
-! CHECK:   %[[LOOP_VAR_NEXT:.*]] = arith.addi %[[LOOP_VAR]], %[[STEP_2]] overflow<nsw> : i32
-! CHECK:   fir.store %[[LOOP_VAR_NEXT]] to %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK:   cf.br ^[[HEADER]]
-! CHECK: ^[[EXIT]]:
+! CHECK:   fir.do_loop %[[IV:.*]] = %c1{{.*}} to %c100{{.*}} step %c2{{.*}} : i32 {
+! CHECK:     fir.store %[[IV]] to
+! CHECK:   }
 ! CHECK:   return
 
 ! Test a three nested unstructured loop. Three nesting is the basic case where
@@ -100,39 +54,12 @@ subroutine nested_unstructured()
     end do
   end do
 end subroutine
-! With the wrap-unstructured-constructs-in-execute-region pass, the innermost
-! k-loop is the only one classified unstructured (the `goto 404`/`404 continue`
-! pattern). It gets wrapped in scf.execute_region, and the outer i and j
-! loops fold back to fir.do_loop.
+! The innermost GOTO stays inside its own body, so every level stays
+! structured and no wrap is needed at all.
 ! CHECK-LABEL: nested_unstructured
-! CHECK:   %[[TRIP_VAR_K_REF:.*]] = fir.alloca i32
-! CHECK:   %[[LOOP_VAR_I_REF:.*]] = fir.alloca i32 {bindc_name = "i", uniq_name = "_QFnested_unstructuredEi"}
-! CHECK:   %[[LOOP_VAR_I_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_I_REF]]
-! CHECK:   %[[LOOP_VAR_J_REF:.*]] = fir.alloca i32 {bindc_name = "j", uniq_name = "_QFnested_unstructuredEj"}
-! CHECK:   %[[LOOP_VAR_J_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_J_REF]]
-! CHECK:   %[[LOOP_VAR_K_REF:.*]] = fir.alloca i32 {bindc_name = "k", uniq_name = "_QFnested_unstructuredEk"}
-! CHECK:   %[[LOOP_VAR_K_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_K_REF]]
-! CHECK:   fir.do_loop %{{[^ ]+}} = %{{.*}} to %{{.*}} step %{{.*}} : i32 {
-! CHECK:     fir.do_loop %{{[^ ]+}} = %{{.*}} to %{{.*}} step %{{.*}} : i32 {
-! CHECK:       scf.execute_region no_inline {
-! CHECK:         cf.br ^[[HEADER_K:.*]]
-! CHECK:       ^[[HEADER_K]]:
-! CHECK:         %[[TRIP_COUNT_K:.*]] = arith.divsi %{{.*}}, %{{.*}} : i32
-! CHECK:         fir.store %[[TRIP_COUNT_K]] to %[[TRIP_VAR_K_REF]] : !fir.ref<i32>
-! CHECK:         fir.store %{{.*}} to %[[LOOP_VAR_K_DECL]]#0 : !fir.ref<i32>
-! CHECK:         cf.br ^[[HEADER_K_BODY:.*]]
-! CHECK:       ^[[HEADER_K_BODY]]:
-! CHECK:         %[[TRIP_VAR_K:.*]] = fir.load %[[TRIP_VAR_K_REF]] : !fir.ref<i32>
-! CHECK:         %[[COND_K:.*]] = arith.cmpi sgt, %[[TRIP_VAR_K]], %{{.*}} : i32
-! CHECK:         cf.cond_br %[[COND_K]], ^[[BODY_K:.*]], ^[[EXIT_K:.*]]
-! CHECK:       ^[[BODY_K]]:
-! CHECK:         cf.br ^{{.*}}
-! CHECK:         cf.br ^[[HEADER_K_BODY]]
-! CHECK:       ^[[EXIT_K]]:
-! CHECK:         scf.yield
-! CHECK:       }
-! CHECK:     }
-! CHECK:   }
+! CHECK:   fir.do_loop %{{.*}} = %c1{{.*}} to %c100{{.*}} step %c1{{.*}} : i32 {
+! CHECK:     fir.do_loop %{{.*}} = %c1{{.*}} to %c200{{.*}} step %c1{{.*}} : i32 {
+! CHECK:       fir.do_loop %{{.*}} = %c1{{.*}} to %c300{{.*}} step %c1{{.*}} : i32 {
 ! CHECK:   return
 
 ! Test the existence of a structured loop inside an unstructured loop.
@@ -146,54 +73,13 @@ subroutine nested_structured_in_unstructured()
     404 continue
   end do
 end subroutine
+! The GOTO follows an inner loop, so the outer body needs raw blocks and is
+! wrapped, while the inner loop stays a plain fir.do_loop inside the wrap.
 ! CHECK-LABEL: nested_structured_in_unstructured
-! CHECK:   %[[TRIP_VAR_I_REF:.*]] = fir.alloca i32
-! CHECK:   %[[LOOP_VAR_I_REF:.*]] = fir.alloca i32 {bindc_name = "i", uniq_name = "_QFnested_structured_in_unstructuredEi"}
-! CHECK:   %[[LOOP_VAR_I_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_I_REF]]
-! CHECK:   %[[LOOP_VAR_J_REF:.*]] = fir.alloca i32 {bindc_name = "j", uniq_name = "_QFnested_structured_in_unstructuredEj"}
-! CHECK:   %[[LOOP_VAR_J_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_J_REF]]
-! CHECK:   %[[I_START:.*]] = arith.constant 1 : i32
-! CHECK:   %[[I_END:.*]] = arith.constant 100 : i32
-! CHECK:   %[[I_STEP:.*]] = arith.constant 1 : i32
-! CHECK:   %[[TMP1:.*]] = arith.subi %[[I_END]], %[[I_START]] : i32
-! CHECK:   %[[TMP2:.*]] = arith.addi %[[TMP1]], %[[I_STEP]] : i32
-! CHECK:   %[[TRIP_COUNT:.*]] = arith.divsi %[[TMP2]], %[[I_STEP]] : i32
-! CHECK:   fir.store %[[TRIP_COUNT]] to %[[TRIP_VAR_I_REF]] : !fir.ref<i32>
-! CHECK:   fir.store %[[I_START]] to %[[LOOP_VAR_I_DECL]]#0 : !fir.ref<i32>
-! CHECK:   cf.br ^[[HEADER:.*]]
-! CHECK: ^[[HEADER]]:
-! CHECK:   %[[TRIP_VAR:.*]] = fir.load %[[TRIP_VAR_I_REF]] : !fir.ref<i32>
-! CHECK:   %[[ZERO:.*]] = arith.constant 0 : i32
-! CHECK:   %[[COND:.*]] = arith.cmpi sgt, %[[TRIP_VAR]], %[[ZERO]] : i32
-! CHECK:   cf.cond_br %[[COND]], ^[[BODY:.*]], ^[[EXIT:.*]]
-! CHECK: ^[[BODY]]:
-! CHECK:   fir.do_loop %[[J_IV:[^ ]*]] =
-! CHECK-SAME: %[[J_LB:[^ ]*]] to %[[J_UB:[^ ]*]] step %[[J_ST:[^ ]*]] : i32 {
-! CHECK:     fir.store %[[J_IV]] to %[[LOOP_VAR_J_DECL]]#0 : !fir.ref<i32>
-! CHECK:   }
-! CHECK:   %[[J_LBIDX:.*]] = fir.convert %[[J_LB]] : (i32) -> index
-! CHECK:   %[[J_UBIDX:.*]] = fir.convert %[[J_UB]] : (i32) -> index
-! CHECK:   %[[J_STIDX:.*]] = fir.convert %[[J_ST]] : (i32) -> index
-! CHECK:   %[[J_C0:.*]] = arith.constant 0 : index
-! CHECK:   %[[J_DIFF:.*]] = arith.subi %[[J_UBIDX]], %[[J_LBIDX]] : index
-! CHECK:   %[[J_ADD:.*]] = arith.addi %[[J_DIFF]], %[[J_STIDX]] : index
-! CHECK:   %[[J_TRIP:.*]] = arith.divsi %[[J_ADD]], %[[J_STIDX]] : index
-! CHECK:   %[[J_CMP:.*]] = arith.cmpi slt, %[[J_TRIP]], %[[J_C0]] : index
-! CHECK:   %[[J_SEL:.*]] = arith.select %[[J_CMP]], %[[J_C0]], %[[J_TRIP]] : index
-! CHECK:   %[[J_MUL:.*]] = arith.muli %[[J_SEL]], %[[J_STIDX]] : index
-! CHECK:   %[[J_LASTIDX:.*]] = arith.addi %[[J_LBIDX]], %[[J_MUL]] : index
-! CHECK:   %[[J_LAST:.*]] = fir.convert %[[J_LASTIDX]] : (index) -> i32
-! CHECK:   fir.store %[[J_LAST]] to %[[LOOP_VAR_J_DECL]]#0 : !fir.ref<i32>
-! CHECK:   %[[TRIP_VAR_I:.*]] = fir.load %[[TRIP_VAR_I_REF]] : !fir.ref<i32>
-! CHECK:   %[[C1_3:.*]] = arith.constant 1 : i32
-! CHECK:   %[[TRIP_VAR_I_NEXT:.*]] = arith.subi %[[TRIP_VAR_I]], %[[C1_3]] : i32
-! CHECK:   fir.store %[[TRIP_VAR_I_NEXT]] to %[[TRIP_VAR_I_REF]] : !fir.ref<i32>
-! CHECK:   %[[LOOP_VAR_I:.*]] = fir.load %[[LOOP_VAR_I_DECL]]#0 : !fir.ref<i32>
-! CHECK:   %[[I_STEP_2:.*]] = arith.constant 1 : i32
-! CHECK:   %[[LOOP_VAR_I_NEXT:.*]] = arith.addi %[[LOOP_VAR_I]], %[[I_STEP_2]] overflow<nsw> : i32
-! CHECK:   fir.store %[[LOOP_VAR_I_NEXT]] to %[[LOOP_VAR_I_DECL]]#0 : !fir.ref<i32>
-! CHECK:   cf.br ^[[HEADER]]
-! CHECK: ^[[EXIT]]:
+! CHECK:   fir.do_loop %{{.*}} = %c1{{.*}} to %c100{{.*}} step %c1{{.*}} : i32 {
+! CHECK:     scf.execute_region no_inline {
+! CHECK:       fir.do_loop %{{.*}} = %c1{{.*}} to %c100{{.*}} step %c1{{.*}} : i32 {
+! CHECK:       scf.yield
 ! CHECK:   return
 
 subroutine unstructured_do_concurrent
diff --git a/flang/test/Lower/select-case-statement.f90 b/flang/test/Lower/select-case-statement.f90
index b289024da2810..c1298bf177c8e 100644
--- a/flang/test/Lower/select-case-statement.f90
+++ b/flang/test/Lower/select-case-statement.f90
@@ -348,9 +348,12 @@ subroutine sempty(n)
   ! select case with goto exit
   subroutine sgoto
     n = 0
-    ! CHECK: cf.cond_br
+    ! The SELECT CASE and its GOTOs branch only within the loop body, so the
+    ! loop keeps its structured form and the raw blocks are confined to a wrap.
+    ! CHECK: fir.do_loop
+    ! CHECK: scf.execute_region no_inline {
     do i=1,8
-      ! CHECK: fir.select_case %8 : i32 [#fir.upper, %c2_i32, ^bb{{.*}}, #fir.lower, %c5_i32, ^bb{{.*}}, unit, ^bb{{.*}}]
+      ! CHECK: fir.select_case %{{[0-9]+}} : i32 [#fir.upper, %c2_i32, ^bb{{.*}}, #fir.lower, %c5_i32, ^bb{{.*}}, unit, ^bb{{.*}}]
       select case(i)
       case (:2)
         ! CHECK-DAG: arith.muli {{.*}}, %c10_i32 : i32



More information about the llvm-branch-commits mailing list